hkjc-bench 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,3 @@
1
+ exclude README.md
2
+ prune data
3
+ prune examples
@@ -0,0 +1,38 @@
1
+ Metadata-Version: 2.4
2
+ Name: hkjc-bench
3
+ Version: 0.1.0
4
+ Summary: Compare your HKJC model with the predictor's best model on the same races, via its model test API
5
+ Project-URL: Homepage, https://hashedcookies.com
6
+ Classifier: Programming Language :: Python :: 3
7
+ Classifier: Operating System :: OS Independent
8
+ Requires-Python: >=3.10
9
+ Description-Content-Type: text/markdown
10
+ Requires-Dist: pandas>=2.0
11
+
12
+ # hkjc-bench
13
+
14
+ Compare your Hong Kong Jockey Club race model with the [HKJC predictor](https://hashedcookies.com)'s best model
15
+ (Model C), the current model and the betting market, on exactly the same races, without anyone's model code
16
+ changing hands.
17
+
18
+ You need an account on the predictor and an API key (dashboard → Profile → Model test API).
19
+
20
+ ```python
21
+ from hkjc_bench import Client
22
+
23
+ c = Client("hkjc_...") # or set HKJC_API_KEY
24
+ c.status() # what the benchmark covers
25
+ races = c.races(since="2024-09-01") # races you can be scored on
26
+ report = c.score(preds, basis="form") # preds: DataFrame of race_key, horse_no, p_win for every runner
27
+ print(report.summary())
28
+ report.by_season()
29
+ ```
30
+
31
+ - **basis="form"** compares with the models' form-only chances (for a model that doesn't use the odds);
32
+ **"blended"** with their final chances, which include the odds.
33
+ - The benchmark is every race since 2010, each predicted by a version of the model trained only on earlier races;
34
+ scores are log-loss on the winner (lower is better), with the chance your model is really better.
35
+ - Your predictions are scored in memory on the server and not stored.
36
+ - `c.predictions(race_keys)` returns Model C's own chances only if an admin has allowed it for your key.
37
+
38
+ Requires Python 3.10+ and pandas.
@@ -0,0 +1,27 @@
1
+ # hkjc-bench
2
+
3
+ Compare your Hong Kong Jockey Club race model with the [HKJC predictor](https://hashedcookies.com)'s best model
4
+ (Model C), the current model and the betting market, on exactly the same races, without anyone's model code
5
+ changing hands.
6
+
7
+ You need an account on the predictor and an API key (dashboard → Profile → Model test API).
8
+
9
+ ```python
10
+ from hkjc_bench import Client
11
+
12
+ c = Client("hkjc_...") # or set HKJC_API_KEY
13
+ c.status() # what the benchmark covers
14
+ races = c.races(since="2024-09-01") # races you can be scored on
15
+ report = c.score(preds, basis="form") # preds: DataFrame of race_key, horse_no, p_win for every runner
16
+ print(report.summary())
17
+ report.by_season()
18
+ ```
19
+
20
+ - **basis="form"** compares with the models' form-only chances (for a model that doesn't use the odds);
21
+ **"blended"** with their final chances, which include the odds.
22
+ - The benchmark is every race since 2010, each predicted by a version of the model trained only on earlier races;
23
+ scores are log-loss on the winner (lower is better), with the chance your model is really better.
24
+ - Your predictions are scored in memory on the server and not stored.
25
+ - `c.predictions(race_keys)` returns Model C's own chances only if an admin has allowed it for your key.
26
+
27
+ Requires Python 3.10+ and pandas.
@@ -0,0 +1,94 @@
1
+ """hkjc-bench: compare your model with the HKJC predictor's best model (Model C) on exactly the same races,
2
+ without either side sharing model code.
3
+
4
+ from hkjc_bench import Client
5
+ c = Client("hkjc_...") # or set HKJC_API_KEY; make a key in the dashboard's Profile tab
6
+ c.status() # what the benchmark covers
7
+ races = c.races(since="2024-09-01") # the races you can be scored on
8
+ report = c.score(preds, basis="form") # preds: DataFrame with race_key, horse_no, p_win
9
+ print(report.summary())
10
+
11
+ basis="form" compares with the models' form-only chances (right for a model that doesn't use the betting odds,
12
+ like the kit's examples); basis="blended" with their final chances, which include the odds. Give chances for
13
+ every runner in a race; races with missing runners are skipped. Your predictions are scored in memory on the
14
+ server and not stored.
15
+ """
16
+ import json
17
+ import os
18
+ import urllib.error
19
+ import urllib.request
20
+
21
+ import pandas as pd
22
+
23
+ __version__ = "0.1.0"
24
+ DEFAULT_URL = "https://hashedcookies.com/api/v1"
25
+
26
+
27
+ class APIError(RuntimeError):
28
+ pass
29
+
30
+
31
+ class Report(dict):
32
+ """The server's comparison, with helpers."""
33
+
34
+ def by_season(self) -> pd.DataFrame:
35
+ return pd.DataFrame(self.get("by_season", []))
36
+
37
+ def summary(self) -> str:
38
+ if not self.get("races"):
39
+ return f"No complete benchmark races in the upload (skipped: {self.get('skipped')})."
40
+ ll = self["logloss"]
41
+ better = self["chance_better_than_model_c"]
42
+ verdict = ("better than Model C" if self["vs_model_c"] < 0 and better >= 0.9 else
43
+ "possibly better than Model C (not proven)" if self["vs_model_c"] < 0 else "not better than Model C")
44
+ return (f"{self['races']:,} races ({self['basis']} chances), log-loss on the winner, lower is better:\n"
45
+ f" yours {ll['yours']:.4f}\n Model C {ll['model_c']:.4f}\n"
46
+ f" current model {ll['current']:.4f}\n market {ll['market']:.4f} (final odds)\n"
47
+ f"Yours vs Model C: {self['vs_model_c']:+.4f} -> {verdict} "
48
+ f"({better:.0%} chance it's really better; treat under 90% as not proven).")
49
+
50
+
51
+ class Client:
52
+ def __init__(self, api_key: str | None = None, url: str | None = None, timeout: float = 300):
53
+ self.key = api_key or os.environ.get("HKJC_API_KEY")
54
+ if not self.key:
55
+ raise APIError("No API key: pass Client(key) or set HKJC_API_KEY (create one in the dashboard's Profile tab).")
56
+ self.url = (url or os.environ.get("HKJC_API_URL") or DEFAULT_URL).rstrip("/")
57
+ self.timeout = timeout
58
+
59
+ def _call(self, path: str, body: dict | None = None) -> dict:
60
+ req = urllib.request.Request(self.url + path, data=None if body is None else json.dumps(body).encode(),
61
+ headers={"Authorization": f"Bearer {self.key}", "Content-Type": "application/json",
62
+ "User-Agent": f"hkjc-bench/{__version__}"})
63
+ try:
64
+ with urllib.request.urlopen(req, timeout=self.timeout) as r:
65
+ return json.loads(r.read())
66
+ except urllib.error.HTTPError as e:
67
+ try:
68
+ msg = json.loads(e.read()).get("error", e.reason)
69
+ except Exception:
70
+ msg = e.reason
71
+ raise APIError(f"{e.code}: {msg}") from None
72
+
73
+ def status(self) -> dict:
74
+ return self._call("/status")
75
+
76
+ def races(self, since: str | None = None, until: str | None = None) -> pd.DataFrame:
77
+ q = "&".join(f"{k}={v}" for k, v in (("since", since), ("until", until)) if v)
78
+ return pd.DataFrame(self._call("/races" + (f"?{q}" if q else ""))["races"])
79
+
80
+ def score(self, predictions: pd.DataFrame, basis: str = "form") -> Report:
81
+ """predictions: race_key, horse_no and p_win (or a score column you pass as p_win) for every runner."""
82
+ cols = {"race_key", "horse_no", "p_win"}
83
+ missing = cols - set(predictions.columns)
84
+ if missing:
85
+ raise ValueError(f"predictions need columns {sorted(cols)}; missing {sorted(missing)}")
86
+ rows = predictions[list(cols)].astype({"race_key": str}).to_dict("records")
87
+ return Report(self._call("/score", {"basis": basis, "predictions": rows}))
88
+
89
+ def predictions(self, race_keys: list[str]) -> pd.DataFrame:
90
+ """Model C's out-of-sample win chances (only for keys an admin has allowed; at most 200 races a call)."""
91
+ out = []
92
+ for i in range(0, len(race_keys), 200):
93
+ out += self._call("/predictions?race_keys=" + ",".join(race_keys[i:i + 200]))["predictions"]
94
+ return pd.DataFrame(out)
@@ -0,0 +1,38 @@
1
+ Metadata-Version: 2.4
2
+ Name: hkjc-bench
3
+ Version: 0.1.0
4
+ Summary: Compare your HKJC model with the predictor's best model on the same races, via its model test API
5
+ Project-URL: Homepage, https://hashedcookies.com
6
+ Classifier: Programming Language :: Python :: 3
7
+ Classifier: Operating System :: OS Independent
8
+ Requires-Python: >=3.10
9
+ Description-Content-Type: text/markdown
10
+ Requires-Dist: pandas>=2.0
11
+
12
+ # hkjc-bench
13
+
14
+ Compare your Hong Kong Jockey Club race model with the [HKJC predictor](https://hashedcookies.com)'s best model
15
+ (Model C), the current model and the betting market, on exactly the same races, without anyone's model code
16
+ changing hands.
17
+
18
+ You need an account on the predictor and an API key (dashboard → Profile → Model test API).
19
+
20
+ ```python
21
+ from hkjc_bench import Client
22
+
23
+ c = Client("hkjc_...") # or set HKJC_API_KEY
24
+ c.status() # what the benchmark covers
25
+ races = c.races(since="2024-09-01") # races you can be scored on
26
+ report = c.score(preds, basis="form") # preds: DataFrame of race_key, horse_no, p_win for every runner
27
+ print(report.summary())
28
+ report.by_season()
29
+ ```
30
+
31
+ - **basis="form"** compares with the models' form-only chances (for a model that doesn't use the odds);
32
+ **"blended"** with their final chances, which include the odds.
33
+ - The benchmark is every race since 2010, each predicted by a version of the model trained only on earlier races;
34
+ scores are log-loss on the winner (lower is better), with the chance your model is really better.
35
+ - Your predictions are scored in memory on the server and not stored.
36
+ - `c.predictions(race_keys)` returns Model C's own chances only if an admin has allowed it for your key.
37
+
38
+ Requires Python 3.10+ and pandas.
@@ -0,0 +1,10 @@
1
+ MANIFEST.in
2
+ PYPI.md
3
+ pyproject.toml
4
+ hkjc_bench/__init__.py
5
+ hkjc_bench.egg-info/PKG-INFO
6
+ hkjc_bench.egg-info/SOURCES.txt
7
+ hkjc_bench.egg-info/dependency_links.txt
8
+ hkjc_bench.egg-info/requires.txt
9
+ hkjc_bench.egg-info/top_level.txt
10
+ tests/test_bench.py
@@ -0,0 +1 @@
1
+ pandas>=2.0
@@ -0,0 +1 @@
1
+ hkjc_bench
@@ -0,0 +1,18 @@
1
+ [build-system]
2
+ requires = ["setuptools>=68"]
3
+ build-backend = "setuptools.build_meta"
4
+
5
+ [project]
6
+ name = "hkjc-bench"
7
+ version = "0.1.0"
8
+ description = "Compare your HKJC model with the predictor's best model on the same races, via its model test API"
9
+ readme = "PYPI.md"
10
+ requires-python = ">=3.10"
11
+ classifiers = ["Programming Language :: Python :: 3", "Operating System :: OS Independent"]
12
+ dependencies = ["pandas>=2.0"]
13
+
14
+ [project.urls]
15
+ Homepage = "https://hashedcookies.com"
16
+
17
+ [tool.setuptools]
18
+ packages = ["hkjc_bench"]
@@ -0,0 +1,4 @@
1
+ [egg_info]
2
+ tag_build =
3
+ tag_date = 0
4
+
@@ -0,0 +1,55 @@
1
+ import json
2
+ import threading
3
+ from http.server import BaseHTTPRequestHandler, HTTPServer
4
+
5
+ import pandas as pd
6
+ import pytest
7
+
8
+ from hkjc_bench import APIError, Client, Report
9
+
10
+
11
+ class _Stub(BaseHTTPRequestHandler):
12
+ def log_message(self, *a):
13
+ pass
14
+
15
+ def _send(self, code, body):
16
+ data = json.dumps(body).encode()
17
+ self.send_response(code)
18
+ self.send_header("Content-Type", "application/json")
19
+ self.end_headers()
20
+ self.wfile.write(data)
21
+
22
+ def do_GET(self):
23
+ if self.headers.get("Authorization") != "Bearer hkjc_good":
24
+ return self._send(401, {"error": "missing or invalid API key"})
25
+ self._send(200, {"races": [{"race_key": "2026-10-01_ST_01", "runners": 2}]})
26
+
27
+ def do_POST(self):
28
+ body = json.loads(self.rfile.read(int(self.headers["Content-Length"])))
29
+ assert body["basis"] == "form" and len(body["predictions"]) == 2
30
+ self._send(200, {"races": 1, "basis": "form", "skipped": {}, "vs_model_c": -0.01,
31
+ "chance_better_than_model_c": 0.95,
32
+ "logloss": {"yours": 2.0, "model_c": 2.01, "current": 2.02, "market": 1.99},
33
+ "by_season": [{"season": 2025, "yours": 2.0, "races": 1}]})
34
+
35
+
36
+ @pytest.fixture()
37
+ def url():
38
+ s = HTTPServer(("127.0.0.1", 0), _Stub)
39
+ threading.Thread(target=s.serve_forever, daemon=True).start()
40
+ yield f"http://127.0.0.1:{s.server_port}"
41
+ s.shutdown()
42
+
43
+
44
+ def test_score_and_races(url):
45
+ c = Client("hkjc_good", url)
46
+ assert c.races().iloc[0]["race_key"] == "2026-10-01_ST_01"
47
+ r = c.score(pd.DataFrame({"race_key": ["a", "a"], "horse_no": [1, 2], "p_win": [0.6, 0.4]}))
48
+ assert isinstance(r, Report) and "better than Model C" in r.summary() and len(r.by_season()) == 1
49
+
50
+
51
+ def test_errors_are_clear(url):
52
+ with pytest.raises(APIError, match="401"):
53
+ Client("hkjc_bad", url).races()
54
+ with pytest.raises(ValueError, match="missing"):
55
+ Client("hkjc_good", url).score(pd.DataFrame({"race_key": ["a"]}))