prime-radiant 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- prime_radiant-0.1.0/.gitignore +23 -0
- prime_radiant-0.1.0/CHANGELOG.md +32 -0
- prime_radiant-0.1.0/CITATION.cff +24 -0
- prime_radiant-0.1.0/LICENSE +21 -0
- prime_radiant-0.1.0/PKG-INFO +179 -0
- prime_radiant-0.1.0/README.md +147 -0
- prime_radiant-0.1.0/psr_parser.py +40 -0
- prime_radiant-0.1.0/pyproject.toml +160 -0
- prime_radiant-0.1.0/src/prime_radiant/__init__.py +1 -0
- prime_radiant-0.1.0/src/prime_radiant/config.py +18 -0
- prime_radiant-0.1.0/src/prime_radiant/epi/__init__.py +0 -0
- prime_radiant-0.1.0/src/prime_radiant/epi/backtest/__init__.py +1 -0
- prime_radiant-0.1.0/src/prime_radiant/epi/backtest/report.py +289 -0
- prime_radiant-0.1.0/src/prime_radiant/epi/backtest/rolling.py +128 -0
- prime_radiant-0.1.0/src/prime_radiant/epi/cli.py +161 -0
- prime_radiant-0.1.0/src/prime_radiant/epi/data/__init__.py +0 -0
- prime_radiant-0.1.0/src/prime_radiant/epi/data/benchmarks.py +84 -0
- prime_radiant-0.1.0/src/prime_radiant/epi/data/epiweek.py +22 -0
- prime_radiant-0.1.0/src/prime_radiant/epi/data/hub.py +60 -0
- prime_radiant-0.1.0/src/prime_radiant/epi/data/locations.py +50 -0
- prime_radiant-0.1.0/src/prime_radiant/epi/data/nhsn.py +68 -0
- prime_radiant-0.1.0/src/prime_radiant/epi/data/vintages.py +99 -0
- prime_radiant-0.1.0/src/prime_radiant/epi/features/__init__.py +1 -0
- prime_radiant-0.1.0/src/prime_radiant/epi/features/assemble.py +130 -0
- prime_radiant-0.1.0/src/prime_radiant/epi/features/lags.py +44 -0
- prime_radiant-0.1.0/src/prime_radiant/epi/features/seasonal.py +24 -0
- prime_radiant-0.1.0/src/prime_radiant/epi/features/transform.py +82 -0
- prime_radiant-0.1.0/src/prime_radiant/epi/models/__init__.py +1 -0
- prime_radiant-0.1.0/src/prime_radiant/epi/models/baseline.py +119 -0
- prime_radiant-0.1.0/src/prime_radiant/epi/models/ensemble.py +24 -0
- prime_radiant-0.1.0/src/prime_radiant/epi/models/lgbm_quantile.py +85 -0
- prime_radiant-0.1.0/src/prime_radiant/epi/models/postprocess.py +17 -0
- prime_radiant-0.1.0/src/prime_radiant/epi/models/seasonal.py +61 -0
- prime_radiant-0.1.0/src/prime_radiant/epi/replication.py +75 -0
- prime_radiant-0.1.0/src/prime_radiant/epi/schemas.py +193 -0
- prime_radiant-0.1.0/src/prime_radiant/epi/serve/__init__.py +0 -0
- prime_radiant-0.1.0/src/prime_radiant/epi/serve/bundle.py +176 -0
- prime_radiant-0.1.0/src/prime_radiant/epi/submission/__init__.py +1 -0
- prime_radiant-0.1.0/src/prime_radiant/epi/submission/format.py +39 -0
- prime_radiant-0.1.0/src/prime_radiant/epi/submission/metadata.py +89 -0
- prime_radiant-0.1.0/src/prime_radiant/epi/submission/validate.py +93 -0
- prime_radiant-0.1.0/src/prime_radiant/epi/submission/write.py +57 -0
- prime_radiant-0.1.0/src/prime_radiant/eval/__init__.py +2 -0
- prime_radiant-0.1.0/src/prime_radiant/eval/scoring.py +80 -0
- prime_radiant-0.1.0/src/prime_radiant/eval/wis.py +139 -0
- prime_radiant-0.1.0/src/prime_radiant/metaculus.py +69 -0
- prime_radiant-0.1.0/uv.lock +4850 -0
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
# Secrets — never commit
|
|
2
|
+
.env
|
|
3
|
+
|
|
4
|
+
# Data and run artifacts — root-anchored: src/prime_radiant/epi/data is CODE
|
|
5
|
+
/data/
|
|
6
|
+
/model-output/
|
|
7
|
+
*.duckdb
|
|
8
|
+
logs/
|
|
9
|
+
.coverage
|
|
10
|
+
|
|
11
|
+
# Python
|
|
12
|
+
.venv/
|
|
13
|
+
__pycache__/
|
|
14
|
+
*.pyc
|
|
15
|
+
.pytest_cache/
|
|
16
|
+
.ruff_cache/
|
|
17
|
+
dist/
|
|
18
|
+
|
|
19
|
+
# OS
|
|
20
|
+
.DS_Store
|
|
21
|
+
|
|
22
|
+
# MkDocs build output
|
|
23
|
+
site/
|
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
# Changelog
|
|
2
|
+
|
|
3
|
+
All notable changes to this project are documented in this file.
|
|
4
|
+
|
|
5
|
+
The format follows [Keep a Changelog](https://keepachangelog.com/en/1.1.0/),
|
|
6
|
+
and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html).
|
|
7
|
+
Releases below the insertion marker are generated by python-semantic-release
|
|
8
|
+
from the conventional commit history.
|
|
9
|
+
|
|
10
|
+
<!-- version list -->
|
|
11
|
+
|
|
12
|
+
## v0.1.0 (2026-09-01)
|
|
13
|
+
|
|
14
|
+
- Initial Release
|
|
15
|
+
|
|
16
|
+
## Pre-release history (unversioned, 2026-08)
|
|
17
|
+
|
|
18
|
+
The project was built in adversarially-verified phases before its first tagged
|
|
19
|
+
release; the full record lives in `NOTES/restart.md` and the `[claude]`-prefixed
|
|
20
|
+
commit history.
|
|
21
|
+
|
|
22
|
+
- **Phase 2A–2B** — FluSight data layer on git-vintage truth; WIS scorer
|
|
23
|
+
cross-validated against the official pipeline (relative WIS 0.999989).
|
|
24
|
+
- **Phase 2C** — pooled LightGBM quantile model; beats FluSight-baseline.
|
|
25
|
+
- **Phase 2D** — three-season league tables; LightGBM wins 2025-26 (relative
|
|
26
|
+
WIS 0.609 vs UMass-flusion 0.625), loses the two earlier seasons — printed.
|
|
27
|
+
- **Phase 2E** — submission automation: `prime-radiant epi forecast|validate`,
|
|
28
|
+
hub-valid CSV writer, registration metadata, five-gate inert live path.
|
|
29
|
+
- **Phase F** — Gradio dashboard on a precomputed serve bundle; deployed to
|
|
30
|
+
Hugging Face Spaces; FluSight registration PR opened.
|
|
31
|
+
- **Phase G** — release hardening: CI, pre-commit, Dependabot, release and
|
|
32
|
+
publish automation, Docker, docs, and the public flip.
|
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
cff-version: 1.2.0
|
|
2
|
+
message: "If you use this software, please cite it as below."
|
|
3
|
+
title: "Prime Radiant"
|
|
4
|
+
type: software
|
|
5
|
+
authors:
|
|
6
|
+
- family-names: Gracey
|
|
7
|
+
given-names: Jeremy
|
|
8
|
+
email: jeremy.a.gracey@gmail.com
|
|
9
|
+
version: 0.1.0
|
|
10
|
+
date-released: 2026-08-31
|
|
11
|
+
license: MIT
|
|
12
|
+
repository-code: "https://github.com/JeremyGracey-AI/prime-radiant"
|
|
13
|
+
url: "https://huggingface.co/spaces/jeremygracey-ai/prime-radiant"
|
|
14
|
+
abstract: >-
|
|
15
|
+
Calibrated forecasting: a CDC FluSight quantile forecaster (pooled LightGBM
|
|
16
|
+
quantile regression ensembled with a FluSight-baseline replica, WIS-scored on
|
|
17
|
+
vintage-honest rolling-origin backtests) and a parked Metaculus LLM
|
|
18
|
+
forecasting thread.
|
|
19
|
+
keywords:
|
|
20
|
+
- forecasting
|
|
21
|
+
- calibration
|
|
22
|
+
- flusight
|
|
23
|
+
- epidemiology
|
|
24
|
+
- quantile-regression
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Jeremy Gracey
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,179 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: prime-radiant
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Calibrated forecasting: Metaculus LLM bot (Brier) and CDC FluSight quantile forecaster (WIS).
|
|
5
|
+
Project-URL: Homepage, https://github.com/JeremyGracey-AI/prime-radiant
|
|
6
|
+
Project-URL: Repository, https://github.com/JeremyGracey-AI/prime-radiant
|
|
7
|
+
Author-email: Jeremy Gracey <jeremy.a.gracey@gmail.com>
|
|
8
|
+
License-Expression: MIT
|
|
9
|
+
License-File: LICENSE
|
|
10
|
+
Keywords: calibration,epidemiology,flusight,forecasting,metaculus,quantile
|
|
11
|
+
Classifier: Development Status :: 3 - Alpha
|
|
12
|
+
Classifier: Intended Audience :: Science/Research
|
|
13
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
14
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
17
|
+
Classifier: Topic :: Scientific/Engineering
|
|
18
|
+
Requires-Python: >=3.11
|
|
19
|
+
Requires-Dist: epiweeks>=2.3
|
|
20
|
+
Requires-Dist: forecasting-tools>=0.2.92
|
|
21
|
+
Requires-Dist: httpx>=0.27
|
|
22
|
+
Requires-Dist: lightgbm==4.7.0
|
|
23
|
+
Requires-Dist: matplotlib<3.12,>=3.11
|
|
24
|
+
Requires-Dist: pandas>=2.2
|
|
25
|
+
Requires-Dist: pandera>=0.20
|
|
26
|
+
Requires-Dist: pyarrow>=17
|
|
27
|
+
Requires-Dist: pydantic-settings>=2.5
|
|
28
|
+
Requires-Dist: pydantic>=2.9
|
|
29
|
+
Requires-Dist: python-dotenv>=1.0
|
|
30
|
+
Requires-Dist: pyyaml>=6.0.3
|
|
31
|
+
Description-Content-Type: text/markdown
|
|
32
|
+
|
|
33
|
+
# Prime Radiant
|
|
34
|
+
|
|
35
|
+
[](https://github.com/JeremyGracey-AI/prime-radiant/actions/workflows/ci.yml)
|
|
36
|
+
[](./AI-USE.md)
|
|
37
|
+
[](./LICENSE)
|
|
38
|
+
[](https://huggingface.co/spaces/jeremygracey-ai/prime-radiant)
|
|
39
|
+
[](https://jeremygracey-ai.github.io/prime-radiant/)
|
|
40
|
+
<!-- At first release / first coverage upload these join (kept out until true):
|
|
41
|
+
codecov, PyPI version — see NOTES/go-live-runbook.md -->
|
|
42
|
+
|
|
43
|
+
Calibrated forecasting, measured honestly. Two threads share this repo:
|
|
44
|
+
|
|
45
|
+
- **FluSight epi forecaster** (`prime_radiant.epi`) — CDC FluSight-format quantile
|
|
46
|
+
forecasts of weekly US influenza hospital admissions: LightGBM quantile regression
|
|
47
|
+
in the shape of the FluSight-2023/24-winning flusion model, ensembled with a
|
|
48
|
+
validated replica of CDC's own baseline. Pure statistical pipeline; no LLM calls.
|
|
49
|
+
- **Metaculus bot** (`prime_radiant.metaculus`) — LLM forecasting on binary
|
|
50
|
+
questions (parked at its retrieval phase; client + tests remain green).
|
|
51
|
+
|
|
52
|
+
## Backtest results (three seasons, vintage-honest)
|
|
53
|
+
|
|
54
|
+
Rolling-origin backtests over 85 weekly origins across the 2023-24, 2024-25 and
|
|
55
|
+
2025-26 seasons. Every forecast is trained only on the data snapshot a live
|
|
56
|
+
forecaster would have held on the Wednesday submission evening (the hub's git
|
|
57
|
+
history is the vintage store); scored against final truth as of 2026-07-09,
|
|
58
|
+
horizons 0-3, all 53 jurisdictions. `wis_rel` = mean WIS relative to the official
|
|
59
|
+
FluSight-baseline on the common task set across all six models (lower is better;
|
|
60
|
+
the official "scaled relative skill" collapses to this ratio on identical sets).
|
|
61
|
+
Coverage columns are computed on each model's full scored set; relative-skill
|
|
62
|
+
columns on the common set — `n` and `n_relative` in the CSVs disclose both.
|
|
63
|
+
|
|
64
|
+
**2025-26** — our model leads the table on the natural scale (on the log(x+1)
|
|
65
|
+
scale UMass-flusion edges it, 0.583 vs 0.585 — both columns are in the CSVs):
|
|
66
|
+
|
|
67
|
+
| model | wis_rel | 50% cov | 95% cov |
|
|
68
|
+
|---|---|---|---|
|
|
69
|
+
| **prime-radiant-lgbm** | **0.609** | 0.398 | 0.818 |
|
|
70
|
+
| UMass-flusion | 0.625 | 0.353 | 0.823 |
|
|
71
|
+
| FluSight-ensemble | 0.666 | 0.526 | 0.903 |
|
|
72
|
+
| prime-radiant-ensemble | 0.764 | 0.464 | 0.880 |
|
|
73
|
+
| FluSight-baseline | 1.000 | 0.433 | 0.864 |
|
|
74
|
+
|
|
75
|
+
**2024-25** — the multi-model ensembles beat us; we beat the baseline:
|
|
76
|
+
|
|
77
|
+
| model | wis_rel | 50% cov | 95% cov |
|
|
78
|
+
|---|---|---|---|
|
|
79
|
+
| UMass-flusion | 0.669 | 0.397 | 0.820 |
|
|
80
|
+
| FluSight-ensemble | 0.675 | 0.519 | 0.818 |
|
|
81
|
+
| **prime-radiant-lgbm** | **0.796** | 0.338 | 0.738 |
|
|
82
|
+
| prime-radiant-ensemble | 0.900 | 0.371 | 0.744 |
|
|
83
|
+
| FluSight-baseline | 1.000 | 0.317 | 0.719 |
|
|
84
|
+
|
|
85
|
+
**2023-24** — flusion dominates; our ensemble edges FluSight-ensemble:
|
|
86
|
+
|
|
87
|
+
| model | wis_rel | 50% cov | 95% cov |
|
|
88
|
+
|---|---|---|---|
|
|
89
|
+
| UMass-flusion | 0.569 | 0.569 | 0.964 |
|
|
90
|
+
| **prime-radiant-ensemble** | **0.716** | 0.422 | 0.909 |
|
|
91
|
+
| FluSight-ensemble | 0.730 | 0.488 | 0.920 |
|
|
92
|
+
| prime-radiant-lgbm | 0.841 | 0.369 | 0.831 |
|
|
93
|
+
| FluSight-baseline | 1.000 | 0.282 | 0.897 |
|
|
94
|
+
|
|
95
|
+
Full tables (per-horizon rows, log-scale variants, AE-median, task counts):
|
|
96
|
+
[`reports/backtest_<season>.csv`](reports/). Calibration:
|
|
97
|
+
|
|
98
|
+

|
|
99
|
+
|
|
100
|
+
## Honest framing
|
|
101
|
+
|
|
102
|
+
- **The backtest is net biased against us, and we keep it that way.** Adversarial
|
|
103
|
+
verification proved that on three 2024-25 holiday weeks the official baseline's
|
|
104
|
+
run saw data committed Thursday+, which our live-Wednesday vintage discipline
|
|
105
|
+
refuses; on the 24 information-equal origins our relative WIS improves (lgbm
|
|
106
|
+
~0.73). It is not one-way: at three October-2023 origins our vintages were
|
|
107
|
+
fresher than what the official run used. One 2023-24 origin (2024-04-13) had a
|
|
108
|
+
week-stale vintage — for our models only; the officials ran on fresh data there.
|
|
109
|
+
- **Our intervals are too narrow.** The lgbm's 50% intervals cover 34-40% and its
|
|
110
|
+
95% intervals 74-83% — under-dispersed, visible in the calibration curves.
|
|
111
|
+
FluSight-ensemble is better calibrated even where we beat it on WIS. This is
|
|
112
|
+
the main modeling debt; season-level bagging (flusion's stabilizer) is the
|
|
113
|
+
known lever.
|
|
114
|
+
- **Wins and losses both stand.** We lead 2025-26 outright; UMass-flusion beats
|
|
115
|
+
us clearly in 2023-24 and 2024-25. The 2023-24 lgbm ran on ~1.2 seasons of
|
|
116
|
+
training history.
|
|
117
|
+
- **The scorer and baseline replica are self-validating — for the season they
|
|
118
|
+
were validated on.** The replica reproduces the official 2024-25 baseline to
|
|
119
|
+
relative WIS 0.99999 on fingerprint-matched vintages (Phase B). In 2023-24 the
|
|
120
|
+
replica scores 0.976 vs the official baseline — a spread-construction
|
|
121
|
+
divergence at matched anchors that Phase B's validation (2024-25-era official
|
|
122
|
+
code) does not cover; adversarially checked: substituting the actual official
|
|
123
|
+
baseline into our ensemble changes no table ordering.
|
|
124
|
+
|
|
125
|
+
## Methodology in one paragraph
|
|
126
|
+
|
|
127
|
+
NHSN weekly admissions (hub target data, git-vintaged) → per-100k 4th-root
|
|
128
|
+
transform with per-location scale/center fitted per origin → pooled LightGBM
|
|
129
|
+
quantile regression (23 levels, horizon-as-feature, deterministic, exact-pinned
|
|
130
|
+
4.7.0) → monotone sort, inversion, integer rounding at the hub boundary →
|
|
131
|
+
per-quantile-median ensemble with the baseline replica → WIS/coverage scoring on
|
|
132
|
+
natural and log(x+1) scales, pinned-truth, common-task relative skill. Every
|
|
133
|
+
stage boundary carries a pandera contract; leakage invariants (vintage as-of,
|
|
134
|
+
feature cut, scaler fit) are hypothesis-tested properties.
|
|
135
|
+
|
|
136
|
+
## Setup
|
|
137
|
+
|
|
138
|
+
```sh
|
|
139
|
+
uv sync --all-groups
|
|
140
|
+
make check # offline gates: ruff + format + pyright + pytest-cov
|
|
141
|
+
make test-integration # network: real hub clone, S3 benchmarks, gate + reports
|
|
142
|
+
```
|
|
143
|
+
|
|
144
|
+
Requires Homebrew `libomp` on macOS for LightGBM. Built with
|
|
145
|
+
[Claude Code](https://claude.com/claude-code) (agentic coding assistant by
|
|
146
|
+
Anthropic); every phase was adversarially verified
|
|
147
|
+
by refuter agents before being declared done — see `NOTES/restart.md` and the
|
|
148
|
+
`[claude]`-prefixed commit history.
|
|
149
|
+
|
|
150
|
+
## Dashboard
|
|
151
|
+
|
|
152
|
+
**Live: https://huggingface.co/spaces/jeremygracey-ai/prime-radiant**
|
|
153
|
+
|
|
154
|
+
A Gradio dashboard (US choropleth of predicted 3-week change, per-state fan
|
|
155
|
+
charts, reliability curves, model-vs-baseline league tables) serves the frozen
|
|
156
|
+
backtest record from `serve_data/`, a ~1.7MB precomputed bundle:
|
|
157
|
+
|
|
158
|
+
```sh
|
|
159
|
+
make bundle # offline: rebuilds serve_data/ deterministically
|
|
160
|
+
uv run python dashboard/app.py # local: http://127.0.0.1:7860
|
|
161
|
+
```
|
|
162
|
+
|
|
163
|
+
The Space (`jeremygracey-ai/prime-radiant`, CPU-basic) installs only
|
|
164
|
+
gradio/plotly/pandas/pyarrow — never this package, so no LightGBM/libomp and no
|
|
165
|
+
hub clone at serve time. `.github/workflows/space-deploy.yml` stages and
|
|
166
|
+
validates the Space tree on every dispatch; the actual push is triple-gated
|
|
167
|
+
(manual `deploy=true` + `SPACE_LIVE=1` repo var + `HF_TOKEN` secret) and stays
|
|
168
|
+
inert until go-live.
|
|
169
|
+
|
|
170
|
+
## Limitations
|
|
171
|
+
|
|
172
|
+
- Single data source (NHSN), single target (`wk inc flu hosp`); no rate-change /
|
|
173
|
+
peak / ED-visit targets yet.
|
|
174
|
+
- Interval under-dispersion as above; no season-bagging yet.
|
|
175
|
+
- 2022-23 is not backtestable (the hub's vintage history begins Oct 2023).
|
|
176
|
+
- Live submission machinery exists (Phase E) but is structurally inert: the
|
|
177
|
+
weekly workflow's live path is triple-gated and unimplemented past its gates,
|
|
178
|
+
and nothing submits anywhere without the go-live runbook being executed by
|
|
179
|
+
hand. The dashboard serves a frozen bundle, not a live feed.
|
|
@@ -0,0 +1,147 @@
|
|
|
1
|
+
# Prime Radiant
|
|
2
|
+
|
|
3
|
+
[](https://github.com/JeremyGracey-AI/prime-radiant/actions/workflows/ci.yml)
|
|
4
|
+
[](./AI-USE.md)
|
|
5
|
+
[](./LICENSE)
|
|
6
|
+
[](https://huggingface.co/spaces/jeremygracey-ai/prime-radiant)
|
|
7
|
+
[](https://jeremygracey-ai.github.io/prime-radiant/)
|
|
8
|
+
<!-- At first release / first coverage upload these join (kept out until true):
|
|
9
|
+
codecov, PyPI version — see NOTES/go-live-runbook.md -->
|
|
10
|
+
|
|
11
|
+
Calibrated forecasting, measured honestly. Two threads share this repo:
|
|
12
|
+
|
|
13
|
+
- **FluSight epi forecaster** (`prime_radiant.epi`) — CDC FluSight-format quantile
|
|
14
|
+
forecasts of weekly US influenza hospital admissions: LightGBM quantile regression
|
|
15
|
+
in the shape of the FluSight-2023/24-winning flusion model, ensembled with a
|
|
16
|
+
validated replica of CDC's own baseline. Pure statistical pipeline; no LLM calls.
|
|
17
|
+
- **Metaculus bot** (`prime_radiant.metaculus`) — LLM forecasting on binary
|
|
18
|
+
questions (parked at its retrieval phase; client + tests remain green).
|
|
19
|
+
|
|
20
|
+
## Backtest results (three seasons, vintage-honest)
|
|
21
|
+
|
|
22
|
+
Rolling-origin backtests over 85 weekly origins across the 2023-24, 2024-25 and
|
|
23
|
+
2025-26 seasons. Every forecast is trained only on the data snapshot a live
|
|
24
|
+
forecaster would have held on the Wednesday submission evening (the hub's git
|
|
25
|
+
history is the vintage store); scored against final truth as of 2026-07-09,
|
|
26
|
+
horizons 0-3, all 53 jurisdictions. `wis_rel` = mean WIS relative to the official
|
|
27
|
+
FluSight-baseline on the common task set across all six models (lower is better;
|
|
28
|
+
the official "scaled relative skill" collapses to this ratio on identical sets).
|
|
29
|
+
Coverage columns are computed on each model's full scored set; relative-skill
|
|
30
|
+
columns on the common set — `n` and `n_relative` in the CSVs disclose both.
|
|
31
|
+
|
|
32
|
+
**2025-26** — our model leads the table on the natural scale (on the log(x+1)
|
|
33
|
+
scale UMass-flusion edges it, 0.583 vs 0.585 — both columns are in the CSVs):
|
|
34
|
+
|
|
35
|
+
| model | wis_rel | 50% cov | 95% cov |
|
|
36
|
+
|---|---|---|---|
|
|
37
|
+
| **prime-radiant-lgbm** | **0.609** | 0.398 | 0.818 |
|
|
38
|
+
| UMass-flusion | 0.625 | 0.353 | 0.823 |
|
|
39
|
+
| FluSight-ensemble | 0.666 | 0.526 | 0.903 |
|
|
40
|
+
| prime-radiant-ensemble | 0.764 | 0.464 | 0.880 |
|
|
41
|
+
| FluSight-baseline | 1.000 | 0.433 | 0.864 |
|
|
42
|
+
|
|
43
|
+
**2024-25** — the multi-model ensembles beat us; we beat the baseline:
|
|
44
|
+
|
|
45
|
+
| model | wis_rel | 50% cov | 95% cov |
|
|
46
|
+
|---|---|---|---|
|
|
47
|
+
| UMass-flusion | 0.669 | 0.397 | 0.820 |
|
|
48
|
+
| FluSight-ensemble | 0.675 | 0.519 | 0.818 |
|
|
49
|
+
| **prime-radiant-lgbm** | **0.796** | 0.338 | 0.738 |
|
|
50
|
+
| prime-radiant-ensemble | 0.900 | 0.371 | 0.744 |
|
|
51
|
+
| FluSight-baseline | 1.000 | 0.317 | 0.719 |
|
|
52
|
+
|
|
53
|
+
**2023-24** — flusion dominates; our ensemble edges FluSight-ensemble:
|
|
54
|
+
|
|
55
|
+
| model | wis_rel | 50% cov | 95% cov |
|
|
56
|
+
|---|---|---|---|
|
|
57
|
+
| UMass-flusion | 0.569 | 0.569 | 0.964 |
|
|
58
|
+
| **prime-radiant-ensemble** | **0.716** | 0.422 | 0.909 |
|
|
59
|
+
| FluSight-ensemble | 0.730 | 0.488 | 0.920 |
|
|
60
|
+
| prime-radiant-lgbm | 0.841 | 0.369 | 0.831 |
|
|
61
|
+
| FluSight-baseline | 1.000 | 0.282 | 0.897 |
|
|
62
|
+
|
|
63
|
+
Full tables (per-horizon rows, log-scale variants, AE-median, task counts):
|
|
64
|
+
[`reports/backtest_<season>.csv`](reports/). Calibration:
|
|
65
|
+
|
|
66
|
+

|
|
67
|
+
|
|
68
|
+
## Honest framing
|
|
69
|
+
|
|
70
|
+
- **The backtest is net biased against us, and we keep it that way.** Adversarial
|
|
71
|
+
verification proved that on three 2024-25 holiday weeks the official baseline's
|
|
72
|
+
run saw data committed Thursday+, which our live-Wednesday vintage discipline
|
|
73
|
+
refuses; on the 24 information-equal origins our relative WIS improves (lgbm
|
|
74
|
+
~0.73). It is not one-way: at three October-2023 origins our vintages were
|
|
75
|
+
fresher than what the official run used. One 2023-24 origin (2024-04-13) had a
|
|
76
|
+
week-stale vintage — for our models only; the officials ran on fresh data there.
|
|
77
|
+
- **Our intervals are too narrow.** The lgbm's 50% intervals cover 34-40% and its
|
|
78
|
+
95% intervals 74-83% — under-dispersed, visible in the calibration curves.
|
|
79
|
+
FluSight-ensemble is better calibrated even where we beat it on WIS. This is
|
|
80
|
+
the main modeling debt; season-level bagging (flusion's stabilizer) is the
|
|
81
|
+
known lever.
|
|
82
|
+
- **Wins and losses both stand.** We lead 2025-26 outright; UMass-flusion beats
|
|
83
|
+
us clearly in 2023-24 and 2024-25. The 2023-24 lgbm ran on ~1.2 seasons of
|
|
84
|
+
training history.
|
|
85
|
+
- **The scorer and baseline replica are self-validating — for the season they
|
|
86
|
+
were validated on.** The replica reproduces the official 2024-25 baseline to
|
|
87
|
+
relative WIS 0.99999 on fingerprint-matched vintages (Phase B). In 2023-24 the
|
|
88
|
+
replica scores 0.976 vs the official baseline — a spread-construction
|
|
89
|
+
divergence at matched anchors that Phase B's validation (2024-25-era official
|
|
90
|
+
code) does not cover; adversarially checked: substituting the actual official
|
|
91
|
+
baseline into our ensemble changes no table ordering.
|
|
92
|
+
|
|
93
|
+
## Methodology in one paragraph
|
|
94
|
+
|
|
95
|
+
NHSN weekly admissions (hub target data, git-vintaged) → per-100k 4th-root
|
|
96
|
+
transform with per-location scale/center fitted per origin → pooled LightGBM
|
|
97
|
+
quantile regression (23 levels, horizon-as-feature, deterministic, exact-pinned
|
|
98
|
+
4.7.0) → monotone sort, inversion, integer rounding at the hub boundary →
|
|
99
|
+
per-quantile-median ensemble with the baseline replica → WIS/coverage scoring on
|
|
100
|
+
natural and log(x+1) scales, pinned-truth, common-task relative skill. Every
|
|
101
|
+
stage boundary carries a pandera contract; leakage invariants (vintage as-of,
|
|
102
|
+
feature cut, scaler fit) are hypothesis-tested properties.
|
|
103
|
+
|
|
104
|
+
## Setup
|
|
105
|
+
|
|
106
|
+
```sh
|
|
107
|
+
uv sync --all-groups
|
|
108
|
+
make check # offline gates: ruff + format + pyright + pytest-cov
|
|
109
|
+
make test-integration # network: real hub clone, S3 benchmarks, gate + reports
|
|
110
|
+
```
|
|
111
|
+
|
|
112
|
+
Requires Homebrew `libomp` on macOS for LightGBM. Built with
|
|
113
|
+
[Claude Code](https://claude.com/claude-code) (agentic coding assistant by
|
|
114
|
+
Anthropic); every phase was adversarially verified
|
|
115
|
+
by refuter agents before being declared done — see `NOTES/restart.md` and the
|
|
116
|
+
`[claude]`-prefixed commit history.
|
|
117
|
+
|
|
118
|
+
## Dashboard
|
|
119
|
+
|
|
120
|
+
**Live: https://huggingface.co/spaces/jeremygracey-ai/prime-radiant**
|
|
121
|
+
|
|
122
|
+
A Gradio dashboard (US choropleth of predicted 3-week change, per-state fan
|
|
123
|
+
charts, reliability curves, model-vs-baseline league tables) serves the frozen
|
|
124
|
+
backtest record from `serve_data/`, a ~1.7MB precomputed bundle:
|
|
125
|
+
|
|
126
|
+
```sh
|
|
127
|
+
make bundle # offline: rebuilds serve_data/ deterministically
|
|
128
|
+
uv run python dashboard/app.py # local: http://127.0.0.1:7860
|
|
129
|
+
```
|
|
130
|
+
|
|
131
|
+
The Space (`jeremygracey-ai/prime-radiant`, CPU-basic) installs only
|
|
132
|
+
gradio/plotly/pandas/pyarrow — never this package, so no LightGBM/libomp and no
|
|
133
|
+
hub clone at serve time. `.github/workflows/space-deploy.yml` stages and
|
|
134
|
+
validates the Space tree on every dispatch; the actual push is triple-gated
|
|
135
|
+
(manual `deploy=true` + `SPACE_LIVE=1` repo var + `HF_TOKEN` secret) and stays
|
|
136
|
+
inert until go-live.
|
|
137
|
+
|
|
138
|
+
## Limitations
|
|
139
|
+
|
|
140
|
+
- Single data source (NHSN), single target (`wk inc flu hosp`); no rate-change /
|
|
141
|
+
peak / ED-visit targets yet.
|
|
142
|
+
- Interval under-dispersion as above; no season-bagging yet.
|
|
143
|
+
- 2022-23 is not backtestable (the hub's vintage history begins Oct 2023).
|
|
144
|
+
- Live submission machinery exists (Phase E) but is structurally inert: the
|
|
145
|
+
weekly workflow's live path is triple-gated and unimplemented past its gates,
|
|
146
|
+
and nothing submits anywhere without the go-live runbook being executed by
|
|
147
|
+
hand. The dashboard serves a frozen bundle, not a live feed.
|
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
"""Commit parser for python-semantic-release tolerating `[actor] type(scope):`.
|
|
2
|
+
|
|
3
|
+
The house convention prefixes agent-authored commits with `[claude] ` before a
|
|
4
|
+
conventional-commits subject. psr's stock ConventionalCommitParser anchors its
|
|
5
|
+
regex at string start, so every prefixed commit fails to parse (verified
|
|
6
|
+
empirically against psr 10.6.2). This subclass strips exactly one leading
|
|
7
|
+
`[word] ` token, then defers to the stock parser.
|
|
8
|
+
|
|
9
|
+
Loaded by psr via the FILE-PATH form in pyproject
|
|
10
|
+
(`commit_parser = "psr_parser.py:ActorPrefixConventionalParser"`) — the bare
|
|
11
|
+
module form fails because the repo root is not on sys.path when psr imports.
|
|
12
|
+
psr itself never lives in the project env (its tomlkit floor conflicts with
|
|
13
|
+
gradio's cap), so the semantic_release import below only resolves under
|
|
14
|
+
`uvx python-semantic-release==10.6.2` or the release workflow's action; the
|
|
15
|
+
pure strip logic stays importable (and unit-tested) without it.
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
import re
|
|
19
|
+
|
|
20
|
+
_ACTOR_PREFIX = re.compile(r"^\[[\w.-]+\]\s+")
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def strip_actor_prefix(message: str) -> str:
|
|
24
|
+
"""Remove one leading `[actor] ` token from a commit message, if present."""
|
|
25
|
+
return _ACTOR_PREFIX.sub("", message, count=1)
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
try:
|
|
29
|
+
# psr is never in the project env (tomlkit conflict with gradio) — resolved
|
|
30
|
+
# only under uvx / the release workflow, hence the scoped pyright ignore.
|
|
31
|
+
from semantic_release.commit_parser.conventional import ( # pyright: ignore[reportMissingImports]
|
|
32
|
+
ConventionalCommitParser,
|
|
33
|
+
)
|
|
34
|
+
except ImportError: # pragma: no cover — project env; psr runs isolated
|
|
35
|
+
ConventionalCommitParser = None # type: ignore[assignment, misc]
|
|
36
|
+
else:
|
|
37
|
+
|
|
38
|
+
class ActorPrefixConventionalParser(ConventionalCommitParser):
|
|
39
|
+
def parse_message(self, message: str): # noqa: ANN201 — psr's own signature
|
|
40
|
+
return super().parse_message(strip_actor_prefix(message))
|
|
@@ -0,0 +1,160 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["hatchling"]
|
|
3
|
+
build-backend = "hatchling.build"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "prime-radiant"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "Calibrated forecasting: Metaculus LLM bot (Brier) and CDC FluSight quantile forecaster (WIS)."
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.11"
|
|
11
|
+
license = "MIT"
|
|
12
|
+
authors = [{ name = "Jeremy Gracey", email = "jeremy.a.gracey@gmail.com" }]
|
|
13
|
+
keywords = ["forecasting", "calibration", "flusight", "metaculus", "quantile", "epidemiology"]
|
|
14
|
+
classifiers = [
|
|
15
|
+
"Development Status :: 3 - Alpha",
|
|
16
|
+
"Intended Audience :: Science/Research",
|
|
17
|
+
"License :: OSI Approved :: MIT License",
|
|
18
|
+
"Programming Language :: Python :: 3.11",
|
|
19
|
+
"Programming Language :: Python :: 3.12",
|
|
20
|
+
"Programming Language :: Python :: 3.13",
|
|
21
|
+
"Topic :: Scientific/Engineering",
|
|
22
|
+
]
|
|
23
|
+
dependencies = [
|
|
24
|
+
# Metaculus thread
|
|
25
|
+
"forecasting-tools>=0.2.92",
|
|
26
|
+
"pydantic>=2.9",
|
|
27
|
+
"pydantic-settings>=2.5",
|
|
28
|
+
"python-dotenv>=1.0",
|
|
29
|
+
# FluSight epi thread (data layer)
|
|
30
|
+
"pandas>=2.2",
|
|
31
|
+
"pandera>=0.20",
|
|
32
|
+
"epiweeks>=2.3",
|
|
33
|
+
"httpx>=0.27",
|
|
34
|
+
"pyarrow>=17",
|
|
35
|
+
# EXACT pin: LightGBM determinism holds only within one compiled binary —
|
|
36
|
+
# docs state results differ across versions/compilers/systems.
|
|
37
|
+
"lightgbm==4.7.0",
|
|
38
|
+
# <3.12 cap: report PNGs are committed artifacts; hold the rendering series
|
|
39
|
+
# stable within a phase so regeneration diffs stay meaningful.
|
|
40
|
+
"matplotlib>=3.11,<3.12",
|
|
41
|
+
"pyyaml>=6.0.3",
|
|
42
|
+
]
|
|
43
|
+
|
|
44
|
+
[project.scripts]
|
|
45
|
+
prime-radiant = "prime_radiant.epi.cli:main"
|
|
46
|
+
|
|
47
|
+
[project.urls]
|
|
48
|
+
Homepage = "https://github.com/JeremyGracey-AI/prime-radiant"
|
|
49
|
+
Repository = "https://github.com/JeremyGracey-AI/prime-radiant"
|
|
50
|
+
|
|
51
|
+
[dependency-groups]
|
|
52
|
+
dev = [
|
|
53
|
+
"pytest>=8",
|
|
54
|
+
"pytest-cov>=5",
|
|
55
|
+
"hypothesis>=6.100",
|
|
56
|
+
"pytest-recording>=0.13",
|
|
57
|
+
# floors match .pre-commit-config.yaml revs so hooks and `uv run` agree
|
|
58
|
+
"ruff>=0.16.5",
|
|
59
|
+
"pyright>=1.1.411",
|
|
60
|
+
"jsonschema>=4.26.0",
|
|
61
|
+
"types-pyyaml>=6.0.12.20260815",
|
|
62
|
+
]
|
|
63
|
+
# NOTE: python-semantic-release is deliberately NOT a dependency group — its
|
|
64
|
+
# tomlkit>=0.15 floor conflicts with gradio 6.26's tomlkit<0.15 cap. It runs
|
|
65
|
+
# isolated instead: `uvx python-semantic-release==10.6.2` locally, the
|
|
66
|
+
# containerized psr action (same version) in release.yml.
|
|
67
|
+
docs = [
|
|
68
|
+
"mkdocs-material>=9.7",
|
|
69
|
+
]
|
|
70
|
+
# Dev-only deps for the HF Space dashboard (dashboard/). The Space itself never
|
|
71
|
+
# installs this package — its requirements.txt is generated by space-deploy.yml.
|
|
72
|
+
dashboard = [
|
|
73
|
+
# ==6.26.*: match the Space's README sdk_version exactly — FetchMerck commit
|
|
74
|
+
# 8732e99 is the recorded cost of developing against a different Gradio than
|
|
75
|
+
# the Space runs.
|
|
76
|
+
"gradio==6.26.*",
|
|
77
|
+
# <7 cap: plotly 7 flips layout.geo.fitbounds default, removes
|
|
78
|
+
# figure_factory.create_choropleth and *mapbox traces, and Gradio's bundled
|
|
79
|
+
# plotly.js has no verified plotly-7 support. Declared directly per the
|
|
80
|
+
# Phase F brief (transitive-only via forecasting-tools before this).
|
|
81
|
+
"plotly>=6.9,<7",
|
|
82
|
+
]
|
|
83
|
+
|
|
84
|
+
[tool.hatch.build.targets.wheel]
|
|
85
|
+
packages = ["src/prime_radiant"]
|
|
86
|
+
|
|
87
|
+
# Default hatchling sdists ship the entire working tree (NOTES, handoff briefs,
|
|
88
|
+
# serve_data) — scope to the package + release-relevant files. LICENSE rides
|
|
89
|
+
# along via the license-files mechanism.
|
|
90
|
+
[tool.hatch.build.targets.sdist]
|
|
91
|
+
only-include = [
|
|
92
|
+
"src/prime_radiant",
|
|
93
|
+
"psr_parser.py",
|
|
94
|
+
"README.md",
|
|
95
|
+
"CHANGELOG.md",
|
|
96
|
+
"CITATION.cff",
|
|
97
|
+
"uv.lock",
|
|
98
|
+
]
|
|
99
|
+
|
|
100
|
+
[tool.ruff]
|
|
101
|
+
line-length = 100
|
|
102
|
+
target-version = "py311"
|
|
103
|
+
|
|
104
|
+
[tool.ruff.lint]
|
|
105
|
+
select = ["E", "F", "W", "I", "UP", "B", "SIM"]
|
|
106
|
+
|
|
107
|
+
[tool.ruff.lint.isort]
|
|
108
|
+
# dashboard/ modules ship flat to the HF Space root and import as siblings.
|
|
109
|
+
known-first-party = ["prime_radiant", "panel_data", "panel_plots"]
|
|
110
|
+
|
|
111
|
+
[tool.pyright]
|
|
112
|
+
pythonVersion = "3.11"
|
|
113
|
+
typeCheckingMode = "basic"
|
|
114
|
+
venvPath = "."
|
|
115
|
+
venv = ".venv"
|
|
116
|
+
|
|
117
|
+
# dashboard/ is not part of the package: it deploys flat to the Space root, so
|
|
118
|
+
# its modules resolve as siblings there and need the same path here and in tests.
|
|
119
|
+
[[tool.pyright.executionEnvironments]]
|
|
120
|
+
root = "dashboard"
|
|
121
|
+
extraPaths = ["dashboard"]
|
|
122
|
+
|
|
123
|
+
[[tool.pyright.executionEnvironments]]
|
|
124
|
+
root = "tests"
|
|
125
|
+
extraPaths = ["dashboard"]
|
|
126
|
+
|
|
127
|
+
[tool.semantic_release]
|
|
128
|
+
# Runs isolated: `uvx python-semantic-release==10.6.2` locally, the psr action
|
|
129
|
+
# (same version) in release.yml — see the dependency-group note above.
|
|
130
|
+
version_toml = ["pyproject.toml:project.version"]
|
|
131
|
+
# FILE-PATH form is load-bearing: the bare module form cannot import from cwd.
|
|
132
|
+
commit_parser = "psr_parser.py:ActorPrefixConventionalParser"
|
|
133
|
+
commit_message = "[claude] chore(release): v{version}"
|
|
134
|
+
# House rule: one git identity everywhere, agent commits included.
|
|
135
|
+
commit_author = "Jeremy Gracey <jeremy.a.gracey@gmail.com>"
|
|
136
|
+
# uv.lock carries the project's own version; regenerate AFTER stamping and
|
|
137
|
+
# ship it inside the release commit.
|
|
138
|
+
build_command = "uv lock && uv build"
|
|
139
|
+
assets = ["uv.lock"]
|
|
140
|
+
allow_zero_version = true
|
|
141
|
+
major_on_zero = false
|
|
142
|
+
tag_format = "v{version}"
|
|
143
|
+
|
|
144
|
+
[tool.semantic_release.changelog]
|
|
145
|
+
mode = "update"
|
|
146
|
+
|
|
147
|
+
[tool.pytest.ini_options]
|
|
148
|
+
testpaths = ["tests"]
|
|
149
|
+
markers = [
|
|
150
|
+
"unit: fast, offline, fixture-backed",
|
|
151
|
+
"integration: touches network or the real hub clone",
|
|
152
|
+
"contract: schema/property invariants (offline)",
|
|
153
|
+
"e2e: full pipeline runs",
|
|
154
|
+
]
|
|
155
|
+
|
|
156
|
+
[tool.coverage.run]
|
|
157
|
+
source = ["prime_radiant"]
|
|
158
|
+
|
|
159
|
+
[tool.coverage.report]
|
|
160
|
+
show_missing = true
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
"""Prime Radiant: calibrated LLM forecasting bot for Metaculus binary questions."""
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
"""Settings loaded from environment / .env — secrets and cost caps.
|
|
2
|
+
|
|
3
|
+
Cost caps are hard limits enforced by the run loop (Phase C): the bot must
|
|
4
|
+
stop spending when a cap is hit, and every run logs token spend against it.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from pydantic_settings import BaseSettings, SettingsConfigDict
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
class Settings(BaseSettings):
|
|
11
|
+
model_config = SettingsConfigDict(env_file=".env", env_file_encoding="utf-8", extra="ignore")
|
|
12
|
+
|
|
13
|
+
anthropic_api_key: str = ""
|
|
14
|
+
metaculus_token: str = ""
|
|
15
|
+
news_api_key: str = ""
|
|
16
|
+
|
|
17
|
+
per_question_budget_usd: float = 0.25
|
|
18
|
+
per_run_budget_usd: float = 2.50
|
|
File without changes
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
"""Rolling-origin backtesting over target-data vintages."""
|