prime-radiant 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (47) hide show
  1. prime_radiant-0.1.0/.gitignore +23 -0
  2. prime_radiant-0.1.0/CHANGELOG.md +32 -0
  3. prime_radiant-0.1.0/CITATION.cff +24 -0
  4. prime_radiant-0.1.0/LICENSE +21 -0
  5. prime_radiant-0.1.0/PKG-INFO +179 -0
  6. prime_radiant-0.1.0/README.md +147 -0
  7. prime_radiant-0.1.0/psr_parser.py +40 -0
  8. prime_radiant-0.1.0/pyproject.toml +160 -0
  9. prime_radiant-0.1.0/src/prime_radiant/__init__.py +1 -0
  10. prime_radiant-0.1.0/src/prime_radiant/config.py +18 -0
  11. prime_radiant-0.1.0/src/prime_radiant/epi/__init__.py +0 -0
  12. prime_radiant-0.1.0/src/prime_radiant/epi/backtest/__init__.py +1 -0
  13. prime_radiant-0.1.0/src/prime_radiant/epi/backtest/report.py +289 -0
  14. prime_radiant-0.1.0/src/prime_radiant/epi/backtest/rolling.py +128 -0
  15. prime_radiant-0.1.0/src/prime_radiant/epi/cli.py +161 -0
  16. prime_radiant-0.1.0/src/prime_radiant/epi/data/__init__.py +0 -0
  17. prime_radiant-0.1.0/src/prime_radiant/epi/data/benchmarks.py +84 -0
  18. prime_radiant-0.1.0/src/prime_radiant/epi/data/epiweek.py +22 -0
  19. prime_radiant-0.1.0/src/prime_radiant/epi/data/hub.py +60 -0
  20. prime_radiant-0.1.0/src/prime_radiant/epi/data/locations.py +50 -0
  21. prime_radiant-0.1.0/src/prime_radiant/epi/data/nhsn.py +68 -0
  22. prime_radiant-0.1.0/src/prime_radiant/epi/data/vintages.py +99 -0
  23. prime_radiant-0.1.0/src/prime_radiant/epi/features/__init__.py +1 -0
  24. prime_radiant-0.1.0/src/prime_radiant/epi/features/assemble.py +130 -0
  25. prime_radiant-0.1.0/src/prime_radiant/epi/features/lags.py +44 -0
  26. prime_radiant-0.1.0/src/prime_radiant/epi/features/seasonal.py +24 -0
  27. prime_radiant-0.1.0/src/prime_radiant/epi/features/transform.py +82 -0
  28. prime_radiant-0.1.0/src/prime_radiant/epi/models/__init__.py +1 -0
  29. prime_radiant-0.1.0/src/prime_radiant/epi/models/baseline.py +119 -0
  30. prime_radiant-0.1.0/src/prime_radiant/epi/models/ensemble.py +24 -0
  31. prime_radiant-0.1.0/src/prime_radiant/epi/models/lgbm_quantile.py +85 -0
  32. prime_radiant-0.1.0/src/prime_radiant/epi/models/postprocess.py +17 -0
  33. prime_radiant-0.1.0/src/prime_radiant/epi/models/seasonal.py +61 -0
  34. prime_radiant-0.1.0/src/prime_radiant/epi/replication.py +75 -0
  35. prime_radiant-0.1.0/src/prime_radiant/epi/schemas.py +193 -0
  36. prime_radiant-0.1.0/src/prime_radiant/epi/serve/__init__.py +0 -0
  37. prime_radiant-0.1.0/src/prime_radiant/epi/serve/bundle.py +176 -0
  38. prime_radiant-0.1.0/src/prime_radiant/epi/submission/__init__.py +1 -0
  39. prime_radiant-0.1.0/src/prime_radiant/epi/submission/format.py +39 -0
  40. prime_radiant-0.1.0/src/prime_radiant/epi/submission/metadata.py +89 -0
  41. prime_radiant-0.1.0/src/prime_radiant/epi/submission/validate.py +93 -0
  42. prime_radiant-0.1.0/src/prime_radiant/epi/submission/write.py +57 -0
  43. prime_radiant-0.1.0/src/prime_radiant/eval/__init__.py +2 -0
  44. prime_radiant-0.1.0/src/prime_radiant/eval/scoring.py +80 -0
  45. prime_radiant-0.1.0/src/prime_radiant/eval/wis.py +139 -0
  46. prime_radiant-0.1.0/src/prime_radiant/metaculus.py +69 -0
  47. prime_radiant-0.1.0/uv.lock +4850 -0
@@ -0,0 +1,23 @@
1
+ # Secrets — never commit
2
+ .env
3
+
4
+ # Data and run artifacts — root-anchored: src/prime_radiant/epi/data is CODE
5
+ /data/
6
+ /model-output/
7
+ *.duckdb
8
+ logs/
9
+ .coverage
10
+
11
+ # Python
12
+ .venv/
13
+ __pycache__/
14
+ *.pyc
15
+ .pytest_cache/
16
+ .ruff_cache/
17
+ dist/
18
+
19
+ # OS
20
+ .DS_Store
21
+
22
+ # MkDocs build output
23
+ site/
@@ -0,0 +1,32 @@
1
+ # Changelog
2
+
3
+ All notable changes to this project are documented in this file.
4
+
5
+ The format follows [Keep a Changelog](https://keepachangelog.com/en/1.1.0/),
6
+ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html).
7
+ Releases below the insertion marker are generated by python-semantic-release
8
+ from the conventional commit history.
9
+
10
+ <!-- version list -->
11
+
12
+ ## v0.1.0 (2026-09-01)
13
+
14
+ - Initial Release
15
+
16
+ ## Pre-release history (unversioned, 2026-08)
17
+
18
+ The project was built in adversarially-verified phases before its first tagged
19
+ release; the full record lives in `NOTES/restart.md` and the `[claude]`-prefixed
20
+ commit history.
21
+
22
+ - **Phase 2A–2B** — FluSight data layer on git-vintage truth; WIS scorer
23
+ cross-validated against the official pipeline (relative WIS 0.999989).
24
+ - **Phase 2C** — pooled LightGBM quantile model; beats FluSight-baseline.
25
+ - **Phase 2D** — three-season league tables; LightGBM wins 2025-26 (relative
26
+ WIS 0.609 vs UMass-flusion 0.625), loses the two earlier seasons — printed.
27
+ - **Phase 2E** — submission automation: `prime-radiant epi forecast|validate`,
28
+ hub-valid CSV writer, registration metadata, five-gate inert live path.
29
+ - **Phase F** — Gradio dashboard on a precomputed serve bundle; deployed to
30
+ Hugging Face Spaces; FluSight registration PR opened.
31
+ - **Phase G** — release hardening: CI, pre-commit, Dependabot, release and
32
+ publish automation, Docker, docs, and the public flip.
@@ -0,0 +1,24 @@
1
+ cff-version: 1.2.0
2
+ message: "If you use this software, please cite it as below."
3
+ title: "Prime Radiant"
4
+ type: software
5
+ authors:
6
+ - family-names: Gracey
7
+ given-names: Jeremy
8
+ email: jeremy.a.gracey@gmail.com
9
+ version: 0.1.0
10
+ date-released: 2026-08-31
11
+ license: MIT
12
+ repository-code: "https://github.com/JeremyGracey-AI/prime-radiant"
13
+ url: "https://huggingface.co/spaces/jeremygracey-ai/prime-radiant"
14
+ abstract: >-
15
+ Calibrated forecasting: a CDC FluSight quantile forecaster (pooled LightGBM
16
+ quantile regression ensembled with a FluSight-baseline replica, WIS-scored on
17
+ vintage-honest rolling-origin backtests) and a parked Metaculus LLM
18
+ forecasting thread.
19
+ keywords:
20
+ - forecasting
21
+ - calibration
22
+ - flusight
23
+ - epidemiology
24
+ - quantile-regression
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Jeremy Gracey
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,179 @@
1
+ Metadata-Version: 2.5
2
+ Name: prime-radiant
3
+ Version: 0.1.0
4
+ Summary: Calibrated forecasting: Metaculus LLM bot (Brier) and CDC FluSight quantile forecaster (WIS).
5
+ Project-URL: Homepage, https://github.com/JeremyGracey-AI/prime-radiant
6
+ Project-URL: Repository, https://github.com/JeremyGracey-AI/prime-radiant
7
+ Author-email: Jeremy Gracey <jeremy.a.gracey@gmail.com>
8
+ License-Expression: MIT
9
+ License-File: LICENSE
10
+ Keywords: calibration,epidemiology,flusight,forecasting,metaculus,quantile
11
+ Classifier: Development Status :: 3 - Alpha
12
+ Classifier: Intended Audience :: Science/Research
13
+ Classifier: License :: OSI Approved :: MIT License
14
+ Classifier: Programming Language :: Python :: 3.11
15
+ Classifier: Programming Language :: Python :: 3.12
16
+ Classifier: Programming Language :: Python :: 3.13
17
+ Classifier: Topic :: Scientific/Engineering
18
+ Requires-Python: >=3.11
19
+ Requires-Dist: epiweeks>=2.3
20
+ Requires-Dist: forecasting-tools>=0.2.92
21
+ Requires-Dist: httpx>=0.27
22
+ Requires-Dist: lightgbm==4.7.0
23
+ Requires-Dist: matplotlib<3.12,>=3.11
24
+ Requires-Dist: pandas>=2.2
25
+ Requires-Dist: pandera>=0.20
26
+ Requires-Dist: pyarrow>=17
27
+ Requires-Dist: pydantic-settings>=2.5
28
+ Requires-Dist: pydantic>=2.9
29
+ Requires-Dist: python-dotenv>=1.0
30
+ Requires-Dist: pyyaml>=6.0.3
31
+ Description-Content-Type: text/markdown
32
+
33
+ # Prime Radiant
34
+
35
+ [![ci](https://github.com/JeremyGracey-AI/prime-radiant/actions/workflows/ci.yml/badge.svg)](https://github.com/JeremyGracey-AI/prime-radiant/actions/workflows/ci.yml)
36
+ [![AI-USE: declared](https://img.shields.io/badge/AI--USE-declared-2ea44f)](./AI-USE.md)
37
+ [![License: MIT](https://img.shields.io/badge/License-MIT-yellow.svg)](./LICENSE)
38
+ [![Dashboard](https://img.shields.io/badge/%F0%9F%A4%97%20Space-live-blue)](https://huggingface.co/spaces/jeremygracey-ai/prime-radiant)
39
+ [![docs](https://img.shields.io/badge/docs-live-2ea44f)](https://jeremygracey-ai.github.io/prime-radiant/)
40
+ <!-- At first release / first coverage upload these join (kept out until true):
41
+ codecov, PyPI version — see NOTES/go-live-runbook.md -->
42
+
43
+ Calibrated forecasting, measured honestly. Two threads share this repo:
44
+
45
+ - **FluSight epi forecaster** (`prime_radiant.epi`) — CDC FluSight-format quantile
46
+ forecasts of weekly US influenza hospital admissions: LightGBM quantile regression
47
+ in the shape of the FluSight-2023/24-winning flusion model, ensembled with a
48
+ validated replica of CDC's own baseline. Pure statistical pipeline; no LLM calls.
49
+ - **Metaculus bot** (`prime_radiant.metaculus`) — LLM forecasting on binary
50
+ questions (parked at its retrieval phase; client + tests remain green).
51
+
52
+ ## Backtest results (three seasons, vintage-honest)
53
+
54
+ Rolling-origin backtests over 85 weekly origins across the 2023-24, 2024-25 and
55
+ 2025-26 seasons. Every forecast is trained only on the data snapshot a live
56
+ forecaster would have held on the Wednesday submission evening (the hub's git
57
+ history is the vintage store); scored against final truth as of 2026-07-09,
58
+ horizons 0-3, all 53 jurisdictions. `wis_rel` = mean WIS relative to the official
59
+ FluSight-baseline on the common task set across all six models (lower is better;
60
+ the official "scaled relative skill" collapses to this ratio on identical sets).
61
+ Coverage columns are computed on each model's full scored set; relative-skill
62
+ columns on the common set — `n` and `n_relative` in the CSVs disclose both.
63
+
64
+ **2025-26** — our model leads the table on the natural scale (on the log(x+1)
65
+ scale UMass-flusion edges it, 0.583 vs 0.585 — both columns are in the CSVs):
66
+
67
+ | model | wis_rel | 50% cov | 95% cov |
68
+ |---|---|---|---|
69
+ | **prime-radiant-lgbm** | **0.609** | 0.398 | 0.818 |
70
+ | UMass-flusion | 0.625 | 0.353 | 0.823 |
71
+ | FluSight-ensemble | 0.666 | 0.526 | 0.903 |
72
+ | prime-radiant-ensemble | 0.764 | 0.464 | 0.880 |
73
+ | FluSight-baseline | 1.000 | 0.433 | 0.864 |
74
+
75
+ **2024-25** — the multi-model ensembles beat us; we beat the baseline:
76
+
77
+ | model | wis_rel | 50% cov | 95% cov |
78
+ |---|---|---|---|
79
+ | UMass-flusion | 0.669 | 0.397 | 0.820 |
80
+ | FluSight-ensemble | 0.675 | 0.519 | 0.818 |
81
+ | **prime-radiant-lgbm** | **0.796** | 0.338 | 0.738 |
82
+ | prime-radiant-ensemble | 0.900 | 0.371 | 0.744 |
83
+ | FluSight-baseline | 1.000 | 0.317 | 0.719 |
84
+
85
+ **2023-24** — flusion dominates; our ensemble edges FluSight-ensemble:
86
+
87
+ | model | wis_rel | 50% cov | 95% cov |
88
+ |---|---|---|---|
89
+ | UMass-flusion | 0.569 | 0.569 | 0.964 |
90
+ | **prime-radiant-ensemble** | **0.716** | 0.422 | 0.909 |
91
+ | FluSight-ensemble | 0.730 | 0.488 | 0.920 |
92
+ | prime-radiant-lgbm | 0.841 | 0.369 | 0.831 |
93
+ | FluSight-baseline | 1.000 | 0.282 | 0.897 |
94
+
95
+ Full tables (per-horizon rows, log-scale variants, AE-median, task counts):
96
+ [`reports/backtest_<season>.csv`](reports/). Calibration:
97
+
98
+ ![Calibration curves](reports/calibration.png)
99
+
100
+ ## Honest framing
101
+
102
+ - **The backtest is net biased against us, and we keep it that way.** Adversarial
103
+ verification proved that on three 2024-25 holiday weeks the official baseline's
104
+ run saw data committed Thursday+, which our live-Wednesday vintage discipline
105
+ refuses; on the 24 information-equal origins our relative WIS improves (lgbm
106
+ ~0.73). It is not one-way: at three October-2023 origins our vintages were
107
+ fresher than what the official run used. One 2023-24 origin (2024-04-13) had a
108
+ week-stale vintage — for our models only; the officials ran on fresh data there.
109
+ - **Our intervals are too narrow.** The lgbm's 50% intervals cover 34-40% and its
110
+ 95% intervals 74-83% — under-dispersed, visible in the calibration curves.
111
+ FluSight-ensemble is better calibrated even where we beat it on WIS. This is
112
+ the main modeling debt; season-level bagging (flusion's stabilizer) is the
113
+ known lever.
114
+ - **Wins and losses both stand.** We lead 2025-26 outright; UMass-flusion beats
115
+ us clearly in 2023-24 and 2024-25. The 2023-24 lgbm ran on ~1.2 seasons of
116
+ training history.
117
+ - **The scorer and baseline replica are self-validating — for the season they
118
+ were validated on.** The replica reproduces the official 2024-25 baseline to
119
+ relative WIS 0.99999 on fingerprint-matched vintages (Phase B). In 2023-24 the
120
+ replica scores 0.976 vs the official baseline — a spread-construction
121
+ divergence at matched anchors that Phase B's validation (2024-25-era official
122
+ code) does not cover; adversarially checked: substituting the actual official
123
+ baseline into our ensemble changes no table ordering.
124
+
125
+ ## Methodology in one paragraph
126
+
127
+ NHSN weekly admissions (hub target data, git-vintaged) → per-100k 4th-root
128
+ transform with per-location scale/center fitted per origin → pooled LightGBM
129
+ quantile regression (23 levels, horizon-as-feature, deterministic, exact-pinned
130
+ 4.7.0) → monotone sort, inversion, integer rounding at the hub boundary →
131
+ per-quantile-median ensemble with the baseline replica → WIS/coverage scoring on
132
+ natural and log(x+1) scales, pinned-truth, common-task relative skill. Every
133
+ stage boundary carries a pandera contract; leakage invariants (vintage as-of,
134
+ feature cut, scaler fit) are hypothesis-tested properties.
135
+
136
+ ## Setup
137
+
138
+ ```sh
139
+ uv sync --all-groups
140
+ make check # offline gates: ruff + format + pyright + pytest-cov
141
+ make test-integration # network: real hub clone, S3 benchmarks, gate + reports
142
+ ```
143
+
144
+ Requires Homebrew `libomp` on macOS for LightGBM. Built with
145
+ [Claude Code](https://claude.com/claude-code) (agentic coding assistant by
146
+ Anthropic); every phase was adversarially verified
147
+ by refuter agents before being declared done — see `NOTES/restart.md` and the
148
+ `[claude]`-prefixed commit history.
149
+
150
+ ## Dashboard
151
+
152
+ **Live: https://huggingface.co/spaces/jeremygracey-ai/prime-radiant**
153
+
154
+ A Gradio dashboard (US choropleth of predicted 3-week change, per-state fan
155
+ charts, reliability curves, model-vs-baseline league tables) serves the frozen
156
+ backtest record from `serve_data/`, a ~1.7MB precomputed bundle:
157
+
158
+ ```sh
159
+ make bundle # offline: rebuilds serve_data/ deterministically
160
+ uv run python dashboard/app.py # local: http://127.0.0.1:7860
161
+ ```
162
+
163
+ The Space (`jeremygracey-ai/prime-radiant`, CPU-basic) installs only
164
+ gradio/plotly/pandas/pyarrow — never this package, so no LightGBM/libomp and no
165
+ hub clone at serve time. `.github/workflows/space-deploy.yml` stages and
166
+ validates the Space tree on every dispatch; the actual push is triple-gated
167
+ (manual `deploy=true` + `SPACE_LIVE=1` repo var + `HF_TOKEN` secret) and stays
168
+ inert until go-live.
169
+
170
+ ## Limitations
171
+
172
+ - Single data source (NHSN), single target (`wk inc flu hosp`); no rate-change /
173
+ peak / ED-visit targets yet.
174
+ - Interval under-dispersion as above; no season-bagging yet.
175
+ - 2022-23 is not backtestable (the hub's vintage history begins Oct 2023).
176
+ - Live submission machinery exists (Phase E) but is structurally inert: the
177
+ weekly workflow's live path is triple-gated and unimplemented past its gates,
178
+ and nothing submits anywhere without the go-live runbook being executed by
179
+ hand. The dashboard serves a frozen bundle, not a live feed.
@@ -0,0 +1,147 @@
1
+ # Prime Radiant
2
+
3
+ [![ci](https://github.com/JeremyGracey-AI/prime-radiant/actions/workflows/ci.yml/badge.svg)](https://github.com/JeremyGracey-AI/prime-radiant/actions/workflows/ci.yml)
4
+ [![AI-USE: declared](https://img.shields.io/badge/AI--USE-declared-2ea44f)](./AI-USE.md)
5
+ [![License: MIT](https://img.shields.io/badge/License-MIT-yellow.svg)](./LICENSE)
6
+ [![Dashboard](https://img.shields.io/badge/%F0%9F%A4%97%20Space-live-blue)](https://huggingface.co/spaces/jeremygracey-ai/prime-radiant)
7
+ [![docs](https://img.shields.io/badge/docs-live-2ea44f)](https://jeremygracey-ai.github.io/prime-radiant/)
8
+ <!-- At first release / first coverage upload these join (kept out until true):
9
+ codecov, PyPI version — see NOTES/go-live-runbook.md -->
10
+
11
+ Calibrated forecasting, measured honestly. Two threads share this repo:
12
+
13
+ - **FluSight epi forecaster** (`prime_radiant.epi`) — CDC FluSight-format quantile
14
+ forecasts of weekly US influenza hospital admissions: LightGBM quantile regression
15
+ in the shape of the FluSight-2023/24-winning flusion model, ensembled with a
16
+ validated replica of CDC's own baseline. Pure statistical pipeline; no LLM calls.
17
+ - **Metaculus bot** (`prime_radiant.metaculus`) — LLM forecasting on binary
18
+ questions (parked at its retrieval phase; client + tests remain green).
19
+
20
+ ## Backtest results (three seasons, vintage-honest)
21
+
22
+ Rolling-origin backtests over 85 weekly origins across the 2023-24, 2024-25 and
23
+ 2025-26 seasons. Every forecast is trained only on the data snapshot a live
24
+ forecaster would have held on the Wednesday submission evening (the hub's git
25
+ history is the vintage store); scored against final truth as of 2026-07-09,
26
+ horizons 0-3, all 53 jurisdictions. `wis_rel` = mean WIS relative to the official
27
+ FluSight-baseline on the common task set across all six models (lower is better;
28
+ the official "scaled relative skill" collapses to this ratio on identical sets).
29
+ Coverage columns are computed on each model's full scored set; relative-skill
30
+ columns on the common set — `n` and `n_relative` in the CSVs disclose both.
31
+
32
+ **2025-26** — our model leads the table on the natural scale (on the log(x+1)
33
+ scale UMass-flusion edges it, 0.583 vs 0.585 — both columns are in the CSVs):
34
+
35
+ | model | wis_rel | 50% cov | 95% cov |
36
+ |---|---|---|---|
37
+ | **prime-radiant-lgbm** | **0.609** | 0.398 | 0.818 |
38
+ | UMass-flusion | 0.625 | 0.353 | 0.823 |
39
+ | FluSight-ensemble | 0.666 | 0.526 | 0.903 |
40
+ | prime-radiant-ensemble | 0.764 | 0.464 | 0.880 |
41
+ | FluSight-baseline | 1.000 | 0.433 | 0.864 |
42
+
43
+ **2024-25** — the multi-model ensembles beat us; we beat the baseline:
44
+
45
+ | model | wis_rel | 50% cov | 95% cov |
46
+ |---|---|---|---|
47
+ | UMass-flusion | 0.669 | 0.397 | 0.820 |
48
+ | FluSight-ensemble | 0.675 | 0.519 | 0.818 |
49
+ | **prime-radiant-lgbm** | **0.796** | 0.338 | 0.738 |
50
+ | prime-radiant-ensemble | 0.900 | 0.371 | 0.744 |
51
+ | FluSight-baseline | 1.000 | 0.317 | 0.719 |
52
+
53
+ **2023-24** — flusion dominates; our ensemble edges FluSight-ensemble:
54
+
55
+ | model | wis_rel | 50% cov | 95% cov |
56
+ |---|---|---|---|
57
+ | UMass-flusion | 0.569 | 0.569 | 0.964 |
58
+ | **prime-radiant-ensemble** | **0.716** | 0.422 | 0.909 |
59
+ | FluSight-ensemble | 0.730 | 0.488 | 0.920 |
60
+ | prime-radiant-lgbm | 0.841 | 0.369 | 0.831 |
61
+ | FluSight-baseline | 1.000 | 0.282 | 0.897 |
62
+
63
+ Full tables (per-horizon rows, log-scale variants, AE-median, task counts):
64
+ [`reports/backtest_<season>.csv`](reports/). Calibration:
65
+
66
+ ![Calibration curves](reports/calibration.png)
67
+
68
+ ## Honest framing
69
+
70
+ - **The backtest is net biased against us, and we keep it that way.** Adversarial
71
+ verification proved that on three 2024-25 holiday weeks the official baseline's
72
+ run saw data committed Thursday+, which our live-Wednesday vintage discipline
73
+ refuses; on the 24 information-equal origins our relative WIS improves (lgbm
74
+ ~0.73). It is not one-way: at three October-2023 origins our vintages were
75
+ fresher than what the official run used. One 2023-24 origin (2024-04-13) had a
76
+ week-stale vintage — for our models only; the officials ran on fresh data there.
77
+ - **Our intervals are too narrow.** The lgbm's 50% intervals cover 34-40% and its
78
+ 95% intervals 74-83% — under-dispersed, visible in the calibration curves.
79
+ FluSight-ensemble is better calibrated even where we beat it on WIS. This is
80
+ the main modeling debt; season-level bagging (flusion's stabilizer) is the
81
+ known lever.
82
+ - **Wins and losses both stand.** We lead 2025-26 outright; UMass-flusion beats
83
+ us clearly in 2023-24 and 2024-25. The 2023-24 lgbm ran on ~1.2 seasons of
84
+ training history.
85
+ - **The scorer and baseline replica are self-validating — for the season they
86
+ were validated on.** The replica reproduces the official 2024-25 baseline to
87
+ relative WIS 0.99999 on fingerprint-matched vintages (Phase B). In 2023-24 the
88
+ replica scores 0.976 vs the official baseline — a spread-construction
89
+ divergence at matched anchors that Phase B's validation (2024-25-era official
90
+ code) does not cover; adversarially checked: substituting the actual official
91
+ baseline into our ensemble changes no table ordering.
92
+
93
+ ## Methodology in one paragraph
94
+
95
+ NHSN weekly admissions (hub target data, git-vintaged) → per-100k 4th-root
96
+ transform with per-location scale/center fitted per origin → pooled LightGBM
97
+ quantile regression (23 levels, horizon-as-feature, deterministic, exact-pinned
98
+ 4.7.0) → monotone sort, inversion, integer rounding at the hub boundary →
99
+ per-quantile-median ensemble with the baseline replica → WIS/coverage scoring on
100
+ natural and log(x+1) scales, pinned-truth, common-task relative skill. Every
101
+ stage boundary carries a pandera contract; leakage invariants (vintage as-of,
102
+ feature cut, scaler fit) are hypothesis-tested properties.
103
+
104
+ ## Setup
105
+
106
+ ```sh
107
+ uv sync --all-groups
108
+ make check # offline gates: ruff + format + pyright + pytest-cov
109
+ make test-integration # network: real hub clone, S3 benchmarks, gate + reports
110
+ ```
111
+
112
+ Requires Homebrew `libomp` on macOS for LightGBM. Built with
113
+ [Claude Code](https://claude.com/claude-code) (agentic coding assistant by
114
+ Anthropic); every phase was adversarially verified
115
+ by refuter agents before being declared done — see `NOTES/restart.md` and the
116
+ `[claude]`-prefixed commit history.
117
+
118
+ ## Dashboard
119
+
120
+ **Live: https://huggingface.co/spaces/jeremygracey-ai/prime-radiant**
121
+
122
+ A Gradio dashboard (US choropleth of predicted 3-week change, per-state fan
123
+ charts, reliability curves, model-vs-baseline league tables) serves the frozen
124
+ backtest record from `serve_data/`, a ~1.7MB precomputed bundle:
125
+
126
+ ```sh
127
+ make bundle # offline: rebuilds serve_data/ deterministically
128
+ uv run python dashboard/app.py # local: http://127.0.0.1:7860
129
+ ```
130
+
131
+ The Space (`jeremygracey-ai/prime-radiant`, CPU-basic) installs only
132
+ gradio/plotly/pandas/pyarrow — never this package, so no LightGBM/libomp and no
133
+ hub clone at serve time. `.github/workflows/space-deploy.yml` stages and
134
+ validates the Space tree on every dispatch; the actual push is triple-gated
135
+ (manual `deploy=true` + `SPACE_LIVE=1` repo var + `HF_TOKEN` secret) and stays
136
+ inert until go-live.
137
+
138
+ ## Limitations
139
+
140
+ - Single data source (NHSN), single target (`wk inc flu hosp`); no rate-change /
141
+ peak / ED-visit targets yet.
142
+ - Interval under-dispersion as above; no season-bagging yet.
143
+ - 2022-23 is not backtestable (the hub's vintage history begins Oct 2023).
144
+ - Live submission machinery exists (Phase E) but is structurally inert: the
145
+ weekly workflow's live path is triple-gated and unimplemented past its gates,
146
+ and nothing submits anywhere without the go-live runbook being executed by
147
+ hand. The dashboard serves a frozen bundle, not a live feed.
@@ -0,0 +1,40 @@
1
+ """Commit parser for python-semantic-release tolerating `[actor] type(scope):`.
2
+
3
+ The house convention prefixes agent-authored commits with `[claude] ` before a
4
+ conventional-commits subject. psr's stock ConventionalCommitParser anchors its
5
+ regex at string start, so every prefixed commit fails to parse (verified
6
+ empirically against psr 10.6.2). This subclass strips exactly one leading
7
+ `[word] ` token, then defers to the stock parser.
8
+
9
+ Loaded by psr via the FILE-PATH form in pyproject
10
+ (`commit_parser = "psr_parser.py:ActorPrefixConventionalParser"`) — the bare
11
+ module form fails because the repo root is not on sys.path when psr imports.
12
+ psr itself never lives in the project env (its tomlkit floor conflicts with
13
+ gradio's cap), so the semantic_release import below only resolves under
14
+ `uvx python-semantic-release==10.6.2` or the release workflow's action; the
15
+ pure strip logic stays importable (and unit-tested) without it.
16
+ """
17
+
18
+ import re
19
+
20
+ _ACTOR_PREFIX = re.compile(r"^\[[\w.-]+\]\s+")
21
+
22
+
23
+ def strip_actor_prefix(message: str) -> str:
24
+ """Remove one leading `[actor] ` token from a commit message, if present."""
25
+ return _ACTOR_PREFIX.sub("", message, count=1)
26
+
27
+
28
+ try:
29
+ # psr is never in the project env (tomlkit conflict with gradio) — resolved
30
+ # only under uvx / the release workflow, hence the scoped pyright ignore.
31
+ from semantic_release.commit_parser.conventional import ( # pyright: ignore[reportMissingImports]
32
+ ConventionalCommitParser,
33
+ )
34
+ except ImportError: # pragma: no cover — project env; psr runs isolated
35
+ ConventionalCommitParser = None # type: ignore[assignment, misc]
36
+ else:
37
+
38
+ class ActorPrefixConventionalParser(ConventionalCommitParser):
39
+ def parse_message(self, message: str): # noqa: ANN201 — psr's own signature
40
+ return super().parse_message(strip_actor_prefix(message))
@@ -0,0 +1,160 @@
1
+ [build-system]
2
+ requires = ["hatchling"]
3
+ build-backend = "hatchling.build"
4
+
5
+ [project]
6
+ name = "prime-radiant"
7
+ version = "0.1.0"
8
+ description = "Calibrated forecasting: Metaculus LLM bot (Brier) and CDC FluSight quantile forecaster (WIS)."
9
+ readme = "README.md"
10
+ requires-python = ">=3.11"
11
+ license = "MIT"
12
+ authors = [{ name = "Jeremy Gracey", email = "jeremy.a.gracey@gmail.com" }]
13
+ keywords = ["forecasting", "calibration", "flusight", "metaculus", "quantile", "epidemiology"]
14
+ classifiers = [
15
+ "Development Status :: 3 - Alpha",
16
+ "Intended Audience :: Science/Research",
17
+ "License :: OSI Approved :: MIT License",
18
+ "Programming Language :: Python :: 3.11",
19
+ "Programming Language :: Python :: 3.12",
20
+ "Programming Language :: Python :: 3.13",
21
+ "Topic :: Scientific/Engineering",
22
+ ]
23
+ dependencies = [
24
+ # Metaculus thread
25
+ "forecasting-tools>=0.2.92",
26
+ "pydantic>=2.9",
27
+ "pydantic-settings>=2.5",
28
+ "python-dotenv>=1.0",
29
+ # FluSight epi thread (data layer)
30
+ "pandas>=2.2",
31
+ "pandera>=0.20",
32
+ "epiweeks>=2.3",
33
+ "httpx>=0.27",
34
+ "pyarrow>=17",
35
+ # EXACT pin: LightGBM determinism holds only within one compiled binary —
36
+ # docs state results differ across versions/compilers/systems.
37
+ "lightgbm==4.7.0",
38
+ # <3.12 cap: report PNGs are committed artifacts; hold the rendering series
39
+ # stable within a phase so regeneration diffs stay meaningful.
40
+ "matplotlib>=3.11,<3.12",
41
+ "pyyaml>=6.0.3",
42
+ ]
43
+
44
+ [project.scripts]
45
+ prime-radiant = "prime_radiant.epi.cli:main"
46
+
47
+ [project.urls]
48
+ Homepage = "https://github.com/JeremyGracey-AI/prime-radiant"
49
+ Repository = "https://github.com/JeremyGracey-AI/prime-radiant"
50
+
51
+ [dependency-groups]
52
+ dev = [
53
+ "pytest>=8",
54
+ "pytest-cov>=5",
55
+ "hypothesis>=6.100",
56
+ "pytest-recording>=0.13",
57
+ # floors match .pre-commit-config.yaml revs so hooks and `uv run` agree
58
+ "ruff>=0.16.5",
59
+ "pyright>=1.1.411",
60
+ "jsonschema>=4.26.0",
61
+ "types-pyyaml>=6.0.12.20260815",
62
+ ]
63
+ # NOTE: python-semantic-release is deliberately NOT a dependency group — its
64
+ # tomlkit>=0.15 floor conflicts with gradio 6.26's tomlkit<0.15 cap. It runs
65
+ # isolated instead: `uvx python-semantic-release==10.6.2` locally, the
66
+ # containerized psr action (same version) in release.yml.
67
+ docs = [
68
+ "mkdocs-material>=9.7",
69
+ ]
70
+ # Dev-only deps for the HF Space dashboard (dashboard/). The Space itself never
71
+ # installs this package — its requirements.txt is generated by space-deploy.yml.
72
+ dashboard = [
73
+ # ==6.26.*: match the Space's README sdk_version exactly — FetchMerck commit
74
+ # 8732e99 is the recorded cost of developing against a different Gradio than
75
+ # the Space runs.
76
+ "gradio==6.26.*",
77
+ # <7 cap: plotly 7 flips layout.geo.fitbounds default, removes
78
+ # figure_factory.create_choropleth and *mapbox traces, and Gradio's bundled
79
+ # plotly.js has no verified plotly-7 support. Declared directly per the
80
+ # Phase F brief (transitive-only via forecasting-tools before this).
81
+ "plotly>=6.9,<7",
82
+ ]
83
+
84
+ [tool.hatch.build.targets.wheel]
85
+ packages = ["src/prime_radiant"]
86
+
87
+ # Default hatchling sdists ship the entire working tree (NOTES, handoff briefs,
88
+ # serve_data) — scope to the package + release-relevant files. LICENSE rides
89
+ # along via the license-files mechanism.
90
+ [tool.hatch.build.targets.sdist]
91
+ only-include = [
92
+ "src/prime_radiant",
93
+ "psr_parser.py",
94
+ "README.md",
95
+ "CHANGELOG.md",
96
+ "CITATION.cff",
97
+ "uv.lock",
98
+ ]
99
+
100
+ [tool.ruff]
101
+ line-length = 100
102
+ target-version = "py311"
103
+
104
+ [tool.ruff.lint]
105
+ select = ["E", "F", "W", "I", "UP", "B", "SIM"]
106
+
107
+ [tool.ruff.lint.isort]
108
+ # dashboard/ modules ship flat to the HF Space root and import as siblings.
109
+ known-first-party = ["prime_radiant", "panel_data", "panel_plots"]
110
+
111
+ [tool.pyright]
112
+ pythonVersion = "3.11"
113
+ typeCheckingMode = "basic"
114
+ venvPath = "."
115
+ venv = ".venv"
116
+
117
+ # dashboard/ is not part of the package: it deploys flat to the Space root, so
118
+ # its modules resolve as siblings there and need the same path here and in tests.
119
+ [[tool.pyright.executionEnvironments]]
120
+ root = "dashboard"
121
+ extraPaths = ["dashboard"]
122
+
123
+ [[tool.pyright.executionEnvironments]]
124
+ root = "tests"
125
+ extraPaths = ["dashboard"]
126
+
127
+ [tool.semantic_release]
128
+ # Runs isolated: `uvx python-semantic-release==10.6.2` locally, the psr action
129
+ # (same version) in release.yml — see the dependency-group note above.
130
+ version_toml = ["pyproject.toml:project.version"]
131
+ # FILE-PATH form is load-bearing: the bare module form cannot import from cwd.
132
+ commit_parser = "psr_parser.py:ActorPrefixConventionalParser"
133
+ commit_message = "[claude] chore(release): v{version}"
134
+ # House rule: one git identity everywhere, agent commits included.
135
+ commit_author = "Jeremy Gracey <jeremy.a.gracey@gmail.com>"
136
+ # uv.lock carries the project's own version; regenerate AFTER stamping and
137
+ # ship it inside the release commit.
138
+ build_command = "uv lock && uv build"
139
+ assets = ["uv.lock"]
140
+ allow_zero_version = true
141
+ major_on_zero = false
142
+ tag_format = "v{version}"
143
+
144
+ [tool.semantic_release.changelog]
145
+ mode = "update"
146
+
147
+ [tool.pytest.ini_options]
148
+ testpaths = ["tests"]
149
+ markers = [
150
+ "unit: fast, offline, fixture-backed",
151
+ "integration: touches network or the real hub clone",
152
+ "contract: schema/property invariants (offline)",
153
+ "e2e: full pipeline runs",
154
+ ]
155
+
156
+ [tool.coverage.run]
157
+ source = ["prime_radiant"]
158
+
159
+ [tool.coverage.report]
160
+ show_missing = true
@@ -0,0 +1 @@
1
+ """Prime Radiant: calibrated LLM forecasting bot for Metaculus binary questions."""
@@ -0,0 +1,18 @@
1
+ """Settings loaded from environment / .env — secrets and cost caps.
2
+
3
+ Cost caps are hard limits enforced by the run loop (Phase C): the bot must
4
+ stop spending when a cap is hit, and every run logs token spend against it.
5
+ """
6
+
7
+ from pydantic_settings import BaseSettings, SettingsConfigDict
8
+
9
+
10
+ class Settings(BaseSettings):
11
+ model_config = SettingsConfigDict(env_file=".env", env_file_encoding="utf-8", extra="ignore")
12
+
13
+ anthropic_api_key: str = ""
14
+ metaculus_token: str = ""
15
+ news_api_key: str = ""
16
+
17
+ per_question_budget_usd: float = 0.25
18
+ per_run_budget_usd: float = 2.50
File without changes
@@ -0,0 +1 @@
1
+ """Rolling-origin backtesting over target-data vintages."""