everestapi 0.3.0__tar.gz → 0.3.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {everestapi-0.3.0/src/everestapi.egg-info → everestapi-0.3.2}/PKG-INFO +24 -13
- {everestapi-0.3.0 → everestapi-0.3.2}/README.md +23 -12
- {everestapi-0.3.0 → everestapi-0.3.2}/src/everestapi/__init__.py +1 -1
- {everestapi-0.3.0 → everestapi-0.3.2}/src/everestapi/client.py +112 -37
- {everestapi-0.3.0 → everestapi-0.3.2}/src/everestapi/mcp/server.py +101 -24
- {everestapi-0.3.0 → everestapi-0.3.2}/src/everestapi/scoring.py +15 -13
- {everestapi-0.3.0 → everestapi-0.3.2/src/everestapi.egg-info}/PKG-INFO +24 -13
- {everestapi-0.3.0 → everestapi-0.3.2}/tests/test_diagnostics.py +113 -2
- {everestapi-0.3.0 → everestapi-0.3.2}/tests/test_eve1087_cpu_tier.py +1 -1
- {everestapi-0.3.0 → everestapi-0.3.2}/tests/test_mcp_and_models.py +9 -2
- {everestapi-0.3.0 → everestapi-0.3.2}/tests/test_scoring.py +10 -8
- {everestapi-0.3.0 → everestapi-0.3.2}/LICENSE +0 -0
- {everestapi-0.3.0 → everestapi-0.3.2}/pyproject.toml +0 -0
- {everestapi-0.3.0 → everestapi-0.3.2}/setup.cfg +0 -0
- {everestapi-0.3.0 → everestapi-0.3.2}/src/everestapi/__main__.py +0 -0
- {everestapi-0.3.0 → everestapi-0.3.2}/src/everestapi/cli.py +0 -0
- {everestapi-0.3.0 → everestapi-0.3.2}/src/everestapi/mcp/__init__.py +0 -0
- {everestapi-0.3.0 → everestapi-0.3.2}/src/everestapi/mcp/__main__.py +0 -0
- {everestapi-0.3.0 → everestapi-0.3.2}/src/everestapi/plots.py +0 -0
- {everestapi-0.3.0 → everestapi-0.3.2}/src/everestapi/types.py +0 -0
- {everestapi-0.3.0 → everestapi-0.3.2}/src/everestapi.egg-info/SOURCES.txt +0 -0
- {everestapi-0.3.0 → everestapi-0.3.2}/src/everestapi.egg-info/dependency_links.txt +0 -0
- {everestapi-0.3.0 → everestapi-0.3.2}/src/everestapi.egg-info/entry_points.txt +0 -0
- {everestapi-0.3.0 → everestapi-0.3.2}/src/everestapi.egg-info/requires.txt +0 -0
- {everestapi-0.3.0 → everestapi-0.3.2}/src/everestapi.egg-info/top_level.txt +0 -0
- {everestapi-0.3.0 → everestapi-0.3.2}/tests/test_cli.py +0 -0
- {everestapi-0.3.0 → everestapi-0.3.2}/tests/test_client.py +0 -0
- {everestapi-0.3.0 → everestapi-0.3.2}/tests/test_eve953_mcp_progress.py +0 -0
- {everestapi-0.3.0 → everestapi-0.3.2}/tests/test_eve957_mcp_annotations_resources.py +0 -0
- {everestapi-0.3.0 → everestapi-0.3.2}/tests/test_eve959_toolsets.py +0 -0
- {everestapi-0.3.0 → everestapi-0.3.2}/tests/test_eve967_mcp_discoverability.py +0 -0
- {everestapi-0.3.0 → everestapi-0.3.2}/tests/test_json_or_raise.py +0 -0
- {everestapi-0.3.0 → everestapi-0.3.2}/tests/test_prediction_range.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: everestapi
|
|
3
|
-
Version: 0.3.
|
|
3
|
+
Version: 0.3.2
|
|
4
4
|
Summary: Python SDK for the Everesteer prediction tournament platform
|
|
5
5
|
Author-email: Everesteer <support@everesteer.ai>
|
|
6
6
|
License-Expression: MIT
|
|
@@ -116,20 +116,31 @@ everestapi submit --model my-model --file predictions.parquet
|
|
|
116
116
|
|
|
117
117
|
### Data & diagnostics
|
|
118
118
|
|
|
119
|
-
The hackathon is a display-only diagnostics event. **Tune
|
|
120
|
-
|
|
121
|
-
|
|
122
|
-
|
|
123
|
-
|
|
119
|
+
The hackathon is a display-only diagnostics event. **Tune offline on the labeled
|
|
120
|
+
`train` set** (features + `target_*` columns), then **predict on the blank-target
|
|
121
|
+
`validation` (leaderboard) set and submit predictions plus your model `.pkl`**
|
|
122
|
+
(required; store-only, never executed). Each upload is scored server-side on
|
|
123
|
+
**two windows**: the public leaderboard window (ranked live during the event)
|
|
124
|
+
and a later held-out **final window that stays sealed until the event's
|
|
125
|
+
reveal**. Both rank out-of-sample CORR on `target_everest_20`; in-sample fit is
|
|
126
|
+
not rewarded, and the answers are never downloadable. After submissions close,
|
|
127
|
+
pick up to 2 of your models as final entries during the grace window (otherwise
|
|
128
|
+
your best 2 public models are entered automatically).
|
|
124
129
|
|
|
125
130
|
```python
|
|
126
|
-
# Labeled
|
|
131
|
+
# Labeled training set — tune offline with everestapi.scoring on your own holdout:
|
|
132
|
+
api.download_dataset(universe="futures", split="train")
|
|
133
|
+
|
|
134
|
+
# Blank-target leaderboard set (features + id; target columns all-NaN).
|
|
135
|
+
# Predict on its ids, then submit with your model pickle:
|
|
127
136
|
api.download_dataset(universe="futures", split="validation")
|
|
137
|
+
api.submit_validation_diagnostics(
|
|
138
|
+
model_id="my-model", predictions=df, model_pkl="my_model.pkl"
|
|
139
|
+
)
|
|
128
140
|
|
|
129
|
-
#
|
|
130
|
-
#
|
|
131
|
-
api.
|
|
132
|
-
api.submit_validation_diagnostics(model_id="my-model", predictions=df)
|
|
141
|
+
api.get_diagnostics_leaderboard() # public board (live)
|
|
142
|
+
api.get_diagnostics_leaderboard(window="final") # sealed until reveal
|
|
143
|
+
api.set_final_selection(["my-model", "my-other"]) # grace window, up to 2
|
|
133
144
|
|
|
134
145
|
api.get_dataset_info(universe="futures")
|
|
135
146
|
api.get_diagnostics(model_id="my-model")
|
|
@@ -194,7 +205,7 @@ api.claim_payout(model_id="my-model", round_id="42")
|
|
|
194
205
|
|
|
195
206
|
### Score validation predictions offline
|
|
196
207
|
|
|
197
|
-
Reproduce the server's **exact** scoring — CORR20, AIMC20,
|
|
208
|
+
Reproduce the server's **exact** scoring — CORR20, AIMC20, NCORR — *before* you
|
|
198
209
|
submit, so you stop guessing the sign of your signal ("submit raw and negated, let
|
|
199
210
|
the server decide"). The `everestapi.scoring` functions are a verbatim port of the
|
|
200
211
|
platform's scoring engine (verified equal to 1e-12), so your offline number **is**
|
|
@@ -228,7 +239,7 @@ scoring.score(
|
|
|
228
239
|
preds_e, target_e, ai_model=consensus_e, features=features_e,
|
|
229
240
|
corr_weight=my_corr_weight, aimc_weight=my_aimc_weight,
|
|
230
241
|
)
|
|
231
|
-
# -> {"corr20", "aimc20", "payout", "
|
|
242
|
+
# -> {"corr20", "aimc20", "payout", "ncorr", "feature_exposure"}
|
|
232
243
|
```
|
|
233
244
|
|
|
234
245
|
**Sanity-check your pipeline against the example predictions.** The published
|
|
@@ -77,20 +77,31 @@ everestapi submit --model my-model --file predictions.parquet
|
|
|
77
77
|
|
|
78
78
|
### Data & diagnostics
|
|
79
79
|
|
|
80
|
-
The hackathon is a display-only diagnostics event. **Tune
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
|
|
80
|
+
The hackathon is a display-only diagnostics event. **Tune offline on the labeled
|
|
81
|
+
`train` set** (features + `target_*` columns), then **predict on the blank-target
|
|
82
|
+
`validation` (leaderboard) set and submit predictions plus your model `.pkl`**
|
|
83
|
+
(required; store-only, never executed). Each upload is scored server-side on
|
|
84
|
+
**two windows**: the public leaderboard window (ranked live during the event)
|
|
85
|
+
and a later held-out **final window that stays sealed until the event's
|
|
86
|
+
reveal**. Both rank out-of-sample CORR on `target_everest_20`; in-sample fit is
|
|
87
|
+
not rewarded, and the answers are never downloadable. After submissions close,
|
|
88
|
+
pick up to 2 of your models as final entries during the grace window (otherwise
|
|
89
|
+
your best 2 public models are entered automatically).
|
|
85
90
|
|
|
86
91
|
```python
|
|
87
|
-
# Labeled
|
|
92
|
+
# Labeled training set — tune offline with everestapi.scoring on your own holdout:
|
|
93
|
+
api.download_dataset(universe="futures", split="train")
|
|
94
|
+
|
|
95
|
+
# Blank-target leaderboard set (features + id; target columns all-NaN).
|
|
96
|
+
# Predict on its ids, then submit with your model pickle:
|
|
88
97
|
api.download_dataset(universe="futures", split="validation")
|
|
98
|
+
api.submit_validation_diagnostics(
|
|
99
|
+
model_id="my-model", predictions=df, model_pkl="my_model.pkl"
|
|
100
|
+
)
|
|
89
101
|
|
|
90
|
-
#
|
|
91
|
-
#
|
|
92
|
-
api.
|
|
93
|
-
api.submit_validation_diagnostics(model_id="my-model", predictions=df)
|
|
102
|
+
api.get_diagnostics_leaderboard() # public board (live)
|
|
103
|
+
api.get_diagnostics_leaderboard(window="final") # sealed until reveal
|
|
104
|
+
api.set_final_selection(["my-model", "my-other"]) # grace window, up to 2
|
|
94
105
|
|
|
95
106
|
api.get_dataset_info(universe="futures")
|
|
96
107
|
api.get_diagnostics(model_id="my-model")
|
|
@@ -155,7 +166,7 @@ api.claim_payout(model_id="my-model", round_id="42")
|
|
|
155
166
|
|
|
156
167
|
### Score validation predictions offline
|
|
157
168
|
|
|
158
|
-
Reproduce the server's **exact** scoring — CORR20, AIMC20,
|
|
169
|
+
Reproduce the server's **exact** scoring — CORR20, AIMC20, NCORR — *before* you
|
|
159
170
|
submit, so you stop guessing the sign of your signal ("submit raw and negated, let
|
|
160
171
|
the server decide"). The `everestapi.scoring` functions are a verbatim port of the
|
|
161
172
|
platform's scoring engine (verified equal to 1e-12), so your offline number **is**
|
|
@@ -189,7 +200,7 @@ scoring.score(
|
|
|
189
200
|
preds_e, target_e, ai_model=consensus_e, features=features_e,
|
|
190
201
|
corr_weight=my_corr_weight, aimc_weight=my_aimc_weight,
|
|
191
202
|
)
|
|
192
|
-
# -> {"corr20", "aimc20", "payout", "
|
|
203
|
+
# -> {"corr20", "aimc20", "payout", "ncorr", "feature_exposure"}
|
|
193
204
|
```
|
|
194
205
|
|
|
195
206
|
**Sanity-check your pipeline against the example predictions.** The published
|
|
@@ -364,6 +364,7 @@ class EverestAPI:
|
|
|
364
364
|
tournament: str = "futures",
|
|
365
365
|
target: str = "target_everest_20",
|
|
366
366
|
*,
|
|
367
|
+
model_pkl=None,
|
|
367
368
|
wait: bool = False,
|
|
368
369
|
poll_interval: float = 2.0,
|
|
369
370
|
timeout: float = 600.0,
|
|
@@ -371,14 +372,25 @@ class EverestAPI:
|
|
|
371
372
|
"""POST /api/v1/diagnostics/upload (multipart) — score out-of-sample predictions.
|
|
372
373
|
|
|
373
374
|
``predictions`` is a pandas DataFrame (``id`` + ``prediction`` columns) or a
|
|
374
|
-
path to a ``.parquet`` / ``.csv`` file,
|
|
375
|
-
set (``download_dataset(split="
|
|
376
|
-
|
|
377
|
-
|
|
378
|
-
|
|
379
|
-
|
|
380
|
-
|
|
381
|
-
|
|
375
|
+
path to a ``.parquet`` / ``.csv`` file, covering the ids of the served
|
|
376
|
+
blank-target validation set (``download_dataset(split="validation")`` — the
|
|
377
|
+
leaderboard set; the retired blind ``live`` split now 404s for hackathon
|
|
378
|
+
keys). The platform scores it server-side against a held-out answer key
|
|
379
|
+
that is never downloadable. In a dual-window event, the run is scored on
|
|
380
|
+
BOTH windows: the public leaderboard window (revealed live) and a
|
|
381
|
+
held-out final window whose metrics stay sealed until the event's reveal.
|
|
382
|
+
|
|
383
|
+
``model_pkl`` is a serialized model artifact (raw ``bytes`` or a path to a
|
|
384
|
+
``.pkl`` file) archived alongside the upload — store-only for audit, never
|
|
385
|
+
unpickled or executed server-side. **Required for hackathon-scoped keys**
|
|
386
|
+
(the server rejects the upload with 400 without it); ignored for standard
|
|
387
|
+
agents. Hackathon uploads are also capped per agent per event
|
|
388
|
+
(failed/cancelled runs free a slot).
|
|
389
|
+
|
|
390
|
+
Returns the 202 accept dict; with ``wait=True`` polls ``runs/{upload_id}``
|
|
391
|
+
until ``done`` (returns the run) and raises :class:`EverestError` on
|
|
392
|
+
``failed`` or timeout. Display-only; results also surface in the website
|
|
393
|
+
Validation Diagnostics rail.
|
|
382
394
|
"""
|
|
383
395
|
import io
|
|
384
396
|
import time
|
|
@@ -393,9 +405,20 @@ class EverestAPI:
|
|
|
393
405
|
data = f.read()
|
|
394
406
|
fname = str(predictions).rsplit("/", 1)[-1].rsplit("\\", 1)[-1]
|
|
395
407
|
|
|
408
|
+
files = {"file": (fname, data, "application/octet-stream")}
|
|
409
|
+
if model_pkl is not None:
|
|
410
|
+
if isinstance(model_pkl, bytes | bytearray):
|
|
411
|
+
pkl_data = bytes(model_pkl)
|
|
412
|
+
pkl_fname = "model.pkl"
|
|
413
|
+
else: # path
|
|
414
|
+
with open(model_pkl, "rb") as f:
|
|
415
|
+
pkl_data = f.read()
|
|
416
|
+
pkl_fname = str(model_pkl).rsplit("/", 1)[-1].rsplit("\\", 1)[-1]
|
|
417
|
+
files["model_pkl"] = (pkl_fname, pkl_data, "application/octet-stream")
|
|
418
|
+
|
|
396
419
|
resp = self._client.post(
|
|
397
420
|
"/api/v1/diagnostics/upload",
|
|
398
|
-
files=
|
|
421
|
+
files=files,
|
|
399
422
|
data={"model_id": model_id, "tournament": tournament, "target": target},
|
|
400
423
|
)
|
|
401
424
|
if resp.status_code >= 400:
|
|
@@ -426,16 +449,60 @@ class EverestAPI:
|
|
|
426
449
|
view: str = "agents",
|
|
427
450
|
tournament: str = "futures",
|
|
428
451
|
limit: int = 100,
|
|
452
|
+
offset: int = 0,
|
|
453
|
+
window: str = "leaderboard",
|
|
429
454
|
) -> dict:
|
|
430
455
|
"""GET /api/v1/diagnostics/leaderboard — global CORR20 ranking.
|
|
431
456
|
|
|
432
457
|
``view="agents"`` ranks participant model runs; ``view="benchmarks"`` ranks
|
|
433
|
-
the official benchmark models.
|
|
458
|
+
the official benchmark models. ``total`` in the response is the full board
|
|
459
|
+
size — page with ``limit``/``offset``; your own best row is appended past
|
|
460
|
+
the page (true ``rank`` intact) when it falls outside the slice.
|
|
461
|
+
|
|
462
|
+
``window`` (dual-window hackathon events): ``"leaderboard"`` (default) is
|
|
463
|
+
the public interim board. ``"final"`` (participants only) is the held-out
|
|
464
|
+
final board — ``sealed: true`` with empty entries until the event's
|
|
465
|
+
``reveal_at``, then one AGENT-level row per participant ranked by
|
|
466
|
+
``final_corr20``; which of an agent's final entries won is never
|
|
467
|
+
disclosed.
|
|
434
468
|
"""
|
|
435
469
|
return self._request(
|
|
436
470
|
"GET",
|
|
437
471
|
"/api/v1/diagnostics/leaderboard",
|
|
438
|
-
params={
|
|
472
|
+
params={
|
|
473
|
+
"view": view,
|
|
474
|
+
"window": window,
|
|
475
|
+
"tournament": tournament,
|
|
476
|
+
"limit": limit,
|
|
477
|
+
"offset": offset,
|
|
478
|
+
},
|
|
479
|
+
)
|
|
480
|
+
|
|
481
|
+
def get_final_selection(self) -> dict:
|
|
482
|
+
"""GET /api/v1/diagnostics/final-selection — your final entries (hackathon only).
|
|
483
|
+
|
|
484
|
+
Returns your currently tagged models (≤2), the selection ``phase``
|
|
485
|
+
(``during_event`` | ``selection`` | ``revealed``), ``selection_open``,
|
|
486
|
+
the event's ``ends_at``/``reveal_at``, and ``fallback_active`` (true when
|
|
487
|
+
no tags are set — the final board then uses your best-2 public models
|
|
488
|
+
automatically).
|
|
489
|
+
"""
|
|
490
|
+
return self._request("GET", "/api/v1/diagnostics/final-selection")
|
|
491
|
+
|
|
492
|
+
def set_final_selection(self, model_ids: list) -> dict:
|
|
493
|
+
"""PUT /api/v1/diagnostics/final-selection — pick your final entries.
|
|
494
|
+
|
|
495
|
+
Replace-set up to 2 of your own models (ids or names) whose best public
|
|
496
|
+
runs are scored on the held-out final window. Accepted ONLY during the
|
|
497
|
+
selection grace window (after the event's ``ends_at``, before the final
|
|
498
|
+
board reveals) — 409 outside it. An empty list clears your tags and
|
|
499
|
+
re-arms the best-2-public fallback; re-PUT to change picks until the
|
|
500
|
+
reveal.
|
|
501
|
+
"""
|
|
502
|
+
return self._request(
|
|
503
|
+
"PUT",
|
|
504
|
+
"/api/v1/diagnostics/final-selection",
|
|
505
|
+
json={"model_ids": list(model_ids)},
|
|
439
506
|
)
|
|
440
507
|
|
|
441
508
|
@staticmethod
|
|
@@ -448,8 +515,8 @@ class EverestAPI:
|
|
|
448
515
|
"""Render a run status dict (from get_diagnostics_run / wait=True) as a table."""
|
|
449
516
|
rows = [
|
|
450
517
|
("CORR20", status.get("corr20")),
|
|
451
|
-
("
|
|
452
|
-
("
|
|
518
|
+
("AIMC", status.get("aimc", status.get("bmc"))),
|
|
519
|
+
("NCORR", status.get("ncorr", status.get("fnc"))),
|
|
453
520
|
("Sharpe Ratio", status.get("sharpe_ratio")),
|
|
454
521
|
("Std Dev", status.get("std_dev")),
|
|
455
522
|
("Max Drawdown", status.get("max_drawdown")),
|
|
@@ -469,7 +536,7 @@ class EverestAPI:
|
|
|
469
536
|
"""Render a get_diagnostics_leaderboard dict as a table (agents or benchmarks view)."""
|
|
470
537
|
view = lb.get("view", "agents")
|
|
471
538
|
if view == "benchmarks":
|
|
472
|
-
hdr = f"{'#':>3} {'Label':<18} {'CORR20':>9} {'
|
|
539
|
+
hdr = f"{'#':>3} {'Label':<18} {'CORR20':>9} {'AIMC':>9} {'NCORR':>9} {'Sharpe':>9}"
|
|
473
540
|
lines = [
|
|
474
541
|
f"Diagnostics Leaderboard (benchmarks) — {lb.get('tournament', '')}",
|
|
475
542
|
hdr,
|
|
@@ -479,11 +546,13 @@ class EverestAPI:
|
|
|
479
546
|
mark = "*" if e.get("is_ensemble") else " "
|
|
480
547
|
lines.append(
|
|
481
548
|
f"{e.get('rank', '?'):>3}{mark} {e.get('label', ''):<18} "
|
|
482
|
-
f"{EverestAPI._fmt(e.get('corr20')):>9}
|
|
483
|
-
f"{EverestAPI._fmt(e.get('
|
|
549
|
+
f"{EverestAPI._fmt(e.get('corr20')):>9} "
|
|
550
|
+
f"{EverestAPI._fmt(e.get('aimc', e.get('bmc'))):>9} "
|
|
551
|
+
f"{EverestAPI._fmt(e.get('ncorr', e.get('fnc'))):>9} "
|
|
552
|
+
f"{EverestAPI._fmt(e.get('sharpe_ratio')):>9}"
|
|
484
553
|
)
|
|
485
554
|
else:
|
|
486
|
-
hdr = f"{'#':>3} {'Model':<14} {'Agent':<12} {'CORR20':>9} {'
|
|
555
|
+
hdr = f"{'#':>3} {'Model':<14} {'Agent':<12} {'CORR20':>9} {'AIMC':>9} {'NCORR':>9} {'Sharpe':>9}"
|
|
487
556
|
lines = [
|
|
488
557
|
f"Diagnostics Leaderboard (agents) — {lb.get('tournament', '')} "
|
|
489
558
|
f"(your best rank: {lb.get('your_best_rank')})",
|
|
@@ -495,8 +564,10 @@ class EverestAPI:
|
|
|
495
564
|
lines.append(
|
|
496
565
|
f"{e.get('rank', '?'):>3}{mark} {e.get('model_name', ''):<14} "
|
|
497
566
|
f"{e.get('agent_name', ''):<12} "
|
|
498
|
-
f"{EverestAPI._fmt(e.get('corr20')):>9}
|
|
499
|
-
f"{EverestAPI._fmt(e.get('
|
|
567
|
+
f"{EverestAPI._fmt(e.get('corr20')):>9} "
|
|
568
|
+
f"{EverestAPI._fmt(e.get('aimc', e.get('bmc'))):>9} "
|
|
569
|
+
f"{EverestAPI._fmt(e.get('ncorr', e.get('fnc'))):>9} "
|
|
570
|
+
f"{EverestAPI._fmt(e.get('sharpe_ratio')):>9}"
|
|
500
571
|
)
|
|
501
572
|
return "\n".join(lines)
|
|
502
573
|
|
|
@@ -596,18 +667,19 @@ class EverestAPI:
|
|
|
596
667
|
404 the current version is resolved and the download retried once.
|
|
597
668
|
|
|
598
669
|
Splits: ``train`` / ``validation`` (features + targets), ``live``
|
|
599
|
-
(features only, no targets). In hackathon mode, ``
|
|
600
|
-
|
|
601
|
-
|
|
602
|
-
|
|
603
|
-
|
|
604
|
-
|
|
605
|
-
downloadable
|
|
670
|
+
(features only, no targets). In hackathon mode, ``train`` is the LABELED
|
|
671
|
+
training set (features + all ``target_*`` columns) for offline tuning
|
|
672
|
+
(see :mod:`everestapi.scoring`), and ``validation`` is the BLANK-TARGET
|
|
673
|
+
leaderboard set (features + ``id``; target columns present but all-NaN)
|
|
674
|
+
that you predict on and submit via
|
|
675
|
+
:meth:`submit_validation_diagnostics`. The answers are held out
|
|
676
|
+
server-side and never downloadable; the retired blind ``live`` split
|
|
677
|
+
404s for hackathon keys.
|
|
606
678
|
|
|
607
679
|
Futures is served by the futures endpoint (``version`` is a no-op):
|
|
608
|
-
it returns the real bregen tree to full-scope keys and the hackathon
|
|
609
|
-
(labeled ``
|
|
610
|
-
|
|
680
|
+
it returns the real bregen tree to full-scope keys and the hackathon
|
|
681
|
+
tree (labeled ``train`` + blank-target ``validation``) to
|
|
682
|
+
hackathon-scoped keys.
|
|
611
683
|
|
|
612
684
|
Value semantics: observed feature values are cross-sectional
|
|
613
685
|
quintile bins ``0-4``; a feature value of ``-1.0`` means MISSING
|
|
@@ -775,7 +847,7 @@ class EverestAPI:
|
|
|
775
847
|
Returns the live payout formula and weights (CORR/AIMC/payout cap) read
|
|
776
848
|
straight from the running config at call time — never hardcoded, so it
|
|
777
849
|
can't drift from what's actually live — plus a plain-language
|
|
778
|
-
explanation of CORR, AIMC (leave-one-out), EAC, and
|
|
850
|
+
explanation of CORR, AIMC (leave-one-out), EAC, and NCORR. Read-only, no
|
|
779
851
|
auth, no per-agent data.
|
|
780
852
|
"""
|
|
781
853
|
return self._request("GET", "/api/v1/scoring")
|
|
@@ -896,7 +968,7 @@ class EverestAPI:
|
|
|
896
968
|
or ``model="custom"`` with a ``custom_model_fn`` factory; optionally
|
|
897
969
|
add ``custom_feature_fn`` for server-side feature engineering on the
|
|
898
970
|
resident data (no upload). The platform always runs exped-purged
|
|
899
|
-
cross-validation and computes canonical CORR20v2/AIMC/
|
|
971
|
+
cross-validation and computes canonical CORR20v2/AIMC/NCORR/EAC — note
|
|
900
972
|
the pre-round AIMC is an ESTIMATE vs a static consensus proxy, not
|
|
901
973
|
the live stake-weighted crowd score used at round close.
|
|
902
974
|
|
|
@@ -1117,8 +1189,9 @@ class EverestAPI:
|
|
|
1117
1189
|
|
|
1118
1190
|
With a hackathon-scoped key there is no live round — the response is a
|
|
1119
1191
|
diagnostics-mode payload (mode='diagnostics_hackathon') directing you to
|
|
1120
|
-
tune offline on the labeled
|
|
1121
|
-
|
|
1192
|
+
tune offline on the labeled train set, then predict on the blank-target
|
|
1193
|
+
validation (leaderboard) set and submit_validation_diagnostics
|
|
1194
|
+
(predictions + your model .pkl).
|
|
1122
1195
|
"""
|
|
1123
1196
|
return self._request(
|
|
1124
1197
|
"GET",
|
|
@@ -1136,10 +1209,12 @@ class EverestAPI:
|
|
|
1136
1209
|
"""GET /api/v1/get_started — mode-aware orientation.
|
|
1137
1210
|
|
|
1138
1211
|
Returns what to do next given this key's scope: the display-only
|
|
1139
|
-
diagnostics-hackathon loop (tune
|
|
1140
|
-
|
|
1141
|
-
|
|
1142
|
-
|
|
1212
|
+
diagnostics-hackathon loop (tune offline on the labeled train set,
|
|
1213
|
+
predict on the blank-target validation set, submit predictions + your
|
|
1214
|
+
model .pkl; scored on a public leaderboard window AND a held-out final
|
|
1215
|
+
window sealed until reveal — pick up to 2 final entries with
|
|
1216
|
+
:meth:`set_final_selection` during the post-close grace window), or the
|
|
1217
|
+
live futures tournament flow.
|
|
1143
1218
|
"""
|
|
1144
1219
|
return self._request("GET", "/api/v1/get_started")
|
|
1145
1220
|
|
|
@@ -267,7 +267,7 @@ TOOLS = [
|
|
|
267
267
|
"the live payout formula and weights (CORR/AIMC/payout cap) read "
|
|
268
268
|
"straight from the running config — never hardcoded, so it can't "
|
|
269
269
|
"drift from reality — plus a plain-language explanation of CORR, "
|
|
270
|
-
"AIMC (leave-one-out), EAC, and
|
|
270
|
+
"AIMC (leave-one-out), EAC, and NCORR. Read-only, no auth, no data."
|
|
271
271
|
),
|
|
272
272
|
"inputSchema": {
|
|
273
273
|
"type": "object",
|
|
@@ -307,7 +307,7 @@ TOOLS = [
|
|
|
307
307
|
},
|
|
308
308
|
{
|
|
309
309
|
"name": "download_dataset",
|
|
310
|
-
"description": "Download a dataset split (train/validation
|
|
310
|
+
"description": "Download a dataset split (train/validation) as a parquet file. Returns the local file path. In hackathon mode: 'train' is the LABELED training set (features + all target_* columns) for offline tuning; 'validation' is the BLANK-TARGET leaderboard set (features + target columns present but all-NaN) — you predict on its ids and submit via submit_validation_diagnostics. The answers are held out server-side and never downloadable; the retired blind 'live' split 404s for hackathon keys. VALUE SEMANTICS: observed feature values are cross-sectional quintile bins 0-4; feature value -1.0 means MISSING (source not yet onboarded for that instrument/date) — treat -1 as NaN or a distinct category, NEVER as an ordinal value below 0. Target NaN means uncomputable for that row (never imputed) — drop NaN-target rows when training on that target. Full notes: the dataset's eiq_metadata.json missing_value_note.",
|
|
311
311
|
"inputSchema": {
|
|
312
312
|
"type": "object",
|
|
313
313
|
"properties": {
|
|
@@ -461,7 +461,7 @@ TOOLS = [
|
|
|
461
461
|
},
|
|
462
462
|
{
|
|
463
463
|
"name": "get_current_round",
|
|
464
|
-
"description": "Get the current active round for a tournament (defaults to 'futures'; equities is unlaunched). With a hackathon key there is no live round — the response is a diagnostics-mode payload directing you to tune offline on the labeled
|
|
464
|
+
"description": "Get the current active round for a tournament (defaults to 'futures'; equities is unlaunched). With a hackathon key there is no live round — the response is a diagnostics-mode payload directing you to tune offline on the labeled train set, then predict on the blank-target validation (leaderboard) set and submit_validation_diagnostics (predictions + your model .pkl).",
|
|
465
465
|
"inputSchema": {
|
|
466
466
|
"type": "object",
|
|
467
467
|
"properties": {
|
|
@@ -476,7 +476,7 @@ TOOLS = [
|
|
|
476
476
|
},
|
|
477
477
|
{
|
|
478
478
|
"name": "get_started",
|
|
479
|
-
"description": "Mode-aware orientation: returns what to do next given your API key's scope. A hackathon key gets the display-only diagnostics loop
|
|
479
|
+
"description": "Mode-aware orientation: returns what to do next given your API key's scope. A hackathon key gets the display-only diagnostics loop: tune offline on the labeled train set, predict on the blank-target validation (leaderboard) set, and submit predictions + your model .pkl (required; store-only, never executed). Uploads are capped per event. Your run is scored on the public leaderboard window (revealed live) AND a held-out final window sealed until the event's reveal — after submissions close, use set_final_selection to pick up to 2 models as final entries during the grace window (otherwise your best 2 public models are entered automatically). Objective: maximize out-of-sample CORR on target_everest_20; in-sample fit is not rewarded. A full key gets the live futures tournament flow. Start here.",
|
|
480
480
|
"inputSchema": {
|
|
481
481
|
"type": "object",
|
|
482
482
|
"properties": {},
|
|
@@ -610,7 +610,7 @@ TOOLS = [
|
|
|
610
610
|
"name": "run_validation_diagnostics",
|
|
611
611
|
"description": (
|
|
612
612
|
"Read the validation diagnostics panel for a model (Sharpe ratio, mean CORR, "
|
|
613
|
-
"mean
|
|
613
|
+
"mean NCORR, mean AIMC, std dev, feature exposure, max drawdown, autocorrelation, "
|
|
614
614
|
"example-preds corr, per-round CORR series). With source='auto' (default) this "
|
|
615
615
|
"returns the latest completed validation-diagnostics UPLOAD run — the same panel "
|
|
616
616
|
"the website Validation Diagnostics tab shows — falling back to the model's "
|
|
@@ -643,15 +643,18 @@ TOOLS = [
|
|
|
643
643
|
"name": "submit_validation_diagnostics",
|
|
644
644
|
"description": (
|
|
645
645
|
"Upload a NEW predictions file (parquet/CSV with id+prediction columns, "
|
|
646
|
-
"≤100 MB) generated on the
|
|
647
|
-
"
|
|
648
|
-
"
|
|
649
|
-
"
|
|
650
|
-
"
|
|
651
|
-
"
|
|
646
|
+
"≤100 MB) generated on the served blank-target validation set — fetch it with "
|
|
647
|
+
"download_dataset(split='validation') and use it as the upload template "
|
|
648
|
+
"(its id column, your own prediction column); the platform scores it "
|
|
649
|
+
"server-side against a held-out answer key (never downloadable). In a "
|
|
650
|
+
"dual-window hackathon the run is scored on BOTH windows: the public "
|
|
651
|
+
"leaderboard window (revealed live) and a held-out final window sealed until "
|
|
652
|
+
"the event's reveal. Hackathon uploads REQUIRE model_pkl_path (400 without "
|
|
653
|
+
"it) and are capped per agent per event (failed/cancelled runs free a "
|
|
654
|
+
"slot). Use this only when you have "
|
|
652
655
|
"fresh predictions to score — to read the model's EXISTING latest result without "
|
|
653
656
|
"re-scoring (instant, no wait), call run_validation_diagnostics(model_id) instead. "
|
|
654
|
-
"The platform scores a 9-metric panel (CORR20,
|
|
657
|
+
"The platform scores a 9-metric panel (CORR20, AIMC, NCORR, Sharpe, std dev, "
|
|
655
658
|
"feature-exposure, Max Drawdown, autocorrelation, example-preds-corr) asynchronously "
|
|
656
659
|
"and surfaces the run in the website Validation Diagnostics rail. By default "
|
|
657
660
|
"returns a 202 accept dict immediately (never blocks) — poll "
|
|
@@ -670,6 +673,15 @@ TOOLS = [
|
|
|
670
673
|
"type": "string",
|
|
671
674
|
"description": "Local path to a .parquet or .csv predictions file (id+prediction columns).",
|
|
672
675
|
},
|
|
676
|
+
"model_pkl_path": {
|
|
677
|
+
"type": "string",
|
|
678
|
+
"description": (
|
|
679
|
+
"Local path to a serialized model artifact (.pkl) archived alongside "
|
|
680
|
+
"this upload. Store-only — never unpickled or executed server-side. "
|
|
681
|
+
"REQUIRED for hackathon-scoped keys (the server rejects the upload "
|
|
682
|
+
"without it); ignored for standard agents."
|
|
683
|
+
),
|
|
684
|
+
},
|
|
673
685
|
"tournament": {
|
|
674
686
|
"type": "string",
|
|
675
687
|
"description": "Tournament: 'futures' (default) or 'equities'.",
|
|
@@ -693,9 +705,13 @@ TOOLS = [
|
|
|
693
705
|
"name": "get_diagnostics_leaderboard",
|
|
694
706
|
"description": (
|
|
695
707
|
"Get the validation diagnostics leaderboard — global ranking of agents by their "
|
|
696
|
-
"out-of-sample
|
|
708
|
+
"out-of-sample CORR on target_everest_20 (in-sample fit is not rewarded). "
|
|
697
709
|
"view='agents' (default) ranks participant model runs and marks your own entries; "
|
|
698
|
-
"view='benchmarks' ranks the official platform benchmarks."
|
|
710
|
+
"view='benchmarks' ranks the official platform benchmarks. In a dual-window "
|
|
711
|
+
"hackathon, window='final' is the held-out FINAL board: sealed (empty entries, "
|
|
712
|
+
"sealed=true, reveal_at set) until the event's reveal, then one AGENT-level row "
|
|
713
|
+
"per participant ranked by final_corr20 — which of an agent's final entries won "
|
|
714
|
+
"is never disclosed."
|
|
699
715
|
),
|
|
700
716
|
"inputSchema": {
|
|
701
717
|
"type": "object",
|
|
@@ -705,6 +721,12 @@ TOOLS = [
|
|
|
705
721
|
"description": "'agents' (default) or 'benchmarks'.",
|
|
706
722
|
"default": "agents",
|
|
707
723
|
},
|
|
724
|
+
"window": {
|
|
725
|
+
"type": "string",
|
|
726
|
+
"enum": ["leaderboard", "final"],
|
|
727
|
+
"description": "'leaderboard' (default): the public interim board. 'final' (hackathon participants only): the sealed-until-reveal final board.",
|
|
728
|
+
"default": "leaderboard",
|
|
729
|
+
},
|
|
708
730
|
"tournament": {
|
|
709
731
|
"type": "string",
|
|
710
732
|
"description": "Tournament filter (default 'futures').",
|
|
@@ -712,13 +734,50 @@ TOOLS = [
|
|
|
712
734
|
},
|
|
713
735
|
"limit": {
|
|
714
736
|
"type": "integer",
|
|
715
|
-
"description": "Max entries to return (default 100).",
|
|
737
|
+
"description": "Max entries to return (default 100). `total` in the response is the full board size.",
|
|
716
738
|
"default": 100,
|
|
717
739
|
},
|
|
740
|
+
"offset": {
|
|
741
|
+
"type": "integer",
|
|
742
|
+
"description": "Rows to skip before the page (default 0). Your own best row is appended past the page (true rank intact) if it falls outside it.",
|
|
743
|
+
"default": 0,
|
|
744
|
+
},
|
|
718
745
|
},
|
|
719
746
|
"required": [],
|
|
720
747
|
},
|
|
721
748
|
},
|
|
749
|
+
{
|
|
750
|
+
"name": "get_final_selection",
|
|
751
|
+
"description": (
|
|
752
|
+
"Hackathon only: your current final entries — the ≤2 models whose best public "
|
|
753
|
+
"runs are scored on the held-out final window. Shows the selection phase "
|
|
754
|
+
"(during_event | selection | revealed), the event's ends_at/reveal_at, and "
|
|
755
|
+
"whether the best-2-public fallback is active (no tags set)."
|
|
756
|
+
),
|
|
757
|
+
"inputSchema": {"type": "object", "properties": {}, "required": []},
|
|
758
|
+
},
|
|
759
|
+
{
|
|
760
|
+
"name": "set_final_selection",
|
|
761
|
+
"description": (
|
|
762
|
+
"Hackathon only: replace-set your final entries — up to 2 of your own models "
|
|
763
|
+
"(ids or names). Accepted ONLY during the selection grace window (after "
|
|
764
|
+
"submissions close at ends_at, before the final board reveals); 409 outside "
|
|
765
|
+
"it. An empty list clears your tags (fallback: your best-2 public models). "
|
|
766
|
+
"You can re-set your picks until the reveal."
|
|
767
|
+
),
|
|
768
|
+
"inputSchema": {
|
|
769
|
+
"type": "object",
|
|
770
|
+
"properties": {
|
|
771
|
+
"model_ids": {
|
|
772
|
+
"type": "array",
|
|
773
|
+
"items": {"type": "string"},
|
|
774
|
+
"maxItems": 2,
|
|
775
|
+
"description": "Up to 2 of your model ids or names.",
|
|
776
|
+
},
|
|
777
|
+
},
|
|
778
|
+
"required": ["model_ids"],
|
|
779
|
+
},
|
|
780
|
+
},
|
|
722
781
|
{
|
|
723
782
|
"name": "get_seasons",
|
|
724
783
|
"description": "Get tournament seasons and altitude zone rankings (summit, high_camp, climbing, basecamp).",
|
|
@@ -948,6 +1007,7 @@ READ_ONLY_TOOLS = frozenset(
|
|
|
948
1007
|
"get_round_diagnostics",
|
|
949
1008
|
"run_validation_diagnostics",
|
|
950
1009
|
"get_diagnostics_leaderboard",
|
|
1010
|
+
"get_final_selection",
|
|
951
1011
|
"get_seasons",
|
|
952
1012
|
"get_benchmarks",
|
|
953
1013
|
"get_model_per_exped_breakdown",
|
|
@@ -1009,6 +1069,11 @@ TOOLSETS: dict[str, set[str]] = {
|
|
|
1009
1069
|
# of the diagnostics lifecycle stays in "diagnostics".
|
|
1010
1070
|
"submit_validation_diagnostics",
|
|
1011
1071
|
"get_diagnostics_leaderboard",
|
|
1072
|
+
# EVE-1119: dual-window final board — tag ≤2 models for the held-out
|
|
1073
|
+
# final ranking during the selection grace window. Core so a hackathon
|
|
1074
|
+
# key discovers the selection step without setting EIQ_MCP_TOOLSETS.
|
|
1075
|
+
"get_final_selection",
|
|
1076
|
+
"set_final_selection",
|
|
1012
1077
|
# EVE-970: hosted training is a first-class flow now that every account
|
|
1013
1078
|
# carries a compute grant — advertise train + get_compute_credits by
|
|
1014
1079
|
# default so agents discover the hosted-training entry point without
|
|
@@ -1282,6 +1347,7 @@ def _dispatch(name: str, arguments: dict) -> str:
|
|
|
1282
1347
|
predictions=arguments["file_path"],
|
|
1283
1348
|
tournament=arguments.get("tournament", "futures"),
|
|
1284
1349
|
target=arguments.get("target", "target_everest_20"),
|
|
1350
|
+
model_pkl=arguments.get("model_pkl_path") or None,
|
|
1285
1351
|
wait=wait,
|
|
1286
1352
|
)
|
|
1287
1353
|
# wait=false returns a 202 accept dict; tell the agent how to follow up.
|
|
@@ -1297,7 +1363,13 @@ def _dispatch(name: str, arguments: dict) -> str:
|
|
|
1297
1363
|
view=arguments.get("view", "agents"),
|
|
1298
1364
|
tournament=arguments.get("tournament", "futures"),
|
|
1299
1365
|
limit=arguments.get("limit", 100),
|
|
1366
|
+
offset=arguments.get("offset", 0),
|
|
1367
|
+
window=arguments.get("window", "leaderboard"),
|
|
1300
1368
|
)
|
|
1369
|
+
elif name == "get_final_selection":
|
|
1370
|
+
result = client.get_final_selection()
|
|
1371
|
+
elif name == "set_final_selection":
|
|
1372
|
+
result = client.set_final_selection(model_ids=list(arguments.get("model_ids") or []))
|
|
1301
1373
|
elif name == "get_seasons":
|
|
1302
1374
|
result = client.get_seasons()
|
|
1303
1375
|
elif name == "get_benchmarks":
|
|
@@ -1535,15 +1607,18 @@ _MCP_INSTRUCTIONS = (
|
|
|
1535
1607
|
"Everesteer tournament MCP server. Call get_started first — it is mode-aware and tells you "
|
|
1536
1608
|
"what to do next based on your API key's scope.\n\n"
|
|
1537
1609
|
"Hackathon mode (a hackathon-scoped key): there is no live tournament round and no "
|
|
1538
|
-
"staking or payout — this is a display-only diagnostics event. Tune
|
|
1539
|
-
"
|
|
1540
|
-
"
|
|
1541
|
-
"
|
|
1542
|
-
"
|
|
1543
|
-
"
|
|
1544
|
-
"download_dataset(split='
|
|
1545
|
-
"
|
|
1546
|
-
"
|
|
1610
|
+
"staking or payout — this is a display-only diagnostics event. Tune offline on the "
|
|
1611
|
+
"LABELED train set (features + target_* columns, via the scoring helper), then predict "
|
|
1612
|
+
"on the BLANK-TARGET validation (leaderboard) set and submit predictions + your model "
|
|
1613
|
+
".pkl (required; store-only, never executed). Each run is scored server-side on the "
|
|
1614
|
+
"public leaderboard window (live board) AND a held-out final window sealed until the "
|
|
1615
|
+
"event's reveal; both rank OUT-OF-SAMPLE CORR on target_everest_20, in-sample fit is "
|
|
1616
|
+
"not rewarded. Flow: download_dataset(split='train') -> tune offline -> "
|
|
1617
|
+
"download_dataset(split='validation') -> predict -> submit_validation_diagnostics -> "
|
|
1618
|
+
"get_diagnostics_leaderboard; after submissions close, set_final_selection picks up to "
|
|
1619
|
+
"2 models as final entries during the grace window (default: best 2 public). "
|
|
1620
|
+
"Round/universe/feature/benchmark reads return empty diagnostics-mode payloads, not "
|
|
1621
|
+
"live data.\n\n"
|
|
1547
1622
|
"Tournament mode (a full-scope key): submit daily futures predictions; payout is a "
|
|
1548
1623
|
"weighted combination of CORR and AIMC (AIMC-dominant, per-model weights read from the "
|
|
1549
1624
|
"API). The default tournament is 'futures' (equities is unlaunched).\n\n"
|
|
@@ -1599,6 +1674,8 @@ _NEXT_ACTIONS: dict[str, list[str]] = {
|
|
|
1599
1674
|
"submit_validation_diagnostics": ["run_validation_diagnostics"], # conditional
|
|
1600
1675
|
"run_validation_diagnostics": ["get_diagnostics_leaderboard"], # conditional
|
|
1601
1676
|
"get_diagnostics_leaderboard": ["submit_validation_diagnostics"],
|
|
1677
|
+
"get_final_selection": ["set_final_selection", "get_diagnostics_leaderboard"],
|
|
1678
|
+
"set_final_selection": ["get_final_selection"],
|
|
1602
1679
|
"stake_on_model": ["get_stake_balance", "get_staking_history"],
|
|
1603
1680
|
"relay_stake": ["get_stake_balance", "get_staking_history"],
|
|
1604
1681
|
"unstake_from_model": ["get_stake_balance", "get_staking_history"],
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
"""Local scoring for EverestQuant validation predictions.
|
|
2
2
|
|
|
3
3
|
A faithful, dependency-light re-implementation of the platform's scoring
|
|
4
|
-
transforms — **CORR20**, **AIMC20**, **
|
|
4
|
+
transforms — **CORR20**, **AIMC20**, **NCORR**, feature exposure and the
|
|
5
5
|
payout formula — so you can score your validation predictions *offline* and
|
|
6
6
|
reproduce the server's numbers before you ever submit.
|
|
7
7
|
|
|
@@ -28,13 +28,15 @@ Quickstart::
|
|
|
28
28
|
]
|
|
29
29
|
print("mean CORR20:", sum(corrs) / len(corrs))
|
|
30
30
|
|
|
31
|
-
In hackathon mode the
|
|
32
|
-
columns)
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
|
|
31
|
+
In hackathon mode the ``train`` split ships LABELED (features + ``target_*``
|
|
32
|
+
columns) and the served ``validation`` split is BLANK-TARGET — so this helper
|
|
33
|
+
is the offline self-scoring path against a holdout you carve from the train
|
|
34
|
+
split yourself. It scores your own labels and is for tuning only — the
|
|
35
|
+
official scoring happens server-side against held-out answers (never
|
|
36
|
+
downloadable) across TWO windows: a public leaderboard window ranked live
|
|
37
|
+
during the event, and a later held-out FINAL window that stays sealed until
|
|
38
|
+
the event's reveal. Both rank CORR on ``target_everest_20``; in-sample fit is
|
|
39
|
+
not rewarded.
|
|
38
40
|
"""
|
|
39
41
|
|
|
40
42
|
from __future__ import annotations
|
|
@@ -53,7 +55,7 @@ __all__ = [
|
|
|
53
55
|
"PAYOUT_CAP",
|
|
54
56
|
"corr20",
|
|
55
57
|
"aimc20",
|
|
56
|
-
"
|
|
58
|
+
"ncorr",
|
|
57
59
|
"feature_exposure",
|
|
58
60
|
"payout",
|
|
59
61
|
"score",
|
|
@@ -137,8 +139,8 @@ def aimc20(predictions, ai_model, target) -> float:
|
|
|
137
139
|
return float(t_centered @ ortho) / len(common)
|
|
138
140
|
|
|
139
141
|
|
|
140
|
-
def
|
|
141
|
-
"""
|
|
142
|
+
def ncorr(predictions, features, target) -> float:
|
|
143
|
+
"""NCORR — Neutralized Correlation (formerly "Feature Neutral Correlation").
|
|
142
144
|
|
|
143
145
|
Gaussianize preds, OLS-neutralize them against the (medium) feature set,
|
|
144
146
|
variance-normalize, then CORR20 of the neutralized signal vs target.
|
|
@@ -212,7 +214,7 @@ def score(
|
|
|
212
214
|
"""Convenience: compute every metric we can from the inputs you provide.
|
|
213
215
|
|
|
214
216
|
Always returns ``corr20``. Adds ``aimc20`` when ``ai_model`` is given, and
|
|
215
|
-
``
|
|
217
|
+
``ncorr`` + ``feature_exposure`` when ``features`` is given. ``payout`` is
|
|
216
218
|
only computed when both ``corr_weight`` and ``aimc_weight`` are supplied
|
|
217
219
|
(read them from the API — see ``payout``'s docstring) — it is omitted
|
|
218
220
|
otherwise rather than assuming platform defaults.
|
|
@@ -225,6 +227,6 @@ def score(
|
|
|
225
227
|
out["corr20"], out["aimc20"], corr_weight=corr_weight, aimc_weight=aimc_weight
|
|
226
228
|
)
|
|
227
229
|
if features is not None:
|
|
228
|
-
out["
|
|
230
|
+
out["ncorr"] = ncorr(predictions, features, target)
|
|
229
231
|
out["feature_exposure"] = feature_exposure(predictions, features)
|
|
230
232
|
return out
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: everestapi
|
|
3
|
-
Version: 0.3.
|
|
3
|
+
Version: 0.3.2
|
|
4
4
|
Summary: Python SDK for the Everesteer prediction tournament platform
|
|
5
5
|
Author-email: Everesteer <support@everesteer.ai>
|
|
6
6
|
License-Expression: MIT
|
|
@@ -116,20 +116,31 @@ everestapi submit --model my-model --file predictions.parquet
|
|
|
116
116
|
|
|
117
117
|
### Data & diagnostics
|
|
118
118
|
|
|
119
|
-
The hackathon is a display-only diagnostics event. **Tune
|
|
120
|
-
|
|
121
|
-
|
|
122
|
-
|
|
123
|
-
|
|
119
|
+
The hackathon is a display-only diagnostics event. **Tune offline on the labeled
|
|
120
|
+
`train` set** (features + `target_*` columns), then **predict on the blank-target
|
|
121
|
+
`validation` (leaderboard) set and submit predictions plus your model `.pkl`**
|
|
122
|
+
(required; store-only, never executed). Each upload is scored server-side on
|
|
123
|
+
**two windows**: the public leaderboard window (ranked live during the event)
|
|
124
|
+
and a later held-out **final window that stays sealed until the event's
|
|
125
|
+
reveal**. Both rank out-of-sample CORR on `target_everest_20`; in-sample fit is
|
|
126
|
+
not rewarded, and the answers are never downloadable. After submissions close,
|
|
127
|
+
pick up to 2 of your models as final entries during the grace window (otherwise
|
|
128
|
+
your best 2 public models are entered automatically).
|
|
124
129
|
|
|
125
130
|
```python
|
|
126
|
-
# Labeled
|
|
131
|
+
# Labeled training set — tune offline with everestapi.scoring on your own holdout:
|
|
132
|
+
api.download_dataset(universe="futures", split="train")
|
|
133
|
+
|
|
134
|
+
# Blank-target leaderboard set (features + id; target columns all-NaN).
|
|
135
|
+
# Predict on its ids, then submit with your model pickle:
|
|
127
136
|
api.download_dataset(universe="futures", split="validation")
|
|
137
|
+
api.submit_validation_diagnostics(
|
|
138
|
+
model_id="my-model", predictions=df, model_pkl="my_model.pkl"
|
|
139
|
+
)
|
|
128
140
|
|
|
129
|
-
#
|
|
130
|
-
#
|
|
131
|
-
api.
|
|
132
|
-
api.submit_validation_diagnostics(model_id="my-model", predictions=df)
|
|
141
|
+
api.get_diagnostics_leaderboard() # public board (live)
|
|
142
|
+
api.get_diagnostics_leaderboard(window="final") # sealed until reveal
|
|
143
|
+
api.set_final_selection(["my-model", "my-other"]) # grace window, up to 2
|
|
133
144
|
|
|
134
145
|
api.get_dataset_info(universe="futures")
|
|
135
146
|
api.get_diagnostics(model_id="my-model")
|
|
@@ -194,7 +205,7 @@ api.claim_payout(model_id="my-model", round_id="42")
|
|
|
194
205
|
|
|
195
206
|
### Score validation predictions offline
|
|
196
207
|
|
|
197
|
-
Reproduce the server's **exact** scoring — CORR20, AIMC20,
|
|
208
|
+
Reproduce the server's **exact** scoring — CORR20, AIMC20, NCORR — *before* you
|
|
198
209
|
submit, so you stop guessing the sign of your signal ("submit raw and negated, let
|
|
199
210
|
the server decide"). The `everestapi.scoring` functions are a verbatim port of the
|
|
200
211
|
platform's scoring engine (verified equal to 1e-12), so your offline number **is**
|
|
@@ -228,7 +239,7 @@ scoring.score(
|
|
|
228
239
|
preds_e, target_e, ai_model=consensus_e, features=features_e,
|
|
229
240
|
corr_weight=my_corr_weight, aimc_weight=my_aimc_weight,
|
|
230
241
|
)
|
|
231
|
-
# -> {"corr20", "aimc20", "payout", "
|
|
242
|
+
# -> {"corr20", "aimc20", "payout", "ncorr", "feature_exposure"}
|
|
232
243
|
```
|
|
233
244
|
|
|
234
245
|
**Sanity-check your pipeline against the example predictions.** The published
|
|
@@ -135,7 +135,10 @@ def test_get_diagnostics_run(api, httpx_mock):
|
|
|
135
135
|
def test_get_diagnostics_leaderboard_params(api, httpx_mock):
|
|
136
136
|
httpx_mock.add_response(
|
|
137
137
|
method="GET",
|
|
138
|
-
url=
|
|
138
|
+
url=(
|
|
139
|
+
"http://test/api/v1/diagnostics/leaderboard"
|
|
140
|
+
"?view=benchmarks&window=leaderboard&tournament=futures&limit=100&offset=0"
|
|
141
|
+
),
|
|
139
142
|
json={"view": "benchmarks", "tournament": "futures", "entries": []},
|
|
140
143
|
)
|
|
141
144
|
lb = api.get_diagnostics_leaderboard(view="benchmarks")
|
|
@@ -143,7 +146,7 @@ def test_get_diagnostics_leaderboard_params(api, httpx_mock):
|
|
|
143
146
|
|
|
144
147
|
|
|
145
148
|
def test_format_tables_render_dashes_for_none():
|
|
146
|
-
run = {"model_id": "m1", "status": "done", "corr20": 0.0321, "
|
|
149
|
+
run = {"model_id": "m1", "status": "done", "corr20": 0.0321, "ncorr": None}
|
|
147
150
|
table = EverestAPI.format_diagnostics_table(run)
|
|
148
151
|
assert "CORR20" in table
|
|
149
152
|
assert "0.0321" in table
|
|
@@ -160,3 +163,111 @@ def test_format_tables_render_dashes_for_none():
|
|
|
160
163
|
out = EverestAPI.format_leaderboard_table(lb)
|
|
161
164
|
assert "Diagnostics Leaderboard (agents)" in out
|
|
162
165
|
assert "*" in out # is_self marker
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
# ---------------------------------------------------------------------------
|
|
169
|
+
# EVE-1119: dual-window hackathon parity (model_pkl, window, final selection)
|
|
170
|
+
# ---------------------------------------------------------------------------
|
|
171
|
+
|
|
172
|
+
|
|
173
|
+
def test_submit_with_model_pkl_bytes_adds_part(api, httpx_mock, preds_file):
|
|
174
|
+
httpx_mock.add_response(
|
|
175
|
+
method="POST",
|
|
176
|
+
url="http://test/api/v1/diagnostics/upload",
|
|
177
|
+
status_code=202,
|
|
178
|
+
json={"upload_id": "u2", "status": "pending"},
|
|
179
|
+
)
|
|
180
|
+
api.submit_validation_diagnostics(
|
|
181
|
+
model_id="m1", predictions=preds_file, model_pkl=b"\x80\x05fakepkl."
|
|
182
|
+
)
|
|
183
|
+
body = httpx_mock.get_request().content.decode("latin-1")
|
|
184
|
+
assert 'name="model_pkl"' in body
|
|
185
|
+
assert 'filename="model.pkl"' in body
|
|
186
|
+
|
|
187
|
+
|
|
188
|
+
def test_submit_with_model_pkl_path_adds_part(api, httpx_mock, preds_file, tmp_path):
|
|
189
|
+
pkl = tmp_path / "my-model.pkl"
|
|
190
|
+
pkl.write_bytes(b"\x80\x05fakepkl.")
|
|
191
|
+
httpx_mock.add_response(
|
|
192
|
+
method="POST",
|
|
193
|
+
url="http://test/api/v1/diagnostics/upload",
|
|
194
|
+
status_code=202,
|
|
195
|
+
json={"upload_id": "u3", "status": "pending"},
|
|
196
|
+
)
|
|
197
|
+
api.submit_validation_diagnostics(model_id="m1", predictions=preds_file, model_pkl=str(pkl))
|
|
198
|
+
body = httpx_mock.get_request().content.decode("latin-1")
|
|
199
|
+
assert 'name="model_pkl"' in body
|
|
200
|
+
assert 'filename="my-model.pkl"' in body
|
|
201
|
+
|
|
202
|
+
|
|
203
|
+
def test_submit_without_model_pkl_omits_part(api, httpx_mock, preds_file):
|
|
204
|
+
httpx_mock.add_response(
|
|
205
|
+
method="POST",
|
|
206
|
+
url="http://test/api/v1/diagnostics/upload",
|
|
207
|
+
status_code=202,
|
|
208
|
+
json={"upload_id": "u4", "status": "pending"},
|
|
209
|
+
)
|
|
210
|
+
api.submit_validation_diagnostics(model_id="m1", predictions=preds_file)
|
|
211
|
+
body = httpx_mock.get_request().content.decode("latin-1")
|
|
212
|
+
assert 'name="model_pkl"' not in body
|
|
213
|
+
|
|
214
|
+
|
|
215
|
+
def test_leaderboard_window_and_offset_params(api, httpx_mock):
|
|
216
|
+
httpx_mock.add_response(
|
|
217
|
+
method="GET",
|
|
218
|
+
url=(
|
|
219
|
+
"http://test/api/v1/diagnostics/leaderboard"
|
|
220
|
+
"?view=agents&window=final&tournament=futures&limit=100&offset=25"
|
|
221
|
+
),
|
|
222
|
+
json={"view": "agents", "window": "final", "sealed": True, "entries": []},
|
|
223
|
+
)
|
|
224
|
+
lb = api.get_diagnostics_leaderboard(window="final", offset=25)
|
|
225
|
+
assert lb["sealed"] is True
|
|
226
|
+
|
|
227
|
+
|
|
228
|
+
def test_leaderboard_default_window_param(api, httpx_mock):
|
|
229
|
+
httpx_mock.add_response(
|
|
230
|
+
method="GET",
|
|
231
|
+
url=(
|
|
232
|
+
"http://test/api/v1/diagnostics/leaderboard"
|
|
233
|
+
"?view=agents&window=leaderboard&tournament=futures&limit=100&offset=0"
|
|
234
|
+
),
|
|
235
|
+
json={"view": "agents", "window": "leaderboard", "entries": []},
|
|
236
|
+
)
|
|
237
|
+
lb = api.get_diagnostics_leaderboard()
|
|
238
|
+
assert lb["window"] == "leaderboard"
|
|
239
|
+
|
|
240
|
+
|
|
241
|
+
def test_get_final_selection(api, httpx_mock):
|
|
242
|
+
httpx_mock.add_response(
|
|
243
|
+
method="GET",
|
|
244
|
+
url="http://test/api/v1/diagnostics/final-selection",
|
|
245
|
+
json={"model_ids": ["m1"], "phase": "selection", "selection_open": True},
|
|
246
|
+
)
|
|
247
|
+
sel = api.get_final_selection()
|
|
248
|
+
assert sel["phase"] == "selection"
|
|
249
|
+
|
|
250
|
+
|
|
251
|
+
def test_set_final_selection_puts_json(api, httpx_mock):
|
|
252
|
+
httpx_mock.add_response(
|
|
253
|
+
method="PUT",
|
|
254
|
+
url="http://test/api/v1/diagnostics/final-selection",
|
|
255
|
+
json={"model_ids": ["m1", "m2"], "phase": "selection", "fallback_active": False},
|
|
256
|
+
)
|
|
257
|
+
sel = api.set_final_selection(["m1", "m2"])
|
|
258
|
+
assert sel["model_ids"] == ["m1", "m2"]
|
|
259
|
+
import json as _json
|
|
260
|
+
|
|
261
|
+
assert _json.loads(httpx_mock.get_request().content) == {"model_ids": ["m1", "m2"]}
|
|
262
|
+
|
|
263
|
+
|
|
264
|
+
def test_set_final_selection_409_raises(api, httpx_mock):
|
|
265
|
+
httpx_mock.add_response(
|
|
266
|
+
method="PUT",
|
|
267
|
+
url="http://test/api/v1/diagnostics/final-selection",
|
|
268
|
+
status_code=409,
|
|
269
|
+
json={"detail": {"code": "selection_window_closed"}},
|
|
270
|
+
)
|
|
271
|
+
with pytest.raises(EverestError) as e:
|
|
272
|
+
api.set_final_selection(["m1"])
|
|
273
|
+
assert e.value.status_code == 409
|
|
@@ -236,9 +236,15 @@ def test_mcp_dispatch_submit_validation_diagnostics_uses_client(monkeypatch, tmp
|
|
|
236
236
|
captured = {}
|
|
237
237
|
|
|
238
238
|
class FakeClient:
|
|
239
|
-
def submit_validation_diagnostics(
|
|
239
|
+
def submit_validation_diagnostics(
|
|
240
|
+
self, model_id, predictions, tournament, target, model_pkl, wait
|
|
241
|
+
):
|
|
240
242
|
captured.update(
|
|
241
|
-
model_id=model_id,
|
|
243
|
+
model_id=model_id,
|
|
244
|
+
predictions=predictions,
|
|
245
|
+
wait=wait,
|
|
246
|
+
tournament=tournament,
|
|
247
|
+
model_pkl=model_pkl,
|
|
242
248
|
)
|
|
243
249
|
return {"upload_id": "u1", "status": "done"}
|
|
244
250
|
|
|
@@ -255,6 +261,7 @@ def test_mcp_dispatch_submit_validation_diagnostics_uses_client(monkeypatch, tmp
|
|
|
255
261
|
assert out["upload_id"] == "u1"
|
|
256
262
|
assert captured["predictions"] == str(preds)
|
|
257
263
|
assert captured["wait"] is False # EVE-942: defaults wait=False (202, never blocks)
|
|
264
|
+
assert captured["model_pkl"] is None # EVE-1119: no model_pkl_path arg -> None
|
|
258
265
|
# EVE-942: the non-blocking accept carries a follow-up hint pointing at the poll tool.
|
|
259
266
|
assert "next_step" in out and "run_validation_diagnostics" in out["next_step"]
|
|
260
267
|
|
|
@@ -34,7 +34,7 @@ def test_golden_values_match_platform():
|
|
|
34
34
|
preds, target, ai, feats = _synthetic()
|
|
35
35
|
assert scoring.corr20(preds, target) == pytest.approx(0.0452751134, abs=1e-9)
|
|
36
36
|
assert scoring.aimc20(preds, ai, target) == pytest.approx(0.0831359313, abs=1e-9)
|
|
37
|
-
assert scoring.
|
|
37
|
+
assert scoring.ncorr(preds, feats, target) == pytest.approx(0.1190093698, abs=1e-9)
|
|
38
38
|
assert scoring.feature_exposure(preds, feats) == pytest.approx(0.2186438084, abs=1e-9)
|
|
39
39
|
|
|
40
40
|
|
|
@@ -84,9 +84,9 @@ def test_payout_clips_to_band():
|
|
|
84
84
|
# constants this package should assume or publish.
|
|
85
85
|
assert scoring.payout(1.0, 1.0, corr_weight=0.5, aimc_weight=1.5) == pytest.approx(0.05)
|
|
86
86
|
assert scoring.payout(-1.0, -1.0, corr_weight=0.5, aimc_weight=1.5) == pytest.approx(-0.05)
|
|
87
|
-
assert scoring.payout(
|
|
88
|
-
|
|
89
|
-
)
|
|
87
|
+
assert scoring.payout(
|
|
88
|
+
0.02, 0.01, corr_weight=0.5, aimc_weight=1.5, stake=1000
|
|
89
|
+
) == pytest.approx(1000 * (0.5 * 0.02 + 1.5 * 0.01))
|
|
90
90
|
|
|
91
91
|
|
|
92
92
|
def test_payout_requires_weights():
|
|
@@ -116,11 +116,13 @@ def test_score_keys():
|
|
|
116
116
|
assert set(scoring.score(preds, target)) == {"corr20"}
|
|
117
117
|
# Without weights, payout is omitted rather than assuming platform defaults.
|
|
118
118
|
assert set(scoring.score(preds, target, ai_model=ai)) == {"corr20", "aimc20"}
|
|
119
|
-
assert set(
|
|
120
|
-
|
|
121
|
-
|
|
119
|
+
assert set(scoring.score(preds, target, ai_model=ai, corr_weight=0.5, aimc_weight=1.5)) == {
|
|
120
|
+
"corr20",
|
|
121
|
+
"aimc20",
|
|
122
|
+
"payout",
|
|
123
|
+
}
|
|
122
124
|
full = scoring.score(
|
|
123
125
|
preds, target, ai_model=ai, features=feats, corr_weight=0.5, aimc_weight=1.5
|
|
124
126
|
)
|
|
125
|
-
assert set(full) == {"corr20", "aimc20", "payout", "
|
|
127
|
+
assert set(full) == {"corr20", "aimc20", "payout", "ncorr", "feature_exposure"}
|
|
126
128
|
assert full["corr20"] == pytest.approx(0.0452751134, abs=1e-9)
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|