everestapi 0.3.1__tar.gz → 0.3.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (33) hide show
  1. {everestapi-0.3.1/src/everestapi.egg-info → everestapi-0.3.2}/PKG-INFO +22 -11
  2. {everestapi-0.3.1 → everestapi-0.3.2}/README.md +21 -10
  3. {everestapi-0.3.1 → everestapi-0.3.2}/src/everestapi/__init__.py +1 -1
  4. {everestapi-0.3.1 → everestapi-0.3.2}/src/everestapi/client.py +98 -27
  5. {everestapi-0.3.1 → everestapi-0.3.2}/src/everestapi/mcp/server.py +98 -21
  6. {everestapi-0.3.1 → everestapi-0.3.2}/src/everestapi/scoring.py +9 -7
  7. {everestapi-0.3.1 → everestapi-0.3.2/src/everestapi.egg-info}/PKG-INFO +22 -11
  8. {everestapi-0.3.1 → everestapi-0.3.2}/tests/test_diagnostics.py +112 -1
  9. {everestapi-0.3.1 → everestapi-0.3.2}/tests/test_eve1087_cpu_tier.py +1 -1
  10. {everestapi-0.3.1 → everestapi-0.3.2}/tests/test_mcp_and_models.py +9 -2
  11. {everestapi-0.3.1 → everestapi-0.3.2}/tests/test_scoring.py +8 -6
  12. {everestapi-0.3.1 → everestapi-0.3.2}/LICENSE +0 -0
  13. {everestapi-0.3.1 → everestapi-0.3.2}/pyproject.toml +0 -0
  14. {everestapi-0.3.1 → everestapi-0.3.2}/setup.cfg +0 -0
  15. {everestapi-0.3.1 → everestapi-0.3.2}/src/everestapi/__main__.py +0 -0
  16. {everestapi-0.3.1 → everestapi-0.3.2}/src/everestapi/cli.py +0 -0
  17. {everestapi-0.3.1 → everestapi-0.3.2}/src/everestapi/mcp/__init__.py +0 -0
  18. {everestapi-0.3.1 → everestapi-0.3.2}/src/everestapi/mcp/__main__.py +0 -0
  19. {everestapi-0.3.1 → everestapi-0.3.2}/src/everestapi/plots.py +0 -0
  20. {everestapi-0.3.1 → everestapi-0.3.2}/src/everestapi/types.py +0 -0
  21. {everestapi-0.3.1 → everestapi-0.3.2}/src/everestapi.egg-info/SOURCES.txt +0 -0
  22. {everestapi-0.3.1 → everestapi-0.3.2}/src/everestapi.egg-info/dependency_links.txt +0 -0
  23. {everestapi-0.3.1 → everestapi-0.3.2}/src/everestapi.egg-info/entry_points.txt +0 -0
  24. {everestapi-0.3.1 → everestapi-0.3.2}/src/everestapi.egg-info/requires.txt +0 -0
  25. {everestapi-0.3.1 → everestapi-0.3.2}/src/everestapi.egg-info/top_level.txt +0 -0
  26. {everestapi-0.3.1 → everestapi-0.3.2}/tests/test_cli.py +0 -0
  27. {everestapi-0.3.1 → everestapi-0.3.2}/tests/test_client.py +0 -0
  28. {everestapi-0.3.1 → everestapi-0.3.2}/tests/test_eve953_mcp_progress.py +0 -0
  29. {everestapi-0.3.1 → everestapi-0.3.2}/tests/test_eve957_mcp_annotations_resources.py +0 -0
  30. {everestapi-0.3.1 → everestapi-0.3.2}/tests/test_eve959_toolsets.py +0 -0
  31. {everestapi-0.3.1 → everestapi-0.3.2}/tests/test_eve967_mcp_discoverability.py +0 -0
  32. {everestapi-0.3.1 → everestapi-0.3.2}/tests/test_json_or_raise.py +0 -0
  33. {everestapi-0.3.1 → everestapi-0.3.2}/tests/test_prediction_range.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: everestapi
3
- Version: 0.3.1
3
+ Version: 0.3.2
4
4
  Summary: Python SDK for the Everesteer prediction tournament platform
5
5
  Author-email: Everesteer <support@everesteer.ai>
6
6
  License-Expression: MIT
@@ -116,20 +116,31 @@ everestapi submit --model my-model --file predictions.parquet
116
116
 
117
117
  ### Data & diagnostics
118
118
 
119
- The hackathon is a display-only diagnostics event. **Tune and self-score offline
120
- on the labeled validation set** (features + `target_*` columns), then **predict on
121
- the blind `eiq_live_2026` set and submit** — the leaderboard ranks your
122
- out-of-sample 2026 CORR on `target_everest_20`. In-sample fit is not rewarded. The
123
- labeled `eiq_live_2026` answers are held out server-side and are never downloadable.
119
+ The hackathon is a display-only diagnostics event. **Tune offline on the labeled
120
+ `train` set** (features + `target_*` columns), then **predict on the blank-target
121
+ `validation` (leaderboard) set and submit predictions plus your model `.pkl`**
122
+ (required; store-only, never executed). Each upload is scored server-side on
123
+ **two windows**: the public leaderboard window (ranked live during the event)
124
+ and a later held-out **final window that stays sealed until the event's
125
+ reveal**. Both rank out-of-sample CORR on `target_everest_20`; in-sample fit is
126
+ not rewarded, and the answers are never downloadable. After submissions close,
127
+ pick up to 2 of your models as final entries during the grace window (otherwise
128
+ your best 2 public models are entered automatically).
124
129
 
125
130
  ```python
126
- # Labeled practice set — tune + self-score offline with everestapi.scoring:
131
+ # Labeled training set — tune offline with everestapi.scoring on your own holdout:
132
+ api.download_dataset(universe="futures", split="train")
133
+
134
+ # Blank-target leaderboard set (features + id; target columns all-NaN).
135
+ # Predict on its ids, then submit with your model pickle:
127
136
  api.download_dataset(universe="futures", split="validation")
137
+ api.submit_validation_diagnostics(
138
+ model_id="my-model", predictions=df, model_pkl="my_model.pkl"
139
+ )
128
140
 
129
- # Blind scored set (columns: exped, exped_date, instrument, id — no targets).
130
- # Predict on it, then submit; it is also the upload id template.
131
- api.download_dataset(universe="futures", split="live")
132
- api.submit_validation_diagnostics(model_id="my-model", predictions=df)
141
+ api.get_diagnostics_leaderboard() # public board (live)
142
+ api.get_diagnostics_leaderboard(window="final") # sealed until reveal
143
+ api.set_final_selection(["my-model", "my-other"]) # grace window, up to 2
133
144
 
134
145
  api.get_dataset_info(universe="futures")
135
146
  api.get_diagnostics(model_id="my-model")
@@ -77,20 +77,31 @@ everestapi submit --model my-model --file predictions.parquet
77
77
 
78
78
  ### Data & diagnostics
79
79
 
80
- The hackathon is a display-only diagnostics event. **Tune and self-score offline
81
- on the labeled validation set** (features + `target_*` columns), then **predict on
82
- the blind `eiq_live_2026` set and submit** — the leaderboard ranks your
83
- out-of-sample 2026 CORR on `target_everest_20`. In-sample fit is not rewarded. The
84
- labeled `eiq_live_2026` answers are held out server-side and are never downloadable.
80
+ The hackathon is a display-only diagnostics event. **Tune offline on the labeled
81
+ `train` set** (features + `target_*` columns), then **predict on the blank-target
82
+ `validation` (leaderboard) set and submit predictions plus your model `.pkl`**
83
+ (required; store-only, never executed). Each upload is scored server-side on
84
+ **two windows**: the public leaderboard window (ranked live during the event)
85
+ and a later held-out **final window that stays sealed until the event's
86
+ reveal**. Both rank out-of-sample CORR on `target_everest_20`; in-sample fit is
87
+ not rewarded, and the answers are never downloadable. After submissions close,
88
+ pick up to 2 of your models as final entries during the grace window (otherwise
89
+ your best 2 public models are entered automatically).
85
90
 
86
91
  ```python
87
- # Labeled practice set — tune + self-score offline with everestapi.scoring:
92
+ # Labeled training set — tune offline with everestapi.scoring on your own holdout:
93
+ api.download_dataset(universe="futures", split="train")
94
+
95
+ # Blank-target leaderboard set (features + id; target columns all-NaN).
96
+ # Predict on its ids, then submit with your model pickle:
88
97
  api.download_dataset(universe="futures", split="validation")
98
+ api.submit_validation_diagnostics(
99
+ model_id="my-model", predictions=df, model_pkl="my_model.pkl"
100
+ )
89
101
 
90
- # Blind scored set (columns: exped, exped_date, instrument, id — no targets).
91
- # Predict on it, then submit; it is also the upload id template.
92
- api.download_dataset(universe="futures", split="live")
93
- api.submit_validation_diagnostics(model_id="my-model", predictions=df)
102
+ api.get_diagnostics_leaderboard() # public board (live)
103
+ api.get_diagnostics_leaderboard(window="final") # sealed until reveal
104
+ api.set_final_selection(["my-model", "my-other"]) # grace window, up to 2
94
105
 
95
106
  api.get_dataset_info(universe="futures")
96
107
  api.get_diagnostics(model_id="my-model")
@@ -1,6 +1,6 @@
1
1
  """EverestAPI — Python SDK for the Everesteer prediction tournament platform."""
2
2
 
3
- __version__ = "0.3.1"
3
+ __version__ = "0.3.2"
4
4
 
5
5
  from everestapi.client import EverestAPI, EverestError
6
6
  from everestapi.types import (
@@ -364,6 +364,7 @@ class EverestAPI:
364
364
  tournament: str = "futures",
365
365
  target: str = "target_everest_20",
366
366
  *,
367
+ model_pkl=None,
367
368
  wait: bool = False,
368
369
  poll_interval: float = 2.0,
369
370
  timeout: float = 600.0,
@@ -371,14 +372,25 @@ class EverestAPI:
371
372
  """POST /api/v1/diagnostics/upload (multipart) — score out-of-sample predictions.
372
373
 
373
374
  ``predictions`` is a pandas DataFrame (``id`` + ``prediction`` columns) or a
374
- path to a ``.parquet`` / ``.csv`` file, generated on the blind eiq_live_2026
375
- set (``download_dataset(split="live")``). The platform scores it server-side
376
- against the held-out labeled eiq_live_2026 answers — those answers are never
377
- downloadable. Tune and self-score offline on the labeled validation set first
378
- (see :mod:`everestapi.scoring`). Returns the 202 accept dict; with
379
- ``wait=True`` polls ``runs/{upload_id}`` until ``done`` (returns the run) and
380
- raises :class:`EverestError` on ``failed`` or timeout. Display-only; results
381
- also surface in the website Validation Diagnostics rail.
375
+ path to a ``.parquet`` / ``.csv`` file, covering the ids of the served
376
+ blank-target validation set (``download_dataset(split="validation")`` — the
377
+ leaderboard set; the retired blind ``live`` split now 404s for hackathon
378
+ keys). The platform scores it server-side against a held-out answer key
379
+ that is never downloadable. In a dual-window event, the run is scored on
380
+ BOTH windows: the public leaderboard window (revealed live) and a
381
+ held-out final window whose metrics stay sealed until the event's reveal.
382
+
383
+ ``model_pkl`` is a serialized model artifact (raw ``bytes`` or a path to a
384
+ ``.pkl`` file) archived alongside the upload — store-only for audit, never
385
+ unpickled or executed server-side. **Required for hackathon-scoped keys**
386
+ (the server rejects the upload with 400 without it); ignored for standard
387
+ agents. Hackathon uploads are also capped per agent per event
388
+ (failed/cancelled runs free a slot).
389
+
390
+ Returns the 202 accept dict; with ``wait=True`` polls ``runs/{upload_id}``
391
+ until ``done`` (returns the run) and raises :class:`EverestError` on
392
+ ``failed`` or timeout. Display-only; results also surface in the website
393
+ Validation Diagnostics rail.
382
394
  """
383
395
  import io
384
396
  import time
@@ -393,9 +405,20 @@ class EverestAPI:
393
405
  data = f.read()
394
406
  fname = str(predictions).rsplit("/", 1)[-1].rsplit("\\", 1)[-1]
395
407
 
408
+ files = {"file": (fname, data, "application/octet-stream")}
409
+ if model_pkl is not None:
410
+ if isinstance(model_pkl, bytes | bytearray):
411
+ pkl_data = bytes(model_pkl)
412
+ pkl_fname = "model.pkl"
413
+ else: # path
414
+ with open(model_pkl, "rb") as f:
415
+ pkl_data = f.read()
416
+ pkl_fname = str(model_pkl).rsplit("/", 1)[-1].rsplit("\\", 1)[-1]
417
+ files["model_pkl"] = (pkl_fname, pkl_data, "application/octet-stream")
418
+
396
419
  resp = self._client.post(
397
420
  "/api/v1/diagnostics/upload",
398
- files={"file": (fname, data, "application/octet-stream")},
421
+ files=files,
399
422
  data={"model_id": model_id, "tournament": tournament, "target": target},
400
423
  )
401
424
  if resp.status_code >= 400:
@@ -426,16 +449,60 @@ class EverestAPI:
426
449
  view: str = "agents",
427
450
  tournament: str = "futures",
428
451
  limit: int = 100,
452
+ offset: int = 0,
453
+ window: str = "leaderboard",
429
454
  ) -> dict:
430
455
  """GET /api/v1/diagnostics/leaderboard — global CORR20 ranking.
431
456
 
432
457
  ``view="agents"`` ranks participant model runs; ``view="benchmarks"`` ranks
433
- the official benchmark models.
458
+ the official benchmark models. ``total`` in the response is the full board
459
+ size — page with ``limit``/``offset``; your own best row is appended past
460
+ the page (true ``rank`` intact) when it falls outside the slice.
461
+
462
+ ``window`` (dual-window hackathon events): ``"leaderboard"`` (default) is
463
+ the public interim board. ``"final"`` (participants only) is the held-out
464
+ final board — ``sealed: true`` with empty entries until the event's
465
+ ``reveal_at``, then one AGENT-level row per participant ranked by
466
+ ``final_corr20``; which of an agent's final entries won is never
467
+ disclosed.
434
468
  """
435
469
  return self._request(
436
470
  "GET",
437
471
  "/api/v1/diagnostics/leaderboard",
438
- params={"view": view, "tournament": tournament, "limit": limit},
472
+ params={
473
+ "view": view,
474
+ "window": window,
475
+ "tournament": tournament,
476
+ "limit": limit,
477
+ "offset": offset,
478
+ },
479
+ )
480
+
481
+ def get_final_selection(self) -> dict:
482
+ """GET /api/v1/diagnostics/final-selection — your final entries (hackathon only).
483
+
484
+ Returns your currently tagged models (≤2), the selection ``phase``
485
+ (``during_event`` | ``selection`` | ``revealed``), ``selection_open``,
486
+ the event's ``ends_at``/``reveal_at``, and ``fallback_active`` (true when
487
+ no tags are set — the final board then uses your best-2 public models
488
+ automatically).
489
+ """
490
+ return self._request("GET", "/api/v1/diagnostics/final-selection")
491
+
492
+ def set_final_selection(self, model_ids: list) -> dict:
493
+ """PUT /api/v1/diagnostics/final-selection — pick your final entries.
494
+
495
+ Replace-set up to 2 of your own models (ids or names) whose best public
496
+ runs are scored on the held-out final window. Accepted ONLY during the
497
+ selection grace window (after the event's ``ends_at``, before the final
498
+ board reveals) — 409 outside it. An empty list clears your tags and
499
+ re-arms the best-2-public fallback; re-PUT to change picks until the
500
+ reveal.
501
+ """
502
+ return self._request(
503
+ "PUT",
504
+ "/api/v1/diagnostics/final-selection",
505
+ json={"model_ids": list(model_ids)},
439
506
  )
440
507
 
441
508
  @staticmethod
@@ -600,18 +667,19 @@ class EverestAPI:
600
667
  404 the current version is resolved and the download retried once.
601
668
 
602
669
  Splits: ``train`` / ``validation`` (features + targets), ``live``
603
- (features only, no targets). In hackathon mode, ``validation`` is the
604
- LABELED practice set — features + all ``target_*`` columns — that you tune
605
- and self-score on offline (see :mod:`everestapi.scoring`), and ``live`` is
606
- the BLIND eiq_live_2026 out-of-sample set (columns exactly ``exped``,
607
- ``exped_date``, ``instrument``, ``id`` — no targets) that you predict on and
608
- submit. The labeled eiq_live_2026 answers are held out server-side and never
609
- downloadable.
670
+ (features only, no targets). In hackathon mode, ``train`` is the LABELED
671
+ training set (features + all ``target_*`` columns) for offline tuning
672
+ (see :mod:`everestapi.scoring`), and ``validation`` is the BLANK-TARGET
673
+ leaderboard set (features + ``id``; target columns present but all-NaN)
674
+ that you predict on and submit via
675
+ :meth:`submit_validation_diagnostics`. The answers are held out
676
+ server-side and never downloadable; the retired blind ``live`` split
677
+ 404s for hackathon keys.
610
678
 
611
679
  Futures is served by the futures endpoint (``version`` is a no-op):
612
- it returns the real bregen tree to full-scope keys and the hackathon tree
613
- (labeled ``validation`` practice set + blind ``live`` eiq_live_2026 scored
614
- set) to hackathon-scoped keys.
680
+ it returns the real bregen tree to full-scope keys and the hackathon
681
+ tree (labeled ``train`` + blank-target ``validation``) to
682
+ hackathon-scoped keys.
615
683
 
616
684
  Value semantics: observed feature values are cross-sectional
617
685
  quintile bins ``0-4``; a feature value of ``-1.0`` means MISSING
@@ -1121,8 +1189,9 @@ class EverestAPI:
1121
1189
 
1122
1190
  With a hackathon-scoped key there is no live round — the response is a
1123
1191
  diagnostics-mode payload (mode='diagnostics_hackathon') directing you to
1124
- tune offline on the labeled validation set, then predict on the blind
1125
- eiq_live_2026 set and submit_validation_diagnostics.
1192
+ tune offline on the labeled train set, then predict on the blank-target
1193
+ validation (leaderboard) set and submit_validation_diagnostics
1194
+ (predictions + your model .pkl).
1126
1195
  """
1127
1196
  return self._request(
1128
1197
  "GET",
@@ -1140,10 +1209,12 @@ class EverestAPI:
1140
1209
  """GET /api/v1/get_started — mode-aware orientation.
1141
1210
 
1142
1211
  Returns what to do next given this key's scope: the display-only
1143
- diagnostics-hackathon loop (tune and self-score offline on the labeled
1144
- validation set, then predict on the blind eiq_live_2026 set and submit —
1145
- ranked on out-of-sample 2026 CORR on target_everest_20), or the live
1146
- futures tournament flow.
1212
+ diagnostics-hackathon loop (tune offline on the labeled train set,
1213
+ predict on the blank-target validation set, submit predictions + your
1214
+ model .pkl; scored on a public leaderboard window AND a held-out final
1215
+ window sealed until reveal — pick up to 2 final entries with
1216
+ :meth:`set_final_selection` during the post-close grace window), or the
1217
+ live futures tournament flow.
1147
1218
  """
1148
1219
  return self._request("GET", "/api/v1/get_started")
1149
1220
 
@@ -307,7 +307,7 @@ TOOLS = [
307
307
  },
308
308
  {
309
309
  "name": "download_dataset",
310
- "description": "Download a dataset split (train/validation/live) as a parquet file. Returns the local file path. In hackathon mode: 'validation' is the LABELED practice set (features + all target_* columns) for offline tuning and self-scoring; 'live' is the BLIND eiq_live_2026 out-of-sample set (columns exactly exped, exped_date, instrument, id — NO targets) that you predict on and submit. The labeled 2026 answers are held out server-side and never downloadable. VALUE SEMANTICS: observed feature values are cross-sectional quintile bins 0-4; feature value -1.0 means MISSING (source not yet onboarded for that instrument/date) — treat -1 as NaN or a distinct category, NEVER as an ordinal value below 0. Target NaN means uncomputable for that row (never imputed) — drop NaN-target rows when training on that target. Full notes: the dataset's eiq_metadata.json missing_value_note.",
310
+ "description": "Download a dataset split (train/validation) as a parquet file. Returns the local file path. In hackathon mode: 'train' is the LABELED training set (features + all target_* columns) for offline tuning; 'validation' is the BLANK-TARGET leaderboard set (features + target columns present but all-NaN) — you predict on its ids and submit via submit_validation_diagnostics. The answers are held out server-side and never downloadable; the retired blind 'live' split 404s for hackathon keys. VALUE SEMANTICS: observed feature values are cross-sectional quintile bins 0-4; feature value -1.0 means MISSING (source not yet onboarded for that instrument/date) — treat -1 as NaN or a distinct category, NEVER as an ordinal value below 0. Target NaN means uncomputable for that row (never imputed) — drop NaN-target rows when training on that target. Full notes: the dataset's eiq_metadata.json missing_value_note.",
311
311
  "inputSchema": {
312
312
  "type": "object",
313
313
  "properties": {
@@ -461,7 +461,7 @@ TOOLS = [
461
461
  },
462
462
  {
463
463
  "name": "get_current_round",
464
- "description": "Get the current active round for a tournament (defaults to 'futures'; equities is unlaunched). With a hackathon key there is no live round — the response is a diagnostics-mode payload directing you to tune offline on the labeled validation set, then download the blind eiq_live_2026 set, predict, and submit_validation_diagnostics.",
464
+ "description": "Get the current active round for a tournament (defaults to 'futures'; equities is unlaunched). With a hackathon key there is no live round — the response is a diagnostics-mode payload directing you to tune offline on the labeled train set, then predict on the blank-target validation (leaderboard) set and submit_validation_diagnostics (predictions + your model .pkl).",
465
465
  "inputSchema": {
466
466
  "type": "object",
467
467
  "properties": {
@@ -476,7 +476,7 @@ TOOLS = [
476
476
  },
477
477
  {
478
478
  "name": "get_started",
479
- "description": "Mode-aware orientation: returns what to do next given your API key's scope. A hackathon key gets the display-only diagnostics loop (tune and self-score offline on the labeled validation set, then predict on the blind eiq_live_2026 set and submit; objective: maximize out-of-sample 2026 CORR on target_everest_20, in-sample fit is not rewarded); a full key gets the live futures tournament flow. Start here.",
479
+ "description": "Mode-aware orientation: returns what to do next given your API key's scope. A hackathon key gets the display-only diagnostics loop: tune offline on the labeled train set, predict on the blank-target validation (leaderboard) set, and submit predictions + your model .pkl (required; store-only, never executed). Uploads are capped per event. Your run is scored on the public leaderboard window (revealed live) AND a held-out final window sealed until the event's reveal — after submissions close, use set_final_selection to pick up to 2 models as final entries during the grace window (otherwise your best 2 public models are entered automatically). Objective: maximize out-of-sample CORR on target_everest_20; in-sample fit is not rewarded. A full key gets the live futures tournament flow. Start here.",
480
480
  "inputSchema": {
481
481
  "type": "object",
482
482
  "properties": {},
@@ -643,12 +643,15 @@ TOOLS = [
643
643
  "name": "submit_validation_diagnostics",
644
644
  "description": (
645
645
  "Upload a NEW predictions file (parquet/CSV with id+prediction columns, "
646
- "≤100 MB) generated on the blind eiq_live_2026 out-of-sample set; the platform "
647
- "scores it server-side against the held-out labeled eiq_live_2026 answers (never "
648
- "downloadable). The required id set is the blind eiq_live_2026 set — fetch it with "
649
- "download_dataset(split='live') and use it as the upload template "
650
- "(its id column, your own prediction column). Tune and self-score offline on the "
651
- "labeled validation set first (see the scoring helper). Use this only when you have "
646
+ "≤100 MB) generated on the served blank-target validation set — fetch it with "
647
+ "download_dataset(split='validation') and use it as the upload template "
648
+ "(its id column, your own prediction column); the platform scores it "
649
+ "server-side against a held-out answer key (never downloadable). In a "
650
+ "dual-window hackathon the run is scored on BOTH windows: the public "
651
+ "leaderboard window (revealed live) and a held-out final window sealed until "
652
+ "the event's reveal. Hackathon uploads REQUIRE model_pkl_path (400 without "
653
+ "it) and are capped per agent per event (failed/cancelled runs free a "
654
+ "slot). Use this only when you have "
652
655
  "fresh predictions to score — to read the model's EXISTING latest result without "
653
656
  "re-scoring (instant, no wait), call run_validation_diagnostics(model_id) instead. "
654
657
  "The platform scores a 9-metric panel (CORR20, AIMC, NCORR, Sharpe, std dev, "
@@ -670,6 +673,15 @@ TOOLS = [
670
673
  "type": "string",
671
674
  "description": "Local path to a .parquet or .csv predictions file (id+prediction columns).",
672
675
  },
676
+ "model_pkl_path": {
677
+ "type": "string",
678
+ "description": (
679
+ "Local path to a serialized model artifact (.pkl) archived alongside "
680
+ "this upload. Store-only — never unpickled or executed server-side. "
681
+ "REQUIRED for hackathon-scoped keys (the server rejects the upload "
682
+ "without it); ignored for standard agents."
683
+ ),
684
+ },
673
685
  "tournament": {
674
686
  "type": "string",
675
687
  "description": "Tournament: 'futures' (default) or 'equities'.",
@@ -693,9 +705,13 @@ TOOLS = [
693
705
  "name": "get_diagnostics_leaderboard",
694
706
  "description": (
695
707
  "Get the validation diagnostics leaderboard — global ranking of agents by their "
696
- "out-of-sample eiq_live_2026 CORR on target_everest_20 (in-sample fit is not rewarded). "
708
+ "out-of-sample CORR on target_everest_20 (in-sample fit is not rewarded). "
697
709
  "view='agents' (default) ranks participant model runs and marks your own entries; "
698
- "view='benchmarks' ranks the official platform benchmarks."
710
+ "view='benchmarks' ranks the official platform benchmarks. In a dual-window "
711
+ "hackathon, window='final' is the held-out FINAL board: sealed (empty entries, "
712
+ "sealed=true, reveal_at set) until the event's reveal, then one AGENT-level row "
713
+ "per participant ranked by final_corr20 — which of an agent's final entries won "
714
+ "is never disclosed."
699
715
  ),
700
716
  "inputSchema": {
701
717
  "type": "object",
@@ -705,6 +721,12 @@ TOOLS = [
705
721
  "description": "'agents' (default) or 'benchmarks'.",
706
722
  "default": "agents",
707
723
  },
724
+ "window": {
725
+ "type": "string",
726
+ "enum": ["leaderboard", "final"],
727
+ "description": "'leaderboard' (default): the public interim board. 'final' (hackathon participants only): the sealed-until-reveal final board.",
728
+ "default": "leaderboard",
729
+ },
708
730
  "tournament": {
709
731
  "type": "string",
710
732
  "description": "Tournament filter (default 'futures').",
@@ -712,13 +734,50 @@ TOOLS = [
712
734
  },
713
735
  "limit": {
714
736
  "type": "integer",
715
- "description": "Max entries to return (default 100).",
737
+ "description": "Max entries to return (default 100). `total` in the response is the full board size.",
716
738
  "default": 100,
717
739
  },
740
+ "offset": {
741
+ "type": "integer",
742
+ "description": "Rows to skip before the page (default 0). Your own best row is appended past the page (true rank intact) if it falls outside it.",
743
+ "default": 0,
744
+ },
718
745
  },
719
746
  "required": [],
720
747
  },
721
748
  },
749
+ {
750
+ "name": "get_final_selection",
751
+ "description": (
752
+ "Hackathon only: your current final entries — the ≤2 models whose best public "
753
+ "runs are scored on the held-out final window. Shows the selection phase "
754
+ "(during_event | selection | revealed), the event's ends_at/reveal_at, and "
755
+ "whether the best-2-public fallback is active (no tags set)."
756
+ ),
757
+ "inputSchema": {"type": "object", "properties": {}, "required": []},
758
+ },
759
+ {
760
+ "name": "set_final_selection",
761
+ "description": (
762
+ "Hackathon only: replace-set your final entries — up to 2 of your own models "
763
+ "(ids or names). Accepted ONLY during the selection grace window (after "
764
+ "submissions close at ends_at, before the final board reveals); 409 outside "
765
+ "it. An empty list clears your tags (fallback: your best-2 public models). "
766
+ "You can re-set your picks until the reveal."
767
+ ),
768
+ "inputSchema": {
769
+ "type": "object",
770
+ "properties": {
771
+ "model_ids": {
772
+ "type": "array",
773
+ "items": {"type": "string"},
774
+ "maxItems": 2,
775
+ "description": "Up to 2 of your model ids or names.",
776
+ },
777
+ },
778
+ "required": ["model_ids"],
779
+ },
780
+ },
722
781
  {
723
782
  "name": "get_seasons",
724
783
  "description": "Get tournament seasons and altitude zone rankings (summit, high_camp, climbing, basecamp).",
@@ -948,6 +1007,7 @@ READ_ONLY_TOOLS = frozenset(
948
1007
  "get_round_diagnostics",
949
1008
  "run_validation_diagnostics",
950
1009
  "get_diagnostics_leaderboard",
1010
+ "get_final_selection",
951
1011
  "get_seasons",
952
1012
  "get_benchmarks",
953
1013
  "get_model_per_exped_breakdown",
@@ -1009,6 +1069,11 @@ TOOLSETS: dict[str, set[str]] = {
1009
1069
  # of the diagnostics lifecycle stays in "diagnostics".
1010
1070
  "submit_validation_diagnostics",
1011
1071
  "get_diagnostics_leaderboard",
1072
+ # EVE-1119: dual-window final board — tag ≤2 models for the held-out
1073
+ # final ranking during the selection grace window. Core so a hackathon
1074
+ # key discovers the selection step without setting EIQ_MCP_TOOLSETS.
1075
+ "get_final_selection",
1076
+ "set_final_selection",
1012
1077
  # EVE-970: hosted training is a first-class flow now that every account
1013
1078
  # carries a compute grant — advertise train + get_compute_credits by
1014
1079
  # default so agents discover the hosted-training entry point without
@@ -1282,6 +1347,7 @@ def _dispatch(name: str, arguments: dict) -> str:
1282
1347
  predictions=arguments["file_path"],
1283
1348
  tournament=arguments.get("tournament", "futures"),
1284
1349
  target=arguments.get("target", "target_everest_20"),
1350
+ model_pkl=arguments.get("model_pkl_path") or None,
1285
1351
  wait=wait,
1286
1352
  )
1287
1353
  # wait=false returns a 202 accept dict; tell the agent how to follow up.
@@ -1297,7 +1363,13 @@ def _dispatch(name: str, arguments: dict) -> str:
1297
1363
  view=arguments.get("view", "agents"),
1298
1364
  tournament=arguments.get("tournament", "futures"),
1299
1365
  limit=arguments.get("limit", 100),
1366
+ offset=arguments.get("offset", 0),
1367
+ window=arguments.get("window", "leaderboard"),
1300
1368
  )
1369
+ elif name == "get_final_selection":
1370
+ result = client.get_final_selection()
1371
+ elif name == "set_final_selection":
1372
+ result = client.set_final_selection(model_ids=list(arguments.get("model_ids") or []))
1301
1373
  elif name == "get_seasons":
1302
1374
  result = client.get_seasons()
1303
1375
  elif name == "get_benchmarks":
@@ -1535,15 +1607,18 @@ _MCP_INSTRUCTIONS = (
1535
1607
  "Everesteer tournament MCP server. Call get_started first — it is mode-aware and tells you "
1536
1608
  "what to do next based on your API key's scope.\n\n"
1537
1609
  "Hackathon mode (a hackathon-scoped key): there is no live tournament round and no "
1538
- "staking or payout — this is a display-only diagnostics event. Tune and self-score "
1539
- "offline on the LABELED validation set (features + target_* columns, via the scoring "
1540
- "helper), then predict on the BLIND eiq_live_2026 set (columns exped, exped_date, "
1541
- "instrument, id — no targets) and submit; the leaderboard ranks your OUT-OF-SAMPLE 2026 "
1542
- "CORR on target_everest_20. In-sample fit is not rewarded. Flow: "
1543
- "download_dataset(split='validation') -> tune + self-score offline -> "
1544
- "download_dataset(split='live') -> predict -> submit_validation_diagnostics -> "
1545
- "get_diagnostics_leaderboard. Round/universe/feature/benchmark reads return empty "
1546
- "diagnostics-mode payloads, not live data.\n\n"
1610
+ "staking or payout — this is a display-only diagnostics event. Tune offline on the "
1611
+ "LABELED train set (features + target_* columns, via the scoring helper), then predict "
1612
+ "on the BLANK-TARGET validation (leaderboard) set and submit predictions + your model "
1613
+ ".pkl (required; store-only, never executed). Each run is scored server-side on the "
1614
+ "public leaderboard window (live board) AND a held-out final window sealed until the "
1615
+ "event's reveal; both rank OUT-OF-SAMPLE CORR on target_everest_20, in-sample fit is "
1616
+ "not rewarded. Flow: download_dataset(split='train') -> tune offline -> "
1617
+ "download_dataset(split='validation') -> predict -> submit_validation_diagnostics -> "
1618
+ "get_diagnostics_leaderboard; after submissions close, set_final_selection picks up to "
1619
+ "2 models as final entries during the grace window (default: best 2 public). "
1620
+ "Round/universe/feature/benchmark reads return empty diagnostics-mode payloads, not "
1621
+ "live data.\n\n"
1547
1622
  "Tournament mode (a full-scope key): submit daily futures predictions; payout is a "
1548
1623
  "weighted combination of CORR and AIMC (AIMC-dominant, per-model weights read from the "
1549
1624
  "API). The default tournament is 'futures' (equities is unlaunched).\n\n"
@@ -1599,6 +1674,8 @@ _NEXT_ACTIONS: dict[str, list[str]] = {
1599
1674
  "submit_validation_diagnostics": ["run_validation_diagnostics"], # conditional
1600
1675
  "run_validation_diagnostics": ["get_diagnostics_leaderboard"], # conditional
1601
1676
  "get_diagnostics_leaderboard": ["submit_validation_diagnostics"],
1677
+ "get_final_selection": ["set_final_selection", "get_diagnostics_leaderboard"],
1678
+ "set_final_selection": ["get_final_selection"],
1602
1679
  "stake_on_model": ["get_stake_balance", "get_staking_history"],
1603
1680
  "relay_stake": ["get_stake_balance", "get_staking_history"],
1604
1681
  "unstake_from_model": ["get_stake_balance", "get_staking_history"],
@@ -28,13 +28,15 @@ Quickstart::
28
28
  ]
29
29
  print("mean CORR20:", sum(corrs) / len(corrs))
30
30
 
31
- In hackathon mode the validation set ships LABELED (features + ``target_*``
32
- columns), so this helper is the offline self-scoring path: use it to tune your
33
- model before you submit. It scores in-sample and is for tuning only — the
34
- official hackathon leaderboard scores your submitted predictions on the BLIND
35
- ``eiq_live_2026`` out-of-sample set server-side (its labeled answers are held
36
- out and never downloadable), ranking your 2026 CORR on ``target_everest_20``.
37
- In-sample fit is not rewarded.
31
+ In hackathon mode the ``train`` split ships LABELED (features + ``target_*``
32
+ columns) and the served ``validation`` split is BLANK-TARGET — so this helper
33
+ is the offline self-scoring path against a holdout you carve from the train
34
+ split yourself. It scores your own labels and is for tuning only — the
35
+ official scoring happens server-side against held-out answers (never
36
+ downloadable) across TWO windows: a public leaderboard window ranked live
37
+ during the event, and a later held-out FINAL window that stays sealed until
38
+ the event's reveal. Both rank CORR on ``target_everest_20``; in-sample fit is
39
+ not rewarded.
38
40
  """
39
41
 
40
42
  from __future__ import annotations
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: everestapi
3
- Version: 0.3.1
3
+ Version: 0.3.2
4
4
  Summary: Python SDK for the Everesteer prediction tournament platform
5
5
  Author-email: Everesteer <support@everesteer.ai>
6
6
  License-Expression: MIT
@@ -116,20 +116,31 @@ everestapi submit --model my-model --file predictions.parquet
116
116
 
117
117
  ### Data & diagnostics
118
118
 
119
- The hackathon is a display-only diagnostics event. **Tune and self-score offline
120
- on the labeled validation set** (features + `target_*` columns), then **predict on
121
- the blind `eiq_live_2026` set and submit** — the leaderboard ranks your
122
- out-of-sample 2026 CORR on `target_everest_20`. In-sample fit is not rewarded. The
123
- labeled `eiq_live_2026` answers are held out server-side and are never downloadable.
119
+ The hackathon is a display-only diagnostics event. **Tune offline on the labeled
120
+ `train` set** (features + `target_*` columns), then **predict on the blank-target
121
+ `validation` (leaderboard) set and submit predictions plus your model `.pkl`**
122
+ (required; store-only, never executed). Each upload is scored server-side on
123
+ **two windows**: the public leaderboard window (ranked live during the event)
124
+ and a later held-out **final window that stays sealed until the event's
125
+ reveal**. Both rank out-of-sample CORR on `target_everest_20`; in-sample fit is
126
+ not rewarded, and the answers are never downloadable. After submissions close,
127
+ pick up to 2 of your models as final entries during the grace window (otherwise
128
+ your best 2 public models are entered automatically).
124
129
 
125
130
  ```python
126
- # Labeled practice set — tune + self-score offline with everestapi.scoring:
131
+ # Labeled training set — tune offline with everestapi.scoring on your own holdout:
132
+ api.download_dataset(universe="futures", split="train")
133
+
134
+ # Blank-target leaderboard set (features + id; target columns all-NaN).
135
+ # Predict on its ids, then submit with your model pickle:
127
136
  api.download_dataset(universe="futures", split="validation")
137
+ api.submit_validation_diagnostics(
138
+ model_id="my-model", predictions=df, model_pkl="my_model.pkl"
139
+ )
128
140
 
129
- # Blind scored set (columns: exped, exped_date, instrument, id — no targets).
130
- # Predict on it, then submit; it is also the upload id template.
131
- api.download_dataset(universe="futures", split="live")
132
- api.submit_validation_diagnostics(model_id="my-model", predictions=df)
141
+ api.get_diagnostics_leaderboard() # public board (live)
142
+ api.get_diagnostics_leaderboard(window="final") # sealed until reveal
143
+ api.set_final_selection(["my-model", "my-other"]) # grace window, up to 2
133
144
 
134
145
  api.get_dataset_info(universe="futures")
135
146
  api.get_diagnostics(model_id="my-model")
@@ -135,7 +135,10 @@ def test_get_diagnostics_run(api, httpx_mock):
135
135
  def test_get_diagnostics_leaderboard_params(api, httpx_mock):
136
136
  httpx_mock.add_response(
137
137
  method="GET",
138
- url="http://test/api/v1/diagnostics/leaderboard?view=benchmarks&tournament=futures&limit=100",
138
+ url=(
139
+ "http://test/api/v1/diagnostics/leaderboard"
140
+ "?view=benchmarks&window=leaderboard&tournament=futures&limit=100&offset=0"
141
+ ),
139
142
  json={"view": "benchmarks", "tournament": "futures", "entries": []},
140
143
  )
141
144
  lb = api.get_diagnostics_leaderboard(view="benchmarks")
@@ -160,3 +163,111 @@ def test_format_tables_render_dashes_for_none():
160
163
  out = EverestAPI.format_leaderboard_table(lb)
161
164
  assert "Diagnostics Leaderboard (agents)" in out
162
165
  assert "*" in out # is_self marker
166
+
167
+
168
+ # ---------------------------------------------------------------------------
169
+ # EVE-1119: dual-window hackathon parity (model_pkl, window, final selection)
170
+ # ---------------------------------------------------------------------------
171
+
172
+
173
+ def test_submit_with_model_pkl_bytes_adds_part(api, httpx_mock, preds_file):
174
+ httpx_mock.add_response(
175
+ method="POST",
176
+ url="http://test/api/v1/diagnostics/upload",
177
+ status_code=202,
178
+ json={"upload_id": "u2", "status": "pending"},
179
+ )
180
+ api.submit_validation_diagnostics(
181
+ model_id="m1", predictions=preds_file, model_pkl=b"\x80\x05fakepkl."
182
+ )
183
+ body = httpx_mock.get_request().content.decode("latin-1")
184
+ assert 'name="model_pkl"' in body
185
+ assert 'filename="model.pkl"' in body
186
+
187
+
188
+ def test_submit_with_model_pkl_path_adds_part(api, httpx_mock, preds_file, tmp_path):
189
+ pkl = tmp_path / "my-model.pkl"
190
+ pkl.write_bytes(b"\x80\x05fakepkl.")
191
+ httpx_mock.add_response(
192
+ method="POST",
193
+ url="http://test/api/v1/diagnostics/upload",
194
+ status_code=202,
195
+ json={"upload_id": "u3", "status": "pending"},
196
+ )
197
+ api.submit_validation_diagnostics(model_id="m1", predictions=preds_file, model_pkl=str(pkl))
198
+ body = httpx_mock.get_request().content.decode("latin-1")
199
+ assert 'name="model_pkl"' in body
200
+ assert 'filename="my-model.pkl"' in body
201
+
202
+
203
+ def test_submit_without_model_pkl_omits_part(api, httpx_mock, preds_file):
204
+ httpx_mock.add_response(
205
+ method="POST",
206
+ url="http://test/api/v1/diagnostics/upload",
207
+ status_code=202,
208
+ json={"upload_id": "u4", "status": "pending"},
209
+ )
210
+ api.submit_validation_diagnostics(model_id="m1", predictions=preds_file)
211
+ body = httpx_mock.get_request().content.decode("latin-1")
212
+ assert 'name="model_pkl"' not in body
213
+
214
+
215
+ def test_leaderboard_window_and_offset_params(api, httpx_mock):
216
+ httpx_mock.add_response(
217
+ method="GET",
218
+ url=(
219
+ "http://test/api/v1/diagnostics/leaderboard"
220
+ "?view=agents&window=final&tournament=futures&limit=100&offset=25"
221
+ ),
222
+ json={"view": "agents", "window": "final", "sealed": True, "entries": []},
223
+ )
224
+ lb = api.get_diagnostics_leaderboard(window="final", offset=25)
225
+ assert lb["sealed"] is True
226
+
227
+
228
+ def test_leaderboard_default_window_param(api, httpx_mock):
229
+ httpx_mock.add_response(
230
+ method="GET",
231
+ url=(
232
+ "http://test/api/v1/diagnostics/leaderboard"
233
+ "?view=agents&window=leaderboard&tournament=futures&limit=100&offset=0"
234
+ ),
235
+ json={"view": "agents", "window": "leaderboard", "entries": []},
236
+ )
237
+ lb = api.get_diagnostics_leaderboard()
238
+ assert lb["window"] == "leaderboard"
239
+
240
+
241
+ def test_get_final_selection(api, httpx_mock):
242
+ httpx_mock.add_response(
243
+ method="GET",
244
+ url="http://test/api/v1/diagnostics/final-selection",
245
+ json={"model_ids": ["m1"], "phase": "selection", "selection_open": True},
246
+ )
247
+ sel = api.get_final_selection()
248
+ assert sel["phase"] == "selection"
249
+
250
+
251
+ def test_set_final_selection_puts_json(api, httpx_mock):
252
+ httpx_mock.add_response(
253
+ method="PUT",
254
+ url="http://test/api/v1/diagnostics/final-selection",
255
+ json={"model_ids": ["m1", "m2"], "phase": "selection", "fallback_active": False},
256
+ )
257
+ sel = api.set_final_selection(["m1", "m2"])
258
+ assert sel["model_ids"] == ["m1", "m2"]
259
+ import json as _json
260
+
261
+ assert _json.loads(httpx_mock.get_request().content) == {"model_ids": ["m1", "m2"]}
262
+
263
+
264
+ def test_set_final_selection_409_raises(api, httpx_mock):
265
+ httpx_mock.add_response(
266
+ method="PUT",
267
+ url="http://test/api/v1/diagnostics/final-selection",
268
+ status_code=409,
269
+ json={"detail": {"code": "selection_window_closed"}},
270
+ )
271
+ with pytest.raises(EverestError) as e:
272
+ api.set_final_selection(["m1"])
273
+ assert e.value.status_code == 409
@@ -17,4 +17,4 @@ def test_train_gpu_enum_includes_cpu():
17
17
 
18
18
 
19
19
  def test_version_bumped():
20
- assert everestapi.__version__ == "0.3.1"
20
+ assert everestapi.__version__ == "0.3.2"
@@ -236,9 +236,15 @@ def test_mcp_dispatch_submit_validation_diagnostics_uses_client(monkeypatch, tmp
236
236
  captured = {}
237
237
 
238
238
  class FakeClient:
239
- def submit_validation_diagnostics(self, model_id, predictions, tournament, target, wait):
239
+ def submit_validation_diagnostics(
240
+ self, model_id, predictions, tournament, target, model_pkl, wait
241
+ ):
240
242
  captured.update(
241
- model_id=model_id, predictions=predictions, wait=wait, tournament=tournament
243
+ model_id=model_id,
244
+ predictions=predictions,
245
+ wait=wait,
246
+ tournament=tournament,
247
+ model_pkl=model_pkl,
242
248
  )
243
249
  return {"upload_id": "u1", "status": "done"}
244
250
 
@@ -255,6 +261,7 @@ def test_mcp_dispatch_submit_validation_diagnostics_uses_client(monkeypatch, tmp
255
261
  assert out["upload_id"] == "u1"
256
262
  assert captured["predictions"] == str(preds)
257
263
  assert captured["wait"] is False # EVE-942: defaults wait=False (202, never blocks)
264
+ assert captured["model_pkl"] is None # EVE-1119: no model_pkl_path arg -> None
258
265
  # EVE-942: the non-blocking accept carries a follow-up hint pointing at the poll tool.
259
266
  assert "next_step" in out and "run_validation_diagnostics" in out["next_step"]
260
267
 
@@ -84,9 +84,9 @@ def test_payout_clips_to_band():
84
84
  # constants this package should assume or publish.
85
85
  assert scoring.payout(1.0, 1.0, corr_weight=0.5, aimc_weight=1.5) == pytest.approx(0.05)
86
86
  assert scoring.payout(-1.0, -1.0, corr_weight=0.5, aimc_weight=1.5) == pytest.approx(-0.05)
87
- assert scoring.payout(0.02, 0.01, corr_weight=0.5, aimc_weight=1.5, stake=1000) == pytest.approx(
88
- 1000 * (0.5 * 0.02 + 1.5 * 0.01)
89
- )
87
+ assert scoring.payout(
88
+ 0.02, 0.01, corr_weight=0.5, aimc_weight=1.5, stake=1000
89
+ ) == pytest.approx(1000 * (0.5 * 0.02 + 1.5 * 0.01))
90
90
 
91
91
 
92
92
  def test_payout_requires_weights():
@@ -116,9 +116,11 @@ def test_score_keys():
116
116
  assert set(scoring.score(preds, target)) == {"corr20"}
117
117
  # Without weights, payout is omitted rather than assuming platform defaults.
118
118
  assert set(scoring.score(preds, target, ai_model=ai)) == {"corr20", "aimc20"}
119
- assert set(
120
- scoring.score(preds, target, ai_model=ai, corr_weight=0.5, aimc_weight=1.5)
121
- ) == {"corr20", "aimc20", "payout"}
119
+ assert set(scoring.score(preds, target, ai_model=ai, corr_weight=0.5, aimc_weight=1.5)) == {
120
+ "corr20",
121
+ "aimc20",
122
+ "payout",
123
+ }
122
124
  full = scoring.score(
123
125
  preds, target, ai_model=ai, features=feats, corr_weight=0.5, aimc_weight=1.5
124
126
  )
File without changes
File without changes
File without changes
File without changes