everestapi 0.2.7__tar.gz → 0.2.9__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {everestapi-0.2.7/src/everestapi.egg-info → everestapi-0.2.9}/PKG-INFO +27 -13
- {everestapi-0.2.7 → everestapi-0.2.9}/README.md +26 -12
- {everestapi-0.2.7 → everestapi-0.2.9}/src/everestapi/__init__.py +1 -1
- {everestapi-0.2.7 → everestapi-0.2.9}/src/everestapi/client.py +123 -32
- {everestapi-0.2.7 → everestapi-0.2.9}/src/everestapi/mcp/server.py +107 -69
- {everestapi-0.2.7 → everestapi-0.2.9}/src/everestapi/scoring.py +8 -0
- {everestapi-0.2.7 → everestapi-0.2.9/src/everestapi.egg-info}/PKG-INFO +27 -13
- {everestapi-0.2.7 → everestapi-0.2.9}/src/everestapi.egg-info/SOURCES.txt +1 -0
- everestapi-0.2.9/tests/test_eve967_mcp_discoverability.py +40 -0
- {everestapi-0.2.7 → everestapi-0.2.9}/tests/test_mcp_and_models.py +7 -0
- {everestapi-0.2.7 → everestapi-0.2.9}/LICENSE +0 -0
- {everestapi-0.2.7 → everestapi-0.2.9}/pyproject.toml +0 -0
- {everestapi-0.2.7 → everestapi-0.2.9}/setup.cfg +0 -0
- {everestapi-0.2.7 → everestapi-0.2.9}/src/everestapi/__main__.py +0 -0
- {everestapi-0.2.7 → everestapi-0.2.9}/src/everestapi/cli.py +0 -0
- {everestapi-0.2.7 → everestapi-0.2.9}/src/everestapi/mcp/__init__.py +0 -0
- {everestapi-0.2.7 → everestapi-0.2.9}/src/everestapi/mcp/__main__.py +0 -0
- {everestapi-0.2.7 → everestapi-0.2.9}/src/everestapi/plots.py +0 -0
- {everestapi-0.2.7 → everestapi-0.2.9}/src/everestapi/types.py +0 -0
- {everestapi-0.2.7 → everestapi-0.2.9}/src/everestapi.egg-info/dependency_links.txt +0 -0
- {everestapi-0.2.7 → everestapi-0.2.9}/src/everestapi.egg-info/entry_points.txt +0 -0
- {everestapi-0.2.7 → everestapi-0.2.9}/src/everestapi.egg-info/requires.txt +0 -0
- {everestapi-0.2.7 → everestapi-0.2.9}/src/everestapi.egg-info/top_level.txt +0 -0
- {everestapi-0.2.7 → everestapi-0.2.9}/tests/test_cli.py +0 -0
- {everestapi-0.2.7 → everestapi-0.2.9}/tests/test_client.py +0 -0
- {everestapi-0.2.7 → everestapi-0.2.9}/tests/test_diagnostics.py +0 -0
- {everestapi-0.2.7 → everestapi-0.2.9}/tests/test_eve953_mcp_progress.py +0 -0
- {everestapi-0.2.7 → everestapi-0.2.9}/tests/test_eve957_mcp_annotations_resources.py +0 -0
- {everestapi-0.2.7 → everestapi-0.2.9}/tests/test_eve959_toolsets.py +0 -0
- {everestapi-0.2.7 → everestapi-0.2.9}/tests/test_json_or_raise.py +0 -0
- {everestapi-0.2.7 → everestapi-0.2.9}/tests/test_prediction_range.py +0 -0
- {everestapi-0.2.7 → everestapi-0.2.9}/tests/test_scoring.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: everestapi
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.9
|
|
4
4
|
Summary: Python SDK for the Everesteer prediction tournament platform
|
|
5
5
|
Author-email: Everesteer <support@everesteer.ai>
|
|
6
6
|
License-Expression: MIT
|
|
@@ -116,16 +116,23 @@ everestapi submit --model my-model --file predictions.parquet
|
|
|
116
116
|
|
|
117
117
|
### Data & diagnostics
|
|
118
118
|
|
|
119
|
+
The hackathon is a display-only diagnostics event. **Tune and self-score offline
|
|
120
|
+
on the labeled validation set** (features + `target_*` columns), then **predict on
|
|
121
|
+
the blind `eiq_live_2026` set and submit** — the leaderboard ranks your
|
|
122
|
+
out-of-sample 2026 CORR on `target_everest_20`. In-sample fit is not rewarded. The
|
|
123
|
+
labeled `eiq_live_2026` answers are held out server-side and are never downloadable.
|
|
124
|
+
|
|
119
125
|
```python
|
|
120
|
-
|
|
121
|
-
api.
|
|
122
|
-
api.get_dataset_info(universe="futures")
|
|
123
|
-
api.get_diagnostics(model_id="my-model")
|
|
126
|
+
# Labeled practice set — tune + self-score offline with everestapi.scoring:
|
|
127
|
+
api.download_dataset(universe="futures", split="validation")
|
|
124
128
|
|
|
125
|
-
#
|
|
126
|
-
#
|
|
127
|
-
api.download_dataset(universe="futures", split="
|
|
129
|
+
# Blind scored set (columns: exped, exped_date, instrument, id — no targets).
|
|
130
|
+
# Predict on it, then submit; it is also the upload id template.
|
|
131
|
+
api.download_dataset(universe="futures", split="live")
|
|
128
132
|
api.submit_validation_diagnostics(model_id="my-model", predictions=df)
|
|
133
|
+
|
|
134
|
+
api.get_dataset_info(universe="futures")
|
|
135
|
+
api.get_diagnostics(model_id="my-model")
|
|
129
136
|
```
|
|
130
137
|
|
|
131
138
|
### Plotting (optional `viz` extra)
|
|
@@ -157,11 +164,18 @@ df = pd.DataFrame(lb["entries"])
|
|
|
157
164
|
### Serverless compute
|
|
158
165
|
|
|
159
166
|
```python
|
|
160
|
-
#
|
|
161
|
-
|
|
162
|
-
|
|
163
|
-
|
|
164
|
-
|
|
167
|
+
# Built-in preset (lightgbm/xgboost/ridge/mlp/random_forest) — no data upload,
|
|
168
|
+
# the platform trains against the same obfuscated dataset you download.
|
|
169
|
+
job = api.train(model="lightgbm", features="small", target="target_everest_20")
|
|
170
|
+
|
|
171
|
+
# model="custom" — your own model factory, run server-side in an isolated,
|
|
172
|
+
# network-denied sandbox (no filesystem access, never sees held-out targets)
|
|
173
|
+
job = api.train(
|
|
174
|
+
model="custom",
|
|
175
|
+
custom_model_fn="def build_model(params):\n from sklearn.linear_model import Ridge\n return Ridge(**params)",
|
|
176
|
+
gpu="A100",
|
|
177
|
+
max_hours=2.0,
|
|
178
|
+
)
|
|
165
179
|
|
|
166
180
|
# Wait and download
|
|
167
181
|
result = api.wait_for_job(job["job_id"])
|
|
@@ -77,16 +77,23 @@ everestapi submit --model my-model --file predictions.parquet
|
|
|
77
77
|
|
|
78
78
|
### Data & diagnostics
|
|
79
79
|
|
|
80
|
+
The hackathon is a display-only diagnostics event. **Tune and self-score offline
|
|
81
|
+
on the labeled validation set** (features + `target_*` columns), then **predict on
|
|
82
|
+
the blind `eiq_live_2026` set and submit** — the leaderboard ranks your
|
|
83
|
+
out-of-sample 2026 CORR on `target_everest_20`. In-sample fit is not rewarded. The
|
|
84
|
+
labeled `eiq_live_2026` answers are held out server-side and are never downloadable.
|
|
85
|
+
|
|
80
86
|
```python
|
|
81
|
-
|
|
82
|
-
api.
|
|
83
|
-
api.get_dataset_info(universe="futures")
|
|
84
|
-
api.get_diagnostics(model_id="my-model")
|
|
87
|
+
# Labeled practice set — tune + self-score offline with everestapi.scoring:
|
|
88
|
+
api.download_dataset(universe="futures", split="validation")
|
|
85
89
|
|
|
86
|
-
#
|
|
87
|
-
#
|
|
88
|
-
api.download_dataset(universe="futures", split="
|
|
90
|
+
# Blind scored set (columns: exped, exped_date, instrument, id — no targets).
|
|
91
|
+
# Predict on it, then submit; it is also the upload id template.
|
|
92
|
+
api.download_dataset(universe="futures", split="live")
|
|
89
93
|
api.submit_validation_diagnostics(model_id="my-model", predictions=df)
|
|
94
|
+
|
|
95
|
+
api.get_dataset_info(universe="futures")
|
|
96
|
+
api.get_diagnostics(model_id="my-model")
|
|
90
97
|
```
|
|
91
98
|
|
|
92
99
|
### Plotting (optional `viz` extra)
|
|
@@ -118,11 +125,18 @@ df = pd.DataFrame(lb["entries"])
|
|
|
118
125
|
### Serverless compute
|
|
119
126
|
|
|
120
127
|
```python
|
|
121
|
-
#
|
|
122
|
-
|
|
123
|
-
|
|
124
|
-
|
|
125
|
-
|
|
128
|
+
# Built-in preset (lightgbm/xgboost/ridge/mlp/random_forest) — no data upload,
|
|
129
|
+
# the platform trains against the same obfuscated dataset you download.
|
|
130
|
+
job = api.train(model="lightgbm", features="small", target="target_everest_20")
|
|
131
|
+
|
|
132
|
+
# model="custom" — your own model factory, run server-side in an isolated,
|
|
133
|
+
# network-denied sandbox (no filesystem access, never sees held-out targets)
|
|
134
|
+
job = api.train(
|
|
135
|
+
model="custom",
|
|
136
|
+
custom_model_fn="def build_model(params):\n from sklearn.linear_model import Ridge\n return Ridge(**params)",
|
|
137
|
+
gpu="A100",
|
|
138
|
+
max_hours=2.0,
|
|
139
|
+
)
|
|
126
140
|
|
|
127
141
|
# Wait and download
|
|
128
142
|
result = api.wait_for_job(job["job_id"])
|
|
@@ -26,6 +26,7 @@ Usage::
|
|
|
26
26
|
from __future__ import annotations
|
|
27
27
|
|
|
28
28
|
import os
|
|
29
|
+
import warnings
|
|
29
30
|
from pathlib import PurePath
|
|
30
31
|
from typing import Any
|
|
31
32
|
|
|
@@ -366,10 +367,14 @@ class EverestAPI:
|
|
|
366
367
|
poll_interval: float = 2.0,
|
|
367
368
|
timeout: float = 600.0,
|
|
368
369
|
) -> dict:
|
|
369
|
-
"""POST /api/v1/diagnostics/upload (multipart) — score
|
|
370
|
+
"""POST /api/v1/diagnostics/upload (multipart) — score out-of-sample predictions.
|
|
370
371
|
|
|
371
372
|
``predictions`` is a pandas DataFrame (``id`` + ``prediction`` columns) or a
|
|
372
|
-
path to a ``.parquet`` / ``.csv`` file
|
|
373
|
+
path to a ``.parquet`` / ``.csv`` file, generated on the blind eiq_live_2026
|
|
374
|
+
set (``download_dataset(split="live")``). The platform scores it server-side
|
|
375
|
+
against the held-out labeled eiq_live_2026 answers — those answers are never
|
|
376
|
+
downloadable. Tune and self-score offline on the labeled validation set first
|
|
377
|
+
(see :mod:`everestapi.scoring`). Returns the 202 accept dict; with
|
|
373
378
|
``wait=True`` polls ``runs/{upload_id}`` until ``done`` (returns the run) and
|
|
374
379
|
raises :class:`EverestError` on ``failed`` or timeout. Display-only; results
|
|
375
380
|
also surface in the website Validation Diagnostics rail.
|
|
@@ -590,11 +595,18 @@ class EverestAPI:
|
|
|
590
595
|
404 the current version is resolved and the download retried once.
|
|
591
596
|
|
|
592
597
|
Splits: ``train`` / ``validation`` (features + targets), ``live``
|
|
593
|
-
(
|
|
598
|
+
(features only, no targets). In hackathon mode, ``validation`` is the
|
|
599
|
+
LABELED practice set — features + all ``target_*`` columns — that you tune
|
|
600
|
+
and self-score on offline (see :mod:`everestapi.scoring`), and ``live`` is
|
|
601
|
+
the BLIND eiq_live_2026 out-of-sample set (columns exactly ``exped``,
|
|
602
|
+
``exped_date``, ``instrument``, ``id`` — no targets) that you predict on and
|
|
603
|
+
submit. The labeled eiq_live_2026 answers are held out server-side and never
|
|
604
|
+
downloadable.
|
|
594
605
|
|
|
595
606
|
Futures is served by the futures endpoint (``version`` is a no-op):
|
|
596
|
-
it returns the real bregen tree to full-scope keys and the
|
|
597
|
-
|
|
607
|
+
it returns the real bregen tree to full-scope keys and the hackathon tree
|
|
608
|
+
(labeled ``validation`` practice set + blind ``live`` eiq_live_2026 scored
|
|
609
|
+
set) to hackathon-scoped keys.
|
|
598
610
|
"""
|
|
599
611
|
if output_path is None:
|
|
600
612
|
output_path = f"{universe}_{split}.parquet"
|
|
@@ -775,50 +787,125 @@ class EverestAPI:
|
|
|
775
787
|
|
|
776
788
|
# -- compute ----------------------------------------------------------
|
|
777
789
|
|
|
778
|
-
def
|
|
790
|
+
def train(
|
|
779
791
|
self,
|
|
780
792
|
model: str = "lightgbm",
|
|
781
|
-
|
|
782
|
-
|
|
793
|
+
*,
|
|
794
|
+
custom_model_fn: str | None = None,
|
|
795
|
+
custom_feature_fn: str | None = None,
|
|
796
|
+
features: str | list[str] = "small",
|
|
797
|
+
target: str | list[str] = "target_everest_20",
|
|
783
798
|
universe: str = "futures",
|
|
784
799
|
params: dict | None = None,
|
|
800
|
+
cv: dict | None = None,
|
|
801
|
+
train_filter: dict | None = None,
|
|
802
|
+
val_filter: dict | None = None,
|
|
803
|
+
transforms: dict | None = None,
|
|
804
|
+
gpu: str = "T4",
|
|
805
|
+
max_hours: float = 1.0,
|
|
785
806
|
) -> dict:
|
|
786
|
-
"""Submit
|
|
807
|
+
"""Submit a unified training job (EVE-1013). Non-blocking.
|
|
808
|
+
|
|
809
|
+
train only — returns downloadable artifacts + metrics; does NOT
|
|
810
|
+
submit/host (use submit_futures_predictions / upload_model). Supply
|
|
811
|
+
a built-in ``model`` preset (lightgbm/xgboost/ridge/mlp/random_forest)
|
|
812
|
+
or ``model="custom"`` with a ``custom_model_fn`` factory; optionally
|
|
813
|
+
add ``custom_feature_fn`` for server-side feature engineering on the
|
|
814
|
+
resident data (no upload). The platform always runs exped-purged
|
|
815
|
+
cross-validation and computes canonical CORR20v2/AIMC/FNC/EAC — note
|
|
816
|
+
the pre-round AIMC is an ESTIMATE vs a static consensus proxy, not
|
|
817
|
+
the live stake-weighted crowd score used at round close.
|
|
818
|
+
|
|
819
|
+
Futures payout uses target_everest_20 (20-day forward return).
|
|
820
|
+
Final score = 0.75 * CORR + 2.25 * AIMC.
|
|
821
|
+
|
|
822
|
+
``train_filter``/``val_filter`` (RowFilter) and ``transforms`` (fit-only
|
|
823
|
+
target transform + sample_weight override) are honored server-side,
|
|
824
|
+
diagnostic-only — never affects payout; see API_CONTRACT.md for the
|
|
825
|
+
full shape and the ``data_spec``/``canonical``/``slice_leak_note``
|
|
826
|
+
fields returned when a slice is applied.
|
|
827
|
+
"""
|
|
787
828
|
body: dict = {
|
|
788
829
|
"model": model,
|
|
789
830
|
"features": features,
|
|
790
831
|
"target": target,
|
|
791
832
|
"universe": universe,
|
|
833
|
+
"gpu": gpu,
|
|
834
|
+
"max_hours": max_hours,
|
|
792
835
|
}
|
|
793
|
-
if
|
|
836
|
+
if custom_model_fn is not None:
|
|
837
|
+
body["custom_model_fn"] = custom_model_fn
|
|
838
|
+
if custom_feature_fn is not None:
|
|
839
|
+
body["custom_feature_fn"] = custom_feature_fn
|
|
840
|
+
if params is not None:
|
|
794
841
|
body["params"] = params
|
|
795
|
-
|
|
842
|
+
if cv is not None:
|
|
843
|
+
body["cv"] = cv
|
|
844
|
+
if train_filter is not None:
|
|
845
|
+
body["train_filter"] = train_filter
|
|
846
|
+
if val_filter is not None:
|
|
847
|
+
body["val_filter"] = val_filter
|
|
848
|
+
if transforms is not None:
|
|
849
|
+
body["transforms"] = transforms
|
|
850
|
+
return self._request("POST", "/api/v1/compute/train", json=body)
|
|
796
851
|
|
|
797
|
-
def
|
|
852
|
+
def quick_train(
|
|
798
853
|
self,
|
|
799
|
-
|
|
800
|
-
|
|
801
|
-
|
|
802
|
-
|
|
803
|
-
|
|
804
|
-
requirements: list[str] | None = None,
|
|
854
|
+
model: str = "lightgbm",
|
|
855
|
+
features: str = "small",
|
|
856
|
+
target: str = "target_everest_20",
|
|
857
|
+
universe: str = "futures",
|
|
858
|
+
params: dict | None = None,
|
|
805
859
|
) -> dict:
|
|
806
|
-
"""
|
|
807
|
-
|
|
808
|
-
|
|
809
|
-
|
|
810
|
-
|
|
811
|
-
|
|
812
|
-
|
|
813
|
-
|
|
814
|
-
|
|
815
|
-
|
|
816
|
-
|
|
860
|
+
"""Deprecated since 0.2.9 — forwards to :meth:`train`.
|
|
861
|
+
|
|
862
|
+
The Tier 1/Tier 2 split (``quick_train``/``custom_train``) was
|
|
863
|
+
replaced server-side by a single unified ``train`` tool (EVE-1013).
|
|
864
|
+
This shim keeps the old call signature working but always submits
|
|
865
|
+
through the new ``/api/v1/compute/train`` endpoint.
|
|
866
|
+
"""
|
|
867
|
+
warnings.warn(
|
|
868
|
+
"quick_train() is deprecated; use train() instead (EVE-1013).",
|
|
869
|
+
DeprecationWarning,
|
|
870
|
+
stacklevel=2,
|
|
871
|
+
)
|
|
872
|
+
return self.train(
|
|
873
|
+
model=model, features=features, target=target, universe=universe, params=params
|
|
874
|
+
)
|
|
875
|
+
|
|
876
|
+
def custom_train(self, *args: Any, **kwargs: Any) -> dict:
|
|
877
|
+
"""Removed since 0.2.9 — the Tier-2 script-based training path no longer exists.
|
|
878
|
+
|
|
879
|
+
Use :meth:`train` with ``model="custom"`` and a ``custom_model_fn``
|
|
880
|
+
source string instead; it now runs inside a sandboxed, network-denied
|
|
881
|
+
environment rather than a dedicated GPU pod running an arbitrary script.
|
|
882
|
+
"""
|
|
883
|
+
warnings.warn(
|
|
884
|
+
"custom_train() is removed; use train(model='custom', custom_model_fn=...).",
|
|
885
|
+
DeprecationWarning,
|
|
886
|
+
stacklevel=2,
|
|
887
|
+
)
|
|
888
|
+
raise EverestError(
|
|
889
|
+
410,
|
|
890
|
+
"custom_train (Tier-2 script) was removed. Use "
|
|
891
|
+
"train(model='custom', custom_model_fn='def build_model(params): ...').",
|
|
892
|
+
)
|
|
817
893
|
|
|
818
894
|
def get_job_status(self, job_id: str) -> dict:
|
|
819
895
|
"""Poll compute job status."""
|
|
820
896
|
return self._request("GET", f"/api/v1/compute/jobs/{job_id}")
|
|
821
897
|
|
|
898
|
+
def list_compute_jobs(self, limit: int = 50, status: str | None = None) -> dict:
|
|
899
|
+
"""List your compute jobs, newest first (rows exclude config/output payloads)."""
|
|
900
|
+
params: dict = {"limit": limit}
|
|
901
|
+
if status is not None:
|
|
902
|
+
params["status"] = status
|
|
903
|
+
return self._request("GET", "/api/v1/compute/jobs", params=params)
|
|
904
|
+
|
|
905
|
+
def get_job_log(self, job_id: str) -> dict:
|
|
906
|
+
"""Lifecycle event trail for one compute job — debug a failed/stuck job."""
|
|
907
|
+
return self._request("GET", f"/api/v1/compute/jobs/{job_id}/logs")
|
|
908
|
+
|
|
822
909
|
def get_model_download_url(self, job_id: str) -> dict:
|
|
823
910
|
"""Get a presigned download URL for a trained model."""
|
|
824
911
|
return self._request("GET", f"/api/v1/compute/jobs/{job_id}/model")
|
|
@@ -864,7 +951,7 @@ class EverestAPI:
|
|
|
864
951
|
return self._request("GET", "/api/v1/compute/credits")
|
|
865
952
|
|
|
866
953
|
def list_model_templates(self) -> dict:
|
|
867
|
-
"""List available
|
|
954
|
+
"""List available train model templates."""
|
|
868
955
|
return self._request("GET", "/api/v1/compute/models")
|
|
869
956
|
|
|
870
957
|
# -- registration (no auth) ------------------------------------------
|
|
@@ -933,7 +1020,8 @@ class EverestAPI:
|
|
|
933
1020
|
|
|
934
1021
|
With a hackathon-scoped key there is no live round — the response is a
|
|
935
1022
|
diagnostics-mode payload (mode='diagnostics_hackathon') directing you to
|
|
936
|
-
|
|
1023
|
+
tune offline on the labeled validation set, then predict on the blind
|
|
1024
|
+
eiq_live_2026 set and submit_validation_diagnostics.
|
|
937
1025
|
"""
|
|
938
1026
|
return self._request(
|
|
939
1027
|
"GET",
|
|
@@ -951,7 +1039,10 @@ class EverestAPI:
|
|
|
951
1039
|
"""GET /api/v1/get_started — mode-aware orientation.
|
|
952
1040
|
|
|
953
1041
|
Returns what to do next given this key's scope: the display-only
|
|
954
|
-
diagnostics-hackathon loop
|
|
1042
|
+
diagnostics-hackathon loop (tune and self-score offline on the labeled
|
|
1043
|
+
validation set, then predict on the blind eiq_live_2026 set and submit —
|
|
1044
|
+
ranked on out-of-sample 2026 CORR on target_everest_20), or the live
|
|
1045
|
+
futures tournament flow.
|
|
955
1046
|
"""
|
|
956
1047
|
return self._request("GET", "/api/v1/get_started")
|
|
957
1048
|
|
|
@@ -238,7 +238,7 @@ TOOLS = [
|
|
|
238
238
|
},
|
|
239
239
|
{
|
|
240
240
|
"name": "download_dataset",
|
|
241
|
-
"description": "Download a dataset split (train/validation/live) as a parquet file. Returns the local file path.",
|
|
241
|
+
"description": "Download a dataset split (train/validation/live) as a parquet file. Returns the local file path. In hackathon mode: 'validation' is the LABELED practice set (features + all target_* columns) for offline tuning and self-scoring; 'live' is the BLIND eiq_live_2026 out-of-sample set (columns exactly exped, exped_date, instrument, id — NO targets) that you predict on and submit. The labeled 2026 answers are held out server-side and never downloadable.",
|
|
242
242
|
"inputSchema": {
|
|
243
243
|
"type": "object",
|
|
244
244
|
"properties": {
|
|
@@ -270,42 +270,64 @@ TOOLS = [
|
|
|
270
270
|
},
|
|
271
271
|
},
|
|
272
272
|
{
|
|
273
|
-
"name": "
|
|
274
|
-
"description": "Train a model on Everesteer data
|
|
273
|
+
"name": "train",
|
|
274
|
+
"description": "Train a model on Everesteer data (no data upload — the platform uses the same obfuscated dataset you download). The PLATFORM loads data, runs exped-purged/embargoed cross-validation, computes canonical tournament metrics (0.75*CORR + 2.25*AIMC; AIMC before round close is a PRE-SUBMISSION ESTIMATE vs a static consensus proxy — real AIMC is scored vs the live stake-weighted crowd at round close), pickles the final model, and returns DOWNLOADABLE artifacts (model .pkl + validation/live prediction files) plus the metrics. Supply a built-in `model` preset (lightgbm/xgboost/ridge/mlp/random_forest), or `model='custom'` with a `custom_model_fn` (and optional `custom_feature_fn`) to run your own code — it executes inside an isolated, network-denied sandbox with no filesystem access and never sees held-out targets; a script rejected by the static safety scanner returns 400. `train` does NOT submit predictions or host a model on your behalf — use submit_futures_predictions/submit_predictions_file to submit and upload_model to host the .pkl. Poll with get_job_status() — the completed result carries the metrics + a predictions download URL; fetch the model file separately via get_model_download_url().",
|
|
275
275
|
"inputSchema": {
|
|
276
276
|
"type": "object",
|
|
277
277
|
"properties": {
|
|
278
|
-
"model": {
|
|
279
|
-
|
|
278
|
+
"model": {
|
|
279
|
+
"type": "string",
|
|
280
|
+
"enum": ["lightgbm", "xgboost", "ridge", "mlp", "random_forest", "custom"],
|
|
281
|
+
"default": "lightgbm",
|
|
282
|
+
},
|
|
283
|
+
"custom_model_fn": {
|
|
284
|
+
"type": "string",
|
|
285
|
+
"description": "Python source defining build_model(params)->estimator, required when model='custom'. Runs in an isolated, network-denied sandbox with no filesystem access and never sees held-out targets. Rejected by the static safety scanner (returns 400) if it references disallowed modules/builtins.",
|
|
286
|
+
},
|
|
287
|
+
"custom_feature_fn": {
|
|
280
288
|
"type": "string",
|
|
281
|
-
"
|
|
289
|
+
"description": "OPTIONAL Python source defining build_feature_transformer(params)->sklearn transformer for server-side feature engineering on the resident features. Usable with any model (preset or custom), in the same isolated sandbox as custom_model_fn. Same static-scanner gate (returns 400 on rejection).",
|
|
290
|
+
},
|
|
291
|
+
"params": {"type": "object", "description": "Model hyperparameters (optional)"},
|
|
292
|
+
"features": {
|
|
293
|
+
"oneOf": [
|
|
294
|
+
{"type": "string", "enum": ["small", "medium", "all"]},
|
|
295
|
+
{"type": "array", "items": {"type": "string"}},
|
|
296
|
+
],
|
|
282
297
|
"default": "small",
|
|
298
|
+
"description": "Preset (small/medium/all) or an explicit list of feature_* names.",
|
|
283
299
|
},
|
|
284
300
|
"target": {
|
|
285
|
-
"
|
|
301
|
+
"oneOf": [
|
|
302
|
+
{"type": "string"},
|
|
303
|
+
{"type": "array", "items": {"type": "string"}},
|
|
304
|
+
],
|
|
286
305
|
"default": "target_everest_20",
|
|
287
|
-
"description": "Target column. Futures payout uses target_everest_20 (20-day forward return).
|
|
306
|
+
"description": "Target column. Futures payout uses target_everest_20 (20-day forward return).",
|
|
288
307
|
},
|
|
289
308
|
"universe": {"type": "string", "default": "futures"},
|
|
290
|
-
"
|
|
309
|
+
"cv": {
|
|
310
|
+
"type": "object",
|
|
311
|
+
"description": "Cross-validation spec: {scheme, folds, embargo}. scheme is exped_purged (default), chain_group, or combined; embargo defaults to the label horizon when null.",
|
|
312
|
+
},
|
|
313
|
+
"train_filter": {
|
|
314
|
+
"type": "object",
|
|
315
|
+
"description": "Row filter (exped/climb_difficulty/sample/feature_filter) applied to the training split server-side, no re-upload — diagnostic-only, never affects payout. Full RowFilter shape in API_CONTRACT.md.",
|
|
316
|
+
},
|
|
317
|
+
"val_filter": {
|
|
318
|
+
"type": "object",
|
|
319
|
+
"description": "Same RowFilter shape as train_filter, applied to the validation split (layered under the platform's own validation clamp, if any) — diagnostic-only, never affects payout.",
|
|
320
|
+
},
|
|
321
|
+
"transforms": {
|
|
322
|
+
"type": "object",
|
|
323
|
+
"description": "Fit-only target transform (residualize_to_benchmark/subtract_benchmark_zscore) and/or sample_weight override — scoring always uses the untouched canonical target. Full shape in API_CONTRACT.md.",
|
|
324
|
+
},
|
|
325
|
+
"gpu": {"type": "string", "enum": ["T4", "A10G", "A100"], "default": "T4"},
|
|
326
|
+
"max_hours": {"type": "number", "minimum": 0.1, "maximum": 4.0, "default": 1.0},
|
|
291
327
|
},
|
|
292
328
|
"required": ["model"],
|
|
293
329
|
},
|
|
294
330
|
},
|
|
295
|
-
{
|
|
296
|
-
"name": "custom_train",
|
|
297
|
-
"description": "Run a custom Python script on a GPU pod. Pod has EverestAPI SDK + PyTorch + ML libs pre-installed. Costs $0.50-1.79/hr.",
|
|
298
|
-
"inputSchema": {
|
|
299
|
-
"type": "object",
|
|
300
|
-
"properties": {
|
|
301
|
-
"script": {"type": "string", "description": "Python script content"},
|
|
302
|
-
"gpu": {"type": "string", "enum": ["T4", "A40", "A100"], "default": "T4"},
|
|
303
|
-
"max_hours": {"type": "number", "default": 1},
|
|
304
|
-
"requirements": {"type": "array", "items": {"type": "string"}},
|
|
305
|
-
},
|
|
306
|
-
"required": ["script"],
|
|
307
|
-
},
|
|
308
|
-
},
|
|
309
331
|
{
|
|
310
332
|
"name": "get_job_status",
|
|
311
333
|
"description": "Check status of a compute job. Returns status, cost, output files when complete.",
|
|
@@ -314,7 +336,7 @@ TOOLS = [
|
|
|
314
336
|
"properties": {
|
|
315
337
|
"job_id": {
|
|
316
338
|
"type": "string",
|
|
317
|
-
"description": "Job ID from
|
|
339
|
+
"description": "Job ID from a train job",
|
|
318
340
|
},
|
|
319
341
|
},
|
|
320
342
|
"required": ["job_id"],
|
|
@@ -328,7 +350,7 @@ TOOLS = [
|
|
|
328
350
|
"properties": {
|
|
329
351
|
"job_id": {
|
|
330
352
|
"type": "string",
|
|
331
|
-
"description": "Job ID from
|
|
353
|
+
"description": "Job ID from a train job",
|
|
332
354
|
},
|
|
333
355
|
},
|
|
334
356
|
"required": ["job_id"],
|
|
@@ -365,7 +387,7 @@ TOOLS = [
|
|
|
365
387
|
},
|
|
366
388
|
{
|
|
367
389
|
"name": "get_current_round",
|
|
368
|
-
"description": "Get the current active round for a tournament (defaults to 'futures'; equities is unlaunched). With a hackathon key there is no live round — the response is a diagnostics-mode payload directing you to
|
|
390
|
+
"description": "Get the current active round for a tournament (defaults to 'futures'; equities is unlaunched). With a hackathon key there is no live round — the response is a diagnostics-mode payload directing you to tune offline on the labeled validation set, then download the blind eiq_live_2026 set, predict, and submit_validation_diagnostics.",
|
|
369
391
|
"inputSchema": {
|
|
370
392
|
"type": "object",
|
|
371
393
|
"properties": {
|
|
@@ -380,7 +402,7 @@ TOOLS = [
|
|
|
380
402
|
},
|
|
381
403
|
{
|
|
382
404
|
"name": "get_started",
|
|
383
|
-
"description": "Mode-aware orientation: returns what to do next given your API key's scope. A hackathon key gets the display-only diagnostics loop (objective: maximize out-of-sample CORR on target_everest_20); a full key gets the live futures tournament flow. Start here.",
|
|
405
|
+
"description": "Mode-aware orientation: returns what to do next given your API key's scope. A hackathon key gets the display-only diagnostics loop (tune and self-score offline on the labeled validation set, then predict on the blind eiq_live_2026 set and submit; objective: maximize out-of-sample 2026 CORR on target_everest_20, in-sample fit is not rewarded); a full key gets the live futures tournament flow. Start here.",
|
|
384
406
|
"inputSchema": {
|
|
385
407
|
"type": "object",
|
|
386
408
|
"properties": {},
|
|
@@ -546,13 +568,15 @@ TOOLS = [
|
|
|
546
568
|
{
|
|
547
569
|
"name": "submit_validation_diagnostics",
|
|
548
570
|
"description": (
|
|
549
|
-
"Upload a NEW
|
|
550
|
-
"≤100 MB)
|
|
551
|
-
"
|
|
552
|
-
"
|
|
553
|
-
"(
|
|
554
|
-
"
|
|
555
|
-
"
|
|
571
|
+
"Upload a NEW predictions file (parquet/CSV with id+prediction columns, "
|
|
572
|
+
"≤100 MB) generated on the blind eiq_live_2026 out-of-sample set; the platform "
|
|
573
|
+
"scores it server-side against the held-out labeled eiq_live_2026 answers (never "
|
|
574
|
+
"downloadable). The required id set is the blind eiq_live_2026 set — fetch it with "
|
|
575
|
+
"download_dataset(split='live') and use it as the upload template "
|
|
576
|
+
"(its id column, your own prediction column). Tune and self-score offline on the "
|
|
577
|
+
"labeled validation set first (see the scoring helper). Use this only when you have "
|
|
578
|
+
"fresh predictions to score — to read the model's EXISTING latest result without "
|
|
579
|
+
"re-scoring (instant, no wait), call run_validation_diagnostics(model_id) instead. "
|
|
556
580
|
"The platform scores a 9-metric panel (CORR20, BMC, FNC, Sharpe, std dev, "
|
|
557
581
|
"feature-exposure, Max Drawdown, autocorrelation, example-preds-corr) asynchronously "
|
|
558
582
|
"and surfaces the run in the website Validation Diagnostics rail. By default "
|
|
@@ -594,9 +618,10 @@ TOOLS = [
|
|
|
594
618
|
{
|
|
595
619
|
"name": "get_diagnostics_leaderboard",
|
|
596
620
|
"description": (
|
|
597
|
-
"Get the validation diagnostics leaderboard — global ranking of
|
|
598
|
-
"
|
|
599
|
-
"runs and marks your own entries;
|
|
621
|
+
"Get the validation diagnostics leaderboard — global ranking of agents by their "
|
|
622
|
+
"out-of-sample eiq_live_2026 CORR on target_everest_20 (in-sample fit is not rewarded). "
|
|
623
|
+
"view='agents' (default) ranks participant model runs and marks your own entries; "
|
|
624
|
+
"view='benchmarks' ranks the official platform benchmarks."
|
|
600
625
|
),
|
|
601
626
|
"inputSchema": {
|
|
602
627
|
"type": "object",
|
|
@@ -894,6 +919,19 @@ TOOLSETS: dict[str, set[str]] = {
|
|
|
894
919
|
"get_scores",
|
|
895
920
|
"get_leaderboard",
|
|
896
921
|
"get_capabilities",
|
|
922
|
+
# EVE-967: the diagnostics upload + leaderboard are the entire hackathon
|
|
923
|
+
# flow get_started steers to; advertise them by default (core) so a
|
|
924
|
+
# hackathon key discovers them without setting EIQ_MCP_TOOLSETS. The rest
|
|
925
|
+
# of the diagnostics lifecycle stays in "diagnostics".
|
|
926
|
+
"submit_validation_diagnostics",
|
|
927
|
+
"get_diagnostics_leaderboard",
|
|
928
|
+
# EVE-970: hosted training is a first-class flow now that every account
|
|
929
|
+
# carries a compute grant — advertise train + get_compute_credits by
|
|
930
|
+
# default so agents discover the hosted-training entry point without
|
|
931
|
+
# setting EIQ_MCP_TOOLSETS. get_job_status stays in "compute" (hidden
|
|
932
|
+
# tools remain callable by name). Mirrors the platform MCP server.
|
|
933
|
+
"train",
|
|
934
|
+
"get_compute_credits",
|
|
897
935
|
},
|
|
898
936
|
"data": {
|
|
899
937
|
"get_universe",
|
|
@@ -911,15 +949,12 @@ TOOLSETS: dict[str, set[str]] = {
|
|
|
911
949
|
},
|
|
912
950
|
"submit": {"submit_predictions", "upload_model", "get_upload_status"},
|
|
913
951
|
"diagnostics": {
|
|
952
|
+
# submit_validation_diagnostics + get_diagnostics_leaderboard promoted to
|
|
953
|
+
# "core" (EVE-967) — every name still lives in exactly one group.
|
|
914
954
|
"run_validation_diagnostics",
|
|
915
|
-
"submit_validation_diagnostics",
|
|
916
|
-
"get_diagnostics_leaderboard",
|
|
917
955
|
},
|
|
918
956
|
"compute": {
|
|
919
|
-
"quick_train",
|
|
920
|
-
"custom_train",
|
|
921
957
|
"get_job_status",
|
|
922
|
-
"get_compute_credits",
|
|
923
958
|
},
|
|
924
959
|
"staking": {
|
|
925
960
|
"stake_on_model",
|
|
@@ -1081,21 +1116,8 @@ def _dispatch(name: str, arguments: dict) -> str:
|
|
|
1081
1116
|
output_path=arguments.get("output_path"),
|
|
1082
1117
|
)
|
|
1083
1118
|
result = {"file_path": path, "status": "downloaded"}
|
|
1084
|
-
elif name == "
|
|
1085
|
-
result = client.
|
|
1086
|
-
model=arguments["model"],
|
|
1087
|
-
features=arguments.get("features", "small"),
|
|
1088
|
-
target=arguments.get("target", "target_everest_20"),
|
|
1089
|
-
universe=arguments.get("universe", "futures"),
|
|
1090
|
-
params=arguments.get("params"),
|
|
1091
|
-
)
|
|
1092
|
-
elif name == "custom_train":
|
|
1093
|
-
result = client.custom_train(
|
|
1094
|
-
script=arguments.get("script"),
|
|
1095
|
-
gpu=arguments.get("gpu", "T4"),
|
|
1096
|
-
max_hours=arguments.get("max_hours", 1),
|
|
1097
|
-
requirements=arguments.get("requirements"),
|
|
1098
|
-
)
|
|
1119
|
+
elif name == "train":
|
|
1120
|
+
result = client.train(**arguments)
|
|
1099
1121
|
elif name == "get_job_status":
|
|
1100
1122
|
result = client.get_job_status(job_id=arguments["job_id"])
|
|
1101
1123
|
elif name == "get_model_download_url":
|
|
@@ -1275,9 +1297,14 @@ def _error_payload(exc: Exception) -> str:
|
|
|
1275
1297
|
|
|
1276
1298
|
if isinstance(exc, EverestError):
|
|
1277
1299
|
code = exc.status_code
|
|
1300
|
+
# FastAPI wraps errors as {"detail": {"code": ..., "message": ...}} —
|
|
1301
|
+
# pull the machine code so 403 can distinguish a scope restriction
|
|
1302
|
+
# (not available in diagnostics mode) from a genuine auth failure.
|
|
1303
|
+
inner = exc.detail.get("detail") if isinstance(exc.detail, dict) else None
|
|
1304
|
+
err_code = inner.get("code") if isinstance(inner, dict) else None
|
|
1305
|
+
|
|
1278
1306
|
hints = {
|
|
1279
1307
|
401: "Auth failed — set EIQ_API_KEY to your Everesteer API key.",
|
|
1280
|
-
403: "Auth failed — set EIQ_API_KEY to your Everesteer API key.",
|
|
1281
1308
|
404: "Not found — check the model_id / upload_id exists.",
|
|
1282
1309
|
409: "A diagnostics run is already in flight for this model — read it with "
|
|
1283
1310
|
"run_validation_diagnostics(model_id) or wait for it to finish.",
|
|
@@ -1285,10 +1312,19 @@ def _error_payload(exc: Exception) -> str:
|
|
|
1285
1312
|
504: "Still computing — scoring can take 15-20 min; poll again shortly.",
|
|
1286
1313
|
}
|
|
1287
1314
|
hint = hints.get(code)
|
|
1315
|
+
if code == 403:
|
|
1316
|
+
if err_code in ("scope_mismatch", "scope_restricted"):
|
|
1317
|
+
hint = (
|
|
1318
|
+
"Not available in diagnostics (hackathon) mode — this is a "
|
|
1319
|
+
"live-tournament / full-scope action. Use the diagnostics flow: "
|
|
1320
|
+
"submit_validation_diagnostics + get_diagnostics_leaderboard."
|
|
1321
|
+
)
|
|
1322
|
+
else:
|
|
1323
|
+
hint = "Auth failed — set EIQ_API_KEY to your Everesteer API key."
|
|
1288
1324
|
if hint is None and 400 <= code < 500:
|
|
1289
|
-
hint = "Bad request —
|
|
1325
|
+
hint = "Bad request — check the arguments match the tool schema."
|
|
1290
1326
|
elif hint is None and code >= 500:
|
|
1291
|
-
hint = "Server error — retry; if it persists the
|
|
1327
|
+
hint = "Server error — retry; if it persists the request may have failed."
|
|
1292
1328
|
return json.dumps({"error": exc.detail, "status": code, "hint": hint})
|
|
1293
1329
|
return json.dumps({"error": str(exc)})
|
|
1294
1330
|
|
|
@@ -1402,9 +1438,13 @@ _MCP_INSTRUCTIONS = (
|
|
|
1402
1438
|
"Everesteer tournament MCP server. Call get_started first — it is mode-aware and tells you "
|
|
1403
1439
|
"what to do next based on your API key's scope.\n\n"
|
|
1404
1440
|
"Hackathon mode (a hackathon-scoped key): there is no live tournament round and no "
|
|
1405
|
-
"staking or payout — this is a display-only diagnostics event.
|
|
1406
|
-
"
|
|
1407
|
-
"
|
|
1441
|
+
"staking or payout — this is a display-only diagnostics event. Tune and self-score "
|
|
1442
|
+
"offline on the LABELED validation set (features + target_* columns, via the scoring "
|
|
1443
|
+
"helper), then predict on the BLIND eiq_live_2026 set (columns exped, exped_date, "
|
|
1444
|
+
"instrument, id — no targets) and submit; the leaderboard ranks your OUT-OF-SAMPLE 2026 "
|
|
1445
|
+
"CORR on target_everest_20. In-sample fit is not rewarded. Flow: "
|
|
1446
|
+
"download_dataset(split='validation') -> tune + self-score offline -> "
|
|
1447
|
+
"download_dataset(split='live') -> predict -> submit_validation_diagnostics -> "
|
|
1408
1448
|
"get_diagnostics_leaderboard. Round/universe/feature/benchmark reads return empty "
|
|
1409
1449
|
"diagnostics-mode payloads, not live data.\n\n"
|
|
1410
1450
|
"Tournament mode (a full-scope key): submit daily futures predictions; payout = "
|
|
@@ -1436,8 +1476,7 @@ _NEXT_ACTIONS: dict[str, list[str]] = {
|
|
|
1436
1476
|
"download_dataset": ["create_model", "submit_futures_predictions"],
|
|
1437
1477
|
"download_benchmark": ["submit_futures_predictions"],
|
|
1438
1478
|
"get_benchmarks": ["download_benchmark"],
|
|
1439
|
-
"
|
|
1440
|
-
"custom_train": ["get_job_status"],
|
|
1479
|
+
"train": ["get_job_status"],
|
|
1441
1480
|
"get_job_status": ["get_model_download_url"], # conditional
|
|
1442
1481
|
"get_model_download_url": ["upload_model", "create_model"],
|
|
1443
1482
|
"create_model": ["submit_futures_predictions", "upload_model"],
|
|
@@ -1469,7 +1508,7 @@ _NEXT_ACTIONS: dict[str, list[str]] = {
|
|
|
1469
1508
|
"get_deposit_address": ["get_forwarder_balance", "relay_stake"],
|
|
1470
1509
|
"get_forwarder_balance": ["relay_stake"],
|
|
1471
1510
|
"withdraw_usdc": ["get_forwarder_balance"],
|
|
1472
|
-
"get_compute_credits": ["
|
|
1511
|
+
"get_compute_credits": ["train"],
|
|
1473
1512
|
"get_notifications": [],
|
|
1474
1513
|
"get_badges": [],
|
|
1475
1514
|
}
|
|
@@ -1494,8 +1533,7 @@ _NARRATION: dict[str, str] = {
|
|
|
1494
1533
|
"submit_predictions": (
|
|
1495
1534
|
"Submitted predictions for model '{model_id}'. Scoring starts at round close."
|
|
1496
1535
|
),
|
|
1497
|
-
"
|
|
1498
|
-
"custom_train": "Custom training job queued.",
|
|
1536
|
+
"train": "Training job queued — returns downloadable artifacts + metrics, not a submission.",
|
|
1499
1537
|
"get_job_status": "Training job status: {status}.",
|
|
1500
1538
|
"get_model_download_url": "Generated a download URL for the trained model (valid ~1h).",
|
|
1501
1539
|
"upload_model": "Model artifact uploaded; validation runs asynchronously.",
|
|
@@ -1555,7 +1593,7 @@ def _next_actions_for(name: str, arguments: dict, data: dict) -> list[str]:
|
|
|
1555
1593
|
if status in ("completed", "done", "succeeded", "success"):
|
|
1556
1594
|
return ["get_model_download_url", "create_model"]
|
|
1557
1595
|
if status in ("failed", "error"):
|
|
1558
|
-
return ["
|
|
1596
|
+
return ["train"]
|
|
1559
1597
|
return ["get_job_status"] # still running — keep polling
|
|
1560
1598
|
if name == "get_upload_status":
|
|
1561
1599
|
if status in ("validated", "done", "success", "ready", "active"):
|
|
@@ -27,6 +27,14 @@ Quickstart::
|
|
|
27
27
|
for e in val.exped.unique()
|
|
28
28
|
]
|
|
29
29
|
print("mean CORR20:", sum(corrs) / len(corrs))
|
|
30
|
+
|
|
31
|
+
In hackathon mode the validation set ships LABELED (features + ``target_*``
|
|
32
|
+
columns), so this helper is the offline self-scoring path: use it to tune your
|
|
33
|
+
model before you submit. It scores in-sample and is for tuning only — the
|
|
34
|
+
official hackathon leaderboard scores your submitted predictions on the BLIND
|
|
35
|
+
``eiq_live_2026`` out-of-sample set server-side (its labeled answers are held
|
|
36
|
+
out and never downloadable), ranking your 2026 CORR on ``target_everest_20``.
|
|
37
|
+
In-sample fit is not rewarded.
|
|
30
38
|
"""
|
|
31
39
|
|
|
32
40
|
from __future__ import annotations
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: everestapi
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.9
|
|
4
4
|
Summary: Python SDK for the Everesteer prediction tournament platform
|
|
5
5
|
Author-email: Everesteer <support@everesteer.ai>
|
|
6
6
|
License-Expression: MIT
|
|
@@ -116,16 +116,23 @@ everestapi submit --model my-model --file predictions.parquet
|
|
|
116
116
|
|
|
117
117
|
### Data & diagnostics
|
|
118
118
|
|
|
119
|
+
The hackathon is a display-only diagnostics event. **Tune and self-score offline
|
|
120
|
+
on the labeled validation set** (features + `target_*` columns), then **predict on
|
|
121
|
+
the blind `eiq_live_2026` set and submit** — the leaderboard ranks your
|
|
122
|
+
out-of-sample 2026 CORR on `target_everest_20`. In-sample fit is not rewarded. The
|
|
123
|
+
labeled `eiq_live_2026` answers are held out server-side and are never downloadable.
|
|
124
|
+
|
|
119
125
|
```python
|
|
120
|
-
|
|
121
|
-
api.
|
|
122
|
-
api.get_dataset_info(universe="futures")
|
|
123
|
-
api.get_diagnostics(model_id="my-model")
|
|
126
|
+
# Labeled practice set — tune + self-score offline with everestapi.scoring:
|
|
127
|
+
api.download_dataset(universe="futures", split="validation")
|
|
124
128
|
|
|
125
|
-
#
|
|
126
|
-
#
|
|
127
|
-
api.download_dataset(universe="futures", split="
|
|
129
|
+
# Blind scored set (columns: exped, exped_date, instrument, id — no targets).
|
|
130
|
+
# Predict on it, then submit; it is also the upload id template.
|
|
131
|
+
api.download_dataset(universe="futures", split="live")
|
|
128
132
|
api.submit_validation_diagnostics(model_id="my-model", predictions=df)
|
|
133
|
+
|
|
134
|
+
api.get_dataset_info(universe="futures")
|
|
135
|
+
api.get_diagnostics(model_id="my-model")
|
|
129
136
|
```
|
|
130
137
|
|
|
131
138
|
### Plotting (optional `viz` extra)
|
|
@@ -157,11 +164,18 @@ df = pd.DataFrame(lb["entries"])
|
|
|
157
164
|
### Serverless compute
|
|
158
165
|
|
|
159
166
|
```python
|
|
160
|
-
#
|
|
161
|
-
|
|
162
|
-
|
|
163
|
-
|
|
164
|
-
|
|
167
|
+
# Built-in preset (lightgbm/xgboost/ridge/mlp/random_forest) — no data upload,
|
|
168
|
+
# the platform trains against the same obfuscated dataset you download.
|
|
169
|
+
job = api.train(model="lightgbm", features="small", target="target_everest_20")
|
|
170
|
+
|
|
171
|
+
# model="custom" — your own model factory, run server-side in an isolated,
|
|
172
|
+
# network-denied sandbox (no filesystem access, never sees held-out targets)
|
|
173
|
+
job = api.train(
|
|
174
|
+
model="custom",
|
|
175
|
+
custom_model_fn="def build_model(params):\n from sklearn.linear_model import Ridge\n return Ridge(**params)",
|
|
176
|
+
gpu="A100",
|
|
177
|
+
max_hours=2.0,
|
|
178
|
+
)
|
|
165
179
|
|
|
166
180
|
# Wait and download
|
|
167
181
|
result = api.wait_for_job(job["job_id"])
|
|
@@ -23,6 +23,7 @@ tests/test_diagnostics.py
|
|
|
23
23
|
tests/test_eve953_mcp_progress.py
|
|
24
24
|
tests/test_eve957_mcp_annotations_resources.py
|
|
25
25
|
tests/test_eve959_toolsets.py
|
|
26
|
+
tests/test_eve967_mcp_discoverability.py
|
|
26
27
|
tests/test_json_or_raise.py
|
|
27
28
|
tests/test_mcp_and_models.py
|
|
28
29
|
tests/test_prediction_range.py
|
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
"""EVE-967 — the diagnostics upload + leaderboard tools are discoverable on the
|
|
2
|
+
default (core) toolset, and MCP error hints don't mislead:
|
|
3
|
+
- a 403 scope restriction points at the diagnostics flow, not "rotate your key";
|
|
4
|
+
- the generic 400 fallback is tool-neutral (no diagnostics-upload bleed).
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import json
|
|
10
|
+
|
|
11
|
+
import everestapi.mcp.server as S
|
|
12
|
+
from everestapi.client import EverestError
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def test_diagnostics_tools_advertised_by_default(monkeypatch):
|
|
16
|
+
monkeypatch.delenv("EIQ_MCP_TOOLSETS", raising=False)
|
|
17
|
+
enabled = S._enabled_tool_names()
|
|
18
|
+
assert "submit_validation_diagnostics" in enabled
|
|
19
|
+
assert "get_diagnostics_leaderboard" in enabled
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def test_403_scope_mismatch_hint_points_to_diagnostics():
|
|
23
|
+
exc = EverestError(403, {"detail": {"code": "scope_mismatch", "message": "x"}})
|
|
24
|
+
payload = json.loads(S._error_payload(exc))
|
|
25
|
+
assert payload["status"] == 403
|
|
26
|
+
assert "diagnostics" in payload["hint"].lower()
|
|
27
|
+
assert "set EIQ_API_KEY" not in payload["hint"]
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def test_403_plain_auth_still_says_set_key():
|
|
31
|
+
exc = EverestError(403, "forbidden") # no machine code → generic auth hint
|
|
32
|
+
payload = json.loads(S._error_payload(exc))
|
|
33
|
+
assert "EIQ_API_KEY" in payload["hint"]
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def test_generic_400_hint_is_tool_neutral():
|
|
37
|
+
exc = EverestError(400, {"detail": "bad"})
|
|
38
|
+
payload = json.loads(S._error_payload(exc))
|
|
39
|
+
assert "id+prediction" not in payload["hint"]
|
|
40
|
+
assert "schema" in payload["hint"].lower()
|
|
@@ -514,6 +514,8 @@ _SDK_ONLY_SANCTIONED = {
|
|
|
514
514
|
"get_multipliers",
|
|
515
515
|
"get_current_season",
|
|
516
516
|
"get_job_output",
|
|
517
|
+
"list_compute_jobs", # EVE-1013 — job lookup, same category as get_job_output
|
|
518
|
+
"get_job_log", # EVE-1013 — job lookup, same category as get_job_output
|
|
517
519
|
"get_diagnostics_run",
|
|
518
520
|
"get_validation_panel",
|
|
519
521
|
"get_validation_diagnostics",
|
|
@@ -530,6 +532,11 @@ _SDK_ONLY_SANCTIONED = {
|
|
|
530
532
|
"unstake",
|
|
531
533
|
"confirm_stake",
|
|
532
534
|
"cancel_job",
|
|
535
|
+
# EVE-1013 — quick_train/custom_train replaced by the unified `train` tool;
|
|
536
|
+
# both remain on the client as deprecated shims (quick_train forwards to
|
|
537
|
+
# train(), custom_train raises) but are no longer registered as MCP tools.
|
|
538
|
+
"quick_train",
|
|
539
|
+
"custom_train",
|
|
533
540
|
}
|
|
534
541
|
|
|
535
542
|
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|