everestapi 0.2.7__tar.gz → 0.2.9__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (32) hide show
  1. {everestapi-0.2.7/src/everestapi.egg-info → everestapi-0.2.9}/PKG-INFO +27 -13
  2. {everestapi-0.2.7 → everestapi-0.2.9}/README.md +26 -12
  3. {everestapi-0.2.7 → everestapi-0.2.9}/src/everestapi/__init__.py +1 -1
  4. {everestapi-0.2.7 → everestapi-0.2.9}/src/everestapi/client.py +123 -32
  5. {everestapi-0.2.7 → everestapi-0.2.9}/src/everestapi/mcp/server.py +107 -69
  6. {everestapi-0.2.7 → everestapi-0.2.9}/src/everestapi/scoring.py +8 -0
  7. {everestapi-0.2.7 → everestapi-0.2.9/src/everestapi.egg-info}/PKG-INFO +27 -13
  8. {everestapi-0.2.7 → everestapi-0.2.9}/src/everestapi.egg-info/SOURCES.txt +1 -0
  9. everestapi-0.2.9/tests/test_eve967_mcp_discoverability.py +40 -0
  10. {everestapi-0.2.7 → everestapi-0.2.9}/tests/test_mcp_and_models.py +7 -0
  11. {everestapi-0.2.7 → everestapi-0.2.9}/LICENSE +0 -0
  12. {everestapi-0.2.7 → everestapi-0.2.9}/pyproject.toml +0 -0
  13. {everestapi-0.2.7 → everestapi-0.2.9}/setup.cfg +0 -0
  14. {everestapi-0.2.7 → everestapi-0.2.9}/src/everestapi/__main__.py +0 -0
  15. {everestapi-0.2.7 → everestapi-0.2.9}/src/everestapi/cli.py +0 -0
  16. {everestapi-0.2.7 → everestapi-0.2.9}/src/everestapi/mcp/__init__.py +0 -0
  17. {everestapi-0.2.7 → everestapi-0.2.9}/src/everestapi/mcp/__main__.py +0 -0
  18. {everestapi-0.2.7 → everestapi-0.2.9}/src/everestapi/plots.py +0 -0
  19. {everestapi-0.2.7 → everestapi-0.2.9}/src/everestapi/types.py +0 -0
  20. {everestapi-0.2.7 → everestapi-0.2.9}/src/everestapi.egg-info/dependency_links.txt +0 -0
  21. {everestapi-0.2.7 → everestapi-0.2.9}/src/everestapi.egg-info/entry_points.txt +0 -0
  22. {everestapi-0.2.7 → everestapi-0.2.9}/src/everestapi.egg-info/requires.txt +0 -0
  23. {everestapi-0.2.7 → everestapi-0.2.9}/src/everestapi.egg-info/top_level.txt +0 -0
  24. {everestapi-0.2.7 → everestapi-0.2.9}/tests/test_cli.py +0 -0
  25. {everestapi-0.2.7 → everestapi-0.2.9}/tests/test_client.py +0 -0
  26. {everestapi-0.2.7 → everestapi-0.2.9}/tests/test_diagnostics.py +0 -0
  27. {everestapi-0.2.7 → everestapi-0.2.9}/tests/test_eve953_mcp_progress.py +0 -0
  28. {everestapi-0.2.7 → everestapi-0.2.9}/tests/test_eve957_mcp_annotations_resources.py +0 -0
  29. {everestapi-0.2.7 → everestapi-0.2.9}/tests/test_eve959_toolsets.py +0 -0
  30. {everestapi-0.2.7 → everestapi-0.2.9}/tests/test_json_or_raise.py +0 -0
  31. {everestapi-0.2.7 → everestapi-0.2.9}/tests/test_prediction_range.py +0 -0
  32. {everestapi-0.2.7 → everestapi-0.2.9}/tests/test_scoring.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: everestapi
3
- Version: 0.2.7
3
+ Version: 0.2.9
4
4
  Summary: Python SDK for the Everesteer prediction tournament platform
5
5
  Author-email: Everesteer <support@everesteer.ai>
6
6
  License-Expression: MIT
@@ -116,16 +116,23 @@ everestapi submit --model my-model --file predictions.parquet
116
116
 
117
117
  ### Data & diagnostics
118
118
 
119
+ The hackathon is a display-only diagnostics event. **Tune and self-score offline
120
+ on the labeled validation set** (features + `target_*` columns), then **predict on
121
+ the blind `eiq_live_2026` set and submit** — the leaderboard ranks your
122
+ out-of-sample 2026 CORR on `target_everest_20`. In-sample fit is not rewarded. The
123
+ labeled `eiq_live_2026` answers are held out server-side and are never downloadable.
124
+
119
125
  ```python
120
- api.download_dataset(universe="futures", split="train")
121
- api.download_benchmark(universe="futures", split="validation")
122
- api.get_dataset_info(universe="futures")
123
- api.get_diagnostics(model_id="my-model")
126
+ # Labeled practice set — tune + self-score offline with everestapi.scoring:
127
+ api.download_dataset(universe="futures", split="validation")
124
128
 
125
- # Validation diagnostics: eiq_validation_example_preds.parquet is the upload template
126
- # (same `id` set, your own `prediction` column).
127
- api.download_dataset(universe="futures", split="validation_example_preds")
129
+ # Blind scored set (columns: exped, exped_date, instrument, id — no targets).
130
+ # Predict on it, then submit; it is also the upload id template.
131
+ api.download_dataset(universe="futures", split="live")
128
132
  api.submit_validation_diagnostics(model_id="my-model", predictions=df)
133
+
134
+ api.get_dataset_info(universe="futures")
135
+ api.get_diagnostics(model_id="my-model")
129
136
  ```
130
137
 
131
138
  ### Plotting (optional `viz` extra)
@@ -157,11 +164,18 @@ df = pd.DataFrame(lb["entries"])
157
164
  ### Serverless compute
158
165
 
159
166
  ```python
160
- # Tier 1 — quick-train with built-in templates
161
- job = api.quick_train(model="lightgbm", features="small", target="target_everest_20")
162
-
163
- # Tier 2 — custom script on GPU
164
- job = api.custom_train(script_path="train.py", gpu="A100", max_hours=2.0)
167
+ # Built-in preset (lightgbm/xgboost/ridge/mlp/random_forest) — no data upload,
168
+ # the platform trains against the same obfuscated dataset you download.
169
+ job = api.train(model="lightgbm", features="small", target="target_everest_20")
170
+
171
+ # model="custom" — your own model factory, run server-side in an isolated,
172
+ # network-denied sandbox (no filesystem access, never sees held-out targets)
173
+ job = api.train(
174
+ model="custom",
175
+ custom_model_fn="def build_model(params):\n from sklearn.linear_model import Ridge\n return Ridge(**params)",
176
+ gpu="A100",
177
+ max_hours=2.0,
178
+ )
165
179
 
166
180
  # Wait and download
167
181
  result = api.wait_for_job(job["job_id"])
@@ -77,16 +77,23 @@ everestapi submit --model my-model --file predictions.parquet
77
77
 
78
78
  ### Data & diagnostics
79
79
 
80
+ The hackathon is a display-only diagnostics event. **Tune and self-score offline
81
+ on the labeled validation set** (features + `target_*` columns), then **predict on
82
+ the blind `eiq_live_2026` set and submit** — the leaderboard ranks your
83
+ out-of-sample 2026 CORR on `target_everest_20`. In-sample fit is not rewarded. The
84
+ labeled `eiq_live_2026` answers are held out server-side and are never downloadable.
85
+
80
86
  ```python
81
- api.download_dataset(universe="futures", split="train")
82
- api.download_benchmark(universe="futures", split="validation")
83
- api.get_dataset_info(universe="futures")
84
- api.get_diagnostics(model_id="my-model")
87
+ # Labeled practice set — tune + self-score offline with everestapi.scoring:
88
+ api.download_dataset(universe="futures", split="validation")
85
89
 
86
- # Validation diagnostics: eiq_validation_example_preds.parquet is the upload template
87
- # (same `id` set, your own `prediction` column).
88
- api.download_dataset(universe="futures", split="validation_example_preds")
90
+ # Blind scored set (columns: exped, exped_date, instrument, id — no targets).
91
+ # Predict on it, then submit; it is also the upload id template.
92
+ api.download_dataset(universe="futures", split="live")
89
93
  api.submit_validation_diagnostics(model_id="my-model", predictions=df)
94
+
95
+ api.get_dataset_info(universe="futures")
96
+ api.get_diagnostics(model_id="my-model")
90
97
  ```
91
98
 
92
99
  ### Plotting (optional `viz` extra)
@@ -118,11 +125,18 @@ df = pd.DataFrame(lb["entries"])
118
125
  ### Serverless compute
119
126
 
120
127
  ```python
121
- # Tier 1 — quick-train with built-in templates
122
- job = api.quick_train(model="lightgbm", features="small", target="target_everest_20")
123
-
124
- # Tier 2 — custom script on GPU
125
- job = api.custom_train(script_path="train.py", gpu="A100", max_hours=2.0)
128
+ # Built-in preset (lightgbm/xgboost/ridge/mlp/random_forest) — no data upload,
129
+ # the platform trains against the same obfuscated dataset you download.
130
+ job = api.train(model="lightgbm", features="small", target="target_everest_20")
131
+
132
+ # model="custom" — your own model factory, run server-side in an isolated,
133
+ # network-denied sandbox (no filesystem access, never sees held-out targets)
134
+ job = api.train(
135
+ model="custom",
136
+ custom_model_fn="def build_model(params):\n from sklearn.linear_model import Ridge\n return Ridge(**params)",
137
+ gpu="A100",
138
+ max_hours=2.0,
139
+ )
126
140
 
127
141
  # Wait and download
128
142
  result = api.wait_for_job(job["job_id"])
@@ -1,6 +1,6 @@
1
1
  """EverestAPI — Python SDK for the Everesteer prediction tournament platform."""
2
2
 
3
- __version__ = "0.2.7"
3
+ __version__ = "0.2.9"
4
4
 
5
5
  from everestapi.client import EverestAPI, EverestError
6
6
  from everestapi.types import (
@@ -26,6 +26,7 @@ Usage::
26
26
  from __future__ import annotations
27
27
 
28
28
  import os
29
+ import warnings
29
30
  from pathlib import PurePath
30
31
  from typing import Any
31
32
 
@@ -366,10 +367,14 @@ class EverestAPI:
366
367
  poll_interval: float = 2.0,
367
368
  timeout: float = 600.0,
368
369
  ) -> dict:
369
- """POST /api/v1/diagnostics/upload (multipart) — score validation predictions.
370
+ """POST /api/v1/diagnostics/upload (multipart) — score out-of-sample predictions.
370
371
 
371
372
  ``predictions`` is a pandas DataFrame (``id`` + ``prediction`` columns) or a
372
- path to a ``.parquet`` / ``.csv`` file. Returns the 202 accept dict; with
373
+ path to a ``.parquet`` / ``.csv`` file, generated on the blind eiq_live_2026
374
+ set (``download_dataset(split="live")``). The platform scores it server-side
375
+ against the held-out labeled eiq_live_2026 answers — those answers are never
376
+ downloadable. Tune and self-score offline on the labeled validation set first
377
+ (see :mod:`everestapi.scoring`). Returns the 202 accept dict; with
373
378
  ``wait=True`` polls ``runs/{upload_id}`` until ``done`` (returns the run) and
374
379
  raises :class:`EverestError` on ``failed`` or timeout. Display-only; results
375
380
  also surface in the website Validation Diagnostics rail.
@@ -590,11 +595,18 @@ class EverestAPI:
590
595
  404 the current version is resolved and the download retried once.
591
596
 
592
597
  Splits: ``train`` / ``validation`` (features + targets), ``live``
593
- (current round features only).
598
+ (features only, no targets). In hackathon mode, ``validation`` is the
599
+ LABELED practice set — features + all ``target_*`` columns — that you tune
600
+ and self-score on offline (see :mod:`everestapi.scoring`), and ``live`` is
601
+ the BLIND eiq_live_2026 out-of-sample set (columns exactly ``exped``,
602
+ ``exped_date``, ``instrument``, ``id`` — no targets) that you predict on and
603
+ submit. The labeled eiq_live_2026 answers are held out server-side and never
604
+ downloadable.
594
605
 
595
606
  Futures is served by the futures endpoint (``version`` is a no-op):
596
- it returns the real bregen tree to full-scope keys and the
597
- target-stripped hackathon tree to hackathon-scoped keys (EVE-931).
607
+ it returns the real bregen tree to full-scope keys and the hackathon tree
608
+ (labeled ``validation`` practice set + blind ``live`` eiq_live_2026 scored
609
+ set) to hackathon-scoped keys.
598
610
  """
599
611
  if output_path is None:
600
612
  output_path = f"{universe}_{split}.parquet"
@@ -775,50 +787,125 @@ class EverestAPI:
775
787
 
776
788
  # -- compute ----------------------------------------------------------
777
789
 
778
- def quick_train(
790
+ def train(
779
791
  self,
780
792
  model: str = "lightgbm",
781
- features: str = "small",
782
- target: str = "target_everest_20",
793
+ *,
794
+ custom_model_fn: str | None = None,
795
+ custom_feature_fn: str | None = None,
796
+ features: str | list[str] = "small",
797
+ target: str | list[str] = "target_everest_20",
783
798
  universe: str = "futures",
784
799
  params: dict | None = None,
800
+ cv: dict | None = None,
801
+ train_filter: dict | None = None,
802
+ val_filter: dict | None = None,
803
+ transforms: dict | None = None,
804
+ gpu: str = "T4",
805
+ max_hours: float = 1.0,
785
806
  ) -> dict:
786
- """Submit Tier 1 serverless training job."""
807
+ """Submit a unified training job (EVE-1013). Non-blocking.
808
+
809
+ train only — returns downloadable artifacts + metrics; does NOT
810
+ submit/host (use submit_futures_predictions / upload_model). Supply
811
+ a built-in ``model`` preset (lightgbm/xgboost/ridge/mlp/random_forest)
812
+ or ``model="custom"`` with a ``custom_model_fn`` factory; optionally
813
+ add ``custom_feature_fn`` for server-side feature engineering on the
814
+ resident data (no upload). The platform always runs exped-purged
815
+ cross-validation and computes canonical CORR20v2/AIMC/FNC/EAC — note
816
+ the pre-round AIMC is an ESTIMATE vs a static consensus proxy, not
817
+ the live stake-weighted crowd score used at round close.
818
+
819
+ Futures payout uses target_everest_20 (20-day forward return).
820
+ Final score = 0.75 * CORR + 2.25 * AIMC.
821
+
822
+ ``train_filter``/``val_filter`` (RowFilter) and ``transforms`` (fit-only
823
+ target transform + sample_weight override) are honored server-side,
824
+ diagnostic-only — never affects payout; see API_CONTRACT.md for the
825
+ full shape and the ``data_spec``/``canonical``/``slice_leak_note``
826
+ fields returned when a slice is applied.
827
+ """
787
828
  body: dict = {
788
829
  "model": model,
789
830
  "features": features,
790
831
  "target": target,
791
832
  "universe": universe,
833
+ "gpu": gpu,
834
+ "max_hours": max_hours,
792
835
  }
793
- if params:
836
+ if custom_model_fn is not None:
837
+ body["custom_model_fn"] = custom_model_fn
838
+ if custom_feature_fn is not None:
839
+ body["custom_feature_fn"] = custom_feature_fn
840
+ if params is not None:
794
841
  body["params"] = params
795
- return self._request("POST", "/api/v1/compute/quick-train", json=body)
842
+ if cv is not None:
843
+ body["cv"] = cv
844
+ if train_filter is not None:
845
+ body["train_filter"] = train_filter
846
+ if val_filter is not None:
847
+ body["val_filter"] = val_filter
848
+ if transforms is not None:
849
+ body["transforms"] = transforms
850
+ return self._request("POST", "/api/v1/compute/train", json=body)
796
851
 
797
- def custom_train(
852
+ def quick_train(
798
853
  self,
799
- script: str | None = None,
800
- script_path: str | None = None,
801
- files: dict[str, str] | None = None,
802
- gpu: str = "T4",
803
- max_hours: float = 1.0,
804
- requirements: list[str] | None = None,
854
+ model: str = "lightgbm",
855
+ features: str = "small",
856
+ target: str = "target_everest_20",
857
+ universe: str = "futures",
858
+ params: dict | None = None,
805
859
  ) -> dict:
806
- """Submit Tier 2 custom training job."""
807
- if script_path and not script:
808
- with open(script_path) as f:
809
- script = f.read()
810
- if files and not script:
811
- parts = [f"# --- FILE: {name} ---\n{content}" for name, content in files.items()]
812
- script = "\n\n".join(parts)
813
- body: dict = {"script": script, "gpu": gpu, "max_hours": max_hours}
814
- if requirements:
815
- body["requirements"] = requirements
816
- return self._request("POST", "/api/v1/compute/custom-train", json=body)
860
+ """Deprecated since 0.2.9 — forwards to :meth:`train`.
861
+
862
+ The Tier 1/Tier 2 split (``quick_train``/``custom_train``) was
863
+ replaced server-side by a single unified ``train`` tool (EVE-1013).
864
+ This shim keeps the old call signature working but always submits
865
+ through the new ``/api/v1/compute/train`` endpoint.
866
+ """
867
+ warnings.warn(
868
+ "quick_train() is deprecated; use train() instead (EVE-1013).",
869
+ DeprecationWarning,
870
+ stacklevel=2,
871
+ )
872
+ return self.train(
873
+ model=model, features=features, target=target, universe=universe, params=params
874
+ )
875
+
876
+ def custom_train(self, *args: Any, **kwargs: Any) -> dict:
877
+ """Removed since 0.2.9 — the Tier-2 script-based training path no longer exists.
878
+
879
+ Use :meth:`train` with ``model="custom"`` and a ``custom_model_fn``
880
+ source string instead; it now runs inside a sandboxed, network-denied
881
+ environment rather than a dedicated GPU pod running an arbitrary script.
882
+ """
883
+ warnings.warn(
884
+ "custom_train() is removed; use train(model='custom', custom_model_fn=...).",
885
+ DeprecationWarning,
886
+ stacklevel=2,
887
+ )
888
+ raise EverestError(
889
+ 410,
890
+ "custom_train (Tier-2 script) was removed. Use "
891
+ "train(model='custom', custom_model_fn='def build_model(params): ...').",
892
+ )
817
893
 
818
894
  def get_job_status(self, job_id: str) -> dict:
819
895
  """Poll compute job status."""
820
896
  return self._request("GET", f"/api/v1/compute/jobs/{job_id}")
821
897
 
898
+ def list_compute_jobs(self, limit: int = 50, status: str | None = None) -> dict:
899
+ """List your compute jobs, newest first (rows exclude config/output payloads)."""
900
+ params: dict = {"limit": limit}
901
+ if status is not None:
902
+ params["status"] = status
903
+ return self._request("GET", "/api/v1/compute/jobs", params=params)
904
+
905
+ def get_job_log(self, job_id: str) -> dict:
906
+ """Lifecycle event trail for one compute job — debug a failed/stuck job."""
907
+ return self._request("GET", f"/api/v1/compute/jobs/{job_id}/logs")
908
+
822
909
  def get_model_download_url(self, job_id: str) -> dict:
823
910
  """Get a presigned download URL for a trained model."""
824
911
  return self._request("GET", f"/api/v1/compute/jobs/{job_id}/model")
@@ -864,7 +951,7 @@ class EverestAPI:
864
951
  return self._request("GET", "/api/v1/compute/credits")
865
952
 
866
953
  def list_model_templates(self) -> dict:
867
- """List available quick-train model templates."""
954
+ """List available train model templates."""
868
955
  return self._request("GET", "/api/v1/compute/models")
869
956
 
870
957
  # -- registration (no auth) ------------------------------------------
@@ -933,7 +1020,8 @@ class EverestAPI:
933
1020
 
934
1021
  With a hackathon-scoped key there is no live round — the response is a
935
1022
  diagnostics-mode payload (mode='diagnostics_hackathon') directing you to
936
- download_dataset + submit_validation_diagnostics.
1023
+ tune offline on the labeled validation set, then predict on the blind
1024
+ eiq_live_2026 set and submit_validation_diagnostics.
937
1025
  """
938
1026
  return self._request(
939
1027
  "GET",
@@ -951,7 +1039,10 @@ class EverestAPI:
951
1039
  """GET /api/v1/get_started — mode-aware orientation.
952
1040
 
953
1041
  Returns what to do next given this key's scope: the display-only
954
- diagnostics-hackathon loop, or the live futures tournament flow.
1042
+ diagnostics-hackathon loop (tune and self-score offline on the labeled
1043
+ validation set, then predict on the blind eiq_live_2026 set and submit —
1044
+ ranked on out-of-sample 2026 CORR on target_everest_20), or the live
1045
+ futures tournament flow.
955
1046
  """
956
1047
  return self._request("GET", "/api/v1/get_started")
957
1048
 
@@ -238,7 +238,7 @@ TOOLS = [
238
238
  },
239
239
  {
240
240
  "name": "download_dataset",
241
- "description": "Download a dataset split (train/validation/live) as a parquet file. Returns the local file path.",
241
+ "description": "Download a dataset split (train/validation/live) as a parquet file. Returns the local file path. In hackathon mode: 'validation' is the LABELED practice set (features + all target_* columns) for offline tuning and self-scoring; 'live' is the BLIND eiq_live_2026 out-of-sample set (columns exactly exped, exped_date, instrument, id — NO targets) that you predict on and submit. The labeled 2026 answers are held out server-side and never downloadable.",
242
242
  "inputSchema": {
243
243
  "type": "object",
244
244
  "properties": {
@@ -270,42 +270,64 @@ TOOLS = [
270
270
  },
271
271
  },
272
272
  {
273
- "name": "quick_train",
274
- "description": "Train a model on Everesteer data using serverless GPU. Returns job_id. Poll with get_job_status(). Models: lightgbm, xgboost, ridge, mlp. Costs $0.005-0.05. Futures payout target: target_everest_20 (20-day forward return). Final score = 0.75*CORR + 2.25*AIMC.",
273
+ "name": "train",
274
+ "description": "Train a model on Everesteer data (no data upload — the platform uses the same obfuscated dataset you download). The PLATFORM loads data, runs exped-purged/embargoed cross-validation, computes canonical tournament metrics (0.75*CORR + 2.25*AIMC; AIMC before round close is a PRE-SUBMISSION ESTIMATE vs a static consensus proxy — real AIMC is scored vs the live stake-weighted crowd at round close), pickles the final model, and returns DOWNLOADABLE artifacts (model .pkl + validation/live prediction files) plus the metrics. Supply a built-in `model` preset (lightgbm/xgboost/ridge/mlp/random_forest), or `model='custom'` with a `custom_model_fn` (and optional `custom_feature_fn`) to run your own code — it executes inside an isolated, network-denied sandbox with no filesystem access and never sees held-out targets; a script rejected by the static safety scanner returns 400. `train` does NOT submit predictions or host a model on your behalf — use submit_futures_predictions/submit_predictions_file to submit and upload_model to host the .pkl. Poll with get_job_status() — the completed result carries the metrics + a predictions download URL; fetch the model file separately via get_model_download_url().",
275
275
  "inputSchema": {
276
276
  "type": "object",
277
277
  "properties": {
278
- "model": {"type": "string", "enum": ["lightgbm", "xgboost", "ridge", "mlp"]},
279
- "features": {
278
+ "model": {
279
+ "type": "string",
280
+ "enum": ["lightgbm", "xgboost", "ridge", "mlp", "random_forest", "custom"],
281
+ "default": "lightgbm",
282
+ },
283
+ "custom_model_fn": {
284
+ "type": "string",
285
+ "description": "Python source defining build_model(params)->estimator, required when model='custom'. Runs in an isolated, network-denied sandbox with no filesystem access and never sees held-out targets. Rejected by the static safety scanner (returns 400) if it references disallowed modules/builtins.",
286
+ },
287
+ "custom_feature_fn": {
280
288
  "type": "string",
281
- "enum": ["small", "medium", "all"],
289
+ "description": "OPTIONAL Python source defining build_feature_transformer(params)->sklearn transformer for server-side feature engineering on the resident features. Usable with any model (preset or custom), in the same isolated sandbox as custom_model_fn. Same static-scanner gate (returns 400 on rejection).",
290
+ },
291
+ "params": {"type": "object", "description": "Model hyperparameters (optional)"},
292
+ "features": {
293
+ "oneOf": [
294
+ {"type": "string", "enum": ["small", "medium", "all"]},
295
+ {"type": "array", "items": {"type": "string"}},
296
+ ],
282
297
  "default": "small",
298
+ "description": "Preset (small/medium/all) or an explicit list of feature_* names.",
283
299
  },
284
300
  "target": {
285
- "type": "string",
301
+ "oneOf": [
302
+ {"type": "string"},
303
+ {"type": "array", "items": {"type": "string"}},
304
+ ],
286
305
  "default": "target_everest_20",
287
- "description": "Target column. Futures payout uses target_everest_20 (20-day forward return). Multiple target types available. See get_dataset_schema for full list.",
306
+ "description": "Target column. Futures payout uses target_everest_20 (20-day forward return).",
288
307
  },
289
308
  "universe": {"type": "string", "default": "futures"},
290
- "params": {"type": "object", "description": "Model hyperparameters (optional)"},
309
+ "cv": {
310
+ "type": "object",
311
+ "description": "Cross-validation spec: {scheme, folds, embargo}. scheme is exped_purged (default), chain_group, or combined; embargo defaults to the label horizon when null.",
312
+ },
313
+ "train_filter": {
314
+ "type": "object",
315
+ "description": "Row filter (exped/climb_difficulty/sample/feature_filter) applied to the training split server-side, no re-upload — diagnostic-only, never affects payout. Full RowFilter shape in API_CONTRACT.md.",
316
+ },
317
+ "val_filter": {
318
+ "type": "object",
319
+ "description": "Same RowFilter shape as train_filter, applied to the validation split (layered under the platform's own validation clamp, if any) — diagnostic-only, never affects payout.",
320
+ },
321
+ "transforms": {
322
+ "type": "object",
323
+ "description": "Fit-only target transform (residualize_to_benchmark/subtract_benchmark_zscore) and/or sample_weight override — scoring always uses the untouched canonical target. Full shape in API_CONTRACT.md.",
324
+ },
325
+ "gpu": {"type": "string", "enum": ["T4", "A10G", "A100"], "default": "T4"},
326
+ "max_hours": {"type": "number", "minimum": 0.1, "maximum": 4.0, "default": 1.0},
291
327
  },
292
328
  "required": ["model"],
293
329
  },
294
330
  },
295
- {
296
- "name": "custom_train",
297
- "description": "Run a custom Python script on a GPU pod. Pod has EverestAPI SDK + PyTorch + ML libs pre-installed. Costs $0.50-1.79/hr.",
298
- "inputSchema": {
299
- "type": "object",
300
- "properties": {
301
- "script": {"type": "string", "description": "Python script content"},
302
- "gpu": {"type": "string", "enum": ["T4", "A40", "A100"], "default": "T4"},
303
- "max_hours": {"type": "number", "default": 1},
304
- "requirements": {"type": "array", "items": {"type": "string"}},
305
- },
306
- "required": ["script"],
307
- },
308
- },
309
331
  {
310
332
  "name": "get_job_status",
311
333
  "description": "Check status of a compute job. Returns status, cost, output files when complete.",
@@ -314,7 +336,7 @@ TOOLS = [
314
336
  "properties": {
315
337
  "job_id": {
316
338
  "type": "string",
317
- "description": "Job ID from quick_train or custom_train",
339
+ "description": "Job ID from a train job",
318
340
  },
319
341
  },
320
342
  "required": ["job_id"],
@@ -328,7 +350,7 @@ TOOLS = [
328
350
  "properties": {
329
351
  "job_id": {
330
352
  "type": "string",
331
- "description": "Job ID from quick_train or custom_train",
353
+ "description": "Job ID from a train job",
332
354
  },
333
355
  },
334
356
  "required": ["job_id"],
@@ -365,7 +387,7 @@ TOOLS = [
365
387
  },
366
388
  {
367
389
  "name": "get_current_round",
368
- "description": "Get the current active round for a tournament (defaults to 'futures'; equities is unlaunched). With a hackathon key there is no live round — the response is a diagnostics-mode payload directing you to download_dataset + submit_validation_diagnostics.",
390
+ "description": "Get the current active round for a tournament (defaults to 'futures'; equities is unlaunched). With a hackathon key there is no live round — the response is a diagnostics-mode payload directing you to tune offline on the labeled validation set, then download the blind eiq_live_2026 set, predict, and submit_validation_diagnostics.",
369
391
  "inputSchema": {
370
392
  "type": "object",
371
393
  "properties": {
@@ -380,7 +402,7 @@ TOOLS = [
380
402
  },
381
403
  {
382
404
  "name": "get_started",
383
- "description": "Mode-aware orientation: returns what to do next given your API key's scope. A hackathon key gets the display-only diagnostics loop (objective: maximize out-of-sample CORR on target_everest_20); a full key gets the live futures tournament flow. Start here.",
405
+ "description": "Mode-aware orientation: returns what to do next given your API key's scope. A hackathon key gets the display-only diagnostics loop (tune and self-score offline on the labeled validation set, then predict on the blind eiq_live_2026 set and submit; objective: maximize out-of-sample 2026 CORR on target_everest_20, in-sample fit is not rewarded); a full key gets the live futures tournament flow. Start here.",
384
406
  "inputSchema": {
385
407
  "type": "object",
386
408
  "properties": {},
@@ -546,13 +568,15 @@ TOOLS = [
546
568
  {
547
569
  "name": "submit_validation_diagnostics",
548
570
  "description": (
549
- "Upload a NEW validation predictions file (parquet/CSV with id+prediction columns, "
550
- "≤100 MB) and score it against eiq_validation.parquet. The required schema mirrors "
551
- "eiq_validation_example_preds.parquet — fetch it with "
552
- "download_dataset(split='validation_example_preds') and use it as the upload template "
553
- "(same id set, your own prediction column). Use this only when you have fresh "
554
- "predictions to score — to read the model's EXISTING latest result without re-scoring "
555
- "(instant, no wait), call run_validation_diagnostics(model_id) instead. "
571
+ "Upload a NEW predictions file (parquet/CSV with id+prediction columns, "
572
+ "≤100 MB) generated on the blind eiq_live_2026 out-of-sample set; the platform "
573
+ "scores it server-side against the held-out labeled eiq_live_2026 answers (never "
574
+ "downloadable). The required id set is the blind eiq_live_2026 set — fetch it with "
575
+ "download_dataset(split='live') and use it as the upload template "
576
+ "(its id column, your own prediction column). Tune and self-score offline on the "
577
+ "labeled validation set first (see the scoring helper). Use this only when you have "
578
+ "fresh predictions to score — to read the model's EXISTING latest result without "
579
+ "re-scoring (instant, no wait), call run_validation_diagnostics(model_id) instead. "
556
580
  "The platform scores a 9-metric panel (CORR20, BMC, FNC, Sharpe, std dev, "
557
581
  "feature-exposure, Max Drawdown, autocorrelation, example-preds-corr) asynchronously "
558
582
  "and surfaces the run in the website Validation Diagnostics rail. By default "
@@ -594,9 +618,10 @@ TOOLS = [
594
618
  {
595
619
  "name": "get_diagnostics_leaderboard",
596
620
  "description": (
597
- "Get the validation diagnostics leaderboard — global ranking of uploaded validation "
598
- "runs by CORR20, BMC, FNC, and Sharpe. view='agents' (default) ranks participant model "
599
- "runs and marks your own entries; view='benchmarks' ranks the official platform benchmarks."
621
+ "Get the validation diagnostics leaderboard — global ranking of agents by their "
622
+ "out-of-sample eiq_live_2026 CORR on target_everest_20 (in-sample fit is not rewarded). "
623
+ "view='agents' (default) ranks participant model runs and marks your own entries; "
624
+ "view='benchmarks' ranks the official platform benchmarks."
600
625
  ),
601
626
  "inputSchema": {
602
627
  "type": "object",
@@ -894,6 +919,19 @@ TOOLSETS: dict[str, set[str]] = {
894
919
  "get_scores",
895
920
  "get_leaderboard",
896
921
  "get_capabilities",
922
+ # EVE-967: the diagnostics upload + leaderboard are the entire hackathon
923
+ # flow get_started steers to; advertise them by default (core) so a
924
+ # hackathon key discovers them without setting EIQ_MCP_TOOLSETS. The rest
925
+ # of the diagnostics lifecycle stays in "diagnostics".
926
+ "submit_validation_diagnostics",
927
+ "get_diagnostics_leaderboard",
928
+ # EVE-970: hosted training is a first-class flow now that every account
929
+ # carries a compute grant — advertise train + get_compute_credits by
930
+ # default so agents discover the hosted-training entry point without
931
+ # setting EIQ_MCP_TOOLSETS. get_job_status stays in "compute" (hidden
932
+ # tools remain callable by name). Mirrors the platform MCP server.
933
+ "train",
934
+ "get_compute_credits",
897
935
  },
898
936
  "data": {
899
937
  "get_universe",
@@ -911,15 +949,12 @@ TOOLSETS: dict[str, set[str]] = {
911
949
  },
912
950
  "submit": {"submit_predictions", "upload_model", "get_upload_status"},
913
951
  "diagnostics": {
952
+ # submit_validation_diagnostics + get_diagnostics_leaderboard promoted to
953
+ # "core" (EVE-967) — every name still lives in exactly one group.
914
954
  "run_validation_diagnostics",
915
- "submit_validation_diagnostics",
916
- "get_diagnostics_leaderboard",
917
955
  },
918
956
  "compute": {
919
- "quick_train",
920
- "custom_train",
921
957
  "get_job_status",
922
- "get_compute_credits",
923
958
  },
924
959
  "staking": {
925
960
  "stake_on_model",
@@ -1081,21 +1116,8 @@ def _dispatch(name: str, arguments: dict) -> str:
1081
1116
  output_path=arguments.get("output_path"),
1082
1117
  )
1083
1118
  result = {"file_path": path, "status": "downloaded"}
1084
- elif name == "quick_train":
1085
- result = client.quick_train(
1086
- model=arguments["model"],
1087
- features=arguments.get("features", "small"),
1088
- target=arguments.get("target", "target_everest_20"),
1089
- universe=arguments.get("universe", "futures"),
1090
- params=arguments.get("params"),
1091
- )
1092
- elif name == "custom_train":
1093
- result = client.custom_train(
1094
- script=arguments.get("script"),
1095
- gpu=arguments.get("gpu", "T4"),
1096
- max_hours=arguments.get("max_hours", 1),
1097
- requirements=arguments.get("requirements"),
1098
- )
1119
+ elif name == "train":
1120
+ result = client.train(**arguments)
1099
1121
  elif name == "get_job_status":
1100
1122
  result = client.get_job_status(job_id=arguments["job_id"])
1101
1123
  elif name == "get_model_download_url":
@@ -1275,9 +1297,14 @@ def _error_payload(exc: Exception) -> str:
1275
1297
 
1276
1298
  if isinstance(exc, EverestError):
1277
1299
  code = exc.status_code
1300
+ # FastAPI wraps errors as {"detail": {"code": ..., "message": ...}} —
1301
+ # pull the machine code so 403 can distinguish a scope restriction
1302
+ # (not available in diagnostics mode) from a genuine auth failure.
1303
+ inner = exc.detail.get("detail") if isinstance(exc.detail, dict) else None
1304
+ err_code = inner.get("code") if isinstance(inner, dict) else None
1305
+
1278
1306
  hints = {
1279
1307
  401: "Auth failed — set EIQ_API_KEY to your Everesteer API key.",
1280
- 403: "Auth failed — set EIQ_API_KEY to your Everesteer API key.",
1281
1308
  404: "Not found — check the model_id / upload_id exists.",
1282
1309
  409: "A diagnostics run is already in flight for this model — read it with "
1283
1310
  "run_validation_diagnostics(model_id) or wait for it to finish.",
@@ -1285,10 +1312,19 @@ def _error_payload(exc: Exception) -> str:
1285
1312
  504: "Still computing — scoring can take 15-20 min; poll again shortly.",
1286
1313
  }
1287
1314
  hint = hints.get(code)
1315
+ if code == 403:
1316
+ if err_code in ("scope_mismatch", "scope_restricted"):
1317
+ hint = (
1318
+ "Not available in diagnostics (hackathon) mode — this is a "
1319
+ "live-tournament / full-scope action. Use the diagnostics flow: "
1320
+ "submit_validation_diagnostics + get_diagnostics_leaderboard."
1321
+ )
1322
+ else:
1323
+ hint = "Auth failed — set EIQ_API_KEY to your Everesteer API key."
1288
1324
  if hint is None and 400 <= code < 500:
1289
- hint = "Bad request — file needs id+prediction columns and an existing model_id."
1325
+ hint = "Bad request — check the arguments match the tool schema."
1290
1326
  elif hint is None and code >= 500:
1291
- hint = "Server error — retry; if it persists the run may have failed (e.g. predictions didn't overlap the validation universe)."
1327
+ hint = "Server error — retry; if it persists the request may have failed."
1292
1328
  return json.dumps({"error": exc.detail, "status": code, "hint": hint})
1293
1329
  return json.dumps({"error": str(exc)})
1294
1330
 
@@ -1402,9 +1438,13 @@ _MCP_INSTRUCTIONS = (
1402
1438
  "Everesteer tournament MCP server. Call get_started first — it is mode-aware and tells you "
1403
1439
  "what to do next based on your API key's scope.\n\n"
1404
1440
  "Hackathon mode (a hackathon-scoped key): there is no live tournament round and no "
1405
- "staking or payout — this is a display-only diagnostics event. Your objective is to "
1406
- "maximize OUT-OF-SAMPLE CORR on target_everest_20; in-sample fit is not rewarded. Flow: "
1407
- "download_dataset -> score offline -> submit_validation_diagnostics -> "
1441
+ "staking or payout — this is a display-only diagnostics event. Tune and self-score "
1442
+ "offline on the LABELED validation set (features + target_* columns, via the scoring "
1443
+ "helper), then predict on the BLIND eiq_live_2026 set (columns exped, exped_date, "
1444
+ "instrument, id — no targets) and submit; the leaderboard ranks your OUT-OF-SAMPLE 2026 "
1445
+ "CORR on target_everest_20. In-sample fit is not rewarded. Flow: "
1446
+ "download_dataset(split='validation') -> tune + self-score offline -> "
1447
+ "download_dataset(split='live') -> predict -> submit_validation_diagnostics -> "
1408
1448
  "get_diagnostics_leaderboard. Round/universe/feature/benchmark reads return empty "
1409
1449
  "diagnostics-mode payloads, not live data.\n\n"
1410
1450
  "Tournament mode (a full-scope key): submit daily futures predictions; payout = "
@@ -1436,8 +1476,7 @@ _NEXT_ACTIONS: dict[str, list[str]] = {
1436
1476
  "download_dataset": ["create_model", "submit_futures_predictions"],
1437
1477
  "download_benchmark": ["submit_futures_predictions"],
1438
1478
  "get_benchmarks": ["download_benchmark"],
1439
- "quick_train": ["get_job_status"],
1440
- "custom_train": ["get_job_status"],
1479
+ "train": ["get_job_status"],
1441
1480
  "get_job_status": ["get_model_download_url"], # conditional
1442
1481
  "get_model_download_url": ["upload_model", "create_model"],
1443
1482
  "create_model": ["submit_futures_predictions", "upload_model"],
@@ -1469,7 +1508,7 @@ _NEXT_ACTIONS: dict[str, list[str]] = {
1469
1508
  "get_deposit_address": ["get_forwarder_balance", "relay_stake"],
1470
1509
  "get_forwarder_balance": ["relay_stake"],
1471
1510
  "withdraw_usdc": ["get_forwarder_balance"],
1472
- "get_compute_credits": ["quick_train"],
1511
+ "get_compute_credits": ["train"],
1473
1512
  "get_notifications": [],
1474
1513
  "get_badges": [],
1475
1514
  }
@@ -1494,8 +1533,7 @@ _NARRATION: dict[str, str] = {
1494
1533
  "submit_predictions": (
1495
1534
  "Submitted predictions for model '{model_id}'. Scoring starts at round close."
1496
1535
  ),
1497
- "quick_train": "Training job queued.",
1498
- "custom_train": "Custom training job queued.",
1536
+ "train": "Training job queued — returns downloadable artifacts + metrics, not a submission.",
1499
1537
  "get_job_status": "Training job status: {status}.",
1500
1538
  "get_model_download_url": "Generated a download URL for the trained model (valid ~1h).",
1501
1539
  "upload_model": "Model artifact uploaded; validation runs asynchronously.",
@@ -1555,7 +1593,7 @@ def _next_actions_for(name: str, arguments: dict, data: dict) -> list[str]:
1555
1593
  if status in ("completed", "done", "succeeded", "success"):
1556
1594
  return ["get_model_download_url", "create_model"]
1557
1595
  if status in ("failed", "error"):
1558
- return ["quick_train", "custom_train"]
1596
+ return ["train"]
1559
1597
  return ["get_job_status"] # still running — keep polling
1560
1598
  if name == "get_upload_status":
1561
1599
  if status in ("validated", "done", "success", "ready", "active"):
@@ -27,6 +27,14 @@ Quickstart::
27
27
  for e in val.exped.unique()
28
28
  ]
29
29
  print("mean CORR20:", sum(corrs) / len(corrs))
30
+
31
+ In hackathon mode the validation set ships LABELED (features + ``target_*``
32
+ columns), so this helper is the offline self-scoring path: use it to tune your
33
+ model before you submit. It scores in-sample and is for tuning only — the
34
+ official hackathon leaderboard scores your submitted predictions on the BLIND
35
+ ``eiq_live_2026`` out-of-sample set server-side (its labeled answers are held
36
+ out and never downloadable), ranking your 2026 CORR on ``target_everest_20``.
37
+ In-sample fit is not rewarded.
30
38
  """
31
39
 
32
40
  from __future__ import annotations
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: everestapi
3
- Version: 0.2.7
3
+ Version: 0.2.9
4
4
  Summary: Python SDK for the Everesteer prediction tournament platform
5
5
  Author-email: Everesteer <support@everesteer.ai>
6
6
  License-Expression: MIT
@@ -116,16 +116,23 @@ everestapi submit --model my-model --file predictions.parquet
116
116
 
117
117
  ### Data & diagnostics
118
118
 
119
+ The hackathon is a display-only diagnostics event. **Tune and self-score offline
120
+ on the labeled validation set** (features + `target_*` columns), then **predict on
121
+ the blind `eiq_live_2026` set and submit** — the leaderboard ranks your
122
+ out-of-sample 2026 CORR on `target_everest_20`. In-sample fit is not rewarded. The
123
+ labeled `eiq_live_2026` answers are held out server-side and are never downloadable.
124
+
119
125
  ```python
120
- api.download_dataset(universe="futures", split="train")
121
- api.download_benchmark(universe="futures", split="validation")
122
- api.get_dataset_info(universe="futures")
123
- api.get_diagnostics(model_id="my-model")
126
+ # Labeled practice set — tune + self-score offline with everestapi.scoring:
127
+ api.download_dataset(universe="futures", split="validation")
124
128
 
125
- # Validation diagnostics: eiq_validation_example_preds.parquet is the upload template
126
- # (same `id` set, your own `prediction` column).
127
- api.download_dataset(universe="futures", split="validation_example_preds")
129
+ # Blind scored set (columns: exped, exped_date, instrument, id — no targets).
130
+ # Predict on it, then submit; it is also the upload id template.
131
+ api.download_dataset(universe="futures", split="live")
128
132
  api.submit_validation_diagnostics(model_id="my-model", predictions=df)
133
+
134
+ api.get_dataset_info(universe="futures")
135
+ api.get_diagnostics(model_id="my-model")
129
136
  ```
130
137
 
131
138
  ### Plotting (optional `viz` extra)
@@ -157,11 +164,18 @@ df = pd.DataFrame(lb["entries"])
157
164
  ### Serverless compute
158
165
 
159
166
  ```python
160
- # Tier 1 — quick-train with built-in templates
161
- job = api.quick_train(model="lightgbm", features="small", target="target_everest_20")
162
-
163
- # Tier 2 — custom script on GPU
164
- job = api.custom_train(script_path="train.py", gpu="A100", max_hours=2.0)
167
+ # Built-in preset (lightgbm/xgboost/ridge/mlp/random_forest) — no data upload,
168
+ # the platform trains against the same obfuscated dataset you download.
169
+ job = api.train(model="lightgbm", features="small", target="target_everest_20")
170
+
171
+ # model="custom" — your own model factory, run server-side in an isolated,
172
+ # network-denied sandbox (no filesystem access, never sees held-out targets)
173
+ job = api.train(
174
+ model="custom",
175
+ custom_model_fn="def build_model(params):\n from sklearn.linear_model import Ridge\n return Ridge(**params)",
176
+ gpu="A100",
177
+ max_hours=2.0,
178
+ )
165
179
 
166
180
  # Wait and download
167
181
  result = api.wait_for_job(job["job_id"])
@@ -23,6 +23,7 @@ tests/test_diagnostics.py
23
23
  tests/test_eve953_mcp_progress.py
24
24
  tests/test_eve957_mcp_annotations_resources.py
25
25
  tests/test_eve959_toolsets.py
26
+ tests/test_eve967_mcp_discoverability.py
26
27
  tests/test_json_or_raise.py
27
28
  tests/test_mcp_and_models.py
28
29
  tests/test_prediction_range.py
@@ -0,0 +1,40 @@
1
+ """EVE-967 — the diagnostics upload + leaderboard tools are discoverable on the
2
+ default (core) toolset, and MCP error hints don't mislead:
3
+ - a 403 scope restriction points at the diagnostics flow, not "rotate your key";
4
+ - the generic 400 fallback is tool-neutral (no diagnostics-upload bleed).
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ import json
10
+
11
+ import everestapi.mcp.server as S
12
+ from everestapi.client import EverestError
13
+
14
+
15
+ def test_diagnostics_tools_advertised_by_default(monkeypatch):
16
+ monkeypatch.delenv("EIQ_MCP_TOOLSETS", raising=False)
17
+ enabled = S._enabled_tool_names()
18
+ assert "submit_validation_diagnostics" in enabled
19
+ assert "get_diagnostics_leaderboard" in enabled
20
+
21
+
22
+ def test_403_scope_mismatch_hint_points_to_diagnostics():
23
+ exc = EverestError(403, {"detail": {"code": "scope_mismatch", "message": "x"}})
24
+ payload = json.loads(S._error_payload(exc))
25
+ assert payload["status"] == 403
26
+ assert "diagnostics" in payload["hint"].lower()
27
+ assert "set EIQ_API_KEY" not in payload["hint"]
28
+
29
+
30
+ def test_403_plain_auth_still_says_set_key():
31
+ exc = EverestError(403, "forbidden") # no machine code → generic auth hint
32
+ payload = json.loads(S._error_payload(exc))
33
+ assert "EIQ_API_KEY" in payload["hint"]
34
+
35
+
36
+ def test_generic_400_hint_is_tool_neutral():
37
+ exc = EverestError(400, {"detail": "bad"})
38
+ payload = json.loads(S._error_payload(exc))
39
+ assert "id+prediction" not in payload["hint"]
40
+ assert "schema" in payload["hint"].lower()
@@ -514,6 +514,8 @@ _SDK_ONLY_SANCTIONED = {
514
514
  "get_multipliers",
515
515
  "get_current_season",
516
516
  "get_job_output",
517
+ "list_compute_jobs", # EVE-1013 — job lookup, same category as get_job_output
518
+ "get_job_log", # EVE-1013 — job lookup, same category as get_job_output
517
519
  "get_diagnostics_run",
518
520
  "get_validation_panel",
519
521
  "get_validation_diagnostics",
@@ -530,6 +532,11 @@ _SDK_ONLY_SANCTIONED = {
530
532
  "unstake",
531
533
  "confirm_stake",
532
534
  "cancel_job",
535
+ # EVE-1013 — quick_train/custom_train replaced by the unified `train` tool;
536
+ # both remain on the client as deprecated shims (quick_train forwards to
537
+ # train(), custom_train raises) but are no longer registered as MCP tools.
538
+ "quick_train",
539
+ "custom_train",
533
540
  }
534
541
 
535
542
 
File without changes
File without changes
File without changes
File without changes