everestapi 0.2.7__tar.gz → 0.2.8__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {everestapi-0.2.7/src/everestapi.egg-info → everestapi-0.2.8}/PKG-INFO +15 -8
- {everestapi-0.2.7 → everestapi-0.2.8}/README.md +14 -7
- {everestapi-0.2.7 → everestapi-0.2.8}/src/everestapi/__init__.py +1 -1
- {everestapi-0.2.7 → everestapi-0.2.8}/src/everestapi/client.py +22 -7
- {everestapi-0.2.7 → everestapi-0.2.8}/src/everestapi/mcp/server.py +48 -21
- {everestapi-0.2.7 → everestapi-0.2.8}/src/everestapi/scoring.py +8 -0
- {everestapi-0.2.7 → everestapi-0.2.8/src/everestapi.egg-info}/PKG-INFO +15 -8
- {everestapi-0.2.7 → everestapi-0.2.8}/src/everestapi.egg-info/SOURCES.txt +1 -0
- everestapi-0.2.8/tests/test_eve967_mcp_discoverability.py +40 -0
- {everestapi-0.2.7 → everestapi-0.2.8}/LICENSE +0 -0
- {everestapi-0.2.7 → everestapi-0.2.8}/pyproject.toml +0 -0
- {everestapi-0.2.7 → everestapi-0.2.8}/setup.cfg +0 -0
- {everestapi-0.2.7 → everestapi-0.2.8}/src/everestapi/__main__.py +0 -0
- {everestapi-0.2.7 → everestapi-0.2.8}/src/everestapi/cli.py +0 -0
- {everestapi-0.2.7 → everestapi-0.2.8}/src/everestapi/mcp/__init__.py +0 -0
- {everestapi-0.2.7 → everestapi-0.2.8}/src/everestapi/mcp/__main__.py +0 -0
- {everestapi-0.2.7 → everestapi-0.2.8}/src/everestapi/plots.py +0 -0
- {everestapi-0.2.7 → everestapi-0.2.8}/src/everestapi/types.py +0 -0
- {everestapi-0.2.7 → everestapi-0.2.8}/src/everestapi.egg-info/dependency_links.txt +0 -0
- {everestapi-0.2.7 → everestapi-0.2.8}/src/everestapi.egg-info/entry_points.txt +0 -0
- {everestapi-0.2.7 → everestapi-0.2.8}/src/everestapi.egg-info/requires.txt +0 -0
- {everestapi-0.2.7 → everestapi-0.2.8}/src/everestapi.egg-info/top_level.txt +0 -0
- {everestapi-0.2.7 → everestapi-0.2.8}/tests/test_cli.py +0 -0
- {everestapi-0.2.7 → everestapi-0.2.8}/tests/test_client.py +0 -0
- {everestapi-0.2.7 → everestapi-0.2.8}/tests/test_diagnostics.py +0 -0
- {everestapi-0.2.7 → everestapi-0.2.8}/tests/test_eve953_mcp_progress.py +0 -0
- {everestapi-0.2.7 → everestapi-0.2.8}/tests/test_eve957_mcp_annotations_resources.py +0 -0
- {everestapi-0.2.7 → everestapi-0.2.8}/tests/test_eve959_toolsets.py +0 -0
- {everestapi-0.2.7 → everestapi-0.2.8}/tests/test_json_or_raise.py +0 -0
- {everestapi-0.2.7 → everestapi-0.2.8}/tests/test_mcp_and_models.py +0 -0
- {everestapi-0.2.7 → everestapi-0.2.8}/tests/test_prediction_range.py +0 -0
- {everestapi-0.2.7 → everestapi-0.2.8}/tests/test_scoring.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: everestapi
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.8
|
|
4
4
|
Summary: Python SDK for the Everesteer prediction tournament platform
|
|
5
5
|
Author-email: Everesteer <support@everesteer.ai>
|
|
6
6
|
License-Expression: MIT
|
|
@@ -116,16 +116,23 @@ everestapi submit --model my-model --file predictions.parquet
|
|
|
116
116
|
|
|
117
117
|
### Data & diagnostics
|
|
118
118
|
|
|
119
|
+
The hackathon is a display-only diagnostics event. **Tune and self-score offline
|
|
120
|
+
on the labeled validation set** (features + `target_*` columns), then **predict on
|
|
121
|
+
the blind `eiq_live_2026` set and submit** — the leaderboard ranks your
|
|
122
|
+
out-of-sample 2026 CORR on `target_everest_20`. In-sample fit is not rewarded. The
|
|
123
|
+
labeled `eiq_live_2026` answers are held out server-side and are never downloadable.
|
|
124
|
+
|
|
119
125
|
```python
|
|
120
|
-
|
|
121
|
-
api.
|
|
122
|
-
api.get_dataset_info(universe="futures")
|
|
123
|
-
api.get_diagnostics(model_id="my-model")
|
|
126
|
+
# Labeled practice set — tune + self-score offline with everestapi.scoring:
|
|
127
|
+
api.download_dataset(universe="futures", split="validation")
|
|
124
128
|
|
|
125
|
-
#
|
|
126
|
-
#
|
|
127
|
-
api.download_dataset(universe="futures", split="
|
|
129
|
+
# Blind scored set (columns: exped, exped_date, instrument, id — no targets).
|
|
130
|
+
# Predict on it, then submit; it is also the upload id template.
|
|
131
|
+
api.download_dataset(universe="futures", split="live")
|
|
128
132
|
api.submit_validation_diagnostics(model_id="my-model", predictions=df)
|
|
133
|
+
|
|
134
|
+
api.get_dataset_info(universe="futures")
|
|
135
|
+
api.get_diagnostics(model_id="my-model")
|
|
129
136
|
```
|
|
130
137
|
|
|
131
138
|
### Plotting (optional `viz` extra)
|
|
@@ -77,16 +77,23 @@ everestapi submit --model my-model --file predictions.parquet
|
|
|
77
77
|
|
|
78
78
|
### Data & diagnostics
|
|
79
79
|
|
|
80
|
+
The hackathon is a display-only diagnostics event. **Tune and self-score offline
|
|
81
|
+
on the labeled validation set** (features + `target_*` columns), then **predict on
|
|
82
|
+
the blind `eiq_live_2026` set and submit** — the leaderboard ranks your
|
|
83
|
+
out-of-sample 2026 CORR on `target_everest_20`. In-sample fit is not rewarded. The
|
|
84
|
+
labeled `eiq_live_2026` answers are held out server-side and are never downloadable.
|
|
85
|
+
|
|
80
86
|
```python
|
|
81
|
-
|
|
82
|
-
api.
|
|
83
|
-
api.get_dataset_info(universe="futures")
|
|
84
|
-
api.get_diagnostics(model_id="my-model")
|
|
87
|
+
# Labeled practice set — tune + self-score offline with everestapi.scoring:
|
|
88
|
+
api.download_dataset(universe="futures", split="validation")
|
|
85
89
|
|
|
86
|
-
#
|
|
87
|
-
#
|
|
88
|
-
api.download_dataset(universe="futures", split="
|
|
90
|
+
# Blind scored set (columns: exped, exped_date, instrument, id — no targets).
|
|
91
|
+
# Predict on it, then submit; it is also the upload id template.
|
|
92
|
+
api.download_dataset(universe="futures", split="live")
|
|
89
93
|
api.submit_validation_diagnostics(model_id="my-model", predictions=df)
|
|
94
|
+
|
|
95
|
+
api.get_dataset_info(universe="futures")
|
|
96
|
+
api.get_diagnostics(model_id="my-model")
|
|
90
97
|
```
|
|
91
98
|
|
|
92
99
|
### Plotting (optional `viz` extra)
|
|
@@ -366,10 +366,14 @@ class EverestAPI:
|
|
|
366
366
|
poll_interval: float = 2.0,
|
|
367
367
|
timeout: float = 600.0,
|
|
368
368
|
) -> dict:
|
|
369
|
-
"""POST /api/v1/diagnostics/upload (multipart) — score
|
|
369
|
+
"""POST /api/v1/diagnostics/upload (multipart) — score out-of-sample predictions.
|
|
370
370
|
|
|
371
371
|
``predictions`` is a pandas DataFrame (``id`` + ``prediction`` columns) or a
|
|
372
|
-
path to a ``.parquet`` / ``.csv`` file
|
|
372
|
+
path to a ``.parquet`` / ``.csv`` file, generated on the blind eiq_live_2026
|
|
373
|
+
set (``download_dataset(split="live")``). The platform scores it server-side
|
|
374
|
+
against the held-out labeled eiq_live_2026 answers — those answers are never
|
|
375
|
+
downloadable. Tune and self-score offline on the labeled validation set first
|
|
376
|
+
(see :mod:`everestapi.scoring`). Returns the 202 accept dict; with
|
|
373
377
|
``wait=True`` polls ``runs/{upload_id}`` until ``done`` (returns the run) and
|
|
374
378
|
raises :class:`EverestError` on ``failed`` or timeout. Display-only; results
|
|
375
379
|
also surface in the website Validation Diagnostics rail.
|
|
@@ -590,11 +594,18 @@ class EverestAPI:
|
|
|
590
594
|
404 the current version is resolved and the download retried once.
|
|
591
595
|
|
|
592
596
|
Splits: ``train`` / ``validation`` (features + targets), ``live``
|
|
593
|
-
(
|
|
597
|
+
(features only, no targets). In hackathon mode, ``validation`` is the
|
|
598
|
+
LABELED practice set — features + all ``target_*`` columns — that you tune
|
|
599
|
+
and self-score on offline (see :mod:`everestapi.scoring`), and ``live`` is
|
|
600
|
+
the BLIND eiq_live_2026 out-of-sample set (columns exactly ``exped``,
|
|
601
|
+
``exped_date``, ``instrument``, ``id`` — no targets) that you predict on and
|
|
602
|
+
submit. The labeled eiq_live_2026 answers are held out server-side and never
|
|
603
|
+
downloadable.
|
|
594
604
|
|
|
595
605
|
Futures is served by the futures endpoint (``version`` is a no-op):
|
|
596
|
-
it returns the real bregen tree to full-scope keys and the
|
|
597
|
-
|
|
606
|
+
it returns the real bregen tree to full-scope keys and the hackathon tree
|
|
607
|
+
(labeled ``validation`` practice set + blind ``live`` eiq_live_2026 scored
|
|
608
|
+
set) to hackathon-scoped keys.
|
|
598
609
|
"""
|
|
599
610
|
if output_path is None:
|
|
600
611
|
output_path = f"{universe}_{split}.parquet"
|
|
@@ -933,7 +944,8 @@ class EverestAPI:
|
|
|
933
944
|
|
|
934
945
|
With a hackathon-scoped key there is no live round — the response is a
|
|
935
946
|
diagnostics-mode payload (mode='diagnostics_hackathon') directing you to
|
|
936
|
-
|
|
947
|
+
tune offline on the labeled validation set, then predict on the blind
|
|
948
|
+
eiq_live_2026 set and submit_validation_diagnostics.
|
|
937
949
|
"""
|
|
938
950
|
return self._request(
|
|
939
951
|
"GET",
|
|
@@ -951,7 +963,10 @@ class EverestAPI:
|
|
|
951
963
|
"""GET /api/v1/get_started — mode-aware orientation.
|
|
952
964
|
|
|
953
965
|
Returns what to do next given this key's scope: the display-only
|
|
954
|
-
diagnostics-hackathon loop
|
|
966
|
+
diagnostics-hackathon loop (tune and self-score offline on the labeled
|
|
967
|
+
validation set, then predict on the blind eiq_live_2026 set and submit —
|
|
968
|
+
ranked on out-of-sample 2026 CORR on target_everest_20), or the live
|
|
969
|
+
futures tournament flow.
|
|
955
970
|
"""
|
|
956
971
|
return self._request("GET", "/api/v1/get_started")
|
|
957
972
|
|
|
@@ -238,7 +238,7 @@ TOOLS = [
|
|
|
238
238
|
},
|
|
239
239
|
{
|
|
240
240
|
"name": "download_dataset",
|
|
241
|
-
"description": "Download a dataset split (train/validation/live) as a parquet file. Returns the local file path.",
|
|
241
|
+
"description": "Download a dataset split (train/validation/live) as a parquet file. Returns the local file path. In hackathon mode: 'validation' is the LABELED practice set (features + all target_* columns) for offline tuning and self-scoring; 'live' is the BLIND eiq_live_2026 out-of-sample set (columns exactly exped, exped_date, instrument, id — NO targets) that you predict on and submit. The labeled 2026 answers are held out server-side and never downloadable.",
|
|
242
242
|
"inputSchema": {
|
|
243
243
|
"type": "object",
|
|
244
244
|
"properties": {
|
|
@@ -365,7 +365,7 @@ TOOLS = [
|
|
|
365
365
|
},
|
|
366
366
|
{
|
|
367
367
|
"name": "get_current_round",
|
|
368
|
-
"description": "Get the current active round for a tournament (defaults to 'futures'; equities is unlaunched). With a hackathon key there is no live round — the response is a diagnostics-mode payload directing you to
|
|
368
|
+
"description": "Get the current active round for a tournament (defaults to 'futures'; equities is unlaunched). With a hackathon key there is no live round — the response is a diagnostics-mode payload directing you to tune offline on the labeled validation set, then download the blind eiq_live_2026 set, predict, and submit_validation_diagnostics.",
|
|
369
369
|
"inputSchema": {
|
|
370
370
|
"type": "object",
|
|
371
371
|
"properties": {
|
|
@@ -380,7 +380,7 @@ TOOLS = [
|
|
|
380
380
|
},
|
|
381
381
|
{
|
|
382
382
|
"name": "get_started",
|
|
383
|
-
"description": "Mode-aware orientation: returns what to do next given your API key's scope. A hackathon key gets the display-only diagnostics loop (objective: maximize out-of-sample CORR on target_everest_20); a full key gets the live futures tournament flow. Start here.",
|
|
383
|
+
"description": "Mode-aware orientation: returns what to do next given your API key's scope. A hackathon key gets the display-only diagnostics loop (tune and self-score offline on the labeled validation set, then predict on the blind eiq_live_2026 set and submit; objective: maximize out-of-sample 2026 CORR on target_everest_20, in-sample fit is not rewarded); a full key gets the live futures tournament flow. Start here.",
|
|
384
384
|
"inputSchema": {
|
|
385
385
|
"type": "object",
|
|
386
386
|
"properties": {},
|
|
@@ -546,13 +546,15 @@ TOOLS = [
|
|
|
546
546
|
{
|
|
547
547
|
"name": "submit_validation_diagnostics",
|
|
548
548
|
"description": (
|
|
549
|
-
"Upload a NEW
|
|
550
|
-
"≤100 MB)
|
|
551
|
-
"
|
|
552
|
-
"
|
|
553
|
-
"(
|
|
554
|
-
"
|
|
555
|
-
"
|
|
549
|
+
"Upload a NEW predictions file (parquet/CSV with id+prediction columns, "
|
|
550
|
+
"≤100 MB) generated on the blind eiq_live_2026 out-of-sample set; the platform "
|
|
551
|
+
"scores it server-side against the held-out labeled eiq_live_2026 answers (never "
|
|
552
|
+
"downloadable). The required id set is the blind eiq_live_2026 set — fetch it with "
|
|
553
|
+
"download_dataset(split='live') and use it as the upload template "
|
|
554
|
+
"(its id column, your own prediction column). Tune and self-score offline on the "
|
|
555
|
+
"labeled validation set first (see the scoring helper). Use this only when you have "
|
|
556
|
+
"fresh predictions to score — to read the model's EXISTING latest result without "
|
|
557
|
+
"re-scoring (instant, no wait), call run_validation_diagnostics(model_id) instead. "
|
|
556
558
|
"The platform scores a 9-metric panel (CORR20, BMC, FNC, Sharpe, std dev, "
|
|
557
559
|
"feature-exposure, Max Drawdown, autocorrelation, example-preds-corr) asynchronously "
|
|
558
560
|
"and surfaces the run in the website Validation Diagnostics rail. By default "
|
|
@@ -594,9 +596,10 @@ TOOLS = [
|
|
|
594
596
|
{
|
|
595
597
|
"name": "get_diagnostics_leaderboard",
|
|
596
598
|
"description": (
|
|
597
|
-
"Get the validation diagnostics leaderboard — global ranking of
|
|
598
|
-
"
|
|
599
|
-
"runs and marks your own entries;
|
|
599
|
+
"Get the validation diagnostics leaderboard — global ranking of agents by their "
|
|
600
|
+
"out-of-sample eiq_live_2026 CORR on target_everest_20 (in-sample fit is not rewarded). "
|
|
601
|
+
"view='agents' (default) ranks participant model runs and marks your own entries; "
|
|
602
|
+
"view='benchmarks' ranks the official platform benchmarks."
|
|
600
603
|
),
|
|
601
604
|
"inputSchema": {
|
|
602
605
|
"type": "object",
|
|
@@ -894,6 +897,12 @@ TOOLSETS: dict[str, set[str]] = {
|
|
|
894
897
|
"get_scores",
|
|
895
898
|
"get_leaderboard",
|
|
896
899
|
"get_capabilities",
|
|
900
|
+
# EVE-967: the diagnostics upload + leaderboard are the entire hackathon
|
|
901
|
+
# flow get_started steers to; advertise them by default (core) so a
|
|
902
|
+
# hackathon key discovers them without setting EIQ_MCP_TOOLSETS. The rest
|
|
903
|
+
# of the diagnostics lifecycle stays in "diagnostics".
|
|
904
|
+
"submit_validation_diagnostics",
|
|
905
|
+
"get_diagnostics_leaderboard",
|
|
897
906
|
},
|
|
898
907
|
"data": {
|
|
899
908
|
"get_universe",
|
|
@@ -911,9 +920,9 @@ TOOLSETS: dict[str, set[str]] = {
|
|
|
911
920
|
},
|
|
912
921
|
"submit": {"submit_predictions", "upload_model", "get_upload_status"},
|
|
913
922
|
"diagnostics": {
|
|
923
|
+
# submit_validation_diagnostics + get_diagnostics_leaderboard promoted to
|
|
924
|
+
# "core" (EVE-967) — every name still lives in exactly one group.
|
|
914
925
|
"run_validation_diagnostics",
|
|
915
|
-
"submit_validation_diagnostics",
|
|
916
|
-
"get_diagnostics_leaderboard",
|
|
917
926
|
},
|
|
918
927
|
"compute": {
|
|
919
928
|
"quick_train",
|
|
@@ -1275,9 +1284,14 @@ def _error_payload(exc: Exception) -> str:
|
|
|
1275
1284
|
|
|
1276
1285
|
if isinstance(exc, EverestError):
|
|
1277
1286
|
code = exc.status_code
|
|
1287
|
+
# FastAPI wraps errors as {"detail": {"code": ..., "message": ...}} —
|
|
1288
|
+
# pull the machine code so 403 can distinguish a scope restriction
|
|
1289
|
+
# (not available in diagnostics mode) from a genuine auth failure.
|
|
1290
|
+
inner = exc.detail.get("detail") if isinstance(exc.detail, dict) else None
|
|
1291
|
+
err_code = inner.get("code") if isinstance(inner, dict) else None
|
|
1292
|
+
|
|
1278
1293
|
hints = {
|
|
1279
1294
|
401: "Auth failed — set EIQ_API_KEY to your Everesteer API key.",
|
|
1280
|
-
403: "Auth failed — set EIQ_API_KEY to your Everesteer API key.",
|
|
1281
1295
|
404: "Not found — check the model_id / upload_id exists.",
|
|
1282
1296
|
409: "A diagnostics run is already in flight for this model — read it with "
|
|
1283
1297
|
"run_validation_diagnostics(model_id) or wait for it to finish.",
|
|
@@ -1285,10 +1299,19 @@ def _error_payload(exc: Exception) -> str:
|
|
|
1285
1299
|
504: "Still computing — scoring can take 15-20 min; poll again shortly.",
|
|
1286
1300
|
}
|
|
1287
1301
|
hint = hints.get(code)
|
|
1302
|
+
if code == 403:
|
|
1303
|
+
if err_code in ("scope_mismatch", "scope_restricted"):
|
|
1304
|
+
hint = (
|
|
1305
|
+
"Not available in diagnostics (hackathon) mode — this is a "
|
|
1306
|
+
"live-tournament / full-scope action. Use the diagnostics flow: "
|
|
1307
|
+
"submit_validation_diagnostics + get_diagnostics_leaderboard."
|
|
1308
|
+
)
|
|
1309
|
+
else:
|
|
1310
|
+
hint = "Auth failed — set EIQ_API_KEY to your Everesteer API key."
|
|
1288
1311
|
if hint is None and 400 <= code < 500:
|
|
1289
|
-
hint = "Bad request —
|
|
1312
|
+
hint = "Bad request — check the arguments match the tool schema."
|
|
1290
1313
|
elif hint is None and code >= 500:
|
|
1291
|
-
hint = "Server error — retry; if it persists the
|
|
1314
|
+
hint = "Server error — retry; if it persists the request may have failed."
|
|
1292
1315
|
return json.dumps({"error": exc.detail, "status": code, "hint": hint})
|
|
1293
1316
|
return json.dumps({"error": str(exc)})
|
|
1294
1317
|
|
|
@@ -1402,9 +1425,13 @@ _MCP_INSTRUCTIONS = (
|
|
|
1402
1425
|
"Everesteer tournament MCP server. Call get_started first — it is mode-aware and tells you "
|
|
1403
1426
|
"what to do next based on your API key's scope.\n\n"
|
|
1404
1427
|
"Hackathon mode (a hackathon-scoped key): there is no live tournament round and no "
|
|
1405
|
-
"staking or payout — this is a display-only diagnostics event.
|
|
1406
|
-
"
|
|
1407
|
-
"
|
|
1428
|
+
"staking or payout — this is a display-only diagnostics event. Tune and self-score "
|
|
1429
|
+
"offline on the LABELED validation set (features + target_* columns, via the scoring "
|
|
1430
|
+
"helper), then predict on the BLIND eiq_live_2026 set (columns exped, exped_date, "
|
|
1431
|
+
"instrument, id — no targets) and submit; the leaderboard ranks your OUT-OF-SAMPLE 2026 "
|
|
1432
|
+
"CORR on target_everest_20. In-sample fit is not rewarded. Flow: "
|
|
1433
|
+
"download_dataset(split='validation') -> tune + self-score offline -> "
|
|
1434
|
+
"download_dataset(split='live') -> predict -> submit_validation_diagnostics -> "
|
|
1408
1435
|
"get_diagnostics_leaderboard. Round/universe/feature/benchmark reads return empty "
|
|
1409
1436
|
"diagnostics-mode payloads, not live data.\n\n"
|
|
1410
1437
|
"Tournament mode (a full-scope key): submit daily futures predictions; payout = "
|
|
@@ -27,6 +27,14 @@ Quickstart::
|
|
|
27
27
|
for e in val.exped.unique()
|
|
28
28
|
]
|
|
29
29
|
print("mean CORR20:", sum(corrs) / len(corrs))
|
|
30
|
+
|
|
31
|
+
In hackathon mode the validation set ships LABELED (features + ``target_*``
|
|
32
|
+
columns), so this helper is the offline self-scoring path: use it to tune your
|
|
33
|
+
model before you submit. It scores in-sample and is for tuning only — the
|
|
34
|
+
official hackathon leaderboard scores your submitted predictions on the BLIND
|
|
35
|
+
``eiq_live_2026`` out-of-sample set server-side (its labeled answers are held
|
|
36
|
+
out and never downloadable), ranking your 2026 CORR on ``target_everest_20``.
|
|
37
|
+
In-sample fit is not rewarded.
|
|
30
38
|
"""
|
|
31
39
|
|
|
32
40
|
from __future__ import annotations
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: everestapi
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.8
|
|
4
4
|
Summary: Python SDK for the Everesteer prediction tournament platform
|
|
5
5
|
Author-email: Everesteer <support@everesteer.ai>
|
|
6
6
|
License-Expression: MIT
|
|
@@ -116,16 +116,23 @@ everestapi submit --model my-model --file predictions.parquet
|
|
|
116
116
|
|
|
117
117
|
### Data & diagnostics
|
|
118
118
|
|
|
119
|
+
The hackathon is a display-only diagnostics event. **Tune and self-score offline
|
|
120
|
+
on the labeled validation set** (features + `target_*` columns), then **predict on
|
|
121
|
+
the blind `eiq_live_2026` set and submit** — the leaderboard ranks your
|
|
122
|
+
out-of-sample 2026 CORR on `target_everest_20`. In-sample fit is not rewarded. The
|
|
123
|
+
labeled `eiq_live_2026` answers are held out server-side and are never downloadable.
|
|
124
|
+
|
|
119
125
|
```python
|
|
120
|
-
|
|
121
|
-
api.
|
|
122
|
-
api.get_dataset_info(universe="futures")
|
|
123
|
-
api.get_diagnostics(model_id="my-model")
|
|
126
|
+
# Labeled practice set — tune + self-score offline with everestapi.scoring:
|
|
127
|
+
api.download_dataset(universe="futures", split="validation")
|
|
124
128
|
|
|
125
|
-
#
|
|
126
|
-
#
|
|
127
|
-
api.download_dataset(universe="futures", split="
|
|
129
|
+
# Blind scored set (columns: exped, exped_date, instrument, id — no targets).
|
|
130
|
+
# Predict on it, then submit; it is also the upload id template.
|
|
131
|
+
api.download_dataset(universe="futures", split="live")
|
|
128
132
|
api.submit_validation_diagnostics(model_id="my-model", predictions=df)
|
|
133
|
+
|
|
134
|
+
api.get_dataset_info(universe="futures")
|
|
135
|
+
api.get_diagnostics(model_id="my-model")
|
|
129
136
|
```
|
|
130
137
|
|
|
131
138
|
### Plotting (optional `viz` extra)
|
|
@@ -23,6 +23,7 @@ tests/test_diagnostics.py
|
|
|
23
23
|
tests/test_eve953_mcp_progress.py
|
|
24
24
|
tests/test_eve957_mcp_annotations_resources.py
|
|
25
25
|
tests/test_eve959_toolsets.py
|
|
26
|
+
tests/test_eve967_mcp_discoverability.py
|
|
26
27
|
tests/test_json_or_raise.py
|
|
27
28
|
tests/test_mcp_and_models.py
|
|
28
29
|
tests/test_prediction_range.py
|
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
"""EVE-967 — the diagnostics upload + leaderboard tools are discoverable on the
|
|
2
|
+
default (core) toolset, and MCP error hints don't mislead:
|
|
3
|
+
- a 403 scope restriction points at the diagnostics flow, not "rotate your key";
|
|
4
|
+
- the generic 400 fallback is tool-neutral (no diagnostics-upload bleed).
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import json
|
|
10
|
+
|
|
11
|
+
import everestapi.mcp.server as S
|
|
12
|
+
from everestapi.client import EverestError
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def test_diagnostics_tools_advertised_by_default(monkeypatch):
|
|
16
|
+
monkeypatch.delenv("EIQ_MCP_TOOLSETS", raising=False)
|
|
17
|
+
enabled = S._enabled_tool_names()
|
|
18
|
+
assert "submit_validation_diagnostics" in enabled
|
|
19
|
+
assert "get_diagnostics_leaderboard" in enabled
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def test_403_scope_mismatch_hint_points_to_diagnostics():
|
|
23
|
+
exc = EverestError(403, {"detail": {"code": "scope_mismatch", "message": "x"}})
|
|
24
|
+
payload = json.loads(S._error_payload(exc))
|
|
25
|
+
assert payload["status"] == 403
|
|
26
|
+
assert "diagnostics" in payload["hint"].lower()
|
|
27
|
+
assert "set EIQ_API_KEY" not in payload["hint"]
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def test_403_plain_auth_still_says_set_key():
|
|
31
|
+
exc = EverestError(403, "forbidden") # no machine code → generic auth hint
|
|
32
|
+
payload = json.loads(S._error_payload(exc))
|
|
33
|
+
assert "EIQ_API_KEY" in payload["hint"]
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def test_generic_400_hint_is_tool_neutral():
|
|
37
|
+
exc = EverestError(400, {"detail": "bad"})
|
|
38
|
+
payload = json.loads(S._error_payload(exc))
|
|
39
|
+
assert "id+prediction" not in payload["hint"]
|
|
40
|
+
assert "schema" in payload["hint"].lower()
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|