everestapi 0.3.2__tar.gz → 0.3.6__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (34) hide show
  1. {everestapi-0.3.2/src/everestapi.egg-info → everestapi-0.3.6}/PKG-INFO +1 -1
  2. {everestapi-0.3.2 → everestapi-0.3.6}/src/everestapi/__init__.py +1 -1
  3. {everestapi-0.3.2 → everestapi-0.3.6}/src/everestapi/client.py +97 -20
  4. {everestapi-0.3.2 → everestapi-0.3.6}/src/everestapi/mcp/server.py +117 -16
  5. {everestapi-0.3.2 → everestapi-0.3.6/src/everestapi.egg-info}/PKG-INFO +1 -1
  6. {everestapi-0.3.2 → everestapi-0.3.6}/src/everestapi.egg-info/SOURCES.txt +1 -0
  7. {everestapi-0.3.2 → everestapi-0.3.6}/tests/test_client.py +1 -1
  8. everestapi-0.3.6/tests/test_client_env.py +12 -0
  9. {everestapi-0.3.2 → everestapi-0.3.6}/tests/test_diagnostics.py +103 -4
  10. {everestapi-0.3.2 → everestapi-0.3.6}/tests/test_eve1087_cpu_tier.py +1 -1
  11. {everestapi-0.3.2 → everestapi-0.3.6}/tests/test_eve959_toolsets.py +17 -0
  12. {everestapi-0.3.2 → everestapi-0.3.6}/tests/test_mcp_and_models.py +18 -7
  13. {everestapi-0.3.2 → everestapi-0.3.6}/LICENSE +0 -0
  14. {everestapi-0.3.2 → everestapi-0.3.6}/README.md +0 -0
  15. {everestapi-0.3.2 → everestapi-0.3.6}/pyproject.toml +0 -0
  16. {everestapi-0.3.2 → everestapi-0.3.6}/setup.cfg +0 -0
  17. {everestapi-0.3.2 → everestapi-0.3.6}/src/everestapi/__main__.py +0 -0
  18. {everestapi-0.3.2 → everestapi-0.3.6}/src/everestapi/cli.py +0 -0
  19. {everestapi-0.3.2 → everestapi-0.3.6}/src/everestapi/mcp/__init__.py +0 -0
  20. {everestapi-0.3.2 → everestapi-0.3.6}/src/everestapi/mcp/__main__.py +0 -0
  21. {everestapi-0.3.2 → everestapi-0.3.6}/src/everestapi/plots.py +0 -0
  22. {everestapi-0.3.2 → everestapi-0.3.6}/src/everestapi/scoring.py +0 -0
  23. {everestapi-0.3.2 → everestapi-0.3.6}/src/everestapi/types.py +0 -0
  24. {everestapi-0.3.2 → everestapi-0.3.6}/src/everestapi.egg-info/dependency_links.txt +0 -0
  25. {everestapi-0.3.2 → everestapi-0.3.6}/src/everestapi.egg-info/entry_points.txt +0 -0
  26. {everestapi-0.3.2 → everestapi-0.3.6}/src/everestapi.egg-info/requires.txt +0 -0
  27. {everestapi-0.3.2 → everestapi-0.3.6}/src/everestapi.egg-info/top_level.txt +0 -0
  28. {everestapi-0.3.2 → everestapi-0.3.6}/tests/test_cli.py +0 -0
  29. {everestapi-0.3.2 → everestapi-0.3.6}/tests/test_eve953_mcp_progress.py +0 -0
  30. {everestapi-0.3.2 → everestapi-0.3.6}/tests/test_eve957_mcp_annotations_resources.py +0 -0
  31. {everestapi-0.3.2 → everestapi-0.3.6}/tests/test_eve967_mcp_discoverability.py +0 -0
  32. {everestapi-0.3.2 → everestapi-0.3.6}/tests/test_json_or_raise.py +0 -0
  33. {everestapi-0.3.2 → everestapi-0.3.6}/tests/test_prediction_range.py +0 -0
  34. {everestapi-0.3.2 → everestapi-0.3.6}/tests/test_scoring.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: everestapi
3
- Version: 0.3.2
3
+ Version: 0.3.6
4
4
  Summary: Python SDK for the Everesteer prediction tournament platform
5
5
  Author-email: Everesteer <support@everesteer.ai>
6
6
  License-Expression: MIT
@@ -1,6 +1,6 @@
1
1
  """EverestAPI — Python SDK for the Everesteer prediction tournament platform."""
2
2
 
3
- __version__ = "0.3.2"
3
+ __version__ = "0.3.6"
4
4
 
5
5
  from everestapi.client import EverestAPI, EverestError
6
6
  from everestapi.types import (
@@ -405,29 +405,40 @@ class EverestAPI:
405
405
  data = f.read()
406
406
  fname = str(predictions).rsplit("/", 1)[-1].rsplit("\\", 1)[-1]
407
407
 
408
- files = {"file": (fname, data, "application/octet-stream")}
408
+ pkl_data = None
409
+ pkl_fname = "model.pkl"
409
410
  if model_pkl is not None:
410
411
  if isinstance(model_pkl, bytes | bytearray):
411
412
  pkl_data = bytes(model_pkl)
412
- pkl_fname = "model.pkl"
413
413
  else: # path
414
414
  with open(model_pkl, "rb") as f:
415
415
  pkl_data = f.read()
416
416
  pkl_fname = str(model_pkl).rsplit("/", 1)[-1].rsplit("\\", 1)[-1]
417
- files["model_pkl"] = (pkl_fname, pkl_data, "application/octet-stream")
418
417
 
419
- resp = self._client.post(
420
- "/api/v1/diagnostics/upload",
421
- files=files,
422
- data={"model_id": model_id, "tournament": tournament, "target": target},
418
+ # EVE-1159: prefer the presigned direct-to-CDN path — it uploads the file
419
+ # straight to S3 through a DNS-only CloudFront host, bypassing the
420
+ # Cloudflare proxy edge that drops large POSTs. Falls back to the multipart
421
+ # endpoint when the server/env doesn't offer it (old server, or the CDN
422
+ # isn't configured), so behaviour is unchanged everywhere else.
423
+ accepted = self._diagnostics_presigned_accept(
424
+ model_id, data, fname, pkl_data, pkl_fname, tournament, target
423
425
  )
424
- if resp.status_code >= 400:
425
- try:
426
- detail = resp.json()
427
- except Exception:
428
- detail = resp.text
429
- raise EverestError(resp.status_code, detail)
430
- accepted = resp.json()
426
+ if accepted is None:
427
+ files = {"file": (fname, data, "application/octet-stream")}
428
+ if pkl_data is not None:
429
+ files["model_pkl"] = (pkl_fname, pkl_data, "application/octet-stream")
430
+ resp = self._client.post(
431
+ "/api/v1/diagnostics/upload",
432
+ files=files,
433
+ data={"model_id": model_id, "tournament": tournament, "target": target},
434
+ )
435
+ if resp.status_code >= 400:
436
+ try:
437
+ detail = resp.json()
438
+ except Exception:
439
+ detail = resp.text
440
+ raise EverestError(resp.status_code, detail)
441
+ accepted = resp.json()
431
442
  if not wait:
432
443
  return accepted
433
444
  deadline = time.time() + timeout
@@ -440,6 +451,65 @@ class EverestAPI:
440
451
  time.sleep(poll_interval)
441
452
  raise EverestError(504, "diagnostics run timed out")
442
453
 
454
+ def _diagnostics_presigned_accept(
455
+ self,
456
+ model_id: str,
457
+ data: bytes,
458
+ fname: str,
459
+ pkl_data: bytes | None,
460
+ pkl_fname: str,
461
+ tournament: str,
462
+ target: str,
463
+ ) -> dict | None:
464
+ """EVE-1159: attempt the presigned direct-to-CDN upload.
465
+
466
+ Returns the accept dict on success, or ``None`` when the server/env
467
+ doesn't offer it (endpoint 404, or ``available: false``) so the caller
468
+ falls back to the multipart endpoint. Raises :class:`EverestError` on a
469
+ real failure (bad file, storage error) so it is NOT silently retried.
470
+ """
471
+ body: dict = {
472
+ "model_id": model_id,
473
+ "filename": fname,
474
+ "tournament": tournament,
475
+ "target": target,
476
+ }
477
+ if pkl_data is not None:
478
+ body["model_pkl_filename"] = pkl_fname
479
+ try:
480
+ presign = self._request("POST", "/api/v1/diagnostics/upload/presign", json=body)
481
+ except EverestError as e:
482
+ if e.status_code == 404:
483
+ return None # server predates the presigned path
484
+ raise
485
+ if not presign.get("available"):
486
+ return None # upload CDN not configured in this env
487
+ self._s3_post(presign["predictions"], fname, data)
488
+ if pkl_data is not None and presign.get("model_pkl"):
489
+ self._s3_post(presign["model_pkl"], pkl_fname, pkl_data)
490
+ complete: dict = {
491
+ "upload_id": presign["upload_id"],
492
+ "model_id": model_id,
493
+ "filename": fname,
494
+ "tournament": tournament,
495
+ "target": target,
496
+ }
497
+ return self._request("POST", "/api/v1/diagnostics/upload/complete", json=complete)
498
+
499
+ @staticmethod
500
+ def _s3_post(post: dict, fname: str, data: bytes) -> None:
501
+ """POST one object straight to S3 via the presigned CloudFront form. The
502
+ file field MUST come last per the S3 POST policy — httpx sends the ``data``
503
+ form fields before the ``files`` part, so this ordering is correct."""
504
+ resp = httpx.post(
505
+ post["url"],
506
+ data=post["fields"],
507
+ files={"file": (fname, data, "application/octet-stream")},
508
+ timeout=300.0,
509
+ )
510
+ if resp.status_code not in (200, 201, 204):
511
+ raise EverestError(resp.status_code, f"direct upload failed: {resp.text[:300]}")
512
+
443
513
  def get_diagnostics_run(self, upload_id: str) -> dict:
444
514
  """GET /api/v1/diagnostics/runs/{upload_id} — status + metrics for one run."""
445
515
  return self._request("GET", f"/api/v1/diagnostics/runs/{upload_id}")
@@ -662,9 +732,10 @@ class EverestAPI:
662
732
  The download route is version-scoped
663
733
  (``/api/v1/data/download/{version}/{universe}/{split}``). Pass
664
734
  ``version="latest"`` (the default) to resolve the current version from
665
- ``/api/v1/data/versions`` automatically, or an explicit version
666
- (e.g. ``"bregen"``) to pin it. A stale pinned version self-heals: on a
667
- 404 the current version is resolved and the download retried once.
735
+ ``/api/v1/data/versions`` automatically, or an explicit version string
736
+ (as returned by :meth:`list_versions`) to pin it. A stale pinned
737
+ version self-heals: on a 404 the current version is resolved and the
738
+ download retried once.
668
739
 
669
740
  Splits: ``train`` / ``validation`` (features + targets), ``live``
670
741
  (features only, no targets). In hackathon mode, ``train`` is the LABELED
@@ -677,8 +748,8 @@ class EverestAPI:
677
748
  404s for hackathon keys.
678
749
 
679
750
  Futures is served by the futures endpoint (``version`` is a no-op):
680
- it returns the real bregen tree to full-scope keys and the hackathon
681
- tree (labeled ``train`` + blank-target ``validation``) to
751
+ it returns the current production dataset to full-scope keys and the
752
+ hackathon tree (labeled ``train`` + blank-target ``validation``) to
682
753
  hackathon-scoped keys.
683
754
 
684
755
  Value semantics: observed feature values are cross-sectional
@@ -895,12 +966,18 @@ class EverestAPI:
895
966
 
896
967
  # -- data discovery ---------------------------------------------------
897
968
 
898
- def get_dataset_schema(self, version: str = "bregen", verbose: bool = False) -> dict:
969
+ def get_dataset_schema(self, version: str = "latest", verbose: bool = False) -> dict:
899
970
  """Get the dataset schema for a version.
900
971
 
972
+ ``version="latest"`` (the default) resolves the current version from
973
+ ``/api/v1/data/versions`` automatically, or pass an explicit version
974
+ string (as returned by :meth:`list_versions`) to pin it.
975
+
901
976
  Default is a compact summary (feature-set counts, target list, which splits
902
977
  are labeled, the CV embargo, distinct feature count, active version). Pass
903
978
  ``verbose=True`` for the full per-feature listing."""
979
+ if version in (None, "latest", "current"):
980
+ version = self._current_dataset_version("futures") or "v0"
904
981
  path = f"/api/v1/data/{version}/schema"
905
982
  if verbose:
906
983
  path += "?verbose=true"
@@ -298,8 +298,8 @@ TOOLS = [
298
298
  "properties": {
299
299
  "version": {
300
300
  "type": "string",
301
- "description": "Dataset version (default: bregen)",
302
- "default": "bregen",
301
+ "description": "Dataset version (default: latest, resolved from /api/v1/data/versions)",
302
+ "default": "latest",
303
303
  },
304
304
  },
305
305
  "required": [],
@@ -340,7 +340,7 @@ TOOLS = [
340
340
  },
341
341
  {
342
342
  "name": "train",
343
- "description": "Train a model on Everesteer data (no data upload — the platform uses the same obfuscated dataset you download). The PLATFORM loads data, runs exped-purged/embargoed cross-validation, computes canonical tournament metrics (a weighted combination of CORR and AIMC, AIMC-dominant, with per-model weights read from the API; AIMC before round close is a PRE-SUBMISSION ESTIMATE vs a static consensus proxy — real AIMC is scored vs the live stake-weighted crowd at round close), pickles the final model, and returns DOWNLOADABLE artifacts (model .pkl + validation/live prediction files) plus the metrics. Supply a built-in `model` preset (lightgbm/xgboost/ridge/mlp/random_forest), or `model='custom'` with a `custom_model_fn` (and optional `custom_feature_fn`) to run your own code — it executes inside an isolated, network-denied sandbox with no filesystem access and never sees held-out targets; a script rejected by the static safety scanner returns 400. `train` does NOT submit predictions or host a model on your behalf — use submit_futures_predictions/submit_predictions_file to submit and upload_model to host the .pkl. Poll with get_job_status() — the completed result carries the metrics + a predictions download URL; fetch the model file separately via get_model_download_url(). Call get_compute_credits for the live per-hour GPU rate card and your balance.",
343
+ "description": "Train a model REMOTELY on Everesteer's hosted compute — nothing executes on your machine. The call returns immediately with a job id and may run for up to 45 minutes. Poll get_job_status and get_job_log every 10–15 seconds for a reconnect-safe estimated percentage, stage, and fold-level messages. The platform loads its obfuscated data, runs purged CV, computes canonical metrics, and returns downloadable model/prediction artifacts; it does not submit or host the model for you. Custom code runs in an isolated network-denied sandbox without filesystem access or held-out targets. Call get_compute_credits for the live rate card and balance.",
344
344
  "inputSchema": {
345
345
  "type": "object",
346
346
  "properties": {
@@ -404,7 +404,7 @@ TOOLS = [
404
404
  },
405
405
  {
406
406
  "name": "get_job_status",
407
- "description": "Check status of a compute job. Returns status, cost, output files when complete.",
407
+ "description": "Check status plus reconnect-safe estimated percentage, stage, message, and update time. Poll every 10–15 seconds while running; use get_job_log for the full fold-level trail.",
408
408
  "inputSchema": {
409
409
  "type": "object",
410
410
  "properties": {
@@ -416,6 +416,17 @@ TOOLS = [
416
416
  "required": ["job_id"],
417
417
  },
418
418
  },
419
+ {
420
+ "name": "get_job_log",
421
+ "description": "Return the estimated progress summary and lifecycle/per-fold event trail for one training job. Poll every 10–15 seconds while running. Percentages are workflow estimates, not an ETA.",
422
+ "inputSchema": {
423
+ "type": "object",
424
+ "properties": {
425
+ "job_id": {"type": "string", "description": "Job ID returned by train"},
426
+ },
427
+ "required": ["job_id"],
428
+ },
429
+ },
419
430
  {
420
431
  "name": "get_model_download_url",
421
432
  "description": "Get a presigned download URL for a trained model from a completed job. URL valid for 1 hour. Only you can download your own models.",
@@ -430,6 +441,38 @@ TOOLS = [
430
441
  "required": ["job_id"],
431
442
  },
432
443
  },
444
+ {
445
+ "name": "get_job_predictions_url",
446
+ "description": "Get a presigned download URL for the prediction files (validation + live parquet) produced by a completed train job. URL valid for ~1 hour. Only you can download your own job's predictions. Fetch the model .pkl separately with get_model_download_url.",
447
+ "inputSchema": {
448
+ "type": "object",
449
+ "properties": {
450
+ "job_id": {
451
+ "type": "string",
452
+ "description": "Job ID from a train job",
453
+ },
454
+ },
455
+ "required": ["job_id"],
456
+ },
457
+ },
458
+ {
459
+ "name": "list_compute_jobs",
460
+ "description": "List your compute training jobs, newest first. Rows carry status, GPU, cost and the provider job id but exclude the large config/output payloads — use get_job_status for one job's full record. Optionally filter by status.",
461
+ "inputSchema": {
462
+ "type": "object",
463
+ "properties": {
464
+ "limit": {
465
+ "type": "integer",
466
+ "description": "Max rows (default 50, capped at 200)",
467
+ },
468
+ "status": {
469
+ "type": "string",
470
+ "description": "Filter to one status",
471
+ "enum": ["pending", "running", "completed", "failed", "timeout", "cancelled"],
472
+ },
473
+ },
474
+ },
475
+ },
433
476
  {
434
477
  "name": "get_compute_credits",
435
478
  "description": "Check your compute credit balance and usage. Purchase credits with USDC before using GPU resources.",
@@ -994,7 +1037,10 @@ READ_ONLY_TOOLS = frozenset(
994
1037
  "download_dataset",
995
1038
  "download_benchmark",
996
1039
  "get_job_status",
1040
+ "get_job_log",
1041
+ "list_compute_jobs",
997
1042
  "get_model_download_url",
1043
+ "get_job_predictions_url",
998
1044
  "get_compute_credits",
999
1045
  "get_rounds",
1000
1046
  "get_current_round",
@@ -1077,10 +1123,19 @@ TOOLSETS: dict[str, set[str]] = {
1077
1123
  # EVE-970: hosted training is a first-class flow now that every account
1078
1124
  # carries a compute grant — advertise train + get_compute_credits by
1079
1125
  # default so agents discover the hosted-training entry point without
1080
- # setting EIQ_MCP_TOOLSETS. get_job_status stays in "compute" (hidden
1081
- # tools remain callable by name). Mirrors the platform MCP server.
1126
+ # setting EIQ_MCP_TOOLSETS. EVE-1145 (platform parity): the whole
1127
+ # job-lifecycle group is core too. train's description tells the caller
1128
+ # to poll get_job_status()/get_job_log() and fetch artifacts via
1129
+ # get_model_download_url()/get_job_predictions_url() — hiding those in a
1130
+ # non-default group made hosted training an unfinishable dead end for an
1131
+ # agent that only sees the default toolset. Mirrors the platform MCP server.
1082
1132
  "train",
1083
1133
  "get_compute_credits",
1134
+ "get_job_status",
1135
+ "get_job_log",
1136
+ "list_compute_jobs",
1137
+ "get_model_download_url",
1138
+ "get_job_predictions_url",
1084
1139
  },
1085
1140
  "data": {
1086
1141
  "get_universe",
@@ -1093,7 +1148,6 @@ TOOLSETS: dict[str, set[str]] = {
1093
1148
  "get_benchmarks",
1094
1149
  "download_benchmark",
1095
1150
  "get_models",
1096
- "get_model_download_url",
1097
1151
  "get_model_per_exped_breakdown",
1098
1152
  },
1099
1153
  "submit": {"submit_predictions", "upload_model", "get_upload_status"},
@@ -1102,9 +1156,10 @@ TOOLSETS: dict[str, set[str]] = {
1102
1156
  # "core" (EVE-967) — every name still lives in exactly one group.
1103
1157
  "run_validation_diagnostics",
1104
1158
  },
1105
- "compute": {
1106
- "get_job_status",
1107
- },
1159
+ # EVE-1145: the job-lifecycle tools (get_job_status/get_job_log/
1160
+ # list_compute_jobs/get_model_download_url/get_job_predictions_url) were
1161
+ # promoted to "core". Kept as a stable, addressable EIQ_MCP_TOOLSETS name.
1162
+ "compute": set(),
1108
1163
  "staking": {
1109
1164
  "stake_on_model",
1110
1165
  "unstake_from_model",
@@ -1221,6 +1276,33 @@ def _validate_predictions_file(path: str) -> None:
1221
1276
  )
1222
1277
 
1223
1278
 
1279
+ def _client_env() -> dict:
1280
+ """Self-report the running interpreter + the command that upgrades IT.
1281
+
1282
+ The MCP runs from a dedicated venv (~/.eiq-agent) that a bare `pip install`
1283
+ in the user's shell does NOT target — so an agent told to "update everestapi"
1284
+ upgrades the wrong Python and leaves this one stale (EVE-1163). Build the
1285
+ upgrade command from sys.executable so the handed-back command always hits
1286
+ the venv the MCP actually runs from.
1287
+ """
1288
+ try:
1289
+ from importlib.metadata import version
1290
+
1291
+ ver = version("everestapi")
1292
+ except Exception: # metadata is always present in a real install
1293
+ ver = "unknown"
1294
+ return {
1295
+ "everestapi_version": ver,
1296
+ "python": sys.executable,
1297
+ "upgrade_command": f'"{sys.executable}" -m pip install -U "everestapi[mcp]"',
1298
+ "note": (
1299
+ "This MCP runs from the interpreter above. A bare `pip install` in "
1300
+ "your shell may target a different Python and leave this one stale — "
1301
+ "use upgrade_command (or re-run the everesteer installer)."
1302
+ ),
1303
+ }
1304
+
1305
+
1224
1306
  def _dispatch(name: str, arguments: dict) -> str:
1225
1307
  """Call the appropriate SDK method and return JSON string."""
1226
1308
  name = name.removeprefix(TOOL_NAMESPACE)
@@ -1263,7 +1345,7 @@ def _dispatch(name: str, arguments: dict) -> str:
1263
1345
  elif name == "get_leaderboard":
1264
1346
  result = client.get_leaderboard(period=arguments.get("period", "30d"))
1265
1347
  elif name == "get_dataset_schema":
1266
- result = client.get_dataset_schema(version=arguments.get("version", "bregen"))
1348
+ result = client.get_dataset_schema(version=arguments.get("version", "latest"))
1267
1349
  elif name == "download_dataset":
1268
1350
  path = client.download_dataset(
1269
1351
  universe=arguments.get("universe", "futures"),
@@ -1282,8 +1364,17 @@ def _dispatch(name: str, arguments: dict) -> str:
1282
1364
  result = client.train(**arguments)
1283
1365
  elif name == "get_job_status":
1284
1366
  result = client.get_job_status(job_id=arguments["job_id"])
1367
+ elif name == "get_job_log":
1368
+ result = client.get_job_log(job_id=arguments["job_id"])
1285
1369
  elif name == "get_model_download_url":
1286
1370
  result = client.get_model_download_url(job_id=arguments["job_id"])
1371
+ elif name == "get_job_predictions_url":
1372
+ result = client.get_job_predictions_url(job_id=arguments["job_id"])
1373
+ elif name == "list_compute_jobs":
1374
+ result = client.list_compute_jobs(
1375
+ limit=arguments.get("limit", 50),
1376
+ status=arguments.get("status"),
1377
+ )
1287
1378
  elif name == "get_compute_credits":
1288
1379
  result = client.get_compute_credits()
1289
1380
  elif name == "get_rounds":
@@ -1297,6 +1388,10 @@ def _dispatch(name: str, arguments: dict) -> str:
1297
1388
  )
1298
1389
  elif name == "get_started":
1299
1390
  result = client.get_started()
1391
+ # EVE-1163: tell a self-updating agent exactly which interpreter to
1392
+ # upgrade, so it can't miss the isolated ~/.eiq-agent venv.
1393
+ if isinstance(result, dict):
1394
+ result["client_env"] = _client_env()
1300
1395
  elif name == "get_status":
1301
1396
  # GET-proxy, same fallback shape as get_started — works on an older
1302
1397
  # standalone everestapi build without the get_status() method.
@@ -1618,7 +1713,8 @@ _MCP_INSTRUCTIONS = (
1618
1713
  "get_diagnostics_leaderboard; after submissions close, set_final_selection picks up to "
1619
1714
  "2 models as final entries during the grace window (default: best 2 public). "
1620
1715
  "Round/universe/feature/benchmark reads return empty diagnostics-mode payloads, not "
1621
- "live data.\n\n"
1716
+ "live data. Working example scripts (train/predict/submit) live at "
1717
+ "https://github.com/everestquant/example-scripts.\n\n"
1622
1718
  "Tournament mode (a full-scope key): submit daily futures predictions; payout is a "
1623
1719
  "weighted combination of CORR and AIMC (AIMC-dominant, per-model weights read from the "
1624
1720
  "API). The default tournament is 'futures' (equities is unlaunched).\n\n"
@@ -1649,9 +1745,12 @@ _NEXT_ACTIONS: dict[str, list[str]] = {
1649
1745
  "download_dataset": ["create_model", "submit_futures_predictions"],
1650
1746
  "download_benchmark": ["submit_futures_predictions"],
1651
1747
  "get_benchmarks": ["download_benchmark"],
1652
- "train": ["get_job_status"],
1653
- "get_job_status": ["get_model_download_url"], # conditional
1748
+ "train": ["get_job_status", "get_job_log"],
1749
+ "get_job_status": ["get_model_download_url", "get_job_predictions_url"], # conditional
1750
+ "get_job_log": ["get_job_status", "train"],
1751
+ "list_compute_jobs": ["get_job_status", "get_job_log"],
1654
1752
  "get_model_download_url": ["upload_model", "create_model"],
1753
+ "get_job_predictions_url": ["submit_futures_predictions", "create_model"],
1655
1754
  "create_model": ["submit_futures_predictions", "upload_model"],
1656
1755
  "upload_model": ["get_upload_status"],
1657
1756
  "get_upload_status": ["submit_futures_predictions"], # conditional
@@ -1714,6 +1813,8 @@ _NARRATION: dict[str, str] = {
1714
1813
  "train": "Training job queued — returns downloadable artifacts + metrics, not a submission.",
1715
1814
  "get_job_status": "Training job status: {status}.",
1716
1815
  "get_model_download_url": "Generated a download URL for the trained model (valid ~1h).",
1816
+ "get_job_predictions_url": "Generated a download URL for the job's predictions (valid ~1h).",
1817
+ "list_compute_jobs": "Returned your compute jobs.",
1717
1818
  "upload_model": "Model artifact uploaded; validation runs asynchronously.",
1718
1819
  "get_upload_status": "Model upload status: {status}.",
1719
1820
  "get_scores": "Returned the score history for model '{model_id}'.",
@@ -1771,8 +1872,8 @@ def _next_actions_for(name: str, arguments: dict, data: dict) -> list[str]:
1771
1872
  if status in ("completed", "done", "succeeded", "success"):
1772
1873
  return ["get_model_download_url", "create_model"]
1773
1874
  if status in ("failed", "error"):
1774
- return ["train"]
1775
- return ["get_job_status"] # still running — keep polling
1875
+ return ["get_job_log", "train"]
1876
+ return ["get_job_status", "get_job_log"]
1776
1877
  if name == "get_upload_status":
1777
1878
  if status in ("validated", "done", "success", "ready", "active"):
1778
1879
  return ["submit_futures_predictions", "get_scores"]
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: everestapi
3
- Version: 0.3.2
3
+ Version: 0.3.6
4
4
  Summary: Python SDK for the Everesteer prediction tournament platform
5
5
  Author-email: Everesteer <support@everesteer.ai>
6
6
  License-Expression: MIT
@@ -19,6 +19,7 @@ src/everestapi/mcp/__main__.py
19
19
  src/everestapi/mcp/server.py
20
20
  tests/test_cli.py
21
21
  tests/test_client.py
22
+ tests/test_client_env.py
22
23
  tests/test_diagnostics.py
23
24
  tests/test_eve1087_cpu_tier.py
24
25
  tests/test_eve953_mcp_progress.py
@@ -130,7 +130,7 @@ def test_download_dataset_futures_uses_futures_endpoint(httpx_mock, api, tmp_pat
130
130
  universe="futures",
131
131
  split="validation",
132
132
  output_path=str(tmp_path / "f.parquet"),
133
- version="bregen",
133
+ version="some-version",
134
134
  )
135
135
  assert out == str(tmp_path / "f.parquet")
136
136
  paths = [r.url.path for r in httpx_mock.get_requests()]
@@ -0,0 +1,12 @@
1
+ import sys
2
+
3
+ from everestapi.mcp.server import _client_env
4
+
5
+
6
+ def test_client_env_targets_this_interpreter():
7
+ """The upgrade command must point at the running venv, not bare `pip`."""
8
+ env = _client_env()
9
+ assert env["python"] == sys.executable
10
+ assert sys.executable in env["upgrade_command"]
11
+ assert "everestapi[mcp]" in env["upgrade_command"]
12
+ assert env["everestapi_version"] # resolved, non-empty
@@ -41,7 +41,21 @@ def preds_file(tmp_path):
41
41
  return str(p)
42
42
 
43
43
 
44
+ # EVE-1159: submit_validation_diagnostics now tries the presigned direct-to-CDN
45
+ # path first. These helpers drive the two branches: fall back to multipart when
46
+ # the CDN is unavailable, or take the presigned path when it is.
47
+ _PRESIGN_URL = "http://test/api/v1/diagnostics/upload/presign"
48
+ _COMPLETE_URL = "http://test/api/v1/diagnostics/upload/complete"
49
+ _UPLOAD_URL = "http://test/api/v1/diagnostics/upload"
50
+
51
+
52
+ def _presign_unavailable(httpx_mock):
53
+ """Make /presign report the CDN is off so the client falls back to multipart."""
54
+ httpx_mock.add_response(method="POST", url=_PRESIGN_URL, json={"available": False})
55
+
56
+
44
57
  def test_submit_no_wait_returns_accept(api, httpx_mock, preds_file):
58
+ _presign_unavailable(httpx_mock) # → multipart fallback
45
59
  httpx_mock.add_response(
46
60
  method="POST",
47
61
  url="http://test/api/v1/diagnostics/upload",
@@ -53,7 +67,7 @@ def test_submit_no_wait_returns_accept(api, httpx_mock, preds_file):
53
67
  assert result["status"] == "pending"
54
68
 
55
69
  # Multipart body carries the file part plus the three form fields.
56
- body = httpx_mock.get_request().content.decode("latin-1")
70
+ body = httpx_mock.get_request(url=_UPLOAD_URL).content.decode("latin-1")
57
71
  assert 'name="file"' in body
58
72
  assert 'name="model_id"' in body
59
73
  assert "m1" in body
@@ -62,6 +76,7 @@ def test_submit_no_wait_returns_accept(api, httpx_mock, preds_file):
62
76
 
63
77
 
64
78
  def test_submit_wait_polls_until_done(api, httpx_mock, preds_file):
79
+ _presign_unavailable(httpx_mock)
65
80
  httpx_mock.add_response(
66
81
  method="POST",
67
82
  url="http://test/api/v1/diagnostics/upload",
@@ -86,6 +101,7 @@ def test_submit_wait_polls_until_done(api, httpx_mock, preds_file):
86
101
 
87
102
 
88
103
  def test_submit_wait_raises_on_failed(api, httpx_mock, preds_file):
104
+ _presign_unavailable(httpx_mock)
89
105
  httpx_mock.add_response(
90
106
  method="POST",
91
107
  url="http://test/api/v1/diagnostics/upload",
@@ -106,6 +122,7 @@ def test_submit_wait_raises_on_failed(api, httpx_mock, preds_file):
106
122
 
107
123
 
108
124
  def test_submit_wait_raises_on_timeout(api, httpx_mock, preds_file):
125
+ _presign_unavailable(httpx_mock)
109
126
  httpx_mock.add_response(
110
127
  method="POST",
111
128
  url="http://test/api/v1/diagnostics/upload",
@@ -171,6 +188,7 @@ def test_format_tables_render_dashes_for_none():
171
188
 
172
189
 
173
190
  def test_submit_with_model_pkl_bytes_adds_part(api, httpx_mock, preds_file):
191
+ _presign_unavailable(httpx_mock)
174
192
  httpx_mock.add_response(
175
193
  method="POST",
176
194
  url="http://test/api/v1/diagnostics/upload",
@@ -180,7 +198,7 @@ def test_submit_with_model_pkl_bytes_adds_part(api, httpx_mock, preds_file):
180
198
  api.submit_validation_diagnostics(
181
199
  model_id="m1", predictions=preds_file, model_pkl=b"\x80\x05fakepkl."
182
200
  )
183
- body = httpx_mock.get_request().content.decode("latin-1")
201
+ body = httpx_mock.get_request(url=_UPLOAD_URL).content.decode("latin-1")
184
202
  assert 'name="model_pkl"' in body
185
203
  assert 'filename="model.pkl"' in body
186
204
 
@@ -188,6 +206,7 @@ def test_submit_with_model_pkl_bytes_adds_part(api, httpx_mock, preds_file):
188
206
  def test_submit_with_model_pkl_path_adds_part(api, httpx_mock, preds_file, tmp_path):
189
207
  pkl = tmp_path / "my-model.pkl"
190
208
  pkl.write_bytes(b"\x80\x05fakepkl.")
209
+ _presign_unavailable(httpx_mock)
191
210
  httpx_mock.add_response(
192
211
  method="POST",
193
212
  url="http://test/api/v1/diagnostics/upload",
@@ -195,12 +214,13 @@ def test_submit_with_model_pkl_path_adds_part(api, httpx_mock, preds_file, tmp_p
195
214
  json={"upload_id": "u3", "status": "pending"},
196
215
  )
197
216
  api.submit_validation_diagnostics(model_id="m1", predictions=preds_file, model_pkl=str(pkl))
198
- body = httpx_mock.get_request().content.decode("latin-1")
217
+ body = httpx_mock.get_request(url=_UPLOAD_URL).content.decode("latin-1")
199
218
  assert 'name="model_pkl"' in body
200
219
  assert 'filename="my-model.pkl"' in body
201
220
 
202
221
 
203
222
  def test_submit_without_model_pkl_omits_part(api, httpx_mock, preds_file):
223
+ _presign_unavailable(httpx_mock)
204
224
  httpx_mock.add_response(
205
225
  method="POST",
206
226
  url="http://test/api/v1/diagnostics/upload",
@@ -208,10 +228,89 @@ def test_submit_without_model_pkl_omits_part(api, httpx_mock, preds_file):
208
228
  json={"upload_id": "u4", "status": "pending"},
209
229
  )
210
230
  api.submit_validation_diagnostics(model_id="m1", predictions=preds_file)
211
- body = httpx_mock.get_request().content.decode("latin-1")
231
+ body = httpx_mock.get_request(url=_UPLOAD_URL).content.decode("latin-1")
212
232
  assert 'name="model_pkl"' not in body
213
233
 
214
234
 
235
+ # EVE-1159: the presigned direct-to-CDN path (the new default when available).
236
+ def test_submit_presigned_path_uploads_to_cdn_then_completes(api, httpx_mock, preds_file):
237
+ httpx_mock.add_response(
238
+ method="POST",
239
+ url=_PRESIGN_URL,
240
+ json={
241
+ "available": True,
242
+ "upload_id": "up9",
243
+ "model_id": "m1",
244
+ "cdn_base": "http://cdn.test",
245
+ "predictions": {
246
+ "url": "http://cdn.test/",
247
+ "fields": {"key": "Hackathon/a/m/up9.csv", "policy": "p", "x-amz-signature": "s"},
248
+ "key": "Hackathon/a/m/up9.csv",
249
+ },
250
+ "model_pkl": None,
251
+ "expires_in": 900,
252
+ },
253
+ )
254
+ # The direct-to-S3 POST goes to the CDN host (204 = S3 success).
255
+ httpx_mock.add_response(method="POST", url="http://cdn.test/", status_code=204)
256
+ httpx_mock.add_response(
257
+ method="POST",
258
+ url=_COMPLETE_URL,
259
+ status_code=202,
260
+ json={"upload_id": "up9", "status": "pending", "poll_url": "/api/v1/diagnostics/runs/up9"},
261
+ )
262
+ result = api.submit_validation_diagnostics(model_id="m1", predictions=preds_file)
263
+ assert result["upload_id"] == "up9"
264
+ # The big file went to the CDN, never multipart to the API.
265
+ cdn_body = httpx_mock.get_request(url="http://cdn.test/").content.decode("latin-1")
266
+ assert 'name="file"' in cdn_body
267
+ assert 'name="key"' in cdn_body # presigned form field present
268
+ # complete carried the upload_id, not the file bytes.
269
+ import json as _json
270
+
271
+ complete_body = _json.loads(httpx_mock.get_request(url=_COMPLETE_URL).content)
272
+ assert complete_body["upload_id"] == "up9"
273
+
274
+
275
+ def test_submit_presigned_uploads_pkl_when_present(api, httpx_mock, preds_file):
276
+ httpx_mock.add_response(
277
+ method="POST",
278
+ url=_PRESIGN_URL,
279
+ json={
280
+ "available": True,
281
+ "upload_id": "up10",
282
+ "model_id": "m1",
283
+ "cdn_base": "http://cdn.test",
284
+ "predictions": {
285
+ "url": "http://cdn.test/",
286
+ "fields": {"key": "Hackathon/a/m/up10.csv"},
287
+ "key": "Hackathon/a/m/up10.csv",
288
+ },
289
+ "model_pkl": {
290
+ "url": "http://cdn.test/",
291
+ "fields": {"key": "uploads/a/m/up10.pkl"},
292
+ "key": "uploads/a/m/up10.pkl",
293
+ },
294
+ "expires_in": 900,
295
+ },
296
+ )
297
+ # Two CDN POSTs: predictions + pkl.
298
+ httpx_mock.add_response(method="POST", url="http://cdn.test/", status_code=204)
299
+ httpx_mock.add_response(method="POST", url="http://cdn.test/", status_code=204)
300
+ httpx_mock.add_response(
301
+ method="POST",
302
+ url=_COMPLETE_URL,
303
+ status_code=202,
304
+ json={"upload_id": "up10", "status": "pending"},
305
+ )
306
+ result = api.submit_validation_diagnostics(
307
+ model_id="m1", predictions=preds_file, model_pkl=b"\x80\x05fakepkl."
308
+ )
309
+ assert result["upload_id"] == "up10"
310
+ cdn_posts = [r for r in httpx_mock.get_requests() if str(r.url) == "http://cdn.test/"]
311
+ assert len(cdn_posts) == 2
312
+
313
+
215
314
  def test_leaderboard_window_and_offset_params(api, httpx_mock):
216
315
  httpx_mock.add_response(
217
316
  method="GET",
@@ -17,4 +17,4 @@ def test_train_gpu_enum_includes_cpu():
17
17
 
18
18
 
19
19
  def test_version_bumped():
20
- assert everestapi.__version__ == "0.3.2"
20
+ assert everestapi.__version__ == "0.3.6"
@@ -32,6 +32,23 @@ def test_toolset_coverage_both_directions():
32
32
  assert len(grouped) == len(grouped_set), "a tool name appears in more than one toolset"
33
33
 
34
34
 
35
+ def test_job_lifecycle_tools_are_core():
36
+ """Regression: the hosted-train lifecycle tools must be discoverable in the
37
+ DEFAULT toolset. train's own description points the caller at get_job_status/
38
+ get_job_log/get_model_download_url/get_job_predictions_url, so hiding them in
39
+ a non-default group made hosted training an unfinishable dead end — the very
40
+ gap that the everestapi package lagged eiq-platform (EVE-1145) on."""
41
+ core = S.TOOLSETS["core"]
42
+ for name in (
43
+ "get_job_status",
44
+ "get_job_log",
45
+ "list_compute_jobs",
46
+ "get_model_download_url",
47
+ "get_job_predictions_url",
48
+ ):
49
+ assert name in core, f"{name} must be in core (default toolset) to finish a train job"
50
+
51
+
35
52
  def test_default_advertises_core_only(monkeypatch):
36
53
  monkeypatch.delenv("EIQ_MCP_TOOLSETS", raising=False)
37
54
  assert S._enabled_tool_names() == set(S.TOOLSETS["core"])
@@ -79,16 +79,16 @@ def test_get_models(httpx_mock, api):
79
79
  # via the futures endpoint, EVE-931 — see test_client.py).
80
80
  def test_download_dataset_uses_versioned_route(httpx_mock, api, tmp_path):
81
81
  httpx_mock.add_response(
82
- url="http://test/api/v1/data/download/bregen/equities/train",
82
+ url="http://test/api/v1/data/download/test-version/equities/train",
83
83
  content=b"PAR1data",
84
84
  )
85
85
  out = api.download_dataset(
86
86
  universe="equities",
87
87
  split="train",
88
- version="bregen",
88
+ version="test-version",
89
89
  output_path=str(tmp_path / "t.parquet"),
90
90
  )
91
- assert httpx_mock.get_request().url.path == "/api/v1/data/download/bregen/equities/train"
91
+ assert httpx_mock.get_request().url.path == "/api/v1/data/download/test-version/equities/train"
92
92
  with open(out, "rb") as f:
93
93
  assert f.read() == b"PAR1data"
94
94
 
@@ -96,10 +96,10 @@ def test_download_dataset_uses_versioned_route(httpx_mock, api, tmp_path):
96
96
  def test_download_dataset_latest_resolves_version(httpx_mock, api, tmp_path):
97
97
  httpx_mock.add_response(
98
98
  url="http://test/api/v1/data/versions",
99
- json={"versions": [{"version": "bregen", "universes": ["equities"]}]},
99
+ json={"versions": [{"version": "test-version", "universes": ["equities"]}]},
100
100
  )
101
101
  httpx_mock.add_response(
102
- url="http://test/api/v1/data/download/bregen/equities/live",
102
+ url="http://test/api/v1/data/download/test-version/equities/live",
103
103
  content=b"PAR1live",
104
104
  )
105
105
  out = api.download_dataset(
@@ -159,6 +159,14 @@ def test_mcp_tools_include_get_started():
159
159
  assert "get_started" in names
160
160
 
161
161
 
162
+ def test_mcp_instructions_point_at_example_scripts():
163
+ # The on-connect banner directs agents to the example-scripts repo (mirrors
164
+ # the eiq-platform hosted server + get_started payload).
165
+ from everestapi.mcp import server
166
+
167
+ assert "https://github.com/everestquant/example-scripts" in server._MCP_INSTRUCTIONS
168
+
169
+
162
170
  def test_mcp_dispatch_get_started_uses_client(monkeypatch):
163
171
  from everestapi.mcp import server
164
172
 
@@ -168,7 +176,10 @@ def test_mcp_dispatch_get_started_uses_client(monkeypatch):
168
176
 
169
177
  monkeypatch.setattr(server, "_client", FakeClient())
170
178
  out = json.loads(server._dispatch("get_started", {}))
171
- assert out == {"mode": "diagnostics_hackathon"}
179
+ assert out["mode"] == "diagnostics_hackathon"
180
+ # EVE-1163: dispatch augments the client payload with the running-interpreter
181
+ # env so a self-updating agent upgrades the venv the MCP actually runs from.
182
+ assert "client_env" in out
172
183
 
173
184
 
174
185
  def test_mcp_tools_include_get_status():
@@ -327,7 +338,7 @@ def test_enrich_next_actions_are_result_conditional():
327
338
  _, running = server._enrich_result(
328
339
  "get_job_status", {"job_id": "j"}, json.dumps({"status": "running"})
329
340
  )
330
- assert running["next_actions"] == ["get_job_status"]
341
+ assert running["next_actions"] == ["get_job_status", "get_job_log"]
331
342
  _, done = server._enrich_result(
332
343
  "get_job_status", {"job_id": "j"}, json.dumps({"status": "completed"})
333
344
  )
File without changes
File without changes
File without changes
File without changes
File without changes