daishi 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,3 @@
1
+ __pycache__/
2
+ dist/
3
+ *.egg-info/
daishi-0.1.0/LICENSE ADDED
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Arkeous LLC
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
daishi-0.1.0/PKG-INFO ADDED
@@ -0,0 +1,88 @@
1
+ Metadata-Version: 2.5
2
+ Name: daishi
3
+ Version: 0.1.0
4
+ Summary: Typed client for the Daishi Studio API: scenarios, runs, series batches and paired experiments, with retries and waiting helpers.
5
+ Project-URL: Homepage, https://daishi.ai
6
+ Project-URL: Documentation, https://daishi.ai/docs/developer
7
+ Author: Arkeous LLC
8
+ License-Expression: MIT
9
+ License-File: LICENSE
10
+ Keywords: agents,ai,benchmark,daishi,evaluation,llm,multi-agent
11
+ Classifier: Development Status :: 3 - Alpha
12
+ Classifier: Intended Audience :: Developers
13
+ Classifier: Intended Audience :: Science/Research
14
+ Classifier: Operating System :: OS Independent
15
+ Classifier: Programming Language :: Python :: 3
16
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
17
+ Classifier: Typing :: Typed
18
+ Requires-Python: >=3.10
19
+ Description-Content-Type: text/markdown
20
+
21
+ # daishi
22
+
23
+ Typed client for the Studio API of [Daishi](https://daishi.ai), an independent agent testing and evaluation platform: save scenarios, launch runs, series batches and paired experiments on your own model keys, and read the scored results.
24
+
25
+ ```bash
26
+ pip install daishi
27
+ ```
28
+
29
+ Python 3.10 or later. No dependencies: the standard library only.
30
+
31
+ ## Quick start
32
+
33
+ In the [Studio](https://daishi.ai/studio), open **Account > Developer**, create an access token with **runs: launch and cancel** ticked, and export it:
34
+
35
+ ```bash
36
+ export DAISHI_TOKEN=dsk_... # macOS, Linux
37
+ $env:DAISHI_TOKEN = "dsk_..." # Windows PowerShell
38
+ ```
39
+
40
+ ```python
41
+ from daishi import Daishi
42
+
43
+ daishi = Daishi() # reads DAISHI_TOKEN
44
+
45
+ run = daishi.launch({
46
+ "scenario_id": "daishi:famine-v1",
47
+ "roster": [{"model": "<provider>/<model>"}, {"model": "<provider>/<model>"}],
48
+ "max_spend_usd": 5,
49
+ })["run"]
50
+
51
+ done = daishi.wait_for_run(run["run_id"])
52
+ for agent in (done["results"] or {}).get("agents", []):
53
+ print(agent["name"], agent.get("fitness_index"), agent.get("grade"))
54
+ ```
55
+
56
+ Seats play on the provider keys stored on your account. Every plan can read and validate; creating, launching and cancelling need a paid plan.
57
+
58
+ ## What it does
59
+
60
+ - **One method per API operation**, named after its operation id: `get_run`, `list_runs`, `launch`, `cancel`, `get_batch`, `create_scenario`, `create_experiment`, `get_experiment` and the rest. Responses are plain dicts typed as `TypedDict`s generated from the API's OpenAPI document, so editors and type checkers know every field.
61
+ - **Errors** raise `DaishiError` with the API's `code` (`unknown_run`, `plan_quota`, ...), `status`, `message`, `retry_after_seconds`, and `docs`, a link to what the code means.
62
+ - **Retries** are safe by construction. A 429 is retried after the seconds its `Retry-After` header names, and a cancel refused with `run_launching` after a pause, on any method, because the server refused before doing anything. A gateway error or a dropped connection is retried on GET only, so a write is never sent twice.
63
+ - **Waiting.** `wait_for_run`, `wait_for_batch` and `wait_for_experiment` poll every 30 seconds until the run, batch or experiment ends. `wait_for_run` also waits for a finished run's results, which arrive when the match archives. Pass `timeout=` (seconds) for a deadline; it raises `TimeoutError`. To be told instead of asking, set up [run-end notices](https://daishi.ai/docs/developer/runs#run-end-notices).
64
+
65
+ ```python
66
+ from daishi import Daishi, DaishiError
67
+
68
+ try:
69
+ daishi.get_run("run_...")
70
+ except DaishiError as err:
71
+ if err.code != "unknown_run":
72
+ raise
73
+ print(err.docs)
74
+ ```
75
+
76
+ ## Options
77
+
78
+ `Daishi(token=None, *, base_url="https://daishi.ai/api/v1", max_retries=3, timeout=60.0)`. `token` defaults to `DAISHI_TOKEN`; `max_retries=0` turns retrying off; `timeout` is seconds per attempt.
79
+
80
+ ## Types
81
+
82
+ Every schema in the API is importable: `Run`, `RunResults`, `Scenario`, `Skill`, `BatchReport`, `Experiment`, `ExperimentReport`, the request bodies (`RunBody`, `ScenarioBody`, `ExperimentBody`, ...), and `ErrorCode`. Fields whose shape follows the scenario, the provider or the meter are `JsonObject` (`dict[str, Any]`). A few operations (`me`, `usage`, the estimates, logs and invites) answer `JsonObject` until the API types them; `run_log` answers the log's NDJSON text.
83
+
84
+ ## Docs
85
+
86
+ [Developer docs](https://daishi.ai/docs/developer) and the [API reference](https://daishi.ai/docs/developer/rest).
87
+
88
+ MIT licensed.
daishi-0.1.0/README.md ADDED
@@ -0,0 +1,68 @@
1
+ # daishi
2
+
3
+ Typed client for the Studio API of [Daishi](https://daishi.ai), an independent agent testing and evaluation platform: save scenarios, launch runs, series batches and paired experiments on your own model keys, and read the scored results.
4
+
5
+ ```bash
6
+ pip install daishi
7
+ ```
8
+
9
+ Python 3.10 or later. No dependencies: the standard library only.
10
+
11
+ ## Quick start
12
+
13
+ In the [Studio](https://daishi.ai/studio), open **Account > Developer**, create an access token with **runs: launch and cancel** ticked, and export it:
14
+
15
+ ```bash
16
+ export DAISHI_TOKEN=dsk_... # macOS, Linux
17
+ $env:DAISHI_TOKEN = "dsk_..." # Windows PowerShell
18
+ ```
19
+
20
+ ```python
21
+ from daishi import Daishi
22
+
23
+ daishi = Daishi() # reads DAISHI_TOKEN
24
+
25
+ run = daishi.launch({
26
+ "scenario_id": "daishi:famine-v1",
27
+ "roster": [{"model": "<provider>/<model>"}, {"model": "<provider>/<model>"}],
28
+ "max_spend_usd": 5,
29
+ })["run"]
30
+
31
+ done = daishi.wait_for_run(run["run_id"])
32
+ for agent in (done["results"] or {}).get("agents", []):
33
+ print(agent["name"], agent.get("fitness_index"), agent.get("grade"))
34
+ ```
35
+
36
+ Seats play on the provider keys stored on your account. Every plan can read and validate; creating, launching and cancelling need a paid plan.
37
+
38
+ ## What it does
39
+
40
+ - **One method per API operation**, named after its operation id: `get_run`, `list_runs`, `launch`, `cancel`, `get_batch`, `create_scenario`, `create_experiment`, `get_experiment` and the rest. Responses are plain dicts typed as `TypedDict`s generated from the API's OpenAPI document, so editors and type checkers know every field.
41
+ - **Errors** raise `DaishiError` with the API's `code` (`unknown_run`, `plan_quota`, ...), `status`, `message`, `retry_after_seconds`, and `docs`, a link to what the code means.
42
+ - **Retries** are safe by construction. A 429 is retried after the seconds its `Retry-After` header names, and a cancel refused with `run_launching` after a pause, on any method, because the server refused before doing anything. A gateway error or a dropped connection is retried on GET only, so a write is never sent twice.
43
+ - **Waiting.** `wait_for_run`, `wait_for_batch` and `wait_for_experiment` poll every 30 seconds until the run, batch or experiment ends. `wait_for_run` also waits for a finished run's results, which arrive when the match archives. Pass `timeout=` (seconds) for a deadline; it raises `TimeoutError`. To be told instead of asking, set up [run-end notices](https://daishi.ai/docs/developer/runs#run-end-notices).
44
+
45
+ ```python
46
+ from daishi import Daishi, DaishiError
47
+
48
+ try:
49
+ daishi.get_run("run_...")
50
+ except DaishiError as err:
51
+ if err.code != "unknown_run":
52
+ raise
53
+ print(err.docs)
54
+ ```
55
+
56
+ ## Options
57
+
58
+ `Daishi(token=None, *, base_url="https://daishi.ai/api/v1", max_retries=3, timeout=60.0)`. `token` defaults to `DAISHI_TOKEN`; `max_retries=0` turns retrying off; `timeout` is seconds per attempt.
59
+
60
+ ## Types
61
+
62
+ Every schema in the API is importable: `Run`, `RunResults`, `Scenario`, `Skill`, `BatchReport`, `Experiment`, `ExperimentReport`, the request bodies (`RunBody`, `ScenarioBody`, `ExperimentBody`, ...), and `ErrorCode`. Fields whose shape follows the scenario, the provider or the meter are `JsonObject` (`dict[str, Any]`). A few operations (`me`, `usage`, the estimates, logs and invites) answer `JsonObject` until the API types them; `run_log` answers the log's NDJSON text.
63
+
64
+ ## Docs
65
+
66
+ [Developer docs](https://daishi.ai/docs/developer) and the [API reference](https://daishi.ai/docs/developer/rest).
67
+
68
+ MIT licensed.
@@ -0,0 +1,34 @@
1
+ [build-system]
2
+ requires = ["hatchling>=1.27"]
3
+ build-backend = "hatchling.build"
4
+
5
+ [project]
6
+ name = "daishi"
7
+ version = "0.1.0"
8
+ description = "Typed client for the Daishi Studio API: scenarios, runs, series batches and paired experiments, with retries and waiting helpers."
9
+ readme = "README.md"
10
+ license = "MIT"
11
+ license-files = ["LICENSE"]
12
+ authors = [{ name = "Arkeous LLC" }]
13
+ requires-python = ">=3.10"
14
+ dependencies = []
15
+ keywords = ["daishi", "ai", "agents", "evaluation", "benchmark", "llm", "multi-agent"]
16
+ classifiers = [
17
+ "Development Status :: 3 - Alpha",
18
+ "Intended Audience :: Developers",
19
+ "Intended Audience :: Science/Research",
20
+ "Operating System :: OS Independent",
21
+ "Programming Language :: Python :: 3",
22
+ "Topic :: Scientific/Engineering :: Artificial Intelligence",
23
+ "Typing :: Typed",
24
+ ]
25
+
26
+ [project.urls]
27
+ Homepage = "https://daishi.ai"
28
+ Documentation = "https://daishi.ai/docs/developer"
29
+
30
+ [tool.hatch.build.targets.wheel]
31
+ packages = ["src/daishi"]
32
+
33
+ [tool.hatch.build.targets.sdist]
34
+ include = ["src/daishi", "README.md", "LICENSE", "pyproject.toml"]
@@ -0,0 +1,18 @@
1
+ """Typed client for the Daishi Studio API.
2
+
3
+ from daishi import Daishi
4
+
5
+ client = Daishi() # reads DAISHI_TOKEN
6
+ run = client.launch({"scenario_id": "daishi:famine-v1", "roster": [{"model": "<provider>/<model>"}]})["run"]
7
+ done = client.wait_for_run(run["run_id"])
8
+
9
+ Docs: https://daishi.ai/docs/developer
10
+ """
11
+
12
+ from . import _generated
13
+ from ._client import DEFAULT_BASE_URL, VERSION, Daishi, DaishiError
14
+ from ._generated import * # noqa: F401,F403 (the API's types)
15
+
16
+ __version__ = VERSION
17
+
18
+ __all__ = ["Daishi", "DaishiError", "DEFAULT_BASE_URL", "VERSION", *_generated.__all__]
@@ -0,0 +1,174 @@
1
+ from __future__ import annotations
2
+
3
+ import email.utils
4
+ import json
5
+ import os
6
+ import random
7
+ import time
8
+ import urllib.error
9
+ import urllib.parse
10
+ import urllib.request
11
+ from typing import Any, Callable, Dict, Optional, TypeVar
12
+
13
+ from ._generated import GetBatchResponse, GetExperimentResponse, GetRunResponse, _Operations
14
+
15
+ __all__ = ["Daishi", "DaishiError", "DEFAULT_BASE_URL", "VERSION"]
16
+
17
+ VERSION = "0.1.0"
18
+ DEFAULT_BASE_URL = "https://daishi.ai/api/v1"
19
+
20
+ _RUN_DONE = {"finished", "failed", "cancelled"}
21
+ _EXPERIMENT_DONE = {"finished", "cancelled"}
22
+ _GATEWAY = {502, 503, 504}
23
+
24
+ T = TypeVar("T")
25
+
26
+
27
+ class DaishiError(Exception):
28
+ """A refusal from the API, or a transport failure after the retries ran out."""
29
+
30
+ def __init__(self, status: int, code: str, message: str, retry_after_seconds: Optional[float] = None, body: Any = None) -> None:
31
+ super().__init__(message)
32
+ #: HTTP status; 0 when no response arrived.
33
+ self.status = status
34
+ #: The API's error code (``unknown_run``, ``rate_limited``, ...), or ``http_<status>`` when the body carried none.
35
+ self.code = code
36
+ self.message = message
37
+ #: Seconds the ``Retry-After`` header asked for, when it sent one.
38
+ self.retry_after_seconds = retry_after_seconds
39
+ #: The parsed error body, when there was one.
40
+ self.body = body
41
+
42
+ @property
43
+ def docs(self) -> str:
44
+ """Where the code is explained."""
45
+ return f"https://daishi.ai/docs/developer/errors#{self.code}"
46
+
47
+ def __repr__(self) -> str:
48
+ return f"DaishiError(status={self.status}, code={self.code!r}, message={self.message!r})"
49
+
50
+
51
+ class Daishi(_Operations):
52
+ """The Daishi Studio API.
53
+
54
+ Every operation is a method (``get_run``, ``launch``, ``create_scenario``, ...)
55
+ typed from the API's OpenAPI document; the ``wait_for_*`` helpers poll until
56
+ a run, batch or experiment ends.
57
+
58
+ Retries: a 429 is retried after the seconds its ``Retry-After`` names, and a
59
+ ``run_launching`` refusal after a short pause, on any method, because the
60
+ server refused before doing anything. A gateway error (502, 503, 504) or a
61
+ dropped connection is retried on GET only: a write that may have landed is
62
+ never sent twice.
63
+
64
+ :param token: A Studio access token (``dsk_...``). Default: the ``DAISHI_TOKEN`` environment variable.
65
+ :param base_url: Default ``https://daishi.ai/api/v1``.
66
+ :param max_retries: Retries after a refusal that is safe to repeat. Default 3; 0 turns retrying off.
67
+ :param timeout: Per-attempt timeout in seconds. Default 60.
68
+ """
69
+
70
+ def __init__(self, token: Optional[str] = None, *, base_url: str = DEFAULT_BASE_URL, max_retries: int = 3, timeout: float = 60.0) -> None:
71
+ self.token = token if token is not None else (os.environ.get("DAISHI_TOKEN") or None)
72
+ self.base_url = base_url.rstrip("/")
73
+ self.max_retries = max_retries
74
+ self.timeout = timeout
75
+
76
+ def __repr__(self) -> str:
77
+ return f"Daishi(base_url={self.base_url!r})"
78
+
79
+ def _request(self, method: str, path: str, *, query: Optional[Dict[str, Any]] = None, body: Any = None, text: bool = False) -> Any:
80
+ url = self.base_url + ("" if path == "/" else path)
81
+ params = {k: v for k, v in (query or {}).items() if v is not None}
82
+ if params:
83
+ url += "?" + urllib.parse.urlencode(params)
84
+ headers = {
85
+ "Accept": "application/x-ndjson, application/json" if text else "application/json",
86
+ "User-Agent": f"daishi-python/{VERSION}",
87
+ }
88
+ if self.token:
89
+ headers["Authorization"] = f"Bearer {self.token}"
90
+ data = None
91
+ if body is not None:
92
+ headers["Content-Type"] = "application/json"
93
+ data = json.dumps(body).encode("utf-8")
94
+
95
+ attempt = 0
96
+ while True:
97
+ can_retry = attempt < self.max_retries
98
+ req = urllib.request.Request(url, data=data, headers=headers, method=method)
99
+ try:
100
+ with urllib.request.urlopen(req, timeout=self.timeout) as res:
101
+ raw = res.read().decode("utf-8")
102
+ return raw if text else json.loads(raw)
103
+ except urllib.error.HTTPError as err:
104
+ raw = err.read().decode("utf-8", "replace")
105
+ try:
106
+ parsed = json.loads(raw)
107
+ except ValueError:
108
+ parsed = None
109
+ parsed_obj = parsed if isinstance(parsed, dict) else {}
110
+ code = parsed_obj.get("error") or f"http_{err.code}"
111
+ retry_after = _parse_retry_after(err.headers.get("Retry-After") if err.headers else None)
112
+ retryable = err.code == 429 or code == "run_launching" or (method == "GET" and err.code in _GATEWAY)
113
+ if retryable and can_retry:
114
+ time.sleep(retry_after if retry_after is not None else _backoff(attempt))
115
+ attempt += 1
116
+ continue
117
+ message = parsed_obj.get("message") or raw[:200] or err.reason
118
+ raise DaishiError(err.code, code, str(message), retry_after, parsed) from None
119
+ except (urllib.error.URLError, TimeoutError, ConnectionError) as err:
120
+ if method == "GET" and can_retry:
121
+ time.sleep(_backoff(attempt))
122
+ attempt += 1
123
+ continue
124
+ reason = getattr(err, "reason", err)
125
+ raise DaishiError(0, "network_error", f"{method} {urllib.parse.urlsplit(url).path} failed: {reason}") from err
126
+
127
+ def wait_for_run(self, id: str, *, interval: float = 30.0, timeout: Optional[float] = None, on_poll: Optional[Callable[[GetRunResponse], None]] = None) -> GetRunResponse:
128
+ """Poll a run until it ends (finished, failed or cancelled) and return the last read.
129
+
130
+ A finished run is read until its ``results`` arrive, which is when the match archives.
131
+ Raises ``TimeoutError`` if ``timeout`` seconds pass first.
132
+ """
133
+ return self._poll(lambda: self.get_run(id), lambda r: r["run"]["status"] in _RUN_DONE and (r["run"]["status"] != "finished" or r["results"] is not None), interval, timeout, on_poll)
134
+
135
+ def wait_for_batch(self, id: str, *, interval: float = 30.0, timeout: Optional[float] = None, on_poll: Optional[Callable[[GetBatchResponse], None]] = None) -> GetBatchResponse:
136
+ """Poll a series batch until no trial is still to play."""
137
+ return self._poll(lambda: self.get_batch(id), lambda r: r["batch"]["pending"] == 0, interval, timeout, on_poll)
138
+
139
+ def wait_for_experiment(self, id: str, *, interval: float = 30.0, timeout: Optional[float] = None, on_poll: Optional[Callable[[GetExperimentResponse], None]] = None) -> GetExperimentResponse:
140
+ """Poll a paired experiment until it finishes or is cancelled."""
141
+ return self._poll(lambda: self.get_experiment(id), lambda r: r["experiment"]["status"] in _EXPERIMENT_DONE, interval, timeout, on_poll)
142
+
143
+ def _poll(self, read: Callable[[], T], done: Callable[[T], bool], interval: float, timeout: Optional[float], on_poll: Optional[Callable[[T], None]]) -> T:
144
+ deadline = None if timeout is None else time.monotonic() + timeout
145
+ while True:
146
+ answer = read()
147
+ if on_poll:
148
+ on_poll(answer)
149
+ if done(answer):
150
+ return answer
151
+ if deadline is not None and time.monotonic() + interval > deadline:
152
+ raise TimeoutError(f"still not done after {timeout} seconds")
153
+ time.sleep(interval)
154
+
155
+
156
+ def _parse_retry_after(header: Optional[str]) -> Optional[float]:
157
+ if not header:
158
+ return None
159
+ try:
160
+ seconds = float(header)
161
+ return seconds if seconds >= 0 else None
162
+ except ValueError:
163
+ pass
164
+ try:
165
+ at = email.utils.parsedate_to_datetime(header)
166
+ except (TypeError, ValueError):
167
+ return None
168
+ return max(0.0, at.timestamp() - time.time())
169
+
170
+
171
+ def _backoff(attempt: int) -> float:
172
+ """1 s, 2 s, 4 s ... with up to a quarter of jitter, capped at 30 s."""
173
+ base = min(30.0, 2.0**attempt)
174
+ return base + random.random() * base * 0.25