bonito-cli 0.9.2__tar.gz → 0.9.4__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- bonito_cli-0.9.4/.gitignore +32 -0
- {bonito_cli-0.9.2 → bonito_cli-0.9.4}/CHANGELOG.md +40 -0
- {bonito_cli-0.9.2 → bonito_cli-0.9.4}/PKG-INFO +2 -2
- {bonito_cli-0.9.2 → bonito_cli-0.9.4}/bonito_cli/api.py +36 -0
- {bonito_cli-0.9.2 → bonito_cli-0.9.4}/bonito_cli/app.py +6 -0
- {bonito_cli-0.9.2 → bonito_cli-0.9.4}/bonito_cli/commands/agents.py +108 -0
- bonito_cli-0.9.4/bonito_cli/commands/artifacts.py +145 -0
- bonito_cli-0.9.4/bonito_cli/commands/evals.py +183 -0
- {bonito_cli-0.9.2 → bonito_cli-0.9.4}/bonito_cli/commands/gateway.py +168 -0
- {bonito_cli-0.9.2 → bonito_cli-0.9.4}/bonito_cli/commands/policies.py +42 -5
- {bonito_cli-0.9.2 → bonito_cli-0.9.4}/bonito_cli/commands/providers.py +18 -1
- bonito_cli-0.9.4/bonito_cli/commands/routing.py +486 -0
- bonito_cli-0.9.4/bonito_cli/commands/skills.py +163 -0
- {bonito_cli-0.9.2 → bonito_cli-0.9.4}/pyproject.toml +1 -1
- bonito_cli-0.9.2/.gitignore +0 -15
- bonito_cli-0.9.2/bonito_cli/commands/routing.py +0 -178
- {bonito_cli-0.9.2 → bonito_cli-0.9.4}/CLI_SPEC.md +0 -0
- {bonito_cli-0.9.2 → bonito_cli-0.9.4}/README.md +0 -0
- {bonito_cli-0.9.2 → bonito_cli-0.9.4}/bonito_cli/__init__.py +0 -0
- {bonito_cli-0.9.2 → bonito_cli-0.9.4}/bonito_cli/__main__.py +0 -0
- {bonito_cli-0.9.2 → bonito_cli-0.9.4}/bonito_cli/commands/__init__.py +0 -0
- {bonito_cli-0.9.2 → bonito_cli-0.9.4}/bonito_cli/commands/admin.py +0 -0
- {bonito_cli-0.9.2 → bonito_cli-0.9.4}/bonito_cli/commands/analytics.py +0 -0
- {bonito_cli-0.9.2 → bonito_cli-0.9.4}/bonito_cli/commands/approval.py +0 -0
- {bonito_cli-0.9.2 → bonito_cli-0.9.4}/bonito_cli/commands/audit.py +0 -0
- {bonito_cli-0.9.2 → bonito_cli-0.9.4}/bonito_cli/commands/auth.py +0 -0
- {bonito_cli-0.9.2 → bonito_cli-0.9.4}/bonito_cli/commands/chat.py +0 -0
- {bonito_cli-0.9.2 → bonito_cli-0.9.4}/bonito_cli/commands/compliance.py +0 -0
- {bonito_cli-0.9.2 → bonito_cli-0.9.4}/bonito_cli/commands/deploy.py +0 -0
- {bonito_cli-0.9.2 → bonito_cli-0.9.4}/bonito_cli/commands/deployments.py +0 -0
- {bonito_cli-0.9.2 → bonito_cli-0.9.4}/bonito_cli/commands/groups.py +0 -0
- {bonito_cli-0.9.2 → bonito_cli-0.9.4}/bonito_cli/commands/init.py +0 -0
- {bonito_cli-0.9.2 → bonito_cli-0.9.4}/bonito_cli/commands/kb.py +0 -0
- {bonito_cli-0.9.2 → bonito_cli-0.9.4}/bonito_cli/commands/mcp.py +0 -0
- {bonito_cli-0.9.2 → bonito_cli-0.9.4}/bonito_cli/commands/memory.py +0 -0
- {bonito_cli-0.9.2 → bonito_cli-0.9.4}/bonito_cli/commands/models.py +0 -0
- {bonito_cli-0.9.2 → bonito_cli-0.9.4}/bonito_cli/commands/plan.py +0 -0
- {bonito_cli-0.9.2 → bonito_cli-0.9.4}/bonito_cli/commands/projects.py +0 -0
- {bonito_cli-0.9.2 → bonito_cli-0.9.4}/bonito_cli/commands/rbac.py +0 -0
- {bonito_cli-0.9.2 → bonito_cli-0.9.4}/bonito_cli/commands/scheduler.py +0 -0
- {bonito_cli-0.9.2 → bonito_cli-0.9.4}/bonito_cli/commands/secrets.py +0 -0
- {bonito_cli-0.9.2 → bonito_cli-0.9.4}/bonito_cli/commands/sso.py +0 -0
- {bonito_cli-0.9.2 → bonito_cli-0.9.4}/bonito_cli/config.py +0 -0
- {bonito_cli-0.9.2 → bonito_cli-0.9.4}/bonito_cli/examples/__init__.py +0 -0
- {bonito_cli-0.9.2 → bonito_cli-0.9.4}/bonito_cli/examples/enterprise.yaml +0 -0
- {bonito_cli-0.9.2 → bonito_cli-0.9.4}/bonito_cli/examples/multi-provider-failover.yaml +0 -0
- {bonito_cli-0.9.2 → bonito_cli-0.9.4}/bonito_cli/examples/quickstart.yaml +0 -0
- {bonito_cli-0.9.2 → bonito_cli-0.9.4}/bonito_cli/examples/support-agent.yaml +0 -0
- {bonito_cli-0.9.2 → bonito_cli-0.9.4}/bonito_cli/utils/__init__.py +0 -0
- {bonito_cli-0.9.2 → bonito_cli-0.9.4}/bonito_cli/utils/auth.py +0 -0
- {bonito_cli-0.9.2 → bonito_cli-0.9.4}/bonito_cli/utils/display.py +0 -0
- {bonito_cli-0.9.2 → bonito_cli-0.9.4}/bonito_cli/utils/feature_gate.py +0 -0
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
node_modules/
|
|
2
|
+
.next/
|
|
3
|
+
__pycache__/
|
|
4
|
+
*.pyc
|
|
5
|
+
.env
|
|
6
|
+
.venv/
|
|
7
|
+
dist/
|
|
8
|
+
*.egg-info/
|
|
9
|
+
.env.secrets
|
|
10
|
+
secrets/age-key.txt
|
|
11
|
+
secrets/*.yaml
|
|
12
|
+
!secrets/*.enc.yaml
|
|
13
|
+
benchmarks/locomo-repo
|
|
14
|
+
bonito-kb-sa-key.json
|
|
15
|
+
.claude/worktrees/
|
|
16
|
+
|
|
17
|
+
# Local dev overrides
|
|
18
|
+
docker-compose.override.yml
|
|
19
|
+
|
|
20
|
+
# Build artifacts
|
|
21
|
+
frontend/tsconfig.tsbuildinfo
|
|
22
|
+
|
|
23
|
+
# Test output
|
|
24
|
+
frontend/test-results/
|
|
25
|
+
|
|
26
|
+
# Business documents — never commit
|
|
27
|
+
invoices/
|
|
28
|
+
|
|
29
|
+
# Generated demo captures & audit reports
|
|
30
|
+
studio-demo/
|
|
31
|
+
studio-demo-haiku-clean/
|
|
32
|
+
studio-*-audit.html
|
|
@@ -2,6 +2,46 @@
|
|
|
2
2
|
|
|
3
3
|
All notable changes to the Bonito CLI will be documented in this file.
|
|
4
4
|
|
|
5
|
+
## 0.9.4
|
|
6
|
+
|
|
7
|
+
- `bonito agents runs list <agent-id>` — recent runs with health rollups
|
|
8
|
+
(duration, llm/tool call counts, errors, cost).
|
|
9
|
+
- `bonito agents runs show <agent-id> <run-id>` — the full trace tree for one
|
|
10
|
+
run, with gateway calls (model, tokens, cost) joined onto their llm spans.
|
|
11
|
+
- New `bonito skills` group — org skill library (agent how-tos): list, add
|
|
12
|
+
from markdown (Poseidon-compatible frontmatter), show, enable/disable,
|
|
13
|
+
remove. Agents load skills on demand via the new `use_skill` tool.
|
|
14
|
+
- New `bonito gateway sessions` group — inspect server-held /v1 conversations
|
|
15
|
+
(clients opt in with the X-Bonito-Session header): list, show (summary +
|
|
16
|
+
stored transcript), clear.
|
|
17
|
+
- New `bonito gateway tokens` group — ephemeral browser tokens (`be-`):
|
|
18
|
+
mint (TTL, model allow-list, rpm, request cap), list, revoke. The safe
|
|
19
|
+
credential for browser apps calling /v1 directly; never ship a bn- key
|
|
20
|
+
to a browser.
|
|
21
|
+
- New `bonito artifacts` group — org artifact store (files produced by
|
|
22
|
+
agents via the new `save_artifact` tool, or uploaded by harness runs):
|
|
23
|
+
list (filter by trace run/session), upload, download, show, remove.
|
|
24
|
+
- New `bonito evals` group — quality checks over runs: add from a
|
|
25
|
+
json/yaml spec (deterministic trace/output checks + LLM-as-judge),
|
|
26
|
+
run against a trace run or output text, results history, remove.
|
|
27
|
+
Evals bound to an agent with `auto: true` fire after every turn.
|
|
28
|
+
- `bonito providers add deepinfra` — connect DeepInfra (provider #8):
|
|
29
|
+
OpenAI-compatible OSS inference (Llama, Qwen, DeepSeek, Mixtral, gpt-oss).
|
|
30
|
+
Also declarable in `bonito.yaml` (`providers: [{name: deepinfra, api_key: ...}]`).
|
|
31
|
+
|
|
32
|
+
## [0.9.3] - 2026-08-18
|
|
33
|
+
|
|
34
|
+
### Fixed
|
|
35
|
+
- **`bonito policies create` offered strategies the gateway does not implement.** The list contained `quality_optimized` and `round_robin`, neither of which exists server-side, so choosing one created a policy that silently fell through to the default and served whichever model was listed first. It also omitted `balanced`, `failover` and `ab_test`, which are real. The list now matches the gateway exactly, each strategy explains what it does, and a `--strategy` passed non-interactively is validated instead of posted blindly.
|
|
36
|
+
- **Model roles and weights were written incorrectly.** Every model was sent as `{"role": "primary", "weight": 50}`, so a `failover` policy had no fallbacks for the chain to advance to, and an `ab_test` policy's weights summed to 50 x N instead of 100, meaning models past the first never received traffic. Roles are now assigned per strategy and `ab_test` weights sum to exactly 100.
|
|
37
|
+
- The interactive strategy prompt was hardcoded to `Select (1-4)` regardless of how many strategies existed.
|
|
38
|
+
|
|
39
|
+
### Changed
|
|
40
|
+
- `bonito policies test` now reports the real blended cost and expected latency for the selected model. Both were previously hardcoded server-side as a "Demo value" (`$0.01`, `150ms`) and returned identically for every policy and every model.
|
|
41
|
+
|
|
42
|
+
### Note
|
|
43
|
+
Requires a backend running Bonito `959b5c0` or later, where `cost_optimized`, `latency_optimized` and `balanced` actually rank models. Before that commit those strategies returned the first configured model regardless of price or latency.
|
|
44
|
+
|
|
5
45
|
## [0.8.0] - 2026-05-27
|
|
6
46
|
|
|
7
47
|
### Added
|
|
@@ -166,6 +166,42 @@ class BonitoAPI:
|
|
|
166
166
|
def delete(self, endpoint: str) -> Any:
|
|
167
167
|
return self._request("DELETE", endpoint)
|
|
168
168
|
|
|
169
|
+
def post_multipart(
|
|
170
|
+
self,
|
|
171
|
+
endpoint: str,
|
|
172
|
+
files: Dict[str, Any],
|
|
173
|
+
data: Optional[Dict[str, Any]] = None,
|
|
174
|
+
) -> Any:
|
|
175
|
+
"""POST a multipart form (file uploads). `files` is httpx's
|
|
176
|
+
{field: (filename, bytes, mime)} shape."""
|
|
177
|
+
url = endpoint if endpoint.startswith("http") else f"/api{endpoint}"
|
|
178
|
+
headers = {k: v for k, v in self._headers().items()
|
|
179
|
+
if k.lower() != "content-type"}
|
|
180
|
+
try:
|
|
181
|
+
resp = self.client.request("POST", url, files=files, data=data,
|
|
182
|
+
headers=headers, timeout=120.0)
|
|
183
|
+
except httpx.RequestError as exc:
|
|
184
|
+
raise APIError(f"Connection failed: {exc}") from exc
|
|
185
|
+
if resp.status_code >= 400:
|
|
186
|
+
try:
|
|
187
|
+
detail = resp.json().get("detail", f"HTTP {resp.status_code}")
|
|
188
|
+
except Exception:
|
|
189
|
+
detail = f"HTTP {resp.status_code}: {resp.text[:200]}"
|
|
190
|
+
raise APIError(str(detail), resp.status_code)
|
|
191
|
+
return resp.json()
|
|
192
|
+
|
|
193
|
+
def get_bytes(self, endpoint: str) -> bytes:
|
|
194
|
+
"""GET raw bytes (artifact downloads)."""
|
|
195
|
+
url = endpoint if endpoint.startswith("http") else f"/api{endpoint}"
|
|
196
|
+
try:
|
|
197
|
+
resp = self.client.request("GET", url, headers=self._headers(),
|
|
198
|
+
timeout=120.0)
|
|
199
|
+
except httpx.RequestError as exc:
|
|
200
|
+
raise APIError(f"Connection failed: {exc}") from exc
|
|
201
|
+
if resp.status_code >= 400:
|
|
202
|
+
raise APIError(f"HTTP {resp.status_code}", resp.status_code)
|
|
203
|
+
return resp.content
|
|
204
|
+
|
|
169
205
|
# ── streaming (SSE) ─────────────────────────────────────────
|
|
170
206
|
|
|
171
207
|
def stream_post(
|
|
@@ -32,6 +32,9 @@ from .commands.approval import app as approval_app
|
|
|
32
32
|
from .commands.scheduler import app as scheduler_app
|
|
33
33
|
from .commands.audit import app as audit_app
|
|
34
34
|
from .commands.memory import app as memory_app
|
|
35
|
+
from .commands.skills import app as skills_app
|
|
36
|
+
from .commands.artifacts import app as artifacts_app
|
|
37
|
+
from .commands.evals import app as evals_app
|
|
35
38
|
from .commands.mcp import app as mcp_app
|
|
36
39
|
from .commands.routing import app as routing_app
|
|
37
40
|
|
|
@@ -102,6 +105,9 @@ app.add_typer(approval_app, name="approval", help="✅ Agent approval work
|
|
|
102
105
|
app.add_typer(scheduler_app, name="scheduler", help="⏰ Agent scheduling")
|
|
103
106
|
app.add_typer(audit_app, name="audit", help="📝 Audit logs")
|
|
104
107
|
app.add_typer(memory_app, name="memory", help="🧠 Agent persistent memory")
|
|
108
|
+
app.add_typer(skills_app, name="skills", help="📚 Org skill library (agent how-tos)")
|
|
109
|
+
app.add_typer(artifacts_app, name="artifacts", help="📦 Artifact store (agent/harness outputs)")
|
|
110
|
+
app.add_typer(evals_app, name="evals", help="✅ Evals (quality checks over runs)")
|
|
105
111
|
app.add_typer(mcp_app, name="mcp", help="🔌 MCP server management")
|
|
106
112
|
app.add_typer(routing_app, name="routing", help="🔀 Routing rule management")
|
|
107
113
|
|
|
@@ -556,6 +556,114 @@ def list_triggers(
|
|
|
556
556
|
|
|
557
557
|
# ─── Scaling / HPA Commands ───
|
|
558
558
|
|
|
559
|
+
# ── Trace tree: agent runs ───────────────────────────────────────────────────
|
|
560
|
+
|
|
561
|
+
runs_app = typer.Typer(help="Per-run traces: what the agent did, called, and cost")
|
|
562
|
+
app.add_typer(runs_app, name="runs")
|
|
563
|
+
|
|
564
|
+
|
|
565
|
+
@runs_app.command("list")
|
|
566
|
+
def runs_list(
|
|
567
|
+
agent_id: str = typer.Argument(..., help="Agent ID"),
|
|
568
|
+
limit: int = typer.Option(20, "--limit", "-n", help="How many runs"),
|
|
569
|
+
status: Optional[str] = typer.Option(None, "--status", help="Filter: ok | error | denied"),
|
|
570
|
+
json_output: bool = typer.Option(False, "--json", help="Output in JSON format"),
|
|
571
|
+
):
|
|
572
|
+
"""Recent runs with health rollups — duration, calls, errors, cost."""
|
|
573
|
+
ensure_authenticated()
|
|
574
|
+
try:
|
|
575
|
+
qs = f"?limit={limit}" + (f"&status={status}" if status else "")
|
|
576
|
+
data = api.get(f"/agents/{agent_id}/runs{qs}")
|
|
577
|
+
except APIError as e:
|
|
578
|
+
print_error(f"Failed to list runs: {e}")
|
|
579
|
+
raise typer.Exit(1)
|
|
580
|
+
|
|
581
|
+
runs = data.get("runs", [])
|
|
582
|
+
if json_output:
|
|
583
|
+
console.print_json(json.dumps(runs))
|
|
584
|
+
return
|
|
585
|
+
if not runs:
|
|
586
|
+
print_info("No traced runs yet — runs appear after the agent executes.")
|
|
587
|
+
return
|
|
588
|
+
|
|
589
|
+
table = Table(title=f"Runs for agent {agent_id}")
|
|
590
|
+
table.add_column("Run", style="cyan", no_wrap=True)
|
|
591
|
+
table.add_column("When")
|
|
592
|
+
table.add_column("Status")
|
|
593
|
+
table.add_column("Duration")
|
|
594
|
+
table.add_column("LLM")
|
|
595
|
+
table.add_column("Tools")
|
|
596
|
+
table.add_column("Errors")
|
|
597
|
+
table.add_column("Cost", justify="right")
|
|
598
|
+
for r in runs:
|
|
599
|
+
table.add_row(
|
|
600
|
+
r["run_id"][:8],
|
|
601
|
+
format_timestamp(r.get("started_at")),
|
|
602
|
+
format_status(r.get("status", "")),
|
|
603
|
+
f"{(r.get('duration_ms') or 0) / 1000:.1f}s",
|
|
604
|
+
str(r.get("llm_calls", 0)),
|
|
605
|
+
str(r.get("tool_calls", 0)),
|
|
606
|
+
str(r.get("errors", 0) + r.get("denied", 0)),
|
|
607
|
+
format_cost(r.get("cost", 0)),
|
|
608
|
+
)
|
|
609
|
+
console.print(table)
|
|
610
|
+
console.print("[dim]bonito agents runs show <agent-id> <run-id> for the full tree[/dim]")
|
|
611
|
+
|
|
612
|
+
|
|
613
|
+
@runs_app.command("show")
|
|
614
|
+
def runs_show(
|
|
615
|
+
agent_id: str = typer.Argument(..., help="Agent ID"),
|
|
616
|
+
run_id: str = typer.Argument(..., help="Run ID (full UUID, or copy from runs list)"),
|
|
617
|
+
json_output: bool = typer.Option(False, "--json", help="Output in JSON format"),
|
|
618
|
+
):
|
|
619
|
+
"""One run as a tree: spans nested, gateway calls joined onto llm spans."""
|
|
620
|
+
ensure_authenticated()
|
|
621
|
+
try:
|
|
622
|
+
data = api.get(f"/agents/{agent_id}/runs/{run_id}")
|
|
623
|
+
except APIError as e:
|
|
624
|
+
print_error(f"Failed to fetch run: {e}")
|
|
625
|
+
raise typer.Exit(1)
|
|
626
|
+
|
|
627
|
+
if json_output:
|
|
628
|
+
console.print_json(json.dumps(data))
|
|
629
|
+
return
|
|
630
|
+
|
|
631
|
+
from rich.tree import Tree as RichTree
|
|
632
|
+
|
|
633
|
+
def label(node):
|
|
634
|
+
status = node.get("status", "ok")
|
|
635
|
+
colour = {"ok": "green", "error": "red", "denied": "yellow",
|
|
636
|
+
"timeout": "red"}.get(status, "white")
|
|
637
|
+
base = (f"[bold]{node['kind']}[/bold] {node['name']} "
|
|
638
|
+
f"[{colour}]{status}[/{colour}] "
|
|
639
|
+
f"[dim]{(node.get('duration_ms') or 0) / 1000:.2f}s[/dim]")
|
|
640
|
+
if node.get("error"):
|
|
641
|
+
base += f"\n[red dim]{node['error'][:120]}[/red dim]"
|
|
642
|
+
for g in node.get("gateway_calls", []):
|
|
643
|
+
base += (f"\n[dim]→ {g.get('model_used') or g.get('model_requested')} "
|
|
644
|
+
f"({g.get('provider') or '?'}) "
|
|
645
|
+
f"{g.get('input_tokens', 0)}in/{g.get('output_tokens', 0)}out "
|
|
646
|
+
f"{format_cost(g.get('cost', 0))}[/dim]")
|
|
647
|
+
return base
|
|
648
|
+
|
|
649
|
+
def attach(rich_node, node):
|
|
650
|
+
for child in node.get("children", []):
|
|
651
|
+
attach(rich_node.add(label(child)), child)
|
|
652
|
+
|
|
653
|
+
root = data["tree"]
|
|
654
|
+
tree_view = RichTree(label(root))
|
|
655
|
+
attach(tree_view, root)
|
|
656
|
+
console.print(tree_view)
|
|
657
|
+
|
|
658
|
+
t = data.get("totals", {})
|
|
659
|
+
console.print(
|
|
660
|
+
f"\n[bold]Totals:[/bold] {t.get('spans', 0)} spans · "
|
|
661
|
+
f"{t.get('gateway_calls', 0)} model calls · "
|
|
662
|
+
f"{t.get('input_tokens', 0)}in/{t.get('output_tokens', 0)}out tokens · "
|
|
663
|
+
f"{format_cost(t.get('cost', 0))}"
|
|
664
|
+
)
|
|
665
|
+
|
|
666
|
+
|
|
559
667
|
scaling_app = typer.Typer(help="Agent autoscaling (HPA) management")
|
|
560
668
|
app.add_typer(scaling_app, name="scaling")
|
|
561
669
|
|
|
@@ -0,0 +1,145 @@
|
|
|
1
|
+
"""Artifact store — durable outputs from agents and harness runs."""
|
|
2
|
+
|
|
3
|
+
import json
|
|
4
|
+
from pathlib import Path
|
|
5
|
+
from typing import Optional
|
|
6
|
+
|
|
7
|
+
import typer
|
|
8
|
+
from rich.console import Console
|
|
9
|
+
from rich.table import Table
|
|
10
|
+
|
|
11
|
+
from ..api import api, APIError
|
|
12
|
+
from ..utils.auth import ensure_authenticated
|
|
13
|
+
from ..utils.display import print_error, print_info, print_success
|
|
14
|
+
|
|
15
|
+
console = Console()
|
|
16
|
+
app = typer.Typer(help="📦 Artifacts — files produced by agents and harness runs")
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
@app.command("list")
|
|
20
|
+
def list_artifacts(
|
|
21
|
+
run_id: Optional[str] = typer.Option(None, "--run", help="Filter by trace run id"),
|
|
22
|
+
session: Optional[str] = typer.Option(None, "--session", help="Filter by session key"),
|
|
23
|
+
json_output: bool = typer.Option(False, "--json", help="Output as JSON"),
|
|
24
|
+
):
|
|
25
|
+
"""List the org's artifacts."""
|
|
26
|
+
ensure_authenticated()
|
|
27
|
+
params = {}
|
|
28
|
+
if run_id:
|
|
29
|
+
params["run_id"] = run_id
|
|
30
|
+
if session:
|
|
31
|
+
params["session_key"] = session
|
|
32
|
+
try:
|
|
33
|
+
data = api.get("/artifacts", params=params or None)
|
|
34
|
+
except APIError as e:
|
|
35
|
+
print_error(f"Failed to list artifacts: {e}")
|
|
36
|
+
raise typer.Exit(1)
|
|
37
|
+
artifacts = data.get("artifacts", [])
|
|
38
|
+
if json_output:
|
|
39
|
+
console.print_json(json.dumps(artifacts))
|
|
40
|
+
return
|
|
41
|
+
if not artifacts:
|
|
42
|
+
print_info("No artifacts yet.")
|
|
43
|
+
return
|
|
44
|
+
table = Table(title="Artifacts")
|
|
45
|
+
table.add_column("ID", style="cyan")
|
|
46
|
+
table.add_column("Name")
|
|
47
|
+
table.add_column("Type")
|
|
48
|
+
table.add_column("Size", justify="right")
|
|
49
|
+
table.add_column("By")
|
|
50
|
+
table.add_column("Created")
|
|
51
|
+
for a in artifacts:
|
|
52
|
+
table.add_row(a["id"][:8], a["name"], a["mime"],
|
|
53
|
+
f"{a['size_bytes']:,}", a.get("created_by") or "-",
|
|
54
|
+
(a.get("created_at") or "")[:19])
|
|
55
|
+
console.print(table)
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
@app.command("upload")
|
|
59
|
+
def upload_artifact(
|
|
60
|
+
file: Path = typer.Argument(..., exists=True, readable=True),
|
|
61
|
+
name: Optional[str] = typer.Option(None, "--name", help="Override the filename"),
|
|
62
|
+
run_id: Optional[str] = typer.Option(None, "--run", help="Link to a trace run id"),
|
|
63
|
+
):
|
|
64
|
+
"""Upload a file as an artifact."""
|
|
65
|
+
ensure_authenticated()
|
|
66
|
+
import mimetypes
|
|
67
|
+
mime = mimetypes.guess_type(str(file))[0] or "application/octet-stream"
|
|
68
|
+
data = {}
|
|
69
|
+
if name:
|
|
70
|
+
data["name"] = name
|
|
71
|
+
if run_id:
|
|
72
|
+
data["run_id"] = run_id
|
|
73
|
+
try:
|
|
74
|
+
result = api.post_multipart(
|
|
75
|
+
"/artifacts",
|
|
76
|
+
files={"file": (file.name, file.read_bytes(), mime)},
|
|
77
|
+
data=data or None)
|
|
78
|
+
print_success(f"Uploaded '{result['name']}' ({result['size_bytes']:,} bytes) "
|
|
79
|
+
f"— id {result['id'][:8]}")
|
|
80
|
+
except APIError as e:
|
|
81
|
+
print_error(f"Upload failed: {e}")
|
|
82
|
+
raise typer.Exit(1)
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def _resolve_id(ref: str) -> str:
|
|
86
|
+
"""Accept an 8-char prefix from `list`."""
|
|
87
|
+
if len(ref) >= 36:
|
|
88
|
+
return ref
|
|
89
|
+
data = api.get("/artifacts")
|
|
90
|
+
matches = [a["id"] for a in data.get("artifacts", []) if a["id"].startswith(ref)]
|
|
91
|
+
if len(matches) != 1:
|
|
92
|
+
print_error(f"Prefix '{ref}' matches {len(matches)} artifacts.")
|
|
93
|
+
raise typer.Exit(1)
|
|
94
|
+
return matches[0]
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
@app.command("download")
|
|
98
|
+
def download_artifact(
|
|
99
|
+
ref: str = typer.Argument(..., help="Artifact id (full or prefix)"),
|
|
100
|
+
output: Optional[Path] = typer.Option(None, "--output", "-o", help="Output path"),
|
|
101
|
+
):
|
|
102
|
+
"""Download an artifact's content."""
|
|
103
|
+
ensure_authenticated()
|
|
104
|
+
try:
|
|
105
|
+
art_id = _resolve_id(ref)
|
|
106
|
+
meta = api.get(f"/artifacts/{art_id}")
|
|
107
|
+
content = api.get_bytes(f"/artifacts/{art_id}/content")
|
|
108
|
+
except APIError as e:
|
|
109
|
+
print_error(f"Download failed: {e}")
|
|
110
|
+
raise typer.Exit(1)
|
|
111
|
+
dest = output or Path(meta["name"])
|
|
112
|
+
dest.write_bytes(content)
|
|
113
|
+
print_success(f"Saved {len(content):,} bytes to {dest}")
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
@app.command("show")
|
|
117
|
+
def show_artifact(
|
|
118
|
+
ref: str = typer.Argument(..., help="Artifact id (full or prefix)"),
|
|
119
|
+
):
|
|
120
|
+
"""Show an artifact's metadata."""
|
|
121
|
+
ensure_authenticated()
|
|
122
|
+
try:
|
|
123
|
+
meta = api.get(f"/artifacts/{_resolve_id(ref)}")
|
|
124
|
+
except APIError as e:
|
|
125
|
+
print_error(f"Failed: {e}")
|
|
126
|
+
raise typer.Exit(1)
|
|
127
|
+
console.print_json(json.dumps(meta))
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
@app.command("remove")
|
|
131
|
+
def remove_artifact(
|
|
132
|
+
ref: str = typer.Argument(..., help="Artifact id (full or prefix)"),
|
|
133
|
+
yes: bool = typer.Option(False, "--yes", "-y", help="Skip confirmation"),
|
|
134
|
+
):
|
|
135
|
+
"""Delete an artifact."""
|
|
136
|
+
ensure_authenticated()
|
|
137
|
+
art_id = _resolve_id(ref)
|
|
138
|
+
if not yes and not typer.confirm(f"Delete artifact {art_id[:8]}?"):
|
|
139
|
+
raise typer.Exit(0)
|
|
140
|
+
try:
|
|
141
|
+
api.delete(f"/artifacts/{art_id}")
|
|
142
|
+
print_success("Artifact deleted.")
|
|
143
|
+
except APIError as e:
|
|
144
|
+
print_error(f"Failed to delete: {e}")
|
|
145
|
+
raise typer.Exit(1)
|
|
@@ -0,0 +1,183 @@
|
|
|
1
|
+
"""Evals — quality checks over agent/harness runs (deterministic + LLM judge)."""
|
|
2
|
+
|
|
3
|
+
import json
|
|
4
|
+
from pathlib import Path
|
|
5
|
+
from typing import Optional
|
|
6
|
+
|
|
7
|
+
import typer
|
|
8
|
+
from rich.console import Console
|
|
9
|
+
from rich.table import Table
|
|
10
|
+
|
|
11
|
+
from ..api import api, APIError
|
|
12
|
+
from ..utils.auth import ensure_authenticated
|
|
13
|
+
from ..utils.display import print_error, print_info, print_success
|
|
14
|
+
|
|
15
|
+
console = Console()
|
|
16
|
+
app = typer.Typer(help="✅ Evals — quality checks over agent runs and outputs")
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
@app.command("list")
|
|
20
|
+
def list_evals(
|
|
21
|
+
json_output: bool = typer.Option(False, "--json", help="Output as JSON"),
|
|
22
|
+
):
|
|
23
|
+
"""List the org's evals."""
|
|
24
|
+
ensure_authenticated()
|
|
25
|
+
try:
|
|
26
|
+
data = api.get("/evals")
|
|
27
|
+
except APIError as e:
|
|
28
|
+
print_error(f"Failed to list evals: {e}")
|
|
29
|
+
raise typer.Exit(1)
|
|
30
|
+
evals = data.get("evals", [])
|
|
31
|
+
if json_output:
|
|
32
|
+
console.print_json(json.dumps(evals))
|
|
33
|
+
return
|
|
34
|
+
if not evals:
|
|
35
|
+
print_info("No evals yet — bonito evals add <spec.json|.yaml> to create one.")
|
|
36
|
+
return
|
|
37
|
+
table = Table(title="Evals")
|
|
38
|
+
table.add_column("Name", style="cyan")
|
|
39
|
+
table.add_column("Checks", justify="right")
|
|
40
|
+
table.add_column("Agent")
|
|
41
|
+
table.add_column("Auto")
|
|
42
|
+
table.add_column("Enabled")
|
|
43
|
+
for e in evals:
|
|
44
|
+
table.add_row(e["name"], str(len(e["checks"])),
|
|
45
|
+
(e.get("agent_id") or "-")[:8],
|
|
46
|
+
"yes" if e["auto"] else "no",
|
|
47
|
+
"yes" if e["enabled"] else "no")
|
|
48
|
+
console.print(table)
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
@app.command("add")
|
|
52
|
+
def add_eval(
|
|
53
|
+
file: Path = typer.Argument(..., exists=True, readable=True,
|
|
54
|
+
help="Eval spec (.json or .yaml): {name, description, checks, agent_id?, auto?}"),
|
|
55
|
+
update: bool = typer.Option(False, "--update", "-u",
|
|
56
|
+
help="Overwrite if the eval already exists"),
|
|
57
|
+
):
|
|
58
|
+
"""Add an eval from a spec file."""
|
|
59
|
+
ensure_authenticated()
|
|
60
|
+
raw = file.read_text(encoding="utf-8")
|
|
61
|
+
if file.suffix in (".yaml", ".yml"):
|
|
62
|
+
import yaml
|
|
63
|
+
spec = yaml.safe_load(raw)
|
|
64
|
+
else:
|
|
65
|
+
spec = json.loads(raw)
|
|
66
|
+
try:
|
|
67
|
+
api.post("/evals", spec)
|
|
68
|
+
print_success(f"Eval '{spec.get('name')}' added.")
|
|
69
|
+
except APIError as e:
|
|
70
|
+
if update and "already exists" in str(e):
|
|
71
|
+
name = spec.pop("name")
|
|
72
|
+
api.patch(f"/evals/{name}", spec)
|
|
73
|
+
print_success(f"Eval '{name}' updated.")
|
|
74
|
+
else:
|
|
75
|
+
print_error(f"Failed to add eval: {e}")
|
|
76
|
+
raise typer.Exit(1)
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
@app.command("show")
|
|
80
|
+
def show_eval(name: str = typer.Argument(..., help="Eval name")):
|
|
81
|
+
"""Show an eval's checks."""
|
|
82
|
+
ensure_authenticated()
|
|
83
|
+
try:
|
|
84
|
+
e = api.get(f"/evals/{name}")
|
|
85
|
+
except APIError as e2:
|
|
86
|
+
print_error(f"Failed: {e2}")
|
|
87
|
+
raise typer.Exit(1)
|
|
88
|
+
console.print_json(json.dumps(e))
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
@app.command("run")
|
|
92
|
+
def run_eval(
|
|
93
|
+
name: str = typer.Argument(..., help="Eval name"),
|
|
94
|
+
run_id: Optional[str] = typer.Option(None, "--run", help="Trace run id to evaluate"),
|
|
95
|
+
output: Optional[str] = typer.Option(None, "--output", help="Output text to evaluate"),
|
|
96
|
+
output_file: Optional[Path] = typer.Option(None, "--output-file",
|
|
97
|
+
help="Read output text from a file"),
|
|
98
|
+
judge_model: Optional[str] = typer.Option(None, "--judge-model",
|
|
99
|
+
help="Model for judge checks"),
|
|
100
|
+
):
|
|
101
|
+
"""Run an eval against a trace run and/or an output text."""
|
|
102
|
+
ensure_authenticated()
|
|
103
|
+
payload = {}
|
|
104
|
+
if run_id:
|
|
105
|
+
payload["run_id"] = run_id
|
|
106
|
+
if output_file:
|
|
107
|
+
payload["output"] = output_file.read_text(encoding="utf-8")
|
|
108
|
+
elif output:
|
|
109
|
+
payload["output"] = output
|
|
110
|
+
if judge_model:
|
|
111
|
+
payload["judge_model"] = judge_model
|
|
112
|
+
if not payload.get("run_id") and payload.get("output") is None:
|
|
113
|
+
print_error("Provide --run and/or --output/--output-file.")
|
|
114
|
+
raise typer.Exit(1)
|
|
115
|
+
try:
|
|
116
|
+
res = api.post(f"/evals/{name}/run", payload)
|
|
117
|
+
except APIError as e:
|
|
118
|
+
print_error(f"Eval run failed: {e}")
|
|
119
|
+
raise typer.Exit(1)
|
|
120
|
+
verdict = "[green]PASSED[/green]" if res["passed"] else "[red]FAILED[/red]"
|
|
121
|
+
score = f" (score {res['score']:.0f})" if res.get("score") is not None else ""
|
|
122
|
+
console.print(f"{verdict}{score}")
|
|
123
|
+
for c in res["checks"]:
|
|
124
|
+
mark = {"pass": "[green]✓[/green]", "fail": "[red]✗[/red]"}.get(
|
|
125
|
+
c["status"], "[dim]○[/dim]")
|
|
126
|
+
detail = c.get("reasoning") or c.get("detail") or ""
|
|
127
|
+
extra = ""
|
|
128
|
+
if "actual" in c:
|
|
129
|
+
extra = f" (actual {c['actual']}, limit {c.get('limit', '-')})"
|
|
130
|
+
elif "score" in c:
|
|
131
|
+
extra = f" (score {c['score']:.0f} vs min {c['min_score']})"
|
|
132
|
+
console.print(f" {mark} {c['type']}{extra} {detail}")
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
@app.command("results")
|
|
136
|
+
def eval_results(
|
|
137
|
+
name: str = typer.Argument(..., help="Eval name"),
|
|
138
|
+
limit: int = typer.Option(20, "--limit"),
|
|
139
|
+
json_output: bool = typer.Option(False, "--json", help="Output as JSON"),
|
|
140
|
+
):
|
|
141
|
+
"""Show recent results for an eval."""
|
|
142
|
+
ensure_authenticated()
|
|
143
|
+
try:
|
|
144
|
+
data = api.get(f"/evals/{name}/results", params={"limit": limit})
|
|
145
|
+
except APIError as e:
|
|
146
|
+
print_error(f"Failed: {e}")
|
|
147
|
+
raise typer.Exit(1)
|
|
148
|
+
results = data.get("results", [])
|
|
149
|
+
if json_output:
|
|
150
|
+
console.print_json(json.dumps(results))
|
|
151
|
+
return
|
|
152
|
+
if not results:
|
|
153
|
+
print_info("No results yet.")
|
|
154
|
+
return
|
|
155
|
+
table = Table(title=f"Results — {name}")
|
|
156
|
+
table.add_column("When")
|
|
157
|
+
table.add_column("Verdict")
|
|
158
|
+
table.add_column("Score", justify="right")
|
|
159
|
+
table.add_column("Trigger")
|
|
160
|
+
table.add_column("Run")
|
|
161
|
+
for r in results:
|
|
162
|
+
table.add_row((r.get("created_at") or "")[:19],
|
|
163
|
+
"[green]pass[/green]" if r["passed"] else "[red]fail[/red]",
|
|
164
|
+
f"{r['score']:.0f}" if r.get("score") is not None else "-",
|
|
165
|
+
r["trigger"], (r.get("run_id") or "-")[:8])
|
|
166
|
+
console.print(table)
|
|
167
|
+
|
|
168
|
+
|
|
169
|
+
@app.command("remove")
|
|
170
|
+
def remove_eval(
|
|
171
|
+
name: str = typer.Argument(...),
|
|
172
|
+
yes: bool = typer.Option(False, "--yes", "-y", help="Skip confirmation"),
|
|
173
|
+
):
|
|
174
|
+
"""Delete an eval (results cascade)."""
|
|
175
|
+
ensure_authenticated()
|
|
176
|
+
if not yes and not typer.confirm(f"Delete eval '{name}' and its results?"):
|
|
177
|
+
raise typer.Exit(0)
|
|
178
|
+
try:
|
|
179
|
+
api.delete(f"/evals/{name}")
|
|
180
|
+
print_success(f"Eval '{name}' deleted.")
|
|
181
|
+
except APIError as e:
|
|
182
|
+
print_error(f"Failed to delete: {e}")
|
|
183
|
+
raise typer.Exit(1)
|