bonito-cli 0.9.2__tar.gz → 0.9.4__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (52) hide show
  1. bonito_cli-0.9.4/.gitignore +32 -0
  2. {bonito_cli-0.9.2 → bonito_cli-0.9.4}/CHANGELOG.md +40 -0
  3. {bonito_cli-0.9.2 → bonito_cli-0.9.4}/PKG-INFO +2 -2
  4. {bonito_cli-0.9.2 → bonito_cli-0.9.4}/bonito_cli/api.py +36 -0
  5. {bonito_cli-0.9.2 → bonito_cli-0.9.4}/bonito_cli/app.py +6 -0
  6. {bonito_cli-0.9.2 → bonito_cli-0.9.4}/bonito_cli/commands/agents.py +108 -0
  7. bonito_cli-0.9.4/bonito_cli/commands/artifacts.py +145 -0
  8. bonito_cli-0.9.4/bonito_cli/commands/evals.py +183 -0
  9. {bonito_cli-0.9.2 → bonito_cli-0.9.4}/bonito_cli/commands/gateway.py +168 -0
  10. {bonito_cli-0.9.2 → bonito_cli-0.9.4}/bonito_cli/commands/policies.py +42 -5
  11. {bonito_cli-0.9.2 → bonito_cli-0.9.4}/bonito_cli/commands/providers.py +18 -1
  12. bonito_cli-0.9.4/bonito_cli/commands/routing.py +486 -0
  13. bonito_cli-0.9.4/bonito_cli/commands/skills.py +163 -0
  14. {bonito_cli-0.9.2 → bonito_cli-0.9.4}/pyproject.toml +1 -1
  15. bonito_cli-0.9.2/.gitignore +0 -15
  16. bonito_cli-0.9.2/bonito_cli/commands/routing.py +0 -178
  17. {bonito_cli-0.9.2 → bonito_cli-0.9.4}/CLI_SPEC.md +0 -0
  18. {bonito_cli-0.9.2 → bonito_cli-0.9.4}/README.md +0 -0
  19. {bonito_cli-0.9.2 → bonito_cli-0.9.4}/bonito_cli/__init__.py +0 -0
  20. {bonito_cli-0.9.2 → bonito_cli-0.9.4}/bonito_cli/__main__.py +0 -0
  21. {bonito_cli-0.9.2 → bonito_cli-0.9.4}/bonito_cli/commands/__init__.py +0 -0
  22. {bonito_cli-0.9.2 → bonito_cli-0.9.4}/bonito_cli/commands/admin.py +0 -0
  23. {bonito_cli-0.9.2 → bonito_cli-0.9.4}/bonito_cli/commands/analytics.py +0 -0
  24. {bonito_cli-0.9.2 → bonito_cli-0.9.4}/bonito_cli/commands/approval.py +0 -0
  25. {bonito_cli-0.9.2 → bonito_cli-0.9.4}/bonito_cli/commands/audit.py +0 -0
  26. {bonito_cli-0.9.2 → bonito_cli-0.9.4}/bonito_cli/commands/auth.py +0 -0
  27. {bonito_cli-0.9.2 → bonito_cli-0.9.4}/bonito_cli/commands/chat.py +0 -0
  28. {bonito_cli-0.9.2 → bonito_cli-0.9.4}/bonito_cli/commands/compliance.py +0 -0
  29. {bonito_cli-0.9.2 → bonito_cli-0.9.4}/bonito_cli/commands/deploy.py +0 -0
  30. {bonito_cli-0.9.2 → bonito_cli-0.9.4}/bonito_cli/commands/deployments.py +0 -0
  31. {bonito_cli-0.9.2 → bonito_cli-0.9.4}/bonito_cli/commands/groups.py +0 -0
  32. {bonito_cli-0.9.2 → bonito_cli-0.9.4}/bonito_cli/commands/init.py +0 -0
  33. {bonito_cli-0.9.2 → bonito_cli-0.9.4}/bonito_cli/commands/kb.py +0 -0
  34. {bonito_cli-0.9.2 → bonito_cli-0.9.4}/bonito_cli/commands/mcp.py +0 -0
  35. {bonito_cli-0.9.2 → bonito_cli-0.9.4}/bonito_cli/commands/memory.py +0 -0
  36. {bonito_cli-0.9.2 → bonito_cli-0.9.4}/bonito_cli/commands/models.py +0 -0
  37. {bonito_cli-0.9.2 → bonito_cli-0.9.4}/bonito_cli/commands/plan.py +0 -0
  38. {bonito_cli-0.9.2 → bonito_cli-0.9.4}/bonito_cli/commands/projects.py +0 -0
  39. {bonito_cli-0.9.2 → bonito_cli-0.9.4}/bonito_cli/commands/rbac.py +0 -0
  40. {bonito_cli-0.9.2 → bonito_cli-0.9.4}/bonito_cli/commands/scheduler.py +0 -0
  41. {bonito_cli-0.9.2 → bonito_cli-0.9.4}/bonito_cli/commands/secrets.py +0 -0
  42. {bonito_cli-0.9.2 → bonito_cli-0.9.4}/bonito_cli/commands/sso.py +0 -0
  43. {bonito_cli-0.9.2 → bonito_cli-0.9.4}/bonito_cli/config.py +0 -0
  44. {bonito_cli-0.9.2 → bonito_cli-0.9.4}/bonito_cli/examples/__init__.py +0 -0
  45. {bonito_cli-0.9.2 → bonito_cli-0.9.4}/bonito_cli/examples/enterprise.yaml +0 -0
  46. {bonito_cli-0.9.2 → bonito_cli-0.9.4}/bonito_cli/examples/multi-provider-failover.yaml +0 -0
  47. {bonito_cli-0.9.2 → bonito_cli-0.9.4}/bonito_cli/examples/quickstart.yaml +0 -0
  48. {bonito_cli-0.9.2 → bonito_cli-0.9.4}/bonito_cli/examples/support-agent.yaml +0 -0
  49. {bonito_cli-0.9.2 → bonito_cli-0.9.4}/bonito_cli/utils/__init__.py +0 -0
  50. {bonito_cli-0.9.2 → bonito_cli-0.9.4}/bonito_cli/utils/auth.py +0 -0
  51. {bonito_cli-0.9.2 → bonito_cli-0.9.4}/bonito_cli/utils/display.py +0 -0
  52. {bonito_cli-0.9.2 → bonito_cli-0.9.4}/bonito_cli/utils/feature_gate.py +0 -0
@@ -0,0 +1,32 @@
1
+ node_modules/
2
+ .next/
3
+ __pycache__/
4
+ *.pyc
5
+ .env
6
+ .venv/
7
+ dist/
8
+ *.egg-info/
9
+ .env.secrets
10
+ secrets/age-key.txt
11
+ secrets/*.yaml
12
+ !secrets/*.enc.yaml
13
+ benchmarks/locomo-repo
14
+ bonito-kb-sa-key.json
15
+ .claude/worktrees/
16
+
17
+ # Local dev overrides
18
+ docker-compose.override.yml
19
+
20
+ # Build artifacts
21
+ frontend/tsconfig.tsbuildinfo
22
+
23
+ # Test output
24
+ frontend/test-results/
25
+
26
+ # Business documents — never commit
27
+ invoices/
28
+
29
+ # Generated demo captures & audit reports
30
+ studio-demo/
31
+ studio-demo-haiku-clean/
32
+ studio-*-audit.html
@@ -2,6 +2,46 @@
2
2
 
3
3
  All notable changes to the Bonito CLI will be documented in this file.
4
4
 
5
+ ## 0.9.4
6
+
7
+ - `bonito agents runs list <agent-id>` — recent runs with health rollups
8
+ (duration, llm/tool call counts, errors, cost).
9
+ - `bonito agents runs show <agent-id> <run-id>` — the full trace tree for one
10
+ run, with gateway calls (model, tokens, cost) joined onto their llm spans.
11
+ - New `bonito skills` group — org skill library (agent how-tos): list, add
12
+ from markdown (Poseidon-compatible frontmatter), show, enable/disable,
13
+ remove. Agents load skills on demand via the new `use_skill` tool.
14
+ - New `bonito gateway sessions` group — inspect server-held /v1 conversations
15
+ (clients opt in with the X-Bonito-Session header): list, show (summary +
16
+ stored transcript), clear.
17
+ - New `bonito gateway tokens` group — ephemeral browser tokens (`be-`):
18
+ mint (TTL, model allow-list, rpm, request cap), list, revoke. The safe
19
+ credential for browser apps calling /v1 directly; never ship a bn- key
20
+ to a browser.
21
+ - New `bonito artifacts` group — org artifact store (files produced by
22
+ agents via the new `save_artifact` tool, or uploaded by harness runs):
23
+ list (filter by trace run/session), upload, download, show, remove.
24
+ - New `bonito evals` group — quality checks over runs: add from a
25
+ json/yaml spec (deterministic trace/output checks + LLM-as-judge),
26
+ run against a trace run or output text, results history, remove.
27
+ Evals bound to an agent with `auto: true` fire after every turn.
28
+ - `bonito providers add deepinfra` — connect DeepInfra (provider #8):
29
+ OpenAI-compatible OSS inference (Llama, Qwen, DeepSeek, Mixtral, gpt-oss).
30
+ Also declarable in `bonito.yaml` (`providers: [{name: deepinfra, api_key: ...}]`).
31
+
32
+ ## [0.9.3] - 2026-08-18
33
+
34
+ ### Fixed
35
+ - **`bonito policies create` offered strategies the gateway does not implement.** The list contained `quality_optimized` and `round_robin`, neither of which exists server-side, so choosing one created a policy that silently fell through to the default and served whichever model was listed first. It also omitted `balanced`, `failover` and `ab_test`, which are real. The list now matches the gateway exactly, each strategy explains what it does, and a `--strategy` passed non-interactively is validated instead of posted blindly.
36
+ - **Model roles and weights were written incorrectly.** Every model was sent as `{"role": "primary", "weight": 50}`, so a `failover` policy had no fallbacks for the chain to advance to, and an `ab_test` policy's weights summed to 50 x N instead of 100, meaning models past the first never received traffic. Roles are now assigned per strategy and `ab_test` weights sum to exactly 100.
37
+ - The interactive strategy prompt was hardcoded to `Select (1-4)` regardless of how many strategies existed.
38
+
39
+ ### Changed
40
+ - `bonito policies test` now reports the real blended cost and expected latency for the selected model. Both were previously hardcoded server-side as a "Demo value" (`$0.01`, `150ms`) and returned identically for every policy and every model.
41
+
42
+ ### Note
43
+ Requires a backend running Bonito `959b5c0` or later, where `cost_optimized`, `latency_optimized` and `balanced` actually rank models. Before that commit those strategies returned the first configured model regardless of price or latency.
44
+
5
45
  ## [0.8.0] - 2026-05-27
6
46
 
7
47
  ### Added
@@ -1,6 +1,6 @@
1
- Metadata-Version: 2.4
1
+ Metadata-Version: 2.5
2
2
  Name: bonito-cli
3
- Version: 0.9.2
3
+ Version: 0.9.4
4
4
  Summary: Bonito CLI — Unified multi-cloud AI management from your terminal
5
5
  Author-email: Bonito <hello@getbonito.com>
6
6
  License: MIT
@@ -166,6 +166,42 @@ class BonitoAPI:
166
166
  def delete(self, endpoint: str) -> Any:
167
167
  return self._request("DELETE", endpoint)
168
168
 
169
+ def post_multipart(
170
+ self,
171
+ endpoint: str,
172
+ files: Dict[str, Any],
173
+ data: Optional[Dict[str, Any]] = None,
174
+ ) -> Any:
175
+ """POST a multipart form (file uploads). `files` is httpx's
176
+ {field: (filename, bytes, mime)} shape."""
177
+ url = endpoint if endpoint.startswith("http") else f"/api{endpoint}"
178
+ headers = {k: v for k, v in self._headers().items()
179
+ if k.lower() != "content-type"}
180
+ try:
181
+ resp = self.client.request("POST", url, files=files, data=data,
182
+ headers=headers, timeout=120.0)
183
+ except httpx.RequestError as exc:
184
+ raise APIError(f"Connection failed: {exc}") from exc
185
+ if resp.status_code >= 400:
186
+ try:
187
+ detail = resp.json().get("detail", f"HTTP {resp.status_code}")
188
+ except Exception:
189
+ detail = f"HTTP {resp.status_code}: {resp.text[:200]}"
190
+ raise APIError(str(detail), resp.status_code)
191
+ return resp.json()
192
+
193
+ def get_bytes(self, endpoint: str) -> bytes:
194
+ """GET raw bytes (artifact downloads)."""
195
+ url = endpoint if endpoint.startswith("http") else f"/api{endpoint}"
196
+ try:
197
+ resp = self.client.request("GET", url, headers=self._headers(),
198
+ timeout=120.0)
199
+ except httpx.RequestError as exc:
200
+ raise APIError(f"Connection failed: {exc}") from exc
201
+ if resp.status_code >= 400:
202
+ raise APIError(f"HTTP {resp.status_code}", resp.status_code)
203
+ return resp.content
204
+
169
205
  # ── streaming (SSE) ─────────────────────────────────────────
170
206
 
171
207
  def stream_post(
@@ -32,6 +32,9 @@ from .commands.approval import app as approval_app
32
32
  from .commands.scheduler import app as scheduler_app
33
33
  from .commands.audit import app as audit_app
34
34
  from .commands.memory import app as memory_app
35
+ from .commands.skills import app as skills_app
36
+ from .commands.artifacts import app as artifacts_app
37
+ from .commands.evals import app as evals_app
35
38
  from .commands.mcp import app as mcp_app
36
39
  from .commands.routing import app as routing_app
37
40
 
@@ -102,6 +105,9 @@ app.add_typer(approval_app, name="approval", help="✅ Agent approval work
102
105
  app.add_typer(scheduler_app, name="scheduler", help="⏰ Agent scheduling")
103
106
  app.add_typer(audit_app, name="audit", help="📝 Audit logs")
104
107
  app.add_typer(memory_app, name="memory", help="🧠 Agent persistent memory")
108
+ app.add_typer(skills_app, name="skills", help="📚 Org skill library (agent how-tos)")
109
+ app.add_typer(artifacts_app, name="artifacts", help="📦 Artifact store (agent/harness outputs)")
110
+ app.add_typer(evals_app, name="evals", help="✅ Evals (quality checks over runs)")
105
111
  app.add_typer(mcp_app, name="mcp", help="🔌 MCP server management")
106
112
  app.add_typer(routing_app, name="routing", help="🔀 Routing rule management")
107
113
 
@@ -556,6 +556,114 @@ def list_triggers(
556
556
 
557
557
  # ─── Scaling / HPA Commands ───
558
558
 
559
+ # ── Trace tree: agent runs ───────────────────────────────────────────────────
560
+
561
+ runs_app = typer.Typer(help="Per-run traces: what the agent did, called, and cost")
562
+ app.add_typer(runs_app, name="runs")
563
+
564
+
565
+ @runs_app.command("list")
566
+ def runs_list(
567
+ agent_id: str = typer.Argument(..., help="Agent ID"),
568
+ limit: int = typer.Option(20, "--limit", "-n", help="How many runs"),
569
+ status: Optional[str] = typer.Option(None, "--status", help="Filter: ok | error | denied"),
570
+ json_output: bool = typer.Option(False, "--json", help="Output in JSON format"),
571
+ ):
572
+ """Recent runs with health rollups — duration, calls, errors, cost."""
573
+ ensure_authenticated()
574
+ try:
575
+ qs = f"?limit={limit}" + (f"&status={status}" if status else "")
576
+ data = api.get(f"/agents/{agent_id}/runs{qs}")
577
+ except APIError as e:
578
+ print_error(f"Failed to list runs: {e}")
579
+ raise typer.Exit(1)
580
+
581
+ runs = data.get("runs", [])
582
+ if json_output:
583
+ console.print_json(json.dumps(runs))
584
+ return
585
+ if not runs:
586
+ print_info("No traced runs yet — runs appear after the agent executes.")
587
+ return
588
+
589
+ table = Table(title=f"Runs for agent {agent_id}")
590
+ table.add_column("Run", style="cyan", no_wrap=True)
591
+ table.add_column("When")
592
+ table.add_column("Status")
593
+ table.add_column("Duration")
594
+ table.add_column("LLM")
595
+ table.add_column("Tools")
596
+ table.add_column("Errors")
597
+ table.add_column("Cost", justify="right")
598
+ for r in runs:
599
+ table.add_row(
600
+ r["run_id"][:8],
601
+ format_timestamp(r.get("started_at")),
602
+ format_status(r.get("status", "")),
603
+ f"{(r.get('duration_ms') or 0) / 1000:.1f}s",
604
+ str(r.get("llm_calls", 0)),
605
+ str(r.get("tool_calls", 0)),
606
+ str(r.get("errors", 0) + r.get("denied", 0)),
607
+ format_cost(r.get("cost", 0)),
608
+ )
609
+ console.print(table)
610
+ console.print("[dim]bonito agents runs show <agent-id> <run-id> for the full tree[/dim]")
611
+
612
+
613
+ @runs_app.command("show")
614
+ def runs_show(
615
+ agent_id: str = typer.Argument(..., help="Agent ID"),
616
+ run_id: str = typer.Argument(..., help="Run ID (full UUID, or copy from runs list)"),
617
+ json_output: bool = typer.Option(False, "--json", help="Output in JSON format"),
618
+ ):
619
+ """One run as a tree: spans nested, gateway calls joined onto llm spans."""
620
+ ensure_authenticated()
621
+ try:
622
+ data = api.get(f"/agents/{agent_id}/runs/{run_id}")
623
+ except APIError as e:
624
+ print_error(f"Failed to fetch run: {e}")
625
+ raise typer.Exit(1)
626
+
627
+ if json_output:
628
+ console.print_json(json.dumps(data))
629
+ return
630
+
631
+ from rich.tree import Tree as RichTree
632
+
633
+ def label(node):
634
+ status = node.get("status", "ok")
635
+ colour = {"ok": "green", "error": "red", "denied": "yellow",
636
+ "timeout": "red"}.get(status, "white")
637
+ base = (f"[bold]{node['kind']}[/bold] {node['name']} "
638
+ f"[{colour}]{status}[/{colour}] "
639
+ f"[dim]{(node.get('duration_ms') or 0) / 1000:.2f}s[/dim]")
640
+ if node.get("error"):
641
+ base += f"\n[red dim]{node['error'][:120]}[/red dim]"
642
+ for g in node.get("gateway_calls", []):
643
+ base += (f"\n[dim]→ {g.get('model_used') or g.get('model_requested')} "
644
+ f"({g.get('provider') or '?'}) "
645
+ f"{g.get('input_tokens', 0)}in/{g.get('output_tokens', 0)}out "
646
+ f"{format_cost(g.get('cost', 0))}[/dim]")
647
+ return base
648
+
649
+ def attach(rich_node, node):
650
+ for child in node.get("children", []):
651
+ attach(rich_node.add(label(child)), child)
652
+
653
+ root = data["tree"]
654
+ tree_view = RichTree(label(root))
655
+ attach(tree_view, root)
656
+ console.print(tree_view)
657
+
658
+ t = data.get("totals", {})
659
+ console.print(
660
+ f"\n[bold]Totals:[/bold] {t.get('spans', 0)} spans · "
661
+ f"{t.get('gateway_calls', 0)} model calls · "
662
+ f"{t.get('input_tokens', 0)}in/{t.get('output_tokens', 0)}out tokens · "
663
+ f"{format_cost(t.get('cost', 0))}"
664
+ )
665
+
666
+
559
667
  scaling_app = typer.Typer(help="Agent autoscaling (HPA) management")
560
668
  app.add_typer(scaling_app, name="scaling")
561
669
 
@@ -0,0 +1,145 @@
1
+ """Artifact store — durable outputs from agents and harness runs."""
2
+
3
+ import json
4
+ from pathlib import Path
5
+ from typing import Optional
6
+
7
+ import typer
8
+ from rich.console import Console
9
+ from rich.table import Table
10
+
11
+ from ..api import api, APIError
12
+ from ..utils.auth import ensure_authenticated
13
+ from ..utils.display import print_error, print_info, print_success
14
+
15
+ console = Console()
16
+ app = typer.Typer(help="📦 Artifacts — files produced by agents and harness runs")
17
+
18
+
19
+ @app.command("list")
20
+ def list_artifacts(
21
+ run_id: Optional[str] = typer.Option(None, "--run", help="Filter by trace run id"),
22
+ session: Optional[str] = typer.Option(None, "--session", help="Filter by session key"),
23
+ json_output: bool = typer.Option(False, "--json", help="Output as JSON"),
24
+ ):
25
+ """List the org's artifacts."""
26
+ ensure_authenticated()
27
+ params = {}
28
+ if run_id:
29
+ params["run_id"] = run_id
30
+ if session:
31
+ params["session_key"] = session
32
+ try:
33
+ data = api.get("/artifacts", params=params or None)
34
+ except APIError as e:
35
+ print_error(f"Failed to list artifacts: {e}")
36
+ raise typer.Exit(1)
37
+ artifacts = data.get("artifacts", [])
38
+ if json_output:
39
+ console.print_json(json.dumps(artifacts))
40
+ return
41
+ if not artifacts:
42
+ print_info("No artifacts yet.")
43
+ return
44
+ table = Table(title="Artifacts")
45
+ table.add_column("ID", style="cyan")
46
+ table.add_column("Name")
47
+ table.add_column("Type")
48
+ table.add_column("Size", justify="right")
49
+ table.add_column("By")
50
+ table.add_column("Created")
51
+ for a in artifacts:
52
+ table.add_row(a["id"][:8], a["name"], a["mime"],
53
+ f"{a['size_bytes']:,}", a.get("created_by") or "-",
54
+ (a.get("created_at") or "")[:19])
55
+ console.print(table)
56
+
57
+
58
+ @app.command("upload")
59
+ def upload_artifact(
60
+ file: Path = typer.Argument(..., exists=True, readable=True),
61
+ name: Optional[str] = typer.Option(None, "--name", help="Override the filename"),
62
+ run_id: Optional[str] = typer.Option(None, "--run", help="Link to a trace run id"),
63
+ ):
64
+ """Upload a file as an artifact."""
65
+ ensure_authenticated()
66
+ import mimetypes
67
+ mime = mimetypes.guess_type(str(file))[0] or "application/octet-stream"
68
+ data = {}
69
+ if name:
70
+ data["name"] = name
71
+ if run_id:
72
+ data["run_id"] = run_id
73
+ try:
74
+ result = api.post_multipart(
75
+ "/artifacts",
76
+ files={"file": (file.name, file.read_bytes(), mime)},
77
+ data=data or None)
78
+ print_success(f"Uploaded '{result['name']}' ({result['size_bytes']:,} bytes) "
79
+ f"— id {result['id'][:8]}")
80
+ except APIError as e:
81
+ print_error(f"Upload failed: {e}")
82
+ raise typer.Exit(1)
83
+
84
+
85
+ def _resolve_id(ref: str) -> str:
86
+ """Accept an 8-char prefix from `list`."""
87
+ if len(ref) >= 36:
88
+ return ref
89
+ data = api.get("/artifacts")
90
+ matches = [a["id"] for a in data.get("artifacts", []) if a["id"].startswith(ref)]
91
+ if len(matches) != 1:
92
+ print_error(f"Prefix '{ref}' matches {len(matches)} artifacts.")
93
+ raise typer.Exit(1)
94
+ return matches[0]
95
+
96
+
97
+ @app.command("download")
98
+ def download_artifact(
99
+ ref: str = typer.Argument(..., help="Artifact id (full or prefix)"),
100
+ output: Optional[Path] = typer.Option(None, "--output", "-o", help="Output path"),
101
+ ):
102
+ """Download an artifact's content."""
103
+ ensure_authenticated()
104
+ try:
105
+ art_id = _resolve_id(ref)
106
+ meta = api.get(f"/artifacts/{art_id}")
107
+ content = api.get_bytes(f"/artifacts/{art_id}/content")
108
+ except APIError as e:
109
+ print_error(f"Download failed: {e}")
110
+ raise typer.Exit(1)
111
+ dest = output or Path(meta["name"])
112
+ dest.write_bytes(content)
113
+ print_success(f"Saved {len(content):,} bytes to {dest}")
114
+
115
+
116
+ @app.command("show")
117
+ def show_artifact(
118
+ ref: str = typer.Argument(..., help="Artifact id (full or prefix)"),
119
+ ):
120
+ """Show an artifact's metadata."""
121
+ ensure_authenticated()
122
+ try:
123
+ meta = api.get(f"/artifacts/{_resolve_id(ref)}")
124
+ except APIError as e:
125
+ print_error(f"Failed: {e}")
126
+ raise typer.Exit(1)
127
+ console.print_json(json.dumps(meta))
128
+
129
+
130
+ @app.command("remove")
131
+ def remove_artifact(
132
+ ref: str = typer.Argument(..., help="Artifact id (full or prefix)"),
133
+ yes: bool = typer.Option(False, "--yes", "-y", help="Skip confirmation"),
134
+ ):
135
+ """Delete an artifact."""
136
+ ensure_authenticated()
137
+ art_id = _resolve_id(ref)
138
+ if not yes and not typer.confirm(f"Delete artifact {art_id[:8]}?"):
139
+ raise typer.Exit(0)
140
+ try:
141
+ api.delete(f"/artifacts/{art_id}")
142
+ print_success("Artifact deleted.")
143
+ except APIError as e:
144
+ print_error(f"Failed to delete: {e}")
145
+ raise typer.Exit(1)
@@ -0,0 +1,183 @@
1
+ """Evals — quality checks over agent/harness runs (deterministic + LLM judge)."""
2
+
3
+ import json
4
+ from pathlib import Path
5
+ from typing import Optional
6
+
7
+ import typer
8
+ from rich.console import Console
9
+ from rich.table import Table
10
+
11
+ from ..api import api, APIError
12
+ from ..utils.auth import ensure_authenticated
13
+ from ..utils.display import print_error, print_info, print_success
14
+
15
+ console = Console()
16
+ app = typer.Typer(help="✅ Evals — quality checks over agent runs and outputs")
17
+
18
+
19
+ @app.command("list")
20
+ def list_evals(
21
+ json_output: bool = typer.Option(False, "--json", help="Output as JSON"),
22
+ ):
23
+ """List the org's evals."""
24
+ ensure_authenticated()
25
+ try:
26
+ data = api.get("/evals")
27
+ except APIError as e:
28
+ print_error(f"Failed to list evals: {e}")
29
+ raise typer.Exit(1)
30
+ evals = data.get("evals", [])
31
+ if json_output:
32
+ console.print_json(json.dumps(evals))
33
+ return
34
+ if not evals:
35
+ print_info("No evals yet — bonito evals add <spec.json|.yaml> to create one.")
36
+ return
37
+ table = Table(title="Evals")
38
+ table.add_column("Name", style="cyan")
39
+ table.add_column("Checks", justify="right")
40
+ table.add_column("Agent")
41
+ table.add_column("Auto")
42
+ table.add_column("Enabled")
43
+ for e in evals:
44
+ table.add_row(e["name"], str(len(e["checks"])),
45
+ (e.get("agent_id") or "-")[:8],
46
+ "yes" if e["auto"] else "no",
47
+ "yes" if e["enabled"] else "no")
48
+ console.print(table)
49
+
50
+
51
+ @app.command("add")
52
+ def add_eval(
53
+ file: Path = typer.Argument(..., exists=True, readable=True,
54
+ help="Eval spec (.json or .yaml): {name, description, checks, agent_id?, auto?}"),
55
+ update: bool = typer.Option(False, "--update", "-u",
56
+ help="Overwrite if the eval already exists"),
57
+ ):
58
+ """Add an eval from a spec file."""
59
+ ensure_authenticated()
60
+ raw = file.read_text(encoding="utf-8")
61
+ if file.suffix in (".yaml", ".yml"):
62
+ import yaml
63
+ spec = yaml.safe_load(raw)
64
+ else:
65
+ spec = json.loads(raw)
66
+ try:
67
+ api.post("/evals", spec)
68
+ print_success(f"Eval '{spec.get('name')}' added.")
69
+ except APIError as e:
70
+ if update and "already exists" in str(e):
71
+ name = spec.pop("name")
72
+ api.patch(f"/evals/{name}", spec)
73
+ print_success(f"Eval '{name}' updated.")
74
+ else:
75
+ print_error(f"Failed to add eval: {e}")
76
+ raise typer.Exit(1)
77
+
78
+
79
+ @app.command("show")
80
+ def show_eval(name: str = typer.Argument(..., help="Eval name")):
81
+ """Show an eval's checks."""
82
+ ensure_authenticated()
83
+ try:
84
+ e = api.get(f"/evals/{name}")
85
+ except APIError as e2:
86
+ print_error(f"Failed: {e2}")
87
+ raise typer.Exit(1)
88
+ console.print_json(json.dumps(e))
89
+
90
+
91
+ @app.command("run")
92
+ def run_eval(
93
+ name: str = typer.Argument(..., help="Eval name"),
94
+ run_id: Optional[str] = typer.Option(None, "--run", help="Trace run id to evaluate"),
95
+ output: Optional[str] = typer.Option(None, "--output", help="Output text to evaluate"),
96
+ output_file: Optional[Path] = typer.Option(None, "--output-file",
97
+ help="Read output text from a file"),
98
+ judge_model: Optional[str] = typer.Option(None, "--judge-model",
99
+ help="Model for judge checks"),
100
+ ):
101
+ """Run an eval against a trace run and/or an output text."""
102
+ ensure_authenticated()
103
+ payload = {}
104
+ if run_id:
105
+ payload["run_id"] = run_id
106
+ if output_file:
107
+ payload["output"] = output_file.read_text(encoding="utf-8")
108
+ elif output:
109
+ payload["output"] = output
110
+ if judge_model:
111
+ payload["judge_model"] = judge_model
112
+ if not payload.get("run_id") and payload.get("output") is None:
113
+ print_error("Provide --run and/or --output/--output-file.")
114
+ raise typer.Exit(1)
115
+ try:
116
+ res = api.post(f"/evals/{name}/run", payload)
117
+ except APIError as e:
118
+ print_error(f"Eval run failed: {e}")
119
+ raise typer.Exit(1)
120
+ verdict = "[green]PASSED[/green]" if res["passed"] else "[red]FAILED[/red]"
121
+ score = f" (score {res['score']:.0f})" if res.get("score") is not None else ""
122
+ console.print(f"{verdict}{score}")
123
+ for c in res["checks"]:
124
+ mark = {"pass": "[green]✓[/green]", "fail": "[red]✗[/red]"}.get(
125
+ c["status"], "[dim]○[/dim]")
126
+ detail = c.get("reasoning") or c.get("detail") or ""
127
+ extra = ""
128
+ if "actual" in c:
129
+ extra = f" (actual {c['actual']}, limit {c.get('limit', '-')})"
130
+ elif "score" in c:
131
+ extra = f" (score {c['score']:.0f} vs min {c['min_score']})"
132
+ console.print(f" {mark} {c['type']}{extra} {detail}")
133
+
134
+
135
+ @app.command("results")
136
+ def eval_results(
137
+ name: str = typer.Argument(..., help="Eval name"),
138
+ limit: int = typer.Option(20, "--limit"),
139
+ json_output: bool = typer.Option(False, "--json", help="Output as JSON"),
140
+ ):
141
+ """Show recent results for an eval."""
142
+ ensure_authenticated()
143
+ try:
144
+ data = api.get(f"/evals/{name}/results", params={"limit": limit})
145
+ except APIError as e:
146
+ print_error(f"Failed: {e}")
147
+ raise typer.Exit(1)
148
+ results = data.get("results", [])
149
+ if json_output:
150
+ console.print_json(json.dumps(results))
151
+ return
152
+ if not results:
153
+ print_info("No results yet.")
154
+ return
155
+ table = Table(title=f"Results — {name}")
156
+ table.add_column("When")
157
+ table.add_column("Verdict")
158
+ table.add_column("Score", justify="right")
159
+ table.add_column("Trigger")
160
+ table.add_column("Run")
161
+ for r in results:
162
+ table.add_row((r.get("created_at") or "")[:19],
163
+ "[green]pass[/green]" if r["passed"] else "[red]fail[/red]",
164
+ f"{r['score']:.0f}" if r.get("score") is not None else "-",
165
+ r["trigger"], (r.get("run_id") or "-")[:8])
166
+ console.print(table)
167
+
168
+
169
+ @app.command("remove")
170
+ def remove_eval(
171
+ name: str = typer.Argument(...),
172
+ yes: bool = typer.Option(False, "--yes", "-y", help="Skip confirmation"),
173
+ ):
174
+ """Delete an eval (results cascade)."""
175
+ ensure_authenticated()
176
+ if not yes and not typer.confirm(f"Delete eval '{name}' and its results?"):
177
+ raise typer.Exit(0)
178
+ try:
179
+ api.delete(f"/evals/{name}")
180
+ print_success(f"Eval '{name}' deleted.")
181
+ except APIError as e:
182
+ print_error(f"Failed to delete: {e}")
183
+ raise typer.Exit(1)