otto-cli-agent 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent/README.md +22 -0
- agent/__init__.py +0 -0
- agent/cli/README.md +77 -0
- agent/cli/__init__.py +0 -0
- agent/cli/art.py +371 -0
- agent/cli/chat.py +309 -0
- agent/cli/clipboard.py +106 -0
- agent/cli/context.py +56 -0
- agent/cli/doctor.py +60 -0
- agent/cli/errors.py +58 -0
- agent/cli/eval.py +130 -0
- agent/cli/eval_claw.py +414 -0
- agent/cli/eval_compaction.py +106 -0
- agent/cli/eval_hle.py +92 -0
- agent/cli/eval_memory.py +253 -0
- agent/cli/eval_swe.py +172 -0
- agent/cli/lessons.py +97 -0
- agent/cli/main.py +67 -0
- agent/cli/modals.py +570 -0
- agent/cli/models.py +64 -0
- agent/cli/output.py +54 -0
- agent/cli/route.py +78 -0
- agent/cli/sessions.py +125 -0
- agent/cli/setup_screen.py +562 -0
- agent/cli/shell.py +548 -0
- agent/cli/tui.py +1807 -0
- agent/cli/ui.py +14 -0
- agent/cli/usage_panel.py +159 -0
- agent/config/README.md +7 -0
- agent/config/__init__.py +0 -0
- agent/config/envfile.py +76 -0
- agent/eval/README.md +76 -0
- agent/eval/__init__.py +0 -0
- agent/eval/claw_bench.py +1031 -0
- agent/eval/compaction_bench.py +229 -0
- agent/eval/data/README.md +10 -0
- agent/eval/data/claw/README.md +108 -0
- agent/eval/data/claw/llm_judge-gemini.patch +57 -0
- agent/eval/data/claw/otto.yaml +35 -0
- agent/eval/failures.py +276 -0
- agent/eval/golden/README.md +33 -0
- agent/eval/golden/code_01.json +6 -0
- agent/eval/golden/code_02.json +6 -0
- agent/eval/golden/code_03.json +6 -0
- agent/eval/golden/code_04.json +6 -0
- agent/eval/golden/code_05.json +6 -0
- agent/eval/golden/code_06.json +6 -0
- agent/eval/golden/math_01.json +6 -0
- agent/eval/golden/math_02.json +6 -0
- agent/eval/golden/math_03.json +6 -0
- agent/eval/golden/math_04.json +6 -0
- agent/eval/golden/math_05.json +6 -0
- agent/eval/golden/math_06.json +6 -0
- agent/eval/golden/nphard_gcp_01.json +6 -0
- agent/eval/golden/nphard_ksp_01.json +6 -0
- agent/eval/golden/nphard_math_binpacking_01.json +6 -0
- agent/eval/golden/nphard_math_clique_01.json +6 -0
- agent/eval/golden/nphard_math_setcover_01.json +6 -0
- agent/eval/golden/nphard_math_subsetsum_01.json +6 -0
- agent/eval/golden/nphard_tsp_01.json +6 -0
- agent/eval/golden/nphard_tsp_02.json +6 -0
- agent/eval/hle_bench.py +273 -0
- agent/eval/langfuse_sync.py +172 -0
- agent/eval/memory_bench.py +538 -0
- agent/eval/runner.py +174 -0
- agent/eval/single_agent.py +120 -0
- agent/eval/swe_bench.py +604 -0
- agent/eval/terminal_bench.py +345 -0
- agent/memory/README.md +102 -0
- agent/memory/__init__.py +42 -0
- agent/memory/embeddings.py +302 -0
- agent/memory/hashing.py +15 -0
- agent/memory/lessons.py +483 -0
- agent/memory/queue.py +531 -0
- agent/memory/retrieval.py +493 -0
- agent/memory/session.py +60 -0
- agent/memory/sessions.py +436 -0
- agent/memory/store.py +429 -0
- agent/memory/tokens.py +60 -0
- agent/memory/wiring.py +146 -0
- agent/pipeline/README.md +135 -0
- agent/pipeline/__init__.py +0 -0
- agent/pipeline/browsing.py +609 -0
- agent/pipeline/budget.py +403 -0
- agent/pipeline/codemap.py +254 -0
- agent/pipeline/evidence.py +325 -0
- agent/pipeline/execution.py +67 -0
- agent/pipeline/modes.py +137 -0
- agent/pipeline/native.py +1137 -0
- agent/pipeline/nodes.py +3644 -0
- agent/pipeline/pricing.py +209 -0
- agent/pipeline/progress.py +139 -0
- agent/pipeline/rag.py +139 -0
- agent/pipeline/research.py +1325 -0
- agent/pipeline/run.py +528 -0
- agent/pipeline/screen.py +77 -0
- agent/pipeline/state.py +220 -0
- agent/pipeline/toolkit.py +328 -0
- agent/pipeline/tools.py +1990 -0
- agent/pipeline/tracing.py +147 -0
- agent/pipeline/usage.py +251 -0
- agent/pipeline/vision.py +84 -0
- agent/pipeline/walkthrough.py +735 -0
- agent/pipeline/workspace.py +229 -0
- agent/router/README.md +60 -0
- agent/router/__init__.py +0 -0
- agent/router/automap.py +114 -0
- agent/router/health.py +229 -0
- agent/router/llm_provider/README.md +38 -0
- agent/router/llm_provider/__init__.py +202 -0
- agent/router/llm_provider/anthropic_provider.py +128 -0
- agent/router/llm_provider/base.py +507 -0
- agent/router/llm_provider/custom.py +152 -0
- agent/router/llm_provider/gemini_provider.py +122 -0
- agent/router/llm_provider/inception_provider.py +687 -0
- agent/router/llm_provider/openai_provider.py +151 -0
- agent/router/llm_provider/retired.py +145 -0
- agent/router/llm_provider/temperature.py +371 -0
- agent/router/mapping.py +579 -0
- agent/router/outcomes.py +363 -0
- agent/router/overrides.py +389 -0
- agent/router/reload.py +28 -0
- agent/router/router.py +413 -0
- agent/router/setup.py +123 -0
- otto_cli_agent-0.1.0.dist-info/METADATA +115 -0
- otto_cli_agent-0.1.0.dist-info/RECORD +129 -0
- otto_cli_agent-0.1.0.dist-info/WHEEL +4 -0
- otto_cli_agent-0.1.0.dist-info/entry_points.txt +2 -0
- otto_cli_agent-0.1.0.dist-info/licenses/LICENSE +21 -0
agent/cli/eval.py
ADDED
|
@@ -0,0 +1,130 @@
|
|
|
1
|
+
"""`otto eval`: run the golden dataset through the real pipeline and report
|
|
2
|
+
pass/fail per item. See agent/eval/runner.py for what "checked" means, and
|
|
3
|
+
agent/eval/golden/ for the items themselves.
|
|
4
|
+
|
|
5
|
+
By default this tracks the run as a Langfuse Dataset Run (agent/eval/
|
|
6
|
+
langfuse_sync.py): the golden set is synced to a Langfuse Dataset once,
|
|
7
|
+
then every `otto eval` invocation becomes one named, comparable run
|
|
8
|
+
against it in the Langfuse UI -- every item's trace, its pass/fail score,
|
|
9
|
+
and the run as a whole. Pass --no-experiment for the old local-only path
|
|
10
|
+
(no Langfuse project needed -- useful offline or in CI without Langfuse
|
|
11
|
+
creds), which prints a plain pass/fail table and nothing else.
|
|
12
|
+
|
|
13
|
+
No --agents option (2026-09-10): the router/planner/solver/summarizer/
|
|
14
|
+
finder/evaluator graph that replaced the swarm pipeline has nothing to
|
|
15
|
+
size -- see agent/pipeline/run.py's module docstring.
|
|
16
|
+
"""
|
|
17
|
+
from typing import Annotated, Optional
|
|
18
|
+
|
|
19
|
+
import typer
|
|
20
|
+
from rich import box
|
|
21
|
+
from rich.table import Table
|
|
22
|
+
|
|
23
|
+
from agent.cli.ui import err, out
|
|
24
|
+
from agent.eval.runner import run_golden
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def eval_cmd(
|
|
28
|
+
domain: Annotated[Optional[str], typer.Option(help="Only 'code' or 'math'.")] = None,
|
|
29
|
+
experiment: Annotated[
|
|
30
|
+
bool,
|
|
31
|
+
typer.Option(
|
|
32
|
+
"--experiment/--no-experiment",
|
|
33
|
+
help="Track this run as a Langfuse Dataset Run (default) vs. a local-only table.",
|
|
34
|
+
),
|
|
35
|
+
] = True,
|
|
36
|
+
run_name: Annotated[
|
|
37
|
+
Optional[str], typer.Option(help="Name for the Langfuse experiment run (--experiment only).")
|
|
38
|
+
] = None,
|
|
39
|
+
) -> None:
|
|
40
|
+
"""Run the golden dataset through the pipeline and report pass/fail."""
|
|
41
|
+
if domain is not None and domain not in ("code", "math"):
|
|
42
|
+
err.print("[bad]--domain must be 'code' or 'math'[/]")
|
|
43
|
+
raise typer.Exit(2)
|
|
44
|
+
|
|
45
|
+
if experiment:
|
|
46
|
+
_run_as_langfuse_experiment(domain=domain, run_name=run_name)
|
|
47
|
+
return
|
|
48
|
+
|
|
49
|
+
with err.status("running golden set…"):
|
|
50
|
+
results = run_golden(only_domain=domain)
|
|
51
|
+
|
|
52
|
+
if not results:
|
|
53
|
+
out.print("[muted]no golden items matched[/]")
|
|
54
|
+
return
|
|
55
|
+
|
|
56
|
+
t = Table(box=box.SIMPLE, header_style="muted")
|
|
57
|
+
t.add_column("id", style="spec")
|
|
58
|
+
t.add_column("domain", style="muted")
|
|
59
|
+
t.add_column("result")
|
|
60
|
+
t.add_column("seconds", justify="right")
|
|
61
|
+
t.add_column("evidence", style="muted")
|
|
62
|
+
passed = 0
|
|
63
|
+
for r in results:
|
|
64
|
+
mark = "[ok]pass[/]" if r.passed else "[bad]fail[/]"
|
|
65
|
+
passed += r.passed
|
|
66
|
+
evidence = r.evidence if len(r.evidence) <= 80 else r.evidence[:80] + "…"
|
|
67
|
+
t.add_row(r.item_id, r.domain, mark, f"{r.seconds:.1f}", evidence.replace("\n", " "))
|
|
68
|
+
out.print(t)
|
|
69
|
+
out.print(f"[muted]{passed}/{len(results)} passed[/]")
|
|
70
|
+
|
|
71
|
+
_print_full_failures((r.item_id, r.evidence) for r in results if not r.passed)
|
|
72
|
+
if passed < len(results):
|
|
73
|
+
raise typer.Exit(1)
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def _print_full_failures(failures) -> None:
|
|
77
|
+
"""The overview table truncates evidence to 80 chars for readability --
|
|
78
|
+
fine for a scan, useless for actually debugging a failure (it cuts a
|
|
79
|
+
checker's stderr right where the real exception would start). Print
|
|
80
|
+
each failing item's full evidence separately so that's never the
|
|
81
|
+
reason you have to go dig through the Langfuse UI.
|
|
82
|
+
"""
|
|
83
|
+
failures = list(failures)
|
|
84
|
+
if not failures:
|
|
85
|
+
return
|
|
86
|
+
out.print("\n[bad]failures, in full:[/]")
|
|
87
|
+
for item_id, evidence in failures:
|
|
88
|
+
out.print(f"[spec]{item_id}[/]")
|
|
89
|
+
out.print(evidence)
|
|
90
|
+
out.print("")
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def _run_as_langfuse_experiment(*, domain: Optional[str], run_name: Optional[str]) -> None:
|
|
94
|
+
from agent.eval.langfuse_sync import run_golden_experiment
|
|
95
|
+
|
|
96
|
+
with err.status("syncing golden set to langfuse and running experiment…"):
|
|
97
|
+
try:
|
|
98
|
+
result = run_golden_experiment(only_domain=domain, run_name=run_name)
|
|
99
|
+
except Exception as exc:
|
|
100
|
+
err.print(f"[bad]langfuse experiment failed: {exc}[/]")
|
|
101
|
+
err.print("[muted]is LANGFUSE_PUBLIC_KEY / LANGFUSE_SECRET_KEY set? falling back to --no-experiment still works offline.[/]")
|
|
102
|
+
raise typer.Exit(1) from exc
|
|
103
|
+
|
|
104
|
+
t = Table(box=box.SIMPLE, header_style="muted")
|
|
105
|
+
t.add_column("id", style="spec")
|
|
106
|
+
t.add_column("result")
|
|
107
|
+
t.add_column("evidence", style="muted")
|
|
108
|
+
all_passed = True
|
|
109
|
+
for item_result in result.item_results:
|
|
110
|
+
golden_id = getattr(item_result.item, "id", "?")
|
|
111
|
+
ev = next((e for e in item_result.evaluations if e.name == "golden_pass"), None)
|
|
112
|
+
item_passed = bool(ev.value) if ev is not None else False
|
|
113
|
+
all_passed &= item_passed
|
|
114
|
+
mark = "[ok]pass[/]" if item_passed else "[bad]fail[/]"
|
|
115
|
+
comment = (ev.comment or "") if ev is not None else "no golden_pass score"
|
|
116
|
+
comment = comment if len(comment) <= 80 else comment[:80] + "…"
|
|
117
|
+
t.add_row(golden_id, mark, comment.replace("\n", " "))
|
|
118
|
+
out.print(t)
|
|
119
|
+
|
|
120
|
+
n = len(result.item_results)
|
|
121
|
+
n_passed = sum(
|
|
122
|
+
1 for r in result.item_results
|
|
123
|
+
if any(e.name == "golden_pass" and e.value for e in r.evaluations)
|
|
124
|
+
)
|
|
125
|
+
out.print(f"[muted]{n_passed}/{n} passed[/]")
|
|
126
|
+
if result.dataset_run_url:
|
|
127
|
+
out.print(f"[muted]langfuse dataset run: {result.dataset_run_url}[/]")
|
|
128
|
+
|
|
129
|
+
if not all_passed:
|
|
130
|
+
raise typer.Exit(1)
|
agent/cli/eval_claw.py
ADDED
|
@@ -0,0 +1,414 @@
|
|
|
1
|
+
"""`otto eval-claw` -- Claw-Eval's 300 tasks against Otto's own agent.
|
|
2
|
+
|
|
3
|
+
Their `claw-eval run` can only point at a model endpoint, so the number it
|
|
4
|
+
gives is about a model inside their agent loop. This command runs OTTO on the
|
|
5
|
+
task -- its graph, its memory, its tools, its evaluator -- and hands the
|
|
6
|
+
resulting trace to their own graders, so the score is about the agent.
|
|
7
|
+
See agent/eval/claw_bench.py for the seam that makes that possible.
|
|
8
|
+
|
|
9
|
+
Needs a Claw-Eval checkout (--claw-root or $CLAW_EVAL_ROOT) with its
|
|
10
|
+
requirements installed, and Docker for the 169 tasks whose files live in a
|
|
11
|
+
container. Start with --limit: every task is a full agent run.
|
|
12
|
+
"""
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
import json
|
|
16
|
+
import time
|
|
17
|
+
from contextlib import ExitStack
|
|
18
|
+
from pathlib import Path
|
|
19
|
+
from types import SimpleNamespace
|
|
20
|
+
from typing import Optional
|
|
21
|
+
|
|
22
|
+
import typer
|
|
23
|
+
from typing_extensions import Annotated
|
|
24
|
+
|
|
25
|
+
from agent.cli.ui import err, out
|
|
26
|
+
from agent.eval import failures
|
|
27
|
+
from agent.memory.lessons import bind_bank, read_only
|
|
28
|
+
from agent.memory.store import MemoryStore
|
|
29
|
+
from agent.router.outcomes import bind_log
|
|
30
|
+
from agent.router.outcomes import read_only as routing_read_only
|
|
31
|
+
from agent.eval.claw_bench import (
|
|
32
|
+
ClawEvalUnavailable,
|
|
33
|
+
claw_root,
|
|
34
|
+
grading_fingerprint,
|
|
35
|
+
load_claw,
|
|
36
|
+
missing_service_keys,
|
|
37
|
+
run_task_file,
|
|
38
|
+
select_tasks,
|
|
39
|
+
split_tasks,
|
|
40
|
+
)
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
#: How many trials per task before a Claw-Eval number is evidence of
|
|
44
|
+
#: anything.
|
|
45
|
+
#:
|
|
46
|
+
#: Measured, not chosen. Task T093 on identical code and configuration, three
|
|
47
|
+
#: runs: 0.86, 0.60, 0.96 -- a spread of 0.36 on a 0-1 scale. Over two trials
|
|
48
|
+
#: pass^2 came out 0.25 against pass@2 of 1.00: it got there once and not
|
|
49
|
+
#: twice. Completion is LLM-judged for 260 of the 300 tasks, which is where
|
|
50
|
+
#: most of that comes from.
|
|
51
|
+
#:
|
|
52
|
+
#: The consequence is retroactive and worth stating plainly: any single-run
|
|
53
|
+
#: before/after difference smaller than about 0.3 on this benchmark is inside
|
|
54
|
+
#: the noise. Much of the optimisation work on this branch was read from one
|
|
55
|
+
#: run per task, and the only difference in that set large enough to survive
|
|
56
|
+
#: was a crash going from 0.00 to 0.955.
|
|
57
|
+
#:
|
|
58
|
+
#: Three, matching what the issues here already ask for when they say how to
|
|
59
|
+
#: settle something ("--trials 3").
|
|
60
|
+
MIN_TRIALS_FOR_EVIDENCE = 3
|
|
61
|
+
|
|
62
|
+
#: The observed spread above, quoted in the warning so the number a reader is
|
|
63
|
+
#: being told to distrust comes with the reason.
|
|
64
|
+
OBSERVED_SINGLE_RUN_SPREAD = 0.36
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def _summary(outcomes: list, claw=None) -> dict:
|
|
68
|
+
scored = [o for o in outcomes if o is not None]
|
|
69
|
+
if not scored:
|
|
70
|
+
return {"tasks": 0}
|
|
71
|
+
n = len(scored)
|
|
72
|
+
summary = {
|
|
73
|
+
"tasks": n,
|
|
74
|
+
"passed": sum(1 for o in scored if o.passed),
|
|
75
|
+
"pass_rate": round(sum(1 for o in scored if o.passed) / n, 4),
|
|
76
|
+
"mean_task_score": round(sum(o.task_score for o in scored) / n, 4),
|
|
77
|
+
"mean_completion": round(sum(o.completion for o in scored) / n, 4),
|
|
78
|
+
"mean_robustness": round(sum(o.robustness for o in scored) / n, 4),
|
|
79
|
+
"mean_communication": round(sum(o.communication for o in scored) / n, 4),
|
|
80
|
+
"errors": sum(1 for o in scored if o.error),
|
|
81
|
+
"mean_wall_time_s": round(sum(o.wall_time_s for o in scored) / n, 2),
|
|
82
|
+
# The cost axis. None of the 300 graders read it, so without this a
|
|
83
|
+
# change that doubles spend for a tenth of a point reads as a win.
|
|
84
|
+
"mean_model_calls": round(sum(o.model_calls for o in scored) / n, 2),
|
|
85
|
+
"total_model_calls": sum(o.model_calls for o in scored),
|
|
86
|
+
}
|
|
87
|
+
|
|
88
|
+
# Always present, so a reader of otto_summary.json never has to infer the
|
|
89
|
+
# trial count from whether a reliability key happens to exist.
|
|
90
|
+
summary["trials"] = min((len(o.trials) for o in scored), default=0)
|
|
91
|
+
#: Whether this run is enough trials to be read as a measurement at all --
|
|
92
|
+
#: MIN_TRIALS_FOR_EVIDENCE. False does NOT mean the numbers are wrong; it
|
|
93
|
+
#: means the difference between them and another run's is not attributable
|
|
94
|
+
#: to anything. Carried in the JSON as well as printed, because the JSON is
|
|
95
|
+
#: what gets pasted into a comparison months later.
|
|
96
|
+
summary["is_evidence"] = summary["trials"] >= MIN_TRIALS_FOR_EVIDENCE
|
|
97
|
+
|
|
98
|
+
# What KIND of failure, and what the tools cost -- agent/eval/failures.py,
|
|
99
|
+
# read off the action lines every run already produced, with no model
|
|
100
|
+
# call. A score says something got worse; "38% of the failures changed no
|
|
101
|
+
# files at all" says where to look.
|
|
102
|
+
failed = {o.task_id: getattr(o, "actions", None)
|
|
103
|
+
for o in scored if not o.passed}
|
|
104
|
+
if failed:
|
|
105
|
+
summary["failures"] = failures.summarise(failed)
|
|
106
|
+
summary["tool_cost"] = failures.total_tool_cost(
|
|
107
|
+
getattr(o, "actions", None) for o in scored
|
|
108
|
+
).to_dict()
|
|
109
|
+
|
|
110
|
+
repeated = [o for o in scored if len(o.trials) > 1]
|
|
111
|
+
if repeated and claw is not None:
|
|
112
|
+
k = min(len(o.trials) for o in repeated)
|
|
113
|
+
summary["trials"] = k
|
|
114
|
+
# Per task first, then averaged: pass^k over a task's own repeats says
|
|
115
|
+
# "does it do this reliably", which is the property a self-improving
|
|
116
|
+
# loop is most able to fake by finding one lucky trial.
|
|
117
|
+
summary["mean_pass_hat_k"] = round(
|
|
118
|
+
sum(claw.compute_pass_hat_k(o.trials, k=k) for o in repeated) / len(repeated), 4
|
|
119
|
+
)
|
|
120
|
+
summary["mean_pass_at_k"] = round(
|
|
121
|
+
sum(claw.compute_pass_at_k(o.trials, k=k) for o in repeated) / len(repeated), 4
|
|
122
|
+
)
|
|
123
|
+
summary["score_spread"] = round(
|
|
124
|
+
sum(max(o.trials) - min(o.trials) for o in repeated) / len(repeated), 4
|
|
125
|
+
)
|
|
126
|
+
return summary
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
def eval_claw_cmd(
|
|
130
|
+
claw_root_opt: Annotated[
|
|
131
|
+
Optional[Path],
|
|
132
|
+
typer.Option("--claw-root", help="Claw-Eval checkout; defaults to $CLAW_EVAL_ROOT."),
|
|
133
|
+
] = None,
|
|
134
|
+
tag: Annotated[
|
|
135
|
+
Optional[str],
|
|
136
|
+
typer.Option(help="Capability to run: general, multimodal, user_agent, multi_service."),
|
|
137
|
+
] = None,
|
|
138
|
+
task: Annotated[
|
|
139
|
+
Optional[str], typer.Option(help="Substring of a task directory name, e.g. 'T01' or 'M001'."),
|
|
140
|
+
] = None,
|
|
141
|
+
limit: Annotated[Optional[int], typer.Option(help="Run at most this many tasks.")] = 5,
|
|
142
|
+
architecture: Annotated[
|
|
143
|
+
str, typer.Option(help="'graph' (the real pipeline) or 'single' (one-conversation control)."),
|
|
144
|
+
] = "graph",
|
|
145
|
+
config: Annotated[
|
|
146
|
+
Optional[Path], typer.Option(help="Claw-Eval config.yaml (judge, sandbox, user-agent models)."),
|
|
147
|
+
] = None,
|
|
148
|
+
trace_dir: Annotated[
|
|
149
|
+
Optional[Path], typer.Option(help="Where to write traces (default: <checkout>/traces/otto_<time>).")
|
|
150
|
+
] = None,
|
|
151
|
+
no_judge: Annotated[bool, typer.Option("--no-judge", help="Skip the LLM judge.")] = False,
|
|
152
|
+
port_offset: Annotated[int, typer.Option(help="Shift every mock service port, for parallel runs.")] = 0,
|
|
153
|
+
max_seconds: Annotated[
|
|
154
|
+
Optional[float],
|
|
155
|
+
typer.Option(help="Cap each task's budget below its own (tasks allow 120-900s). "
|
|
156
|
+
"Cheaper samples, and lower scores -- say so when reporting."),
|
|
157
|
+
] = None,
|
|
158
|
+
trials: Annotated[
|
|
159
|
+
int,
|
|
160
|
+
typer.Option(help="Run each task this many times and report pass^k and the "
|
|
161
|
+
"spread. One run cannot tell a real change from judge "
|
|
162
|
+
"variance -- 260 of the 300 tasks are LLM-judged."),
|
|
163
|
+
] = 1,
|
|
164
|
+
split: Annotated[
|
|
165
|
+
str,
|
|
166
|
+
typer.Option(help="'all', 'dev' (tune against these), or 'holdout' (never "
|
|
167
|
+
"tuned against; the honest number). The split is a hash "
|
|
168
|
+
"of the task id, so it does not move between runs."),
|
|
169
|
+
] = "all",
|
|
170
|
+
lesson_bank: Annotated[
|
|
171
|
+
Optional[Path],
|
|
172
|
+
typer.Option(help="Lesson bank to read and write (default: ~/.otto/memory/"
|
|
173
|
+
"lessons.db). Point a measurement at its own file so it "
|
|
174
|
+
"does not learn from -- or teach -- your working one."),
|
|
175
|
+
] = None,
|
|
176
|
+
no_learning: Annotated[
|
|
177
|
+
bool,
|
|
178
|
+
typer.Option("--no-learning", help="Read no lessons and write none. This is "
|
|
179
|
+
"the compute-matched baseline every "
|
|
180
|
+
"self-improvement claim has to be shown "
|
|
181
|
+
"beside."),
|
|
182
|
+
] = False,
|
|
183
|
+
as_json: Annotated[bool, typer.Option("--json", help="Print the full report as JSON.")] = False,
|
|
184
|
+
) -> None:
|
|
185
|
+
if architecture not in {"graph", "single"}:
|
|
186
|
+
err.print(f"architecture must be 'graph' or 'single', not {architecture!r}")
|
|
187
|
+
raise typer.Exit(2)
|
|
188
|
+
if split not in {"all", "dev", "holdout"}:
|
|
189
|
+
err.print(f"split must be 'all', 'dev' or 'holdout', not {split!r}")
|
|
190
|
+
raise typer.Exit(2)
|
|
191
|
+
if trials < 1:
|
|
192
|
+
err.print("trials must be at least 1")
|
|
193
|
+
raise typer.Exit(2)
|
|
194
|
+
|
|
195
|
+
try:
|
|
196
|
+
root = claw_root(claw_root_opt)
|
|
197
|
+
claw = load_claw(root)
|
|
198
|
+
except ClawEvalUnavailable as exc:
|
|
199
|
+
err.print(str(exc))
|
|
200
|
+
raise typer.Exit(2) from None
|
|
201
|
+
|
|
202
|
+
cfg = claw.load_config(str(config) if config else None)
|
|
203
|
+
# Their own factory, so the judge model, key and base URL come from the
|
|
204
|
+
# same config a `claw-eval run` would use rather than from a second copy.
|
|
205
|
+
judge = claw.cli._make_judge(cfg, SimpleNamespace(no_judge=no_judge, judge_model=None))
|
|
206
|
+
|
|
207
|
+
# Split BEFORE the limit, so --limit takes the first N of the chosen side
|
|
208
|
+
# rather than trimming the pool and then splitting a different set each
|
|
209
|
+
# time the pool changes.
|
|
210
|
+
tasks = select_tasks(root, tag=tag, pattern=task)
|
|
211
|
+
if split != "all":
|
|
212
|
+
development, reserved = split_tasks(tasks)
|
|
213
|
+
tasks = reserved if split == "holdout" else development
|
|
214
|
+
if limit:
|
|
215
|
+
tasks = tasks[:limit]
|
|
216
|
+
if not tasks:
|
|
217
|
+
err.print("no tasks matched")
|
|
218
|
+
raise typer.Exit(1)
|
|
219
|
+
|
|
220
|
+
out_dir = Path(trace_dir) if trace_dir else root / "traces" / f"otto-{architecture}-{time.strftime('%y%m%d-%H%M')}"
|
|
221
|
+
out_dir.mkdir(parents=True, exist_ok=True)
|
|
222
|
+
err.print(f"{len(tasks)} task(s) -> {out_dir}")
|
|
223
|
+
|
|
224
|
+
# What the loop is allowed to learn, decided once, here, rather than per
|
|
225
|
+
# task -- the discipline is a property of the whole measurement.
|
|
226
|
+
#
|
|
227
|
+
# --no-learning nothing read, nothing written. The baseline.
|
|
228
|
+
# --split holdout lessons read, none written back. Tests TRANSFER: the
|
|
229
|
+
# loop never tunes against these tasks, which is the
|
|
230
|
+
# difference one evolved system got 31.7 points wrong.
|
|
231
|
+
# otherwise read and write. This is where lessons come from.
|
|
232
|
+
learning = ExitStack()
|
|
233
|
+
if no_learning:
|
|
234
|
+
learning.enter_context(bind_bank(None))
|
|
235
|
+
# Routing adapts from observed outcomes too, so the baseline has to
|
|
236
|
+
# hold that still as well. A "no learning" arm that quietly reordered
|
|
237
|
+
# the model chain partway through would be measuring two things.
|
|
238
|
+
learning.enter_context(bind_log(None))
|
|
239
|
+
else:
|
|
240
|
+
if lesson_bank:
|
|
241
|
+
learning.enter_context(bind_bank(MemoryStore(lesson_bank)))
|
|
242
|
+
# Beside the bank, so a measurement's routing evidence travels
|
|
243
|
+
# with its lessons instead of leaking into the working install.
|
|
244
|
+
learning.enter_context(bind_log(Path(lesson_bank).with_suffix(".seats.db")))
|
|
245
|
+
if split == "holdout":
|
|
246
|
+
# Read what the development runs learned, write nothing back --
|
|
247
|
+
# lessons and routing evidence alike. That is what makes the
|
|
248
|
+
# held-out number answer "does this TRANSFER" rather than "did the
|
|
249
|
+
# loop find something that works on what it was tuned on".
|
|
250
|
+
learning.enter_context(read_only())
|
|
251
|
+
learning.enter_context(routing_read_only())
|
|
252
|
+
err.print(
|
|
253
|
+
"learning: " + ("off (baseline)" if no_learning else
|
|
254
|
+
"read-only (held out)" if split == "holdout" else "on")
|
|
255
|
+
)
|
|
256
|
+
|
|
257
|
+
outcomes = []
|
|
258
|
+
try:
|
|
259
|
+
_run_tasks(claw, tasks, outcomes, out_dir=out_dir, cfg=cfg, judge=judge,
|
|
260
|
+
architecture=architecture, port_offset=port_offset,
|
|
261
|
+
max_seconds=max_seconds, trials=trials)
|
|
262
|
+
finally:
|
|
263
|
+
learning.close()
|
|
264
|
+
|
|
265
|
+
report = _build_report(
|
|
266
|
+
outcomes, claw, architecture=architecture, tag=tag, split=split,
|
|
267
|
+
trials=trials, max_seconds=max_seconds, out_dir=out_dir, cfg=cfg,
|
|
268
|
+
judge=judge, no_learning=no_learning,
|
|
269
|
+
)
|
|
270
|
+
(out_dir / "otto_summary.json").write_text(json.dumps(report, indent=2))
|
|
271
|
+
|
|
272
|
+
if as_json:
|
|
273
|
+
out.print_json(data=report)
|
|
274
|
+
return
|
|
275
|
+
_print_summary(report, out_dir, split)
|
|
276
|
+
|
|
277
|
+
|
|
278
|
+
def _run_tasks(claw, tasks, outcomes, *, out_dir, cfg, judge, architecture,
|
|
279
|
+
port_offset, max_seconds, trials) -> None:
|
|
280
|
+
for i, task_yaml in enumerate(tasks, 1):
|
|
281
|
+
name = task_yaml.parent.name
|
|
282
|
+
err.print(f"[{i}/{len(tasks)}] {name}")
|
|
283
|
+
missing = missing_service_keys(claw.TaskDefinition.from_yaml(task_yaml))
|
|
284
|
+
if missing:
|
|
285
|
+
err.print(
|
|
286
|
+
f" [warn]{', '.join(missing)} not set[/] -- this task's service "
|
|
287
|
+
"reaches the real internet and will return nothing, so the score "
|
|
288
|
+
"below measures the environment, not the agent"
|
|
289
|
+
)
|
|
290
|
+
try:
|
|
291
|
+
outcome = run_task_file(
|
|
292
|
+
claw, task_yaml,
|
|
293
|
+
trace_dir=out_dir, cfg=cfg, judge=judge,
|
|
294
|
+
architecture=architecture, port_offset=port_offset,
|
|
295
|
+
max_seconds=max_seconds, trials=trials,
|
|
296
|
+
)
|
|
297
|
+
except Exception as exc:
|
|
298
|
+
# One task's container or service failing is not a reason to lose
|
|
299
|
+
# the other 299 -- report it and carry on, the way their batch does.
|
|
300
|
+
err.print(f" [bad]harness error[/] {type(exc).__name__}: {exc}")
|
|
301
|
+
continue
|
|
302
|
+
outcomes.append(outcome)
|
|
303
|
+
flag = "[ok]pass[/]" if outcome.passed else "[warn]fail[/]"
|
|
304
|
+
for item in outcome.checklist:
|
|
305
|
+
err.print(f" [{item.get('status', '?')}] {item.get('text', '')[:90]}")
|
|
306
|
+
detail = f" ({outcome.error})" if outcome.error else ""
|
|
307
|
+
spread = (
|
|
308
|
+
" trials=" + "/".join(f"{t:.2f}" for t in outcome.trials)
|
|
309
|
+
if len(outcome.trials) > 1 else ""
|
|
310
|
+
)
|
|
311
|
+
out.print(
|
|
312
|
+
f" {flag} score={outcome.task_score:.2f} "
|
|
313
|
+
f"completion={outcome.completion:.2f} service-tools={outcome.tool_calls} "
|
|
314
|
+
f"actions={outcome.agent_actions} calls={outcome.model_calls} "
|
|
315
|
+
f"{outcome.wall_time_s:.0f}s{spread}{detail}"
|
|
316
|
+
)
|
|
317
|
+
|
|
318
|
+
|
|
319
|
+
def _build_report(outcomes, claw, *, architecture, tag, split, trials,
|
|
320
|
+
max_seconds, out_dir, cfg, judge, no_learning) -> dict:
|
|
321
|
+
return {
|
|
322
|
+
"architecture": architecture,
|
|
323
|
+
"learning": "off" if no_learning else "read-only" if split == "holdout" else "on",
|
|
324
|
+
"tag": tag,
|
|
325
|
+
"max_seconds": max_seconds,
|
|
326
|
+
"split": split,
|
|
327
|
+
"trials": trials,
|
|
328
|
+
"trace_dir": str(out_dir),
|
|
329
|
+
# What the scores mean. Two reports whose fingerprints differ are not
|
|
330
|
+
# comparable, however similar the numbers look.
|
|
331
|
+
"grading": grading_fingerprint(claw, cfg, judge),
|
|
332
|
+
"summary": _summary(outcomes, claw),
|
|
333
|
+
"tasks": [
|
|
334
|
+
{
|
|
335
|
+
"task_id": o.task_id, "task_score": o.task_score, "passed": o.passed,
|
|
336
|
+
"completion": o.completion, "robustness": o.robustness,
|
|
337
|
+
"communication": o.communication, "safety": o.safety,
|
|
338
|
+
"tool_calls": o.tool_calls, "agent_actions": o.agent_actions,
|
|
339
|
+
"checklist": o.checklist, "model_calls": o.model_calls,
|
|
340
|
+
"trials": o.trials,
|
|
341
|
+
"wall_time_s": o.wall_time_s, "error": o.error,
|
|
342
|
+
}
|
|
343
|
+
for o in outcomes
|
|
344
|
+
],
|
|
345
|
+
}
|
|
346
|
+
|
|
347
|
+
|
|
348
|
+
def _print_summary(report: dict, out_dir: Path, split: str) -> None:
|
|
349
|
+
s = report["summary"]
|
|
350
|
+
|
|
351
|
+
# Said BEFORE the numbers, not after. A caveat printed underneath a mean
|
|
352
|
+
# score is read after the score has already been believed, and the whole
|
|
353
|
+
# failure this guards against is a single run being quoted as a result.
|
|
354
|
+
if s.get("tasks") and not s.get("is_evidence", True):
|
|
355
|
+
trials = s.get("trials", 1)
|
|
356
|
+
err.print(
|
|
357
|
+
f"\n[bad]NOT EVIDENCE: {trials} trial(s) per task.[/] "
|
|
358
|
+
f"[warn]Identical code and configuration have scored "
|
|
359
|
+
f"{OBSERVED_SINGLE_RUN_SPREAD:.2f} apart on a single task here, so "
|
|
360
|
+
f"any before/after difference below roughly 0.3 in what follows is "
|
|
361
|
+
f"noise. Do not quote these numbers as a result.[/]\n"
|
|
362
|
+
f"[muted]Re-run with --trials {MIN_TRIALS_FOR_EVIDENCE} and read "
|
|
363
|
+
f"pass^k, not the mean.[/]"
|
|
364
|
+
)
|
|
365
|
+
|
|
366
|
+
out.print(
|
|
367
|
+
f"\n{s.get('passed', 0)}/{s.get('tasks', 0)} passed "
|
|
368
|
+
f"({s.get('pass_rate', 0):.1%}) "
|
|
369
|
+
f"{'mean score' if s.get('is_evidence', True) else 'score (1 run)'} "
|
|
370
|
+
f"{s.get('mean_task_score', 0):.3f} "
|
|
371
|
+
f"completion {s.get('mean_completion', 0):.3f} "
|
|
372
|
+
f"{s.get('errors', 0)} harness/agent error(s)"
|
|
373
|
+
)
|
|
374
|
+
out.print(
|
|
375
|
+
f"cost: {s.get('mean_model_calls', 0):.1f} model calls and "
|
|
376
|
+
f"{s.get('mean_wall_time_s', 0):.0f}s per task"
|
|
377
|
+
)
|
|
378
|
+
if "mean_pass_hat_k" in s:
|
|
379
|
+
k = s["trials"]
|
|
380
|
+
out.print(
|
|
381
|
+
f"reliability over {k} trials: pass^{k} {s['mean_pass_hat_k']:.3f} "
|
|
382
|
+
f"pass@{k} {s['mean_pass_at_k']:.3f} "
|
|
383
|
+
f"mean spread {s['score_spread']:.3f}"
|
|
384
|
+
)
|
|
385
|
+
kinds = (s.get("failures") or {}).get("failures_by_kind") or {}
|
|
386
|
+
if kinds:
|
|
387
|
+
total_failed = s.get("tasks", 0) - s.get("passed", 0)
|
|
388
|
+
out.print(
|
|
389
|
+
"failures by kind: "
|
|
390
|
+
+ " ".join(f"{kind} {count}" for kind, count in kinds.items())
|
|
391
|
+
+ f" (of {total_failed} failed)"
|
|
392
|
+
)
|
|
393
|
+
cost = s.get("tool_cost") or {}
|
|
394
|
+
if cost.get("calls"):
|
|
395
|
+
# Per SEAT is the half that answers the question: Otto routes across
|
|
396
|
+
# four vendors and several capability tiers, and model strength is
|
|
397
|
+
# what decides whether an agent manages a seventeen-tool menu at all.
|
|
398
|
+
busiest = " ".join(f"{tool} {n}" for tool, n in
|
|
399
|
+
list(cost["by_tool"].items())[:5])
|
|
400
|
+
out.print(
|
|
401
|
+
f"tools: {cost['calls']} calls, {cost['failures']} failed | "
|
|
402
|
+
f"busiest: {busiest}"
|
|
403
|
+
)
|
|
404
|
+
out.print(
|
|
405
|
+
"per seat: "
|
|
406
|
+
+ " ".join(f"{seat} {n}" for seat, n in cost["by_seat"].items())
|
|
407
|
+
)
|
|
408
|
+
|
|
409
|
+
grading = report["grading"]
|
|
410
|
+
out.print(
|
|
411
|
+
f"grading: {grading['otto_grading_path']} / claw {grading['claw_eval_revision'] or '?'} "
|
|
412
|
+
f"/ judge {grading['judge']} / split {split}"
|
|
413
|
+
)
|
|
414
|
+
out.print(f"report: {out_dir / 'otto_summary.json'}")
|
|
@@ -0,0 +1,106 @@
|
|
|
1
|
+
"""`otto eval-compaction` -- what each compaction policy actually loses.
|
|
2
|
+
|
|
3
|
+
The instrument Otto did not have. `otto eval-memory` scores whether raw
|
|
4
|
+
evidence stays RETRIEVABLE, which the chunk store nearly guarantees; this
|
|
5
|
+
scores whether the constraint is still in what the prompt reads. Those are
|
|
6
|
+
different questions and only the second one distinguishes compaction policies.
|
|
7
|
+
|
|
8
|
+
Offline and free by default: the summariser is a deterministic stand-in, so a
|
|
9
|
+
policy comparison measures the POLICY rather than whichever model happened to
|
|
10
|
+
summarise that day.
|
|
11
|
+
"""
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
import json
|
|
15
|
+
import tempfile
|
|
16
|
+
from pathlib import Path
|
|
17
|
+
from typing import Optional
|
|
18
|
+
|
|
19
|
+
import typer
|
|
20
|
+
from rich import box
|
|
21
|
+
from rich.table import Table
|
|
22
|
+
from typing_extensions import Annotated
|
|
23
|
+
|
|
24
|
+
from agent.cli.ui import err, out
|
|
25
|
+
from agent.eval.compaction_bench import POLICIES, run_matrix
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def eval_compaction_cmd(
|
|
29
|
+
turns: Annotated[
|
|
30
|
+
int, typer.Option(help="How long a conversation to replay. Longer means "
|
|
31
|
+
"more compaction rounds, which is where policies diverge."),
|
|
32
|
+
] = 120,
|
|
33
|
+
policy: Annotated[
|
|
34
|
+
Optional[str],
|
|
35
|
+
typer.Option(help="Run one policy instead of the whole matrix."),
|
|
36
|
+
] = None,
|
|
37
|
+
live: Annotated[
|
|
38
|
+
bool,
|
|
39
|
+
typer.Option("--live", help="Use the real Task.SUMMARIZE model instead of the "
|
|
40
|
+
"deterministic stand-in. Answers a question about "
|
|
41
|
+
"the MODEL, not about the policy, and costs calls."),
|
|
42
|
+
] = False,
|
|
43
|
+
out_dir: Annotated[
|
|
44
|
+
Optional[Path], typer.Option(help="Where to put the stores (default: a temp dir)."),
|
|
45
|
+
] = None,
|
|
46
|
+
as_json: Annotated[bool, typer.Option("--json", help="Print the report as JSON.")] = False,
|
|
47
|
+
) -> None:
|
|
48
|
+
chosen = POLICIES
|
|
49
|
+
if policy:
|
|
50
|
+
if policy not in POLICIES:
|
|
51
|
+
err.print(f"no policy {policy!r} -- have {', '.join(POLICIES)}")
|
|
52
|
+
raise typer.Exit(2)
|
|
53
|
+
chosen = {policy: POLICIES[policy]}
|
|
54
|
+
|
|
55
|
+
summarize = None
|
|
56
|
+
if live:
|
|
57
|
+
from agent.memory.wiring import summarize_for_memory
|
|
58
|
+
|
|
59
|
+
summarize = summarize_for_memory
|
|
60
|
+
err.print("[warn]--live[/] spends model calls, one per compaction round per policy")
|
|
61
|
+
|
|
62
|
+
root = Path(out_dir) if out_dir else Path(tempfile.mkdtemp(prefix="otto-compaction-"))
|
|
63
|
+
root.mkdir(parents=True, exist_ok=True)
|
|
64
|
+
err.print(f"{len(chosen)} policy/policies over {turns} turns -> {root}")
|
|
65
|
+
|
|
66
|
+
results = run_matrix(root, turns=turns, policies=chosen, summarize=summarize)
|
|
67
|
+
|
|
68
|
+
if as_json:
|
|
69
|
+
out.print_json(data={
|
|
70
|
+
"turns": turns, "live": live,
|
|
71
|
+
"policies": [
|
|
72
|
+
{
|
|
73
|
+
"name": r.name, "survives": r.survives, "recoverable": r.recoverable,
|
|
74
|
+
"total": r.total, "survival_rate": round(r.survival_rate, 3),
|
|
75
|
+
"recovery_rate": round(r.recovery_rate, 3),
|
|
76
|
+
"view_chars": r.view_chars, "compactions": r.compactions,
|
|
77
|
+
"missing": r.missing,
|
|
78
|
+
}
|
|
79
|
+
for r in results
|
|
80
|
+
],
|
|
81
|
+
})
|
|
82
|
+
return
|
|
83
|
+
|
|
84
|
+
table = Table(box=box.SIMPLE, pad_edge=False)
|
|
85
|
+
table.add_column("policy")
|
|
86
|
+
table.add_column("in the view", justify="right")
|
|
87
|
+
table.add_column("recoverable", justify="right")
|
|
88
|
+
table.add_column("view chars", justify="right")
|
|
89
|
+
table.add_column("rounds", justify="right")
|
|
90
|
+
for r in results:
|
|
91
|
+
# The number that matters is the first: what the model will actually
|
|
92
|
+
# read without being told to go looking.
|
|
93
|
+
style = "ok" if r.survival_rate >= 0.75 else "warn" if r.survival_rate else "bad"
|
|
94
|
+
table.add_row(
|
|
95
|
+
r.name,
|
|
96
|
+
f"[{style}]{r.survives}/{r.total}[/]",
|
|
97
|
+
f"{r.recoverable}/{r.total}",
|
|
98
|
+
f"{r.view_chars:,}",
|
|
99
|
+
str(r.compactions),
|
|
100
|
+
)
|
|
101
|
+
out.print(table)
|
|
102
|
+
out.print(
|
|
103
|
+
"'in the view' is what a prompt is built from; 'recoverable' is what "
|
|
104
|
+
"recall_memory could still find. The gap between them is text the "
|
|
105
|
+
"model will not see unless it thinks to go looking."
|
|
106
|
+
)
|