otto-cli-agent 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent/README.md +22 -0
- agent/__init__.py +0 -0
- agent/cli/README.md +77 -0
- agent/cli/__init__.py +0 -0
- agent/cli/art.py +371 -0
- agent/cli/chat.py +309 -0
- agent/cli/clipboard.py +106 -0
- agent/cli/context.py +56 -0
- agent/cli/doctor.py +60 -0
- agent/cli/errors.py +58 -0
- agent/cli/eval.py +130 -0
- agent/cli/eval_claw.py +414 -0
- agent/cli/eval_compaction.py +106 -0
- agent/cli/eval_hle.py +92 -0
- agent/cli/eval_memory.py +253 -0
- agent/cli/eval_swe.py +172 -0
- agent/cli/lessons.py +97 -0
- agent/cli/main.py +67 -0
- agent/cli/modals.py +570 -0
- agent/cli/models.py +64 -0
- agent/cli/output.py +54 -0
- agent/cli/route.py +78 -0
- agent/cli/sessions.py +125 -0
- agent/cli/setup_screen.py +562 -0
- agent/cli/shell.py +548 -0
- agent/cli/tui.py +1807 -0
- agent/cli/ui.py +14 -0
- agent/cli/usage_panel.py +159 -0
- agent/config/README.md +7 -0
- agent/config/__init__.py +0 -0
- agent/config/envfile.py +76 -0
- agent/eval/README.md +76 -0
- agent/eval/__init__.py +0 -0
- agent/eval/claw_bench.py +1031 -0
- agent/eval/compaction_bench.py +229 -0
- agent/eval/data/README.md +10 -0
- agent/eval/data/claw/README.md +108 -0
- agent/eval/data/claw/llm_judge-gemini.patch +57 -0
- agent/eval/data/claw/otto.yaml +35 -0
- agent/eval/failures.py +276 -0
- agent/eval/golden/README.md +33 -0
- agent/eval/golden/code_01.json +6 -0
- agent/eval/golden/code_02.json +6 -0
- agent/eval/golden/code_03.json +6 -0
- agent/eval/golden/code_04.json +6 -0
- agent/eval/golden/code_05.json +6 -0
- agent/eval/golden/code_06.json +6 -0
- agent/eval/golden/math_01.json +6 -0
- agent/eval/golden/math_02.json +6 -0
- agent/eval/golden/math_03.json +6 -0
- agent/eval/golden/math_04.json +6 -0
- agent/eval/golden/math_05.json +6 -0
- agent/eval/golden/math_06.json +6 -0
- agent/eval/golden/nphard_gcp_01.json +6 -0
- agent/eval/golden/nphard_ksp_01.json +6 -0
- agent/eval/golden/nphard_math_binpacking_01.json +6 -0
- agent/eval/golden/nphard_math_clique_01.json +6 -0
- agent/eval/golden/nphard_math_setcover_01.json +6 -0
- agent/eval/golden/nphard_math_subsetsum_01.json +6 -0
- agent/eval/golden/nphard_tsp_01.json +6 -0
- agent/eval/golden/nphard_tsp_02.json +6 -0
- agent/eval/hle_bench.py +273 -0
- agent/eval/langfuse_sync.py +172 -0
- agent/eval/memory_bench.py +538 -0
- agent/eval/runner.py +174 -0
- agent/eval/single_agent.py +120 -0
- agent/eval/swe_bench.py +604 -0
- agent/eval/terminal_bench.py +345 -0
- agent/memory/README.md +102 -0
- agent/memory/__init__.py +42 -0
- agent/memory/embeddings.py +302 -0
- agent/memory/hashing.py +15 -0
- agent/memory/lessons.py +483 -0
- agent/memory/queue.py +531 -0
- agent/memory/retrieval.py +493 -0
- agent/memory/session.py +60 -0
- agent/memory/sessions.py +436 -0
- agent/memory/store.py +429 -0
- agent/memory/tokens.py +60 -0
- agent/memory/wiring.py +146 -0
- agent/pipeline/README.md +135 -0
- agent/pipeline/__init__.py +0 -0
- agent/pipeline/browsing.py +609 -0
- agent/pipeline/budget.py +403 -0
- agent/pipeline/codemap.py +254 -0
- agent/pipeline/evidence.py +325 -0
- agent/pipeline/execution.py +67 -0
- agent/pipeline/modes.py +137 -0
- agent/pipeline/native.py +1137 -0
- agent/pipeline/nodes.py +3644 -0
- agent/pipeline/pricing.py +209 -0
- agent/pipeline/progress.py +139 -0
- agent/pipeline/rag.py +139 -0
- agent/pipeline/research.py +1325 -0
- agent/pipeline/run.py +528 -0
- agent/pipeline/screen.py +77 -0
- agent/pipeline/state.py +220 -0
- agent/pipeline/toolkit.py +328 -0
- agent/pipeline/tools.py +1990 -0
- agent/pipeline/tracing.py +147 -0
- agent/pipeline/usage.py +251 -0
- agent/pipeline/vision.py +84 -0
- agent/pipeline/walkthrough.py +735 -0
- agent/pipeline/workspace.py +229 -0
- agent/router/README.md +60 -0
- agent/router/__init__.py +0 -0
- agent/router/automap.py +114 -0
- agent/router/health.py +229 -0
- agent/router/llm_provider/README.md +38 -0
- agent/router/llm_provider/__init__.py +202 -0
- agent/router/llm_provider/anthropic_provider.py +128 -0
- agent/router/llm_provider/base.py +507 -0
- agent/router/llm_provider/custom.py +152 -0
- agent/router/llm_provider/gemini_provider.py +122 -0
- agent/router/llm_provider/inception_provider.py +687 -0
- agent/router/llm_provider/openai_provider.py +151 -0
- agent/router/llm_provider/retired.py +145 -0
- agent/router/llm_provider/temperature.py +371 -0
- agent/router/mapping.py +579 -0
- agent/router/outcomes.py +363 -0
- agent/router/overrides.py +389 -0
- agent/router/reload.py +28 -0
- agent/router/router.py +413 -0
- agent/router/setup.py +123 -0
- otto_cli_agent-0.1.0.dist-info/METADATA +115 -0
- otto_cli_agent-0.1.0.dist-info/RECORD +129 -0
- otto_cli_agent-0.1.0.dist-info/WHEEL +4 -0
- otto_cli_agent-0.1.0.dist-info/entry_points.txt +2 -0
- otto_cli_agent-0.1.0.dist-info/licenses/LICENSE +21 -0
agent/cli/eval_hle.py
ADDED
|
@@ -0,0 +1,92 @@
|
|
|
1
|
+
"""`otto eval-hle` -- Humanity's Last Exam against Otto (agent/eval/hle_bench.py).
|
|
2
|
+
|
|
3
|
+
Two modes on the same sample, because the interesting number is not HLE's
|
|
4
|
+
published figure for this model but what Otto's graph does to it: `raw` asks
|
|
5
|
+
the model once, `agent` runs the whole router/planner/solver/evaluator graph
|
|
6
|
+
over the identical question and reports how many model calls that cost.
|
|
7
|
+
|
|
8
|
+
The dataset is gated -- accept the terms at huggingface.co/datasets/cais/hle
|
|
9
|
+
and set HF_TOKEN. Start with a small --limit: `agent` mode spends a full graph
|
|
10
|
+
run per question.
|
|
11
|
+
"""
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
import json
|
|
15
|
+
from pathlib import Path
|
|
16
|
+
from typing import Optional
|
|
17
|
+
|
|
18
|
+
import typer
|
|
19
|
+
from typing_extensions import Annotated
|
|
20
|
+
|
|
21
|
+
from agent.cli.ui import err, out
|
|
22
|
+
from agent.eval.hle_bench import (
|
|
23
|
+
DEFAULT_CACHE,
|
|
24
|
+
DatasetGated,
|
|
25
|
+
download_hle,
|
|
26
|
+
load_hle,
|
|
27
|
+
run_hle,
|
|
28
|
+
sample_questions,
|
|
29
|
+
summarise,
|
|
30
|
+
)
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def eval_hle_cmd(
|
|
34
|
+
mode: Annotated[
|
|
35
|
+
str, typer.Option(help="'raw' (one model call per question) or 'agent' (the whole graph)."),
|
|
36
|
+
] = "raw",
|
|
37
|
+
limit: Annotated[
|
|
38
|
+
Optional[int], typer.Option(help="Score only this many questions (sampled deterministically)."),
|
|
39
|
+
] = 25,
|
|
40
|
+
seed: Annotated[int, typer.Option(help="Sampling seed, so two modes score the SAME questions.")] = 0,
|
|
41
|
+
data_path: Annotated[
|
|
42
|
+
Optional[Path], typer.Option(help=f"Cached HLE parquet (default: {DEFAULT_CACHE}).")
|
|
43
|
+
] = None,
|
|
44
|
+
hf_token: Annotated[
|
|
45
|
+
Optional[str], typer.Option(help="Hugging Face token; defaults to $HF_TOKEN.")
|
|
46
|
+
] = None,
|
|
47
|
+
show_items: Annotated[
|
|
48
|
+
bool, typer.Option("--show-items", help="Print every question, answer and verdict."),
|
|
49
|
+
] = False,
|
|
50
|
+
as_json: Annotated[bool, typer.Option("--json", help="Print the full report as JSON.")] = False,
|
|
51
|
+
) -> None:
|
|
52
|
+
if mode not in {"raw", "agent"}:
|
|
53
|
+
err.print(f"mode must be 'raw' or 'agent', not {mode!r}")
|
|
54
|
+
raise typer.Exit(2)
|
|
55
|
+
|
|
56
|
+
path = data_path or DEFAULT_CACHE
|
|
57
|
+
try:
|
|
58
|
+
download_hle(path, token=hf_token)
|
|
59
|
+
except DatasetGated as exc:
|
|
60
|
+
err.print(str(exc))
|
|
61
|
+
raise typer.Exit(1)
|
|
62
|
+
|
|
63
|
+
rows, skipped = sample_questions(load_hle(path), limit=limit, seed=seed)
|
|
64
|
+
out.print(f"scoring {len(rows)} text-only question(s) in {mode} mode "
|
|
65
|
+
f"({skipped} image question(s) skipped -- Otto's chat path is text-only)")
|
|
66
|
+
|
|
67
|
+
def _progress(r):
|
|
68
|
+
mark = "correct" if r.correct else "wrong "
|
|
69
|
+
out.print(f" {mark} {r.llm_calls:>3} call(s) {r.question[:72]}")
|
|
70
|
+
|
|
71
|
+
results = run_hle(rows, mode=mode, on_result=None if as_json else _progress)
|
|
72
|
+
report = summarise(results)
|
|
73
|
+
|
|
74
|
+
if as_json:
|
|
75
|
+
out.print(json.dumps({"mode": mode, "skipped_image_questions": skipped,
|
|
76
|
+
"summary": report,
|
|
77
|
+
"items": [r.__dict__ for r in results]}, indent=2))
|
|
78
|
+
return
|
|
79
|
+
|
|
80
|
+
out.print("")
|
|
81
|
+
out.print(f"accuracy: {report['accuracy']:.1%} of {report['n']}"
|
|
82
|
+
f" model calls: {report['llm_calls_total']} "
|
|
83
|
+
f"({report['llm_calls_per_question']:.1f} per question)")
|
|
84
|
+
for name, stats in report["by_category"].items():
|
|
85
|
+
out.print(f" {name:<34} {stats['accuracy']:>6.1%} (n={stats['n']})")
|
|
86
|
+
|
|
87
|
+
if show_items:
|
|
88
|
+
out.print("")
|
|
89
|
+
for r in results:
|
|
90
|
+
out.print(f"[{'correct' if r.correct else 'wrong'}] {r.question[:110]}")
|
|
91
|
+
out.print(f" expected: {r.correct_answer[:110]}")
|
|
92
|
+
out.print(f" extracted: {r.extracted[:110]}")
|
agent/cli/eval_memory.py
ADDED
|
@@ -0,0 +1,253 @@
|
|
|
1
|
+
"""`otto eval-memory`: run agent/eval/memory_bench.py's LoCoMo-based memory
|
|
2
|
+
benchmark and report coverage. See that module's own docstring for what's
|
|
3
|
+
actually measured (store/visible-verbatim/recall/answerable coverage per QA
|
|
4
|
+
category, plus a compression ratio) and why the two `--x-budget`/`--y-budget`
|
|
5
|
+
options exist: real LoCoMo conversations are small enough that Otto's real
|
|
6
|
+
production budget (agent/memory/queue.py's X_BUDGET/Y_BUDGET) never triggers
|
|
7
|
+
compaction at all, so a run at the defaults mostly reports "everything is
|
|
8
|
+
still verbatim, nothing needed compacting" -- correct, but it doesn't
|
|
9
|
+
exercise the compaction+recall code path. Passing smaller values here forces
|
|
10
|
+
that path to actually run, at the cost of no longer matching Otto's real
|
|
11
|
+
per-turn budget.
|
|
12
|
+
|
|
13
|
+
No Langfuse tracking here (unlike `otto eval`, agent/cli/eval.py) -- this
|
|
14
|
+
benchmark doesn't run the pipeline or the router at all in its default,
|
|
15
|
+
offline mode (`--live` opts into one real LLM call per compaction, via
|
|
16
|
+
agent.memory.wiring.summarize_for_memory, agent/eval/memory_bench.py's own
|
|
17
|
+
module docstring), so there's no per-item trace worth syncing there.
|
|
18
|
+
|
|
19
|
+
`--show-items failures|all` prints each scored QA item's evidence text next
|
|
20
|
+
to recall()'s raw output -- worth reaching for whenever a coverage number
|
|
21
|
+
looks suspicious, since `recalled` is an exact-substring match (module
|
|
22
|
+
docstring point 3), not a fuzzy one: reading the two side by side is how
|
|
23
|
+
you tell a real semantic-search find from a coincidental match. It also
|
|
24
|
+
surfaces a real gap in the OFFLINE (`--no-live`, default) summarizer:
|
|
25
|
+
_canned_summarize groups raw turns into fixed-size blocks with placeholder
|
|
26
|
+
bullet text, and recall() returns a matched bullet's ENTIRE underlying raw
|
|
27
|
+
text -- so offline, "recalled" mostly tests "did the right block rank in
|
|
28
|
+
top-k," not "did it find the right sentence." `--live`'s real, topic-
|
|
29
|
+
scoped bullet summaries are the stricter version of this same test.
|
|
30
|
+
"""
|
|
31
|
+
from __future__ import annotations
|
|
32
|
+
|
|
33
|
+
import json as json_module
|
|
34
|
+
from pathlib import Path
|
|
35
|
+
from typing import Annotated, Optional
|
|
36
|
+
|
|
37
|
+
import typer
|
|
38
|
+
from rich import box
|
|
39
|
+
from rich.table import Table
|
|
40
|
+
|
|
41
|
+
from agent.cli.ui import err, out
|
|
42
|
+
from agent.memory.retrieval import (
|
|
43
|
+
DEFAULT_MAX_CHUNKS,
|
|
44
|
+
DEFAULT_NEIGHBOUR_WINDOW,
|
|
45
|
+
DEFAULT_TOKEN_BUDGET,
|
|
46
|
+
)
|
|
47
|
+
from agent.eval.memory_bench import (
|
|
48
|
+
BenchmarkReport,
|
|
49
|
+
DEFAULT_CACHE,
|
|
50
|
+
download_locomo,
|
|
51
|
+
load_locomo,
|
|
52
|
+
recalled_text_has_unparsed_bullet,
|
|
53
|
+
run_benchmark,
|
|
54
|
+
)
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def eval_memory_cmd(
|
|
58
|
+
live: Annotated[
|
|
59
|
+
bool,
|
|
60
|
+
typer.Option(
|
|
61
|
+
"--live/--no-live",
|
|
62
|
+
help="Use the real Task.SUMMARIZE model + local embeddings (needs INCEPTION_API_KEY) "
|
|
63
|
+
"instead of the deterministic offline canned summarizer.",
|
|
64
|
+
),
|
|
65
|
+
] = False,
|
|
66
|
+
samples: Annotated[
|
|
67
|
+
Optional[int], typer.Option(help="Only run the first N of LoCoMo's 10 conversations.")
|
|
68
|
+
] = None,
|
|
69
|
+
max_turns: Annotated[
|
|
70
|
+
Optional[int], typer.Option(help="Truncate each conversation to its first N turns (fast smoke run).")
|
|
71
|
+
] = None,
|
|
72
|
+
x_budget: Annotated[
|
|
73
|
+
Optional[int],
|
|
74
|
+
typer.Option(help="TieredQueue's X budget (tokens). Defaults to the "
|
|
75
|
+
"PRODUCTION budget, which LoCoMo's shorter "
|
|
76
|
+
"conversations do not overflow -- so the default "
|
|
77
|
+
"compacts nothing and the run is refused. Try 600."),
|
|
78
|
+
] = None,
|
|
79
|
+
y_budget: Annotated[
|
|
80
|
+
Optional[int],
|
|
81
|
+
typer.Option(help="TieredQueue's Y budget (tokens). Same as --x-budget: "
|
|
82
|
+
"the production default never binds here. Try 1500."),
|
|
83
|
+
] = None,
|
|
84
|
+
top_k: Annotated[int, typer.Option(help="How many bullets recall() considers per question (stage 1).")] = 5,
|
|
85
|
+
max_chunks: Annotated[
|
|
86
|
+
int,
|
|
87
|
+
typer.Option(help="How many raw chunks recall() actually returns (stage 2's cap)."),
|
|
88
|
+
] = DEFAULT_MAX_CHUNKS,
|
|
89
|
+
neighbours: Annotated[
|
|
90
|
+
int,
|
|
91
|
+
typer.Option(help="How many chunks either side of each returned chunk to include."),
|
|
92
|
+
] = DEFAULT_NEIGHBOUR_WINDOW,
|
|
93
|
+
token_budget: Annotated[
|
|
94
|
+
int,
|
|
95
|
+
typer.Option(help="Hard ceiling (tokens) on how much text one recall() returns."),
|
|
96
|
+
] = DEFAULT_TOKEN_BUDGET,
|
|
97
|
+
data_path: Annotated[
|
|
98
|
+
Optional[Path], typer.Option(help="Path to a cached locomo10.json (default: agent/eval/data/locomo10.json).")
|
|
99
|
+
] = None,
|
|
100
|
+
force_download: Annotated[
|
|
101
|
+
bool, typer.Option("--force-download", help="Re-download locomo10.json even if a cached copy exists.")
|
|
102
|
+
] = False,
|
|
103
|
+
json: Annotated[
|
|
104
|
+
bool, typer.Option("--json", help="Print the full report as JSON instead of a table.")
|
|
105
|
+
] = False,
|
|
106
|
+
show_items: Annotated[
|
|
107
|
+
str,
|
|
108
|
+
typer.Option(
|
|
109
|
+
help="Print per-question detail after the table: 'none' (default), 'failures' "
|
|
110
|
+
"(answerable=False items only -- the interesting/debug case), or 'all'. Each "
|
|
111
|
+
"item shows its evidence text and recall()'s raw output side by side, since "
|
|
112
|
+
"`recalled` is an exact-substring match, not a fuzzy one -- useful for checking "
|
|
113
|
+
"whether a pass/miss is real or an artifact of the conversation repeating similar "
|
|
114
|
+
"phrasing. Ignored when --json is set (the JSON report already includes both).",
|
|
115
|
+
),
|
|
116
|
+
] = "none",
|
|
117
|
+
) -> None:
|
|
118
|
+
"""Score agent/memory/'s tiered queue against the LoCoMo long-conversation-memory dataset."""
|
|
119
|
+
if show_items not in ("none", "failures", "all"):
|
|
120
|
+
err.print("[bad]--show-items must be 'none', 'failures', or 'all'[/]")
|
|
121
|
+
raise typer.Exit(2)
|
|
122
|
+
|
|
123
|
+
path = data_path or DEFAULT_CACHE
|
|
124
|
+
with err.status(f"fetching LoCoMo dataset ({path})…"):
|
|
125
|
+
try:
|
|
126
|
+
download_locomo(path, force=force_download)
|
|
127
|
+
except Exception as exc:
|
|
128
|
+
err.print(f"[bad]could not download the LoCoMo dataset: {exc}[/]")
|
|
129
|
+
err.print("[muted]pass --data-path to point at an already-downloaded locomo10.json[/]")
|
|
130
|
+
raise typer.Exit(1) from exc
|
|
131
|
+
all_samples = load_locomo(path)
|
|
132
|
+
|
|
133
|
+
if samples is not None:
|
|
134
|
+
all_samples = all_samples[:samples]
|
|
135
|
+
if not all_samples:
|
|
136
|
+
err.print("[bad]no LoCoMo samples to run[/]")
|
|
137
|
+
raise typer.Exit(1)
|
|
138
|
+
|
|
139
|
+
with err.status(f"running {len(all_samples)} conversation(s), live={live}…"):
|
|
140
|
+
report = run_benchmark(
|
|
141
|
+
all_samples, live=live, top_k=top_k, max_chunks=max_chunks,
|
|
142
|
+
neighbour_window=neighbours, token_budget=token_budget, max_turns=max_turns,
|
|
143
|
+
x_budget=x_budget, y_budget=y_budget,
|
|
144
|
+
)
|
|
145
|
+
|
|
146
|
+
if json:
|
|
147
|
+
out.print(json_module.dumps(report.to_dict(), indent=2))
|
|
148
|
+
return
|
|
149
|
+
|
|
150
|
+
summary = report.summary()
|
|
151
|
+
|
|
152
|
+
# REFUSE, rather than print a perfect-looking empty result. A run at the
|
|
153
|
+
# production budgets never compacts LoCoMo's shorter conversations, so
|
|
154
|
+
# every citation stays in the verbatim window, retrieval is never asked a
|
|
155
|
+
# question, and the table comes out `recalled 0%` beside `answerable
|
|
156
|
+
# 100%` -- which reads as a pass. The numbers are not wrong so much as
|
|
157
|
+
# measured off nothing, and printing them at all is what let this happen
|
|
158
|
+
# unnoticed. See memory_bench.NO_COMPACTION_RATIO.
|
|
159
|
+
if summary["conversation_count"] and not summary["recall_exercised"]:
|
|
160
|
+
ratio = summary["overall_compression_ratio"]
|
|
161
|
+
err.print(
|
|
162
|
+
"[bad]nothing was compacted at these budgets, so recall was never "
|
|
163
|
+
"exercised -- this run measured nothing and no coverage table is "
|
|
164
|
+
"printed for it.[/]\n"
|
|
165
|
+
f"[muted]final_view/raw token ratio "
|
|
166
|
+
f"{'-' if ratio is None else f'{ratio:.3f}'} over "
|
|
167
|
+
f"{summary['conversation_count']} conversation(s), at "
|
|
168
|
+
f"--x-budget {x_budget} --y-budget {y_budget}.[/]\n"
|
|
169
|
+
"[warn]re-run with smaller --x-budget/--y-budget -- e.g. "
|
|
170
|
+
"--x-budget 600 --y-budget 1500 -- so the conversations overflow "
|
|
171
|
+
"and the retrieval path actually runs.[/]"
|
|
172
|
+
)
|
|
173
|
+
raise typer.Exit(1)
|
|
174
|
+
|
|
175
|
+
t = Table(box=box.SIMPLE, header_style="muted")
|
|
176
|
+
t.add_column("category", style="spec")
|
|
177
|
+
t.add_column("n", justify="right")
|
|
178
|
+
t.add_column("store", justify="right")
|
|
179
|
+
t.add_column("verbatim", justify="right")
|
|
180
|
+
t.add_column("recalled", justify="right")
|
|
181
|
+
t.add_column("answerable", justify="right")
|
|
182
|
+
|
|
183
|
+
def _pct(value: Optional[float]) -> str:
|
|
184
|
+
return "-" if value is None else f"{value * 100:.0f}%"
|
|
185
|
+
|
|
186
|
+
for name, rates in summary["by_category"].items():
|
|
187
|
+
t.add_row(
|
|
188
|
+
name, str(rates["n"]), _pct(rates["store_coverage"]),
|
|
189
|
+
_pct(rates["visible_verbatim_coverage"]), _pct(rates["recall_coverage"]),
|
|
190
|
+
_pct(rates["answerable_coverage"]),
|
|
191
|
+
)
|
|
192
|
+
overall = summary["overall"]
|
|
193
|
+
t.add_row(
|
|
194
|
+
"[chosen]overall[/]", str(overall["n"]), _pct(overall["store_coverage"]),
|
|
195
|
+
_pct(overall["visible_verbatim_coverage"]), _pct(overall["recall_coverage"]),
|
|
196
|
+
_pct(overall["answerable_coverage"]),
|
|
197
|
+
)
|
|
198
|
+
out.print(t)
|
|
199
|
+
|
|
200
|
+
ratio = summary["overall_compression_ratio"]
|
|
201
|
+
ratio_text = "-" if ratio is None else f"{ratio:.3f}"
|
|
202
|
+
out.print(
|
|
203
|
+
f"[muted]{summary['conversation_count']} conversation(s) -- "
|
|
204
|
+
f"final_view/raw token ratio: {ratio_text}[/]"
|
|
205
|
+
)
|
|
206
|
+
|
|
207
|
+
unparsed = summary["unparsed_bullet_count"]
|
|
208
|
+
if unparsed:
|
|
209
|
+
err.print(
|
|
210
|
+
f"[warn]{unparsed}/{summary['bullet_count']} stored bullets are unparsed-summary "
|
|
211
|
+
f"placeholders (the summarizer's reply didn't parse for that compaction) -- "
|
|
212
|
+
f"recall_coverage is partly measuring ranking against content-free text, not a "
|
|
213
|
+
f"real summary. --show-items all will show which questions land on one.[/]"
|
|
214
|
+
)
|
|
215
|
+
|
|
216
|
+
if show_items != "none":
|
|
217
|
+
_print_items(report, show_items)
|
|
218
|
+
|
|
219
|
+
if overall["n"] and overall["store_coverage"] is not None and overall["store_coverage"] < 1.0:
|
|
220
|
+
err.print("[bad]store_coverage < 100% -- the engine lost cited evidence; this is a real bug[/]")
|
|
221
|
+
raise typer.Exit(1)
|
|
222
|
+
|
|
223
|
+
|
|
224
|
+
def _snippet(text: str, limit: int = 280) -> str:
|
|
225
|
+
text = " ".join(text.split())
|
|
226
|
+
return text if len(text) <= limit else text[:limit] + "…"
|
|
227
|
+
|
|
228
|
+
|
|
229
|
+
def _print_items(report: BenchmarkReport, which: str) -> None:
|
|
230
|
+
"""Per-question detail: for every scored QA item matching `which`, the
|
|
231
|
+
exact evidence text next to recall()'s raw output -- the pair
|
|
232
|
+
`recalled` was computed from (an exact substring check, module
|
|
233
|
+
docstring point 3), so reading them side by side is how you tell a
|
|
234
|
+
real find from a coincidental match in a conversation that repeats
|
|
235
|
+
similar phrasing often.
|
|
236
|
+
"""
|
|
237
|
+
out.print("")
|
|
238
|
+
shown = 0
|
|
239
|
+
for conv in report.conversations:
|
|
240
|
+
for r in conv.scored_results():
|
|
241
|
+
if which == "failures" and r.answerable:
|
|
242
|
+
continue
|
|
243
|
+
shown += 1
|
|
244
|
+
verdict = "[ok]answerable[/]" if r.answerable else "[bad]MISS[/]"
|
|
245
|
+
via = "verbatim" if r.visible_verbatim else ("recalled" if r.recalled else "neither")
|
|
246
|
+
flag = " [warn]unparsed-bullet[/]" if recalled_text_has_unparsed_bullet(r.recalled_text) else ""
|
|
247
|
+
out.print(f"[spec]{conv.sample_id}[/] [muted]({via})[/] {verdict}{flag}")
|
|
248
|
+
out.print(f" Q: {r.question}")
|
|
249
|
+
out.print(f" evidence: {_snippet(r.evidence_text)}")
|
|
250
|
+
out.print(f" recalled: {_snippet(r.recalled_text)}")
|
|
251
|
+
out.print("")
|
|
252
|
+
if shown == 0:
|
|
253
|
+
out.print("[muted](nothing matched --show-items filter)[/]")
|
agent/cli/eval_swe.py
ADDED
|
@@ -0,0 +1,172 @@
|
|
|
1
|
+
"""`otto eval-swe` -- SWE-bench Verified, graded by each repository's own tests.
|
|
2
|
+
|
|
3
|
+
The only benchmark here whose verdict nobody involved can argue with: 500 real
|
|
4
|
+
GitHub issues, and the maintainers' own test suite decides. See
|
|
5
|
+
agent/eval/swe_bench.py for what Otto is and is not told.
|
|
6
|
+
|
|
7
|
+
Needs Docker. The official per-instance images are about 3.5GB each and are
|
|
8
|
+
pulled on first use, so start with --limit 1 and expect the first run of a
|
|
9
|
+
repository to spend several minutes downloading before the agent starts.
|
|
10
|
+
"""
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
import json
|
|
14
|
+
import time
|
|
15
|
+
from pathlib import Path
|
|
16
|
+
from typing import Optional
|
|
17
|
+
|
|
18
|
+
import typer
|
|
19
|
+
from typing_extensions import Annotated
|
|
20
|
+
|
|
21
|
+
from agent.cli.ui import err, out
|
|
22
|
+
from agent.eval.swe_bench import (
|
|
23
|
+
SweBenchUnavailable, load_dataset, resolve_image, run_instance, select,
|
|
24
|
+
)
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def eval_swe_cmd(
|
|
28
|
+
repo: Annotated[
|
|
29
|
+
Optional[str], typer.Option(help="Only instances from repositories matching this."),
|
|
30
|
+
] = None,
|
|
31
|
+
instance: Annotated[
|
|
32
|
+
Optional[str], typer.Option(help="Substring of an instance id, e.g. 'astropy-12907'."),
|
|
33
|
+
] = None,
|
|
34
|
+
difficulty: Annotated[
|
|
35
|
+
Optional[str],
|
|
36
|
+
typer.Option(help="Filter on the dataset's own estimate: '<15', '15 min', '1-4', '>4'."),
|
|
37
|
+
] = None,
|
|
38
|
+
limit: Annotated[int, typer.Option(help="Run at most this many instances.")] = 1,
|
|
39
|
+
max_seconds: Annotated[
|
|
40
|
+
float, typer.Option(help="Wall-clock budget per instance, before grading."),
|
|
41
|
+
] = 1800.0,
|
|
42
|
+
max_model_calls: Annotated[
|
|
43
|
+
Optional[int], typer.Option(help="Override the default spend ceiling per instance."),
|
|
44
|
+
] = None,
|
|
45
|
+
trace_dir: Annotated[
|
|
46
|
+
Optional[Path], typer.Option(help="Where to write each instance's diff."),
|
|
47
|
+
] = None,
|
|
48
|
+
dataset: Annotated[
|
|
49
|
+
Optional[Path], typer.Option(help="Cached parquet (default: agent/eval/data/)."),
|
|
50
|
+
] = None,
|
|
51
|
+
as_json: Annotated[bool, typer.Option("--json", help="Print the report as JSON.")] = False,
|
|
52
|
+
) -> None:
|
|
53
|
+
try:
|
|
54
|
+
instances = select(
|
|
55
|
+
load_dataset(dataset), repo=repo, pattern=instance,
|
|
56
|
+
difficulty=difficulty, limit=limit,
|
|
57
|
+
)
|
|
58
|
+
except SweBenchUnavailable as exc:
|
|
59
|
+
err.print(str(exc))
|
|
60
|
+
raise typer.Exit(2) from None
|
|
61
|
+
|
|
62
|
+
if not instances:
|
|
63
|
+
err.print("no instances matched")
|
|
64
|
+
raise typer.Exit(1)
|
|
65
|
+
|
|
66
|
+
out_dir = Path(trace_dir) if trace_dir else Path("/tmp") / f"otto-swe-{time.strftime('%y%m%d-%H%M')}"
|
|
67
|
+
err.print(f"{len(instances)} instance(s) -> {out_dir}")
|
|
68
|
+
|
|
69
|
+
outcomes = []
|
|
70
|
+
for i, item in enumerate(instances, 1):
|
|
71
|
+
err.print(f"[{i}/{len(instances)}] {item.instance_id} ({item.difficulty or 'unrated'})")
|
|
72
|
+
image, emulated = resolve_image(item.instance_id)
|
|
73
|
+
err.print(f" image {image}"
|
|
74
|
+
+ (" [warn](x86_64 under emulation -- no arm64 build)[/]"
|
|
75
|
+
if emulated else ""))
|
|
76
|
+
try:
|
|
77
|
+
outcome = run_instance(
|
|
78
|
+
item, max_seconds=max_seconds,
|
|
79
|
+
max_model_calls=max_model_calls, trace_dir=out_dir,
|
|
80
|
+
)
|
|
81
|
+
except SweBenchUnavailable as exc:
|
|
82
|
+
err.print(f" [bad]harness error[/] {exc}")
|
|
83
|
+
continue
|
|
84
|
+
outcomes.append(outcome)
|
|
85
|
+
flag = "[ok]resolved[/]" if outcome.resolved else "[warn]not resolved[/]"
|
|
86
|
+
detail = f" ({outcome.error})" if outcome.error else ""
|
|
87
|
+
out.print(
|
|
88
|
+
f" {flag} {outcome.grade.summary()} "
|
|
89
|
+
f"{outcome.diff_lines} changed line(s) calls={outcome.model_calls} "
|
|
90
|
+
f"{outcome.wall_time_s:.0f}s{detail}"
|
|
91
|
+
)
|
|
92
|
+
if outcome.grade.error:
|
|
93
|
+
err.print(f" [bad]{outcome.grade.error}[/]")
|
|
94
|
+
for failure in outcome.grade.failures[:3]:
|
|
95
|
+
err.print(f" still failing: {failure[:100]}")
|
|
96
|
+
|
|
97
|
+
if not outcomes:
|
|
98
|
+
raise typer.Exit(1)
|
|
99
|
+
|
|
100
|
+
resolved = sum(1 for o in outcomes if o.resolved)
|
|
101
|
+
# How much of this number came from emulated instances, and how it splits.
|
|
102
|
+
#
|
|
103
|
+
# Upstream publishes arm64 for only part of the set, and django -- 231 of
|
|
104
|
+
# the 500 instances -- had no arm64 build in every instance tried. So on
|
|
105
|
+
# Apple silicon roughly half the benchmark runs under emulation, several
|
|
106
|
+
# times slower, and a resolve rate that mixes the two is not a sample of
|
|
107
|
+
# SWE-bench. swe_bench.EMULATION_TIME_FACTOR stops the budget being the
|
|
108
|
+
# thing that decides those instances; it does NOT make them comparable,
|
|
109
|
+
# which is why the split is reported rather than smoothed away.
|
|
110
|
+
emulated = [o for o in outcomes if o.emulated]
|
|
111
|
+
native = [o for o in outcomes if not o.emulated]
|
|
112
|
+
|
|
113
|
+
def _rate(group):
|
|
114
|
+
if not group:
|
|
115
|
+
return None
|
|
116
|
+
return round(sum(1 for o in group if o.resolved) / len(group), 4)
|
|
117
|
+
|
|
118
|
+
report = {
|
|
119
|
+
"instances": len(outcomes),
|
|
120
|
+
"resolved": resolved,
|
|
121
|
+
"resolve_rate": round(resolved / len(outcomes), 4),
|
|
122
|
+
"emulated_instances": len(emulated),
|
|
123
|
+
"native_instances": len(native),
|
|
124
|
+
"emulated_resolve_rate": _rate(emulated),
|
|
125
|
+
"native_resolve_rate": _rate(native),
|
|
126
|
+
#: The repositories whose instances ran emulated, which is the thing
|
|
127
|
+
#: a reader needs to judge whether the sample is representative --
|
|
128
|
+
#: "half of it was django, emulated" is a different number from the
|
|
129
|
+
#: same rate measured natively.
|
|
130
|
+
"emulated_repos": sorted({o.instance_id.split("__")[0] for o in emulated}),
|
|
131
|
+
"mean_model_calls": round(sum(o.model_calls for o in outcomes) / len(outcomes), 1),
|
|
132
|
+
"mean_wall_time_s": round(sum(o.wall_time_s for o in outcomes) / len(outcomes), 1),
|
|
133
|
+
"trace_dir": str(out_dir),
|
|
134
|
+
"results": [
|
|
135
|
+
{
|
|
136
|
+
"instance_id": o.instance_id, "resolved": o.resolved,
|
|
137
|
+
"fail_to_pass": [o.grade.fail_to_pass_passed, o.grade.fail_to_pass_total],
|
|
138
|
+
"pass_to_pass": [o.grade.pass_to_pass_passed, o.grade.pass_to_pass_total],
|
|
139
|
+
"diff_lines": o.diff_lines, "model_calls": o.model_calls,
|
|
140
|
+
"emulated": o.emulated,
|
|
141
|
+
"wall_time_s": round(o.wall_time_s, 1),
|
|
142
|
+
"error": o.error or o.grade.error,
|
|
143
|
+
}
|
|
144
|
+
for o in outcomes
|
|
145
|
+
],
|
|
146
|
+
}
|
|
147
|
+
out_dir.mkdir(parents=True, exist_ok=True)
|
|
148
|
+
(out_dir / "report.json").write_text(json.dumps(report, indent=2))
|
|
149
|
+
|
|
150
|
+
if as_json:
|
|
151
|
+
out.print_json(data=report)
|
|
152
|
+
return
|
|
153
|
+
out.print(
|
|
154
|
+
f"\n{resolved}/{len(outcomes)} resolved ({report['resolve_rate']:.1%}) "
|
|
155
|
+
f"{report['mean_model_calls']:.0f} model calls and "
|
|
156
|
+
f"{report['mean_wall_time_s']:.0f}s per instance"
|
|
157
|
+
)
|
|
158
|
+
if emulated:
|
|
159
|
+
# A SWE-bench number from a machine that emulated part of its sample
|
|
160
|
+
# has to say so, in the same breath as the number.
|
|
161
|
+
native_text = (
|
|
162
|
+
f"{report['native_resolve_rate']:.1%}" if native else "no native instances"
|
|
163
|
+
)
|
|
164
|
+
err.print(
|
|
165
|
+
f"[warn]{len(emulated)}/{len(outcomes)} instances ran under x86_64 "
|
|
166
|
+
f"emulation (no arm64 build): "
|
|
167
|
+
f"{', '.join(report['emulated_repos'])}.[/]\n"
|
|
168
|
+
f"[muted]emulated {report['emulated_resolve_rate']:.1%} vs native "
|
|
169
|
+
f"{native_text} -- quote the split, not the combined rate, and say "
|
|
170
|
+
f"this machine emulated part of the sample.[/]"
|
|
171
|
+
)
|
|
172
|
+
out.print(f"report: {out_dir / 'report.json'}")
|
agent/cli/lessons.py
ADDED
|
@@ -0,0 +1,97 @@
|
|
|
1
|
+
"""`otto lessons` -- read, move, and if need be empty, what Otto has learned.
|
|
2
|
+
|
|
3
|
+
A self-improving system whose learned state cannot be read is not a system
|
|
4
|
+
anyone should trust, and this is the cheapest possible remedy: the bank is a
|
|
5
|
+
few short lines of text, so print them.
|
|
6
|
+
|
|
7
|
+
There is a second reason it exists. Evolution that is not working is a real
|
|
8
|
+
and documented outcome -- across five methods and three frontier models the
|
|
9
|
+
measured gains were around +1%, and negative in every regime for the
|
|
10
|
+
strongest model. Being able to look at the bank and say "these lessons are
|
|
11
|
+
rubbish" is how that gets noticed, and `--clear` is how it gets undone.
|
|
12
|
+
|
|
13
|
+
`--export` / `--import` (2026-09-12) move a bank between machines as JSON,
|
|
14
|
+
through the same duplicate adjudication a run's own lessons face. The TUI's
|
|
15
|
+
"Export lessons…" / "Import lessons…" palette entries call the same two
|
|
16
|
+
functions (agent/memory/lessons.py).
|
|
17
|
+
"""
|
|
18
|
+
from __future__ import annotations
|
|
19
|
+
|
|
20
|
+
from pathlib import Path
|
|
21
|
+
from typing import Optional
|
|
22
|
+
|
|
23
|
+
import typer
|
|
24
|
+
from rich import box
|
|
25
|
+
from rich.table import Table
|
|
26
|
+
from rich.text import Text
|
|
27
|
+
from typing_extensions import Annotated
|
|
28
|
+
|
|
29
|
+
from agent.cli.ui import err, out
|
|
30
|
+
from agent.memory.lessons import (
|
|
31
|
+
Lesson, all_lessons, bank_path, bind_bank, clear_bank, export_lessons, import_lessons,
|
|
32
|
+
)
|
|
33
|
+
from agent.memory.store import MemoryStore
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def lessons_table(lessons: list[Lesson], path: Path | None = None) -> Table:
|
|
37
|
+
"""The bank as a table. Primitive Rich styles only, so it renders the same
|
|
38
|
+
in the REPL (whose THEME maps ok/warn to exactly these) and inside a
|
|
39
|
+
Textual Static, which never sees that THEME."""
|
|
40
|
+
title = f"{len(lessons)} lesson(s)" + (f" in {path}" if path else "")
|
|
41
|
+
t = Table(box=box.SIMPLE, header_style="dim", title=title, title_justify="left")
|
|
42
|
+
t.add_column("outcome")
|
|
43
|
+
t.add_column("when")
|
|
44
|
+
t.add_column("do", overflow="fold")
|
|
45
|
+
for lesson in lessons:
|
|
46
|
+
mark = Text("worked", style="bold green") if lesson.outcome == "worked" else Text("failed", style="yellow")
|
|
47
|
+
t.add_row(mark, lesson.cue, lesson.action)
|
|
48
|
+
return t
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def lessons_cmd(
|
|
52
|
+
bank: Annotated[
|
|
53
|
+
Optional[Path],
|
|
54
|
+
typer.Option(help="Which bank to read (default: ~/.otto/memory/lessons.db)."),
|
|
55
|
+
] = None,
|
|
56
|
+
clear: Annotated[
|
|
57
|
+
bool,
|
|
58
|
+
typer.Option("--clear", help="Delete every lesson. Not undoable."),
|
|
59
|
+
] = False,
|
|
60
|
+
export: Annotated[
|
|
61
|
+
Optional[Path],
|
|
62
|
+
typer.Option("--export", help="Write the bank to this JSON file."),
|
|
63
|
+
] = None,
|
|
64
|
+
import_from: Annotated[
|
|
65
|
+
Optional[Path],
|
|
66
|
+
typer.Option("--import", help="Merge lessons from this JSON file into the bank."),
|
|
67
|
+
] = None,
|
|
68
|
+
replace: Annotated[
|
|
69
|
+
bool,
|
|
70
|
+
typer.Option("--replace", help="With --import: empty the bank first."),
|
|
71
|
+
] = False,
|
|
72
|
+
) -> None:
|
|
73
|
+
path = bank or bank_path()
|
|
74
|
+
if import_from is None and not path.exists():
|
|
75
|
+
out.print(f"no lesson bank at {path} -- nothing learned yet")
|
|
76
|
+
return
|
|
77
|
+
|
|
78
|
+
store = MemoryStore(path)
|
|
79
|
+
with bind_bank(store):
|
|
80
|
+
if clear:
|
|
81
|
+
removed = clear_bank()
|
|
82
|
+
err.print(f"cleared {removed} lesson(s) from {path}")
|
|
83
|
+
return
|
|
84
|
+
if export is not None:
|
|
85
|
+
n = export_lessons(export)
|
|
86
|
+
err.print(f"exported {n} lesson(s) to {export}")
|
|
87
|
+
return
|
|
88
|
+
if import_from is not None:
|
|
89
|
+
report = import_lessons(import_from, replace=replace)
|
|
90
|
+
err.print(f"imported from {import_from}: {report.summary()}")
|
|
91
|
+
return
|
|
92
|
+
|
|
93
|
+
learned = all_lessons()
|
|
94
|
+
if not learned:
|
|
95
|
+
out.print(f"{path} is empty")
|
|
96
|
+
return
|
|
97
|
+
out.print(lessons_table(learned, path))
|
agent/cli/main.py
ADDED
|
@@ -0,0 +1,67 @@
|
|
|
1
|
+
from dotenv import load_dotenv
|
|
2
|
+
from typing import Annotated
|
|
3
|
+
import typer
|
|
4
|
+
|
|
5
|
+
# Loaded here, at import time, before anything below it is imported -- not
|
|
6
|
+
# inside bootstrap(). agent.cli.chat now pulls in the whole pipeline
|
|
7
|
+
# (agent.pipeline.nodes), which constructs a module-level Router() the
|
|
8
|
+
# moment it's imported, which reads *_API_KEY from the environment
|
|
9
|
+
# immediately. bootstrap() only runs once Typer has already finished
|
|
10
|
+
# importing every command module, which is too late for that first
|
|
11
|
+
# Router() call -- it would always see an environment with no keys in it,
|
|
12
|
+
# whatever's actually in .env.
|
|
13
|
+
from agent.config.envfile import ENV_PATH # the one file the setup screen writes
|
|
14
|
+
|
|
15
|
+
load_dotenv(ENV_PATH)
|
|
16
|
+
|
|
17
|
+
# Also before anything below: ~/.otto/routes.json may name custom endpoints
|
|
18
|
+
# and per-task pins, and the module-level Router() in agent.pipeline.nodes
|
|
19
|
+
# snapshots which providers exist the moment it is imported.
|
|
20
|
+
from agent.router.overrides import apply_at_startup # noqa: E402
|
|
21
|
+
|
|
22
|
+
apply_at_startup()
|
|
23
|
+
|
|
24
|
+
from agent.cli.context import AppContext
|
|
25
|
+
from agent.cli.errors import friendly
|
|
26
|
+
from agent.cli import doctor as doctor_cmd
|
|
27
|
+
from agent.cli import eval as eval_cmd
|
|
28
|
+
from agent.cli import eval_claw as eval_claw_cmd
|
|
29
|
+
from agent.cli import eval_compaction as eval_compaction_cmd
|
|
30
|
+
from agent.cli import eval_swe as eval_swe_cmd
|
|
31
|
+
from agent.cli import eval_hle as eval_hle_cmd
|
|
32
|
+
from agent.cli import eval_memory as eval_memory_cmd
|
|
33
|
+
from agent.cli import lessons as lessons_cmd
|
|
34
|
+
from agent.cli import models as model_cmd
|
|
35
|
+
from agent.cli import route as route_cmd
|
|
36
|
+
from agent.cli import sessions as sessions_cmd
|
|
37
|
+
from agent.cli import chat as chat_cmd
|
|
38
|
+
from agent.cli import tui as tui_cmd
|
|
39
|
+
|
|
40
|
+
app = typer.Typer()
|
|
41
|
+
|
|
42
|
+
@app.callback()
|
|
43
|
+
def bootstrap(ctx: typer.Context, strict: Annotated[bool, typer.Option()] = False) -> None:
|
|
44
|
+
ctx.obj = AppContext(strict=strict)
|
|
45
|
+
|
|
46
|
+
# `friendly` is applied here, once, rather than as a decorator on each command
|
|
47
|
+
# module. One registration site means a new command cannot forget it, and the
|
|
48
|
+
# command modules stay free of CLI exit-code concerns.
|
|
49
|
+
for _name, _fn in (
|
|
50
|
+
("doctor", doctor_cmd.doctor),
|
|
51
|
+
("models", model_cmd.models),
|
|
52
|
+
("route", route_cmd.route),
|
|
53
|
+
("chat", chat_cmd.chat),
|
|
54
|
+
("tui", tui_cmd.tui),
|
|
55
|
+
("eval", eval_cmd.eval_cmd),
|
|
56
|
+
("eval-memory", eval_memory_cmd.eval_memory_cmd),
|
|
57
|
+
("eval-hle", eval_hle_cmd.eval_hle_cmd),
|
|
58
|
+
("eval-claw", eval_claw_cmd.eval_claw_cmd),
|
|
59
|
+
("eval-compaction", eval_compaction_cmd.eval_compaction_cmd),
|
|
60
|
+
("eval-swe", eval_swe_cmd.eval_swe_cmd),
|
|
61
|
+
("lessons", lessons_cmd.lessons_cmd),
|
|
62
|
+
("sessions", sessions_cmd.sessions_cmd),
|
|
63
|
+
):
|
|
64
|
+
app.command(_name)(friendly(_fn))
|
|
65
|
+
|
|
66
|
+
if __name__ == "__main__":
|
|
67
|
+
app()
|