otto-cli-agent 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (129) hide show
  1. agent/README.md +22 -0
  2. agent/__init__.py +0 -0
  3. agent/cli/README.md +77 -0
  4. agent/cli/__init__.py +0 -0
  5. agent/cli/art.py +371 -0
  6. agent/cli/chat.py +309 -0
  7. agent/cli/clipboard.py +106 -0
  8. agent/cli/context.py +56 -0
  9. agent/cli/doctor.py +60 -0
  10. agent/cli/errors.py +58 -0
  11. agent/cli/eval.py +130 -0
  12. agent/cli/eval_claw.py +414 -0
  13. agent/cli/eval_compaction.py +106 -0
  14. agent/cli/eval_hle.py +92 -0
  15. agent/cli/eval_memory.py +253 -0
  16. agent/cli/eval_swe.py +172 -0
  17. agent/cli/lessons.py +97 -0
  18. agent/cli/main.py +67 -0
  19. agent/cli/modals.py +570 -0
  20. agent/cli/models.py +64 -0
  21. agent/cli/output.py +54 -0
  22. agent/cli/route.py +78 -0
  23. agent/cli/sessions.py +125 -0
  24. agent/cli/setup_screen.py +562 -0
  25. agent/cli/shell.py +548 -0
  26. agent/cli/tui.py +1807 -0
  27. agent/cli/ui.py +14 -0
  28. agent/cli/usage_panel.py +159 -0
  29. agent/config/README.md +7 -0
  30. agent/config/__init__.py +0 -0
  31. agent/config/envfile.py +76 -0
  32. agent/eval/README.md +76 -0
  33. agent/eval/__init__.py +0 -0
  34. agent/eval/claw_bench.py +1031 -0
  35. agent/eval/compaction_bench.py +229 -0
  36. agent/eval/data/README.md +10 -0
  37. agent/eval/data/claw/README.md +108 -0
  38. agent/eval/data/claw/llm_judge-gemini.patch +57 -0
  39. agent/eval/data/claw/otto.yaml +35 -0
  40. agent/eval/failures.py +276 -0
  41. agent/eval/golden/README.md +33 -0
  42. agent/eval/golden/code_01.json +6 -0
  43. agent/eval/golden/code_02.json +6 -0
  44. agent/eval/golden/code_03.json +6 -0
  45. agent/eval/golden/code_04.json +6 -0
  46. agent/eval/golden/code_05.json +6 -0
  47. agent/eval/golden/code_06.json +6 -0
  48. agent/eval/golden/math_01.json +6 -0
  49. agent/eval/golden/math_02.json +6 -0
  50. agent/eval/golden/math_03.json +6 -0
  51. agent/eval/golden/math_04.json +6 -0
  52. agent/eval/golden/math_05.json +6 -0
  53. agent/eval/golden/math_06.json +6 -0
  54. agent/eval/golden/nphard_gcp_01.json +6 -0
  55. agent/eval/golden/nphard_ksp_01.json +6 -0
  56. agent/eval/golden/nphard_math_binpacking_01.json +6 -0
  57. agent/eval/golden/nphard_math_clique_01.json +6 -0
  58. agent/eval/golden/nphard_math_setcover_01.json +6 -0
  59. agent/eval/golden/nphard_math_subsetsum_01.json +6 -0
  60. agent/eval/golden/nphard_tsp_01.json +6 -0
  61. agent/eval/golden/nphard_tsp_02.json +6 -0
  62. agent/eval/hle_bench.py +273 -0
  63. agent/eval/langfuse_sync.py +172 -0
  64. agent/eval/memory_bench.py +538 -0
  65. agent/eval/runner.py +174 -0
  66. agent/eval/single_agent.py +120 -0
  67. agent/eval/swe_bench.py +604 -0
  68. agent/eval/terminal_bench.py +345 -0
  69. agent/memory/README.md +102 -0
  70. agent/memory/__init__.py +42 -0
  71. agent/memory/embeddings.py +302 -0
  72. agent/memory/hashing.py +15 -0
  73. agent/memory/lessons.py +483 -0
  74. agent/memory/queue.py +531 -0
  75. agent/memory/retrieval.py +493 -0
  76. agent/memory/session.py +60 -0
  77. agent/memory/sessions.py +436 -0
  78. agent/memory/store.py +429 -0
  79. agent/memory/tokens.py +60 -0
  80. agent/memory/wiring.py +146 -0
  81. agent/pipeline/README.md +135 -0
  82. agent/pipeline/__init__.py +0 -0
  83. agent/pipeline/browsing.py +609 -0
  84. agent/pipeline/budget.py +403 -0
  85. agent/pipeline/codemap.py +254 -0
  86. agent/pipeline/evidence.py +325 -0
  87. agent/pipeline/execution.py +67 -0
  88. agent/pipeline/modes.py +137 -0
  89. agent/pipeline/native.py +1137 -0
  90. agent/pipeline/nodes.py +3644 -0
  91. agent/pipeline/pricing.py +209 -0
  92. agent/pipeline/progress.py +139 -0
  93. agent/pipeline/rag.py +139 -0
  94. agent/pipeline/research.py +1325 -0
  95. agent/pipeline/run.py +528 -0
  96. agent/pipeline/screen.py +77 -0
  97. agent/pipeline/state.py +220 -0
  98. agent/pipeline/toolkit.py +328 -0
  99. agent/pipeline/tools.py +1990 -0
  100. agent/pipeline/tracing.py +147 -0
  101. agent/pipeline/usage.py +251 -0
  102. agent/pipeline/vision.py +84 -0
  103. agent/pipeline/walkthrough.py +735 -0
  104. agent/pipeline/workspace.py +229 -0
  105. agent/router/README.md +60 -0
  106. agent/router/__init__.py +0 -0
  107. agent/router/automap.py +114 -0
  108. agent/router/health.py +229 -0
  109. agent/router/llm_provider/README.md +38 -0
  110. agent/router/llm_provider/__init__.py +202 -0
  111. agent/router/llm_provider/anthropic_provider.py +128 -0
  112. agent/router/llm_provider/base.py +507 -0
  113. agent/router/llm_provider/custom.py +152 -0
  114. agent/router/llm_provider/gemini_provider.py +122 -0
  115. agent/router/llm_provider/inception_provider.py +687 -0
  116. agent/router/llm_provider/openai_provider.py +151 -0
  117. agent/router/llm_provider/retired.py +145 -0
  118. agent/router/llm_provider/temperature.py +371 -0
  119. agent/router/mapping.py +579 -0
  120. agent/router/outcomes.py +363 -0
  121. agent/router/overrides.py +389 -0
  122. agent/router/reload.py +28 -0
  123. agent/router/router.py +413 -0
  124. agent/router/setup.py +123 -0
  125. otto_cli_agent-0.1.0.dist-info/METADATA +115 -0
  126. otto_cli_agent-0.1.0.dist-info/RECORD +129 -0
  127. otto_cli_agent-0.1.0.dist-info/WHEEL +4 -0
  128. otto_cli_agent-0.1.0.dist-info/entry_points.txt +2 -0
  129. otto_cli_agent-0.1.0.dist-info/licenses/LICENSE +21 -0
agent/cli/eval_hle.py ADDED
@@ -0,0 +1,92 @@
1
+ """`otto eval-hle` -- Humanity's Last Exam against Otto (agent/eval/hle_bench.py).
2
+
3
+ Two modes on the same sample, because the interesting number is not HLE's
4
+ published figure for this model but what Otto's graph does to it: `raw` asks
5
+ the model once, `agent` runs the whole router/planner/solver/evaluator graph
6
+ over the identical question and reports how many model calls that cost.
7
+
8
+ The dataset is gated -- accept the terms at huggingface.co/datasets/cais/hle
9
+ and set HF_TOKEN. Start with a small --limit: `agent` mode spends a full graph
10
+ run per question.
11
+ """
12
+ from __future__ import annotations
13
+
14
+ import json
15
+ from pathlib import Path
16
+ from typing import Optional
17
+
18
+ import typer
19
+ from typing_extensions import Annotated
20
+
21
+ from agent.cli.ui import err, out
22
+ from agent.eval.hle_bench import (
23
+ DEFAULT_CACHE,
24
+ DatasetGated,
25
+ download_hle,
26
+ load_hle,
27
+ run_hle,
28
+ sample_questions,
29
+ summarise,
30
+ )
31
+
32
+
33
+ def eval_hle_cmd(
34
+ mode: Annotated[
35
+ str, typer.Option(help="'raw' (one model call per question) or 'agent' (the whole graph)."),
36
+ ] = "raw",
37
+ limit: Annotated[
38
+ Optional[int], typer.Option(help="Score only this many questions (sampled deterministically)."),
39
+ ] = 25,
40
+ seed: Annotated[int, typer.Option(help="Sampling seed, so two modes score the SAME questions.")] = 0,
41
+ data_path: Annotated[
42
+ Optional[Path], typer.Option(help=f"Cached HLE parquet (default: {DEFAULT_CACHE}).")
43
+ ] = None,
44
+ hf_token: Annotated[
45
+ Optional[str], typer.Option(help="Hugging Face token; defaults to $HF_TOKEN.")
46
+ ] = None,
47
+ show_items: Annotated[
48
+ bool, typer.Option("--show-items", help="Print every question, answer and verdict."),
49
+ ] = False,
50
+ as_json: Annotated[bool, typer.Option("--json", help="Print the full report as JSON.")] = False,
51
+ ) -> None:
52
+ if mode not in {"raw", "agent"}:
53
+ err.print(f"mode must be 'raw' or 'agent', not {mode!r}")
54
+ raise typer.Exit(2)
55
+
56
+ path = data_path or DEFAULT_CACHE
57
+ try:
58
+ download_hle(path, token=hf_token)
59
+ except DatasetGated as exc:
60
+ err.print(str(exc))
61
+ raise typer.Exit(1)
62
+
63
+ rows, skipped = sample_questions(load_hle(path), limit=limit, seed=seed)
64
+ out.print(f"scoring {len(rows)} text-only question(s) in {mode} mode "
65
+ f"({skipped} image question(s) skipped -- Otto's chat path is text-only)")
66
+
67
+ def _progress(r):
68
+ mark = "correct" if r.correct else "wrong "
69
+ out.print(f" {mark} {r.llm_calls:>3} call(s) {r.question[:72]}")
70
+
71
+ results = run_hle(rows, mode=mode, on_result=None if as_json else _progress)
72
+ report = summarise(results)
73
+
74
+ if as_json:
75
+ out.print(json.dumps({"mode": mode, "skipped_image_questions": skipped,
76
+ "summary": report,
77
+ "items": [r.__dict__ for r in results]}, indent=2))
78
+ return
79
+
80
+ out.print("")
81
+ out.print(f"accuracy: {report['accuracy']:.1%} of {report['n']}"
82
+ f" model calls: {report['llm_calls_total']} "
83
+ f"({report['llm_calls_per_question']:.1f} per question)")
84
+ for name, stats in report["by_category"].items():
85
+ out.print(f" {name:<34} {stats['accuracy']:>6.1%} (n={stats['n']})")
86
+
87
+ if show_items:
88
+ out.print("")
89
+ for r in results:
90
+ out.print(f"[{'correct' if r.correct else 'wrong'}] {r.question[:110]}")
91
+ out.print(f" expected: {r.correct_answer[:110]}")
92
+ out.print(f" extracted: {r.extracted[:110]}")
@@ -0,0 +1,253 @@
1
+ """`otto eval-memory`: run agent/eval/memory_bench.py's LoCoMo-based memory
2
+ benchmark and report coverage. See that module's own docstring for what's
3
+ actually measured (store/visible-verbatim/recall/answerable coverage per QA
4
+ category, plus a compression ratio) and why the two `--x-budget`/`--y-budget`
5
+ options exist: real LoCoMo conversations are small enough that Otto's real
6
+ production budget (agent/memory/queue.py's X_BUDGET/Y_BUDGET) never triggers
7
+ compaction at all, so a run at the defaults mostly reports "everything is
8
+ still verbatim, nothing needed compacting" -- correct, but it doesn't
9
+ exercise the compaction+recall code path. Passing smaller values here forces
10
+ that path to actually run, at the cost of no longer matching Otto's real
11
+ per-turn budget.
12
+
13
+ No Langfuse tracking here (unlike `otto eval`, agent/cli/eval.py) -- this
14
+ benchmark doesn't run the pipeline or the router at all in its default,
15
+ offline mode (`--live` opts into one real LLM call per compaction, via
16
+ agent.memory.wiring.summarize_for_memory, agent/eval/memory_bench.py's own
17
+ module docstring), so there's no per-item trace worth syncing there.
18
+
19
+ `--show-items failures|all` prints each scored QA item's evidence text next
20
+ to recall()'s raw output -- worth reaching for whenever a coverage number
21
+ looks suspicious, since `recalled` is an exact-substring match (module
22
+ docstring point 3), not a fuzzy one: reading the two side by side is how
23
+ you tell a real semantic-search find from a coincidental match. It also
24
+ surfaces a real gap in the OFFLINE (`--no-live`, default) summarizer:
25
+ _canned_summarize groups raw turns into fixed-size blocks with placeholder
26
+ bullet text, and recall() returns a matched bullet's ENTIRE underlying raw
27
+ text -- so offline, "recalled" mostly tests "did the right block rank in
28
+ top-k," not "did it find the right sentence." `--live`'s real, topic-
29
+ scoped bullet summaries are the stricter version of this same test.
30
+ """
31
+ from __future__ import annotations
32
+
33
+ import json as json_module
34
+ from pathlib import Path
35
+ from typing import Annotated, Optional
36
+
37
+ import typer
38
+ from rich import box
39
+ from rich.table import Table
40
+
41
+ from agent.cli.ui import err, out
42
+ from agent.memory.retrieval import (
43
+ DEFAULT_MAX_CHUNKS,
44
+ DEFAULT_NEIGHBOUR_WINDOW,
45
+ DEFAULT_TOKEN_BUDGET,
46
+ )
47
+ from agent.eval.memory_bench import (
48
+ BenchmarkReport,
49
+ DEFAULT_CACHE,
50
+ download_locomo,
51
+ load_locomo,
52
+ recalled_text_has_unparsed_bullet,
53
+ run_benchmark,
54
+ )
55
+
56
+
57
+ def eval_memory_cmd(
58
+ live: Annotated[
59
+ bool,
60
+ typer.Option(
61
+ "--live/--no-live",
62
+ help="Use the real Task.SUMMARIZE model + local embeddings (needs INCEPTION_API_KEY) "
63
+ "instead of the deterministic offline canned summarizer.",
64
+ ),
65
+ ] = False,
66
+ samples: Annotated[
67
+ Optional[int], typer.Option(help="Only run the first N of LoCoMo's 10 conversations.")
68
+ ] = None,
69
+ max_turns: Annotated[
70
+ Optional[int], typer.Option(help="Truncate each conversation to its first N turns (fast smoke run).")
71
+ ] = None,
72
+ x_budget: Annotated[
73
+ Optional[int],
74
+ typer.Option(help="TieredQueue's X budget (tokens). Defaults to the "
75
+ "PRODUCTION budget, which LoCoMo's shorter "
76
+ "conversations do not overflow -- so the default "
77
+ "compacts nothing and the run is refused. Try 600."),
78
+ ] = None,
79
+ y_budget: Annotated[
80
+ Optional[int],
81
+ typer.Option(help="TieredQueue's Y budget (tokens). Same as --x-budget: "
82
+ "the production default never binds here. Try 1500."),
83
+ ] = None,
84
+ top_k: Annotated[int, typer.Option(help="How many bullets recall() considers per question (stage 1).")] = 5,
85
+ max_chunks: Annotated[
86
+ int,
87
+ typer.Option(help="How many raw chunks recall() actually returns (stage 2's cap)."),
88
+ ] = DEFAULT_MAX_CHUNKS,
89
+ neighbours: Annotated[
90
+ int,
91
+ typer.Option(help="How many chunks either side of each returned chunk to include."),
92
+ ] = DEFAULT_NEIGHBOUR_WINDOW,
93
+ token_budget: Annotated[
94
+ int,
95
+ typer.Option(help="Hard ceiling (tokens) on how much text one recall() returns."),
96
+ ] = DEFAULT_TOKEN_BUDGET,
97
+ data_path: Annotated[
98
+ Optional[Path], typer.Option(help="Path to a cached locomo10.json (default: agent/eval/data/locomo10.json).")
99
+ ] = None,
100
+ force_download: Annotated[
101
+ bool, typer.Option("--force-download", help="Re-download locomo10.json even if a cached copy exists.")
102
+ ] = False,
103
+ json: Annotated[
104
+ bool, typer.Option("--json", help="Print the full report as JSON instead of a table.")
105
+ ] = False,
106
+ show_items: Annotated[
107
+ str,
108
+ typer.Option(
109
+ help="Print per-question detail after the table: 'none' (default), 'failures' "
110
+ "(answerable=False items only -- the interesting/debug case), or 'all'. Each "
111
+ "item shows its evidence text and recall()'s raw output side by side, since "
112
+ "`recalled` is an exact-substring match, not a fuzzy one -- useful for checking "
113
+ "whether a pass/miss is real or an artifact of the conversation repeating similar "
114
+ "phrasing. Ignored when --json is set (the JSON report already includes both).",
115
+ ),
116
+ ] = "none",
117
+ ) -> None:
118
+ """Score agent/memory/'s tiered queue against the LoCoMo long-conversation-memory dataset."""
119
+ if show_items not in ("none", "failures", "all"):
120
+ err.print("[bad]--show-items must be 'none', 'failures', or 'all'[/]")
121
+ raise typer.Exit(2)
122
+
123
+ path = data_path or DEFAULT_CACHE
124
+ with err.status(f"fetching LoCoMo dataset ({path})…"):
125
+ try:
126
+ download_locomo(path, force=force_download)
127
+ except Exception as exc:
128
+ err.print(f"[bad]could not download the LoCoMo dataset: {exc}[/]")
129
+ err.print("[muted]pass --data-path to point at an already-downloaded locomo10.json[/]")
130
+ raise typer.Exit(1) from exc
131
+ all_samples = load_locomo(path)
132
+
133
+ if samples is not None:
134
+ all_samples = all_samples[:samples]
135
+ if not all_samples:
136
+ err.print("[bad]no LoCoMo samples to run[/]")
137
+ raise typer.Exit(1)
138
+
139
+ with err.status(f"running {len(all_samples)} conversation(s), live={live}…"):
140
+ report = run_benchmark(
141
+ all_samples, live=live, top_k=top_k, max_chunks=max_chunks,
142
+ neighbour_window=neighbours, token_budget=token_budget, max_turns=max_turns,
143
+ x_budget=x_budget, y_budget=y_budget,
144
+ )
145
+
146
+ if json:
147
+ out.print(json_module.dumps(report.to_dict(), indent=2))
148
+ return
149
+
150
+ summary = report.summary()
151
+
152
+ # REFUSE, rather than print a perfect-looking empty result. A run at the
153
+ # production budgets never compacts LoCoMo's shorter conversations, so
154
+ # every citation stays in the verbatim window, retrieval is never asked a
155
+ # question, and the table comes out `recalled 0%` beside `answerable
156
+ # 100%` -- which reads as a pass. The numbers are not wrong so much as
157
+ # measured off nothing, and printing them at all is what let this happen
158
+ # unnoticed. See memory_bench.NO_COMPACTION_RATIO.
159
+ if summary["conversation_count"] and not summary["recall_exercised"]:
160
+ ratio = summary["overall_compression_ratio"]
161
+ err.print(
162
+ "[bad]nothing was compacted at these budgets, so recall was never "
163
+ "exercised -- this run measured nothing and no coverage table is "
164
+ "printed for it.[/]\n"
165
+ f"[muted]final_view/raw token ratio "
166
+ f"{'-' if ratio is None else f'{ratio:.3f}'} over "
167
+ f"{summary['conversation_count']} conversation(s), at "
168
+ f"--x-budget {x_budget} --y-budget {y_budget}.[/]\n"
169
+ "[warn]re-run with smaller --x-budget/--y-budget -- e.g. "
170
+ "--x-budget 600 --y-budget 1500 -- so the conversations overflow "
171
+ "and the retrieval path actually runs.[/]"
172
+ )
173
+ raise typer.Exit(1)
174
+
175
+ t = Table(box=box.SIMPLE, header_style="muted")
176
+ t.add_column("category", style="spec")
177
+ t.add_column("n", justify="right")
178
+ t.add_column("store", justify="right")
179
+ t.add_column("verbatim", justify="right")
180
+ t.add_column("recalled", justify="right")
181
+ t.add_column("answerable", justify="right")
182
+
183
+ def _pct(value: Optional[float]) -> str:
184
+ return "-" if value is None else f"{value * 100:.0f}%"
185
+
186
+ for name, rates in summary["by_category"].items():
187
+ t.add_row(
188
+ name, str(rates["n"]), _pct(rates["store_coverage"]),
189
+ _pct(rates["visible_verbatim_coverage"]), _pct(rates["recall_coverage"]),
190
+ _pct(rates["answerable_coverage"]),
191
+ )
192
+ overall = summary["overall"]
193
+ t.add_row(
194
+ "[chosen]overall[/]", str(overall["n"]), _pct(overall["store_coverage"]),
195
+ _pct(overall["visible_verbatim_coverage"]), _pct(overall["recall_coverage"]),
196
+ _pct(overall["answerable_coverage"]),
197
+ )
198
+ out.print(t)
199
+
200
+ ratio = summary["overall_compression_ratio"]
201
+ ratio_text = "-" if ratio is None else f"{ratio:.3f}"
202
+ out.print(
203
+ f"[muted]{summary['conversation_count']} conversation(s) -- "
204
+ f"final_view/raw token ratio: {ratio_text}[/]"
205
+ )
206
+
207
+ unparsed = summary["unparsed_bullet_count"]
208
+ if unparsed:
209
+ err.print(
210
+ f"[warn]{unparsed}/{summary['bullet_count']} stored bullets are unparsed-summary "
211
+ f"placeholders (the summarizer's reply didn't parse for that compaction) -- "
212
+ f"recall_coverage is partly measuring ranking against content-free text, not a "
213
+ f"real summary. --show-items all will show which questions land on one.[/]"
214
+ )
215
+
216
+ if show_items != "none":
217
+ _print_items(report, show_items)
218
+
219
+ if overall["n"] and overall["store_coverage"] is not None and overall["store_coverage"] < 1.0:
220
+ err.print("[bad]store_coverage < 100% -- the engine lost cited evidence; this is a real bug[/]")
221
+ raise typer.Exit(1)
222
+
223
+
224
+ def _snippet(text: str, limit: int = 280) -> str:
225
+ text = " ".join(text.split())
226
+ return text if len(text) <= limit else text[:limit] + "…"
227
+
228
+
229
+ def _print_items(report: BenchmarkReport, which: str) -> None:
230
+ """Per-question detail: for every scored QA item matching `which`, the
231
+ exact evidence text next to recall()'s raw output -- the pair
232
+ `recalled` was computed from (an exact substring check, module
233
+ docstring point 3), so reading them side by side is how you tell a
234
+ real find from a coincidental match in a conversation that repeats
235
+ similar phrasing often.
236
+ """
237
+ out.print("")
238
+ shown = 0
239
+ for conv in report.conversations:
240
+ for r in conv.scored_results():
241
+ if which == "failures" and r.answerable:
242
+ continue
243
+ shown += 1
244
+ verdict = "[ok]answerable[/]" if r.answerable else "[bad]MISS[/]"
245
+ via = "verbatim" if r.visible_verbatim else ("recalled" if r.recalled else "neither")
246
+ flag = " [warn]unparsed-bullet[/]" if recalled_text_has_unparsed_bullet(r.recalled_text) else ""
247
+ out.print(f"[spec]{conv.sample_id}[/] [muted]({via})[/] {verdict}{flag}")
248
+ out.print(f" Q: {r.question}")
249
+ out.print(f" evidence: {_snippet(r.evidence_text)}")
250
+ out.print(f" recalled: {_snippet(r.recalled_text)}")
251
+ out.print("")
252
+ if shown == 0:
253
+ out.print("[muted](nothing matched --show-items filter)[/]")
agent/cli/eval_swe.py ADDED
@@ -0,0 +1,172 @@
1
+ """`otto eval-swe` -- SWE-bench Verified, graded by each repository's own tests.
2
+
3
+ The only benchmark here whose verdict nobody involved can argue with: 500 real
4
+ GitHub issues, and the maintainers' own test suite decides. See
5
+ agent/eval/swe_bench.py for what Otto is and is not told.
6
+
7
+ Needs Docker. The official per-instance images are about 3.5GB each and are
8
+ pulled on first use, so start with --limit 1 and expect the first run of a
9
+ repository to spend several minutes downloading before the agent starts.
10
+ """
11
+ from __future__ import annotations
12
+
13
+ import json
14
+ import time
15
+ from pathlib import Path
16
+ from typing import Optional
17
+
18
+ import typer
19
+ from typing_extensions import Annotated
20
+
21
+ from agent.cli.ui import err, out
22
+ from agent.eval.swe_bench import (
23
+ SweBenchUnavailable, load_dataset, resolve_image, run_instance, select,
24
+ )
25
+
26
+
27
+ def eval_swe_cmd(
28
+ repo: Annotated[
29
+ Optional[str], typer.Option(help="Only instances from repositories matching this."),
30
+ ] = None,
31
+ instance: Annotated[
32
+ Optional[str], typer.Option(help="Substring of an instance id, e.g. 'astropy-12907'."),
33
+ ] = None,
34
+ difficulty: Annotated[
35
+ Optional[str],
36
+ typer.Option(help="Filter on the dataset's own estimate: '<15', '15 min', '1-4', '>4'."),
37
+ ] = None,
38
+ limit: Annotated[int, typer.Option(help="Run at most this many instances.")] = 1,
39
+ max_seconds: Annotated[
40
+ float, typer.Option(help="Wall-clock budget per instance, before grading."),
41
+ ] = 1800.0,
42
+ max_model_calls: Annotated[
43
+ Optional[int], typer.Option(help="Override the default spend ceiling per instance."),
44
+ ] = None,
45
+ trace_dir: Annotated[
46
+ Optional[Path], typer.Option(help="Where to write each instance's diff."),
47
+ ] = None,
48
+ dataset: Annotated[
49
+ Optional[Path], typer.Option(help="Cached parquet (default: agent/eval/data/)."),
50
+ ] = None,
51
+ as_json: Annotated[bool, typer.Option("--json", help="Print the report as JSON.")] = False,
52
+ ) -> None:
53
+ try:
54
+ instances = select(
55
+ load_dataset(dataset), repo=repo, pattern=instance,
56
+ difficulty=difficulty, limit=limit,
57
+ )
58
+ except SweBenchUnavailable as exc:
59
+ err.print(str(exc))
60
+ raise typer.Exit(2) from None
61
+
62
+ if not instances:
63
+ err.print("no instances matched")
64
+ raise typer.Exit(1)
65
+
66
+ out_dir = Path(trace_dir) if trace_dir else Path("/tmp") / f"otto-swe-{time.strftime('%y%m%d-%H%M')}"
67
+ err.print(f"{len(instances)} instance(s) -> {out_dir}")
68
+
69
+ outcomes = []
70
+ for i, item in enumerate(instances, 1):
71
+ err.print(f"[{i}/{len(instances)}] {item.instance_id} ({item.difficulty or 'unrated'})")
72
+ image, emulated = resolve_image(item.instance_id)
73
+ err.print(f" image {image}"
74
+ + (" [warn](x86_64 under emulation -- no arm64 build)[/]"
75
+ if emulated else ""))
76
+ try:
77
+ outcome = run_instance(
78
+ item, max_seconds=max_seconds,
79
+ max_model_calls=max_model_calls, trace_dir=out_dir,
80
+ )
81
+ except SweBenchUnavailable as exc:
82
+ err.print(f" [bad]harness error[/] {exc}")
83
+ continue
84
+ outcomes.append(outcome)
85
+ flag = "[ok]resolved[/]" if outcome.resolved else "[warn]not resolved[/]"
86
+ detail = f" ({outcome.error})" if outcome.error else ""
87
+ out.print(
88
+ f" {flag} {outcome.grade.summary()} "
89
+ f"{outcome.diff_lines} changed line(s) calls={outcome.model_calls} "
90
+ f"{outcome.wall_time_s:.0f}s{detail}"
91
+ )
92
+ if outcome.grade.error:
93
+ err.print(f" [bad]{outcome.grade.error}[/]")
94
+ for failure in outcome.grade.failures[:3]:
95
+ err.print(f" still failing: {failure[:100]}")
96
+
97
+ if not outcomes:
98
+ raise typer.Exit(1)
99
+
100
+ resolved = sum(1 for o in outcomes if o.resolved)
101
+ # How much of this number came from emulated instances, and how it splits.
102
+ #
103
+ # Upstream publishes arm64 for only part of the set, and django -- 231 of
104
+ # the 500 instances -- had no arm64 build in every instance tried. So on
105
+ # Apple silicon roughly half the benchmark runs under emulation, several
106
+ # times slower, and a resolve rate that mixes the two is not a sample of
107
+ # SWE-bench. swe_bench.EMULATION_TIME_FACTOR stops the budget being the
108
+ # thing that decides those instances; it does NOT make them comparable,
109
+ # which is why the split is reported rather than smoothed away.
110
+ emulated = [o for o in outcomes if o.emulated]
111
+ native = [o for o in outcomes if not o.emulated]
112
+
113
+ def _rate(group):
114
+ if not group:
115
+ return None
116
+ return round(sum(1 for o in group if o.resolved) / len(group), 4)
117
+
118
+ report = {
119
+ "instances": len(outcomes),
120
+ "resolved": resolved,
121
+ "resolve_rate": round(resolved / len(outcomes), 4),
122
+ "emulated_instances": len(emulated),
123
+ "native_instances": len(native),
124
+ "emulated_resolve_rate": _rate(emulated),
125
+ "native_resolve_rate": _rate(native),
126
+ #: The repositories whose instances ran emulated, which is the thing
127
+ #: a reader needs to judge whether the sample is representative --
128
+ #: "half of it was django, emulated" is a different number from the
129
+ #: same rate measured natively.
130
+ "emulated_repos": sorted({o.instance_id.split("__")[0] for o in emulated}),
131
+ "mean_model_calls": round(sum(o.model_calls for o in outcomes) / len(outcomes), 1),
132
+ "mean_wall_time_s": round(sum(o.wall_time_s for o in outcomes) / len(outcomes), 1),
133
+ "trace_dir": str(out_dir),
134
+ "results": [
135
+ {
136
+ "instance_id": o.instance_id, "resolved": o.resolved,
137
+ "fail_to_pass": [o.grade.fail_to_pass_passed, o.grade.fail_to_pass_total],
138
+ "pass_to_pass": [o.grade.pass_to_pass_passed, o.grade.pass_to_pass_total],
139
+ "diff_lines": o.diff_lines, "model_calls": o.model_calls,
140
+ "emulated": o.emulated,
141
+ "wall_time_s": round(o.wall_time_s, 1),
142
+ "error": o.error or o.grade.error,
143
+ }
144
+ for o in outcomes
145
+ ],
146
+ }
147
+ out_dir.mkdir(parents=True, exist_ok=True)
148
+ (out_dir / "report.json").write_text(json.dumps(report, indent=2))
149
+
150
+ if as_json:
151
+ out.print_json(data=report)
152
+ return
153
+ out.print(
154
+ f"\n{resolved}/{len(outcomes)} resolved ({report['resolve_rate']:.1%}) "
155
+ f"{report['mean_model_calls']:.0f} model calls and "
156
+ f"{report['mean_wall_time_s']:.0f}s per instance"
157
+ )
158
+ if emulated:
159
+ # A SWE-bench number from a machine that emulated part of its sample
160
+ # has to say so, in the same breath as the number.
161
+ native_text = (
162
+ f"{report['native_resolve_rate']:.1%}" if native else "no native instances"
163
+ )
164
+ err.print(
165
+ f"[warn]{len(emulated)}/{len(outcomes)} instances ran under x86_64 "
166
+ f"emulation (no arm64 build): "
167
+ f"{', '.join(report['emulated_repos'])}.[/]\n"
168
+ f"[muted]emulated {report['emulated_resolve_rate']:.1%} vs native "
169
+ f"{native_text} -- quote the split, not the combined rate, and say "
170
+ f"this machine emulated part of the sample.[/]"
171
+ )
172
+ out.print(f"report: {out_dir / 'report.json'}")
agent/cli/lessons.py ADDED
@@ -0,0 +1,97 @@
1
+ """`otto lessons` -- read, move, and if need be empty, what Otto has learned.
2
+
3
+ A self-improving system whose learned state cannot be read is not a system
4
+ anyone should trust, and this is the cheapest possible remedy: the bank is a
5
+ few short lines of text, so print them.
6
+
7
+ There is a second reason it exists. Evolution that is not working is a real
8
+ and documented outcome -- across five methods and three frontier models the
9
+ measured gains were around +1%, and negative in every regime for the
10
+ strongest model. Being able to look at the bank and say "these lessons are
11
+ rubbish" is how that gets noticed, and `--clear` is how it gets undone.
12
+
13
+ `--export` / `--import` (2026-09-12) move a bank between machines as JSON,
14
+ through the same duplicate adjudication a run's own lessons face. The TUI's
15
+ "Export lessons…" / "Import lessons…" palette entries call the same two
16
+ functions (agent/memory/lessons.py).
17
+ """
18
+ from __future__ import annotations
19
+
20
+ from pathlib import Path
21
+ from typing import Optional
22
+
23
+ import typer
24
+ from rich import box
25
+ from rich.table import Table
26
+ from rich.text import Text
27
+ from typing_extensions import Annotated
28
+
29
+ from agent.cli.ui import err, out
30
+ from agent.memory.lessons import (
31
+ Lesson, all_lessons, bank_path, bind_bank, clear_bank, export_lessons, import_lessons,
32
+ )
33
+ from agent.memory.store import MemoryStore
34
+
35
+
36
+ def lessons_table(lessons: list[Lesson], path: Path | None = None) -> Table:
37
+ """The bank as a table. Primitive Rich styles only, so it renders the same
38
+ in the REPL (whose THEME maps ok/warn to exactly these) and inside a
39
+ Textual Static, which never sees that THEME."""
40
+ title = f"{len(lessons)} lesson(s)" + (f" in {path}" if path else "")
41
+ t = Table(box=box.SIMPLE, header_style="dim", title=title, title_justify="left")
42
+ t.add_column("outcome")
43
+ t.add_column("when")
44
+ t.add_column("do", overflow="fold")
45
+ for lesson in lessons:
46
+ mark = Text("worked", style="bold green") if lesson.outcome == "worked" else Text("failed", style="yellow")
47
+ t.add_row(mark, lesson.cue, lesson.action)
48
+ return t
49
+
50
+
51
+ def lessons_cmd(
52
+ bank: Annotated[
53
+ Optional[Path],
54
+ typer.Option(help="Which bank to read (default: ~/.otto/memory/lessons.db)."),
55
+ ] = None,
56
+ clear: Annotated[
57
+ bool,
58
+ typer.Option("--clear", help="Delete every lesson. Not undoable."),
59
+ ] = False,
60
+ export: Annotated[
61
+ Optional[Path],
62
+ typer.Option("--export", help="Write the bank to this JSON file."),
63
+ ] = None,
64
+ import_from: Annotated[
65
+ Optional[Path],
66
+ typer.Option("--import", help="Merge lessons from this JSON file into the bank."),
67
+ ] = None,
68
+ replace: Annotated[
69
+ bool,
70
+ typer.Option("--replace", help="With --import: empty the bank first."),
71
+ ] = False,
72
+ ) -> None:
73
+ path = bank or bank_path()
74
+ if import_from is None and not path.exists():
75
+ out.print(f"no lesson bank at {path} -- nothing learned yet")
76
+ return
77
+
78
+ store = MemoryStore(path)
79
+ with bind_bank(store):
80
+ if clear:
81
+ removed = clear_bank()
82
+ err.print(f"cleared {removed} lesson(s) from {path}")
83
+ return
84
+ if export is not None:
85
+ n = export_lessons(export)
86
+ err.print(f"exported {n} lesson(s) to {export}")
87
+ return
88
+ if import_from is not None:
89
+ report = import_lessons(import_from, replace=replace)
90
+ err.print(f"imported from {import_from}: {report.summary()}")
91
+ return
92
+
93
+ learned = all_lessons()
94
+ if not learned:
95
+ out.print(f"{path} is empty")
96
+ return
97
+ out.print(lessons_table(learned, path))
agent/cli/main.py ADDED
@@ -0,0 +1,67 @@
1
+ from dotenv import load_dotenv
2
+ from typing import Annotated
3
+ import typer
4
+
5
+ # Loaded here, at import time, before anything below it is imported -- not
6
+ # inside bootstrap(). agent.cli.chat now pulls in the whole pipeline
7
+ # (agent.pipeline.nodes), which constructs a module-level Router() the
8
+ # moment it's imported, which reads *_API_KEY from the environment
9
+ # immediately. bootstrap() only runs once Typer has already finished
10
+ # importing every command module, which is too late for that first
11
+ # Router() call -- it would always see an environment with no keys in it,
12
+ # whatever's actually in .env.
13
+ from agent.config.envfile import ENV_PATH # the one file the setup screen writes
14
+
15
+ load_dotenv(ENV_PATH)
16
+
17
+ # Also before anything below: ~/.otto/routes.json may name custom endpoints
18
+ # and per-task pins, and the module-level Router() in agent.pipeline.nodes
19
+ # snapshots which providers exist the moment it is imported.
20
+ from agent.router.overrides import apply_at_startup # noqa: E402
21
+
22
+ apply_at_startup()
23
+
24
+ from agent.cli.context import AppContext
25
+ from agent.cli.errors import friendly
26
+ from agent.cli import doctor as doctor_cmd
27
+ from agent.cli import eval as eval_cmd
28
+ from agent.cli import eval_claw as eval_claw_cmd
29
+ from agent.cli import eval_compaction as eval_compaction_cmd
30
+ from agent.cli import eval_swe as eval_swe_cmd
31
+ from agent.cli import eval_hle as eval_hle_cmd
32
+ from agent.cli import eval_memory as eval_memory_cmd
33
+ from agent.cli import lessons as lessons_cmd
34
+ from agent.cli import models as model_cmd
35
+ from agent.cli import route as route_cmd
36
+ from agent.cli import sessions as sessions_cmd
37
+ from agent.cli import chat as chat_cmd
38
+ from agent.cli import tui as tui_cmd
39
+
40
+ app = typer.Typer()
41
+
42
+ @app.callback()
43
+ def bootstrap(ctx: typer.Context, strict: Annotated[bool, typer.Option()] = False) -> None:
44
+ ctx.obj = AppContext(strict=strict)
45
+
46
+ # `friendly` is applied here, once, rather than as a decorator on each command
47
+ # module. One registration site means a new command cannot forget it, and the
48
+ # command modules stay free of CLI exit-code concerns.
49
+ for _name, _fn in (
50
+ ("doctor", doctor_cmd.doctor),
51
+ ("models", model_cmd.models),
52
+ ("route", route_cmd.route),
53
+ ("chat", chat_cmd.chat),
54
+ ("tui", tui_cmd.tui),
55
+ ("eval", eval_cmd.eval_cmd),
56
+ ("eval-memory", eval_memory_cmd.eval_memory_cmd),
57
+ ("eval-hle", eval_hle_cmd.eval_hle_cmd),
58
+ ("eval-claw", eval_claw_cmd.eval_claw_cmd),
59
+ ("eval-compaction", eval_compaction_cmd.eval_compaction_cmd),
60
+ ("eval-swe", eval_swe_cmd.eval_swe_cmd),
61
+ ("lessons", lessons_cmd.lessons_cmd),
62
+ ("sessions", sessions_cmd.sessions_cmd),
63
+ ):
64
+ app.command(_name)(friendly(_fn))
65
+
66
+ if __name__ == "__main__":
67
+ app()