otto-cli-agent 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (129) hide show
  1. agent/README.md +22 -0
  2. agent/__init__.py +0 -0
  3. agent/cli/README.md +77 -0
  4. agent/cli/__init__.py +0 -0
  5. agent/cli/art.py +371 -0
  6. agent/cli/chat.py +309 -0
  7. agent/cli/clipboard.py +106 -0
  8. agent/cli/context.py +56 -0
  9. agent/cli/doctor.py +60 -0
  10. agent/cli/errors.py +58 -0
  11. agent/cli/eval.py +130 -0
  12. agent/cli/eval_claw.py +414 -0
  13. agent/cli/eval_compaction.py +106 -0
  14. agent/cli/eval_hle.py +92 -0
  15. agent/cli/eval_memory.py +253 -0
  16. agent/cli/eval_swe.py +172 -0
  17. agent/cli/lessons.py +97 -0
  18. agent/cli/main.py +67 -0
  19. agent/cli/modals.py +570 -0
  20. agent/cli/models.py +64 -0
  21. agent/cli/output.py +54 -0
  22. agent/cli/route.py +78 -0
  23. agent/cli/sessions.py +125 -0
  24. agent/cli/setup_screen.py +562 -0
  25. agent/cli/shell.py +548 -0
  26. agent/cli/tui.py +1807 -0
  27. agent/cli/ui.py +14 -0
  28. agent/cli/usage_panel.py +159 -0
  29. agent/config/README.md +7 -0
  30. agent/config/__init__.py +0 -0
  31. agent/config/envfile.py +76 -0
  32. agent/eval/README.md +76 -0
  33. agent/eval/__init__.py +0 -0
  34. agent/eval/claw_bench.py +1031 -0
  35. agent/eval/compaction_bench.py +229 -0
  36. agent/eval/data/README.md +10 -0
  37. agent/eval/data/claw/README.md +108 -0
  38. agent/eval/data/claw/llm_judge-gemini.patch +57 -0
  39. agent/eval/data/claw/otto.yaml +35 -0
  40. agent/eval/failures.py +276 -0
  41. agent/eval/golden/README.md +33 -0
  42. agent/eval/golden/code_01.json +6 -0
  43. agent/eval/golden/code_02.json +6 -0
  44. agent/eval/golden/code_03.json +6 -0
  45. agent/eval/golden/code_04.json +6 -0
  46. agent/eval/golden/code_05.json +6 -0
  47. agent/eval/golden/code_06.json +6 -0
  48. agent/eval/golden/math_01.json +6 -0
  49. agent/eval/golden/math_02.json +6 -0
  50. agent/eval/golden/math_03.json +6 -0
  51. agent/eval/golden/math_04.json +6 -0
  52. agent/eval/golden/math_05.json +6 -0
  53. agent/eval/golden/math_06.json +6 -0
  54. agent/eval/golden/nphard_gcp_01.json +6 -0
  55. agent/eval/golden/nphard_ksp_01.json +6 -0
  56. agent/eval/golden/nphard_math_binpacking_01.json +6 -0
  57. agent/eval/golden/nphard_math_clique_01.json +6 -0
  58. agent/eval/golden/nphard_math_setcover_01.json +6 -0
  59. agent/eval/golden/nphard_math_subsetsum_01.json +6 -0
  60. agent/eval/golden/nphard_tsp_01.json +6 -0
  61. agent/eval/golden/nphard_tsp_02.json +6 -0
  62. agent/eval/hle_bench.py +273 -0
  63. agent/eval/langfuse_sync.py +172 -0
  64. agent/eval/memory_bench.py +538 -0
  65. agent/eval/runner.py +174 -0
  66. agent/eval/single_agent.py +120 -0
  67. agent/eval/swe_bench.py +604 -0
  68. agent/eval/terminal_bench.py +345 -0
  69. agent/memory/README.md +102 -0
  70. agent/memory/__init__.py +42 -0
  71. agent/memory/embeddings.py +302 -0
  72. agent/memory/hashing.py +15 -0
  73. agent/memory/lessons.py +483 -0
  74. agent/memory/queue.py +531 -0
  75. agent/memory/retrieval.py +493 -0
  76. agent/memory/session.py +60 -0
  77. agent/memory/sessions.py +436 -0
  78. agent/memory/store.py +429 -0
  79. agent/memory/tokens.py +60 -0
  80. agent/memory/wiring.py +146 -0
  81. agent/pipeline/README.md +135 -0
  82. agent/pipeline/__init__.py +0 -0
  83. agent/pipeline/browsing.py +609 -0
  84. agent/pipeline/budget.py +403 -0
  85. agent/pipeline/codemap.py +254 -0
  86. agent/pipeline/evidence.py +325 -0
  87. agent/pipeline/execution.py +67 -0
  88. agent/pipeline/modes.py +137 -0
  89. agent/pipeline/native.py +1137 -0
  90. agent/pipeline/nodes.py +3644 -0
  91. agent/pipeline/pricing.py +209 -0
  92. agent/pipeline/progress.py +139 -0
  93. agent/pipeline/rag.py +139 -0
  94. agent/pipeline/research.py +1325 -0
  95. agent/pipeline/run.py +528 -0
  96. agent/pipeline/screen.py +77 -0
  97. agent/pipeline/state.py +220 -0
  98. agent/pipeline/toolkit.py +328 -0
  99. agent/pipeline/tools.py +1990 -0
  100. agent/pipeline/tracing.py +147 -0
  101. agent/pipeline/usage.py +251 -0
  102. agent/pipeline/vision.py +84 -0
  103. agent/pipeline/walkthrough.py +735 -0
  104. agent/pipeline/workspace.py +229 -0
  105. agent/router/README.md +60 -0
  106. agent/router/__init__.py +0 -0
  107. agent/router/automap.py +114 -0
  108. agent/router/health.py +229 -0
  109. agent/router/llm_provider/README.md +38 -0
  110. agent/router/llm_provider/__init__.py +202 -0
  111. agent/router/llm_provider/anthropic_provider.py +128 -0
  112. agent/router/llm_provider/base.py +507 -0
  113. agent/router/llm_provider/custom.py +152 -0
  114. agent/router/llm_provider/gemini_provider.py +122 -0
  115. agent/router/llm_provider/inception_provider.py +687 -0
  116. agent/router/llm_provider/openai_provider.py +151 -0
  117. agent/router/llm_provider/retired.py +145 -0
  118. agent/router/llm_provider/temperature.py +371 -0
  119. agent/router/mapping.py +579 -0
  120. agent/router/outcomes.py +363 -0
  121. agent/router/overrides.py +389 -0
  122. agent/router/reload.py +28 -0
  123. agent/router/router.py +413 -0
  124. agent/router/setup.py +123 -0
  125. otto_cli_agent-0.1.0.dist-info/METADATA +115 -0
  126. otto_cli_agent-0.1.0.dist-info/RECORD +129 -0
  127. otto_cli_agent-0.1.0.dist-info/WHEEL +4 -0
  128. otto_cli_agent-0.1.0.dist-info/entry_points.txt +2 -0
  129. otto_cli_agent-0.1.0.dist-info/licenses/LICENSE +21 -0
agent/cli/eval.py ADDED
@@ -0,0 +1,130 @@
1
+ """`otto eval`: run the golden dataset through the real pipeline and report
2
+ pass/fail per item. See agent/eval/runner.py for what "checked" means, and
3
+ agent/eval/golden/ for the items themselves.
4
+
5
+ By default this tracks the run as a Langfuse Dataset Run (agent/eval/
6
+ langfuse_sync.py): the golden set is synced to a Langfuse Dataset once,
7
+ then every `otto eval` invocation becomes one named, comparable run
8
+ against it in the Langfuse UI -- every item's trace, its pass/fail score,
9
+ and the run as a whole. Pass --no-experiment for the old local-only path
10
+ (no Langfuse project needed -- useful offline or in CI without Langfuse
11
+ creds), which prints a plain pass/fail table and nothing else.
12
+
13
+ No --agents option (2026-09-10): the router/planner/solver/summarizer/
14
+ finder/evaluator graph that replaced the swarm pipeline has nothing to
15
+ size -- see agent/pipeline/run.py's module docstring.
16
+ """
17
+ from typing import Annotated, Optional
18
+
19
+ import typer
20
+ from rich import box
21
+ from rich.table import Table
22
+
23
+ from agent.cli.ui import err, out
24
+ from agent.eval.runner import run_golden
25
+
26
+
27
+ def eval_cmd(
28
+ domain: Annotated[Optional[str], typer.Option(help="Only 'code' or 'math'.")] = None,
29
+ experiment: Annotated[
30
+ bool,
31
+ typer.Option(
32
+ "--experiment/--no-experiment",
33
+ help="Track this run as a Langfuse Dataset Run (default) vs. a local-only table.",
34
+ ),
35
+ ] = True,
36
+ run_name: Annotated[
37
+ Optional[str], typer.Option(help="Name for the Langfuse experiment run (--experiment only).")
38
+ ] = None,
39
+ ) -> None:
40
+ """Run the golden dataset through the pipeline and report pass/fail."""
41
+ if domain is not None and domain not in ("code", "math"):
42
+ err.print("[bad]--domain must be 'code' or 'math'[/]")
43
+ raise typer.Exit(2)
44
+
45
+ if experiment:
46
+ _run_as_langfuse_experiment(domain=domain, run_name=run_name)
47
+ return
48
+
49
+ with err.status("running golden set…"):
50
+ results = run_golden(only_domain=domain)
51
+
52
+ if not results:
53
+ out.print("[muted]no golden items matched[/]")
54
+ return
55
+
56
+ t = Table(box=box.SIMPLE, header_style="muted")
57
+ t.add_column("id", style="spec")
58
+ t.add_column("domain", style="muted")
59
+ t.add_column("result")
60
+ t.add_column("seconds", justify="right")
61
+ t.add_column("evidence", style="muted")
62
+ passed = 0
63
+ for r in results:
64
+ mark = "[ok]pass[/]" if r.passed else "[bad]fail[/]"
65
+ passed += r.passed
66
+ evidence = r.evidence if len(r.evidence) <= 80 else r.evidence[:80] + "…"
67
+ t.add_row(r.item_id, r.domain, mark, f"{r.seconds:.1f}", evidence.replace("\n", " "))
68
+ out.print(t)
69
+ out.print(f"[muted]{passed}/{len(results)} passed[/]")
70
+
71
+ _print_full_failures((r.item_id, r.evidence) for r in results if not r.passed)
72
+ if passed < len(results):
73
+ raise typer.Exit(1)
74
+
75
+
76
+ def _print_full_failures(failures) -> None:
77
+ """The overview table truncates evidence to 80 chars for readability --
78
+ fine for a scan, useless for actually debugging a failure (it cuts a
79
+ checker's stderr right where the real exception would start). Print
80
+ each failing item's full evidence separately so that's never the
81
+ reason you have to go dig through the Langfuse UI.
82
+ """
83
+ failures = list(failures)
84
+ if not failures:
85
+ return
86
+ out.print("\n[bad]failures, in full:[/]")
87
+ for item_id, evidence in failures:
88
+ out.print(f"[spec]{item_id}[/]")
89
+ out.print(evidence)
90
+ out.print("")
91
+
92
+
93
+ def _run_as_langfuse_experiment(*, domain: Optional[str], run_name: Optional[str]) -> None:
94
+ from agent.eval.langfuse_sync import run_golden_experiment
95
+
96
+ with err.status("syncing golden set to langfuse and running experiment…"):
97
+ try:
98
+ result = run_golden_experiment(only_domain=domain, run_name=run_name)
99
+ except Exception as exc:
100
+ err.print(f"[bad]langfuse experiment failed: {exc}[/]")
101
+ err.print("[muted]is LANGFUSE_PUBLIC_KEY / LANGFUSE_SECRET_KEY set? falling back to --no-experiment still works offline.[/]")
102
+ raise typer.Exit(1) from exc
103
+
104
+ t = Table(box=box.SIMPLE, header_style="muted")
105
+ t.add_column("id", style="spec")
106
+ t.add_column("result")
107
+ t.add_column("evidence", style="muted")
108
+ all_passed = True
109
+ for item_result in result.item_results:
110
+ golden_id = getattr(item_result.item, "id", "?")
111
+ ev = next((e for e in item_result.evaluations if e.name == "golden_pass"), None)
112
+ item_passed = bool(ev.value) if ev is not None else False
113
+ all_passed &= item_passed
114
+ mark = "[ok]pass[/]" if item_passed else "[bad]fail[/]"
115
+ comment = (ev.comment or "") if ev is not None else "no golden_pass score"
116
+ comment = comment if len(comment) <= 80 else comment[:80] + "…"
117
+ t.add_row(golden_id, mark, comment.replace("\n", " "))
118
+ out.print(t)
119
+
120
+ n = len(result.item_results)
121
+ n_passed = sum(
122
+ 1 for r in result.item_results
123
+ if any(e.name == "golden_pass" and e.value for e in r.evaluations)
124
+ )
125
+ out.print(f"[muted]{n_passed}/{n} passed[/]")
126
+ if result.dataset_run_url:
127
+ out.print(f"[muted]langfuse dataset run: {result.dataset_run_url}[/]")
128
+
129
+ if not all_passed:
130
+ raise typer.Exit(1)
agent/cli/eval_claw.py ADDED
@@ -0,0 +1,414 @@
1
+ """`otto eval-claw` -- Claw-Eval's 300 tasks against Otto's own agent.
2
+
3
+ Their `claw-eval run` can only point at a model endpoint, so the number it
4
+ gives is about a model inside their agent loop. This command runs OTTO on the
5
+ task -- its graph, its memory, its tools, its evaluator -- and hands the
6
+ resulting trace to their own graders, so the score is about the agent.
7
+ See agent/eval/claw_bench.py for the seam that makes that possible.
8
+
9
+ Needs a Claw-Eval checkout (--claw-root or $CLAW_EVAL_ROOT) with its
10
+ requirements installed, and Docker for the 169 tasks whose files live in a
11
+ container. Start with --limit: every task is a full agent run.
12
+ """
13
+ from __future__ import annotations
14
+
15
+ import json
16
+ import time
17
+ from contextlib import ExitStack
18
+ from pathlib import Path
19
+ from types import SimpleNamespace
20
+ from typing import Optional
21
+
22
+ import typer
23
+ from typing_extensions import Annotated
24
+
25
+ from agent.cli.ui import err, out
26
+ from agent.eval import failures
27
+ from agent.memory.lessons import bind_bank, read_only
28
+ from agent.memory.store import MemoryStore
29
+ from agent.router.outcomes import bind_log
30
+ from agent.router.outcomes import read_only as routing_read_only
31
+ from agent.eval.claw_bench import (
32
+ ClawEvalUnavailable,
33
+ claw_root,
34
+ grading_fingerprint,
35
+ load_claw,
36
+ missing_service_keys,
37
+ run_task_file,
38
+ select_tasks,
39
+ split_tasks,
40
+ )
41
+
42
+
43
+ #: How many trials per task before a Claw-Eval number is evidence of
44
+ #: anything.
45
+ #:
46
+ #: Measured, not chosen. Task T093 on identical code and configuration, three
47
+ #: runs: 0.86, 0.60, 0.96 -- a spread of 0.36 on a 0-1 scale. Over two trials
48
+ #: pass^2 came out 0.25 against pass@2 of 1.00: it got there once and not
49
+ #: twice. Completion is LLM-judged for 260 of the 300 tasks, which is where
50
+ #: most of that comes from.
51
+ #:
52
+ #: The consequence is retroactive and worth stating plainly: any single-run
53
+ #: before/after difference smaller than about 0.3 on this benchmark is inside
54
+ #: the noise. Much of the optimisation work on this branch was read from one
55
+ #: run per task, and the only difference in that set large enough to survive
56
+ #: was a crash going from 0.00 to 0.955.
57
+ #:
58
+ #: Three, matching what the issues here already ask for when they say how to
59
+ #: settle something ("--trials 3").
60
+ MIN_TRIALS_FOR_EVIDENCE = 3
61
+
62
+ #: The observed spread above, quoted in the warning so the number a reader is
63
+ #: being told to distrust comes with the reason.
64
+ OBSERVED_SINGLE_RUN_SPREAD = 0.36
65
+
66
+
67
+ def _summary(outcomes: list, claw=None) -> dict:
68
+ scored = [o for o in outcomes if o is not None]
69
+ if not scored:
70
+ return {"tasks": 0}
71
+ n = len(scored)
72
+ summary = {
73
+ "tasks": n,
74
+ "passed": sum(1 for o in scored if o.passed),
75
+ "pass_rate": round(sum(1 for o in scored if o.passed) / n, 4),
76
+ "mean_task_score": round(sum(o.task_score for o in scored) / n, 4),
77
+ "mean_completion": round(sum(o.completion for o in scored) / n, 4),
78
+ "mean_robustness": round(sum(o.robustness for o in scored) / n, 4),
79
+ "mean_communication": round(sum(o.communication for o in scored) / n, 4),
80
+ "errors": sum(1 for o in scored if o.error),
81
+ "mean_wall_time_s": round(sum(o.wall_time_s for o in scored) / n, 2),
82
+ # The cost axis. None of the 300 graders read it, so without this a
83
+ # change that doubles spend for a tenth of a point reads as a win.
84
+ "mean_model_calls": round(sum(o.model_calls for o in scored) / n, 2),
85
+ "total_model_calls": sum(o.model_calls for o in scored),
86
+ }
87
+
88
+ # Always present, so a reader of otto_summary.json never has to infer the
89
+ # trial count from whether a reliability key happens to exist.
90
+ summary["trials"] = min((len(o.trials) for o in scored), default=0)
91
+ #: Whether this run is enough trials to be read as a measurement at all --
92
+ #: MIN_TRIALS_FOR_EVIDENCE. False does NOT mean the numbers are wrong; it
93
+ #: means the difference between them and another run's is not attributable
94
+ #: to anything. Carried in the JSON as well as printed, because the JSON is
95
+ #: what gets pasted into a comparison months later.
96
+ summary["is_evidence"] = summary["trials"] >= MIN_TRIALS_FOR_EVIDENCE
97
+
98
+ # What KIND of failure, and what the tools cost -- agent/eval/failures.py,
99
+ # read off the action lines every run already produced, with no model
100
+ # call. A score says something got worse; "38% of the failures changed no
101
+ # files at all" says where to look.
102
+ failed = {o.task_id: getattr(o, "actions", None)
103
+ for o in scored if not o.passed}
104
+ if failed:
105
+ summary["failures"] = failures.summarise(failed)
106
+ summary["tool_cost"] = failures.total_tool_cost(
107
+ getattr(o, "actions", None) for o in scored
108
+ ).to_dict()
109
+
110
+ repeated = [o for o in scored if len(o.trials) > 1]
111
+ if repeated and claw is not None:
112
+ k = min(len(o.trials) for o in repeated)
113
+ summary["trials"] = k
114
+ # Per task first, then averaged: pass^k over a task's own repeats says
115
+ # "does it do this reliably", which is the property a self-improving
116
+ # loop is most able to fake by finding one lucky trial.
117
+ summary["mean_pass_hat_k"] = round(
118
+ sum(claw.compute_pass_hat_k(o.trials, k=k) for o in repeated) / len(repeated), 4
119
+ )
120
+ summary["mean_pass_at_k"] = round(
121
+ sum(claw.compute_pass_at_k(o.trials, k=k) for o in repeated) / len(repeated), 4
122
+ )
123
+ summary["score_spread"] = round(
124
+ sum(max(o.trials) - min(o.trials) for o in repeated) / len(repeated), 4
125
+ )
126
+ return summary
127
+
128
+
129
+ def eval_claw_cmd(
130
+ claw_root_opt: Annotated[
131
+ Optional[Path],
132
+ typer.Option("--claw-root", help="Claw-Eval checkout; defaults to $CLAW_EVAL_ROOT."),
133
+ ] = None,
134
+ tag: Annotated[
135
+ Optional[str],
136
+ typer.Option(help="Capability to run: general, multimodal, user_agent, multi_service."),
137
+ ] = None,
138
+ task: Annotated[
139
+ Optional[str], typer.Option(help="Substring of a task directory name, e.g. 'T01' or 'M001'."),
140
+ ] = None,
141
+ limit: Annotated[Optional[int], typer.Option(help="Run at most this many tasks.")] = 5,
142
+ architecture: Annotated[
143
+ str, typer.Option(help="'graph' (the real pipeline) or 'single' (one-conversation control)."),
144
+ ] = "graph",
145
+ config: Annotated[
146
+ Optional[Path], typer.Option(help="Claw-Eval config.yaml (judge, sandbox, user-agent models)."),
147
+ ] = None,
148
+ trace_dir: Annotated[
149
+ Optional[Path], typer.Option(help="Where to write traces (default: <checkout>/traces/otto_<time>).")
150
+ ] = None,
151
+ no_judge: Annotated[bool, typer.Option("--no-judge", help="Skip the LLM judge.")] = False,
152
+ port_offset: Annotated[int, typer.Option(help="Shift every mock service port, for parallel runs.")] = 0,
153
+ max_seconds: Annotated[
154
+ Optional[float],
155
+ typer.Option(help="Cap each task's budget below its own (tasks allow 120-900s). "
156
+ "Cheaper samples, and lower scores -- say so when reporting."),
157
+ ] = None,
158
+ trials: Annotated[
159
+ int,
160
+ typer.Option(help="Run each task this many times and report pass^k and the "
161
+ "spread. One run cannot tell a real change from judge "
162
+ "variance -- 260 of the 300 tasks are LLM-judged."),
163
+ ] = 1,
164
+ split: Annotated[
165
+ str,
166
+ typer.Option(help="'all', 'dev' (tune against these), or 'holdout' (never "
167
+ "tuned against; the honest number). The split is a hash "
168
+ "of the task id, so it does not move between runs."),
169
+ ] = "all",
170
+ lesson_bank: Annotated[
171
+ Optional[Path],
172
+ typer.Option(help="Lesson bank to read and write (default: ~/.otto/memory/"
173
+ "lessons.db). Point a measurement at its own file so it "
174
+ "does not learn from -- or teach -- your working one."),
175
+ ] = None,
176
+ no_learning: Annotated[
177
+ bool,
178
+ typer.Option("--no-learning", help="Read no lessons and write none. This is "
179
+ "the compute-matched baseline every "
180
+ "self-improvement claim has to be shown "
181
+ "beside."),
182
+ ] = False,
183
+ as_json: Annotated[bool, typer.Option("--json", help="Print the full report as JSON.")] = False,
184
+ ) -> None:
185
+ if architecture not in {"graph", "single"}:
186
+ err.print(f"architecture must be 'graph' or 'single', not {architecture!r}")
187
+ raise typer.Exit(2)
188
+ if split not in {"all", "dev", "holdout"}:
189
+ err.print(f"split must be 'all', 'dev' or 'holdout', not {split!r}")
190
+ raise typer.Exit(2)
191
+ if trials < 1:
192
+ err.print("trials must be at least 1")
193
+ raise typer.Exit(2)
194
+
195
+ try:
196
+ root = claw_root(claw_root_opt)
197
+ claw = load_claw(root)
198
+ except ClawEvalUnavailable as exc:
199
+ err.print(str(exc))
200
+ raise typer.Exit(2) from None
201
+
202
+ cfg = claw.load_config(str(config) if config else None)
203
+ # Their own factory, so the judge model, key and base URL come from the
204
+ # same config a `claw-eval run` would use rather than from a second copy.
205
+ judge = claw.cli._make_judge(cfg, SimpleNamespace(no_judge=no_judge, judge_model=None))
206
+
207
+ # Split BEFORE the limit, so --limit takes the first N of the chosen side
208
+ # rather than trimming the pool and then splitting a different set each
209
+ # time the pool changes.
210
+ tasks = select_tasks(root, tag=tag, pattern=task)
211
+ if split != "all":
212
+ development, reserved = split_tasks(tasks)
213
+ tasks = reserved if split == "holdout" else development
214
+ if limit:
215
+ tasks = tasks[:limit]
216
+ if not tasks:
217
+ err.print("no tasks matched")
218
+ raise typer.Exit(1)
219
+
220
+ out_dir = Path(trace_dir) if trace_dir else root / "traces" / f"otto-{architecture}-{time.strftime('%y%m%d-%H%M')}"
221
+ out_dir.mkdir(parents=True, exist_ok=True)
222
+ err.print(f"{len(tasks)} task(s) -> {out_dir}")
223
+
224
+ # What the loop is allowed to learn, decided once, here, rather than per
225
+ # task -- the discipline is a property of the whole measurement.
226
+ #
227
+ # --no-learning nothing read, nothing written. The baseline.
228
+ # --split holdout lessons read, none written back. Tests TRANSFER: the
229
+ # loop never tunes against these tasks, which is the
230
+ # difference one evolved system got 31.7 points wrong.
231
+ # otherwise read and write. This is where lessons come from.
232
+ learning = ExitStack()
233
+ if no_learning:
234
+ learning.enter_context(bind_bank(None))
235
+ # Routing adapts from observed outcomes too, so the baseline has to
236
+ # hold that still as well. A "no learning" arm that quietly reordered
237
+ # the model chain partway through would be measuring two things.
238
+ learning.enter_context(bind_log(None))
239
+ else:
240
+ if lesson_bank:
241
+ learning.enter_context(bind_bank(MemoryStore(lesson_bank)))
242
+ # Beside the bank, so a measurement's routing evidence travels
243
+ # with its lessons instead of leaking into the working install.
244
+ learning.enter_context(bind_log(Path(lesson_bank).with_suffix(".seats.db")))
245
+ if split == "holdout":
246
+ # Read what the development runs learned, write nothing back --
247
+ # lessons and routing evidence alike. That is what makes the
248
+ # held-out number answer "does this TRANSFER" rather than "did the
249
+ # loop find something that works on what it was tuned on".
250
+ learning.enter_context(read_only())
251
+ learning.enter_context(routing_read_only())
252
+ err.print(
253
+ "learning: " + ("off (baseline)" if no_learning else
254
+ "read-only (held out)" if split == "holdout" else "on")
255
+ )
256
+
257
+ outcomes = []
258
+ try:
259
+ _run_tasks(claw, tasks, outcomes, out_dir=out_dir, cfg=cfg, judge=judge,
260
+ architecture=architecture, port_offset=port_offset,
261
+ max_seconds=max_seconds, trials=trials)
262
+ finally:
263
+ learning.close()
264
+
265
+ report = _build_report(
266
+ outcomes, claw, architecture=architecture, tag=tag, split=split,
267
+ trials=trials, max_seconds=max_seconds, out_dir=out_dir, cfg=cfg,
268
+ judge=judge, no_learning=no_learning,
269
+ )
270
+ (out_dir / "otto_summary.json").write_text(json.dumps(report, indent=2))
271
+
272
+ if as_json:
273
+ out.print_json(data=report)
274
+ return
275
+ _print_summary(report, out_dir, split)
276
+
277
+
278
+ def _run_tasks(claw, tasks, outcomes, *, out_dir, cfg, judge, architecture,
279
+ port_offset, max_seconds, trials) -> None:
280
+ for i, task_yaml in enumerate(tasks, 1):
281
+ name = task_yaml.parent.name
282
+ err.print(f"[{i}/{len(tasks)}] {name}")
283
+ missing = missing_service_keys(claw.TaskDefinition.from_yaml(task_yaml))
284
+ if missing:
285
+ err.print(
286
+ f" [warn]{', '.join(missing)} not set[/] -- this task's service "
287
+ "reaches the real internet and will return nothing, so the score "
288
+ "below measures the environment, not the agent"
289
+ )
290
+ try:
291
+ outcome = run_task_file(
292
+ claw, task_yaml,
293
+ trace_dir=out_dir, cfg=cfg, judge=judge,
294
+ architecture=architecture, port_offset=port_offset,
295
+ max_seconds=max_seconds, trials=trials,
296
+ )
297
+ except Exception as exc:
298
+ # One task's container or service failing is not a reason to lose
299
+ # the other 299 -- report it and carry on, the way their batch does.
300
+ err.print(f" [bad]harness error[/] {type(exc).__name__}: {exc}")
301
+ continue
302
+ outcomes.append(outcome)
303
+ flag = "[ok]pass[/]" if outcome.passed else "[warn]fail[/]"
304
+ for item in outcome.checklist:
305
+ err.print(f" [{item.get('status', '?')}] {item.get('text', '')[:90]}")
306
+ detail = f" ({outcome.error})" if outcome.error else ""
307
+ spread = (
308
+ " trials=" + "/".join(f"{t:.2f}" for t in outcome.trials)
309
+ if len(outcome.trials) > 1 else ""
310
+ )
311
+ out.print(
312
+ f" {flag} score={outcome.task_score:.2f} "
313
+ f"completion={outcome.completion:.2f} service-tools={outcome.tool_calls} "
314
+ f"actions={outcome.agent_actions} calls={outcome.model_calls} "
315
+ f"{outcome.wall_time_s:.0f}s{spread}{detail}"
316
+ )
317
+
318
+
319
+ def _build_report(outcomes, claw, *, architecture, tag, split, trials,
320
+ max_seconds, out_dir, cfg, judge, no_learning) -> dict:
321
+ return {
322
+ "architecture": architecture,
323
+ "learning": "off" if no_learning else "read-only" if split == "holdout" else "on",
324
+ "tag": tag,
325
+ "max_seconds": max_seconds,
326
+ "split": split,
327
+ "trials": trials,
328
+ "trace_dir": str(out_dir),
329
+ # What the scores mean. Two reports whose fingerprints differ are not
330
+ # comparable, however similar the numbers look.
331
+ "grading": grading_fingerprint(claw, cfg, judge),
332
+ "summary": _summary(outcomes, claw),
333
+ "tasks": [
334
+ {
335
+ "task_id": o.task_id, "task_score": o.task_score, "passed": o.passed,
336
+ "completion": o.completion, "robustness": o.robustness,
337
+ "communication": o.communication, "safety": o.safety,
338
+ "tool_calls": o.tool_calls, "agent_actions": o.agent_actions,
339
+ "checklist": o.checklist, "model_calls": o.model_calls,
340
+ "trials": o.trials,
341
+ "wall_time_s": o.wall_time_s, "error": o.error,
342
+ }
343
+ for o in outcomes
344
+ ],
345
+ }
346
+
347
+
348
+ def _print_summary(report: dict, out_dir: Path, split: str) -> None:
349
+ s = report["summary"]
350
+
351
+ # Said BEFORE the numbers, not after. A caveat printed underneath a mean
352
+ # score is read after the score has already been believed, and the whole
353
+ # failure this guards against is a single run being quoted as a result.
354
+ if s.get("tasks") and not s.get("is_evidence", True):
355
+ trials = s.get("trials", 1)
356
+ err.print(
357
+ f"\n[bad]NOT EVIDENCE: {trials} trial(s) per task.[/] "
358
+ f"[warn]Identical code and configuration have scored "
359
+ f"{OBSERVED_SINGLE_RUN_SPREAD:.2f} apart on a single task here, so "
360
+ f"any before/after difference below roughly 0.3 in what follows is "
361
+ f"noise. Do not quote these numbers as a result.[/]\n"
362
+ f"[muted]Re-run with --trials {MIN_TRIALS_FOR_EVIDENCE} and read "
363
+ f"pass^k, not the mean.[/]"
364
+ )
365
+
366
+ out.print(
367
+ f"\n{s.get('passed', 0)}/{s.get('tasks', 0)} passed "
368
+ f"({s.get('pass_rate', 0):.1%}) "
369
+ f"{'mean score' if s.get('is_evidence', True) else 'score (1 run)'} "
370
+ f"{s.get('mean_task_score', 0):.3f} "
371
+ f"completion {s.get('mean_completion', 0):.3f} "
372
+ f"{s.get('errors', 0)} harness/agent error(s)"
373
+ )
374
+ out.print(
375
+ f"cost: {s.get('mean_model_calls', 0):.1f} model calls and "
376
+ f"{s.get('mean_wall_time_s', 0):.0f}s per task"
377
+ )
378
+ if "mean_pass_hat_k" in s:
379
+ k = s["trials"]
380
+ out.print(
381
+ f"reliability over {k} trials: pass^{k} {s['mean_pass_hat_k']:.3f} "
382
+ f"pass@{k} {s['mean_pass_at_k']:.3f} "
383
+ f"mean spread {s['score_spread']:.3f}"
384
+ )
385
+ kinds = (s.get("failures") or {}).get("failures_by_kind") or {}
386
+ if kinds:
387
+ total_failed = s.get("tasks", 0) - s.get("passed", 0)
388
+ out.print(
389
+ "failures by kind: "
390
+ + " ".join(f"{kind} {count}" for kind, count in kinds.items())
391
+ + f" (of {total_failed} failed)"
392
+ )
393
+ cost = s.get("tool_cost") or {}
394
+ if cost.get("calls"):
395
+ # Per SEAT is the half that answers the question: Otto routes across
396
+ # four vendors and several capability tiers, and model strength is
397
+ # what decides whether an agent manages a seventeen-tool menu at all.
398
+ busiest = " ".join(f"{tool} {n}" for tool, n in
399
+ list(cost["by_tool"].items())[:5])
400
+ out.print(
401
+ f"tools: {cost['calls']} calls, {cost['failures']} failed | "
402
+ f"busiest: {busiest}"
403
+ )
404
+ out.print(
405
+ "per seat: "
406
+ + " ".join(f"{seat} {n}" for seat, n in cost["by_seat"].items())
407
+ )
408
+
409
+ grading = report["grading"]
410
+ out.print(
411
+ f"grading: {grading['otto_grading_path']} / claw {grading['claw_eval_revision'] or '?'} "
412
+ f"/ judge {grading['judge']} / split {split}"
413
+ )
414
+ out.print(f"report: {out_dir / 'otto_summary.json'}")
@@ -0,0 +1,106 @@
1
+ """`otto eval-compaction` -- what each compaction policy actually loses.
2
+
3
+ The instrument Otto did not have. `otto eval-memory` scores whether raw
4
+ evidence stays RETRIEVABLE, which the chunk store nearly guarantees; this
5
+ scores whether the constraint is still in what the prompt reads. Those are
6
+ different questions and only the second one distinguishes compaction policies.
7
+
8
+ Offline and free by default: the summariser is a deterministic stand-in, so a
9
+ policy comparison measures the POLICY rather than whichever model happened to
10
+ summarise that day.
11
+ """
12
+ from __future__ import annotations
13
+
14
+ import json
15
+ import tempfile
16
+ from pathlib import Path
17
+ from typing import Optional
18
+
19
+ import typer
20
+ from rich import box
21
+ from rich.table import Table
22
+ from typing_extensions import Annotated
23
+
24
+ from agent.cli.ui import err, out
25
+ from agent.eval.compaction_bench import POLICIES, run_matrix
26
+
27
+
28
+ def eval_compaction_cmd(
29
+ turns: Annotated[
30
+ int, typer.Option(help="How long a conversation to replay. Longer means "
31
+ "more compaction rounds, which is where policies diverge."),
32
+ ] = 120,
33
+ policy: Annotated[
34
+ Optional[str],
35
+ typer.Option(help="Run one policy instead of the whole matrix."),
36
+ ] = None,
37
+ live: Annotated[
38
+ bool,
39
+ typer.Option("--live", help="Use the real Task.SUMMARIZE model instead of the "
40
+ "deterministic stand-in. Answers a question about "
41
+ "the MODEL, not about the policy, and costs calls."),
42
+ ] = False,
43
+ out_dir: Annotated[
44
+ Optional[Path], typer.Option(help="Where to put the stores (default: a temp dir)."),
45
+ ] = None,
46
+ as_json: Annotated[bool, typer.Option("--json", help="Print the report as JSON.")] = False,
47
+ ) -> None:
48
+ chosen = POLICIES
49
+ if policy:
50
+ if policy not in POLICIES:
51
+ err.print(f"no policy {policy!r} -- have {', '.join(POLICIES)}")
52
+ raise typer.Exit(2)
53
+ chosen = {policy: POLICIES[policy]}
54
+
55
+ summarize = None
56
+ if live:
57
+ from agent.memory.wiring import summarize_for_memory
58
+
59
+ summarize = summarize_for_memory
60
+ err.print("[warn]--live[/] spends model calls, one per compaction round per policy")
61
+
62
+ root = Path(out_dir) if out_dir else Path(tempfile.mkdtemp(prefix="otto-compaction-"))
63
+ root.mkdir(parents=True, exist_ok=True)
64
+ err.print(f"{len(chosen)} policy/policies over {turns} turns -> {root}")
65
+
66
+ results = run_matrix(root, turns=turns, policies=chosen, summarize=summarize)
67
+
68
+ if as_json:
69
+ out.print_json(data={
70
+ "turns": turns, "live": live,
71
+ "policies": [
72
+ {
73
+ "name": r.name, "survives": r.survives, "recoverable": r.recoverable,
74
+ "total": r.total, "survival_rate": round(r.survival_rate, 3),
75
+ "recovery_rate": round(r.recovery_rate, 3),
76
+ "view_chars": r.view_chars, "compactions": r.compactions,
77
+ "missing": r.missing,
78
+ }
79
+ for r in results
80
+ ],
81
+ })
82
+ return
83
+
84
+ table = Table(box=box.SIMPLE, pad_edge=False)
85
+ table.add_column("policy")
86
+ table.add_column("in the view", justify="right")
87
+ table.add_column("recoverable", justify="right")
88
+ table.add_column("view chars", justify="right")
89
+ table.add_column("rounds", justify="right")
90
+ for r in results:
91
+ # The number that matters is the first: what the model will actually
92
+ # read without being told to go looking.
93
+ style = "ok" if r.survival_rate >= 0.75 else "warn" if r.survival_rate else "bad"
94
+ table.add_row(
95
+ r.name,
96
+ f"[{style}]{r.survives}/{r.total}[/]",
97
+ f"{r.recoverable}/{r.total}",
98
+ f"{r.view_chars:,}",
99
+ str(r.compactions),
100
+ )
101
+ out.print(table)
102
+ out.print(
103
+ "'in the view' is what a prompt is built from; 'recoverable' is what "
104
+ "recall_memory could still find. The gap between them is text the "
105
+ "model will not see unless it thinks to go looking."
106
+ )