acco 1.15.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (317) hide show
  1. acco-1.15.0/BENCHMARKING.md +880 -0
  2. acco-1.15.0/CHANGELOG.md +1995 -0
  3. acco-1.15.0/INTEGRATIONS.md +318 -0
  4. acco-1.15.0/LICENSE +21 -0
  5. acco-1.15.0/MANIFEST.in +3 -0
  6. acco-1.15.0/PKG-INFO +1246 -0
  7. acco-1.15.0/README.md +1210 -0
  8. acco-1.15.0/VALIDATION.md +859 -0
  9. acco-1.15.0/benchmarks/agent-runs.example.json +49 -0
  10. acco-1.15.0/benchmarks/claude-sonnet-5-rates-2026-09-19.json +9 -0
  11. acco-1.15.0/benchmarks/cli-output-compression-ratchet-v1.frozen.json +663 -0
  12. acco-1.15.0/benchmarks/cli-output-real-v1/manifest.json +367 -0
  13. acco-1.15.0/benchmarks/cli-output-real-v2/freeze.json +21 -0
  14. acco-1.15.0/benchmarks/cli-output-real-v3/manifest.json +415 -0
  15. acco-1.15.0/benchmarks/context-quality.json +27 -0
  16. acco-1.15.0/benchmarks/e2e-suite.example.json +68 -0
  17. acco-1.15.0/benchmarks/e2e-swebench-24.frozen.json +675 -0
  18. acco-1.15.0/benchmarks/holdout-external-10.json +910 -0
  19. acco-1.15.0/benchmarks/holdout-external-10.result.json +665 -0
  20. acco-1.15.0/benchmarks/holdout-external-11.json +1134 -0
  21. acco-1.15.0/benchmarks/holdout-external-11.result.json +827 -0
  22. acco-1.15.0/benchmarks/holdout-external-12-candidate-rejected.json +484 -0
  23. acco-1.15.0/benchmarks/holdout-external-12-candidate-rejected.result.json +362 -0
  24. acco-1.15.0/benchmarks/holdout-external-12-pilot-rejected.json +484 -0
  25. acco-1.15.0/benchmarks/holdout-external-12-pilot-rejected.result.json +362 -0
  26. acco-1.15.0/benchmarks/holdout-external-12.json +1358 -0
  27. acco-1.15.0/benchmarks/holdout-external-12.result.json +989 -0
  28. acco-1.15.0/benchmarks/holdout-external-2.json +510 -0
  29. acco-1.15.0/benchmarks/holdout-external-2.result.json +431 -0
  30. acco-1.15.0/benchmarks/holdout-external-3.json +382 -0
  31. acco-1.15.0/benchmarks/holdout-external-3.result.json +320 -0
  32. acco-1.15.0/benchmarks/holdout-external-4.json +488 -0
  33. acco-1.15.0/benchmarks/holdout-external-4.result.json +422 -0
  34. acco-1.15.0/benchmarks/holdout-external-5.json +368 -0
  35. acco-1.15.0/benchmarks/holdout-external-5.result.json +320 -0
  36. acco-1.15.0/benchmarks/holdout-external-6.json +470 -0
  37. acco-1.15.0/benchmarks/holdout-external-6.result.json +328 -0
  38. acco-1.15.0/benchmarks/holdout-external-7.json +910 -0
  39. acco-1.15.0/benchmarks/holdout-external-7.result.json +608 -0
  40. acco-1.15.0/benchmarks/holdout-external-8.json +910 -0
  41. acco-1.15.0/benchmarks/holdout-external-8.result.json +608 -0
  42. acco-1.15.0/benchmarks/holdout-external-9.json +910 -0
  43. acco-1.15.0/benchmarks/holdout-external-9.result.json +665 -0
  44. acco-1.15.0/benchmarks/holdout-external.floor.json +45 -0
  45. acco-1.15.0/benchmarks/holdout-external.json +63 -0
  46. acco-1.15.0/benchmarks/holdout-external.result.json +80 -0
  47. acco-1.15.0/benchmarks/holdout.example.json +44 -0
  48. acco-1.15.0/benchmarks/knowledge-efficiency-swebench-24.frozen.json +716 -0
  49. acco-1.15.0/benchmarks/output-quality-session-v17.frozen.json +109 -0
  50. acco-1.15.0/benchmarks/output-quality.example.json +25 -0
  51. acco-1.15.0/benchmarks/semantic-holdout-13.frozen.json +496 -0
  52. acco-1.15.0/benchmarks/semantic-holdout-13.query-freeze.json +262 -0
  53. acco-1.15.0/benchmarks/semantic-holdout-14.frozen.json +473 -0
  54. acco-1.15.0/benchmarks/semantic-holdout-14.query-freeze.json +253 -0
  55. acco-1.15.0/benchmarks/semantic-holdout-15.query-freeze.json +253 -0
  56. acco-1.15.0/benchmarks/session-efficiency-swebench-24.frozen.json +711 -0
  57. acco-1.15.0/integrations/claude-code.mcp.json +8 -0
  58. acco-1.15.0/integrations/codex.config.toml +3 -0
  59. acco-1.15.0/integrations/cursor.mcp.json +8 -0
  60. acco-1.15.0/integrations/github-actions.yml +18 -0
  61. acco-1.15.0/pyproject.toml +70 -0
  62. acco-1.15.0/setup.cfg +4 -0
  63. acco-1.15.0/src/acco/__init__.py +3 -0
  64. acco-1.15.0/src/acco/agent_eval.py +336 -0
  65. acco-1.15.0/src/acco/audit.py +242 -0
  66. acco-1.15.0/src/acco/benchmark.py +354 -0
  67. acco-1.15.0/src/acco/blind_grader.py +469 -0
  68. acco-1.15.0/src/acco/browser_context.py +197 -0
  69. acco-1.15.0/src/acco/budget.py +72 -0
  70. acco-1.15.0/src/acco/cache_economics.py +145 -0
  71. acco-1.15.0/src/acco/claude_docker.py +181 -0
  72. acco-1.15.0/src/acco/claude_grader.py +114 -0
  73. acco-1.15.0/src/acco/claude_plugin.py +203 -0
  74. acco-1.15.0/src/acco/cli.py +721 -0
  75. acco-1.15.0/src/acco/client_capabilities.py +259 -0
  76. acco-1.15.0/src/acco/closure.py +181 -0
  77. acco-1.15.0/src/acco/command_handlers/__init__.py +6 -0
  78. acco-1.15.0/src/acco/command_handlers/context.py +322 -0
  79. acco-1.15.0/src/acco/command_handlers/efficiency.py +371 -0
  80. acco-1.15.0/src/acco/command_handlers/evaluation.py +238 -0
  81. acco-1.15.0/src/acco/command_handlers/experiment.py +412 -0
  82. acco-1.15.0/src/acco/command_handlers/host.py +290 -0
  83. acco-1.15.0/src/acco/command_handlers/ingress.py +51 -0
  84. acco-1.15.0/src/acco/command_handlers/model_routing.py +142 -0
  85. acco-1.15.0/src/acco/command_handlers/optimization.py +272 -0
  86. acco-1.15.0/src/acco/command_handlers/output.py +388 -0
  87. acco-1.15.0/src/acco/command_handlers/patch.py +81 -0
  88. acco-1.15.0/src/acco/command_handlers/pricing.py +82 -0
  89. acco-1.15.0/src/acco/command_registry.py +183 -0
  90. acco-1.15.0/src/acco/commands.py +84 -0
  91. acco-1.15.0/src/acco/config.py +24 -0
  92. acco-1.15.0/src/acco/context_browser.py +93 -0
  93. acco-1.15.0/src/acco/context_router.py +332 -0
  94. acco-1.15.0/src/acco/cost_report.py +462 -0
  95. acco-1.15.0/src/acco/delta_context.py +288 -0
  96. acco-1.15.0/src/acco/efficiency/__init__.py +20 -0
  97. acco-1.15.0/src/acco/efficiency/advisor.py +438 -0
  98. acco-1.15.0/src/acco/efficiency/dashboard.py +113 -0
  99. acco-1.15.0/src/acco/efficiency/report.py +103 -0
  100. acco-1.15.0/src/acco/efficiency/service.py +534 -0
  101. acco-1.15.0/src/acco/efficiency/store.py +198 -0
  102. acco-1.15.0/src/acco/entry.py +40 -0
  103. acco-1.15.0/src/acco/estimate.py +201 -0
  104. acco-1.15.0/src/acco/evaluate.py +321 -0
  105. acco-1.15.0/src/acco/evidence_pipeline.py +164 -0
  106. acco-1.15.0/src/acco/experiment.py +1057 -0
  107. acco-1.15.0/src/acco/fastpath.py +210 -0
  108. acco-1.15.0/src/acco/feedback.py +47 -0
  109. acco-1.15.0/src/acco/filter_output.py +43 -0
  110. acco-1.15.0/src/acco/generation_policy.py +229 -0
  111. acco-1.15.0/src/acco/guard.py +398 -0
  112. acco-1.15.0/src/acco/hook.py +201 -0
  113. acco-1.15.0/src/acco/hook_runtime.py +669 -0
  114. acco-1.15.0/src/acco/host_configs.py +542 -0
  115. acco-1.15.0/src/acco/host_validate.py +178 -0
  116. acco-1.15.0/src/acco/ignore.py +81 -0
  117. acco-1.15.0/src/acco/images.py +130 -0
  118. acco-1.15.0/src/acco/impact.py +114 -0
  119. acco-1.15.0/src/acco/ingress.py +279 -0
  120. acco-1.15.0/src/acco/install.py +124 -0
  121. acco-1.15.0/src/acco/integration_setup.py +759 -0
  122. acco-1.15.0/src/acco/knowledge.py +733 -0
  123. acco-1.15.0/src/acco/knowledge_holdout.py +409 -0
  124. acco-1.15.0/src/acco/knowledge_holdout_docker.py +208 -0
  125. acco-1.15.0/src/acco/knowledge_holdout_pipeline.py +109 -0
  126. acco-1.15.0/src/acco/lexical.py +199 -0
  127. acco-1.15.0/src/acco/mapstat.py +33 -0
  128. acco-1.15.0/src/acco/mcp.py +154 -0
  129. acco-1.15.0/src/acco/mcp_server/__init__.py +19 -0
  130. acco-1.15.0/src/acco/mcp_server/contracts.py +121 -0
  131. acco-1.15.0/src/acco/mcp_server/protocol.py +196 -0
  132. acco-1.15.0/src/acco/mcp_server/services.py +13 -0
  133. acco-1.15.0/src/acco/mcp_server/tool_surface.py +139 -0
  134. acco-1.15.0/src/acco/mcp_server/tools.py +830 -0
  135. acco-1.15.0/src/acco/mcp_server/transport.py +41 -0
  136. acco-1.15.0/src/acco/model_routing.py +1118 -0
  137. acco-1.15.0/src/acco/optimizer.py +330 -0
  138. acco-1.15.0/src/acco/output/__init__.py +41 -0
  139. acco-1.15.0/src/acco/output/contracts.py +55 -0
  140. acco-1.15.0/src/acco/output/pipeline.py +134 -0
  141. acco-1.15.0/src/acco/output/processors.py +677 -0
  142. acco-1.15.0/src/acco/output/registry.py +31 -0
  143. acco-1.15.0/src/acco/output/specialized_processors.py +250 -0
  144. acco-1.15.0/src/acco/output/text.py +146 -0
  145. acco-1.15.0/src/acco/output_benchmark.py +89 -0
  146. acco-1.15.0/src/acco/output_budget.py +351 -0
  147. acco-1.15.0/src/acco/output_effectiveness.py +836 -0
  148. acco-1.15.0/src/acco/output_processors.py +47 -0
  149. acco-1.15.0/src/acco/output_quality.py +237 -0
  150. acco-1.15.0/src/acco/output_saver.py +355 -0
  151. acco-1.15.0/src/acco/output_store.py +37 -0
  152. acco-1.15.0/src/acco/output_telemetry.py +622 -0
  153. acco-1.15.0/src/acco/pack.py +431 -0
  154. acco-1.15.0/src/acco/pack_cli.py +144 -0
  155. acco-1.15.0/src/acco/packing/__init__.py +23 -0
  156. acco-1.15.0/src/acco/packing/contracts.py +78 -0
  157. acco-1.15.0/src/acco/packing/file_scoring.py +435 -0
  158. acco-1.15.0/src/acco/packing/graph_rerank.py +517 -0
  159. acco-1.15.0/src/acco/packing/observability.py +77 -0
  160. acco-1.15.0/src/acco/packing/query_analysis.py +227 -0
  161. acco-1.15.0/src/acco/packing/ranking.py +154 -0
  162. acco-1.15.0/src/acco/packing/ranking_defaults.py +11 -0
  163. acco-1.15.0/src/acco/packing/ranking_stages.py +116 -0
  164. acco-1.15.0/src/acco/packing/render.py +71 -0
  165. acco-1.15.0/src/acco/packing/symbol_scoring.py +521 -0
  166. acco-1.15.0/src/acco/packing/symbol_windows.py +251 -0
  167. acco-1.15.0/src/acco/packing/symbols.py +41 -0
  168. acco-1.15.0/src/acco/paired_conditions.py +18 -0
  169. acco-1.15.0/src/acco/patch_context.py +332 -0
  170. acco-1.15.0/src/acco/policy.py +222 -0
  171. acco-1.15.0/src/acco/prefix_cache.py +139 -0
  172. acco-1.15.0/src/acco/pricing.py +224 -0
  173. acco-1.15.0/src/acco/pricing_registry.json +70 -0
  174. acco-1.15.0/src/acco/processor_mining.py +173 -0
  175. acco-1.15.0/src/acco/provider_proxy.py +266 -0
  176. acco-1.15.0/src/acco/provider_transform.py +268 -0
  177. acco-1.15.0/src/acco/prune.py +48 -0
  178. acco-1.15.0/src/acco/ranking_calibration.py +336 -0
  179. acco-1.15.0/src/acco/ranking_regression.py +555 -0
  180. acco-1.15.0/src/acco/recovery.py +234 -0
  181. acco-1.15.0/src/acco/repo_index.py +640 -0
  182. acco-1.15.0/src/acco/repository_service.py +348 -0
  183. acco-1.15.0/src/acco/retrieval_cache.py +272 -0
  184. acco-1.15.0/src/acco/retrieval_vnext.py +119 -0
  185. acco-1.15.0/src/acco/runtime_config.py +676 -0
  186. acco-1.15.0/src/acco/security.py +74 -0
  187. acco-1.15.0/src/acco/semantic_holdout.py +602 -0
  188. acco-1.15.0/src/acco/semantic_retrieval.py +1076 -0
  189. acco-1.15.0/src/acco/semantic_ts.py +381 -0
  190. acco-1.15.0/src/acco/serve.py +53 -0
  191. acco-1.15.0/src/acco/session_holdout.py +590 -0
  192. acco-1.15.0/src/acco/session_holdout_docker.py +388 -0
  193. acco-1.15.0/src/acco/session_holdout_pipeline.py +158 -0
  194. acco-1.15.0/src/acco/session_metrics.py +211 -0
  195. acco-1.15.0/src/acco/sessions.py +339 -0
  196. acco-1.15.0/src/acco/skeleton.py +698 -0
  197. acco-1.15.0/src/acco/snippet.py +95 -0
  198. acco-1.15.0/src/acco/state.py +147 -0
  199. acco-1.15.0/src/acco/swebench_docker.py +136 -0
  200. acco-1.15.0/src/acco/syntax.py +914 -0
  201. acco-1.15.0/src/acco/tool_proxy.py +534 -0
  202. acco-1.15.0/src/acco/tool_schema.py +176 -0
  203. acco-1.15.0/src/acco/unified_audit.py +226 -0
  204. acco-1.15.0/src/acco/working_set.py +45 -0
  205. acco-1.15.0/src/acco.egg-info/PKG-INFO +1246 -0
  206. acco-1.15.0/src/acco.egg-info/SOURCES.txt +315 -0
  207. acco-1.15.0/src/acco.egg-info/dependency_links.txt +1 -0
  208. acco-1.15.0/src/acco.egg-info/entry_points.txt +3 -0
  209. acco-1.15.0/src/acco.egg-info/requires.txt +33 -0
  210. acco-1.15.0/src/acco.egg-info/top_level.txt +1 -0
  211. acco-1.15.0/tests/test_adaptive_budget.py +93 -0
  212. acco-1.15.0/tests/test_architecture_boundaries.py +134 -0
  213. acco-1.15.0/tests/test_audit.py +171 -0
  214. acco-1.15.0/tests/test_benchmark_evidence.py +71 -0
  215. acco-1.15.0/tests/test_blind_grader.py +186 -0
  216. acco-1.15.0/tests/test_budget_map.py +106 -0
  217. acco-1.15.0/tests/test_cache_economics.py +86 -0
  218. acco-1.15.0/tests/test_callable_identity_v5.py +264 -0
  219. acco-1.15.0/tests/test_callable_symbol_ranking_v3.py +166 -0
  220. acco-1.15.0/tests/test_check_holdout.py +67 -0
  221. acco-1.15.0/tests/test_claude_docker.py +233 -0
  222. acco-1.15.0/tests/test_claude_grader.py +44 -0
  223. acco-1.15.0/tests/test_claude_plugin.py +74 -0
  224. acco-1.15.0/tests/test_cli.py +231 -0
  225. acco-1.15.0/tests/test_client_capabilities.py +60 -0
  226. acco-1.15.0/tests/test_context_browser_mcp.py +26 -0
  227. acco-1.15.0/tests/test_context_router.py +84 -0
  228. acco-1.15.0/tests/test_cost_advisor.py +206 -0
  229. acco-1.15.0/tests/test_cost_report.py +334 -0
  230. acco-1.15.0/tests/test_csharp14_extension_blocks.py +154 -0
  231. acco-1.15.0/tests/test_delta_context.py +94 -0
  232. acco-1.15.0/tests/test_documentation.py +301 -0
  233. acco-1.15.0/tests/test_efficiency_cli.py +103 -0
  234. acco-1.15.0/tests/test_entry.py +240 -0
  235. acco-1.15.0/tests/test_estimate.py +122 -0
  236. acco-1.15.0/tests/test_evaluate_holdout.py +135 -0
  237. acco-1.15.0/tests/test_evaluate_scoped_recall.py +66 -0
  238. acco-1.15.0/tests/test_evidence.py +210 -0
  239. acco-1.15.0/tests/test_evidence_pipeline.py +267 -0
  240. acco-1.15.0/tests/test_experiment.py +551 -0
  241. acco-1.15.0/tests/test_fastpath.py +103 -0
  242. acco-1.15.0/tests/test_filter.py +110 -0
  243. acco-1.15.0/tests/test_filter_dedupe.py +20 -0
  244. acco-1.15.0/tests/test_frozen_e2e_suite.py +27 -0
  245. acco-1.15.0/tests/test_fuzzy_symbol.py +71 -0
  246. acco-1.15.0/tests/test_generation_policy.py +249 -0
  247. acco-1.15.0/tests/test_guard.py +326 -0
  248. acco-1.15.0/tests/test_holdout11_failure_classes.py +250 -0
  249. acco-1.15.0/tests/test_hook.py +308 -0
  250. acco-1.15.0/tests/test_hook_runtime.py +701 -0
  251. acco-1.15.0/tests/test_images.py +90 -0
  252. acco-1.15.0/tests/test_index_robustness.py +83 -0
  253. acco-1.15.0/tests/test_ingress.py +112 -0
  254. acco-1.15.0/tests/test_integration_setup.py +578 -0
  255. acco-1.15.0/tests/test_knowledge.py +280 -0
  256. acco-1.15.0/tests/test_knowledge_holdout.py +336 -0
  257. acco-1.15.0/tests/test_large_output_example.py +99 -0
  258. acco-1.15.0/tests/test_lexical.py +69 -0
  259. acco-1.15.0/tests/test_mapstat.py +21 -0
  260. acco-1.15.0/tests/test_mcp.py +91 -0
  261. acco-1.15.0/tests/test_mcp_profiles.py +267 -0
  262. acco-1.15.0/tests/test_mcp_server_boundaries.py +143 -0
  263. acco-1.15.0/tests/test_model_routing.py +596 -0
  264. acco-1.15.0/tests/test_multi_host_integrations.py +452 -0
  265. acco-1.15.0/tests/test_multilang_syntax.py +246 -0
  266. acco-1.15.0/tests/test_optimization_platform.py +373 -0
  267. acco-1.15.0/tests/test_output_benchmark.py +67 -0
  268. acco-1.15.0/tests/test_output_budget.py +352 -0
  269. acco-1.15.0/tests/test_output_effectiveness.py +496 -0
  270. acco-1.15.0/tests/test_output_processor_expansion.py +353 -0
  271. acco-1.15.0/tests/test_output_processors.py +230 -0
  272. acco-1.15.0/tests/test_output_quality.py +209 -0
  273. acco-1.15.0/tests/test_output_saver.py +131 -0
  274. acco-1.15.0/tests/test_output_saver_mcp.py +45 -0
  275. acco-1.15.0/tests/test_output_telemetry.py +418 -0
  276. acco-1.15.0/tests/test_overload_resolution_v4.py +203 -0
  277. acco-1.15.0/tests/test_pack.py +700 -0
  278. acco-1.15.0/tests/test_pack_pipeline_boundaries.py +107 -0
  279. acco-1.15.0/tests/test_policy.py +61 -0
  280. acco-1.15.0/tests/test_pricing_registry.py +123 -0
  281. acco-1.15.0/tests/test_processor_mining.py +62 -0
  282. acco-1.15.0/tests/test_prune.py +13 -0
  283. acco-1.15.0/tests/test_ranking_artifact_collector.py +250 -0
  284. acco-1.15.0/tests/test_ranking_calibration.py +239 -0
  285. acco-1.15.0/tests/test_ranking_calibration_workflow.py +85 -0
  286. acco-1.15.0/tests/test_ranking_ci_workflow.py +46 -0
  287. acco-1.15.0/tests/test_ranking_observability.py +230 -0
  288. acco-1.15.0/tests/test_ranking_pipeline_boundaries.py +141 -0
  289. acco-1.15.0/tests/test_ranking_regression.py +275 -0
  290. acco-1.15.0/tests/test_ranking_stage_registry.py +215 -0
  291. acco-1.15.0/tests/test_ranking_stages.py +203 -0
  292. acco-1.15.0/tests/test_release_regressions.py +251 -0
  293. acco-1.15.0/tests/test_repo_index.py +174 -0
  294. acco-1.15.0/tests/test_repository_service.py +157 -0
  295. acco-1.15.0/tests/test_retrieval_cache.py +128 -0
  296. acco-1.15.0/tests/test_retrieval_vnext.py +59 -0
  297. acco-1.15.0/tests/test_semantic_holdout.py +329 -0
  298. acco-1.15.0/tests/test_semantic_ref_closure.py +167 -0
  299. acco-1.15.0/tests/test_semantic_retrieval.py +829 -0
  300. acco-1.15.0/tests/test_semantic_ts.py +140 -0
  301. acco-1.15.0/tests/test_session_efficiency.py +261 -0
  302. acco-1.15.0/tests/test_session_holdout.py +313 -0
  303. acco-1.15.0/tests/test_session_holdout_docker.py +172 -0
  304. acco-1.15.0/tests/test_session_metrics.py +142 -0
  305. acco-1.15.0/tests/test_sessions.py +266 -0
  306. acco-1.15.0/tests/test_skeleton.py +305 -0
  307. acco-1.15.0/tests/test_snippet.py +36 -0
  308. acco-1.15.0/tests/test_structural_authority.py +100 -0
  309. acco-1.15.0/tests/test_structural_symbol_identity.py +128 -0
  310. acco-1.15.0/tests/test_swebench_grading.py +230 -0
  311. acco-1.15.0/tests/test_symbol_pipeline_boundaries.py +110 -0
  312. acco-1.15.0/tests/test_tool_proxy.py +220 -0
  313. acco-1.15.0/tests/test_unified_audit.py +38 -0
  314. acco-1.15.0/tests/test_v09.py +225 -0
  315. acco-1.15.0/tests/test_v1.py +477 -0
  316. acco-1.15.0/tests/test_v12_ports.py +96 -0
  317. acco-1.15.0/tests/test_version.py +12 -0
@@ -0,0 +1,880 @@
1
+ > **Branding note:** frozen benchmark artifacts created before the ACCO rename retain their original Token Saver identifiers and hashes. The documentation uses the current ACCO product name; frozen evidence files are not rewritten.
2
+
3
+ # Verify integration, then benchmark successful work
4
+
5
+ ## Deterministic context-quality benchmark
6
+
7
+ Before paid paired-agent trials, run the included 25-task ground-truth selector
8
+ benchmark. It measures relevant-file recall, relevant-symbol recall, and context
9
+ reduction; selection metrics alone do not prove agent success.
10
+
11
+ ```bash
12
+ acco evaluate benchmarks/context-quality.json --path . --max-tokens 6000
13
+ ```
14
+
15
+ Each item also reports `symbol_recall_in_expected_files`, which counts a
16
+ symbol only when it was selected from one of the task's expected files. Bare
17
+ `symbol_recall` can be satisfied by a same-named symbol in an unrelated file;
18
+ prefer the scoped figure (or `qualified_symbols` / `symbol_identities`) when
19
+ names are common.
20
+
21
+ Add project-specific tasks using `query`, `files`, and `symbols`. Keep the
22
+ manifest under version control so ranking changes can be compared reproducibly.
23
+
24
+ ## Multi-repository holdout benchmark
25
+
26
+ ### Query-construction protocol
27
+
28
+ A frozen hash prevents post-hoc edits to the **query text**, expected files, and
29
+ expected symbols. It does not make an easy query difficult. New holdouts must
30
+ therefore declare which retrieval behavior they are measuring before the first
31
+ ACCO run.
32
+
33
+ For a **semantic natural-language holdout**, construct each query from the
34
+ upstream issue, bug report, user request, or behavior description **before
35
+ looking up the answer identity**. The query must not contain:
36
+
37
+ - the target symbol/member name;
38
+ - the target's containing class/type/module name when that name directly
39
+ identifies the answer;
40
+ - the target file basename/path;
41
+ - the exact qualified symbol identity used as ground truth;
42
+ - a distinctive implementation literal copied from the answer solely to make
43
+ retrieval easier.
44
+
45
+ Example:
46
+
47
+ ```text
48
+ acceptable: "reject malformed email addresses before schema validation"
49
+ not semantic: "where is emailRegex in regexes.ts"
50
+ ```
51
+
52
+ If the real user task already contains an identifier (for example, a compiler
53
+ error names `refreshSession`), keep it: removing genuine task evidence would
54
+ make the benchmark artificial. Classify that task/suite as **identifier-bearing**
55
+ rather than semantic-natural-language and report it separately.
56
+
57
+ Do not present identifier-bearing recall as a continuation of a
58
+ natural-language difficulty trend. The two answer different questions:
59
+
60
+ - semantic suites test whether ACCO can discover the identity from a
61
+ behavior/problem description;
62
+ - identifier-bearing suites test exact navigation, overload resolution,
63
+ scoping, and identity recovery once some answer vocabulary is already known.
64
+
65
+ For future headline semantic holdouts:
66
+
67
+ 1. pre-register the query source/rule and freeze the exact query text;
68
+ 2. keep target symbol, containing type, file path, and qualified identity out of
69
+ the query unless they genuinely appeared in the upstream task;
70
+ 3. record any unavoidable identifier-bearing tasks explicitly;
71
+ 4. run a trivial lexical baseline (for example grep/ctags/exact identifier
72
+ lookup where applicable) on the same frozen tasks;
73
+ 5. report ACCO recall alongside that baseline rather than quoting only an
74
+ absolute recall percentage;
75
+ 6. never rewrite queries after seeing retrieval misses.
76
+
77
+ The query source/rule belongs in the same committed evidence package as the
78
+ manifest. The manifest's query strings themselves are part of the ground-truth
79
+ freeze hash, so changing the wording after freeze changes benchmark identity.
80
+
81
+ The repository-local selector benchmark is useful for regressions, but because
82
+ ACCO is developed against this codebase it is not independent evidence of
83
+ generalization. For unseen evaluation, define ground truth before running the
84
+ tool and point one manifest at repositories that were excluded from ranking
85
+ work/tuning. `benchmarks/holdout.example.json` contains the full schema.
86
+
87
+ Repository paths are resolved relative to the manifest. Pin exact Git commits so
88
+ the corpus cannot move between runs. A publishable holdout also records a freeze
89
+ timestamp and a SHA-256 of the task/repository ground-truth definition:
90
+
91
+ ```json
92
+ {
93
+ "suite_version": 1,
94
+ "protocol": {
95
+ "ground_truth_frozen": true,
96
+ "development_excluded": true,
97
+ "frozen_at": "2026-09-17T12:00:00Z",
98
+ "ground_truth_sha256": "HASH_FROM_COMMAND_BELOW"
99
+ },
100
+ "repositories": {
101
+ "app-a": {"path": "../app-a", "revision": "ACTUAL_COMMIT_SHA"},
102
+ "app-b": {"path": "../app-b", "revision": "ACTUAL_COMMIT_SHA"}
103
+ },
104
+ "tasks": [
105
+ {
106
+ "id": "auth-refresh",
107
+ "repository": "app-a",
108
+ "query": "session refresh after logout",
109
+ "files": ["src/auth/session.ts"],
110
+ "symbols": ["refreshSession"]
111
+ }
112
+ ]
113
+ }
114
+ ```
115
+
116
+ After the tasks, expected evidence, and revision pins are final, calculate the
117
+ freeze hash without running retrieval:
118
+
119
+ ```bash
120
+ acco evaluate benchmarks/holdout.json --print-ground-truth-hash
121
+ ```
122
+
123
+ Put that value in `protocol.ground_truth_sha256`, commit the manifest, then run:
124
+
125
+ ```bash
126
+ acco evaluate benchmarks/holdout.json --require-holdout --max-tokens 6000
127
+ ```
128
+
129
+ `--require-holdout` rejects a manifest unless both protocol flags are true,
130
+ `frozen_at` is present, the SHA-256 still matches the frozen task definition,
131
+ and every pinned repository `HEAD` matches its declared revision. Filesystem
132
+ paths are excluded from the freeze hash so the same manifest can be replicated
133
+ on another machine without changing the benchmark identity. The output includes
134
+ aggregate metrics, per-repository summaries, and the verified hash.
135
+
136
+ A valid hash proves the evaluated definition did not change after it was frozen;
137
+ it does not by itself prove the labels were independently authored before tuning.
138
+ Preserve manifest history and the task-definition process as audit evidence.
139
+
140
+ ### Frozen semantic holdout #13
141
+
142
+ Version 1.10 includes a fresh **no-identifier-leakage semantic holdout** for
143
+ measuring whether hybrid chunk retrieval improves natural-language file
144
+ discovery beyond ACCO's lexical/structural pipeline.
145
+
146
+ The benchmark uses 24 behavior descriptions from public upstream issues across
147
+ six repositories that were not used in external holdouts #1-#12:
148
+
149
+ - Python: `Kludex/uvicorn`;
150
+ - Go: `spf13/afero`;
151
+ - Rust: `rust-lang/regex`;
152
+ - Java: `resilience4j/resilience4j`;
153
+ - JavaScript: `fastify/fastify`;
154
+ - C#: `dotnet/command-line-api`.
155
+
156
+ The audit sequence is intentionally stronger than a single final-manifest hash.
157
+ The exact 24 queries and repository revisions were committed **before target
158
+ files or fixes were inspected** in
159
+ `benchmarks/semantic-holdout-13.query-freeze.json`.
160
+
161
+ Query-only freeze:
162
+
163
+ - commit: `f3247c1d4c8e388de653aea1a5de4fa4624f82ce`;
164
+ - canonical SHA-256:
165
+ `8cdf871ea2fcbe59161a332f1f0ce0f93b087a6c3560acebadb5ee338c4169bf`.
166
+
167
+ Ground truth was then collected without changing those queries. A post-freeze
168
+ answer-identity audit excluded two tasks from the semantic headline rather than
169
+ rewriting them: one Uvicorn task contains exact target member names
170
+ (`restart` / `shutdown`), and one System.CommandLine task is too close to
171
+ the public `CustomParser` member identity. Both remain visible in the frozen
172
+ manifest with exclusion reasons. The headline cohort is therefore **22 of the
173
+ 24 frozen tasks**.
174
+
175
+ Final semantic ground-truth SHA-256:
176
+
177
+ `dc6ea6c3641db573b5b05473f2bc4ee13e0f05a4f093cb86f803e68c7b265d25`
178
+
179
+ Every eligible task is evaluated under the same file-count and token limits
180
+ against three arms:
181
+
182
+ 1. **ACCO lexical/structural** — the current validated pipeline with
183
+ semantic retrieval disabled;
184
+ 2. **ACCO hybrid semantic** — the same pipeline with persistent
185
+ chunk-level semantic retrieval enabled;
186
+ 3. **trivial lexical baseline** — distinct normalized query-term overlap only,
187
+ with no structural authority, fuzzy correction, dependency graph, feedback,
188
+ or semantic evidence.
189
+
190
+ The semantic arm is pinned to `all-MiniLM-L6-v2` revision
191
+ `bc57282bc374d33e0d6c4de27f12dc1c2a87f37a`. Model revision participates in
192
+ ACCO's semantic vector/query-cache identity. The evidence workflow
193
+ explicitly removes `hnswlib`, so the canonical first run uses exact cosine
194
+ over the persistent vectors rather than approximate nearest-neighbor search.
195
+
196
+ The first real evaluation ran once in GitHub Actions
197
+ **35537362040** and permanently burned this suite for tuning. The exact-cosine
198
+ result over the 22 eligible tasks was:
199
+
200
+ | Arm | File recall |
201
+ | --- | ---: |
202
+ | ACCO hybrid semantic | **50.00% (11/22)** |
203
+ | ACCO lexical/structural | **45.45% (10/22)** |
204
+ | Trivial lexical baseline | **40.91% (9/22)** |
205
+
206
+ Hybrid semantic retrieval recovered one task missed by ACCO's lexical
207
+ arm and regressed none. Mean estimated context reduction was effectively flat
208
+ (97.8374% vs 97.8373%).
209
+
210
+ The result is useful precisely because it is not inflated: the semantic
211
+ mechanism shows a real but small improvement, while several repositories still
212
+ have difficult behavior-only misses. Holdout #13 must not be used as the tuning
213
+ loop for those misses.
214
+
215
+ Fresh holdout #14 was subsequently frozen before evaluation with 24
216
+ issue-derived natural-language tasks across six repositories; four were
217
+ conservatively excluded after ground-truth review, leaving 20 eligible tasks.
218
+ Its first complete exact-cosine run was GitHub Actions **35615316639**:
219
+
220
+ | Arm | File recall |
221
+ | --- | ---: |
222
+ | ACCO hybrid semantic | **82.50%** |
223
+ | ACCO lexical/structural | **80.00%** |
224
+ | Trivial lexical baseline | **70.00%** |
225
+
226
+ The semantic arm improved aggregate file recall by 2.5 percentage points with
227
+ zero regressions. One two-file target was partially recovered, so the run
228
+ reported zero complete `semantic_recovered_tasks`. Mean estimated context
229
+ reduction remained essentially identical (98.9516% semantic vs 98.9515%
230
+ lexical). That first run burned #14.
231
+
232
+ Any rerun after tuning against #14 is development evidence only. In particular,
233
+ the later 87.5% semantic development result must not be reported as fresh
234
+ generalization evidence. The next independent cohort is semantic holdout #15: its queries and
235
+ pinned repository revisions are frozen, but ground truth and the first
236
+ evaluation are still pending.
237
+
238
+ These are retrieval benchmarks. File-recall improvement does not by itself
239
+ establish lower API cost, coding-task success, or cost per successful task.
240
+
241
+ ## Automated end-to-end cost-per-success experiment
242
+
243
+ For evidence that supports a public cost claim, use the executable experiment
244
+ harness rather than hand-assembling a few runs. It exports history-isolated snapshots at pinned revisions, randomizes
245
+ baseline/enabled order deterministically, runs multiple trials, applies Token
246
+ Saver only in the enabled arm, executes the task's verifier outside the agent,
247
+ captures the real Claude Code transcript, and checkpoints after every run so an
248
+ interrupted paid experiment can resume safely. The exported snapshot is
249
+ re-initialized as a one-commit Git repository, so an agent cannot recover the
250
+ historical gold fix from later commits or remote branches.
251
+
252
+ The publication gate deliberately requires **at least 20 distinct tasks and at
253
+ least 3 paired trials per task**. A 20-task suite therefore means 120 agent runs
254
+ (20 tasks × 3 trials × 2 conditions). Prefer 20–50 historical real-world bug
255
+ fixes/refactors across several repositories and task types rather than many near-
256
+ duplicates from one project.
257
+
258
+ Start from `benchmarks/e2e-suite.example.json`. Each task must pin a repository
259
+ revision, preserve the exact prompt (plus its SHA-256), and specify one or more
260
+ independent verifier commands. Do not use the agent's own "done" statement as
261
+ the success label.
262
+
263
+ A production-ready frozen suite is checked in as
264
+ `benchmarks/e2e-swebench-24.frozen.json`. It contains 24 historical SWE-bench
265
+ Verified issues across scikit-learn, pytest, Astropy, Pylint, Requests, Xarray,
266
+ and Seaborn, with three paired trials per task (**144 agent runs**). The public
267
+ repository URL, task revision, prompt, hidden regression patch, official
268
+ SWE-bench evaluation image, and canonical test command are frozen into the suite
269
+ hash; only the machine-local clone path is excluded.
270
+
271
+ Prepare its external repositories without checking them into this repository:
272
+
273
+ ```bash
274
+ python scripts/prepare_e2e_repos.py benchmarks/e2e-swebench-24.frozen.json
275
+ ```
276
+
277
+ For these tasks, the hidden regression patch is not present while the coding
278
+ agent runs. After the agent exits, its diff is captured, then a fresh official
279
+ SWE-bench Docker image applies the agent patch and hidden test patch and executes
280
+ the canonical test command inside the benchmark's prepared environment.
281
+
282
+ A run is graded per test, not by the exit code of the whole command, because
283
+ official images can carry pre-existing errors (for example broken fixtures)
284
+ that make it nonzero even for a correct fix. A run is **resolved** when every
285
+ `source.fail_to_pass` test passes and no test that passed on the unpatched
286
+ reference regressed. The reference is the same image with the hidden tests
287
+ applied and no agent patch; it is computed once per task and cached under
288
+ `<out>.artifacts/_reference/`. Agent edits to files the hidden test patch
289
+ modifies are removed before verification (the full patch stays recorded as
290
+ `agent.patch`; the verified one is `agent.graded.patch`), so an agent that also
291
+ edits a test file cannot make the hidden tests fail to apply.
292
+
293
+ Before any paid run, finalize the task definitions and experimental design, then
294
+ freeze them:
295
+
296
+ ```bash
297
+ acco experiment benchmarks/e2e-suite.json \
298
+ --print-task-definition-hash
299
+ ```
300
+
301
+ Copy that SHA-256 into `protocol.task_definition_sha256`, set `frozen_at`,
302
+ commit the suite, and inspect the randomized schedule without calling a model:
303
+
304
+ ```bash
305
+ acco experiment benchmarks/e2e-suite.json \
306
+ --out benchmark-runs.json \
307
+ --dry-run
308
+ ```
309
+
310
+ Run the experiment:
311
+
312
+ ```bash
313
+ acco experiment benchmarks/e2e-suite.json \
314
+ --out benchmark-runs.json
315
+ ```
316
+
317
+ The baseline arm exports `ACCO_DISABLED=1`, which makes any inherited
318
+ ACCO hook a true no-op. The enabled arm installs project-local hooks.
319
+ Both arms receive the same history-isolated source snapshot, exact prompt, model,
320
+ turn limit, and verifier. Hidden SWE-bench regression tests are applied only
321
+ after the agent process has ended.
322
+ To avoid double instrumentation, the runner refuses to start when it detects a
323
+ user-level `acco hook`; use a clean host configuration for publishable
324
+ runs rather than bypassing that guard.
325
+
326
+ For development/smoke experiments with fewer tasks or an unfrozen suite, pass
327
+ `--allow-development`. Those runs are intentionally rejected by the publication
328
+ gate.
329
+
330
+ After the runs finish, price the exact recorded model usage and require the broad
331
+ protocol:
332
+
333
+ ```bash
334
+ acco benchmark benchmark-runs.json \
335
+ --rates rates.json \
336
+ --require-publishable
337
+ ```
338
+
339
+ The result includes success rates, failed-run cost, cost per success, per-task
340
+ improvements/regressions, and deterministic **95% task-cluster bootstrap
341
+ intervals**. Repeated trials are resampled as one task cluster; three trials of
342
+ one task are not treated as three independent tasks. The report separately marks
343
+ whether the frozen protocol is valid, whether a quality-parity cost claim is
344
+ allowed, and whether the 95% interval for cost-per-success reduction is entirely
345
+ above zero.
346
+
347
+ Raw transcripts and verifier outputs can contain source code or secrets. Keep
348
+ them private when needed; the checked-in suite definition, revision pins, prompt
349
+ hashes, verifier definitions, rates, and aggregate result are sufficient to make
350
+ the experimental design auditable.
351
+
352
+ ### v1.13 optimization-platform treatment exposure
353
+
354
+ The v1.13 recovery/schema/prefix/proxy/browser/optimizer additions are not
355
+ assigned a new savings percentage merely because their mechanism tests pass.
356
+ A publishable bundle experiment must keep the task, repository revision, model,
357
+ turn/tool limits, verifier, and pricing source identical between paired arms.
358
+
359
+ For the broad platform bundle, the control should use the same installed Token
360
+ Saver binary with the new treatment surfaces disabled or left at their
361
+ backward-compatible defaults. The treatment may enable adaptive MCP disclosure,
362
+ recoverable schema compression, provider request transformation, and other
363
+ declared v1.13 surfaces. **Do not change the model between arms** when the goal
364
+ is to measure this bundle; model-routing savings require their own calibrated
365
+ experiment or a design that explicitly isolates model choice.
366
+
367
+ The experiment artifact must prove treatment exposure rather than assuming that
368
+ configuration implies use. At minimum, report:
369
+
370
+ - MCP profile/tool-list exposure and whether schema compression actually changed
371
+ an advertised catalog;
372
+ - provider transform activation and before/after request-token estimates;
373
+ - stable-prefix hit/miss counters when prefix tracking is part of the treatment;
374
+ - recovery handles emitted and successful exact-recovery spot checks;
375
+ - memory/tool-result/browser optimizations only on tasks where those surfaces
376
+ were actually exercised;
377
+ - total tool calls, repeated Reads/commands, provider input/cache/output usage,
378
+ latency, verifier success, and blind response quality.
379
+
380
+ A feature that never activates is not evidence for that feature. Bundle-level
381
+ cost-per-success may still be measured when the randomized treatment is the
382
+ whole declared platform, but the report must preserve per-feature activation so
383
+ readers can distinguish “enabled” from “used.”
384
+
385
+ The existing publication gate remains authoritative: enough distinct tasks and
386
+ paired trials, independent verification, blind quality parity, complete
387
+ model/cache-aware pricing evidence, and a strictly positive task-cluster 95%
388
+ confidence-interval lower bound for cost-per-success reduction. There is
389
+ currently **no completed publishable v1.13 bundle experiment** in the repository.
390
+
391
+ ## Paired agent outcomes
392
+
393
+ Record independently validated baseline and ACCO runs using the schema in
394
+ `benchmarks/agent-runs.example.json`, then run:
395
+
396
+ ```bash
397
+ acco agent-evaluate benchmarks/agent-runs.json
398
+ ```
399
+
400
+ The evaluator pairs runs by `(task, trial, condition)`. `trial` defaults to
401
+ `1` for backward-compatible single-pair manifests, but repeated experiments
402
+ should number trials explicitly. It reports input and output tokens, output
403
+ tokens per successful run, retries, elapsed time, context failures, success
404
+ rate, and total tokens per success. It also summarizes paired per-trial
405
+ reductions with median, p10/p90, standard deviation, and a deterministic 95%
406
+ bootstrap confidence interval.
407
+
408
+ Raw token reductions remain measurable without a quality grader, but
409
+ `claim_allowed` is **false unless blind quality evidence is present**, task
410
+ success is at parity, correctness/safety and weighted quality remain within the
411
+ documented tolerance, blockers do not increase, and the per-success denominator
412
+ is available.
413
+
414
+ For generation-time output-policy experiments, the same manifest can carry
415
+ optional **blind response-quality evidence**. Score both conditions on identical
416
+ tasks/trials after relabelling them so the grader cannot see which response came
417
+ from ACCO. The built-in rubric weights correctness 40%, completeness
418
+ 20%, actionability 15%, safety 15%, and concision 10%.
419
+
420
+ ```json
421
+ {
422
+ "quality_evaluation": {
423
+ "blinded": true,
424
+ "judge": "independent-response-grader"
425
+ },
426
+ "runs": [
427
+ {
428
+ "task": "fix-session-refresh",
429
+ "trial": 1,
430
+ "condition": "baseline",
431
+ "success": true,
432
+ "input_tokens": 18000,
433
+ "output_tokens": 1200,
434
+ "quality": {
435
+ "correctness": 5,
436
+ "completeness": 5,
437
+ "actionability": 4,
438
+ "safety": 5,
439
+ "concision": 3
440
+ },
441
+ "blocker": false
442
+ }
443
+ ]
444
+ }
445
+ ```
446
+
447
+ When quality scores are supplied, every paired run must be scored. ACCO
448
+ then requires task-success parity, no material correctness/safety regression,
449
+ no increase in blockers, and weighted-quality parity before authorizing a
450
+ savings claim. `raw_output_token_reduction` remains a descriptive measurement;
451
+ `output_tokens_per_success_reduction` is the stronger generated-output metric,
452
+ and `tokens_per_success_reduction` measures total input + output efficiency per
453
+ successful run. Unblinded or missing quality evidence is reported through
454
+ `claim_blockers` and cannot authorize a savings claim.
455
+
456
+ For a publishable output-cost claim, use at least 20 distinct frozen tasks and
457
+ three randomized paired trials per task, keep the model/prompt/tool settings
458
+ identical, independently verify task success, and report output-token reduction
459
+ next to cost per success rather than treating response length alone as quality.
460
+
461
+ ### Automated end-to-end evidence pipeline
462
+
463
+ The high-level production path is now:
464
+
465
+ ```bash
466
+ acco evidence-run benchmarks/e2e-swebench-24.frozen.json \
467
+ --out benchmark-runs.json \
468
+ --require-publishable
469
+ ```
470
+
471
+ The command is checkpointed across paid agent runs and judge calls. It runs the
472
+ frozen randomized experiment, executes independent hidden verification, extracts
473
+ only final assistant response text for a balanced deterministic blind A/B judge,
474
+ writes quality scores back into the same run identities, evaluates exact
475
+ cache-TTL-aware cost per successful task, and emits an adaptive-budget
476
+ calibration artifact. The judge never receives baseline/ACCO labels,
477
+ patches, verifier outcomes, or billing data.
478
+
479
+ The repository also ships a paid GitHub workflow for the existing **24-task ×
480
+ 3-trial SWE-bench Verified suite**. A smoke pair must first prove real agent
481
+ usage, output-policy telemetry, blind grading, and pricing. Only then does the
482
+ 24-task matrix run. Aggregation blind-grades all 72 pairs and enforces the
483
+ publishability gate. The workflow requires an explicit paid-run confirmation;
484
+ merging the feature alone is not a savings result.
485
+
486
+ ### Joined output-effectiveness evidence
487
+
488
+ The paired experiment artifact now carries transcript-measured fresh input,
489
+ cache creation, cache read, output tokens, and model calls for both arms.
490
+ Enabled runs also carry the selected output task/mode/budget when Claude hooks
491
+ emit policy telemetry. After blind response grading has been attached to the
492
+ same task/trial records, run:
493
+
494
+ ```bash
495
+ acco output-effectiveness benchmark-runs.json \
496
+ --fresh-input-per-million <rate> \
497
+ --cache-creation-5m-per-million <rate> \
498
+ --cache-creation-1h-per-million <rate> \
499
+ --cache-creation-unknown-per-million <rate> \
500
+ --cache-read-per-million <rate> \
501
+ --output-per-million <rate> \
502
+ --require-publishable
503
+ ```
504
+
505
+ The publication gate requires the existing broad design (>=20 distinct frozen
506
+ tasks and >=3 trials/task), recomputes and verifies the frozen task-definition
507
+ SHA-256, and checks each run's model/revision/prompt hash against that frozen
508
+ suite. It also requires no task-success regression, blind correctness/safety
509
+ and weighted-quality parity, complete optimized-arm policy telemetry with a
510
+ measured task/mode/budget, exact telemetry/transcript usage agreement, and
511
+ complete cost evidence. Cache-creation usage is split into 5-minute, 1-hour,
512
+ and unknown-TTL buckets; any nonzero bucket without a supplied rate makes
513
+ derived cost incomplete. A positive point estimate is not enough: the 95%
514
+ task-cluster bootstrap interval for cost-per-success reduction must remain
515
+ strictly above zero. Repeated trials are resampled as one task cluster rather
516
+ than treated as independent evidence.
517
+
518
+ This is the preferred end-to-end output-cost claim surface. Raw context
519
+ reduction, response length, or a low budget-utilization ratio are not
520
+ substitutes for cost per independently verified successful task.
521
+
522
+ ### Frozen session-efficiency causal holdout
523
+
524
+ Operational dashboard savings are not enough to establish that continuity,
525
+ cross-turn dedup, or waste prevention improve coding-agent economics. Version
526
+ 1.8 therefore adds a separate frozen paired-agent holdout:
527
+
528
+ ```bash
529
+ acco session-holdout \
530
+ benchmarks/session-efficiency-swebench-24.frozen.json \
531
+ --out session-holdout-runs.json \
532
+ --require-publishable
533
+ ```
534
+
535
+ The suite reuses the already frozen 24 SWE-bench Verified tasks and hidden
536
+ verification definitions, with three randomized trials per task. The control is
537
+ **not a historical 1.6 executable**. Both arms install the same current Token
538
+ Saver build; the baseline sets only the four session-efficiency controls to
539
+ zero, while treatment sets them to one. This holds retrieval, output processors,
540
+ model, host adapter, prompt, revision, and grader constant.
541
+
542
+ Every arm uses the same two-session protocol:
543
+
544
+ 1. phase 1 receives the frozen task and is restricted to Read/Grep/Glob/Bash;
545
+ 2. phase 1 must leave benchmark-visible repository state unchanged;
546
+ 3. the runner invokes ACCO's actual `SessionStart:resume` hook;
547
+ 4. phase 2 starts in a new Claude home/session and implements/verifies the task;
548
+ 5. treatment may receive the structured continuity checkpoint; control cannot.
549
+
550
+ That forced boundary gives continuity a deterministic opportunity to affect the
551
+ second session without feeding phase-1 prose directly to phase 2.
552
+
553
+ #### Independent outcome metrics
554
+
555
+ The evaluator derives behavioral outcomes from raw Claude transcripts for both
556
+ arms:
557
+
558
+ - total tool calls;
559
+ - total input tokens, with exact cache fields retained for pricing;
560
+ - normalized repeated Bash calls;
561
+ - repeated identical failing Bash-result attempts;
562
+ - duplicate full-file Reads;
563
+ - independent task success;
564
+ - blind final-response quality;
565
+ - cache-TTL-aware cost per successful task.
566
+
567
+ Treatment-side ACCO efficiency events are kept in a separate
568
+ `feature_activation` block. They prove whether continuity, dedup, and waste
569
+ signals fired, but they are not substituted for outcome metrics.
570
+
571
+ #### Statistical design
572
+
573
+ The frozen suite contains 24 task clusters and 3 trials/task. Point estimates
574
+ are accompanied by deterministic 2,000-sample task-cluster bootstrap 95%
575
+ intervals for tool-call reduction, input-token reduction, retry reduction, and
576
+ cost-per-success reduction. Trials from the same task are sampled together to
577
+ avoid treating repeated trials as independent tasks.
578
+
579
+ The publication gate requires:
580
+
581
+ - at least 20 distinct tasks and 3 trials/task;
582
+ - exact frozen definition hash and isolated condition profiles;
583
+ - matching model/prompt/revision identity between arms;
584
+ - no manual intervention;
585
+ - independent task-success parity;
586
+ - blind response-quality parity;
587
+ - complete cache-TTL-aware pricing;
588
+ - zero session-efficiency events in control;
589
+ - exactly the forced continuity exposure path, with at least one restore for
590
+ every treatment arm-run;
591
+ - observed dedup, continuity, and waste feature families somewhere in treatment;
592
+ - positive cost-per-success point reduction;
593
+ - cost-per-success 95% task-cluster CI lower bound strictly above zero.
594
+
595
+ The design estimates the **combined session-efficiency bundle**. Feature
596
+ activation counts do not identify each mechanism's individual causal effect; a
597
+ future ablation design would be required for that.
598
+
599
+ The dedicated GitHub workflow requires `RUN_SESSION_288` because the 144
600
+ arm-runs contain 288 paid Claude task phases, before blind-grader calls. A paid
601
+ smoke proves the two-phase runner, control isolation, continuity hook, transcript
602
+ metrics, blind grader, and pricing path before the full matrix starts.
603
+
604
+ No session-efficiency savings percentage should be published from the frozen
605
+ definition alone. A claim starts only after the paid workflow completes and this
606
+ gate passes.
607
+
608
+ ### Frozen session-efficiency output-quality gate
609
+
610
+ The session-efficiency layer does not replace retrieval or end-to-end evidence.
611
+ Its expanded command processors have a separate deterministic frozen fixture:
612
+
613
+ ```bash
614
+ acco output-replay \
615
+ benchmarks/output-quality-session-v17.frozen.json \
616
+ --require-frozen
617
+ ```
618
+
619
+ The fixture definition is SHA-256 frozen and covers search, lint, typecheck,
620
+ compiled tests, build diagnostics, git status, and container logs. Each case can
621
+ require exact preserved evidence, minimum token reduction, and strings that the
622
+ transformer must not introduce. CI runs this gate in addition to the external
623
+ retrieval holdout and base-vs-candidate ranking checks.
624
+
625
+ Cross-turn dedup itself is intentionally simpler than fuzzy compression: it only
626
+ fires when the normalized command and exact output digest match. Unchanged
627
+ full-file Read dedup uses the existing verified read digest. These mechanics are
628
+ covered by deterministic tests; any real end-to-end savings claim still belongs
629
+ to the randomized `evidence-run` protocol with task success and blind quality.
630
+
631
+ The local `dashboard` is also an operational surface, not a benchmark. Its
632
+ tool-context savings are estimated from observed before/after text, while its
633
+ Claude usage counters come from available transcript billing fields. Those
634
+ categories stay separate and are never promoted to cost-per-success evidence.
635
+
636
+ ### Runtime budget telemetry
637
+
638
+ For ordinary Claude Code use, `output-telemetry` records actual transcript
639
+ usage counters against the selected policy budget without storing conversation
640
+ content. Use it to discover candidate task/mode groups for future experiments:
641
+
642
+ ```bash
643
+ acco output-telemetry . --json
644
+ ```
645
+
646
+ A low p90 budget-utilization ratio is only an **observational tuning signal**.
647
+ Do not feed it directly into calibration. A completed `Stop` turn is not proof
648
+ of task success, and telemetry has no blind response-quality score. Validate
649
+ candidate budget changes with paired successful runs before calibrating them.
650
+
651
+ ### Calibrating adaptive output budgets
652
+
653
+ ACCO can turn the same blind paired evidence into conservative learned
654
+ task/mode bases. Add `output_task` and `output_mode` to ACCO runs, then:
655
+
656
+ ```bash
657
+ acco output-calibrate benchmarks/agent-runs.json \
658
+ --out .acco.output-calibration.json
659
+ ```
660
+
661
+ The calibrator considers only pairs where baseline and ACCO both succeed,
662
+ the ACCO response has no blocker, and correctness, safety, and weighted
663
+ blind quality remain within the evaluator parity tolerance. At least three valid samples spanning at least three distinct task IDs are
664
+ required per task/mode. The recommendation is p90 observed output
665
+ tokens plus a 15% safety margin, bounded by the mode safety range. Failed,
666
+ unblinded, or degraded short runs therefore cannot train the controller toward
667
+ an artificially small budget.
668
+
669
+ ## Live host validation is a separate manual gate
670
+
671
+ Start with the consolidated configuration/index check, then run the deeper host
672
+ transport check before a live host trial:
673
+
674
+ ```bash
675
+ acco doctor . --require-ready
676
+ acco host-check . --require-ready
677
+ ```
678
+
679
+ This checks project/user hook configuration, probes the host executable version,
680
+ runs a synthetic 500-line Bash response through the real PostToolUse hook, and
681
+ retrieves an omitted middle line from ACCO's saved original output. That
682
+ is a local transport test; it does **not** prove the host actually feeds
683
+ `hookSpecificOutput.updatedToolOutput` back to the model.
684
+
685
+ For that final gate, capture a real host debug transcript and supply it explicitly:
686
+
687
+ ```bash
688
+ acco host-check . \
689
+ --live-evidence /path/to/claude-debug.log \
690
+ --require-live
691
+ ```
692
+
693
+ `live_verified=true` is reported only when the supplied evidence contains both
694
+ the host replacement field and ACCO's filtered-output recovery marker.
695
+ The manual validation protocol remains:
696
+
697
+ 1. Record `claude --version`, model ID, configuration, and ACCO version.
698
+ 2. Use a disposable project. Install the package and its project hooks. Check
699
+ that inherited user hooks do not run ACCO a second time.
700
+ 3. Start Claude Code with debugging enabled. Ask it to run a harmless command
701
+ that prints 500 distinct progress lines. Do not use a command with side effects.
702
+ 4. Verify the host accepts `hookSpecificOutput.updatedToolOutput` without a
703
+ validation error. The model-visible result should include the recovery note,
704
+ show the shortened head/tail, and preserve the structured Bash fields.
705
+ Hook stdout alone is not sufficient evidence of host acceptance.
706
+ 5. Ask it to retrieve a known omitted middle line using the supplied
707
+ `acco output` command. Verify the exact line without rerunning the
708
+ original command. Test stderr independently and test a failing Jest-style
709
+ output containing a test name, stack location and multiline assertion diff.
710
+ 6. Test a >220-line source read. A bounded Read must return actual bytes, and a
711
+ subsequent Edit must succeed. Repeat after `/compact`; prior full-read state
712
+ must not block access to needed source. Test two independent sessions.
713
+ 7. Save the version, debug evidence and outcomes. If the installed host does not
714
+ support structured replacement, use manual `filter` until upgraded. Do not
715
+ claim automatic savings based on an ignored hook field.
716
+
717
+ No paid model call or live Claude Code session was executed in the repair
718
+ workspace. Automated tests cover the documented contract, subprocess transport,
719
+ state isolation, retrieval, and accounting. Live behavior remains a separate gate.
720
+
721
+ ## Paired task protocol
722
+
723
+ - Choose representative tasks before measuring: bug fixes, refactors, unfamiliar
724
+ repo navigation, noisy passing tests, and failures requiring deep diagnostics.
725
+ - Use independent fresh worktrees at the same commit and the same exact prompt,
726
+ model, effort, tools, system instructions, and approval configuration.
727
+ - In baseline, disable all ACCO hooks (including user-scope hooks). In
728
+ enabled runs, use the release's hooks. Do not add orientation maps to only one
729
+ arm unless that is the specific intervention being tested.
730
+ - Randomize condition order and run multiple trials per task. Record cache
731
+ policy and cold/warm conditions rather than assuming a five-minute TTL.
732
+ - Keep tests/evaluation independent of the agent. Record success, regression
733
+ checks, elapsed time, retries and any manual intervention. A smaller context
734
+ that fails the task is not a win.
735
+ - Preserve each run's complete transcripts, including nested subagent
736
+ transcripts where applicable. Do not reuse a transcript across runs.
737
+ - Keep raw transcripts local; they may contain code or secrets.
738
+
739
+ `acco benchmark` analyzes recorded runs; it does not execute coding agents
740
+ or certify evaluator outcomes. It verifies paired task/trial identity, revision,
741
+ model and prompt metadata, requires measured usage and explicit prices, and
742
+ rejects incomplete or duplicate runs. Metadata must be recorded honestly by the
743
+ runner; the evaluator does not independently attest your checkout or prompt.
744
+
745
+ ## Run manifest
746
+
747
+ Paths are relative to the manifest file. `success` is determined by independent
748
+ validation, not by the agent's own assertion. Use the actual SHA-256 of the prompt.
749
+
750
+ ```json
751
+ {
752
+ "runs": [
753
+ {
754
+ "task": "fix-user-lookup",
755
+ "trial": 1,
756
+ "condition": "baseline",
757
+ "revision": "ACTUAL_COMMIT",
758
+ "model": "EXACT_MODEL_ID",
759
+ "prompt_sha256": "ACTUAL_PROMPT_SHA256",
760
+ "success": true,
761
+ "validation": "Independent tests passed; no regressions in required suite",
762
+ "seconds": 120.5,
763
+ "transcripts": ["runs/baseline.jsonl"]
764
+ },
765
+ {
766
+ "task": "fix-user-lookup",
767
+ "trial": 1,
768
+ "condition": "enabled",
769
+ "revision": "ACTUAL_COMMIT",
770
+ "model": "EXACT_MODEL_ID",
771
+ "prompt_sha256": "ACTUAL_PROMPT_SHA256",
772
+ "success": true,
773
+ "validation": "Same independent checks passed",
774
+ "seconds": 118.2,
775
+ "transcripts": ["runs/enabled.jsonl"]
776
+ }
777
+ ]
778
+ }
779
+ ```
780
+
781
+ The times above demonstrate the schema; they are not measured results.
782
+
783
+ ## Rate file
784
+
785
+ Create a JSON object mapping each exact model ID to numeric USD-per-million
786
+ rates: `input`, `cache_write_5m`, `cache_write_1h`, `cache_read`, `output`.
787
+ For example, this object is syntactically valid but deliberately uses a fictitious
788
+ model and synthetic prices. **It must not be used to price actual models.**
789
+
790
+ ```json
791
+ {
792
+ "synthetic-test-model": {
793
+ "input": 1,
794
+ "cache_write_5m": 1.25,
795
+ "cache_write_1h": 2,
796
+ "cache_read": 0.1,
797
+ "output": 5
798
+ }
799
+ }
800
+ ```
801
+
802
+ Only add `cache_write_unknown` when the run's actual cache configuration permits
803
+ an explicit rate for missing TTL details. Missing prices or unexplained cache
804
+ creation prevent a complete cost result.
805
+
806
+ ```bash
807
+ acco benchmark runs.json --rates rates.json
808
+ ```
809
+
810
+ The output includes per-condition success rates, total tokens by usage type,
811
+ tool-result counts, repeated reads, elapsed time, total cost, and cost per success.
812
+ Costs from failed attempts are included. A reduced success rate suppresses the
813
+ headline cost-per-success reduction. Inspect per-task outcomes too: aggregate
814
+ success parity does not prove each task retained the same quality.
815
+
816
+ For the automated experiment path above, `acco benchmark` reports a
817
+ deterministic task-cluster bootstrap 95% interval. This is still an empirical
818
+ benchmark, not a proof that savings generalize to every repository or model.
819
+ Before publishing savings, retain the frozen task definitions, repeated trials,
820
+ model and host versions, prices, independent checks, and raw results needed to
821
+ reproduce the claim.
822
+
823
+
824
+ ## Frozen knowledge-efficiency holdout
825
+
826
+ Knowledge-assisted read avoidance is evaluated separately from retrieval recall
827
+ and from the existing session-efficiency bundle. The frozen definition is:
828
+
829
+ ```text
830
+ benchmarks/knowledge-efficiency-swebench-24.frozen.json
831
+ 24 SWE-bench Verified tasks
832
+ 3 trials per task
833
+ 2 conditions
834
+ = 144 arm-runs
835
+ = 288 fresh Claude task phases
836
+ ```
837
+
838
+ The causal contract intentionally equalizes memory creation. Phase 1 is
839
+ investigation-only in both arms and requires the agent to persist 1–3
840
+ `verified` ACCO findings backed by source it actually inspected.
841
+ Repository state must remain unchanged. Phase 2 uses a fresh Claude home/session
842
+ and receives no conversation transcript or continuity checkpoint.
843
+
844
+ The control and treatment install the **same current ACCO binary**.
845
+ Both disable continuity, cross-turn command/read dedup, reread blocking, and
846
+ behavioral waste detection. The control disables knowledge read avoidance and
847
+ cache economics; the treatment enables those two switches. This tests the
848
+ combined **knowledge read-avoidance + cache gate** effect without attributing
849
+ savings to unrelated 1.7 session features.
850
+
851
+ Outcome metrics come from independent evidence:
852
+
853
+ - raw transcripts: total tool calls, input tokens, and duplicate Reads;
854
+ - hidden task verifier: task success;
855
+ - blinded judge: response quality parity;
856
+ - transcript cache-usage fields plus frozen rates: billed cost/cost per success;
857
+ - ACCO's efficiency ledger: feature exposure only (knowledge seeds,
858
+ read-avoidance interventions, and cache-economics-approved interventions).
859
+
860
+ The ledger never grades its own success. A publishable claim requires at least
861
+ 20 tasks, three trials per task, all paired identities intact, no manual
862
+ intervention, success parity, blind-quality verification, complete pricing,
863
+ verified knowledge seeding in every arm-run, zero control avoidance activation,
864
+ observed treatment activation, positive cost-per-success reduction, and a
865
+ task-cluster 95% confidence interval with a strictly positive lower bound.
866
+
867
+ Run the frozen preflight without paid execution:
868
+
869
+ ```bash
870
+ acco knowledge-holdout \
871
+ benchmarks/knowledge-efficiency-swebench-24.frozen.json \
872
+ --out /tmp/knowledge-holdout-runs.json \
873
+ --dry-run
874
+ ```
875
+
876
+ The paid workflow is
877
+ `.github/workflows/knowledge-efficiency-holdout.yml` and requires the explicit
878
+ `RUN_KNOWLEDGE_288` confirmation. Until it completes successfully, the
879
+ mechanism has **no publishable end-to-end savings percentage**.
880
+