acco 1.15.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- acco-1.15.0/BENCHMARKING.md +880 -0
- acco-1.15.0/CHANGELOG.md +1995 -0
- acco-1.15.0/INTEGRATIONS.md +318 -0
- acco-1.15.0/LICENSE +21 -0
- acco-1.15.0/MANIFEST.in +3 -0
- acco-1.15.0/PKG-INFO +1246 -0
- acco-1.15.0/README.md +1210 -0
- acco-1.15.0/VALIDATION.md +859 -0
- acco-1.15.0/benchmarks/agent-runs.example.json +49 -0
- acco-1.15.0/benchmarks/claude-sonnet-5-rates-2026-09-19.json +9 -0
- acco-1.15.0/benchmarks/cli-output-compression-ratchet-v1.frozen.json +663 -0
- acco-1.15.0/benchmarks/cli-output-real-v1/manifest.json +367 -0
- acco-1.15.0/benchmarks/cli-output-real-v2/freeze.json +21 -0
- acco-1.15.0/benchmarks/cli-output-real-v3/manifest.json +415 -0
- acco-1.15.0/benchmarks/context-quality.json +27 -0
- acco-1.15.0/benchmarks/e2e-suite.example.json +68 -0
- acco-1.15.0/benchmarks/e2e-swebench-24.frozen.json +675 -0
- acco-1.15.0/benchmarks/holdout-external-10.json +910 -0
- acco-1.15.0/benchmarks/holdout-external-10.result.json +665 -0
- acco-1.15.0/benchmarks/holdout-external-11.json +1134 -0
- acco-1.15.0/benchmarks/holdout-external-11.result.json +827 -0
- acco-1.15.0/benchmarks/holdout-external-12-candidate-rejected.json +484 -0
- acco-1.15.0/benchmarks/holdout-external-12-candidate-rejected.result.json +362 -0
- acco-1.15.0/benchmarks/holdout-external-12-pilot-rejected.json +484 -0
- acco-1.15.0/benchmarks/holdout-external-12-pilot-rejected.result.json +362 -0
- acco-1.15.0/benchmarks/holdout-external-12.json +1358 -0
- acco-1.15.0/benchmarks/holdout-external-12.result.json +989 -0
- acco-1.15.0/benchmarks/holdout-external-2.json +510 -0
- acco-1.15.0/benchmarks/holdout-external-2.result.json +431 -0
- acco-1.15.0/benchmarks/holdout-external-3.json +382 -0
- acco-1.15.0/benchmarks/holdout-external-3.result.json +320 -0
- acco-1.15.0/benchmarks/holdout-external-4.json +488 -0
- acco-1.15.0/benchmarks/holdout-external-4.result.json +422 -0
- acco-1.15.0/benchmarks/holdout-external-5.json +368 -0
- acco-1.15.0/benchmarks/holdout-external-5.result.json +320 -0
- acco-1.15.0/benchmarks/holdout-external-6.json +470 -0
- acco-1.15.0/benchmarks/holdout-external-6.result.json +328 -0
- acco-1.15.0/benchmarks/holdout-external-7.json +910 -0
- acco-1.15.0/benchmarks/holdout-external-7.result.json +608 -0
- acco-1.15.0/benchmarks/holdout-external-8.json +910 -0
- acco-1.15.0/benchmarks/holdout-external-8.result.json +608 -0
- acco-1.15.0/benchmarks/holdout-external-9.json +910 -0
- acco-1.15.0/benchmarks/holdout-external-9.result.json +665 -0
- acco-1.15.0/benchmarks/holdout-external.floor.json +45 -0
- acco-1.15.0/benchmarks/holdout-external.json +63 -0
- acco-1.15.0/benchmarks/holdout-external.result.json +80 -0
- acco-1.15.0/benchmarks/holdout.example.json +44 -0
- acco-1.15.0/benchmarks/knowledge-efficiency-swebench-24.frozen.json +716 -0
- acco-1.15.0/benchmarks/output-quality-session-v17.frozen.json +109 -0
- acco-1.15.0/benchmarks/output-quality.example.json +25 -0
- acco-1.15.0/benchmarks/semantic-holdout-13.frozen.json +496 -0
- acco-1.15.0/benchmarks/semantic-holdout-13.query-freeze.json +262 -0
- acco-1.15.0/benchmarks/semantic-holdout-14.frozen.json +473 -0
- acco-1.15.0/benchmarks/semantic-holdout-14.query-freeze.json +253 -0
- acco-1.15.0/benchmarks/semantic-holdout-15.query-freeze.json +253 -0
- acco-1.15.0/benchmarks/session-efficiency-swebench-24.frozen.json +711 -0
- acco-1.15.0/integrations/claude-code.mcp.json +8 -0
- acco-1.15.0/integrations/codex.config.toml +3 -0
- acco-1.15.0/integrations/cursor.mcp.json +8 -0
- acco-1.15.0/integrations/github-actions.yml +18 -0
- acco-1.15.0/pyproject.toml +70 -0
- acco-1.15.0/setup.cfg +4 -0
- acco-1.15.0/src/acco/__init__.py +3 -0
- acco-1.15.0/src/acco/agent_eval.py +336 -0
- acco-1.15.0/src/acco/audit.py +242 -0
- acco-1.15.0/src/acco/benchmark.py +354 -0
- acco-1.15.0/src/acco/blind_grader.py +469 -0
- acco-1.15.0/src/acco/browser_context.py +197 -0
- acco-1.15.0/src/acco/budget.py +72 -0
- acco-1.15.0/src/acco/cache_economics.py +145 -0
- acco-1.15.0/src/acco/claude_docker.py +181 -0
- acco-1.15.0/src/acco/claude_grader.py +114 -0
- acco-1.15.0/src/acco/claude_plugin.py +203 -0
- acco-1.15.0/src/acco/cli.py +721 -0
- acco-1.15.0/src/acco/client_capabilities.py +259 -0
- acco-1.15.0/src/acco/closure.py +181 -0
- acco-1.15.0/src/acco/command_handlers/__init__.py +6 -0
- acco-1.15.0/src/acco/command_handlers/context.py +322 -0
- acco-1.15.0/src/acco/command_handlers/efficiency.py +371 -0
- acco-1.15.0/src/acco/command_handlers/evaluation.py +238 -0
- acco-1.15.0/src/acco/command_handlers/experiment.py +412 -0
- acco-1.15.0/src/acco/command_handlers/host.py +290 -0
- acco-1.15.0/src/acco/command_handlers/ingress.py +51 -0
- acco-1.15.0/src/acco/command_handlers/model_routing.py +142 -0
- acco-1.15.0/src/acco/command_handlers/optimization.py +272 -0
- acco-1.15.0/src/acco/command_handlers/output.py +388 -0
- acco-1.15.0/src/acco/command_handlers/patch.py +81 -0
- acco-1.15.0/src/acco/command_handlers/pricing.py +82 -0
- acco-1.15.0/src/acco/command_registry.py +183 -0
- acco-1.15.0/src/acco/commands.py +84 -0
- acco-1.15.0/src/acco/config.py +24 -0
- acco-1.15.0/src/acco/context_browser.py +93 -0
- acco-1.15.0/src/acco/context_router.py +332 -0
- acco-1.15.0/src/acco/cost_report.py +462 -0
- acco-1.15.0/src/acco/delta_context.py +288 -0
- acco-1.15.0/src/acco/efficiency/__init__.py +20 -0
- acco-1.15.0/src/acco/efficiency/advisor.py +438 -0
- acco-1.15.0/src/acco/efficiency/dashboard.py +113 -0
- acco-1.15.0/src/acco/efficiency/report.py +103 -0
- acco-1.15.0/src/acco/efficiency/service.py +534 -0
- acco-1.15.0/src/acco/efficiency/store.py +198 -0
- acco-1.15.0/src/acco/entry.py +40 -0
- acco-1.15.0/src/acco/estimate.py +201 -0
- acco-1.15.0/src/acco/evaluate.py +321 -0
- acco-1.15.0/src/acco/evidence_pipeline.py +164 -0
- acco-1.15.0/src/acco/experiment.py +1057 -0
- acco-1.15.0/src/acco/fastpath.py +210 -0
- acco-1.15.0/src/acco/feedback.py +47 -0
- acco-1.15.0/src/acco/filter_output.py +43 -0
- acco-1.15.0/src/acco/generation_policy.py +229 -0
- acco-1.15.0/src/acco/guard.py +398 -0
- acco-1.15.0/src/acco/hook.py +201 -0
- acco-1.15.0/src/acco/hook_runtime.py +669 -0
- acco-1.15.0/src/acco/host_configs.py +542 -0
- acco-1.15.0/src/acco/host_validate.py +178 -0
- acco-1.15.0/src/acco/ignore.py +81 -0
- acco-1.15.0/src/acco/images.py +130 -0
- acco-1.15.0/src/acco/impact.py +114 -0
- acco-1.15.0/src/acco/ingress.py +279 -0
- acco-1.15.0/src/acco/install.py +124 -0
- acco-1.15.0/src/acco/integration_setup.py +759 -0
- acco-1.15.0/src/acco/knowledge.py +733 -0
- acco-1.15.0/src/acco/knowledge_holdout.py +409 -0
- acco-1.15.0/src/acco/knowledge_holdout_docker.py +208 -0
- acco-1.15.0/src/acco/knowledge_holdout_pipeline.py +109 -0
- acco-1.15.0/src/acco/lexical.py +199 -0
- acco-1.15.0/src/acco/mapstat.py +33 -0
- acco-1.15.0/src/acco/mcp.py +154 -0
- acco-1.15.0/src/acco/mcp_server/__init__.py +19 -0
- acco-1.15.0/src/acco/mcp_server/contracts.py +121 -0
- acco-1.15.0/src/acco/mcp_server/protocol.py +196 -0
- acco-1.15.0/src/acco/mcp_server/services.py +13 -0
- acco-1.15.0/src/acco/mcp_server/tool_surface.py +139 -0
- acco-1.15.0/src/acco/mcp_server/tools.py +830 -0
- acco-1.15.0/src/acco/mcp_server/transport.py +41 -0
- acco-1.15.0/src/acco/model_routing.py +1118 -0
- acco-1.15.0/src/acco/optimizer.py +330 -0
- acco-1.15.0/src/acco/output/__init__.py +41 -0
- acco-1.15.0/src/acco/output/contracts.py +55 -0
- acco-1.15.0/src/acco/output/pipeline.py +134 -0
- acco-1.15.0/src/acco/output/processors.py +677 -0
- acco-1.15.0/src/acco/output/registry.py +31 -0
- acco-1.15.0/src/acco/output/specialized_processors.py +250 -0
- acco-1.15.0/src/acco/output/text.py +146 -0
- acco-1.15.0/src/acco/output_benchmark.py +89 -0
- acco-1.15.0/src/acco/output_budget.py +351 -0
- acco-1.15.0/src/acco/output_effectiveness.py +836 -0
- acco-1.15.0/src/acco/output_processors.py +47 -0
- acco-1.15.0/src/acco/output_quality.py +237 -0
- acco-1.15.0/src/acco/output_saver.py +355 -0
- acco-1.15.0/src/acco/output_store.py +37 -0
- acco-1.15.0/src/acco/output_telemetry.py +622 -0
- acco-1.15.0/src/acco/pack.py +431 -0
- acco-1.15.0/src/acco/pack_cli.py +144 -0
- acco-1.15.0/src/acco/packing/__init__.py +23 -0
- acco-1.15.0/src/acco/packing/contracts.py +78 -0
- acco-1.15.0/src/acco/packing/file_scoring.py +435 -0
- acco-1.15.0/src/acco/packing/graph_rerank.py +517 -0
- acco-1.15.0/src/acco/packing/observability.py +77 -0
- acco-1.15.0/src/acco/packing/query_analysis.py +227 -0
- acco-1.15.0/src/acco/packing/ranking.py +154 -0
- acco-1.15.0/src/acco/packing/ranking_defaults.py +11 -0
- acco-1.15.0/src/acco/packing/ranking_stages.py +116 -0
- acco-1.15.0/src/acco/packing/render.py +71 -0
- acco-1.15.0/src/acco/packing/symbol_scoring.py +521 -0
- acco-1.15.0/src/acco/packing/symbol_windows.py +251 -0
- acco-1.15.0/src/acco/packing/symbols.py +41 -0
- acco-1.15.0/src/acco/paired_conditions.py +18 -0
- acco-1.15.0/src/acco/patch_context.py +332 -0
- acco-1.15.0/src/acco/policy.py +222 -0
- acco-1.15.0/src/acco/prefix_cache.py +139 -0
- acco-1.15.0/src/acco/pricing.py +224 -0
- acco-1.15.0/src/acco/pricing_registry.json +70 -0
- acco-1.15.0/src/acco/processor_mining.py +173 -0
- acco-1.15.0/src/acco/provider_proxy.py +266 -0
- acco-1.15.0/src/acco/provider_transform.py +268 -0
- acco-1.15.0/src/acco/prune.py +48 -0
- acco-1.15.0/src/acco/ranking_calibration.py +336 -0
- acco-1.15.0/src/acco/ranking_regression.py +555 -0
- acco-1.15.0/src/acco/recovery.py +234 -0
- acco-1.15.0/src/acco/repo_index.py +640 -0
- acco-1.15.0/src/acco/repository_service.py +348 -0
- acco-1.15.0/src/acco/retrieval_cache.py +272 -0
- acco-1.15.0/src/acco/retrieval_vnext.py +119 -0
- acco-1.15.0/src/acco/runtime_config.py +676 -0
- acco-1.15.0/src/acco/security.py +74 -0
- acco-1.15.0/src/acco/semantic_holdout.py +602 -0
- acco-1.15.0/src/acco/semantic_retrieval.py +1076 -0
- acco-1.15.0/src/acco/semantic_ts.py +381 -0
- acco-1.15.0/src/acco/serve.py +53 -0
- acco-1.15.0/src/acco/session_holdout.py +590 -0
- acco-1.15.0/src/acco/session_holdout_docker.py +388 -0
- acco-1.15.0/src/acco/session_holdout_pipeline.py +158 -0
- acco-1.15.0/src/acco/session_metrics.py +211 -0
- acco-1.15.0/src/acco/sessions.py +339 -0
- acco-1.15.0/src/acco/skeleton.py +698 -0
- acco-1.15.0/src/acco/snippet.py +95 -0
- acco-1.15.0/src/acco/state.py +147 -0
- acco-1.15.0/src/acco/swebench_docker.py +136 -0
- acco-1.15.0/src/acco/syntax.py +914 -0
- acco-1.15.0/src/acco/tool_proxy.py +534 -0
- acco-1.15.0/src/acco/tool_schema.py +176 -0
- acco-1.15.0/src/acco/unified_audit.py +226 -0
- acco-1.15.0/src/acco/working_set.py +45 -0
- acco-1.15.0/src/acco.egg-info/PKG-INFO +1246 -0
- acco-1.15.0/src/acco.egg-info/SOURCES.txt +315 -0
- acco-1.15.0/src/acco.egg-info/dependency_links.txt +1 -0
- acco-1.15.0/src/acco.egg-info/entry_points.txt +3 -0
- acco-1.15.0/src/acco.egg-info/requires.txt +33 -0
- acco-1.15.0/src/acco.egg-info/top_level.txt +1 -0
- acco-1.15.0/tests/test_adaptive_budget.py +93 -0
- acco-1.15.0/tests/test_architecture_boundaries.py +134 -0
- acco-1.15.0/tests/test_audit.py +171 -0
- acco-1.15.0/tests/test_benchmark_evidence.py +71 -0
- acco-1.15.0/tests/test_blind_grader.py +186 -0
- acco-1.15.0/tests/test_budget_map.py +106 -0
- acco-1.15.0/tests/test_cache_economics.py +86 -0
- acco-1.15.0/tests/test_callable_identity_v5.py +264 -0
- acco-1.15.0/tests/test_callable_symbol_ranking_v3.py +166 -0
- acco-1.15.0/tests/test_check_holdout.py +67 -0
- acco-1.15.0/tests/test_claude_docker.py +233 -0
- acco-1.15.0/tests/test_claude_grader.py +44 -0
- acco-1.15.0/tests/test_claude_plugin.py +74 -0
- acco-1.15.0/tests/test_cli.py +231 -0
- acco-1.15.0/tests/test_client_capabilities.py +60 -0
- acco-1.15.0/tests/test_context_browser_mcp.py +26 -0
- acco-1.15.0/tests/test_context_router.py +84 -0
- acco-1.15.0/tests/test_cost_advisor.py +206 -0
- acco-1.15.0/tests/test_cost_report.py +334 -0
- acco-1.15.0/tests/test_csharp14_extension_blocks.py +154 -0
- acco-1.15.0/tests/test_delta_context.py +94 -0
- acco-1.15.0/tests/test_documentation.py +301 -0
- acco-1.15.0/tests/test_efficiency_cli.py +103 -0
- acco-1.15.0/tests/test_entry.py +240 -0
- acco-1.15.0/tests/test_estimate.py +122 -0
- acco-1.15.0/tests/test_evaluate_holdout.py +135 -0
- acco-1.15.0/tests/test_evaluate_scoped_recall.py +66 -0
- acco-1.15.0/tests/test_evidence.py +210 -0
- acco-1.15.0/tests/test_evidence_pipeline.py +267 -0
- acco-1.15.0/tests/test_experiment.py +551 -0
- acco-1.15.0/tests/test_fastpath.py +103 -0
- acco-1.15.0/tests/test_filter.py +110 -0
- acco-1.15.0/tests/test_filter_dedupe.py +20 -0
- acco-1.15.0/tests/test_frozen_e2e_suite.py +27 -0
- acco-1.15.0/tests/test_fuzzy_symbol.py +71 -0
- acco-1.15.0/tests/test_generation_policy.py +249 -0
- acco-1.15.0/tests/test_guard.py +326 -0
- acco-1.15.0/tests/test_holdout11_failure_classes.py +250 -0
- acco-1.15.0/tests/test_hook.py +308 -0
- acco-1.15.0/tests/test_hook_runtime.py +701 -0
- acco-1.15.0/tests/test_images.py +90 -0
- acco-1.15.0/tests/test_index_robustness.py +83 -0
- acco-1.15.0/tests/test_ingress.py +112 -0
- acco-1.15.0/tests/test_integration_setup.py +578 -0
- acco-1.15.0/tests/test_knowledge.py +280 -0
- acco-1.15.0/tests/test_knowledge_holdout.py +336 -0
- acco-1.15.0/tests/test_large_output_example.py +99 -0
- acco-1.15.0/tests/test_lexical.py +69 -0
- acco-1.15.0/tests/test_mapstat.py +21 -0
- acco-1.15.0/tests/test_mcp.py +91 -0
- acco-1.15.0/tests/test_mcp_profiles.py +267 -0
- acco-1.15.0/tests/test_mcp_server_boundaries.py +143 -0
- acco-1.15.0/tests/test_model_routing.py +596 -0
- acco-1.15.0/tests/test_multi_host_integrations.py +452 -0
- acco-1.15.0/tests/test_multilang_syntax.py +246 -0
- acco-1.15.0/tests/test_optimization_platform.py +373 -0
- acco-1.15.0/tests/test_output_benchmark.py +67 -0
- acco-1.15.0/tests/test_output_budget.py +352 -0
- acco-1.15.0/tests/test_output_effectiveness.py +496 -0
- acco-1.15.0/tests/test_output_processor_expansion.py +353 -0
- acco-1.15.0/tests/test_output_processors.py +230 -0
- acco-1.15.0/tests/test_output_quality.py +209 -0
- acco-1.15.0/tests/test_output_saver.py +131 -0
- acco-1.15.0/tests/test_output_saver_mcp.py +45 -0
- acco-1.15.0/tests/test_output_telemetry.py +418 -0
- acco-1.15.0/tests/test_overload_resolution_v4.py +203 -0
- acco-1.15.0/tests/test_pack.py +700 -0
- acco-1.15.0/tests/test_pack_pipeline_boundaries.py +107 -0
- acco-1.15.0/tests/test_policy.py +61 -0
- acco-1.15.0/tests/test_pricing_registry.py +123 -0
- acco-1.15.0/tests/test_processor_mining.py +62 -0
- acco-1.15.0/tests/test_prune.py +13 -0
- acco-1.15.0/tests/test_ranking_artifact_collector.py +250 -0
- acco-1.15.0/tests/test_ranking_calibration.py +239 -0
- acco-1.15.0/tests/test_ranking_calibration_workflow.py +85 -0
- acco-1.15.0/tests/test_ranking_ci_workflow.py +46 -0
- acco-1.15.0/tests/test_ranking_observability.py +230 -0
- acco-1.15.0/tests/test_ranking_pipeline_boundaries.py +141 -0
- acco-1.15.0/tests/test_ranking_regression.py +275 -0
- acco-1.15.0/tests/test_ranking_stage_registry.py +215 -0
- acco-1.15.0/tests/test_ranking_stages.py +203 -0
- acco-1.15.0/tests/test_release_regressions.py +251 -0
- acco-1.15.0/tests/test_repo_index.py +174 -0
- acco-1.15.0/tests/test_repository_service.py +157 -0
- acco-1.15.0/tests/test_retrieval_cache.py +128 -0
- acco-1.15.0/tests/test_retrieval_vnext.py +59 -0
- acco-1.15.0/tests/test_semantic_holdout.py +329 -0
- acco-1.15.0/tests/test_semantic_ref_closure.py +167 -0
- acco-1.15.0/tests/test_semantic_retrieval.py +829 -0
- acco-1.15.0/tests/test_semantic_ts.py +140 -0
- acco-1.15.0/tests/test_session_efficiency.py +261 -0
- acco-1.15.0/tests/test_session_holdout.py +313 -0
- acco-1.15.0/tests/test_session_holdout_docker.py +172 -0
- acco-1.15.0/tests/test_session_metrics.py +142 -0
- acco-1.15.0/tests/test_sessions.py +266 -0
- acco-1.15.0/tests/test_skeleton.py +305 -0
- acco-1.15.0/tests/test_snippet.py +36 -0
- acco-1.15.0/tests/test_structural_authority.py +100 -0
- acco-1.15.0/tests/test_structural_symbol_identity.py +128 -0
- acco-1.15.0/tests/test_swebench_grading.py +230 -0
- acco-1.15.0/tests/test_symbol_pipeline_boundaries.py +110 -0
- acco-1.15.0/tests/test_tool_proxy.py +220 -0
- acco-1.15.0/tests/test_unified_audit.py +38 -0
- acco-1.15.0/tests/test_v09.py +225 -0
- acco-1.15.0/tests/test_v1.py +477 -0
- acco-1.15.0/tests/test_v12_ports.py +96 -0
- acco-1.15.0/tests/test_version.py +12 -0
|
@@ -0,0 +1,880 @@
|
|
|
1
|
+
> **Branding note:** frozen benchmark artifacts created before the ACCO rename retain their original Token Saver identifiers and hashes. The documentation uses the current ACCO product name; frozen evidence files are not rewritten.
|
|
2
|
+
|
|
3
|
+
# Verify integration, then benchmark successful work
|
|
4
|
+
|
|
5
|
+
## Deterministic context-quality benchmark
|
|
6
|
+
|
|
7
|
+
Before paid paired-agent trials, run the included 25-task ground-truth selector
|
|
8
|
+
benchmark. It measures relevant-file recall, relevant-symbol recall, and context
|
|
9
|
+
reduction; selection metrics alone do not prove agent success.
|
|
10
|
+
|
|
11
|
+
```bash
|
|
12
|
+
acco evaluate benchmarks/context-quality.json --path . --max-tokens 6000
|
|
13
|
+
```
|
|
14
|
+
|
|
15
|
+
Each item also reports `symbol_recall_in_expected_files`, which counts a
|
|
16
|
+
symbol only when it was selected from one of the task's expected files. Bare
|
|
17
|
+
`symbol_recall` can be satisfied by a same-named symbol in an unrelated file;
|
|
18
|
+
prefer the scoped figure (or `qualified_symbols` / `symbol_identities`) when
|
|
19
|
+
names are common.
|
|
20
|
+
|
|
21
|
+
Add project-specific tasks using `query`, `files`, and `symbols`. Keep the
|
|
22
|
+
manifest under version control so ranking changes can be compared reproducibly.
|
|
23
|
+
|
|
24
|
+
## Multi-repository holdout benchmark
|
|
25
|
+
|
|
26
|
+
### Query-construction protocol
|
|
27
|
+
|
|
28
|
+
A frozen hash prevents post-hoc edits to the **query text**, expected files, and
|
|
29
|
+
expected symbols. It does not make an easy query difficult. New holdouts must
|
|
30
|
+
therefore declare which retrieval behavior they are measuring before the first
|
|
31
|
+
ACCO run.
|
|
32
|
+
|
|
33
|
+
For a **semantic natural-language holdout**, construct each query from the
|
|
34
|
+
upstream issue, bug report, user request, or behavior description **before
|
|
35
|
+
looking up the answer identity**. The query must not contain:
|
|
36
|
+
|
|
37
|
+
- the target symbol/member name;
|
|
38
|
+
- the target's containing class/type/module name when that name directly
|
|
39
|
+
identifies the answer;
|
|
40
|
+
- the target file basename/path;
|
|
41
|
+
- the exact qualified symbol identity used as ground truth;
|
|
42
|
+
- a distinctive implementation literal copied from the answer solely to make
|
|
43
|
+
retrieval easier.
|
|
44
|
+
|
|
45
|
+
Example:
|
|
46
|
+
|
|
47
|
+
```text
|
|
48
|
+
acceptable: "reject malformed email addresses before schema validation"
|
|
49
|
+
not semantic: "where is emailRegex in regexes.ts"
|
|
50
|
+
```
|
|
51
|
+
|
|
52
|
+
If the real user task already contains an identifier (for example, a compiler
|
|
53
|
+
error names `refreshSession`), keep it: removing genuine task evidence would
|
|
54
|
+
make the benchmark artificial. Classify that task/suite as **identifier-bearing**
|
|
55
|
+
rather than semantic-natural-language and report it separately.
|
|
56
|
+
|
|
57
|
+
Do not present identifier-bearing recall as a continuation of a
|
|
58
|
+
natural-language difficulty trend. The two answer different questions:
|
|
59
|
+
|
|
60
|
+
- semantic suites test whether ACCO can discover the identity from a
|
|
61
|
+
behavior/problem description;
|
|
62
|
+
- identifier-bearing suites test exact navigation, overload resolution,
|
|
63
|
+
scoping, and identity recovery once some answer vocabulary is already known.
|
|
64
|
+
|
|
65
|
+
For future headline semantic holdouts:
|
|
66
|
+
|
|
67
|
+
1. pre-register the query source/rule and freeze the exact query text;
|
|
68
|
+
2. keep target symbol, containing type, file path, and qualified identity out of
|
|
69
|
+
the query unless they genuinely appeared in the upstream task;
|
|
70
|
+
3. record any unavoidable identifier-bearing tasks explicitly;
|
|
71
|
+
4. run a trivial lexical baseline (for example grep/ctags/exact identifier
|
|
72
|
+
lookup where applicable) on the same frozen tasks;
|
|
73
|
+
5. report ACCO recall alongside that baseline rather than quoting only an
|
|
74
|
+
absolute recall percentage;
|
|
75
|
+
6. never rewrite queries after seeing retrieval misses.
|
|
76
|
+
|
|
77
|
+
The query source/rule belongs in the same committed evidence package as the
|
|
78
|
+
manifest. The manifest's query strings themselves are part of the ground-truth
|
|
79
|
+
freeze hash, so changing the wording after freeze changes benchmark identity.
|
|
80
|
+
|
|
81
|
+
The repository-local selector benchmark is useful for regressions, but because
|
|
82
|
+
ACCO is developed against this codebase it is not independent evidence of
|
|
83
|
+
generalization. For unseen evaluation, define ground truth before running the
|
|
84
|
+
tool and point one manifest at repositories that were excluded from ranking
|
|
85
|
+
work/tuning. `benchmarks/holdout.example.json` contains the full schema.
|
|
86
|
+
|
|
87
|
+
Repository paths are resolved relative to the manifest. Pin exact Git commits so
|
|
88
|
+
the corpus cannot move between runs. A publishable holdout also records a freeze
|
|
89
|
+
timestamp and a SHA-256 of the task/repository ground-truth definition:
|
|
90
|
+
|
|
91
|
+
```json
|
|
92
|
+
{
|
|
93
|
+
"suite_version": 1,
|
|
94
|
+
"protocol": {
|
|
95
|
+
"ground_truth_frozen": true,
|
|
96
|
+
"development_excluded": true,
|
|
97
|
+
"frozen_at": "2026-09-17T12:00:00Z",
|
|
98
|
+
"ground_truth_sha256": "HASH_FROM_COMMAND_BELOW"
|
|
99
|
+
},
|
|
100
|
+
"repositories": {
|
|
101
|
+
"app-a": {"path": "../app-a", "revision": "ACTUAL_COMMIT_SHA"},
|
|
102
|
+
"app-b": {"path": "../app-b", "revision": "ACTUAL_COMMIT_SHA"}
|
|
103
|
+
},
|
|
104
|
+
"tasks": [
|
|
105
|
+
{
|
|
106
|
+
"id": "auth-refresh",
|
|
107
|
+
"repository": "app-a",
|
|
108
|
+
"query": "session refresh after logout",
|
|
109
|
+
"files": ["src/auth/session.ts"],
|
|
110
|
+
"symbols": ["refreshSession"]
|
|
111
|
+
}
|
|
112
|
+
]
|
|
113
|
+
}
|
|
114
|
+
```
|
|
115
|
+
|
|
116
|
+
After the tasks, expected evidence, and revision pins are final, calculate the
|
|
117
|
+
freeze hash without running retrieval:
|
|
118
|
+
|
|
119
|
+
```bash
|
|
120
|
+
acco evaluate benchmarks/holdout.json --print-ground-truth-hash
|
|
121
|
+
```
|
|
122
|
+
|
|
123
|
+
Put that value in `protocol.ground_truth_sha256`, commit the manifest, then run:
|
|
124
|
+
|
|
125
|
+
```bash
|
|
126
|
+
acco evaluate benchmarks/holdout.json --require-holdout --max-tokens 6000
|
|
127
|
+
```
|
|
128
|
+
|
|
129
|
+
`--require-holdout` rejects a manifest unless both protocol flags are true,
|
|
130
|
+
`frozen_at` is present, the SHA-256 still matches the frozen task definition,
|
|
131
|
+
and every pinned repository `HEAD` matches its declared revision. Filesystem
|
|
132
|
+
paths are excluded from the freeze hash so the same manifest can be replicated
|
|
133
|
+
on another machine without changing the benchmark identity. The output includes
|
|
134
|
+
aggregate metrics, per-repository summaries, and the verified hash.
|
|
135
|
+
|
|
136
|
+
A valid hash proves the evaluated definition did not change after it was frozen;
|
|
137
|
+
it does not by itself prove the labels were independently authored before tuning.
|
|
138
|
+
Preserve manifest history and the task-definition process as audit evidence.
|
|
139
|
+
|
|
140
|
+
### Frozen semantic holdout #13
|
|
141
|
+
|
|
142
|
+
Version 1.10 includes a fresh **no-identifier-leakage semantic holdout** for
|
|
143
|
+
measuring whether hybrid chunk retrieval improves natural-language file
|
|
144
|
+
discovery beyond ACCO's lexical/structural pipeline.
|
|
145
|
+
|
|
146
|
+
The benchmark uses 24 behavior descriptions from public upstream issues across
|
|
147
|
+
six repositories that were not used in external holdouts #1-#12:
|
|
148
|
+
|
|
149
|
+
- Python: `Kludex/uvicorn`;
|
|
150
|
+
- Go: `spf13/afero`;
|
|
151
|
+
- Rust: `rust-lang/regex`;
|
|
152
|
+
- Java: `resilience4j/resilience4j`;
|
|
153
|
+
- JavaScript: `fastify/fastify`;
|
|
154
|
+
- C#: `dotnet/command-line-api`.
|
|
155
|
+
|
|
156
|
+
The audit sequence is intentionally stronger than a single final-manifest hash.
|
|
157
|
+
The exact 24 queries and repository revisions were committed **before target
|
|
158
|
+
files or fixes were inspected** in
|
|
159
|
+
`benchmarks/semantic-holdout-13.query-freeze.json`.
|
|
160
|
+
|
|
161
|
+
Query-only freeze:
|
|
162
|
+
|
|
163
|
+
- commit: `f3247c1d4c8e388de653aea1a5de4fa4624f82ce`;
|
|
164
|
+
- canonical SHA-256:
|
|
165
|
+
`8cdf871ea2fcbe59161a332f1f0ce0f93b087a6c3560acebadb5ee338c4169bf`.
|
|
166
|
+
|
|
167
|
+
Ground truth was then collected without changing those queries. A post-freeze
|
|
168
|
+
answer-identity audit excluded two tasks from the semantic headline rather than
|
|
169
|
+
rewriting them: one Uvicorn task contains exact target member names
|
|
170
|
+
(`restart` / `shutdown`), and one System.CommandLine task is too close to
|
|
171
|
+
the public `CustomParser` member identity. Both remain visible in the frozen
|
|
172
|
+
manifest with exclusion reasons. The headline cohort is therefore **22 of the
|
|
173
|
+
24 frozen tasks**.
|
|
174
|
+
|
|
175
|
+
Final semantic ground-truth SHA-256:
|
|
176
|
+
|
|
177
|
+
`dc6ea6c3641db573b5b05473f2bc4ee13e0f05a4f093cb86f803e68c7b265d25`
|
|
178
|
+
|
|
179
|
+
Every eligible task is evaluated under the same file-count and token limits
|
|
180
|
+
against three arms:
|
|
181
|
+
|
|
182
|
+
1. **ACCO lexical/structural** — the current validated pipeline with
|
|
183
|
+
semantic retrieval disabled;
|
|
184
|
+
2. **ACCO hybrid semantic** — the same pipeline with persistent
|
|
185
|
+
chunk-level semantic retrieval enabled;
|
|
186
|
+
3. **trivial lexical baseline** — distinct normalized query-term overlap only,
|
|
187
|
+
with no structural authority, fuzzy correction, dependency graph, feedback,
|
|
188
|
+
or semantic evidence.
|
|
189
|
+
|
|
190
|
+
The semantic arm is pinned to `all-MiniLM-L6-v2` revision
|
|
191
|
+
`bc57282bc374d33e0d6c4de27f12dc1c2a87f37a`. Model revision participates in
|
|
192
|
+
ACCO's semantic vector/query-cache identity. The evidence workflow
|
|
193
|
+
explicitly removes `hnswlib`, so the canonical first run uses exact cosine
|
|
194
|
+
over the persistent vectors rather than approximate nearest-neighbor search.
|
|
195
|
+
|
|
196
|
+
The first real evaluation ran once in GitHub Actions
|
|
197
|
+
**35537362040** and permanently burned this suite for tuning. The exact-cosine
|
|
198
|
+
result over the 22 eligible tasks was:
|
|
199
|
+
|
|
200
|
+
| Arm | File recall |
|
|
201
|
+
| --- | ---: |
|
|
202
|
+
| ACCO hybrid semantic | **50.00% (11/22)** |
|
|
203
|
+
| ACCO lexical/structural | **45.45% (10/22)** |
|
|
204
|
+
| Trivial lexical baseline | **40.91% (9/22)** |
|
|
205
|
+
|
|
206
|
+
Hybrid semantic retrieval recovered one task missed by ACCO's lexical
|
|
207
|
+
arm and regressed none. Mean estimated context reduction was effectively flat
|
|
208
|
+
(97.8374% vs 97.8373%).
|
|
209
|
+
|
|
210
|
+
The result is useful precisely because it is not inflated: the semantic
|
|
211
|
+
mechanism shows a real but small improvement, while several repositories still
|
|
212
|
+
have difficult behavior-only misses. Holdout #13 must not be used as the tuning
|
|
213
|
+
loop for those misses.
|
|
214
|
+
|
|
215
|
+
Fresh holdout #14 was subsequently frozen before evaluation with 24
|
|
216
|
+
issue-derived natural-language tasks across six repositories; four were
|
|
217
|
+
conservatively excluded after ground-truth review, leaving 20 eligible tasks.
|
|
218
|
+
Its first complete exact-cosine run was GitHub Actions **35615316639**:
|
|
219
|
+
|
|
220
|
+
| Arm | File recall |
|
|
221
|
+
| --- | ---: |
|
|
222
|
+
| ACCO hybrid semantic | **82.50%** |
|
|
223
|
+
| ACCO lexical/structural | **80.00%** |
|
|
224
|
+
| Trivial lexical baseline | **70.00%** |
|
|
225
|
+
|
|
226
|
+
The semantic arm improved aggregate file recall by 2.5 percentage points with
|
|
227
|
+
zero regressions. One two-file target was partially recovered, so the run
|
|
228
|
+
reported zero complete `semantic_recovered_tasks`. Mean estimated context
|
|
229
|
+
reduction remained essentially identical (98.9516% semantic vs 98.9515%
|
|
230
|
+
lexical). That first run burned #14.
|
|
231
|
+
|
|
232
|
+
Any rerun after tuning against #14 is development evidence only. In particular,
|
|
233
|
+
the later 87.5% semantic development result must not be reported as fresh
|
|
234
|
+
generalization evidence. The next independent cohort is semantic holdout #15: its queries and
|
|
235
|
+
pinned repository revisions are frozen, but ground truth and the first
|
|
236
|
+
evaluation are still pending.
|
|
237
|
+
|
|
238
|
+
These are retrieval benchmarks. File-recall improvement does not by itself
|
|
239
|
+
establish lower API cost, coding-task success, or cost per successful task.
|
|
240
|
+
|
|
241
|
+
## Automated end-to-end cost-per-success experiment
|
|
242
|
+
|
|
243
|
+
For evidence that supports a public cost claim, use the executable experiment
|
|
244
|
+
harness rather than hand-assembling a few runs. It exports history-isolated snapshots at pinned revisions, randomizes
|
|
245
|
+
baseline/enabled order deterministically, runs multiple trials, applies Token
|
|
246
|
+
Saver only in the enabled arm, executes the task's verifier outside the agent,
|
|
247
|
+
captures the real Claude Code transcript, and checkpoints after every run so an
|
|
248
|
+
interrupted paid experiment can resume safely. The exported snapshot is
|
|
249
|
+
re-initialized as a one-commit Git repository, so an agent cannot recover the
|
|
250
|
+
historical gold fix from later commits or remote branches.
|
|
251
|
+
|
|
252
|
+
The publication gate deliberately requires **at least 20 distinct tasks and at
|
|
253
|
+
least 3 paired trials per task**. A 20-task suite therefore means 120 agent runs
|
|
254
|
+
(20 tasks × 3 trials × 2 conditions). Prefer 20–50 historical real-world bug
|
|
255
|
+
fixes/refactors across several repositories and task types rather than many near-
|
|
256
|
+
duplicates from one project.
|
|
257
|
+
|
|
258
|
+
Start from `benchmarks/e2e-suite.example.json`. Each task must pin a repository
|
|
259
|
+
revision, preserve the exact prompt (plus its SHA-256), and specify one or more
|
|
260
|
+
independent verifier commands. Do not use the agent's own "done" statement as
|
|
261
|
+
the success label.
|
|
262
|
+
|
|
263
|
+
A production-ready frozen suite is checked in as
|
|
264
|
+
`benchmarks/e2e-swebench-24.frozen.json`. It contains 24 historical SWE-bench
|
|
265
|
+
Verified issues across scikit-learn, pytest, Astropy, Pylint, Requests, Xarray,
|
|
266
|
+
and Seaborn, with three paired trials per task (**144 agent runs**). The public
|
|
267
|
+
repository URL, task revision, prompt, hidden regression patch, official
|
|
268
|
+
SWE-bench evaluation image, and canonical test command are frozen into the suite
|
|
269
|
+
hash; only the machine-local clone path is excluded.
|
|
270
|
+
|
|
271
|
+
Prepare its external repositories without checking them into this repository:
|
|
272
|
+
|
|
273
|
+
```bash
|
|
274
|
+
python scripts/prepare_e2e_repos.py benchmarks/e2e-swebench-24.frozen.json
|
|
275
|
+
```
|
|
276
|
+
|
|
277
|
+
For these tasks, the hidden regression patch is not present while the coding
|
|
278
|
+
agent runs. After the agent exits, its diff is captured, then a fresh official
|
|
279
|
+
SWE-bench Docker image applies the agent patch and hidden test patch and executes
|
|
280
|
+
the canonical test command inside the benchmark's prepared environment.
|
|
281
|
+
|
|
282
|
+
A run is graded per test, not by the exit code of the whole command, because
|
|
283
|
+
official images can carry pre-existing errors (for example broken fixtures)
|
|
284
|
+
that make it nonzero even for a correct fix. A run is **resolved** when every
|
|
285
|
+
`source.fail_to_pass` test passes and no test that passed on the unpatched
|
|
286
|
+
reference regressed. The reference is the same image with the hidden tests
|
|
287
|
+
applied and no agent patch; it is computed once per task and cached under
|
|
288
|
+
`<out>.artifacts/_reference/`. Agent edits to files the hidden test patch
|
|
289
|
+
modifies are removed before verification (the full patch stays recorded as
|
|
290
|
+
`agent.patch`; the verified one is `agent.graded.patch`), so an agent that also
|
|
291
|
+
edits a test file cannot make the hidden tests fail to apply.
|
|
292
|
+
|
|
293
|
+
Before any paid run, finalize the task definitions and experimental design, then
|
|
294
|
+
freeze them:
|
|
295
|
+
|
|
296
|
+
```bash
|
|
297
|
+
acco experiment benchmarks/e2e-suite.json \
|
|
298
|
+
--print-task-definition-hash
|
|
299
|
+
```
|
|
300
|
+
|
|
301
|
+
Copy that SHA-256 into `protocol.task_definition_sha256`, set `frozen_at`,
|
|
302
|
+
commit the suite, and inspect the randomized schedule without calling a model:
|
|
303
|
+
|
|
304
|
+
```bash
|
|
305
|
+
acco experiment benchmarks/e2e-suite.json \
|
|
306
|
+
--out benchmark-runs.json \
|
|
307
|
+
--dry-run
|
|
308
|
+
```
|
|
309
|
+
|
|
310
|
+
Run the experiment:
|
|
311
|
+
|
|
312
|
+
```bash
|
|
313
|
+
acco experiment benchmarks/e2e-suite.json \
|
|
314
|
+
--out benchmark-runs.json
|
|
315
|
+
```
|
|
316
|
+
|
|
317
|
+
The baseline arm exports `ACCO_DISABLED=1`, which makes any inherited
|
|
318
|
+
ACCO hook a true no-op. The enabled arm installs project-local hooks.
|
|
319
|
+
Both arms receive the same history-isolated source snapshot, exact prompt, model,
|
|
320
|
+
turn limit, and verifier. Hidden SWE-bench regression tests are applied only
|
|
321
|
+
after the agent process has ended.
|
|
322
|
+
To avoid double instrumentation, the runner refuses to start when it detects a
|
|
323
|
+
user-level `acco hook`; use a clean host configuration for publishable
|
|
324
|
+
runs rather than bypassing that guard.
|
|
325
|
+
|
|
326
|
+
For development/smoke experiments with fewer tasks or an unfrozen suite, pass
|
|
327
|
+
`--allow-development`. Those runs are intentionally rejected by the publication
|
|
328
|
+
gate.
|
|
329
|
+
|
|
330
|
+
After the runs finish, price the exact recorded model usage and require the broad
|
|
331
|
+
protocol:
|
|
332
|
+
|
|
333
|
+
```bash
|
|
334
|
+
acco benchmark benchmark-runs.json \
|
|
335
|
+
--rates rates.json \
|
|
336
|
+
--require-publishable
|
|
337
|
+
```
|
|
338
|
+
|
|
339
|
+
The result includes success rates, failed-run cost, cost per success, per-task
|
|
340
|
+
improvements/regressions, and deterministic **95% task-cluster bootstrap
|
|
341
|
+
intervals**. Repeated trials are resampled as one task cluster; three trials of
|
|
342
|
+
one task are not treated as three independent tasks. The report separately marks
|
|
343
|
+
whether the frozen protocol is valid, whether a quality-parity cost claim is
|
|
344
|
+
allowed, and whether the 95% interval for cost-per-success reduction is entirely
|
|
345
|
+
above zero.
|
|
346
|
+
|
|
347
|
+
Raw transcripts and verifier outputs can contain source code or secrets. Keep
|
|
348
|
+
them private when needed; the checked-in suite definition, revision pins, prompt
|
|
349
|
+
hashes, verifier definitions, rates, and aggregate result are sufficient to make
|
|
350
|
+
the experimental design auditable.
|
|
351
|
+
|
|
352
|
+
### v1.13 optimization-platform treatment exposure
|
|
353
|
+
|
|
354
|
+
The v1.13 recovery/schema/prefix/proxy/browser/optimizer additions are not
|
|
355
|
+
assigned a new savings percentage merely because their mechanism tests pass.
|
|
356
|
+
A publishable bundle experiment must keep the task, repository revision, model,
|
|
357
|
+
turn/tool limits, verifier, and pricing source identical between paired arms.
|
|
358
|
+
|
|
359
|
+
For the broad platform bundle, the control should use the same installed Token
|
|
360
|
+
Saver binary with the new treatment surfaces disabled or left at their
|
|
361
|
+
backward-compatible defaults. The treatment may enable adaptive MCP disclosure,
|
|
362
|
+
recoverable schema compression, provider request transformation, and other
|
|
363
|
+
declared v1.13 surfaces. **Do not change the model between arms** when the goal
|
|
364
|
+
is to measure this bundle; model-routing savings require their own calibrated
|
|
365
|
+
experiment or a design that explicitly isolates model choice.
|
|
366
|
+
|
|
367
|
+
The experiment artifact must prove treatment exposure rather than assuming that
|
|
368
|
+
configuration implies use. At minimum, report:
|
|
369
|
+
|
|
370
|
+
- MCP profile/tool-list exposure and whether schema compression actually changed
|
|
371
|
+
an advertised catalog;
|
|
372
|
+
- provider transform activation and before/after request-token estimates;
|
|
373
|
+
- stable-prefix hit/miss counters when prefix tracking is part of the treatment;
|
|
374
|
+
- recovery handles emitted and successful exact-recovery spot checks;
|
|
375
|
+
- memory/tool-result/browser optimizations only on tasks where those surfaces
|
|
376
|
+
were actually exercised;
|
|
377
|
+
- total tool calls, repeated Reads/commands, provider input/cache/output usage,
|
|
378
|
+
latency, verifier success, and blind response quality.
|
|
379
|
+
|
|
380
|
+
A feature that never activates is not evidence for that feature. Bundle-level
|
|
381
|
+
cost-per-success may still be measured when the randomized treatment is the
|
|
382
|
+
whole declared platform, but the report must preserve per-feature activation so
|
|
383
|
+
readers can distinguish “enabled” from “used.”
|
|
384
|
+
|
|
385
|
+
The existing publication gate remains authoritative: enough distinct tasks and
|
|
386
|
+
paired trials, independent verification, blind quality parity, complete
|
|
387
|
+
model/cache-aware pricing evidence, and a strictly positive task-cluster 95%
|
|
388
|
+
confidence-interval lower bound for cost-per-success reduction. There is
|
|
389
|
+
currently **no completed publishable v1.13 bundle experiment** in the repository.
|
|
390
|
+
|
|
391
|
+
## Paired agent outcomes
|
|
392
|
+
|
|
393
|
+
Record independently validated baseline and ACCO runs using the schema in
|
|
394
|
+
`benchmarks/agent-runs.example.json`, then run:
|
|
395
|
+
|
|
396
|
+
```bash
|
|
397
|
+
acco agent-evaluate benchmarks/agent-runs.json
|
|
398
|
+
```
|
|
399
|
+
|
|
400
|
+
The evaluator pairs runs by `(task, trial, condition)`. `trial` defaults to
|
|
401
|
+
`1` for backward-compatible single-pair manifests, but repeated experiments
|
|
402
|
+
should number trials explicitly. It reports input and output tokens, output
|
|
403
|
+
tokens per successful run, retries, elapsed time, context failures, success
|
|
404
|
+
rate, and total tokens per success. It also summarizes paired per-trial
|
|
405
|
+
reductions with median, p10/p90, standard deviation, and a deterministic 95%
|
|
406
|
+
bootstrap confidence interval.
|
|
407
|
+
|
|
408
|
+
Raw token reductions remain measurable without a quality grader, but
|
|
409
|
+
`claim_allowed` is **false unless blind quality evidence is present**, task
|
|
410
|
+
success is at parity, correctness/safety and weighted quality remain within the
|
|
411
|
+
documented tolerance, blockers do not increase, and the per-success denominator
|
|
412
|
+
is available.
|
|
413
|
+
|
|
414
|
+
For generation-time output-policy experiments, the same manifest can carry
|
|
415
|
+
optional **blind response-quality evidence**. Score both conditions on identical
|
|
416
|
+
tasks/trials after relabelling them so the grader cannot see which response came
|
|
417
|
+
from ACCO. The built-in rubric weights correctness 40%, completeness
|
|
418
|
+
20%, actionability 15%, safety 15%, and concision 10%.
|
|
419
|
+
|
|
420
|
+
```json
|
|
421
|
+
{
|
|
422
|
+
"quality_evaluation": {
|
|
423
|
+
"blinded": true,
|
|
424
|
+
"judge": "independent-response-grader"
|
|
425
|
+
},
|
|
426
|
+
"runs": [
|
|
427
|
+
{
|
|
428
|
+
"task": "fix-session-refresh",
|
|
429
|
+
"trial": 1,
|
|
430
|
+
"condition": "baseline",
|
|
431
|
+
"success": true,
|
|
432
|
+
"input_tokens": 18000,
|
|
433
|
+
"output_tokens": 1200,
|
|
434
|
+
"quality": {
|
|
435
|
+
"correctness": 5,
|
|
436
|
+
"completeness": 5,
|
|
437
|
+
"actionability": 4,
|
|
438
|
+
"safety": 5,
|
|
439
|
+
"concision": 3
|
|
440
|
+
},
|
|
441
|
+
"blocker": false
|
|
442
|
+
}
|
|
443
|
+
]
|
|
444
|
+
}
|
|
445
|
+
```
|
|
446
|
+
|
|
447
|
+
When quality scores are supplied, every paired run must be scored. ACCO
|
|
448
|
+
then requires task-success parity, no material correctness/safety regression,
|
|
449
|
+
no increase in blockers, and weighted-quality parity before authorizing a
|
|
450
|
+
savings claim. `raw_output_token_reduction` remains a descriptive measurement;
|
|
451
|
+
`output_tokens_per_success_reduction` is the stronger generated-output metric,
|
|
452
|
+
and `tokens_per_success_reduction` measures total input + output efficiency per
|
|
453
|
+
successful run. Unblinded or missing quality evidence is reported through
|
|
454
|
+
`claim_blockers` and cannot authorize a savings claim.
|
|
455
|
+
|
|
456
|
+
For a publishable output-cost claim, use at least 20 distinct frozen tasks and
|
|
457
|
+
three randomized paired trials per task, keep the model/prompt/tool settings
|
|
458
|
+
identical, independently verify task success, and report output-token reduction
|
|
459
|
+
next to cost per success rather than treating response length alone as quality.
|
|
460
|
+
|
|
461
|
+
### Automated end-to-end evidence pipeline
|
|
462
|
+
|
|
463
|
+
The high-level production path is now:
|
|
464
|
+
|
|
465
|
+
```bash
|
|
466
|
+
acco evidence-run benchmarks/e2e-swebench-24.frozen.json \
|
|
467
|
+
--out benchmark-runs.json \
|
|
468
|
+
--require-publishable
|
|
469
|
+
```
|
|
470
|
+
|
|
471
|
+
The command is checkpointed across paid agent runs and judge calls. It runs the
|
|
472
|
+
frozen randomized experiment, executes independent hidden verification, extracts
|
|
473
|
+
only final assistant response text for a balanced deterministic blind A/B judge,
|
|
474
|
+
writes quality scores back into the same run identities, evaluates exact
|
|
475
|
+
cache-TTL-aware cost per successful task, and emits an adaptive-budget
|
|
476
|
+
calibration artifact. The judge never receives baseline/ACCO labels,
|
|
477
|
+
patches, verifier outcomes, or billing data.
|
|
478
|
+
|
|
479
|
+
The repository also ships a paid GitHub workflow for the existing **24-task ×
|
|
480
|
+
3-trial SWE-bench Verified suite**. A smoke pair must first prove real agent
|
|
481
|
+
usage, output-policy telemetry, blind grading, and pricing. Only then does the
|
|
482
|
+
24-task matrix run. Aggregation blind-grades all 72 pairs and enforces the
|
|
483
|
+
publishability gate. The workflow requires an explicit paid-run confirmation;
|
|
484
|
+
merging the feature alone is not a savings result.
|
|
485
|
+
|
|
486
|
+
### Joined output-effectiveness evidence
|
|
487
|
+
|
|
488
|
+
The paired experiment artifact now carries transcript-measured fresh input,
|
|
489
|
+
cache creation, cache read, output tokens, and model calls for both arms.
|
|
490
|
+
Enabled runs also carry the selected output task/mode/budget when Claude hooks
|
|
491
|
+
emit policy telemetry. After blind response grading has been attached to the
|
|
492
|
+
same task/trial records, run:
|
|
493
|
+
|
|
494
|
+
```bash
|
|
495
|
+
acco output-effectiveness benchmark-runs.json \
|
|
496
|
+
--fresh-input-per-million <rate> \
|
|
497
|
+
--cache-creation-5m-per-million <rate> \
|
|
498
|
+
--cache-creation-1h-per-million <rate> \
|
|
499
|
+
--cache-creation-unknown-per-million <rate> \
|
|
500
|
+
--cache-read-per-million <rate> \
|
|
501
|
+
--output-per-million <rate> \
|
|
502
|
+
--require-publishable
|
|
503
|
+
```
|
|
504
|
+
|
|
505
|
+
The publication gate requires the existing broad design (>=20 distinct frozen
|
|
506
|
+
tasks and >=3 trials/task), recomputes and verifies the frozen task-definition
|
|
507
|
+
SHA-256, and checks each run's model/revision/prompt hash against that frozen
|
|
508
|
+
suite. It also requires no task-success regression, blind correctness/safety
|
|
509
|
+
and weighted-quality parity, complete optimized-arm policy telemetry with a
|
|
510
|
+
measured task/mode/budget, exact telemetry/transcript usage agreement, and
|
|
511
|
+
complete cost evidence. Cache-creation usage is split into 5-minute, 1-hour,
|
|
512
|
+
and unknown-TTL buckets; any nonzero bucket without a supplied rate makes
|
|
513
|
+
derived cost incomplete. A positive point estimate is not enough: the 95%
|
|
514
|
+
task-cluster bootstrap interval for cost-per-success reduction must remain
|
|
515
|
+
strictly above zero. Repeated trials are resampled as one task cluster rather
|
|
516
|
+
than treated as independent evidence.
|
|
517
|
+
|
|
518
|
+
This is the preferred end-to-end output-cost claim surface. Raw context
|
|
519
|
+
reduction, response length, or a low budget-utilization ratio are not
|
|
520
|
+
substitutes for cost per independently verified successful task.
|
|
521
|
+
|
|
522
|
+
### Frozen session-efficiency causal holdout
|
|
523
|
+
|
|
524
|
+
Operational dashboard savings are not enough to establish that continuity,
|
|
525
|
+
cross-turn dedup, or waste prevention improve coding-agent economics. Version
|
|
526
|
+
1.8 therefore adds a separate frozen paired-agent holdout:
|
|
527
|
+
|
|
528
|
+
```bash
|
|
529
|
+
acco session-holdout \
|
|
530
|
+
benchmarks/session-efficiency-swebench-24.frozen.json \
|
|
531
|
+
--out session-holdout-runs.json \
|
|
532
|
+
--require-publishable
|
|
533
|
+
```
|
|
534
|
+
|
|
535
|
+
The suite reuses the already frozen 24 SWE-bench Verified tasks and hidden
|
|
536
|
+
verification definitions, with three randomized trials per task. The control is
|
|
537
|
+
**not a historical 1.6 executable**. Both arms install the same current Token
|
|
538
|
+
Saver build; the baseline sets only the four session-efficiency controls to
|
|
539
|
+
zero, while treatment sets them to one. This holds retrieval, output processors,
|
|
540
|
+
model, host adapter, prompt, revision, and grader constant.
|
|
541
|
+
|
|
542
|
+
Every arm uses the same two-session protocol:
|
|
543
|
+
|
|
544
|
+
1. phase 1 receives the frozen task and is restricted to Read/Grep/Glob/Bash;
|
|
545
|
+
2. phase 1 must leave benchmark-visible repository state unchanged;
|
|
546
|
+
3. the runner invokes ACCO's actual `SessionStart:resume` hook;
|
|
547
|
+
4. phase 2 starts in a new Claude home/session and implements/verifies the task;
|
|
548
|
+
5. treatment may receive the structured continuity checkpoint; control cannot.
|
|
549
|
+
|
|
550
|
+
That forced boundary gives continuity a deterministic opportunity to affect the
|
|
551
|
+
second session without feeding phase-1 prose directly to phase 2.
|
|
552
|
+
|
|
553
|
+
#### Independent outcome metrics
|
|
554
|
+
|
|
555
|
+
The evaluator derives behavioral outcomes from raw Claude transcripts for both
|
|
556
|
+
arms:
|
|
557
|
+
|
|
558
|
+
- total tool calls;
|
|
559
|
+
- total input tokens, with exact cache fields retained for pricing;
|
|
560
|
+
- normalized repeated Bash calls;
|
|
561
|
+
- repeated identical failing Bash-result attempts;
|
|
562
|
+
- duplicate full-file Reads;
|
|
563
|
+
- independent task success;
|
|
564
|
+
- blind final-response quality;
|
|
565
|
+
- cache-TTL-aware cost per successful task.
|
|
566
|
+
|
|
567
|
+
Treatment-side ACCO efficiency events are kept in a separate
|
|
568
|
+
`feature_activation` block. They prove whether continuity, dedup, and waste
|
|
569
|
+
signals fired, but they are not substituted for outcome metrics.
|
|
570
|
+
|
|
571
|
+
#### Statistical design
|
|
572
|
+
|
|
573
|
+
The frozen suite contains 24 task clusters and 3 trials/task. Point estimates
|
|
574
|
+
are accompanied by deterministic 2,000-sample task-cluster bootstrap 95%
|
|
575
|
+
intervals for tool-call reduction, input-token reduction, retry reduction, and
|
|
576
|
+
cost-per-success reduction. Trials from the same task are sampled together to
|
|
577
|
+
avoid treating repeated trials as independent tasks.
|
|
578
|
+
|
|
579
|
+
The publication gate requires:
|
|
580
|
+
|
|
581
|
+
- at least 20 distinct tasks and 3 trials/task;
|
|
582
|
+
- exact frozen definition hash and isolated condition profiles;
|
|
583
|
+
- matching model/prompt/revision identity between arms;
|
|
584
|
+
- no manual intervention;
|
|
585
|
+
- independent task-success parity;
|
|
586
|
+
- blind response-quality parity;
|
|
587
|
+
- complete cache-TTL-aware pricing;
|
|
588
|
+
- zero session-efficiency events in control;
|
|
589
|
+
- exactly the forced continuity exposure path, with at least one restore for
|
|
590
|
+
every treatment arm-run;
|
|
591
|
+
- observed dedup, continuity, and waste feature families somewhere in treatment;
|
|
592
|
+
- positive cost-per-success point reduction;
|
|
593
|
+
- cost-per-success 95% task-cluster CI lower bound strictly above zero.
|
|
594
|
+
|
|
595
|
+
The design estimates the **combined session-efficiency bundle**. Feature
|
|
596
|
+
activation counts do not identify each mechanism's individual causal effect; a
|
|
597
|
+
future ablation design would be required for that.
|
|
598
|
+
|
|
599
|
+
The dedicated GitHub workflow requires `RUN_SESSION_288` because the 144
|
|
600
|
+
arm-runs contain 288 paid Claude task phases, before blind-grader calls. A paid
|
|
601
|
+
smoke proves the two-phase runner, control isolation, continuity hook, transcript
|
|
602
|
+
metrics, blind grader, and pricing path before the full matrix starts.
|
|
603
|
+
|
|
604
|
+
No session-efficiency savings percentage should be published from the frozen
|
|
605
|
+
definition alone. A claim starts only after the paid workflow completes and this
|
|
606
|
+
gate passes.
|
|
607
|
+
|
|
608
|
+
### Frozen session-efficiency output-quality gate
|
|
609
|
+
|
|
610
|
+
The session-efficiency layer does not replace retrieval or end-to-end evidence.
|
|
611
|
+
Its expanded command processors have a separate deterministic frozen fixture:
|
|
612
|
+
|
|
613
|
+
```bash
|
|
614
|
+
acco output-replay \
|
|
615
|
+
benchmarks/output-quality-session-v17.frozen.json \
|
|
616
|
+
--require-frozen
|
|
617
|
+
```
|
|
618
|
+
|
|
619
|
+
The fixture definition is SHA-256 frozen and covers search, lint, typecheck,
|
|
620
|
+
compiled tests, build diagnostics, git status, and container logs. Each case can
|
|
621
|
+
require exact preserved evidence, minimum token reduction, and strings that the
|
|
622
|
+
transformer must not introduce. CI runs this gate in addition to the external
|
|
623
|
+
retrieval holdout and base-vs-candidate ranking checks.
|
|
624
|
+
|
|
625
|
+
Cross-turn dedup itself is intentionally simpler than fuzzy compression: it only
|
|
626
|
+
fires when the normalized command and exact output digest match. Unchanged
|
|
627
|
+
full-file Read dedup uses the existing verified read digest. These mechanics are
|
|
628
|
+
covered by deterministic tests; any real end-to-end savings claim still belongs
|
|
629
|
+
to the randomized `evidence-run` protocol with task success and blind quality.
|
|
630
|
+
|
|
631
|
+
The local `dashboard` is also an operational surface, not a benchmark. Its
|
|
632
|
+
tool-context savings are estimated from observed before/after text, while its
|
|
633
|
+
Claude usage counters come from available transcript billing fields. Those
|
|
634
|
+
categories stay separate and are never promoted to cost-per-success evidence.
|
|
635
|
+
|
|
636
|
+
### Runtime budget telemetry
|
|
637
|
+
|
|
638
|
+
For ordinary Claude Code use, `output-telemetry` records actual transcript
|
|
639
|
+
usage counters against the selected policy budget without storing conversation
|
|
640
|
+
content. Use it to discover candidate task/mode groups for future experiments:
|
|
641
|
+
|
|
642
|
+
```bash
|
|
643
|
+
acco output-telemetry . --json
|
|
644
|
+
```
|
|
645
|
+
|
|
646
|
+
A low p90 budget-utilization ratio is only an **observational tuning signal**.
|
|
647
|
+
Do not feed it directly into calibration. A completed `Stop` turn is not proof
|
|
648
|
+
of task success, and telemetry has no blind response-quality score. Validate
|
|
649
|
+
candidate budget changes with paired successful runs before calibrating them.
|
|
650
|
+
|
|
651
|
+
### Calibrating adaptive output budgets
|
|
652
|
+
|
|
653
|
+
ACCO can turn the same blind paired evidence into conservative learned
|
|
654
|
+
task/mode bases. Add `output_task` and `output_mode` to ACCO runs, then:
|
|
655
|
+
|
|
656
|
+
```bash
|
|
657
|
+
acco output-calibrate benchmarks/agent-runs.json \
|
|
658
|
+
--out .acco.output-calibration.json
|
|
659
|
+
```
|
|
660
|
+
|
|
661
|
+
The calibrator considers only pairs where baseline and ACCO both succeed,
|
|
662
|
+
the ACCO response has no blocker, and correctness, safety, and weighted
|
|
663
|
+
blind quality remain within the evaluator parity tolerance. At least three valid samples spanning at least three distinct task IDs are
|
|
664
|
+
required per task/mode. The recommendation is p90 observed output
|
|
665
|
+
tokens plus a 15% safety margin, bounded by the mode safety range. Failed,
|
|
666
|
+
unblinded, or degraded short runs therefore cannot train the controller toward
|
|
667
|
+
an artificially small budget.
|
|
668
|
+
|
|
669
|
+
## Live host validation is a separate manual gate
|
|
670
|
+
|
|
671
|
+
Start with the consolidated configuration/index check, then run the deeper host
|
|
672
|
+
transport check before a live host trial:
|
|
673
|
+
|
|
674
|
+
```bash
|
|
675
|
+
acco doctor . --require-ready
|
|
676
|
+
acco host-check . --require-ready
|
|
677
|
+
```
|
|
678
|
+
|
|
679
|
+
This checks project/user hook configuration, probes the host executable version,
|
|
680
|
+
runs a synthetic 500-line Bash response through the real PostToolUse hook, and
|
|
681
|
+
retrieves an omitted middle line from ACCO's saved original output. That
|
|
682
|
+
is a local transport test; it does **not** prove the host actually feeds
|
|
683
|
+
`hookSpecificOutput.updatedToolOutput` back to the model.
|
|
684
|
+
|
|
685
|
+
For that final gate, capture a real host debug transcript and supply it explicitly:
|
|
686
|
+
|
|
687
|
+
```bash
|
|
688
|
+
acco host-check . \
|
|
689
|
+
--live-evidence /path/to/claude-debug.log \
|
|
690
|
+
--require-live
|
|
691
|
+
```
|
|
692
|
+
|
|
693
|
+
`live_verified=true` is reported only when the supplied evidence contains both
|
|
694
|
+
the host replacement field and ACCO's filtered-output recovery marker.
|
|
695
|
+
The manual validation protocol remains:
|
|
696
|
+
|
|
697
|
+
1. Record `claude --version`, model ID, configuration, and ACCO version.
|
|
698
|
+
2. Use a disposable project. Install the package and its project hooks. Check
|
|
699
|
+
that inherited user hooks do not run ACCO a second time.
|
|
700
|
+
3. Start Claude Code with debugging enabled. Ask it to run a harmless command
|
|
701
|
+
that prints 500 distinct progress lines. Do not use a command with side effects.
|
|
702
|
+
4. Verify the host accepts `hookSpecificOutput.updatedToolOutput` without a
|
|
703
|
+
validation error. The model-visible result should include the recovery note,
|
|
704
|
+
show the shortened head/tail, and preserve the structured Bash fields.
|
|
705
|
+
Hook stdout alone is not sufficient evidence of host acceptance.
|
|
706
|
+
5. Ask it to retrieve a known omitted middle line using the supplied
|
|
707
|
+
`acco output` command. Verify the exact line without rerunning the
|
|
708
|
+
original command. Test stderr independently and test a failing Jest-style
|
|
709
|
+
output containing a test name, stack location and multiline assertion diff.
|
|
710
|
+
6. Test a >220-line source read. A bounded Read must return actual bytes, and a
|
|
711
|
+
subsequent Edit must succeed. Repeat after `/compact`; prior full-read state
|
|
712
|
+
must not block access to needed source. Test two independent sessions.
|
|
713
|
+
7. Save the version, debug evidence and outcomes. If the installed host does not
|
|
714
|
+
support structured replacement, use manual `filter` until upgraded. Do not
|
|
715
|
+
claim automatic savings based on an ignored hook field.
|
|
716
|
+
|
|
717
|
+
No paid model call or live Claude Code session was executed in the repair
|
|
718
|
+
workspace. Automated tests cover the documented contract, subprocess transport,
|
|
719
|
+
state isolation, retrieval, and accounting. Live behavior remains a separate gate.
|
|
720
|
+
|
|
721
|
+
## Paired task protocol
|
|
722
|
+
|
|
723
|
+
- Choose representative tasks before measuring: bug fixes, refactors, unfamiliar
|
|
724
|
+
repo navigation, noisy passing tests, and failures requiring deep diagnostics.
|
|
725
|
+
- Use independent fresh worktrees at the same commit and the same exact prompt,
|
|
726
|
+
model, effort, tools, system instructions, and approval configuration.
|
|
727
|
+
- In baseline, disable all ACCO hooks (including user-scope hooks). In
|
|
728
|
+
enabled runs, use the release's hooks. Do not add orientation maps to only one
|
|
729
|
+
arm unless that is the specific intervention being tested.
|
|
730
|
+
- Randomize condition order and run multiple trials per task. Record cache
|
|
731
|
+
policy and cold/warm conditions rather than assuming a five-minute TTL.
|
|
732
|
+
- Keep tests/evaluation independent of the agent. Record success, regression
|
|
733
|
+
checks, elapsed time, retries and any manual intervention. A smaller context
|
|
734
|
+
that fails the task is not a win.
|
|
735
|
+
- Preserve each run's complete transcripts, including nested subagent
|
|
736
|
+
transcripts where applicable. Do not reuse a transcript across runs.
|
|
737
|
+
- Keep raw transcripts local; they may contain code or secrets.
|
|
738
|
+
|
|
739
|
+
`acco benchmark` analyzes recorded runs; it does not execute coding agents
|
|
740
|
+
or certify evaluator outcomes. It verifies paired task/trial identity, revision,
|
|
741
|
+
model and prompt metadata, requires measured usage and explicit prices, and
|
|
742
|
+
rejects incomplete or duplicate runs. Metadata must be recorded honestly by the
|
|
743
|
+
runner; the evaluator does not independently attest your checkout or prompt.
|
|
744
|
+
|
|
745
|
+
## Run manifest
|
|
746
|
+
|
|
747
|
+
Paths are relative to the manifest file. `success` is determined by independent
|
|
748
|
+
validation, not by the agent's own assertion. Use the actual SHA-256 of the prompt.
|
|
749
|
+
|
|
750
|
+
```json
|
|
751
|
+
{
|
|
752
|
+
"runs": [
|
|
753
|
+
{
|
|
754
|
+
"task": "fix-user-lookup",
|
|
755
|
+
"trial": 1,
|
|
756
|
+
"condition": "baseline",
|
|
757
|
+
"revision": "ACTUAL_COMMIT",
|
|
758
|
+
"model": "EXACT_MODEL_ID",
|
|
759
|
+
"prompt_sha256": "ACTUAL_PROMPT_SHA256",
|
|
760
|
+
"success": true,
|
|
761
|
+
"validation": "Independent tests passed; no regressions in required suite",
|
|
762
|
+
"seconds": 120.5,
|
|
763
|
+
"transcripts": ["runs/baseline.jsonl"]
|
|
764
|
+
},
|
|
765
|
+
{
|
|
766
|
+
"task": "fix-user-lookup",
|
|
767
|
+
"trial": 1,
|
|
768
|
+
"condition": "enabled",
|
|
769
|
+
"revision": "ACTUAL_COMMIT",
|
|
770
|
+
"model": "EXACT_MODEL_ID",
|
|
771
|
+
"prompt_sha256": "ACTUAL_PROMPT_SHA256",
|
|
772
|
+
"success": true,
|
|
773
|
+
"validation": "Same independent checks passed",
|
|
774
|
+
"seconds": 118.2,
|
|
775
|
+
"transcripts": ["runs/enabled.jsonl"]
|
|
776
|
+
}
|
|
777
|
+
]
|
|
778
|
+
}
|
|
779
|
+
```
|
|
780
|
+
|
|
781
|
+
The times above demonstrate the schema; they are not measured results.
|
|
782
|
+
|
|
783
|
+
## Rate file
|
|
784
|
+
|
|
785
|
+
Create a JSON object mapping each exact model ID to numeric USD-per-million
|
|
786
|
+
rates: `input`, `cache_write_5m`, `cache_write_1h`, `cache_read`, `output`.
|
|
787
|
+
For example, this object is syntactically valid but deliberately uses a fictitious
|
|
788
|
+
model and synthetic prices. **It must not be used to price actual models.**
|
|
789
|
+
|
|
790
|
+
```json
|
|
791
|
+
{
|
|
792
|
+
"synthetic-test-model": {
|
|
793
|
+
"input": 1,
|
|
794
|
+
"cache_write_5m": 1.25,
|
|
795
|
+
"cache_write_1h": 2,
|
|
796
|
+
"cache_read": 0.1,
|
|
797
|
+
"output": 5
|
|
798
|
+
}
|
|
799
|
+
}
|
|
800
|
+
```
|
|
801
|
+
|
|
802
|
+
Only add `cache_write_unknown` when the run's actual cache configuration permits
|
|
803
|
+
an explicit rate for missing TTL details. Missing prices or unexplained cache
|
|
804
|
+
creation prevent a complete cost result.
|
|
805
|
+
|
|
806
|
+
```bash
|
|
807
|
+
acco benchmark runs.json --rates rates.json
|
|
808
|
+
```
|
|
809
|
+
|
|
810
|
+
The output includes per-condition success rates, total tokens by usage type,
|
|
811
|
+
tool-result counts, repeated reads, elapsed time, total cost, and cost per success.
|
|
812
|
+
Costs from failed attempts are included. A reduced success rate suppresses the
|
|
813
|
+
headline cost-per-success reduction. Inspect per-task outcomes too: aggregate
|
|
814
|
+
success parity does not prove each task retained the same quality.
|
|
815
|
+
|
|
816
|
+
For the automated experiment path above, `acco benchmark` reports a
|
|
817
|
+
deterministic task-cluster bootstrap 95% interval. This is still an empirical
|
|
818
|
+
benchmark, not a proof that savings generalize to every repository or model.
|
|
819
|
+
Before publishing savings, retain the frozen task definitions, repeated trials,
|
|
820
|
+
model and host versions, prices, independent checks, and raw results needed to
|
|
821
|
+
reproduce the claim.
|
|
822
|
+
|
|
823
|
+
|
|
824
|
+
## Frozen knowledge-efficiency holdout
|
|
825
|
+
|
|
826
|
+
Knowledge-assisted read avoidance is evaluated separately from retrieval recall
|
|
827
|
+
and from the existing session-efficiency bundle. The frozen definition is:
|
|
828
|
+
|
|
829
|
+
```text
|
|
830
|
+
benchmarks/knowledge-efficiency-swebench-24.frozen.json
|
|
831
|
+
24 SWE-bench Verified tasks
|
|
832
|
+
3 trials per task
|
|
833
|
+
2 conditions
|
|
834
|
+
= 144 arm-runs
|
|
835
|
+
= 288 fresh Claude task phases
|
|
836
|
+
```
|
|
837
|
+
|
|
838
|
+
The causal contract intentionally equalizes memory creation. Phase 1 is
|
|
839
|
+
investigation-only in both arms and requires the agent to persist 1–3
|
|
840
|
+
`verified` ACCO findings backed by source it actually inspected.
|
|
841
|
+
Repository state must remain unchanged. Phase 2 uses a fresh Claude home/session
|
|
842
|
+
and receives no conversation transcript or continuity checkpoint.
|
|
843
|
+
|
|
844
|
+
The control and treatment install the **same current ACCO binary**.
|
|
845
|
+
Both disable continuity, cross-turn command/read dedup, reread blocking, and
|
|
846
|
+
behavioral waste detection. The control disables knowledge read avoidance and
|
|
847
|
+
cache economics; the treatment enables those two switches. This tests the
|
|
848
|
+
combined **knowledge read-avoidance + cache gate** effect without attributing
|
|
849
|
+
savings to unrelated 1.7 session features.
|
|
850
|
+
|
|
851
|
+
Outcome metrics come from independent evidence:
|
|
852
|
+
|
|
853
|
+
- raw transcripts: total tool calls, input tokens, and duplicate Reads;
|
|
854
|
+
- hidden task verifier: task success;
|
|
855
|
+
- blinded judge: response quality parity;
|
|
856
|
+
- transcript cache-usage fields plus frozen rates: billed cost/cost per success;
|
|
857
|
+
- ACCO's efficiency ledger: feature exposure only (knowledge seeds,
|
|
858
|
+
read-avoidance interventions, and cache-economics-approved interventions).
|
|
859
|
+
|
|
860
|
+
The ledger never grades its own success. A publishable claim requires at least
|
|
861
|
+
20 tasks, three trials per task, all paired identities intact, no manual
|
|
862
|
+
intervention, success parity, blind-quality verification, complete pricing,
|
|
863
|
+
verified knowledge seeding in every arm-run, zero control avoidance activation,
|
|
864
|
+
observed treatment activation, positive cost-per-success reduction, and a
|
|
865
|
+
task-cluster 95% confidence interval with a strictly positive lower bound.
|
|
866
|
+
|
|
867
|
+
Run the frozen preflight without paid execution:
|
|
868
|
+
|
|
869
|
+
```bash
|
|
870
|
+
acco knowledge-holdout \
|
|
871
|
+
benchmarks/knowledge-efficiency-swebench-24.frozen.json \
|
|
872
|
+
--out /tmp/knowledge-holdout-runs.json \
|
|
873
|
+
--dry-run
|
|
874
|
+
```
|
|
875
|
+
|
|
876
|
+
The paid workflow is
|
|
877
|
+
`.github/workflows/knowledge-efficiency-holdout.yml` and requires the explicit
|
|
878
|
+
`RUN_KNOWLEDGE_288` confirmation. Until it completes successfully, the
|
|
879
|
+
mechanism has **no publishable end-to-end savings percentage**.
|
|
880
|
+
|