drydock-cli 3.1.23__tar.gz → 3.1.27__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {drydock_cli-3.1.23/drydock_cli.egg-info → drydock_cli-3.1.27}/PKG-INFO +1 -1
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/drydock/agent.py +25 -1
- drydock_cli-3.1.27/drydock/bottleneck.py +218 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/drydock/cli.py +10 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/drydock/config.py +14 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/drydock/graphrag.py +20 -0
- drydock_cli-3.1.27/drydock/groundtruth.py +167 -0
- drydock_cli-3.1.27/drydock/predictions.py +177 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/drydock/ratchet.py +37 -0
- drydock_cli-3.1.27/drydock/swarm.py +1010 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/drydock/tools/__init__.py +191 -31
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/drydock/tui/app.py +145 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27/drydock_cli.egg-info}/PKG-INFO +1 -1
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/drydock_cli.egg-info/SOURCES.txt +9 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/pyproject.toml +1 -1
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_bash_background.py +42 -0
- drydock_cli-3.1.27/tests/test_bottleneck.py +103 -0
- drydock_cli-3.1.27/tests/test_groundtruth_predictions.py +124 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_jobs.py +23 -0
- drydock_cli-3.1.27/tests/test_knowledge_tool_available.py +56 -0
- drydock_cli-3.1.27/tests/test_ratchet_offer.py +64 -0
- drydock_cli-3.1.27/tests/test_swarm.py +426 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/LICENSE +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/NOTICE +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/README.md +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/drydock/__init__.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/drydock/__main__.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/drydock/advisor.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/drydock/bash_safety.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/drydock/budget.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/drydock/builtin_skills/__init__.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/drydock/builtin_skills/document-canvas.md +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/drydock/builtin_skills/fiar-assess.md +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/drydock/builtin_skills/fiar-cap.md +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/drydock/builtin_skills/fiar-evidence.md +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/drydock/builtin_skills/fiar-readiness.md +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/drydock/builtin_skills/ml-data.md +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/drydock/builtin_skills/ml-debug.md +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/drydock/builtin_skills/ml-finetune.md +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/drydock/builtin_skills/ml-metrics.md +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/drydock/builtin_skills/ml-rl.md +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/drydock/builtin_skills/ml-train.md +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/drydock/builtin_skills/nist-ai-rmf.md +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/drydock/builtin_skills/nist-csf.md +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/drydock/builtin_skills/rmf-categorize.md +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/drydock/builtin_skills/rmf-control.md +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/drydock/builtin_skills/rmf-poam.md +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/drydock/builtin_skills/rmf-review.md +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/drydock/builtin_skills/stig-assess.md +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/drydock/builtin_skills/stig-remediate.md +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/drydock/cci.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/drydock/comms/__init__.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/drydock/comms/attention.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/drydock/comms/channels.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/drydock/comms/events.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/drydock/compaction.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/drydock/detect.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/drydock/doccanvas.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/drydock/eratchet.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/drydock/events.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/drydock/extract.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/drydock/fiar.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/drydock/gittools.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/drydock/guards.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/drydock/jobs.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/drydock/loop_detect.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/drydock/mcp.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/drydock/pdfredbox.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/drydock/phases.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/drydock/poam.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/drydock/progress.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/drydock/providers.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/drydock/recipes.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/drydock/recovery.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/drydock/resume.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/drydock/rmf.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/drydock/rmf_graph.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/drydock/skills.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/drydock/stig.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/drydock/subagents.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/drydock/suggest.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/drydock/task_state.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/drydock/tool_policy.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/drydock/tool_registry.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/drydock/tool_result.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/drydock/tool_select.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/drydock/tool_validate.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/drydock/trajectory.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/drydock/tui/__init__.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/drydock/tui/approval.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/drydock/tui/messages.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/drydock/tui/widgets.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/drydock/tuning.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/drydock/verification.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/drydock/web.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/drydock_cli.egg-info/dependency_links.txt +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/drydock_cli.egg-info/entry_points.txt +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/drydock_cli.egg-info/requires.txt +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/drydock_cli.egg-info/top_level.txt +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/setup.cfg +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_advisor.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_approval.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_back_command.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_bash_binary_output.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_bash_crossplatform.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_bash_output_bounding.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_bash_process_group.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_bash_safety.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_bash_sanitize.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_bash_shell.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_bash_stdin.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_bash_stop_partial.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_bash_timeout_network.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_bash_timeout_param.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_budget.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_cci.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_cli_agents.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_comms.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_compact_command.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_compaction.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_config.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_config_migration.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_context_limit_config.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_context_limit_issue25.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_criteria_coverage.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_cycling_detection.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_declared_deps.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_degenerate_argument.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_detect.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_dispatch.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_doccanvas.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_e2e_connected.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_edit_replace_all.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_edit_thrash_and_compaction.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_effort_governor.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_empty_response.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_eratchet.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_events.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_extract.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_failure_loop.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_fiar.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_first_run_setup.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_gittools.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_glob_edit_edges.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_graphify_example.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_graphrag.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_graphrag_quoted_path.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_graphrag_sqlite.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_grep_and_read_robust.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_guards_and_tools.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_hallucinated_tools.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_leaked_tool_call.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_loop_detect.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_mcp.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_model_registry.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_oneshot_unreachable.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_overthink_interrupt.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_pdfredbox.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_phases.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_plan_autocontinue.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_poam.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_progress.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_providers_unreachable.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_ratchet.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_read_index.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_recipes.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_recovery.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_recovery_config.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_recovery_integration.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_repeated_outcome.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_repetition_interrupt.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_resume_events.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_resume_restore.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_resume_snapshot.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_rmf.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_rmf_graph.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_rmf_stig_graph.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_rolling_plan.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_runaway_repetition.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_screenshot.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_server_probe.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_skills.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_sqlite_events.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_stall_retry.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_stig.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_stop.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_streaming_newlines.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_subagent.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_subagents.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_suggest.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_system_prompt_help.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_task_state.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_timeline.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_todo.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_tool_arg_coercion.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_tool_arg_coercion_more.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_tool_arg_parsing.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_tool_canonical.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_tool_policy.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_tool_result.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_tool_select.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_tool_validate.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_tools_undo.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_trajectory.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_tui.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_tuning.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_verification_gate.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_viewimage.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_vision_input.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_web_tools.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_windows_shell.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_worker_subagent.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_write_content_coerce.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.27}/tests/test_xccdf.py +0 -0
|
@@ -107,6 +107,8 @@ class AgentState:
|
|
|
107
107
|
events: "EventLog | SQLiteEventLog | None" = None # optional durable execution trace
|
|
108
108
|
recovery_stage: int = 0 # live recovery escalation stage (0 = normal); for the TUI
|
|
109
109
|
progress_streak: int = 0 # consecutive no-progress (stall) actions; for the TUI
|
|
110
|
+
verify_fail_streak: int = 0 # consecutive FAILING checks; drives the /ratchet offer
|
|
111
|
+
last_verify_cmd: str = "" # the command behind that streak (pre-fills the offer)
|
|
110
112
|
budget: "BudgetState" = field(default_factory=lambda: BudgetState()) # scoped budgets
|
|
111
113
|
|
|
112
114
|
|
|
@@ -308,12 +310,24 @@ def run(
|
|
|
308
310
|
# relevant to this task + phase, capped at max_tools (default 12).
|
|
309
311
|
# Trims nothing when already under the cap; core coding tools are
|
|
310
312
|
# never dropped.
|
|
313
|
+
# Pin Knowledge when the project HAS a GraphRAG index. Its purpose
|
|
314
|
+
# is to answer plain questions from the user's own corpus, but the
|
|
315
|
+
# keyword gate only fires on knowledge/graph/entity/graphrag/ingest
|
|
316
|
+
# — words a user asking "what does the auth module do?" never types
|
|
317
|
+
# — so it was trimmed and the call came back "not available here".
|
|
318
|
+
_pins = list(turn_config.get("pin_tools") or [])
|
|
319
|
+
try:
|
|
320
|
+
from drydock.graphrag import knowledge_base_exists
|
|
321
|
+
if "Knowledge" not in _pins and knowledge_base_exists(os.getcwd()):
|
|
322
|
+
_pins.append("Knowledge")
|
|
323
|
+
except Exception:
|
|
324
|
+
pass # advisory: never let this break the turn
|
|
311
325
|
available = select_tools(
|
|
312
326
|
available,
|
|
313
327
|
phase=str(state.task.phase),
|
|
314
328
|
task_text=state.task.objective,
|
|
315
329
|
max_tools=turn_config.get("max_tools", DEFAULT_MAX_TOOLS),
|
|
316
|
-
pin_tools=
|
|
330
|
+
pin_tools=_pins,
|
|
317
331
|
)
|
|
318
332
|
for event in stream(
|
|
319
333
|
model=turn_config["model"],
|
|
@@ -745,6 +759,16 @@ def run(
|
|
|
745
759
|
_vcmd = (tc.get("input") or {}).get("command", "")
|
|
746
760
|
if looks_like_verification(_vcmd):
|
|
747
761
|
last_verification = parse_evidence(_vcmd, result)
|
|
762
|
+
# Streak of consecutive FAILING checks, for the TUI's proactive
|
|
763
|
+
# /ratchet offer (ratchet.ratchet_offer). Reset on any pass, so it
|
|
764
|
+
# only fires while the model is genuinely stuck re-running the same
|
|
765
|
+
# check — which is precisely the ratchet's use case. Advisory: this
|
|
766
|
+
# counter never changes the agent's own behaviour.
|
|
767
|
+
if last_verification.status == "fail":
|
|
768
|
+
state.verify_fail_streak += 1
|
|
769
|
+
state.last_verify_cmd = _vcmd
|
|
770
|
+
elif last_verification.status == "pass":
|
|
771
|
+
state.verify_fail_streak = 0
|
|
748
772
|
# Accumulate what the checks touched (command + result text),
|
|
749
773
|
# capped, for criteria-coverage at completion time.
|
|
750
774
|
if len(verification_text) < 200_000:
|
|
@@ -0,0 +1,218 @@
|
|
|
1
|
+
"""Bottleneck register — after a system is decomposed, rank its KNOWN components by
|
|
2
|
+
how much the objective actually moves if you improve each one, and attack the one
|
|
3
|
+
limiting factor instead of optimising everything evenly.
|
|
4
|
+
|
|
5
|
+
WHY THIS IS A SEPARATE STEP FROM THE LEDGER
|
|
6
|
+
-------------------------------------------
|
|
7
|
+
`groundtruth.py` ranks UNKNOWNS by decision impact — it answers "what don't I know
|
|
8
|
+
that could change the plan?" That is an epistemic question. The bottleneck is a
|
|
9
|
+
different question, and conflating the two loses it: the limiting factor is usually
|
|
10
|
+
something you already KNOW with no uncertainty at all. You can be perfectly certain
|
|
11
|
+
which stage is slowest and still pour effort into a faster one.
|
|
12
|
+
|
|
13
|
+
This project is its own evidence. Two of its largest course-corrections were bottleneck
|
|
14
|
+
findings, not discoveries:
|
|
15
|
+
· "throughput is the master lever" — the search was being made cleverer for weeks while
|
|
16
|
+
the objective barely moved, because the limiting factor was how many traces the box
|
|
17
|
+
could produce per hour, not how good each one was.
|
|
18
|
+
· "the real lever is training solves back into the model (write-back), not fancier
|
|
19
|
+
search" — the same shape a second time: effort concentrated on a component with lots
|
|
20
|
+
of headroom but little SHARE of the objective, while the component with the share sat
|
|
21
|
+
untouched.
|
|
22
|
+
Neither of those was an unknown. `next_test()` would not have surfaced either, because
|
|
23
|
+
there was nothing to learn — only a limiting factor to name and attack.
|
|
24
|
+
|
|
25
|
+
THE ONE NUMBER IT COMPUTES
|
|
26
|
+
--------------------------
|
|
27
|
+
For a component that controls a fraction `share` of the objective's gap and still has a
|
|
28
|
+
fraction `headroom` left to improve, the realizable gain from perfecting it is
|
|
29
|
+
`share * headroom`. The bottleneck is the component with the largest realizable gain —
|
|
30
|
+
NOT the one that is easiest to improve (high headroom, low share: the trap this whole
|
|
31
|
+
module exists to name), and NOT the biggest cost that happens to be a fundamental
|
|
32
|
+
constraint you cannot move (high share, zero headroom: a wall, not a lever).
|
|
33
|
+
|
|
34
|
+
The Amdahl ceiling `share` is reported alongside: it is the most that component could
|
|
35
|
+
EVER buy you, even with an infinitely good improvement. A lever with a 5% ceiling is
|
|
36
|
+
capped at 5% no matter how brilliant the optimisation — that ceiling is the sentence
|
|
37
|
+
that ends most premature-optimisation arguments.
|
|
38
|
+
|
|
39
|
+
WHY THIS IS NOT A PROMPT
|
|
40
|
+
Same reason as the other two modules: a reasoning checklist telling the model to "find
|
|
41
|
+
the bottleneck first" was measured on terminal-bench-2 and lost to a plain retry. This
|
|
42
|
+
gives the model somewhere to put the decomposition and computes the one comparison it is
|
|
43
|
+
bad at doing in its head — the share×headroom product across a handful of components.
|
|
44
|
+
|
|
45
|
+
Advisory by contract: nothing here raises on bad input, blocks a call, or edits a plan.
|
|
46
|
+
It records, ranks, and renders. See also `groundtruth.py` and `predictions.py`.
|
|
47
|
+
"""
|
|
48
|
+
from __future__ import annotations
|
|
49
|
+
|
|
50
|
+
import json
|
|
51
|
+
from dataclasses import dataclass, asdict
|
|
52
|
+
from pathlib import Path
|
|
53
|
+
|
|
54
|
+
# A component is a "wall" (fundamental constraint, not a lever) once its headroom is
|
|
55
|
+
# this low: no realistic optimisation moves it, so naming it stops you re-attacking it.
|
|
56
|
+
_WALL_HEADROOM = 0.05
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def _clamp01(x: object, default: float = 0.0) -> float:
|
|
60
|
+
try:
|
|
61
|
+
v = float(x) # type: ignore[arg-type]
|
|
62
|
+
except (TypeError, ValueError):
|
|
63
|
+
return default
|
|
64
|
+
if v != v: # NaN
|
|
65
|
+
return default
|
|
66
|
+
return 0.0 if v < 0.0 else 1.0 if v > 1.0 else v
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
@dataclass
|
|
70
|
+
class Component:
|
|
71
|
+
"""One piece of the decomposed system, and its two numbers that matter."""
|
|
72
|
+
id: str
|
|
73
|
+
name: str = ""
|
|
74
|
+
# Fraction of the objective's gap this component controls (its Amdahl share). If the
|
|
75
|
+
# objective is end-to-end latency and this stage is 70% of it, share = 0.7. This is a
|
|
76
|
+
# SENSITIVITY, not the effort currently spent on it — the whole point is that the two
|
|
77
|
+
# come apart.
|
|
78
|
+
share: float = 0.0
|
|
79
|
+
# Fraction of THIS component still improvable. 0 = a fundamental constraint (physics,
|
|
80
|
+
# a hard resource limit, already optimal) you cannot move; 1 = could be eliminated
|
|
81
|
+
# entirely. This is where step 4 (fundamental constraints) enters the arithmetic.
|
|
82
|
+
headroom: float = 1.0
|
|
83
|
+
note: str = ""
|
|
84
|
+
|
|
85
|
+
def gain(self) -> float:
|
|
86
|
+
"""Realizable objective gain from exploiting this component's headroom."""
|
|
87
|
+
return self.share * self.headroom
|
|
88
|
+
|
|
89
|
+
def ceiling(self) -> float:
|
|
90
|
+
"""Amdahl ceiling: the most this component could EVER buy, headroom aside."""
|
|
91
|
+
return self.share
|
|
92
|
+
|
|
93
|
+
def is_wall(self) -> bool:
|
|
94
|
+
return self.share > 0 and self.headroom <= _WALL_HEADROOM
|
|
95
|
+
|
|
96
|
+
def summary(self) -> str:
|
|
97
|
+
mark = "▮" if self.is_wall() else "▲"
|
|
98
|
+
return (f"{mark} {self.id} {self.name[:56]} "
|
|
99
|
+
f"[gain {self.gain():.0%} · ceiling {self.ceiling():.0%} · "
|
|
100
|
+
f"headroom {self.headroom:.0%}]")
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
class Bottlenecks:
|
|
104
|
+
"""The decomposed components of one problem, persisted next to the run."""
|
|
105
|
+
|
|
106
|
+
def __init__(self, path: str | Path | None = None):
|
|
107
|
+
self.path = Path(path) if path else None
|
|
108
|
+
self.items: list[Component] = []
|
|
109
|
+
if self.path and self.path.exists():
|
|
110
|
+
self.load()
|
|
111
|
+
|
|
112
|
+
# ── recording ────────────────────────────────────────────────────────────
|
|
113
|
+
def add(self, name: str, *, share: float = 0.0, headroom: float = 1.0,
|
|
114
|
+
note: str = "") -> Component:
|
|
115
|
+
"""Record a decomposed component. Out-of-range numbers are clamped to [0,1]
|
|
116
|
+
rather than raising — a register that throws would take down the turn it is
|
|
117
|
+
meant to help."""
|
|
118
|
+
c = Component(id=f"c{len(self.items) + 1}",
|
|
119
|
+
name=(name or "").strip(),
|
|
120
|
+
share=_clamp01(share, 0.0),
|
|
121
|
+
headroom=_clamp01(headroom, 1.0),
|
|
122
|
+
note=(note or "").strip())
|
|
123
|
+
self.items.append(c)
|
|
124
|
+
self._save()
|
|
125
|
+
return c
|
|
126
|
+
|
|
127
|
+
def update(self, cid: str, *, share: float | None = None,
|
|
128
|
+
headroom: float | None = None, note: str | None = None) -> Component | None:
|
|
129
|
+
"""Revise a component's numbers as measurement replaces estimate. Attacking the
|
|
130
|
+
bottleneck moves its headroom toward 0, at which point the ranking shifts to the
|
|
131
|
+
next limiting factor on its own."""
|
|
132
|
+
for c in self.items:
|
|
133
|
+
if c.id == cid:
|
|
134
|
+
if share is not None:
|
|
135
|
+
c.share = _clamp01(share, c.share)
|
|
136
|
+
if headroom is not None:
|
|
137
|
+
c.headroom = _clamp01(headroom, c.headroom)
|
|
138
|
+
if note is not None:
|
|
139
|
+
c.note = (note or "").strip()
|
|
140
|
+
self._save()
|
|
141
|
+
return c
|
|
142
|
+
return None
|
|
143
|
+
|
|
144
|
+
# ── the ranking that is the point of the module ──────────────────────────
|
|
145
|
+
def _levers(self) -> list[Component]:
|
|
146
|
+
"""Components that are actually worth attacking — positive gain and not a wall.
|
|
147
|
+
A wall's sliver of headroom is not a lever, so it is excluded here even though its
|
|
148
|
+
gain is technically > 0."""
|
|
149
|
+
return [c for c in self.items if c.gain() > 0 and not c.is_wall()]
|
|
150
|
+
|
|
151
|
+
def bottleneck(self) -> Component | None:
|
|
152
|
+
"""The one limiting factor to attack first: the largest realizable gain
|
|
153
|
+
(share × headroom). Ties break toward the larger share — the higher ceiling is
|
|
154
|
+
the safer bet once the immediate gain is equal."""
|
|
155
|
+
movable = self._levers()
|
|
156
|
+
if not movable:
|
|
157
|
+
return None
|
|
158
|
+
return sorted(movable, key=lambda c: (-c.gain(), -c.share, c.id))[0]
|
|
159
|
+
|
|
160
|
+
def ranked(self) -> list[Component]:
|
|
161
|
+
return sorted(self.items, key=lambda c: (-c.gain(), -c.share, c.id))
|
|
162
|
+
|
|
163
|
+
def walls(self) -> list[Component]:
|
|
164
|
+
"""High-share components with (almost) no headroom — fundamental constraints.
|
|
165
|
+
Naming them is what stops the loop re-attacking a wall as if it were a lever."""
|
|
166
|
+
return [c for c in self.items if c.is_wall()]
|
|
167
|
+
|
|
168
|
+
def misplaced_effort(self) -> Component | None:
|
|
169
|
+
"""The trap this module exists to catch: the component with the MOST headroom is
|
|
170
|
+
not the bottleneck — improving it feels productive but is capped by its small
|
|
171
|
+
share. Returns that decoy when it differs from the real bottleneck, else None."""
|
|
172
|
+
movable = self._levers()
|
|
173
|
+
if len(movable) < 2:
|
|
174
|
+
return None
|
|
175
|
+
by_headroom = max(movable, key=lambda c: (c.headroom, -c.share))
|
|
176
|
+
bn = self.bottleneck()
|
|
177
|
+
return by_headroom if bn is not None and by_headroom.id != bn.id else None
|
|
178
|
+
|
|
179
|
+
# ── rendering ────────────────────────────────────────────────────────────
|
|
180
|
+
def render(self) -> str:
|
|
181
|
+
if not self.items:
|
|
182
|
+
return "(no components decomposed yet)"
|
|
183
|
+
lines = [c.summary() for c in self.ranked()]
|
|
184
|
+
bn = self.bottleneck()
|
|
185
|
+
if bn is not None:
|
|
186
|
+
lines.append(f"→ attack first: {bn.id} ({bn.name[:50]}) — "
|
|
187
|
+
f"realizable gain {bn.gain():.0%}, ceiling {bn.ceiling():.0%}")
|
|
188
|
+
else:
|
|
189
|
+
lines.append("→ no movable component: every lever is a wall "
|
|
190
|
+
"(headroom ~0). The objective may need a different decomposition.")
|
|
191
|
+
decoy = self.misplaced_effort()
|
|
192
|
+
if decoy is not None:
|
|
193
|
+
lines.append(f"⚠ {decoy.id} has the most headroom but only a "
|
|
194
|
+
f"{decoy.ceiling():.0%} ceiling — optimising it is capped there.")
|
|
195
|
+
total_share = sum(c.share for c in self.items)
|
|
196
|
+
if total_share > 1.05:
|
|
197
|
+
lines.append(f"⚠ shares sum to {total_share:.0%} (>100%): the decomposition "
|
|
198
|
+
"overlaps — components are not independent.")
|
|
199
|
+
return "\n".join(lines)
|
|
200
|
+
|
|
201
|
+
# ── persistence ──────────────────────────────────────────────────────────
|
|
202
|
+
def _save(self) -> None:
|
|
203
|
+
if not self.path:
|
|
204
|
+
return
|
|
205
|
+
try:
|
|
206
|
+
self.path.parent.mkdir(parents=True, exist_ok=True)
|
|
207
|
+
self.path.write_text(
|
|
208
|
+
json.dumps([asdict(c) for c in self.items], indent=2), encoding="utf-8")
|
|
209
|
+
except OSError:
|
|
210
|
+
pass # advisory: never fail a run over bookkeeping
|
|
211
|
+
|
|
212
|
+
def load(self) -> None:
|
|
213
|
+
try:
|
|
214
|
+
raw = json.loads(self.path.read_text(encoding="utf-8")) if self.path else []
|
|
215
|
+
self.items = [Component(**{k: v for k, v in d.items()
|
|
216
|
+
if k in Component.__dataclass_fields__}) for d in raw]
|
|
217
|
+
except (OSError, json.JSONDecodeError, TypeError):
|
|
218
|
+
self.items = []
|
|
@@ -373,6 +373,16 @@ def main():
|
|
|
373
373
|
cfg["cwd"] = os.getcwd()
|
|
374
374
|
sys.exit(run_cli(sys.argv[2:], config=cfg))
|
|
375
375
|
|
|
376
|
+
# `drydock swarm <objective> …` / `swarm status|list|resume` — multi-agent swarm.
|
|
377
|
+
# Same argv-intercept pattern as eratchet so the subcommand parses its own flags.
|
|
378
|
+
if len(sys.argv) > 1 and sys.argv[1] == "swarm":
|
|
379
|
+
from drydock import config as cfgmod
|
|
380
|
+
from drydock.swarm import run_cli as swarm_run_cli
|
|
381
|
+
cfg = cfgmod.resolve({}, cfgmod.default_config_path())
|
|
382
|
+
cfgmod.resolve_active_model(cfg)
|
|
383
|
+
cfg["cwd"] = os.getcwd()
|
|
384
|
+
sys.exit(swarm_run_cli(sys.argv[2:], config=cfg))
|
|
385
|
+
|
|
376
386
|
parser = argparse.ArgumentParser(description="DryDock — local coding agent")
|
|
377
387
|
parser.add_argument(
|
|
378
388
|
"--version", "-V", action="version", version=f"drydock {__version__}"
|
|
@@ -40,6 +40,15 @@ DEFAULTS: dict[str, object] = {
|
|
|
40
40
|
# generation usually isn't stalled — a known gemma/llama.cpp hang). 0 = off.
|
|
41
41
|
# Set to e.g. 600 on a stall-prone local server. Bounded to a few retries.
|
|
42
42
|
"stall_retry_secs": 0,
|
|
43
|
+
# Idle-output watchdog for foreground Bash: if a running command produces NO
|
|
44
|
+
# output for this many seconds, drydock stops BLOCKING on it and adopts it as a
|
|
45
|
+
# background job (it keeps running; the agent gets the prompt back + a nudge to
|
|
46
|
+
# reconsider). Complements the total-runtime auto-background (_AUTO_BG_TIMEOUT):
|
|
47
|
+
# catches a command hung SILENTLY on a long timeout (e.g. a dead network
|
|
48
|
+
# download the model set timeout:1800 for) far sooner than the total cap.
|
|
49
|
+
# Non-destructive (backgrounds, never kills). 0 = off. Set e.g. 300 on the fleet
|
|
50
|
+
# where a weak local model loops on a silently-stuck command.
|
|
51
|
+
"bash_idle_bg_secs": 0,
|
|
43
52
|
# Inject bundled technique recipes (drydock/recipes.py) relevant to the task
|
|
44
53
|
# into the system prompt, so a local model has the *method* a task needs
|
|
45
54
|
# instead of guessing. Retrieval is keyword-overlap; only relevant recipes are
|
|
@@ -88,6 +97,11 @@ DEFAULTS: dict[str, object] = {
|
|
|
88
97
|
# web tools surfaced even on tasks whose text never mentions the web. Names
|
|
89
98
|
# not in the registry are ignored.
|
|
90
99
|
"pin_tools": [],
|
|
100
|
+
# When a single agent stays stuck on a KNOWN verifier (it keeps failing), the harness
|
|
101
|
+
# may auto-escalate to an in-process multi-agent swarm to explore approaches in parallel
|
|
102
|
+
# (docs/multi_agent_swarm_prd.md). On by default; set false to keep every request single-
|
|
103
|
+
# agent unless the user runs /swarm explicitly.
|
|
104
|
+
"swarm_auto_escalate": True,
|
|
91
105
|
# URL substrings the web tools refuse: WebSearch drops matching results,
|
|
92
106
|
# WebFetch declines matching URLs (with a plain message, never an error).
|
|
93
107
|
# Used to keep benchmark/solution sites out of harvested training runs.
|
|
@@ -65,6 +65,26 @@ def default_store_path(cwd: str) -> Path:
|
|
|
65
65
|
return Path(cwd) / ".drydock" / "graphrag.db"
|
|
66
66
|
|
|
67
67
|
|
|
68
|
+
def knowledge_base_exists(cwd: str) -> bool:
|
|
69
|
+
"""True when this project has a built GraphRAG index (SQLite or legacy JSON).
|
|
70
|
+
|
|
71
|
+
Used to PIN the `Knowledge` tool for the turn. Dynamic tool selection
|
|
72
|
+
(tool_select) otherwise gates Knowledge behind the keywords
|
|
73
|
+
knowledge/graph/entity/graphrag/ingest, which the user has no reason to type:
|
|
74
|
+
the whole point of the feature is that a plain question like "what does the
|
|
75
|
+
auth module do?" is answered FROM their index. With 46 tools and a cap of 12,
|
|
76
|
+
that question dropped Knowledge from the toolset and the model's call came
|
|
77
|
+
back "[The 'Knowledge' tool is not available here]" — the bug reported
|
|
78
|
+
2026-09-03. Knowledge is read-only and cheap, so pinning it whenever an index
|
|
79
|
+
exists costs one slot and restores the documented behaviour; when no index
|
|
80
|
+
exists nothing is pinned and the tool stays gated as before.
|
|
81
|
+
"""
|
|
82
|
+
try:
|
|
83
|
+
return _resolve_store(default_store_path(cwd)).exists()
|
|
84
|
+
except Exception: # never let a tool-selection helper break the turn
|
|
85
|
+
return False
|
|
86
|
+
|
|
87
|
+
|
|
68
88
|
def _resolve_store(store_path) -> Path:
|
|
69
89
|
"""Given a requested store path, return the one that actually exists — a
|
|
70
90
|
``.db`` (SQLite) if present, else a legacy ``.json`` sibling, else the path
|
|
@@ -0,0 +1,167 @@
|
|
|
1
|
+
"""Ground-truth ledger — separate what is KNOWN from what is ASSUMED, and rank the
|
|
2
|
+
unknowns by how much they could change the decision.
|
|
3
|
+
|
|
4
|
+
WHY THIS EXISTS, and why it is a LEDGER rather than a prompt
|
|
5
|
+
------------------------------------------------------------
|
|
6
|
+
A reasoning checklist was measured on terminal-bench-2 and did not work: applied to
|
|
7
|
+
every task it caused regressions (it overthinks work that was already succeeding), and
|
|
8
|
+
applied only after a failure it was BEATEN by simply retrying (4.6%/4.7% rescue vs a
|
|
9
|
+
plain retry's 6.2%). A longer ten-step variant did worse than a short one, and
|
|
10
|
+
compliance explained why — the long form was followed less often (criteria written 3/8
|
|
11
|
+
vs 7/8). More instructions bought less behaviour.
|
|
12
|
+
|
|
13
|
+
So this module deliberately does NOT tell the model how to think. It gives it somewhere
|
|
14
|
+
to PUT things, and it computes one ranking the model is bad at doing in its head.
|
|
15
|
+
|
|
16
|
+
The specific thing it computes — `next_test()` — comes from a failure in this project's
|
|
17
|
+
own research log rather than from theory. A claim about a prompt scaffold was reported
|
|
18
|
+
at +6.8 points, corrected to +3.4, then to +0.0. Every correction came from a control
|
|
19
|
+
that could have been run on day one; the highest-value unknown was always "would a plain
|
|
20
|
+
retry do the same thing?", and it went unasked for six days while cheaper, less
|
|
21
|
+
decisive experiments ran. Ranking unknowns by DECISION IMPACT (not by how interesting or
|
|
22
|
+
how easy they are) is the step that would have caught it immediately.
|
|
23
|
+
|
|
24
|
+
Advisory by contract: nothing here blocks a tool call, raises on bad input, or edits the
|
|
25
|
+
model's plan. It records, ranks, and renders. See also `predictions.py`, which closes the
|
|
26
|
+
other half of the loop (predict before you look).
|
|
27
|
+
"""
|
|
28
|
+
from __future__ import annotations
|
|
29
|
+
|
|
30
|
+
import json
|
|
31
|
+
from dataclasses import dataclass, asdict
|
|
32
|
+
from pathlib import Path
|
|
33
|
+
|
|
34
|
+
# Confidence bands. Kept coarse on purpose: a model asked for a 0-100 confidence emits
|
|
35
|
+
# noise, but "did I VERIFY this or am I assuming it" is a distinction it can actually make.
|
|
36
|
+
FACT = "fact" # verified by looking (a file read, a command's output)
|
|
37
|
+
ASSUMPTION = "assumption" # believed, not checked — the usual source of a stuck run
|
|
38
|
+
UNKNOWN = "unknown" # explicitly not known
|
|
39
|
+
KINDS = (FACT, ASSUMPTION, UNKNOWN)
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
@dataclass
|
|
43
|
+
class Item:
|
|
44
|
+
"""One thing believed about the problem, and where that belief came from."""
|
|
45
|
+
id: str
|
|
46
|
+
kind: str = ASSUMPTION
|
|
47
|
+
statement: str = ""
|
|
48
|
+
# For FACT: how it was verified (the command, the file). Empty evidence on a FACT is
|
|
49
|
+
# itself a smell — it means something got promoted without being checked.
|
|
50
|
+
evidence: str = ""
|
|
51
|
+
# For ASSUMPTION/UNKNOWN: how much the decision changes if this turns out false.
|
|
52
|
+
# 0 = irrelevant, 3 = the approach is invalid. This is the field that does the work.
|
|
53
|
+
impact: int = 0
|
|
54
|
+
# Rough cost to find out (0 = seconds, 3 = hours). Used only to break impact ties, so
|
|
55
|
+
# a cheap decisive test wins over an expensive one — never to avoid a decisive test.
|
|
56
|
+
cost: int = 1
|
|
57
|
+
resolved: bool = False
|
|
58
|
+
|
|
59
|
+
def summary(self) -> str:
|
|
60
|
+
mark = {FACT: "✓", ASSUMPTION: "?", UNKNOWN: "·"}.get(self.kind, "·")
|
|
61
|
+
if self.resolved:
|
|
62
|
+
mark = "✔"
|
|
63
|
+
tag = f" [impact {self.impact}/cost {self.cost}]" if self.kind != FACT else ""
|
|
64
|
+
return f"{mark} {self.id} {self.statement[:70]}{tag}"
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
class Ledger:
|
|
68
|
+
"""Facts, assumptions and unknowns for one task, persisted next to the run."""
|
|
69
|
+
|
|
70
|
+
def __init__(self, path: str | Path | None = None):
|
|
71
|
+
self.path = Path(path) if path else None
|
|
72
|
+
self.items: list[Item] = []
|
|
73
|
+
if self.path and self.path.exists():
|
|
74
|
+
self.load()
|
|
75
|
+
|
|
76
|
+
# ── recording ────────────────────────────────────────────────────────────
|
|
77
|
+
def add(self, statement: str, kind: str = ASSUMPTION, *, evidence: str = "",
|
|
78
|
+
impact: int = 0, cost: int = 1) -> Item:
|
|
79
|
+
"""Record a belief. Unknown kinds degrade to ASSUMPTION rather than raising —
|
|
80
|
+
a ledger that throws would take down the turn it is supposed to be helping."""
|
|
81
|
+
if kind not in KINDS:
|
|
82
|
+
kind = ASSUMPTION
|
|
83
|
+
it = Item(id=f"i{len(self.items) + 1}", kind=kind,
|
|
84
|
+
statement=(statement or "").strip(),
|
|
85
|
+
evidence=(evidence or "").strip(),
|
|
86
|
+
impact=max(0, min(3, int(impact or 0))),
|
|
87
|
+
cost=max(0, min(3, int(cost or 1))))
|
|
88
|
+
self.items.append(it)
|
|
89
|
+
self._save()
|
|
90
|
+
return it
|
|
91
|
+
|
|
92
|
+
def verify(self, item_id: str, evidence: str) -> Item | None:
|
|
93
|
+
"""Promote an assumption to a fact, with the evidence that earned it. This is the
|
|
94
|
+
only path to FACT — a belief cannot become a fact by being restated."""
|
|
95
|
+
for it in self.items:
|
|
96
|
+
if it.id == item_id:
|
|
97
|
+
it.kind = FACT
|
|
98
|
+
it.evidence = (evidence or "").strip()
|
|
99
|
+
it.resolved = True
|
|
100
|
+
self._save()
|
|
101
|
+
return it
|
|
102
|
+
return None
|
|
103
|
+
|
|
104
|
+
def refute(self, item_id: str, evidence: str) -> Item | None:
|
|
105
|
+
"""Mark an assumption FALSE. Kept rather than deleted: a refuted assumption is
|
|
106
|
+
the most valuable row in the ledger — it is the one that changed the approach."""
|
|
107
|
+
for it in self.items:
|
|
108
|
+
if it.id == item_id:
|
|
109
|
+
it.statement = f"[REFUTED] {it.statement}"
|
|
110
|
+
it.kind = FACT
|
|
111
|
+
it.evidence = (evidence or "").strip()
|
|
112
|
+
it.resolved = True
|
|
113
|
+
self._save()
|
|
114
|
+
return it
|
|
115
|
+
return None
|
|
116
|
+
|
|
117
|
+
# ── the ranking that is the point of the module ──────────────────────────
|
|
118
|
+
def open_uncertainties(self) -> list[Item]:
|
|
119
|
+
return [i for i in self.items if not i.resolved and i.kind != FACT]
|
|
120
|
+
|
|
121
|
+
def next_test(self) -> Item | None:
|
|
122
|
+
"""The unknown worth attacking first: highest decision-impact, cheapest to settle
|
|
123
|
+
among equals. NOT the easiest, and NOT the most interesting — those are the two
|
|
124
|
+
attractors that let a project run six days of experiments that could not have
|
|
125
|
+
changed its own conclusion."""
|
|
126
|
+
openq = [i for i in self.open_uncertainties() if i.impact > 0]
|
|
127
|
+
if not openq:
|
|
128
|
+
return None
|
|
129
|
+
return sorted(openq, key=lambda i: (-i.impact, i.cost, i.id))[0]
|
|
130
|
+
|
|
131
|
+
def unevidenced_facts(self) -> list[Item]:
|
|
132
|
+
"""FACTs carrying no evidence — i.e. assumptions wearing a fact's badge."""
|
|
133
|
+
return [i for i in self.items if i.kind == FACT and not i.evidence]
|
|
134
|
+
|
|
135
|
+
# ── rendering ────────────────────────────────────────────────────────────
|
|
136
|
+
def render(self) -> str:
|
|
137
|
+
if not self.items:
|
|
138
|
+
return "(ledger empty)"
|
|
139
|
+
lines = [i.summary() for i in self.items]
|
|
140
|
+
nxt = self.next_test()
|
|
141
|
+
if nxt:
|
|
142
|
+
lines.append(f"→ test first: {nxt.id} ({nxt.statement[:60]}) "
|
|
143
|
+
f"— impact {nxt.impact}, cost {nxt.cost}")
|
|
144
|
+
stale = self.unevidenced_facts()
|
|
145
|
+
if stale:
|
|
146
|
+
lines.append(f"⚠ {len(stale)} 'fact(s)' with no evidence: "
|
|
147
|
+
f"{', '.join(i.id for i in stale)}")
|
|
148
|
+
return "\n".join(lines)
|
|
149
|
+
|
|
150
|
+
# ── persistence ──────────────────────────────────────────────────────────
|
|
151
|
+
def _save(self) -> None:
|
|
152
|
+
if not self.path:
|
|
153
|
+
return
|
|
154
|
+
try:
|
|
155
|
+
self.path.parent.mkdir(parents=True, exist_ok=True)
|
|
156
|
+
self.path.write_text(
|
|
157
|
+
json.dumps([asdict(i) for i in self.items], indent=2), encoding="utf-8")
|
|
158
|
+
except OSError:
|
|
159
|
+
pass # advisory: never fail a run over bookkeeping
|
|
160
|
+
|
|
161
|
+
def load(self) -> None:
|
|
162
|
+
try:
|
|
163
|
+
raw = json.loads(self.path.read_text(encoding="utf-8")) if self.path else []
|
|
164
|
+
self.items = [Item(**{k: v for k, v in d.items()
|
|
165
|
+
if k in Item.__dataclass_fields__}) for d in raw]
|
|
166
|
+
except (OSError, json.JSONDecodeError, TypeError):
|
|
167
|
+
self.items = []
|