drydock-cli 3.1.30__tar.gz → 3.1.31__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {drydock_cli-3.1.30/drydock_cli.egg-info → drydock_cli-3.1.31}/PKG-INFO +1 -1
- drydock_cli-3.1.31/drydock/asr.py +227 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/drydock/config.py +6 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/drydock/mission_run.py +18 -8
- {drydock_cli-3.1.30 → drydock_cli-3.1.31/drydock_cli.egg-info}/PKG-INFO +1 -1
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/drydock_cli.egg-info/SOURCES.txt +3 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/pyproject.toml +1 -1
- drydock_cli-3.1.31/tests/test_asr.py +140 -0
- drydock_cli-3.1.31/tests/test_main_api_key.py +76 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_mission_run.py +27 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/LICENSE +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/NOTICE +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/README.md +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/drydock/__init__.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/drydock/__main__.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/drydock/advisor.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/drydock/agent.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/drydock/bash_safety.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/drydock/bottleneck.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/drydock/budget.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/drydock/builtin_skills/__init__.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/drydock/builtin_skills/document-canvas.md +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/drydock/builtin_skills/fiar-assess.md +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/drydock/builtin_skills/fiar-cap.md +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/drydock/builtin_skills/fiar-evidence.md +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/drydock/builtin_skills/fiar-readiness.md +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/drydock/builtin_skills/ml-data.md +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/drydock/builtin_skills/ml-debug.md +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/drydock/builtin_skills/ml-finetune.md +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/drydock/builtin_skills/ml-metrics.md +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/drydock/builtin_skills/ml-rl.md +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/drydock/builtin_skills/ml-train.md +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/drydock/builtin_skills/nist-ai-rmf.md +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/drydock/builtin_skills/nist-csf.md +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/drydock/builtin_skills/rmf-categorize.md +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/drydock/builtin_skills/rmf-control.md +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/drydock/builtin_skills/rmf-poam.md +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/drydock/builtin_skills/rmf-review.md +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/drydock/builtin_skills/stig-assess.md +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/drydock/builtin_skills/stig-remediate.md +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/drydock/capacity.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/drydock/cci.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/drydock/cli.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/drydock/comms/__init__.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/drydock/comms/attention.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/drydock/comms/channels.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/drydock/comms/events.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/drydock/compaction.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/drydock/detect.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/drydock/doccanvas.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/drydock/eratchet.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/drydock/events.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/drydock/extract.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/drydock/fiar.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/drydock/gittools.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/drydock/graphrag.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/drydock/groundtruth.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/drydock/guards.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/drydock/jobs.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/drydock/loop_detect.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/drydock/mcp.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/drydock/mission.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/drydock/mission_cli.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/drydock/pdfredbox.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/drydock/phases.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/drydock/poam.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/drydock/predictions.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/drydock/progress.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/drydock/providers.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/drydock/ratchet.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/drydock/recipes.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/drydock/recovery.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/drydock/resume.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/drydock/rmf.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/drydock/rmf_graph.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/drydock/skills.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/drydock/stig.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/drydock/subagents.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/drydock/suggest.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/drydock/swarm.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/drydock/task_state.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/drydock/tool_policy.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/drydock/tool_registry.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/drydock/tool_result.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/drydock/tool_select.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/drydock/tool_validate.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/drydock/tools/__init__.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/drydock/trajectory.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/drydock/tui/__init__.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/drydock/tui/app.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/drydock/tui/approval.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/drydock/tui/messages.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/drydock/tui/widgets.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/drydock/tuning.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/drydock/verification.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/drydock/web.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/drydock_cli.egg-info/dependency_links.txt +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/drydock_cli.egg-info/entry_points.txt +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/drydock_cli.egg-info/requires.txt +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/drydock_cli.egg-info/top_level.txt +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/setup.cfg +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_advisor.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_approval.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_back_command.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_bash_background.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_bash_binary_output.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_bash_crossplatform.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_bash_output_bounding.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_bash_process_group.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_bash_safety.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_bash_sanitize.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_bash_shell.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_bash_stdin.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_bash_stop_partial.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_bash_timeout_network.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_bash_timeout_param.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_bottleneck.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_budget.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_capacity.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_cci.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_cli_agents.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_comms.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_compact_command.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_compaction.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_config.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_config_migration.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_context_limit_config.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_context_limit_issue25.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_criteria_coverage.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_cycling_detection.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_declared_deps.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_degenerate_argument.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_detect.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_dispatch.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_doccanvas.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_e2e_connected.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_edit_replace_all.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_edit_thrash_and_compaction.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_effort_governor.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_empty_response.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_eratchet.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_events.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_extract.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_failure_loop.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_fiar.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_first_run_setup.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_gittools.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_glob_edit_edges.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_graphify_example.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_graphrag.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_graphrag_quoted_path.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_graphrag_sqlite.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_grep_and_read_robust.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_groundtruth_predictions.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_guards_and_tools.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_hallucinated_tools.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_jobs.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_knowledge_tool_available.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_leaked_tool_call.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_loop_detect.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_mcp.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_mission.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_mission_cli.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_model_registry.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_oneshot_unreachable.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_overthink_interrupt.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_pdfredbox.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_phases.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_plan_autocontinue.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_poam.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_progress.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_providers_unreachable.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_ratchet.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_ratchet_offer.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_read_index.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_recipes.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_recovery.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_recovery_config.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_recovery_integration.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_repeated_outcome.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_repetition_interrupt.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_resume_events.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_resume_restore.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_resume_snapshot.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_rmf.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_rmf_graph.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_rmf_stig_graph.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_rolling_plan.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_runaway_repetition.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_screenshot.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_server_probe.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_skills.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_sqlite_events.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_stall_retry.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_stig.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_stop.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_streaming_newlines.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_subagent.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_subagents.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_suggest.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_swarm.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_system_prompt_help.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_task_state.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_timeline.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_todo.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_tool_arg_coercion.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_tool_arg_coercion_more.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_tool_arg_parsing.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_tool_canonical.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_tool_policy.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_tool_result.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_tool_select.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_tool_validate.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_tools_undo.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_trajectory.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_tui.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_tuning.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_verification_gate.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_viewimage.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_vision_input.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_web_tools.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_windows_shell.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_worker_subagent.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_write_content_coerce.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.31}/tests/test_xccdf.py +0 -0
|
@@ -0,0 +1,227 @@
|
|
|
1
|
+
"""Adaptive Swarm Runtime (ASR) — the search-decision core.
|
|
2
|
+
|
|
3
|
+
This is the resource-agnostic brain of the adaptive swarm (PRD "Resource-Aware Adaptive
|
|
4
|
+
Swarms"): it turns a compute/time BUDGET into decisions about *where the next unit of inference
|
|
5
|
+
should go* and *when to stop*. It deliberately holds no model, GPU, or scheduler state — those
|
|
6
|
+
live in capacity.py (concurrency), swarm.py (blackboard/candidates/worktrees) and the mission
|
|
7
|
+
store (durable state). Keeping the decisions pure makes them unit-testable and reusable.
|
|
8
|
+
|
|
9
|
+
Three pieces, in dependency order:
|
|
10
|
+
* Budget — a finite termination envelope parsed from a user string (§15).
|
|
11
|
+
* Convergence — should the swarm keep going? Stops on success / budget / no-improvement /
|
|
12
|
+
when marginal value Δquality/Δcompute falls below a floor (§23).
|
|
13
|
+
* AdaptiveAllocator — split the next compute pool across live branches by expected value,
|
|
14
|
+
pruning the weak ones (§17). Bad ideas must not get the same compute as good.
|
|
15
|
+
"""
|
|
16
|
+
from __future__ import annotations
|
|
17
|
+
|
|
18
|
+
import re
|
|
19
|
+
from dataclasses import dataclass, field
|
|
20
|
+
|
|
21
|
+
# ── Budget (§15) ────────────────────────────────────────────────────────────
|
|
22
|
+
_DUR = re.compile(r"^\s*([\d.]+)\s*([hms])\s*$", re.I)
|
|
23
|
+
_TOK = re.compile(r"^\s*([\d.]+)\s*([kmg]?)\s*[-_]?tokens?\s*$", re.I)
|
|
24
|
+
_MULT = {"": 1, "k": 1_000, "m": 1_000_000, "g": 1_000_000_000}
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
@dataclass
|
|
28
|
+
class Budget:
|
|
29
|
+
"""A finite stopping envelope (§10/§15). Every field is optional, but a Budget with no
|
|
30
|
+
finite limit is a bug — `is_finite` guards against "run forever"."""
|
|
31
|
+
wall_secs: float | None = None
|
|
32
|
+
tokens: int | None = None
|
|
33
|
+
generations: int | None = None
|
|
34
|
+
|
|
35
|
+
@property
|
|
36
|
+
def is_finite(self) -> bool:
|
|
37
|
+
return any(v is not None for v in (self.wall_secs, self.tokens, self.generations))
|
|
38
|
+
|
|
39
|
+
@classmethod
|
|
40
|
+
def parse(cls, spec: str) -> "Budget":
|
|
41
|
+
"""Parse one user token: '2h' / '30m' / '90s' (wall) or '100M-tokens' / '50k-tokens'
|
|
42
|
+
(compute). Unknown strings raise ValueError so a typo never silently means 'unbounded'."""
|
|
43
|
+
s = (spec or "").strip()
|
|
44
|
+
md = _DUR.match(s)
|
|
45
|
+
if md:
|
|
46
|
+
n, unit = float(md.group(1)), md.group(2).lower()
|
|
47
|
+
return cls(wall_secs=n * {"h": 3600, "m": 60, "s": 1}[unit])
|
|
48
|
+
mt = _TOK.match(s)
|
|
49
|
+
if mt:
|
|
50
|
+
return cls(tokens=int(float(mt.group(1)) * _MULT[mt.group(2).lower()]))
|
|
51
|
+
raise ValueError(f"unparseable budget: {spec!r} (try '2h', '30m', '100M-tokens')")
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
# ── Convergence (§23) ─────────────────────────────────────────────────────────
|
|
55
|
+
@dataclass
|
|
56
|
+
class GenerationRecord:
|
|
57
|
+
gen: int
|
|
58
|
+
best_score: float # best branch quality so far (higher = better)
|
|
59
|
+
cumulative_tokens: int # total inference tokens spent through this generation
|
|
60
|
+
elapsed_s: float # wall-clock since start
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
@dataclass
|
|
64
|
+
class Convergence:
|
|
65
|
+
"""Decides whether the swarm should run another generation (§23). "More agents are not
|
|
66
|
+
automatically better": we stop the moment any finite condition trips, INCLUDING the marginal
|
|
67
|
+
one — when Δquality per unit compute drops below `min_marginal`, extra inference is waste."""
|
|
68
|
+
budget: Budget
|
|
69
|
+
success_score: float = 100.0 # stop when best branch reaches this (e.g. all tests pass)
|
|
70
|
+
no_improve_gens: int = 3 # stop after this many generations with no best-score gain
|
|
71
|
+
min_marginal: float = 0.0 # floor on Δscore per 1k tokens; <=0 disables the check
|
|
72
|
+
eps: float = 1e-9
|
|
73
|
+
history: list[GenerationRecord] = field(default_factory=list)
|
|
74
|
+
|
|
75
|
+
def record(self, best_score: float, cumulative_tokens: int, elapsed_s: float) -> None:
|
|
76
|
+
self.history.append(GenerationRecord(len(self.history) + 1, best_score,
|
|
77
|
+
cumulative_tokens, elapsed_s))
|
|
78
|
+
|
|
79
|
+
def _gens_since_improvement(self) -> int:
|
|
80
|
+
"""Generations since the best score was FIRST reached (a plateau at the best counts)."""
|
|
81
|
+
if not self.history:
|
|
82
|
+
return 0
|
|
83
|
+
best = max(r.best_score for r in self.history)
|
|
84
|
+
first_best = next(i for i, r in enumerate(self.history) if r.best_score >= best - self.eps)
|
|
85
|
+
return (len(self.history) - 1) - first_best
|
|
86
|
+
|
|
87
|
+
def marginal_value(self) -> float | None:
|
|
88
|
+
"""Δscore per 1,000 tokens over the last generation, or None with <2 records."""
|
|
89
|
+
if len(self.history) < 2:
|
|
90
|
+
return None
|
|
91
|
+
a, b = self.history[-2], self.history[-1]
|
|
92
|
+
dtok = b.cumulative_tokens - a.cumulative_tokens
|
|
93
|
+
if dtok <= 0:
|
|
94
|
+
return None
|
|
95
|
+
return (b.best_score - a.best_score) / (dtok / 1000.0)
|
|
96
|
+
|
|
97
|
+
def should_continue(self) -> tuple[bool, str]:
|
|
98
|
+
"""(keep_going?, reason). Checked in priority order; the reason names the stop trigger."""
|
|
99
|
+
last = self.history[-1] if self.history else None
|
|
100
|
+
if last is not None and last.best_score >= self.success_score - self.eps:
|
|
101
|
+
return False, "success score reached"
|
|
102
|
+
b = self.budget
|
|
103
|
+
if b.wall_secs is not None and last is not None and last.elapsed_s >= b.wall_secs:
|
|
104
|
+
return False, "wall-clock budget exhausted"
|
|
105
|
+
if b.tokens is not None and last is not None and last.cumulative_tokens >= b.tokens:
|
|
106
|
+
return False, "token budget exhausted"
|
|
107
|
+
if b.generations is not None and len(self.history) >= b.generations:
|
|
108
|
+
return False, "generation budget reached"
|
|
109
|
+
if self.no_improve_gens > 0 and self._gens_since_improvement() >= self.no_improve_gens:
|
|
110
|
+
return False, f"no improvement in {self.no_improve_gens} generations"
|
|
111
|
+
mv = self.marginal_value()
|
|
112
|
+
if self.min_marginal > 0 and mv is not None and mv < self.min_marginal:
|
|
113
|
+
return False, f"marginal value {mv:.3f} < floor {self.min_marginal:.3f} (Δquality/Δcompute)"
|
|
114
|
+
return True, "continue"
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
# ── Adaptive compute allocation (§17) ──────────────────────────────────────────
|
|
118
|
+
@dataclass
|
|
119
|
+
class Allocation:
|
|
120
|
+
grants: dict[str, int] # branch id -> tokens granted this round
|
|
121
|
+
terminated: list[str] # branch ids pruned (0 tokens, killed)
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
def allocate(branches: list[tuple[str, float]], pool_tokens: int, *,
|
|
125
|
+
prune_below: float = 0.0, keep_top: int | None = None,
|
|
126
|
+
bias: float = 1.5, min_grant: int = 0) -> Allocation:
|
|
127
|
+
"""Split `pool_tokens` across branches by expected value (§17).
|
|
128
|
+
|
|
129
|
+
branches: (id, score) with higher score = more promising. A branch is TERMINATED (gets 0)
|
|
130
|
+
if its score is below `prune_below` or it falls outside the top `keep_top`. Survivors share
|
|
131
|
+
the pool in proportion to score**bias, so leaders get disproportionately more (bias>1) — the
|
|
132
|
+
system must not spend equally on good and bad ideas. `min_grant` guarantees a floor so a
|
|
133
|
+
kept-but-trailing branch still gets a probe. Deterministic; ties broken by input order."""
|
|
134
|
+
if pool_tokens <= 0 or not branches:
|
|
135
|
+
return Allocation({}, [b for b, _ in branches])
|
|
136
|
+
ranked = sorted(branches, key=lambda x: x[1], reverse=True)
|
|
137
|
+
survivors = [(b, s) for b, s in ranked if s >= prune_below]
|
|
138
|
+
if keep_top is not None:
|
|
139
|
+
survivors = survivors[:max(0, keep_top)]
|
|
140
|
+
kept_ids = {b for b, _ in survivors}
|
|
141
|
+
terminated = [b for b, _ in branches if b not in kept_ids]
|
|
142
|
+
if not survivors:
|
|
143
|
+
return Allocation({}, terminated)
|
|
144
|
+
weights = [(b, max(s, 0.0) ** bias) for b, s in survivors]
|
|
145
|
+
total_w = sum(w for _, w in weights) or float(len(weights))
|
|
146
|
+
grants: dict[str, int] = {}
|
|
147
|
+
for b, w in weights:
|
|
148
|
+
share = (w / total_w) if total_w else 1.0 / len(weights)
|
|
149
|
+
grants[b] = max(min_grant, int(round(pool_tokens * share)))
|
|
150
|
+
# correct rounding drift so grants sum to the pool (adjust the top branch)
|
|
151
|
+
drift = pool_tokens - sum(grants.values())
|
|
152
|
+
if drift and weights:
|
|
153
|
+
top = weights[0][0]
|
|
154
|
+
grants[top] = max(min_grant, grants[top] + drift)
|
|
155
|
+
return Allocation(grants, terminated)
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
# ── Planner: budget + intensity + hardware → population plan (§15/§16, §5/§9) ──
|
|
159
|
+
# Diverse roles for solution-space coverage (§9/§10). Correlated agents waste compute.
|
|
160
|
+
ROLES = [
|
|
161
|
+
"first-principles solver", "conventional solver", "skeptic", "debugger",
|
|
162
|
+
"security reviewer", "researcher", "minimalist", "adversarial critic",
|
|
163
|
+
"performance specialist", "alternative-architecture agent",
|
|
164
|
+
]
|
|
165
|
+
|
|
166
|
+
# Per-intensity search shape (§16): initial population, critic count, generation cap, and how
|
|
167
|
+
# many branches survive each prune. Population is a SEARCH choice — independent of hardware.
|
|
168
|
+
INTENSITY = {
|
|
169
|
+
"light": {"population": 3, "critics": 1, "generations": 2, "keep": 2},
|
|
170
|
+
"standard": {"population": 8, "critics": 2, "generations": 4, "keep": 4},
|
|
171
|
+
"deep": {"population": 16, "critics": 3, "generations": 6, "keep": 6},
|
|
172
|
+
"extreme": {"population": 24, "critics": 5, "generations": 10, "keep": 8},
|
|
173
|
+
}
|
|
174
|
+
|
|
175
|
+
|
|
176
|
+
@dataclass
|
|
177
|
+
class SwarmPlan:
|
|
178
|
+
intensity: str
|
|
179
|
+
initial_population: int # logical agents in the explore stage (§9)
|
|
180
|
+
max_concurrency: int # simultaneous inference — HARDWARE bound (§5), != population
|
|
181
|
+
critics: int # judge/critic agents (§11)
|
|
182
|
+
generations: int # generation cap (a convergence bound, §23)
|
|
183
|
+
keep: int # survivors promoted past each prune (§12)
|
|
184
|
+
roles: list[str] # diversified assignments (§10)
|
|
185
|
+
estimated_logical: int # rough total agents incl. spawned descendants (§13) — info only
|
|
186
|
+
|
|
187
|
+
@property
|
|
188
|
+
def queued_at_start(self) -> int:
|
|
189
|
+
"""Logical agents that must wait because concurrency < population (§18)."""
|
|
190
|
+
return max(0, self.initial_population - self.max_concurrency)
|
|
191
|
+
|
|
192
|
+
|
|
193
|
+
def choose_intensity(budget: Budget) -> str:
|
|
194
|
+
"""Map a budget to a search intensity for `--swarm auto` (§16). Bigger budget → deeper
|
|
195
|
+
search. Falls back to 'standard' when the budget carries no size signal."""
|
|
196
|
+
if budget.wall_secs is not None:
|
|
197
|
+
s = budget.wall_secs
|
|
198
|
+
return "light" if s < 900 else "standard" if s < 3600 else "deep" if s < 14400 else "extreme"
|
|
199
|
+
if budget.tokens is not None:
|
|
200
|
+
t = budget.tokens
|
|
201
|
+
return ("light" if t < 1_000_000 else "standard" if t < 10_000_000
|
|
202
|
+
else "deep" if t < 50_000_000 else "extreme")
|
|
203
|
+
if budget.generations is not None:
|
|
204
|
+
g = budget.generations
|
|
205
|
+
return "light" if g <= 2 else "standard" if g <= 4 else "deep" if g <= 6 else "extreme"
|
|
206
|
+
return "standard"
|
|
207
|
+
|
|
208
|
+
|
|
209
|
+
def plan(budget: Budget, concurrency: int, *, intensity: str = "auto") -> SwarmPlan:
|
|
210
|
+
"""Turn a budget + measured hardware concurrency into a population plan (§15/§16).
|
|
211
|
+
|
|
212
|
+
KEY: logical population comes from the intensity (a search decision); `concurrency` only
|
|
213
|
+
caps how many run at once (§5). On a 1-GPU box a 16-agent 'deep' plan still has 16 logical
|
|
214
|
+
agents — the rest queue (§18); the penalty is wall-clock, not lost search capability. This
|
|
215
|
+
is the opposite of capacity.swarm_size(), which caps logical count AT concurrency."""
|
|
216
|
+
intensity = choose_intensity(budget) if intensity == "auto" else intensity
|
|
217
|
+
if intensity not in INTENSITY:
|
|
218
|
+
raise ValueError(f"unknown intensity {intensity!r} (light|standard|deep|extreme|auto)")
|
|
219
|
+
p = INTENSITY[intensity]
|
|
220
|
+
pop = p["population"]
|
|
221
|
+
roles = [ROLES[i % len(ROLES)] for i in range(pop)]
|
|
222
|
+
# estimate total logical agents: initial + ~one descendant per survivor per later generation
|
|
223
|
+
estimated = pop + p["keep"] * max(0, p["generations"] - 1)
|
|
224
|
+
return SwarmPlan(intensity=intensity, initial_population=pop,
|
|
225
|
+
max_concurrency=max(1, concurrency), critics=p["critics"],
|
|
226
|
+
generations=p["generations"], keep=p["keep"], roles=roles,
|
|
227
|
+
estimated_logical=estimated)
|
|
@@ -27,6 +27,12 @@ DEFAULTS: dict[str, object] = {
|
|
|
27
27
|
# default invisibly, which left base_url absent from the file. For a non-vllm
|
|
28
28
|
# provider, override base_url (or use --base-url / the first-run prompt).
|
|
29
29
|
"base_url": "http://localhost:8000/v1",
|
|
30
|
+
# API key for the MAIN model endpoint, sent as `Authorization: Bearer <key>`.
|
|
31
|
+
# Leave "" for a local server that needs no auth (llama.cpp/Ollama/LM Studio);
|
|
32
|
+
# set it for a hosted OpenAI-compatible endpoint (e.g. a vLLM behind a gateway,
|
|
33
|
+
# OpenRouter, Together). Mirrors `advisor_api_key`. Can also come from the
|
|
34
|
+
# provider's api_key_env (e.g. OPENAI_API_KEY); the config value wins.
|
|
35
|
+
"api_key": "",
|
|
30
36
|
"max_tokens": 8192, # 4096 truncated large file writes mid-JSON (→ _raw fail)
|
|
31
37
|
"temperature": 0.2,
|
|
32
38
|
# The model server's context window (llama.cpp -c / vLLM --max-model-len).
|
|
@@ -270,14 +270,21 @@ class TaskOutcome:
|
|
|
270
270
|
|
|
271
271
|
def run_task(store: M.MissionStore, task: dict, *, mission: dict, cwd: str, repo: str,
|
|
272
272
|
worker: WorkerFn, evaluator: EvaluatorFn, base_config: dict | None = None,
|
|
273
|
-
worker_id: str = "worker-1"
|
|
274
|
-
|
|
273
|
+
worker_id: str = "worker-1",
|
|
274
|
+
checkpoint_fn: "Callable[[str], str]" = checkpoint,
|
|
275
|
+
restore_fn: "Callable[[str, str], bool]" = restore) -> TaskOutcome | None:
|
|
276
|
+
"""Execute one leased task with checkpoint + KEEP/REVERT. None if the lease was lost.
|
|
277
|
+
|
|
278
|
+
checkpoint_fn/restore_fn default to git (`repo` = a checkout dir), but are injectable so a
|
|
279
|
+
non-git target can plug in its own snapshot mechanism — e.g. `docker commit` for a task that
|
|
280
|
+
lives in a container (the tbench/ddt harness). `repo` is then just an opaque handle passed
|
|
281
|
+
to those callables; an empty `repo` disables checkpointing."""
|
|
275
282
|
mid = mission["id"]
|
|
276
283
|
tid = task["id"]
|
|
277
284
|
if not store.claim_task(tid, worker_id):
|
|
278
285
|
return None
|
|
279
286
|
before = mission.get("current_metric") or 0.0
|
|
280
|
-
cp =
|
|
287
|
+
cp = checkpoint_fn(repo) if repo else ""
|
|
281
288
|
store.event(mid, "task_started", task=tid, objective=task.get("objective", ""))
|
|
282
289
|
# Context reconstruction (§13): surface known failed approaches so the worker doesn't
|
|
283
290
|
# repeat them (§16). Key the lookup on the SPECIFIC target (task reason, e.g. the failing
|
|
@@ -295,7 +302,7 @@ def run_task(store: M.MissionStore, task: dict, *, mission: dict, cwd: str, repo
|
|
|
295
302
|
store.add_usage(mid, tokens=wr.in_tokens + wr.out_tokens, wall_s=time.monotonic() - t0)
|
|
296
303
|
if not wr.ok:
|
|
297
304
|
if repo and cp:
|
|
298
|
-
|
|
305
|
+
restore_fn(repo, cp)
|
|
299
306
|
store.complete_task(tid, M.T_FAILED, {"error": wr.error})
|
|
300
307
|
return TaskOutcome(tid, accept=False, metric_after=before, reverted=True, error=wr.error)
|
|
301
308
|
|
|
@@ -309,7 +316,7 @@ def run_task(store: M.MissionStore, task: dict, *, mission: dict, cwd: str, repo
|
|
|
309
316
|
tamper = tampered_paths(changed, protected)
|
|
310
317
|
if tamper:
|
|
311
318
|
if repo and cp:
|
|
312
|
-
|
|
319
|
+
restore_fn(repo, cp)
|
|
313
320
|
store.add_usage(mid, experiments=1)
|
|
314
321
|
store.complete_task(tid, M.T_FAILED,
|
|
315
322
|
{"summary": wr.summary, "decision": "REVERT", "reverted": True,
|
|
@@ -327,7 +334,7 @@ def run_task(store: M.MissionStore, task: dict, *, mission: dict, cwd: str, repo
|
|
|
327
334
|
ev = evaluator(task, mission, cwd, before)
|
|
328
335
|
store.add_usage(mid, experiments=1)
|
|
329
336
|
if ev.accept:
|
|
330
|
-
end =
|
|
337
|
+
end = checkpoint_fn(repo) if repo else ""
|
|
331
338
|
store.set_metric(mid, ev.metric_after)
|
|
332
339
|
store.complete_task(tid, M.T_COMPLETED,
|
|
333
340
|
{"summary": wr.summary, "metric": ev.metric_after, "decision": "KEEP",
|
|
@@ -340,7 +347,7 @@ def run_task(store: M.MissionStore, task: dict, *, mission: dict, cwd: str, repo
|
|
|
340
347
|
f"{(wr.summary or '')[:220]}", confidence=0.7, sources=[tid])
|
|
341
348
|
else:
|
|
342
349
|
if repo and cp:
|
|
343
|
-
|
|
350
|
+
restore_fn(repo, cp) # auto-revert the regression (§19)
|
|
344
351
|
store.complete_task(tid, M.T_COMPLETED,
|
|
345
352
|
{"summary": wr.summary, "decision": "REVERT", "reverted": True})
|
|
346
353
|
# negative knowledge stays in history even though the code is reverted (§16/§19)
|
|
@@ -505,6 +512,8 @@ def run_mission(store: M.MissionStore, mission_id: str, *, cwd: str, repo: str =
|
|
|
505
512
|
review_every_tasks: int = 10, review_every_secs: float = 3600.0,
|
|
506
513
|
reviewer: Callable[[M.MissionStore, dict, dict], dict] | None = None,
|
|
507
514
|
critic: Callable[[M.MissionStore, dict, dict], str] | None = None,
|
|
515
|
+
checkpoint_fn: "Callable[[str], str]" = checkpoint,
|
|
516
|
+
restore_fn: "Callable[[str, str], bool]" = restore,
|
|
508
517
|
on_event: Callable[[str, dict], None] | None = None) -> RunSummary:
|
|
509
518
|
"""Drive a mission to a stopping condition with a SINGLE worker (§41/§49). Stops on
|
|
510
519
|
success (§30), budget (§30), stagnation (§22), or an empty queue (planner is a later
|
|
@@ -557,7 +566,8 @@ def run_mission(store: M.MissionStore, mission_id: str, *, cwd: str, repo: str =
|
|
|
557
566
|
task = ready[0]
|
|
558
567
|
out = run_task(store, task, mission=m, cwd=cwd, repo=repo, worker=worker,
|
|
559
568
|
evaluator=(evaluator or (lambda *_: Evaluation(accept=True))),
|
|
560
|
-
base_config=base_config, worker_id=worker_id
|
|
569
|
+
base_config=base_config, worker_id=worker_id,
|
|
570
|
+
checkpoint_fn=checkpoint_fn, restore_fn=restore_fn)
|
|
561
571
|
progressed = bool(out and out.accept)
|
|
562
572
|
_ev("task_done", task=task["id"], accepted=progressed,
|
|
563
573
|
metric=out.metric_after if out else None)
|
|
@@ -6,6 +6,7 @@ drydock/__init__.py
|
|
|
6
6
|
drydock/__main__.py
|
|
7
7
|
drydock/advisor.py
|
|
8
8
|
drydock/agent.py
|
|
9
|
+
drydock/asr.py
|
|
9
10
|
drydock/bash_safety.py
|
|
10
11
|
drydock/bottleneck.py
|
|
11
12
|
drydock/budget.py
|
|
@@ -95,6 +96,7 @@ drydock_cli.egg-info/requires.txt
|
|
|
95
96
|
drydock_cli.egg-info/top_level.txt
|
|
96
97
|
tests/test_advisor.py
|
|
97
98
|
tests/test_approval.py
|
|
99
|
+
tests/test_asr.py
|
|
98
100
|
tests/test_back_command.py
|
|
99
101
|
tests/test_bash_background.py
|
|
100
102
|
tests/test_bash_binary_output.py
|
|
@@ -152,6 +154,7 @@ tests/test_jobs.py
|
|
|
152
154
|
tests/test_knowledge_tool_available.py
|
|
153
155
|
tests/test_leaked_tool_call.py
|
|
154
156
|
tests/test_loop_detect.py
|
|
157
|
+
tests/test_main_api_key.py
|
|
155
158
|
tests/test_mcp.py
|
|
156
159
|
tests/test_mission.py
|
|
157
160
|
tests/test_mission_cli.py
|
|
@@ -7,7 +7,7 @@ build-backend = "setuptools.build_meta"
|
|
|
7
7
|
# PyPI distribution name is drydock-cli (the established install name, continued
|
|
8
8
|
# from the retired v2 fork); the import package + CLI command stay `drydock`.
|
|
9
9
|
name = "drydock-cli"
|
|
10
|
-
version = "3.1.
|
|
10
|
+
version = "3.1.31"
|
|
11
11
|
description = "Drydock — a local, provider-agnostic terminal coding agent for local LLMs"
|
|
12
12
|
readme = "README.md"
|
|
13
13
|
requires-python = ">=3.11"
|
|
@@ -0,0 +1,140 @@
|
|
|
1
|
+
"""Tests for the Adaptive Swarm Runtime search core (drydock/asr.py) — pure, no model/GPU."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
import pytest
|
|
5
|
+
|
|
6
|
+
from drydock import asr
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
# ── Budget (§15) ────────────────────────────────────────────────────────────
|
|
10
|
+
def test_budget_parse_time_and_tokens():
|
|
11
|
+
assert asr.Budget.parse("2h").wall_secs == 7200
|
|
12
|
+
assert asr.Budget.parse("30m").wall_secs == 1800
|
|
13
|
+
assert asr.Budget.parse("90s").wall_secs == 90
|
|
14
|
+
assert asr.Budget.parse("100M-tokens").tokens == 100_000_000
|
|
15
|
+
assert asr.Budget.parse("50k-tokens").tokens == 50_000
|
|
16
|
+
assert asr.Budget.parse("2h").is_finite
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def test_budget_rejects_garbage_never_silently_unbounded():
|
|
20
|
+
with pytest.raises(ValueError):
|
|
21
|
+
asr.Budget.parse("whenever")
|
|
22
|
+
assert asr.Budget().is_finite is False # empty budget is explicitly not finite
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
# ── Convergence (§23) ─────────────────────────────────────────────────────────
|
|
26
|
+
def _conv(**kw):
|
|
27
|
+
return asr.Convergence(budget=asr.Budget(generations=100), **kw)
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def test_convergence_stops_on_success():
|
|
31
|
+
c = _conv(success_score=100.0)
|
|
32
|
+
c.record(best_score=100.0, cumulative_tokens=5000, elapsed_s=60)
|
|
33
|
+
go, why = c.should_continue()
|
|
34
|
+
assert not go and "success" in why
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def test_convergence_stops_on_budget():
|
|
38
|
+
c = asr.Convergence(budget=asr.Budget(tokens=10_000))
|
|
39
|
+
c.record(best_score=40.0, cumulative_tokens=12_000, elapsed_s=30)
|
|
40
|
+
go, why = c.should_continue()
|
|
41
|
+
assert not go and "token budget" in why
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def test_convergence_stops_on_no_improvement():
|
|
45
|
+
c = _conv(no_improve_gens=2, success_score=100.0)
|
|
46
|
+
c.record(50.0, 1000, 10)
|
|
47
|
+
c.record(50.0, 2000, 20) # no gain
|
|
48
|
+
c.record(50.0, 3000, 30) # still no gain -> 2 stagnant gens
|
|
49
|
+
go, why = c.should_continue()
|
|
50
|
+
assert not go and "no improvement" in why
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def test_convergence_stops_on_low_marginal_value():
|
|
54
|
+
# big jump, then a tiny gain for a lot of tokens -> marginal value below floor
|
|
55
|
+
c = _conv(min_marginal=1.0, no_improve_gens=0, success_score=100.0)
|
|
56
|
+
c.record(40.0, 1000, 10)
|
|
57
|
+
c.record(80.0, 3000, 20) # +40 over 2k tok = 20/1k -> continues
|
|
58
|
+
assert c.should_continue()[0] is True
|
|
59
|
+
c.record(80.5, 13000, 40) # +0.5 over 10k tok = 0.05/1k -> below 1.0 floor
|
|
60
|
+
go, why = c.should_continue()
|
|
61
|
+
assert not go and "marginal value" in why
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def test_convergence_continues_while_improving():
|
|
65
|
+
c = _conv(no_improve_gens=3, min_marginal=0.5, success_score=100.0)
|
|
66
|
+
c.record(40.0, 1000, 10)
|
|
67
|
+
c.record(60.0, 2000, 20) # +20/1k, improving, under budget
|
|
68
|
+
assert c.should_continue() == (True, "continue")
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def test_marginal_value_needs_two_records():
|
|
72
|
+
c = _conv()
|
|
73
|
+
assert c.marginal_value() is None
|
|
74
|
+
c.record(10.0, 500, 5)
|
|
75
|
+
assert c.marginal_value() is None
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
# ── Adaptive allocation (§17) ──────────────────────────────────────────────────
|
|
79
|
+
def test_allocate_favors_leaders_and_prunes_weak():
|
|
80
|
+
branches = [("A", 0.91), ("B", 0.86), ("C", 0.43), ("D", 0.21)]
|
|
81
|
+
alloc = asr.allocate(branches, 8000, prune_below=0.3)
|
|
82
|
+
assert "D" in alloc.terminated and alloc.grants.get("D", 0) == 0 # weakest pruned
|
|
83
|
+
assert set(alloc.grants) == {"A", "B", "C"}
|
|
84
|
+
assert sum(alloc.grants.values()) == 8000 # whole pool spent
|
|
85
|
+
assert alloc.grants["A"] > alloc.grants["B"] > alloc.grants["C"] # EV ordering
|
|
86
|
+
assert alloc.grants["C"] < alloc.grants["A"] / 2 # bias toward the leader
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def test_allocate_keep_top_terminates_the_rest():
|
|
90
|
+
branches = [("A", 0.9), ("B", 0.8), ("C", 0.7), ("D", 0.6)]
|
|
91
|
+
alloc = asr.allocate(branches, 6000, keep_top=2)
|
|
92
|
+
assert set(alloc.grants) == {"A", "B"}
|
|
93
|
+
assert sorted(alloc.terminated) == ["C", "D"]
|
|
94
|
+
assert sum(alloc.grants.values()) == 6000
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
def test_allocate_empty_pool_terminates_all():
|
|
98
|
+
alloc = asr.allocate([("A", 0.9), ("B", 0.5)], 0)
|
|
99
|
+
assert alloc.grants == {} and sorted(alloc.terminated) == ["A", "B"]
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
def test_allocate_min_grant_floor_for_kept_branches():
|
|
103
|
+
branches = [("A", 0.99), ("B", 0.01)]
|
|
104
|
+
alloc = asr.allocate(branches, 10_000, min_grant=200)
|
|
105
|
+
assert alloc.grants["B"] >= 200 # trailing-but-kept branch still probed
|
|
106
|
+
assert alloc.grants["A"] > alloc.grants["B"]
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
# ── Planner (§15/§16, §5/§18) ──────────────────────────────────────────────────
|
|
110
|
+
def test_choose_intensity_scales_with_budget():
|
|
111
|
+
assert asr.choose_intensity(asr.Budget.parse("10m")) == "light"
|
|
112
|
+
assert asr.choose_intensity(asr.Budget.parse("45m")) == "standard"
|
|
113
|
+
assert asr.choose_intensity(asr.Budget.parse("2h")) == "deep"
|
|
114
|
+
assert asr.choose_intensity(asr.Budget.parse("8h")) == "extreme"
|
|
115
|
+
assert asr.choose_intensity(asr.Budget.parse("5M-tokens")) == "standard"
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
def test_plan_decouples_logical_population_from_concurrency():
|
|
119
|
+
# THE core ASR property (§5/§18): a deep plan on a 1-slot box keeps all 16 logical agents;
|
|
120
|
+
# they queue, they aren't discarded down to concurrency (unlike capacity.swarm_size).
|
|
121
|
+
weak = asr.plan(asr.Budget.parse("2h"), concurrency=1, intensity="deep")
|
|
122
|
+
assert weak.initial_population == 16 and weak.max_concurrency == 1
|
|
123
|
+
assert weak.queued_at_start == 15
|
|
124
|
+
strong = asr.plan(asr.Budget.parse("2h"), concurrency=8, intensity="deep")
|
|
125
|
+
assert strong.initial_population == 16 and strong.max_concurrency == 8
|
|
126
|
+
assert strong.queued_at_start == 8 # same search, more parallelism → less queue
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
def test_plan_auto_and_intensity_ladder():
|
|
130
|
+
auto = asr.plan(asr.Budget.parse("30m"), concurrency=4) # auto -> standard
|
|
131
|
+
assert auto.intensity == "standard"
|
|
132
|
+
pops = [asr.plan(asr.Budget.parse("2h"), 4, intensity=i).initial_population
|
|
133
|
+
for i in ("light", "standard", "deep", "extreme")]
|
|
134
|
+
assert pops == sorted(pops) and pops[0] < pops[-1] # deeper => bigger population
|
|
135
|
+
assert len(auto.roles) == auto.initial_population and len(set(auto.roles)) > 1 # diverse
|
|
136
|
+
|
|
137
|
+
|
|
138
|
+
def test_plan_rejects_unknown_intensity():
|
|
139
|
+
with pytest.raises(ValueError):
|
|
140
|
+
asr.plan(asr.Budget.parse("1h"), 4, intensity="ludicrous")
|
|
@@ -0,0 +1,76 @@
|
|
|
1
|
+
"""Main-model API key in config.toml (mirrors advisor_api_key)."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
from drydock import config as C
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
def test_api_key_is_a_declared_default():
|
|
8
|
+
# the fix: api_key is a first-class default (so it appears in generated config.toml
|
|
9
|
+
# and survives resolve), not just something providers.py happened to read.
|
|
10
|
+
assert "api_key" in C.DEFAULTS and C.DEFAULTS["api_key"] == ""
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def test_config_file_api_key_round_trips(tmp_path):
|
|
14
|
+
p = tmp_path / "config.toml"
|
|
15
|
+
C.save_file({**C.DEFAULTS, "api_key": "sk-main-123"}, p)
|
|
16
|
+
assert "api_key" in p.read_text() # written to the file (discoverable)
|
|
17
|
+
cfg = C.resolve({}, p)
|
|
18
|
+
assert cfg["api_key"] == "sk-main-123" # and resolved back
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def test_generated_default_config_includes_api_key(tmp_path):
|
|
22
|
+
p = tmp_path / "config.toml"
|
|
23
|
+
C.resolve({}, p) # non-existent -> writes DEFAULTS
|
|
24
|
+
assert "api_key" in p.read_text()
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def test_stream_sends_config_api_key_to_client(monkeypatch):
|
|
28
|
+
import openai
|
|
29
|
+
|
|
30
|
+
from drydock import providers as P
|
|
31
|
+
captured: dict = {}
|
|
32
|
+
|
|
33
|
+
class _Stop(Exception):
|
|
34
|
+
pass
|
|
35
|
+
|
|
36
|
+
class _FakeClient:
|
|
37
|
+
def __init__(self, **kw):
|
|
38
|
+
captured.update(kw)
|
|
39
|
+
|
|
40
|
+
def __getattr__(self, _name): # any use (client.chat...) aborts after construction
|
|
41
|
+
raise _Stop()
|
|
42
|
+
|
|
43
|
+
monkeypatch.setattr(openai, "OpenAI", _FakeClient)
|
|
44
|
+
gen = P.stream("m", "sys", [{"role": "user", "content": "hi"}], [],
|
|
45
|
+
{"provider": "vllm", "base_url": "http://x/v1", "api_key": "sk-MAIN"})
|
|
46
|
+
try:
|
|
47
|
+
next(gen)
|
|
48
|
+
except Exception:
|
|
49
|
+
pass
|
|
50
|
+
assert captured.get("api_key") == "sk-MAIN" # main-model key reaches the client
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def test_stream_empty_api_key_falls_back_not_crashes(monkeypatch):
|
|
54
|
+
import openai
|
|
55
|
+
|
|
56
|
+
from drydock import providers as P
|
|
57
|
+
captured: dict = {}
|
|
58
|
+
|
|
59
|
+
class _Stop(Exception):
|
|
60
|
+
pass
|
|
61
|
+
|
|
62
|
+
class _FakeClient:
|
|
63
|
+
def __init__(self, **kw):
|
|
64
|
+
captured.update(kw)
|
|
65
|
+
|
|
66
|
+
def __getattr__(self, _name):
|
|
67
|
+
raise _Stop()
|
|
68
|
+
|
|
69
|
+
monkeypatch.setattr(openai, "OpenAI", _FakeClient)
|
|
70
|
+
gen = P.stream("m", "sys", [{"role": "user", "content": "hi"}], [],
|
|
71
|
+
{"provider": "vllm", "base_url": "http://x/v1", "api_key": ""})
|
|
72
|
+
try:
|
|
73
|
+
next(gen)
|
|
74
|
+
except Exception:
|
|
75
|
+
pass
|
|
76
|
+
assert captured.get("api_key") == "dummy" # empty -> provider default (unchanged behavior)
|
|
@@ -347,6 +347,33 @@ def test_restore_preserves_mission_workspace(tmp_path):
|
|
|
347
347
|
assert (ws / "state.db").read_text() == "STATE-V2" # mission state SURVIVES the revert
|
|
348
348
|
|
|
349
349
|
|
|
350
|
+
def test_injectable_checkpoint_restore(tmp_path):
|
|
351
|
+
# a non-git target (e.g. a container via `docker commit`) plugs in its own snapshot fns
|
|
352
|
+
snaps = {"made": [], "restored": []}
|
|
353
|
+
|
|
354
|
+
def cp_fn(handle):
|
|
355
|
+
ref = f"snap-{len(snaps['made'])}"
|
|
356
|
+
snaps["made"].append(ref)
|
|
357
|
+
return ref
|
|
358
|
+
|
|
359
|
+
def rs_fn(handle, ref):
|
|
360
|
+
snaps["restored"].append(ref)
|
|
361
|
+
return True
|
|
362
|
+
|
|
363
|
+
s = M.MissionStore(tmp_path / "s.db")
|
|
364
|
+
mid = s.create_mission("obj")
|
|
365
|
+
s.set_metric(mid, 50.0)
|
|
366
|
+
tid = s.add_task(mid, "risky")
|
|
367
|
+
out = R.run_task(s, s.get_task(tid), mission=s.get_mission(mid), cwd=".", repo="ctr:task",
|
|
368
|
+
worker=lambda *a: R.WorkerResult(ok=True, summary="tried"),
|
|
369
|
+
evaluator=lambda *a: R.Evaluation(accept=False, metric_before=50, metric_after=40,
|
|
370
|
+
reason="worse"),
|
|
371
|
+
checkpoint_fn=cp_fn, restore_fn=rs_fn)
|
|
372
|
+
assert out and out.reverted
|
|
373
|
+
assert snaps["made"] == ["snap-0"] # checkpointed via the injected fn (not git)
|
|
374
|
+
assert snaps["restored"] == ["snap-0"] # reverted via the injected fn
|
|
375
|
+
|
|
376
|
+
|
|
350
377
|
def test_verify_passes_reflects_the_tree(tmp_path):
|
|
351
378
|
repo = _repo(tmp_path)
|
|
352
379
|
(Path(repo) / "answer.txt").write_text("PASS")
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|