drydock-cli 3.1.30__tar.gz → 3.1.32__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {drydock_cli-3.1.30/drydock_cli.egg-info → drydock_cli-3.1.32}/PKG-INFO +1 -1
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/advisor.py +29 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/agent.py +3 -0
- drydock_cli-3.1.32/drydock/asr.py +227 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/config.py +6 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/mission_run.py +18 -8
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/tools/__init__.py +8 -1
- {drydock_cli-3.1.30 → drydock_cli-3.1.32/drydock_cli.egg-info}/PKG-INFO +1 -1
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock_cli.egg-info/SOURCES.txt +3 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/pyproject.toml +1 -1
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_advisor.py +55 -0
- drydock_cli-3.1.32/tests/test_asr.py +140 -0
- drydock_cli-3.1.32/tests/test_main_api_key.py +76 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_mission_run.py +27 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/LICENSE +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/NOTICE +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/README.md +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/__init__.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/__main__.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/bash_safety.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/bottleneck.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/budget.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/builtin_skills/__init__.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/builtin_skills/document-canvas.md +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/builtin_skills/fiar-assess.md +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/builtin_skills/fiar-cap.md +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/builtin_skills/fiar-evidence.md +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/builtin_skills/fiar-readiness.md +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/builtin_skills/ml-data.md +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/builtin_skills/ml-debug.md +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/builtin_skills/ml-finetune.md +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/builtin_skills/ml-metrics.md +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/builtin_skills/ml-rl.md +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/builtin_skills/ml-train.md +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/builtin_skills/nist-ai-rmf.md +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/builtin_skills/nist-csf.md +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/builtin_skills/rmf-categorize.md +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/builtin_skills/rmf-control.md +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/builtin_skills/rmf-poam.md +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/builtin_skills/rmf-review.md +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/builtin_skills/stig-assess.md +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/builtin_skills/stig-remediate.md +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/capacity.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/cci.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/cli.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/comms/__init__.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/comms/attention.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/comms/channels.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/comms/events.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/compaction.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/detect.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/doccanvas.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/eratchet.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/events.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/extract.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/fiar.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/gittools.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/graphrag.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/groundtruth.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/guards.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/jobs.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/loop_detect.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/mcp.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/mission.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/mission_cli.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/pdfredbox.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/phases.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/poam.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/predictions.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/progress.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/providers.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/ratchet.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/recipes.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/recovery.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/resume.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/rmf.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/rmf_graph.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/skills.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/stig.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/subagents.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/suggest.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/swarm.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/task_state.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/tool_policy.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/tool_registry.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/tool_result.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/tool_select.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/tool_validate.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/trajectory.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/tui/__init__.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/tui/app.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/tui/approval.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/tui/messages.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/tui/widgets.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/tuning.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/verification.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/web.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock_cli.egg-info/dependency_links.txt +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock_cli.egg-info/entry_points.txt +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock_cli.egg-info/requires.txt +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock_cli.egg-info/top_level.txt +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/setup.cfg +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_approval.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_back_command.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_bash_background.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_bash_binary_output.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_bash_crossplatform.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_bash_output_bounding.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_bash_process_group.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_bash_safety.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_bash_sanitize.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_bash_shell.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_bash_stdin.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_bash_stop_partial.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_bash_timeout_network.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_bash_timeout_param.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_bottleneck.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_budget.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_capacity.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_cci.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_cli_agents.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_comms.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_compact_command.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_compaction.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_config.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_config_migration.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_context_limit_config.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_context_limit_issue25.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_criteria_coverage.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_cycling_detection.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_declared_deps.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_degenerate_argument.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_detect.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_dispatch.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_doccanvas.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_e2e_connected.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_edit_replace_all.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_edit_thrash_and_compaction.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_effort_governor.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_empty_response.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_eratchet.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_events.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_extract.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_failure_loop.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_fiar.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_first_run_setup.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_gittools.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_glob_edit_edges.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_graphify_example.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_graphrag.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_graphrag_quoted_path.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_graphrag_sqlite.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_grep_and_read_robust.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_groundtruth_predictions.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_guards_and_tools.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_hallucinated_tools.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_jobs.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_knowledge_tool_available.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_leaked_tool_call.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_loop_detect.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_mcp.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_mission.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_mission_cli.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_model_registry.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_oneshot_unreachable.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_overthink_interrupt.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_pdfredbox.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_phases.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_plan_autocontinue.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_poam.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_progress.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_providers_unreachable.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_ratchet.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_ratchet_offer.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_read_index.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_recipes.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_recovery.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_recovery_config.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_recovery_integration.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_repeated_outcome.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_repetition_interrupt.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_resume_events.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_resume_restore.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_resume_snapshot.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_rmf.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_rmf_graph.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_rmf_stig_graph.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_rolling_plan.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_runaway_repetition.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_screenshot.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_server_probe.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_skills.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_sqlite_events.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_stall_retry.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_stig.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_stop.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_streaming_newlines.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_subagent.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_subagents.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_suggest.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_swarm.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_system_prompt_help.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_task_state.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_timeline.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_todo.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_tool_arg_coercion.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_tool_arg_coercion_more.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_tool_arg_parsing.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_tool_canonical.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_tool_policy.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_tool_result.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_tool_select.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_tool_validate.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_tools_undo.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_trajectory.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_tui.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_tuning.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_verification_gate.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_viewimage.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_vision_input.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_web_tools.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_windows_shell.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_worker_subagent.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_write_content_coerce.py +0 -0
- {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_xccdf.py +0 -0
|
@@ -24,6 +24,35 @@ _ADVISOR_SYSTEM = (
|
|
|
24
24
|
)
|
|
25
25
|
|
|
26
26
|
|
|
27
|
+
def _msg_text(content) -> str:
|
|
28
|
+
"""Flatten a message's content (str or multimodal block list) to plain text."""
|
|
29
|
+
if isinstance(content, str):
|
|
30
|
+
return content
|
|
31
|
+
if isinstance(content, list):
|
|
32
|
+
parts = [str(b.get("text") or b.get("content") or "") if isinstance(b, dict) else str(b)
|
|
33
|
+
for b in content]
|
|
34
|
+
return " ".join(p for p in parts if p)
|
|
35
|
+
return str(content or "")
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def recent_context(messages: list, *, max_msgs: int = 12, max_chars: int = 6000) -> str:
|
|
39
|
+
"""Compact the tail of the transcript to brief the advisor (§ Consult). The calling model —
|
|
40
|
+
especially a small local one — often passes little/no `context`, leaving the advisor blind;
|
|
41
|
+
this auto-attaches the last few turns (assistant reasoning + tool results/errors) so it can
|
|
42
|
+
actually help. Keeps the MOST RECENT content when truncating."""
|
|
43
|
+
if not messages:
|
|
44
|
+
return ""
|
|
45
|
+
out = []
|
|
46
|
+
for m in messages[-max_msgs:]:
|
|
47
|
+
if not isinstance(m, dict) or m.get("role") == "system":
|
|
48
|
+
continue
|
|
49
|
+
txt = _msg_text(m.get("content")).strip()
|
|
50
|
+
if txt:
|
|
51
|
+
out.append(f"[{m.get('role', '?')}] {txt[:1200]}")
|
|
52
|
+
blob = "\n".join(out)
|
|
53
|
+
return ("…" + blob[-max_chars:]) if len(blob) > max_chars else blob
|
|
54
|
+
|
|
55
|
+
|
|
27
56
|
def is_configured(config: dict) -> bool:
|
|
28
57
|
"""True when an advisor endpoint + model are set."""
|
|
29
58
|
return bool((config.get("advisor_base_url") or "").strip()
|
|
@@ -670,6 +670,9 @@ def run(
|
|
|
670
670
|
canonical=canonical_name(tc["name"]),
|
|
671
671
|
effect=tool_effect_of(tc["name"], _ro).value,
|
|
672
672
|
input=str(tc.get("input"))[:200])
|
|
673
|
+
# Expose the recent transcript so context-hungry tools (Consult) can brief
|
|
674
|
+
# the advisor without relying on the model to hand-pass context.
|
|
675
|
+
config["_recent_messages"] = state.messages
|
|
673
676
|
tool_result = execute_structured(tc["name"], tc["input"], config)
|
|
674
677
|
result = tool_result.text
|
|
675
678
|
# Emit the completion event IMMEDIATELY after execution (before
|
|
@@ -0,0 +1,227 @@
|
|
|
1
|
+
"""Adaptive Swarm Runtime (ASR) — the search-decision core.
|
|
2
|
+
|
|
3
|
+
This is the resource-agnostic brain of the adaptive swarm (PRD "Resource-Aware Adaptive
|
|
4
|
+
Swarms"): it turns a compute/time BUDGET into decisions about *where the next unit of inference
|
|
5
|
+
should go* and *when to stop*. It deliberately holds no model, GPU, or scheduler state — those
|
|
6
|
+
live in capacity.py (concurrency), swarm.py (blackboard/candidates/worktrees) and the mission
|
|
7
|
+
store (durable state). Keeping the decisions pure makes them unit-testable and reusable.
|
|
8
|
+
|
|
9
|
+
Three pieces, in dependency order:
|
|
10
|
+
* Budget — a finite termination envelope parsed from a user string (§15).
|
|
11
|
+
* Convergence — should the swarm keep going? Stops on success / budget / no-improvement /
|
|
12
|
+
when marginal value Δquality/Δcompute falls below a floor (§23).
|
|
13
|
+
* AdaptiveAllocator — split the next compute pool across live branches by expected value,
|
|
14
|
+
pruning the weak ones (§17). Bad ideas must not get the same compute as good.
|
|
15
|
+
"""
|
|
16
|
+
from __future__ import annotations
|
|
17
|
+
|
|
18
|
+
import re
|
|
19
|
+
from dataclasses import dataclass, field
|
|
20
|
+
|
|
21
|
+
# ── Budget (§15) ────────────────────────────────────────────────────────────
|
|
22
|
+
_DUR = re.compile(r"^\s*([\d.]+)\s*([hms])\s*$", re.I)
|
|
23
|
+
_TOK = re.compile(r"^\s*([\d.]+)\s*([kmg]?)\s*[-_]?tokens?\s*$", re.I)
|
|
24
|
+
_MULT = {"": 1, "k": 1_000, "m": 1_000_000, "g": 1_000_000_000}
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
@dataclass
|
|
28
|
+
class Budget:
|
|
29
|
+
"""A finite stopping envelope (§10/§15). Every field is optional, but a Budget with no
|
|
30
|
+
finite limit is a bug — `is_finite` guards against "run forever"."""
|
|
31
|
+
wall_secs: float | None = None
|
|
32
|
+
tokens: int | None = None
|
|
33
|
+
generations: int | None = None
|
|
34
|
+
|
|
35
|
+
@property
|
|
36
|
+
def is_finite(self) -> bool:
|
|
37
|
+
return any(v is not None for v in (self.wall_secs, self.tokens, self.generations))
|
|
38
|
+
|
|
39
|
+
@classmethod
|
|
40
|
+
def parse(cls, spec: str) -> "Budget":
|
|
41
|
+
"""Parse one user token: '2h' / '30m' / '90s' (wall) or '100M-tokens' / '50k-tokens'
|
|
42
|
+
(compute). Unknown strings raise ValueError so a typo never silently means 'unbounded'."""
|
|
43
|
+
s = (spec or "").strip()
|
|
44
|
+
md = _DUR.match(s)
|
|
45
|
+
if md:
|
|
46
|
+
n, unit = float(md.group(1)), md.group(2).lower()
|
|
47
|
+
return cls(wall_secs=n * {"h": 3600, "m": 60, "s": 1}[unit])
|
|
48
|
+
mt = _TOK.match(s)
|
|
49
|
+
if mt:
|
|
50
|
+
return cls(tokens=int(float(mt.group(1)) * _MULT[mt.group(2).lower()]))
|
|
51
|
+
raise ValueError(f"unparseable budget: {spec!r} (try '2h', '30m', '100M-tokens')")
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
# ── Convergence (§23) ─────────────────────────────────────────────────────────
|
|
55
|
+
@dataclass
|
|
56
|
+
class GenerationRecord:
|
|
57
|
+
gen: int
|
|
58
|
+
best_score: float # best branch quality so far (higher = better)
|
|
59
|
+
cumulative_tokens: int # total inference tokens spent through this generation
|
|
60
|
+
elapsed_s: float # wall-clock since start
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
@dataclass
|
|
64
|
+
class Convergence:
|
|
65
|
+
"""Decides whether the swarm should run another generation (§23). "More agents are not
|
|
66
|
+
automatically better": we stop the moment any finite condition trips, INCLUDING the marginal
|
|
67
|
+
one — when Δquality per unit compute drops below `min_marginal`, extra inference is waste."""
|
|
68
|
+
budget: Budget
|
|
69
|
+
success_score: float = 100.0 # stop when best branch reaches this (e.g. all tests pass)
|
|
70
|
+
no_improve_gens: int = 3 # stop after this many generations with no best-score gain
|
|
71
|
+
min_marginal: float = 0.0 # floor on Δscore per 1k tokens; <=0 disables the check
|
|
72
|
+
eps: float = 1e-9
|
|
73
|
+
history: list[GenerationRecord] = field(default_factory=list)
|
|
74
|
+
|
|
75
|
+
def record(self, best_score: float, cumulative_tokens: int, elapsed_s: float) -> None:
|
|
76
|
+
self.history.append(GenerationRecord(len(self.history) + 1, best_score,
|
|
77
|
+
cumulative_tokens, elapsed_s))
|
|
78
|
+
|
|
79
|
+
def _gens_since_improvement(self) -> int:
|
|
80
|
+
"""Generations since the best score was FIRST reached (a plateau at the best counts)."""
|
|
81
|
+
if not self.history:
|
|
82
|
+
return 0
|
|
83
|
+
best = max(r.best_score for r in self.history)
|
|
84
|
+
first_best = next(i for i, r in enumerate(self.history) if r.best_score >= best - self.eps)
|
|
85
|
+
return (len(self.history) - 1) - first_best
|
|
86
|
+
|
|
87
|
+
def marginal_value(self) -> float | None:
|
|
88
|
+
"""Δscore per 1,000 tokens over the last generation, or None with <2 records."""
|
|
89
|
+
if len(self.history) < 2:
|
|
90
|
+
return None
|
|
91
|
+
a, b = self.history[-2], self.history[-1]
|
|
92
|
+
dtok = b.cumulative_tokens - a.cumulative_tokens
|
|
93
|
+
if dtok <= 0:
|
|
94
|
+
return None
|
|
95
|
+
return (b.best_score - a.best_score) / (dtok / 1000.0)
|
|
96
|
+
|
|
97
|
+
def should_continue(self) -> tuple[bool, str]:
|
|
98
|
+
"""(keep_going?, reason). Checked in priority order; the reason names the stop trigger."""
|
|
99
|
+
last = self.history[-1] if self.history else None
|
|
100
|
+
if last is not None and last.best_score >= self.success_score - self.eps:
|
|
101
|
+
return False, "success score reached"
|
|
102
|
+
b = self.budget
|
|
103
|
+
if b.wall_secs is not None and last is not None and last.elapsed_s >= b.wall_secs:
|
|
104
|
+
return False, "wall-clock budget exhausted"
|
|
105
|
+
if b.tokens is not None and last is not None and last.cumulative_tokens >= b.tokens:
|
|
106
|
+
return False, "token budget exhausted"
|
|
107
|
+
if b.generations is not None and len(self.history) >= b.generations:
|
|
108
|
+
return False, "generation budget reached"
|
|
109
|
+
if self.no_improve_gens > 0 and self._gens_since_improvement() >= self.no_improve_gens:
|
|
110
|
+
return False, f"no improvement in {self.no_improve_gens} generations"
|
|
111
|
+
mv = self.marginal_value()
|
|
112
|
+
if self.min_marginal > 0 and mv is not None and mv < self.min_marginal:
|
|
113
|
+
return False, f"marginal value {mv:.3f} < floor {self.min_marginal:.3f} (Δquality/Δcompute)"
|
|
114
|
+
return True, "continue"
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
# ── Adaptive compute allocation (§17) ──────────────────────────────────────────
|
|
118
|
+
@dataclass
|
|
119
|
+
class Allocation:
|
|
120
|
+
grants: dict[str, int] # branch id -> tokens granted this round
|
|
121
|
+
terminated: list[str] # branch ids pruned (0 tokens, killed)
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
def allocate(branches: list[tuple[str, float]], pool_tokens: int, *,
|
|
125
|
+
prune_below: float = 0.0, keep_top: int | None = None,
|
|
126
|
+
bias: float = 1.5, min_grant: int = 0) -> Allocation:
|
|
127
|
+
"""Split `pool_tokens` across branches by expected value (§17).
|
|
128
|
+
|
|
129
|
+
branches: (id, score) with higher score = more promising. A branch is TERMINATED (gets 0)
|
|
130
|
+
if its score is below `prune_below` or it falls outside the top `keep_top`. Survivors share
|
|
131
|
+
the pool in proportion to score**bias, so leaders get disproportionately more (bias>1) — the
|
|
132
|
+
system must not spend equally on good and bad ideas. `min_grant` guarantees a floor so a
|
|
133
|
+
kept-but-trailing branch still gets a probe. Deterministic; ties broken by input order."""
|
|
134
|
+
if pool_tokens <= 0 or not branches:
|
|
135
|
+
return Allocation({}, [b for b, _ in branches])
|
|
136
|
+
ranked = sorted(branches, key=lambda x: x[1], reverse=True)
|
|
137
|
+
survivors = [(b, s) for b, s in ranked if s >= prune_below]
|
|
138
|
+
if keep_top is not None:
|
|
139
|
+
survivors = survivors[:max(0, keep_top)]
|
|
140
|
+
kept_ids = {b for b, _ in survivors}
|
|
141
|
+
terminated = [b for b, _ in branches if b not in kept_ids]
|
|
142
|
+
if not survivors:
|
|
143
|
+
return Allocation({}, terminated)
|
|
144
|
+
weights = [(b, max(s, 0.0) ** bias) for b, s in survivors]
|
|
145
|
+
total_w = sum(w for _, w in weights) or float(len(weights))
|
|
146
|
+
grants: dict[str, int] = {}
|
|
147
|
+
for b, w in weights:
|
|
148
|
+
share = (w / total_w) if total_w else 1.0 / len(weights)
|
|
149
|
+
grants[b] = max(min_grant, int(round(pool_tokens * share)))
|
|
150
|
+
# correct rounding drift so grants sum to the pool (adjust the top branch)
|
|
151
|
+
drift = pool_tokens - sum(grants.values())
|
|
152
|
+
if drift and weights:
|
|
153
|
+
top = weights[0][0]
|
|
154
|
+
grants[top] = max(min_grant, grants[top] + drift)
|
|
155
|
+
return Allocation(grants, terminated)
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
# ── Planner: budget + intensity + hardware → population plan (§15/§16, §5/§9) ──
|
|
159
|
+
# Diverse roles for solution-space coverage (§9/§10). Correlated agents waste compute.
|
|
160
|
+
ROLES = [
|
|
161
|
+
"first-principles solver", "conventional solver", "skeptic", "debugger",
|
|
162
|
+
"security reviewer", "researcher", "minimalist", "adversarial critic",
|
|
163
|
+
"performance specialist", "alternative-architecture agent",
|
|
164
|
+
]
|
|
165
|
+
|
|
166
|
+
# Per-intensity search shape (§16): initial population, critic count, generation cap, and how
|
|
167
|
+
# many branches survive each prune. Population is a SEARCH choice — independent of hardware.
|
|
168
|
+
INTENSITY = {
|
|
169
|
+
"light": {"population": 3, "critics": 1, "generations": 2, "keep": 2},
|
|
170
|
+
"standard": {"population": 8, "critics": 2, "generations": 4, "keep": 4},
|
|
171
|
+
"deep": {"population": 16, "critics": 3, "generations": 6, "keep": 6},
|
|
172
|
+
"extreme": {"population": 24, "critics": 5, "generations": 10, "keep": 8},
|
|
173
|
+
}
|
|
174
|
+
|
|
175
|
+
|
|
176
|
+
@dataclass
|
|
177
|
+
class SwarmPlan:
|
|
178
|
+
intensity: str
|
|
179
|
+
initial_population: int # logical agents in the explore stage (§9)
|
|
180
|
+
max_concurrency: int # simultaneous inference — HARDWARE bound (§5), != population
|
|
181
|
+
critics: int # judge/critic agents (§11)
|
|
182
|
+
generations: int # generation cap (a convergence bound, §23)
|
|
183
|
+
keep: int # survivors promoted past each prune (§12)
|
|
184
|
+
roles: list[str] # diversified assignments (§10)
|
|
185
|
+
estimated_logical: int # rough total agents incl. spawned descendants (§13) — info only
|
|
186
|
+
|
|
187
|
+
@property
|
|
188
|
+
def queued_at_start(self) -> int:
|
|
189
|
+
"""Logical agents that must wait because concurrency < population (§18)."""
|
|
190
|
+
return max(0, self.initial_population - self.max_concurrency)
|
|
191
|
+
|
|
192
|
+
|
|
193
|
+
def choose_intensity(budget: Budget) -> str:
|
|
194
|
+
"""Map a budget to a search intensity for `--swarm auto` (§16). Bigger budget → deeper
|
|
195
|
+
search. Falls back to 'standard' when the budget carries no size signal."""
|
|
196
|
+
if budget.wall_secs is not None:
|
|
197
|
+
s = budget.wall_secs
|
|
198
|
+
return "light" if s < 900 else "standard" if s < 3600 else "deep" if s < 14400 else "extreme"
|
|
199
|
+
if budget.tokens is not None:
|
|
200
|
+
t = budget.tokens
|
|
201
|
+
return ("light" if t < 1_000_000 else "standard" if t < 10_000_000
|
|
202
|
+
else "deep" if t < 50_000_000 else "extreme")
|
|
203
|
+
if budget.generations is not None:
|
|
204
|
+
g = budget.generations
|
|
205
|
+
return "light" if g <= 2 else "standard" if g <= 4 else "deep" if g <= 6 else "extreme"
|
|
206
|
+
return "standard"
|
|
207
|
+
|
|
208
|
+
|
|
209
|
+
def plan(budget: Budget, concurrency: int, *, intensity: str = "auto") -> SwarmPlan:
|
|
210
|
+
"""Turn a budget + measured hardware concurrency into a population plan (§15/§16).
|
|
211
|
+
|
|
212
|
+
KEY: logical population comes from the intensity (a search decision); `concurrency` only
|
|
213
|
+
caps how many run at once (§5). On a 1-GPU box a 16-agent 'deep' plan still has 16 logical
|
|
214
|
+
agents — the rest queue (§18); the penalty is wall-clock, not lost search capability. This
|
|
215
|
+
is the opposite of capacity.swarm_size(), which caps logical count AT concurrency."""
|
|
216
|
+
intensity = choose_intensity(budget) if intensity == "auto" else intensity
|
|
217
|
+
if intensity not in INTENSITY:
|
|
218
|
+
raise ValueError(f"unknown intensity {intensity!r} (light|standard|deep|extreme|auto)")
|
|
219
|
+
p = INTENSITY[intensity]
|
|
220
|
+
pop = p["population"]
|
|
221
|
+
roles = [ROLES[i % len(ROLES)] for i in range(pop)]
|
|
222
|
+
# estimate total logical agents: initial + ~one descendant per survivor per later generation
|
|
223
|
+
estimated = pop + p["keep"] * max(0, p["generations"] - 1)
|
|
224
|
+
return SwarmPlan(intensity=intensity, initial_population=pop,
|
|
225
|
+
max_concurrency=max(1, concurrency), critics=p["critics"],
|
|
226
|
+
generations=p["generations"], keep=p["keep"], roles=roles,
|
|
227
|
+
estimated_logical=estimated)
|
|
@@ -27,6 +27,12 @@ DEFAULTS: dict[str, object] = {
|
|
|
27
27
|
# default invisibly, which left base_url absent from the file. For a non-vllm
|
|
28
28
|
# provider, override base_url (or use --base-url / the first-run prompt).
|
|
29
29
|
"base_url": "http://localhost:8000/v1",
|
|
30
|
+
# API key for the MAIN model endpoint, sent as `Authorization: Bearer <key>`.
|
|
31
|
+
# Leave "" for a local server that needs no auth (llama.cpp/Ollama/LM Studio);
|
|
32
|
+
# set it for a hosted OpenAI-compatible endpoint (e.g. a vLLM behind a gateway,
|
|
33
|
+
# OpenRouter, Together). Mirrors `advisor_api_key`. Can also come from the
|
|
34
|
+
# provider's api_key_env (e.g. OPENAI_API_KEY); the config value wins.
|
|
35
|
+
"api_key": "",
|
|
30
36
|
"max_tokens": 8192, # 4096 truncated large file writes mid-JSON (→ _raw fail)
|
|
31
37
|
"temperature": 0.2,
|
|
32
38
|
# The model server's context window (llama.cpp -c / vLLM --max-model-len).
|
|
@@ -270,14 +270,21 @@ class TaskOutcome:
|
|
|
270
270
|
|
|
271
271
|
def run_task(store: M.MissionStore, task: dict, *, mission: dict, cwd: str, repo: str,
|
|
272
272
|
worker: WorkerFn, evaluator: EvaluatorFn, base_config: dict | None = None,
|
|
273
|
-
worker_id: str = "worker-1"
|
|
274
|
-
|
|
273
|
+
worker_id: str = "worker-1",
|
|
274
|
+
checkpoint_fn: "Callable[[str], str]" = checkpoint,
|
|
275
|
+
restore_fn: "Callable[[str, str], bool]" = restore) -> TaskOutcome | None:
|
|
276
|
+
"""Execute one leased task with checkpoint + KEEP/REVERT. None if the lease was lost.
|
|
277
|
+
|
|
278
|
+
checkpoint_fn/restore_fn default to git (`repo` = a checkout dir), but are injectable so a
|
|
279
|
+
non-git target can plug in its own snapshot mechanism — e.g. `docker commit` for a task that
|
|
280
|
+
lives in a container (the tbench/ddt harness). `repo` is then just an opaque handle passed
|
|
281
|
+
to those callables; an empty `repo` disables checkpointing."""
|
|
275
282
|
mid = mission["id"]
|
|
276
283
|
tid = task["id"]
|
|
277
284
|
if not store.claim_task(tid, worker_id):
|
|
278
285
|
return None
|
|
279
286
|
before = mission.get("current_metric") or 0.0
|
|
280
|
-
cp =
|
|
287
|
+
cp = checkpoint_fn(repo) if repo else ""
|
|
281
288
|
store.event(mid, "task_started", task=tid, objective=task.get("objective", ""))
|
|
282
289
|
# Context reconstruction (§13): surface known failed approaches so the worker doesn't
|
|
283
290
|
# repeat them (§16). Key the lookup on the SPECIFIC target (task reason, e.g. the failing
|
|
@@ -295,7 +302,7 @@ def run_task(store: M.MissionStore, task: dict, *, mission: dict, cwd: str, repo
|
|
|
295
302
|
store.add_usage(mid, tokens=wr.in_tokens + wr.out_tokens, wall_s=time.monotonic() - t0)
|
|
296
303
|
if not wr.ok:
|
|
297
304
|
if repo and cp:
|
|
298
|
-
|
|
305
|
+
restore_fn(repo, cp)
|
|
299
306
|
store.complete_task(tid, M.T_FAILED, {"error": wr.error})
|
|
300
307
|
return TaskOutcome(tid, accept=False, metric_after=before, reverted=True, error=wr.error)
|
|
301
308
|
|
|
@@ -309,7 +316,7 @@ def run_task(store: M.MissionStore, task: dict, *, mission: dict, cwd: str, repo
|
|
|
309
316
|
tamper = tampered_paths(changed, protected)
|
|
310
317
|
if tamper:
|
|
311
318
|
if repo and cp:
|
|
312
|
-
|
|
319
|
+
restore_fn(repo, cp)
|
|
313
320
|
store.add_usage(mid, experiments=1)
|
|
314
321
|
store.complete_task(tid, M.T_FAILED,
|
|
315
322
|
{"summary": wr.summary, "decision": "REVERT", "reverted": True,
|
|
@@ -327,7 +334,7 @@ def run_task(store: M.MissionStore, task: dict, *, mission: dict, cwd: str, repo
|
|
|
327
334
|
ev = evaluator(task, mission, cwd, before)
|
|
328
335
|
store.add_usage(mid, experiments=1)
|
|
329
336
|
if ev.accept:
|
|
330
|
-
end =
|
|
337
|
+
end = checkpoint_fn(repo) if repo else ""
|
|
331
338
|
store.set_metric(mid, ev.metric_after)
|
|
332
339
|
store.complete_task(tid, M.T_COMPLETED,
|
|
333
340
|
{"summary": wr.summary, "metric": ev.metric_after, "decision": "KEEP",
|
|
@@ -340,7 +347,7 @@ def run_task(store: M.MissionStore, task: dict, *, mission: dict, cwd: str, repo
|
|
|
340
347
|
f"{(wr.summary or '')[:220]}", confidence=0.7, sources=[tid])
|
|
341
348
|
else:
|
|
342
349
|
if repo and cp:
|
|
343
|
-
|
|
350
|
+
restore_fn(repo, cp) # auto-revert the regression (§19)
|
|
344
351
|
store.complete_task(tid, M.T_COMPLETED,
|
|
345
352
|
{"summary": wr.summary, "decision": "REVERT", "reverted": True})
|
|
346
353
|
# negative knowledge stays in history even though the code is reverted (§16/§19)
|
|
@@ -505,6 +512,8 @@ def run_mission(store: M.MissionStore, mission_id: str, *, cwd: str, repo: str =
|
|
|
505
512
|
review_every_tasks: int = 10, review_every_secs: float = 3600.0,
|
|
506
513
|
reviewer: Callable[[M.MissionStore, dict, dict], dict] | None = None,
|
|
507
514
|
critic: Callable[[M.MissionStore, dict, dict], str] | None = None,
|
|
515
|
+
checkpoint_fn: "Callable[[str], str]" = checkpoint,
|
|
516
|
+
restore_fn: "Callable[[str, str], bool]" = restore,
|
|
508
517
|
on_event: Callable[[str, dict], None] | None = None) -> RunSummary:
|
|
509
518
|
"""Drive a mission to a stopping condition with a SINGLE worker (§41/§49). Stops on
|
|
510
519
|
success (§30), budget (§30), stagnation (§22), or an empty queue (planner is a later
|
|
@@ -557,7 +566,8 @@ def run_mission(store: M.MissionStore, mission_id: str, *, cwd: str, repo: str =
|
|
|
557
566
|
task = ready[0]
|
|
558
567
|
out = run_task(store, task, mission=m, cwd=cwd, repo=repo, worker=worker,
|
|
559
568
|
evaluator=(evaluator or (lambda *_: Evaluation(accept=True))),
|
|
560
|
-
base_config=base_config, worker_id=worker_id
|
|
569
|
+
base_config=base_config, worker_id=worker_id,
|
|
570
|
+
checkpoint_fn=checkpoint_fn, restore_fn=restore_fn)
|
|
561
571
|
progressed = bool(out and out.accept)
|
|
562
572
|
_ev("task_done", task=task["id"], accepted=progressed,
|
|
563
573
|
metric=out.metric_after if out else None)
|
|
@@ -2054,7 +2054,14 @@ def tool_consult(params: dict, config: dict) -> str:
|
|
|
2054
2054
|
question = _as_str_arg(params.get("question") or params.get("prompt")).strip()
|
|
2055
2055
|
if not question:
|
|
2056
2056
|
return "Error: Consult needs a `question`."
|
|
2057
|
-
|
|
2057
|
+
ctx = _as_str_arg(params.get("context") or "").strip()
|
|
2058
|
+
# Auto-enrich: the advisor is useless without knowing what's going on, and the calling
|
|
2059
|
+
# model often passes thin/no context. Attach the recent transcript (errors + reasoning)
|
|
2060
|
+
# that the agent stashed in config so the advisor always has real information to work with.
|
|
2061
|
+
auto = advisor.recent_context(config.get("_recent_messages") or [])
|
|
2062
|
+
if auto:
|
|
2063
|
+
ctx = (ctx + "\n\n" if ctx else "") + "Recent activity (auto-attached):\n" + auto
|
|
2064
|
+
return advisor.consult(question, config, context=ctx)
|
|
2058
2065
|
|
|
2059
2066
|
|
|
2060
2067
|
# A sub-agent's whole job is to keep its investigation OUT of the main agent's
|
|
@@ -6,6 +6,7 @@ drydock/__init__.py
|
|
|
6
6
|
drydock/__main__.py
|
|
7
7
|
drydock/advisor.py
|
|
8
8
|
drydock/agent.py
|
|
9
|
+
drydock/asr.py
|
|
9
10
|
drydock/bash_safety.py
|
|
10
11
|
drydock/bottleneck.py
|
|
11
12
|
drydock/budget.py
|
|
@@ -95,6 +96,7 @@ drydock_cli.egg-info/requires.txt
|
|
|
95
96
|
drydock_cli.egg-info/top_level.txt
|
|
96
97
|
tests/test_advisor.py
|
|
97
98
|
tests/test_approval.py
|
|
99
|
+
tests/test_asr.py
|
|
98
100
|
tests/test_back_command.py
|
|
99
101
|
tests/test_bash_background.py
|
|
100
102
|
tests/test_bash_binary_output.py
|
|
@@ -152,6 +154,7 @@ tests/test_jobs.py
|
|
|
152
154
|
tests/test_knowledge_tool_available.py
|
|
153
155
|
tests/test_leaked_tool_call.py
|
|
154
156
|
tests/test_loop_detect.py
|
|
157
|
+
tests/test_main_api_key.py
|
|
155
158
|
tests/test_mcp.py
|
|
156
159
|
tests/test_mission.py
|
|
157
160
|
tests/test_mission_cli.py
|
|
@@ -7,7 +7,7 @@ build-backend = "setuptools.build_meta"
|
|
|
7
7
|
# PyPI distribution name is drydock-cli (the established install name, continued
|
|
8
8
|
# from the retired v2 fork); the import package + CLI command stay `drydock`.
|
|
9
9
|
name = "drydock-cli"
|
|
10
|
-
version = "3.1.
|
|
10
|
+
version = "3.1.32"
|
|
11
11
|
description = "Drydock — a local, provider-agnostic terminal coding agent for local LLMs"
|
|
12
12
|
readme = "README.md"
|
|
13
13
|
requires-python = ">=3.11"
|
|
@@ -88,3 +88,58 @@ def test_test_connection_timeout_says_reachable_but_slow(monkeypatch):
|
|
|
88
88
|
monkeypatch.setattr(advisor, "_call", slow)
|
|
89
89
|
out = advisor.test_connection({"advisor_base_url": "http://b/v1", "advisor_model": "m"})
|
|
90
90
|
assert out.startswith("✗") and "REACHABLE but slow" in out and "/ask" in out
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
# ── recent-context auto-enrichment (advisor gets info even w/ thin caller context) ──
|
|
94
|
+
def test_recent_context_formats_and_keeps_newest():
|
|
95
|
+
msgs = [
|
|
96
|
+
{"role": "system", "content": "SYS should be skipped"},
|
|
97
|
+
{"role": "assistant", "content": "I'll run the tests."},
|
|
98
|
+
{"role": "tool", "content": "FAILED test_x - AssertionError: expected 42"},
|
|
99
|
+
{"role": "assistant", "content": "The assertion failed; investigating add()."},
|
|
100
|
+
]
|
|
101
|
+
out = advisor.recent_context(msgs)
|
|
102
|
+
assert "SYS should be skipped" not in out # system dropped
|
|
103
|
+
assert "FAILED test_x" in out and "investigating add()" in out
|
|
104
|
+
assert "[tool]" in out and "[assistant]" in out
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
def test_recent_context_truncates_to_recent_chars():
|
|
108
|
+
msgs = [{"role": "assistant", "content": "x" * 500} for _ in range(40)]
|
|
109
|
+
out = advisor.recent_context(msgs, max_chars=1000)
|
|
110
|
+
assert len(out) <= 1001 and out.startswith("…") # kept the tail
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
def test_recent_context_handles_multimodal_blocks():
|
|
114
|
+
msgs = [{"role": "user", "content": [{"type": "text", "text": "look at this"}, {"type": "image"}]}]
|
|
115
|
+
assert "look at this" in advisor.recent_context(msgs)
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
def test_consult_tool_auto_attaches_recent_transcript(monkeypatch):
|
|
119
|
+
from drydock import tools as T
|
|
120
|
+
captured = {}
|
|
121
|
+
monkeypatch.setattr(advisor, "consult",
|
|
122
|
+
lambda q, cfg, context="": captured.update(question=q, context=context) or "ok")
|
|
123
|
+
cfg = {
|
|
124
|
+
"advisor_base_url": "http://b/v1", "advisor_model": "m",
|
|
125
|
+
"_recent_messages": [
|
|
126
|
+
{"role": "tool", "content": "FAILED test_reconnect - state lost after reconnect"},
|
|
127
|
+
{"role": "assistant", "content": "Trying a lock around the queue."},
|
|
128
|
+
],
|
|
129
|
+
}
|
|
130
|
+
# model passed only a bare question (thin context) — the fix must attach the transcript
|
|
131
|
+
T.tool_consult({"question": "why does reconnect lose state?"}, cfg)
|
|
132
|
+
assert "FAILED test_reconnect" in captured["context"] # advisor now sees the error
|
|
133
|
+
assert "lock around the queue" in captured["context"]
|
|
134
|
+
assert "Recent activity" in captured["context"]
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
def test_consult_tool_keeps_model_context_and_adds_recent(monkeypatch):
|
|
138
|
+
from drydock import tools as T
|
|
139
|
+
captured = {}
|
|
140
|
+
monkeypatch.setattr(advisor, "consult",
|
|
141
|
+
lambda q, cfg, context="": captured.update(context=context) or "ok")
|
|
142
|
+
cfg = {"advisor_base_url": "http://b/v1", "advisor_model": "m",
|
|
143
|
+
"_recent_messages": [{"role": "tool", "content": "ERR: boom"}]}
|
|
144
|
+
T.tool_consult({"question": "help", "context": "my hand-written context"}, cfg)
|
|
145
|
+
assert "my hand-written context" in captured["context"] and "ERR: boom" in captured["context"]
|
|
@@ -0,0 +1,140 @@
|
|
|
1
|
+
"""Tests for the Adaptive Swarm Runtime search core (drydock/asr.py) — pure, no model/GPU."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
import pytest
|
|
5
|
+
|
|
6
|
+
from drydock import asr
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
# ── Budget (§15) ────────────────────────────────────────────────────────────
|
|
10
|
+
def test_budget_parse_time_and_tokens():
|
|
11
|
+
assert asr.Budget.parse("2h").wall_secs == 7200
|
|
12
|
+
assert asr.Budget.parse("30m").wall_secs == 1800
|
|
13
|
+
assert asr.Budget.parse("90s").wall_secs == 90
|
|
14
|
+
assert asr.Budget.parse("100M-tokens").tokens == 100_000_000
|
|
15
|
+
assert asr.Budget.parse("50k-tokens").tokens == 50_000
|
|
16
|
+
assert asr.Budget.parse("2h").is_finite
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def test_budget_rejects_garbage_never_silently_unbounded():
|
|
20
|
+
with pytest.raises(ValueError):
|
|
21
|
+
asr.Budget.parse("whenever")
|
|
22
|
+
assert asr.Budget().is_finite is False # empty budget is explicitly not finite
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
# ── Convergence (§23) ─────────────────────────────────────────────────────────
|
|
26
|
+
def _conv(**kw):
|
|
27
|
+
return asr.Convergence(budget=asr.Budget(generations=100), **kw)
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def test_convergence_stops_on_success():
|
|
31
|
+
c = _conv(success_score=100.0)
|
|
32
|
+
c.record(best_score=100.0, cumulative_tokens=5000, elapsed_s=60)
|
|
33
|
+
go, why = c.should_continue()
|
|
34
|
+
assert not go and "success" in why
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def test_convergence_stops_on_budget():
|
|
38
|
+
c = asr.Convergence(budget=asr.Budget(tokens=10_000))
|
|
39
|
+
c.record(best_score=40.0, cumulative_tokens=12_000, elapsed_s=30)
|
|
40
|
+
go, why = c.should_continue()
|
|
41
|
+
assert not go and "token budget" in why
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def test_convergence_stops_on_no_improvement():
|
|
45
|
+
c = _conv(no_improve_gens=2, success_score=100.0)
|
|
46
|
+
c.record(50.0, 1000, 10)
|
|
47
|
+
c.record(50.0, 2000, 20) # no gain
|
|
48
|
+
c.record(50.0, 3000, 30) # still no gain -> 2 stagnant gens
|
|
49
|
+
go, why = c.should_continue()
|
|
50
|
+
assert not go and "no improvement" in why
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def test_convergence_stops_on_low_marginal_value():
|
|
54
|
+
# big jump, then a tiny gain for a lot of tokens -> marginal value below floor
|
|
55
|
+
c = _conv(min_marginal=1.0, no_improve_gens=0, success_score=100.0)
|
|
56
|
+
c.record(40.0, 1000, 10)
|
|
57
|
+
c.record(80.0, 3000, 20) # +40 over 2k tok = 20/1k -> continues
|
|
58
|
+
assert c.should_continue()[0] is True
|
|
59
|
+
c.record(80.5, 13000, 40) # +0.5 over 10k tok = 0.05/1k -> below 1.0 floor
|
|
60
|
+
go, why = c.should_continue()
|
|
61
|
+
assert not go and "marginal value" in why
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def test_convergence_continues_while_improving():
|
|
65
|
+
c = _conv(no_improve_gens=3, min_marginal=0.5, success_score=100.0)
|
|
66
|
+
c.record(40.0, 1000, 10)
|
|
67
|
+
c.record(60.0, 2000, 20) # +20/1k, improving, under budget
|
|
68
|
+
assert c.should_continue() == (True, "continue")
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def test_marginal_value_needs_two_records():
|
|
72
|
+
c = _conv()
|
|
73
|
+
assert c.marginal_value() is None
|
|
74
|
+
c.record(10.0, 500, 5)
|
|
75
|
+
assert c.marginal_value() is None
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
# ── Adaptive allocation (§17) ──────────────────────────────────────────────────
|
|
79
|
+
def test_allocate_favors_leaders_and_prunes_weak():
|
|
80
|
+
branches = [("A", 0.91), ("B", 0.86), ("C", 0.43), ("D", 0.21)]
|
|
81
|
+
alloc = asr.allocate(branches, 8000, prune_below=0.3)
|
|
82
|
+
assert "D" in alloc.terminated and alloc.grants.get("D", 0) == 0 # weakest pruned
|
|
83
|
+
assert set(alloc.grants) == {"A", "B", "C"}
|
|
84
|
+
assert sum(alloc.grants.values()) == 8000 # whole pool spent
|
|
85
|
+
assert alloc.grants["A"] > alloc.grants["B"] > alloc.grants["C"] # EV ordering
|
|
86
|
+
assert alloc.grants["C"] < alloc.grants["A"] / 2 # bias toward the leader
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def test_allocate_keep_top_terminates_the_rest():
|
|
90
|
+
branches = [("A", 0.9), ("B", 0.8), ("C", 0.7), ("D", 0.6)]
|
|
91
|
+
alloc = asr.allocate(branches, 6000, keep_top=2)
|
|
92
|
+
assert set(alloc.grants) == {"A", "B"}
|
|
93
|
+
assert sorted(alloc.terminated) == ["C", "D"]
|
|
94
|
+
assert sum(alloc.grants.values()) == 6000
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
def test_allocate_empty_pool_terminates_all():
|
|
98
|
+
alloc = asr.allocate([("A", 0.9), ("B", 0.5)], 0)
|
|
99
|
+
assert alloc.grants == {} and sorted(alloc.terminated) == ["A", "B"]
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
def test_allocate_min_grant_floor_for_kept_branches():
|
|
103
|
+
branches = [("A", 0.99), ("B", 0.01)]
|
|
104
|
+
alloc = asr.allocate(branches, 10_000, min_grant=200)
|
|
105
|
+
assert alloc.grants["B"] >= 200 # trailing-but-kept branch still probed
|
|
106
|
+
assert alloc.grants["A"] > alloc.grants["B"]
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
# ── Planner (§15/§16, §5/§18) ──────────────────────────────────────────────────
|
|
110
|
+
def test_choose_intensity_scales_with_budget():
|
|
111
|
+
assert asr.choose_intensity(asr.Budget.parse("10m")) == "light"
|
|
112
|
+
assert asr.choose_intensity(asr.Budget.parse("45m")) == "standard"
|
|
113
|
+
assert asr.choose_intensity(asr.Budget.parse("2h")) == "deep"
|
|
114
|
+
assert asr.choose_intensity(asr.Budget.parse("8h")) == "extreme"
|
|
115
|
+
assert asr.choose_intensity(asr.Budget.parse("5M-tokens")) == "standard"
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
def test_plan_decouples_logical_population_from_concurrency():
|
|
119
|
+
# THE core ASR property (§5/§18): a deep plan on a 1-slot box keeps all 16 logical agents;
|
|
120
|
+
# they queue, they aren't discarded down to concurrency (unlike capacity.swarm_size).
|
|
121
|
+
weak = asr.plan(asr.Budget.parse("2h"), concurrency=1, intensity="deep")
|
|
122
|
+
assert weak.initial_population == 16 and weak.max_concurrency == 1
|
|
123
|
+
assert weak.queued_at_start == 15
|
|
124
|
+
strong = asr.plan(asr.Budget.parse("2h"), concurrency=8, intensity="deep")
|
|
125
|
+
assert strong.initial_population == 16 and strong.max_concurrency == 8
|
|
126
|
+
assert strong.queued_at_start == 8 # same search, more parallelism → less queue
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
def test_plan_auto_and_intensity_ladder():
|
|
130
|
+
auto = asr.plan(asr.Budget.parse("30m"), concurrency=4) # auto -> standard
|
|
131
|
+
assert auto.intensity == "standard"
|
|
132
|
+
pops = [asr.plan(asr.Budget.parse("2h"), 4, intensity=i).initial_population
|
|
133
|
+
for i in ("light", "standard", "deep", "extreme")]
|
|
134
|
+
assert pops == sorted(pops) and pops[0] < pops[-1] # deeper => bigger population
|
|
135
|
+
assert len(auto.roles) == auto.initial_population and len(set(auto.roles)) > 1 # diverse
|
|
136
|
+
|
|
137
|
+
|
|
138
|
+
def test_plan_rejects_unknown_intensity():
|
|
139
|
+
with pytest.raises(ValueError):
|
|
140
|
+
asr.plan(asr.Budget.parse("1h"), 4, intensity="ludicrous")
|