drydock-cli 3.1.37__tar.gz → 3.1.39__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {drydock_cli-3.1.37/drydock_cli.egg-info → drydock_cli-3.1.39}/PKG-INFO +1 -1
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/drydock/agent.py +25 -2
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/drydock/providers.py +68 -11
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/drydock/ratchet.py +29 -9
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/drydock/swarm.py +30 -10
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/drydock/tui/app.py +19 -6
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/drydock/tuning.py +4 -3
- {drydock_cli-3.1.37 → drydock_cli-3.1.39/drydock_cli.egg-info}/PKG-INFO +1 -1
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/drydock_cli.egg-info/SOURCES.txt +1 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/pyproject.toml +1 -1
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_providers_unreachable.py +51 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_swarm.py +42 -0
- drydock_cli-3.1.39/tests/test_truncated_turn.py +45 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/LICENSE +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/NOTICE +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/README.md +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/drydock/__init__.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/drydock/__main__.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/drydock/advisor.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/drydock/asr.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/drydock/bash_safety.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/drydock/bottleneck.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/drydock/budget.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/drydock/builtin_skills/__init__.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/drydock/builtin_skills/document-canvas.md +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/drydock/builtin_skills/ml-data.md +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/drydock/builtin_skills/ml-debug.md +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/drydock/builtin_skills/ml-finetune.md +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/drydock/builtin_skills/ml-metrics.md +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/drydock/builtin_skills/ml-rl.md +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/drydock/builtin_skills/ml-train.md +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/drydock/builtin_skills/nist-ai-rmf.md +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/drydock/builtin_skills/nist-csf.md +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/drydock/builtin_skills/rmf-categorize.md +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/drydock/builtin_skills/rmf-control.md +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/drydock/builtin_skills/rmf-poam.md +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/drydock/builtin_skills/rmf-review.md +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/drydock/builtin_skills/stig-assess.md +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/drydock/builtin_skills/stig-remediate.md +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/drydock/capacity.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/drydock/cci.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/drydock/cli.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/drydock/comms/__init__.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/drydock/comms/attention.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/drydock/comms/channels.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/drydock/comms/events.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/drydock/compaction.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/drydock/config.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/drydock/detect.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/drydock/doccanvas.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/drydock/eratchet.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/drydock/events.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/drydock/extract.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/drydock/gittools.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/drydock/graphrag.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/drydock/groundtruth.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/drydock/guards.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/drydock/jobs.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/drydock/loop_detect.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/drydock/mcp.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/drydock/mission.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/drydock/mission_cli.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/drydock/mission_run.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/drydock/pdfredbox.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/drydock/phases.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/drydock/poam.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/drydock/predictions.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/drydock/progress.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/drydock/recipes.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/drydock/recovery.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/drydock/resume.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/drydock/rmf.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/drydock/rmf_graph.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/drydock/skills.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/drydock/stig.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/drydock/subagents.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/drydock/suggest.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/drydock/task_state.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/drydock/tool_policy.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/drydock/tool_registry.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/drydock/tool_result.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/drydock/tool_select.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/drydock/tool_validate.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/drydock/tools/__init__.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/drydock/trajectory.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/drydock/tui/__init__.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/drydock/tui/approval.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/drydock/tui/messages.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/drydock/tui/widgets.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/drydock/verification.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/drydock/web.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/drydock_cli.egg-info/dependency_links.txt +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/drydock_cli.egg-info/entry_points.txt +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/drydock_cli.egg-info/requires.txt +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/drydock_cli.egg-info/top_level.txt +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/setup.cfg +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_advisor.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_approval.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_asr.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_back_command.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_bash_background.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_bash_binary_output.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_bash_crossplatform.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_bash_output_bounding.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_bash_process_group.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_bash_safety.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_bash_sanitize.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_bash_shell.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_bash_stdin.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_bash_stop_partial.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_bash_timeout_network.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_bash_timeout_param.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_bottleneck.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_budget.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_capacity.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_cci.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_cli_agents.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_comms.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_compact_command.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_compaction.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_config.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_config_migration.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_context_limit_config.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_context_limit_issue25.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_criteria_coverage.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_cycling_detection.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_declared_deps.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_degenerate_argument.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_detect.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_dispatch.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_doccanvas.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_e2e_connected.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_edit_replace_all.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_edit_thrash_and_compaction.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_effort_governor.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_empty_response.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_eratchet.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_events.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_extract.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_failure_loop.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_first_run_setup.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_gittools.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_glob_edit_edges.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_graphify_example.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_graphrag.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_graphrag_quoted_path.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_graphrag_sqlite.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_grep_and_read_robust.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_groundtruth_predictions.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_guards_and_tools.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_hallucinated_tools.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_jobs.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_knowledge_tool_available.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_leaked_tool_call.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_loop_detect.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_main_api_key.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_mcp.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_mission.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_mission_cli.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_mission_run.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_model_registry.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_oneshot_unreachable.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_overthink_interrupt.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_pdfredbox.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_phases.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_plan_autocontinue.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_poam.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_progress.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_ratchet.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_ratchet_offer.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_read_index.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_recipes.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_recovery.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_recovery_config.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_recovery_integration.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_repeated_outcome.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_repetition_interrupt.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_resume_events.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_resume_restore.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_resume_snapshot.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_rmf.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_rmf_graph.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_rmf_stig_graph.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_rolling_plan.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_runaway_repetition.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_screenshot.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_server_probe.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_skills.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_sqlite_events.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_stall_retry.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_stig.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_stop.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_streaming_newlines.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_subagent.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_subagents.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_suggest.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_system_prompt_help.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_task_state.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_text_only_model.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_timeline.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_todo.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_tool_arg_coercion.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_tool_arg_coercion_more.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_tool_arg_parsing.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_tool_canonical.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_tool_policy.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_tool_result.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_tool_select.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_tool_validate.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_tools_undo.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_trajectory.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_tui.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_tuning.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_verification_gate.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_viewimage.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_vision_input.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_web_tools.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_windows_shell.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_worker_subagent.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_write_content_coerce.py +0 -0
- {drydock_cli-3.1.37 → drydock_cli-3.1.39}/tests/test_xccdf.py +0 -0
|
@@ -217,6 +217,7 @@ def run(
|
|
|
217
217
|
leaked_call_retries = 0
|
|
218
218
|
plan_continue_nudges = 0 # consecutive "you stopped mid-plan" nudges
|
|
219
219
|
empty_response_nudges = 0 # consecutive "you returned nothing" nudges
|
|
220
|
+
truncated_nudges = 0 # "you hit the output limit without acting" nudges
|
|
220
221
|
# Safety valve for a degenerate loop: the SAME tool call run over and over
|
|
221
222
|
# with the SAME result — success OR failure (seen failing a Write 160×, and
|
|
222
223
|
# re-running an identical passing `pytest` 92× for 25 min). The advisory
|
|
@@ -436,10 +437,15 @@ def run(
|
|
|
436
437
|
if assistant_turn is None:
|
|
437
438
|
break
|
|
438
439
|
|
|
439
|
-
# Record assistant message
|
|
440
|
+
# Record assistant message. A turn cut off at max_tokens with no tool call is
|
|
441
|
+
# (almost always) runaway planning — keep only its tail so it doesn't eat the
|
|
442
|
+
# context window on every later request.
|
|
443
|
+
_content = assistant_turn.text
|
|
444
|
+
if assistant_turn.truncated and not assistant_turn.tool_calls and len(_content) > 2000:
|
|
445
|
+
_content = "[…long planning cut off at the output limit…]\n" + _content[-1500:]
|
|
440
446
|
state.messages.append({
|
|
441
447
|
"role": "assistant",
|
|
442
|
-
"content":
|
|
448
|
+
"content": _content,
|
|
443
449
|
"tool_calls": assistant_turn.tool_calls,
|
|
444
450
|
})
|
|
445
451
|
|
|
@@ -456,6 +462,22 @@ def run(
|
|
|
456
462
|
# case nudge it to use the real function interface and retry, instead
|
|
457
463
|
# of ending the turn with nothing done. Capped so it can never spin.
|
|
458
464
|
if not assistant_turn.tool_calls:
|
|
465
|
+
# Hit the output-token limit before acting (long planning / a whole-file write
|
|
466
|
+
# that didn't fit). That is not "done" — nudge it to act in smaller steps.
|
|
467
|
+
if assistant_turn.truncated and truncated_nudges < 3:
|
|
468
|
+
truncated_nudges += 1
|
|
469
|
+
_emit(state, "truncated_turn", nudge=truncated_nudges)
|
|
470
|
+
yield TextChunk("\n[output limit reached before any action — nudging to act in smaller steps]\n")
|
|
471
|
+
state.messages.append({
|
|
472
|
+
"role": "user",
|
|
473
|
+
"content": (
|
|
474
|
+
"[SYSTEM] Your reply hit the output-length limit before you took any "
|
|
475
|
+
"action, so nothing was done. Do not re-plan. Act NOW with one tool "
|
|
476
|
+
"call, and keep each call small: implement or edit ONE function or "
|
|
477
|
+
"section at a time (Edit), then continue with the next."
|
|
478
|
+
),
|
|
479
|
+
})
|
|
480
|
+
continue
|
|
459
481
|
if assistant_turn.had_leaked_call and leaked_call_retries < 2:
|
|
460
482
|
leaked_call_retries += 1
|
|
461
483
|
state.messages.append({
|
|
@@ -930,6 +952,7 @@ def run(
|
|
|
930
952
|
# bounds back-to-back stalls).
|
|
931
953
|
plan_continue_nudges = 0
|
|
932
954
|
empty_response_nudges = 0
|
|
955
|
+
truncated_nudges = 0
|
|
933
956
|
|
|
934
957
|
# Nudge: if past 15 tool calls without any edits, inject gentle guidance
|
|
935
958
|
if tool_call_count == 15 and not session_has_edited and config.get("force_first_tool"):
|
|
@@ -17,7 +17,10 @@ from typing import Generator
|
|
|
17
17
|
# HTTP client does NOT interrupt an in-flight blocking read, so instead we wait
|
|
18
18
|
# on a future and, when the user cancels, return immediately and let the
|
|
19
19
|
# orphaned request finish (and be discarded) in the background.
|
|
20
|
-
|
|
20
|
+
# Sized for in-process swarms: every in-flight request holds a thread (a streamed one holds
|
|
21
|
+
# a second for its reader), and abandoned requests keep theirs until they finish — 8 threads
|
|
22
|
+
# made an 8-agent swarm in one TUI queue behind itself.
|
|
23
|
+
_LLM_POOL = concurrent.futures.ThreadPoolExecutor(max_workers=64, thread_name_prefix="llm")
|
|
21
24
|
|
|
22
25
|
|
|
23
26
|
class _StopRequested(Exception):
|
|
@@ -139,19 +142,57 @@ def _friendly_timeout(base_url: str, timeout_s: float) -> str:
|
|
|
139
142
|
)
|
|
140
143
|
|
|
141
144
|
|
|
145
|
+
def _friendly_connect_timeout(base_url: str) -> str:
|
|
146
|
+
return (
|
|
147
|
+
f"Could not open a connection to the model server at {base_url} in time "
|
|
148
|
+
f"(connect timeout, retried). The server may be overloaded or the network slow.\n"
|
|
149
|
+
f" Your last message was not lost — just send it again."
|
|
150
|
+
)
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
def _is_connect_timeout(e: BaseException) -> bool:
|
|
154
|
+
"""An SDK APITimeoutError covers BOTH connect and read timeouts; tell them apart."""
|
|
155
|
+
import httpx
|
|
156
|
+
|
|
157
|
+
cur: BaseException | None = e
|
|
158
|
+
for _ in range(5):
|
|
159
|
+
if cur is None:
|
|
160
|
+
break
|
|
161
|
+
if isinstance(cur, (httpx.ConnectTimeout, httpx.PoolTimeout)):
|
|
162
|
+
return True
|
|
163
|
+
cur = cur.__cause__ or cur.__context__
|
|
164
|
+
return False
|
|
165
|
+
|
|
166
|
+
|
|
167
|
+
_CONNECT_RETRIES = 2 # transient connect timeouts (busy server / slow port proxy)
|
|
168
|
+
_CONNECT_BACKOFF_S = (2.0, 5.0)
|
|
169
|
+
|
|
170
|
+
|
|
171
|
+
def _timeout_error(e: BaseException, base_url: str, timeout_s: float) -> "LLMUnreachable":
|
|
172
|
+
if _is_connect_timeout(e):
|
|
173
|
+
return LLMUnreachable(_friendly_connect_timeout(base_url))
|
|
174
|
+
return LLMUnreachable(_friendly_timeout(base_url, timeout_s))
|
|
175
|
+
|
|
176
|
+
|
|
142
177
|
def _safe_create(client, kwargs: dict, base_url: str, provider: str, timeout_s: float = 600.0):
|
|
143
178
|
"""Call chat.completions.create, mapping a connection failure to a clean
|
|
144
179
|
LLMUnreachable instead of a raw traceback / 12-minute hang."""
|
|
145
180
|
import openai
|
|
146
181
|
|
|
147
|
-
|
|
148
|
-
|
|
149
|
-
|
|
150
|
-
|
|
151
|
-
|
|
152
|
-
|
|
153
|
-
|
|
154
|
-
|
|
182
|
+
import time as _time
|
|
183
|
+
|
|
184
|
+
for attempt in range(_CONNECT_RETRIES + 1):
|
|
185
|
+
try:
|
|
186
|
+
return client.chat.completions.create(**kwargs)
|
|
187
|
+
# APITimeoutError subclasses APIConnectionError — catch it FIRST so a slow
|
|
188
|
+
# (but alive) server gets the accurate "timed out" message, not "down".
|
|
189
|
+
except openai.APITimeoutError as e:
|
|
190
|
+
if _is_connect_timeout(e) and attempt < _CONNECT_RETRIES:
|
|
191
|
+
_time.sleep(_CONNECT_BACKOFF_S[attempt])
|
|
192
|
+
continue
|
|
193
|
+
raise _timeout_error(e, base_url, timeout_s) from e
|
|
194
|
+
except openai.APIConnectionError as e:
|
|
195
|
+
raise LLMUnreachable(_friendly_unreachable(base_url, provider)) from e
|
|
155
196
|
|
|
156
197
|
|
|
157
198
|
def _create_abortable(client, kwargs: dict, base_url: str, provider: str, cancel,
|
|
@@ -167,6 +208,7 @@ def _create_abortable(client, kwargs: dict, base_url: str, provider: str, cancel
|
|
|
167
208
|
|
|
168
209
|
fut = _LLM_POOL.submit(client.chat.completions.create, **kwargs)
|
|
169
210
|
start = _time.monotonic()
|
|
211
|
+
connect_retries = 0
|
|
170
212
|
while True:
|
|
171
213
|
try:
|
|
172
214
|
return fut.result(timeout=0.2)
|
|
@@ -177,7 +219,17 @@ def _create_abortable(client, kwargs: dict, base_url: str, provider: str, cancel
|
|
|
177
219
|
raise StallRetry(stall_secs) # orphan the wedged request; caller retries
|
|
178
220
|
# APITimeoutError subclasses APIConnectionError — catch it FIRST.
|
|
179
221
|
except openai.APITimeoutError as e:
|
|
180
|
-
|
|
222
|
+
if _is_connect_timeout(e) and connect_retries < _CONNECT_RETRIES:
|
|
223
|
+
# transient: the server/proxy was slow to ACCEPT — re-issue (cancellable wait)
|
|
224
|
+
wait_until = _time.monotonic() + _CONNECT_BACKOFF_S[connect_retries]
|
|
225
|
+
connect_retries += 1
|
|
226
|
+
while _time.monotonic() < wait_until:
|
|
227
|
+
if cancel is not None and cancel.is_set():
|
|
228
|
+
raise _StopRequested
|
|
229
|
+
_time.sleep(0.1)
|
|
230
|
+
fut = _LLM_POOL.submit(client.chat.completions.create, **kwargs)
|
|
231
|
+
continue
|
|
232
|
+
raise _timeout_error(e, base_url, timeout_s) from e
|
|
181
233
|
except openai.APIConnectionError as e:
|
|
182
234
|
raise LLMUnreachable(_friendly_unreachable(base_url, provider)) from e
|
|
183
235
|
|
|
@@ -246,6 +298,7 @@ class AssistantTurn:
|
|
|
246
298
|
input_tokens: int
|
|
247
299
|
output_tokens: int
|
|
248
300
|
had_leaked_call: bool = False # model emitted a tool call as text, not a call
|
|
301
|
+
truncated: bool = False # generation stopped at max_tokens (finish_reason == "length")
|
|
249
302
|
|
|
250
303
|
|
|
251
304
|
# ── Tool-call argument parsing ────────────────────────────────────────────
|
|
@@ -650,6 +703,7 @@ def stream(
|
|
|
650
703
|
return
|
|
651
704
|
|
|
652
705
|
text = ""
|
|
706
|
+
truncated = False
|
|
653
707
|
tool_buf: dict = {} # index → {id, name, args}
|
|
654
708
|
in_tok = out_tok = 0
|
|
655
709
|
# Runaway-repetition guard: a weak model can collapse into streaming one
|
|
@@ -672,6 +726,8 @@ def stream(
|
|
|
672
726
|
|
|
673
727
|
choice = chunk.choices[0]
|
|
674
728
|
delta = choice.delta
|
|
729
|
+
if getattr(choice, "finish_reason", None) == "length":
|
|
730
|
+
truncated = True
|
|
675
731
|
|
|
676
732
|
if delta.content:
|
|
677
733
|
# Strip any leaked thinking-token markers (best-effort per chunk).
|
|
@@ -741,7 +797,7 @@ def stream(
|
|
|
741
797
|
# A template-opened <think> leaves "reasoning…</think>answer" in the content —
|
|
742
798
|
# keep only the answer in the recorded turn (it was already shown while streaming).
|
|
743
799
|
_, text = split_think_tags(text)
|
|
744
|
-
yield AssistantTurn(text, tool_calls, in_tok, out_tok)
|
|
800
|
+
yield AssistantTurn(text, tool_calls, in_tok, out_tok, truncated=truncated)
|
|
745
801
|
|
|
746
802
|
|
|
747
803
|
def _complete_nonstreaming(
|
|
@@ -795,4 +851,5 @@ def _complete_nonstreaming(
|
|
|
795
851
|
yield AssistantTurn(
|
|
796
852
|
text, tool_calls, in_tok, out_tok,
|
|
797
853
|
had_leaked_call=had_leak and not tool_calls,
|
|
854
|
+
truncated=getattr(choice, "finish_reason", None) == "length",
|
|
798
855
|
)
|
|
@@ -87,6 +87,30 @@ class VerifyResult:
|
|
|
87
87
|
return self.total > 0 and self.passed >= self.total
|
|
88
88
|
|
|
89
89
|
|
|
90
|
+
def run_shell_bounded(command: str, cwd: str | None, timeout: float) -> tuple[str, int, bool]:
|
|
91
|
+
"""Run a shell command with a timeout that actually holds: the command runs in its
|
|
92
|
+
own process group and the whole group is killed on timeout. (subprocess.run with
|
|
93
|
+
shell=True only kills /bin/sh; a hung child — e.g. pytest stuck in an infinite loop —
|
|
94
|
+
keeps the output pipes open and the call blocks far past the timeout.)
|
|
95
|
+
Returns (combined output, returncode, timed_out)."""
|
|
96
|
+
import signal
|
|
97
|
+
|
|
98
|
+
proc = subprocess.Popen(command, shell=True, cwd=cwd, stdout=subprocess.PIPE,
|
|
99
|
+
stderr=subprocess.PIPE, text=True, start_new_session=True)
|
|
100
|
+
try:
|
|
101
|
+
out, err = proc.communicate(timeout=timeout)
|
|
102
|
+
timed_out = False
|
|
103
|
+
except subprocess.TimeoutExpired:
|
|
104
|
+
try:
|
|
105
|
+
os.killpg(proc.pid, signal.SIGKILL)
|
|
106
|
+
except (ProcessLookupError, PermissionError):
|
|
107
|
+
proc.kill()
|
|
108
|
+
out, err = proc.communicate()
|
|
109
|
+
timed_out = True
|
|
110
|
+
text = (out or "") + ("\n" + err if err else "")
|
|
111
|
+
return text, (124 if timed_out else proc.returncode), timed_out
|
|
112
|
+
|
|
113
|
+
|
|
90
114
|
@dataclass
|
|
91
115
|
class Verifier:
|
|
92
116
|
"""Runs a shell command and scores it. The command is the ground truth the
|
|
@@ -99,15 +123,11 @@ class Verifier:
|
|
|
99
123
|
|
|
100
124
|
def run(self) -> VerifyResult:
|
|
101
125
|
try:
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
106
|
-
out
|
|
107
|
-
rc = proc.returncode
|
|
108
|
-
except subprocess.TimeoutExpired as e:
|
|
109
|
-
out = (e.output or "") + "\n[verifier timed out]"
|
|
110
|
-
rc = 124
|
|
126
|
+
out, rc, timed_out = run_shell_bounded(self.command, self.cwd, self.timeout)
|
|
127
|
+
except OSError as e:
|
|
128
|
+
out, rc, timed_out = f"[verifier failed to start: {e}]", 127, False
|
|
129
|
+
if timed_out:
|
|
130
|
+
out += "\n[verifier timed out]"
|
|
111
131
|
p, t = score_output(out, self.mode, rc)
|
|
112
132
|
return VerifyResult(p, t, rc, out)
|
|
113
133
|
|
|
@@ -589,21 +589,21 @@ VerifyFn = Callable[["Candidate"], "tuple[int, int]"]
|
|
|
589
589
|
|
|
590
590
|
|
|
591
591
|
def make_shell_verifier(verify_cmd: str, fitness: str = "auto",
|
|
592
|
-
timeout: int =
|
|
592
|
+
timeout: int = 600) -> VerifyFn:
|
|
593
593
|
"""A verifier that runs `verify_cmd` in the candidate's worktree and scores the output
|
|
594
594
|
with drydock.ratchet.score_output — the same scorer the ratchet/eratchet use, so a
|
|
595
595
|
candidate is judged by a real test run, never by the builder's self-report (§18/§20)."""
|
|
596
|
-
from drydock.ratchet import score_output
|
|
596
|
+
from drydock.ratchet import run_shell_bounded, score_output
|
|
597
597
|
|
|
598
598
|
def verify(cand: Candidate) -> tuple[int, int]:
|
|
599
599
|
if not cand.worktree or not Path(cand.worktree).exists():
|
|
600
600
|
return 0, 0
|
|
601
601
|
try:
|
|
602
|
-
|
|
603
|
-
capture_output=True, text=True, timeout=timeout)
|
|
602
|
+
out, rc, _ = run_shell_bounded(verify_cmd, cand.worktree, timeout)
|
|
604
603
|
except (OSError, subprocess.SubprocessError):
|
|
605
604
|
return 0, 1
|
|
606
|
-
|
|
605
|
+
# a hung candidate (killed at the timeout) scores whatever finished before the kill
|
|
606
|
+
return score_output(out, fitness, rc)
|
|
607
607
|
|
|
608
608
|
return verify
|
|
609
609
|
|
|
@@ -764,12 +764,14 @@ def run_swarm(cwd: str | Path, objective: str, *, agents: "int | str" = 4,
|
|
|
764
764
|
runner: AgentRunner = default_agent_runner, verify: VerifyFn | None = None,
|
|
765
765
|
on_event: "Callable[[str, dict], None] | None" = None,
|
|
766
766
|
share: bool = False, waves: int = 2,
|
|
767
|
-
swarm_id: str | None = None) -> SwarmResult:
|
|
767
|
+
swarm_id: str | None = None, cancel=None) -> SwarmResult:
|
|
768
768
|
"""Coordinate a parallel-strategy swarm end to end (§24 Parallel, the MVP default):
|
|
769
769
|
decompose into N diversity-injected Builders (§15), fan them out concurrently over one
|
|
770
770
|
shared inference server (§22), verify each candidate independently (§18), and converge on
|
|
771
771
|
the best by evidence (§20). Never raises on an individual worker — failures are contained
|
|
772
|
-
(§32). `runner`/`verify` are injectable for testing without a live model.
|
|
772
|
+
(§32). `runner`/`verify` are injectable for testing without a live model. `cancel` (a
|
|
773
|
+
threading.Event) stops the swarm: running agents end their turn loop, queued ones never
|
|
774
|
+
start, and verification is skipped — whatever was produced is still judged."""
|
|
773
775
|
from concurrent.futures import ThreadPoolExecutor
|
|
774
776
|
|
|
775
777
|
repo = repo_root(cwd)
|
|
@@ -777,6 +779,12 @@ def run_swarm(cwd: str | Path, objective: str, *, agents: "int | str" = 4,
|
|
|
777
779
|
raise ValueError("swarm needs a git repository (run `git init` first) for worktree "
|
|
778
780
|
"isolation")
|
|
779
781
|
bc = base_config or {}
|
|
782
|
+
if cancel is not None:
|
|
783
|
+
# workers inherit this as their agent-loop stop signal (agent.run checks _cancel)
|
|
784
|
+
base_config = bc = {**bc, "_cancel": cancel}
|
|
785
|
+
|
|
786
|
+
def _cancelled() -> bool:
|
|
787
|
+
return cancel is not None and cancel.is_set()
|
|
780
788
|
# Auto-size to the server's real concurrency (§21/§22): hammer a big box, tone down a
|
|
781
789
|
# small one. N = min(hardware concurrency, task-demand, budget) — never maximized.
|
|
782
790
|
detected = 0
|
|
@@ -830,10 +838,19 @@ def run_swarm(cwd: str | Path, objective: str, *, agents: "int | str" = 4,
|
|
|
830
838
|
|
|
831
839
|
# 2. fan out workers over the shared server (§22). Each is isolated + safe. `extra_sys`
|
|
832
840
|
# carries peers' notes (blackboard consumption, §10) for waves after the first.
|
|
841
|
+
# Workers are full coding agents: give them drydock's own coding prompt (tool use,
|
|
842
|
+
# verify-before-done, act-don't-plan), then the swarm role + diversity angle on top.
|
|
843
|
+
# Without it a worker got only the 3-line role text and tended to plan until it ran
|
|
844
|
+
# out of output tokens instead of editing.
|
|
845
|
+
from drydock.tuning import system_prompt_for_model
|
|
846
|
+
base_sys = system_prompt_for_model(str(bc.get("model") or ""), worker=True) + "\n\n"
|
|
847
|
+
|
|
833
848
|
def _spawn(i: int, extra_sys: str) -> WorkerOutcome:
|
|
849
|
+
if _cancelled():
|
|
850
|
+
return WorkerOutcome(agent=f"agent-{i + 1}", ok=False, error="cancelled")
|
|
834
851
|
angle, sysprompt = angles[i]
|
|
835
852
|
out = run_worker(bb, repo, base_ref, f"agent-{i + 1}", objective,
|
|
836
|
-
role="builder", system_prompt=sysprompt + extra_sys,
|
|
853
|
+
role="builder", system_prompt=base_sys + sysprompt + extra_sys,
|
|
837
854
|
base_config=base_config, max_turns=max_turns,
|
|
838
855
|
max_tool_calls=max_tool_calls, runner=runner)
|
|
839
856
|
_ev("worker_done", agent=out.agent, ok=out.ok, files=out.files_changed,
|
|
@@ -842,7 +859,7 @@ def run_swarm(cwd: str | Path, objective: str, *, agents: "int | str" = 4,
|
|
|
842
859
|
|
|
843
860
|
def _verify_unscored() -> None:
|
|
844
861
|
"""Independently score any candidate not yet verified (§18)."""
|
|
845
|
-
if verify is None:
|
|
862
|
+
if verify is None or _cancelled():
|
|
846
863
|
return
|
|
847
864
|
for c in bb.candidates():
|
|
848
865
|
if c.commit and c.tests_total == 0:
|
|
@@ -858,7 +875,7 @@ def run_swarm(cwd: str | Path, objective: str, *, agents: "int | str" = 4,
|
|
|
858
875
|
# notes the next wave reads are already scored.
|
|
859
876
|
per = -(-n // max(2, waves)) # ceil(n / waves)
|
|
860
877
|
i0, w = 0, 0
|
|
861
|
-
while i0 < n:
|
|
878
|
+
while i0 < n and not _cancelled():
|
|
862
879
|
idxs = list(range(i0, min(i0 + per, n)))
|
|
863
880
|
i0 += per
|
|
864
881
|
notes = _peer_notes(bb.candidates()) if w > 0 else ""
|
|
@@ -878,6 +895,9 @@ def run_swarm(cwd: str | Path, objective: str, *, agents: "int | str" = 4,
|
|
|
878
895
|
_verify_unscored()
|
|
879
896
|
|
|
880
897
|
# 4. judge on evidence (§20) and check convergence (§19).
|
|
898
|
+
if _cancelled():
|
|
899
|
+
bb.emit("SWARM_CANCELLED")
|
|
900
|
+
_ev("cancelled")
|
|
881
901
|
final = bb.candidates()
|
|
882
902
|
winner = judge(final)
|
|
883
903
|
converged = bool(winner and winner.tests_total > 0 and winner.tests_passed == winner.tests_total)
|
|
@@ -181,9 +181,15 @@ class DrydockApp(App):
|
|
|
181
181
|
if self._erx is not None:
|
|
182
182
|
self._erx["cancel"].set()
|
|
183
183
|
self._erx = None
|
|
184
|
+
swarm_cancel = (self._swarm or {}).get("cancel")
|
|
185
|
+
swarm_active = swarm_cancel is not None
|
|
186
|
+
if swarm_cancel is not None:
|
|
187
|
+
swarm_cancel.set()
|
|
184
188
|
if not self._busy:
|
|
185
189
|
if erx_active:
|
|
186
190
|
self._info("⏹ eratchet stopping…")
|
|
191
|
+
if swarm_active:
|
|
192
|
+
self._info("⏹ swarm stopping — agents finish their current step…")
|
|
187
193
|
return
|
|
188
194
|
self._cancel.set()
|
|
189
195
|
self._abort_inflight()
|
|
@@ -414,9 +420,11 @@ class DrydockApp(App):
|
|
|
414
420
|
used = self._ctx_tokens
|
|
415
421
|
pct = min(100, round(used / limit * 100)) if limit else 0
|
|
416
422
|
ctx = f"ctx {used:,}/{limit // 1000}k ({pct}%)"
|
|
417
|
-
|
|
423
|
+
# A swarm / eratchet runs off-thread without setting _busy — still show it as running.
|
|
424
|
+
bg = "swarm" if self._swarm else ("eratchet" if self._erx else "")
|
|
425
|
+
flag = "⚓ working" if self._busy else (f"⚓ {bg} running" if bg else "⚓ ready")
|
|
418
426
|
# Busy → show the interrupt keys; idle → the quit hint.
|
|
419
|
-
keys = "Esc/Ctrl+C stop" if self._busy else "Ctrl+C×2 quit"
|
|
427
|
+
keys = "Esc/Ctrl+C stop" if (self._busy or bg) else "Ctrl+C×2 quit"
|
|
420
428
|
# Show the task phase once it's past the default (understand).
|
|
421
429
|
ph = getattr(self.state, "task", None)
|
|
422
430
|
phase = f" · {ph.phase}" if (ph and ph.is_set() and ph.phase != "understand") else ""
|
|
@@ -1230,17 +1238,21 @@ class DrydockApp(App):
|
|
|
1230
1238
|
harness escalating on its own (no user command); `verify_cmd` pins the checker
|
|
1231
1239
|
(e.g. the one the single agent kept failing). agents="auto" sizes to the server's
|
|
1232
1240
|
detected concurrency (hammer a big box, tone down a small one)."""
|
|
1233
|
-
|
|
1241
|
+
import threading
|
|
1242
|
+
|
|
1243
|
+
cancel = threading.Event()
|
|
1244
|
+
self._swarm = {"active": True, "cancel": cancel}
|
|
1245
|
+
self._refresh_status()
|
|
1234
1246
|
head = ("⚙ the harness is escalating to a parallel swarm" if auto
|
|
1235
1247
|
else f"⇶ /swarm launching: {objective!r}")
|
|
1236
1248
|
count = f"{agents}" if isinstance(agents, int) else "auto-sized"
|
|
1237
1249
|
self._info(f"{head} — {count} agents exploring in isolated worktrees "
|
|
1238
1250
|
"(in-process, your working tree is left alone). Esc to stop.")
|
|
1239
1251
|
self.run_worker(
|
|
1240
|
-
lambda: self._swarm_worker(cwd, objective, agents, verify_cmd), thread=True)
|
|
1252
|
+
lambda: self._swarm_worker(cwd, objective, agents, verify_cmd, cancel), thread=True)
|
|
1241
1253
|
|
|
1242
1254
|
def _swarm_worker(self, cwd: str, objective: str, agents: "int | str",
|
|
1243
|
-
verify_cmd: str | None = None) -> None:
|
|
1255
|
+
verify_cmd: str | None = None, cancel=None) -> None:
|
|
1244
1256
|
"""Off-thread: drive run_swarm, streaming milestones into this session's transcript."""
|
|
1245
1257
|
from drydock import swarm as swarmmod
|
|
1246
1258
|
|
|
@@ -1267,7 +1279,8 @@ class DrydockApp(App):
|
|
|
1267
1279
|
# share=True: later-wave agents read peers' verified attempts off the blackboard
|
|
1268
1280
|
# and compare notes, instead of exploring fully blind (§10).
|
|
1269
1281
|
res = swarmmod.run_swarm(cwd, objective, agents=agents, base_config=self.config,
|
|
1270
|
-
verify_cmd=verify_cmd, on_event=on_event, share=True
|
|
1282
|
+
verify_cmd=verify_cmd, on_event=on_event, share=True,
|
|
1283
|
+
cancel=cancel)
|
|
1271
1284
|
if res.converged and res.winner is not None:
|
|
1272
1285
|
tail = (f"\nApply it: git cherry-pick {res.winner.commit}")
|
|
1273
1286
|
elif res.winner is not None:
|
|
@@ -328,10 +328,11 @@ def _shell_env_note() -> str:
|
|
|
328
328
|
)
|
|
329
329
|
|
|
330
330
|
|
|
331
|
-
def system_prompt_for_model(model: str | None) -> str:
|
|
332
|
-
"""Return the system prompt best suited to the model.
|
|
331
|
+
def system_prompt_for_model(model: str | None, *, worker: bool = False) -> str:
|
|
332
|
+
"""Return the system prompt best suited to the model. worker=True drops the
|
|
333
|
+
user-facing slash-command help (a background swarm worker never talks to the user)."""
|
|
333
334
|
base = _GEMMA_SYSTEM_PROMPT if is_gemma(model) else _DEFAULT_SYSTEM_PROMPT
|
|
334
|
-
return base + _DRYDOCK_COMMANDS_HELP + _shell_env_note()
|
|
335
|
+
return base + ("" if worker else _DRYDOCK_COMMANDS_HELP) + _shell_env_note()
|
|
335
336
|
|
|
336
337
|
|
|
337
338
|
def thinking_level_for_turn(turn_count: int, is_user_turn: bool) -> str:
|
|
@@ -7,7 +7,7 @@ build-backend = "setuptools.build_meta"
|
|
|
7
7
|
# PyPI distribution name is drydock-cli (the established install name, continued
|
|
8
8
|
# from the retired v2 fork); the import package + CLI command stay `drydock`.
|
|
9
9
|
name = "drydock-cli"
|
|
10
|
-
version = "3.1.
|
|
10
|
+
version = "3.1.39"
|
|
11
11
|
description = "Drydock — a local, provider-agnostic terminal coding agent for local LLMs"
|
|
12
12
|
readme = "README.md"
|
|
13
13
|
requires-python = ">=3.11"
|
|
@@ -93,3 +93,54 @@ def test_friendly_timeout_renders_minutes_and_remediation():
|
|
|
93
93
|
assert "30 min" in msg
|
|
94
94
|
assert "request_timeout" in msg # how to raise the limit
|
|
95
95
|
assert "reasoning budget" in msg # how to shorten turns
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
# ── Connect timeouts: retried, and reported as connect (not read) timeouts ──
|
|
99
|
+
class _ConnectTimeoutClient:
|
|
100
|
+
def __init__(self, fail_times):
|
|
101
|
+
self.fail_times = fail_times
|
|
102
|
+
self.calls = 0
|
|
103
|
+
outer = self
|
|
104
|
+
|
|
105
|
+
class _Completions:
|
|
106
|
+
@staticmethod
|
|
107
|
+
def create(**kwargs):
|
|
108
|
+
outer.calls += 1
|
|
109
|
+
if outer.calls <= outer.fail_times:
|
|
110
|
+
req = httpx.Request("POST", "http://localhost:8000/v1")
|
|
111
|
+
try:
|
|
112
|
+
raise httpx.ConnectTimeout("connect timed out", request=req)
|
|
113
|
+
except httpx.ConnectTimeout as e:
|
|
114
|
+
raise openai.APITimeoutError(request=req) from e
|
|
115
|
+
return "OK"
|
|
116
|
+
|
|
117
|
+
class _Chat:
|
|
118
|
+
completions = _Completions
|
|
119
|
+
|
|
120
|
+
self.chat = _Chat
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
def _fast_backoff(monkeypatch):
|
|
124
|
+
from drydock import providers
|
|
125
|
+
monkeypatch.setattr(providers, "_CONNECT_BACKOFF_S", (0.01, 0.01))
|
|
126
|
+
|
|
127
|
+
|
|
128
|
+
def test_connect_timeout_is_retried(monkeypatch):
|
|
129
|
+
_fast_backoff(monkeypatch)
|
|
130
|
+
c = _ConnectTimeoutClient(fail_times=2)
|
|
131
|
+
assert _safe_create(c, {}, "http://localhost:8000/v1", "vllm", 1800.0) == "OK"
|
|
132
|
+
c2 = _ConnectTimeoutClient(fail_times=2)
|
|
133
|
+
assert _create_abortable(c2, {}, "http://localhost:8000/v1", "vllm", None, 1800.0) == "OK"
|
|
134
|
+
assert c.calls == 3 and c2.calls == 3
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
def test_persistent_connect_timeout_says_connect_not_read(monkeypatch):
|
|
138
|
+
_fast_backoff(monkeypatch)
|
|
139
|
+
for call in (lambda c: _safe_create(c, {}, "http://x:8000/v1", "vllm", 1800.0),
|
|
140
|
+
lambda c: _create_abortable(c, {}, "http://x:8000/v1", "vllm", None, 1800.0)):
|
|
141
|
+
c = _ConnectTimeoutClient(fail_times=99)
|
|
142
|
+
with pytest.raises(LLMUnreachable) as ei:
|
|
143
|
+
call(c)
|
|
144
|
+
assert "connect timeout" in str(ei.value)
|
|
145
|
+
assert "read timeout" not in str(ei.value)
|
|
146
|
+
assert c.calls == 3
|
|
@@ -468,3 +468,45 @@ def test_cli_solve_dispatches_to_run_swarm(tmp_path, capsys, monkeypatch):
|
|
|
468
468
|
assert seen == {"objective": "fix the bug", "agents": 3, "verify_cmd": "pytest -q"}
|
|
469
469
|
out = capsys.readouterr().out
|
|
470
470
|
assert "SWARM CONVERGED" in out and "cherry-pick abc123" in out
|
|
471
|
+
|
|
472
|
+
|
|
473
|
+
def test_run_swarm_cancel_stops_queued_waves_and_verification(tmp_path):
|
|
474
|
+
import threading
|
|
475
|
+
|
|
476
|
+
repo = _init_repo(tmp_path / "repo")
|
|
477
|
+
cancel = threading.Event()
|
|
478
|
+
seen_cancel, verified = [], []
|
|
479
|
+
|
|
480
|
+
def runner(objective, cwd, base_config, system_prompt, allow, mt, mtc):
|
|
481
|
+
seen_cancel.append(base_config.get("_cancel") is cancel)
|
|
482
|
+
(Path(cwd) / "answer.txt").write_text("x")
|
|
483
|
+
cancel.set() # user hits Esc while the first wave is running
|
|
484
|
+
return "wrote"
|
|
485
|
+
|
|
486
|
+
def verify(cand):
|
|
487
|
+
verified.append(cand.id)
|
|
488
|
+
return (1, 1)
|
|
489
|
+
|
|
490
|
+
events = []
|
|
491
|
+
res = swarm.run_swarm(repo, "obj", agents=4, base_config={}, runner=runner,
|
|
492
|
+
verify=verify, share=True, waves=2, max_workers=1,
|
|
493
|
+
cancel=cancel, on_event=lambda k, d: events.append(k))
|
|
494
|
+
assert seen_cancel and all(seen_cancel) # workers get the swarm's stop signal
|
|
495
|
+
assert len(seen_cancel) == 1 # queued agents never started
|
|
496
|
+
assert verified == [] # verification skipped on cancel
|
|
497
|
+
assert "cancelled" in events and "judge" in events
|
|
498
|
+
assert len([c for c in res.candidates if c.commit]) == 1
|
|
499
|
+
|
|
500
|
+
|
|
501
|
+
def test_shell_verifier_timeout_kills_hung_child(tmp_path):
|
|
502
|
+
import time
|
|
503
|
+
|
|
504
|
+
wt = tmp_path / "wt"
|
|
505
|
+
wt.mkdir()
|
|
506
|
+
cand = swarm.Candidate(id="c-1", agent="a", commit="x", worktree=str(wt))
|
|
507
|
+
# the shell spawns a child that outlives the shell's own kill and holds the pipes open
|
|
508
|
+
verify = swarm.make_shell_verifier("sleep 30 & sleep 30; echo done", timeout=1)
|
|
509
|
+
t0 = time.monotonic()
|
|
510
|
+
passed, _total = verify(cand)
|
|
511
|
+
assert time.monotonic() - t0 < 10
|
|
512
|
+
assert passed == 0
|
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
"""A turn that hits max_tokens without a tool call (runaway planning, or a whole-file
|
|
2
|
+
write that didn't fit) is not "done": the agent nudges it to act in smaller steps, trims
|
|
3
|
+
the runaway text from history, and stays bounded."""
|
|
4
|
+
from __future__ import annotations
|
|
5
|
+
|
|
6
|
+
from drydock import agent
|
|
7
|
+
from drydock.agent import AgentState
|
|
8
|
+
from drydock.providers import AssistantTurn
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
def _drive(monkeypatch, truncations):
|
|
12
|
+
calls = []
|
|
13
|
+
|
|
14
|
+
def fake_stream(model, system, messages, tool_schemas, config):
|
|
15
|
+
calls.append([m.get("content") for m in messages])
|
|
16
|
+
if len(calls) <= truncations:
|
|
17
|
+
yield AssistantTurn(text="plan " * 5000, tool_calls=[], input_tokens=10,
|
|
18
|
+
output_tokens=8192, truncated=True)
|
|
19
|
+
else:
|
|
20
|
+
yield AssistantTurn(text="done", tool_calls=[], input_tokens=10, output_tokens=5)
|
|
21
|
+
|
|
22
|
+
monkeypatch.setattr(agent, "stream", fake_stream)
|
|
23
|
+
state = AgentState()
|
|
24
|
+
cfg = {"model": "m", "max_tokens": 8192, "context_limit": 65536}
|
|
25
|
+
events = list(agent.run("do it", state, cfg, "SYS"))
|
|
26
|
+
return calls, state, events
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def test_truncated_turn_nudges_and_trims(monkeypatch):
|
|
30
|
+
calls, state, events = _drive(monkeypatch, truncations=1)
|
|
31
|
+
assert len(calls) == 2
|
|
32
|
+
assert any("output-length limit" in (c or "") for c in calls[1])
|
|
33
|
+
planning = [m for m in state.messages if m["role"] == "assistant"][0]["content"]
|
|
34
|
+
assert len(planning) < 2000 and "cut off" in planning
|
|
35
|
+
assert any("output limit reached" in getattr(e, "text", "") for e in events)
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def test_truncation_nudges_are_bounded(monkeypatch):
|
|
39
|
+
calls, _, _ = _drive(monkeypatch, truncations=99)
|
|
40
|
+
assert len(calls) == 4 # 1 + 3 nudges, then accepted as the end
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def test_normal_text_turn_is_not_nudged(monkeypatch):
|
|
44
|
+
calls, _, _ = _drive(monkeypatch, truncations=0)
|
|
45
|
+
assert len(calls) == 1
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|