drydock-cli 3.1.23__tar.gz → 3.1.28__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {drydock_cli-3.1.23/drydock_cli.egg-info → drydock_cli-3.1.28}/PKG-INFO +1 -1
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/drydock/agent.py +25 -1
- drydock_cli-3.1.28/drydock/bottleneck.py +218 -0
- drydock_cli-3.1.28/drydock/capacity.py +135 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/drydock/cli.py +10 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/drydock/config.py +20 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/drydock/eratchet.py +60 -8
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/drydock/graphrag.py +20 -0
- drydock_cli-3.1.28/drydock/groundtruth.py +167 -0
- drydock_cli-3.1.28/drydock/predictions.py +177 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/drydock/ratchet.py +37 -0
- drydock_cli-3.1.28/drydock/swarm.py +1024 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/drydock/tools/__init__.py +191 -31
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/drydock/tui/app.py +147 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28/drydock_cli.egg-info}/PKG-INFO +1 -1
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/drydock_cli.egg-info/SOURCES.txt +11 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/pyproject.toml +1 -1
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_bash_background.py +42 -0
- drydock_cli-3.1.28/tests/test_bottleneck.py +103 -0
- drydock_cli-3.1.28/tests/test_capacity.py +78 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_eratchet.py +49 -0
- drydock_cli-3.1.28/tests/test_groundtruth_predictions.py +124 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_jobs.py +23 -0
- drydock_cli-3.1.28/tests/test_knowledge_tool_available.py +56 -0
- drydock_cli-3.1.28/tests/test_ratchet_offer.py +64 -0
- drydock_cli-3.1.28/tests/test_swarm.py +426 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/LICENSE +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/NOTICE +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/README.md +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/drydock/__init__.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/drydock/__main__.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/drydock/advisor.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/drydock/bash_safety.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/drydock/budget.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/drydock/builtin_skills/__init__.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/drydock/builtin_skills/document-canvas.md +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/drydock/builtin_skills/fiar-assess.md +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/drydock/builtin_skills/fiar-cap.md +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/drydock/builtin_skills/fiar-evidence.md +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/drydock/builtin_skills/fiar-readiness.md +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/drydock/builtin_skills/ml-data.md +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/drydock/builtin_skills/ml-debug.md +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/drydock/builtin_skills/ml-finetune.md +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/drydock/builtin_skills/ml-metrics.md +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/drydock/builtin_skills/ml-rl.md +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/drydock/builtin_skills/ml-train.md +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/drydock/builtin_skills/nist-ai-rmf.md +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/drydock/builtin_skills/nist-csf.md +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/drydock/builtin_skills/rmf-categorize.md +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/drydock/builtin_skills/rmf-control.md +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/drydock/builtin_skills/rmf-poam.md +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/drydock/builtin_skills/rmf-review.md +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/drydock/builtin_skills/stig-assess.md +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/drydock/builtin_skills/stig-remediate.md +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/drydock/cci.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/drydock/comms/__init__.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/drydock/comms/attention.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/drydock/comms/channels.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/drydock/comms/events.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/drydock/compaction.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/drydock/detect.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/drydock/doccanvas.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/drydock/events.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/drydock/extract.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/drydock/fiar.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/drydock/gittools.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/drydock/guards.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/drydock/jobs.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/drydock/loop_detect.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/drydock/mcp.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/drydock/pdfredbox.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/drydock/phases.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/drydock/poam.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/drydock/progress.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/drydock/providers.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/drydock/recipes.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/drydock/recovery.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/drydock/resume.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/drydock/rmf.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/drydock/rmf_graph.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/drydock/skills.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/drydock/stig.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/drydock/subagents.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/drydock/suggest.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/drydock/task_state.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/drydock/tool_policy.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/drydock/tool_registry.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/drydock/tool_result.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/drydock/tool_select.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/drydock/tool_validate.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/drydock/trajectory.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/drydock/tui/__init__.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/drydock/tui/approval.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/drydock/tui/messages.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/drydock/tui/widgets.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/drydock/tuning.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/drydock/verification.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/drydock/web.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/drydock_cli.egg-info/dependency_links.txt +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/drydock_cli.egg-info/entry_points.txt +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/drydock_cli.egg-info/requires.txt +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/drydock_cli.egg-info/top_level.txt +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/setup.cfg +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_advisor.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_approval.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_back_command.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_bash_binary_output.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_bash_crossplatform.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_bash_output_bounding.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_bash_process_group.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_bash_safety.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_bash_sanitize.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_bash_shell.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_bash_stdin.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_bash_stop_partial.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_bash_timeout_network.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_bash_timeout_param.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_budget.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_cci.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_cli_agents.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_comms.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_compact_command.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_compaction.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_config.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_config_migration.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_context_limit_config.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_context_limit_issue25.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_criteria_coverage.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_cycling_detection.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_declared_deps.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_degenerate_argument.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_detect.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_dispatch.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_doccanvas.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_e2e_connected.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_edit_replace_all.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_edit_thrash_and_compaction.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_effort_governor.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_empty_response.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_events.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_extract.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_failure_loop.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_fiar.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_first_run_setup.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_gittools.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_glob_edit_edges.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_graphify_example.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_graphrag.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_graphrag_quoted_path.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_graphrag_sqlite.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_grep_and_read_robust.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_guards_and_tools.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_hallucinated_tools.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_leaked_tool_call.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_loop_detect.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_mcp.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_model_registry.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_oneshot_unreachable.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_overthink_interrupt.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_pdfredbox.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_phases.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_plan_autocontinue.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_poam.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_progress.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_providers_unreachable.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_ratchet.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_read_index.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_recipes.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_recovery.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_recovery_config.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_recovery_integration.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_repeated_outcome.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_repetition_interrupt.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_resume_events.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_resume_restore.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_resume_snapshot.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_rmf.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_rmf_graph.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_rmf_stig_graph.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_rolling_plan.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_runaway_repetition.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_screenshot.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_server_probe.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_skills.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_sqlite_events.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_stall_retry.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_stig.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_stop.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_streaming_newlines.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_subagent.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_subagents.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_suggest.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_system_prompt_help.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_task_state.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_timeline.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_todo.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_tool_arg_coercion.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_tool_arg_coercion_more.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_tool_arg_parsing.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_tool_canonical.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_tool_policy.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_tool_result.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_tool_select.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_tool_validate.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_tools_undo.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_trajectory.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_tui.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_tuning.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_verification_gate.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_viewimage.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_vision_input.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_web_tools.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_windows_shell.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_worker_subagent.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_write_content_coerce.py +0 -0
- {drydock_cli-3.1.23 → drydock_cli-3.1.28}/tests/test_xccdf.py +0 -0
|
@@ -107,6 +107,8 @@ class AgentState:
|
|
|
107
107
|
events: "EventLog | SQLiteEventLog | None" = None # optional durable execution trace
|
|
108
108
|
recovery_stage: int = 0 # live recovery escalation stage (0 = normal); for the TUI
|
|
109
109
|
progress_streak: int = 0 # consecutive no-progress (stall) actions; for the TUI
|
|
110
|
+
verify_fail_streak: int = 0 # consecutive FAILING checks; drives the /ratchet offer
|
|
111
|
+
last_verify_cmd: str = "" # the command behind that streak (pre-fills the offer)
|
|
110
112
|
budget: "BudgetState" = field(default_factory=lambda: BudgetState()) # scoped budgets
|
|
111
113
|
|
|
112
114
|
|
|
@@ -308,12 +310,24 @@ def run(
|
|
|
308
310
|
# relevant to this task + phase, capped at max_tools (default 12).
|
|
309
311
|
# Trims nothing when already under the cap; core coding tools are
|
|
310
312
|
# never dropped.
|
|
313
|
+
# Pin Knowledge when the project HAS a GraphRAG index. Its purpose
|
|
314
|
+
# is to answer plain questions from the user's own corpus, but the
|
|
315
|
+
# keyword gate only fires on knowledge/graph/entity/graphrag/ingest
|
|
316
|
+
# — words a user asking "what does the auth module do?" never types
|
|
317
|
+
# — so it was trimmed and the call came back "not available here".
|
|
318
|
+
_pins = list(turn_config.get("pin_tools") or [])
|
|
319
|
+
try:
|
|
320
|
+
from drydock.graphrag import knowledge_base_exists
|
|
321
|
+
if "Knowledge" not in _pins and knowledge_base_exists(os.getcwd()):
|
|
322
|
+
_pins.append("Knowledge")
|
|
323
|
+
except Exception:
|
|
324
|
+
pass # advisory: never let this break the turn
|
|
311
325
|
available = select_tools(
|
|
312
326
|
available,
|
|
313
327
|
phase=str(state.task.phase),
|
|
314
328
|
task_text=state.task.objective,
|
|
315
329
|
max_tools=turn_config.get("max_tools", DEFAULT_MAX_TOOLS),
|
|
316
|
-
pin_tools=
|
|
330
|
+
pin_tools=_pins,
|
|
317
331
|
)
|
|
318
332
|
for event in stream(
|
|
319
333
|
model=turn_config["model"],
|
|
@@ -745,6 +759,16 @@ def run(
|
|
|
745
759
|
_vcmd = (tc.get("input") or {}).get("command", "")
|
|
746
760
|
if looks_like_verification(_vcmd):
|
|
747
761
|
last_verification = parse_evidence(_vcmd, result)
|
|
762
|
+
# Streak of consecutive FAILING checks, for the TUI's proactive
|
|
763
|
+
# /ratchet offer (ratchet.ratchet_offer). Reset on any pass, so it
|
|
764
|
+
# only fires while the model is genuinely stuck re-running the same
|
|
765
|
+
# check — which is precisely the ratchet's use case. Advisory: this
|
|
766
|
+
# counter never changes the agent's own behaviour.
|
|
767
|
+
if last_verification.status == "fail":
|
|
768
|
+
state.verify_fail_streak += 1
|
|
769
|
+
state.last_verify_cmd = _vcmd
|
|
770
|
+
elif last_verification.status == "pass":
|
|
771
|
+
state.verify_fail_streak = 0
|
|
748
772
|
# Accumulate what the checks touched (command + result text),
|
|
749
773
|
# capped, for criteria-coverage at completion time.
|
|
750
774
|
if len(verification_text) < 200_000:
|
|
@@ -0,0 +1,218 @@
|
|
|
1
|
+
"""Bottleneck register — after a system is decomposed, rank its KNOWN components by
|
|
2
|
+
how much the objective actually moves if you improve each one, and attack the one
|
|
3
|
+
limiting factor instead of optimising everything evenly.
|
|
4
|
+
|
|
5
|
+
WHY THIS IS A SEPARATE STEP FROM THE LEDGER
|
|
6
|
+
-------------------------------------------
|
|
7
|
+
`groundtruth.py` ranks UNKNOWNS by decision impact — it answers "what don't I know
|
|
8
|
+
that could change the plan?" That is an epistemic question. The bottleneck is a
|
|
9
|
+
different question, and conflating the two loses it: the limiting factor is usually
|
|
10
|
+
something you already KNOW with no uncertainty at all. You can be perfectly certain
|
|
11
|
+
which stage is slowest and still pour effort into a faster one.
|
|
12
|
+
|
|
13
|
+
This project is its own evidence. Two of its largest course-corrections were bottleneck
|
|
14
|
+
findings, not discoveries:
|
|
15
|
+
· "throughput is the master lever" — the search was being made cleverer for weeks while
|
|
16
|
+
the objective barely moved, because the limiting factor was how many traces the box
|
|
17
|
+
could produce per hour, not how good each one was.
|
|
18
|
+
· "the real lever is training solves back into the model (write-back), not fancier
|
|
19
|
+
search" — the same shape a second time: effort concentrated on a component with lots
|
|
20
|
+
of headroom but little SHARE of the objective, while the component with the share sat
|
|
21
|
+
untouched.
|
|
22
|
+
Neither of those was an unknown. `next_test()` would not have surfaced either, because
|
|
23
|
+
there was nothing to learn — only a limiting factor to name and attack.
|
|
24
|
+
|
|
25
|
+
THE ONE NUMBER IT COMPUTES
|
|
26
|
+
--------------------------
|
|
27
|
+
For a component that controls a fraction `share` of the objective's gap and still has a
|
|
28
|
+
fraction `headroom` left to improve, the realizable gain from perfecting it is
|
|
29
|
+
`share * headroom`. The bottleneck is the component with the largest realizable gain —
|
|
30
|
+
NOT the one that is easiest to improve (high headroom, low share: the trap this whole
|
|
31
|
+
module exists to name), and NOT the biggest cost that happens to be a fundamental
|
|
32
|
+
constraint you cannot move (high share, zero headroom: a wall, not a lever).
|
|
33
|
+
|
|
34
|
+
The Amdahl ceiling `share` is reported alongside: it is the most that component could
|
|
35
|
+
EVER buy you, even with an infinitely good improvement. A lever with a 5% ceiling is
|
|
36
|
+
capped at 5% no matter how brilliant the optimisation — that ceiling is the sentence
|
|
37
|
+
that ends most premature-optimisation arguments.
|
|
38
|
+
|
|
39
|
+
WHY THIS IS NOT A PROMPT
|
|
40
|
+
Same reason as the other two modules: a reasoning checklist telling the model to "find
|
|
41
|
+
the bottleneck first" was measured on terminal-bench-2 and lost to a plain retry. This
|
|
42
|
+
gives the model somewhere to put the decomposition and computes the one comparison it is
|
|
43
|
+
bad at doing in its head — the share×headroom product across a handful of components.
|
|
44
|
+
|
|
45
|
+
Advisory by contract: nothing here raises on bad input, blocks a call, or edits a plan.
|
|
46
|
+
It records, ranks, and renders. See also `groundtruth.py` and `predictions.py`.
|
|
47
|
+
"""
|
|
48
|
+
from __future__ import annotations
|
|
49
|
+
|
|
50
|
+
import json
|
|
51
|
+
from dataclasses import dataclass, asdict
|
|
52
|
+
from pathlib import Path
|
|
53
|
+
|
|
54
|
+
# A component is a "wall" (fundamental constraint, not a lever) once its headroom is
|
|
55
|
+
# this low: no realistic optimisation moves it, so naming it stops you re-attacking it.
|
|
56
|
+
_WALL_HEADROOM = 0.05
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def _clamp01(x: object, default: float = 0.0) -> float:
|
|
60
|
+
try:
|
|
61
|
+
v = float(x) # type: ignore[arg-type]
|
|
62
|
+
except (TypeError, ValueError):
|
|
63
|
+
return default
|
|
64
|
+
if v != v: # NaN
|
|
65
|
+
return default
|
|
66
|
+
return 0.0 if v < 0.0 else 1.0 if v > 1.0 else v
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
@dataclass
|
|
70
|
+
class Component:
|
|
71
|
+
"""One piece of the decomposed system, and its two numbers that matter."""
|
|
72
|
+
id: str
|
|
73
|
+
name: str = ""
|
|
74
|
+
# Fraction of the objective's gap this component controls (its Amdahl share). If the
|
|
75
|
+
# objective is end-to-end latency and this stage is 70% of it, share = 0.7. This is a
|
|
76
|
+
# SENSITIVITY, not the effort currently spent on it — the whole point is that the two
|
|
77
|
+
# come apart.
|
|
78
|
+
share: float = 0.0
|
|
79
|
+
# Fraction of THIS component still improvable. 0 = a fundamental constraint (physics,
|
|
80
|
+
# a hard resource limit, already optimal) you cannot move; 1 = could be eliminated
|
|
81
|
+
# entirely. This is where step 4 (fundamental constraints) enters the arithmetic.
|
|
82
|
+
headroom: float = 1.0
|
|
83
|
+
note: str = ""
|
|
84
|
+
|
|
85
|
+
def gain(self) -> float:
|
|
86
|
+
"""Realizable objective gain from exploiting this component's headroom."""
|
|
87
|
+
return self.share * self.headroom
|
|
88
|
+
|
|
89
|
+
def ceiling(self) -> float:
|
|
90
|
+
"""Amdahl ceiling: the most this component could EVER buy, headroom aside."""
|
|
91
|
+
return self.share
|
|
92
|
+
|
|
93
|
+
def is_wall(self) -> bool:
|
|
94
|
+
return self.share > 0 and self.headroom <= _WALL_HEADROOM
|
|
95
|
+
|
|
96
|
+
def summary(self) -> str:
|
|
97
|
+
mark = "▮" if self.is_wall() else "▲"
|
|
98
|
+
return (f"{mark} {self.id} {self.name[:56]} "
|
|
99
|
+
f"[gain {self.gain():.0%} · ceiling {self.ceiling():.0%} · "
|
|
100
|
+
f"headroom {self.headroom:.0%}]")
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
class Bottlenecks:
|
|
104
|
+
"""The decomposed components of one problem, persisted next to the run."""
|
|
105
|
+
|
|
106
|
+
def __init__(self, path: str | Path | None = None):
|
|
107
|
+
self.path = Path(path) if path else None
|
|
108
|
+
self.items: list[Component] = []
|
|
109
|
+
if self.path and self.path.exists():
|
|
110
|
+
self.load()
|
|
111
|
+
|
|
112
|
+
# ── recording ────────────────────────────────────────────────────────────
|
|
113
|
+
def add(self, name: str, *, share: float = 0.0, headroom: float = 1.0,
|
|
114
|
+
note: str = "") -> Component:
|
|
115
|
+
"""Record a decomposed component. Out-of-range numbers are clamped to [0,1]
|
|
116
|
+
rather than raising — a register that throws would take down the turn it is
|
|
117
|
+
meant to help."""
|
|
118
|
+
c = Component(id=f"c{len(self.items) + 1}",
|
|
119
|
+
name=(name or "").strip(),
|
|
120
|
+
share=_clamp01(share, 0.0),
|
|
121
|
+
headroom=_clamp01(headroom, 1.0),
|
|
122
|
+
note=(note or "").strip())
|
|
123
|
+
self.items.append(c)
|
|
124
|
+
self._save()
|
|
125
|
+
return c
|
|
126
|
+
|
|
127
|
+
def update(self, cid: str, *, share: float | None = None,
|
|
128
|
+
headroom: float | None = None, note: str | None = None) -> Component | None:
|
|
129
|
+
"""Revise a component's numbers as measurement replaces estimate. Attacking the
|
|
130
|
+
bottleneck moves its headroom toward 0, at which point the ranking shifts to the
|
|
131
|
+
next limiting factor on its own."""
|
|
132
|
+
for c in self.items:
|
|
133
|
+
if c.id == cid:
|
|
134
|
+
if share is not None:
|
|
135
|
+
c.share = _clamp01(share, c.share)
|
|
136
|
+
if headroom is not None:
|
|
137
|
+
c.headroom = _clamp01(headroom, c.headroom)
|
|
138
|
+
if note is not None:
|
|
139
|
+
c.note = (note or "").strip()
|
|
140
|
+
self._save()
|
|
141
|
+
return c
|
|
142
|
+
return None
|
|
143
|
+
|
|
144
|
+
# ── the ranking that is the point of the module ──────────────────────────
|
|
145
|
+
def _levers(self) -> list[Component]:
|
|
146
|
+
"""Components that are actually worth attacking — positive gain and not a wall.
|
|
147
|
+
A wall's sliver of headroom is not a lever, so it is excluded here even though its
|
|
148
|
+
gain is technically > 0."""
|
|
149
|
+
return [c for c in self.items if c.gain() > 0 and not c.is_wall()]
|
|
150
|
+
|
|
151
|
+
def bottleneck(self) -> Component | None:
|
|
152
|
+
"""The one limiting factor to attack first: the largest realizable gain
|
|
153
|
+
(share × headroom). Ties break toward the larger share — the higher ceiling is
|
|
154
|
+
the safer bet once the immediate gain is equal."""
|
|
155
|
+
movable = self._levers()
|
|
156
|
+
if not movable:
|
|
157
|
+
return None
|
|
158
|
+
return sorted(movable, key=lambda c: (-c.gain(), -c.share, c.id))[0]
|
|
159
|
+
|
|
160
|
+
def ranked(self) -> list[Component]:
|
|
161
|
+
return sorted(self.items, key=lambda c: (-c.gain(), -c.share, c.id))
|
|
162
|
+
|
|
163
|
+
def walls(self) -> list[Component]:
|
|
164
|
+
"""High-share components with (almost) no headroom — fundamental constraints.
|
|
165
|
+
Naming them is what stops the loop re-attacking a wall as if it were a lever."""
|
|
166
|
+
return [c for c in self.items if c.is_wall()]
|
|
167
|
+
|
|
168
|
+
def misplaced_effort(self) -> Component | None:
|
|
169
|
+
"""The trap this module exists to catch: the component with the MOST headroom is
|
|
170
|
+
not the bottleneck — improving it feels productive but is capped by its small
|
|
171
|
+
share. Returns that decoy when it differs from the real bottleneck, else None."""
|
|
172
|
+
movable = self._levers()
|
|
173
|
+
if len(movable) < 2:
|
|
174
|
+
return None
|
|
175
|
+
by_headroom = max(movable, key=lambda c: (c.headroom, -c.share))
|
|
176
|
+
bn = self.bottleneck()
|
|
177
|
+
return by_headroom if bn is not None and by_headroom.id != bn.id else None
|
|
178
|
+
|
|
179
|
+
# ── rendering ────────────────────────────────────────────────────────────
|
|
180
|
+
def render(self) -> str:
|
|
181
|
+
if not self.items:
|
|
182
|
+
return "(no components decomposed yet)"
|
|
183
|
+
lines = [c.summary() for c in self.ranked()]
|
|
184
|
+
bn = self.bottleneck()
|
|
185
|
+
if bn is not None:
|
|
186
|
+
lines.append(f"→ attack first: {bn.id} ({bn.name[:50]}) — "
|
|
187
|
+
f"realizable gain {bn.gain():.0%}, ceiling {bn.ceiling():.0%}")
|
|
188
|
+
else:
|
|
189
|
+
lines.append("→ no movable component: every lever is a wall "
|
|
190
|
+
"(headroom ~0). The objective may need a different decomposition.")
|
|
191
|
+
decoy = self.misplaced_effort()
|
|
192
|
+
if decoy is not None:
|
|
193
|
+
lines.append(f"⚠ {decoy.id} has the most headroom but only a "
|
|
194
|
+
f"{decoy.ceiling():.0%} ceiling — optimising it is capped there.")
|
|
195
|
+
total_share = sum(c.share for c in self.items)
|
|
196
|
+
if total_share > 1.05:
|
|
197
|
+
lines.append(f"⚠ shares sum to {total_share:.0%} (>100%): the decomposition "
|
|
198
|
+
"overlaps — components are not independent.")
|
|
199
|
+
return "\n".join(lines)
|
|
200
|
+
|
|
201
|
+
# ── persistence ──────────────────────────────────────────────────────────
|
|
202
|
+
def _save(self) -> None:
|
|
203
|
+
if not self.path:
|
|
204
|
+
return
|
|
205
|
+
try:
|
|
206
|
+
self.path.parent.mkdir(parents=True, exist_ok=True)
|
|
207
|
+
self.path.write_text(
|
|
208
|
+
json.dumps([asdict(c) for c in self.items], indent=2), encoding="utf-8")
|
|
209
|
+
except OSError:
|
|
210
|
+
pass # advisory: never fail a run over bookkeeping
|
|
211
|
+
|
|
212
|
+
def load(self) -> None:
|
|
213
|
+
try:
|
|
214
|
+
raw = json.loads(self.path.read_text(encoding="utf-8")) if self.path else []
|
|
215
|
+
self.items = [Component(**{k: v for k, v in d.items()
|
|
216
|
+
if k in Component.__dataclass_fields__}) for d in raw]
|
|
217
|
+
except (OSError, json.JSONDecodeError, TypeError):
|
|
218
|
+
self.items = []
|
|
@@ -0,0 +1,135 @@
|
|
|
1
|
+
"""Inference-server capacity detection for swarm auto-sizing.
|
|
2
|
+
|
|
3
|
+
The swarm's useful agent count is bounded by how many requests the server actually serves
|
|
4
|
+
IN PARALLEL (§21/§22): the agents share one server, so past its real concurrency, extra
|
|
5
|
+
agents just queue — more tokens, no more wall-clock. So "hammer the 8-GPU vLLM box, tone
|
|
6
|
+
down the -np 2 laptop" reduces to: estimate the server's concurrency C, then size the swarm
|
|
7
|
+
to min(C, task-demand, budget). This module estimates C.
|
|
8
|
+
|
|
9
|
+
Detection order (first hit wins), all swallow-error / never-raise:
|
|
10
|
+
1. explicit — `config['swarm_concurrency']` or $DRYDOCK_SWARM_CONCURRENCY (the operator
|
|
11
|
+
knows their hardware; always the most reliable);
|
|
12
|
+
2. llama.cpp — GET /slots (its `-np` parallel slots);
|
|
13
|
+
3. empirical probe — fire a burst of tiny concurrent requests and infer parallelism from
|
|
14
|
+
wall-time vs single-request latency (server-agnostic; the honest signal for vLLM, whose
|
|
15
|
+
max_num_seqs isn't exposed over the OpenAI API);
|
|
16
|
+
4. fallback — a conservative default.
|
|
17
|
+
|
|
18
|
+
Stdlib-only; the probe funcs are injectable so this is testable without a live server.
|
|
19
|
+
"""
|
|
20
|
+
from __future__ import annotations
|
|
21
|
+
|
|
22
|
+
import concurrent.futures
|
|
23
|
+
import json
|
|
24
|
+
import os
|
|
25
|
+
import time
|
|
26
|
+
import urllib.request
|
|
27
|
+
from collections.abc import Callable
|
|
28
|
+
|
|
29
|
+
DEFAULT_CONCURRENCY = 4 # conservative when nothing else is known
|
|
30
|
+
DEFAULT_TASK_DEMAND = 6 # distinct approaches worth trying before agents just duplicate
|
|
31
|
+
HARD_MAX = 64 # never spawn more than this regardless of hardware
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def _server_root(base_url: str) -> str:
|
|
35
|
+
"""Strip a trailing /v1 so we can hit sibling paths like /slots."""
|
|
36
|
+
root = base_url.rstrip("/")
|
|
37
|
+
if root.endswith("/v1"):
|
|
38
|
+
root = root[:-3].rstrip("/")
|
|
39
|
+
return root
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def _get_json(url: str, timeout: float = 4.0):
|
|
43
|
+
try:
|
|
44
|
+
with urllib.request.urlopen(url, timeout=timeout) as r: # noqa: S310 — operator's own server
|
|
45
|
+
return json.loads(r.read().decode("utf-8"))
|
|
46
|
+
except Exception: # noqa: BLE001 — detection is best-effort, never raises
|
|
47
|
+
return None
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def slots_concurrency(base_url: str, *, getter: Callable[[str], object] | None = None) -> int | None:
|
|
51
|
+
"""llama.cpp exposes its `-np` parallel slots at /slots (a list). len = concurrency."""
|
|
52
|
+
get = getter or (lambda u: _get_json(u))
|
|
53
|
+
d = get(_server_root(base_url) + "/slots")
|
|
54
|
+
if isinstance(d, list) and d:
|
|
55
|
+
return len(d)
|
|
56
|
+
return None
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def _default_requester(base_url: str, model: str) -> Callable[[], None]:
|
|
60
|
+
def req() -> None:
|
|
61
|
+
payload = json.dumps({
|
|
62
|
+
"model": model,
|
|
63
|
+
"messages": [{"role": "user", "content": "ping"}],
|
|
64
|
+
"max_tokens": 1, "temperature": 0,
|
|
65
|
+
}).encode("utf-8")
|
|
66
|
+
r = urllib.request.Request( # noqa: S310 — operator's own server
|
|
67
|
+
base_url.rstrip("/") + "/chat/completions", data=payload,
|
|
68
|
+
headers={"Content-Type": "application/json"})
|
|
69
|
+
with urllib.request.urlopen(r, timeout=60) as resp:
|
|
70
|
+
resp.read()
|
|
71
|
+
return req
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def probe_concurrency(base_url: str, model: str, *, burst: int = 16,
|
|
75
|
+
requester: Callable[[], None] | None = None) -> int:
|
|
76
|
+
"""Empirically estimate useful parallelism: one warm request costs t1; `burst` concurrent
|
|
77
|
+
requests finish in wall time W. Total work ≈ burst·t1 done in W ⇒ parallelism ≈ burst·t1/W
|
|
78
|
+
(≈burst if fully batched, ≈1 if serialized). Server-agnostic. Returns 1..burst."""
|
|
79
|
+
req = requester or _default_requester(base_url, model)
|
|
80
|
+
try:
|
|
81
|
+
t0 = time.monotonic()
|
|
82
|
+
req()
|
|
83
|
+
t1 = time.monotonic() - t0
|
|
84
|
+
except Exception: # noqa: BLE001
|
|
85
|
+
return 1
|
|
86
|
+
if t1 <= 0:
|
|
87
|
+
return burst
|
|
88
|
+
start = time.monotonic()
|
|
89
|
+
try:
|
|
90
|
+
with concurrent.futures.ThreadPoolExecutor(max_workers=burst) as ex:
|
|
91
|
+
list(ex.map(lambda _: _safe_call(req), range(burst)))
|
|
92
|
+
except Exception: # noqa: BLE001
|
|
93
|
+
return 1
|
|
94
|
+
wall = time.monotonic() - start
|
|
95
|
+
if wall <= 0:
|
|
96
|
+
return burst
|
|
97
|
+
return max(1, min(burst, round(burst * t1 / wall)))
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def _safe_call(req: Callable[[], None]) -> None:
|
|
101
|
+
try:
|
|
102
|
+
req()
|
|
103
|
+
except Exception: # noqa: BLE001
|
|
104
|
+
pass
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
def detect_concurrency(base_url: str, *, provider: str = "vllm", model: str = "",
|
|
108
|
+
config: dict | None = None, probe: bool = True) -> int:
|
|
109
|
+
"""Estimate the server's useful concurrency C (see module docstring for the order)."""
|
|
110
|
+
config = config or {}
|
|
111
|
+
override = config.get("swarm_concurrency") or os.environ.get("DRYDOCK_SWARM_CONCURRENCY")
|
|
112
|
+
if override:
|
|
113
|
+
try:
|
|
114
|
+
return max(1, int(override))
|
|
115
|
+
except (TypeError, ValueError):
|
|
116
|
+
pass
|
|
117
|
+
slots = slots_concurrency(base_url)
|
|
118
|
+
if slots:
|
|
119
|
+
return slots
|
|
120
|
+
if probe and model:
|
|
121
|
+
c = probe_concurrency(base_url, model)
|
|
122
|
+
if c:
|
|
123
|
+
return c
|
|
124
|
+
return int(config.get("swarm_concurrency_default", DEFAULT_CONCURRENCY))
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
def swarm_size(concurrency: int, *, task_demand: int = DEFAULT_TASK_DEMAND,
|
|
128
|
+
budget_agents: int | None = None, hard_max: int = HARD_MAX) -> int:
|
|
129
|
+
"""N = min(hardware concurrency, task-demand, budget) — never maximized (§39: more
|
|
130
|
+
agents ≠ more intelligence). On an 8-GPU box concurrency is high so task-demand binds
|
|
131
|
+
(no wasteful cloning); on a small box concurrency binds (auto tone-down)."""
|
|
132
|
+
n = min(max(1, concurrency), max(1, task_demand), hard_max)
|
|
133
|
+
if budget_agents:
|
|
134
|
+
n = min(n, max(1, budget_agents))
|
|
135
|
+
return max(1, n)
|
|
@@ -373,6 +373,16 @@ def main():
|
|
|
373
373
|
cfg["cwd"] = os.getcwd()
|
|
374
374
|
sys.exit(run_cli(sys.argv[2:], config=cfg))
|
|
375
375
|
|
|
376
|
+
# `drydock swarm <objective> …` / `swarm status|list|resume` — multi-agent swarm.
|
|
377
|
+
# Same argv-intercept pattern as eratchet so the subcommand parses its own flags.
|
|
378
|
+
if len(sys.argv) > 1 and sys.argv[1] == "swarm":
|
|
379
|
+
from drydock import config as cfgmod
|
|
380
|
+
from drydock.swarm import run_cli as swarm_run_cli
|
|
381
|
+
cfg = cfgmod.resolve({}, cfgmod.default_config_path())
|
|
382
|
+
cfgmod.resolve_active_model(cfg)
|
|
383
|
+
cfg["cwd"] = os.getcwd()
|
|
384
|
+
sys.exit(swarm_run_cli(sys.argv[2:], config=cfg))
|
|
385
|
+
|
|
376
386
|
parser = argparse.ArgumentParser(description="DryDock — local coding agent")
|
|
377
387
|
parser.add_argument(
|
|
378
388
|
"--version", "-V", action="version", version=f"drydock {__version__}"
|
|
@@ -40,6 +40,15 @@ DEFAULTS: dict[str, object] = {
|
|
|
40
40
|
# generation usually isn't stalled — a known gemma/llama.cpp hang). 0 = off.
|
|
41
41
|
# Set to e.g. 600 on a stall-prone local server. Bounded to a few retries.
|
|
42
42
|
"stall_retry_secs": 0,
|
|
43
|
+
# Idle-output watchdog for foreground Bash: if a running command produces NO
|
|
44
|
+
# output for this many seconds, drydock stops BLOCKING on it and adopts it as a
|
|
45
|
+
# background job (it keeps running; the agent gets the prompt back + a nudge to
|
|
46
|
+
# reconsider). Complements the total-runtime auto-background (_AUTO_BG_TIMEOUT):
|
|
47
|
+
# catches a command hung SILENTLY on a long timeout (e.g. a dead network
|
|
48
|
+
# download the model set timeout:1800 for) far sooner than the total cap.
|
|
49
|
+
# Non-destructive (backgrounds, never kills). 0 = off. Set e.g. 300 on the fleet
|
|
50
|
+
# where a weak local model loops on a silently-stuck command.
|
|
51
|
+
"bash_idle_bg_secs": 0,
|
|
43
52
|
# Inject bundled technique recipes (drydock/recipes.py) relevant to the task
|
|
44
53
|
# into the system prompt, so a local model has the *method* a task needs
|
|
45
54
|
# instead of guessing. Retrieval is keyword-overlap; only relevant recipes are
|
|
@@ -88,6 +97,17 @@ DEFAULTS: dict[str, object] = {
|
|
|
88
97
|
# web tools surfaced even on tasks whose text never mentions the web. Names
|
|
89
98
|
# not in the registry are ignored.
|
|
90
99
|
"pin_tools": [],
|
|
100
|
+
# When a single agent stays stuck on a KNOWN verifier (it keeps failing), the harness
|
|
101
|
+
# may auto-escalate to an in-process multi-agent swarm to explore approaches in parallel
|
|
102
|
+
# (docs/multi_agent_swarm_prd.md). On by default; set false to keep every request single-
|
|
103
|
+
# agent unless the user runs /swarm explicitly.
|
|
104
|
+
"swarm_auto_escalate": True,
|
|
105
|
+
# Swarm sizing: the harness caps agents at the inference server's real concurrency
|
|
106
|
+
# (drydock/capacity.py) — hammer an 8-GPU vLLM box, tone down a -np 2 laptop. Set
|
|
107
|
+
# swarm_concurrency to your server's parallel capacity to skip auto-detection (most
|
|
108
|
+
# reliable); 0 = auto-detect (llama.cpp /slots, else an empirical burst probe).
|
|
109
|
+
"swarm_concurrency": 0,
|
|
110
|
+
"swarm_concurrency_default": 4,
|
|
91
111
|
# URL substrings the web tools refuse: WebSearch drops matching results,
|
|
92
112
|
# WebFetch declines matching URLs (with a plain message, never an error).
|
|
93
113
|
# Used to keep benchmark/solution sites out of harvested training runs.
|
|
@@ -106,6 +106,7 @@ def run_eratchet(
|
|
|
106
106
|
on_event: Optional[Callable[[str, dict], None]] = None,
|
|
107
107
|
should_stop: Optional[Callable[[], bool]] = None,
|
|
108
108
|
capture: Optional[Callable[[dict], None]] = None,
|
|
109
|
+
share: bool = False,
|
|
109
110
|
) -> EratchetResult:
|
|
110
111
|
"""Drive the parallel evolutionary ratchet. Pure control flow: every side
|
|
111
112
|
effect (running an attempt, verifying, snapshotting) happens inside the
|
|
@@ -137,6 +138,7 @@ def run_eratchet(
|
|
|
137
138
|
base_ref: Optional[str] = None
|
|
138
139
|
prev_best = -1
|
|
139
140
|
flat = 0
|
|
141
|
+
peer_log: list = [] # (generation, passed, total, summary) — the blackboard when share=True
|
|
140
142
|
|
|
141
143
|
emit("start", goal=goal, effort=effort, generations=gens,
|
|
142
144
|
fanout=policy.fanout, servers=list(servers))
|
|
@@ -154,6 +156,15 @@ def run_eratchet(
|
|
|
154
156
|
if pairs:
|
|
155
157
|
xplan = plan_crossover(pairs[0][0], pairs[0][1], g)
|
|
156
158
|
|
|
159
|
+
# BLACKBOARD (§10): feed prior variants' attempts+scores into this generation's
|
|
160
|
+
# specs so its variants generate INFORMED rather than blind. share=False → today's
|
|
161
|
+
# blind eratchet (the baseline arm of the eratchet ± blackboard control).
|
|
162
|
+
if share and peer_log:
|
|
163
|
+
notes = _peer_notes_erx(peer_log)
|
|
164
|
+
for s in specs:
|
|
165
|
+
s["peer_notes"] = notes
|
|
166
|
+
emit("share", generation=g, prior_attempts=len(peer_log))
|
|
167
|
+
|
|
157
168
|
emit("generation_start", generation=g, operator=op,
|
|
158
169
|
variants=len(specs), modes=[s.get("mode") for s in specs])
|
|
159
170
|
|
|
@@ -169,6 +180,8 @@ def run_eratchet(
|
|
|
169
180
|
cost=float(g),
|
|
170
181
|
)
|
|
171
182
|
archive.consider(cand)
|
|
183
|
+
if share:
|
|
184
|
+
peer_log.append((g, oc.passed, oc.total, _last_msg(oc.messages)))
|
|
172
185
|
if capture:
|
|
173
186
|
capture({
|
|
174
187
|
"goal": goal, "generation": g, "index": i,
|
|
@@ -251,21 +264,54 @@ def _git(args: list, cwd: str) -> subprocess.CompletedProcess:
|
|
|
251
264
|
return subprocess.run(["git", *args], cwd=cwd, capture_output=True, text=True)
|
|
252
265
|
|
|
253
266
|
|
|
267
|
+
def _last_msg(messages) -> str:
|
|
268
|
+
"""The variant's final assistant line — a one-line 'what it tried' for the blackboard."""
|
|
269
|
+
for m in reversed(messages or []):
|
|
270
|
+
if isinstance(m, dict) and m.get("role") == "assistant":
|
|
271
|
+
c = (m.get("content") or "").strip()
|
|
272
|
+
if c:
|
|
273
|
+
return c.splitlines()[0][:160]
|
|
274
|
+
return ""
|
|
275
|
+
|
|
276
|
+
|
|
277
|
+
def _peer_notes_erx(log: list, limit: int = 8) -> str:
|
|
278
|
+
"""Digest of what earlier variants already tried, best-scoring first (§10). Turns
|
|
279
|
+
eratchet's blind parallel search into a coordinated one: a new variant sees the prior
|
|
280
|
+
attempts' scores + approaches so it builds on the partials and skips the dead ends.
|
|
281
|
+
`log` items are (generation, passed, total, summary)."""
|
|
282
|
+
ranked = sorted(log, key=lambda r: ((r[1] / r[2]) if r[2] else 0.0, r[1]), reverse=True)
|
|
283
|
+
lines = []
|
|
284
|
+
for (g, p, t, summ) in ranked[:limit]:
|
|
285
|
+
score = f"{p}/{t}" if t else "unverified"
|
|
286
|
+
lines.append(f"- gen{g} [{score}]: {summ or '(no summary)'}")
|
|
287
|
+
return "\n".join(lines)
|
|
288
|
+
|
|
289
|
+
|
|
254
290
|
def _variant_prompt(goal: str, spec: dict, progress: dict, xplan, donor_path: str) -> str:
|
|
255
291
|
passed, total = progress.get("passed", 0), progress.get("total", 0)
|
|
256
292
|
mode = spec.get("mode", "continue")
|
|
257
293
|
if mode == "crossover" and xplan and donor_path:
|
|
258
294
|
wants = ", ".join(xplan.get("wants", [])) or "the checks it passes that this tree fails"
|
|
259
|
-
|
|
295
|
+
base = (
|
|
260
296
|
f"{goal}\n\n[CROSSOVER — a COMPLEMENTARY solution is checked out at "
|
|
261
297
|
f"{donor_path}. It passes: {wants}. Study how it does so and fold ONLY "
|
|
262
298
|
f"those parts into THIS working tree, keeping everything that already "
|
|
263
299
|
f"passes here. A verifier will score the result.]"
|
|
264
300
|
)
|
|
265
|
-
|
|
266
|
-
|
|
267
|
-
|
|
268
|
-
|
|
301
|
+
else:
|
|
302
|
+
op = _MODE_OP.get(mode, "exploit")
|
|
303
|
+
if op in ("diversify", "restart"):
|
|
304
|
+
base = diversify_prompt(goal, passed, total, op)
|
|
305
|
+
else:
|
|
306
|
+
base = continuation_prompt(goal, passed, total)
|
|
307
|
+
# BLACKBOARD (§10): if the orchestrator attached peers' notes to this spec, the variant
|
|
308
|
+
# generates INFORMED instead of blind — this is what turns eratchet from redundant
|
|
309
|
+
# parallel search into coordinated search.
|
|
310
|
+
notes = spec.get("peer_notes")
|
|
311
|
+
if notes:
|
|
312
|
+
base += (f"\n\n[WHAT OTHER VARIANTS ALREADY TRIED on this SAME goal — build past the "
|
|
313
|
+
f"highest partial below; do NOT repeat the failed approaches:\n{notes}]")
|
|
314
|
+
return base
|
|
269
315
|
|
|
270
316
|
|
|
271
317
|
@dataclass
|
|
@@ -403,10 +449,15 @@ def parse_eratchet(tokens: list) -> dict:
|
|
|
403
449
|
capture = ""
|
|
404
450
|
servers: list = []
|
|
405
451
|
fanout = generations = 0
|
|
452
|
+
share = False
|
|
406
453
|
i, seen_flag = 0, False
|
|
407
454
|
while i < len(tokens):
|
|
408
455
|
t = tokens[i]
|
|
409
|
-
if t
|
|
456
|
+
if t == "--share":
|
|
457
|
+
# blackboard: later generations' variants read prior attempts' scores+approaches
|
|
458
|
+
share = True
|
|
459
|
+
seen_flag = True
|
|
460
|
+
elif t in ("--effort", "--verify", "--servers", "--fanout", "--generations",
|
|
410
461
|
"--model", "--provider", "--base-url", "--capture"):
|
|
411
462
|
seen_flag = True
|
|
412
463
|
i += 1
|
|
@@ -443,7 +494,7 @@ def parse_eratchet(tokens: list) -> dict:
|
|
|
443
494
|
return {"goal": " ".join(goal), "effort": effort, "verify": verify,
|
|
444
495
|
"servers": servers, "fanout": fanout, "generations": generations,
|
|
445
496
|
"model": model, "provider": provider, "base_url": base_url,
|
|
446
|
-
"capture": capture}
|
|
497
|
+
"capture": capture, "share": share}
|
|
447
498
|
|
|
448
499
|
|
|
449
500
|
def resolve_config(opts: dict, config: dict) -> "ExecConfig | str":
|
|
@@ -511,7 +562,8 @@ def run_cli(argv: list, config: dict | None = None) -> int:
|
|
|
511
562
|
try:
|
|
512
563
|
res = run_eratchet(cfg.goal, servers=servers, runner=make_variant_runner(cfg),
|
|
513
564
|
effort=opts["effort"], max_generations=opts["generations"],
|
|
514
|
-
fanout=opts["fanout"], on_event=on_event, capture=capture
|
|
565
|
+
fanout=opts["fanout"], on_event=on_event, capture=capture,
|
|
566
|
+
share=opts.get("share", False))
|
|
515
567
|
finally:
|
|
516
568
|
if cap_fh:
|
|
517
569
|
cap_fh.close()
|
|
@@ -65,6 +65,26 @@ def default_store_path(cwd: str) -> Path:
|
|
|
65
65
|
return Path(cwd) / ".drydock" / "graphrag.db"
|
|
66
66
|
|
|
67
67
|
|
|
68
|
+
def knowledge_base_exists(cwd: str) -> bool:
|
|
69
|
+
"""True when this project has a built GraphRAG index (SQLite or legacy JSON).
|
|
70
|
+
|
|
71
|
+
Used to PIN the `Knowledge` tool for the turn. Dynamic tool selection
|
|
72
|
+
(tool_select) otherwise gates Knowledge behind the keywords
|
|
73
|
+
knowledge/graph/entity/graphrag/ingest, which the user has no reason to type:
|
|
74
|
+
the whole point of the feature is that a plain question like "what does the
|
|
75
|
+
auth module do?" is answered FROM their index. With 46 tools and a cap of 12,
|
|
76
|
+
that question dropped Knowledge from the toolset and the model's call came
|
|
77
|
+
back "[The 'Knowledge' tool is not available here]" — the bug reported
|
|
78
|
+
2026-09-03. Knowledge is read-only and cheap, so pinning it whenever an index
|
|
79
|
+
exists costs one slot and restores the documented behaviour; when no index
|
|
80
|
+
exists nothing is pinned and the tool stays gated as before.
|
|
81
|
+
"""
|
|
82
|
+
try:
|
|
83
|
+
return _resolve_store(default_store_path(cwd)).exists()
|
|
84
|
+
except Exception: # never let a tool-selection helper break the turn
|
|
85
|
+
return False
|
|
86
|
+
|
|
87
|
+
|
|
68
88
|
def _resolve_store(store_path) -> Path:
|
|
69
89
|
"""Given a requested store path, return the one that actually exists — a
|
|
70
90
|
``.db`` (SQLite) if present, else a legacy ``.json`` sibling, else the path
|