drydock-cli 3.1.35__tar.gz → 3.1.37__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {drydock_cli-3.1.35/drydock_cli.egg-info → drydock_cli-3.1.37}/PKG-INFO +49 -26
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/README.md +48 -25
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/agent.py +12 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/cli.py +62 -1
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/compaction.py +13 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/config.py +6 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/providers.py +43 -7
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/swarm.py +23 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/tool_select.py +7 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/tools/__init__.py +76 -324
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/tui/app.py +1 -1
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/tuning.py +21 -3
- {drydock_cli-3.1.35 → drydock_cli-3.1.37/drydock_cli.egg-info}/PKG-INFO +49 -26
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock_cli.egg-info/SOURCES.txt +1 -6
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/pyproject.toml +1 -1
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_skills.py +0 -1
- drydock_cli-3.1.37/tests/test_text_only_model.py +78 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_tuning.py +11 -0
- drydock_cli-3.1.35/drydock/builtin_skills/fiar-assess.md +0 -33
- drydock_cli-3.1.35/drydock/builtin_skills/fiar-cap.md +0 -26
- drydock_cli-3.1.35/drydock/builtin_skills/fiar-evidence.md +0 -33
- drydock_cli-3.1.35/drydock/builtin_skills/fiar-readiness.md +0 -28
- drydock_cli-3.1.35/drydock/fiar.py +0 -680
- drydock_cli-3.1.35/tests/test_fiar.py +0 -268
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/LICENSE +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/NOTICE +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/__init__.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/__main__.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/advisor.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/asr.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/bash_safety.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/bottleneck.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/budget.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/builtin_skills/__init__.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/builtin_skills/document-canvas.md +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/builtin_skills/ml-data.md +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/builtin_skills/ml-debug.md +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/builtin_skills/ml-finetune.md +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/builtin_skills/ml-metrics.md +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/builtin_skills/ml-rl.md +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/builtin_skills/ml-train.md +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/builtin_skills/nist-ai-rmf.md +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/builtin_skills/nist-csf.md +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/builtin_skills/rmf-categorize.md +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/builtin_skills/rmf-control.md +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/builtin_skills/rmf-poam.md +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/builtin_skills/rmf-review.md +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/builtin_skills/stig-assess.md +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/builtin_skills/stig-remediate.md +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/capacity.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/cci.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/comms/__init__.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/comms/attention.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/comms/channels.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/comms/events.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/detect.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/doccanvas.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/eratchet.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/events.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/extract.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/gittools.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/graphrag.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/groundtruth.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/guards.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/jobs.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/loop_detect.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/mcp.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/mission.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/mission_cli.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/mission_run.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/pdfredbox.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/phases.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/poam.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/predictions.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/progress.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/ratchet.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/recipes.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/recovery.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/resume.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/rmf.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/rmf_graph.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/skills.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/stig.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/subagents.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/suggest.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/task_state.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/tool_policy.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/tool_registry.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/tool_result.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/tool_validate.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/trajectory.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/tui/__init__.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/tui/approval.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/tui/messages.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/tui/widgets.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/verification.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/web.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock_cli.egg-info/dependency_links.txt +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock_cli.egg-info/entry_points.txt +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock_cli.egg-info/requires.txt +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock_cli.egg-info/top_level.txt +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/setup.cfg +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_advisor.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_approval.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_asr.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_back_command.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_bash_background.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_bash_binary_output.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_bash_crossplatform.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_bash_output_bounding.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_bash_process_group.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_bash_safety.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_bash_sanitize.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_bash_shell.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_bash_stdin.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_bash_stop_partial.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_bash_timeout_network.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_bash_timeout_param.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_bottleneck.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_budget.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_capacity.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_cci.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_cli_agents.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_comms.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_compact_command.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_compaction.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_config.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_config_migration.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_context_limit_config.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_context_limit_issue25.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_criteria_coverage.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_cycling_detection.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_declared_deps.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_degenerate_argument.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_detect.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_dispatch.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_doccanvas.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_e2e_connected.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_edit_replace_all.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_edit_thrash_and_compaction.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_effort_governor.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_empty_response.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_eratchet.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_events.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_extract.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_failure_loop.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_first_run_setup.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_gittools.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_glob_edit_edges.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_graphify_example.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_graphrag.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_graphrag_quoted_path.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_graphrag_sqlite.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_grep_and_read_robust.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_groundtruth_predictions.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_guards_and_tools.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_hallucinated_tools.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_jobs.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_knowledge_tool_available.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_leaked_tool_call.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_loop_detect.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_main_api_key.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_mcp.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_mission.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_mission_cli.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_mission_run.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_model_registry.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_oneshot_unreachable.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_overthink_interrupt.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_pdfredbox.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_phases.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_plan_autocontinue.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_poam.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_progress.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_providers_unreachable.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_ratchet.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_ratchet_offer.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_read_index.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_recipes.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_recovery.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_recovery_config.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_recovery_integration.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_repeated_outcome.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_repetition_interrupt.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_resume_events.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_resume_restore.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_resume_snapshot.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_rmf.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_rmf_graph.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_rmf_stig_graph.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_rolling_plan.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_runaway_repetition.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_screenshot.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_server_probe.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_sqlite_events.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_stall_retry.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_stig.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_stop.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_streaming_newlines.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_subagent.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_subagents.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_suggest.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_swarm.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_system_prompt_help.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_task_state.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_timeline.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_todo.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_tool_arg_coercion.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_tool_arg_coercion_more.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_tool_arg_parsing.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_tool_canonical.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_tool_policy.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_tool_result.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_tool_select.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_tool_validate.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_tools_undo.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_trajectory.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_tui.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_verification_gate.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_viewimage.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_vision_input.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_web_tools.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_windows_shell.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_worker_subagent.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_write_content_coerce.py +0 -0
- {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_xccdf.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: drydock-cli
|
|
3
|
-
Version: 3.1.
|
|
3
|
+
Version: 3.1.37
|
|
4
4
|
Summary: Drydock — a local, provider-agnostic terminal coding agent for local LLMs
|
|
5
5
|
Author: Frank Bobe III
|
|
6
6
|
License-Expression: Apache-2.0
|
|
@@ -32,11 +32,16 @@ Dynamic: license-file
|
|
|
32
32
|
|
|
33
33
|
# ⚓ Drydock
|
|
34
34
|
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
35
|
+
**The air-gapped terminal coding agent that automates your NIST compliance** —
|
|
36
|
+
RMF (800-53), CSF 2.0, AI RMF, and DISA STIG — running entirely on your own
|
|
37
|
+
hardware against a local LLM. No accounts, no telemetry, no cloud: your code,
|
|
38
|
+
controls, and CUI never leave the box.
|
|
39
|
+
|
|
40
|
+
Under the compliance skills it's a full agentic coding harness — the same
|
|
41
|
+
air-gapped agent whether you're assessing a control or shipping a feature. The
|
|
42
|
+
only outbound calls are to the model endpoint you configure and (optionally) the
|
|
43
|
+
web-search tools you invoke. Primary target: **dense Gemma-4-31B** (QAT, 64K)
|
|
44
|
+
served by llama.cpp on a single workstation.
|
|
40
45
|
|
|
41
46
|
> **v3 — clean-room rebuild.** Drydock is being rebuilt as an original,
|
|
42
47
|
> Apache-2.0 codebase owned end to end (no upstream fork). Every release is
|
|
@@ -46,10 +51,14 @@ single workstation.
|
|
|
46
51
|
|
|
47
52
|
## Why
|
|
48
53
|
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
54
|
+
Regulated and air-gapped environments — federal, defense, critical-infrastructure,
|
|
55
|
+
and any org adopting NIST CSF 2.0 — can't send code, controls, or CUI to a cloud
|
|
56
|
+
agent, yet still need real automation. Drydock runs entirely against a local model,
|
|
57
|
+
feels like a first-class terminal agent, ingests the NIST 800-53 catalog and DISA
|
|
58
|
+
STIG benchmarks to assess / remediate / export POA&Ms **offline**, and keeps its
|
|
59
|
+
data plane on your box. It's a governable agent by design: a deterministic control
|
|
60
|
+
loop, a verification gate on "done", and a durable, tamper-evident event trace —
|
|
61
|
+
the audit record for what an autonomous agent did in your environment.
|
|
53
62
|
|
|
54
63
|
## Status
|
|
55
64
|
|
|
@@ -60,6 +69,36 @@ line, and a multi-line prompt. The agent loop, OpenAI-compatible provider,
|
|
|
60
69
|
two-tier compaction, and the full agentic toolset (below) are in, with Gemma
|
|
61
70
|
reliability hardening verified hands-on.
|
|
62
71
|
|
|
72
|
+
## Quickstart
|
|
73
|
+
|
|
74
|
+
**One command** — installs Drydock, then either serves a strong local model or points
|
|
75
|
+
at a keyed endpoint, and launches you into the agent:
|
|
76
|
+
|
|
77
|
+
```bash
|
|
78
|
+
curl -fsSL https://raw.githubusercontent.com/fbobe321/drydock/main/scripts/quickstart.sh | bash
|
|
79
|
+
```
|
|
80
|
+
|
|
81
|
+
It offers two backends:
|
|
82
|
+
|
|
83
|
+
- **Local (air-gapped)** — downloads **Qwen2.5-Coder-32B** (one time) and serves it with
|
|
84
|
+
`llama.cpp`, then runs entirely offline. Needs a GPU box with `llama-server` on `PATH`.
|
|
85
|
+
- **Frontier (no GPU)** — try it in a minute against a keyed OpenAI-compatible endpoint;
|
|
86
|
+
only that call leaves the box:
|
|
87
|
+
```bash
|
|
88
|
+
OPENAI_API_KEY=sk-... DRYDOCK_QUICKSTART_MODE=frontier \
|
|
89
|
+
curl -fsSL https://raw.githubusercontent.com/fbobe321/drydock/main/scripts/quickstart.sh | bash
|
|
90
|
+
```
|
|
91
|
+
|
|
92
|
+
Prefer to wire it up by hand?
|
|
93
|
+
|
|
94
|
+
```bash
|
|
95
|
+
pip install drydock-cli
|
|
96
|
+
# point at any OpenAI-compatible server (llama.cpp / vLLM / Ollama / LM Studio):
|
|
97
|
+
drydock --provider vllm --base-url http://localhost:8000/v1 --model <served-model-name>
|
|
98
|
+
# or a keyed frontier model (key via OPENAI_API_KEY or config `api_key`):
|
|
99
|
+
OPENAI_API_KEY=sk-... drydock --provider openai --base-url https://api.openai.com/v1 --model gpt-4o
|
|
100
|
+
```
|
|
101
|
+
|
|
63
102
|
## Capabilities
|
|
64
103
|
|
|
65
104
|
A full agentic CLI harness — every tool below is clean-room and dependency-free
|
|
@@ -282,22 +321,6 @@ exports the open findings to a deterministic **eMASS POA&M CSV** — Control (fr
|
|
|
282
321
|
the CCI map), Vulnerability Description, `POA&M Status=Ongoing`, Milestone (the
|
|
283
322
|
Fix Text), and Severity (CAT I/II/III → High/Moderate/Low) — no LLM, stdlib only.
|
|
284
323
|
|
|
285
|
-
### FIAR audit-readiness (DoD financial-statement audit)
|
|
286
|
-
|
|
287
|
-
Model a Financial Improvement and Audit Readiness engagement — a seeded key-control
|
|
288
|
-
matrix per business cycle (FBWT, P2P, PP&E, INV, CIVPAY, REIM, FR, ITGC), the five FS
|
|
289
|
-
assertions, a deterministic **evidence-chain validator** (a control can't be called
|
|
290
|
-
effective on an incomplete population→sample→…→GL→assertion trace), findings (NFRs), and
|
|
291
|
-
CAPs. `python -m drydock.fiar new|controls|control|assess|reconcile|package`; skills
|
|
292
|
-
`/fiar-assess /fiar-evidence /fiar-readiness /fiar-cap`.
|
|
293
|
-
|
|
294
|
-
**KSD evidence packaging.** `fiar package <engagement> <out> --evidence-dir <dir>` assembles
|
|
295
|
-
an audit binder — `index.md` + `index.json` mapping every control → assertions → KSDs →
|
|
296
|
-
status → evidence-chain completeness → findings — and **collects the actual evidence files
|
|
297
|
-
each test cited**, zipped. Add `--redact "<term>"` (repeatable) to **red-box** names/secrets
|
|
298
|
-
in the evidence *and* the manifest for a releasable version (verified via the Document
|
|
299
|
-
Canvas); the original engagement is never touched. Also exposed as the `FiarPackage` tool.
|
|
300
|
-
|
|
301
324
|
### Document Canvas — editing documents far larger than the context window
|
|
302
325
|
|
|
303
326
|
Edit a 300-, 800-, 1000-page document the way you edit a large codebase: **search,
|
|
@@ -1,10 +1,15 @@
|
|
|
1
1
|
# ⚓ Drydock
|
|
2
2
|
|
|
3
|
-
|
|
4
|
-
|
|
5
|
-
|
|
6
|
-
|
|
7
|
-
|
|
3
|
+
**The air-gapped terminal coding agent that automates your NIST compliance** —
|
|
4
|
+
RMF (800-53), CSF 2.0, AI RMF, and DISA STIG — running entirely on your own
|
|
5
|
+
hardware against a local LLM. No accounts, no telemetry, no cloud: your code,
|
|
6
|
+
controls, and CUI never leave the box.
|
|
7
|
+
|
|
8
|
+
Under the compliance skills it's a full agentic coding harness — the same
|
|
9
|
+
air-gapped agent whether you're assessing a control or shipping a feature. The
|
|
10
|
+
only outbound calls are to the model endpoint you configure and (optionally) the
|
|
11
|
+
web-search tools you invoke. Primary target: **dense Gemma-4-31B** (QAT, 64K)
|
|
12
|
+
served by llama.cpp on a single workstation.
|
|
8
13
|
|
|
9
14
|
> **v3 — clean-room rebuild.** Drydock is being rebuilt as an original,
|
|
10
15
|
> Apache-2.0 codebase owned end to end (no upstream fork). Every release is
|
|
@@ -14,10 +19,14 @@ single workstation.
|
|
|
14
19
|
|
|
15
20
|
## Why
|
|
16
21
|
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
|
|
22
|
+
Regulated and air-gapped environments — federal, defense, critical-infrastructure,
|
|
23
|
+
and any org adopting NIST CSF 2.0 — can't send code, controls, or CUI to a cloud
|
|
24
|
+
agent, yet still need real automation. Drydock runs entirely against a local model,
|
|
25
|
+
feels like a first-class terminal agent, ingests the NIST 800-53 catalog and DISA
|
|
26
|
+
STIG benchmarks to assess / remediate / export POA&Ms **offline**, and keeps its
|
|
27
|
+
data plane on your box. It's a governable agent by design: a deterministic control
|
|
28
|
+
loop, a verification gate on "done", and a durable, tamper-evident event trace —
|
|
29
|
+
the audit record for what an autonomous agent did in your environment.
|
|
21
30
|
|
|
22
31
|
## Status
|
|
23
32
|
|
|
@@ -28,6 +37,36 @@ line, and a multi-line prompt. The agent loop, OpenAI-compatible provider,
|
|
|
28
37
|
two-tier compaction, and the full agentic toolset (below) are in, with Gemma
|
|
29
38
|
reliability hardening verified hands-on.
|
|
30
39
|
|
|
40
|
+
## Quickstart
|
|
41
|
+
|
|
42
|
+
**One command** — installs Drydock, then either serves a strong local model or points
|
|
43
|
+
at a keyed endpoint, and launches you into the agent:
|
|
44
|
+
|
|
45
|
+
```bash
|
|
46
|
+
curl -fsSL https://raw.githubusercontent.com/fbobe321/drydock/main/scripts/quickstart.sh | bash
|
|
47
|
+
```
|
|
48
|
+
|
|
49
|
+
It offers two backends:
|
|
50
|
+
|
|
51
|
+
- **Local (air-gapped)** — downloads **Qwen2.5-Coder-32B** (one time) and serves it with
|
|
52
|
+
`llama.cpp`, then runs entirely offline. Needs a GPU box with `llama-server` on `PATH`.
|
|
53
|
+
- **Frontier (no GPU)** — try it in a minute against a keyed OpenAI-compatible endpoint;
|
|
54
|
+
only that call leaves the box:
|
|
55
|
+
```bash
|
|
56
|
+
OPENAI_API_KEY=sk-... DRYDOCK_QUICKSTART_MODE=frontier \
|
|
57
|
+
curl -fsSL https://raw.githubusercontent.com/fbobe321/drydock/main/scripts/quickstart.sh | bash
|
|
58
|
+
```
|
|
59
|
+
|
|
60
|
+
Prefer to wire it up by hand?
|
|
61
|
+
|
|
62
|
+
```bash
|
|
63
|
+
pip install drydock-cli
|
|
64
|
+
# point at any OpenAI-compatible server (llama.cpp / vLLM / Ollama / LM Studio):
|
|
65
|
+
drydock --provider vllm --base-url http://localhost:8000/v1 --model <served-model-name>
|
|
66
|
+
# or a keyed frontier model (key via OPENAI_API_KEY or config `api_key`):
|
|
67
|
+
OPENAI_API_KEY=sk-... drydock --provider openai --base-url https://api.openai.com/v1 --model gpt-4o
|
|
68
|
+
```
|
|
69
|
+
|
|
31
70
|
## Capabilities
|
|
32
71
|
|
|
33
72
|
A full agentic CLI harness — every tool below is clean-room and dependency-free
|
|
@@ -250,22 +289,6 @@ exports the open findings to a deterministic **eMASS POA&M CSV** — Control (fr
|
|
|
250
289
|
the CCI map), Vulnerability Description, `POA&M Status=Ongoing`, Milestone (the
|
|
251
290
|
Fix Text), and Severity (CAT I/II/III → High/Moderate/Low) — no LLM, stdlib only.
|
|
252
291
|
|
|
253
|
-
### FIAR audit-readiness (DoD financial-statement audit)
|
|
254
|
-
|
|
255
|
-
Model a Financial Improvement and Audit Readiness engagement — a seeded key-control
|
|
256
|
-
matrix per business cycle (FBWT, P2P, PP&E, INV, CIVPAY, REIM, FR, ITGC), the five FS
|
|
257
|
-
assertions, a deterministic **evidence-chain validator** (a control can't be called
|
|
258
|
-
effective on an incomplete population→sample→…→GL→assertion trace), findings (NFRs), and
|
|
259
|
-
CAPs. `python -m drydock.fiar new|controls|control|assess|reconcile|package`; skills
|
|
260
|
-
`/fiar-assess /fiar-evidence /fiar-readiness /fiar-cap`.
|
|
261
|
-
|
|
262
|
-
**KSD evidence packaging.** `fiar package <engagement> <out> --evidence-dir <dir>` assembles
|
|
263
|
-
an audit binder — `index.md` + `index.json` mapping every control → assertions → KSDs →
|
|
264
|
-
status → evidence-chain completeness → findings — and **collects the actual evidence files
|
|
265
|
-
each test cited**, zipped. Add `--redact "<term>"` (repeatable) to **red-box** names/secrets
|
|
266
|
-
in the evidence *and* the manifest for a releasable version (verified via the Document
|
|
267
|
-
Canvas); the original engagement is never touched. Also exposed as the `FiarPackage` tool.
|
|
268
|
-
|
|
269
292
|
### Document Canvas — editing documents far larger than the context window
|
|
270
293
|
|
|
271
294
|
Edit a 300-, 800-, 1000-page document the way you edit a large codebase: **search,
|
|
@@ -47,6 +47,7 @@ from drydock.tools import register_all
|
|
|
47
47
|
register_all()
|
|
48
48
|
from drydock.compaction import (
|
|
49
49
|
maybe_compact, emergency_compact, is_context_length_error, is_image_load_error,
|
|
50
|
+
is_text_only_model_error,
|
|
50
51
|
extract_server_n_ctx,
|
|
51
52
|
)
|
|
52
53
|
from drydock.loop_detect import LoopTracker, degenerate_argument
|
|
@@ -409,6 +410,17 @@ def run(
|
|
|
409
410
|
if retries >= 2:
|
|
410
411
|
raise
|
|
411
412
|
yield TextChunk("\n[context limit hit — compacting and retrying...]\n")
|
|
413
|
+
elif is_text_only_model_error(err):
|
|
414
|
+
# The model can't take images at all. Stop attaching them for this
|
|
415
|
+
# endpoint and retry the step as text — a vision-less model can
|
|
416
|
+
# still do the task (read the file with tools), it just can't SEE it.
|
|
417
|
+
from drydock.providers import mark_text_only
|
|
418
|
+
if not mark_text_only(config):
|
|
419
|
+
raise
|
|
420
|
+
yield TextChunk(
|
|
421
|
+
"\n[This model is text-only — sending image references as plain "
|
|
422
|
+
"text from now on and retrying...]\n"
|
|
423
|
+
)
|
|
412
424
|
elif is_image_load_error(err):
|
|
413
425
|
# Corrupt/truncated/unsupported image the server couldn't decode —
|
|
414
426
|
# end the turn cleanly rather than dumping the raw 400.
|
|
@@ -149,6 +149,42 @@ def run_oneshot(prompt: str, config: dict) -> None:
|
|
|
149
149
|
sys.exit(2)
|
|
150
150
|
|
|
151
151
|
|
|
152
|
+
def run_oneshot_swarm(prompt: str, config: dict) -> None:
|
|
153
|
+
"""One-shot via a competitive SWARM (config `auto_swarm` / --swarm): N agents attempt the
|
|
154
|
+
task in parallel in isolated worktrees, the verified winner is applied to the working tree.
|
|
155
|
+
Falls back to a single agent if the swarm can't run or converge."""
|
|
156
|
+
import subprocess
|
|
157
|
+
from drydock.swarm import run_swarm
|
|
158
|
+
cwd = config.get("cwd") or os.getcwd()
|
|
159
|
+
try:
|
|
160
|
+
agents = max(2, min(8, int(config.get("auto_swarm_agents", 4) or 4)))
|
|
161
|
+
except (TypeError, ValueError):
|
|
162
|
+
agents = 4
|
|
163
|
+
print(f"⚓ auto-swarm: {agents} parallel agents on: {prompt}", file=sys.stderr, flush=True)
|
|
164
|
+
try:
|
|
165
|
+
res = run_swarm(cwd, prompt, agents=agents, base_config=config, share=True, waves=2)
|
|
166
|
+
except Exception as e: # noqa: BLE001
|
|
167
|
+
print(f"(swarm couldn't run: {e} — falling back to single agent)", file=sys.stderr, flush=True)
|
|
168
|
+
return run_oneshot(prompt, config)
|
|
169
|
+
if not res.converged or res.winner is None:
|
|
170
|
+
print(f"(swarm ran {agents} agents, {len(res.candidates)} candidate(s), none converged "
|
|
171
|
+
f"— falling back to single agent)", file=sys.stderr, flush=True)
|
|
172
|
+
return run_oneshot(prompt, config)
|
|
173
|
+
w = res.winner
|
|
174
|
+
try:
|
|
175
|
+
r = subprocess.run(["git", "checkout", w.commit, "--", "."], cwd=cwd,
|
|
176
|
+
capture_output=True, text=True)
|
|
177
|
+
if r.returncode != 0:
|
|
178
|
+
print(f"Swarm winner {w.commit[:12]} not auto-applied ({r.stderr.strip()}). "
|
|
179
|
+
f"Apply: git checkout {w.commit} -- .", file=sys.stderr, flush=True)
|
|
180
|
+
return
|
|
181
|
+
except Exception as e: # noqa: BLE001
|
|
182
|
+
print(f"Swarm converged but apply errored: {e}", file=sys.stderr, flush=True)
|
|
183
|
+
return
|
|
184
|
+
print(f"✓ swarm converged: winner by {w.agent}, {w.tests_passed}/{w.tests_total} tests, "
|
|
185
|
+
f"{w.files_changed} file(s) changed, applied {w.commit[:12]}.", flush=True)
|
|
186
|
+
|
|
187
|
+
|
|
152
188
|
def handle_command(cmd: str, state: AgentState, config: dict) -> bool:
|
|
153
189
|
"""Handle slash commands. Returns True if handled."""
|
|
154
190
|
raw = cmd.strip() # case-preserved (paths/args must not be lowercased)
|
|
@@ -414,6 +450,10 @@ def main():
|
|
|
414
450
|
help="Write the full run transcript (system + messages) here — for harvesting training data")
|
|
415
451
|
parser.add_argument("--force-first-tool", action="store_true", help="Force tool_choice=required on first turn")
|
|
416
452
|
parser.add_argument("--cli", action="store_true", help="Plain readline mode instead of the TUI")
|
|
453
|
+
parser.add_argument("--swarm", action="store_true", dest="swarm",
|
|
454
|
+
help="One-shot: solve the task with a competitive parallel swarm (winner applied)")
|
|
455
|
+
parser.add_argument("--no-swarm", action="store_true", dest="no_swarm",
|
|
456
|
+
help="One-shot: force a single agent even if auto_swarm is on")
|
|
417
457
|
parser.add_argument(
|
|
418
458
|
"--dangerously-skip-permissions",
|
|
419
459
|
dest="dangerously_skip_permissions",
|
|
@@ -467,6 +507,15 @@ def main():
|
|
|
467
507
|
# the PRD" content fought the system prompt and risked turning a plain
|
|
468
508
|
# greeting into a runaway build. The TUI reads this from config; CLI modes
|
|
469
509
|
# call load_project_instructions directly (see run_interactive/run_oneshot).
|
|
510
|
+
# A context_limit above the server's real window overflows it mid-task and the
|
|
511
|
+
# ctx gauge lies. Unless the user pinned --context-limit, ask the server once and
|
|
512
|
+
# adopt its window when smaller (never grow past config). Best-effort, short timeout.
|
|
513
|
+
if not args.context_limit and config.get("base_url"):
|
|
514
|
+
from drydock.providers import probe_server_context
|
|
515
|
+
n_ctx = probe_server_context(config["base_url"], timeout=2.0)
|
|
516
|
+
if n_ctx and n_ctx < int(config.get("context_limit") or 65536):
|
|
517
|
+
config["context_limit"] = n_ctx
|
|
518
|
+
|
|
470
519
|
config["project_instructions"] = load_project_instructions()
|
|
471
520
|
# In a git project, drop a discoverable per-project prompt template at
|
|
472
521
|
# <project>/.drydock/system_prompt.md (inert until edited). Then load the
|
|
@@ -481,7 +530,19 @@ def main():
|
|
|
481
530
|
_connect_mcp(config)
|
|
482
531
|
|
|
483
532
|
if args.prompt:
|
|
484
|
-
|
|
533
|
+
# Route to the swarm when forced (--swarm) or when auto_swarm is on AND the task
|
|
534
|
+
# looks substantial; --no-swarm and the triviality gate keep single-agent the default.
|
|
535
|
+
from drydock.swarm import looks_substantial
|
|
536
|
+
if getattr(args, "no_swarm", False):
|
|
537
|
+
use_swarm = False
|
|
538
|
+
elif getattr(args, "swarm", False):
|
|
539
|
+
use_swarm = True
|
|
540
|
+
else:
|
|
541
|
+
use_swarm = bool(config.get("auto_swarm")) and looks_substantial(args.prompt)
|
|
542
|
+
if use_swarm:
|
|
543
|
+
run_oneshot_swarm(args.prompt, config)
|
|
544
|
+
else:
|
|
545
|
+
run_oneshot(args.prompt, config)
|
|
485
546
|
elif args.cli:
|
|
486
547
|
run_interactive(config)
|
|
487
548
|
else:
|
|
@@ -51,6 +51,19 @@ def extract_server_n_ctx(err: str) -> int | None:
|
|
|
51
51
|
return None
|
|
52
52
|
|
|
53
53
|
|
|
54
|
+
def is_text_only_model_error(err: str) -> bool:
|
|
55
|
+
"""Whether a provider 400 says the MODEL can't take images at all (a text-only
|
|
56
|
+
model, e.g. vLLM "X is not a multimodal model", llama.cpp "image input is not
|
|
57
|
+
supported" when no --mmproj is loaded). Distinct from a bad image: the fix is to
|
|
58
|
+
drop attachments and retry, not to ask the user for a different file."""
|
|
59
|
+
e = (err or "").lower()
|
|
60
|
+
return (
|
|
61
|
+
"not a multimodal model" in e
|
|
62
|
+
or "image input is not supported" in e
|
|
63
|
+
or ("does not support" in e and ("image" in e or "multimodal" in e or "vision" in e))
|
|
64
|
+
)
|
|
65
|
+
|
|
66
|
+
|
|
54
67
|
def is_image_load_error(err: str) -> bool:
|
|
55
68
|
"""Whether a provider 400 is the server failing to decode an attached image
|
|
56
69
|
(corrupt / truncated / unsupported), so the agent ends the turn with a clean
|
|
@@ -33,6 +33,12 @@ DEFAULTS: dict[str, object] = {
|
|
|
33
33
|
# OpenRouter, Together). Mirrors `advisor_api_key`. Can also come from the
|
|
34
34
|
# provider's api_key_env (e.g. OPENAI_API_KEY); the config value wins.
|
|
35
35
|
"api_key": "",
|
|
36
|
+
# Auto-swarm (opt-in): when true, one-shot `-p` tasks that look like a substantial,
|
|
37
|
+
# self-contained implementation (swarm.looks_substantial) are solved by a competitive
|
|
38
|
+
# swarm of `auto_swarm_agents` parallel agents (winner applied), instead of one agent.
|
|
39
|
+
# Trivial edits/questions/reads always stay single-agent. Off by default.
|
|
40
|
+
"auto_swarm": False,
|
|
41
|
+
"auto_swarm_agents": 4,
|
|
36
42
|
"max_tokens": 8192, # 4096 truncated large file writes mid-JSON (→ _raw fail)
|
|
37
43
|
"temperature": 0.2,
|
|
38
44
|
# The model server's context window (llama.cpp -c / vLLM --max-model-len).
|
|
@@ -39,7 +39,9 @@ class RepetitionDetected(StallRetry):
|
|
|
39
39
|
re-issue would just loop again, so the agent goes straight to decisive mode."""
|
|
40
40
|
|
|
41
41
|
from drydock.loop_detect import runaway_repetition_len
|
|
42
|
-
from drydock.tuning import
|
|
42
|
+
from drydock.tuning import (
|
|
43
|
+
extract_thinking, split_think_tags, strip_leaked_tool_calls, strip_thinking_tokens, use_streaming,
|
|
44
|
+
)
|
|
43
45
|
|
|
44
46
|
# ── Provider registry ─────────────────────────────────────────────────────
|
|
45
47
|
|
|
@@ -430,6 +432,31 @@ def _image_url_block(path: str) -> dict | None:
|
|
|
430
432
|
return {"type": "image_url", "image_url": {"url": f"data:{mime};base64,{b64}"}}
|
|
431
433
|
|
|
432
434
|
|
|
435
|
+
# (base_url, model) pairs whose server rejected image input as text-only. Process-wide
|
|
436
|
+
# so every later request (and turn) sends plain text instead of re-hitting the 400.
|
|
437
|
+
_TEXT_ONLY: set[tuple[str, str]] = set()
|
|
438
|
+
|
|
439
|
+
|
|
440
|
+
def _vision_key(config: dict) -> tuple[str, str]:
|
|
441
|
+
prov = PROVIDERS.get(config.get("provider", "vllm"), PROVIDERS["vllm"])
|
|
442
|
+
base_url = config.get("base_url") or prov.get("base_url", "http://localhost:8000/v1")
|
|
443
|
+
return (str(base_url).rstrip("/"), str(config.get("model") or ""))
|
|
444
|
+
|
|
445
|
+
|
|
446
|
+
def mark_text_only(config: dict) -> bool:
|
|
447
|
+
"""Record that this endpoint's model can't take images. Returns True if newly marked."""
|
|
448
|
+
key = _vision_key(config)
|
|
449
|
+
if key in _TEXT_ONLY:
|
|
450
|
+
return False
|
|
451
|
+
_TEXT_ONLY.add(key)
|
|
452
|
+
return True
|
|
453
|
+
|
|
454
|
+
|
|
455
|
+
def vision_enabled(config: dict) -> bool:
|
|
456
|
+
"""False when vision is disabled in config or the server proved text-only."""
|
|
457
|
+
return config.get("vision", True) is not False and _vision_key(config) not in _TEXT_ONLY
|
|
458
|
+
|
|
459
|
+
|
|
433
460
|
def _content_with_images(content):
|
|
434
461
|
"""Turn text that references on-disk image paths into a multimodal content
|
|
435
462
|
list ([text, image_url…]); text without resolvable images passes through
|
|
@@ -451,15 +478,17 @@ def _content_with_images(content):
|
|
|
451
478
|
_user_content_with_images = _content_with_images
|
|
452
479
|
|
|
453
480
|
|
|
454
|
-
def messages_to_openai(messages: list, system: str) -> list:
|
|
455
|
-
"""Convert neutral messages to OpenAI API format.
|
|
481
|
+
def messages_to_openai(messages: list, system: str, vision: bool = True) -> list:
|
|
482
|
+
"""Convert neutral messages to OpenAI API format. vision=False sends image
|
|
483
|
+
references as plain text (text-only model)."""
|
|
456
484
|
# value type is mixed (str content, multimodal list content, tool_calls list)
|
|
457
485
|
result: list[dict] = [{"role": "system", "content": system}]
|
|
458
486
|
for m in messages:
|
|
459
487
|
role = m["role"]
|
|
460
488
|
if role == "user":
|
|
461
489
|
result.append({"role": "user",
|
|
462
|
-
"content": _user_content_with_images(m["content"])
|
|
490
|
+
"content": (_user_content_with_images(m["content"])
|
|
491
|
+
if vision else m["content"])})
|
|
463
492
|
elif role == "assistant":
|
|
464
493
|
msg = {"role": "assistant", "content": m.get("content") or None}
|
|
465
494
|
tcs = m.get("tool_calls", [])
|
|
@@ -482,7 +511,7 @@ def messages_to_openai(messages: list, system: str) -> list:
|
|
|
482
511
|
# to look at — attach it so the model SEES it (servers read images
|
|
483
512
|
# from tool messages too, verified). Only ViewImage, so an unrelated
|
|
484
513
|
# tool mentioning a .png path never balloons the request.
|
|
485
|
-
if m.get("name") == "ViewImage" and isinstance(content, str):
|
|
514
|
+
if vision and m.get("name") == "ViewImage" and isinstance(content, str):
|
|
486
515
|
content = _content_with_images(content)
|
|
487
516
|
result.append({
|
|
488
517
|
"role": "tool",
|
|
@@ -559,7 +588,7 @@ def stream(
|
|
|
559
588
|
# dict(config) shallow-copy) so the TUI's STOP can reach this exact client.
|
|
560
589
|
config.setdefault("_abort", {})["client"] = client
|
|
561
590
|
|
|
562
|
-
oai_messages = messages_to_openai(messages, system)
|
|
591
|
+
oai_messages = messages_to_openai(messages, system, vision=vision_enabled(config))
|
|
563
592
|
|
|
564
593
|
has_tools = bool(tool_schemas)
|
|
565
594
|
# Gemma (and any local server whose model name we can't trust) corrupts
|
|
@@ -647,6 +676,9 @@ def stream(
|
|
|
647
676
|
if delta.content:
|
|
648
677
|
# Strip any leaked thinking-token markers (best-effort per chunk).
|
|
649
678
|
chunk_text = strip_thinking_tokens(delta.content)
|
|
679
|
+
# <think>/</think> tags usually arrive as whole tokens; keep them in the
|
|
680
|
+
# buffer (split off at the end) but don't display the bare markers.
|
|
681
|
+
shown = chunk_text.replace("<think>", "").replace("</think>", "")
|
|
650
682
|
# Keep ANY non-empty chunk — including whitespace-only ones. A
|
|
651
683
|
# `.strip()` guard here used to DROP newline-only chunks (the blank
|
|
652
684
|
# line a model streams between markdown blocks arrives as its own
|
|
@@ -654,7 +686,8 @@ def stream(
|
|
|
654
686
|
# skips chunks that stripping emptied (e.g. a lone thinking marker).
|
|
655
687
|
if chunk_text:
|
|
656
688
|
text += chunk_text
|
|
657
|
-
|
|
689
|
+
if shown:
|
|
690
|
+
yield TextChunk(shown)
|
|
658
691
|
# Throttled: only scan once text is long enough and every ~200
|
|
659
692
|
# new chars, so the common path pays almost nothing.
|
|
660
693
|
if len(text) - _rep_checked_at >= 200:
|
|
@@ -705,6 +738,9 @@ def stream(
|
|
|
705
738
|
"input": inp,
|
|
706
739
|
})
|
|
707
740
|
|
|
741
|
+
# A template-opened <think> leaves "reasoning…</think>answer" in the content —
|
|
742
|
+
# keep only the answer in the recorded turn (it was already shown while streaming).
|
|
743
|
+
_, text = split_think_tags(text)
|
|
708
744
|
yield AssistantTurn(text, tool_calls, in_tok, out_tok)
|
|
709
745
|
|
|
710
746
|
|
|
@@ -734,6 +734,29 @@ class SwarmResult:
|
|
|
734
734
|
candidates: list[Candidate]
|
|
735
735
|
|
|
736
736
|
|
|
737
|
+
# Auto-swarm triviality gate: route a task to the swarm only when it looks like a
|
|
738
|
+
# substantial, self-contained implementation — never a one-line fix, question, or read.
|
|
739
|
+
_SUBSTANTIAL = ("implement", "build the", "build a", "refactor", "rewrite", "create the",
|
|
740
|
+
"all functions", "each function", "whole module", "entire", "feature",
|
|
741
|
+
"make the tests", "tests pass", "test suite", "port ", "migrate",
|
|
742
|
+
"add support", "write the", "flesh out", "scaffold", "from scratch")
|
|
743
|
+
_TRIVIAL = ("typo", "rename", "one line", "one-line", "single line", "the comment",
|
|
744
|
+
"what is", "what's", "explain", "why ", "show me", "read ", "print the",
|
|
745
|
+
"list the", "how do", "how does", "?")
|
|
746
|
+
|
|
747
|
+
|
|
748
|
+
def looks_substantial(task: str) -> bool:
|
|
749
|
+
"""Heuristic gate for auto-swarm (config `auto_swarm`): True only when the task is a
|
|
750
|
+
substantial, self-contained implementation worth parallel attempts — not a trivial edit,
|
|
751
|
+
a question, or a read. Conservative: unsure → False (single agent stays the default)."""
|
|
752
|
+
t = (task or "").lower().strip()
|
|
753
|
+
if len(t.split()) < 4:
|
|
754
|
+
return False
|
|
755
|
+
if any(x in t for x in _TRIVIAL):
|
|
756
|
+
return False
|
|
757
|
+
return any(x in t for x in _SUBSTANTIAL)
|
|
758
|
+
|
|
759
|
+
|
|
737
760
|
def run_swarm(cwd: str | Path, objective: str, *, agents: "int | str" = 4,
|
|
738
761
|
base_config: dict | None = None,
|
|
739
762
|
base_ref: str = "HEAD", verify_cmd: str | None = None, fitness: str = "auto",
|
|
@@ -44,6 +44,13 @@ _FAMILIES: list[tuple[frozenset[str], tuple[str, ...]]] = [
|
|
|
44
44
|
(frozenset({"Jobs"}),
|
|
45
45
|
("job", "background", "training", "still running", "long-running", "long running",
|
|
46
46
|
"status of", "check on", "how is the", "how's the", "in the background")),
|
|
47
|
+
# Multi-agent delegation/parallelism: surface Swarm (competitive parallel agents),
|
|
48
|
+
# Worker (delegate a subtask), and Dispatch (parallel investigation) when the task is
|
|
49
|
+
# a substantial implementation/refactor where parallel attempts or delegation help.
|
|
50
|
+
(frozenset({"Swarm", "Worker", "Dispatch"}),
|
|
51
|
+
("implement", "module", "feature", "refactor", "rewrite", "build the", "all functions",
|
|
52
|
+
"each function", "tests pass", "make the tests", "parallel", "swarm", "several",
|
|
53
|
+
"whole", "entire", "port ", "migrate")),
|
|
47
54
|
]
|
|
48
55
|
|
|
49
56
|
# External/mutating tools deprioritised while merely exploring (PRD F1.4: read &
|