drydock-cli 3.1.35__tar.gz → 3.1.37__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (226) hide show
  1. {drydock_cli-3.1.35/drydock_cli.egg-info → drydock_cli-3.1.37}/PKG-INFO +49 -26
  2. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/README.md +48 -25
  3. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/agent.py +12 -0
  4. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/cli.py +62 -1
  5. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/compaction.py +13 -0
  6. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/config.py +6 -0
  7. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/providers.py +43 -7
  8. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/swarm.py +23 -0
  9. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/tool_select.py +7 -0
  10. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/tools/__init__.py +76 -324
  11. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/tui/app.py +1 -1
  12. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/tuning.py +21 -3
  13. {drydock_cli-3.1.35 → drydock_cli-3.1.37/drydock_cli.egg-info}/PKG-INFO +49 -26
  14. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock_cli.egg-info/SOURCES.txt +1 -6
  15. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/pyproject.toml +1 -1
  16. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_skills.py +0 -1
  17. drydock_cli-3.1.37/tests/test_text_only_model.py +78 -0
  18. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_tuning.py +11 -0
  19. drydock_cli-3.1.35/drydock/builtin_skills/fiar-assess.md +0 -33
  20. drydock_cli-3.1.35/drydock/builtin_skills/fiar-cap.md +0 -26
  21. drydock_cli-3.1.35/drydock/builtin_skills/fiar-evidence.md +0 -33
  22. drydock_cli-3.1.35/drydock/builtin_skills/fiar-readiness.md +0 -28
  23. drydock_cli-3.1.35/drydock/fiar.py +0 -680
  24. drydock_cli-3.1.35/tests/test_fiar.py +0 -268
  25. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/LICENSE +0 -0
  26. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/NOTICE +0 -0
  27. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/__init__.py +0 -0
  28. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/__main__.py +0 -0
  29. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/advisor.py +0 -0
  30. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/asr.py +0 -0
  31. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/bash_safety.py +0 -0
  32. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/bottleneck.py +0 -0
  33. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/budget.py +0 -0
  34. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/builtin_skills/__init__.py +0 -0
  35. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/builtin_skills/document-canvas.md +0 -0
  36. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/builtin_skills/ml-data.md +0 -0
  37. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/builtin_skills/ml-debug.md +0 -0
  38. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/builtin_skills/ml-finetune.md +0 -0
  39. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/builtin_skills/ml-metrics.md +0 -0
  40. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/builtin_skills/ml-rl.md +0 -0
  41. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/builtin_skills/ml-train.md +0 -0
  42. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/builtin_skills/nist-ai-rmf.md +0 -0
  43. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/builtin_skills/nist-csf.md +0 -0
  44. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/builtin_skills/rmf-categorize.md +0 -0
  45. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/builtin_skills/rmf-control.md +0 -0
  46. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/builtin_skills/rmf-poam.md +0 -0
  47. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/builtin_skills/rmf-review.md +0 -0
  48. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/builtin_skills/stig-assess.md +0 -0
  49. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/builtin_skills/stig-remediate.md +0 -0
  50. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/capacity.py +0 -0
  51. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/cci.py +0 -0
  52. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/comms/__init__.py +0 -0
  53. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/comms/attention.py +0 -0
  54. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/comms/channels.py +0 -0
  55. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/comms/events.py +0 -0
  56. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/detect.py +0 -0
  57. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/doccanvas.py +0 -0
  58. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/eratchet.py +0 -0
  59. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/events.py +0 -0
  60. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/extract.py +0 -0
  61. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/gittools.py +0 -0
  62. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/graphrag.py +0 -0
  63. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/groundtruth.py +0 -0
  64. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/guards.py +0 -0
  65. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/jobs.py +0 -0
  66. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/loop_detect.py +0 -0
  67. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/mcp.py +0 -0
  68. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/mission.py +0 -0
  69. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/mission_cli.py +0 -0
  70. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/mission_run.py +0 -0
  71. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/pdfredbox.py +0 -0
  72. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/phases.py +0 -0
  73. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/poam.py +0 -0
  74. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/predictions.py +0 -0
  75. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/progress.py +0 -0
  76. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/ratchet.py +0 -0
  77. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/recipes.py +0 -0
  78. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/recovery.py +0 -0
  79. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/resume.py +0 -0
  80. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/rmf.py +0 -0
  81. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/rmf_graph.py +0 -0
  82. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/skills.py +0 -0
  83. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/stig.py +0 -0
  84. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/subagents.py +0 -0
  85. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/suggest.py +0 -0
  86. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/task_state.py +0 -0
  87. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/tool_policy.py +0 -0
  88. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/tool_registry.py +0 -0
  89. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/tool_result.py +0 -0
  90. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/tool_validate.py +0 -0
  91. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/trajectory.py +0 -0
  92. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/tui/__init__.py +0 -0
  93. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/tui/approval.py +0 -0
  94. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/tui/messages.py +0 -0
  95. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/tui/widgets.py +0 -0
  96. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/verification.py +0 -0
  97. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock/web.py +0 -0
  98. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock_cli.egg-info/dependency_links.txt +0 -0
  99. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock_cli.egg-info/entry_points.txt +0 -0
  100. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock_cli.egg-info/requires.txt +0 -0
  101. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/drydock_cli.egg-info/top_level.txt +0 -0
  102. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/setup.cfg +0 -0
  103. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_advisor.py +0 -0
  104. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_approval.py +0 -0
  105. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_asr.py +0 -0
  106. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_back_command.py +0 -0
  107. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_bash_background.py +0 -0
  108. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_bash_binary_output.py +0 -0
  109. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_bash_crossplatform.py +0 -0
  110. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_bash_output_bounding.py +0 -0
  111. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_bash_process_group.py +0 -0
  112. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_bash_safety.py +0 -0
  113. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_bash_sanitize.py +0 -0
  114. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_bash_shell.py +0 -0
  115. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_bash_stdin.py +0 -0
  116. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_bash_stop_partial.py +0 -0
  117. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_bash_timeout_network.py +0 -0
  118. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_bash_timeout_param.py +0 -0
  119. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_bottleneck.py +0 -0
  120. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_budget.py +0 -0
  121. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_capacity.py +0 -0
  122. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_cci.py +0 -0
  123. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_cli_agents.py +0 -0
  124. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_comms.py +0 -0
  125. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_compact_command.py +0 -0
  126. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_compaction.py +0 -0
  127. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_config.py +0 -0
  128. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_config_migration.py +0 -0
  129. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_context_limit_config.py +0 -0
  130. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_context_limit_issue25.py +0 -0
  131. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_criteria_coverage.py +0 -0
  132. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_cycling_detection.py +0 -0
  133. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_declared_deps.py +0 -0
  134. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_degenerate_argument.py +0 -0
  135. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_detect.py +0 -0
  136. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_dispatch.py +0 -0
  137. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_doccanvas.py +0 -0
  138. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_e2e_connected.py +0 -0
  139. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_edit_replace_all.py +0 -0
  140. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_edit_thrash_and_compaction.py +0 -0
  141. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_effort_governor.py +0 -0
  142. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_empty_response.py +0 -0
  143. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_eratchet.py +0 -0
  144. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_events.py +0 -0
  145. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_extract.py +0 -0
  146. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_failure_loop.py +0 -0
  147. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_first_run_setup.py +0 -0
  148. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_gittools.py +0 -0
  149. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_glob_edit_edges.py +0 -0
  150. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_graphify_example.py +0 -0
  151. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_graphrag.py +0 -0
  152. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_graphrag_quoted_path.py +0 -0
  153. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_graphrag_sqlite.py +0 -0
  154. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_grep_and_read_robust.py +0 -0
  155. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_groundtruth_predictions.py +0 -0
  156. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_guards_and_tools.py +0 -0
  157. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_hallucinated_tools.py +0 -0
  158. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_jobs.py +0 -0
  159. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_knowledge_tool_available.py +0 -0
  160. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_leaked_tool_call.py +0 -0
  161. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_loop_detect.py +0 -0
  162. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_main_api_key.py +0 -0
  163. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_mcp.py +0 -0
  164. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_mission.py +0 -0
  165. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_mission_cli.py +0 -0
  166. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_mission_run.py +0 -0
  167. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_model_registry.py +0 -0
  168. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_oneshot_unreachable.py +0 -0
  169. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_overthink_interrupt.py +0 -0
  170. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_pdfredbox.py +0 -0
  171. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_phases.py +0 -0
  172. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_plan_autocontinue.py +0 -0
  173. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_poam.py +0 -0
  174. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_progress.py +0 -0
  175. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_providers_unreachable.py +0 -0
  176. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_ratchet.py +0 -0
  177. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_ratchet_offer.py +0 -0
  178. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_read_index.py +0 -0
  179. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_recipes.py +0 -0
  180. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_recovery.py +0 -0
  181. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_recovery_config.py +0 -0
  182. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_recovery_integration.py +0 -0
  183. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_repeated_outcome.py +0 -0
  184. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_repetition_interrupt.py +0 -0
  185. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_resume_events.py +0 -0
  186. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_resume_restore.py +0 -0
  187. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_resume_snapshot.py +0 -0
  188. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_rmf.py +0 -0
  189. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_rmf_graph.py +0 -0
  190. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_rmf_stig_graph.py +0 -0
  191. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_rolling_plan.py +0 -0
  192. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_runaway_repetition.py +0 -0
  193. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_screenshot.py +0 -0
  194. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_server_probe.py +0 -0
  195. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_sqlite_events.py +0 -0
  196. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_stall_retry.py +0 -0
  197. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_stig.py +0 -0
  198. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_stop.py +0 -0
  199. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_streaming_newlines.py +0 -0
  200. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_subagent.py +0 -0
  201. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_subagents.py +0 -0
  202. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_suggest.py +0 -0
  203. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_swarm.py +0 -0
  204. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_system_prompt_help.py +0 -0
  205. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_task_state.py +0 -0
  206. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_timeline.py +0 -0
  207. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_todo.py +0 -0
  208. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_tool_arg_coercion.py +0 -0
  209. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_tool_arg_coercion_more.py +0 -0
  210. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_tool_arg_parsing.py +0 -0
  211. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_tool_canonical.py +0 -0
  212. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_tool_policy.py +0 -0
  213. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_tool_result.py +0 -0
  214. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_tool_select.py +0 -0
  215. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_tool_validate.py +0 -0
  216. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_tools_undo.py +0 -0
  217. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_trajectory.py +0 -0
  218. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_tui.py +0 -0
  219. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_verification_gate.py +0 -0
  220. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_viewimage.py +0 -0
  221. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_vision_input.py +0 -0
  222. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_web_tools.py +0 -0
  223. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_windows_shell.py +0 -0
  224. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_worker_subagent.py +0 -0
  225. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_write_content_coerce.py +0 -0
  226. {drydock_cli-3.1.35 → drydock_cli-3.1.37}/tests/test_xccdf.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: drydock-cli
3
- Version: 3.1.35
3
+ Version: 3.1.37
4
4
  Summary: Drydock — a local, provider-agnostic terminal coding agent for local LLMs
5
5
  Author: Frank Bobe III
6
6
  License-Expression: Apache-2.0
@@ -32,11 +32,16 @@ Dynamic: license-file
32
32
 
33
33
  # ⚓ Drydock
34
34
 
35
- A local-first, provider-agnostic **terminal coding agent** for your own LLM.
36
- No accounts, no telemetry, no cloud — the only outbound calls are to the model
37
- endpoint you configure and (optionally) the web-search tools you invoke.
38
- Primary target: **dense Gemma-4-31B** (QAT, 64K) served by llama.cpp on a
39
- single workstation.
35
+ **The air-gapped terminal coding agent that automates your NIST compliance** —
36
+ RMF (800-53), CSF 2.0, AI RMF, and DISA STIG — running entirely on your own
37
+ hardware against a local LLM. No accounts, no telemetry, no cloud: your code,
38
+ controls, and CUI never leave the box.
39
+
40
+ Under the compliance skills it's a full agentic coding harness — the same
41
+ air-gapped agent whether you're assessing a control or shipping a feature. The
42
+ only outbound calls are to the model endpoint you configure and (optionally) the
43
+ web-search tools you invoke. Primary target: **dense Gemma-4-31B** (QAT, 64K)
44
+ served by llama.cpp on a single workstation.
40
45
 
41
46
  > **v3 — clean-room rebuild.** Drydock is being rebuilt as an original,
42
47
  > Apache-2.0 codebase owned end to end (no upstream fork). Every release is
@@ -46,10 +51,14 @@ single workstation.
46
51
 
47
52
  ## Why
48
53
 
49
- A coding agent should build real projects from your machine without sending
50
- your code or credentials anywhere. Drydock runs entirely against a local
51
- model, feels like a first-class terminal agent, and keeps its data plane on
52
- your box.
54
+ Regulated and air-gapped environments — federal, defense, critical-infrastructure,
55
+ and any org adopting NIST CSF 2.0 — can't send code, controls, or CUI to a cloud
56
+ agent, yet still need real automation. Drydock runs entirely against a local model,
57
+ feels like a first-class terminal agent, ingests the NIST 800-53 catalog and DISA
58
+ STIG benchmarks to assess / remediate / export POA&Ms **offline**, and keeps its
59
+ data plane on your box. It's a governable agent by design: a deterministic control
60
+ loop, a verification gate on "done", and a durable, tamper-evident event trace —
61
+ the audit record for what an autonomous agent did in your environment.
53
62
 
54
63
  ## Status
55
64
 
@@ -60,6 +69,36 @@ line, and a multi-line prompt. The agent loop, OpenAI-compatible provider,
60
69
  two-tier compaction, and the full agentic toolset (below) are in, with Gemma
61
70
  reliability hardening verified hands-on.
62
71
 
72
+ ## Quickstart
73
+
74
+ **One command** — installs Drydock, then either serves a strong local model or points
75
+ at a keyed endpoint, and launches you into the agent:
76
+
77
+ ```bash
78
+ curl -fsSL https://raw.githubusercontent.com/fbobe321/drydock/main/scripts/quickstart.sh | bash
79
+ ```
80
+
81
+ It offers two backends:
82
+
83
+ - **Local (air-gapped)** — downloads **Qwen2.5-Coder-32B** (one time) and serves it with
84
+ `llama.cpp`, then runs entirely offline. Needs a GPU box with `llama-server` on `PATH`.
85
+ - **Frontier (no GPU)** — try it in a minute against a keyed OpenAI-compatible endpoint;
86
+ only that call leaves the box:
87
+ ```bash
88
+ OPENAI_API_KEY=sk-... DRYDOCK_QUICKSTART_MODE=frontier \
89
+ curl -fsSL https://raw.githubusercontent.com/fbobe321/drydock/main/scripts/quickstart.sh | bash
90
+ ```
91
+
92
+ Prefer to wire it up by hand?
93
+
94
+ ```bash
95
+ pip install drydock-cli
96
+ # point at any OpenAI-compatible server (llama.cpp / vLLM / Ollama / LM Studio):
97
+ drydock --provider vllm --base-url http://localhost:8000/v1 --model <served-model-name>
98
+ # or a keyed frontier model (key via OPENAI_API_KEY or config `api_key`):
99
+ OPENAI_API_KEY=sk-... drydock --provider openai --base-url https://api.openai.com/v1 --model gpt-4o
100
+ ```
101
+
63
102
  ## Capabilities
64
103
 
65
104
  A full agentic CLI harness — every tool below is clean-room and dependency-free
@@ -282,22 +321,6 @@ exports the open findings to a deterministic **eMASS POA&M CSV** — Control (fr
282
321
  the CCI map), Vulnerability Description, `POA&M Status=Ongoing`, Milestone (the
283
322
  Fix Text), and Severity (CAT I/II/III → High/Moderate/Low) — no LLM, stdlib only.
284
323
 
285
- ### FIAR audit-readiness (DoD financial-statement audit)
286
-
287
- Model a Financial Improvement and Audit Readiness engagement — a seeded key-control
288
- matrix per business cycle (FBWT, P2P, PP&E, INV, CIVPAY, REIM, FR, ITGC), the five FS
289
- assertions, a deterministic **evidence-chain validator** (a control can't be called
290
- effective on an incomplete population→sample→…→GL→assertion trace), findings (NFRs), and
291
- CAPs. `python -m drydock.fiar new|controls|control|assess|reconcile|package`; skills
292
- `/fiar-assess /fiar-evidence /fiar-readiness /fiar-cap`.
293
-
294
- **KSD evidence packaging.** `fiar package <engagement> <out> --evidence-dir <dir>` assembles
295
- an audit binder — `index.md` + `index.json` mapping every control → assertions → KSDs →
296
- status → evidence-chain completeness → findings — and **collects the actual evidence files
297
- each test cited**, zipped. Add `--redact "<term>"` (repeatable) to **red-box** names/secrets
298
- in the evidence *and* the manifest for a releasable version (verified via the Document
299
- Canvas); the original engagement is never touched. Also exposed as the `FiarPackage` tool.
300
-
301
324
  ### Document Canvas — editing documents far larger than the context window
302
325
 
303
326
  Edit a 300-, 800-, 1000-page document the way you edit a large codebase: **search,
@@ -1,10 +1,15 @@
1
1
  # ⚓ Drydock
2
2
 
3
- A local-first, provider-agnostic **terminal coding agent** for your own LLM.
4
- No accounts, no telemetry, no cloud — the only outbound calls are to the model
5
- endpoint you configure and (optionally) the web-search tools you invoke.
6
- Primary target: **dense Gemma-4-31B** (QAT, 64K) served by llama.cpp on a
7
- single workstation.
3
+ **The air-gapped terminal coding agent that automates your NIST compliance** —
4
+ RMF (800-53), CSF 2.0, AI RMF, and DISA STIG — running entirely on your own
5
+ hardware against a local LLM. No accounts, no telemetry, no cloud: your code,
6
+ controls, and CUI never leave the box.
7
+
8
+ Under the compliance skills it's a full agentic coding harness — the same
9
+ air-gapped agent whether you're assessing a control or shipping a feature. The
10
+ only outbound calls are to the model endpoint you configure and (optionally) the
11
+ web-search tools you invoke. Primary target: **dense Gemma-4-31B** (QAT, 64K)
12
+ served by llama.cpp on a single workstation.
8
13
 
9
14
  > **v3 — clean-room rebuild.** Drydock is being rebuilt as an original,
10
15
  > Apache-2.0 codebase owned end to end (no upstream fork). Every release is
@@ -14,10 +19,14 @@ single workstation.
14
19
 
15
20
  ## Why
16
21
 
17
- A coding agent should build real projects from your machine without sending
18
- your code or credentials anywhere. Drydock runs entirely against a local
19
- model, feels like a first-class terminal agent, and keeps its data plane on
20
- your box.
22
+ Regulated and air-gapped environments — federal, defense, critical-infrastructure,
23
+ and any org adopting NIST CSF 2.0 — can't send code, controls, or CUI to a cloud
24
+ agent, yet still need real automation. Drydock runs entirely against a local model,
25
+ feels like a first-class terminal agent, ingests the NIST 800-53 catalog and DISA
26
+ STIG benchmarks to assess / remediate / export POA&Ms **offline**, and keeps its
27
+ data plane on your box. It's a governable agent by design: a deterministic control
28
+ loop, a verification gate on "done", and a durable, tamper-evident event trace —
29
+ the audit record for what an autonomous agent did in your environment.
21
30
 
22
31
  ## Status
23
32
 
@@ -28,6 +37,36 @@ line, and a multi-line prompt. The agent loop, OpenAI-compatible provider,
28
37
  two-tier compaction, and the full agentic toolset (below) are in, with Gemma
29
38
  reliability hardening verified hands-on.
30
39
 
40
+ ## Quickstart
41
+
42
+ **One command** — installs Drydock, then either serves a strong local model or points
43
+ at a keyed endpoint, and launches you into the agent:
44
+
45
+ ```bash
46
+ curl -fsSL https://raw.githubusercontent.com/fbobe321/drydock/main/scripts/quickstart.sh | bash
47
+ ```
48
+
49
+ It offers two backends:
50
+
51
+ - **Local (air-gapped)** — downloads **Qwen2.5-Coder-32B** (one time) and serves it with
52
+ `llama.cpp`, then runs entirely offline. Needs a GPU box with `llama-server` on `PATH`.
53
+ - **Frontier (no GPU)** — try it in a minute against a keyed OpenAI-compatible endpoint;
54
+ only that call leaves the box:
55
+ ```bash
56
+ OPENAI_API_KEY=sk-... DRYDOCK_QUICKSTART_MODE=frontier \
57
+ curl -fsSL https://raw.githubusercontent.com/fbobe321/drydock/main/scripts/quickstart.sh | bash
58
+ ```
59
+
60
+ Prefer to wire it up by hand?
61
+
62
+ ```bash
63
+ pip install drydock-cli
64
+ # point at any OpenAI-compatible server (llama.cpp / vLLM / Ollama / LM Studio):
65
+ drydock --provider vllm --base-url http://localhost:8000/v1 --model <served-model-name>
66
+ # or a keyed frontier model (key via OPENAI_API_KEY or config `api_key`):
67
+ OPENAI_API_KEY=sk-... drydock --provider openai --base-url https://api.openai.com/v1 --model gpt-4o
68
+ ```
69
+
31
70
  ## Capabilities
32
71
 
33
72
  A full agentic CLI harness — every tool below is clean-room and dependency-free
@@ -250,22 +289,6 @@ exports the open findings to a deterministic **eMASS POA&M CSV** — Control (fr
250
289
  the CCI map), Vulnerability Description, `POA&M Status=Ongoing`, Milestone (the
251
290
  Fix Text), and Severity (CAT I/II/III → High/Moderate/Low) — no LLM, stdlib only.
252
291
 
253
- ### FIAR audit-readiness (DoD financial-statement audit)
254
-
255
- Model a Financial Improvement and Audit Readiness engagement — a seeded key-control
256
- matrix per business cycle (FBWT, P2P, PP&E, INV, CIVPAY, REIM, FR, ITGC), the five FS
257
- assertions, a deterministic **evidence-chain validator** (a control can't be called
258
- effective on an incomplete population→sample→…→GL→assertion trace), findings (NFRs), and
259
- CAPs. `python -m drydock.fiar new|controls|control|assess|reconcile|package`; skills
260
- `/fiar-assess /fiar-evidence /fiar-readiness /fiar-cap`.
261
-
262
- **KSD evidence packaging.** `fiar package <engagement> <out> --evidence-dir <dir>` assembles
263
- an audit binder — `index.md` + `index.json` mapping every control → assertions → KSDs →
264
- status → evidence-chain completeness → findings — and **collects the actual evidence files
265
- each test cited**, zipped. Add `--redact "<term>"` (repeatable) to **red-box** names/secrets
266
- in the evidence *and* the manifest for a releasable version (verified via the Document
267
- Canvas); the original engagement is never touched. Also exposed as the `FiarPackage` tool.
268
-
269
292
  ### Document Canvas — editing documents far larger than the context window
270
293
 
271
294
  Edit a 300-, 800-, 1000-page document the way you edit a large codebase: **search,
@@ -47,6 +47,7 @@ from drydock.tools import register_all
47
47
  register_all()
48
48
  from drydock.compaction import (
49
49
  maybe_compact, emergency_compact, is_context_length_error, is_image_load_error,
50
+ is_text_only_model_error,
50
51
  extract_server_n_ctx,
51
52
  )
52
53
  from drydock.loop_detect import LoopTracker, degenerate_argument
@@ -409,6 +410,17 @@ def run(
409
410
  if retries >= 2:
410
411
  raise
411
412
  yield TextChunk("\n[context limit hit — compacting and retrying...]\n")
413
+ elif is_text_only_model_error(err):
414
+ # The model can't take images at all. Stop attaching them for this
415
+ # endpoint and retry the step as text — a vision-less model can
416
+ # still do the task (read the file with tools), it just can't SEE it.
417
+ from drydock.providers import mark_text_only
418
+ if not mark_text_only(config):
419
+ raise
420
+ yield TextChunk(
421
+ "\n[This model is text-only — sending image references as plain "
422
+ "text from now on and retrying...]\n"
423
+ )
412
424
  elif is_image_load_error(err):
413
425
  # Corrupt/truncated/unsupported image the server couldn't decode —
414
426
  # end the turn cleanly rather than dumping the raw 400.
@@ -149,6 +149,42 @@ def run_oneshot(prompt: str, config: dict) -> None:
149
149
  sys.exit(2)
150
150
 
151
151
 
152
+ def run_oneshot_swarm(prompt: str, config: dict) -> None:
153
+ """One-shot via a competitive SWARM (config `auto_swarm` / --swarm): N agents attempt the
154
+ task in parallel in isolated worktrees, the verified winner is applied to the working tree.
155
+ Falls back to a single agent if the swarm can't run or converge."""
156
+ import subprocess
157
+ from drydock.swarm import run_swarm
158
+ cwd = config.get("cwd") or os.getcwd()
159
+ try:
160
+ agents = max(2, min(8, int(config.get("auto_swarm_agents", 4) or 4)))
161
+ except (TypeError, ValueError):
162
+ agents = 4
163
+ print(f"⚓ auto-swarm: {agents} parallel agents on: {prompt}", file=sys.stderr, flush=True)
164
+ try:
165
+ res = run_swarm(cwd, prompt, agents=agents, base_config=config, share=True, waves=2)
166
+ except Exception as e: # noqa: BLE001
167
+ print(f"(swarm couldn't run: {e} — falling back to single agent)", file=sys.stderr, flush=True)
168
+ return run_oneshot(prompt, config)
169
+ if not res.converged or res.winner is None:
170
+ print(f"(swarm ran {agents} agents, {len(res.candidates)} candidate(s), none converged "
171
+ f"— falling back to single agent)", file=sys.stderr, flush=True)
172
+ return run_oneshot(prompt, config)
173
+ w = res.winner
174
+ try:
175
+ r = subprocess.run(["git", "checkout", w.commit, "--", "."], cwd=cwd,
176
+ capture_output=True, text=True)
177
+ if r.returncode != 0:
178
+ print(f"Swarm winner {w.commit[:12]} not auto-applied ({r.stderr.strip()}). "
179
+ f"Apply: git checkout {w.commit} -- .", file=sys.stderr, flush=True)
180
+ return
181
+ except Exception as e: # noqa: BLE001
182
+ print(f"Swarm converged but apply errored: {e}", file=sys.stderr, flush=True)
183
+ return
184
+ print(f"✓ swarm converged: winner by {w.agent}, {w.tests_passed}/{w.tests_total} tests, "
185
+ f"{w.files_changed} file(s) changed, applied {w.commit[:12]}.", flush=True)
186
+
187
+
152
188
  def handle_command(cmd: str, state: AgentState, config: dict) -> bool:
153
189
  """Handle slash commands. Returns True if handled."""
154
190
  raw = cmd.strip() # case-preserved (paths/args must not be lowercased)
@@ -414,6 +450,10 @@ def main():
414
450
  help="Write the full run transcript (system + messages) here — for harvesting training data")
415
451
  parser.add_argument("--force-first-tool", action="store_true", help="Force tool_choice=required on first turn")
416
452
  parser.add_argument("--cli", action="store_true", help="Plain readline mode instead of the TUI")
453
+ parser.add_argument("--swarm", action="store_true", dest="swarm",
454
+ help="One-shot: solve the task with a competitive parallel swarm (winner applied)")
455
+ parser.add_argument("--no-swarm", action="store_true", dest="no_swarm",
456
+ help="One-shot: force a single agent even if auto_swarm is on")
417
457
  parser.add_argument(
418
458
  "--dangerously-skip-permissions",
419
459
  dest="dangerously_skip_permissions",
@@ -467,6 +507,15 @@ def main():
467
507
  # the PRD" content fought the system prompt and risked turning a plain
468
508
  # greeting into a runaway build. The TUI reads this from config; CLI modes
469
509
  # call load_project_instructions directly (see run_interactive/run_oneshot).
510
+ # A context_limit above the server's real window overflows it mid-task and the
511
+ # ctx gauge lies. Unless the user pinned --context-limit, ask the server once and
512
+ # adopt its window when smaller (never grow past config). Best-effort, short timeout.
513
+ if not args.context_limit and config.get("base_url"):
514
+ from drydock.providers import probe_server_context
515
+ n_ctx = probe_server_context(config["base_url"], timeout=2.0)
516
+ if n_ctx and n_ctx < int(config.get("context_limit") or 65536):
517
+ config["context_limit"] = n_ctx
518
+
470
519
  config["project_instructions"] = load_project_instructions()
471
520
  # In a git project, drop a discoverable per-project prompt template at
472
521
  # <project>/.drydock/system_prompt.md (inert until edited). Then load the
@@ -481,7 +530,19 @@ def main():
481
530
  _connect_mcp(config)
482
531
 
483
532
  if args.prompt:
484
- run_oneshot(args.prompt, config)
533
+ # Route to the swarm when forced (--swarm) or when auto_swarm is on AND the task
534
+ # looks substantial; --no-swarm and the triviality gate keep single-agent the default.
535
+ from drydock.swarm import looks_substantial
536
+ if getattr(args, "no_swarm", False):
537
+ use_swarm = False
538
+ elif getattr(args, "swarm", False):
539
+ use_swarm = True
540
+ else:
541
+ use_swarm = bool(config.get("auto_swarm")) and looks_substantial(args.prompt)
542
+ if use_swarm:
543
+ run_oneshot_swarm(args.prompt, config)
544
+ else:
545
+ run_oneshot(args.prompt, config)
485
546
  elif args.cli:
486
547
  run_interactive(config)
487
548
  else:
@@ -51,6 +51,19 @@ def extract_server_n_ctx(err: str) -> int | None:
51
51
  return None
52
52
 
53
53
 
54
+ def is_text_only_model_error(err: str) -> bool:
55
+ """Whether a provider 400 says the MODEL can't take images at all (a text-only
56
+ model, e.g. vLLM "X is not a multimodal model", llama.cpp "image input is not
57
+ supported" when no --mmproj is loaded). Distinct from a bad image: the fix is to
58
+ drop attachments and retry, not to ask the user for a different file."""
59
+ e = (err or "").lower()
60
+ return (
61
+ "not a multimodal model" in e
62
+ or "image input is not supported" in e
63
+ or ("does not support" in e and ("image" in e or "multimodal" in e or "vision" in e))
64
+ )
65
+
66
+
54
67
  def is_image_load_error(err: str) -> bool:
55
68
  """Whether a provider 400 is the server failing to decode an attached image
56
69
  (corrupt / truncated / unsupported), so the agent ends the turn with a clean
@@ -33,6 +33,12 @@ DEFAULTS: dict[str, object] = {
33
33
  # OpenRouter, Together). Mirrors `advisor_api_key`. Can also come from the
34
34
  # provider's api_key_env (e.g. OPENAI_API_KEY); the config value wins.
35
35
  "api_key": "",
36
+ # Auto-swarm (opt-in): when true, one-shot `-p` tasks that look like a substantial,
37
+ # self-contained implementation (swarm.looks_substantial) are solved by a competitive
38
+ # swarm of `auto_swarm_agents` parallel agents (winner applied), instead of one agent.
39
+ # Trivial edits/questions/reads always stay single-agent. Off by default.
40
+ "auto_swarm": False,
41
+ "auto_swarm_agents": 4,
36
42
  "max_tokens": 8192, # 4096 truncated large file writes mid-JSON (→ _raw fail)
37
43
  "temperature": 0.2,
38
44
  # The model server's context window (llama.cpp -c / vLLM --max-model-len).
@@ -39,7 +39,9 @@ class RepetitionDetected(StallRetry):
39
39
  re-issue would just loop again, so the agent goes straight to decisive mode."""
40
40
 
41
41
  from drydock.loop_detect import runaway_repetition_len
42
- from drydock.tuning import extract_thinking, strip_leaked_tool_calls, strip_thinking_tokens, use_streaming
42
+ from drydock.tuning import (
43
+ extract_thinking, split_think_tags, strip_leaked_tool_calls, strip_thinking_tokens, use_streaming,
44
+ )
43
45
 
44
46
  # ── Provider registry ─────────────────────────────────────────────────────
45
47
 
@@ -430,6 +432,31 @@ def _image_url_block(path: str) -> dict | None:
430
432
  return {"type": "image_url", "image_url": {"url": f"data:{mime};base64,{b64}"}}
431
433
 
432
434
 
435
+ # (base_url, model) pairs whose server rejected image input as text-only. Process-wide
436
+ # so every later request (and turn) sends plain text instead of re-hitting the 400.
437
+ _TEXT_ONLY: set[tuple[str, str]] = set()
438
+
439
+
440
+ def _vision_key(config: dict) -> tuple[str, str]:
441
+ prov = PROVIDERS.get(config.get("provider", "vllm"), PROVIDERS["vllm"])
442
+ base_url = config.get("base_url") or prov.get("base_url", "http://localhost:8000/v1")
443
+ return (str(base_url).rstrip("/"), str(config.get("model") or ""))
444
+
445
+
446
+ def mark_text_only(config: dict) -> bool:
447
+ """Record that this endpoint's model can't take images. Returns True if newly marked."""
448
+ key = _vision_key(config)
449
+ if key in _TEXT_ONLY:
450
+ return False
451
+ _TEXT_ONLY.add(key)
452
+ return True
453
+
454
+
455
+ def vision_enabled(config: dict) -> bool:
456
+ """False when vision is disabled in config or the server proved text-only."""
457
+ return config.get("vision", True) is not False and _vision_key(config) not in _TEXT_ONLY
458
+
459
+
433
460
  def _content_with_images(content):
434
461
  """Turn text that references on-disk image paths into a multimodal content
435
462
  list ([text, image_url…]); text without resolvable images passes through
@@ -451,15 +478,17 @@ def _content_with_images(content):
451
478
  _user_content_with_images = _content_with_images
452
479
 
453
480
 
454
- def messages_to_openai(messages: list, system: str) -> list:
455
- """Convert neutral messages to OpenAI API format."""
481
+ def messages_to_openai(messages: list, system: str, vision: bool = True) -> list:
482
+ """Convert neutral messages to OpenAI API format. vision=False sends image
483
+ references as plain text (text-only model)."""
456
484
  # value type is mixed (str content, multimodal list content, tool_calls list)
457
485
  result: list[dict] = [{"role": "system", "content": system}]
458
486
  for m in messages:
459
487
  role = m["role"]
460
488
  if role == "user":
461
489
  result.append({"role": "user",
462
- "content": _user_content_with_images(m["content"])})
490
+ "content": (_user_content_with_images(m["content"])
491
+ if vision else m["content"])})
463
492
  elif role == "assistant":
464
493
  msg = {"role": "assistant", "content": m.get("content") or None}
465
494
  tcs = m.get("tool_calls", [])
@@ -482,7 +511,7 @@ def messages_to_openai(messages: list, system: str) -> list:
482
511
  # to look at — attach it so the model SEES it (servers read images
483
512
  # from tool messages too, verified). Only ViewImage, so an unrelated
484
513
  # tool mentioning a .png path never balloons the request.
485
- if m.get("name") == "ViewImage" and isinstance(content, str):
514
+ if vision and m.get("name") == "ViewImage" and isinstance(content, str):
486
515
  content = _content_with_images(content)
487
516
  result.append({
488
517
  "role": "tool",
@@ -559,7 +588,7 @@ def stream(
559
588
  # dict(config) shallow-copy) so the TUI's STOP can reach this exact client.
560
589
  config.setdefault("_abort", {})["client"] = client
561
590
 
562
- oai_messages = messages_to_openai(messages, system)
591
+ oai_messages = messages_to_openai(messages, system, vision=vision_enabled(config))
563
592
 
564
593
  has_tools = bool(tool_schemas)
565
594
  # Gemma (and any local server whose model name we can't trust) corrupts
@@ -647,6 +676,9 @@ def stream(
647
676
  if delta.content:
648
677
  # Strip any leaked thinking-token markers (best-effort per chunk).
649
678
  chunk_text = strip_thinking_tokens(delta.content)
679
+ # <think>/</think> tags usually arrive as whole tokens; keep them in the
680
+ # buffer (split off at the end) but don't display the bare markers.
681
+ shown = chunk_text.replace("<think>", "").replace("</think>", "")
650
682
  # Keep ANY non-empty chunk — including whitespace-only ones. A
651
683
  # `.strip()` guard here used to DROP newline-only chunks (the blank
652
684
  # line a model streams between markdown blocks arrives as its own
@@ -654,7 +686,8 @@ def stream(
654
686
  # skips chunks that stripping emptied (e.g. a lone thinking marker).
655
687
  if chunk_text:
656
688
  text += chunk_text
657
- yield TextChunk(chunk_text)
689
+ if shown:
690
+ yield TextChunk(shown)
658
691
  # Throttled: only scan once text is long enough and every ~200
659
692
  # new chars, so the common path pays almost nothing.
660
693
  if len(text) - _rep_checked_at >= 200:
@@ -705,6 +738,9 @@ def stream(
705
738
  "input": inp,
706
739
  })
707
740
 
741
+ # A template-opened <think> leaves "reasoning…</think>answer" in the content —
742
+ # keep only the answer in the recorded turn (it was already shown while streaming).
743
+ _, text = split_think_tags(text)
708
744
  yield AssistantTurn(text, tool_calls, in_tok, out_tok)
709
745
 
710
746
 
@@ -734,6 +734,29 @@ class SwarmResult:
734
734
  candidates: list[Candidate]
735
735
 
736
736
 
737
+ # Auto-swarm triviality gate: route a task to the swarm only when it looks like a
738
+ # substantial, self-contained implementation — never a one-line fix, question, or read.
739
+ _SUBSTANTIAL = ("implement", "build the", "build a", "refactor", "rewrite", "create the",
740
+ "all functions", "each function", "whole module", "entire", "feature",
741
+ "make the tests", "tests pass", "test suite", "port ", "migrate",
742
+ "add support", "write the", "flesh out", "scaffold", "from scratch")
743
+ _TRIVIAL = ("typo", "rename", "one line", "one-line", "single line", "the comment",
744
+ "what is", "what's", "explain", "why ", "show me", "read ", "print the",
745
+ "list the", "how do", "how does", "?")
746
+
747
+
748
+ def looks_substantial(task: str) -> bool:
749
+ """Heuristic gate for auto-swarm (config `auto_swarm`): True only when the task is a
750
+ substantial, self-contained implementation worth parallel attempts — not a trivial edit,
751
+ a question, or a read. Conservative: unsure → False (single agent stays the default)."""
752
+ t = (task or "").lower().strip()
753
+ if len(t.split()) < 4:
754
+ return False
755
+ if any(x in t for x in _TRIVIAL):
756
+ return False
757
+ return any(x in t for x in _SUBSTANTIAL)
758
+
759
+
737
760
  def run_swarm(cwd: str | Path, objective: str, *, agents: "int | str" = 4,
738
761
  base_config: dict | None = None,
739
762
  base_ref: str = "HEAD", verify_cmd: str | None = None, fitness: str = "auto",
@@ -44,6 +44,13 @@ _FAMILIES: list[tuple[frozenset[str], tuple[str, ...]]] = [
44
44
  (frozenset({"Jobs"}),
45
45
  ("job", "background", "training", "still running", "long-running", "long running",
46
46
  "status of", "check on", "how is the", "how's the", "in the background")),
47
+ # Multi-agent delegation/parallelism: surface Swarm (competitive parallel agents),
48
+ # Worker (delegate a subtask), and Dispatch (parallel investigation) when the task is
49
+ # a substantial implementation/refactor where parallel attempts or delegation help.
50
+ (frozenset({"Swarm", "Worker", "Dispatch"}),
51
+ ("implement", "module", "feature", "refactor", "rewrite", "build the", "all functions",
52
+ "each function", "tests pass", "make the tests", "parallel", "swarm", "several",
53
+ "whole", "entire", "port ", "migrate")),
47
54
  ]
48
55
 
49
56
  # External/mutating tools deprioritised while merely exploring (PRD F1.4: read &