drydock-cli 3.1.30__tar.gz → 3.1.32__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (225) hide show
  1. {drydock_cli-3.1.30/drydock_cli.egg-info → drydock_cli-3.1.32}/PKG-INFO +1 -1
  2. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/advisor.py +29 -0
  3. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/agent.py +3 -0
  4. drydock_cli-3.1.32/drydock/asr.py +227 -0
  5. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/config.py +6 -0
  6. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/mission_run.py +18 -8
  7. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/tools/__init__.py +8 -1
  8. {drydock_cli-3.1.30 → drydock_cli-3.1.32/drydock_cli.egg-info}/PKG-INFO +1 -1
  9. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock_cli.egg-info/SOURCES.txt +3 -0
  10. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/pyproject.toml +1 -1
  11. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_advisor.py +55 -0
  12. drydock_cli-3.1.32/tests/test_asr.py +140 -0
  13. drydock_cli-3.1.32/tests/test_main_api_key.py +76 -0
  14. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_mission_run.py +27 -0
  15. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/LICENSE +0 -0
  16. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/NOTICE +0 -0
  17. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/README.md +0 -0
  18. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/__init__.py +0 -0
  19. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/__main__.py +0 -0
  20. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/bash_safety.py +0 -0
  21. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/bottleneck.py +0 -0
  22. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/budget.py +0 -0
  23. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/builtin_skills/__init__.py +0 -0
  24. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/builtin_skills/document-canvas.md +0 -0
  25. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/builtin_skills/fiar-assess.md +0 -0
  26. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/builtin_skills/fiar-cap.md +0 -0
  27. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/builtin_skills/fiar-evidence.md +0 -0
  28. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/builtin_skills/fiar-readiness.md +0 -0
  29. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/builtin_skills/ml-data.md +0 -0
  30. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/builtin_skills/ml-debug.md +0 -0
  31. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/builtin_skills/ml-finetune.md +0 -0
  32. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/builtin_skills/ml-metrics.md +0 -0
  33. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/builtin_skills/ml-rl.md +0 -0
  34. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/builtin_skills/ml-train.md +0 -0
  35. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/builtin_skills/nist-ai-rmf.md +0 -0
  36. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/builtin_skills/nist-csf.md +0 -0
  37. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/builtin_skills/rmf-categorize.md +0 -0
  38. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/builtin_skills/rmf-control.md +0 -0
  39. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/builtin_skills/rmf-poam.md +0 -0
  40. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/builtin_skills/rmf-review.md +0 -0
  41. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/builtin_skills/stig-assess.md +0 -0
  42. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/builtin_skills/stig-remediate.md +0 -0
  43. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/capacity.py +0 -0
  44. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/cci.py +0 -0
  45. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/cli.py +0 -0
  46. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/comms/__init__.py +0 -0
  47. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/comms/attention.py +0 -0
  48. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/comms/channels.py +0 -0
  49. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/comms/events.py +0 -0
  50. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/compaction.py +0 -0
  51. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/detect.py +0 -0
  52. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/doccanvas.py +0 -0
  53. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/eratchet.py +0 -0
  54. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/events.py +0 -0
  55. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/extract.py +0 -0
  56. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/fiar.py +0 -0
  57. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/gittools.py +0 -0
  58. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/graphrag.py +0 -0
  59. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/groundtruth.py +0 -0
  60. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/guards.py +0 -0
  61. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/jobs.py +0 -0
  62. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/loop_detect.py +0 -0
  63. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/mcp.py +0 -0
  64. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/mission.py +0 -0
  65. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/mission_cli.py +0 -0
  66. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/pdfredbox.py +0 -0
  67. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/phases.py +0 -0
  68. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/poam.py +0 -0
  69. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/predictions.py +0 -0
  70. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/progress.py +0 -0
  71. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/providers.py +0 -0
  72. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/ratchet.py +0 -0
  73. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/recipes.py +0 -0
  74. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/recovery.py +0 -0
  75. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/resume.py +0 -0
  76. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/rmf.py +0 -0
  77. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/rmf_graph.py +0 -0
  78. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/skills.py +0 -0
  79. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/stig.py +0 -0
  80. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/subagents.py +0 -0
  81. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/suggest.py +0 -0
  82. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/swarm.py +0 -0
  83. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/task_state.py +0 -0
  84. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/tool_policy.py +0 -0
  85. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/tool_registry.py +0 -0
  86. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/tool_result.py +0 -0
  87. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/tool_select.py +0 -0
  88. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/tool_validate.py +0 -0
  89. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/trajectory.py +0 -0
  90. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/tui/__init__.py +0 -0
  91. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/tui/app.py +0 -0
  92. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/tui/approval.py +0 -0
  93. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/tui/messages.py +0 -0
  94. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/tui/widgets.py +0 -0
  95. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/tuning.py +0 -0
  96. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/verification.py +0 -0
  97. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock/web.py +0 -0
  98. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock_cli.egg-info/dependency_links.txt +0 -0
  99. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock_cli.egg-info/entry_points.txt +0 -0
  100. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock_cli.egg-info/requires.txt +0 -0
  101. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/drydock_cli.egg-info/top_level.txt +0 -0
  102. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/setup.cfg +0 -0
  103. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_approval.py +0 -0
  104. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_back_command.py +0 -0
  105. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_bash_background.py +0 -0
  106. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_bash_binary_output.py +0 -0
  107. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_bash_crossplatform.py +0 -0
  108. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_bash_output_bounding.py +0 -0
  109. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_bash_process_group.py +0 -0
  110. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_bash_safety.py +0 -0
  111. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_bash_sanitize.py +0 -0
  112. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_bash_shell.py +0 -0
  113. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_bash_stdin.py +0 -0
  114. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_bash_stop_partial.py +0 -0
  115. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_bash_timeout_network.py +0 -0
  116. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_bash_timeout_param.py +0 -0
  117. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_bottleneck.py +0 -0
  118. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_budget.py +0 -0
  119. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_capacity.py +0 -0
  120. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_cci.py +0 -0
  121. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_cli_agents.py +0 -0
  122. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_comms.py +0 -0
  123. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_compact_command.py +0 -0
  124. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_compaction.py +0 -0
  125. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_config.py +0 -0
  126. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_config_migration.py +0 -0
  127. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_context_limit_config.py +0 -0
  128. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_context_limit_issue25.py +0 -0
  129. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_criteria_coverage.py +0 -0
  130. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_cycling_detection.py +0 -0
  131. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_declared_deps.py +0 -0
  132. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_degenerate_argument.py +0 -0
  133. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_detect.py +0 -0
  134. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_dispatch.py +0 -0
  135. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_doccanvas.py +0 -0
  136. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_e2e_connected.py +0 -0
  137. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_edit_replace_all.py +0 -0
  138. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_edit_thrash_and_compaction.py +0 -0
  139. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_effort_governor.py +0 -0
  140. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_empty_response.py +0 -0
  141. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_eratchet.py +0 -0
  142. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_events.py +0 -0
  143. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_extract.py +0 -0
  144. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_failure_loop.py +0 -0
  145. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_fiar.py +0 -0
  146. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_first_run_setup.py +0 -0
  147. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_gittools.py +0 -0
  148. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_glob_edit_edges.py +0 -0
  149. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_graphify_example.py +0 -0
  150. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_graphrag.py +0 -0
  151. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_graphrag_quoted_path.py +0 -0
  152. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_graphrag_sqlite.py +0 -0
  153. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_grep_and_read_robust.py +0 -0
  154. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_groundtruth_predictions.py +0 -0
  155. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_guards_and_tools.py +0 -0
  156. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_hallucinated_tools.py +0 -0
  157. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_jobs.py +0 -0
  158. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_knowledge_tool_available.py +0 -0
  159. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_leaked_tool_call.py +0 -0
  160. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_loop_detect.py +0 -0
  161. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_mcp.py +0 -0
  162. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_mission.py +0 -0
  163. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_mission_cli.py +0 -0
  164. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_model_registry.py +0 -0
  165. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_oneshot_unreachable.py +0 -0
  166. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_overthink_interrupt.py +0 -0
  167. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_pdfredbox.py +0 -0
  168. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_phases.py +0 -0
  169. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_plan_autocontinue.py +0 -0
  170. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_poam.py +0 -0
  171. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_progress.py +0 -0
  172. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_providers_unreachable.py +0 -0
  173. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_ratchet.py +0 -0
  174. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_ratchet_offer.py +0 -0
  175. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_read_index.py +0 -0
  176. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_recipes.py +0 -0
  177. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_recovery.py +0 -0
  178. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_recovery_config.py +0 -0
  179. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_recovery_integration.py +0 -0
  180. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_repeated_outcome.py +0 -0
  181. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_repetition_interrupt.py +0 -0
  182. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_resume_events.py +0 -0
  183. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_resume_restore.py +0 -0
  184. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_resume_snapshot.py +0 -0
  185. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_rmf.py +0 -0
  186. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_rmf_graph.py +0 -0
  187. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_rmf_stig_graph.py +0 -0
  188. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_rolling_plan.py +0 -0
  189. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_runaway_repetition.py +0 -0
  190. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_screenshot.py +0 -0
  191. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_server_probe.py +0 -0
  192. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_skills.py +0 -0
  193. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_sqlite_events.py +0 -0
  194. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_stall_retry.py +0 -0
  195. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_stig.py +0 -0
  196. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_stop.py +0 -0
  197. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_streaming_newlines.py +0 -0
  198. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_subagent.py +0 -0
  199. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_subagents.py +0 -0
  200. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_suggest.py +0 -0
  201. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_swarm.py +0 -0
  202. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_system_prompt_help.py +0 -0
  203. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_task_state.py +0 -0
  204. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_timeline.py +0 -0
  205. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_todo.py +0 -0
  206. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_tool_arg_coercion.py +0 -0
  207. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_tool_arg_coercion_more.py +0 -0
  208. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_tool_arg_parsing.py +0 -0
  209. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_tool_canonical.py +0 -0
  210. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_tool_policy.py +0 -0
  211. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_tool_result.py +0 -0
  212. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_tool_select.py +0 -0
  213. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_tool_validate.py +0 -0
  214. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_tools_undo.py +0 -0
  215. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_trajectory.py +0 -0
  216. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_tui.py +0 -0
  217. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_tuning.py +0 -0
  218. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_verification_gate.py +0 -0
  219. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_viewimage.py +0 -0
  220. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_vision_input.py +0 -0
  221. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_web_tools.py +0 -0
  222. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_windows_shell.py +0 -0
  223. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_worker_subagent.py +0 -0
  224. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_write_content_coerce.py +0 -0
  225. {drydock_cli-3.1.30 → drydock_cli-3.1.32}/tests/test_xccdf.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: drydock-cli
3
- Version: 3.1.30
3
+ Version: 3.1.32
4
4
  Summary: Drydock — a local, provider-agnostic terminal coding agent for local LLMs
5
5
  Author: Frank Bobe III
6
6
  License-Expression: Apache-2.0
@@ -24,6 +24,35 @@ _ADVISOR_SYSTEM = (
24
24
  )
25
25
 
26
26
 
27
+ def _msg_text(content) -> str:
28
+ """Flatten a message's content (str or multimodal block list) to plain text."""
29
+ if isinstance(content, str):
30
+ return content
31
+ if isinstance(content, list):
32
+ parts = [str(b.get("text") or b.get("content") or "") if isinstance(b, dict) else str(b)
33
+ for b in content]
34
+ return " ".join(p for p in parts if p)
35
+ return str(content or "")
36
+
37
+
38
+ def recent_context(messages: list, *, max_msgs: int = 12, max_chars: int = 6000) -> str:
39
+ """Compact the tail of the transcript to brief the advisor (§ Consult). The calling model —
40
+ especially a small local one — often passes little/no `context`, leaving the advisor blind;
41
+ this auto-attaches the last few turns (assistant reasoning + tool results/errors) so it can
42
+ actually help. Keeps the MOST RECENT content when truncating."""
43
+ if not messages:
44
+ return ""
45
+ out = []
46
+ for m in messages[-max_msgs:]:
47
+ if not isinstance(m, dict) or m.get("role") == "system":
48
+ continue
49
+ txt = _msg_text(m.get("content")).strip()
50
+ if txt:
51
+ out.append(f"[{m.get('role', '?')}] {txt[:1200]}")
52
+ blob = "\n".join(out)
53
+ return ("…" + blob[-max_chars:]) if len(blob) > max_chars else blob
54
+
55
+
27
56
  def is_configured(config: dict) -> bool:
28
57
  """True when an advisor endpoint + model are set."""
29
58
  return bool((config.get("advisor_base_url") or "").strip()
@@ -670,6 +670,9 @@ def run(
670
670
  canonical=canonical_name(tc["name"]),
671
671
  effect=tool_effect_of(tc["name"], _ro).value,
672
672
  input=str(tc.get("input"))[:200])
673
+ # Expose the recent transcript so context-hungry tools (Consult) can brief
674
+ # the advisor without relying on the model to hand-pass context.
675
+ config["_recent_messages"] = state.messages
673
676
  tool_result = execute_structured(tc["name"], tc["input"], config)
674
677
  result = tool_result.text
675
678
  # Emit the completion event IMMEDIATELY after execution (before
@@ -0,0 +1,227 @@
1
+ """Adaptive Swarm Runtime (ASR) — the search-decision core.
2
+
3
+ This is the resource-agnostic brain of the adaptive swarm (PRD "Resource-Aware Adaptive
4
+ Swarms"): it turns a compute/time BUDGET into decisions about *where the next unit of inference
5
+ should go* and *when to stop*. It deliberately holds no model, GPU, or scheduler state — those
6
+ live in capacity.py (concurrency), swarm.py (blackboard/candidates/worktrees) and the mission
7
+ store (durable state). Keeping the decisions pure makes them unit-testable and reusable.
8
+
9
+ Three pieces, in dependency order:
10
+ * Budget — a finite termination envelope parsed from a user string (§15).
11
+ * Convergence — should the swarm keep going? Stops on success / budget / no-improvement /
12
+ when marginal value Δquality/Δcompute falls below a floor (§23).
13
+ * AdaptiveAllocator — split the next compute pool across live branches by expected value,
14
+ pruning the weak ones (§17). Bad ideas must not get the same compute as good.
15
+ """
16
+ from __future__ import annotations
17
+
18
+ import re
19
+ from dataclasses import dataclass, field
20
+
21
+ # ── Budget (§15) ────────────────────────────────────────────────────────────
22
+ _DUR = re.compile(r"^\s*([\d.]+)\s*([hms])\s*$", re.I)
23
+ _TOK = re.compile(r"^\s*([\d.]+)\s*([kmg]?)\s*[-_]?tokens?\s*$", re.I)
24
+ _MULT = {"": 1, "k": 1_000, "m": 1_000_000, "g": 1_000_000_000}
25
+
26
+
27
+ @dataclass
28
+ class Budget:
29
+ """A finite stopping envelope (§10/§15). Every field is optional, but a Budget with no
30
+ finite limit is a bug — `is_finite` guards against "run forever"."""
31
+ wall_secs: float | None = None
32
+ tokens: int | None = None
33
+ generations: int | None = None
34
+
35
+ @property
36
+ def is_finite(self) -> bool:
37
+ return any(v is not None for v in (self.wall_secs, self.tokens, self.generations))
38
+
39
+ @classmethod
40
+ def parse(cls, spec: str) -> "Budget":
41
+ """Parse one user token: '2h' / '30m' / '90s' (wall) or '100M-tokens' / '50k-tokens'
42
+ (compute). Unknown strings raise ValueError so a typo never silently means 'unbounded'."""
43
+ s = (spec or "").strip()
44
+ md = _DUR.match(s)
45
+ if md:
46
+ n, unit = float(md.group(1)), md.group(2).lower()
47
+ return cls(wall_secs=n * {"h": 3600, "m": 60, "s": 1}[unit])
48
+ mt = _TOK.match(s)
49
+ if mt:
50
+ return cls(tokens=int(float(mt.group(1)) * _MULT[mt.group(2).lower()]))
51
+ raise ValueError(f"unparseable budget: {spec!r} (try '2h', '30m', '100M-tokens')")
52
+
53
+
54
+ # ── Convergence (§23) ─────────────────────────────────────────────────────────
55
+ @dataclass
56
+ class GenerationRecord:
57
+ gen: int
58
+ best_score: float # best branch quality so far (higher = better)
59
+ cumulative_tokens: int # total inference tokens spent through this generation
60
+ elapsed_s: float # wall-clock since start
61
+
62
+
63
+ @dataclass
64
+ class Convergence:
65
+ """Decides whether the swarm should run another generation (§23). "More agents are not
66
+ automatically better": we stop the moment any finite condition trips, INCLUDING the marginal
67
+ one — when Δquality per unit compute drops below `min_marginal`, extra inference is waste."""
68
+ budget: Budget
69
+ success_score: float = 100.0 # stop when best branch reaches this (e.g. all tests pass)
70
+ no_improve_gens: int = 3 # stop after this many generations with no best-score gain
71
+ min_marginal: float = 0.0 # floor on Δscore per 1k tokens; <=0 disables the check
72
+ eps: float = 1e-9
73
+ history: list[GenerationRecord] = field(default_factory=list)
74
+
75
+ def record(self, best_score: float, cumulative_tokens: int, elapsed_s: float) -> None:
76
+ self.history.append(GenerationRecord(len(self.history) + 1, best_score,
77
+ cumulative_tokens, elapsed_s))
78
+
79
+ def _gens_since_improvement(self) -> int:
80
+ """Generations since the best score was FIRST reached (a plateau at the best counts)."""
81
+ if not self.history:
82
+ return 0
83
+ best = max(r.best_score for r in self.history)
84
+ first_best = next(i for i, r in enumerate(self.history) if r.best_score >= best - self.eps)
85
+ return (len(self.history) - 1) - first_best
86
+
87
+ def marginal_value(self) -> float | None:
88
+ """Δscore per 1,000 tokens over the last generation, or None with <2 records."""
89
+ if len(self.history) < 2:
90
+ return None
91
+ a, b = self.history[-2], self.history[-1]
92
+ dtok = b.cumulative_tokens - a.cumulative_tokens
93
+ if dtok <= 0:
94
+ return None
95
+ return (b.best_score - a.best_score) / (dtok / 1000.0)
96
+
97
+ def should_continue(self) -> tuple[bool, str]:
98
+ """(keep_going?, reason). Checked in priority order; the reason names the stop trigger."""
99
+ last = self.history[-1] if self.history else None
100
+ if last is not None and last.best_score >= self.success_score - self.eps:
101
+ return False, "success score reached"
102
+ b = self.budget
103
+ if b.wall_secs is not None and last is not None and last.elapsed_s >= b.wall_secs:
104
+ return False, "wall-clock budget exhausted"
105
+ if b.tokens is not None and last is not None and last.cumulative_tokens >= b.tokens:
106
+ return False, "token budget exhausted"
107
+ if b.generations is not None and len(self.history) >= b.generations:
108
+ return False, "generation budget reached"
109
+ if self.no_improve_gens > 0 and self._gens_since_improvement() >= self.no_improve_gens:
110
+ return False, f"no improvement in {self.no_improve_gens} generations"
111
+ mv = self.marginal_value()
112
+ if self.min_marginal > 0 and mv is not None and mv < self.min_marginal:
113
+ return False, f"marginal value {mv:.3f} < floor {self.min_marginal:.3f} (Δquality/Δcompute)"
114
+ return True, "continue"
115
+
116
+
117
+ # ── Adaptive compute allocation (§17) ──────────────────────────────────────────
118
+ @dataclass
119
+ class Allocation:
120
+ grants: dict[str, int] # branch id -> tokens granted this round
121
+ terminated: list[str] # branch ids pruned (0 tokens, killed)
122
+
123
+
124
+ def allocate(branches: list[tuple[str, float]], pool_tokens: int, *,
125
+ prune_below: float = 0.0, keep_top: int | None = None,
126
+ bias: float = 1.5, min_grant: int = 0) -> Allocation:
127
+ """Split `pool_tokens` across branches by expected value (§17).
128
+
129
+ branches: (id, score) with higher score = more promising. A branch is TERMINATED (gets 0)
130
+ if its score is below `prune_below` or it falls outside the top `keep_top`. Survivors share
131
+ the pool in proportion to score**bias, so leaders get disproportionately more (bias>1) — the
132
+ system must not spend equally on good and bad ideas. `min_grant` guarantees a floor so a
133
+ kept-but-trailing branch still gets a probe. Deterministic; ties broken by input order."""
134
+ if pool_tokens <= 0 or not branches:
135
+ return Allocation({}, [b for b, _ in branches])
136
+ ranked = sorted(branches, key=lambda x: x[1], reverse=True)
137
+ survivors = [(b, s) for b, s in ranked if s >= prune_below]
138
+ if keep_top is not None:
139
+ survivors = survivors[:max(0, keep_top)]
140
+ kept_ids = {b for b, _ in survivors}
141
+ terminated = [b for b, _ in branches if b not in kept_ids]
142
+ if not survivors:
143
+ return Allocation({}, terminated)
144
+ weights = [(b, max(s, 0.0) ** bias) for b, s in survivors]
145
+ total_w = sum(w for _, w in weights) or float(len(weights))
146
+ grants: dict[str, int] = {}
147
+ for b, w in weights:
148
+ share = (w / total_w) if total_w else 1.0 / len(weights)
149
+ grants[b] = max(min_grant, int(round(pool_tokens * share)))
150
+ # correct rounding drift so grants sum to the pool (adjust the top branch)
151
+ drift = pool_tokens - sum(grants.values())
152
+ if drift and weights:
153
+ top = weights[0][0]
154
+ grants[top] = max(min_grant, grants[top] + drift)
155
+ return Allocation(grants, terminated)
156
+
157
+
158
+ # ── Planner: budget + intensity + hardware → population plan (§15/§16, §5/§9) ──
159
+ # Diverse roles for solution-space coverage (§9/§10). Correlated agents waste compute.
160
+ ROLES = [
161
+ "first-principles solver", "conventional solver", "skeptic", "debugger",
162
+ "security reviewer", "researcher", "minimalist", "adversarial critic",
163
+ "performance specialist", "alternative-architecture agent",
164
+ ]
165
+
166
+ # Per-intensity search shape (§16): initial population, critic count, generation cap, and how
167
+ # many branches survive each prune. Population is a SEARCH choice — independent of hardware.
168
+ INTENSITY = {
169
+ "light": {"population": 3, "critics": 1, "generations": 2, "keep": 2},
170
+ "standard": {"population": 8, "critics": 2, "generations": 4, "keep": 4},
171
+ "deep": {"population": 16, "critics": 3, "generations": 6, "keep": 6},
172
+ "extreme": {"population": 24, "critics": 5, "generations": 10, "keep": 8},
173
+ }
174
+
175
+
176
+ @dataclass
177
+ class SwarmPlan:
178
+ intensity: str
179
+ initial_population: int # logical agents in the explore stage (§9)
180
+ max_concurrency: int # simultaneous inference — HARDWARE bound (§5), != population
181
+ critics: int # judge/critic agents (§11)
182
+ generations: int # generation cap (a convergence bound, §23)
183
+ keep: int # survivors promoted past each prune (§12)
184
+ roles: list[str] # diversified assignments (§10)
185
+ estimated_logical: int # rough total agents incl. spawned descendants (§13) — info only
186
+
187
+ @property
188
+ def queued_at_start(self) -> int:
189
+ """Logical agents that must wait because concurrency < population (§18)."""
190
+ return max(0, self.initial_population - self.max_concurrency)
191
+
192
+
193
+ def choose_intensity(budget: Budget) -> str:
194
+ """Map a budget to a search intensity for `--swarm auto` (§16). Bigger budget → deeper
195
+ search. Falls back to 'standard' when the budget carries no size signal."""
196
+ if budget.wall_secs is not None:
197
+ s = budget.wall_secs
198
+ return "light" if s < 900 else "standard" if s < 3600 else "deep" if s < 14400 else "extreme"
199
+ if budget.tokens is not None:
200
+ t = budget.tokens
201
+ return ("light" if t < 1_000_000 else "standard" if t < 10_000_000
202
+ else "deep" if t < 50_000_000 else "extreme")
203
+ if budget.generations is not None:
204
+ g = budget.generations
205
+ return "light" if g <= 2 else "standard" if g <= 4 else "deep" if g <= 6 else "extreme"
206
+ return "standard"
207
+
208
+
209
+ def plan(budget: Budget, concurrency: int, *, intensity: str = "auto") -> SwarmPlan:
210
+ """Turn a budget + measured hardware concurrency into a population plan (§15/§16).
211
+
212
+ KEY: logical population comes from the intensity (a search decision); `concurrency` only
213
+ caps how many run at once (§5). On a 1-GPU box a 16-agent 'deep' plan still has 16 logical
214
+ agents — the rest queue (§18); the penalty is wall-clock, not lost search capability. This
215
+ is the opposite of capacity.swarm_size(), which caps logical count AT concurrency."""
216
+ intensity = choose_intensity(budget) if intensity == "auto" else intensity
217
+ if intensity not in INTENSITY:
218
+ raise ValueError(f"unknown intensity {intensity!r} (light|standard|deep|extreme|auto)")
219
+ p = INTENSITY[intensity]
220
+ pop = p["population"]
221
+ roles = [ROLES[i % len(ROLES)] for i in range(pop)]
222
+ # estimate total logical agents: initial + ~one descendant per survivor per later generation
223
+ estimated = pop + p["keep"] * max(0, p["generations"] - 1)
224
+ return SwarmPlan(intensity=intensity, initial_population=pop,
225
+ max_concurrency=max(1, concurrency), critics=p["critics"],
226
+ generations=p["generations"], keep=p["keep"], roles=roles,
227
+ estimated_logical=estimated)
@@ -27,6 +27,12 @@ DEFAULTS: dict[str, object] = {
27
27
  # default invisibly, which left base_url absent from the file. For a non-vllm
28
28
  # provider, override base_url (or use --base-url / the first-run prompt).
29
29
  "base_url": "http://localhost:8000/v1",
30
+ # API key for the MAIN model endpoint, sent as `Authorization: Bearer <key>`.
31
+ # Leave "" for a local server that needs no auth (llama.cpp/Ollama/LM Studio);
32
+ # set it for a hosted OpenAI-compatible endpoint (e.g. a vLLM behind a gateway,
33
+ # OpenRouter, Together). Mirrors `advisor_api_key`. Can also come from the
34
+ # provider's api_key_env (e.g. OPENAI_API_KEY); the config value wins.
35
+ "api_key": "",
30
36
  "max_tokens": 8192, # 4096 truncated large file writes mid-JSON (→ _raw fail)
31
37
  "temperature": 0.2,
32
38
  # The model server's context window (llama.cpp -c / vLLM --max-model-len).
@@ -270,14 +270,21 @@ class TaskOutcome:
270
270
 
271
271
  def run_task(store: M.MissionStore, task: dict, *, mission: dict, cwd: str, repo: str,
272
272
  worker: WorkerFn, evaluator: EvaluatorFn, base_config: dict | None = None,
273
- worker_id: str = "worker-1") -> TaskOutcome | None:
274
- """Execute one leased task with checkpoint + KEEP/REVERT. None if the lease was lost."""
273
+ worker_id: str = "worker-1",
274
+ checkpoint_fn: "Callable[[str], str]" = checkpoint,
275
+ restore_fn: "Callable[[str, str], bool]" = restore) -> TaskOutcome | None:
276
+ """Execute one leased task with checkpoint + KEEP/REVERT. None if the lease was lost.
277
+
278
+ checkpoint_fn/restore_fn default to git (`repo` = a checkout dir), but are injectable so a
279
+ non-git target can plug in its own snapshot mechanism — e.g. `docker commit` for a task that
280
+ lives in a container (the tbench/ddt harness). `repo` is then just an opaque handle passed
281
+ to those callables; an empty `repo` disables checkpointing."""
275
282
  mid = mission["id"]
276
283
  tid = task["id"]
277
284
  if not store.claim_task(tid, worker_id):
278
285
  return None
279
286
  before = mission.get("current_metric") or 0.0
280
- cp = checkpoint(repo) if repo else ""
287
+ cp = checkpoint_fn(repo) if repo else ""
281
288
  store.event(mid, "task_started", task=tid, objective=task.get("objective", ""))
282
289
  # Context reconstruction (§13): surface known failed approaches so the worker doesn't
283
290
  # repeat them (§16). Key the lookup on the SPECIFIC target (task reason, e.g. the failing
@@ -295,7 +302,7 @@ def run_task(store: M.MissionStore, task: dict, *, mission: dict, cwd: str, repo
295
302
  store.add_usage(mid, tokens=wr.in_tokens + wr.out_tokens, wall_s=time.monotonic() - t0)
296
303
  if not wr.ok:
297
304
  if repo and cp:
298
- restore(repo, cp)
305
+ restore_fn(repo, cp)
299
306
  store.complete_task(tid, M.T_FAILED, {"error": wr.error})
300
307
  return TaskOutcome(tid, accept=False, metric_after=before, reverted=True, error=wr.error)
301
308
 
@@ -309,7 +316,7 @@ def run_task(store: M.MissionStore, task: dict, *, mission: dict, cwd: str, repo
309
316
  tamper = tampered_paths(changed, protected)
310
317
  if tamper:
311
318
  if repo and cp:
312
- restore(repo, cp)
319
+ restore_fn(repo, cp)
313
320
  store.add_usage(mid, experiments=1)
314
321
  store.complete_task(tid, M.T_FAILED,
315
322
  {"summary": wr.summary, "decision": "REVERT", "reverted": True,
@@ -327,7 +334,7 @@ def run_task(store: M.MissionStore, task: dict, *, mission: dict, cwd: str, repo
327
334
  ev = evaluator(task, mission, cwd, before)
328
335
  store.add_usage(mid, experiments=1)
329
336
  if ev.accept:
330
- end = checkpoint(repo) if repo else ""
337
+ end = checkpoint_fn(repo) if repo else ""
331
338
  store.set_metric(mid, ev.metric_after)
332
339
  store.complete_task(tid, M.T_COMPLETED,
333
340
  {"summary": wr.summary, "metric": ev.metric_after, "decision": "KEEP",
@@ -340,7 +347,7 @@ def run_task(store: M.MissionStore, task: dict, *, mission: dict, cwd: str, repo
340
347
  f"{(wr.summary or '')[:220]}", confidence=0.7, sources=[tid])
341
348
  else:
342
349
  if repo and cp:
343
- restore(repo, cp) # auto-revert the regression (§19)
350
+ restore_fn(repo, cp) # auto-revert the regression (§19)
344
351
  store.complete_task(tid, M.T_COMPLETED,
345
352
  {"summary": wr.summary, "decision": "REVERT", "reverted": True})
346
353
  # negative knowledge stays in history even though the code is reverted (§16/§19)
@@ -505,6 +512,8 @@ def run_mission(store: M.MissionStore, mission_id: str, *, cwd: str, repo: str =
505
512
  review_every_tasks: int = 10, review_every_secs: float = 3600.0,
506
513
  reviewer: Callable[[M.MissionStore, dict, dict], dict] | None = None,
507
514
  critic: Callable[[M.MissionStore, dict, dict], str] | None = None,
515
+ checkpoint_fn: "Callable[[str], str]" = checkpoint,
516
+ restore_fn: "Callable[[str, str], bool]" = restore,
508
517
  on_event: Callable[[str, dict], None] | None = None) -> RunSummary:
509
518
  """Drive a mission to a stopping condition with a SINGLE worker (§41/§49). Stops on
510
519
  success (§30), budget (§30), stagnation (§22), or an empty queue (planner is a later
@@ -557,7 +566,8 @@ def run_mission(store: M.MissionStore, mission_id: str, *, cwd: str, repo: str =
557
566
  task = ready[0]
558
567
  out = run_task(store, task, mission=m, cwd=cwd, repo=repo, worker=worker,
559
568
  evaluator=(evaluator or (lambda *_: Evaluation(accept=True))),
560
- base_config=base_config, worker_id=worker_id)
569
+ base_config=base_config, worker_id=worker_id,
570
+ checkpoint_fn=checkpoint_fn, restore_fn=restore_fn)
561
571
  progressed = bool(out and out.accept)
562
572
  _ev("task_done", task=task["id"], accepted=progressed,
563
573
  metric=out.metric_after if out else None)
@@ -2054,7 +2054,14 @@ def tool_consult(params: dict, config: dict) -> str:
2054
2054
  question = _as_str_arg(params.get("question") or params.get("prompt")).strip()
2055
2055
  if not question:
2056
2056
  return "Error: Consult needs a `question`."
2057
- return advisor.consult(question, config, context=params.get("context", "") or "")
2057
+ ctx = _as_str_arg(params.get("context") or "").strip()
2058
+ # Auto-enrich: the advisor is useless without knowing what's going on, and the calling
2059
+ # model often passes thin/no context. Attach the recent transcript (errors + reasoning)
2060
+ # that the agent stashed in config so the advisor always has real information to work with.
2061
+ auto = advisor.recent_context(config.get("_recent_messages") or [])
2062
+ if auto:
2063
+ ctx = (ctx + "\n\n" if ctx else "") + "Recent activity (auto-attached):\n" + auto
2064
+ return advisor.consult(question, config, context=ctx)
2058
2065
 
2059
2066
 
2060
2067
  # A sub-agent's whole job is to keep its investigation OUT of the main agent's
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: drydock-cli
3
- Version: 3.1.30
3
+ Version: 3.1.32
4
4
  Summary: Drydock — a local, provider-agnostic terminal coding agent for local LLMs
5
5
  Author: Frank Bobe III
6
6
  License-Expression: Apache-2.0
@@ -6,6 +6,7 @@ drydock/__init__.py
6
6
  drydock/__main__.py
7
7
  drydock/advisor.py
8
8
  drydock/agent.py
9
+ drydock/asr.py
9
10
  drydock/bash_safety.py
10
11
  drydock/bottleneck.py
11
12
  drydock/budget.py
@@ -95,6 +96,7 @@ drydock_cli.egg-info/requires.txt
95
96
  drydock_cli.egg-info/top_level.txt
96
97
  tests/test_advisor.py
97
98
  tests/test_approval.py
99
+ tests/test_asr.py
98
100
  tests/test_back_command.py
99
101
  tests/test_bash_background.py
100
102
  tests/test_bash_binary_output.py
@@ -152,6 +154,7 @@ tests/test_jobs.py
152
154
  tests/test_knowledge_tool_available.py
153
155
  tests/test_leaked_tool_call.py
154
156
  tests/test_loop_detect.py
157
+ tests/test_main_api_key.py
155
158
  tests/test_mcp.py
156
159
  tests/test_mission.py
157
160
  tests/test_mission_cli.py
@@ -7,7 +7,7 @@ build-backend = "setuptools.build_meta"
7
7
  # PyPI distribution name is drydock-cli (the established install name, continued
8
8
  # from the retired v2 fork); the import package + CLI command stay `drydock`.
9
9
  name = "drydock-cli"
10
- version = "3.1.30"
10
+ version = "3.1.32"
11
11
  description = "Drydock — a local, provider-agnostic terminal coding agent for local LLMs"
12
12
  readme = "README.md"
13
13
  requires-python = ">=3.11"
@@ -88,3 +88,58 @@ def test_test_connection_timeout_says_reachable_but_slow(monkeypatch):
88
88
  monkeypatch.setattr(advisor, "_call", slow)
89
89
  out = advisor.test_connection({"advisor_base_url": "http://b/v1", "advisor_model": "m"})
90
90
  assert out.startswith("✗") and "REACHABLE but slow" in out and "/ask" in out
91
+
92
+
93
+ # ── recent-context auto-enrichment (advisor gets info even w/ thin caller context) ──
94
+ def test_recent_context_formats_and_keeps_newest():
95
+ msgs = [
96
+ {"role": "system", "content": "SYS should be skipped"},
97
+ {"role": "assistant", "content": "I'll run the tests."},
98
+ {"role": "tool", "content": "FAILED test_x - AssertionError: expected 42"},
99
+ {"role": "assistant", "content": "The assertion failed; investigating add()."},
100
+ ]
101
+ out = advisor.recent_context(msgs)
102
+ assert "SYS should be skipped" not in out # system dropped
103
+ assert "FAILED test_x" in out and "investigating add()" in out
104
+ assert "[tool]" in out and "[assistant]" in out
105
+
106
+
107
+ def test_recent_context_truncates_to_recent_chars():
108
+ msgs = [{"role": "assistant", "content": "x" * 500} for _ in range(40)]
109
+ out = advisor.recent_context(msgs, max_chars=1000)
110
+ assert len(out) <= 1001 and out.startswith("…") # kept the tail
111
+
112
+
113
+ def test_recent_context_handles_multimodal_blocks():
114
+ msgs = [{"role": "user", "content": [{"type": "text", "text": "look at this"}, {"type": "image"}]}]
115
+ assert "look at this" in advisor.recent_context(msgs)
116
+
117
+
118
+ def test_consult_tool_auto_attaches_recent_transcript(monkeypatch):
119
+ from drydock import tools as T
120
+ captured = {}
121
+ monkeypatch.setattr(advisor, "consult",
122
+ lambda q, cfg, context="": captured.update(question=q, context=context) or "ok")
123
+ cfg = {
124
+ "advisor_base_url": "http://b/v1", "advisor_model": "m",
125
+ "_recent_messages": [
126
+ {"role": "tool", "content": "FAILED test_reconnect - state lost after reconnect"},
127
+ {"role": "assistant", "content": "Trying a lock around the queue."},
128
+ ],
129
+ }
130
+ # model passed only a bare question (thin context) — the fix must attach the transcript
131
+ T.tool_consult({"question": "why does reconnect lose state?"}, cfg)
132
+ assert "FAILED test_reconnect" in captured["context"] # advisor now sees the error
133
+ assert "lock around the queue" in captured["context"]
134
+ assert "Recent activity" in captured["context"]
135
+
136
+
137
+ def test_consult_tool_keeps_model_context_and_adds_recent(monkeypatch):
138
+ from drydock import tools as T
139
+ captured = {}
140
+ monkeypatch.setattr(advisor, "consult",
141
+ lambda q, cfg, context="": captured.update(context=context) or "ok")
142
+ cfg = {"advisor_base_url": "http://b/v1", "advisor_model": "m",
143
+ "_recent_messages": [{"role": "tool", "content": "ERR: boom"}]}
144
+ T.tool_consult({"question": "help", "context": "my hand-written context"}, cfg)
145
+ assert "my hand-written context" in captured["context"] and "ERR: boom" in captured["context"]
@@ -0,0 +1,140 @@
1
+ """Tests for the Adaptive Swarm Runtime search core (drydock/asr.py) — pure, no model/GPU."""
2
+ from __future__ import annotations
3
+
4
+ import pytest
5
+
6
+ from drydock import asr
7
+
8
+
9
+ # ── Budget (§15) ────────────────────────────────────────────────────────────
10
+ def test_budget_parse_time_and_tokens():
11
+ assert asr.Budget.parse("2h").wall_secs == 7200
12
+ assert asr.Budget.parse("30m").wall_secs == 1800
13
+ assert asr.Budget.parse("90s").wall_secs == 90
14
+ assert asr.Budget.parse("100M-tokens").tokens == 100_000_000
15
+ assert asr.Budget.parse("50k-tokens").tokens == 50_000
16
+ assert asr.Budget.parse("2h").is_finite
17
+
18
+
19
+ def test_budget_rejects_garbage_never_silently_unbounded():
20
+ with pytest.raises(ValueError):
21
+ asr.Budget.parse("whenever")
22
+ assert asr.Budget().is_finite is False # empty budget is explicitly not finite
23
+
24
+
25
+ # ── Convergence (§23) ─────────────────────────────────────────────────────────
26
+ def _conv(**kw):
27
+ return asr.Convergence(budget=asr.Budget(generations=100), **kw)
28
+
29
+
30
+ def test_convergence_stops_on_success():
31
+ c = _conv(success_score=100.0)
32
+ c.record(best_score=100.0, cumulative_tokens=5000, elapsed_s=60)
33
+ go, why = c.should_continue()
34
+ assert not go and "success" in why
35
+
36
+
37
+ def test_convergence_stops_on_budget():
38
+ c = asr.Convergence(budget=asr.Budget(tokens=10_000))
39
+ c.record(best_score=40.0, cumulative_tokens=12_000, elapsed_s=30)
40
+ go, why = c.should_continue()
41
+ assert not go and "token budget" in why
42
+
43
+
44
+ def test_convergence_stops_on_no_improvement():
45
+ c = _conv(no_improve_gens=2, success_score=100.0)
46
+ c.record(50.0, 1000, 10)
47
+ c.record(50.0, 2000, 20) # no gain
48
+ c.record(50.0, 3000, 30) # still no gain -> 2 stagnant gens
49
+ go, why = c.should_continue()
50
+ assert not go and "no improvement" in why
51
+
52
+
53
+ def test_convergence_stops_on_low_marginal_value():
54
+ # big jump, then a tiny gain for a lot of tokens -> marginal value below floor
55
+ c = _conv(min_marginal=1.0, no_improve_gens=0, success_score=100.0)
56
+ c.record(40.0, 1000, 10)
57
+ c.record(80.0, 3000, 20) # +40 over 2k tok = 20/1k -> continues
58
+ assert c.should_continue()[0] is True
59
+ c.record(80.5, 13000, 40) # +0.5 over 10k tok = 0.05/1k -> below 1.0 floor
60
+ go, why = c.should_continue()
61
+ assert not go and "marginal value" in why
62
+
63
+
64
+ def test_convergence_continues_while_improving():
65
+ c = _conv(no_improve_gens=3, min_marginal=0.5, success_score=100.0)
66
+ c.record(40.0, 1000, 10)
67
+ c.record(60.0, 2000, 20) # +20/1k, improving, under budget
68
+ assert c.should_continue() == (True, "continue")
69
+
70
+
71
+ def test_marginal_value_needs_two_records():
72
+ c = _conv()
73
+ assert c.marginal_value() is None
74
+ c.record(10.0, 500, 5)
75
+ assert c.marginal_value() is None
76
+
77
+
78
+ # ── Adaptive allocation (§17) ──────────────────────────────────────────────────
79
+ def test_allocate_favors_leaders_and_prunes_weak():
80
+ branches = [("A", 0.91), ("B", 0.86), ("C", 0.43), ("D", 0.21)]
81
+ alloc = asr.allocate(branches, 8000, prune_below=0.3)
82
+ assert "D" in alloc.terminated and alloc.grants.get("D", 0) == 0 # weakest pruned
83
+ assert set(alloc.grants) == {"A", "B", "C"}
84
+ assert sum(alloc.grants.values()) == 8000 # whole pool spent
85
+ assert alloc.grants["A"] > alloc.grants["B"] > alloc.grants["C"] # EV ordering
86
+ assert alloc.grants["C"] < alloc.grants["A"] / 2 # bias toward the leader
87
+
88
+
89
+ def test_allocate_keep_top_terminates_the_rest():
90
+ branches = [("A", 0.9), ("B", 0.8), ("C", 0.7), ("D", 0.6)]
91
+ alloc = asr.allocate(branches, 6000, keep_top=2)
92
+ assert set(alloc.grants) == {"A", "B"}
93
+ assert sorted(alloc.terminated) == ["C", "D"]
94
+ assert sum(alloc.grants.values()) == 6000
95
+
96
+
97
+ def test_allocate_empty_pool_terminates_all():
98
+ alloc = asr.allocate([("A", 0.9), ("B", 0.5)], 0)
99
+ assert alloc.grants == {} and sorted(alloc.terminated) == ["A", "B"]
100
+
101
+
102
+ def test_allocate_min_grant_floor_for_kept_branches():
103
+ branches = [("A", 0.99), ("B", 0.01)]
104
+ alloc = asr.allocate(branches, 10_000, min_grant=200)
105
+ assert alloc.grants["B"] >= 200 # trailing-but-kept branch still probed
106
+ assert alloc.grants["A"] > alloc.grants["B"]
107
+
108
+
109
+ # ── Planner (§15/§16, §5/§18) ──────────────────────────────────────────────────
110
+ def test_choose_intensity_scales_with_budget():
111
+ assert asr.choose_intensity(asr.Budget.parse("10m")) == "light"
112
+ assert asr.choose_intensity(asr.Budget.parse("45m")) == "standard"
113
+ assert asr.choose_intensity(asr.Budget.parse("2h")) == "deep"
114
+ assert asr.choose_intensity(asr.Budget.parse("8h")) == "extreme"
115
+ assert asr.choose_intensity(asr.Budget.parse("5M-tokens")) == "standard"
116
+
117
+
118
+ def test_plan_decouples_logical_population_from_concurrency():
119
+ # THE core ASR property (§5/§18): a deep plan on a 1-slot box keeps all 16 logical agents;
120
+ # they queue, they aren't discarded down to concurrency (unlike capacity.swarm_size).
121
+ weak = asr.plan(asr.Budget.parse("2h"), concurrency=1, intensity="deep")
122
+ assert weak.initial_population == 16 and weak.max_concurrency == 1
123
+ assert weak.queued_at_start == 15
124
+ strong = asr.plan(asr.Budget.parse("2h"), concurrency=8, intensity="deep")
125
+ assert strong.initial_population == 16 and strong.max_concurrency == 8
126
+ assert strong.queued_at_start == 8 # same search, more parallelism → less queue
127
+
128
+
129
+ def test_plan_auto_and_intensity_ladder():
130
+ auto = asr.plan(asr.Budget.parse("30m"), concurrency=4) # auto -> standard
131
+ assert auto.intensity == "standard"
132
+ pops = [asr.plan(asr.Budget.parse("2h"), 4, intensity=i).initial_population
133
+ for i in ("light", "standard", "deep", "extreme")]
134
+ assert pops == sorted(pops) and pops[0] < pops[-1] # deeper => bigger population
135
+ assert len(auto.roles) == auto.initial_population and len(set(auto.roles)) > 1 # diverse
136
+
137
+
138
+ def test_plan_rejects_unknown_intensity():
139
+ with pytest.raises(ValueError):
140
+ asr.plan(asr.Budget.parse("1h"), 4, intensity="ludicrous")