drydock-cli 3.1.2__tar.gz → 3.1.3__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (190) hide show
  1. {drydock_cli-3.1.2/drydock_cli.egg-info → drydock_cli-3.1.3}/PKG-INFO +42 -1
  2. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/README.md +41 -0
  3. drydock_cli-3.1.3/drydock/builtin_skills/document-canvas.md +40 -0
  4. drydock_cli-3.1.3/drydock/doccanvas.py +525 -0
  5. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/tools/__init__.py +376 -0
  6. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/tui/widgets.py +1 -1
  7. {drydock_cli-3.1.2 → drydock_cli-3.1.3/drydock_cli.egg-info}/PKG-INFO +42 -1
  8. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock_cli.egg-info/SOURCES.txt +3 -0
  9. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/pyproject.toml +1 -1
  10. drydock_cli-3.1.3/tests/test_doccanvas.py +252 -0
  11. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_skills.py +1 -0
  12. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/LICENSE +0 -0
  13. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/NOTICE +0 -0
  14. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/__init__.py +0 -0
  15. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/__main__.py +0 -0
  16. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/advisor.py +0 -0
  17. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/agent.py +0 -0
  18. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/bash_safety.py +0 -0
  19. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/budget.py +0 -0
  20. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/builtin_skills/__init__.py +0 -0
  21. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/builtin_skills/fiar-assess.md +0 -0
  22. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/builtin_skills/fiar-cap.md +0 -0
  23. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/builtin_skills/fiar-evidence.md +0 -0
  24. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/builtin_skills/fiar-readiness.md +0 -0
  25. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/builtin_skills/ml-data.md +0 -0
  26. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/builtin_skills/ml-debug.md +0 -0
  27. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/builtin_skills/ml-finetune.md +0 -0
  28. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/builtin_skills/ml-metrics.md +0 -0
  29. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/builtin_skills/ml-rl.md +0 -0
  30. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/builtin_skills/ml-train.md +0 -0
  31. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/builtin_skills/nist-ai-rmf.md +0 -0
  32. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/builtin_skills/nist-csf.md +0 -0
  33. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/builtin_skills/rmf-categorize.md +0 -0
  34. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/builtin_skills/rmf-control.md +0 -0
  35. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/builtin_skills/rmf-poam.md +0 -0
  36. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/builtin_skills/rmf-review.md +0 -0
  37. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/builtin_skills/stig-assess.md +0 -0
  38. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/builtin_skills/stig-remediate.md +0 -0
  39. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/cci.py +0 -0
  40. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/cli.py +0 -0
  41. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/compaction.py +0 -0
  42. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/config.py +0 -0
  43. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/detect.py +0 -0
  44. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/events.py +0 -0
  45. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/extract.py +0 -0
  46. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/fiar.py +0 -0
  47. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/gittools.py +0 -0
  48. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/graphrag.py +0 -0
  49. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/guards.py +0 -0
  50. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/loop_detect.py +0 -0
  51. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/mcp.py +0 -0
  52. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/phases.py +0 -0
  53. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/poam.py +0 -0
  54. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/progress.py +0 -0
  55. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/providers.py +0 -0
  56. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/recipes.py +0 -0
  57. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/recovery.py +0 -0
  58. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/resume.py +0 -0
  59. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/rmf.py +0 -0
  60. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/rmf_graph.py +0 -0
  61. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/skills.py +0 -0
  62. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/stig.py +0 -0
  63. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/subagents.py +0 -0
  64. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/suggest.py +0 -0
  65. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/task_state.py +0 -0
  66. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/tool_policy.py +0 -0
  67. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/tool_registry.py +0 -0
  68. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/tool_result.py +0 -0
  69. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/tool_select.py +0 -0
  70. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/tool_validate.py +0 -0
  71. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/trajectory.py +0 -0
  72. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/tui/__init__.py +0 -0
  73. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/tui/app.py +0 -0
  74. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/tui/approval.py +0 -0
  75. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/tui/messages.py +0 -0
  76. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/tuning.py +0 -0
  77. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/verification.py +0 -0
  78. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/web.py +0 -0
  79. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock_cli.egg-info/dependency_links.txt +0 -0
  80. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock_cli.egg-info/entry_points.txt +0 -0
  81. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock_cli.egg-info/requires.txt +0 -0
  82. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock_cli.egg-info/top_level.txt +0 -0
  83. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/setup.cfg +0 -0
  84. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_advisor.py +0 -0
  85. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_approval.py +0 -0
  86. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_back_command.py +0 -0
  87. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_bash_background.py +0 -0
  88. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_bash_binary_output.py +0 -0
  89. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_bash_crossplatform.py +0 -0
  90. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_bash_output_bounding.py +0 -0
  91. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_bash_process_group.py +0 -0
  92. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_bash_safety.py +0 -0
  93. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_bash_sanitize.py +0 -0
  94. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_bash_shell.py +0 -0
  95. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_bash_stdin.py +0 -0
  96. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_bash_stop_partial.py +0 -0
  97. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_bash_timeout_network.py +0 -0
  98. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_bash_timeout_param.py +0 -0
  99. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_budget.py +0 -0
  100. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_cci.py +0 -0
  101. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_cli_agents.py +0 -0
  102. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_compact_command.py +0 -0
  103. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_compaction.py +0 -0
  104. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_config.py +0 -0
  105. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_config_migration.py +0 -0
  106. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_context_limit_config.py +0 -0
  107. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_context_limit_issue25.py +0 -0
  108. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_criteria_coverage.py +0 -0
  109. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_cycling_detection.py +0 -0
  110. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_degenerate_argument.py +0 -0
  111. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_detect.py +0 -0
  112. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_dispatch.py +0 -0
  113. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_e2e_connected.py +0 -0
  114. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_edit_replace_all.py +0 -0
  115. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_effort_governor.py +0 -0
  116. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_empty_response.py +0 -0
  117. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_events.py +0 -0
  118. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_extract.py +0 -0
  119. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_failure_loop.py +0 -0
  120. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_fiar.py +0 -0
  121. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_first_run_setup.py +0 -0
  122. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_gittools.py +0 -0
  123. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_glob_edit_edges.py +0 -0
  124. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_graphify_example.py +0 -0
  125. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_graphrag.py +0 -0
  126. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_graphrag_quoted_path.py +0 -0
  127. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_graphrag_sqlite.py +0 -0
  128. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_grep_and_read_robust.py +0 -0
  129. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_guards_and_tools.py +0 -0
  130. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_hallucinated_tools.py +0 -0
  131. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_leaked_tool_call.py +0 -0
  132. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_loop_detect.py +0 -0
  133. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_mcp.py +0 -0
  134. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_model_registry.py +0 -0
  135. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_oneshot_unreachable.py +0 -0
  136. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_overthink_interrupt.py +0 -0
  137. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_phases.py +0 -0
  138. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_plan_autocontinue.py +0 -0
  139. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_poam.py +0 -0
  140. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_progress.py +0 -0
  141. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_providers_unreachable.py +0 -0
  142. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_read_index.py +0 -0
  143. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_recipes.py +0 -0
  144. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_recovery.py +0 -0
  145. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_recovery_config.py +0 -0
  146. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_recovery_integration.py +0 -0
  147. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_repeated_outcome.py +0 -0
  148. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_repetition_interrupt.py +0 -0
  149. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_resume_events.py +0 -0
  150. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_resume_restore.py +0 -0
  151. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_resume_snapshot.py +0 -0
  152. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_rmf.py +0 -0
  153. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_rmf_graph.py +0 -0
  154. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_rmf_stig_graph.py +0 -0
  155. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_rolling_plan.py +0 -0
  156. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_runaway_repetition.py +0 -0
  157. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_screenshot.py +0 -0
  158. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_server_probe.py +0 -0
  159. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_sqlite_events.py +0 -0
  160. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_stall_retry.py +0 -0
  161. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_stig.py +0 -0
  162. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_stop.py +0 -0
  163. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_streaming_newlines.py +0 -0
  164. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_subagent.py +0 -0
  165. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_subagents.py +0 -0
  166. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_suggest.py +0 -0
  167. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_system_prompt_help.py +0 -0
  168. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_task_state.py +0 -0
  169. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_timeline.py +0 -0
  170. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_todo.py +0 -0
  171. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_tool_arg_coercion.py +0 -0
  172. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_tool_arg_coercion_more.py +0 -0
  173. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_tool_arg_parsing.py +0 -0
  174. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_tool_canonical.py +0 -0
  175. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_tool_policy.py +0 -0
  176. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_tool_result.py +0 -0
  177. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_tool_select.py +0 -0
  178. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_tool_validate.py +0 -0
  179. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_tools_undo.py +0 -0
  180. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_trajectory.py +0 -0
  181. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_tui.py +0 -0
  182. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_tuning.py +0 -0
  183. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_verification_gate.py +0 -0
  184. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_viewimage.py +0 -0
  185. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_vision_input.py +0 -0
  186. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_web_tools.py +0 -0
  187. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_windows_shell.py +0 -0
  188. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_worker_subagent.py +0 -0
  189. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_write_content_coerce.py +0 -0
  190. {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_xccdf.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: drydock-cli
3
- Version: 3.1.2
3
+ Version: 3.1.3
4
4
  Summary: Drydock — a local, provider-agnostic terminal coding agent for local LLMs
5
5
  Author: Frank Bobe III
6
6
  License-Expression: Apache-2.0
@@ -267,6 +267,47 @@ exports the open findings to a deterministic **eMASS POA&M CSV** — Control (fr
267
267
  the CCI map), Vulnerability Description, `POA&M Status=Ongoing`, Milestone (the
268
268
  Fix Text), and Severity (CAT I/II/III → High/Moderate/Low) — no LLM, stdlib only.
269
269
 
270
+ ### Document Canvas — editing documents far larger than the context window
271
+
272
+ Edit a 300-, 800-, 1000-page document the way you edit a large codebase: **search,
273
+ open a small region, apply a hash-guarded patch, validate, commit** — the model
274
+ never loads the whole document into context. Drydock parses the source into
275
+ addressable **blocks** with stable ids (`sec-0004`, `para-0182`, …) and content
276
+ hashes, and gives the model these tools:
277
+
278
+ | Tool | What it does |
279
+ |---|---|
280
+ | `DocOpen` | Parse a document into the canvas and show its outline |
281
+ | `DocOutline` / `DocSearch` / `DocRead` | Navigate + read small windows (never the whole file) |
282
+ | `DocReplace` | Global search-and-replace across the whole document in one pass |
283
+ | `DocPatch` | Hash-guarded, transactional edit of a block (replace / insert / delete) |
284
+ | `DocRedact` | Permanently remove text and **verify** it can't be recovered ("red boxing") |
285
+ | `DocDiff` / `DocValidate` | Preview staged changes; run structural + phrase checks |
286
+ | `DocCommit` / `DocRollback` | Write the source (original kept as `<file>.orig`) or discard |
287
+
288
+ **You don't call these tools yourself — you describe the task in plain English** and
289
+ the model drives them (just like Read/Edit/Bash). It picks the canvas over
290
+ `Read`/`Edit` automatically for documents too big to hold in context. For example,
291
+ type into the TUI:
292
+
293
+ ```
294
+ Open report.md and change every "single-factor authentication" to
295
+ "phishing-resistant MFA", then validate and commit.
296
+
297
+ Find the definition of "serendipity" in dictionary.pdf.
298
+
299
+ Redact every SSN and phone number in disclosure.md, then commit a release copy.
300
+ ```
301
+
302
+ Or run the bundled skill: `/document-canvas report.md redact all phone numbers`.
303
+
304
+ **Formats:** `.md` / `.markdown` / `.txt` are edited in place. `.pdf` / `.docx` are
305
+ imported read-only (text extracted; edits written to a `<file>.canvas.md` sidecar so
306
+ the binary original is never overwritten). Every edit is **staged** in a working copy
307
+ and hash-guarded, so the model can never clobber stale content; `DocCommit` writes the
308
+ file and preserves the untouched original as `<file>.orig`. Redaction is the one
309
+ **blocking** check — it refuses if the removed text would still be recoverable.
310
+
270
311
  ## Install
271
312
 
272
313
  ```bash
@@ -243,6 +243,47 @@ exports the open findings to a deterministic **eMASS POA&M CSV** — Control (fr
243
243
  the CCI map), Vulnerability Description, `POA&M Status=Ongoing`, Milestone (the
244
244
  Fix Text), and Severity (CAT I/II/III → High/Moderate/Low) — no LLM, stdlib only.
245
245
 
246
+ ### Document Canvas — editing documents far larger than the context window
247
+
248
+ Edit a 300-, 800-, 1000-page document the way you edit a large codebase: **search,
249
+ open a small region, apply a hash-guarded patch, validate, commit** — the model
250
+ never loads the whole document into context. Drydock parses the source into
251
+ addressable **blocks** with stable ids (`sec-0004`, `para-0182`, …) and content
252
+ hashes, and gives the model these tools:
253
+
254
+ | Tool | What it does |
255
+ |---|---|
256
+ | `DocOpen` | Parse a document into the canvas and show its outline |
257
+ | `DocOutline` / `DocSearch` / `DocRead` | Navigate + read small windows (never the whole file) |
258
+ | `DocReplace` | Global search-and-replace across the whole document in one pass |
259
+ | `DocPatch` | Hash-guarded, transactional edit of a block (replace / insert / delete) |
260
+ | `DocRedact` | Permanently remove text and **verify** it can't be recovered ("red boxing") |
261
+ | `DocDiff` / `DocValidate` | Preview staged changes; run structural + phrase checks |
262
+ | `DocCommit` / `DocRollback` | Write the source (original kept as `<file>.orig`) or discard |
263
+
264
+ **You don't call these tools yourself — you describe the task in plain English** and
265
+ the model drives them (just like Read/Edit/Bash). It picks the canvas over
266
+ `Read`/`Edit` automatically for documents too big to hold in context. For example,
267
+ type into the TUI:
268
+
269
+ ```
270
+ Open report.md and change every "single-factor authentication" to
271
+ "phishing-resistant MFA", then validate and commit.
272
+
273
+ Find the definition of "serendipity" in dictionary.pdf.
274
+
275
+ Redact every SSN and phone number in disclosure.md, then commit a release copy.
276
+ ```
277
+
278
+ Or run the bundled skill: `/document-canvas report.md redact all phone numbers`.
279
+
280
+ **Formats:** `.md` / `.markdown` / `.txt` are edited in place. `.pdf` / `.docx` are
281
+ imported read-only (text extracted; edits written to a `<file>.canvas.md` sidecar so
282
+ the binary original is never overwritten). Every edit is **staged** in a working copy
283
+ and hash-guarded, so the model can never clobber stale content; `DocCommit` writes the
284
+ file and preserves the untouched original as `<file>.orig`. Redaction is the one
285
+ **blocking** check — it refuses if the removed text would still be recoverable.
286
+
246
287
  ## Install
247
288
 
248
289
  ```bash
@@ -0,0 +1,40 @@
1
+ ---
2
+ name: document-canvas
3
+ description: Edit a large document (md/txt) via the Document Canvas — search, patch, validate, commit, without loading it all
4
+ ---
5
+ Edit the document as a CANVAS — never load the whole thing into context. Target
6
+ for this task: $ARGS
7
+
8
+ Work like a programmer editing a large codebase. Address blocks by their stable
9
+ id (para-0182, sec-0004, …), not by line numbers or page numbers.
10
+
11
+ 1. **Open.** Call DocOpen with the file path. Read the outline it returns to
12
+ understand the structure. (The document name is the filename.)
13
+
14
+ 2. **Locate — don't read everything.** Use DocSearch to find the blocks relevant
15
+ to the change (exact text, or regex:true for patterns like requirement numbers).
16
+ Use DocOutline to navigate the hierarchy. Only the blocks you open enter context.
17
+
18
+ 3. **Read the target region.** Call DocRead on each candidate block with a small
19
+ `before`/`after` window so you understand its context. Note each target's
20
+ content **hash** — you must pass it back as `expected_hash`.
21
+
22
+ 4. **Patch — hash-guarded, staged.** Call DocPatch with op (replace / insert_before
23
+ / insert_after / delete), target_id, the `expected_hash` from DocRead, and
24
+ new_text plus a short `reason`. Patches are STAGED in a working copy — the source
25
+ file is NOT changed yet. If a hash is stale the patch is rejected; re-DocRead and
26
+ retry. For several related edits, pass a `patches` list so they apply atomically.
27
+
28
+ 5. **Preview.** Call DocDiff to see every staged change as a unified diff. Confirm
29
+ you changed only what you intended and did not over-edit neighbouring text.
30
+
31
+ 6. **Validate.** Call DocValidate (pass `prohibited` for terms that must be gone and
32
+ `required` for terms that must remain). Resolve any ⚠ findings before committing.
33
+
34
+ 7. **Commit.** Only when the diff and validation look right, call DocCommit. It
35
+ writes the source file and preserves the untouched original as <file>.orig. Use
36
+ DocRollback to discard staged edits instead.
37
+
38
+ For a global change (e.g. a terminology update), search for ALL occurrences, then
39
+ process them in small batches — you do not need every passage in context at once.
40
+ Report which blocks you changed and the final validation result.
@@ -0,0 +1,525 @@
1
+ """Document Canvas — random-access, structure-aware editing of large documents.
2
+
3
+ Design (clean-room, stdlib only — no third-party deps, no embeddings). The
4
+ model treats a big document like a codebase: get an outline, search, open a
5
+ small region, apply a HASH-GUARDED patch inside a transaction, validate, and
6
+ commit. It never holds the whole document.
7
+
8
+ Core ideas
9
+ • A source file (Markdown or plain text for this MVP) is parsed once into a
10
+ CANONICAL MODEL: an ordered list of addressable BLOCKS (heading, paragraph,
11
+ code, table, list item, blank). Every block gets a PERMANENT id
12
+ (``sec-0001``/``para-0004``/…) and a ``content_hash`` (sha256 prefix).
13
+ • The model is persisted as JSON under ``~/.drydock/doccanvas/<name>.json`` so
14
+ ids and hashes survive across turns and edits. Page numbers are navigation
15
+ hints only; ids are the stable address.
16
+ • Edits are STRUCTURED PATCHES against ids, guarded by ``expected_hash`` so the
17
+ model can never edit stale content. Patches run in a TRANSACTION: stage →
18
+ validate → preview diff → commit or rollback. Commit re-renders the blocks
19
+ back to the source format and writes it, preserving the original as an
20
+ immutable ``.orig`` artifact on first commit.
21
+
22
+ This module is pure logic (no TUI, no tool wiring); ``drydock/tools`` exposes it
23
+ as Doc* tools and the audit rides on ``drydock.events``.
24
+ """
25
+ from __future__ import annotations
26
+
27
+ import hashlib
28
+ import json
29
+ import re
30
+ from dataclasses import asdict, dataclass, field
31
+ from pathlib import Path
32
+
33
+ STORE_DIR = Path.home() / ".drydock" / "doccanvas"
34
+
35
+ _HEADING = re.compile(r"^(#{1,6})\s+(.*)$")
36
+ _TABLE_ROW = re.compile(r"^\s*\|.*\|\s*$")
37
+ _LIST_ITEM = re.compile(r"^\s*([-*+]|\d+[.)])\s+\S")
38
+ _FENCE = re.compile(r"^\s*(```|~~~)")
39
+
40
+
41
+ def _hash(text: str) -> str:
42
+ return hashlib.sha256(text.encode("utf-8", "replace")).hexdigest()[:12]
43
+
44
+
45
+ @dataclass
46
+ class Block:
47
+ """One addressable document object."""
48
+ id: str
49
+ type: str # heading|paragraph|code|table|list|blank
50
+ text: str
51
+ content_hash: str = ""
52
+ level: int = 0 # heading depth (1-6); 0 otherwise
53
+ meta: dict = field(default_factory=dict)
54
+
55
+ def __post_init__(self):
56
+ if not self.content_hash:
57
+ self.content_hash = _hash(self.text)
58
+
59
+ def rehash(self) -> None:
60
+ self.content_hash = _hash(self.text)
61
+
62
+
63
+ @dataclass
64
+ class Document:
65
+ """Canonical model: source pointer + ordered blocks + id counters."""
66
+ source_path: str
67
+ fmt: str # markdown|text
68
+ blocks: list = field(default_factory=list) # list[Block]
69
+ counters: dict = field(default_factory=dict) # type -> last integer used
70
+ source_hash: str = "" # hash of the source file at parse time
71
+ committed: bool = False # has an .orig artifact been written yet
72
+ redactions: list = field(default_factory=list) # audit of redactions applied
73
+ meta: dict = field(default_factory=dict) # e.g. imported_from for pdf/docx
74
+
75
+ # ---- id allocation -------------------------------------------------
76
+ _PREFIX = {"heading": "sec", "paragraph": "para", "code": "code",
77
+ "table": "table", "list": "list", "blank": "blank"}
78
+
79
+ def _new_id(self, btype: str) -> str:
80
+ n = self.counters.get(btype, 0) + 1
81
+ self.counters[btype] = n
82
+ return f"{self._PREFIX.get(btype, 'blk')}-{n:04d}"
83
+
84
+ # ---- lookups -------------------------------------------------------
85
+ def index_of(self, block_id: str) -> int:
86
+ for i, b in enumerate(self.blocks):
87
+ if b.id == block_id:
88
+ return i
89
+ return -1
90
+
91
+ def get(self, block_id: str):
92
+ i = self.index_of(block_id)
93
+ return self.blocks[i] if i >= 0 else None
94
+
95
+
96
+ # ---------------------------------------------------------------------------
97
+ # Parsing (source -> canonical model)
98
+ # ---------------------------------------------------------------------------
99
+ def _classify_para(text: str) -> str:
100
+ lines = [ln for ln in text.splitlines() if ln.strip()]
101
+ if lines and all(_TABLE_ROW.match(ln) for ln in lines):
102
+ return "table"
103
+ if lines and all(_LIST_ITEM.match(ln) for ln in lines):
104
+ return "list"
105
+ return "paragraph"
106
+
107
+
108
+ def parse(text: str, fmt: str, source_path: str) -> Document:
109
+ """Parse raw text into a Document. Markdown recognises headings, fenced
110
+ code, tables and lists; plain text splits on blank lines only."""
111
+ doc = Document(source_path=source_path, fmt=fmt, source_hash=_hash(text))
112
+ lines = text.split("\n")
113
+ i, n = 0, len(lines)
114
+ buf: list[str] = []
115
+
116
+ def flush_para():
117
+ if not buf:
118
+ return
119
+ raw = "\n".join(buf).strip("\n")
120
+ buf.clear()
121
+ if not raw.strip():
122
+ return
123
+ btype = _classify_para(raw) if fmt == "markdown" else "paragraph"
124
+ doc.blocks.append(Block(id=doc._new_id(btype), type=btype, text=raw))
125
+
126
+ while i < n:
127
+ line = lines[i]
128
+ if fmt == "markdown":
129
+ m = _HEADING.match(line)
130
+ if m:
131
+ flush_para()
132
+ doc.blocks.append(Block(id=doc._new_id("heading"), type="heading",
133
+ text=line.rstrip(), level=len(m.group(1))))
134
+ i += 1
135
+ continue
136
+ fm = _FENCE.match(line)
137
+ if fm:
138
+ flush_para()
139
+ fence = line.strip()[:3]
140
+ code = [line]
141
+ i += 1
142
+ while i < n:
143
+ code.append(lines[i])
144
+ if lines[i].strip().startswith(fence):
145
+ i += 1
146
+ break
147
+ i += 1
148
+ doc.blocks.append(Block(id=doc._new_id("code"), type="code",
149
+ text="\n".join(code)))
150
+ continue
151
+ if not line.strip():
152
+ flush_para()
153
+ else:
154
+ buf.append(line)
155
+ i += 1
156
+ flush_para()
157
+ return doc
158
+
159
+
160
+ # ---------------------------------------------------------------------------
161
+ # Rendering (canonical model -> source text)
162
+ # ---------------------------------------------------------------------------
163
+ def render(doc: Document) -> str:
164
+ """Serialise blocks back to source text. Blocks are separated by a blank
165
+ line (the natural Markdown/prose separator); code and headings render as-is."""
166
+ return "\n\n".join(b.text for b in doc.blocks) + "\n"
167
+
168
+
169
+ # ---------------------------------------------------------------------------
170
+ # Persistence
171
+ # ---------------------------------------------------------------------------
172
+ def _store_path(name: str) -> Path:
173
+ safe = re.sub(r"[^A-Za-z0-9._-]", "_", name)
174
+ return STORE_DIR / f"{safe}.json"
175
+
176
+
177
+ def store_name(source_path: str) -> str:
178
+ return Path(source_path).name
179
+
180
+
181
+ def save(doc: Document) -> Path:
182
+ STORE_DIR.mkdir(parents=True, exist_ok=True)
183
+ p = _store_path(store_name(doc.source_path))
184
+ data = asdict(doc)
185
+ p.write_text(json.dumps(data, ensure_ascii=False, indent=1), encoding="utf-8")
186
+ return p
187
+
188
+
189
+ def load(name: str) -> Document | None:
190
+ p = _store_path(name)
191
+ if not p.exists():
192
+ return None
193
+ d = json.loads(p.read_text(encoding="utf-8"))
194
+ blocks = [Block(**b) for b in d.pop("blocks", [])]
195
+ doc = Document(**d)
196
+ doc.blocks = blocks
197
+ return doc
198
+
199
+
200
+ _IMPORT_EXT = (".pdf", ".docx")
201
+
202
+
203
+ def open_document(source_path: str, reparse: bool = False) -> Document:
204
+ """Open a source file into a canonical model. Reuses a persisted store if
205
+ the source is unchanged (preserving ids); reparses on change or reparse=1.
206
+
207
+ .md/.markdown/.txt are edited IN PLACE. .pdf/.docx are IMPORTED read-only
208
+ (text extracted via drydock.extract): edits go to a Markdown sidecar
209
+ ``<source>.canvas.md`` and the binary original is never overwritten."""
210
+ path = Path(source_path).expanduser()
211
+ if not path.exists():
212
+ raise FileNotFoundError(source_path)
213
+ ext = path.suffix.lower()
214
+
215
+ if ext in _IMPORT_EXT:
216
+ from drydock import extract
217
+ if ext == ".pdf" and not extract.pdf_backend_available():
218
+ raise RuntimeError(
219
+ "PDF import needs a backend: install `pdftotext` (poppler) or "
220
+ "`pip install drydock-cli[pdf]`.")
221
+ text = extract.extract_document(path)
222
+ if not text:
223
+ raise RuntimeError(f"could not extract any text from {path.name}")
224
+ edit_path = path.with_name(path.name + ".canvas.md")
225
+ name = edit_path.name
226
+ existing = load(name)
227
+ if existing and not reparse and existing.source_hash == _hash(text):
228
+ return existing
229
+ doc = parse(text, "markdown", str(edit_path))
230
+ doc.meta = {"imported_from": str(path), "import_ext": ext}
231
+ save(doc)
232
+ return doc
233
+
234
+ text = path.read_text(encoding="utf-8", errors="replace")
235
+ fmt = "markdown" if ext in (".md", ".markdown") else "text"
236
+ existing = load(path.name)
237
+ if existing and not reparse and existing.source_hash == _hash(text):
238
+ return existing
239
+ doc = parse(text, fmt, str(path))
240
+ save(doc)
241
+ return doc
242
+
243
+
244
+ # ---------------------------------------------------------------------------
245
+ # Read / navigate
246
+ # ---------------------------------------------------------------------------
247
+ def outline(doc: Document) -> list:
248
+ """Heading hierarchy with the id and a running count of child blocks."""
249
+ out = []
250
+ for i, b in enumerate(doc.blocks):
251
+ if b.type == "heading":
252
+ title = _HEADING.match(b.text)
253
+ out.append({"id": b.id, "level": b.level,
254
+ "title": title.group(2) if title else b.text})
255
+ return out
256
+
257
+
258
+ def search(doc: Document, query: str, regex: bool = False,
259
+ max_hits: int = 100, ignore_case: bool = True) -> list:
260
+ """Lexical search over block text. Returns id/type/snippet per hit.
261
+ Case-insensitive by default (find loosely); pass ignore_case=False to match
262
+ exactly — the same casing DocReplace/DocRedact use by default."""
263
+ hits = []
264
+ flags = re.IGNORECASE if ignore_case else 0
265
+ try:
266
+ pat = re.compile(query if regex else re.escape(query), flags)
267
+ except re.error as e:
268
+ return [{"error": f"bad regex: {e}"}]
269
+ for b in doc.blocks:
270
+ m = pat.search(b.text)
271
+ if m:
272
+ s = max(0, m.start() - 40)
273
+ snippet = b.text[s:m.end() + 40].replace("\n", " ")
274
+ hits.append({"id": b.id, "type": b.type, "snippet": snippet})
275
+ if len(hits) >= max_hits:
276
+ break
277
+ return hits
278
+
279
+
280
+ def read(doc: Document, block_id: str, before: int = 0, after: int = 0) -> dict:
281
+ """Return a block plus N neighbours on each side (window into the doc)."""
282
+ i = doc.index_of(block_id)
283
+ if i < 0:
284
+ return {"error": f"no such block: {block_id}"}
285
+ lo, hi = max(0, i - before), min(len(doc.blocks), i + after + 1)
286
+ window = [{"id": b.id, "type": b.type, "hash": b.content_hash,
287
+ "text": b.text, "target": b.id == block_id}
288
+ for b in doc.blocks[lo:hi]]
289
+ return {"target": block_id, "window": window}
290
+
291
+
292
+ # ---------------------------------------------------------------------------
293
+ # Transactional patching
294
+ # ---------------------------------------------------------------------------
295
+ class PatchError(Exception):
296
+ pass
297
+
298
+
299
+ _OPS = {"replace", "insert_before", "insert_after", "delete"}
300
+
301
+
302
+ def apply_patches(doc: Document, patches: list) -> list:
303
+ """Validate + apply a list of patches to `doc` in order, atomically: if ANY
304
+ patch fails validation the document is left untouched and PatchError is
305
+ raised. Each patch: {op, target_id, expected_hash?, new_text?, reason?}.
306
+
307
+ Returns a per-patch changelog (for the audit trail + diff)."""
308
+ # 1) validate everything first (atomic)
309
+ for p in patches:
310
+ op = p.get("op")
311
+ if op not in _OPS:
312
+ raise PatchError(f"unknown op: {op!r} (use {sorted(_OPS)})")
313
+ tid = p.get("target_id")
314
+ idx = doc.index_of(tid) if tid else -1
315
+ if idx < 0:
316
+ raise PatchError(f"target_id not found: {tid!r}")
317
+ exp = p.get("expected_hash")
318
+ if exp and doc.blocks[idx].content_hash != exp:
319
+ raise PatchError(
320
+ f"stale content for {tid}: expected_hash {exp} but current is "
321
+ f"{doc.blocks[idx].content_hash}. Re-read the block before patching.")
322
+ if op in ("replace", "insert_before", "insert_after") and not p.get("new_text", "").strip():
323
+ raise PatchError(f"{op} on {tid} needs non-empty new_text")
324
+
325
+ # 2) apply (target ids are resolved fresh each step; inserts shift indices)
326
+ changelog = []
327
+ for p in patches:
328
+ op, tid = p["op"], p["target_id"]
329
+ idx = doc.index_of(tid)
330
+ if op == "replace":
331
+ old = doc.blocks[idx].text
332
+ doc.blocks[idx].text = p["new_text"]
333
+ doc.blocks[idx].rehash()
334
+ changelog.append({"op": op, "id": tid, "old": old, "new": p["new_text"],
335
+ "reason": p.get("reason", "")})
336
+ elif op == "delete":
337
+ old = doc.blocks.pop(idx)
338
+ changelog.append({"op": op, "id": tid, "old": old.text, "new": "",
339
+ "reason": p.get("reason", "")})
340
+ else: # insert_before / insert_after
341
+ btype = p.get("block_type", "paragraph")
342
+ nb = Block(id=doc._new_id(btype), type=btype, text=p["new_text"],
343
+ level=int(p.get("level", 0)))
344
+ at = idx if op == "insert_before" else idx + 1
345
+ doc.blocks.insert(at, nb)
346
+ changelog.append({"op": op, "id": nb.id, "anchor": tid, "old": "",
347
+ "new": nb.text, "reason": p.get("reason", "")})
348
+ return changelog
349
+
350
+
351
+ def diff_text(changelog: list) -> str:
352
+ """Human-readable unified-ish diff of a changelog."""
353
+ import difflib
354
+ out = []
355
+ for c in changelog:
356
+ head = f"@@ {c['op']} {c.get('id','')}"
357
+ if c.get("anchor"):
358
+ head += f" (anchor {c['anchor']})"
359
+ if c.get("reason"):
360
+ head += f" — {c['reason']}"
361
+ out.append(head)
362
+ old = (c.get("old") or "").splitlines()
363
+ new = (c.get("new") or "").splitlines()
364
+ for line in difflib.unified_diff(old, new, lineterm="", n=1):
365
+ if line.startswith(("---", "+++", "@@")):
366
+ continue
367
+ out.append(line)
368
+ return "\n".join(out)
369
+
370
+
371
+ # ---------------------------------------------------------------------------
372
+ # Bulk search-and-update (one O(n) pass — the scalable primitive for global
373
+ # changes on a large document, instead of one patch call per occurrence)
374
+ # ---------------------------------------------------------------------------
375
+ def replace_all(doc: Document, query: str, replacement: str,
376
+ regex: bool = False, ignore_case: bool = False) -> dict:
377
+ """Replace EVERY occurrence of `query` across all blocks in a single pass.
378
+ Case-SENSITIVE by default (precise; preserves other casings); pass
379
+ ignore_case=True to replace all case variants. Returns a count + touched ids."""
380
+ flags = re.IGNORECASE if ignore_case else 0
381
+ try:
382
+ pat = re.compile(query if regex else re.escape(query), flags)
383
+ except re.error as e:
384
+ raise PatchError(f"bad regex: {e}")
385
+ count = 0
386
+ touched = []
387
+ for b in doc.blocks:
388
+ new, n = pat.subn(replacement, b.text)
389
+ if n:
390
+ b.text = new
391
+ b.rehash()
392
+ count += n
393
+ touched.append(b.id)
394
+ return {"replacements": count, "blocks_touched": touched}
395
+
396
+
397
+ # ---------------------------------------------------------------------------
398
+ # Redaction ("red boxing") — PERMANENTLY remove text, then VERIFY it cannot be
399
+ # recovered from the document. Verification is BLOCKING (unlike the advisory
400
+ # validators): a failed check aborts and the store is left unchanged.
401
+ # ---------------------------------------------------------------------------
402
+ REDACT_MARKER = "[REDACTED]"
403
+
404
+
405
+ def redact(doc: Document, query: str | None = None, block_id: str | None = None,
406
+ regex: bool = False, marker: str = REDACT_MARKER,
407
+ ignore_case: bool = True) -> dict:
408
+ """Redact by whole block (block_id) or by matched text (query). The removed
409
+ text is permanently replaced by `marker`; only a HASH of it is recorded (the
410
+ store never keeps the sensitive text). Then verifies none of the removed
411
+ strings survive in the rendered document — raises PatchError if any do.
412
+ Case-INSENSITIVE by default so every casing of a secret is caught."""
413
+ removed: list[str] = []
414
+ if block_id:
415
+ b = doc.get(block_id)
416
+ if b is None:
417
+ raise PatchError(f"no such block: {block_id}")
418
+ removed.append(b.text)
419
+ b.text = marker
420
+ b.rehash()
421
+ count = 1
422
+ elif query:
423
+ if not regex and query in marker:
424
+ raise PatchError("redaction marker contains the query — pick a different marker")
425
+ try:
426
+ pat = re.compile(query if regex else re.escape(query),
427
+ re.IGNORECASE if ignore_case else 0)
428
+ except re.error as e:
429
+ raise PatchError(f"bad regex: {e}")
430
+ count = 0
431
+ for b in doc.blocks:
432
+ def _sub(m):
433
+ nonlocal count
434
+ removed.append(m.group(0))
435
+ count += 1
436
+ return marker
437
+ b.text = pat.subn(_sub, b.text)[0]
438
+ b.rehash()
439
+ if count == 0:
440
+ raise PatchError(f"nothing matched {query!r} — nothing redacted")
441
+ else:
442
+ raise PatchError("redact needs a block_id or a query")
443
+
444
+ # BLOCKING verification: no removed string may still be recoverable.
445
+ full = render(doc)
446
+ leaked = sorted({s for s in removed if s and s != marker and s in full})
447
+ if leaked:
448
+ raise PatchError(
449
+ f"redaction verification FAILED: {len(leaked)} removed string(s) still "
450
+ "recoverable — redaction aborted, document unchanged")
451
+ doc.redactions.append({"mode": "block" if block_id else "query",
452
+ "target": block_id or query, "count": count,
453
+ "content_hash": _hash("\x00".join(removed)), "marker": marker})
454
+ return {"redacted": count, "verified": True, "marker": marker}
455
+
456
+
457
+ # ---------------------------------------------------------------------------
458
+ # Validation (deterministic, run before any LLM review)
459
+ # ---------------------------------------------------------------------------
460
+ def validate(doc: Document, required: list | None = None,
461
+ prohibited: list | None = None) -> dict:
462
+ """Structural + content checks. Advisory: returns pass/warn findings; callers
463
+ decide whether to block (redaction is the only hard-blocking case, elsewhere)."""
464
+ findings = []
465
+ ids = [b.id for b in doc.blocks]
466
+ dupes = {x for x in ids if ids.count(x) > 1}
467
+ findings.append({"check": "unique_ids", "ok": not dupes,
468
+ "detail": f"duplicate ids: {sorted(dupes)}" if dupes else ""})
469
+ bad_hash = [b.id for b in doc.blocks if b.content_hash != _hash(b.text)]
470
+ findings.append({"check": "hashes_consistent", "ok": not bad_hash,
471
+ "detail": f"stale hashes: {bad_hash}" if bad_hash else ""})
472
+ empty = [b.id for b in doc.blocks if not b.text.strip()]
473
+ findings.append({"check": "no_empty_blocks", "ok": not empty,
474
+ "detail": f"empty blocks: {empty}" if empty else ""})
475
+ full = render(doc)
476
+ for phrase in (required or []):
477
+ findings.append({"check": f"required:{phrase!r}", "ok": phrase in full,
478
+ "detail": "" if phrase in full else "missing"})
479
+ for phrase in (prohibited or []):
480
+ findings.append({"check": f"prohibited:{phrase!r}", "ok": phrase not in full,
481
+ "detail": "" if phrase not in full else "still present"})
482
+ findings.append({"check": "heading_levels_monotonic",
483
+ "ok": _levels_ok(doc), "detail": _levels_detail(doc)})
484
+ return {"ok": all(f["ok"] for f in findings), "findings": findings}
485
+
486
+
487
+ def _levels_ok(doc: Document) -> bool:
488
+ prev = 0
489
+ for b in doc.blocks:
490
+ if b.type == "heading":
491
+ if prev and b.level > prev + 1:
492
+ return False
493
+ prev = b.level
494
+ return True
495
+
496
+
497
+ def _levels_detail(doc: Document) -> str:
498
+ prev = 0
499
+ for b in doc.blocks:
500
+ if b.type == "heading":
501
+ if prev and b.level > prev + 1:
502
+ return f"{b.id} jumps from h{prev} to h{b.level}"
503
+ prev = b.level
504
+ return ""
505
+
506
+
507
+ # ---------------------------------------------------------------------------
508
+ # Commit (canonical model -> source file, original preserved)
509
+ # ---------------------------------------------------------------------------
510
+ def commit(doc: Document) -> dict:
511
+ """Render the canonical model back to the source file. On the first commit,
512
+ the untouched original is preserved as ``<source>.orig`` (immutable artifact)."""
513
+ src = Path(doc.source_path)
514
+ if not doc.committed:
515
+ orig = src.with_suffix(src.suffix + ".orig")
516
+ if src.exists() and not orig.exists():
517
+ orig.write_text(src.read_text(encoding="utf-8", errors="replace"),
518
+ encoding="utf-8")
519
+ doc.committed = True
520
+ text = render(doc)
521
+ src.write_text(text, encoding="utf-8")
522
+ doc.source_hash = _hash(text)
523
+ save(doc)
524
+ return {"source": str(src), "blocks": len(doc.blocks),
525
+ "original": str(src.with_suffix(src.suffix + ".orig"))}