drydock-cli 3.1.2__tar.gz → 3.1.3__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {drydock_cli-3.1.2/drydock_cli.egg-info → drydock_cli-3.1.3}/PKG-INFO +42 -1
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/README.md +41 -0
- drydock_cli-3.1.3/drydock/builtin_skills/document-canvas.md +40 -0
- drydock_cli-3.1.3/drydock/doccanvas.py +525 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/tools/__init__.py +376 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/tui/widgets.py +1 -1
- {drydock_cli-3.1.2 → drydock_cli-3.1.3/drydock_cli.egg-info}/PKG-INFO +42 -1
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock_cli.egg-info/SOURCES.txt +3 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/pyproject.toml +1 -1
- drydock_cli-3.1.3/tests/test_doccanvas.py +252 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_skills.py +1 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/LICENSE +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/NOTICE +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/__init__.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/__main__.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/advisor.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/agent.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/bash_safety.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/budget.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/builtin_skills/__init__.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/builtin_skills/fiar-assess.md +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/builtin_skills/fiar-cap.md +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/builtin_skills/fiar-evidence.md +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/builtin_skills/fiar-readiness.md +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/builtin_skills/ml-data.md +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/builtin_skills/ml-debug.md +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/builtin_skills/ml-finetune.md +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/builtin_skills/ml-metrics.md +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/builtin_skills/ml-rl.md +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/builtin_skills/ml-train.md +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/builtin_skills/nist-ai-rmf.md +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/builtin_skills/nist-csf.md +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/builtin_skills/rmf-categorize.md +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/builtin_skills/rmf-control.md +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/builtin_skills/rmf-poam.md +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/builtin_skills/rmf-review.md +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/builtin_skills/stig-assess.md +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/builtin_skills/stig-remediate.md +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/cci.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/cli.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/compaction.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/config.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/detect.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/events.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/extract.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/fiar.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/gittools.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/graphrag.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/guards.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/loop_detect.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/mcp.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/phases.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/poam.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/progress.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/providers.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/recipes.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/recovery.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/resume.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/rmf.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/rmf_graph.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/skills.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/stig.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/subagents.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/suggest.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/task_state.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/tool_policy.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/tool_registry.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/tool_result.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/tool_select.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/tool_validate.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/trajectory.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/tui/__init__.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/tui/app.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/tui/approval.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/tui/messages.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/tuning.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/verification.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock/web.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock_cli.egg-info/dependency_links.txt +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock_cli.egg-info/entry_points.txt +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock_cli.egg-info/requires.txt +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/drydock_cli.egg-info/top_level.txt +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/setup.cfg +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_advisor.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_approval.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_back_command.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_bash_background.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_bash_binary_output.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_bash_crossplatform.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_bash_output_bounding.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_bash_process_group.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_bash_safety.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_bash_sanitize.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_bash_shell.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_bash_stdin.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_bash_stop_partial.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_bash_timeout_network.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_bash_timeout_param.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_budget.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_cci.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_cli_agents.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_compact_command.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_compaction.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_config.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_config_migration.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_context_limit_config.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_context_limit_issue25.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_criteria_coverage.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_cycling_detection.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_degenerate_argument.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_detect.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_dispatch.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_e2e_connected.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_edit_replace_all.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_effort_governor.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_empty_response.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_events.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_extract.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_failure_loop.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_fiar.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_first_run_setup.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_gittools.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_glob_edit_edges.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_graphify_example.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_graphrag.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_graphrag_quoted_path.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_graphrag_sqlite.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_grep_and_read_robust.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_guards_and_tools.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_hallucinated_tools.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_leaked_tool_call.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_loop_detect.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_mcp.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_model_registry.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_oneshot_unreachable.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_overthink_interrupt.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_phases.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_plan_autocontinue.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_poam.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_progress.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_providers_unreachable.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_read_index.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_recipes.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_recovery.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_recovery_config.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_recovery_integration.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_repeated_outcome.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_repetition_interrupt.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_resume_events.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_resume_restore.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_resume_snapshot.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_rmf.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_rmf_graph.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_rmf_stig_graph.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_rolling_plan.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_runaway_repetition.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_screenshot.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_server_probe.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_sqlite_events.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_stall_retry.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_stig.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_stop.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_streaming_newlines.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_subagent.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_subagents.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_suggest.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_system_prompt_help.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_task_state.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_timeline.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_todo.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_tool_arg_coercion.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_tool_arg_coercion_more.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_tool_arg_parsing.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_tool_canonical.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_tool_policy.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_tool_result.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_tool_select.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_tool_validate.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_tools_undo.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_trajectory.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_tui.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_tuning.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_verification_gate.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_viewimage.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_vision_input.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_web_tools.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_windows_shell.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_worker_subagent.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_write_content_coerce.py +0 -0
- {drydock_cli-3.1.2 → drydock_cli-3.1.3}/tests/test_xccdf.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: drydock-cli
|
|
3
|
-
Version: 3.1.
|
|
3
|
+
Version: 3.1.3
|
|
4
4
|
Summary: Drydock — a local, provider-agnostic terminal coding agent for local LLMs
|
|
5
5
|
Author: Frank Bobe III
|
|
6
6
|
License-Expression: Apache-2.0
|
|
@@ -267,6 +267,47 @@ exports the open findings to a deterministic **eMASS POA&M CSV** — Control (fr
|
|
|
267
267
|
the CCI map), Vulnerability Description, `POA&M Status=Ongoing`, Milestone (the
|
|
268
268
|
Fix Text), and Severity (CAT I/II/III → High/Moderate/Low) — no LLM, stdlib only.
|
|
269
269
|
|
|
270
|
+
### Document Canvas — editing documents far larger than the context window
|
|
271
|
+
|
|
272
|
+
Edit a 300-, 800-, 1000-page document the way you edit a large codebase: **search,
|
|
273
|
+
open a small region, apply a hash-guarded patch, validate, commit** — the model
|
|
274
|
+
never loads the whole document into context. Drydock parses the source into
|
|
275
|
+
addressable **blocks** with stable ids (`sec-0004`, `para-0182`, …) and content
|
|
276
|
+
hashes, and gives the model these tools:
|
|
277
|
+
|
|
278
|
+
| Tool | What it does |
|
|
279
|
+
|---|---|
|
|
280
|
+
| `DocOpen` | Parse a document into the canvas and show its outline |
|
|
281
|
+
| `DocOutline` / `DocSearch` / `DocRead` | Navigate + read small windows (never the whole file) |
|
|
282
|
+
| `DocReplace` | Global search-and-replace across the whole document in one pass |
|
|
283
|
+
| `DocPatch` | Hash-guarded, transactional edit of a block (replace / insert / delete) |
|
|
284
|
+
| `DocRedact` | Permanently remove text and **verify** it can't be recovered ("red boxing") |
|
|
285
|
+
| `DocDiff` / `DocValidate` | Preview staged changes; run structural + phrase checks |
|
|
286
|
+
| `DocCommit` / `DocRollback` | Write the source (original kept as `<file>.orig`) or discard |
|
|
287
|
+
|
|
288
|
+
**You don't call these tools yourself — you describe the task in plain English** and
|
|
289
|
+
the model drives them (just like Read/Edit/Bash). It picks the canvas over
|
|
290
|
+
`Read`/`Edit` automatically for documents too big to hold in context. For example,
|
|
291
|
+
type into the TUI:
|
|
292
|
+
|
|
293
|
+
```
|
|
294
|
+
Open report.md and change every "single-factor authentication" to
|
|
295
|
+
"phishing-resistant MFA", then validate and commit.
|
|
296
|
+
|
|
297
|
+
Find the definition of "serendipity" in dictionary.pdf.
|
|
298
|
+
|
|
299
|
+
Redact every SSN and phone number in disclosure.md, then commit a release copy.
|
|
300
|
+
```
|
|
301
|
+
|
|
302
|
+
Or run the bundled skill: `/document-canvas report.md redact all phone numbers`.
|
|
303
|
+
|
|
304
|
+
**Formats:** `.md` / `.markdown` / `.txt` are edited in place. `.pdf` / `.docx` are
|
|
305
|
+
imported read-only (text extracted; edits written to a `<file>.canvas.md` sidecar so
|
|
306
|
+
the binary original is never overwritten). Every edit is **staged** in a working copy
|
|
307
|
+
and hash-guarded, so the model can never clobber stale content; `DocCommit` writes the
|
|
308
|
+
file and preserves the untouched original as `<file>.orig`. Redaction is the one
|
|
309
|
+
**blocking** check — it refuses if the removed text would still be recoverable.
|
|
310
|
+
|
|
270
311
|
## Install
|
|
271
312
|
|
|
272
313
|
```bash
|
|
@@ -243,6 +243,47 @@ exports the open findings to a deterministic **eMASS POA&M CSV** — Control (fr
|
|
|
243
243
|
the CCI map), Vulnerability Description, `POA&M Status=Ongoing`, Milestone (the
|
|
244
244
|
Fix Text), and Severity (CAT I/II/III → High/Moderate/Low) — no LLM, stdlib only.
|
|
245
245
|
|
|
246
|
+
### Document Canvas — editing documents far larger than the context window
|
|
247
|
+
|
|
248
|
+
Edit a 300-, 800-, 1000-page document the way you edit a large codebase: **search,
|
|
249
|
+
open a small region, apply a hash-guarded patch, validate, commit** — the model
|
|
250
|
+
never loads the whole document into context. Drydock parses the source into
|
|
251
|
+
addressable **blocks** with stable ids (`sec-0004`, `para-0182`, …) and content
|
|
252
|
+
hashes, and gives the model these tools:
|
|
253
|
+
|
|
254
|
+
| Tool | What it does |
|
|
255
|
+
|---|---|
|
|
256
|
+
| `DocOpen` | Parse a document into the canvas and show its outline |
|
|
257
|
+
| `DocOutline` / `DocSearch` / `DocRead` | Navigate + read small windows (never the whole file) |
|
|
258
|
+
| `DocReplace` | Global search-and-replace across the whole document in one pass |
|
|
259
|
+
| `DocPatch` | Hash-guarded, transactional edit of a block (replace / insert / delete) |
|
|
260
|
+
| `DocRedact` | Permanently remove text and **verify** it can't be recovered ("red boxing") |
|
|
261
|
+
| `DocDiff` / `DocValidate` | Preview staged changes; run structural + phrase checks |
|
|
262
|
+
| `DocCommit` / `DocRollback` | Write the source (original kept as `<file>.orig`) or discard |
|
|
263
|
+
|
|
264
|
+
**You don't call these tools yourself — you describe the task in plain English** and
|
|
265
|
+
the model drives them (just like Read/Edit/Bash). It picks the canvas over
|
|
266
|
+
`Read`/`Edit` automatically for documents too big to hold in context. For example,
|
|
267
|
+
type into the TUI:
|
|
268
|
+
|
|
269
|
+
```
|
|
270
|
+
Open report.md and change every "single-factor authentication" to
|
|
271
|
+
"phishing-resistant MFA", then validate and commit.
|
|
272
|
+
|
|
273
|
+
Find the definition of "serendipity" in dictionary.pdf.
|
|
274
|
+
|
|
275
|
+
Redact every SSN and phone number in disclosure.md, then commit a release copy.
|
|
276
|
+
```
|
|
277
|
+
|
|
278
|
+
Or run the bundled skill: `/document-canvas report.md redact all phone numbers`.
|
|
279
|
+
|
|
280
|
+
**Formats:** `.md` / `.markdown` / `.txt` are edited in place. `.pdf` / `.docx` are
|
|
281
|
+
imported read-only (text extracted; edits written to a `<file>.canvas.md` sidecar so
|
|
282
|
+
the binary original is never overwritten). Every edit is **staged** in a working copy
|
|
283
|
+
and hash-guarded, so the model can never clobber stale content; `DocCommit` writes the
|
|
284
|
+
file and preserves the untouched original as `<file>.orig`. Redaction is the one
|
|
285
|
+
**blocking** check — it refuses if the removed text would still be recoverable.
|
|
286
|
+
|
|
246
287
|
## Install
|
|
247
288
|
|
|
248
289
|
```bash
|
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: document-canvas
|
|
3
|
+
description: Edit a large document (md/txt) via the Document Canvas — search, patch, validate, commit, without loading it all
|
|
4
|
+
---
|
|
5
|
+
Edit the document as a CANVAS — never load the whole thing into context. Target
|
|
6
|
+
for this task: $ARGS
|
|
7
|
+
|
|
8
|
+
Work like a programmer editing a large codebase. Address blocks by their stable
|
|
9
|
+
id (para-0182, sec-0004, …), not by line numbers or page numbers.
|
|
10
|
+
|
|
11
|
+
1. **Open.** Call DocOpen with the file path. Read the outline it returns to
|
|
12
|
+
understand the structure. (The document name is the filename.)
|
|
13
|
+
|
|
14
|
+
2. **Locate — don't read everything.** Use DocSearch to find the blocks relevant
|
|
15
|
+
to the change (exact text, or regex:true for patterns like requirement numbers).
|
|
16
|
+
Use DocOutline to navigate the hierarchy. Only the blocks you open enter context.
|
|
17
|
+
|
|
18
|
+
3. **Read the target region.** Call DocRead on each candidate block with a small
|
|
19
|
+
`before`/`after` window so you understand its context. Note each target's
|
|
20
|
+
content **hash** — you must pass it back as `expected_hash`.
|
|
21
|
+
|
|
22
|
+
4. **Patch — hash-guarded, staged.** Call DocPatch with op (replace / insert_before
|
|
23
|
+
/ insert_after / delete), target_id, the `expected_hash` from DocRead, and
|
|
24
|
+
new_text plus a short `reason`. Patches are STAGED in a working copy — the source
|
|
25
|
+
file is NOT changed yet. If a hash is stale the patch is rejected; re-DocRead and
|
|
26
|
+
retry. For several related edits, pass a `patches` list so they apply atomically.
|
|
27
|
+
|
|
28
|
+
5. **Preview.** Call DocDiff to see every staged change as a unified diff. Confirm
|
|
29
|
+
you changed only what you intended and did not over-edit neighbouring text.
|
|
30
|
+
|
|
31
|
+
6. **Validate.** Call DocValidate (pass `prohibited` for terms that must be gone and
|
|
32
|
+
`required` for terms that must remain). Resolve any ⚠ findings before committing.
|
|
33
|
+
|
|
34
|
+
7. **Commit.** Only when the diff and validation look right, call DocCommit. It
|
|
35
|
+
writes the source file and preserves the untouched original as <file>.orig. Use
|
|
36
|
+
DocRollback to discard staged edits instead.
|
|
37
|
+
|
|
38
|
+
For a global change (e.g. a terminology update), search for ALL occurrences, then
|
|
39
|
+
process them in small batches — you do not need every passage in context at once.
|
|
40
|
+
Report which blocks you changed and the final validation result.
|
|
@@ -0,0 +1,525 @@
|
|
|
1
|
+
"""Document Canvas — random-access, structure-aware editing of large documents.
|
|
2
|
+
|
|
3
|
+
Design (clean-room, stdlib only — no third-party deps, no embeddings). The
|
|
4
|
+
model treats a big document like a codebase: get an outline, search, open a
|
|
5
|
+
small region, apply a HASH-GUARDED patch inside a transaction, validate, and
|
|
6
|
+
commit. It never holds the whole document.
|
|
7
|
+
|
|
8
|
+
Core ideas
|
|
9
|
+
• A source file (Markdown or plain text for this MVP) is parsed once into a
|
|
10
|
+
CANONICAL MODEL: an ordered list of addressable BLOCKS (heading, paragraph,
|
|
11
|
+
code, table, list item, blank). Every block gets a PERMANENT id
|
|
12
|
+
(``sec-0001``/``para-0004``/…) and a ``content_hash`` (sha256 prefix).
|
|
13
|
+
• The model is persisted as JSON under ``~/.drydock/doccanvas/<name>.json`` so
|
|
14
|
+
ids and hashes survive across turns and edits. Page numbers are navigation
|
|
15
|
+
hints only; ids are the stable address.
|
|
16
|
+
• Edits are STRUCTURED PATCHES against ids, guarded by ``expected_hash`` so the
|
|
17
|
+
model can never edit stale content. Patches run in a TRANSACTION: stage →
|
|
18
|
+
validate → preview diff → commit or rollback. Commit re-renders the blocks
|
|
19
|
+
back to the source format and writes it, preserving the original as an
|
|
20
|
+
immutable ``.orig`` artifact on first commit.
|
|
21
|
+
|
|
22
|
+
This module is pure logic (no TUI, no tool wiring); ``drydock/tools`` exposes it
|
|
23
|
+
as Doc* tools and the audit rides on ``drydock.events``.
|
|
24
|
+
"""
|
|
25
|
+
from __future__ import annotations
|
|
26
|
+
|
|
27
|
+
import hashlib
|
|
28
|
+
import json
|
|
29
|
+
import re
|
|
30
|
+
from dataclasses import asdict, dataclass, field
|
|
31
|
+
from pathlib import Path
|
|
32
|
+
|
|
33
|
+
STORE_DIR = Path.home() / ".drydock" / "doccanvas"
|
|
34
|
+
|
|
35
|
+
_HEADING = re.compile(r"^(#{1,6})\s+(.*)$")
|
|
36
|
+
_TABLE_ROW = re.compile(r"^\s*\|.*\|\s*$")
|
|
37
|
+
_LIST_ITEM = re.compile(r"^\s*([-*+]|\d+[.)])\s+\S")
|
|
38
|
+
_FENCE = re.compile(r"^\s*(```|~~~)")
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def _hash(text: str) -> str:
|
|
42
|
+
return hashlib.sha256(text.encode("utf-8", "replace")).hexdigest()[:12]
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
@dataclass
|
|
46
|
+
class Block:
|
|
47
|
+
"""One addressable document object."""
|
|
48
|
+
id: str
|
|
49
|
+
type: str # heading|paragraph|code|table|list|blank
|
|
50
|
+
text: str
|
|
51
|
+
content_hash: str = ""
|
|
52
|
+
level: int = 0 # heading depth (1-6); 0 otherwise
|
|
53
|
+
meta: dict = field(default_factory=dict)
|
|
54
|
+
|
|
55
|
+
def __post_init__(self):
|
|
56
|
+
if not self.content_hash:
|
|
57
|
+
self.content_hash = _hash(self.text)
|
|
58
|
+
|
|
59
|
+
def rehash(self) -> None:
|
|
60
|
+
self.content_hash = _hash(self.text)
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
@dataclass
|
|
64
|
+
class Document:
|
|
65
|
+
"""Canonical model: source pointer + ordered blocks + id counters."""
|
|
66
|
+
source_path: str
|
|
67
|
+
fmt: str # markdown|text
|
|
68
|
+
blocks: list = field(default_factory=list) # list[Block]
|
|
69
|
+
counters: dict = field(default_factory=dict) # type -> last integer used
|
|
70
|
+
source_hash: str = "" # hash of the source file at parse time
|
|
71
|
+
committed: bool = False # has an .orig artifact been written yet
|
|
72
|
+
redactions: list = field(default_factory=list) # audit of redactions applied
|
|
73
|
+
meta: dict = field(default_factory=dict) # e.g. imported_from for pdf/docx
|
|
74
|
+
|
|
75
|
+
# ---- id allocation -------------------------------------------------
|
|
76
|
+
_PREFIX = {"heading": "sec", "paragraph": "para", "code": "code",
|
|
77
|
+
"table": "table", "list": "list", "blank": "blank"}
|
|
78
|
+
|
|
79
|
+
def _new_id(self, btype: str) -> str:
|
|
80
|
+
n = self.counters.get(btype, 0) + 1
|
|
81
|
+
self.counters[btype] = n
|
|
82
|
+
return f"{self._PREFIX.get(btype, 'blk')}-{n:04d}"
|
|
83
|
+
|
|
84
|
+
# ---- lookups -------------------------------------------------------
|
|
85
|
+
def index_of(self, block_id: str) -> int:
|
|
86
|
+
for i, b in enumerate(self.blocks):
|
|
87
|
+
if b.id == block_id:
|
|
88
|
+
return i
|
|
89
|
+
return -1
|
|
90
|
+
|
|
91
|
+
def get(self, block_id: str):
|
|
92
|
+
i = self.index_of(block_id)
|
|
93
|
+
return self.blocks[i] if i >= 0 else None
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
# ---------------------------------------------------------------------------
|
|
97
|
+
# Parsing (source -> canonical model)
|
|
98
|
+
# ---------------------------------------------------------------------------
|
|
99
|
+
def _classify_para(text: str) -> str:
|
|
100
|
+
lines = [ln for ln in text.splitlines() if ln.strip()]
|
|
101
|
+
if lines and all(_TABLE_ROW.match(ln) for ln in lines):
|
|
102
|
+
return "table"
|
|
103
|
+
if lines and all(_LIST_ITEM.match(ln) for ln in lines):
|
|
104
|
+
return "list"
|
|
105
|
+
return "paragraph"
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
def parse(text: str, fmt: str, source_path: str) -> Document:
|
|
109
|
+
"""Parse raw text into a Document. Markdown recognises headings, fenced
|
|
110
|
+
code, tables and lists; plain text splits on blank lines only."""
|
|
111
|
+
doc = Document(source_path=source_path, fmt=fmt, source_hash=_hash(text))
|
|
112
|
+
lines = text.split("\n")
|
|
113
|
+
i, n = 0, len(lines)
|
|
114
|
+
buf: list[str] = []
|
|
115
|
+
|
|
116
|
+
def flush_para():
|
|
117
|
+
if not buf:
|
|
118
|
+
return
|
|
119
|
+
raw = "\n".join(buf).strip("\n")
|
|
120
|
+
buf.clear()
|
|
121
|
+
if not raw.strip():
|
|
122
|
+
return
|
|
123
|
+
btype = _classify_para(raw) if fmt == "markdown" else "paragraph"
|
|
124
|
+
doc.blocks.append(Block(id=doc._new_id(btype), type=btype, text=raw))
|
|
125
|
+
|
|
126
|
+
while i < n:
|
|
127
|
+
line = lines[i]
|
|
128
|
+
if fmt == "markdown":
|
|
129
|
+
m = _HEADING.match(line)
|
|
130
|
+
if m:
|
|
131
|
+
flush_para()
|
|
132
|
+
doc.blocks.append(Block(id=doc._new_id("heading"), type="heading",
|
|
133
|
+
text=line.rstrip(), level=len(m.group(1))))
|
|
134
|
+
i += 1
|
|
135
|
+
continue
|
|
136
|
+
fm = _FENCE.match(line)
|
|
137
|
+
if fm:
|
|
138
|
+
flush_para()
|
|
139
|
+
fence = line.strip()[:3]
|
|
140
|
+
code = [line]
|
|
141
|
+
i += 1
|
|
142
|
+
while i < n:
|
|
143
|
+
code.append(lines[i])
|
|
144
|
+
if lines[i].strip().startswith(fence):
|
|
145
|
+
i += 1
|
|
146
|
+
break
|
|
147
|
+
i += 1
|
|
148
|
+
doc.blocks.append(Block(id=doc._new_id("code"), type="code",
|
|
149
|
+
text="\n".join(code)))
|
|
150
|
+
continue
|
|
151
|
+
if not line.strip():
|
|
152
|
+
flush_para()
|
|
153
|
+
else:
|
|
154
|
+
buf.append(line)
|
|
155
|
+
i += 1
|
|
156
|
+
flush_para()
|
|
157
|
+
return doc
|
|
158
|
+
|
|
159
|
+
|
|
160
|
+
# ---------------------------------------------------------------------------
|
|
161
|
+
# Rendering (canonical model -> source text)
|
|
162
|
+
# ---------------------------------------------------------------------------
|
|
163
|
+
def render(doc: Document) -> str:
|
|
164
|
+
"""Serialise blocks back to source text. Blocks are separated by a blank
|
|
165
|
+
line (the natural Markdown/prose separator); code and headings render as-is."""
|
|
166
|
+
return "\n\n".join(b.text for b in doc.blocks) + "\n"
|
|
167
|
+
|
|
168
|
+
|
|
169
|
+
# ---------------------------------------------------------------------------
|
|
170
|
+
# Persistence
|
|
171
|
+
# ---------------------------------------------------------------------------
|
|
172
|
+
def _store_path(name: str) -> Path:
|
|
173
|
+
safe = re.sub(r"[^A-Za-z0-9._-]", "_", name)
|
|
174
|
+
return STORE_DIR / f"{safe}.json"
|
|
175
|
+
|
|
176
|
+
|
|
177
|
+
def store_name(source_path: str) -> str:
|
|
178
|
+
return Path(source_path).name
|
|
179
|
+
|
|
180
|
+
|
|
181
|
+
def save(doc: Document) -> Path:
|
|
182
|
+
STORE_DIR.mkdir(parents=True, exist_ok=True)
|
|
183
|
+
p = _store_path(store_name(doc.source_path))
|
|
184
|
+
data = asdict(doc)
|
|
185
|
+
p.write_text(json.dumps(data, ensure_ascii=False, indent=1), encoding="utf-8")
|
|
186
|
+
return p
|
|
187
|
+
|
|
188
|
+
|
|
189
|
+
def load(name: str) -> Document | None:
|
|
190
|
+
p = _store_path(name)
|
|
191
|
+
if not p.exists():
|
|
192
|
+
return None
|
|
193
|
+
d = json.loads(p.read_text(encoding="utf-8"))
|
|
194
|
+
blocks = [Block(**b) for b in d.pop("blocks", [])]
|
|
195
|
+
doc = Document(**d)
|
|
196
|
+
doc.blocks = blocks
|
|
197
|
+
return doc
|
|
198
|
+
|
|
199
|
+
|
|
200
|
+
_IMPORT_EXT = (".pdf", ".docx")
|
|
201
|
+
|
|
202
|
+
|
|
203
|
+
def open_document(source_path: str, reparse: bool = False) -> Document:
|
|
204
|
+
"""Open a source file into a canonical model. Reuses a persisted store if
|
|
205
|
+
the source is unchanged (preserving ids); reparses on change or reparse=1.
|
|
206
|
+
|
|
207
|
+
.md/.markdown/.txt are edited IN PLACE. .pdf/.docx are IMPORTED read-only
|
|
208
|
+
(text extracted via drydock.extract): edits go to a Markdown sidecar
|
|
209
|
+
``<source>.canvas.md`` and the binary original is never overwritten."""
|
|
210
|
+
path = Path(source_path).expanduser()
|
|
211
|
+
if not path.exists():
|
|
212
|
+
raise FileNotFoundError(source_path)
|
|
213
|
+
ext = path.suffix.lower()
|
|
214
|
+
|
|
215
|
+
if ext in _IMPORT_EXT:
|
|
216
|
+
from drydock import extract
|
|
217
|
+
if ext == ".pdf" and not extract.pdf_backend_available():
|
|
218
|
+
raise RuntimeError(
|
|
219
|
+
"PDF import needs a backend: install `pdftotext` (poppler) or "
|
|
220
|
+
"`pip install drydock-cli[pdf]`.")
|
|
221
|
+
text = extract.extract_document(path)
|
|
222
|
+
if not text:
|
|
223
|
+
raise RuntimeError(f"could not extract any text from {path.name}")
|
|
224
|
+
edit_path = path.with_name(path.name + ".canvas.md")
|
|
225
|
+
name = edit_path.name
|
|
226
|
+
existing = load(name)
|
|
227
|
+
if existing and not reparse and existing.source_hash == _hash(text):
|
|
228
|
+
return existing
|
|
229
|
+
doc = parse(text, "markdown", str(edit_path))
|
|
230
|
+
doc.meta = {"imported_from": str(path), "import_ext": ext}
|
|
231
|
+
save(doc)
|
|
232
|
+
return doc
|
|
233
|
+
|
|
234
|
+
text = path.read_text(encoding="utf-8", errors="replace")
|
|
235
|
+
fmt = "markdown" if ext in (".md", ".markdown") else "text"
|
|
236
|
+
existing = load(path.name)
|
|
237
|
+
if existing and not reparse and existing.source_hash == _hash(text):
|
|
238
|
+
return existing
|
|
239
|
+
doc = parse(text, fmt, str(path))
|
|
240
|
+
save(doc)
|
|
241
|
+
return doc
|
|
242
|
+
|
|
243
|
+
|
|
244
|
+
# ---------------------------------------------------------------------------
|
|
245
|
+
# Read / navigate
|
|
246
|
+
# ---------------------------------------------------------------------------
|
|
247
|
+
def outline(doc: Document) -> list:
|
|
248
|
+
"""Heading hierarchy with the id and a running count of child blocks."""
|
|
249
|
+
out = []
|
|
250
|
+
for i, b in enumerate(doc.blocks):
|
|
251
|
+
if b.type == "heading":
|
|
252
|
+
title = _HEADING.match(b.text)
|
|
253
|
+
out.append({"id": b.id, "level": b.level,
|
|
254
|
+
"title": title.group(2) if title else b.text})
|
|
255
|
+
return out
|
|
256
|
+
|
|
257
|
+
|
|
258
|
+
def search(doc: Document, query: str, regex: bool = False,
|
|
259
|
+
max_hits: int = 100, ignore_case: bool = True) -> list:
|
|
260
|
+
"""Lexical search over block text. Returns id/type/snippet per hit.
|
|
261
|
+
Case-insensitive by default (find loosely); pass ignore_case=False to match
|
|
262
|
+
exactly — the same casing DocReplace/DocRedact use by default."""
|
|
263
|
+
hits = []
|
|
264
|
+
flags = re.IGNORECASE if ignore_case else 0
|
|
265
|
+
try:
|
|
266
|
+
pat = re.compile(query if regex else re.escape(query), flags)
|
|
267
|
+
except re.error as e:
|
|
268
|
+
return [{"error": f"bad regex: {e}"}]
|
|
269
|
+
for b in doc.blocks:
|
|
270
|
+
m = pat.search(b.text)
|
|
271
|
+
if m:
|
|
272
|
+
s = max(0, m.start() - 40)
|
|
273
|
+
snippet = b.text[s:m.end() + 40].replace("\n", " ")
|
|
274
|
+
hits.append({"id": b.id, "type": b.type, "snippet": snippet})
|
|
275
|
+
if len(hits) >= max_hits:
|
|
276
|
+
break
|
|
277
|
+
return hits
|
|
278
|
+
|
|
279
|
+
|
|
280
|
+
def read(doc: Document, block_id: str, before: int = 0, after: int = 0) -> dict:
|
|
281
|
+
"""Return a block plus N neighbours on each side (window into the doc)."""
|
|
282
|
+
i = doc.index_of(block_id)
|
|
283
|
+
if i < 0:
|
|
284
|
+
return {"error": f"no such block: {block_id}"}
|
|
285
|
+
lo, hi = max(0, i - before), min(len(doc.blocks), i + after + 1)
|
|
286
|
+
window = [{"id": b.id, "type": b.type, "hash": b.content_hash,
|
|
287
|
+
"text": b.text, "target": b.id == block_id}
|
|
288
|
+
for b in doc.blocks[lo:hi]]
|
|
289
|
+
return {"target": block_id, "window": window}
|
|
290
|
+
|
|
291
|
+
|
|
292
|
+
# ---------------------------------------------------------------------------
|
|
293
|
+
# Transactional patching
|
|
294
|
+
# ---------------------------------------------------------------------------
|
|
295
|
+
class PatchError(Exception):
|
|
296
|
+
pass
|
|
297
|
+
|
|
298
|
+
|
|
299
|
+
_OPS = {"replace", "insert_before", "insert_after", "delete"}
|
|
300
|
+
|
|
301
|
+
|
|
302
|
+
def apply_patches(doc: Document, patches: list) -> list:
|
|
303
|
+
"""Validate + apply a list of patches to `doc` in order, atomically: if ANY
|
|
304
|
+
patch fails validation the document is left untouched and PatchError is
|
|
305
|
+
raised. Each patch: {op, target_id, expected_hash?, new_text?, reason?}.
|
|
306
|
+
|
|
307
|
+
Returns a per-patch changelog (for the audit trail + diff)."""
|
|
308
|
+
# 1) validate everything first (atomic)
|
|
309
|
+
for p in patches:
|
|
310
|
+
op = p.get("op")
|
|
311
|
+
if op not in _OPS:
|
|
312
|
+
raise PatchError(f"unknown op: {op!r} (use {sorted(_OPS)})")
|
|
313
|
+
tid = p.get("target_id")
|
|
314
|
+
idx = doc.index_of(tid) if tid else -1
|
|
315
|
+
if idx < 0:
|
|
316
|
+
raise PatchError(f"target_id not found: {tid!r}")
|
|
317
|
+
exp = p.get("expected_hash")
|
|
318
|
+
if exp and doc.blocks[idx].content_hash != exp:
|
|
319
|
+
raise PatchError(
|
|
320
|
+
f"stale content for {tid}: expected_hash {exp} but current is "
|
|
321
|
+
f"{doc.blocks[idx].content_hash}. Re-read the block before patching.")
|
|
322
|
+
if op in ("replace", "insert_before", "insert_after") and not p.get("new_text", "").strip():
|
|
323
|
+
raise PatchError(f"{op} on {tid} needs non-empty new_text")
|
|
324
|
+
|
|
325
|
+
# 2) apply (target ids are resolved fresh each step; inserts shift indices)
|
|
326
|
+
changelog = []
|
|
327
|
+
for p in patches:
|
|
328
|
+
op, tid = p["op"], p["target_id"]
|
|
329
|
+
idx = doc.index_of(tid)
|
|
330
|
+
if op == "replace":
|
|
331
|
+
old = doc.blocks[idx].text
|
|
332
|
+
doc.blocks[idx].text = p["new_text"]
|
|
333
|
+
doc.blocks[idx].rehash()
|
|
334
|
+
changelog.append({"op": op, "id": tid, "old": old, "new": p["new_text"],
|
|
335
|
+
"reason": p.get("reason", "")})
|
|
336
|
+
elif op == "delete":
|
|
337
|
+
old = doc.blocks.pop(idx)
|
|
338
|
+
changelog.append({"op": op, "id": tid, "old": old.text, "new": "",
|
|
339
|
+
"reason": p.get("reason", "")})
|
|
340
|
+
else: # insert_before / insert_after
|
|
341
|
+
btype = p.get("block_type", "paragraph")
|
|
342
|
+
nb = Block(id=doc._new_id(btype), type=btype, text=p["new_text"],
|
|
343
|
+
level=int(p.get("level", 0)))
|
|
344
|
+
at = idx if op == "insert_before" else idx + 1
|
|
345
|
+
doc.blocks.insert(at, nb)
|
|
346
|
+
changelog.append({"op": op, "id": nb.id, "anchor": tid, "old": "",
|
|
347
|
+
"new": nb.text, "reason": p.get("reason", "")})
|
|
348
|
+
return changelog
|
|
349
|
+
|
|
350
|
+
|
|
351
|
+
def diff_text(changelog: list) -> str:
|
|
352
|
+
"""Human-readable unified-ish diff of a changelog."""
|
|
353
|
+
import difflib
|
|
354
|
+
out = []
|
|
355
|
+
for c in changelog:
|
|
356
|
+
head = f"@@ {c['op']} {c.get('id','')}"
|
|
357
|
+
if c.get("anchor"):
|
|
358
|
+
head += f" (anchor {c['anchor']})"
|
|
359
|
+
if c.get("reason"):
|
|
360
|
+
head += f" — {c['reason']}"
|
|
361
|
+
out.append(head)
|
|
362
|
+
old = (c.get("old") or "").splitlines()
|
|
363
|
+
new = (c.get("new") or "").splitlines()
|
|
364
|
+
for line in difflib.unified_diff(old, new, lineterm="", n=1):
|
|
365
|
+
if line.startswith(("---", "+++", "@@")):
|
|
366
|
+
continue
|
|
367
|
+
out.append(line)
|
|
368
|
+
return "\n".join(out)
|
|
369
|
+
|
|
370
|
+
|
|
371
|
+
# ---------------------------------------------------------------------------
|
|
372
|
+
# Bulk search-and-update (one O(n) pass — the scalable primitive for global
|
|
373
|
+
# changes on a large document, instead of one patch call per occurrence)
|
|
374
|
+
# ---------------------------------------------------------------------------
|
|
375
|
+
def replace_all(doc: Document, query: str, replacement: str,
|
|
376
|
+
regex: bool = False, ignore_case: bool = False) -> dict:
|
|
377
|
+
"""Replace EVERY occurrence of `query` across all blocks in a single pass.
|
|
378
|
+
Case-SENSITIVE by default (precise; preserves other casings); pass
|
|
379
|
+
ignore_case=True to replace all case variants. Returns a count + touched ids."""
|
|
380
|
+
flags = re.IGNORECASE if ignore_case else 0
|
|
381
|
+
try:
|
|
382
|
+
pat = re.compile(query if regex else re.escape(query), flags)
|
|
383
|
+
except re.error as e:
|
|
384
|
+
raise PatchError(f"bad regex: {e}")
|
|
385
|
+
count = 0
|
|
386
|
+
touched = []
|
|
387
|
+
for b in doc.blocks:
|
|
388
|
+
new, n = pat.subn(replacement, b.text)
|
|
389
|
+
if n:
|
|
390
|
+
b.text = new
|
|
391
|
+
b.rehash()
|
|
392
|
+
count += n
|
|
393
|
+
touched.append(b.id)
|
|
394
|
+
return {"replacements": count, "blocks_touched": touched}
|
|
395
|
+
|
|
396
|
+
|
|
397
|
+
# ---------------------------------------------------------------------------
|
|
398
|
+
# Redaction ("red boxing") — PERMANENTLY remove text, then VERIFY it cannot be
|
|
399
|
+
# recovered from the document. Verification is BLOCKING (unlike the advisory
|
|
400
|
+
# validators): a failed check aborts and the store is left unchanged.
|
|
401
|
+
# ---------------------------------------------------------------------------
|
|
402
|
+
REDACT_MARKER = "[REDACTED]"
|
|
403
|
+
|
|
404
|
+
|
|
405
|
+
def redact(doc: Document, query: str | None = None, block_id: str | None = None,
|
|
406
|
+
regex: bool = False, marker: str = REDACT_MARKER,
|
|
407
|
+
ignore_case: bool = True) -> dict:
|
|
408
|
+
"""Redact by whole block (block_id) or by matched text (query). The removed
|
|
409
|
+
text is permanently replaced by `marker`; only a HASH of it is recorded (the
|
|
410
|
+
store never keeps the sensitive text). Then verifies none of the removed
|
|
411
|
+
strings survive in the rendered document — raises PatchError if any do.
|
|
412
|
+
Case-INSENSITIVE by default so every casing of a secret is caught."""
|
|
413
|
+
removed: list[str] = []
|
|
414
|
+
if block_id:
|
|
415
|
+
b = doc.get(block_id)
|
|
416
|
+
if b is None:
|
|
417
|
+
raise PatchError(f"no such block: {block_id}")
|
|
418
|
+
removed.append(b.text)
|
|
419
|
+
b.text = marker
|
|
420
|
+
b.rehash()
|
|
421
|
+
count = 1
|
|
422
|
+
elif query:
|
|
423
|
+
if not regex and query in marker:
|
|
424
|
+
raise PatchError("redaction marker contains the query — pick a different marker")
|
|
425
|
+
try:
|
|
426
|
+
pat = re.compile(query if regex else re.escape(query),
|
|
427
|
+
re.IGNORECASE if ignore_case else 0)
|
|
428
|
+
except re.error as e:
|
|
429
|
+
raise PatchError(f"bad regex: {e}")
|
|
430
|
+
count = 0
|
|
431
|
+
for b in doc.blocks:
|
|
432
|
+
def _sub(m):
|
|
433
|
+
nonlocal count
|
|
434
|
+
removed.append(m.group(0))
|
|
435
|
+
count += 1
|
|
436
|
+
return marker
|
|
437
|
+
b.text = pat.subn(_sub, b.text)[0]
|
|
438
|
+
b.rehash()
|
|
439
|
+
if count == 0:
|
|
440
|
+
raise PatchError(f"nothing matched {query!r} — nothing redacted")
|
|
441
|
+
else:
|
|
442
|
+
raise PatchError("redact needs a block_id or a query")
|
|
443
|
+
|
|
444
|
+
# BLOCKING verification: no removed string may still be recoverable.
|
|
445
|
+
full = render(doc)
|
|
446
|
+
leaked = sorted({s for s in removed if s and s != marker and s in full})
|
|
447
|
+
if leaked:
|
|
448
|
+
raise PatchError(
|
|
449
|
+
f"redaction verification FAILED: {len(leaked)} removed string(s) still "
|
|
450
|
+
"recoverable — redaction aborted, document unchanged")
|
|
451
|
+
doc.redactions.append({"mode": "block" if block_id else "query",
|
|
452
|
+
"target": block_id or query, "count": count,
|
|
453
|
+
"content_hash": _hash("\x00".join(removed)), "marker": marker})
|
|
454
|
+
return {"redacted": count, "verified": True, "marker": marker}
|
|
455
|
+
|
|
456
|
+
|
|
457
|
+
# ---------------------------------------------------------------------------
|
|
458
|
+
# Validation (deterministic, run before any LLM review)
|
|
459
|
+
# ---------------------------------------------------------------------------
|
|
460
|
+
def validate(doc: Document, required: list | None = None,
|
|
461
|
+
prohibited: list | None = None) -> dict:
|
|
462
|
+
"""Structural + content checks. Advisory: returns pass/warn findings; callers
|
|
463
|
+
decide whether to block (redaction is the only hard-blocking case, elsewhere)."""
|
|
464
|
+
findings = []
|
|
465
|
+
ids = [b.id for b in doc.blocks]
|
|
466
|
+
dupes = {x for x in ids if ids.count(x) > 1}
|
|
467
|
+
findings.append({"check": "unique_ids", "ok": not dupes,
|
|
468
|
+
"detail": f"duplicate ids: {sorted(dupes)}" if dupes else ""})
|
|
469
|
+
bad_hash = [b.id for b in doc.blocks if b.content_hash != _hash(b.text)]
|
|
470
|
+
findings.append({"check": "hashes_consistent", "ok": not bad_hash,
|
|
471
|
+
"detail": f"stale hashes: {bad_hash}" if bad_hash else ""})
|
|
472
|
+
empty = [b.id for b in doc.blocks if not b.text.strip()]
|
|
473
|
+
findings.append({"check": "no_empty_blocks", "ok": not empty,
|
|
474
|
+
"detail": f"empty blocks: {empty}" if empty else ""})
|
|
475
|
+
full = render(doc)
|
|
476
|
+
for phrase in (required or []):
|
|
477
|
+
findings.append({"check": f"required:{phrase!r}", "ok": phrase in full,
|
|
478
|
+
"detail": "" if phrase in full else "missing"})
|
|
479
|
+
for phrase in (prohibited or []):
|
|
480
|
+
findings.append({"check": f"prohibited:{phrase!r}", "ok": phrase not in full,
|
|
481
|
+
"detail": "" if phrase not in full else "still present"})
|
|
482
|
+
findings.append({"check": "heading_levels_monotonic",
|
|
483
|
+
"ok": _levels_ok(doc), "detail": _levels_detail(doc)})
|
|
484
|
+
return {"ok": all(f["ok"] for f in findings), "findings": findings}
|
|
485
|
+
|
|
486
|
+
|
|
487
|
+
def _levels_ok(doc: Document) -> bool:
|
|
488
|
+
prev = 0
|
|
489
|
+
for b in doc.blocks:
|
|
490
|
+
if b.type == "heading":
|
|
491
|
+
if prev and b.level > prev + 1:
|
|
492
|
+
return False
|
|
493
|
+
prev = b.level
|
|
494
|
+
return True
|
|
495
|
+
|
|
496
|
+
|
|
497
|
+
def _levels_detail(doc: Document) -> str:
|
|
498
|
+
prev = 0
|
|
499
|
+
for b in doc.blocks:
|
|
500
|
+
if b.type == "heading":
|
|
501
|
+
if prev and b.level > prev + 1:
|
|
502
|
+
return f"{b.id} jumps from h{prev} to h{b.level}"
|
|
503
|
+
prev = b.level
|
|
504
|
+
return ""
|
|
505
|
+
|
|
506
|
+
|
|
507
|
+
# ---------------------------------------------------------------------------
|
|
508
|
+
# Commit (canonical model -> source file, original preserved)
|
|
509
|
+
# ---------------------------------------------------------------------------
|
|
510
|
+
def commit(doc: Document) -> dict:
|
|
511
|
+
"""Render the canonical model back to the source file. On the first commit,
|
|
512
|
+
the untouched original is preserved as ``<source>.orig`` (immutable artifact)."""
|
|
513
|
+
src = Path(doc.source_path)
|
|
514
|
+
if not doc.committed:
|
|
515
|
+
orig = src.with_suffix(src.suffix + ".orig")
|
|
516
|
+
if src.exists() and not orig.exists():
|
|
517
|
+
orig.write_text(src.read_text(encoding="utf-8", errors="replace"),
|
|
518
|
+
encoding="utf-8")
|
|
519
|
+
doc.committed = True
|
|
520
|
+
text = render(doc)
|
|
521
|
+
src.write_text(text, encoding="utf-8")
|
|
522
|
+
doc.source_hash = _hash(text)
|
|
523
|
+
save(doc)
|
|
524
|
+
return {"source": str(src), "blocks": len(doc.blocks),
|
|
525
|
+
"original": str(src.with_suffix(src.suffix + ".orig"))}
|