woods 2.0.0.beta2 → 2.0.0.beta3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (218) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +262 -1
  3. data/CONTRIBUTING.md +173 -9
  4. data/README.md +7 -3
  5. data/SECURITY.md +9 -6
  6. data/docs/AGENT_GUIDE.md +83 -4
  7. data/docs/AGENT_SETUP.md +82 -1
  8. data/docs/BACKEND_MATRIX.md +20 -0
  9. data/docs/CLIENT_HOOKS.md +111 -0
  10. data/docs/CONFIGURATION_REFERENCE.md +199 -14
  11. data/docs/CONSOLE_MCP_SETUP.md +35 -5
  12. data/docs/DOCKER_SETUP.md +21 -2
  13. data/docs/EVALUATION.md +464 -1
  14. data/docs/EXTRACTOR_REFERENCE.md +36 -5
  15. data/docs/FAQ.md +11 -12
  16. data/docs/GETTING_STARTED.md +17 -5
  17. data/docs/INCREMENTAL_EXTRACTION.md +117 -1
  18. data/docs/INDEX_LAYOUT.md +382 -0
  19. data/docs/INTERNALS.md +7 -2
  20. data/docs/MCP_SERVERS.md +221 -5
  21. data/docs/MCP_TOOL_COOKBOOK.md +33 -18
  22. data/docs/NOTION_INTEGRATION.md +13 -0
  23. data/docs/OBSIDIAN_INTEGRATION.md +57 -9
  24. data/docs/PUBLISHED_INDEX.md +55 -0
  25. data/docs/README.md +7 -0
  26. data/docs/RETRIEVAL_GUIDE.md +253 -11
  27. data/docs/RUNTIME_TRACING.md +71 -0
  28. data/docs/SOURCE_FRESHNESS.md +143 -0
  29. data/docs/TROUBLESHOOTING.md +117 -5
  30. data/docs/UNBLOCKED_INTEGRATION.md +25 -0
  31. data/docs/UPGRADING_TO_2.md +44 -22
  32. data/docs/WATCH_DAEMON.md +259 -59
  33. data/exe/woods-agent-config +6 -0
  34. data/exe/woods-extract +5 -0
  35. data/exe/woods-hook-context +6 -0
  36. data/lib/generators/woods/templates/woods.rb.tt +1 -3
  37. data/lib/tasks/woods.rake +47 -397
  38. data/lib/woods/agent_configuration/applier.rb +133 -0
  39. data/lib/woods/agent_configuration/cli.rb +101 -0
  40. data/lib/woods/agent_configuration/cli_options.rb +29 -0
  41. data/lib/woods/agent_configuration/document.rb +105 -0
  42. data/lib/woods/agent_configuration/error.rb +7 -0
  43. data/lib/woods/agent_configuration/launcher.rb +75 -0
  44. data/lib/woods/agent_configuration/layout.rb +59 -0
  45. data/lib/woods/agent_configuration/managed_section.rb +62 -0
  46. data/lib/woods/agent_configuration/plan.rb +98 -0
  47. data/lib/woods/agent_configuration/plan_diff.rb +38 -0
  48. data/lib/woods/agent_configuration/planned_files.rb +61 -0
  49. data/lib/woods/agent_configuration/planner.rb +63 -0
  50. data/lib/woods/agent_configuration/planner_validation.rb +77 -0
  51. data/lib/woods/agent_configuration/preflight.rb +100 -0
  52. data/lib/woods/agent_configuration/recovery.rb +49 -0
  53. data/lib/woods/ast/node.rb +2 -0
  54. data/lib/woods/ast/parser.rb +38 -5
  55. data/lib/woods/builder.rb +21 -5
  56. data/lib/woods/cache/cache_middleware.rb +28 -7
  57. data/lib/woods/cache/cache_store.rb +4 -5
  58. data/lib/woods/change_set.rb +5 -4
  59. data/lib/woods/console/credential_index.rb +20 -2
  60. data/lib/woods/console/credential_scanner.rb +14 -14
  61. data/lib/woods/console/credential_scanner_registry.rb +36 -0
  62. data/lib/woods/console/embedded_executor.rb +1 -1
  63. data/lib/woods/console/encrypted_credential_snapshot.rb +16 -0
  64. data/lib/woods/console/rack_middleware.rb +22 -13
  65. data/lib/woods/console/server.rb +18 -16
  66. data/lib/woods/dependency_graph.rb +65 -13
  67. data/lib/woods/embedding/corpus.rb +94 -0
  68. data/lib/woods/embedding/indexer.rb +90 -46
  69. data/lib/woods/embedding/openai.rb +17 -6
  70. data/lib/woods/evaluation/ablation_executor.rb +6 -1
  71. data/lib/woods/evaluation/ablation_timed_executor.rb +22 -4
  72. data/lib/woods/export/typed_reader.rb +56 -0
  73. data/lib/woods/extractor.rb +232 -137
  74. data/lib/woods/extractors/action_cable_extractor.rb +3 -1
  75. data/lib/woods/extractors/behavioral_profile.rb +9 -7
  76. data/lib/woods/extractors/caching_extractor.rb +3 -1
  77. data/lib/woods/extractors/concern_extractor.rb +64 -6
  78. data/lib/woods/extractors/configuration_extractor.rb +7 -3
  79. data/lib/woods/extractors/controller_extractor.rb +13 -4
  80. data/lib/woods/extractors/database_view_extractor.rb +3 -1
  81. data/lib/woods/extractors/decorator_extractor.rb +3 -1
  82. data/lib/woods/extractors/engine_extractor.rb +3 -1
  83. data/lib/woods/extractors/event_extractor.rb +4 -2
  84. data/lib/woods/extractors/factory_extractor.rb +3 -1
  85. data/lib/woods/extractors/graphql_extractor.rb +8 -2
  86. data/lib/woods/extractors/i18n_extractor.rb +3 -1
  87. data/lib/woods/extractors/job_extractor.rb +6 -19
  88. data/lib/woods/extractors/lib_extractor.rb +3 -1
  89. data/lib/woods/extractors/mailer_extractor.rb +20 -5
  90. data/lib/woods/extractors/manager_extractor.rb +3 -1
  91. data/lib/woods/extractors/method_parameters.rb +53 -0
  92. data/lib/woods/extractors/middleware_argument.rb +65 -0
  93. data/lib/woods/extractors/middleware_extractor.rb +9 -3
  94. data/lib/woods/extractors/migration_extractor.rb +3 -1
  95. data/lib/woods/extractors/model_extractor.rb +39 -33
  96. data/lib/woods/extractors/package_extractor.rb +24 -4
  97. data/lib/woods/extractors/phlex_extractor.rb +3 -1
  98. data/lib/woods/extractors/policy_extractor.rb +3 -1
  99. data/lib/woods/extractors/poro_extractor.rb +3 -1
  100. data/lib/woods/extractors/pundit_extractor.rb +3 -1
  101. data/lib/woods/extractors/rails_source_extractor.rb +4 -2
  102. data/lib/woods/extractors/rake_task_extractor.rb +4 -2
  103. data/lib/woods/extractors/route_extractor.rb +3 -1
  104. data/lib/woods/extractors/route_helper_resolver.rb +10 -33
  105. data/lib/woods/extractors/scheduled_job_extractor.rb +41 -15
  106. data/lib/woods/extractors/serializer_extractor.rb +4 -2
  107. data/lib/woods/extractors/service_extractor.rb +3 -1
  108. data/lib/woods/extractors/shared_dependency_scanner.rb +2 -2
  109. data/lib/woods/extractors/shared_utility_methods.rb +27 -15
  110. data/lib/woods/extractors/source_nesting.rb +1 -1
  111. data/lib/woods/extractors/state_machine_extractor.rb +3 -1
  112. data/lib/woods/extractors/test_mapping_extractor.rb +3 -1
  113. data/lib/woods/extractors/validator_extractor.rb +3 -1
  114. data/lib/woods/extractors/view_component_extractor.rb +3 -1
  115. data/lib/woods/extractors/view_template_extractor.rb +3 -1
  116. data/lib/woods/gem_mapper.rb +2 -0
  117. data/lib/woods/git_history.rb +116 -0
  118. data/lib/woods/graph_analyzer.rb +35 -6
  119. data/lib/woods/hooks/context_cli.rb +54 -0
  120. data/lib/woods/hooks/context_event.rb +88 -0
  121. data/lib/woods/hooks/context_hint.rb +73 -0
  122. data/lib/woods/hooks/context_impact.rb +77 -0
  123. data/lib/woods/hooks/context_output.rb +47 -0
  124. data/lib/woods/hooks/context_state.rb +102 -0
  125. data/lib/woods/hooks/refresh.rb +79 -0
  126. data/lib/woods/hooks/rule_projection.rb +78 -0
  127. data/lib/woods/input_rules.rb +19 -0
  128. data/lib/woods/mcp/bearer_auth.rb +20 -12
  129. data/lib/woods/mcp/bootstrapper.rb +62 -0
  130. data/lib/woods/mcp/index_reader.rb +323 -160
  131. data/lib/woods/mcp/initialization_guidance.rb +27 -0
  132. data/lib/woods/mcp/origin_guard.rb +17 -9
  133. data/lib/woods/mcp/published_lexical_retriever.rb +115 -0
  134. data/lib/woods/mcp/renderers/markdown_renderer.rb +8 -1
  135. data/lib/woods/mcp/renderers/plain_renderer.rb +7 -1
  136. data/lib/woods/mcp/search_results.rb +74 -0
  137. data/lib/woods/mcp/server.rb +158 -37
  138. data/lib/woods/mcp/tool_contract.rb +2 -0
  139. data/lib/woods/mcp/tool_response_renderer.rb +25 -0
  140. data/lib/woods/mcp/traversal_evidence.rb +113 -0
  141. data/lib/woods/mcp/traversal_evidence_index.rb +100 -0
  142. data/lib/woods/mcp/traversal_evidence_page.rb +41 -0
  143. data/lib/woods/mcp/traversal_evidence_text.rb +52 -0
  144. data/lib/woods/notion/exporter.rb +56 -17
  145. data/lib/woods/obsidian/destination_plan.rb +98 -0
  146. data/lib/woods/obsidian/name_mapper.rb +19 -3
  147. data/lib/woods/obsidian/note_builder.rb +19 -10
  148. data/lib/woods/obsidian/vault_exporter.rb +88 -32
  149. data/lib/woods/operator/pipeline_guard.rb +18 -13
  150. data/lib/woods/path_dispatcher.rb +7 -1
  151. data/lib/woods/payload_store.rb +27 -26
  152. data/lib/woods/railtie.rb +3 -3
  153. data/lib/woods/railtie_support.rb +12 -12
  154. data/lib/woods/rake_helpers.rb +392 -0
  155. data/lib/woods/resilience/graph_invariant_validator/membership_checks.rb +71 -0
  156. data/lib/woods/resilience/graph_invariant_validator/node_checks.rb +61 -0
  157. data/lib/woods/resilience/graph_invariant_validator/reverse_relationship_checks.rb +46 -0
  158. data/lib/woods/resilience/graph_invariant_validator.rb +119 -0
  159. data/lib/woods/resilience/index_validator/graph_checks.rb +80 -0
  160. data/lib/woods/resilience/index_validator.rb +112 -23
  161. data/lib/woods/retrieval/context_assembler.rb +50 -15
  162. data/lib/woods/retrieval/lexical_assembler.rb +73 -0
  163. data/lib/woods/retrieval/lexical_index.rb +119 -0
  164. data/lib/woods/retrieval/ranker.rb +4 -2
  165. data/lib/woods/retrieval/scope.rb +108 -0
  166. data/lib/woods/retrieval/scoped_graph_store.rb +32 -0
  167. data/lib/woods/retrieval/scoped_vector_store.rb +55 -0
  168. data/lib/woods/retrieval/search_executor.rb +86 -27
  169. data/lib/woods/retrieval/source_evidence.rb +200 -0
  170. data/lib/woods/retriever.rb +98 -22
  171. data/lib/woods/ruby_analyzer/trace_enricher.rb +77 -38
  172. data/lib/woods/session_tracer/middleware.rb +10 -12
  173. data/lib/woods/session_tracer/redis_store.rb +22 -6
  174. data/lib/woods/session_tracer/session_flow_assembler.rb +23 -17
  175. data/lib/woods/session_tracer/solid_cache_coordination.rb +6 -4
  176. data/lib/woods/session_tracer/unit_resolver.rb +63 -0
  177. data/lib/woods/source_inputs/consumer_errors.rb +27 -0
  178. data/lib/woods/source_inputs/handoff.rb +102 -0
  179. data/lib/woods/source_inputs/launcher.rb +157 -0
  180. data/lib/woods/source_inputs/manifest.rb +124 -0
  181. data/lib/woods/source_inputs/private_key.rb +55 -0
  182. data/lib/woods/source_inputs/scanner.rb +171 -0
  183. data/lib/woods/source_inputs/scopes.rb +71 -0
  184. data/lib/woods/source_inputs/session.rb +214 -0
  185. data/lib/woods/source_inputs/status.rb +84 -0
  186. data/lib/woods/source_inputs/verifier.rb +107 -0
  187. data/lib/woods/storage/metadata_store.rb +25 -25
  188. data/lib/woods/storage/pgvector.rb +29 -8
  189. data/lib/woods/storage/qdrant.rb +17 -7
  190. data/lib/woods/storage/vector_store.rb +18 -6
  191. data/lib/woods/tasks.rb +3 -2
  192. data/lib/woods/temporal/json_snapshot_store.rb +29 -8
  193. data/lib/woods/unblocked/exporter.rb +59 -70
  194. data/lib/woods/version.rb +1 -1
  195. data/lib/woods/watch/boot_snapshot.rb +52 -0
  196. data/lib/woods/watch/daemon.rb +136 -28
  197. data/lib/woods/watch/listen_watcher.rb +4 -0
  198. data/lib/woods/watch/polling_watcher.rb +5 -1
  199. data/lib/woods/watch/status.rb +20 -15
  200. data/lib/woods/watch/tree_scan.rb +21 -13
  201. data/lib/woods/watch/watcher.rb +4 -1
  202. data/lib/woods.rb +50 -11
  203. data/plugin/.claude-plugin/plugin.json +1 -1
  204. data/plugin/hooks/adapters/normalize.jq +15 -0
  205. data/plugin/hooks/adapters/normalize.rb +63 -0
  206. data/plugin/hooks/hooks.json +20 -0
  207. data/plugin/hooks/woods-context.sh +50 -0
  208. data/plugin/hooks/woods-input-rules.sh +159 -0
  209. data/plugin/hooks/woods-opencode.mjs +65 -0
  210. data/plugin/hooks/woods-post-edit.sh +2 -225
  211. data/plugin/hooks/woods-refresh.sh +260 -0
  212. data/plugin/hooks/woods-session-start.sh +47 -55
  213. data/plugin/skills/woods-agent-enable/SKILL.md +13 -0
  214. data/plugin/skills/woods-diagnose/SKILL.md +288 -1
  215. data/plugin/skills/woods-investigate/SKILL.md +106 -0
  216. data/plugin/skills/woods-mcp-config/SKILL.md +89 -1
  217. data/plugin/skills/woods-setup/SKILL.md +107 -6
  218. metadata +84 -5
@@ -0,0 +1,392 @@
1
+ # frozen_string_literal: true
2
+
3
+ require 'woods/atomic_file'
4
+ require 'woods/generation'
5
+ require 'woods/git_command'
6
+
7
+ module Woods
8
+ # Shared rake implementation, kept off the host application's Object.
9
+ module RakeHelpers # rubocop:disable Metrics/ModuleLength -- existing task helpers grouped without changing behavior
10
+ module_function
11
+
12
+ # ── Multi-instance helpers (#164 phase 4) ────────────────────────────────
13
+ #
14
+ # Worktrees are disjoint by construction (each has its own Rails.root and
15
+ # its own output dir), so these only ever mediate writers against the *same*
16
+ # index: a manual rake run, a hook-triggered sync, and the watch daemon.
17
+
18
+ # Run a block holding the extraction lock, waiting for another writer to
19
+ # finish.
20
+ #
21
+ # This used to proceed *without* the lock after 30s, on the reasoning that a
22
+ # daemon cycle is milliseconds so a longer wait meant something unusual. That
23
+ # reasoning was wrong in the case that matters: a cycle includes a
24
+ # storm-triggered `extract_all`, which on a large host app runs for minutes.
25
+ # Proceeding then means two writers load `dependency_graph.json`, mutate
26
+ # divergent copies, and the last one silently discards the other's work — then
27
+ # bumps the generation, marking the clobbered graph fresh. Per-file atomic
28
+ # writes do not help, because the file *set* is not atomic.
29
+ #
30
+ # So the wait is now generous and the failure explicit. `WOODS_LOCK_WAIT`
31
+ # overrides it; exceeding it exits non-zero rather than corrupting the index,
32
+ # which is the outcome a CI job or a developer can actually act on.
33
+ # Fail the task when a run reported per-item errors (INF-4 / EXP-4).
34
+ #
35
+ # A printed-but-green run is invisible in the CI/cron pipelines these tasks
36
+ # target: a revoked API key, a full vector store, or a 401 on every page
37
+ # prints `Errors: N` and the job stays green while the index or the export
38
+ # target rots. `woods:unblocked_sync` and `woods:obsidian` already exit 1 on
39
+ # the same shape, and #270 fixed it for the extraction family; the embedding
40
+ # and Notion tasks carry identical CI exposure.
41
+ #
42
+ # Partial progress is deliberately *not* a carve-out here. The unblocked
43
+ # exporter has one (budget exhaustion with progress is the expected
44
+ # cold-start shape and converges on the next run); an embedding or Notion
45
+ # error is a genuine failure per unit and does not self-heal, so any
46
+ # non-zero count fails.
47
+ #
48
+ # @param errors [Integer, Array, nil] the run's error count or error list
49
+ # @return [void]
50
+ def woods_exit_on_reported_errors(errors)
51
+ # The embed indexer reports a count; the exporters report a list.
52
+ # (`Integer#size` exists and answers 8, so this must branch on the type.)
53
+ count = errors.is_a?(Numeric) ? errors.to_i : Array(errors).size
54
+ return if count.zero?
55
+
56
+ puts
57
+ puts 'Run completed with errors — failing so CI surfaces it.'
58
+ exit 1
59
+ end
60
+
61
+ def woods_with_extraction_lock(output_dir, wait: nil, raise_on_timeout: false, &block)
62
+ # Requires first. The default wait reads a constant from the daemon, so
63
+ # resolving it above these lines NameError'd every write task — the same
64
+ # load-order bug as the missing require in `woods:watch`, reintroduced one
65
+ # method over by the fix for it.
66
+ require 'woods/coordination/pipeline_lock'
67
+ require 'woods/coordination/lock_heartbeat'
68
+ require 'woods/watch/daemon'
69
+
70
+ wait ||= Float(ENV.fetch('WOODS_LOCK_WAIT', Woods::Watch::Daemon::LOCK_STALE_TIMEOUT))
71
+
72
+ lock = Woods::Coordination::PipelineLock.new(
73
+ lock_dir: output_dir.to_s,
74
+ name: Woods::Watch::Daemon::LOCK_NAME,
75
+ stale_timeout: Woods::Watch::Daemon::LOCK_STALE_TIMEOUT
76
+ )
77
+
78
+ woods_abort_on_lock_timeout(wait, raise_error: raise_on_timeout) unless woods_acquire_within(lock, wait)
79
+
80
+ begin
81
+ Woods::Coordination::LockHeartbeat.run(lock, &block)
82
+ ensure
83
+ lock.release
84
+ end
85
+ end
86
+
87
+ # Poll for the lock until `wait` seconds have elapsed.
88
+ #
89
+ # Monotonic, so a clock adjustment mid-wait cannot cut the window short or
90
+ # extend it indefinitely.
91
+ #
92
+ # @return [Boolean] whether the lock was acquired
93
+ def woods_acquire_within(lock, wait)
94
+ deadline = Process.clock_gettime(Process::CLOCK_MONOTONIC) + wait
95
+ acquired = lock.acquire
96
+ until acquired || Process.clock_gettime(Process::CLOCK_MONOTONIC) > deadline
97
+ sleep 0.25
98
+ acquired = lock.acquire
99
+ end
100
+ acquired
101
+ end
102
+
103
+ def woods_abort_on_lock_timeout(wait, raise_error: false)
104
+ if raise_error
105
+ raise Coordination::LockError,
106
+ "Another writer holds the extraction lock after #{wait}s; set WOODS_LOCK_WAIT to wait longer"
107
+ end
108
+
109
+ warn "ERROR: another writer has held the extraction lock for #{wait.round}s."
110
+ warn 'Refusing to write: two concurrent writers rewrite the dependency graph from divergent'
111
+ warn 'copies, and the loser\'s work is discarded under a generation that says "fresh".'
112
+ warn 'Set WOODS_LOCK_WAIT to wait longer, or stop the other writer.'
113
+ exit 1
114
+ end
115
+
116
+ # Is a watch daemon already maintaining this index?
117
+ #
118
+ # A session-start or worktree hook that fires `woods:incremental` on a tree
119
+ # a daemon is already watching is pure duplicated work — and it contends for
120
+ # the lock the daemon needs. Set WOODS_IGNORE_WATCH=1 to run anyway.
121
+ #
122
+ # Liveness alone is not coverage, and the difference matters. A `:running`
123
+ # daemon reconciles everything modified since the index's last successful
124
+ # publish when it starts (Daemon#catch_up), so changes that predate it are
125
+ # covered whether or not it witnessed them — that is what makes standing down
126
+ # safe. A `:degraded` daemon is alive but *cannot* currently update, so
127
+ # standing down for it would report success over work nothing is doing.
128
+ #
129
+ # @return [Symbol] `:none`, `:running`, or `:degraded`
130
+ def woods_daemon_coverage(output_dir)
131
+ return :none if ENV['WOODS_IGNORE_WATCH'] == '1'
132
+
133
+ require 'woods/watch/status'
134
+ status = Woods::Watch::Status.new(output_dir: output_dir)
135
+ return :none unless status.alive?
136
+
137
+ status.read['state'] == 'degraded' ? :degraded : :running
138
+ rescue StandardError
139
+ :none
140
+ end
141
+
142
+ # Delete the index without yanking it out from under another writer (#170).
143
+ #
144
+ # The old `woods:clean` body was a bare rm_rf. Run mid-daemon-cycle it
145
+ # deleted the daemon's own extraction lock, so the "writers serialize on
146
+ # PipelineLock" invariant evaporated at the exact moment a writer was
147
+ # mid-graph-rewrite. Now: refuse while a daemon is alive (same stand-down
148
+ # check as `woods:incremental`; WOODS_IGNORE_WATCH=1 overrides), then take
149
+ # the extraction lock like every other writer, delete everything EXCEPT the
150
+ # lock file itself, and let release remove the lock last.
151
+ #
152
+ # The directory the published generation's payload lives in.
153
+ #
154
+ # Read-side tasks resolve through this rather than the output root: an index
155
+ # publishing per-generation payloads keeps only `generation.json`, `dumps/`,
156
+ # `payloads/` and the lock files at the root. A flat index resolves to the
157
+ # root unchanged.
158
+ #
159
+ # @param output_dir [Pathname, String] index directory
160
+ # @return [Pathname]
161
+ def woods_payload_dir(output_dir)
162
+ Woods::Generation.new(output_dir: output_dir).payload_dir
163
+ end
164
+
165
+ # @param output_dir [Pathname, String] index directory
166
+ # @param wait [Numeric, nil] seconds to wait for the lock (default: the
167
+ # shared writer wait; injectable so a spec need not sit out the window)
168
+ # @return [Symbol] `:cleaned`, or `:refused` when a live daemon is
169
+ # maintaining this index
170
+ def woods_clean_index(output_dir, wait: nil)
171
+ require 'woods/watch/daemon'
172
+
173
+ output_dir = Pathname.new(output_dir)
174
+
175
+ unless woods_daemon_coverage(output_dir) == :none
176
+ warn 'ERROR: a watch daemon is maintaining this index — refusing to delete it out from under it.'
177
+ warn 'Stop the daemon first, or set WOODS_IGNORE_WATCH=1 to clean anyway.'
178
+ return :refused
179
+ end
180
+
181
+ lock_name = Woods::Watch::Daemon::LOCK_NAME
182
+ woods_with_extraction_lock(output_dir, wait: wait) do
183
+ woods_sweep_index_dir(output_dir, lock_name)
184
+ end
185
+
186
+ # The lock was released (and its file removed) above; drop the guard so
187
+ # the directory can empty, then remove the now-empty directory. Another
188
+ # writer may legitimately have recreated content between release and here
189
+ # — a non-empty directory is left alone rather than forced.
190
+ woods_remove_if_empty(output_dir, lock_name)
191
+ :cleaned
192
+ end
193
+
194
+ # Delete every index artifact except the lock file and its transaction
195
+ # guard. The guard lives inside the lock directory and a contender may hold
196
+ # its flock right now — deleting it mid-sweep would split the flock across
197
+ # two inodes and defeat the mutual exclusion it provides.
198
+ def woods_sweep_index_dir(output_dir, lock_name)
199
+ preserved = [
200
+ output_dir.join("#{lock_name}.lock"),
201
+ output_dir.join(Woods::Coordination::PipelineLock.guard_filename(lock_name))
202
+ ]
203
+ output_dir.children.each do |entry|
204
+ FileUtils.rm_rf(entry) unless preserved.include?(entry)
205
+ end
206
+ end
207
+
208
+ def woods_remove_if_empty(output_dir, lock_name)
209
+ FileUtils.rm_f(output_dir.join(Woods::Coordination::PipelineLock.guard_filename(lock_name)))
210
+ Dir.rmdir(output_dir)
211
+ rescue SystemCallError
212
+ nil
213
+ end
214
+
215
+ # The root containing the Rakefile that loaded this task file.
216
+ #
217
+ # `woods:watch_status` intentionally avoids Rails boot, so it cannot ask
218
+ # `Rails.root` for the conventional output path. `Dir.pwd` is not a stable
219
+ # substitute: `rake -f /app/Rakefile` and worktree launchers may invoke the
220
+ # task from somewhere else. Rake retains the selected Rakefile even when it
221
+ # does not chdir, which gives the same application root without loading the
222
+ # environment.
223
+ #
224
+ # @return [String] absolute directory containing the active Rakefile
225
+ def woods_task_root
226
+ rakefile = Rake.application.rakefile
227
+ return Rake.application.original_dir if rakefile.nil? || rakefile.empty?
228
+
229
+ # Rake may keep this relative (`Rakefile`) after searching upward from a
230
+ # nested invocation. At task execution its cwd is the directory it loaded
231
+ # that relative file from. An explicit absolute `-f` is already complete.
232
+ File.dirname(File.expand_path(rakefile))
233
+ end
234
+
235
+ # Changed paths across a git range, for `woods:incremental`'s CI branches.
236
+ #
237
+ # `git diff --name-only` split on lines corrupted three things at once: a
238
+ # path containing a newline split into two entries; a non-ASCII path came
239
+ # back octal-escaped inside quotes under git's default `core.quotePath`,
240
+ # which the dispatcher then can't match to any rule; and a rename reported
241
+ # only the new path, so the old path's unit was never pruned. `-z` +
242
+ # `--name-status` + `--no-renames` fixes all three: NUL-delimited records,
243
+ # `core.quotePath=false` unescaped, and a rename decomposed by git itself
244
+ # into a separate `A <new>` and `D <old>` record rather than one `R` record
245
+ # naming both.
246
+ #
247
+ # The diff is rooted at the extracted application (`git -C Rails.root`),
248
+ # consistent with {Woods::GitProvenance} (#262): extraction launched from
249
+ # another checkout must not diff that checkout's history.
250
+ #
251
+ # The child status is carried out, not discarded (M1): `Open3.capture2`
252
+ # turned an unresolvable range — a GitLab zero-SHA, an unfetched GitHub base
253
+ # ref, garbage — into an empty change set the caller could not tell apart
254
+ # from "nothing changed", and `woods:incremental` exited 0 over a sync that
255
+ # never ran.
256
+ #
257
+ # @param range [String] a git diff range/revision expression
258
+ # @param root [Pathname, String] repository root the diff runs against
259
+ # @return [Array(Array<String>, String, nil)] the changed paths (both halves
260
+ # of any rename included), or nil paths plus a human-readable failure when
261
+ # the range could not be resolved
262
+ def woods_changed_paths_for_range(range, root: Rails.root)
263
+ require 'open3'
264
+ output, error, status = Open3.capture3(
265
+ *Woods::GitCommand.argv(
266
+ root, '-c', 'core.quotePath=false',
267
+ 'diff', '--name-status', '-z', '--no-renames', range
268
+ )
269
+ )
270
+ return [woods_parse_git_diff_name_status(output), nil] if status.success?
271
+
272
+ [nil, "#{error.strip} (git exited #{status.exitstatus})"]
273
+ rescue SystemCallError => e
274
+ # No git binary at all (a slim container image, git removed after
275
+ # checkout). Errno::ENOENT out of `Open3.capture3` used to kill the task
276
+ # with a backtrace: non-zero, so safe, but it bypassed the decision below
277
+ # — the `:running`-daemon stand-down branch was unreachable, and the
278
+ # operator got a stack trace instead of the remediation text. An absent
279
+ # binary is an unresolvable range like any other (INF-12).
280
+ [nil, "git unavailable: #{e.message}"]
281
+ end
282
+
283
+ # The changed-path set `woods:incremental` will process, or a stand-down.
284
+ #
285
+ # An explicit `CHANGED_FILES` list bypasses git entirely. Otherwise the
286
+ # range comes from the CI environment (GitLab's before-SHA, GitHub's base
287
+ # ref) or defaults to the last commit.
288
+ #
289
+ # A failed range is a decision, not a skip (M1): the changed-file set is
290
+ # unknown, so silently extracting nothing would exit 0 over work that never
291
+ # happened and leave CI drift unbounded. When a `:running` daemon maintains
292
+ # the index, its start-up catch-up covers whatever changed, so standing down
293
+ # with a printed reason is safe; a degraded daemon covers nothing, so the
294
+ # run fails like any other uncovered case. The decision runs BEFORE the
295
+ # task's empty-range exit — a failed range must never be mistaken for an
296
+ # empty one.
297
+ #
298
+ # @param output_dir [Pathname, String] index directory, for daemon coverage
299
+ # @return [Array<String>] changed paths
300
+ def woods_incremental_changed_paths(output_dir)
301
+ explicit = ENV.fetch('CHANGED_FILES', nil)
302
+ return explicit.split(',').map(&:strip) if explicit
303
+
304
+ range = woods_incremental_range
305
+ changed_files, failure = woods_changed_paths_for_range(range)
306
+ return changed_files unless failure
307
+
308
+ if woods_daemon_coverage(output_dir) == :running
309
+ puts "Could not resolve the git diff range #{range.inspect} (#{failure})."
310
+ puts 'A watch daemon is maintaining this index, so its catch-up covers the changed paths — standing down.'
311
+ puts 'To extract now anyway, repair or provide the range (check the CI env refs),'
312
+ puts 'set CHANGED_FILES explicitly, or run a full woods:extract.'
313
+ exit 0
314
+ end
315
+
316
+ warn "ERROR: could not resolve the git diff range #{range.inspect}: #{failure}"
317
+ warn 'The changed-file set is unknown, so incremental extraction would silently index nothing.'
318
+ warn 'Fix the range (a zero SHA or an unfetched base ref resolve to nothing), or run a full woods:extract.'
319
+ exit 1
320
+ end
321
+
322
+ # The git range `woods:incremental` diffs, from the CI environment or the
323
+ # last-commit default.
324
+ #
325
+ # @return [String]
326
+ def woods_incremental_range
327
+ if ENV['CI_COMMIT_BEFORE_SHA']
328
+ # GitLab CI
329
+ "#{ENV['CI_COMMIT_BEFORE_SHA']}..#{ENV.fetch('CI_COMMIT_SHA', nil)}"
330
+ elsif ENV['GITHUB_BASE_REF']
331
+ # GitHub Actions PR
332
+ "origin/#{ENV['GITHUB_BASE_REF']}...HEAD"
333
+ else
334
+ # Default: changes since last commit
335
+ 'HEAD~1'
336
+ end
337
+ end
338
+
339
+ # Parse NUL-delimited `git diff --name-status -z --no-renames` output.
340
+ #
341
+ # Each record is `<status>\0<path>\0` — `--no-renames` guarantees a single
342
+ # path per record, since it is what stops git emitting a two-path `R`/`C`
343
+ # record in the first place. Status letters are not inspected beyond "did
344
+ # git report anything at all"; a deleted path still needs to reach the
345
+ # change set so its unit can be pruned.
346
+ #
347
+ # A bare `Open3.capture2` read is tagged with the process's default
348
+ # external encoding — US-ASCII under `LANG=C`, this daemon's usual
349
+ # environment (see `AtomicFile.read`'s gotcha) — so a UTF-8 path is a
350
+ # US-ASCII string containing invalid bytes until re-tagged.
351
+ #
352
+ # @param output [String] raw NUL-delimited git output
353
+ # @return [Array<String>] changed paths
354
+ def woods_parse_git_diff_name_status(output)
355
+ fields = output.dup.force_encoding(Encoding::UTF_8).split("\x00")
356
+ paths = []
357
+ fields.each_slice(2) do |status, path|
358
+ break if path.nil?
359
+
360
+ paths << path unless status.nil? || status.empty?
361
+ end
362
+ paths
363
+ end
364
+
365
+ # Resolve the configured retrieval stack and run one ad-hoc query (#178).
366
+ #
367
+ # Every backend — embedding provider, vector store, metadata store, graph
368
+ # store — is resolved through Woods::Builder from Woods.configuration,
369
+ # the same wiring `woods:embed` writes through
370
+ # (Woods::Tasks.build_embed_indexer). The old task body hardcoded
371
+ # Ollama + InMemory + SQLite + Memory, so on any other configured stack it
372
+ # queried backends the embed run never wrote to and silently returned
373
+ # nothing.
374
+ #
375
+ # In-memory stores start empty in a fresh process; hosts on the :local /
376
+ # :shared_filesystem presets should query through woods-mcp, which
377
+ # hydrates them from the dumps on disk.
378
+ #
379
+ # @param query [String] natural-language retrieval query
380
+ # @return [String] human-formatted retrieval output
381
+ def woods_run_retrieval(query)
382
+ require 'woods'
383
+ require 'woods/formatting/human_adapter'
384
+
385
+ config = Woods.configuration
386
+ retriever = Woods::Builder.new(config).build_retriever
387
+ result = retriever.retrieve(query, budget: config.max_context_tokens)
388
+
389
+ Woods::Formatting::HumanAdapter.new.format(result)
390
+ end
391
+ end
392
+ end
@@ -0,0 +1,71 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Woods
4
+ module Resilience
5
+ class GraphInvariantValidator
6
+ # Set-membership and index agreement checks, sharing the validator
7
+ # diagnostics and typed nodes. Keeping these separate leaves graph
8
+ # edge/variant validation in the main validator.
9
+ module MembershipChecks
10
+ private
11
+
12
+ def validate_membership(section, expected, scalar: false)
13
+ actual = object_section(section).each_with_object({}) do |(key, entries), result|
14
+ entries = [entries] if scalar && entries.is_a?(String)
15
+ values = membership_entries(key, entries, section)
16
+ result[key] = values if values
17
+ end
18
+ (expected.keys | actual.keys).each do |key|
19
+ compare_membership(section, key, expected.fetch(key, Set.new), actual.fetch(key, Set.new))
20
+ end
21
+ end
22
+
23
+ def compare_membership(section, key, wanted, recorded)
24
+ (wanted - recorded).each { |id| error("#{section}[#{key.inspect}]", "missing #{id.inspect}") }
25
+ (recorded - wanted).each { |id| error("#{section}[#{key.inspect}]", "unexpected #{id.inspect}") }
26
+ end
27
+
28
+ def membership_entries(key, entries, section)
29
+ return entries.to_set if name?(key) && entries.is_a?(Array) && entries.all? { |entry| name?(entry) }
30
+
31
+ error("#{section}[#{key.inspect}]", 'expected a named bucket containing identifiers')
32
+ nil
33
+ end
34
+
35
+ def validate_index_agreement
36
+ unless @index_entries.is_a?(Array)
37
+ error('unit indexes', 'expected typed index entries')
38
+ return
39
+ end
40
+
41
+ entries = {}
42
+ @index_entries.each { |entry| collect_index_entry(entry, entries) }
43
+ (@typed_nodes.keys - entries.keys).each do |identifier, type|
44
+ error('unit indexes', "graph node #{type}:#{identifier} is not indexed")
45
+ end
46
+ (entries.keys - @typed_nodes.keys).each do |identifier, type|
47
+ error('nodes', "missing indexed unit #{type}:#{identifier}")
48
+ end
49
+ end
50
+
51
+ def collect_index_entry(entry, entries)
52
+ unless typed_entry?(entry)
53
+ error('unit indexes', 'entry requires a nonempty identifier and type')
54
+ return
55
+ end
56
+ key = [entry['identifier'], entry['type']]
57
+ error('unit indexes', "duplicate typed entry #{key.last}:#{key.first}") if entries.key?(key)
58
+ entries[key] = entry
59
+ node = @typed_nodes[key]
60
+ return unless node && entry.key?('file_path') && node['file_path'] != entry['file_path']
61
+
62
+ error('unit indexes', "file_path differs for #{key.last}:#{key.first}")
63
+ end
64
+
65
+ def typed_entry?(entry)
66
+ entry.is_a?(Hash) && name?(entry['identifier']) && name?(entry['type'])
67
+ end
68
+ end
69
+ end
70
+ end
71
+ end
@@ -0,0 +1,61 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Woods
4
+ module Resilience
5
+ class GraphInvariantValidator
6
+ # Collect typed primary/variant identities without normalizing malformed
7
+ # records away; derive type and file membership from those identities.
8
+ module NodeChecks
9
+ private
10
+
11
+ def collect_nodes
12
+ object_section('nodes').each do |identifier, node|
13
+ type = register_node(identifier, node, "nodes[#{identifier.inspect}]")
14
+ @primary_types[identifier] = type if type
15
+ end
16
+ @variants = @graph.fetch('variants', [])
17
+ unless @variants.is_a?(Array)
18
+ error('variants', 'expected an array')
19
+ @variants = []
20
+ end
21
+ @variants.each_with_index { |record, index| collect_variant(record, index) }
22
+ end
23
+
24
+ def collect_variant(record, index)
25
+ label = "variants[#{index}]"
26
+ unless record.is_a?(Hash)
27
+ error(label, 'expected an object')
28
+ return
29
+ end
30
+
31
+ identifier = record['identifier']
32
+ error(label, "missing primary node for #{identifier.inspect}") unless @primary_types.key?(identifier)
33
+ register_node(identifier, record, label)
34
+ end
35
+
36
+ def register_node(identifier, node, label)
37
+ unless name?(identifier) && node.is_a?(Hash) && name?(node['type'])
38
+ error(label, 'expected a nonempty identifier and an object with a nonempty type')
39
+ return
40
+ end
41
+ type = node['type']
42
+ key = [identifier, type]
43
+ if @typed_nodes.key?(key)
44
+ error(label, "duplicate typed node #{type}:#{identifier}")
45
+ return
46
+ end
47
+
48
+ @typed_nodes[key] = node
49
+ @expected_types[type].add(identifier)
50
+ path = node['file_path']
51
+ if path.is_a?(String)
52
+ @expected_files[path].add(identifier)
53
+ elsif !path.nil?
54
+ error(label, 'file_path must be a string or null')
55
+ end
56
+ type
57
+ end
58
+ end
59
+ end
60
+ end
61
+ end
@@ -0,0 +1,46 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Woods
4
+ module Resilience
5
+ class GraphInvariantValidator
6
+ # The additive reverse relationship index must preserve every typed
7
+ # forward record. Legacy graphs can omit it entirely.
8
+ module ReverseRelationshipChecks
9
+ RELATIONSHIP_KEYS = %w[via through through_db disable_joins].freeze
10
+
11
+ private
12
+
13
+ def record_reverse_relationship(target, source, source_type, edge)
14
+ return unless name?(source) && name?(source_type)
15
+
16
+ attributes = edge.is_a?(Hash) ? edge.slice(*RELATIONSHIP_KEYS) : {}
17
+ record = { 'source' => source, 'source_type' => source_type, 'via' => nil }.merge(attributes)
18
+ @expected_reverse_via[target] << record
19
+ end
20
+
21
+ def validate_reverse_relationships
22
+ return unless @graph.key?('reverse_via')
23
+
24
+ actual = object_section('reverse_via')
25
+ (@expected_reverse_via.keys | actual.keys).each do |target|
26
+ compare_reverse_relationships(target, actual.fetch(target, []))
27
+ end
28
+ end
29
+
30
+ def compare_reverse_relationships(target, rows)
31
+ label = "reverse_via[#{target.inspect}]"
32
+ unless name?(target) && rows.is_a?(Array) && rows.all?(Hash)
33
+ error(label, 'expected a named bucket containing relationship objects')
34
+ return
35
+ end
36
+ expected = @expected_reverse_via.fetch(target, [])
37
+ keys = %w[source source_type] + RELATIONSHIP_KEYS
38
+ recorded = rows.map { |row| row.slice(*keys) }
39
+ return if expected.tally == recorded.tally
40
+
41
+ error(label, 'relationship records differ from typed forward edges')
42
+ end
43
+ end
44
+ end
45
+ end
46
+ end
@@ -0,0 +1,119 @@
1
+ # frozen_string_literal: true
2
+
3
+ require 'set'
4
+ require_relative 'graph_invariant_validator/membership_checks'
5
+ require_relative 'graph_invariant_validator/node_checks'
6
+ require_relative 'graph_invariant_validator/reverse_relationship_checks'
7
+
8
+ module Woods
9
+ module Resilience
10
+ # Checks raw published graph data against typed per-directory index entries.
11
+ # It deliberately does not use DependencyGraph.from_h: normalization there
12
+ # can discard malformed records before a validator has seen them.
13
+ #
14
+ # Target identifiers are not necessarily graph nodes (external/unresolved
15
+ # dependencies are valid), and reverse/file/type indexes contain bare names,
16
+ # not typed endpoints. Expected membership is the union over all variants.
17
+ class GraphInvariantValidator
18
+ include MembershipChecks
19
+ include NodeChecks
20
+ include ReverseRelationshipChecks
21
+
22
+ def initialize(graph:, index_entries:)
23
+ @graph = graph
24
+ @index_entries = index_entries
25
+ end
26
+
27
+ # @return [Array<String>] semantic errors, without changing either input
28
+ def validate
29
+ @errors = []
30
+ return ['dependency_graph.json: expected an object'] unless @graph.is_a?(Hash)
31
+
32
+ @typed_nodes = {}
33
+ @primary_types = {}
34
+ @expected_reverse = membership_map
35
+ @expected_reverse_via = Hash.new { |hash, key| hash[key] = [] }
36
+ @expected_files = membership_map
37
+ @expected_types = membership_map
38
+ collect_nodes
39
+ validate_edges
40
+ validate_reverse_relationships
41
+ validate_membership('reverse', @expected_reverse)
42
+ validate_membership('file_map', @expected_files, scalar: true)
43
+ validate_membership('type_index', @expected_types)
44
+ validate_index_agreement
45
+ @errors
46
+ end
47
+
48
+ private
49
+
50
+ def membership_map
51
+ Hash.new { |hash, key| hash[key] = Set.new }
52
+ end
53
+
54
+ def error(label, message)
55
+ @errors << "dependency_graph.json #{label}: #{message}"
56
+ end
57
+
58
+ def object_section(name)
59
+ value = @graph[name]
60
+ return value if value.is_a?(Hash)
61
+
62
+ error(name, 'expected an object')
63
+ {}
64
+ end
65
+
66
+ def name?(value)
67
+ value.is_a?(String) && !value.empty?
68
+ end
69
+
70
+ def validate_edges
71
+ edges = object_section('edges')
72
+ @primary_types.each_key do |identifier|
73
+ error('edges', "missing edge list for #{identifier.inspect}") unless edges.key?(identifier)
74
+ end
75
+ edges.each do |identifier, list|
76
+ label = "edges[#{identifier.inspect}]"
77
+ error(label, 'source has no primary node') unless @primary_types.key?(identifier)
78
+ validate_edge_list(identifier, @primary_types[identifier], list, label)
79
+ end
80
+ @variants.each_with_index do |record, index|
81
+ next unless record.is_a?(Hash)
82
+
83
+ validate_edge_list(record['identifier'], record['type'], record['edges'], "variants[#{index}].edges")
84
+ end
85
+ end
86
+
87
+ def validate_edge_list(source, source_type, list, label)
88
+ unless list.is_a?(Array)
89
+ error(label, 'expected an array')
90
+ return
91
+ end
92
+
93
+ list.each_with_index do |edge, index|
94
+ validate_edge(source, source_type, edge, "#{label}[#{index}]")
95
+ end
96
+ end
97
+
98
+ def validate_edge(source, source_type, edge, label)
99
+ target = edge.is_a?(Hash) ? edge['target'] : edge
100
+ unless name?(target)
101
+ error(label, 'expected a target identifier or an object with a target identifier')
102
+ return
103
+ end
104
+ validate_edge_attributes(edge, label) if edge.is_a?(Hash)
105
+ @expected_reverse[target].add(source) if name?(source)
106
+ record_reverse_relationship(target, source, source_type, edge)
107
+ end
108
+
109
+ def validate_edge_attributes(edge, label)
110
+ %w[via through through_db].each do |key|
111
+ error(label, "#{key} must be a string or null") unless edge[key].nil? || edge[key].is_a?(String)
112
+ end
113
+ return if [nil, true, false].include?(edge['disable_joins'])
114
+
115
+ error(label, 'disable_joins must be a boolean or null')
116
+ end
117
+ end
118
+ end
119
+ end