woods 1.6.4 → 2.0.0.beta1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (282) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +1879 -37
  3. data/CONTRIBUTING.md +195 -137
  4. data/README.md +162 -520
  5. data/SECURITY.md +92 -0
  6. data/assets/woods-wordmark-white-with-bg.png +0 -0
  7. data/docs/AGENT_GUIDE.md +204 -0
  8. data/docs/AGENT_SETUP.md +205 -0
  9. data/docs/BACKEND_MATRIX.md +470 -0
  10. data/docs/CONFIGURATION_REFERENCE.md +620 -0
  11. data/docs/CONSOLE_MCP_SETUP.md +829 -0
  12. data/docs/DOCKER_SETUP.md +454 -0
  13. data/docs/EMBEDDING_MODELS.md +136 -0
  14. data/docs/EVALUATION.md +91 -0
  15. data/docs/EXTRACTOR_REFERENCE.md +765 -0
  16. data/docs/FAQ.md +544 -0
  17. data/docs/GETTING_STARTED.md +183 -0
  18. data/docs/INCREMENTAL_EXTRACTION.md +415 -0
  19. data/docs/INTERNALS.md +415 -0
  20. data/docs/MCP_HTTP_TRANSPORT.md +144 -0
  21. data/docs/MCP_SERVERS.md +231 -0
  22. data/docs/MCP_TOOL_COOKBOOK.md +987 -0
  23. data/docs/MCP_WORKTREE_SETUP.md +127 -0
  24. data/docs/NOTION_INTEGRATION.md +283 -0
  25. data/docs/OBSIDIAN_INTEGRATION.md +170 -0
  26. data/docs/PUBLISHED_INDEX.md +197 -0
  27. data/docs/README.md +94 -0
  28. data/docs/RETRIEVAL_GUIDE.md +267 -0
  29. data/docs/TOKEN_BENCHMARK.md +68 -0
  30. data/docs/TROUBLESHOOTING.md +841 -0
  31. data/docs/UNBLOCKED_INTEGRATION.md +279 -0
  32. data/docs/UPGRADING_TO_2.md +321 -0
  33. data/docs/WATCH_DAEMON.md +667 -0
  34. data/docs/WHY_WOODS.md +219 -0
  35. data/exe/woods-console +39 -3
  36. data/exe/woods-console-mcp +21 -35
  37. data/exe/woods-mcp +20 -7
  38. data/exe/woods-mcp-http +78 -24
  39. data/exe/woods-mcp-start +57 -52
  40. data/lib/generators/woods/install_generator.rb +6 -5
  41. data/lib/generators/woods/pgvector_generator.rb +6 -3
  42. data/lib/generators/woods/templates/add_pgvector_to_woods.rb.erb +29 -9
  43. data/lib/generators/woods/templates/create_woods_tables.rb.erb +5 -1
  44. data/lib/generators/woods/templates/woods.rb.tt +49 -28
  45. data/lib/tasks/woods.rake +622 -168
  46. data/lib/tasks/woods_checks.rake +107 -0
  47. data/lib/tasks/woods_evaluation.rake +164 -80
  48. data/lib/woods/ast/call_site_extractor.rb +6 -15
  49. data/lib/woods/ast/method_extractor.rb +19 -9
  50. data/lib/woods/ast/parser.rb +54 -8
  51. data/lib/woods/atomic_file.rb +40 -1
  52. data/lib/woods/builder.rb +310 -22
  53. data/lib/woods/cache/cache_middleware.rb +18 -13
  54. data/lib/woods/cache/cache_store.rb +9 -1
  55. data/lib/woods/cache/solid_cache_store.rb +6 -4
  56. data/lib/woods/change_set.rb +88 -0
  57. data/lib/woods/checks/generation_resolution.rb +34 -0
  58. data/lib/woods/checks/moved_messages.rb +186 -0
  59. data/lib/woods/chunking/semantic_chunker.rb +160 -18
  60. data/lib/woods/console/audit_logger.rb +12 -3
  61. data/lib/woods/console/bridge_protocol.rb +3 -16
  62. data/lib/woods/console/connection_manager.rb +51 -136
  63. data/lib/woods/console/credential_index.rb +5 -53
  64. data/lib/woods/console/credential_scanner.rb +15 -16
  65. data/lib/woods/console/dispatch_pipeline.rb +46 -34
  66. data/lib/woods/console/embedded_executor.rb +806 -257
  67. data/lib/woods/console/eval_guard.rb +27 -20
  68. data/lib/woods/console/input_contract.rb +78 -0
  69. data/lib/woods/console/model_validator.rb +24 -6
  70. data/lib/woods/console/rack_middleware.rb +62 -63
  71. data/lib/woods/console/redactor.rb +10 -24
  72. data/lib/woods/console/safe_context.rb +45 -45
  73. data/lib/woods/console/scope_predicate_parser.rb +41 -0
  74. data/lib/woods/console/server.rb +136 -267
  75. data/lib/woods/console/sql_noise_stripper.rb +20 -51
  76. data/lib/woods/console/sql_table_scanner.rb +39 -90
  77. data/lib/woods/console/sql_validator.rb +455 -85
  78. data/lib/woods/console/table_gate.rb +2 -2
  79. data/lib/woods/console/tool_specs.rb +462 -88
  80. data/lib/woods/console/tools/tier1.rb +0 -3
  81. data/lib/woods/console/tools/tier4.rb +17 -7
  82. data/lib/woods/coordination/lock_heartbeat.rb +103 -0
  83. data/lib/woods/coordination/pipeline_lock.rb +263 -53
  84. data/lib/woods/db/migrations/007_typed_snapshot_units.rb +45 -0
  85. data/lib/woods/db/migrator.rb +3 -9
  86. data/lib/woods/db/schema_version.rb +47 -2
  87. data/lib/woods/dependency_graph.rb +898 -64
  88. data/lib/woods/embedding/fake.rb +138 -0
  89. data/lib/woods/embedding/indexer.rb +832 -40
  90. data/lib/woods/embedding/openai.rb +77 -19
  91. data/lib/woods/embedding/provider.rb +189 -11
  92. data/lib/woods/embedding/text_preparer.rb +1 -1
  93. data/lib/woods/embedding/token_counter.rb +0 -7
  94. data/lib/woods/evaluation/ablation_agent_payload.rb +38 -0
  95. data/lib/woods/evaluation/ablation_executor.rb +67 -0
  96. data/lib/woods/evaluation/ablation_provenance.rb +38 -0
  97. data/lib/woods/evaluation/ablation_report_writer.rb +43 -0
  98. data/lib/woods/evaluation/ablation_runner.rb +173 -0
  99. data/lib/woods/evaluation/ablation_summary.rb +65 -0
  100. data/lib/woods/evaluation/ablation_task.rb +66 -0
  101. data/lib/woods/evaluation/ablation_task_set.rb +77 -0
  102. data/lib/woods/evaluation/ablation_timed_executor.rb +91 -0
  103. data/lib/woods/evaluation/ablation_worktree.rb +71 -0
  104. data/lib/woods/evaluation/baseline.rb +60 -0
  105. data/lib/woods/evaluation/baseline_runner.rb +11 -3
  106. data/lib/woods/evaluation/evaluator.rb +41 -8
  107. data/lib/woods/evaluation/query_set.rb +79 -13
  108. data/lib/woods/evaluation/report_generator.rb +20 -1
  109. data/lib/woods/export/unit_facts.rb +0 -11
  110. data/lib/woods/extracted_unit.rb +22 -63
  111. data/lib/woods/extractor.rb +2503 -192
  112. data/lib/woods/extractors/action_cable_extractor.rb +9 -4
  113. data/lib/woods/extractors/ast_source_extraction.rb +20 -2
  114. data/lib/woods/extractors/caching_extractor.rb +46 -12
  115. data/lib/woods/extractors/callback_analyzer.rb +39 -9
  116. data/lib/woods/extractors/component_discovery.rb +123 -0
  117. data/lib/woods/extractors/concern_extractor.rb +17 -3
  118. data/lib/woods/extractors/controller_extractor.rb +389 -29
  119. data/lib/woods/extractors/decorator_extractor.rb +7 -14
  120. data/lib/woods/extractors/engine_extractor.rb +53 -8
  121. data/lib/woods/extractors/event_extractor.rb +55 -4
  122. data/lib/woods/extractors/factory_extractor.rb +49 -11
  123. data/lib/woods/extractors/graphql_extractor.rb +162 -66
  124. data/lib/woods/extractors/i18n_extractor.rb +6 -1
  125. data/lib/woods/extractors/job_extractor.rb +51 -21
  126. data/lib/woods/extractors/lib_extractor.rb +23 -17
  127. data/lib/woods/extractors/line_neutralizer.rb +171 -0
  128. data/lib/woods/extractors/mailer_extractor.rb +9 -1
  129. data/lib/woods/extractors/manager_extractor.rb +19 -2
  130. data/lib/woods/extractors/migration_extractor.rb +22 -11
  131. data/lib/woods/extractors/model_extractor.rb +292 -57
  132. data/lib/woods/extractors/package_extractor.rb +154 -0
  133. data/lib/woods/extractors/phlex_extractor.rb +18 -3
  134. data/lib/woods/extractors/policy_extractor.rb +6 -5
  135. data/lib/woods/extractors/poro_extractor.rb +13 -14
  136. data/lib/woods/extractors/pundit_extractor.rb +3 -3
  137. data/lib/woods/extractors/rails_source_extractor.rb +24 -7
  138. data/lib/woods/extractors/rake_task_extractor.rb +158 -30
  139. data/lib/woods/extractors/reference_patterns.rb +38 -0
  140. data/lib/woods/extractors/route_extractor.rb +58 -2
  141. data/lib/woods/extractors/scheduled_job_extractor.rb +51 -35
  142. data/lib/woods/extractors/serializer_extractor.rb +3 -4
  143. data/lib/woods/extractors/service_extractor.rb +11 -1
  144. data/lib/woods/extractors/shared_dependency_scanner.rb +24 -34
  145. data/lib/woods/extractors/shared_utility_methods.rb +36 -6
  146. data/lib/woods/extractors/source_nesting.rb +560 -0
  147. data/lib/woods/extractors/state_machine_extractor.rb +30 -18
  148. data/lib/woods/extractors/test_mapping_extractor.rb +26 -9
  149. data/lib/woods/extractors/view_component_extractor.rb +28 -3
  150. data/lib/woods/extractors/view_engines/erb.rb +17 -3
  151. data/lib/woods/feedback/gap_detector.rb +9 -3
  152. data/lib/woods/feedback/store.rb +7 -1
  153. data/lib/woods/filename_utils.rb +29 -1
  154. data/lib/woods/flow_analysis/operation_extractor.rb +22 -10
  155. data/lib/woods/flow_assembler.rb +63 -21
  156. data/lib/woods/flow_document.rb +1 -0
  157. data/lib/woods/flow_precomputer.rb +138 -22
  158. data/lib/woods/gem_mapper.rb +285 -0
  159. data/lib/woods/generation.rb +185 -0
  160. data/lib/woods/git_command.rb +38 -0
  161. data/lib/woods/git_provenance.rb +16 -2
  162. data/lib/woods/graph_analyzer.rb +408 -34
  163. data/lib/woods/index_artifact.rb +93 -23
  164. data/lib/woods/mcp/bearer_auth.rb +92 -22
  165. data/lib/woods/mcp/bootstrap_state.rb +77 -0
  166. data/lib/woods/mcp/bootstrapper.rb +582 -77
  167. data/lib/woods/mcp/config_resolver.rb +66 -6
  168. data/lib/woods/mcp/errors.rb +60 -0
  169. data/lib/woods/mcp/index_reader.rb +836 -117
  170. data/lib/woods/mcp/index_reader_pinning.rb +78 -0
  171. data/lib/woods/mcp/origin_guard.rb +108 -23
  172. data/lib/woods/mcp/protocol_policy.rb +98 -0
  173. data/lib/woods/mcp/provider_probe.rb +45 -6
  174. data/lib/woods/mcp/renderers/markdown_renderer.rb +72 -4
  175. data/lib/woods/mcp/renderers/plain_renderer.rb +54 -6
  176. data/lib/woods/mcp/server.rb +907 -154
  177. data/lib/woods/mcp/tasks/extension.rb +196 -0
  178. data/lib/woods/mcp/tasks/request_capture.rb +45 -0
  179. data/lib/woods/mcp/tasks/store.rb +518 -0
  180. data/lib/woods/mcp/tool_contract.rb +171 -0
  181. data/lib/woods/mcp/tool_response_renderer.rb +7 -0
  182. data/lib/woods/mcp/version_aware_tool_dispatch.rb +3 -9
  183. data/lib/woods/model_name_cache.rb +19 -1
  184. data/lib/woods/notion/client.rb +132 -36
  185. data/lib/woods/notion/exporter.rb +456 -61
  186. data/lib/woods/notion/mappers/column_mapper.rb +34 -5
  187. data/lib/woods/notion/mappers/migration_mapper.rb +32 -8
  188. data/lib/woods/notion/mappers/model_mapper.rb +21 -6
  189. data/lib/woods/notion/mappers/shared.rb +45 -3
  190. data/lib/woods/notion/sync_manifest.rb +258 -0
  191. data/lib/woods/obsidian/errors.rb +6 -0
  192. data/lib/woods/obsidian/name_mapper.rb +40 -24
  193. data/lib/woods/obsidian/vault_exporter.rb +103 -36
  194. data/lib/woods/operator/pipeline_guard.rb +118 -21
  195. data/lib/woods/operator/status_reporter.rb +20 -3
  196. data/lib/woods/path_dispatcher.rb +276 -0
  197. data/lib/woods/payload_store.rb +223 -0
  198. data/lib/woods/published_index/edge_shaper.rb +61 -0
  199. data/lib/woods/published_index/generation_catalog.rb +72 -0
  200. data/lib/woods/published_index/typed_unit_reader.rb +48 -0
  201. data/lib/woods/published_index.rb +287 -0
  202. data/lib/woods/railtie.rb +70 -38
  203. data/lib/woods/railtie_support.rb +167 -0
  204. data/lib/woods/release.rb +12 -0
  205. data/lib/woods/reload_policy.rb +206 -0
  206. data/lib/woods/resilience/circuit_breaker.rb +47 -8
  207. data/lib/woods/resilience/index_validator.rb +296 -10
  208. data/lib/woods/resilience/retryable_provider.rb +71 -6
  209. data/lib/woods/resolved_config.rb +55 -11
  210. data/lib/woods/retrieval/context_assembler.rb +132 -40
  211. data/lib/woods/retrieval/query_classifier.rb +25 -6
  212. data/lib/woods/retrieval/ranker.rb +193 -28
  213. data/lib/woods/retrieval/search_executor.rb +206 -39
  214. data/lib/woods/retriever.rb +317 -71
  215. data/lib/woods/retry_after.rb +22 -2
  216. data/lib/woods/ruby_analyzer/class_analyzer.rb +10 -14
  217. data/lib/woods/ruby_analyzer/fqn_builder.rb +2 -0
  218. data/lib/woods/ruby_analyzer/mermaid_renderer.rb +14 -4
  219. data/lib/woods/ruby_analyzer/method_analyzer.rb +1 -1
  220. data/lib/woods/ruby_analyzer.rb +21 -5
  221. data/lib/woods/session_tracer/file_store.rb +138 -19
  222. data/lib/woods/session_tracer/redis_store.rb +122 -12
  223. data/lib/woods/session_tracer/session_flow_assembler.rb +54 -11
  224. data/lib/woods/session_tracer/session_flow_document.rb +52 -6
  225. data/lib/woods/session_tracer/solid_cache_coordination.rb +192 -0
  226. data/lib/woods/session_tracer/solid_cache_store.rb +560 -91
  227. data/lib/woods/session_tracer/store.rb +14 -1
  228. data/lib/woods/storage/metadata_store.rb +230 -26
  229. data/lib/woods/storage/pgvector.rb +180 -22
  230. data/lib/woods/storage/qdrant.rb +367 -41
  231. data/lib/woods/storage/snapshotter/metadata.rb +79 -16
  232. data/lib/woods/storage/snapshotter/vector.rb +128 -17
  233. data/lib/woods/storage/snapshotter.rb +23 -5
  234. data/lib/woods/storage/vector_store.rb +49 -8
  235. data/lib/woods/storage_identity.rb +28 -0
  236. data/lib/woods/tasks.rb +53 -2
  237. data/lib/woods/temporal/json_snapshot_store.rb +112 -42
  238. data/lib/woods/temporal/snapshot_store.rb +139 -42
  239. data/lib/woods/unblocked/client.rb +119 -17
  240. data/lib/woods/unblocked/document_builder.rb +34 -2
  241. data/lib/woods/unblocked/exporter.rb +63 -27
  242. data/lib/woods/unblocked/rate_limiter.rb +23 -9
  243. data/lib/woods/unblocked/sync_manifest.rb +16 -8
  244. data/lib/woods/update_check.rb +24 -1
  245. data/lib/woods/util/uuid5.rb +124 -0
  246. data/lib/woods/version.rb +1 -1
  247. data/lib/woods/watch/daemon.rb +1345 -0
  248. data/lib/woods/watch/listen_watcher.rb +81 -0
  249. data/lib/woods/watch/polling_watcher.rb +137 -0
  250. data/lib/woods/watch/status.rb +169 -0
  251. data/lib/woods/watch/tree_scan.rb +163 -0
  252. data/lib/woods/watch/watcher.rb +100 -0
  253. data/lib/woods.rb +53 -9
  254. data/plugin/.claude-plugin/plugin.json +18 -0
  255. data/plugin/hooks/hooks.json +29 -0
  256. data/plugin/hooks/woods-post-edit.sh +226 -0
  257. data/plugin/hooks/woods-session-start.sh +77 -0
  258. data/plugin/skills/woods-agent-enable/SKILL.md +51 -0
  259. data/plugin/skills/woods-diagnose/SKILL.md +75 -0
  260. data/plugin/skills/woods-investigate/SKILL.md +39 -0
  261. data/plugin/skills/woods-mcp-config/SKILL.md +101 -0
  262. data/plugin/skills/woods-setup/SKILL.md +99 -0
  263. metadata +102 -30
  264. data/lib/woods/console/adapter_family.rb +0 -39
  265. data/lib/woods/console/adapters/cache_adapter.rb +0 -58
  266. data/lib/woods/console/adapters/good_job_adapter.rb +0 -33
  267. data/lib/woods/console/adapters/job_adapter.rb +0 -74
  268. data/lib/woods/console/adapters/sidekiq_adapter.rb +0 -33
  269. data/lib/woods/console/adapters/solid_queue_adapter.rb +0 -33
  270. data/lib/woods/console/bridge.rb +0 -210
  271. data/lib/woods/console/credential_scanner_registry.rb +0 -36
  272. data/lib/woods/console/encrypted_credential_snapshot.rb +0 -16
  273. data/lib/woods/console/sql_output_policy.rb +0 -535
  274. data/lib/woods/console/sqlite_read_guard.rb +0 -46
  275. data/lib/woods/formatting/claude_adapter.rb +0 -98
  276. data/lib/woods/formatting/generic_adapter.rb +0 -56
  277. data/lib/woods/formatting/gpt_adapter.rb +0 -64
  278. data/lib/woods/mcp/http_transport_options.rb +0 -15
  279. data/lib/woods/mcp/origin_policy.rb +0 -113
  280. data/lib/woods/notion/mapper.rb +0 -40
  281. data/lib/woods/observability/health_check.rb +0 -79
  282. data/lib/woods/observability/instrumentation.rb +0 -34
@@ -7,10 +7,12 @@ require 'open3'
7
7
  require 'pathname'
8
8
  require 'set'
9
9
 
10
+ require_relative 'atomic_file'
10
11
  require_relative 'filename_utils'
11
12
  require_relative 'token_utils'
12
13
  require_relative 'extracted_unit'
13
14
  require_relative 'dependency_graph'
15
+ require_relative 'payload_store'
14
16
  require_relative 'git_provenance'
15
17
  require_relative 'extractors/model_extractor'
16
18
  require_relative 'extractors/controller_extractor'
@@ -46,9 +48,13 @@ require_relative 'extractors/factory_extractor'
46
48
  require_relative 'extractors/test_mapping_extractor'
47
49
  require_relative 'extractors/poro_extractor'
48
50
  require_relative 'extractors/lib_extractor'
51
+ require_relative 'extractors/package_extractor'
49
52
  require_relative 'graph_analyzer'
50
53
  require_relative 'model_name_cache'
51
54
  require_relative 'flow_precomputer'
55
+ require_relative 'change_set'
56
+ require_relative 'generation'
57
+ require_relative 'path_dispatcher'
52
58
 
53
59
  module Woods
54
60
  # Extractor is the main orchestrator for codebase extraction.
@@ -66,6 +72,7 @@ module Woods
66
72
  #
67
73
  class Extractor
68
74
  include FilenameUtils
75
+ include Extractors::SourceNesting
69
76
 
70
77
  # Directories under app/ that contain classes we need to extract.
71
78
  # Used by eager_load_extraction_directories as a fallback when
@@ -126,7 +133,8 @@ module Woods
126
133
  test_mappings: Extractors::TestMappingExtractor,
127
134
  rails_source: Extractors::RailsSourceExtractor,
128
135
  poros: Extractors::PoroExtractor,
129
- libs: Extractors::LibExtractor
136
+ libs: Extractors::LibExtractor,
137
+ packages: Extractors::PackageExtractor
130
138
  }.freeze
131
139
 
132
140
  # Maps singular unit types (as stored in ExtractedUnit/graph nodes)
@@ -169,8 +177,17 @@ module Woods
169
177
  factory: :factories,
170
178
  test_mapping: :test_mappings,
171
179
  rails_source: :rails_source,
180
+ # RailsSourceExtractor emits BOTH :rails_source and :gem_source units
181
+ # into the rails_source/ output directory. Without this entry the
182
+ # wholesale-replacement path could not prune a stale gem_source unit
183
+ # ({EXTRACTOR_KEY_TO_TYPES} would not list the type) and {#remove_unit}
184
+ # could not resolve its JSON file on disk (#169) — the GraphQL history
185
+ # in CLAUDE.md is the cautionary tale for an extractor whose emitted
186
+ # types are not all mapped.
187
+ gem_source: :rails_source,
172
188
  poro: :poros,
173
- lib: :libs
189
+ lib: :libs,
190
+ package: :packages
174
191
  }.freeze
175
192
 
176
193
  # Maps unit types to class-based extractor methods (constantize + call).
@@ -203,13 +220,170 @@ module Woods
203
220
  # GraphQL types all use the same extractor method.
204
221
  GRAPHQL_TYPES = %i[graphql_type graphql_mutation graphql_resolver graphql_query].freeze
205
222
 
223
+ # File-based unit types the full path ALSO discovers by class, with the
224
+ # class-based entry point to re-extract them through. Re-extracting one
225
+ # of these by file is faithful only when the file names the unit; see
226
+ # {#re_extracted_units}.
227
+ CLASS_DISCOVERED_FALLBACK = { job: :extract_job_class }.freeze
228
+
229
+ # Unit types each extractor owns — the inverse of {TYPE_TO_EXTRACTOR_KEY}.
230
+ #
231
+ # Wholesale replacement of an extractor's output has to know every type it
232
+ # can produce, and one extractor can own several (`graphql` covers types,
233
+ # mutations, resolvers and queries).
234
+ #
235
+ # @return [Hash{Symbol => Array<Symbol>}]
236
+ EXTRACTOR_KEY_TO_TYPES = TYPE_TO_EXTRACTOR_KEY.each_with_object({}) do |(type, key), map|
237
+ (map[key] ||= []) << type
238
+ end.freeze
239
+
240
+ # Class-based extractors, which discover their units by walking runtime
241
+ # descendants rather than by globbing files. The incremental path
242
+ # reconciles each extractor's `discoverable_classes` against the
243
+ # identifiers already in the graph, so a class added since the last
244
+ # extraction is found without having to guess a constant name from a
245
+ # path (#164, gap 1).
246
+ #
247
+ # `reconcile_removals: false` opts an entry out of the removal half. Only
248
+ # GraphQL sets it, and the reason is that its unit type is produced by two
249
+ # independent discovery mechanisms whose union is the truth: runtime schema
250
+ # introspection *and* a static file pass over `app/graphql`. Every other
251
+ # entry here owns its type outright, which is what makes
252
+ # "in the graph but not in `discoverable_classes`" mean "deleted".
253
+ #
254
+ # For GraphQL that inference is wrong twice over. Without graphql-ruby
255
+ # loaded the runtime set is empty, so removal would delete every GraphQL
256
+ # unit in the index; with it loaded, a type defined in a file but not
257
+ # attached to the schema is legitimately absent from the type map, so
258
+ # removal would delete units a full extraction still emits. Additions are
259
+ # safe in both worlds — they are exactly the runtime-only types no changed
260
+ # path can dispatch to, which is the #167 divergence.
261
+ #
262
+ # The consequence, stated so it does not read as a bug later: a runtime-only
263
+ # type that *disappears* is never removed by an incremental run. It survives
264
+ # until a full extraction rebuilds the type. Confirmed against a live schema
265
+ # during review — probe units outlived deletion of the file that defined
266
+ # them. That is the cost of the opt-out, and it is the right side to err on:
267
+ # a stale unit is recoverable by one full run, whereas the removal half
268
+ # would delete units a full extraction still emits.
269
+ #
270
+ # @return [Hash{Symbol => Hash}] extractor key => { type:, method: }
271
+ CLASS_BASED_DISCOVERY = {
272
+ models: { type: :model, method: :extract_model },
273
+ controllers: { type: :controller, method: :extract_controller },
274
+ mailers: { type: :mailer, method: :extract_mailer },
275
+ components: { type: :component, method: :extract_component },
276
+ view_components: { type: :view_component, method: :extract_component },
277
+ action_cable_channels: { type: :action_cable_channel, method: :extract_channel },
278
+ # Additions only — see `reconcile_removals: false`. GraphQL is the one
279
+ # entry whose discovery set is not authoritative for its unit type
280
+ # (#167).
281
+ # `types:` because this extractor emits four unit types, not one:
282
+ # `classify_runtime_type` returns :graphql_type, :graphql_mutation,
283
+ # :graphql_resolver or :graphql_query. Declaring only :graphql_type made
284
+ # `known` miss the others — the schema's query root is always in
285
+ # `Schema.types` and classifies as :graphql_query — so they were
286
+ # re-"discovered" and re-added on *every* incremental run. That leaves
287
+ # `touched` non-empty on a genuine no-op, which rewrites the manifest and
288
+ # bumps the generation each cycle: the #164 gap-4 symptom, reintroduced.
289
+ graphql: { type: :graphql_type, types: GRAPHQL_TYPES,
290
+ method: :extract_from_runtime_type, reconcile_removals: false }
291
+ }.freeze
292
+
293
+ # Extractors with no per-file entry point: they scan the whole app (or
294
+ # introspect the whole runtime) in one pass, so an incremental run
295
+ # replaces their output wholesale rather than per unit. Before #164
296
+ # these types were simply skipped by incremental runs while
297
+ # `config/routes.rb` still triggered one — the run rewrote the manifest,
298
+ # zeroing the staleness clock, without updating the data.
299
+ #
300
+ # @return [Hash{Symbol => Symbol}] extractor key => unit type
301
+ WHOLE_APP_EXTRACTORS = {
302
+ routes: :route,
303
+ middleware: :middleware,
304
+ engines: :engine,
305
+ scheduled_jobs: :scheduled_job,
306
+ state_machines: :state_machine,
307
+ factories: :factory,
308
+ events: :event,
309
+ # DatabaseViewExtractor keeps only the highest `_vNN` of each Scenic
310
+ # view, so its unit set is a function of the whole directory rather
311
+ # than of each file independently: dispatching db/views/foo_v01.sql to
312
+ # the per-file method would index a version a full extraction drops.
313
+ database_views: :database_view,
314
+ # Framework/gem sources are a function of the installed dependency set,
315
+ # so `Gemfile.lock` is their trigger path (#169). Before this the only
316
+ # incremental writer was `woods:extract_framework`, which hand-wrote
317
+ # JSON around the whole pipeline — no AtomicFile, no _index.json, no
318
+ # manifest counts, no lock, no generation bump — and its output then
319
+ # sat stale forever. The value here names the primary unit type, like
320
+ # every other entry; the extractor also owns :gem_source, and
321
+ # {EXTRACTOR_KEY_TO_TYPES} (which wholesale replacement consults) lists
322
+ # both. Participation is gated by `include_framework_sources` — see
323
+ # {#skip_by_configuration?}.
324
+ rails_source: :rails_source,
325
+ # A package root decides which package every other unit belongs to,
326
+ # and the undeclared-edge report reads the whole declared set, so any
327
+ # package.yml change re-runs the extractor wholesale (#280). Task 8
328
+ # re-annotates unit membership in the same run.
329
+ packages: :package
330
+ }.freeze
331
+
332
+ # Extractors whose output embeds the route table, and which therefore go
333
+ # stale when routes change even though none of their own files did.
334
+ #
335
+ # ControllerExtractor writes each action's routes into unit metadata and
336
+ # into the action chunks; everything that includes `RouteHelperResolver`
337
+ # resolves `_path`/`_url` references into navigation edges against the
338
+ # same table. The dependency graph can't express this — a route unit
339
+ # depends *on* its controller, so walking dependents from
340
+ # `config/routes.rb` never reaches it — so the relationship is declared
341
+ # here instead and these types are re-extracted wholesale whenever the
342
+ # route set is re-run (#164).
343
+ #
344
+ # @return [Array<Symbol>] extractor keys
345
+ ROUTE_CONSUMER_EXTRACTORS = %i[
346
+ controllers
347
+ mailers
348
+ components
349
+ view_components
350
+ view_templates
351
+ ].freeze
352
+
353
+ # Payload artifacts that live at the top of a payload directory rather
354
+ # than inside a per-type directory. Used when seeding a payload from a
355
+ # flat index — the output root also holds `generation.json`, `dumps/`,
356
+ # `tasks/`, `woods.sqlite3` and `payloads/` itself, none of which belong
357
+ # to a generation's payload.
358
+ PAYLOAD_FILES = %w[manifest.json dependency_graph.json graph_analysis.json SUMMARY.md].freeze
359
+
360
+ # Payload directories that are not per-type unit directories.
361
+ PAYLOAD_DIRS = %w[flows].freeze
362
+
206
363
  attr_reader :output_dir, :dependency_graph
207
364
 
208
365
  def initialize(output_dir: nil)
209
366
  @output_dir = Pathname.new(output_dir || Rails.root.join('tmp/woods'))
367
+ @payload_store = PayloadStore.new(@output_dir)
368
+ @payload_dir = nil
369
+ @payload_generation = nil
210
370
  @dependency_graph = DependencyGraph.new
211
371
  @results = {}
212
372
  @extractors = {}
373
+ @publication_error = nil
374
+ end
375
+
376
+ # Where this run reads and writes payload artifacts.
377
+ #
378
+ # A run publishes into an immutable per-generation directory, so that the
379
+ # single atomic write of `generation.json` commits the whole payload at
380
+ # once. Falls back to the output root when no payload directory could be
381
+ # opened — a flat index is non-atomic but perfectly readable, and failing
382
+ # the extraction over it would be worse.
383
+ #
384
+ # @return [Pathname]
385
+ def payload_dir
386
+ @payload_dir || @output_dir
213
387
  end
214
388
 
215
389
  # ══════════════════════════════════════════════════════════════════════
@@ -222,6 +396,15 @@ module Woods
222
396
  def extract_all
223
397
  setup_output_directory
224
398
  ModelNameCache.reset!
399
+ # @package_resolver alone is not enough: #package_resolver builds
400
+ # through #extractor_for, which memoizes into @incremental_extractors.
401
+ # Without this reset, a second full run on the same instance would
402
+ # resolve membership through the first run's PackageExtractor and its
403
+ # already-memoized (now stale) package_files/package_roots, missing a
404
+ # package added between the two runs.
405
+ @package_resolver = nil
406
+ @incremental_extractors = nil
407
+ begin_payload!
225
408
 
226
409
  # Eager load once — all extractors need loaded classes for introspection.
227
410
  safe_eager_load!
@@ -237,8 +420,15 @@ module Woods
237
420
  Rails.logger.info '[Woods] Deduplicating results...'
238
421
  deduplicate_results
239
422
 
240
- # Rebuild graph from deduped results — Phase 1 registered all units including
241
- # duplicates, and DependencyGraph has no remove/unregister API.
423
+ # Phase 1.6: Package membership. Runs before the graph is rebuilt so
424
+ # registration copies metadata[:package] onto the node (#280).
425
+ annotate_packages
426
+
427
+ # Rebuild the graph from deduped results. #164 gave DependencyGraph
428
+ # `#remove`/`#unregister`, so surgical removal is now possible — but a
429
+ # full extraction has just registered every unit including duplicates,
430
+ # and rebuilding from the deduped set is both cheaper and less
431
+ # error-prone than unwinding registrations one at a time.
242
432
  @dependency_graph = DependencyGraph.new
243
433
  @results.each_value { |units| units.each { |u| @dependency_graph.register(u) } }
244
434
 
@@ -246,13 +436,15 @@ module Woods
246
436
  Rails.logger.info '[Woods] Resolving dependents...'
247
437
  resolve_dependents
248
438
 
249
- # Phase 3: Graph analysis (PageRank, structural metrics)
250
- Rails.logger.info '[Woods] Analyzing dependency graph...'
251
- @graph_analysis = GraphAnalyzer.new(@dependency_graph).analyze
252
-
253
- # Phase 4: Enrich with git data
439
+ # Phase 3: Enrich with git data. Runs BEFORE analysis now: the
440
+ # volatile_dependencies report reads commit counts off graph nodes.
254
441
  Rails.logger.info '[Woods] Enriching with git data...'
255
442
  enrich_with_git_data
443
+ annotate_graph_with_git_data
444
+
445
+ # Phase 4: Graph analysis (PageRank, structural metrics)
446
+ Rails.logger.info '[Woods] Analyzing dependency graph...'
447
+ @graph_analysis = build_graph_analyzer.analyze
256
448
 
257
449
  # Phase 4.5: Normalize file_path to relative paths
258
450
  Rails.logger.info '[Woods] Normalizing file paths...'
@@ -262,11 +454,28 @@ module Woods
262
454
  Rails.logger.info '[Woods] Writing output...'
263
455
  write_results
264
456
 
457
+ # Phase 5.1: Sweep unit files no current unit accounts for (#177). Must
458
+ # run after write_results — the just-written set is what defines
459
+ # "legitimate" — and belongs to the full path only; the incremental path
460
+ # deletes through the graph instead. See {#sweep_orphaned_unit_files}.
461
+ sweep_orphaned_unit_files
462
+
265
463
  # Phase 5.5: Precompute request flows (opt-in). Must run AFTER
266
464
  # write_results — FlowAssembler loads unit JSON from disk, so running
267
465
  # earlier assembled every flow from absent (fresh output dir) or
268
466
  # stale (previous run's) data. precompute_flows re-writes the
269
467
  # controller units it annotates with metadata[:flow_paths].
468
+ #
469
+ # Fail closed (M3 review round 3): a failure anywhere in the family —
470
+ # assembly, index write, annotation rewrite, sweep — raises and aborts
471
+ # the run here, BEFORE the graph, manifest, snapshot, and generation
472
+ # publish below. The previously published generation stays resolved
473
+ # and readable; the aborted run's payload directory is left
474
+ # unreachable for the next run's PayloadStore#create to reclaim.
475
+ # Publishing a partial flow index, stale prior flow artifacts
476
+ # alongside a new graph, or half-rewritten annotations would violate
477
+ # the atomic-generation and full/incremental-equivalence contracts;
478
+ # being opt-in does not excuse it.
270
479
  if Woods.configuration.precompute_flows
271
480
  Rails.logger.info '[Woods] Precomputing request flows...'
272
481
  precompute_flows
@@ -277,6 +486,7 @@ module Woods
277
486
  write_manifest
278
487
  write_structural_summary
279
488
  capture_snapshot
489
+ publish_generation('full')
280
490
 
281
491
  log_summary
282
492
 
@@ -287,57 +497,470 @@ module Woods
287
497
  # Incremental Extraction
288
498
  # ══════════════════════════════════════════════════════════════════════
289
499
 
290
- # Extract only units affected by changed files
291
- # Used for incremental indexing in CI
500
+ # Extract only units affected by changed files.
501
+ #
502
+ # Used for incremental indexing in CI and by any caller that maintains a
503
+ # live index. The goal is equivalence: after this returns, the on-disk
504
+ # index should match what a cold full extraction of the same tree would
505
+ # have produced.
506
+ #
507
+ # The run proceeds in a fixed order, and the order matters:
508
+ #
509
+ # 1. Blast radius is computed against the *pre-change* graph, so
510
+ # dependents of a file that just disappeared still get re-extracted.
511
+ # 2. Known units in the blast radius are re-extracted.
512
+ # 3. Changed paths the index has never seen are dispatched to an
513
+ # extractor by path ({PathDispatcher}).
514
+ # 4. Class-based types are reconciled against their runtime discovery
515
+ # sets, catching classes added since the last extraction.
516
+ # 5. Whole-app extractors whose trigger paths changed are re-run and
517
+ # their type replaced wholesale.
518
+ # 6. Units whose source file has vanished are pruned last, so anything
519
+ # resurrected by steps 2–5 against a deleted file is swept in the
520
+ # same run rather than surviving as a ghost until the next one.
292
521
  #
293
522
  # @param changed_files [Array<String>] List of changed file paths
294
- # @return [Array<String>] List of re-extracted unit identifiers
523
+ # @return [Array<String>] Identifiers of units re-extracted, added, or removed
295
524
  def extract_changed(changed_files)
296
- # Load existing graph
297
- graph_path = @output_dir.join('dependency_graph.json')
298
- @dependency_graph = DependencyGraph.from_h(JSON.parse(File.read(graph_path))) if graph_path.exist?
525
+ prepare_incremental_run
299
526
 
300
- ModelNameCache.reset!
301
-
302
- # Eager load to ensure newly-added classes are discoverable.
303
- safe_eager_load!
527
+ change_set = ChangeSet.new(paths: changed_files, root: Rails.root)
528
+ affected_types = Set.new
304
529
 
305
- # Normalize relative paths (from git diff) to absolute (as stored in file_map)
306
- absolute_files = changed_files.map do |f|
307
- Pathname.new(f).absolute? ? f : Rails.root.join(f).to_s
308
- end
530
+ # Blast radius from the pre-change graph.
531
+ affected_ids = @dependency_graph.affected_by(change_set.absolute_paths)
532
+ Rails.logger.info "[Woods] #{change_set.size} changed files affect #{affected_ids.size} units"
309
533
 
310
- # Compute affected units
311
- affected_ids = @dependency_graph.affected_by(absolute_files)
312
- Rails.logger.info "[Woods] #{changed_files.size} changed files affect #{affected_ids.size} units"
534
+ touched = reconcile_changed_paths(change_set, affected_types)
313
535
 
314
- # Re-extract affected units
315
- affected_types = Set.new
316
- affected_ids.each do |unit_id|
317
- re_extract_unit(unit_id, affected_types: affected_types)
536
+ (affected_ids - touched.to_a).each do |unit_id|
537
+ touched.add(unit_id) if re_extract_unit(unit_id, affected_types: affected_types)
318
538
  end
319
539
 
540
+ touched.merge(reconcile_class_based_types(affected_types))
541
+ touched.merge(rerun_whole_app_extractors(change_set, affected_types))
542
+ touched.merge(reannotate_packages(change_set, affected_types))
543
+ pruned = prune_vanished_units(change_set, affected_types)
544
+ touched.merge(pruned)
545
+
546
+ # Reconcile once more, because pruning can un-know a class the first pass
547
+ # skipped. A class-based file moved between autoload directories with its
548
+ # constant unchanged is still registered under the old path when
549
+ # reconciliation runs, so it looks known and is not re-extracted; the
550
+ # prune that follows then removes it for its vanished path. This pass
551
+ # re-adds it in the same run (M1) instead of leaving the unit missing
552
+ # until some later run happens to notice. Idempotent when nothing was
553
+ # pruned: the discovery set is compared against the graph, so an
554
+ # already-registered class is skipped.
555
+ #
556
+ # But not everything pruning removed may come back. `except:` keeps the
557
+ # *deletion* shape pruned: without a reload, a constant outlives the file
558
+ # that defined it — so deleting `app/models/user.rb` prunes `User`, and
559
+ # this pass finds `User` still in `ActiveRecord::Base.descendants`.
560
+ # Re-registering it would pin the unit to a path that no longer exists,
561
+ # and nothing could ever remove it: the sweep excludes class-based units
562
+ # and no future change set names that path again. A resident daemon
563
+ # processing a batch before its reload hits this every time. What
564
+ # separates the two shapes is the filesystem — only pruned identifiers
565
+ # that a still-existing file in the change set actually declares are
566
+ # re-addable. See {#readdable_pruned_classes}.
567
+ touched.merge(reconcile_class_based_types(
568
+ affected_types, except: pruned - readdable_pruned_classes(pruned, change_set)
569
+ ))
570
+
571
+ finalize_incremental_unit_json(affected_types)
572
+
320
573
  # Regenerate type indexes for affected types
321
574
  affected_types.each do |type_key|
322
575
  regenerate_type_index(type_key)
323
576
  end
324
577
 
325
- # Update graph, manifest, and summary. No capture_snapshot here:
326
- # snapshots must hash the FULL unit set, and incremental runs only
327
- # re-extract affected units (@results stays empty) — capturing would
328
- # record a snapshot whose diff reports every unit as deleted.
329
- # Snapshots are captured on full extraction only.
578
+ finalize_incremental_run(touched)
579
+
580
+ touched.to_a
581
+ end
582
+
583
+ # ══════════════════════════════════════════════════════════════════════
584
+ # Targeted Refresh
585
+ # ══════════════════════════════════════════════════════════════════════
586
+
587
+ # Re-run one or more extractors wholesale against the already-booted app,
588
+ # replacing every unit of the types they own.
589
+ #
590
+ # This is the escape hatch for the unit types that have no per-file entry
591
+ # point — routes are derived from `Rails.application.routes`, the
592
+ # middleware unit from the live stack, events from a two-pass scan of all
593
+ # of `app/`. Incremental runs reach them by trigger path
594
+ # ({PathDispatcher.whole_app_rules}); this reaches them by name, for a
595
+ # caller that already knows what went stale: a resident process that just
596
+ # reloaded, a `woods:refresh` invocation after editing `config/routes.rb`,
597
+ # a deploy hook after a dependency bump.
598
+ #
599
+ # "Full extraction required for these types" was always a cold-boot
600
+ # artifact rather than something inherent: in a booted process re-running
601
+ # one extractor is seconds.
602
+ #
603
+ # Any extractor key works, not just the whole-app ones — `refresh(:models)`
604
+ # is a legitimate way to re-derive every model after a schema change.
605
+ #
606
+ # @example After editing config/routes.rb
607
+ # Woods::Extractor.new(output_dir: "tmp/woods").refresh(:routes)
608
+ #
609
+ # @param keys [Array<Symbol>] keys into {EXTRACTORS}
610
+ # @return [Hash] `{ types:, touched:, unknown: }` — the extractors that
611
+ # ran, the identifiers written or removed, and any key that isn't an
612
+ # extractor
613
+ # @raise [ArgumentError] when no recognized key is given
614
+ def refresh(*keys)
615
+ keys = Array(keys).flatten.map(&:to_sym).uniq
616
+ known, unknown = keys.partition { |key| EXTRACTORS.key?(key) }
617
+ raise ArgumentError, "No known extractor in #{keys.inspect}" if known.empty?
618
+
619
+ known += ROUTE_CONSUMER_EXTRACTORS if known.include?(:routes)
620
+ known.uniq!
621
+
622
+ prepare_incremental_run
623
+ affected_types = Set.new
624
+ touched = known.each_with_object(Set.new) do |key, acc|
625
+ acc.merge(replace_type_wholesale(key, affected_types))
626
+ end
627
+
628
+ finalize_incremental_unit_json(affected_types)
629
+ affected_types.each { |type_key| regenerate_type_index(type_key) }
630
+ finalize_incremental_run(touched, reason: "refresh:#{known.sort.join(',')}")
631
+
632
+ { types: known, touched: touched.to_a, unknown: unknown }
633
+ end
634
+
635
+ # Raise when the most recent extraction run wrote a payload but could not
636
+ # publish its generation marker.
637
+ #
638
+ # The extractor records this failure instead of raising immediately so
639
+ # the resident watch daemon can keep its recoverable posture: it detects
640
+ # the unchanged generation, reports degraded, and carries the paths into
641
+ # a later cycle. One-shot callers have no later cycle, so the rake tasks
642
+ # call this method before reporting success and receive a typed non-zero
643
+ # failure instead of claiming an unreachable payload was published.
644
+ #
645
+ # @return [void]
646
+ # @raise [Woods::ExtractionError] when the generation marker could not be
647
+ # published; the previously published generation remains active
648
+ def raise_on_publication_failure!
649
+ raise @publication_error if @publication_error
650
+ end
651
+
652
+ private
653
+
654
+ # Load the persisted graph and reset the per-run bookkeeping that the
655
+ # incremental helpers read. Shared by {#extract_changed} and {#refresh};
656
+ # calling either without this leaves `@dependents_dirty` and
657
+ # `@incremental_written` holding a previous run's state.
658
+ #
659
+ # `begin_payload!(strict: true)`: an incremental write set is only the
660
+ # touched units, so a degrade to flat here (see {#begin_payload!}) would
661
+ # both read the wrong baseline graph below and publish an index missing
662
+ # every unit it didn't touch. Raises rather than degrading; the caller
663
+ # sees {Woods::ExtractionError} and the generation is left unbumped.
664
+ #
665
+ # @return [void]
666
+ # @raise [Woods::ExtractionError] see {#begin_payload!}
667
+ def prepare_incremental_run
668
+ begin_payload!(strict: true)
669
+ graph_path = payload_dir.join('dependency_graph.json')
670
+ ensure_incremental_baseline!(graph_path)
671
+ @dependency_graph = DependencyGraph.from_h(JSON.parse(AtomicFile.read(graph_path))) if graph_path.exist?
672
+
673
+ ModelNameCache.reset!
674
+ safe_eager_load!
675
+
676
+ @dependents_dirty = Set.new
677
+ @incremental_written = {}
678
+ @incremental_extractors = nil
679
+ @active_record_names = nil
680
+ @package_resolver = nil
681
+ end
682
+
683
+ # Write the graph and the derived artifacts after an incremental run.
684
+ #
685
+ # The manifest is rewritten only when the run actually changed something.
686
+ # `staleness_seconds` is derived from the manifest timestamp, so touching
687
+ # it after a no-op run reports the index as freshly synced when nothing
688
+ # was re-read — the misleading half of #164 gap 4. The graph write itself
689
+ # is unconditional, but on a no-op run it is inert, not meaningful: under
690
+ # payloads it lands in this run's not-yet-published directory, and a
691
+ # no-op run returns below before {#publish_generation}, so nothing ever
692
+ # points at what it just wrote — the next run's {PayloadStore#create}
693
+ # empties that same (unbumped-generation-numbered) directory before
694
+ # writing into it again. The graph's content is unchanged on a no-op run
695
+ # regardless (no touched units, no new edges), so nothing is lost; it is
696
+ # the write itself, not the recomputation, that goes nowhere.
697
+ #
698
+ # No `capture_snapshot` here: snapshots must hash the FULL unit set, and
699
+ # incremental runs only hold changed units in memory — capturing would
700
+ # record a snapshot whose diff reports every other unit as deleted.
701
+ #
702
+ # @param touched [Set<String>] identifiers added, re-extracted, or removed
703
+ # @param reason [String] what produced this run, recorded on the generation
704
+ # so `woods_status` can distinguish a targeted refresh from a file-driven
705
+ # incremental run — they have different blast radii and an operator
706
+ # reading "incremental" after a `woods:refresh[routes]` is being misled
707
+ # @return [void]
708
+ def finalize_incremental_run(touched, reason: 'incremental')
330
709
  write_dependency_graph
710
+
711
+ if touched.empty?
712
+ Rails.logger.info '[Woods] Incremental run changed nothing — leaving manifest timestamp untouched'
713
+ return
714
+ end
715
+
716
+ write_incremental_graph_analysis
717
+ refresh_incremental_flows(touched)
331
718
  write_manifest(incremental: true)
332
719
  write_structural_summary
333
- if Woods.configuration.enable_snapshots
334
- Rails.logger.info '[Woods] Skipping snapshot capture — snapshots are captured on full extraction only'
720
+ publish_generation(reason)
721
+
722
+ return unless Woods.configuration.enable_snapshots
723
+
724
+ Rails.logger.info '[Woods] Skipping snapshot capture — snapshots are captured on full extraction only'
725
+ end
726
+
727
+ # Publish a new generation — the last write of any successful run.
728
+ #
729
+ # Every extraction mode does this, so a long-lived reader can detect that
730
+ # the index moved without stat-ing the whole directory, whatever produced
731
+ # the change: a full run, an incremental run, a targeted refresh, or the
732
+ # watch daemon. Ordering is the contract: the generation goes last, so a
733
+ # reader that sees generation N knows N's files are already on disk. A run
734
+ # that raised, or that changed nothing, never reaches this.
735
+ #
736
+ # @param reason [String] what produced this generation
737
+ # @return [void]
738
+ def publish_generation(reason)
739
+ generation = Generation.new(output_dir: @output_dir)
740
+ marker = generation.bump!(reason: reason, payload: publishable_payload_name(generation))
741
+ prune_payloads(marker.number)
742
+ marker
743
+ rescue StandardError => e
744
+ # A failed bump must not fail the extraction that produced a perfectly
745
+ # good index. But "readers keep their current view until the next run" is
746
+ # too comfortable a way to put it: the generation *is* the freshness
747
+ # contract, so readers keep serving the old index for as long as the
748
+ # cause persists, and the next incremental may be a no-op that bumps
749
+ # nothing either. Error, not warn — and `Watch::Daemon` cross-checks that
750
+ # the number actually moved so the daemon reports degraded rather than
751
+ # running.
752
+ @publication_error = Woods::ExtractionError.new(
753
+ "Could not publish generation for #{@output_dir} (#{e.class}: #{e.message}); " \
754
+ 'the previous generation remains active'
755
+ )
756
+ Rails.logger.error "[Woods] #{@publication_error.message}"
757
+ nil
758
+ end
759
+
760
+ # Open the payload directory this run publishes into, seeded from the
761
+ # generation currently on disk.
762
+ #
763
+ # Seeding is what lets a run that touches ten files publish a whole
764
+ # generation: the unchanged artifacts are hardlinked in, and the run
765
+ # overwrites only what it changed. It also means every payload read during
766
+ # the run — the persisted graph, the per-type indexes, the unit JSON an
767
+ # incremental run patches — sees the previous generation, exactly as it
768
+ # did when the index was flat.
769
+ #
770
+ # A failure here degrades to the flat layout rather than failing the run
771
+ # — but only when a flat publish can actually be complete. A full
772
+ # extraction's write set is every unit in the app, so a flat publish from
773
+ # it is a whole index; the next successful run restores the payload
774
+ # boundary. `strict:` opts out of that degrade for {#extract_changed} and
775
+ # {#refresh}, whose write set is only the units they touched: over a
776
+ # payload-born index (`marker.payload` set), the flat root holds nothing
777
+ # newer than the last time this index was flat — possibly nothing at all
778
+ # — so publishing there would both compute the wrong incremental baseline
779
+ # ({#prepare_incremental_run} reads its graph from wherever `payload_dir`
780
+ # resolves to) and, even with the right baseline, redirect every reader
781
+ # to a near-empty directory missing every untouched unit. There is no
782
+ # complete flat index an incremental degrade can produce, so this raises
783
+ # instead — see {Woods::ExtractionError}. The generation is never bumped
784
+ # over a raised run, so readers keep serving the last good index.
785
+ #
786
+ # @param strict [Boolean] raise instead of degrading to a flat publish
787
+ # when the published generation names a payload
788
+ # @return [void]
789
+ # @raise [Woods::ExtractionError] when `strict` and a payload-born index's
790
+ # payload directory could not be opened for this run
791
+ def begin_payload!(strict: false)
792
+ @publication_error = nil
793
+ marker = Generation.new(output_dir: @output_dir).current
794
+ @payload_generation = marker.number + 1
795
+ @payload_dir = @payload_store.create(@payload_generation)
796
+ seed_payload(marker)
797
+ rescue StandardError => e
798
+ if strict && marker&.payload
799
+ raise Woods::ExtractionError, <<~MSG.tr("\n", ' ').strip
800
+ Could not open a payload directory for this incremental run
801
+ (#{e.class}: #{e.message}), and generation #{marker.number}'s payload
802
+ lives in #{marker.payload} rather than the flat output root. An
803
+ incremental run only writes the units it touched, so publishing flat
804
+ here would produce an index missing every untouched unit. Fix the
805
+ underlying filesystem issue (permissions, free space, a broken
806
+ mount), or run a full `woods:extract` to rebuild a complete flat
807
+ index before incremental runs resume.
808
+ MSG
335
809
  end
336
810
 
337
- affected_ids
811
+ Rails.logger.warn(
812
+ "[Woods] Could not open a payload directory (#{e.class}: #{e.message}) — publishing flat"
813
+ )
814
+ @payload_dir = nil
815
+ @payload_generation = nil
338
816
  end
339
817
 
340
- private
818
+ # Refuse an incremental run that has no baseline to be incremental
819
+ # against (CORE-2). With no published generation and no dependency
820
+ # graph — a failed CI cache restore, a typo'd WOODS_OUTPUT, a first run
821
+ # on a fresh runner — {#extract_changed} would compute an empty blast
822
+ # radius, dispatch only the diffed paths, and publish generation 1: a
823
+ # near-empty index that readers, `woods:validate`, and retrieval all
824
+ # treat as the complete truth of the app, and that nothing self-heals
825
+ # until a full extraction. The watch daemon already enforces this
826
+ # invariant on its side (a missing generation marker means one full
827
+ # extraction); the one-shot entry points must refuse rather than
828
+ # publish silently. A pre-generation flat index passes: its graph was
829
+ # seeded into this run's payload directory by {#begin_payload!}.
830
+ #
831
+ # @param graph_path [Pathname] the seeded payload's dependency graph
832
+ # @return [void]
833
+ # @raise [Woods::ExtractionError] when no baseline exists
834
+ def ensure_incremental_baseline!(graph_path)
835
+ return if graph_path.exist?
836
+ return if Generation.new(output_dir: @output_dir).current.number.positive?
837
+ # An embedding caller that already holds a populated graph (a prior
838
+ # in-process run, or seeded registrations) IS the baseline —
839
+ # {#prepare_incremental_run} keeps the in-memory graph whenever no
840
+ # disk graph exists.
841
+ return unless @dependency_graph.nil? || @dependency_graph.empty?
842
+
843
+ raise Woods::ExtractionError, <<~MSG.tr("\n", ' ').strip
844
+ No baseline index found under #{@output_dir}: no generation has been
845
+ published and no dependency_graph.json exists. An incremental run
846
+ only re-extracts what changed relative to an existing index, so
847
+ running it here would publish a near-empty index as the complete
848
+ truth of the application. Run a full `woods:extract` first (or point
849
+ WOODS_OUTPUT at the directory holding the existing index).
850
+ MSG
851
+ end
852
+
853
+ # Copy the published generation's payload into this run's directory.
854
+ #
855
+ # Two sources. A generation that already names a payload directory is
856
+ # cloned wholesale — by construction it holds payload artifacts and
857
+ # nothing else. A flat index (every index written before payloads, and any
858
+ # index whose last run degraded) is cloned entry by entry from an
859
+ # allowlist, because the output root also holds `generation.json`,
860
+ # `dumps/`, `tasks/`, `woods.sqlite3` and `payloads/` itself.
861
+ #
862
+ # @param marker [Woods::Generation::Marker] the published generation
863
+ # @return [void]
864
+ def seed_payload(marker)
865
+ if marker.payload && (source = @output_dir.join(marker.payload)).directory?
866
+ @payload_store.clone(source, @payload_dir)
867
+ else
868
+ seed_payload_from_flat_root
869
+ end
870
+ end
871
+
872
+ # Uses {PayloadStore#link_or_copy} rather than a bare +FileUtils.ln+ for
873
+ # the same reason {PayloadStore#clone} does: a filesystem that disallows
874
+ # hardlinks (EXDEV/EPERM/EMLINK/NotImplementedError) must still seed the
875
+ # payload, just by copying. Before this, that raise was caught by
876
+ # {#begin_payload!}'s rescue and degraded every run on such a filesystem
877
+ # to a flat publish — the fallback existed in {PayloadStore} but this,
878
+ # the only other file-level linker, never reached it.
879
+ #
880
+ # @return [void]
881
+ def seed_payload_from_flat_root
882
+ (PAYLOAD_FILES + payload_entry_dirs).each do |entry|
883
+ source = @output_dir.join(entry)
884
+ next unless source.exist?
885
+
886
+ @payload_store.clone(source, @payload_dir.join(entry)) if source.directory?
887
+ @payload_store.link_or_copy(source, @payload_dir.join(entry)) if source.file?
888
+ end
889
+ end
890
+
891
+ # @return [Array<String>] every directory name a payload can contain
892
+ def payload_entry_dirs
893
+ EXTRACTORS.keys.map(&:to_s) + PAYLOAD_DIRS
894
+ end
895
+
896
+ # The pointer to publish, or nil when this run built no payload directory.
897
+ #
898
+ # The directory was named from the generation number this run expected to
899
+ # publish as. Writers serialize on `PipelineLock`, so that prediction holds
900
+ # — but if it ever did not, the pointer would name a directory belonging to
901
+ # a different generation, so the directory is renamed to match rather than
902
+ # published under a name that lies.
903
+ #
904
+ # @param generation [Woods::Generation]
905
+ # @return [String, nil]
906
+ def publishable_payload_name(generation)
907
+ return nil unless @payload_dir
908
+
909
+ actual = generation.current.number + 1
910
+ if actual != @payload_generation
911
+ @payload_dir = rename_payload(actual)
912
+ @payload_generation = actual
913
+ end
914
+
915
+ PayloadStore.name_for(@payload_generation)
916
+ end
917
+
918
+ # @param number [Integer] the generation number actually being published
919
+ # @return [Pathname] the payload directory under its corrected name
920
+ def rename_payload(number)
921
+ destination = @payload_store.path_for(number)
922
+ FileUtils.rm_rf(destination.to_s)
923
+ FileUtils.mv(@payload_dir.to_s, destination.to_s)
924
+ destination
925
+ end
926
+
927
+ # @param published [Integer] the generation just published
928
+ # @return [void]
929
+ def prune_payloads(published)
930
+ return unless @payload_dir
931
+
932
+ @payload_store.prune(keep: payload_retention, protect: published)
933
+ rescue StandardError => e
934
+ Rails.logger.warn "[Woods] Payload retention failed: #{e.message}"
935
+ end
936
+
937
+ # @return [Integer] how many payload generations to keep on disk
938
+ def payload_retention
939
+ value = ENV.fetch('WOODS_PAYLOAD_RETENTION', nil).to_i
940
+ value.positive? ? value : PayloadStore::DEFAULT_RETENTION
941
+ end
942
+
943
+ # Recompute graph_analysis.json after an incremental graph write.
944
+ #
945
+ # Structural analysis (orphans, hubs, cycles, bridges) is derived purely
946
+ # from the graph, so leaving it at the last full extraction's values let
947
+ # it drift continuously between full runs (#164 gap 5).
948
+ #
949
+ # Re-raises rather than warn-and-continue: a caller that swallowed this
950
+ # went on to write the manifest and publish a fresh generation over an
951
+ # index whose graph_analysis.json never updated — a failed analysis write
952
+ # must abort the run the same way a failed extraction does, not ship
953
+ # under a generation that says everything landed.
954
+ #
955
+ # @return [void]
956
+ # @raise [StandardError] whatever GraphAnalyzer or the write raised
957
+ def write_incremental_graph_analysis
958
+ @graph_analysis = build_graph_analyzer.analyze
959
+ write_graph_analysis
960
+ rescue StandardError => e
961
+ Rails.logger.error "[Woods] Incremental graph analysis failed: #{e.message}"
962
+ raise
963
+ end
341
964
 
342
965
  # ──────────────────────────────────────────────────────────────────────
343
966
  # Eager Loading
@@ -351,9 +974,16 @@ module Woods
351
974
  # loads only the directories we actually need for extraction.
352
975
  def safe_eager_load!
353
976
  Rails.application.eager_load!
977
+ # Recorded because it decides whether the runtime discovery sets are
978
+ # *complete*. On the fallback path they are known-partial — whole
979
+ # directories may have failed to load — and treating "absent from
980
+ # descendants" as "deleted" would then erase live units by the type.
981
+ # See {#stale_class_based_units}.
982
+ @eager_load_complete = true
354
983
  rescue NameError => e
355
984
  Rails.logger.warn "[Woods] eager_load! hit NameError: #{e.message}"
356
985
  Rails.logger.warn '[Woods] Falling back to per-directory eager loading'
986
+ @eager_load_complete = false
357
987
  eager_load_extraction_directories
358
988
  end
359
989
 
@@ -387,8 +1017,36 @@ module Woods
387
1017
  # Extraction Strategies
388
1018
  # ──────────────────────────────────────────────────────────────────────
389
1019
 
1020
+ # Should this extractor be skipped under the current configuration?
1021
+ #
1022
+ # `include_framework_sources` (default: true — see
1023
+ # docs/CONFIGURATION_REFERENCE.md) was documented and consumed by nothing
1024
+ # (#169). It now gates both automatic paths at once: the full-run pass
1025
+ # over {EXTRACTORS} and the `Gemfile.lock` whole-app trigger
1026
+ # ({#rerun_whole_app_extractors}). The {PathDispatcher} rule itself stays
1027
+ # ungated — rules are memoized per-process while configuration is not,
1028
+ # and `relevant?` stays correct either way because `Gemfile.lock` already
1029
+ # triggers :engines and :middleware — so the gate sits where the
1030
+ # extractor would actually run. Gating only one side would either extract
1031
+ # units a knob-off full run does not produce (a phantom `touched` cycle
1032
+ # bumping the generation) or leave a knob-on host with framework sources
1033
+ # that never refresh on a dependency bump.
1034
+ #
1035
+ # A by-name {#refresh} (and its `woods:extract_framework` alias) is
1036
+ # deliberately NOT gated: the knob controls automatic participation, and
1037
+ # the explicit refresh is the escape hatch for hosts that want framework
1038
+ # sources on demand without paying for them every run.
1039
+ #
1040
+ # @param key [Symbol] key into {EXTRACTORS}
1041
+ # @return [Boolean]
1042
+ def skip_by_configuration?(key)
1043
+ key == :rails_source && !Woods.configuration.include_framework_sources
1044
+ end
1045
+
390
1046
  def extract_all_sequential
391
1047
  EXTRACTORS.each do |type, extractor_class|
1048
+ next if skip_by_configuration?(type)
1049
+
392
1050
  Rails.logger.info "[Woods] Extracting #{type}..."
393
1051
  start_time = Time.current
394
1052
 
@@ -427,7 +1085,9 @@ module Woods
427
1085
  ModelNameCache.short_names_regex if ModelNameCache.respond_to?(:short_names_regex)
428
1086
 
429
1087
  results_mutex = Mutex.new
430
- threads = EXTRACTORS.map do |type, extractor_class|
1088
+ failures = []
1089
+ active = EXTRACTORS.reject { |type, _| skip_by_configuration?(type) }
1090
+ threads = active.map do |type, extractor_class|
431
1091
  Thread.new do
432
1092
  Rails.logger.info "[Woods] [Thread] Extracting #{type}..."
433
1093
  start_time = Time.current
@@ -445,12 +1105,21 @@ module Woods
445
1105
  end
446
1106
  rescue StandardError => e
447
1107
  Rails.logger.error "[Woods] [Thread] #{type} failed: #{e.message}"
448
- results_mutex.synchronize { @results[type] = [] }
1108
+ results_mutex.synchronize { failures << [type, e] }
449
1109
  end
450
1110
  end
451
1111
 
452
1112
  threads.each(&:join)
453
1113
 
1114
+ # Fail closed, matching sequential extraction: a raise there aborts
1115
+ # before registering anything for the run. Silently substituting `[]`
1116
+ # for a failed type let the run finish, register a partial graph, and
1117
+ # publish it under a generation that says every type succeeded.
1118
+ if failures.any?
1119
+ names = failures.map { |(type, _e)| type }.sort.join(', ')
1120
+ raise Woods::ExtractionError, "Concurrent extraction failed for: #{names}"
1121
+ end
1122
+
454
1123
  # Register into dependency graph sequentially — DependencyGraph is not thread-safe
455
1124
  EXTRACTORS.each_key do |type|
456
1125
  (@results[type] || []).each { |unit| @dependency_graph.register(unit) }
@@ -464,7 +1133,7 @@ module Woods
464
1133
  def setup_output_directory
465
1134
  FileUtils.mkdir_p(@output_dir)
466
1135
  EXTRACTORS.each_key do |type|
467
- FileUtils.mkdir_p(@output_dir.join(type.to_s))
1136
+ FileUtils.mkdir_p(payload_dir.join(type.to_s))
468
1137
  end
469
1138
  end
470
1139
 
@@ -472,56 +1141,122 @@ module Woods
472
1141
  # Dependency Resolution
473
1142
  # ──────────────────────────────────────────────────────────────────────
474
1143
 
1144
+ # `unit_map` is identifier => Array<unit>, not identifier => unit (#225).
1145
+ # An identifier is not unique across types (a Scenic view and a factory
1146
+ # can both be `reports`), so a single-valued map let the later
1147
+ # registration overwrite the earlier one and every dependent land on
1148
+ # whichever unit happened to be indexed last — the other unit serialized
1149
+ # `dependents: []` forever. {#rewrite_unit_json_of_type} already treats
1150
+ # `dependents` as a property of the identifier and writes the same list
1151
+ # to every type that identifier owns; this brings the full-extraction
1152
+ # path into agreement with it.
475
1153
  def resolve_dependents
476
1154
  # Build complete unit map first (cross-type dependencies require all units indexed).
477
- unit_map = @results.each_with_object({}) do |(_type, units), map|
478
- units.each { |u| map[u.identifier] = u }
1155
+ unit_map = @results.each_with_object(Hash.new { |h, k| h[k] = [] }) do |(_type, units), map|
1156
+ units.each { |u| map[u.identifier] << u }
479
1157
  end
480
1158
 
481
1159
  # Resolve dependents using the complete map.
482
1160
  @results.each_value do |units|
483
1161
  units.each do |unit|
484
1162
  unit.dependencies.each do |dep|
485
- target_unit = unit_map[dep[:target]]
486
- next unless target_unit
487
-
488
- target_unit.dependents ||= []
489
- target_unit.dependents << {
490
- type: unit.type,
491
- identifier: unit.identifier
492
- }
1163
+ unit_map[dep[:target]].each do |target_unit|
1164
+ target_unit.dependents ||= []
1165
+ target_unit.dependents << {
1166
+ type: unit.type,
1167
+ identifier: unit.identifier
1168
+ }
1169
+ end
493
1170
  end
494
1171
  end
495
1172
  end
496
1173
  end
497
1174
 
498
1175
  # Remove duplicate units (same identifier) within each type, keeping the first occurrence.
499
- # Duplicates arise when multiple extractors produce the same unit (e.g., engine-mounted
500
- # routes duplicating app routes). Without dedup, downstream phases would produce inflated
501
- # counts, duplicate _index.json entries, and last-writer-wins file overwrites.
1176
+ # Duplicates arise when the same identifier is legitimately re-derived from the same
1177
+ # source (e.g., engine-mounted routes duplicating app routes: both carry no file path).
1178
+ # Without dedup, downstream phases would produce inflated counts, duplicate
1179
+ # _index.json entries, and last-writer-wins file overwrites.
1180
+ #
1181
+ # A duplicate derived from a DIFFERENT file is not a duplicate — it is two distinct
1182
+ # real-world constants collapsing onto one identifier (finding G-1: wrapper-named
1183
+ # sources all indexed as the wrapper class). Only one could ever be indexed; the
1184
+ # other silently vanished. That is unrepresentable — `variants:` carries cross-type
1185
+ # graph identity only — so extraction aborts with both files named.
502
1186
  def deduplicate_results
503
1187
  @results.each do |type, units|
504
- deduped = units.uniq(&:identifier)
505
- dropped = units.size - deduped.size
1188
+ @results[type] = deduplicate_type_units(type, units)
1189
+ end
1190
+ end
1191
+
1192
+ # First-occurrence dedup for one type, fail-closed on cross-file collisions.
1193
+ #
1194
+ # A retained identifier remembers its source file. A later unit with the
1195
+ # same identifier and the same file (or the same absence of one — runtime
1196
+ # and engine units carry nil) is the legitimate re-derivation {#deduplicate_results}
1197
+ # documents and is dropped. A later unit from a different file raises
1198
+ # {Woods::ExtractionError} listing both paths.
1199
+ #
1200
+ # @param type [Symbol] Result type, for the log line and error message
1201
+ # @param units [Array<ExtractedUnit>]
1202
+ # @return [Array<ExtractedUnit>] Deduplicated units, first occurrence order
1203
+ # @raise [Woods::ExtractionError] when one type+identifier is derived
1204
+ # from two different files
1205
+ def deduplicate_type_units(type, units)
1206
+ deduped = []
1207
+ dropped = 0
1208
+ retained_paths = {}
1209
+
1210
+ units.each do |unit|
1211
+ if retained_paths.key?(unit.identifier)
1212
+ prior_path = retained_paths[unit.identifier]
1213
+ if unit.file_path != prior_path
1214
+ raise Woods::ExtractionError, same_type_collision_message(type, unit, prior_path)
1215
+ end
506
1216
 
507
- Rails.logger.warn "[Woods] Deduplicated #{type}: dropped #{dropped} duplicate(s)" if dropped.positive?
1217
+ dropped += 1
1218
+ next
1219
+ end
508
1220
 
509
- @results[type] = deduped
1221
+ retained_paths[unit.identifier] = unit.file_path
1222
+ deduped << unit
510
1223
  end
1224
+
1225
+ Rails.logger.warn "[Woods] Deduplicated #{type}: dropped #{dropped} duplicate(s)" if dropped.positive?
1226
+
1227
+ deduped
1228
+ end
1229
+
1230
+ # @param type [Symbol]
1231
+ # @param unit [ExtractedUnit] The colliding (later) unit
1232
+ # @param prior_path [String, nil] File path of the retained unit
1233
+ # @return [String]
1234
+ def same_type_collision_message(type, unit, prior_path)
1235
+ "same-type identifier collision: #{type.to_s.singularize} '#{unit.identifier}' derived from " \
1236
+ "two different sources ('#{prior_path || 'no file'}' and '#{unit.file_path || 'no file'}'); " \
1237
+ 'only one unit could ever be indexed, so extraction aborted — either merge the ' \
1238
+ 'declarations into one file or split them into distinct constants'
511
1239
  end
512
1240
 
513
1241
  # ──────────────────────────────────────────────────────────────────────
514
1242
  # Flow Precomputation
515
1243
  # ──────────────────────────────────────────────────────────────────────
516
1244
 
1245
+ # Full-path counterpart of {#refresh_incremental_flows}: assemble every
1246
+ # controller's flows, annotate the in-memory units, rewrite them, and
1247
+ # sweep orphans. Failure-closed — any error raises to {#extract_all},
1248
+ # which aborts before write_manifest/publish_generation, so a partial
1249
+ # flow index or half-rewritten annotations can never be published.
1250
+ #
1251
+ # @return [void]
1252
+ # @raise [Woods::ExtractionError] when any flow-family step fails
517
1253
  def precompute_flows
518
1254
  all_units = @results.values.flatten(1)
519
- precomputer = FlowPrecomputer.new(units: all_units, graph: @dependency_graph, output_dir: @output_dir.to_s)
1255
+ precomputer = FlowPrecomputer.new(units: all_units, graph: @dependency_graph, output_dir: payload_dir.to_s)
520
1256
  flow_map = precomputer.precompute
521
1257
  rewrite_flow_annotated_units
1258
+ sweep_orphaned_flow_files
522
1259
  Rails.logger.info "[Woods] Precomputed #{flow_map.size} request flows"
523
- rescue StandardError => e
524
- Rails.logger.error "[Woods] Flow precomputation failed: #{e.message}"
525
1260
  end
526
1261
 
527
1262
  # Precompute runs after write_results (FlowAssembler reads unit JSON
@@ -540,20 +1275,217 @@ module Woods
540
1275
  annotated = units.select { |u| u.metadata[:flow_paths] }
541
1276
  next if annotated.empty?
542
1277
 
543
- type_dir = @output_dir.join(type.to_s)
1278
+ type_dir = payload_dir.join(type.to_s)
544
1279
  annotated.each do |unit|
545
- File.write(
1280
+ AtomicFile.write(
546
1281
  type_dir.join(collision_safe_filename(unit.identifier)),
547
1282
  json_serialize(unit.to_h)
548
1283
  )
549
1284
  end
550
- File.write(
1285
+ AtomicFile.write(
551
1286
  type_dir.join('_index.json'),
552
1287
  json_serialize(type_index_entries(units))
553
1288
  )
554
1289
  end
555
1290
  end
556
1291
 
1292
+ # Incremental counterpart of Phase 5.5 (M3): the run's controller delta
1293
+ # gets the same flow annotations a full run would give it, flow
1294
+ # documents and flow_index.json are refreshed for the touched
1295
+ # controllers while untouched controllers' entries carry forward, and
1296
+ # flows/ documents no entry references are swept. Without this, a
1297
+ # controller re-extracted incrementally lost metadata[:flow_paths],
1298
+ # flow_index.json described pre-change routes, and flows/ files for
1299
+ # deleted or renamed controllers persisted across every generation —
1300
+ # seeded forward by PAYLOAD_DIRS.
1301
+ #
1302
+ # Runs inside {#finalize_incremental_run} so both extract_changed and
1303
+ # refresh (whose routes cascade rewrites controllers wholesale) get it,
1304
+ # and so a no-op run — which returns before this — leaves flows alone.
1305
+ #
1306
+ # Skip rule: the refresh participates whenever the gate is on and
1307
+ # something is touched. A genuinely absent family — no flows/ directory,
1308
+ # or an empty one — skips: there is nothing to carry forward, publication
1309
+ # proceeds, and the next full extraction with the gate on builds the
1310
+ # family. Once the family holds ANY artifact it is authoritative: a
1311
+ # missing flow_index.json among documents, a corrupt one, a failed
1312
+ # rehydration, write, patch, sweep, or type-index regeneration raises,
1313
+ # and the raise propagates out of {#finalize_incremental_run} BEFORE
1314
+ # {#publish_generation} — no generation bump, the preceding generation
1315
+ # stays resolved and readable. ( woods:validate applies the same rule
1316
+ # for a populated family without an index.)
1317
+ #
1318
+ # @param touched [Set<String>] identifiers added, re-extracted, or removed
1319
+ # @return [void]
1320
+ # @raise [Woods::ExtractionError] when authoritative flow state is
1321
+ # missing or corrupt, or a flow-family write fails
1322
+ def refresh_incremental_flows(touched)
1323
+ return unless Woods.configuration.precompute_flows
1324
+ return if touched.empty?
1325
+ return unless flow_family_present?
1326
+
1327
+ controllers_dir = payload_dir.join('controllers')
1328
+ reextracted = touched.select { |id| controllers_dir.join(collision_safe_filename(id)).exist? }
1329
+ removed = previous_flow_index_controllers & (touched.to_set - reextracted.to_set)
1330
+ return if reextracted.empty? && removed.empty?
1331
+
1332
+ Rails.logger.info "[Woods] Refreshing flows for #{reextracted.size} controller(s), " \
1333
+ "#{removed.size} removed..."
1334
+ precomputer = FlowPrecomputer.new(units: [], graph: @dependency_graph, output_dir: payload_dir.to_s)
1335
+ annotations = precomputer.recompute_delta(
1336
+ touched_units: reextracted.filter_map { |id| unit_from_payload(:controllers, id) },
1337
+ removed_identifiers: removed.to_a
1338
+ )
1339
+ patch_flow_annotations(annotations)
1340
+ sweep_orphaned_flow_files
1341
+ # The annotation patch changed controller JSON after the run's type
1342
+ # index regeneration; the index carries estimated_tokens, which the
1343
+ # flow_paths are part of — a full run builds its index from the
1344
+ # annotated in-memory units, so the incremental one re-derives it from
1345
+ # the annotated files to match. A failure here raises like every
1346
+ # other refresh failure.
1347
+ regenerate_type_index(:controllers) if annotations.any?
1348
+ end
1349
+
1350
+ # Does the run's seeded payload hold a flow family at all? An absent
1351
+ # family (no flows/ directory, or an empty one) is a genuine absence —
1352
+ # typically an index built while the gate was off — and skips the
1353
+ # refresh rather than publishing a delta-only index.
1354
+ #
1355
+ # @return [Boolean]
1356
+ def flow_family_present?
1357
+ flows_dir = payload_dir.join('flows')
1358
+ return false unless flows_dir.directory?
1359
+
1360
+ !Dir[flows_dir.join('*.json')].empty?
1361
+ end
1362
+
1363
+ # Controller identifiers that hold entries in the previous generation's
1364
+ # flow_index.json. Only these can be flow-removed this run: a controller
1365
+ # with no index entries has no documents to sweep and no annotation to
1366
+ # clear.
1367
+ #
1368
+ # The read is authoritative ({#flow_family_present?} guaranteed the
1369
+ # family holds artifacts): a missing index among documents, or a corrupt
1370
+ # one, raises so publication aborts.
1371
+ #
1372
+ # @return [Set<String>]
1373
+ # @raise [Woods::ExtractionError] when the index is missing or corrupt
1374
+ def previous_flow_index_controllers
1375
+ index_path = payload_dir.join('flows', 'flow_index.json')
1376
+ raise Woods::ExtractionError, 'flows/ is populated but flow_index.json is missing' unless index_path.exist?
1377
+
1378
+ JSON.parse(AtomicFile.read(index_path))
1379
+ .keys.to_set { |entry_point| entry_point.to_s.split('#', 2).first }
1380
+ rescue JSON::ParserError => e
1381
+ raise Woods::ExtractionError, "previous flow_index.json does not parse: #{e.message}"
1382
+ end
1383
+
1384
+ # Rehydrate one unit from its payload JSON for the incremental flow
1385
+ # pass. FlowPrecomputer only needs the identifier and the actions
1386
+ # metadata; the source lives on disk, where FlowAssembler reads it.
1387
+ #
1388
+ # @param type_key [Symbol] extractor key naming the payload directory
1389
+ # @param identifier [String]
1390
+ # @return [ExtractedUnit, nil]
1391
+ # @raise [Woods::ExtractionError] when the unit JSON does not parse —
1392
+ # skipping it would publish annotations against stale state
1393
+ def unit_from_payload(type_key, identifier)
1394
+ path = payload_dir.join(type_key.to_s, collision_safe_filename(identifier))
1395
+ return unless File.exist?(path)
1396
+
1397
+ data = JSON.parse(AtomicFile.read(path))
1398
+ unit = ExtractedUnit.new(
1399
+ type: data['type'],
1400
+ identifier: data['identifier'] || identifier,
1401
+ file_path: data['file_path']
1402
+ )
1403
+ unit.metadata = data['metadata'] || {}
1404
+ unit.source_code = data['source_code']
1405
+ unit
1406
+ rescue JSON::ParserError => e
1407
+ raise Woods::ExtractionError, "could not rehydrate #{identifier} for flow refresh: #{e.message}"
1408
+ end
1409
+
1410
+ # Write metadata[:flow_paths] into the re-extracted controllers' unit
1411
+ # JSON — the incremental counterpart of {#rewrite_flow_annotated_units}.
1412
+ # A controller with no flows this run loses any annotation a previous
1413
+ # run had written, which is what a full run would have produced for the
1414
+ # same tree. Read-compare-write, like {#rewrite_unit_json_of_type}; a
1415
+ # read or write failure raises so publication aborts.
1416
+ #
1417
+ # @param annotations [Hash{String => Hash{String => String}}] from
1418
+ # {FlowPrecomputer#recompute_delta}
1419
+ # @return [void]
1420
+ def patch_flow_annotations(annotations)
1421
+ annotations.each do |identifier, flow_paths|
1422
+ path = payload_dir.join('controllers', collision_safe_filename(identifier))
1423
+ next unless File.exist?(path)
1424
+
1425
+ data = JSON.parse(AtomicFile.read(path))
1426
+ before = JSON.generate(data)
1427
+
1428
+ metadata = (data['metadata'] ||= {})
1429
+ if flow_paths.any?
1430
+ metadata['flow_paths'] = flow_paths
1431
+ else
1432
+ metadata.delete('flow_paths')
1433
+ end
1434
+ next if JSON.generate(data) == before
1435
+
1436
+ AtomicFile.write(path, json_serialize(data))
1437
+ end
1438
+ end
1439
+
1440
+ # Remove flows/ documents no entry of flow_index.json references.
1441
+ #
1442
+ # Deliberately NOT part of {#sweep_orphaned_unit_files}: that sweep
1443
+ # deletes per-type unit files no in-memory unit accounts for, while
1444
+ # flows/ holds neither units nor an _index.json. The flow family is
1445
+ # defined by flow_index.json's references, and this validates against
1446
+ # exactly those (M3) — the same artifact the validator treats separately
1447
+ # (G-2). Before this, a controller deleted or renamed incrementally left
1448
+ # its flow documents behind forever.
1449
+ #
1450
+ # Reconciled skip rule: a genuinely empty directory is an absence and is
1451
+ # skipped; a populated one whose index is missing, or an index that does
1452
+ # not parse, is corruption and raises — with nothing to validate
1453
+ # against, deleting every document would be the one outcome worse than
1454
+ # keeping orphans. Failures raise so the incremental caller aborts
1455
+ # publication.
1456
+ #
1457
+ # @return [void]
1458
+ # @raise [Woods::ExtractionError] when authoritative flow state is
1459
+ # missing or corrupt
1460
+ def sweep_orphaned_flow_files
1461
+ flows_dir = payload_dir.join('flows')
1462
+ return unless flows_dir.directory?
1463
+
1464
+ index_path = flows_dir.join('flow_index.json')
1465
+ unless index_path.exist?
1466
+ return if Dir[flows_dir.join('*.json')].empty?
1467
+
1468
+ raise Woods::ExtractionError, 'flows/ is populated but flow_index.json is missing'
1469
+ end
1470
+
1471
+ keep = parse_flow_index_for_sweep(index_path)
1472
+ .to_set { |relative| File.basename(relative.to_s) } << 'flow_index.json'
1473
+ orphans = Dir[flows_dir.join('*.json').to_s].reject { |file| keep.include?(File.basename(file)) }
1474
+ return if orphans.empty?
1475
+
1476
+ orphans.each { |file| FileUtils.rm_f(file) }
1477
+ Rails.logger.info "[Woods] Swept #{orphans.size} orphaned flow file(s)"
1478
+ end
1479
+
1480
+ # @param index_path [Pathname]
1481
+ # @return [Array<String>] the referenced document paths
1482
+ # @raise [Woods::ExtractionError] when the index does not parse
1483
+ def parse_flow_index_for_sweep(index_path)
1484
+ JSON.parse(AtomicFile.read(index_path)).values
1485
+ rescue JSON::ParserError => e
1486
+ raise Woods::ExtractionError, "flow_index.json does not parse: #{e.message}"
1487
+ end
1488
+
557
1489
  # ──────────────────────────────────────────────────────────────────────
558
1490
  # Git Enrichment
559
1491
  # ──────────────────────────────────────────────────────────────────────
@@ -561,19 +1493,21 @@ module Woods
561
1493
  def enrich_with_git_data
562
1494
  return unless git_available?
563
1495
 
564
- # Collect all file paths that need git data
1496
+ # Collect all file paths that need git data. Only app-owned paths under
1497
+ # Rails.root qualify — see {#git_enrichable_path?}.
1498
+ root = "#{Rails.root}/"
565
1499
  file_paths = []
566
1500
  @results.each do |type, units|
567
1501
  next if %i[rails_source gem_source].include?(type)
568
1502
 
569
1503
  units.each do |unit|
570
- file_paths << unit.file_path if unit.file_path && File.exist?(unit.file_path)
1504
+ path = unit.file_path
1505
+ file_paths << path if git_enrichable_path?(path, root)
571
1506
  end
572
1507
  end
573
1508
 
574
1509
  # Batch-fetch all git data in minimal subprocess calls
575
1510
  git_data = batch_git_data(file_paths)
576
- root = "#{Rails.root}/"
577
1511
 
578
1512
  # Assign results to units
579
1513
  @results.each do |type, units|
@@ -588,58 +1522,344 @@ module Woods
588
1522
  end
589
1523
  end
590
1524
 
591
- # Normalize all unit file_paths to relative paths (relative to Rails.root).
592
- #
593
- # Extractors set file_path via source_location, which returns absolute paths.
594
- # This normalization ensures consistent relative paths (e.g., "app/models/user.rb")
595
- # across all environments (local, Docker, CI) where Rails.root differs.
1525
+ # Copy git facts onto graph nodes so the analyzer can read them without
1526
+ # the units (an incremental run never holds every unit in memory).
596
1527
  #
597
- # Must run after enrich_with_git_data, which needs absolute paths for
598
- # File.exist? checks and git log commands.
599
- def normalize_file_paths
1528
+ # @return [void]
1529
+ # build_file_metadata always emits commit_count and change_frequency together
1530
+ # and non-nil, so no nil compaction is needed here (contrast annotate_node_from_git).
1531
+ def annotate_graph_with_git_data
600
1532
  @results.each_value do |units|
601
1533
  units.each do |unit|
602
- unit.file_path = normalize_file_path(unit.file_path)
1534
+ git = unit.metadata[:git]
1535
+ next unless git.is_a?(Hash)
1536
+
1537
+ @dependency_graph.annotate(
1538
+ unit.identifier,
1539
+ type: unit.type,
1540
+ commit_count: git[:commit_count],
1541
+ change_frequency: git[:change_frequency]
1542
+ )
603
1543
  end
604
1544
  end
605
1545
  end
606
1546
 
607
- # Strip Rails.root prefix from a file path, converting it to a relative path.
1547
+ # The one constructor for the analyzer both extraction paths use.
1548
+ # `Woods.configuration` can be nil in specs that reset it; fall back to
1549
+ # the analyzer's own default rather than raising mid-run.
608
1550
  #
609
- # @param path [String, nil] Absolute or relative file path
610
- # @return [String, nil] Relative path, or the original value if already relative,
611
- # nil, or not under Rails.root (e.g., a gem path)
612
- def normalize_file_path(path)
613
- return path unless path
614
-
615
- root = Rails.root.to_s
616
- prefix = root.end_with?('/') ? root : "#{root}/"
617
- path.start_with?(prefix) ? path.sub(prefix, '') : path
1551
+ # @return [GraphAnalyzer]
1552
+ def build_graph_analyzer
1553
+ ratio = Woods.configuration&.volatile_dependency_ratio || GraphAnalyzer::DEFAULT_VOLATILE_RATIO
1554
+ GraphAnalyzer.new(@dependency_graph, volatile_ratio: ratio)
618
1555
  end
619
1556
 
620
- def git_available?
621
- return @git_available if defined?(@git_available)
1557
+ # ──────────────────────────────────────────────────────────────────────
1558
+ # Package membership (#280)
1559
+ # ──────────────────────────────────────────────────────────────────────
622
1560
 
623
- @git_available = begin
624
- _, status = Open3.capture2('git', 'rev-parse', '--git-dir')
625
- status.success?
626
- rescue StandardError
627
- false
1561
+ # The package extractor instance this run resolves membership through.
1562
+ # Always built through {#extractor_for}, never `@extractors[:packages]`
1563
+ # (the instance Phase 1 used to extract package units): reading that one
1564
+ # would tie package-membership lookups to whichever instance happened to
1565
+ # run first, instead of to this run's own memoized, on-demand build.
1566
+ # Reset at the start of every run (both {#extract_all} and
1567
+ # {#prepare_incremental_run}), so a later run never resolves membership
1568
+ # against a prior run's package set.
1569
+ #
1570
+ # @return [Extractors::PackageExtractor]
1571
+ def package_resolver
1572
+ @package_resolver ||= extractor_for(:packages) || Extractors::PackageExtractor.new
1573
+ end
1574
+
1575
+ # Set or clear metadata[:package] on one unit. Framework units and
1576
+ # units with no path are never members. The key is deleted rather than
1577
+ # set to nil so a unit outside every package serializes as before.
1578
+ #
1579
+ # `unit.file_path` may be absolute (the full path, before Phase 4.5
1580
+ # relativization, and {#register_and_write}'s incremental path, before
1581
+ # its own normalization) or Rails.root-relative (unit JSON already on
1582
+ # disk); {Extractors::PackageExtractor#package_for} accepts either.
1583
+ #
1584
+ # @param unit [ExtractedUnit]
1585
+ # @return [String, nil] the package name
1586
+ def annotate_package(unit)
1587
+ return nil if %i[rails_source gem_source].include?(unit.type) || unit.file_path.nil?
1588
+
1589
+ package = package_resolver.package_for(unit.file_path)
1590
+ if package
1591
+ unit.metadata[:package] = package
1592
+ else
1593
+ unit.metadata.delete(:package)
628
1594
  end
1595
+ package
629
1596
  end
630
1597
 
631
- # Safe git command execution — no shell interpolation
1598
+ # Full-path pass over every extracted unit. Skipped entirely when the
1599
+ # app declares no packages, so a non-Packwerk app's output is unchanged
1600
+ # (I5): no work, and no `package` key on any unit or node.
632
1601
  #
633
- # @param args [Array<String>] Git command arguments
634
- # @return [String] Command output (empty string on failure)
635
- def run_git(*args)
636
- output, status = Open3.capture2('git', *args)
637
- status.success? ? output.strip : ''
638
- rescue StandardError
639
- ''
1602
+ # @return [void]
1603
+ def annotate_packages
1604
+ return if package_resolver.package_roots.empty?
1605
+
1606
+ @results.each_value { |units| units.each { |unit| annotate_package(unit) } }
640
1607
  end
641
1608
 
642
- # Batch-fetch git data for all file paths in two git commands.
1609
+ # Incremental counterpart: when a package file changed, membership of
1610
+ # units this run never touched may have changed too (a new package
1611
+ # root, a renamed package). Walk the payload's unit JSON, rewrite the
1612
+ # ones whose package differs, and annotate their nodes. Bounded to runs
1613
+ # where a package trigger fired, which is rare; every other run pays
1614
+ # nothing.
1615
+ #
1616
+ # @param change_set [ChangeSet]
1617
+ # @param affected_types [Set<Symbol>]
1618
+ # @return [Set<String>] identifiers rewritten
1619
+ def reannotate_packages(change_set, affected_types)
1620
+ keys = PathDispatcher.new.whole_app_keys_for_all(change_set.relative_paths)
1621
+ return Set.new unless keys.include?(:packages)
1622
+
1623
+ Dir[payload_dir.join('*', '*.json').to_s].each_with_object(Set.new) do |file, touched|
1624
+ next if File.basename(file) == '_index.json'
1625
+
1626
+ type_dir = File.basename(File.dirname(file))
1627
+ next if type_dir == 'rails_source' || PAYLOAD_DIRS.include?(type_dir)
1628
+
1629
+ identifier = reannotate_unit_file(file, type_dir)
1630
+ next unless identifier
1631
+
1632
+ touched.add(identifier)
1633
+ affected_types&.add(type_dir.to_sym)
1634
+ end
1635
+ end
1636
+
1637
+ # @param file [String] absolute path to one unit JSON file
1638
+ # @param type_dir [String] the extractor key directory it lives in
1639
+ # @return [String, nil] the identifier when the file was rewritten
1640
+ def reannotate_unit_file(file, type_dir)
1641
+ data = JSON.parse(AtomicFile.read(file))
1642
+ relative_path = data['file_path']
1643
+ return nil if relative_path.nil? || relative_path.start_with?('/')
1644
+
1645
+ package = package_resolver.package_for(relative_path)
1646
+ metadata = (data['metadata'] ||= {})
1647
+ return nil if metadata['package'] == package
1648
+
1649
+ if package
1650
+ metadata['package'] = package
1651
+ else
1652
+ metadata.delete('package')
1653
+ end
1654
+ AtomicFile.write(file, json_serialize(data))
1655
+
1656
+ identifier = data['identifier']
1657
+ type = (data['type'] || type_dir.singularize).to_sym
1658
+ @dependency_graph.annotate(identifier, type: type, package: package)
1659
+ identifier
1660
+ rescue JSON::ParserError => e
1661
+ Rails.logger.warn "[Woods] Could not re-annotate package on #{file}: #{e.message}"
1662
+ nil
1663
+ end
1664
+
1665
+ # Is this a path worth asking git about?
1666
+ #
1667
+ # A gem-owned unit (an engine model) carries its real path. Outside
1668
+ # Rails.root, git refuses the whole `log` invocation when any pathspec is
1669
+ # outside the repository — one gem path would erase the git metadata of
1670
+ # the other 499 units in its 500-path batch. Inside Rails.root, a bundle
1671
+ # vendored at `vendor/bundle` puts the same gem files under the root
1672
+ # prefix, gitignored, so sending them is wasted pathspec work every run.
1673
+ # Same exclusions as {Extractors::SharedUtilityMethods#app_source?}.
1674
+ #
1675
+ # @param path [String, nil] absolute file path
1676
+ # @param root [String] Rails.root with a trailing separator
1677
+ # @return [Boolean]
1678
+ def git_enrichable_path?(path, root)
1679
+ return false unless path&.start_with?(root)
1680
+ return false if path.include?('/vendor/') || path.include?('/node_modules/')
1681
+
1682
+ File.exist?(path)
1683
+ end
1684
+
1685
+ # Normalize all unit file_paths to relative paths (relative to Rails.root).
1686
+ #
1687
+ # Extractors set file_path via source_location, which returns absolute paths.
1688
+ # This normalization ensures consistent relative paths (e.g., "app/models/user.rb")
1689
+ # across all environments (local, Docker, CI) where Rails.root differs.
1690
+ #
1691
+ # Must run after enrich_with_git_data, which needs absolute paths for
1692
+ # File.exist? checks and git log commands.
1693
+ # Write a unit's JSON, unless the bytes on disk are already exactly that.
1694
+ #
1695
+ # A routes change replaces every `ROUTE_CONSUMER_EXTRACTORS` type wholesale
1696
+ # — on a production-shaped host that measured 1,707 units, roughly a quarter
1697
+ # of the index — and almost all of them re-serialize to the bytes already
1698
+ # there. `AtomicFile.write` is a tempfile plus an fsync plus a rename each
1699
+ # time, so the fsync is the cost being avoided here; the comparison read is
1700
+ # cheaper than the write it replaces.
1701
+ #
1702
+ # Only the *write* is skipped. Graph registration, the dependents marking
1703
+ # and `@incremental_written` all still happen for every unit, because those
1704
+ # are what equivalence and the git-enrichment pass depend on — skipping any
1705
+ # of them would make an unchanged unit differ from a full extraction.
1706
+ #
1707
+ # Compared as bytes: `AtomicFile.write` is binmode, and the encoding a read
1708
+ # comes back tagged with depends on the process's default external encoding
1709
+ # (US-ASCII under `LANG=C`, which is where the daemon runs).
1710
+ #
1711
+ # @param path [Pathname] destination
1712
+ # @param unit [ExtractedUnit] unit to serialize
1713
+ # @return [void]
1714
+ def write_unit_file(path, unit)
1715
+ payload = json_serialize(unit.to_h)
1716
+ return if identical_on_disk?(path, payload)
1717
+
1718
+ AtomicFile.write(path, payload)
1719
+ end
1720
+
1721
+ # The serialized `extracted_at` scalar as {ExtractedUnit#to_h} +
1722
+ # {#json_serialize} emit it, compact or pretty. `Time#iso8601` produces
1723
+ # exactly this value shape — no fractional seconds, `Z` or a `±hh:mm`
1724
+ # offset — and `spec/extracted_unit_spec.rb` pins that, so a change to the
1725
+ # stamp's shape fails a spec instead of quietly un-matching this mask.
1726
+ # The value constraint is what keeps the mask honest against user code: a
1727
+ # bare `"extracted_at":` cannot occur inside any JSON *string* value
1728
+ # (interior quotes serialize as `\"`), so only a real JSON key can match,
1729
+ # and only when it holds a timestamp — which no extractor emits below the
1730
+ # top level.
1731
+ EXTRACTED_AT_SCALAR =
1732
+ /("extracted_at":\s*")\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}(?:Z|[+-]\d{2}:\d{2})(?=")/
1733
+ # An implementation detail of the byte comparison, not part of the
1734
+ # extractor's surface (`private` does not scope constants).
1735
+ private_constant :EXTRACTED_AT_SCALAR
1736
+
1737
+ # @return [Boolean] true when the file already holds exactly these bytes,
1738
+ # the `extracted_at` stamp aside
1739
+ def identical_on_disk?(path, payload)
1740
+ return false unless File.exist?(path)
1741
+
1742
+ mask_extracted_at(AtomicFile.read(path).b) == mask_extracted_at(payload.b)
1743
+ rescue StandardError
1744
+ # An unreadable or half-written file is not a match; fall through and
1745
+ # rewrite it.
1746
+ false
1747
+ end
1748
+
1749
+ # Blank the `extracted_at` value so the byte comparison ignores it.
1750
+ #
1751
+ # {ExtractedUnit#to_h} stamps `extracted_at: Time.now.iso8601` on every
1752
+ # serialization, so compared raw the two sides *always* differed across
1753
+ # runs and the skip was inert (#208): every write did the comparison read
1754
+ # and then paid the fsync anyway. Masked with a regex rather than parsed,
1755
+ # because a JSON parse of both sides would cost more than the write being
1756
+ # avoided. A skipped write leaves the older stamp on disk deliberately —
1757
+ # the equivalence oracle (`spec/support/index_comparison.rb`) lists
1758
+ # `extracted_at` in `VOLATILE_UNIT_KEYS`: "a unit an incremental run
1759
+ # correctly left alone keeps an older stamp".
1760
+ #
1761
+ # @param bytes [String] serialized unit JSON, binary-tagged (`.b`) — the
1762
+ # ASCII-only pattern matches bytewise regardless of the content's
1763
+ # original encoding
1764
+ # @return [String] the bytes with the stamp's value removed
1765
+ def mask_extracted_at(bytes)
1766
+ bytes.gsub(EXTRACTED_AT_SCALAR, '\1')
1767
+ end
1768
+
1769
+ def normalize_file_paths
1770
+ @results.each_value do |units|
1771
+ units.each do |unit|
1772
+ unit.file_path = normalize_file_path(unit.file_path)
1773
+ end
1774
+ end
1775
+ end
1776
+
1777
+ # Strip Rails.root prefix from a file path, converting it to a relative path.
1778
+ #
1779
+ # @param path [String, nil] Absolute or relative file path
1780
+ # @return [String, nil] Relative path, or the original value if already relative,
1781
+ # nil, or not under Rails.root (e.g., a gem path)
1782
+ def normalize_file_path(path)
1783
+ return path unless path
1784
+
1785
+ root = Rails.root.to_s
1786
+ prefix = root.end_with?('/') ? root : "#{root}/"
1787
+ path.start_with?(prefix) ? path.sub(prefix, '') : path
1788
+ end
1789
+
1790
+ # Can this run's git calls produce real facts?
1791
+ #
1792
+ # `rev-parse --git-dir` alone is not enough. Over a linked worktree whose
1793
+ # private git directory is reachable but whose `commondir` is not, it
1794
+ # answers while every ref lookup fails and `git log` exits 0 with nothing
1795
+ # to say. Enrichment then wrote `commit_count: 0` and
1796
+ # `change_frequency: new` onto every unit, which reads exactly like a file
1797
+ # that was never committed, where an absent git directory correctly omits
1798
+ # the keys (B-186). HEAD has to resolve.
1799
+ #
1800
+ # Memoized, so the warning below is emitted at most once per run.
1801
+ #
1802
+ # @return [Boolean]
1803
+ def git_available?
1804
+ return @git_available if defined?(@git_available)
1805
+
1806
+ _output, error, status = Open3.capture3(*git_argv('rev-parse', 'HEAD'))
1807
+ @git_available = status.success?
1808
+ warn_unresolvable_git(error) unless @git_available
1809
+ @git_available
1810
+ rescue StandardError
1811
+ @git_available = false
1812
+ end
1813
+
1814
+ # Say once why no unit will carry git metadata, but only when there is a
1815
+ # working tree to explain. No `.git` at the root is the ordinary source
1816
+ # tarball or `COPY`-without-`.git` case, and it is not a fault.
1817
+ #
1818
+ # @param error [String] git's own stderr
1819
+ # @return [void]
1820
+ def warn_unresolvable_git(error)
1821
+ return unless File.exist?(File.join(Rails.root.to_s, '.git'))
1822
+
1823
+ cause = error.to_s.lines.first.to_s.strip
1824
+ Rails.logger.warn(
1825
+ '[Woods] git cannot resolve HEAD for this working tree, so no unit will carry git ' \
1826
+ "metadata: #{cause}. Over a linked worktree in a container, mount the canonical git " \
1827
+ 'directory and point WOODS_GIT_DIR at it; GIT_DIR alone is not enough, because the ' \
1828
+ "worktree's private git directory reaches the shared one through a relative pointer."
1829
+ )
1830
+ end
1831
+
1832
+ # The git command line every enrichment call runs.
1833
+ #
1834
+ # `-C <root>` keeps the result independent of the process working
1835
+ # directory. `WOODS_GIT_DIR` wins when set: it names the canonical git
1836
+ # directory directly, which is the escape hatch for a container that can
1837
+ # mount that directory but not the host path a worktree pointer names
1838
+ # (B-181).
1839
+ #
1840
+ # @param args [Array<String>] git arguments
1841
+ # @return [Array<String>] full argv
1842
+ def git_argv(*args)
1843
+ GitCommand.argv(Rails.root, *args)
1844
+ end
1845
+
1846
+ # Safe git command execution — no shell interpolation
1847
+ #
1848
+ # @param args [Array<String>] Git command arguments
1849
+ # @return [String] Command output (empty string on failure)
1850
+ def run_git(*args)
1851
+ output, _error, status = Open3.capture3(*git_argv(*args))
1852
+ status.success? ? output.strip : ''
1853
+ rescue StandardError
1854
+ ''
1855
+ end
1856
+
1857
+ # Batch-fetch git data for all file paths in two git commands.
1858
+ #
1859
+ # Duplicate paths are collapsed before slicing (audit P9d): many units
1860
+ # share one file_path, and duplicates only repeat a pathspec another
1861
+ # batch also sent. The result is keyed by relative path, so the output
1862
+ # is identical.
643
1863
  #
644
1864
  # @param file_paths [Array<String>] Absolute file paths
645
1865
  # @return [Hash{String => Hash}] Keyed by relative path
@@ -647,7 +1867,7 @@ module Woods
647
1867
  return {} if file_paths.empty?
648
1868
 
649
1869
  root = "#{Rails.root}/"
650
- relative_paths = file_paths.map { |f| f.sub(root, '') }
1870
+ relative_paths = file_paths.map { |f| f.sub(root, '') }.uniq
651
1871
  result = {}
652
1872
  relative_paths.each { |rp| result[rp] = {} }
653
1873
 
@@ -735,23 +1955,89 @@ module Woods
735
1955
 
736
1956
  def write_results
737
1957
  @results.each do |type, units|
738
- type_dir = @output_dir.join(type.to_s)
1958
+ type_dir = payload_dir.join(type.to_s)
739
1959
 
740
1960
  units.each do |unit|
741
- File.write(
1961
+ AtomicFile.write(
742
1962
  type_dir.join(collision_safe_filename(unit.identifier)),
743
1963
  json_serialize(unit.to_h)
744
1964
  )
745
1965
  end
746
1966
 
747
1967
  # Also write a type index for fast lookups
748
- File.write(
1968
+ AtomicFile.write(
749
1969
  type_dir.join('_index.json'),
750
1970
  json_serialize(type_index_entries(units))
751
1971
  )
752
1972
  end
753
1973
  end
754
1974
 
1975
+ # Delete unit JSON files in the just-written type directories that this
1976
+ # full run did not write (#177).
1977
+ #
1978
+ # A full extraction never wipes the output directory —
1979
+ # {#setup_output_directory} only mkdir_p's — so extracting into a
1980
+ # directory that saw a different version of the app left the previous
1981
+ # run's `<Unit>_<digest>.json` files behind. The manifest and the freshly
1982
+ # written `_index.json` were correct, but the NEXT incremental run's
1983
+ # {#regenerate_type_index} rebuilds the index from a disk glob of the type
1984
+ # dir, resurrecting every orphan into the listed index, and
1985
+ # {#persisted_counts} then writes the inflated counts into the manifest —
1986
+ # orphans became listed, retrievable-by-listing units the graph knows
1987
+ # nothing about. On a full run the in-memory `@results` are authoritative
1988
+ # — the same argument {#rewrite_flow_annotated_units} makes for refusing
1989
+ # the disk glob — so a unit file no current unit accounts for is stale by
1990
+ # definition and is removed here.
1991
+ #
1992
+ # The boundary is deliberately narrow:
1993
+ #
1994
+ # * Only the per-type directories this run produced — `@results` keys —
1995
+ # are entered. When `include_framework_sources` gates rails_source out
1996
+ # of the run ({#skip_by_configuration?}), `@results` holds no
1997
+ # `:rails_source` key, so a knob-off full run leaves explicitly
1998
+ # extracted framework units (`woods:extract_framework`) alone — the
1999
+ # #169 preservation decision. Nothing outside those directories is
2000
+ # reachable: manifest, graph, generation, `woods.json`, `dumps/`,
2001
+ # `flows/` and the lock/status files all live in the output root or in
2002
+ # directories no extractor key names.
2003
+ # * Only `*.json` directly inside a type dir is considered, and
2004
+ # `_index.json` is always kept. Non-JSON files and subdirectories are
2005
+ # never touched.
2006
+ # * Legitimate names are exactly what {#write_results} just wrote —
2007
+ # {FilenameUtils#collision_safe_filename} over the type's deduped units
2008
+ # (which covers every unit type the extractor emits: gem_source shares
2009
+ # `@results[:rails_source]`). Legacy {FilenameUtils#safe_filename} names
2010
+ # carry no digest suffix, so they never match a current name and are by
2011
+ # definition orphans of a pre-collision-safe run.
2012
+ #
2013
+ # The incremental path deliberately has no counterpart: it holds only
2014
+ # changed units in memory, so "not in memory" means nothing there —
2015
+ # deletion on that path is driven by the graph and the change set
2016
+ # ({#prune_vanished_units}, {#remove_replaced_units}).
2017
+ #
2018
+ # @return [void]
2019
+ def sweep_orphaned_unit_files
2020
+ @results.each do |type, units|
2021
+ type_dir = payload_dir.join(type.to_s)
2022
+ next unless type_dir.directory?
2023
+
2024
+ keep = units.to_set { |unit| collision_safe_filename(unit.identifier) }
2025
+ keep << '_index.json'
2026
+
2027
+ orphans = Dir[type_dir.join('*.json').to_s].reject { |file| keep.include?(File.basename(file)) }
2028
+ next if orphans.empty?
2029
+
2030
+ orphans.each { |file| FileUtils.rm_f(file) }
2031
+ Rails.logger.info "[Woods] Swept #{orphans.size} orphaned unit file(s) from #{type}/"
2032
+ end
2033
+ rescue StandardError => e
2034
+ # The extraction itself already succeeded; aborting the run over cleanup
2035
+ # would trade orphaned files for a lost index. Error, not warn: the
2036
+ # files this leaves behind are exactly what the next incremental run's
2037
+ # disk glob resurrects.
2038
+ Rails.logger.error "[Woods] Orphaned-unit sweep failed: #{e.message}"
2039
+ end
2040
+
755
2041
  # Build the `_index.json` entry list for a set of in-memory units.
756
2042
  # Shared by {#write_results} and {#rewrite_flow_annotated_units} so both
757
2043
  # emit the index from the authoritative in-memory `@results` rather than
@@ -773,10 +2059,13 @@ module Woods
773
2059
 
774
2060
  def write_dependency_graph
775
2061
  graph_data = @dependency_graph.to_h
776
- graph_data[:pagerank] = @dependency_graph.pagerank
2062
+ # Key-sorted for the same reason `to_h` sorts its own sections: the
2063
+ # digest published as `graph_sha` covers these bytes, so node
2064
+ # registration order must not reach it (B-180).
2065
+ graph_data[:pagerank] = @dependency_graph.pagerank.sort_by { |identifier, _| identifier }.to_h
777
2066
 
778
- File.write(
779
- @output_dir.join('dependency_graph.json'),
2067
+ AtomicFile.write(
2068
+ payload_dir.join('dependency_graph.json'),
780
2069
  json_serialize(graph_data)
781
2070
  )
782
2071
  end
@@ -787,12 +2076,12 @@ module Woods
787
2076
  enriched = @graph_analysis.merge(
788
2077
  generated_at: Time.current.iso8601,
789
2078
  graph_sha: Digest::SHA256.hexdigest(
790
- File.read(@output_dir.join('dependency_graph.json'))
2079
+ AtomicFile.read(payload_dir.join('dependency_graph.json'))
791
2080
  )
792
2081
  )
793
2082
 
794
- File.write(
795
- @output_dir.join('graph_analysis.json'),
2083
+ AtomicFile.write(
2084
+ payload_dir.join('graph_analysis.json'),
796
2085
  json_serialize(enriched)
797
2086
  )
798
2087
  end
@@ -836,8 +2125,8 @@ module Woods
836
2125
  schema_sha: schema_sha
837
2126
  }
838
2127
 
839
- File.write(
840
- @output_dir.join('manifest.json'),
2128
+ AtomicFile.write(
2129
+ payload_dir.join('manifest.json'),
841
2130
  json_serialize(manifest)
842
2131
  )
843
2132
  end
@@ -851,8 +2140,8 @@ module Woods
851
2140
  counts = {}
852
2141
  chunks = 0
853
2142
 
854
- Dir[@output_dir.join('*/_index.json').to_s].each do |index_path|
855
- entries = JSON.parse(File.read(index_path))
2143
+ Dir[payload_dir.join('*/_index.json').to_s].each do |index_path|
2144
+ entries = JSON.parse(AtomicFile.read(index_path))
856
2145
  counts[File.basename(File.dirname(index_path)).to_sym] = entries.size
857
2146
  chunks += entries.sum { |e| e['chunk_count'].to_i }
858
2147
  rescue JSON::ParserError => e
@@ -876,10 +2165,10 @@ module Woods
876
2165
  def capture_snapshot
877
2166
  return unless Woods.configuration.enable_snapshots
878
2167
 
879
- manifest_path = @output_dir.join('manifest.json')
2168
+ manifest_path = payload_dir.join('manifest.json')
880
2169
  return unless manifest_path.exist?
881
2170
 
882
- manifest = JSON.parse(File.read(manifest_path))
2171
+ manifest = JSON.parse(AtomicFile.read(manifest_path))
883
2172
  # Snapshots are keyed on the commit SHA — an unresolvable provenance
884
2173
  # ("unknown", see GitProvenance/#137) must not key or collide a snapshot.
885
2174
  git_sha = manifest['git_sha']
@@ -932,32 +2221,39 @@ module Woods
932
2221
  # category with count and top-5 namespace breakdown, rather than enumerating
933
2222
  # every unit. Per-unit detail is available in the per-category _index.json files.
934
2223
  #
2224
+ # The full path derives its numbers from the in-memory results. An
2225
+ # incremental run holds only changed units in memory, so it derives the
2226
+ # same shape from the persisted per-type _index.json files — the same
2227
+ # source {#persisted_counts} feeds the manifest from, which is what keeps
2228
+ # SUMMARY.md's totals agreeing with manifest.json. (M4: the hardlinked
2229
+ # previous generation's summary this path used to leave in place carried
2230
+ # stale totals after any run that added or removed units.) The `Generated:`
2231
+ # stamp keeps its meaning either way: it names the moment the summary was
2232
+ # written.
2233
+ #
935
2234
  # @return [void]
936
2235
  def write_structural_summary
937
- return if @results.empty?
2236
+ stats = @results.empty? ? persisted_summary_stats : results_summary_stats
2237
+ return unless stats
938
2238
 
939
- total_units = @results.values.sum(&:size)
940
- total_chunks = @results.sum { |_, units| units.sum { |u| [u.chunks.size, 1].max } }
941
- category_count = @results.count { |_, units| units.any? }
2239
+ total_units = stats.sum { |_, s| s[:count] }
2240
+ # Matches the manifest's count (`write_manifest`/`persisted_counts`
2241
+ # both sum chunk counts directly) — the previous `[size, 1].max`
2242
+ # floor made SUMMARY.md disagree with manifest.json for every
2243
+ # unchunked unit.
2244
+ total_chunks = stats.sum { |_, s| s[:chunks] }
942
2245
 
943
2246
  summary = []
944
2247
  summary << '# Codebase Index Summary'
945
2248
  summary << "Generated: #{Time.current.iso8601}"
946
2249
  summary << "Rails #{Rails.version} / Ruby #{RUBY_VERSION}"
947
- summary << "Units: #{total_units} | Chunks: #{total_chunks} | Categories: #{category_count}"
2250
+ summary << "Units: #{total_units} | Chunks: #{total_chunks} | Categories: #{stats.size}"
948
2251
  summary << ''
949
2252
 
950
- @results.each do |type, units|
951
- next if units.empty?
952
-
953
- summary << "## #{type.to_s.titleize} (#{units.size})"
954
-
955
- ns_counts = units
956
- .group_by { |u| u.namespace.nil? || u.namespace.empty? ? '(root)' : u.namespace }
957
- .transform_values(&:size)
958
- .sort_by { |_, count| -count }
959
- .first(5)
2253
+ stats.each do |type, s|
2254
+ summary << "## #{type.to_s.titleize} (#{s[:count]})"
960
2255
 
2256
+ ns_counts = s[:namespaces].sort_by { |_, count| -count }.first(5)
961
2257
  ns_parts = ns_counts.map { |ns, count| "#{ns} #{count}" }
962
2258
  summary << "Namespaces: #{ns_parts.join(', ')}" unless ns_parts.empty?
963
2259
  summary << ''
@@ -979,25 +2275,88 @@ module Woods
979
2275
  hub_names = significant_hubs.map { |h| h[:identifier] }.join(', ')
980
2276
  summary << "- Hub nodes (>20 dependents): #{hub_names}"
981
2277
  end
2278
+
2279
+ volatile = Array(@graph_analysis[:volatile_dependencies]).first(5)
2280
+ if volatile.any?
2281
+ lines = volatile.map { |v| "#{v[:from]} -> #{v[:to]} (#{v[:from_commits]} vs #{v[:to_commits]} commits)" }
2282
+ summary << "- Volatile dependencies (top #{lines.size}): #{lines.join('; ')}"
2283
+ end
982
2284
  end
983
2285
 
984
2286
  summary << ''
985
2287
 
986
- File.write(
987
- @output_dir.join('SUMMARY.md'),
2288
+ AtomicFile.write(
2289
+ payload_dir.join('SUMMARY.md'),
988
2290
  summary.join("\n")
989
2291
  )
990
2292
  end
991
2293
 
2294
+ # Per-type `{ count:, chunks:, namespaces: }` for {#write_structural_summary},
2295
+ # derived from the in-memory results a full extraction holds.
2296
+ #
2297
+ # @return [Hash{Symbol => Hash}]
2298
+ def results_summary_stats
2299
+ @results.each_with_object({}) do |(type, units), stats|
2300
+ next if units.empty?
2301
+
2302
+ stats[type] = {
2303
+ count: units.size,
2304
+ chunks: units.sum { |u| u.chunks.size },
2305
+ namespaces: namespace_histogram(units.map(&:namespace))
2306
+ }
2307
+ end
2308
+ end
2309
+
2310
+ # The incremental path's counterpart to {#results_summary_stats}: the same
2311
+ # shape read back from the persisted per-type _index.json files, the source
2312
+ # {#persisted_counts} uses for the manifest — so the two artifacts cannot
2313
+ # disagree about totals. Sorted for determinism, since the glob order is
2314
+ # the filesystem's.
2315
+ #
2316
+ # @return [Hash{Symbol => Hash}, nil] nil when the payload holds no type
2317
+ # indexes at all, so a bare index writes no summary — matching the full
2318
+ # path's early return when it extracted nothing
2319
+ def persisted_summary_stats
2320
+ stats = {}
2321
+
2322
+ Dir[payload_dir.join('*/_index.json').to_s].each do |index_path|
2323
+ entries = JSON.parse(AtomicFile.read(index_path))
2324
+ next if entries.empty?
2325
+
2326
+ stats[File.basename(File.dirname(index_path)).to_sym] = {
2327
+ count: entries.size,
2328
+ chunks: entries.sum { |e| e['chunk_count'].to_i },
2329
+ namespaces: namespace_histogram(entries.map { |e| e['namespace'] })
2330
+ }
2331
+ rescue JSON::ParserError => e
2332
+ # Same posture as {#persisted_counts}: an unreadable index drops that
2333
+ # type from the manifest and the summary alike, so the two still agree.
2334
+ type = File.basename(File.dirname(index_path))
2335
+ Rails.logger.warn("[Woods] Skipping unreadable #{type}/_index.json in summary totals: #{e.message}")
2336
+ end
2337
+
2338
+ stats.empty? ? nil : stats
2339
+ end
2340
+
2341
+ # Group namespaces for a SUMMARY.md section: a missing or empty namespace
2342
+ # is the root, everything else counts as-is.
2343
+ #
2344
+ # @param namespaces [Array<String, nil>]
2345
+ # @return [Hash{String => Integer}]
2346
+ def namespace_histogram(namespaces)
2347
+ namespaces.group_by { |ns| ns.nil? || ns.empty? ? '(root)' : ns }
2348
+ .transform_values(&:size)
2349
+ end
2350
+
992
2351
  def regenerate_type_index(type_key)
993
- type_dir = @output_dir.join(type_key.to_s)
2352
+ type_dir = payload_dir.join(type_key.to_s)
994
2353
  return unless type_dir.directory?
995
2354
 
996
2355
  # Scan existing unit JSON files (exclude _index.json)
997
2356
  index = Dir[type_dir.join('*.json')].filter_map do |file|
998
2357
  next if File.basename(file) == '_index.json'
999
2358
 
1000
- data = JSON.parse(File.read(file))
2359
+ data = JSON.parse(AtomicFile.read(file))
1001
2360
  {
1002
2361
  identifier: data['identifier'],
1003
2362
  file_path: data['file_path'],
@@ -1010,7 +2369,7 @@ module Woods
1010
2369
  }
1011
2370
  end
1012
2371
 
1013
- File.write(
2372
+ AtomicFile.write(
1014
2373
  type_dir.join('_index.json'),
1015
2374
  json_serialize(index)
1016
2375
  )
@@ -1019,6 +2378,15 @@ module Woods
1019
2378
  # Token estimate for a unit parsed back from JSON, mirroring
1020
2379
  # ExtractedUnit#estimated_tokens (see docs/TOKEN_BENCHMARK.md).
1021
2380
  #
2381
+ # `JSON.generate`, not `Hash#to_json`: with ActiveSupport loaded the
2382
+ # latter applies HTML-safe escaping, rendering `>` as `\\u003e` — six
2383
+ # characters where the file on disk has one. A model with
2384
+ # `scope :recent, -> { ... }` in its metadata therefore estimated one
2385
+ # token higher here than {ExtractedUnit#estimated_tokens} did over the
2386
+ # same content, so a full and an incremental run disagreed on that unit's
2387
+ # `_index.json` entry (#164). Both sides now measure the serialization
2388
+ # {#json_serialize} actually writes.
2389
+ #
1022
2390
  # @param data [Hash] Parsed unit JSON (string keys)
1023
2391
  # @return [Integer]
1024
2392
  def estimated_tokens_from(data)
@@ -1026,7 +2394,7 @@ module Woods
1026
2394
  metadata = data['metadata'] || {}
1027
2395
 
1028
2396
  source_tokens = source ? TokenUtils.estimate_tokens(source) : 0
1029
- metadata_tokens = metadata.any? ? TokenUtils.estimate_tokens(metadata.to_json) : 0
2397
+ metadata_tokens = metadata.any? ? TokenUtils.estimate_tokens(JSON.generate(metadata)) : 0
1030
2398
  source_tokens + metadata_tokens
1031
2399
  end
1032
2400
 
@@ -1087,78 +2455,1021 @@ module Woods
1087
2455
  # Incremental Re-extraction
1088
2456
  # ──────────────────────────────────────────────────────────────────────
1089
2457
 
2458
+ # Extractor instances are reused across a single incremental run, the way
2459
+ # full extraction uses one instance per type. Several of them do real
2460
+ # setup work in `initialize` (route-helper maps, routes maps), so a fresh
2461
+ # instance per file would make an N-file change N times more expensive.
2462
+ #
2463
+ # @param key [Symbol] key into {EXTRACTORS}
2464
+ # @return [Object, nil] the memoized extractor instance
2465
+ def extractor_for(key)
2466
+ @incremental_extractors ||= {}
2467
+ return @incremental_extractors[key] if @incremental_extractors.key?(key)
2468
+
2469
+ @incremental_extractors[key] = EXTRACTORS[key]&.new
2470
+ rescue StandardError => e
2471
+ Rails.logger.warn "[Woods] Could not build #{key} extractor: #{e.message}"
2472
+ @incremental_extractors[key] = nil
2473
+ end
2474
+
2475
+ # ActiveRecord model names, needed by PoroExtractor to tell a plain class
2476
+ # under app/models apart from a persisted model. Computed once per run.
2477
+ #
2478
+ # @return [Set<String>]
2479
+ def active_record_names
2480
+ @active_record_names ||=
2481
+ if defined?(ActiveRecord::Base)
2482
+ ActiveRecord::Base.descendants.filter_map(&:name).to_set
2483
+ else
2484
+ Set.new
2485
+ end
2486
+ end
2487
+
2488
+ # Re-extract every file-based unit defined by the changed paths that still
2489
+ # exist on disk, and prune the ones those paths no longer define.
2490
+ #
2491
+ # This is the fix for #164 gap 1 (a path the index has never seen routed
2492
+ # nowhere, so new files were silently ignored) and the per-path half of
2493
+ # gap 3 (a file that defines several units — a `.rake` file with multiple
2494
+ # tasks, an i18n YAML — could only ever resolve to one identifier).
2495
+ # Reconciling the whole path at once means a task deleted from a
2496
+ # multi-task file is removed rather than left behind.
2497
+ #
2498
+ # Removal is scoped to the unit types the matching rules could have
2499
+ # produced, so a class-based unit sharing the path (the `User` model unit
2500
+ # for `app/models/user.rb`) is never collaterally deleted here.
2501
+ #
2502
+ # @param change_set [ChangeSet]
2503
+ # @param affected_types [Set<Symbol>] out-param of touched extractor keys
2504
+ # @return [Set<String>] identifiers written or removed
2505
+ def reconcile_changed_paths(change_set, affected_types)
2506
+ dispatcher = PathDispatcher.new
2507
+ touched = Set.new
2508
+
2509
+ change_set.existing_paths.each do |absolute_path|
2510
+ rules = dispatcher.file_rules_for(change_set.relativize(absolute_path))
2511
+ next if rules.empty?
2512
+
2513
+ produced = Set.new
2514
+ # A rule whose extraction *raised* tells us nothing about what the path
2515
+ # defines, so it must not license pruning what the path defined before.
2516
+ raised = false
2517
+ rules.each do |rule|
2518
+ units = extract_with_rule(rule, absolute_path)
2519
+ if units.nil?
2520
+ raised = true
2521
+ next
2522
+ end
2523
+
2524
+ # (identifier, type) pairs, not bare identifiers: two rules can
2525
+ # claim one path (policies/pundit_policies) and mint the same
2526
+ # identifier for different unit types. When one of them stops
2527
+ # producing, an identifier-keyed set would let the survivor shield
2528
+ # the stale sibling-type node from the prune below (the #225 shape,
2529
+ # one method over — CORE-1).
2530
+ produced.merge(units.map { |unit| [unit.identifier, unit.type] })
2531
+ touched.merge(register_and_write(rule.extractor_key, units, affected_types))
2532
+ end
2533
+
2534
+ next if raised
2535
+
2536
+ touched.merge(
2537
+ prune_path_leftovers(absolute_path, rules, produced, affected_types)
2538
+ )
2539
+ end
2540
+
2541
+ touched
2542
+ end
2543
+
2544
+ # Run one {PathDispatcher::Rule} against one file.
2545
+ #
2546
+ # @param rule [PathDispatcher::Rule]
2547
+ # @param absolute_path [String]
2548
+ # @return [Array<ExtractedUnit>, nil] the units the path defines, or nil
2549
+ # when the attempt says nothing about what it defines (see the rescue)
2550
+ def extract_with_rule(rule, absolute_path)
2551
+ extractor = extractor_for(rule.extractor_key)
2552
+ # nil, not [] — the same distinction the rescue below defends (#198).
2553
+ # {#extractor_for} rescues a raising *constructor* and memoizes nil, and
2554
+ # nil fails the respond_to? guard — so a failed construction used to fall
2555
+ # through to the [] "this path defines nothing any more" answer, and one
2556
+ # broken constructor licensed pruning every previously-registered unit
2557
+ # on every changed path of that type, with the generation bumped over
2558
+ # the loss. Construction failure tells us nothing about the path; only
2559
+ # a genuinely constructed extractor that lacks the method earns the [].
2560
+ return nil if extractor.nil?
2561
+ return [] unless extractor.respond_to?(rule.method_name)
2562
+
2563
+ result =
2564
+ if rule.extractor_key == :poros
2565
+ # PoroExtractor needs the AR name set to reject persisted models;
2566
+ # its default is an empty set, which would misfile every model
2567
+ # under app/models as a PORO on the incremental path.
2568
+ extractor.public_send(rule.method_name, absolute_path, ar_names: active_record_names)
2569
+ else
2570
+ extractor.public_send(rule.method_name, absolute_path)
2571
+ end
2572
+
2573
+ Array(result).compact
2574
+ rescue StandardError => e
2575
+ Rails.logger.warn "[Woods] #{rule.extractor_key} re-extraction of #{absolute_path} failed: #{e.message}"
2576
+ # `nil`, not `[]`. The caller treats an empty result as "this path defines
2577
+ # nothing any more" and prunes the units previously registered to it — so
2578
+ # returning [] here turns a *transient* failure (a watcher batch catching
2579
+ # an editor mid-write, an encoding hiccup) into silent deletion of a
2580
+ # perfectly good unit, which nothing restores until that file is next
2581
+ # touched. "Extraction raised" and "extracted successfully, defines
2582
+ # nothing" are different answers and only the second licenses a prune.
2583
+ nil
2584
+ end
2585
+
2586
+ # Remove units the graph still attributes to a path that the path no
2587
+ # longer produces — but only for the types the matching rules cover.
2588
+ #
2589
+ # @param absolute_path [String]
2590
+ # @param rules [Array<PathDispatcher::Rule>]
2591
+ # @param produced [Set<Array(String, Symbol)>] (identifier, type) pairs
2592
+ # just extracted from the path
2593
+ # @param affected_types [Set<Symbol>]
2594
+ # @return [Set<String>] identifiers removed
2595
+ def prune_path_leftovers(absolute_path, rules, produced, affected_types)
2596
+ covered_keys = rules.to_set(&:extractor_key)
2597
+ removed = Set.new
2598
+
2599
+ @dependency_graph.units_for_path(absolute_path).each do |identifier, node_type|
2600
+ next if produced.include?([identifier, node_type])
2601
+ next unless covered_keys.include?(TYPE_TO_EXTRACTOR_KEY[node_type])
2602
+
2603
+ removed.add(identifier) if remove_unit(identifier, affected_types, type: node_type)
2604
+ end
2605
+
2606
+ removed
2607
+ end
2608
+
2609
+ # Reconcile class-based types against their runtime discovery sets.
2610
+ #
2611
+ # Models, controllers, mailers, components and channels are discovered by
2612
+ # walking descendants, not by globbing files, so a class added since the
2613
+ # last extraction is invisible to any path-based dispatch. Comparing each
2614
+ # extractor's `discoverable_classes` against the identifiers already in
2615
+ # the graph finds exactly those additions, using the same discovery code
2616
+ # a full extraction uses.
2617
+ #
2618
+ # Removals are handled here too, but only against a *complete* discovery
2619
+ # set — see {#stale_class_based_units} for why that qualifier carries the
2620
+ # whole safety argument.
2621
+ #
2622
+ # @param affected_types [Set<Symbol>]
2623
+ # @return [Set<String>] identifiers added or removed
2624
+ def reconcile_class_based_types(affected_types, except: nil)
2625
+ touched = Set.new
2626
+ excluded = except&.to_set || Set.new
2627
+
2628
+ CLASS_BASED_DISCOVERY.each do |key, spec|
2629
+ extractor = extractor_for(key)
2630
+ next unless extractor.respond_to?(:discoverable_classes)
2631
+
2632
+ discovered = extractor.discoverable_classes.reject { |k| k.name.nil? }
2633
+ known = Array(spec[:types] || spec[:type])
2634
+ .flat_map { |type| @dependency_graph.units_of_type(type) }.to_set
2635
+
2636
+ touched.merge(add_discovered_classes(key, spec, discovered, known, excluded, affected_types))
2637
+ next if spec[:reconcile_removals] == false
2638
+
2639
+ touched.merge(remove_stale_classes(spec, discovered, known, affected_types))
2640
+ end
2641
+
2642
+ touched
2643
+ end
2644
+
2645
+ # Pruned class-based identifiers the tree still governs, and that the
2646
+ # second reconciliation pass may therefore re-add.
2647
+ #
2648
+ # The class-based *move* shape — the file moved, the constant did not
2649
+ # (M1) — prunes the unit for the vanished old path while the class stays
2650
+ # in the discovery set; the move target is a changed file the active
2651
+ # loader governs for exactly that constant, so re-adding produces the
2652
+ # unit a full extraction produces. The *deletion* shape must stay
2653
+ # pruned: without a reload a constant outlives the file that defined
2654
+ # it, so liveness alone proves nothing.
2655
+ #
2656
+ # Identity is loader-derived, not textual: {SourceNesting#
2657
+ # governed_class_name} returns the constant path the active Zeitwerk
2658
+ # loader expects the changed file to define (its inflector, ignores,
2659
+ # and root namespaces decide, via +cpath_expected_at+), gated on the
2660
+ # file actually declaring it. A loader non-claim — an unmanaged or
2661
+ # declined path — is authoritative and re-adds nothing, exactly the
2662
+ # governed-naming contract. That kills the two resurrection shapes a
2663
+ # demodulized class-name regex allowed: a changed file declaring
2664
+ # `Public::User` no longer resurrects a pruned `Admin::User`, and a
2665
+ # file whose comments or string literals mention `class User` no
2666
+ # longer resurrects anything.
2667
+ #
2668
+ # @param pruned [Set<String>] identifiers removed by {#prune_vanished_units}
2669
+ # @param change_set [ChangeSet]
2670
+ # @return [Set<String>] identifiers safe to re-add this run
2671
+ def readdable_pruned_classes(pruned, change_set)
2672
+ readdable = Set.new
2673
+ return readdable if pruned.empty?
2674
+
2675
+ claimed = changed_governed_names(change_set)
2676
+ return readdable if claimed.empty?
2677
+
2678
+ CLASS_BASED_DISCOVERY.each_key do |key|
2679
+ extractor = extractor_for(key)
2680
+ next unless extractor.respond_to?(:discoverable_classes)
2681
+
2682
+ extractor.discoverable_classes.each do |klass|
2683
+ next if klass.name.nil? || !pruned.include?(klass.name)
2684
+
2685
+ readdable.add(klass.name) if claimed.include?(klass.name)
2686
+ end
2687
+ end
2688
+
2689
+ readdable
2690
+ rescue StandardError => e
2691
+ # A failure here must not turn a deletion into a resurrection: the
2692
+ # empty set keeps every pruned identifier excluded, which is the
2693
+ # pre-M1 behavior.
2694
+ Rails.logger.warn "[Woods] Could not determine re-addable pruned classes: #{e.message}"
2695
+ Set.new
2696
+ end
2697
+
2698
+ # Governed constant names of the change set's still-existing Ruby
2699
+ # files. The shared governed-naming helper consults the active Zeitwerk
2700
+ # loader and returns nil for anything it does not claim, so unmanaged
2701
+ # paths (anything outside the loader's roots, under an old Zeitwerk
2702
+ # without +cpath_expected_at+, or a declined file) contribute nothing.
2703
+ #
2704
+ # @param change_set [ChangeSet]
2705
+ # @return [Set<String>] governed constant names of changed files
2706
+ def changed_governed_names(change_set)
2707
+ change_set.existing_paths.filter_map do |path|
2708
+ path = path.to_s
2709
+ next unless path.end_with?('.rb')
2710
+
2711
+ governed_class_name(path, File.read(path))
2712
+ end.compact.to_set
2713
+ end
2714
+
2715
+ def add_discovered_classes(key, spec, discovered, known, excluded, affected_types)
2716
+ new_classes = discovered.reject { |k| known.include?(k.name) || excluded.include?(k.name) }
2717
+ return Set.new if new_classes.empty?
2718
+
2719
+ units = new_classes.filter_map do |klass|
2720
+ extractor_for(key).public_send(spec[:method], klass)
2721
+ rescue StandardError => e
2722
+ Rails.logger.warn "[Woods] #{key} extraction of #{klass} failed: #{e.message}"
2723
+ nil
2724
+ end
2725
+
2726
+ register_and_write(key, units, affected_types)
2727
+ end
2728
+
2729
+ def remove_stale_classes(spec, discovered, known, affected_types)
2730
+ stale = stale_class_based_units(spec[:type], discovered, known)
2731
+ return Set.new if stale.empty?
2732
+
2733
+ Rails.logger.info "[Woods] removing #{stale.size} #{spec[:type]} unit(s) whose class no longer exists"
2734
+ stale.each_with_object(Set.new) do |identifier, removed|
2735
+ removed.add(identifier) if remove_unit(identifier, affected_types, type: spec[:type])
2736
+ end
2737
+ end
2738
+
2739
+ # Class-based units the graph still holds that a full extraction would not
2740
+ # produce.
2741
+ #
2742
+ # {#prune_vanished_units} keys on the source file being gone, which cannot
2743
+ # see this case: a class removed from a file that still exists leaves no
2744
+ # missing path, and class-based units register a *convention* path derived
2745
+ # from the constant name, so a class defined somewhere unconventional was
2746
+ # never attributed to the file it actually lived in. Two models in one
2747
+ # `.rb`, one of them deleted, and the survivor's own re-extraction says
2748
+ # nothing about the other. Nothing else in the run removes it, so it
2749
+ # outlives every subsequent incremental — a permanent divergence from a
2750
+ # full run, not a transient one.
2751
+ #
2752
+ # For all six class-based extractors `extract_all` is literally
2753
+ # `discoverable_classes.map { ... }.compact`, so absence from that set is
2754
+ # exactly "a full extraction would not produce this" — the equivalence the
2755
+ # incremental path is held to.
2756
+ #
2757
+ # The `@eager_load_complete` gate is the whole safety argument, and it is
2758
+ # why this is not simply the inverse of the addition pass:
2759
+ #
2760
+ # * **A partial eager load.** The documented NameError fallback loads only
2761
+ # `EXTRACTION_DIRECTORIES`, so descendants are known-incomplete and the
2762
+ # difference here would be most of the app. Deleting by the type is far
2763
+ # worse than a stale unit, so a partial load removes nothing.
2764
+ # * **A constant outliving its file.** A resident daemon that has not
2765
+ # reloaded still holds a deleted class as a descendant, so it is *in* the
2766
+ # set and not stale — which is correct for that process, and the
2767
+ # subsequent reload is what makes it removable.
2768
+ #
2769
+ # @param type [Symbol] unit type
2770
+ # @param discovered [Array<Class>] the extractor's current discovery set
2771
+ # @param known [Set<String>] identifiers of that type already in the graph
2772
+ # @return [Array<String>] identifiers to remove
2773
+ def stale_class_based_units(type, discovered, known)
2774
+ return [] unless @eager_load_complete
2775
+
2776
+ live = discovered.to_set(&:name)
2777
+ known.reject { |identifier| live.include?(identifier) }
2778
+ # Not redundant with `units_of_type`: an identifier can be listed in
2779
+ # this type's index while also naming a unit of another type, and
2780
+ # the caller removes by (identifier, type), so it has to be told
2781
+ # which node it is allowed to take.
2782
+ .select { |identifier| @dependency_graph.node_types(identifier).include?(type) }
2783
+ end
2784
+
2785
+ # Re-run whole-app extractors whose trigger paths changed, replacing that
2786
+ # unit type wholesale.
2787
+ #
2788
+ # These extractors have no per-file entry point — a route unit is derived
2789
+ # from `Rails.application.routes`, the middleware unit from the live
2790
+ # stack, events from a two-pass scan of all of `app/`. Before this,
2791
+ # incremental runs skipped them entirely, so a routes-only change
2792
+ # triggered a run that re-extracted nothing while still rewriting the
2793
+ # manifest (#164 gap 4). In an already-booted process re-running them is
2794
+ # cheap, which is what makes wholesale replacement the right shape.
2795
+ #
2796
+ # @param change_set [ChangeSet]
2797
+ # @param affected_types [Set<Symbol>]
2798
+ # @return [Set<String>] identifiers written or removed
2799
+ def rerun_whole_app_extractors(change_set, affected_types)
2800
+ keys = PathDispatcher.new.whole_app_keys_for_all(change_set.relative_paths)
2801
+ # The dispatch rules are configuration-blind (memoized per-process), so
2802
+ # the configuration gate applies here — a knob-off host must not re-run
2803
+ # an extractor whose units a full extraction of the same tree would not
2804
+ # produce. See {#skip_by_configuration?}.
2805
+ keys = keys.reject { |key| skip_by_configuration?(key) }.to_set
2806
+ return Set.new if keys.empty?
2807
+
2808
+ keys += ROUTE_CONSUMER_EXTRACTORS if keys.include?(:routes)
2809
+
2810
+ keys.each_with_object(Set.new) do |key, touched|
2811
+ touched.merge(replace_type_wholesale(key, affected_types))
2812
+ end
2813
+ end
2814
+
2815
+ # Replace every unit an extractor owns with a fresh extraction.
2816
+ #
2817
+ # Units of the extractor's types that the fresh run no longer produces are
2818
+ # removed, which is what makes this a replacement rather than an upsert —
2819
+ # subject to the same eager-load gate the reconciler applies, see
2820
+ # {#remove_replaced_units}.
2821
+ #
2822
+ # Fail closed (M8): the rescue exists so a re-run that learned nothing —
2823
+ # `extract_all` itself raising — costs the run nothing. It must not swallow
2824
+ # a failure that lands AFTER the replacement started mutating state a
2825
+ # published generation would carry: {#register_and_write} registers the
2826
+ # graph node before writing the unit file, and {#remove_unit_of_type} rm_f's
2827
+ # the file before dropping the graph node, so a raise in either window
2828
+ # leaves the graph and the payload directory disagreeing. Swallowed, the run
2829
+ # went on to publish a generation whose dependency_graph.json held nodes
2830
+ # with no unit file — `dependencies`/`dependents` reported `found: true`
2831
+ # while lookup returned nil. The counter from {#note_wholesale_mutation}
2832
+ # turns any such failure into a re-raise: the run aborts before
2833
+ # {#publish_generation}, and the preceding generation stays resolved, the
2834
+ # same posture the flow family takes on the full path.
2835
+ #
2836
+ # Two counter rules make that decision sound. The marker is placed BEFORE
2837
+ # each mutation, because registration itself can fail mid-mutation (a
2838
+ # malformed dependency raises after the node is inserted) and a rm_f can
2839
+ # fail part-way; a marker placed after the fact would miss the window it
2840
+ # exists for, at the cost of a conservative abort when the marked mutation
2841
+ # then fails before changing anything. And the counter is reset BEFORE
2842
+ # `extract_all`, not after: {#register_and_write} is shared with the
2843
+ # reconcile paths that run earlier in the same pass, and one of those
2844
+ # leaving the counter positive must not turn a later, mutation-free
2845
+ # wholesale failure into an abort — only the current key's wholesale pass
2846
+ # contributes to the decision.
2847
+ #
2848
+ # @param key [Symbol] extractor key
2849
+ # @param affected_types [Set<Symbol>]
2850
+ # @return [Set<String>] identifiers written or removed
2851
+ # @raise [Woods::ExtractionError] when the replacement failed after
2852
+ # mutating durable state
2853
+ def replace_type_wholesale(key, affected_types)
2854
+ extractor = extractor_for(key)
2855
+ return Set.new unless extractor.respond_to?(:extract_all)
2856
+
2857
+ @wholesale_mutations = 0
2858
+ units = Array(extractor.extract_all).compact.uniq(&:identifier)
2859
+ Rails.logger.info "[Woods] Re-ran #{key} wholesale: #{units.size} units"
2860
+
2861
+ touched = register_and_write(key, units, affected_types)
2862
+ touched.merge(remove_replaced_units(key, units, affected_types))
2863
+ rescue StandardError => e
2864
+ if @wholesale_mutations.to_i.positive?
2865
+ raise Woods::ExtractionError, <<~MSG.tr("\n", ' ').strip
2866
+ Wholesale re-run of #{key} failed after this run had already
2867
+ registered, written, or removed #{@wholesale_mutations} unit
2868
+ artifact(s) (#{e.class}: #{e.message}). Continuing would publish a
2869
+ generation whose dependency graph disagrees with the unit files on
2870
+ disk, so the run aborts before publication and the previous
2871
+ generation stays resolved.
2872
+ MSG
2873
+ end
2874
+
2875
+ Rails.logger.error "[Woods] Wholesale re-run of #{key} failed: #{e.message}"
2876
+ Set.new
2877
+ end
2878
+
2879
+ # Record that the in-flight wholesale replacement is about to mutate state
2880
+ # a published generation would carry: a graph registration, a unit-file
2881
+ # write, or a unit-file removal. Callers mark BEFORE mutating — register
2882
+ # itself can raise mid-mutation (a malformed dependency raises after the
2883
+ # node is inserted), so a marker placed after the fact can miss the very
2884
+ # window it exists for. The cost is a conservative abort when the marked
2885
+ # mutation then fails before changing anything; that direction is safe.
2886
+ # {#replace_type_wholesale} resets the counter to 0 before `extract_all` —
2887
+ # so only the current key's wholesale pass contributes, never the earlier
2888
+ # reconcile passes that share {#register_and_write} — and re-raises from
2889
+ # its rescue when the count is non-zero. Increments from non-wholesale
2890
+ # callers of {#register_and_write} are inert outside that window: nothing
2891
+ # reads the counter between resets.
2892
+ #
2893
+ # @return [void]
2894
+ def note_wholesale_mutation
2895
+ @wholesale_mutations = @wholesale_mutations.to_i + 1
2896
+ end
2897
+
2898
+ # The removal half of a wholesale replacement: drop every unit of `key`'s
2899
+ # types that the fresh run no longer produced.
2900
+ #
2901
+ # Gated exactly the way {#stale_class_based_units} is, and for the same
2902
+ # reason (#198): for the extractors in {CLASS_BASED_DISCOVERY},
2903
+ # `extract_all` is `discoverable_classes.map { ... }` — a function of the
2904
+ # runtime descendants — so on the documented NameError-fallback boot the
2905
+ # fresh set is known-partial and "absent from it" does not mean "deleted".
2906
+ # The reconciler's gate ("the whole safety argument") never protected this
2907
+ # path, and {ROUTE_CONSUMER_EXTRACTORS} routes four descendants-discovered
2908
+ # types through here on *every* routes change — so one routes edit on a
2909
+ # fallback boot mass-deleted most controller units. Registration above is
2910
+ # not gated: additions are safe against a partial set (the same asymmetry
2911
+ # {#reconcile_class_based_types} relies on); only removal needs the
2912
+ # complete one.
2913
+ #
2914
+ # File-derived whole-app types (routes, view_templates, events, ...) keep
2915
+ # unconditional removal — their `extract_all` globs files or introspects
2916
+ # structures that do not depend on eager loading, so their fresh set is
2917
+ # authoritative on any boot.
2918
+ #
2919
+ # The skip is warned, not silent: until a run with a clean boot removes
2920
+ # them, the type may hold stale units, and an operator chasing a ghost
2921
+ # unit needs to know which run declined to delete it and why.
2922
+ #
2923
+ # `fresh` is computed PER unit_type, not once for the whole extractor key
2924
+ # (#225). An identifier is not unique across types — a Scenic view and a
2925
+ # factory can both be named `reports` — so a single identifier-level
2926
+ # `fresh` set let a factories re-run's `database_view` deletion pass see
2927
+ # the surviving `factories` identifier as "still fresh" and skip removing
2928
+ # nothing, then later delete both nodes anyway once `remove_unit` was
2929
+ # called without `type:` (the same call also had to be fixed — see
2930
+ # below). A per-type set also covers a unit reclassified between two
2931
+ # types this same key owns (GraphQL's four): the old type's identifier is
2932
+ # absent from ITS OWN fresh set even though the identifier as a whole is
2933
+ # still "fresh" under the new type, so the stale old-type node is
2934
+ # correctly dropped instead of surviving because the identifier still
2935
+ # exists somewhere in the extractor's output.
2936
+ #
2937
+ # `remove_unit` is called WITH `type: unit_type` for the same reason:
2938
+ # `DependencyGraph#remove`'s own doc warns that a typeless removal fans
2939
+ # over every type registered under the identifier, so calling it here
2940
+ # without `type:` deleted the sibling type's node and JSON file too —
2941
+ # reproduced live as a factories re-run deleting a same-named Scenic-view
2942
+ # unit.
2943
+ #
2944
+ # @param key [Symbol] extractor key
2945
+ # @param units [Array<ExtractedUnit>] the fresh extraction
2946
+ # @param affected_types [Set<Symbol>]
2947
+ # @return [Set<String>] identifiers removed
2948
+ def remove_replaced_units(key, units, affected_types)
2949
+ if CLASS_BASED_DISCOVERY.key?(key) && !@eager_load_complete
2950
+ Rails.logger.warn(
2951
+ "[Woods] Skipping stale-unit removal for #{key}: the eager load was incomplete, " \
2952
+ 'so its discovery set is known-partial — the type may hold stale units until a clean boot'
2953
+ )
2954
+ return Set.new
2955
+ end
2956
+
2957
+ EXTRACTOR_KEY_TO_TYPES.fetch(key, []).each_with_object(Set.new) do |unit_type, removed|
2958
+ fresh = units.select { |u| u.type == unit_type }.to_set(&:identifier)
2959
+ (@dependency_graph.units_of_type(unit_type) - fresh.to_a).each do |stale|
2960
+ removed.add(stale) if remove_unit(stale, affected_types, type: unit_type)
2961
+ end
2962
+ end
2963
+ end
2964
+
2965
+ # Prune units whose source file no longer exists (#164 gap 2).
2966
+ #
2967
+ # Two inputs, with deliberately different authority:
2968
+ #
2969
+ # * **The change set.** A path the caller reports as changed which is no
2970
+ # longer on disk is authoritative — every unit the graph attributes to
2971
+ # it goes, whatever its type. This is the path that handles a deleted
2972
+ # model or controller, and the old side of a rename (git's
2973
+ # `--no-renames` semantics).
2974
+ # * **A sweep** over registered paths, for callers whose change set is
2975
+ # incomplete: a git diff that omits deletions, a watcher that missed an
2976
+ # unlink, a branch switch. The sweep is a heuristic, so it is bounded
2977
+ # twice — see {#sweep_candidates} and the class-based exclusion below.
2978
+ #
2979
+ # Only paths under `Rails.root` are considered either way. Framework
2980
+ # units point at gem paths, and an index restored from a CI artifact can
2981
+ # carry paths produced under a different root; neither is a deletion.
2982
+ #
2983
+ # @param change_set [ChangeSet]
2984
+ # @param affected_types [Set<Symbol>]
2985
+ # @return [Set<String>] identifiers removed
2986
+ def prune_vanished_units(change_set, affected_types)
2987
+ removed = prune_paths(change_set.missing_paths, affected_types, class_based: true)
2988
+ removed.merge(prune_paths(sweep_candidates(change_set), affected_types, class_based: false))
2989
+
2990
+ Rails.logger.info "[Woods] Pruned #{removed.size} unit(s) whose source file is gone" if removed.any?
2991
+ removed
2992
+ end
2993
+
2994
+ # Registered paths that have vanished and that the sweep is allowed to act
2995
+ # on: under Rails.root, gone from disk, and claimed by a file dispatch
2996
+ # rule.
2997
+ #
2998
+ # The file-rule bound exists because some units point at a *nominal* path
2999
+ # rather than a source file — `BehavioralProfile` names
3000
+ # `config/application.rb`, which no rule claims — and sweeping those would
3001
+ # delete units a full extraction still produces.
3002
+ #
3003
+ # @param change_set [ChangeSet]
3004
+ # @return [Array<String>] absolute paths
3005
+ def sweep_candidates(change_set)
3006
+ root_prefix = "#{Rails.root}/"
3007
+ dispatcher = PathDispatcher.new
3008
+ already_named = change_set.missing_paths.to_set
3009
+
3010
+ @dependency_graph.registered_paths.reject do |path|
3011
+ already_named.include?(path) ||
3012
+ !path.to_s.start_with?(root_prefix) ||
3013
+ File.exist?(path) ||
3014
+ dispatcher.file_rules_for(change_set.relativize(path)).empty?
3015
+ end
3016
+ end
3017
+
3018
+ # Remove every unit the graph attributes to each of `paths`.
3019
+ #
3020
+ # @param paths [Enumerable<String>] absolute paths believed to be gone
3021
+ # @param affected_types [Set<Symbol>]
3022
+ # @param class_based [Boolean] whether class-based units may be pruned
3023
+ # @return [Set<String>] identifiers removed
3024
+ def prune_paths(paths, affected_types, class_based:)
3025
+ root_prefix = "#{Rails.root}/"
3026
+
3027
+ paths.each_with_object(Set.new) do |path, removed|
3028
+ next unless path.to_s.start_with?(root_prefix)
3029
+ next if File.exist?(path)
3030
+
3031
+ @dependency_graph.units_for_path(path).each do |identifier, type|
3032
+ next if !class_based && convention_path_unit?(type)
3033
+
3034
+ removed.add(identifier) if remove_unit(identifier, affected_types, type: type)
3035
+ end
3036
+ end
3037
+ end
3038
+
3039
+ # Does this unit's `file_path` name a *convention* that need not exist?
3040
+ #
3041
+ # The sweep deletes units whose file is gone, which is wrong for anything
3042
+ # that derives its path from a constant name rather than from a file it was
3043
+ # read out of. Class-based types have always been excluded for that reason.
3044
+ #
3045
+ # Today the predicate keys on {CLASS_BASED} and nothing else (L4): the six
3046
+ # class-based families — models, controllers, components, view components,
3047
+ # mailers, channels. GRAPHQL_TYPES are deliberately NOT spared. An earlier
3048
+ # version of this comment narrated GraphQL units being spared through a
3049
+ # `source_file_for_class` convention-path fallback, but that fallback now
3050
+ # returns nil rather than fabricating a convention path
3051
+ # (`graphql_extractor.rb`), a runtime-defined GraphQL unit registers no
3052
+ # path at all ({DependencyGraph#register} skips nil), and the static file
3053
+ # pass records the real file it read — so the sweep's own bounds decide,
3054
+ # and this predicate has no say in GraphQL either way.
3055
+ #
3056
+ # The original case: on Rails < 7.1 `ActiveRecord::SchemaMigration` and
3057
+ # `InternalMetadata` are real `ActiveRecord::Base` descendants a full
3058
+ # extraction emits with `app/models/active_record/schema_migration.rb` as
3059
+ # their path — a file no application has, and one the PORO rule *does*
3060
+ # claim, so the sweep's file-rule bound doesn't cover it and this check has
3061
+ # to. Deleting such a unit requires the caller to name the path; the sweep
3062
+ # never infers it.
3063
+ #
3064
+ # KNOWN COST (B-070 / #171): this keys on unit *type*, but the property is
3065
+ # per-unit — CLASS_BASED is also true of units whose recorded path has
3066
+ # since been moved elsewhere, which is exactly the move shape
3067
+ # {#readdable_pruned_classes} exists to re-add in the same run. For
3068
+ # deletions the cost stays: a class-based unit whose file is gone survives
3069
+ # an unnamed-path sweep. Named-path deletion still works, so
3070
+ # `woods:incremental` on a git diff is unaffected; the exposed caller is the
3071
+ # daemon's catch-up, which runs an empty change set precisely because
3072
+ # deletions leave no mtime.
3073
+ #
3074
+ # Two obvious fixes were tried and **both are wrong** — do not re-attempt
3075
+ # them without reading this:
3076
+ #
3077
+ # 1. *Compare the recorded path against the convention path derived from the
3078
+ # constant.* `Types::ForgottenType` conventionally lives at
3079
+ # `app/graphql/types/forgotten_type.rb`, which IS its convention path — so
3080
+ # a conventionally-named file-defined type is indistinguishable from a
3081
+ # runtime-defined one, and the common case stays spared. Disproven by
3082
+ # `spec/integration/incremental_equivalence_spec.rb`'s pending example.
3083
+ # 2. *Let the sweep prune GraphQL and rely on the reconciler's addition half
3084
+ # to re-add whatever the schema still holds.* This is what `except: pruned`
3085
+ # exists to prevent (see the comment at its call site): without a reload a
3086
+ # constant outlives its deleted file, and the re-add resurrects the deleted
3087
+ # unit against a path that no longer exists — permanently, since the sweep
3088
+ # then spares it.
3089
+ #
3090
+ # What would actually work is provenance: record at extraction time whether a
3091
+ # source file existed for the unit and spare only the units that never had
3092
+ # one. That needs the graph node to carry the flag, so it is a serialization
3093
+ # change rather than a predicate tweak.
3094
+ #
3095
+ # @param type [Symbol, nil] the unit type the caller is about to remove
3096
+ # @return [Boolean]
3097
+ def convention_path_unit?(type)
3098
+ CLASS_BASED.key?(type)
3099
+ end
3100
+
3101
+ # Register a batch of freshly-extracted units and write their JSON.
3102
+ #
3103
+ # Registration happens BEFORE path normalization — the graph's file map
3104
+ # stores absolute paths (that is what changed files are matched against),
3105
+ # exactly as full extraction registers in Phase 1 and only normalizes in
3106
+ # Phase 4.5. Unit JSON carries the relative path.
3107
+ #
3108
+ # @param extractor_key [Symbol]
3109
+ # @param units [Array<ExtractedUnit>]
3110
+ # @param affected_types [Set<Symbol>]
3111
+ # @return [Set<String>] identifiers written
3112
+ def register_and_write(extractor_key, units, affected_types)
3113
+ units = Array(units).compact
3114
+ return Set.new if units.empty?
3115
+
3116
+ affected_types&.add(extractor_key)
3117
+ type_dir = payload_dir.join(extractor_key.to_s)
3118
+ FileUtils.mkdir_p(type_dir)
3119
+
3120
+ units.each_with_object(Set.new) do |unit, written|
3121
+ annotate_package(unit)
3122
+ mark_dependents_dirty(unit.identifier)
3123
+ # Marked BEFORE registration: DependencyGraph#register inserts the
3124
+ # node before it iterates the unit's dependencies, so a malformed
3125
+ # dependency raises with the graph already mutated — a marker placed
3126
+ # after the call would never run, and the phantom would ship. The
3127
+ # cost of this ordering is a conservative abort when registration
3128
+ # fails before mutating anything; that is the safe direction to err
3129
+ # in. See {#note_wholesale_mutation}.
3130
+ note_wholesale_mutation
3131
+ @dependency_graph.register(unit)
3132
+ mark_dependents_dirty(unit.identifier)
3133
+
3134
+ unit.file_path = normalize_file_path(unit.file_path)
3135
+ # Keyed by relative path, which is how batch_git_data keys its result.
3136
+ (@incremental_written ||= {})[unit.identifier] = unit.file_path
3137
+
3138
+ write_unit_file(type_dir.join(collision_safe_filename(unit.identifier)), unit)
3139
+ written.add(unit.identifier)
3140
+ end
3141
+ end
3142
+
3143
+ # Remove a unit from the graph and delete its JSON from the index.
3144
+ #
3145
+ # Callers that know which type they mean must say so. An identifier can
3146
+ # name units of several types (a Scenic view `reports` and a factory
3147
+ # `reports`), each with its own `<extractor_key>/<identifier>.json`, and
3148
+ # removing the identifier wholesale takes the sibling with it.
3149
+ #
3150
+ # @param identifier [String]
3151
+ # @param affected_types [Set<Symbol>]
3152
+ # @param type [Symbol, nil] remove only this type; without it, every type
3153
+ # registered under the identifier
3154
+ # @return [String, nil] the identifier when it existed and was removed
3155
+ def remove_unit(identifier, affected_types, type: nil)
3156
+ types = type ? [type] : @dependency_graph.node_types(identifier)
3157
+ types = types.select { |t| @dependency_graph.node(identifier, type: t) }
3158
+ return nil if types.empty?
3159
+
3160
+ # Before the removals: this reads the identifier's forward edges, which
3161
+ # go with the nodes.
3162
+ mark_dependents_dirty(identifier)
3163
+
3164
+ types.each { |t| remove_unit_of_type(identifier, t, affected_types) }
3165
+ Rails.logger.debug { "[Woods] Removed #{identifier}" }
3166
+ identifier
3167
+ end
3168
+
3169
+ # @param identifier [String]
3170
+ # @param type [Symbol]
3171
+ # @param affected_types [Set<Symbol>]
3172
+ # @return [void]
3173
+ def remove_unit_of_type(identifier, type, affected_types)
3174
+ extractor_key = TYPE_TO_EXTRACTOR_KEY[type]
3175
+
3176
+ if extractor_key
3177
+ affected_types&.add(extractor_key)
3178
+ path = payload_dir.join(extractor_key.to_s, collision_safe_filename(identifier))
3179
+ # Marked BEFORE the removal, same ordering as the write half: a
3180
+ # rm_f that fails part-way (or a failure immediately after it) must
3181
+ # not leave the counter at zero while the file is already gone. See
3182
+ # {#note_wholesale_mutation}.
3183
+ note_wholesale_mutation
3184
+ FileUtils.rm_f(path)
3185
+ end
3186
+
3187
+ @dependency_graph.remove(identifier, type: type)
3188
+ end
3189
+
3190
+ # Record that a unit's edges changed, so every target it points at (and
3191
+ # the unit itself) has its `dependents` list rewritten at the end of the
3192
+ # run.
3193
+ #
3194
+ # @param identifier [String]
3195
+ # @return [void]
3196
+ def mark_dependents_dirty(identifier)
3197
+ @dependents_dirty ||= Set.new
3198
+ @dependents_dirty.add(identifier)
3199
+ @dependency_graph.dependencies_of(identifier).each { |target| @dependents_dirty.add(target) }
3200
+ end
3201
+
3202
+ # Second pass over the unit JSON an incremental run touched, mirroring the
3203
+ # two full-extraction phases that operate on already-extracted units:
3204
+ # {#resolve_dependents} (Phase 2) and {#enrich_with_git_data} (Phase 4).
3205
+ #
3206
+ # Both were full-extraction-only, so incremental runs left `dependents`
3207
+ # and `metadata.git` frozen at whatever the last full run wrote —
3208
+ # divergence that compounds run over run in an incremental CI chain.
3209
+ #
3210
+ # @param affected_types [Set<Symbol>]
3211
+ # @return [void]
3212
+ def finalize_incremental_unit_json(affected_types)
3213
+ dependents_dirty = @dependents_dirty || Set.new
3214
+ git_dirty = @incremental_written || {}
3215
+ git_data = incremental_git_data(git_dirty.keys)
3216
+
3217
+ (dependents_dirty | git_dirty.keys).each do |identifier|
3218
+ rewrite_unit_json(identifier, affected_types,
3219
+ refresh_dependents: dependents_dirty.include?(identifier),
3220
+ git_data: git_dirty.key?(identifier) ? git_data : nil)
3221
+ end
3222
+ end
3223
+
3224
+ # Read one unit's JSON, apply the second-pass patches, and write it back
3225
+ # only if something actually changed.
3226
+ #
3227
+ # @param identifier [String]
3228
+ # @param affected_types [Set<Symbol>]
3229
+ # @param refresh_dependents [Boolean]
3230
+ # @param git_data [Hash{String => Hash}, nil] batch git data keyed by
3231
+ # relative path (per {#batch_git_data}), or nil when this identifier
3232
+ # isn't git-dirty this run
3233
+ # @return [void]
3234
+ def rewrite_unit_json(identifier, affected_types, refresh_dependents:, git_data:)
3235
+ @dependency_graph.node_types(identifier).each do |type|
3236
+ rewrite_unit_json_of_type(identifier, type, affected_types,
3237
+ refresh_dependents: refresh_dependents, git_data: git_data)
3238
+ end
3239
+ end
3240
+
3241
+ # One (identifier, type) pair's JSON. The `dependents` list is a property
3242
+ # of the identifier, not of the type, so every file the identifier owns
3243
+ # gets the same refreshed list — which is what a full extraction writes.
3244
+ #
3245
+ # Git metadata is NOT a property of the identifier (#225): a colliding
3246
+ # identifier's types each have their own `file_path` (a Scenic view and a
3247
+ # factory both named `reports` live in different files with different
3248
+ # histories), so this resolves git data against THIS type's own node
3249
+ # rather than a single pre-resolved hash shared across every type — the
3250
+ # previous shape let one type's commit history land in every colliding
3251
+ # type's `metadata.git`, keyed by whichever type {#register_and_write}
3252
+ # happened to touch last for that identifier.
3253
+ #
3254
+ # @param identifier [String]
3255
+ # @param type [Symbol]
3256
+ # @param affected_types [Set<Symbol>]
3257
+ # @param refresh_dependents [Boolean]
3258
+ # @param git_data [Hash{String => Hash}, nil]
3259
+ # @return [void]
3260
+ def rewrite_unit_json_of_type(identifier, type, affected_types, refresh_dependents:, git_data:)
3261
+ extractor_key = TYPE_TO_EXTRACTOR_KEY[type]
3262
+ return unless extractor_key
3263
+
3264
+ path = payload_dir.join(extractor_key.to_s, collision_safe_filename(identifier))
3265
+ return unless File.exist?(path)
3266
+
3267
+ data = JSON.parse(AtomicFile.read(path))
3268
+ # Serialized comparison, not `data.dup`: the git patch mutates the
3269
+ # nested metadata hash in place, which a shallow copy would follow.
3270
+ before = JSON.generate(data)
3271
+
3272
+ if refresh_dependents
3273
+ data['dependents'] = @dependency_graph.dependents_detail(identifier)
3274
+ .map { |d| { 'type' => d[:type].to_s, 'identifier' => d[:identifier] } }
3275
+ end
3276
+
3277
+ git = git_data && git_for_type(identifier, type, git_data)
3278
+ if git
3279
+ (data['metadata'] ||= {})['git'] = JSON.parse(JSON.generate(git))
3280
+ annotate_node_from_git(identifier, type, git)
3281
+ end
3282
+
3283
+ return if JSON.generate(data) == before
3284
+
3285
+ AtomicFile.write(path, json_serialize(data))
3286
+ affected_types&.add(extractor_key)
3287
+ rescue JSON::ParserError => e
3288
+ Rails.logger.warn "[Woods] Could not finalize #{identifier}: #{e.message}"
3289
+ end
3290
+
3291
+ # This (identifier, type) pair's own git data, looked up by its own
3292
+ # `file_path` rather than by identifier — see {#rewrite_unit_json_of_type}.
3293
+ #
3294
+ # @param identifier [String]
3295
+ # @param type [Symbol]
3296
+ # @param git_data [Hash{String => Hash}] keyed by relative path
3297
+ # @return [Hash, nil]
3298
+ def git_for_type(identifier, type, git_data)
3299
+ node = @dependency_graph.node(identifier, type: type)
3300
+ return nil unless node && node[:file_path]
3301
+
3302
+ git_data[normalize_file_path(node[:file_path])]
3303
+ end
3304
+
3305
+ # Mirror of {#annotate_graph_with_git_data} for one incrementally
3306
+ # patched unit. Git data arrives symbol-keyed from {#batch_git_data} and
3307
+ # string-keyed when a caller hands in parsed JSON; both are accepted.
3308
+ #
3309
+ # Unlike {#annotate_graph_with_git_data}, a missing key is never forwarded
3310
+ # as an explicit `nil`: {DependencyGraph#annotate} treats `nil` as
3311
+ # "clear this attribute", and this batch's git data can be missing one of
3312
+ # the two keys without meaning the other's last-known value is gone.
3313
+ #
3314
+ # @param identifier [String]
3315
+ # @param type [Symbol]
3316
+ # @param git [Hash]
3317
+ # @return [void]
3318
+ def annotate_node_from_git(identifier, type, git)
3319
+ attributes = {
3320
+ commit_count: git[:commit_count] || git['commit_count'],
3321
+ change_frequency: git[:change_frequency] || git['change_frequency']
3322
+ }.compact
3323
+ return if attributes.empty?
3324
+
3325
+ @dependency_graph.annotate(identifier, type: type, **attributes)
3326
+ end
3327
+
3328
+ # Batch-fetch git metadata for the units written by this run, in a single
3329
+ # git invocation, keyed by Rails.root-relative path the way
3330
+ # {#batch_git_data} returns it.
3331
+ #
3332
+ # @param identifiers [Array<String>]
3333
+ # @return [Hash{String => Hash}]
3334
+ def incremental_git_data(identifiers)
3335
+ return {} if identifiers.empty? || !git_available?
3336
+
3337
+ paths = identifiers.flat_map do |identifier|
3338
+ @dependency_graph.nodes_for(identifier).filter_map do |node|
3339
+ next if %i[rails_source gem_source].include?(node[:type])
3340
+
3341
+ node[:file_path] if node[:file_path] && File.exist?(node[:file_path])
3342
+ end
3343
+ end
3344
+
3345
+ batch_git_data(paths.uniq)
3346
+ rescue StandardError => e
3347
+ Rails.logger.warn "[Woods] Incremental git enrichment failed: #{e.message}"
3348
+ {}
3349
+ end
3350
+
3351
+ # Re-extract a single known unit, identified by its graph node.
3352
+ #
3353
+ # Used for the transitive half of the blast radius — units whose own file
3354
+ # did not change but which depend on something that did. Files that *did*
3355
+ # change go through {#reconcile_changed_paths} instead, which reconciles
3356
+ # the whole path rather than one identifier.
3357
+ #
3358
+ # @param unit_id [String]
3359
+ # @param affected_types [Set<Symbol>, nil]
3360
+ # @return [String, nil] the identifier when it was re-extracted and written
1090
3361
  def re_extract_unit(unit_id, affected_types: nil)
1091
3362
  # Framework source only changes on version updates
1092
3363
  if unit_id.start_with?('rails/') || unit_id.start_with?('gems/')
1093
- Rails.logger.debug "[Woods] Skipping framework re-extraction for #{unit_id}"
1094
- return
3364
+ Rails.logger.debug { "[Woods] Skipping framework re-extraction for #{unit_id}" }
3365
+ return nil
1095
3366
  end
1096
3367
 
1097
- # Find the unit's type from the graph
1098
- node = @dependency_graph.to_h[:nodes][unit_id]
1099
- return unless node
3368
+ # An identifier can name units of several types, each with its own
3369
+ # extractor and its own file; re-extracting one and calling it done would
3370
+ # leave the others frozen at the pre-change extraction.
3371
+ types = @dependency_graph.node_types(unit_id)
3372
+ return nil if types.empty?
1100
3373
 
1101
- type = node[:type]&.to_sym
1102
- file_path = node[:file_path]
3374
+ re_extracted = types.count { |type| re_extract_unit_of_type(unit_id, type, affected_types) }
3375
+ return nil if re_extracted.zero?
1103
3376
 
1104
- return unless file_path && File.exist?(file_path)
3377
+ Rails.logger.info "[Woods] Re-extracted #{unit_id}"
3378
+ unit_id
3379
+ end
3380
+
3381
+ # @param unit_id [String]
3382
+ # @param type [Symbol]
3383
+ # @param affected_types [Set<Symbol>, nil]
3384
+ # @return [String, nil] the identifier when this type was re-extracted and written
3385
+ def re_extract_unit_of_type(unit_id, type, affected_types)
3386
+ node = @dependency_graph.node(unit_id, type: type)
3387
+ file_path = node && node[:file_path]
3388
+
3389
+ # A vanished file is not re-extractable; {#prune_vanished_units} owns it.
3390
+ return nil unless file_path && File.exist?(file_path)
1105
3391
 
1106
- # Re-extract based on type
1107
3392
  extractor_key = TYPE_TO_EXTRACTOR_KEY[type]
1108
- return unless extractor_key
3393
+ return nil unless extractor_key
1109
3394
 
1110
- extractor = EXTRACTORS[extractor_key]&.new
1111
- return unless extractor
1112
-
1113
- unit = if (method = CLASS_BASED[type])
1114
- klass = if unit_id.match?(/\A[A-Z][A-Za-z0-9_:]*\z/)
1115
- begin
1116
- unit_id.constantize
1117
- rescue StandardError
1118
- nil
1119
- end
1120
- end
1121
- extractor.public_send(method, klass) if klass
1122
- elsif (method = FILE_BASED[type])
1123
- extractor.public_send(method, file_path)
1124
- elsif GRAPHQL_TYPES.include?(type)
1125
- extractor.extract_graphql_file(file_path)
1126
- end
1127
-
1128
- return unless unit
3395
+ extractor = extractor_for(extractor_key)
3396
+ return nil unless extractor
1129
3397
 
1130
3398
  # File-based extractors can return several units from one file (a .rake
1131
3399
  # file defining multiple tasks, etc.); class-based extractors return one.
1132
- # Normalize to an array so every unit is registered and written — passing
1133
- # an Array straight to DependencyGraph#register crashes on unit.identifier.
1134
- units = unit.is_a?(Array) ? unit : [unit]
1135
- return if units.empty?
3400
+ units = Array(re_extracted_units(extractor, type, unit_id, file_path, extractor_key)).compact
3401
+ return nil if units.empty?
1136
3402
 
1137
- # Track which type was affected
1138
- affected_types&.add(extractor_key)
3403
+ register_and_write(extractor_key, units, affected_types)
3404
+ unit_id
3405
+ end
3406
+
3407
+ # Dispatch one re-extraction to the right extractor entry point.
3408
+ #
3409
+ # @return [ExtractedUnit, Array<ExtractedUnit>, nil]
3410
+ def re_extracted_units(extractor, type, unit_id, file_path, extractor_key)
3411
+ if (method = CLASS_BASED[type])
3412
+ klass = constant_for_identifier(unit_id)
3413
+ klass && extractor.public_send(method, klass)
3414
+ elsif (method = FILE_BASED[type])
3415
+ units = Array(extract_file_based_unit(extractor, method, file_path, extractor_key)).compact
3416
+ return units unless CLASS_DISCOVERED_FALLBACK.key?(type)
3417
+ return units if units.any? { |unit| unit.identifier == unit_id }
3418
+
3419
+ re_extract_by_class(extractor, type, unit_id)
3420
+ elsif GRAPHQL_TYPES.include?(type)
3421
+ extractor.extract_graphql_file(file_path)
3422
+ end
3423
+ end
1139
3424
 
1140
- type_dir = @output_dir.join(extractor_key.to_s)
3425
+ # Re-extract a unit the full path discovered by class, not by file.
3426
+ #
3427
+ # Jobs are found two ways ({Extractors::JobExtractor#extract_all}): a scan
3428
+ # of the job directories, then a descendant walk for every job class the
3429
+ # scan did not name. A class-discovered job can live in a file whose
3430
+ # governed constant is not the job — a job nested inside a model — so
3431
+ # re-deriving it from that file names the enclosing class instead, and
3432
+ # registered a second job unit under the PORO's identifier that no full
3433
+ # extraction emits: the wrapper-naming collision, reached incrementally.
3434
+ # When the file entry point does not reproduce the unit, the full path
3435
+ # found it by class; do the same, and register nothing the file scan
3436
+ # would not have produced there either.
3437
+ #
3438
+ # Only a class the extractor itself would discover qualifies. A stale
3439
+ # pre-2.0 wrapper identifier can still constantize (to the wrapper class),
3440
+ # and re-extracting *that* by class would pin the stale unit with fresh
3441
+ # metadata; it stays as it is until a full extraction replaces it.
3442
+ #
3443
+ # @return [ExtractedUnit, nil]
3444
+ def re_extract_by_class(extractor, type, unit_id)
3445
+ klass = constant_for_identifier(unit_id)
3446
+ return nil unless klass && extractor.discoverable_classes.include?(klass)
1141
3447
 
1142
- units.each do |extracted|
1143
- # Update dependency graph. Register BEFORE normalizing the path —
1144
- # the graph's file_map stores absolute paths (affected_by matches
1145
- # changed files against them), exactly as full extraction registers
1146
- # in Phase 1 and only normalizes in Phase 4.5.
1147
- @dependency_graph.register(extracted)
3448
+ extractor.public_send(CLASS_DISCOVERED_FALLBACK[type], klass)
3449
+ end
1148
3450
 
1149
- # Unit JSON carries Rails.root-relative paths (full extraction's
1150
- # Phase 4.5); writing the raw absolute source_location here would
1151
- # leak container-absolute paths into the index after incremental runs.
1152
- extracted.file_path = normalize_file_path(extracted.file_path)
3451
+ # @param unit_id [String]
3452
+ # @return [Class, nil] the constant an identifier names, when it names one
3453
+ def constant_for_identifier(unit_id)
3454
+ return nil unless unit_id.match?(/\A[A-Z][A-Za-z0-9_:]*\z/)
1153
3455
 
1154
- # Write updated unit
1155
- File.write(
1156
- type_dir.join(collision_safe_filename(extracted.identifier)),
1157
- json_serialize(extracted.to_h)
1158
- )
3456
+ begin
3457
+ unit_id.constantize
3458
+ rescue StandardError
3459
+ nil
1159
3460
  end
3461
+ end
1160
3462
 
1161
- Rails.logger.info "[Woods] Re-extracted #{unit_id}"
3463
+ # Invoke a file-based extraction method, supplying the extra arguments
3464
+ # the handful of non-uniform signatures need.
3465
+ #
3466
+ # @return [ExtractedUnit, Array<ExtractedUnit>, nil]
3467
+ def extract_file_based_unit(extractor, method, file_path, extractor_key)
3468
+ if extractor_key == :poros
3469
+ extractor.public_send(method, file_path, ar_names: active_record_names)
3470
+ else
3471
+ extractor.public_send(method, file_path)
3472
+ end
1162
3473
  end
1163
3474
  end
1164
3475
  end