woods 1.6.1 → 2.0.0.beta2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (274) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +2035 -0
  3. data/CONTRIBUTING.md +253 -87
  4. data/README.md +161 -513
  5. data/SECURITY.md +92 -0
  6. data/assets/woods-wordmark-white-with-bg.png +0 -0
  7. data/docs/AGENT_GUIDE.md +204 -0
  8. data/docs/AGENT_SETUP.md +205 -0
  9. data/docs/BACKEND_MATRIX.md +470 -0
  10. data/docs/CONFIGURATION_REFERENCE.md +655 -0
  11. data/docs/CONSOLE_MCP_SETUP.md +829 -0
  12. data/docs/DOCKER_SETUP.md +454 -0
  13. data/docs/EMBEDDING_MODELS.md +136 -0
  14. data/docs/EVALUATION.md +91 -0
  15. data/docs/EXTRACTOR_REFERENCE.md +765 -0
  16. data/docs/FAQ.md +544 -0
  17. data/docs/GETTING_STARTED.md +183 -0
  18. data/docs/INCREMENTAL_EXTRACTION.md +455 -0
  19. data/docs/INTERNALS.md +418 -0
  20. data/docs/MCP_HTTP_TRANSPORT.md +144 -0
  21. data/docs/MCP_SERVERS.md +231 -0
  22. data/docs/MCP_TOOL_COOKBOOK.md +987 -0
  23. data/docs/MCP_WORKTREE_SETUP.md +127 -0
  24. data/docs/NOTION_INTEGRATION.md +283 -0
  25. data/docs/OBSIDIAN_INTEGRATION.md +170 -0
  26. data/docs/PUBLISHED_INDEX.md +213 -0
  27. data/docs/README.md +94 -0
  28. data/docs/RETRIEVAL_GUIDE.md +267 -0
  29. data/docs/TOKEN_BENCHMARK.md +68 -0
  30. data/docs/TROUBLESHOOTING.md +841 -0
  31. data/docs/UNBLOCKED_INTEGRATION.md +279 -0
  32. data/docs/UPGRADING_TO_2.md +321 -0
  33. data/docs/WATCH_DAEMON.md +667 -0
  34. data/docs/WHY_WOODS.md +219 -0
  35. data/exe/woods-console +40 -4
  36. data/exe/woods-console-mcp +21 -35
  37. data/exe/woods-mcp +20 -7
  38. data/exe/woods-mcp-http +80 -11
  39. data/exe/woods-mcp-start +57 -52
  40. data/lib/generators/woods/install_generator.rb +6 -5
  41. data/lib/generators/woods/pgvector_generator.rb +6 -3
  42. data/lib/generators/woods/templates/add_pgvector_to_woods.rb.erb +29 -9
  43. data/lib/generators/woods/templates/create_woods_tables.rb.erb +5 -1
  44. data/lib/generators/woods/templates/woods.rb.tt +49 -28
  45. data/lib/tasks/woods.rake +622 -168
  46. data/lib/tasks/woods_checks.rake +107 -0
  47. data/lib/tasks/woods_evaluation.rake +164 -80
  48. data/lib/woods/ast/call_site_extractor.rb +6 -15
  49. data/lib/woods/ast/method_extractor.rb +19 -9
  50. data/lib/woods/ast/parser.rb +54 -8
  51. data/lib/woods/atomic_file.rb +171 -2
  52. data/lib/woods/builder.rb +310 -22
  53. data/lib/woods/cache/cache_middleware.rb +7 -2
  54. data/lib/woods/cache/cache_store.rb +9 -1
  55. data/lib/woods/cache/solid_cache_store.rb +6 -4
  56. data/lib/woods/change_set.rb +88 -0
  57. data/lib/woods/checks/generation_resolution.rb +34 -0
  58. data/lib/woods/checks/moved_messages.rb +186 -0
  59. data/lib/woods/chunking/semantic_chunker.rb +160 -18
  60. data/lib/woods/console/audit_logger.rb +12 -3
  61. data/lib/woods/console/bridge_protocol.rb +3 -16
  62. data/lib/woods/console/connection_manager.rb +51 -136
  63. data/lib/woods/console/dispatch_pipeline.rb +42 -12
  64. data/lib/woods/console/embedded_executor.rb +806 -149
  65. data/lib/woods/console/eval_guard.rb +27 -20
  66. data/lib/woods/console/input_contract.rb +78 -0
  67. data/lib/woods/console/model_validator.rb +29 -1
  68. data/lib/woods/console/rack_middleware.rb +65 -42
  69. data/lib/woods/console/redactor.rb +26 -8
  70. data/lib/woods/console/safe_context.rb +58 -10
  71. data/lib/woods/console/scope_predicate_parser.rb +41 -0
  72. data/lib/woods/console/server.rb +119 -247
  73. data/lib/woods/console/sql_noise_stripper.rb +125 -16
  74. data/lib/woods/console/sql_table_scanner.rb +82 -22
  75. data/lib/woods/console/sql_validator.rb +459 -29
  76. data/lib/woods/console/table_gate.rb +2 -2
  77. data/lib/woods/console/tool_specs.rb +463 -90
  78. data/lib/woods/console/tools/tier1.rb +1 -5
  79. data/lib/woods/console/tools/tier4.rb +18 -9
  80. data/lib/woods/coordination/lock_heartbeat.rb +103 -0
  81. data/lib/woods/coordination/pipeline_lock.rb +263 -53
  82. data/lib/woods/db/migrations/007_typed_snapshot_units.rb +45 -0
  83. data/lib/woods/db/migrator.rb +3 -9
  84. data/lib/woods/db/schema_version.rb +47 -2
  85. data/lib/woods/dependency_graph.rb +898 -64
  86. data/lib/woods/embedding/fake.rb +138 -0
  87. data/lib/woods/embedding/indexer.rb +832 -40
  88. data/lib/woods/embedding/openai.rb +77 -19
  89. data/lib/woods/embedding/provider.rb +189 -11
  90. data/lib/woods/embedding/text_preparer.rb +1 -1
  91. data/lib/woods/embedding/token_counter.rb +0 -7
  92. data/lib/woods/evaluation/ablation_agent_payload.rb +38 -0
  93. data/lib/woods/evaluation/ablation_executor.rb +67 -0
  94. data/lib/woods/evaluation/ablation_provenance.rb +38 -0
  95. data/lib/woods/evaluation/ablation_report_writer.rb +43 -0
  96. data/lib/woods/evaluation/ablation_runner.rb +173 -0
  97. data/lib/woods/evaluation/ablation_summary.rb +65 -0
  98. data/lib/woods/evaluation/ablation_task.rb +66 -0
  99. data/lib/woods/evaluation/ablation_task_set.rb +77 -0
  100. data/lib/woods/evaluation/ablation_timed_executor.rb +91 -0
  101. data/lib/woods/evaluation/ablation_worktree.rb +71 -0
  102. data/lib/woods/evaluation/baseline.rb +60 -0
  103. data/lib/woods/evaluation/baseline_runner.rb +11 -3
  104. data/lib/woods/evaluation/evaluator.rb +41 -8
  105. data/lib/woods/evaluation/query_set.rb +79 -13
  106. data/lib/woods/evaluation/report_generator.rb +20 -1
  107. data/lib/woods/export/unit_facts.rb +0 -11
  108. data/lib/woods/extracted_unit.rb +22 -63
  109. data/lib/woods/extractor.rb +2783 -238
  110. data/lib/woods/extractors/action_cable_extractor.rb +9 -4
  111. data/lib/woods/extractors/ast_source_extraction.rb +20 -2
  112. data/lib/woods/extractors/caching_extractor.rb +46 -12
  113. data/lib/woods/extractors/callback_analyzer.rb +39 -9
  114. data/lib/woods/extractors/component_discovery.rb +123 -0
  115. data/lib/woods/extractors/concern_extractor.rb +17 -3
  116. data/lib/woods/extractors/controller_extractor.rb +389 -29
  117. data/lib/woods/extractors/decorator_extractor.rb +7 -14
  118. data/lib/woods/extractors/engine_extractor.rb +53 -8
  119. data/lib/woods/extractors/event_extractor.rb +55 -4
  120. data/lib/woods/extractors/factory_extractor.rb +49 -11
  121. data/lib/woods/extractors/graphql_extractor.rb +162 -66
  122. data/lib/woods/extractors/i18n_extractor.rb +6 -1
  123. data/lib/woods/extractors/job_extractor.rb +51 -21
  124. data/lib/woods/extractors/lib_extractor.rb +23 -17
  125. data/lib/woods/extractors/line_neutralizer.rb +171 -0
  126. data/lib/woods/extractors/mailer_extractor.rb +9 -1
  127. data/lib/woods/extractors/manager_extractor.rb +19 -2
  128. data/lib/woods/extractors/migration_extractor.rb +22 -11
  129. data/lib/woods/extractors/model_extractor.rb +292 -57
  130. data/lib/woods/extractors/package_extractor.rb +154 -0
  131. data/lib/woods/extractors/phlex_extractor.rb +18 -3
  132. data/lib/woods/extractors/policy_extractor.rb +6 -5
  133. data/lib/woods/extractors/poro_extractor.rb +13 -14
  134. data/lib/woods/extractors/pundit_extractor.rb +3 -3
  135. data/lib/woods/extractors/rails_source_extractor.rb +24 -7
  136. data/lib/woods/extractors/rake_task_extractor.rb +158 -30
  137. data/lib/woods/extractors/reference_patterns.rb +38 -0
  138. data/lib/woods/extractors/route_extractor.rb +58 -2
  139. data/lib/woods/extractors/scheduled_job_extractor.rb +51 -35
  140. data/lib/woods/extractors/serializer_extractor.rb +3 -4
  141. data/lib/woods/extractors/service_extractor.rb +11 -1
  142. data/lib/woods/extractors/shared_dependency_scanner.rb +24 -34
  143. data/lib/woods/extractors/shared_utility_methods.rb +36 -6
  144. data/lib/woods/extractors/source_nesting.rb +560 -0
  145. data/lib/woods/extractors/state_machine_extractor.rb +30 -18
  146. data/lib/woods/extractors/test_mapping_extractor.rb +26 -9
  147. data/lib/woods/extractors/view_component_extractor.rb +28 -3
  148. data/lib/woods/extractors/view_engines/erb.rb +17 -3
  149. data/lib/woods/feedback/gap_detector.rb +9 -3
  150. data/lib/woods/feedback/store.rb +7 -1
  151. data/lib/woods/filename_utils.rb +29 -1
  152. data/lib/woods/flow_analysis/operation_extractor.rb +22 -10
  153. data/lib/woods/flow_assembler.rb +147 -26
  154. data/lib/woods/flow_document.rb +1 -0
  155. data/lib/woods/flow_precomputer.rb +175 -22
  156. data/lib/woods/gem_mapper.rb +285 -0
  157. data/lib/woods/generation.rb +185 -0
  158. data/lib/woods/git_command.rb +38 -0
  159. data/lib/woods/git_provenance.rb +16 -2
  160. data/lib/woods/graph_analyzer.rb +564 -87
  161. data/lib/woods/index_artifact.rb +93 -23
  162. data/lib/woods/mcp/bearer_auth.rb +102 -13
  163. data/lib/woods/mcp/bootstrap_state.rb +77 -0
  164. data/lib/woods/mcp/bootstrapper.rb +582 -77
  165. data/lib/woods/mcp/config_resolver.rb +66 -6
  166. data/lib/woods/mcp/errors.rb +60 -0
  167. data/lib/woods/mcp/index_reader.rb +836 -117
  168. data/lib/woods/mcp/index_reader_pinning.rb +78 -0
  169. data/lib/woods/mcp/origin_guard.rb +66 -7
  170. data/lib/woods/mcp/protocol_policy.rb +98 -0
  171. data/lib/woods/mcp/provider_probe.rb +45 -6
  172. data/lib/woods/mcp/renderers/markdown_renderer.rb +72 -4
  173. data/lib/woods/mcp/renderers/plain_renderer.rb +54 -6
  174. data/lib/woods/mcp/server.rb +898 -152
  175. data/lib/woods/mcp/tasks/extension.rb +196 -0
  176. data/lib/woods/mcp/tasks/request_capture.rb +45 -0
  177. data/lib/woods/mcp/tasks/store.rb +518 -0
  178. data/lib/woods/mcp/tool_contract.rb +171 -0
  179. data/lib/woods/mcp/tool_response_renderer.rb +7 -0
  180. data/lib/woods/model_name_cache.rb +19 -1
  181. data/lib/woods/notion/client.rb +132 -36
  182. data/lib/woods/notion/exporter.rb +456 -61
  183. data/lib/woods/notion/mappers/column_mapper.rb +34 -5
  184. data/lib/woods/notion/mappers/migration_mapper.rb +32 -8
  185. data/lib/woods/notion/mappers/model_mapper.rb +21 -6
  186. data/lib/woods/notion/mappers/shared.rb +45 -3
  187. data/lib/woods/notion/sync_manifest.rb +258 -0
  188. data/lib/woods/obsidian/errors.rb +6 -0
  189. data/lib/woods/obsidian/name_mapper.rb +40 -24
  190. data/lib/woods/obsidian/vault_exporter.rb +103 -36
  191. data/lib/woods/operator/pipeline_guard.rb +118 -21
  192. data/lib/woods/operator/status_reporter.rb +20 -3
  193. data/lib/woods/path_dispatcher.rb +276 -0
  194. data/lib/woods/payload_store.rb +236 -0
  195. data/lib/woods/published_index/edge_shaper.rb +61 -0
  196. data/lib/woods/published_index/generation_catalog.rb +72 -0
  197. data/lib/woods/published_index/typed_unit_reader.rb +48 -0
  198. data/lib/woods/published_index.rb +287 -0
  199. data/lib/woods/railtie.rb +69 -30
  200. data/lib/woods/railtie_support.rb +167 -0
  201. data/lib/woods/release.rb +12 -0
  202. data/lib/woods/reload_policy.rb +206 -0
  203. data/lib/woods/resilience/circuit_breaker.rb +47 -8
  204. data/lib/woods/resilience/index_validator.rb +296 -10
  205. data/lib/woods/resilience/retryable_provider.rb +71 -6
  206. data/lib/woods/resolved_config.rb +55 -11
  207. data/lib/woods/retrieval/context_assembler.rb +132 -40
  208. data/lib/woods/retrieval/query_classifier.rb +26 -8
  209. data/lib/woods/retrieval/ranker.rb +193 -28
  210. data/lib/woods/retrieval/search_executor.rb +206 -39
  211. data/lib/woods/retriever.rb +317 -71
  212. data/lib/woods/retry_after.rb +22 -2
  213. data/lib/woods/ruby_analyzer/class_analyzer.rb +10 -14
  214. data/lib/woods/ruby_analyzer/fqn_builder.rb +2 -0
  215. data/lib/woods/ruby_analyzer/mermaid_renderer.rb +14 -4
  216. data/lib/woods/ruby_analyzer/method_analyzer.rb +1 -1
  217. data/lib/woods/ruby_analyzer/trace_enricher.rb +3 -0
  218. data/lib/woods/ruby_analyzer.rb +21 -5
  219. data/lib/woods/session_tracer/file_store.rb +138 -19
  220. data/lib/woods/session_tracer/middleware.rb +1 -2
  221. data/lib/woods/session_tracer/redis_store.rb +122 -12
  222. data/lib/woods/session_tracer/session_flow_assembler.rb +57 -17
  223. data/lib/woods/session_tracer/session_flow_document.rb +56 -14
  224. data/lib/woods/session_tracer/solid_cache_coordination.rb +192 -0
  225. data/lib/woods/session_tracer/solid_cache_store.rb +560 -91
  226. data/lib/woods/session_tracer/store.rb +14 -1
  227. data/lib/woods/storage/metadata_store.rb +230 -26
  228. data/lib/woods/storage/pgvector.rb +180 -22
  229. data/lib/woods/storage/qdrant.rb +367 -41
  230. data/lib/woods/storage/snapshotter/metadata.rb +79 -16
  231. data/lib/woods/storage/snapshotter/vector.rb +128 -17
  232. data/lib/woods/storage/snapshotter.rb +23 -5
  233. data/lib/woods/storage/vector_store.rb +49 -8
  234. data/lib/woods/storage_identity.rb +28 -0
  235. data/lib/woods/tasks.rb +53 -2
  236. data/lib/woods/temporal/json_snapshot_store.rb +112 -42
  237. data/lib/woods/temporal/snapshot_store.rb +139 -42
  238. data/lib/woods/unblocked/client.rb +119 -17
  239. data/lib/woods/unblocked/document_builder.rb +34 -2
  240. data/lib/woods/unblocked/exporter.rb +63 -27
  241. data/lib/woods/unblocked/rate_limiter.rb +23 -9
  242. data/lib/woods/unblocked/sync_manifest.rb +16 -8
  243. data/lib/woods/update_check.rb +24 -1
  244. data/lib/woods/util/uuid5.rb +124 -0
  245. data/lib/woods/version.rb +1 -1
  246. data/lib/woods/watch/daemon.rb +1345 -0
  247. data/lib/woods/watch/listen_watcher.rb +81 -0
  248. data/lib/woods/watch/polling_watcher.rb +137 -0
  249. data/lib/woods/watch/status.rb +169 -0
  250. data/lib/woods/watch/tree_scan.rb +163 -0
  251. data/lib/woods/watch/watcher.rb +100 -0
  252. data/lib/woods.rb +138 -9
  253. data/plugin/.claude-plugin/plugin.json +18 -0
  254. data/plugin/hooks/hooks.json +29 -0
  255. data/plugin/hooks/woods-post-edit.sh +226 -0
  256. data/plugin/hooks/woods-session-start.sh +77 -0
  257. data/plugin/skills/woods-agent-enable/SKILL.md +51 -0
  258. data/plugin/skills/woods-diagnose/SKILL.md +75 -0
  259. data/plugin/skills/woods-investigate/SKILL.md +39 -0
  260. data/plugin/skills/woods-mcp-config/SKILL.md +101 -0
  261. data/plugin/skills/woods-setup/SKILL.md +99 -0
  262. metadata +134 -23
  263. data/lib/woods/console/adapters/cache_adapter.rb +0 -58
  264. data/lib/woods/console/adapters/good_job_adapter.rb +0 -33
  265. data/lib/woods/console/adapters/job_adapter.rb +0 -74
  266. data/lib/woods/console/adapters/sidekiq_adapter.rb +0 -33
  267. data/lib/woods/console/adapters/solid_queue_adapter.rb +0 -33
  268. data/lib/woods/console/bridge.rb +0 -210
  269. data/lib/woods/formatting/claude_adapter.rb +0 -98
  270. data/lib/woods/formatting/generic_adapter.rb +0 -56
  271. data/lib/woods/formatting/gpt_adapter.rb +0 -64
  272. data/lib/woods/notion/mapper.rb +0 -40
  273. data/lib/woods/observability/health_check.rb +0 -79
  274. data/lib/woods/observability/instrumentation.rb +0 -34
@@ -7,10 +7,12 @@ require 'open3'
7
7
  require 'pathname'
8
8
  require 'set'
9
9
 
10
+ require_relative 'atomic_file'
10
11
  require_relative 'filename_utils'
11
12
  require_relative 'token_utils'
12
13
  require_relative 'extracted_unit'
13
14
  require_relative 'dependency_graph'
15
+ require_relative 'payload_store'
14
16
  require_relative 'git_provenance'
15
17
  require_relative 'extractors/model_extractor'
16
18
  require_relative 'extractors/controller_extractor'
@@ -46,9 +48,13 @@ require_relative 'extractors/factory_extractor'
46
48
  require_relative 'extractors/test_mapping_extractor'
47
49
  require_relative 'extractors/poro_extractor'
48
50
  require_relative 'extractors/lib_extractor'
51
+ require_relative 'extractors/package_extractor'
49
52
  require_relative 'graph_analyzer'
50
53
  require_relative 'model_name_cache'
51
54
  require_relative 'flow_precomputer'
55
+ require_relative 'change_set'
56
+ require_relative 'generation'
57
+ require_relative 'path_dispatcher'
52
58
 
53
59
  module Woods
54
60
  # Extractor is the main orchestrator for codebase extraction.
@@ -66,6 +72,7 @@ module Woods
66
72
  #
67
73
  class Extractor
68
74
  include FilenameUtils
75
+ include Extractors::SourceNesting
69
76
 
70
77
  # Directories under app/ that contain classes we need to extract.
71
78
  # Used by eager_load_extraction_directories as a fallback when
@@ -126,7 +133,8 @@ module Woods
126
133
  test_mappings: Extractors::TestMappingExtractor,
127
134
  rails_source: Extractors::RailsSourceExtractor,
128
135
  poros: Extractors::PoroExtractor,
129
- libs: Extractors::LibExtractor
136
+ libs: Extractors::LibExtractor,
137
+ packages: Extractors::PackageExtractor
130
138
  }.freeze
131
139
 
132
140
  # Maps singular unit types (as stored in ExtractedUnit/graph nodes)
@@ -169,8 +177,17 @@ module Woods
169
177
  factory: :factories,
170
178
  test_mapping: :test_mappings,
171
179
  rails_source: :rails_source,
180
+ # RailsSourceExtractor emits BOTH :rails_source and :gem_source units
181
+ # into the rails_source/ output directory. Without this entry the
182
+ # wholesale-replacement path could not prune a stale gem_source unit
183
+ # ({EXTRACTOR_KEY_TO_TYPES} would not list the type) and {#remove_unit}
184
+ # could not resolve its JSON file on disk (#169) — the GraphQL history
185
+ # in CLAUDE.md is the cautionary tale for an extractor whose emitted
186
+ # types are not all mapped.
187
+ gem_source: :rails_source,
172
188
  poro: :poros,
173
- lib: :libs
189
+ lib: :libs,
190
+ package: :packages
174
191
  }.freeze
175
192
 
176
193
  # Maps unit types to class-based extractor methods (constantize + call).
@@ -203,13 +220,170 @@ module Woods
203
220
  # GraphQL types all use the same extractor method.
204
221
  GRAPHQL_TYPES = %i[graphql_type graphql_mutation graphql_resolver graphql_query].freeze
205
222
 
223
+ # File-based unit types the full path ALSO discovers by class, with the
224
+ # class-based entry point to re-extract them through. Re-extracting one
225
+ # of these by file is faithful only when the file names the unit; see
226
+ # {#re_extracted_units}.
227
+ CLASS_DISCOVERED_FALLBACK = { job: :extract_job_class }.freeze
228
+
229
+ # Unit types each extractor owns — the inverse of {TYPE_TO_EXTRACTOR_KEY}.
230
+ #
231
+ # Wholesale replacement of an extractor's output has to know every type it
232
+ # can produce, and one extractor can own several (`graphql` covers types,
233
+ # mutations, resolvers and queries).
234
+ #
235
+ # @return [Hash{Symbol => Array<Symbol>}]
236
+ EXTRACTOR_KEY_TO_TYPES = TYPE_TO_EXTRACTOR_KEY.each_with_object({}) do |(type, key), map|
237
+ (map[key] ||= []) << type
238
+ end.freeze
239
+
240
+ # Class-based extractors, which discover their units by walking runtime
241
+ # descendants rather than by globbing files. The incremental path
242
+ # reconciles each extractor's `discoverable_classes` against the
243
+ # identifiers already in the graph, so a class added since the last
244
+ # extraction is found without having to guess a constant name from a
245
+ # path (#164, gap 1).
246
+ #
247
+ # `reconcile_removals: false` opts an entry out of the removal half. Only
248
+ # GraphQL sets it, and the reason is that its unit type is produced by two
249
+ # independent discovery mechanisms whose union is the truth: runtime schema
250
+ # introspection *and* a static file pass over `app/graphql`. Every other
251
+ # entry here owns its type outright, which is what makes
252
+ # "in the graph but not in `discoverable_classes`" mean "deleted".
253
+ #
254
+ # For GraphQL that inference is wrong twice over. Without graphql-ruby
255
+ # loaded the runtime set is empty, so removal would delete every GraphQL
256
+ # unit in the index; with it loaded, a type defined in a file but not
257
+ # attached to the schema is legitimately absent from the type map, so
258
+ # removal would delete units a full extraction still emits. Additions are
259
+ # safe in both worlds — they are exactly the runtime-only types no changed
260
+ # path can dispatch to, which is the #167 divergence.
261
+ #
262
+ # The consequence, stated so it does not read as a bug later: a runtime-only
263
+ # type that *disappears* is never removed by an incremental run. It survives
264
+ # until a full extraction rebuilds the type. Confirmed against a live schema
265
+ # during review — probe units outlived deletion of the file that defined
266
+ # them. That is the cost of the opt-out, and it is the right side to err on:
267
+ # a stale unit is recoverable by one full run, whereas the removal half
268
+ # would delete units a full extraction still emits.
269
+ #
270
+ # @return [Hash{Symbol => Hash}] extractor key => { type:, method: }
271
+ CLASS_BASED_DISCOVERY = {
272
+ models: { type: :model, method: :extract_model },
273
+ controllers: { type: :controller, method: :extract_controller },
274
+ mailers: { type: :mailer, method: :extract_mailer },
275
+ components: { type: :component, method: :extract_component },
276
+ view_components: { type: :view_component, method: :extract_component },
277
+ action_cable_channels: { type: :action_cable_channel, method: :extract_channel },
278
+ # Additions only — see `reconcile_removals: false`. GraphQL is the one
279
+ # entry whose discovery set is not authoritative for its unit type
280
+ # (#167).
281
+ # `types:` because this extractor emits four unit types, not one:
282
+ # `classify_runtime_type` returns :graphql_type, :graphql_mutation,
283
+ # :graphql_resolver or :graphql_query. Declaring only :graphql_type made
284
+ # `known` miss the others — the schema's query root is always in
285
+ # `Schema.types` and classifies as :graphql_query — so they were
286
+ # re-"discovered" and re-added on *every* incremental run. That leaves
287
+ # `touched` non-empty on a genuine no-op, which rewrites the manifest and
288
+ # bumps the generation each cycle: the #164 gap-4 symptom, reintroduced.
289
+ graphql: { type: :graphql_type, types: GRAPHQL_TYPES,
290
+ method: :extract_from_runtime_type, reconcile_removals: false }
291
+ }.freeze
292
+
293
+ # Extractors with no per-file entry point: they scan the whole app (or
294
+ # introspect the whole runtime) in one pass, so an incremental run
295
+ # replaces their output wholesale rather than per unit. Before #164
296
+ # these types were simply skipped by incremental runs while
297
+ # `config/routes.rb` still triggered one — the run rewrote the manifest,
298
+ # zeroing the staleness clock, without updating the data.
299
+ #
300
+ # @return [Hash{Symbol => Symbol}] extractor key => unit type
301
+ WHOLE_APP_EXTRACTORS = {
302
+ routes: :route,
303
+ middleware: :middleware,
304
+ engines: :engine,
305
+ scheduled_jobs: :scheduled_job,
306
+ state_machines: :state_machine,
307
+ factories: :factory,
308
+ events: :event,
309
+ # DatabaseViewExtractor keeps only the highest `_vNN` of each Scenic
310
+ # view, so its unit set is a function of the whole directory rather
311
+ # than of each file independently: dispatching db/views/foo_v01.sql to
312
+ # the per-file method would index a version a full extraction drops.
313
+ database_views: :database_view,
314
+ # Framework/gem sources are a function of the installed dependency set,
315
+ # so `Gemfile.lock` is their trigger path (#169). Before this the only
316
+ # incremental writer was `woods:extract_framework`, which hand-wrote
317
+ # JSON around the whole pipeline — no AtomicFile, no _index.json, no
318
+ # manifest counts, no lock, no generation bump — and its output then
319
+ # sat stale forever. The value here names the primary unit type, like
320
+ # every other entry; the extractor also owns :gem_source, and
321
+ # {EXTRACTOR_KEY_TO_TYPES} (which wholesale replacement consults) lists
322
+ # both. Participation is gated by `include_framework_sources` — see
323
+ # {#skip_by_configuration?}.
324
+ rails_source: :rails_source,
325
+ # A package root decides which package every other unit belongs to,
326
+ # and the undeclared-edge report reads the whole declared set, so any
327
+ # package.yml change re-runs the extractor wholesale (#280). Task 8
328
+ # re-annotates unit membership in the same run.
329
+ packages: :package
330
+ }.freeze
331
+
332
+ # Extractors whose output embeds the route table, and which therefore go
333
+ # stale when routes change even though none of their own files did.
334
+ #
335
+ # ControllerExtractor writes each action's routes into unit metadata and
336
+ # into the action chunks; everything that includes `RouteHelperResolver`
337
+ # resolves `_path`/`_url` references into navigation edges against the
338
+ # same table. The dependency graph can't express this — a route unit
339
+ # depends *on* its controller, so walking dependents from
340
+ # `config/routes.rb` never reaches it — so the relationship is declared
341
+ # here instead and these types are re-extracted wholesale whenever the
342
+ # route set is re-run (#164).
343
+ #
344
+ # @return [Array<Symbol>] extractor keys
345
+ ROUTE_CONSUMER_EXTRACTORS = %i[
346
+ controllers
347
+ mailers
348
+ components
349
+ view_components
350
+ view_templates
351
+ ].freeze
352
+
353
+ # Payload artifacts that live at the top of a payload directory rather
354
+ # than inside a per-type directory. Used when seeding a payload from a
355
+ # flat index — the output root also holds `generation.json`, `dumps/`,
356
+ # `tasks/`, `woods.sqlite3` and `payloads/` itself, none of which belong
357
+ # to a generation's payload.
358
+ PAYLOAD_FILES = %w[manifest.json dependency_graph.json graph_analysis.json SUMMARY.md].freeze
359
+
360
+ # Payload directories that are not per-type unit directories.
361
+ PAYLOAD_DIRS = %w[flows].freeze
362
+
206
363
  attr_reader :output_dir, :dependency_graph
207
364
 
208
365
  def initialize(output_dir: nil)
209
366
  @output_dir = Pathname.new(output_dir || Rails.root.join('tmp/woods'))
367
+ @payload_store = PayloadStore.new(@output_dir)
368
+ @payload_dir = nil
369
+ @payload_generation = nil
210
370
  @dependency_graph = DependencyGraph.new
211
371
  @results = {}
212
372
  @extractors = {}
373
+ @publication_error = nil
374
+ end
375
+
376
+ # Where this run reads and writes payload artifacts.
377
+ #
378
+ # A run publishes into an immutable per-generation directory, so that the
379
+ # single atomic write of `generation.json` commits the whole payload at
380
+ # once. Falls back to the output root when no payload directory could be
381
+ # opened — a flat index is non-atomic but perfectly readable, and failing
382
+ # the extraction over it would be worse.
383
+ #
384
+ # @return [Pathname]
385
+ def payload_dir
386
+ @payload_dir || @output_dir
213
387
  end
214
388
 
215
389
  # ══════════════════════════════════════════════════════════════════════
@@ -222,23 +396,43 @@ module Woods
222
396
  def extract_all
223
397
  setup_output_directory
224
398
  ModelNameCache.reset!
399
+ # @package_resolver alone is not enough: #package_resolver builds
400
+ # through #extractor_for, which memoizes into @incremental_extractors.
401
+ # Without this reset, a second full run on the same instance would
402
+ # resolve membership through the first run's PackageExtractor and its
403
+ # already-memoized (now stale) package_files/package_roots, missing a
404
+ # package added between the two runs.
405
+ @package_resolver = nil
406
+ @incremental_extractors = nil
407
+ @persisted_index_stats = nil
408
+ @graph_sha = nil
409
+ profile_phase('payload seed') { begin_payload! }
225
410
 
226
411
  # Eager load once — all extractors need loaded classes for introspection.
227
- safe_eager_load!
412
+ profile_phase('eager load') { safe_eager_load! }
228
413
 
229
414
  # Phase 1: Extract all units
230
- if Woods.configuration.concurrent_extraction
231
- extract_all_concurrent
232
- else
233
- extract_all_sequential
415
+ profile_phase('extraction') do
416
+ if Woods.configuration.concurrent_extraction
417
+ extract_all_concurrent
418
+ else
419
+ extract_all_sequential
420
+ end
234
421
  end
235
422
 
236
423
  # Phase 1.5: Deduplicate results
237
424
  Rails.logger.info '[Woods] Deduplicating results...'
238
425
  deduplicate_results
239
426
 
240
- # Rebuild graph from deduped results — Phase 1 registered all units including
241
- # duplicates, and DependencyGraph has no remove/unregister API.
427
+ # Phase 1.6: Package membership. Runs before the graph is rebuilt so
428
+ # registration copies metadata[:package] onto the node (#280).
429
+ annotate_packages
430
+
431
+ # Rebuild the graph from deduped results. #164 gave DependencyGraph
432
+ # `#remove`/`#unregister`, so surgical removal is now possible — but a
433
+ # full extraction has just registered every unit including duplicates,
434
+ # and rebuilding from the deduped set is both cheaper and less
435
+ # error-prone than unwinding registrations one at a time.
242
436
  @dependency_graph = DependencyGraph.new
243
437
  @results.each_value { |units| units.each { |u| @dependency_graph.register(u) } }
244
438
 
@@ -246,13 +440,15 @@ module Woods
246
440
  Rails.logger.info '[Woods] Resolving dependents...'
247
441
  resolve_dependents
248
442
 
249
- # Phase 3: Graph analysis (PageRank, structural metrics)
250
- Rails.logger.info '[Woods] Analyzing dependency graph...'
251
- @graph_analysis = GraphAnalyzer.new(@dependency_graph).analyze
252
-
253
- # Phase 4: Enrich with git data
443
+ # Phase 3: Enrich with git data. Runs BEFORE analysis now: the
444
+ # volatile_dependencies report reads commit counts off graph nodes.
254
445
  Rails.logger.info '[Woods] Enriching with git data...'
255
446
  enrich_with_git_data
447
+ annotate_graph_with_git_data
448
+
449
+ # Phase 4: Graph analysis (PageRank, structural metrics)
450
+ Rails.logger.info '[Woods] Analyzing dependency graph...'
451
+ @graph_analysis = profile_phase('graph analysis') { build_graph_analyzer.analyze }
256
452
 
257
453
  # Phase 4.5: Normalize file_path to relative paths
258
454
  Rails.logger.info '[Woods] Normalizing file paths...'
@@ -260,23 +456,43 @@ module Woods
260
456
 
261
457
  # Phase 5: Write output
262
458
  Rails.logger.info '[Woods] Writing output...'
263
- write_results
459
+ profile_phase('write results') { write_results }
460
+
461
+ # Phase 5.1: Sweep unit files no current unit accounts for (#177). Must
462
+ # run after write_results — the just-written set is what defines
463
+ # "legitimate" — and belongs to the full path only; the incremental path
464
+ # deletes through the graph instead. See {#sweep_orphaned_unit_files}.
465
+ sweep_orphaned_unit_files
264
466
 
265
467
  # Phase 5.5: Precompute request flows (opt-in). Must run AFTER
266
468
  # write_results — FlowAssembler loads unit JSON from disk, so running
267
469
  # earlier assembled every flow from absent (fresh output dir) or
268
470
  # stale (previous run's) data. precompute_flows re-writes the
269
471
  # controller units it annotates with metadata[:flow_paths].
472
+ #
473
+ # Fail closed (M3 review round 3): a failure anywhere in the family —
474
+ # assembly, index write, annotation rewrite, sweep — raises and aborts
475
+ # the run here, BEFORE the graph, manifest, snapshot, and generation
476
+ # publish below. The previously published generation stays resolved
477
+ # and readable; the aborted run's payload directory is left
478
+ # unreachable for the next run's PayloadStore#create to reclaim.
479
+ # Publishing a partial flow index, stale prior flow artifacts
480
+ # alongside a new graph, or half-rewritten annotations would violate
481
+ # the atomic-generation and full/incremental-equivalence contracts;
482
+ # being opt-in does not excuse it.
270
483
  if Woods.configuration.precompute_flows
271
484
  Rails.logger.info '[Woods] Precomputing request flows...'
272
- precompute_flows
485
+ profile_phase('flows') { precompute_flows }
273
486
  end
274
487
 
275
488
  write_dependency_graph
276
489
  write_graph_analysis
277
- write_manifest
278
- write_structural_summary
490
+ profile_phase('manifest and summary') do
491
+ write_manifest
492
+ write_structural_summary
493
+ end
279
494
  capture_snapshot
495
+ profile_phase('publish') { publish_generation('full') }
280
496
 
281
497
  log_summary
282
498
 
@@ -287,57 +503,600 @@ module Woods
287
503
  # Incremental Extraction
288
504
  # ══════════════════════════════════════════════════════════════════════
289
505
 
290
- # Extract only units affected by changed files
291
- # Used for incremental indexing in CI
506
+ # Extract only units affected by changed files.
507
+ #
508
+ # Used for incremental indexing in CI and by any caller that maintains a
509
+ # live index. The goal is equivalence: after this returns, the on-disk
510
+ # index should match what a cold full extraction of the same tree would
511
+ # have produced.
512
+ #
513
+ # The run proceeds in a fixed order, and the order matters:
514
+ #
515
+ # 1. Blast radius is computed against the *pre-change* graph, so
516
+ # dependents of a file that just disappeared still get re-extracted.
517
+ # 2. Known units in the blast radius are re-extracted.
518
+ # 3. Changed paths the index has never seen are dispatched to an
519
+ # extractor by path ({PathDispatcher}).
520
+ # 4. Class-based types are reconciled against their runtime discovery
521
+ # sets, catching classes added since the last extraction.
522
+ # 5. Whole-app extractors whose trigger paths changed are re-run and
523
+ # their type replaced wholesale.
524
+ # 6. Units whose source file has vanished are pruned last, so anything
525
+ # resurrected by steps 2–5 against a deleted file is swept in the
526
+ # same run rather than surviving as a ghost until the next one.
292
527
  #
293
528
  # @param changed_files [Array<String>] List of changed file paths
294
- # @return [Array<String>] List of re-extracted unit identifiers
529
+ # @return [Array<String>] Identifiers of units re-extracted, added, or removed
295
530
  def extract_changed(changed_files)
296
- # Load existing graph
297
- graph_path = @output_dir.join('dependency_graph.json')
298
- @dependency_graph = DependencyGraph.from_h(JSON.parse(File.read(graph_path))) if graph_path.exist?
531
+ prepare_incremental_run
299
532
 
300
- ModelNameCache.reset!
533
+ change_set = ChangeSet.new(paths: changed_files, root: Rails.root)
534
+ affected_types = Set.new
535
+
536
+ # Blast radius from the pre-change graph, bounded by
537
+ # `incremental_blast_radius_depth` (nil = the unbounded closure).
538
+ affected_ids = profile_phase('blast radius') do
539
+ @dependency_graph.affected_by(change_set.absolute_paths, max_depth: blast_radius_depth)
540
+ end
541
+ @flow_scope = profile_phase('flow radius') { flow_scope_for(change_set) }
542
+ Rails.logger.info "[Woods] #{change_set.size} changed files affect #{affected_ids.size} units"
543
+
544
+ touched = profile_phase('re-extraction') do
545
+ acc = reconcile_changed_paths(change_set, affected_types)
546
+
547
+ (affected_ids - acc.to_a).each do |unit_id|
548
+ acc.add(unit_id) if re_extract_unit(unit_id, affected_types: affected_types)
549
+ end
550
+
551
+ acc
552
+ end
301
553
 
302
- # Eager load to ensure newly-added classes are discoverable.
303
- safe_eager_load!
554
+ touched.merge(reconcile_class_based_types(affected_types))
555
+ touched.merge(rerun_whole_app_extractors(change_set, affected_types))
556
+ touched.merge(reannotate_packages(change_set, affected_types))
557
+ pruned = prune_vanished_units(change_set, affected_types)
558
+ touched.merge(pruned)
559
+
560
+ # Reconcile once more, because pruning can un-know a class the first pass
561
+ # skipped. A class-based file moved between autoload directories with its
562
+ # constant unchanged is still registered under the old path when
563
+ # reconciliation runs, so it looks known and is not re-extracted; the
564
+ # prune that follows then removes it for its vanished path. This pass
565
+ # re-adds it in the same run (M1) instead of leaving the unit missing
566
+ # until some later run happens to notice. Idempotent when nothing was
567
+ # pruned: the discovery set is compared against the graph, so an
568
+ # already-registered class is skipped.
569
+ #
570
+ # But not everything pruning removed may come back. `except:` keeps the
571
+ # *deletion* shape pruned: without a reload, a constant outlives the file
572
+ # that defined it — so deleting `app/models/user.rb` prunes `User`, and
573
+ # this pass finds `User` still in `ActiveRecord::Base.descendants`.
574
+ # Re-registering it would pin the unit to a path that no longer exists,
575
+ # and nothing could ever remove it: the sweep excludes class-based units
576
+ # and no future change set names that path again. A resident daemon
577
+ # processing a batch before its reload hits this every time. What
578
+ # separates the two shapes is the filesystem — only pruned identifiers
579
+ # that a still-existing file in the change set actually declares are
580
+ # re-addable. See {#readdable_pruned_classes}.
581
+ touched.merge(reconcile_class_based_types(
582
+ affected_types, except: pruned - readdable_pruned_classes(pruned, change_set)
583
+ ))
584
+
585
+ finalize_incremental_unit_json(affected_types)
304
586
 
305
- # Normalize relative paths (from git diff) to absolute (as stored in file_map)
306
- absolute_files = changed_files.map do |f|
307
- Pathname.new(f).absolute? ? f : Rails.root.join(f).to_s
587
+ # Regenerate type indexes for affected types
588
+ profile_phase('type index') do
589
+ affected_types.each do |type_key|
590
+ regenerate_type_index(type_key)
591
+ end
308
592
  end
309
593
 
310
- # Compute affected units
311
- affected_ids = @dependency_graph.affected_by(absolute_files)
312
- Rails.logger.info "[Woods] #{changed_files.size} changed files affect #{affected_ids.size} units"
594
+ finalize_incremental_run(touched)
595
+
596
+ touched.to_a
597
+ end
598
+
599
+ # ══════════════════════════════════════════════════════════════════════
600
+ # Targeted Refresh
601
+ # ══════════════════════════════════════════════════════════════════════
313
602
 
314
- # Re-extract affected units
603
+ # Re-run one or more extractors wholesale against the already-booted app,
604
+ # replacing every unit of the types they own.
605
+ #
606
+ # This is the escape hatch for the unit types that have no per-file entry
607
+ # point — routes are derived from `Rails.application.routes`, the
608
+ # middleware unit from the live stack, events from a two-pass scan of all
609
+ # of `app/`. Incremental runs reach them by trigger path
610
+ # ({PathDispatcher.whole_app_rules}); this reaches them by name, for a
611
+ # caller that already knows what went stale: a resident process that just
612
+ # reloaded, a `woods:refresh` invocation after editing `config/routes.rb`,
613
+ # a deploy hook after a dependency bump.
614
+ #
615
+ # "Full extraction required for these types" was always a cold-boot
616
+ # artifact rather than something inherent: in a booted process re-running
617
+ # one extractor is seconds.
618
+ #
619
+ # Any extractor key works, not just the whole-app ones — `refresh(:models)`
620
+ # is a legitimate way to re-derive every model after a schema change.
621
+ #
622
+ # @example After editing config/routes.rb
623
+ # Woods::Extractor.new(output_dir: "tmp/woods").refresh(:routes)
624
+ #
625
+ # @param keys [Array<Symbol>] keys into {EXTRACTORS}
626
+ # @return [Hash] `{ types:, touched:, unknown: }` — the extractors that
627
+ # ran, the identifiers written or removed, and any key that isn't an
628
+ # extractor
629
+ # @raise [ArgumentError] when no recognized key is given
630
+ def refresh(*keys)
631
+ keys = Array(keys).flatten.map(&:to_sym).uniq
632
+ known, unknown = keys.partition { |key| EXTRACTORS.key?(key) }
633
+ raise ArgumentError, "No known extractor in #{keys.inspect}" if known.empty?
634
+
635
+ known += ROUTE_CONSUMER_EXTRACTORS if known.include?(:routes)
636
+ known.uniq!
637
+
638
+ prepare_incremental_run
315
639
  affected_types = Set.new
316
- affected_ids.each do |unit_id|
317
- re_extract_unit(unit_id, affected_types: affected_types)
640
+ touched = known.each_with_object(Set.new) do |key, acc|
641
+ acc.merge(replace_type_wholesale(key, affected_types))
318
642
  end
319
643
 
320
- # Regenerate type indexes for affected types
321
- affected_types.each do |type_key|
322
- regenerate_type_index(type_key)
644
+ finalize_incremental_unit_json(affected_types)
645
+ affected_types.each { |type_key| regenerate_type_index(type_key) }
646
+ finalize_incremental_run(touched, reason: "refresh:#{known.sort.join(',')}")
647
+
648
+ { types: known, touched: touched.to_a, unknown: unknown }
649
+ end
650
+
651
+ # Raise when the most recent extraction run wrote a payload but could not
652
+ # publish its generation marker.
653
+ #
654
+ # The extractor records this failure instead of raising immediately so
655
+ # the resident watch daemon can keep its recoverable posture: it detects
656
+ # the unchanged generation, reports degraded, and carries the paths into
657
+ # a later cycle. One-shot callers have no later cycle, so the rake tasks
658
+ # call this method before reporting success and receive a typed non-zero
659
+ # failure instead of claiming an unreachable payload was published.
660
+ #
661
+ # @return [void]
662
+ # @raise [Woods::ExtractionError] when the generation marker could not be
663
+ # published; the previously published generation remains active
664
+ def raise_on_publication_failure!
665
+ raise @publication_error if @publication_error
666
+ end
667
+
668
+ private
669
+
670
+ # Time one phase of a run and log how long it took, when WOODS_PROFILE=1.
671
+ #
672
+ # The per-extractor lines (see {#extract_all_sequential}) already report
673
+ # extraction itself. Everything after it (the graph load, the analysis,
674
+ # the flows, the publish) was unattributed, so a slow run could only be
675
+ # split by guessing. Off by default and free when off: the block is
676
+ # yielded directly, with no timing and no log line. Timed on the
677
+ # monotonic clock, so a wall-clock adjustment mid-run cannot produce a
678
+ # negative phase.
679
+ #
680
+ # @param name [String] phase name, as it appears in the log line
681
+ # @return [Object] whatever the block returned
682
+ def profile_phase(name)
683
+ return yield unless profiling?
684
+
685
+ start_time = Process.clock_gettime(Process::CLOCK_MONOTONIC)
686
+ result = yield
687
+ elapsed = Process.clock_gettime(Process::CLOCK_MONOTONIC) - start_time
688
+ Rails.logger.info "[Woods] [profile] #{name} in #{elapsed.round(2)}s"
689
+ result
690
+ end
691
+
692
+ # How many reverse hops {#extract_changed} walks from the changed files
693
+ # before it stops re-extracting dependents.
694
+ #
695
+ # Unbounded by default. A unit outside the cap keeps its content, and the
696
+ # one derived field a re-extraction elsewhere can change (`dependents`)
697
+ # is refreshed regardless: {#register_and_write} marks every target of a
698
+ # re-extracted unit's edges, before and after registration, so a unit
699
+ # that gains or loses an inbound edge is rewritten by
700
+ # {#finalize_incremental_unit_json} whether or not the walk reached it.
701
+ #
702
+ # @return [Integer, nil]
703
+ def blast_radius_depth
704
+ Woods.configuration&.incremental_blast_radius_depth
705
+ end
706
+
707
+ # The units a controller's flow document could reach from this run's
708
+ # changed files, read off the pre-change graph.
709
+ #
710
+ # {FlowAssembler} stops expanding at {FlowPrecomputer::DEFAULT_MAX_DEPTH},
711
+ # so a controller further than that from everything the run changed
712
+ # assembles the same document it already has. The bound is the
713
+ # assembler's own constant, not a second literal: raising the assembly
714
+ # depth widens this walk with it.
715
+ #
716
+ # Reverse reachability is the same relation the refresh already rested on
717
+ # (`touched` is itself the graph's reverse closure), so this narrows the
718
+ # distance without changing the edge set behind the decision.
719
+ #
720
+ # @param change_set [Woods::ChangeSet]
721
+ # @return [Set<String>]
722
+ def flow_scope_for(change_set)
723
+ return Set.new unless Woods.configuration.precompute_flows
724
+
725
+ @dependency_graph.affected_by(
726
+ change_set.absolute_paths, max_depth: FlowPrecomputer::DEFAULT_MAX_DEPTH
727
+ ).to_set
728
+ end
729
+
730
+ # @return [Boolean] whether phase timing is enabled for this process
731
+ def profiling?
732
+ ENV.fetch('WOODS_PROFILE', nil) == '1'
733
+ end
734
+
735
+ # Load the persisted graph and reset the per-run bookkeeping that the
736
+ # incremental helpers read. Shared by {#extract_changed} and {#refresh};
737
+ # calling either without this leaves `@dependents_dirty` and
738
+ # `@incremental_written` holding a previous run's state.
739
+ #
740
+ # `begin_payload!(strict: true)`: an incremental write set is only the
741
+ # touched units, so a degrade to flat here (see {#begin_payload!}) would
742
+ # both read the wrong baseline graph below and publish an index missing
743
+ # every unit it didn't touch. Raises rather than degrading; the caller
744
+ # sees {Woods::ExtractionError} and the generation is left unbumped.
745
+ #
746
+ # @return [void]
747
+ # @raise [Woods::ExtractionError] see {#begin_payload!}
748
+ def prepare_incremental_run
749
+ profile_phase('payload seed') { begin_payload!(strict: true) }
750
+ graph_path = payload_dir.join('dependency_graph.json')
751
+ ensure_incremental_baseline!(graph_path)
752
+ profile_phase('previous graph load') do
753
+ @dependency_graph = DependencyGraph.from_h(JSON.parse(AtomicFile.read(graph_path))) if graph_path.exist?
323
754
  end
324
755
 
325
- # Update graph, manifest, and summary. No capture_snapshot here:
326
- # snapshots must hash the FULL unit set, and incremental runs only
327
- # re-extract affected units (@results stays empty) — capturing would
328
- # record a snapshot whose diff reports every unit as deleted.
329
- # Snapshots are captured on full extraction only.
756
+ ModelNameCache.reset!
757
+ profile_phase('eager load') { safe_eager_load! }
758
+
759
+ @dependents_dirty = Set.new
760
+ @incremental_written = {}
761
+ # nil = no scope was computed for this run, so every re-extracted
762
+ # controller has its flows reassembled. {#refresh} leaves it that way.
763
+ @flow_scope = nil
764
+ @previous_flow_index_entries = nil
765
+ @incremental_extractors = nil
766
+ @active_record_names = nil
767
+ @package_resolver = nil
768
+ @persisted_index_stats = nil
769
+ @graph_sha = nil
770
+ end
771
+
772
+ # Write the graph and the derived artifacts after an incremental run.
773
+ #
774
+ # The manifest is rewritten only when the run actually changed something.
775
+ # `staleness_seconds` is derived from the manifest timestamp, so touching
776
+ # it after a no-op run reports the index as freshly synced when nothing
777
+ # was re-read — the misleading half of #164 gap 4. The graph write itself
778
+ # is unconditional, but on a no-op run it is inert, not meaningful: under
779
+ # payloads it lands in this run's not-yet-published directory, and a
780
+ # no-op run returns below before {#publish_generation}, so nothing ever
781
+ # points at what it just wrote — the next run's {PayloadStore#create}
782
+ # empties that same (unbumped-generation-numbered) directory before
783
+ # writing into it again. The graph's content is unchanged on a no-op run
784
+ # regardless (no touched units, no new edges), so nothing is lost; it is
785
+ # the write itself, not the recomputation, that goes nowhere.
786
+ #
787
+ # No `capture_snapshot` here: snapshots must hash the FULL unit set, and
788
+ # incremental runs only hold changed units in memory — capturing would
789
+ # record a snapshot whose diff reports every other unit as deleted.
790
+ #
791
+ # @param touched [Set<String>] identifiers added, re-extracted, or removed
792
+ # @param reason [String] what produced this run, recorded on the generation
793
+ # so `woods_status` can distinguish a targeted refresh from a file-driven
794
+ # incremental run — they have different blast radii and an operator
795
+ # reading "incremental" after a `woods:refresh[routes]` is being misled
796
+ # @return [void]
797
+ def finalize_incremental_run(touched, reason: 'incremental')
330
798
  write_dependency_graph
331
- write_manifest(incremental: true)
332
- write_structural_summary
333
- if Woods.configuration.enable_snapshots
334
- Rails.logger.info '[Woods] Skipping snapshot capture — snapshots are captured on full extraction only'
799
+
800
+ if touched.empty?
801
+ Rails.logger.info '[Woods] Incremental run changed nothing — leaving manifest timestamp untouched'
802
+ return
335
803
  end
336
804
 
337
- affected_ids
805
+ profile_phase('graph analysis') { write_incremental_graph_analysis }
806
+ profile_phase('flows') { refresh_incremental_flows(touched) }
807
+ profile_phase('manifest and summary') do
808
+ write_manifest(incremental: true)
809
+ write_structural_summary
810
+ end
811
+ profile_phase('publish') { publish_generation(reason) }
812
+
813
+ return unless Woods.configuration.enable_snapshots
814
+
815
+ Rails.logger.info '[Woods] Skipping snapshot capture — snapshots are captured on full extraction only'
338
816
  end
339
817
 
340
- private
818
+ # Publish a new generation — the last write of any successful run.
819
+ #
820
+ # Every extraction mode does this, so a long-lived reader can detect that
821
+ # the index moved without stat-ing the whole directory, whatever produced
822
+ # the change: a full run, an incremental run, a targeted refresh, or the
823
+ # watch daemon. Ordering is the contract: the generation goes last, so a
824
+ # reader that sees generation N knows N's files are already on disk. A run
825
+ # that raised, or that changed nothing, never reaches this.
826
+ #
827
+ # @param reason [String] what produced this generation
828
+ # @return [void]
829
+ def publish_generation(reason)
830
+ generation = Generation.new(output_dir: @output_dir)
831
+ # Resolve (and if necessary rename) the payload first, so the flush
832
+ # below covers the directory under the name the pointer will carry.
833
+ payload = publishable_payload_name(generation)
834
+ profile_phase('payload sync') { sync_payload }
835
+ marker = generation.bump!(reason: reason, payload: payload)
836
+ prune_payloads(marker.number)
837
+ marker
838
+ rescue StandardError => e
839
+ # A failed bump must not fail the extraction that produced a perfectly
840
+ # good index. But "readers keep their current view until the next run" is
841
+ # too comfortable a way to put it: the generation *is* the freshness
842
+ # contract, so readers keep serving the old index for as long as the
843
+ # cause persists, and the next incremental may be a no-op that bumps
844
+ # nothing either. Error, not warn — and `Watch::Daemon` cross-checks that
845
+ # the number actually moved so the daemon reports degraded rather than
846
+ # running.
847
+ @publication_error = Woods::ExtractionError.new(
848
+ "Could not publish generation for #{@output_dir} (#{e.class}: #{e.message}); " \
849
+ 'the previous generation remains active'
850
+ )
851
+ Rails.logger.error "[Woods] #{@publication_error.message}"
852
+ nil
853
+ end
854
+
855
+ # Open the payload directory this run publishes into, seeded from the
856
+ # generation currently on disk.
857
+ #
858
+ # Seeding is what lets a run that touches ten files publish a whole
859
+ # generation: the unchanged artifacts are hardlinked in, and the run
860
+ # overwrites only what it changed. It also means every payload read during
861
+ # the run — the persisted graph, the per-type indexes, the unit JSON an
862
+ # incremental run patches — sees the previous generation, exactly as it
863
+ # did when the index was flat.
864
+ #
865
+ # A failure here degrades to the flat layout rather than failing the run
866
+ # — but only when a flat publish can actually be complete. A full
867
+ # extraction's write set is every unit in the app, so a flat publish from
868
+ # it is a whole index; the next successful run restores the payload
869
+ # boundary. `strict:` opts out of that degrade for {#extract_changed} and
870
+ # {#refresh}, whose write set is only the units they touched: over a
871
+ # payload-born index (`marker.payload` set), the flat root holds nothing
872
+ # newer than the last time this index was flat — possibly nothing at all
873
+ # — so publishing there would both compute the wrong incremental baseline
874
+ # ({#prepare_incremental_run} reads its graph from wherever `payload_dir`
875
+ # resolves to) and, even with the right baseline, redirect every reader
876
+ # to a near-empty directory missing every untouched unit. There is no
877
+ # complete flat index an incremental degrade can produce, so this raises
878
+ # instead — see {Woods::ExtractionError}. The generation is never bumped
879
+ # over a raised run, so readers keep serving the last good index.
880
+ #
881
+ # @param strict [Boolean] raise instead of degrading to a flat publish
882
+ # when the published generation names a payload
883
+ # @return [void]
884
+ # @raise [Woods::ExtractionError] when `strict` and a payload-born index's
885
+ # payload directory could not be opened for this run
886
+ def begin_payload!(strict: false)
887
+ @publication_error = nil
888
+ marker = Generation.new(output_dir: @output_dir).current
889
+ @payload_generation = marker.number + 1
890
+ @payload_dir = @payload_store.create(@payload_generation)
891
+ seed_payload(marker)
892
+ rescue StandardError => e
893
+ if strict && marker&.payload
894
+ raise Woods::ExtractionError, <<~MSG.tr("\n", ' ').strip
895
+ Could not open a payload directory for this incremental run
896
+ (#{e.class}: #{e.message}), and generation #{marker.number}'s payload
897
+ lives in #{marker.payload} rather than the flat output root. An
898
+ incremental run only writes the units it touched, so publishing flat
899
+ here would produce an index missing every untouched unit. Fix the
900
+ underlying filesystem issue (permissions, free space, a broken
901
+ mount), or run a full `woods:extract` to rebuild a complete flat
902
+ index before incremental runs resume.
903
+ MSG
904
+ end
905
+
906
+ Rails.logger.warn(
907
+ "[Woods] Could not open a payload directory (#{e.class}: #{e.message}) — publishing flat"
908
+ )
909
+ @payload_dir = nil
910
+ @payload_generation = nil
911
+ end
912
+
913
+ # Refuse an incremental run that has no baseline to be incremental
914
+ # against (CORE-2). With no published generation and no dependency
915
+ # graph — a failed CI cache restore, a typo'd WOODS_OUTPUT, a first run
916
+ # on a fresh runner — {#extract_changed} would compute an empty blast
917
+ # radius, dispatch only the diffed paths, and publish generation 1: a
918
+ # near-empty index that readers, `woods:validate`, and retrieval all
919
+ # treat as the complete truth of the app, and that nothing self-heals
920
+ # until a full extraction. The watch daemon already enforces this
921
+ # invariant on its side (a missing generation marker means one full
922
+ # extraction); the one-shot entry points must refuse rather than
923
+ # publish silently. A pre-generation flat index passes: its graph was
924
+ # seeded into this run's payload directory by {#begin_payload!}.
925
+ #
926
+ # @param graph_path [Pathname] the seeded payload's dependency graph
927
+ # @return [void]
928
+ # @raise [Woods::ExtractionError] when no baseline exists
929
+ def ensure_incremental_baseline!(graph_path)
930
+ return if graph_path.exist?
931
+ return if Generation.new(output_dir: @output_dir).current.number.positive?
932
+ # An embedding caller that already holds a populated graph (a prior
933
+ # in-process run, or seeded registrations) IS the baseline —
934
+ # {#prepare_incremental_run} keeps the in-memory graph whenever no
935
+ # disk graph exists.
936
+ return unless @dependency_graph.nil? || @dependency_graph.empty?
937
+
938
+ raise Woods::ExtractionError, <<~MSG.tr("\n", ' ').strip
939
+ No baseline index found under #{@output_dir}: no generation has been
940
+ published and no dependency_graph.json exists. An incremental run
941
+ only re-extracts what changed relative to an existing index, so
942
+ running it here would publish a near-empty index as the complete
943
+ truth of the application. Run a full `woods:extract` first (or point
944
+ WOODS_OUTPUT at the directory holding the existing index).
945
+ MSG
946
+ end
947
+
948
+ # Copy the published generation's payload into this run's directory.
949
+ #
950
+ # Two sources. A generation that already names a payload directory is
951
+ # cloned wholesale — by construction it holds payload artifacts and
952
+ # nothing else. A flat index (every index written before payloads, and any
953
+ # index whose last run degraded) is cloned entry by entry from an
954
+ # allowlist, because the output root also holds `generation.json`,
955
+ # `dumps/`, `tasks/`, `woods.sqlite3` and `payloads/` itself.
956
+ #
957
+ # @param marker [Woods::Generation::Marker] the published generation
958
+ # @return [void]
959
+ def seed_payload(marker)
960
+ if marker.payload && (source = @output_dir.join(marker.payload)).directory?
961
+ @payload_store.clone(source, @payload_dir)
962
+ else
963
+ seed_payload_from_flat_root
964
+ end
965
+ end
966
+
967
+ # Uses {PayloadStore#link_or_copy} rather than a bare +FileUtils.ln+ for
968
+ # the same reason {PayloadStore#clone} does: a filesystem that disallows
969
+ # hardlinks (EXDEV/EPERM/EMLINK/NotImplementedError) must still seed the
970
+ # payload, just by copying. Before this, that raise was caught by
971
+ # {#begin_payload!}'s rescue and degraded every run on such a filesystem
972
+ # to a flat publish — the fallback existed in {PayloadStore} but this,
973
+ # the only other file-level linker, never reached it.
974
+ #
975
+ # @return [void]
976
+ def seed_payload_from_flat_root
977
+ (PAYLOAD_FILES + payload_entry_dirs).each do |entry|
978
+ source = @output_dir.join(entry)
979
+ next unless source.exist?
980
+
981
+ @payload_store.clone(source, @payload_dir.join(entry)) if source.directory?
982
+ @payload_store.link_or_copy(source, @payload_dir.join(entry)) if source.file?
983
+ end
984
+ end
985
+
986
+ # @return [Array<String>] every directory name a payload can contain
987
+ def payload_entry_dirs
988
+ EXTRACTORS.keys.map(&:to_s) + PAYLOAD_DIRS
989
+ end
990
+
991
+ # Whether payload files pay for their own durability as they are written.
992
+ #
993
+ # False by default and false on every host that has not asked otherwise:
994
+ # {#sync_payload} makes the whole payload durable in one flush before the
995
+ # pointer names it, so a per-file fsync buys only the window in which
996
+ # nothing can read the file anyway. `durable_payload_writes = true`
997
+ # restores the per-file cost for a host that wants it; it does not, and
998
+ # cannot, remove the publish flush.
999
+ #
1000
+ # @return [Boolean]
1001
+ def payload_writes_durable?
1002
+ Woods.configuration&.durable_payload_writes ? true : false
1003
+ end
1004
+
1005
+ # One filesystem flush that makes this run's whole payload durable,
1006
+ # immediately before the pointer that names it is written durably.
1007
+ #
1008
+ # This is the guarantee that replaces the per-file fsync {#write_results}
1009
+ # used to pay for every unit: **when `generation.json` is durable, every
1010
+ # file in the payload it names is durable.** What is given up is an
1011
+ # individual payload file being durable before the pointer exists, and
1012
+ # nobody reads a payload file in that window: every reader resolves
1013
+ # through the pointer, and a crash leaves an unreferenced partial payload
1014
+ # that the next run prunes.
1015
+ #
1016
+ # Unconditional on purpose. `durable_payload_writes` decides whether the
1017
+ # per-file fsyncs are *also* paid; it can never turn this off, because
1018
+ # that would be the one change that actually weakens the contract.
1019
+ # {Woods::AtomicFile.sync_directory_tree} degrades through `sync -f`, a
1020
+ # bare `sync`, and finally a per-file fsync pass, so the flush cannot
1021
+ # silently become a no-op either.
1022
+ #
1023
+ # A run that degraded to a flat publish has no payload directory;
1024
+ # {#payload_dir} then returns the output directory, which is the tree that
1025
+ # needs flushing in exactly the same way.
1026
+ #
1027
+ # @return [Symbol, nil] the strategy {Woods::AtomicFile} used
1028
+ def sync_payload
1029
+ AtomicFile.sync_directory_tree(payload_dir)
1030
+ end
1031
+
1032
+ # The pointer to publish, or nil when this run built no payload directory.
1033
+ #
1034
+ # The directory was named from the generation number this run expected to
1035
+ # publish as. Writers serialize on `PipelineLock`, so that prediction holds
1036
+ # — but if it ever did not, the pointer would name a directory belonging to
1037
+ # a different generation, so the directory is renamed to match rather than
1038
+ # published under a name that lies.
1039
+ #
1040
+ # @param generation [Woods::Generation]
1041
+ # @return [String, nil]
1042
+ def publishable_payload_name(generation)
1043
+ return nil unless @payload_dir
1044
+
1045
+ actual = generation.current.number + 1
1046
+ if actual != @payload_generation
1047
+ @payload_dir = rename_payload(actual)
1048
+ @payload_generation = actual
1049
+ end
1050
+
1051
+ PayloadStore.name_for(@payload_generation)
1052
+ end
1053
+
1054
+ # @param number [Integer] the generation number actually being published
1055
+ # @return [Pathname] the payload directory under its corrected name
1056
+ def rename_payload(number)
1057
+ destination = @payload_store.path_for(number)
1058
+ FileUtils.rm_rf(destination.to_s)
1059
+ FileUtils.mv(@payload_dir.to_s, destination.to_s)
1060
+ destination
1061
+ end
1062
+
1063
+ # @param published [Integer] the generation just published
1064
+ # @return [void]
1065
+ def prune_payloads(published)
1066
+ return unless @payload_dir
1067
+
1068
+ @payload_store.prune(keep: payload_retention, protect: published)
1069
+ rescue StandardError => e
1070
+ Rails.logger.warn "[Woods] Payload retention failed: #{e.message}"
1071
+ end
1072
+
1073
+ # @return [Integer] how many payload generations to keep on disk
1074
+ def payload_retention
1075
+ value = ENV.fetch('WOODS_PAYLOAD_RETENTION', nil).to_i
1076
+ value.positive? ? value : PayloadStore::DEFAULT_RETENTION
1077
+ end
1078
+
1079
+ # Recompute graph_analysis.json after an incremental graph write.
1080
+ #
1081
+ # Structural analysis (orphans, hubs, cycles, bridges) is derived purely
1082
+ # from the graph, so leaving it at the last full extraction's values let
1083
+ # it drift continuously between full runs (#164 gap 5).
1084
+ #
1085
+ # Re-raises rather than warn-and-continue: a caller that swallowed this
1086
+ # went on to write the manifest and publish a fresh generation over an
1087
+ # index whose graph_analysis.json never updated — a failed analysis write
1088
+ # must abort the run the same way a failed extraction does, not ship
1089
+ # under a generation that says everything landed.
1090
+ #
1091
+ # @return [void]
1092
+ # @raise [StandardError] whatever GraphAnalyzer or the write raised
1093
+ def write_incremental_graph_analysis
1094
+ @graph_analysis = build_graph_analyzer.analyze
1095
+ write_graph_analysis
1096
+ rescue StandardError => e
1097
+ Rails.logger.error "[Woods] Incremental graph analysis failed: #{e.message}"
1098
+ raise
1099
+ end
341
1100
 
342
1101
  # ──────────────────────────────────────────────────────────────────────
343
1102
  # Eager Loading
@@ -351,9 +1110,16 @@ module Woods
351
1110
  # loads only the directories we actually need for extraction.
352
1111
  def safe_eager_load!
353
1112
  Rails.application.eager_load!
1113
+ # Recorded because it decides whether the runtime discovery sets are
1114
+ # *complete*. On the fallback path they are known-partial — whole
1115
+ # directories may have failed to load — and treating "absent from
1116
+ # descendants" as "deleted" would then erase live units by the type.
1117
+ # See {#stale_class_based_units}.
1118
+ @eager_load_complete = true
354
1119
  rescue NameError => e
355
1120
  Rails.logger.warn "[Woods] eager_load! hit NameError: #{e.message}"
356
1121
  Rails.logger.warn '[Woods] Falling back to per-directory eager loading'
1122
+ @eager_load_complete = false
357
1123
  eager_load_extraction_directories
358
1124
  end
359
1125
 
@@ -387,8 +1153,36 @@ module Woods
387
1153
  # Extraction Strategies
388
1154
  # ──────────────────────────────────────────────────────────────────────
389
1155
 
1156
+ # Should this extractor be skipped under the current configuration?
1157
+ #
1158
+ # `include_framework_sources` (default: true — see
1159
+ # docs/CONFIGURATION_REFERENCE.md) was documented and consumed by nothing
1160
+ # (#169). It now gates both automatic paths at once: the full-run pass
1161
+ # over {EXTRACTORS} and the `Gemfile.lock` whole-app trigger
1162
+ # ({#rerun_whole_app_extractors}). The {PathDispatcher} rule itself stays
1163
+ # ungated — rules are memoized per-process while configuration is not,
1164
+ # and `relevant?` stays correct either way because `Gemfile.lock` already
1165
+ # triggers :engines and :middleware — so the gate sits where the
1166
+ # extractor would actually run. Gating only one side would either extract
1167
+ # units a knob-off full run does not produce (a phantom `touched` cycle
1168
+ # bumping the generation) or leave a knob-on host with framework sources
1169
+ # that never refresh on a dependency bump.
1170
+ #
1171
+ # A by-name {#refresh} (and its `woods:extract_framework` alias) is
1172
+ # deliberately NOT gated: the knob controls automatic participation, and
1173
+ # the explicit refresh is the escape hatch for hosts that want framework
1174
+ # sources on demand without paying for them every run.
1175
+ #
1176
+ # @param key [Symbol] key into {EXTRACTORS}
1177
+ # @return [Boolean]
1178
+ def skip_by_configuration?(key)
1179
+ key == :rails_source && !Woods.configuration.include_framework_sources
1180
+ end
1181
+
390
1182
  def extract_all_sequential
391
1183
  EXTRACTORS.each do |type, extractor_class|
1184
+ next if skip_by_configuration?(type)
1185
+
392
1186
  Rails.logger.info "[Woods] Extracting #{type}..."
393
1187
  start_time = Time.current
394
1188
 
@@ -427,7 +1221,9 @@ module Woods
427
1221
  ModelNameCache.short_names_regex if ModelNameCache.respond_to?(:short_names_regex)
428
1222
 
429
1223
  results_mutex = Mutex.new
430
- threads = EXTRACTORS.map do |type, extractor_class|
1224
+ failures = []
1225
+ active = EXTRACTORS.reject { |type, _| skip_by_configuration?(type) }
1226
+ threads = active.map do |type, extractor_class|
431
1227
  Thread.new do
432
1228
  Rails.logger.info "[Woods] [Thread] Extracting #{type}..."
433
1229
  start_time = Time.current
@@ -445,12 +1241,21 @@ module Woods
445
1241
  end
446
1242
  rescue StandardError => e
447
1243
  Rails.logger.error "[Woods] [Thread] #{type} failed: #{e.message}"
448
- results_mutex.synchronize { @results[type] = [] }
1244
+ results_mutex.synchronize { failures << [type, e] }
449
1245
  end
450
1246
  end
451
1247
 
452
1248
  threads.each(&:join)
453
1249
 
1250
+ # Fail closed, matching sequential extraction: a raise there aborts
1251
+ # before registering anything for the run. Silently substituting `[]`
1252
+ # for a failed type let the run finish, register a partial graph, and
1253
+ # publish it under a generation that says every type succeeded.
1254
+ if failures.any?
1255
+ names = failures.map { |(type, _e)| type }.sort.join(', ')
1256
+ raise Woods::ExtractionError, "Concurrent extraction failed for: #{names}"
1257
+ end
1258
+
454
1259
  # Register into dependency graph sequentially — DependencyGraph is not thread-safe
455
1260
  EXTRACTORS.each_key do |type|
456
1261
  (@results[type] || []).each { |unit| @dependency_graph.register(unit) }
@@ -464,7 +1269,7 @@ module Woods
464
1269
  def setup_output_directory
465
1270
  FileUtils.mkdir_p(@output_dir)
466
1271
  EXTRACTORS.each_key do |type|
467
- FileUtils.mkdir_p(@output_dir.join(type.to_s))
1272
+ FileUtils.mkdir_p(payload_dir.join(type.to_s))
468
1273
  end
469
1274
  end
470
1275
 
@@ -472,56 +1277,122 @@ module Woods
472
1277
  # Dependency Resolution
473
1278
  # ──────────────────────────────────────────────────────────────────────
474
1279
 
1280
+ # `unit_map` is identifier => Array<unit>, not identifier => unit (#225).
1281
+ # An identifier is not unique across types (a Scenic view and a factory
1282
+ # can both be `reports`), so a single-valued map let the later
1283
+ # registration overwrite the earlier one and every dependent land on
1284
+ # whichever unit happened to be indexed last — the other unit serialized
1285
+ # `dependents: []` forever. {#rewrite_unit_json_of_type} already treats
1286
+ # `dependents` as a property of the identifier and writes the same list
1287
+ # to every type that identifier owns; this brings the full-extraction
1288
+ # path into agreement with it.
475
1289
  def resolve_dependents
476
1290
  # Build complete unit map first (cross-type dependencies require all units indexed).
477
- unit_map = @results.each_with_object({}) do |(_type, units), map|
478
- units.each { |u| map[u.identifier] = u }
1291
+ unit_map = @results.each_with_object(Hash.new { |h, k| h[k] = [] }) do |(_type, units), map|
1292
+ units.each { |u| map[u.identifier] << u }
479
1293
  end
480
1294
 
481
1295
  # Resolve dependents using the complete map.
482
1296
  @results.each_value do |units|
483
1297
  units.each do |unit|
484
1298
  unit.dependencies.each do |dep|
485
- target_unit = unit_map[dep[:target]]
486
- next unless target_unit
487
-
488
- target_unit.dependents ||= []
489
- target_unit.dependents << {
490
- type: unit.type,
491
- identifier: unit.identifier
492
- }
1299
+ unit_map[dep[:target]].each do |target_unit|
1300
+ target_unit.dependents ||= []
1301
+ target_unit.dependents << {
1302
+ type: unit.type,
1303
+ identifier: unit.identifier
1304
+ }
1305
+ end
493
1306
  end
494
1307
  end
495
1308
  end
496
1309
  end
497
1310
 
498
1311
  # Remove duplicate units (same identifier) within each type, keeping the first occurrence.
499
- # Duplicates arise when multiple extractors produce the same unit (e.g., engine-mounted
500
- # routes duplicating app routes). Without dedup, downstream phases would produce inflated
501
- # counts, duplicate _index.json entries, and last-writer-wins file overwrites.
1312
+ # Duplicates arise when the same identifier is legitimately re-derived from the same
1313
+ # source (e.g., engine-mounted routes duplicating app routes: both carry no file path).
1314
+ # Without dedup, downstream phases would produce inflated counts, duplicate
1315
+ # _index.json entries, and last-writer-wins file overwrites.
1316
+ #
1317
+ # A duplicate derived from a DIFFERENT file is not a duplicate — it is two distinct
1318
+ # real-world constants collapsing onto one identifier (finding G-1: wrapper-named
1319
+ # sources all indexed as the wrapper class). Only one could ever be indexed; the
1320
+ # other silently vanished. That is unrepresentable — `variants:` carries cross-type
1321
+ # graph identity only — so extraction aborts with both files named.
502
1322
  def deduplicate_results
503
1323
  @results.each do |type, units|
504
- deduped = units.uniq(&:identifier)
505
- dropped = units.size - deduped.size
1324
+ @results[type] = deduplicate_type_units(type, units)
1325
+ end
1326
+ end
1327
+
1328
+ # First-occurrence dedup for one type, fail-closed on cross-file collisions.
1329
+ #
1330
+ # A retained identifier remembers its source file. A later unit with the
1331
+ # same identifier and the same file (or the same absence of one — runtime
1332
+ # and engine units carry nil) is the legitimate re-derivation {#deduplicate_results}
1333
+ # documents and is dropped. A later unit from a different file raises
1334
+ # {Woods::ExtractionError} listing both paths.
1335
+ #
1336
+ # @param type [Symbol] Result type, for the log line and error message
1337
+ # @param units [Array<ExtractedUnit>]
1338
+ # @return [Array<ExtractedUnit>] Deduplicated units, first occurrence order
1339
+ # @raise [Woods::ExtractionError] when one type+identifier is derived
1340
+ # from two different files
1341
+ def deduplicate_type_units(type, units)
1342
+ deduped = []
1343
+ dropped = 0
1344
+ retained_paths = {}
1345
+
1346
+ units.each do |unit|
1347
+ if retained_paths.key?(unit.identifier)
1348
+ prior_path = retained_paths[unit.identifier]
1349
+ if unit.file_path != prior_path
1350
+ raise Woods::ExtractionError, same_type_collision_message(type, unit, prior_path)
1351
+ end
506
1352
 
507
- Rails.logger.warn "[Woods] Deduplicated #{type}: dropped #{dropped} duplicate(s)" if dropped.positive?
1353
+ dropped += 1
1354
+ next
1355
+ end
508
1356
 
509
- @results[type] = deduped
1357
+ retained_paths[unit.identifier] = unit.file_path
1358
+ deduped << unit
510
1359
  end
1360
+
1361
+ Rails.logger.warn "[Woods] Deduplicated #{type}: dropped #{dropped} duplicate(s)" if dropped.positive?
1362
+
1363
+ deduped
1364
+ end
1365
+
1366
+ # @param type [Symbol]
1367
+ # @param unit [ExtractedUnit] The colliding (later) unit
1368
+ # @param prior_path [String, nil] File path of the retained unit
1369
+ # @return [String]
1370
+ def same_type_collision_message(type, unit, prior_path)
1371
+ "same-type identifier collision: #{type.to_s.singularize} '#{unit.identifier}' derived from " \
1372
+ "two different sources ('#{prior_path || 'no file'}' and '#{unit.file_path || 'no file'}'); " \
1373
+ 'only one unit could ever be indexed, so extraction aborted — either merge the ' \
1374
+ 'declarations into one file or split them into distinct constants'
511
1375
  end
512
1376
 
513
1377
  # ──────────────────────────────────────────────────────────────────────
514
1378
  # Flow Precomputation
515
1379
  # ──────────────────────────────────────────────────────────────────────
516
1380
 
1381
+ # Full-path counterpart of {#refresh_incremental_flows}: assemble every
1382
+ # controller's flows, annotate the in-memory units, rewrite them, and
1383
+ # sweep orphans. Failure-closed — any error raises to {#extract_all},
1384
+ # which aborts before write_manifest/publish_generation, so a partial
1385
+ # flow index or half-rewritten annotations can never be published.
1386
+ #
1387
+ # @return [void]
1388
+ # @raise [Woods::ExtractionError] when any flow-family step fails
517
1389
  def precompute_flows
518
1390
  all_units = @results.values.flatten(1)
519
- precomputer = FlowPrecomputer.new(units: all_units, graph: @dependency_graph, output_dir: @output_dir.to_s)
1391
+ precomputer = FlowPrecomputer.new(units: all_units, graph: @dependency_graph, output_dir: payload_dir.to_s)
520
1392
  flow_map = precomputer.precompute
521
1393
  rewrite_flow_annotated_units
1394
+ sweep_orphaned_flow_files
522
1395
  Rails.logger.info "[Woods] Precomputed #{flow_map.size} request flows"
523
- rescue StandardError => e
524
- Rails.logger.error "[Woods] Flow precomputation failed: #{e.message}"
525
1396
  end
526
1397
 
527
1398
  # Precompute runs after write_results (FlowAssembler reads unit JSON
@@ -540,75 +1411,578 @@ module Woods
540
1411
  annotated = units.select { |u| u.metadata[:flow_paths] }
541
1412
  next if annotated.empty?
542
1413
 
543
- type_dir = @output_dir.join(type.to_s)
1414
+ type_dir = payload_dir.join(type.to_s)
544
1415
  annotated.each do |unit|
545
- File.write(
1416
+ AtomicFile.write(
546
1417
  type_dir.join(collision_safe_filename(unit.identifier)),
547
- json_serialize(unit.to_h)
1418
+ json_serialize(unit.to_h),
1419
+ durable: payload_writes_durable?
548
1420
  )
549
1421
  end
550
- File.write(
1422
+ AtomicFile.write(
551
1423
  type_dir.join('_index.json'),
552
- json_serialize(type_index_entries(units))
1424
+ json_serialize(type_index_entries(units)),
1425
+ durable: payload_writes_durable?
553
1426
  )
554
1427
  end
555
1428
  end
556
1429
 
557
- # ──────────────────────────────────────────────────────────────────────
558
- # Git Enrichment
559
- # ──────────────────────────────────────────────────────────────────────
560
-
561
- def enrich_with_git_data
562
- return unless git_available?
563
-
564
- # Collect all file paths that need git data
565
- file_paths = []
566
- @results.each do |type, units|
567
- next if %i[rails_source gem_source].include?(type)
1430
+ # Incremental counterpart of Phase 5.5 (M3): the run's controller delta
1431
+ # gets the same flow annotations a full run would give it, flow
1432
+ # documents and flow_index.json are refreshed for the touched
1433
+ # controllers while untouched controllers' entries carry forward, and
1434
+ # flows/ documents no entry references are swept. Without this, a
1435
+ # controller re-extracted incrementally lost metadata[:flow_paths],
1436
+ # flow_index.json described pre-change routes, and flows/ files for
1437
+ # deleted or renamed controllers persisted across every generation —
1438
+ # seeded forward by PAYLOAD_DIRS.
1439
+ #
1440
+ # Runs inside {#finalize_incremental_run} so both extract_changed and
1441
+ # refresh (whose routes cascade rewrites controllers wholesale) get it,
1442
+ # and so a no-op run — which returns before this — leaves flows alone.
1443
+ #
1444
+ # Skip rule: the refresh participates whenever the gate is on and
1445
+ # something is touched. A genuinely absent family — no flows/ directory,
1446
+ # or an empty one — skips: there is nothing to carry forward, publication
1447
+ # proceeds, and the next full extraction with the gate on builds the
1448
+ # family. Once the family holds ANY artifact it is authoritative: a
1449
+ # missing flow_index.json among documents, a corrupt one, a failed
1450
+ # rehydration, write, patch, sweep, or type-index regeneration raises,
1451
+ # and the raise propagates out of {#finalize_incremental_run} BEFORE
1452
+ # {#publish_generation} — no generation bump, the preceding generation
1453
+ # stays resolved and readable. ( woods:validate applies the same rule
1454
+ # for a populated family without an index.)
1455
+ #
1456
+ # @param touched [Set<String>] identifiers added, re-extracted, or removed
1457
+ # @return [void]
1458
+ # @raise [Woods::ExtractionError] when authoritative flow state is
1459
+ # missing or corrupt, or a flow-family write fails
1460
+ def refresh_incremental_flows(touched)
1461
+ return unless Woods.configuration.precompute_flows
1462
+ return if touched.empty?
1463
+ return unless flow_family_present?
1464
+
1465
+ controllers_dir = payload_dir.join('controllers')
1466
+ reextracted = touched.select { |id| controllers_dir.join(collision_safe_filename(id)).exist? }
1467
+ removed = previous_flow_index_controllers & (touched.to_set - reextracted.to_set)
1468
+ return if reextracted.empty? && removed.empty?
1469
+
1470
+ reassemble, carried = partition_flow_controllers(
1471
+ reextracted.filter_map { |id| unit_from_payload(:controllers, id) }
1472
+ )
1473
+ Rails.logger.info "[Woods] Refreshing flows for #{reassemble.size} controller(s), " \
1474
+ "#{carried.size} carried forward, #{removed.size} removed..."
1475
+ precomputer = FlowPrecomputer.new(units: [], graph: @dependency_graph, output_dir: payload_dir.to_s)
1476
+ annotations = precomputer.recompute_delta(
1477
+ touched_units: reassemble,
1478
+ removed_identifiers: removed.to_a,
1479
+ carried_identifiers: carried.map(&:identifier)
1480
+ )
1481
+ patch_flow_annotations(annotations)
1482
+ sweep_orphaned_flow_files
1483
+ # The annotation patch changed controller JSON after the run's type
1484
+ # index regeneration; the index carries estimated_tokens, which the
1485
+ # flow_paths are part of — a full run builds its index from the
1486
+ # annotated in-memory units, so the incremental one re-derives it from
1487
+ # the annotated files to match. A failure here raises like every
1488
+ # other refresh failure.
1489
+ regenerate_type_index(:controllers) if annotations.any?
1490
+ end
568
1491
 
569
- units.each do |unit|
570
- file_paths << unit.file_path if unit.file_path && File.exist?(unit.file_path)
571
- end
1492
+ # Split the run's re-extracted controllers into the ones whose flow
1493
+ # documents have to be reassembled and the ones that only need their
1494
+ # annotation back.
1495
+ #
1496
+ # A controller is reassembled when the run changed something its flow can
1497
+ # reach ({#flow_scope_for}), or when its own action set no longer matches
1498
+ # the previous generation's index. The second test is what catches an
1499
+ # action arriving from further up a controller inheritance chain than the
1500
+ # flow radius reaches, and a controller the index has never seen.
1501
+ #
1502
+ # Without a scope (a targeted {#refresh}, or a routes re-run) everything
1503
+ # is reassembled, which is what this path always did.
1504
+ #
1505
+ # @param units [Array<ExtractedUnit>] the run's re-extracted controllers
1506
+ # @return [Array(Array<ExtractedUnit>, Array<ExtractedUnit>)]
1507
+ def partition_flow_controllers(units)
1508
+ return [units, []] if @flow_scope.nil?
1509
+
1510
+ previous = previous_flow_index_actions
1511
+ units.partition do |unit|
1512
+ @flow_scope.include?(unit.identifier) || flow_actions_of(unit) != previous[unit.identifier]
572
1513
  end
1514
+ end
573
1515
 
574
- # Batch-fetch all git data in minimal subprocess calls
575
- git_data = batch_git_data(file_paths)
576
- root = "#{Rails.root}/"
1516
+ # The actions a unit's metadata declares, as the index records them.
1517
+ #
1518
+ # @param unit [ExtractedUnit]
1519
+ # @return [Set<String>]
1520
+ def flow_actions_of(unit)
1521
+ Array(unit.metadata[:actions] || unit.metadata['actions']).to_set(&:to_s)
1522
+ end
577
1523
 
578
- # Assign results to units
579
- @results.each do |type, units|
580
- next if %i[rails_source gem_source].include?(type)
1524
+ # The previous generation's flow index, grouped by controller.
1525
+ #
1526
+ # @return [Hash{String => Set<String>}] controller identifier to its
1527
+ # recorded action names, defaulting to an empty set
1528
+ # @raise [Woods::ExtractionError] when the index is missing or corrupt
1529
+ def previous_flow_index_actions
1530
+ previous_flow_index_entries.keys.each_with_object(Hash.new { Set.new }) do |entry_point, grouped|
1531
+ controller, action = entry_point.to_s.split('#', 2)
1532
+ next unless action
1533
+
1534
+ grouped[controller] = grouped[controller] + [action]
1535
+ end
1536
+ end
581
1537
 
582
- units.each do |unit|
583
- next unless unit.file_path
1538
+ # Does the run's seeded payload hold a flow family at all? An absent
1539
+ # family (no flows/ directory, or an empty one) is a genuine absence —
1540
+ # typically an index built while the gate was off — and skips the
1541
+ # refresh rather than publishing a delta-only index.
1542
+ #
1543
+ # @return [Boolean]
1544
+ def flow_family_present?
1545
+ flows_dir = payload_dir.join('flows')
1546
+ return false unless flows_dir.directory?
584
1547
 
585
- relative = unit.file_path.sub(root, '')
586
- unit.metadata[:git] = git_data[relative] if git_data[relative]
587
- end
588
- end
1548
+ !Dir[flows_dir.join('*.json')].empty?
589
1549
  end
590
1550
 
591
- # Normalize all unit file_paths to relative paths (relative to Rails.root).
1551
+ # Controller identifiers that hold entries in the previous generation's
1552
+ # flow_index.json. Only these can be flow-removed this run: a controller
1553
+ # with no index entries has no documents to sweep and no annotation to
1554
+ # clear.
592
1555
  #
593
- # Extractors set file_path via source_location, which returns absolute paths.
594
- # This normalization ensures consistent relative paths (e.g., "app/models/user.rb")
595
- # across all environments (local, Docker, CI) where Rails.root differs.
1556
+ # The read is authoritative ({#flow_family_present?} guaranteed the
1557
+ # family holds artifacts): a missing index among documents, or a corrupt
1558
+ # one, raises so publication aborts.
596
1559
  #
597
- # Must run after enrich_with_git_data, which needs absolute paths for
598
- # File.exist? checks and git log commands.
599
- def normalize_file_paths
600
- @results.each_value do |units|
601
- units.each do |unit|
602
- unit.file_path = normalize_file_path(unit.file_path)
603
- end
604
- end
1560
+ # @return [Set<String>]
1561
+ # @raise [Woods::ExtractionError] when the index is missing or corrupt
1562
+ def previous_flow_index_controllers
1563
+ previous_flow_index_entries.keys.to_set { |entry_point| entry_point.to_s.split('#', 2).first }
605
1564
  end
606
1565
 
607
- # Strip Rails.root prefix from a file path, converting it to a relative path.
1566
+ # The previous generation's flow index itself, read once per run: both
1567
+ # the removal set and the reassembly partition are derived from it.
608
1568
  #
609
- # @param path [String, nil] Absolute or relative file path
610
- # @return [String, nil] Relative path, or the original value if already relative,
611
- # nil, or not under Rails.root (e.g., a gem path)
1569
+ # @return [Hash{String => String}] entry point to relative document path
1570
+ # @raise [Woods::ExtractionError] when the index is missing or corrupt
1571
+ def previous_flow_index_entries
1572
+ return @previous_flow_index_entries if @previous_flow_index_entries
1573
+
1574
+ index_path = payload_dir.join('flows', 'flow_index.json')
1575
+ raise Woods::ExtractionError, 'flows/ is populated but flow_index.json is missing' unless index_path.exist?
1576
+
1577
+ @previous_flow_index_entries = JSON.parse(AtomicFile.read(index_path))
1578
+ rescue JSON::ParserError => e
1579
+ raise Woods::ExtractionError, "previous flow_index.json does not parse: #{e.message}"
1580
+ end
1581
+
1582
+ # Rehydrate one unit from its payload JSON for the incremental flow
1583
+ # pass. FlowPrecomputer only needs the identifier and the actions
1584
+ # metadata; the source lives on disk, where FlowAssembler reads it.
1585
+ #
1586
+ # @param type_key [Symbol] extractor key naming the payload directory
1587
+ # @param identifier [String]
1588
+ # @return [ExtractedUnit, nil]
1589
+ # @raise [Woods::ExtractionError] when the unit JSON does not parse —
1590
+ # skipping it would publish annotations against stale state
1591
+ def unit_from_payload(type_key, identifier)
1592
+ path = payload_dir.join(type_key.to_s, collision_safe_filename(identifier))
1593
+ return unless File.exist?(path)
1594
+
1595
+ data = JSON.parse(AtomicFile.read(path))
1596
+ unit = ExtractedUnit.new(
1597
+ type: data['type'],
1598
+ identifier: data['identifier'] || identifier,
1599
+ file_path: data['file_path']
1600
+ )
1601
+ unit.metadata = data['metadata'] || {}
1602
+ unit.source_code = data['source_code']
1603
+ unit
1604
+ rescue JSON::ParserError => e
1605
+ raise Woods::ExtractionError, "could not rehydrate #{identifier} for flow refresh: #{e.message}"
1606
+ end
1607
+
1608
+ # Write metadata[:flow_paths] into the re-extracted controllers' unit
1609
+ # JSON — the incremental counterpart of {#rewrite_flow_annotated_units}.
1610
+ # A controller with no flows this run loses any annotation a previous
1611
+ # run had written, which is what a full run would have produced for the
1612
+ # same tree. Read-compare-write, like {#rewrite_unit_json_of_type}; a
1613
+ # read or write failure raises so publication aborts.
1614
+ #
1615
+ # @param annotations [Hash{String => Hash{String => String}}] from
1616
+ # {FlowPrecomputer#recompute_delta}
1617
+ # @return [void]
1618
+ def patch_flow_annotations(annotations)
1619
+ annotations.each do |identifier, flow_paths|
1620
+ path = payload_dir.join('controllers', collision_safe_filename(identifier))
1621
+ next unless File.exist?(path)
1622
+
1623
+ data = JSON.parse(AtomicFile.read(path))
1624
+ before = JSON.generate(data)
1625
+
1626
+ metadata = (data['metadata'] ||= {})
1627
+ if flow_paths.any?
1628
+ metadata['flow_paths'] = flow_paths
1629
+ else
1630
+ metadata.delete('flow_paths')
1631
+ end
1632
+ next if JSON.generate(data) == before
1633
+
1634
+ AtomicFile.write(path, json_serialize(data), durable: payload_writes_durable?)
1635
+ end
1636
+ end
1637
+
1638
+ # Remove flows/ documents no entry of flow_index.json references.
1639
+ #
1640
+ # Deliberately NOT part of {#sweep_orphaned_unit_files}: that sweep
1641
+ # deletes per-type unit files no in-memory unit accounts for, while
1642
+ # flows/ holds neither units nor an _index.json. The flow family is
1643
+ # defined by flow_index.json's references, and this validates against
1644
+ # exactly those (M3) — the same artifact the validator treats separately
1645
+ # (G-2). Before this, a controller deleted or renamed incrementally left
1646
+ # its flow documents behind forever.
1647
+ #
1648
+ # Reconciled skip rule: a genuinely empty directory is an absence and is
1649
+ # skipped; a populated one whose index is missing, or an index that does
1650
+ # not parse, is corruption and raises — with nothing to validate
1651
+ # against, deleting every document would be the one outcome worse than
1652
+ # keeping orphans. Failures raise so the incremental caller aborts
1653
+ # publication.
1654
+ #
1655
+ # @return [void]
1656
+ # @raise [Woods::ExtractionError] when authoritative flow state is
1657
+ # missing or corrupt
1658
+ def sweep_orphaned_flow_files
1659
+ flows_dir = payload_dir.join('flows')
1660
+ return unless flows_dir.directory?
1661
+
1662
+ index_path = flows_dir.join('flow_index.json')
1663
+ unless index_path.exist?
1664
+ return if Dir[flows_dir.join('*.json')].empty?
1665
+
1666
+ raise Woods::ExtractionError, 'flows/ is populated but flow_index.json is missing'
1667
+ end
1668
+
1669
+ keep = parse_flow_index_for_sweep(index_path)
1670
+ .to_set { |relative| File.basename(relative.to_s) } << 'flow_index.json'
1671
+ orphans = Dir[flows_dir.join('*.json').to_s].reject { |file| keep.include?(File.basename(file)) }
1672
+ return if orphans.empty?
1673
+
1674
+ orphans.each { |file| FileUtils.rm_f(file) }
1675
+ Rails.logger.info "[Woods] Swept #{orphans.size} orphaned flow file(s)"
1676
+ end
1677
+
1678
+ # @param index_path [Pathname]
1679
+ # @return [Array<String>] the referenced document paths
1680
+ # @raise [Woods::ExtractionError] when the index does not parse
1681
+ def parse_flow_index_for_sweep(index_path)
1682
+ JSON.parse(AtomicFile.read(index_path)).values
1683
+ rescue JSON::ParserError => e
1684
+ raise Woods::ExtractionError, "flow_index.json does not parse: #{e.message}"
1685
+ end
1686
+
1687
+ # ──────────────────────────────────────────────────────────────────────
1688
+ # Git Enrichment
1689
+ # ──────────────────────────────────────────────────────────────────────
1690
+
1691
+ def enrich_with_git_data
1692
+ return unless git_available?
1693
+
1694
+ # Collect all file paths that need git data. Only app-owned paths under
1695
+ # Rails.root qualify — see {#git_enrichable_path?}.
1696
+ root = "#{Rails.root}/"
1697
+ file_paths = []
1698
+ @results.each do |type, units|
1699
+ next if %i[rails_source gem_source].include?(type)
1700
+
1701
+ units.each do |unit|
1702
+ path = unit.file_path
1703
+ file_paths << path if git_enrichable_path?(path, root)
1704
+ end
1705
+ end
1706
+
1707
+ # Batch-fetch all git data in minimal subprocess calls
1708
+ git_data = batch_git_data(file_paths)
1709
+
1710
+ # Assign results to units
1711
+ @results.each do |type, units|
1712
+ next if %i[rails_source gem_source].include?(type)
1713
+
1714
+ units.each do |unit|
1715
+ next unless unit.file_path
1716
+
1717
+ relative = unit.file_path.sub(root, '')
1718
+ unit.metadata[:git] = git_data[relative] if git_data[relative]
1719
+ end
1720
+ end
1721
+ end
1722
+
1723
+ # Copy git facts onto graph nodes so the analyzer can read them without
1724
+ # the units (an incremental run never holds every unit in memory).
1725
+ #
1726
+ # @return [void]
1727
+ # build_file_metadata always emits commit_count and change_frequency together
1728
+ # and non-nil, so no nil compaction is needed here (contrast annotate_node_from_git).
1729
+ def annotate_graph_with_git_data
1730
+ @results.each_value do |units|
1731
+ units.each do |unit|
1732
+ git = unit.metadata[:git]
1733
+ next unless git.is_a?(Hash)
1734
+
1735
+ @dependency_graph.annotate(
1736
+ unit.identifier,
1737
+ type: unit.type,
1738
+ commit_count: git[:commit_count],
1739
+ change_frequency: git[:change_frequency]
1740
+ )
1741
+ end
1742
+ end
1743
+ end
1744
+
1745
+ # The one constructor for the analyzer both extraction paths use.
1746
+ # `Woods.configuration` can be nil in specs that reset it; fall back to
1747
+ # the analyzer's own default rather than raising mid-run.
1748
+ #
1749
+ # @return [GraphAnalyzer]
1750
+ def build_graph_analyzer
1751
+ config = Woods.configuration
1752
+ ratio = config&.volatile_dependency_ratio || GraphAnalyzer::DEFAULT_VOLATILE_RATIO
1753
+ GraphAnalyzer.new(
1754
+ @dependency_graph,
1755
+ volatile_ratio: ratio,
1756
+ cycle_limit: config ? config.graph_cycle_limit : GraphAnalyzer::DEFAULT_CYCLE_LIMIT,
1757
+ cycle_max_length: config ? config.graph_cycle_max_length : GraphAnalyzer::DEFAULT_CYCLE_MAX_LENGTH
1758
+ )
1759
+ end
1760
+
1761
+ # ──────────────────────────────────────────────────────────────────────
1762
+ # Package membership (#280)
1763
+ # ──────────────────────────────────────────────────────────────────────
1764
+
1765
+ # The package extractor instance this run resolves membership through.
1766
+ # Always built through {#extractor_for}, never `@extractors[:packages]`
1767
+ # (the instance Phase 1 used to extract package units): reading that one
1768
+ # would tie package-membership lookups to whichever instance happened to
1769
+ # run first, instead of to this run's own memoized, on-demand build.
1770
+ # Reset at the start of every run (both {#extract_all} and
1771
+ # {#prepare_incremental_run}), so a later run never resolves membership
1772
+ # against a prior run's package set.
1773
+ #
1774
+ # @return [Extractors::PackageExtractor]
1775
+ def package_resolver
1776
+ @package_resolver ||= extractor_for(:packages) || Extractors::PackageExtractor.new
1777
+ end
1778
+
1779
+ # Set or clear metadata[:package] on one unit. Framework units and
1780
+ # units with no path are never members. The key is deleted rather than
1781
+ # set to nil so a unit outside every package serializes as before.
1782
+ #
1783
+ # `unit.file_path` may be absolute (the full path, before Phase 4.5
1784
+ # relativization, and {#register_and_write}'s incremental path, before
1785
+ # its own normalization) or Rails.root-relative (unit JSON already on
1786
+ # disk); {Extractors::PackageExtractor#package_for} accepts either.
1787
+ #
1788
+ # @param unit [ExtractedUnit]
1789
+ # @return [String, nil] the package name
1790
+ def annotate_package(unit)
1791
+ return nil if %i[rails_source gem_source].include?(unit.type) || unit.file_path.nil?
1792
+
1793
+ package = package_resolver.package_for(unit.file_path)
1794
+ if package
1795
+ unit.metadata[:package] = package
1796
+ else
1797
+ unit.metadata.delete(:package)
1798
+ end
1799
+ package
1800
+ end
1801
+
1802
+ # Full-path pass over every extracted unit. Skipped entirely when the
1803
+ # app declares no packages, so a non-Packwerk app's output is unchanged
1804
+ # (I5): no work, and no `package` key on any unit or node.
1805
+ #
1806
+ # @return [void]
1807
+ def annotate_packages
1808
+ return if package_resolver.package_roots.empty?
1809
+
1810
+ @results.each_value { |units| units.each { |unit| annotate_package(unit) } }
1811
+ end
1812
+
1813
+ # Incremental counterpart: when a package file changed, membership of
1814
+ # units this run never touched may have changed too (a new package
1815
+ # root, a renamed package). Walk the payload's unit JSON, rewrite the
1816
+ # ones whose package differs, and annotate their nodes. Bounded to runs
1817
+ # where a package trigger fired, which is rare; every other run pays
1818
+ # nothing.
1819
+ #
1820
+ # @param change_set [ChangeSet]
1821
+ # @param affected_types [Set<Symbol>]
1822
+ # @return [Set<String>] identifiers rewritten
1823
+ def reannotate_packages(change_set, affected_types)
1824
+ keys = PathDispatcher.new.whole_app_keys_for_all(change_set.relative_paths)
1825
+ return Set.new unless keys.include?(:packages)
1826
+
1827
+ Dir[payload_dir.join('*', '*.json').to_s].each_with_object(Set.new) do |file, touched|
1828
+ next if File.basename(file) == '_index.json'
1829
+
1830
+ type_dir = File.basename(File.dirname(file))
1831
+ next if type_dir == 'rails_source' || PAYLOAD_DIRS.include?(type_dir)
1832
+
1833
+ identifier = reannotate_unit_file(file, type_dir)
1834
+ next unless identifier
1835
+
1836
+ touched.add(identifier)
1837
+ affected_types&.add(type_dir.to_sym)
1838
+ end
1839
+ end
1840
+
1841
+ # @param file [String] absolute path to one unit JSON file
1842
+ # @param type_dir [String] the extractor key directory it lives in
1843
+ # @return [String, nil] the identifier when the file was rewritten
1844
+ def reannotate_unit_file(file, type_dir)
1845
+ data = JSON.parse(AtomicFile.read(file))
1846
+ relative_path = data['file_path']
1847
+ return nil if relative_path.nil? || relative_path.start_with?('/')
1848
+
1849
+ package = package_resolver.package_for(relative_path)
1850
+ metadata = (data['metadata'] ||= {})
1851
+ return nil if metadata['package'] == package
1852
+
1853
+ if package
1854
+ metadata['package'] = package
1855
+ else
1856
+ metadata.delete('package')
1857
+ end
1858
+ AtomicFile.write(file, json_serialize(data), durable: payload_writes_durable?)
1859
+
1860
+ identifier = data['identifier']
1861
+ type = (data['type'] || type_dir.singularize).to_sym
1862
+ @dependency_graph.annotate(identifier, type: type, package: package)
1863
+ identifier
1864
+ rescue JSON::ParserError => e
1865
+ Rails.logger.warn "[Woods] Could not re-annotate package on #{file}: #{e.message}"
1866
+ nil
1867
+ end
1868
+
1869
+ # Is this a path worth asking git about?
1870
+ #
1871
+ # A gem-owned unit (an engine model) carries its real path. Outside
1872
+ # Rails.root, git refuses the whole `log` invocation when any pathspec is
1873
+ # outside the repository — one gem path would erase the git metadata of
1874
+ # the other 499 units in its 500-path batch. Inside Rails.root, a bundle
1875
+ # vendored at `vendor/bundle` puts the same gem files under the root
1876
+ # prefix, gitignored, so sending them is wasted pathspec work every run.
1877
+ # Same exclusions as {Extractors::SharedUtilityMethods#app_source?}.
1878
+ #
1879
+ # @param path [String, nil] absolute file path
1880
+ # @param root [String] Rails.root with a trailing separator
1881
+ # @return [Boolean]
1882
+ def git_enrichable_path?(path, root)
1883
+ return false unless path&.start_with?(root)
1884
+ return false if path.include?('/vendor/') || path.include?('/node_modules/')
1885
+
1886
+ File.exist?(path)
1887
+ end
1888
+
1889
+ # Normalize all unit file_paths to relative paths (relative to Rails.root).
1890
+ #
1891
+ # Extractors set file_path via source_location, which returns absolute paths.
1892
+ # This normalization ensures consistent relative paths (e.g., "app/models/user.rb")
1893
+ # across all environments (local, Docker, CI) where Rails.root differs.
1894
+ #
1895
+ # Must run after enrich_with_git_data, which needs absolute paths for
1896
+ # File.exist? checks and git log commands.
1897
+ # Write a unit's JSON, unless the bytes on disk are already exactly that.
1898
+ #
1899
+ # A routes change replaces every `ROUTE_CONSUMER_EXTRACTORS` type wholesale
1900
+ # — on a production-shaped host that measured 1,707 units, roughly a quarter
1901
+ # of the index — and almost all of them re-serialize to the bytes already
1902
+ # there. `AtomicFile.write` is a tempfile plus an fsync plus a rename each
1903
+ # time, so the fsync is the cost being avoided here; the comparison read is
1904
+ # cheaper than the write it replaces.
1905
+ #
1906
+ # Only the *write* is skipped. Graph registration, the dependents marking
1907
+ # and `@incremental_written` all still happen for every unit, because those
1908
+ # are what equivalence and the git-enrichment pass depend on — skipping any
1909
+ # of them would make an unchanged unit differ from a full extraction.
1910
+ #
1911
+ # Compared as bytes: `AtomicFile.write` is binmode, and the encoding a read
1912
+ # comes back tagged with depends on the process's default external encoding
1913
+ # (US-ASCII under `LANG=C`, which is where the daemon runs).
1914
+ #
1915
+ # @param path [Pathname] destination
1916
+ # @param unit [ExtractedUnit] unit to serialize
1917
+ # @return [void]
1918
+ def write_unit_file(path, unit)
1919
+ payload = json_serialize(unit.to_h)
1920
+ return if identical_on_disk?(path, payload)
1921
+
1922
+ AtomicFile.write(path, payload, durable: payload_writes_durable?)
1923
+ end
1924
+
1925
+ # The serialized `extracted_at` scalar as {ExtractedUnit#to_h} +
1926
+ # {#json_serialize} emit it, compact or pretty. `Time#iso8601` produces
1927
+ # exactly this value shape — no fractional seconds, `Z` or a `±hh:mm`
1928
+ # offset — and `spec/extracted_unit_spec.rb` pins that, so a change to the
1929
+ # stamp's shape fails a spec instead of quietly un-matching this mask.
1930
+ # The value constraint is what keeps the mask honest against user code: a
1931
+ # bare `"extracted_at":` cannot occur inside any JSON *string* value
1932
+ # (interior quotes serialize as `\"`), so only a real JSON key can match,
1933
+ # and only when it holds a timestamp — which no extractor emits below the
1934
+ # top level.
1935
+ EXTRACTED_AT_SCALAR =
1936
+ /("extracted_at":\s*")\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}(?:Z|[+-]\d{2}:\d{2})(?=")/
1937
+ # An implementation detail of the byte comparison, not part of the
1938
+ # extractor's surface (`private` does not scope constants).
1939
+ private_constant :EXTRACTED_AT_SCALAR
1940
+
1941
+ # @return [Boolean] true when the file already holds exactly these bytes,
1942
+ # the `extracted_at` stamp aside
1943
+ def identical_on_disk?(path, payload)
1944
+ return false unless File.exist?(path)
1945
+
1946
+ mask_extracted_at(AtomicFile.read(path).b) == mask_extracted_at(payload.b)
1947
+ rescue StandardError
1948
+ # An unreadable or half-written file is not a match; fall through and
1949
+ # rewrite it.
1950
+ false
1951
+ end
1952
+
1953
+ # Blank the `extracted_at` value so the byte comparison ignores it.
1954
+ #
1955
+ # {ExtractedUnit#to_h} stamps `extracted_at: Time.now.iso8601` on every
1956
+ # serialization, so compared raw the two sides *always* differed across
1957
+ # runs and the skip was inert (#208): every write did the comparison read
1958
+ # and then paid the fsync anyway. Masked with a regex rather than parsed,
1959
+ # because a JSON parse of both sides would cost more than the write being
1960
+ # avoided. A skipped write leaves the older stamp on disk deliberately —
1961
+ # the equivalence oracle (`spec/support/index_comparison.rb`) lists
1962
+ # `extracted_at` in `VOLATILE_UNIT_KEYS`: "a unit an incremental run
1963
+ # correctly left alone keeps an older stamp".
1964
+ #
1965
+ # @param bytes [String] serialized unit JSON, binary-tagged (`.b`) — the
1966
+ # ASCII-only pattern matches bytewise regardless of the content's
1967
+ # original encoding
1968
+ # @return [String] the bytes with the stamp's value removed
1969
+ def mask_extracted_at(bytes)
1970
+ bytes.gsub(EXTRACTED_AT_SCALAR, '\1')
1971
+ end
1972
+
1973
+ def normalize_file_paths
1974
+ @results.each_value do |units|
1975
+ units.each do |unit|
1976
+ unit.file_path = normalize_file_path(unit.file_path)
1977
+ end
1978
+ end
1979
+ end
1980
+
1981
+ # Strip Rails.root prefix from a file path, converting it to a relative path.
1982
+ #
1983
+ # @param path [String, nil] Absolute or relative file path
1984
+ # @return [String, nil] Relative path, or the original value if already relative,
1985
+ # nil, or not under Rails.root (e.g., a gem path)
612
1986
  def normalize_file_path(path)
613
1987
  return path unless path
614
1988
 
@@ -617,15 +1991,60 @@ module Woods
617
1991
  path.start_with?(prefix) ? path.sub(prefix, '') : path
618
1992
  end
619
1993
 
1994
+ # Can this run's git calls produce real facts?
1995
+ #
1996
+ # `rev-parse --git-dir` alone is not enough. Over a linked worktree whose
1997
+ # private git directory is reachable but whose `commondir` is not, it
1998
+ # answers while every ref lookup fails and `git log` exits 0 with nothing
1999
+ # to say. Enrichment then wrote `commit_count: 0` and
2000
+ # `change_frequency: new` onto every unit, which reads exactly like a file
2001
+ # that was never committed, where an absent git directory correctly omits
2002
+ # the keys (B-186). HEAD has to resolve.
2003
+ #
2004
+ # Memoized, so the warning below is emitted at most once per run.
2005
+ #
2006
+ # @return [Boolean]
620
2007
  def git_available?
621
2008
  return @git_available if defined?(@git_available)
622
2009
 
623
- @git_available = begin
624
- _, status = Open3.capture2('git', 'rev-parse', '--git-dir')
625
- status.success?
626
- rescue StandardError
627
- false
628
- end
2010
+ _output, error, status = Open3.capture3(*git_argv('rev-parse', 'HEAD'))
2011
+ @git_available = status.success?
2012
+ warn_unresolvable_git(error) unless @git_available
2013
+ @git_available
2014
+ rescue StandardError
2015
+ @git_available = false
2016
+ end
2017
+
2018
+ # Say once why no unit will carry git metadata, but only when there is a
2019
+ # working tree to explain. No `.git` at the root is the ordinary source
2020
+ # tarball or `COPY`-without-`.git` case, and it is not a fault.
2021
+ #
2022
+ # @param error [String] git's own stderr
2023
+ # @return [void]
2024
+ def warn_unresolvable_git(error)
2025
+ return unless File.exist?(File.join(Rails.root.to_s, '.git'))
2026
+
2027
+ cause = error.to_s.lines.first.to_s.strip
2028
+ Rails.logger.warn(
2029
+ '[Woods] git cannot resolve HEAD for this working tree, so no unit will carry git ' \
2030
+ "metadata: #{cause}. Over a linked worktree in a container, mount the canonical git " \
2031
+ 'directory and point WOODS_GIT_DIR at it; GIT_DIR alone is not enough, because the ' \
2032
+ "worktree's private git directory reaches the shared one through a relative pointer."
2033
+ )
2034
+ end
2035
+
2036
+ # The git command line every enrichment call runs.
2037
+ #
2038
+ # `-C <root>` keeps the result independent of the process working
2039
+ # directory. `WOODS_GIT_DIR` wins when set: it names the canonical git
2040
+ # directory directly, which is the escape hatch for a container that can
2041
+ # mount that directory but not the host path a worktree pointer names
2042
+ # (B-181).
2043
+ #
2044
+ # @param args [Array<String>] git arguments
2045
+ # @return [Array<String>] full argv
2046
+ def git_argv(*args)
2047
+ GitCommand.argv(Rails.root, *args)
629
2048
  end
630
2049
 
631
2050
  # Safe git command execution — no shell interpolation
@@ -633,7 +2052,7 @@ module Woods
633
2052
  # @param args [Array<String>] Git command arguments
634
2053
  # @return [String] Command output (empty string on failure)
635
2054
  def run_git(*args)
636
- output, status = Open3.capture2('git', *args)
2055
+ output, _error, status = Open3.capture3(*git_argv(*args))
637
2056
  status.success? ? output.strip : ''
638
2057
  rescue StandardError
639
2058
  ''
@@ -641,13 +2060,18 @@ module Woods
641
2060
 
642
2061
  # Batch-fetch git data for all file paths in two git commands.
643
2062
  #
2063
+ # Duplicate paths are collapsed before slicing (audit P9d): many units
2064
+ # share one file_path, and duplicates only repeat a pathspec another
2065
+ # batch also sent. The result is keyed by relative path, so the output
2066
+ # is identical.
2067
+ #
644
2068
  # @param file_paths [Array<String>] Absolute file paths
645
2069
  # @return [Hash{String => Hash}] Keyed by relative path
646
2070
  def batch_git_data(file_paths)
647
2071
  return {} if file_paths.empty?
648
2072
 
649
2073
  root = "#{Rails.root}/"
650
- relative_paths = file_paths.map { |f| f.sub(root, '') }
2074
+ relative_paths = file_paths.map { |f| f.sub(root, '') }.uniq
651
2075
  result = {}
652
2076
  relative_paths.each { |rp| result[rp] = {} }
653
2077
 
@@ -735,23 +2159,91 @@ module Woods
735
2159
 
736
2160
  def write_results
737
2161
  @results.each do |type, units|
738
- type_dir = @output_dir.join(type.to_s)
2162
+ type_dir = payload_dir.join(type.to_s)
739
2163
 
740
2164
  units.each do |unit|
741
- File.write(
2165
+ AtomicFile.write(
742
2166
  type_dir.join(collision_safe_filename(unit.identifier)),
743
- json_serialize(unit.to_h)
2167
+ json_serialize(unit.to_h),
2168
+ durable: payload_writes_durable?
744
2169
  )
745
2170
  end
746
2171
 
747
2172
  # Also write a type index for fast lookups
748
- File.write(
2173
+ AtomicFile.write(
749
2174
  type_dir.join('_index.json'),
750
- json_serialize(type_index_entries(units))
2175
+ json_serialize(type_index_entries(units)),
2176
+ durable: payload_writes_durable?
751
2177
  )
752
2178
  end
753
2179
  end
754
2180
 
2181
+ # Delete unit JSON files in the just-written type directories that this
2182
+ # full run did not write (#177).
2183
+ #
2184
+ # A full extraction never wipes the output directory —
2185
+ # {#setup_output_directory} only mkdir_p's — so extracting into a
2186
+ # directory that saw a different version of the app left the previous
2187
+ # run's `<Unit>_<digest>.json` files behind. The manifest and the freshly
2188
+ # written `_index.json` were correct, but the NEXT incremental run's
2189
+ # {#regenerate_type_index} rebuilds the index from a disk glob of the type
2190
+ # dir, resurrecting every orphan into the listed index, and
2191
+ # {#persisted_counts} then writes the inflated counts into the manifest —
2192
+ # orphans became listed, retrievable-by-listing units the graph knows
2193
+ # nothing about. On a full run the in-memory `@results` are authoritative
2194
+ # — the same argument {#rewrite_flow_annotated_units} makes for refusing
2195
+ # the disk glob — so a unit file no current unit accounts for is stale by
2196
+ # definition and is removed here.
2197
+ #
2198
+ # The boundary is deliberately narrow:
2199
+ #
2200
+ # * Only the per-type directories this run produced — `@results` keys —
2201
+ # are entered. When `include_framework_sources` gates rails_source out
2202
+ # of the run ({#skip_by_configuration?}), `@results` holds no
2203
+ # `:rails_source` key, so a knob-off full run leaves explicitly
2204
+ # extracted framework units (`woods:extract_framework`) alone — the
2205
+ # #169 preservation decision. Nothing outside those directories is
2206
+ # reachable: manifest, graph, generation, `woods.json`, `dumps/`,
2207
+ # `flows/` and the lock/status files all live in the output root or in
2208
+ # directories no extractor key names.
2209
+ # * Only `*.json` directly inside a type dir is considered, and
2210
+ # `_index.json` is always kept. Non-JSON files and subdirectories are
2211
+ # never touched.
2212
+ # * Legitimate names are exactly what {#write_results} just wrote —
2213
+ # {FilenameUtils#collision_safe_filename} over the type's deduped units
2214
+ # (which covers every unit type the extractor emits: gem_source shares
2215
+ # `@results[:rails_source]`). Legacy {FilenameUtils#safe_filename} names
2216
+ # carry no digest suffix, so they never match a current name and are by
2217
+ # definition orphans of a pre-collision-safe run.
2218
+ #
2219
+ # The incremental path deliberately has no counterpart: it holds only
2220
+ # changed units in memory, so "not in memory" means nothing there —
2221
+ # deletion on that path is driven by the graph and the change set
2222
+ # ({#prune_vanished_units}, {#remove_replaced_units}).
2223
+ #
2224
+ # @return [void]
2225
+ def sweep_orphaned_unit_files
2226
+ @results.each do |type, units|
2227
+ type_dir = payload_dir.join(type.to_s)
2228
+ next unless type_dir.directory?
2229
+
2230
+ keep = units.to_set { |unit| collision_safe_filename(unit.identifier) }
2231
+ keep << '_index.json'
2232
+
2233
+ orphans = Dir[type_dir.join('*.json').to_s].reject { |file| keep.include?(File.basename(file)) }
2234
+ next if orphans.empty?
2235
+
2236
+ orphans.each { |file| FileUtils.rm_f(file) }
2237
+ Rails.logger.info "[Woods] Swept #{orphans.size} orphaned unit file(s) from #{type}/"
2238
+ end
2239
+ rescue StandardError => e
2240
+ # The extraction itself already succeeded; aborting the run over cleanup
2241
+ # would trade orphaned files for a lost index. Error, not warn: the
2242
+ # files this leaves behind are exactly what the next incremental run's
2243
+ # disk glob resurrects.
2244
+ Rails.logger.error "[Woods] Orphaned-unit sweep failed: #{e.message}"
2245
+ end
2246
+
755
2247
  # Build the `_index.json` entry list for a set of in-memory units.
756
2248
  # Shared by {#write_results} and {#rewrite_flow_annotated_units} so both
757
2249
  # emit the index from the authoritative in-memory `@results` rather than
@@ -773,12 +2265,28 @@ module Woods
773
2265
 
774
2266
  def write_dependency_graph
775
2267
  graph_data = @dependency_graph.to_h
776
- graph_data[:pagerank] = @dependency_graph.pagerank
2268
+ # Key-sorted for the same reason `to_h` sorts its own sections: the
2269
+ # digest published as `graph_sha` covers these bytes, so node
2270
+ # registration order must not reach it (B-180).
2271
+ graph_data[:pagerank] = @dependency_graph.pagerank.sort_by { |identifier, _| identifier }.to_h
2272
+
2273
+ payload = json_serialize(graph_data)
2274
+ # The bytes about to land on disk are the bytes `graph_sha` covers, so
2275
+ # keep the digest here rather than reading a whole large-app graph back
2276
+ # to compute it. AtomicFile writes in binary mode and reads back as
2277
+ # UTF-8, so the two digests are the same either way.
2278
+ @graph_sha = Digest::SHA256.hexdigest(payload)
2279
+ AtomicFile.write(payload_dir.join('dependency_graph.json'), payload, durable: payload_writes_durable?)
2280
+ end
777
2281
 
778
- File.write(
779
- @output_dir.join('dependency_graph.json'),
780
- json_serialize(graph_data)
781
- )
2282
+ # The digest of the dependency graph this run wrote.
2283
+ #
2284
+ # Falls back to reading the file for a caller that writes the analysis
2285
+ # without having written the graph in the same run.
2286
+ #
2287
+ # @return [String] hex SHA256 of dependency_graph.json
2288
+ def graph_sha
2289
+ @graph_sha || Digest::SHA256.hexdigest(AtomicFile.read(payload_dir.join('dependency_graph.json')))
782
2290
  end
783
2291
 
784
2292
  def write_graph_analysis
@@ -786,14 +2294,13 @@ module Woods
786
2294
 
787
2295
  enriched = @graph_analysis.merge(
788
2296
  generated_at: Time.current.iso8601,
789
- graph_sha: Digest::SHA256.hexdigest(
790
- File.read(@output_dir.join('dependency_graph.json'))
791
- )
2297
+ graph_sha: graph_sha
792
2298
  )
793
2299
 
794
- File.write(
795
- @output_dir.join('graph_analysis.json'),
796
- json_serialize(enriched)
2300
+ AtomicFile.write(
2301
+ payload_dir.join('graph_analysis.json'),
2302
+ json_serialize(enriched),
2303
+ durable: payload_writes_durable?
797
2304
  )
798
2305
  end
799
2306
 
@@ -836,34 +2343,57 @@ module Woods
836
2343
  schema_sha: schema_sha
837
2344
  }
838
2345
 
839
- File.write(
840
- @output_dir.join('manifest.json'),
841
- json_serialize(manifest)
2346
+ AtomicFile.write(
2347
+ payload_dir.join('manifest.json'),
2348
+ json_serialize(manifest),
2349
+ durable: payload_writes_durable?
842
2350
  )
843
2351
  end
844
2352
 
845
2353
  # Unit and chunk counts derived from the per-type _index.json files on
846
- # disk — the source of truth after an incremental run, where only the
2354
+ # disk: the source of truth after an incremental run, where only the
847
2355
  # affected units were re-extracted.
848
2356
  #
849
2357
  # @return [Array(Hash{Symbol => Integer}, Integer)] counts by type, total chunk count
850
2358
  def persisted_counts
851
- counts = {}
852
- chunks = 0
2359
+ stats = persisted_index_stats
2360
+ [stats.transform_values { |type_stats| type_stats[:count] },
2361
+ stats.sum { |_type, type_stats| type_stats[:chunks] }]
2362
+ end
853
2363
 
854
- Dir[@output_dir.join('*/_index.json').to_s].each do |index_path|
855
- entries = JSON.parse(File.read(index_path))
856
- counts[File.basename(File.dirname(index_path)).to_sym] = entries.size
857
- chunks += entries.sum { |e| e['chunk_count'].to_i }
2364
+ # One pass over the persisted per-type _index.json files.
2365
+ #
2366
+ # {#persisted_counts} feeds the manifest and {#persisted_summary_stats}
2367
+ # feeds SUMMARY.md, they run back to back at the end of every incremental
2368
+ # run, and each used to parse every type index for itself. Reading once and
2369
+ # deriving both is also what keeps the two artifacts from disagreeing about
2370
+ # totals, which was previously a property of them using the same source
2371
+ # rather than the same read.
2372
+ #
2373
+ # An unreadable index drops that whole type from the manifest and the
2374
+ # summary alike, with one warning rather than the two the separate passes
2375
+ # emitted. Sorted for determinism, since the glob order is the
2376
+ # filesystem's.
2377
+ #
2378
+ # Memoized per run: invalidated at the start of {#extract_all} and
2379
+ # {#prepare_incremental_run}, and again whenever {#regenerate_type_index}
2380
+ # rewrites an index underneath it.
2381
+ #
2382
+ # @return [Hash{Symbol => Hash}] type => `{ count:, chunks:, namespaces: }`
2383
+ def persisted_index_stats
2384
+ @persisted_index_stats ||= Dir[payload_dir.join('*/_index.json').to_s].each_with_object({}) do |path, stats|
2385
+ entries = JSON.parse(AtomicFile.read(path))
2386
+ stats[File.basename(File.dirname(path)).to_sym] = {
2387
+ count: entries.size,
2388
+ chunks: entries.sum { |entry| entry['chunk_count'].to_i },
2389
+ namespaces: namespace_histogram(entries.map { |entry| entry['namespace'] })
2390
+ }
858
2391
  rescue JSON::ParserError => e
859
- # An unreadable index silently drops that whole type from the manifest
860
- # counts — warn rather than undercount without a trace.
861
- type = File.basename(File.dirname(index_path))
862
- Rails.logger.warn("[Woods] Skipping unreadable #{type}/_index.json in manifest counts: #{e.message}")
863
- next
2392
+ type = File.basename(File.dirname(path))
2393
+ Rails.logger.warn(
2394
+ "[Woods] Skipping unreadable #{type}/_index.json in manifest counts and summary totals: #{e.message}"
2395
+ )
864
2396
  end
865
-
866
- [counts, chunks]
867
2397
  end
868
2398
 
869
2399
  # Capture a temporal snapshot after extraction completes.
@@ -876,10 +2406,10 @@ module Woods
876
2406
  def capture_snapshot
877
2407
  return unless Woods.configuration.enable_snapshots
878
2408
 
879
- manifest_path = @output_dir.join('manifest.json')
2409
+ manifest_path = payload_dir.join('manifest.json')
880
2410
  return unless manifest_path.exist?
881
2411
 
882
- manifest = JSON.parse(File.read(manifest_path))
2412
+ manifest = JSON.parse(AtomicFile.read(manifest_path))
883
2413
  # Snapshots are keyed on the commit SHA — an unresolvable provenance
884
2414
  # ("unknown", see GitProvenance/#137) must not key or collide a snapshot.
885
2415
  git_sha = manifest['git_sha']
@@ -932,32 +2462,39 @@ module Woods
932
2462
  # category with count and top-5 namespace breakdown, rather than enumerating
933
2463
  # every unit. Per-unit detail is available in the per-category _index.json files.
934
2464
  #
2465
+ # The full path derives its numbers from the in-memory results. An
2466
+ # incremental run holds only changed units in memory, so it derives the
2467
+ # same shape from the persisted per-type _index.json files — the same
2468
+ # source {#persisted_counts} feeds the manifest from, which is what keeps
2469
+ # SUMMARY.md's totals agreeing with manifest.json. (M4: the hardlinked
2470
+ # previous generation's summary this path used to leave in place carried
2471
+ # stale totals after any run that added or removed units.) The `Generated:`
2472
+ # stamp keeps its meaning either way: it names the moment the summary was
2473
+ # written.
2474
+ #
935
2475
  # @return [void]
936
2476
  def write_structural_summary
937
- return if @results.empty?
2477
+ stats = @results.empty? ? persisted_summary_stats : results_summary_stats
2478
+ return unless stats
938
2479
 
939
- total_units = @results.values.sum(&:size)
940
- total_chunks = @results.sum { |_, units| units.sum { |u| [u.chunks.size, 1].max } }
941
- category_count = @results.count { |_, units| units.any? }
2480
+ total_units = stats.sum { |_, s| s[:count] }
2481
+ # Matches the manifest's count (`write_manifest`/`persisted_counts`
2482
+ # both sum chunk counts directly) — the previous `[size, 1].max`
2483
+ # floor made SUMMARY.md disagree with manifest.json for every
2484
+ # unchunked unit.
2485
+ total_chunks = stats.sum { |_, s| s[:chunks] }
942
2486
 
943
2487
  summary = []
944
2488
  summary << '# Codebase Index Summary'
945
2489
  summary << "Generated: #{Time.current.iso8601}"
946
2490
  summary << "Rails #{Rails.version} / Ruby #{RUBY_VERSION}"
947
- summary << "Units: #{total_units} | Chunks: #{total_chunks} | Categories: #{category_count}"
2491
+ summary << "Units: #{total_units} | Chunks: #{total_chunks} | Categories: #{stats.size}"
948
2492
  summary << ''
949
2493
 
950
- @results.each do |type, units|
951
- next if units.empty?
952
-
953
- summary << "## #{type.to_s.titleize} (#{units.size})"
954
-
955
- ns_counts = units
956
- .group_by { |u| u.namespace.nil? || u.namespace.empty? ? '(root)' : u.namespace }
957
- .transform_values(&:size)
958
- .sort_by { |_, count| -count }
959
- .first(5)
2494
+ stats.each do |type, s|
2495
+ summary << "## #{type.to_s.titleize} (#{s[:count]})"
960
2496
 
2497
+ ns_counts = s[:namespaces].sort_by { |_, count| -count }.first(5)
961
2498
  ns_parts = ns_counts.map { |ns, count| "#{ns} #{count}" }
962
2499
  summary << "Namespaces: #{ns_parts.join(', ')}" unless ns_parts.empty?
963
2500
  summary << ''
@@ -979,25 +2516,75 @@ module Woods
979
2516
  hub_names = significant_hubs.map { |h| h[:identifier] }.join(', ')
980
2517
  summary << "- Hub nodes (>20 dependents): #{hub_names}"
981
2518
  end
2519
+
2520
+ volatile = Array(@graph_analysis[:volatile_dependencies]).first(5)
2521
+ if volatile.any?
2522
+ lines = volatile.map { |v| "#{v[:from]} -> #{v[:to]} (#{v[:from_commits]} vs #{v[:to_commits]} commits)" }
2523
+ summary << "- Volatile dependencies (top #{lines.size}): #{lines.join('; ')}"
2524
+ end
982
2525
  end
983
2526
 
984
2527
  summary << ''
985
2528
 
986
- File.write(
987
- @output_dir.join('SUMMARY.md'),
988
- summary.join("\n")
2529
+ AtomicFile.write(
2530
+ payload_dir.join('SUMMARY.md'),
2531
+ summary.join("\n"),
2532
+ durable: payload_writes_durable?
989
2533
  )
990
2534
  end
991
2535
 
2536
+ # Per-type `{ count:, chunks:, namespaces: }` for {#write_structural_summary},
2537
+ # derived from the in-memory results a full extraction holds.
2538
+ #
2539
+ # @return [Hash{Symbol => Hash}]
2540
+ def results_summary_stats
2541
+ @results.each_with_object({}) do |(type, units), stats|
2542
+ next if units.empty?
2543
+
2544
+ stats[type] = {
2545
+ count: units.size,
2546
+ chunks: units.sum { |u| u.chunks.size },
2547
+ namespaces: namespace_histogram(units.map(&:namespace))
2548
+ }
2549
+ end
2550
+ end
2551
+
2552
+ # The incremental path's counterpart to {#results_summary_stats}: the same
2553
+ # shape, read back through {#persisted_index_stats}. A type whose index is
2554
+ # empty is left out, where the manifest still counts it as zero.
2555
+ #
2556
+ # @return [Hash{Symbol => Hash}, nil] nil when the payload holds no
2557
+ # non-empty type index, so a bare index writes no summary, matching the
2558
+ # full path's early return when it extracted nothing
2559
+ def persisted_summary_stats
2560
+ stats = persisted_index_stats.reject { |_type, type_stats| type_stats[:count].zero? }
2561
+
2562
+ stats.empty? ? nil : stats
2563
+ end
2564
+
2565
+ # Group namespaces for a SUMMARY.md section: a missing or empty namespace
2566
+ # is the root, everything else counts as-is.
2567
+ #
2568
+ # @param namespaces [Array<String, nil>]
2569
+ # @return [Hash{String => Integer}]
2570
+ def namespace_histogram(namespaces)
2571
+ namespaces.group_by { |ns| ns.nil? || ns.empty? ? '(root)' : ns }
2572
+ .transform_values(&:size)
2573
+ end
2574
+
992
2575
  def regenerate_type_index(type_key)
993
- type_dir = @output_dir.join(type_key.to_s)
2576
+ type_dir = payload_dir.join(type_key.to_s)
994
2577
  return unless type_dir.directory?
995
2578
 
2579
+ # This run's counts and summary totals are read from the index files;
2580
+ # rewriting one invalidates whatever was read before.
2581
+ @persisted_index_stats = nil
2582
+
996
2583
  # Scan existing unit JSON files (exclude _index.json)
997
2584
  index = Dir[type_dir.join('*.json')].filter_map do |file|
998
2585
  next if File.basename(file) == '_index.json'
999
2586
 
1000
- data = JSON.parse(File.read(file))
2587
+ data = JSON.parse(AtomicFile.read(file))
1001
2588
  {
1002
2589
  identifier: data['identifier'],
1003
2590
  file_path: data['file_path'],
@@ -1010,15 +2597,25 @@ module Woods
1010
2597
  }
1011
2598
  end
1012
2599
 
1013
- File.write(
2600
+ AtomicFile.write(
1014
2601
  type_dir.join('_index.json'),
1015
- json_serialize(index)
2602
+ json_serialize(index),
2603
+ durable: payload_writes_durable?
1016
2604
  )
1017
2605
  end
1018
2606
 
1019
2607
  # Token estimate for a unit parsed back from JSON, mirroring
1020
2608
  # ExtractedUnit#estimated_tokens (see docs/TOKEN_BENCHMARK.md).
1021
2609
  #
2610
+ # `JSON.generate`, not `Hash#to_json`: with ActiveSupport loaded the
2611
+ # latter applies HTML-safe escaping, rendering `>` as `\\u003e` — six
2612
+ # characters where the file on disk has one. A model with
2613
+ # `scope :recent, -> { ... }` in its metadata therefore estimated one
2614
+ # token higher here than {ExtractedUnit#estimated_tokens} did over the
2615
+ # same content, so a full and an incremental run disagreed on that unit's
2616
+ # `_index.json` entry (#164). Both sides now measure the serialization
2617
+ # {#json_serialize} actually writes.
2618
+ #
1022
2619
  # @param data [Hash] Parsed unit JSON (string keys)
1023
2620
  # @return [Integer]
1024
2621
  def estimated_tokens_from(data)
@@ -1026,7 +2623,7 @@ module Woods
1026
2623
  metadata = data['metadata'] || {}
1027
2624
 
1028
2625
  source_tokens = source ? TokenUtils.estimate_tokens(source) : 0
1029
- metadata_tokens = metadata.any? ? TokenUtils.estimate_tokens(metadata.to_json) : 0
2626
+ metadata_tokens = metadata.any? ? TokenUtils.estimate_tokens(JSON.generate(metadata)) : 0
1030
2627
  source_tokens + metadata_tokens
1031
2628
  end
1032
2629
 
@@ -1087,78 +2684,1026 @@ module Woods
1087
2684
  # Incremental Re-extraction
1088
2685
  # ──────────────────────────────────────────────────────────────────────
1089
2686
 
2687
+ # Extractor instances are reused across a single incremental run, the way
2688
+ # full extraction uses one instance per type. Several of them do real
2689
+ # setup work in `initialize` (route-helper maps, routes maps), so a fresh
2690
+ # instance per file would make an N-file change N times more expensive.
2691
+ #
2692
+ # @param key [Symbol] key into {EXTRACTORS}
2693
+ # @return [Object, nil] the memoized extractor instance
2694
+ def extractor_for(key)
2695
+ @incremental_extractors ||= {}
2696
+ return @incremental_extractors[key] if @incremental_extractors.key?(key)
2697
+
2698
+ @incremental_extractors[key] = EXTRACTORS[key]&.new
2699
+ rescue StandardError => e
2700
+ Rails.logger.warn "[Woods] Could not build #{key} extractor: #{e.message}"
2701
+ @incremental_extractors[key] = nil
2702
+ end
2703
+
2704
+ # ActiveRecord model names, needed by PoroExtractor to tell a plain class
2705
+ # under app/models apart from a persisted model. Computed once per run.
2706
+ #
2707
+ # @return [Set<String>]
2708
+ def active_record_names
2709
+ @active_record_names ||=
2710
+ if defined?(ActiveRecord::Base)
2711
+ ActiveRecord::Base.descendants.filter_map(&:name).to_set
2712
+ else
2713
+ Set.new
2714
+ end
2715
+ end
2716
+
2717
+ # Re-extract every file-based unit defined by the changed paths that still
2718
+ # exist on disk, and prune the ones those paths no longer define.
2719
+ #
2720
+ # This is the fix for #164 gap 1 (a path the index has never seen routed
2721
+ # nowhere, so new files were silently ignored) and the per-path half of
2722
+ # gap 3 (a file that defines several units — a `.rake` file with multiple
2723
+ # tasks, an i18n YAML — could only ever resolve to one identifier).
2724
+ # Reconciling the whole path at once means a task deleted from a
2725
+ # multi-task file is removed rather than left behind.
2726
+ #
2727
+ # Removal is scoped to the unit types the matching rules could have
2728
+ # produced, so a class-based unit sharing the path (the `User` model unit
2729
+ # for `app/models/user.rb`) is never collaterally deleted here.
2730
+ #
2731
+ # @param change_set [ChangeSet]
2732
+ # @param affected_types [Set<Symbol>] out-param of touched extractor keys
2733
+ # @return [Set<String>] identifiers written or removed
2734
+ def reconcile_changed_paths(change_set, affected_types)
2735
+ dispatcher = PathDispatcher.new
2736
+ touched = Set.new
2737
+
2738
+ change_set.existing_paths.each do |absolute_path|
2739
+ rules = dispatcher.file_rules_for(change_set.relativize(absolute_path))
2740
+ next if rules.empty?
2741
+
2742
+ produced = Set.new
2743
+ # A rule whose extraction *raised* tells us nothing about what the path
2744
+ # defines, so it must not license pruning what the path defined before.
2745
+ raised = false
2746
+ rules.each do |rule|
2747
+ units = extract_with_rule(rule, absolute_path)
2748
+ if units.nil?
2749
+ raised = true
2750
+ next
2751
+ end
2752
+
2753
+ # (identifier, type) pairs, not bare identifiers: two rules can
2754
+ # claim one path (policies/pundit_policies) and mint the same
2755
+ # identifier for different unit types. When one of them stops
2756
+ # producing, an identifier-keyed set would let the survivor shield
2757
+ # the stale sibling-type node from the prune below (the #225 shape,
2758
+ # one method over — CORE-1).
2759
+ produced.merge(units.map { |unit| [unit.identifier, unit.type] })
2760
+ touched.merge(register_and_write(rule.extractor_key, units, affected_types))
2761
+ end
2762
+
2763
+ next if raised
2764
+
2765
+ touched.merge(
2766
+ prune_path_leftovers(absolute_path, rules, produced, affected_types)
2767
+ )
2768
+ end
2769
+
2770
+ touched
2771
+ end
2772
+
2773
+ # Run one {PathDispatcher::Rule} against one file.
2774
+ #
2775
+ # @param rule [PathDispatcher::Rule]
2776
+ # @param absolute_path [String]
2777
+ # @return [Array<ExtractedUnit>, nil] the units the path defines, or nil
2778
+ # when the attempt says nothing about what it defines (see the rescue)
2779
+ def extract_with_rule(rule, absolute_path)
2780
+ extractor = extractor_for(rule.extractor_key)
2781
+ # nil, not [] — the same distinction the rescue below defends (#198).
2782
+ # {#extractor_for} rescues a raising *constructor* and memoizes nil, and
2783
+ # nil fails the respond_to? guard — so a failed construction used to fall
2784
+ # through to the [] "this path defines nothing any more" answer, and one
2785
+ # broken constructor licensed pruning every previously-registered unit
2786
+ # on every changed path of that type, with the generation bumped over
2787
+ # the loss. Construction failure tells us nothing about the path; only
2788
+ # a genuinely constructed extractor that lacks the method earns the [].
2789
+ return nil if extractor.nil?
2790
+ return [] unless extractor.respond_to?(rule.method_name)
2791
+
2792
+ result =
2793
+ if rule.extractor_key == :poros
2794
+ # PoroExtractor needs the AR name set to reject persisted models;
2795
+ # its default is an empty set, which would misfile every model
2796
+ # under app/models as a PORO on the incremental path.
2797
+ extractor.public_send(rule.method_name, absolute_path, ar_names: active_record_names)
2798
+ else
2799
+ extractor.public_send(rule.method_name, absolute_path)
2800
+ end
2801
+
2802
+ Array(result).compact
2803
+ rescue StandardError => e
2804
+ Rails.logger.warn "[Woods] #{rule.extractor_key} re-extraction of #{absolute_path} failed: #{e.message}"
2805
+ # `nil`, not `[]`. The caller treats an empty result as "this path defines
2806
+ # nothing any more" and prunes the units previously registered to it — so
2807
+ # returning [] here turns a *transient* failure (a watcher batch catching
2808
+ # an editor mid-write, an encoding hiccup) into silent deletion of a
2809
+ # perfectly good unit, which nothing restores until that file is next
2810
+ # touched. "Extraction raised" and "extracted successfully, defines
2811
+ # nothing" are different answers and only the second licenses a prune.
2812
+ nil
2813
+ end
2814
+
2815
+ # Remove units the graph still attributes to a path that the path no
2816
+ # longer produces — but only for the types the matching rules cover.
2817
+ #
2818
+ # @param absolute_path [String]
2819
+ # @param rules [Array<PathDispatcher::Rule>]
2820
+ # @param produced [Set<Array(String, Symbol)>] (identifier, type) pairs
2821
+ # just extracted from the path
2822
+ # @param affected_types [Set<Symbol>]
2823
+ # @return [Set<String>] identifiers removed
2824
+ def prune_path_leftovers(absolute_path, rules, produced, affected_types)
2825
+ covered_keys = rules.to_set(&:extractor_key)
2826
+ removed = Set.new
2827
+
2828
+ @dependency_graph.units_for_path(absolute_path).each do |identifier, node_type|
2829
+ next if produced.include?([identifier, node_type])
2830
+ next unless covered_keys.include?(TYPE_TO_EXTRACTOR_KEY[node_type])
2831
+
2832
+ removed.add(identifier) if remove_unit(identifier, affected_types, type: node_type)
2833
+ end
2834
+
2835
+ removed
2836
+ end
2837
+
2838
+ # Reconcile class-based types against their runtime discovery sets.
2839
+ #
2840
+ # Models, controllers, mailers, components and channels are discovered by
2841
+ # walking descendants, not by globbing files, so a class added since the
2842
+ # last extraction is invisible to any path-based dispatch. Comparing each
2843
+ # extractor's `discoverable_classes` against the identifiers already in
2844
+ # the graph finds exactly those additions, using the same discovery code
2845
+ # a full extraction uses.
2846
+ #
2847
+ # Removals are handled here too, but only against a *complete* discovery
2848
+ # set — see {#stale_class_based_units} for why that qualifier carries the
2849
+ # whole safety argument.
2850
+ #
2851
+ # @param affected_types [Set<Symbol>]
2852
+ # @return [Set<String>] identifiers added or removed
2853
+ def reconcile_class_based_types(affected_types, except: nil)
2854
+ touched = Set.new
2855
+ excluded = except&.to_set || Set.new
2856
+
2857
+ CLASS_BASED_DISCOVERY.each do |key, spec|
2858
+ extractor = extractor_for(key)
2859
+ next unless extractor.respond_to?(:discoverable_classes)
2860
+
2861
+ discovered = extractor.discoverable_classes.reject { |k| k.name.nil? }
2862
+ known = Array(spec[:types] || spec[:type])
2863
+ .flat_map { |type| @dependency_graph.units_of_type(type) }.to_set
2864
+
2865
+ touched.merge(add_discovered_classes(key, spec, discovered, known, excluded, affected_types))
2866
+ next if spec[:reconcile_removals] == false
2867
+
2868
+ touched.merge(remove_stale_classes(spec, discovered, known, affected_types))
2869
+ end
2870
+
2871
+ touched
2872
+ end
2873
+
2874
+ # Pruned class-based identifiers the tree still governs, and that the
2875
+ # second reconciliation pass may therefore re-add.
2876
+ #
2877
+ # The class-based *move* shape — the file moved, the constant did not
2878
+ # (M1) — prunes the unit for the vanished old path while the class stays
2879
+ # in the discovery set; the move target is a changed file the active
2880
+ # loader governs for exactly that constant, so re-adding produces the
2881
+ # unit a full extraction produces. The *deletion* shape must stay
2882
+ # pruned: without a reload a constant outlives the file that defined
2883
+ # it, so liveness alone proves nothing.
2884
+ #
2885
+ # Identity is loader-derived, not textual: {SourceNesting#
2886
+ # governed_class_name} returns the constant path the active Zeitwerk
2887
+ # loader expects the changed file to define (its inflector, ignores,
2888
+ # and root namespaces decide, via +cpath_expected_at+), gated on the
2889
+ # file actually declaring it. A loader non-claim — an unmanaged or
2890
+ # declined path — is authoritative and re-adds nothing, exactly the
2891
+ # governed-naming contract. That kills the two resurrection shapes a
2892
+ # demodulized class-name regex allowed: a changed file declaring
2893
+ # `Public::User` no longer resurrects a pruned `Admin::User`, and a
2894
+ # file whose comments or string literals mention `class User` no
2895
+ # longer resurrects anything.
2896
+ #
2897
+ # @param pruned [Set<String>] identifiers removed by {#prune_vanished_units}
2898
+ # @param change_set [ChangeSet]
2899
+ # @return [Set<String>] identifiers safe to re-add this run
2900
+ def readdable_pruned_classes(pruned, change_set)
2901
+ readdable = Set.new
2902
+ return readdable if pruned.empty?
2903
+
2904
+ claimed = changed_governed_names(change_set)
2905
+ return readdable if claimed.empty?
2906
+
2907
+ CLASS_BASED_DISCOVERY.each_key do |key|
2908
+ extractor = extractor_for(key)
2909
+ next unless extractor.respond_to?(:discoverable_classes)
2910
+
2911
+ extractor.discoverable_classes.each do |klass|
2912
+ next if klass.name.nil? || !pruned.include?(klass.name)
2913
+
2914
+ readdable.add(klass.name) if claimed.include?(klass.name)
2915
+ end
2916
+ end
2917
+
2918
+ readdable
2919
+ rescue StandardError => e
2920
+ # A failure here must not turn a deletion into a resurrection: the
2921
+ # empty set keeps every pruned identifier excluded, which is the
2922
+ # pre-M1 behavior.
2923
+ Rails.logger.warn "[Woods] Could not determine re-addable pruned classes: #{e.message}"
2924
+ Set.new
2925
+ end
2926
+
2927
+ # Governed constant names of the change set's still-existing Ruby
2928
+ # files. The shared governed-naming helper consults the active Zeitwerk
2929
+ # loader and returns nil for anything it does not claim, so unmanaged
2930
+ # paths (anything outside the loader's roots, under an old Zeitwerk
2931
+ # without +cpath_expected_at+, or a declined file) contribute nothing.
2932
+ #
2933
+ # @param change_set [ChangeSet]
2934
+ # @return [Set<String>] governed constant names of changed files
2935
+ def changed_governed_names(change_set)
2936
+ change_set.existing_paths.filter_map do |path|
2937
+ path = path.to_s
2938
+ next unless path.end_with?('.rb')
2939
+
2940
+ governed_class_name(path, File.read(path))
2941
+ end.compact.to_set
2942
+ end
2943
+
2944
+ def add_discovered_classes(key, spec, discovered, known, excluded, affected_types)
2945
+ new_classes = discovered.reject { |k| known.include?(k.name) || excluded.include?(k.name) }
2946
+ return Set.new if new_classes.empty?
2947
+
2948
+ units = new_classes.filter_map do |klass|
2949
+ extractor_for(key).public_send(spec[:method], klass)
2950
+ rescue StandardError => e
2951
+ Rails.logger.warn "[Woods] #{key} extraction of #{klass} failed: #{e.message}"
2952
+ nil
2953
+ end
2954
+
2955
+ register_and_write(key, units, affected_types)
2956
+ end
2957
+
2958
+ def remove_stale_classes(spec, discovered, known, affected_types)
2959
+ stale = stale_class_based_units(spec[:type], discovered, known)
2960
+ return Set.new if stale.empty?
2961
+
2962
+ Rails.logger.info "[Woods] removing #{stale.size} #{spec[:type]} unit(s) whose class no longer exists"
2963
+ stale.each_with_object(Set.new) do |identifier, removed|
2964
+ removed.add(identifier) if remove_unit(identifier, affected_types, type: spec[:type])
2965
+ end
2966
+ end
2967
+
2968
+ # Class-based units the graph still holds that a full extraction would not
2969
+ # produce.
2970
+ #
2971
+ # {#prune_vanished_units} keys on the source file being gone, which cannot
2972
+ # see this case: a class removed from a file that still exists leaves no
2973
+ # missing path, and class-based units register a *convention* path derived
2974
+ # from the constant name, so a class defined somewhere unconventional was
2975
+ # never attributed to the file it actually lived in. Two models in one
2976
+ # `.rb`, one of them deleted, and the survivor's own re-extraction says
2977
+ # nothing about the other. Nothing else in the run removes it, so it
2978
+ # outlives every subsequent incremental — a permanent divergence from a
2979
+ # full run, not a transient one.
2980
+ #
2981
+ # For all six class-based extractors `extract_all` is literally
2982
+ # `discoverable_classes.map { ... }.compact`, so absence from that set is
2983
+ # exactly "a full extraction would not produce this" — the equivalence the
2984
+ # incremental path is held to.
2985
+ #
2986
+ # The `@eager_load_complete` gate is the whole safety argument, and it is
2987
+ # why this is not simply the inverse of the addition pass:
2988
+ #
2989
+ # * **A partial eager load.** The documented NameError fallback loads only
2990
+ # `EXTRACTION_DIRECTORIES`, so descendants are known-incomplete and the
2991
+ # difference here would be most of the app. Deleting by the type is far
2992
+ # worse than a stale unit, so a partial load removes nothing.
2993
+ # * **A constant outliving its file.** A resident daemon that has not
2994
+ # reloaded still holds a deleted class as a descendant, so it is *in* the
2995
+ # set and not stale — which is correct for that process, and the
2996
+ # subsequent reload is what makes it removable.
2997
+ #
2998
+ # @param type [Symbol] unit type
2999
+ # @param discovered [Array<Class>] the extractor's current discovery set
3000
+ # @param known [Set<String>] identifiers of that type already in the graph
3001
+ # @return [Array<String>] identifiers to remove
3002
+ def stale_class_based_units(type, discovered, known)
3003
+ return [] unless @eager_load_complete
3004
+
3005
+ live = discovered.to_set(&:name)
3006
+ known.reject { |identifier| live.include?(identifier) }
3007
+ # Not redundant with `units_of_type`: an identifier can be listed in
3008
+ # this type's index while also naming a unit of another type, and
3009
+ # the caller removes by (identifier, type), so it has to be told
3010
+ # which node it is allowed to take.
3011
+ .select { |identifier| @dependency_graph.node_types(identifier).include?(type) }
3012
+ end
3013
+
3014
+ # Re-run whole-app extractors whose trigger paths changed, replacing that
3015
+ # unit type wholesale.
3016
+ #
3017
+ # These extractors have no per-file entry point — a route unit is derived
3018
+ # from `Rails.application.routes`, the middleware unit from the live
3019
+ # stack, events from a two-pass scan of all of `app/`. Before this,
3020
+ # incremental runs skipped them entirely, so a routes-only change
3021
+ # triggered a run that re-extracted nothing while still rewriting the
3022
+ # manifest (#164 gap 4). In an already-booted process re-running them is
3023
+ # cheap, which is what makes wholesale replacement the right shape.
3024
+ #
3025
+ # @param change_set [ChangeSet]
3026
+ # @param affected_types [Set<Symbol>]
3027
+ # @return [Set<String>] identifiers written or removed
3028
+ def rerun_whole_app_extractors(change_set, affected_types)
3029
+ keys = PathDispatcher.new.whole_app_keys_for_all(change_set.relative_paths)
3030
+ # The dispatch rules are configuration-blind (memoized per-process), so
3031
+ # the configuration gate applies here — a knob-off host must not re-run
3032
+ # an extractor whose units a full extraction of the same tree would not
3033
+ # produce. See {#skip_by_configuration?}.
3034
+ keys = keys.reject { |key| skip_by_configuration?(key) }.to_set
3035
+ return Set.new if keys.empty?
3036
+
3037
+ keys += ROUTE_CONSUMER_EXTRACTORS if keys.include?(:routes)
3038
+ # A routes re-run replaces every controller, and a flow document
3039
+ # carries the route itself, which no dependency edge connects to the
3040
+ # controller. Nothing about that is reachable by a graph walk, so the
3041
+ # run drops its flow scope and reassembles every touched controller.
3042
+ @flow_scope = nil if keys.include?(:routes)
3043
+
3044
+ keys.each_with_object(Set.new) do |key, touched|
3045
+ touched.merge(replace_type_wholesale(key, affected_types))
3046
+ end
3047
+ end
3048
+
3049
+ # Replace every unit an extractor owns with a fresh extraction.
3050
+ #
3051
+ # Units of the extractor's types that the fresh run no longer produces are
3052
+ # removed, which is what makes this a replacement rather than an upsert —
3053
+ # subject to the same eager-load gate the reconciler applies, see
3054
+ # {#remove_replaced_units}.
3055
+ #
3056
+ # Fail closed (M8): the rescue exists so a re-run that learned nothing —
3057
+ # `extract_all` itself raising — costs the run nothing. It must not swallow
3058
+ # a failure that lands AFTER the replacement started mutating state a
3059
+ # published generation would carry: {#register_and_write} registers the
3060
+ # graph node before writing the unit file, and {#remove_unit_of_type} rm_f's
3061
+ # the file before dropping the graph node, so a raise in either window
3062
+ # leaves the graph and the payload directory disagreeing. Swallowed, the run
3063
+ # went on to publish a generation whose dependency_graph.json held nodes
3064
+ # with no unit file — `dependencies`/`dependents` reported `found: true`
3065
+ # while lookup returned nil. The counter from {#note_wholesale_mutation}
3066
+ # turns any such failure into a re-raise: the run aborts before
3067
+ # {#publish_generation}, and the preceding generation stays resolved, the
3068
+ # same posture the flow family takes on the full path.
3069
+ #
3070
+ # Two counter rules make that decision sound. The marker is placed BEFORE
3071
+ # each mutation, because registration itself can fail mid-mutation (a
3072
+ # malformed dependency raises after the node is inserted) and a rm_f can
3073
+ # fail part-way; a marker placed after the fact would miss the window it
3074
+ # exists for, at the cost of a conservative abort when the marked mutation
3075
+ # then fails before changing anything. And the counter is reset BEFORE
3076
+ # `extract_all`, not after: {#register_and_write} is shared with the
3077
+ # reconcile paths that run earlier in the same pass, and one of those
3078
+ # leaving the counter positive must not turn a later, mutation-free
3079
+ # wholesale failure into an abort — only the current key's wholesale pass
3080
+ # contributes to the decision.
3081
+ #
3082
+ # @param key [Symbol] extractor key
3083
+ # @param affected_types [Set<Symbol>]
3084
+ # @return [Set<String>] identifiers written or removed
3085
+ # @raise [Woods::ExtractionError] when the replacement failed after
3086
+ # mutating durable state
3087
+ def replace_type_wholesale(key, affected_types)
3088
+ extractor = extractor_for(key)
3089
+ return Set.new unless extractor.respond_to?(:extract_all)
3090
+
3091
+ @wholesale_mutations = 0
3092
+ units = Array(extractor.extract_all).compact.uniq(&:identifier)
3093
+ Rails.logger.info "[Woods] Re-ran #{key} wholesale: #{units.size} units"
3094
+
3095
+ touched = register_and_write(key, units, affected_types)
3096
+ touched.merge(remove_replaced_units(key, units, affected_types))
3097
+ rescue StandardError => e
3098
+ if @wholesale_mutations.to_i.positive?
3099
+ raise Woods::ExtractionError, <<~MSG.tr("\n", ' ').strip
3100
+ Wholesale re-run of #{key} failed after this run had already
3101
+ registered, written, or removed #{@wholesale_mutations} unit
3102
+ artifact(s) (#{e.class}: #{e.message}). Continuing would publish a
3103
+ generation whose dependency graph disagrees with the unit files on
3104
+ disk, so the run aborts before publication and the previous
3105
+ generation stays resolved.
3106
+ MSG
3107
+ end
3108
+
3109
+ Rails.logger.error "[Woods] Wholesale re-run of #{key} failed: #{e.message}"
3110
+ Set.new
3111
+ end
3112
+
3113
+ # Record that the in-flight wholesale replacement is about to mutate state
3114
+ # a published generation would carry: a graph registration, a unit-file
3115
+ # write, or a unit-file removal. Callers mark BEFORE mutating — register
3116
+ # itself can raise mid-mutation (a malformed dependency raises after the
3117
+ # node is inserted), so a marker placed after the fact can miss the very
3118
+ # window it exists for. The cost is a conservative abort when the marked
3119
+ # mutation then fails before changing anything; that direction is safe.
3120
+ # {#replace_type_wholesale} resets the counter to 0 before `extract_all` —
3121
+ # so only the current key's wholesale pass contributes, never the earlier
3122
+ # reconcile passes that share {#register_and_write} — and re-raises from
3123
+ # its rescue when the count is non-zero. Increments from non-wholesale
3124
+ # callers of {#register_and_write} are inert outside that window: nothing
3125
+ # reads the counter between resets.
3126
+ #
3127
+ # @return [void]
3128
+ def note_wholesale_mutation
3129
+ @wholesale_mutations = @wholesale_mutations.to_i + 1
3130
+ end
3131
+
3132
+ # The removal half of a wholesale replacement: drop every unit of `key`'s
3133
+ # types that the fresh run no longer produced.
3134
+ #
3135
+ # Gated exactly the way {#stale_class_based_units} is, and for the same
3136
+ # reason (#198): for the extractors in {CLASS_BASED_DISCOVERY},
3137
+ # `extract_all` is `discoverable_classes.map { ... }` — a function of the
3138
+ # runtime descendants — so on the documented NameError-fallback boot the
3139
+ # fresh set is known-partial and "absent from it" does not mean "deleted".
3140
+ # The reconciler's gate ("the whole safety argument") never protected this
3141
+ # path, and {ROUTE_CONSUMER_EXTRACTORS} routes four descendants-discovered
3142
+ # types through here on *every* routes change — so one routes edit on a
3143
+ # fallback boot mass-deleted most controller units. Registration above is
3144
+ # not gated: additions are safe against a partial set (the same asymmetry
3145
+ # {#reconcile_class_based_types} relies on); only removal needs the
3146
+ # complete one.
3147
+ #
3148
+ # File-derived whole-app types (routes, view_templates, events, ...) keep
3149
+ # unconditional removal — their `extract_all` globs files or introspects
3150
+ # structures that do not depend on eager loading, so their fresh set is
3151
+ # authoritative on any boot.
3152
+ #
3153
+ # The skip is warned, not silent: until a run with a clean boot removes
3154
+ # them, the type may hold stale units, and an operator chasing a ghost
3155
+ # unit needs to know which run declined to delete it and why.
3156
+ #
3157
+ # `fresh` is computed PER unit_type, not once for the whole extractor key
3158
+ # (#225). An identifier is not unique across types — a Scenic view and a
3159
+ # factory can both be named `reports` — so a single identifier-level
3160
+ # `fresh` set let a factories re-run's `database_view` deletion pass see
3161
+ # the surviving `factories` identifier as "still fresh" and skip removing
3162
+ # nothing, then later delete both nodes anyway once `remove_unit` was
3163
+ # called without `type:` (the same call also had to be fixed — see
3164
+ # below). A per-type set also covers a unit reclassified between two
3165
+ # types this same key owns (GraphQL's four): the old type's identifier is
3166
+ # absent from ITS OWN fresh set even though the identifier as a whole is
3167
+ # still "fresh" under the new type, so the stale old-type node is
3168
+ # correctly dropped instead of surviving because the identifier still
3169
+ # exists somewhere in the extractor's output.
3170
+ #
3171
+ # `remove_unit` is called WITH `type: unit_type` for the same reason:
3172
+ # `DependencyGraph#remove`'s own doc warns that a typeless removal fans
3173
+ # over every type registered under the identifier, so calling it here
3174
+ # without `type:` deleted the sibling type's node and JSON file too —
3175
+ # reproduced live as a factories re-run deleting a same-named Scenic-view
3176
+ # unit.
3177
+ #
3178
+ # @param key [Symbol] extractor key
3179
+ # @param units [Array<ExtractedUnit>] the fresh extraction
3180
+ # @param affected_types [Set<Symbol>]
3181
+ # @return [Set<String>] identifiers removed
3182
+ def remove_replaced_units(key, units, affected_types)
3183
+ if CLASS_BASED_DISCOVERY.key?(key) && !@eager_load_complete
3184
+ Rails.logger.warn(
3185
+ "[Woods] Skipping stale-unit removal for #{key}: the eager load was incomplete, " \
3186
+ 'so its discovery set is known-partial — the type may hold stale units until a clean boot'
3187
+ )
3188
+ return Set.new
3189
+ end
3190
+
3191
+ EXTRACTOR_KEY_TO_TYPES.fetch(key, []).each_with_object(Set.new) do |unit_type, removed|
3192
+ fresh = units.select { |u| u.type == unit_type }.to_set(&:identifier)
3193
+ (@dependency_graph.units_of_type(unit_type) - fresh.to_a).each do |stale|
3194
+ removed.add(stale) if remove_unit(stale, affected_types, type: unit_type)
3195
+ end
3196
+ end
3197
+ end
3198
+
3199
+ # Prune units whose source file no longer exists (#164 gap 2).
3200
+ #
3201
+ # Two inputs, with deliberately different authority:
3202
+ #
3203
+ # * **The change set.** A path the caller reports as changed which is no
3204
+ # longer on disk is authoritative — every unit the graph attributes to
3205
+ # it goes, whatever its type. This is the path that handles a deleted
3206
+ # model or controller, and the old side of a rename (git's
3207
+ # `--no-renames` semantics).
3208
+ # * **A sweep** over registered paths, for callers whose change set is
3209
+ # incomplete: a git diff that omits deletions, a watcher that missed an
3210
+ # unlink, a branch switch. The sweep is a heuristic, so it is bounded
3211
+ # twice — see {#sweep_candidates} and the class-based exclusion below.
3212
+ #
3213
+ # Only paths under `Rails.root` are considered either way. Framework
3214
+ # units point at gem paths, and an index restored from a CI artifact can
3215
+ # carry paths produced under a different root; neither is a deletion.
3216
+ #
3217
+ # @param change_set [ChangeSet]
3218
+ # @param affected_types [Set<Symbol>]
3219
+ # @return [Set<String>] identifiers removed
3220
+ def prune_vanished_units(change_set, affected_types)
3221
+ removed = prune_paths(change_set.missing_paths, affected_types, class_based: true)
3222
+ removed.merge(prune_paths(sweep_candidates(change_set), affected_types, class_based: false))
3223
+
3224
+ Rails.logger.info "[Woods] Pruned #{removed.size} unit(s) whose source file is gone" if removed.any?
3225
+ removed
3226
+ end
3227
+
3228
+ # Registered paths that have vanished and that the sweep is allowed to act
3229
+ # on: under Rails.root, gone from disk, and claimed by a file dispatch
3230
+ # rule.
3231
+ #
3232
+ # The file-rule bound exists because some units point at a *nominal* path
3233
+ # rather than a source file — `BehavioralProfile` names
3234
+ # `config/application.rb`, which no rule claims — and sweeping those would
3235
+ # delete units a full extraction still produces.
3236
+ #
3237
+ # @param change_set [ChangeSet]
3238
+ # @return [Array<String>] absolute paths
3239
+ def sweep_candidates(change_set)
3240
+ root_prefix = "#{Rails.root}/"
3241
+ dispatcher = PathDispatcher.new
3242
+ already_named = change_set.missing_paths.to_set
3243
+
3244
+ @dependency_graph.registered_paths.reject do |path|
3245
+ already_named.include?(path) ||
3246
+ !path.to_s.start_with?(root_prefix) ||
3247
+ File.exist?(path) ||
3248
+ dispatcher.file_rules_for(change_set.relativize(path)).empty?
3249
+ end
3250
+ end
3251
+
3252
+ # Remove every unit the graph attributes to each of `paths`.
3253
+ #
3254
+ # @param paths [Enumerable<String>] absolute paths believed to be gone
3255
+ # @param affected_types [Set<Symbol>]
3256
+ # @param class_based [Boolean] whether class-based units may be pruned
3257
+ # @return [Set<String>] identifiers removed
3258
+ def prune_paths(paths, affected_types, class_based:)
3259
+ root_prefix = "#{Rails.root}/"
3260
+
3261
+ paths.each_with_object(Set.new) do |path, removed|
3262
+ next unless path.to_s.start_with?(root_prefix)
3263
+ next if File.exist?(path)
3264
+
3265
+ @dependency_graph.units_for_path(path).each do |identifier, type|
3266
+ next if !class_based && convention_path_unit?(type)
3267
+
3268
+ removed.add(identifier) if remove_unit(identifier, affected_types, type: type)
3269
+ end
3270
+ end
3271
+ end
3272
+
3273
+ # Does this unit's `file_path` name a *convention* that need not exist?
3274
+ #
3275
+ # The sweep deletes units whose file is gone, which is wrong for anything
3276
+ # that derives its path from a constant name rather than from a file it was
3277
+ # read out of. Class-based types have always been excluded for that reason.
3278
+ #
3279
+ # Today the predicate keys on {CLASS_BASED} and nothing else (L4): the six
3280
+ # class-based families — models, controllers, components, view components,
3281
+ # mailers, channels. GRAPHQL_TYPES are deliberately NOT spared. An earlier
3282
+ # version of this comment narrated GraphQL units being spared through a
3283
+ # `source_file_for_class` convention-path fallback, but that fallback now
3284
+ # returns nil rather than fabricating a convention path
3285
+ # (`graphql_extractor.rb`), a runtime-defined GraphQL unit registers no
3286
+ # path at all ({DependencyGraph#register} skips nil), and the static file
3287
+ # pass records the real file it read — so the sweep's own bounds decide,
3288
+ # and this predicate has no say in GraphQL either way.
3289
+ #
3290
+ # The original case: on Rails < 7.1 `ActiveRecord::SchemaMigration` and
3291
+ # `InternalMetadata` are real `ActiveRecord::Base` descendants a full
3292
+ # extraction emits with `app/models/active_record/schema_migration.rb` as
3293
+ # their path — a file no application has, and one the PORO rule *does*
3294
+ # claim, so the sweep's file-rule bound doesn't cover it and this check has
3295
+ # to. Deleting such a unit requires the caller to name the path; the sweep
3296
+ # never infers it.
3297
+ #
3298
+ # KNOWN COST (B-070 / #171): this keys on unit *type*, but the property is
3299
+ # per-unit — CLASS_BASED is also true of units whose recorded path has
3300
+ # since been moved elsewhere, which is exactly the move shape
3301
+ # {#readdable_pruned_classes} exists to re-add in the same run. For
3302
+ # deletions the cost stays: a class-based unit whose file is gone survives
3303
+ # an unnamed-path sweep. Named-path deletion still works, so
3304
+ # `woods:incremental` on a git diff is unaffected; the exposed caller is the
3305
+ # daemon's catch-up, which runs an empty change set precisely because
3306
+ # deletions leave no mtime.
3307
+ #
3308
+ # Two obvious fixes were tried and **both are wrong** — do not re-attempt
3309
+ # them without reading this:
3310
+ #
3311
+ # 1. *Compare the recorded path against the convention path derived from the
3312
+ # constant.* `Types::ForgottenType` conventionally lives at
3313
+ # `app/graphql/types/forgotten_type.rb`, which IS its convention path — so
3314
+ # a conventionally-named file-defined type is indistinguishable from a
3315
+ # runtime-defined one, and the common case stays spared. Disproven by
3316
+ # `spec/integration/incremental_equivalence_spec.rb`'s pending example.
3317
+ # 2. *Let the sweep prune GraphQL and rely on the reconciler's addition half
3318
+ # to re-add whatever the schema still holds.* This is what `except: pruned`
3319
+ # exists to prevent (see the comment at its call site): without a reload a
3320
+ # constant outlives its deleted file, and the re-add resurrects the deleted
3321
+ # unit against a path that no longer exists — permanently, since the sweep
3322
+ # then spares it.
3323
+ #
3324
+ # What would actually work is provenance: record at extraction time whether a
3325
+ # source file existed for the unit and spare only the units that never had
3326
+ # one. That needs the graph node to carry the flag, so it is a serialization
3327
+ # change rather than a predicate tweak.
3328
+ #
3329
+ # @param type [Symbol, nil] the unit type the caller is about to remove
3330
+ # @return [Boolean]
3331
+ def convention_path_unit?(type)
3332
+ CLASS_BASED.key?(type)
3333
+ end
3334
+
3335
+ # Register a batch of freshly-extracted units and write their JSON.
3336
+ #
3337
+ # Registration happens BEFORE path normalization — the graph's file map
3338
+ # stores absolute paths (that is what changed files are matched against),
3339
+ # exactly as full extraction registers in Phase 1 and only normalizes in
3340
+ # Phase 4.5. Unit JSON carries the relative path.
3341
+ #
3342
+ # @param extractor_key [Symbol]
3343
+ # @param units [Array<ExtractedUnit>]
3344
+ # @param affected_types [Set<Symbol>]
3345
+ # @return [Set<String>] identifiers written
3346
+ def register_and_write(extractor_key, units, affected_types)
3347
+ units = Array(units).compact
3348
+ return Set.new if units.empty?
3349
+
3350
+ affected_types&.add(extractor_key)
3351
+ type_dir = payload_dir.join(extractor_key.to_s)
3352
+ FileUtils.mkdir_p(type_dir)
3353
+
3354
+ units.each_with_object(Set.new) do |unit, written|
3355
+ annotate_package(unit)
3356
+ mark_dependents_dirty(unit.identifier)
3357
+ # Marked BEFORE registration: DependencyGraph#register inserts the
3358
+ # node before it iterates the unit's dependencies, so a malformed
3359
+ # dependency raises with the graph already mutated — a marker placed
3360
+ # after the call would never run, and the phantom would ship. The
3361
+ # cost of this ordering is a conservative abort when registration
3362
+ # fails before mutating anything; that is the safe direction to err
3363
+ # in. See {#note_wholesale_mutation}.
3364
+ note_wholesale_mutation
3365
+ @dependency_graph.register(unit)
3366
+ mark_dependents_dirty(unit.identifier)
3367
+
3368
+ unit.file_path = normalize_file_path(unit.file_path)
3369
+ # Keyed by relative path, which is how batch_git_data keys its result.
3370
+ (@incremental_written ||= {})[unit.identifier] = unit.file_path
3371
+
3372
+ write_unit_file(type_dir.join(collision_safe_filename(unit.identifier)), unit)
3373
+ written.add(unit.identifier)
3374
+ end
3375
+ end
3376
+
3377
+ # Remove a unit from the graph and delete its JSON from the index.
3378
+ #
3379
+ # Callers that know which type they mean must say so. An identifier can
3380
+ # name units of several types (a Scenic view `reports` and a factory
3381
+ # `reports`), each with its own `<extractor_key>/<identifier>.json`, and
3382
+ # removing the identifier wholesale takes the sibling with it.
3383
+ #
3384
+ # @param identifier [String]
3385
+ # @param affected_types [Set<Symbol>]
3386
+ # @param type [Symbol, nil] remove only this type; without it, every type
3387
+ # registered under the identifier
3388
+ # @return [String, nil] the identifier when it existed and was removed
3389
+ def remove_unit(identifier, affected_types, type: nil)
3390
+ types = type ? [type] : @dependency_graph.node_types(identifier)
3391
+ types = types.select { |t| @dependency_graph.node(identifier, type: t) }
3392
+ return nil if types.empty?
3393
+
3394
+ # Before the removals: this reads the identifier's forward edges, which
3395
+ # go with the nodes.
3396
+ mark_dependents_dirty(identifier)
3397
+
3398
+ types.each { |t| remove_unit_of_type(identifier, t, affected_types) }
3399
+ Rails.logger.debug { "[Woods] Removed #{identifier}" }
3400
+ identifier
3401
+ end
3402
+
3403
+ # @param identifier [String]
3404
+ # @param type [Symbol]
3405
+ # @param affected_types [Set<Symbol>]
3406
+ # @return [void]
3407
+ def remove_unit_of_type(identifier, type, affected_types)
3408
+ extractor_key = TYPE_TO_EXTRACTOR_KEY[type]
3409
+
3410
+ if extractor_key
3411
+ affected_types&.add(extractor_key)
3412
+ path = payload_dir.join(extractor_key.to_s, collision_safe_filename(identifier))
3413
+ # Marked BEFORE the removal, same ordering as the write half: a
3414
+ # rm_f that fails part-way (or a failure immediately after it) must
3415
+ # not leave the counter at zero while the file is already gone. See
3416
+ # {#note_wholesale_mutation}.
3417
+ note_wholesale_mutation
3418
+ FileUtils.rm_f(path)
3419
+ end
3420
+
3421
+ @dependency_graph.remove(identifier, type: type)
3422
+ end
3423
+
3424
+ # Record that a unit's edges changed, so every target it points at (and
3425
+ # the unit itself) has its `dependents` list rewritten at the end of the
3426
+ # run.
3427
+ #
3428
+ # @param identifier [String]
3429
+ # @return [void]
3430
+ def mark_dependents_dirty(identifier)
3431
+ @dependents_dirty ||= Set.new
3432
+ @dependents_dirty.add(identifier)
3433
+ @dependency_graph.dependencies_of(identifier).each { |target| @dependents_dirty.add(target) }
3434
+ end
3435
+
3436
+ # Second pass over the unit JSON an incremental run touched, mirroring the
3437
+ # two full-extraction phases that operate on already-extracted units:
3438
+ # {#resolve_dependents} (Phase 2) and {#enrich_with_git_data} (Phase 4).
3439
+ #
3440
+ # Both were full-extraction-only, so incremental runs left `dependents`
3441
+ # and `metadata.git` frozen at whatever the last full run wrote —
3442
+ # divergence that compounds run over run in an incremental CI chain.
3443
+ #
3444
+ # @param affected_types [Set<Symbol>]
3445
+ # @return [void]
3446
+ def finalize_incremental_unit_json(affected_types)
3447
+ dependents_dirty = @dependents_dirty || Set.new
3448
+ git_dirty = @incremental_written || {}
3449
+ git_data = incremental_git_data(git_dirty.keys)
3450
+
3451
+ (dependents_dirty | git_dirty.keys).each do |identifier|
3452
+ rewrite_unit_json(identifier, affected_types,
3453
+ refresh_dependents: dependents_dirty.include?(identifier),
3454
+ git_data: git_dirty.key?(identifier) ? git_data : nil)
3455
+ end
3456
+ end
3457
+
3458
+ # Read one unit's JSON, apply the second-pass patches, and write it back
3459
+ # only if something actually changed.
3460
+ #
3461
+ # @param identifier [String]
3462
+ # @param affected_types [Set<Symbol>]
3463
+ # @param refresh_dependents [Boolean]
3464
+ # @param git_data [Hash{String => Hash}, nil] batch git data keyed by
3465
+ # relative path (per {#batch_git_data}), or nil when this identifier
3466
+ # isn't git-dirty this run
3467
+ # @return [void]
3468
+ def rewrite_unit_json(identifier, affected_types, refresh_dependents:, git_data:)
3469
+ @dependency_graph.node_types(identifier).each do |type|
3470
+ rewrite_unit_json_of_type(identifier, type, affected_types,
3471
+ refresh_dependents: refresh_dependents, git_data: git_data)
3472
+ end
3473
+ end
3474
+
3475
+ # One (identifier, type) pair's JSON. The `dependents` list is a property
3476
+ # of the identifier, not of the type, so every file the identifier owns
3477
+ # gets the same refreshed list — which is what a full extraction writes.
3478
+ #
3479
+ # Git metadata is NOT a property of the identifier (#225): a colliding
3480
+ # identifier's types each have their own `file_path` (a Scenic view and a
3481
+ # factory both named `reports` live in different files with different
3482
+ # histories), so this resolves git data against THIS type's own node
3483
+ # rather than a single pre-resolved hash shared across every type — the
3484
+ # previous shape let one type's commit history land in every colliding
3485
+ # type's `metadata.git`, keyed by whichever type {#register_and_write}
3486
+ # happened to touch last for that identifier.
3487
+ #
3488
+ # @param identifier [String]
3489
+ # @param type [Symbol]
3490
+ # @param affected_types [Set<Symbol>]
3491
+ # @param refresh_dependents [Boolean]
3492
+ # @param git_data [Hash{String => Hash}, nil]
3493
+ # @return [void]
3494
+ def rewrite_unit_json_of_type(identifier, type, affected_types, refresh_dependents:, git_data:)
3495
+ extractor_key = TYPE_TO_EXTRACTOR_KEY[type]
3496
+ return unless extractor_key
3497
+
3498
+ path = payload_dir.join(extractor_key.to_s, collision_safe_filename(identifier))
3499
+ return unless File.exist?(path)
3500
+
3501
+ data = JSON.parse(AtomicFile.read(path))
3502
+ # Serialized comparison, not `data.dup`: the git patch mutates the
3503
+ # nested metadata hash in place, which a shallow copy would follow.
3504
+ before = JSON.generate(data)
3505
+
3506
+ if refresh_dependents
3507
+ data['dependents'] = @dependency_graph.dependents_detail(identifier)
3508
+ .map { |d| { 'type' => d[:type].to_s, 'identifier' => d[:identifier] } }
3509
+ end
3510
+
3511
+ git = git_data && git_for_type(identifier, type, git_data)
3512
+ if git
3513
+ (data['metadata'] ||= {})['git'] = JSON.parse(JSON.generate(git))
3514
+ annotate_node_from_git(identifier, type, git)
3515
+ end
3516
+
3517
+ return if JSON.generate(data) == before
3518
+
3519
+ AtomicFile.write(path, json_serialize(data), durable: payload_writes_durable?)
3520
+ affected_types&.add(extractor_key)
3521
+ rescue JSON::ParserError => e
3522
+ Rails.logger.warn "[Woods] Could not finalize #{identifier}: #{e.message}"
3523
+ end
3524
+
3525
+ # This (identifier, type) pair's own git data, looked up by its own
3526
+ # `file_path` rather than by identifier — see {#rewrite_unit_json_of_type}.
3527
+ #
3528
+ # @param identifier [String]
3529
+ # @param type [Symbol]
3530
+ # @param git_data [Hash{String => Hash}] keyed by relative path
3531
+ # @return [Hash, nil]
3532
+ def git_for_type(identifier, type, git_data)
3533
+ node = @dependency_graph.node(identifier, type: type)
3534
+ return nil unless node && node[:file_path]
3535
+
3536
+ git_data[normalize_file_path(node[:file_path])]
3537
+ end
3538
+
3539
+ # Mirror of {#annotate_graph_with_git_data} for one incrementally
3540
+ # patched unit. Git data arrives symbol-keyed from {#batch_git_data} and
3541
+ # string-keyed when a caller hands in parsed JSON; both are accepted.
3542
+ #
3543
+ # Unlike {#annotate_graph_with_git_data}, a missing key is never forwarded
3544
+ # as an explicit `nil`: {DependencyGraph#annotate} treats `nil` as
3545
+ # "clear this attribute", and this batch's git data can be missing one of
3546
+ # the two keys without meaning the other's last-known value is gone.
3547
+ #
3548
+ # @param identifier [String]
3549
+ # @param type [Symbol]
3550
+ # @param git [Hash]
3551
+ # @return [void]
3552
+ def annotate_node_from_git(identifier, type, git)
3553
+ attributes = {
3554
+ commit_count: git[:commit_count] || git['commit_count'],
3555
+ change_frequency: git[:change_frequency] || git['change_frequency']
3556
+ }.compact
3557
+ return if attributes.empty?
3558
+
3559
+ @dependency_graph.annotate(identifier, type: type, **attributes)
3560
+ end
3561
+
3562
+ # Batch-fetch git metadata for the units written by this run, in a single
3563
+ # git invocation, keyed by Rails.root-relative path the way
3564
+ # {#batch_git_data} returns it.
3565
+ #
3566
+ # @param identifiers [Array<String>]
3567
+ # @return [Hash{String => Hash}]
3568
+ def incremental_git_data(identifiers)
3569
+ return {} if identifiers.empty? || !git_available?
3570
+
3571
+ paths = identifiers.flat_map do |identifier|
3572
+ @dependency_graph.nodes_for(identifier).filter_map do |node|
3573
+ next if %i[rails_source gem_source].include?(node[:type])
3574
+
3575
+ node[:file_path] if node[:file_path] && File.exist?(node[:file_path])
3576
+ end
3577
+ end
3578
+
3579
+ batch_git_data(paths.uniq)
3580
+ rescue StandardError => e
3581
+ Rails.logger.warn "[Woods] Incremental git enrichment failed: #{e.message}"
3582
+ {}
3583
+ end
3584
+
3585
+ # Re-extract a single known unit, identified by its graph node.
3586
+ #
3587
+ # Used for the transitive half of the blast radius — units whose own file
3588
+ # did not change but which depend on something that did. Files that *did*
3589
+ # change go through {#reconcile_changed_paths} instead, which reconciles
3590
+ # the whole path rather than one identifier.
3591
+ #
3592
+ # @param unit_id [String]
3593
+ # @param affected_types [Set<Symbol>, nil]
3594
+ # @return [String, nil] the identifier when it was re-extracted and written
1090
3595
  def re_extract_unit(unit_id, affected_types: nil)
1091
3596
  # Framework source only changes on version updates
1092
3597
  if unit_id.start_with?('rails/') || unit_id.start_with?('gems/')
1093
- Rails.logger.debug "[Woods] Skipping framework re-extraction for #{unit_id}"
1094
- return
3598
+ Rails.logger.debug { "[Woods] Skipping framework re-extraction for #{unit_id}" }
3599
+ return nil
1095
3600
  end
1096
3601
 
1097
- # Find the unit's type from the graph
1098
- node = @dependency_graph.to_h[:nodes][unit_id]
1099
- return unless node
3602
+ # An identifier can name units of several types, each with its own
3603
+ # extractor and its own file; re-extracting one and calling it done would
3604
+ # leave the others frozen at the pre-change extraction.
3605
+ types = @dependency_graph.node_types(unit_id)
3606
+ return nil if types.empty?
3607
+
3608
+ re_extracted = types.count { |type| re_extract_unit_of_type(unit_id, type, affected_types) }
3609
+ return nil if re_extracted.zero?
3610
+
3611
+ Rails.logger.info "[Woods] Re-extracted #{unit_id}"
3612
+ unit_id
3613
+ end
1100
3614
 
1101
- type = node[:type]&.to_sym
1102
- file_path = node[:file_path]
3615
+ # @param unit_id [String]
3616
+ # @param type [Symbol]
3617
+ # @param affected_types [Set<Symbol>, nil]
3618
+ # @return [String, nil] the identifier when this type was re-extracted and written
3619
+ def re_extract_unit_of_type(unit_id, type, affected_types)
3620
+ node = @dependency_graph.node(unit_id, type: type)
3621
+ file_path = node && node[:file_path]
1103
3622
 
1104
- return unless file_path && File.exist?(file_path)
3623
+ # A vanished file is not re-extractable; {#prune_vanished_units} owns it.
3624
+ return nil unless file_path && File.exist?(file_path)
1105
3625
 
1106
- # Re-extract based on type
1107
3626
  extractor_key = TYPE_TO_EXTRACTOR_KEY[type]
1108
- return unless extractor_key
3627
+ return nil unless extractor_key
1109
3628
 
1110
- extractor = EXTRACTORS[extractor_key]&.new
1111
- return unless extractor
1112
-
1113
- unit = if (method = CLASS_BASED[type])
1114
- klass = if unit_id.match?(/\A[A-Z][A-Za-z0-9_:]*\z/)
1115
- begin
1116
- unit_id.constantize
1117
- rescue StandardError
1118
- nil
1119
- end
1120
- end
1121
- extractor.public_send(method, klass) if klass
1122
- elsif (method = FILE_BASED[type])
1123
- extractor.public_send(method, file_path)
1124
- elsif GRAPHQL_TYPES.include?(type)
1125
- extractor.extract_graphql_file(file_path)
1126
- end
1127
-
1128
- return unless unit
3629
+ extractor = extractor_for(extractor_key)
3630
+ return nil unless extractor
1129
3631
 
1130
3632
  # File-based extractors can return several units from one file (a .rake
1131
3633
  # file defining multiple tasks, etc.); class-based extractors return one.
1132
- # Normalize to an array so every unit is registered and written — passing
1133
- # an Array straight to DependencyGraph#register crashes on unit.identifier.
1134
- units = unit.is_a?(Array) ? unit : [unit]
1135
- return if units.empty?
3634
+ units = Array(re_extracted_units(extractor, type, unit_id, file_path, extractor_key)).compact
3635
+ return nil if units.empty?
1136
3636
 
1137
- # Track which type was affected
1138
- affected_types&.add(extractor_key)
3637
+ register_and_write(extractor_key, units, affected_types)
3638
+ unit_id
3639
+ end
1139
3640
 
1140
- type_dir = @output_dir.join(extractor_key.to_s)
3641
+ # Dispatch one re-extraction to the right extractor entry point.
3642
+ #
3643
+ # @return [ExtractedUnit, Array<ExtractedUnit>, nil]
3644
+ def re_extracted_units(extractor, type, unit_id, file_path, extractor_key)
3645
+ if (method = CLASS_BASED[type])
3646
+ klass = constant_for_identifier(unit_id)
3647
+ klass && extractor.public_send(method, klass)
3648
+ elsif (method = FILE_BASED[type])
3649
+ units = Array(extract_file_based_unit(extractor, method, file_path, extractor_key)).compact
3650
+ return units unless CLASS_DISCOVERED_FALLBACK.key?(type)
3651
+ return units if units.any? { |unit| unit.identifier == unit_id }
3652
+
3653
+ re_extract_by_class(extractor, type, unit_id)
3654
+ elsif GRAPHQL_TYPES.include?(type)
3655
+ extractor.extract_graphql_file(file_path)
3656
+ end
3657
+ end
1141
3658
 
1142
- units.each do |extracted|
1143
- # Update dependency graph. Register BEFORE normalizing the path —
1144
- # the graph's file_map stores absolute paths (affected_by matches
1145
- # changed files against them), exactly as full extraction registers
1146
- # in Phase 1 and only normalizes in Phase 4.5.
1147
- @dependency_graph.register(extracted)
3659
+ # Re-extract a unit the full path discovered by class, not by file.
3660
+ #
3661
+ # Jobs are found two ways ({Extractors::JobExtractor#extract_all}): a scan
3662
+ # of the job directories, then a descendant walk for every job class the
3663
+ # scan did not name. A class-discovered job can live in a file whose
3664
+ # governed constant is not the job — a job nested inside a model — so
3665
+ # re-deriving it from that file names the enclosing class instead, and
3666
+ # registered a second job unit under the PORO's identifier that no full
3667
+ # extraction emits: the wrapper-naming collision, reached incrementally.
3668
+ # When the file entry point does not reproduce the unit, the full path
3669
+ # found it by class; do the same, and register nothing the file scan
3670
+ # would not have produced there either.
3671
+ #
3672
+ # Only a class the extractor itself would discover qualifies. A stale
3673
+ # pre-2.0 wrapper identifier can still constantize (to the wrapper class),
3674
+ # and re-extracting *that* by class would pin the stale unit with fresh
3675
+ # metadata; it stays as it is until a full extraction replaces it.
3676
+ #
3677
+ # @return [ExtractedUnit, nil]
3678
+ def re_extract_by_class(extractor, type, unit_id)
3679
+ klass = constant_for_identifier(unit_id)
3680
+ return nil unless klass && extractor.discoverable_classes.include?(klass)
1148
3681
 
1149
- # Unit JSON carries Rails.root-relative paths (full extraction's
1150
- # Phase 4.5); writing the raw absolute source_location here would
1151
- # leak container-absolute paths into the index after incremental runs.
1152
- extracted.file_path = normalize_file_path(extracted.file_path)
3682
+ extractor.public_send(CLASS_DISCOVERED_FALLBACK[type], klass)
3683
+ end
1153
3684
 
1154
- # Write updated unit
1155
- File.write(
1156
- type_dir.join(collision_safe_filename(extracted.identifier)),
1157
- json_serialize(extracted.to_h)
1158
- )
3685
+ # @param unit_id [String]
3686
+ # @return [Class, nil] the constant an identifier names, when it names one
3687
+ def constant_for_identifier(unit_id)
3688
+ return nil unless unit_id.match?(/\A[A-Z][A-Za-z0-9_:]*\z/)
3689
+
3690
+ begin
3691
+ unit_id.constantize
3692
+ rescue StandardError
3693
+ nil
1159
3694
  end
3695
+ end
1160
3696
 
1161
- Rails.logger.info "[Woods] Re-extracted #{unit_id}"
3697
+ # Invoke a file-based extraction method, supplying the extra arguments
3698
+ # the handful of non-uniform signatures need.
3699
+ #
3700
+ # @return [ExtractedUnit, Array<ExtractedUnit>, nil]
3701
+ def extract_file_based_unit(extractor, method, file_path, extractor_key)
3702
+ if extractor_key == :poros
3703
+ extractor.public_send(method, file_path, ar_names: active_record_names)
3704
+ else
3705
+ extractor.public_send(method, file_path)
3706
+ end
1162
3707
  end
1163
3708
  end
1164
3709
  end