woods 2.0.0.beta2 → 2.0.0.beta4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (233) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +339 -1
  3. data/CONTRIBUTING.md +188 -12
  4. data/README.md +93 -174
  5. data/SECURITY.md +9 -6
  6. data/docs/AGENT_GUIDE.md +109 -8
  7. data/docs/AGENT_SETUP.md +98 -7
  8. data/docs/BACKEND_MATRIX.md +25 -0
  9. data/docs/CLIENT_HOOKS.md +111 -0
  10. data/docs/CONFIGURATION_REFERENCE.md +267 -16
  11. data/docs/CONSOLE_MCP_SETUP.md +80 -7
  12. data/docs/DOCKER_SETUP.md +22 -3
  13. data/docs/EVALUATION.md +464 -1
  14. data/docs/EXTRACTOR_REFERENCE.md +45 -6
  15. data/docs/FAQ.md +11 -12
  16. data/docs/GETTING_STARTED.md +17 -5
  17. data/docs/INCREMENTAL_EXTRACTION.md +147 -7
  18. data/docs/INDEX_LAYOUT.md +382 -0
  19. data/docs/INTERNALS.md +7 -2
  20. data/docs/MCP_SERVERS.md +276 -5
  21. data/docs/MCP_TOOL_COOKBOOK.md +37 -22
  22. data/docs/MCP_WORKTREE_SETUP.md +43 -83
  23. data/docs/NOTION_INTEGRATION.md +13 -0
  24. data/docs/OBSIDIAN_INTEGRATION.md +57 -9
  25. data/docs/PUBLISHED_INDEX.md +72 -0
  26. data/docs/README.md +7 -0
  27. data/docs/RETRIEVAL_GUIDE.md +273 -12
  28. data/docs/RUNTIME_TRACING.md +71 -0
  29. data/docs/SOURCE_FRESHNESS.md +143 -0
  30. data/docs/TROUBLESHOOTING.md +129 -18
  31. data/docs/UNBLOCKED_INTEGRATION.md +25 -0
  32. data/docs/UPGRADING_TO_2.md +48 -22
  33. data/docs/WATCH_DAEMON.md +277 -67
  34. data/exe/woods-agent-config +6 -0
  35. data/exe/woods-extract +5 -0
  36. data/exe/woods-hook-context +6 -0
  37. data/exe/woods-mcp-start +14 -9
  38. data/lib/generators/woods/pgvector_generator.rb +8 -2
  39. data/lib/generators/woods/templates/woods.rb.tt +1 -3
  40. data/lib/tasks/woods.rake +47 -397
  41. data/lib/woods/agent_configuration/applier.rb +135 -0
  42. data/lib/woods/agent_configuration/cli.rb +101 -0
  43. data/lib/woods/agent_configuration/cli_options.rb +29 -0
  44. data/lib/woods/agent_configuration/document.rb +105 -0
  45. data/lib/woods/agent_configuration/error.rb +7 -0
  46. data/lib/woods/agent_configuration/launcher.rb +75 -0
  47. data/lib/woods/agent_configuration/layout.rb +72 -0
  48. data/lib/woods/agent_configuration/managed_section.rb +62 -0
  49. data/lib/woods/agent_configuration/plan.rb +98 -0
  50. data/lib/woods/agent_configuration/plan_diff.rb +38 -0
  51. data/lib/woods/agent_configuration/planned_files.rb +61 -0
  52. data/lib/woods/agent_configuration/planner.rb +63 -0
  53. data/lib/woods/agent_configuration/planner_validation.rb +77 -0
  54. data/lib/woods/agent_configuration/preflight.rb +100 -0
  55. data/lib/woods/agent_configuration/recovery.rb +49 -0
  56. data/lib/woods/ast/node.rb +2 -0
  57. data/lib/woods/ast/parser.rb +38 -5
  58. data/lib/woods/builder.rb +21 -5
  59. data/lib/woods/cache/cache_middleware.rb +28 -7
  60. data/lib/woods/cache/cache_store.rb +4 -5
  61. data/lib/woods/change_set.rb +5 -4
  62. data/lib/woods/console/credential_index.rb +20 -2
  63. data/lib/woods/console/credential_scanner.rb +18 -17
  64. data/lib/woods/console/credential_scanner_registry.rb +36 -0
  65. data/lib/woods/console/dispatch_pipeline.rb +7 -0
  66. data/lib/woods/console/embedded_executor.rb +32 -10
  67. data/lib/woods/console/encrypted_credential_snapshot.rb +16 -0
  68. data/lib/woods/console/rack_middleware.rb +22 -13
  69. data/lib/woods/console/server.rb +18 -16
  70. data/lib/woods/console/sql_noise_stripper.rb +9 -7
  71. data/lib/woods/console/sql_table_scanner.rb +47 -7
  72. data/lib/woods/console/sql_validator.rb +49 -9
  73. data/lib/woods/console/sqlite_read_guard.rb +46 -0
  74. data/lib/woods/coordination/pipeline_lock.rb +3 -2
  75. data/lib/woods/dependency_graph.rb +65 -13
  76. data/lib/woods/embedding/corpus.rb +94 -0
  77. data/lib/woods/embedding/indexer.rb +114 -60
  78. data/lib/woods/embedding/openai.rb +17 -6
  79. data/lib/woods/evaluation/ablation_executor.rb +6 -1
  80. data/lib/woods/evaluation/ablation_timed_executor.rb +22 -4
  81. data/lib/woods/export/typed_reader.rb +56 -0
  82. data/lib/woods/extractor.rb +277 -149
  83. data/lib/woods/extractors/action_cable_extractor.rb +3 -1
  84. data/lib/woods/extractors/behavioral_profile.rb +9 -7
  85. data/lib/woods/extractors/caching_extractor.rb +3 -1
  86. data/lib/woods/extractors/concern_extractor.rb +64 -6
  87. data/lib/woods/extractors/configuration_extractor.rb +7 -3
  88. data/lib/woods/extractors/controller_extractor.rb +13 -4
  89. data/lib/woods/extractors/database_view_extractor.rb +3 -1
  90. data/lib/woods/extractors/declared_parent.rb +55 -0
  91. data/lib/woods/extractors/decorator_extractor.rb +3 -1
  92. data/lib/woods/extractors/engine_extractor.rb +3 -1
  93. data/lib/woods/extractors/event_extractor.rb +4 -2
  94. data/lib/woods/extractors/factory_extractor.rb +3 -1
  95. data/lib/woods/extractors/graphql_extractor.rb +10 -13
  96. data/lib/woods/extractors/i18n_extractor.rb +3 -1
  97. data/lib/woods/extractors/job_extractor.rb +6 -19
  98. data/lib/woods/extractors/lib_extractor.rb +13 -9
  99. data/lib/woods/extractors/mailer_extractor.rb +26 -15
  100. data/lib/woods/extractors/manager_extractor.rb +3 -1
  101. data/lib/woods/extractors/method_parameters.rb +53 -0
  102. data/lib/woods/extractors/middleware_argument.rb +65 -0
  103. data/lib/woods/extractors/middleware_extractor.rb +9 -3
  104. data/lib/woods/extractors/migration_extractor.rb +3 -1
  105. data/lib/woods/extractors/model_extractor.rb +26 -34
  106. data/lib/woods/extractors/package_extractor.rb +24 -4
  107. data/lib/woods/extractors/phlex_extractor.rb +3 -1
  108. data/lib/woods/extractors/policy_extractor.rb +3 -1
  109. data/lib/woods/extractors/poro_extractor.rb +13 -9
  110. data/lib/woods/extractors/pundit_extractor.rb +3 -1
  111. data/lib/woods/extractors/rails_source_extractor.rb +4 -2
  112. data/lib/woods/extractors/rake_task_extractor.rb +4 -2
  113. data/lib/woods/extractors/route_extractor.rb +3 -1
  114. data/lib/woods/extractors/route_helper_resolver.rb +10 -33
  115. data/lib/woods/extractors/scheduled_job_extractor.rb +41 -15
  116. data/lib/woods/extractors/serializer_extractor.rb +4 -2
  117. data/lib/woods/extractors/service_extractor.rb +3 -1
  118. data/lib/woods/extractors/shared_dependency_scanner.rb +2 -2
  119. data/lib/woods/extractors/shared_utility_methods.rb +48 -19
  120. data/lib/woods/extractors/source_nesting.rb +1 -1
  121. data/lib/woods/extractors/state_machine_extractor.rb +3 -1
  122. data/lib/woods/extractors/test_mapping_extractor.rb +3 -1
  123. data/lib/woods/extractors/validator_extractor.rb +3 -1
  124. data/lib/woods/extractors/view_component_extractor.rb +3 -1
  125. data/lib/woods/extractors/view_template_extractor.rb +3 -1
  126. data/lib/woods/gem_mapper.rb +2 -0
  127. data/lib/woods/git_history.rb +116 -0
  128. data/lib/woods/graph_analyzer.rb +35 -6
  129. data/lib/woods/hooks/context_cli.rb +54 -0
  130. data/lib/woods/hooks/context_event.rb +88 -0
  131. data/lib/woods/hooks/context_hint.rb +73 -0
  132. data/lib/woods/hooks/context_impact.rb +77 -0
  133. data/lib/woods/hooks/context_output.rb +47 -0
  134. data/lib/woods/hooks/context_state.rb +102 -0
  135. data/lib/woods/hooks/refresh.rb +79 -0
  136. data/lib/woods/hooks/rule_projection.rb +78 -0
  137. data/lib/woods/input_rules.rb +19 -0
  138. data/lib/woods/mcp/bearer_auth.rb +22 -13
  139. data/lib/woods/mcp/bootstrapper.rb +79 -4
  140. data/lib/woods/mcp/config_resolver.rb +2 -1
  141. data/lib/woods/mcp/index_reader.rb +334 -162
  142. data/lib/woods/mcp/initialization_guidance.rb +27 -0
  143. data/lib/woods/mcp/origin_guard.rb +17 -9
  144. data/lib/woods/mcp/published_lexical_retriever.rb +115 -0
  145. data/lib/woods/mcp/renderers/markdown_renderer.rb +22 -9
  146. data/lib/woods/mcp/renderers/plain_renderer.rb +18 -8
  147. data/lib/woods/mcp/search_results.rb +74 -0
  148. data/lib/woods/mcp/server.rb +178 -63
  149. data/lib/woods/mcp/tool_contract.rb +3 -1
  150. data/lib/woods/mcp/tool_response_renderer.rb +41 -0
  151. data/lib/woods/mcp/traversal_evidence.rb +113 -0
  152. data/lib/woods/mcp/traversal_evidence_index.rb +100 -0
  153. data/lib/woods/mcp/traversal_evidence_page.rb +41 -0
  154. data/lib/woods/mcp/traversal_evidence_text.rb +52 -0
  155. data/lib/woods/mcp/traversal_response.rb +22 -0
  156. data/lib/woods/notion/exporter.rb +56 -17
  157. data/lib/woods/obsidian/destination_plan.rb +98 -0
  158. data/lib/woods/obsidian/name_mapper.rb +19 -3
  159. data/lib/woods/obsidian/note_builder.rb +19 -10
  160. data/lib/woods/obsidian/vault_exporter.rb +88 -32
  161. data/lib/woods/operator/pipeline_guard.rb +18 -13
  162. data/lib/woods/path_dispatcher.rb +13 -6
  163. data/lib/woods/payload_store.rb +27 -26
  164. data/lib/woods/published_index/typed_unit_reader.rb +40 -3
  165. data/lib/woods/published_index.rb +2 -2
  166. data/lib/woods/railtie.rb +3 -3
  167. data/lib/woods/railtie_support.rb +12 -12
  168. data/lib/woods/rake_helpers.rb +382 -0
  169. data/lib/woods/resilience/graph_invariant_validator/membership_checks.rb +71 -0
  170. data/lib/woods/resilience/graph_invariant_validator/node_checks.rb +61 -0
  171. data/lib/woods/resilience/graph_invariant_validator/reverse_relationship_checks.rb +46 -0
  172. data/lib/woods/resilience/graph_invariant_validator.rb +119 -0
  173. data/lib/woods/resilience/index_validator/graph_checks.rb +80 -0
  174. data/lib/woods/resilience/index_validator.rb +112 -23
  175. data/lib/woods/retrieval/context_assembler.rb +50 -15
  176. data/lib/woods/retrieval/lexical_assembler.rb +84 -0
  177. data/lib/woods/retrieval/lexical_index.rb +120 -0
  178. data/lib/woods/retrieval/ranker.rb +4 -2
  179. data/lib/woods/retrieval/scope.rb +108 -0
  180. data/lib/woods/retrieval/scoped_graph_store.rb +32 -0
  181. data/lib/woods/retrieval/scoped_vector_store.rb +55 -0
  182. data/lib/woods/retrieval/search_executor.rb +86 -27
  183. data/lib/woods/retrieval/source_evidence.rb +200 -0
  184. data/lib/woods/retriever.rb +98 -22
  185. data/lib/woods/ruby_analyzer/trace_enricher.rb +77 -38
  186. data/lib/woods/session_tracer/file_store.rb +6 -1
  187. data/lib/woods/session_tracer/middleware.rb +10 -12
  188. data/lib/woods/session_tracer/redis_store.rb +22 -6
  189. data/lib/woods/session_tracer/session_flow_assembler.rb +23 -17
  190. data/lib/woods/session_tracer/solid_cache_coordination.rb +6 -4
  191. data/lib/woods/session_tracer/unit_resolver.rb +63 -0
  192. data/lib/woods/source_inputs/consumer_errors.rb +31 -0
  193. data/lib/woods/source_inputs/handoff.rb +102 -0
  194. data/lib/woods/source_inputs/launcher.rb +157 -0
  195. data/lib/woods/source_inputs/manifest.rb +124 -0
  196. data/lib/woods/source_inputs/private_key.rb +55 -0
  197. data/lib/woods/source_inputs/scanner.rb +171 -0
  198. data/lib/woods/source_inputs/scopes.rb +71 -0
  199. data/lib/woods/source_inputs/session.rb +214 -0
  200. data/lib/woods/source_inputs/status.rb +84 -0
  201. data/lib/woods/source_inputs/verifier.rb +107 -0
  202. data/lib/woods/storage/metadata_store.rb +25 -25
  203. data/lib/woods/storage/pgvector.rb +35 -10
  204. data/lib/woods/storage/qdrant.rb +17 -7
  205. data/lib/woods/storage/vector_store.rb +18 -6
  206. data/lib/woods/tasks.rb +3 -2
  207. data/lib/woods/temporal/json_snapshot_store.rb +58 -9
  208. data/lib/woods/unblocked/exporter.rb +59 -70
  209. data/lib/woods/version.rb +1 -1
  210. data/lib/woods/watch/boot_snapshot.rb +52 -0
  211. data/lib/woods/watch/daemon.rb +154 -32
  212. data/lib/woods/watch/listen_watcher.rb +4 -0
  213. data/lib/woods/watch/polling_watcher.rb +5 -1
  214. data/lib/woods/watch/status.rb +20 -15
  215. data/lib/woods/watch/tree_scan.rb +21 -13
  216. data/lib/woods/watch/watcher.rb +4 -1
  217. data/lib/woods.rb +50 -11
  218. data/plugin/.claude-plugin/plugin.json +1 -1
  219. data/plugin/hooks/adapters/normalize.jq +15 -0
  220. data/plugin/hooks/adapters/normalize.rb +63 -0
  221. data/plugin/hooks/hooks.json +20 -0
  222. data/plugin/hooks/woods-context.sh +50 -0
  223. data/plugin/hooks/woods-input-rules.sh +159 -0
  224. data/plugin/hooks/woods-opencode.mjs +65 -0
  225. data/plugin/hooks/woods-post-edit.sh +2 -225
  226. data/plugin/hooks/woods-refresh.sh +260 -0
  227. data/plugin/hooks/woods-session-start.sh +47 -55
  228. data/plugin/skills/woods-agent-enable/SKILL.md +19 -0
  229. data/plugin/skills/woods-diagnose/SKILL.md +319 -1
  230. data/plugin/skills/woods-investigate/SKILL.md +145 -0
  231. data/plugin/skills/woods-mcp-config/SKILL.md +90 -2
  232. data/plugin/skills/woods-setup/SKILL.md +110 -6
  233. metadata +87 -5
@@ -8,12 +8,14 @@ require 'pathname'
8
8
  require 'set'
9
9
 
10
10
  require_relative 'atomic_file'
11
+ require_relative 'version'
11
12
  require_relative 'filename_utils'
12
13
  require_relative 'token_utils'
13
14
  require_relative 'extracted_unit'
14
15
  require_relative 'dependency_graph'
15
16
  require_relative 'payload_store'
16
17
  require_relative 'git_provenance'
18
+ require_relative 'git_history'
17
19
  require_relative 'extractors/model_extractor'
18
20
  require_relative 'extractors/controller_extractor'
19
21
  require_relative 'extractors/phlex_extractor'
@@ -55,6 +57,7 @@ require_relative 'flow_precomputer'
55
57
  require_relative 'change_set'
56
58
  require_relative 'generation'
57
59
  require_relative 'path_dispatcher'
60
+ require_relative 'source_inputs/session'
58
61
 
59
62
  module Woods
60
63
  # Extractor is the main orchestrator for codebase extraction.
@@ -208,7 +211,6 @@ module Woods
208
211
  configuration: :extract_configuration_file,
209
212
  view_template: :extract_view_template_file,
210
213
  migration: :extract_migration_file,
211
- rake_task: :extract_rake_file,
212
214
  decorator: :extract_decorator_file,
213
215
  database_view: :extract_view_file,
214
216
  caching: :extract_caching_file,
@@ -290,7 +292,7 @@ module Woods
290
292
  method: :extract_from_runtime_type, reconcile_removals: false }
291
293
  }.freeze
292
294
 
293
- # Extractors with no per-file entry point: they scan the whole app (or
295
+ # Extractors requiring a complete source set: they scan the whole app (or
294
296
  # introspect the whole runtime) in one pass, so an incremental run
295
297
  # replaces their output wholesale rather than per unit. Before #164
296
298
  # these types were simply skipped by incremental runs while
@@ -299,6 +301,7 @@ module Woods
299
301
  #
300
302
  # @return [Hash{Symbol => Symbol}] extractor key => unit type
301
303
  WHOLE_APP_EXTRACTORS = {
304
+ rake_tasks: :rake_task,
302
305
  routes: :route,
303
306
  middleware: :middleware,
304
307
  engines: :engine,
@@ -355,7 +358,7 @@ module Woods
355
358
  # flat index — the output root also holds `generation.json`, `dumps/`,
356
359
  # `tasks/`, `woods.sqlite3` and `payloads/` itself, none of which belong
357
360
  # to a generation's payload.
358
- PAYLOAD_FILES = %w[manifest.json dependency_graph.json graph_analysis.json SUMMARY.md].freeze
361
+ PAYLOAD_FILES = %w[manifest.json source_inputs.json dependency_graph.json graph_analysis.json SUMMARY.md].freeze
359
362
 
360
363
  # Payload directories that are not per-type unit directories.
361
364
  PAYLOAD_DIRS = %w[flows].freeze
@@ -394,7 +397,9 @@ module Woods
394
397
  #
395
398
  # @return [Hash] Results keyed by extractor type
396
399
  def extract_all
400
+ profile_started = Process.clock_gettime(Process::CLOCK_MONOTONIC) if profiling?
397
401
  setup_output_directory
402
+ profile_phase('source capture') { begin_source_inputs('full') }
398
403
  ModelNameCache.reset!
399
404
  # @package_resolver alone is not enough: #package_resolver builds
400
405
  # through #extractor_for, which memoizes into @incremental_extractors.
@@ -422,29 +427,33 @@ module Woods
422
427
 
423
428
  # Phase 1.5: Deduplicate results
424
429
  Rails.logger.info '[Woods] Deduplicating results...'
425
- deduplicate_results
430
+ profile_phase('deduplication') { deduplicate_results }
426
431
 
427
432
  # Phase 1.6: Package membership. Runs before the graph is rebuilt so
428
433
  # registration copies metadata[:package] onto the node (#280).
429
- annotate_packages
434
+ profile_phase('package annotation') { annotate_packages }
430
435
 
431
436
  # Rebuild the graph from deduped results. #164 gave DependencyGraph
432
437
  # `#remove`/`#unregister`, so surgical removal is now possible — but a
433
438
  # full extraction has just registered every unit including duplicates,
434
439
  # and rebuilding from the deduped set is both cheaper and less
435
440
  # error-prone than unwinding registrations one at a time.
436
- @dependency_graph = DependencyGraph.new
437
- @results.each_value { |units| units.each { |u| @dependency_graph.register(u) } }
441
+ profile_phase('graph rebuild') do
442
+ @dependency_graph = DependencyGraph.new
443
+ @results.each_value { |units| units.each { |u| @dependency_graph.register(u) } }
444
+ end
438
445
 
439
446
  # Phase 2: Resolve dependents (reverse dependencies)
440
447
  Rails.logger.info '[Woods] Resolving dependents...'
441
- resolve_dependents
448
+ profile_phase('dependents') { resolve_dependents }
442
449
 
443
450
  # Phase 3: Enrich with git data. Runs BEFORE analysis now: the
444
451
  # volatile_dependencies report reads commit counts off graph nodes.
445
452
  Rails.logger.info '[Woods] Enriching with git data...'
446
- enrich_with_git_data
447
- annotate_graph_with_git_data
453
+ profile_phase('git enrichment') do
454
+ enrich_with_git_data
455
+ annotate_graph_with_git_data
456
+ end
448
457
 
449
458
  # Phase 4: Graph analysis (PageRank, structural metrics)
450
459
  Rails.logger.info '[Woods] Analyzing dependency graph...'
@@ -452,7 +461,7 @@ module Woods
452
461
 
453
462
  # Phase 4.5: Normalize file_path to relative paths
454
463
  Rails.logger.info '[Woods] Normalizing file paths...'
455
- normalize_file_paths
464
+ profile_phase('path normalization') { normalize_file_paths }
456
465
 
457
466
  # Phase 5: Write output
458
467
  Rails.logger.info '[Woods] Writing output...'
@@ -462,7 +471,7 @@ module Woods
462
471
  # run after write_results — the just-written set is what defines
463
472
  # "legitimate" — and belongs to the full path only; the incremental path
464
473
  # deletes through the graph instead. See {#sweep_orphaned_unit_files}.
465
- sweep_orphaned_unit_files
474
+ profile_phase('orphan sweep') { sweep_orphaned_unit_files }
466
475
 
467
476
  # Phase 5.5: Precompute request flows (opt-in). Must run AFTER
468
477
  # write_results — FlowAssembler loads unit JSON from disk, so running
@@ -485,18 +494,23 @@ module Woods
485
494
  profile_phase('flows') { precompute_flows }
486
495
  end
487
496
 
488
- write_dependency_graph
489
- write_graph_analysis
497
+ profile_phase('graph write') do
498
+ write_dependency_graph
499
+ write_graph_analysis
500
+ end
490
501
  profile_phase('manifest and summary') do
491
502
  write_manifest
492
503
  write_structural_summary
493
504
  end
494
- capture_snapshot
495
- profile_phase('publish') { publish_generation('full') }
505
+ profile_phase('snapshot') { capture_snapshot }
506
+ @source_inputs.full_units(@results, consumers: @extractors)
507
+ publish_generation('full')
496
508
 
497
509
  log_summary
498
510
 
499
511
  @results
512
+ ensure
513
+ log_profile_total('full', profile_started)
500
514
  end
501
515
 
502
516
  # ══════════════════════════════════════════════════════════════════════
@@ -528,7 +542,8 @@ module Woods
528
542
  # @param changed_files [Array<String>] List of changed file paths
529
543
  # @return [Array<String>] Identifiers of units re-extracted, added, or removed
530
544
  def extract_changed(changed_files)
531
- prepare_incremental_run
545
+ profile_started = Process.clock_gettime(Process::CLOCK_MONOTONIC) if profiling?
546
+ prepare_incremental_run(operation: 'incremental')
532
547
 
533
548
  change_set = ChangeSet.new(paths: changed_files, root: Rails.root)
534
549
  affected_types = Set.new
@@ -551,37 +566,41 @@ module Woods
551
566
  acc
552
567
  end
553
568
 
554
- touched.merge(reconcile_class_based_types(affected_types))
555
- touched.merge(rerun_whole_app_extractors(change_set, affected_types))
556
- touched.merge(reannotate_packages(change_set, affected_types))
557
- pruned = prune_vanished_units(change_set, affected_types)
558
- touched.merge(pruned)
559
-
560
- # Reconcile once more, because pruning can un-know a class the first pass
561
- # skipped. A class-based file moved between autoload directories with its
562
- # constant unchanged is still registered under the old path when
563
- # reconciliation runs, so it looks known and is not re-extracted; the
564
- # prune that follows then removes it for its vanished path. This pass
565
- # re-adds it in the same run (M1) instead of leaving the unit missing
566
- # until some later run happens to notice. Idempotent when nothing was
567
- # pruned: the discovery set is compared against the graph, so an
568
- # already-registered class is skipped.
569
- #
570
- # But not everything pruning removed may come back. `except:` keeps the
571
- # *deletion* shape pruned: without a reload, a constant outlives the file
572
- # that defined it — so deleting `app/models/user.rb` prunes `User`, and
573
- # this pass finds `User` still in `ActiveRecord::Base.descendants`.
574
- # Re-registering it would pin the unit to a path that no longer exists,
575
- # and nothing could ever remove it: the sweep excludes class-based units
576
- # and no future change set names that path again. A resident daemon
577
- # processing a batch before its reload hits this every time. What
578
- # separates the two shapes is the filesystem — only pruned identifiers
579
- # that a still-existing file in the change set actually declares are
580
- # re-addable. See {#readdable_pruned_classes}.
581
- touched.merge(reconcile_class_based_types(
582
- affected_types, except: pruned - readdable_pruned_classes(pruned, change_set)
583
- ))
584
-
569
+ profile_phase('reconciliation') do
570
+ touched.merge(reconcile_class_based_types(affected_types))
571
+ touched.merge(reconcile_model_mixins(affected_types))
572
+ touched.merge(rerun_whole_app_extractors(change_set, affected_types))
573
+ touched.merge(reannotate_packages(change_set, affected_types))
574
+ pruned = prune_vanished_units(change_set, affected_types)
575
+ touched.merge(pruned)
576
+
577
+ # Reconcile once more, because pruning can un-know a class the first pass
578
+ # skipped. A class-based file moved between autoload directories with its
579
+ # constant unchanged is still registered under the old path when
580
+ # reconciliation runs, so it looks known and is not re-extracted; the
581
+ # prune that follows then removes it for its vanished path. This pass
582
+ # re-adds it in the same run (M1) instead of leaving the unit missing
583
+ # until some later run happens to notice. Idempotent when nothing was
584
+ # pruned: the discovery set is compared against the graph, so an
585
+ # already-registered class is skipped.
586
+ #
587
+ # But not everything pruning removed may come back. `except:` keeps the
588
+ # *deletion* shape pruned: without a reload, a constant outlives the file
589
+ # that defined it — so deleting `app/models/user.rb` prunes `User`, and
590
+ # this pass finds `User` still in `ActiveRecord::Base.descendants`.
591
+ # Re-registering it would pin the unit to a path that no longer exists,
592
+ # and nothing could ever remove it: the sweep excludes class-based units
593
+ # and no future change set names that path again. A resident daemon
594
+ # processing a batch before its reload hits this every time. What
595
+ # separates the two shapes is the filesystem — only pruned identifiers
596
+ # that a still-existing file in the change set actually declares are
597
+ # re-addable. See {#readdable_pruned_classes}.
598
+ touched.merge(reconcile_class_based_types(
599
+ affected_types, except: pruned - readdable_pruned_classes(pruned, change_set)
600
+ ))
601
+ end
602
+
603
+ raise_on_handled_extraction_failure!
585
604
  finalize_incremental_unit_json(affected_types)
586
605
 
587
606
  # Regenerate type indexes for affected types
@@ -594,6 +613,8 @@ module Woods
594
613
  finalize_incremental_run(touched)
595
614
 
596
615
  touched.to_a
616
+ ensure
617
+ log_profile_total('incremental', profile_started)
597
618
  end
598
619
 
599
620
  # ══════════════════════════════════════════════════════════════════════
@@ -628,6 +649,7 @@ module Woods
628
649
  # extractor
629
650
  # @raise [ArgumentError] when no recognized key is given
630
651
  def refresh(*keys)
652
+ profile_started = Process.clock_gettime(Process::CLOCK_MONOTONIC) if profiling?
631
653
  keys = Array(keys).flatten.map(&:to_sym).uniq
632
654
  known, unknown = keys.partition { |key| EXTRACTORS.key?(key) }
633
655
  raise ArgumentError, "No known extractor in #{keys.inspect}" if known.empty?
@@ -635,17 +657,20 @@ module Woods
635
657
  known += ROUTE_CONSUMER_EXTRACTORS if known.include?(:routes)
636
658
  known.uniq!
637
659
 
638
- prepare_incremental_run
660
+ prepare_incremental_run(operation: 'refresh')
639
661
  affected_types = Set.new
640
662
  touched = known.each_with_object(Set.new) do |key, acc|
641
663
  acc.merge(replace_type_wholesale(key, affected_types))
642
664
  end
643
665
 
666
+ raise_on_handled_extraction_failure!
644
667
  finalize_incremental_unit_json(affected_types)
645
- affected_types.each { |type_key| regenerate_type_index(type_key) }
668
+ profile_phase('type index') { affected_types.each { |type_key| regenerate_type_index(type_key) } }
646
669
  finalize_incremental_run(touched, reason: "refresh:#{known.sort.join(',')}")
647
670
 
648
671
  { types: known, touched: touched.to_a, unknown: unknown }
672
+ ensure
673
+ log_profile_total('refresh', profile_started)
649
674
  end
650
675
 
651
676
  # Raise when the most recent extraction run wrote a payload but could not
@@ -667,6 +692,15 @@ module Woods
667
692
 
668
693
  private
669
694
 
695
+ # Whole-run wall time, including unprofiled setup and failed runs. This
696
+ # separate log family must never be added to the individual phase times.
697
+ def log_profile_total(name, started)
698
+ return unless started
699
+
700
+ elapsed = Process.clock_gettime(Process::CLOCK_MONOTONIC) - started
701
+ Rails.logger.info "[Woods] [profile total] #{name} in #{elapsed.round(2)}s"
702
+ end
703
+
670
704
  # Time one phase of a run and log how long it took, when WOODS_PROFILE=1.
671
705
  #
672
706
  # The per-extractor lines (see {#extract_all_sequential}) already report
@@ -745,7 +779,8 @@ module Woods
745
779
  #
746
780
  # @return [void]
747
781
  # @raise [Woods::ExtractionError] see {#begin_payload!}
748
- def prepare_incremental_run
782
+ def prepare_incremental_run(operation: 'incremental')
783
+ profile_phase('source capture') { begin_source_inputs(operation) }
749
784
  profile_phase('payload seed') { begin_payload!(strict: true) }
750
785
  graph_path = payload_dir.join('dependency_graph.json')
751
786
  ensure_incremental_baseline!(graph_path)
@@ -795,7 +830,7 @@ module Woods
795
830
  # reading "incremental" after a `woods:refresh[routes]` is being misled
796
831
  # @return [void]
797
832
  def finalize_incremental_run(touched, reason: 'incremental')
798
- write_dependency_graph
833
+ profile_phase('graph write') { write_dependency_graph }
799
834
 
800
835
  if touched.empty?
801
836
  Rails.logger.info '[Woods] Incremental run changed nothing — leaving manifest timestamp untouched'
@@ -808,7 +843,7 @@ module Woods
808
843
  write_manifest(incremental: true)
809
844
  write_structural_summary
810
845
  end
811
- profile_phase('publish') { publish_generation(reason) }
846
+ publish_generation(reason)
812
847
 
813
848
  return unless Woods.configuration.enable_snapshots
814
849
 
@@ -831,9 +866,10 @@ module Woods
831
866
  # Resolve (and if necessary rename) the payload first, so the flush
832
867
  # below covers the directory under the name the pointer will carry.
833
868
  payload = publishable_payload_name(generation)
869
+ profile_phase('source verification') { write_source_inputs } if payload
834
870
  profile_phase('payload sync') { sync_payload }
835
- marker = generation.bump!(reason: reason, payload: payload)
836
- prune_payloads(marker.number)
871
+ marker = profile_phase('publish') { generation.bump!(reason: reason, payload: payload) }
872
+ profile_phase('payload prune') { prune_payloads(marker.number) }
837
873
  marker
838
874
  rescue StandardError => e
839
875
  # A failed bump must not fail the extraction that produced a perfectly
@@ -852,6 +888,61 @@ module Woods
852
888
  nil
853
889
  end
854
890
 
891
+ # Capture before eager loading or extraction; only an explicit fresh-launch
892
+ # handoff can additionally establish the pre-Bundler/Rails boot boundary.
893
+ def begin_source_inputs(operation)
894
+ @failed_consumers = Set.new
895
+ @source_inputs = SourceInputs::Session.new(root: Rails.root, output_dir: @output_dir,
896
+ baseline_path: source_input_baseline_path,
897
+ operation: operation)
898
+ end
899
+
900
+ def source_input_baseline_path
901
+ generation = Generation.new(output_dir: @output_dir)
902
+ marker = generation.current
903
+ directory = generation.payload_dir(marker)
904
+ return nil unless marker.payload && directory != generation.root
905
+
906
+ directory.join(SourceInputs::Manifest::FILE_NAME)
907
+ rescue TypeError, NoMethodError
908
+ nil
909
+ end
910
+
911
+ def source_consumer_failed?(key, consumer = extractor_for(key))
912
+ failed = consumer.nil? || SourceInputs::ConsumerErrors.failed?(consumer)
913
+ @source_inputs&.unverified("extractor:#{key}") if failed
914
+ failed
915
+ end
916
+
917
+ # A rescued consumer error is not a successful empty result. Reset the
918
+ # per-call flag because one extractor instance serves several paths, but
919
+ # retain the failed scope for the run so a later success cannot authorize
920
+ # publication. Watch then carries the complete batch forward for retry.
921
+ def checked_extraction(key, consumer)
922
+ (@failed_consumers ||= Set.new).add(key) if SourceInputs::ConsumerErrors.failed?(consumer)
923
+ SourceInputs::ConsumerErrors.reset(consumer)
924
+ result = yield
925
+ return result unless source_consumer_failed?(key, consumer)
926
+
927
+ (@failed_consumers ||= Set.new).add(key)
928
+ nil
929
+ end
930
+
931
+ def raise_on_handled_extraction_failure!
932
+ return if @failed_consumers.nil? || @failed_consumers.empty?
933
+
934
+ raise Woods::ExtractionError,
935
+ "Extraction failed for #{@failed_consumers.to_a.sort.join(', ')}; " \
936
+ 'the previous generation remains active. Fix the logged source errors and retry the complete batch.'
937
+ end
938
+
939
+ def write_source_inputs
940
+ return unless @source_inputs
941
+
942
+ manifest = @source_inputs.finish(generation: @payload_generation, eager_load_complete: @eager_load_complete)
943
+ AtomicFile.write(payload_dir.join(SourceInputs::Manifest::FILE_NAME), JSON.pretty_generate(manifest.data))
944
+ end
945
+
855
946
  # Open the payload directory this run publishes into, seeded from the
856
947
  # generation currently on disk.
857
948
  #
@@ -1268,9 +1359,6 @@ module Woods
1268
1359
 
1269
1360
  def setup_output_directory
1270
1361
  FileUtils.mkdir_p(@output_dir)
1271
- EXTRACTORS.each_key do |type|
1272
- FileUtils.mkdir_p(payload_dir.join(type.to_s))
1273
- end
1274
1362
  end
1275
1363
 
1276
1364
  # ──────────────────────────────────────────────────────────────────────
@@ -1370,8 +1458,10 @@ module Woods
1370
1458
  def same_type_collision_message(type, unit, prior_path)
1371
1459
  "same-type identifier collision: #{type.to_s.singularize} '#{unit.identifier}' derived from " \
1372
1460
  "two different sources ('#{prior_path || 'no file'}' and '#{unit.file_path || 'no file'}'); " \
1373
- 'only one unit could ever be indexed, so extraction aborted — either merge the ' \
1374
- 'declarations into one file or split them into distinct constants'
1461
+ 'only one unit could ever be indexed, so extraction aborted. ' \
1462
+ 'Wrapper-nested class naming requires Zeitwerk mode with Zeitwerk >= 2.6.9; on older loaders or ' \
1463
+ 'classic-mode hosts, check that support before changing valid namespace wrappers. ' \
1464
+ 'For a genuine duplicate, merge the declarations into one file or split them into distinct constants'
1375
1465
  end
1376
1466
 
1377
1467
  # ──────────────────────────────────────────────────────────────────────
@@ -1753,6 +1843,7 @@ module Woods
1753
1843
  GraphAnalyzer.new(
1754
1844
  @dependency_graph,
1755
1845
  volatile_ratio: ratio,
1846
+ volatile_limit_per_target: config&.volatile_dependency_limit_per_target,
1756
1847
  cycle_limit: config ? config.graph_cycle_limit : GraphAnalyzer::DEFAULT_CYCLE_LIMIT,
1757
1848
  cycle_max_length: config ? config.graph_cycle_max_length : GraphAnalyzer::DEFAULT_CYCLE_MAX_LENGTH
1758
1849
  )
@@ -1869,11 +1960,9 @@ module Woods
1869
1960
  # Is this a path worth asking git about?
1870
1961
  #
1871
1962
  # A gem-owned unit (an engine model) carries its real path. Outside
1872
- # Rails.root, git refuses the whole `log` invocation when any pathspec is
1873
- # outside the repository — one gem path would erase the git metadata of
1874
- # the other 499 units in its 500-path batch. Inside Rails.root, a bundle
1963
+ # Rails.root, it has no app repository history. Inside Rails.root, a bundle
1875
1964
  # vendored at `vendor/bundle` puts the same gem files under the root
1876
- # prefix, gitignored, so sending them is wasted pathspec work every run.
1965
+ # prefix, gitignored, so requesting their history serves no app-owned unit.
1877
1966
  # Same exclusions as {Extractors::SharedUtilityMethods#app_source?}.
1878
1967
  #
1879
1968
  # @param path [String, nil] absolute file path
@@ -1927,13 +2016,13 @@ module Woods
1927
2016
  # exactly this value shape — no fractional seconds, `Z` or a `±hh:mm`
1928
2017
  # offset — and `spec/extracted_unit_spec.rb` pins that, so a change to the
1929
2018
  # stamp's shape fails a spec instead of quietly un-matching this mask.
1930
- # The value constraint is what keeps the mask honest against user code: a
1931
- # bare `"extracted_at":` cannot occur inside any JSON *string* value
1932
- # (interior quotes serialize as `\"`), so only a real JSON key can match,
1933
- # and only when it holds a timestamp — which no extractor emits below the
1934
- # top level.
2019
+ # Match only the final top-level stamp, followed by the source_hash field
2020
+ # and the document's closing brace. Nested metadata may use the same key
2021
+ # and timestamp shape; changing it must still rewrite the unit. Escaped
2022
+ # quotes inside string values cannot match these JSON field boundaries.
1935
2023
  EXTRACTED_AT_SCALAR =
1936
- /("extracted_at":\s*")\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}(?:Z|[+-]\d{2}:\d{2})(?=")/
2024
+ /("extracted_at":\s*")\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}(?:Z|[+-]\d{2}:\d{2})
2025
+ (?=",\s*"source_hash":\s*"[0-9a-f]{64}"\s*}\s*\z)/x
1937
2026
  # An implementation detail of the byte comparison, not part of the
1938
2027
  # extractor's surface (`private` does not scope constants).
1939
2028
  private_constant :EXTRACTED_AT_SCALAR
@@ -1967,7 +2056,7 @@ module Woods
1967
2056
  # original encoding
1968
2057
  # @return [String] the bytes with the stamp's value removed
1969
2058
  def mask_extracted_at(bytes)
1970
- bytes.gsub(EXTRACTED_AT_SCALAR, '\1')
2059
+ bytes.sub(EXTRACTED_AT_SCALAR, '\1')
1971
2060
  end
1972
2061
 
1973
2062
  def normalize_file_paths
@@ -1999,7 +2088,7 @@ module Woods
1999
2088
  # to say. Enrichment then wrote `commit_count: 0` and
2000
2089
  # `change_frequency: new` onto every unit, which reads exactly like a file
2001
2090
  # that was never committed, where an absent git directory correctly omits
2002
- # the keys (B-186). HEAD has to resolve.
2091
+ # the keys (B-186). HEAD has to resolve, with complete ancestry (B-189).
2003
2092
  #
2004
2093
  # Memoized, so the warning below is emitted at most once per run.
2005
2094
  #
@@ -2008,13 +2097,31 @@ module Woods
2008
2097
  return @git_available if defined?(@git_available)
2009
2098
 
2010
2099
  _output, error, status = Open3.capture3(*git_argv('rev-parse', 'HEAD'))
2011
- @git_available = status.success?
2012
- warn_unresolvable_git(error) unless @git_available
2013
- @git_available
2100
+ unless status.success?
2101
+ warn_unresolvable_git(error)
2102
+ return @git_available = false
2103
+ end
2104
+
2105
+ @git_available = complete_git_history?
2014
2106
  rescue StandardError
2015
2107
  @git_available = false
2016
2108
  end
2017
2109
 
2110
+ # A shallow HEAD resolves but represents an incomplete ancestry. Do not
2111
+ # turn that boundary into apparent one-commit/new-file churn facts.
2112
+ def complete_git_history?
2113
+ output, _error, status = Open3.capture3(*git_argv('rev-parse', '--is-shallow-repository'))
2114
+ return true if status.success? && output.strip == 'false'
2115
+
2116
+ shallow = status.success? && output.strip == 'true'
2117
+ reason = shallow ? 'shallow repository' : 'repository depth could not be verified'
2118
+ Rails.logger.warn(
2119
+ "[Woods] Git enrichment omitted: #{reason}. Fetch full history with git fetch --unshallow " \
2120
+ '(or actions/checkout fetch-depth: 0), then run full extraction to refresh git metadata.'
2121
+ )
2122
+ false
2123
+ end
2124
+
2018
2125
  # Say once why no unit will carry git metadata, but only when there is a
2019
2126
  # working tree to explain. No `.git` at the root is the ordinary source
2020
2127
  # tarball or `COPY`-without-`.git` case, and it is not a fault.
@@ -2058,63 +2165,19 @@ module Woods
2058
2165
  ''
2059
2166
  end
2060
2167
 
2061
- # Batch-fetch git data for all file paths in two git commands.
2062
- #
2063
- # Duplicate paths are collapsed before slicing (audit P9d): many units
2064
- # share one file_path, and duplicates only repeat a pathspec another
2065
- # batch also sent. The result is keyed by relative path, so the output
2066
- # is identical.
2067
- #
2068
- # @param file_paths [Array<String>] Absolute file paths
2069
- # @return [Hash{String => Hash}] Keyed by relative path
2168
+ # One HEAD history walk, independent of requested-path grouping. See
2169
+ # GitHistory for explicit merge semantics and binary record framing.
2170
+ # @param file_paths [Array<String>] absolute file paths
2171
+ # @return [Hash{String => Hash}] keyed by Rails.root-relative path
2070
2172
  def batch_git_data(file_paths)
2071
2173
  return {} if file_paths.empty?
2072
2174
 
2073
- root = "#{Rails.root}/"
2074
- relative_paths = file_paths.map { |f| f.sub(root, '') }.uniq
2075
- result = {}
2076
- relative_paths.each { |rp| result[rp] = {} }
2077
-
2078
- path_set = relative_paths.to_set
2079
- relative_paths.each_slice(500) do |batch|
2080
- log_output = run_git(
2081
- 'log', '--all', '--name-only',
2082
- '--format=__COMMIT__%H|||%an|||%cI|||%s',
2083
- '--since=365 days ago',
2084
- '--', *batch
2085
- )
2086
- parse_git_log_output(log_output, path_set, result)
2087
- end
2088
-
2089
- ninety_days_ago = (Time.current - 90.days).iso8601
2090
- result.each do |relative_path, data|
2091
- result[relative_path] = build_file_metadata(data, ninety_days_ago)
2092
- end
2093
-
2094
- result
2095
- end
2096
-
2097
- # Parse git log output line-by-line, populating result with per-file commit data.
2098
- def parse_git_log_output(log_output, path_set, result)
2099
- current_commit = nil
2100
-
2101
- log_output.each_line do |line|
2102
- line = line.strip
2103
- next if line.empty?
2175
+ relative_paths = file_paths.map { |path| normalize_file_path(path) }.uniq
2176
+ recent_after = Time.current - 90.days
2177
+ raw = GitHistory.new(root: Rails.root, logger: Rails.logger).read(relative_paths, recent_after: recent_after)
2178
+ return {} unless raw
2104
2179
 
2105
- if line.start_with?('__COMMIT__')
2106
- parts = line.sub('__COMMIT__', '').split('|||', 4)
2107
- current_commit = { sha: parts[0], author: parts[1], date: parts[2], message: parts[3] }
2108
- elsif current_commit && path_set.include?(line)
2109
- entry = result[line] ||= {}
2110
- unless entry[:last_modified]
2111
- entry[:last_modified] = current_commit[:date]
2112
- entry[:last_author] = current_commit[:author]
2113
- end
2114
- (entry[:commits] ||= []) << current_commit
2115
- (entry[:contributors] ||= Hash.new(0))[current_commit[:author]] += 1
2116
- end
2117
- end
2180
+ raw.transform_values { |data| build_file_metadata(data, recent_after.iso8601) }
2118
2181
  end
2119
2182
 
2120
2183
  # Classify how frequently a file changes based on commit counts.
@@ -2136,12 +2199,13 @@ module Woods
2136
2199
  def build_file_metadata(data, ninety_days_ago)
2137
2200
  all_commits = data[:commits] || []
2138
2201
  contributor_counts = data[:contributors] || {}
2139
- recent_count = all_commits.count { |c| c[:date] && c[:date] > ninety_days_ago }
2202
+ recent_count = data.fetch(:recent_count) { all_commits.count { |c| c[:date] && c[:date] > ninety_days_ago } }
2203
+ total_count = data.fetch(:commit_count, all_commits.size)
2140
2204
 
2141
2205
  {
2142
2206
  last_modified: data[:last_modified],
2143
2207
  last_author: data[:last_author],
2144
- commit_count: all_commits.size,
2208
+ commit_count: total_count,
2145
2209
  contributors: contributor_counts
2146
2210
  .sort_by { |_, count| -count }
2147
2211
  .first(5)
@@ -2149,7 +2213,7 @@ module Woods
2149
2213
  recent_commits: all_commits.first(5).map do |c|
2150
2214
  { sha: c[:sha]&.first(8), message: c[:message], date: c[:date], author: c[:author] }
2151
2215
  end,
2152
- change_frequency: classify_change_frequency(all_commits.size, recent_count)
2216
+ change_frequency: classify_change_frequency(total_count, recent_count)
2153
2217
  }
2154
2218
  end
2155
2219
 
@@ -2324,6 +2388,7 @@ module Woods
2324
2388
 
2325
2389
  manifest = {
2326
2390
  extracted_at: Time.current.iso8601,
2391
+ woods_version: Woods::VERSION,
2327
2392
  rails_version: Rails.version,
2328
2393
  ruby_version: RUBY_VERSION,
2329
2394
 
@@ -2697,6 +2762,7 @@ module Woods
2697
2762
 
2698
2763
  @incremental_extractors[key] = EXTRACTORS[key]&.new
2699
2764
  rescue StandardError => e
2765
+ @source_inputs&.unverified("extractor:#{key}")
2700
2766
  Rails.logger.warn "[Woods] Could not build #{key} extractor: #{e.message}"
2701
2767
  @incremental_extractors[key] = nil
2702
2768
  end
@@ -2719,10 +2785,11 @@ module Woods
2719
2785
  #
2720
2786
  # This is the fix for #164 gap 1 (a path the index has never seen routed
2721
2787
  # nowhere, so new files were silently ignored) and the per-path half of
2722
- # gap 3 (a file that defines several units — a `.rake` file with multiple
2723
- # tasks, an i18n YAML — could only ever resolve to one identifier).
2724
- # Reconciling the whole path at once means a task deleted from a
2725
- # multi-task file is removed rather than left behind.
2788
+ # gap 3 (a file defining several units could only ever resolve to one
2789
+ # identifier). Rake tasks need the wholesale path because definitions
2790
+ # of one task can span several files.
2791
+ # Reconciling the whole path removes definitions deleted from a surviving
2792
+ # file rather than leaving them behind.
2726
2793
  #
2727
2794
  # Removal is scoped to the unit types the matching rules could have
2728
2795
  # produced, so a class-based unit sharing the path (the `User` model unit
@@ -2758,6 +2825,10 @@ module Woods
2758
2825
  # one method over — CORE-1).
2759
2826
  produced.merge(units.map { |unit| [unit.identifier, unit.type] })
2760
2827
  touched.merge(register_and_write(rule.extractor_key, units, affected_types))
2828
+ unless source_consumer_failed?(rule.extractor_key)
2829
+ @source_inputs&.consume_file(rule.extractor_key,
2830
+ absolute_path)
2831
+ end
2761
2832
  end
2762
2833
 
2763
2834
  next if raised
@@ -2786,10 +2857,13 @@ module Woods
2786
2857
  # on every changed path of that type, with the generation bumped over
2787
2858
  # the loss. Construction failure tells us nothing about the path; only
2788
2859
  # a genuinely constructed extractor that lacks the method earns the [].
2789
- return nil if extractor.nil?
2860
+ if extractor.nil?
2861
+ source_consumer_failed?(rule.extractor_key, extractor)
2862
+ return nil
2863
+ end
2790
2864
  return [] unless extractor.respond_to?(rule.method_name)
2791
2865
 
2792
- result =
2866
+ result = checked_extraction(rule.extractor_key, extractor) do
2793
2867
  if rule.extractor_key == :poros
2794
2868
  # PoroExtractor needs the AR name set to reject persisted models;
2795
2869
  # its default is an empty set, which would misfile every model
@@ -2798,9 +2872,12 @@ module Woods
2798
2872
  else
2799
2873
  extractor.public_send(rule.method_name, absolute_path)
2800
2874
  end
2875
+ end
2876
+ return nil if result.nil? && SourceInputs::ConsumerErrors.failed?(extractor)
2801
2877
 
2802
2878
  Array(result).compact
2803
2879
  rescue StandardError => e
2880
+ @source_inputs&.unverified("extractor:#{rule.extractor_key}")
2804
2881
  Rails.logger.warn "[Woods] #{rule.extractor_key} re-extraction of #{absolute_path} failed: #{e.message}"
2805
2882
  # `nil`, not `[]`. The caller treats an empty result as "this path defines
2806
2883
  # nothing any more" and prunes the units previously registered to it — so
@@ -2871,6 +2948,34 @@ module Woods
2871
2948
  touched
2872
2949
  end
2873
2950
 
2951
+ # Runtime-only model mixins can enter or leave discovery when their
2952
+ # includer changes, even if the mixin file itself is untouched.
2953
+ # @param affected_types [Set<Symbol>]
2954
+ # @return [Set<String>] Added or removed concern identifiers
2955
+ def reconcile_model_mixins(affected_types)
2956
+ extractor = extractor_for(:concerns)
2957
+ return Set.new unless extractor.respond_to?(:runtime_model_mixins)
2958
+
2959
+ live = extractor.runtime_model_mixins
2960
+ known = @dependency_graph.units_of_type(:concern).to_set
2961
+ added = live.flat_map do |path, modules|
2962
+ next [] if modules.all? { |mod| known.include?(mod.name) }
2963
+
2964
+ Array(extractor.extract_model_mixin_file(path)).reject { |unit| known.include?(unit.identifier) }
2965
+ end
2966
+ touched = register_and_write(:concerns, added, affected_types)
2967
+ return touched unless @eager_load_complete
2968
+
2969
+ live_names = live.values.flatten.to_set(&:name)
2970
+ known.each do |identifier|
2971
+ path = @dependency_graph.node(identifier, type: :concern)[:file_path]
2972
+ next if extractor.conventional_concern_path?(path) || live_names.include?(identifier)
2973
+
2974
+ touched.add(identifier) if remove_unit(identifier, affected_types, type: :concern)
2975
+ end
2976
+ touched
2977
+ end
2978
+
2874
2979
  # Pruned class-based identifiers the tree still governs, and that the
2875
2980
  # second reconciliation pass may therefore re-add.
2876
2981
  #
@@ -2948,10 +3053,12 @@ module Woods
2948
3053
  units = new_classes.filter_map do |klass|
2949
3054
  extractor_for(key).public_send(spec[:method], klass)
2950
3055
  rescue StandardError => e
3056
+ @source_inputs&.unverified("extractor:#{key}")
2951
3057
  Rails.logger.warn "[Woods] #{key} extraction of #{klass} failed: #{e.message}"
2952
3058
  nil
2953
3059
  end
2954
3060
 
3061
+ source_consumer_failed?(key)
2955
3062
  register_and_write(key, units, affected_types)
2956
3063
  end
2957
3064
 
@@ -3086,14 +3193,23 @@ module Woods
3086
3193
  # mutating durable state
3087
3194
  def replace_type_wholesale(key, affected_types)
3088
3195
  extractor = extractor_for(key)
3196
+ if extractor.nil?
3197
+ source_consumer_failed?(key, extractor)
3198
+ return Set.new
3199
+ end
3089
3200
  return Set.new unless extractor.respond_to?(:extract_all)
3090
3201
 
3091
3202
  @wholesale_mutations = 0
3092
- units = Array(extractor.extract_all).compact.uniq(&:identifier)
3203
+ result = checked_extraction(key, extractor) { extractor.extract_all }
3204
+ return Set.new if SourceInputs::ConsumerErrors.failed?(extractor)
3205
+
3206
+ units = Array(result).compact.uniq(&:identifier)
3093
3207
  Rails.logger.info "[Woods] Re-ran #{key} wholesale: #{units.size} units"
3094
3208
 
3095
3209
  touched = register_and_write(key, units, affected_types)
3096
3210
  touched.merge(remove_replaced_units(key, units, affected_types))
3211
+ @source_inputs&.consume_extractor(key, units) unless source_consumer_failed?(key, extractor)
3212
+ touched
3097
3213
  rescue StandardError => e
3098
3214
  if @wholesale_mutations.to_i.positive?
3099
3215
  raise Woods::ExtractionError, <<~MSG.tr("\n", ' ').strip
@@ -3106,6 +3222,7 @@ module Woods
3106
3222
  MSG
3107
3223
  end
3108
3224
 
3225
+ @source_inputs&.unverified("extractor:#{key}")
3109
3226
  Rails.logger.error "[Woods] Wholesale re-run of #{key} failed: #{e.message}"
3110
3227
  Set.new
3111
3228
  end
@@ -3267,6 +3384,7 @@ module Woods
3267
3384
 
3268
3385
  removed.add(identifier) if remove_unit(identifier, affected_types, type: type)
3269
3386
  end
3387
+ @source_inputs&.consume_deleted(path)
3270
3388
  end
3271
3389
  end
3272
3390
 
@@ -3370,6 +3488,7 @@ module Woods
3370
3488
  (@incremental_written ||= {})[unit.identifier] = unit.file_path
3371
3489
 
3372
3490
  write_unit_file(type_dir.join(collision_safe_filename(unit.identifier)), unit)
3491
+ @source_inputs&.consume_unit(extractor_key, unit.file_path) unless source_consumer_failed?(extractor_key)
3373
3492
  written.add(unit.identifier)
3374
3493
  end
3375
3494
  end
@@ -3446,12 +3565,14 @@ module Woods
3446
3565
  def finalize_incremental_unit_json(affected_types)
3447
3566
  dependents_dirty = @dependents_dirty || Set.new
3448
3567
  git_dirty = @incremental_written || {}
3449
- git_data = incremental_git_data(git_dirty.keys)
3568
+ git_data = profile_phase('git enrichment') { incremental_git_data(git_dirty.keys) }
3450
3569
 
3451
- (dependents_dirty | git_dirty.keys).each do |identifier|
3452
- rewrite_unit_json(identifier, affected_types,
3453
- refresh_dependents: dependents_dirty.include?(identifier),
3454
- git_data: git_dirty.key?(identifier) ? git_data : nil)
3570
+ profile_phase('unit finalization') do
3571
+ (dependents_dirty | git_dirty.keys).each do |identifier|
3572
+ rewrite_unit_json(identifier, affected_types,
3573
+ refresh_dependents: dependents_dirty.include?(identifier),
3574
+ git_data: git_dirty.key?(identifier) ? git_data : nil)
3575
+ end
3455
3576
  end
3456
3577
  end
3457
3578
 
@@ -3560,7 +3681,7 @@ module Woods
3560
3681
  end
3561
3682
 
3562
3683
  # Batch-fetch git metadata for the units written by this run, in a single
3563
- # git invocation, keyed by Rails.root-relative path the way
3684
+ # history walk, keyed by Rails.root-relative path the way
3564
3685
  # {#batch_git_data} returns it.
3565
3686
  #
3566
3687
  # @param identifiers [Array<String>]
@@ -3568,11 +3689,12 @@ module Woods
3568
3689
  def incremental_git_data(identifiers)
3569
3690
  return {} if identifiers.empty? || !git_available?
3570
3691
 
3692
+ root = "#{Rails.root}/"
3571
3693
  paths = identifiers.flat_map do |identifier|
3572
3694
  @dependency_graph.nodes_for(identifier).filter_map do |node|
3573
3695
  next if %i[rails_source gem_source].include?(node[:type])
3574
3696
 
3575
- node[:file_path] if node[:file_path] && File.exist?(node[:file_path])
3697
+ node[:file_path] if git_enrichable_path?(node[:file_path], root)
3576
3698
  end
3577
3699
  end
3578
3700
 
@@ -3627,11 +3749,17 @@ module Woods
3627
3749
  return nil unless extractor_key
3628
3750
 
3629
3751
  extractor = extractor_for(extractor_key)
3630
- return nil unless extractor
3752
+ if extractor.nil?
3753
+ source_consumer_failed?(extractor_key, extractor)
3754
+ return nil
3755
+ end
3631
3756
 
3632
- # File-based extractors can return several units from one file (a .rake
3633
- # file defining multiple tasks, etc.); class-based extractors return one.
3634
- units = Array(re_extracted_units(extractor, type, unit_id, file_path, extractor_key)).compact
3757
+ # File-based extractors can return several units from one file;
3758
+ # class-based extractors return one.
3759
+ result = checked_extraction(extractor_key, extractor) do
3760
+ re_extracted_units(extractor, type, unit_id, file_path, extractor_key)
3761
+ end
3762
+ units = Array(result).compact
3635
3763
  return nil if units.empty?
3636
3764
 
3637
3765
  register_and_write(extractor_key, units, affected_types)