woods 2.0.0.beta2 → 2.0.0.beta3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (218) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +262 -1
  3. data/CONTRIBUTING.md +173 -9
  4. data/README.md +7 -3
  5. data/SECURITY.md +9 -6
  6. data/docs/AGENT_GUIDE.md +83 -4
  7. data/docs/AGENT_SETUP.md +82 -1
  8. data/docs/BACKEND_MATRIX.md +20 -0
  9. data/docs/CLIENT_HOOKS.md +111 -0
  10. data/docs/CONFIGURATION_REFERENCE.md +199 -14
  11. data/docs/CONSOLE_MCP_SETUP.md +35 -5
  12. data/docs/DOCKER_SETUP.md +21 -2
  13. data/docs/EVALUATION.md +464 -1
  14. data/docs/EXTRACTOR_REFERENCE.md +36 -5
  15. data/docs/FAQ.md +11 -12
  16. data/docs/GETTING_STARTED.md +17 -5
  17. data/docs/INCREMENTAL_EXTRACTION.md +117 -1
  18. data/docs/INDEX_LAYOUT.md +382 -0
  19. data/docs/INTERNALS.md +7 -2
  20. data/docs/MCP_SERVERS.md +221 -5
  21. data/docs/MCP_TOOL_COOKBOOK.md +33 -18
  22. data/docs/NOTION_INTEGRATION.md +13 -0
  23. data/docs/OBSIDIAN_INTEGRATION.md +57 -9
  24. data/docs/PUBLISHED_INDEX.md +55 -0
  25. data/docs/README.md +7 -0
  26. data/docs/RETRIEVAL_GUIDE.md +253 -11
  27. data/docs/RUNTIME_TRACING.md +71 -0
  28. data/docs/SOURCE_FRESHNESS.md +143 -0
  29. data/docs/TROUBLESHOOTING.md +117 -5
  30. data/docs/UNBLOCKED_INTEGRATION.md +25 -0
  31. data/docs/UPGRADING_TO_2.md +44 -22
  32. data/docs/WATCH_DAEMON.md +259 -59
  33. data/exe/woods-agent-config +6 -0
  34. data/exe/woods-extract +5 -0
  35. data/exe/woods-hook-context +6 -0
  36. data/lib/generators/woods/templates/woods.rb.tt +1 -3
  37. data/lib/tasks/woods.rake +47 -397
  38. data/lib/woods/agent_configuration/applier.rb +133 -0
  39. data/lib/woods/agent_configuration/cli.rb +101 -0
  40. data/lib/woods/agent_configuration/cli_options.rb +29 -0
  41. data/lib/woods/agent_configuration/document.rb +105 -0
  42. data/lib/woods/agent_configuration/error.rb +7 -0
  43. data/lib/woods/agent_configuration/launcher.rb +75 -0
  44. data/lib/woods/agent_configuration/layout.rb +59 -0
  45. data/lib/woods/agent_configuration/managed_section.rb +62 -0
  46. data/lib/woods/agent_configuration/plan.rb +98 -0
  47. data/lib/woods/agent_configuration/plan_diff.rb +38 -0
  48. data/lib/woods/agent_configuration/planned_files.rb +61 -0
  49. data/lib/woods/agent_configuration/planner.rb +63 -0
  50. data/lib/woods/agent_configuration/planner_validation.rb +77 -0
  51. data/lib/woods/agent_configuration/preflight.rb +100 -0
  52. data/lib/woods/agent_configuration/recovery.rb +49 -0
  53. data/lib/woods/ast/node.rb +2 -0
  54. data/lib/woods/ast/parser.rb +38 -5
  55. data/lib/woods/builder.rb +21 -5
  56. data/lib/woods/cache/cache_middleware.rb +28 -7
  57. data/lib/woods/cache/cache_store.rb +4 -5
  58. data/lib/woods/change_set.rb +5 -4
  59. data/lib/woods/console/credential_index.rb +20 -2
  60. data/lib/woods/console/credential_scanner.rb +14 -14
  61. data/lib/woods/console/credential_scanner_registry.rb +36 -0
  62. data/lib/woods/console/embedded_executor.rb +1 -1
  63. data/lib/woods/console/encrypted_credential_snapshot.rb +16 -0
  64. data/lib/woods/console/rack_middleware.rb +22 -13
  65. data/lib/woods/console/server.rb +18 -16
  66. data/lib/woods/dependency_graph.rb +65 -13
  67. data/lib/woods/embedding/corpus.rb +94 -0
  68. data/lib/woods/embedding/indexer.rb +90 -46
  69. data/lib/woods/embedding/openai.rb +17 -6
  70. data/lib/woods/evaluation/ablation_executor.rb +6 -1
  71. data/lib/woods/evaluation/ablation_timed_executor.rb +22 -4
  72. data/lib/woods/export/typed_reader.rb +56 -0
  73. data/lib/woods/extractor.rb +232 -137
  74. data/lib/woods/extractors/action_cable_extractor.rb +3 -1
  75. data/lib/woods/extractors/behavioral_profile.rb +9 -7
  76. data/lib/woods/extractors/caching_extractor.rb +3 -1
  77. data/lib/woods/extractors/concern_extractor.rb +64 -6
  78. data/lib/woods/extractors/configuration_extractor.rb +7 -3
  79. data/lib/woods/extractors/controller_extractor.rb +13 -4
  80. data/lib/woods/extractors/database_view_extractor.rb +3 -1
  81. data/lib/woods/extractors/decorator_extractor.rb +3 -1
  82. data/lib/woods/extractors/engine_extractor.rb +3 -1
  83. data/lib/woods/extractors/event_extractor.rb +4 -2
  84. data/lib/woods/extractors/factory_extractor.rb +3 -1
  85. data/lib/woods/extractors/graphql_extractor.rb +8 -2
  86. data/lib/woods/extractors/i18n_extractor.rb +3 -1
  87. data/lib/woods/extractors/job_extractor.rb +6 -19
  88. data/lib/woods/extractors/lib_extractor.rb +3 -1
  89. data/lib/woods/extractors/mailer_extractor.rb +20 -5
  90. data/lib/woods/extractors/manager_extractor.rb +3 -1
  91. data/lib/woods/extractors/method_parameters.rb +53 -0
  92. data/lib/woods/extractors/middleware_argument.rb +65 -0
  93. data/lib/woods/extractors/middleware_extractor.rb +9 -3
  94. data/lib/woods/extractors/migration_extractor.rb +3 -1
  95. data/lib/woods/extractors/model_extractor.rb +39 -33
  96. data/lib/woods/extractors/package_extractor.rb +24 -4
  97. data/lib/woods/extractors/phlex_extractor.rb +3 -1
  98. data/lib/woods/extractors/policy_extractor.rb +3 -1
  99. data/lib/woods/extractors/poro_extractor.rb +3 -1
  100. data/lib/woods/extractors/pundit_extractor.rb +3 -1
  101. data/lib/woods/extractors/rails_source_extractor.rb +4 -2
  102. data/lib/woods/extractors/rake_task_extractor.rb +4 -2
  103. data/lib/woods/extractors/route_extractor.rb +3 -1
  104. data/lib/woods/extractors/route_helper_resolver.rb +10 -33
  105. data/lib/woods/extractors/scheduled_job_extractor.rb +41 -15
  106. data/lib/woods/extractors/serializer_extractor.rb +4 -2
  107. data/lib/woods/extractors/service_extractor.rb +3 -1
  108. data/lib/woods/extractors/shared_dependency_scanner.rb +2 -2
  109. data/lib/woods/extractors/shared_utility_methods.rb +27 -15
  110. data/lib/woods/extractors/source_nesting.rb +1 -1
  111. data/lib/woods/extractors/state_machine_extractor.rb +3 -1
  112. data/lib/woods/extractors/test_mapping_extractor.rb +3 -1
  113. data/lib/woods/extractors/validator_extractor.rb +3 -1
  114. data/lib/woods/extractors/view_component_extractor.rb +3 -1
  115. data/lib/woods/extractors/view_template_extractor.rb +3 -1
  116. data/lib/woods/gem_mapper.rb +2 -0
  117. data/lib/woods/git_history.rb +116 -0
  118. data/lib/woods/graph_analyzer.rb +35 -6
  119. data/lib/woods/hooks/context_cli.rb +54 -0
  120. data/lib/woods/hooks/context_event.rb +88 -0
  121. data/lib/woods/hooks/context_hint.rb +73 -0
  122. data/lib/woods/hooks/context_impact.rb +77 -0
  123. data/lib/woods/hooks/context_output.rb +47 -0
  124. data/lib/woods/hooks/context_state.rb +102 -0
  125. data/lib/woods/hooks/refresh.rb +79 -0
  126. data/lib/woods/hooks/rule_projection.rb +78 -0
  127. data/lib/woods/input_rules.rb +19 -0
  128. data/lib/woods/mcp/bearer_auth.rb +20 -12
  129. data/lib/woods/mcp/bootstrapper.rb +62 -0
  130. data/lib/woods/mcp/index_reader.rb +323 -160
  131. data/lib/woods/mcp/initialization_guidance.rb +27 -0
  132. data/lib/woods/mcp/origin_guard.rb +17 -9
  133. data/lib/woods/mcp/published_lexical_retriever.rb +115 -0
  134. data/lib/woods/mcp/renderers/markdown_renderer.rb +8 -1
  135. data/lib/woods/mcp/renderers/plain_renderer.rb +7 -1
  136. data/lib/woods/mcp/search_results.rb +74 -0
  137. data/lib/woods/mcp/server.rb +158 -37
  138. data/lib/woods/mcp/tool_contract.rb +2 -0
  139. data/lib/woods/mcp/tool_response_renderer.rb +25 -0
  140. data/lib/woods/mcp/traversal_evidence.rb +113 -0
  141. data/lib/woods/mcp/traversal_evidence_index.rb +100 -0
  142. data/lib/woods/mcp/traversal_evidence_page.rb +41 -0
  143. data/lib/woods/mcp/traversal_evidence_text.rb +52 -0
  144. data/lib/woods/notion/exporter.rb +56 -17
  145. data/lib/woods/obsidian/destination_plan.rb +98 -0
  146. data/lib/woods/obsidian/name_mapper.rb +19 -3
  147. data/lib/woods/obsidian/note_builder.rb +19 -10
  148. data/lib/woods/obsidian/vault_exporter.rb +88 -32
  149. data/lib/woods/operator/pipeline_guard.rb +18 -13
  150. data/lib/woods/path_dispatcher.rb +7 -1
  151. data/lib/woods/payload_store.rb +27 -26
  152. data/lib/woods/railtie.rb +3 -3
  153. data/lib/woods/railtie_support.rb +12 -12
  154. data/lib/woods/rake_helpers.rb +392 -0
  155. data/lib/woods/resilience/graph_invariant_validator/membership_checks.rb +71 -0
  156. data/lib/woods/resilience/graph_invariant_validator/node_checks.rb +61 -0
  157. data/lib/woods/resilience/graph_invariant_validator/reverse_relationship_checks.rb +46 -0
  158. data/lib/woods/resilience/graph_invariant_validator.rb +119 -0
  159. data/lib/woods/resilience/index_validator/graph_checks.rb +80 -0
  160. data/lib/woods/resilience/index_validator.rb +112 -23
  161. data/lib/woods/retrieval/context_assembler.rb +50 -15
  162. data/lib/woods/retrieval/lexical_assembler.rb +73 -0
  163. data/lib/woods/retrieval/lexical_index.rb +119 -0
  164. data/lib/woods/retrieval/ranker.rb +4 -2
  165. data/lib/woods/retrieval/scope.rb +108 -0
  166. data/lib/woods/retrieval/scoped_graph_store.rb +32 -0
  167. data/lib/woods/retrieval/scoped_vector_store.rb +55 -0
  168. data/lib/woods/retrieval/search_executor.rb +86 -27
  169. data/lib/woods/retrieval/source_evidence.rb +200 -0
  170. data/lib/woods/retriever.rb +98 -22
  171. data/lib/woods/ruby_analyzer/trace_enricher.rb +77 -38
  172. data/lib/woods/session_tracer/middleware.rb +10 -12
  173. data/lib/woods/session_tracer/redis_store.rb +22 -6
  174. data/lib/woods/session_tracer/session_flow_assembler.rb +23 -17
  175. data/lib/woods/session_tracer/solid_cache_coordination.rb +6 -4
  176. data/lib/woods/session_tracer/unit_resolver.rb +63 -0
  177. data/lib/woods/source_inputs/consumer_errors.rb +27 -0
  178. data/lib/woods/source_inputs/handoff.rb +102 -0
  179. data/lib/woods/source_inputs/launcher.rb +157 -0
  180. data/lib/woods/source_inputs/manifest.rb +124 -0
  181. data/lib/woods/source_inputs/private_key.rb +55 -0
  182. data/lib/woods/source_inputs/scanner.rb +171 -0
  183. data/lib/woods/source_inputs/scopes.rb +71 -0
  184. data/lib/woods/source_inputs/session.rb +214 -0
  185. data/lib/woods/source_inputs/status.rb +84 -0
  186. data/lib/woods/source_inputs/verifier.rb +107 -0
  187. data/lib/woods/storage/metadata_store.rb +25 -25
  188. data/lib/woods/storage/pgvector.rb +29 -8
  189. data/lib/woods/storage/qdrant.rb +17 -7
  190. data/lib/woods/storage/vector_store.rb +18 -6
  191. data/lib/woods/tasks.rb +3 -2
  192. data/lib/woods/temporal/json_snapshot_store.rb +29 -8
  193. data/lib/woods/unblocked/exporter.rb +59 -70
  194. data/lib/woods/version.rb +1 -1
  195. data/lib/woods/watch/boot_snapshot.rb +52 -0
  196. data/lib/woods/watch/daemon.rb +136 -28
  197. data/lib/woods/watch/listen_watcher.rb +4 -0
  198. data/lib/woods/watch/polling_watcher.rb +5 -1
  199. data/lib/woods/watch/status.rb +20 -15
  200. data/lib/woods/watch/tree_scan.rb +21 -13
  201. data/lib/woods/watch/watcher.rb +4 -1
  202. data/lib/woods.rb +50 -11
  203. data/plugin/.claude-plugin/plugin.json +1 -1
  204. data/plugin/hooks/adapters/normalize.jq +15 -0
  205. data/plugin/hooks/adapters/normalize.rb +63 -0
  206. data/plugin/hooks/hooks.json +20 -0
  207. data/plugin/hooks/woods-context.sh +50 -0
  208. data/plugin/hooks/woods-input-rules.sh +159 -0
  209. data/plugin/hooks/woods-opencode.mjs +65 -0
  210. data/plugin/hooks/woods-post-edit.sh +2 -225
  211. data/plugin/hooks/woods-refresh.sh +260 -0
  212. data/plugin/hooks/woods-session-start.sh +47 -55
  213. data/plugin/skills/woods-agent-enable/SKILL.md +13 -0
  214. data/plugin/skills/woods-diagnose/SKILL.md +288 -1
  215. data/plugin/skills/woods-investigate/SKILL.md +106 -0
  216. data/plugin/skills/woods-mcp-config/SKILL.md +89 -1
  217. data/plugin/skills/woods-setup/SKILL.md +107 -6
  218. metadata +84 -5
@@ -8,12 +8,14 @@ require 'pathname'
8
8
  require 'set'
9
9
 
10
10
  require_relative 'atomic_file'
11
+ require_relative 'version'
11
12
  require_relative 'filename_utils'
12
13
  require_relative 'token_utils'
13
14
  require_relative 'extracted_unit'
14
15
  require_relative 'dependency_graph'
15
16
  require_relative 'payload_store'
16
17
  require_relative 'git_provenance'
18
+ require_relative 'git_history'
17
19
  require_relative 'extractors/model_extractor'
18
20
  require_relative 'extractors/controller_extractor'
19
21
  require_relative 'extractors/phlex_extractor'
@@ -55,6 +57,7 @@ require_relative 'flow_precomputer'
55
57
  require_relative 'change_set'
56
58
  require_relative 'generation'
57
59
  require_relative 'path_dispatcher'
60
+ require_relative 'source_inputs/session'
58
61
 
59
62
  module Woods
60
63
  # Extractor is the main orchestrator for codebase extraction.
@@ -355,7 +358,7 @@ module Woods
355
358
  # flat index — the output root also holds `generation.json`, `dumps/`,
356
359
  # `tasks/`, `woods.sqlite3` and `payloads/` itself, none of which belong
357
360
  # to a generation's payload.
358
- PAYLOAD_FILES = %w[manifest.json dependency_graph.json graph_analysis.json SUMMARY.md].freeze
361
+ PAYLOAD_FILES = %w[manifest.json source_inputs.json dependency_graph.json graph_analysis.json SUMMARY.md].freeze
359
362
 
360
363
  # Payload directories that are not per-type unit directories.
361
364
  PAYLOAD_DIRS = %w[flows].freeze
@@ -394,7 +397,9 @@ module Woods
394
397
  #
395
398
  # @return [Hash] Results keyed by extractor type
396
399
  def extract_all
400
+ profile_started = Process.clock_gettime(Process::CLOCK_MONOTONIC) if profiling?
397
401
  setup_output_directory
402
+ profile_phase('source capture') { begin_source_inputs('full') }
398
403
  ModelNameCache.reset!
399
404
  # @package_resolver alone is not enough: #package_resolver builds
400
405
  # through #extractor_for, which memoizes into @incremental_extractors.
@@ -422,29 +427,33 @@ module Woods
422
427
 
423
428
  # Phase 1.5: Deduplicate results
424
429
  Rails.logger.info '[Woods] Deduplicating results...'
425
- deduplicate_results
430
+ profile_phase('deduplication') { deduplicate_results }
426
431
 
427
432
  # Phase 1.6: Package membership. Runs before the graph is rebuilt so
428
433
  # registration copies metadata[:package] onto the node (#280).
429
- annotate_packages
434
+ profile_phase('package annotation') { annotate_packages }
430
435
 
431
436
  # Rebuild the graph from deduped results. #164 gave DependencyGraph
432
437
  # `#remove`/`#unregister`, so surgical removal is now possible — but a
433
438
  # full extraction has just registered every unit including duplicates,
434
439
  # and rebuilding from the deduped set is both cheaper and less
435
440
  # error-prone than unwinding registrations one at a time.
436
- @dependency_graph = DependencyGraph.new
437
- @results.each_value { |units| units.each { |u| @dependency_graph.register(u) } }
441
+ profile_phase('graph rebuild') do
442
+ @dependency_graph = DependencyGraph.new
443
+ @results.each_value { |units| units.each { |u| @dependency_graph.register(u) } }
444
+ end
438
445
 
439
446
  # Phase 2: Resolve dependents (reverse dependencies)
440
447
  Rails.logger.info '[Woods] Resolving dependents...'
441
- resolve_dependents
448
+ profile_phase('dependents') { resolve_dependents }
442
449
 
443
450
  # Phase 3: Enrich with git data. Runs BEFORE analysis now: the
444
451
  # volatile_dependencies report reads commit counts off graph nodes.
445
452
  Rails.logger.info '[Woods] Enriching with git data...'
446
- enrich_with_git_data
447
- annotate_graph_with_git_data
453
+ profile_phase('git enrichment') do
454
+ enrich_with_git_data
455
+ annotate_graph_with_git_data
456
+ end
448
457
 
449
458
  # Phase 4: Graph analysis (PageRank, structural metrics)
450
459
  Rails.logger.info '[Woods] Analyzing dependency graph...'
@@ -452,7 +461,7 @@ module Woods
452
461
 
453
462
  # Phase 4.5: Normalize file_path to relative paths
454
463
  Rails.logger.info '[Woods] Normalizing file paths...'
455
- normalize_file_paths
464
+ profile_phase('path normalization') { normalize_file_paths }
456
465
 
457
466
  # Phase 5: Write output
458
467
  Rails.logger.info '[Woods] Writing output...'
@@ -462,7 +471,7 @@ module Woods
462
471
  # run after write_results — the just-written set is what defines
463
472
  # "legitimate" — and belongs to the full path only; the incremental path
464
473
  # deletes through the graph instead. See {#sweep_orphaned_unit_files}.
465
- sweep_orphaned_unit_files
474
+ profile_phase('orphan sweep') { sweep_orphaned_unit_files }
466
475
 
467
476
  # Phase 5.5: Precompute request flows (opt-in). Must run AFTER
468
477
  # write_results — FlowAssembler loads unit JSON from disk, so running
@@ -485,18 +494,23 @@ module Woods
485
494
  profile_phase('flows') { precompute_flows }
486
495
  end
487
496
 
488
- write_dependency_graph
489
- write_graph_analysis
497
+ profile_phase('graph write') do
498
+ write_dependency_graph
499
+ write_graph_analysis
500
+ end
490
501
  profile_phase('manifest and summary') do
491
502
  write_manifest
492
503
  write_structural_summary
493
504
  end
494
- capture_snapshot
495
- profile_phase('publish') { publish_generation('full') }
505
+ profile_phase('snapshot') { capture_snapshot }
506
+ @source_inputs.full_units(@results, consumers: @extractors)
507
+ publish_generation('full')
496
508
 
497
509
  log_summary
498
510
 
499
511
  @results
512
+ ensure
513
+ log_profile_total('full', profile_started)
500
514
  end
501
515
 
502
516
  # ══════════════════════════════════════════════════════════════════════
@@ -528,7 +542,8 @@ module Woods
528
542
  # @param changed_files [Array<String>] List of changed file paths
529
543
  # @return [Array<String>] Identifiers of units re-extracted, added, or removed
530
544
  def extract_changed(changed_files)
531
- prepare_incremental_run
545
+ profile_started = Process.clock_gettime(Process::CLOCK_MONOTONIC) if profiling?
546
+ prepare_incremental_run(operation: 'incremental')
532
547
 
533
548
  change_set = ChangeSet.new(paths: changed_files, root: Rails.root)
534
549
  affected_types = Set.new
@@ -551,36 +566,39 @@ module Woods
551
566
  acc
552
567
  end
553
568
 
554
- touched.merge(reconcile_class_based_types(affected_types))
555
- touched.merge(rerun_whole_app_extractors(change_set, affected_types))
556
- touched.merge(reannotate_packages(change_set, affected_types))
557
- pruned = prune_vanished_units(change_set, affected_types)
558
- touched.merge(pruned)
559
-
560
- # Reconcile once more, because pruning can un-know a class the first pass
561
- # skipped. A class-based file moved between autoload directories with its
562
- # constant unchanged is still registered under the old path when
563
- # reconciliation runs, so it looks known and is not re-extracted; the
564
- # prune that follows then removes it for its vanished path. This pass
565
- # re-adds it in the same run (M1) instead of leaving the unit missing
566
- # until some later run happens to notice. Idempotent when nothing was
567
- # pruned: the discovery set is compared against the graph, so an
568
- # already-registered class is skipped.
569
- #
570
- # But not everything pruning removed may come back. `except:` keeps the
571
- # *deletion* shape pruned: without a reload, a constant outlives the file
572
- # that defined it — so deleting `app/models/user.rb` prunes `User`, and
573
- # this pass finds `User` still in `ActiveRecord::Base.descendants`.
574
- # Re-registering it would pin the unit to a path that no longer exists,
575
- # and nothing could ever remove it: the sweep excludes class-based units
576
- # and no future change set names that path again. A resident daemon
577
- # processing a batch before its reload hits this every time. What
578
- # separates the two shapes is the filesystem — only pruned identifiers
579
- # that a still-existing file in the change set actually declares are
580
- # re-addable. See {#readdable_pruned_classes}.
581
- touched.merge(reconcile_class_based_types(
582
- affected_types, except: pruned - readdable_pruned_classes(pruned, change_set)
583
- ))
569
+ profile_phase('reconciliation') do
570
+ touched.merge(reconcile_class_based_types(affected_types))
571
+ touched.merge(reconcile_model_mixins(affected_types))
572
+ touched.merge(rerun_whole_app_extractors(change_set, affected_types))
573
+ touched.merge(reannotate_packages(change_set, affected_types))
574
+ pruned = prune_vanished_units(change_set, affected_types)
575
+ touched.merge(pruned)
576
+
577
+ # Reconcile once more, because pruning can un-know a class the first pass
578
+ # skipped. A class-based file moved between autoload directories with its
579
+ # constant unchanged is still registered under the old path when
580
+ # reconciliation runs, so it looks known and is not re-extracted; the
581
+ # prune that follows then removes it for its vanished path. This pass
582
+ # re-adds it in the same run (M1) instead of leaving the unit missing
583
+ # until some later run happens to notice. Idempotent when nothing was
584
+ # pruned: the discovery set is compared against the graph, so an
585
+ # already-registered class is skipped.
586
+ #
587
+ # But not everything pruning removed may come back. `except:` keeps the
588
+ # *deletion* shape pruned: without a reload, a constant outlives the file
589
+ # that defined it — so deleting `app/models/user.rb` prunes `User`, and
590
+ # this pass finds `User` still in `ActiveRecord::Base.descendants`.
591
+ # Re-registering it would pin the unit to a path that no longer exists,
592
+ # and nothing could ever remove it: the sweep excludes class-based units
593
+ # and no future change set names that path again. A resident daemon
594
+ # processing a batch before its reload hits this every time. What
595
+ # separates the two shapes is the filesystem — only pruned identifiers
596
+ # that a still-existing file in the change set actually declares are
597
+ # re-addable. See {#readdable_pruned_classes}.
598
+ touched.merge(reconcile_class_based_types(
599
+ affected_types, except: pruned - readdable_pruned_classes(pruned, change_set)
600
+ ))
601
+ end
584
602
 
585
603
  finalize_incremental_unit_json(affected_types)
586
604
 
@@ -594,6 +612,8 @@ module Woods
594
612
  finalize_incremental_run(touched)
595
613
 
596
614
  touched.to_a
615
+ ensure
616
+ log_profile_total('incremental', profile_started)
597
617
  end
598
618
 
599
619
  # ══════════════════════════════════════════════════════════════════════
@@ -628,6 +648,7 @@ module Woods
628
648
  # extractor
629
649
  # @raise [ArgumentError] when no recognized key is given
630
650
  def refresh(*keys)
651
+ profile_started = Process.clock_gettime(Process::CLOCK_MONOTONIC) if profiling?
631
652
  keys = Array(keys).flatten.map(&:to_sym).uniq
632
653
  known, unknown = keys.partition { |key| EXTRACTORS.key?(key) }
633
654
  raise ArgumentError, "No known extractor in #{keys.inspect}" if known.empty?
@@ -635,17 +656,19 @@ module Woods
635
656
  known += ROUTE_CONSUMER_EXTRACTORS if known.include?(:routes)
636
657
  known.uniq!
637
658
 
638
- prepare_incremental_run
659
+ prepare_incremental_run(operation: 'refresh')
639
660
  affected_types = Set.new
640
661
  touched = known.each_with_object(Set.new) do |key, acc|
641
662
  acc.merge(replace_type_wholesale(key, affected_types))
642
663
  end
643
664
 
644
665
  finalize_incremental_unit_json(affected_types)
645
- affected_types.each { |type_key| regenerate_type_index(type_key) }
666
+ profile_phase('type index') { affected_types.each { |type_key| regenerate_type_index(type_key) } }
646
667
  finalize_incremental_run(touched, reason: "refresh:#{known.sort.join(',')}")
647
668
 
648
669
  { types: known, touched: touched.to_a, unknown: unknown }
670
+ ensure
671
+ log_profile_total('refresh', profile_started)
649
672
  end
650
673
 
651
674
  # Raise when the most recent extraction run wrote a payload but could not
@@ -667,6 +690,15 @@ module Woods
667
690
 
668
691
  private
669
692
 
693
+ # Whole-run wall time, including unprofiled setup and failed runs. This
694
+ # separate log family must never be added to the individual phase times.
695
+ def log_profile_total(name, started)
696
+ return unless started
697
+
698
+ elapsed = Process.clock_gettime(Process::CLOCK_MONOTONIC) - started
699
+ Rails.logger.info "[Woods] [profile total] #{name} in #{elapsed.round(2)}s"
700
+ end
701
+
670
702
  # Time one phase of a run and log how long it took, when WOODS_PROFILE=1.
671
703
  #
672
704
  # The per-extractor lines (see {#extract_all_sequential}) already report
@@ -745,7 +777,8 @@ module Woods
745
777
  #
746
778
  # @return [void]
747
779
  # @raise [Woods::ExtractionError] see {#begin_payload!}
748
- def prepare_incremental_run
780
+ def prepare_incremental_run(operation: 'incremental')
781
+ profile_phase('source capture') { begin_source_inputs(operation) }
749
782
  profile_phase('payload seed') { begin_payload!(strict: true) }
750
783
  graph_path = payload_dir.join('dependency_graph.json')
751
784
  ensure_incremental_baseline!(graph_path)
@@ -795,7 +828,7 @@ module Woods
795
828
  # reading "incremental" after a `woods:refresh[routes]` is being misled
796
829
  # @return [void]
797
830
  def finalize_incremental_run(touched, reason: 'incremental')
798
- write_dependency_graph
831
+ profile_phase('graph write') { write_dependency_graph }
799
832
 
800
833
  if touched.empty?
801
834
  Rails.logger.info '[Woods] Incremental run changed nothing — leaving manifest timestamp untouched'
@@ -808,7 +841,7 @@ module Woods
808
841
  write_manifest(incremental: true)
809
842
  write_structural_summary
810
843
  end
811
- profile_phase('publish') { publish_generation(reason) }
844
+ publish_generation(reason)
812
845
 
813
846
  return unless Woods.configuration.enable_snapshots
814
847
 
@@ -831,9 +864,10 @@ module Woods
831
864
  # Resolve (and if necessary rename) the payload first, so the flush
832
865
  # below covers the directory under the name the pointer will carry.
833
866
  payload = publishable_payload_name(generation)
867
+ profile_phase('source verification') { write_source_inputs } if payload
834
868
  profile_phase('payload sync') { sync_payload }
835
- marker = generation.bump!(reason: reason, payload: payload)
836
- prune_payloads(marker.number)
869
+ marker = profile_phase('publish') { generation.bump!(reason: reason, payload: payload) }
870
+ profile_phase('payload prune') { prune_payloads(marker.number) }
837
871
  marker
838
872
  rescue StandardError => e
839
873
  # A failed bump must not fail the extraction that produced a perfectly
@@ -852,6 +886,38 @@ module Woods
852
886
  nil
853
887
  end
854
888
 
889
+ # Capture before eager loading or extraction; only an explicit fresh-launch
890
+ # handoff can additionally establish the pre-Bundler/Rails boot boundary.
891
+ def begin_source_inputs(operation)
892
+ @source_inputs = SourceInputs::Session.new(root: Rails.root, output_dir: @output_dir,
893
+ baseline_path: source_input_baseline_path,
894
+ operation: operation)
895
+ end
896
+
897
+ def source_input_baseline_path
898
+ generation = Generation.new(output_dir: @output_dir)
899
+ marker = generation.current
900
+ directory = generation.payload_dir(marker)
901
+ return nil unless marker.payload && directory != generation.root
902
+
903
+ directory.join(SourceInputs::Manifest::FILE_NAME)
904
+ rescue TypeError, NoMethodError
905
+ nil
906
+ end
907
+
908
+ def source_consumer_failed?(key, consumer = extractor_for(key))
909
+ failed = consumer.nil? || SourceInputs::ConsumerErrors.failed?(consumer)
910
+ @source_inputs&.unverified("extractor:#{key}") if failed
911
+ failed
912
+ end
913
+
914
+ def write_source_inputs
915
+ return unless @source_inputs
916
+
917
+ manifest = @source_inputs.finish(generation: @payload_generation, eager_load_complete: @eager_load_complete)
918
+ AtomicFile.write(payload_dir.join(SourceInputs::Manifest::FILE_NAME), JSON.pretty_generate(manifest.data))
919
+ end
920
+
855
921
  # Open the payload directory this run publishes into, seeded from the
856
922
  # generation currently on disk.
857
923
  #
@@ -1268,9 +1334,6 @@ module Woods
1268
1334
 
1269
1335
  def setup_output_directory
1270
1336
  FileUtils.mkdir_p(@output_dir)
1271
- EXTRACTORS.each_key do |type|
1272
- FileUtils.mkdir_p(payload_dir.join(type.to_s))
1273
- end
1274
1337
  end
1275
1338
 
1276
1339
  # ──────────────────────────────────────────────────────────────────────
@@ -1370,8 +1433,10 @@ module Woods
1370
1433
  def same_type_collision_message(type, unit, prior_path)
1371
1434
  "same-type identifier collision: #{type.to_s.singularize} '#{unit.identifier}' derived from " \
1372
1435
  "two different sources ('#{prior_path || 'no file'}' and '#{unit.file_path || 'no file'}'); " \
1373
- 'only one unit could ever be indexed, so extraction aborted — either merge the ' \
1374
- 'declarations into one file or split them into distinct constants'
1436
+ 'only one unit could ever be indexed, so extraction aborted. ' \
1437
+ 'Wrapper-nested class naming requires Zeitwerk mode with Zeitwerk >= 2.6.9; on older loaders or ' \
1438
+ 'classic-mode hosts, check that support before changing valid namespace wrappers. ' \
1439
+ 'For a genuine duplicate, merge the declarations into one file or split them into distinct constants'
1375
1440
  end
1376
1441
 
1377
1442
  # ──────────────────────────────────────────────────────────────────────
@@ -1753,6 +1818,7 @@ module Woods
1753
1818
  GraphAnalyzer.new(
1754
1819
  @dependency_graph,
1755
1820
  volatile_ratio: ratio,
1821
+ volatile_limit_per_target: config&.volatile_dependency_limit_per_target,
1756
1822
  cycle_limit: config ? config.graph_cycle_limit : GraphAnalyzer::DEFAULT_CYCLE_LIMIT,
1757
1823
  cycle_max_length: config ? config.graph_cycle_max_length : GraphAnalyzer::DEFAULT_CYCLE_MAX_LENGTH
1758
1824
  )
@@ -1869,11 +1935,9 @@ module Woods
1869
1935
  # Is this a path worth asking git about?
1870
1936
  #
1871
1937
  # A gem-owned unit (an engine model) carries its real path. Outside
1872
- # Rails.root, git refuses the whole `log` invocation when any pathspec is
1873
- # outside the repository — one gem path would erase the git metadata of
1874
- # the other 499 units in its 500-path batch. Inside Rails.root, a bundle
1938
+ # Rails.root, it has no app repository history. Inside Rails.root, a bundle
1875
1939
  # vendored at `vendor/bundle` puts the same gem files under the root
1876
- # prefix, gitignored, so sending them is wasted pathspec work every run.
1940
+ # prefix, gitignored, so requesting their history serves no app-owned unit.
1877
1941
  # Same exclusions as {Extractors::SharedUtilityMethods#app_source?}.
1878
1942
  #
1879
1943
  # @param path [String, nil] absolute file path
@@ -1927,13 +1991,13 @@ module Woods
1927
1991
  # exactly this value shape — no fractional seconds, `Z` or a `±hh:mm`
1928
1992
  # offset — and `spec/extracted_unit_spec.rb` pins that, so a change to the
1929
1993
  # stamp's shape fails a spec instead of quietly un-matching this mask.
1930
- # The value constraint is what keeps the mask honest against user code: a
1931
- # bare `"extracted_at":` cannot occur inside any JSON *string* value
1932
- # (interior quotes serialize as `\"`), so only a real JSON key can match,
1933
- # and only when it holds a timestamp — which no extractor emits below the
1934
- # top level.
1994
+ # Match only the final top-level stamp, followed by the source_hash field
1995
+ # and the document's closing brace. Nested metadata may use the same key
1996
+ # and timestamp shape; changing it must still rewrite the unit. Escaped
1997
+ # quotes inside string values cannot match these JSON field boundaries.
1935
1998
  EXTRACTED_AT_SCALAR =
1936
- /("extracted_at":\s*")\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}(?:Z|[+-]\d{2}:\d{2})(?=")/
1999
+ /("extracted_at":\s*")\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}(?:Z|[+-]\d{2}:\d{2})
2000
+ (?=",\s*"source_hash":\s*"[0-9a-f]{64}"\s*}\s*\z)/x
1937
2001
  # An implementation detail of the byte comparison, not part of the
1938
2002
  # extractor's surface (`private` does not scope constants).
1939
2003
  private_constant :EXTRACTED_AT_SCALAR
@@ -1967,7 +2031,7 @@ module Woods
1967
2031
  # original encoding
1968
2032
  # @return [String] the bytes with the stamp's value removed
1969
2033
  def mask_extracted_at(bytes)
1970
- bytes.gsub(EXTRACTED_AT_SCALAR, '\1')
2034
+ bytes.sub(EXTRACTED_AT_SCALAR, '\1')
1971
2035
  end
1972
2036
 
1973
2037
  def normalize_file_paths
@@ -1999,7 +2063,7 @@ module Woods
1999
2063
  # to say. Enrichment then wrote `commit_count: 0` and
2000
2064
  # `change_frequency: new` onto every unit, which reads exactly like a file
2001
2065
  # that was never committed, where an absent git directory correctly omits
2002
- # the keys (B-186). HEAD has to resolve.
2066
+ # the keys (B-186). HEAD has to resolve, with complete ancestry (B-189).
2003
2067
  #
2004
2068
  # Memoized, so the warning below is emitted at most once per run.
2005
2069
  #
@@ -2008,13 +2072,31 @@ module Woods
2008
2072
  return @git_available if defined?(@git_available)
2009
2073
 
2010
2074
  _output, error, status = Open3.capture3(*git_argv('rev-parse', 'HEAD'))
2011
- @git_available = status.success?
2012
- warn_unresolvable_git(error) unless @git_available
2013
- @git_available
2075
+ unless status.success?
2076
+ warn_unresolvable_git(error)
2077
+ return @git_available = false
2078
+ end
2079
+
2080
+ @git_available = complete_git_history?
2014
2081
  rescue StandardError
2015
2082
  @git_available = false
2016
2083
  end
2017
2084
 
2085
+ # A shallow HEAD resolves but represents an incomplete ancestry. Do not
2086
+ # turn that boundary into apparent one-commit/new-file churn facts.
2087
+ def complete_git_history?
2088
+ output, _error, status = Open3.capture3(*git_argv('rev-parse', '--is-shallow-repository'))
2089
+ return true if status.success? && output.strip == 'false'
2090
+
2091
+ shallow = status.success? && output.strip == 'true'
2092
+ reason = shallow ? 'shallow repository' : 'repository depth could not be verified'
2093
+ Rails.logger.warn(
2094
+ "[Woods] Git enrichment omitted: #{reason}. Fetch full history with git fetch --unshallow " \
2095
+ '(or actions/checkout fetch-depth: 0), then run full extraction to refresh git metadata.'
2096
+ )
2097
+ false
2098
+ end
2099
+
2018
2100
  # Say once why no unit will carry git metadata, but only when there is a
2019
2101
  # working tree to explain. No `.git` at the root is the ordinary source
2020
2102
  # tarball or `COPY`-without-`.git` case, and it is not a fault.
@@ -2058,63 +2140,19 @@ module Woods
2058
2140
  ''
2059
2141
  end
2060
2142
 
2061
- # Batch-fetch git data for all file paths in two git commands.
2062
- #
2063
- # Duplicate paths are collapsed before slicing (audit P9d): many units
2064
- # share one file_path, and duplicates only repeat a pathspec another
2065
- # batch also sent. The result is keyed by relative path, so the output
2066
- # is identical.
2067
- #
2068
- # @param file_paths [Array<String>] Absolute file paths
2069
- # @return [Hash{String => Hash}] Keyed by relative path
2143
+ # One HEAD history walk, independent of requested-path grouping. See
2144
+ # GitHistory for explicit merge semantics and binary record framing.
2145
+ # @param file_paths [Array<String>] absolute file paths
2146
+ # @return [Hash{String => Hash}] keyed by Rails.root-relative path
2070
2147
  def batch_git_data(file_paths)
2071
2148
  return {} if file_paths.empty?
2072
2149
 
2073
- root = "#{Rails.root}/"
2074
- relative_paths = file_paths.map { |f| f.sub(root, '') }.uniq
2075
- result = {}
2076
- relative_paths.each { |rp| result[rp] = {} }
2077
-
2078
- path_set = relative_paths.to_set
2079
- relative_paths.each_slice(500) do |batch|
2080
- log_output = run_git(
2081
- 'log', '--all', '--name-only',
2082
- '--format=__COMMIT__%H|||%an|||%cI|||%s',
2083
- '--since=365 days ago',
2084
- '--', *batch
2085
- )
2086
- parse_git_log_output(log_output, path_set, result)
2087
- end
2088
-
2089
- ninety_days_ago = (Time.current - 90.days).iso8601
2090
- result.each do |relative_path, data|
2091
- result[relative_path] = build_file_metadata(data, ninety_days_ago)
2092
- end
2150
+ relative_paths = file_paths.map { |path| normalize_file_path(path) }.uniq
2151
+ recent_after = Time.current - 90.days
2152
+ raw = GitHistory.new(root: Rails.root, logger: Rails.logger).read(relative_paths, recent_after: recent_after)
2153
+ return {} unless raw
2093
2154
 
2094
- result
2095
- end
2096
-
2097
- # Parse git log output line-by-line, populating result with per-file commit data.
2098
- def parse_git_log_output(log_output, path_set, result)
2099
- current_commit = nil
2100
-
2101
- log_output.each_line do |line|
2102
- line = line.strip
2103
- next if line.empty?
2104
-
2105
- if line.start_with?('__COMMIT__')
2106
- parts = line.sub('__COMMIT__', '').split('|||', 4)
2107
- current_commit = { sha: parts[0], author: parts[1], date: parts[2], message: parts[3] }
2108
- elsif current_commit && path_set.include?(line)
2109
- entry = result[line] ||= {}
2110
- unless entry[:last_modified]
2111
- entry[:last_modified] = current_commit[:date]
2112
- entry[:last_author] = current_commit[:author]
2113
- end
2114
- (entry[:commits] ||= []) << current_commit
2115
- (entry[:contributors] ||= Hash.new(0))[current_commit[:author]] += 1
2116
- end
2117
- end
2155
+ raw.transform_values { |data| build_file_metadata(data, recent_after.iso8601) }
2118
2156
  end
2119
2157
 
2120
2158
  # Classify how frequently a file changes based on commit counts.
@@ -2136,12 +2174,13 @@ module Woods
2136
2174
  def build_file_metadata(data, ninety_days_ago)
2137
2175
  all_commits = data[:commits] || []
2138
2176
  contributor_counts = data[:contributors] || {}
2139
- recent_count = all_commits.count { |c| c[:date] && c[:date] > ninety_days_ago }
2177
+ recent_count = data.fetch(:recent_count) { all_commits.count { |c| c[:date] && c[:date] > ninety_days_ago } }
2178
+ total_count = data.fetch(:commit_count, all_commits.size)
2140
2179
 
2141
2180
  {
2142
2181
  last_modified: data[:last_modified],
2143
2182
  last_author: data[:last_author],
2144
- commit_count: all_commits.size,
2183
+ commit_count: total_count,
2145
2184
  contributors: contributor_counts
2146
2185
  .sort_by { |_, count| -count }
2147
2186
  .first(5)
@@ -2149,7 +2188,7 @@ module Woods
2149
2188
  recent_commits: all_commits.first(5).map do |c|
2150
2189
  { sha: c[:sha]&.first(8), message: c[:message], date: c[:date], author: c[:author] }
2151
2190
  end,
2152
- change_frequency: classify_change_frequency(all_commits.size, recent_count)
2191
+ change_frequency: classify_change_frequency(total_count, recent_count)
2153
2192
  }
2154
2193
  end
2155
2194
 
@@ -2324,6 +2363,7 @@ module Woods
2324
2363
 
2325
2364
  manifest = {
2326
2365
  extracted_at: Time.current.iso8601,
2366
+ woods_version: Woods::VERSION,
2327
2367
  rails_version: Rails.version,
2328
2368
  ruby_version: RUBY_VERSION,
2329
2369
 
@@ -2697,6 +2737,7 @@ module Woods
2697
2737
 
2698
2738
  @incremental_extractors[key] = EXTRACTORS[key]&.new
2699
2739
  rescue StandardError => e
2740
+ @source_inputs&.unverified("extractor:#{key}")
2700
2741
  Rails.logger.warn "[Woods] Could not build #{key} extractor: #{e.message}"
2701
2742
  @incremental_extractors[key] = nil
2702
2743
  end
@@ -2758,6 +2799,10 @@ module Woods
2758
2799
  # one method over — CORE-1).
2759
2800
  produced.merge(units.map { |unit| [unit.identifier, unit.type] })
2760
2801
  touched.merge(register_and_write(rule.extractor_key, units, affected_types))
2802
+ unless source_consumer_failed?(rule.extractor_key)
2803
+ @source_inputs&.consume_file(rule.extractor_key,
2804
+ absolute_path)
2805
+ end
2761
2806
  end
2762
2807
 
2763
2808
  next if raised
@@ -2786,7 +2831,10 @@ module Woods
2786
2831
  # on every changed path of that type, with the generation bumped over
2787
2832
  # the loss. Construction failure tells us nothing about the path; only
2788
2833
  # a genuinely constructed extractor that lacks the method earns the [].
2789
- return nil if extractor.nil?
2834
+ if extractor.nil?
2835
+ source_consumer_failed?(rule.extractor_key, extractor)
2836
+ return nil
2837
+ end
2790
2838
  return [] unless extractor.respond_to?(rule.method_name)
2791
2839
 
2792
2840
  result =
@@ -2801,6 +2849,7 @@ module Woods
2801
2849
 
2802
2850
  Array(result).compact
2803
2851
  rescue StandardError => e
2852
+ @source_inputs&.unverified("extractor:#{rule.extractor_key}")
2804
2853
  Rails.logger.warn "[Woods] #{rule.extractor_key} re-extraction of #{absolute_path} failed: #{e.message}"
2805
2854
  # `nil`, not `[]`. The caller treats an empty result as "this path defines
2806
2855
  # nothing any more" and prunes the units previously registered to it — so
@@ -2871,6 +2920,34 @@ module Woods
2871
2920
  touched
2872
2921
  end
2873
2922
 
2923
+ # Runtime-only model mixins can enter or leave discovery when their
2924
+ # includer changes, even if the mixin file itself is untouched.
2925
+ # @param affected_types [Set<Symbol>]
2926
+ # @return [Set<String>] Added or removed concern identifiers
2927
+ def reconcile_model_mixins(affected_types)
2928
+ extractor = extractor_for(:concerns)
2929
+ return Set.new unless extractor.respond_to?(:runtime_model_mixins)
2930
+
2931
+ live = extractor.runtime_model_mixins
2932
+ known = @dependency_graph.units_of_type(:concern).to_set
2933
+ added = live.flat_map do |path, modules|
2934
+ next [] if modules.all? { |mod| known.include?(mod.name) }
2935
+
2936
+ Array(extractor.extract_model_mixin_file(path)).reject { |unit| known.include?(unit.identifier) }
2937
+ end
2938
+ touched = register_and_write(:concerns, added, affected_types)
2939
+ return touched unless @eager_load_complete
2940
+
2941
+ live_names = live.values.flatten.to_set(&:name)
2942
+ known.each do |identifier|
2943
+ path = @dependency_graph.node(identifier, type: :concern)[:file_path]
2944
+ next if extractor.conventional_concern_path?(path) || live_names.include?(identifier)
2945
+
2946
+ touched.add(identifier) if remove_unit(identifier, affected_types, type: :concern)
2947
+ end
2948
+ touched
2949
+ end
2950
+
2874
2951
  # Pruned class-based identifiers the tree still governs, and that the
2875
2952
  # second reconciliation pass may therefore re-add.
2876
2953
  #
@@ -2948,10 +3025,12 @@ module Woods
2948
3025
  units = new_classes.filter_map do |klass|
2949
3026
  extractor_for(key).public_send(spec[:method], klass)
2950
3027
  rescue StandardError => e
3028
+ @source_inputs&.unverified("extractor:#{key}")
2951
3029
  Rails.logger.warn "[Woods] #{key} extraction of #{klass} failed: #{e.message}"
2952
3030
  nil
2953
3031
  end
2954
3032
 
3033
+ source_consumer_failed?(key)
2955
3034
  register_and_write(key, units, affected_types)
2956
3035
  end
2957
3036
 
@@ -3086,6 +3165,10 @@ module Woods
3086
3165
  # mutating durable state
3087
3166
  def replace_type_wholesale(key, affected_types)
3088
3167
  extractor = extractor_for(key)
3168
+ if extractor.nil?
3169
+ source_consumer_failed?(key, extractor)
3170
+ return Set.new
3171
+ end
3089
3172
  return Set.new unless extractor.respond_to?(:extract_all)
3090
3173
 
3091
3174
  @wholesale_mutations = 0
@@ -3094,6 +3177,8 @@ module Woods
3094
3177
 
3095
3178
  touched = register_and_write(key, units, affected_types)
3096
3179
  touched.merge(remove_replaced_units(key, units, affected_types))
3180
+ @source_inputs&.consume_extractor(key, units) unless source_consumer_failed?(key, extractor)
3181
+ touched
3097
3182
  rescue StandardError => e
3098
3183
  if @wholesale_mutations.to_i.positive?
3099
3184
  raise Woods::ExtractionError, <<~MSG.tr("\n", ' ').strip
@@ -3106,6 +3191,7 @@ module Woods
3106
3191
  MSG
3107
3192
  end
3108
3193
 
3194
+ @source_inputs&.unverified("extractor:#{key}")
3109
3195
  Rails.logger.error "[Woods] Wholesale re-run of #{key} failed: #{e.message}"
3110
3196
  Set.new
3111
3197
  end
@@ -3267,6 +3353,7 @@ module Woods
3267
3353
 
3268
3354
  removed.add(identifier) if remove_unit(identifier, affected_types, type: type)
3269
3355
  end
3356
+ @source_inputs&.consume_deleted(path)
3270
3357
  end
3271
3358
  end
3272
3359
 
@@ -3370,6 +3457,7 @@ module Woods
3370
3457
  (@incremental_written ||= {})[unit.identifier] = unit.file_path
3371
3458
 
3372
3459
  write_unit_file(type_dir.join(collision_safe_filename(unit.identifier)), unit)
3460
+ @source_inputs&.consume_unit(extractor_key, unit.file_path) unless source_consumer_failed?(extractor_key)
3373
3461
  written.add(unit.identifier)
3374
3462
  end
3375
3463
  end
@@ -3446,12 +3534,14 @@ module Woods
3446
3534
  def finalize_incremental_unit_json(affected_types)
3447
3535
  dependents_dirty = @dependents_dirty || Set.new
3448
3536
  git_dirty = @incremental_written || {}
3449
- git_data = incremental_git_data(git_dirty.keys)
3537
+ git_data = profile_phase('git enrichment') { incremental_git_data(git_dirty.keys) }
3450
3538
 
3451
- (dependents_dirty | git_dirty.keys).each do |identifier|
3452
- rewrite_unit_json(identifier, affected_types,
3453
- refresh_dependents: dependents_dirty.include?(identifier),
3454
- git_data: git_dirty.key?(identifier) ? git_data : nil)
3539
+ profile_phase('unit finalization') do
3540
+ (dependents_dirty | git_dirty.keys).each do |identifier|
3541
+ rewrite_unit_json(identifier, affected_types,
3542
+ refresh_dependents: dependents_dirty.include?(identifier),
3543
+ git_data: git_dirty.key?(identifier) ? git_data : nil)
3544
+ end
3455
3545
  end
3456
3546
  end
3457
3547
 
@@ -3560,7 +3650,7 @@ module Woods
3560
3650
  end
3561
3651
 
3562
3652
  # Batch-fetch git metadata for the units written by this run, in a single
3563
- # git invocation, keyed by Rails.root-relative path the way
3653
+ # history walk, keyed by Rails.root-relative path the way
3564
3654
  # {#batch_git_data} returns it.
3565
3655
  #
3566
3656
  # @param identifiers [Array<String>]
@@ -3568,11 +3658,12 @@ module Woods
3568
3658
  def incremental_git_data(identifiers)
3569
3659
  return {} if identifiers.empty? || !git_available?
3570
3660
 
3661
+ root = "#{Rails.root}/"
3571
3662
  paths = identifiers.flat_map do |identifier|
3572
3663
  @dependency_graph.nodes_for(identifier).filter_map do |node|
3573
3664
  next if %i[rails_source gem_source].include?(node[:type])
3574
3665
 
3575
- node[:file_path] if node[:file_path] && File.exist?(node[:file_path])
3666
+ node[:file_path] if git_enrichable_path?(node[:file_path], root)
3576
3667
  end
3577
3668
  end
3578
3669
 
@@ -3627,11 +3718,15 @@ module Woods
3627
3718
  return nil unless extractor_key
3628
3719
 
3629
3720
  extractor = extractor_for(extractor_key)
3630
- return nil unless extractor
3721
+ if extractor.nil?
3722
+ source_consumer_failed?(extractor_key, extractor)
3723
+ return nil
3724
+ end
3631
3725
 
3632
3726
  # File-based extractors can return several units from one file (a .rake
3633
3727
  # file defining multiple tasks, etc.); class-based extractors return one.
3634
3728
  units = Array(re_extracted_units(extractor, type, unit_id, file_path, extractor_key)).compact
3729
+ source_consumer_failed?(extractor_key, extractor)
3635
3730
  return nil if units.empty?
3636
3731
 
3637
3732
  register_and_write(extractor_key, units, affected_types)