woods 2.0.0.beta1 → 2.0.0.beta3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (221) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +400 -1
  3. data/CONTRIBUTING.md +224 -9
  4. data/README.md +7 -3
  5. data/SECURITY.md +9 -6
  6. data/docs/AGENT_GUIDE.md +83 -4
  7. data/docs/AGENT_SETUP.md +82 -1
  8. data/docs/BACKEND_MATRIX.md +20 -0
  9. data/docs/CLIENT_HOOKS.md +111 -0
  10. data/docs/CONFIGURATION_REFERENCE.md +233 -13
  11. data/docs/CONSOLE_MCP_SETUP.md +35 -5
  12. data/docs/DOCKER_SETUP.md +21 -2
  13. data/docs/EVALUATION.md +464 -1
  14. data/docs/EXTRACTOR_REFERENCE.md +36 -5
  15. data/docs/FAQ.md +11 -12
  16. data/docs/GETTING_STARTED.md +17 -5
  17. data/docs/INCREMENTAL_EXTRACTION.md +158 -2
  18. data/docs/INDEX_LAYOUT.md +382 -0
  19. data/docs/INTERNALS.md +15 -7
  20. data/docs/MCP_SERVERS.md +221 -5
  21. data/docs/MCP_TOOL_COOKBOOK.md +33 -18
  22. data/docs/NOTION_INTEGRATION.md +13 -0
  23. data/docs/OBSIDIAN_INTEGRATION.md +57 -9
  24. data/docs/PUBLISHED_INDEX.md +71 -0
  25. data/docs/README.md +7 -0
  26. data/docs/RETRIEVAL_GUIDE.md +253 -11
  27. data/docs/RUNTIME_TRACING.md +71 -0
  28. data/docs/SOURCE_FRESHNESS.md +143 -0
  29. data/docs/TROUBLESHOOTING.md +117 -5
  30. data/docs/UNBLOCKED_INTEGRATION.md +25 -0
  31. data/docs/UPGRADING_TO_2.md +44 -22
  32. data/docs/WATCH_DAEMON.md +259 -59
  33. data/exe/woods-agent-config +6 -0
  34. data/exe/woods-extract +5 -0
  35. data/exe/woods-hook-context +6 -0
  36. data/lib/generators/woods/templates/woods.rb.tt +1 -3
  37. data/lib/tasks/woods.rake +47 -397
  38. data/lib/woods/agent_configuration/applier.rb +133 -0
  39. data/lib/woods/agent_configuration/cli.rb +101 -0
  40. data/lib/woods/agent_configuration/cli_options.rb +29 -0
  41. data/lib/woods/agent_configuration/document.rb +105 -0
  42. data/lib/woods/agent_configuration/error.rb +7 -0
  43. data/lib/woods/agent_configuration/launcher.rb +75 -0
  44. data/lib/woods/agent_configuration/layout.rb +59 -0
  45. data/lib/woods/agent_configuration/managed_section.rb +62 -0
  46. data/lib/woods/agent_configuration/plan.rb +98 -0
  47. data/lib/woods/agent_configuration/plan_diff.rb +38 -0
  48. data/lib/woods/agent_configuration/planned_files.rb +61 -0
  49. data/lib/woods/agent_configuration/planner.rb +63 -0
  50. data/lib/woods/agent_configuration/planner_validation.rb +77 -0
  51. data/lib/woods/agent_configuration/preflight.rb +100 -0
  52. data/lib/woods/agent_configuration/recovery.rb +49 -0
  53. data/lib/woods/ast/node.rb +2 -0
  54. data/lib/woods/ast/parser.rb +38 -5
  55. data/lib/woods/atomic_file.rb +133 -3
  56. data/lib/woods/builder.rb +21 -5
  57. data/lib/woods/cache/cache_middleware.rb +28 -7
  58. data/lib/woods/cache/cache_store.rb +4 -5
  59. data/lib/woods/change_set.rb +5 -4
  60. data/lib/woods/console/credential_index.rb +20 -2
  61. data/lib/woods/console/credential_scanner.rb +14 -14
  62. data/lib/woods/console/credential_scanner_registry.rb +36 -0
  63. data/lib/woods/console/embedded_executor.rb +1 -1
  64. data/lib/woods/console/encrypted_credential_snapshot.rb +16 -0
  65. data/lib/woods/console/rack_middleware.rb +22 -13
  66. data/lib/woods/console/server.rb +18 -16
  67. data/lib/woods/dependency_graph.rb +65 -13
  68. data/lib/woods/embedding/corpus.rb +94 -0
  69. data/lib/woods/embedding/indexer.rb +90 -46
  70. data/lib/woods/embedding/openai.rb +17 -6
  71. data/lib/woods/evaluation/ablation_executor.rb +6 -1
  72. data/lib/woods/evaluation/ablation_timed_executor.rb +22 -4
  73. data/lib/woods/export/typed_reader.rb +56 -0
  74. data/lib/woods/extractor.rb +557 -228
  75. data/lib/woods/extractors/action_cable_extractor.rb +3 -1
  76. data/lib/woods/extractors/behavioral_profile.rb +9 -7
  77. data/lib/woods/extractors/caching_extractor.rb +3 -1
  78. data/lib/woods/extractors/concern_extractor.rb +64 -6
  79. data/lib/woods/extractors/configuration_extractor.rb +7 -3
  80. data/lib/woods/extractors/controller_extractor.rb +13 -4
  81. data/lib/woods/extractors/database_view_extractor.rb +3 -1
  82. data/lib/woods/extractors/decorator_extractor.rb +3 -1
  83. data/lib/woods/extractors/engine_extractor.rb +3 -1
  84. data/lib/woods/extractors/event_extractor.rb +4 -2
  85. data/lib/woods/extractors/factory_extractor.rb +3 -1
  86. data/lib/woods/extractors/graphql_extractor.rb +8 -2
  87. data/lib/woods/extractors/i18n_extractor.rb +3 -1
  88. data/lib/woods/extractors/job_extractor.rb +6 -19
  89. data/lib/woods/extractors/lib_extractor.rb +3 -1
  90. data/lib/woods/extractors/mailer_extractor.rb +20 -5
  91. data/lib/woods/extractors/manager_extractor.rb +3 -1
  92. data/lib/woods/extractors/method_parameters.rb +53 -0
  93. data/lib/woods/extractors/middleware_argument.rb +65 -0
  94. data/lib/woods/extractors/middleware_extractor.rb +9 -3
  95. data/lib/woods/extractors/migration_extractor.rb +3 -1
  96. data/lib/woods/extractors/model_extractor.rb +39 -33
  97. data/lib/woods/extractors/package_extractor.rb +24 -4
  98. data/lib/woods/extractors/phlex_extractor.rb +3 -1
  99. data/lib/woods/extractors/policy_extractor.rb +3 -1
  100. data/lib/woods/extractors/poro_extractor.rb +3 -1
  101. data/lib/woods/extractors/pundit_extractor.rb +3 -1
  102. data/lib/woods/extractors/rails_source_extractor.rb +4 -2
  103. data/lib/woods/extractors/rake_task_extractor.rb +4 -2
  104. data/lib/woods/extractors/route_extractor.rb +3 -1
  105. data/lib/woods/extractors/route_helper_resolver.rb +10 -33
  106. data/lib/woods/extractors/scheduled_job_extractor.rb +41 -15
  107. data/lib/woods/extractors/serializer_extractor.rb +4 -2
  108. data/lib/woods/extractors/service_extractor.rb +3 -1
  109. data/lib/woods/extractors/shared_dependency_scanner.rb +2 -2
  110. data/lib/woods/extractors/shared_utility_methods.rb +27 -15
  111. data/lib/woods/extractors/source_nesting.rb +1 -1
  112. data/lib/woods/extractors/state_machine_extractor.rb +3 -1
  113. data/lib/woods/extractors/test_mapping_extractor.rb +3 -1
  114. data/lib/woods/extractors/validator_extractor.rb +3 -1
  115. data/lib/woods/extractors/view_component_extractor.rb +3 -1
  116. data/lib/woods/extractors/view_template_extractor.rb +3 -1
  117. data/lib/woods/flow_assembler.rb +87 -8
  118. data/lib/woods/flow_precomputer.rb +44 -7
  119. data/lib/woods/gem_mapper.rb +2 -0
  120. data/lib/woods/git_history.rb +116 -0
  121. data/lib/woods/graph_analyzer.rb +195 -63
  122. data/lib/woods/hooks/context_cli.rb +54 -0
  123. data/lib/woods/hooks/context_event.rb +88 -0
  124. data/lib/woods/hooks/context_hint.rb +73 -0
  125. data/lib/woods/hooks/context_impact.rb +77 -0
  126. data/lib/woods/hooks/context_output.rb +47 -0
  127. data/lib/woods/hooks/context_state.rb +102 -0
  128. data/lib/woods/hooks/refresh.rb +79 -0
  129. data/lib/woods/hooks/rule_projection.rb +78 -0
  130. data/lib/woods/input_rules.rb +19 -0
  131. data/lib/woods/mcp/bearer_auth.rb +20 -12
  132. data/lib/woods/mcp/bootstrapper.rb +62 -0
  133. data/lib/woods/mcp/index_reader.rb +323 -160
  134. data/lib/woods/mcp/initialization_guidance.rb +27 -0
  135. data/lib/woods/mcp/origin_guard.rb +17 -9
  136. data/lib/woods/mcp/published_lexical_retriever.rb +115 -0
  137. data/lib/woods/mcp/renderers/markdown_renderer.rb +8 -1
  138. data/lib/woods/mcp/renderers/plain_renderer.rb +7 -1
  139. data/lib/woods/mcp/search_results.rb +74 -0
  140. data/lib/woods/mcp/server.rb +158 -37
  141. data/lib/woods/mcp/tool_contract.rb +2 -0
  142. data/lib/woods/mcp/tool_response_renderer.rb +25 -0
  143. data/lib/woods/mcp/traversal_evidence.rb +113 -0
  144. data/lib/woods/mcp/traversal_evidence_index.rb +100 -0
  145. data/lib/woods/mcp/traversal_evidence_page.rb +41 -0
  146. data/lib/woods/mcp/traversal_evidence_text.rb +52 -0
  147. data/lib/woods/notion/exporter.rb +56 -17
  148. data/lib/woods/obsidian/destination_plan.rb +98 -0
  149. data/lib/woods/obsidian/name_mapper.rb +19 -3
  150. data/lib/woods/obsidian/note_builder.rb +19 -10
  151. data/lib/woods/obsidian/vault_exporter.rb +88 -32
  152. data/lib/woods/operator/pipeline_guard.rb +18 -13
  153. data/lib/woods/path_dispatcher.rb +7 -1
  154. data/lib/woods/payload_store.rb +29 -15
  155. data/lib/woods/railtie.rb +3 -3
  156. data/lib/woods/railtie_support.rb +12 -12
  157. data/lib/woods/rake_helpers.rb +392 -0
  158. data/lib/woods/resilience/graph_invariant_validator/membership_checks.rb +71 -0
  159. data/lib/woods/resilience/graph_invariant_validator/node_checks.rb +61 -0
  160. data/lib/woods/resilience/graph_invariant_validator/reverse_relationship_checks.rb +46 -0
  161. data/lib/woods/resilience/graph_invariant_validator.rb +119 -0
  162. data/lib/woods/resilience/index_validator/graph_checks.rb +80 -0
  163. data/lib/woods/resilience/index_validator.rb +112 -23
  164. data/lib/woods/retrieval/context_assembler.rb +50 -15
  165. data/lib/woods/retrieval/lexical_assembler.rb +73 -0
  166. data/lib/woods/retrieval/lexical_index.rb +119 -0
  167. data/lib/woods/retrieval/ranker.rb +4 -2
  168. data/lib/woods/retrieval/scope.rb +108 -0
  169. data/lib/woods/retrieval/scoped_graph_store.rb +32 -0
  170. data/lib/woods/retrieval/scoped_vector_store.rb +55 -0
  171. data/lib/woods/retrieval/search_executor.rb +86 -27
  172. data/lib/woods/retrieval/source_evidence.rb +200 -0
  173. data/lib/woods/retriever.rb +98 -22
  174. data/lib/woods/ruby_analyzer/trace_enricher.rb +80 -38
  175. data/lib/woods/session_tracer/middleware.rb +10 -12
  176. data/lib/woods/session_tracer/redis_store.rb +22 -6
  177. data/lib/woods/session_tracer/session_flow_assembler.rb +23 -17
  178. data/lib/woods/session_tracer/solid_cache_coordination.rb +6 -4
  179. data/lib/woods/session_tracer/unit_resolver.rb +63 -0
  180. data/lib/woods/source_inputs/consumer_errors.rb +27 -0
  181. data/lib/woods/source_inputs/handoff.rb +102 -0
  182. data/lib/woods/source_inputs/launcher.rb +157 -0
  183. data/lib/woods/source_inputs/manifest.rb +124 -0
  184. data/lib/woods/source_inputs/private_key.rb +55 -0
  185. data/lib/woods/source_inputs/scanner.rb +171 -0
  186. data/lib/woods/source_inputs/scopes.rb +71 -0
  187. data/lib/woods/source_inputs/session.rb +214 -0
  188. data/lib/woods/source_inputs/status.rb +84 -0
  189. data/lib/woods/source_inputs/verifier.rb +107 -0
  190. data/lib/woods/storage/metadata_store.rb +25 -25
  191. data/lib/woods/storage/pgvector.rb +29 -8
  192. data/lib/woods/storage/qdrant.rb +17 -7
  193. data/lib/woods/storage/vector_store.rb +18 -6
  194. data/lib/woods/tasks.rb +3 -2
  195. data/lib/woods/temporal/json_snapshot_store.rb +29 -8
  196. data/lib/woods/unblocked/exporter.rb +59 -70
  197. data/lib/woods/version.rb +1 -1
  198. data/lib/woods/watch/boot_snapshot.rb +52 -0
  199. data/lib/woods/watch/daemon.rb +136 -28
  200. data/lib/woods/watch/listen_watcher.rb +4 -0
  201. data/lib/woods/watch/polling_watcher.rb +5 -1
  202. data/lib/woods/watch/status.rb +20 -15
  203. data/lib/woods/watch/tree_scan.rb +21 -13
  204. data/lib/woods/watch/watcher.rb +4 -1
  205. data/lib/woods.rb +135 -11
  206. data/plugin/.claude-plugin/plugin.json +1 -1
  207. data/plugin/hooks/adapters/normalize.jq +15 -0
  208. data/plugin/hooks/adapters/normalize.rb +63 -0
  209. data/plugin/hooks/hooks.json +20 -0
  210. data/plugin/hooks/woods-context.sh +50 -0
  211. data/plugin/hooks/woods-input-rules.sh +159 -0
  212. data/plugin/hooks/woods-opencode.mjs +65 -0
  213. data/plugin/hooks/woods-post-edit.sh +2 -225
  214. data/plugin/hooks/woods-refresh.sh +260 -0
  215. data/plugin/hooks/woods-session-start.sh +47 -55
  216. data/plugin/skills/woods-agent-enable/SKILL.md +13 -0
  217. data/plugin/skills/woods-diagnose/SKILL.md +288 -1
  218. data/plugin/skills/woods-investigate/SKILL.md +106 -0
  219. data/plugin/skills/woods-mcp-config/SKILL.md +89 -1
  220. data/plugin/skills/woods-setup/SKILL.md +107 -6
  221. metadata +84 -5
@@ -8,12 +8,14 @@ require 'pathname'
8
8
  require 'set'
9
9
 
10
10
  require_relative 'atomic_file'
11
+ require_relative 'version'
11
12
  require_relative 'filename_utils'
12
13
  require_relative 'token_utils'
13
14
  require_relative 'extracted_unit'
14
15
  require_relative 'dependency_graph'
15
16
  require_relative 'payload_store'
16
17
  require_relative 'git_provenance'
18
+ require_relative 'git_history'
17
19
  require_relative 'extractors/model_extractor'
18
20
  require_relative 'extractors/controller_extractor'
19
21
  require_relative 'extractors/phlex_extractor'
@@ -55,6 +57,7 @@ require_relative 'flow_precomputer'
55
57
  require_relative 'change_set'
56
58
  require_relative 'generation'
57
59
  require_relative 'path_dispatcher'
60
+ require_relative 'source_inputs/session'
58
61
 
59
62
  module Woods
60
63
  # Extractor is the main orchestrator for codebase extraction.
@@ -355,7 +358,7 @@ module Woods
355
358
  # flat index — the output root also holds `generation.json`, `dumps/`,
356
359
  # `tasks/`, `woods.sqlite3` and `payloads/` itself, none of which belong
357
360
  # to a generation's payload.
358
- PAYLOAD_FILES = %w[manifest.json dependency_graph.json graph_analysis.json SUMMARY.md].freeze
361
+ PAYLOAD_FILES = %w[manifest.json source_inputs.json dependency_graph.json graph_analysis.json SUMMARY.md].freeze
359
362
 
360
363
  # Payload directories that are not per-type unit directories.
361
364
  PAYLOAD_DIRS = %w[flows].freeze
@@ -394,7 +397,9 @@ module Woods
394
397
  #
395
398
  # @return [Hash] Results keyed by extractor type
396
399
  def extract_all
400
+ profile_started = Process.clock_gettime(Process::CLOCK_MONOTONIC) if profiling?
397
401
  setup_output_directory
402
+ profile_phase('source capture') { begin_source_inputs('full') }
398
403
  ModelNameCache.reset!
399
404
  # @package_resolver alone is not enough: #package_resolver builds
400
405
  # through #extractor_for, which memoizes into @incremental_extractors.
@@ -404,61 +409,69 @@ module Woods
404
409
  # package added between the two runs.
405
410
  @package_resolver = nil
406
411
  @incremental_extractors = nil
407
- begin_payload!
412
+ @persisted_index_stats = nil
413
+ @graph_sha = nil
414
+ profile_phase('payload seed') { begin_payload! }
408
415
 
409
416
  # Eager load once — all extractors need loaded classes for introspection.
410
- safe_eager_load!
417
+ profile_phase('eager load') { safe_eager_load! }
411
418
 
412
419
  # Phase 1: Extract all units
413
- if Woods.configuration.concurrent_extraction
414
- extract_all_concurrent
415
- else
416
- extract_all_sequential
420
+ profile_phase('extraction') do
421
+ if Woods.configuration.concurrent_extraction
422
+ extract_all_concurrent
423
+ else
424
+ extract_all_sequential
425
+ end
417
426
  end
418
427
 
419
428
  # Phase 1.5: Deduplicate results
420
429
  Rails.logger.info '[Woods] Deduplicating results...'
421
- deduplicate_results
430
+ profile_phase('deduplication') { deduplicate_results }
422
431
 
423
432
  # Phase 1.6: Package membership. Runs before the graph is rebuilt so
424
433
  # registration copies metadata[:package] onto the node (#280).
425
- annotate_packages
434
+ profile_phase('package annotation') { annotate_packages }
426
435
 
427
436
  # Rebuild the graph from deduped results. #164 gave DependencyGraph
428
437
  # `#remove`/`#unregister`, so surgical removal is now possible — but a
429
438
  # full extraction has just registered every unit including duplicates,
430
439
  # and rebuilding from the deduped set is both cheaper and less
431
440
  # error-prone than unwinding registrations one at a time.
432
- @dependency_graph = DependencyGraph.new
433
- @results.each_value { |units| units.each { |u| @dependency_graph.register(u) } }
441
+ profile_phase('graph rebuild') do
442
+ @dependency_graph = DependencyGraph.new
443
+ @results.each_value { |units| units.each { |u| @dependency_graph.register(u) } }
444
+ end
434
445
 
435
446
  # Phase 2: Resolve dependents (reverse dependencies)
436
447
  Rails.logger.info '[Woods] Resolving dependents...'
437
- resolve_dependents
448
+ profile_phase('dependents') { resolve_dependents }
438
449
 
439
450
  # Phase 3: Enrich with git data. Runs BEFORE analysis now: the
440
451
  # volatile_dependencies report reads commit counts off graph nodes.
441
452
  Rails.logger.info '[Woods] Enriching with git data...'
442
- enrich_with_git_data
443
- annotate_graph_with_git_data
453
+ profile_phase('git enrichment') do
454
+ enrich_with_git_data
455
+ annotate_graph_with_git_data
456
+ end
444
457
 
445
458
  # Phase 4: Graph analysis (PageRank, structural metrics)
446
459
  Rails.logger.info '[Woods] Analyzing dependency graph...'
447
- @graph_analysis = build_graph_analyzer.analyze
460
+ @graph_analysis = profile_phase('graph analysis') { build_graph_analyzer.analyze }
448
461
 
449
462
  # Phase 4.5: Normalize file_path to relative paths
450
463
  Rails.logger.info '[Woods] Normalizing file paths...'
451
- normalize_file_paths
464
+ profile_phase('path normalization') { normalize_file_paths }
452
465
 
453
466
  # Phase 5: Write output
454
467
  Rails.logger.info '[Woods] Writing output...'
455
- write_results
468
+ profile_phase('write results') { write_results }
456
469
 
457
470
  # Phase 5.1: Sweep unit files no current unit accounts for (#177). Must
458
471
  # run after write_results — the just-written set is what defines
459
472
  # "legitimate" — and belongs to the full path only; the incremental path
460
473
  # deletes through the graph instead. See {#sweep_orphaned_unit_files}.
461
- sweep_orphaned_unit_files
474
+ profile_phase('orphan sweep') { sweep_orphaned_unit_files }
462
475
 
463
476
  # Phase 5.5: Precompute request flows (opt-in). Must run AFTER
464
477
  # write_results — FlowAssembler loads unit JSON from disk, so running
@@ -478,19 +491,26 @@ module Woods
478
491
  # being opt-in does not excuse it.
479
492
  if Woods.configuration.precompute_flows
480
493
  Rails.logger.info '[Woods] Precomputing request flows...'
481
- precompute_flows
494
+ profile_phase('flows') { precompute_flows }
482
495
  end
483
496
 
484
- write_dependency_graph
485
- write_graph_analysis
486
- write_manifest
487
- write_structural_summary
488
- capture_snapshot
497
+ profile_phase('graph write') do
498
+ write_dependency_graph
499
+ write_graph_analysis
500
+ end
501
+ profile_phase('manifest and summary') do
502
+ write_manifest
503
+ write_structural_summary
504
+ end
505
+ profile_phase('snapshot') { capture_snapshot }
506
+ @source_inputs.full_units(@results, consumers: @extractors)
489
507
  publish_generation('full')
490
508
 
491
509
  log_summary
492
510
 
493
511
  @results
512
+ ensure
513
+ log_profile_total('full', profile_started)
494
514
  end
495
515
 
496
516
  # ══════════════════════════════════════════════════════════════════════
@@ -522,62 +542,78 @@ module Woods
522
542
  # @param changed_files [Array<String>] List of changed file paths
523
543
  # @return [Array<String>] Identifiers of units re-extracted, added, or removed
524
544
  def extract_changed(changed_files)
525
- prepare_incremental_run
545
+ profile_started = Process.clock_gettime(Process::CLOCK_MONOTONIC) if profiling?
546
+ prepare_incremental_run(operation: 'incremental')
526
547
 
527
548
  change_set = ChangeSet.new(paths: changed_files, root: Rails.root)
528
549
  affected_types = Set.new
529
550
 
530
- # Blast radius from the pre-change graph.
531
- affected_ids = @dependency_graph.affected_by(change_set.absolute_paths)
551
+ # Blast radius from the pre-change graph, bounded by
552
+ # `incremental_blast_radius_depth` (nil = the unbounded closure).
553
+ affected_ids = profile_phase('blast radius') do
554
+ @dependency_graph.affected_by(change_set.absolute_paths, max_depth: blast_radius_depth)
555
+ end
556
+ @flow_scope = profile_phase('flow radius') { flow_scope_for(change_set) }
532
557
  Rails.logger.info "[Woods] #{change_set.size} changed files affect #{affected_ids.size} units"
533
558
 
534
- touched = reconcile_changed_paths(change_set, affected_types)
535
-
536
- (affected_ids - touched.to_a).each do |unit_id|
537
- touched.add(unit_id) if re_extract_unit(unit_id, affected_types: affected_types)
538
- end
559
+ touched = profile_phase('re-extraction') do
560
+ acc = reconcile_changed_paths(change_set, affected_types)
539
561
 
540
- touched.merge(reconcile_class_based_types(affected_types))
541
- touched.merge(rerun_whole_app_extractors(change_set, affected_types))
542
- touched.merge(reannotate_packages(change_set, affected_types))
543
- pruned = prune_vanished_units(change_set, affected_types)
544
- touched.merge(pruned)
562
+ (affected_ids - acc.to_a).each do |unit_id|
563
+ acc.add(unit_id) if re_extract_unit(unit_id, affected_types: affected_types)
564
+ end
545
565
 
546
- # Reconcile once more, because pruning can un-know a class the first pass
547
- # skipped. A class-based file moved between autoload directories with its
548
- # constant unchanged is still registered under the old path when
549
- # reconciliation runs, so it looks known and is not re-extracted; the
550
- # prune that follows then removes it for its vanished path. This pass
551
- # re-adds it in the same run (M1) instead of leaving the unit missing
552
- # until some later run happens to notice. Idempotent when nothing was
553
- # pruned: the discovery set is compared against the graph, so an
554
- # already-registered class is skipped.
555
- #
556
- # But not everything pruning removed may come back. `except:` keeps the
557
- # *deletion* shape pruned: without a reload, a constant outlives the file
558
- # that defined it — so deleting `app/models/user.rb` prunes `User`, and
559
- # this pass finds `User` still in `ActiveRecord::Base.descendants`.
560
- # Re-registering it would pin the unit to a path that no longer exists,
561
- # and nothing could ever remove it: the sweep excludes class-based units
562
- # and no future change set names that path again. A resident daemon
563
- # processing a batch before its reload hits this every time. What
564
- # separates the two shapes is the filesystem — only pruned identifiers
565
- # that a still-existing file in the change set actually declares are
566
- # re-addable. See {#readdable_pruned_classes}.
567
- touched.merge(reconcile_class_based_types(
568
- affected_types, except: pruned - readdable_pruned_classes(pruned, change_set)
569
- ))
566
+ acc
567
+ end
568
+
569
+ profile_phase('reconciliation') do
570
+ touched.merge(reconcile_class_based_types(affected_types))
571
+ touched.merge(reconcile_model_mixins(affected_types))
572
+ touched.merge(rerun_whole_app_extractors(change_set, affected_types))
573
+ touched.merge(reannotate_packages(change_set, affected_types))
574
+ pruned = prune_vanished_units(change_set, affected_types)
575
+ touched.merge(pruned)
576
+
577
+ # Reconcile once more, because pruning can un-know a class the first pass
578
+ # skipped. A class-based file moved between autoload directories with its
579
+ # constant unchanged is still registered under the old path when
580
+ # reconciliation runs, so it looks known and is not re-extracted; the
581
+ # prune that follows then removes it for its vanished path. This pass
582
+ # re-adds it in the same run (M1) instead of leaving the unit missing
583
+ # until some later run happens to notice. Idempotent when nothing was
584
+ # pruned: the discovery set is compared against the graph, so an
585
+ # already-registered class is skipped.
586
+ #
587
+ # But not everything pruning removed may come back. `except:` keeps the
588
+ # *deletion* shape pruned: without a reload, a constant outlives the file
589
+ # that defined it — so deleting `app/models/user.rb` prunes `User`, and
590
+ # this pass finds `User` still in `ActiveRecord::Base.descendants`.
591
+ # Re-registering it would pin the unit to a path that no longer exists,
592
+ # and nothing could ever remove it: the sweep excludes class-based units
593
+ # and no future change set names that path again. A resident daemon
594
+ # processing a batch before its reload hits this every time. What
595
+ # separates the two shapes is the filesystem — only pruned identifiers
596
+ # that a still-existing file in the change set actually declares are
597
+ # re-addable. See {#readdable_pruned_classes}.
598
+ touched.merge(reconcile_class_based_types(
599
+ affected_types, except: pruned - readdable_pruned_classes(pruned, change_set)
600
+ ))
601
+ end
570
602
 
571
603
  finalize_incremental_unit_json(affected_types)
572
604
 
573
605
  # Regenerate type indexes for affected types
574
- affected_types.each do |type_key|
575
- regenerate_type_index(type_key)
606
+ profile_phase('type index') do
607
+ affected_types.each do |type_key|
608
+ regenerate_type_index(type_key)
609
+ end
576
610
  end
577
611
 
578
612
  finalize_incremental_run(touched)
579
613
 
580
614
  touched.to_a
615
+ ensure
616
+ log_profile_total('incremental', profile_started)
581
617
  end
582
618
 
583
619
  # ══════════════════════════════════════════════════════════════════════
@@ -612,6 +648,7 @@ module Woods
612
648
  # extractor
613
649
  # @raise [ArgumentError] when no recognized key is given
614
650
  def refresh(*keys)
651
+ profile_started = Process.clock_gettime(Process::CLOCK_MONOTONIC) if profiling?
615
652
  keys = Array(keys).flatten.map(&:to_sym).uniq
616
653
  known, unknown = keys.partition { |key| EXTRACTORS.key?(key) }
617
654
  raise ArgumentError, "No known extractor in #{keys.inspect}" if known.empty?
@@ -619,17 +656,19 @@ module Woods
619
656
  known += ROUTE_CONSUMER_EXTRACTORS if known.include?(:routes)
620
657
  known.uniq!
621
658
 
622
- prepare_incremental_run
659
+ prepare_incremental_run(operation: 'refresh')
623
660
  affected_types = Set.new
624
661
  touched = known.each_with_object(Set.new) do |key, acc|
625
662
  acc.merge(replace_type_wholesale(key, affected_types))
626
663
  end
627
664
 
628
665
  finalize_incremental_unit_json(affected_types)
629
- affected_types.each { |type_key| regenerate_type_index(type_key) }
666
+ profile_phase('type index') { affected_types.each { |type_key| regenerate_type_index(type_key) } }
630
667
  finalize_incremental_run(touched, reason: "refresh:#{known.sort.join(',')}")
631
668
 
632
669
  { types: known, touched: touched.to_a, unknown: unknown }
670
+ ensure
671
+ log_profile_total('refresh', profile_started)
633
672
  end
634
673
 
635
674
  # Raise when the most recent extraction run wrote a payload but could not
@@ -651,6 +690,80 @@ module Woods
651
690
 
652
691
  private
653
692
 
693
+ # Whole-run wall time, including unprofiled setup and failed runs. This
694
+ # separate log family must never be added to the individual phase times.
695
+ def log_profile_total(name, started)
696
+ return unless started
697
+
698
+ elapsed = Process.clock_gettime(Process::CLOCK_MONOTONIC) - started
699
+ Rails.logger.info "[Woods] [profile total] #{name} in #{elapsed.round(2)}s"
700
+ end
701
+
702
+ # Time one phase of a run and log how long it took, when WOODS_PROFILE=1.
703
+ #
704
+ # The per-extractor lines (see {#extract_all_sequential}) already report
705
+ # extraction itself. Everything after it (the graph load, the analysis,
706
+ # the flows, the publish) was unattributed, so a slow run could only be
707
+ # split by guessing. Off by default and free when off: the block is
708
+ # yielded directly, with no timing and no log line. Timed on the
709
+ # monotonic clock, so a wall-clock adjustment mid-run cannot produce a
710
+ # negative phase.
711
+ #
712
+ # @param name [String] phase name, as it appears in the log line
713
+ # @return [Object] whatever the block returned
714
+ def profile_phase(name)
715
+ return yield unless profiling?
716
+
717
+ start_time = Process.clock_gettime(Process::CLOCK_MONOTONIC)
718
+ result = yield
719
+ elapsed = Process.clock_gettime(Process::CLOCK_MONOTONIC) - start_time
720
+ Rails.logger.info "[Woods] [profile] #{name} in #{elapsed.round(2)}s"
721
+ result
722
+ end
723
+
724
+ # How many reverse hops {#extract_changed} walks from the changed files
725
+ # before it stops re-extracting dependents.
726
+ #
727
+ # Unbounded by default. A unit outside the cap keeps its content, and the
728
+ # one derived field a re-extraction elsewhere can change (`dependents`)
729
+ # is refreshed regardless: {#register_and_write} marks every target of a
730
+ # re-extracted unit's edges, before and after registration, so a unit
731
+ # that gains or loses an inbound edge is rewritten by
732
+ # {#finalize_incremental_unit_json} whether or not the walk reached it.
733
+ #
734
+ # @return [Integer, nil]
735
+ def blast_radius_depth
736
+ Woods.configuration&.incremental_blast_radius_depth
737
+ end
738
+
739
+ # The units a controller's flow document could reach from this run's
740
+ # changed files, read off the pre-change graph.
741
+ #
742
+ # {FlowAssembler} stops expanding at {FlowPrecomputer::DEFAULT_MAX_DEPTH},
743
+ # so a controller further than that from everything the run changed
744
+ # assembles the same document it already has. The bound is the
745
+ # assembler's own constant, not a second literal: raising the assembly
746
+ # depth widens this walk with it.
747
+ #
748
+ # Reverse reachability is the same relation the refresh already rested on
749
+ # (`touched` is itself the graph's reverse closure), so this narrows the
750
+ # distance without changing the edge set behind the decision.
751
+ #
752
+ # @param change_set [Woods::ChangeSet]
753
+ # @return [Set<String>]
754
+ def flow_scope_for(change_set)
755
+ return Set.new unless Woods.configuration.precompute_flows
756
+
757
+ @dependency_graph.affected_by(
758
+ change_set.absolute_paths, max_depth: FlowPrecomputer::DEFAULT_MAX_DEPTH
759
+ ).to_set
760
+ end
761
+
762
+ # @return [Boolean] whether phase timing is enabled for this process
763
+ def profiling?
764
+ ENV.fetch('WOODS_PROFILE', nil) == '1'
765
+ end
766
+
654
767
  # Load the persisted graph and reset the per-run bookkeeping that the
655
768
  # incremental helpers read. Shared by {#extract_changed} and {#refresh};
656
769
  # calling either without this leaves `@dependents_dirty` and
@@ -664,20 +777,29 @@ module Woods
664
777
  #
665
778
  # @return [void]
666
779
  # @raise [Woods::ExtractionError] see {#begin_payload!}
667
- def prepare_incremental_run
668
- begin_payload!(strict: true)
780
+ def prepare_incremental_run(operation: 'incremental')
781
+ profile_phase('source capture') { begin_source_inputs(operation) }
782
+ profile_phase('payload seed') { begin_payload!(strict: true) }
669
783
  graph_path = payload_dir.join('dependency_graph.json')
670
784
  ensure_incremental_baseline!(graph_path)
671
- @dependency_graph = DependencyGraph.from_h(JSON.parse(AtomicFile.read(graph_path))) if graph_path.exist?
785
+ profile_phase('previous graph load') do
786
+ @dependency_graph = DependencyGraph.from_h(JSON.parse(AtomicFile.read(graph_path))) if graph_path.exist?
787
+ end
672
788
 
673
789
  ModelNameCache.reset!
674
- safe_eager_load!
790
+ profile_phase('eager load') { safe_eager_load! }
675
791
 
676
792
  @dependents_dirty = Set.new
677
793
  @incremental_written = {}
794
+ # nil = no scope was computed for this run, so every re-extracted
795
+ # controller has its flows reassembled. {#refresh} leaves it that way.
796
+ @flow_scope = nil
797
+ @previous_flow_index_entries = nil
678
798
  @incremental_extractors = nil
679
799
  @active_record_names = nil
680
800
  @package_resolver = nil
801
+ @persisted_index_stats = nil
802
+ @graph_sha = nil
681
803
  end
682
804
 
683
805
  # Write the graph and the derived artifacts after an incremental run.
@@ -706,17 +828,19 @@ module Woods
706
828
  # reading "incremental" after a `woods:refresh[routes]` is being misled
707
829
  # @return [void]
708
830
  def finalize_incremental_run(touched, reason: 'incremental')
709
- write_dependency_graph
831
+ profile_phase('graph write') { write_dependency_graph }
710
832
 
711
833
  if touched.empty?
712
834
  Rails.logger.info '[Woods] Incremental run changed nothing — leaving manifest timestamp untouched'
713
835
  return
714
836
  end
715
837
 
716
- write_incremental_graph_analysis
717
- refresh_incremental_flows(touched)
718
- write_manifest(incremental: true)
719
- write_structural_summary
838
+ profile_phase('graph analysis') { write_incremental_graph_analysis }
839
+ profile_phase('flows') { refresh_incremental_flows(touched) }
840
+ profile_phase('manifest and summary') do
841
+ write_manifest(incremental: true)
842
+ write_structural_summary
843
+ end
720
844
  publish_generation(reason)
721
845
 
722
846
  return unless Woods.configuration.enable_snapshots
@@ -737,8 +861,13 @@ module Woods
737
861
  # @return [void]
738
862
  def publish_generation(reason)
739
863
  generation = Generation.new(output_dir: @output_dir)
740
- marker = generation.bump!(reason: reason, payload: publishable_payload_name(generation))
741
- prune_payloads(marker.number)
864
+ # Resolve (and if necessary rename) the payload first, so the flush
865
+ # below covers the directory under the name the pointer will carry.
866
+ payload = publishable_payload_name(generation)
867
+ profile_phase('source verification') { write_source_inputs } if payload
868
+ profile_phase('payload sync') { sync_payload }
869
+ marker = profile_phase('publish') { generation.bump!(reason: reason, payload: payload) }
870
+ profile_phase('payload prune') { prune_payloads(marker.number) }
742
871
  marker
743
872
  rescue StandardError => e
744
873
  # A failed bump must not fail the extraction that produced a perfectly
@@ -757,6 +886,38 @@ module Woods
757
886
  nil
758
887
  end
759
888
 
889
+ # Capture before eager loading or extraction; only an explicit fresh-launch
890
+ # handoff can additionally establish the pre-Bundler/Rails boot boundary.
891
+ def begin_source_inputs(operation)
892
+ @source_inputs = SourceInputs::Session.new(root: Rails.root, output_dir: @output_dir,
893
+ baseline_path: source_input_baseline_path,
894
+ operation: operation)
895
+ end
896
+
897
+ def source_input_baseline_path
898
+ generation = Generation.new(output_dir: @output_dir)
899
+ marker = generation.current
900
+ directory = generation.payload_dir(marker)
901
+ return nil unless marker.payload && directory != generation.root
902
+
903
+ directory.join(SourceInputs::Manifest::FILE_NAME)
904
+ rescue TypeError, NoMethodError
905
+ nil
906
+ end
907
+
908
+ def source_consumer_failed?(key, consumer = extractor_for(key))
909
+ failed = consumer.nil? || SourceInputs::ConsumerErrors.failed?(consumer)
910
+ @source_inputs&.unverified("extractor:#{key}") if failed
911
+ failed
912
+ end
913
+
914
+ def write_source_inputs
915
+ return unless @source_inputs
916
+
917
+ manifest = @source_inputs.finish(generation: @payload_generation, eager_load_complete: @eager_load_complete)
918
+ AtomicFile.write(payload_dir.join(SourceInputs::Manifest::FILE_NAME), JSON.pretty_generate(manifest.data))
919
+ end
920
+
760
921
  # Open the payload directory this run publishes into, seeded from the
761
922
  # generation currently on disk.
762
923
  #
@@ -893,6 +1054,47 @@ module Woods
893
1054
  EXTRACTORS.keys.map(&:to_s) + PAYLOAD_DIRS
894
1055
  end
895
1056
 
1057
+ # Whether payload files pay for their own durability as they are written.
1058
+ #
1059
+ # False by default and false on every host that has not asked otherwise:
1060
+ # {#sync_payload} makes the whole payload durable in one flush before the
1061
+ # pointer names it, so a per-file fsync buys only the window in which
1062
+ # nothing can read the file anyway. `durable_payload_writes = true`
1063
+ # restores the per-file cost for a host that wants it; it does not, and
1064
+ # cannot, remove the publish flush.
1065
+ #
1066
+ # @return [Boolean]
1067
+ def payload_writes_durable?
1068
+ Woods.configuration&.durable_payload_writes ? true : false
1069
+ end
1070
+
1071
+ # One filesystem flush that makes this run's whole payload durable,
1072
+ # immediately before the pointer that names it is written durably.
1073
+ #
1074
+ # This is the guarantee that replaces the per-file fsync {#write_results}
1075
+ # used to pay for every unit: **when `generation.json` is durable, every
1076
+ # file in the payload it names is durable.** What is given up is an
1077
+ # individual payload file being durable before the pointer exists, and
1078
+ # nobody reads a payload file in that window: every reader resolves
1079
+ # through the pointer, and a crash leaves an unreferenced partial payload
1080
+ # that the next run prunes.
1081
+ #
1082
+ # Unconditional on purpose. `durable_payload_writes` decides whether the
1083
+ # per-file fsyncs are *also* paid; it can never turn this off, because
1084
+ # that would be the one change that actually weakens the contract.
1085
+ # {Woods::AtomicFile.sync_directory_tree} degrades through `sync -f`, a
1086
+ # bare `sync`, and finally a per-file fsync pass, so the flush cannot
1087
+ # silently become a no-op either.
1088
+ #
1089
+ # A run that degraded to a flat publish has no payload directory;
1090
+ # {#payload_dir} then returns the output directory, which is the tree that
1091
+ # needs flushing in exactly the same way.
1092
+ #
1093
+ # @return [Symbol, nil] the strategy {Woods::AtomicFile} used
1094
+ def sync_payload
1095
+ AtomicFile.sync_directory_tree(payload_dir)
1096
+ end
1097
+
896
1098
  # The pointer to publish, or nil when this run built no payload directory.
897
1099
  #
898
1100
  # The directory was named from the generation number this run expected to
@@ -1132,9 +1334,6 @@ module Woods
1132
1334
 
1133
1335
  def setup_output_directory
1134
1336
  FileUtils.mkdir_p(@output_dir)
1135
- EXTRACTORS.each_key do |type|
1136
- FileUtils.mkdir_p(payload_dir.join(type.to_s))
1137
- end
1138
1337
  end
1139
1338
 
1140
1339
  # ──────────────────────────────────────────────────────────────────────
@@ -1234,8 +1433,10 @@ module Woods
1234
1433
  def same_type_collision_message(type, unit, prior_path)
1235
1434
  "same-type identifier collision: #{type.to_s.singularize} '#{unit.identifier}' derived from " \
1236
1435
  "two different sources ('#{prior_path || 'no file'}' and '#{unit.file_path || 'no file'}'); " \
1237
- 'only one unit could ever be indexed, so extraction aborted — either merge the ' \
1238
- 'declarations into one file or split them into distinct constants'
1436
+ 'only one unit could ever be indexed, so extraction aborted. ' \
1437
+ 'Wrapper-nested class naming requires Zeitwerk mode with Zeitwerk >= 2.6.9; on older loaders or ' \
1438
+ 'classic-mode hosts, check that support before changing valid namespace wrappers. ' \
1439
+ 'For a genuine duplicate, merge the declarations into one file or split them into distinct constants'
1239
1440
  end
1240
1441
 
1241
1442
  # ──────────────────────────────────────────────────────────────────────
@@ -1279,12 +1480,14 @@ module Woods
1279
1480
  annotated.each do |unit|
1280
1481
  AtomicFile.write(
1281
1482
  type_dir.join(collision_safe_filename(unit.identifier)),
1282
- json_serialize(unit.to_h)
1483
+ json_serialize(unit.to_h),
1484
+ durable: payload_writes_durable?
1283
1485
  )
1284
1486
  end
1285
1487
  AtomicFile.write(
1286
1488
  type_dir.join('_index.json'),
1287
- json_serialize(type_index_entries(units))
1489
+ json_serialize(type_index_entries(units)),
1490
+ durable: payload_writes_durable?
1288
1491
  )
1289
1492
  end
1290
1493
  end
@@ -1329,12 +1532,16 @@ module Woods
1329
1532
  removed = previous_flow_index_controllers & (touched.to_set - reextracted.to_set)
1330
1533
  return if reextracted.empty? && removed.empty?
1331
1534
 
1332
- Rails.logger.info "[Woods] Refreshing flows for #{reextracted.size} controller(s), " \
1333
- "#{removed.size} removed..."
1535
+ reassemble, carried = partition_flow_controllers(
1536
+ reextracted.filter_map { |id| unit_from_payload(:controllers, id) }
1537
+ )
1538
+ Rails.logger.info "[Woods] Refreshing flows for #{reassemble.size} controller(s), " \
1539
+ "#{carried.size} carried forward, #{removed.size} removed..."
1334
1540
  precomputer = FlowPrecomputer.new(units: [], graph: @dependency_graph, output_dir: payload_dir.to_s)
1335
1541
  annotations = precomputer.recompute_delta(
1336
- touched_units: reextracted.filter_map { |id| unit_from_payload(:controllers, id) },
1337
- removed_identifiers: removed.to_a
1542
+ touched_units: reassemble,
1543
+ removed_identifiers: removed.to_a,
1544
+ carried_identifiers: carried.map(&:identifier)
1338
1545
  )
1339
1546
  patch_flow_annotations(annotations)
1340
1547
  sweep_orphaned_flow_files
@@ -1347,6 +1554,52 @@ module Woods
1347
1554
  regenerate_type_index(:controllers) if annotations.any?
1348
1555
  end
1349
1556
 
1557
+ # Split the run's re-extracted controllers into the ones whose flow
1558
+ # documents have to be reassembled and the ones that only need their
1559
+ # annotation back.
1560
+ #
1561
+ # A controller is reassembled when the run changed something its flow can
1562
+ # reach ({#flow_scope_for}), or when its own action set no longer matches
1563
+ # the previous generation's index. The second test is what catches an
1564
+ # action arriving from further up a controller inheritance chain than the
1565
+ # flow radius reaches, and a controller the index has never seen.
1566
+ #
1567
+ # Without a scope (a targeted {#refresh}, or a routes re-run) everything
1568
+ # is reassembled, which is what this path always did.
1569
+ #
1570
+ # @param units [Array<ExtractedUnit>] the run's re-extracted controllers
1571
+ # @return [Array(Array<ExtractedUnit>, Array<ExtractedUnit>)]
1572
+ def partition_flow_controllers(units)
1573
+ return [units, []] if @flow_scope.nil?
1574
+
1575
+ previous = previous_flow_index_actions
1576
+ units.partition do |unit|
1577
+ @flow_scope.include?(unit.identifier) || flow_actions_of(unit) != previous[unit.identifier]
1578
+ end
1579
+ end
1580
+
1581
+ # The actions a unit's metadata declares, as the index records them.
1582
+ #
1583
+ # @param unit [ExtractedUnit]
1584
+ # @return [Set<String>]
1585
+ def flow_actions_of(unit)
1586
+ Array(unit.metadata[:actions] || unit.metadata['actions']).to_set(&:to_s)
1587
+ end
1588
+
1589
+ # The previous generation's flow index, grouped by controller.
1590
+ #
1591
+ # @return [Hash{String => Set<String>}] controller identifier to its
1592
+ # recorded action names, defaulting to an empty set
1593
+ # @raise [Woods::ExtractionError] when the index is missing or corrupt
1594
+ def previous_flow_index_actions
1595
+ previous_flow_index_entries.keys.each_with_object(Hash.new { Set.new }) do |entry_point, grouped|
1596
+ controller, action = entry_point.to_s.split('#', 2)
1597
+ next unless action
1598
+
1599
+ grouped[controller] = grouped[controller] + [action]
1600
+ end
1601
+ end
1602
+
1350
1603
  # Does the run's seeded payload hold a flow family at all? An absent
1351
1604
  # family (no flows/ directory, or an empty one) is a genuine absence —
1352
1605
  # typically an index built while the gate was off — and skips the
@@ -1372,11 +1625,21 @@ module Woods
1372
1625
  # @return [Set<String>]
1373
1626
  # @raise [Woods::ExtractionError] when the index is missing or corrupt
1374
1627
  def previous_flow_index_controllers
1628
+ previous_flow_index_entries.keys.to_set { |entry_point| entry_point.to_s.split('#', 2).first }
1629
+ end
1630
+
1631
+ # The previous generation's flow index itself, read once per run: both
1632
+ # the removal set and the reassembly partition are derived from it.
1633
+ #
1634
+ # @return [Hash{String => String}] entry point to relative document path
1635
+ # @raise [Woods::ExtractionError] when the index is missing or corrupt
1636
+ def previous_flow_index_entries
1637
+ return @previous_flow_index_entries if @previous_flow_index_entries
1638
+
1375
1639
  index_path = payload_dir.join('flows', 'flow_index.json')
1376
1640
  raise Woods::ExtractionError, 'flows/ is populated but flow_index.json is missing' unless index_path.exist?
1377
1641
 
1378
- JSON.parse(AtomicFile.read(index_path))
1379
- .keys.to_set { |entry_point| entry_point.to_s.split('#', 2).first }
1642
+ @previous_flow_index_entries = JSON.parse(AtomicFile.read(index_path))
1380
1643
  rescue JSON::ParserError => e
1381
1644
  raise Woods::ExtractionError, "previous flow_index.json does not parse: #{e.message}"
1382
1645
  end
@@ -1433,7 +1696,7 @@ module Woods
1433
1696
  end
1434
1697
  next if JSON.generate(data) == before
1435
1698
 
1436
- AtomicFile.write(path, json_serialize(data))
1699
+ AtomicFile.write(path, json_serialize(data), durable: payload_writes_durable?)
1437
1700
  end
1438
1701
  end
1439
1702
 
@@ -1550,8 +1813,15 @@ module Woods
1550
1813
  #
1551
1814
  # @return [GraphAnalyzer]
1552
1815
  def build_graph_analyzer
1553
- ratio = Woods.configuration&.volatile_dependency_ratio || GraphAnalyzer::DEFAULT_VOLATILE_RATIO
1554
- GraphAnalyzer.new(@dependency_graph, volatile_ratio: ratio)
1816
+ config = Woods.configuration
1817
+ ratio = config&.volatile_dependency_ratio || GraphAnalyzer::DEFAULT_VOLATILE_RATIO
1818
+ GraphAnalyzer.new(
1819
+ @dependency_graph,
1820
+ volatile_ratio: ratio,
1821
+ volatile_limit_per_target: config&.volatile_dependency_limit_per_target,
1822
+ cycle_limit: config ? config.graph_cycle_limit : GraphAnalyzer::DEFAULT_CYCLE_LIMIT,
1823
+ cycle_max_length: config ? config.graph_cycle_max_length : GraphAnalyzer::DEFAULT_CYCLE_MAX_LENGTH
1824
+ )
1555
1825
  end
1556
1826
 
1557
1827
  # ──────────────────────────────────────────────────────────────────────
@@ -1651,7 +1921,7 @@ module Woods
1651
1921
  else
1652
1922
  metadata.delete('package')
1653
1923
  end
1654
- AtomicFile.write(file, json_serialize(data))
1924
+ AtomicFile.write(file, json_serialize(data), durable: payload_writes_durable?)
1655
1925
 
1656
1926
  identifier = data['identifier']
1657
1927
  type = (data['type'] || type_dir.singularize).to_sym
@@ -1665,11 +1935,9 @@ module Woods
1665
1935
  # Is this a path worth asking git about?
1666
1936
  #
1667
1937
  # A gem-owned unit (an engine model) carries its real path. Outside
1668
- # Rails.root, git refuses the whole `log` invocation when any pathspec is
1669
- # outside the repository — one gem path would erase the git metadata of
1670
- # the other 499 units in its 500-path batch. Inside Rails.root, a bundle
1938
+ # Rails.root, it has no app repository history. Inside Rails.root, a bundle
1671
1939
  # vendored at `vendor/bundle` puts the same gem files under the root
1672
- # prefix, gitignored, so sending them is wasted pathspec work every run.
1940
+ # prefix, gitignored, so requesting their history serves no app-owned unit.
1673
1941
  # Same exclusions as {Extractors::SharedUtilityMethods#app_source?}.
1674
1942
  #
1675
1943
  # @param path [String, nil] absolute file path
@@ -1715,7 +1983,7 @@ module Woods
1715
1983
  payload = json_serialize(unit.to_h)
1716
1984
  return if identical_on_disk?(path, payload)
1717
1985
 
1718
- AtomicFile.write(path, payload)
1986
+ AtomicFile.write(path, payload, durable: payload_writes_durable?)
1719
1987
  end
1720
1988
 
1721
1989
  # The serialized `extracted_at` scalar as {ExtractedUnit#to_h} +
@@ -1723,13 +1991,13 @@ module Woods
1723
1991
  # exactly this value shape — no fractional seconds, `Z` or a `±hh:mm`
1724
1992
  # offset — and `spec/extracted_unit_spec.rb` pins that, so a change to the
1725
1993
  # stamp's shape fails a spec instead of quietly un-matching this mask.
1726
- # The value constraint is what keeps the mask honest against user code: a
1727
- # bare `"extracted_at":` cannot occur inside any JSON *string* value
1728
- # (interior quotes serialize as `\"`), so only a real JSON key can match,
1729
- # and only when it holds a timestamp — which no extractor emits below the
1730
- # top level.
1994
+ # Match only the final top-level stamp, followed by the source_hash field
1995
+ # and the document's closing brace. Nested metadata may use the same key
1996
+ # and timestamp shape; changing it must still rewrite the unit. Escaped
1997
+ # quotes inside string values cannot match these JSON field boundaries.
1731
1998
  EXTRACTED_AT_SCALAR =
1732
- /("extracted_at":\s*")\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}(?:Z|[+-]\d{2}:\d{2})(?=")/
1999
+ /("extracted_at":\s*")\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}(?:Z|[+-]\d{2}:\d{2})
2000
+ (?=",\s*"source_hash":\s*"[0-9a-f]{64}"\s*}\s*\z)/x
1733
2001
  # An implementation detail of the byte comparison, not part of the
1734
2002
  # extractor's surface (`private` does not scope constants).
1735
2003
  private_constant :EXTRACTED_AT_SCALAR
@@ -1763,7 +2031,7 @@ module Woods
1763
2031
  # original encoding
1764
2032
  # @return [String] the bytes with the stamp's value removed
1765
2033
  def mask_extracted_at(bytes)
1766
- bytes.gsub(EXTRACTED_AT_SCALAR, '\1')
2034
+ bytes.sub(EXTRACTED_AT_SCALAR, '\1')
1767
2035
  end
1768
2036
 
1769
2037
  def normalize_file_paths
@@ -1795,7 +2063,7 @@ module Woods
1795
2063
  # to say. Enrichment then wrote `commit_count: 0` and
1796
2064
  # `change_frequency: new` onto every unit, which reads exactly like a file
1797
2065
  # that was never committed, where an absent git directory correctly omits
1798
- # the keys (B-186). HEAD has to resolve.
2066
+ # the keys (B-186). HEAD has to resolve, with complete ancestry (B-189).
1799
2067
  #
1800
2068
  # Memoized, so the warning below is emitted at most once per run.
1801
2069
  #
@@ -1804,13 +2072,31 @@ module Woods
1804
2072
  return @git_available if defined?(@git_available)
1805
2073
 
1806
2074
  _output, error, status = Open3.capture3(*git_argv('rev-parse', 'HEAD'))
1807
- @git_available = status.success?
1808
- warn_unresolvable_git(error) unless @git_available
1809
- @git_available
2075
+ unless status.success?
2076
+ warn_unresolvable_git(error)
2077
+ return @git_available = false
2078
+ end
2079
+
2080
+ @git_available = complete_git_history?
1810
2081
  rescue StandardError
1811
2082
  @git_available = false
1812
2083
  end
1813
2084
 
2085
+ # A shallow HEAD resolves but represents an incomplete ancestry. Do not
2086
+ # turn that boundary into apparent one-commit/new-file churn facts.
2087
+ def complete_git_history?
2088
+ output, _error, status = Open3.capture3(*git_argv('rev-parse', '--is-shallow-repository'))
2089
+ return true if status.success? && output.strip == 'false'
2090
+
2091
+ shallow = status.success? && output.strip == 'true'
2092
+ reason = shallow ? 'shallow repository' : 'repository depth could not be verified'
2093
+ Rails.logger.warn(
2094
+ "[Woods] Git enrichment omitted: #{reason}. Fetch full history with git fetch --unshallow " \
2095
+ '(or actions/checkout fetch-depth: 0), then run full extraction to refresh git metadata.'
2096
+ )
2097
+ false
2098
+ end
2099
+
1814
2100
  # Say once why no unit will carry git metadata, but only when there is a
1815
2101
  # working tree to explain. No `.git` at the root is the ordinary source
1816
2102
  # tarball or `COPY`-without-`.git` case, and it is not a fault.
@@ -1854,63 +2140,19 @@ module Woods
1854
2140
  ''
1855
2141
  end
1856
2142
 
1857
- # Batch-fetch git data for all file paths in two git commands.
1858
- #
1859
- # Duplicate paths are collapsed before slicing (audit P9d): many units
1860
- # share one file_path, and duplicates only repeat a pathspec another
1861
- # batch also sent. The result is keyed by relative path, so the output
1862
- # is identical.
1863
- #
1864
- # @param file_paths [Array<String>] Absolute file paths
1865
- # @return [Hash{String => Hash}] Keyed by relative path
2143
+ # One HEAD history walk, independent of requested-path grouping. See
2144
+ # GitHistory for explicit merge semantics and binary record framing.
2145
+ # @param file_paths [Array<String>] absolute file paths
2146
+ # @return [Hash{String => Hash}] keyed by Rails.root-relative path
1866
2147
  def batch_git_data(file_paths)
1867
2148
  return {} if file_paths.empty?
1868
2149
 
1869
- root = "#{Rails.root}/"
1870
- relative_paths = file_paths.map { |f| f.sub(root, '') }.uniq
1871
- result = {}
1872
- relative_paths.each { |rp| result[rp] = {} }
1873
-
1874
- path_set = relative_paths.to_set
1875
- relative_paths.each_slice(500) do |batch|
1876
- log_output = run_git(
1877
- 'log', '--all', '--name-only',
1878
- '--format=__COMMIT__%H|||%an|||%cI|||%s',
1879
- '--since=365 days ago',
1880
- '--', *batch
1881
- )
1882
- parse_git_log_output(log_output, path_set, result)
1883
- end
1884
-
1885
- ninety_days_ago = (Time.current - 90.days).iso8601
1886
- result.each do |relative_path, data|
1887
- result[relative_path] = build_file_metadata(data, ninety_days_ago)
1888
- end
2150
+ relative_paths = file_paths.map { |path| normalize_file_path(path) }.uniq
2151
+ recent_after = Time.current - 90.days
2152
+ raw = GitHistory.new(root: Rails.root, logger: Rails.logger).read(relative_paths, recent_after: recent_after)
2153
+ return {} unless raw
1889
2154
 
1890
- result
1891
- end
1892
-
1893
- # Parse git log output line-by-line, populating result with per-file commit data.
1894
- def parse_git_log_output(log_output, path_set, result)
1895
- current_commit = nil
1896
-
1897
- log_output.each_line do |line|
1898
- line = line.strip
1899
- next if line.empty?
1900
-
1901
- if line.start_with?('__COMMIT__')
1902
- parts = line.sub('__COMMIT__', '').split('|||', 4)
1903
- current_commit = { sha: parts[0], author: parts[1], date: parts[2], message: parts[3] }
1904
- elsif current_commit && path_set.include?(line)
1905
- entry = result[line] ||= {}
1906
- unless entry[:last_modified]
1907
- entry[:last_modified] = current_commit[:date]
1908
- entry[:last_author] = current_commit[:author]
1909
- end
1910
- (entry[:commits] ||= []) << current_commit
1911
- (entry[:contributors] ||= Hash.new(0))[current_commit[:author]] += 1
1912
- end
1913
- end
2155
+ raw.transform_values { |data| build_file_metadata(data, recent_after.iso8601) }
1914
2156
  end
1915
2157
 
1916
2158
  # Classify how frequently a file changes based on commit counts.
@@ -1932,12 +2174,13 @@ module Woods
1932
2174
  def build_file_metadata(data, ninety_days_ago)
1933
2175
  all_commits = data[:commits] || []
1934
2176
  contributor_counts = data[:contributors] || {}
1935
- recent_count = all_commits.count { |c| c[:date] && c[:date] > ninety_days_ago }
2177
+ recent_count = data.fetch(:recent_count) { all_commits.count { |c| c[:date] && c[:date] > ninety_days_ago } }
2178
+ total_count = data.fetch(:commit_count, all_commits.size)
1936
2179
 
1937
2180
  {
1938
2181
  last_modified: data[:last_modified],
1939
2182
  last_author: data[:last_author],
1940
- commit_count: all_commits.size,
2183
+ commit_count: total_count,
1941
2184
  contributors: contributor_counts
1942
2185
  .sort_by { |_, count| -count }
1943
2186
  .first(5)
@@ -1945,7 +2188,7 @@ module Woods
1945
2188
  recent_commits: all_commits.first(5).map do |c|
1946
2189
  { sha: c[:sha]&.first(8), message: c[:message], date: c[:date], author: c[:author] }
1947
2190
  end,
1948
- change_frequency: classify_change_frequency(all_commits.size, recent_count)
2191
+ change_frequency: classify_change_frequency(total_count, recent_count)
1949
2192
  }
1950
2193
  end
1951
2194
 
@@ -1960,14 +2203,16 @@ module Woods
1960
2203
  units.each do |unit|
1961
2204
  AtomicFile.write(
1962
2205
  type_dir.join(collision_safe_filename(unit.identifier)),
1963
- json_serialize(unit.to_h)
2206
+ json_serialize(unit.to_h),
2207
+ durable: payload_writes_durable?
1964
2208
  )
1965
2209
  end
1966
2210
 
1967
2211
  # Also write a type index for fast lookups
1968
2212
  AtomicFile.write(
1969
2213
  type_dir.join('_index.json'),
1970
- json_serialize(type_index_entries(units))
2214
+ json_serialize(type_index_entries(units)),
2215
+ durable: payload_writes_durable?
1971
2216
  )
1972
2217
  end
1973
2218
  end
@@ -2064,10 +2309,23 @@ module Woods
2064
2309
  # registration order must not reach it (B-180).
2065
2310
  graph_data[:pagerank] = @dependency_graph.pagerank.sort_by { |identifier, _| identifier }.to_h
2066
2311
 
2067
- AtomicFile.write(
2068
- payload_dir.join('dependency_graph.json'),
2069
- json_serialize(graph_data)
2070
- )
2312
+ payload = json_serialize(graph_data)
2313
+ # The bytes about to land on disk are the bytes `graph_sha` covers, so
2314
+ # keep the digest here rather than reading a whole large-app graph back
2315
+ # to compute it. AtomicFile writes in binary mode and reads back as
2316
+ # UTF-8, so the two digests are the same either way.
2317
+ @graph_sha = Digest::SHA256.hexdigest(payload)
2318
+ AtomicFile.write(payload_dir.join('dependency_graph.json'), payload, durable: payload_writes_durable?)
2319
+ end
2320
+
2321
+ # The digest of the dependency graph this run wrote.
2322
+ #
2323
+ # Falls back to reading the file for a caller that writes the analysis
2324
+ # without having written the graph in the same run.
2325
+ #
2326
+ # @return [String] hex SHA256 of dependency_graph.json
2327
+ def graph_sha
2328
+ @graph_sha || Digest::SHA256.hexdigest(AtomicFile.read(payload_dir.join('dependency_graph.json')))
2071
2329
  end
2072
2330
 
2073
2331
  def write_graph_analysis
@@ -2075,14 +2333,13 @@ module Woods
2075
2333
 
2076
2334
  enriched = @graph_analysis.merge(
2077
2335
  generated_at: Time.current.iso8601,
2078
- graph_sha: Digest::SHA256.hexdigest(
2079
- AtomicFile.read(payload_dir.join('dependency_graph.json'))
2080
- )
2336
+ graph_sha: graph_sha
2081
2337
  )
2082
2338
 
2083
2339
  AtomicFile.write(
2084
2340
  payload_dir.join('graph_analysis.json'),
2085
- json_serialize(enriched)
2341
+ json_serialize(enriched),
2342
+ durable: payload_writes_durable?
2086
2343
  )
2087
2344
  end
2088
2345
 
@@ -2106,6 +2363,7 @@ module Woods
2106
2363
 
2107
2364
  manifest = {
2108
2365
  extracted_at: Time.current.iso8601,
2366
+ woods_version: Woods::VERSION,
2109
2367
  rails_version: Rails.version,
2110
2368
  ruby_version: RUBY_VERSION,
2111
2369
 
@@ -2127,32 +2385,55 @@ module Woods
2127
2385
 
2128
2386
  AtomicFile.write(
2129
2387
  payload_dir.join('manifest.json'),
2130
- json_serialize(manifest)
2388
+ json_serialize(manifest),
2389
+ durable: payload_writes_durable?
2131
2390
  )
2132
2391
  end
2133
2392
 
2134
2393
  # Unit and chunk counts derived from the per-type _index.json files on
2135
- # disk — the source of truth after an incremental run, where only the
2394
+ # disk: the source of truth after an incremental run, where only the
2136
2395
  # affected units were re-extracted.
2137
2396
  #
2138
2397
  # @return [Array(Hash{Symbol => Integer}, Integer)] counts by type, total chunk count
2139
2398
  def persisted_counts
2140
- counts = {}
2141
- chunks = 0
2142
-
2143
- Dir[payload_dir.join('*/_index.json').to_s].each do |index_path|
2144
- entries = JSON.parse(AtomicFile.read(index_path))
2145
- counts[File.basename(File.dirname(index_path)).to_sym] = entries.size
2146
- chunks += entries.sum { |e| e['chunk_count'].to_i }
2399
+ stats = persisted_index_stats
2400
+ [stats.transform_values { |type_stats| type_stats[:count] },
2401
+ stats.sum { |_type, type_stats| type_stats[:chunks] }]
2402
+ end
2403
+
2404
+ # One pass over the persisted per-type _index.json files.
2405
+ #
2406
+ # {#persisted_counts} feeds the manifest and {#persisted_summary_stats}
2407
+ # feeds SUMMARY.md, they run back to back at the end of every incremental
2408
+ # run, and each used to parse every type index for itself. Reading once and
2409
+ # deriving both is also what keeps the two artifacts from disagreeing about
2410
+ # totals, which was previously a property of them using the same source
2411
+ # rather than the same read.
2412
+ #
2413
+ # An unreadable index drops that whole type from the manifest and the
2414
+ # summary alike, with one warning rather than the two the separate passes
2415
+ # emitted. Sorted for determinism, since the glob order is the
2416
+ # filesystem's.
2417
+ #
2418
+ # Memoized per run: invalidated at the start of {#extract_all} and
2419
+ # {#prepare_incremental_run}, and again whenever {#regenerate_type_index}
2420
+ # rewrites an index underneath it.
2421
+ #
2422
+ # @return [Hash{Symbol => Hash}] type => `{ count:, chunks:, namespaces: }`
2423
+ def persisted_index_stats
2424
+ @persisted_index_stats ||= Dir[payload_dir.join('*/_index.json').to_s].each_with_object({}) do |path, stats|
2425
+ entries = JSON.parse(AtomicFile.read(path))
2426
+ stats[File.basename(File.dirname(path)).to_sym] = {
2427
+ count: entries.size,
2428
+ chunks: entries.sum { |entry| entry['chunk_count'].to_i },
2429
+ namespaces: namespace_histogram(entries.map { |entry| entry['namespace'] })
2430
+ }
2147
2431
  rescue JSON::ParserError => e
2148
- # An unreadable index silently drops that whole type from the manifest
2149
- # counts — warn rather than undercount without a trace.
2150
- type = File.basename(File.dirname(index_path))
2151
- Rails.logger.warn("[Woods] Skipping unreadable #{type}/_index.json in manifest counts: #{e.message}")
2152
- next
2432
+ type = File.basename(File.dirname(path))
2433
+ Rails.logger.warn(
2434
+ "[Woods] Skipping unreadable #{type}/_index.json in manifest counts and summary totals: #{e.message}"
2435
+ )
2153
2436
  end
2154
-
2155
- [counts, chunks]
2156
2437
  end
2157
2438
 
2158
2439
  # Capture a temporal snapshot after extraction completes.
@@ -2287,7 +2568,8 @@ module Woods
2287
2568
 
2288
2569
  AtomicFile.write(
2289
2570
  payload_dir.join('SUMMARY.md'),
2290
- summary.join("\n")
2571
+ summary.join("\n"),
2572
+ durable: payload_writes_durable?
2291
2573
  )
2292
2574
  end
2293
2575
 
@@ -2308,32 +2590,14 @@ module Woods
2308
2590
  end
2309
2591
 
2310
2592
  # The incremental path's counterpart to {#results_summary_stats}: the same
2311
- # shape read back from the persisted per-type _index.json files, the source
2312
- # {#persisted_counts} uses for the manifest — so the two artifacts cannot
2313
- # disagree about totals. Sorted for determinism, since the glob order is
2314
- # the filesystem's.
2315
- #
2316
- # @return [Hash{Symbol => Hash}, nil] nil when the payload holds no type
2317
- # indexes at all, so a bare index writes no summary — matching the full
2318
- # path's early return when it extracted nothing
2593
+ # shape, read back through {#persisted_index_stats}. A type whose index is
2594
+ # empty is left out, where the manifest still counts it as zero.
2595
+ #
2596
+ # @return [Hash{Symbol => Hash}, nil] nil when the payload holds no
2597
+ # non-empty type index, so a bare index writes no summary, matching the
2598
+ # full path's early return when it extracted nothing
2319
2599
  def persisted_summary_stats
2320
- stats = {}
2321
-
2322
- Dir[payload_dir.join('*/_index.json').to_s].each do |index_path|
2323
- entries = JSON.parse(AtomicFile.read(index_path))
2324
- next if entries.empty?
2325
-
2326
- stats[File.basename(File.dirname(index_path)).to_sym] = {
2327
- count: entries.size,
2328
- chunks: entries.sum { |e| e['chunk_count'].to_i },
2329
- namespaces: namespace_histogram(entries.map { |e| e['namespace'] })
2330
- }
2331
- rescue JSON::ParserError => e
2332
- # Same posture as {#persisted_counts}: an unreadable index drops that
2333
- # type from the manifest and the summary alike, so the two still agree.
2334
- type = File.basename(File.dirname(index_path))
2335
- Rails.logger.warn("[Woods] Skipping unreadable #{type}/_index.json in summary totals: #{e.message}")
2336
- end
2600
+ stats = persisted_index_stats.reject { |_type, type_stats| type_stats[:count].zero? }
2337
2601
 
2338
2602
  stats.empty? ? nil : stats
2339
2603
  end
@@ -2352,6 +2616,10 @@ module Woods
2352
2616
  type_dir = payload_dir.join(type_key.to_s)
2353
2617
  return unless type_dir.directory?
2354
2618
 
2619
+ # This run's counts and summary totals are read from the index files;
2620
+ # rewriting one invalidates whatever was read before.
2621
+ @persisted_index_stats = nil
2622
+
2355
2623
  # Scan existing unit JSON files (exclude _index.json)
2356
2624
  index = Dir[type_dir.join('*.json')].filter_map do |file|
2357
2625
  next if File.basename(file) == '_index.json'
@@ -2371,7 +2639,8 @@ module Woods
2371
2639
 
2372
2640
  AtomicFile.write(
2373
2641
  type_dir.join('_index.json'),
2374
- json_serialize(index)
2642
+ json_serialize(index),
2643
+ durable: payload_writes_durable?
2375
2644
  )
2376
2645
  end
2377
2646
 
@@ -2468,6 +2737,7 @@ module Woods
2468
2737
 
2469
2738
  @incremental_extractors[key] = EXTRACTORS[key]&.new
2470
2739
  rescue StandardError => e
2740
+ @source_inputs&.unverified("extractor:#{key}")
2471
2741
  Rails.logger.warn "[Woods] Could not build #{key} extractor: #{e.message}"
2472
2742
  @incremental_extractors[key] = nil
2473
2743
  end
@@ -2529,6 +2799,10 @@ module Woods
2529
2799
  # one method over — CORE-1).
2530
2800
  produced.merge(units.map { |unit| [unit.identifier, unit.type] })
2531
2801
  touched.merge(register_and_write(rule.extractor_key, units, affected_types))
2802
+ unless source_consumer_failed?(rule.extractor_key)
2803
+ @source_inputs&.consume_file(rule.extractor_key,
2804
+ absolute_path)
2805
+ end
2532
2806
  end
2533
2807
 
2534
2808
  next if raised
@@ -2557,7 +2831,10 @@ module Woods
2557
2831
  # on every changed path of that type, with the generation bumped over
2558
2832
  # the loss. Construction failure tells us nothing about the path; only
2559
2833
  # a genuinely constructed extractor that lacks the method earns the [].
2560
- return nil if extractor.nil?
2834
+ if extractor.nil?
2835
+ source_consumer_failed?(rule.extractor_key, extractor)
2836
+ return nil
2837
+ end
2561
2838
  return [] unless extractor.respond_to?(rule.method_name)
2562
2839
 
2563
2840
  result =
@@ -2572,6 +2849,7 @@ module Woods
2572
2849
 
2573
2850
  Array(result).compact
2574
2851
  rescue StandardError => e
2852
+ @source_inputs&.unverified("extractor:#{rule.extractor_key}")
2575
2853
  Rails.logger.warn "[Woods] #{rule.extractor_key} re-extraction of #{absolute_path} failed: #{e.message}"
2576
2854
  # `nil`, not `[]`. The caller treats an empty result as "this path defines
2577
2855
  # nothing any more" and prunes the units previously registered to it — so
@@ -2642,6 +2920,34 @@ module Woods
2642
2920
  touched
2643
2921
  end
2644
2922
 
2923
+ # Runtime-only model mixins can enter or leave discovery when their
2924
+ # includer changes, even if the mixin file itself is untouched.
2925
+ # @param affected_types [Set<Symbol>]
2926
+ # @return [Set<String>] Added or removed concern identifiers
2927
+ def reconcile_model_mixins(affected_types)
2928
+ extractor = extractor_for(:concerns)
2929
+ return Set.new unless extractor.respond_to?(:runtime_model_mixins)
2930
+
2931
+ live = extractor.runtime_model_mixins
2932
+ known = @dependency_graph.units_of_type(:concern).to_set
2933
+ added = live.flat_map do |path, modules|
2934
+ next [] if modules.all? { |mod| known.include?(mod.name) }
2935
+
2936
+ Array(extractor.extract_model_mixin_file(path)).reject { |unit| known.include?(unit.identifier) }
2937
+ end
2938
+ touched = register_and_write(:concerns, added, affected_types)
2939
+ return touched unless @eager_load_complete
2940
+
2941
+ live_names = live.values.flatten.to_set(&:name)
2942
+ known.each do |identifier|
2943
+ path = @dependency_graph.node(identifier, type: :concern)[:file_path]
2944
+ next if extractor.conventional_concern_path?(path) || live_names.include?(identifier)
2945
+
2946
+ touched.add(identifier) if remove_unit(identifier, affected_types, type: :concern)
2947
+ end
2948
+ touched
2949
+ end
2950
+
2645
2951
  # Pruned class-based identifiers the tree still governs, and that the
2646
2952
  # second reconciliation pass may therefore re-add.
2647
2953
  #
@@ -2719,10 +3025,12 @@ module Woods
2719
3025
  units = new_classes.filter_map do |klass|
2720
3026
  extractor_for(key).public_send(spec[:method], klass)
2721
3027
  rescue StandardError => e
3028
+ @source_inputs&.unverified("extractor:#{key}")
2722
3029
  Rails.logger.warn "[Woods] #{key} extraction of #{klass} failed: #{e.message}"
2723
3030
  nil
2724
3031
  end
2725
3032
 
3033
+ source_consumer_failed?(key)
2726
3034
  register_and_write(key, units, affected_types)
2727
3035
  end
2728
3036
 
@@ -2806,6 +3114,11 @@ module Woods
2806
3114
  return Set.new if keys.empty?
2807
3115
 
2808
3116
  keys += ROUTE_CONSUMER_EXTRACTORS if keys.include?(:routes)
3117
+ # A routes re-run replaces every controller, and a flow document
3118
+ # carries the route itself, which no dependency edge connects to the
3119
+ # controller. Nothing about that is reachable by a graph walk, so the
3120
+ # run drops its flow scope and reassembles every touched controller.
3121
+ @flow_scope = nil if keys.include?(:routes)
2809
3122
 
2810
3123
  keys.each_with_object(Set.new) do |key, touched|
2811
3124
  touched.merge(replace_type_wholesale(key, affected_types))
@@ -2852,6 +3165,10 @@ module Woods
2852
3165
  # mutating durable state
2853
3166
  def replace_type_wholesale(key, affected_types)
2854
3167
  extractor = extractor_for(key)
3168
+ if extractor.nil?
3169
+ source_consumer_failed?(key, extractor)
3170
+ return Set.new
3171
+ end
2855
3172
  return Set.new unless extractor.respond_to?(:extract_all)
2856
3173
 
2857
3174
  @wholesale_mutations = 0
@@ -2860,6 +3177,8 @@ module Woods
2860
3177
 
2861
3178
  touched = register_and_write(key, units, affected_types)
2862
3179
  touched.merge(remove_replaced_units(key, units, affected_types))
3180
+ @source_inputs&.consume_extractor(key, units) unless source_consumer_failed?(key, extractor)
3181
+ touched
2863
3182
  rescue StandardError => e
2864
3183
  if @wholesale_mutations.to_i.positive?
2865
3184
  raise Woods::ExtractionError, <<~MSG.tr("\n", ' ').strip
@@ -2872,6 +3191,7 @@ module Woods
2872
3191
  MSG
2873
3192
  end
2874
3193
 
3194
+ @source_inputs&.unverified("extractor:#{key}")
2875
3195
  Rails.logger.error "[Woods] Wholesale re-run of #{key} failed: #{e.message}"
2876
3196
  Set.new
2877
3197
  end
@@ -3033,6 +3353,7 @@ module Woods
3033
3353
 
3034
3354
  removed.add(identifier) if remove_unit(identifier, affected_types, type: type)
3035
3355
  end
3356
+ @source_inputs&.consume_deleted(path)
3036
3357
  end
3037
3358
  end
3038
3359
 
@@ -3136,6 +3457,7 @@ module Woods
3136
3457
  (@incremental_written ||= {})[unit.identifier] = unit.file_path
3137
3458
 
3138
3459
  write_unit_file(type_dir.join(collision_safe_filename(unit.identifier)), unit)
3460
+ @source_inputs&.consume_unit(extractor_key, unit.file_path) unless source_consumer_failed?(extractor_key)
3139
3461
  written.add(unit.identifier)
3140
3462
  end
3141
3463
  end
@@ -3212,12 +3534,14 @@ module Woods
3212
3534
  def finalize_incremental_unit_json(affected_types)
3213
3535
  dependents_dirty = @dependents_dirty || Set.new
3214
3536
  git_dirty = @incremental_written || {}
3215
- git_data = incremental_git_data(git_dirty.keys)
3537
+ git_data = profile_phase('git enrichment') { incremental_git_data(git_dirty.keys) }
3216
3538
 
3217
- (dependents_dirty | git_dirty.keys).each do |identifier|
3218
- rewrite_unit_json(identifier, affected_types,
3219
- refresh_dependents: dependents_dirty.include?(identifier),
3220
- git_data: git_dirty.key?(identifier) ? git_data : nil)
3539
+ profile_phase('unit finalization') do
3540
+ (dependents_dirty | git_dirty.keys).each do |identifier|
3541
+ rewrite_unit_json(identifier, affected_types,
3542
+ refresh_dependents: dependents_dirty.include?(identifier),
3543
+ git_data: git_dirty.key?(identifier) ? git_data : nil)
3544
+ end
3221
3545
  end
3222
3546
  end
3223
3547
 
@@ -3282,7 +3606,7 @@ module Woods
3282
3606
 
3283
3607
  return if JSON.generate(data) == before
3284
3608
 
3285
- AtomicFile.write(path, json_serialize(data))
3609
+ AtomicFile.write(path, json_serialize(data), durable: payload_writes_durable?)
3286
3610
  affected_types&.add(extractor_key)
3287
3611
  rescue JSON::ParserError => e
3288
3612
  Rails.logger.warn "[Woods] Could not finalize #{identifier}: #{e.message}"
@@ -3326,7 +3650,7 @@ module Woods
3326
3650
  end
3327
3651
 
3328
3652
  # Batch-fetch git metadata for the units written by this run, in a single
3329
- # git invocation, keyed by Rails.root-relative path the way
3653
+ # history walk, keyed by Rails.root-relative path the way
3330
3654
  # {#batch_git_data} returns it.
3331
3655
  #
3332
3656
  # @param identifiers [Array<String>]
@@ -3334,11 +3658,12 @@ module Woods
3334
3658
  def incremental_git_data(identifiers)
3335
3659
  return {} if identifiers.empty? || !git_available?
3336
3660
 
3661
+ root = "#{Rails.root}/"
3337
3662
  paths = identifiers.flat_map do |identifier|
3338
3663
  @dependency_graph.nodes_for(identifier).filter_map do |node|
3339
3664
  next if %i[rails_source gem_source].include?(node[:type])
3340
3665
 
3341
- node[:file_path] if node[:file_path] && File.exist?(node[:file_path])
3666
+ node[:file_path] if git_enrichable_path?(node[:file_path], root)
3342
3667
  end
3343
3668
  end
3344
3669
 
@@ -3393,11 +3718,15 @@ module Woods
3393
3718
  return nil unless extractor_key
3394
3719
 
3395
3720
  extractor = extractor_for(extractor_key)
3396
- return nil unless extractor
3721
+ if extractor.nil?
3722
+ source_consumer_failed?(extractor_key, extractor)
3723
+ return nil
3724
+ end
3397
3725
 
3398
3726
  # File-based extractors can return several units from one file (a .rake
3399
3727
  # file defining multiple tasks, etc.); class-based extractors return one.
3400
3728
  units = Array(re_extracted_units(extractor, type, unit_id, file_path, extractor_key)).compact
3729
+ source_consumer_failed?(extractor_key, extractor)
3401
3730
  return nil if units.empty?
3402
3731
 
3403
3732
  register_and_write(extractor_key, units, affected_types)