woods 2.0.1 → 2.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (154) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +94 -7
  3. data/CONTRIBUTING.md +134 -19
  4. data/README.md +1 -1
  5. data/docs/AGENT_GUIDE.md +19 -0
  6. data/docs/AGENT_SETUP.md +22 -2
  7. data/docs/BACKEND_MATRIX.md +7 -0
  8. data/docs/CLIENT_HOOKS.md +6 -0
  9. data/docs/CONFIGURATION_REFERENCE.md +133 -25
  10. data/docs/CONSOLE_MCP_SETUP.md +82 -30
  11. data/docs/EMBEDDING_MODELS.md +16 -19
  12. data/docs/EXTRACTOR_REFERENCE.md +219 -21
  13. data/docs/FAQ.md +11 -25
  14. data/docs/GETTING_STARTED.md +7 -1
  15. data/docs/INCREMENTAL_EXTRACTION.md +261 -19
  16. data/docs/INDEX_LAYOUT.md +5 -0
  17. data/docs/INTERNALS.md +9 -0
  18. data/docs/MCP_HTTP_TRANSPORT.md +20 -15
  19. data/docs/MCP_SERVERS.md +87 -8
  20. data/docs/MCP_TOOL_COOKBOOK.md +13 -55
  21. data/docs/NOTION_INTEGRATION.md +7 -1
  22. data/docs/PUBLISHED_INDEX.md +6 -0
  23. data/docs/README.md +6 -1
  24. data/docs/RETRIEVAL_GUIDE.md +17 -0
  25. data/docs/SOURCE_FRESHNESS.md +157 -5
  26. data/docs/TOKEN_BENCHMARK.md +10 -18
  27. data/docs/TROUBLESHOOTING.md +70 -14
  28. data/docs/UNBLOCKED_INTEGRATION.md +60 -8
  29. data/docs/UPGRADING_TO_2.md +153 -38
  30. data/docs/WATCH_DAEMON.md +97 -14
  31. data/exe/woods-console-mcp +2 -2
  32. data/lib/generators/woods/templates/woods.rb.tt +2 -1
  33. data/lib/tasks/woods.rake +23 -7
  34. data/lib/tasks/woods_checks.rake +2 -2
  35. data/lib/woods/agent_configuration/cli.rb +1 -1
  36. data/lib/woods/agent_configuration/layout.rb +16 -2
  37. data/lib/woods/agent_configuration/plan.rb +13 -3
  38. data/lib/woods/agent_configuration/planner_validation.rb +4 -2
  39. data/lib/woods/agent_configuration/preflight.rb +5 -3
  40. data/lib/woods/builder.rb +17 -57
  41. data/lib/woods/cache/cache_middleware.rb +56 -30
  42. data/lib/woods/chunking/contributor_chunks.rb +119 -0
  43. data/lib/woods/chunking/semantic_chunker.rb +44 -21
  44. data/lib/woods/console/connection_manager.rb +56 -3
  45. data/lib/woods/console/embedded_executor.rb +30 -5
  46. data/lib/woods/console/rack_middleware.rb +29 -1
  47. data/lib/woods/dependency_graph.rb +34 -10
  48. data/lib/woods/embedding/fake.rb +12 -0
  49. data/lib/woods/embedding/indexer.rb +195 -98
  50. data/lib/woods/embedding/input_budget.rb +67 -0
  51. data/lib/woods/embedding/openai.rb +70 -20
  52. data/lib/woods/embedding/provider.rb +37 -25
  53. data/lib/woods/embedding/text_preparer.rb +76 -32
  54. data/lib/woods/embedding/token_counter.rb +18 -81
  55. data/lib/woods/embedding/vector_configuration.rb +48 -0
  56. data/lib/woods/extraction_identities.rb +175 -0
  57. data/lib/woods/extractor.rb +304 -107
  58. data/lib/woods/extractors/action_cable_extractor.rb +8 -3
  59. data/lib/woods/extractors/assigned_value_discovery.rb +74 -0
  60. data/lib/woods/extractors/class_declarations.rb +121 -0
  61. data/lib/woods/extractors/configuration_extractor.rb +11 -3
  62. data/lib/woods/extractors/declaration_ancestry.rb +92 -0
  63. data/lib/woods/extractors/event_extractor.rb +8 -0
  64. data/lib/woods/extractors/graphql_extractor.rb +134 -77
  65. data/lib/woods/extractors/job_extractor.rb +5 -1
  66. data/lib/woods/extractors/lib_extractor.rb +132 -15
  67. data/lib/woods/extractors/mailer_extractor.rb +3 -5
  68. data/lib/woods/extractors/manager_extractor.rb +7 -21
  69. data/lib/woods/extractors/migration_declaration.rb +87 -0
  70. data/lib/woods/extractors/migration_extractor.rb +5 -39
  71. data/lib/woods/extractors/phlex_extractor.rb +6 -2
  72. data/lib/woods/extractors/policy_extractor.rb +9 -5
  73. data/lib/woods/extractors/poro_extractor.rb +112 -53
  74. data/lib/woods/extractors/pundit_extractor.rb +11 -6
  75. data/lib/woods/extractors/scheduled_job_extractor.rb +45 -4
  76. data/lib/woods/extractors/serializer_extractor.rb +34 -22
  77. data/lib/woods/extractors/shared_utility_methods.rb +18 -1
  78. data/lib/woods/extractors/source_nesting.rb +142 -106
  79. data/lib/woods/extractors/standalone_module_discovery.rb +123 -0
  80. data/lib/woods/extractors/state_machine_extractor.rb +46 -40
  81. data/lib/woods/extractors/view_component_extractor.rb +9 -7
  82. data/lib/woods/flow_assembler.rb +4 -1
  83. data/lib/woods/generation.rb +25 -0
  84. data/lib/woods/hooks/context_hint.rb +7 -2
  85. data/lib/woods/mcp/bootstrapper.rb +33 -7
  86. data/lib/woods/mcp/config_resolver.rb +26 -7
  87. data/lib/woods/mcp/index_reader.rb +125 -24
  88. data/lib/woods/mcp/index_reader_pinning.rb +16 -0
  89. data/lib/woods/mcp/renderers/markdown_renderer.rb +7 -1
  90. data/lib/woods/mcp/renderers/plain_renderer.rb +3 -1
  91. data/lib/woods/mcp/search_results.rb +7 -1
  92. data/lib/woods/mcp/server.rb +24 -4
  93. data/lib/woods/module_reconciliation.rb +151 -0
  94. data/lib/woods/path_dispatcher.rb +7 -2
  95. data/lib/woods/rake_helpers.rb +43 -11
  96. data/lib/woods/release.rb +1 -1
  97. data/lib/woods/resilience/index_validator.rb +8 -3
  98. data/lib/woods/resilience/retryable_provider.rb +18 -1
  99. data/lib/woods/resolved_config.rb +68 -8
  100. data/lib/woods/retrieval/context_assembler.rb +3 -3
  101. data/lib/woods/retrieval/lexical_assembler.rb +3 -2
  102. data/lib/woods/retrieval/scope.rb +18 -2
  103. data/lib/woods/retrieval/source_evidence.rb +14 -2
  104. data/lib/woods/source_contributor_validation.rb +78 -0
  105. data/lib/woods/source_contributors.rb +116 -0
  106. data/lib/woods/source_inputs/handoff.rb +37 -0
  107. data/lib/woods/source_inputs/launcher.rb +53 -13
  108. data/lib/woods/source_inputs/manifest.rb +84 -3
  109. data/lib/woods/source_inputs/private_key.rb +44 -12
  110. data/lib/woods/source_inputs/scanner.rb +98 -27
  111. data/lib/woods/source_inputs/scopes.rb +1 -1
  112. data/lib/woods/source_inputs/session.rb +147 -15
  113. data/lib/woods/source_inputs/stable_reader.rb +127 -0
  114. data/lib/woods/source_inputs/status.rb +40 -8
  115. data/lib/woods/source_inputs/verifier.rb +28 -5
  116. data/lib/woods/source_path_encoding.rb +33 -0
  117. data/lib/woods/source_references/cache.rb +284 -0
  118. data/lib/woods/source_references/collector.rb +120 -0
  119. data/lib/woods/source_references/extraction.rb +185 -0
  120. data/lib/woods/source_references/inputs.rb +134 -0
  121. data/lib/woods/source_references/parser_adapter.rb +134 -0
  122. data/lib/woods/source_references/pass.rb +152 -0
  123. data/lib/woods/source_references/prism_adapter.rb +116 -0
  124. data/lib/woods/source_references/registry.rb +178 -0
  125. data/lib/woods/source_references/runtime_lookup.rb +127 -0
  126. data/lib/woods/source_references/value_class.rb +82 -0
  127. data/lib/woods/storage/metadata_store.rb +4 -1
  128. data/lib/woods/storage/qdrant.rb +2 -2
  129. data/lib/woods/unblocked/client.rb +12 -7
  130. data/lib/woods/unblocked/document_builder.rb +4 -1
  131. data/lib/woods/unblocked/exporter.rb +127 -37
  132. data/lib/woods/unblocked/sync_manifest.rb +137 -21
  133. data/lib/woods/unblocked/uri_migration.rb +105 -0
  134. data/lib/woods/util/host_guard.rb +3 -2
  135. data/lib/woods/version.rb +1 -1
  136. data/lib/woods/watch/catch_up.rb +138 -0
  137. data/lib/woods/watch/claim_lease.rb +150 -0
  138. data/lib/woods/watch/cli.rb +26 -2
  139. data/lib/woods/watch/daemon.rb +80 -59
  140. data/lib/woods/watch/installation/options.rb +1 -1
  141. data/lib/woods/watch/installation/receipt.rb +6 -1
  142. data/lib/woods/watch/managed_child.rb +1 -1
  143. data/lib/woods/watch/supervisor.rb +1 -1
  144. data/lib/woods/watch/tree_scan.rb +14 -2
  145. data/plugin/.claude-plugin/plugin.json +1 -1
  146. data/plugin/hooks/adapters/normalize.rb +3 -2
  147. data/plugin/hooks/woods-input-rules.sh +4 -0
  148. data/plugin/hooks/woods-refresh.sh +15 -7
  149. data/plugin/hooks/woods-session-start.sh +60 -3
  150. data/plugin/skills/woods-diagnose/SKILL.md +334 -11
  151. data/plugin/skills/woods-investigate/SKILL.md +11 -0
  152. data/plugin/skills/woods-mcp-config/SKILL.md +79 -8
  153. data/plugin/skills/woods-setup/SKILL.md +53 -6
  154. metadata +32 -5
@@ -0,0 +1,87 @@
1
+ # frozen_string_literal: true
2
+
3
+ require_relative 'class_declarations'
4
+ require_relative '../source_references/runtime_lookup'
5
+
6
+ module Woods
7
+ module Extractors
8
+ # Identifies migration declarations without running historical migration code.
9
+ # Filename ownership disambiguates a migration from its local migration base.
10
+ class MigrationDeclaration
11
+ class Ambiguous < StandardError; end
12
+
13
+ # @param source [String] original Ruby source
14
+ # @param file_path [String] migration filename
15
+ def initialize(source:, file_path:)
16
+ @declarations = ClassDeclarations.read(source)
17
+ @lookup = SourceReferences::RuntimeLookup.new
18
+ @expected = File.basename(file_path, '.rb').sub(/\A\d+_/, '').camelize
19
+ end
20
+
21
+ # @return [Hash, nil] selected identifier and its declaration source
22
+ # @raise [Ambiguous] when ownership is ambiguous
23
+ def call
24
+ candidates = @declarations.select { |record| migration?(record, []) }.group_by { |record| record[:identifier] }
25
+ selected = select_identity(candidates.keys)
26
+ return unless selected
27
+
28
+ source = @declarations.select { |record| record[:identifier] == selected }.map { |record| record[:source] }
29
+ { identifier: selected, source: source.join("\n") }
30
+ end
31
+
32
+ private
33
+
34
+ def select_identity(identifiers)
35
+ matching = identifiers.select { |name| name == @expected }
36
+ matching = identifiers.select { |name| name.split('::').last == @expected } if matching.empty?
37
+ matching = identifiers if matching.empty?
38
+ return matching.first if matching.size <= 1
39
+
40
+ raise Ambiguous, "Ambiguous migration declaration: #{matching.join(', ')}"
41
+ end
42
+
43
+ def migration?(record, visiting)
44
+ identifier = record.fetch(:identifier)
45
+ return false if visiting.include?(identifier)
46
+
47
+ parent = ClassDeclarations.parent_name(record)
48
+ return false unless parent
49
+
50
+ local = local_parent(parent, record.fetch(:nesting))
51
+ return local.any? { |candidate| migration?(candidate, [*visiting, identifier]) } unless local.empty?
52
+
53
+ runtime_parent?(parent, record.fetch(:nesting))
54
+ end
55
+
56
+ def local_parent(parent, nesting)
57
+ candidates = if parent.start_with?('::')
58
+ [parent.delete_prefix('::')]
59
+ else
60
+ nesting.map { |scope| "#{scope}::#{parent}" } + [parent]
61
+ end
62
+ candidates.each do |name|
63
+ records = @declarations.select { |record| record[:identifier] == name }
64
+ return records unless records.empty?
65
+ end
66
+ []
67
+ end
68
+
69
+ def runtime_parent?(parent, nesting)
70
+ result = @lookup.call(parent, nesting: nesting, allow_private: true)
71
+ if result[:status] == :resolved
72
+ return false unless @lookup.class_object?(result[:value])
73
+
74
+ return @lookup.reflect(result[:value], :ancestors).any? do |ancestor|
75
+ @lookup.reflect(ancestor, :name) == 'ActiveRecord::Migration'
76
+ end
77
+ end
78
+ return false unless %w[constant_missing unloaded_scope autoload_pending].include?(result[:reason])
79
+
80
+ # The standard Rails declaration is structural evidence even when the
81
+ # migration namespace has never been loaded. Unknown custom bases are
82
+ # not inferred from their spelling or evaluated to find out.
83
+ parent.delete_prefix('::') == 'ActiveRecord::Migration'
84
+ end
85
+ end
86
+ end
87
+ end
@@ -4,6 +4,7 @@ require_relative '../source_inputs/consumer_errors'
4
4
 
5
5
  require_relative 'shared_utility_methods'
6
6
  require_relative 'shared_dependency_scanner'
7
+ require_relative 'migration_declaration'
7
8
 
8
9
  module Woods
9
10
  module Extractors
@@ -106,10 +107,10 @@ module Woods
106
107
  # @return [ExtractedUnit, nil] The extracted unit or nil if not a migration
107
108
  def extract_migration_file(file_path)
108
109
  source = File.read(file_path)
109
- class_name = extract_class_name(source)
110
+ declaration = MigrationDeclaration.new(source: source, file_path: file_path).call
111
+ return nil unless declaration
110
112
 
111
- return nil unless class_name
112
- return nil unless migration_class?(source)
113
+ class_name = declaration.fetch(:identifier)
113
114
 
114
115
  unit = ExtractedUnit.new(
115
116
  type: :migration,
@@ -118,7 +119,7 @@ module Woods
118
119
  )
119
120
 
120
121
  unit.namespace = extract_namespace(class_name)
121
- unit.metadata = extract_metadata(source, file_path)
122
+ unit.metadata = extract_metadata(declaration.fetch(:source), file_path)
122
123
  unit.source_code = annotate_source(source, class_name, unit.metadata)
123
124
  unit.dependencies = extract_dependencies(source, unit.metadata)
124
125
 
@@ -130,41 +131,6 @@ module Woods
130
131
 
131
132
  private
132
133
 
133
- # ──────────────────────────────────────────────────────────────────────
134
- # Class Discovery
135
- # ──────────────────────────────────────────────────────────────────────
136
-
137
- # Extract the class name from migration source code.
138
- #
139
- # @param source [String] Ruby source code
140
- # @return [String, nil] The class name or nil
141
- def extract_class_name(source)
142
- # Match namespaced or plain class declarations
143
- namespaces = source.scan(/^\s*module\s+([\w:]+)/).flatten
144
- class_match = source.match(/^\s*class\s+([\w:]+)\s*</)
145
- return nil unless class_match
146
-
147
- base_class = class_match[1]
148
- if namespaces.any? && !base_class.include?('::')
149
- "#{namespaces.join('::')}::#{base_class}"
150
- else
151
- base_class
152
- end
153
- end
154
-
155
- # Check whether the source defines an ActiveRecord::Migration subclass.
156
- #
157
- # `[\w:]+`, matching {#extract_class_name}'s class-name pattern — a
158
- # compact-form declaration (`class Billing::AddFoo < ...`) satisfies
159
- # extract_class_name but failed the plain `\w+` here, so the file was
160
- # silently skipped even though it had a resolvable identifier.
161
- #
162
- # @param source [String] Ruby source code
163
- # @return [Boolean]
164
- def migration_class?(source)
165
- source.match?(/class\s+[\w:]+\s*<\s*ActiveRecord::Migration/)
166
- end
167
-
168
134
  # ──────────────────────────────────────────────────────────────────────
169
135
  # Metadata Extraction
170
136
  # ──────────────────────────────────────────────────────────────────────
@@ -66,7 +66,7 @@ module Woods
66
66
  return [] unless @component_base
67
67
 
68
68
  load_component_files
69
- @component_base.descendants
69
+ @component_base.descendants.select { |component| app_component?(component) }
70
70
  end
71
71
 
72
72
  # Extract a single component
@@ -74,7 +74,7 @@ module Woods
74
74
  # @param component [Class] The component class
75
75
  # @return [ExtractedUnit] The extracted unit
76
76
  def extract_component(component)
77
- return nil if component.name.nil?
77
+ return nil unless app_component?(component)
78
78
 
79
79
  unit = ExtractedUnit.new(
80
80
  type: :component,
@@ -95,6 +95,10 @@ module Woods
95
95
 
96
96
  private
97
97
 
98
+ def app_component?(component)
99
+ @component_base && component.name && component < @component_base && app_source_file?(source_file_for(component))
100
+ end
101
+
98
102
  # Find the base component class used in the application.
99
103
  # Skips ApplicationComponent if it's actually a ViewComponent subclass
100
104
  # to avoid extracting ViewComponent classes with Phlex-specific metadata.
@@ -4,6 +4,7 @@ require_relative '../source_inputs/consumer_errors'
4
4
 
5
5
  require_relative 'shared_utility_methods'
6
6
  require_relative 'shared_dependency_scanner'
7
+ require_relative 'declaration_ancestry'
7
8
 
8
9
  module Woods
9
10
  module Extractors
@@ -66,7 +67,8 @@ module Woods
66
67
 
67
68
  unit.namespace = extract_namespace(class_name)
68
69
  unit.source_code = annotate_source(source, class_name)
69
- unit.metadata = extract_metadata(source, class_name)
70
+ ancestry = DeclarationAncestry.new(source: source, identifier: class_name, file_path: file_path)
71
+ unit.metadata = extract_metadata(source, class_name, ancestry)
70
72
  unit.dependencies = extract_dependencies(source, class_name)
71
73
 
72
74
  unit
@@ -100,14 +102,14 @@ module Woods
100
102
  # Metadata Extraction
101
103
  # ──────────────────────────────────────────────────────────────────────
102
104
 
103
- def extract_metadata(source, class_name)
105
+ def extract_metadata(source, class_name, ancestry)
104
106
  {
105
107
  evaluated_models: detect_evaluated_models(source, class_name),
106
108
  decision_methods: detect_decision_methods(source),
107
109
  public_methods: extract_public_methods(source),
108
110
  class_methods: extract_class_methods(source),
109
111
  initialize_params: extract_initialize_params(source),
110
- is_pundit: pundit_policy?(source),
112
+ is_pundit: pundit_policy?(source, ancestry),
111
113
  custom_errors: extract_custom_errors(source),
112
114
  loc: source.lines.count { |l| l.strip.length.positive? && !l.strip.start_with?('#') },
113
115
  method_count: source.scan(/def\s+(?:self\.)?\w+/).size
@@ -165,8 +167,10 @@ module Woods
165
167
  models.uniq
166
168
  end
167
169
 
168
- def pundit_policy?(source)
169
- source.match?(/< ApplicationPolicy/) ||
170
+ def pundit_policy?(source, ancestry)
171
+ return false if ancestry.foreign?
172
+
173
+ ancestry.application_policy? ||
170
174
  source.match?(/def\s+initialize\s*\(\s*user\s*,/) ||
171
175
  source.match?(/attr_reader\s+:user\s*,\s*:record/)
172
176
  end
@@ -5,6 +5,9 @@ require_relative '../source_inputs/consumer_errors'
5
5
  require_relative 'shared_utility_methods'
6
6
  require_relative 'shared_dependency_scanner'
7
7
  require_relative 'source_nesting'
8
+ require_relative '../source_references/collector'
9
+ require_relative 'standalone_module_discovery'
10
+ require_relative 'assigned_value_discovery'
8
11
 
9
12
  module Woods
10
13
  module Extractors
@@ -16,7 +19,8 @@ module Woods
16
19
  # wrappers, and any other non-AR class living alongside AR models.
17
20
  #
18
21
  # Files under app/models/concerns/ are excluded — those are handled by
19
- # ConcernExtractor. Module-only files are also excluded.
22
+ # ConcernExtractor. Callable standalone modules use the existing poro type,
23
+ # with explicit module metadata and verified runtime/source ownership.
20
24
  #
21
25
  # @example
22
26
  # extractor = PoroExtractor.new
@@ -51,76 +55,121 @@ module Woods
51
55
 
52
56
  ar_names = ActiveRecord::Base.descendants.filter_map(&:name).to_set
53
57
 
54
- Dir[Rails.root.join(MODELS_GLOB)].filter_map do |file|
55
- next if file.include?(CONCERNS_SEGMENT)
58
+ @module_discovery = StandaloneModuleDiscovery.new
59
+ Dir[Rails.root.join(MODELS_GLOB)].flat_map do |file|
60
+ next [] if file.include?(CONCERNS_SEGMENT)
56
61
 
57
- extract_poro_file(file, ar_names: ar_names)
62
+ extract_poro_units(file, ar_names: ar_names)
58
63
  end
59
64
  end
60
65
 
61
- # Extract a single PORO file.
62
- #
63
- # Returns nil if the file is not a PORO (e.g., module-only, no class
64
- # or PORO pattern found, or the inferred class is an AR descendant).
66
+ # Preserve the historical single-unit return contract. A class remains
67
+ # primary when a file also declares callable standalone modules.
65
68
  #
66
- # @param file_path [String] Absolute path to the Ruby file
67
- # @param ar_names [Set<String>] Set of AR descendant names to skip
68
- # @return [ExtractedUnit, nil] The extracted unit or nil
69
+ # @param file_path [String] original Ruby file
70
+ # @param ar_names [Set<String>] Active Record identities to exclude
71
+ # @return [ExtractedUnit, nil] primary class or first standalone module
69
72
  def extract_poro_file(file_path, ar_names: Set.new)
73
+ extract_poro_units(file_path, ar_names: ar_names).first
74
+ end
75
+
76
+ # Extract every owned unit from one file; incremental dispatch uses this
77
+ # form so a concern and a standalone sibling can share source safely.
78
+ #
79
+ # @param file_path [String] original Ruby file
80
+ # @param ar_names [Set<String>] Active Record identities to exclude
81
+ # @return [Array<ExtractedUnit>] legacy class followed by standalone modules
82
+ def extract_poro_units(file_path, ar_names: Set.new)
70
83
  source = File.read(file_path)
84
+ analysis = SourceReferences::Collector.new.call(source)
85
+ primary = extract_class_unit(file_path, source, ar_names, analysis)
86
+ discovery = (@module_discovery ||= StandaloneModuleDiscovery.new)
87
+ modules = discovery.call(file_path, analysis: analysis).map { |record| module_unit(file_path, source, record) }
88
+ [primary, *modules].compact
89
+ rescue StandardError => e
90
+ SourceInputs::ConsumerErrors.log(self, "Failed to extract PORO #{file_path}: #{e.message}")
91
+ []
92
+ end
93
+
94
+ # Recompute unclaimed module identities for includer-only reconciliation.
95
+ # The root pipeline owns cross-family migration and source-consumption.
96
+ #
97
+ # @return [Hash<String, Array<ExtractedUnit>>] absolute paths and module units
98
+ def standalone_modules
99
+ @module_discovery = StandaloneModuleDiscovery.new
100
+ return {} unless @models_dir.directory?
101
+
102
+ Dir[Rails.root.join(MODELS_GLOB)].each_with_object({}) do |file, result|
103
+ next if file.include?(CONCERNS_SEGMENT)
104
+
105
+ units = extract_standalone_module_file(file)
106
+ result[file] = units unless units.empty?
107
+ end
108
+ end
109
+
110
+ private
111
+
112
+ def extract_standalone_module_file(file)
113
+ source = File.read(file)
114
+ analysis = SourceReferences::Collector.new.call(source)
115
+ @module_discovery.call(file, analysis: analysis).map { |record| module_unit(file, source, record) }
116
+ rescue StandardError => e
117
+ SourceInputs::ConsumerErrors.log(self, "Failed to extract standalone module #{file}: #{e.message}")
118
+ []
119
+ end
71
120
 
72
- return nil unless poro_file?(source)
73
- return nil if module_only?(source)
121
+ def extract_class_unit(file_path, source, ar_names, analysis)
122
+ return nil unless class_source?(source, analysis)
74
123
 
75
- class_name = infer_class_name(file_path, source)
124
+ class_name = infer_class_name(file_path, source, analysis)
76
125
  return nil unless class_name
77
126
  return nil if ar_names.include?(class_name)
127
+ return nil if analysis.fetch('declarations').any? do |declaration|
128
+ declaration['owner'] == class_name && declaration['kind'] == 'module'
129
+ end
78
130
 
79
- unit = ExtractedUnit.new(
80
- type: :poro,
81
- identifier: class_name,
82
- file_path: file_path
83
- )
84
-
131
+ unit = ExtractedUnit.new(type: :poro, identifier: class_name, file_path: file_path)
85
132
  parent_class = extract_parent_class(source, class_name)
86
-
87
- unit.namespace = extract_namespace(class_name)
88
- unit.source_code = annotate_source(source, class_name, parent_class)
89
- unit.metadata = extract_metadata(source, parent_class)
133
+ unit.namespace = extract_namespace(class_name)
134
+ unit.source_code = annotate_source(source, class_name, parent_class)
135
+ unit.metadata = extract_metadata(source, parent_class)
90
136
  unit.dependencies = extract_dependencies(source)
91
-
92
137
  unit
93
- rescue StandardError => e
94
- SourceInputs::ConsumerErrors.log(self, "Failed to extract PORO #{file_path}: #{e.message}")
95
- nil
96
138
  end
97
139
 
98
- private
140
+ def module_unit(file_path, source, record)
141
+ identifier = record.fetch(:identifier)
142
+ unit = ExtractedUnit.new(type: :poro, identifier: identifier, file_path: file_path)
143
+ unit.namespace = extract_namespace(identifier)
144
+ unit.source_code = annotate_source(source, identifier, nil)
145
+ unit.metadata = record.except(:identifier).merge(ruby_kind: 'module', parent_class: nil,
146
+ initialize_params: [], loc: count_loc(source))
147
+ # Shared-file regex scans cannot attribute a sibling's references safely.
148
+ # The source-reference pass owns method/body references for these units.
149
+ unit.dependencies = []
150
+ unit
151
+ end
99
152
 
100
153
  # ──────────────────────────────────────────────────────────────────────
101
154
  # File Classification
102
155
  # ──────────────────────────────────────────────────────────────────────
103
156
 
104
- # Determine whether a file is worth examining as a PORO.
105
- #
106
- # A file qualifies if it contains a class definition OR uses one of the
107
- # common PORO-without-class patterns (Struct.new, Data.define).
108
- # Plain constant assignments and module-only files are excluded upstream.
109
- #
110
- # @param source [String] Ruby source code
111
- # @return [Boolean]
112
- def poro_file?(source)
113
- source.match?(/^\s*class\s+/) ||
114
- source.match?(/\bStruct\.new\b/) ||
115
- source.match?(/\bData\.define\b/)
116
- end
117
-
118
- # Return true when the file defines only modules, no class keyword.
119
- #
120
- # @param source [String] Ruby source code
121
- # @return [Boolean]
122
- def module_only?(source)
123
- source.match?(/^\s*module\s+\w+/) && !source.match?(/^\s*class\s+/)
157
+ # Singleton-class syntax does not establish a class-owned unit. Keep
158
+ # legacy Struct/Data handling while requiring an actual class declaration.
159
+ def class_source?(source, analysis)
160
+ # Preserve the legacy single-class excerpt behavior on invalid fragments;
161
+ # no runtime module ownership is inferred from an unsuccessful parse.
162
+ has_class = if analysis['parse_error']
163
+ source.match?(/^\s*class\s+/)
164
+ else
165
+ analysis.fetch('declarations').any? do |declaration|
166
+ declaration['kind'] == 'class' && declaration.fetch('singleton_depth', 0).zero?
167
+ end
168
+ end
169
+ return true if has_class
170
+ return false if source.match?(/^\s*module\s+\w+/)
171
+
172
+ source.match?(/\bStruct\.new\b/) || source.match?(/\bData\.define\b/)
124
173
  end
125
174
 
126
175
  # ──────────────────────────────────────────────────────────────────────
@@ -140,15 +189,25 @@ module Woods
140
189
  # @param file_path [String] Absolute path to the file
141
190
  # @param source [String] Ruby source code
142
191
  # @return [String, nil] The inferred class name
143
- def infer_class_name(file_path, source)
192
+ def infer_class_name(file_path, source, analysis = SourceReferences::Collector.new.call(source))
144
193
  # Explicit class keyword — Zeitwerk-governed naming first (G-1), then
145
194
  # enclosing modules joined by position (#174)
195
+ assigned = AssignedValueDiscovery.new.call(file_path, analysis: analysis,
196
+ expected: managed_constant_path(file_path.to_s))
197
+ return assigned if assigned
198
+
146
199
  qualified = governed_class_name(file_path, source) || qualified_first_class_name(source)
147
200
  return qualified if qualified
148
201
 
149
- # Struct.new / Data.define: ConstantName = Struct.new(...)
150
- struct_match = source.match(/^(\w[\w:]*)\s*=\s*(?:Struct\.new|Data\.define)/)
151
- return struct_match[1] if struct_match
202
+ # Preserve historical top-level assignment lookup identities, including
203
+ # named Struct aliases. Reference targets still require verified ownership.
204
+ assignments = analysis.fetch('declarations').select { |record| record['constructor'] }
205
+ unless assignments.empty?
206
+ legacy = assignments.find do |record|
207
+ record['enclosing_nesting'].empty? && record.fetch('singleton_depth', 0).zero?
208
+ end
209
+ return legacy && legacy['owner']
210
+ end
152
211
 
153
212
  # Fall back: derive from file path using Rails naming convention
154
213
  path_based_class_name(file_path)
@@ -4,6 +4,7 @@ require_relative '../source_inputs/consumer_errors'
4
4
 
5
5
  require_relative 'shared_utility_methods'
6
6
  require_relative 'shared_dependency_scanner'
7
+ require_relative 'declaration_ancestry'
7
8
 
8
9
  module Woods
9
10
  module Extractors
@@ -54,7 +55,9 @@ module Woods
54
55
  class_name = extract_class_name(file_path, source)
55
56
 
56
57
  return nil unless class_name
57
- return nil unless pundit_policy?(source)
58
+
59
+ ancestry = DeclarationAncestry.new(source: source, identifier: class_name, file_path: file_path)
60
+ return nil unless pundit_policy?(source, ancestry)
58
61
 
59
62
  unit = ExtractedUnit.new(
60
63
  type: :pundit_policy,
@@ -64,7 +67,7 @@ module Woods
64
67
 
65
68
  unit.namespace = extract_namespace(class_name)
66
69
  unit.source_code = annotate_source(source, class_name)
67
- unit.metadata = extract_metadata(source, class_name)
70
+ unit.metadata = extract_metadata(source, class_name, ancestry)
68
71
  unit.dependencies = extract_dependencies(source, class_name)
69
72
 
70
73
  unit
@@ -100,8 +103,10 @@ module Woods
100
103
  #
101
104
  # @param source [String] Ruby source code
102
105
  # @return [Boolean]
103
- def pundit_policy?(source)
104
- source.match?(/< ApplicationPolicy/) ||
106
+ def pundit_policy?(source, ancestry)
107
+ return false if ancestry.foreign?
108
+
109
+ ancestry.application_policy? ||
105
110
  (source.match?(/attr_reader\s+:user/) && source.match?(/attr_reader.*:record/)) ||
106
111
  (source.match?(/def\s+initialize\s*\(\s*user\s*,/) && source.match?(/def\s+\w+\?/))
107
112
  end
@@ -135,7 +140,7 @@ module Woods
135
140
  # @param source [String]
136
141
  # @param class_name [String]
137
142
  # @return [Hash]
138
- def extract_metadata(source, class_name)
143
+ def extract_metadata(source, class_name, ancestry)
139
144
  actions = detect_authorization_actions(source)
140
145
  {
141
146
  model: infer_model(class_name),
@@ -143,7 +148,7 @@ module Woods
143
148
  standard_actions: actions & PUNDIT_ACTIONS,
144
149
  custom_actions: actions - PUNDIT_ACTIONS,
145
150
  has_scope_class: source.match?(/class\s+Scope\b/) || false,
146
- inherits_application_policy: source.match?(/< ApplicationPolicy/) || false,
151
+ inherits_application_policy: ancestry.application_policy?,
147
152
  public_methods: extract_public_methods(source),
148
153
  class_methods: extract_class_methods(source),
149
154
  loc: source.lines.count { |l| l.strip.length.positive? && !l.strip.start_with?('#') },
@@ -3,6 +3,7 @@
3
3
  require_relative '../source_inputs/consumer_errors'
4
4
 
5
5
  require 'yaml'
6
+ require 'set'
6
7
  begin
7
8
  require 'active_support/configuration_file'
8
9
  rescue LoadError
@@ -62,9 +63,7 @@ module Woods
62
63
  #
63
64
  # @return [Array<ExtractedUnit>] List of scheduled job units
64
65
  def extract_all
65
- @schedule_files.flat_map do |file_path, format|
66
- extract_scheduled_job_file(file_path, format)
67
- end
66
+ allocate_identifiers(schedule_units(@schedule_files))
68
67
  end
69
68
 
70
69
  # Extract scheduled job entries from a single schedule file.
@@ -76,6 +75,18 @@ module Woods
76
75
  # @param format [Symbol] One of :solid_queue, :sidekiq_cron, :whenever
77
76
  # @return [Array<ExtractedUnit>] List of scheduled job units
78
77
  def extract_scheduled_job_file(file_path, format)
78
+ path = File.expand_path(file_path.to_s)
79
+ files = @schedule_files.merge(path => format)
80
+ allocate_identifiers(schedule_units(files)).select { |unit| unit.file_path == path }
81
+ end
82
+
83
+ private
84
+
85
+ def schedule_units(files)
86
+ files.flat_map { |path, format| extract_schedule_file(path, format) }
87
+ end
88
+
89
+ def extract_schedule_file(file_path, format)
79
90
  case format
80
91
  when :solid_queue, :sidekiq_cron
81
92
  extract_yaml_schedule(file_path, format)
@@ -89,7 +100,35 @@ module Woods
89
100
  []
90
101
  end
91
102
 
92
- private
103
+ # Keep every unique legacy identifier. Reserve those first, then qualify
104
+ # conflicting names in deterministic format order without stealing a
105
+ # literal task name that already looks like one of our generated names.
106
+ def allocate_identifiers(units)
107
+ groups = units.group_by(&:identifier)
108
+ reserved = groups.select { |_identifier, entries| entries.one? }.keys.to_set
109
+ groups.keys.sort.each do |identifier|
110
+ entries = groups.fetch(identifier)
111
+ next if entries.one?
112
+
113
+ formats = entries.map { |unit| unit.metadata.fetch(:schedule_format) }
114
+ if formats.uniq.size != formats.size
115
+ raise ArgumentError, "Ambiguous same-format schedule name: #{identifier.inspect}"
116
+ end
117
+
118
+ entries.sort_by { |unit| unit.metadata.fetch(:schedule_format).to_s }.each do |unit|
119
+ base = "scheduled:#{unit.metadata.fetch(:schedule_format)}:#{identifier.delete_prefix('scheduled:')}"
120
+ candidate = base
121
+ suffix = 1
122
+ while reserved.include?(candidate)
123
+ suffix += 1
124
+ candidate = "#{base}:#{suffix}"
125
+ end
126
+ unit.identifier = candidate
127
+ reserved.add(candidate)
128
+ end
129
+ end
130
+ units
131
+ end
93
132
 
94
133
  # ──────────────────────────────────────────────────────────────────────
95
134
  # YAML-based formats (Solid Queue, Sidekiq-Cron)
@@ -190,6 +229,7 @@ module Woods
190
229
  unit.source_code = source
191
230
  unit.metadata = {
192
231
  schedule_format: format,
232
+ task_name: task_name.to_s,
193
233
  job_class: job_class,
194
234
  cron_expression: cron,
195
235
  queue: config['queue'],
@@ -324,6 +364,7 @@ module Woods
324
364
  unit.source_code = source
325
365
  unit.metadata = {
326
366
  schedule_format: :whenever,
367
+ task_name: identifier.delete_prefix('scheduled:'),
327
368
  job_class: block[:job_class],
328
369
  cron_expression: block[:frequency],
329
370
  command_type: block[:command_type],