woods 1.6.1 → 2.0.0.beta2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +2035 -0
- data/CONTRIBUTING.md +253 -87
- data/README.md +161 -513
- data/SECURITY.md +92 -0
- data/assets/woods-wordmark-white-with-bg.png +0 -0
- data/docs/AGENT_GUIDE.md +204 -0
- data/docs/AGENT_SETUP.md +205 -0
- data/docs/BACKEND_MATRIX.md +470 -0
- data/docs/CONFIGURATION_REFERENCE.md +655 -0
- data/docs/CONSOLE_MCP_SETUP.md +829 -0
- data/docs/DOCKER_SETUP.md +454 -0
- data/docs/EMBEDDING_MODELS.md +136 -0
- data/docs/EVALUATION.md +91 -0
- data/docs/EXTRACTOR_REFERENCE.md +765 -0
- data/docs/FAQ.md +544 -0
- data/docs/GETTING_STARTED.md +183 -0
- data/docs/INCREMENTAL_EXTRACTION.md +455 -0
- data/docs/INTERNALS.md +418 -0
- data/docs/MCP_HTTP_TRANSPORT.md +144 -0
- data/docs/MCP_SERVERS.md +231 -0
- data/docs/MCP_TOOL_COOKBOOK.md +987 -0
- data/docs/MCP_WORKTREE_SETUP.md +127 -0
- data/docs/NOTION_INTEGRATION.md +283 -0
- data/docs/OBSIDIAN_INTEGRATION.md +170 -0
- data/docs/PUBLISHED_INDEX.md +213 -0
- data/docs/README.md +94 -0
- data/docs/RETRIEVAL_GUIDE.md +267 -0
- data/docs/TOKEN_BENCHMARK.md +68 -0
- data/docs/TROUBLESHOOTING.md +841 -0
- data/docs/UNBLOCKED_INTEGRATION.md +279 -0
- data/docs/UPGRADING_TO_2.md +321 -0
- data/docs/WATCH_DAEMON.md +667 -0
- data/docs/WHY_WOODS.md +219 -0
- data/exe/woods-console +40 -4
- data/exe/woods-console-mcp +21 -35
- data/exe/woods-mcp +20 -7
- data/exe/woods-mcp-http +80 -11
- data/exe/woods-mcp-start +57 -52
- data/lib/generators/woods/install_generator.rb +6 -5
- data/lib/generators/woods/pgvector_generator.rb +6 -3
- data/lib/generators/woods/templates/add_pgvector_to_woods.rb.erb +29 -9
- data/lib/generators/woods/templates/create_woods_tables.rb.erb +5 -1
- data/lib/generators/woods/templates/woods.rb.tt +49 -28
- data/lib/tasks/woods.rake +622 -168
- data/lib/tasks/woods_checks.rake +107 -0
- data/lib/tasks/woods_evaluation.rake +164 -80
- data/lib/woods/ast/call_site_extractor.rb +6 -15
- data/lib/woods/ast/method_extractor.rb +19 -9
- data/lib/woods/ast/parser.rb +54 -8
- data/lib/woods/atomic_file.rb +171 -2
- data/lib/woods/builder.rb +310 -22
- data/lib/woods/cache/cache_middleware.rb +7 -2
- data/lib/woods/cache/cache_store.rb +9 -1
- data/lib/woods/cache/solid_cache_store.rb +6 -4
- data/lib/woods/change_set.rb +88 -0
- data/lib/woods/checks/generation_resolution.rb +34 -0
- data/lib/woods/checks/moved_messages.rb +186 -0
- data/lib/woods/chunking/semantic_chunker.rb +160 -18
- data/lib/woods/console/audit_logger.rb +12 -3
- data/lib/woods/console/bridge_protocol.rb +3 -16
- data/lib/woods/console/connection_manager.rb +51 -136
- data/lib/woods/console/dispatch_pipeline.rb +42 -12
- data/lib/woods/console/embedded_executor.rb +806 -149
- data/lib/woods/console/eval_guard.rb +27 -20
- data/lib/woods/console/input_contract.rb +78 -0
- data/lib/woods/console/model_validator.rb +29 -1
- data/lib/woods/console/rack_middleware.rb +65 -42
- data/lib/woods/console/redactor.rb +26 -8
- data/lib/woods/console/safe_context.rb +58 -10
- data/lib/woods/console/scope_predicate_parser.rb +41 -0
- data/lib/woods/console/server.rb +119 -247
- data/lib/woods/console/sql_noise_stripper.rb +125 -16
- data/lib/woods/console/sql_table_scanner.rb +82 -22
- data/lib/woods/console/sql_validator.rb +459 -29
- data/lib/woods/console/table_gate.rb +2 -2
- data/lib/woods/console/tool_specs.rb +463 -90
- data/lib/woods/console/tools/tier1.rb +1 -5
- data/lib/woods/console/tools/tier4.rb +18 -9
- data/lib/woods/coordination/lock_heartbeat.rb +103 -0
- data/lib/woods/coordination/pipeline_lock.rb +263 -53
- data/lib/woods/db/migrations/007_typed_snapshot_units.rb +45 -0
- data/lib/woods/db/migrator.rb +3 -9
- data/lib/woods/db/schema_version.rb +47 -2
- data/lib/woods/dependency_graph.rb +898 -64
- data/lib/woods/embedding/fake.rb +138 -0
- data/lib/woods/embedding/indexer.rb +832 -40
- data/lib/woods/embedding/openai.rb +77 -19
- data/lib/woods/embedding/provider.rb +189 -11
- data/lib/woods/embedding/text_preparer.rb +1 -1
- data/lib/woods/embedding/token_counter.rb +0 -7
- data/lib/woods/evaluation/ablation_agent_payload.rb +38 -0
- data/lib/woods/evaluation/ablation_executor.rb +67 -0
- data/lib/woods/evaluation/ablation_provenance.rb +38 -0
- data/lib/woods/evaluation/ablation_report_writer.rb +43 -0
- data/lib/woods/evaluation/ablation_runner.rb +173 -0
- data/lib/woods/evaluation/ablation_summary.rb +65 -0
- data/lib/woods/evaluation/ablation_task.rb +66 -0
- data/lib/woods/evaluation/ablation_task_set.rb +77 -0
- data/lib/woods/evaluation/ablation_timed_executor.rb +91 -0
- data/lib/woods/evaluation/ablation_worktree.rb +71 -0
- data/lib/woods/evaluation/baseline.rb +60 -0
- data/lib/woods/evaluation/baseline_runner.rb +11 -3
- data/lib/woods/evaluation/evaluator.rb +41 -8
- data/lib/woods/evaluation/query_set.rb +79 -13
- data/lib/woods/evaluation/report_generator.rb +20 -1
- data/lib/woods/export/unit_facts.rb +0 -11
- data/lib/woods/extracted_unit.rb +22 -63
- data/lib/woods/extractor.rb +2783 -238
- data/lib/woods/extractors/action_cable_extractor.rb +9 -4
- data/lib/woods/extractors/ast_source_extraction.rb +20 -2
- data/lib/woods/extractors/caching_extractor.rb +46 -12
- data/lib/woods/extractors/callback_analyzer.rb +39 -9
- data/lib/woods/extractors/component_discovery.rb +123 -0
- data/lib/woods/extractors/concern_extractor.rb +17 -3
- data/lib/woods/extractors/controller_extractor.rb +389 -29
- data/lib/woods/extractors/decorator_extractor.rb +7 -14
- data/lib/woods/extractors/engine_extractor.rb +53 -8
- data/lib/woods/extractors/event_extractor.rb +55 -4
- data/lib/woods/extractors/factory_extractor.rb +49 -11
- data/lib/woods/extractors/graphql_extractor.rb +162 -66
- data/lib/woods/extractors/i18n_extractor.rb +6 -1
- data/lib/woods/extractors/job_extractor.rb +51 -21
- data/lib/woods/extractors/lib_extractor.rb +23 -17
- data/lib/woods/extractors/line_neutralizer.rb +171 -0
- data/lib/woods/extractors/mailer_extractor.rb +9 -1
- data/lib/woods/extractors/manager_extractor.rb +19 -2
- data/lib/woods/extractors/migration_extractor.rb +22 -11
- data/lib/woods/extractors/model_extractor.rb +292 -57
- data/lib/woods/extractors/package_extractor.rb +154 -0
- data/lib/woods/extractors/phlex_extractor.rb +18 -3
- data/lib/woods/extractors/policy_extractor.rb +6 -5
- data/lib/woods/extractors/poro_extractor.rb +13 -14
- data/lib/woods/extractors/pundit_extractor.rb +3 -3
- data/lib/woods/extractors/rails_source_extractor.rb +24 -7
- data/lib/woods/extractors/rake_task_extractor.rb +158 -30
- data/lib/woods/extractors/reference_patterns.rb +38 -0
- data/lib/woods/extractors/route_extractor.rb +58 -2
- data/lib/woods/extractors/scheduled_job_extractor.rb +51 -35
- data/lib/woods/extractors/serializer_extractor.rb +3 -4
- data/lib/woods/extractors/service_extractor.rb +11 -1
- data/lib/woods/extractors/shared_dependency_scanner.rb +24 -34
- data/lib/woods/extractors/shared_utility_methods.rb +36 -6
- data/lib/woods/extractors/source_nesting.rb +560 -0
- data/lib/woods/extractors/state_machine_extractor.rb +30 -18
- data/lib/woods/extractors/test_mapping_extractor.rb +26 -9
- data/lib/woods/extractors/view_component_extractor.rb +28 -3
- data/lib/woods/extractors/view_engines/erb.rb +17 -3
- data/lib/woods/feedback/gap_detector.rb +9 -3
- data/lib/woods/feedback/store.rb +7 -1
- data/lib/woods/filename_utils.rb +29 -1
- data/lib/woods/flow_analysis/operation_extractor.rb +22 -10
- data/lib/woods/flow_assembler.rb +147 -26
- data/lib/woods/flow_document.rb +1 -0
- data/lib/woods/flow_precomputer.rb +175 -22
- data/lib/woods/gem_mapper.rb +285 -0
- data/lib/woods/generation.rb +185 -0
- data/lib/woods/git_command.rb +38 -0
- data/lib/woods/git_provenance.rb +16 -2
- data/lib/woods/graph_analyzer.rb +564 -87
- data/lib/woods/index_artifact.rb +93 -23
- data/lib/woods/mcp/bearer_auth.rb +102 -13
- data/lib/woods/mcp/bootstrap_state.rb +77 -0
- data/lib/woods/mcp/bootstrapper.rb +582 -77
- data/lib/woods/mcp/config_resolver.rb +66 -6
- data/lib/woods/mcp/errors.rb +60 -0
- data/lib/woods/mcp/index_reader.rb +836 -117
- data/lib/woods/mcp/index_reader_pinning.rb +78 -0
- data/lib/woods/mcp/origin_guard.rb +66 -7
- data/lib/woods/mcp/protocol_policy.rb +98 -0
- data/lib/woods/mcp/provider_probe.rb +45 -6
- data/lib/woods/mcp/renderers/markdown_renderer.rb +72 -4
- data/lib/woods/mcp/renderers/plain_renderer.rb +54 -6
- data/lib/woods/mcp/server.rb +898 -152
- data/lib/woods/mcp/tasks/extension.rb +196 -0
- data/lib/woods/mcp/tasks/request_capture.rb +45 -0
- data/lib/woods/mcp/tasks/store.rb +518 -0
- data/lib/woods/mcp/tool_contract.rb +171 -0
- data/lib/woods/mcp/tool_response_renderer.rb +7 -0
- data/lib/woods/model_name_cache.rb +19 -1
- data/lib/woods/notion/client.rb +132 -36
- data/lib/woods/notion/exporter.rb +456 -61
- data/lib/woods/notion/mappers/column_mapper.rb +34 -5
- data/lib/woods/notion/mappers/migration_mapper.rb +32 -8
- data/lib/woods/notion/mappers/model_mapper.rb +21 -6
- data/lib/woods/notion/mappers/shared.rb +45 -3
- data/lib/woods/notion/sync_manifest.rb +258 -0
- data/lib/woods/obsidian/errors.rb +6 -0
- data/lib/woods/obsidian/name_mapper.rb +40 -24
- data/lib/woods/obsidian/vault_exporter.rb +103 -36
- data/lib/woods/operator/pipeline_guard.rb +118 -21
- data/lib/woods/operator/status_reporter.rb +20 -3
- data/lib/woods/path_dispatcher.rb +276 -0
- data/lib/woods/payload_store.rb +236 -0
- data/lib/woods/published_index/edge_shaper.rb +61 -0
- data/lib/woods/published_index/generation_catalog.rb +72 -0
- data/lib/woods/published_index/typed_unit_reader.rb +48 -0
- data/lib/woods/published_index.rb +287 -0
- data/lib/woods/railtie.rb +69 -30
- data/lib/woods/railtie_support.rb +167 -0
- data/lib/woods/release.rb +12 -0
- data/lib/woods/reload_policy.rb +206 -0
- data/lib/woods/resilience/circuit_breaker.rb +47 -8
- data/lib/woods/resilience/index_validator.rb +296 -10
- data/lib/woods/resilience/retryable_provider.rb +71 -6
- data/lib/woods/resolved_config.rb +55 -11
- data/lib/woods/retrieval/context_assembler.rb +132 -40
- data/lib/woods/retrieval/query_classifier.rb +26 -8
- data/lib/woods/retrieval/ranker.rb +193 -28
- data/lib/woods/retrieval/search_executor.rb +206 -39
- data/lib/woods/retriever.rb +317 -71
- data/lib/woods/retry_after.rb +22 -2
- data/lib/woods/ruby_analyzer/class_analyzer.rb +10 -14
- data/lib/woods/ruby_analyzer/fqn_builder.rb +2 -0
- data/lib/woods/ruby_analyzer/mermaid_renderer.rb +14 -4
- data/lib/woods/ruby_analyzer/method_analyzer.rb +1 -1
- data/lib/woods/ruby_analyzer/trace_enricher.rb +3 -0
- data/lib/woods/ruby_analyzer.rb +21 -5
- data/lib/woods/session_tracer/file_store.rb +138 -19
- data/lib/woods/session_tracer/middleware.rb +1 -2
- data/lib/woods/session_tracer/redis_store.rb +122 -12
- data/lib/woods/session_tracer/session_flow_assembler.rb +57 -17
- data/lib/woods/session_tracer/session_flow_document.rb +56 -14
- data/lib/woods/session_tracer/solid_cache_coordination.rb +192 -0
- data/lib/woods/session_tracer/solid_cache_store.rb +560 -91
- data/lib/woods/session_tracer/store.rb +14 -1
- data/lib/woods/storage/metadata_store.rb +230 -26
- data/lib/woods/storage/pgvector.rb +180 -22
- data/lib/woods/storage/qdrant.rb +367 -41
- data/lib/woods/storage/snapshotter/metadata.rb +79 -16
- data/lib/woods/storage/snapshotter/vector.rb +128 -17
- data/lib/woods/storage/snapshotter.rb +23 -5
- data/lib/woods/storage/vector_store.rb +49 -8
- data/lib/woods/storage_identity.rb +28 -0
- data/lib/woods/tasks.rb +53 -2
- data/lib/woods/temporal/json_snapshot_store.rb +112 -42
- data/lib/woods/temporal/snapshot_store.rb +139 -42
- data/lib/woods/unblocked/client.rb +119 -17
- data/lib/woods/unblocked/document_builder.rb +34 -2
- data/lib/woods/unblocked/exporter.rb +63 -27
- data/lib/woods/unblocked/rate_limiter.rb +23 -9
- data/lib/woods/unblocked/sync_manifest.rb +16 -8
- data/lib/woods/update_check.rb +24 -1
- data/lib/woods/util/uuid5.rb +124 -0
- data/lib/woods/version.rb +1 -1
- data/lib/woods/watch/daemon.rb +1345 -0
- data/lib/woods/watch/listen_watcher.rb +81 -0
- data/lib/woods/watch/polling_watcher.rb +137 -0
- data/lib/woods/watch/status.rb +169 -0
- data/lib/woods/watch/tree_scan.rb +163 -0
- data/lib/woods/watch/watcher.rb +100 -0
- data/lib/woods.rb +138 -9
- data/plugin/.claude-plugin/plugin.json +18 -0
- data/plugin/hooks/hooks.json +29 -0
- data/plugin/hooks/woods-post-edit.sh +226 -0
- data/plugin/hooks/woods-session-start.sh +77 -0
- data/plugin/skills/woods-agent-enable/SKILL.md +51 -0
- data/plugin/skills/woods-diagnose/SKILL.md +75 -0
- data/plugin/skills/woods-investigate/SKILL.md +39 -0
- data/plugin/skills/woods-mcp-config/SKILL.md +101 -0
- data/plugin/skills/woods-setup/SKILL.md +99 -0
- metadata +134 -23
- data/lib/woods/console/adapters/cache_adapter.rb +0 -58
- data/lib/woods/console/adapters/good_job_adapter.rb +0 -33
- data/lib/woods/console/adapters/job_adapter.rb +0 -74
- data/lib/woods/console/adapters/sidekiq_adapter.rb +0 -33
- data/lib/woods/console/adapters/solid_queue_adapter.rb +0 -33
- data/lib/woods/console/bridge.rb +0 -210
- data/lib/woods/formatting/claude_adapter.rb +0 -98
- data/lib/woods/formatting/generic_adapter.rb +0 -56
- data/lib/woods/formatting/gpt_adapter.rb +0 -64
- data/lib/woods/notion/mapper.rb +0 -40
- data/lib/woods/observability/health_check.rb +0 -79
- data/lib/woods/observability/instrumentation.rb +0 -34
data/lib/woods/extractor.rb
CHANGED
|
@@ -7,10 +7,12 @@ require 'open3'
|
|
|
7
7
|
require 'pathname'
|
|
8
8
|
require 'set'
|
|
9
9
|
|
|
10
|
+
require_relative 'atomic_file'
|
|
10
11
|
require_relative 'filename_utils'
|
|
11
12
|
require_relative 'token_utils'
|
|
12
13
|
require_relative 'extracted_unit'
|
|
13
14
|
require_relative 'dependency_graph'
|
|
15
|
+
require_relative 'payload_store'
|
|
14
16
|
require_relative 'git_provenance'
|
|
15
17
|
require_relative 'extractors/model_extractor'
|
|
16
18
|
require_relative 'extractors/controller_extractor'
|
|
@@ -46,9 +48,13 @@ require_relative 'extractors/factory_extractor'
|
|
|
46
48
|
require_relative 'extractors/test_mapping_extractor'
|
|
47
49
|
require_relative 'extractors/poro_extractor'
|
|
48
50
|
require_relative 'extractors/lib_extractor'
|
|
51
|
+
require_relative 'extractors/package_extractor'
|
|
49
52
|
require_relative 'graph_analyzer'
|
|
50
53
|
require_relative 'model_name_cache'
|
|
51
54
|
require_relative 'flow_precomputer'
|
|
55
|
+
require_relative 'change_set'
|
|
56
|
+
require_relative 'generation'
|
|
57
|
+
require_relative 'path_dispatcher'
|
|
52
58
|
|
|
53
59
|
module Woods
|
|
54
60
|
# Extractor is the main orchestrator for codebase extraction.
|
|
@@ -66,6 +72,7 @@ module Woods
|
|
|
66
72
|
#
|
|
67
73
|
class Extractor
|
|
68
74
|
include FilenameUtils
|
|
75
|
+
include Extractors::SourceNesting
|
|
69
76
|
|
|
70
77
|
# Directories under app/ that contain classes we need to extract.
|
|
71
78
|
# Used by eager_load_extraction_directories as a fallback when
|
|
@@ -126,7 +133,8 @@ module Woods
|
|
|
126
133
|
test_mappings: Extractors::TestMappingExtractor,
|
|
127
134
|
rails_source: Extractors::RailsSourceExtractor,
|
|
128
135
|
poros: Extractors::PoroExtractor,
|
|
129
|
-
libs: Extractors::LibExtractor
|
|
136
|
+
libs: Extractors::LibExtractor,
|
|
137
|
+
packages: Extractors::PackageExtractor
|
|
130
138
|
}.freeze
|
|
131
139
|
|
|
132
140
|
# Maps singular unit types (as stored in ExtractedUnit/graph nodes)
|
|
@@ -169,8 +177,17 @@ module Woods
|
|
|
169
177
|
factory: :factories,
|
|
170
178
|
test_mapping: :test_mappings,
|
|
171
179
|
rails_source: :rails_source,
|
|
180
|
+
# RailsSourceExtractor emits BOTH :rails_source and :gem_source units
|
|
181
|
+
# into the rails_source/ output directory. Without this entry the
|
|
182
|
+
# wholesale-replacement path could not prune a stale gem_source unit
|
|
183
|
+
# ({EXTRACTOR_KEY_TO_TYPES} would not list the type) and {#remove_unit}
|
|
184
|
+
# could not resolve its JSON file on disk (#169) — the GraphQL history
|
|
185
|
+
# in CLAUDE.md is the cautionary tale for an extractor whose emitted
|
|
186
|
+
# types are not all mapped.
|
|
187
|
+
gem_source: :rails_source,
|
|
172
188
|
poro: :poros,
|
|
173
|
-
lib: :libs
|
|
189
|
+
lib: :libs,
|
|
190
|
+
package: :packages
|
|
174
191
|
}.freeze
|
|
175
192
|
|
|
176
193
|
# Maps unit types to class-based extractor methods (constantize + call).
|
|
@@ -203,13 +220,170 @@ module Woods
|
|
|
203
220
|
# GraphQL types all use the same extractor method.
|
|
204
221
|
GRAPHQL_TYPES = %i[graphql_type graphql_mutation graphql_resolver graphql_query].freeze
|
|
205
222
|
|
|
223
|
+
# File-based unit types the full path ALSO discovers by class, with the
|
|
224
|
+
# class-based entry point to re-extract them through. Re-extracting one
|
|
225
|
+
# of these by file is faithful only when the file names the unit; see
|
|
226
|
+
# {#re_extracted_units}.
|
|
227
|
+
CLASS_DISCOVERED_FALLBACK = { job: :extract_job_class }.freeze
|
|
228
|
+
|
|
229
|
+
# Unit types each extractor owns — the inverse of {TYPE_TO_EXTRACTOR_KEY}.
|
|
230
|
+
#
|
|
231
|
+
# Wholesale replacement of an extractor's output has to know every type it
|
|
232
|
+
# can produce, and one extractor can own several (`graphql` covers types,
|
|
233
|
+
# mutations, resolvers and queries).
|
|
234
|
+
#
|
|
235
|
+
# @return [Hash{Symbol => Array<Symbol>}]
|
|
236
|
+
EXTRACTOR_KEY_TO_TYPES = TYPE_TO_EXTRACTOR_KEY.each_with_object({}) do |(type, key), map|
|
|
237
|
+
(map[key] ||= []) << type
|
|
238
|
+
end.freeze
|
|
239
|
+
|
|
240
|
+
# Class-based extractors, which discover their units by walking runtime
|
|
241
|
+
# descendants rather than by globbing files. The incremental path
|
|
242
|
+
# reconciles each extractor's `discoverable_classes` against the
|
|
243
|
+
# identifiers already in the graph, so a class added since the last
|
|
244
|
+
# extraction is found without having to guess a constant name from a
|
|
245
|
+
# path (#164, gap 1).
|
|
246
|
+
#
|
|
247
|
+
# `reconcile_removals: false` opts an entry out of the removal half. Only
|
|
248
|
+
# GraphQL sets it, and the reason is that its unit type is produced by two
|
|
249
|
+
# independent discovery mechanisms whose union is the truth: runtime schema
|
|
250
|
+
# introspection *and* a static file pass over `app/graphql`. Every other
|
|
251
|
+
# entry here owns its type outright, which is what makes
|
|
252
|
+
# "in the graph but not in `discoverable_classes`" mean "deleted".
|
|
253
|
+
#
|
|
254
|
+
# For GraphQL that inference is wrong twice over. Without graphql-ruby
|
|
255
|
+
# loaded the runtime set is empty, so removal would delete every GraphQL
|
|
256
|
+
# unit in the index; with it loaded, a type defined in a file but not
|
|
257
|
+
# attached to the schema is legitimately absent from the type map, so
|
|
258
|
+
# removal would delete units a full extraction still emits. Additions are
|
|
259
|
+
# safe in both worlds — they are exactly the runtime-only types no changed
|
|
260
|
+
# path can dispatch to, which is the #167 divergence.
|
|
261
|
+
#
|
|
262
|
+
# The consequence, stated so it does not read as a bug later: a runtime-only
|
|
263
|
+
# type that *disappears* is never removed by an incremental run. It survives
|
|
264
|
+
# until a full extraction rebuilds the type. Confirmed against a live schema
|
|
265
|
+
# during review — probe units outlived deletion of the file that defined
|
|
266
|
+
# them. That is the cost of the opt-out, and it is the right side to err on:
|
|
267
|
+
# a stale unit is recoverable by one full run, whereas the removal half
|
|
268
|
+
# would delete units a full extraction still emits.
|
|
269
|
+
#
|
|
270
|
+
# @return [Hash{Symbol => Hash}] extractor key => { type:, method: }
|
|
271
|
+
CLASS_BASED_DISCOVERY = {
|
|
272
|
+
models: { type: :model, method: :extract_model },
|
|
273
|
+
controllers: { type: :controller, method: :extract_controller },
|
|
274
|
+
mailers: { type: :mailer, method: :extract_mailer },
|
|
275
|
+
components: { type: :component, method: :extract_component },
|
|
276
|
+
view_components: { type: :view_component, method: :extract_component },
|
|
277
|
+
action_cable_channels: { type: :action_cable_channel, method: :extract_channel },
|
|
278
|
+
# Additions only — see `reconcile_removals: false`. GraphQL is the one
|
|
279
|
+
# entry whose discovery set is not authoritative for its unit type
|
|
280
|
+
# (#167).
|
|
281
|
+
# `types:` because this extractor emits four unit types, not one:
|
|
282
|
+
# `classify_runtime_type` returns :graphql_type, :graphql_mutation,
|
|
283
|
+
# :graphql_resolver or :graphql_query. Declaring only :graphql_type made
|
|
284
|
+
# `known` miss the others — the schema's query root is always in
|
|
285
|
+
# `Schema.types` and classifies as :graphql_query — so they were
|
|
286
|
+
# re-"discovered" and re-added on *every* incremental run. That leaves
|
|
287
|
+
# `touched` non-empty on a genuine no-op, which rewrites the manifest and
|
|
288
|
+
# bumps the generation each cycle: the #164 gap-4 symptom, reintroduced.
|
|
289
|
+
graphql: { type: :graphql_type, types: GRAPHQL_TYPES,
|
|
290
|
+
method: :extract_from_runtime_type, reconcile_removals: false }
|
|
291
|
+
}.freeze
|
|
292
|
+
|
|
293
|
+
# Extractors with no per-file entry point: they scan the whole app (or
|
|
294
|
+
# introspect the whole runtime) in one pass, so an incremental run
|
|
295
|
+
# replaces their output wholesale rather than per unit. Before #164
|
|
296
|
+
# these types were simply skipped by incremental runs while
|
|
297
|
+
# `config/routes.rb` still triggered one — the run rewrote the manifest,
|
|
298
|
+
# zeroing the staleness clock, without updating the data.
|
|
299
|
+
#
|
|
300
|
+
# @return [Hash{Symbol => Symbol}] extractor key => unit type
|
|
301
|
+
WHOLE_APP_EXTRACTORS = {
|
|
302
|
+
routes: :route,
|
|
303
|
+
middleware: :middleware,
|
|
304
|
+
engines: :engine,
|
|
305
|
+
scheduled_jobs: :scheduled_job,
|
|
306
|
+
state_machines: :state_machine,
|
|
307
|
+
factories: :factory,
|
|
308
|
+
events: :event,
|
|
309
|
+
# DatabaseViewExtractor keeps only the highest `_vNN` of each Scenic
|
|
310
|
+
# view, so its unit set is a function of the whole directory rather
|
|
311
|
+
# than of each file independently: dispatching db/views/foo_v01.sql to
|
|
312
|
+
# the per-file method would index a version a full extraction drops.
|
|
313
|
+
database_views: :database_view,
|
|
314
|
+
# Framework/gem sources are a function of the installed dependency set,
|
|
315
|
+
# so `Gemfile.lock` is their trigger path (#169). Before this the only
|
|
316
|
+
# incremental writer was `woods:extract_framework`, which hand-wrote
|
|
317
|
+
# JSON around the whole pipeline — no AtomicFile, no _index.json, no
|
|
318
|
+
# manifest counts, no lock, no generation bump — and its output then
|
|
319
|
+
# sat stale forever. The value here names the primary unit type, like
|
|
320
|
+
# every other entry; the extractor also owns :gem_source, and
|
|
321
|
+
# {EXTRACTOR_KEY_TO_TYPES} (which wholesale replacement consults) lists
|
|
322
|
+
# both. Participation is gated by `include_framework_sources` — see
|
|
323
|
+
# {#skip_by_configuration?}.
|
|
324
|
+
rails_source: :rails_source,
|
|
325
|
+
# A package root decides which package every other unit belongs to,
|
|
326
|
+
# and the undeclared-edge report reads the whole declared set, so any
|
|
327
|
+
# package.yml change re-runs the extractor wholesale (#280). Task 8
|
|
328
|
+
# re-annotates unit membership in the same run.
|
|
329
|
+
packages: :package
|
|
330
|
+
}.freeze
|
|
331
|
+
|
|
332
|
+
# Extractors whose output embeds the route table, and which therefore go
|
|
333
|
+
# stale when routes change even though none of their own files did.
|
|
334
|
+
#
|
|
335
|
+
# ControllerExtractor writes each action's routes into unit metadata and
|
|
336
|
+
# into the action chunks; everything that includes `RouteHelperResolver`
|
|
337
|
+
# resolves `_path`/`_url` references into navigation edges against the
|
|
338
|
+
# same table. The dependency graph can't express this — a route unit
|
|
339
|
+
# depends *on* its controller, so walking dependents from
|
|
340
|
+
# `config/routes.rb` never reaches it — so the relationship is declared
|
|
341
|
+
# here instead and these types are re-extracted wholesale whenever the
|
|
342
|
+
# route set is re-run (#164).
|
|
343
|
+
#
|
|
344
|
+
# @return [Array<Symbol>] extractor keys
|
|
345
|
+
ROUTE_CONSUMER_EXTRACTORS = %i[
|
|
346
|
+
controllers
|
|
347
|
+
mailers
|
|
348
|
+
components
|
|
349
|
+
view_components
|
|
350
|
+
view_templates
|
|
351
|
+
].freeze
|
|
352
|
+
|
|
353
|
+
# Payload artifacts that live at the top of a payload directory rather
|
|
354
|
+
# than inside a per-type directory. Used when seeding a payload from a
|
|
355
|
+
# flat index — the output root also holds `generation.json`, `dumps/`,
|
|
356
|
+
# `tasks/`, `woods.sqlite3` and `payloads/` itself, none of which belong
|
|
357
|
+
# to a generation's payload.
|
|
358
|
+
PAYLOAD_FILES = %w[manifest.json dependency_graph.json graph_analysis.json SUMMARY.md].freeze
|
|
359
|
+
|
|
360
|
+
# Payload directories that are not per-type unit directories.
|
|
361
|
+
PAYLOAD_DIRS = %w[flows].freeze
|
|
362
|
+
|
|
206
363
|
attr_reader :output_dir, :dependency_graph
|
|
207
364
|
|
|
208
365
|
def initialize(output_dir: nil)
|
|
209
366
|
@output_dir = Pathname.new(output_dir || Rails.root.join('tmp/woods'))
|
|
367
|
+
@payload_store = PayloadStore.new(@output_dir)
|
|
368
|
+
@payload_dir = nil
|
|
369
|
+
@payload_generation = nil
|
|
210
370
|
@dependency_graph = DependencyGraph.new
|
|
211
371
|
@results = {}
|
|
212
372
|
@extractors = {}
|
|
373
|
+
@publication_error = nil
|
|
374
|
+
end
|
|
375
|
+
|
|
376
|
+
# Where this run reads and writes payload artifacts.
|
|
377
|
+
#
|
|
378
|
+
# A run publishes into an immutable per-generation directory, so that the
|
|
379
|
+
# single atomic write of `generation.json` commits the whole payload at
|
|
380
|
+
# once. Falls back to the output root when no payload directory could be
|
|
381
|
+
# opened — a flat index is non-atomic but perfectly readable, and failing
|
|
382
|
+
# the extraction over it would be worse.
|
|
383
|
+
#
|
|
384
|
+
# @return [Pathname]
|
|
385
|
+
def payload_dir
|
|
386
|
+
@payload_dir || @output_dir
|
|
213
387
|
end
|
|
214
388
|
|
|
215
389
|
# ══════════════════════════════════════════════════════════════════════
|
|
@@ -222,23 +396,43 @@ module Woods
|
|
|
222
396
|
def extract_all
|
|
223
397
|
setup_output_directory
|
|
224
398
|
ModelNameCache.reset!
|
|
399
|
+
# @package_resolver alone is not enough: #package_resolver builds
|
|
400
|
+
# through #extractor_for, which memoizes into @incremental_extractors.
|
|
401
|
+
# Without this reset, a second full run on the same instance would
|
|
402
|
+
# resolve membership through the first run's PackageExtractor and its
|
|
403
|
+
# already-memoized (now stale) package_files/package_roots, missing a
|
|
404
|
+
# package added between the two runs.
|
|
405
|
+
@package_resolver = nil
|
|
406
|
+
@incremental_extractors = nil
|
|
407
|
+
@persisted_index_stats = nil
|
|
408
|
+
@graph_sha = nil
|
|
409
|
+
profile_phase('payload seed') { begin_payload! }
|
|
225
410
|
|
|
226
411
|
# Eager load once — all extractors need loaded classes for introspection.
|
|
227
|
-
safe_eager_load!
|
|
412
|
+
profile_phase('eager load') { safe_eager_load! }
|
|
228
413
|
|
|
229
414
|
# Phase 1: Extract all units
|
|
230
|
-
|
|
231
|
-
|
|
232
|
-
|
|
233
|
-
|
|
415
|
+
profile_phase('extraction') do
|
|
416
|
+
if Woods.configuration.concurrent_extraction
|
|
417
|
+
extract_all_concurrent
|
|
418
|
+
else
|
|
419
|
+
extract_all_sequential
|
|
420
|
+
end
|
|
234
421
|
end
|
|
235
422
|
|
|
236
423
|
# Phase 1.5: Deduplicate results
|
|
237
424
|
Rails.logger.info '[Woods] Deduplicating results...'
|
|
238
425
|
deduplicate_results
|
|
239
426
|
|
|
240
|
-
#
|
|
241
|
-
#
|
|
427
|
+
# Phase 1.6: Package membership. Runs before the graph is rebuilt so
|
|
428
|
+
# registration copies metadata[:package] onto the node (#280).
|
|
429
|
+
annotate_packages
|
|
430
|
+
|
|
431
|
+
# Rebuild the graph from deduped results. #164 gave DependencyGraph
|
|
432
|
+
# `#remove`/`#unregister`, so surgical removal is now possible — but a
|
|
433
|
+
# full extraction has just registered every unit including duplicates,
|
|
434
|
+
# and rebuilding from the deduped set is both cheaper and less
|
|
435
|
+
# error-prone than unwinding registrations one at a time.
|
|
242
436
|
@dependency_graph = DependencyGraph.new
|
|
243
437
|
@results.each_value { |units| units.each { |u| @dependency_graph.register(u) } }
|
|
244
438
|
|
|
@@ -246,13 +440,15 @@ module Woods
|
|
|
246
440
|
Rails.logger.info '[Woods] Resolving dependents...'
|
|
247
441
|
resolve_dependents
|
|
248
442
|
|
|
249
|
-
# Phase 3:
|
|
250
|
-
|
|
251
|
-
@graph_analysis = GraphAnalyzer.new(@dependency_graph).analyze
|
|
252
|
-
|
|
253
|
-
# Phase 4: Enrich with git data
|
|
443
|
+
# Phase 3: Enrich with git data. Runs BEFORE analysis now: the
|
|
444
|
+
# volatile_dependencies report reads commit counts off graph nodes.
|
|
254
445
|
Rails.logger.info '[Woods] Enriching with git data...'
|
|
255
446
|
enrich_with_git_data
|
|
447
|
+
annotate_graph_with_git_data
|
|
448
|
+
|
|
449
|
+
# Phase 4: Graph analysis (PageRank, structural metrics)
|
|
450
|
+
Rails.logger.info '[Woods] Analyzing dependency graph...'
|
|
451
|
+
@graph_analysis = profile_phase('graph analysis') { build_graph_analyzer.analyze }
|
|
256
452
|
|
|
257
453
|
# Phase 4.5: Normalize file_path to relative paths
|
|
258
454
|
Rails.logger.info '[Woods] Normalizing file paths...'
|
|
@@ -260,23 +456,43 @@ module Woods
|
|
|
260
456
|
|
|
261
457
|
# Phase 5: Write output
|
|
262
458
|
Rails.logger.info '[Woods] Writing output...'
|
|
263
|
-
write_results
|
|
459
|
+
profile_phase('write results') { write_results }
|
|
460
|
+
|
|
461
|
+
# Phase 5.1: Sweep unit files no current unit accounts for (#177). Must
|
|
462
|
+
# run after write_results — the just-written set is what defines
|
|
463
|
+
# "legitimate" — and belongs to the full path only; the incremental path
|
|
464
|
+
# deletes through the graph instead. See {#sweep_orphaned_unit_files}.
|
|
465
|
+
sweep_orphaned_unit_files
|
|
264
466
|
|
|
265
467
|
# Phase 5.5: Precompute request flows (opt-in). Must run AFTER
|
|
266
468
|
# write_results — FlowAssembler loads unit JSON from disk, so running
|
|
267
469
|
# earlier assembled every flow from absent (fresh output dir) or
|
|
268
470
|
# stale (previous run's) data. precompute_flows re-writes the
|
|
269
471
|
# controller units it annotates with metadata[:flow_paths].
|
|
472
|
+
#
|
|
473
|
+
# Fail closed (M3 review round 3): a failure anywhere in the family —
|
|
474
|
+
# assembly, index write, annotation rewrite, sweep — raises and aborts
|
|
475
|
+
# the run here, BEFORE the graph, manifest, snapshot, and generation
|
|
476
|
+
# publish below. The previously published generation stays resolved
|
|
477
|
+
# and readable; the aborted run's payload directory is left
|
|
478
|
+
# unreachable for the next run's PayloadStore#create to reclaim.
|
|
479
|
+
# Publishing a partial flow index, stale prior flow artifacts
|
|
480
|
+
# alongside a new graph, or half-rewritten annotations would violate
|
|
481
|
+
# the atomic-generation and full/incremental-equivalence contracts;
|
|
482
|
+
# being opt-in does not excuse it.
|
|
270
483
|
if Woods.configuration.precompute_flows
|
|
271
484
|
Rails.logger.info '[Woods] Precomputing request flows...'
|
|
272
|
-
precompute_flows
|
|
485
|
+
profile_phase('flows') { precompute_flows }
|
|
273
486
|
end
|
|
274
487
|
|
|
275
488
|
write_dependency_graph
|
|
276
489
|
write_graph_analysis
|
|
277
|
-
|
|
278
|
-
|
|
490
|
+
profile_phase('manifest and summary') do
|
|
491
|
+
write_manifest
|
|
492
|
+
write_structural_summary
|
|
493
|
+
end
|
|
279
494
|
capture_snapshot
|
|
495
|
+
profile_phase('publish') { publish_generation('full') }
|
|
280
496
|
|
|
281
497
|
log_summary
|
|
282
498
|
|
|
@@ -287,57 +503,600 @@ module Woods
|
|
|
287
503
|
# Incremental Extraction
|
|
288
504
|
# ══════════════════════════════════════════════════════════════════════
|
|
289
505
|
|
|
290
|
-
# Extract only units affected by changed files
|
|
291
|
-
#
|
|
506
|
+
# Extract only units affected by changed files.
|
|
507
|
+
#
|
|
508
|
+
# Used for incremental indexing in CI and by any caller that maintains a
|
|
509
|
+
# live index. The goal is equivalence: after this returns, the on-disk
|
|
510
|
+
# index should match what a cold full extraction of the same tree would
|
|
511
|
+
# have produced.
|
|
512
|
+
#
|
|
513
|
+
# The run proceeds in a fixed order, and the order matters:
|
|
514
|
+
#
|
|
515
|
+
# 1. Blast radius is computed against the *pre-change* graph, so
|
|
516
|
+
# dependents of a file that just disappeared still get re-extracted.
|
|
517
|
+
# 2. Known units in the blast radius are re-extracted.
|
|
518
|
+
# 3. Changed paths the index has never seen are dispatched to an
|
|
519
|
+
# extractor by path ({PathDispatcher}).
|
|
520
|
+
# 4. Class-based types are reconciled against their runtime discovery
|
|
521
|
+
# sets, catching classes added since the last extraction.
|
|
522
|
+
# 5. Whole-app extractors whose trigger paths changed are re-run and
|
|
523
|
+
# their type replaced wholesale.
|
|
524
|
+
# 6. Units whose source file has vanished are pruned last, so anything
|
|
525
|
+
# resurrected by steps 2–5 against a deleted file is swept in the
|
|
526
|
+
# same run rather than surviving as a ghost until the next one.
|
|
292
527
|
#
|
|
293
528
|
# @param changed_files [Array<String>] List of changed file paths
|
|
294
|
-
# @return [Array<String>]
|
|
529
|
+
# @return [Array<String>] Identifiers of units re-extracted, added, or removed
|
|
295
530
|
def extract_changed(changed_files)
|
|
296
|
-
|
|
297
|
-
graph_path = @output_dir.join('dependency_graph.json')
|
|
298
|
-
@dependency_graph = DependencyGraph.from_h(JSON.parse(File.read(graph_path))) if graph_path.exist?
|
|
531
|
+
prepare_incremental_run
|
|
299
532
|
|
|
300
|
-
|
|
533
|
+
change_set = ChangeSet.new(paths: changed_files, root: Rails.root)
|
|
534
|
+
affected_types = Set.new
|
|
535
|
+
|
|
536
|
+
# Blast radius from the pre-change graph, bounded by
|
|
537
|
+
# `incremental_blast_radius_depth` (nil = the unbounded closure).
|
|
538
|
+
affected_ids = profile_phase('blast radius') do
|
|
539
|
+
@dependency_graph.affected_by(change_set.absolute_paths, max_depth: blast_radius_depth)
|
|
540
|
+
end
|
|
541
|
+
@flow_scope = profile_phase('flow radius') { flow_scope_for(change_set) }
|
|
542
|
+
Rails.logger.info "[Woods] #{change_set.size} changed files affect #{affected_ids.size} units"
|
|
543
|
+
|
|
544
|
+
touched = profile_phase('re-extraction') do
|
|
545
|
+
acc = reconcile_changed_paths(change_set, affected_types)
|
|
546
|
+
|
|
547
|
+
(affected_ids - acc.to_a).each do |unit_id|
|
|
548
|
+
acc.add(unit_id) if re_extract_unit(unit_id, affected_types: affected_types)
|
|
549
|
+
end
|
|
550
|
+
|
|
551
|
+
acc
|
|
552
|
+
end
|
|
301
553
|
|
|
302
|
-
|
|
303
|
-
|
|
554
|
+
touched.merge(reconcile_class_based_types(affected_types))
|
|
555
|
+
touched.merge(rerun_whole_app_extractors(change_set, affected_types))
|
|
556
|
+
touched.merge(reannotate_packages(change_set, affected_types))
|
|
557
|
+
pruned = prune_vanished_units(change_set, affected_types)
|
|
558
|
+
touched.merge(pruned)
|
|
559
|
+
|
|
560
|
+
# Reconcile once more, because pruning can un-know a class the first pass
|
|
561
|
+
# skipped. A class-based file moved between autoload directories with its
|
|
562
|
+
# constant unchanged is still registered under the old path when
|
|
563
|
+
# reconciliation runs, so it looks known and is not re-extracted; the
|
|
564
|
+
# prune that follows then removes it for its vanished path. This pass
|
|
565
|
+
# re-adds it in the same run (M1) instead of leaving the unit missing
|
|
566
|
+
# until some later run happens to notice. Idempotent when nothing was
|
|
567
|
+
# pruned: the discovery set is compared against the graph, so an
|
|
568
|
+
# already-registered class is skipped.
|
|
569
|
+
#
|
|
570
|
+
# But not everything pruning removed may come back. `except:` keeps the
|
|
571
|
+
# *deletion* shape pruned: without a reload, a constant outlives the file
|
|
572
|
+
# that defined it — so deleting `app/models/user.rb` prunes `User`, and
|
|
573
|
+
# this pass finds `User` still in `ActiveRecord::Base.descendants`.
|
|
574
|
+
# Re-registering it would pin the unit to a path that no longer exists,
|
|
575
|
+
# and nothing could ever remove it: the sweep excludes class-based units
|
|
576
|
+
# and no future change set names that path again. A resident daemon
|
|
577
|
+
# processing a batch before its reload hits this every time. What
|
|
578
|
+
# separates the two shapes is the filesystem — only pruned identifiers
|
|
579
|
+
# that a still-existing file in the change set actually declares are
|
|
580
|
+
# re-addable. See {#readdable_pruned_classes}.
|
|
581
|
+
touched.merge(reconcile_class_based_types(
|
|
582
|
+
affected_types, except: pruned - readdable_pruned_classes(pruned, change_set)
|
|
583
|
+
))
|
|
584
|
+
|
|
585
|
+
finalize_incremental_unit_json(affected_types)
|
|
304
586
|
|
|
305
|
-
#
|
|
306
|
-
|
|
307
|
-
|
|
587
|
+
# Regenerate type indexes for affected types
|
|
588
|
+
profile_phase('type index') do
|
|
589
|
+
affected_types.each do |type_key|
|
|
590
|
+
regenerate_type_index(type_key)
|
|
591
|
+
end
|
|
308
592
|
end
|
|
309
593
|
|
|
310
|
-
|
|
311
|
-
|
|
312
|
-
|
|
594
|
+
finalize_incremental_run(touched)
|
|
595
|
+
|
|
596
|
+
touched.to_a
|
|
597
|
+
end
|
|
598
|
+
|
|
599
|
+
# ══════════════════════════════════════════════════════════════════════
|
|
600
|
+
# Targeted Refresh
|
|
601
|
+
# ══════════════════════════════════════════════════════════════════════
|
|
313
602
|
|
|
314
|
-
|
|
603
|
+
# Re-run one or more extractors wholesale against the already-booted app,
|
|
604
|
+
# replacing every unit of the types they own.
|
|
605
|
+
#
|
|
606
|
+
# This is the escape hatch for the unit types that have no per-file entry
|
|
607
|
+
# point — routes are derived from `Rails.application.routes`, the
|
|
608
|
+
# middleware unit from the live stack, events from a two-pass scan of all
|
|
609
|
+
# of `app/`. Incremental runs reach them by trigger path
|
|
610
|
+
# ({PathDispatcher.whole_app_rules}); this reaches them by name, for a
|
|
611
|
+
# caller that already knows what went stale: a resident process that just
|
|
612
|
+
# reloaded, a `woods:refresh` invocation after editing `config/routes.rb`,
|
|
613
|
+
# a deploy hook after a dependency bump.
|
|
614
|
+
#
|
|
615
|
+
# "Full extraction required for these types" was always a cold-boot
|
|
616
|
+
# artifact rather than something inherent: in a booted process re-running
|
|
617
|
+
# one extractor is seconds.
|
|
618
|
+
#
|
|
619
|
+
# Any extractor key works, not just the whole-app ones — `refresh(:models)`
|
|
620
|
+
# is a legitimate way to re-derive every model after a schema change.
|
|
621
|
+
#
|
|
622
|
+
# @example After editing config/routes.rb
|
|
623
|
+
# Woods::Extractor.new(output_dir: "tmp/woods").refresh(:routes)
|
|
624
|
+
#
|
|
625
|
+
# @param keys [Array<Symbol>] keys into {EXTRACTORS}
|
|
626
|
+
# @return [Hash] `{ types:, touched:, unknown: }` — the extractors that
|
|
627
|
+
# ran, the identifiers written or removed, and any key that isn't an
|
|
628
|
+
# extractor
|
|
629
|
+
# @raise [ArgumentError] when no recognized key is given
|
|
630
|
+
def refresh(*keys)
|
|
631
|
+
keys = Array(keys).flatten.map(&:to_sym).uniq
|
|
632
|
+
known, unknown = keys.partition { |key| EXTRACTORS.key?(key) }
|
|
633
|
+
raise ArgumentError, "No known extractor in #{keys.inspect}" if known.empty?
|
|
634
|
+
|
|
635
|
+
known += ROUTE_CONSUMER_EXTRACTORS if known.include?(:routes)
|
|
636
|
+
known.uniq!
|
|
637
|
+
|
|
638
|
+
prepare_incremental_run
|
|
315
639
|
affected_types = Set.new
|
|
316
|
-
|
|
317
|
-
|
|
640
|
+
touched = known.each_with_object(Set.new) do |key, acc|
|
|
641
|
+
acc.merge(replace_type_wholesale(key, affected_types))
|
|
318
642
|
end
|
|
319
643
|
|
|
320
|
-
|
|
321
|
-
affected_types.each
|
|
322
|
-
|
|
644
|
+
finalize_incremental_unit_json(affected_types)
|
|
645
|
+
affected_types.each { |type_key| regenerate_type_index(type_key) }
|
|
646
|
+
finalize_incremental_run(touched, reason: "refresh:#{known.sort.join(',')}")
|
|
647
|
+
|
|
648
|
+
{ types: known, touched: touched.to_a, unknown: unknown }
|
|
649
|
+
end
|
|
650
|
+
|
|
651
|
+
# Raise when the most recent extraction run wrote a payload but could not
|
|
652
|
+
# publish its generation marker.
|
|
653
|
+
#
|
|
654
|
+
# The extractor records this failure instead of raising immediately so
|
|
655
|
+
# the resident watch daemon can keep its recoverable posture: it detects
|
|
656
|
+
# the unchanged generation, reports degraded, and carries the paths into
|
|
657
|
+
# a later cycle. One-shot callers have no later cycle, so the rake tasks
|
|
658
|
+
# call this method before reporting success and receive a typed non-zero
|
|
659
|
+
# failure instead of claiming an unreachable payload was published.
|
|
660
|
+
#
|
|
661
|
+
# @return [void]
|
|
662
|
+
# @raise [Woods::ExtractionError] when the generation marker could not be
|
|
663
|
+
# published; the previously published generation remains active
|
|
664
|
+
def raise_on_publication_failure!
|
|
665
|
+
raise @publication_error if @publication_error
|
|
666
|
+
end
|
|
667
|
+
|
|
668
|
+
private
|
|
669
|
+
|
|
670
|
+
# Time one phase of a run and log how long it took, when WOODS_PROFILE=1.
|
|
671
|
+
#
|
|
672
|
+
# The per-extractor lines (see {#extract_all_sequential}) already report
|
|
673
|
+
# extraction itself. Everything after it (the graph load, the analysis,
|
|
674
|
+
# the flows, the publish) was unattributed, so a slow run could only be
|
|
675
|
+
# split by guessing. Off by default and free when off: the block is
|
|
676
|
+
# yielded directly, with no timing and no log line. Timed on the
|
|
677
|
+
# monotonic clock, so a wall-clock adjustment mid-run cannot produce a
|
|
678
|
+
# negative phase.
|
|
679
|
+
#
|
|
680
|
+
# @param name [String] phase name, as it appears in the log line
|
|
681
|
+
# @return [Object] whatever the block returned
|
|
682
|
+
def profile_phase(name)
|
|
683
|
+
return yield unless profiling?
|
|
684
|
+
|
|
685
|
+
start_time = Process.clock_gettime(Process::CLOCK_MONOTONIC)
|
|
686
|
+
result = yield
|
|
687
|
+
elapsed = Process.clock_gettime(Process::CLOCK_MONOTONIC) - start_time
|
|
688
|
+
Rails.logger.info "[Woods] [profile] #{name} in #{elapsed.round(2)}s"
|
|
689
|
+
result
|
|
690
|
+
end
|
|
691
|
+
|
|
692
|
+
# How many reverse hops {#extract_changed} walks from the changed files
|
|
693
|
+
# before it stops re-extracting dependents.
|
|
694
|
+
#
|
|
695
|
+
# Unbounded by default. A unit outside the cap keeps its content, and the
|
|
696
|
+
# one derived field a re-extraction elsewhere can change (`dependents`)
|
|
697
|
+
# is refreshed regardless: {#register_and_write} marks every target of a
|
|
698
|
+
# re-extracted unit's edges, before and after registration, so a unit
|
|
699
|
+
# that gains or loses an inbound edge is rewritten by
|
|
700
|
+
# {#finalize_incremental_unit_json} whether or not the walk reached it.
|
|
701
|
+
#
|
|
702
|
+
# @return [Integer, nil]
|
|
703
|
+
def blast_radius_depth
|
|
704
|
+
Woods.configuration&.incremental_blast_radius_depth
|
|
705
|
+
end
|
|
706
|
+
|
|
707
|
+
# The units a controller's flow document could reach from this run's
|
|
708
|
+
# changed files, read off the pre-change graph.
|
|
709
|
+
#
|
|
710
|
+
# {FlowAssembler} stops expanding at {FlowPrecomputer::DEFAULT_MAX_DEPTH},
|
|
711
|
+
# so a controller further than that from everything the run changed
|
|
712
|
+
# assembles the same document it already has. The bound is the
|
|
713
|
+
# assembler's own constant, not a second literal: raising the assembly
|
|
714
|
+
# depth widens this walk with it.
|
|
715
|
+
#
|
|
716
|
+
# Reverse reachability is the same relation the refresh already rested on
|
|
717
|
+
# (`touched` is itself the graph's reverse closure), so this narrows the
|
|
718
|
+
# distance without changing the edge set behind the decision.
|
|
719
|
+
#
|
|
720
|
+
# @param change_set [Woods::ChangeSet]
|
|
721
|
+
# @return [Set<String>]
|
|
722
|
+
def flow_scope_for(change_set)
|
|
723
|
+
return Set.new unless Woods.configuration.precompute_flows
|
|
724
|
+
|
|
725
|
+
@dependency_graph.affected_by(
|
|
726
|
+
change_set.absolute_paths, max_depth: FlowPrecomputer::DEFAULT_MAX_DEPTH
|
|
727
|
+
).to_set
|
|
728
|
+
end
|
|
729
|
+
|
|
730
|
+
# @return [Boolean] whether phase timing is enabled for this process
|
|
731
|
+
def profiling?
|
|
732
|
+
ENV.fetch('WOODS_PROFILE', nil) == '1'
|
|
733
|
+
end
|
|
734
|
+
|
|
735
|
+
# Load the persisted graph and reset the per-run bookkeeping that the
|
|
736
|
+
# incremental helpers read. Shared by {#extract_changed} and {#refresh};
|
|
737
|
+
# calling either without this leaves `@dependents_dirty` and
|
|
738
|
+
# `@incremental_written` holding a previous run's state.
|
|
739
|
+
#
|
|
740
|
+
# `begin_payload!(strict: true)`: an incremental write set is only the
|
|
741
|
+
# touched units, so a degrade to flat here (see {#begin_payload!}) would
|
|
742
|
+
# both read the wrong baseline graph below and publish an index missing
|
|
743
|
+
# every unit it didn't touch. Raises rather than degrading; the caller
|
|
744
|
+
# sees {Woods::ExtractionError} and the generation is left unbumped.
|
|
745
|
+
#
|
|
746
|
+
# @return [void]
|
|
747
|
+
# @raise [Woods::ExtractionError] see {#begin_payload!}
|
|
748
|
+
def prepare_incremental_run
|
|
749
|
+
profile_phase('payload seed') { begin_payload!(strict: true) }
|
|
750
|
+
graph_path = payload_dir.join('dependency_graph.json')
|
|
751
|
+
ensure_incremental_baseline!(graph_path)
|
|
752
|
+
profile_phase('previous graph load') do
|
|
753
|
+
@dependency_graph = DependencyGraph.from_h(JSON.parse(AtomicFile.read(graph_path))) if graph_path.exist?
|
|
323
754
|
end
|
|
324
755
|
|
|
325
|
-
|
|
326
|
-
|
|
327
|
-
|
|
328
|
-
|
|
329
|
-
|
|
756
|
+
ModelNameCache.reset!
|
|
757
|
+
profile_phase('eager load') { safe_eager_load! }
|
|
758
|
+
|
|
759
|
+
@dependents_dirty = Set.new
|
|
760
|
+
@incremental_written = {}
|
|
761
|
+
# nil = no scope was computed for this run, so every re-extracted
|
|
762
|
+
# controller has its flows reassembled. {#refresh} leaves it that way.
|
|
763
|
+
@flow_scope = nil
|
|
764
|
+
@previous_flow_index_entries = nil
|
|
765
|
+
@incremental_extractors = nil
|
|
766
|
+
@active_record_names = nil
|
|
767
|
+
@package_resolver = nil
|
|
768
|
+
@persisted_index_stats = nil
|
|
769
|
+
@graph_sha = nil
|
|
770
|
+
end
|
|
771
|
+
|
|
772
|
+
# Write the graph and the derived artifacts after an incremental run.
|
|
773
|
+
#
|
|
774
|
+
# The manifest is rewritten only when the run actually changed something.
|
|
775
|
+
# `staleness_seconds` is derived from the manifest timestamp, so touching
|
|
776
|
+
# it after a no-op run reports the index as freshly synced when nothing
|
|
777
|
+
# was re-read — the misleading half of #164 gap 4. The graph write itself
|
|
778
|
+
# is unconditional, but on a no-op run it is inert, not meaningful: under
|
|
779
|
+
# payloads it lands in this run's not-yet-published directory, and a
|
|
780
|
+
# no-op run returns below before {#publish_generation}, so nothing ever
|
|
781
|
+
# points at what it just wrote — the next run's {PayloadStore#create}
|
|
782
|
+
# empties that same (unbumped-generation-numbered) directory before
|
|
783
|
+
# writing into it again. The graph's content is unchanged on a no-op run
|
|
784
|
+
# regardless (no touched units, no new edges), so nothing is lost; it is
|
|
785
|
+
# the write itself, not the recomputation, that goes nowhere.
|
|
786
|
+
#
|
|
787
|
+
# No `capture_snapshot` here: snapshots must hash the FULL unit set, and
|
|
788
|
+
# incremental runs only hold changed units in memory — capturing would
|
|
789
|
+
# record a snapshot whose diff reports every other unit as deleted.
|
|
790
|
+
#
|
|
791
|
+
# @param touched [Set<String>] identifiers added, re-extracted, or removed
|
|
792
|
+
# @param reason [String] what produced this run, recorded on the generation
|
|
793
|
+
# so `woods_status` can distinguish a targeted refresh from a file-driven
|
|
794
|
+
# incremental run — they have different blast radii and an operator
|
|
795
|
+
# reading "incremental" after a `woods:refresh[routes]` is being misled
|
|
796
|
+
# @return [void]
|
|
797
|
+
def finalize_incremental_run(touched, reason: 'incremental')
|
|
330
798
|
write_dependency_graph
|
|
331
|
-
|
|
332
|
-
|
|
333
|
-
|
|
334
|
-
|
|
799
|
+
|
|
800
|
+
if touched.empty?
|
|
801
|
+
Rails.logger.info '[Woods] Incremental run changed nothing — leaving manifest timestamp untouched'
|
|
802
|
+
return
|
|
335
803
|
end
|
|
336
804
|
|
|
337
|
-
|
|
805
|
+
profile_phase('graph analysis') { write_incremental_graph_analysis }
|
|
806
|
+
profile_phase('flows') { refresh_incremental_flows(touched) }
|
|
807
|
+
profile_phase('manifest and summary') do
|
|
808
|
+
write_manifest(incremental: true)
|
|
809
|
+
write_structural_summary
|
|
810
|
+
end
|
|
811
|
+
profile_phase('publish') { publish_generation(reason) }
|
|
812
|
+
|
|
813
|
+
return unless Woods.configuration.enable_snapshots
|
|
814
|
+
|
|
815
|
+
Rails.logger.info '[Woods] Skipping snapshot capture — snapshots are captured on full extraction only'
|
|
338
816
|
end
|
|
339
817
|
|
|
340
|
-
|
|
818
|
+
# Publish a new generation — the last write of any successful run.
|
|
819
|
+
#
|
|
820
|
+
# Every extraction mode does this, so a long-lived reader can detect that
|
|
821
|
+
# the index moved without stat-ing the whole directory, whatever produced
|
|
822
|
+
# the change: a full run, an incremental run, a targeted refresh, or the
|
|
823
|
+
# watch daemon. Ordering is the contract: the generation goes last, so a
|
|
824
|
+
# reader that sees generation N knows N's files are already on disk. A run
|
|
825
|
+
# that raised, or that changed nothing, never reaches this.
|
|
826
|
+
#
|
|
827
|
+
# @param reason [String] what produced this generation
|
|
828
|
+
# @return [void]
|
|
829
|
+
def publish_generation(reason)
|
|
830
|
+
generation = Generation.new(output_dir: @output_dir)
|
|
831
|
+
# Resolve (and if necessary rename) the payload first, so the flush
|
|
832
|
+
# below covers the directory under the name the pointer will carry.
|
|
833
|
+
payload = publishable_payload_name(generation)
|
|
834
|
+
profile_phase('payload sync') { sync_payload }
|
|
835
|
+
marker = generation.bump!(reason: reason, payload: payload)
|
|
836
|
+
prune_payloads(marker.number)
|
|
837
|
+
marker
|
|
838
|
+
rescue StandardError => e
|
|
839
|
+
# A failed bump must not fail the extraction that produced a perfectly
|
|
840
|
+
# good index. But "readers keep their current view until the next run" is
|
|
841
|
+
# too comfortable a way to put it: the generation *is* the freshness
|
|
842
|
+
# contract, so readers keep serving the old index for as long as the
|
|
843
|
+
# cause persists, and the next incremental may be a no-op that bumps
|
|
844
|
+
# nothing either. Error, not warn — and `Watch::Daemon` cross-checks that
|
|
845
|
+
# the number actually moved so the daemon reports degraded rather than
|
|
846
|
+
# running.
|
|
847
|
+
@publication_error = Woods::ExtractionError.new(
|
|
848
|
+
"Could not publish generation for #{@output_dir} (#{e.class}: #{e.message}); " \
|
|
849
|
+
'the previous generation remains active'
|
|
850
|
+
)
|
|
851
|
+
Rails.logger.error "[Woods] #{@publication_error.message}"
|
|
852
|
+
nil
|
|
853
|
+
end
|
|
854
|
+
|
|
855
|
+
# Open the payload directory this run publishes into, seeded from the
|
|
856
|
+
# generation currently on disk.
|
|
857
|
+
#
|
|
858
|
+
# Seeding is what lets a run that touches ten files publish a whole
|
|
859
|
+
# generation: the unchanged artifacts are hardlinked in, and the run
|
|
860
|
+
# overwrites only what it changed. It also means every payload read during
|
|
861
|
+
# the run — the persisted graph, the per-type indexes, the unit JSON an
|
|
862
|
+
# incremental run patches — sees the previous generation, exactly as it
|
|
863
|
+
# did when the index was flat.
|
|
864
|
+
#
|
|
865
|
+
# A failure here degrades to the flat layout rather than failing the run
|
|
866
|
+
# — but only when a flat publish can actually be complete. A full
|
|
867
|
+
# extraction's write set is every unit in the app, so a flat publish from
|
|
868
|
+
# it is a whole index; the next successful run restores the payload
|
|
869
|
+
# boundary. `strict:` opts out of that degrade for {#extract_changed} and
|
|
870
|
+
# {#refresh}, whose write set is only the units they touched: over a
|
|
871
|
+
# payload-born index (`marker.payload` set), the flat root holds nothing
|
|
872
|
+
# newer than the last time this index was flat — possibly nothing at all
|
|
873
|
+
# — so publishing there would both compute the wrong incremental baseline
|
|
874
|
+
# ({#prepare_incremental_run} reads its graph from wherever `payload_dir`
|
|
875
|
+
# resolves to) and, even with the right baseline, redirect every reader
|
|
876
|
+
# to a near-empty directory missing every untouched unit. There is no
|
|
877
|
+
# complete flat index an incremental degrade can produce, so this raises
|
|
878
|
+
# instead — see {Woods::ExtractionError}. The generation is never bumped
|
|
879
|
+
# over a raised run, so readers keep serving the last good index.
|
|
880
|
+
#
|
|
881
|
+
# @param strict [Boolean] raise instead of degrading to a flat publish
|
|
882
|
+
# when the published generation names a payload
|
|
883
|
+
# @return [void]
|
|
884
|
+
# @raise [Woods::ExtractionError] when `strict` and a payload-born index's
|
|
885
|
+
# payload directory could not be opened for this run
|
|
886
|
+
def begin_payload!(strict: false)
|
|
887
|
+
@publication_error = nil
|
|
888
|
+
marker = Generation.new(output_dir: @output_dir).current
|
|
889
|
+
@payload_generation = marker.number + 1
|
|
890
|
+
@payload_dir = @payload_store.create(@payload_generation)
|
|
891
|
+
seed_payload(marker)
|
|
892
|
+
rescue StandardError => e
|
|
893
|
+
if strict && marker&.payload
|
|
894
|
+
raise Woods::ExtractionError, <<~MSG.tr("\n", ' ').strip
|
|
895
|
+
Could not open a payload directory for this incremental run
|
|
896
|
+
(#{e.class}: #{e.message}), and generation #{marker.number}'s payload
|
|
897
|
+
lives in #{marker.payload} rather than the flat output root. An
|
|
898
|
+
incremental run only writes the units it touched, so publishing flat
|
|
899
|
+
here would produce an index missing every untouched unit. Fix the
|
|
900
|
+
underlying filesystem issue (permissions, free space, a broken
|
|
901
|
+
mount), or run a full `woods:extract` to rebuild a complete flat
|
|
902
|
+
index before incremental runs resume.
|
|
903
|
+
MSG
|
|
904
|
+
end
|
|
905
|
+
|
|
906
|
+
Rails.logger.warn(
|
|
907
|
+
"[Woods] Could not open a payload directory (#{e.class}: #{e.message}) — publishing flat"
|
|
908
|
+
)
|
|
909
|
+
@payload_dir = nil
|
|
910
|
+
@payload_generation = nil
|
|
911
|
+
end
|
|
912
|
+
|
|
913
|
+
# Refuse an incremental run that has no baseline to be incremental
|
|
914
|
+
# against (CORE-2). With no published generation and no dependency
|
|
915
|
+
# graph — a failed CI cache restore, a typo'd WOODS_OUTPUT, a first run
|
|
916
|
+
# on a fresh runner — {#extract_changed} would compute an empty blast
|
|
917
|
+
# radius, dispatch only the diffed paths, and publish generation 1: a
|
|
918
|
+
# near-empty index that readers, `woods:validate`, and retrieval all
|
|
919
|
+
# treat as the complete truth of the app, and that nothing self-heals
|
|
920
|
+
# until a full extraction. The watch daemon already enforces this
|
|
921
|
+
# invariant on its side (a missing generation marker means one full
|
|
922
|
+
# extraction); the one-shot entry points must refuse rather than
|
|
923
|
+
# publish silently. A pre-generation flat index passes: its graph was
|
|
924
|
+
# seeded into this run's payload directory by {#begin_payload!}.
|
|
925
|
+
#
|
|
926
|
+
# @param graph_path [Pathname] the seeded payload's dependency graph
|
|
927
|
+
# @return [void]
|
|
928
|
+
# @raise [Woods::ExtractionError] when no baseline exists
|
|
929
|
+
def ensure_incremental_baseline!(graph_path)
|
|
930
|
+
return if graph_path.exist?
|
|
931
|
+
return if Generation.new(output_dir: @output_dir).current.number.positive?
|
|
932
|
+
# An embedding caller that already holds a populated graph (a prior
|
|
933
|
+
# in-process run, or seeded registrations) IS the baseline —
|
|
934
|
+
# {#prepare_incremental_run} keeps the in-memory graph whenever no
|
|
935
|
+
# disk graph exists.
|
|
936
|
+
return unless @dependency_graph.nil? || @dependency_graph.empty?
|
|
937
|
+
|
|
938
|
+
raise Woods::ExtractionError, <<~MSG.tr("\n", ' ').strip
|
|
939
|
+
No baseline index found under #{@output_dir}: no generation has been
|
|
940
|
+
published and no dependency_graph.json exists. An incremental run
|
|
941
|
+
only re-extracts what changed relative to an existing index, so
|
|
942
|
+
running it here would publish a near-empty index as the complete
|
|
943
|
+
truth of the application. Run a full `woods:extract` first (or point
|
|
944
|
+
WOODS_OUTPUT at the directory holding the existing index).
|
|
945
|
+
MSG
|
|
946
|
+
end
|
|
947
|
+
|
|
948
|
+
# Copy the published generation's payload into this run's directory.
|
|
949
|
+
#
|
|
950
|
+
# Two sources. A generation that already names a payload directory is
|
|
951
|
+
# cloned wholesale — by construction it holds payload artifacts and
|
|
952
|
+
# nothing else. A flat index (every index written before payloads, and any
|
|
953
|
+
# index whose last run degraded) is cloned entry by entry from an
|
|
954
|
+
# allowlist, because the output root also holds `generation.json`,
|
|
955
|
+
# `dumps/`, `tasks/`, `woods.sqlite3` and `payloads/` itself.
|
|
956
|
+
#
|
|
957
|
+
# @param marker [Woods::Generation::Marker] the published generation
|
|
958
|
+
# @return [void]
|
|
959
|
+
def seed_payload(marker)
|
|
960
|
+
if marker.payload && (source = @output_dir.join(marker.payload)).directory?
|
|
961
|
+
@payload_store.clone(source, @payload_dir)
|
|
962
|
+
else
|
|
963
|
+
seed_payload_from_flat_root
|
|
964
|
+
end
|
|
965
|
+
end
|
|
966
|
+
|
|
967
|
+
# Uses {PayloadStore#link_or_copy} rather than a bare +FileUtils.ln+ for
|
|
968
|
+
# the same reason {PayloadStore#clone} does: a filesystem that disallows
|
|
969
|
+
# hardlinks (EXDEV/EPERM/EMLINK/NotImplementedError) must still seed the
|
|
970
|
+
# payload, just by copying. Before this, that raise was caught by
|
|
971
|
+
# {#begin_payload!}'s rescue and degraded every run on such a filesystem
|
|
972
|
+
# to a flat publish — the fallback existed in {PayloadStore} but this,
|
|
973
|
+
# the only other file-level linker, never reached it.
|
|
974
|
+
#
|
|
975
|
+
# @return [void]
|
|
976
|
+
def seed_payload_from_flat_root
|
|
977
|
+
(PAYLOAD_FILES + payload_entry_dirs).each do |entry|
|
|
978
|
+
source = @output_dir.join(entry)
|
|
979
|
+
next unless source.exist?
|
|
980
|
+
|
|
981
|
+
@payload_store.clone(source, @payload_dir.join(entry)) if source.directory?
|
|
982
|
+
@payload_store.link_or_copy(source, @payload_dir.join(entry)) if source.file?
|
|
983
|
+
end
|
|
984
|
+
end
|
|
985
|
+
|
|
986
|
+
# @return [Array<String>] every directory name a payload can contain
|
|
987
|
+
def payload_entry_dirs
|
|
988
|
+
EXTRACTORS.keys.map(&:to_s) + PAYLOAD_DIRS
|
|
989
|
+
end
|
|
990
|
+
|
|
991
|
+
# Whether payload files pay for their own durability as they are written.
|
|
992
|
+
#
|
|
993
|
+
# False by default and false on every host that has not asked otherwise:
|
|
994
|
+
# {#sync_payload} makes the whole payload durable in one flush before the
|
|
995
|
+
# pointer names it, so a per-file fsync buys only the window in which
|
|
996
|
+
# nothing can read the file anyway. `durable_payload_writes = true`
|
|
997
|
+
# restores the per-file cost for a host that wants it; it does not, and
|
|
998
|
+
# cannot, remove the publish flush.
|
|
999
|
+
#
|
|
1000
|
+
# @return [Boolean]
|
|
1001
|
+
def payload_writes_durable?
|
|
1002
|
+
Woods.configuration&.durable_payload_writes ? true : false
|
|
1003
|
+
end
|
|
1004
|
+
|
|
1005
|
+
# One filesystem flush that makes this run's whole payload durable,
|
|
1006
|
+
# immediately before the pointer that names it is written durably.
|
|
1007
|
+
#
|
|
1008
|
+
# This is the guarantee that replaces the per-file fsync {#write_results}
|
|
1009
|
+
# used to pay for every unit: **when `generation.json` is durable, every
|
|
1010
|
+
# file in the payload it names is durable.** What is given up is an
|
|
1011
|
+
# individual payload file being durable before the pointer exists, and
|
|
1012
|
+
# nobody reads a payload file in that window: every reader resolves
|
|
1013
|
+
# through the pointer, and a crash leaves an unreferenced partial payload
|
|
1014
|
+
# that the next run prunes.
|
|
1015
|
+
#
|
|
1016
|
+
# Unconditional on purpose. `durable_payload_writes` decides whether the
|
|
1017
|
+
# per-file fsyncs are *also* paid; it can never turn this off, because
|
|
1018
|
+
# that would be the one change that actually weakens the contract.
|
|
1019
|
+
# {Woods::AtomicFile.sync_directory_tree} degrades through `sync -f`, a
|
|
1020
|
+
# bare `sync`, and finally a per-file fsync pass, so the flush cannot
|
|
1021
|
+
# silently become a no-op either.
|
|
1022
|
+
#
|
|
1023
|
+
# A run that degraded to a flat publish has no payload directory;
|
|
1024
|
+
# {#payload_dir} then returns the output directory, which is the tree that
|
|
1025
|
+
# needs flushing in exactly the same way.
|
|
1026
|
+
#
|
|
1027
|
+
# @return [Symbol, nil] the strategy {Woods::AtomicFile} used
|
|
1028
|
+
def sync_payload
|
|
1029
|
+
AtomicFile.sync_directory_tree(payload_dir)
|
|
1030
|
+
end
|
|
1031
|
+
|
|
1032
|
+
# The pointer to publish, or nil when this run built no payload directory.
|
|
1033
|
+
#
|
|
1034
|
+
# The directory was named from the generation number this run expected to
|
|
1035
|
+
# publish as. Writers serialize on `PipelineLock`, so that prediction holds
|
|
1036
|
+
# — but if it ever did not, the pointer would name a directory belonging to
|
|
1037
|
+
# a different generation, so the directory is renamed to match rather than
|
|
1038
|
+
# published under a name that lies.
|
|
1039
|
+
#
|
|
1040
|
+
# @param generation [Woods::Generation]
|
|
1041
|
+
# @return [String, nil]
|
|
1042
|
+
def publishable_payload_name(generation)
|
|
1043
|
+
return nil unless @payload_dir
|
|
1044
|
+
|
|
1045
|
+
actual = generation.current.number + 1
|
|
1046
|
+
if actual != @payload_generation
|
|
1047
|
+
@payload_dir = rename_payload(actual)
|
|
1048
|
+
@payload_generation = actual
|
|
1049
|
+
end
|
|
1050
|
+
|
|
1051
|
+
PayloadStore.name_for(@payload_generation)
|
|
1052
|
+
end
|
|
1053
|
+
|
|
1054
|
+
# @param number [Integer] the generation number actually being published
|
|
1055
|
+
# @return [Pathname] the payload directory under its corrected name
|
|
1056
|
+
def rename_payload(number)
|
|
1057
|
+
destination = @payload_store.path_for(number)
|
|
1058
|
+
FileUtils.rm_rf(destination.to_s)
|
|
1059
|
+
FileUtils.mv(@payload_dir.to_s, destination.to_s)
|
|
1060
|
+
destination
|
|
1061
|
+
end
|
|
1062
|
+
|
|
1063
|
+
# @param published [Integer] the generation just published
|
|
1064
|
+
# @return [void]
|
|
1065
|
+
def prune_payloads(published)
|
|
1066
|
+
return unless @payload_dir
|
|
1067
|
+
|
|
1068
|
+
@payload_store.prune(keep: payload_retention, protect: published)
|
|
1069
|
+
rescue StandardError => e
|
|
1070
|
+
Rails.logger.warn "[Woods] Payload retention failed: #{e.message}"
|
|
1071
|
+
end
|
|
1072
|
+
|
|
1073
|
+
# @return [Integer] how many payload generations to keep on disk
|
|
1074
|
+
def payload_retention
|
|
1075
|
+
value = ENV.fetch('WOODS_PAYLOAD_RETENTION', nil).to_i
|
|
1076
|
+
value.positive? ? value : PayloadStore::DEFAULT_RETENTION
|
|
1077
|
+
end
|
|
1078
|
+
|
|
1079
|
+
# Recompute graph_analysis.json after an incremental graph write.
|
|
1080
|
+
#
|
|
1081
|
+
# Structural analysis (orphans, hubs, cycles, bridges) is derived purely
|
|
1082
|
+
# from the graph, so leaving it at the last full extraction's values let
|
|
1083
|
+
# it drift continuously between full runs (#164 gap 5).
|
|
1084
|
+
#
|
|
1085
|
+
# Re-raises rather than warn-and-continue: a caller that swallowed this
|
|
1086
|
+
# went on to write the manifest and publish a fresh generation over an
|
|
1087
|
+
# index whose graph_analysis.json never updated — a failed analysis write
|
|
1088
|
+
# must abort the run the same way a failed extraction does, not ship
|
|
1089
|
+
# under a generation that says everything landed.
|
|
1090
|
+
#
|
|
1091
|
+
# @return [void]
|
|
1092
|
+
# @raise [StandardError] whatever GraphAnalyzer or the write raised
|
|
1093
|
+
def write_incremental_graph_analysis
|
|
1094
|
+
@graph_analysis = build_graph_analyzer.analyze
|
|
1095
|
+
write_graph_analysis
|
|
1096
|
+
rescue StandardError => e
|
|
1097
|
+
Rails.logger.error "[Woods] Incremental graph analysis failed: #{e.message}"
|
|
1098
|
+
raise
|
|
1099
|
+
end
|
|
341
1100
|
|
|
342
1101
|
# ──────────────────────────────────────────────────────────────────────
|
|
343
1102
|
# Eager Loading
|
|
@@ -351,9 +1110,16 @@ module Woods
|
|
|
351
1110
|
# loads only the directories we actually need for extraction.
|
|
352
1111
|
def safe_eager_load!
|
|
353
1112
|
Rails.application.eager_load!
|
|
1113
|
+
# Recorded because it decides whether the runtime discovery sets are
|
|
1114
|
+
# *complete*. On the fallback path they are known-partial — whole
|
|
1115
|
+
# directories may have failed to load — and treating "absent from
|
|
1116
|
+
# descendants" as "deleted" would then erase live units by the type.
|
|
1117
|
+
# See {#stale_class_based_units}.
|
|
1118
|
+
@eager_load_complete = true
|
|
354
1119
|
rescue NameError => e
|
|
355
1120
|
Rails.logger.warn "[Woods] eager_load! hit NameError: #{e.message}"
|
|
356
1121
|
Rails.logger.warn '[Woods] Falling back to per-directory eager loading'
|
|
1122
|
+
@eager_load_complete = false
|
|
357
1123
|
eager_load_extraction_directories
|
|
358
1124
|
end
|
|
359
1125
|
|
|
@@ -387,8 +1153,36 @@ module Woods
|
|
|
387
1153
|
# Extraction Strategies
|
|
388
1154
|
# ──────────────────────────────────────────────────────────────────────
|
|
389
1155
|
|
|
1156
|
+
# Should this extractor be skipped under the current configuration?
|
|
1157
|
+
#
|
|
1158
|
+
# `include_framework_sources` (default: true — see
|
|
1159
|
+
# docs/CONFIGURATION_REFERENCE.md) was documented and consumed by nothing
|
|
1160
|
+
# (#169). It now gates both automatic paths at once: the full-run pass
|
|
1161
|
+
# over {EXTRACTORS} and the `Gemfile.lock` whole-app trigger
|
|
1162
|
+
# ({#rerun_whole_app_extractors}). The {PathDispatcher} rule itself stays
|
|
1163
|
+
# ungated — rules are memoized per-process while configuration is not,
|
|
1164
|
+
# and `relevant?` stays correct either way because `Gemfile.lock` already
|
|
1165
|
+
# triggers :engines and :middleware — so the gate sits where the
|
|
1166
|
+
# extractor would actually run. Gating only one side would either extract
|
|
1167
|
+
# units a knob-off full run does not produce (a phantom `touched` cycle
|
|
1168
|
+
# bumping the generation) or leave a knob-on host with framework sources
|
|
1169
|
+
# that never refresh on a dependency bump.
|
|
1170
|
+
#
|
|
1171
|
+
# A by-name {#refresh} (and its `woods:extract_framework` alias) is
|
|
1172
|
+
# deliberately NOT gated: the knob controls automatic participation, and
|
|
1173
|
+
# the explicit refresh is the escape hatch for hosts that want framework
|
|
1174
|
+
# sources on demand without paying for them every run.
|
|
1175
|
+
#
|
|
1176
|
+
# @param key [Symbol] key into {EXTRACTORS}
|
|
1177
|
+
# @return [Boolean]
|
|
1178
|
+
def skip_by_configuration?(key)
|
|
1179
|
+
key == :rails_source && !Woods.configuration.include_framework_sources
|
|
1180
|
+
end
|
|
1181
|
+
|
|
390
1182
|
def extract_all_sequential
|
|
391
1183
|
EXTRACTORS.each do |type, extractor_class|
|
|
1184
|
+
next if skip_by_configuration?(type)
|
|
1185
|
+
|
|
392
1186
|
Rails.logger.info "[Woods] Extracting #{type}..."
|
|
393
1187
|
start_time = Time.current
|
|
394
1188
|
|
|
@@ -427,7 +1221,9 @@ module Woods
|
|
|
427
1221
|
ModelNameCache.short_names_regex if ModelNameCache.respond_to?(:short_names_regex)
|
|
428
1222
|
|
|
429
1223
|
results_mutex = Mutex.new
|
|
430
|
-
|
|
1224
|
+
failures = []
|
|
1225
|
+
active = EXTRACTORS.reject { |type, _| skip_by_configuration?(type) }
|
|
1226
|
+
threads = active.map do |type, extractor_class|
|
|
431
1227
|
Thread.new do
|
|
432
1228
|
Rails.logger.info "[Woods] [Thread] Extracting #{type}..."
|
|
433
1229
|
start_time = Time.current
|
|
@@ -445,12 +1241,21 @@ module Woods
|
|
|
445
1241
|
end
|
|
446
1242
|
rescue StandardError => e
|
|
447
1243
|
Rails.logger.error "[Woods] [Thread] #{type} failed: #{e.message}"
|
|
448
|
-
results_mutex.synchronize {
|
|
1244
|
+
results_mutex.synchronize { failures << [type, e] }
|
|
449
1245
|
end
|
|
450
1246
|
end
|
|
451
1247
|
|
|
452
1248
|
threads.each(&:join)
|
|
453
1249
|
|
|
1250
|
+
# Fail closed, matching sequential extraction: a raise there aborts
|
|
1251
|
+
# before registering anything for the run. Silently substituting `[]`
|
|
1252
|
+
# for a failed type let the run finish, register a partial graph, and
|
|
1253
|
+
# publish it under a generation that says every type succeeded.
|
|
1254
|
+
if failures.any?
|
|
1255
|
+
names = failures.map { |(type, _e)| type }.sort.join(', ')
|
|
1256
|
+
raise Woods::ExtractionError, "Concurrent extraction failed for: #{names}"
|
|
1257
|
+
end
|
|
1258
|
+
|
|
454
1259
|
# Register into dependency graph sequentially — DependencyGraph is not thread-safe
|
|
455
1260
|
EXTRACTORS.each_key do |type|
|
|
456
1261
|
(@results[type] || []).each { |unit| @dependency_graph.register(unit) }
|
|
@@ -464,7 +1269,7 @@ module Woods
|
|
|
464
1269
|
def setup_output_directory
|
|
465
1270
|
FileUtils.mkdir_p(@output_dir)
|
|
466
1271
|
EXTRACTORS.each_key do |type|
|
|
467
|
-
FileUtils.mkdir_p(
|
|
1272
|
+
FileUtils.mkdir_p(payload_dir.join(type.to_s))
|
|
468
1273
|
end
|
|
469
1274
|
end
|
|
470
1275
|
|
|
@@ -472,56 +1277,122 @@ module Woods
|
|
|
472
1277
|
# Dependency Resolution
|
|
473
1278
|
# ──────────────────────────────────────────────────────────────────────
|
|
474
1279
|
|
|
1280
|
+
# `unit_map` is identifier => Array<unit>, not identifier => unit (#225).
|
|
1281
|
+
# An identifier is not unique across types (a Scenic view and a factory
|
|
1282
|
+
# can both be `reports`), so a single-valued map let the later
|
|
1283
|
+
# registration overwrite the earlier one and every dependent land on
|
|
1284
|
+
# whichever unit happened to be indexed last — the other unit serialized
|
|
1285
|
+
# `dependents: []` forever. {#rewrite_unit_json_of_type} already treats
|
|
1286
|
+
# `dependents` as a property of the identifier and writes the same list
|
|
1287
|
+
# to every type that identifier owns; this brings the full-extraction
|
|
1288
|
+
# path into agreement with it.
|
|
475
1289
|
def resolve_dependents
|
|
476
1290
|
# Build complete unit map first (cross-type dependencies require all units indexed).
|
|
477
|
-
unit_map = @results.each_with_object({}) do |(_type, units), map|
|
|
478
|
-
units.each { |u| map[u.identifier]
|
|
1291
|
+
unit_map = @results.each_with_object(Hash.new { |h, k| h[k] = [] }) do |(_type, units), map|
|
|
1292
|
+
units.each { |u| map[u.identifier] << u }
|
|
479
1293
|
end
|
|
480
1294
|
|
|
481
1295
|
# Resolve dependents using the complete map.
|
|
482
1296
|
@results.each_value do |units|
|
|
483
1297
|
units.each do |unit|
|
|
484
1298
|
unit.dependencies.each do |dep|
|
|
485
|
-
|
|
486
|
-
|
|
487
|
-
|
|
488
|
-
|
|
489
|
-
|
|
490
|
-
|
|
491
|
-
|
|
492
|
-
}
|
|
1299
|
+
unit_map[dep[:target]].each do |target_unit|
|
|
1300
|
+
target_unit.dependents ||= []
|
|
1301
|
+
target_unit.dependents << {
|
|
1302
|
+
type: unit.type,
|
|
1303
|
+
identifier: unit.identifier
|
|
1304
|
+
}
|
|
1305
|
+
end
|
|
493
1306
|
end
|
|
494
1307
|
end
|
|
495
1308
|
end
|
|
496
1309
|
end
|
|
497
1310
|
|
|
498
1311
|
# Remove duplicate units (same identifier) within each type, keeping the first occurrence.
|
|
499
|
-
# Duplicates arise when
|
|
500
|
-
# routes duplicating app routes
|
|
501
|
-
#
|
|
1312
|
+
# Duplicates arise when the same identifier is legitimately re-derived from the same
|
|
1313
|
+
# source (e.g., engine-mounted routes duplicating app routes: both carry no file path).
|
|
1314
|
+
# Without dedup, downstream phases would produce inflated counts, duplicate
|
|
1315
|
+
# _index.json entries, and last-writer-wins file overwrites.
|
|
1316
|
+
#
|
|
1317
|
+
# A duplicate derived from a DIFFERENT file is not a duplicate — it is two distinct
|
|
1318
|
+
# real-world constants collapsing onto one identifier (finding G-1: wrapper-named
|
|
1319
|
+
# sources all indexed as the wrapper class). Only one could ever be indexed; the
|
|
1320
|
+
# other silently vanished. That is unrepresentable — `variants:` carries cross-type
|
|
1321
|
+
# graph identity only — so extraction aborts with both files named.
|
|
502
1322
|
def deduplicate_results
|
|
503
1323
|
@results.each do |type, units|
|
|
504
|
-
|
|
505
|
-
|
|
1324
|
+
@results[type] = deduplicate_type_units(type, units)
|
|
1325
|
+
end
|
|
1326
|
+
end
|
|
1327
|
+
|
|
1328
|
+
# First-occurrence dedup for one type, fail-closed on cross-file collisions.
|
|
1329
|
+
#
|
|
1330
|
+
# A retained identifier remembers its source file. A later unit with the
|
|
1331
|
+
# same identifier and the same file (or the same absence of one — runtime
|
|
1332
|
+
# and engine units carry nil) is the legitimate re-derivation {#deduplicate_results}
|
|
1333
|
+
# documents and is dropped. A later unit from a different file raises
|
|
1334
|
+
# {Woods::ExtractionError} listing both paths.
|
|
1335
|
+
#
|
|
1336
|
+
# @param type [Symbol] Result type, for the log line and error message
|
|
1337
|
+
# @param units [Array<ExtractedUnit>]
|
|
1338
|
+
# @return [Array<ExtractedUnit>] Deduplicated units, first occurrence order
|
|
1339
|
+
# @raise [Woods::ExtractionError] when one type+identifier is derived
|
|
1340
|
+
# from two different files
|
|
1341
|
+
def deduplicate_type_units(type, units)
|
|
1342
|
+
deduped = []
|
|
1343
|
+
dropped = 0
|
|
1344
|
+
retained_paths = {}
|
|
1345
|
+
|
|
1346
|
+
units.each do |unit|
|
|
1347
|
+
if retained_paths.key?(unit.identifier)
|
|
1348
|
+
prior_path = retained_paths[unit.identifier]
|
|
1349
|
+
if unit.file_path != prior_path
|
|
1350
|
+
raise Woods::ExtractionError, same_type_collision_message(type, unit, prior_path)
|
|
1351
|
+
end
|
|
506
1352
|
|
|
507
|
-
|
|
1353
|
+
dropped += 1
|
|
1354
|
+
next
|
|
1355
|
+
end
|
|
508
1356
|
|
|
509
|
-
|
|
1357
|
+
retained_paths[unit.identifier] = unit.file_path
|
|
1358
|
+
deduped << unit
|
|
510
1359
|
end
|
|
1360
|
+
|
|
1361
|
+
Rails.logger.warn "[Woods] Deduplicated #{type}: dropped #{dropped} duplicate(s)" if dropped.positive?
|
|
1362
|
+
|
|
1363
|
+
deduped
|
|
1364
|
+
end
|
|
1365
|
+
|
|
1366
|
+
# @param type [Symbol]
|
|
1367
|
+
# @param unit [ExtractedUnit] The colliding (later) unit
|
|
1368
|
+
# @param prior_path [String, nil] File path of the retained unit
|
|
1369
|
+
# @return [String]
|
|
1370
|
+
def same_type_collision_message(type, unit, prior_path)
|
|
1371
|
+
"same-type identifier collision: #{type.to_s.singularize} '#{unit.identifier}' derived from " \
|
|
1372
|
+
"two different sources ('#{prior_path || 'no file'}' and '#{unit.file_path || 'no file'}'); " \
|
|
1373
|
+
'only one unit could ever be indexed, so extraction aborted — either merge the ' \
|
|
1374
|
+
'declarations into one file or split them into distinct constants'
|
|
511
1375
|
end
|
|
512
1376
|
|
|
513
1377
|
# ──────────────────────────────────────────────────────────────────────
|
|
514
1378
|
# Flow Precomputation
|
|
515
1379
|
# ──────────────────────────────────────────────────────────────────────
|
|
516
1380
|
|
|
1381
|
+
# Full-path counterpart of {#refresh_incremental_flows}: assemble every
|
|
1382
|
+
# controller's flows, annotate the in-memory units, rewrite them, and
|
|
1383
|
+
# sweep orphans. Failure-closed — any error raises to {#extract_all},
|
|
1384
|
+
# which aborts before write_manifest/publish_generation, so a partial
|
|
1385
|
+
# flow index or half-rewritten annotations can never be published.
|
|
1386
|
+
#
|
|
1387
|
+
# @return [void]
|
|
1388
|
+
# @raise [Woods::ExtractionError] when any flow-family step fails
|
|
517
1389
|
def precompute_flows
|
|
518
1390
|
all_units = @results.values.flatten(1)
|
|
519
|
-
precomputer = FlowPrecomputer.new(units: all_units, graph: @dependency_graph, output_dir:
|
|
1391
|
+
precomputer = FlowPrecomputer.new(units: all_units, graph: @dependency_graph, output_dir: payload_dir.to_s)
|
|
520
1392
|
flow_map = precomputer.precompute
|
|
521
1393
|
rewrite_flow_annotated_units
|
|
1394
|
+
sweep_orphaned_flow_files
|
|
522
1395
|
Rails.logger.info "[Woods] Precomputed #{flow_map.size} request flows"
|
|
523
|
-
rescue StandardError => e
|
|
524
|
-
Rails.logger.error "[Woods] Flow precomputation failed: #{e.message}"
|
|
525
1396
|
end
|
|
526
1397
|
|
|
527
1398
|
# Precompute runs after write_results (FlowAssembler reads unit JSON
|
|
@@ -540,75 +1411,578 @@ module Woods
|
|
|
540
1411
|
annotated = units.select { |u| u.metadata[:flow_paths] }
|
|
541
1412
|
next if annotated.empty?
|
|
542
1413
|
|
|
543
|
-
type_dir =
|
|
1414
|
+
type_dir = payload_dir.join(type.to_s)
|
|
544
1415
|
annotated.each do |unit|
|
|
545
|
-
|
|
1416
|
+
AtomicFile.write(
|
|
546
1417
|
type_dir.join(collision_safe_filename(unit.identifier)),
|
|
547
|
-
json_serialize(unit.to_h)
|
|
1418
|
+
json_serialize(unit.to_h),
|
|
1419
|
+
durable: payload_writes_durable?
|
|
548
1420
|
)
|
|
549
1421
|
end
|
|
550
|
-
|
|
1422
|
+
AtomicFile.write(
|
|
551
1423
|
type_dir.join('_index.json'),
|
|
552
|
-
json_serialize(type_index_entries(units))
|
|
1424
|
+
json_serialize(type_index_entries(units)),
|
|
1425
|
+
durable: payload_writes_durable?
|
|
553
1426
|
)
|
|
554
1427
|
end
|
|
555
1428
|
end
|
|
556
1429
|
|
|
557
|
-
#
|
|
558
|
-
#
|
|
559
|
-
#
|
|
560
|
-
|
|
561
|
-
|
|
562
|
-
|
|
563
|
-
|
|
564
|
-
|
|
565
|
-
|
|
566
|
-
|
|
567
|
-
|
|
1430
|
+
# Incremental counterpart of Phase 5.5 (M3): the run's controller delta
|
|
1431
|
+
# gets the same flow annotations a full run would give it, flow
|
|
1432
|
+
# documents and flow_index.json are refreshed for the touched
|
|
1433
|
+
# controllers while untouched controllers' entries carry forward, and
|
|
1434
|
+
# flows/ documents no entry references are swept. Without this, a
|
|
1435
|
+
# controller re-extracted incrementally lost metadata[:flow_paths],
|
|
1436
|
+
# flow_index.json described pre-change routes, and flows/ files for
|
|
1437
|
+
# deleted or renamed controllers persisted across every generation —
|
|
1438
|
+
# seeded forward by PAYLOAD_DIRS.
|
|
1439
|
+
#
|
|
1440
|
+
# Runs inside {#finalize_incremental_run} so both extract_changed and
|
|
1441
|
+
# refresh (whose routes cascade rewrites controllers wholesale) get it,
|
|
1442
|
+
# and so a no-op run — which returns before this — leaves flows alone.
|
|
1443
|
+
#
|
|
1444
|
+
# Skip rule: the refresh participates whenever the gate is on and
|
|
1445
|
+
# something is touched. A genuinely absent family — no flows/ directory,
|
|
1446
|
+
# or an empty one — skips: there is nothing to carry forward, publication
|
|
1447
|
+
# proceeds, and the next full extraction with the gate on builds the
|
|
1448
|
+
# family. Once the family holds ANY artifact it is authoritative: a
|
|
1449
|
+
# missing flow_index.json among documents, a corrupt one, a failed
|
|
1450
|
+
# rehydration, write, patch, sweep, or type-index regeneration raises,
|
|
1451
|
+
# and the raise propagates out of {#finalize_incremental_run} BEFORE
|
|
1452
|
+
# {#publish_generation} — no generation bump, the preceding generation
|
|
1453
|
+
# stays resolved and readable. ( woods:validate applies the same rule
|
|
1454
|
+
# for a populated family without an index.)
|
|
1455
|
+
#
|
|
1456
|
+
# @param touched [Set<String>] identifiers added, re-extracted, or removed
|
|
1457
|
+
# @return [void]
|
|
1458
|
+
# @raise [Woods::ExtractionError] when authoritative flow state is
|
|
1459
|
+
# missing or corrupt, or a flow-family write fails
|
|
1460
|
+
def refresh_incremental_flows(touched)
|
|
1461
|
+
return unless Woods.configuration.precompute_flows
|
|
1462
|
+
return if touched.empty?
|
|
1463
|
+
return unless flow_family_present?
|
|
1464
|
+
|
|
1465
|
+
controllers_dir = payload_dir.join('controllers')
|
|
1466
|
+
reextracted = touched.select { |id| controllers_dir.join(collision_safe_filename(id)).exist? }
|
|
1467
|
+
removed = previous_flow_index_controllers & (touched.to_set - reextracted.to_set)
|
|
1468
|
+
return if reextracted.empty? && removed.empty?
|
|
1469
|
+
|
|
1470
|
+
reassemble, carried = partition_flow_controllers(
|
|
1471
|
+
reextracted.filter_map { |id| unit_from_payload(:controllers, id) }
|
|
1472
|
+
)
|
|
1473
|
+
Rails.logger.info "[Woods] Refreshing flows for #{reassemble.size} controller(s), " \
|
|
1474
|
+
"#{carried.size} carried forward, #{removed.size} removed..."
|
|
1475
|
+
precomputer = FlowPrecomputer.new(units: [], graph: @dependency_graph, output_dir: payload_dir.to_s)
|
|
1476
|
+
annotations = precomputer.recompute_delta(
|
|
1477
|
+
touched_units: reassemble,
|
|
1478
|
+
removed_identifiers: removed.to_a,
|
|
1479
|
+
carried_identifiers: carried.map(&:identifier)
|
|
1480
|
+
)
|
|
1481
|
+
patch_flow_annotations(annotations)
|
|
1482
|
+
sweep_orphaned_flow_files
|
|
1483
|
+
# The annotation patch changed controller JSON after the run's type
|
|
1484
|
+
# index regeneration; the index carries estimated_tokens, which the
|
|
1485
|
+
# flow_paths are part of — a full run builds its index from the
|
|
1486
|
+
# annotated in-memory units, so the incremental one re-derives it from
|
|
1487
|
+
# the annotated files to match. A failure here raises like every
|
|
1488
|
+
# other refresh failure.
|
|
1489
|
+
regenerate_type_index(:controllers) if annotations.any?
|
|
1490
|
+
end
|
|
568
1491
|
|
|
569
|
-
|
|
570
|
-
|
|
571
|
-
|
|
1492
|
+
# Split the run's re-extracted controllers into the ones whose flow
|
|
1493
|
+
# documents have to be reassembled and the ones that only need their
|
|
1494
|
+
# annotation back.
|
|
1495
|
+
#
|
|
1496
|
+
# A controller is reassembled when the run changed something its flow can
|
|
1497
|
+
# reach ({#flow_scope_for}), or when its own action set no longer matches
|
|
1498
|
+
# the previous generation's index. The second test is what catches an
|
|
1499
|
+
# action arriving from further up a controller inheritance chain than the
|
|
1500
|
+
# flow radius reaches, and a controller the index has never seen.
|
|
1501
|
+
#
|
|
1502
|
+
# Without a scope (a targeted {#refresh}, or a routes re-run) everything
|
|
1503
|
+
# is reassembled, which is what this path always did.
|
|
1504
|
+
#
|
|
1505
|
+
# @param units [Array<ExtractedUnit>] the run's re-extracted controllers
|
|
1506
|
+
# @return [Array(Array<ExtractedUnit>, Array<ExtractedUnit>)]
|
|
1507
|
+
def partition_flow_controllers(units)
|
|
1508
|
+
return [units, []] if @flow_scope.nil?
|
|
1509
|
+
|
|
1510
|
+
previous = previous_flow_index_actions
|
|
1511
|
+
units.partition do |unit|
|
|
1512
|
+
@flow_scope.include?(unit.identifier) || flow_actions_of(unit) != previous[unit.identifier]
|
|
572
1513
|
end
|
|
1514
|
+
end
|
|
573
1515
|
|
|
574
|
-
|
|
575
|
-
|
|
576
|
-
|
|
1516
|
+
# The actions a unit's metadata declares, as the index records them.
|
|
1517
|
+
#
|
|
1518
|
+
# @param unit [ExtractedUnit]
|
|
1519
|
+
# @return [Set<String>]
|
|
1520
|
+
def flow_actions_of(unit)
|
|
1521
|
+
Array(unit.metadata[:actions] || unit.metadata['actions']).to_set(&:to_s)
|
|
1522
|
+
end
|
|
577
1523
|
|
|
578
|
-
|
|
579
|
-
|
|
580
|
-
|
|
1524
|
+
# The previous generation's flow index, grouped by controller.
|
|
1525
|
+
#
|
|
1526
|
+
# @return [Hash{String => Set<String>}] controller identifier to its
|
|
1527
|
+
# recorded action names, defaulting to an empty set
|
|
1528
|
+
# @raise [Woods::ExtractionError] when the index is missing or corrupt
|
|
1529
|
+
def previous_flow_index_actions
|
|
1530
|
+
previous_flow_index_entries.keys.each_with_object(Hash.new { Set.new }) do |entry_point, grouped|
|
|
1531
|
+
controller, action = entry_point.to_s.split('#', 2)
|
|
1532
|
+
next unless action
|
|
1533
|
+
|
|
1534
|
+
grouped[controller] = grouped[controller] + [action]
|
|
1535
|
+
end
|
|
1536
|
+
end
|
|
581
1537
|
|
|
582
|
-
|
|
583
|
-
|
|
1538
|
+
# Does the run's seeded payload hold a flow family at all? An absent
|
|
1539
|
+
# family (no flows/ directory, or an empty one) is a genuine absence —
|
|
1540
|
+
# typically an index built while the gate was off — and skips the
|
|
1541
|
+
# refresh rather than publishing a delta-only index.
|
|
1542
|
+
#
|
|
1543
|
+
# @return [Boolean]
|
|
1544
|
+
def flow_family_present?
|
|
1545
|
+
flows_dir = payload_dir.join('flows')
|
|
1546
|
+
return false unless flows_dir.directory?
|
|
584
1547
|
|
|
585
|
-
|
|
586
|
-
unit.metadata[:git] = git_data[relative] if git_data[relative]
|
|
587
|
-
end
|
|
588
|
-
end
|
|
1548
|
+
!Dir[flows_dir.join('*.json')].empty?
|
|
589
1549
|
end
|
|
590
1550
|
|
|
591
|
-
#
|
|
1551
|
+
# Controller identifiers that hold entries in the previous generation's
|
|
1552
|
+
# flow_index.json. Only these can be flow-removed this run: a controller
|
|
1553
|
+
# with no index entries has no documents to sweep and no annotation to
|
|
1554
|
+
# clear.
|
|
592
1555
|
#
|
|
593
|
-
#
|
|
594
|
-
#
|
|
595
|
-
#
|
|
1556
|
+
# The read is authoritative ({#flow_family_present?} guaranteed the
|
|
1557
|
+
# family holds artifacts): a missing index among documents, or a corrupt
|
|
1558
|
+
# one, raises so publication aborts.
|
|
596
1559
|
#
|
|
597
|
-
#
|
|
598
|
-
#
|
|
599
|
-
def
|
|
600
|
-
|
|
601
|
-
units.each do |unit|
|
|
602
|
-
unit.file_path = normalize_file_path(unit.file_path)
|
|
603
|
-
end
|
|
604
|
-
end
|
|
1560
|
+
# @return [Set<String>]
|
|
1561
|
+
# @raise [Woods::ExtractionError] when the index is missing or corrupt
|
|
1562
|
+
def previous_flow_index_controllers
|
|
1563
|
+
previous_flow_index_entries.keys.to_set { |entry_point| entry_point.to_s.split('#', 2).first }
|
|
605
1564
|
end
|
|
606
1565
|
|
|
607
|
-
#
|
|
1566
|
+
# The previous generation's flow index itself, read once per run: both
|
|
1567
|
+
# the removal set and the reassembly partition are derived from it.
|
|
608
1568
|
#
|
|
609
|
-
# @
|
|
610
|
-
# @
|
|
611
|
-
|
|
1569
|
+
# @return [Hash{String => String}] entry point to relative document path
|
|
1570
|
+
# @raise [Woods::ExtractionError] when the index is missing or corrupt
|
|
1571
|
+
def previous_flow_index_entries
|
|
1572
|
+
return @previous_flow_index_entries if @previous_flow_index_entries
|
|
1573
|
+
|
|
1574
|
+
index_path = payload_dir.join('flows', 'flow_index.json')
|
|
1575
|
+
raise Woods::ExtractionError, 'flows/ is populated but flow_index.json is missing' unless index_path.exist?
|
|
1576
|
+
|
|
1577
|
+
@previous_flow_index_entries = JSON.parse(AtomicFile.read(index_path))
|
|
1578
|
+
rescue JSON::ParserError => e
|
|
1579
|
+
raise Woods::ExtractionError, "previous flow_index.json does not parse: #{e.message}"
|
|
1580
|
+
end
|
|
1581
|
+
|
|
1582
|
+
# Rehydrate one unit from its payload JSON for the incremental flow
|
|
1583
|
+
# pass. FlowPrecomputer only needs the identifier and the actions
|
|
1584
|
+
# metadata; the source lives on disk, where FlowAssembler reads it.
|
|
1585
|
+
#
|
|
1586
|
+
# @param type_key [Symbol] extractor key naming the payload directory
|
|
1587
|
+
# @param identifier [String]
|
|
1588
|
+
# @return [ExtractedUnit, nil]
|
|
1589
|
+
# @raise [Woods::ExtractionError] when the unit JSON does not parse —
|
|
1590
|
+
# skipping it would publish annotations against stale state
|
|
1591
|
+
def unit_from_payload(type_key, identifier)
|
|
1592
|
+
path = payload_dir.join(type_key.to_s, collision_safe_filename(identifier))
|
|
1593
|
+
return unless File.exist?(path)
|
|
1594
|
+
|
|
1595
|
+
data = JSON.parse(AtomicFile.read(path))
|
|
1596
|
+
unit = ExtractedUnit.new(
|
|
1597
|
+
type: data['type'],
|
|
1598
|
+
identifier: data['identifier'] || identifier,
|
|
1599
|
+
file_path: data['file_path']
|
|
1600
|
+
)
|
|
1601
|
+
unit.metadata = data['metadata'] || {}
|
|
1602
|
+
unit.source_code = data['source_code']
|
|
1603
|
+
unit
|
|
1604
|
+
rescue JSON::ParserError => e
|
|
1605
|
+
raise Woods::ExtractionError, "could not rehydrate #{identifier} for flow refresh: #{e.message}"
|
|
1606
|
+
end
|
|
1607
|
+
|
|
1608
|
+
# Write metadata[:flow_paths] into the re-extracted controllers' unit
|
|
1609
|
+
# JSON — the incremental counterpart of {#rewrite_flow_annotated_units}.
|
|
1610
|
+
# A controller with no flows this run loses any annotation a previous
|
|
1611
|
+
# run had written, which is what a full run would have produced for the
|
|
1612
|
+
# same tree. Read-compare-write, like {#rewrite_unit_json_of_type}; a
|
|
1613
|
+
# read or write failure raises so publication aborts.
|
|
1614
|
+
#
|
|
1615
|
+
# @param annotations [Hash{String => Hash{String => String}}] from
|
|
1616
|
+
# {FlowPrecomputer#recompute_delta}
|
|
1617
|
+
# @return [void]
|
|
1618
|
+
def patch_flow_annotations(annotations)
|
|
1619
|
+
annotations.each do |identifier, flow_paths|
|
|
1620
|
+
path = payload_dir.join('controllers', collision_safe_filename(identifier))
|
|
1621
|
+
next unless File.exist?(path)
|
|
1622
|
+
|
|
1623
|
+
data = JSON.parse(AtomicFile.read(path))
|
|
1624
|
+
before = JSON.generate(data)
|
|
1625
|
+
|
|
1626
|
+
metadata = (data['metadata'] ||= {})
|
|
1627
|
+
if flow_paths.any?
|
|
1628
|
+
metadata['flow_paths'] = flow_paths
|
|
1629
|
+
else
|
|
1630
|
+
metadata.delete('flow_paths')
|
|
1631
|
+
end
|
|
1632
|
+
next if JSON.generate(data) == before
|
|
1633
|
+
|
|
1634
|
+
AtomicFile.write(path, json_serialize(data), durable: payload_writes_durable?)
|
|
1635
|
+
end
|
|
1636
|
+
end
|
|
1637
|
+
|
|
1638
|
+
# Remove flows/ documents no entry of flow_index.json references.
|
|
1639
|
+
#
|
|
1640
|
+
# Deliberately NOT part of {#sweep_orphaned_unit_files}: that sweep
|
|
1641
|
+
# deletes per-type unit files no in-memory unit accounts for, while
|
|
1642
|
+
# flows/ holds neither units nor an _index.json. The flow family is
|
|
1643
|
+
# defined by flow_index.json's references, and this validates against
|
|
1644
|
+
# exactly those (M3) — the same artifact the validator treats separately
|
|
1645
|
+
# (G-2). Before this, a controller deleted or renamed incrementally left
|
|
1646
|
+
# its flow documents behind forever.
|
|
1647
|
+
#
|
|
1648
|
+
# Reconciled skip rule: a genuinely empty directory is an absence and is
|
|
1649
|
+
# skipped; a populated one whose index is missing, or an index that does
|
|
1650
|
+
# not parse, is corruption and raises — with nothing to validate
|
|
1651
|
+
# against, deleting every document would be the one outcome worse than
|
|
1652
|
+
# keeping orphans. Failures raise so the incremental caller aborts
|
|
1653
|
+
# publication.
|
|
1654
|
+
#
|
|
1655
|
+
# @return [void]
|
|
1656
|
+
# @raise [Woods::ExtractionError] when authoritative flow state is
|
|
1657
|
+
# missing or corrupt
|
|
1658
|
+
def sweep_orphaned_flow_files
|
|
1659
|
+
flows_dir = payload_dir.join('flows')
|
|
1660
|
+
return unless flows_dir.directory?
|
|
1661
|
+
|
|
1662
|
+
index_path = flows_dir.join('flow_index.json')
|
|
1663
|
+
unless index_path.exist?
|
|
1664
|
+
return if Dir[flows_dir.join('*.json')].empty?
|
|
1665
|
+
|
|
1666
|
+
raise Woods::ExtractionError, 'flows/ is populated but flow_index.json is missing'
|
|
1667
|
+
end
|
|
1668
|
+
|
|
1669
|
+
keep = parse_flow_index_for_sweep(index_path)
|
|
1670
|
+
.to_set { |relative| File.basename(relative.to_s) } << 'flow_index.json'
|
|
1671
|
+
orphans = Dir[flows_dir.join('*.json').to_s].reject { |file| keep.include?(File.basename(file)) }
|
|
1672
|
+
return if orphans.empty?
|
|
1673
|
+
|
|
1674
|
+
orphans.each { |file| FileUtils.rm_f(file) }
|
|
1675
|
+
Rails.logger.info "[Woods] Swept #{orphans.size} orphaned flow file(s)"
|
|
1676
|
+
end
|
|
1677
|
+
|
|
1678
|
+
# @param index_path [Pathname]
|
|
1679
|
+
# @return [Array<String>] the referenced document paths
|
|
1680
|
+
# @raise [Woods::ExtractionError] when the index does not parse
|
|
1681
|
+
def parse_flow_index_for_sweep(index_path)
|
|
1682
|
+
JSON.parse(AtomicFile.read(index_path)).values
|
|
1683
|
+
rescue JSON::ParserError => e
|
|
1684
|
+
raise Woods::ExtractionError, "flow_index.json does not parse: #{e.message}"
|
|
1685
|
+
end
|
|
1686
|
+
|
|
1687
|
+
# ──────────────────────────────────────────────────────────────────────
|
|
1688
|
+
# Git Enrichment
|
|
1689
|
+
# ──────────────────────────────────────────────────────────────────────
|
|
1690
|
+
|
|
1691
|
+
def enrich_with_git_data
|
|
1692
|
+
return unless git_available?
|
|
1693
|
+
|
|
1694
|
+
# Collect all file paths that need git data. Only app-owned paths under
|
|
1695
|
+
# Rails.root qualify — see {#git_enrichable_path?}.
|
|
1696
|
+
root = "#{Rails.root}/"
|
|
1697
|
+
file_paths = []
|
|
1698
|
+
@results.each do |type, units|
|
|
1699
|
+
next if %i[rails_source gem_source].include?(type)
|
|
1700
|
+
|
|
1701
|
+
units.each do |unit|
|
|
1702
|
+
path = unit.file_path
|
|
1703
|
+
file_paths << path if git_enrichable_path?(path, root)
|
|
1704
|
+
end
|
|
1705
|
+
end
|
|
1706
|
+
|
|
1707
|
+
# Batch-fetch all git data in minimal subprocess calls
|
|
1708
|
+
git_data = batch_git_data(file_paths)
|
|
1709
|
+
|
|
1710
|
+
# Assign results to units
|
|
1711
|
+
@results.each do |type, units|
|
|
1712
|
+
next if %i[rails_source gem_source].include?(type)
|
|
1713
|
+
|
|
1714
|
+
units.each do |unit|
|
|
1715
|
+
next unless unit.file_path
|
|
1716
|
+
|
|
1717
|
+
relative = unit.file_path.sub(root, '')
|
|
1718
|
+
unit.metadata[:git] = git_data[relative] if git_data[relative]
|
|
1719
|
+
end
|
|
1720
|
+
end
|
|
1721
|
+
end
|
|
1722
|
+
|
|
1723
|
+
# Copy git facts onto graph nodes so the analyzer can read them without
|
|
1724
|
+
# the units (an incremental run never holds every unit in memory).
|
|
1725
|
+
#
|
|
1726
|
+
# @return [void]
|
|
1727
|
+
# build_file_metadata always emits commit_count and change_frequency together
|
|
1728
|
+
# and non-nil, so no nil compaction is needed here (contrast annotate_node_from_git).
|
|
1729
|
+
def annotate_graph_with_git_data
|
|
1730
|
+
@results.each_value do |units|
|
|
1731
|
+
units.each do |unit|
|
|
1732
|
+
git = unit.metadata[:git]
|
|
1733
|
+
next unless git.is_a?(Hash)
|
|
1734
|
+
|
|
1735
|
+
@dependency_graph.annotate(
|
|
1736
|
+
unit.identifier,
|
|
1737
|
+
type: unit.type,
|
|
1738
|
+
commit_count: git[:commit_count],
|
|
1739
|
+
change_frequency: git[:change_frequency]
|
|
1740
|
+
)
|
|
1741
|
+
end
|
|
1742
|
+
end
|
|
1743
|
+
end
|
|
1744
|
+
|
|
1745
|
+
# The one constructor for the analyzer both extraction paths use.
|
|
1746
|
+
# `Woods.configuration` can be nil in specs that reset it; fall back to
|
|
1747
|
+
# the analyzer's own default rather than raising mid-run.
|
|
1748
|
+
#
|
|
1749
|
+
# @return [GraphAnalyzer]
|
|
1750
|
+
def build_graph_analyzer
|
|
1751
|
+
config = Woods.configuration
|
|
1752
|
+
ratio = config&.volatile_dependency_ratio || GraphAnalyzer::DEFAULT_VOLATILE_RATIO
|
|
1753
|
+
GraphAnalyzer.new(
|
|
1754
|
+
@dependency_graph,
|
|
1755
|
+
volatile_ratio: ratio,
|
|
1756
|
+
cycle_limit: config ? config.graph_cycle_limit : GraphAnalyzer::DEFAULT_CYCLE_LIMIT,
|
|
1757
|
+
cycle_max_length: config ? config.graph_cycle_max_length : GraphAnalyzer::DEFAULT_CYCLE_MAX_LENGTH
|
|
1758
|
+
)
|
|
1759
|
+
end
|
|
1760
|
+
|
|
1761
|
+
# ──────────────────────────────────────────────────────────────────────
|
|
1762
|
+
# Package membership (#280)
|
|
1763
|
+
# ──────────────────────────────────────────────────────────────────────
|
|
1764
|
+
|
|
1765
|
+
# The package extractor instance this run resolves membership through.
|
|
1766
|
+
# Always built through {#extractor_for}, never `@extractors[:packages]`
|
|
1767
|
+
# (the instance Phase 1 used to extract package units): reading that one
|
|
1768
|
+
# would tie package-membership lookups to whichever instance happened to
|
|
1769
|
+
# run first, instead of to this run's own memoized, on-demand build.
|
|
1770
|
+
# Reset at the start of every run (both {#extract_all} and
|
|
1771
|
+
# {#prepare_incremental_run}), so a later run never resolves membership
|
|
1772
|
+
# against a prior run's package set.
|
|
1773
|
+
#
|
|
1774
|
+
# @return [Extractors::PackageExtractor]
|
|
1775
|
+
def package_resolver
|
|
1776
|
+
@package_resolver ||= extractor_for(:packages) || Extractors::PackageExtractor.new
|
|
1777
|
+
end
|
|
1778
|
+
|
|
1779
|
+
# Set or clear metadata[:package] on one unit. Framework units and
|
|
1780
|
+
# units with no path are never members. The key is deleted rather than
|
|
1781
|
+
# set to nil so a unit outside every package serializes as before.
|
|
1782
|
+
#
|
|
1783
|
+
# `unit.file_path` may be absolute (the full path, before Phase 4.5
|
|
1784
|
+
# relativization, and {#register_and_write}'s incremental path, before
|
|
1785
|
+
# its own normalization) or Rails.root-relative (unit JSON already on
|
|
1786
|
+
# disk); {Extractors::PackageExtractor#package_for} accepts either.
|
|
1787
|
+
#
|
|
1788
|
+
# @param unit [ExtractedUnit]
|
|
1789
|
+
# @return [String, nil] the package name
|
|
1790
|
+
def annotate_package(unit)
|
|
1791
|
+
return nil if %i[rails_source gem_source].include?(unit.type) || unit.file_path.nil?
|
|
1792
|
+
|
|
1793
|
+
package = package_resolver.package_for(unit.file_path)
|
|
1794
|
+
if package
|
|
1795
|
+
unit.metadata[:package] = package
|
|
1796
|
+
else
|
|
1797
|
+
unit.metadata.delete(:package)
|
|
1798
|
+
end
|
|
1799
|
+
package
|
|
1800
|
+
end
|
|
1801
|
+
|
|
1802
|
+
# Full-path pass over every extracted unit. Skipped entirely when the
|
|
1803
|
+
# app declares no packages, so a non-Packwerk app's output is unchanged
|
|
1804
|
+
# (I5): no work, and no `package` key on any unit or node.
|
|
1805
|
+
#
|
|
1806
|
+
# @return [void]
|
|
1807
|
+
def annotate_packages
|
|
1808
|
+
return if package_resolver.package_roots.empty?
|
|
1809
|
+
|
|
1810
|
+
@results.each_value { |units| units.each { |unit| annotate_package(unit) } }
|
|
1811
|
+
end
|
|
1812
|
+
|
|
1813
|
+
# Incremental counterpart: when a package file changed, membership of
|
|
1814
|
+
# units this run never touched may have changed too (a new package
|
|
1815
|
+
# root, a renamed package). Walk the payload's unit JSON, rewrite the
|
|
1816
|
+
# ones whose package differs, and annotate their nodes. Bounded to runs
|
|
1817
|
+
# where a package trigger fired, which is rare; every other run pays
|
|
1818
|
+
# nothing.
|
|
1819
|
+
#
|
|
1820
|
+
# @param change_set [ChangeSet]
|
|
1821
|
+
# @param affected_types [Set<Symbol>]
|
|
1822
|
+
# @return [Set<String>] identifiers rewritten
|
|
1823
|
+
def reannotate_packages(change_set, affected_types)
|
|
1824
|
+
keys = PathDispatcher.new.whole_app_keys_for_all(change_set.relative_paths)
|
|
1825
|
+
return Set.new unless keys.include?(:packages)
|
|
1826
|
+
|
|
1827
|
+
Dir[payload_dir.join('*', '*.json').to_s].each_with_object(Set.new) do |file, touched|
|
|
1828
|
+
next if File.basename(file) == '_index.json'
|
|
1829
|
+
|
|
1830
|
+
type_dir = File.basename(File.dirname(file))
|
|
1831
|
+
next if type_dir == 'rails_source' || PAYLOAD_DIRS.include?(type_dir)
|
|
1832
|
+
|
|
1833
|
+
identifier = reannotate_unit_file(file, type_dir)
|
|
1834
|
+
next unless identifier
|
|
1835
|
+
|
|
1836
|
+
touched.add(identifier)
|
|
1837
|
+
affected_types&.add(type_dir.to_sym)
|
|
1838
|
+
end
|
|
1839
|
+
end
|
|
1840
|
+
|
|
1841
|
+
# @param file [String] absolute path to one unit JSON file
|
|
1842
|
+
# @param type_dir [String] the extractor key directory it lives in
|
|
1843
|
+
# @return [String, nil] the identifier when the file was rewritten
|
|
1844
|
+
def reannotate_unit_file(file, type_dir)
|
|
1845
|
+
data = JSON.parse(AtomicFile.read(file))
|
|
1846
|
+
relative_path = data['file_path']
|
|
1847
|
+
return nil if relative_path.nil? || relative_path.start_with?('/')
|
|
1848
|
+
|
|
1849
|
+
package = package_resolver.package_for(relative_path)
|
|
1850
|
+
metadata = (data['metadata'] ||= {})
|
|
1851
|
+
return nil if metadata['package'] == package
|
|
1852
|
+
|
|
1853
|
+
if package
|
|
1854
|
+
metadata['package'] = package
|
|
1855
|
+
else
|
|
1856
|
+
metadata.delete('package')
|
|
1857
|
+
end
|
|
1858
|
+
AtomicFile.write(file, json_serialize(data), durable: payload_writes_durable?)
|
|
1859
|
+
|
|
1860
|
+
identifier = data['identifier']
|
|
1861
|
+
type = (data['type'] || type_dir.singularize).to_sym
|
|
1862
|
+
@dependency_graph.annotate(identifier, type: type, package: package)
|
|
1863
|
+
identifier
|
|
1864
|
+
rescue JSON::ParserError => e
|
|
1865
|
+
Rails.logger.warn "[Woods] Could not re-annotate package on #{file}: #{e.message}"
|
|
1866
|
+
nil
|
|
1867
|
+
end
|
|
1868
|
+
|
|
1869
|
+
# Is this a path worth asking git about?
|
|
1870
|
+
#
|
|
1871
|
+
# A gem-owned unit (an engine model) carries its real path. Outside
|
|
1872
|
+
# Rails.root, git refuses the whole `log` invocation when any pathspec is
|
|
1873
|
+
# outside the repository — one gem path would erase the git metadata of
|
|
1874
|
+
# the other 499 units in its 500-path batch. Inside Rails.root, a bundle
|
|
1875
|
+
# vendored at `vendor/bundle` puts the same gem files under the root
|
|
1876
|
+
# prefix, gitignored, so sending them is wasted pathspec work every run.
|
|
1877
|
+
# Same exclusions as {Extractors::SharedUtilityMethods#app_source?}.
|
|
1878
|
+
#
|
|
1879
|
+
# @param path [String, nil] absolute file path
|
|
1880
|
+
# @param root [String] Rails.root with a trailing separator
|
|
1881
|
+
# @return [Boolean]
|
|
1882
|
+
def git_enrichable_path?(path, root)
|
|
1883
|
+
return false unless path&.start_with?(root)
|
|
1884
|
+
return false if path.include?('/vendor/') || path.include?('/node_modules/')
|
|
1885
|
+
|
|
1886
|
+
File.exist?(path)
|
|
1887
|
+
end
|
|
1888
|
+
|
|
1889
|
+
# Normalize all unit file_paths to relative paths (relative to Rails.root).
|
|
1890
|
+
#
|
|
1891
|
+
# Extractors set file_path via source_location, which returns absolute paths.
|
|
1892
|
+
# This normalization ensures consistent relative paths (e.g., "app/models/user.rb")
|
|
1893
|
+
# across all environments (local, Docker, CI) where Rails.root differs.
|
|
1894
|
+
#
|
|
1895
|
+
# Must run after enrich_with_git_data, which needs absolute paths for
|
|
1896
|
+
# File.exist? checks and git log commands.
|
|
1897
|
+
# Write a unit's JSON, unless the bytes on disk are already exactly that.
|
|
1898
|
+
#
|
|
1899
|
+
# A routes change replaces every `ROUTE_CONSUMER_EXTRACTORS` type wholesale
|
|
1900
|
+
# — on a production-shaped host that measured 1,707 units, roughly a quarter
|
|
1901
|
+
# of the index — and almost all of them re-serialize to the bytes already
|
|
1902
|
+
# there. `AtomicFile.write` is a tempfile plus an fsync plus a rename each
|
|
1903
|
+
# time, so the fsync is the cost being avoided here; the comparison read is
|
|
1904
|
+
# cheaper than the write it replaces.
|
|
1905
|
+
#
|
|
1906
|
+
# Only the *write* is skipped. Graph registration, the dependents marking
|
|
1907
|
+
# and `@incremental_written` all still happen for every unit, because those
|
|
1908
|
+
# are what equivalence and the git-enrichment pass depend on — skipping any
|
|
1909
|
+
# of them would make an unchanged unit differ from a full extraction.
|
|
1910
|
+
#
|
|
1911
|
+
# Compared as bytes: `AtomicFile.write` is binmode, and the encoding a read
|
|
1912
|
+
# comes back tagged with depends on the process's default external encoding
|
|
1913
|
+
# (US-ASCII under `LANG=C`, which is where the daemon runs).
|
|
1914
|
+
#
|
|
1915
|
+
# @param path [Pathname] destination
|
|
1916
|
+
# @param unit [ExtractedUnit] unit to serialize
|
|
1917
|
+
# @return [void]
|
|
1918
|
+
def write_unit_file(path, unit)
|
|
1919
|
+
payload = json_serialize(unit.to_h)
|
|
1920
|
+
return if identical_on_disk?(path, payload)
|
|
1921
|
+
|
|
1922
|
+
AtomicFile.write(path, payload, durable: payload_writes_durable?)
|
|
1923
|
+
end
|
|
1924
|
+
|
|
1925
|
+
# The serialized `extracted_at` scalar as {ExtractedUnit#to_h} +
|
|
1926
|
+
# {#json_serialize} emit it, compact or pretty. `Time#iso8601` produces
|
|
1927
|
+
# exactly this value shape — no fractional seconds, `Z` or a `±hh:mm`
|
|
1928
|
+
# offset — and `spec/extracted_unit_spec.rb` pins that, so a change to the
|
|
1929
|
+
# stamp's shape fails a spec instead of quietly un-matching this mask.
|
|
1930
|
+
# The value constraint is what keeps the mask honest against user code: a
|
|
1931
|
+
# bare `"extracted_at":` cannot occur inside any JSON *string* value
|
|
1932
|
+
# (interior quotes serialize as `\"`), so only a real JSON key can match,
|
|
1933
|
+
# and only when it holds a timestamp — which no extractor emits below the
|
|
1934
|
+
# top level.
|
|
1935
|
+
EXTRACTED_AT_SCALAR =
|
|
1936
|
+
/("extracted_at":\s*")\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}(?:Z|[+-]\d{2}:\d{2})(?=")/
|
|
1937
|
+
# An implementation detail of the byte comparison, not part of the
|
|
1938
|
+
# extractor's surface (`private` does not scope constants).
|
|
1939
|
+
private_constant :EXTRACTED_AT_SCALAR
|
|
1940
|
+
|
|
1941
|
+
# @return [Boolean] true when the file already holds exactly these bytes,
|
|
1942
|
+
# the `extracted_at` stamp aside
|
|
1943
|
+
def identical_on_disk?(path, payload)
|
|
1944
|
+
return false unless File.exist?(path)
|
|
1945
|
+
|
|
1946
|
+
mask_extracted_at(AtomicFile.read(path).b) == mask_extracted_at(payload.b)
|
|
1947
|
+
rescue StandardError
|
|
1948
|
+
# An unreadable or half-written file is not a match; fall through and
|
|
1949
|
+
# rewrite it.
|
|
1950
|
+
false
|
|
1951
|
+
end
|
|
1952
|
+
|
|
1953
|
+
# Blank the `extracted_at` value so the byte comparison ignores it.
|
|
1954
|
+
#
|
|
1955
|
+
# {ExtractedUnit#to_h} stamps `extracted_at: Time.now.iso8601` on every
|
|
1956
|
+
# serialization, so compared raw the two sides *always* differed across
|
|
1957
|
+
# runs and the skip was inert (#208): every write did the comparison read
|
|
1958
|
+
# and then paid the fsync anyway. Masked with a regex rather than parsed,
|
|
1959
|
+
# because a JSON parse of both sides would cost more than the write being
|
|
1960
|
+
# avoided. A skipped write leaves the older stamp on disk deliberately —
|
|
1961
|
+
# the equivalence oracle (`spec/support/index_comparison.rb`) lists
|
|
1962
|
+
# `extracted_at` in `VOLATILE_UNIT_KEYS`: "a unit an incremental run
|
|
1963
|
+
# correctly left alone keeps an older stamp".
|
|
1964
|
+
#
|
|
1965
|
+
# @param bytes [String] serialized unit JSON, binary-tagged (`.b`) — the
|
|
1966
|
+
# ASCII-only pattern matches bytewise regardless of the content's
|
|
1967
|
+
# original encoding
|
|
1968
|
+
# @return [String] the bytes with the stamp's value removed
|
|
1969
|
+
def mask_extracted_at(bytes)
|
|
1970
|
+
bytes.gsub(EXTRACTED_AT_SCALAR, '\1')
|
|
1971
|
+
end
|
|
1972
|
+
|
|
1973
|
+
def normalize_file_paths
|
|
1974
|
+
@results.each_value do |units|
|
|
1975
|
+
units.each do |unit|
|
|
1976
|
+
unit.file_path = normalize_file_path(unit.file_path)
|
|
1977
|
+
end
|
|
1978
|
+
end
|
|
1979
|
+
end
|
|
1980
|
+
|
|
1981
|
+
# Strip Rails.root prefix from a file path, converting it to a relative path.
|
|
1982
|
+
#
|
|
1983
|
+
# @param path [String, nil] Absolute or relative file path
|
|
1984
|
+
# @return [String, nil] Relative path, or the original value if already relative,
|
|
1985
|
+
# nil, or not under Rails.root (e.g., a gem path)
|
|
612
1986
|
def normalize_file_path(path)
|
|
613
1987
|
return path unless path
|
|
614
1988
|
|
|
@@ -617,15 +1991,60 @@ module Woods
|
|
|
617
1991
|
path.start_with?(prefix) ? path.sub(prefix, '') : path
|
|
618
1992
|
end
|
|
619
1993
|
|
|
1994
|
+
# Can this run's git calls produce real facts?
|
|
1995
|
+
#
|
|
1996
|
+
# `rev-parse --git-dir` alone is not enough. Over a linked worktree whose
|
|
1997
|
+
# private git directory is reachable but whose `commondir` is not, it
|
|
1998
|
+
# answers while every ref lookup fails and `git log` exits 0 with nothing
|
|
1999
|
+
# to say. Enrichment then wrote `commit_count: 0` and
|
|
2000
|
+
# `change_frequency: new` onto every unit, which reads exactly like a file
|
|
2001
|
+
# that was never committed, where an absent git directory correctly omits
|
|
2002
|
+
# the keys (B-186). HEAD has to resolve.
|
|
2003
|
+
#
|
|
2004
|
+
# Memoized, so the warning below is emitted at most once per run.
|
|
2005
|
+
#
|
|
2006
|
+
# @return [Boolean]
|
|
620
2007
|
def git_available?
|
|
621
2008
|
return @git_available if defined?(@git_available)
|
|
622
2009
|
|
|
623
|
-
|
|
624
|
-
|
|
625
|
-
|
|
626
|
-
|
|
627
|
-
|
|
628
|
-
|
|
2010
|
+
_output, error, status = Open3.capture3(*git_argv('rev-parse', 'HEAD'))
|
|
2011
|
+
@git_available = status.success?
|
|
2012
|
+
warn_unresolvable_git(error) unless @git_available
|
|
2013
|
+
@git_available
|
|
2014
|
+
rescue StandardError
|
|
2015
|
+
@git_available = false
|
|
2016
|
+
end
|
|
2017
|
+
|
|
2018
|
+
# Say once why no unit will carry git metadata, but only when there is a
|
|
2019
|
+
# working tree to explain. No `.git` at the root is the ordinary source
|
|
2020
|
+
# tarball or `COPY`-without-`.git` case, and it is not a fault.
|
|
2021
|
+
#
|
|
2022
|
+
# @param error [String] git's own stderr
|
|
2023
|
+
# @return [void]
|
|
2024
|
+
def warn_unresolvable_git(error)
|
|
2025
|
+
return unless File.exist?(File.join(Rails.root.to_s, '.git'))
|
|
2026
|
+
|
|
2027
|
+
cause = error.to_s.lines.first.to_s.strip
|
|
2028
|
+
Rails.logger.warn(
|
|
2029
|
+
'[Woods] git cannot resolve HEAD for this working tree, so no unit will carry git ' \
|
|
2030
|
+
"metadata: #{cause}. Over a linked worktree in a container, mount the canonical git " \
|
|
2031
|
+
'directory and point WOODS_GIT_DIR at it; GIT_DIR alone is not enough, because the ' \
|
|
2032
|
+
"worktree's private git directory reaches the shared one through a relative pointer."
|
|
2033
|
+
)
|
|
2034
|
+
end
|
|
2035
|
+
|
|
2036
|
+
# The git command line every enrichment call runs.
|
|
2037
|
+
#
|
|
2038
|
+
# `-C <root>` keeps the result independent of the process working
|
|
2039
|
+
# directory. `WOODS_GIT_DIR` wins when set: it names the canonical git
|
|
2040
|
+
# directory directly, which is the escape hatch for a container that can
|
|
2041
|
+
# mount that directory but not the host path a worktree pointer names
|
|
2042
|
+
# (B-181).
|
|
2043
|
+
#
|
|
2044
|
+
# @param args [Array<String>] git arguments
|
|
2045
|
+
# @return [Array<String>] full argv
|
|
2046
|
+
def git_argv(*args)
|
|
2047
|
+
GitCommand.argv(Rails.root, *args)
|
|
629
2048
|
end
|
|
630
2049
|
|
|
631
2050
|
# Safe git command execution — no shell interpolation
|
|
@@ -633,7 +2052,7 @@ module Woods
|
|
|
633
2052
|
# @param args [Array<String>] Git command arguments
|
|
634
2053
|
# @return [String] Command output (empty string on failure)
|
|
635
2054
|
def run_git(*args)
|
|
636
|
-
output, status = Open3.
|
|
2055
|
+
output, _error, status = Open3.capture3(*git_argv(*args))
|
|
637
2056
|
status.success? ? output.strip : ''
|
|
638
2057
|
rescue StandardError
|
|
639
2058
|
''
|
|
@@ -641,13 +2060,18 @@ module Woods
|
|
|
641
2060
|
|
|
642
2061
|
# Batch-fetch git data for all file paths in two git commands.
|
|
643
2062
|
#
|
|
2063
|
+
# Duplicate paths are collapsed before slicing (audit P9d): many units
|
|
2064
|
+
# share one file_path, and duplicates only repeat a pathspec another
|
|
2065
|
+
# batch also sent. The result is keyed by relative path, so the output
|
|
2066
|
+
# is identical.
|
|
2067
|
+
#
|
|
644
2068
|
# @param file_paths [Array<String>] Absolute file paths
|
|
645
2069
|
# @return [Hash{String => Hash}] Keyed by relative path
|
|
646
2070
|
def batch_git_data(file_paths)
|
|
647
2071
|
return {} if file_paths.empty?
|
|
648
2072
|
|
|
649
2073
|
root = "#{Rails.root}/"
|
|
650
|
-
relative_paths = file_paths.map { |f| f.sub(root, '') }
|
|
2074
|
+
relative_paths = file_paths.map { |f| f.sub(root, '') }.uniq
|
|
651
2075
|
result = {}
|
|
652
2076
|
relative_paths.each { |rp| result[rp] = {} }
|
|
653
2077
|
|
|
@@ -735,23 +2159,91 @@ module Woods
|
|
|
735
2159
|
|
|
736
2160
|
def write_results
|
|
737
2161
|
@results.each do |type, units|
|
|
738
|
-
type_dir =
|
|
2162
|
+
type_dir = payload_dir.join(type.to_s)
|
|
739
2163
|
|
|
740
2164
|
units.each do |unit|
|
|
741
|
-
|
|
2165
|
+
AtomicFile.write(
|
|
742
2166
|
type_dir.join(collision_safe_filename(unit.identifier)),
|
|
743
|
-
json_serialize(unit.to_h)
|
|
2167
|
+
json_serialize(unit.to_h),
|
|
2168
|
+
durable: payload_writes_durable?
|
|
744
2169
|
)
|
|
745
2170
|
end
|
|
746
2171
|
|
|
747
2172
|
# Also write a type index for fast lookups
|
|
748
|
-
|
|
2173
|
+
AtomicFile.write(
|
|
749
2174
|
type_dir.join('_index.json'),
|
|
750
|
-
json_serialize(type_index_entries(units))
|
|
2175
|
+
json_serialize(type_index_entries(units)),
|
|
2176
|
+
durable: payload_writes_durable?
|
|
751
2177
|
)
|
|
752
2178
|
end
|
|
753
2179
|
end
|
|
754
2180
|
|
|
2181
|
+
# Delete unit JSON files in the just-written type directories that this
|
|
2182
|
+
# full run did not write (#177).
|
|
2183
|
+
#
|
|
2184
|
+
# A full extraction never wipes the output directory —
|
|
2185
|
+
# {#setup_output_directory} only mkdir_p's — so extracting into a
|
|
2186
|
+
# directory that saw a different version of the app left the previous
|
|
2187
|
+
# run's `<Unit>_<digest>.json` files behind. The manifest and the freshly
|
|
2188
|
+
# written `_index.json` were correct, but the NEXT incremental run's
|
|
2189
|
+
# {#regenerate_type_index} rebuilds the index from a disk glob of the type
|
|
2190
|
+
# dir, resurrecting every orphan into the listed index, and
|
|
2191
|
+
# {#persisted_counts} then writes the inflated counts into the manifest —
|
|
2192
|
+
# orphans became listed, retrievable-by-listing units the graph knows
|
|
2193
|
+
# nothing about. On a full run the in-memory `@results` are authoritative
|
|
2194
|
+
# — the same argument {#rewrite_flow_annotated_units} makes for refusing
|
|
2195
|
+
# the disk glob — so a unit file no current unit accounts for is stale by
|
|
2196
|
+
# definition and is removed here.
|
|
2197
|
+
#
|
|
2198
|
+
# The boundary is deliberately narrow:
|
|
2199
|
+
#
|
|
2200
|
+
# * Only the per-type directories this run produced — `@results` keys —
|
|
2201
|
+
# are entered. When `include_framework_sources` gates rails_source out
|
|
2202
|
+
# of the run ({#skip_by_configuration?}), `@results` holds no
|
|
2203
|
+
# `:rails_source` key, so a knob-off full run leaves explicitly
|
|
2204
|
+
# extracted framework units (`woods:extract_framework`) alone — the
|
|
2205
|
+
# #169 preservation decision. Nothing outside those directories is
|
|
2206
|
+
# reachable: manifest, graph, generation, `woods.json`, `dumps/`,
|
|
2207
|
+
# `flows/` and the lock/status files all live in the output root or in
|
|
2208
|
+
# directories no extractor key names.
|
|
2209
|
+
# * Only `*.json` directly inside a type dir is considered, and
|
|
2210
|
+
# `_index.json` is always kept. Non-JSON files and subdirectories are
|
|
2211
|
+
# never touched.
|
|
2212
|
+
# * Legitimate names are exactly what {#write_results} just wrote —
|
|
2213
|
+
# {FilenameUtils#collision_safe_filename} over the type's deduped units
|
|
2214
|
+
# (which covers every unit type the extractor emits: gem_source shares
|
|
2215
|
+
# `@results[:rails_source]`). Legacy {FilenameUtils#safe_filename} names
|
|
2216
|
+
# carry no digest suffix, so they never match a current name and are by
|
|
2217
|
+
# definition orphans of a pre-collision-safe run.
|
|
2218
|
+
#
|
|
2219
|
+
# The incremental path deliberately has no counterpart: it holds only
|
|
2220
|
+
# changed units in memory, so "not in memory" means nothing there —
|
|
2221
|
+
# deletion on that path is driven by the graph and the change set
|
|
2222
|
+
# ({#prune_vanished_units}, {#remove_replaced_units}).
|
|
2223
|
+
#
|
|
2224
|
+
# @return [void]
|
|
2225
|
+
def sweep_orphaned_unit_files
|
|
2226
|
+
@results.each do |type, units|
|
|
2227
|
+
type_dir = payload_dir.join(type.to_s)
|
|
2228
|
+
next unless type_dir.directory?
|
|
2229
|
+
|
|
2230
|
+
keep = units.to_set { |unit| collision_safe_filename(unit.identifier) }
|
|
2231
|
+
keep << '_index.json'
|
|
2232
|
+
|
|
2233
|
+
orphans = Dir[type_dir.join('*.json').to_s].reject { |file| keep.include?(File.basename(file)) }
|
|
2234
|
+
next if orphans.empty?
|
|
2235
|
+
|
|
2236
|
+
orphans.each { |file| FileUtils.rm_f(file) }
|
|
2237
|
+
Rails.logger.info "[Woods] Swept #{orphans.size} orphaned unit file(s) from #{type}/"
|
|
2238
|
+
end
|
|
2239
|
+
rescue StandardError => e
|
|
2240
|
+
# The extraction itself already succeeded; aborting the run over cleanup
|
|
2241
|
+
# would trade orphaned files for a lost index. Error, not warn: the
|
|
2242
|
+
# files this leaves behind are exactly what the next incremental run's
|
|
2243
|
+
# disk glob resurrects.
|
|
2244
|
+
Rails.logger.error "[Woods] Orphaned-unit sweep failed: #{e.message}"
|
|
2245
|
+
end
|
|
2246
|
+
|
|
755
2247
|
# Build the `_index.json` entry list for a set of in-memory units.
|
|
756
2248
|
# Shared by {#write_results} and {#rewrite_flow_annotated_units} so both
|
|
757
2249
|
# emit the index from the authoritative in-memory `@results` rather than
|
|
@@ -773,12 +2265,28 @@ module Woods
|
|
|
773
2265
|
|
|
774
2266
|
def write_dependency_graph
|
|
775
2267
|
graph_data = @dependency_graph.to_h
|
|
776
|
-
|
|
2268
|
+
# Key-sorted for the same reason `to_h` sorts its own sections: the
|
|
2269
|
+
# digest published as `graph_sha` covers these bytes, so node
|
|
2270
|
+
# registration order must not reach it (B-180).
|
|
2271
|
+
graph_data[:pagerank] = @dependency_graph.pagerank.sort_by { |identifier, _| identifier }.to_h
|
|
2272
|
+
|
|
2273
|
+
payload = json_serialize(graph_data)
|
|
2274
|
+
# The bytes about to land on disk are the bytes `graph_sha` covers, so
|
|
2275
|
+
# keep the digest here rather than reading a whole large-app graph back
|
|
2276
|
+
# to compute it. AtomicFile writes in binary mode and reads back as
|
|
2277
|
+
# UTF-8, so the two digests are the same either way.
|
|
2278
|
+
@graph_sha = Digest::SHA256.hexdigest(payload)
|
|
2279
|
+
AtomicFile.write(payload_dir.join('dependency_graph.json'), payload, durable: payload_writes_durable?)
|
|
2280
|
+
end
|
|
777
2281
|
|
|
778
|
-
|
|
779
|
-
|
|
780
|
-
|
|
781
|
-
|
|
2282
|
+
# The digest of the dependency graph this run wrote.
|
|
2283
|
+
#
|
|
2284
|
+
# Falls back to reading the file for a caller that writes the analysis
|
|
2285
|
+
# without having written the graph in the same run.
|
|
2286
|
+
#
|
|
2287
|
+
# @return [String] hex SHA256 of dependency_graph.json
|
|
2288
|
+
def graph_sha
|
|
2289
|
+
@graph_sha || Digest::SHA256.hexdigest(AtomicFile.read(payload_dir.join('dependency_graph.json')))
|
|
782
2290
|
end
|
|
783
2291
|
|
|
784
2292
|
def write_graph_analysis
|
|
@@ -786,14 +2294,13 @@ module Woods
|
|
|
786
2294
|
|
|
787
2295
|
enriched = @graph_analysis.merge(
|
|
788
2296
|
generated_at: Time.current.iso8601,
|
|
789
|
-
graph_sha:
|
|
790
|
-
File.read(@output_dir.join('dependency_graph.json'))
|
|
791
|
-
)
|
|
2297
|
+
graph_sha: graph_sha
|
|
792
2298
|
)
|
|
793
2299
|
|
|
794
|
-
|
|
795
|
-
|
|
796
|
-
json_serialize(enriched)
|
|
2300
|
+
AtomicFile.write(
|
|
2301
|
+
payload_dir.join('graph_analysis.json'),
|
|
2302
|
+
json_serialize(enriched),
|
|
2303
|
+
durable: payload_writes_durable?
|
|
797
2304
|
)
|
|
798
2305
|
end
|
|
799
2306
|
|
|
@@ -836,34 +2343,57 @@ module Woods
|
|
|
836
2343
|
schema_sha: schema_sha
|
|
837
2344
|
}
|
|
838
2345
|
|
|
839
|
-
|
|
840
|
-
|
|
841
|
-
json_serialize(manifest)
|
|
2346
|
+
AtomicFile.write(
|
|
2347
|
+
payload_dir.join('manifest.json'),
|
|
2348
|
+
json_serialize(manifest),
|
|
2349
|
+
durable: payload_writes_durable?
|
|
842
2350
|
)
|
|
843
2351
|
end
|
|
844
2352
|
|
|
845
2353
|
# Unit and chunk counts derived from the per-type _index.json files on
|
|
846
|
-
# disk
|
|
2354
|
+
# disk: the source of truth after an incremental run, where only the
|
|
847
2355
|
# affected units were re-extracted.
|
|
848
2356
|
#
|
|
849
2357
|
# @return [Array(Hash{Symbol => Integer}, Integer)] counts by type, total chunk count
|
|
850
2358
|
def persisted_counts
|
|
851
|
-
|
|
852
|
-
|
|
2359
|
+
stats = persisted_index_stats
|
|
2360
|
+
[stats.transform_values { |type_stats| type_stats[:count] },
|
|
2361
|
+
stats.sum { |_type, type_stats| type_stats[:chunks] }]
|
|
2362
|
+
end
|
|
853
2363
|
|
|
854
|
-
|
|
855
|
-
|
|
856
|
-
|
|
857
|
-
|
|
2364
|
+
# One pass over the persisted per-type _index.json files.
|
|
2365
|
+
#
|
|
2366
|
+
# {#persisted_counts} feeds the manifest and {#persisted_summary_stats}
|
|
2367
|
+
# feeds SUMMARY.md, they run back to back at the end of every incremental
|
|
2368
|
+
# run, and each used to parse every type index for itself. Reading once and
|
|
2369
|
+
# deriving both is also what keeps the two artifacts from disagreeing about
|
|
2370
|
+
# totals, which was previously a property of them using the same source
|
|
2371
|
+
# rather than the same read.
|
|
2372
|
+
#
|
|
2373
|
+
# An unreadable index drops that whole type from the manifest and the
|
|
2374
|
+
# summary alike, with one warning rather than the two the separate passes
|
|
2375
|
+
# emitted. Sorted for determinism, since the glob order is the
|
|
2376
|
+
# filesystem's.
|
|
2377
|
+
#
|
|
2378
|
+
# Memoized per run: invalidated at the start of {#extract_all} and
|
|
2379
|
+
# {#prepare_incremental_run}, and again whenever {#regenerate_type_index}
|
|
2380
|
+
# rewrites an index underneath it.
|
|
2381
|
+
#
|
|
2382
|
+
# @return [Hash{Symbol => Hash}] type => `{ count:, chunks:, namespaces: }`
|
|
2383
|
+
def persisted_index_stats
|
|
2384
|
+
@persisted_index_stats ||= Dir[payload_dir.join('*/_index.json').to_s].each_with_object({}) do |path, stats|
|
|
2385
|
+
entries = JSON.parse(AtomicFile.read(path))
|
|
2386
|
+
stats[File.basename(File.dirname(path)).to_sym] = {
|
|
2387
|
+
count: entries.size,
|
|
2388
|
+
chunks: entries.sum { |entry| entry['chunk_count'].to_i },
|
|
2389
|
+
namespaces: namespace_histogram(entries.map { |entry| entry['namespace'] })
|
|
2390
|
+
}
|
|
858
2391
|
rescue JSON::ParserError => e
|
|
859
|
-
|
|
860
|
-
|
|
861
|
-
|
|
862
|
-
|
|
863
|
-
next
|
|
2392
|
+
type = File.basename(File.dirname(path))
|
|
2393
|
+
Rails.logger.warn(
|
|
2394
|
+
"[Woods] Skipping unreadable #{type}/_index.json in manifest counts and summary totals: #{e.message}"
|
|
2395
|
+
)
|
|
864
2396
|
end
|
|
865
|
-
|
|
866
|
-
[counts, chunks]
|
|
867
2397
|
end
|
|
868
2398
|
|
|
869
2399
|
# Capture a temporal snapshot after extraction completes.
|
|
@@ -876,10 +2406,10 @@ module Woods
|
|
|
876
2406
|
def capture_snapshot
|
|
877
2407
|
return unless Woods.configuration.enable_snapshots
|
|
878
2408
|
|
|
879
|
-
manifest_path =
|
|
2409
|
+
manifest_path = payload_dir.join('manifest.json')
|
|
880
2410
|
return unless manifest_path.exist?
|
|
881
2411
|
|
|
882
|
-
manifest = JSON.parse(
|
|
2412
|
+
manifest = JSON.parse(AtomicFile.read(manifest_path))
|
|
883
2413
|
# Snapshots are keyed on the commit SHA — an unresolvable provenance
|
|
884
2414
|
# ("unknown", see GitProvenance/#137) must not key or collide a snapshot.
|
|
885
2415
|
git_sha = manifest['git_sha']
|
|
@@ -932,32 +2462,39 @@ module Woods
|
|
|
932
2462
|
# category with count and top-5 namespace breakdown, rather than enumerating
|
|
933
2463
|
# every unit. Per-unit detail is available in the per-category _index.json files.
|
|
934
2464
|
#
|
|
2465
|
+
# The full path derives its numbers from the in-memory results. An
|
|
2466
|
+
# incremental run holds only changed units in memory, so it derives the
|
|
2467
|
+
# same shape from the persisted per-type _index.json files — the same
|
|
2468
|
+
# source {#persisted_counts} feeds the manifest from, which is what keeps
|
|
2469
|
+
# SUMMARY.md's totals agreeing with manifest.json. (M4: the hardlinked
|
|
2470
|
+
# previous generation's summary this path used to leave in place carried
|
|
2471
|
+
# stale totals after any run that added or removed units.) The `Generated:`
|
|
2472
|
+
# stamp keeps its meaning either way: it names the moment the summary was
|
|
2473
|
+
# written.
|
|
2474
|
+
#
|
|
935
2475
|
# @return [void]
|
|
936
2476
|
def write_structural_summary
|
|
937
|
-
|
|
2477
|
+
stats = @results.empty? ? persisted_summary_stats : results_summary_stats
|
|
2478
|
+
return unless stats
|
|
938
2479
|
|
|
939
|
-
total_units =
|
|
940
|
-
|
|
941
|
-
|
|
2480
|
+
total_units = stats.sum { |_, s| s[:count] }
|
|
2481
|
+
# Matches the manifest's count (`write_manifest`/`persisted_counts`
|
|
2482
|
+
# both sum chunk counts directly) — the previous `[size, 1].max`
|
|
2483
|
+
# floor made SUMMARY.md disagree with manifest.json for every
|
|
2484
|
+
# unchunked unit.
|
|
2485
|
+
total_chunks = stats.sum { |_, s| s[:chunks] }
|
|
942
2486
|
|
|
943
2487
|
summary = []
|
|
944
2488
|
summary << '# Codebase Index Summary'
|
|
945
2489
|
summary << "Generated: #{Time.current.iso8601}"
|
|
946
2490
|
summary << "Rails #{Rails.version} / Ruby #{RUBY_VERSION}"
|
|
947
|
-
summary << "Units: #{total_units} | Chunks: #{total_chunks} | Categories: #{
|
|
2491
|
+
summary << "Units: #{total_units} | Chunks: #{total_chunks} | Categories: #{stats.size}"
|
|
948
2492
|
summary << ''
|
|
949
2493
|
|
|
950
|
-
|
|
951
|
-
|
|
952
|
-
|
|
953
|
-
summary << "## #{type.to_s.titleize} (#{units.size})"
|
|
954
|
-
|
|
955
|
-
ns_counts = units
|
|
956
|
-
.group_by { |u| u.namespace.nil? || u.namespace.empty? ? '(root)' : u.namespace }
|
|
957
|
-
.transform_values(&:size)
|
|
958
|
-
.sort_by { |_, count| -count }
|
|
959
|
-
.first(5)
|
|
2494
|
+
stats.each do |type, s|
|
|
2495
|
+
summary << "## #{type.to_s.titleize} (#{s[:count]})"
|
|
960
2496
|
|
|
2497
|
+
ns_counts = s[:namespaces].sort_by { |_, count| -count }.first(5)
|
|
961
2498
|
ns_parts = ns_counts.map { |ns, count| "#{ns} #{count}" }
|
|
962
2499
|
summary << "Namespaces: #{ns_parts.join(', ')}" unless ns_parts.empty?
|
|
963
2500
|
summary << ''
|
|
@@ -979,25 +2516,75 @@ module Woods
|
|
|
979
2516
|
hub_names = significant_hubs.map { |h| h[:identifier] }.join(', ')
|
|
980
2517
|
summary << "- Hub nodes (>20 dependents): #{hub_names}"
|
|
981
2518
|
end
|
|
2519
|
+
|
|
2520
|
+
volatile = Array(@graph_analysis[:volatile_dependencies]).first(5)
|
|
2521
|
+
if volatile.any?
|
|
2522
|
+
lines = volatile.map { |v| "#{v[:from]} -> #{v[:to]} (#{v[:from_commits]} vs #{v[:to_commits]} commits)" }
|
|
2523
|
+
summary << "- Volatile dependencies (top #{lines.size}): #{lines.join('; ')}"
|
|
2524
|
+
end
|
|
982
2525
|
end
|
|
983
2526
|
|
|
984
2527
|
summary << ''
|
|
985
2528
|
|
|
986
|
-
|
|
987
|
-
|
|
988
|
-
summary.join("\n")
|
|
2529
|
+
AtomicFile.write(
|
|
2530
|
+
payload_dir.join('SUMMARY.md'),
|
|
2531
|
+
summary.join("\n"),
|
|
2532
|
+
durable: payload_writes_durable?
|
|
989
2533
|
)
|
|
990
2534
|
end
|
|
991
2535
|
|
|
2536
|
+
# Per-type `{ count:, chunks:, namespaces: }` for {#write_structural_summary},
|
|
2537
|
+
# derived from the in-memory results a full extraction holds.
|
|
2538
|
+
#
|
|
2539
|
+
# @return [Hash{Symbol => Hash}]
|
|
2540
|
+
def results_summary_stats
|
|
2541
|
+
@results.each_with_object({}) do |(type, units), stats|
|
|
2542
|
+
next if units.empty?
|
|
2543
|
+
|
|
2544
|
+
stats[type] = {
|
|
2545
|
+
count: units.size,
|
|
2546
|
+
chunks: units.sum { |u| u.chunks.size },
|
|
2547
|
+
namespaces: namespace_histogram(units.map(&:namespace))
|
|
2548
|
+
}
|
|
2549
|
+
end
|
|
2550
|
+
end
|
|
2551
|
+
|
|
2552
|
+
# The incremental path's counterpart to {#results_summary_stats}: the same
|
|
2553
|
+
# shape, read back through {#persisted_index_stats}. A type whose index is
|
|
2554
|
+
# empty is left out, where the manifest still counts it as zero.
|
|
2555
|
+
#
|
|
2556
|
+
# @return [Hash{Symbol => Hash}, nil] nil when the payload holds no
|
|
2557
|
+
# non-empty type index, so a bare index writes no summary, matching the
|
|
2558
|
+
# full path's early return when it extracted nothing
|
|
2559
|
+
def persisted_summary_stats
|
|
2560
|
+
stats = persisted_index_stats.reject { |_type, type_stats| type_stats[:count].zero? }
|
|
2561
|
+
|
|
2562
|
+
stats.empty? ? nil : stats
|
|
2563
|
+
end
|
|
2564
|
+
|
|
2565
|
+
# Group namespaces for a SUMMARY.md section: a missing or empty namespace
|
|
2566
|
+
# is the root, everything else counts as-is.
|
|
2567
|
+
#
|
|
2568
|
+
# @param namespaces [Array<String, nil>]
|
|
2569
|
+
# @return [Hash{String => Integer}]
|
|
2570
|
+
def namespace_histogram(namespaces)
|
|
2571
|
+
namespaces.group_by { |ns| ns.nil? || ns.empty? ? '(root)' : ns }
|
|
2572
|
+
.transform_values(&:size)
|
|
2573
|
+
end
|
|
2574
|
+
|
|
992
2575
|
def regenerate_type_index(type_key)
|
|
993
|
-
type_dir =
|
|
2576
|
+
type_dir = payload_dir.join(type_key.to_s)
|
|
994
2577
|
return unless type_dir.directory?
|
|
995
2578
|
|
|
2579
|
+
# This run's counts and summary totals are read from the index files;
|
|
2580
|
+
# rewriting one invalidates whatever was read before.
|
|
2581
|
+
@persisted_index_stats = nil
|
|
2582
|
+
|
|
996
2583
|
# Scan existing unit JSON files (exclude _index.json)
|
|
997
2584
|
index = Dir[type_dir.join('*.json')].filter_map do |file|
|
|
998
2585
|
next if File.basename(file) == '_index.json'
|
|
999
2586
|
|
|
1000
|
-
data = JSON.parse(
|
|
2587
|
+
data = JSON.parse(AtomicFile.read(file))
|
|
1001
2588
|
{
|
|
1002
2589
|
identifier: data['identifier'],
|
|
1003
2590
|
file_path: data['file_path'],
|
|
@@ -1010,15 +2597,25 @@ module Woods
|
|
|
1010
2597
|
}
|
|
1011
2598
|
end
|
|
1012
2599
|
|
|
1013
|
-
|
|
2600
|
+
AtomicFile.write(
|
|
1014
2601
|
type_dir.join('_index.json'),
|
|
1015
|
-
json_serialize(index)
|
|
2602
|
+
json_serialize(index),
|
|
2603
|
+
durable: payload_writes_durable?
|
|
1016
2604
|
)
|
|
1017
2605
|
end
|
|
1018
2606
|
|
|
1019
2607
|
# Token estimate for a unit parsed back from JSON, mirroring
|
|
1020
2608
|
# ExtractedUnit#estimated_tokens (see docs/TOKEN_BENCHMARK.md).
|
|
1021
2609
|
#
|
|
2610
|
+
# `JSON.generate`, not `Hash#to_json`: with ActiveSupport loaded the
|
|
2611
|
+
# latter applies HTML-safe escaping, rendering `>` as `\\u003e` — six
|
|
2612
|
+
# characters where the file on disk has one. A model with
|
|
2613
|
+
# `scope :recent, -> { ... }` in its metadata therefore estimated one
|
|
2614
|
+
# token higher here than {ExtractedUnit#estimated_tokens} did over the
|
|
2615
|
+
# same content, so a full and an incremental run disagreed on that unit's
|
|
2616
|
+
# `_index.json` entry (#164). Both sides now measure the serialization
|
|
2617
|
+
# {#json_serialize} actually writes.
|
|
2618
|
+
#
|
|
1022
2619
|
# @param data [Hash] Parsed unit JSON (string keys)
|
|
1023
2620
|
# @return [Integer]
|
|
1024
2621
|
def estimated_tokens_from(data)
|
|
@@ -1026,7 +2623,7 @@ module Woods
|
|
|
1026
2623
|
metadata = data['metadata'] || {}
|
|
1027
2624
|
|
|
1028
2625
|
source_tokens = source ? TokenUtils.estimate_tokens(source) : 0
|
|
1029
|
-
metadata_tokens = metadata.any? ? TokenUtils.estimate_tokens(metadata
|
|
2626
|
+
metadata_tokens = metadata.any? ? TokenUtils.estimate_tokens(JSON.generate(metadata)) : 0
|
|
1030
2627
|
source_tokens + metadata_tokens
|
|
1031
2628
|
end
|
|
1032
2629
|
|
|
@@ -1087,78 +2684,1026 @@ module Woods
|
|
|
1087
2684
|
# Incremental Re-extraction
|
|
1088
2685
|
# ──────────────────────────────────────────────────────────────────────
|
|
1089
2686
|
|
|
2687
|
+
# Extractor instances are reused across a single incremental run, the way
|
|
2688
|
+
# full extraction uses one instance per type. Several of them do real
|
|
2689
|
+
# setup work in `initialize` (route-helper maps, routes maps), so a fresh
|
|
2690
|
+
# instance per file would make an N-file change N times more expensive.
|
|
2691
|
+
#
|
|
2692
|
+
# @param key [Symbol] key into {EXTRACTORS}
|
|
2693
|
+
# @return [Object, nil] the memoized extractor instance
|
|
2694
|
+
def extractor_for(key)
|
|
2695
|
+
@incremental_extractors ||= {}
|
|
2696
|
+
return @incremental_extractors[key] if @incremental_extractors.key?(key)
|
|
2697
|
+
|
|
2698
|
+
@incremental_extractors[key] = EXTRACTORS[key]&.new
|
|
2699
|
+
rescue StandardError => e
|
|
2700
|
+
Rails.logger.warn "[Woods] Could not build #{key} extractor: #{e.message}"
|
|
2701
|
+
@incremental_extractors[key] = nil
|
|
2702
|
+
end
|
|
2703
|
+
|
|
2704
|
+
# ActiveRecord model names, needed by PoroExtractor to tell a plain class
|
|
2705
|
+
# under app/models apart from a persisted model. Computed once per run.
|
|
2706
|
+
#
|
|
2707
|
+
# @return [Set<String>]
|
|
2708
|
+
def active_record_names
|
|
2709
|
+
@active_record_names ||=
|
|
2710
|
+
if defined?(ActiveRecord::Base)
|
|
2711
|
+
ActiveRecord::Base.descendants.filter_map(&:name).to_set
|
|
2712
|
+
else
|
|
2713
|
+
Set.new
|
|
2714
|
+
end
|
|
2715
|
+
end
|
|
2716
|
+
|
|
2717
|
+
# Re-extract every file-based unit defined by the changed paths that still
|
|
2718
|
+
# exist on disk, and prune the ones those paths no longer define.
|
|
2719
|
+
#
|
|
2720
|
+
# This is the fix for #164 gap 1 (a path the index has never seen routed
|
|
2721
|
+
# nowhere, so new files were silently ignored) and the per-path half of
|
|
2722
|
+
# gap 3 (a file that defines several units — a `.rake` file with multiple
|
|
2723
|
+
# tasks, an i18n YAML — could only ever resolve to one identifier).
|
|
2724
|
+
# Reconciling the whole path at once means a task deleted from a
|
|
2725
|
+
# multi-task file is removed rather than left behind.
|
|
2726
|
+
#
|
|
2727
|
+
# Removal is scoped to the unit types the matching rules could have
|
|
2728
|
+
# produced, so a class-based unit sharing the path (the `User` model unit
|
|
2729
|
+
# for `app/models/user.rb`) is never collaterally deleted here.
|
|
2730
|
+
#
|
|
2731
|
+
# @param change_set [ChangeSet]
|
|
2732
|
+
# @param affected_types [Set<Symbol>] out-param of touched extractor keys
|
|
2733
|
+
# @return [Set<String>] identifiers written or removed
|
|
2734
|
+
def reconcile_changed_paths(change_set, affected_types)
|
|
2735
|
+
dispatcher = PathDispatcher.new
|
|
2736
|
+
touched = Set.new
|
|
2737
|
+
|
|
2738
|
+
change_set.existing_paths.each do |absolute_path|
|
|
2739
|
+
rules = dispatcher.file_rules_for(change_set.relativize(absolute_path))
|
|
2740
|
+
next if rules.empty?
|
|
2741
|
+
|
|
2742
|
+
produced = Set.new
|
|
2743
|
+
# A rule whose extraction *raised* tells us nothing about what the path
|
|
2744
|
+
# defines, so it must not license pruning what the path defined before.
|
|
2745
|
+
raised = false
|
|
2746
|
+
rules.each do |rule|
|
|
2747
|
+
units = extract_with_rule(rule, absolute_path)
|
|
2748
|
+
if units.nil?
|
|
2749
|
+
raised = true
|
|
2750
|
+
next
|
|
2751
|
+
end
|
|
2752
|
+
|
|
2753
|
+
# (identifier, type) pairs, not bare identifiers: two rules can
|
|
2754
|
+
# claim one path (policies/pundit_policies) and mint the same
|
|
2755
|
+
# identifier for different unit types. When one of them stops
|
|
2756
|
+
# producing, an identifier-keyed set would let the survivor shield
|
|
2757
|
+
# the stale sibling-type node from the prune below (the #225 shape,
|
|
2758
|
+
# one method over — CORE-1).
|
|
2759
|
+
produced.merge(units.map { |unit| [unit.identifier, unit.type] })
|
|
2760
|
+
touched.merge(register_and_write(rule.extractor_key, units, affected_types))
|
|
2761
|
+
end
|
|
2762
|
+
|
|
2763
|
+
next if raised
|
|
2764
|
+
|
|
2765
|
+
touched.merge(
|
|
2766
|
+
prune_path_leftovers(absolute_path, rules, produced, affected_types)
|
|
2767
|
+
)
|
|
2768
|
+
end
|
|
2769
|
+
|
|
2770
|
+
touched
|
|
2771
|
+
end
|
|
2772
|
+
|
|
2773
|
+
# Run one {PathDispatcher::Rule} against one file.
|
|
2774
|
+
#
|
|
2775
|
+
# @param rule [PathDispatcher::Rule]
|
|
2776
|
+
# @param absolute_path [String]
|
|
2777
|
+
# @return [Array<ExtractedUnit>, nil] the units the path defines, or nil
|
|
2778
|
+
# when the attempt says nothing about what it defines (see the rescue)
|
|
2779
|
+
def extract_with_rule(rule, absolute_path)
|
|
2780
|
+
extractor = extractor_for(rule.extractor_key)
|
|
2781
|
+
# nil, not [] — the same distinction the rescue below defends (#198).
|
|
2782
|
+
# {#extractor_for} rescues a raising *constructor* and memoizes nil, and
|
|
2783
|
+
# nil fails the respond_to? guard — so a failed construction used to fall
|
|
2784
|
+
# through to the [] "this path defines nothing any more" answer, and one
|
|
2785
|
+
# broken constructor licensed pruning every previously-registered unit
|
|
2786
|
+
# on every changed path of that type, with the generation bumped over
|
|
2787
|
+
# the loss. Construction failure tells us nothing about the path; only
|
|
2788
|
+
# a genuinely constructed extractor that lacks the method earns the [].
|
|
2789
|
+
return nil if extractor.nil?
|
|
2790
|
+
return [] unless extractor.respond_to?(rule.method_name)
|
|
2791
|
+
|
|
2792
|
+
result =
|
|
2793
|
+
if rule.extractor_key == :poros
|
|
2794
|
+
# PoroExtractor needs the AR name set to reject persisted models;
|
|
2795
|
+
# its default is an empty set, which would misfile every model
|
|
2796
|
+
# under app/models as a PORO on the incremental path.
|
|
2797
|
+
extractor.public_send(rule.method_name, absolute_path, ar_names: active_record_names)
|
|
2798
|
+
else
|
|
2799
|
+
extractor.public_send(rule.method_name, absolute_path)
|
|
2800
|
+
end
|
|
2801
|
+
|
|
2802
|
+
Array(result).compact
|
|
2803
|
+
rescue StandardError => e
|
|
2804
|
+
Rails.logger.warn "[Woods] #{rule.extractor_key} re-extraction of #{absolute_path} failed: #{e.message}"
|
|
2805
|
+
# `nil`, not `[]`. The caller treats an empty result as "this path defines
|
|
2806
|
+
# nothing any more" and prunes the units previously registered to it — so
|
|
2807
|
+
# returning [] here turns a *transient* failure (a watcher batch catching
|
|
2808
|
+
# an editor mid-write, an encoding hiccup) into silent deletion of a
|
|
2809
|
+
# perfectly good unit, which nothing restores until that file is next
|
|
2810
|
+
# touched. "Extraction raised" and "extracted successfully, defines
|
|
2811
|
+
# nothing" are different answers and only the second licenses a prune.
|
|
2812
|
+
nil
|
|
2813
|
+
end
|
|
2814
|
+
|
|
2815
|
+
# Remove units the graph still attributes to a path that the path no
|
|
2816
|
+
# longer produces — but only for the types the matching rules cover.
|
|
2817
|
+
#
|
|
2818
|
+
# @param absolute_path [String]
|
|
2819
|
+
# @param rules [Array<PathDispatcher::Rule>]
|
|
2820
|
+
# @param produced [Set<Array(String, Symbol)>] (identifier, type) pairs
|
|
2821
|
+
# just extracted from the path
|
|
2822
|
+
# @param affected_types [Set<Symbol>]
|
|
2823
|
+
# @return [Set<String>] identifiers removed
|
|
2824
|
+
def prune_path_leftovers(absolute_path, rules, produced, affected_types)
|
|
2825
|
+
covered_keys = rules.to_set(&:extractor_key)
|
|
2826
|
+
removed = Set.new
|
|
2827
|
+
|
|
2828
|
+
@dependency_graph.units_for_path(absolute_path).each do |identifier, node_type|
|
|
2829
|
+
next if produced.include?([identifier, node_type])
|
|
2830
|
+
next unless covered_keys.include?(TYPE_TO_EXTRACTOR_KEY[node_type])
|
|
2831
|
+
|
|
2832
|
+
removed.add(identifier) if remove_unit(identifier, affected_types, type: node_type)
|
|
2833
|
+
end
|
|
2834
|
+
|
|
2835
|
+
removed
|
|
2836
|
+
end
|
|
2837
|
+
|
|
2838
|
+
# Reconcile class-based types against their runtime discovery sets.
|
|
2839
|
+
#
|
|
2840
|
+
# Models, controllers, mailers, components and channels are discovered by
|
|
2841
|
+
# walking descendants, not by globbing files, so a class added since the
|
|
2842
|
+
# last extraction is invisible to any path-based dispatch. Comparing each
|
|
2843
|
+
# extractor's `discoverable_classes` against the identifiers already in
|
|
2844
|
+
# the graph finds exactly those additions, using the same discovery code
|
|
2845
|
+
# a full extraction uses.
|
|
2846
|
+
#
|
|
2847
|
+
# Removals are handled here too, but only against a *complete* discovery
|
|
2848
|
+
# set — see {#stale_class_based_units} for why that qualifier carries the
|
|
2849
|
+
# whole safety argument.
|
|
2850
|
+
#
|
|
2851
|
+
# @param affected_types [Set<Symbol>]
|
|
2852
|
+
# @return [Set<String>] identifiers added or removed
|
|
2853
|
+
def reconcile_class_based_types(affected_types, except: nil)
|
|
2854
|
+
touched = Set.new
|
|
2855
|
+
excluded = except&.to_set || Set.new
|
|
2856
|
+
|
|
2857
|
+
CLASS_BASED_DISCOVERY.each do |key, spec|
|
|
2858
|
+
extractor = extractor_for(key)
|
|
2859
|
+
next unless extractor.respond_to?(:discoverable_classes)
|
|
2860
|
+
|
|
2861
|
+
discovered = extractor.discoverable_classes.reject { |k| k.name.nil? }
|
|
2862
|
+
known = Array(spec[:types] || spec[:type])
|
|
2863
|
+
.flat_map { |type| @dependency_graph.units_of_type(type) }.to_set
|
|
2864
|
+
|
|
2865
|
+
touched.merge(add_discovered_classes(key, spec, discovered, known, excluded, affected_types))
|
|
2866
|
+
next if spec[:reconcile_removals] == false
|
|
2867
|
+
|
|
2868
|
+
touched.merge(remove_stale_classes(spec, discovered, known, affected_types))
|
|
2869
|
+
end
|
|
2870
|
+
|
|
2871
|
+
touched
|
|
2872
|
+
end
|
|
2873
|
+
|
|
2874
|
+
# Pruned class-based identifiers the tree still governs, and that the
|
|
2875
|
+
# second reconciliation pass may therefore re-add.
|
|
2876
|
+
#
|
|
2877
|
+
# The class-based *move* shape — the file moved, the constant did not
|
|
2878
|
+
# (M1) — prunes the unit for the vanished old path while the class stays
|
|
2879
|
+
# in the discovery set; the move target is a changed file the active
|
|
2880
|
+
# loader governs for exactly that constant, so re-adding produces the
|
|
2881
|
+
# unit a full extraction produces. The *deletion* shape must stay
|
|
2882
|
+
# pruned: without a reload a constant outlives the file that defined
|
|
2883
|
+
# it, so liveness alone proves nothing.
|
|
2884
|
+
#
|
|
2885
|
+
# Identity is loader-derived, not textual: {SourceNesting#
|
|
2886
|
+
# governed_class_name} returns the constant path the active Zeitwerk
|
|
2887
|
+
# loader expects the changed file to define (its inflector, ignores,
|
|
2888
|
+
# and root namespaces decide, via +cpath_expected_at+), gated on the
|
|
2889
|
+
# file actually declaring it. A loader non-claim — an unmanaged or
|
|
2890
|
+
# declined path — is authoritative and re-adds nothing, exactly the
|
|
2891
|
+
# governed-naming contract. That kills the two resurrection shapes a
|
|
2892
|
+
# demodulized class-name regex allowed: a changed file declaring
|
|
2893
|
+
# `Public::User` no longer resurrects a pruned `Admin::User`, and a
|
|
2894
|
+
# file whose comments or string literals mention `class User` no
|
|
2895
|
+
# longer resurrects anything.
|
|
2896
|
+
#
|
|
2897
|
+
# @param pruned [Set<String>] identifiers removed by {#prune_vanished_units}
|
|
2898
|
+
# @param change_set [ChangeSet]
|
|
2899
|
+
# @return [Set<String>] identifiers safe to re-add this run
|
|
2900
|
+
def readdable_pruned_classes(pruned, change_set)
|
|
2901
|
+
readdable = Set.new
|
|
2902
|
+
return readdable if pruned.empty?
|
|
2903
|
+
|
|
2904
|
+
claimed = changed_governed_names(change_set)
|
|
2905
|
+
return readdable if claimed.empty?
|
|
2906
|
+
|
|
2907
|
+
CLASS_BASED_DISCOVERY.each_key do |key|
|
|
2908
|
+
extractor = extractor_for(key)
|
|
2909
|
+
next unless extractor.respond_to?(:discoverable_classes)
|
|
2910
|
+
|
|
2911
|
+
extractor.discoverable_classes.each do |klass|
|
|
2912
|
+
next if klass.name.nil? || !pruned.include?(klass.name)
|
|
2913
|
+
|
|
2914
|
+
readdable.add(klass.name) if claimed.include?(klass.name)
|
|
2915
|
+
end
|
|
2916
|
+
end
|
|
2917
|
+
|
|
2918
|
+
readdable
|
|
2919
|
+
rescue StandardError => e
|
|
2920
|
+
# A failure here must not turn a deletion into a resurrection: the
|
|
2921
|
+
# empty set keeps every pruned identifier excluded, which is the
|
|
2922
|
+
# pre-M1 behavior.
|
|
2923
|
+
Rails.logger.warn "[Woods] Could not determine re-addable pruned classes: #{e.message}"
|
|
2924
|
+
Set.new
|
|
2925
|
+
end
|
|
2926
|
+
|
|
2927
|
+
# Governed constant names of the change set's still-existing Ruby
|
|
2928
|
+
# files. The shared governed-naming helper consults the active Zeitwerk
|
|
2929
|
+
# loader and returns nil for anything it does not claim, so unmanaged
|
|
2930
|
+
# paths (anything outside the loader's roots, under an old Zeitwerk
|
|
2931
|
+
# without +cpath_expected_at+, or a declined file) contribute nothing.
|
|
2932
|
+
#
|
|
2933
|
+
# @param change_set [ChangeSet]
|
|
2934
|
+
# @return [Set<String>] governed constant names of changed files
|
|
2935
|
+
def changed_governed_names(change_set)
|
|
2936
|
+
change_set.existing_paths.filter_map do |path|
|
|
2937
|
+
path = path.to_s
|
|
2938
|
+
next unless path.end_with?('.rb')
|
|
2939
|
+
|
|
2940
|
+
governed_class_name(path, File.read(path))
|
|
2941
|
+
end.compact.to_set
|
|
2942
|
+
end
|
|
2943
|
+
|
|
2944
|
+
def add_discovered_classes(key, spec, discovered, known, excluded, affected_types)
|
|
2945
|
+
new_classes = discovered.reject { |k| known.include?(k.name) || excluded.include?(k.name) }
|
|
2946
|
+
return Set.new if new_classes.empty?
|
|
2947
|
+
|
|
2948
|
+
units = new_classes.filter_map do |klass|
|
|
2949
|
+
extractor_for(key).public_send(spec[:method], klass)
|
|
2950
|
+
rescue StandardError => e
|
|
2951
|
+
Rails.logger.warn "[Woods] #{key} extraction of #{klass} failed: #{e.message}"
|
|
2952
|
+
nil
|
|
2953
|
+
end
|
|
2954
|
+
|
|
2955
|
+
register_and_write(key, units, affected_types)
|
|
2956
|
+
end
|
|
2957
|
+
|
|
2958
|
+
def remove_stale_classes(spec, discovered, known, affected_types)
|
|
2959
|
+
stale = stale_class_based_units(spec[:type], discovered, known)
|
|
2960
|
+
return Set.new if stale.empty?
|
|
2961
|
+
|
|
2962
|
+
Rails.logger.info "[Woods] removing #{stale.size} #{spec[:type]} unit(s) whose class no longer exists"
|
|
2963
|
+
stale.each_with_object(Set.new) do |identifier, removed|
|
|
2964
|
+
removed.add(identifier) if remove_unit(identifier, affected_types, type: spec[:type])
|
|
2965
|
+
end
|
|
2966
|
+
end
|
|
2967
|
+
|
|
2968
|
+
# Class-based units the graph still holds that a full extraction would not
|
|
2969
|
+
# produce.
|
|
2970
|
+
#
|
|
2971
|
+
# {#prune_vanished_units} keys on the source file being gone, which cannot
|
|
2972
|
+
# see this case: a class removed from a file that still exists leaves no
|
|
2973
|
+
# missing path, and class-based units register a *convention* path derived
|
|
2974
|
+
# from the constant name, so a class defined somewhere unconventional was
|
|
2975
|
+
# never attributed to the file it actually lived in. Two models in one
|
|
2976
|
+
# `.rb`, one of them deleted, and the survivor's own re-extraction says
|
|
2977
|
+
# nothing about the other. Nothing else in the run removes it, so it
|
|
2978
|
+
# outlives every subsequent incremental — a permanent divergence from a
|
|
2979
|
+
# full run, not a transient one.
|
|
2980
|
+
#
|
|
2981
|
+
# For all six class-based extractors `extract_all` is literally
|
|
2982
|
+
# `discoverable_classes.map { ... }.compact`, so absence from that set is
|
|
2983
|
+
# exactly "a full extraction would not produce this" — the equivalence the
|
|
2984
|
+
# incremental path is held to.
|
|
2985
|
+
#
|
|
2986
|
+
# The `@eager_load_complete` gate is the whole safety argument, and it is
|
|
2987
|
+
# why this is not simply the inverse of the addition pass:
|
|
2988
|
+
#
|
|
2989
|
+
# * **A partial eager load.** The documented NameError fallback loads only
|
|
2990
|
+
# `EXTRACTION_DIRECTORIES`, so descendants are known-incomplete and the
|
|
2991
|
+
# difference here would be most of the app. Deleting by the type is far
|
|
2992
|
+
# worse than a stale unit, so a partial load removes nothing.
|
|
2993
|
+
# * **A constant outliving its file.** A resident daemon that has not
|
|
2994
|
+
# reloaded still holds a deleted class as a descendant, so it is *in* the
|
|
2995
|
+
# set and not stale — which is correct for that process, and the
|
|
2996
|
+
# subsequent reload is what makes it removable.
|
|
2997
|
+
#
|
|
2998
|
+
# @param type [Symbol] unit type
|
|
2999
|
+
# @param discovered [Array<Class>] the extractor's current discovery set
|
|
3000
|
+
# @param known [Set<String>] identifiers of that type already in the graph
|
|
3001
|
+
# @return [Array<String>] identifiers to remove
|
|
3002
|
+
def stale_class_based_units(type, discovered, known)
|
|
3003
|
+
return [] unless @eager_load_complete
|
|
3004
|
+
|
|
3005
|
+
live = discovered.to_set(&:name)
|
|
3006
|
+
known.reject { |identifier| live.include?(identifier) }
|
|
3007
|
+
# Not redundant with `units_of_type`: an identifier can be listed in
|
|
3008
|
+
# this type's index while also naming a unit of another type, and
|
|
3009
|
+
# the caller removes by (identifier, type), so it has to be told
|
|
3010
|
+
# which node it is allowed to take.
|
|
3011
|
+
.select { |identifier| @dependency_graph.node_types(identifier).include?(type) }
|
|
3012
|
+
end
|
|
3013
|
+
|
|
3014
|
+
# Re-run whole-app extractors whose trigger paths changed, replacing that
|
|
3015
|
+
# unit type wholesale.
|
|
3016
|
+
#
|
|
3017
|
+
# These extractors have no per-file entry point — a route unit is derived
|
|
3018
|
+
# from `Rails.application.routes`, the middleware unit from the live
|
|
3019
|
+
# stack, events from a two-pass scan of all of `app/`. Before this,
|
|
3020
|
+
# incremental runs skipped them entirely, so a routes-only change
|
|
3021
|
+
# triggered a run that re-extracted nothing while still rewriting the
|
|
3022
|
+
# manifest (#164 gap 4). In an already-booted process re-running them is
|
|
3023
|
+
# cheap, which is what makes wholesale replacement the right shape.
|
|
3024
|
+
#
|
|
3025
|
+
# @param change_set [ChangeSet]
|
|
3026
|
+
# @param affected_types [Set<Symbol>]
|
|
3027
|
+
# @return [Set<String>] identifiers written or removed
|
|
3028
|
+
def rerun_whole_app_extractors(change_set, affected_types)
|
|
3029
|
+
keys = PathDispatcher.new.whole_app_keys_for_all(change_set.relative_paths)
|
|
3030
|
+
# The dispatch rules are configuration-blind (memoized per-process), so
|
|
3031
|
+
# the configuration gate applies here — a knob-off host must not re-run
|
|
3032
|
+
# an extractor whose units a full extraction of the same tree would not
|
|
3033
|
+
# produce. See {#skip_by_configuration?}.
|
|
3034
|
+
keys = keys.reject { |key| skip_by_configuration?(key) }.to_set
|
|
3035
|
+
return Set.new if keys.empty?
|
|
3036
|
+
|
|
3037
|
+
keys += ROUTE_CONSUMER_EXTRACTORS if keys.include?(:routes)
|
|
3038
|
+
# A routes re-run replaces every controller, and a flow document
|
|
3039
|
+
# carries the route itself, which no dependency edge connects to the
|
|
3040
|
+
# controller. Nothing about that is reachable by a graph walk, so the
|
|
3041
|
+
# run drops its flow scope and reassembles every touched controller.
|
|
3042
|
+
@flow_scope = nil if keys.include?(:routes)
|
|
3043
|
+
|
|
3044
|
+
keys.each_with_object(Set.new) do |key, touched|
|
|
3045
|
+
touched.merge(replace_type_wholesale(key, affected_types))
|
|
3046
|
+
end
|
|
3047
|
+
end
|
|
3048
|
+
|
|
3049
|
+
# Replace every unit an extractor owns with a fresh extraction.
|
|
3050
|
+
#
|
|
3051
|
+
# Units of the extractor's types that the fresh run no longer produces are
|
|
3052
|
+
# removed, which is what makes this a replacement rather than an upsert —
|
|
3053
|
+
# subject to the same eager-load gate the reconciler applies, see
|
|
3054
|
+
# {#remove_replaced_units}.
|
|
3055
|
+
#
|
|
3056
|
+
# Fail closed (M8): the rescue exists so a re-run that learned nothing —
|
|
3057
|
+
# `extract_all` itself raising — costs the run nothing. It must not swallow
|
|
3058
|
+
# a failure that lands AFTER the replacement started mutating state a
|
|
3059
|
+
# published generation would carry: {#register_and_write} registers the
|
|
3060
|
+
# graph node before writing the unit file, and {#remove_unit_of_type} rm_f's
|
|
3061
|
+
# the file before dropping the graph node, so a raise in either window
|
|
3062
|
+
# leaves the graph and the payload directory disagreeing. Swallowed, the run
|
|
3063
|
+
# went on to publish a generation whose dependency_graph.json held nodes
|
|
3064
|
+
# with no unit file — `dependencies`/`dependents` reported `found: true`
|
|
3065
|
+
# while lookup returned nil. The counter from {#note_wholesale_mutation}
|
|
3066
|
+
# turns any such failure into a re-raise: the run aborts before
|
|
3067
|
+
# {#publish_generation}, and the preceding generation stays resolved, the
|
|
3068
|
+
# same posture the flow family takes on the full path.
|
|
3069
|
+
#
|
|
3070
|
+
# Two counter rules make that decision sound. The marker is placed BEFORE
|
|
3071
|
+
# each mutation, because registration itself can fail mid-mutation (a
|
|
3072
|
+
# malformed dependency raises after the node is inserted) and a rm_f can
|
|
3073
|
+
# fail part-way; a marker placed after the fact would miss the window it
|
|
3074
|
+
# exists for, at the cost of a conservative abort when the marked mutation
|
|
3075
|
+
# then fails before changing anything. And the counter is reset BEFORE
|
|
3076
|
+
# `extract_all`, not after: {#register_and_write} is shared with the
|
|
3077
|
+
# reconcile paths that run earlier in the same pass, and one of those
|
|
3078
|
+
# leaving the counter positive must not turn a later, mutation-free
|
|
3079
|
+
# wholesale failure into an abort — only the current key's wholesale pass
|
|
3080
|
+
# contributes to the decision.
|
|
3081
|
+
#
|
|
3082
|
+
# @param key [Symbol] extractor key
|
|
3083
|
+
# @param affected_types [Set<Symbol>]
|
|
3084
|
+
# @return [Set<String>] identifiers written or removed
|
|
3085
|
+
# @raise [Woods::ExtractionError] when the replacement failed after
|
|
3086
|
+
# mutating durable state
|
|
3087
|
+
def replace_type_wholesale(key, affected_types)
|
|
3088
|
+
extractor = extractor_for(key)
|
|
3089
|
+
return Set.new unless extractor.respond_to?(:extract_all)
|
|
3090
|
+
|
|
3091
|
+
@wholesale_mutations = 0
|
|
3092
|
+
units = Array(extractor.extract_all).compact.uniq(&:identifier)
|
|
3093
|
+
Rails.logger.info "[Woods] Re-ran #{key} wholesale: #{units.size} units"
|
|
3094
|
+
|
|
3095
|
+
touched = register_and_write(key, units, affected_types)
|
|
3096
|
+
touched.merge(remove_replaced_units(key, units, affected_types))
|
|
3097
|
+
rescue StandardError => e
|
|
3098
|
+
if @wholesale_mutations.to_i.positive?
|
|
3099
|
+
raise Woods::ExtractionError, <<~MSG.tr("\n", ' ').strip
|
|
3100
|
+
Wholesale re-run of #{key} failed after this run had already
|
|
3101
|
+
registered, written, or removed #{@wholesale_mutations} unit
|
|
3102
|
+
artifact(s) (#{e.class}: #{e.message}). Continuing would publish a
|
|
3103
|
+
generation whose dependency graph disagrees with the unit files on
|
|
3104
|
+
disk, so the run aborts before publication and the previous
|
|
3105
|
+
generation stays resolved.
|
|
3106
|
+
MSG
|
|
3107
|
+
end
|
|
3108
|
+
|
|
3109
|
+
Rails.logger.error "[Woods] Wholesale re-run of #{key} failed: #{e.message}"
|
|
3110
|
+
Set.new
|
|
3111
|
+
end
|
|
3112
|
+
|
|
3113
|
+
# Record that the in-flight wholesale replacement is about to mutate state
|
|
3114
|
+
# a published generation would carry: a graph registration, a unit-file
|
|
3115
|
+
# write, or a unit-file removal. Callers mark BEFORE mutating — register
|
|
3116
|
+
# itself can raise mid-mutation (a malformed dependency raises after the
|
|
3117
|
+
# node is inserted), so a marker placed after the fact can miss the very
|
|
3118
|
+
# window it exists for. The cost is a conservative abort when the marked
|
|
3119
|
+
# mutation then fails before changing anything; that direction is safe.
|
|
3120
|
+
# {#replace_type_wholesale} resets the counter to 0 before `extract_all` —
|
|
3121
|
+
# so only the current key's wholesale pass contributes, never the earlier
|
|
3122
|
+
# reconcile passes that share {#register_and_write} — and re-raises from
|
|
3123
|
+
# its rescue when the count is non-zero. Increments from non-wholesale
|
|
3124
|
+
# callers of {#register_and_write} are inert outside that window: nothing
|
|
3125
|
+
# reads the counter between resets.
|
|
3126
|
+
#
|
|
3127
|
+
# @return [void]
|
|
3128
|
+
def note_wholesale_mutation
|
|
3129
|
+
@wholesale_mutations = @wholesale_mutations.to_i + 1
|
|
3130
|
+
end
|
|
3131
|
+
|
|
3132
|
+
# The removal half of a wholesale replacement: drop every unit of `key`'s
|
|
3133
|
+
# types that the fresh run no longer produced.
|
|
3134
|
+
#
|
|
3135
|
+
# Gated exactly the way {#stale_class_based_units} is, and for the same
|
|
3136
|
+
# reason (#198): for the extractors in {CLASS_BASED_DISCOVERY},
|
|
3137
|
+
# `extract_all` is `discoverable_classes.map { ... }` — a function of the
|
|
3138
|
+
# runtime descendants — so on the documented NameError-fallback boot the
|
|
3139
|
+
# fresh set is known-partial and "absent from it" does not mean "deleted".
|
|
3140
|
+
# The reconciler's gate ("the whole safety argument") never protected this
|
|
3141
|
+
# path, and {ROUTE_CONSUMER_EXTRACTORS} routes four descendants-discovered
|
|
3142
|
+
# types through here on *every* routes change — so one routes edit on a
|
|
3143
|
+
# fallback boot mass-deleted most controller units. Registration above is
|
|
3144
|
+
# not gated: additions are safe against a partial set (the same asymmetry
|
|
3145
|
+
# {#reconcile_class_based_types} relies on); only removal needs the
|
|
3146
|
+
# complete one.
|
|
3147
|
+
#
|
|
3148
|
+
# File-derived whole-app types (routes, view_templates, events, ...) keep
|
|
3149
|
+
# unconditional removal — their `extract_all` globs files or introspects
|
|
3150
|
+
# structures that do not depend on eager loading, so their fresh set is
|
|
3151
|
+
# authoritative on any boot.
|
|
3152
|
+
#
|
|
3153
|
+
# The skip is warned, not silent: until a run with a clean boot removes
|
|
3154
|
+
# them, the type may hold stale units, and an operator chasing a ghost
|
|
3155
|
+
# unit needs to know which run declined to delete it and why.
|
|
3156
|
+
#
|
|
3157
|
+
# `fresh` is computed PER unit_type, not once for the whole extractor key
|
|
3158
|
+
# (#225). An identifier is not unique across types — a Scenic view and a
|
|
3159
|
+
# factory can both be named `reports` — so a single identifier-level
|
|
3160
|
+
# `fresh` set let a factories re-run's `database_view` deletion pass see
|
|
3161
|
+
# the surviving `factories` identifier as "still fresh" and skip removing
|
|
3162
|
+
# nothing, then later delete both nodes anyway once `remove_unit` was
|
|
3163
|
+
# called without `type:` (the same call also had to be fixed — see
|
|
3164
|
+
# below). A per-type set also covers a unit reclassified between two
|
|
3165
|
+
# types this same key owns (GraphQL's four): the old type's identifier is
|
|
3166
|
+
# absent from ITS OWN fresh set even though the identifier as a whole is
|
|
3167
|
+
# still "fresh" under the new type, so the stale old-type node is
|
|
3168
|
+
# correctly dropped instead of surviving because the identifier still
|
|
3169
|
+
# exists somewhere in the extractor's output.
|
|
3170
|
+
#
|
|
3171
|
+
# `remove_unit` is called WITH `type: unit_type` for the same reason:
|
|
3172
|
+
# `DependencyGraph#remove`'s own doc warns that a typeless removal fans
|
|
3173
|
+
# over every type registered under the identifier, so calling it here
|
|
3174
|
+
# without `type:` deleted the sibling type's node and JSON file too —
|
|
3175
|
+
# reproduced live as a factories re-run deleting a same-named Scenic-view
|
|
3176
|
+
# unit.
|
|
3177
|
+
#
|
|
3178
|
+
# @param key [Symbol] extractor key
|
|
3179
|
+
# @param units [Array<ExtractedUnit>] the fresh extraction
|
|
3180
|
+
# @param affected_types [Set<Symbol>]
|
|
3181
|
+
# @return [Set<String>] identifiers removed
|
|
3182
|
+
def remove_replaced_units(key, units, affected_types)
|
|
3183
|
+
if CLASS_BASED_DISCOVERY.key?(key) && !@eager_load_complete
|
|
3184
|
+
Rails.logger.warn(
|
|
3185
|
+
"[Woods] Skipping stale-unit removal for #{key}: the eager load was incomplete, " \
|
|
3186
|
+
'so its discovery set is known-partial — the type may hold stale units until a clean boot'
|
|
3187
|
+
)
|
|
3188
|
+
return Set.new
|
|
3189
|
+
end
|
|
3190
|
+
|
|
3191
|
+
EXTRACTOR_KEY_TO_TYPES.fetch(key, []).each_with_object(Set.new) do |unit_type, removed|
|
|
3192
|
+
fresh = units.select { |u| u.type == unit_type }.to_set(&:identifier)
|
|
3193
|
+
(@dependency_graph.units_of_type(unit_type) - fresh.to_a).each do |stale|
|
|
3194
|
+
removed.add(stale) if remove_unit(stale, affected_types, type: unit_type)
|
|
3195
|
+
end
|
|
3196
|
+
end
|
|
3197
|
+
end
|
|
3198
|
+
|
|
3199
|
+
# Prune units whose source file no longer exists (#164 gap 2).
|
|
3200
|
+
#
|
|
3201
|
+
# Two inputs, with deliberately different authority:
|
|
3202
|
+
#
|
|
3203
|
+
# * **The change set.** A path the caller reports as changed which is no
|
|
3204
|
+
# longer on disk is authoritative — every unit the graph attributes to
|
|
3205
|
+
# it goes, whatever its type. This is the path that handles a deleted
|
|
3206
|
+
# model or controller, and the old side of a rename (git's
|
|
3207
|
+
# `--no-renames` semantics).
|
|
3208
|
+
# * **A sweep** over registered paths, for callers whose change set is
|
|
3209
|
+
# incomplete: a git diff that omits deletions, a watcher that missed an
|
|
3210
|
+
# unlink, a branch switch. The sweep is a heuristic, so it is bounded
|
|
3211
|
+
# twice — see {#sweep_candidates} and the class-based exclusion below.
|
|
3212
|
+
#
|
|
3213
|
+
# Only paths under `Rails.root` are considered either way. Framework
|
|
3214
|
+
# units point at gem paths, and an index restored from a CI artifact can
|
|
3215
|
+
# carry paths produced under a different root; neither is a deletion.
|
|
3216
|
+
#
|
|
3217
|
+
# @param change_set [ChangeSet]
|
|
3218
|
+
# @param affected_types [Set<Symbol>]
|
|
3219
|
+
# @return [Set<String>] identifiers removed
|
|
3220
|
+
def prune_vanished_units(change_set, affected_types)
|
|
3221
|
+
removed = prune_paths(change_set.missing_paths, affected_types, class_based: true)
|
|
3222
|
+
removed.merge(prune_paths(sweep_candidates(change_set), affected_types, class_based: false))
|
|
3223
|
+
|
|
3224
|
+
Rails.logger.info "[Woods] Pruned #{removed.size} unit(s) whose source file is gone" if removed.any?
|
|
3225
|
+
removed
|
|
3226
|
+
end
|
|
3227
|
+
|
|
3228
|
+
# Registered paths that have vanished and that the sweep is allowed to act
|
|
3229
|
+
# on: under Rails.root, gone from disk, and claimed by a file dispatch
|
|
3230
|
+
# rule.
|
|
3231
|
+
#
|
|
3232
|
+
# The file-rule bound exists because some units point at a *nominal* path
|
|
3233
|
+
# rather than a source file — `BehavioralProfile` names
|
|
3234
|
+
# `config/application.rb`, which no rule claims — and sweeping those would
|
|
3235
|
+
# delete units a full extraction still produces.
|
|
3236
|
+
#
|
|
3237
|
+
# @param change_set [ChangeSet]
|
|
3238
|
+
# @return [Array<String>] absolute paths
|
|
3239
|
+
def sweep_candidates(change_set)
|
|
3240
|
+
root_prefix = "#{Rails.root}/"
|
|
3241
|
+
dispatcher = PathDispatcher.new
|
|
3242
|
+
already_named = change_set.missing_paths.to_set
|
|
3243
|
+
|
|
3244
|
+
@dependency_graph.registered_paths.reject do |path|
|
|
3245
|
+
already_named.include?(path) ||
|
|
3246
|
+
!path.to_s.start_with?(root_prefix) ||
|
|
3247
|
+
File.exist?(path) ||
|
|
3248
|
+
dispatcher.file_rules_for(change_set.relativize(path)).empty?
|
|
3249
|
+
end
|
|
3250
|
+
end
|
|
3251
|
+
|
|
3252
|
+
# Remove every unit the graph attributes to each of `paths`.
|
|
3253
|
+
#
|
|
3254
|
+
# @param paths [Enumerable<String>] absolute paths believed to be gone
|
|
3255
|
+
# @param affected_types [Set<Symbol>]
|
|
3256
|
+
# @param class_based [Boolean] whether class-based units may be pruned
|
|
3257
|
+
# @return [Set<String>] identifiers removed
|
|
3258
|
+
def prune_paths(paths, affected_types, class_based:)
|
|
3259
|
+
root_prefix = "#{Rails.root}/"
|
|
3260
|
+
|
|
3261
|
+
paths.each_with_object(Set.new) do |path, removed|
|
|
3262
|
+
next unless path.to_s.start_with?(root_prefix)
|
|
3263
|
+
next if File.exist?(path)
|
|
3264
|
+
|
|
3265
|
+
@dependency_graph.units_for_path(path).each do |identifier, type|
|
|
3266
|
+
next if !class_based && convention_path_unit?(type)
|
|
3267
|
+
|
|
3268
|
+
removed.add(identifier) if remove_unit(identifier, affected_types, type: type)
|
|
3269
|
+
end
|
|
3270
|
+
end
|
|
3271
|
+
end
|
|
3272
|
+
|
|
3273
|
+
# Does this unit's `file_path` name a *convention* that need not exist?
|
|
3274
|
+
#
|
|
3275
|
+
# The sweep deletes units whose file is gone, which is wrong for anything
|
|
3276
|
+
# that derives its path from a constant name rather than from a file it was
|
|
3277
|
+
# read out of. Class-based types have always been excluded for that reason.
|
|
3278
|
+
#
|
|
3279
|
+
# Today the predicate keys on {CLASS_BASED} and nothing else (L4): the six
|
|
3280
|
+
# class-based families — models, controllers, components, view components,
|
|
3281
|
+
# mailers, channels. GRAPHQL_TYPES are deliberately NOT spared. An earlier
|
|
3282
|
+
# version of this comment narrated GraphQL units being spared through a
|
|
3283
|
+
# `source_file_for_class` convention-path fallback, but that fallback now
|
|
3284
|
+
# returns nil rather than fabricating a convention path
|
|
3285
|
+
# (`graphql_extractor.rb`), a runtime-defined GraphQL unit registers no
|
|
3286
|
+
# path at all ({DependencyGraph#register} skips nil), and the static file
|
|
3287
|
+
# pass records the real file it read — so the sweep's own bounds decide,
|
|
3288
|
+
# and this predicate has no say in GraphQL either way.
|
|
3289
|
+
#
|
|
3290
|
+
# The original case: on Rails < 7.1 `ActiveRecord::SchemaMigration` and
|
|
3291
|
+
# `InternalMetadata` are real `ActiveRecord::Base` descendants a full
|
|
3292
|
+
# extraction emits with `app/models/active_record/schema_migration.rb` as
|
|
3293
|
+
# their path — a file no application has, and one the PORO rule *does*
|
|
3294
|
+
# claim, so the sweep's file-rule bound doesn't cover it and this check has
|
|
3295
|
+
# to. Deleting such a unit requires the caller to name the path; the sweep
|
|
3296
|
+
# never infers it.
|
|
3297
|
+
#
|
|
3298
|
+
# KNOWN COST (B-070 / #171): this keys on unit *type*, but the property is
|
|
3299
|
+
# per-unit — CLASS_BASED is also true of units whose recorded path has
|
|
3300
|
+
# since been moved elsewhere, which is exactly the move shape
|
|
3301
|
+
# {#readdable_pruned_classes} exists to re-add in the same run. For
|
|
3302
|
+
# deletions the cost stays: a class-based unit whose file is gone survives
|
|
3303
|
+
# an unnamed-path sweep. Named-path deletion still works, so
|
|
3304
|
+
# `woods:incremental` on a git diff is unaffected; the exposed caller is the
|
|
3305
|
+
# daemon's catch-up, which runs an empty change set precisely because
|
|
3306
|
+
# deletions leave no mtime.
|
|
3307
|
+
#
|
|
3308
|
+
# Two obvious fixes were tried and **both are wrong** — do not re-attempt
|
|
3309
|
+
# them without reading this:
|
|
3310
|
+
#
|
|
3311
|
+
# 1. *Compare the recorded path against the convention path derived from the
|
|
3312
|
+
# constant.* `Types::ForgottenType` conventionally lives at
|
|
3313
|
+
# `app/graphql/types/forgotten_type.rb`, which IS its convention path — so
|
|
3314
|
+
# a conventionally-named file-defined type is indistinguishable from a
|
|
3315
|
+
# runtime-defined one, and the common case stays spared. Disproven by
|
|
3316
|
+
# `spec/integration/incremental_equivalence_spec.rb`'s pending example.
|
|
3317
|
+
# 2. *Let the sweep prune GraphQL and rely on the reconciler's addition half
|
|
3318
|
+
# to re-add whatever the schema still holds.* This is what `except: pruned`
|
|
3319
|
+
# exists to prevent (see the comment at its call site): without a reload a
|
|
3320
|
+
# constant outlives its deleted file, and the re-add resurrects the deleted
|
|
3321
|
+
# unit against a path that no longer exists — permanently, since the sweep
|
|
3322
|
+
# then spares it.
|
|
3323
|
+
#
|
|
3324
|
+
# What would actually work is provenance: record at extraction time whether a
|
|
3325
|
+
# source file existed for the unit and spare only the units that never had
|
|
3326
|
+
# one. That needs the graph node to carry the flag, so it is a serialization
|
|
3327
|
+
# change rather than a predicate tweak.
|
|
3328
|
+
#
|
|
3329
|
+
# @param type [Symbol, nil] the unit type the caller is about to remove
|
|
3330
|
+
# @return [Boolean]
|
|
3331
|
+
def convention_path_unit?(type)
|
|
3332
|
+
CLASS_BASED.key?(type)
|
|
3333
|
+
end
|
|
3334
|
+
|
|
3335
|
+
# Register a batch of freshly-extracted units and write their JSON.
|
|
3336
|
+
#
|
|
3337
|
+
# Registration happens BEFORE path normalization — the graph's file map
|
|
3338
|
+
# stores absolute paths (that is what changed files are matched against),
|
|
3339
|
+
# exactly as full extraction registers in Phase 1 and only normalizes in
|
|
3340
|
+
# Phase 4.5. Unit JSON carries the relative path.
|
|
3341
|
+
#
|
|
3342
|
+
# @param extractor_key [Symbol]
|
|
3343
|
+
# @param units [Array<ExtractedUnit>]
|
|
3344
|
+
# @param affected_types [Set<Symbol>]
|
|
3345
|
+
# @return [Set<String>] identifiers written
|
|
3346
|
+
def register_and_write(extractor_key, units, affected_types)
|
|
3347
|
+
units = Array(units).compact
|
|
3348
|
+
return Set.new if units.empty?
|
|
3349
|
+
|
|
3350
|
+
affected_types&.add(extractor_key)
|
|
3351
|
+
type_dir = payload_dir.join(extractor_key.to_s)
|
|
3352
|
+
FileUtils.mkdir_p(type_dir)
|
|
3353
|
+
|
|
3354
|
+
units.each_with_object(Set.new) do |unit, written|
|
|
3355
|
+
annotate_package(unit)
|
|
3356
|
+
mark_dependents_dirty(unit.identifier)
|
|
3357
|
+
# Marked BEFORE registration: DependencyGraph#register inserts the
|
|
3358
|
+
# node before it iterates the unit's dependencies, so a malformed
|
|
3359
|
+
# dependency raises with the graph already mutated — a marker placed
|
|
3360
|
+
# after the call would never run, and the phantom would ship. The
|
|
3361
|
+
# cost of this ordering is a conservative abort when registration
|
|
3362
|
+
# fails before mutating anything; that is the safe direction to err
|
|
3363
|
+
# in. See {#note_wholesale_mutation}.
|
|
3364
|
+
note_wholesale_mutation
|
|
3365
|
+
@dependency_graph.register(unit)
|
|
3366
|
+
mark_dependents_dirty(unit.identifier)
|
|
3367
|
+
|
|
3368
|
+
unit.file_path = normalize_file_path(unit.file_path)
|
|
3369
|
+
# Keyed by relative path, which is how batch_git_data keys its result.
|
|
3370
|
+
(@incremental_written ||= {})[unit.identifier] = unit.file_path
|
|
3371
|
+
|
|
3372
|
+
write_unit_file(type_dir.join(collision_safe_filename(unit.identifier)), unit)
|
|
3373
|
+
written.add(unit.identifier)
|
|
3374
|
+
end
|
|
3375
|
+
end
|
|
3376
|
+
|
|
3377
|
+
# Remove a unit from the graph and delete its JSON from the index.
|
|
3378
|
+
#
|
|
3379
|
+
# Callers that know which type they mean must say so. An identifier can
|
|
3380
|
+
# name units of several types (a Scenic view `reports` and a factory
|
|
3381
|
+
# `reports`), each with its own `<extractor_key>/<identifier>.json`, and
|
|
3382
|
+
# removing the identifier wholesale takes the sibling with it.
|
|
3383
|
+
#
|
|
3384
|
+
# @param identifier [String]
|
|
3385
|
+
# @param affected_types [Set<Symbol>]
|
|
3386
|
+
# @param type [Symbol, nil] remove only this type; without it, every type
|
|
3387
|
+
# registered under the identifier
|
|
3388
|
+
# @return [String, nil] the identifier when it existed and was removed
|
|
3389
|
+
def remove_unit(identifier, affected_types, type: nil)
|
|
3390
|
+
types = type ? [type] : @dependency_graph.node_types(identifier)
|
|
3391
|
+
types = types.select { |t| @dependency_graph.node(identifier, type: t) }
|
|
3392
|
+
return nil if types.empty?
|
|
3393
|
+
|
|
3394
|
+
# Before the removals: this reads the identifier's forward edges, which
|
|
3395
|
+
# go with the nodes.
|
|
3396
|
+
mark_dependents_dirty(identifier)
|
|
3397
|
+
|
|
3398
|
+
types.each { |t| remove_unit_of_type(identifier, t, affected_types) }
|
|
3399
|
+
Rails.logger.debug { "[Woods] Removed #{identifier}" }
|
|
3400
|
+
identifier
|
|
3401
|
+
end
|
|
3402
|
+
|
|
3403
|
+
# @param identifier [String]
|
|
3404
|
+
# @param type [Symbol]
|
|
3405
|
+
# @param affected_types [Set<Symbol>]
|
|
3406
|
+
# @return [void]
|
|
3407
|
+
def remove_unit_of_type(identifier, type, affected_types)
|
|
3408
|
+
extractor_key = TYPE_TO_EXTRACTOR_KEY[type]
|
|
3409
|
+
|
|
3410
|
+
if extractor_key
|
|
3411
|
+
affected_types&.add(extractor_key)
|
|
3412
|
+
path = payload_dir.join(extractor_key.to_s, collision_safe_filename(identifier))
|
|
3413
|
+
# Marked BEFORE the removal, same ordering as the write half: a
|
|
3414
|
+
# rm_f that fails part-way (or a failure immediately after it) must
|
|
3415
|
+
# not leave the counter at zero while the file is already gone. See
|
|
3416
|
+
# {#note_wholesale_mutation}.
|
|
3417
|
+
note_wholesale_mutation
|
|
3418
|
+
FileUtils.rm_f(path)
|
|
3419
|
+
end
|
|
3420
|
+
|
|
3421
|
+
@dependency_graph.remove(identifier, type: type)
|
|
3422
|
+
end
|
|
3423
|
+
|
|
3424
|
+
# Record that a unit's edges changed, so every target it points at (and
|
|
3425
|
+
# the unit itself) has its `dependents` list rewritten at the end of the
|
|
3426
|
+
# run.
|
|
3427
|
+
#
|
|
3428
|
+
# @param identifier [String]
|
|
3429
|
+
# @return [void]
|
|
3430
|
+
def mark_dependents_dirty(identifier)
|
|
3431
|
+
@dependents_dirty ||= Set.new
|
|
3432
|
+
@dependents_dirty.add(identifier)
|
|
3433
|
+
@dependency_graph.dependencies_of(identifier).each { |target| @dependents_dirty.add(target) }
|
|
3434
|
+
end
|
|
3435
|
+
|
|
3436
|
+
# Second pass over the unit JSON an incremental run touched, mirroring the
|
|
3437
|
+
# two full-extraction phases that operate on already-extracted units:
|
|
3438
|
+
# {#resolve_dependents} (Phase 2) and {#enrich_with_git_data} (Phase 4).
|
|
3439
|
+
#
|
|
3440
|
+
# Both were full-extraction-only, so incremental runs left `dependents`
|
|
3441
|
+
# and `metadata.git` frozen at whatever the last full run wrote —
|
|
3442
|
+
# divergence that compounds run over run in an incremental CI chain.
|
|
3443
|
+
#
|
|
3444
|
+
# @param affected_types [Set<Symbol>]
|
|
3445
|
+
# @return [void]
|
|
3446
|
+
def finalize_incremental_unit_json(affected_types)
|
|
3447
|
+
dependents_dirty = @dependents_dirty || Set.new
|
|
3448
|
+
git_dirty = @incremental_written || {}
|
|
3449
|
+
git_data = incremental_git_data(git_dirty.keys)
|
|
3450
|
+
|
|
3451
|
+
(dependents_dirty | git_dirty.keys).each do |identifier|
|
|
3452
|
+
rewrite_unit_json(identifier, affected_types,
|
|
3453
|
+
refresh_dependents: dependents_dirty.include?(identifier),
|
|
3454
|
+
git_data: git_dirty.key?(identifier) ? git_data : nil)
|
|
3455
|
+
end
|
|
3456
|
+
end
|
|
3457
|
+
|
|
3458
|
+
# Read one unit's JSON, apply the second-pass patches, and write it back
|
|
3459
|
+
# only if something actually changed.
|
|
3460
|
+
#
|
|
3461
|
+
# @param identifier [String]
|
|
3462
|
+
# @param affected_types [Set<Symbol>]
|
|
3463
|
+
# @param refresh_dependents [Boolean]
|
|
3464
|
+
# @param git_data [Hash{String => Hash}, nil] batch git data keyed by
|
|
3465
|
+
# relative path (per {#batch_git_data}), or nil when this identifier
|
|
3466
|
+
# isn't git-dirty this run
|
|
3467
|
+
# @return [void]
|
|
3468
|
+
def rewrite_unit_json(identifier, affected_types, refresh_dependents:, git_data:)
|
|
3469
|
+
@dependency_graph.node_types(identifier).each do |type|
|
|
3470
|
+
rewrite_unit_json_of_type(identifier, type, affected_types,
|
|
3471
|
+
refresh_dependents: refresh_dependents, git_data: git_data)
|
|
3472
|
+
end
|
|
3473
|
+
end
|
|
3474
|
+
|
|
3475
|
+
# One (identifier, type) pair's JSON. The `dependents` list is a property
|
|
3476
|
+
# of the identifier, not of the type, so every file the identifier owns
|
|
3477
|
+
# gets the same refreshed list — which is what a full extraction writes.
|
|
3478
|
+
#
|
|
3479
|
+
# Git metadata is NOT a property of the identifier (#225): a colliding
|
|
3480
|
+
# identifier's types each have their own `file_path` (a Scenic view and a
|
|
3481
|
+
# factory both named `reports` live in different files with different
|
|
3482
|
+
# histories), so this resolves git data against THIS type's own node
|
|
3483
|
+
# rather than a single pre-resolved hash shared across every type — the
|
|
3484
|
+
# previous shape let one type's commit history land in every colliding
|
|
3485
|
+
# type's `metadata.git`, keyed by whichever type {#register_and_write}
|
|
3486
|
+
# happened to touch last for that identifier.
|
|
3487
|
+
#
|
|
3488
|
+
# @param identifier [String]
|
|
3489
|
+
# @param type [Symbol]
|
|
3490
|
+
# @param affected_types [Set<Symbol>]
|
|
3491
|
+
# @param refresh_dependents [Boolean]
|
|
3492
|
+
# @param git_data [Hash{String => Hash}, nil]
|
|
3493
|
+
# @return [void]
|
|
3494
|
+
def rewrite_unit_json_of_type(identifier, type, affected_types, refresh_dependents:, git_data:)
|
|
3495
|
+
extractor_key = TYPE_TO_EXTRACTOR_KEY[type]
|
|
3496
|
+
return unless extractor_key
|
|
3497
|
+
|
|
3498
|
+
path = payload_dir.join(extractor_key.to_s, collision_safe_filename(identifier))
|
|
3499
|
+
return unless File.exist?(path)
|
|
3500
|
+
|
|
3501
|
+
data = JSON.parse(AtomicFile.read(path))
|
|
3502
|
+
# Serialized comparison, not `data.dup`: the git patch mutates the
|
|
3503
|
+
# nested metadata hash in place, which a shallow copy would follow.
|
|
3504
|
+
before = JSON.generate(data)
|
|
3505
|
+
|
|
3506
|
+
if refresh_dependents
|
|
3507
|
+
data['dependents'] = @dependency_graph.dependents_detail(identifier)
|
|
3508
|
+
.map { |d| { 'type' => d[:type].to_s, 'identifier' => d[:identifier] } }
|
|
3509
|
+
end
|
|
3510
|
+
|
|
3511
|
+
git = git_data && git_for_type(identifier, type, git_data)
|
|
3512
|
+
if git
|
|
3513
|
+
(data['metadata'] ||= {})['git'] = JSON.parse(JSON.generate(git))
|
|
3514
|
+
annotate_node_from_git(identifier, type, git)
|
|
3515
|
+
end
|
|
3516
|
+
|
|
3517
|
+
return if JSON.generate(data) == before
|
|
3518
|
+
|
|
3519
|
+
AtomicFile.write(path, json_serialize(data), durable: payload_writes_durable?)
|
|
3520
|
+
affected_types&.add(extractor_key)
|
|
3521
|
+
rescue JSON::ParserError => e
|
|
3522
|
+
Rails.logger.warn "[Woods] Could not finalize #{identifier}: #{e.message}"
|
|
3523
|
+
end
|
|
3524
|
+
|
|
3525
|
+
# This (identifier, type) pair's own git data, looked up by its own
|
|
3526
|
+
# `file_path` rather than by identifier — see {#rewrite_unit_json_of_type}.
|
|
3527
|
+
#
|
|
3528
|
+
# @param identifier [String]
|
|
3529
|
+
# @param type [Symbol]
|
|
3530
|
+
# @param git_data [Hash{String => Hash}] keyed by relative path
|
|
3531
|
+
# @return [Hash, nil]
|
|
3532
|
+
def git_for_type(identifier, type, git_data)
|
|
3533
|
+
node = @dependency_graph.node(identifier, type: type)
|
|
3534
|
+
return nil unless node && node[:file_path]
|
|
3535
|
+
|
|
3536
|
+
git_data[normalize_file_path(node[:file_path])]
|
|
3537
|
+
end
|
|
3538
|
+
|
|
3539
|
+
# Mirror of {#annotate_graph_with_git_data} for one incrementally
|
|
3540
|
+
# patched unit. Git data arrives symbol-keyed from {#batch_git_data} and
|
|
3541
|
+
# string-keyed when a caller hands in parsed JSON; both are accepted.
|
|
3542
|
+
#
|
|
3543
|
+
# Unlike {#annotate_graph_with_git_data}, a missing key is never forwarded
|
|
3544
|
+
# as an explicit `nil`: {DependencyGraph#annotate} treats `nil` as
|
|
3545
|
+
# "clear this attribute", and this batch's git data can be missing one of
|
|
3546
|
+
# the two keys without meaning the other's last-known value is gone.
|
|
3547
|
+
#
|
|
3548
|
+
# @param identifier [String]
|
|
3549
|
+
# @param type [Symbol]
|
|
3550
|
+
# @param git [Hash]
|
|
3551
|
+
# @return [void]
|
|
3552
|
+
def annotate_node_from_git(identifier, type, git)
|
|
3553
|
+
attributes = {
|
|
3554
|
+
commit_count: git[:commit_count] || git['commit_count'],
|
|
3555
|
+
change_frequency: git[:change_frequency] || git['change_frequency']
|
|
3556
|
+
}.compact
|
|
3557
|
+
return if attributes.empty?
|
|
3558
|
+
|
|
3559
|
+
@dependency_graph.annotate(identifier, type: type, **attributes)
|
|
3560
|
+
end
|
|
3561
|
+
|
|
3562
|
+
# Batch-fetch git metadata for the units written by this run, in a single
|
|
3563
|
+
# git invocation, keyed by Rails.root-relative path the way
|
|
3564
|
+
# {#batch_git_data} returns it.
|
|
3565
|
+
#
|
|
3566
|
+
# @param identifiers [Array<String>]
|
|
3567
|
+
# @return [Hash{String => Hash}]
|
|
3568
|
+
def incremental_git_data(identifiers)
|
|
3569
|
+
return {} if identifiers.empty? || !git_available?
|
|
3570
|
+
|
|
3571
|
+
paths = identifiers.flat_map do |identifier|
|
|
3572
|
+
@dependency_graph.nodes_for(identifier).filter_map do |node|
|
|
3573
|
+
next if %i[rails_source gem_source].include?(node[:type])
|
|
3574
|
+
|
|
3575
|
+
node[:file_path] if node[:file_path] && File.exist?(node[:file_path])
|
|
3576
|
+
end
|
|
3577
|
+
end
|
|
3578
|
+
|
|
3579
|
+
batch_git_data(paths.uniq)
|
|
3580
|
+
rescue StandardError => e
|
|
3581
|
+
Rails.logger.warn "[Woods] Incremental git enrichment failed: #{e.message}"
|
|
3582
|
+
{}
|
|
3583
|
+
end
|
|
3584
|
+
|
|
3585
|
+
# Re-extract a single known unit, identified by its graph node.
|
|
3586
|
+
#
|
|
3587
|
+
# Used for the transitive half of the blast radius — units whose own file
|
|
3588
|
+
# did not change but which depend on something that did. Files that *did*
|
|
3589
|
+
# change go through {#reconcile_changed_paths} instead, which reconciles
|
|
3590
|
+
# the whole path rather than one identifier.
|
|
3591
|
+
#
|
|
3592
|
+
# @param unit_id [String]
|
|
3593
|
+
# @param affected_types [Set<Symbol>, nil]
|
|
3594
|
+
# @return [String, nil] the identifier when it was re-extracted and written
|
|
1090
3595
|
def re_extract_unit(unit_id, affected_types: nil)
|
|
1091
3596
|
# Framework source only changes on version updates
|
|
1092
3597
|
if unit_id.start_with?('rails/') || unit_id.start_with?('gems/')
|
|
1093
|
-
Rails.logger.debug "[Woods] Skipping framework re-extraction for #{unit_id}"
|
|
1094
|
-
return
|
|
3598
|
+
Rails.logger.debug { "[Woods] Skipping framework re-extraction for #{unit_id}" }
|
|
3599
|
+
return nil
|
|
1095
3600
|
end
|
|
1096
3601
|
|
|
1097
|
-
#
|
|
1098
|
-
|
|
1099
|
-
|
|
3602
|
+
# An identifier can name units of several types, each with its own
|
|
3603
|
+
# extractor and its own file; re-extracting one and calling it done would
|
|
3604
|
+
# leave the others frozen at the pre-change extraction.
|
|
3605
|
+
types = @dependency_graph.node_types(unit_id)
|
|
3606
|
+
return nil if types.empty?
|
|
3607
|
+
|
|
3608
|
+
re_extracted = types.count { |type| re_extract_unit_of_type(unit_id, type, affected_types) }
|
|
3609
|
+
return nil if re_extracted.zero?
|
|
3610
|
+
|
|
3611
|
+
Rails.logger.info "[Woods] Re-extracted #{unit_id}"
|
|
3612
|
+
unit_id
|
|
3613
|
+
end
|
|
1100
3614
|
|
|
1101
|
-
|
|
1102
|
-
|
|
3615
|
+
# @param unit_id [String]
|
|
3616
|
+
# @param type [Symbol]
|
|
3617
|
+
# @param affected_types [Set<Symbol>, nil]
|
|
3618
|
+
# @return [String, nil] the identifier when this type was re-extracted and written
|
|
3619
|
+
def re_extract_unit_of_type(unit_id, type, affected_types)
|
|
3620
|
+
node = @dependency_graph.node(unit_id, type: type)
|
|
3621
|
+
file_path = node && node[:file_path]
|
|
1103
3622
|
|
|
1104
|
-
|
|
3623
|
+
# A vanished file is not re-extractable; {#prune_vanished_units} owns it.
|
|
3624
|
+
return nil unless file_path && File.exist?(file_path)
|
|
1105
3625
|
|
|
1106
|
-
# Re-extract based on type
|
|
1107
3626
|
extractor_key = TYPE_TO_EXTRACTOR_KEY[type]
|
|
1108
|
-
return unless extractor_key
|
|
3627
|
+
return nil unless extractor_key
|
|
1109
3628
|
|
|
1110
|
-
extractor =
|
|
1111
|
-
return unless extractor
|
|
1112
|
-
|
|
1113
|
-
unit = if (method = CLASS_BASED[type])
|
|
1114
|
-
klass = if unit_id.match?(/\A[A-Z][A-Za-z0-9_:]*\z/)
|
|
1115
|
-
begin
|
|
1116
|
-
unit_id.constantize
|
|
1117
|
-
rescue StandardError
|
|
1118
|
-
nil
|
|
1119
|
-
end
|
|
1120
|
-
end
|
|
1121
|
-
extractor.public_send(method, klass) if klass
|
|
1122
|
-
elsif (method = FILE_BASED[type])
|
|
1123
|
-
extractor.public_send(method, file_path)
|
|
1124
|
-
elsif GRAPHQL_TYPES.include?(type)
|
|
1125
|
-
extractor.extract_graphql_file(file_path)
|
|
1126
|
-
end
|
|
1127
|
-
|
|
1128
|
-
return unless unit
|
|
3629
|
+
extractor = extractor_for(extractor_key)
|
|
3630
|
+
return nil unless extractor
|
|
1129
3631
|
|
|
1130
3632
|
# File-based extractors can return several units from one file (a .rake
|
|
1131
3633
|
# file defining multiple tasks, etc.); class-based extractors return one.
|
|
1132
|
-
|
|
1133
|
-
|
|
1134
|
-
units = unit.is_a?(Array) ? unit : [unit]
|
|
1135
|
-
return if units.empty?
|
|
3634
|
+
units = Array(re_extracted_units(extractor, type, unit_id, file_path, extractor_key)).compact
|
|
3635
|
+
return nil if units.empty?
|
|
1136
3636
|
|
|
1137
|
-
|
|
1138
|
-
|
|
3637
|
+
register_and_write(extractor_key, units, affected_types)
|
|
3638
|
+
unit_id
|
|
3639
|
+
end
|
|
1139
3640
|
|
|
1140
|
-
|
|
3641
|
+
# Dispatch one re-extraction to the right extractor entry point.
|
|
3642
|
+
#
|
|
3643
|
+
# @return [ExtractedUnit, Array<ExtractedUnit>, nil]
|
|
3644
|
+
def re_extracted_units(extractor, type, unit_id, file_path, extractor_key)
|
|
3645
|
+
if (method = CLASS_BASED[type])
|
|
3646
|
+
klass = constant_for_identifier(unit_id)
|
|
3647
|
+
klass && extractor.public_send(method, klass)
|
|
3648
|
+
elsif (method = FILE_BASED[type])
|
|
3649
|
+
units = Array(extract_file_based_unit(extractor, method, file_path, extractor_key)).compact
|
|
3650
|
+
return units unless CLASS_DISCOVERED_FALLBACK.key?(type)
|
|
3651
|
+
return units if units.any? { |unit| unit.identifier == unit_id }
|
|
3652
|
+
|
|
3653
|
+
re_extract_by_class(extractor, type, unit_id)
|
|
3654
|
+
elsif GRAPHQL_TYPES.include?(type)
|
|
3655
|
+
extractor.extract_graphql_file(file_path)
|
|
3656
|
+
end
|
|
3657
|
+
end
|
|
1141
3658
|
|
|
1142
|
-
|
|
1143
|
-
|
|
1144
|
-
|
|
1145
|
-
|
|
1146
|
-
|
|
1147
|
-
|
|
3659
|
+
# Re-extract a unit the full path discovered by class, not by file.
|
|
3660
|
+
#
|
|
3661
|
+
# Jobs are found two ways ({Extractors::JobExtractor#extract_all}): a scan
|
|
3662
|
+
# of the job directories, then a descendant walk for every job class the
|
|
3663
|
+
# scan did not name. A class-discovered job can live in a file whose
|
|
3664
|
+
# governed constant is not the job — a job nested inside a model — so
|
|
3665
|
+
# re-deriving it from that file names the enclosing class instead, and
|
|
3666
|
+
# registered a second job unit under the PORO's identifier that no full
|
|
3667
|
+
# extraction emits: the wrapper-naming collision, reached incrementally.
|
|
3668
|
+
# When the file entry point does not reproduce the unit, the full path
|
|
3669
|
+
# found it by class; do the same, and register nothing the file scan
|
|
3670
|
+
# would not have produced there either.
|
|
3671
|
+
#
|
|
3672
|
+
# Only a class the extractor itself would discover qualifies. A stale
|
|
3673
|
+
# pre-2.0 wrapper identifier can still constantize (to the wrapper class),
|
|
3674
|
+
# and re-extracting *that* by class would pin the stale unit with fresh
|
|
3675
|
+
# metadata; it stays as it is until a full extraction replaces it.
|
|
3676
|
+
#
|
|
3677
|
+
# @return [ExtractedUnit, nil]
|
|
3678
|
+
def re_extract_by_class(extractor, type, unit_id)
|
|
3679
|
+
klass = constant_for_identifier(unit_id)
|
|
3680
|
+
return nil unless klass && extractor.discoverable_classes.include?(klass)
|
|
1148
3681
|
|
|
1149
|
-
|
|
1150
|
-
|
|
1151
|
-
# leak container-absolute paths into the index after incremental runs.
|
|
1152
|
-
extracted.file_path = normalize_file_path(extracted.file_path)
|
|
3682
|
+
extractor.public_send(CLASS_DISCOVERED_FALLBACK[type], klass)
|
|
3683
|
+
end
|
|
1153
3684
|
|
|
1154
|
-
|
|
1155
|
-
|
|
1156
|
-
|
|
1157
|
-
|
|
1158
|
-
|
|
3685
|
+
# @param unit_id [String]
|
|
3686
|
+
# @return [Class, nil] the constant an identifier names, when it names one
|
|
3687
|
+
def constant_for_identifier(unit_id)
|
|
3688
|
+
return nil unless unit_id.match?(/\A[A-Z][A-Za-z0-9_:]*\z/)
|
|
3689
|
+
|
|
3690
|
+
begin
|
|
3691
|
+
unit_id.constantize
|
|
3692
|
+
rescue StandardError
|
|
3693
|
+
nil
|
|
1159
3694
|
end
|
|
3695
|
+
end
|
|
1160
3696
|
|
|
1161
|
-
|
|
3697
|
+
# Invoke a file-based extraction method, supplying the extra arguments
|
|
3698
|
+
# the handful of non-uniform signatures need.
|
|
3699
|
+
#
|
|
3700
|
+
# @return [ExtractedUnit, Array<ExtractedUnit>, nil]
|
|
3701
|
+
def extract_file_based_unit(extractor, method, file_path, extractor_key)
|
|
3702
|
+
if extractor_key == :poros
|
|
3703
|
+
extractor.public_send(method, file_path, ar_names: active_record_names)
|
|
3704
|
+
else
|
|
3705
|
+
extractor.public_send(method, file_path)
|
|
3706
|
+
end
|
|
1162
3707
|
end
|
|
1163
3708
|
end
|
|
1164
3709
|
end
|