woods 2.0.0.beta2 → 2.0.0.beta4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +339 -1
- data/CONTRIBUTING.md +188 -12
- data/README.md +93 -174
- data/SECURITY.md +9 -6
- data/docs/AGENT_GUIDE.md +109 -8
- data/docs/AGENT_SETUP.md +98 -7
- data/docs/BACKEND_MATRIX.md +25 -0
- data/docs/CLIENT_HOOKS.md +111 -0
- data/docs/CONFIGURATION_REFERENCE.md +267 -16
- data/docs/CONSOLE_MCP_SETUP.md +80 -7
- data/docs/DOCKER_SETUP.md +22 -3
- data/docs/EVALUATION.md +464 -1
- data/docs/EXTRACTOR_REFERENCE.md +45 -6
- data/docs/FAQ.md +11 -12
- data/docs/GETTING_STARTED.md +17 -5
- data/docs/INCREMENTAL_EXTRACTION.md +147 -7
- data/docs/INDEX_LAYOUT.md +382 -0
- data/docs/INTERNALS.md +7 -2
- data/docs/MCP_SERVERS.md +276 -5
- data/docs/MCP_TOOL_COOKBOOK.md +37 -22
- data/docs/MCP_WORKTREE_SETUP.md +43 -83
- data/docs/NOTION_INTEGRATION.md +13 -0
- data/docs/OBSIDIAN_INTEGRATION.md +57 -9
- data/docs/PUBLISHED_INDEX.md +72 -0
- data/docs/README.md +7 -0
- data/docs/RETRIEVAL_GUIDE.md +273 -12
- data/docs/RUNTIME_TRACING.md +71 -0
- data/docs/SOURCE_FRESHNESS.md +143 -0
- data/docs/TROUBLESHOOTING.md +129 -18
- data/docs/UNBLOCKED_INTEGRATION.md +25 -0
- data/docs/UPGRADING_TO_2.md +48 -22
- data/docs/WATCH_DAEMON.md +277 -67
- data/exe/woods-agent-config +6 -0
- data/exe/woods-extract +5 -0
- data/exe/woods-hook-context +6 -0
- data/exe/woods-mcp-start +14 -9
- data/lib/generators/woods/pgvector_generator.rb +8 -2
- data/lib/generators/woods/templates/woods.rb.tt +1 -3
- data/lib/tasks/woods.rake +47 -397
- data/lib/woods/agent_configuration/applier.rb +135 -0
- data/lib/woods/agent_configuration/cli.rb +101 -0
- data/lib/woods/agent_configuration/cli_options.rb +29 -0
- data/lib/woods/agent_configuration/document.rb +105 -0
- data/lib/woods/agent_configuration/error.rb +7 -0
- data/lib/woods/agent_configuration/launcher.rb +75 -0
- data/lib/woods/agent_configuration/layout.rb +72 -0
- data/lib/woods/agent_configuration/managed_section.rb +62 -0
- data/lib/woods/agent_configuration/plan.rb +98 -0
- data/lib/woods/agent_configuration/plan_diff.rb +38 -0
- data/lib/woods/agent_configuration/planned_files.rb +61 -0
- data/lib/woods/agent_configuration/planner.rb +63 -0
- data/lib/woods/agent_configuration/planner_validation.rb +77 -0
- data/lib/woods/agent_configuration/preflight.rb +100 -0
- data/lib/woods/agent_configuration/recovery.rb +49 -0
- data/lib/woods/ast/node.rb +2 -0
- data/lib/woods/ast/parser.rb +38 -5
- data/lib/woods/builder.rb +21 -5
- data/lib/woods/cache/cache_middleware.rb +28 -7
- data/lib/woods/cache/cache_store.rb +4 -5
- data/lib/woods/change_set.rb +5 -4
- data/lib/woods/console/credential_index.rb +20 -2
- data/lib/woods/console/credential_scanner.rb +18 -17
- data/lib/woods/console/credential_scanner_registry.rb +36 -0
- data/lib/woods/console/dispatch_pipeline.rb +7 -0
- data/lib/woods/console/embedded_executor.rb +32 -10
- data/lib/woods/console/encrypted_credential_snapshot.rb +16 -0
- data/lib/woods/console/rack_middleware.rb +22 -13
- data/lib/woods/console/server.rb +18 -16
- data/lib/woods/console/sql_noise_stripper.rb +9 -7
- data/lib/woods/console/sql_table_scanner.rb +47 -7
- data/lib/woods/console/sql_validator.rb +49 -9
- data/lib/woods/console/sqlite_read_guard.rb +46 -0
- data/lib/woods/coordination/pipeline_lock.rb +3 -2
- data/lib/woods/dependency_graph.rb +65 -13
- data/lib/woods/embedding/corpus.rb +94 -0
- data/lib/woods/embedding/indexer.rb +114 -60
- data/lib/woods/embedding/openai.rb +17 -6
- data/lib/woods/evaluation/ablation_executor.rb +6 -1
- data/lib/woods/evaluation/ablation_timed_executor.rb +22 -4
- data/lib/woods/export/typed_reader.rb +56 -0
- data/lib/woods/extractor.rb +277 -149
- data/lib/woods/extractors/action_cable_extractor.rb +3 -1
- data/lib/woods/extractors/behavioral_profile.rb +9 -7
- data/lib/woods/extractors/caching_extractor.rb +3 -1
- data/lib/woods/extractors/concern_extractor.rb +64 -6
- data/lib/woods/extractors/configuration_extractor.rb +7 -3
- data/lib/woods/extractors/controller_extractor.rb +13 -4
- data/lib/woods/extractors/database_view_extractor.rb +3 -1
- data/lib/woods/extractors/declared_parent.rb +55 -0
- data/lib/woods/extractors/decorator_extractor.rb +3 -1
- data/lib/woods/extractors/engine_extractor.rb +3 -1
- data/lib/woods/extractors/event_extractor.rb +4 -2
- data/lib/woods/extractors/factory_extractor.rb +3 -1
- data/lib/woods/extractors/graphql_extractor.rb +10 -13
- data/lib/woods/extractors/i18n_extractor.rb +3 -1
- data/lib/woods/extractors/job_extractor.rb +6 -19
- data/lib/woods/extractors/lib_extractor.rb +13 -9
- data/lib/woods/extractors/mailer_extractor.rb +26 -15
- data/lib/woods/extractors/manager_extractor.rb +3 -1
- data/lib/woods/extractors/method_parameters.rb +53 -0
- data/lib/woods/extractors/middleware_argument.rb +65 -0
- data/lib/woods/extractors/middleware_extractor.rb +9 -3
- data/lib/woods/extractors/migration_extractor.rb +3 -1
- data/lib/woods/extractors/model_extractor.rb +26 -34
- data/lib/woods/extractors/package_extractor.rb +24 -4
- data/lib/woods/extractors/phlex_extractor.rb +3 -1
- data/lib/woods/extractors/policy_extractor.rb +3 -1
- data/lib/woods/extractors/poro_extractor.rb +13 -9
- data/lib/woods/extractors/pundit_extractor.rb +3 -1
- data/lib/woods/extractors/rails_source_extractor.rb +4 -2
- data/lib/woods/extractors/rake_task_extractor.rb +4 -2
- data/lib/woods/extractors/route_extractor.rb +3 -1
- data/lib/woods/extractors/route_helper_resolver.rb +10 -33
- data/lib/woods/extractors/scheduled_job_extractor.rb +41 -15
- data/lib/woods/extractors/serializer_extractor.rb +4 -2
- data/lib/woods/extractors/service_extractor.rb +3 -1
- data/lib/woods/extractors/shared_dependency_scanner.rb +2 -2
- data/lib/woods/extractors/shared_utility_methods.rb +48 -19
- data/lib/woods/extractors/source_nesting.rb +1 -1
- data/lib/woods/extractors/state_machine_extractor.rb +3 -1
- data/lib/woods/extractors/test_mapping_extractor.rb +3 -1
- data/lib/woods/extractors/validator_extractor.rb +3 -1
- data/lib/woods/extractors/view_component_extractor.rb +3 -1
- data/lib/woods/extractors/view_template_extractor.rb +3 -1
- data/lib/woods/gem_mapper.rb +2 -0
- data/lib/woods/git_history.rb +116 -0
- data/lib/woods/graph_analyzer.rb +35 -6
- data/lib/woods/hooks/context_cli.rb +54 -0
- data/lib/woods/hooks/context_event.rb +88 -0
- data/lib/woods/hooks/context_hint.rb +73 -0
- data/lib/woods/hooks/context_impact.rb +77 -0
- data/lib/woods/hooks/context_output.rb +47 -0
- data/lib/woods/hooks/context_state.rb +102 -0
- data/lib/woods/hooks/refresh.rb +79 -0
- data/lib/woods/hooks/rule_projection.rb +78 -0
- data/lib/woods/input_rules.rb +19 -0
- data/lib/woods/mcp/bearer_auth.rb +22 -13
- data/lib/woods/mcp/bootstrapper.rb +79 -4
- data/lib/woods/mcp/config_resolver.rb +2 -1
- data/lib/woods/mcp/index_reader.rb +334 -162
- data/lib/woods/mcp/initialization_guidance.rb +27 -0
- data/lib/woods/mcp/origin_guard.rb +17 -9
- data/lib/woods/mcp/published_lexical_retriever.rb +115 -0
- data/lib/woods/mcp/renderers/markdown_renderer.rb +22 -9
- data/lib/woods/mcp/renderers/plain_renderer.rb +18 -8
- data/lib/woods/mcp/search_results.rb +74 -0
- data/lib/woods/mcp/server.rb +178 -63
- data/lib/woods/mcp/tool_contract.rb +3 -1
- data/lib/woods/mcp/tool_response_renderer.rb +41 -0
- data/lib/woods/mcp/traversal_evidence.rb +113 -0
- data/lib/woods/mcp/traversal_evidence_index.rb +100 -0
- data/lib/woods/mcp/traversal_evidence_page.rb +41 -0
- data/lib/woods/mcp/traversal_evidence_text.rb +52 -0
- data/lib/woods/mcp/traversal_response.rb +22 -0
- data/lib/woods/notion/exporter.rb +56 -17
- data/lib/woods/obsidian/destination_plan.rb +98 -0
- data/lib/woods/obsidian/name_mapper.rb +19 -3
- data/lib/woods/obsidian/note_builder.rb +19 -10
- data/lib/woods/obsidian/vault_exporter.rb +88 -32
- data/lib/woods/operator/pipeline_guard.rb +18 -13
- data/lib/woods/path_dispatcher.rb +13 -6
- data/lib/woods/payload_store.rb +27 -26
- data/lib/woods/published_index/typed_unit_reader.rb +40 -3
- data/lib/woods/published_index.rb +2 -2
- data/lib/woods/railtie.rb +3 -3
- data/lib/woods/railtie_support.rb +12 -12
- data/lib/woods/rake_helpers.rb +382 -0
- data/lib/woods/resilience/graph_invariant_validator/membership_checks.rb +71 -0
- data/lib/woods/resilience/graph_invariant_validator/node_checks.rb +61 -0
- data/lib/woods/resilience/graph_invariant_validator/reverse_relationship_checks.rb +46 -0
- data/lib/woods/resilience/graph_invariant_validator.rb +119 -0
- data/lib/woods/resilience/index_validator/graph_checks.rb +80 -0
- data/lib/woods/resilience/index_validator.rb +112 -23
- data/lib/woods/retrieval/context_assembler.rb +50 -15
- data/lib/woods/retrieval/lexical_assembler.rb +84 -0
- data/lib/woods/retrieval/lexical_index.rb +120 -0
- data/lib/woods/retrieval/ranker.rb +4 -2
- data/lib/woods/retrieval/scope.rb +108 -0
- data/lib/woods/retrieval/scoped_graph_store.rb +32 -0
- data/lib/woods/retrieval/scoped_vector_store.rb +55 -0
- data/lib/woods/retrieval/search_executor.rb +86 -27
- data/lib/woods/retrieval/source_evidence.rb +200 -0
- data/lib/woods/retriever.rb +98 -22
- data/lib/woods/ruby_analyzer/trace_enricher.rb +77 -38
- data/lib/woods/session_tracer/file_store.rb +6 -1
- data/lib/woods/session_tracer/middleware.rb +10 -12
- data/lib/woods/session_tracer/redis_store.rb +22 -6
- data/lib/woods/session_tracer/session_flow_assembler.rb +23 -17
- data/lib/woods/session_tracer/solid_cache_coordination.rb +6 -4
- data/lib/woods/session_tracer/unit_resolver.rb +63 -0
- data/lib/woods/source_inputs/consumer_errors.rb +31 -0
- data/lib/woods/source_inputs/handoff.rb +102 -0
- data/lib/woods/source_inputs/launcher.rb +157 -0
- data/lib/woods/source_inputs/manifest.rb +124 -0
- data/lib/woods/source_inputs/private_key.rb +55 -0
- data/lib/woods/source_inputs/scanner.rb +171 -0
- data/lib/woods/source_inputs/scopes.rb +71 -0
- data/lib/woods/source_inputs/session.rb +214 -0
- data/lib/woods/source_inputs/status.rb +84 -0
- data/lib/woods/source_inputs/verifier.rb +107 -0
- data/lib/woods/storage/metadata_store.rb +25 -25
- data/lib/woods/storage/pgvector.rb +35 -10
- data/lib/woods/storage/qdrant.rb +17 -7
- data/lib/woods/storage/vector_store.rb +18 -6
- data/lib/woods/tasks.rb +3 -2
- data/lib/woods/temporal/json_snapshot_store.rb +58 -9
- data/lib/woods/unblocked/exporter.rb +59 -70
- data/lib/woods/version.rb +1 -1
- data/lib/woods/watch/boot_snapshot.rb +52 -0
- data/lib/woods/watch/daemon.rb +154 -32
- data/lib/woods/watch/listen_watcher.rb +4 -0
- data/lib/woods/watch/polling_watcher.rb +5 -1
- data/lib/woods/watch/status.rb +20 -15
- data/lib/woods/watch/tree_scan.rb +21 -13
- data/lib/woods/watch/watcher.rb +4 -1
- data/lib/woods.rb +50 -11
- data/plugin/.claude-plugin/plugin.json +1 -1
- data/plugin/hooks/adapters/normalize.jq +15 -0
- data/plugin/hooks/adapters/normalize.rb +63 -0
- data/plugin/hooks/hooks.json +20 -0
- data/plugin/hooks/woods-context.sh +50 -0
- data/plugin/hooks/woods-input-rules.sh +159 -0
- data/plugin/hooks/woods-opencode.mjs +65 -0
- data/plugin/hooks/woods-post-edit.sh +2 -225
- data/plugin/hooks/woods-refresh.sh +260 -0
- data/plugin/hooks/woods-session-start.sh +47 -55
- data/plugin/skills/woods-agent-enable/SKILL.md +19 -0
- data/plugin/skills/woods-diagnose/SKILL.md +319 -1
- data/plugin/skills/woods-investigate/SKILL.md +145 -0
- data/plugin/skills/woods-mcp-config/SKILL.md +90 -2
- data/plugin/skills/woods-setup/SKILL.md +110 -6
- metadata +87 -5
data/lib/woods/extractor.rb
CHANGED
|
@@ -8,12 +8,14 @@ require 'pathname'
|
|
|
8
8
|
require 'set'
|
|
9
9
|
|
|
10
10
|
require_relative 'atomic_file'
|
|
11
|
+
require_relative 'version'
|
|
11
12
|
require_relative 'filename_utils'
|
|
12
13
|
require_relative 'token_utils'
|
|
13
14
|
require_relative 'extracted_unit'
|
|
14
15
|
require_relative 'dependency_graph'
|
|
15
16
|
require_relative 'payload_store'
|
|
16
17
|
require_relative 'git_provenance'
|
|
18
|
+
require_relative 'git_history'
|
|
17
19
|
require_relative 'extractors/model_extractor'
|
|
18
20
|
require_relative 'extractors/controller_extractor'
|
|
19
21
|
require_relative 'extractors/phlex_extractor'
|
|
@@ -55,6 +57,7 @@ require_relative 'flow_precomputer'
|
|
|
55
57
|
require_relative 'change_set'
|
|
56
58
|
require_relative 'generation'
|
|
57
59
|
require_relative 'path_dispatcher'
|
|
60
|
+
require_relative 'source_inputs/session'
|
|
58
61
|
|
|
59
62
|
module Woods
|
|
60
63
|
# Extractor is the main orchestrator for codebase extraction.
|
|
@@ -208,7 +211,6 @@ module Woods
|
|
|
208
211
|
configuration: :extract_configuration_file,
|
|
209
212
|
view_template: :extract_view_template_file,
|
|
210
213
|
migration: :extract_migration_file,
|
|
211
|
-
rake_task: :extract_rake_file,
|
|
212
214
|
decorator: :extract_decorator_file,
|
|
213
215
|
database_view: :extract_view_file,
|
|
214
216
|
caching: :extract_caching_file,
|
|
@@ -290,7 +292,7 @@ module Woods
|
|
|
290
292
|
method: :extract_from_runtime_type, reconcile_removals: false }
|
|
291
293
|
}.freeze
|
|
292
294
|
|
|
293
|
-
# Extractors
|
|
295
|
+
# Extractors requiring a complete source set: they scan the whole app (or
|
|
294
296
|
# introspect the whole runtime) in one pass, so an incremental run
|
|
295
297
|
# replaces their output wholesale rather than per unit. Before #164
|
|
296
298
|
# these types were simply skipped by incremental runs while
|
|
@@ -299,6 +301,7 @@ module Woods
|
|
|
299
301
|
#
|
|
300
302
|
# @return [Hash{Symbol => Symbol}] extractor key => unit type
|
|
301
303
|
WHOLE_APP_EXTRACTORS = {
|
|
304
|
+
rake_tasks: :rake_task,
|
|
302
305
|
routes: :route,
|
|
303
306
|
middleware: :middleware,
|
|
304
307
|
engines: :engine,
|
|
@@ -355,7 +358,7 @@ module Woods
|
|
|
355
358
|
# flat index — the output root also holds `generation.json`, `dumps/`,
|
|
356
359
|
# `tasks/`, `woods.sqlite3` and `payloads/` itself, none of which belong
|
|
357
360
|
# to a generation's payload.
|
|
358
|
-
PAYLOAD_FILES = %w[manifest.json dependency_graph.json graph_analysis.json SUMMARY.md].freeze
|
|
361
|
+
PAYLOAD_FILES = %w[manifest.json source_inputs.json dependency_graph.json graph_analysis.json SUMMARY.md].freeze
|
|
359
362
|
|
|
360
363
|
# Payload directories that are not per-type unit directories.
|
|
361
364
|
PAYLOAD_DIRS = %w[flows].freeze
|
|
@@ -394,7 +397,9 @@ module Woods
|
|
|
394
397
|
#
|
|
395
398
|
# @return [Hash] Results keyed by extractor type
|
|
396
399
|
def extract_all
|
|
400
|
+
profile_started = Process.clock_gettime(Process::CLOCK_MONOTONIC) if profiling?
|
|
397
401
|
setup_output_directory
|
|
402
|
+
profile_phase('source capture') { begin_source_inputs('full') }
|
|
398
403
|
ModelNameCache.reset!
|
|
399
404
|
# @package_resolver alone is not enough: #package_resolver builds
|
|
400
405
|
# through #extractor_for, which memoizes into @incremental_extractors.
|
|
@@ -422,29 +427,33 @@ module Woods
|
|
|
422
427
|
|
|
423
428
|
# Phase 1.5: Deduplicate results
|
|
424
429
|
Rails.logger.info '[Woods] Deduplicating results...'
|
|
425
|
-
deduplicate_results
|
|
430
|
+
profile_phase('deduplication') { deduplicate_results }
|
|
426
431
|
|
|
427
432
|
# Phase 1.6: Package membership. Runs before the graph is rebuilt so
|
|
428
433
|
# registration copies metadata[:package] onto the node (#280).
|
|
429
|
-
annotate_packages
|
|
434
|
+
profile_phase('package annotation') { annotate_packages }
|
|
430
435
|
|
|
431
436
|
# Rebuild the graph from deduped results. #164 gave DependencyGraph
|
|
432
437
|
# `#remove`/`#unregister`, so surgical removal is now possible — but a
|
|
433
438
|
# full extraction has just registered every unit including duplicates,
|
|
434
439
|
# and rebuilding from the deduped set is both cheaper and less
|
|
435
440
|
# error-prone than unwinding registrations one at a time.
|
|
436
|
-
|
|
437
|
-
|
|
441
|
+
profile_phase('graph rebuild') do
|
|
442
|
+
@dependency_graph = DependencyGraph.new
|
|
443
|
+
@results.each_value { |units| units.each { |u| @dependency_graph.register(u) } }
|
|
444
|
+
end
|
|
438
445
|
|
|
439
446
|
# Phase 2: Resolve dependents (reverse dependencies)
|
|
440
447
|
Rails.logger.info '[Woods] Resolving dependents...'
|
|
441
|
-
resolve_dependents
|
|
448
|
+
profile_phase('dependents') { resolve_dependents }
|
|
442
449
|
|
|
443
450
|
# Phase 3: Enrich with git data. Runs BEFORE analysis now: the
|
|
444
451
|
# volatile_dependencies report reads commit counts off graph nodes.
|
|
445
452
|
Rails.logger.info '[Woods] Enriching with git data...'
|
|
446
|
-
|
|
447
|
-
|
|
453
|
+
profile_phase('git enrichment') do
|
|
454
|
+
enrich_with_git_data
|
|
455
|
+
annotate_graph_with_git_data
|
|
456
|
+
end
|
|
448
457
|
|
|
449
458
|
# Phase 4: Graph analysis (PageRank, structural metrics)
|
|
450
459
|
Rails.logger.info '[Woods] Analyzing dependency graph...'
|
|
@@ -452,7 +461,7 @@ module Woods
|
|
|
452
461
|
|
|
453
462
|
# Phase 4.5: Normalize file_path to relative paths
|
|
454
463
|
Rails.logger.info '[Woods] Normalizing file paths...'
|
|
455
|
-
normalize_file_paths
|
|
464
|
+
profile_phase('path normalization') { normalize_file_paths }
|
|
456
465
|
|
|
457
466
|
# Phase 5: Write output
|
|
458
467
|
Rails.logger.info '[Woods] Writing output...'
|
|
@@ -462,7 +471,7 @@ module Woods
|
|
|
462
471
|
# run after write_results — the just-written set is what defines
|
|
463
472
|
# "legitimate" — and belongs to the full path only; the incremental path
|
|
464
473
|
# deletes through the graph instead. See {#sweep_orphaned_unit_files}.
|
|
465
|
-
sweep_orphaned_unit_files
|
|
474
|
+
profile_phase('orphan sweep') { sweep_orphaned_unit_files }
|
|
466
475
|
|
|
467
476
|
# Phase 5.5: Precompute request flows (opt-in). Must run AFTER
|
|
468
477
|
# write_results — FlowAssembler loads unit JSON from disk, so running
|
|
@@ -485,18 +494,23 @@ module Woods
|
|
|
485
494
|
profile_phase('flows') { precompute_flows }
|
|
486
495
|
end
|
|
487
496
|
|
|
488
|
-
|
|
489
|
-
|
|
497
|
+
profile_phase('graph write') do
|
|
498
|
+
write_dependency_graph
|
|
499
|
+
write_graph_analysis
|
|
500
|
+
end
|
|
490
501
|
profile_phase('manifest and summary') do
|
|
491
502
|
write_manifest
|
|
492
503
|
write_structural_summary
|
|
493
504
|
end
|
|
494
|
-
capture_snapshot
|
|
495
|
-
|
|
505
|
+
profile_phase('snapshot') { capture_snapshot }
|
|
506
|
+
@source_inputs.full_units(@results, consumers: @extractors)
|
|
507
|
+
publish_generation('full')
|
|
496
508
|
|
|
497
509
|
log_summary
|
|
498
510
|
|
|
499
511
|
@results
|
|
512
|
+
ensure
|
|
513
|
+
log_profile_total('full', profile_started)
|
|
500
514
|
end
|
|
501
515
|
|
|
502
516
|
# ══════════════════════════════════════════════════════════════════════
|
|
@@ -528,7 +542,8 @@ module Woods
|
|
|
528
542
|
# @param changed_files [Array<String>] List of changed file paths
|
|
529
543
|
# @return [Array<String>] Identifiers of units re-extracted, added, or removed
|
|
530
544
|
def extract_changed(changed_files)
|
|
531
|
-
|
|
545
|
+
profile_started = Process.clock_gettime(Process::CLOCK_MONOTONIC) if profiling?
|
|
546
|
+
prepare_incremental_run(operation: 'incremental')
|
|
532
547
|
|
|
533
548
|
change_set = ChangeSet.new(paths: changed_files, root: Rails.root)
|
|
534
549
|
affected_types = Set.new
|
|
@@ -551,37 +566,41 @@ module Woods
|
|
|
551
566
|
acc
|
|
552
567
|
end
|
|
553
568
|
|
|
554
|
-
|
|
555
|
-
|
|
556
|
-
|
|
557
|
-
|
|
558
|
-
|
|
559
|
-
|
|
560
|
-
|
|
561
|
-
|
|
562
|
-
|
|
563
|
-
|
|
564
|
-
|
|
565
|
-
|
|
566
|
-
|
|
567
|
-
|
|
568
|
-
|
|
569
|
-
|
|
570
|
-
|
|
571
|
-
|
|
572
|
-
|
|
573
|
-
|
|
574
|
-
|
|
575
|
-
|
|
576
|
-
|
|
577
|
-
|
|
578
|
-
|
|
579
|
-
|
|
580
|
-
|
|
581
|
-
|
|
582
|
-
|
|
583
|
-
|
|
584
|
-
|
|
569
|
+
profile_phase('reconciliation') do
|
|
570
|
+
touched.merge(reconcile_class_based_types(affected_types))
|
|
571
|
+
touched.merge(reconcile_model_mixins(affected_types))
|
|
572
|
+
touched.merge(rerun_whole_app_extractors(change_set, affected_types))
|
|
573
|
+
touched.merge(reannotate_packages(change_set, affected_types))
|
|
574
|
+
pruned = prune_vanished_units(change_set, affected_types)
|
|
575
|
+
touched.merge(pruned)
|
|
576
|
+
|
|
577
|
+
# Reconcile once more, because pruning can un-know a class the first pass
|
|
578
|
+
# skipped. A class-based file moved between autoload directories with its
|
|
579
|
+
# constant unchanged is still registered under the old path when
|
|
580
|
+
# reconciliation runs, so it looks known and is not re-extracted; the
|
|
581
|
+
# prune that follows then removes it for its vanished path. This pass
|
|
582
|
+
# re-adds it in the same run (M1) instead of leaving the unit missing
|
|
583
|
+
# until some later run happens to notice. Idempotent when nothing was
|
|
584
|
+
# pruned: the discovery set is compared against the graph, so an
|
|
585
|
+
# already-registered class is skipped.
|
|
586
|
+
#
|
|
587
|
+
# But not everything pruning removed may come back. `except:` keeps the
|
|
588
|
+
# *deletion* shape pruned: without a reload, a constant outlives the file
|
|
589
|
+
# that defined it — so deleting `app/models/user.rb` prunes `User`, and
|
|
590
|
+
# this pass finds `User` still in `ActiveRecord::Base.descendants`.
|
|
591
|
+
# Re-registering it would pin the unit to a path that no longer exists,
|
|
592
|
+
# and nothing could ever remove it: the sweep excludes class-based units
|
|
593
|
+
# and no future change set names that path again. A resident daemon
|
|
594
|
+
# processing a batch before its reload hits this every time. What
|
|
595
|
+
# separates the two shapes is the filesystem — only pruned identifiers
|
|
596
|
+
# that a still-existing file in the change set actually declares are
|
|
597
|
+
# re-addable. See {#readdable_pruned_classes}.
|
|
598
|
+
touched.merge(reconcile_class_based_types(
|
|
599
|
+
affected_types, except: pruned - readdable_pruned_classes(pruned, change_set)
|
|
600
|
+
))
|
|
601
|
+
end
|
|
602
|
+
|
|
603
|
+
raise_on_handled_extraction_failure!
|
|
585
604
|
finalize_incremental_unit_json(affected_types)
|
|
586
605
|
|
|
587
606
|
# Regenerate type indexes for affected types
|
|
@@ -594,6 +613,8 @@ module Woods
|
|
|
594
613
|
finalize_incremental_run(touched)
|
|
595
614
|
|
|
596
615
|
touched.to_a
|
|
616
|
+
ensure
|
|
617
|
+
log_profile_total('incremental', profile_started)
|
|
597
618
|
end
|
|
598
619
|
|
|
599
620
|
# ══════════════════════════════════════════════════════════════════════
|
|
@@ -628,6 +649,7 @@ module Woods
|
|
|
628
649
|
# extractor
|
|
629
650
|
# @raise [ArgumentError] when no recognized key is given
|
|
630
651
|
def refresh(*keys)
|
|
652
|
+
profile_started = Process.clock_gettime(Process::CLOCK_MONOTONIC) if profiling?
|
|
631
653
|
keys = Array(keys).flatten.map(&:to_sym).uniq
|
|
632
654
|
known, unknown = keys.partition { |key| EXTRACTORS.key?(key) }
|
|
633
655
|
raise ArgumentError, "No known extractor in #{keys.inspect}" if known.empty?
|
|
@@ -635,17 +657,20 @@ module Woods
|
|
|
635
657
|
known += ROUTE_CONSUMER_EXTRACTORS if known.include?(:routes)
|
|
636
658
|
known.uniq!
|
|
637
659
|
|
|
638
|
-
prepare_incremental_run
|
|
660
|
+
prepare_incremental_run(operation: 'refresh')
|
|
639
661
|
affected_types = Set.new
|
|
640
662
|
touched = known.each_with_object(Set.new) do |key, acc|
|
|
641
663
|
acc.merge(replace_type_wholesale(key, affected_types))
|
|
642
664
|
end
|
|
643
665
|
|
|
666
|
+
raise_on_handled_extraction_failure!
|
|
644
667
|
finalize_incremental_unit_json(affected_types)
|
|
645
|
-
affected_types.each { |type_key| regenerate_type_index(type_key) }
|
|
668
|
+
profile_phase('type index') { affected_types.each { |type_key| regenerate_type_index(type_key) } }
|
|
646
669
|
finalize_incremental_run(touched, reason: "refresh:#{known.sort.join(',')}")
|
|
647
670
|
|
|
648
671
|
{ types: known, touched: touched.to_a, unknown: unknown }
|
|
672
|
+
ensure
|
|
673
|
+
log_profile_total('refresh', profile_started)
|
|
649
674
|
end
|
|
650
675
|
|
|
651
676
|
# Raise when the most recent extraction run wrote a payload but could not
|
|
@@ -667,6 +692,15 @@ module Woods
|
|
|
667
692
|
|
|
668
693
|
private
|
|
669
694
|
|
|
695
|
+
# Whole-run wall time, including unprofiled setup and failed runs. This
|
|
696
|
+
# separate log family must never be added to the individual phase times.
|
|
697
|
+
def log_profile_total(name, started)
|
|
698
|
+
return unless started
|
|
699
|
+
|
|
700
|
+
elapsed = Process.clock_gettime(Process::CLOCK_MONOTONIC) - started
|
|
701
|
+
Rails.logger.info "[Woods] [profile total] #{name} in #{elapsed.round(2)}s"
|
|
702
|
+
end
|
|
703
|
+
|
|
670
704
|
# Time one phase of a run and log how long it took, when WOODS_PROFILE=1.
|
|
671
705
|
#
|
|
672
706
|
# The per-extractor lines (see {#extract_all_sequential}) already report
|
|
@@ -745,7 +779,8 @@ module Woods
|
|
|
745
779
|
#
|
|
746
780
|
# @return [void]
|
|
747
781
|
# @raise [Woods::ExtractionError] see {#begin_payload!}
|
|
748
|
-
def prepare_incremental_run
|
|
782
|
+
def prepare_incremental_run(operation: 'incremental')
|
|
783
|
+
profile_phase('source capture') { begin_source_inputs(operation) }
|
|
749
784
|
profile_phase('payload seed') { begin_payload!(strict: true) }
|
|
750
785
|
graph_path = payload_dir.join('dependency_graph.json')
|
|
751
786
|
ensure_incremental_baseline!(graph_path)
|
|
@@ -795,7 +830,7 @@ module Woods
|
|
|
795
830
|
# reading "incremental" after a `woods:refresh[routes]` is being misled
|
|
796
831
|
# @return [void]
|
|
797
832
|
def finalize_incremental_run(touched, reason: 'incremental')
|
|
798
|
-
write_dependency_graph
|
|
833
|
+
profile_phase('graph write') { write_dependency_graph }
|
|
799
834
|
|
|
800
835
|
if touched.empty?
|
|
801
836
|
Rails.logger.info '[Woods] Incremental run changed nothing — leaving manifest timestamp untouched'
|
|
@@ -808,7 +843,7 @@ module Woods
|
|
|
808
843
|
write_manifest(incremental: true)
|
|
809
844
|
write_structural_summary
|
|
810
845
|
end
|
|
811
|
-
|
|
846
|
+
publish_generation(reason)
|
|
812
847
|
|
|
813
848
|
return unless Woods.configuration.enable_snapshots
|
|
814
849
|
|
|
@@ -831,9 +866,10 @@ module Woods
|
|
|
831
866
|
# Resolve (and if necessary rename) the payload first, so the flush
|
|
832
867
|
# below covers the directory under the name the pointer will carry.
|
|
833
868
|
payload = publishable_payload_name(generation)
|
|
869
|
+
profile_phase('source verification') { write_source_inputs } if payload
|
|
834
870
|
profile_phase('payload sync') { sync_payload }
|
|
835
|
-
marker = generation.bump!(reason: reason, payload: payload)
|
|
836
|
-
prune_payloads(marker.number)
|
|
871
|
+
marker = profile_phase('publish') { generation.bump!(reason: reason, payload: payload) }
|
|
872
|
+
profile_phase('payload prune') { prune_payloads(marker.number) }
|
|
837
873
|
marker
|
|
838
874
|
rescue StandardError => e
|
|
839
875
|
# A failed bump must not fail the extraction that produced a perfectly
|
|
@@ -852,6 +888,61 @@ module Woods
|
|
|
852
888
|
nil
|
|
853
889
|
end
|
|
854
890
|
|
|
891
|
+
# Capture before eager loading or extraction; only an explicit fresh-launch
|
|
892
|
+
# handoff can additionally establish the pre-Bundler/Rails boot boundary.
|
|
893
|
+
def begin_source_inputs(operation)
|
|
894
|
+
@failed_consumers = Set.new
|
|
895
|
+
@source_inputs = SourceInputs::Session.new(root: Rails.root, output_dir: @output_dir,
|
|
896
|
+
baseline_path: source_input_baseline_path,
|
|
897
|
+
operation: operation)
|
|
898
|
+
end
|
|
899
|
+
|
|
900
|
+
def source_input_baseline_path
|
|
901
|
+
generation = Generation.new(output_dir: @output_dir)
|
|
902
|
+
marker = generation.current
|
|
903
|
+
directory = generation.payload_dir(marker)
|
|
904
|
+
return nil unless marker.payload && directory != generation.root
|
|
905
|
+
|
|
906
|
+
directory.join(SourceInputs::Manifest::FILE_NAME)
|
|
907
|
+
rescue TypeError, NoMethodError
|
|
908
|
+
nil
|
|
909
|
+
end
|
|
910
|
+
|
|
911
|
+
def source_consumer_failed?(key, consumer = extractor_for(key))
|
|
912
|
+
failed = consumer.nil? || SourceInputs::ConsumerErrors.failed?(consumer)
|
|
913
|
+
@source_inputs&.unverified("extractor:#{key}") if failed
|
|
914
|
+
failed
|
|
915
|
+
end
|
|
916
|
+
|
|
917
|
+
# A rescued consumer error is not a successful empty result. Reset the
|
|
918
|
+
# per-call flag because one extractor instance serves several paths, but
|
|
919
|
+
# retain the failed scope for the run so a later success cannot authorize
|
|
920
|
+
# publication. Watch then carries the complete batch forward for retry.
|
|
921
|
+
def checked_extraction(key, consumer)
|
|
922
|
+
(@failed_consumers ||= Set.new).add(key) if SourceInputs::ConsumerErrors.failed?(consumer)
|
|
923
|
+
SourceInputs::ConsumerErrors.reset(consumer)
|
|
924
|
+
result = yield
|
|
925
|
+
return result unless source_consumer_failed?(key, consumer)
|
|
926
|
+
|
|
927
|
+
(@failed_consumers ||= Set.new).add(key)
|
|
928
|
+
nil
|
|
929
|
+
end
|
|
930
|
+
|
|
931
|
+
def raise_on_handled_extraction_failure!
|
|
932
|
+
return if @failed_consumers.nil? || @failed_consumers.empty?
|
|
933
|
+
|
|
934
|
+
raise Woods::ExtractionError,
|
|
935
|
+
"Extraction failed for #{@failed_consumers.to_a.sort.join(', ')}; " \
|
|
936
|
+
'the previous generation remains active. Fix the logged source errors and retry the complete batch.'
|
|
937
|
+
end
|
|
938
|
+
|
|
939
|
+
def write_source_inputs
|
|
940
|
+
return unless @source_inputs
|
|
941
|
+
|
|
942
|
+
manifest = @source_inputs.finish(generation: @payload_generation, eager_load_complete: @eager_load_complete)
|
|
943
|
+
AtomicFile.write(payload_dir.join(SourceInputs::Manifest::FILE_NAME), JSON.pretty_generate(manifest.data))
|
|
944
|
+
end
|
|
945
|
+
|
|
855
946
|
# Open the payload directory this run publishes into, seeded from the
|
|
856
947
|
# generation currently on disk.
|
|
857
948
|
#
|
|
@@ -1268,9 +1359,6 @@ module Woods
|
|
|
1268
1359
|
|
|
1269
1360
|
def setup_output_directory
|
|
1270
1361
|
FileUtils.mkdir_p(@output_dir)
|
|
1271
|
-
EXTRACTORS.each_key do |type|
|
|
1272
|
-
FileUtils.mkdir_p(payload_dir.join(type.to_s))
|
|
1273
|
-
end
|
|
1274
1362
|
end
|
|
1275
1363
|
|
|
1276
1364
|
# ──────────────────────────────────────────────────────────────────────
|
|
@@ -1370,8 +1458,10 @@ module Woods
|
|
|
1370
1458
|
def same_type_collision_message(type, unit, prior_path)
|
|
1371
1459
|
"same-type identifier collision: #{type.to_s.singularize} '#{unit.identifier}' derived from " \
|
|
1372
1460
|
"two different sources ('#{prior_path || 'no file'}' and '#{unit.file_path || 'no file'}'); " \
|
|
1373
|
-
'only one unit could ever be indexed, so extraction aborted
|
|
1374
|
-
'
|
|
1461
|
+
'only one unit could ever be indexed, so extraction aborted. ' \
|
|
1462
|
+
'Wrapper-nested class naming requires Zeitwerk mode with Zeitwerk >= 2.6.9; on older loaders or ' \
|
|
1463
|
+
'classic-mode hosts, check that support before changing valid namespace wrappers. ' \
|
|
1464
|
+
'For a genuine duplicate, merge the declarations into one file or split them into distinct constants'
|
|
1375
1465
|
end
|
|
1376
1466
|
|
|
1377
1467
|
# ──────────────────────────────────────────────────────────────────────
|
|
@@ -1753,6 +1843,7 @@ module Woods
|
|
|
1753
1843
|
GraphAnalyzer.new(
|
|
1754
1844
|
@dependency_graph,
|
|
1755
1845
|
volatile_ratio: ratio,
|
|
1846
|
+
volatile_limit_per_target: config&.volatile_dependency_limit_per_target,
|
|
1756
1847
|
cycle_limit: config ? config.graph_cycle_limit : GraphAnalyzer::DEFAULT_CYCLE_LIMIT,
|
|
1757
1848
|
cycle_max_length: config ? config.graph_cycle_max_length : GraphAnalyzer::DEFAULT_CYCLE_MAX_LENGTH
|
|
1758
1849
|
)
|
|
@@ -1869,11 +1960,9 @@ module Woods
|
|
|
1869
1960
|
# Is this a path worth asking git about?
|
|
1870
1961
|
#
|
|
1871
1962
|
# A gem-owned unit (an engine model) carries its real path. Outside
|
|
1872
|
-
# Rails.root,
|
|
1873
|
-
# outside the repository — one gem path would erase the git metadata of
|
|
1874
|
-
# the other 499 units in its 500-path batch. Inside Rails.root, a bundle
|
|
1963
|
+
# Rails.root, it has no app repository history. Inside Rails.root, a bundle
|
|
1875
1964
|
# vendored at `vendor/bundle` puts the same gem files under the root
|
|
1876
|
-
# prefix, gitignored, so
|
|
1965
|
+
# prefix, gitignored, so requesting their history serves no app-owned unit.
|
|
1877
1966
|
# Same exclusions as {Extractors::SharedUtilityMethods#app_source?}.
|
|
1878
1967
|
#
|
|
1879
1968
|
# @param path [String, nil] absolute file path
|
|
@@ -1927,13 +2016,13 @@ module Woods
|
|
|
1927
2016
|
# exactly this value shape — no fractional seconds, `Z` or a `±hh:mm`
|
|
1928
2017
|
# offset — and `spec/extracted_unit_spec.rb` pins that, so a change to the
|
|
1929
2018
|
# stamp's shape fails a spec instead of quietly un-matching this mask.
|
|
1930
|
-
#
|
|
1931
|
-
#
|
|
1932
|
-
#
|
|
1933
|
-
#
|
|
1934
|
-
# top level.
|
|
2019
|
+
# Match only the final top-level stamp, followed by the source_hash field
|
|
2020
|
+
# and the document's closing brace. Nested metadata may use the same key
|
|
2021
|
+
# and timestamp shape; changing it must still rewrite the unit. Escaped
|
|
2022
|
+
# quotes inside string values cannot match these JSON field boundaries.
|
|
1935
2023
|
EXTRACTED_AT_SCALAR =
|
|
1936
|
-
/("extracted_at":\s*")\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}(?:Z|[+-]\d{2}:\d{2})
|
|
2024
|
+
/("extracted_at":\s*")\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}(?:Z|[+-]\d{2}:\d{2})
|
|
2025
|
+
(?=",\s*"source_hash":\s*"[0-9a-f]{64}"\s*}\s*\z)/x
|
|
1937
2026
|
# An implementation detail of the byte comparison, not part of the
|
|
1938
2027
|
# extractor's surface (`private` does not scope constants).
|
|
1939
2028
|
private_constant :EXTRACTED_AT_SCALAR
|
|
@@ -1967,7 +2056,7 @@ module Woods
|
|
|
1967
2056
|
# original encoding
|
|
1968
2057
|
# @return [String] the bytes with the stamp's value removed
|
|
1969
2058
|
def mask_extracted_at(bytes)
|
|
1970
|
-
bytes.
|
|
2059
|
+
bytes.sub(EXTRACTED_AT_SCALAR, '\1')
|
|
1971
2060
|
end
|
|
1972
2061
|
|
|
1973
2062
|
def normalize_file_paths
|
|
@@ -1999,7 +2088,7 @@ module Woods
|
|
|
1999
2088
|
# to say. Enrichment then wrote `commit_count: 0` and
|
|
2000
2089
|
# `change_frequency: new` onto every unit, which reads exactly like a file
|
|
2001
2090
|
# that was never committed, where an absent git directory correctly omits
|
|
2002
|
-
# the keys (B-186). HEAD has to resolve.
|
|
2091
|
+
# the keys (B-186). HEAD has to resolve, with complete ancestry (B-189).
|
|
2003
2092
|
#
|
|
2004
2093
|
# Memoized, so the warning below is emitted at most once per run.
|
|
2005
2094
|
#
|
|
@@ -2008,13 +2097,31 @@ module Woods
|
|
|
2008
2097
|
return @git_available if defined?(@git_available)
|
|
2009
2098
|
|
|
2010
2099
|
_output, error, status = Open3.capture3(*git_argv('rev-parse', 'HEAD'))
|
|
2011
|
-
|
|
2012
|
-
|
|
2013
|
-
|
|
2100
|
+
unless status.success?
|
|
2101
|
+
warn_unresolvable_git(error)
|
|
2102
|
+
return @git_available = false
|
|
2103
|
+
end
|
|
2104
|
+
|
|
2105
|
+
@git_available = complete_git_history?
|
|
2014
2106
|
rescue StandardError
|
|
2015
2107
|
@git_available = false
|
|
2016
2108
|
end
|
|
2017
2109
|
|
|
2110
|
+
# A shallow HEAD resolves but represents an incomplete ancestry. Do not
|
|
2111
|
+
# turn that boundary into apparent one-commit/new-file churn facts.
|
|
2112
|
+
def complete_git_history?
|
|
2113
|
+
output, _error, status = Open3.capture3(*git_argv('rev-parse', '--is-shallow-repository'))
|
|
2114
|
+
return true if status.success? && output.strip == 'false'
|
|
2115
|
+
|
|
2116
|
+
shallow = status.success? && output.strip == 'true'
|
|
2117
|
+
reason = shallow ? 'shallow repository' : 'repository depth could not be verified'
|
|
2118
|
+
Rails.logger.warn(
|
|
2119
|
+
"[Woods] Git enrichment omitted: #{reason}. Fetch full history with git fetch --unshallow " \
|
|
2120
|
+
'(or actions/checkout fetch-depth: 0), then run full extraction to refresh git metadata.'
|
|
2121
|
+
)
|
|
2122
|
+
false
|
|
2123
|
+
end
|
|
2124
|
+
|
|
2018
2125
|
# Say once why no unit will carry git metadata, but only when there is a
|
|
2019
2126
|
# working tree to explain. No `.git` at the root is the ordinary source
|
|
2020
2127
|
# tarball or `COPY`-without-`.git` case, and it is not a fault.
|
|
@@ -2058,63 +2165,19 @@ module Woods
|
|
|
2058
2165
|
''
|
|
2059
2166
|
end
|
|
2060
2167
|
|
|
2061
|
-
#
|
|
2062
|
-
#
|
|
2063
|
-
#
|
|
2064
|
-
#
|
|
2065
|
-
# batch also sent. The result is keyed by relative path, so the output
|
|
2066
|
-
# is identical.
|
|
2067
|
-
#
|
|
2068
|
-
# @param file_paths [Array<String>] Absolute file paths
|
|
2069
|
-
# @return [Hash{String => Hash}] Keyed by relative path
|
|
2168
|
+
# One HEAD history walk, independent of requested-path grouping. See
|
|
2169
|
+
# GitHistory for explicit merge semantics and binary record framing.
|
|
2170
|
+
# @param file_paths [Array<String>] absolute file paths
|
|
2171
|
+
# @return [Hash{String => Hash}] keyed by Rails.root-relative path
|
|
2070
2172
|
def batch_git_data(file_paths)
|
|
2071
2173
|
return {} if file_paths.empty?
|
|
2072
2174
|
|
|
2073
|
-
|
|
2074
|
-
|
|
2075
|
-
|
|
2076
|
-
|
|
2077
|
-
|
|
2078
|
-
path_set = relative_paths.to_set
|
|
2079
|
-
relative_paths.each_slice(500) do |batch|
|
|
2080
|
-
log_output = run_git(
|
|
2081
|
-
'log', '--all', '--name-only',
|
|
2082
|
-
'--format=__COMMIT__%H|||%an|||%cI|||%s',
|
|
2083
|
-
'--since=365 days ago',
|
|
2084
|
-
'--', *batch
|
|
2085
|
-
)
|
|
2086
|
-
parse_git_log_output(log_output, path_set, result)
|
|
2087
|
-
end
|
|
2088
|
-
|
|
2089
|
-
ninety_days_ago = (Time.current - 90.days).iso8601
|
|
2090
|
-
result.each do |relative_path, data|
|
|
2091
|
-
result[relative_path] = build_file_metadata(data, ninety_days_ago)
|
|
2092
|
-
end
|
|
2093
|
-
|
|
2094
|
-
result
|
|
2095
|
-
end
|
|
2096
|
-
|
|
2097
|
-
# Parse git log output line-by-line, populating result with per-file commit data.
|
|
2098
|
-
def parse_git_log_output(log_output, path_set, result)
|
|
2099
|
-
current_commit = nil
|
|
2100
|
-
|
|
2101
|
-
log_output.each_line do |line|
|
|
2102
|
-
line = line.strip
|
|
2103
|
-
next if line.empty?
|
|
2175
|
+
relative_paths = file_paths.map { |path| normalize_file_path(path) }.uniq
|
|
2176
|
+
recent_after = Time.current - 90.days
|
|
2177
|
+
raw = GitHistory.new(root: Rails.root, logger: Rails.logger).read(relative_paths, recent_after: recent_after)
|
|
2178
|
+
return {} unless raw
|
|
2104
2179
|
|
|
2105
|
-
|
|
2106
|
-
parts = line.sub('__COMMIT__', '').split('|||', 4)
|
|
2107
|
-
current_commit = { sha: parts[0], author: parts[1], date: parts[2], message: parts[3] }
|
|
2108
|
-
elsif current_commit && path_set.include?(line)
|
|
2109
|
-
entry = result[line] ||= {}
|
|
2110
|
-
unless entry[:last_modified]
|
|
2111
|
-
entry[:last_modified] = current_commit[:date]
|
|
2112
|
-
entry[:last_author] = current_commit[:author]
|
|
2113
|
-
end
|
|
2114
|
-
(entry[:commits] ||= []) << current_commit
|
|
2115
|
-
(entry[:contributors] ||= Hash.new(0))[current_commit[:author]] += 1
|
|
2116
|
-
end
|
|
2117
|
-
end
|
|
2180
|
+
raw.transform_values { |data| build_file_metadata(data, recent_after.iso8601) }
|
|
2118
2181
|
end
|
|
2119
2182
|
|
|
2120
2183
|
# Classify how frequently a file changes based on commit counts.
|
|
@@ -2136,12 +2199,13 @@ module Woods
|
|
|
2136
2199
|
def build_file_metadata(data, ninety_days_ago)
|
|
2137
2200
|
all_commits = data[:commits] || []
|
|
2138
2201
|
contributor_counts = data[:contributors] || {}
|
|
2139
|
-
recent_count = all_commits.count { |c| c[:date] && c[:date] > ninety_days_ago }
|
|
2202
|
+
recent_count = data.fetch(:recent_count) { all_commits.count { |c| c[:date] && c[:date] > ninety_days_ago } }
|
|
2203
|
+
total_count = data.fetch(:commit_count, all_commits.size)
|
|
2140
2204
|
|
|
2141
2205
|
{
|
|
2142
2206
|
last_modified: data[:last_modified],
|
|
2143
2207
|
last_author: data[:last_author],
|
|
2144
|
-
commit_count:
|
|
2208
|
+
commit_count: total_count,
|
|
2145
2209
|
contributors: contributor_counts
|
|
2146
2210
|
.sort_by { |_, count| -count }
|
|
2147
2211
|
.first(5)
|
|
@@ -2149,7 +2213,7 @@ module Woods
|
|
|
2149
2213
|
recent_commits: all_commits.first(5).map do |c|
|
|
2150
2214
|
{ sha: c[:sha]&.first(8), message: c[:message], date: c[:date], author: c[:author] }
|
|
2151
2215
|
end,
|
|
2152
|
-
change_frequency: classify_change_frequency(
|
|
2216
|
+
change_frequency: classify_change_frequency(total_count, recent_count)
|
|
2153
2217
|
}
|
|
2154
2218
|
end
|
|
2155
2219
|
|
|
@@ -2324,6 +2388,7 @@ module Woods
|
|
|
2324
2388
|
|
|
2325
2389
|
manifest = {
|
|
2326
2390
|
extracted_at: Time.current.iso8601,
|
|
2391
|
+
woods_version: Woods::VERSION,
|
|
2327
2392
|
rails_version: Rails.version,
|
|
2328
2393
|
ruby_version: RUBY_VERSION,
|
|
2329
2394
|
|
|
@@ -2697,6 +2762,7 @@ module Woods
|
|
|
2697
2762
|
|
|
2698
2763
|
@incremental_extractors[key] = EXTRACTORS[key]&.new
|
|
2699
2764
|
rescue StandardError => e
|
|
2765
|
+
@source_inputs&.unverified("extractor:#{key}")
|
|
2700
2766
|
Rails.logger.warn "[Woods] Could not build #{key} extractor: #{e.message}"
|
|
2701
2767
|
@incremental_extractors[key] = nil
|
|
2702
2768
|
end
|
|
@@ -2719,10 +2785,11 @@ module Woods
|
|
|
2719
2785
|
#
|
|
2720
2786
|
# This is the fix for #164 gap 1 (a path the index has never seen routed
|
|
2721
2787
|
# nowhere, so new files were silently ignored) and the per-path half of
|
|
2722
|
-
# gap 3 (a file
|
|
2723
|
-
#
|
|
2724
|
-
#
|
|
2725
|
-
#
|
|
2788
|
+
# gap 3 (a file defining several units could only ever resolve to one
|
|
2789
|
+
# identifier). Rake tasks need the wholesale path because definitions
|
|
2790
|
+
# of one task can span several files.
|
|
2791
|
+
# Reconciling the whole path removes definitions deleted from a surviving
|
|
2792
|
+
# file rather than leaving them behind.
|
|
2726
2793
|
#
|
|
2727
2794
|
# Removal is scoped to the unit types the matching rules could have
|
|
2728
2795
|
# produced, so a class-based unit sharing the path (the `User` model unit
|
|
@@ -2758,6 +2825,10 @@ module Woods
|
|
|
2758
2825
|
# one method over — CORE-1).
|
|
2759
2826
|
produced.merge(units.map { |unit| [unit.identifier, unit.type] })
|
|
2760
2827
|
touched.merge(register_and_write(rule.extractor_key, units, affected_types))
|
|
2828
|
+
unless source_consumer_failed?(rule.extractor_key)
|
|
2829
|
+
@source_inputs&.consume_file(rule.extractor_key,
|
|
2830
|
+
absolute_path)
|
|
2831
|
+
end
|
|
2761
2832
|
end
|
|
2762
2833
|
|
|
2763
2834
|
next if raised
|
|
@@ -2786,10 +2857,13 @@ module Woods
|
|
|
2786
2857
|
# on every changed path of that type, with the generation bumped over
|
|
2787
2858
|
# the loss. Construction failure tells us nothing about the path; only
|
|
2788
2859
|
# a genuinely constructed extractor that lacks the method earns the [].
|
|
2789
|
-
|
|
2860
|
+
if extractor.nil?
|
|
2861
|
+
source_consumer_failed?(rule.extractor_key, extractor)
|
|
2862
|
+
return nil
|
|
2863
|
+
end
|
|
2790
2864
|
return [] unless extractor.respond_to?(rule.method_name)
|
|
2791
2865
|
|
|
2792
|
-
result =
|
|
2866
|
+
result = checked_extraction(rule.extractor_key, extractor) do
|
|
2793
2867
|
if rule.extractor_key == :poros
|
|
2794
2868
|
# PoroExtractor needs the AR name set to reject persisted models;
|
|
2795
2869
|
# its default is an empty set, which would misfile every model
|
|
@@ -2798,9 +2872,12 @@ module Woods
|
|
|
2798
2872
|
else
|
|
2799
2873
|
extractor.public_send(rule.method_name, absolute_path)
|
|
2800
2874
|
end
|
|
2875
|
+
end
|
|
2876
|
+
return nil if result.nil? && SourceInputs::ConsumerErrors.failed?(extractor)
|
|
2801
2877
|
|
|
2802
2878
|
Array(result).compact
|
|
2803
2879
|
rescue StandardError => e
|
|
2880
|
+
@source_inputs&.unverified("extractor:#{rule.extractor_key}")
|
|
2804
2881
|
Rails.logger.warn "[Woods] #{rule.extractor_key} re-extraction of #{absolute_path} failed: #{e.message}"
|
|
2805
2882
|
# `nil`, not `[]`. The caller treats an empty result as "this path defines
|
|
2806
2883
|
# nothing any more" and prunes the units previously registered to it — so
|
|
@@ -2871,6 +2948,34 @@ module Woods
|
|
|
2871
2948
|
touched
|
|
2872
2949
|
end
|
|
2873
2950
|
|
|
2951
|
+
# Runtime-only model mixins can enter or leave discovery when their
|
|
2952
|
+
# includer changes, even if the mixin file itself is untouched.
|
|
2953
|
+
# @param affected_types [Set<Symbol>]
|
|
2954
|
+
# @return [Set<String>] Added or removed concern identifiers
|
|
2955
|
+
def reconcile_model_mixins(affected_types)
|
|
2956
|
+
extractor = extractor_for(:concerns)
|
|
2957
|
+
return Set.new unless extractor.respond_to?(:runtime_model_mixins)
|
|
2958
|
+
|
|
2959
|
+
live = extractor.runtime_model_mixins
|
|
2960
|
+
known = @dependency_graph.units_of_type(:concern).to_set
|
|
2961
|
+
added = live.flat_map do |path, modules|
|
|
2962
|
+
next [] if modules.all? { |mod| known.include?(mod.name) }
|
|
2963
|
+
|
|
2964
|
+
Array(extractor.extract_model_mixin_file(path)).reject { |unit| known.include?(unit.identifier) }
|
|
2965
|
+
end
|
|
2966
|
+
touched = register_and_write(:concerns, added, affected_types)
|
|
2967
|
+
return touched unless @eager_load_complete
|
|
2968
|
+
|
|
2969
|
+
live_names = live.values.flatten.to_set(&:name)
|
|
2970
|
+
known.each do |identifier|
|
|
2971
|
+
path = @dependency_graph.node(identifier, type: :concern)[:file_path]
|
|
2972
|
+
next if extractor.conventional_concern_path?(path) || live_names.include?(identifier)
|
|
2973
|
+
|
|
2974
|
+
touched.add(identifier) if remove_unit(identifier, affected_types, type: :concern)
|
|
2975
|
+
end
|
|
2976
|
+
touched
|
|
2977
|
+
end
|
|
2978
|
+
|
|
2874
2979
|
# Pruned class-based identifiers the tree still governs, and that the
|
|
2875
2980
|
# second reconciliation pass may therefore re-add.
|
|
2876
2981
|
#
|
|
@@ -2948,10 +3053,12 @@ module Woods
|
|
|
2948
3053
|
units = new_classes.filter_map do |klass|
|
|
2949
3054
|
extractor_for(key).public_send(spec[:method], klass)
|
|
2950
3055
|
rescue StandardError => e
|
|
3056
|
+
@source_inputs&.unverified("extractor:#{key}")
|
|
2951
3057
|
Rails.logger.warn "[Woods] #{key} extraction of #{klass} failed: #{e.message}"
|
|
2952
3058
|
nil
|
|
2953
3059
|
end
|
|
2954
3060
|
|
|
3061
|
+
source_consumer_failed?(key)
|
|
2955
3062
|
register_and_write(key, units, affected_types)
|
|
2956
3063
|
end
|
|
2957
3064
|
|
|
@@ -3086,14 +3193,23 @@ module Woods
|
|
|
3086
3193
|
# mutating durable state
|
|
3087
3194
|
def replace_type_wholesale(key, affected_types)
|
|
3088
3195
|
extractor = extractor_for(key)
|
|
3196
|
+
if extractor.nil?
|
|
3197
|
+
source_consumer_failed?(key, extractor)
|
|
3198
|
+
return Set.new
|
|
3199
|
+
end
|
|
3089
3200
|
return Set.new unless extractor.respond_to?(:extract_all)
|
|
3090
3201
|
|
|
3091
3202
|
@wholesale_mutations = 0
|
|
3092
|
-
|
|
3203
|
+
result = checked_extraction(key, extractor) { extractor.extract_all }
|
|
3204
|
+
return Set.new if SourceInputs::ConsumerErrors.failed?(extractor)
|
|
3205
|
+
|
|
3206
|
+
units = Array(result).compact.uniq(&:identifier)
|
|
3093
3207
|
Rails.logger.info "[Woods] Re-ran #{key} wholesale: #{units.size} units"
|
|
3094
3208
|
|
|
3095
3209
|
touched = register_and_write(key, units, affected_types)
|
|
3096
3210
|
touched.merge(remove_replaced_units(key, units, affected_types))
|
|
3211
|
+
@source_inputs&.consume_extractor(key, units) unless source_consumer_failed?(key, extractor)
|
|
3212
|
+
touched
|
|
3097
3213
|
rescue StandardError => e
|
|
3098
3214
|
if @wholesale_mutations.to_i.positive?
|
|
3099
3215
|
raise Woods::ExtractionError, <<~MSG.tr("\n", ' ').strip
|
|
@@ -3106,6 +3222,7 @@ module Woods
|
|
|
3106
3222
|
MSG
|
|
3107
3223
|
end
|
|
3108
3224
|
|
|
3225
|
+
@source_inputs&.unverified("extractor:#{key}")
|
|
3109
3226
|
Rails.logger.error "[Woods] Wholesale re-run of #{key} failed: #{e.message}"
|
|
3110
3227
|
Set.new
|
|
3111
3228
|
end
|
|
@@ -3267,6 +3384,7 @@ module Woods
|
|
|
3267
3384
|
|
|
3268
3385
|
removed.add(identifier) if remove_unit(identifier, affected_types, type: type)
|
|
3269
3386
|
end
|
|
3387
|
+
@source_inputs&.consume_deleted(path)
|
|
3270
3388
|
end
|
|
3271
3389
|
end
|
|
3272
3390
|
|
|
@@ -3370,6 +3488,7 @@ module Woods
|
|
|
3370
3488
|
(@incremental_written ||= {})[unit.identifier] = unit.file_path
|
|
3371
3489
|
|
|
3372
3490
|
write_unit_file(type_dir.join(collision_safe_filename(unit.identifier)), unit)
|
|
3491
|
+
@source_inputs&.consume_unit(extractor_key, unit.file_path) unless source_consumer_failed?(extractor_key)
|
|
3373
3492
|
written.add(unit.identifier)
|
|
3374
3493
|
end
|
|
3375
3494
|
end
|
|
@@ -3446,12 +3565,14 @@ module Woods
|
|
|
3446
3565
|
def finalize_incremental_unit_json(affected_types)
|
|
3447
3566
|
dependents_dirty = @dependents_dirty || Set.new
|
|
3448
3567
|
git_dirty = @incremental_written || {}
|
|
3449
|
-
git_data = incremental_git_data(git_dirty.keys)
|
|
3568
|
+
git_data = profile_phase('git enrichment') { incremental_git_data(git_dirty.keys) }
|
|
3450
3569
|
|
|
3451
|
-
(
|
|
3452
|
-
|
|
3453
|
-
|
|
3454
|
-
|
|
3570
|
+
profile_phase('unit finalization') do
|
|
3571
|
+
(dependents_dirty | git_dirty.keys).each do |identifier|
|
|
3572
|
+
rewrite_unit_json(identifier, affected_types,
|
|
3573
|
+
refresh_dependents: dependents_dirty.include?(identifier),
|
|
3574
|
+
git_data: git_dirty.key?(identifier) ? git_data : nil)
|
|
3575
|
+
end
|
|
3455
3576
|
end
|
|
3456
3577
|
end
|
|
3457
3578
|
|
|
@@ -3560,7 +3681,7 @@ module Woods
|
|
|
3560
3681
|
end
|
|
3561
3682
|
|
|
3562
3683
|
# Batch-fetch git metadata for the units written by this run, in a single
|
|
3563
|
-
#
|
|
3684
|
+
# history walk, keyed by Rails.root-relative path the way
|
|
3564
3685
|
# {#batch_git_data} returns it.
|
|
3565
3686
|
#
|
|
3566
3687
|
# @param identifiers [Array<String>]
|
|
@@ -3568,11 +3689,12 @@ module Woods
|
|
|
3568
3689
|
def incremental_git_data(identifiers)
|
|
3569
3690
|
return {} if identifiers.empty? || !git_available?
|
|
3570
3691
|
|
|
3692
|
+
root = "#{Rails.root}/"
|
|
3571
3693
|
paths = identifiers.flat_map do |identifier|
|
|
3572
3694
|
@dependency_graph.nodes_for(identifier).filter_map do |node|
|
|
3573
3695
|
next if %i[rails_source gem_source].include?(node[:type])
|
|
3574
3696
|
|
|
3575
|
-
node[:file_path] if
|
|
3697
|
+
node[:file_path] if git_enrichable_path?(node[:file_path], root)
|
|
3576
3698
|
end
|
|
3577
3699
|
end
|
|
3578
3700
|
|
|
@@ -3627,11 +3749,17 @@ module Woods
|
|
|
3627
3749
|
return nil unless extractor_key
|
|
3628
3750
|
|
|
3629
3751
|
extractor = extractor_for(extractor_key)
|
|
3630
|
-
|
|
3752
|
+
if extractor.nil?
|
|
3753
|
+
source_consumer_failed?(extractor_key, extractor)
|
|
3754
|
+
return nil
|
|
3755
|
+
end
|
|
3631
3756
|
|
|
3632
|
-
# File-based extractors can return several units from one file
|
|
3633
|
-
#
|
|
3634
|
-
|
|
3757
|
+
# File-based extractors can return several units from one file;
|
|
3758
|
+
# class-based extractors return one.
|
|
3759
|
+
result = checked_extraction(extractor_key, extractor) do
|
|
3760
|
+
re_extracted_units(extractor, type, unit_id, file_path, extractor_key)
|
|
3761
|
+
end
|
|
3762
|
+
units = Array(result).compact
|
|
3635
3763
|
return nil if units.empty?
|
|
3636
3764
|
|
|
3637
3765
|
register_and_write(extractor_key, units, affected_types)
|