woods 2.0.0.beta2 → 2.0.0.beta3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +262 -1
- data/CONTRIBUTING.md +173 -9
- data/README.md +7 -3
- data/SECURITY.md +9 -6
- data/docs/AGENT_GUIDE.md +83 -4
- data/docs/AGENT_SETUP.md +82 -1
- data/docs/BACKEND_MATRIX.md +20 -0
- data/docs/CLIENT_HOOKS.md +111 -0
- data/docs/CONFIGURATION_REFERENCE.md +199 -14
- data/docs/CONSOLE_MCP_SETUP.md +35 -5
- data/docs/DOCKER_SETUP.md +21 -2
- data/docs/EVALUATION.md +464 -1
- data/docs/EXTRACTOR_REFERENCE.md +36 -5
- data/docs/FAQ.md +11 -12
- data/docs/GETTING_STARTED.md +17 -5
- data/docs/INCREMENTAL_EXTRACTION.md +117 -1
- data/docs/INDEX_LAYOUT.md +382 -0
- data/docs/INTERNALS.md +7 -2
- data/docs/MCP_SERVERS.md +221 -5
- data/docs/MCP_TOOL_COOKBOOK.md +33 -18
- data/docs/NOTION_INTEGRATION.md +13 -0
- data/docs/OBSIDIAN_INTEGRATION.md +57 -9
- data/docs/PUBLISHED_INDEX.md +55 -0
- data/docs/README.md +7 -0
- data/docs/RETRIEVAL_GUIDE.md +253 -11
- data/docs/RUNTIME_TRACING.md +71 -0
- data/docs/SOURCE_FRESHNESS.md +143 -0
- data/docs/TROUBLESHOOTING.md +117 -5
- data/docs/UNBLOCKED_INTEGRATION.md +25 -0
- data/docs/UPGRADING_TO_2.md +44 -22
- data/docs/WATCH_DAEMON.md +259 -59
- data/exe/woods-agent-config +6 -0
- data/exe/woods-extract +5 -0
- data/exe/woods-hook-context +6 -0
- data/lib/generators/woods/templates/woods.rb.tt +1 -3
- data/lib/tasks/woods.rake +47 -397
- data/lib/woods/agent_configuration/applier.rb +133 -0
- data/lib/woods/agent_configuration/cli.rb +101 -0
- data/lib/woods/agent_configuration/cli_options.rb +29 -0
- data/lib/woods/agent_configuration/document.rb +105 -0
- data/lib/woods/agent_configuration/error.rb +7 -0
- data/lib/woods/agent_configuration/launcher.rb +75 -0
- data/lib/woods/agent_configuration/layout.rb +59 -0
- data/lib/woods/agent_configuration/managed_section.rb +62 -0
- data/lib/woods/agent_configuration/plan.rb +98 -0
- data/lib/woods/agent_configuration/plan_diff.rb +38 -0
- data/lib/woods/agent_configuration/planned_files.rb +61 -0
- data/lib/woods/agent_configuration/planner.rb +63 -0
- data/lib/woods/agent_configuration/planner_validation.rb +77 -0
- data/lib/woods/agent_configuration/preflight.rb +100 -0
- data/lib/woods/agent_configuration/recovery.rb +49 -0
- data/lib/woods/ast/node.rb +2 -0
- data/lib/woods/ast/parser.rb +38 -5
- data/lib/woods/builder.rb +21 -5
- data/lib/woods/cache/cache_middleware.rb +28 -7
- data/lib/woods/cache/cache_store.rb +4 -5
- data/lib/woods/change_set.rb +5 -4
- data/lib/woods/console/credential_index.rb +20 -2
- data/lib/woods/console/credential_scanner.rb +14 -14
- data/lib/woods/console/credential_scanner_registry.rb +36 -0
- data/lib/woods/console/embedded_executor.rb +1 -1
- data/lib/woods/console/encrypted_credential_snapshot.rb +16 -0
- data/lib/woods/console/rack_middleware.rb +22 -13
- data/lib/woods/console/server.rb +18 -16
- data/lib/woods/dependency_graph.rb +65 -13
- data/lib/woods/embedding/corpus.rb +94 -0
- data/lib/woods/embedding/indexer.rb +90 -46
- data/lib/woods/embedding/openai.rb +17 -6
- data/lib/woods/evaluation/ablation_executor.rb +6 -1
- data/lib/woods/evaluation/ablation_timed_executor.rb +22 -4
- data/lib/woods/export/typed_reader.rb +56 -0
- data/lib/woods/extractor.rb +232 -137
- data/lib/woods/extractors/action_cable_extractor.rb +3 -1
- data/lib/woods/extractors/behavioral_profile.rb +9 -7
- data/lib/woods/extractors/caching_extractor.rb +3 -1
- data/lib/woods/extractors/concern_extractor.rb +64 -6
- data/lib/woods/extractors/configuration_extractor.rb +7 -3
- data/lib/woods/extractors/controller_extractor.rb +13 -4
- data/lib/woods/extractors/database_view_extractor.rb +3 -1
- data/lib/woods/extractors/decorator_extractor.rb +3 -1
- data/lib/woods/extractors/engine_extractor.rb +3 -1
- data/lib/woods/extractors/event_extractor.rb +4 -2
- data/lib/woods/extractors/factory_extractor.rb +3 -1
- data/lib/woods/extractors/graphql_extractor.rb +8 -2
- data/lib/woods/extractors/i18n_extractor.rb +3 -1
- data/lib/woods/extractors/job_extractor.rb +6 -19
- data/lib/woods/extractors/lib_extractor.rb +3 -1
- data/lib/woods/extractors/mailer_extractor.rb +20 -5
- data/lib/woods/extractors/manager_extractor.rb +3 -1
- data/lib/woods/extractors/method_parameters.rb +53 -0
- data/lib/woods/extractors/middleware_argument.rb +65 -0
- data/lib/woods/extractors/middleware_extractor.rb +9 -3
- data/lib/woods/extractors/migration_extractor.rb +3 -1
- data/lib/woods/extractors/model_extractor.rb +39 -33
- data/lib/woods/extractors/package_extractor.rb +24 -4
- data/lib/woods/extractors/phlex_extractor.rb +3 -1
- data/lib/woods/extractors/policy_extractor.rb +3 -1
- data/lib/woods/extractors/poro_extractor.rb +3 -1
- data/lib/woods/extractors/pundit_extractor.rb +3 -1
- data/lib/woods/extractors/rails_source_extractor.rb +4 -2
- data/lib/woods/extractors/rake_task_extractor.rb +4 -2
- data/lib/woods/extractors/route_extractor.rb +3 -1
- data/lib/woods/extractors/route_helper_resolver.rb +10 -33
- data/lib/woods/extractors/scheduled_job_extractor.rb +41 -15
- data/lib/woods/extractors/serializer_extractor.rb +4 -2
- data/lib/woods/extractors/service_extractor.rb +3 -1
- data/lib/woods/extractors/shared_dependency_scanner.rb +2 -2
- data/lib/woods/extractors/shared_utility_methods.rb +27 -15
- data/lib/woods/extractors/source_nesting.rb +1 -1
- data/lib/woods/extractors/state_machine_extractor.rb +3 -1
- data/lib/woods/extractors/test_mapping_extractor.rb +3 -1
- data/lib/woods/extractors/validator_extractor.rb +3 -1
- data/lib/woods/extractors/view_component_extractor.rb +3 -1
- data/lib/woods/extractors/view_template_extractor.rb +3 -1
- data/lib/woods/gem_mapper.rb +2 -0
- data/lib/woods/git_history.rb +116 -0
- data/lib/woods/graph_analyzer.rb +35 -6
- data/lib/woods/hooks/context_cli.rb +54 -0
- data/lib/woods/hooks/context_event.rb +88 -0
- data/lib/woods/hooks/context_hint.rb +73 -0
- data/lib/woods/hooks/context_impact.rb +77 -0
- data/lib/woods/hooks/context_output.rb +47 -0
- data/lib/woods/hooks/context_state.rb +102 -0
- data/lib/woods/hooks/refresh.rb +79 -0
- data/lib/woods/hooks/rule_projection.rb +78 -0
- data/lib/woods/input_rules.rb +19 -0
- data/lib/woods/mcp/bearer_auth.rb +20 -12
- data/lib/woods/mcp/bootstrapper.rb +62 -0
- data/lib/woods/mcp/index_reader.rb +323 -160
- data/lib/woods/mcp/initialization_guidance.rb +27 -0
- data/lib/woods/mcp/origin_guard.rb +17 -9
- data/lib/woods/mcp/published_lexical_retriever.rb +115 -0
- data/lib/woods/mcp/renderers/markdown_renderer.rb +8 -1
- data/lib/woods/mcp/renderers/plain_renderer.rb +7 -1
- data/lib/woods/mcp/search_results.rb +74 -0
- data/lib/woods/mcp/server.rb +158 -37
- data/lib/woods/mcp/tool_contract.rb +2 -0
- data/lib/woods/mcp/tool_response_renderer.rb +25 -0
- data/lib/woods/mcp/traversal_evidence.rb +113 -0
- data/lib/woods/mcp/traversal_evidence_index.rb +100 -0
- data/lib/woods/mcp/traversal_evidence_page.rb +41 -0
- data/lib/woods/mcp/traversal_evidence_text.rb +52 -0
- data/lib/woods/notion/exporter.rb +56 -17
- data/lib/woods/obsidian/destination_plan.rb +98 -0
- data/lib/woods/obsidian/name_mapper.rb +19 -3
- data/lib/woods/obsidian/note_builder.rb +19 -10
- data/lib/woods/obsidian/vault_exporter.rb +88 -32
- data/lib/woods/operator/pipeline_guard.rb +18 -13
- data/lib/woods/path_dispatcher.rb +7 -1
- data/lib/woods/payload_store.rb +27 -26
- data/lib/woods/railtie.rb +3 -3
- data/lib/woods/railtie_support.rb +12 -12
- data/lib/woods/rake_helpers.rb +392 -0
- data/lib/woods/resilience/graph_invariant_validator/membership_checks.rb +71 -0
- data/lib/woods/resilience/graph_invariant_validator/node_checks.rb +61 -0
- data/lib/woods/resilience/graph_invariant_validator/reverse_relationship_checks.rb +46 -0
- data/lib/woods/resilience/graph_invariant_validator.rb +119 -0
- data/lib/woods/resilience/index_validator/graph_checks.rb +80 -0
- data/lib/woods/resilience/index_validator.rb +112 -23
- data/lib/woods/retrieval/context_assembler.rb +50 -15
- data/lib/woods/retrieval/lexical_assembler.rb +73 -0
- data/lib/woods/retrieval/lexical_index.rb +119 -0
- data/lib/woods/retrieval/ranker.rb +4 -2
- data/lib/woods/retrieval/scope.rb +108 -0
- data/lib/woods/retrieval/scoped_graph_store.rb +32 -0
- data/lib/woods/retrieval/scoped_vector_store.rb +55 -0
- data/lib/woods/retrieval/search_executor.rb +86 -27
- data/lib/woods/retrieval/source_evidence.rb +200 -0
- data/lib/woods/retriever.rb +98 -22
- data/lib/woods/ruby_analyzer/trace_enricher.rb +77 -38
- data/lib/woods/session_tracer/middleware.rb +10 -12
- data/lib/woods/session_tracer/redis_store.rb +22 -6
- data/lib/woods/session_tracer/session_flow_assembler.rb +23 -17
- data/lib/woods/session_tracer/solid_cache_coordination.rb +6 -4
- data/lib/woods/session_tracer/unit_resolver.rb +63 -0
- data/lib/woods/source_inputs/consumer_errors.rb +27 -0
- data/lib/woods/source_inputs/handoff.rb +102 -0
- data/lib/woods/source_inputs/launcher.rb +157 -0
- data/lib/woods/source_inputs/manifest.rb +124 -0
- data/lib/woods/source_inputs/private_key.rb +55 -0
- data/lib/woods/source_inputs/scanner.rb +171 -0
- data/lib/woods/source_inputs/scopes.rb +71 -0
- data/lib/woods/source_inputs/session.rb +214 -0
- data/lib/woods/source_inputs/status.rb +84 -0
- data/lib/woods/source_inputs/verifier.rb +107 -0
- data/lib/woods/storage/metadata_store.rb +25 -25
- data/lib/woods/storage/pgvector.rb +29 -8
- data/lib/woods/storage/qdrant.rb +17 -7
- data/lib/woods/storage/vector_store.rb +18 -6
- data/lib/woods/tasks.rb +3 -2
- data/lib/woods/temporal/json_snapshot_store.rb +29 -8
- data/lib/woods/unblocked/exporter.rb +59 -70
- data/lib/woods/version.rb +1 -1
- data/lib/woods/watch/boot_snapshot.rb +52 -0
- data/lib/woods/watch/daemon.rb +136 -28
- data/lib/woods/watch/listen_watcher.rb +4 -0
- data/lib/woods/watch/polling_watcher.rb +5 -1
- data/lib/woods/watch/status.rb +20 -15
- data/lib/woods/watch/tree_scan.rb +21 -13
- data/lib/woods/watch/watcher.rb +4 -1
- data/lib/woods.rb +50 -11
- data/plugin/.claude-plugin/plugin.json +1 -1
- data/plugin/hooks/adapters/normalize.jq +15 -0
- data/plugin/hooks/adapters/normalize.rb +63 -0
- data/plugin/hooks/hooks.json +20 -0
- data/plugin/hooks/woods-context.sh +50 -0
- data/plugin/hooks/woods-input-rules.sh +159 -0
- data/plugin/hooks/woods-opencode.mjs +65 -0
- data/plugin/hooks/woods-post-edit.sh +2 -225
- data/plugin/hooks/woods-refresh.sh +260 -0
- data/plugin/hooks/woods-session-start.sh +47 -55
- data/plugin/skills/woods-agent-enable/SKILL.md +13 -0
- data/plugin/skills/woods-diagnose/SKILL.md +288 -1
- data/plugin/skills/woods-investigate/SKILL.md +106 -0
- data/plugin/skills/woods-mcp-config/SKILL.md +89 -1
- data/plugin/skills/woods-setup/SKILL.md +107 -6
- metadata +84 -5
data/lib/woods/extractor.rb
CHANGED
|
@@ -8,12 +8,14 @@ require 'pathname'
|
|
|
8
8
|
require 'set'
|
|
9
9
|
|
|
10
10
|
require_relative 'atomic_file'
|
|
11
|
+
require_relative 'version'
|
|
11
12
|
require_relative 'filename_utils'
|
|
12
13
|
require_relative 'token_utils'
|
|
13
14
|
require_relative 'extracted_unit'
|
|
14
15
|
require_relative 'dependency_graph'
|
|
15
16
|
require_relative 'payload_store'
|
|
16
17
|
require_relative 'git_provenance'
|
|
18
|
+
require_relative 'git_history'
|
|
17
19
|
require_relative 'extractors/model_extractor'
|
|
18
20
|
require_relative 'extractors/controller_extractor'
|
|
19
21
|
require_relative 'extractors/phlex_extractor'
|
|
@@ -55,6 +57,7 @@ require_relative 'flow_precomputer'
|
|
|
55
57
|
require_relative 'change_set'
|
|
56
58
|
require_relative 'generation'
|
|
57
59
|
require_relative 'path_dispatcher'
|
|
60
|
+
require_relative 'source_inputs/session'
|
|
58
61
|
|
|
59
62
|
module Woods
|
|
60
63
|
# Extractor is the main orchestrator for codebase extraction.
|
|
@@ -355,7 +358,7 @@ module Woods
|
|
|
355
358
|
# flat index — the output root also holds `generation.json`, `dumps/`,
|
|
356
359
|
# `tasks/`, `woods.sqlite3` and `payloads/` itself, none of which belong
|
|
357
360
|
# to a generation's payload.
|
|
358
|
-
PAYLOAD_FILES = %w[manifest.json dependency_graph.json graph_analysis.json SUMMARY.md].freeze
|
|
361
|
+
PAYLOAD_FILES = %w[manifest.json source_inputs.json dependency_graph.json graph_analysis.json SUMMARY.md].freeze
|
|
359
362
|
|
|
360
363
|
# Payload directories that are not per-type unit directories.
|
|
361
364
|
PAYLOAD_DIRS = %w[flows].freeze
|
|
@@ -394,7 +397,9 @@ module Woods
|
|
|
394
397
|
#
|
|
395
398
|
# @return [Hash] Results keyed by extractor type
|
|
396
399
|
def extract_all
|
|
400
|
+
profile_started = Process.clock_gettime(Process::CLOCK_MONOTONIC) if profiling?
|
|
397
401
|
setup_output_directory
|
|
402
|
+
profile_phase('source capture') { begin_source_inputs('full') }
|
|
398
403
|
ModelNameCache.reset!
|
|
399
404
|
# @package_resolver alone is not enough: #package_resolver builds
|
|
400
405
|
# through #extractor_for, which memoizes into @incremental_extractors.
|
|
@@ -422,29 +427,33 @@ module Woods
|
|
|
422
427
|
|
|
423
428
|
# Phase 1.5: Deduplicate results
|
|
424
429
|
Rails.logger.info '[Woods] Deduplicating results...'
|
|
425
|
-
deduplicate_results
|
|
430
|
+
profile_phase('deduplication') { deduplicate_results }
|
|
426
431
|
|
|
427
432
|
# Phase 1.6: Package membership. Runs before the graph is rebuilt so
|
|
428
433
|
# registration copies metadata[:package] onto the node (#280).
|
|
429
|
-
annotate_packages
|
|
434
|
+
profile_phase('package annotation') { annotate_packages }
|
|
430
435
|
|
|
431
436
|
# Rebuild the graph from deduped results. #164 gave DependencyGraph
|
|
432
437
|
# `#remove`/`#unregister`, so surgical removal is now possible — but a
|
|
433
438
|
# full extraction has just registered every unit including duplicates,
|
|
434
439
|
# and rebuilding from the deduped set is both cheaper and less
|
|
435
440
|
# error-prone than unwinding registrations one at a time.
|
|
436
|
-
|
|
437
|
-
|
|
441
|
+
profile_phase('graph rebuild') do
|
|
442
|
+
@dependency_graph = DependencyGraph.new
|
|
443
|
+
@results.each_value { |units| units.each { |u| @dependency_graph.register(u) } }
|
|
444
|
+
end
|
|
438
445
|
|
|
439
446
|
# Phase 2: Resolve dependents (reverse dependencies)
|
|
440
447
|
Rails.logger.info '[Woods] Resolving dependents...'
|
|
441
|
-
resolve_dependents
|
|
448
|
+
profile_phase('dependents') { resolve_dependents }
|
|
442
449
|
|
|
443
450
|
# Phase 3: Enrich with git data. Runs BEFORE analysis now: the
|
|
444
451
|
# volatile_dependencies report reads commit counts off graph nodes.
|
|
445
452
|
Rails.logger.info '[Woods] Enriching with git data...'
|
|
446
|
-
|
|
447
|
-
|
|
453
|
+
profile_phase('git enrichment') do
|
|
454
|
+
enrich_with_git_data
|
|
455
|
+
annotate_graph_with_git_data
|
|
456
|
+
end
|
|
448
457
|
|
|
449
458
|
# Phase 4: Graph analysis (PageRank, structural metrics)
|
|
450
459
|
Rails.logger.info '[Woods] Analyzing dependency graph...'
|
|
@@ -452,7 +461,7 @@ module Woods
|
|
|
452
461
|
|
|
453
462
|
# Phase 4.5: Normalize file_path to relative paths
|
|
454
463
|
Rails.logger.info '[Woods] Normalizing file paths...'
|
|
455
|
-
normalize_file_paths
|
|
464
|
+
profile_phase('path normalization') { normalize_file_paths }
|
|
456
465
|
|
|
457
466
|
# Phase 5: Write output
|
|
458
467
|
Rails.logger.info '[Woods] Writing output...'
|
|
@@ -462,7 +471,7 @@ module Woods
|
|
|
462
471
|
# run after write_results — the just-written set is what defines
|
|
463
472
|
# "legitimate" — and belongs to the full path only; the incremental path
|
|
464
473
|
# deletes through the graph instead. See {#sweep_orphaned_unit_files}.
|
|
465
|
-
sweep_orphaned_unit_files
|
|
474
|
+
profile_phase('orphan sweep') { sweep_orphaned_unit_files }
|
|
466
475
|
|
|
467
476
|
# Phase 5.5: Precompute request flows (opt-in). Must run AFTER
|
|
468
477
|
# write_results — FlowAssembler loads unit JSON from disk, so running
|
|
@@ -485,18 +494,23 @@ module Woods
|
|
|
485
494
|
profile_phase('flows') { precompute_flows }
|
|
486
495
|
end
|
|
487
496
|
|
|
488
|
-
|
|
489
|
-
|
|
497
|
+
profile_phase('graph write') do
|
|
498
|
+
write_dependency_graph
|
|
499
|
+
write_graph_analysis
|
|
500
|
+
end
|
|
490
501
|
profile_phase('manifest and summary') do
|
|
491
502
|
write_manifest
|
|
492
503
|
write_structural_summary
|
|
493
504
|
end
|
|
494
|
-
capture_snapshot
|
|
495
|
-
|
|
505
|
+
profile_phase('snapshot') { capture_snapshot }
|
|
506
|
+
@source_inputs.full_units(@results, consumers: @extractors)
|
|
507
|
+
publish_generation('full')
|
|
496
508
|
|
|
497
509
|
log_summary
|
|
498
510
|
|
|
499
511
|
@results
|
|
512
|
+
ensure
|
|
513
|
+
log_profile_total('full', profile_started)
|
|
500
514
|
end
|
|
501
515
|
|
|
502
516
|
# ══════════════════════════════════════════════════════════════════════
|
|
@@ -528,7 +542,8 @@ module Woods
|
|
|
528
542
|
# @param changed_files [Array<String>] List of changed file paths
|
|
529
543
|
# @return [Array<String>] Identifiers of units re-extracted, added, or removed
|
|
530
544
|
def extract_changed(changed_files)
|
|
531
|
-
|
|
545
|
+
profile_started = Process.clock_gettime(Process::CLOCK_MONOTONIC) if profiling?
|
|
546
|
+
prepare_incremental_run(operation: 'incremental')
|
|
532
547
|
|
|
533
548
|
change_set = ChangeSet.new(paths: changed_files, root: Rails.root)
|
|
534
549
|
affected_types = Set.new
|
|
@@ -551,36 +566,39 @@ module Woods
|
|
|
551
566
|
acc
|
|
552
567
|
end
|
|
553
568
|
|
|
554
|
-
|
|
555
|
-
|
|
556
|
-
|
|
557
|
-
|
|
558
|
-
|
|
559
|
-
|
|
560
|
-
|
|
561
|
-
|
|
562
|
-
|
|
563
|
-
|
|
564
|
-
|
|
565
|
-
|
|
566
|
-
|
|
567
|
-
|
|
568
|
-
|
|
569
|
-
|
|
570
|
-
|
|
571
|
-
|
|
572
|
-
|
|
573
|
-
|
|
574
|
-
|
|
575
|
-
|
|
576
|
-
|
|
577
|
-
|
|
578
|
-
|
|
579
|
-
|
|
580
|
-
|
|
581
|
-
|
|
582
|
-
|
|
583
|
-
|
|
569
|
+
profile_phase('reconciliation') do
|
|
570
|
+
touched.merge(reconcile_class_based_types(affected_types))
|
|
571
|
+
touched.merge(reconcile_model_mixins(affected_types))
|
|
572
|
+
touched.merge(rerun_whole_app_extractors(change_set, affected_types))
|
|
573
|
+
touched.merge(reannotate_packages(change_set, affected_types))
|
|
574
|
+
pruned = prune_vanished_units(change_set, affected_types)
|
|
575
|
+
touched.merge(pruned)
|
|
576
|
+
|
|
577
|
+
# Reconcile once more, because pruning can un-know a class the first pass
|
|
578
|
+
# skipped. A class-based file moved between autoload directories with its
|
|
579
|
+
# constant unchanged is still registered under the old path when
|
|
580
|
+
# reconciliation runs, so it looks known and is not re-extracted; the
|
|
581
|
+
# prune that follows then removes it for its vanished path. This pass
|
|
582
|
+
# re-adds it in the same run (M1) instead of leaving the unit missing
|
|
583
|
+
# until some later run happens to notice. Idempotent when nothing was
|
|
584
|
+
# pruned: the discovery set is compared against the graph, so an
|
|
585
|
+
# already-registered class is skipped.
|
|
586
|
+
#
|
|
587
|
+
# But not everything pruning removed may come back. `except:` keeps the
|
|
588
|
+
# *deletion* shape pruned: without a reload, a constant outlives the file
|
|
589
|
+
# that defined it — so deleting `app/models/user.rb` prunes `User`, and
|
|
590
|
+
# this pass finds `User` still in `ActiveRecord::Base.descendants`.
|
|
591
|
+
# Re-registering it would pin the unit to a path that no longer exists,
|
|
592
|
+
# and nothing could ever remove it: the sweep excludes class-based units
|
|
593
|
+
# and no future change set names that path again. A resident daemon
|
|
594
|
+
# processing a batch before its reload hits this every time. What
|
|
595
|
+
# separates the two shapes is the filesystem — only pruned identifiers
|
|
596
|
+
# that a still-existing file in the change set actually declares are
|
|
597
|
+
# re-addable. See {#readdable_pruned_classes}.
|
|
598
|
+
touched.merge(reconcile_class_based_types(
|
|
599
|
+
affected_types, except: pruned - readdable_pruned_classes(pruned, change_set)
|
|
600
|
+
))
|
|
601
|
+
end
|
|
584
602
|
|
|
585
603
|
finalize_incremental_unit_json(affected_types)
|
|
586
604
|
|
|
@@ -594,6 +612,8 @@ module Woods
|
|
|
594
612
|
finalize_incremental_run(touched)
|
|
595
613
|
|
|
596
614
|
touched.to_a
|
|
615
|
+
ensure
|
|
616
|
+
log_profile_total('incremental', profile_started)
|
|
597
617
|
end
|
|
598
618
|
|
|
599
619
|
# ══════════════════════════════════════════════════════════════════════
|
|
@@ -628,6 +648,7 @@ module Woods
|
|
|
628
648
|
# extractor
|
|
629
649
|
# @raise [ArgumentError] when no recognized key is given
|
|
630
650
|
def refresh(*keys)
|
|
651
|
+
profile_started = Process.clock_gettime(Process::CLOCK_MONOTONIC) if profiling?
|
|
631
652
|
keys = Array(keys).flatten.map(&:to_sym).uniq
|
|
632
653
|
known, unknown = keys.partition { |key| EXTRACTORS.key?(key) }
|
|
633
654
|
raise ArgumentError, "No known extractor in #{keys.inspect}" if known.empty?
|
|
@@ -635,17 +656,19 @@ module Woods
|
|
|
635
656
|
known += ROUTE_CONSUMER_EXTRACTORS if known.include?(:routes)
|
|
636
657
|
known.uniq!
|
|
637
658
|
|
|
638
|
-
prepare_incremental_run
|
|
659
|
+
prepare_incremental_run(operation: 'refresh')
|
|
639
660
|
affected_types = Set.new
|
|
640
661
|
touched = known.each_with_object(Set.new) do |key, acc|
|
|
641
662
|
acc.merge(replace_type_wholesale(key, affected_types))
|
|
642
663
|
end
|
|
643
664
|
|
|
644
665
|
finalize_incremental_unit_json(affected_types)
|
|
645
|
-
affected_types.each { |type_key| regenerate_type_index(type_key) }
|
|
666
|
+
profile_phase('type index') { affected_types.each { |type_key| regenerate_type_index(type_key) } }
|
|
646
667
|
finalize_incremental_run(touched, reason: "refresh:#{known.sort.join(',')}")
|
|
647
668
|
|
|
648
669
|
{ types: known, touched: touched.to_a, unknown: unknown }
|
|
670
|
+
ensure
|
|
671
|
+
log_profile_total('refresh', profile_started)
|
|
649
672
|
end
|
|
650
673
|
|
|
651
674
|
# Raise when the most recent extraction run wrote a payload but could not
|
|
@@ -667,6 +690,15 @@ module Woods
|
|
|
667
690
|
|
|
668
691
|
private
|
|
669
692
|
|
|
693
|
+
# Whole-run wall time, including unprofiled setup and failed runs. This
|
|
694
|
+
# separate log family must never be added to the individual phase times.
|
|
695
|
+
def log_profile_total(name, started)
|
|
696
|
+
return unless started
|
|
697
|
+
|
|
698
|
+
elapsed = Process.clock_gettime(Process::CLOCK_MONOTONIC) - started
|
|
699
|
+
Rails.logger.info "[Woods] [profile total] #{name} in #{elapsed.round(2)}s"
|
|
700
|
+
end
|
|
701
|
+
|
|
670
702
|
# Time one phase of a run and log how long it took, when WOODS_PROFILE=1.
|
|
671
703
|
#
|
|
672
704
|
# The per-extractor lines (see {#extract_all_sequential}) already report
|
|
@@ -745,7 +777,8 @@ module Woods
|
|
|
745
777
|
#
|
|
746
778
|
# @return [void]
|
|
747
779
|
# @raise [Woods::ExtractionError] see {#begin_payload!}
|
|
748
|
-
def prepare_incremental_run
|
|
780
|
+
def prepare_incremental_run(operation: 'incremental')
|
|
781
|
+
profile_phase('source capture') { begin_source_inputs(operation) }
|
|
749
782
|
profile_phase('payload seed') { begin_payload!(strict: true) }
|
|
750
783
|
graph_path = payload_dir.join('dependency_graph.json')
|
|
751
784
|
ensure_incremental_baseline!(graph_path)
|
|
@@ -795,7 +828,7 @@ module Woods
|
|
|
795
828
|
# reading "incremental" after a `woods:refresh[routes]` is being misled
|
|
796
829
|
# @return [void]
|
|
797
830
|
def finalize_incremental_run(touched, reason: 'incremental')
|
|
798
|
-
write_dependency_graph
|
|
831
|
+
profile_phase('graph write') { write_dependency_graph }
|
|
799
832
|
|
|
800
833
|
if touched.empty?
|
|
801
834
|
Rails.logger.info '[Woods] Incremental run changed nothing — leaving manifest timestamp untouched'
|
|
@@ -808,7 +841,7 @@ module Woods
|
|
|
808
841
|
write_manifest(incremental: true)
|
|
809
842
|
write_structural_summary
|
|
810
843
|
end
|
|
811
|
-
|
|
844
|
+
publish_generation(reason)
|
|
812
845
|
|
|
813
846
|
return unless Woods.configuration.enable_snapshots
|
|
814
847
|
|
|
@@ -831,9 +864,10 @@ module Woods
|
|
|
831
864
|
# Resolve (and if necessary rename) the payload first, so the flush
|
|
832
865
|
# below covers the directory under the name the pointer will carry.
|
|
833
866
|
payload = publishable_payload_name(generation)
|
|
867
|
+
profile_phase('source verification') { write_source_inputs } if payload
|
|
834
868
|
profile_phase('payload sync') { sync_payload }
|
|
835
|
-
marker = generation.bump!(reason: reason, payload: payload)
|
|
836
|
-
prune_payloads(marker.number)
|
|
869
|
+
marker = profile_phase('publish') { generation.bump!(reason: reason, payload: payload) }
|
|
870
|
+
profile_phase('payload prune') { prune_payloads(marker.number) }
|
|
837
871
|
marker
|
|
838
872
|
rescue StandardError => e
|
|
839
873
|
# A failed bump must not fail the extraction that produced a perfectly
|
|
@@ -852,6 +886,38 @@ module Woods
|
|
|
852
886
|
nil
|
|
853
887
|
end
|
|
854
888
|
|
|
889
|
+
# Capture before eager loading or extraction; only an explicit fresh-launch
|
|
890
|
+
# handoff can additionally establish the pre-Bundler/Rails boot boundary.
|
|
891
|
+
def begin_source_inputs(operation)
|
|
892
|
+
@source_inputs = SourceInputs::Session.new(root: Rails.root, output_dir: @output_dir,
|
|
893
|
+
baseline_path: source_input_baseline_path,
|
|
894
|
+
operation: operation)
|
|
895
|
+
end
|
|
896
|
+
|
|
897
|
+
def source_input_baseline_path
|
|
898
|
+
generation = Generation.new(output_dir: @output_dir)
|
|
899
|
+
marker = generation.current
|
|
900
|
+
directory = generation.payload_dir(marker)
|
|
901
|
+
return nil unless marker.payload && directory != generation.root
|
|
902
|
+
|
|
903
|
+
directory.join(SourceInputs::Manifest::FILE_NAME)
|
|
904
|
+
rescue TypeError, NoMethodError
|
|
905
|
+
nil
|
|
906
|
+
end
|
|
907
|
+
|
|
908
|
+
def source_consumer_failed?(key, consumer = extractor_for(key))
|
|
909
|
+
failed = consumer.nil? || SourceInputs::ConsumerErrors.failed?(consumer)
|
|
910
|
+
@source_inputs&.unverified("extractor:#{key}") if failed
|
|
911
|
+
failed
|
|
912
|
+
end
|
|
913
|
+
|
|
914
|
+
def write_source_inputs
|
|
915
|
+
return unless @source_inputs
|
|
916
|
+
|
|
917
|
+
manifest = @source_inputs.finish(generation: @payload_generation, eager_load_complete: @eager_load_complete)
|
|
918
|
+
AtomicFile.write(payload_dir.join(SourceInputs::Manifest::FILE_NAME), JSON.pretty_generate(manifest.data))
|
|
919
|
+
end
|
|
920
|
+
|
|
855
921
|
# Open the payload directory this run publishes into, seeded from the
|
|
856
922
|
# generation currently on disk.
|
|
857
923
|
#
|
|
@@ -1268,9 +1334,6 @@ module Woods
|
|
|
1268
1334
|
|
|
1269
1335
|
def setup_output_directory
|
|
1270
1336
|
FileUtils.mkdir_p(@output_dir)
|
|
1271
|
-
EXTRACTORS.each_key do |type|
|
|
1272
|
-
FileUtils.mkdir_p(payload_dir.join(type.to_s))
|
|
1273
|
-
end
|
|
1274
1337
|
end
|
|
1275
1338
|
|
|
1276
1339
|
# ──────────────────────────────────────────────────────────────────────
|
|
@@ -1370,8 +1433,10 @@ module Woods
|
|
|
1370
1433
|
def same_type_collision_message(type, unit, prior_path)
|
|
1371
1434
|
"same-type identifier collision: #{type.to_s.singularize} '#{unit.identifier}' derived from " \
|
|
1372
1435
|
"two different sources ('#{prior_path || 'no file'}' and '#{unit.file_path || 'no file'}'); " \
|
|
1373
|
-
'only one unit could ever be indexed, so extraction aborted
|
|
1374
|
-
'
|
|
1436
|
+
'only one unit could ever be indexed, so extraction aborted. ' \
|
|
1437
|
+
'Wrapper-nested class naming requires Zeitwerk mode with Zeitwerk >= 2.6.9; on older loaders or ' \
|
|
1438
|
+
'classic-mode hosts, check that support before changing valid namespace wrappers. ' \
|
|
1439
|
+
'For a genuine duplicate, merge the declarations into one file or split them into distinct constants'
|
|
1375
1440
|
end
|
|
1376
1441
|
|
|
1377
1442
|
# ──────────────────────────────────────────────────────────────────────
|
|
@@ -1753,6 +1818,7 @@ module Woods
|
|
|
1753
1818
|
GraphAnalyzer.new(
|
|
1754
1819
|
@dependency_graph,
|
|
1755
1820
|
volatile_ratio: ratio,
|
|
1821
|
+
volatile_limit_per_target: config&.volatile_dependency_limit_per_target,
|
|
1756
1822
|
cycle_limit: config ? config.graph_cycle_limit : GraphAnalyzer::DEFAULT_CYCLE_LIMIT,
|
|
1757
1823
|
cycle_max_length: config ? config.graph_cycle_max_length : GraphAnalyzer::DEFAULT_CYCLE_MAX_LENGTH
|
|
1758
1824
|
)
|
|
@@ -1869,11 +1935,9 @@ module Woods
|
|
|
1869
1935
|
# Is this a path worth asking git about?
|
|
1870
1936
|
#
|
|
1871
1937
|
# A gem-owned unit (an engine model) carries its real path. Outside
|
|
1872
|
-
# Rails.root,
|
|
1873
|
-
# outside the repository — one gem path would erase the git metadata of
|
|
1874
|
-
# the other 499 units in its 500-path batch. Inside Rails.root, a bundle
|
|
1938
|
+
# Rails.root, it has no app repository history. Inside Rails.root, a bundle
|
|
1875
1939
|
# vendored at `vendor/bundle` puts the same gem files under the root
|
|
1876
|
-
# prefix, gitignored, so
|
|
1940
|
+
# prefix, gitignored, so requesting their history serves no app-owned unit.
|
|
1877
1941
|
# Same exclusions as {Extractors::SharedUtilityMethods#app_source?}.
|
|
1878
1942
|
#
|
|
1879
1943
|
# @param path [String, nil] absolute file path
|
|
@@ -1927,13 +1991,13 @@ module Woods
|
|
|
1927
1991
|
# exactly this value shape — no fractional seconds, `Z` or a `±hh:mm`
|
|
1928
1992
|
# offset — and `spec/extracted_unit_spec.rb` pins that, so a change to the
|
|
1929
1993
|
# stamp's shape fails a spec instead of quietly un-matching this mask.
|
|
1930
|
-
#
|
|
1931
|
-
#
|
|
1932
|
-
#
|
|
1933
|
-
#
|
|
1934
|
-
# top level.
|
|
1994
|
+
# Match only the final top-level stamp, followed by the source_hash field
|
|
1995
|
+
# and the document's closing brace. Nested metadata may use the same key
|
|
1996
|
+
# and timestamp shape; changing it must still rewrite the unit. Escaped
|
|
1997
|
+
# quotes inside string values cannot match these JSON field boundaries.
|
|
1935
1998
|
EXTRACTED_AT_SCALAR =
|
|
1936
|
-
/("extracted_at":\s*")\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}(?:Z|[+-]\d{2}:\d{2})
|
|
1999
|
+
/("extracted_at":\s*")\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}(?:Z|[+-]\d{2}:\d{2})
|
|
2000
|
+
(?=",\s*"source_hash":\s*"[0-9a-f]{64}"\s*}\s*\z)/x
|
|
1937
2001
|
# An implementation detail of the byte comparison, not part of the
|
|
1938
2002
|
# extractor's surface (`private` does not scope constants).
|
|
1939
2003
|
private_constant :EXTRACTED_AT_SCALAR
|
|
@@ -1967,7 +2031,7 @@ module Woods
|
|
|
1967
2031
|
# original encoding
|
|
1968
2032
|
# @return [String] the bytes with the stamp's value removed
|
|
1969
2033
|
def mask_extracted_at(bytes)
|
|
1970
|
-
bytes.
|
|
2034
|
+
bytes.sub(EXTRACTED_AT_SCALAR, '\1')
|
|
1971
2035
|
end
|
|
1972
2036
|
|
|
1973
2037
|
def normalize_file_paths
|
|
@@ -1999,7 +2063,7 @@ module Woods
|
|
|
1999
2063
|
# to say. Enrichment then wrote `commit_count: 0` and
|
|
2000
2064
|
# `change_frequency: new` onto every unit, which reads exactly like a file
|
|
2001
2065
|
# that was never committed, where an absent git directory correctly omits
|
|
2002
|
-
# the keys (B-186). HEAD has to resolve.
|
|
2066
|
+
# the keys (B-186). HEAD has to resolve, with complete ancestry (B-189).
|
|
2003
2067
|
#
|
|
2004
2068
|
# Memoized, so the warning below is emitted at most once per run.
|
|
2005
2069
|
#
|
|
@@ -2008,13 +2072,31 @@ module Woods
|
|
|
2008
2072
|
return @git_available if defined?(@git_available)
|
|
2009
2073
|
|
|
2010
2074
|
_output, error, status = Open3.capture3(*git_argv('rev-parse', 'HEAD'))
|
|
2011
|
-
|
|
2012
|
-
|
|
2013
|
-
|
|
2075
|
+
unless status.success?
|
|
2076
|
+
warn_unresolvable_git(error)
|
|
2077
|
+
return @git_available = false
|
|
2078
|
+
end
|
|
2079
|
+
|
|
2080
|
+
@git_available = complete_git_history?
|
|
2014
2081
|
rescue StandardError
|
|
2015
2082
|
@git_available = false
|
|
2016
2083
|
end
|
|
2017
2084
|
|
|
2085
|
+
# A shallow HEAD resolves but represents an incomplete ancestry. Do not
|
|
2086
|
+
# turn that boundary into apparent one-commit/new-file churn facts.
|
|
2087
|
+
def complete_git_history?
|
|
2088
|
+
output, _error, status = Open3.capture3(*git_argv('rev-parse', '--is-shallow-repository'))
|
|
2089
|
+
return true if status.success? && output.strip == 'false'
|
|
2090
|
+
|
|
2091
|
+
shallow = status.success? && output.strip == 'true'
|
|
2092
|
+
reason = shallow ? 'shallow repository' : 'repository depth could not be verified'
|
|
2093
|
+
Rails.logger.warn(
|
|
2094
|
+
"[Woods] Git enrichment omitted: #{reason}. Fetch full history with git fetch --unshallow " \
|
|
2095
|
+
'(or actions/checkout fetch-depth: 0), then run full extraction to refresh git metadata.'
|
|
2096
|
+
)
|
|
2097
|
+
false
|
|
2098
|
+
end
|
|
2099
|
+
|
|
2018
2100
|
# Say once why no unit will carry git metadata, but only when there is a
|
|
2019
2101
|
# working tree to explain. No `.git` at the root is the ordinary source
|
|
2020
2102
|
# tarball or `COPY`-without-`.git` case, and it is not a fault.
|
|
@@ -2058,63 +2140,19 @@ module Woods
|
|
|
2058
2140
|
''
|
|
2059
2141
|
end
|
|
2060
2142
|
|
|
2061
|
-
#
|
|
2062
|
-
#
|
|
2063
|
-
#
|
|
2064
|
-
#
|
|
2065
|
-
# batch also sent. The result is keyed by relative path, so the output
|
|
2066
|
-
# is identical.
|
|
2067
|
-
#
|
|
2068
|
-
# @param file_paths [Array<String>] Absolute file paths
|
|
2069
|
-
# @return [Hash{String => Hash}] Keyed by relative path
|
|
2143
|
+
# One HEAD history walk, independent of requested-path grouping. See
|
|
2144
|
+
# GitHistory for explicit merge semantics and binary record framing.
|
|
2145
|
+
# @param file_paths [Array<String>] absolute file paths
|
|
2146
|
+
# @return [Hash{String => Hash}] keyed by Rails.root-relative path
|
|
2070
2147
|
def batch_git_data(file_paths)
|
|
2071
2148
|
return {} if file_paths.empty?
|
|
2072
2149
|
|
|
2073
|
-
|
|
2074
|
-
|
|
2075
|
-
|
|
2076
|
-
|
|
2077
|
-
|
|
2078
|
-
path_set = relative_paths.to_set
|
|
2079
|
-
relative_paths.each_slice(500) do |batch|
|
|
2080
|
-
log_output = run_git(
|
|
2081
|
-
'log', '--all', '--name-only',
|
|
2082
|
-
'--format=__COMMIT__%H|||%an|||%cI|||%s',
|
|
2083
|
-
'--since=365 days ago',
|
|
2084
|
-
'--', *batch
|
|
2085
|
-
)
|
|
2086
|
-
parse_git_log_output(log_output, path_set, result)
|
|
2087
|
-
end
|
|
2088
|
-
|
|
2089
|
-
ninety_days_ago = (Time.current - 90.days).iso8601
|
|
2090
|
-
result.each do |relative_path, data|
|
|
2091
|
-
result[relative_path] = build_file_metadata(data, ninety_days_ago)
|
|
2092
|
-
end
|
|
2150
|
+
relative_paths = file_paths.map { |path| normalize_file_path(path) }.uniq
|
|
2151
|
+
recent_after = Time.current - 90.days
|
|
2152
|
+
raw = GitHistory.new(root: Rails.root, logger: Rails.logger).read(relative_paths, recent_after: recent_after)
|
|
2153
|
+
return {} unless raw
|
|
2093
2154
|
|
|
2094
|
-
|
|
2095
|
-
end
|
|
2096
|
-
|
|
2097
|
-
# Parse git log output line-by-line, populating result with per-file commit data.
|
|
2098
|
-
def parse_git_log_output(log_output, path_set, result)
|
|
2099
|
-
current_commit = nil
|
|
2100
|
-
|
|
2101
|
-
log_output.each_line do |line|
|
|
2102
|
-
line = line.strip
|
|
2103
|
-
next if line.empty?
|
|
2104
|
-
|
|
2105
|
-
if line.start_with?('__COMMIT__')
|
|
2106
|
-
parts = line.sub('__COMMIT__', '').split('|||', 4)
|
|
2107
|
-
current_commit = { sha: parts[0], author: parts[1], date: parts[2], message: parts[3] }
|
|
2108
|
-
elsif current_commit && path_set.include?(line)
|
|
2109
|
-
entry = result[line] ||= {}
|
|
2110
|
-
unless entry[:last_modified]
|
|
2111
|
-
entry[:last_modified] = current_commit[:date]
|
|
2112
|
-
entry[:last_author] = current_commit[:author]
|
|
2113
|
-
end
|
|
2114
|
-
(entry[:commits] ||= []) << current_commit
|
|
2115
|
-
(entry[:contributors] ||= Hash.new(0))[current_commit[:author]] += 1
|
|
2116
|
-
end
|
|
2117
|
-
end
|
|
2155
|
+
raw.transform_values { |data| build_file_metadata(data, recent_after.iso8601) }
|
|
2118
2156
|
end
|
|
2119
2157
|
|
|
2120
2158
|
# Classify how frequently a file changes based on commit counts.
|
|
@@ -2136,12 +2174,13 @@ module Woods
|
|
|
2136
2174
|
def build_file_metadata(data, ninety_days_ago)
|
|
2137
2175
|
all_commits = data[:commits] || []
|
|
2138
2176
|
contributor_counts = data[:contributors] || {}
|
|
2139
|
-
recent_count = all_commits.count { |c| c[:date] && c[:date] > ninety_days_ago }
|
|
2177
|
+
recent_count = data.fetch(:recent_count) { all_commits.count { |c| c[:date] && c[:date] > ninety_days_ago } }
|
|
2178
|
+
total_count = data.fetch(:commit_count, all_commits.size)
|
|
2140
2179
|
|
|
2141
2180
|
{
|
|
2142
2181
|
last_modified: data[:last_modified],
|
|
2143
2182
|
last_author: data[:last_author],
|
|
2144
|
-
commit_count:
|
|
2183
|
+
commit_count: total_count,
|
|
2145
2184
|
contributors: contributor_counts
|
|
2146
2185
|
.sort_by { |_, count| -count }
|
|
2147
2186
|
.first(5)
|
|
@@ -2149,7 +2188,7 @@ module Woods
|
|
|
2149
2188
|
recent_commits: all_commits.first(5).map do |c|
|
|
2150
2189
|
{ sha: c[:sha]&.first(8), message: c[:message], date: c[:date], author: c[:author] }
|
|
2151
2190
|
end,
|
|
2152
|
-
change_frequency: classify_change_frequency(
|
|
2191
|
+
change_frequency: classify_change_frequency(total_count, recent_count)
|
|
2153
2192
|
}
|
|
2154
2193
|
end
|
|
2155
2194
|
|
|
@@ -2324,6 +2363,7 @@ module Woods
|
|
|
2324
2363
|
|
|
2325
2364
|
manifest = {
|
|
2326
2365
|
extracted_at: Time.current.iso8601,
|
|
2366
|
+
woods_version: Woods::VERSION,
|
|
2327
2367
|
rails_version: Rails.version,
|
|
2328
2368
|
ruby_version: RUBY_VERSION,
|
|
2329
2369
|
|
|
@@ -2697,6 +2737,7 @@ module Woods
|
|
|
2697
2737
|
|
|
2698
2738
|
@incremental_extractors[key] = EXTRACTORS[key]&.new
|
|
2699
2739
|
rescue StandardError => e
|
|
2740
|
+
@source_inputs&.unverified("extractor:#{key}")
|
|
2700
2741
|
Rails.logger.warn "[Woods] Could not build #{key} extractor: #{e.message}"
|
|
2701
2742
|
@incremental_extractors[key] = nil
|
|
2702
2743
|
end
|
|
@@ -2758,6 +2799,10 @@ module Woods
|
|
|
2758
2799
|
# one method over — CORE-1).
|
|
2759
2800
|
produced.merge(units.map { |unit| [unit.identifier, unit.type] })
|
|
2760
2801
|
touched.merge(register_and_write(rule.extractor_key, units, affected_types))
|
|
2802
|
+
unless source_consumer_failed?(rule.extractor_key)
|
|
2803
|
+
@source_inputs&.consume_file(rule.extractor_key,
|
|
2804
|
+
absolute_path)
|
|
2805
|
+
end
|
|
2761
2806
|
end
|
|
2762
2807
|
|
|
2763
2808
|
next if raised
|
|
@@ -2786,7 +2831,10 @@ module Woods
|
|
|
2786
2831
|
# on every changed path of that type, with the generation bumped over
|
|
2787
2832
|
# the loss. Construction failure tells us nothing about the path; only
|
|
2788
2833
|
# a genuinely constructed extractor that lacks the method earns the [].
|
|
2789
|
-
|
|
2834
|
+
if extractor.nil?
|
|
2835
|
+
source_consumer_failed?(rule.extractor_key, extractor)
|
|
2836
|
+
return nil
|
|
2837
|
+
end
|
|
2790
2838
|
return [] unless extractor.respond_to?(rule.method_name)
|
|
2791
2839
|
|
|
2792
2840
|
result =
|
|
@@ -2801,6 +2849,7 @@ module Woods
|
|
|
2801
2849
|
|
|
2802
2850
|
Array(result).compact
|
|
2803
2851
|
rescue StandardError => e
|
|
2852
|
+
@source_inputs&.unverified("extractor:#{rule.extractor_key}")
|
|
2804
2853
|
Rails.logger.warn "[Woods] #{rule.extractor_key} re-extraction of #{absolute_path} failed: #{e.message}"
|
|
2805
2854
|
# `nil`, not `[]`. The caller treats an empty result as "this path defines
|
|
2806
2855
|
# nothing any more" and prunes the units previously registered to it — so
|
|
@@ -2871,6 +2920,34 @@ module Woods
|
|
|
2871
2920
|
touched
|
|
2872
2921
|
end
|
|
2873
2922
|
|
|
2923
|
+
# Runtime-only model mixins can enter or leave discovery when their
|
|
2924
|
+
# includer changes, even if the mixin file itself is untouched.
|
|
2925
|
+
# @param affected_types [Set<Symbol>]
|
|
2926
|
+
# @return [Set<String>] Added or removed concern identifiers
|
|
2927
|
+
def reconcile_model_mixins(affected_types)
|
|
2928
|
+
extractor = extractor_for(:concerns)
|
|
2929
|
+
return Set.new unless extractor.respond_to?(:runtime_model_mixins)
|
|
2930
|
+
|
|
2931
|
+
live = extractor.runtime_model_mixins
|
|
2932
|
+
known = @dependency_graph.units_of_type(:concern).to_set
|
|
2933
|
+
added = live.flat_map do |path, modules|
|
|
2934
|
+
next [] if modules.all? { |mod| known.include?(mod.name) }
|
|
2935
|
+
|
|
2936
|
+
Array(extractor.extract_model_mixin_file(path)).reject { |unit| known.include?(unit.identifier) }
|
|
2937
|
+
end
|
|
2938
|
+
touched = register_and_write(:concerns, added, affected_types)
|
|
2939
|
+
return touched unless @eager_load_complete
|
|
2940
|
+
|
|
2941
|
+
live_names = live.values.flatten.to_set(&:name)
|
|
2942
|
+
known.each do |identifier|
|
|
2943
|
+
path = @dependency_graph.node(identifier, type: :concern)[:file_path]
|
|
2944
|
+
next if extractor.conventional_concern_path?(path) || live_names.include?(identifier)
|
|
2945
|
+
|
|
2946
|
+
touched.add(identifier) if remove_unit(identifier, affected_types, type: :concern)
|
|
2947
|
+
end
|
|
2948
|
+
touched
|
|
2949
|
+
end
|
|
2950
|
+
|
|
2874
2951
|
# Pruned class-based identifiers the tree still governs, and that the
|
|
2875
2952
|
# second reconciliation pass may therefore re-add.
|
|
2876
2953
|
#
|
|
@@ -2948,10 +3025,12 @@ module Woods
|
|
|
2948
3025
|
units = new_classes.filter_map do |klass|
|
|
2949
3026
|
extractor_for(key).public_send(spec[:method], klass)
|
|
2950
3027
|
rescue StandardError => e
|
|
3028
|
+
@source_inputs&.unverified("extractor:#{key}")
|
|
2951
3029
|
Rails.logger.warn "[Woods] #{key} extraction of #{klass} failed: #{e.message}"
|
|
2952
3030
|
nil
|
|
2953
3031
|
end
|
|
2954
3032
|
|
|
3033
|
+
source_consumer_failed?(key)
|
|
2955
3034
|
register_and_write(key, units, affected_types)
|
|
2956
3035
|
end
|
|
2957
3036
|
|
|
@@ -3086,6 +3165,10 @@ module Woods
|
|
|
3086
3165
|
# mutating durable state
|
|
3087
3166
|
def replace_type_wholesale(key, affected_types)
|
|
3088
3167
|
extractor = extractor_for(key)
|
|
3168
|
+
if extractor.nil?
|
|
3169
|
+
source_consumer_failed?(key, extractor)
|
|
3170
|
+
return Set.new
|
|
3171
|
+
end
|
|
3089
3172
|
return Set.new unless extractor.respond_to?(:extract_all)
|
|
3090
3173
|
|
|
3091
3174
|
@wholesale_mutations = 0
|
|
@@ -3094,6 +3177,8 @@ module Woods
|
|
|
3094
3177
|
|
|
3095
3178
|
touched = register_and_write(key, units, affected_types)
|
|
3096
3179
|
touched.merge(remove_replaced_units(key, units, affected_types))
|
|
3180
|
+
@source_inputs&.consume_extractor(key, units) unless source_consumer_failed?(key, extractor)
|
|
3181
|
+
touched
|
|
3097
3182
|
rescue StandardError => e
|
|
3098
3183
|
if @wholesale_mutations.to_i.positive?
|
|
3099
3184
|
raise Woods::ExtractionError, <<~MSG.tr("\n", ' ').strip
|
|
@@ -3106,6 +3191,7 @@ module Woods
|
|
|
3106
3191
|
MSG
|
|
3107
3192
|
end
|
|
3108
3193
|
|
|
3194
|
+
@source_inputs&.unverified("extractor:#{key}")
|
|
3109
3195
|
Rails.logger.error "[Woods] Wholesale re-run of #{key} failed: #{e.message}"
|
|
3110
3196
|
Set.new
|
|
3111
3197
|
end
|
|
@@ -3267,6 +3353,7 @@ module Woods
|
|
|
3267
3353
|
|
|
3268
3354
|
removed.add(identifier) if remove_unit(identifier, affected_types, type: type)
|
|
3269
3355
|
end
|
|
3356
|
+
@source_inputs&.consume_deleted(path)
|
|
3270
3357
|
end
|
|
3271
3358
|
end
|
|
3272
3359
|
|
|
@@ -3370,6 +3457,7 @@ module Woods
|
|
|
3370
3457
|
(@incremental_written ||= {})[unit.identifier] = unit.file_path
|
|
3371
3458
|
|
|
3372
3459
|
write_unit_file(type_dir.join(collision_safe_filename(unit.identifier)), unit)
|
|
3460
|
+
@source_inputs&.consume_unit(extractor_key, unit.file_path) unless source_consumer_failed?(extractor_key)
|
|
3373
3461
|
written.add(unit.identifier)
|
|
3374
3462
|
end
|
|
3375
3463
|
end
|
|
@@ -3446,12 +3534,14 @@ module Woods
|
|
|
3446
3534
|
def finalize_incremental_unit_json(affected_types)
|
|
3447
3535
|
dependents_dirty = @dependents_dirty || Set.new
|
|
3448
3536
|
git_dirty = @incremental_written || {}
|
|
3449
|
-
git_data = incremental_git_data(git_dirty.keys)
|
|
3537
|
+
git_data = profile_phase('git enrichment') { incremental_git_data(git_dirty.keys) }
|
|
3450
3538
|
|
|
3451
|
-
(
|
|
3452
|
-
|
|
3453
|
-
|
|
3454
|
-
|
|
3539
|
+
profile_phase('unit finalization') do
|
|
3540
|
+
(dependents_dirty | git_dirty.keys).each do |identifier|
|
|
3541
|
+
rewrite_unit_json(identifier, affected_types,
|
|
3542
|
+
refresh_dependents: dependents_dirty.include?(identifier),
|
|
3543
|
+
git_data: git_dirty.key?(identifier) ? git_data : nil)
|
|
3544
|
+
end
|
|
3455
3545
|
end
|
|
3456
3546
|
end
|
|
3457
3547
|
|
|
@@ -3560,7 +3650,7 @@ module Woods
|
|
|
3560
3650
|
end
|
|
3561
3651
|
|
|
3562
3652
|
# Batch-fetch git metadata for the units written by this run, in a single
|
|
3563
|
-
#
|
|
3653
|
+
# history walk, keyed by Rails.root-relative path the way
|
|
3564
3654
|
# {#batch_git_data} returns it.
|
|
3565
3655
|
#
|
|
3566
3656
|
# @param identifiers [Array<String>]
|
|
@@ -3568,11 +3658,12 @@ module Woods
|
|
|
3568
3658
|
def incremental_git_data(identifiers)
|
|
3569
3659
|
return {} if identifiers.empty? || !git_available?
|
|
3570
3660
|
|
|
3661
|
+
root = "#{Rails.root}/"
|
|
3571
3662
|
paths = identifiers.flat_map do |identifier|
|
|
3572
3663
|
@dependency_graph.nodes_for(identifier).filter_map do |node|
|
|
3573
3664
|
next if %i[rails_source gem_source].include?(node[:type])
|
|
3574
3665
|
|
|
3575
|
-
node[:file_path] if
|
|
3666
|
+
node[:file_path] if git_enrichable_path?(node[:file_path], root)
|
|
3576
3667
|
end
|
|
3577
3668
|
end
|
|
3578
3669
|
|
|
@@ -3627,11 +3718,15 @@ module Woods
|
|
|
3627
3718
|
return nil unless extractor_key
|
|
3628
3719
|
|
|
3629
3720
|
extractor = extractor_for(extractor_key)
|
|
3630
|
-
|
|
3721
|
+
if extractor.nil?
|
|
3722
|
+
source_consumer_failed?(extractor_key, extractor)
|
|
3723
|
+
return nil
|
|
3724
|
+
end
|
|
3631
3725
|
|
|
3632
3726
|
# File-based extractors can return several units from one file (a .rake
|
|
3633
3727
|
# file defining multiple tasks, etc.); class-based extractors return one.
|
|
3634
3728
|
units = Array(re_extracted_units(extractor, type, unit_id, file_path, extractor_key)).compact
|
|
3729
|
+
source_consumer_failed?(extractor_key, extractor)
|
|
3635
3730
|
return nil if units.empty?
|
|
3636
3731
|
|
|
3637
3732
|
register_and_write(extractor_key, units, affected_types)
|