woods 2.0.0.beta2 → 2.0.0.beta4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +339 -1
- data/CONTRIBUTING.md +188 -12
- data/README.md +93 -174
- data/SECURITY.md +9 -6
- data/docs/AGENT_GUIDE.md +109 -8
- data/docs/AGENT_SETUP.md +98 -7
- data/docs/BACKEND_MATRIX.md +25 -0
- data/docs/CLIENT_HOOKS.md +111 -0
- data/docs/CONFIGURATION_REFERENCE.md +267 -16
- data/docs/CONSOLE_MCP_SETUP.md +80 -7
- data/docs/DOCKER_SETUP.md +22 -3
- data/docs/EVALUATION.md +464 -1
- data/docs/EXTRACTOR_REFERENCE.md +45 -6
- data/docs/FAQ.md +11 -12
- data/docs/GETTING_STARTED.md +17 -5
- data/docs/INCREMENTAL_EXTRACTION.md +147 -7
- data/docs/INDEX_LAYOUT.md +382 -0
- data/docs/INTERNALS.md +7 -2
- data/docs/MCP_SERVERS.md +276 -5
- data/docs/MCP_TOOL_COOKBOOK.md +37 -22
- data/docs/MCP_WORKTREE_SETUP.md +43 -83
- data/docs/NOTION_INTEGRATION.md +13 -0
- data/docs/OBSIDIAN_INTEGRATION.md +57 -9
- data/docs/PUBLISHED_INDEX.md +72 -0
- data/docs/README.md +7 -0
- data/docs/RETRIEVAL_GUIDE.md +273 -12
- data/docs/RUNTIME_TRACING.md +71 -0
- data/docs/SOURCE_FRESHNESS.md +143 -0
- data/docs/TROUBLESHOOTING.md +129 -18
- data/docs/UNBLOCKED_INTEGRATION.md +25 -0
- data/docs/UPGRADING_TO_2.md +48 -22
- data/docs/WATCH_DAEMON.md +277 -67
- data/exe/woods-agent-config +6 -0
- data/exe/woods-extract +5 -0
- data/exe/woods-hook-context +6 -0
- data/exe/woods-mcp-start +14 -9
- data/lib/generators/woods/pgvector_generator.rb +8 -2
- data/lib/generators/woods/templates/woods.rb.tt +1 -3
- data/lib/tasks/woods.rake +47 -397
- data/lib/woods/agent_configuration/applier.rb +135 -0
- data/lib/woods/agent_configuration/cli.rb +101 -0
- data/lib/woods/agent_configuration/cli_options.rb +29 -0
- data/lib/woods/agent_configuration/document.rb +105 -0
- data/lib/woods/agent_configuration/error.rb +7 -0
- data/lib/woods/agent_configuration/launcher.rb +75 -0
- data/lib/woods/agent_configuration/layout.rb +72 -0
- data/lib/woods/agent_configuration/managed_section.rb +62 -0
- data/lib/woods/agent_configuration/plan.rb +98 -0
- data/lib/woods/agent_configuration/plan_diff.rb +38 -0
- data/lib/woods/agent_configuration/planned_files.rb +61 -0
- data/lib/woods/agent_configuration/planner.rb +63 -0
- data/lib/woods/agent_configuration/planner_validation.rb +77 -0
- data/lib/woods/agent_configuration/preflight.rb +100 -0
- data/lib/woods/agent_configuration/recovery.rb +49 -0
- data/lib/woods/ast/node.rb +2 -0
- data/lib/woods/ast/parser.rb +38 -5
- data/lib/woods/builder.rb +21 -5
- data/lib/woods/cache/cache_middleware.rb +28 -7
- data/lib/woods/cache/cache_store.rb +4 -5
- data/lib/woods/change_set.rb +5 -4
- data/lib/woods/console/credential_index.rb +20 -2
- data/lib/woods/console/credential_scanner.rb +18 -17
- data/lib/woods/console/credential_scanner_registry.rb +36 -0
- data/lib/woods/console/dispatch_pipeline.rb +7 -0
- data/lib/woods/console/embedded_executor.rb +32 -10
- data/lib/woods/console/encrypted_credential_snapshot.rb +16 -0
- data/lib/woods/console/rack_middleware.rb +22 -13
- data/lib/woods/console/server.rb +18 -16
- data/lib/woods/console/sql_noise_stripper.rb +9 -7
- data/lib/woods/console/sql_table_scanner.rb +47 -7
- data/lib/woods/console/sql_validator.rb +49 -9
- data/lib/woods/console/sqlite_read_guard.rb +46 -0
- data/lib/woods/coordination/pipeline_lock.rb +3 -2
- data/lib/woods/dependency_graph.rb +65 -13
- data/lib/woods/embedding/corpus.rb +94 -0
- data/lib/woods/embedding/indexer.rb +114 -60
- data/lib/woods/embedding/openai.rb +17 -6
- data/lib/woods/evaluation/ablation_executor.rb +6 -1
- data/lib/woods/evaluation/ablation_timed_executor.rb +22 -4
- data/lib/woods/export/typed_reader.rb +56 -0
- data/lib/woods/extractor.rb +277 -149
- data/lib/woods/extractors/action_cable_extractor.rb +3 -1
- data/lib/woods/extractors/behavioral_profile.rb +9 -7
- data/lib/woods/extractors/caching_extractor.rb +3 -1
- data/lib/woods/extractors/concern_extractor.rb +64 -6
- data/lib/woods/extractors/configuration_extractor.rb +7 -3
- data/lib/woods/extractors/controller_extractor.rb +13 -4
- data/lib/woods/extractors/database_view_extractor.rb +3 -1
- data/lib/woods/extractors/declared_parent.rb +55 -0
- data/lib/woods/extractors/decorator_extractor.rb +3 -1
- data/lib/woods/extractors/engine_extractor.rb +3 -1
- data/lib/woods/extractors/event_extractor.rb +4 -2
- data/lib/woods/extractors/factory_extractor.rb +3 -1
- data/lib/woods/extractors/graphql_extractor.rb +10 -13
- data/lib/woods/extractors/i18n_extractor.rb +3 -1
- data/lib/woods/extractors/job_extractor.rb +6 -19
- data/lib/woods/extractors/lib_extractor.rb +13 -9
- data/lib/woods/extractors/mailer_extractor.rb +26 -15
- data/lib/woods/extractors/manager_extractor.rb +3 -1
- data/lib/woods/extractors/method_parameters.rb +53 -0
- data/lib/woods/extractors/middleware_argument.rb +65 -0
- data/lib/woods/extractors/middleware_extractor.rb +9 -3
- data/lib/woods/extractors/migration_extractor.rb +3 -1
- data/lib/woods/extractors/model_extractor.rb +26 -34
- data/lib/woods/extractors/package_extractor.rb +24 -4
- data/lib/woods/extractors/phlex_extractor.rb +3 -1
- data/lib/woods/extractors/policy_extractor.rb +3 -1
- data/lib/woods/extractors/poro_extractor.rb +13 -9
- data/lib/woods/extractors/pundit_extractor.rb +3 -1
- data/lib/woods/extractors/rails_source_extractor.rb +4 -2
- data/lib/woods/extractors/rake_task_extractor.rb +4 -2
- data/lib/woods/extractors/route_extractor.rb +3 -1
- data/lib/woods/extractors/route_helper_resolver.rb +10 -33
- data/lib/woods/extractors/scheduled_job_extractor.rb +41 -15
- data/lib/woods/extractors/serializer_extractor.rb +4 -2
- data/lib/woods/extractors/service_extractor.rb +3 -1
- data/lib/woods/extractors/shared_dependency_scanner.rb +2 -2
- data/lib/woods/extractors/shared_utility_methods.rb +48 -19
- data/lib/woods/extractors/source_nesting.rb +1 -1
- data/lib/woods/extractors/state_machine_extractor.rb +3 -1
- data/lib/woods/extractors/test_mapping_extractor.rb +3 -1
- data/lib/woods/extractors/validator_extractor.rb +3 -1
- data/lib/woods/extractors/view_component_extractor.rb +3 -1
- data/lib/woods/extractors/view_template_extractor.rb +3 -1
- data/lib/woods/gem_mapper.rb +2 -0
- data/lib/woods/git_history.rb +116 -0
- data/lib/woods/graph_analyzer.rb +35 -6
- data/lib/woods/hooks/context_cli.rb +54 -0
- data/lib/woods/hooks/context_event.rb +88 -0
- data/lib/woods/hooks/context_hint.rb +73 -0
- data/lib/woods/hooks/context_impact.rb +77 -0
- data/lib/woods/hooks/context_output.rb +47 -0
- data/lib/woods/hooks/context_state.rb +102 -0
- data/lib/woods/hooks/refresh.rb +79 -0
- data/lib/woods/hooks/rule_projection.rb +78 -0
- data/lib/woods/input_rules.rb +19 -0
- data/lib/woods/mcp/bearer_auth.rb +22 -13
- data/lib/woods/mcp/bootstrapper.rb +79 -4
- data/lib/woods/mcp/config_resolver.rb +2 -1
- data/lib/woods/mcp/index_reader.rb +334 -162
- data/lib/woods/mcp/initialization_guidance.rb +27 -0
- data/lib/woods/mcp/origin_guard.rb +17 -9
- data/lib/woods/mcp/published_lexical_retriever.rb +115 -0
- data/lib/woods/mcp/renderers/markdown_renderer.rb +22 -9
- data/lib/woods/mcp/renderers/plain_renderer.rb +18 -8
- data/lib/woods/mcp/search_results.rb +74 -0
- data/lib/woods/mcp/server.rb +178 -63
- data/lib/woods/mcp/tool_contract.rb +3 -1
- data/lib/woods/mcp/tool_response_renderer.rb +41 -0
- data/lib/woods/mcp/traversal_evidence.rb +113 -0
- data/lib/woods/mcp/traversal_evidence_index.rb +100 -0
- data/lib/woods/mcp/traversal_evidence_page.rb +41 -0
- data/lib/woods/mcp/traversal_evidence_text.rb +52 -0
- data/lib/woods/mcp/traversal_response.rb +22 -0
- data/lib/woods/notion/exporter.rb +56 -17
- data/lib/woods/obsidian/destination_plan.rb +98 -0
- data/lib/woods/obsidian/name_mapper.rb +19 -3
- data/lib/woods/obsidian/note_builder.rb +19 -10
- data/lib/woods/obsidian/vault_exporter.rb +88 -32
- data/lib/woods/operator/pipeline_guard.rb +18 -13
- data/lib/woods/path_dispatcher.rb +13 -6
- data/lib/woods/payload_store.rb +27 -26
- data/lib/woods/published_index/typed_unit_reader.rb +40 -3
- data/lib/woods/published_index.rb +2 -2
- data/lib/woods/railtie.rb +3 -3
- data/lib/woods/railtie_support.rb +12 -12
- data/lib/woods/rake_helpers.rb +382 -0
- data/lib/woods/resilience/graph_invariant_validator/membership_checks.rb +71 -0
- data/lib/woods/resilience/graph_invariant_validator/node_checks.rb +61 -0
- data/lib/woods/resilience/graph_invariant_validator/reverse_relationship_checks.rb +46 -0
- data/lib/woods/resilience/graph_invariant_validator.rb +119 -0
- data/lib/woods/resilience/index_validator/graph_checks.rb +80 -0
- data/lib/woods/resilience/index_validator.rb +112 -23
- data/lib/woods/retrieval/context_assembler.rb +50 -15
- data/lib/woods/retrieval/lexical_assembler.rb +84 -0
- data/lib/woods/retrieval/lexical_index.rb +120 -0
- data/lib/woods/retrieval/ranker.rb +4 -2
- data/lib/woods/retrieval/scope.rb +108 -0
- data/lib/woods/retrieval/scoped_graph_store.rb +32 -0
- data/lib/woods/retrieval/scoped_vector_store.rb +55 -0
- data/lib/woods/retrieval/search_executor.rb +86 -27
- data/lib/woods/retrieval/source_evidence.rb +200 -0
- data/lib/woods/retriever.rb +98 -22
- data/lib/woods/ruby_analyzer/trace_enricher.rb +77 -38
- data/lib/woods/session_tracer/file_store.rb +6 -1
- data/lib/woods/session_tracer/middleware.rb +10 -12
- data/lib/woods/session_tracer/redis_store.rb +22 -6
- data/lib/woods/session_tracer/session_flow_assembler.rb +23 -17
- data/lib/woods/session_tracer/solid_cache_coordination.rb +6 -4
- data/lib/woods/session_tracer/unit_resolver.rb +63 -0
- data/lib/woods/source_inputs/consumer_errors.rb +31 -0
- data/lib/woods/source_inputs/handoff.rb +102 -0
- data/lib/woods/source_inputs/launcher.rb +157 -0
- data/lib/woods/source_inputs/manifest.rb +124 -0
- data/lib/woods/source_inputs/private_key.rb +55 -0
- data/lib/woods/source_inputs/scanner.rb +171 -0
- data/lib/woods/source_inputs/scopes.rb +71 -0
- data/lib/woods/source_inputs/session.rb +214 -0
- data/lib/woods/source_inputs/status.rb +84 -0
- data/lib/woods/source_inputs/verifier.rb +107 -0
- data/lib/woods/storage/metadata_store.rb +25 -25
- data/lib/woods/storage/pgvector.rb +35 -10
- data/lib/woods/storage/qdrant.rb +17 -7
- data/lib/woods/storage/vector_store.rb +18 -6
- data/lib/woods/tasks.rb +3 -2
- data/lib/woods/temporal/json_snapshot_store.rb +58 -9
- data/lib/woods/unblocked/exporter.rb +59 -70
- data/lib/woods/version.rb +1 -1
- data/lib/woods/watch/boot_snapshot.rb +52 -0
- data/lib/woods/watch/daemon.rb +154 -32
- data/lib/woods/watch/listen_watcher.rb +4 -0
- data/lib/woods/watch/polling_watcher.rb +5 -1
- data/lib/woods/watch/status.rb +20 -15
- data/lib/woods/watch/tree_scan.rb +21 -13
- data/lib/woods/watch/watcher.rb +4 -1
- data/lib/woods.rb +50 -11
- data/plugin/.claude-plugin/plugin.json +1 -1
- data/plugin/hooks/adapters/normalize.jq +15 -0
- data/plugin/hooks/adapters/normalize.rb +63 -0
- data/plugin/hooks/hooks.json +20 -0
- data/plugin/hooks/woods-context.sh +50 -0
- data/plugin/hooks/woods-input-rules.sh +159 -0
- data/plugin/hooks/woods-opencode.mjs +65 -0
- data/plugin/hooks/woods-post-edit.sh +2 -225
- data/plugin/hooks/woods-refresh.sh +260 -0
- data/plugin/hooks/woods-session-start.sh +47 -55
- data/plugin/skills/woods-agent-enable/SKILL.md +19 -0
- data/plugin/skills/woods-diagnose/SKILL.md +319 -1
- data/plugin/skills/woods-investigate/SKILL.md +145 -0
- data/plugin/skills/woods-mcp-config/SKILL.md +90 -2
- data/plugin/skills/woods-setup/SKILL.md +110 -6
- metadata +87 -5
|
@@ -8,6 +8,10 @@ require 'pathname'
|
|
|
8
8
|
require 'set'
|
|
9
9
|
|
|
10
10
|
require_relative '../generation'
|
|
11
|
+
require_relative '../source_inputs/status'
|
|
12
|
+
require_relative 'search_results'
|
|
13
|
+
require_relative '../retrieval/scope'
|
|
14
|
+
require_relative 'traversal_evidence'
|
|
11
15
|
|
|
12
16
|
module Woods
|
|
13
17
|
module MCP
|
|
@@ -40,6 +44,15 @@ module Woods
|
|
|
40
44
|
|
|
41
45
|
TYPE_TO_DIR = DIR_TO_TYPE.invert.freeze
|
|
42
46
|
|
|
47
|
+
# Most output directories contain one unit type. RailsSourceExtractor
|
|
48
|
+
# emits gem_source units in rails_source/, while GraphQL publishes four
|
|
49
|
+
# subtypes in graphql/. Directory-filtered search retains its historical
|
|
50
|
+
# family labels while artifact readers preserve the actual unit type.
|
|
51
|
+
UNIT_TYPES_BY_DIR = DIR_TO_TYPE.transform_values { |type| [type].freeze }
|
|
52
|
+
.merge('rails_source' => %w[rails_source gem_source].freeze,
|
|
53
|
+
'graphql' => %w[graphql_type graphql_mutation graphql_resolver graphql_query].freeze)
|
|
54
|
+
.freeze
|
|
55
|
+
|
|
43
56
|
# Maximum number of loaded unit files to cache in memory.
|
|
44
57
|
MAX_UNIT_CACHE = 50
|
|
45
58
|
|
|
@@ -47,11 +60,16 @@ module Woods
|
|
|
47
60
|
# @param auto_refresh [Boolean] re-read the index when its published
|
|
48
61
|
# generation moves. On by default; specs that assert caching behaviour
|
|
49
62
|
# turn it off.
|
|
50
|
-
# @raise [ArgumentError] if directory doesn't exist or
|
|
63
|
+
# @raise [ArgumentError] if directory doesn't exist or no published index resolves
|
|
51
64
|
def initialize(index_dir, auto_refresh: true)
|
|
52
65
|
@index_dir = Pathname.new(index_dir)
|
|
53
66
|
raise ArgumentError, "Index directory does not exist: #{index_dir}" unless @index_dir.directory?
|
|
54
|
-
|
|
67
|
+
unless manifest_present?
|
|
68
|
+
raise ArgumentError, "Could not resolve a published Woods index in: #{@index_dir.expand_path}\n" \
|
|
69
|
+
'Expected generation.json pointing to a payload manifest.json, or a legacy flat manifest.json. ' \
|
|
70
|
+
'Point IndexReader at an existing index root. ' \
|
|
71
|
+
'If no index exists, run `bundle exec rake woods:extract` in your Rails app.'
|
|
72
|
+
end
|
|
55
73
|
|
|
56
74
|
@unit_cache = {}
|
|
57
75
|
@unit_cache_signatures = {}
|
|
@@ -89,6 +107,13 @@ module Woods
|
|
|
89
107
|
# @return [Integer, nil] nil until something has been read
|
|
90
108
|
attr_reader :loaded_generation
|
|
91
109
|
|
|
110
|
+
# Cache identity includes the publish token: overlapping writers can
|
|
111
|
+
# legitimately publish the same number with a different immutable payload.
|
|
112
|
+
def generation_identity
|
|
113
|
+
ensure_fresh!
|
|
114
|
+
[@loaded_generation, @loaded_token, current_payload_dir.to_s].freeze
|
|
115
|
+
end
|
|
116
|
+
|
|
92
117
|
# Drop caches if the index has been rewritten since they were populated.
|
|
93
118
|
#
|
|
94
119
|
# This is what makes the MCP `reload` tool an optimization rather than a
|
|
@@ -274,6 +299,7 @@ module Woods
|
|
|
274
299
|
@raw_graph_data = nil
|
|
275
300
|
@normalized_graph_edges = nil
|
|
276
301
|
@graph_node_types = nil
|
|
302
|
+
@traversal_evidence_index = nil
|
|
277
303
|
remove_instance_variable(:@multi_database_graph) if defined?(@multi_database_graph)
|
|
278
304
|
end
|
|
279
305
|
|
|
@@ -295,6 +321,15 @@ module Woods
|
|
|
295
321
|
current_payload_dir
|
|
296
322
|
end
|
|
297
323
|
|
|
324
|
+
# Verify the immutable payload this reader is actually serving. Do not
|
|
325
|
+
# cache source results: an edit can occur without an index generation move.
|
|
326
|
+
def source_freshness(mode: 'quick')
|
|
327
|
+
with_pinned_generation do
|
|
328
|
+
SourceInputs::Status.new(output_dir: @index_dir, payload_dir: current_payload_dir,
|
|
329
|
+
generation: @payload_dir ? @loaded_generation : 0, mode: mode).call
|
|
330
|
+
end
|
|
331
|
+
end
|
|
332
|
+
|
|
298
333
|
# @return [Hash] Parsed manifest.json
|
|
299
334
|
def manifest
|
|
300
335
|
ensure_fresh!
|
|
@@ -337,7 +372,19 @@ module Woods
|
|
|
337
372
|
#
|
|
338
373
|
# @param identifier [String] Unit identifier (e.g. "Post", "Api::V1::HealthController")
|
|
339
374
|
# @return [Hash, nil] Full unit data or nil if not found
|
|
340
|
-
def find_unit(identifier)
|
|
375
|
+
def find_unit(identifier, type: nil)
|
|
376
|
+
if type
|
|
377
|
+
return with_pinned_generation do
|
|
378
|
+
dir = UNIT_TYPES_BY_DIR.find { |_, types| types.include?(type) }&.first
|
|
379
|
+
next nil unless dir
|
|
380
|
+
raise IOError, "symlink unit directory: #{dir}" if current_payload_dir.join(dir).symlink?
|
|
381
|
+
|
|
382
|
+
next nil unless search_index_entries(dir).any? { |entry| entry['identifier'] == identifier }
|
|
383
|
+
|
|
384
|
+
unit = read_published_unit(dir, identifier)
|
|
385
|
+
unit if unit['type'] == type
|
|
386
|
+
end
|
|
387
|
+
end
|
|
341
388
|
ensure_fresh!
|
|
342
389
|
location = identifier_map[identifier]
|
|
343
390
|
return nil unless location
|
|
@@ -361,6 +408,32 @@ module Woods
|
|
|
361
408
|
dirs.flat_map { |dir| read_index(dir) }
|
|
362
409
|
end
|
|
363
410
|
|
|
411
|
+
# Enumerate complete typed published units, holding one generation pin.
|
|
412
|
+
# Bulk consumers must fail closed on a corrupt/missing entry instead of
|
|
413
|
+
# silently constructing a partial retrieval index. Bare-name lookup cannot
|
|
414
|
+
# do this because multiple types can legitimately share an identifier.
|
|
415
|
+
def each_unit
|
|
416
|
+
return enum_for(__method__) unless block_given?
|
|
417
|
+
|
|
418
|
+
with_pinned_generation do
|
|
419
|
+
TYPE_DIRS.each do |dir|
|
|
420
|
+
directory = current_payload_dir.join(dir)
|
|
421
|
+
raise IOError, "symlink unit directory: #{dir}" if directory.symlink?
|
|
422
|
+
|
|
423
|
+
entries = published_unit_entries(dir)
|
|
424
|
+
seen = Set.new
|
|
425
|
+
entries.each do |entry|
|
|
426
|
+
id = entry.is_a?(Hash) && entry['identifier']
|
|
427
|
+
unless id.is_a?(String) && !id.empty? && seen.add?(id)
|
|
428
|
+
raise IOError, "invalid or duplicate unit identifier in #{dir}/_index.json"
|
|
429
|
+
end
|
|
430
|
+
|
|
431
|
+
yield read_published_unit(dir, id)
|
|
432
|
+
end
|
|
433
|
+
end
|
|
434
|
+
end
|
|
435
|
+
end
|
|
436
|
+
|
|
364
437
|
# Default maximum number of unit files to load during phase-2 search.
|
|
365
438
|
# Override with WOODS_SEARCH_MAX_SCAN env var.
|
|
366
439
|
DEFAULT_SEARCH_MAX_SCAN = 500
|
|
@@ -400,23 +473,23 @@ module Woods
|
|
|
400
473
|
# @param limit [Integer] Maximum results to return
|
|
401
474
|
# @param exact_prefix [String, nil] Literal identifier prefix filter (case-insensitive)
|
|
402
475
|
# @param exact_suffix [String, nil] Literal identifier suffix filter (case-insensitive)
|
|
403
|
-
# @return [Hash]
|
|
476
|
+
# @return [Hash] results, optional note/partial, and explicit completeness evidence
|
|
404
477
|
# @raise [ArgumentError] when all of query, exact_prefix, and exact_suffix are blank
|
|
405
|
-
def search(query = nil, types: nil, fields: %w[identifier], limit: 20, exact_prefix: nil, exact_suffix: nil
|
|
406
|
-
|
|
407
|
-
#
|
|
408
|
-
#
|
|
409
|
-
# result list still held identifiers from the previous generation — one
|
|
410
|
-
# response describing two indexes.
|
|
478
|
+
def search(query = nil, types: nil, fields: %w[identifier], limit: 20, exact_prefix: nil, exact_suffix: nil,
|
|
479
|
+
packages: nil, source_paths: nil)
|
|
480
|
+
# Keep summaries, typed deep reads and lookahead on one generation,
|
|
481
|
+
# including when publication advances after the result page fills.
|
|
411
482
|
with_pinned_generation do
|
|
412
483
|
search_within_pin(query, types: types, fields: fields, limit: limit,
|
|
413
|
-
exact_prefix: exact_prefix, exact_suffix: exact_suffix
|
|
484
|
+
exact_prefix: exact_prefix, exact_suffix: exact_suffix,
|
|
485
|
+
packages: packages, source_paths: source_paths)
|
|
414
486
|
end
|
|
415
487
|
end
|
|
416
488
|
|
|
417
489
|
# @api private
|
|
418
490
|
def search_within_pin(query = nil, types: nil, fields: %w[identifier], limit: 20,
|
|
419
|
-
exact_prefix: nil, exact_suffix: nil)
|
|
491
|
+
exact_prefix: nil, exact_suffix: nil, packages: nil, source_paths: nil)
|
|
492
|
+
scope = search_scope(packages, source_paths, types)
|
|
420
493
|
prefix = exact_prefix.blank? ? nil : exact_prefix.downcase
|
|
421
494
|
suffix = exact_suffix.blank? ? nil : exact_suffix.downcase
|
|
422
495
|
if query.blank? && !prefix && !suffix
|
|
@@ -430,114 +503,126 @@ module Woods
|
|
|
430
503
|
max_scan = max_scan_env.empty? ? DEFAULT_SEARCH_MAX_SCAN : max_scan_env.to_i
|
|
431
504
|
max_scan = DEFAULT_SEARCH_MAX_SCAN if max_scan <= 0
|
|
432
505
|
|
|
433
|
-
results =
|
|
506
|
+
results = SearchResults.new(limit: limit)
|
|
434
507
|
notes = []
|
|
435
508
|
phase2_scanned = 0
|
|
436
|
-
partial = false
|
|
437
509
|
|
|
438
510
|
begin
|
|
439
|
-
dirs =
|
|
440
|
-
|
|
441
|
-
|
|
442
|
-
TYPE_DIRS
|
|
443
|
-
end
|
|
444
|
-
|
|
445
|
-
# Phase 2 candidates are collected per-dir and then scanned in
|
|
446
|
-
# round-robin across dirs. Exhausting the per-run scan cap linearly
|
|
447
|
-
# down TYPE_DIRS order would starve later types (`concerns` at pos
|
|
448
|
-
# 13, `test_mappings` at pos 31) on any codebase where the earlier
|
|
449
|
-
# dirs together exceed max_scan entries. Interleaving guarantees
|
|
450
|
-
# every type contributes to the scanned set.
|
|
511
|
+
dirs = scope || !types ? TYPE_DIRS : types.filter_map { |type| TYPE_TO_DIR[type] }.uniq
|
|
512
|
+
# Identifier matches retain priority. Deep candidates are interleaved
|
|
513
|
+
# across types so an early large directory cannot consume their budget.
|
|
451
514
|
phase2_queues = {}
|
|
452
|
-
|
|
453
|
-
|
|
454
|
-
|
|
455
|
-
|
|
456
|
-
|
|
457
|
-
|
|
458
|
-
|
|
459
|
-
|
|
460
|
-
|
|
461
|
-
|
|
462
|
-
|
|
463
|
-
|
|
515
|
+
catch(:search_done) do
|
|
516
|
+
dirs.each do |dir|
|
|
517
|
+
type_name = DIR_TO_TYPE[dir]
|
|
518
|
+
entries = search_index_entries(dir)
|
|
519
|
+
entries = scoped_search_entries(entries, dir, scope) if scope
|
|
520
|
+
if entries.size > 1
|
|
521
|
+
matching_count = entries.count do |entry|
|
|
522
|
+
identifier_passes_filters?(entry['identifier'], pattern, prefix, suffix)
|
|
523
|
+
end
|
|
524
|
+
if matching_count > entries.size / 2.0
|
|
525
|
+
notes << "broad pattern matched #{matching_count}/#{entries.size} entries in #{dir}"
|
|
526
|
+
end
|
|
464
527
|
end
|
|
465
|
-
end
|
|
466
528
|
|
|
467
|
-
|
|
468
|
-
|
|
469
|
-
|
|
470
|
-
|
|
471
|
-
# Phase 1: identifier matching (still in-order per dir)
|
|
472
|
-
if fields.include?('identifier') && pattern.match?(id)
|
|
473
|
-
next if results.size >= limit
|
|
529
|
+
entries.each do |entry|
|
|
530
|
+
type_name = entry.fetch('scope_type', DIR_TO_TYPE[dir])
|
|
531
|
+
id = entry['identifier']
|
|
532
|
+
next unless identifier_passes_prefix_suffix?(id, prefix, suffix)
|
|
474
533
|
|
|
475
|
-
|
|
476
|
-
|
|
477
|
-
|
|
534
|
+
if fields.include?('identifier') && pattern.match?(id)
|
|
535
|
+
results.add(identifier: id, type: type_name, match_field: 'identifier')
|
|
536
|
+
throw :search_done if results.result_limit_reached?
|
|
478
537
|
|
|
479
|
-
|
|
480
|
-
|
|
538
|
+
next
|
|
539
|
+
end
|
|
540
|
+
next unless fields.include?('metadata') || fields.include?('source_code')
|
|
481
541
|
|
|
482
|
-
|
|
542
|
+
(phase2_queues[dir] ||= []) << [type_name, id]
|
|
543
|
+
end
|
|
483
544
|
end
|
|
484
|
-
end
|
|
485
|
-
|
|
486
|
-
if results.size < limit && phase2_queues.any?
|
|
487
|
-
queues = phase2_queues.values.map(&:dup)
|
|
488
|
-
catch(:phase2_done) do
|
|
489
|
-
loop do
|
|
490
|
-
progressed = false
|
|
491
|
-
queues.each do |queue|
|
|
492
|
-
next if queue.empty?
|
|
493
|
-
|
|
494
|
-
throw :phase2_done if results.size >= limit
|
|
495
|
-
|
|
496
|
-
if phase2_scanned >= max_scan
|
|
497
|
-
partial = true
|
|
498
|
-
throw :phase2_done
|
|
499
|
-
end
|
|
500
|
-
|
|
501
|
-
type_name, id = queue.shift
|
|
502
|
-
progressed = true
|
|
503
545
|
|
|
504
|
-
|
|
505
|
-
|
|
546
|
+
queues = phase2_queues.values
|
|
547
|
+
loop do
|
|
548
|
+
progressed = false
|
|
549
|
+
queues.each do |queue|
|
|
550
|
+
next if queue.empty?
|
|
506
551
|
|
|
507
|
-
|
|
508
|
-
|
|
509
|
-
|
|
510
|
-
results << { identifier: id, type: type_name, match_field: 'source_code' }
|
|
511
|
-
elsif fields.include?('metadata') && unit['metadata'] && pattern.match?(unit['metadata'].to_json)
|
|
512
|
-
results << { identifier: id, type: type_name, match_field: 'metadata' }
|
|
513
|
-
end
|
|
552
|
+
if phase2_scanned >= max_scan
|
|
553
|
+
results.stop('scan_budget')
|
|
554
|
+
throw :search_done
|
|
514
555
|
end
|
|
515
|
-
|
|
556
|
+
type_name, id = queue.shift
|
|
557
|
+
progressed = true
|
|
558
|
+
phase2_scanned += 1
|
|
559
|
+
unit = scope ? scope.metadata_store.find(StorageIdentity.key(id, type_name)) : load_search_unit(type_name, id)
|
|
560
|
+
field = if fields.include?('source_code') && unit['source_code'] && pattern.match?(unit['source_code'])
|
|
561
|
+
'source_code'
|
|
562
|
+
elsif fields.include?('metadata') && unit['metadata'] && pattern.match?(unit['metadata'].to_json)
|
|
563
|
+
'metadata'
|
|
564
|
+
end
|
|
565
|
+
next unless field
|
|
566
|
+
|
|
567
|
+
results.add(identifier: id, type: type_name, match_field: field)
|
|
568
|
+
throw :search_done if results.result_limit_reached?
|
|
516
569
|
end
|
|
570
|
+
break unless progressed
|
|
517
571
|
end
|
|
518
572
|
end
|
|
519
573
|
rescue StandardError => e
|
|
520
574
|
raise unless regexp_timeout_error?(e)
|
|
521
575
|
|
|
522
576
|
notes << "search aborted: the pattern exceeded the #{SEARCH_PATTERN_TIMEOUT}s per-match limit"
|
|
523
|
-
|
|
577
|
+
results.stop('regex_timeout')
|
|
524
578
|
end
|
|
525
579
|
|
|
526
|
-
response =
|
|
527
|
-
response[:
|
|
528
|
-
response[:partial] = true if partial
|
|
580
|
+
response = results.finish.response(note: notes.join('; '))
|
|
581
|
+
response[:applied_scope] = scope.summary if scope
|
|
529
582
|
response
|
|
530
583
|
end
|
|
531
584
|
|
|
585
|
+
# Scope preparation reads the complete pinned unit snapshot. Search's
|
|
586
|
+
# deep-field scan budget still governs matching work after this read.
|
|
587
|
+
def search_scope(packages, source_paths, types)
|
|
588
|
+
return unless Retrieval::Scope.requested?(packages: packages, source_paths: source_paths)
|
|
589
|
+
|
|
590
|
+
metadata = Storage::MetadataStore::InMemory.new
|
|
591
|
+
each_unit { |unit| metadata.store(StorageIdentity.key(unit.fetch('identifier'), unit.fetch('type')), unit) }
|
|
592
|
+
Retrieval::Scope.new(metadata_store: metadata, packages: packages, source_paths: source_paths, types: types)
|
|
593
|
+
end
|
|
594
|
+
private :search_scope
|
|
595
|
+
|
|
596
|
+
def scoped_search_entries(entries, dir, scope)
|
|
597
|
+
entries.flat_map do |entry|
|
|
598
|
+
UNIT_TYPES_BY_DIR.fetch(dir).filter_map do |type|
|
|
599
|
+
key = StorageIdentity.key(entry['identifier'], type)
|
|
600
|
+
entry.merge('scope_type' => type) if scope.include?(key)
|
|
601
|
+
end
|
|
602
|
+
end
|
|
603
|
+
end
|
|
604
|
+
private :scoped_search_entries
|
|
605
|
+
|
|
532
606
|
# BFS traversal of forward dependencies.
|
|
533
607
|
#
|
|
534
608
|
# @param identifier [String] Starting unit identifier
|
|
535
609
|
# @param depth [Integer] Maximum traversal depth
|
|
536
610
|
# @param types [Array<String>, nil] Filter to these singular type names
|
|
537
611
|
# @param via [Array<String>, nil] Filter to these relationship types (e.g. ["link_to", "redirect_to"])
|
|
538
|
-
# @
|
|
539
|
-
|
|
540
|
-
|
|
612
|
+
# @param max_nodes [Integer] Distinct admitted nodes including root (default 1000; clamped to 1..10_000)
|
|
613
|
+
# @param max_edges [Integer] Edge checks before filters, including reverse via checks
|
|
614
|
+
# (default 10_000; clamped to 1..100_000). Graph loading/cache preparation is not budgeted.
|
|
615
|
+
# @return [Hash] { root:, found:, nodes: { id => { type:, depth:, deps: [] } } };
|
|
616
|
+
# incomplete walks add partial: true, partial_reason (node_budget or edge_budget),
|
|
617
|
+
# and traversal_budget (max_nodes, max_edges, visited_nodes, visited_edges).
|
|
618
|
+
# Already discovered rows remain; empty deps in a partial result need not mean a leaf.
|
|
619
|
+
def traverse_dependencies(identifier, depth: 2, types: nil, via: nil, max_nodes: 1000, max_edges: 10_000, explain: false)
|
|
620
|
+
if explain
|
|
621
|
+
return traverse_with_evidence(identifier, depth: depth, types: types, via: via, direction: :forward,
|
|
622
|
+
max_nodes: max_nodes, max_edges: max_edges)
|
|
623
|
+
end
|
|
624
|
+
|
|
625
|
+
traverse(identifier, depth: depth, types: types, via: via, direction: :forward, max_nodes: max_nodes, max_edges: max_edges)
|
|
541
626
|
end
|
|
542
627
|
|
|
543
628
|
# BFS traversal of reverse dependencies (dependents).
|
|
@@ -546,9 +631,20 @@ module Woods
|
|
|
546
631
|
# @param depth [Integer] Maximum traversal depth
|
|
547
632
|
# @param types [Array<String>, nil] Filter to these singular type names
|
|
548
633
|
# @param via [Array<String>, nil] Filter to these relationship types (e.g. ["link_to", "redirect_to"])
|
|
549
|
-
# @
|
|
550
|
-
|
|
551
|
-
|
|
634
|
+
# @param max_nodes [Integer] Distinct admitted nodes including root (default 1000; clamped to 1..10_000)
|
|
635
|
+
# @param max_edges [Integer] Edge checks before filters, including reverse via checks
|
|
636
|
+
# (default 10_000; clamped to 1..100_000). Graph loading/cache preparation is not budgeted.
|
|
637
|
+
# @return [Hash] { root:, found:, nodes: { id => { type:, depth:, deps: [] } } };
|
|
638
|
+
# incomplete walks add partial: true, partial_reason (node_budget or edge_budget),
|
|
639
|
+
# and traversal_budget (max_nodes, max_edges, visited_nodes, visited_edges).
|
|
640
|
+
# Already discovered rows remain; empty deps in a partial result need not mean a leaf.
|
|
641
|
+
def traverse_dependents(identifier, depth: 2, types: nil, via: nil, max_nodes: 1000, max_edges: 10_000, explain: false)
|
|
642
|
+
if explain
|
|
643
|
+
return traverse_with_evidence(identifier, depth: depth, types: types, via: via, direction: :reverse,
|
|
644
|
+
max_nodes: max_nodes, max_edges: max_edges)
|
|
645
|
+
end
|
|
646
|
+
|
|
647
|
+
traverse(identifier, depth: depth, types: types, via: via, direction: :reverse, max_nodes: max_nodes, max_edges: max_edges)
|
|
552
648
|
end
|
|
553
649
|
|
|
554
650
|
# Search rails_source units by concept keyword.
|
|
@@ -578,7 +674,7 @@ module Woods
|
|
|
578
674
|
break if results.size >= limit
|
|
579
675
|
|
|
580
676
|
id = entry['identifier']
|
|
581
|
-
unit =
|
|
677
|
+
unit = load_search_unit('rails_source', id)
|
|
582
678
|
next unless unit
|
|
583
679
|
|
|
584
680
|
metadata_json = unit['metadata']&.to_json
|
|
@@ -627,7 +723,7 @@ module Woods
|
|
|
627
723
|
entries = read_index(dir)
|
|
628
724
|
entries.each do |entry|
|
|
629
725
|
id = entry['identifier']
|
|
630
|
-
unit =
|
|
726
|
+
unit = load_search_unit(DIR_TO_TYPE.fetch(dir), id)
|
|
631
727
|
next unless unit
|
|
632
728
|
|
|
633
729
|
last_modified = unit.dig('metadata', 'git', 'last_modified')
|
|
@@ -872,6 +968,10 @@ module Woods
|
|
|
872
968
|
|
|
873
969
|
marker = Woods::Generation.new(output_dir: @index_dir).current
|
|
874
970
|
resolve_payload_dir(marker).join('manifest.json').file?
|
|
971
|
+
rescue TypeError, NoMethodError
|
|
972
|
+
# Match Bootstrapper's startup preflight for malformed marker shapes.
|
|
973
|
+
# Keep this local: errors during later generation refresh still surface.
|
|
974
|
+
false
|
|
875
975
|
end
|
|
876
976
|
|
|
877
977
|
# The loaded generation's payload directory, without a freshness check.
|
|
@@ -1102,15 +1202,80 @@ module Woods
|
|
|
1102
1202
|
entries = read_index(dir)
|
|
1103
1203
|
entries.each do |entry|
|
|
1104
1204
|
id = entry['identifier']
|
|
1105
|
-
|
|
1106
|
-
digest = Digest::SHA256.hexdigest(id)[0, 8]
|
|
1107
|
-
filename = "#{base}_#{digest}.json"
|
|
1108
|
-
map[id] = { type_dir: dir, filename: filename }
|
|
1205
|
+
map[id] = { type_dir: dir, filename: unit_filename(id) }
|
|
1109
1206
|
end
|
|
1110
1207
|
end
|
|
1111
1208
|
map
|
|
1112
1209
|
end
|
|
1113
1210
|
|
|
1211
|
+
# Read the selected type bucket, never the identifier-only lookup map.
|
|
1212
|
+
# Keep the existing per-file signature checks and LRU cache for deep reads.
|
|
1213
|
+
def load_search_unit(type, identifier)
|
|
1214
|
+
data = load_unit(TYPE_TO_DIR.fetch(type), unit_filename(identifier))
|
|
1215
|
+
unless valid_published_unit?(data, TYPE_TO_DIR.fetch(type), identifier)
|
|
1216
|
+
raise IOError, "typed unit identity mismatch: #{type}:#{identifier}"
|
|
1217
|
+
end
|
|
1218
|
+
|
|
1219
|
+
data
|
|
1220
|
+
end
|
|
1221
|
+
|
|
1222
|
+
def unit_filename(identifier)
|
|
1223
|
+
base = identifier.gsub('::', '__').gsub(/[^a-zA-Z0-9_-]/, '_')
|
|
1224
|
+
"#{base}_#{Digest::SHA256.hexdigest(identifier)[0, 8]}.json"
|
|
1225
|
+
end
|
|
1226
|
+
|
|
1227
|
+
def search_index_entries(dir)
|
|
1228
|
+
entries = published_unit_entries(dir)
|
|
1229
|
+
seen = Set.new
|
|
1230
|
+
entries.each do |entry|
|
|
1231
|
+
id = entry.is_a?(Hash) && entry['identifier']
|
|
1232
|
+
unless id.is_a?(String) && !id.empty? && seen.add?(id)
|
|
1233
|
+
raise IOError, "invalid or duplicate unit identifier in #{dir}/_index.json"
|
|
1234
|
+
end
|
|
1235
|
+
end
|
|
1236
|
+
entries
|
|
1237
|
+
end
|
|
1238
|
+
|
|
1239
|
+
def published_unit_entries(dir)
|
|
1240
|
+
index_path = current_payload_dir.join(dir, '_index.json')
|
|
1241
|
+
raise IOError, "symlink unit index: #{dir}" if index_path.symlink?
|
|
1242
|
+
|
|
1243
|
+
expected = manifest.dig('counts', dir)
|
|
1244
|
+
raise IOError, "missing unit index: #{dir}/_index.json" if !index_path.file? && expected.to_i.positive?
|
|
1245
|
+
|
|
1246
|
+
entries = read_index(dir)
|
|
1247
|
+
raise IOError, "invalid unit index: #{dir}" unless entries.is_a?(Array)
|
|
1248
|
+
if expected.is_a?(Integer) && entries.size != expected
|
|
1249
|
+
raise IOError, "unit count mismatch in #{dir}: expected #{expected}, found #{entries.size}"
|
|
1250
|
+
end
|
|
1251
|
+
|
|
1252
|
+
entries
|
|
1253
|
+
end
|
|
1254
|
+
|
|
1255
|
+
def read_published_unit(dir, identifier)
|
|
1256
|
+
base = identifier.gsub('::', '__').gsub(/[^a-zA-Z0-9_-]/, '_')
|
|
1257
|
+
filename = "#{base}_#{Digest::SHA256.hexdigest(identifier)[0, 8]}.json"
|
|
1258
|
+
path = current_payload_dir.join(dir, filename)
|
|
1259
|
+
raise IOError, "symlink unit file: #{dir}/#{filename}" if path.symlink?
|
|
1260
|
+
|
|
1261
|
+
raise IOError, "non-regular unit file: #{dir}/#{filename}" unless path.lstat.file?
|
|
1262
|
+
|
|
1263
|
+
data = File.open(path, File::RDONLY | File::NOFOLLOW | File::NONBLOCK) do |file|
|
|
1264
|
+
raise IOError, "non-regular unit file: #{dir}/#{filename}" unless file.stat.file?
|
|
1265
|
+
|
|
1266
|
+
JSON.parse(file.read)
|
|
1267
|
+
end
|
|
1268
|
+
unless valid_published_unit?(data, dir, identifier)
|
|
1269
|
+
raise IOError, "typed unit identity mismatch: #{dir}/#{filename}"
|
|
1270
|
+
end
|
|
1271
|
+
|
|
1272
|
+
data
|
|
1273
|
+
end
|
|
1274
|
+
|
|
1275
|
+
def valid_published_unit?(data, dir, identifier)
|
|
1276
|
+
data.is_a?(Hash) && data['identifier'] == identifier && UNIT_TYPES_BY_DIR.fetch(dir).include?(data['type'])
|
|
1277
|
+
end
|
|
1278
|
+
|
|
1114
1279
|
# Read and cache an _index.json file for a type directory.
|
|
1115
1280
|
def read_index(dir)
|
|
1116
1281
|
@index_cache ||= {}
|
|
@@ -1229,6 +1394,16 @@ module Woods
|
|
|
1229
1394
|
end
|
|
1230
1395
|
end
|
|
1231
1396
|
|
|
1397
|
+
# Ownership/type preparation is cached once per retained generation, like
|
|
1398
|
+
# graph_node_types. It never flattens adjacency; the per-query walker charges
|
|
1399
|
+
# every candidate edge, including legacy reverse evidence recovery.
|
|
1400
|
+
def traverse_with_evidence(identifier, **options)
|
|
1401
|
+
with_pinned_generation do
|
|
1402
|
+
@traversal_evidence_index ||= TraversalEvidenceIndex.new(raw_graph_data)
|
|
1403
|
+
TraversalEvidence.new(@traversal_evidence_index).call(identifier, **options)
|
|
1404
|
+
end
|
|
1405
|
+
end
|
|
1406
|
+
|
|
1232
1407
|
# BFS traversal in either direction.
|
|
1233
1408
|
#
|
|
1234
1409
|
# Edges may be stored as bare strings (old format) or as
|
|
@@ -1241,70 +1416,89 @@ module Woods
|
|
|
1241
1416
|
# @param via [Array<String>, nil] Filter to these relationship types
|
|
1242
1417
|
# @param direction [:forward, :reverse] Traversal direction
|
|
1243
1418
|
# @return [Hash]
|
|
1244
|
-
def traverse(identifier, depth:, types:, via:, direction:)
|
|
1419
|
+
def traverse(identifier, depth:, types:, via:, direction:, max_nodes:, max_edges:)
|
|
1245
1420
|
graph_data = raw_graph_data
|
|
1246
1421
|
nodes_data = graph_data['nodes'] || {}
|
|
1247
|
-
|
|
1248
1422
|
return { root: identifier, found: false, nodes: {} } unless nodes_data.key?(identifier)
|
|
1249
1423
|
|
|
1250
|
-
|
|
1251
|
-
|
|
1252
|
-
|
|
1424
|
+
budget = {
|
|
1425
|
+
max_nodes: max_nodes.to_i.clamp(1, 10_000),
|
|
1426
|
+
max_edges: max_edges.to_i.clamp(1, 100_000),
|
|
1427
|
+
visited_nodes: 1, visited_edges: 0
|
|
1428
|
+
}
|
|
1253
1429
|
type_set = types&.to_set
|
|
1254
1430
|
via_set = via&.to_set
|
|
1255
1431
|
visited = Set.new([identifier])
|
|
1256
1432
|
queue = [[identifier, 0]]
|
|
1257
1433
|
result_nodes = {}
|
|
1434
|
+
cursor = 0
|
|
1435
|
+
partial_reason = nil
|
|
1258
1436
|
|
|
1259
|
-
while queue.
|
|
1260
|
-
current, current_depth = queue
|
|
1261
|
-
|
|
1262
|
-
neighbors = if direction == :forward
|
|
1263
|
-
resolve_forward_neighbors(normalized_edges, current, via_set)
|
|
1264
|
-
else
|
|
1265
|
-
resolve_reverse_neighbors(graph_data, normalized_edges, current, via_set)
|
|
1266
|
-
end
|
|
1267
|
-
|
|
1268
|
-
# Filter by node type if requested. An identifier naming units of
|
|
1269
|
-
# several types matches when any of them does — excluding it because
|
|
1270
|
-
# the type that happens to sort first is not the requested one would
|
|
1271
|
-
# hide a unit the filter asked for.
|
|
1272
|
-
filtered = if type_set
|
|
1273
|
-
neighbors.select { |n| graph_node_types[n]&.any? { |t| type_set.include?(t) } }
|
|
1274
|
-
else
|
|
1275
|
-
neighbors
|
|
1276
|
-
end
|
|
1277
|
-
|
|
1278
|
-
# At max depth, record the node with empty deps so the renderer
|
|
1279
|
-
# doesn't emit an extra level of unexpanded neighbors. The parent
|
|
1280
|
-
# node's deps list already shows this node as a child.
|
|
1281
|
-
will_expand = current_depth < depth
|
|
1437
|
+
while cursor < queue.size
|
|
1438
|
+
current, current_depth = queue[cursor]
|
|
1439
|
+
cursor += 1
|
|
1282
1440
|
node_meta = nodes_data[current]
|
|
1283
|
-
entry = {
|
|
1284
|
-
type: node_meta&.dig('type'),
|
|
1285
|
-
depth: current_depth,
|
|
1286
|
-
deps: will_expand ? filtered : []
|
|
1287
|
-
}
|
|
1288
|
-
# Only when the identifier is genuinely ambiguous, so the shape is
|
|
1289
|
-
# unchanged for every node in an index with no shared identifiers.
|
|
1441
|
+
entry = { type: node_meta&.dig('type'), depth: current_depth, deps: [] }
|
|
1290
1442
|
node_types = graph_node_types[current] || []
|
|
1291
1443
|
entry[:types] = node_types if node_types.size > 1
|
|
1292
|
-
# Likewise: a single-database app learns nothing from a column that
|
|
1293
|
-
# answers the same thing on every row (B-183).
|
|
1294
1444
|
entry[:database] = node_meta&.dig('database') if multi_database_graph?
|
|
1295
1445
|
result_nodes[current] = entry
|
|
1446
|
+
next if partial_reason || current_depth >= depth
|
|
1447
|
+
|
|
1448
|
+
# Keep every discovered row, even after expansion stops. No dangling
|
|
1449
|
+
# deps are introduced by the budget, and paging stays independent.
|
|
1450
|
+
partial_reason = catch(:traversal_budget) do
|
|
1451
|
+
each_traversal_neighbor(graph_data, current, direction, via_set, budget) do |neighbor|
|
|
1452
|
+
next if type_set && !graph_node_types[neighbor]&.any? { |type| type_set.include?(type) }
|
|
1453
|
+
|
|
1454
|
+
unless visited.include?(neighbor)
|
|
1455
|
+
throw :traversal_budget, 'node_budget' if visited.size >= budget[:max_nodes]
|
|
1456
|
+
|
|
1457
|
+
visited.add(neighbor)
|
|
1458
|
+
budget[:visited_nodes] += 1
|
|
1459
|
+
queue << [neighbor, current_depth + 1]
|
|
1460
|
+
end
|
|
1461
|
+
entry[:deps] << neighbor
|
|
1462
|
+
end
|
|
1463
|
+
nil
|
|
1464
|
+
end
|
|
1465
|
+
end
|
|
1296
1466
|
|
|
1297
|
-
|
|
1467
|
+
result = { root: identifier, found: true, nodes: result_nodes }
|
|
1468
|
+
result.merge!(partial: true, partial_reason: partial_reason, traversal_budget: budget) if partial_reason
|
|
1469
|
+
result
|
|
1470
|
+
end
|
|
1298
1471
|
|
|
1299
|
-
|
|
1300
|
-
|
|
1301
|
-
|
|
1302
|
-
|
|
1472
|
+
# Walk stored adjacency order lazily. In reverse traversal a via filter
|
|
1473
|
+
# must inspect forward edges too; those checks consume the same budget.
|
|
1474
|
+
# Whole-generation JSON loading is deliberately outside the walk budget.
|
|
1475
|
+
def each_traversal_neighbor(graph_data, identifier, direction, via_set, budget)
|
|
1476
|
+
edges = normalized_graph_edges
|
|
1477
|
+
if direction == :forward
|
|
1478
|
+
(edges[identifier] || []).each do |edge|
|
|
1479
|
+
consume_traversal_edge(budget)
|
|
1480
|
+
edge = { 'target' => edge } unless edge.is_a?(Hash)
|
|
1481
|
+
yield edge['target'] unless via_set && !via_set.include?(edge['via'])
|
|
1482
|
+
end
|
|
1483
|
+
else
|
|
1484
|
+
((graph_data['reverse'] || {})[identifier] || []).each do |dependent|
|
|
1485
|
+
consume_traversal_edge(budget)
|
|
1486
|
+
if via_set
|
|
1487
|
+
matches = (edges[dependent] || []).any? do |edge|
|
|
1488
|
+
consume_traversal_edge(budget)
|
|
1489
|
+
edge.is_a?(Hash) && edge['target'] == identifier && via_set.include?(edge['via'])
|
|
1490
|
+
end
|
|
1491
|
+
next unless matches
|
|
1303
1492
|
end
|
|
1493
|
+
yield dependent
|
|
1304
1494
|
end
|
|
1305
1495
|
end
|
|
1496
|
+
end
|
|
1497
|
+
|
|
1498
|
+
def consume_traversal_edge(budget)
|
|
1499
|
+
throw :traversal_budget, 'edge_budget' if budget[:visited_edges] >= budget[:max_edges]
|
|
1306
1500
|
|
|
1307
|
-
|
|
1501
|
+
budget[:visited_edges] += 1
|
|
1308
1502
|
end
|
|
1309
1503
|
|
|
1310
1504
|
# Normalize all edge arrays once, converting bare strings to hashes.
|
|
@@ -1321,28 +1515,6 @@ module Woods
|
|
|
1321
1515
|
entries.map { |e| e.is_a?(Hash) ? e : { 'target' => e } }
|
|
1322
1516
|
end
|
|
1323
1517
|
end
|
|
1324
|
-
|
|
1325
|
-
# Extract forward neighbor identifiers, optionally filtered by via type.
|
|
1326
|
-
# Expects pre-normalized edges (all entries are hashes).
|
|
1327
|
-
def resolve_forward_neighbors(normalized_edges, identifier, via_set)
|
|
1328
|
-
edges = normalized_edges[identifier] || []
|
|
1329
|
-
edges = edges.select { |e| via_set.include?(e['via']) } if via_set
|
|
1330
|
-
edges.map { |e| e['target'] }
|
|
1331
|
-
end
|
|
1332
|
-
|
|
1333
|
-
# Extract reverse neighbor identifiers, optionally filtered by via type.
|
|
1334
|
-
# Reverse edges are stored as bare identifier arrays. When via filtering
|
|
1335
|
-
# is requested, checks each dependent's pre-normalized forward edges to
|
|
1336
|
-
# find those pointing at +identifier+ with a matching via type.
|
|
1337
|
-
def resolve_reverse_neighbors(graph_data, normalized_edges, identifier, via_set)
|
|
1338
|
-
dependents = (graph_data['reverse'] || {})[identifier] || []
|
|
1339
|
-
return dependents unless via_set
|
|
1340
|
-
|
|
1341
|
-
dependents.select do |dep|
|
|
1342
|
-
forward = normalized_edges[dep] || []
|
|
1343
|
-
forward.any? { |e| e['target'] == identifier && via_set.include?(e['via']) }
|
|
1344
|
-
end
|
|
1345
|
-
end
|
|
1346
1518
|
end
|
|
1347
1519
|
end
|
|
1348
1520
|
end
|