woods 2.0.0.beta1 → 2.0.0.beta3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (221) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +400 -1
  3. data/CONTRIBUTING.md +224 -9
  4. data/README.md +7 -3
  5. data/SECURITY.md +9 -6
  6. data/docs/AGENT_GUIDE.md +83 -4
  7. data/docs/AGENT_SETUP.md +82 -1
  8. data/docs/BACKEND_MATRIX.md +20 -0
  9. data/docs/CLIENT_HOOKS.md +111 -0
  10. data/docs/CONFIGURATION_REFERENCE.md +233 -13
  11. data/docs/CONSOLE_MCP_SETUP.md +35 -5
  12. data/docs/DOCKER_SETUP.md +21 -2
  13. data/docs/EVALUATION.md +464 -1
  14. data/docs/EXTRACTOR_REFERENCE.md +36 -5
  15. data/docs/FAQ.md +11 -12
  16. data/docs/GETTING_STARTED.md +17 -5
  17. data/docs/INCREMENTAL_EXTRACTION.md +158 -2
  18. data/docs/INDEX_LAYOUT.md +382 -0
  19. data/docs/INTERNALS.md +15 -7
  20. data/docs/MCP_SERVERS.md +221 -5
  21. data/docs/MCP_TOOL_COOKBOOK.md +33 -18
  22. data/docs/NOTION_INTEGRATION.md +13 -0
  23. data/docs/OBSIDIAN_INTEGRATION.md +57 -9
  24. data/docs/PUBLISHED_INDEX.md +71 -0
  25. data/docs/README.md +7 -0
  26. data/docs/RETRIEVAL_GUIDE.md +253 -11
  27. data/docs/RUNTIME_TRACING.md +71 -0
  28. data/docs/SOURCE_FRESHNESS.md +143 -0
  29. data/docs/TROUBLESHOOTING.md +117 -5
  30. data/docs/UNBLOCKED_INTEGRATION.md +25 -0
  31. data/docs/UPGRADING_TO_2.md +44 -22
  32. data/docs/WATCH_DAEMON.md +259 -59
  33. data/exe/woods-agent-config +6 -0
  34. data/exe/woods-extract +5 -0
  35. data/exe/woods-hook-context +6 -0
  36. data/lib/generators/woods/templates/woods.rb.tt +1 -3
  37. data/lib/tasks/woods.rake +47 -397
  38. data/lib/woods/agent_configuration/applier.rb +133 -0
  39. data/lib/woods/agent_configuration/cli.rb +101 -0
  40. data/lib/woods/agent_configuration/cli_options.rb +29 -0
  41. data/lib/woods/agent_configuration/document.rb +105 -0
  42. data/lib/woods/agent_configuration/error.rb +7 -0
  43. data/lib/woods/agent_configuration/launcher.rb +75 -0
  44. data/lib/woods/agent_configuration/layout.rb +59 -0
  45. data/lib/woods/agent_configuration/managed_section.rb +62 -0
  46. data/lib/woods/agent_configuration/plan.rb +98 -0
  47. data/lib/woods/agent_configuration/plan_diff.rb +38 -0
  48. data/lib/woods/agent_configuration/planned_files.rb +61 -0
  49. data/lib/woods/agent_configuration/planner.rb +63 -0
  50. data/lib/woods/agent_configuration/planner_validation.rb +77 -0
  51. data/lib/woods/agent_configuration/preflight.rb +100 -0
  52. data/lib/woods/agent_configuration/recovery.rb +49 -0
  53. data/lib/woods/ast/node.rb +2 -0
  54. data/lib/woods/ast/parser.rb +38 -5
  55. data/lib/woods/atomic_file.rb +133 -3
  56. data/lib/woods/builder.rb +21 -5
  57. data/lib/woods/cache/cache_middleware.rb +28 -7
  58. data/lib/woods/cache/cache_store.rb +4 -5
  59. data/lib/woods/change_set.rb +5 -4
  60. data/lib/woods/console/credential_index.rb +20 -2
  61. data/lib/woods/console/credential_scanner.rb +14 -14
  62. data/lib/woods/console/credential_scanner_registry.rb +36 -0
  63. data/lib/woods/console/embedded_executor.rb +1 -1
  64. data/lib/woods/console/encrypted_credential_snapshot.rb +16 -0
  65. data/lib/woods/console/rack_middleware.rb +22 -13
  66. data/lib/woods/console/server.rb +18 -16
  67. data/lib/woods/dependency_graph.rb +65 -13
  68. data/lib/woods/embedding/corpus.rb +94 -0
  69. data/lib/woods/embedding/indexer.rb +90 -46
  70. data/lib/woods/embedding/openai.rb +17 -6
  71. data/lib/woods/evaluation/ablation_executor.rb +6 -1
  72. data/lib/woods/evaluation/ablation_timed_executor.rb +22 -4
  73. data/lib/woods/export/typed_reader.rb +56 -0
  74. data/lib/woods/extractor.rb +557 -228
  75. data/lib/woods/extractors/action_cable_extractor.rb +3 -1
  76. data/lib/woods/extractors/behavioral_profile.rb +9 -7
  77. data/lib/woods/extractors/caching_extractor.rb +3 -1
  78. data/lib/woods/extractors/concern_extractor.rb +64 -6
  79. data/lib/woods/extractors/configuration_extractor.rb +7 -3
  80. data/lib/woods/extractors/controller_extractor.rb +13 -4
  81. data/lib/woods/extractors/database_view_extractor.rb +3 -1
  82. data/lib/woods/extractors/decorator_extractor.rb +3 -1
  83. data/lib/woods/extractors/engine_extractor.rb +3 -1
  84. data/lib/woods/extractors/event_extractor.rb +4 -2
  85. data/lib/woods/extractors/factory_extractor.rb +3 -1
  86. data/lib/woods/extractors/graphql_extractor.rb +8 -2
  87. data/lib/woods/extractors/i18n_extractor.rb +3 -1
  88. data/lib/woods/extractors/job_extractor.rb +6 -19
  89. data/lib/woods/extractors/lib_extractor.rb +3 -1
  90. data/lib/woods/extractors/mailer_extractor.rb +20 -5
  91. data/lib/woods/extractors/manager_extractor.rb +3 -1
  92. data/lib/woods/extractors/method_parameters.rb +53 -0
  93. data/lib/woods/extractors/middleware_argument.rb +65 -0
  94. data/lib/woods/extractors/middleware_extractor.rb +9 -3
  95. data/lib/woods/extractors/migration_extractor.rb +3 -1
  96. data/lib/woods/extractors/model_extractor.rb +39 -33
  97. data/lib/woods/extractors/package_extractor.rb +24 -4
  98. data/lib/woods/extractors/phlex_extractor.rb +3 -1
  99. data/lib/woods/extractors/policy_extractor.rb +3 -1
  100. data/lib/woods/extractors/poro_extractor.rb +3 -1
  101. data/lib/woods/extractors/pundit_extractor.rb +3 -1
  102. data/lib/woods/extractors/rails_source_extractor.rb +4 -2
  103. data/lib/woods/extractors/rake_task_extractor.rb +4 -2
  104. data/lib/woods/extractors/route_extractor.rb +3 -1
  105. data/lib/woods/extractors/route_helper_resolver.rb +10 -33
  106. data/lib/woods/extractors/scheduled_job_extractor.rb +41 -15
  107. data/lib/woods/extractors/serializer_extractor.rb +4 -2
  108. data/lib/woods/extractors/service_extractor.rb +3 -1
  109. data/lib/woods/extractors/shared_dependency_scanner.rb +2 -2
  110. data/lib/woods/extractors/shared_utility_methods.rb +27 -15
  111. data/lib/woods/extractors/source_nesting.rb +1 -1
  112. data/lib/woods/extractors/state_machine_extractor.rb +3 -1
  113. data/lib/woods/extractors/test_mapping_extractor.rb +3 -1
  114. data/lib/woods/extractors/validator_extractor.rb +3 -1
  115. data/lib/woods/extractors/view_component_extractor.rb +3 -1
  116. data/lib/woods/extractors/view_template_extractor.rb +3 -1
  117. data/lib/woods/flow_assembler.rb +87 -8
  118. data/lib/woods/flow_precomputer.rb +44 -7
  119. data/lib/woods/gem_mapper.rb +2 -0
  120. data/lib/woods/git_history.rb +116 -0
  121. data/lib/woods/graph_analyzer.rb +195 -63
  122. data/lib/woods/hooks/context_cli.rb +54 -0
  123. data/lib/woods/hooks/context_event.rb +88 -0
  124. data/lib/woods/hooks/context_hint.rb +73 -0
  125. data/lib/woods/hooks/context_impact.rb +77 -0
  126. data/lib/woods/hooks/context_output.rb +47 -0
  127. data/lib/woods/hooks/context_state.rb +102 -0
  128. data/lib/woods/hooks/refresh.rb +79 -0
  129. data/lib/woods/hooks/rule_projection.rb +78 -0
  130. data/lib/woods/input_rules.rb +19 -0
  131. data/lib/woods/mcp/bearer_auth.rb +20 -12
  132. data/lib/woods/mcp/bootstrapper.rb +62 -0
  133. data/lib/woods/mcp/index_reader.rb +323 -160
  134. data/lib/woods/mcp/initialization_guidance.rb +27 -0
  135. data/lib/woods/mcp/origin_guard.rb +17 -9
  136. data/lib/woods/mcp/published_lexical_retriever.rb +115 -0
  137. data/lib/woods/mcp/renderers/markdown_renderer.rb +8 -1
  138. data/lib/woods/mcp/renderers/plain_renderer.rb +7 -1
  139. data/lib/woods/mcp/search_results.rb +74 -0
  140. data/lib/woods/mcp/server.rb +158 -37
  141. data/lib/woods/mcp/tool_contract.rb +2 -0
  142. data/lib/woods/mcp/tool_response_renderer.rb +25 -0
  143. data/lib/woods/mcp/traversal_evidence.rb +113 -0
  144. data/lib/woods/mcp/traversal_evidence_index.rb +100 -0
  145. data/lib/woods/mcp/traversal_evidence_page.rb +41 -0
  146. data/lib/woods/mcp/traversal_evidence_text.rb +52 -0
  147. data/lib/woods/notion/exporter.rb +56 -17
  148. data/lib/woods/obsidian/destination_plan.rb +98 -0
  149. data/lib/woods/obsidian/name_mapper.rb +19 -3
  150. data/lib/woods/obsidian/note_builder.rb +19 -10
  151. data/lib/woods/obsidian/vault_exporter.rb +88 -32
  152. data/lib/woods/operator/pipeline_guard.rb +18 -13
  153. data/lib/woods/path_dispatcher.rb +7 -1
  154. data/lib/woods/payload_store.rb +29 -15
  155. data/lib/woods/railtie.rb +3 -3
  156. data/lib/woods/railtie_support.rb +12 -12
  157. data/lib/woods/rake_helpers.rb +392 -0
  158. data/lib/woods/resilience/graph_invariant_validator/membership_checks.rb +71 -0
  159. data/lib/woods/resilience/graph_invariant_validator/node_checks.rb +61 -0
  160. data/lib/woods/resilience/graph_invariant_validator/reverse_relationship_checks.rb +46 -0
  161. data/lib/woods/resilience/graph_invariant_validator.rb +119 -0
  162. data/lib/woods/resilience/index_validator/graph_checks.rb +80 -0
  163. data/lib/woods/resilience/index_validator.rb +112 -23
  164. data/lib/woods/retrieval/context_assembler.rb +50 -15
  165. data/lib/woods/retrieval/lexical_assembler.rb +73 -0
  166. data/lib/woods/retrieval/lexical_index.rb +119 -0
  167. data/lib/woods/retrieval/ranker.rb +4 -2
  168. data/lib/woods/retrieval/scope.rb +108 -0
  169. data/lib/woods/retrieval/scoped_graph_store.rb +32 -0
  170. data/lib/woods/retrieval/scoped_vector_store.rb +55 -0
  171. data/lib/woods/retrieval/search_executor.rb +86 -27
  172. data/lib/woods/retrieval/source_evidence.rb +200 -0
  173. data/lib/woods/retriever.rb +98 -22
  174. data/lib/woods/ruby_analyzer/trace_enricher.rb +80 -38
  175. data/lib/woods/session_tracer/middleware.rb +10 -12
  176. data/lib/woods/session_tracer/redis_store.rb +22 -6
  177. data/lib/woods/session_tracer/session_flow_assembler.rb +23 -17
  178. data/lib/woods/session_tracer/solid_cache_coordination.rb +6 -4
  179. data/lib/woods/session_tracer/unit_resolver.rb +63 -0
  180. data/lib/woods/source_inputs/consumer_errors.rb +27 -0
  181. data/lib/woods/source_inputs/handoff.rb +102 -0
  182. data/lib/woods/source_inputs/launcher.rb +157 -0
  183. data/lib/woods/source_inputs/manifest.rb +124 -0
  184. data/lib/woods/source_inputs/private_key.rb +55 -0
  185. data/lib/woods/source_inputs/scanner.rb +171 -0
  186. data/lib/woods/source_inputs/scopes.rb +71 -0
  187. data/lib/woods/source_inputs/session.rb +214 -0
  188. data/lib/woods/source_inputs/status.rb +84 -0
  189. data/lib/woods/source_inputs/verifier.rb +107 -0
  190. data/lib/woods/storage/metadata_store.rb +25 -25
  191. data/lib/woods/storage/pgvector.rb +29 -8
  192. data/lib/woods/storage/qdrant.rb +17 -7
  193. data/lib/woods/storage/vector_store.rb +18 -6
  194. data/lib/woods/tasks.rb +3 -2
  195. data/lib/woods/temporal/json_snapshot_store.rb +29 -8
  196. data/lib/woods/unblocked/exporter.rb +59 -70
  197. data/lib/woods/version.rb +1 -1
  198. data/lib/woods/watch/boot_snapshot.rb +52 -0
  199. data/lib/woods/watch/daemon.rb +136 -28
  200. data/lib/woods/watch/listen_watcher.rb +4 -0
  201. data/lib/woods/watch/polling_watcher.rb +5 -1
  202. data/lib/woods/watch/status.rb +20 -15
  203. data/lib/woods/watch/tree_scan.rb +21 -13
  204. data/lib/woods/watch/watcher.rb +4 -1
  205. data/lib/woods.rb +135 -11
  206. data/plugin/.claude-plugin/plugin.json +1 -1
  207. data/plugin/hooks/adapters/normalize.jq +15 -0
  208. data/plugin/hooks/adapters/normalize.rb +63 -0
  209. data/plugin/hooks/hooks.json +20 -0
  210. data/plugin/hooks/woods-context.sh +50 -0
  211. data/plugin/hooks/woods-input-rules.sh +159 -0
  212. data/plugin/hooks/woods-opencode.mjs +65 -0
  213. data/plugin/hooks/woods-post-edit.sh +2 -225
  214. data/plugin/hooks/woods-refresh.sh +260 -0
  215. data/plugin/hooks/woods-session-start.sh +47 -55
  216. data/plugin/skills/woods-agent-enable/SKILL.md +13 -0
  217. data/plugin/skills/woods-diagnose/SKILL.md +288 -1
  218. data/plugin/skills/woods-investigate/SKILL.md +106 -0
  219. data/plugin/skills/woods-mcp-config/SKILL.md +89 -1
  220. data/plugin/skills/woods-setup/SKILL.md +107 -6
  221. metadata +84 -5
@@ -8,6 +8,10 @@ require 'pathname'
8
8
  require 'set'
9
9
 
10
10
  require_relative '../generation'
11
+ require_relative '../source_inputs/status'
12
+ require_relative 'search_results'
13
+ require_relative '../retrieval/scope'
14
+ require_relative 'traversal_evidence'
11
15
 
12
16
  module Woods
13
17
  module MCP
@@ -40,6 +44,15 @@ module Woods
40
44
 
41
45
  TYPE_TO_DIR = DIR_TO_TYPE.invert.freeze
42
46
 
47
+ # Most output directories contain one unit type. RailsSourceExtractor
48
+ # emits gem_source units in rails_source/, while GraphQL publishes four
49
+ # subtypes in graphql/. Directory-filtered search retains its historical
50
+ # family labels while artifact readers preserve the actual unit type.
51
+ UNIT_TYPES_BY_DIR = DIR_TO_TYPE.transform_values { |type| [type].freeze }
52
+ .merge('rails_source' => %w[rails_source gem_source].freeze,
53
+ 'graphql' => %w[graphql_type graphql_mutation graphql_resolver graphql_query].freeze)
54
+ .freeze
55
+
43
56
  # Maximum number of loaded unit files to cache in memory.
44
57
  MAX_UNIT_CACHE = 50
45
58
 
@@ -89,6 +102,13 @@ module Woods
89
102
  # @return [Integer, nil] nil until something has been read
90
103
  attr_reader :loaded_generation
91
104
 
105
+ # Cache identity includes the publish token: overlapping writers can
106
+ # legitimately publish the same number with a different immutable payload.
107
+ def generation_identity
108
+ ensure_fresh!
109
+ [@loaded_generation, @loaded_token, current_payload_dir.to_s].freeze
110
+ end
111
+
92
112
  # Drop caches if the index has been rewritten since they were populated.
93
113
  #
94
114
  # This is what makes the MCP `reload` tool an optimization rather than a
@@ -274,6 +294,7 @@ module Woods
274
294
  @raw_graph_data = nil
275
295
  @normalized_graph_edges = nil
276
296
  @graph_node_types = nil
297
+ @traversal_evidence_index = nil
277
298
  remove_instance_variable(:@multi_database_graph) if defined?(@multi_database_graph)
278
299
  end
279
300
 
@@ -295,6 +316,15 @@ module Woods
295
316
  current_payload_dir
296
317
  end
297
318
 
319
+ # Verify the immutable payload this reader is actually serving. Do not
320
+ # cache source results: an edit can occur without an index generation move.
321
+ def source_freshness(mode: 'quick')
322
+ with_pinned_generation do
323
+ SourceInputs::Status.new(output_dir: @index_dir, payload_dir: current_payload_dir,
324
+ generation: @payload_dir ? @loaded_generation : 0, mode: mode).call
325
+ end
326
+ end
327
+
298
328
  # @return [Hash] Parsed manifest.json
299
329
  def manifest
300
330
  ensure_fresh!
@@ -337,7 +367,19 @@ module Woods
337
367
  #
338
368
  # @param identifier [String] Unit identifier (e.g. "Post", "Api::V1::HealthController")
339
369
  # @return [Hash, nil] Full unit data or nil if not found
340
- def find_unit(identifier)
370
+ def find_unit(identifier, type: nil)
371
+ if type
372
+ return with_pinned_generation do
373
+ dir = UNIT_TYPES_BY_DIR.find { |_, types| types.include?(type) }&.first
374
+ next nil unless dir
375
+ raise IOError, "symlink unit directory: #{dir}" if current_payload_dir.join(dir).symlink?
376
+
377
+ next nil unless search_index_entries(dir).any? { |entry| entry['identifier'] == identifier }
378
+
379
+ unit = read_published_unit(dir, identifier)
380
+ unit if unit['type'] == type
381
+ end
382
+ end
341
383
  ensure_fresh!
342
384
  location = identifier_map[identifier]
343
385
  return nil unless location
@@ -361,6 +403,32 @@ module Woods
361
403
  dirs.flat_map { |dir| read_index(dir) }
362
404
  end
363
405
 
406
+ # Enumerate complete typed published units, holding one generation pin.
407
+ # Bulk consumers must fail closed on a corrupt/missing entry instead of
408
+ # silently constructing a partial retrieval index. Bare-name lookup cannot
409
+ # do this because multiple types can legitimately share an identifier.
410
+ def each_unit
411
+ return enum_for(__method__) unless block_given?
412
+
413
+ with_pinned_generation do
414
+ TYPE_DIRS.each do |dir|
415
+ directory = current_payload_dir.join(dir)
416
+ raise IOError, "symlink unit directory: #{dir}" if directory.symlink?
417
+
418
+ entries = published_unit_entries(dir)
419
+ seen = Set.new
420
+ entries.each do |entry|
421
+ id = entry.is_a?(Hash) && entry['identifier']
422
+ unless id.is_a?(String) && !id.empty? && seen.add?(id)
423
+ raise IOError, "invalid or duplicate unit identifier in #{dir}/_index.json"
424
+ end
425
+
426
+ yield read_published_unit(dir, id)
427
+ end
428
+ end
429
+ end
430
+ end
431
+
364
432
  # Default maximum number of unit files to load during phase-2 search.
365
433
  # Override with WOODS_SEARCH_MAX_SCAN env var.
366
434
  DEFAULT_SEARCH_MAX_SCAN = 500
@@ -400,23 +468,23 @@ module Woods
400
468
  # @param limit [Integer] Maximum results to return
401
469
  # @param exact_prefix [String, nil] Literal identifier prefix filter (case-insensitive)
402
470
  # @param exact_suffix [String, nil] Literal identifier suffix filter (case-insensitive)
403
- # @return [Hash] { results: Array<Hash>, note: String|nil, partial: Boolean }
471
+ # @return [Hash] results, optional note/partial, and explicit completeness evidence
404
472
  # @raise [ArgumentError] when all of query, exact_prefix, and exact_suffix are blank
405
- def search(query = nil, types: nil, fields: %w[identifier], limit: 20, exact_prefix: nil, exact_suffix: nil)
406
- # Pinned, not merely checked-once. This walks the identifier map and
407
- # then loads units for the hits; each nested `find_unit` re-checks
408
- # freshness, so a publish landing mid-walk rebuilt the caches while the
409
- # result list still held identifiers from the previous generation — one
410
- # response describing two indexes.
473
+ def search(query = nil, types: nil, fields: %w[identifier], limit: 20, exact_prefix: nil, exact_suffix: nil,
474
+ packages: nil, source_paths: nil)
475
+ # Keep summaries, typed deep reads and lookahead on one generation,
476
+ # including when publication advances after the result page fills.
411
477
  with_pinned_generation do
412
478
  search_within_pin(query, types: types, fields: fields, limit: limit,
413
- exact_prefix: exact_prefix, exact_suffix: exact_suffix)
479
+ exact_prefix: exact_prefix, exact_suffix: exact_suffix,
480
+ packages: packages, source_paths: source_paths)
414
481
  end
415
482
  end
416
483
 
417
484
  # @api private
418
485
  def search_within_pin(query = nil, types: nil, fields: %w[identifier], limit: 20,
419
- exact_prefix: nil, exact_suffix: nil)
486
+ exact_prefix: nil, exact_suffix: nil, packages: nil, source_paths: nil)
487
+ scope = search_scope(packages, source_paths, types)
420
488
  prefix = exact_prefix.blank? ? nil : exact_prefix.downcase
421
489
  suffix = exact_suffix.blank? ? nil : exact_suffix.downcase
422
490
  if query.blank? && !prefix && !suffix
@@ -430,114 +498,126 @@ module Woods
430
498
  max_scan = max_scan_env.empty? ? DEFAULT_SEARCH_MAX_SCAN : max_scan_env.to_i
431
499
  max_scan = DEFAULT_SEARCH_MAX_SCAN if max_scan <= 0
432
500
 
433
- results = []
501
+ results = SearchResults.new(limit: limit)
434
502
  notes = []
435
503
  phase2_scanned = 0
436
- partial = false
437
504
 
438
505
  begin
439
- dirs = if types
440
- types.filter_map { |t| TYPE_TO_DIR[t] }
441
- else
442
- TYPE_DIRS
443
- end
444
-
445
- # Phase 2 candidates are collected per-dir and then scanned in
446
- # round-robin across dirs. Exhausting the per-run scan cap linearly
447
- # down TYPE_DIRS order would starve later types (`concerns` at pos
448
- # 13, `test_mappings` at pos 31) on any codebase where the earlier
449
- # dirs together exceed max_scan entries. Interleaving guarantees
450
- # every type contributes to the scanned set.
506
+ dirs = scope || !types ? TYPE_DIRS : types.filter_map { |type| TYPE_TO_DIR[type] }.uniq
507
+ # Identifier matches retain priority. Deep candidates are interleaved
508
+ # across types so an early large directory cannot consume their budget.
451
509
  phase2_queues = {}
452
-
453
- dirs.each do |dir|
454
- type_name = DIR_TO_TYPE[dir]
455
- entries = read_index(dir)
456
-
457
- # Broad-match detection: warn when pattern matches >50% of dir entries
458
- if entries.size > 1
459
- matching_count = entries.count do |e|
460
- identifier_passes_filters?(e['identifier'], pattern, prefix, suffix)
461
- end
462
- if matching_count > entries.size / 2.0
463
- notes << "broad pattern matched #{matching_count}/#{entries.size} entries in #{dir}"
510
+ catch(:search_done) do
511
+ dirs.each do |dir|
512
+ type_name = DIR_TO_TYPE[dir]
513
+ entries = search_index_entries(dir)
514
+ entries = scoped_search_entries(entries, dir, scope) if scope
515
+ if entries.size > 1
516
+ matching_count = entries.count do |entry|
517
+ identifier_passes_filters?(entry['identifier'], pattern, prefix, suffix)
518
+ end
519
+ if matching_count > entries.size / 2.0
520
+ notes << "broad pattern matched #{matching_count}/#{entries.size} entries in #{dir}"
521
+ end
464
522
  end
465
- end
466
523
 
467
- entries.each do |entry|
468
- id = entry['identifier']
469
- next unless identifier_passes_prefix_suffix?(id, prefix, suffix)
524
+ entries.each do |entry|
525
+ type_name = entry.fetch('scope_type', DIR_TO_TYPE[dir])
526
+ id = entry['identifier']
527
+ next unless identifier_passes_prefix_suffix?(id, prefix, suffix)
470
528
 
471
- # Phase 1: identifier matching (still in-order per dir)
472
- if fields.include?('identifier') && pattern.match?(id)
473
- next if results.size >= limit
529
+ if fields.include?('identifier') && pattern.match?(id)
530
+ results.add(identifier: id, type: type_name, match_field: 'identifier')
531
+ throw :search_done if results.result_limit_reached?
474
532
 
475
- results << { identifier: id, type: type_name, match_field: 'identifier' }
476
- next
477
- end
478
-
479
- # Phase 2 is only reached when the caller opted into deeper fields.
480
- next unless fields.include?('metadata') || fields.include?('source_code')
533
+ next
534
+ end
535
+ next unless fields.include?('metadata') || fields.include?('source_code')
481
536
 
482
- (phase2_queues[dir] ||= []) << [type_name, id]
537
+ (phase2_queues[dir] ||= []) << [type_name, id]
538
+ end
483
539
  end
484
- end
485
-
486
- if results.size < limit && phase2_queues.any?
487
- queues = phase2_queues.values.map(&:dup)
488
- catch(:phase2_done) do
489
- loop do
490
- progressed = false
491
- queues.each do |queue|
492
- next if queue.empty?
493
-
494
- throw :phase2_done if results.size >= limit
495
-
496
- if phase2_scanned >= max_scan
497
- partial = true
498
- throw :phase2_done
499
- end
500
-
501
- type_name, id = queue.shift
502
- progressed = true
503
540
 
504
- unit = find_unit(id)
505
- next unless unit
541
+ queues = phase2_queues.values
542
+ loop do
543
+ progressed = false
544
+ queues.each do |queue|
545
+ next if queue.empty?
506
546
 
507
- phase2_scanned += 1
508
-
509
- if fields.include?('source_code') && unit['source_code'] && pattern.match?(unit['source_code'])
510
- results << { identifier: id, type: type_name, match_field: 'source_code' }
511
- elsif fields.include?('metadata') && unit['metadata'] && pattern.match?(unit['metadata'].to_json)
512
- results << { identifier: id, type: type_name, match_field: 'metadata' }
513
- end
547
+ if phase2_scanned >= max_scan
548
+ results.stop('scan_budget')
549
+ throw :search_done
514
550
  end
515
- break unless progressed
551
+ type_name, id = queue.shift
552
+ progressed = true
553
+ phase2_scanned += 1
554
+ unit = scope ? scope.metadata_store.find(StorageIdentity.key(id, type_name)) : load_search_unit(type_name, id)
555
+ field = if fields.include?('source_code') && unit['source_code'] && pattern.match?(unit['source_code'])
556
+ 'source_code'
557
+ elsif fields.include?('metadata') && unit['metadata'] && pattern.match?(unit['metadata'].to_json)
558
+ 'metadata'
559
+ end
560
+ next unless field
561
+
562
+ results.add(identifier: id, type: type_name, match_field: field)
563
+ throw :search_done if results.result_limit_reached?
516
564
  end
565
+ break unless progressed
517
566
  end
518
567
  end
519
568
  rescue StandardError => e
520
569
  raise unless regexp_timeout_error?(e)
521
570
 
522
571
  notes << "search aborted: the pattern exceeded the #{SEARCH_PATTERN_TIMEOUT}s per-match limit"
523
- partial = true
572
+ results.stop('regex_timeout')
524
573
  end
525
574
 
526
- response = { results: results.first(limit) }
527
- response[:note] = notes.join('; ') unless notes.empty?
528
- response[:partial] = true if partial
575
+ response = results.finish.response(note: notes.join('; '))
576
+ response[:applied_scope] = scope.summary if scope
529
577
  response
530
578
  end
531
579
 
580
+ # Scope preparation reads the complete pinned unit snapshot. Search's
581
+ # deep-field scan budget still governs matching work after this read.
582
+ def search_scope(packages, source_paths, types)
583
+ return unless Retrieval::Scope.requested?(packages: packages, source_paths: source_paths)
584
+
585
+ metadata = Storage::MetadataStore::InMemory.new
586
+ each_unit { |unit| metadata.store(StorageIdentity.key(unit.fetch('identifier'), unit.fetch('type')), unit) }
587
+ Retrieval::Scope.new(metadata_store: metadata, packages: packages, source_paths: source_paths, types: types)
588
+ end
589
+ private :search_scope
590
+
591
+ def scoped_search_entries(entries, dir, scope)
592
+ entries.flat_map do |entry|
593
+ UNIT_TYPES_BY_DIR.fetch(dir).filter_map do |type|
594
+ key = StorageIdentity.key(entry['identifier'], type)
595
+ entry.merge('scope_type' => type) if scope.include?(key)
596
+ end
597
+ end
598
+ end
599
+ private :scoped_search_entries
600
+
532
601
  # BFS traversal of forward dependencies.
533
602
  #
534
603
  # @param identifier [String] Starting unit identifier
535
604
  # @param depth [Integer] Maximum traversal depth
536
605
  # @param types [Array<String>, nil] Filter to these singular type names
537
606
  # @param via [Array<String>, nil] Filter to these relationship types (e.g. ["link_to", "redirect_to"])
538
- # @return [Hash] { root:, nodes: { id => { type:, depth:, deps: [] } } }
539
- def traverse_dependencies(identifier, depth: 2, types: nil, via: nil)
540
- traverse(identifier, depth: depth, types: types, via: via, direction: :forward)
607
+ # @param max_nodes [Integer] Distinct admitted nodes including root (default 1000; clamped to 1..10_000)
608
+ # @param max_edges [Integer] Edge checks before filters, including reverse via checks
609
+ # (default 10_000; clamped to 1..100_000). Graph loading/cache preparation is not budgeted.
610
+ # @return [Hash] { root:, found:, nodes: { id => { type:, depth:, deps: [] } } };
611
+ # incomplete walks add partial: true, partial_reason (node_budget or edge_budget),
612
+ # and traversal_budget (max_nodes, max_edges, visited_nodes, visited_edges).
613
+ # Already discovered rows remain; empty deps in a partial result need not mean a leaf.
614
+ def traverse_dependencies(identifier, depth: 2, types: nil, via: nil, max_nodes: 1000, max_edges: 10_000, explain: false)
615
+ if explain
616
+ return traverse_with_evidence(identifier, depth: depth, types: types, via: via, direction: :forward,
617
+ max_nodes: max_nodes, max_edges: max_edges)
618
+ end
619
+
620
+ traverse(identifier, depth: depth, types: types, via: via, direction: :forward, max_nodes: max_nodes, max_edges: max_edges)
541
621
  end
542
622
 
543
623
  # BFS traversal of reverse dependencies (dependents).
@@ -546,9 +626,20 @@ module Woods
546
626
  # @param depth [Integer] Maximum traversal depth
547
627
  # @param types [Array<String>, nil] Filter to these singular type names
548
628
  # @param via [Array<String>, nil] Filter to these relationship types (e.g. ["link_to", "redirect_to"])
549
- # @return [Hash] { root:, nodes: { id => { type:, depth:, deps: [] } } }
550
- def traverse_dependents(identifier, depth: 2, types: nil, via: nil)
551
- traverse(identifier, depth: depth, types: types, via: via, direction: :reverse)
629
+ # @param max_nodes [Integer] Distinct admitted nodes including root (default 1000; clamped to 1..10_000)
630
+ # @param max_edges [Integer] Edge checks before filters, including reverse via checks
631
+ # (default 10_000; clamped to 1..100_000). Graph loading/cache preparation is not budgeted.
632
+ # @return [Hash] { root:, found:, nodes: { id => { type:, depth:, deps: [] } } };
633
+ # incomplete walks add partial: true, partial_reason (node_budget or edge_budget),
634
+ # and traversal_budget (max_nodes, max_edges, visited_nodes, visited_edges).
635
+ # Already discovered rows remain; empty deps in a partial result need not mean a leaf.
636
+ def traverse_dependents(identifier, depth: 2, types: nil, via: nil, max_nodes: 1000, max_edges: 10_000, explain: false)
637
+ if explain
638
+ return traverse_with_evidence(identifier, depth: depth, types: types, via: via, direction: :reverse,
639
+ max_nodes: max_nodes, max_edges: max_edges)
640
+ end
641
+
642
+ traverse(identifier, depth: depth, types: types, via: via, direction: :reverse, max_nodes: max_nodes, max_edges: max_edges)
552
643
  end
553
644
 
554
645
  # Search rails_source units by concept keyword.
@@ -578,7 +669,7 @@ module Woods
578
669
  break if results.size >= limit
579
670
 
580
671
  id = entry['identifier']
581
- unit = find_unit(id)
672
+ unit = load_search_unit('rails_source', id)
582
673
  next unless unit
583
674
 
584
675
  metadata_json = unit['metadata']&.to_json
@@ -627,7 +718,7 @@ module Woods
627
718
  entries = read_index(dir)
628
719
  entries.each do |entry|
629
720
  id = entry['identifier']
630
- unit = find_unit(id)
721
+ unit = load_search_unit(DIR_TO_TYPE.fetch(dir), id)
631
722
  next unless unit
632
723
 
633
724
  last_modified = unit.dig('metadata', 'git', 'last_modified')
@@ -1102,15 +1193,80 @@ module Woods
1102
1193
  entries = read_index(dir)
1103
1194
  entries.each do |entry|
1104
1195
  id = entry['identifier']
1105
- base = id.gsub('::', '__').gsub(/[^a-zA-Z0-9_-]/, '_')
1106
- digest = Digest::SHA256.hexdigest(id)[0, 8]
1107
- filename = "#{base}_#{digest}.json"
1108
- map[id] = { type_dir: dir, filename: filename }
1196
+ map[id] = { type_dir: dir, filename: unit_filename(id) }
1109
1197
  end
1110
1198
  end
1111
1199
  map
1112
1200
  end
1113
1201
 
1202
+ # Read the selected type bucket, never the identifier-only lookup map.
1203
+ # Keep the existing per-file signature checks and LRU cache for deep reads.
1204
+ def load_search_unit(type, identifier)
1205
+ data = load_unit(TYPE_TO_DIR.fetch(type), unit_filename(identifier))
1206
+ unless valid_published_unit?(data, TYPE_TO_DIR.fetch(type), identifier)
1207
+ raise IOError, "typed unit identity mismatch: #{type}:#{identifier}"
1208
+ end
1209
+
1210
+ data
1211
+ end
1212
+
1213
+ def unit_filename(identifier)
1214
+ base = identifier.gsub('::', '__').gsub(/[^a-zA-Z0-9_-]/, '_')
1215
+ "#{base}_#{Digest::SHA256.hexdigest(identifier)[0, 8]}.json"
1216
+ end
1217
+
1218
+ def search_index_entries(dir)
1219
+ entries = published_unit_entries(dir)
1220
+ seen = Set.new
1221
+ entries.each do |entry|
1222
+ id = entry.is_a?(Hash) && entry['identifier']
1223
+ unless id.is_a?(String) && !id.empty? && seen.add?(id)
1224
+ raise IOError, "invalid or duplicate unit identifier in #{dir}/_index.json"
1225
+ end
1226
+ end
1227
+ entries
1228
+ end
1229
+
1230
+ def published_unit_entries(dir)
1231
+ index_path = current_payload_dir.join(dir, '_index.json')
1232
+ raise IOError, "symlink unit index: #{dir}" if index_path.symlink?
1233
+
1234
+ expected = manifest.dig('counts', dir)
1235
+ raise IOError, "missing unit index: #{dir}/_index.json" if !index_path.file? && expected.to_i.positive?
1236
+
1237
+ entries = read_index(dir)
1238
+ raise IOError, "invalid unit index: #{dir}" unless entries.is_a?(Array)
1239
+ if expected.is_a?(Integer) && entries.size != expected
1240
+ raise IOError, "unit count mismatch in #{dir}: expected #{expected}, found #{entries.size}"
1241
+ end
1242
+
1243
+ entries
1244
+ end
1245
+
1246
+ def read_published_unit(dir, identifier)
1247
+ base = identifier.gsub('::', '__').gsub(/[^a-zA-Z0-9_-]/, '_')
1248
+ filename = "#{base}_#{Digest::SHA256.hexdigest(identifier)[0, 8]}.json"
1249
+ path = current_payload_dir.join(dir, filename)
1250
+ raise IOError, "symlink unit file: #{dir}/#{filename}" if path.symlink?
1251
+
1252
+ raise IOError, "non-regular unit file: #{dir}/#{filename}" unless path.lstat.file?
1253
+
1254
+ data = File.open(path, File::RDONLY | File::NOFOLLOW | File::NONBLOCK) do |file|
1255
+ raise IOError, "non-regular unit file: #{dir}/#{filename}" unless file.stat.file?
1256
+
1257
+ JSON.parse(file.read)
1258
+ end
1259
+ unless valid_published_unit?(data, dir, identifier)
1260
+ raise IOError, "typed unit identity mismatch: #{dir}/#{filename}"
1261
+ end
1262
+
1263
+ data
1264
+ end
1265
+
1266
+ def valid_published_unit?(data, dir, identifier)
1267
+ data.is_a?(Hash) && data['identifier'] == identifier && UNIT_TYPES_BY_DIR.fetch(dir).include?(data['type'])
1268
+ end
1269
+
1114
1270
  # Read and cache an _index.json file for a type directory.
1115
1271
  def read_index(dir)
1116
1272
  @index_cache ||= {}
@@ -1229,6 +1385,16 @@ module Woods
1229
1385
  end
1230
1386
  end
1231
1387
 
1388
+ # Ownership/type preparation is cached once per retained generation, like
1389
+ # graph_node_types. It never flattens adjacency; the per-query walker charges
1390
+ # every candidate edge, including legacy reverse evidence recovery.
1391
+ def traverse_with_evidence(identifier, **options)
1392
+ with_pinned_generation do
1393
+ @traversal_evidence_index ||= TraversalEvidenceIndex.new(raw_graph_data)
1394
+ TraversalEvidence.new(@traversal_evidence_index).call(identifier, **options)
1395
+ end
1396
+ end
1397
+
1232
1398
  # BFS traversal in either direction.
1233
1399
  #
1234
1400
  # Edges may be stored as bare strings (old format) or as
@@ -1241,70 +1407,89 @@ module Woods
1241
1407
  # @param via [Array<String>, nil] Filter to these relationship types
1242
1408
  # @param direction [:forward, :reverse] Traversal direction
1243
1409
  # @return [Hash]
1244
- def traverse(identifier, depth:, types:, via:, direction:)
1410
+ def traverse(identifier, depth:, types:, via:, direction:, max_nodes:, max_edges:)
1245
1411
  graph_data = raw_graph_data
1246
1412
  nodes_data = graph_data['nodes'] || {}
1247
-
1248
1413
  return { root: identifier, found: false, nodes: {} } unless nodes_data.key?(identifier)
1249
1414
 
1250
- # Normalize edges once per graph load — memoized alongside raw_graph_data
1251
- normalized_edges = normalized_graph_edges
1252
-
1415
+ budget = {
1416
+ max_nodes: max_nodes.to_i.clamp(1, 10_000),
1417
+ max_edges: max_edges.to_i.clamp(1, 100_000),
1418
+ visited_nodes: 1, visited_edges: 0
1419
+ }
1253
1420
  type_set = types&.to_set
1254
1421
  via_set = via&.to_set
1255
1422
  visited = Set.new([identifier])
1256
1423
  queue = [[identifier, 0]]
1257
1424
  result_nodes = {}
1425
+ cursor = 0
1426
+ partial_reason = nil
1258
1427
 
1259
- while queue.any?
1260
- current, current_depth = queue.shift
1261
-
1262
- neighbors = if direction == :forward
1263
- resolve_forward_neighbors(normalized_edges, current, via_set)
1264
- else
1265
- resolve_reverse_neighbors(graph_data, normalized_edges, current, via_set)
1266
- end
1267
-
1268
- # Filter by node type if requested. An identifier naming units of
1269
- # several types matches when any of them does — excluding it because
1270
- # the type that happens to sort first is not the requested one would
1271
- # hide a unit the filter asked for.
1272
- filtered = if type_set
1273
- neighbors.select { |n| graph_node_types[n]&.any? { |t| type_set.include?(t) } }
1274
- else
1275
- neighbors
1276
- end
1277
-
1278
- # At max depth, record the node with empty deps so the renderer
1279
- # doesn't emit an extra level of unexpanded neighbors. The parent
1280
- # node's deps list already shows this node as a child.
1281
- will_expand = current_depth < depth
1428
+ while cursor < queue.size
1429
+ current, current_depth = queue[cursor]
1430
+ cursor += 1
1282
1431
  node_meta = nodes_data[current]
1283
- entry = {
1284
- type: node_meta&.dig('type'),
1285
- depth: current_depth,
1286
- deps: will_expand ? filtered : []
1287
- }
1288
- # Only when the identifier is genuinely ambiguous, so the shape is
1289
- # unchanged for every node in an index with no shared identifiers.
1432
+ entry = { type: node_meta&.dig('type'), depth: current_depth, deps: [] }
1290
1433
  node_types = graph_node_types[current] || []
1291
1434
  entry[:types] = node_types if node_types.size > 1
1292
- # Likewise: a single-database app learns nothing from a column that
1293
- # answers the same thing on every row (B-183).
1294
1435
  entry[:database] = node_meta&.dig('database') if multi_database_graph?
1295
1436
  result_nodes[current] = entry
1437
+ next if partial_reason || current_depth >= depth
1438
+
1439
+ # Keep every discovered row, even after expansion stops. No dangling
1440
+ # deps are introduced by the budget, and paging stays independent.
1441
+ partial_reason = catch(:traversal_budget) do
1442
+ each_traversal_neighbor(graph_data, current, direction, via_set, budget) do |neighbor|
1443
+ next if type_set && !graph_node_types[neighbor]&.any? { |type| type_set.include?(type) }
1444
+
1445
+ unless visited.include?(neighbor)
1446
+ throw :traversal_budget, 'node_budget' if visited.size >= budget[:max_nodes]
1447
+
1448
+ visited.add(neighbor)
1449
+ budget[:visited_nodes] += 1
1450
+ queue << [neighbor, current_depth + 1]
1451
+ end
1452
+ entry[:deps] << neighbor
1453
+ end
1454
+ nil
1455
+ end
1456
+ end
1296
1457
 
1297
- next unless will_expand
1458
+ result = { root: identifier, found: true, nodes: result_nodes }
1459
+ result.merge!(partial: true, partial_reason: partial_reason, traversal_budget: budget) if partial_reason
1460
+ result
1461
+ end
1298
1462
 
1299
- filtered.each do |neighbor|
1300
- unless visited.include?(neighbor)
1301
- visited.add(neighbor)
1302
- queue.push([neighbor, current_depth + 1])
1463
+ # Walk stored adjacency order lazily. In reverse traversal a via filter
1464
+ # must inspect forward edges too; those checks consume the same budget.
1465
+ # Whole-generation JSON loading is deliberately outside the walk budget.
1466
+ def each_traversal_neighbor(graph_data, identifier, direction, via_set, budget)
1467
+ edges = normalized_graph_edges
1468
+ if direction == :forward
1469
+ (edges[identifier] || []).each do |edge|
1470
+ consume_traversal_edge(budget)
1471
+ edge = { 'target' => edge } unless edge.is_a?(Hash)
1472
+ yield edge['target'] unless via_set && !via_set.include?(edge['via'])
1473
+ end
1474
+ else
1475
+ ((graph_data['reverse'] || {})[identifier] || []).each do |dependent|
1476
+ consume_traversal_edge(budget)
1477
+ if via_set
1478
+ matches = (edges[dependent] || []).any? do |edge|
1479
+ consume_traversal_edge(budget)
1480
+ edge.is_a?(Hash) && edge['target'] == identifier && via_set.include?(edge['via'])
1481
+ end
1482
+ next unless matches
1303
1483
  end
1484
+ yield dependent
1304
1485
  end
1305
1486
  end
1487
+ end
1488
+
1489
+ def consume_traversal_edge(budget)
1490
+ throw :traversal_budget, 'edge_budget' if budget[:visited_edges] >= budget[:max_edges]
1306
1491
 
1307
- { root: identifier, found: true, nodes: result_nodes }
1492
+ budget[:visited_edges] += 1
1308
1493
  end
1309
1494
 
1310
1495
  # Normalize all edge arrays once, converting bare strings to hashes.
@@ -1321,28 +1506,6 @@ module Woods
1321
1506
  entries.map { |e| e.is_a?(Hash) ? e : { 'target' => e } }
1322
1507
  end
1323
1508
  end
1324
-
1325
- # Extract forward neighbor identifiers, optionally filtered by via type.
1326
- # Expects pre-normalized edges (all entries are hashes).
1327
- def resolve_forward_neighbors(normalized_edges, identifier, via_set)
1328
- edges = normalized_edges[identifier] || []
1329
- edges = edges.select { |e| via_set.include?(e['via']) } if via_set
1330
- edges.map { |e| e['target'] }
1331
- end
1332
-
1333
- # Extract reverse neighbor identifiers, optionally filtered by via type.
1334
- # Reverse edges are stored as bare identifier arrays. When via filtering
1335
- # is requested, checks each dependent's pre-normalized forward edges to
1336
- # find those pointing at +identifier+ with a matching via type.
1337
- def resolve_reverse_neighbors(graph_data, normalized_edges, identifier, via_set)
1338
- dependents = (graph_data['reverse'] || {})[identifier] || []
1339
- return dependents unless via_set
1340
-
1341
- dependents.select do |dep|
1342
- forward = normalized_edges[dep] || []
1343
- forward.any? { |e| e['target'] == identifier && via_set.include?(e['via']) }
1344
- end
1345
- end
1346
1509
  end
1347
1510
  end
1348
1511
  end