woods 2.0.0.beta2 → 2.0.0.beta4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +339 -1
- data/CONTRIBUTING.md +188 -12
- data/README.md +93 -174
- data/SECURITY.md +9 -6
- data/docs/AGENT_GUIDE.md +109 -8
- data/docs/AGENT_SETUP.md +98 -7
- data/docs/BACKEND_MATRIX.md +25 -0
- data/docs/CLIENT_HOOKS.md +111 -0
- data/docs/CONFIGURATION_REFERENCE.md +267 -16
- data/docs/CONSOLE_MCP_SETUP.md +80 -7
- data/docs/DOCKER_SETUP.md +22 -3
- data/docs/EVALUATION.md +464 -1
- data/docs/EXTRACTOR_REFERENCE.md +45 -6
- data/docs/FAQ.md +11 -12
- data/docs/GETTING_STARTED.md +17 -5
- data/docs/INCREMENTAL_EXTRACTION.md +147 -7
- data/docs/INDEX_LAYOUT.md +382 -0
- data/docs/INTERNALS.md +7 -2
- data/docs/MCP_SERVERS.md +276 -5
- data/docs/MCP_TOOL_COOKBOOK.md +37 -22
- data/docs/MCP_WORKTREE_SETUP.md +43 -83
- data/docs/NOTION_INTEGRATION.md +13 -0
- data/docs/OBSIDIAN_INTEGRATION.md +57 -9
- data/docs/PUBLISHED_INDEX.md +72 -0
- data/docs/README.md +7 -0
- data/docs/RETRIEVAL_GUIDE.md +273 -12
- data/docs/RUNTIME_TRACING.md +71 -0
- data/docs/SOURCE_FRESHNESS.md +143 -0
- data/docs/TROUBLESHOOTING.md +129 -18
- data/docs/UNBLOCKED_INTEGRATION.md +25 -0
- data/docs/UPGRADING_TO_2.md +48 -22
- data/docs/WATCH_DAEMON.md +277 -67
- data/exe/woods-agent-config +6 -0
- data/exe/woods-extract +5 -0
- data/exe/woods-hook-context +6 -0
- data/exe/woods-mcp-start +14 -9
- data/lib/generators/woods/pgvector_generator.rb +8 -2
- data/lib/generators/woods/templates/woods.rb.tt +1 -3
- data/lib/tasks/woods.rake +47 -397
- data/lib/woods/agent_configuration/applier.rb +135 -0
- data/lib/woods/agent_configuration/cli.rb +101 -0
- data/lib/woods/agent_configuration/cli_options.rb +29 -0
- data/lib/woods/agent_configuration/document.rb +105 -0
- data/lib/woods/agent_configuration/error.rb +7 -0
- data/lib/woods/agent_configuration/launcher.rb +75 -0
- data/lib/woods/agent_configuration/layout.rb +72 -0
- data/lib/woods/agent_configuration/managed_section.rb +62 -0
- data/lib/woods/agent_configuration/plan.rb +98 -0
- data/lib/woods/agent_configuration/plan_diff.rb +38 -0
- data/lib/woods/agent_configuration/planned_files.rb +61 -0
- data/lib/woods/agent_configuration/planner.rb +63 -0
- data/lib/woods/agent_configuration/planner_validation.rb +77 -0
- data/lib/woods/agent_configuration/preflight.rb +100 -0
- data/lib/woods/agent_configuration/recovery.rb +49 -0
- data/lib/woods/ast/node.rb +2 -0
- data/lib/woods/ast/parser.rb +38 -5
- data/lib/woods/builder.rb +21 -5
- data/lib/woods/cache/cache_middleware.rb +28 -7
- data/lib/woods/cache/cache_store.rb +4 -5
- data/lib/woods/change_set.rb +5 -4
- data/lib/woods/console/credential_index.rb +20 -2
- data/lib/woods/console/credential_scanner.rb +18 -17
- data/lib/woods/console/credential_scanner_registry.rb +36 -0
- data/lib/woods/console/dispatch_pipeline.rb +7 -0
- data/lib/woods/console/embedded_executor.rb +32 -10
- data/lib/woods/console/encrypted_credential_snapshot.rb +16 -0
- data/lib/woods/console/rack_middleware.rb +22 -13
- data/lib/woods/console/server.rb +18 -16
- data/lib/woods/console/sql_noise_stripper.rb +9 -7
- data/lib/woods/console/sql_table_scanner.rb +47 -7
- data/lib/woods/console/sql_validator.rb +49 -9
- data/lib/woods/console/sqlite_read_guard.rb +46 -0
- data/lib/woods/coordination/pipeline_lock.rb +3 -2
- data/lib/woods/dependency_graph.rb +65 -13
- data/lib/woods/embedding/corpus.rb +94 -0
- data/lib/woods/embedding/indexer.rb +114 -60
- data/lib/woods/embedding/openai.rb +17 -6
- data/lib/woods/evaluation/ablation_executor.rb +6 -1
- data/lib/woods/evaluation/ablation_timed_executor.rb +22 -4
- data/lib/woods/export/typed_reader.rb +56 -0
- data/lib/woods/extractor.rb +277 -149
- data/lib/woods/extractors/action_cable_extractor.rb +3 -1
- data/lib/woods/extractors/behavioral_profile.rb +9 -7
- data/lib/woods/extractors/caching_extractor.rb +3 -1
- data/lib/woods/extractors/concern_extractor.rb +64 -6
- data/lib/woods/extractors/configuration_extractor.rb +7 -3
- data/lib/woods/extractors/controller_extractor.rb +13 -4
- data/lib/woods/extractors/database_view_extractor.rb +3 -1
- data/lib/woods/extractors/declared_parent.rb +55 -0
- data/lib/woods/extractors/decorator_extractor.rb +3 -1
- data/lib/woods/extractors/engine_extractor.rb +3 -1
- data/lib/woods/extractors/event_extractor.rb +4 -2
- data/lib/woods/extractors/factory_extractor.rb +3 -1
- data/lib/woods/extractors/graphql_extractor.rb +10 -13
- data/lib/woods/extractors/i18n_extractor.rb +3 -1
- data/lib/woods/extractors/job_extractor.rb +6 -19
- data/lib/woods/extractors/lib_extractor.rb +13 -9
- data/lib/woods/extractors/mailer_extractor.rb +26 -15
- data/lib/woods/extractors/manager_extractor.rb +3 -1
- data/lib/woods/extractors/method_parameters.rb +53 -0
- data/lib/woods/extractors/middleware_argument.rb +65 -0
- data/lib/woods/extractors/middleware_extractor.rb +9 -3
- data/lib/woods/extractors/migration_extractor.rb +3 -1
- data/lib/woods/extractors/model_extractor.rb +26 -34
- data/lib/woods/extractors/package_extractor.rb +24 -4
- data/lib/woods/extractors/phlex_extractor.rb +3 -1
- data/lib/woods/extractors/policy_extractor.rb +3 -1
- data/lib/woods/extractors/poro_extractor.rb +13 -9
- data/lib/woods/extractors/pundit_extractor.rb +3 -1
- data/lib/woods/extractors/rails_source_extractor.rb +4 -2
- data/lib/woods/extractors/rake_task_extractor.rb +4 -2
- data/lib/woods/extractors/route_extractor.rb +3 -1
- data/lib/woods/extractors/route_helper_resolver.rb +10 -33
- data/lib/woods/extractors/scheduled_job_extractor.rb +41 -15
- data/lib/woods/extractors/serializer_extractor.rb +4 -2
- data/lib/woods/extractors/service_extractor.rb +3 -1
- data/lib/woods/extractors/shared_dependency_scanner.rb +2 -2
- data/lib/woods/extractors/shared_utility_methods.rb +48 -19
- data/lib/woods/extractors/source_nesting.rb +1 -1
- data/lib/woods/extractors/state_machine_extractor.rb +3 -1
- data/lib/woods/extractors/test_mapping_extractor.rb +3 -1
- data/lib/woods/extractors/validator_extractor.rb +3 -1
- data/lib/woods/extractors/view_component_extractor.rb +3 -1
- data/lib/woods/extractors/view_template_extractor.rb +3 -1
- data/lib/woods/gem_mapper.rb +2 -0
- data/lib/woods/git_history.rb +116 -0
- data/lib/woods/graph_analyzer.rb +35 -6
- data/lib/woods/hooks/context_cli.rb +54 -0
- data/lib/woods/hooks/context_event.rb +88 -0
- data/lib/woods/hooks/context_hint.rb +73 -0
- data/lib/woods/hooks/context_impact.rb +77 -0
- data/lib/woods/hooks/context_output.rb +47 -0
- data/lib/woods/hooks/context_state.rb +102 -0
- data/lib/woods/hooks/refresh.rb +79 -0
- data/lib/woods/hooks/rule_projection.rb +78 -0
- data/lib/woods/input_rules.rb +19 -0
- data/lib/woods/mcp/bearer_auth.rb +22 -13
- data/lib/woods/mcp/bootstrapper.rb +79 -4
- data/lib/woods/mcp/config_resolver.rb +2 -1
- data/lib/woods/mcp/index_reader.rb +334 -162
- data/lib/woods/mcp/initialization_guidance.rb +27 -0
- data/lib/woods/mcp/origin_guard.rb +17 -9
- data/lib/woods/mcp/published_lexical_retriever.rb +115 -0
- data/lib/woods/mcp/renderers/markdown_renderer.rb +22 -9
- data/lib/woods/mcp/renderers/plain_renderer.rb +18 -8
- data/lib/woods/mcp/search_results.rb +74 -0
- data/lib/woods/mcp/server.rb +178 -63
- data/lib/woods/mcp/tool_contract.rb +3 -1
- data/lib/woods/mcp/tool_response_renderer.rb +41 -0
- data/lib/woods/mcp/traversal_evidence.rb +113 -0
- data/lib/woods/mcp/traversal_evidence_index.rb +100 -0
- data/lib/woods/mcp/traversal_evidence_page.rb +41 -0
- data/lib/woods/mcp/traversal_evidence_text.rb +52 -0
- data/lib/woods/mcp/traversal_response.rb +22 -0
- data/lib/woods/notion/exporter.rb +56 -17
- data/lib/woods/obsidian/destination_plan.rb +98 -0
- data/lib/woods/obsidian/name_mapper.rb +19 -3
- data/lib/woods/obsidian/note_builder.rb +19 -10
- data/lib/woods/obsidian/vault_exporter.rb +88 -32
- data/lib/woods/operator/pipeline_guard.rb +18 -13
- data/lib/woods/path_dispatcher.rb +13 -6
- data/lib/woods/payload_store.rb +27 -26
- data/lib/woods/published_index/typed_unit_reader.rb +40 -3
- data/lib/woods/published_index.rb +2 -2
- data/lib/woods/railtie.rb +3 -3
- data/lib/woods/railtie_support.rb +12 -12
- data/lib/woods/rake_helpers.rb +382 -0
- data/lib/woods/resilience/graph_invariant_validator/membership_checks.rb +71 -0
- data/lib/woods/resilience/graph_invariant_validator/node_checks.rb +61 -0
- data/lib/woods/resilience/graph_invariant_validator/reverse_relationship_checks.rb +46 -0
- data/lib/woods/resilience/graph_invariant_validator.rb +119 -0
- data/lib/woods/resilience/index_validator/graph_checks.rb +80 -0
- data/lib/woods/resilience/index_validator.rb +112 -23
- data/lib/woods/retrieval/context_assembler.rb +50 -15
- data/lib/woods/retrieval/lexical_assembler.rb +84 -0
- data/lib/woods/retrieval/lexical_index.rb +120 -0
- data/lib/woods/retrieval/ranker.rb +4 -2
- data/lib/woods/retrieval/scope.rb +108 -0
- data/lib/woods/retrieval/scoped_graph_store.rb +32 -0
- data/lib/woods/retrieval/scoped_vector_store.rb +55 -0
- data/lib/woods/retrieval/search_executor.rb +86 -27
- data/lib/woods/retrieval/source_evidence.rb +200 -0
- data/lib/woods/retriever.rb +98 -22
- data/lib/woods/ruby_analyzer/trace_enricher.rb +77 -38
- data/lib/woods/session_tracer/file_store.rb +6 -1
- data/lib/woods/session_tracer/middleware.rb +10 -12
- data/lib/woods/session_tracer/redis_store.rb +22 -6
- data/lib/woods/session_tracer/session_flow_assembler.rb +23 -17
- data/lib/woods/session_tracer/solid_cache_coordination.rb +6 -4
- data/lib/woods/session_tracer/unit_resolver.rb +63 -0
- data/lib/woods/source_inputs/consumer_errors.rb +31 -0
- data/lib/woods/source_inputs/handoff.rb +102 -0
- data/lib/woods/source_inputs/launcher.rb +157 -0
- data/lib/woods/source_inputs/manifest.rb +124 -0
- data/lib/woods/source_inputs/private_key.rb +55 -0
- data/lib/woods/source_inputs/scanner.rb +171 -0
- data/lib/woods/source_inputs/scopes.rb +71 -0
- data/lib/woods/source_inputs/session.rb +214 -0
- data/lib/woods/source_inputs/status.rb +84 -0
- data/lib/woods/source_inputs/verifier.rb +107 -0
- data/lib/woods/storage/metadata_store.rb +25 -25
- data/lib/woods/storage/pgvector.rb +35 -10
- data/lib/woods/storage/qdrant.rb +17 -7
- data/lib/woods/storage/vector_store.rb +18 -6
- data/lib/woods/tasks.rb +3 -2
- data/lib/woods/temporal/json_snapshot_store.rb +58 -9
- data/lib/woods/unblocked/exporter.rb +59 -70
- data/lib/woods/version.rb +1 -1
- data/lib/woods/watch/boot_snapshot.rb +52 -0
- data/lib/woods/watch/daemon.rb +154 -32
- data/lib/woods/watch/listen_watcher.rb +4 -0
- data/lib/woods/watch/polling_watcher.rb +5 -1
- data/lib/woods/watch/status.rb +20 -15
- data/lib/woods/watch/tree_scan.rb +21 -13
- data/lib/woods/watch/watcher.rb +4 -1
- data/lib/woods.rb +50 -11
- data/plugin/.claude-plugin/plugin.json +1 -1
- data/plugin/hooks/adapters/normalize.jq +15 -0
- data/plugin/hooks/adapters/normalize.rb +63 -0
- data/plugin/hooks/hooks.json +20 -0
- data/plugin/hooks/woods-context.sh +50 -0
- data/plugin/hooks/woods-input-rules.sh +159 -0
- data/plugin/hooks/woods-opencode.mjs +65 -0
- data/plugin/hooks/woods-post-edit.sh +2 -225
- data/plugin/hooks/woods-refresh.sh +260 -0
- data/plugin/hooks/woods-session-start.sh +47 -55
- data/plugin/skills/woods-agent-enable/SKILL.md +19 -0
- data/plugin/skills/woods-diagnose/SKILL.md +319 -1
- data/plugin/skills/woods-investigate/SKILL.md +145 -0
- data/plugin/skills/woods-mcp-config/SKILL.md +90 -2
- data/plugin/skills/woods-setup/SKILL.md +110 -6
- metadata +87 -5
|
@@ -0,0 +1,120 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require 'json'
|
|
4
|
+
require_relative 'query_classifier'
|
|
5
|
+
require_relative 'search_executor'
|
|
6
|
+
|
|
7
|
+
module Woods
|
|
8
|
+
module Retrieval
|
|
9
|
+
# Immutable field-aware lexical snapshot. Only published unit values enter
|
|
10
|
+
# the vocabulary; bookkeeping and arbitrary JSON keys cannot become hits.
|
|
11
|
+
class LexicalIndex
|
|
12
|
+
FIELD_WEIGHTS = { 'identifier' => 4.0, 'file_path' => 2.0, 'source_code' => 1.0, 'runtime' => 2.0 }.freeze
|
|
13
|
+
RUNTIME_FIELDS = %w[callbacks associations validations scopes concerns included_modules
|
|
14
|
+
methods instance_methods class_methods actions routes columns table_name
|
|
15
|
+
description purpose dependencies].freeze
|
|
16
|
+
DEFAULT_LIMIT = 20
|
|
17
|
+
K1 = 1.2
|
|
18
|
+
B = 0.75
|
|
19
|
+
Document = Struct.new(:key, :unit, :fields, keyword_init: true)
|
|
20
|
+
private_constant :Document
|
|
21
|
+
|
|
22
|
+
def initialize(metadata_store:)
|
|
23
|
+
@documents = metadata_store.all_identifiers.sort.map do |key|
|
|
24
|
+
unit = metadata_store.find(key)
|
|
25
|
+
raise ArgumentError, "missing unit metadata for #{key.inspect}" unless unit.is_a?(Hash)
|
|
26
|
+
|
|
27
|
+
unit = JSON.parse(JSON.generate(unit))
|
|
28
|
+
fields = field_values(unit).transform_values { |value| tokenize(value).tally.freeze }.freeze
|
|
29
|
+
Document.new(key: key.freeze, unit: deep_freeze(unit), fields: fields).freeze
|
|
30
|
+
end.compact.freeze
|
|
31
|
+
@averages = FIELD_WEIGHTS.to_h do |field, _|
|
|
32
|
+
lengths = @documents.map { |doc| doc.fields.fetch(field).values.sum }
|
|
33
|
+
[field, lengths.empty? ? 1.0 : [lengths.sum.fdiv(lengths.size), 1.0].max]
|
|
34
|
+
end.freeze
|
|
35
|
+
@frequencies = Hash.new(0)
|
|
36
|
+
@documents.each { |doc| doc.fields.values.flat_map(&:keys).uniq.each { |term| @frequencies[term] += 1 } }
|
|
37
|
+
@frequencies.freeze
|
|
38
|
+
end
|
|
39
|
+
|
|
40
|
+
def execute(query:, limit: DEFAULT_LIMIT, type_filter: nil, exclude_types: nil)
|
|
41
|
+
terms = tokenize(query).uniq
|
|
42
|
+
candidates = @documents.filter_map do |doc|
|
|
43
|
+
next unless eligible?(doc.unit, type_filter, exclude_types)
|
|
44
|
+
|
|
45
|
+
score, fields = score_document(doc, terms)
|
|
46
|
+
exact = doc.unit['identifier'].to_s.casecmp?(query.strip)
|
|
47
|
+
score += 1.0 if exact
|
|
48
|
+
fields << 'identifier:exact' if exact
|
|
49
|
+
next unless score.positive?
|
|
50
|
+
|
|
51
|
+
SearchExecutor::Candidate.new(identifier: doc.key, score: score, source: :lexical,
|
|
52
|
+
metadata: doc.unit, matched_fields: fields.sort)
|
|
53
|
+
end
|
|
54
|
+
candidates.sort_by! do |candidate|
|
|
55
|
+
[candidate.matched_fields.include?('identifier:exact') ? 0 : 1, -candidate.score, candidate.identifier]
|
|
56
|
+
end
|
|
57
|
+
SearchExecutor::ExecutionResult.new(candidates: candidates.first(limit), strategy: :lexical, query: query)
|
|
58
|
+
end
|
|
59
|
+
|
|
60
|
+
private
|
|
61
|
+
|
|
62
|
+
def tokenize(value)
|
|
63
|
+
value.to_s.gsub(/(\p{Ll}|\d)(\p{Lu})/u, '\1 \2')
|
|
64
|
+
.gsub(/(\p{Lu})(\p{Lu}\p{Ll})/u, '\1 \2').downcase
|
|
65
|
+
.scan(/[\p{L}\p{N}]+/u).reject { |term| QueryClassifier::STOP_WORDS.include?(term) }
|
|
66
|
+
end
|
|
67
|
+
|
|
68
|
+
def field_values(unit)
|
|
69
|
+
metadata = unit['metadata'].is_a?(Hash) ? unit['metadata'] : {}
|
|
70
|
+
runtime = RUNTIME_FIELDS.filter_map { |key| metadata[key] || unit[key] }
|
|
71
|
+
{ 'identifier' => unit['identifier'], 'file_path' => unit['file_path'],
|
|
72
|
+
'source_code' => unit['source_code'], 'runtime' => values_text(runtime) }
|
|
73
|
+
end
|
|
74
|
+
|
|
75
|
+
def values_text(value)
|
|
76
|
+
case value
|
|
77
|
+
when Hash then value.values.map { |child| values_text(child) }.join(' ')
|
|
78
|
+
when Array then value.map { |child| values_text(child) }.join(' ')
|
|
79
|
+
when String, Symbol, Numeric then value.to_s
|
|
80
|
+
else ''
|
|
81
|
+
end
|
|
82
|
+
end
|
|
83
|
+
|
|
84
|
+
def eligible?(unit, allowed, excluded)
|
|
85
|
+
return allowed.map(&:to_s).include?(unit['type']) if allowed && !allowed.empty?
|
|
86
|
+
|
|
87
|
+
!Array(excluded).map(&:to_s).include?(unit['type'])
|
|
88
|
+
end
|
|
89
|
+
|
|
90
|
+
def score_document(doc, terms)
|
|
91
|
+
fields = []
|
|
92
|
+
score = FIELD_WEIGHTS.sum do |field, weight|
|
|
93
|
+
counts = doc.fields.fetch(field)
|
|
94
|
+
normalization = K1 * (1 - B + (B * counts.values.sum / @averages.fetch(field)))
|
|
95
|
+
terms.sum do |term|
|
|
96
|
+
frequency = counts.fetch(term, 0)
|
|
97
|
+
next 0.0 if frequency.zero?
|
|
98
|
+
|
|
99
|
+
fields << "#{field}:#{term}"
|
|
100
|
+
document_frequency = @frequencies.fetch(term)
|
|
101
|
+
idf = Math.log(1 + ((@documents.size - document_frequency + 0.5) / (document_frequency + 0.5)))
|
|
102
|
+
weight * idf * frequency * (K1 + 1) / (frequency + normalization)
|
|
103
|
+
end
|
|
104
|
+
end
|
|
105
|
+
[score, fields]
|
|
106
|
+
end
|
|
107
|
+
|
|
108
|
+
def deep_freeze(value)
|
|
109
|
+
case value
|
|
110
|
+
when Hash then value.each do |key, child|
|
|
111
|
+
key.freeze
|
|
112
|
+
deep_freeze(child)
|
|
113
|
+
end
|
|
114
|
+
when Array then value.each { |child| deep_freeze(child) }
|
|
115
|
+
end
|
|
116
|
+
value.freeze
|
|
117
|
+
end
|
|
118
|
+
end
|
|
119
|
+
end
|
|
120
|
+
end
|
|
@@ -378,7 +378,9 @@ module Woods
|
|
|
378
378
|
|
|
379
379
|
# Lazily-computed rank-percentile map derived from the graph store's PageRank.
|
|
380
380
|
#
|
|
381
|
-
# Top-ranked identifier gets 1.0, bottom-ranked gets 1/n.
|
|
381
|
+
# Top-ranked identifier gets 1.0, bottom-ranked gets 1/n. Equal PageRank scores
|
|
382
|
+
# are ordered lexically by identifier so their ordinal percentiles do not
|
|
383
|
+
# depend on graph insertion order or Ruby's unstable sort. Identifiers absent
|
|
382
384
|
# from PageRank (new units, ephemeral candidates) return nil and fall back
|
|
383
385
|
# to the bucketed importance signal.
|
|
384
386
|
#
|
|
@@ -402,7 +404,7 @@ module Woods
|
|
|
402
404
|
scores = @graph_store.pagerank
|
|
403
405
|
return {} if scores.nil? || scores.empty?
|
|
404
406
|
|
|
405
|
-
ranked = scores.sort_by { |
|
|
407
|
+
ranked = scores.sort_by { |identifier, score| [-score, identifier] }
|
|
406
408
|
total = ranked.size.to_f
|
|
407
409
|
ranked.each_with_index.to_h do |(identifier, _score), rank|
|
|
408
410
|
[identifier, 1.0 - (rank / total)]
|
|
@@ -0,0 +1,108 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require 'json'
|
|
4
|
+
require 'set'
|
|
5
|
+
require_relative '../storage_identity'
|
|
6
|
+
require_relative '../storage/metadata_store'
|
|
7
|
+
|
|
8
|
+
module Woods
|
|
9
|
+
module Retrieval
|
|
10
|
+
# Resolves explicit scope from published metadata, never host filesystem
|
|
11
|
+
# paths. Ownership is exact; path prefixes match complete directory segments.
|
|
12
|
+
# Every eligible unit is selected before any search strategy applies a limit.
|
|
13
|
+
class Scope
|
|
14
|
+
class InvalidScopeError < ArgumentError; end
|
|
15
|
+
|
|
16
|
+
attr_reader :packages, :source_paths, :keys, :metadata_store
|
|
17
|
+
|
|
18
|
+
def self.requested?(packages: nil, source_paths: nil)
|
|
19
|
+
[packages, source_paths].any? { |list| !list.nil? && list != [] }
|
|
20
|
+
end
|
|
21
|
+
|
|
22
|
+
def initialize(metadata_store:, packages: nil, source_paths: nil, types: nil, exclude_types: nil)
|
|
23
|
+
@packages = normalize_list(packages, 'packages').freeze
|
|
24
|
+
@source_paths = normalize_list(source_paths, 'source_paths').map do |path|
|
|
25
|
+
normalize_path(path)
|
|
26
|
+
end.uniq.sort.freeze
|
|
27
|
+
records = metadata_store.all_identifiers.sort.to_h do |key|
|
|
28
|
+
record = metadata_store.find(key)
|
|
29
|
+
raise InvalidScopeError, "missing metadata for scoped unit #{key.inspect}" unless record.is_a?(Hash)
|
|
30
|
+
|
|
31
|
+
[key, JSON.parse(JSON.generate(record))]
|
|
32
|
+
end
|
|
33
|
+
validate_packages!(records.values)
|
|
34
|
+
@metadata_store = Storage::MetadataStore::InMemory.new
|
|
35
|
+
records.each do |key, record|
|
|
36
|
+
next unless eligible?(record, types, exclude_types)
|
|
37
|
+
|
|
38
|
+
@metadata_store.store(key, record)
|
|
39
|
+
end
|
|
40
|
+
@keys = @metadata_store.all_identifiers.sort.freeze
|
|
41
|
+
@key_set = @keys.to_set.freeze
|
|
42
|
+
end
|
|
43
|
+
|
|
44
|
+
def include?(key)
|
|
45
|
+
@key_set.include?(key.to_s.sub(/#chunk_\d+\z/, ''))
|
|
46
|
+
end
|
|
47
|
+
|
|
48
|
+
def summary
|
|
49
|
+
{ packages: packages, source_paths: source_paths, eligible_units: keys.size }
|
|
50
|
+
end
|
|
51
|
+
|
|
52
|
+
private
|
|
53
|
+
|
|
54
|
+
def normalize_list(list, name)
|
|
55
|
+
return [] if list.nil?
|
|
56
|
+
unless list.is_a?(Array) && list.all? { |value| value.is_a?(String) && !value.empty? && !value.include?("\0") }
|
|
57
|
+
raise InvalidScopeError, "#{name} must be an array of nonempty strings"
|
|
58
|
+
end
|
|
59
|
+
|
|
60
|
+
list.uniq.sort.map { |value| value.dup.freeze }
|
|
61
|
+
end
|
|
62
|
+
|
|
63
|
+
def normalize_path(path)
|
|
64
|
+
if path.start_with?('/') || path.match?(/\A[A-Za-z]:/) || path.include?('\\')
|
|
65
|
+
raise InvalidScopeError, "source path must be application-relative: #{path.inspect}"
|
|
66
|
+
end
|
|
67
|
+
|
|
68
|
+
segments = []
|
|
69
|
+
path.split('/').each do |part|
|
|
70
|
+
next if part.empty? || part == '.'
|
|
71
|
+
|
|
72
|
+
if part == '..'
|
|
73
|
+
raise InvalidScopeError, "source path escapes application root: #{path.inspect}" if segments.empty?
|
|
74
|
+
|
|
75
|
+
segments.pop
|
|
76
|
+
else
|
|
77
|
+
segments << part
|
|
78
|
+
end
|
|
79
|
+
end
|
|
80
|
+
(segments.empty? ? '.' : segments.join('/')).freeze
|
|
81
|
+
end
|
|
82
|
+
|
|
83
|
+
def validate_packages!(records)
|
|
84
|
+
names = records.filter_map { |unit| unit['identifier'] if unit['type'] == 'package' }
|
|
85
|
+
names.concat(records.filter_map { |unit| unit.dig('metadata', 'package') })
|
|
86
|
+
unknown = packages - names
|
|
87
|
+
raise InvalidScopeError, "unknown package scope: #{unknown.join(', ')}" unless unknown.empty?
|
|
88
|
+
end
|
|
89
|
+
|
|
90
|
+
def eligible?(unit, types, excluded)
|
|
91
|
+
return false unless packages.empty? || packages.include?(unit.dig('metadata', 'package'))
|
|
92
|
+
return false unless source_paths.empty? || path_match?(unit['file_path'])
|
|
93
|
+
return Array(types).map(&:to_s).include?(unit['type']) if types && !types.empty?
|
|
94
|
+
|
|
95
|
+
!Array(excluded).map(&:to_s).include?(unit['type'])
|
|
96
|
+
end
|
|
97
|
+
|
|
98
|
+
def path_match?(path)
|
|
99
|
+
return false unless path.is_a?(String) && !path.empty? && !path.include?("\0")
|
|
100
|
+
|
|
101
|
+
normalized = normalize_path(path)
|
|
102
|
+
source_paths.any? { |prefix| prefix == '.' || normalized == prefix || normalized.start_with?("#{prefix}/") }
|
|
103
|
+
rescue InvalidScopeError
|
|
104
|
+
false
|
|
105
|
+
end
|
|
106
|
+
end
|
|
107
|
+
end
|
|
108
|
+
end
|
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Woods
|
|
4
|
+
module Retrieval
|
|
5
|
+
# Expansion remains inside caller eligibility. Graphs retain their existing
|
|
6
|
+
# bare-identifier ambiguity: the executor resolves eligible typed variants.
|
|
7
|
+
class ScopedGraphStore
|
|
8
|
+
def initialize(store:, scope:)
|
|
9
|
+
@store = store
|
|
10
|
+
@identifiers = scope.keys.to_set { |key| StorageIdentity.identifier(key) }
|
|
11
|
+
end
|
|
12
|
+
|
|
13
|
+
def dependencies_of(identifier)
|
|
14
|
+
return [] unless @store && @identifiers.include?(identifier)
|
|
15
|
+
|
|
16
|
+
@store.dependencies_of(identifier).select { |target| @identifiers.include?(target) }
|
|
17
|
+
end
|
|
18
|
+
|
|
19
|
+
def dependents_of(identifier)
|
|
20
|
+
return [] unless @store && @identifiers.include?(identifier)
|
|
21
|
+
|
|
22
|
+
@store.dependents_of(identifier).select { |target| @identifiers.include?(target) }
|
|
23
|
+
end
|
|
24
|
+
|
|
25
|
+
def pagerank
|
|
26
|
+
return {} unless @store
|
|
27
|
+
|
|
28
|
+
@store.pagerank.slice(*@identifiers)
|
|
29
|
+
end
|
|
30
|
+
end
|
|
31
|
+
end
|
|
32
|
+
end
|
|
@@ -0,0 +1,55 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Woods
|
|
4
|
+
module Retrieval
|
|
5
|
+
# Complete, bounded native searches across eligible raw vector IDs. Raw IDs
|
|
6
|
+
# preserve typed/chunk identity without requiring a new vector payload or an
|
|
7
|
+
# embedding migration. One outer executor embeds the query once.
|
|
8
|
+
class ScopedVectorStore
|
|
9
|
+
BATCH_SIZE = 100
|
|
10
|
+
|
|
11
|
+
def initialize(store:, scope:)
|
|
12
|
+
@store = store
|
|
13
|
+
@scope = scope
|
|
14
|
+
end
|
|
15
|
+
|
|
16
|
+
def search(query_vector, limit: 10, filters: {})
|
|
17
|
+
unless @store.respond_to?(:supports_id_filter?) && @store.supports_id_filter?
|
|
18
|
+
raise Scope::InvalidScopeError, 'vector adapter does not support complete ID-scoped search'
|
|
19
|
+
end
|
|
20
|
+
|
|
21
|
+
ids = @store.each_id.select { |id| eligible_id?(id, filters) }.uniq.sort
|
|
22
|
+
# Type eligibility comes from authoritative unit metadata, including for
|
|
23
|
+
# legacy vectors without a type payload. Other payload filters stay native.
|
|
24
|
+
filters = filters.reject { |key, _| key.to_s == 'type' }
|
|
25
|
+
best = []
|
|
26
|
+
ids.each_slice(BATCH_SIZE) do |batch|
|
|
27
|
+
# Fetch every bounded batch member before our raw-ID tie-break. A
|
|
28
|
+
# backend may order equal scores differently (or use SQL collation).
|
|
29
|
+
results = @store.search(query_vector, limit: batch.size, filters: filters, ids: batch)
|
|
30
|
+
results.each do |result|
|
|
31
|
+
next if batch.include?(result.id)
|
|
32
|
+
|
|
33
|
+
raise Scope::InvalidScopeError, 'vector adapter returned an ID outside the requested scope'
|
|
34
|
+
end
|
|
35
|
+
best = (best + results).sort_by { |result| [-result.score, result.id] }.first(limit)
|
|
36
|
+
end
|
|
37
|
+
best
|
|
38
|
+
rescue NotImplementedError
|
|
39
|
+
raise Scope::InvalidScopeError, 'vector adapter cannot enumerate IDs for complete scoped search'
|
|
40
|
+
end
|
|
41
|
+
|
|
42
|
+
private
|
|
43
|
+
|
|
44
|
+
def eligible_id?(id, filters)
|
|
45
|
+
return false unless @scope.include?(id)
|
|
46
|
+
|
|
47
|
+
types = filters[:type] || filters['type']
|
|
48
|
+
return true unless types
|
|
49
|
+
|
|
50
|
+
record = @scope.metadata_store.find(id.sub(/#chunk_\d+\z/, ''))
|
|
51
|
+
Array(types).map(&:to_s).include?(record['type'])
|
|
52
|
+
end
|
|
53
|
+
end
|
|
54
|
+
end
|
|
55
|
+
end
|
|
@@ -86,8 +86,11 @@ module Woods
|
|
|
86
86
|
type_filter: type_filter
|
|
87
87
|
)
|
|
88
88
|
|
|
89
|
+
candidates = expand_graph_candidates(candidates)
|
|
90
|
+
candidates = order_hybrid_candidates(candidates) if strategy == :hybrid
|
|
91
|
+
|
|
89
92
|
ExecutionResult.new(
|
|
90
|
-
candidates: bounded_candidates(
|
|
93
|
+
candidates: bounded_candidates(candidates, limit, strategy),
|
|
91
94
|
strategy: strategy,
|
|
92
95
|
query: query
|
|
93
96
|
)
|
|
@@ -155,7 +158,7 @@ module Woods
|
|
|
155
158
|
when :keyword
|
|
156
159
|
execute_keyword(classification: classification, limit: limit)
|
|
157
160
|
when :graph
|
|
158
|
-
execute_graph(
|
|
161
|
+
execute_graph(query: query, limit: limit)
|
|
159
162
|
when :hybrid
|
|
160
163
|
execute_hybrid(query, classification: classification, limit: limit, type_filter: type_filter)
|
|
161
164
|
when :direct
|
|
@@ -318,11 +321,11 @@ module Woods
|
|
|
318
321
|
# Graph strategy: find related units via dependency traversal.
|
|
319
322
|
#
|
|
320
323
|
# @return [Array<Candidate>]
|
|
321
|
-
def execute_graph(
|
|
324
|
+
def execute_graph(query:, limit:)
|
|
322
325
|
candidates = []
|
|
323
326
|
|
|
324
|
-
#
|
|
325
|
-
seeds = find_seed_identifiers(
|
|
327
|
+
# Resolve original query subjects to canonical metadata identifiers
|
|
328
|
+
seeds = find_seed_identifiers(query)
|
|
326
329
|
return [] if seeds.empty?
|
|
327
330
|
|
|
328
331
|
seeds.each do |seed_id|
|
|
@@ -376,7 +379,7 @@ module Woods
|
|
|
376
379
|
|
|
377
380
|
# Graph expansion on top vector results
|
|
378
381
|
graph_candidates = []
|
|
379
|
-
vector_candidates.first(3).each do |candidate|
|
|
382
|
+
order_hybrid_candidates(vector_candidates).first(3).each do |candidate|
|
|
380
383
|
deps = @graph_store.dependencies_of(StorageIdentity.identifier(candidate.identifier.sub(/#chunk_\d+\z/, '')))
|
|
381
384
|
deps.each do |dep|
|
|
382
385
|
graph_candidates << Candidate.new(
|
|
@@ -385,10 +388,16 @@ module Woods
|
|
|
385
388
|
end
|
|
386
389
|
end
|
|
387
390
|
|
|
388
|
-
#
|
|
389
|
-
#
|
|
390
|
-
|
|
391
|
-
|
|
391
|
+
# Keep every source occurrence for RRF. Ordering happens after typed
|
|
392
|
+
# graph identities expand, before the caller truncates unique units.
|
|
393
|
+
vector_candidates + keyword_candidates + graph_candidates
|
|
394
|
+
end
|
|
395
|
+
|
|
396
|
+
# Equal scores must not let insertion order or Ruby's sort implementation
|
|
397
|
+
# choose graph seeds, retained units, or a source's RRF rank. Identity is
|
|
398
|
+
# the tie-break; source makes cross-source occurrences deterministic too.
|
|
399
|
+
def order_hybrid_candidates(candidates)
|
|
400
|
+
candidates.sort_by { |candidate| [-candidate.score, candidate.identifier.to_s, candidate.source.to_s] }
|
|
392
401
|
end
|
|
393
402
|
|
|
394
403
|
# Direct strategy: look up specific identifiers from keywords.
|
|
@@ -449,27 +458,77 @@ module Woods
|
|
|
449
458
|
{ type: type_filter.map(&:to_s) }
|
|
450
459
|
end
|
|
451
460
|
|
|
452
|
-
#
|
|
453
|
-
#
|
|
454
|
-
|
|
455
|
-
|
|
456
|
-
|
|
457
|
-
|
|
458
|
-
|
|
459
|
-
|
|
460
|
-
|
|
461
|
-
|
|
462
|
-
|
|
463
|
-
|
|
461
|
+
# Keep graph subjects separate from intent words and preserve namespace
|
|
462
|
+
# separators/case that the general-purpose classifier keywords discard.
|
|
463
|
+
GRAPH_INTENT_WORDS = %w[trace follow track call calls called depends used find locate search look].freeze
|
|
464
|
+
private_constant :GRAPH_INTENT_WORDS
|
|
465
|
+
|
|
466
|
+
def find_seed_identifiers(query)
|
|
467
|
+
records = graph_query_records(query)
|
|
468
|
+
exact = records.flat_map do |term, matches|
|
|
469
|
+
graph_identifier_matches(matches, term)
|
|
470
|
+
end
|
|
471
|
+
return exact.map { |record| record['id'] }.uniq.sort unless exact.empty?
|
|
472
|
+
|
|
473
|
+
# A qualified miss must not seed another namespace through source-text
|
|
474
|
+
# mentions. Unqualified prose can still fall back to subject metadata.
|
|
475
|
+
records.reject { |term, _| term.include?('::') }.values.flatten
|
|
476
|
+
.map { |record| record['id'] }.uniq.sort.first(3)
|
|
477
|
+
end
|
|
478
|
+
|
|
479
|
+
# Sentence-capitalized instructions are still prose. Keep an instruction
|
|
480
|
+
# token only when it is also an explicitly spelled, existing identifier.
|
|
481
|
+
def graph_query_records(query)
|
|
482
|
+
query.scan(/\w+(?:::\w+)*/).uniq.filter_map do |term|
|
|
483
|
+
instruction = QueryClassifier::STOP_WORDS.include?(term.downcase) ||
|
|
484
|
+
GRAPH_INTENT_WORDS.include?(term.downcase)
|
|
485
|
+
next if instruction && !term.match?(/[A-Z]/)
|
|
486
|
+
|
|
487
|
+
records = graph_subject_records(term)
|
|
488
|
+
records = records.select { |record| graph_record_identifier(record) == term } if instruction
|
|
489
|
+
[term, records]
|
|
490
|
+
end.to_h
|
|
491
|
+
end
|
|
492
|
+
|
|
493
|
+
# Prefer the spelling supplied by the caller before inferring snake_case
|
|
494
|
+
# variants. Metadata search recovers canonical and typed storage identities.
|
|
495
|
+
def graph_subject_records(term)
|
|
496
|
+
direct = @metadata_store.find(term)
|
|
497
|
+
return [direct.merge('id' => term)] if direct
|
|
498
|
+
|
|
499
|
+
records = @metadata_store.search(term)
|
|
500
|
+
exact = records.select { |record| graph_record_identifier(record) == term }
|
|
501
|
+
return exact unless exact.empty?
|
|
502
|
+
|
|
503
|
+
camelized = term.split('::').map { |part| part.split('_').map(&:capitalize).join }.join('::')
|
|
504
|
+
return records if camelized == term
|
|
505
|
+
|
|
506
|
+
inferred = @metadata_store.find(camelized)
|
|
507
|
+
return [inferred.merge('id' => camelized)] if inferred
|
|
508
|
+
|
|
509
|
+
(records + @metadata_store.search(camelized)).uniq { |record| record['id'] }
|
|
510
|
+
end
|
|
511
|
+
|
|
512
|
+
# Exact qualified identity wins over a bounded, deterministic set of
|
|
513
|
+
# unqualified namespace suffix matches; source-text mentions are not roots.
|
|
514
|
+
def graph_identifier_matches(records, term)
|
|
515
|
+
literal = records.select { |record| graph_record_identifier(record) == term }
|
|
516
|
+
return literal unless literal.empty?
|
|
517
|
+
|
|
518
|
+
exact = records.select do |record|
|
|
519
|
+
graph_record_identifier(record).delete('_').casecmp?(term.delete('_'))
|
|
464
520
|
end
|
|
521
|
+
return exact unless exact.empty?
|
|
522
|
+
return [] if term.include?('::')
|
|
465
523
|
|
|
466
|
-
|
|
467
|
-
|
|
468
|
-
results = @metadata_store.search(classification.keywords.join(' '))
|
|
469
|
-
seeds = results.first(3).map { |r| r['id'] }
|
|
524
|
+
suffix_matches = records.select do |record|
|
|
525
|
+
graph_record_identifier(record).split('::').last.delete('_').casecmp?(term.delete('_'))
|
|
470
526
|
end
|
|
527
|
+
suffix_matches.sort_by { |record| record['id'] }.first(3)
|
|
528
|
+
end
|
|
471
529
|
|
|
472
|
-
|
|
530
|
+
def graph_record_identifier(record)
|
|
531
|
+
record['identifier'] || StorageIdentity.identifier(record['id'])
|
|
473
532
|
end
|
|
474
533
|
|
|
475
534
|
# Deduplicate candidates, keeping the highest-scored entry per identifier.
|