woods 2.0.0.beta2 → 2.0.0.beta3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +262 -1
- data/CONTRIBUTING.md +173 -9
- data/README.md +7 -3
- data/SECURITY.md +9 -6
- data/docs/AGENT_GUIDE.md +83 -4
- data/docs/AGENT_SETUP.md +82 -1
- data/docs/BACKEND_MATRIX.md +20 -0
- data/docs/CLIENT_HOOKS.md +111 -0
- data/docs/CONFIGURATION_REFERENCE.md +199 -14
- data/docs/CONSOLE_MCP_SETUP.md +35 -5
- data/docs/DOCKER_SETUP.md +21 -2
- data/docs/EVALUATION.md +464 -1
- data/docs/EXTRACTOR_REFERENCE.md +36 -5
- data/docs/FAQ.md +11 -12
- data/docs/GETTING_STARTED.md +17 -5
- data/docs/INCREMENTAL_EXTRACTION.md +117 -1
- data/docs/INDEX_LAYOUT.md +382 -0
- data/docs/INTERNALS.md +7 -2
- data/docs/MCP_SERVERS.md +221 -5
- data/docs/MCP_TOOL_COOKBOOK.md +33 -18
- data/docs/NOTION_INTEGRATION.md +13 -0
- data/docs/OBSIDIAN_INTEGRATION.md +57 -9
- data/docs/PUBLISHED_INDEX.md +55 -0
- data/docs/README.md +7 -0
- data/docs/RETRIEVAL_GUIDE.md +253 -11
- data/docs/RUNTIME_TRACING.md +71 -0
- data/docs/SOURCE_FRESHNESS.md +143 -0
- data/docs/TROUBLESHOOTING.md +117 -5
- data/docs/UNBLOCKED_INTEGRATION.md +25 -0
- data/docs/UPGRADING_TO_2.md +44 -22
- data/docs/WATCH_DAEMON.md +259 -59
- data/exe/woods-agent-config +6 -0
- data/exe/woods-extract +5 -0
- data/exe/woods-hook-context +6 -0
- data/lib/generators/woods/templates/woods.rb.tt +1 -3
- data/lib/tasks/woods.rake +47 -397
- data/lib/woods/agent_configuration/applier.rb +133 -0
- data/lib/woods/agent_configuration/cli.rb +101 -0
- data/lib/woods/agent_configuration/cli_options.rb +29 -0
- data/lib/woods/agent_configuration/document.rb +105 -0
- data/lib/woods/agent_configuration/error.rb +7 -0
- data/lib/woods/agent_configuration/launcher.rb +75 -0
- data/lib/woods/agent_configuration/layout.rb +59 -0
- data/lib/woods/agent_configuration/managed_section.rb +62 -0
- data/lib/woods/agent_configuration/plan.rb +98 -0
- data/lib/woods/agent_configuration/plan_diff.rb +38 -0
- data/lib/woods/agent_configuration/planned_files.rb +61 -0
- data/lib/woods/agent_configuration/planner.rb +63 -0
- data/lib/woods/agent_configuration/planner_validation.rb +77 -0
- data/lib/woods/agent_configuration/preflight.rb +100 -0
- data/lib/woods/agent_configuration/recovery.rb +49 -0
- data/lib/woods/ast/node.rb +2 -0
- data/lib/woods/ast/parser.rb +38 -5
- data/lib/woods/builder.rb +21 -5
- data/lib/woods/cache/cache_middleware.rb +28 -7
- data/lib/woods/cache/cache_store.rb +4 -5
- data/lib/woods/change_set.rb +5 -4
- data/lib/woods/console/credential_index.rb +20 -2
- data/lib/woods/console/credential_scanner.rb +14 -14
- data/lib/woods/console/credential_scanner_registry.rb +36 -0
- data/lib/woods/console/embedded_executor.rb +1 -1
- data/lib/woods/console/encrypted_credential_snapshot.rb +16 -0
- data/lib/woods/console/rack_middleware.rb +22 -13
- data/lib/woods/console/server.rb +18 -16
- data/lib/woods/dependency_graph.rb +65 -13
- data/lib/woods/embedding/corpus.rb +94 -0
- data/lib/woods/embedding/indexer.rb +90 -46
- data/lib/woods/embedding/openai.rb +17 -6
- data/lib/woods/evaluation/ablation_executor.rb +6 -1
- data/lib/woods/evaluation/ablation_timed_executor.rb +22 -4
- data/lib/woods/export/typed_reader.rb +56 -0
- data/lib/woods/extractor.rb +232 -137
- data/lib/woods/extractors/action_cable_extractor.rb +3 -1
- data/lib/woods/extractors/behavioral_profile.rb +9 -7
- data/lib/woods/extractors/caching_extractor.rb +3 -1
- data/lib/woods/extractors/concern_extractor.rb +64 -6
- data/lib/woods/extractors/configuration_extractor.rb +7 -3
- data/lib/woods/extractors/controller_extractor.rb +13 -4
- data/lib/woods/extractors/database_view_extractor.rb +3 -1
- data/lib/woods/extractors/decorator_extractor.rb +3 -1
- data/lib/woods/extractors/engine_extractor.rb +3 -1
- data/lib/woods/extractors/event_extractor.rb +4 -2
- data/lib/woods/extractors/factory_extractor.rb +3 -1
- data/lib/woods/extractors/graphql_extractor.rb +8 -2
- data/lib/woods/extractors/i18n_extractor.rb +3 -1
- data/lib/woods/extractors/job_extractor.rb +6 -19
- data/lib/woods/extractors/lib_extractor.rb +3 -1
- data/lib/woods/extractors/mailer_extractor.rb +20 -5
- data/lib/woods/extractors/manager_extractor.rb +3 -1
- data/lib/woods/extractors/method_parameters.rb +53 -0
- data/lib/woods/extractors/middleware_argument.rb +65 -0
- data/lib/woods/extractors/middleware_extractor.rb +9 -3
- data/lib/woods/extractors/migration_extractor.rb +3 -1
- data/lib/woods/extractors/model_extractor.rb +39 -33
- data/lib/woods/extractors/package_extractor.rb +24 -4
- data/lib/woods/extractors/phlex_extractor.rb +3 -1
- data/lib/woods/extractors/policy_extractor.rb +3 -1
- data/lib/woods/extractors/poro_extractor.rb +3 -1
- data/lib/woods/extractors/pundit_extractor.rb +3 -1
- data/lib/woods/extractors/rails_source_extractor.rb +4 -2
- data/lib/woods/extractors/rake_task_extractor.rb +4 -2
- data/lib/woods/extractors/route_extractor.rb +3 -1
- data/lib/woods/extractors/route_helper_resolver.rb +10 -33
- data/lib/woods/extractors/scheduled_job_extractor.rb +41 -15
- data/lib/woods/extractors/serializer_extractor.rb +4 -2
- data/lib/woods/extractors/service_extractor.rb +3 -1
- data/lib/woods/extractors/shared_dependency_scanner.rb +2 -2
- data/lib/woods/extractors/shared_utility_methods.rb +27 -15
- data/lib/woods/extractors/source_nesting.rb +1 -1
- data/lib/woods/extractors/state_machine_extractor.rb +3 -1
- data/lib/woods/extractors/test_mapping_extractor.rb +3 -1
- data/lib/woods/extractors/validator_extractor.rb +3 -1
- data/lib/woods/extractors/view_component_extractor.rb +3 -1
- data/lib/woods/extractors/view_template_extractor.rb +3 -1
- data/lib/woods/gem_mapper.rb +2 -0
- data/lib/woods/git_history.rb +116 -0
- data/lib/woods/graph_analyzer.rb +35 -6
- data/lib/woods/hooks/context_cli.rb +54 -0
- data/lib/woods/hooks/context_event.rb +88 -0
- data/lib/woods/hooks/context_hint.rb +73 -0
- data/lib/woods/hooks/context_impact.rb +77 -0
- data/lib/woods/hooks/context_output.rb +47 -0
- data/lib/woods/hooks/context_state.rb +102 -0
- data/lib/woods/hooks/refresh.rb +79 -0
- data/lib/woods/hooks/rule_projection.rb +78 -0
- data/lib/woods/input_rules.rb +19 -0
- data/lib/woods/mcp/bearer_auth.rb +20 -12
- data/lib/woods/mcp/bootstrapper.rb +62 -0
- data/lib/woods/mcp/index_reader.rb +323 -160
- data/lib/woods/mcp/initialization_guidance.rb +27 -0
- data/lib/woods/mcp/origin_guard.rb +17 -9
- data/lib/woods/mcp/published_lexical_retriever.rb +115 -0
- data/lib/woods/mcp/renderers/markdown_renderer.rb +8 -1
- data/lib/woods/mcp/renderers/plain_renderer.rb +7 -1
- data/lib/woods/mcp/search_results.rb +74 -0
- data/lib/woods/mcp/server.rb +158 -37
- data/lib/woods/mcp/tool_contract.rb +2 -0
- data/lib/woods/mcp/tool_response_renderer.rb +25 -0
- data/lib/woods/mcp/traversal_evidence.rb +113 -0
- data/lib/woods/mcp/traversal_evidence_index.rb +100 -0
- data/lib/woods/mcp/traversal_evidence_page.rb +41 -0
- data/lib/woods/mcp/traversal_evidence_text.rb +52 -0
- data/lib/woods/notion/exporter.rb +56 -17
- data/lib/woods/obsidian/destination_plan.rb +98 -0
- data/lib/woods/obsidian/name_mapper.rb +19 -3
- data/lib/woods/obsidian/note_builder.rb +19 -10
- data/lib/woods/obsidian/vault_exporter.rb +88 -32
- data/lib/woods/operator/pipeline_guard.rb +18 -13
- data/lib/woods/path_dispatcher.rb +7 -1
- data/lib/woods/payload_store.rb +27 -26
- data/lib/woods/railtie.rb +3 -3
- data/lib/woods/railtie_support.rb +12 -12
- data/lib/woods/rake_helpers.rb +392 -0
- data/lib/woods/resilience/graph_invariant_validator/membership_checks.rb +71 -0
- data/lib/woods/resilience/graph_invariant_validator/node_checks.rb +61 -0
- data/lib/woods/resilience/graph_invariant_validator/reverse_relationship_checks.rb +46 -0
- data/lib/woods/resilience/graph_invariant_validator.rb +119 -0
- data/lib/woods/resilience/index_validator/graph_checks.rb +80 -0
- data/lib/woods/resilience/index_validator.rb +112 -23
- data/lib/woods/retrieval/context_assembler.rb +50 -15
- data/lib/woods/retrieval/lexical_assembler.rb +73 -0
- data/lib/woods/retrieval/lexical_index.rb +119 -0
- data/lib/woods/retrieval/ranker.rb +4 -2
- data/lib/woods/retrieval/scope.rb +108 -0
- data/lib/woods/retrieval/scoped_graph_store.rb +32 -0
- data/lib/woods/retrieval/scoped_vector_store.rb +55 -0
- data/lib/woods/retrieval/search_executor.rb +86 -27
- data/lib/woods/retrieval/source_evidence.rb +200 -0
- data/lib/woods/retriever.rb +98 -22
- data/lib/woods/ruby_analyzer/trace_enricher.rb +77 -38
- data/lib/woods/session_tracer/middleware.rb +10 -12
- data/lib/woods/session_tracer/redis_store.rb +22 -6
- data/lib/woods/session_tracer/session_flow_assembler.rb +23 -17
- data/lib/woods/session_tracer/solid_cache_coordination.rb +6 -4
- data/lib/woods/session_tracer/unit_resolver.rb +63 -0
- data/lib/woods/source_inputs/consumer_errors.rb +27 -0
- data/lib/woods/source_inputs/handoff.rb +102 -0
- data/lib/woods/source_inputs/launcher.rb +157 -0
- data/lib/woods/source_inputs/manifest.rb +124 -0
- data/lib/woods/source_inputs/private_key.rb +55 -0
- data/lib/woods/source_inputs/scanner.rb +171 -0
- data/lib/woods/source_inputs/scopes.rb +71 -0
- data/lib/woods/source_inputs/session.rb +214 -0
- data/lib/woods/source_inputs/status.rb +84 -0
- data/lib/woods/source_inputs/verifier.rb +107 -0
- data/lib/woods/storage/metadata_store.rb +25 -25
- data/lib/woods/storage/pgvector.rb +29 -8
- data/lib/woods/storage/qdrant.rb +17 -7
- data/lib/woods/storage/vector_store.rb +18 -6
- data/lib/woods/tasks.rb +3 -2
- data/lib/woods/temporal/json_snapshot_store.rb +29 -8
- data/lib/woods/unblocked/exporter.rb +59 -70
- data/lib/woods/version.rb +1 -1
- data/lib/woods/watch/boot_snapshot.rb +52 -0
- data/lib/woods/watch/daemon.rb +136 -28
- data/lib/woods/watch/listen_watcher.rb +4 -0
- data/lib/woods/watch/polling_watcher.rb +5 -1
- data/lib/woods/watch/status.rb +20 -15
- data/lib/woods/watch/tree_scan.rb +21 -13
- data/lib/woods/watch/watcher.rb +4 -1
- data/lib/woods.rb +50 -11
- data/plugin/.claude-plugin/plugin.json +1 -1
- data/plugin/hooks/adapters/normalize.jq +15 -0
- data/plugin/hooks/adapters/normalize.rb +63 -0
- data/plugin/hooks/hooks.json +20 -0
- data/plugin/hooks/woods-context.sh +50 -0
- data/plugin/hooks/woods-input-rules.sh +159 -0
- data/plugin/hooks/woods-opencode.mjs +65 -0
- data/plugin/hooks/woods-post-edit.sh +2 -225
- data/plugin/hooks/woods-refresh.sh +260 -0
- data/plugin/hooks/woods-session-start.sh +47 -55
- data/plugin/skills/woods-agent-enable/SKILL.md +13 -0
- data/plugin/skills/woods-diagnose/SKILL.md +288 -1
- data/plugin/skills/woods-investigate/SKILL.md +106 -0
- data/plugin/skills/woods-mcp-config/SKILL.md +89 -1
- data/plugin/skills/woods-setup/SKILL.md +107 -6
- metadata +84 -5
data/lib/woods/mcp/server.rb
CHANGED
|
@@ -14,10 +14,13 @@ require_relative '../tasks'
|
|
|
14
14
|
require_relative '../watch/status'
|
|
15
15
|
require_relative '../filename_utils'
|
|
16
16
|
require_relative '../update_check'
|
|
17
|
+
require_relative '../retrieval/source_evidence'
|
|
18
|
+
require_relative '../session_tracer/unit_resolver'
|
|
17
19
|
require_relative 'bootstrap_state'
|
|
18
20
|
require_relative 'errors'
|
|
19
21
|
require_relative 'index_reader'
|
|
20
22
|
require_relative 'index_reader_pinning'
|
|
23
|
+
require_relative 'initialization_guidance'
|
|
21
24
|
require_relative 'protocol_policy'
|
|
22
25
|
require_relative 'tasks/extension'
|
|
23
26
|
require_relative 'tasks/request_capture'
|
|
@@ -90,6 +93,7 @@ module Woods
|
|
|
90
93
|
def build(index_dir:, retriever: nil, operator: nil, feedback_store: nil, snapshot_store: nil,
|
|
91
94
|
bootstrap_state: nil, response_format: nil, warmup: true, retriever_reloader: nil)
|
|
92
95
|
reader = IndexReader.new(index_dir)
|
|
96
|
+
retriever.bind_reader(reader) if retriever.respond_to?(:bind_reader)
|
|
93
97
|
reader.warmup! if warmup
|
|
94
98
|
config = Woods.configuration
|
|
95
99
|
format = response_format || (config.respond_to?(:context_format) ? config.context_format : nil) || :markdown
|
|
@@ -156,7 +160,9 @@ module Woods
|
|
|
156
160
|
description: 'Traverse forward dependencies of a unit (what it depends on). ' \
|
|
157
161
|
'Narrow with depth, types and via first: they shrink the answer, ' \
|
|
158
162
|
'while limit and offset only page it. Returns a BFS tree with ' \
|
|
159
|
-
"depth,
|
|
163
|
+
"depth, paged to #{DEFAULT_TRAVERSAL_LIMIT} nodes by default. " \
|
|
164
|
+
'max_nodes/max_edges bound the walk independently; partial_reason reports a budget cutoff. ' \
|
|
165
|
+
'Use explain:true for recorded directed relationships and bounded witnesses; ambiguous types remain explicit.',
|
|
160
166
|
reader_method: :traverse_dependencies,
|
|
161
167
|
render_key: :dependencies)
|
|
162
168
|
define_traversal_tool(server, reader, respond, renderer,
|
|
@@ -164,7 +170,9 @@ module Woods
|
|
|
164
170
|
description: 'Traverse reverse dependencies of a unit (what depends on it). ' \
|
|
165
171
|
'Narrow with depth, types and via first: they shrink the answer, ' \
|
|
166
172
|
'while limit and offset only page it. Returns a BFS tree with ' \
|
|
167
|
-
"depth,
|
|
173
|
+
"depth, paged to #{DEFAULT_TRAVERSAL_LIMIT} nodes by default. " \
|
|
174
|
+
'max_nodes/max_edges bound the walk independently; partial_reason reports a budget cutoff. ' \
|
|
175
|
+
'Use explain:true for recorded directed relationships and bounded witnesses; ambiguous types remain explicit.',
|
|
168
176
|
reader_method: :traverse_dependents,
|
|
169
177
|
render_key: :dependents)
|
|
170
178
|
define_structure_tool(server, reader, respond, renderer)
|
|
@@ -191,6 +199,7 @@ module Woods
|
|
|
191
199
|
register_resource_handler(server, reader)
|
|
192
200
|
ToolContract.apply!(server)
|
|
193
201
|
IndexReaderPinning.install(server, reader: reader)
|
|
202
|
+
server.instructions = InitializationGuidance.for(server.tools.keys)
|
|
194
203
|
|
|
195
204
|
# Last, after every conditional registration above — the whole point is
|
|
196
205
|
# that a host with Notion wired advertises the same tool order as one
|
|
@@ -435,6 +444,11 @@ module Woods
|
|
|
435
444
|
identifier: { type: 'string',
|
|
436
445
|
description: 'Exact unit identifier (e.g. "Post", "PostsController", "Api::V1::HealthController")' },
|
|
437
446
|
name: { type: 'string', description: 'Alias for `identifier`. Either one works.' },
|
|
447
|
+
type: { type: 'string', description: 'Optional actual published unit type, to disambiguate shared identifiers.' },
|
|
448
|
+
evidence: { type: 'string', enum: %w[full compact outline], description: 'Published evidence mode (default full).' },
|
|
449
|
+
query: { type: 'string', description: 'Optional relevance query for compact evidence; absent means API orientation.' },
|
|
450
|
+
budget: { type: 'integer', minimum: 1, description: 'Compact/outline token estimate budget (default 2000); full lookup remains complete.' },
|
|
451
|
+
source_sha256: { type: 'string', description: 'Require the published source SHA256 from an earlier excerpt; refuses changed source.' },
|
|
438
452
|
include_source: { type: 'boolean', description: 'Include source_code in response (default: true)' },
|
|
439
453
|
sections: {
|
|
440
454
|
type: 'array', items: { type: 'string' },
|
|
@@ -445,7 +459,7 @@ module Woods
|
|
|
445
459
|
# accepted alias. The handler validates that one of the two
|
|
446
460
|
# was provided.
|
|
447
461
|
}
|
|
448
|
-
) do |server_context:, identifier: nil, name: nil, include_source: nil, sections: nil|
|
|
462
|
+
) do |server_context:, identifier: nil, name: nil, include_source: nil, sections: nil, type: nil, evidence: 'full', query: nil, budget: nil, source_sha256: nil|
|
|
449
463
|
identifier ||= name
|
|
450
464
|
if identifier.nil? || identifier.empty?
|
|
451
465
|
next respond_err.call(
|
|
@@ -457,8 +471,30 @@ module Woods
|
|
|
457
471
|
)
|
|
458
472
|
end
|
|
459
473
|
sections = coerce.call(sections)
|
|
460
|
-
|
|
474
|
+
begin
|
|
475
|
+
Retrieval::SourceEvidence.validate_mode!(evidence)
|
|
476
|
+
if evidence != 'full' && (include_source == false || sections&.any?)
|
|
477
|
+
raise ArgumentError, 'compact/outline evidence cannot be combined with include_source: false or sections'
|
|
478
|
+
end
|
|
479
|
+
if evidence == 'full' && (!query.nil? || !budget.nil?)
|
|
480
|
+
raise ArgumentError, 'query and budget apply only to compact/outline evidence'
|
|
481
|
+
end
|
|
482
|
+
rescue ArgumentError => e
|
|
483
|
+
next respond_err.call(e.message, code: :unsupported_argument, tool: 'lookup', argument: 'evidence')
|
|
484
|
+
end
|
|
485
|
+
unit = type ? reader.find_unit(identifier, type: type) : reader.find_unit(identifier)
|
|
461
486
|
if unit
|
|
487
|
+
if source_sha256 && Digest::SHA256.hexdigest(unit['source_code'].to_s) != source_sha256
|
|
488
|
+
next respond_err.call('Published source changed since the excerpt; retrieve fresh evidence before verification.',
|
|
489
|
+
code: :stale_index, tool: 'lookup', argument: 'source_sha256')
|
|
490
|
+
end
|
|
491
|
+
if evidence != 'full'
|
|
492
|
+
selected = Retrieval::SourceEvidence.new(unit: unit, query: query, generation: reader.loaded_generation)
|
|
493
|
+
.render(mode: evidence, budget: budget || 2000,
|
|
494
|
+
counter: ->(text) { (text.length / 4.0).ceil })
|
|
495
|
+
next ::MCP::Tool::Response.new([{ type: 'text', text: selected.text }],
|
|
496
|
+
structured_content: { text: selected.text, data: { evidence: selected.provenance } })
|
|
497
|
+
end
|
|
462
498
|
always_include = %w[type identifier file_path namespace]
|
|
463
499
|
filtered = unit
|
|
464
500
|
filtered = filtered.except('source_code') if include_source == false
|
|
@@ -485,8 +521,9 @@ module Woods
|
|
|
485
521
|
server.define_tool(
|
|
486
522
|
name: 'search',
|
|
487
523
|
description: 'Find code units whose identifiers (or source/metadata) match a regex. ' \
|
|
488
|
-
'Example: search("Worker|Job")
|
|
489
|
-
'returns units starting with "Post". Returns [{identifier, type, match_field}]. ' \
|
|
524
|
+
'Example: search("Worker|Job") finds workers and jobs; search("^Post") ' \
|
|
525
|
+
'returns units starting with "Post". Returns [{identifier, type, match_field}] plus completeness. ' \
|
|
526
|
+
'Check completeness before treating discovery as exhaustive; a limit is only a page size. ' \
|
|
490
527
|
'Use `lookup` for exact identifiers, `dependencies`/`dependents` for graph traversal. ' \
|
|
491
528
|
'Gotchas: query is a Ruby regex — literal pipe needs escaping as \\|; ' \
|
|
492
529
|
'types restricts which index directories are scanned (e.g. ["mailer"] scans only ' \
|
|
@@ -500,6 +537,10 @@ module Woods
|
|
|
500
537
|
type: 'array', items: { type: 'string' },
|
|
501
538
|
description: 'Restrict scan to these unit types: model, controller, service, job, mailer, etc.'
|
|
502
539
|
},
|
|
540
|
+
packages: { type: 'array', items: { type: 'string' },
|
|
541
|
+
description: 'Exact published package owners, OR within the list; AND with source_paths and types.' },
|
|
542
|
+
source_paths: { type: 'array', items: { type: 'string' },
|
|
543
|
+
description: 'Application-relative directory prefixes; segment-aware, OR within the list. Applied before limits.' },
|
|
503
544
|
fields: {
|
|
504
545
|
type: 'array', items: { type: 'string', enum: %w[identifier metadata source_code] },
|
|
505
546
|
description: 'Fields to search: identifier (default), source_code, metadata'
|
|
@@ -517,7 +558,8 @@ module Woods
|
|
|
517
558
|
}
|
|
518
559
|
}
|
|
519
560
|
}
|
|
520
|
-
) do |server_context:, query: nil, types: nil, fields: nil, limit: nil, exact_prefix: nil, exact_suffix: nil
|
|
561
|
+
) do |server_context:, query: nil, types: nil, fields: nil, limit: nil, exact_prefix: nil, exact_suffix: nil,
|
|
562
|
+
packages: nil, source_paths: nil|
|
|
521
563
|
if (query.nil? || query.empty?) &&
|
|
522
564
|
(exact_prefix.nil? || exact_prefix.empty?) &&
|
|
523
565
|
(exact_suffix.nil? || exact_suffix.empty?)
|
|
@@ -538,17 +580,29 @@ module Woods
|
|
|
538
580
|
fields: fields || %w[identifier],
|
|
539
581
|
limit: limit || 20,
|
|
540
582
|
exact_prefix: exact_prefix,
|
|
541
|
-
exact_suffix: exact_suffix
|
|
583
|
+
exact_suffix: exact_suffix,
|
|
584
|
+
packages: packages, source_paths: source_paths
|
|
542
585
|
)
|
|
543
586
|
results = search_result[:results]
|
|
544
587
|
payload = {
|
|
545
588
|
query: query,
|
|
546
589
|
result_count: results.size,
|
|
547
|
-
results: results
|
|
590
|
+
results: results,
|
|
591
|
+
completeness: search_result[:completeness]
|
|
548
592
|
}
|
|
593
|
+
payload[:applied_scope] = search_result[:applied_scope] if search_result[:applied_scope]
|
|
549
594
|
payload[:note] = search_result[:note] if search_result[:note]
|
|
550
595
|
payload[:partial] = true if search_result[:partial]
|
|
596
|
+
payload[:hint] = search_result[:hint] if search_result[:hint]
|
|
551
597
|
respond.call(renderer.render(:search, payload))
|
|
598
|
+
rescue Retrieval::Scope::InvalidScopeError => e
|
|
599
|
+
respond_err.call(e.message, code: :unsupported_argument, tool: 'search', argument: 'scope')
|
|
600
|
+
rescue IOError, SystemCallError, JSON::ParserError, EncodingError
|
|
601
|
+
respond_err.call(
|
|
602
|
+
'Search completeness: unknown (unreadable_or_corrupt_source). ' \
|
|
603
|
+
'An Index artifact is unavailable or malformed; inspect woods_status and run woods:validate.',
|
|
604
|
+
code: :corrupt_artifact, tool: 'search', completeness: SearchResults.unavailable
|
|
605
|
+
)
|
|
552
606
|
end
|
|
553
607
|
end
|
|
554
608
|
|
|
@@ -563,6 +617,7 @@ module Woods
|
|
|
563
617
|
properties: {
|
|
564
618
|
identifier: { type: 'string', description: 'Unit identifier to start from' },
|
|
565
619
|
depth: { type: 'integer', description: 'Maximum traversal depth (default: 2)' },
|
|
620
|
+
explain: { type: 'boolean', description: 'Include recorded relationship evidence and shared shortest witnesses (default: false)' },
|
|
566
621
|
types: {
|
|
567
622
|
type: 'array', items: { type: 'string' },
|
|
568
623
|
description: 'Filter to these types'
|
|
@@ -580,22 +635,29 @@ module Woods
|
|
|
580
635
|
},
|
|
581
636
|
limit: { type: 'integer',
|
|
582
637
|
description: "Maximum nodes to return (default: #{DEFAULT_TRAVERSAL_LIMIT})" },
|
|
583
|
-
offset: { type: 'integer', description: 'Skip this many nodes (default: 0)' }
|
|
638
|
+
offset: { type: 'integer', description: 'Skip this many nodes (default: 0)' },
|
|
639
|
+
max_nodes: { type: 'integer', minimum: 1, maximum: 10_000,
|
|
640
|
+
description: 'Visited-node budget including root (default: 1000; maximum: 10000)' },
|
|
641
|
+
max_edges: { type: 'integer', minimum: 1, maximum: 100_000,
|
|
642
|
+
description: 'Edge-check budget before filters, including reverse via checks (default: 10000; maximum: 100000)' }
|
|
584
643
|
},
|
|
585
644
|
required: ['identifier']
|
|
586
645
|
}
|
|
587
|
-
) do |identifier:, server_context:, depth: nil, types: nil, via: nil, limit: nil, offset: nil|
|
|
646
|
+
) do |identifier:, server_context:, depth: nil, types: nil, via: nil, limit: nil, offset: nil, max_nodes: nil, max_edges: nil, explain: nil|
|
|
588
647
|
types = coerce.call(types)
|
|
589
648
|
via = coerce.call(via)
|
|
590
649
|
depth = coerce_int.call(depth)
|
|
591
650
|
limit = coerce_int.call(limit)
|
|
592
651
|
offset = coerce_int.call(offset)
|
|
593
|
-
result = reader.send(reader_method, identifier, depth: depth || 2, types: types, via: via
|
|
652
|
+
result = reader.send(reader_method, identifier, depth: depth || 2, types: types, via: via,
|
|
653
|
+
max_nodes: coerce_int.call(max_nodes) || 1000,
|
|
654
|
+
max_edges: coerce_int.call(max_edges) || 10_000, explain: explain || false)
|
|
594
655
|
if result[:found] == false
|
|
595
656
|
result[:message] =
|
|
596
657
|
"Identifier '#{identifier}' not found in the index. Use 'search' to find valid identifiers."
|
|
597
658
|
end
|
|
598
659
|
paginate_nodes.call(result, limit || DEFAULT_TRAVERSAL_LIMIT, offset || 0)
|
|
660
|
+
TraversalEvidencePage.apply(result)
|
|
599
661
|
respond.call(renderer.render(render_key, result))
|
|
600
662
|
end
|
|
601
663
|
end
|
|
@@ -874,13 +936,14 @@ module Woods
|
|
|
874
936
|
coerce = method(:coerce_array)
|
|
875
937
|
stale_check = method(:stale_index_result?)
|
|
876
938
|
degraded_response = method(:degraded_retrieval_response)
|
|
939
|
+
retrieval_mode = retriever.respond_to?(:mode) ? retriever.mode : :semantic
|
|
877
940
|
server.define_tool(
|
|
878
941
|
name: 'codebase_retrieve',
|
|
879
|
-
description: '
|
|
942
|
+
description: 'Ranked retrieval: relevant code units for a natural-language question. ' \
|
|
880
943
|
'Example: codebase_retrieve("how does billing work?") returns ranked source context. ' \
|
|
881
944
|
'Returns a token-budgeted context string ready to paste into a prompt. ' \
|
|
882
945
|
'Use `search` for exact name/pattern matching; use this for conceptual questions. ' \
|
|
883
|
-
'
|
|
946
|
+
'Uses configured embeddings, or explicit WOODS_RETRIEVAL_MODE=lexical over extraction units. ' \
|
|
884
947
|
'By default excludes test_mappings (~33% of a typical index) so spec filenames do not ' \
|
|
885
948
|
'dominate semantic rank; pass types: ["test_mapping"] to opt back in. ' \
|
|
886
949
|
'Parameter: use `budget` for the token budget (not `limit` — that means result count ' \
|
|
@@ -890,12 +953,15 @@ module Woods
|
|
|
890
953
|
query: { type: 'string',
|
|
891
954
|
description: 'Natural language question (e.g. "How does user authentication work?")' },
|
|
892
955
|
budget: { type: 'integer',
|
|
893
|
-
description: 'Token budget for context assembly (
|
|
956
|
+
description: 'Token budget for context assembly (configured max_context_tokens; otherwise 8000).' },
|
|
957
|
+
evidence: { type: 'string', enum: %w[full compact outline],
|
|
958
|
+
description: 'Explicit complete spans or API outline within each ranked unit; default full retains existing output.' },
|
|
894
959
|
types: {
|
|
895
960
|
type: 'array', items: { type: 'string' },
|
|
896
961
|
description: 'Restrict results to these unit types (model, controller, service, job, mailer, ' \
|
|
897
962
|
'rails_source, test_mapping, etc.). Overrides the default test_mapping exclusion. ' \
|
|
898
|
-
'
|
|
963
|
+
'Lexical mode and explicit package/path scopes filter before limits and omit the global rank table. ' \
|
|
964
|
+
'In semantic mode, when the unfiltered top-K has no requested type, the retriever ' \
|
|
899
965
|
'falls back to rank-within-type so the response is populated whenever units of ' \
|
|
900
966
|
'the requested type exist in the index. The response appends a "Type rank ' \
|
|
901
967
|
'context" table with per-type: source, rank in unfiltered top-K, global_k, ' \
|
|
@@ -904,6 +970,10 @@ module Woods
|
|
|
904
970
|
'(index has this type but other requested types filled the result), absent ' \
|
|
905
971
|
'(zero units of this type in the index).'
|
|
906
972
|
},
|
|
973
|
+
packages: { type: 'array', items: { type: 'string' },
|
|
974
|
+
description: 'Exact published nearest package owners. OR within the list; AND with paths and type eligibility.' },
|
|
975
|
+
source_paths: { type: 'array', items: { type: 'string' },
|
|
976
|
+
description: 'Application-relative directory prefixes. Scope applies before candidate limits; graph expansion stays inside it.' },
|
|
907
977
|
exclude_types: {
|
|
908
978
|
type: 'array', items: { type: 'string' },
|
|
909
979
|
description: 'Additional types to exclude on top of the default test_mapping exclusion.'
|
|
@@ -911,7 +981,7 @@ module Woods
|
|
|
911
981
|
},
|
|
912
982
|
required: ['query']
|
|
913
983
|
}
|
|
914
|
-
) do |query:, server_context:, budget: nil, limit: nil, types: nil, exclude_types: nil|
|
|
984
|
+
) do |query:, server_context:, budget: nil, limit: nil, types: nil, exclude_types: nil, packages: nil, source_paths: nil, evidence: 'full'|
|
|
915
985
|
# `limit` isn't declared in the schema but clients still send it
|
|
916
986
|
# because sibling tools (search, recent_changes, pagerank) use
|
|
917
987
|
# `limit` as a result count. Mapping it to `budget` here would
|
|
@@ -919,10 +989,10 @@ module Woods
|
|
|
919
989
|
# budget). Surface a helpful typed error instead.
|
|
920
990
|
unless limit.nil?
|
|
921
991
|
next respond_err.call(
|
|
922
|
-
'codebase_retrieve uses `budget` (token budget, default
|
|
992
|
+
'codebase_retrieve uses `budget` (token budget, configured default), not `limit`. ' \
|
|
923
993
|
'`limit` is the result-count parameter on sibling tools (search, recent_changes, pagerank). ' \
|
|
924
994
|
"Pass `budget: #{coerce_int.call(limit)}` if you meant a #{coerce_int.call(limit)}-token context, " \
|
|
925
|
-
'or drop the kwarg entirely for the default
|
|
995
|
+
'or drop the kwarg entirely for the configured default.',
|
|
926
996
|
code: :unsupported_argument,
|
|
927
997
|
tool: 'codebase_retrieve',
|
|
928
998
|
argument: 'limit',
|
|
@@ -931,6 +1001,11 @@ module Woods
|
|
|
931
1001
|
)
|
|
932
1002
|
end
|
|
933
1003
|
|
|
1004
|
+
begin
|
|
1005
|
+
Retrieval::SourceEvidence.validate_mode!(evidence)
|
|
1006
|
+
rescue ArgumentError => e
|
|
1007
|
+
next respond_err.call(e.message, code: :unsupported_argument, tool: 'codebase_retrieve', argument: 'evidence')
|
|
1008
|
+
end
|
|
934
1009
|
budget = coerce_int.call(budget)
|
|
935
1010
|
types = coerce.call(types)
|
|
936
1011
|
exclude_types = coerce.call(exclude_types)
|
|
@@ -949,12 +1024,20 @@ module Woods
|
|
|
949
1024
|
end
|
|
950
1025
|
if retriever
|
|
951
1026
|
begin
|
|
1027
|
+
scope_options = if Retrieval::Scope.requested?(packages: packages, source_paths: source_paths)
|
|
1028
|
+
{ packages: packages, source_paths: source_paths }
|
|
1029
|
+
else
|
|
1030
|
+
{}
|
|
1031
|
+
end
|
|
1032
|
+
scope_options[:evidence] = evidence unless evidence == 'full'
|
|
952
1033
|
result = retriever.retrieve(
|
|
953
1034
|
query,
|
|
954
|
-
budget: budget || 8000,
|
|
1035
|
+
budget: budget || (retriever.respond_to?(:default_budget) ? retriever.default_budget : 8000),
|
|
955
1036
|
types: types,
|
|
956
|
-
exclude_types: exclude_types
|
|
1037
|
+
exclude_types: exclude_types, **scope_options
|
|
957
1038
|
)
|
|
1039
|
+
rescue Retrieval::Scope::InvalidScopeError => e
|
|
1040
|
+
next respond_err.call(e.message, code: :unsupported_argument, tool: 'codebase_retrieve', argument: 'scope')
|
|
958
1041
|
rescue Woods::Retriever::StoreError => e
|
|
959
1042
|
# M8: a metadata-store failure mid-query must not surface as
|
|
960
1043
|
# a raw raise through the tool boundary (or as the misleading
|
|
@@ -963,7 +1046,7 @@ module Woods
|
|
|
963
1046
|
respond_err,
|
|
964
1047
|
reason: e.message,
|
|
965
1048
|
stores: [e.store],
|
|
966
|
-
phase: 'query'
|
|
1049
|
+
phase: 'query', mode: retrieval_mode
|
|
967
1050
|
)
|
|
968
1051
|
end
|
|
969
1052
|
if stale_check.call(result)
|
|
@@ -975,7 +1058,15 @@ module Woods
|
|
|
975
1058
|
tool: 'codebase_retrieve'
|
|
976
1059
|
)
|
|
977
1060
|
end
|
|
978
|
-
|
|
1061
|
+
if evidence != 'full' || (result.respond_to?(:applied_scope) && result.applied_scope)
|
|
1062
|
+
::MCP::Tool::Response.new(
|
|
1063
|
+
[{ type: 'text', text: result.context }],
|
|
1064
|
+
structured_content: { text: result.context, data: { applied_scope: result.applied_scope, sources: result.sources } },
|
|
1065
|
+
meta: { applied_scope: result.applied_scope }
|
|
1066
|
+
)
|
|
1067
|
+
else
|
|
1068
|
+
respond.call(result.context)
|
|
1069
|
+
end
|
|
979
1070
|
else
|
|
980
1071
|
respond_err.call(
|
|
981
1072
|
'Semantic search is disabled — no embedding provider is configured. ' \
|
|
@@ -1020,7 +1111,16 @@ module Woods
|
|
|
1020
1111
|
# @param phase [String] 'boot' (hydration failure) or 'query'
|
|
1021
1112
|
# (store failure at query time)
|
|
1022
1113
|
# @return [MCP::Tool::Response]
|
|
1023
|
-
def degraded_retrieval_response(respond_err, reason:, stores:, phase:)
|
|
1114
|
+
def degraded_retrieval_response(respond_err, reason:, stores:, phase:, mode: :semantic)
|
|
1115
|
+
if mode == :lexical
|
|
1116
|
+
return respond_err.call(
|
|
1117
|
+
"Lexical retrieval is degraded: #{reason}. No partial lexical snapshot was served. " \
|
|
1118
|
+
'Inspect woods_status and repair or re-extract the published index, then retry.',
|
|
1119
|
+
code: :degraded_index, tool: 'codebase_retrieve', degraded: true,
|
|
1120
|
+
phase: phase, stores: stores, reason: reason, mode: 'lexical'
|
|
1121
|
+
)
|
|
1122
|
+
end
|
|
1123
|
+
|
|
1024
1124
|
respond_err.call(
|
|
1025
1125
|
"Semantic search is degraded: #{reason}. The affected store(s) return no data, so " \
|
|
1026
1126
|
'queries would come back empty — this is NOT "no results". ' \
|
|
@@ -1129,6 +1229,9 @@ module Woods
|
|
|
1129
1229
|
)
|
|
1130
1230
|
doc = assembler.assemble(session_id, budget: budget || 8000, depth: depth || 1)
|
|
1131
1231
|
respond.call(doc.to_markdown)
|
|
1232
|
+
rescue Woods::SessionTracer::AmbiguousUnitError => e
|
|
1233
|
+
respond_err.call(e.message, code: :ambiguous_identity, tool: 'session_trace',
|
|
1234
|
+
identifier: e.identifier, types: e.types)
|
|
1132
1235
|
rescue StandardError => e
|
|
1133
1236
|
respond_err.call(
|
|
1134
1237
|
"Session trace failed: #{e.message}",
|
|
@@ -1979,13 +2082,17 @@ module Woods
|
|
|
1979
2082
|
description: 'Diagnose whether the Woods index and server are healthy. Returns extraction metadata ' \
|
|
1980
2083
|
'(last run, unit counts, git SHA, staleness in seconds), retriever/embedding configuration, ' \
|
|
1981
2084
|
'bootstrap state (hydrated / degraded / failed + reason), feature flags, and a ready flag. ' \
|
|
1982
|
-
'
|
|
1983
|
-
|
|
1984
|
-
|
|
2085
|
+
'Includes source-content freshness; quick scans have a 250ms budget, explicit deep scans have 5s. ' \
|
|
2086
|
+
'Incomplete evidence is unknown. Call this first on cold connect.',
|
|
2087
|
+
input_schema: { type: 'object', properties: {
|
|
2088
|
+
source_check: { type: 'string', enum: %w[quick deep], default: 'quick',
|
|
2089
|
+
description: 'Bounded source content verification: quick (250ms) or deep (5s).' }
|
|
2090
|
+
} }
|
|
2091
|
+
) do |server_context:, source_check: 'quick'|
|
|
1985
2092
|
_ = server_context
|
|
1986
2093
|
status = Woods::MCP::Server.build_status(
|
|
1987
2094
|
reader: reader, retriever: retriever, index_dir: index_dir,
|
|
1988
|
-
bootstrap_state: bootstrap_state
|
|
2095
|
+
bootstrap_state: bootstrap_state, source_check: source_check
|
|
1989
2096
|
)
|
|
1990
2097
|
respond.call(JSON.pretty_generate(status))
|
|
1991
2098
|
end
|
|
@@ -2004,21 +2111,21 @@ module Woods
|
|
|
2004
2111
|
# provider in use. Without this, operators debugging "wrong provider" see
|
|
2005
2112
|
# status claiming +embedding_model: "text-embedding-3-small"+ next to
|
|
2006
2113
|
# +embedding_provider: "ollama"+ and reasonably distrust every field.
|
|
2007
|
-
def build_status(reader:, retriever:, index_dir:, bootstrap_state: nil)
|
|
2114
|
+
def build_status(reader:, retriever:, index_dir:, bootstrap_state: nil, source_check: 'quick')
|
|
2008
2115
|
# Pin the generation across the whole payload. Without this the
|
|
2009
2116
|
# manifest can be read at generation N and `generation_fields` then
|
|
2010
2117
|
# report N+1 — a status report that describes counts from one index
|
|
2011
2118
|
# while announcing the number of another, which is precisely the
|
|
2012
2119
|
# confusion this tool exists to resolve.
|
|
2013
|
-
return build_status_payload(reader, retriever, index_dir, bootstrap_state) unless
|
|
2120
|
+
return build_status_payload(reader, retriever, index_dir, bootstrap_state, source_check) unless
|
|
2014
2121
|
reader.respond_to?(:with_pinned_generation)
|
|
2015
2122
|
|
|
2016
2123
|
reader.with_pinned_generation do
|
|
2017
|
-
build_status_payload(reader, retriever, index_dir, bootstrap_state)
|
|
2124
|
+
build_status_payload(reader, retriever, index_dir, bootstrap_state, source_check)
|
|
2018
2125
|
end
|
|
2019
2126
|
end
|
|
2020
2127
|
|
|
2021
|
-
def build_status_payload(reader, retriever, index_dir, bootstrap_state)
|
|
2128
|
+
def build_status_payload(reader, retriever, index_dir, bootstrap_state, source_check)
|
|
2022
2129
|
manifest = safe_manifest(reader)
|
|
2023
2130
|
extracted_at = manifest && manifest['extracted_at']
|
|
2024
2131
|
staleness = staleness_seconds(extracted_at)
|
|
@@ -2036,14 +2143,15 @@ module Woods
|
|
|
2036
2143
|
index_dir: index_dir.to_s,
|
|
2037
2144
|
update: Woods::UpdateCheck.status_hash
|
|
2038
2145
|
},
|
|
2039
|
-
index: index_section(manifest, extracted_at, staleness, index_dir, reader),
|
|
2146
|
+
index: index_section(manifest, extracted_at, staleness, index_dir, reader, source_check),
|
|
2040
2147
|
watch: watch_section(index_dir),
|
|
2041
2148
|
retriever: {
|
|
2042
2149
|
configured: !retriever.nil?,
|
|
2043
|
-
class: retriever&.class&.name
|
|
2150
|
+
class: retriever&.class&.name,
|
|
2151
|
+
**(retriever.respond_to?(:mode) && retriever.mode == :lexical ? { mode: 'lexical' } : {})
|
|
2044
2152
|
},
|
|
2045
2153
|
bootstrap: bootstrap_state&.to_h,
|
|
2046
|
-
features:
|
|
2154
|
+
features: retrieval_features(config, resolved, retriever)
|
|
2047
2155
|
}
|
|
2048
2156
|
end
|
|
2049
2157
|
|
|
@@ -2066,10 +2174,11 @@ module Woods
|
|
|
2066
2174
|
# diff directly. This is an observability signal, not a hard gate —
|
|
2067
2175
|
# hard-refusing responses would be much more disruptive than a loudly-
|
|
2068
2176
|
# visible staleness flag that agents can branch on.
|
|
2069
|
-
def index_section(manifest, extracted_at, staleness, index_dir, reader = nil)
|
|
2177
|
+
def index_section(manifest, extracted_at, staleness, index_dir, reader = nil, source_check = 'quick')
|
|
2070
2178
|
base = {
|
|
2071
2179
|
extracted_at: extracted_at,
|
|
2072
2180
|
staleness_seconds: staleness,
|
|
2181
|
+
woods_version: manifest && manifest['woods_version'],
|
|
2073
2182
|
rails_version: manifest && manifest['rails_version'],
|
|
2074
2183
|
ruby_version: manifest && manifest['ruby_version'],
|
|
2075
2184
|
total_units: manifest && manifest['total_units'],
|
|
@@ -2080,6 +2189,11 @@ module Woods
|
|
|
2080
2189
|
schema_sha: manifest && manifest['schema_sha']
|
|
2081
2190
|
}
|
|
2082
2191
|
|
|
2192
|
+
base[:source_freshness] = if reader.respond_to?(:source_freshness)
|
|
2193
|
+
reader.source_freshness(mode: source_check)
|
|
2194
|
+
else
|
|
2195
|
+
{ 'state' => 'unknown', 'reasons' => ['source_reader_unavailable'], 'complete' => false }
|
|
2196
|
+
end
|
|
2083
2197
|
base.merge!(generation_fields(index_dir, reader))
|
|
2084
2198
|
base.merge!(working_tree_fields(index_dir))
|
|
2085
2199
|
|
|
@@ -2210,7 +2324,7 @@ module Woods
|
|
|
2210
2324
|
record = JSON.parse(Woods::AtomicFile.read(path))
|
|
2211
2325
|
# `state` is whatever the daemon last wrote, and a `kill -9`'d daemon
|
|
2212
2326
|
# leaves `running` behind forever. `alive?` adds the two checks that
|
|
2213
|
-
# catch that —
|
|
2327
|
+
# catch that — a recent record and, for local hosts, a live pid — so the
|
|
2214
2328
|
# payload can distinguish "maintaining this index" from "claimed to be,
|
|
2215
2329
|
# once". Reported as a separate field rather than by overwriting
|
|
2216
2330
|
# `state`, because the recorded state and the liveness verdict answer
|
|
@@ -2263,6 +2377,13 @@ module Woods
|
|
|
2263
2377
|
# historic status payloads always reported +false+ regardless of the
|
|
2264
2378
|
# actual console MCP state. Advertising a misleading field is worse
|
|
2265
2379
|
# than not advertising it at all.
|
|
2380
|
+
def retrieval_features(config, resolved, retriever)
|
|
2381
|
+
features = features_from(config, resolved)
|
|
2382
|
+
return features unless retriever.respond_to?(:mode) && retriever.mode == :lexical
|
|
2383
|
+
|
|
2384
|
+
features.merge(retrieval_mode: 'lexical', embedding_model: nil, embedding_provider: nil, vector_store: nil)
|
|
2385
|
+
end
|
|
2386
|
+
|
|
2266
2387
|
def features_from(config, resolved)
|
|
2267
2388
|
provider_hash = resolved&.embedding_provider || {}
|
|
2268
2389
|
resolved_provider = resolved_provider_symbol(provider_hash[:class])
|
|
@@ -1,5 +1,8 @@
|
|
|
1
1
|
# frozen_string_literal: true
|
|
2
2
|
|
|
3
|
+
require_relative 'traversal_evidence_index'
|
|
4
|
+
require_relative 'traversal_evidence_text'
|
|
5
|
+
|
|
3
6
|
module Woods
|
|
4
7
|
module MCP
|
|
5
8
|
# Base class for rendering MCP tool responses in different output formats.
|
|
@@ -67,6 +70,28 @@ module Woods
|
|
|
67
70
|
|
|
68
71
|
private
|
|
69
72
|
|
|
73
|
+
def search_completeness_lines(data)
|
|
74
|
+
evidence = fetch_key(data, :completeness)
|
|
75
|
+
return [] unless evidence.is_a?(Hash)
|
|
76
|
+
|
|
77
|
+
more = { true => 'yes', false => 'no', nil => 'unknown' }.fetch(fetch_key(evidence, :has_more))
|
|
78
|
+
total = fetch_key(evidence, :total_matches)
|
|
79
|
+
lines = [
|
|
80
|
+
"Search completeness: #{fetch_key(evidence, :status)} (#{fetch_key(evidence, :reason)}).",
|
|
81
|
+
"More matches: #{more}; total matches: #{total.nil? ? 'unknown' : total}; " \
|
|
82
|
+
"matched lower bound: #{fetch_key(evidence, :matched_lower_bound)}."
|
|
83
|
+
]
|
|
84
|
+
scope = fetch_key(data, :applied_scope)
|
|
85
|
+
if scope
|
|
86
|
+
lines << "Applied scope: packages=#{fetch_key(scope, :packages).inspect}; " \
|
|
87
|
+
"source_paths=#{fetch_key(scope, :source_paths).inspect}; " \
|
|
88
|
+
"eligible units=#{fetch_key(scope, :eligible_units)}."
|
|
89
|
+
end
|
|
90
|
+
hint = fetch_key(data, :hint)
|
|
91
|
+
lines << hint if hint
|
|
92
|
+
lines
|
|
93
|
+
end
|
|
94
|
+
|
|
70
95
|
# Fetch a value from a hash by symbol or string key, falling back to a default.
|
|
71
96
|
#
|
|
72
97
|
# Handles data hashes that may use either symbol or string keys (e.g., data
|
|
@@ -0,0 +1,113 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require_relative 'traversal_evidence_index'
|
|
4
|
+
require_relative 'traversal_evidence_page'
|
|
5
|
+
|
|
6
|
+
module Woods
|
|
7
|
+
module MCP
|
|
8
|
+
# Per-query identifier-level BFS with truthful typed records and a shared
|
|
9
|
+
# predecessor forest. Ambiguous endpoints never become a claimed typed path.
|
|
10
|
+
class TraversalEvidence
|
|
11
|
+
class Budget
|
|
12
|
+
attr_reader :data
|
|
13
|
+
|
|
14
|
+
def initialize(max_nodes, max_edges)
|
|
15
|
+
@data = { max_nodes: max_nodes.to_i.clamp(1, 10_000), max_edges: max_edges.to_i.clamp(1, 100_000),
|
|
16
|
+
visited_nodes: 1, visited_edges: 0 }
|
|
17
|
+
end
|
|
18
|
+
|
|
19
|
+
def consume_node
|
|
20
|
+
throw :traversal_budget, 'node_budget' if data[:visited_nodes] >= data[:max_nodes]
|
|
21
|
+
|
|
22
|
+
data[:visited_nodes] += 1
|
|
23
|
+
end
|
|
24
|
+
|
|
25
|
+
def consume_edge
|
|
26
|
+
throw :traversal_budget, 'edge_budget' if data[:visited_edges] >= data[:max_edges]
|
|
27
|
+
|
|
28
|
+
data[:visited_edges] += 1
|
|
29
|
+
end
|
|
30
|
+
end
|
|
31
|
+
|
|
32
|
+
def initialize(index)
|
|
33
|
+
@index = index
|
|
34
|
+
end
|
|
35
|
+
|
|
36
|
+
def call(identifier, depth: 2, direction: :forward, types: nil, via: nil, max_nodes: 1000, max_edges: 10_000)
|
|
37
|
+
return { root: identifier, found: false, nodes: {} } unless @index.include?(identifier)
|
|
38
|
+
|
|
39
|
+
prepare(identifier, depth, direction, types, via, max_nodes, max_edges)
|
|
40
|
+
cursor = 0
|
|
41
|
+
while cursor < @queue.size
|
|
42
|
+
current, level = @queue[cursor]
|
|
43
|
+
cursor += 1
|
|
44
|
+
entry = @index.node(current, level)
|
|
45
|
+
@nodes[current] = entry
|
|
46
|
+
next if @partial_reason || level >= @depth
|
|
47
|
+
|
|
48
|
+
@partial_reason = catch(:traversal_budget) do
|
|
49
|
+
@index.each_edge(current, @direction, @budget) { |edge| visit(current, level, entry, edge) }
|
|
50
|
+
nil
|
|
51
|
+
end
|
|
52
|
+
end
|
|
53
|
+
response(identifier)
|
|
54
|
+
end
|
|
55
|
+
|
|
56
|
+
private
|
|
57
|
+
|
|
58
|
+
def prepare(identifier, depth, direction, types, via, max_nodes, max_edges)
|
|
59
|
+
@depth = depth
|
|
60
|
+
@direction = direction
|
|
61
|
+
@types = types&.to_set
|
|
62
|
+
@via = via&.to_set
|
|
63
|
+
@budget = Budget.new(max_nodes, max_edges)
|
|
64
|
+
@queue = [[identifier, 0]]
|
|
65
|
+
@nodes = {}
|
|
66
|
+
@neighbors = Hash.new { |hash, key| hash[key] = Set.new }
|
|
67
|
+
@edges = {}
|
|
68
|
+
@edge_ids = {}
|
|
69
|
+
@partial_reason = nil
|
|
70
|
+
@witnesses = { identifier => { parent: nil, edge_id: nil, impact: 'root',
|
|
71
|
+
typed_path_complete: @index.types(identifier).size == 1 } }
|
|
72
|
+
end
|
|
73
|
+
|
|
74
|
+
def visit(current, level, entry, edge)
|
|
75
|
+
return if @via && !@via.include?(edge[:via])
|
|
76
|
+
|
|
77
|
+
neighbor = edge.fetch(@direction == :forward ? :target : :source).fetch(:identifier)
|
|
78
|
+
return if @types && @index.types(neighbor).none? { |type| @types.include?(type) }
|
|
79
|
+
|
|
80
|
+
unless @witnesses.key?(neighbor)
|
|
81
|
+
@budget.consume_node
|
|
82
|
+
@queue << [neighbor, level + 1]
|
|
83
|
+
@witnesses[neighbor] = witness(current, level, edge)
|
|
84
|
+
end
|
|
85
|
+
edge_id(edge)
|
|
86
|
+
entry[:deps] << neighbor if @neighbors[current].add?(neighbor)
|
|
87
|
+
end
|
|
88
|
+
|
|
89
|
+
def witness(current, level, edge)
|
|
90
|
+
complete = @witnesses.fetch(current)[:typed_path_complete] && !edge[:target][:type].nil? &&
|
|
91
|
+
@index.types(edge[:source][:identifier]).size == 1
|
|
92
|
+
{ parent: current, edge_id: edge_id(edge), impact: level.zero? ? 'direct' : 'transitive',
|
|
93
|
+
typed_path_complete: complete }
|
|
94
|
+
end
|
|
95
|
+
|
|
96
|
+
def edge_id(edge)
|
|
97
|
+
@edge_ids[edge] ||= begin
|
|
98
|
+
id = "e#{@edges.size}"
|
|
99
|
+
@edges[id] = edge
|
|
100
|
+
id
|
|
101
|
+
end
|
|
102
|
+
end
|
|
103
|
+
|
|
104
|
+
def response(identifier)
|
|
105
|
+
result = { root: identifier, found: true, nodes: @nodes,
|
|
106
|
+
explanation: { direction: @direction.to_s, root: @index.identity(identifier),
|
|
107
|
+
edges: @edges, witnesses: @witnesses } }
|
|
108
|
+
result.merge!(partial: true, partial_reason: @partial_reason, traversal_budget: @budget.data) if @partial_reason
|
|
109
|
+
result
|
|
110
|
+
end
|
|
111
|
+
end
|
|
112
|
+
end
|
|
113
|
+
end
|