woods 2.0.0.beta2 → 2.0.0.beta4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +339 -1
- data/CONTRIBUTING.md +188 -12
- data/README.md +93 -174
- data/SECURITY.md +9 -6
- data/docs/AGENT_GUIDE.md +109 -8
- data/docs/AGENT_SETUP.md +98 -7
- data/docs/BACKEND_MATRIX.md +25 -0
- data/docs/CLIENT_HOOKS.md +111 -0
- data/docs/CONFIGURATION_REFERENCE.md +267 -16
- data/docs/CONSOLE_MCP_SETUP.md +80 -7
- data/docs/DOCKER_SETUP.md +22 -3
- data/docs/EVALUATION.md +464 -1
- data/docs/EXTRACTOR_REFERENCE.md +45 -6
- data/docs/FAQ.md +11 -12
- data/docs/GETTING_STARTED.md +17 -5
- data/docs/INCREMENTAL_EXTRACTION.md +147 -7
- data/docs/INDEX_LAYOUT.md +382 -0
- data/docs/INTERNALS.md +7 -2
- data/docs/MCP_SERVERS.md +276 -5
- data/docs/MCP_TOOL_COOKBOOK.md +37 -22
- data/docs/MCP_WORKTREE_SETUP.md +43 -83
- data/docs/NOTION_INTEGRATION.md +13 -0
- data/docs/OBSIDIAN_INTEGRATION.md +57 -9
- data/docs/PUBLISHED_INDEX.md +72 -0
- data/docs/README.md +7 -0
- data/docs/RETRIEVAL_GUIDE.md +273 -12
- data/docs/RUNTIME_TRACING.md +71 -0
- data/docs/SOURCE_FRESHNESS.md +143 -0
- data/docs/TROUBLESHOOTING.md +129 -18
- data/docs/UNBLOCKED_INTEGRATION.md +25 -0
- data/docs/UPGRADING_TO_2.md +48 -22
- data/docs/WATCH_DAEMON.md +277 -67
- data/exe/woods-agent-config +6 -0
- data/exe/woods-extract +5 -0
- data/exe/woods-hook-context +6 -0
- data/exe/woods-mcp-start +14 -9
- data/lib/generators/woods/pgvector_generator.rb +8 -2
- data/lib/generators/woods/templates/woods.rb.tt +1 -3
- data/lib/tasks/woods.rake +47 -397
- data/lib/woods/agent_configuration/applier.rb +135 -0
- data/lib/woods/agent_configuration/cli.rb +101 -0
- data/lib/woods/agent_configuration/cli_options.rb +29 -0
- data/lib/woods/agent_configuration/document.rb +105 -0
- data/lib/woods/agent_configuration/error.rb +7 -0
- data/lib/woods/agent_configuration/launcher.rb +75 -0
- data/lib/woods/agent_configuration/layout.rb +72 -0
- data/lib/woods/agent_configuration/managed_section.rb +62 -0
- data/lib/woods/agent_configuration/plan.rb +98 -0
- data/lib/woods/agent_configuration/plan_diff.rb +38 -0
- data/lib/woods/agent_configuration/planned_files.rb +61 -0
- data/lib/woods/agent_configuration/planner.rb +63 -0
- data/lib/woods/agent_configuration/planner_validation.rb +77 -0
- data/lib/woods/agent_configuration/preflight.rb +100 -0
- data/lib/woods/agent_configuration/recovery.rb +49 -0
- data/lib/woods/ast/node.rb +2 -0
- data/lib/woods/ast/parser.rb +38 -5
- data/lib/woods/builder.rb +21 -5
- data/lib/woods/cache/cache_middleware.rb +28 -7
- data/lib/woods/cache/cache_store.rb +4 -5
- data/lib/woods/change_set.rb +5 -4
- data/lib/woods/console/credential_index.rb +20 -2
- data/lib/woods/console/credential_scanner.rb +18 -17
- data/lib/woods/console/credential_scanner_registry.rb +36 -0
- data/lib/woods/console/dispatch_pipeline.rb +7 -0
- data/lib/woods/console/embedded_executor.rb +32 -10
- data/lib/woods/console/encrypted_credential_snapshot.rb +16 -0
- data/lib/woods/console/rack_middleware.rb +22 -13
- data/lib/woods/console/server.rb +18 -16
- data/lib/woods/console/sql_noise_stripper.rb +9 -7
- data/lib/woods/console/sql_table_scanner.rb +47 -7
- data/lib/woods/console/sql_validator.rb +49 -9
- data/lib/woods/console/sqlite_read_guard.rb +46 -0
- data/lib/woods/coordination/pipeline_lock.rb +3 -2
- data/lib/woods/dependency_graph.rb +65 -13
- data/lib/woods/embedding/corpus.rb +94 -0
- data/lib/woods/embedding/indexer.rb +114 -60
- data/lib/woods/embedding/openai.rb +17 -6
- data/lib/woods/evaluation/ablation_executor.rb +6 -1
- data/lib/woods/evaluation/ablation_timed_executor.rb +22 -4
- data/lib/woods/export/typed_reader.rb +56 -0
- data/lib/woods/extractor.rb +277 -149
- data/lib/woods/extractors/action_cable_extractor.rb +3 -1
- data/lib/woods/extractors/behavioral_profile.rb +9 -7
- data/lib/woods/extractors/caching_extractor.rb +3 -1
- data/lib/woods/extractors/concern_extractor.rb +64 -6
- data/lib/woods/extractors/configuration_extractor.rb +7 -3
- data/lib/woods/extractors/controller_extractor.rb +13 -4
- data/lib/woods/extractors/database_view_extractor.rb +3 -1
- data/lib/woods/extractors/declared_parent.rb +55 -0
- data/lib/woods/extractors/decorator_extractor.rb +3 -1
- data/lib/woods/extractors/engine_extractor.rb +3 -1
- data/lib/woods/extractors/event_extractor.rb +4 -2
- data/lib/woods/extractors/factory_extractor.rb +3 -1
- data/lib/woods/extractors/graphql_extractor.rb +10 -13
- data/lib/woods/extractors/i18n_extractor.rb +3 -1
- data/lib/woods/extractors/job_extractor.rb +6 -19
- data/lib/woods/extractors/lib_extractor.rb +13 -9
- data/lib/woods/extractors/mailer_extractor.rb +26 -15
- data/lib/woods/extractors/manager_extractor.rb +3 -1
- data/lib/woods/extractors/method_parameters.rb +53 -0
- data/lib/woods/extractors/middleware_argument.rb +65 -0
- data/lib/woods/extractors/middleware_extractor.rb +9 -3
- data/lib/woods/extractors/migration_extractor.rb +3 -1
- data/lib/woods/extractors/model_extractor.rb +26 -34
- data/lib/woods/extractors/package_extractor.rb +24 -4
- data/lib/woods/extractors/phlex_extractor.rb +3 -1
- data/lib/woods/extractors/policy_extractor.rb +3 -1
- data/lib/woods/extractors/poro_extractor.rb +13 -9
- data/lib/woods/extractors/pundit_extractor.rb +3 -1
- data/lib/woods/extractors/rails_source_extractor.rb +4 -2
- data/lib/woods/extractors/rake_task_extractor.rb +4 -2
- data/lib/woods/extractors/route_extractor.rb +3 -1
- data/lib/woods/extractors/route_helper_resolver.rb +10 -33
- data/lib/woods/extractors/scheduled_job_extractor.rb +41 -15
- data/lib/woods/extractors/serializer_extractor.rb +4 -2
- data/lib/woods/extractors/service_extractor.rb +3 -1
- data/lib/woods/extractors/shared_dependency_scanner.rb +2 -2
- data/lib/woods/extractors/shared_utility_methods.rb +48 -19
- data/lib/woods/extractors/source_nesting.rb +1 -1
- data/lib/woods/extractors/state_machine_extractor.rb +3 -1
- data/lib/woods/extractors/test_mapping_extractor.rb +3 -1
- data/lib/woods/extractors/validator_extractor.rb +3 -1
- data/lib/woods/extractors/view_component_extractor.rb +3 -1
- data/lib/woods/extractors/view_template_extractor.rb +3 -1
- data/lib/woods/gem_mapper.rb +2 -0
- data/lib/woods/git_history.rb +116 -0
- data/lib/woods/graph_analyzer.rb +35 -6
- data/lib/woods/hooks/context_cli.rb +54 -0
- data/lib/woods/hooks/context_event.rb +88 -0
- data/lib/woods/hooks/context_hint.rb +73 -0
- data/lib/woods/hooks/context_impact.rb +77 -0
- data/lib/woods/hooks/context_output.rb +47 -0
- data/lib/woods/hooks/context_state.rb +102 -0
- data/lib/woods/hooks/refresh.rb +79 -0
- data/lib/woods/hooks/rule_projection.rb +78 -0
- data/lib/woods/input_rules.rb +19 -0
- data/lib/woods/mcp/bearer_auth.rb +22 -13
- data/lib/woods/mcp/bootstrapper.rb +79 -4
- data/lib/woods/mcp/config_resolver.rb +2 -1
- data/lib/woods/mcp/index_reader.rb +334 -162
- data/lib/woods/mcp/initialization_guidance.rb +27 -0
- data/lib/woods/mcp/origin_guard.rb +17 -9
- data/lib/woods/mcp/published_lexical_retriever.rb +115 -0
- data/lib/woods/mcp/renderers/markdown_renderer.rb +22 -9
- data/lib/woods/mcp/renderers/plain_renderer.rb +18 -8
- data/lib/woods/mcp/search_results.rb +74 -0
- data/lib/woods/mcp/server.rb +178 -63
- data/lib/woods/mcp/tool_contract.rb +3 -1
- data/lib/woods/mcp/tool_response_renderer.rb +41 -0
- data/lib/woods/mcp/traversal_evidence.rb +113 -0
- data/lib/woods/mcp/traversal_evidence_index.rb +100 -0
- data/lib/woods/mcp/traversal_evidence_page.rb +41 -0
- data/lib/woods/mcp/traversal_evidence_text.rb +52 -0
- data/lib/woods/mcp/traversal_response.rb +22 -0
- data/lib/woods/notion/exporter.rb +56 -17
- data/lib/woods/obsidian/destination_plan.rb +98 -0
- data/lib/woods/obsidian/name_mapper.rb +19 -3
- data/lib/woods/obsidian/note_builder.rb +19 -10
- data/lib/woods/obsidian/vault_exporter.rb +88 -32
- data/lib/woods/operator/pipeline_guard.rb +18 -13
- data/lib/woods/path_dispatcher.rb +13 -6
- data/lib/woods/payload_store.rb +27 -26
- data/lib/woods/published_index/typed_unit_reader.rb +40 -3
- data/lib/woods/published_index.rb +2 -2
- data/lib/woods/railtie.rb +3 -3
- data/lib/woods/railtie_support.rb +12 -12
- data/lib/woods/rake_helpers.rb +382 -0
- data/lib/woods/resilience/graph_invariant_validator/membership_checks.rb +71 -0
- data/lib/woods/resilience/graph_invariant_validator/node_checks.rb +61 -0
- data/lib/woods/resilience/graph_invariant_validator/reverse_relationship_checks.rb +46 -0
- data/lib/woods/resilience/graph_invariant_validator.rb +119 -0
- data/lib/woods/resilience/index_validator/graph_checks.rb +80 -0
- data/lib/woods/resilience/index_validator.rb +112 -23
- data/lib/woods/retrieval/context_assembler.rb +50 -15
- data/lib/woods/retrieval/lexical_assembler.rb +84 -0
- data/lib/woods/retrieval/lexical_index.rb +120 -0
- data/lib/woods/retrieval/ranker.rb +4 -2
- data/lib/woods/retrieval/scope.rb +108 -0
- data/lib/woods/retrieval/scoped_graph_store.rb +32 -0
- data/lib/woods/retrieval/scoped_vector_store.rb +55 -0
- data/lib/woods/retrieval/search_executor.rb +86 -27
- data/lib/woods/retrieval/source_evidence.rb +200 -0
- data/lib/woods/retriever.rb +98 -22
- data/lib/woods/ruby_analyzer/trace_enricher.rb +77 -38
- data/lib/woods/session_tracer/file_store.rb +6 -1
- data/lib/woods/session_tracer/middleware.rb +10 -12
- data/lib/woods/session_tracer/redis_store.rb +22 -6
- data/lib/woods/session_tracer/session_flow_assembler.rb +23 -17
- data/lib/woods/session_tracer/solid_cache_coordination.rb +6 -4
- data/lib/woods/session_tracer/unit_resolver.rb +63 -0
- data/lib/woods/source_inputs/consumer_errors.rb +31 -0
- data/lib/woods/source_inputs/handoff.rb +102 -0
- data/lib/woods/source_inputs/launcher.rb +157 -0
- data/lib/woods/source_inputs/manifest.rb +124 -0
- data/lib/woods/source_inputs/private_key.rb +55 -0
- data/lib/woods/source_inputs/scanner.rb +171 -0
- data/lib/woods/source_inputs/scopes.rb +71 -0
- data/lib/woods/source_inputs/session.rb +214 -0
- data/lib/woods/source_inputs/status.rb +84 -0
- data/lib/woods/source_inputs/verifier.rb +107 -0
- data/lib/woods/storage/metadata_store.rb +25 -25
- data/lib/woods/storage/pgvector.rb +35 -10
- data/lib/woods/storage/qdrant.rb +17 -7
- data/lib/woods/storage/vector_store.rb +18 -6
- data/lib/woods/tasks.rb +3 -2
- data/lib/woods/temporal/json_snapshot_store.rb +58 -9
- data/lib/woods/unblocked/exporter.rb +59 -70
- data/lib/woods/version.rb +1 -1
- data/lib/woods/watch/boot_snapshot.rb +52 -0
- data/lib/woods/watch/daemon.rb +154 -32
- data/lib/woods/watch/listen_watcher.rb +4 -0
- data/lib/woods/watch/polling_watcher.rb +5 -1
- data/lib/woods/watch/status.rb +20 -15
- data/lib/woods/watch/tree_scan.rb +21 -13
- data/lib/woods/watch/watcher.rb +4 -1
- data/lib/woods.rb +50 -11
- data/plugin/.claude-plugin/plugin.json +1 -1
- data/plugin/hooks/adapters/normalize.jq +15 -0
- data/plugin/hooks/adapters/normalize.rb +63 -0
- data/plugin/hooks/hooks.json +20 -0
- data/plugin/hooks/woods-context.sh +50 -0
- data/plugin/hooks/woods-input-rules.sh +159 -0
- data/plugin/hooks/woods-opencode.mjs +65 -0
- data/plugin/hooks/woods-post-edit.sh +2 -225
- data/plugin/hooks/woods-refresh.sh +260 -0
- data/plugin/hooks/woods-session-start.sh +47 -55
- data/plugin/skills/woods-agent-enable/SKILL.md +19 -0
- data/plugin/skills/woods-diagnose/SKILL.md +319 -1
- data/plugin/skills/woods-investigate/SKILL.md +145 -0
- data/plugin/skills/woods-mcp-config/SKILL.md +90 -2
- data/plugin/skills/woods-setup/SKILL.md +110 -6
- metadata +87 -5
data/lib/woods/mcp/server.rb
CHANGED
|
@@ -14,16 +14,20 @@ require_relative '../tasks'
|
|
|
14
14
|
require_relative '../watch/status'
|
|
15
15
|
require_relative '../filename_utils'
|
|
16
16
|
require_relative '../update_check'
|
|
17
|
+
require_relative '../retrieval/source_evidence'
|
|
18
|
+
require_relative '../session_tracer/unit_resolver'
|
|
17
19
|
require_relative 'bootstrap_state'
|
|
18
20
|
require_relative 'errors'
|
|
19
21
|
require_relative 'index_reader'
|
|
20
22
|
require_relative 'index_reader_pinning'
|
|
23
|
+
require_relative 'initialization_guidance'
|
|
21
24
|
require_relative 'protocol_policy'
|
|
22
25
|
require_relative 'tasks/extension'
|
|
23
26
|
require_relative 'tasks/request_capture'
|
|
24
27
|
require_relative 'tasks/store'
|
|
25
28
|
require_relative 'tool_contract'
|
|
26
29
|
require_relative 'tool_response_renderer'
|
|
30
|
+
require_relative 'traversal_response'
|
|
27
31
|
require_relative 'version_aware_tool_dispatch'
|
|
28
32
|
|
|
29
33
|
module Woods
|
|
@@ -70,6 +74,7 @@ module Woods
|
|
|
70
74
|
# controls that actually shrink the answer are `depth`, `types` and
|
|
71
75
|
# `via`; `limit` and `offset` only page what those leave (B-183).
|
|
72
76
|
DEFAULT_TRAVERSAL_LIMIT = 50
|
|
77
|
+
DEFAULT_GRAPH_ANALYSIS_LIMIT = 20
|
|
73
78
|
|
|
74
79
|
class << self
|
|
75
80
|
# Build a configured MCP::Server with all tools and resources.
|
|
@@ -90,6 +95,7 @@ module Woods
|
|
|
90
95
|
def build(index_dir:, retriever: nil, operator: nil, feedback_store: nil, snapshot_store: nil,
|
|
91
96
|
bootstrap_state: nil, response_format: nil, warmup: true, retriever_reloader: nil)
|
|
92
97
|
reader = IndexReader.new(index_dir)
|
|
98
|
+
retriever.bind_reader(reader) if retriever.respond_to?(:bind_reader)
|
|
93
99
|
reader.warmup! if warmup
|
|
94
100
|
config = Woods.configuration
|
|
95
101
|
format = response_format || (config.respond_to?(:context_format) ? config.context_format : nil) || :markdown
|
|
@@ -156,7 +162,10 @@ module Woods
|
|
|
156
162
|
description: 'Traverse forward dependencies of a unit (what it depends on). ' \
|
|
157
163
|
'Narrow with depth, types and via first: they shrink the answer, ' \
|
|
158
164
|
'while limit and offset only page it. Returns a BFS tree with ' \
|
|
159
|
-
"depth,
|
|
165
|
+
"depth, paged to #{DEFAULT_TRAVERSAL_LIMIT} nodes by default. " \
|
|
166
|
+
'max_nodes/max_edges bound the walk independently; partial_reason reports a budget cutoff and total_is_exact is false. ' \
|
|
167
|
+
'Published relationships are not exhaustive source-reference coverage. ' \
|
|
168
|
+
'Use explain:true for recorded directed relationships and bounded witnesses; ambiguous types remain explicit.',
|
|
160
169
|
reader_method: :traverse_dependencies,
|
|
161
170
|
render_key: :dependencies)
|
|
162
171
|
define_traversal_tool(server, reader, respond, renderer,
|
|
@@ -164,7 +173,10 @@ module Woods
|
|
|
164
173
|
description: 'Traverse reverse dependencies of a unit (what depends on it). ' \
|
|
165
174
|
'Narrow with depth, types and via first: they shrink the answer, ' \
|
|
166
175
|
'while limit and offset only page it. Returns a BFS tree with ' \
|
|
167
|
-
"depth,
|
|
176
|
+
"depth, paged to #{DEFAULT_TRAVERSAL_LIMIT} nodes by default. " \
|
|
177
|
+
'max_nodes/max_edges bound the walk independently; partial_reason reports a budget cutoff and total_is_exact is false. ' \
|
|
178
|
+
'Published relationships are not exhaustive source-reference coverage. ' \
|
|
179
|
+
'Use explain:true for recorded directed relationships and bounded witnesses; ambiguous types remain explicit.',
|
|
168
180
|
reader_method: :traverse_dependents,
|
|
169
181
|
render_key: :dependents)
|
|
170
182
|
define_structure_tool(server, reader, respond, renderer)
|
|
@@ -191,6 +203,7 @@ module Woods
|
|
|
191
203
|
register_resource_handler(server, reader)
|
|
192
204
|
ToolContract.apply!(server)
|
|
193
205
|
IndexReaderPinning.install(server, reader: reader)
|
|
206
|
+
server.instructions = InitializationGuidance.for(server.tools.keys)
|
|
194
207
|
|
|
195
208
|
# Last, after every conditional registration above — the whole point is
|
|
196
209
|
# that a host with Notion wired advertises the same tool order as one
|
|
@@ -238,9 +251,9 @@ module Woods
|
|
|
238
251
|
!token.nil? && ids && !ids.empty?
|
|
239
252
|
end
|
|
240
253
|
|
|
241
|
-
def text_response(text)
|
|
254
|
+
def text_response(text, data: nil)
|
|
242
255
|
structured = { text: text }
|
|
243
|
-
structured[:data] = JSON.parse(text)
|
|
256
|
+
structured[:data] = data.nil? ? JSON.parse(text) : data
|
|
244
257
|
::MCP::Tool::Response.new(
|
|
245
258
|
[{ type: 'text', text: text }],
|
|
246
259
|
structured_content: structured
|
|
@@ -384,7 +397,7 @@ module Woods
|
|
|
384
397
|
|
|
385
398
|
sliced = offset.positive? ? original.drop(offset) : original
|
|
386
399
|
container[key] = limit ? truncate_section(sliced, limit) : sliced
|
|
387
|
-
if
|
|
400
|
+
if offset.positive? || container[key].size < original.size
|
|
388
401
|
container["#{key}_total"] = original.size
|
|
389
402
|
container["#{key}_truncated"] = true
|
|
390
403
|
end
|
|
@@ -394,12 +407,11 @@ module Woods
|
|
|
394
407
|
# Page a traversal result's `nodes` hash in place, in BFS order.
|
|
395
408
|
#
|
|
396
409
|
# Mirrors {#paginate_section}'s metadata keys (`nodes_total`,
|
|
397
|
-
# `nodes_truncated`, `nodes_offset`)
|
|
398
|
-
#
|
|
399
|
-
#
|
|
400
|
-
# exactly as it did before the bound existed (B-183).
|
|
410
|
+
# `nodes_truncated`, `nodes_offset`). A page that holds every admitted
|
|
411
|
+
# node adds no pagination keys (B-183). TraversalResponse separately
|
|
412
|
+
# annotates scope and total exactness before pagination.
|
|
401
413
|
#
|
|
402
|
-
# `nodes_total` marks *any*
|
|
414
|
+
# `nodes_total` marks *any* paged answer, not only one with more
|
|
403
415
|
# behind it. Keying it on `total > offset + limit` left the last page
|
|
404
416
|
# of a walk indistinguishable from a complete one: 21 nodes of 121,
|
|
405
417
|
# with nothing saying 100 were skipped. `nodes_truncated` still means
|
|
@@ -435,6 +447,11 @@ module Woods
|
|
|
435
447
|
identifier: { type: 'string',
|
|
436
448
|
description: 'Exact unit identifier (e.g. "Post", "PostsController", "Api::V1::HealthController")' },
|
|
437
449
|
name: { type: 'string', description: 'Alias for `identifier`. Either one works.' },
|
|
450
|
+
type: { type: 'string', description: 'Optional actual published unit type, to disambiguate shared identifiers.' },
|
|
451
|
+
evidence: { type: 'string', enum: %w[full compact outline], description: 'Published evidence mode (default full).' },
|
|
452
|
+
query: { type: 'string', description: 'Optional relevance query for compact evidence; absent means API orientation.' },
|
|
453
|
+
budget: { type: 'integer', minimum: 1, description: 'Compact/outline token estimate budget (default 2000); full lookup remains complete.' },
|
|
454
|
+
source_sha256: { type: 'string', description: 'Require the published source SHA256 from an earlier excerpt; refuses changed source.' },
|
|
438
455
|
include_source: { type: 'boolean', description: 'Include source_code in response (default: true)' },
|
|
439
456
|
sections: {
|
|
440
457
|
type: 'array', items: { type: 'string' },
|
|
@@ -445,7 +462,7 @@ module Woods
|
|
|
445
462
|
# accepted alias. The handler validates that one of the two
|
|
446
463
|
# was provided.
|
|
447
464
|
}
|
|
448
|
-
) do |server_context:, identifier: nil, name: nil, include_source: nil, sections: nil|
|
|
465
|
+
) do |server_context:, identifier: nil, name: nil, include_source: nil, sections: nil, type: nil, evidence: 'full', query: nil, budget: nil, source_sha256: nil|
|
|
449
466
|
identifier ||= name
|
|
450
467
|
if identifier.nil? || identifier.empty?
|
|
451
468
|
next respond_err.call(
|
|
@@ -457,8 +474,30 @@ module Woods
|
|
|
457
474
|
)
|
|
458
475
|
end
|
|
459
476
|
sections = coerce.call(sections)
|
|
460
|
-
|
|
477
|
+
begin
|
|
478
|
+
Retrieval::SourceEvidence.validate_mode!(evidence)
|
|
479
|
+
if evidence != 'full' && (include_source == false || sections&.any?)
|
|
480
|
+
raise ArgumentError, 'compact/outline evidence cannot be combined with include_source: false or sections'
|
|
481
|
+
end
|
|
482
|
+
if evidence == 'full' && (!query.nil? || !budget.nil?)
|
|
483
|
+
raise ArgumentError, 'query and budget apply only to compact/outline evidence'
|
|
484
|
+
end
|
|
485
|
+
rescue ArgumentError => e
|
|
486
|
+
next respond_err.call(e.message, code: :unsupported_argument, tool: 'lookup', argument: 'evidence')
|
|
487
|
+
end
|
|
488
|
+
unit = type ? reader.find_unit(identifier, type: type) : reader.find_unit(identifier)
|
|
461
489
|
if unit
|
|
490
|
+
if source_sha256 && Digest::SHA256.hexdigest(unit['source_code'].to_s) != source_sha256
|
|
491
|
+
next respond_err.call('Published source changed since the excerpt; retrieve fresh evidence before verification.',
|
|
492
|
+
code: :stale_index, tool: 'lookup', argument: 'source_sha256')
|
|
493
|
+
end
|
|
494
|
+
if evidence != 'full'
|
|
495
|
+
selected = Retrieval::SourceEvidence.new(unit: unit, query: query, generation: reader.loaded_generation)
|
|
496
|
+
.render(mode: evidence, budget: budget || 2000,
|
|
497
|
+
counter: ->(text) { (text.length / 4.0).ceil })
|
|
498
|
+
next ::MCP::Tool::Response.new([{ type: 'text', text: selected.text }],
|
|
499
|
+
structured_content: { text: selected.text, data: { evidence: selected.provenance } })
|
|
500
|
+
end
|
|
462
501
|
always_include = %w[type identifier file_path namespace]
|
|
463
502
|
filtered = unit
|
|
464
503
|
filtered = filtered.except('source_code') if include_source == false
|
|
@@ -485,8 +524,9 @@ module Woods
|
|
|
485
524
|
server.define_tool(
|
|
486
525
|
name: 'search',
|
|
487
526
|
description: 'Find code units whose identifiers (or source/metadata) match a regex. ' \
|
|
488
|
-
'Example: search("Worker|Job")
|
|
489
|
-
'returns units starting with "Post". Returns [{identifier, type, match_field}]. ' \
|
|
527
|
+
'Example: search("Worker|Job") finds workers and jobs; search("^Post") ' \
|
|
528
|
+
'returns units starting with "Post". Returns [{identifier, type, match_field}] plus completeness. ' \
|
|
529
|
+
'Check completeness before treating discovery as exhaustive; a limit is only a page size. ' \
|
|
490
530
|
'Use `lookup` for exact identifiers, `dependencies`/`dependents` for graph traversal. ' \
|
|
491
531
|
'Gotchas: query is a Ruby regex — literal pipe needs escaping as \\|; ' \
|
|
492
532
|
'types restricts which index directories are scanned (e.g. ["mailer"] scans only ' \
|
|
@@ -500,6 +540,10 @@ module Woods
|
|
|
500
540
|
type: 'array', items: { type: 'string' },
|
|
501
541
|
description: 'Restrict scan to these unit types: model, controller, service, job, mailer, etc.'
|
|
502
542
|
},
|
|
543
|
+
packages: { type: 'array', items: { type: 'string' },
|
|
544
|
+
description: 'Exact published package owners, OR within the list; AND with source_paths and types.' },
|
|
545
|
+
source_paths: { type: 'array', items: { type: 'string' },
|
|
546
|
+
description: 'Application-relative directory prefixes; segment-aware, OR within the list. Applied before limits.' },
|
|
503
547
|
fields: {
|
|
504
548
|
type: 'array', items: { type: 'string', enum: %w[identifier metadata source_code] },
|
|
505
549
|
description: 'Fields to search: identifier (default), source_code, metadata'
|
|
@@ -517,7 +561,8 @@ module Woods
|
|
|
517
561
|
}
|
|
518
562
|
}
|
|
519
563
|
}
|
|
520
|
-
) do |server_context:, query: nil, types: nil, fields: nil, limit: nil, exact_prefix: nil, exact_suffix: nil
|
|
564
|
+
) do |server_context:, query: nil, types: nil, fields: nil, limit: nil, exact_prefix: nil, exact_suffix: nil,
|
|
565
|
+
packages: nil, source_paths: nil|
|
|
521
566
|
if (query.nil? || query.empty?) &&
|
|
522
567
|
(exact_prefix.nil? || exact_prefix.empty?) &&
|
|
523
568
|
(exact_suffix.nil? || exact_suffix.empty?)
|
|
@@ -538,17 +583,29 @@ module Woods
|
|
|
538
583
|
fields: fields || %w[identifier],
|
|
539
584
|
limit: limit || 20,
|
|
540
585
|
exact_prefix: exact_prefix,
|
|
541
|
-
exact_suffix: exact_suffix
|
|
586
|
+
exact_suffix: exact_suffix,
|
|
587
|
+
packages: packages, source_paths: source_paths
|
|
542
588
|
)
|
|
543
589
|
results = search_result[:results]
|
|
544
590
|
payload = {
|
|
545
591
|
query: query,
|
|
546
592
|
result_count: results.size,
|
|
547
|
-
results: results
|
|
593
|
+
results: results,
|
|
594
|
+
completeness: search_result[:completeness]
|
|
548
595
|
}
|
|
596
|
+
payload[:applied_scope] = search_result[:applied_scope] if search_result[:applied_scope]
|
|
549
597
|
payload[:note] = search_result[:note] if search_result[:note]
|
|
550
598
|
payload[:partial] = true if search_result[:partial]
|
|
599
|
+
payload[:hint] = search_result[:hint] if search_result[:hint]
|
|
551
600
|
respond.call(renderer.render(:search, payload))
|
|
601
|
+
rescue Retrieval::Scope::InvalidScopeError => e
|
|
602
|
+
respond_err.call(e.message, code: :unsupported_argument, tool: 'search', argument: 'scope')
|
|
603
|
+
rescue IOError, SystemCallError, JSON::ParserError, EncodingError
|
|
604
|
+
respond_err.call(
|
|
605
|
+
'Search completeness: unknown (unreadable_or_corrupt_source). ' \
|
|
606
|
+
'An Index artifact is unavailable or malformed; inspect woods_status and run woods:validate.',
|
|
607
|
+
code: :corrupt_artifact, tool: 'search', completeness: SearchResults.unavailable
|
|
608
|
+
)
|
|
552
609
|
end
|
|
553
610
|
end
|
|
554
611
|
|
|
@@ -563,6 +620,7 @@ module Woods
|
|
|
563
620
|
properties: {
|
|
564
621
|
identifier: { type: 'string', description: 'Unit identifier to start from' },
|
|
565
622
|
depth: { type: 'integer', description: 'Maximum traversal depth (default: 2)' },
|
|
623
|
+
explain: { type: 'boolean', description: 'Include recorded relationship evidence and shared shortest witnesses (default: false)' },
|
|
566
624
|
types: {
|
|
567
625
|
type: 'array', items: { type: 'string' },
|
|
568
626
|
description: 'Filter to these types'
|
|
@@ -580,23 +638,31 @@ module Woods
|
|
|
580
638
|
},
|
|
581
639
|
limit: { type: 'integer',
|
|
582
640
|
description: "Maximum nodes to return (default: #{DEFAULT_TRAVERSAL_LIMIT})" },
|
|
583
|
-
offset: { type: 'integer', description: 'Skip this many nodes (default: 0)' }
|
|
641
|
+
offset: { type: 'integer', description: 'Skip this many nodes (default: 0)' },
|
|
642
|
+
max_nodes: { type: 'integer', minimum: 1, maximum: 10_000,
|
|
643
|
+
description: 'Visited-node budget including root (default: 1000; maximum: 10000)' },
|
|
644
|
+
max_edges: { type: 'integer', minimum: 1, maximum: 100_000,
|
|
645
|
+
description: 'Edge-check budget before filters, including reverse via checks (default: 10000; maximum: 100000)' }
|
|
584
646
|
},
|
|
585
647
|
required: ['identifier']
|
|
586
648
|
}
|
|
587
|
-
) do |identifier:, server_context:, depth: nil, types: nil, via: nil, limit: nil, offset: nil|
|
|
649
|
+
) do |identifier:, server_context:, depth: nil, types: nil, via: nil, limit: nil, offset: nil, max_nodes: nil, max_edges: nil, explain: nil|
|
|
588
650
|
types = coerce.call(types)
|
|
589
651
|
via = coerce.call(via)
|
|
590
652
|
depth = coerce_int.call(depth)
|
|
591
653
|
limit = coerce_int.call(limit)
|
|
592
654
|
offset = coerce_int.call(offset)
|
|
593
|
-
result = reader.send(reader_method, identifier, depth: depth || 2, types: types, via: via
|
|
655
|
+
result = reader.send(reader_method, identifier, depth: depth || 2, types: types, via: via,
|
|
656
|
+
max_nodes: coerce_int.call(max_nodes) || 1000,
|
|
657
|
+
max_edges: coerce_int.call(max_edges) || 10_000, explain: explain || false)
|
|
594
658
|
if result[:found] == false
|
|
595
659
|
result[:message] =
|
|
596
660
|
"Identifier '#{identifier}' not found in the index. Use 'search' to find valid identifiers."
|
|
597
661
|
end
|
|
662
|
+
TraversalResponse.annotate(result)
|
|
598
663
|
paginate_nodes.call(result, limit || DEFAULT_TRAVERSAL_LIMIT, offset || 0)
|
|
599
|
-
|
|
664
|
+
TraversalEvidencePage.apply(result)
|
|
665
|
+
respond.call(renderer.render(render_key, result), data: result)
|
|
600
666
|
end
|
|
601
667
|
end
|
|
602
668
|
|
|
@@ -636,32 +702,20 @@ module Woods
|
|
|
636
702
|
enum: ToolResponseRenderer::GRAPH_ANALYSIS_SECTIONS + %w[all],
|
|
637
703
|
description: 'Which analysis to return. Default: all'
|
|
638
704
|
},
|
|
639
|
-
limit: { type: 'integer', description:
|
|
705
|
+
limit: { type: 'integer', description: "Limit results per section (default: #{DEFAULT_GRAPH_ANALYSIS_LIMIT})" },
|
|
640
706
|
offset: { type: 'integer', description: 'Skip this many results per section (default: 0)' }
|
|
641
707
|
}
|
|
642
708
|
}
|
|
643
709
|
) do |server_context:, analysis: nil, limit: nil, offset: nil|
|
|
644
|
-
limit = coerce_int.call(limit)
|
|
710
|
+
limit = coerce_int.call(limit) || DEFAULT_GRAPH_ANALYSIS_LIMIT
|
|
645
711
|
offset = coerce_int.call(offset)
|
|
646
712
|
data = reader.graph_analysis
|
|
647
713
|
section = analysis || 'all'
|
|
648
714
|
effective_offset = offset || 0
|
|
649
715
|
|
|
650
|
-
result =
|
|
651
|
-
|
|
652
|
-
|
|
653
|
-
ToolResponseRenderer::GRAPH_ANALYSIS_SECTIONS.each do |key|
|
|
654
|
-
paginate.call(truncated, key, limit, effective_offset)
|
|
655
|
-
end
|
|
656
|
-
truncated
|
|
657
|
-
else
|
|
658
|
-
data
|
|
659
|
-
end
|
|
660
|
-
else
|
|
661
|
-
single = { section => data[section] || [], 'stats' => data['stats'] }
|
|
662
|
-
paginate.call(single, section, limit, effective_offset) if limit || effective_offset.positive?
|
|
663
|
-
single
|
|
664
|
-
end
|
|
716
|
+
result = section == 'all' ? data.dup : { section => data[section] || [], 'stats' => data['stats'] }
|
|
717
|
+
sections = section == 'all' ? ToolResponseRenderer::GRAPH_ANALYSIS_SECTIONS : [section]
|
|
718
|
+
sections.each { |key| paginate.call(result, key, limit, effective_offset) }
|
|
665
719
|
|
|
666
720
|
respond.call(renderer.render(:graph_analysis, result))
|
|
667
721
|
end
|
|
@@ -874,13 +928,14 @@ module Woods
|
|
|
874
928
|
coerce = method(:coerce_array)
|
|
875
929
|
stale_check = method(:stale_index_result?)
|
|
876
930
|
degraded_response = method(:degraded_retrieval_response)
|
|
931
|
+
retrieval_mode = retriever.respond_to?(:mode) ? retriever.mode : :semantic
|
|
877
932
|
server.define_tool(
|
|
878
933
|
name: 'codebase_retrieve',
|
|
879
|
-
description: '
|
|
934
|
+
description: 'Ranked retrieval: relevant code units for a natural-language question. ' \
|
|
880
935
|
'Example: codebase_retrieve("how does billing work?") returns ranked source context. ' \
|
|
881
936
|
'Returns a token-budgeted context string ready to paste into a prompt. ' \
|
|
882
937
|
'Use `search` for exact name/pattern matching; use this for conceptual questions. ' \
|
|
883
|
-
'
|
|
938
|
+
'Uses configured embeddings, or explicit WOODS_RETRIEVAL_MODE=lexical over extraction units. ' \
|
|
884
939
|
'By default excludes test_mappings (~33% of a typical index) so spec filenames do not ' \
|
|
885
940
|
'dominate semantic rank; pass types: ["test_mapping"] to opt back in. ' \
|
|
886
941
|
'Parameter: use `budget` for the token budget (not `limit` — that means result count ' \
|
|
@@ -890,12 +945,15 @@ module Woods
|
|
|
890
945
|
query: { type: 'string',
|
|
891
946
|
description: 'Natural language question (e.g. "How does user authentication work?")' },
|
|
892
947
|
budget: { type: 'integer',
|
|
893
|
-
description: 'Token budget for context assembly (
|
|
948
|
+
description: 'Token budget for context assembly (configured max_context_tokens; otherwise 8000).' },
|
|
949
|
+
evidence: { type: 'string', enum: %w[full compact outline],
|
|
950
|
+
description: 'Explicit complete spans or API outline within each ranked unit; default full retains existing output.' },
|
|
894
951
|
types: {
|
|
895
952
|
type: 'array', items: { type: 'string' },
|
|
896
953
|
description: 'Restrict results to these unit types (model, controller, service, job, mailer, ' \
|
|
897
954
|
'rails_source, test_mapping, etc.). Overrides the default test_mapping exclusion. ' \
|
|
898
|
-
'
|
|
955
|
+
'Lexical mode and explicit package/path scopes filter before limits and omit the global rank table. ' \
|
|
956
|
+
'In semantic mode, when the unfiltered top-K has no requested type, the retriever ' \
|
|
899
957
|
'falls back to rank-within-type so the response is populated whenever units of ' \
|
|
900
958
|
'the requested type exist in the index. The response appends a "Type rank ' \
|
|
901
959
|
'context" table with per-type: source, rank in unfiltered top-K, global_k, ' \
|
|
@@ -904,6 +962,10 @@ module Woods
|
|
|
904
962
|
'(index has this type but other requested types filled the result), absent ' \
|
|
905
963
|
'(zero units of this type in the index).'
|
|
906
964
|
},
|
|
965
|
+
packages: { type: 'array', items: { type: 'string' },
|
|
966
|
+
description: 'Exact published nearest package owners. OR within the list; AND with paths and type eligibility.' },
|
|
967
|
+
source_paths: { type: 'array', items: { type: 'string' },
|
|
968
|
+
description: 'Application-relative directory prefixes. Scope applies before candidate limits; graph expansion stays inside it.' },
|
|
907
969
|
exclude_types: {
|
|
908
970
|
type: 'array', items: { type: 'string' },
|
|
909
971
|
description: 'Additional types to exclude on top of the default test_mapping exclusion.'
|
|
@@ -911,7 +973,7 @@ module Woods
|
|
|
911
973
|
},
|
|
912
974
|
required: ['query']
|
|
913
975
|
}
|
|
914
|
-
) do |query:, server_context:, budget: nil, limit: nil, types: nil, exclude_types: nil|
|
|
976
|
+
) do |query:, server_context:, budget: nil, limit: nil, types: nil, exclude_types: nil, packages: nil, source_paths: nil, evidence: 'full'|
|
|
915
977
|
# `limit` isn't declared in the schema but clients still send it
|
|
916
978
|
# because sibling tools (search, recent_changes, pagerank) use
|
|
917
979
|
# `limit` as a result count. Mapping it to `budget` here would
|
|
@@ -919,10 +981,10 @@ module Woods
|
|
|
919
981
|
# budget). Surface a helpful typed error instead.
|
|
920
982
|
unless limit.nil?
|
|
921
983
|
next respond_err.call(
|
|
922
|
-
'codebase_retrieve uses `budget` (token budget, default
|
|
984
|
+
'codebase_retrieve uses `budget` (token budget, configured default), not `limit`. ' \
|
|
923
985
|
'`limit` is the result-count parameter on sibling tools (search, recent_changes, pagerank). ' \
|
|
924
986
|
"Pass `budget: #{coerce_int.call(limit)}` if you meant a #{coerce_int.call(limit)}-token context, " \
|
|
925
|
-
'or drop the kwarg entirely for the default
|
|
987
|
+
'or drop the kwarg entirely for the configured default.',
|
|
926
988
|
code: :unsupported_argument,
|
|
927
989
|
tool: 'codebase_retrieve',
|
|
928
990
|
argument: 'limit',
|
|
@@ -931,6 +993,11 @@ module Woods
|
|
|
931
993
|
)
|
|
932
994
|
end
|
|
933
995
|
|
|
996
|
+
begin
|
|
997
|
+
Retrieval::SourceEvidence.validate_mode!(evidence)
|
|
998
|
+
rescue ArgumentError => e
|
|
999
|
+
next respond_err.call(e.message, code: :unsupported_argument, tool: 'codebase_retrieve', argument: 'evidence')
|
|
1000
|
+
end
|
|
934
1001
|
budget = coerce_int.call(budget)
|
|
935
1002
|
types = coerce.call(types)
|
|
936
1003
|
exclude_types = coerce.call(exclude_types)
|
|
@@ -949,12 +1016,20 @@ module Woods
|
|
|
949
1016
|
end
|
|
950
1017
|
if retriever
|
|
951
1018
|
begin
|
|
1019
|
+
scope_options = if Retrieval::Scope.requested?(packages: packages, source_paths: source_paths)
|
|
1020
|
+
{ packages: packages, source_paths: source_paths }
|
|
1021
|
+
else
|
|
1022
|
+
{}
|
|
1023
|
+
end
|
|
1024
|
+
scope_options[:evidence] = evidence unless evidence == 'full'
|
|
952
1025
|
result = retriever.retrieve(
|
|
953
1026
|
query,
|
|
954
|
-
budget: budget || 8000,
|
|
1027
|
+
budget: budget || (retriever.respond_to?(:default_budget) ? retriever.default_budget : 8000),
|
|
955
1028
|
types: types,
|
|
956
|
-
exclude_types: exclude_types
|
|
1029
|
+
exclude_types: exclude_types, **scope_options
|
|
957
1030
|
)
|
|
1031
|
+
rescue Retrieval::Scope::InvalidScopeError => e
|
|
1032
|
+
next respond_err.call(e.message, code: :unsupported_argument, tool: 'codebase_retrieve', argument: 'scope')
|
|
958
1033
|
rescue Woods::Retriever::StoreError => e
|
|
959
1034
|
# M8: a metadata-store failure mid-query must not surface as
|
|
960
1035
|
# a raw raise through the tool boundary (or as the misleading
|
|
@@ -963,7 +1038,7 @@ module Woods
|
|
|
963
1038
|
respond_err,
|
|
964
1039
|
reason: e.message,
|
|
965
1040
|
stores: [e.store],
|
|
966
|
-
phase: 'query'
|
|
1041
|
+
phase: 'query', mode: retrieval_mode
|
|
967
1042
|
)
|
|
968
1043
|
end
|
|
969
1044
|
if stale_check.call(result)
|
|
@@ -975,12 +1050,22 @@ module Woods
|
|
|
975
1050
|
tool: 'codebase_retrieve'
|
|
976
1051
|
)
|
|
977
1052
|
end
|
|
978
|
-
|
|
1053
|
+
if evidence != 'full' || (result.respond_to?(:applied_scope) && result.applied_scope)
|
|
1054
|
+
::MCP::Tool::Response.new(
|
|
1055
|
+
[{ type: 'text', text: result.context }],
|
|
1056
|
+
structured_content: { text: result.context, data: { applied_scope: result.applied_scope, sources: result.sources } },
|
|
1057
|
+
meta: { applied_scope: result.applied_scope }
|
|
1058
|
+
)
|
|
1059
|
+
else
|
|
1060
|
+
respond.call(result.context)
|
|
1061
|
+
end
|
|
979
1062
|
else
|
|
980
1063
|
respond_err.call(
|
|
981
1064
|
'Semantic search is disabled — no embedding provider is configured. ' \
|
|
982
1065
|
'To enable: set OPENAI_API_KEY, or run Ollama locally ' \
|
|
983
1066
|
'(brew install ollama && ollama serve && ollama pull nomic-embed-text). ' \
|
|
1067
|
+
'For ranked discovery with no embeddings, set WOODS_RETRIEVAL_MODE=lexical in the MCP process ' \
|
|
1068
|
+
'environment and restart the server. See docs/RETRIEVAL_GUIDE.md#embedding-free-lexical-retrieval. ' \
|
|
984
1069
|
'Use the `search` tool for pattern-based matching in the meantime.',
|
|
985
1070
|
code: :not_configured,
|
|
986
1071
|
config_key: 'embedding_provider',
|
|
@@ -1020,7 +1105,16 @@ module Woods
|
|
|
1020
1105
|
# @param phase [String] 'boot' (hydration failure) or 'query'
|
|
1021
1106
|
# (store failure at query time)
|
|
1022
1107
|
# @return [MCP::Tool::Response]
|
|
1023
|
-
def degraded_retrieval_response(respond_err, reason:, stores:, phase:)
|
|
1108
|
+
def degraded_retrieval_response(respond_err, reason:, stores:, phase:, mode: :semantic)
|
|
1109
|
+
if mode == :lexical
|
|
1110
|
+
return respond_err.call(
|
|
1111
|
+
"Lexical retrieval is degraded: #{reason}. No partial lexical snapshot was served. " \
|
|
1112
|
+
'Inspect woods_status and repair or re-extract the published index, then retry.',
|
|
1113
|
+
code: :degraded_index, tool: 'codebase_retrieve', degraded: true,
|
|
1114
|
+
phase: phase, stores: stores, reason: reason, mode: 'lexical'
|
|
1115
|
+
)
|
|
1116
|
+
end
|
|
1117
|
+
|
|
1024
1118
|
respond_err.call(
|
|
1025
1119
|
"Semantic search is degraded: #{reason}. The affected store(s) return no data, so " \
|
|
1026
1120
|
'queries would come back empty — this is NOT "no results". ' \
|
|
@@ -1129,6 +1223,9 @@ module Woods
|
|
|
1129
1223
|
)
|
|
1130
1224
|
doc = assembler.assemble(session_id, budget: budget || 8000, depth: depth || 1)
|
|
1131
1225
|
respond.call(doc.to_markdown)
|
|
1226
|
+
rescue Woods::SessionTracer::AmbiguousUnitError => e
|
|
1227
|
+
respond_err.call(e.message, code: :ambiguous_identity, tool: 'session_trace',
|
|
1228
|
+
identifier: e.identifier, types: e.types)
|
|
1132
1229
|
rescue StandardError => e
|
|
1133
1230
|
respond_err.call(
|
|
1134
1231
|
"Session trace failed: #{e.message}",
|
|
@@ -1979,13 +2076,17 @@ module Woods
|
|
|
1979
2076
|
description: 'Diagnose whether the Woods index and server are healthy. Returns extraction metadata ' \
|
|
1980
2077
|
'(last run, unit counts, git SHA, staleness in seconds), retriever/embedding configuration, ' \
|
|
1981
2078
|
'bootstrap state (hydrated / degraded / failed + reason), feature flags, and a ready flag. ' \
|
|
1982
|
-
'
|
|
1983
|
-
|
|
1984
|
-
|
|
2079
|
+
'Includes source-content freshness; quick scans have a 250ms budget, explicit deep scans have 5s. ' \
|
|
2080
|
+
'Incomplete evidence is unknown. Call this first on cold connect.',
|
|
2081
|
+
input_schema: { type: 'object', properties: {
|
|
2082
|
+
source_check: { type: 'string', enum: %w[quick deep], default: 'quick',
|
|
2083
|
+
description: 'Bounded source content verification: quick (250ms) or deep (5s).' }
|
|
2084
|
+
} }
|
|
2085
|
+
) do |server_context:, source_check: 'quick'|
|
|
1985
2086
|
_ = server_context
|
|
1986
2087
|
status = Woods::MCP::Server.build_status(
|
|
1987
2088
|
reader: reader, retriever: retriever, index_dir: index_dir,
|
|
1988
|
-
bootstrap_state: bootstrap_state
|
|
2089
|
+
bootstrap_state: bootstrap_state, source_check: source_check
|
|
1989
2090
|
)
|
|
1990
2091
|
respond.call(JSON.pretty_generate(status))
|
|
1991
2092
|
end
|
|
@@ -2004,21 +2105,21 @@ module Woods
|
|
|
2004
2105
|
# provider in use. Without this, operators debugging "wrong provider" see
|
|
2005
2106
|
# status claiming +embedding_model: "text-embedding-3-small"+ next to
|
|
2006
2107
|
# +embedding_provider: "ollama"+ and reasonably distrust every field.
|
|
2007
|
-
def build_status(reader:, retriever:, index_dir:, bootstrap_state: nil)
|
|
2108
|
+
def build_status(reader:, retriever:, index_dir:, bootstrap_state: nil, source_check: 'quick')
|
|
2008
2109
|
# Pin the generation across the whole payload. Without this the
|
|
2009
2110
|
# manifest can be read at generation N and `generation_fields` then
|
|
2010
2111
|
# report N+1 — a status report that describes counts from one index
|
|
2011
2112
|
# while announcing the number of another, which is precisely the
|
|
2012
2113
|
# confusion this tool exists to resolve.
|
|
2013
|
-
return build_status_payload(reader, retriever, index_dir, bootstrap_state) unless
|
|
2114
|
+
return build_status_payload(reader, retriever, index_dir, bootstrap_state, source_check) unless
|
|
2014
2115
|
reader.respond_to?(:with_pinned_generation)
|
|
2015
2116
|
|
|
2016
2117
|
reader.with_pinned_generation do
|
|
2017
|
-
build_status_payload(reader, retriever, index_dir, bootstrap_state)
|
|
2118
|
+
build_status_payload(reader, retriever, index_dir, bootstrap_state, source_check)
|
|
2018
2119
|
end
|
|
2019
2120
|
end
|
|
2020
2121
|
|
|
2021
|
-
def build_status_payload(reader, retriever, index_dir, bootstrap_state)
|
|
2122
|
+
def build_status_payload(reader, retriever, index_dir, bootstrap_state, source_check)
|
|
2022
2123
|
manifest = safe_manifest(reader)
|
|
2023
2124
|
extracted_at = manifest && manifest['extracted_at']
|
|
2024
2125
|
staleness = staleness_seconds(extracted_at)
|
|
@@ -2036,14 +2137,15 @@ module Woods
|
|
|
2036
2137
|
index_dir: index_dir.to_s,
|
|
2037
2138
|
update: Woods::UpdateCheck.status_hash
|
|
2038
2139
|
},
|
|
2039
|
-
index: index_section(manifest, extracted_at, staleness, index_dir, reader),
|
|
2140
|
+
index: index_section(manifest, extracted_at, staleness, index_dir, reader, source_check),
|
|
2040
2141
|
watch: watch_section(index_dir),
|
|
2041
2142
|
retriever: {
|
|
2042
2143
|
configured: !retriever.nil?,
|
|
2043
|
-
class: retriever&.class&.name
|
|
2144
|
+
class: retriever&.class&.name,
|
|
2145
|
+
**(retriever.respond_to?(:mode) && retriever.mode == :lexical ? { mode: 'lexical' } : {})
|
|
2044
2146
|
},
|
|
2045
2147
|
bootstrap: bootstrap_state&.to_h,
|
|
2046
|
-
features:
|
|
2148
|
+
features: retrieval_features(config, resolved, retriever)
|
|
2047
2149
|
}
|
|
2048
2150
|
end
|
|
2049
2151
|
|
|
@@ -2066,10 +2168,11 @@ module Woods
|
|
|
2066
2168
|
# diff directly. This is an observability signal, not a hard gate —
|
|
2067
2169
|
# hard-refusing responses would be much more disruptive than a loudly-
|
|
2068
2170
|
# visible staleness flag that agents can branch on.
|
|
2069
|
-
def index_section(manifest, extracted_at, staleness, index_dir, reader = nil)
|
|
2171
|
+
def index_section(manifest, extracted_at, staleness, index_dir, reader = nil, source_check = 'quick')
|
|
2070
2172
|
base = {
|
|
2071
2173
|
extracted_at: extracted_at,
|
|
2072
2174
|
staleness_seconds: staleness,
|
|
2175
|
+
woods_version: manifest && manifest['woods_version'],
|
|
2073
2176
|
rails_version: manifest && manifest['rails_version'],
|
|
2074
2177
|
ruby_version: manifest && manifest['ruby_version'],
|
|
2075
2178
|
total_units: manifest && manifest['total_units'],
|
|
@@ -2080,6 +2183,11 @@ module Woods
|
|
|
2080
2183
|
schema_sha: manifest && manifest['schema_sha']
|
|
2081
2184
|
}
|
|
2082
2185
|
|
|
2186
|
+
base[:source_freshness] = if reader.respond_to?(:source_freshness)
|
|
2187
|
+
reader.source_freshness(mode: source_check)
|
|
2188
|
+
else
|
|
2189
|
+
{ 'state' => 'unknown', 'reasons' => ['source_reader_unavailable'], 'complete' => false }
|
|
2190
|
+
end
|
|
2083
2191
|
base.merge!(generation_fields(index_dir, reader))
|
|
2084
2192
|
base.merge!(working_tree_fields(index_dir))
|
|
2085
2193
|
|
|
@@ -2210,7 +2318,7 @@ module Woods
|
|
|
2210
2318
|
record = JSON.parse(Woods::AtomicFile.read(path))
|
|
2211
2319
|
# `state` is whatever the daemon last wrote, and a `kill -9`'d daemon
|
|
2212
2320
|
# leaves `running` behind forever. `alive?` adds the two checks that
|
|
2213
|
-
# catch that —
|
|
2321
|
+
# catch that — a recent record and, for local hosts, a live pid — so the
|
|
2214
2322
|
# payload can distinguish "maintaining this index" from "claimed to be,
|
|
2215
2323
|
# once". Reported as a separate field rather than by overwriting
|
|
2216
2324
|
# `state`, because the recorded state and the liveness verdict answer
|
|
@@ -2263,6 +2371,13 @@ module Woods
|
|
|
2263
2371
|
# historic status payloads always reported +false+ regardless of the
|
|
2264
2372
|
# actual console MCP state. Advertising a misleading field is worse
|
|
2265
2373
|
# than not advertising it at all.
|
|
2374
|
+
def retrieval_features(config, resolved, retriever)
|
|
2375
|
+
features = features_from(config, resolved)
|
|
2376
|
+
return features unless retriever.respond_to?(:mode) && retriever.mode == :lexical
|
|
2377
|
+
|
|
2378
|
+
features.merge(retrieval_mode: 'lexical', embedding_model: nil, embedding_provider: nil, vector_store: nil)
|
|
2379
|
+
end
|
|
2380
|
+
|
|
2266
2381
|
def features_from(config, resolved)
|
|
2267
2382
|
provider_hash = resolved&.embedding_provider || {}
|
|
2268
2383
|
resolved_provider = resolved_provider_symbol(provider_hash[:class])
|
|
@@ -11,6 +11,8 @@ module Woods
|
|
|
11
11
|
TASK_RESULT_TOOLS = %w[pipeline_embed pipeline_extract].freeze
|
|
12
12
|
|
|
13
13
|
INTEGER_BOUNDS = {
|
|
14
|
+
'max_nodes' => [1, 10_000],
|
|
15
|
+
'max_edges' => [1, 100_000],
|
|
14
16
|
'budget' => [1, 200_000],
|
|
15
17
|
'depth' => [0, 20],
|
|
16
18
|
'limit' => [1, 1_000],
|
|
@@ -25,7 +27,7 @@ module Woods
|
|
|
25
27
|
text: { type: 'string' },
|
|
26
28
|
data: {
|
|
27
29
|
type: %w[object array string number boolean null],
|
|
28
|
-
description: '
|
|
30
|
+
description: 'Structured tool payload, including traversal data in every renderer; otherwise parsed JSON when available.'
|
|
29
31
|
}
|
|
30
32
|
},
|
|
31
33
|
required: ['text'],
|
|
@@ -1,5 +1,8 @@
|
|
|
1
1
|
# frozen_string_literal: true
|
|
2
2
|
|
|
3
|
+
require_relative 'traversal_evidence_index'
|
|
4
|
+
require_relative 'traversal_evidence_text'
|
|
5
|
+
|
|
3
6
|
module Woods
|
|
4
7
|
module MCP
|
|
5
8
|
# Base class for rendering MCP tool responses in different output formats.
|
|
@@ -67,6 +70,44 @@ module Woods
|
|
|
67
70
|
|
|
68
71
|
private
|
|
69
72
|
|
|
73
|
+
def traversal_coverage_lines(data)
|
|
74
|
+
coverage = fetch_key(data, :graph_coverage)
|
|
75
|
+
notice = fetch_key(coverage, :notice) if coverage.is_a?(Hash)
|
|
76
|
+
notice ? [notice] : []
|
|
77
|
+
end
|
|
78
|
+
|
|
79
|
+
def traversal_lower_bound_note(data, shown)
|
|
80
|
+
return unless fetch_key(data, :total_is_exact) == false
|
|
81
|
+
|
|
82
|
+
total = fetch_key(data, :nodes_total, shown)
|
|
83
|
+
offset = fetch_key(data, :nodes_offset, 0)
|
|
84
|
+
position = offset.positive? ? " from offset #{offset}" : ''
|
|
85
|
+
"Showing #{shown} of at least #{total}#{position} " \
|
|
86
|
+
"(total unknown: #{fetch_key(data, :partial_reason)})."
|
|
87
|
+
end
|
|
88
|
+
|
|
89
|
+
def search_completeness_lines(data)
|
|
90
|
+
evidence = fetch_key(data, :completeness)
|
|
91
|
+
return [] unless evidence.is_a?(Hash)
|
|
92
|
+
|
|
93
|
+
more = { true => 'yes', false => 'no', nil => 'unknown' }.fetch(fetch_key(evidence, :has_more))
|
|
94
|
+
total = fetch_key(evidence, :total_matches)
|
|
95
|
+
lines = [
|
|
96
|
+
"Search completeness: #{fetch_key(evidence, :status)} (#{fetch_key(evidence, :reason)}).",
|
|
97
|
+
"More matches: #{more}; total matches: #{total.nil? ? 'unknown' : total}; " \
|
|
98
|
+
"matched lower bound: #{fetch_key(evidence, :matched_lower_bound)}."
|
|
99
|
+
]
|
|
100
|
+
scope = fetch_key(data, :applied_scope)
|
|
101
|
+
if scope
|
|
102
|
+
lines << "Applied scope: packages=#{fetch_key(scope, :packages).inspect}; " \
|
|
103
|
+
"source_paths=#{fetch_key(scope, :source_paths).inspect}; " \
|
|
104
|
+
"eligible units=#{fetch_key(scope, :eligible_units)}."
|
|
105
|
+
end
|
|
106
|
+
hint = fetch_key(data, :hint)
|
|
107
|
+
lines << hint if hint
|
|
108
|
+
lines
|
|
109
|
+
end
|
|
110
|
+
|
|
70
111
|
# Fetch a value from a hash by symbol or string key, falling back to a default.
|
|
71
112
|
#
|
|
72
113
|
# Handles data hashes that may use either symbol or string keys (e.g., data
|