woods 2.0.0.beta2 → 2.0.0.beta4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (233) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +339 -1
  3. data/CONTRIBUTING.md +188 -12
  4. data/README.md +93 -174
  5. data/SECURITY.md +9 -6
  6. data/docs/AGENT_GUIDE.md +109 -8
  7. data/docs/AGENT_SETUP.md +98 -7
  8. data/docs/BACKEND_MATRIX.md +25 -0
  9. data/docs/CLIENT_HOOKS.md +111 -0
  10. data/docs/CONFIGURATION_REFERENCE.md +267 -16
  11. data/docs/CONSOLE_MCP_SETUP.md +80 -7
  12. data/docs/DOCKER_SETUP.md +22 -3
  13. data/docs/EVALUATION.md +464 -1
  14. data/docs/EXTRACTOR_REFERENCE.md +45 -6
  15. data/docs/FAQ.md +11 -12
  16. data/docs/GETTING_STARTED.md +17 -5
  17. data/docs/INCREMENTAL_EXTRACTION.md +147 -7
  18. data/docs/INDEX_LAYOUT.md +382 -0
  19. data/docs/INTERNALS.md +7 -2
  20. data/docs/MCP_SERVERS.md +276 -5
  21. data/docs/MCP_TOOL_COOKBOOK.md +37 -22
  22. data/docs/MCP_WORKTREE_SETUP.md +43 -83
  23. data/docs/NOTION_INTEGRATION.md +13 -0
  24. data/docs/OBSIDIAN_INTEGRATION.md +57 -9
  25. data/docs/PUBLISHED_INDEX.md +72 -0
  26. data/docs/README.md +7 -0
  27. data/docs/RETRIEVAL_GUIDE.md +273 -12
  28. data/docs/RUNTIME_TRACING.md +71 -0
  29. data/docs/SOURCE_FRESHNESS.md +143 -0
  30. data/docs/TROUBLESHOOTING.md +129 -18
  31. data/docs/UNBLOCKED_INTEGRATION.md +25 -0
  32. data/docs/UPGRADING_TO_2.md +48 -22
  33. data/docs/WATCH_DAEMON.md +277 -67
  34. data/exe/woods-agent-config +6 -0
  35. data/exe/woods-extract +5 -0
  36. data/exe/woods-hook-context +6 -0
  37. data/exe/woods-mcp-start +14 -9
  38. data/lib/generators/woods/pgvector_generator.rb +8 -2
  39. data/lib/generators/woods/templates/woods.rb.tt +1 -3
  40. data/lib/tasks/woods.rake +47 -397
  41. data/lib/woods/agent_configuration/applier.rb +135 -0
  42. data/lib/woods/agent_configuration/cli.rb +101 -0
  43. data/lib/woods/agent_configuration/cli_options.rb +29 -0
  44. data/lib/woods/agent_configuration/document.rb +105 -0
  45. data/lib/woods/agent_configuration/error.rb +7 -0
  46. data/lib/woods/agent_configuration/launcher.rb +75 -0
  47. data/lib/woods/agent_configuration/layout.rb +72 -0
  48. data/lib/woods/agent_configuration/managed_section.rb +62 -0
  49. data/lib/woods/agent_configuration/plan.rb +98 -0
  50. data/lib/woods/agent_configuration/plan_diff.rb +38 -0
  51. data/lib/woods/agent_configuration/planned_files.rb +61 -0
  52. data/lib/woods/agent_configuration/planner.rb +63 -0
  53. data/lib/woods/agent_configuration/planner_validation.rb +77 -0
  54. data/lib/woods/agent_configuration/preflight.rb +100 -0
  55. data/lib/woods/agent_configuration/recovery.rb +49 -0
  56. data/lib/woods/ast/node.rb +2 -0
  57. data/lib/woods/ast/parser.rb +38 -5
  58. data/lib/woods/builder.rb +21 -5
  59. data/lib/woods/cache/cache_middleware.rb +28 -7
  60. data/lib/woods/cache/cache_store.rb +4 -5
  61. data/lib/woods/change_set.rb +5 -4
  62. data/lib/woods/console/credential_index.rb +20 -2
  63. data/lib/woods/console/credential_scanner.rb +18 -17
  64. data/lib/woods/console/credential_scanner_registry.rb +36 -0
  65. data/lib/woods/console/dispatch_pipeline.rb +7 -0
  66. data/lib/woods/console/embedded_executor.rb +32 -10
  67. data/lib/woods/console/encrypted_credential_snapshot.rb +16 -0
  68. data/lib/woods/console/rack_middleware.rb +22 -13
  69. data/lib/woods/console/server.rb +18 -16
  70. data/lib/woods/console/sql_noise_stripper.rb +9 -7
  71. data/lib/woods/console/sql_table_scanner.rb +47 -7
  72. data/lib/woods/console/sql_validator.rb +49 -9
  73. data/lib/woods/console/sqlite_read_guard.rb +46 -0
  74. data/lib/woods/coordination/pipeline_lock.rb +3 -2
  75. data/lib/woods/dependency_graph.rb +65 -13
  76. data/lib/woods/embedding/corpus.rb +94 -0
  77. data/lib/woods/embedding/indexer.rb +114 -60
  78. data/lib/woods/embedding/openai.rb +17 -6
  79. data/lib/woods/evaluation/ablation_executor.rb +6 -1
  80. data/lib/woods/evaluation/ablation_timed_executor.rb +22 -4
  81. data/lib/woods/export/typed_reader.rb +56 -0
  82. data/lib/woods/extractor.rb +277 -149
  83. data/lib/woods/extractors/action_cable_extractor.rb +3 -1
  84. data/lib/woods/extractors/behavioral_profile.rb +9 -7
  85. data/lib/woods/extractors/caching_extractor.rb +3 -1
  86. data/lib/woods/extractors/concern_extractor.rb +64 -6
  87. data/lib/woods/extractors/configuration_extractor.rb +7 -3
  88. data/lib/woods/extractors/controller_extractor.rb +13 -4
  89. data/lib/woods/extractors/database_view_extractor.rb +3 -1
  90. data/lib/woods/extractors/declared_parent.rb +55 -0
  91. data/lib/woods/extractors/decorator_extractor.rb +3 -1
  92. data/lib/woods/extractors/engine_extractor.rb +3 -1
  93. data/lib/woods/extractors/event_extractor.rb +4 -2
  94. data/lib/woods/extractors/factory_extractor.rb +3 -1
  95. data/lib/woods/extractors/graphql_extractor.rb +10 -13
  96. data/lib/woods/extractors/i18n_extractor.rb +3 -1
  97. data/lib/woods/extractors/job_extractor.rb +6 -19
  98. data/lib/woods/extractors/lib_extractor.rb +13 -9
  99. data/lib/woods/extractors/mailer_extractor.rb +26 -15
  100. data/lib/woods/extractors/manager_extractor.rb +3 -1
  101. data/lib/woods/extractors/method_parameters.rb +53 -0
  102. data/lib/woods/extractors/middleware_argument.rb +65 -0
  103. data/lib/woods/extractors/middleware_extractor.rb +9 -3
  104. data/lib/woods/extractors/migration_extractor.rb +3 -1
  105. data/lib/woods/extractors/model_extractor.rb +26 -34
  106. data/lib/woods/extractors/package_extractor.rb +24 -4
  107. data/lib/woods/extractors/phlex_extractor.rb +3 -1
  108. data/lib/woods/extractors/policy_extractor.rb +3 -1
  109. data/lib/woods/extractors/poro_extractor.rb +13 -9
  110. data/lib/woods/extractors/pundit_extractor.rb +3 -1
  111. data/lib/woods/extractors/rails_source_extractor.rb +4 -2
  112. data/lib/woods/extractors/rake_task_extractor.rb +4 -2
  113. data/lib/woods/extractors/route_extractor.rb +3 -1
  114. data/lib/woods/extractors/route_helper_resolver.rb +10 -33
  115. data/lib/woods/extractors/scheduled_job_extractor.rb +41 -15
  116. data/lib/woods/extractors/serializer_extractor.rb +4 -2
  117. data/lib/woods/extractors/service_extractor.rb +3 -1
  118. data/lib/woods/extractors/shared_dependency_scanner.rb +2 -2
  119. data/lib/woods/extractors/shared_utility_methods.rb +48 -19
  120. data/lib/woods/extractors/source_nesting.rb +1 -1
  121. data/lib/woods/extractors/state_machine_extractor.rb +3 -1
  122. data/lib/woods/extractors/test_mapping_extractor.rb +3 -1
  123. data/lib/woods/extractors/validator_extractor.rb +3 -1
  124. data/lib/woods/extractors/view_component_extractor.rb +3 -1
  125. data/lib/woods/extractors/view_template_extractor.rb +3 -1
  126. data/lib/woods/gem_mapper.rb +2 -0
  127. data/lib/woods/git_history.rb +116 -0
  128. data/lib/woods/graph_analyzer.rb +35 -6
  129. data/lib/woods/hooks/context_cli.rb +54 -0
  130. data/lib/woods/hooks/context_event.rb +88 -0
  131. data/lib/woods/hooks/context_hint.rb +73 -0
  132. data/lib/woods/hooks/context_impact.rb +77 -0
  133. data/lib/woods/hooks/context_output.rb +47 -0
  134. data/lib/woods/hooks/context_state.rb +102 -0
  135. data/lib/woods/hooks/refresh.rb +79 -0
  136. data/lib/woods/hooks/rule_projection.rb +78 -0
  137. data/lib/woods/input_rules.rb +19 -0
  138. data/lib/woods/mcp/bearer_auth.rb +22 -13
  139. data/lib/woods/mcp/bootstrapper.rb +79 -4
  140. data/lib/woods/mcp/config_resolver.rb +2 -1
  141. data/lib/woods/mcp/index_reader.rb +334 -162
  142. data/lib/woods/mcp/initialization_guidance.rb +27 -0
  143. data/lib/woods/mcp/origin_guard.rb +17 -9
  144. data/lib/woods/mcp/published_lexical_retriever.rb +115 -0
  145. data/lib/woods/mcp/renderers/markdown_renderer.rb +22 -9
  146. data/lib/woods/mcp/renderers/plain_renderer.rb +18 -8
  147. data/lib/woods/mcp/search_results.rb +74 -0
  148. data/lib/woods/mcp/server.rb +178 -63
  149. data/lib/woods/mcp/tool_contract.rb +3 -1
  150. data/lib/woods/mcp/tool_response_renderer.rb +41 -0
  151. data/lib/woods/mcp/traversal_evidence.rb +113 -0
  152. data/lib/woods/mcp/traversal_evidence_index.rb +100 -0
  153. data/lib/woods/mcp/traversal_evidence_page.rb +41 -0
  154. data/lib/woods/mcp/traversal_evidence_text.rb +52 -0
  155. data/lib/woods/mcp/traversal_response.rb +22 -0
  156. data/lib/woods/notion/exporter.rb +56 -17
  157. data/lib/woods/obsidian/destination_plan.rb +98 -0
  158. data/lib/woods/obsidian/name_mapper.rb +19 -3
  159. data/lib/woods/obsidian/note_builder.rb +19 -10
  160. data/lib/woods/obsidian/vault_exporter.rb +88 -32
  161. data/lib/woods/operator/pipeline_guard.rb +18 -13
  162. data/lib/woods/path_dispatcher.rb +13 -6
  163. data/lib/woods/payload_store.rb +27 -26
  164. data/lib/woods/published_index/typed_unit_reader.rb +40 -3
  165. data/lib/woods/published_index.rb +2 -2
  166. data/lib/woods/railtie.rb +3 -3
  167. data/lib/woods/railtie_support.rb +12 -12
  168. data/lib/woods/rake_helpers.rb +382 -0
  169. data/lib/woods/resilience/graph_invariant_validator/membership_checks.rb +71 -0
  170. data/lib/woods/resilience/graph_invariant_validator/node_checks.rb +61 -0
  171. data/lib/woods/resilience/graph_invariant_validator/reverse_relationship_checks.rb +46 -0
  172. data/lib/woods/resilience/graph_invariant_validator.rb +119 -0
  173. data/lib/woods/resilience/index_validator/graph_checks.rb +80 -0
  174. data/lib/woods/resilience/index_validator.rb +112 -23
  175. data/lib/woods/retrieval/context_assembler.rb +50 -15
  176. data/lib/woods/retrieval/lexical_assembler.rb +84 -0
  177. data/lib/woods/retrieval/lexical_index.rb +120 -0
  178. data/lib/woods/retrieval/ranker.rb +4 -2
  179. data/lib/woods/retrieval/scope.rb +108 -0
  180. data/lib/woods/retrieval/scoped_graph_store.rb +32 -0
  181. data/lib/woods/retrieval/scoped_vector_store.rb +55 -0
  182. data/lib/woods/retrieval/search_executor.rb +86 -27
  183. data/lib/woods/retrieval/source_evidence.rb +200 -0
  184. data/lib/woods/retriever.rb +98 -22
  185. data/lib/woods/ruby_analyzer/trace_enricher.rb +77 -38
  186. data/lib/woods/session_tracer/file_store.rb +6 -1
  187. data/lib/woods/session_tracer/middleware.rb +10 -12
  188. data/lib/woods/session_tracer/redis_store.rb +22 -6
  189. data/lib/woods/session_tracer/session_flow_assembler.rb +23 -17
  190. data/lib/woods/session_tracer/solid_cache_coordination.rb +6 -4
  191. data/lib/woods/session_tracer/unit_resolver.rb +63 -0
  192. data/lib/woods/source_inputs/consumer_errors.rb +31 -0
  193. data/lib/woods/source_inputs/handoff.rb +102 -0
  194. data/lib/woods/source_inputs/launcher.rb +157 -0
  195. data/lib/woods/source_inputs/manifest.rb +124 -0
  196. data/lib/woods/source_inputs/private_key.rb +55 -0
  197. data/lib/woods/source_inputs/scanner.rb +171 -0
  198. data/lib/woods/source_inputs/scopes.rb +71 -0
  199. data/lib/woods/source_inputs/session.rb +214 -0
  200. data/lib/woods/source_inputs/status.rb +84 -0
  201. data/lib/woods/source_inputs/verifier.rb +107 -0
  202. data/lib/woods/storage/metadata_store.rb +25 -25
  203. data/lib/woods/storage/pgvector.rb +35 -10
  204. data/lib/woods/storage/qdrant.rb +17 -7
  205. data/lib/woods/storage/vector_store.rb +18 -6
  206. data/lib/woods/tasks.rb +3 -2
  207. data/lib/woods/temporal/json_snapshot_store.rb +58 -9
  208. data/lib/woods/unblocked/exporter.rb +59 -70
  209. data/lib/woods/version.rb +1 -1
  210. data/lib/woods/watch/boot_snapshot.rb +52 -0
  211. data/lib/woods/watch/daemon.rb +154 -32
  212. data/lib/woods/watch/listen_watcher.rb +4 -0
  213. data/lib/woods/watch/polling_watcher.rb +5 -1
  214. data/lib/woods/watch/status.rb +20 -15
  215. data/lib/woods/watch/tree_scan.rb +21 -13
  216. data/lib/woods/watch/watcher.rb +4 -1
  217. data/lib/woods.rb +50 -11
  218. data/plugin/.claude-plugin/plugin.json +1 -1
  219. data/plugin/hooks/adapters/normalize.jq +15 -0
  220. data/plugin/hooks/adapters/normalize.rb +63 -0
  221. data/plugin/hooks/hooks.json +20 -0
  222. data/plugin/hooks/woods-context.sh +50 -0
  223. data/plugin/hooks/woods-input-rules.sh +159 -0
  224. data/plugin/hooks/woods-opencode.mjs +65 -0
  225. data/plugin/hooks/woods-post-edit.sh +2 -225
  226. data/plugin/hooks/woods-refresh.sh +260 -0
  227. data/plugin/hooks/woods-session-start.sh +47 -55
  228. data/plugin/skills/woods-agent-enable/SKILL.md +19 -0
  229. data/plugin/skills/woods-diagnose/SKILL.md +319 -1
  230. data/plugin/skills/woods-investigate/SKILL.md +145 -0
  231. data/plugin/skills/woods-mcp-config/SKILL.md +90 -2
  232. data/plugin/skills/woods-setup/SKILL.md +110 -6
  233. metadata +87 -5
@@ -14,16 +14,20 @@ require_relative '../tasks'
14
14
  require_relative '../watch/status'
15
15
  require_relative '../filename_utils'
16
16
  require_relative '../update_check'
17
+ require_relative '../retrieval/source_evidence'
18
+ require_relative '../session_tracer/unit_resolver'
17
19
  require_relative 'bootstrap_state'
18
20
  require_relative 'errors'
19
21
  require_relative 'index_reader'
20
22
  require_relative 'index_reader_pinning'
23
+ require_relative 'initialization_guidance'
21
24
  require_relative 'protocol_policy'
22
25
  require_relative 'tasks/extension'
23
26
  require_relative 'tasks/request_capture'
24
27
  require_relative 'tasks/store'
25
28
  require_relative 'tool_contract'
26
29
  require_relative 'tool_response_renderer'
30
+ require_relative 'traversal_response'
27
31
  require_relative 'version_aware_tool_dispatch'
28
32
 
29
33
  module Woods
@@ -70,6 +74,7 @@ module Woods
70
74
  # controls that actually shrink the answer are `depth`, `types` and
71
75
  # `via`; `limit` and `offset` only page what those leave (B-183).
72
76
  DEFAULT_TRAVERSAL_LIMIT = 50
77
+ DEFAULT_GRAPH_ANALYSIS_LIMIT = 20
73
78
 
74
79
  class << self
75
80
  # Build a configured MCP::Server with all tools and resources.
@@ -90,6 +95,7 @@ module Woods
90
95
  def build(index_dir:, retriever: nil, operator: nil, feedback_store: nil, snapshot_store: nil,
91
96
  bootstrap_state: nil, response_format: nil, warmup: true, retriever_reloader: nil)
92
97
  reader = IndexReader.new(index_dir)
98
+ retriever.bind_reader(reader) if retriever.respond_to?(:bind_reader)
93
99
  reader.warmup! if warmup
94
100
  config = Woods.configuration
95
101
  format = response_format || (config.respond_to?(:context_format) ? config.context_format : nil) || :markdown
@@ -156,7 +162,10 @@ module Woods
156
162
  description: 'Traverse forward dependencies of a unit (what it depends on). ' \
157
163
  'Narrow with depth, types and via first: they shrink the answer, ' \
158
164
  'while limit and offset only page it. Returns a BFS tree with ' \
159
- "depth, bounded to #{DEFAULT_TRAVERSAL_LIMIT} nodes by default.",
165
+ "depth, paged to #{DEFAULT_TRAVERSAL_LIMIT} nodes by default. " \
166
+ 'max_nodes/max_edges bound the walk independently; partial_reason reports a budget cutoff and total_is_exact is false. ' \
167
+ 'Published relationships are not exhaustive source-reference coverage. ' \
168
+ 'Use explain:true for recorded directed relationships and bounded witnesses; ambiguous types remain explicit.',
160
169
  reader_method: :traverse_dependencies,
161
170
  render_key: :dependencies)
162
171
  define_traversal_tool(server, reader, respond, renderer,
@@ -164,7 +173,10 @@ module Woods
164
173
  description: 'Traverse reverse dependencies of a unit (what depends on it). ' \
165
174
  'Narrow with depth, types and via first: they shrink the answer, ' \
166
175
  'while limit and offset only page it. Returns a BFS tree with ' \
167
- "depth, bounded to #{DEFAULT_TRAVERSAL_LIMIT} nodes by default.",
176
+ "depth, paged to #{DEFAULT_TRAVERSAL_LIMIT} nodes by default. " \
177
+ 'max_nodes/max_edges bound the walk independently; partial_reason reports a budget cutoff and total_is_exact is false. ' \
178
+ 'Published relationships are not exhaustive source-reference coverage. ' \
179
+ 'Use explain:true for recorded directed relationships and bounded witnesses; ambiguous types remain explicit.',
168
180
  reader_method: :traverse_dependents,
169
181
  render_key: :dependents)
170
182
  define_structure_tool(server, reader, respond, renderer)
@@ -191,6 +203,7 @@ module Woods
191
203
  register_resource_handler(server, reader)
192
204
  ToolContract.apply!(server)
193
205
  IndexReaderPinning.install(server, reader: reader)
206
+ server.instructions = InitializationGuidance.for(server.tools.keys)
194
207
 
195
208
  # Last, after every conditional registration above — the whole point is
196
209
  # that a host with Notion wired advertises the same tool order as one
@@ -238,9 +251,9 @@ module Woods
238
251
  !token.nil? && ids && !ids.empty?
239
252
  end
240
253
 
241
- def text_response(text)
254
+ def text_response(text, data: nil)
242
255
  structured = { text: text }
243
- structured[:data] = JSON.parse(text)
256
+ structured[:data] = data.nil? ? JSON.parse(text) : data
244
257
  ::MCP::Tool::Response.new(
245
258
  [{ type: 'text', text: text }],
246
259
  structured_content: structured
@@ -384,7 +397,7 @@ module Woods
384
397
 
385
398
  sliced = offset.positive? ? original.drop(offset) : original
386
399
  container[key] = limit ? truncate_section(sliced, limit) : sliced
387
- if original.size > offset + (limit || original.size)
400
+ if offset.positive? || container[key].size < original.size
388
401
  container["#{key}_total"] = original.size
389
402
  container["#{key}_truncated"] = true
390
403
  end
@@ -394,12 +407,11 @@ module Woods
394
407
  # Page a traversal result's `nodes` hash in place, in BFS order.
395
408
  #
396
409
  # Mirrors {#paginate_section}'s metadata keys (`nodes_total`,
397
- # `nodes_truncated`, `nodes_offset`) so both renderers print the one
398
- # truncation line they already had for `graph_analysis`. A page that
399
- # holds every node adds no keys at all, so a small result renders
400
- # exactly as it did before the bound existed (B-183).
410
+ # `nodes_truncated`, `nodes_offset`). A page that holds every admitted
411
+ # node adds no pagination keys (B-183). TraversalResponse separately
412
+ # annotates scope and total exactness before pagination.
401
413
  #
402
- # `nodes_total` marks *any* partial answer, not only one with more
414
+ # `nodes_total` marks *any* paged answer, not only one with more
403
415
  # behind it. Keying it on `total > offset + limit` left the last page
404
416
  # of a walk indistinguishable from a complete one: 21 nodes of 121,
405
417
  # with nothing saying 100 were skipped. `nodes_truncated` still means
@@ -435,6 +447,11 @@ module Woods
435
447
  identifier: { type: 'string',
436
448
  description: 'Exact unit identifier (e.g. "Post", "PostsController", "Api::V1::HealthController")' },
437
449
  name: { type: 'string', description: 'Alias for `identifier`. Either one works.' },
450
+ type: { type: 'string', description: 'Optional actual published unit type, to disambiguate shared identifiers.' },
451
+ evidence: { type: 'string', enum: %w[full compact outline], description: 'Published evidence mode (default full).' },
452
+ query: { type: 'string', description: 'Optional relevance query for compact evidence; absent means API orientation.' },
453
+ budget: { type: 'integer', minimum: 1, description: 'Compact/outline token estimate budget (default 2000); full lookup remains complete.' },
454
+ source_sha256: { type: 'string', description: 'Require the published source SHA256 from an earlier excerpt; refuses changed source.' },
438
455
  include_source: { type: 'boolean', description: 'Include source_code in response (default: true)' },
439
456
  sections: {
440
457
  type: 'array', items: { type: 'string' },
@@ -445,7 +462,7 @@ module Woods
445
462
  # accepted alias. The handler validates that one of the two
446
463
  # was provided.
447
464
  }
448
- ) do |server_context:, identifier: nil, name: nil, include_source: nil, sections: nil|
465
+ ) do |server_context:, identifier: nil, name: nil, include_source: nil, sections: nil, type: nil, evidence: 'full', query: nil, budget: nil, source_sha256: nil|
449
466
  identifier ||= name
450
467
  if identifier.nil? || identifier.empty?
451
468
  next respond_err.call(
@@ -457,8 +474,30 @@ module Woods
457
474
  )
458
475
  end
459
476
  sections = coerce.call(sections)
460
- unit = reader.find_unit(identifier)
477
+ begin
478
+ Retrieval::SourceEvidence.validate_mode!(evidence)
479
+ if evidence != 'full' && (include_source == false || sections&.any?)
480
+ raise ArgumentError, 'compact/outline evidence cannot be combined with include_source: false or sections'
481
+ end
482
+ if evidence == 'full' && (!query.nil? || !budget.nil?)
483
+ raise ArgumentError, 'query and budget apply only to compact/outline evidence'
484
+ end
485
+ rescue ArgumentError => e
486
+ next respond_err.call(e.message, code: :unsupported_argument, tool: 'lookup', argument: 'evidence')
487
+ end
488
+ unit = type ? reader.find_unit(identifier, type: type) : reader.find_unit(identifier)
461
489
  if unit
490
+ if source_sha256 && Digest::SHA256.hexdigest(unit['source_code'].to_s) != source_sha256
491
+ next respond_err.call('Published source changed since the excerpt; retrieve fresh evidence before verification.',
492
+ code: :stale_index, tool: 'lookup', argument: 'source_sha256')
493
+ end
494
+ if evidence != 'full'
495
+ selected = Retrieval::SourceEvidence.new(unit: unit, query: query, generation: reader.loaded_generation)
496
+ .render(mode: evidence, budget: budget || 2000,
497
+ counter: ->(text) { (text.length / 4.0).ceil })
498
+ next ::MCP::Tool::Response.new([{ type: 'text', text: selected.text }],
499
+ structured_content: { text: selected.text, data: { evidence: selected.provenance } })
500
+ end
462
501
  always_include = %w[type identifier file_path namespace]
463
502
  filtered = unit
464
503
  filtered = filtered.except('source_code') if include_source == false
@@ -485,8 +524,9 @@ module Woods
485
524
  server.define_tool(
486
525
  name: 'search',
487
526
  description: 'Find code units whose identifiers (or source/metadata) match a regex. ' \
488
- 'Example: search("Worker|Job") returns all workers and jobs; search("^Post") ' \
489
- 'returns units starting with "Post". Returns [{identifier, type, match_field}]. ' \
527
+ 'Example: search("Worker|Job") finds workers and jobs; search("^Post") ' \
528
+ 'returns units starting with "Post". Returns [{identifier, type, match_field}] plus completeness. ' \
529
+ 'Check completeness before treating discovery as exhaustive; a limit is only a page size. ' \
490
530
  'Use `lookup` for exact identifiers, `dependencies`/`dependents` for graph traversal. ' \
491
531
  'Gotchas: query is a Ruby regex — literal pipe needs escaping as \\|; ' \
492
532
  'types restricts which index directories are scanned (e.g. ["mailer"] scans only ' \
@@ -500,6 +540,10 @@ module Woods
500
540
  type: 'array', items: { type: 'string' },
501
541
  description: 'Restrict scan to these unit types: model, controller, service, job, mailer, etc.'
502
542
  },
543
+ packages: { type: 'array', items: { type: 'string' },
544
+ description: 'Exact published package owners, OR within the list; AND with source_paths and types.' },
545
+ source_paths: { type: 'array', items: { type: 'string' },
546
+ description: 'Application-relative directory prefixes; segment-aware, OR within the list. Applied before limits.' },
503
547
  fields: {
504
548
  type: 'array', items: { type: 'string', enum: %w[identifier metadata source_code] },
505
549
  description: 'Fields to search: identifier (default), source_code, metadata'
@@ -517,7 +561,8 @@ module Woods
517
561
  }
518
562
  }
519
563
  }
520
- ) do |server_context:, query: nil, types: nil, fields: nil, limit: nil, exact_prefix: nil, exact_suffix: nil|
564
+ ) do |server_context:, query: nil, types: nil, fields: nil, limit: nil, exact_prefix: nil, exact_suffix: nil,
565
+ packages: nil, source_paths: nil|
521
566
  if (query.nil? || query.empty?) &&
522
567
  (exact_prefix.nil? || exact_prefix.empty?) &&
523
568
  (exact_suffix.nil? || exact_suffix.empty?)
@@ -538,17 +583,29 @@ module Woods
538
583
  fields: fields || %w[identifier],
539
584
  limit: limit || 20,
540
585
  exact_prefix: exact_prefix,
541
- exact_suffix: exact_suffix
586
+ exact_suffix: exact_suffix,
587
+ packages: packages, source_paths: source_paths
542
588
  )
543
589
  results = search_result[:results]
544
590
  payload = {
545
591
  query: query,
546
592
  result_count: results.size,
547
- results: results
593
+ results: results,
594
+ completeness: search_result[:completeness]
548
595
  }
596
+ payload[:applied_scope] = search_result[:applied_scope] if search_result[:applied_scope]
549
597
  payload[:note] = search_result[:note] if search_result[:note]
550
598
  payload[:partial] = true if search_result[:partial]
599
+ payload[:hint] = search_result[:hint] if search_result[:hint]
551
600
  respond.call(renderer.render(:search, payload))
601
+ rescue Retrieval::Scope::InvalidScopeError => e
602
+ respond_err.call(e.message, code: :unsupported_argument, tool: 'search', argument: 'scope')
603
+ rescue IOError, SystemCallError, JSON::ParserError, EncodingError
604
+ respond_err.call(
605
+ 'Search completeness: unknown (unreadable_or_corrupt_source). ' \
606
+ 'An Index artifact is unavailable or malformed; inspect woods_status and run woods:validate.',
607
+ code: :corrupt_artifact, tool: 'search', completeness: SearchResults.unavailable
608
+ )
552
609
  end
553
610
  end
554
611
 
@@ -563,6 +620,7 @@ module Woods
563
620
  properties: {
564
621
  identifier: { type: 'string', description: 'Unit identifier to start from' },
565
622
  depth: { type: 'integer', description: 'Maximum traversal depth (default: 2)' },
623
+ explain: { type: 'boolean', description: 'Include recorded relationship evidence and shared shortest witnesses (default: false)' },
566
624
  types: {
567
625
  type: 'array', items: { type: 'string' },
568
626
  description: 'Filter to these types'
@@ -580,23 +638,31 @@ module Woods
580
638
  },
581
639
  limit: { type: 'integer',
582
640
  description: "Maximum nodes to return (default: #{DEFAULT_TRAVERSAL_LIMIT})" },
583
- offset: { type: 'integer', description: 'Skip this many nodes (default: 0)' }
641
+ offset: { type: 'integer', description: 'Skip this many nodes (default: 0)' },
642
+ max_nodes: { type: 'integer', minimum: 1, maximum: 10_000,
643
+ description: 'Visited-node budget including root (default: 1000; maximum: 10000)' },
644
+ max_edges: { type: 'integer', minimum: 1, maximum: 100_000,
645
+ description: 'Edge-check budget before filters, including reverse via checks (default: 10000; maximum: 100000)' }
584
646
  },
585
647
  required: ['identifier']
586
648
  }
587
- ) do |identifier:, server_context:, depth: nil, types: nil, via: nil, limit: nil, offset: nil|
649
+ ) do |identifier:, server_context:, depth: nil, types: nil, via: nil, limit: nil, offset: nil, max_nodes: nil, max_edges: nil, explain: nil|
588
650
  types = coerce.call(types)
589
651
  via = coerce.call(via)
590
652
  depth = coerce_int.call(depth)
591
653
  limit = coerce_int.call(limit)
592
654
  offset = coerce_int.call(offset)
593
- result = reader.send(reader_method, identifier, depth: depth || 2, types: types, via: via)
655
+ result = reader.send(reader_method, identifier, depth: depth || 2, types: types, via: via,
656
+ max_nodes: coerce_int.call(max_nodes) || 1000,
657
+ max_edges: coerce_int.call(max_edges) || 10_000, explain: explain || false)
594
658
  if result[:found] == false
595
659
  result[:message] =
596
660
  "Identifier '#{identifier}' not found in the index. Use 'search' to find valid identifiers."
597
661
  end
662
+ TraversalResponse.annotate(result)
598
663
  paginate_nodes.call(result, limit || DEFAULT_TRAVERSAL_LIMIT, offset || 0)
599
- respond.call(renderer.render(render_key, result))
664
+ TraversalEvidencePage.apply(result)
665
+ respond.call(renderer.render(render_key, result), data: result)
600
666
  end
601
667
  end
602
668
 
@@ -636,32 +702,20 @@ module Woods
636
702
  enum: ToolResponseRenderer::GRAPH_ANALYSIS_SECTIONS + %w[all],
637
703
  description: 'Which analysis to return. Default: all'
638
704
  },
639
- limit: { type: 'integer', description: 'Limit results per section (default: 20)' },
705
+ limit: { type: 'integer', description: "Limit results per section (default: #{DEFAULT_GRAPH_ANALYSIS_LIMIT})" },
640
706
  offset: { type: 'integer', description: 'Skip this many results per section (default: 0)' }
641
707
  }
642
708
  }
643
709
  ) do |server_context:, analysis: nil, limit: nil, offset: nil|
644
- limit = coerce_int.call(limit)
710
+ limit = coerce_int.call(limit) || DEFAULT_GRAPH_ANALYSIS_LIMIT
645
711
  offset = coerce_int.call(offset)
646
712
  data = reader.graph_analysis
647
713
  section = analysis || 'all'
648
714
  effective_offset = offset || 0
649
715
 
650
- result = if section == 'all'
651
- if limit || effective_offset.positive?
652
- truncated = data.dup
653
- ToolResponseRenderer::GRAPH_ANALYSIS_SECTIONS.each do |key|
654
- paginate.call(truncated, key, limit, effective_offset)
655
- end
656
- truncated
657
- else
658
- data
659
- end
660
- else
661
- single = { section => data[section] || [], 'stats' => data['stats'] }
662
- paginate.call(single, section, limit, effective_offset) if limit || effective_offset.positive?
663
- single
664
- end
716
+ result = section == 'all' ? data.dup : { section => data[section] || [], 'stats' => data['stats'] }
717
+ sections = section == 'all' ? ToolResponseRenderer::GRAPH_ANALYSIS_SECTIONS : [section]
718
+ sections.each { |key| paginate.call(result, key, limit, effective_offset) }
665
719
 
666
720
  respond.call(renderer.render(:graph_analysis, result))
667
721
  end
@@ -874,13 +928,14 @@ module Woods
874
928
  coerce = method(:coerce_array)
875
929
  stale_check = method(:stale_index_result?)
876
930
  degraded_response = method(:degraded_retrieval_response)
931
+ retrieval_mode = retriever.respond_to?(:mode) ? retriever.mode : :semantic
877
932
  server.define_tool(
878
933
  name: 'codebase_retrieve',
879
- description: 'Semantic search: retrieve relevant code units for a natural-language question. ' \
934
+ description: 'Ranked retrieval: relevant code units for a natural-language question. ' \
880
935
  'Example: codebase_retrieve("how does billing work?") returns ranked source context. ' \
881
936
  'Returns a token-budgeted context string ready to paste into a prompt. ' \
882
937
  'Use `search` for exact name/pattern matching; use this for conceptual questions. ' \
883
- 'Requires an embedding provider — disabled if OPENAI_API_KEY is unset and Ollama is unreachable. ' \
938
+ 'Uses configured embeddings, or explicit WOODS_RETRIEVAL_MODE=lexical over extraction units. ' \
884
939
  'By default excludes test_mappings (~33% of a typical index) so spec filenames do not ' \
885
940
  'dominate semantic rank; pass types: ["test_mapping"] to opt back in. ' \
886
941
  'Parameter: use `budget` for the token budget (not `limit` — that means result count ' \
@@ -890,12 +945,15 @@ module Woods
890
945
  query: { type: 'string',
891
946
  description: 'Natural language question (e.g. "How does user authentication work?")' },
892
947
  budget: { type: 'integer',
893
- description: 'Token budget for context assembly (default: 8000).' },
948
+ description: 'Token budget for context assembly (configured max_context_tokens; otherwise 8000).' },
949
+ evidence: { type: 'string', enum: %w[full compact outline],
950
+ description: 'Explicit complete spans or API outline within each ranked unit; default full retains existing output.' },
894
951
  types: {
895
952
  type: 'array', items: { type: 'string' },
896
953
  description: 'Restrict results to these unit types (model, controller, service, job, mailer, ' \
897
954
  'rails_source, test_mapping, etc.). Overrides the default test_mapping exclusion. ' \
898
- 'When the unfiltered top-K has no candidate of a requested type, the retriever ' \
955
+ 'Lexical mode and explicit package/path scopes filter before limits and omit the global rank table. ' \
956
+ 'In semantic mode, when the unfiltered top-K has no requested type, the retriever ' \
899
957
  'falls back to rank-within-type so the response is populated whenever units of ' \
900
958
  'the requested type exist in the index. The response appends a "Type rank ' \
901
959
  'context" table with per-type: source, rank in unfiltered top-K, global_k, ' \
@@ -904,6 +962,10 @@ module Woods
904
962
  '(index has this type but other requested types filled the result), absent ' \
905
963
  '(zero units of this type in the index).'
906
964
  },
965
+ packages: { type: 'array', items: { type: 'string' },
966
+ description: 'Exact published nearest package owners. OR within the list; AND with paths and type eligibility.' },
967
+ source_paths: { type: 'array', items: { type: 'string' },
968
+ description: 'Application-relative directory prefixes. Scope applies before candidate limits; graph expansion stays inside it.' },
907
969
  exclude_types: {
908
970
  type: 'array', items: { type: 'string' },
909
971
  description: 'Additional types to exclude on top of the default test_mapping exclusion.'
@@ -911,7 +973,7 @@ module Woods
911
973
  },
912
974
  required: ['query']
913
975
  }
914
- ) do |query:, server_context:, budget: nil, limit: nil, types: nil, exclude_types: nil|
976
+ ) do |query:, server_context:, budget: nil, limit: nil, types: nil, exclude_types: nil, packages: nil, source_paths: nil, evidence: 'full'|
915
977
  # `limit` isn't declared in the schema but clients still send it
916
978
  # because sibling tools (search, recent_changes, pagerank) use
917
979
  # `limit` as a result count. Mapping it to `budget` here would
@@ -919,10 +981,10 @@ module Woods
919
981
  # budget). Surface a helpful typed error instead.
920
982
  unless limit.nil?
921
983
  next respond_err.call(
922
- 'codebase_retrieve uses `budget` (token budget, default 8000), not `limit`. ' \
984
+ 'codebase_retrieve uses `budget` (token budget, configured default), not `limit`. ' \
923
985
  '`limit` is the result-count parameter on sibling tools (search, recent_changes, pagerank). ' \
924
986
  "Pass `budget: #{coerce_int.call(limit)}` if you meant a #{coerce_int.call(limit)}-token context, " \
925
- 'or drop the kwarg entirely for the default 8000.',
987
+ 'or drop the kwarg entirely for the configured default.',
926
988
  code: :unsupported_argument,
927
989
  tool: 'codebase_retrieve',
928
990
  argument: 'limit',
@@ -931,6 +993,11 @@ module Woods
931
993
  )
932
994
  end
933
995
 
996
+ begin
997
+ Retrieval::SourceEvidence.validate_mode!(evidence)
998
+ rescue ArgumentError => e
999
+ next respond_err.call(e.message, code: :unsupported_argument, tool: 'codebase_retrieve', argument: 'evidence')
1000
+ end
934
1001
  budget = coerce_int.call(budget)
935
1002
  types = coerce.call(types)
936
1003
  exclude_types = coerce.call(exclude_types)
@@ -949,12 +1016,20 @@ module Woods
949
1016
  end
950
1017
  if retriever
951
1018
  begin
1019
+ scope_options = if Retrieval::Scope.requested?(packages: packages, source_paths: source_paths)
1020
+ { packages: packages, source_paths: source_paths }
1021
+ else
1022
+ {}
1023
+ end
1024
+ scope_options[:evidence] = evidence unless evidence == 'full'
952
1025
  result = retriever.retrieve(
953
1026
  query,
954
- budget: budget || 8000,
1027
+ budget: budget || (retriever.respond_to?(:default_budget) ? retriever.default_budget : 8000),
955
1028
  types: types,
956
- exclude_types: exclude_types
1029
+ exclude_types: exclude_types, **scope_options
957
1030
  )
1031
+ rescue Retrieval::Scope::InvalidScopeError => e
1032
+ next respond_err.call(e.message, code: :unsupported_argument, tool: 'codebase_retrieve', argument: 'scope')
958
1033
  rescue Woods::Retriever::StoreError => e
959
1034
  # M8: a metadata-store failure mid-query must not surface as
960
1035
  # a raw raise through the tool boundary (or as the misleading
@@ -963,7 +1038,7 @@ module Woods
963
1038
  respond_err,
964
1039
  reason: e.message,
965
1040
  stores: [e.store],
966
- phase: 'query'
1041
+ phase: 'query', mode: retrieval_mode
967
1042
  )
968
1043
  end
969
1044
  if stale_check.call(result)
@@ -975,12 +1050,22 @@ module Woods
975
1050
  tool: 'codebase_retrieve'
976
1051
  )
977
1052
  end
978
- respond.call(result.context)
1053
+ if evidence != 'full' || (result.respond_to?(:applied_scope) && result.applied_scope)
1054
+ ::MCP::Tool::Response.new(
1055
+ [{ type: 'text', text: result.context }],
1056
+ structured_content: { text: result.context, data: { applied_scope: result.applied_scope, sources: result.sources } },
1057
+ meta: { applied_scope: result.applied_scope }
1058
+ )
1059
+ else
1060
+ respond.call(result.context)
1061
+ end
979
1062
  else
980
1063
  respond_err.call(
981
1064
  'Semantic search is disabled — no embedding provider is configured. ' \
982
1065
  'To enable: set OPENAI_API_KEY, or run Ollama locally ' \
983
1066
  '(brew install ollama && ollama serve && ollama pull nomic-embed-text). ' \
1067
+ 'For ranked discovery with no embeddings, set WOODS_RETRIEVAL_MODE=lexical in the MCP process ' \
1068
+ 'environment and restart the server. See docs/RETRIEVAL_GUIDE.md#embedding-free-lexical-retrieval. ' \
984
1069
  'Use the `search` tool for pattern-based matching in the meantime.',
985
1070
  code: :not_configured,
986
1071
  config_key: 'embedding_provider',
@@ -1020,7 +1105,16 @@ module Woods
1020
1105
  # @param phase [String] 'boot' (hydration failure) or 'query'
1021
1106
  # (store failure at query time)
1022
1107
  # @return [MCP::Tool::Response]
1023
- def degraded_retrieval_response(respond_err, reason:, stores:, phase:)
1108
+ def degraded_retrieval_response(respond_err, reason:, stores:, phase:, mode: :semantic)
1109
+ if mode == :lexical
1110
+ return respond_err.call(
1111
+ "Lexical retrieval is degraded: #{reason}. No partial lexical snapshot was served. " \
1112
+ 'Inspect woods_status and repair or re-extract the published index, then retry.',
1113
+ code: :degraded_index, tool: 'codebase_retrieve', degraded: true,
1114
+ phase: phase, stores: stores, reason: reason, mode: 'lexical'
1115
+ )
1116
+ end
1117
+
1024
1118
  respond_err.call(
1025
1119
  "Semantic search is degraded: #{reason}. The affected store(s) return no data, so " \
1026
1120
  'queries would come back empty — this is NOT "no results". ' \
@@ -1129,6 +1223,9 @@ module Woods
1129
1223
  )
1130
1224
  doc = assembler.assemble(session_id, budget: budget || 8000, depth: depth || 1)
1131
1225
  respond.call(doc.to_markdown)
1226
+ rescue Woods::SessionTracer::AmbiguousUnitError => e
1227
+ respond_err.call(e.message, code: :ambiguous_identity, tool: 'session_trace',
1228
+ identifier: e.identifier, types: e.types)
1132
1229
  rescue StandardError => e
1133
1230
  respond_err.call(
1134
1231
  "Session trace failed: #{e.message}",
@@ -1979,13 +2076,17 @@ module Woods
1979
2076
  description: 'Diagnose whether the Woods index and server are healthy. Returns extraction metadata ' \
1980
2077
  '(last run, unit counts, git SHA, staleness in seconds), retriever/embedding configuration, ' \
1981
2078
  'bootstrap state (hydrated / degraded / failed + reason), feature flags, and a ready flag. ' \
1982
- 'Call this first on cold connect to learn what the server knows.',
1983
- input_schema: { type: 'object', properties: {} }
1984
- ) do |server_context:|
2079
+ 'Includes source-content freshness; quick scans have a 250ms budget, explicit deep scans have 5s. ' \
2080
+ 'Incomplete evidence is unknown. Call this first on cold connect.',
2081
+ input_schema: { type: 'object', properties: {
2082
+ source_check: { type: 'string', enum: %w[quick deep], default: 'quick',
2083
+ description: 'Bounded source content verification: quick (250ms) or deep (5s).' }
2084
+ } }
2085
+ ) do |server_context:, source_check: 'quick'|
1985
2086
  _ = server_context
1986
2087
  status = Woods::MCP::Server.build_status(
1987
2088
  reader: reader, retriever: retriever, index_dir: index_dir,
1988
- bootstrap_state: bootstrap_state
2089
+ bootstrap_state: bootstrap_state, source_check: source_check
1989
2090
  )
1990
2091
  respond.call(JSON.pretty_generate(status))
1991
2092
  end
@@ -2004,21 +2105,21 @@ module Woods
2004
2105
  # provider in use. Without this, operators debugging "wrong provider" see
2005
2106
  # status claiming +embedding_model: "text-embedding-3-small"+ next to
2006
2107
  # +embedding_provider: "ollama"+ and reasonably distrust every field.
2007
- def build_status(reader:, retriever:, index_dir:, bootstrap_state: nil)
2108
+ def build_status(reader:, retriever:, index_dir:, bootstrap_state: nil, source_check: 'quick')
2008
2109
  # Pin the generation across the whole payload. Without this the
2009
2110
  # manifest can be read at generation N and `generation_fields` then
2010
2111
  # report N+1 — a status report that describes counts from one index
2011
2112
  # while announcing the number of another, which is precisely the
2012
2113
  # confusion this tool exists to resolve.
2013
- return build_status_payload(reader, retriever, index_dir, bootstrap_state) unless
2114
+ return build_status_payload(reader, retriever, index_dir, bootstrap_state, source_check) unless
2014
2115
  reader.respond_to?(:with_pinned_generation)
2015
2116
 
2016
2117
  reader.with_pinned_generation do
2017
- build_status_payload(reader, retriever, index_dir, bootstrap_state)
2118
+ build_status_payload(reader, retriever, index_dir, bootstrap_state, source_check)
2018
2119
  end
2019
2120
  end
2020
2121
 
2021
- def build_status_payload(reader, retriever, index_dir, bootstrap_state)
2122
+ def build_status_payload(reader, retriever, index_dir, bootstrap_state, source_check)
2022
2123
  manifest = safe_manifest(reader)
2023
2124
  extracted_at = manifest && manifest['extracted_at']
2024
2125
  staleness = staleness_seconds(extracted_at)
@@ -2036,14 +2137,15 @@ module Woods
2036
2137
  index_dir: index_dir.to_s,
2037
2138
  update: Woods::UpdateCheck.status_hash
2038
2139
  },
2039
- index: index_section(manifest, extracted_at, staleness, index_dir, reader),
2140
+ index: index_section(manifest, extracted_at, staleness, index_dir, reader, source_check),
2040
2141
  watch: watch_section(index_dir),
2041
2142
  retriever: {
2042
2143
  configured: !retriever.nil?,
2043
- class: retriever&.class&.name
2144
+ class: retriever&.class&.name,
2145
+ **(retriever.respond_to?(:mode) && retriever.mode == :lexical ? { mode: 'lexical' } : {})
2044
2146
  },
2045
2147
  bootstrap: bootstrap_state&.to_h,
2046
- features: features_from(config, resolved)
2148
+ features: retrieval_features(config, resolved, retriever)
2047
2149
  }
2048
2150
  end
2049
2151
 
@@ -2066,10 +2168,11 @@ module Woods
2066
2168
  # diff directly. This is an observability signal, not a hard gate —
2067
2169
  # hard-refusing responses would be much more disruptive than a loudly-
2068
2170
  # visible staleness flag that agents can branch on.
2069
- def index_section(manifest, extracted_at, staleness, index_dir, reader = nil)
2171
+ def index_section(manifest, extracted_at, staleness, index_dir, reader = nil, source_check = 'quick')
2070
2172
  base = {
2071
2173
  extracted_at: extracted_at,
2072
2174
  staleness_seconds: staleness,
2175
+ woods_version: manifest && manifest['woods_version'],
2073
2176
  rails_version: manifest && manifest['rails_version'],
2074
2177
  ruby_version: manifest && manifest['ruby_version'],
2075
2178
  total_units: manifest && manifest['total_units'],
@@ -2080,6 +2183,11 @@ module Woods
2080
2183
  schema_sha: manifest && manifest['schema_sha']
2081
2184
  }
2082
2185
 
2186
+ base[:source_freshness] = if reader.respond_to?(:source_freshness)
2187
+ reader.source_freshness(mode: source_check)
2188
+ else
2189
+ { 'state' => 'unknown', 'reasons' => ['source_reader_unavailable'], 'complete' => false }
2190
+ end
2083
2191
  base.merge!(generation_fields(index_dir, reader))
2084
2192
  base.merge!(working_tree_fields(index_dir))
2085
2193
 
@@ -2210,7 +2318,7 @@ module Woods
2210
2318
  record = JSON.parse(Woods::AtomicFile.read(path))
2211
2319
  # `state` is whatever the daemon last wrote, and a `kill -9`'d daemon
2212
2320
  # leaves `running` behind forever. `alive?` adds the two checks that
2213
- # catch that — the pid still exists and the record is recent — so the
2321
+ # catch that — a recent record and, for local hosts, a live pid — so the
2214
2322
  # payload can distinguish "maintaining this index" from "claimed to be,
2215
2323
  # once". Reported as a separate field rather than by overwriting
2216
2324
  # `state`, because the recorded state and the liveness verdict answer
@@ -2263,6 +2371,13 @@ module Woods
2263
2371
  # historic status payloads always reported +false+ regardless of the
2264
2372
  # actual console MCP state. Advertising a misleading field is worse
2265
2373
  # than not advertising it at all.
2374
+ def retrieval_features(config, resolved, retriever)
2375
+ features = features_from(config, resolved)
2376
+ return features unless retriever.respond_to?(:mode) && retriever.mode == :lexical
2377
+
2378
+ features.merge(retrieval_mode: 'lexical', embedding_model: nil, embedding_provider: nil, vector_store: nil)
2379
+ end
2380
+
2266
2381
  def features_from(config, resolved)
2267
2382
  provider_hash = resolved&.embedding_provider || {}
2268
2383
  resolved_provider = resolved_provider_symbol(provider_hash[:class])
@@ -11,6 +11,8 @@ module Woods
11
11
  TASK_RESULT_TOOLS = %w[pipeline_embed pipeline_extract].freeze
12
12
 
13
13
  INTEGER_BOUNDS = {
14
+ 'max_nodes' => [1, 10_000],
15
+ 'max_edges' => [1, 100_000],
14
16
  'budget' => [1, 200_000],
15
17
  'depth' => [0, 20],
16
18
  'limit' => [1, 1_000],
@@ -25,7 +27,7 @@ module Woods
25
27
  text: { type: 'string' },
26
28
  data: {
27
29
  type: %w[object array string number boolean null],
28
- description: 'Parsed JSON payload when the selected renderer emits JSON.'
30
+ description: 'Structured tool payload, including traversal data in every renderer; otherwise parsed JSON when available.'
29
31
  }
30
32
  },
31
33
  required: ['text'],
@@ -1,5 +1,8 @@
1
1
  # frozen_string_literal: true
2
2
 
3
+ require_relative 'traversal_evidence_index'
4
+ require_relative 'traversal_evidence_text'
5
+
3
6
  module Woods
4
7
  module MCP
5
8
  # Base class for rendering MCP tool responses in different output formats.
@@ -67,6 +70,44 @@ module Woods
67
70
 
68
71
  private
69
72
 
73
+ def traversal_coverage_lines(data)
74
+ coverage = fetch_key(data, :graph_coverage)
75
+ notice = fetch_key(coverage, :notice) if coverage.is_a?(Hash)
76
+ notice ? [notice] : []
77
+ end
78
+
79
+ def traversal_lower_bound_note(data, shown)
80
+ return unless fetch_key(data, :total_is_exact) == false
81
+
82
+ total = fetch_key(data, :nodes_total, shown)
83
+ offset = fetch_key(data, :nodes_offset, 0)
84
+ position = offset.positive? ? " from offset #{offset}" : ''
85
+ "Showing #{shown} of at least #{total}#{position} " \
86
+ "(total unknown: #{fetch_key(data, :partial_reason)})."
87
+ end
88
+
89
+ def search_completeness_lines(data)
90
+ evidence = fetch_key(data, :completeness)
91
+ return [] unless evidence.is_a?(Hash)
92
+
93
+ more = { true => 'yes', false => 'no', nil => 'unknown' }.fetch(fetch_key(evidence, :has_more))
94
+ total = fetch_key(evidence, :total_matches)
95
+ lines = [
96
+ "Search completeness: #{fetch_key(evidence, :status)} (#{fetch_key(evidence, :reason)}).",
97
+ "More matches: #{more}; total matches: #{total.nil? ? 'unknown' : total}; " \
98
+ "matched lower bound: #{fetch_key(evidence, :matched_lower_bound)}."
99
+ ]
100
+ scope = fetch_key(data, :applied_scope)
101
+ if scope
102
+ lines << "Applied scope: packages=#{fetch_key(scope, :packages).inspect}; " \
103
+ "source_paths=#{fetch_key(scope, :source_paths).inspect}; " \
104
+ "eligible units=#{fetch_key(scope, :eligible_units)}."
105
+ end
106
+ hint = fetch_key(data, :hint)
107
+ lines << hint if hint
108
+ lines
109
+ end
110
+
70
111
  # Fetch a value from a hash by symbol or string key, falling back to a default.
71
112
  #
72
113
  # Handles data hashes that may use either symbol or string keys (e.g., data