woods 2.0.0.beta3 → 2.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +500 -420
- data/CONTRIBUTING.md +29 -17
- data/README.md +78 -178
- data/docs/AGENT_GUIDE.md +52 -11
- data/docs/AGENT_SETUP.md +34 -17
- data/docs/AUTOMATIC_MAINTENANCE.md +222 -0
- data/docs/BACKEND_MATRIX.md +18 -7
- data/docs/CLIENT_HOOKS.md +1 -1
- data/docs/CONFIGURATION_REFERENCE.md +105 -29
- data/docs/CONSOLE_MCP_SETUP.md +54 -9
- data/docs/DOCKER_SETUP.md +16 -1
- data/docs/EVALUATION.md +10 -4
- data/docs/EXTRACTOR_REFERENCE.md +23 -3
- data/docs/FAQ.md +14 -3
- data/docs/GETTING_STARTED.md +18 -17
- data/docs/INCREMENTAL_EXTRACTION.md +37 -8
- data/docs/INDEX_LAYOUT.md +2 -2
- data/docs/MCP_SERVERS.md +79 -7
- data/docs/MCP_TOOL_COOKBOOK.md +5 -5
- data/docs/MCP_WORKTREE_SETUP.md +55 -83
- data/docs/PUBLISHED_INDEX.md +17 -0
- data/docs/README.md +2 -1
- data/docs/RETRIEVAL_GUIDE.md +81 -13
- data/docs/SOURCE_FRESHNESS.md +1 -1
- data/docs/TOKEN_BENCHMARK.md +16 -10
- data/docs/TROUBLESHOOTING.md +142 -47
- data/docs/UPGRADING_TO_2.md +12 -6
- data/docs/WATCH_DAEMON.md +189 -24
- data/docs/WHY_WOODS.md +9 -5
- data/exe/woods-console +13 -11
- data/exe/woods-mcp-start +14 -9
- data/exe/woods-watch +5 -0
- data/lib/generators/woods/pgvector_generator.rb +8 -2
- data/lib/generators/woods/watch_generator.rb +53 -0
- data/lib/puma/plugin/woods.rb +10 -0
- data/lib/tasks/woods.rake +14 -0
- data/lib/woods/agent_configuration/applier.rb +5 -3
- data/lib/woods/agent_configuration/cli.rb +2 -2
- data/lib/woods/agent_configuration/layout.rb +13 -0
- data/lib/woods/cache/cache_middleware.rb +6 -0
- data/lib/woods/console/credential_scanner.rb +4 -3
- data/lib/woods/console/dispatch_pipeline.rb +7 -0
- data/lib/woods/console/embedded_executor.rb +31 -9
- data/lib/woods/console/sql_noise_stripper.rb +9 -7
- data/lib/woods/console/sql_table_scanner.rb +47 -7
- data/lib/woods/console/sql_validator.rb +49 -9
- data/lib/woods/console/sqlite_read_guard.rb +46 -0
- data/lib/woods/console/stdio_transport.rb +27 -0
- data/lib/woods/coordination/pipeline_lock.rb +3 -2
- data/lib/woods/embedding/indexer.rb +24 -14
- data/lib/woods/extractor.rb +70 -19
- data/lib/woods/extractors/declared_parent.rb +55 -0
- data/lib/woods/extractors/graphql_extractor.rb +2 -11
- data/lib/woods/extractors/lib_extractor.rb +10 -8
- data/lib/woods/extractors/mailer_extractor.rb +6 -10
- data/lib/woods/extractors/model_extractor.rb +1 -15
- data/lib/woods/extractors/poro_extractor.rb +10 -8
- data/lib/woods/extractors/shared_utility_methods.rb +22 -5
- data/lib/woods/git_command.rb +6 -7
- data/lib/woods/git_provenance.rb +4 -6
- data/lib/woods/mcp/bearer_auth.rb +2 -1
- data/lib/woods/mcp/bootstrapper.rb +20 -5
- data/lib/woods/mcp/config_resolver.rb +2 -1
- data/lib/woods/mcp/index_reader.rb +11 -2
- data/lib/woods/mcp/initialization_guidance.rb +1 -1
- data/lib/woods/mcp/renderers/markdown_renderer.rb +14 -8
- data/lib/woods/mcp/renderers/plain_renderer.rb +11 -7
- data/lib/woods/mcp/server.rb +63 -37
- data/lib/woods/mcp/tool_contract.rb +1 -1
- data/lib/woods/mcp/tool_response_renderer.rb +16 -0
- data/lib/woods/mcp/traversal_evidence_text.rb +1 -1
- data/lib/woods/mcp/traversal_response.rb +22 -0
- data/lib/woods/path_dispatcher.rb +6 -5
- data/lib/woods/published_index/typed_unit_reader.rb +40 -3
- data/lib/woods/published_index.rb +2 -2
- data/lib/woods/rake_helpers.rb +2 -12
- data/lib/woods/retrieval/corpus_status.rb +46 -0
- data/lib/woods/retrieval/lexical_assembler.rb +14 -3
- data/lib/woods/retrieval/lexical_index.rb +2 -1
- data/lib/woods/retriever.rb +19 -7
- data/lib/woods/session_tracer/file_store.rb +6 -1
- data/lib/woods/source_inputs/consumer_errors.rb +4 -0
- data/lib/woods/storage/local_corpus_stats.rb +32 -0
- data/lib/woods/storage/metadata_store.rb +20 -0
- data/lib/woods/storage/pgvector.rb +6 -2
- data/lib/woods/storage/vector_store.rb +10 -0
- data/lib/woods/temporal/json_snapshot_store.rb +35 -7
- data/lib/woods/version.rb +1 -1
- data/lib/woods/watch/child_environment.rb +30 -0
- data/lib/woods/watch/cli.rb +91 -0
- data/lib/woods/watch/daemon.rb +73 -11
- data/lib/woods/watch/event_stream.rb +70 -0
- data/lib/woods/watch/guardian.rb +142 -0
- data/lib/woods/watch/installation/layout.rb +70 -0
- data/lib/woods/watch/installation/options.rb +128 -0
- data/lib/woods/watch/installation/planner.rb +128 -0
- data/lib/woods/watch/installation/probe.rb +101 -0
- data/lib/woods/watch/installation/receipt.rb +77 -0
- data/lib/woods/watch/installation/recovery.rb +64 -0
- data/lib/woods/watch/installation/templates.rb +58 -0
- data/lib/woods/watch/installation.rb +56 -0
- data/lib/woods/watch/lifecycle.rb +182 -0
- data/lib/woods/watch/managed_child.rb +113 -0
- data/lib/woods/watch/managed_cleanup.rb +48 -0
- data/lib/woods/watch/managed_process.rb +144 -0
- data/lib/woods/watch/puma_adapter.rb +87 -0
- data/lib/woods/watch/puma_child.rb +66 -0
- data/lib/woods/watch/supervision_records.rb +95 -0
- data/lib/woods/watch/supervision_status.rb +104 -0
- data/lib/woods/watch/supervisor.rb +161 -0
- data/lib/woods/watch/supervisor_reporting.rb +46 -0
- data/plugin/.claude-plugin/plugin.json +1 -1
- data/plugin/hooks/woods-input-rules.sh +4 -4
- data/plugin/skills/woods-agent-enable/SKILL.md +7 -1
- data/plugin/skills/woods-diagnose/SKILL.md +134 -34
- data/plugin/skills/woods-investigate/SKILL.md +54 -15
- data/plugin/skills/woods-mcp-config/SKILL.md +38 -11
- data/plugin/skills/woods-setup/SKILL.md +72 -15
- metadata +38 -5
|
@@ -11,7 +11,7 @@ module Woods
|
|
|
11
11
|
|
|
12
12
|
WORKFLOW = <<~TEXT
|
|
13
13
|
Start with woods_status: check index readiness, generation freshness and relevant type counts before relying on results.
|
|
14
|
-
For exact names, discover identifiers with search (prefer literal exact_prefix/exact_suffix), then inspect with lookup. For conceptual questions,
|
|
14
|
+
For exact names, discover identifiers with search (prefer literal exact_prefix/exact_suffix), then inspect with lookup. For conceptual questions, check retrieval mode and data in woods_status before codebase_retrieve; structural readiness alone is insufficient. Otherwise use search and lookup.
|
|
15
15
|
Follow dependencies or dependents at depth 1 or 2; narrow types and via before paging. Respect partial results and limits: a missing match is not proof of absence, and a partial traversal does not establish every dependent or leaf.
|
|
16
16
|
Verify important conclusions against current source and tests. Recorded relationships and inferred downstream impact do not prove runtime execution or test coverage.
|
|
17
17
|
Registration does not authorize extraction, configuration changes, or live Console access. Use only the tools registered here and operate within the user's authorized scope.
|
|
@@ -152,11 +152,11 @@ module Woods
|
|
|
152
152
|
- **units_indexed** (manifest.json, `structure` tool) — total
|
|
153
153
|
ExtractedUnits written by the extractor. Canonical count.
|
|
154
154
|
- **graph_nodes** (`pagerank`, `dependencies`, `dependents`) —
|
|
155
|
-
units present in the dependency graph
|
|
156
|
-
|
|
155
|
+
units present in the dependency graph, including isolated units
|
|
156
|
+
with no incoming or outgoing edges.
|
|
157
157
|
- **searchable_entries** (`codebase_retrieve`) — retriever-store
|
|
158
|
-
entries
|
|
159
|
-
|
|
158
|
+
entries. Semantic mode may include per-chunk rows; lexical mode
|
|
159
|
+
ranks published units. Coverage depends on the configured store.
|
|
160
160
|
GLOSSARY
|
|
161
161
|
end
|
|
162
162
|
|
|
@@ -177,7 +177,7 @@ module Woods
|
|
|
177
177
|
|
|
178
178
|
GRAPH_ANALYSIS_SECTIONS.each do |section|
|
|
179
179
|
items = fetch_key(data, section)
|
|
180
|
-
next unless items.is_a?(Array) && items.any?
|
|
180
|
+
next unless items.is_a?(Array) && (items.any? || fetch_key(data, "#{section}_total"))
|
|
181
181
|
|
|
182
182
|
lines << "### #{section.tr('_', ' ').capitalize}"
|
|
183
183
|
lines << ''
|
|
@@ -194,11 +194,11 @@ module Woods
|
|
|
194
194
|
end
|
|
195
195
|
|
|
196
196
|
total_key = "#{section}_total"
|
|
197
|
-
if data
|
|
197
|
+
if fetch_key(data, total_key)
|
|
198
198
|
lines << ''
|
|
199
199
|
offset = fetch_key(data, "#{section}_offset", 0)
|
|
200
200
|
position = offset.positive? ? " from offset #{offset}" : ''
|
|
201
|
-
lines << "_Showing #{items.size} of #{data
|
|
201
|
+
lines << "_Showing #{items.size} of #{fetch_key(data, total_key)}#{position} (truncated)_"
|
|
202
202
|
end
|
|
203
203
|
lines << ''
|
|
204
204
|
end
|
|
@@ -417,6 +417,8 @@ module Woods
|
|
|
417
417
|
return lines.join("\n").rstrip
|
|
418
418
|
end
|
|
419
419
|
|
|
420
|
+
lines.concat(traversal_coverage_lines(data))
|
|
421
|
+
|
|
420
422
|
nodes.each do |id, info|
|
|
421
423
|
depth = fetch_key(info, :depth) || 0
|
|
422
424
|
deps = fetch_key(info, :deps, [])
|
|
@@ -438,7 +440,11 @@ module Woods
|
|
|
438
440
|
if fetch_key(data, :partial)
|
|
439
441
|
lines << "Partial traversal (#{fetch_key(data, :partial_reason)}); narrow depth/types/via or increase max_nodes/max_edges."
|
|
440
442
|
end
|
|
441
|
-
|
|
443
|
+
if (note = traversal_lower_bound_note(data, nodes.size))
|
|
444
|
+
lines << '' << "_#{note}_"
|
|
445
|
+
elsif fetch_key(data, :nodes_total)
|
|
446
|
+
lines << '' << truncation_note(data, nodes.size)
|
|
447
|
+
end
|
|
442
448
|
|
|
443
449
|
lines.join("\n").rstrip
|
|
444
450
|
end
|
|
@@ -126,9 +126,9 @@ module Woods
|
|
|
126
126
|
lines << 'Denominators:'
|
|
127
127
|
lines << ' units_indexed (manifest, structure): total ExtractedUnits written.'
|
|
128
128
|
lines << ' graph_nodes (pagerank, dependencies, dependents): units in the graph'
|
|
129
|
-
lines << ' (
|
|
130
|
-
lines << ' searchable_entries (codebase_retrieve): retriever-store entries
|
|
131
|
-
lines << ' per-chunk rows.
|
|
129
|
+
lines << ' (includes isolated units with no incoming/outgoing edges).'
|
|
130
|
+
lines << ' searchable_entries (codebase_retrieve): retriever-store entries; semantic mode may include'
|
|
131
|
+
lines << ' per-chunk rows; lexical mode ranks published units. Store coverage varies.'
|
|
132
132
|
|
|
133
133
|
lines.join("\n").rstrip
|
|
134
134
|
end
|
|
@@ -146,7 +146,7 @@ module Woods
|
|
|
146
146
|
|
|
147
147
|
GRAPH_ANALYSIS_SECTIONS.each do |section|
|
|
148
148
|
items = fetch_key(data, section)
|
|
149
|
-
next unless items.is_a?(Array) && items.any?
|
|
149
|
+
next unless items.is_a?(Array) && (items.any? || fetch_key(data, "#{section}_total"))
|
|
150
150
|
|
|
151
151
|
lines << "#{section.tr('_', ' ').upcase}:"
|
|
152
152
|
items.each do |item|
|
|
@@ -161,9 +161,9 @@ module Woods
|
|
|
161
161
|
|
|
162
162
|
total_key = "#{section}_total"
|
|
163
163
|
offset = fetch_key(data, "#{section}_offset", 0)
|
|
164
|
-
if data
|
|
164
|
+
if fetch_key(data, total_key)
|
|
165
165
|
position = offset.positive? ? " from offset #{offset}" : ''
|
|
166
|
-
lines << " (showing #{items.size} of #{data
|
|
166
|
+
lines << " (showing #{items.size} of #{fetch_key(data, total_key)}#{position}; truncated)"
|
|
167
167
|
end
|
|
168
168
|
lines << ''
|
|
169
169
|
end
|
|
@@ -279,6 +279,8 @@ module Woods
|
|
|
279
279
|
return lines.join("\n").rstrip
|
|
280
280
|
end
|
|
281
281
|
|
|
282
|
+
lines.concat(traversal_coverage_lines(data))
|
|
283
|
+
|
|
282
284
|
nodes.each do |id, info|
|
|
283
285
|
depth = fetch_key(info, :depth) || 0
|
|
284
286
|
deps = fetch_key(info, :deps, [])
|
|
@@ -294,7 +296,9 @@ module Woods
|
|
|
294
296
|
if fetch_key(data, :partial)
|
|
295
297
|
lines << "Partial traversal (#{fetch_key(data, :partial_reason)}); narrow depth/types/via or increase max_nodes/max_edges."
|
|
296
298
|
end
|
|
297
|
-
if
|
|
299
|
+
if (note = traversal_lower_bound_note(data, nodes.size))
|
|
300
|
+
lines << note
|
|
301
|
+
elsif fetch_key(data, :nodes_total)
|
|
298
302
|
offset = fetch_key(data, :nodes_offset, 0)
|
|
299
303
|
position = offset.positive? ? " from offset #{offset}" : ''
|
|
300
304
|
lines << " (showing #{nodes.size} of #{fetch_key(data, :nodes_total)}#{position}; truncated)"
|
data/lib/woods/mcp/server.rb
CHANGED
|
@@ -12,6 +12,7 @@ require_relative '../atomic_file'
|
|
|
12
12
|
require_relative '../generation'
|
|
13
13
|
require_relative '../tasks'
|
|
14
14
|
require_relative '../watch/status'
|
|
15
|
+
require_relative '../watch/supervision_status'
|
|
15
16
|
require_relative '../filename_utils'
|
|
16
17
|
require_relative '../update_check'
|
|
17
18
|
require_relative '../retrieval/source_evidence'
|
|
@@ -27,6 +28,7 @@ require_relative 'tasks/request_capture'
|
|
|
27
28
|
require_relative 'tasks/store'
|
|
28
29
|
require_relative 'tool_contract'
|
|
29
30
|
require_relative 'tool_response_renderer'
|
|
31
|
+
require_relative 'traversal_response'
|
|
30
32
|
require_relative 'version_aware_tool_dispatch'
|
|
31
33
|
|
|
32
34
|
module Woods
|
|
@@ -73,6 +75,7 @@ module Woods
|
|
|
73
75
|
# controls that actually shrink the answer are `depth`, `types` and
|
|
74
76
|
# `via`; `limit` and `offset` only page what those leave (B-183).
|
|
75
77
|
DEFAULT_TRAVERSAL_LIMIT = 50
|
|
78
|
+
DEFAULT_GRAPH_ANALYSIS_LIMIT = 20
|
|
76
79
|
|
|
77
80
|
class << self
|
|
78
81
|
# Build a configured MCP::Server with all tools and resources.
|
|
@@ -161,7 +164,8 @@ module Woods
|
|
|
161
164
|
'Narrow with depth, types and via first: they shrink the answer, ' \
|
|
162
165
|
'while limit and offset only page it. Returns a BFS tree with ' \
|
|
163
166
|
"depth, paged to #{DEFAULT_TRAVERSAL_LIMIT} nodes by default. " \
|
|
164
|
-
'max_nodes/max_edges bound the walk independently; partial_reason reports a budget cutoff. ' \
|
|
167
|
+
'max_nodes/max_edges bound the walk independently; partial_reason reports a budget cutoff and total_is_exact is false. ' \
|
|
168
|
+
'Published relationships are not exhaustive source-reference coverage. ' \
|
|
165
169
|
'Use explain:true for recorded directed relationships and bounded witnesses; ambiguous types remain explicit.',
|
|
166
170
|
reader_method: :traverse_dependencies,
|
|
167
171
|
render_key: :dependencies)
|
|
@@ -171,7 +175,8 @@ module Woods
|
|
|
171
175
|
'Narrow with depth, types and via first: they shrink the answer, ' \
|
|
172
176
|
'while limit and offset only page it. Returns a BFS tree with ' \
|
|
173
177
|
"depth, paged to #{DEFAULT_TRAVERSAL_LIMIT} nodes by default. " \
|
|
174
|
-
'max_nodes/max_edges bound the walk independently; partial_reason reports a budget cutoff. ' \
|
|
178
|
+
'max_nodes/max_edges bound the walk independently; partial_reason reports a budget cutoff and total_is_exact is false. ' \
|
|
179
|
+
'Published relationships are not exhaustive source-reference coverage. ' \
|
|
175
180
|
'Use explain:true for recorded directed relationships and bounded witnesses; ambiguous types remain explicit.',
|
|
176
181
|
reader_method: :traverse_dependents,
|
|
177
182
|
render_key: :dependents)
|
|
@@ -247,9 +252,9 @@ module Woods
|
|
|
247
252
|
!token.nil? && ids && !ids.empty?
|
|
248
253
|
end
|
|
249
254
|
|
|
250
|
-
def text_response(text)
|
|
255
|
+
def text_response(text, data: nil)
|
|
251
256
|
structured = { text: text }
|
|
252
|
-
structured[:data] = JSON.parse(text)
|
|
257
|
+
structured[:data] = data.nil? ? JSON.parse(text) : data
|
|
253
258
|
::MCP::Tool::Response.new(
|
|
254
259
|
[{ type: 'text', text: text }],
|
|
255
260
|
structured_content: structured
|
|
@@ -393,7 +398,7 @@ module Woods
|
|
|
393
398
|
|
|
394
399
|
sliced = offset.positive? ? original.drop(offset) : original
|
|
395
400
|
container[key] = limit ? truncate_section(sliced, limit) : sliced
|
|
396
|
-
if
|
|
401
|
+
if offset.positive? || container[key].size < original.size
|
|
397
402
|
container["#{key}_total"] = original.size
|
|
398
403
|
container["#{key}_truncated"] = true
|
|
399
404
|
end
|
|
@@ -403,12 +408,11 @@ module Woods
|
|
|
403
408
|
# Page a traversal result's `nodes` hash in place, in BFS order.
|
|
404
409
|
#
|
|
405
410
|
# Mirrors {#paginate_section}'s metadata keys (`nodes_total`,
|
|
406
|
-
# `nodes_truncated`, `nodes_offset`)
|
|
407
|
-
#
|
|
408
|
-
#
|
|
409
|
-
# exactly as it did before the bound existed (B-183).
|
|
411
|
+
# `nodes_truncated`, `nodes_offset`). A page that holds every admitted
|
|
412
|
+
# node adds no pagination keys (B-183). TraversalResponse separately
|
|
413
|
+
# annotates scope and total exactness before pagination.
|
|
410
414
|
#
|
|
411
|
-
# `nodes_total` marks *any*
|
|
415
|
+
# `nodes_total` marks *any* paged answer, not only one with more
|
|
412
416
|
# behind it. Keying it on `total > offset + limit` left the last page
|
|
413
417
|
# of a walk indistinguishable from a complete one: 21 nodes of 121,
|
|
414
418
|
# with nothing saying 100 were skipped. `nodes_truncated` still means
|
|
@@ -656,9 +660,10 @@ module Woods
|
|
|
656
660
|
result[:message] =
|
|
657
661
|
"Identifier '#{identifier}' not found in the index. Use 'search' to find valid identifiers."
|
|
658
662
|
end
|
|
663
|
+
TraversalResponse.annotate(result)
|
|
659
664
|
paginate_nodes.call(result, limit || DEFAULT_TRAVERSAL_LIMIT, offset || 0)
|
|
660
665
|
TraversalEvidencePage.apply(result)
|
|
661
|
-
respond.call(renderer.render(render_key, result))
|
|
666
|
+
respond.call(renderer.render(render_key, result), data: result)
|
|
662
667
|
end
|
|
663
668
|
end
|
|
664
669
|
|
|
@@ -698,32 +703,20 @@ module Woods
|
|
|
698
703
|
enum: ToolResponseRenderer::GRAPH_ANALYSIS_SECTIONS + %w[all],
|
|
699
704
|
description: 'Which analysis to return. Default: all'
|
|
700
705
|
},
|
|
701
|
-
limit: { type: 'integer', description:
|
|
706
|
+
limit: { type: 'integer', description: "Limit results per section (default: #{DEFAULT_GRAPH_ANALYSIS_LIMIT})" },
|
|
702
707
|
offset: { type: 'integer', description: 'Skip this many results per section (default: 0)' }
|
|
703
708
|
}
|
|
704
709
|
}
|
|
705
710
|
) do |server_context:, analysis: nil, limit: nil, offset: nil|
|
|
706
|
-
limit = coerce_int.call(limit)
|
|
711
|
+
limit = coerce_int.call(limit) || DEFAULT_GRAPH_ANALYSIS_LIMIT
|
|
707
712
|
offset = coerce_int.call(offset)
|
|
708
713
|
data = reader.graph_analysis
|
|
709
714
|
section = analysis || 'all'
|
|
710
715
|
effective_offset = offset || 0
|
|
711
716
|
|
|
712
|
-
result =
|
|
713
|
-
|
|
714
|
-
|
|
715
|
-
ToolResponseRenderer::GRAPH_ANALYSIS_SECTIONS.each do |key|
|
|
716
|
-
paginate.call(truncated, key, limit, effective_offset)
|
|
717
|
-
end
|
|
718
|
-
truncated
|
|
719
|
-
else
|
|
720
|
-
data
|
|
721
|
-
end
|
|
722
|
-
else
|
|
723
|
-
single = { section => data[section] || [], 'stats' => data['stats'] }
|
|
724
|
-
paginate.call(single, section, limit, effective_offset) if limit || effective_offset.positive?
|
|
725
|
-
single
|
|
726
|
-
end
|
|
717
|
+
result = section == 'all' ? data.dup : { section => data[section] || [], 'stats' => data['stats'] }
|
|
718
|
+
sections = section == 'all' ? ToolResponseRenderer::GRAPH_ANALYSIS_SECTIONS : [section]
|
|
719
|
+
sections.each { |key| paginate.call(result, key, limit, effective_offset) }
|
|
727
720
|
|
|
728
721
|
respond.call(renderer.render(:graph_analysis, result))
|
|
729
722
|
end
|
|
@@ -936,6 +929,7 @@ module Woods
|
|
|
936
929
|
coerce = method(:coerce_array)
|
|
937
930
|
stale_check = method(:stale_index_result?)
|
|
938
931
|
degraded_response = method(:degraded_retrieval_response)
|
|
932
|
+
corpus_status = method(:retriever_corpus_status)
|
|
939
933
|
retrieval_mode = retriever.respond_to?(:mode) ? retriever.mode : :semantic
|
|
940
934
|
server.define_tool(
|
|
941
935
|
name: 'codebase_retrieve',
|
|
@@ -962,13 +956,13 @@ module Woods
|
|
|
962
956
|
'rails_source, test_mapping, etc.). Overrides the default test_mapping exclusion. ' \
|
|
963
957
|
'Lexical mode and explicit package/path scopes filter before limits and omit the global rank table. ' \
|
|
964
958
|
'In semantic mode, when the unfiltered top-K has no requested type, the retriever ' \
|
|
965
|
-
'falls back to rank-within-type
|
|
966
|
-
'the requested type exist in the index. The response appends a "Type rank ' \
|
|
959
|
+
'falls back to rank-within-type over the available vectors. The response appends a "Type rank ' \
|
|
967
960
|
'context" table with per-type: source, rank in unfiltered top-K, global_k, ' \
|
|
968
|
-
'total_of_type
|
|
961
|
+
'total_of_type (retrieval metadata records, not structural or embedded units). ' \
|
|
962
|
+
'Read source to tell the cases apart: in_top_k (strong match), ' \
|
|
969
963
|
'within_type_fallback (weak match surfaced by the fallback), outside_top_k ' \
|
|
970
|
-
'(
|
|
971
|
-
'(zero
|
|
964
|
+
'(retrieval metadata has this type but other requested types filled the result), absent ' \
|
|
965
|
+
'(zero retrieval metadata records of this type; structural units may still exist).'
|
|
972
966
|
},
|
|
973
967
|
packages: { type: 'array', items: { type: 'string' },
|
|
974
968
|
description: 'Exact published nearest package owners. OR within the list; AND with paths and type eligibility.' },
|
|
@@ -1023,6 +1017,16 @@ module Woods
|
|
|
1023
1017
|
)
|
|
1024
1018
|
end
|
|
1025
1019
|
if retriever
|
|
1020
|
+
corpus = corpus_status.call(retriever, include_types: false)
|
|
1021
|
+
if corpus && corpus[:state] == 'empty'
|
|
1022
|
+
next respond_err.call(
|
|
1023
|
+
'Semantic retrieval has no vectors or retrieval metadata. The structural index may still be ready. ' \
|
|
1024
|
+
'Run `woods:embed` in the application environment and reload the index, or use ' \
|
|
1025
|
+
'explicit `WOODS_RETRIEVAL_MODE=lexical` and restart for retrieval without embeddings. ' \
|
|
1026
|
+
'Use `search` and `lookup` for structural discovery.',
|
|
1027
|
+
code: :empty_index, tool: 'codebase_retrieve', corpus: corpus
|
|
1028
|
+
)
|
|
1029
|
+
end
|
|
1026
1030
|
begin
|
|
1027
1031
|
scope_options = if Retrieval::Scope.requested?(packages: packages, source_paths: source_paths)
|
|
1028
1032
|
{ packages: packages, source_paths: source_paths }
|
|
@@ -1072,6 +1076,8 @@ module Woods
|
|
|
1072
1076
|
'Semantic search is disabled — no embedding provider is configured. ' \
|
|
1073
1077
|
'To enable: set OPENAI_API_KEY, or run Ollama locally ' \
|
|
1074
1078
|
'(brew install ollama && ollama serve && ollama pull nomic-embed-text). ' \
|
|
1079
|
+
'For ranked discovery with no embeddings, set WOODS_RETRIEVAL_MODE=lexical in the MCP process ' \
|
|
1080
|
+
'environment and restart the server. See docs/RETRIEVAL_GUIDE.md#embedding-free-lexical-retrieval. ' \
|
|
1075
1081
|
'Use the `search` tool for pattern-based matching in the meantime.',
|
|
1076
1082
|
code: :not_configured,
|
|
1077
1083
|
config_key: 'embedding_provider',
|
|
@@ -1144,12 +1150,15 @@ module Woods
|
|
|
1144
1150
|
|
|
1145
1151
|
server.define_tool(
|
|
1146
1152
|
name: 'trace_flow',
|
|
1147
|
-
description: 'Trace
|
|
1153
|
+
description: 'Trace a source-derived flow from an exact indexed unit, optionally scoped by #method. ' \
|
|
1154
|
+
'Receiverless local calls may remain unexpanded; this is not proof of runtime execution.',
|
|
1148
1155
|
input_schema: {
|
|
1149
1156
|
properties: {
|
|
1150
1157
|
entry_point: {
|
|
1151
1158
|
type: 'string',
|
|
1152
|
-
description: '
|
|
1159
|
+
description: 'Exact UnitIdentifier, optionally followed by #method (e.g., UsersController#create ' \
|
|
1160
|
+
'or CheckoutService#order). Bare names identify units, including factories; ' \
|
|
1161
|
+
'a bare method name does not locate its owning class. Use search and lookup first.'
|
|
1153
1162
|
},
|
|
1154
1163
|
depth: {
|
|
1155
1164
|
type: 'integer',
|
|
@@ -2081,7 +2090,10 @@ module Woods
|
|
|
2081
2090
|
name: 'woods_status',
|
|
2082
2091
|
description: 'Diagnose whether the Woods index and server are healthy. Returns extraction metadata ' \
|
|
2083
2092
|
'(last run, unit counts, git SHA, staleness in seconds), retriever/embedding configuration, ' \
|
|
2084
|
-
'bootstrap state (hydrated / degraded / failed + reason), feature flags
|
|
2093
|
+
'bootstrap state (hydrated / degraded / failed + reason), and feature flags. ' \
|
|
2094
|
+
'Top-level `ready` describes structural index availability. ' \
|
|
2095
|
+
'Semantic corpus diagnostics report local vector/metadata record counts separately; ' \
|
|
2096
|
+
'unknown counts are null, and nonempty counts do not prove complete embedding coverage. ' \
|
|
2085
2097
|
'Includes source-content freshness; quick scans have a 250ms budget, explicit deep scans have 5s. ' \
|
|
2086
2098
|
'Incomplete evidence is unknown. Call this first on cold connect.',
|
|
2087
2099
|
input_schema: { type: 'object', properties: {
|
|
@@ -2145,10 +2157,11 @@ module Woods
|
|
|
2145
2157
|
},
|
|
2146
2158
|
index: index_section(manifest, extracted_at, staleness, index_dir, reader, source_check),
|
|
2147
2159
|
watch: watch_section(index_dir),
|
|
2160
|
+
supervision: index_dir ? Woods::Watch::SupervisionStatus.read(index_dir.to_s) : { records: [] },
|
|
2148
2161
|
retriever: {
|
|
2149
2162
|
configured: !retriever.nil?,
|
|
2150
2163
|
class: retriever&.class&.name,
|
|
2151
|
-
**(retriever
|
|
2164
|
+
**retriever_status_fields(retriever)
|
|
2152
2165
|
},
|
|
2153
2166
|
bootstrap: bootstrap_state&.to_h,
|
|
2154
2167
|
features: retrieval_features(config, resolved, retriever)
|
|
@@ -2157,6 +2170,19 @@ module Woods
|
|
|
2157
2170
|
|
|
2158
2171
|
private
|
|
2159
2172
|
|
|
2173
|
+
def retriever_status_fields(retriever)
|
|
2174
|
+
return { mode: 'lexical' } if retriever.respond_to?(:mode) && retriever.mode == :lexical
|
|
2175
|
+
|
|
2176
|
+
corpus = retriever_corpus_status(retriever)
|
|
2177
|
+
corpus ? { corpus: corpus } : {}
|
|
2178
|
+
end
|
|
2179
|
+
|
|
2180
|
+
def retriever_corpus_status(retriever, include_types: true)
|
|
2181
|
+
return if retriever.respond_to?(:mode) && retriever.mode == :lexical
|
|
2182
|
+
|
|
2183
|
+
retriever.corpus_status(include_types: include_types) if retriever.respond_to?(:corpus_status)
|
|
2184
|
+
end
|
|
2185
|
+
|
|
2160
2186
|
# Assemble the +index+ sub-hash of woods_status, including a staleness
|
|
2161
2187
|
# gate that compares +manifest.git_sha+ against the current HEAD. The
|
|
2162
2188
|
# manifest captures +git_sha+ / +gemfile_lock_sha+ / +schema_sha+ at
|
|
@@ -27,7 +27,7 @@ module Woods
|
|
|
27
27
|
text: { type: 'string' },
|
|
28
28
|
data: {
|
|
29
29
|
type: %w[object array string number boolean null],
|
|
30
|
-
description: '
|
|
30
|
+
description: 'Structured tool payload, including traversal data in every renderer; otherwise parsed JSON when available.'
|
|
31
31
|
}
|
|
32
32
|
},
|
|
33
33
|
required: ['text'],
|
|
@@ -70,6 +70,22 @@ module Woods
|
|
|
70
70
|
|
|
71
71
|
private
|
|
72
72
|
|
|
73
|
+
def traversal_coverage_lines(data)
|
|
74
|
+
coverage = fetch_key(data, :graph_coverage)
|
|
75
|
+
notice = fetch_key(coverage, :notice) if coverage.is_a?(Hash)
|
|
76
|
+
notice ? [notice] : []
|
|
77
|
+
end
|
|
78
|
+
|
|
79
|
+
def traversal_lower_bound_note(data, shown)
|
|
80
|
+
return unless fetch_key(data, :total_is_exact) == false
|
|
81
|
+
|
|
82
|
+
total = fetch_key(data, :nodes_total, shown)
|
|
83
|
+
offset = fetch_key(data, :nodes_offset, 0)
|
|
84
|
+
position = offset.positive? ? " from offset #{offset}" : ''
|
|
85
|
+
"Showing #{shown} of at least #{total}#{position} " \
|
|
86
|
+
"(total unknown: #{fetch_key(data, :partial_reason)})."
|
|
87
|
+
end
|
|
88
|
+
|
|
73
89
|
def search_completeness_lines(data)
|
|
74
90
|
evidence = fetch_key(data, :completeness)
|
|
75
91
|
return [] unless evidence.is_a?(Hash)
|
|
@@ -40,7 +40,7 @@ module Woods
|
|
|
40
40
|
complete = value(witness, :typed_path_complete) ? 'yes' : 'no'
|
|
41
41
|
context = value(witness, :context) ? 'yes' : 'no'
|
|
42
42
|
"#{identifier}: #{value(witness, :impact)}; parent=#{parent}; edge=#{edge}; " \
|
|
43
|
-
"
|
|
43
|
+
"witness types unambiguous=#{complete}; context only=#{context}"
|
|
44
44
|
end
|
|
45
45
|
|
|
46
46
|
def self.value(hash, key)
|
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Woods
|
|
4
|
+
module MCP
|
|
5
|
+
# Response scope is independent of traversal evidence and page selection.
|
|
6
|
+
module TraversalResponse
|
|
7
|
+
COVERAGE = {
|
|
8
|
+
scope: 'published_relationships',
|
|
9
|
+
source_references: 'not_exhaustive',
|
|
10
|
+
notice: 'Graph coverage: published relationships only. Arbitrary method-body constant references ' \
|
|
11
|
+
'are not exhaustively captured; missing relationships do not prove no callers or dependencies.'
|
|
12
|
+
}.freeze
|
|
13
|
+
|
|
14
|
+
def self.annotate(result)
|
|
15
|
+
return if result[:found] == false
|
|
16
|
+
|
|
17
|
+
result[:graph_coverage] = COVERAGE
|
|
18
|
+
result[:total_is_exact] = !result[:partial]
|
|
19
|
+
end
|
|
20
|
+
end
|
|
21
|
+
end
|
|
22
|
+
end
|
|
@@ -16,10 +16,9 @@ module Woods
|
|
|
16
16
|
#
|
|
17
17
|
# * {.file_rules} — file-based extractors, whose per-file method can be
|
|
18
18
|
# pointed straight at the new path.
|
|
19
|
-
# * {.whole_app_rules} — extractors
|
|
20
|
-
# middleware,
|
|
21
|
-
#
|
|
22
|
-
# re-run of that extractor, which is cheap in an already-booted process.
|
|
19
|
+
# * {.whole_app_rules} — extractors needing the complete runtime or source
|
|
20
|
+
# set (routes, middleware, merged Rake tasks, scheduled jobs, etc.). Their
|
|
21
|
+
# trigger paths map to a wholesale re-run of that extractor.
|
|
23
22
|
#
|
|
24
23
|
# Class-based extractors (models, controllers, mailers, components,
|
|
25
24
|
# channels) are deliberately *absent* here. They are reconciled against
|
|
@@ -161,7 +160,6 @@ module Woods
|
|
|
161
160
|
file_rule(:graphql, :extract_graphql_file,
|
|
162
161
|
[ex::GraphQLExtractor::GRAPHQL_DIRECTORY], extensions: %w[.rb]),
|
|
163
162
|
file_rule(:i18n, :extract_i18n_file, ex::I18nExtractor::I18N_DIRECTORIES, extensions: %w[.yml]),
|
|
164
|
-
file_rule(:rake_tasks, :extract_rake_file, ex::RakeTaskExtractor::RAKE_DIRECTORIES, extensions: %w[.rake]),
|
|
165
163
|
file_rule(:view_templates, :extract_view_template_file,
|
|
166
164
|
ex::ViewTemplateExtractor::VIEW_DIRECTORIES, extensions: view_template_extensions),
|
|
167
165
|
file_rule(:migrations, :extract_migration_file, %w[db/migrate], recursive: false),
|
|
@@ -188,6 +186,9 @@ module Woods
|
|
|
188
186
|
|
|
189
187
|
def build_whole_app_rules
|
|
190
188
|
[
|
|
189
|
+
# A task may combine definitions from several files; any change or
|
|
190
|
+
# deletion must reconcile the complete task set, not its primary file.
|
|
191
|
+
whole_app_rule(:rake_tasks, Woods::Extractors::RakeTaskExtractor::RAKE_DIRECTORIES, extensions: %w[.rake]),
|
|
191
192
|
whole_app_rule(:routes, %w[config/routes], exact_paths: %w[config/routes.rb]),
|
|
192
193
|
whole_app_rule(:engines, %w[config/routes], exact_paths: %w[config/routes.rb Gemfile.lock]),
|
|
193
194
|
whole_app_rule(:middleware, %w[config/initializers config/environments],
|
|
@@ -21,10 +21,47 @@ module Woods
|
|
|
21
21
|
# @return [Hash, nil] string-keyed unit, or nil when this type has no
|
|
22
22
|
# such identifier
|
|
23
23
|
def self.call(payload_dir, reader, identifier, type)
|
|
24
|
-
dir =
|
|
24
|
+
dir = directory_for(type)
|
|
25
25
|
return nil unless dir
|
|
26
|
-
return nil unless reader.list_units(type: type).any? { |entry| entry['identifier'] == identifier }
|
|
27
26
|
|
|
27
|
+
family = Woods::MCP::IndexReader::DIR_TO_TYPE.fetch(dir)
|
|
28
|
+
return nil unless reader.list_units(type: family).any? { |entry| entry['identifier'] == identifier }
|
|
29
|
+
|
|
30
|
+
unit = read_unit(payload_dir, dir, identifier)
|
|
31
|
+
unit if unit && (unit['type'] == type || type == family)
|
|
32
|
+
end
|
|
33
|
+
|
|
34
|
+
# Resolve actual published types as well as historical directory-family aliases.
|
|
35
|
+
#
|
|
36
|
+
# @param type [String]
|
|
37
|
+
# @return [String, nil]
|
|
38
|
+
def self.directory_for(type)
|
|
39
|
+
Woods::MCP::IndexReader::TYPE_TO_DIR[type] ||
|
|
40
|
+
Woods::MCP::IndexReader::UNIT_TYPES_BY_DIR.find { |_, types| types.include?(type) }&.first
|
|
41
|
+
end
|
|
42
|
+
|
|
43
|
+
# Preserve subtype identity in an index entry from a shared type directory.
|
|
44
|
+
#
|
|
45
|
+
# @param payload_dir [Pathname]
|
|
46
|
+
# @param entry [Hash] published index entry
|
|
47
|
+
# @param dir [String] type directory
|
|
48
|
+
# @param requested_type [String, nil] actual type or family alias
|
|
49
|
+
# @return [Hash, nil] entry with actual type, or nil when filtered out
|
|
50
|
+
def self.entry(payload_dir, entry, dir, requested_type)
|
|
51
|
+
family = Woods::MCP::IndexReader::DIR_TO_TYPE.fetch(dir)
|
|
52
|
+
actual_type = if Woods::MCP::IndexReader::UNIT_TYPES_BY_DIR.fetch(dir).size > 1
|
|
53
|
+
read_unit(payload_dir, dir, entry['identifier'])&.fetch('type')
|
|
54
|
+
else
|
|
55
|
+
family
|
|
56
|
+
end
|
|
57
|
+
return unless actual_type
|
|
58
|
+
return if requested_type && requested_type != family && requested_type != actual_type
|
|
59
|
+
|
|
60
|
+
entry.merge('type' => actual_type)
|
|
61
|
+
end
|
|
62
|
+
|
|
63
|
+
# Read a unit whose directory-index membership has already been established.
|
|
64
|
+
def self.read_unit(payload_dir, dir, identifier)
|
|
28
65
|
path = payload_dir.join(dir, filename_for(identifier))
|
|
29
66
|
return nil unless path.file?
|
|
30
67
|
|
|
@@ -42,7 +79,7 @@ module Woods
|
|
|
42
79
|
"#{base}_#{digest}.json"
|
|
43
80
|
end
|
|
44
81
|
|
|
45
|
-
private_class_method :filename_for
|
|
82
|
+
private_class_method :filename_for, :read_unit
|
|
46
83
|
end
|
|
47
84
|
end
|
|
48
85
|
end
|
|
@@ -167,14 +167,14 @@ module Woods
|
|
|
167
167
|
# @return [Array<Hash>]
|
|
168
168
|
def units(type: nil)
|
|
169
169
|
dirs = if type
|
|
170
|
-
dir =
|
|
170
|
+
dir = TypedUnitReader.directory_for(type.to_s)
|
|
171
171
|
dir ? [dir] : []
|
|
172
172
|
else
|
|
173
173
|
Woods::MCP::IndexReader::TYPE_DIRS
|
|
174
174
|
end
|
|
175
175
|
dirs.flat_map do |dir|
|
|
176
176
|
@reader.list_units(type: Woods::MCP::IndexReader::DIR_TO_TYPE[dir])
|
|
177
|
-
.
|
|
177
|
+
.filter_map { |entry| TypedUnitReader.entry(@payload_dir, entry, dir, type&.to_s) }
|
|
178
178
|
end
|
|
179
179
|
end
|
|
180
180
|
|
data/lib/woods/rake_helpers.rb
CHANGED
|
@@ -183,11 +183,8 @@ module Woods
|
|
|
183
183
|
woods_sweep_index_dir(output_dir, lock_name)
|
|
184
184
|
end
|
|
185
185
|
|
|
186
|
-
#
|
|
187
|
-
# the
|
|
188
|
-
# writer may legitimately have recreated content between release and here
|
|
189
|
-
# — a non-empty directory is left alone rather than forced.
|
|
190
|
-
woods_remove_if_empty(output_dir, lock_name)
|
|
186
|
+
# Retain the stable guard and directory: another writer may already hold
|
|
187
|
+
# the guard after release, before it has created its extraction.lock.
|
|
191
188
|
:cleaned
|
|
192
189
|
end
|
|
193
190
|
|
|
@@ -205,13 +202,6 @@ module Woods
|
|
|
205
202
|
end
|
|
206
203
|
end
|
|
207
204
|
|
|
208
|
-
def woods_remove_if_empty(output_dir, lock_name)
|
|
209
|
-
FileUtils.rm_f(output_dir.join(Woods::Coordination::PipelineLock.guard_filename(lock_name)))
|
|
210
|
-
Dir.rmdir(output_dir)
|
|
211
|
-
rescue SystemCallError
|
|
212
|
-
nil
|
|
213
|
-
end
|
|
214
|
-
|
|
215
205
|
# The root containing the Rakefile that loaded this task file.
|
|
216
206
|
#
|
|
217
207
|
# `woods:watch_status` intentionally avoids Rails boot, so it cannot ask
|
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Woods
|
|
4
|
+
module Retrieval
|
|
5
|
+
# Bounded, local diagnostics: never probe remote stores or providers.
|
|
6
|
+
# Nonempty counts establish presence, not alignment, completeness, or
|
|
7
|
+
# provider health. Metadata-only corpora can still serve non-vector paths.
|
|
8
|
+
module CorpusStatus
|
|
9
|
+
module_function
|
|
10
|
+
|
|
11
|
+
# @return [Hash] state and independently known local store statistics
|
|
12
|
+
def build(vector_store, metadata_store, include_types: true)
|
|
13
|
+
vectors = local_stats(vector_store, include_types: include_types)
|
|
14
|
+
metadata = local_stats(metadata_store, include_types: include_types)
|
|
15
|
+
{ state: state(vectors[:count], metadata[:count]), vectors: vectors, metadata: metadata }
|
|
16
|
+
end
|
|
17
|
+
|
|
18
|
+
def local_stats(store, include_types:)
|
|
19
|
+
return unknown unless store.respond_to?(:local_corpus_stats)
|
|
20
|
+
|
|
21
|
+
stats = store.local_corpus_stats(include_types: include_types)
|
|
22
|
+
return unknown unless stats[:count].is_a?(Integer) && stats[:count] >= 0
|
|
23
|
+
|
|
24
|
+
stats
|
|
25
|
+
rescue StandardError, NotImplementedError
|
|
26
|
+
unknown
|
|
27
|
+
end
|
|
28
|
+
private_class_method :local_stats
|
|
29
|
+
|
|
30
|
+
def unknown
|
|
31
|
+
{ count: nil, by_type: nil, untyped_count: nil }
|
|
32
|
+
end
|
|
33
|
+
private_class_method :unknown
|
|
34
|
+
|
|
35
|
+
def state(vectors, metadata)
|
|
36
|
+
return 'unknown' if vectors.nil? || metadata.nil?
|
|
37
|
+
return 'empty' if vectors.zero? && metadata.zero?
|
|
38
|
+
return 'metadata_only' if vectors.zero?
|
|
39
|
+
return 'vectors_only' if metadata.zero?
|
|
40
|
+
|
|
41
|
+
'nonempty'
|
|
42
|
+
end
|
|
43
|
+
private_class_method :state
|
|
44
|
+
end
|
|
45
|
+
end
|
|
46
|
+
end
|