woods 2.0.0.beta2 → 2.0.0.beta4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (233) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +339 -1
  3. data/CONTRIBUTING.md +188 -12
  4. data/README.md +93 -174
  5. data/SECURITY.md +9 -6
  6. data/docs/AGENT_GUIDE.md +109 -8
  7. data/docs/AGENT_SETUP.md +98 -7
  8. data/docs/BACKEND_MATRIX.md +25 -0
  9. data/docs/CLIENT_HOOKS.md +111 -0
  10. data/docs/CONFIGURATION_REFERENCE.md +267 -16
  11. data/docs/CONSOLE_MCP_SETUP.md +80 -7
  12. data/docs/DOCKER_SETUP.md +22 -3
  13. data/docs/EVALUATION.md +464 -1
  14. data/docs/EXTRACTOR_REFERENCE.md +45 -6
  15. data/docs/FAQ.md +11 -12
  16. data/docs/GETTING_STARTED.md +17 -5
  17. data/docs/INCREMENTAL_EXTRACTION.md +147 -7
  18. data/docs/INDEX_LAYOUT.md +382 -0
  19. data/docs/INTERNALS.md +7 -2
  20. data/docs/MCP_SERVERS.md +276 -5
  21. data/docs/MCP_TOOL_COOKBOOK.md +37 -22
  22. data/docs/MCP_WORKTREE_SETUP.md +43 -83
  23. data/docs/NOTION_INTEGRATION.md +13 -0
  24. data/docs/OBSIDIAN_INTEGRATION.md +57 -9
  25. data/docs/PUBLISHED_INDEX.md +72 -0
  26. data/docs/README.md +7 -0
  27. data/docs/RETRIEVAL_GUIDE.md +273 -12
  28. data/docs/RUNTIME_TRACING.md +71 -0
  29. data/docs/SOURCE_FRESHNESS.md +143 -0
  30. data/docs/TROUBLESHOOTING.md +129 -18
  31. data/docs/UNBLOCKED_INTEGRATION.md +25 -0
  32. data/docs/UPGRADING_TO_2.md +48 -22
  33. data/docs/WATCH_DAEMON.md +277 -67
  34. data/exe/woods-agent-config +6 -0
  35. data/exe/woods-extract +5 -0
  36. data/exe/woods-hook-context +6 -0
  37. data/exe/woods-mcp-start +14 -9
  38. data/lib/generators/woods/pgvector_generator.rb +8 -2
  39. data/lib/generators/woods/templates/woods.rb.tt +1 -3
  40. data/lib/tasks/woods.rake +47 -397
  41. data/lib/woods/agent_configuration/applier.rb +135 -0
  42. data/lib/woods/agent_configuration/cli.rb +101 -0
  43. data/lib/woods/agent_configuration/cli_options.rb +29 -0
  44. data/lib/woods/agent_configuration/document.rb +105 -0
  45. data/lib/woods/agent_configuration/error.rb +7 -0
  46. data/lib/woods/agent_configuration/launcher.rb +75 -0
  47. data/lib/woods/agent_configuration/layout.rb +72 -0
  48. data/lib/woods/agent_configuration/managed_section.rb +62 -0
  49. data/lib/woods/agent_configuration/plan.rb +98 -0
  50. data/lib/woods/agent_configuration/plan_diff.rb +38 -0
  51. data/lib/woods/agent_configuration/planned_files.rb +61 -0
  52. data/lib/woods/agent_configuration/planner.rb +63 -0
  53. data/lib/woods/agent_configuration/planner_validation.rb +77 -0
  54. data/lib/woods/agent_configuration/preflight.rb +100 -0
  55. data/lib/woods/agent_configuration/recovery.rb +49 -0
  56. data/lib/woods/ast/node.rb +2 -0
  57. data/lib/woods/ast/parser.rb +38 -5
  58. data/lib/woods/builder.rb +21 -5
  59. data/lib/woods/cache/cache_middleware.rb +28 -7
  60. data/lib/woods/cache/cache_store.rb +4 -5
  61. data/lib/woods/change_set.rb +5 -4
  62. data/lib/woods/console/credential_index.rb +20 -2
  63. data/lib/woods/console/credential_scanner.rb +18 -17
  64. data/lib/woods/console/credential_scanner_registry.rb +36 -0
  65. data/lib/woods/console/dispatch_pipeline.rb +7 -0
  66. data/lib/woods/console/embedded_executor.rb +32 -10
  67. data/lib/woods/console/encrypted_credential_snapshot.rb +16 -0
  68. data/lib/woods/console/rack_middleware.rb +22 -13
  69. data/lib/woods/console/server.rb +18 -16
  70. data/lib/woods/console/sql_noise_stripper.rb +9 -7
  71. data/lib/woods/console/sql_table_scanner.rb +47 -7
  72. data/lib/woods/console/sql_validator.rb +49 -9
  73. data/lib/woods/console/sqlite_read_guard.rb +46 -0
  74. data/lib/woods/coordination/pipeline_lock.rb +3 -2
  75. data/lib/woods/dependency_graph.rb +65 -13
  76. data/lib/woods/embedding/corpus.rb +94 -0
  77. data/lib/woods/embedding/indexer.rb +114 -60
  78. data/lib/woods/embedding/openai.rb +17 -6
  79. data/lib/woods/evaluation/ablation_executor.rb +6 -1
  80. data/lib/woods/evaluation/ablation_timed_executor.rb +22 -4
  81. data/lib/woods/export/typed_reader.rb +56 -0
  82. data/lib/woods/extractor.rb +277 -149
  83. data/lib/woods/extractors/action_cable_extractor.rb +3 -1
  84. data/lib/woods/extractors/behavioral_profile.rb +9 -7
  85. data/lib/woods/extractors/caching_extractor.rb +3 -1
  86. data/lib/woods/extractors/concern_extractor.rb +64 -6
  87. data/lib/woods/extractors/configuration_extractor.rb +7 -3
  88. data/lib/woods/extractors/controller_extractor.rb +13 -4
  89. data/lib/woods/extractors/database_view_extractor.rb +3 -1
  90. data/lib/woods/extractors/declared_parent.rb +55 -0
  91. data/lib/woods/extractors/decorator_extractor.rb +3 -1
  92. data/lib/woods/extractors/engine_extractor.rb +3 -1
  93. data/lib/woods/extractors/event_extractor.rb +4 -2
  94. data/lib/woods/extractors/factory_extractor.rb +3 -1
  95. data/lib/woods/extractors/graphql_extractor.rb +10 -13
  96. data/lib/woods/extractors/i18n_extractor.rb +3 -1
  97. data/lib/woods/extractors/job_extractor.rb +6 -19
  98. data/lib/woods/extractors/lib_extractor.rb +13 -9
  99. data/lib/woods/extractors/mailer_extractor.rb +26 -15
  100. data/lib/woods/extractors/manager_extractor.rb +3 -1
  101. data/lib/woods/extractors/method_parameters.rb +53 -0
  102. data/lib/woods/extractors/middleware_argument.rb +65 -0
  103. data/lib/woods/extractors/middleware_extractor.rb +9 -3
  104. data/lib/woods/extractors/migration_extractor.rb +3 -1
  105. data/lib/woods/extractors/model_extractor.rb +26 -34
  106. data/lib/woods/extractors/package_extractor.rb +24 -4
  107. data/lib/woods/extractors/phlex_extractor.rb +3 -1
  108. data/lib/woods/extractors/policy_extractor.rb +3 -1
  109. data/lib/woods/extractors/poro_extractor.rb +13 -9
  110. data/lib/woods/extractors/pundit_extractor.rb +3 -1
  111. data/lib/woods/extractors/rails_source_extractor.rb +4 -2
  112. data/lib/woods/extractors/rake_task_extractor.rb +4 -2
  113. data/lib/woods/extractors/route_extractor.rb +3 -1
  114. data/lib/woods/extractors/route_helper_resolver.rb +10 -33
  115. data/lib/woods/extractors/scheduled_job_extractor.rb +41 -15
  116. data/lib/woods/extractors/serializer_extractor.rb +4 -2
  117. data/lib/woods/extractors/service_extractor.rb +3 -1
  118. data/lib/woods/extractors/shared_dependency_scanner.rb +2 -2
  119. data/lib/woods/extractors/shared_utility_methods.rb +48 -19
  120. data/lib/woods/extractors/source_nesting.rb +1 -1
  121. data/lib/woods/extractors/state_machine_extractor.rb +3 -1
  122. data/lib/woods/extractors/test_mapping_extractor.rb +3 -1
  123. data/lib/woods/extractors/validator_extractor.rb +3 -1
  124. data/lib/woods/extractors/view_component_extractor.rb +3 -1
  125. data/lib/woods/extractors/view_template_extractor.rb +3 -1
  126. data/lib/woods/gem_mapper.rb +2 -0
  127. data/lib/woods/git_history.rb +116 -0
  128. data/lib/woods/graph_analyzer.rb +35 -6
  129. data/lib/woods/hooks/context_cli.rb +54 -0
  130. data/lib/woods/hooks/context_event.rb +88 -0
  131. data/lib/woods/hooks/context_hint.rb +73 -0
  132. data/lib/woods/hooks/context_impact.rb +77 -0
  133. data/lib/woods/hooks/context_output.rb +47 -0
  134. data/lib/woods/hooks/context_state.rb +102 -0
  135. data/lib/woods/hooks/refresh.rb +79 -0
  136. data/lib/woods/hooks/rule_projection.rb +78 -0
  137. data/lib/woods/input_rules.rb +19 -0
  138. data/lib/woods/mcp/bearer_auth.rb +22 -13
  139. data/lib/woods/mcp/bootstrapper.rb +79 -4
  140. data/lib/woods/mcp/config_resolver.rb +2 -1
  141. data/lib/woods/mcp/index_reader.rb +334 -162
  142. data/lib/woods/mcp/initialization_guidance.rb +27 -0
  143. data/lib/woods/mcp/origin_guard.rb +17 -9
  144. data/lib/woods/mcp/published_lexical_retriever.rb +115 -0
  145. data/lib/woods/mcp/renderers/markdown_renderer.rb +22 -9
  146. data/lib/woods/mcp/renderers/plain_renderer.rb +18 -8
  147. data/lib/woods/mcp/search_results.rb +74 -0
  148. data/lib/woods/mcp/server.rb +178 -63
  149. data/lib/woods/mcp/tool_contract.rb +3 -1
  150. data/lib/woods/mcp/tool_response_renderer.rb +41 -0
  151. data/lib/woods/mcp/traversal_evidence.rb +113 -0
  152. data/lib/woods/mcp/traversal_evidence_index.rb +100 -0
  153. data/lib/woods/mcp/traversal_evidence_page.rb +41 -0
  154. data/lib/woods/mcp/traversal_evidence_text.rb +52 -0
  155. data/lib/woods/mcp/traversal_response.rb +22 -0
  156. data/lib/woods/notion/exporter.rb +56 -17
  157. data/lib/woods/obsidian/destination_plan.rb +98 -0
  158. data/lib/woods/obsidian/name_mapper.rb +19 -3
  159. data/lib/woods/obsidian/note_builder.rb +19 -10
  160. data/lib/woods/obsidian/vault_exporter.rb +88 -32
  161. data/lib/woods/operator/pipeline_guard.rb +18 -13
  162. data/lib/woods/path_dispatcher.rb +13 -6
  163. data/lib/woods/payload_store.rb +27 -26
  164. data/lib/woods/published_index/typed_unit_reader.rb +40 -3
  165. data/lib/woods/published_index.rb +2 -2
  166. data/lib/woods/railtie.rb +3 -3
  167. data/lib/woods/railtie_support.rb +12 -12
  168. data/lib/woods/rake_helpers.rb +382 -0
  169. data/lib/woods/resilience/graph_invariant_validator/membership_checks.rb +71 -0
  170. data/lib/woods/resilience/graph_invariant_validator/node_checks.rb +61 -0
  171. data/lib/woods/resilience/graph_invariant_validator/reverse_relationship_checks.rb +46 -0
  172. data/lib/woods/resilience/graph_invariant_validator.rb +119 -0
  173. data/lib/woods/resilience/index_validator/graph_checks.rb +80 -0
  174. data/lib/woods/resilience/index_validator.rb +112 -23
  175. data/lib/woods/retrieval/context_assembler.rb +50 -15
  176. data/lib/woods/retrieval/lexical_assembler.rb +84 -0
  177. data/lib/woods/retrieval/lexical_index.rb +120 -0
  178. data/lib/woods/retrieval/ranker.rb +4 -2
  179. data/lib/woods/retrieval/scope.rb +108 -0
  180. data/lib/woods/retrieval/scoped_graph_store.rb +32 -0
  181. data/lib/woods/retrieval/scoped_vector_store.rb +55 -0
  182. data/lib/woods/retrieval/search_executor.rb +86 -27
  183. data/lib/woods/retrieval/source_evidence.rb +200 -0
  184. data/lib/woods/retriever.rb +98 -22
  185. data/lib/woods/ruby_analyzer/trace_enricher.rb +77 -38
  186. data/lib/woods/session_tracer/file_store.rb +6 -1
  187. data/lib/woods/session_tracer/middleware.rb +10 -12
  188. data/lib/woods/session_tracer/redis_store.rb +22 -6
  189. data/lib/woods/session_tracer/session_flow_assembler.rb +23 -17
  190. data/lib/woods/session_tracer/solid_cache_coordination.rb +6 -4
  191. data/lib/woods/session_tracer/unit_resolver.rb +63 -0
  192. data/lib/woods/source_inputs/consumer_errors.rb +31 -0
  193. data/lib/woods/source_inputs/handoff.rb +102 -0
  194. data/lib/woods/source_inputs/launcher.rb +157 -0
  195. data/lib/woods/source_inputs/manifest.rb +124 -0
  196. data/lib/woods/source_inputs/private_key.rb +55 -0
  197. data/lib/woods/source_inputs/scanner.rb +171 -0
  198. data/lib/woods/source_inputs/scopes.rb +71 -0
  199. data/lib/woods/source_inputs/session.rb +214 -0
  200. data/lib/woods/source_inputs/status.rb +84 -0
  201. data/lib/woods/source_inputs/verifier.rb +107 -0
  202. data/lib/woods/storage/metadata_store.rb +25 -25
  203. data/lib/woods/storage/pgvector.rb +35 -10
  204. data/lib/woods/storage/qdrant.rb +17 -7
  205. data/lib/woods/storage/vector_store.rb +18 -6
  206. data/lib/woods/tasks.rb +3 -2
  207. data/lib/woods/temporal/json_snapshot_store.rb +58 -9
  208. data/lib/woods/unblocked/exporter.rb +59 -70
  209. data/lib/woods/version.rb +1 -1
  210. data/lib/woods/watch/boot_snapshot.rb +52 -0
  211. data/lib/woods/watch/daemon.rb +154 -32
  212. data/lib/woods/watch/listen_watcher.rb +4 -0
  213. data/lib/woods/watch/polling_watcher.rb +5 -1
  214. data/lib/woods/watch/status.rb +20 -15
  215. data/lib/woods/watch/tree_scan.rb +21 -13
  216. data/lib/woods/watch/watcher.rb +4 -1
  217. data/lib/woods.rb +50 -11
  218. data/plugin/.claude-plugin/plugin.json +1 -1
  219. data/plugin/hooks/adapters/normalize.jq +15 -0
  220. data/plugin/hooks/adapters/normalize.rb +63 -0
  221. data/plugin/hooks/hooks.json +20 -0
  222. data/plugin/hooks/woods-context.sh +50 -0
  223. data/plugin/hooks/woods-input-rules.sh +159 -0
  224. data/plugin/hooks/woods-opencode.mjs +65 -0
  225. data/plugin/hooks/woods-post-edit.sh +2 -225
  226. data/plugin/hooks/woods-refresh.sh +260 -0
  227. data/plugin/hooks/woods-session-start.sh +47 -55
  228. data/plugin/skills/woods-agent-enable/SKILL.md +19 -0
  229. data/plugin/skills/woods-diagnose/SKILL.md +319 -1
  230. data/plugin/skills/woods-investigate/SKILL.md +145 -0
  231. data/plugin/skills/woods-mcp-config/SKILL.md +90 -2
  232. data/plugin/skills/woods-setup/SKILL.md +110 -6
  233. metadata +87 -5
@@ -0,0 +1,119 @@
1
+ # frozen_string_literal: true
2
+
3
+ require 'set'
4
+ require_relative 'graph_invariant_validator/membership_checks'
5
+ require_relative 'graph_invariant_validator/node_checks'
6
+ require_relative 'graph_invariant_validator/reverse_relationship_checks'
7
+
8
+ module Woods
9
+ module Resilience
10
+ # Checks raw published graph data against typed per-directory index entries.
11
+ # It deliberately does not use DependencyGraph.from_h: normalization there
12
+ # can discard malformed records before a validator has seen them.
13
+ #
14
+ # Target identifiers are not necessarily graph nodes (external/unresolved
15
+ # dependencies are valid), and reverse/file/type indexes contain bare names,
16
+ # not typed endpoints. Expected membership is the union over all variants.
17
+ class GraphInvariantValidator
18
+ include MembershipChecks
19
+ include NodeChecks
20
+ include ReverseRelationshipChecks
21
+
22
+ def initialize(graph:, index_entries:)
23
+ @graph = graph
24
+ @index_entries = index_entries
25
+ end
26
+
27
+ # @return [Array<String>] semantic errors, without changing either input
28
+ def validate
29
+ @errors = []
30
+ return ['dependency_graph.json: expected an object'] unless @graph.is_a?(Hash)
31
+
32
+ @typed_nodes = {}
33
+ @primary_types = {}
34
+ @expected_reverse = membership_map
35
+ @expected_reverse_via = Hash.new { |hash, key| hash[key] = [] }
36
+ @expected_files = membership_map
37
+ @expected_types = membership_map
38
+ collect_nodes
39
+ validate_edges
40
+ validate_reverse_relationships
41
+ validate_membership('reverse', @expected_reverse)
42
+ validate_membership('file_map', @expected_files, scalar: true)
43
+ validate_membership('type_index', @expected_types)
44
+ validate_index_agreement
45
+ @errors
46
+ end
47
+
48
+ private
49
+
50
+ def membership_map
51
+ Hash.new { |hash, key| hash[key] = Set.new }
52
+ end
53
+
54
+ def error(label, message)
55
+ @errors << "dependency_graph.json #{label}: #{message}"
56
+ end
57
+
58
+ def object_section(name)
59
+ value = @graph[name]
60
+ return value if value.is_a?(Hash)
61
+
62
+ error(name, 'expected an object')
63
+ {}
64
+ end
65
+
66
+ def name?(value)
67
+ value.is_a?(String) && !value.empty?
68
+ end
69
+
70
+ def validate_edges
71
+ edges = object_section('edges')
72
+ @primary_types.each_key do |identifier|
73
+ error('edges', "missing edge list for #{identifier.inspect}") unless edges.key?(identifier)
74
+ end
75
+ edges.each do |identifier, list|
76
+ label = "edges[#{identifier.inspect}]"
77
+ error(label, 'source has no primary node') unless @primary_types.key?(identifier)
78
+ validate_edge_list(identifier, @primary_types[identifier], list, label)
79
+ end
80
+ @variants.each_with_index do |record, index|
81
+ next unless record.is_a?(Hash)
82
+
83
+ validate_edge_list(record['identifier'], record['type'], record['edges'], "variants[#{index}].edges")
84
+ end
85
+ end
86
+
87
+ def validate_edge_list(source, source_type, list, label)
88
+ unless list.is_a?(Array)
89
+ error(label, 'expected an array')
90
+ return
91
+ end
92
+
93
+ list.each_with_index do |edge, index|
94
+ validate_edge(source, source_type, edge, "#{label}[#{index}]")
95
+ end
96
+ end
97
+
98
+ def validate_edge(source, source_type, edge, label)
99
+ target = edge.is_a?(Hash) ? edge['target'] : edge
100
+ unless name?(target)
101
+ error(label, 'expected a target identifier or an object with a target identifier')
102
+ return
103
+ end
104
+ validate_edge_attributes(edge, label) if edge.is_a?(Hash)
105
+ @expected_reverse[target].add(source) if name?(source)
106
+ record_reverse_relationship(target, source, source_type, edge)
107
+ end
108
+
109
+ def validate_edge_attributes(edge, label)
110
+ %w[via through through_db].each do |key|
111
+ error(label, "#{key} must be a string or null") unless edge[key].nil? || edge[key].is_a?(String)
112
+ end
113
+ return if [nil, true, false].include?(edge['disable_joins'])
114
+
115
+ error(label, 'disable_joins must be a boolean or null')
116
+ end
117
+ end
118
+ end
119
+ end
@@ -0,0 +1,80 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Woods
4
+ module Resilience
5
+ class IndexValidator
6
+ # Reuse the published reader's retention pin but inspect raw graph and
7
+ # unit artifacts independently, without normalizing away bad records.
8
+ module GraphChecks
9
+ private
10
+
11
+ def with_validation_payload
12
+ root = Pathname.new(@index_dir)
13
+ if root.join('manifest.json').exist? || root.join('generation.json').exist?
14
+ Woods::PublishedIndex::GenerationCatalog.pointer(root)
15
+ reader = Woods::MCP::IndexReader.new(root)
16
+ reader.with_pinned_generation do
17
+ @validation_payload = reader.payload_dir.to_s
18
+ @validation_manifest = reader.manifest
19
+ yield
20
+ end
21
+ else
22
+ @validation_payload = @index_dir
23
+ yield
24
+ end
25
+ ensure
26
+ @validation_payload = nil
27
+ @validation_manifest = nil
28
+ @graph_index_entries = nil
29
+ end
30
+
31
+ def static_source_map?
32
+ manifest = @validation_manifest
33
+ manifest.is_a?(Hash) && manifest['provenance'].is_a?(Hash) &&
34
+ manifest['provenance']['mode'] == 'woods_static_ruby_source'
35
+ end
36
+
37
+ def validation_index_entries(path, errors)
38
+ entries = JSON.parse(Woods::AtomicFile.read(path))
39
+ unless entries.is_a?(Array)
40
+ errors << "#{path}: expected an array"
41
+ return []
42
+ end
43
+ entries.select do |entry|
44
+ valid = entry.is_a?(Hash) && entry['identifier'].is_a?(String) && !entry['identifier'].empty?
45
+ errors << "#{path}: entry requires a nonempty identifier" unless valid
46
+ valid
47
+ end
48
+ end
49
+
50
+ def collect_graph_index_entry(type_dir, entry, data, errors)
51
+ directory = File.basename(type_dir)
52
+ types = Woods::MCP::IndexReader::UNIT_TYPES_BY_DIR.fetch(directory)
53
+ typed_entry = graph_index_entry(entry, data, types)
54
+ @graph_index_entries << typed_entry
55
+ return unless data
56
+
57
+ file = find_unit_file(type_dir, entry['identifier'])
58
+ unless data['identifier'] == entry['identifier'] && types.include?(data['type'])
59
+ errors << "#{file}: expected typed unit #{typed_entry['type']}:#{entry['identifier']}"
60
+ return
61
+ end
62
+ validate_graph_unit_path(file, entry, data, errors)
63
+ end
64
+
65
+ def graph_index_entry(entry, data, types)
66
+ type = data && types.include?(data['type']) ? data['type'] : types.first
67
+ result = entry.merge('type' => type)
68
+ result['file_path'] = data['file_path'] if data&.key?('file_path') && !entry.key?('file_path')
69
+ result
70
+ end
71
+
72
+ def validate_graph_unit_path(file, entry, data, errors)
73
+ return unless entry.key?('file_path') && data.key?('file_path') && entry['file_path'] != data['file_path']
74
+
75
+ errors << "#{file}: file_path differs from #{File.dirname(file)}/_index.json for #{entry['identifier']}"
76
+ end
77
+ end
78
+ end
79
+ end
80
+ end
@@ -2,10 +2,16 @@
2
2
 
3
3
  require 'json'
4
4
  require 'set'
5
+ require 'rubygems/version'
6
+ require_relative '../version'
5
7
  require_relative '../filename_utils'
6
8
  require_relative '../atomic_file'
7
9
 
8
10
  require_relative '../generation'
11
+ require_relative '../published_index'
12
+ require_relative 'graph_invariant_validator'
13
+ require_relative 'index_validator/graph_checks'
14
+ require_relative '../source_inputs/manifest'
9
15
 
10
16
  module Woods
11
17
  module Resilience
@@ -16,6 +22,8 @@ module Woods
16
22
  # - All files referenced in the index exist on disk
17
23
  # - Content hashes (source_hash) match the actual source_code
18
24
  # - No stale unit files exist that aren't listed in the index
25
+ # - Typed graph identities and reverse/file/type memberships agree
26
+ # All checks share one pinned generation; the validator never repairs it.
19
27
  #
20
28
  # **This class knows nothing about vectors or embedding dimensions.** Six
21
29
  # documents used to credit it with detecting dimension mismatches; it never
@@ -23,10 +31,8 @@ module Woods
23
31
  # `Tasks.verify_store_dimensions!` before a durable embed run and by
24
32
  # {Woods::Storage::Snapshotter::Vector} at MCP boot.
25
33
  #
26
- # Consumed by `spec/integration/multi_worktree_spec.rb` as a per-worktree
27
- # integrity oracle. The `woods:validate` rake task performs an overlapping
28
- # check inline rather than calling this — deliberate duplication left alone
29
- # for now, since the task's output format is user-facing.
34
+ # Shared by the `woods:validate` rake task and worktree integration checks.
35
+ # Writer-version mismatches are advisory; structural errors still fail.
30
36
  #
31
37
  # @example
32
38
  # validator = IndexValidator.new(index_dir: "tmp/woods")
@@ -34,6 +40,7 @@ module Woods
34
40
  # puts report.errors if !report.valid?
35
41
  class IndexValidator # rubocop:disable Metrics/ClassLength
36
42
  include Woods::FilenameUtils
43
+ include GraphChecks
37
44
 
38
45
  # Report produced by {#validate}.
39
46
  #
@@ -84,13 +91,19 @@ module Woods
84
91
  return ValidationReport.new(valid?: false, warnings: warnings, errors: errors)
85
92
  end
86
93
 
87
- payload_type_dirs(errors).each do |type_dir|
88
- validate_type_directory(type_dir, warnings, errors)
94
+ with_validation_payload do
95
+ @graph_index_entries = []
96
+ payload_type_dirs(errors).each do |type_dir|
97
+ validate_type_directory(type_dir, warnings, errors)
98
+ end
99
+ validate_flow_artifacts(errors)
100
+ validate_against_manifest(warnings, errors)
89
101
  end
90
- validate_flow_artifacts(errors)
91
- validate_against_manifest(warnings, errors)
92
102
 
93
103
  ValidationReport.new(valid?: errors.empty?, warnings: warnings, errors: errors)
104
+ rescue IOError, SystemCallError, ArgumentError, JSON::ParserError, Woods::PublishedIndex::CorruptPointerError => e
105
+ errors << "Cannot read published index: #{e.class}: #{e.message}"
106
+ ValidationReport.new(valid?: false, warnings: warnings, errors: errors)
94
107
  end
95
108
 
96
109
  # The checks `woods:validate` used to carry inline: manifest counts
@@ -107,19 +120,72 @@ module Woods
107
120
  manifest_path = File.join(payload, 'manifest.json')
108
121
  return unless File.exist?(manifest_path)
109
122
 
110
- manifest = JSON.parse(Woods::AtomicFile.read(manifest_path))
123
+ manifest = read_validation_manifest(manifest_path, errors)
124
+ return unless manifest
125
+
126
+ validate_writer_version(manifest['woods_version'], warnings)
111
127
  unresolvable = Hash.new { |hash, key| hash[key] = [] }
112
128
 
113
- (manifest['counts'] || {}).each do |type, expected_count|
129
+ manifest.fetch('counts', {}).each do |type, expected_count|
114
130
  validate_manifest_type(payload, type, expected_count, unresolvable, warnings, errors)
115
131
  end
116
132
 
117
133
  warn_unresolvable_paths(warnings, unresolvable)
118
134
  validate_dependency_graph(payload, errors)
135
+ validate_source_inputs(payload, errors)
136
+ end
137
+
138
+ # Optional for old generations; malformed new provenance is an artifact
139
+ # integrity error. Source drift itself is advisory and belongs to status.
140
+ def validate_source_inputs(payload, errors)
141
+ path = File.join(payload, SourceInputs::Manifest::FILE_NAME)
142
+ return unless File.exist?(path)
143
+
144
+ File.open(path, File::RDONLY | File::NONBLOCK) do |file|
145
+ raise SourceInputs::Manifest::Invalid unless file.stat.file?
146
+
147
+ SourceInputs::Manifest.parse(file.read(SourceInputs::Manifest::MAX_BYTES + 1))
148
+ end
149
+ rescue SourceInputs::Manifest::Invalid, SystemCallError, IOError
150
+ errors << 'Invalid source_inputs.json provenance artifact'
151
+ end
152
+
153
+ def read_validation_manifest(path, errors)
154
+ manifest = JSON.parse(Woods::AtomicFile.read(path))
155
+ unless manifest.is_a?(Hash)
156
+ errors << 'manifest.json: expected an object'
157
+ return
158
+ end
159
+ unless manifest.fetch('counts', {}).is_a?(Hash)
160
+ errors << 'manifest.json counts: expected an object'
161
+ return
162
+ end
163
+
164
+ manifest
165
+ end
166
+
167
+ # The version records the last publisher, not the producer of every
168
+ # retained unit. Missing provenance is normal for older indexes.
169
+ def validate_writer_version(value, warnings)
170
+ return if value.nil?
171
+
172
+ unless value.is_a?(String) && !value.strip.empty?
173
+ warnings << 'Invalid manifest woods_version; writer provenance is unknown. Run a full woods:extract.'
174
+ return
175
+ end
176
+
177
+ writer = Gem::Version.new(value)
178
+ reader = Gem::Version.new(Woods::VERSION)
179
+ return if writer.segments.first == reader.segments.first
180
+
181
+ warnings << "Index last published by Woods #{value}; reader is Woods #{Woods::VERSION}. " \
182
+ 'Major versions differ. Run a full woods:extract before relying on compatibility.'
183
+ rescue ArgumentError
184
+ warnings << 'Invalid manifest woods_version; writer provenance is unknown. Run a full woods:extract.'
119
185
  end
120
186
 
121
- # Split each type's unresolvable units into app-tree paths and gem-owned
122
- # paths, since only the first has a remedy the operator can act on.
187
+ # Split unresolvable app-tree paths from gem-owned paths so each warning
188
+ # can distinguish a different filesystem environment from a changed bundle.
123
189
  #
124
190
  # @param warnings [Array<String>]
125
191
  # @param unresolvable [Hash{String => Array<Array(String, Boolean)>}]
@@ -143,15 +209,16 @@ module Woods
143
209
 
144
210
  # Absolute paths outside the app root belong to a gem — an engine model,
145
211
  # a framework source — and resolve only where that gem is installed at
146
- # the extracting path. Re-running extraction here cannot change that,
147
- # so the remedy is not offered.
212
+ # the extracting path. A changed bundle requires fresh extraction; a
213
+ # reader on another filesystem instead needs the original gem paths.
148
214
  def warn_unresolvable_gem_paths(warnings, type, identifiers)
149
215
  return if identifiers.empty?
150
216
 
151
217
  warnings << "#{type}: #{identifiers.size} unit(s) whose file_path lies outside the app root " \
152
218
  "and is absent here (e.g. #{identifiers.first(3).join(', ')}). Gem-owned units " \
153
219
  '(engine models, framework sources) resolve only where that gem is installed at ' \
154
- 'the extracting path.'
220
+ 'the extracting path. After a bundle update, run woods:extract in a fresh process ' \
221
+ 'with the updated bundle, then woods:validate.'
155
222
  end
156
223
 
157
224
  # rubocop:disable-next Metrics/ParameterLists
@@ -175,6 +242,10 @@ module Woods
175
242
  # @param errors [Array<String>]
176
243
  def validate_unit_file(file, type, unresolvable, errors)
177
244
  data = JSON.parse(Woods::AtomicFile.read(file))
245
+ unless data.is_a?(Hash)
246
+ errors << "#{file}: expected a unit object"
247
+ return
248
+ end
178
249
  errors << "#{file}: missing identifier" unless data['identifier']
179
250
  errors << "#{file}: missing source_code" unless data['source_code']
180
251
  file_path = data['file_path']
@@ -211,13 +282,14 @@ module Woods
211
282
  return
212
283
  end
213
284
 
214
- JSON.parse(Woods::AtomicFile.read(graph_path))
285
+ graph = JSON.parse(Woods::AtomicFile.read(graph_path))
286
+ errors.concat(GraphInvariantValidator.new(graph: graph, index_entries: @graph_index_entries).validate)
215
287
  rescue JSON::ParserError
216
288
  errors << 'dependency_graph.json: invalid JSON'
217
289
  end
218
290
 
219
291
  def payload_dir
220
- Woods::Generation.new(output_dir: @index_dir).payload_dir.to_s
292
+ @validation_payload || Woods::Generation.new(output_dir: @index_dir).payload_dir.to_s
221
293
  end
222
294
 
223
295
  private
@@ -292,12 +364,18 @@ module Woods
292
364
  # failure RAISES: {#payload_type_dirs} converts it to a validation
293
365
  # error rather than degrading to a silently empty allowlist.
294
366
  #
367
+ # The explicit Woods static-map provenance additionally admits the
368
+ # GemMapper's own type families; arbitrary directories stay excluded.
295
369
  # `flows/` is deliberately absent: it holds `flow_index.json` and
296
370
  # per-flow documents, which {#validate_flow_artifacts} owns.
297
371
  #
298
372
  # @return [Array<String>]
299
373
  def type_directory_allowlist
300
- @type_directory_allowlist ||= self.class.unit_type_directories
374
+ directories = self.class.unit_type_directories
375
+ return directories unless static_source_map?
376
+
377
+ require_relative '../gem_mapper'
378
+ directories | Woods::GemMapper::TYPE_DIRECTORIES.values
301
379
  end
302
380
 
303
381
  # Validate the flows/ artifact family (G-2): `flow_index.json` parses,
@@ -361,13 +439,12 @@ module Woods
361
439
  return
362
440
  end
363
441
 
364
- index_entries = JSON.parse(Woods::AtomicFile.read(index_path))
365
442
  indexed_identifiers = Set.new
366
-
367
- index_entries.each do |entry|
443
+ validation_index_entries(index_path, errors).each do |entry|
368
444
  identifier = entry['identifier']
369
445
  indexed_identifiers << identifier
370
- validate_index_entry(type_dir, type_name, identifier, errors)
446
+ data = validate_index_entry(type_dir, type_name, identifier, errors)
447
+ collect_graph_index_entry(type_dir, entry, data, errors)
371
448
  end
372
449
 
373
450
  check_stale_files(type_dir, type_name, indexed_identifiers, warnings)
@@ -413,11 +490,23 @@ module Woods
413
490
  # @param errors [Array<String>] Accumulated errors
414
491
  def validate_content_hash(unit_file, identifier, errors)
415
492
  data = JSON.parse(Woods::AtomicFile.read(unit_file))
493
+ unless data.is_a?(Hash)
494
+ errors << "#{unit_file}: expected a unit object"
495
+ return
496
+ end
497
+ check_content_hash(data, identifier, errors)
498
+ data
499
+ end
500
+
501
+ def check_content_hash(data, identifier, errors)
416
502
  source_code = data['source_code']
417
503
  stored_hash = data['source_hash']
418
-
419
504
  return unless source_code && stored_hash
420
505
 
506
+ unless source_code.is_a?(String) && stored_hash.is_a?(String)
507
+ errors << "#{identifier}: source_code and source_hash must be strings"
508
+ return
509
+ end
421
510
  expected_hash = Digest::SHA256.hexdigest(source_code)
422
511
  return if stored_hash == expected_hash
423
512
 
@@ -2,6 +2,7 @@
2
2
 
3
3
  require_relative 'search_executor'
4
4
  require_relative '../token_utils'
5
+ require_relative 'source_evidence'
5
6
 
6
7
  module Woods
7
8
  module Retrieval
@@ -103,7 +104,18 @@ module Woods
103
104
  # @param structural_context [String, nil] Optional codebase overview text
104
105
  # @param budget [Integer, nil] Override token budget; falls back to @budget
105
106
  # @return [AssembledContext] Token-budgeted context with source attribution
106
- def assemble(candidates:, classification:, structural_context: nil, budget: nil)
107
+ def assemble(**options)
108
+ # Pipeline assemblers are shared by concurrent Ruby/HTTP callers. Keep
109
+ # per-request metadata, query, mode and generation on a private worker.
110
+ dup.send(:assemble_request, **options)
111
+ end
112
+
113
+ def assemble_request(candidates:, classification:, structural_context: nil, budget: nil,
114
+ evidence: 'full', query: nil, generation: nil)
115
+ SourceEvidence.validate_mode!(evidence)
116
+ @evidence_mode = evidence
117
+ @evidence_query = query
118
+ @evidence_generation = generation
107
119
  effective_budget = budget || @budget
108
120
  sections = []
109
121
  sources = []
@@ -132,6 +144,22 @@ module Woods
132
144
  build_result(sections, sources, effective_budget, @skipped_missing_metadata)
133
145
  end
134
146
 
147
+ private :assemble_request
148
+
149
+ # Estimate token count. Prefers the injected {TokenCounter} — which
150
+ # loads the provider's real tokenizer and returns exact counts — and
151
+ # falls back to the configured chars-per-token ratio when no counter
152
+ # is wired.
153
+ #
154
+ # @param text [String]
155
+ # @return [Integer]
156
+ def estimate_tokens(text)
157
+ return 0 if text.nil? || text.empty?
158
+ return @token_counter.count(text) if @token_counter
159
+
160
+ (text.length / @chars_per_token).ceil
161
+ end
162
+
135
163
  private
136
164
 
137
165
  # Suffix the Indexer appends when a single unit is split into multiple
@@ -293,6 +321,10 @@ module Woods
293
321
  return tokens_used
294
322
  end
295
323
 
324
+ if @evidence_mode != 'full'
325
+ return append_compact_candidate(parts, sources, candidate, unit, budget, tokens_used)
326
+ end
327
+
296
328
  text = format_unit(unit, candidate)
297
329
  tokens = estimate_tokens(text)
298
330
  remaining = budget - tokens_used
@@ -308,6 +340,23 @@ module Woods
308
340
  end
309
341
  end
310
342
 
343
+ def append_compact_candidate(parts, sources, candidate, unit, budget, tokens_used)
344
+ header = "## #{unit_field(unit, :identifier)} (#{unit_field(unit, :type)})\n" \
345
+ "File: #{unit_field(unit, :file_path)}\n\n"
346
+ remaining = budget - tokens_used
347
+ return tokens_used unless remaining.positive?
348
+
349
+ evidence = SourceEvidence.new(unit: unit, query: @evidence_query, generation: @evidence_generation)
350
+ .render(mode: @evidence_mode, budget: remaining,
351
+ counter: ->(text) { estimate_tokens(header + text) })
352
+ return tokens_used if evidence.text.empty?
353
+
354
+ text = header + evidence.text
355
+ parts << text
356
+ sources << build_source_attribution(candidate, unit).merge(evidence: evidence.provenance)
357
+ tokens_used + estimate_tokens(text)
358
+ end
359
+
311
360
  # Format a unit for inclusion in context.
312
361
  #
313
362
  # @param unit [Hash] Unit data from metadata store
@@ -400,20 +449,6 @@ module Woods
400
449
  "#{text[0...target_chars]}\n... [truncated]"
401
450
  end
402
451
 
403
- # Estimate token count. Prefers the injected {TokenCounter} — which
404
- # loads the provider's real tokenizer and returns exact counts — and
405
- # falls back to the configured chars-per-token ratio when no counter
406
- # is wired.
407
- #
408
- # @param text [String]
409
- # @return [Integer]
410
- def estimate_tokens(text)
411
- return 0 if text.nil? || text.empty?
412
- return @token_counter.count(text) if @token_counter
413
-
414
- (text.length / @chars_per_token).ceil
415
- end
416
-
417
452
  # Effective chars-per-token for chunk-size sizing. When an exact
418
453
  # counter is present, prefer its native ratio (e.g. 1.2 for
419
454
  # nomic-embed-text) so truncation and estimation agree. Falls back
@@ -0,0 +1,84 @@
1
+ # frozen_string_literal: true
2
+
3
+ require_relative 'context_assembler'
4
+ require_relative 'lexical_index'
5
+
6
+ module Woods
7
+ module Retrieval
8
+ # Lexical context keeps matching evidence visible and charges every notice,
9
+ # header and truncation marker to the same character-based token estimate.
10
+ class LexicalAssembler
11
+ def estimate_tokens(text)
12
+ (text.length / 4.0).ceil
13
+ end
14
+
15
+ def assemble(candidates:, budget:, evidence: 'full', query: nil, generation: nil, **)
16
+ SourceEvidence.validate_mode!(evidence)
17
+ raise ArgumentError, 'budget must be a positive Integer' unless budget.is_a?(Integer) && budget.positive?
18
+
19
+ # Reserve the largest included count before selecting evidence. Replacing
20
+ # it afterward can only shrink the text; do not refill or reorder sources.
21
+ notice = count_notice(candidates.size, candidates.size)
22
+ context = notice[0, budget * 4]
23
+ sources = []
24
+ candidates.each do |candidate|
25
+ unit = candidate.metadata
26
+ header = "\n\n## #{unit['identifier']} (#{unit['type']})\nFile: #{unit['file_path']}\n" \
27
+ "Matched: #{candidate.matched_fields.join(', ')}\n\n"
28
+ remaining = (budget * 4) - context.length - header.length
29
+ next unless remaining.positive?
30
+
31
+ if evidence != 'full'
32
+ selected = SourceEvidence.new(unit: unit, query: query, generation: generation)
33
+ .render(mode: evidence, budget: budget,
34
+ counter: ->(text) { estimate_tokens(context + header + text) })
35
+ next if selected.text.empty?
36
+
37
+ context += header + selected.text
38
+ sources << { identifier: unit['identifier'], type: unit['type'], file_path: unit['file_path'],
39
+ score: candidate.score, matched_fields: candidate.matched_fields,
40
+ evidence: selected.provenance }
41
+ next
42
+ end
43
+
44
+ source = evidence_text(candidate)
45
+ truncated = source.length > remaining
46
+ marker = "\n[Published evidence truncated; use lookup for the full unit.]"
47
+ next if truncated && remaining <= marker.length
48
+
49
+ context += header + (truncated ? source[0, remaining - marker.length] + marker : source)
50
+ sources << { identifier: unit['identifier'], type: unit['type'], file_path: unit['file_path'],
51
+ score: candidate.score, matched_fields: candidate.matched_fields, truncated: truncated }
52
+ break if truncated
53
+ end
54
+ body = context[notice.length..].to_s
55
+ context = (count_notice(candidates.size, sources.size) + body)[0, budget * 4]
56
+ AssembledContext.new(context: context, tokens_used: estimate_tokens(context), budget: budget,
57
+ sources: sources, sections: [:primary], skipped_missing_metadata: 0)
58
+ end
59
+
60
+ private
61
+
62
+ def count_notice(candidates, included)
63
+ text = "Mode: lexical (field-aware BM25; sources included: #{included}; " \
64
+ "candidates considered: #{candidates}; candidate limit: #{LexicalIndex::DEFAULT_LIMIT}; " \
65
+ 'token counts estimated).'
66
+ text += "\nNo lexical matches." if candidates.zero?
67
+ text
68
+ end
69
+
70
+ def evidence_text(candidate)
71
+ unit = candidate.metadata
72
+ source = unit['source_code'].to_s
73
+ return source unless candidate.matched_fields.any? { |field| field.start_with?('runtime:') }
74
+
75
+ metadata = unit['metadata'].is_a?(Hash) ? unit['metadata'] : {}
76
+ values = LexicalIndex::RUNTIME_FIELDS.each_with_object({}) do |key, selected|
77
+ value = metadata[key] || unit[key]
78
+ selected[key] = value unless value.nil?
79
+ end
80
+ "Published runtime metadata:\n#{JSON.pretty_generate(values)}\n\n#{source}"
81
+ end
82
+ end
83
+ end
84
+ end