woods 2.0.0.beta1 → 2.0.0.beta3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (221) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +400 -1
  3. data/CONTRIBUTING.md +224 -9
  4. data/README.md +7 -3
  5. data/SECURITY.md +9 -6
  6. data/docs/AGENT_GUIDE.md +83 -4
  7. data/docs/AGENT_SETUP.md +82 -1
  8. data/docs/BACKEND_MATRIX.md +20 -0
  9. data/docs/CLIENT_HOOKS.md +111 -0
  10. data/docs/CONFIGURATION_REFERENCE.md +233 -13
  11. data/docs/CONSOLE_MCP_SETUP.md +35 -5
  12. data/docs/DOCKER_SETUP.md +21 -2
  13. data/docs/EVALUATION.md +464 -1
  14. data/docs/EXTRACTOR_REFERENCE.md +36 -5
  15. data/docs/FAQ.md +11 -12
  16. data/docs/GETTING_STARTED.md +17 -5
  17. data/docs/INCREMENTAL_EXTRACTION.md +158 -2
  18. data/docs/INDEX_LAYOUT.md +382 -0
  19. data/docs/INTERNALS.md +15 -7
  20. data/docs/MCP_SERVERS.md +221 -5
  21. data/docs/MCP_TOOL_COOKBOOK.md +33 -18
  22. data/docs/NOTION_INTEGRATION.md +13 -0
  23. data/docs/OBSIDIAN_INTEGRATION.md +57 -9
  24. data/docs/PUBLISHED_INDEX.md +71 -0
  25. data/docs/README.md +7 -0
  26. data/docs/RETRIEVAL_GUIDE.md +253 -11
  27. data/docs/RUNTIME_TRACING.md +71 -0
  28. data/docs/SOURCE_FRESHNESS.md +143 -0
  29. data/docs/TROUBLESHOOTING.md +117 -5
  30. data/docs/UNBLOCKED_INTEGRATION.md +25 -0
  31. data/docs/UPGRADING_TO_2.md +44 -22
  32. data/docs/WATCH_DAEMON.md +259 -59
  33. data/exe/woods-agent-config +6 -0
  34. data/exe/woods-extract +5 -0
  35. data/exe/woods-hook-context +6 -0
  36. data/lib/generators/woods/templates/woods.rb.tt +1 -3
  37. data/lib/tasks/woods.rake +47 -397
  38. data/lib/woods/agent_configuration/applier.rb +133 -0
  39. data/lib/woods/agent_configuration/cli.rb +101 -0
  40. data/lib/woods/agent_configuration/cli_options.rb +29 -0
  41. data/lib/woods/agent_configuration/document.rb +105 -0
  42. data/lib/woods/agent_configuration/error.rb +7 -0
  43. data/lib/woods/agent_configuration/launcher.rb +75 -0
  44. data/lib/woods/agent_configuration/layout.rb +59 -0
  45. data/lib/woods/agent_configuration/managed_section.rb +62 -0
  46. data/lib/woods/agent_configuration/plan.rb +98 -0
  47. data/lib/woods/agent_configuration/plan_diff.rb +38 -0
  48. data/lib/woods/agent_configuration/planned_files.rb +61 -0
  49. data/lib/woods/agent_configuration/planner.rb +63 -0
  50. data/lib/woods/agent_configuration/planner_validation.rb +77 -0
  51. data/lib/woods/agent_configuration/preflight.rb +100 -0
  52. data/lib/woods/agent_configuration/recovery.rb +49 -0
  53. data/lib/woods/ast/node.rb +2 -0
  54. data/lib/woods/ast/parser.rb +38 -5
  55. data/lib/woods/atomic_file.rb +133 -3
  56. data/lib/woods/builder.rb +21 -5
  57. data/lib/woods/cache/cache_middleware.rb +28 -7
  58. data/lib/woods/cache/cache_store.rb +4 -5
  59. data/lib/woods/change_set.rb +5 -4
  60. data/lib/woods/console/credential_index.rb +20 -2
  61. data/lib/woods/console/credential_scanner.rb +14 -14
  62. data/lib/woods/console/credential_scanner_registry.rb +36 -0
  63. data/lib/woods/console/embedded_executor.rb +1 -1
  64. data/lib/woods/console/encrypted_credential_snapshot.rb +16 -0
  65. data/lib/woods/console/rack_middleware.rb +22 -13
  66. data/lib/woods/console/server.rb +18 -16
  67. data/lib/woods/dependency_graph.rb +65 -13
  68. data/lib/woods/embedding/corpus.rb +94 -0
  69. data/lib/woods/embedding/indexer.rb +90 -46
  70. data/lib/woods/embedding/openai.rb +17 -6
  71. data/lib/woods/evaluation/ablation_executor.rb +6 -1
  72. data/lib/woods/evaluation/ablation_timed_executor.rb +22 -4
  73. data/lib/woods/export/typed_reader.rb +56 -0
  74. data/lib/woods/extractor.rb +557 -228
  75. data/lib/woods/extractors/action_cable_extractor.rb +3 -1
  76. data/lib/woods/extractors/behavioral_profile.rb +9 -7
  77. data/lib/woods/extractors/caching_extractor.rb +3 -1
  78. data/lib/woods/extractors/concern_extractor.rb +64 -6
  79. data/lib/woods/extractors/configuration_extractor.rb +7 -3
  80. data/lib/woods/extractors/controller_extractor.rb +13 -4
  81. data/lib/woods/extractors/database_view_extractor.rb +3 -1
  82. data/lib/woods/extractors/decorator_extractor.rb +3 -1
  83. data/lib/woods/extractors/engine_extractor.rb +3 -1
  84. data/lib/woods/extractors/event_extractor.rb +4 -2
  85. data/lib/woods/extractors/factory_extractor.rb +3 -1
  86. data/lib/woods/extractors/graphql_extractor.rb +8 -2
  87. data/lib/woods/extractors/i18n_extractor.rb +3 -1
  88. data/lib/woods/extractors/job_extractor.rb +6 -19
  89. data/lib/woods/extractors/lib_extractor.rb +3 -1
  90. data/lib/woods/extractors/mailer_extractor.rb +20 -5
  91. data/lib/woods/extractors/manager_extractor.rb +3 -1
  92. data/lib/woods/extractors/method_parameters.rb +53 -0
  93. data/lib/woods/extractors/middleware_argument.rb +65 -0
  94. data/lib/woods/extractors/middleware_extractor.rb +9 -3
  95. data/lib/woods/extractors/migration_extractor.rb +3 -1
  96. data/lib/woods/extractors/model_extractor.rb +39 -33
  97. data/lib/woods/extractors/package_extractor.rb +24 -4
  98. data/lib/woods/extractors/phlex_extractor.rb +3 -1
  99. data/lib/woods/extractors/policy_extractor.rb +3 -1
  100. data/lib/woods/extractors/poro_extractor.rb +3 -1
  101. data/lib/woods/extractors/pundit_extractor.rb +3 -1
  102. data/lib/woods/extractors/rails_source_extractor.rb +4 -2
  103. data/lib/woods/extractors/rake_task_extractor.rb +4 -2
  104. data/lib/woods/extractors/route_extractor.rb +3 -1
  105. data/lib/woods/extractors/route_helper_resolver.rb +10 -33
  106. data/lib/woods/extractors/scheduled_job_extractor.rb +41 -15
  107. data/lib/woods/extractors/serializer_extractor.rb +4 -2
  108. data/lib/woods/extractors/service_extractor.rb +3 -1
  109. data/lib/woods/extractors/shared_dependency_scanner.rb +2 -2
  110. data/lib/woods/extractors/shared_utility_methods.rb +27 -15
  111. data/lib/woods/extractors/source_nesting.rb +1 -1
  112. data/lib/woods/extractors/state_machine_extractor.rb +3 -1
  113. data/lib/woods/extractors/test_mapping_extractor.rb +3 -1
  114. data/lib/woods/extractors/validator_extractor.rb +3 -1
  115. data/lib/woods/extractors/view_component_extractor.rb +3 -1
  116. data/lib/woods/extractors/view_template_extractor.rb +3 -1
  117. data/lib/woods/flow_assembler.rb +87 -8
  118. data/lib/woods/flow_precomputer.rb +44 -7
  119. data/lib/woods/gem_mapper.rb +2 -0
  120. data/lib/woods/git_history.rb +116 -0
  121. data/lib/woods/graph_analyzer.rb +195 -63
  122. data/lib/woods/hooks/context_cli.rb +54 -0
  123. data/lib/woods/hooks/context_event.rb +88 -0
  124. data/lib/woods/hooks/context_hint.rb +73 -0
  125. data/lib/woods/hooks/context_impact.rb +77 -0
  126. data/lib/woods/hooks/context_output.rb +47 -0
  127. data/lib/woods/hooks/context_state.rb +102 -0
  128. data/lib/woods/hooks/refresh.rb +79 -0
  129. data/lib/woods/hooks/rule_projection.rb +78 -0
  130. data/lib/woods/input_rules.rb +19 -0
  131. data/lib/woods/mcp/bearer_auth.rb +20 -12
  132. data/lib/woods/mcp/bootstrapper.rb +62 -0
  133. data/lib/woods/mcp/index_reader.rb +323 -160
  134. data/lib/woods/mcp/initialization_guidance.rb +27 -0
  135. data/lib/woods/mcp/origin_guard.rb +17 -9
  136. data/lib/woods/mcp/published_lexical_retriever.rb +115 -0
  137. data/lib/woods/mcp/renderers/markdown_renderer.rb +8 -1
  138. data/lib/woods/mcp/renderers/plain_renderer.rb +7 -1
  139. data/lib/woods/mcp/search_results.rb +74 -0
  140. data/lib/woods/mcp/server.rb +158 -37
  141. data/lib/woods/mcp/tool_contract.rb +2 -0
  142. data/lib/woods/mcp/tool_response_renderer.rb +25 -0
  143. data/lib/woods/mcp/traversal_evidence.rb +113 -0
  144. data/lib/woods/mcp/traversal_evidence_index.rb +100 -0
  145. data/lib/woods/mcp/traversal_evidence_page.rb +41 -0
  146. data/lib/woods/mcp/traversal_evidence_text.rb +52 -0
  147. data/lib/woods/notion/exporter.rb +56 -17
  148. data/lib/woods/obsidian/destination_plan.rb +98 -0
  149. data/lib/woods/obsidian/name_mapper.rb +19 -3
  150. data/lib/woods/obsidian/note_builder.rb +19 -10
  151. data/lib/woods/obsidian/vault_exporter.rb +88 -32
  152. data/lib/woods/operator/pipeline_guard.rb +18 -13
  153. data/lib/woods/path_dispatcher.rb +7 -1
  154. data/lib/woods/payload_store.rb +29 -15
  155. data/lib/woods/railtie.rb +3 -3
  156. data/lib/woods/railtie_support.rb +12 -12
  157. data/lib/woods/rake_helpers.rb +392 -0
  158. data/lib/woods/resilience/graph_invariant_validator/membership_checks.rb +71 -0
  159. data/lib/woods/resilience/graph_invariant_validator/node_checks.rb +61 -0
  160. data/lib/woods/resilience/graph_invariant_validator/reverse_relationship_checks.rb +46 -0
  161. data/lib/woods/resilience/graph_invariant_validator.rb +119 -0
  162. data/lib/woods/resilience/index_validator/graph_checks.rb +80 -0
  163. data/lib/woods/resilience/index_validator.rb +112 -23
  164. data/lib/woods/retrieval/context_assembler.rb +50 -15
  165. data/lib/woods/retrieval/lexical_assembler.rb +73 -0
  166. data/lib/woods/retrieval/lexical_index.rb +119 -0
  167. data/lib/woods/retrieval/ranker.rb +4 -2
  168. data/lib/woods/retrieval/scope.rb +108 -0
  169. data/lib/woods/retrieval/scoped_graph_store.rb +32 -0
  170. data/lib/woods/retrieval/scoped_vector_store.rb +55 -0
  171. data/lib/woods/retrieval/search_executor.rb +86 -27
  172. data/lib/woods/retrieval/source_evidence.rb +200 -0
  173. data/lib/woods/retriever.rb +98 -22
  174. data/lib/woods/ruby_analyzer/trace_enricher.rb +80 -38
  175. data/lib/woods/session_tracer/middleware.rb +10 -12
  176. data/lib/woods/session_tracer/redis_store.rb +22 -6
  177. data/lib/woods/session_tracer/session_flow_assembler.rb +23 -17
  178. data/lib/woods/session_tracer/solid_cache_coordination.rb +6 -4
  179. data/lib/woods/session_tracer/unit_resolver.rb +63 -0
  180. data/lib/woods/source_inputs/consumer_errors.rb +27 -0
  181. data/lib/woods/source_inputs/handoff.rb +102 -0
  182. data/lib/woods/source_inputs/launcher.rb +157 -0
  183. data/lib/woods/source_inputs/manifest.rb +124 -0
  184. data/lib/woods/source_inputs/private_key.rb +55 -0
  185. data/lib/woods/source_inputs/scanner.rb +171 -0
  186. data/lib/woods/source_inputs/scopes.rb +71 -0
  187. data/lib/woods/source_inputs/session.rb +214 -0
  188. data/lib/woods/source_inputs/status.rb +84 -0
  189. data/lib/woods/source_inputs/verifier.rb +107 -0
  190. data/lib/woods/storage/metadata_store.rb +25 -25
  191. data/lib/woods/storage/pgvector.rb +29 -8
  192. data/lib/woods/storage/qdrant.rb +17 -7
  193. data/lib/woods/storage/vector_store.rb +18 -6
  194. data/lib/woods/tasks.rb +3 -2
  195. data/lib/woods/temporal/json_snapshot_store.rb +29 -8
  196. data/lib/woods/unblocked/exporter.rb +59 -70
  197. data/lib/woods/version.rb +1 -1
  198. data/lib/woods/watch/boot_snapshot.rb +52 -0
  199. data/lib/woods/watch/daemon.rb +136 -28
  200. data/lib/woods/watch/listen_watcher.rb +4 -0
  201. data/lib/woods/watch/polling_watcher.rb +5 -1
  202. data/lib/woods/watch/status.rb +20 -15
  203. data/lib/woods/watch/tree_scan.rb +21 -13
  204. data/lib/woods/watch/watcher.rb +4 -1
  205. data/lib/woods.rb +135 -11
  206. data/plugin/.claude-plugin/plugin.json +1 -1
  207. data/plugin/hooks/adapters/normalize.jq +15 -0
  208. data/plugin/hooks/adapters/normalize.rb +63 -0
  209. data/plugin/hooks/hooks.json +20 -0
  210. data/plugin/hooks/woods-context.sh +50 -0
  211. data/plugin/hooks/woods-input-rules.sh +159 -0
  212. data/plugin/hooks/woods-opencode.mjs +65 -0
  213. data/plugin/hooks/woods-post-edit.sh +2 -225
  214. data/plugin/hooks/woods-refresh.sh +260 -0
  215. data/plugin/hooks/woods-session-start.sh +47 -55
  216. data/plugin/skills/woods-agent-enable/SKILL.md +13 -0
  217. data/plugin/skills/woods-diagnose/SKILL.md +288 -1
  218. data/plugin/skills/woods-investigate/SKILL.md +106 -0
  219. data/plugin/skills/woods-mcp-config/SKILL.md +89 -1
  220. data/plugin/skills/woods-setup/SKILL.md +107 -6
  221. metadata +84 -5
@@ -0,0 +1,80 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Woods
4
+ module Resilience
5
+ class IndexValidator
6
+ # Reuse the published reader's retention pin but inspect raw graph and
7
+ # unit artifacts independently, without normalizing away bad records.
8
+ module GraphChecks
9
+ private
10
+
11
+ def with_validation_payload
12
+ root = Pathname.new(@index_dir)
13
+ if root.join('manifest.json').exist? || root.join('generation.json').exist?
14
+ Woods::PublishedIndex::GenerationCatalog.pointer(root)
15
+ reader = Woods::MCP::IndexReader.new(root)
16
+ reader.with_pinned_generation do
17
+ @validation_payload = reader.payload_dir.to_s
18
+ @validation_manifest = reader.manifest
19
+ yield
20
+ end
21
+ else
22
+ @validation_payload = @index_dir
23
+ yield
24
+ end
25
+ ensure
26
+ @validation_payload = nil
27
+ @validation_manifest = nil
28
+ @graph_index_entries = nil
29
+ end
30
+
31
+ def static_source_map?
32
+ manifest = @validation_manifest
33
+ manifest.is_a?(Hash) && manifest['provenance'].is_a?(Hash) &&
34
+ manifest['provenance']['mode'] == 'woods_static_ruby_source'
35
+ end
36
+
37
+ def validation_index_entries(path, errors)
38
+ entries = JSON.parse(Woods::AtomicFile.read(path))
39
+ unless entries.is_a?(Array)
40
+ errors << "#{path}: expected an array"
41
+ return []
42
+ end
43
+ entries.select do |entry|
44
+ valid = entry.is_a?(Hash) && entry['identifier'].is_a?(String) && !entry['identifier'].empty?
45
+ errors << "#{path}: entry requires a nonempty identifier" unless valid
46
+ valid
47
+ end
48
+ end
49
+
50
+ def collect_graph_index_entry(type_dir, entry, data, errors)
51
+ directory = File.basename(type_dir)
52
+ types = Woods::MCP::IndexReader::UNIT_TYPES_BY_DIR.fetch(directory)
53
+ typed_entry = graph_index_entry(entry, data, types)
54
+ @graph_index_entries << typed_entry
55
+ return unless data
56
+
57
+ file = find_unit_file(type_dir, entry['identifier'])
58
+ unless data['identifier'] == entry['identifier'] && types.include?(data['type'])
59
+ errors << "#{file}: expected typed unit #{typed_entry['type']}:#{entry['identifier']}"
60
+ return
61
+ end
62
+ validate_graph_unit_path(file, entry, data, errors)
63
+ end
64
+
65
+ def graph_index_entry(entry, data, types)
66
+ type = data && types.include?(data['type']) ? data['type'] : types.first
67
+ result = entry.merge('type' => type)
68
+ result['file_path'] = data['file_path'] if data&.key?('file_path') && !entry.key?('file_path')
69
+ result
70
+ end
71
+
72
+ def validate_graph_unit_path(file, entry, data, errors)
73
+ return unless entry.key?('file_path') && data.key?('file_path') && entry['file_path'] != data['file_path']
74
+
75
+ errors << "#{file}: file_path differs from #{File.dirname(file)}/_index.json for #{entry['identifier']}"
76
+ end
77
+ end
78
+ end
79
+ end
80
+ end
@@ -2,10 +2,16 @@
2
2
 
3
3
  require 'json'
4
4
  require 'set'
5
+ require 'rubygems/version'
6
+ require_relative '../version'
5
7
  require_relative '../filename_utils'
6
8
  require_relative '../atomic_file'
7
9
 
8
10
  require_relative '../generation'
11
+ require_relative '../published_index'
12
+ require_relative 'graph_invariant_validator'
13
+ require_relative 'index_validator/graph_checks'
14
+ require_relative '../source_inputs/manifest'
9
15
 
10
16
  module Woods
11
17
  module Resilience
@@ -16,6 +22,8 @@ module Woods
16
22
  # - All files referenced in the index exist on disk
17
23
  # - Content hashes (source_hash) match the actual source_code
18
24
  # - No stale unit files exist that aren't listed in the index
25
+ # - Typed graph identities and reverse/file/type memberships agree
26
+ # All checks share one pinned generation; the validator never repairs it.
19
27
  #
20
28
  # **This class knows nothing about vectors or embedding dimensions.** Six
21
29
  # documents used to credit it with detecting dimension mismatches; it never
@@ -23,10 +31,8 @@ module Woods
23
31
  # `Tasks.verify_store_dimensions!` before a durable embed run and by
24
32
  # {Woods::Storage::Snapshotter::Vector} at MCP boot.
25
33
  #
26
- # Consumed by `spec/integration/multi_worktree_spec.rb` as a per-worktree
27
- # integrity oracle. The `woods:validate` rake task performs an overlapping
28
- # check inline rather than calling this — deliberate duplication left alone
29
- # for now, since the task's output format is user-facing.
34
+ # Shared by the `woods:validate` rake task and worktree integration checks.
35
+ # Writer-version mismatches are advisory; structural errors still fail.
30
36
  #
31
37
  # @example
32
38
  # validator = IndexValidator.new(index_dir: "tmp/woods")
@@ -34,6 +40,7 @@ module Woods
34
40
  # puts report.errors if !report.valid?
35
41
  class IndexValidator # rubocop:disable Metrics/ClassLength
36
42
  include Woods::FilenameUtils
43
+ include GraphChecks
37
44
 
38
45
  # Report produced by {#validate}.
39
46
  #
@@ -84,13 +91,19 @@ module Woods
84
91
  return ValidationReport.new(valid?: false, warnings: warnings, errors: errors)
85
92
  end
86
93
 
87
- payload_type_dirs(errors).each do |type_dir|
88
- validate_type_directory(type_dir, warnings, errors)
94
+ with_validation_payload do
95
+ @graph_index_entries = []
96
+ payload_type_dirs(errors).each do |type_dir|
97
+ validate_type_directory(type_dir, warnings, errors)
98
+ end
99
+ validate_flow_artifacts(errors)
100
+ validate_against_manifest(warnings, errors)
89
101
  end
90
- validate_flow_artifacts(errors)
91
- validate_against_manifest(warnings, errors)
92
102
 
93
103
  ValidationReport.new(valid?: errors.empty?, warnings: warnings, errors: errors)
104
+ rescue IOError, SystemCallError, ArgumentError, JSON::ParserError, Woods::PublishedIndex::CorruptPointerError => e
105
+ errors << "Cannot read published index: #{e.class}: #{e.message}"
106
+ ValidationReport.new(valid?: false, warnings: warnings, errors: errors)
94
107
  end
95
108
 
96
109
  # The checks `woods:validate` used to carry inline: manifest counts
@@ -107,19 +120,72 @@ module Woods
107
120
  manifest_path = File.join(payload, 'manifest.json')
108
121
  return unless File.exist?(manifest_path)
109
122
 
110
- manifest = JSON.parse(Woods::AtomicFile.read(manifest_path))
123
+ manifest = read_validation_manifest(manifest_path, errors)
124
+ return unless manifest
125
+
126
+ validate_writer_version(manifest['woods_version'], warnings)
111
127
  unresolvable = Hash.new { |hash, key| hash[key] = [] }
112
128
 
113
- (manifest['counts'] || {}).each do |type, expected_count|
129
+ manifest.fetch('counts', {}).each do |type, expected_count|
114
130
  validate_manifest_type(payload, type, expected_count, unresolvable, warnings, errors)
115
131
  end
116
132
 
117
133
  warn_unresolvable_paths(warnings, unresolvable)
118
134
  validate_dependency_graph(payload, errors)
135
+ validate_source_inputs(payload, errors)
136
+ end
137
+
138
+ # Optional for old generations; malformed new provenance is an artifact
139
+ # integrity error. Source drift itself is advisory and belongs to status.
140
+ def validate_source_inputs(payload, errors)
141
+ path = File.join(payload, SourceInputs::Manifest::FILE_NAME)
142
+ return unless File.exist?(path)
143
+
144
+ File.open(path, File::RDONLY | File::NONBLOCK) do |file|
145
+ raise SourceInputs::Manifest::Invalid unless file.stat.file?
146
+
147
+ SourceInputs::Manifest.parse(file.read(SourceInputs::Manifest::MAX_BYTES + 1))
148
+ end
149
+ rescue SourceInputs::Manifest::Invalid, SystemCallError, IOError
150
+ errors << 'Invalid source_inputs.json provenance artifact'
151
+ end
152
+
153
+ def read_validation_manifest(path, errors)
154
+ manifest = JSON.parse(Woods::AtomicFile.read(path))
155
+ unless manifest.is_a?(Hash)
156
+ errors << 'manifest.json: expected an object'
157
+ return
158
+ end
159
+ unless manifest.fetch('counts', {}).is_a?(Hash)
160
+ errors << 'manifest.json counts: expected an object'
161
+ return
162
+ end
163
+
164
+ manifest
165
+ end
166
+
167
+ # The version records the last publisher, not the producer of every
168
+ # retained unit. Missing provenance is normal for older indexes.
169
+ def validate_writer_version(value, warnings)
170
+ return if value.nil?
171
+
172
+ unless value.is_a?(String) && !value.strip.empty?
173
+ warnings << 'Invalid manifest woods_version; writer provenance is unknown. Run a full woods:extract.'
174
+ return
175
+ end
176
+
177
+ writer = Gem::Version.new(value)
178
+ reader = Gem::Version.new(Woods::VERSION)
179
+ return if writer.segments.first == reader.segments.first
180
+
181
+ warnings << "Index last published by Woods #{value}; reader is Woods #{Woods::VERSION}. " \
182
+ 'Major versions differ. Run a full woods:extract before relying on compatibility.'
183
+ rescue ArgumentError
184
+ warnings << 'Invalid manifest woods_version; writer provenance is unknown. Run a full woods:extract.'
119
185
  end
120
186
 
121
- # Split each type's unresolvable units into app-tree paths and gem-owned
122
- # paths, since only the first has a remedy the operator can act on.
187
+ # Split unresolvable app-tree paths from gem-owned paths so each warning
188
+ # can distinguish a different filesystem environment from a changed bundle.
123
189
  #
124
190
  # @param warnings [Array<String>]
125
191
  # @param unresolvable [Hash{String => Array<Array(String, Boolean)>}]
@@ -143,15 +209,16 @@ module Woods
143
209
 
144
210
  # Absolute paths outside the app root belong to a gem — an engine model,
145
211
  # a framework source — and resolve only where that gem is installed at
146
- # the extracting path. Re-running extraction here cannot change that,
147
- # so the remedy is not offered.
212
+ # the extracting path. A changed bundle requires fresh extraction; a
213
+ # reader on another filesystem instead needs the original gem paths.
148
214
  def warn_unresolvable_gem_paths(warnings, type, identifiers)
149
215
  return if identifiers.empty?
150
216
 
151
217
  warnings << "#{type}: #{identifiers.size} unit(s) whose file_path lies outside the app root " \
152
218
  "and is absent here (e.g. #{identifiers.first(3).join(', ')}). Gem-owned units " \
153
219
  '(engine models, framework sources) resolve only where that gem is installed at ' \
154
- 'the extracting path.'
220
+ 'the extracting path. After a bundle update, run woods:extract in a fresh process ' \
221
+ 'with the updated bundle, then woods:validate.'
155
222
  end
156
223
 
157
224
  # rubocop:disable-next Metrics/ParameterLists
@@ -175,6 +242,10 @@ module Woods
175
242
  # @param errors [Array<String>]
176
243
  def validate_unit_file(file, type, unresolvable, errors)
177
244
  data = JSON.parse(Woods::AtomicFile.read(file))
245
+ unless data.is_a?(Hash)
246
+ errors << "#{file}: expected a unit object"
247
+ return
248
+ end
178
249
  errors << "#{file}: missing identifier" unless data['identifier']
179
250
  errors << "#{file}: missing source_code" unless data['source_code']
180
251
  file_path = data['file_path']
@@ -211,13 +282,14 @@ module Woods
211
282
  return
212
283
  end
213
284
 
214
- JSON.parse(Woods::AtomicFile.read(graph_path))
285
+ graph = JSON.parse(Woods::AtomicFile.read(graph_path))
286
+ errors.concat(GraphInvariantValidator.new(graph: graph, index_entries: @graph_index_entries).validate)
215
287
  rescue JSON::ParserError
216
288
  errors << 'dependency_graph.json: invalid JSON'
217
289
  end
218
290
 
219
291
  def payload_dir
220
- Woods::Generation.new(output_dir: @index_dir).payload_dir.to_s
292
+ @validation_payload || Woods::Generation.new(output_dir: @index_dir).payload_dir.to_s
221
293
  end
222
294
 
223
295
  private
@@ -292,12 +364,18 @@ module Woods
292
364
  # failure RAISES: {#payload_type_dirs} converts it to a validation
293
365
  # error rather than degrading to a silently empty allowlist.
294
366
  #
367
+ # The explicit Woods static-map provenance additionally admits the
368
+ # GemMapper's own type families; arbitrary directories stay excluded.
295
369
  # `flows/` is deliberately absent: it holds `flow_index.json` and
296
370
  # per-flow documents, which {#validate_flow_artifacts} owns.
297
371
  #
298
372
  # @return [Array<String>]
299
373
  def type_directory_allowlist
300
- @type_directory_allowlist ||= self.class.unit_type_directories
374
+ directories = self.class.unit_type_directories
375
+ return directories unless static_source_map?
376
+
377
+ require_relative '../gem_mapper'
378
+ directories | Woods::GemMapper::TYPE_DIRECTORIES.values
301
379
  end
302
380
 
303
381
  # Validate the flows/ artifact family (G-2): `flow_index.json` parses,
@@ -361,13 +439,12 @@ module Woods
361
439
  return
362
440
  end
363
441
 
364
- index_entries = JSON.parse(Woods::AtomicFile.read(index_path))
365
442
  indexed_identifiers = Set.new
366
-
367
- index_entries.each do |entry|
443
+ validation_index_entries(index_path, errors).each do |entry|
368
444
  identifier = entry['identifier']
369
445
  indexed_identifiers << identifier
370
- validate_index_entry(type_dir, type_name, identifier, errors)
446
+ data = validate_index_entry(type_dir, type_name, identifier, errors)
447
+ collect_graph_index_entry(type_dir, entry, data, errors)
371
448
  end
372
449
 
373
450
  check_stale_files(type_dir, type_name, indexed_identifiers, warnings)
@@ -413,11 +490,23 @@ module Woods
413
490
  # @param errors [Array<String>] Accumulated errors
414
491
  def validate_content_hash(unit_file, identifier, errors)
415
492
  data = JSON.parse(Woods::AtomicFile.read(unit_file))
493
+ unless data.is_a?(Hash)
494
+ errors << "#{unit_file}: expected a unit object"
495
+ return
496
+ end
497
+ check_content_hash(data, identifier, errors)
498
+ data
499
+ end
500
+
501
+ def check_content_hash(data, identifier, errors)
416
502
  source_code = data['source_code']
417
503
  stored_hash = data['source_hash']
418
-
419
504
  return unless source_code && stored_hash
420
505
 
506
+ unless source_code.is_a?(String) && stored_hash.is_a?(String)
507
+ errors << "#{identifier}: source_code and source_hash must be strings"
508
+ return
509
+ end
421
510
  expected_hash = Digest::SHA256.hexdigest(source_code)
422
511
  return if stored_hash == expected_hash
423
512
 
@@ -2,6 +2,7 @@
2
2
 
3
3
  require_relative 'search_executor'
4
4
  require_relative '../token_utils'
5
+ require_relative 'source_evidence'
5
6
 
6
7
  module Woods
7
8
  module Retrieval
@@ -103,7 +104,18 @@ module Woods
103
104
  # @param structural_context [String, nil] Optional codebase overview text
104
105
  # @param budget [Integer, nil] Override token budget; falls back to @budget
105
106
  # @return [AssembledContext] Token-budgeted context with source attribution
106
- def assemble(candidates:, classification:, structural_context: nil, budget: nil)
107
+ def assemble(**options)
108
+ # Pipeline assemblers are shared by concurrent Ruby/HTTP callers. Keep
109
+ # per-request metadata, query, mode and generation on a private worker.
110
+ dup.send(:assemble_request, **options)
111
+ end
112
+
113
+ def assemble_request(candidates:, classification:, structural_context: nil, budget: nil,
114
+ evidence: 'full', query: nil, generation: nil)
115
+ SourceEvidence.validate_mode!(evidence)
116
+ @evidence_mode = evidence
117
+ @evidence_query = query
118
+ @evidence_generation = generation
107
119
  effective_budget = budget || @budget
108
120
  sections = []
109
121
  sources = []
@@ -132,6 +144,22 @@ module Woods
132
144
  build_result(sections, sources, effective_budget, @skipped_missing_metadata)
133
145
  end
134
146
 
147
+ private :assemble_request
148
+
149
+ # Estimate token count. Prefers the injected {TokenCounter} — which
150
+ # loads the provider's real tokenizer and returns exact counts — and
151
+ # falls back to the configured chars-per-token ratio when no counter
152
+ # is wired.
153
+ #
154
+ # @param text [String]
155
+ # @return [Integer]
156
+ def estimate_tokens(text)
157
+ return 0 if text.nil? || text.empty?
158
+ return @token_counter.count(text) if @token_counter
159
+
160
+ (text.length / @chars_per_token).ceil
161
+ end
162
+
135
163
  private
136
164
 
137
165
  # Suffix the Indexer appends when a single unit is split into multiple
@@ -293,6 +321,10 @@ module Woods
293
321
  return tokens_used
294
322
  end
295
323
 
324
+ if @evidence_mode != 'full'
325
+ return append_compact_candidate(parts, sources, candidate, unit, budget, tokens_used)
326
+ end
327
+
296
328
  text = format_unit(unit, candidate)
297
329
  tokens = estimate_tokens(text)
298
330
  remaining = budget - tokens_used
@@ -308,6 +340,23 @@ module Woods
308
340
  end
309
341
  end
310
342
 
343
+ def append_compact_candidate(parts, sources, candidate, unit, budget, tokens_used)
344
+ header = "## #{unit_field(unit, :identifier)} (#{unit_field(unit, :type)})\n" \
345
+ "File: #{unit_field(unit, :file_path)}\n\n"
346
+ remaining = budget - tokens_used
347
+ return tokens_used unless remaining.positive?
348
+
349
+ evidence = SourceEvidence.new(unit: unit, query: @evidence_query, generation: @evidence_generation)
350
+ .render(mode: @evidence_mode, budget: remaining,
351
+ counter: ->(text) { estimate_tokens(header + text) })
352
+ return tokens_used if evidence.text.empty?
353
+
354
+ text = header + evidence.text
355
+ parts << text
356
+ sources << build_source_attribution(candidate, unit).merge(evidence: evidence.provenance)
357
+ tokens_used + estimate_tokens(text)
358
+ end
359
+
311
360
  # Format a unit for inclusion in context.
312
361
  #
313
362
  # @param unit [Hash] Unit data from metadata store
@@ -400,20 +449,6 @@ module Woods
400
449
  "#{text[0...target_chars]}\n... [truncated]"
401
450
  end
402
451
 
403
- # Estimate token count. Prefers the injected {TokenCounter} — which
404
- # loads the provider's real tokenizer and returns exact counts — and
405
- # falls back to the configured chars-per-token ratio when no counter
406
- # is wired.
407
- #
408
- # @param text [String]
409
- # @return [Integer]
410
- def estimate_tokens(text)
411
- return 0 if text.nil? || text.empty?
412
- return @token_counter.count(text) if @token_counter
413
-
414
- (text.length / @chars_per_token).ceil
415
- end
416
-
417
452
  # Effective chars-per-token for chunk-size sizing. When an exact
418
453
  # counter is present, prefer its native ratio (e.g. 1.2 for
419
454
  # nomic-embed-text) so truncation and estimation agree. Falls back
@@ -0,0 +1,73 @@
1
+ # frozen_string_literal: true
2
+
3
+ require_relative 'context_assembler'
4
+ require_relative 'lexical_index'
5
+
6
+ module Woods
7
+ module Retrieval
8
+ # Lexical context keeps matching evidence visible and charges every notice,
9
+ # header and truncation marker to the same character-based token estimate.
10
+ class LexicalAssembler
11
+ def estimate_tokens(text)
12
+ (text.length / 4.0).ceil
13
+ end
14
+
15
+ def assemble(candidates:, budget:, evidence: 'full', query: nil, generation: nil, **)
16
+ SourceEvidence.validate_mode!(evidence)
17
+ raise ArgumentError, 'budget must be a positive Integer' unless budget.is_a?(Integer) && budget.positive?
18
+
19
+ context = 'Mode: lexical (field-aware BM25; ranked top 20; token counts estimated).'
20
+ context += "\nNo lexical matches." if candidates.empty?
21
+ context = context[0, budget * 4]
22
+ sources = []
23
+ candidates.each do |candidate|
24
+ unit = candidate.metadata
25
+ header = "\n\n## #{unit['identifier']} (#{unit['type']})\nFile: #{unit['file_path']}\n" \
26
+ "Matched: #{candidate.matched_fields.join(', ')}\n\n"
27
+ remaining = (budget * 4) - context.length - header.length
28
+ next unless remaining.positive?
29
+
30
+ if evidence != 'full'
31
+ selected = SourceEvidence.new(unit: unit, query: query, generation: generation)
32
+ .render(mode: evidence, budget: budget,
33
+ counter: ->(text) { estimate_tokens(context + header + text) })
34
+ next if selected.text.empty?
35
+
36
+ context += header + selected.text
37
+ sources << { identifier: unit['identifier'], type: unit['type'], file_path: unit['file_path'],
38
+ score: candidate.score, matched_fields: candidate.matched_fields,
39
+ evidence: selected.provenance }
40
+ next
41
+ end
42
+
43
+ source = evidence_text(candidate)
44
+ truncated = source.length > remaining
45
+ marker = "\n[Published evidence truncated; use lookup for the full unit.]"
46
+ next if truncated && remaining <= marker.length
47
+
48
+ context += header + (truncated ? source[0, remaining - marker.length] + marker : source)
49
+ sources << { identifier: unit['identifier'], type: unit['type'], file_path: unit['file_path'],
50
+ score: candidate.score, matched_fields: candidate.matched_fields, truncated: truncated }
51
+ break if truncated
52
+ end
53
+ AssembledContext.new(context: context, tokens_used: estimate_tokens(context), budget: budget,
54
+ sources: sources, sections: [:primary], skipped_missing_metadata: 0)
55
+ end
56
+
57
+ private
58
+
59
+ def evidence_text(candidate)
60
+ unit = candidate.metadata
61
+ source = unit['source_code'].to_s
62
+ return source unless candidate.matched_fields.any? { |field| field.start_with?('runtime:') }
63
+
64
+ metadata = unit['metadata'].is_a?(Hash) ? unit['metadata'] : {}
65
+ values = LexicalIndex::RUNTIME_FIELDS.each_with_object({}) do |key, selected|
66
+ value = metadata[key] || unit[key]
67
+ selected[key] = value unless value.nil?
68
+ end
69
+ "Published runtime metadata:\n#{JSON.pretty_generate(values)}\n\n#{source}"
70
+ end
71
+ end
72
+ end
73
+ end
@@ -0,0 +1,119 @@
1
+ # frozen_string_literal: true
2
+
3
+ require 'json'
4
+ require_relative 'query_classifier'
5
+ require_relative 'search_executor'
6
+
7
+ module Woods
8
+ module Retrieval
9
+ # Immutable field-aware lexical snapshot. Only published unit values enter
10
+ # the vocabulary; bookkeeping and arbitrary JSON keys cannot become hits.
11
+ class LexicalIndex
12
+ FIELD_WEIGHTS = { 'identifier' => 4.0, 'file_path' => 2.0, 'source_code' => 1.0, 'runtime' => 2.0 }.freeze
13
+ RUNTIME_FIELDS = %w[callbacks associations validations scopes concerns included_modules
14
+ methods instance_methods class_methods actions routes columns table_name
15
+ description purpose dependencies].freeze
16
+ K1 = 1.2
17
+ B = 0.75
18
+ Document = Struct.new(:key, :unit, :fields, keyword_init: true)
19
+ private_constant :Document
20
+
21
+ def initialize(metadata_store:)
22
+ @documents = metadata_store.all_identifiers.sort.map do |key|
23
+ unit = metadata_store.find(key)
24
+ raise ArgumentError, "missing unit metadata for #{key.inspect}" unless unit.is_a?(Hash)
25
+
26
+ unit = JSON.parse(JSON.generate(unit))
27
+ fields = field_values(unit).transform_values { |value| tokenize(value).tally.freeze }.freeze
28
+ Document.new(key: key.freeze, unit: deep_freeze(unit), fields: fields).freeze
29
+ end.compact.freeze
30
+ @averages = FIELD_WEIGHTS.to_h do |field, _|
31
+ lengths = @documents.map { |doc| doc.fields.fetch(field).values.sum }
32
+ [field, lengths.empty? ? 1.0 : [lengths.sum.fdiv(lengths.size), 1.0].max]
33
+ end.freeze
34
+ @frequencies = Hash.new(0)
35
+ @documents.each { |doc| doc.fields.values.flat_map(&:keys).uniq.each { |term| @frequencies[term] += 1 } }
36
+ @frequencies.freeze
37
+ end
38
+
39
+ def execute(query:, limit: 20, type_filter: nil, exclude_types: nil)
40
+ terms = tokenize(query).uniq
41
+ candidates = @documents.filter_map do |doc|
42
+ next unless eligible?(doc.unit, type_filter, exclude_types)
43
+
44
+ score, fields = score_document(doc, terms)
45
+ exact = doc.unit['identifier'].to_s.casecmp?(query.strip)
46
+ score += 1.0 if exact
47
+ fields << 'identifier:exact' if exact
48
+ next unless score.positive?
49
+
50
+ SearchExecutor::Candidate.new(identifier: doc.key, score: score, source: :lexical,
51
+ metadata: doc.unit, matched_fields: fields.sort)
52
+ end
53
+ candidates.sort_by! do |candidate|
54
+ [candidate.matched_fields.include?('identifier:exact') ? 0 : 1, -candidate.score, candidate.identifier]
55
+ end
56
+ SearchExecutor::ExecutionResult.new(candidates: candidates.first(limit), strategy: :lexical, query: query)
57
+ end
58
+
59
+ private
60
+
61
+ def tokenize(value)
62
+ value.to_s.gsub(/(\p{Ll}|\d)(\p{Lu})/u, '\1 \2')
63
+ .gsub(/(\p{Lu})(\p{Lu}\p{Ll})/u, '\1 \2').downcase
64
+ .scan(/[\p{L}\p{N}]+/u).reject { |term| QueryClassifier::STOP_WORDS.include?(term) }
65
+ end
66
+
67
+ def field_values(unit)
68
+ metadata = unit['metadata'].is_a?(Hash) ? unit['metadata'] : {}
69
+ runtime = RUNTIME_FIELDS.filter_map { |key| metadata[key] || unit[key] }
70
+ { 'identifier' => unit['identifier'], 'file_path' => unit['file_path'],
71
+ 'source_code' => unit['source_code'], 'runtime' => values_text(runtime) }
72
+ end
73
+
74
+ def values_text(value)
75
+ case value
76
+ when Hash then value.values.map { |child| values_text(child) }.join(' ')
77
+ when Array then value.map { |child| values_text(child) }.join(' ')
78
+ when String, Symbol, Numeric then value.to_s
79
+ else ''
80
+ end
81
+ end
82
+
83
+ def eligible?(unit, allowed, excluded)
84
+ return allowed.map(&:to_s).include?(unit['type']) if allowed && !allowed.empty?
85
+
86
+ !Array(excluded).map(&:to_s).include?(unit['type'])
87
+ end
88
+
89
+ def score_document(doc, terms)
90
+ fields = []
91
+ score = FIELD_WEIGHTS.sum do |field, weight|
92
+ counts = doc.fields.fetch(field)
93
+ normalization = K1 * (1 - B + (B * counts.values.sum / @averages.fetch(field)))
94
+ terms.sum do |term|
95
+ frequency = counts.fetch(term, 0)
96
+ next 0.0 if frequency.zero?
97
+
98
+ fields << "#{field}:#{term}"
99
+ document_frequency = @frequencies.fetch(term)
100
+ idf = Math.log(1 + ((@documents.size - document_frequency + 0.5) / (document_frequency + 0.5)))
101
+ weight * idf * frequency * (K1 + 1) / (frequency + normalization)
102
+ end
103
+ end
104
+ [score, fields]
105
+ end
106
+
107
+ def deep_freeze(value)
108
+ case value
109
+ when Hash then value.each do |key, child|
110
+ key.freeze
111
+ deep_freeze(child)
112
+ end
113
+ when Array then value.each { |child| deep_freeze(child) }
114
+ end
115
+ value.freeze
116
+ end
117
+ end
118
+ end
119
+ end
@@ -378,7 +378,9 @@ module Woods
378
378
 
379
379
  # Lazily-computed rank-percentile map derived from the graph store's PageRank.
380
380
  #
381
- # Top-ranked identifier gets 1.0, bottom-ranked gets 1/n. Identifiers absent
381
+ # Top-ranked identifier gets 1.0, bottom-ranked gets 1/n. Equal PageRank scores
382
+ # are ordered lexically by identifier so their ordinal percentiles do not
383
+ # depend on graph insertion order or Ruby's unstable sort. Identifiers absent
382
384
  # from PageRank (new units, ephemeral candidates) return nil and fall back
383
385
  # to the bucketed importance signal.
384
386
  #
@@ -402,7 +404,7 @@ module Woods
402
404
  scores = @graph_store.pagerank
403
405
  return {} if scores.nil? || scores.empty?
404
406
 
405
- ranked = scores.sort_by { |_id, score| -score }
407
+ ranked = scores.sort_by { |identifier, score| [-score, identifier] }
406
408
  total = ranked.size.to_f
407
409
  ranked.each_with_index.to_h do |(identifier, _score), rank|
408
410
  [identifier, 1.0 - (rank / total)]