woods 2.0.0.beta2 → 2.0.0.beta4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (233) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +339 -1
  3. data/CONTRIBUTING.md +188 -12
  4. data/README.md +93 -174
  5. data/SECURITY.md +9 -6
  6. data/docs/AGENT_GUIDE.md +109 -8
  7. data/docs/AGENT_SETUP.md +98 -7
  8. data/docs/BACKEND_MATRIX.md +25 -0
  9. data/docs/CLIENT_HOOKS.md +111 -0
  10. data/docs/CONFIGURATION_REFERENCE.md +267 -16
  11. data/docs/CONSOLE_MCP_SETUP.md +80 -7
  12. data/docs/DOCKER_SETUP.md +22 -3
  13. data/docs/EVALUATION.md +464 -1
  14. data/docs/EXTRACTOR_REFERENCE.md +45 -6
  15. data/docs/FAQ.md +11 -12
  16. data/docs/GETTING_STARTED.md +17 -5
  17. data/docs/INCREMENTAL_EXTRACTION.md +147 -7
  18. data/docs/INDEX_LAYOUT.md +382 -0
  19. data/docs/INTERNALS.md +7 -2
  20. data/docs/MCP_SERVERS.md +276 -5
  21. data/docs/MCP_TOOL_COOKBOOK.md +37 -22
  22. data/docs/MCP_WORKTREE_SETUP.md +43 -83
  23. data/docs/NOTION_INTEGRATION.md +13 -0
  24. data/docs/OBSIDIAN_INTEGRATION.md +57 -9
  25. data/docs/PUBLISHED_INDEX.md +72 -0
  26. data/docs/README.md +7 -0
  27. data/docs/RETRIEVAL_GUIDE.md +273 -12
  28. data/docs/RUNTIME_TRACING.md +71 -0
  29. data/docs/SOURCE_FRESHNESS.md +143 -0
  30. data/docs/TROUBLESHOOTING.md +129 -18
  31. data/docs/UNBLOCKED_INTEGRATION.md +25 -0
  32. data/docs/UPGRADING_TO_2.md +48 -22
  33. data/docs/WATCH_DAEMON.md +277 -67
  34. data/exe/woods-agent-config +6 -0
  35. data/exe/woods-extract +5 -0
  36. data/exe/woods-hook-context +6 -0
  37. data/exe/woods-mcp-start +14 -9
  38. data/lib/generators/woods/pgvector_generator.rb +8 -2
  39. data/lib/generators/woods/templates/woods.rb.tt +1 -3
  40. data/lib/tasks/woods.rake +47 -397
  41. data/lib/woods/agent_configuration/applier.rb +135 -0
  42. data/lib/woods/agent_configuration/cli.rb +101 -0
  43. data/lib/woods/agent_configuration/cli_options.rb +29 -0
  44. data/lib/woods/agent_configuration/document.rb +105 -0
  45. data/lib/woods/agent_configuration/error.rb +7 -0
  46. data/lib/woods/agent_configuration/launcher.rb +75 -0
  47. data/lib/woods/agent_configuration/layout.rb +72 -0
  48. data/lib/woods/agent_configuration/managed_section.rb +62 -0
  49. data/lib/woods/agent_configuration/plan.rb +98 -0
  50. data/lib/woods/agent_configuration/plan_diff.rb +38 -0
  51. data/lib/woods/agent_configuration/planned_files.rb +61 -0
  52. data/lib/woods/agent_configuration/planner.rb +63 -0
  53. data/lib/woods/agent_configuration/planner_validation.rb +77 -0
  54. data/lib/woods/agent_configuration/preflight.rb +100 -0
  55. data/lib/woods/agent_configuration/recovery.rb +49 -0
  56. data/lib/woods/ast/node.rb +2 -0
  57. data/lib/woods/ast/parser.rb +38 -5
  58. data/lib/woods/builder.rb +21 -5
  59. data/lib/woods/cache/cache_middleware.rb +28 -7
  60. data/lib/woods/cache/cache_store.rb +4 -5
  61. data/lib/woods/change_set.rb +5 -4
  62. data/lib/woods/console/credential_index.rb +20 -2
  63. data/lib/woods/console/credential_scanner.rb +18 -17
  64. data/lib/woods/console/credential_scanner_registry.rb +36 -0
  65. data/lib/woods/console/dispatch_pipeline.rb +7 -0
  66. data/lib/woods/console/embedded_executor.rb +32 -10
  67. data/lib/woods/console/encrypted_credential_snapshot.rb +16 -0
  68. data/lib/woods/console/rack_middleware.rb +22 -13
  69. data/lib/woods/console/server.rb +18 -16
  70. data/lib/woods/console/sql_noise_stripper.rb +9 -7
  71. data/lib/woods/console/sql_table_scanner.rb +47 -7
  72. data/lib/woods/console/sql_validator.rb +49 -9
  73. data/lib/woods/console/sqlite_read_guard.rb +46 -0
  74. data/lib/woods/coordination/pipeline_lock.rb +3 -2
  75. data/lib/woods/dependency_graph.rb +65 -13
  76. data/lib/woods/embedding/corpus.rb +94 -0
  77. data/lib/woods/embedding/indexer.rb +114 -60
  78. data/lib/woods/embedding/openai.rb +17 -6
  79. data/lib/woods/evaluation/ablation_executor.rb +6 -1
  80. data/lib/woods/evaluation/ablation_timed_executor.rb +22 -4
  81. data/lib/woods/export/typed_reader.rb +56 -0
  82. data/lib/woods/extractor.rb +277 -149
  83. data/lib/woods/extractors/action_cable_extractor.rb +3 -1
  84. data/lib/woods/extractors/behavioral_profile.rb +9 -7
  85. data/lib/woods/extractors/caching_extractor.rb +3 -1
  86. data/lib/woods/extractors/concern_extractor.rb +64 -6
  87. data/lib/woods/extractors/configuration_extractor.rb +7 -3
  88. data/lib/woods/extractors/controller_extractor.rb +13 -4
  89. data/lib/woods/extractors/database_view_extractor.rb +3 -1
  90. data/lib/woods/extractors/declared_parent.rb +55 -0
  91. data/lib/woods/extractors/decorator_extractor.rb +3 -1
  92. data/lib/woods/extractors/engine_extractor.rb +3 -1
  93. data/lib/woods/extractors/event_extractor.rb +4 -2
  94. data/lib/woods/extractors/factory_extractor.rb +3 -1
  95. data/lib/woods/extractors/graphql_extractor.rb +10 -13
  96. data/lib/woods/extractors/i18n_extractor.rb +3 -1
  97. data/lib/woods/extractors/job_extractor.rb +6 -19
  98. data/lib/woods/extractors/lib_extractor.rb +13 -9
  99. data/lib/woods/extractors/mailer_extractor.rb +26 -15
  100. data/lib/woods/extractors/manager_extractor.rb +3 -1
  101. data/lib/woods/extractors/method_parameters.rb +53 -0
  102. data/lib/woods/extractors/middleware_argument.rb +65 -0
  103. data/lib/woods/extractors/middleware_extractor.rb +9 -3
  104. data/lib/woods/extractors/migration_extractor.rb +3 -1
  105. data/lib/woods/extractors/model_extractor.rb +26 -34
  106. data/lib/woods/extractors/package_extractor.rb +24 -4
  107. data/lib/woods/extractors/phlex_extractor.rb +3 -1
  108. data/lib/woods/extractors/policy_extractor.rb +3 -1
  109. data/lib/woods/extractors/poro_extractor.rb +13 -9
  110. data/lib/woods/extractors/pundit_extractor.rb +3 -1
  111. data/lib/woods/extractors/rails_source_extractor.rb +4 -2
  112. data/lib/woods/extractors/rake_task_extractor.rb +4 -2
  113. data/lib/woods/extractors/route_extractor.rb +3 -1
  114. data/lib/woods/extractors/route_helper_resolver.rb +10 -33
  115. data/lib/woods/extractors/scheduled_job_extractor.rb +41 -15
  116. data/lib/woods/extractors/serializer_extractor.rb +4 -2
  117. data/lib/woods/extractors/service_extractor.rb +3 -1
  118. data/lib/woods/extractors/shared_dependency_scanner.rb +2 -2
  119. data/lib/woods/extractors/shared_utility_methods.rb +48 -19
  120. data/lib/woods/extractors/source_nesting.rb +1 -1
  121. data/lib/woods/extractors/state_machine_extractor.rb +3 -1
  122. data/lib/woods/extractors/test_mapping_extractor.rb +3 -1
  123. data/lib/woods/extractors/validator_extractor.rb +3 -1
  124. data/lib/woods/extractors/view_component_extractor.rb +3 -1
  125. data/lib/woods/extractors/view_template_extractor.rb +3 -1
  126. data/lib/woods/gem_mapper.rb +2 -0
  127. data/lib/woods/git_history.rb +116 -0
  128. data/lib/woods/graph_analyzer.rb +35 -6
  129. data/lib/woods/hooks/context_cli.rb +54 -0
  130. data/lib/woods/hooks/context_event.rb +88 -0
  131. data/lib/woods/hooks/context_hint.rb +73 -0
  132. data/lib/woods/hooks/context_impact.rb +77 -0
  133. data/lib/woods/hooks/context_output.rb +47 -0
  134. data/lib/woods/hooks/context_state.rb +102 -0
  135. data/lib/woods/hooks/refresh.rb +79 -0
  136. data/lib/woods/hooks/rule_projection.rb +78 -0
  137. data/lib/woods/input_rules.rb +19 -0
  138. data/lib/woods/mcp/bearer_auth.rb +22 -13
  139. data/lib/woods/mcp/bootstrapper.rb +79 -4
  140. data/lib/woods/mcp/config_resolver.rb +2 -1
  141. data/lib/woods/mcp/index_reader.rb +334 -162
  142. data/lib/woods/mcp/initialization_guidance.rb +27 -0
  143. data/lib/woods/mcp/origin_guard.rb +17 -9
  144. data/lib/woods/mcp/published_lexical_retriever.rb +115 -0
  145. data/lib/woods/mcp/renderers/markdown_renderer.rb +22 -9
  146. data/lib/woods/mcp/renderers/plain_renderer.rb +18 -8
  147. data/lib/woods/mcp/search_results.rb +74 -0
  148. data/lib/woods/mcp/server.rb +178 -63
  149. data/lib/woods/mcp/tool_contract.rb +3 -1
  150. data/lib/woods/mcp/tool_response_renderer.rb +41 -0
  151. data/lib/woods/mcp/traversal_evidence.rb +113 -0
  152. data/lib/woods/mcp/traversal_evidence_index.rb +100 -0
  153. data/lib/woods/mcp/traversal_evidence_page.rb +41 -0
  154. data/lib/woods/mcp/traversal_evidence_text.rb +52 -0
  155. data/lib/woods/mcp/traversal_response.rb +22 -0
  156. data/lib/woods/notion/exporter.rb +56 -17
  157. data/lib/woods/obsidian/destination_plan.rb +98 -0
  158. data/lib/woods/obsidian/name_mapper.rb +19 -3
  159. data/lib/woods/obsidian/note_builder.rb +19 -10
  160. data/lib/woods/obsidian/vault_exporter.rb +88 -32
  161. data/lib/woods/operator/pipeline_guard.rb +18 -13
  162. data/lib/woods/path_dispatcher.rb +13 -6
  163. data/lib/woods/payload_store.rb +27 -26
  164. data/lib/woods/published_index/typed_unit_reader.rb +40 -3
  165. data/lib/woods/published_index.rb +2 -2
  166. data/lib/woods/railtie.rb +3 -3
  167. data/lib/woods/railtie_support.rb +12 -12
  168. data/lib/woods/rake_helpers.rb +382 -0
  169. data/lib/woods/resilience/graph_invariant_validator/membership_checks.rb +71 -0
  170. data/lib/woods/resilience/graph_invariant_validator/node_checks.rb +61 -0
  171. data/lib/woods/resilience/graph_invariant_validator/reverse_relationship_checks.rb +46 -0
  172. data/lib/woods/resilience/graph_invariant_validator.rb +119 -0
  173. data/lib/woods/resilience/index_validator/graph_checks.rb +80 -0
  174. data/lib/woods/resilience/index_validator.rb +112 -23
  175. data/lib/woods/retrieval/context_assembler.rb +50 -15
  176. data/lib/woods/retrieval/lexical_assembler.rb +84 -0
  177. data/lib/woods/retrieval/lexical_index.rb +120 -0
  178. data/lib/woods/retrieval/ranker.rb +4 -2
  179. data/lib/woods/retrieval/scope.rb +108 -0
  180. data/lib/woods/retrieval/scoped_graph_store.rb +32 -0
  181. data/lib/woods/retrieval/scoped_vector_store.rb +55 -0
  182. data/lib/woods/retrieval/search_executor.rb +86 -27
  183. data/lib/woods/retrieval/source_evidence.rb +200 -0
  184. data/lib/woods/retriever.rb +98 -22
  185. data/lib/woods/ruby_analyzer/trace_enricher.rb +77 -38
  186. data/lib/woods/session_tracer/file_store.rb +6 -1
  187. data/lib/woods/session_tracer/middleware.rb +10 -12
  188. data/lib/woods/session_tracer/redis_store.rb +22 -6
  189. data/lib/woods/session_tracer/session_flow_assembler.rb +23 -17
  190. data/lib/woods/session_tracer/solid_cache_coordination.rb +6 -4
  191. data/lib/woods/session_tracer/unit_resolver.rb +63 -0
  192. data/lib/woods/source_inputs/consumer_errors.rb +31 -0
  193. data/lib/woods/source_inputs/handoff.rb +102 -0
  194. data/lib/woods/source_inputs/launcher.rb +157 -0
  195. data/lib/woods/source_inputs/manifest.rb +124 -0
  196. data/lib/woods/source_inputs/private_key.rb +55 -0
  197. data/lib/woods/source_inputs/scanner.rb +171 -0
  198. data/lib/woods/source_inputs/scopes.rb +71 -0
  199. data/lib/woods/source_inputs/session.rb +214 -0
  200. data/lib/woods/source_inputs/status.rb +84 -0
  201. data/lib/woods/source_inputs/verifier.rb +107 -0
  202. data/lib/woods/storage/metadata_store.rb +25 -25
  203. data/lib/woods/storage/pgvector.rb +35 -10
  204. data/lib/woods/storage/qdrant.rb +17 -7
  205. data/lib/woods/storage/vector_store.rb +18 -6
  206. data/lib/woods/tasks.rb +3 -2
  207. data/lib/woods/temporal/json_snapshot_store.rb +58 -9
  208. data/lib/woods/unblocked/exporter.rb +59 -70
  209. data/lib/woods/version.rb +1 -1
  210. data/lib/woods/watch/boot_snapshot.rb +52 -0
  211. data/lib/woods/watch/daemon.rb +154 -32
  212. data/lib/woods/watch/listen_watcher.rb +4 -0
  213. data/lib/woods/watch/polling_watcher.rb +5 -1
  214. data/lib/woods/watch/status.rb +20 -15
  215. data/lib/woods/watch/tree_scan.rb +21 -13
  216. data/lib/woods/watch/watcher.rb +4 -1
  217. data/lib/woods.rb +50 -11
  218. data/plugin/.claude-plugin/plugin.json +1 -1
  219. data/plugin/hooks/adapters/normalize.jq +15 -0
  220. data/plugin/hooks/adapters/normalize.rb +63 -0
  221. data/plugin/hooks/hooks.json +20 -0
  222. data/plugin/hooks/woods-context.sh +50 -0
  223. data/plugin/hooks/woods-input-rules.sh +159 -0
  224. data/plugin/hooks/woods-opencode.mjs +65 -0
  225. data/plugin/hooks/woods-post-edit.sh +2 -225
  226. data/plugin/hooks/woods-refresh.sh +260 -0
  227. data/plugin/hooks/woods-session-start.sh +47 -55
  228. data/plugin/skills/woods-agent-enable/SKILL.md +19 -0
  229. data/plugin/skills/woods-diagnose/SKILL.md +319 -1
  230. data/plugin/skills/woods-investigate/SKILL.md +145 -0
  231. data/plugin/skills/woods-mcp-config/SKILL.md +90 -2
  232. data/plugin/skills/woods-setup/SKILL.md +110 -6
  233. metadata +87 -5
@@ -0,0 +1,84 @@
1
+ # frozen_string_literal: true
2
+
3
+ require 'base64'
4
+ require 'woods/generation'
5
+ require 'woods/source_inputs/verifier'
6
+
7
+ module Woods
8
+ module SourceInputs
9
+ # Shared CLI/hook/MCP reader. A caller serving an older immutable payload
10
+ # passes its own directory and generation, never the newest marker's proof.
11
+ class Status
12
+ BUDGETS = { 'quick' => 0.25, 'deep' => 5.0 }.freeze
13
+
14
+ def self.from_transport(encoded)
15
+ raise ArgumentError, 'source status options are too large' if encoded.to_s.bytesize > 16_384
16
+
17
+ options = encoded.nil? ? {} : JSON.parse(Base64.strict_decode64(encoded))
18
+ unless options.is_a?(Hash) && (options.keys - %w[output root mode]).empty? && options.values.all?(String)
19
+ raise ArgumentError, 'source status options must contain only output, root and mode strings'
20
+ end
21
+
22
+ new(output_dir: options.fetch('output', ENV.fetch('WOODS_OUTPUT', 'tmp/woods')),
23
+ root: options['root'], mode: options.fetch('mode', 'quick')).call
24
+ rescue JSON::ParserError
25
+ raise ArgumentError, 'invalid source status options'
26
+ end
27
+
28
+ def initialize(output_dir:, payload_dir: nil, generation: nil, root: nil, mode: 'quick')
29
+ raise ArgumentError, 'source status mode must be quick or deep' unless BUDGETS.key?(mode)
30
+
31
+ @output = File.expand_path(output_dir.to_s)
32
+ @root = root
33
+ @mode = mode
34
+ @generation = generation
35
+ @payload = payload_dir
36
+ end
37
+
38
+ def call
39
+ return unknown(@resolution_error || 'atomic_source_manifest_unavailable') unless resolve_payload
40
+
41
+ flags = File::RDONLY | File::NONBLOCK
42
+ flags |= File::NOFOLLOW if defined?(File::NOFOLLOW)
43
+ manifest = File.open(File.join(@payload, Manifest::FILE_NAME), flags) do |file|
44
+ return unknown('invalid_source_manifest') unless file.stat.file?
45
+ return unknown('source_manifest_too_large') if file.stat.size > Manifest::MAX_BYTES
46
+
47
+ Manifest.parse(file.read(Manifest::MAX_BYTES + 1))
48
+ end
49
+ Verifier.new(manifest: manifest, output_dir: @output, root: @root,
50
+ generation: @generation, max_seconds: BUDGETS.fetch(@mode)).call.merge('check' => @mode)
51
+ rescue Errno::ENOENT
52
+ unknown('source_manifest_unavailable')
53
+ rescue Manifest::Invalid, SystemCallError, IOError
54
+ unknown('invalid_source_manifest')
55
+ end
56
+
57
+ private
58
+
59
+ def resolve_payload
60
+ generation = Generation.new(output_dir: @output)
61
+ marker = if @payload
62
+ Generation::Marker.new(payload: @payload.to_s)
63
+ else
64
+ generation.current.tap { |current| @generation = current.number }
65
+ end
66
+ return false unless marker.payload.is_a?(String)
67
+
68
+ resolved = generation.payload_dir(marker)
69
+ return false if resolved == generation.root
70
+
71
+ @payload = resolved
72
+ true
73
+ rescue TypeError, NoMethodError
74
+ @resolution_error = 'invalid_generation'
75
+ false
76
+ end
77
+
78
+ def unknown(reason)
79
+ { 'state' => 'unknown', 'mode' => 'content', 'check' => @mode, 'generation' => @generation,
80
+ 'checked_at' => Time.now.utc.iso8601, 'complete' => false, 'reasons' => [reason] }
81
+ end
82
+ end
83
+ end
84
+ end
@@ -0,0 +1,107 @@
1
+ # frozen_string_literal: true
2
+
3
+ require 'time'
4
+ require 'woods/source_inputs/scanner'
5
+ require 'woods/source_inputs/manifest'
6
+
7
+ module Woods
8
+ module SourceInputs
9
+ # Bounded, byte-verifying source evidence. No stat-only result is called
10
+ # current: the same-size/same-mtime case is deliberately read and hashed.
11
+ class Verifier
12
+ DEFAULT_SECONDS = 0.25
13
+ SUMMARY_LIMIT = 30
14
+
15
+ # rubocop:disable Metrics/ParameterLists -- caller chooses identity, served generation and bounded scan
16
+ def initialize(manifest:, output_dir:, root: nil, generation: nil, max_seconds: DEFAULT_SECONDS, **limits)
17
+ @manifest = manifest.is_a?(Manifest) ? manifest : Manifest.new(manifest)
18
+ @output_dir = output_dir
19
+ @root = root || @manifest.data.fetch('root')
20
+ @generation = generation
21
+ @limits = limits.merge(max_seconds: max_seconds)
22
+ end
23
+
24
+ # rubocop:enable Metrics/ParameterLists
25
+
26
+ def call # rubocop:disable Metrics/AbcSize -- cheap refusal checks precede the only source scan
27
+ return unknown('generation_mismatch') if @generation && @generation != @manifest.data['generation']
28
+ return unknown('source_root_unavailable') unless File.directory?(@root)
29
+
30
+ key = PrivateKey.new(output_dir: @output_dir)
31
+ return unknown('identity_key_mismatch') unless key.identifier == @manifest.data.fetch('key_id')
32
+
33
+ scopes = Scopes.new(extra_roots: @manifest.data.fetch('extra_roots'))
34
+ return unknown('input_rules_changed') unless scopes.fingerprint == @manifest.data.fetch('rules')
35
+
36
+ current = Scanner.new(root: @root, output_dir: @output_dir, key: key, scopes: scopes, **@limits).call
37
+ compare(current)
38
+ rescue PrivateKey::Unavailable => e
39
+ unknown(e.message)
40
+ rescue SystemCallError, IOError
41
+ unknown('source_root_unavailable')
42
+ end
43
+
44
+ private
45
+
46
+ def compare(current)
47
+ previous = @manifest.expanded
48
+ changes = { 'added' => [], 'changed' => [], 'removed' => [] }
49
+ previous.each { |scope, paths| compare_scope(scope, paths, changes, current) }
50
+ add_new_scopes(previous, changes, current) if complete?(current)
51
+ reasons = coverage_reasons(current)
52
+ summarized(changes, reasons, current)
53
+ end
54
+
55
+ def complete?(current)
56
+ @manifest.data['complete'] && current['complete']
57
+ end
58
+
59
+ def compare_scope(scope, paths, changes, current)
60
+ paths.each do |path, identity|
61
+ now = current.fetch('files')[path]
62
+ changes['changed'] << path if now && now != identity
63
+ changes['removed'] << path if now.nil? && current['complete']
64
+ end
65
+ return unless complete?(current)
66
+
67
+ changes['added'].concat(current.fetch('scope_paths').fetch(scope, []) - paths.keys)
68
+ end
69
+
70
+ def add_new_scopes(previous, changes, current)
71
+ new_scopes = current.fetch('scope_paths').keys - previous.keys
72
+ new_scopes.each { |scope| changes['added'].concat(current.fetch('scope_paths').fetch(scope)) }
73
+ end
74
+
75
+ def coverage_reasons(current)
76
+ reasons = (@manifest.data.fetch('errors') + current.fetch('errors')).map { |error| error['reason'] }.compact
77
+ reasons << 'incomplete_capture' unless @manifest.data['complete']
78
+ reasons << 'incomplete_verification' unless current['complete']
79
+ reasons << 'unverified_boot_boundary' unless @manifest.data['boot_verified']
80
+ reasons << 'unverified_consumption_scopes' unless @manifest.data.fetch('unverified_scopes').empty?
81
+ reasons.uniq
82
+ end
83
+
84
+ def summarized(changes, reasons, current)
85
+ changes.transform_values! { |paths| paths.uniq.sort }
86
+ counts = changes.transform_values(&:size)
87
+ { 'state' => state_for(counts, reasons), 'mode' => 'content', 'generation' => @manifest.data['generation'],
88
+ 'checked_at' => Time.now.utc.iso8601, 'reasons' => reasons,
89
+ 'complete' => current['complete'] && @manifest.data['complete'],
90
+ 'counts' => counts, 'truncated' => counts.values.any? { |count| count > SUMMARY_LIMIT },
91
+ 'changes' => changes.transform_values { |paths| paths.first(SUMMARY_LIMIT) },
92
+ 'metrics' => current.fetch('metrics') }
93
+ end
94
+
95
+ def state_for(counts, reasons)
96
+ return 'drifted' if counts.values.any?(&:positive?)
97
+
98
+ reasons.empty? ? 'current' : 'unknown'
99
+ end
100
+
101
+ def unknown(reason)
102
+ { 'state' => 'unknown', 'mode' => 'content', 'generation' => @manifest.data['generation'],
103
+ 'checked_at' => Time.now.utc.iso8601, 'reasons' => [reason], 'complete' => false }
104
+ end
105
+ end
106
+ end
107
+ end
@@ -84,16 +84,19 @@ module Woods
84
84
  # adapter — they were substitutable in name only until the semantics
85
85
  # were written down):
86
86
  #
87
- # - **Substring match**, not word or prefix match.
87
+ # - **Literal substring match**, including embedded NUL, not word or prefix match.
88
88
  # - **Case-insensitive.** InMemory used a case-sensitive
89
89
  # `String#include?` while SQLite used `LIKE`, so the same query
90
90
  # returned different results depending on the configured backend.
91
91
  # Case-insensitive is both the search-like expectation and what the
92
- # durable adapter already did. (SQLite's `LIKE` folds ASCII only, so
92
+ # durable adapter already did. (SQLite's `lower` folds ASCII only, so
93
93
  # non-ASCII case folding remains backend-specific — do not rely on
94
94
  # it either way.)
95
95
  # - **`fields: nil` searches the whole record**, including keys, as
96
96
  # serialized JSON. A query can therefore match a field *name*.
97
+ # - **Field-scoped values use JSON spellings.** Strings are unquoted,
98
+ # objects/arrays use JSON text, Booleans use `true`/`false`, and
99
+ # numbers remain numeric text. Null and absent fields never match.
97
100
  # - **LIKE metacharacters are literal.** `%` and `_` in a query match
98
101
  # themselves rather than acting as wildcards.
99
102
  # - Field names are validated against {SEARCH_FIELD_NAME} by every
@@ -218,15 +221,14 @@ module Woods
218
221
  # @see Interface#search
219
222
  #
220
223
  # Matching is literal substring inclusion — `%` and `_` in the query
221
- # have no special meaning here, and the SQLite adapter escapes them
222
- # so the two adapters agree.
224
+ # have no special meaning here or in the SQLite adapter.
223
225
  #
224
226
  # @raise [ArgumentError] if a field name fails {SEARCH_FIELD_NAME}
225
227
  def search(query, fields: nil)
226
228
  fields = validate_search_fields!(fields)
227
229
  return [] if fields == []
228
230
 
229
- # Case-insensitive, matching the SQLite adapter's `LIKE` (see the
231
+ # Case-insensitive, matching the SQLite adapter's `lower` (see the
230
232
  # contract note on {Interface#search}). This used to be a
231
233
  # case-sensitive `String#include?`, so the same query returned
232
234
  # different results depending on which backend a host had configured.
@@ -312,9 +314,9 @@ module Woods
312
314
  JSON.parse(JSON.generate(metadata))
313
315
  end
314
316
 
315
- # The text SQLite's +json_extract(data, '$.field')+ would compare
316
- # against for one field value: strings come back raw, structured
317
- # values as their JSON text, scalars as their decimal form. A Ruby
317
+ # The searchable text for one field value: strings come back raw,
318
+ # structured values as JSON text, Booleans as true/false, and numbers
319
+ # as their decimal form. A Ruby
318
320
  # +Hash#to_s+ haystack used to leak `=>` and `:sym` syntax that no
319
321
  # JSON document contains (STO-8).
320
322
  #
@@ -425,23 +427,22 @@ module Woods
425
427
  # Field names are interpolated into a `json_extract` JSON-path
426
428
  # literal, so they are validated against {SEARCH_FIELD_NAME} first —
427
429
  # a crafted name could otherwise break out of the literal and alter
428
- # the SQL shape. LIKE metacharacters (`%`, `_`, `\`) in the query are
429
- # escaped (with an explicit ESCAPE clause) so they match literally,
430
- # aligning with the InMemory adapter's substring semantics instead of
431
- # silently broadening matches.
430
+ # the SQL shape. instr and lower perform literal, ASCII-case-insensitive
431
+ # substring matching without LIKE's truncation at embedded NUL bytes.
432
+ # Wildcard characters need no escaping.
432
433
  #
433
434
  # @raise [ArgumentError] if a field name fails the whitelist
434
435
  def search(query, fields: nil)
435
436
  fields = validate_search_fields!(fields)
436
437
  return [] if fields == []
437
438
 
438
- pattern = "%#{escape_like(query)}%"
439
+ needle = query.to_s
439
440
  if fields
440
- conditions = fields.map { "json_extract(data, '$.#{_1}') LIKE ? ESCAPE '\\'" }.join(' OR ')
441
- params = Array.new(fields.size, pattern)
441
+ conditions = fields.map { "instr(lower(#{field_haystack_sql(_1)}), lower(?)) > 0" }.join(' OR ')
442
+ params = Array.new(fields.size, needle)
442
443
  rows = @db.execute("SELECT id, data FROM units WHERE #{conditions}", params)
443
444
  else
444
- rows = @db.execute("SELECT id, data FROM units WHERE data LIKE ? ESCAPE '\\'", [pattern])
445
+ rows = @db.execute('SELECT id, data FROM units WHERE instr(lower(data), lower(?)) > 0', [needle])
445
446
  end
446
447
 
447
448
  rows.map { |row| parse_row(row) }
@@ -489,15 +490,14 @@ module Woods
489
490
  defined?(SQLite3::BusyException) && error.is_a?(SQLite3::BusyException)
490
491
  end
491
492
 
492
- # Escape SQL LIKE metacharacters in a user query so they match
493
- # literally under the `ESCAPE '\'` clause {#search} emits. Without
494
- # this, `%` and `_` in a query act as wildcards and silently broaden
495
- # matches (`"user_name"` would match `"userXname"`).
496
- #
497
- # @param query [String] Raw search query
498
- # @return [String] Query safe for embedding in a LIKE pattern
499
- def escape_like(query)
500
- query.to_s.gsub(/[\\%_]/) { |ch| "\\#{ch}" }
493
+ # JSON1 extracts Boolean scalars as integers; use their JSON type to
494
+ # preserve true/false without conflating them with numeric 1/0.
495
+ # Field names have already passed validate_search_fields!.
496
+ def field_haystack_sql(field)
497
+ path = "$.#{field}"
498
+ "CASE json_type(data, '#{path}') " \
499
+ "WHEN 'true' THEN 'true' WHEN 'false' THEN 'false' " \
500
+ "ELSE json_extract(data, '#{path}') END"
501
501
  end
502
502
 
503
503
  # Parse a database row into a metadata hash with the id field injected.
@@ -26,6 +26,7 @@ module Woods
26
26
  class Pgvector # rubocop:disable Metrics/ClassLength
27
27
  include Interface
28
28
 
29
+ MAX_HNSW_DIMENSIONS = 2000
29
30
  TABLE = 'woods_vectors'
30
31
  TABLE_NAME_PATTERN = /\A[a-z_][a-z0-9_]*\z/
31
32
 
@@ -122,6 +123,9 @@ module Woods
122
123
  SQL
123
124
  end
124
125
 
126
+ # Native raw-ID eligibility is applied before ranking and the limit.
127
+ def supports_id_filter? = true
128
+
125
129
  # Search for similar vectors using cosine distance.
126
130
  #
127
131
  # The query vector is dimension-checked before SQL runs, mirroring the
@@ -136,19 +140,14 @@ module Woods
136
140
  # @raise [Woods::Error] if the query vector's length disagrees with the
137
141
  # configured dimension
138
142
  # @see Interface#search
139
- def search(query_vector, limit: 10, filters: {})
143
+ def search(query_vector, limit: 10, filters: {}, ids: nil)
140
144
  validate_vector!(query_vector)
141
145
  validate_dimensions!(query_vector) if @dimensions
142
146
  vector_literal = build_vector_literal(query_vector)
143
147
  where_clause = build_where(filters)
148
+ where_clause = append_id_filter(where_clause, ids) if ids
144
149
 
145
- sql = <<~SQL
146
- SELECT id, embedding <=> '#{vector_literal}' AS distance, metadata
147
- FROM #{qualified_table}
148
- #{where_clause}
149
- ORDER BY distance ASC
150
- LIMIT #{limit.to_i}
151
- SQL
150
+ sql = search_sql(vector_literal, where_clause, limit, ids)
152
151
 
153
152
  rows = @connection.execute(sql)
154
153
  rows.map { |row| row_to_result(row) }
@@ -221,9 +220,12 @@ module Woods
221
220
  private
222
221
 
223
222
  def normalize_dimensions(value)
224
- return value if value.is_a?(Integer) && value.positive?
223
+ raise ArgumentError, 'dimensions must be a positive Integer' unless value.is_a?(Integer) && value.positive?
224
+ return value if value <= MAX_HNSW_DIMENSIONS
225
225
 
226
- raise ArgumentError, 'dimensions must be a positive Integer'
226
+ raise ArgumentError,
227
+ "pgvector HNSW vector dimensions must be at most #{MAX_HNSW_DIMENSIONS}, got #{value}. " \
228
+ 'Request a supported provider output width or choose another vector backend.'
227
229
  end
228
230
 
229
231
  def validate_identifier!(name, value, original)
@@ -345,6 +347,29 @@ module Woods
345
347
  )
346
348
  end
347
349
 
350
+ def search_sql(vector_literal, where_clause, limit, ids)
351
+ source = ids ? 'eligible' : qualified_table
352
+ prefix = if ids
353
+ 'WITH eligible AS MATERIALIZED (SELECT id, embedding, metadata ' \
354
+ "FROM #{qualified_table} #{where_clause})\n"
355
+ else
356
+ ''
357
+ end
358
+ prefix + <<~SQL
359
+ SELECT id, embedding <=> '#{vector_literal}' AS distance, metadata
360
+ FROM #{source}
361
+ #{where_clause unless ids}
362
+ ORDER BY distance ASC#{', id ASC' if ids}
363
+ LIMIT #{limit.to_i}
364
+ SQL
365
+ end
366
+
367
+ # Raw IDs work with old vectors that have no identifier/package payload.
368
+ def append_id_filter(where_clause, ids)
369
+ condition = ids.empty? ? 'FALSE' : "id IN (#{ids.map { |id| @connection.quote(id.to_s) }.join(', ')})"
370
+ where_clause.empty? ? "WHERE #{condition}" : "#{where_clause} AND #{condition}"
371
+ end
372
+
348
373
  # Build a WHERE clause from metadata filters.
349
374
  #
350
375
  # @param filters [Hash] Metadata key-value pairs
@@ -332,6 +332,9 @@ module Woods
332
332
  request(:put, "/collections/#{@collection}/points#{WAIT_FOR_WRITE}", body)
333
333
  end
334
334
 
335
+ # Native raw-ID eligibility is applied before ranking and the limit.
336
+ def supports_id_filter? = true
337
+
335
338
  # Search for similar vectors.
336
339
  #
337
340
  # The query vector is dimension-checked before the request, mirroring
@@ -348,14 +351,11 @@ module Woods
348
351
  # @raise [Woods::Error] if the query vector's length disagrees with the
349
352
  # configured dimension
350
353
  # @see Interface#search
351
- def search(query_vector, limit: 10, filters: {})
354
+ def search(query_vector, limit: 10, filters: {}, ids: nil)
355
+ return [] if ids == []
356
+
352
357
  validate_dimensions!(query_vector) if @dimensions
353
- body = {
354
- vector: query_vector,
355
- limit: limit,
356
- with_payload: true
357
- }
358
- body[:filter] = build_filter(filters) unless filters.empty?
358
+ body = search_body(query_vector, limit, filters, ids)
359
359
 
360
360
  response = request(:post, "/collections/#{@collection}/points/search", body)
361
361
  results = response['result'] || []
@@ -579,6 +579,16 @@ module Woods
579
579
  "Vector dimension mismatch#{where}: got #{got}, expected #{@dimensions}"
580
580
  end
581
581
 
582
+ def search_body(query_vector, limit, filters, ids)
583
+ body = { vector: query_vector, limit: limit, with_payload: true }
584
+ body[:filter] = build_filter(filters) unless filters.empty? && ids.nil?
585
+ if ids
586
+ body[:filter][:must] << { key: IDENTIFIER_KEY, match: { any: ids } }
587
+ body[:params] = { exact: true }
588
+ end
589
+ body
590
+ end
591
+
582
592
  # Build a Qdrant filter from metadata key-value pairs.
583
593
  #
584
594
  # @param filters [Hash] Metadata filters
@@ -90,12 +90,17 @@ module Woods
90
90
  # @param limit [Integer] Maximum number of results to return
91
91
  # @param filters [Hash] Optional metadata filters — values may be
92
92
  # scalars or Arrays
93
+ # @param ids [Array<String>, nil] Raw vector IDs eligible before ranking;
94
+ # optional capability advertised by #supports_id_filter?. Empty matches none.
93
95
  # @return [Array<SearchResult>] Results sorted by descending similarity
94
96
  # @raise [NotImplementedError] if not implemented by adapter
95
- def search(query_vector, limit: 10, filters: {})
97
+ def search(query_vector, limit: 10, filters: {}, ids: nil)
96
98
  raise NotImplementedError
97
99
  end
98
100
 
101
+ # Whether native raw-ID eligibility is supported before the result limit.
102
+ def supports_id_filter? = false
103
+
99
104
  # Delete a vector by ID.
100
105
  #
101
106
  # @param id [String] The identifier to delete
@@ -220,8 +225,10 @@ module Woods
220
225
  end
221
226
  end
222
227
 
228
+ def supports_id_filter? = true
229
+
223
230
  # @see Interface#search
224
- def search(query_vector, limit: 10, filters: {})
231
+ def search(query_vector, limit: 10, filters: {}, ids: nil)
225
232
  return [] if @dim.nil?
226
233
 
227
234
  unless query_vector.length == @dim
@@ -229,8 +236,8 @@ module Woods
229
236
  "Vector dimension mismatch (#{query_vector.length} vs #{@dim})"
230
237
  end
231
238
 
232
- scored = gather_candidates(query_vector, filters)
233
- scored.sort_by! { |r| -r.score }
239
+ scored = gather_candidates(query_vector, filters, ids)
240
+ scored.sort_by! { |r| ids ? [-r.score, r.id] : [-r.score] }
234
241
  scored.first(limit)
235
242
  end
236
243
 
@@ -300,15 +307,20 @@ module Woods
300
307
  @metadata[idx] = metadata
301
308
  end
302
309
 
310
+ def excluded_index?(idx, allowed_ids)
311
+ @tombstones.include?(idx) || (allowed_ids && !allowed_ids.include?(@ids[idx]))
312
+ end
313
+
303
314
  # Walk every non-tombstoned index, apply filters, score survivors.
304
315
  # Filter check runs BEFORE the cosine kernel — avoids computing
305
316
  # 12k dot products only to discard most of them.
306
- def gather_candidates(query_vector, filters)
317
+ def gather_candidates(query_vector, filters, ids)
307
318
  scored = []
319
+ allowed_ids = ids&.to_set
308
320
  len = @ids.size
309
321
  idx = 0
310
322
  while idx < len
311
- if @tombstones.include?(idx)
323
+ if excluded_index?(idx, allowed_ids)
312
324
  idx += 1
313
325
  next
314
326
  end
data/lib/woods/tasks.rb CHANGED
@@ -27,6 +27,7 @@ module Woods
27
27
  # @return [Embedding::Indexer]
28
28
  def build_embed_indexer
29
29
  config = Woods.configuration
30
+ output_dir = ENV.fetch('WOODS_OUTPUT', config.output_dir)
30
31
  builder = Builder.new(config)
31
32
  provider = builder.build_embedding_provider
32
33
 
@@ -56,11 +57,11 @@ module Woods
56
57
  provider: resilient_provider,
57
58
  text_preparer: builder.build_text_preparer(provider),
58
59
  vector_store: vector_store,
59
- metadata_store: config.metadata_store ? builder.build_metadata_store : nil,
60
+ metadata_store: config.metadata_store ? builder.build_metadata_store(output_dir: output_dir) : nil,
60
61
  resolved_config: build_resolved_config(config, provider: provider),
61
62
  chunker: builder.build_chunker(provider),
62
63
  dump_retention_count: config.dump_retention_count,
63
- output_dir: ENV.fetch('WOODS_OUTPUT', config.output_dir)
64
+ output_dir: output_dir
64
65
  )
65
66
  end
66
67
 
@@ -22,7 +22,8 @@ module Woods
22
22
  #
23
23
  # Implements the same public interface as SnapshotStore so the MCP server
24
24
  # tools work identically.
25
- # Malformed retained snapshots are warned about and treated as absent:
25
+ # Malformed, invalidly shaped, or unreadable retained snapshots (including files
26
+ # removed during retention) are warned about and treated as absent:
26
27
  # +find+ returns nil, +diff+ returns an empty result, and history/list scans
27
28
  # omit the corrupt file.
28
29
  #
@@ -94,7 +95,6 @@ module Woods
94
95
  snapshots = load_all_with_units
95
96
  .sort_by { |s| s[:extracted_at] || '' }
96
97
  .reverse
97
- .first(limit)
98
98
 
99
99
  entries = snapshots.flat_map do |snap|
100
100
  snap[:units].values.select { |unit| unit[:identifier] == identifier }.map do |unit|
@@ -122,9 +122,9 @@ module Woods
122
122
  value.positive? ? value : PayloadStore::DEFAULT_RETENTION
123
123
  end
124
124
 
125
- # Delete snapshots beyond the retention count, oldest by extracted_at
126
- # first. +protect+ names the snapshot just captured; it is the newest
127
- # anyway, and a tie on extracted_at must not delete it.
125
+ # Delete snapshots beyond the retention count, corrupt/unreadable files
126
+ # first, then oldest by extracted_at. +protect+ names the snapshot just
127
+ # captured; even an older timestamp must not cause its deletion.
128
128
  #
129
129
  # Failures are non-fatal: retention is housekeeping, and a failed
130
130
  # prune leaves the store growing as it did before rather than
@@ -133,7 +133,7 @@ module Woods
133
133
  # @param protect [String] git SHA of the just-captured snapshot
134
134
  # @return [void]
135
135
  def prune_snapshots(protect:)
136
- summaries = load_all_summaries
136
+ summaries = retention_summaries
137
137
  overflow = summaries.size - @retention
138
138
  return unless overflow.positive?
139
139
 
@@ -142,7 +142,19 @@ module Woods
142
142
  warn "[Woods] Snapshot retention failed: #{e.message}"
143
143
  end
144
144
 
145
- # Oldest `overflow` snapshots by extracted_at, never the protected one.
145
+ # Include unreadable snapshots in the bound. Only files named like a
146
+ # snapshot are eligible; the filename owns identity, not JSON content.
147
+ def retention_summaries
148
+ Dir.glob(File.join(@dir, '*.json')).filter_map do |path|
149
+ sha = File.basename(path, '.json')
150
+ next unless sha.match?(/\A[0-9a-f]+\z/i) && File.file?(path)
151
+
152
+ data = read_snapshot(path)
153
+ { git_sha: sha, extracted_at: data&.[]('extracted_at') || '', corrupt: data.nil? }
154
+ end
155
+ end
156
+
157
+ # Corrupt files, then oldest `overflow` snapshots, never the protected one.
146
158
  # The protected SHA is rejected before sorting and slicing: it is
147
159
  # captured last, so its extracted_at can tie or precede older entries,
148
160
  # and letting it occupy the victim slice would leave the store one file
@@ -154,7 +166,7 @@ module Woods
154
166
  # @return [Array<String>]
155
167
  def retention_victims(summaries, overflow, protect)
156
168
  summaries.reject { |summary| summary[:git_sha] == protect }
157
- .sort_by { |summary| summary[:extracted_at] || '' }
169
+ .sort_by { |summary| [summary[:corrupt] ? 0 : 1, summary[:extracted_at], summary[:git_sha]] }
158
170
  .first(overflow)
159
171
  .map { |summary| summary[:git_sha] }
160
172
  end
@@ -264,10 +276,47 @@ module Woods
264
276
  end
265
277
 
266
278
  def read_snapshot(path)
267
- JSON.parse(AtomicFile.read(path))
279
+ data = JSON.parse(AtomicFile.read(path))
280
+ validate_snapshot_shape!(data)
281
+ unless data['git_sha'] == File.basename(path, '.json')
282
+ raise JSON::ParserError, 'git_sha does not match the snapshot filename'
283
+ end
284
+
285
+ data
268
286
  rescue JSON::ParserError => e
269
287
  warn "[Woods] Skipping corrupt snapshot #{File.basename(path)}: #{e.message}"
270
288
  nil
289
+ rescue SystemCallError => e
290
+ warn "[Woods] Skipping unreadable snapshot #{File.basename(path)}: #{e.message}"
291
+ nil
292
+ end
293
+
294
+ # Guard only shapes consumed by conversion, sorting, and the next capture.
295
+ # Legacy files can omit units/timestamps and optional per-unit hashes;
296
+ # bare unit keys still supply identifiers when the record does not.
297
+ def validate_snapshot_shape!(data)
298
+ raise JSON::ParserError, 'expected a JSON object' unless data.is_a?(Hash)
299
+
300
+ sha = data['git_sha']
301
+ unless sha.is_a?(String) && sha.match?(/\A[0-9a-f]+\z/i)
302
+ raise JSON::ParserError, 'expected git_sha to be a hexadecimal string'
303
+ end
304
+
305
+ timestamp = data['extracted_at']
306
+ unless timestamp.nil? || timestamp.is_a?(String)
307
+ raise JSON::ParserError, 'expected extracted_at to be a string or null'
308
+ end
309
+
310
+ validate_snapshot_units!(data['units'])
311
+ end
312
+
313
+ def validate_snapshot_units!(units)
314
+ return if units.nil?
315
+
316
+ raise JSON::ParserError, 'expected units to be an object or null' unless units.is_a?(Hash)
317
+ return if units.each_value.all?(Hash)
318
+
319
+ raise JSON::ParserError, 'expected every unit record to be an object'
271
320
  end
272
321
 
273
322
  # @param exclude_sha [String, nil] SHA to leave out of the result