woods 2.0.0.beta1 → 2.0.0.beta3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (221) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +400 -1
  3. data/CONTRIBUTING.md +224 -9
  4. data/README.md +7 -3
  5. data/SECURITY.md +9 -6
  6. data/docs/AGENT_GUIDE.md +83 -4
  7. data/docs/AGENT_SETUP.md +82 -1
  8. data/docs/BACKEND_MATRIX.md +20 -0
  9. data/docs/CLIENT_HOOKS.md +111 -0
  10. data/docs/CONFIGURATION_REFERENCE.md +233 -13
  11. data/docs/CONSOLE_MCP_SETUP.md +35 -5
  12. data/docs/DOCKER_SETUP.md +21 -2
  13. data/docs/EVALUATION.md +464 -1
  14. data/docs/EXTRACTOR_REFERENCE.md +36 -5
  15. data/docs/FAQ.md +11 -12
  16. data/docs/GETTING_STARTED.md +17 -5
  17. data/docs/INCREMENTAL_EXTRACTION.md +158 -2
  18. data/docs/INDEX_LAYOUT.md +382 -0
  19. data/docs/INTERNALS.md +15 -7
  20. data/docs/MCP_SERVERS.md +221 -5
  21. data/docs/MCP_TOOL_COOKBOOK.md +33 -18
  22. data/docs/NOTION_INTEGRATION.md +13 -0
  23. data/docs/OBSIDIAN_INTEGRATION.md +57 -9
  24. data/docs/PUBLISHED_INDEX.md +71 -0
  25. data/docs/README.md +7 -0
  26. data/docs/RETRIEVAL_GUIDE.md +253 -11
  27. data/docs/RUNTIME_TRACING.md +71 -0
  28. data/docs/SOURCE_FRESHNESS.md +143 -0
  29. data/docs/TROUBLESHOOTING.md +117 -5
  30. data/docs/UNBLOCKED_INTEGRATION.md +25 -0
  31. data/docs/UPGRADING_TO_2.md +44 -22
  32. data/docs/WATCH_DAEMON.md +259 -59
  33. data/exe/woods-agent-config +6 -0
  34. data/exe/woods-extract +5 -0
  35. data/exe/woods-hook-context +6 -0
  36. data/lib/generators/woods/templates/woods.rb.tt +1 -3
  37. data/lib/tasks/woods.rake +47 -397
  38. data/lib/woods/agent_configuration/applier.rb +133 -0
  39. data/lib/woods/agent_configuration/cli.rb +101 -0
  40. data/lib/woods/agent_configuration/cli_options.rb +29 -0
  41. data/lib/woods/agent_configuration/document.rb +105 -0
  42. data/lib/woods/agent_configuration/error.rb +7 -0
  43. data/lib/woods/agent_configuration/launcher.rb +75 -0
  44. data/lib/woods/agent_configuration/layout.rb +59 -0
  45. data/lib/woods/agent_configuration/managed_section.rb +62 -0
  46. data/lib/woods/agent_configuration/plan.rb +98 -0
  47. data/lib/woods/agent_configuration/plan_diff.rb +38 -0
  48. data/lib/woods/agent_configuration/planned_files.rb +61 -0
  49. data/lib/woods/agent_configuration/planner.rb +63 -0
  50. data/lib/woods/agent_configuration/planner_validation.rb +77 -0
  51. data/lib/woods/agent_configuration/preflight.rb +100 -0
  52. data/lib/woods/agent_configuration/recovery.rb +49 -0
  53. data/lib/woods/ast/node.rb +2 -0
  54. data/lib/woods/ast/parser.rb +38 -5
  55. data/lib/woods/atomic_file.rb +133 -3
  56. data/lib/woods/builder.rb +21 -5
  57. data/lib/woods/cache/cache_middleware.rb +28 -7
  58. data/lib/woods/cache/cache_store.rb +4 -5
  59. data/lib/woods/change_set.rb +5 -4
  60. data/lib/woods/console/credential_index.rb +20 -2
  61. data/lib/woods/console/credential_scanner.rb +14 -14
  62. data/lib/woods/console/credential_scanner_registry.rb +36 -0
  63. data/lib/woods/console/embedded_executor.rb +1 -1
  64. data/lib/woods/console/encrypted_credential_snapshot.rb +16 -0
  65. data/lib/woods/console/rack_middleware.rb +22 -13
  66. data/lib/woods/console/server.rb +18 -16
  67. data/lib/woods/dependency_graph.rb +65 -13
  68. data/lib/woods/embedding/corpus.rb +94 -0
  69. data/lib/woods/embedding/indexer.rb +90 -46
  70. data/lib/woods/embedding/openai.rb +17 -6
  71. data/lib/woods/evaluation/ablation_executor.rb +6 -1
  72. data/lib/woods/evaluation/ablation_timed_executor.rb +22 -4
  73. data/lib/woods/export/typed_reader.rb +56 -0
  74. data/lib/woods/extractor.rb +557 -228
  75. data/lib/woods/extractors/action_cable_extractor.rb +3 -1
  76. data/lib/woods/extractors/behavioral_profile.rb +9 -7
  77. data/lib/woods/extractors/caching_extractor.rb +3 -1
  78. data/lib/woods/extractors/concern_extractor.rb +64 -6
  79. data/lib/woods/extractors/configuration_extractor.rb +7 -3
  80. data/lib/woods/extractors/controller_extractor.rb +13 -4
  81. data/lib/woods/extractors/database_view_extractor.rb +3 -1
  82. data/lib/woods/extractors/decorator_extractor.rb +3 -1
  83. data/lib/woods/extractors/engine_extractor.rb +3 -1
  84. data/lib/woods/extractors/event_extractor.rb +4 -2
  85. data/lib/woods/extractors/factory_extractor.rb +3 -1
  86. data/lib/woods/extractors/graphql_extractor.rb +8 -2
  87. data/lib/woods/extractors/i18n_extractor.rb +3 -1
  88. data/lib/woods/extractors/job_extractor.rb +6 -19
  89. data/lib/woods/extractors/lib_extractor.rb +3 -1
  90. data/lib/woods/extractors/mailer_extractor.rb +20 -5
  91. data/lib/woods/extractors/manager_extractor.rb +3 -1
  92. data/lib/woods/extractors/method_parameters.rb +53 -0
  93. data/lib/woods/extractors/middleware_argument.rb +65 -0
  94. data/lib/woods/extractors/middleware_extractor.rb +9 -3
  95. data/lib/woods/extractors/migration_extractor.rb +3 -1
  96. data/lib/woods/extractors/model_extractor.rb +39 -33
  97. data/lib/woods/extractors/package_extractor.rb +24 -4
  98. data/lib/woods/extractors/phlex_extractor.rb +3 -1
  99. data/lib/woods/extractors/policy_extractor.rb +3 -1
  100. data/lib/woods/extractors/poro_extractor.rb +3 -1
  101. data/lib/woods/extractors/pundit_extractor.rb +3 -1
  102. data/lib/woods/extractors/rails_source_extractor.rb +4 -2
  103. data/lib/woods/extractors/rake_task_extractor.rb +4 -2
  104. data/lib/woods/extractors/route_extractor.rb +3 -1
  105. data/lib/woods/extractors/route_helper_resolver.rb +10 -33
  106. data/lib/woods/extractors/scheduled_job_extractor.rb +41 -15
  107. data/lib/woods/extractors/serializer_extractor.rb +4 -2
  108. data/lib/woods/extractors/service_extractor.rb +3 -1
  109. data/lib/woods/extractors/shared_dependency_scanner.rb +2 -2
  110. data/lib/woods/extractors/shared_utility_methods.rb +27 -15
  111. data/lib/woods/extractors/source_nesting.rb +1 -1
  112. data/lib/woods/extractors/state_machine_extractor.rb +3 -1
  113. data/lib/woods/extractors/test_mapping_extractor.rb +3 -1
  114. data/lib/woods/extractors/validator_extractor.rb +3 -1
  115. data/lib/woods/extractors/view_component_extractor.rb +3 -1
  116. data/lib/woods/extractors/view_template_extractor.rb +3 -1
  117. data/lib/woods/flow_assembler.rb +87 -8
  118. data/lib/woods/flow_precomputer.rb +44 -7
  119. data/lib/woods/gem_mapper.rb +2 -0
  120. data/lib/woods/git_history.rb +116 -0
  121. data/lib/woods/graph_analyzer.rb +195 -63
  122. data/lib/woods/hooks/context_cli.rb +54 -0
  123. data/lib/woods/hooks/context_event.rb +88 -0
  124. data/lib/woods/hooks/context_hint.rb +73 -0
  125. data/lib/woods/hooks/context_impact.rb +77 -0
  126. data/lib/woods/hooks/context_output.rb +47 -0
  127. data/lib/woods/hooks/context_state.rb +102 -0
  128. data/lib/woods/hooks/refresh.rb +79 -0
  129. data/lib/woods/hooks/rule_projection.rb +78 -0
  130. data/lib/woods/input_rules.rb +19 -0
  131. data/lib/woods/mcp/bearer_auth.rb +20 -12
  132. data/lib/woods/mcp/bootstrapper.rb +62 -0
  133. data/lib/woods/mcp/index_reader.rb +323 -160
  134. data/lib/woods/mcp/initialization_guidance.rb +27 -0
  135. data/lib/woods/mcp/origin_guard.rb +17 -9
  136. data/lib/woods/mcp/published_lexical_retriever.rb +115 -0
  137. data/lib/woods/mcp/renderers/markdown_renderer.rb +8 -1
  138. data/lib/woods/mcp/renderers/plain_renderer.rb +7 -1
  139. data/lib/woods/mcp/search_results.rb +74 -0
  140. data/lib/woods/mcp/server.rb +158 -37
  141. data/lib/woods/mcp/tool_contract.rb +2 -0
  142. data/lib/woods/mcp/tool_response_renderer.rb +25 -0
  143. data/lib/woods/mcp/traversal_evidence.rb +113 -0
  144. data/lib/woods/mcp/traversal_evidence_index.rb +100 -0
  145. data/lib/woods/mcp/traversal_evidence_page.rb +41 -0
  146. data/lib/woods/mcp/traversal_evidence_text.rb +52 -0
  147. data/lib/woods/notion/exporter.rb +56 -17
  148. data/lib/woods/obsidian/destination_plan.rb +98 -0
  149. data/lib/woods/obsidian/name_mapper.rb +19 -3
  150. data/lib/woods/obsidian/note_builder.rb +19 -10
  151. data/lib/woods/obsidian/vault_exporter.rb +88 -32
  152. data/lib/woods/operator/pipeline_guard.rb +18 -13
  153. data/lib/woods/path_dispatcher.rb +7 -1
  154. data/lib/woods/payload_store.rb +29 -15
  155. data/lib/woods/railtie.rb +3 -3
  156. data/lib/woods/railtie_support.rb +12 -12
  157. data/lib/woods/rake_helpers.rb +392 -0
  158. data/lib/woods/resilience/graph_invariant_validator/membership_checks.rb +71 -0
  159. data/lib/woods/resilience/graph_invariant_validator/node_checks.rb +61 -0
  160. data/lib/woods/resilience/graph_invariant_validator/reverse_relationship_checks.rb +46 -0
  161. data/lib/woods/resilience/graph_invariant_validator.rb +119 -0
  162. data/lib/woods/resilience/index_validator/graph_checks.rb +80 -0
  163. data/lib/woods/resilience/index_validator.rb +112 -23
  164. data/lib/woods/retrieval/context_assembler.rb +50 -15
  165. data/lib/woods/retrieval/lexical_assembler.rb +73 -0
  166. data/lib/woods/retrieval/lexical_index.rb +119 -0
  167. data/lib/woods/retrieval/ranker.rb +4 -2
  168. data/lib/woods/retrieval/scope.rb +108 -0
  169. data/lib/woods/retrieval/scoped_graph_store.rb +32 -0
  170. data/lib/woods/retrieval/scoped_vector_store.rb +55 -0
  171. data/lib/woods/retrieval/search_executor.rb +86 -27
  172. data/lib/woods/retrieval/source_evidence.rb +200 -0
  173. data/lib/woods/retriever.rb +98 -22
  174. data/lib/woods/ruby_analyzer/trace_enricher.rb +80 -38
  175. data/lib/woods/session_tracer/middleware.rb +10 -12
  176. data/lib/woods/session_tracer/redis_store.rb +22 -6
  177. data/lib/woods/session_tracer/session_flow_assembler.rb +23 -17
  178. data/lib/woods/session_tracer/solid_cache_coordination.rb +6 -4
  179. data/lib/woods/session_tracer/unit_resolver.rb +63 -0
  180. data/lib/woods/source_inputs/consumer_errors.rb +27 -0
  181. data/lib/woods/source_inputs/handoff.rb +102 -0
  182. data/lib/woods/source_inputs/launcher.rb +157 -0
  183. data/lib/woods/source_inputs/manifest.rb +124 -0
  184. data/lib/woods/source_inputs/private_key.rb +55 -0
  185. data/lib/woods/source_inputs/scanner.rb +171 -0
  186. data/lib/woods/source_inputs/scopes.rb +71 -0
  187. data/lib/woods/source_inputs/session.rb +214 -0
  188. data/lib/woods/source_inputs/status.rb +84 -0
  189. data/lib/woods/source_inputs/verifier.rb +107 -0
  190. data/lib/woods/storage/metadata_store.rb +25 -25
  191. data/lib/woods/storage/pgvector.rb +29 -8
  192. data/lib/woods/storage/qdrant.rb +17 -7
  193. data/lib/woods/storage/vector_store.rb +18 -6
  194. data/lib/woods/tasks.rb +3 -2
  195. data/lib/woods/temporal/json_snapshot_store.rb +29 -8
  196. data/lib/woods/unblocked/exporter.rb +59 -70
  197. data/lib/woods/version.rb +1 -1
  198. data/lib/woods/watch/boot_snapshot.rb +52 -0
  199. data/lib/woods/watch/daemon.rb +136 -28
  200. data/lib/woods/watch/listen_watcher.rb +4 -0
  201. data/lib/woods/watch/polling_watcher.rb +5 -1
  202. data/lib/woods/watch/status.rb +20 -15
  203. data/lib/woods/watch/tree_scan.rb +21 -13
  204. data/lib/woods/watch/watcher.rb +4 -1
  205. data/lib/woods.rb +135 -11
  206. data/plugin/.claude-plugin/plugin.json +1 -1
  207. data/plugin/hooks/adapters/normalize.jq +15 -0
  208. data/plugin/hooks/adapters/normalize.rb +63 -0
  209. data/plugin/hooks/hooks.json +20 -0
  210. data/plugin/hooks/woods-context.sh +50 -0
  211. data/plugin/hooks/woods-input-rules.sh +159 -0
  212. data/plugin/hooks/woods-opencode.mjs +65 -0
  213. data/plugin/hooks/woods-post-edit.sh +2 -225
  214. data/plugin/hooks/woods-refresh.sh +260 -0
  215. data/plugin/hooks/woods-session-start.sh +47 -55
  216. data/plugin/skills/woods-agent-enable/SKILL.md +13 -0
  217. data/plugin/skills/woods-diagnose/SKILL.md +288 -1
  218. data/plugin/skills/woods-investigate/SKILL.md +106 -0
  219. data/plugin/skills/woods-mcp-config/SKILL.md +89 -1
  220. data/plugin/skills/woods-setup/SKILL.md +107 -6
  221. metadata +84 -5
@@ -84,16 +84,19 @@ module Woods
84
84
  # adapter — they were substitutable in name only until the semantics
85
85
  # were written down):
86
86
  #
87
- # - **Substring match**, not word or prefix match.
87
+ # - **Literal substring match**, including embedded NUL, not word or prefix match.
88
88
  # - **Case-insensitive.** InMemory used a case-sensitive
89
89
  # `String#include?` while SQLite used `LIKE`, so the same query
90
90
  # returned different results depending on the configured backend.
91
91
  # Case-insensitive is both the search-like expectation and what the
92
- # durable adapter already did. (SQLite's `LIKE` folds ASCII only, so
92
+ # durable adapter already did. (SQLite's `lower` folds ASCII only, so
93
93
  # non-ASCII case folding remains backend-specific — do not rely on
94
94
  # it either way.)
95
95
  # - **`fields: nil` searches the whole record**, including keys, as
96
96
  # serialized JSON. A query can therefore match a field *name*.
97
+ # - **Field-scoped values use JSON spellings.** Strings are unquoted,
98
+ # objects/arrays use JSON text, Booleans use `true`/`false`, and
99
+ # numbers remain numeric text. Null and absent fields never match.
97
100
  # - **LIKE metacharacters are literal.** `%` and `_` in a query match
98
101
  # themselves rather than acting as wildcards.
99
102
  # - Field names are validated against {SEARCH_FIELD_NAME} by every
@@ -218,15 +221,14 @@ module Woods
218
221
  # @see Interface#search
219
222
  #
220
223
  # Matching is literal substring inclusion — `%` and `_` in the query
221
- # have no special meaning here, and the SQLite adapter escapes them
222
- # so the two adapters agree.
224
+ # have no special meaning here or in the SQLite adapter.
223
225
  #
224
226
  # @raise [ArgumentError] if a field name fails {SEARCH_FIELD_NAME}
225
227
  def search(query, fields: nil)
226
228
  fields = validate_search_fields!(fields)
227
229
  return [] if fields == []
228
230
 
229
- # Case-insensitive, matching the SQLite adapter's `LIKE` (see the
231
+ # Case-insensitive, matching the SQLite adapter's `lower` (see the
230
232
  # contract note on {Interface#search}). This used to be a
231
233
  # case-sensitive `String#include?`, so the same query returned
232
234
  # different results depending on which backend a host had configured.
@@ -312,9 +314,9 @@ module Woods
312
314
  JSON.parse(JSON.generate(metadata))
313
315
  end
314
316
 
315
- # The text SQLite's +json_extract(data, '$.field')+ would compare
316
- # against for one field value: strings come back raw, structured
317
- # values as their JSON text, scalars as their decimal form. A Ruby
317
+ # The searchable text for one field value: strings come back raw,
318
+ # structured values as JSON text, Booleans as true/false, and numbers
319
+ # as their decimal form. A Ruby
318
320
  # +Hash#to_s+ haystack used to leak `=>` and `:sym` syntax that no
319
321
  # JSON document contains (STO-8).
320
322
  #
@@ -425,23 +427,22 @@ module Woods
425
427
  # Field names are interpolated into a `json_extract` JSON-path
426
428
  # literal, so they are validated against {SEARCH_FIELD_NAME} first —
427
429
  # a crafted name could otherwise break out of the literal and alter
428
- # the SQL shape. LIKE metacharacters (`%`, `_`, `\`) in the query are
429
- # escaped (with an explicit ESCAPE clause) so they match literally,
430
- # aligning with the InMemory adapter's substring semantics instead of
431
- # silently broadening matches.
430
+ # the SQL shape. instr and lower perform literal, ASCII-case-insensitive
431
+ # substring matching without LIKE's truncation at embedded NUL bytes.
432
+ # Wildcard characters need no escaping.
432
433
  #
433
434
  # @raise [ArgumentError] if a field name fails the whitelist
434
435
  def search(query, fields: nil)
435
436
  fields = validate_search_fields!(fields)
436
437
  return [] if fields == []
437
438
 
438
- pattern = "%#{escape_like(query)}%"
439
+ needle = query.to_s
439
440
  if fields
440
- conditions = fields.map { "json_extract(data, '$.#{_1}') LIKE ? ESCAPE '\\'" }.join(' OR ')
441
- params = Array.new(fields.size, pattern)
441
+ conditions = fields.map { "instr(lower(#{field_haystack_sql(_1)}), lower(?)) > 0" }.join(' OR ')
442
+ params = Array.new(fields.size, needle)
442
443
  rows = @db.execute("SELECT id, data FROM units WHERE #{conditions}", params)
443
444
  else
444
- rows = @db.execute("SELECT id, data FROM units WHERE data LIKE ? ESCAPE '\\'", [pattern])
445
+ rows = @db.execute('SELECT id, data FROM units WHERE instr(lower(data), lower(?)) > 0', [needle])
445
446
  end
446
447
 
447
448
  rows.map { |row| parse_row(row) }
@@ -489,15 +490,14 @@ module Woods
489
490
  defined?(SQLite3::BusyException) && error.is_a?(SQLite3::BusyException)
490
491
  end
491
492
 
492
- # Escape SQL LIKE metacharacters in a user query so they match
493
- # literally under the `ESCAPE '\'` clause {#search} emits. Without
494
- # this, `%` and `_` in a query act as wildcards and silently broaden
495
- # matches (`"user_name"` would match `"userXname"`).
496
- #
497
- # @param query [String] Raw search query
498
- # @return [String] Query safe for embedding in a LIKE pattern
499
- def escape_like(query)
500
- query.to_s.gsub(/[\\%_]/) { |ch| "\\#{ch}" }
493
+ # JSON1 extracts Boolean scalars as integers; use their JSON type to
494
+ # preserve true/false without conflating them with numeric 1/0.
495
+ # Field names have already passed validate_search_fields!.
496
+ def field_haystack_sql(field)
497
+ path = "$.#{field}"
498
+ "CASE json_type(data, '#{path}') " \
499
+ "WHEN 'true' THEN 'true' WHEN 'false' THEN 'false' " \
500
+ "ELSE json_extract(data, '#{path}') END"
501
501
  end
502
502
 
503
503
  # Parse a database row into a metadata hash with the id field injected.
@@ -122,6 +122,9 @@ module Woods
122
122
  SQL
123
123
  end
124
124
 
125
+ # Native raw-ID eligibility is applied before ranking and the limit.
126
+ def supports_id_filter? = true
127
+
125
128
  # Search for similar vectors using cosine distance.
126
129
  #
127
130
  # The query vector is dimension-checked before SQL runs, mirroring the
@@ -136,19 +139,14 @@ module Woods
136
139
  # @raise [Woods::Error] if the query vector's length disagrees with the
137
140
  # configured dimension
138
141
  # @see Interface#search
139
- def search(query_vector, limit: 10, filters: {})
142
+ def search(query_vector, limit: 10, filters: {}, ids: nil)
140
143
  validate_vector!(query_vector)
141
144
  validate_dimensions!(query_vector) if @dimensions
142
145
  vector_literal = build_vector_literal(query_vector)
143
146
  where_clause = build_where(filters)
147
+ where_clause = append_id_filter(where_clause, ids) if ids
144
148
 
145
- sql = <<~SQL
146
- SELECT id, embedding <=> '#{vector_literal}' AS distance, metadata
147
- FROM #{qualified_table}
148
- #{where_clause}
149
- ORDER BY distance ASC
150
- LIMIT #{limit.to_i}
151
- SQL
149
+ sql = search_sql(vector_literal, where_clause, limit, ids)
152
150
 
153
151
  rows = @connection.execute(sql)
154
152
  rows.map { |row| row_to_result(row) }
@@ -345,6 +343,29 @@ module Woods
345
343
  )
346
344
  end
347
345
 
346
+ def search_sql(vector_literal, where_clause, limit, ids)
347
+ source = ids ? 'eligible' : qualified_table
348
+ prefix = if ids
349
+ 'WITH eligible AS MATERIALIZED (SELECT id, embedding, metadata ' \
350
+ "FROM #{qualified_table} #{where_clause})\n"
351
+ else
352
+ ''
353
+ end
354
+ prefix + <<~SQL
355
+ SELECT id, embedding <=> '#{vector_literal}' AS distance, metadata
356
+ FROM #{source}
357
+ #{where_clause unless ids}
358
+ ORDER BY distance ASC#{', id ASC' if ids}
359
+ LIMIT #{limit.to_i}
360
+ SQL
361
+ end
362
+
363
+ # Raw IDs work with old vectors that have no identifier/package payload.
364
+ def append_id_filter(where_clause, ids)
365
+ condition = ids.empty? ? 'FALSE' : "id IN (#{ids.map { |id| @connection.quote(id.to_s) }.join(', ')})"
366
+ where_clause.empty? ? "WHERE #{condition}" : "#{where_clause} AND #{condition}"
367
+ end
368
+
348
369
  # Build a WHERE clause from metadata filters.
349
370
  #
350
371
  # @param filters [Hash] Metadata key-value pairs
@@ -332,6 +332,9 @@ module Woods
332
332
  request(:put, "/collections/#{@collection}/points#{WAIT_FOR_WRITE}", body)
333
333
  end
334
334
 
335
+ # Native raw-ID eligibility is applied before ranking and the limit.
336
+ def supports_id_filter? = true
337
+
335
338
  # Search for similar vectors.
336
339
  #
337
340
  # The query vector is dimension-checked before the request, mirroring
@@ -348,14 +351,11 @@ module Woods
348
351
  # @raise [Woods::Error] if the query vector's length disagrees with the
349
352
  # configured dimension
350
353
  # @see Interface#search
351
- def search(query_vector, limit: 10, filters: {})
354
+ def search(query_vector, limit: 10, filters: {}, ids: nil)
355
+ return [] if ids == []
356
+
352
357
  validate_dimensions!(query_vector) if @dimensions
353
- body = {
354
- vector: query_vector,
355
- limit: limit,
356
- with_payload: true
357
- }
358
- body[:filter] = build_filter(filters) unless filters.empty?
358
+ body = search_body(query_vector, limit, filters, ids)
359
359
 
360
360
  response = request(:post, "/collections/#{@collection}/points/search", body)
361
361
  results = response['result'] || []
@@ -579,6 +579,16 @@ module Woods
579
579
  "Vector dimension mismatch#{where}: got #{got}, expected #{@dimensions}"
580
580
  end
581
581
 
582
+ def search_body(query_vector, limit, filters, ids)
583
+ body = { vector: query_vector, limit: limit, with_payload: true }
584
+ body[:filter] = build_filter(filters) unless filters.empty? && ids.nil?
585
+ if ids
586
+ body[:filter][:must] << { key: IDENTIFIER_KEY, match: { any: ids } }
587
+ body[:params] = { exact: true }
588
+ end
589
+ body
590
+ end
591
+
582
592
  # Build a Qdrant filter from metadata key-value pairs.
583
593
  #
584
594
  # @param filters [Hash] Metadata filters
@@ -90,12 +90,17 @@ module Woods
90
90
  # @param limit [Integer] Maximum number of results to return
91
91
  # @param filters [Hash] Optional metadata filters — values may be
92
92
  # scalars or Arrays
93
+ # @param ids [Array<String>, nil] Raw vector IDs eligible before ranking;
94
+ # optional capability advertised by #supports_id_filter?. Empty matches none.
93
95
  # @return [Array<SearchResult>] Results sorted by descending similarity
94
96
  # @raise [NotImplementedError] if not implemented by adapter
95
- def search(query_vector, limit: 10, filters: {})
97
+ def search(query_vector, limit: 10, filters: {}, ids: nil)
96
98
  raise NotImplementedError
97
99
  end
98
100
 
101
+ # Whether native raw-ID eligibility is supported before the result limit.
102
+ def supports_id_filter? = false
103
+
99
104
  # Delete a vector by ID.
100
105
  #
101
106
  # @param id [String] The identifier to delete
@@ -220,8 +225,10 @@ module Woods
220
225
  end
221
226
  end
222
227
 
228
+ def supports_id_filter? = true
229
+
223
230
  # @see Interface#search
224
- def search(query_vector, limit: 10, filters: {})
231
+ def search(query_vector, limit: 10, filters: {}, ids: nil)
225
232
  return [] if @dim.nil?
226
233
 
227
234
  unless query_vector.length == @dim
@@ -229,8 +236,8 @@ module Woods
229
236
  "Vector dimension mismatch (#{query_vector.length} vs #{@dim})"
230
237
  end
231
238
 
232
- scored = gather_candidates(query_vector, filters)
233
- scored.sort_by! { |r| -r.score }
239
+ scored = gather_candidates(query_vector, filters, ids)
240
+ scored.sort_by! { |r| ids ? [-r.score, r.id] : [-r.score] }
234
241
  scored.first(limit)
235
242
  end
236
243
 
@@ -300,15 +307,20 @@ module Woods
300
307
  @metadata[idx] = metadata
301
308
  end
302
309
 
310
+ def excluded_index?(idx, allowed_ids)
311
+ @tombstones.include?(idx) || (allowed_ids && !allowed_ids.include?(@ids[idx]))
312
+ end
313
+
303
314
  # Walk every non-tombstoned index, apply filters, score survivors.
304
315
  # Filter check runs BEFORE the cosine kernel — avoids computing
305
316
  # 12k dot products only to discard most of them.
306
- def gather_candidates(query_vector, filters)
317
+ def gather_candidates(query_vector, filters, ids)
307
318
  scored = []
319
+ allowed_ids = ids&.to_set
308
320
  len = @ids.size
309
321
  idx = 0
310
322
  while idx < len
311
- if @tombstones.include?(idx)
323
+ if excluded_index?(idx, allowed_ids)
312
324
  idx += 1
313
325
  next
314
326
  end
data/lib/woods/tasks.rb CHANGED
@@ -27,6 +27,7 @@ module Woods
27
27
  # @return [Embedding::Indexer]
28
28
  def build_embed_indexer
29
29
  config = Woods.configuration
30
+ output_dir = ENV.fetch('WOODS_OUTPUT', config.output_dir)
30
31
  builder = Builder.new(config)
31
32
  provider = builder.build_embedding_provider
32
33
 
@@ -56,11 +57,11 @@ module Woods
56
57
  provider: resilient_provider,
57
58
  text_preparer: builder.build_text_preparer(provider),
58
59
  vector_store: vector_store,
59
- metadata_store: config.metadata_store ? builder.build_metadata_store : nil,
60
+ metadata_store: config.metadata_store ? builder.build_metadata_store(output_dir: output_dir) : nil,
60
61
  resolved_config: build_resolved_config(config, provider: provider),
61
62
  chunker: builder.build_chunker(provider),
62
63
  dump_retention_count: config.dump_retention_count,
63
- output_dir: ENV.fetch('WOODS_OUTPUT', config.output_dir)
64
+ output_dir: output_dir
64
65
  )
65
66
  end
66
67
 
@@ -22,7 +22,8 @@ module Woods
22
22
  #
23
23
  # Implements the same public interface as SnapshotStore so the MCP server
24
24
  # tools work identically.
25
- # Malformed retained snapshots are warned about and treated as absent:
25
+ # Malformed, non-object, or unreadable retained snapshots (including files
26
+ # removed during retention) are warned about and treated as absent:
26
27
  # +find+ returns nil, +diff+ returns an empty result, and history/list scans
27
28
  # omit the corrupt file.
28
29
  #
@@ -122,9 +123,9 @@ module Woods
122
123
  value.positive? ? value : PayloadStore::DEFAULT_RETENTION
123
124
  end
124
125
 
125
- # Delete snapshots beyond the retention count, oldest by extracted_at
126
- # first. +protect+ names the snapshot just captured; it is the newest
127
- # anyway, and a tie on extracted_at must not delete it.
126
+ # Delete snapshots beyond the retention count, corrupt/unreadable files
127
+ # first, then oldest by extracted_at. +protect+ names the snapshot just
128
+ # captured; even an older timestamp must not cause its deletion.
128
129
  #
129
130
  # Failures are non-fatal: retention is housekeeping, and a failed
130
131
  # prune leaves the store growing as it did before rather than
@@ -133,7 +134,7 @@ module Woods
133
134
  # @param protect [String] git SHA of the just-captured snapshot
134
135
  # @return [void]
135
136
  def prune_snapshots(protect:)
136
- summaries = load_all_summaries
137
+ summaries = retention_summaries
137
138
  overflow = summaries.size - @retention
138
139
  return unless overflow.positive?
139
140
 
@@ -142,7 +143,21 @@ module Woods
142
143
  warn "[Woods] Snapshot retention failed: #{e.message}"
143
144
  end
144
145
 
145
- # Oldest `overflow` snapshots by extracted_at, never the protected one.
146
+ # Include unreadable snapshots in the bound. Only files named like a
147
+ # snapshot are eligible; the filename owns identity, not JSON content.
148
+ def retention_summaries
149
+ Dir.glob(File.join(@dir, '*.json')).filter_map do |path|
150
+ sha = File.basename(path, '.json')
151
+ next unless sha.match?(/\A[0-9a-f]+\z/i) && File.file?(path)
152
+
153
+ data = read_snapshot(path) || {}
154
+ timestamp = data['extracted_at']
155
+ valid = data['git_sha'] == sha && timestamp.is_a?(String)
156
+ { git_sha: sha, extracted_at: valid ? timestamp : '', corrupt: !valid }
157
+ end
158
+ end
159
+
160
+ # Corrupt files, then oldest `overflow` snapshots, never the protected one.
146
161
  # The protected SHA is rejected before sorting and slicing: it is
147
162
  # captured last, so its extracted_at can tie or precede older entries,
148
163
  # and letting it occupy the victim slice would leave the store one file
@@ -154,7 +169,7 @@ module Woods
154
169
  # @return [Array<String>]
155
170
  def retention_victims(summaries, overflow, protect)
156
171
  summaries.reject { |summary| summary[:git_sha] == protect }
157
- .sort_by { |summary| summary[:extracted_at] || '' }
172
+ .sort_by { |summary| [summary[:corrupt] ? 0 : 1, summary[:extracted_at], summary[:git_sha]] }
158
173
  .first(overflow)
159
174
  .map { |summary| summary[:git_sha] }
160
175
  end
@@ -264,10 +279,16 @@ module Woods
264
279
  end
265
280
 
266
281
  def read_snapshot(path)
267
- JSON.parse(AtomicFile.read(path))
282
+ data = JSON.parse(AtomicFile.read(path))
283
+ raise JSON::ParserError, 'expected a JSON object' unless data.is_a?(Hash)
284
+
285
+ data
268
286
  rescue JSON::ParserError => e
269
287
  warn "[Woods] Skipping corrupt snapshot #{File.basename(path)}: #{e.message}"
270
288
  nil
289
+ rescue SystemCallError => e
290
+ warn "[Woods] Skipping unreadable snapshot #{File.basename(path)}: #{e.message}"
291
+ nil
271
292
  end
272
293
 
273
294
  # @param exclude_sha [String, nil] SHA to leave out of the result
@@ -8,6 +8,7 @@ require_relative 'client'
8
8
  require_relative 'rate_limiter'
9
9
  require_relative 'document_builder'
10
10
  require_relative 'sync_manifest'
11
+ require_relative '../export/typed_reader'
11
12
 
12
13
  module Woods
13
14
  module Unblocked
@@ -75,6 +76,7 @@ module Woods
75
76
 
76
77
  @client = client || Client.new(api_token: api_token, rate_limiter: limiter)
77
78
  @reader = reader || build_reader(index_dir)
79
+ @typed_reader = Export::TypedReader.new(@reader)
78
80
  # Cite the ref the index was actually extracted from. `main` was
79
81
  # hardcoded, so citations on a `master`-default repo pointed at a
80
82
  # branch that need not exist.
@@ -96,10 +98,11 @@ module Woods
96
98
  #
97
99
  # @return [Hash] { synced:, skipped:, deleted:, errors: }
98
100
  def sync_all
99
- with_pinned_index do
101
+ prepared = false
102
+ with_prepared_index do
103
+ prepared = true
100
104
  @current_uris = Set.new
101
105
  @budget_exhausted = false
102
- build_uri_index
103
106
  reconcile_from_remote if @manifest.empty?
104
107
 
105
108
  synced = 0
@@ -108,6 +111,7 @@ module Woods
108
111
 
109
112
  FULL_SYNC_TYPES.each do |type|
110
113
  break if @budget_exhausted
114
+ next if type.start_with?('graphql_') # the family already includes these actual types
111
115
 
112
116
  result = sync_type(type)
113
117
  synced += result[:synced]
@@ -124,11 +128,11 @@ module Woods
124
128
  errors.concat(result[:errors])
125
129
  end
126
130
 
127
- deleted = @budget_exhausted ? 0 : purge_stale(errors)
131
+ deleted = @budget_exhausted || @ambiguous_uris.any? ? 0 : purge_stale(errors)
128
132
  { synced: synced, skipped: skipped, deleted: deleted, errors: cap_errors(errors) }
129
133
  end
130
134
  ensure
131
- save_manifest
135
+ save_manifest if prepared
132
136
  end
133
137
 
134
138
  # Sync all units of a given type.
@@ -136,10 +140,12 @@ module Woods
136
140
  # @param type [String] Unit type (e.g. "model", "controller")
137
141
  # @return [Hash] { synced:, skipped:, errors: }
138
142
  def sync_type(type)
139
- units = @reader.list_units(type: type)
140
- log " #{type}: #{units.size} units"
143
+ with_prepared_index do
144
+ units = units_for(type)
145
+ log " #{type}: #{units.size} units"
141
146
 
142
- sync_units(units)
147
+ sync_unit_data(units.map { |unit| [unit, unit] })
148
+ end
143
149
  end
144
150
 
145
151
  # Sync the top N most-connected units of a type (by dependent count).
@@ -148,35 +154,42 @@ module Woods
148
154
  # @param max_count [Integer] Maximum units to sync
149
155
  # @return [Hash] { synced:, skipped:, errors: }
150
156
  def sync_type_partial(type, max_count)
151
- units = @reader.list_units(type: type)
152
- return empty_stats if units.empty?
153
-
154
- # Load full data to sort by dependent count
155
- units_with_data = units.filter_map do |entry|
156
- data = @reader.find_unit(entry['identifier'])
157
- next unless data
158
-
159
- dep_count = (data['dependents'] || []).size
160
- { entry: entry, data: data, dep_count: dep_count }
157
+ with_prepared_index do
158
+ units = units_for(type)
159
+ units.each { |unit| track_uri(unit) }
160
+ top = units.sort_by { |unit| -(unit['dependents'] || []).size }.first(max_count)
161
+ log " #{type}: #{top.size}/#{units.size} units (top by dependents)"
162
+ result = sync_unit_data(top.map { |unit| [unit, unit] })
163
+ result[:skipped] += units.size - top.size
164
+ result
161
165
  end
166
+ end
162
167
 
163
- # Every unit of this type still exists — track its URI so partial units
164
- # that fall *out* of the top-N are never mistaken for deletions.
165
- units_with_data.each { |u| track_uri(u[:data]) }
166
-
167
- top_units = units_with_data.sort_by { |u| -u[:dep_count] }.first(max_count)
168
- # Count against what was actually synced — units.size includes entries
169
- # whose unit data was missing (dropped by the filter_map above).
170
- skipped_count = units.size - top_units.size
168
+ private
171
169
 
172
- log " #{type}: #{top_units.size}/#{units.size} units (top by dependents)"
170
+ # Complete preflight precedes all remote writes and destructive pruning.
171
+ # Nested sync methods share this snapshot; standalone methods pin as well.
172
+ def with_prepared_index
173
+ return yield if @prepared_index
173
174
 
174
- result = sync_unit_data(top_units.map { |u| [u[:entry], u[:data]] })
175
- result[:skipped] += skipped_count
176
- result
175
+ with_pinned_index do
176
+ @published_units = @typed_reader.all
177
+ build_uri_index
178
+ @prepared_index = true
179
+ begin
180
+ yield
181
+ ensure
182
+ @prepared_index = false
183
+ @published_units = nil
184
+ end
185
+ end
177
186
  end
178
187
 
179
- private
188
+ def units_for(type)
189
+ # The historical graphql family includes all four actual subtypes.
190
+ types = type == 'graphql' ? MCP::IndexReader::UNIT_TYPES_BY_DIR.fetch('graphql') : [type]
191
+ @published_units.select { |unit| types.include?(unit['type']) }
192
+ end
180
193
 
181
194
  # Run a multi-read export body against one index generation.
182
195
  #
@@ -197,36 +210,6 @@ module Woods
197
210
  @reader.with_pinned_generation(&block)
198
211
  end
199
212
 
200
- def sync_units(units)
201
- synced = 0
202
- skipped = 0
203
- errors = []
204
-
205
- units.each do |entry|
206
- unit_data = @reader.find_unit(entry['identifier'])
207
- unless unit_data
208
- skipped += 1
209
- next
210
- end
211
-
212
- track_uri(unit_data)
213
- if push_document(unit_data) == :skipped
214
- skipped += 1
215
- else
216
- synced += 1
217
- end
218
- rescue Woods::Error => e
219
- errors << "#{entry['identifier']}: #{e.message}"
220
- break if note_budget_exhaustion(e)
221
- rescue StandardError => e
222
- # Include the class — "undefined method for nil" without it is
223
- # unactionable in CI logs.
224
- errors << "#{entry['identifier']}: #{e.class}: #{e.message}"
225
- end
226
-
227
- { synced: synced, skipped: skipped, errors: errors }
228
- end
229
-
230
213
  def sync_unit_data(entries_with_data)
231
214
  synced = 0
232
215
  skipped = 0
@@ -256,6 +239,10 @@ module Woods
256
239
  #
257
240
  # @return [Symbol] :synced or :skipped
258
241
  def push_document(unit_data)
242
+ if @ambiguous_uris.include?(@builder.uri_for(unit_data))
243
+ raise ExtractionError,
244
+ "ambiguous export URI for #{unit_data['type']}:#{unit_data['identifier']} — push skipped"
245
+ end
259
246
  # No file_path → the URI falls back to the bare repo URL, which every
260
247
  # such unit would share: they'd overwrite each other remotely and
261
248
  # ping-pong the manifest hash forever. Skip them.
@@ -411,24 +398,26 @@ module Woods
411
398
  "#{base}?unit=#{URI.encode_www_form_component(unit_data['identifier'])}"
412
399
  end
413
400
 
414
- # One cheap pass over the type indexes (entries already carry file_path,
415
- # and read_index is cached) to find files that define more than one synced
416
- # unit. For each such base URI, the lexically-smallest identifier — the
401
+ # Inspect the complete validated snapshot, including excluded types, for
402
+ # same-name cross-type collisions. For files with distinct synced names,
403
+ # the lexically-smallest identifier — the
417
404
  # outer/top-level class — keeps the bare URI; siblings are suffixed. Solo
418
405
  # files (the overwhelming majority) are absent from the map and unchanged,
419
406
  # so this introduces no churn for them.
420
407
  def build_uri_index
421
408
  groups = Hash.new { |h, k| h[k] = [] }
422
- synced_types.each do |type|
423
- @reader.list_units(type: type).each do |entry|
424
- next unless entry['file_path']
409
+ @published_units.each do |unit|
410
+ next unless unit['file_path']
425
411
 
426
- groups[@builder.uri_for(entry)] << entry['identifier']
427
- end
412
+ groups[@builder.uri_for(unit)] << unit
428
413
  end
429
414
 
430
- @uri_primary = groups.each_with_object({}) do |(uri, identifiers), primary|
431
- unique = identifiers.uniq
415
+ @ambiguous_uris = groups.each_with_object(Set.new) do |(uri, units), ambiguous|
416
+ identities = units.map { |unit| [unit['identifier'], unit['type']] }.uniq
417
+ ambiguous << uri if identities.group_by(&:first).any? { |_, variants| variants.size > 1 }
418
+ end
419
+ @uri_primary = groups.each_with_object({}) do |(uri, units), primary|
420
+ unique = units.select { |unit| synced_types.include?(unit['type']) }.map { |unit| unit['identifier'] }.uniq
432
421
  primary[uri] = unique.min if unique.size > 1
433
422
  end
434
423
  end
data/lib/woods/version.rb CHANGED
@@ -1,5 +1,5 @@
1
1
  # frozen_string_literal: true
2
2
 
3
3
  module Woods
4
- VERSION = '2.0.0.beta1'
4
+ VERSION = '2.0.0.beta3'
5
5
  end
@@ -0,0 +1,52 @@
1
+ # frozen_string_literal: true
2
+
3
+ require_relative '../reload_policy'
4
+ require_relative 'tree_scan'
5
+ require_relative 'watcher'
6
+
7
+ module Woods
8
+ module Watch
9
+ # Files observed before the watch task invokes Rails' environment task.
10
+ # This bounds environment initialization, not earlier Bundler/application
11
+ # loading. Hosts embedding the daemon may provide the same explicit boundary.
12
+ class BootSnapshot
13
+ def initialize(root:, policy: ReloadPolicy.new)
14
+ @root = File.expand_path(root.to_s)
15
+ @policy = policy
16
+ @files = scan
17
+ end
18
+
19
+ # True for unchanged files, including a carried path absent before and
20
+ # after boot. A deleted initializer must not demand another restart.
21
+ def covers?(path)
22
+ absolute = File.expand_path(path, @root)
23
+ @files[absolute] == fingerprint(absolute)
24
+ end
25
+
26
+ # Includes additions and deletions, even if another writer has advanced
27
+ # the index watermark while Rails was booting.
28
+ def changed_paths
29
+ current = scan
30
+ (@files.keys | current.keys).reject { |path| @files[path] == current[path] }
31
+ end
32
+
33
+ private
34
+
35
+ def scan
36
+ TreeScan.files(root: @root, ignored: Watcher::DEFAULT_IGNORED_DIRECTORIES).each_with_object({}) do |path, files|
37
+ next unless %i[restart reload].include?(@policy.classify(path.delete_prefix("#{@root}/")))
38
+
39
+ stamp = fingerprint(path)
40
+ files[path] = stamp if stamp
41
+ end
42
+ end
43
+
44
+ def fingerprint(path)
45
+ stat = File.stat(path)
46
+ [stat.ino, stat.size, stat.mtime, stat.ctime]
47
+ rescue Errno::ENOENT, Errno::ENOTDIR
48
+ nil
49
+ end
50
+ end
51
+ end
52
+ end