woods 2.0.0.beta1 → 2.0.0.beta3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (221) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +400 -1
  3. data/CONTRIBUTING.md +224 -9
  4. data/README.md +7 -3
  5. data/SECURITY.md +9 -6
  6. data/docs/AGENT_GUIDE.md +83 -4
  7. data/docs/AGENT_SETUP.md +82 -1
  8. data/docs/BACKEND_MATRIX.md +20 -0
  9. data/docs/CLIENT_HOOKS.md +111 -0
  10. data/docs/CONFIGURATION_REFERENCE.md +233 -13
  11. data/docs/CONSOLE_MCP_SETUP.md +35 -5
  12. data/docs/DOCKER_SETUP.md +21 -2
  13. data/docs/EVALUATION.md +464 -1
  14. data/docs/EXTRACTOR_REFERENCE.md +36 -5
  15. data/docs/FAQ.md +11 -12
  16. data/docs/GETTING_STARTED.md +17 -5
  17. data/docs/INCREMENTAL_EXTRACTION.md +158 -2
  18. data/docs/INDEX_LAYOUT.md +382 -0
  19. data/docs/INTERNALS.md +15 -7
  20. data/docs/MCP_SERVERS.md +221 -5
  21. data/docs/MCP_TOOL_COOKBOOK.md +33 -18
  22. data/docs/NOTION_INTEGRATION.md +13 -0
  23. data/docs/OBSIDIAN_INTEGRATION.md +57 -9
  24. data/docs/PUBLISHED_INDEX.md +71 -0
  25. data/docs/README.md +7 -0
  26. data/docs/RETRIEVAL_GUIDE.md +253 -11
  27. data/docs/RUNTIME_TRACING.md +71 -0
  28. data/docs/SOURCE_FRESHNESS.md +143 -0
  29. data/docs/TROUBLESHOOTING.md +117 -5
  30. data/docs/UNBLOCKED_INTEGRATION.md +25 -0
  31. data/docs/UPGRADING_TO_2.md +44 -22
  32. data/docs/WATCH_DAEMON.md +259 -59
  33. data/exe/woods-agent-config +6 -0
  34. data/exe/woods-extract +5 -0
  35. data/exe/woods-hook-context +6 -0
  36. data/lib/generators/woods/templates/woods.rb.tt +1 -3
  37. data/lib/tasks/woods.rake +47 -397
  38. data/lib/woods/agent_configuration/applier.rb +133 -0
  39. data/lib/woods/agent_configuration/cli.rb +101 -0
  40. data/lib/woods/agent_configuration/cli_options.rb +29 -0
  41. data/lib/woods/agent_configuration/document.rb +105 -0
  42. data/lib/woods/agent_configuration/error.rb +7 -0
  43. data/lib/woods/agent_configuration/launcher.rb +75 -0
  44. data/lib/woods/agent_configuration/layout.rb +59 -0
  45. data/lib/woods/agent_configuration/managed_section.rb +62 -0
  46. data/lib/woods/agent_configuration/plan.rb +98 -0
  47. data/lib/woods/agent_configuration/plan_diff.rb +38 -0
  48. data/lib/woods/agent_configuration/planned_files.rb +61 -0
  49. data/lib/woods/agent_configuration/planner.rb +63 -0
  50. data/lib/woods/agent_configuration/planner_validation.rb +77 -0
  51. data/lib/woods/agent_configuration/preflight.rb +100 -0
  52. data/lib/woods/agent_configuration/recovery.rb +49 -0
  53. data/lib/woods/ast/node.rb +2 -0
  54. data/lib/woods/ast/parser.rb +38 -5
  55. data/lib/woods/atomic_file.rb +133 -3
  56. data/lib/woods/builder.rb +21 -5
  57. data/lib/woods/cache/cache_middleware.rb +28 -7
  58. data/lib/woods/cache/cache_store.rb +4 -5
  59. data/lib/woods/change_set.rb +5 -4
  60. data/lib/woods/console/credential_index.rb +20 -2
  61. data/lib/woods/console/credential_scanner.rb +14 -14
  62. data/lib/woods/console/credential_scanner_registry.rb +36 -0
  63. data/lib/woods/console/embedded_executor.rb +1 -1
  64. data/lib/woods/console/encrypted_credential_snapshot.rb +16 -0
  65. data/lib/woods/console/rack_middleware.rb +22 -13
  66. data/lib/woods/console/server.rb +18 -16
  67. data/lib/woods/dependency_graph.rb +65 -13
  68. data/lib/woods/embedding/corpus.rb +94 -0
  69. data/lib/woods/embedding/indexer.rb +90 -46
  70. data/lib/woods/embedding/openai.rb +17 -6
  71. data/lib/woods/evaluation/ablation_executor.rb +6 -1
  72. data/lib/woods/evaluation/ablation_timed_executor.rb +22 -4
  73. data/lib/woods/export/typed_reader.rb +56 -0
  74. data/lib/woods/extractor.rb +557 -228
  75. data/lib/woods/extractors/action_cable_extractor.rb +3 -1
  76. data/lib/woods/extractors/behavioral_profile.rb +9 -7
  77. data/lib/woods/extractors/caching_extractor.rb +3 -1
  78. data/lib/woods/extractors/concern_extractor.rb +64 -6
  79. data/lib/woods/extractors/configuration_extractor.rb +7 -3
  80. data/lib/woods/extractors/controller_extractor.rb +13 -4
  81. data/lib/woods/extractors/database_view_extractor.rb +3 -1
  82. data/lib/woods/extractors/decorator_extractor.rb +3 -1
  83. data/lib/woods/extractors/engine_extractor.rb +3 -1
  84. data/lib/woods/extractors/event_extractor.rb +4 -2
  85. data/lib/woods/extractors/factory_extractor.rb +3 -1
  86. data/lib/woods/extractors/graphql_extractor.rb +8 -2
  87. data/lib/woods/extractors/i18n_extractor.rb +3 -1
  88. data/lib/woods/extractors/job_extractor.rb +6 -19
  89. data/lib/woods/extractors/lib_extractor.rb +3 -1
  90. data/lib/woods/extractors/mailer_extractor.rb +20 -5
  91. data/lib/woods/extractors/manager_extractor.rb +3 -1
  92. data/lib/woods/extractors/method_parameters.rb +53 -0
  93. data/lib/woods/extractors/middleware_argument.rb +65 -0
  94. data/lib/woods/extractors/middleware_extractor.rb +9 -3
  95. data/lib/woods/extractors/migration_extractor.rb +3 -1
  96. data/lib/woods/extractors/model_extractor.rb +39 -33
  97. data/lib/woods/extractors/package_extractor.rb +24 -4
  98. data/lib/woods/extractors/phlex_extractor.rb +3 -1
  99. data/lib/woods/extractors/policy_extractor.rb +3 -1
  100. data/lib/woods/extractors/poro_extractor.rb +3 -1
  101. data/lib/woods/extractors/pundit_extractor.rb +3 -1
  102. data/lib/woods/extractors/rails_source_extractor.rb +4 -2
  103. data/lib/woods/extractors/rake_task_extractor.rb +4 -2
  104. data/lib/woods/extractors/route_extractor.rb +3 -1
  105. data/lib/woods/extractors/route_helper_resolver.rb +10 -33
  106. data/lib/woods/extractors/scheduled_job_extractor.rb +41 -15
  107. data/lib/woods/extractors/serializer_extractor.rb +4 -2
  108. data/lib/woods/extractors/service_extractor.rb +3 -1
  109. data/lib/woods/extractors/shared_dependency_scanner.rb +2 -2
  110. data/lib/woods/extractors/shared_utility_methods.rb +27 -15
  111. data/lib/woods/extractors/source_nesting.rb +1 -1
  112. data/lib/woods/extractors/state_machine_extractor.rb +3 -1
  113. data/lib/woods/extractors/test_mapping_extractor.rb +3 -1
  114. data/lib/woods/extractors/validator_extractor.rb +3 -1
  115. data/lib/woods/extractors/view_component_extractor.rb +3 -1
  116. data/lib/woods/extractors/view_template_extractor.rb +3 -1
  117. data/lib/woods/flow_assembler.rb +87 -8
  118. data/lib/woods/flow_precomputer.rb +44 -7
  119. data/lib/woods/gem_mapper.rb +2 -0
  120. data/lib/woods/git_history.rb +116 -0
  121. data/lib/woods/graph_analyzer.rb +195 -63
  122. data/lib/woods/hooks/context_cli.rb +54 -0
  123. data/lib/woods/hooks/context_event.rb +88 -0
  124. data/lib/woods/hooks/context_hint.rb +73 -0
  125. data/lib/woods/hooks/context_impact.rb +77 -0
  126. data/lib/woods/hooks/context_output.rb +47 -0
  127. data/lib/woods/hooks/context_state.rb +102 -0
  128. data/lib/woods/hooks/refresh.rb +79 -0
  129. data/lib/woods/hooks/rule_projection.rb +78 -0
  130. data/lib/woods/input_rules.rb +19 -0
  131. data/lib/woods/mcp/bearer_auth.rb +20 -12
  132. data/lib/woods/mcp/bootstrapper.rb +62 -0
  133. data/lib/woods/mcp/index_reader.rb +323 -160
  134. data/lib/woods/mcp/initialization_guidance.rb +27 -0
  135. data/lib/woods/mcp/origin_guard.rb +17 -9
  136. data/lib/woods/mcp/published_lexical_retriever.rb +115 -0
  137. data/lib/woods/mcp/renderers/markdown_renderer.rb +8 -1
  138. data/lib/woods/mcp/renderers/plain_renderer.rb +7 -1
  139. data/lib/woods/mcp/search_results.rb +74 -0
  140. data/lib/woods/mcp/server.rb +158 -37
  141. data/lib/woods/mcp/tool_contract.rb +2 -0
  142. data/lib/woods/mcp/tool_response_renderer.rb +25 -0
  143. data/lib/woods/mcp/traversal_evidence.rb +113 -0
  144. data/lib/woods/mcp/traversal_evidence_index.rb +100 -0
  145. data/lib/woods/mcp/traversal_evidence_page.rb +41 -0
  146. data/lib/woods/mcp/traversal_evidence_text.rb +52 -0
  147. data/lib/woods/notion/exporter.rb +56 -17
  148. data/lib/woods/obsidian/destination_plan.rb +98 -0
  149. data/lib/woods/obsidian/name_mapper.rb +19 -3
  150. data/lib/woods/obsidian/note_builder.rb +19 -10
  151. data/lib/woods/obsidian/vault_exporter.rb +88 -32
  152. data/lib/woods/operator/pipeline_guard.rb +18 -13
  153. data/lib/woods/path_dispatcher.rb +7 -1
  154. data/lib/woods/payload_store.rb +29 -15
  155. data/lib/woods/railtie.rb +3 -3
  156. data/lib/woods/railtie_support.rb +12 -12
  157. data/lib/woods/rake_helpers.rb +392 -0
  158. data/lib/woods/resilience/graph_invariant_validator/membership_checks.rb +71 -0
  159. data/lib/woods/resilience/graph_invariant_validator/node_checks.rb +61 -0
  160. data/lib/woods/resilience/graph_invariant_validator/reverse_relationship_checks.rb +46 -0
  161. data/lib/woods/resilience/graph_invariant_validator.rb +119 -0
  162. data/lib/woods/resilience/index_validator/graph_checks.rb +80 -0
  163. data/lib/woods/resilience/index_validator.rb +112 -23
  164. data/lib/woods/retrieval/context_assembler.rb +50 -15
  165. data/lib/woods/retrieval/lexical_assembler.rb +73 -0
  166. data/lib/woods/retrieval/lexical_index.rb +119 -0
  167. data/lib/woods/retrieval/ranker.rb +4 -2
  168. data/lib/woods/retrieval/scope.rb +108 -0
  169. data/lib/woods/retrieval/scoped_graph_store.rb +32 -0
  170. data/lib/woods/retrieval/scoped_vector_store.rb +55 -0
  171. data/lib/woods/retrieval/search_executor.rb +86 -27
  172. data/lib/woods/retrieval/source_evidence.rb +200 -0
  173. data/lib/woods/retriever.rb +98 -22
  174. data/lib/woods/ruby_analyzer/trace_enricher.rb +80 -38
  175. data/lib/woods/session_tracer/middleware.rb +10 -12
  176. data/lib/woods/session_tracer/redis_store.rb +22 -6
  177. data/lib/woods/session_tracer/session_flow_assembler.rb +23 -17
  178. data/lib/woods/session_tracer/solid_cache_coordination.rb +6 -4
  179. data/lib/woods/session_tracer/unit_resolver.rb +63 -0
  180. data/lib/woods/source_inputs/consumer_errors.rb +27 -0
  181. data/lib/woods/source_inputs/handoff.rb +102 -0
  182. data/lib/woods/source_inputs/launcher.rb +157 -0
  183. data/lib/woods/source_inputs/manifest.rb +124 -0
  184. data/lib/woods/source_inputs/private_key.rb +55 -0
  185. data/lib/woods/source_inputs/scanner.rb +171 -0
  186. data/lib/woods/source_inputs/scopes.rb +71 -0
  187. data/lib/woods/source_inputs/session.rb +214 -0
  188. data/lib/woods/source_inputs/status.rb +84 -0
  189. data/lib/woods/source_inputs/verifier.rb +107 -0
  190. data/lib/woods/storage/metadata_store.rb +25 -25
  191. data/lib/woods/storage/pgvector.rb +29 -8
  192. data/lib/woods/storage/qdrant.rb +17 -7
  193. data/lib/woods/storage/vector_store.rb +18 -6
  194. data/lib/woods/tasks.rb +3 -2
  195. data/lib/woods/temporal/json_snapshot_store.rb +29 -8
  196. data/lib/woods/unblocked/exporter.rb +59 -70
  197. data/lib/woods/version.rb +1 -1
  198. data/lib/woods/watch/boot_snapshot.rb +52 -0
  199. data/lib/woods/watch/daemon.rb +136 -28
  200. data/lib/woods/watch/listen_watcher.rb +4 -0
  201. data/lib/woods/watch/polling_watcher.rb +5 -1
  202. data/lib/woods/watch/status.rb +20 -15
  203. data/lib/woods/watch/tree_scan.rb +21 -13
  204. data/lib/woods/watch/watcher.rb +4 -1
  205. data/lib/woods.rb +135 -11
  206. data/plugin/.claude-plugin/plugin.json +1 -1
  207. data/plugin/hooks/adapters/normalize.jq +15 -0
  208. data/plugin/hooks/adapters/normalize.rb +63 -0
  209. data/plugin/hooks/hooks.json +20 -0
  210. data/plugin/hooks/woods-context.sh +50 -0
  211. data/plugin/hooks/woods-input-rules.sh +159 -0
  212. data/plugin/hooks/woods-opencode.mjs +65 -0
  213. data/plugin/hooks/woods-post-edit.sh +2 -225
  214. data/plugin/hooks/woods-refresh.sh +260 -0
  215. data/plugin/hooks/woods-session-start.sh +47 -55
  216. data/plugin/skills/woods-agent-enable/SKILL.md +13 -0
  217. data/plugin/skills/woods-diagnose/SKILL.md +288 -1
  218. data/plugin/skills/woods-investigate/SKILL.md +106 -0
  219. data/plugin/skills/woods-mcp-config/SKILL.md +89 -1
  220. data/plugin/skills/woods-setup/SKILL.md +107 -6
  221. metadata +84 -5
@@ -8,6 +8,11 @@ require_relative 'retrieval/query_classifier'
8
8
  require_relative 'retrieval/search_executor'
9
9
  require_relative 'retrieval/ranker'
10
10
  require_relative 'retrieval/context_assembler'
11
+ require_relative 'retrieval/lexical_index'
12
+ require_relative 'retrieval/lexical_assembler'
13
+ require_relative 'retrieval/scope'
14
+ require_relative 'retrieval/scoped_vector_store'
15
+ require_relative 'retrieval/scoped_graph_store'
11
16
  require_relative 'embedding/token_counter'
12
17
  require_relative 'token_utils'
13
18
 
@@ -59,6 +64,7 @@ module Woods
59
64
  ).freeze
60
65
 
61
66
  # Diagnostic trace for retrieval quality analysis.
67
+ # +tokens_used+ counts the final delivered context, including postprocessing.
62
68
  #
63
69
  # +skipped_missing_metadata+ carries {Retrieval::ContextAssembler}'s count
64
70
  # of candidates dropped because the metadata store had no record for
@@ -98,7 +104,7 @@ module Woods
98
104
  #
99
105
  # Nil for unfiltered queries.
100
106
  RetrievalResult = Struct.new(:context, :sources, :classification, :strategy, :tokens_used, :budget, :trace,
101
- :type_rank_context, keyword_init: true)
107
+ :type_rank_context, :applied_scope, keyword_init: true)
102
108
 
103
109
  # Raised when a metadata-store access fails during retrieval (M8). One
104
110
  # shared typed error for every store call site: the retriever used to
@@ -191,7 +197,7 @@ module Woods
191
197
  # The reload transaction swaps the whole struct via {#swap_stores!}.
192
198
  #
193
199
  # @return [Pipeline]
194
- attr_reader :pipeline
200
+ attr_reader :pipeline, :mode, :default_budget
195
201
 
196
202
  # Optional callback invoked with the pipeline struct the moment
197
203
  # {#retrieve} resolves it, before any pipeline work runs. Nil in
@@ -207,7 +213,13 @@ module Woods
207
213
  # @param graph_store [Storage::GraphStore::Interface] Graph store adapter
208
214
  # @param embedding_provider [Embedding::Provider::Interface] Embedding provider
209
215
  # @param formatter [#call, nil] Optional callable to post-process the context string
210
- def initialize(vector_store:, metadata_store:, graph_store:, embedding_provider:, formatter: nil)
216
+ # @param default_budget [Integer] Token budget when retrieve omits budget
217
+ def initialize(vector_store:, metadata_store:, graph_store:, embedding_provider:, formatter: nil, mode: :semantic,
218
+ default_budget: 8000)
219
+ raise ArgumentError, 'unknown retrieval mode' unless %i[semantic lexical].include?(mode)
220
+
221
+ @mode = mode
222
+ @default_budget = default_budget
211
223
  @embedding_provider = embedding_provider
212
224
  @formatter = formatter
213
225
  @classifier = Retrieval::QueryClassifier.new
@@ -246,7 +258,9 @@ module Woods
246
258
  # they make raises the typed {StoreError} instead of a raw adapter error
247
259
  # (MCP-6); the struct keeps the raw adapters for identity and capability
248
260
  # checks.
249
- def build_pipeline(vector_store:, metadata_store:, graph_store:)
261
+ def build_pipeline(vector_store:, metadata_store:, graph_store:) # rubocop:disable Metrics/MethodLength
262
+ return build_lexical_pipeline(metadata_store, graph_store) if @mode == :lexical
263
+
250
264
  translated_vector = translate_store(vector_store, :vector)
251
265
  translated_metadata = translate_store(metadata_store, :metadata)
252
266
  translated_graph = translate_store(graph_store, :graph)
@@ -269,6 +283,14 @@ module Woods
269
283
  graph_store: graph_store
270
284
  )
271
285
  end
286
+
287
+ def build_lexical_pipeline(metadata_store, graph_store)
288
+ executor = Retrieval::LexicalIndex.new(metadata_store: translate_store(metadata_store, :metadata))
289
+ Pipeline.new(executor: executor, assembler: Retrieval::LexicalAssembler.new, metadata_store: metadata_store,
290
+ vector_store: nil, graph_store: graph_store)
291
+ end
292
+ private :build_lexical_pipeline
293
+
272
294
  private :build_pipeline
273
295
 
274
296
  # Wrap one store adapter for the pipeline components. Nil stays nil — a
@@ -357,16 +379,29 @@ module Woods
357
379
  # unit types (overrides DEFAULT_EXCLUDE_TYPES).
358
380
  # @param exclude_types [Array<String, Symbol>, nil] Additional types to
359
381
  # exclude. Applied on top of DEFAULT_EXCLUDE_TYPES unless +types:+ is set.
360
- # @return [RetrievalResult] Complete retrieval result
361
- def retrieve(query, budget: 8000, types: nil, exclude_types: nil)
382
+ # @param packages [Array<String>, nil] Exact published nearest package owners
383
+ # @param source_paths [Array<String>, nil] Application-relative directory prefixes
384
+ # @return [RetrievalResult] Complete retrieval result; scoped calls carry applied_scope
385
+ def retrieve(query, budget: @default_budget, types: nil, exclude_types: nil, packages: nil, source_paths: nil,
386
+ evidence: 'full', evidence_generation: nil) # rubocop:disable Metrics/MethodLength, Metrics/AbcSize
362
387
  validate_query!(query)
388
+ Retrieval::SourceEvidence.validate_mode!(evidence)
389
+ evidence_options = evidence == 'full' ? {} : { evidence: evidence, query: query, generation: evidence_generation }
363
390
  start_time = Process.clock_gettime(Process::CLOCK_MONOTONIC)
364
391
  # One atomic read of the bundle reference: everything this query does
365
392
  # from here on — execution, ranking, filtering, assembly — stays on the
366
393
  # SAME store set even if a reload swaps the pipeline mid-flight (M7).
367
394
  pipeline = @pipeline
368
395
  @pipeline_observer&.call(pipeline)
396
+ scope = resolve_scope(pipeline, packages, source_paths, types, exclude_types)
397
+ pipeline = scoped_pipeline(pipeline, scope) if scope
369
398
  classification = @classifier.classify(query)
399
+ if @mode == :lexical
400
+ result = retrieve_lexical(pipeline, query, classification, budget, types, exclude_types, start_time,
401
+ **evidence_options)
402
+ return attach_scope(result, scope)
403
+ end
404
+
370
405
  execution_result = pipeline.executor.execute(query: query, classification: classification)
371
406
  ranked = pipeline.ranker.rank(execution_result.candidates, classification: classification)
372
407
 
@@ -374,20 +409,61 @@ module Woods
374
409
  filtered, fallback_ran = apply_type_filter(pipeline, ranked, query, classification,
375
410
  types: types, type_list: type_list,
376
411
  exclude_types: exclude_types)
377
- type_rank_context = build_type_rank_context_for(ranked, pipeline, type_list, filtered,
378
- fallback_ran: fallback_ran)
379
-
380
- assembled = assemble_context(pipeline, filtered, classification, budget)
381
- trace = build_trace(classification, execution_result, filtered, assembled, start_time)
412
+ type_rank_context = unless scope
413
+ build_type_rank_context_for(ranked, pipeline, type_list, filtered,
414
+ fallback_ran: fallback_ran)
415
+ end
382
416
 
417
+ assembled = assemble_context(pipeline, filtered, classification, budget, **evidence_options)
383
418
  build_result(
384
- assembled: assembled, classification: classification, strategy: execution_result.strategy,
385
- budget: budget, trace: trace, type_rank_context: type_rank_context
386
- )
419
+ assembled: assembled, assembler: pipeline.assembler, classification: classification,
420
+ strategy: execution_result.strategy, budget: budget, type_rank_context: type_rank_context
421
+ ).tap do |result|
422
+ result.trace = build_trace(result, execution_result, filtered, assembled, start_time)
423
+ attach_scope(result, scope)
424
+ end
387
425
  end
388
426
 
389
427
  private
390
428
 
429
+ def resolve_scope(pipeline, packages, source_paths, types, excluded)
430
+ return unless Retrieval::Scope.requested?(packages: packages, source_paths: source_paths)
431
+
432
+ Retrieval::Scope.new(metadata_store: translate_store(pipeline.metadata_store, :metadata),
433
+ packages: packages, source_paths: source_paths, types: types,
434
+ exclude_types: DEFAULT_EXCLUDE_TYPES + Array(excluded).map(&:to_s))
435
+ end
436
+
437
+ def scoped_pipeline(pipeline, scope)
438
+ vector = Retrieval::ScopedVectorStore.new(store: pipeline.vector_store, scope: scope)
439
+ graph = Retrieval::ScopedGraphStore.new(store: pipeline.graph_store, scope: scope)
440
+ build_pipeline(vector_store: vector, metadata_store: scope.metadata_store, graph_store: graph)
441
+ end
442
+
443
+ def attach_scope(result, scope)
444
+ return result unless scope
445
+
446
+ result.applied_scope = scope.summary.merge(
447
+ outcome: if scope.keys.empty?
448
+ :empty_scope
449
+ else
450
+ (result.trace.ranked_count.zero? ? :no_match : :matched)
451
+ end,
452
+ candidate_count: result.trace.candidate_count, returned_units: result.sources.size
453
+ )
454
+ result
455
+ end
456
+
457
+ def retrieve_lexical(pipeline, query, classification, budget, types, exclude_types, start_time, **evidence_options)
458
+ excluded = DEFAULT_EXCLUDE_TYPES + Array(exclude_types).map(&:to_s)
459
+ execution = pipeline.executor.execute(query: query, type_filter: types, exclude_types: excluded)
460
+ assembled = pipeline.assembler.assemble(candidates: execution.candidates, budget: budget, **evidence_options)
461
+ result = build_result(assembled: assembled, assembler: pipeline.assembler, classification: classification,
462
+ strategy: :lexical, budget: budget)
463
+ result.trace = build_trace(result, execution, execution.candidates, assembled, start_time)
464
+ result
465
+ end
466
+
391
467
  # Validate +query+ before any classify/execute/rank work happens.
392
468
  #
393
469
  # @param query [Object] The caller-supplied query
@@ -464,23 +540,24 @@ module Woods
464
540
  # @param ranked [Array<Candidate>] Ranked search candidates
465
541
  # @param classification [QueryClassifier::Classification] Query classification
466
542
  # @return [AssembledContext]
467
- def assemble_context(pipeline, ranked, classification, budget)
543
+ def assemble_context(pipeline, ranked, classification, budget, **evidence_options)
468
544
  pipeline.assembler.assemble(
469
545
  candidates: ranked,
470
546
  classification: classification,
471
547
  structural_context: build_structural_context(pipeline.metadata_store),
472
- budget: budget
548
+ budget: budget, **evidence_options
473
549
  )
474
550
  end
475
551
 
476
552
  # Build a RetrievalResult from assembled context and pipeline metadata.
477
553
  #
478
554
  # @param assembled [AssembledContext] Assembled context
555
+ # @param assembler [Retrieval::ContextAssembler] Counter configuration used for assembly
479
556
  # @param classification [QueryClassifier::Classification] Query classification
480
557
  # @param strategy [Symbol] Search strategy used
481
558
  # @param budget [Integer] Token budget
482
559
  # @return [RetrievalResult]
483
- def build_result(assembled:, classification:, strategy:, budget:, trace: nil, type_rank_context: nil)
560
+ def build_result(assembled:, assembler:, classification:, strategy:, budget:, type_rank_context: nil)
484
561
  context = @formatter ? @formatter.call(assembled.context) : assembled.context
485
562
  context = append_type_rank_context(context, type_rank_context) if type_rank_context
486
563
 
@@ -489,9 +566,8 @@ module Woods
489
566
  sources: assembled.sources,
490
567
  classification: classification,
491
568
  strategy: strategy,
492
- tokens_used: assembled.tokens_used,
569
+ tokens_used: assembler.estimate_tokens(context),
493
570
  budget: budget,
494
- trace: trace,
495
571
  type_rank_context: type_rank_context
496
572
  )
497
573
  end
@@ -534,14 +610,14 @@ module Woods
534
610
  filter_by_type(pipeline, ranked, types: type_array, exclude_types: exclude_types)
535
611
  end
536
612
 
537
- def build_trace(classification, execution_result, filtered, assembled, start_time)
613
+ def build_trace(result, execution_result, filtered, assembled, start_time)
538
614
  elapsed_ms = ((Process.clock_gettime(Process::CLOCK_MONOTONIC) - start_time) * 1000).round(1)
539
615
  RetrievalTrace.new(
540
- classification: classification,
616
+ classification: result.classification,
541
617
  strategy: execution_result.strategy,
542
618
  candidate_count: execution_result.candidates.size,
543
619
  ranked_count: filtered.size,
544
- tokens_used: assembled.tokens_used,
620
+ tokens_used: result.tokens_used,
545
621
  elapsed_ms: elapsed_ms,
546
622
  skipped_missing_metadata: assembled.skipped_missing_metadata.to_i
547
623
  )
@@ -1,5 +1,7 @@
1
1
  # frozen_string_literal: true
2
2
 
3
+ require 'fiber'
4
+
3
5
  require_relative '../extracted_unit'
4
6
 
5
7
  module Woods
@@ -19,25 +21,26 @@ module Woods
19
21
  class TraceEnricher
20
22
  # Record method calls during block execution using TracePoint.
21
23
  #
24
+ # Records the current thread only, with independent stacks for its fibers.
25
+ # Caller fields identify the nearest observed Ruby method, not native or
26
+ # block frames. Calls entered before recording have an unknown caller.
27
+ #
22
28
  # @yield Block to trace
23
29
  # @return [Array<Hash>] Collected trace events
30
+ # @raise [ArgumentError] if no block is given
24
31
  def self.record(&block)
25
- traces = []
32
+ raise ArgumentError, 'block required' unless block
26
33
 
34
+ traces = []
35
+ stacks = Hash.new { |frames, fiber| frames[fiber] = [] }
27
36
  trace = TracePoint.new(:call, :return) do |tp|
28
- traces << {
29
- class_name: tp.defined_class&.name || tp.defined_class.to_s,
30
- method_name: tp.method_id.to_s,
31
- event: tp.event.to_s,
32
- path: tp.path,
33
- line: tp.lineno,
34
- caller_class: extract_caller_class(tp),
35
- caller_method: extract_caller_method(tp),
36
- return_class: tp.event == :return ? safe_return_class(tp) : nil
37
- }
37
+ fiber = Fiber.current
38
+ stack = stacks[fiber]
39
+ traces << record_event(tp, stack)
40
+ stacks.delete(fiber) if stack.empty?
38
41
  end
39
42
 
40
- trace.enable(&block)
43
+ trace.enable(target_thread: Thread.current, &block)
41
44
  traces
42
45
  end
43
46
 
@@ -52,14 +55,11 @@ module Woods
52
55
  def self.merge(units:, trace_data:)
53
56
  return units if trace_data.nil? || trace_data.empty?
54
57
 
55
- # Index traces by class_name + method_name
58
+ # Index by defining owner, method name, and instance/singleton kind.
56
59
  grouped = group_traces(trace_data)
57
60
 
58
61
  units.each do |unit|
59
- class_name, method_name = parse_identifier(unit.identifier)
60
- next unless class_name && method_name
61
-
62
- key = "#{class_name}##{method_name}"
62
+ key = parse_identifier(unit.identifier)
63
63
  next unless grouped.key?(key)
64
64
 
65
65
  traces = grouped[key]
@@ -72,7 +72,10 @@ module Woods
72
72
  caller_method = fetch_key(t, :caller_method)
73
73
  next unless caller_class
74
74
 
75
- { 'caller_class' => caller_class, 'caller_method' => caller_method }
75
+ caller = { 'caller_class' => caller_class, 'caller_method' => caller_method }
76
+ kind = fetch_key(t, :caller_method_kind)
77
+ caller['caller_method_kind'] = kind.to_s if kind
78
+ caller
76
79
  end
77
80
 
78
81
  return_types = returns.filter_map do |t|
@@ -101,35 +104,74 @@ module Woods
101
104
  method_name = fetch_key(trace, :method_name)
102
105
  next unless class_name && method_name
103
106
 
104
- key = "#{class_name}##{method_name}"
105
- grouped[key] << trace
107
+ # Legacy recorder output with a named owner describes instance
108
+ # methods. Never infer its kind from the units supplied to merge.
109
+ kind = (fetch_key(trace, :method_kind) || 'instance').to_s
110
+ next unless %w[instance singleton].include?(kind)
111
+ next if class_name.start_with?('#<')
112
+
113
+ grouped[[class_name, method_name, kind]] << trace
106
114
  end
107
115
  grouped
108
116
  end
109
117
 
110
118
  def parse_identifier(identifier)
111
- # Handle both "Class#method" and "Class.method" formats
112
- if identifier.include?('#')
113
- identifier.split('#', 2)
114
- elsif identifier.include?('.')
115
- identifier.split('.', 2)
116
- end
119
+ match = /\A(.+?)([#.])(.+)\z/.match(identifier)
120
+ return unless match
121
+
122
+ [match[1], match[3], match[2] == '#' ? 'instance' : 'singleton']
117
123
  end
118
124
 
119
- def extract_caller_class(tp)
120
- binding_obj = tp.binding
121
- receiver = binding_obj.receiver
122
- receiver.is_a?(Class) || receiver.is_a?(Module) ? receiver.name : receiver.class.name
123
- rescue StandardError
124
- nil
125
+ def event_identity(tp)
126
+ owner = tp.defined_class
127
+ singleton = owner&.singleton_class?
128
+ name = singleton ? singleton_owner_name(owner, tp.self) : owner&.name
129
+ { class_name: name, method_name: tp.method_id.to_s,
130
+ method_kind: singleton ? 'singleton' : 'instance' }
125
131
  end
126
132
 
127
- def extract_caller_method(_tp)
128
- # TracePoint doesn't directly expose caller method,
129
- # but we can get it from the call stack
130
- caller_locations(3, 1)&.first&.label
131
- rescue StandardError
132
- nil
133
+ # Ruby 3.0 has no Class#attached_object. Find the defining owner,
134
+ # not just the receiver: Child.run may be defined on Parent's singleton
135
+ # class. Singleton methods on individual objects have no named unit.
136
+ def singleton_owner_name(owner, receiver)
137
+ return unless receiver.is_a?(Module)
138
+
139
+ receiver.ancestors.find { |ancestor| ancestor.singleton_class.equal?(owner) }&.name
140
+ end
141
+
142
+ def record_event(tp, stack)
143
+ identity = event_identity(tp)
144
+ caller = if tp.event == :call
145
+ caller_fields(stack.last)
146
+ else
147
+ returning_caller(identity, stack)
148
+ end
149
+ event = identity.merge(
150
+ event: tp.event.to_s, path: tp.path, line: tp.lineno,
151
+ **caller, return_class: tp.event == :return ? safe_return_class(tp) : nil
152
+ )
153
+ stack << event if tp.event == :call
154
+ event
155
+ end
156
+
157
+ def caller_fields(frame)
158
+ frame = nil unless frame && frame[:class_name]
159
+ { caller_class: frame && frame[:class_name], caller_method: frame && frame[:method_name],
160
+ caller_method_kind: frame && frame[:method_kind] }
161
+ end
162
+
163
+ # Ruby emits :return during exceptional and nonlocal unwinds too. A
164
+ # return whose call predates recording has no known caller; discard an
165
+ # inconsistent stack instead of inventing an edge from unrelated frames.
166
+ def returning_caller(identity, stack)
167
+ frame = stack.pop
168
+ if frame && identity.all? { |key, value| frame[key] == value }
169
+ { caller_class: frame[:caller_class], caller_method: frame[:caller_method],
170
+ caller_method_kind: frame[:caller_method_kind] }
171
+ else
172
+ stack.clear
173
+ caller_fields(nil)
174
+ end
133
175
  end
134
176
 
135
177
  def safe_return_class(tp)
@@ -1,13 +1,14 @@
1
1
  # frozen_string_literal: true
2
2
 
3
3
  require 'time'
4
+ require 'active_support/inflector'
4
5
 
5
6
  module Woods
6
7
  module SessionTracer
7
8
  # Rack middleware that captures request metadata for session tracing.
8
9
  #
9
10
  # Wraps `@app.call(env)`, records after response. Extracts controller/action
10
- # from `env['action_dispatch.request.path_parameters']`. Session ID from
11
+ # from the dispatched controller instance and Rails path parameters. Session ID from
11
12
  # `X-Trace-Session` header first, falls back to `request.session.id`.
12
13
  #
13
14
  # Fire-and-forget writes — `rescue StandardError` on recording, never breaks the request.
@@ -98,8 +99,7 @@ module Woods
98
99
  action = path_params[:action]
99
100
  return unless controller
100
101
 
101
- # Classify controller name (e.g., "orders" -> "OrdersController")
102
- controller_class = classify_controller(controller)
102
+ controller_class = controller_identity(env['action_controller.instance'], controller)
103
103
 
104
104
  request_data = {
105
105
  'session_id' => session_id,
@@ -144,15 +144,13 @@ module Woods
144
144
  @exclude_paths.any? { |prefix| path.start_with?(prefix) }
145
145
  end
146
146
 
147
- # Classify a Rails controller path segment into a controller class name.
148
- #
149
- # @param controller [String] e.g., "orders" or "admin/orders"
150
- # @return [String] e.g., "OrdersController" or "Admin::OrdersController"
151
- def classify_controller(controller)
152
- parts = controller.to_s
153
- .split('/')
154
- .map { |segment| segment.split('_').map(&:capitalize).join }
155
- "#{parts.join('::')}Controller"
147
+ # Runtime identity is authoritative. Rack adapters without a dispatched
148
+ # instance still use Rails' configured acronyms when classifying the route.
149
+ def controller_identity(instance, controller)
150
+ name = instance.class.name if instance
151
+ return name if name && !name.empty?
152
+
153
+ "#{ActiveSupport::Inflector.camelize(controller.to_s)}Controller"
156
154
  end
157
155
 
158
156
  # Extract response format from the Rack env.
@@ -56,6 +56,19 @@ module Woods
56
56
  return 1
57
57
  LUA
58
58
 
59
+ # Read and remove against either index format without converting a
60
+ # legacy SET. Each type check and operation is atomic with respect to
61
+ # a concurrent record's migration. Reader-only upgrades therefore do
62
+ # not prevent older SET writers from continuing to record.
63
+ INDEX_ACCESS_SCRIPT = <<~LUA
64
+ local legacy = redis.call('TYPE', KEYS[1]).ok == 'set'
65
+ if ARGV[1] == 'members' then
66
+ if legacy then return redis.call('SMEMBERS', KEYS[1]) end
67
+ return redis.call('ZRANGE', KEYS[1], 0, -1)
68
+ end
69
+ return redis.call(legacy and 'SREM' or 'ZREM', KEYS[1], ARGV[2])
70
+ LUA
71
+
59
72
  # @param redis [Redis] A Redis client instance
60
73
  # @param ttl [Integer, nil] Time-to-live in seconds for session keys (nil = no expiry)
61
74
  def initialize(redis:, ttl: nil, max_sessions: DEFAULT_MAX_SESSIONS,
@@ -103,14 +116,13 @@ module Woods
103
116
  # @param limit [Integer] Maximum number of sessions to return
104
117
  # @return [Array<Hash>] Session summaries
105
118
  def sessions(limit: 20)
106
- all_ids = @redis.zrange(SESSIONS_KEY, 0, -1)
119
+ all_ids = index_access('members')
107
120
 
108
121
  # Filter to sessions that still have data (TTL may have expired)
109
122
  active = all_ids.select { |id| @redis.exists?(session_key(id)) }
110
123
 
111
124
  # Remove expired session IDs from the index
112
- expired = all_ids - active
113
- expired.each { |id| @redis.zrem(SESSIONS_KEY, id) } if expired.any?
125
+ (all_ids - active).each { |id| index_access('remove', id) }
114
126
 
115
127
  # Redis sorted sets order by score, but a summary's last_request is
116
128
  # the payload timestamp of the session's last record — the same
@@ -133,20 +145,24 @@ module Woods
133
145
  # @return [void]
134
146
  def clear(session_id)
135
147
  @redis.del(session_key(session_id))
136
- @redis.zrem(SESSIONS_KEY, session_id)
148
+ index_access('remove', session_id)
137
149
  end
138
150
 
139
151
  # Remove all session data.
140
152
  #
141
153
  # @return [void]
142
154
  def clear_all
143
- all_ids = @redis.zrange(SESSIONS_KEY, 0, -1)
155
+ all_ids = index_access('members')
144
156
  all_ids.each { |id| @redis.del(session_key(id)) }
145
157
  @redis.del(SESSIONS_KEY)
146
158
  end
147
159
 
148
160
  private
149
161
 
162
+ def index_access(action, session_id = nil)
163
+ @redis.eval(INDEX_ACCESS_SCRIPT, keys: [SESSIONS_KEY], argv: [action, *session_id])
164
+ end
165
+
150
166
  # @param session_id [String]
151
167
  # @return [String] Redis key for this session
152
168
  def session_key(session_id)
@@ -196,7 +212,7 @@ module Woods
196
212
 
197
213
  victims.each do |id|
198
214
  @redis.del(session_key(id))
199
- @redis.zrem(SESSIONS_KEY, id)
215
+ index_access('remove', id)
200
216
  end
201
217
  end
202
218
 
@@ -4,6 +4,7 @@ require 'json'
4
4
  require 'set'
5
5
  require_relative '../token_utils'
6
6
  require_relative 'session_flow_document'
7
+ require_relative 'unit_resolver'
7
8
 
8
9
  module Woods
9
10
  module SessionTracer
@@ -42,8 +43,14 @@ module Woods
42
43
  # @param budget [Integer] Maximum token budget (default: 8000)
43
44
  # @param depth [Integer] Expansion depth (0=metadata only, 1=direct deps, 2+=full flow)
44
45
  # @return [SessionFlowDocument] The assembled document
45
- # rubocop:disable-next Metrics/AbcSize, Metrics/CyclomaticComplexity, Metrics/MethodLength
46
46
  def assemble(session_id, budget: 8000, depth: 1)
47
+ @reader.with_pinned_generation { assemble_pinned(session_id, budget: budget, depth: depth) }
48
+ end
49
+
50
+ private
51
+
52
+ # rubocop:disable-next Metrics/AbcSize, Metrics/CyclomaticComplexity, Metrics/MethodLength
53
+ def assemble_pinned(session_id, budget:, depth:)
47
54
  requests = @store.read(session_id)
48
55
  return empty_document(session_id) if requests.empty?
49
56
 
@@ -52,6 +59,7 @@ module Woods
52
59
  side_effects = []
53
60
  dependency_map = {}
54
61
  seen_units = Set.new
62
+ resolver = UnitResolver.new(@reader)
55
63
 
56
64
  requests.each_with_index do |req, idx|
57
65
  step = build_step(req, idx)
@@ -63,18 +71,17 @@ module Woods
63
71
  next unless controller_id
64
72
 
65
73
  # Resolve controller unit
66
- unit = @reader.find_unit(controller_id)
74
+ unit = @reader.find_unit(controller_id, type: 'controller')
67
75
  if unit && !seen_units.include?(controller_id)
68
76
  seen_units.add(controller_id)
69
77
  context_pool[controller_id] = unit_summary(unit)
70
78
  end
71
- step[:unit_refs] = [controller_id].compact
72
-
73
- # Expand dependencies
79
+ # Expand dependencies only for a resolved controller.
74
80
  next unless unit
75
81
 
82
+ step[:unit_refs] = [controller_id]
76
83
  deps = resolve_dependencies(controller_id, seen_units, context_pool,
77
- side_effects, step, dependency_map, depth)
84
+ side_effects, step, dependency_map, depth, resolver)
78
85
  step[:unit_refs].concat(deps)
79
86
  end
80
87
 
@@ -85,8 +92,6 @@ module Woods
85
92
  budgeted_document(parts, budget)
86
93
  end
87
94
 
88
- private
89
-
90
95
  # Build a timeline step from a request record.
91
96
  #
92
97
  # @param req [Hash] Request data from store
@@ -111,13 +116,13 @@ module Woods
111
116
  # @return [Array<String>] Non-async dependency identifiers added
112
117
  # rubocop:disable-next Metrics/AbcSize, Metrics/CyclomaticComplexity, Metrics/MethodLength, Metrics/ParameterLists, Metrics/PerceivedComplexity
113
118
  def resolve_dependencies(unit_id, seen_units, context_pool,
114
- side_effects, step, dependency_map, depth)
119
+ side_effects, step, dependency_map, depth, resolver)
115
120
  graph = @reader.dependency_graph
116
- dep_ids = graph.dependencies_of(unit_id)
121
+ dep_ids = graph.dependencies_of(unit_id, type: :controller)
117
122
  added = []
118
123
 
119
124
  dep_ids.each do |dep_id|
120
- dep_unit = @reader.find_unit(dep_id)
125
+ dep_unit = resolver.find(dep_id)
121
126
  next unless dep_unit
122
127
 
123
128
  dep_type = dep_unit['type']&.to_s
@@ -137,13 +142,13 @@ module Woods
137
142
  added << dep_id
138
143
 
139
144
  # Depth 2+: expand transitive dependencies
140
- expand_transitive(dep_id, seen_units, context_pool, dependency_map, depth - 1) if depth >= 2
145
+ expand_transitive(dep_id, seen_units, context_pool, dependency_map, depth - 1, resolver) if depth >= 2
141
146
  end
142
147
  end
143
148
  end
144
149
 
145
150
  # Record dependency map for this unit
146
- all_deps = dep_ids.select { |id| @reader.find_unit(id) }
151
+ all_deps = dep_ids.select { |id| resolver.find(id) }
147
152
  dependency_map[unit_id] = all_deps if all_deps.any?
148
153
 
149
154
  added
@@ -156,15 +161,16 @@ module Woods
156
161
  # @param context_pool [Hash] Accumulator for unit data
157
162
  # @param dependency_map [Hash] Accumulator for dependency edges
158
163
  # @param remaining_depth [Integer] Remaining expansion depth
159
- def expand_transitive(unit_id, seen_units, context_pool, dependency_map, remaining_depth)
164
+ # rubocop:disable-next Metrics/ParameterLists
165
+ def expand_transitive(unit_id, seen_units, context_pool, dependency_map, remaining_depth, resolver)
160
166
  return if remaining_depth <= 0
161
167
 
162
168
  graph = @reader.dependency_graph
163
- dep_ids = graph.dependencies_of(unit_id)
169
+ dep_ids = graph.dependencies_of(unit_id, type: resolver.find(unit_id).fetch('type').to_sym)
164
170
  resolved_deps = []
165
171
 
166
172
  dep_ids.each do |dep_id|
167
- dep_unit = @reader.find_unit(dep_id)
173
+ dep_unit = resolver.find(dep_id)
168
174
  next unless dep_unit
169
175
 
170
176
  resolved_deps << dep_id
@@ -173,7 +179,7 @@ module Woods
173
179
  seen_units.add(dep_id)
174
180
  context_pool[dep_id] = unit_summary(dep_unit)
175
181
 
176
- expand_transitive(dep_id, seen_units, context_pool, dependency_map, remaining_depth - 1)
182
+ expand_transitive(dep_id, seen_units, context_pool, dependency_map, remaining_depth - 1, resolver)
177
183
  end
178
184
 
179
185
  dependency_map[unit_id] = resolved_deps if resolved_deps.any?
@@ -157,11 +157,13 @@ module Woods
157
157
  end
158
158
 
159
159
  def inserted?(entry_class, result, key, payload)
160
- affected_rows = result.affected_rows if result.respond_to?(:affected_rows)
161
- return affected_rows.positive? unless affected_rows.nil?
162
- return result.rows.any? if entry_class.connection.supports_insert_returning?
160
+ # MySQL's FOUND_ROWS can count a duplicate no-op as an affected row.
161
+ # The per-attempt CAS version makes equal values distinct payloads, so
162
+ # only our exact stored bytes establish ownership without RETURNING.
163
+ return entry_class.read(key) == payload unless entry_class.connection.supports_insert_returning?
163
164
 
164
- entry_class.read(key) == payload
165
+ affected_rows = result.affected_rows if result.respond_to?(:affected_rows)
166
+ affected_rows.nil? ? result.rows.any? : affected_rows.positive?
165
167
  end
166
168
 
167
169
  def routed_read(key, failsafe)