woods 2.0.0.beta2 → 2.0.0.beta3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (218) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +262 -1
  3. data/CONTRIBUTING.md +173 -9
  4. data/README.md +7 -3
  5. data/SECURITY.md +9 -6
  6. data/docs/AGENT_GUIDE.md +83 -4
  7. data/docs/AGENT_SETUP.md +82 -1
  8. data/docs/BACKEND_MATRIX.md +20 -0
  9. data/docs/CLIENT_HOOKS.md +111 -0
  10. data/docs/CONFIGURATION_REFERENCE.md +199 -14
  11. data/docs/CONSOLE_MCP_SETUP.md +35 -5
  12. data/docs/DOCKER_SETUP.md +21 -2
  13. data/docs/EVALUATION.md +464 -1
  14. data/docs/EXTRACTOR_REFERENCE.md +36 -5
  15. data/docs/FAQ.md +11 -12
  16. data/docs/GETTING_STARTED.md +17 -5
  17. data/docs/INCREMENTAL_EXTRACTION.md +117 -1
  18. data/docs/INDEX_LAYOUT.md +382 -0
  19. data/docs/INTERNALS.md +7 -2
  20. data/docs/MCP_SERVERS.md +221 -5
  21. data/docs/MCP_TOOL_COOKBOOK.md +33 -18
  22. data/docs/NOTION_INTEGRATION.md +13 -0
  23. data/docs/OBSIDIAN_INTEGRATION.md +57 -9
  24. data/docs/PUBLISHED_INDEX.md +55 -0
  25. data/docs/README.md +7 -0
  26. data/docs/RETRIEVAL_GUIDE.md +253 -11
  27. data/docs/RUNTIME_TRACING.md +71 -0
  28. data/docs/SOURCE_FRESHNESS.md +143 -0
  29. data/docs/TROUBLESHOOTING.md +117 -5
  30. data/docs/UNBLOCKED_INTEGRATION.md +25 -0
  31. data/docs/UPGRADING_TO_2.md +44 -22
  32. data/docs/WATCH_DAEMON.md +259 -59
  33. data/exe/woods-agent-config +6 -0
  34. data/exe/woods-extract +5 -0
  35. data/exe/woods-hook-context +6 -0
  36. data/lib/generators/woods/templates/woods.rb.tt +1 -3
  37. data/lib/tasks/woods.rake +47 -397
  38. data/lib/woods/agent_configuration/applier.rb +133 -0
  39. data/lib/woods/agent_configuration/cli.rb +101 -0
  40. data/lib/woods/agent_configuration/cli_options.rb +29 -0
  41. data/lib/woods/agent_configuration/document.rb +105 -0
  42. data/lib/woods/agent_configuration/error.rb +7 -0
  43. data/lib/woods/agent_configuration/launcher.rb +75 -0
  44. data/lib/woods/agent_configuration/layout.rb +59 -0
  45. data/lib/woods/agent_configuration/managed_section.rb +62 -0
  46. data/lib/woods/agent_configuration/plan.rb +98 -0
  47. data/lib/woods/agent_configuration/plan_diff.rb +38 -0
  48. data/lib/woods/agent_configuration/planned_files.rb +61 -0
  49. data/lib/woods/agent_configuration/planner.rb +63 -0
  50. data/lib/woods/agent_configuration/planner_validation.rb +77 -0
  51. data/lib/woods/agent_configuration/preflight.rb +100 -0
  52. data/lib/woods/agent_configuration/recovery.rb +49 -0
  53. data/lib/woods/ast/node.rb +2 -0
  54. data/lib/woods/ast/parser.rb +38 -5
  55. data/lib/woods/builder.rb +21 -5
  56. data/lib/woods/cache/cache_middleware.rb +28 -7
  57. data/lib/woods/cache/cache_store.rb +4 -5
  58. data/lib/woods/change_set.rb +5 -4
  59. data/lib/woods/console/credential_index.rb +20 -2
  60. data/lib/woods/console/credential_scanner.rb +14 -14
  61. data/lib/woods/console/credential_scanner_registry.rb +36 -0
  62. data/lib/woods/console/embedded_executor.rb +1 -1
  63. data/lib/woods/console/encrypted_credential_snapshot.rb +16 -0
  64. data/lib/woods/console/rack_middleware.rb +22 -13
  65. data/lib/woods/console/server.rb +18 -16
  66. data/lib/woods/dependency_graph.rb +65 -13
  67. data/lib/woods/embedding/corpus.rb +94 -0
  68. data/lib/woods/embedding/indexer.rb +90 -46
  69. data/lib/woods/embedding/openai.rb +17 -6
  70. data/lib/woods/evaluation/ablation_executor.rb +6 -1
  71. data/lib/woods/evaluation/ablation_timed_executor.rb +22 -4
  72. data/lib/woods/export/typed_reader.rb +56 -0
  73. data/lib/woods/extractor.rb +232 -137
  74. data/lib/woods/extractors/action_cable_extractor.rb +3 -1
  75. data/lib/woods/extractors/behavioral_profile.rb +9 -7
  76. data/lib/woods/extractors/caching_extractor.rb +3 -1
  77. data/lib/woods/extractors/concern_extractor.rb +64 -6
  78. data/lib/woods/extractors/configuration_extractor.rb +7 -3
  79. data/lib/woods/extractors/controller_extractor.rb +13 -4
  80. data/lib/woods/extractors/database_view_extractor.rb +3 -1
  81. data/lib/woods/extractors/decorator_extractor.rb +3 -1
  82. data/lib/woods/extractors/engine_extractor.rb +3 -1
  83. data/lib/woods/extractors/event_extractor.rb +4 -2
  84. data/lib/woods/extractors/factory_extractor.rb +3 -1
  85. data/lib/woods/extractors/graphql_extractor.rb +8 -2
  86. data/lib/woods/extractors/i18n_extractor.rb +3 -1
  87. data/lib/woods/extractors/job_extractor.rb +6 -19
  88. data/lib/woods/extractors/lib_extractor.rb +3 -1
  89. data/lib/woods/extractors/mailer_extractor.rb +20 -5
  90. data/lib/woods/extractors/manager_extractor.rb +3 -1
  91. data/lib/woods/extractors/method_parameters.rb +53 -0
  92. data/lib/woods/extractors/middleware_argument.rb +65 -0
  93. data/lib/woods/extractors/middleware_extractor.rb +9 -3
  94. data/lib/woods/extractors/migration_extractor.rb +3 -1
  95. data/lib/woods/extractors/model_extractor.rb +39 -33
  96. data/lib/woods/extractors/package_extractor.rb +24 -4
  97. data/lib/woods/extractors/phlex_extractor.rb +3 -1
  98. data/lib/woods/extractors/policy_extractor.rb +3 -1
  99. data/lib/woods/extractors/poro_extractor.rb +3 -1
  100. data/lib/woods/extractors/pundit_extractor.rb +3 -1
  101. data/lib/woods/extractors/rails_source_extractor.rb +4 -2
  102. data/lib/woods/extractors/rake_task_extractor.rb +4 -2
  103. data/lib/woods/extractors/route_extractor.rb +3 -1
  104. data/lib/woods/extractors/route_helper_resolver.rb +10 -33
  105. data/lib/woods/extractors/scheduled_job_extractor.rb +41 -15
  106. data/lib/woods/extractors/serializer_extractor.rb +4 -2
  107. data/lib/woods/extractors/service_extractor.rb +3 -1
  108. data/lib/woods/extractors/shared_dependency_scanner.rb +2 -2
  109. data/lib/woods/extractors/shared_utility_methods.rb +27 -15
  110. data/lib/woods/extractors/source_nesting.rb +1 -1
  111. data/lib/woods/extractors/state_machine_extractor.rb +3 -1
  112. data/lib/woods/extractors/test_mapping_extractor.rb +3 -1
  113. data/lib/woods/extractors/validator_extractor.rb +3 -1
  114. data/lib/woods/extractors/view_component_extractor.rb +3 -1
  115. data/lib/woods/extractors/view_template_extractor.rb +3 -1
  116. data/lib/woods/gem_mapper.rb +2 -0
  117. data/lib/woods/git_history.rb +116 -0
  118. data/lib/woods/graph_analyzer.rb +35 -6
  119. data/lib/woods/hooks/context_cli.rb +54 -0
  120. data/lib/woods/hooks/context_event.rb +88 -0
  121. data/lib/woods/hooks/context_hint.rb +73 -0
  122. data/lib/woods/hooks/context_impact.rb +77 -0
  123. data/lib/woods/hooks/context_output.rb +47 -0
  124. data/lib/woods/hooks/context_state.rb +102 -0
  125. data/lib/woods/hooks/refresh.rb +79 -0
  126. data/lib/woods/hooks/rule_projection.rb +78 -0
  127. data/lib/woods/input_rules.rb +19 -0
  128. data/lib/woods/mcp/bearer_auth.rb +20 -12
  129. data/lib/woods/mcp/bootstrapper.rb +62 -0
  130. data/lib/woods/mcp/index_reader.rb +323 -160
  131. data/lib/woods/mcp/initialization_guidance.rb +27 -0
  132. data/lib/woods/mcp/origin_guard.rb +17 -9
  133. data/lib/woods/mcp/published_lexical_retriever.rb +115 -0
  134. data/lib/woods/mcp/renderers/markdown_renderer.rb +8 -1
  135. data/lib/woods/mcp/renderers/plain_renderer.rb +7 -1
  136. data/lib/woods/mcp/search_results.rb +74 -0
  137. data/lib/woods/mcp/server.rb +158 -37
  138. data/lib/woods/mcp/tool_contract.rb +2 -0
  139. data/lib/woods/mcp/tool_response_renderer.rb +25 -0
  140. data/lib/woods/mcp/traversal_evidence.rb +113 -0
  141. data/lib/woods/mcp/traversal_evidence_index.rb +100 -0
  142. data/lib/woods/mcp/traversal_evidence_page.rb +41 -0
  143. data/lib/woods/mcp/traversal_evidence_text.rb +52 -0
  144. data/lib/woods/notion/exporter.rb +56 -17
  145. data/lib/woods/obsidian/destination_plan.rb +98 -0
  146. data/lib/woods/obsidian/name_mapper.rb +19 -3
  147. data/lib/woods/obsidian/note_builder.rb +19 -10
  148. data/lib/woods/obsidian/vault_exporter.rb +88 -32
  149. data/lib/woods/operator/pipeline_guard.rb +18 -13
  150. data/lib/woods/path_dispatcher.rb +7 -1
  151. data/lib/woods/payload_store.rb +27 -26
  152. data/lib/woods/railtie.rb +3 -3
  153. data/lib/woods/railtie_support.rb +12 -12
  154. data/lib/woods/rake_helpers.rb +392 -0
  155. data/lib/woods/resilience/graph_invariant_validator/membership_checks.rb +71 -0
  156. data/lib/woods/resilience/graph_invariant_validator/node_checks.rb +61 -0
  157. data/lib/woods/resilience/graph_invariant_validator/reverse_relationship_checks.rb +46 -0
  158. data/lib/woods/resilience/graph_invariant_validator.rb +119 -0
  159. data/lib/woods/resilience/index_validator/graph_checks.rb +80 -0
  160. data/lib/woods/resilience/index_validator.rb +112 -23
  161. data/lib/woods/retrieval/context_assembler.rb +50 -15
  162. data/lib/woods/retrieval/lexical_assembler.rb +73 -0
  163. data/lib/woods/retrieval/lexical_index.rb +119 -0
  164. data/lib/woods/retrieval/ranker.rb +4 -2
  165. data/lib/woods/retrieval/scope.rb +108 -0
  166. data/lib/woods/retrieval/scoped_graph_store.rb +32 -0
  167. data/lib/woods/retrieval/scoped_vector_store.rb +55 -0
  168. data/lib/woods/retrieval/search_executor.rb +86 -27
  169. data/lib/woods/retrieval/source_evidence.rb +200 -0
  170. data/lib/woods/retriever.rb +98 -22
  171. data/lib/woods/ruby_analyzer/trace_enricher.rb +77 -38
  172. data/lib/woods/session_tracer/middleware.rb +10 -12
  173. data/lib/woods/session_tracer/redis_store.rb +22 -6
  174. data/lib/woods/session_tracer/session_flow_assembler.rb +23 -17
  175. data/lib/woods/session_tracer/solid_cache_coordination.rb +6 -4
  176. data/lib/woods/session_tracer/unit_resolver.rb +63 -0
  177. data/lib/woods/source_inputs/consumer_errors.rb +27 -0
  178. data/lib/woods/source_inputs/handoff.rb +102 -0
  179. data/lib/woods/source_inputs/launcher.rb +157 -0
  180. data/lib/woods/source_inputs/manifest.rb +124 -0
  181. data/lib/woods/source_inputs/private_key.rb +55 -0
  182. data/lib/woods/source_inputs/scanner.rb +171 -0
  183. data/lib/woods/source_inputs/scopes.rb +71 -0
  184. data/lib/woods/source_inputs/session.rb +214 -0
  185. data/lib/woods/source_inputs/status.rb +84 -0
  186. data/lib/woods/source_inputs/verifier.rb +107 -0
  187. data/lib/woods/storage/metadata_store.rb +25 -25
  188. data/lib/woods/storage/pgvector.rb +29 -8
  189. data/lib/woods/storage/qdrant.rb +17 -7
  190. data/lib/woods/storage/vector_store.rb +18 -6
  191. data/lib/woods/tasks.rb +3 -2
  192. data/lib/woods/temporal/json_snapshot_store.rb +29 -8
  193. data/lib/woods/unblocked/exporter.rb +59 -70
  194. data/lib/woods/version.rb +1 -1
  195. data/lib/woods/watch/boot_snapshot.rb +52 -0
  196. data/lib/woods/watch/daemon.rb +136 -28
  197. data/lib/woods/watch/listen_watcher.rb +4 -0
  198. data/lib/woods/watch/polling_watcher.rb +5 -1
  199. data/lib/woods/watch/status.rb +20 -15
  200. data/lib/woods/watch/tree_scan.rb +21 -13
  201. data/lib/woods/watch/watcher.rb +4 -1
  202. data/lib/woods.rb +50 -11
  203. data/plugin/.claude-plugin/plugin.json +1 -1
  204. data/plugin/hooks/adapters/normalize.jq +15 -0
  205. data/plugin/hooks/adapters/normalize.rb +63 -0
  206. data/plugin/hooks/hooks.json +20 -0
  207. data/plugin/hooks/woods-context.sh +50 -0
  208. data/plugin/hooks/woods-input-rules.sh +159 -0
  209. data/plugin/hooks/woods-opencode.mjs +65 -0
  210. data/plugin/hooks/woods-post-edit.sh +2 -225
  211. data/plugin/hooks/woods-refresh.sh +260 -0
  212. data/plugin/hooks/woods-session-start.sh +47 -55
  213. data/plugin/skills/woods-agent-enable/SKILL.md +13 -0
  214. data/plugin/skills/woods-diagnose/SKILL.md +288 -1
  215. data/plugin/skills/woods-investigate/SKILL.md +106 -0
  216. data/plugin/skills/woods-mcp-config/SKILL.md +89 -1
  217. data/plugin/skills/woods-setup/SKILL.md +107 -6
  218. metadata +84 -5
@@ -26,6 +26,16 @@ query
26
26
  | Assembly | `Woods::Retrieval::ContextAssembler` | Fills a token budget with ranked units, sectioned into structural / primary / supporting / framework blocks |
27
27
  | Orchestration | `Woods::Retriever` | Coordinates all four stages; returns a `RetrievalResult` with `context`, `sources`, `strategy`, `tokens_used`, and `trace` |
28
28
 
29
+ ### PageRank importance
30
+
31
+ When a graph store supplies PageRank, ranking converts its scores to ordinal
32
+ percentiles: the highest-ranked unit receives 1.0 and the lowest receives 1/n,
33
+ where n is the number of entries in the PageRank map.
34
+ Equal PageRank scores are ordered lexically by identifier before assigning
35
+ percentiles. This preserves distinct ordinal weights while making ties independent
36
+ of graph insertion order and Ruby version. It does not assign equal importance
37
+ weights to tied units. Missing graph entries retain the metadata-based fallback.
38
+
29
39
  ### Search strategies
30
40
 
31
41
  `SearchExecutor` selects one of five strategies based on query classification:
@@ -38,6 +48,15 @@ query
38
48
  | `:hybrid` | `comprehensive` or `exploratory` scope | Runs vector + keyword + graph expansion, deduplicates |
39
49
  | `:direct` | `locate`/`reference` + `pinpoint` scope | Looks up identifiers directly in metadata store; falls back to keyword |
40
50
 
51
+ Graph queries preserve identifier spelling and namespaces: `trace Billing::Invoice`,
52
+ `trace ReviewAssignment`, and `trace review_assignment` resolve the subject before
53
+ traversing its dependencies and dependents. An exact identifier takes precedence;
54
+ an unqualified name without an exact match can seed up to three namespace matches
55
+ in storage-identifier order. Use the qualified name to disambiguate. A missing
56
+ qualified identifier does not fall back to a different namespace. For unqualified
57
+ prose, metadata fallback searches subject terms without instructions such as
58
+ `trace`, `follow`, or `who calls`.
59
+
41
60
  Keyword results are scored by how many distinct fields matched (identifier, source, metadata), not by the store's result order. Each matched field adds 0.25, capped at 1.0, so a result matching on identifier and source scores higher than one matching source alone.
42
61
 
43
62
  ---
@@ -124,13 +143,80 @@ bundle exec rake woods:extract
124
143
  bundle exec rake woods:embed
125
144
  ```
126
145
 
146
+ ### Input integrity and source-empty units
147
+
148
+ Embedding validates a native published generation's manifest, type listings and
149
+ unit payloads before changing vector storage, metadata or checkpoints. A missing,
150
+ corrupt or mismatched listed unit raises `Embedding input incomplete`; repair the
151
+ extraction (or run a full `woods:extract`) before retrying embedding.
152
+ `WOODS_ALLOW_PURGE=1` permits intentional mass deletion; it does not bypass this
153
+ integrity check. Corpus reads hold one generation pin while collecting the input.
154
+
155
+ Legacy flat indexes retain arbitrary unit filenames and do not require a manifest
156
+ or type listing. A `payloads/` directory requires a valid publication pointer;
157
+ missing or null pointers refuse embedding even when stale flat files remain.
158
+ Malformed JSON now refuses embedding instead of silently dropping
159
+ that file. Without an authoritative listing, a missing legacy file still denotes
160
+ a deletion; regenerate extraction into the native publication layout for stronger
161
+ completeness checks.
162
+
163
+ When a current unit has no source text to embed, Woods retains its metadata and
164
+ removes its superseded vectors, including old chunks. It records the current
165
+ source hash without calling the provider; later incremental runs verify that the
166
+ unit still prepares no text before accepting a matching no-vector checkpoint.
167
+ Empty-vector reconciliation waits until all batches succeed, so a later provider
168
+ failure does not retire those old vectors.
169
+
170
+ ---
171
+
172
+ ## Embedding-free lexical retrieval
173
+
174
+ Opt into ranked retrieval over an extract-only published index:
175
+
176
+ ```bash
177
+ WOODS_RETRIEVAL_MODE=lexical bundle exec woods-mcp-start ./tmp/woods
178
+ ```
179
+
180
+ For a Ruby-built retriever, set `config.retrieval_mode = :lexical` and supply a
181
+ populated metadata store to `Builder#build_retriever`. The packaged MCP server
182
+ loads published unit JSON itself; it does not boot Rails or read current source
183
+ files. Neither path constructs an embedding provider or vector adapter. The
184
+ existing `:semantic` mode remains the default; provider failures never switch
185
+ modes automatically. A Rails initializer is not loaded by the standalone MCP
186
+ process, so set the environment variable in that process's client configuration.
187
+
188
+ Lexical results use field-aware BM25 scoring over identifiers, source paths,
189
+ published source and selected runtime metadata (including callbacks, associations
190
+ and validations). Exact full identifiers rank first, with ambiguous typed owners
191
+ retained. Other ties are deterministic. Responses name the lexical mode and
192
+ matching fields/terms; runtime-field hits include the selected published runtime
193
+ values. Lexical Ruby results leave the semantic-only `type_rank_context` table
194
+ `nil`; they do not report a global vector rank or vector fallback. The top 20 eligible positive matches are considered for the
195
+ output budget; this is ranked discovery, not an exhaustive match listing. Explicit
196
+ `types` filters override default exclusions, as in semantic retrieval, and apply
197
+ before that limit. A query with no lexical evidence returns no matches; unrelated
198
+ graph hubs are never added. Query-seeded graph ranking is evaluation-only.
199
+
200
+ The budget covers headers, matching explanations and truncation notices using a
201
+ labelled character-based estimate, not an exact provider tokenizer. Full source
202
+ remains available through `lookup`. Very small budgets can omit all sources. The
203
+ reader pins one published generation for building and querying its immutable
204
+ lexical snapshot, rebuilding after publication. Corrupt units fail explicitly;
205
+ they cannot quietly become a successful partial index. Older flat indexes rebuild
206
+ on each query because they lack an immutable generation identity.
207
+
208
+ See [evaluation](EVALUATION.md) for measured query coverage and limits. Lexical
209
+ matching cannot infer synonyms absent from the published text; a miss is not proof
210
+ that application behavior is absent. Static Woods self-maps can opt into the same
211
+ mode but remain static source maps, not resolved Rails runtime evidence.
212
+
127
213
  ---
128
214
 
129
215
  ## Running Retrieval
130
216
 
131
217
  ### MCP tool: `codebase_retrieve`
132
218
 
133
- The primary interface for agents. Available in the Index Server when an embedding provider is configured and `rake woods:embed` has been run.
219
+ The primary interface for agents. Available with explicit lexical mode over extraction output, or with an embedding provider and completed `rake woods:embed` in the default semantic mode.
134
220
 
135
221
  ```
136
222
  codebase_retrieve(query: "how does billing work?")
@@ -165,6 +251,12 @@ result.sources # => [{ identifier: "User", type: "model", score: 0.91, ...
165
251
  result.trace # => RetrievalTrace with elapsed_ms, candidate_count, etc.
166
252
  ```
167
253
 
254
+ `result.tokens_used` and `result.trace.tokens_used` count the final returned
255
+ `context`, including the optional formatter output and type-rank table. Counting
256
+ uses the same injected token counter or chars-per-token estimate as assembly;
257
+ it does not guarantee an exact count for the downstream model. `budget` limits
258
+ context assembly, so postprocessing can make the final count exceed it.
259
+
168
260
  Override the token budget per call:
169
261
 
170
262
  ```ruby
@@ -201,18 +293,29 @@ In all cases, errors in individual components produce empty candidate sets for t
201
293
 
202
294
  ### `similarity_threshold`
203
295
 
204
- Controls which vector search results are considered. Range: `0.0`–`1.0`. Default: `0.7`.
205
-
206
- ```ruby
207
- config.similarity_threshold = 0.6 # Include less similar results (broader)
208
- config.similarity_threshold = 0.8 # Require higher similarity (narrower)
209
- ```
210
-
211
- Lower values return more candidates, which can improve recall for broad queries at the cost of precision. Raise it if results seem loosely related.
296
+ Deprecated and inert. Woods retains numeric validation (`0.0`–`1.0`) and
297
+ readback for compatibility, but setting this option emits a warning. It has not
298
+ filtered retrieval results; changing it does not change scores or candidates.
299
+ Use explicit type, package or source-path scopes to select eligible units, and
300
+ inspect returned ranking evidence to assess relevance. No new score cutoff is
301
+ introduced by this deprecation.
212
302
 
213
303
  ### `max_context_tokens`
214
304
 
215
- Sets the default token budget for context assembly. Default: `8000`. The `budget` parameter on `codebase_retrieve` and `Retriever#retrieve` overrides this per call.
305
+ Sets the default token budget for context assembly. Default: `8000`. Builder
306
+ captures this value when constructing semantic or lexical retrievers, including
307
+ cached retrievers. The `budget` parameter on `codebase_retrieve` and
308
+ `Retriever#retrieve` overrides it per call; omitted and explicitly equal budgets
309
+ share a cache entry. Changing configuration affects newly constructed retrievers.
310
+ Direct `Retriever.new` callers can pass `default_budget:`; otherwise they retain
311
+ 8000. Custom MCP collaborators without `default_budget` also retain the legacy
312
+ 8000 fallback.
313
+
314
+ The setting applies to the serving process's configuration. It is not embedded
315
+ in `woods.json`; a standalone MCP process that does not load the host initializer
316
+ uses its own default unless a tool call supplies `budget`. Token accounting uses
317
+ the configured estimator, and very small budgets can still include formatting
318
+ overhead.
216
319
 
217
320
  ```ruby
218
321
  config.max_context_tokens = 12_000 # More context per retrieval
@@ -264,4 +367,143 @@ bundle exec rake woods:embed
264
367
  | Empty results for a known class name | Keyword strategy not finding the identifier | Try a conceptual query with `codebase_retrieve`; or use `search` for exact name lookup |
265
368
  | Very slow retrieval | Large vector index without HNSW index, or Qdrant cold start | For pgvector: create an HNSW index (see `BACKEND_MATRIX.md`). For Qdrant: check collection status |
266
369
  | `codebase_retrieve` tool listed but disabled | Embedding provider not configured or API key missing | Set `embedding_provider`, run `woods:embed`, and check `woods_status` |
267
- | Results clustered around one type | Diversity penalty insufficient for codebase shape | Lower `similarity_threshold` slightly and widen the query scope |
370
+ | Results clustered around one type | Diversity penalty insufficient for codebase shape | Use explicit type filters or a more specific query; inspect ranking evidence |
371
+
372
+ ## Explicit package and source-path scopes
373
+
374
+ Both `codebase_retrieve` and `search` accept `packages` and `source_paths` arrays.
375
+ The Ruby API uses the same keyword arguments:
376
+
377
+ ```ruby
378
+ retriever.retrieve('How are payments collected?', budget: 1200,
379
+ packages: ['packs/billing'], source_paths: ['packs/billing/app'])
380
+ ```
381
+
382
+ These filters select eligible published units **before candidate limits**. Each
383
+ list uses OR; package, path, and type restrictions combine with AND. Empty or
384
+ omitted scope lists add no restriction. Existing unscoped calls keep their behavior.
385
+
386
+ - **Packages:** exact, case-sensitive published nearest owners. A parent package
387
+ excludes its nested packages unless both names are requested. `.` selects only
388
+ root-owned units. Missing ownership does not match a package restriction;
389
+ it can still match a source path. Unknown package names return an argument error;
390
+ a declared package with no eligible units is a valid empty scope.
391
+ - **Paths:** application-relative directory prefixes, compared at segment
392
+ boundaries. `packs/billing` includes descendants, including nested packages,
393
+ and excludes `packs/billing_admin`. `.` covers all published application-relative
394
+ paths. Repeated separators and `.` are normalized; internal `..` segments are
395
+ resolved, while attempts to escape the root, absolute paths, Windows paths,
396
+ backslashes, and NUL are rejected. External gem paths and missing paths do not
397
+ match. No host `realpath` or current source-file reads determine ownership.
398
+ - **Types and graph:** existing inclusion/exclusion rules remain authoritative.
399
+ Scoped discovery uses actual published unit types, including `gem_source` units
400
+ in the shared `rails_source` directory. Graph seeds and expansion stay within
401
+ eligibility. Existing bare graph edges can still be ambiguous between typed
402
+ units sharing an identifier; scope does not establish a missing edge target type.
403
+
404
+ `RetrievalResult#applied_scope` reports normalized lists, `eligible_units`,
405
+ `candidate_count`, `returned_units`, and `outcome` (`empty_scope`, `no_match`, or
406
+ `matched`). A matched candidate need not fit a tiny output budget. Scoped MCP
407
+ retrieval exposes this object in `structuredContent.data.applied_scope` and `_meta`,
408
+ with typed source provenance in `structuredContent.data.sources`; the budget continues
409
+ to apply to the context text. Scoped results omit `type_rank_context`, whose global
410
+ rank fields would otherwise describe the wrong population.
411
+
412
+ Search adds `applied_scope` beside its existing completeness evidence. Zero
413
+ eligible units means an empty scope; positive eligibility with a complete zero-match
414
+ search means no match. A partial zero-match search remains inconclusive.
415
+
416
+ ### Storage and cost
417
+
418
+ Scope preparation reads the complete metadata snapshot from the selected store
419
+ bundle. Packaged discovery and lexical retrieval keep that read and the query on
420
+ one published generation. Discovery's deep-field scan budget applies to matching
421
+ after scope preparation; it does **not** cap the initial metadata read.
422
+
423
+ Vector search enumerates eligible raw vector IDs, including typed and chunk IDs,
424
+ and searches every batch of at most 100 IDs before merging the best candidates.
425
+ It embeds the query once per vector execution. In-memory, pgvector, and Qdrant
426
+ adapters support this without adding package fields or re-embedding old vectors.
427
+ Scoped pgvector queries materialize eligible rows before exact distance ranking;
428
+ scoped Qdrant requests use exact search. Custom adapters must explicitly support
429
+ native ID filters and ID enumeration; unsupported adapters return a degraded
430
+ error instead of falling back to global results. Broad scopes can cost more than
431
+ ordinary approximate search. Automatic package balancing is not enabled.
432
+
433
+ See [scope evaluation](EVALUATION.md#explicit-scope-comparison) for the matched
434
+ budget replay and its cross-boundary recall tradeoff.
435
+
436
+ ## Compact published evidence and API outlines
437
+
438
+ `codebase_retrieve` and `Retriever#retrieve` accept an explicit `evidence` mode:
439
+
440
+ - `full` (default) preserves existing source formatting and budget truncation.
441
+ - `compact` chooses complete query-relevant methods and published concern display
442
+ blocks within each ranked unit, then relevant resolved runtime metadata.
443
+ - `outline` returns declared method names, kinds, lexical owners and published
444
+ line ranges. It is an API orientation aid, not reconstructed Ruby signatures.
445
+
446
+ ```ruby
447
+ retriever.retrieve('How are payments refunded?', budget: 1200, evidence: 'compact')
448
+ # MCP:
449
+ codebase_retrieve(query: 'How are payments refunded?', budget: 1200, evidence: 'compact')
450
+ lookup(identifier: 'Billing::Invoice', type: 'model', evidence: 'outline', budget: 600)
451
+ lookup(identifier: 'Billing::Invoice', type: 'model', evidence: 'compact', query: 'refund', budget: 800)
452
+ ```
453
+
454
+ Ranking, candidate limits and explicit package/path scopes are unchanged. Selection
455
+ uses the original query: method-name term matches weigh more than body matches;
456
+ source order breaks ties. With no matching terms, or no lookup query, source order
457
+ provides deterministic orientation. Nested blocks and definitions stay inside their
458
+ complete containing method. A method that cannot fit is omitted whole; compact mode
459
+ never substitutes a broken prefix. If a heredoc body lies outside its method's
460
+ syntactic range, a `whole_source_fallback` span retains the complete published
461
+ source or omits it whole when it cannot fit. Metadata fields are also included whole. More
462
+ unit names fitting in an outline does not establish that their implementation was
463
+ shown or that retrieval quality improved.
464
+
465
+ The existing retrieval token counter also charges compact headers, notices and
466
+ source spans. Compact lookup defaults to 2000 estimated tokens, using four
467
+ characters per token. These are text-context budgets; MCP JSON envelopes and
468
+ structured provenance add transport/output tokens. Very small budgets may return
469
+ no evidence text. Check omission counts and increase the budget or request full
470
+ source. `query` and `budget` apply only to compact/outline lookup; combining those
471
+ modes with `include_source: false` or nonempty `sections` is an argument error.
472
+
473
+ ### Provenance and full-source verification
474
+
475
+ MCP retrieval exposes evidence under `structuredContent.data.sources[].evidence`;
476
+ compact lookup uses `structuredContent.data.evidence`. Ruby retrieval carries the
477
+ same data in `result.sources`. Every record includes the typed unit owner, original
478
+ published display path, source SHA256, generation status, selected span hashes,
479
+ zero-based byte offsets (exclusive end), one-based line ranges and omitted spans.
480
+ Span owners are **lexical declarations**, not inferred runtime dispatch owners.
481
+
482
+ Coordinates are explicitly `published_unit`: they refer to the exact published
483
+ `source_code` bytes. Physical source coordinates are unavailable. Model/controller
484
+ units may contain synthesized headers or commented inlined concerns; the excerpt
485
+ preserves those comments verbatim and labels concern display blocks. Inherited
486
+ behavior may be described by runtime metadata without a corresponding source span.
487
+ The server never opens host application source files to fill these gaps.
488
+
489
+ Published lexical retrieval and lookup report the pinned generation when present.
490
+ Legacy flat indexes and standalone semantic metadata stores do not establish a
491
+ publication generation and report it unavailable. A current reader generation must
492
+ not be substituted for an older semantic artifact's unknown generation.
493
+
494
+ Each record supplies a `full_evidence` lookup call with the actual `type`,
495
+ `identifier`, `evidence: 'full'` and `source_sha256`. Pass that hash back to verify
496
+ the same source bytes; changed source produces a typed `stale_index` refusal.
497
+ Ordinary full lookup remains available without the hash when deliberately
498
+ inspecting the newest publication. Optional `type` disambiguates names shared by
499
+ multiple unit types and preserves actual types within mixed storage directories.
500
+ Full source is never disabled by compact mode. See [evaluation](EVALUATION.md#compact-evidence-comparison-403)
501
+ for measured tradeoffs and the limits of returned-unit metrics.
502
+
503
+ ### Source-owner selection remains an offline experiment
504
+
505
+ The [owner-overlap evaluation](EVALUATION.md#source-owner-overlap-experiment-412)
506
+ compares complete-span overlap against current score order. It adds no retrieval
507
+ configuration or MCP option. Display paths alone cannot prove original ownership,
508
+ especially for inlined concern display; representative runtime task evidence is
509
+ still needed before changing which units receive context budget.
@@ -0,0 +1,71 @@
1
+ # Ruby runtime trace enrichment
2
+
3
+ `Woods::RubyAnalyzer::TraceEnricher` records Ruby method events during a block
4
+ and can attach that evidence to Ruby method units. This is a Ruby API; Rails
5
+ session tracing and the MCP `trace_flow` tool are separate features.
6
+
7
+ ```ruby
8
+ require 'woods/ruby_analyzer'
9
+
10
+ traces = Woods::RubyAnalyzer::TraceEnricher.record { application.run }
11
+ units = Woods::RubyAnalyzer.analyze(paths: ['lib'], trace_data: traces)
12
+ # Or enrich units already analyzed:
13
+ Woods::RubyAnalyzer::TraceEnricher.merge(units: units, trace_data: traces)
14
+ ```
15
+
16
+ ## Caller evidence
17
+
18
+ Each event contains `class_name`, `method_name`, `event` (`call` or `return`),
19
+ `method_kind`, `path`, `line`, `caller_class`, `caller_method`,
20
+ `caller_method_kind`, and `return_class`.
21
+ For an observed `Caller#invoke` calling `Callee#run`, the callee's call and
22
+ return events record `caller_class: 'Caller'` and `caller_method: 'invoke'`.
23
+ The caller method field contains the method name, not a qualified backtrace label.
24
+
25
+ The caller is the **nearest Ruby method observed by this recording**. The
26
+ recorder does not capture native method frames or block frames, and does not
27
+ infer methods already on the stack when recording begins. An unknown caller
28
+ has all three caller fields set to `nil`; merge omits those unknown caller entries.
29
+ This is scoped execution evidence, not a complete application call graph.
30
+
31
+ Stacks belong to one recording and are separate for each fiber. Recording
32
+ captures the current thread, including fibers running on it, and excludes
33
+ other threads. To record work in another thread, invoke `record` in that thread.
34
+ Recursion and Ruby return events during exception or nonlocal unwinds balance
35
+ the stack. An unmatched return, such as a fiber method entered before recording,
36
+ has no inferred caller. Escaping exceptions still propagate and disable the
37
+ recorder; a later recording starts with fresh stacks.
38
+
39
+ `merge` adds `metadata[:trace]` with `call_count`, `callers`, and `return_types`.
40
+ A return event during exception unwinding does not establish normal completion;
41
+ `return_types` summarizes the return-event values Ruby exposes. Recording does
42
+ not change existing method identifiers.
43
+
44
+ ## Method identity
45
+
46
+ `method_kind` is `instance` or `singleton`. Merge matches the defining owner,
47
+ method name, and kind: `Example#run` and `Example.run` keep separate call counts
48
+ and return types. This identity survives JSON serialization; symbol keys and
49
+ symbol kind values are accepted too. Caller entries retain `caller_method_kind`
50
+ when present, so same-name instance and singleton callers remain distinct.
51
+
52
+ Singleton methods use their defining named class or module. If `Child.run`
53
+ inherits `Parent.run`, its events belong to `Parent.run`. Methods supplied by
54
+ an extended module retain that module's instance-method identity. Anonymous
55
+ owners and singleton methods on individual objects have no named unit;
56
+ their `class_name` is `nil`, and merge skips them.
57
+
58
+ ## Older recordings
59
+
60
+ Older Woods recordings derived `caller_class` from the callee receiver and
61
+ `caller_method` from an interpreter-dependent backtrace offset. Their caller
62
+ fields can disagree or describe the callee itself. Record again to obtain
63
+ corrected caller evidence; merging old JSON cannot reconstruct missing callers.
64
+
65
+ Recordings without `method_kind` match **instance methods only**. The older
66
+ recorder used a normal owner name for instance methods and an opaque string
67
+ such as `#<Class:Example>` for singleton methods. Those opaque singleton-owner
68
+ records are skipped; record again to obtain a usable singleton identity.
69
+ Handwritten fixtures for `Example.run` must specify `method_kind: 'singleton'`.
70
+ Merge never guesses the kind from which units happen to be supplied. Unknown
71
+ explicit kinds are skipped. Old caller entries without a kind remain untyped.
@@ -0,0 +1,143 @@
1
+ # Source freshness
2
+
3
+ `woods_status.index.source_freshness` compares the served generation's captured
4
+ application inputs with source bytes visible to the reader. It is separate from
5
+ index age, HEAD equality, daemon liveness and external database/runtime state.
6
+
7
+ This capability is unreleased after `2.0.0.beta2`. Check the installed gem's
8
+ `woods-extract --help`, `rake -T woods:source_status`, and `woods_status` schema
9
+ before using it; upgrading the plugin alone does not upgrade Woods.
10
+
11
+ ## Read the result
12
+
13
+ | State | Meaning | Next step |
14
+ |---|---|---|
15
+ | `current` | All covered inputs match their consumer baselines, with a verified fresh boot boundary and complete checks. A dirty checkout can be current. | Use the indexed facts within the coverage below. |
16
+ | `drifted` | At least one captured input differs, was removed, or a relevant input was added. | Inspect the changed paths; choose a full run or a justified targeted refresh. |
17
+ | `unknown` | Evidence is incomplete: for example an old index, missing source/key, scan limit, opaque symlink directory, or unproved boot/consumer boundary. | Inspect `reasons`; use a deep check or fresh full capture as appropriate. |
18
+
19
+ Drift can coexist with incomplete coverage. `reasons` reports both; absence of a
20
+ listed change never proves an omitted input is unchanged. `complete` describes
21
+ capture/traversal completion, while boot and consumer qualifications remain in
22
+ `reasons`. `counts` contains total observed added/changed/removed paths; each
23
+ `changes` list contains at most 30 paths and `truncated` marks longer lists.
24
+
25
+ ```json
26
+ {"source_check":"deep"}
27
+ ```
28
+
29
+ Pass this to `woods_status` for an explicit five-second content scan. The default
30
+ `quick` scan has a 250ms budget. Both read and HMAC source bytes; there is no
31
+ stat-only `current` shortcut, so same-size/same-mtime edits are detected. Limits
32
+ also cap traversal at 50,000 visited files and 128 MiB read. Budgets are checked
33
+ between filesystem operations; an operating-system read that itself stalls can
34
+ outlast the scan deadline. A limit produces unknown coverage, never current.
35
+
36
+ The result is tied to the **served** generation, including a reader holding an
37
+ older generation during a concurrent publication. It is recomputed on each call;
38
+ a source edit does not require an index generation change to become visible.
39
+
40
+ ## Establish a fresh baseline
41
+
42
+ Run the launcher through the application's installed bundle:
43
+
44
+ ```bash
45
+ bundle exec woods-extract full
46
+ bundle exec woods-extract incremental app/services/checkout.rb app/views/orders/show.html.erb
47
+ bundle exec woods-extract refresh routes controllers
48
+ ```
49
+
50
+ `--root PATH` selects the application root. `--output PATH` selects the index
51
+ (default `WOODS_OUTPUT`, otherwise `tmp/woods`); relative output paths resolve
52
+ under the application root. Repeat `--source-root PATH` to include additional
53
+ application-relative directories used by custom loaders. Use the same declaration
54
+ on subsequent launcher runs. Paths are separate arguments, preserving spaces,
55
+ commas and newlines. Refresh accepts known extractor names. Invalid arguments
56
+ fail before extraction; the launcher propagates the child's failure or daemon
57
+ stand-down exit 75. Split oversized incremental batches or choose full.
58
+
59
+ The launcher captures source before a **fresh child** evaluates its Gemfile,
60
+ Rakefile, Rails boot and eager loading. A private one-use handoff binds the capture
61
+ to root, output, action, rules, nonce and parent process. It waits for the child
62
+ and removes the handoff when the child exits. Source is checked again before
63
+ publication; edits during boot/extraction retain the earlier identity and are
64
+ reported, rather than being silently adopted as a current baseline.
65
+
66
+ Existing Rake tasks, direct `Extractor` calls and the watch daemon remain usable.
67
+ Their post-boot capture is marked `unverified_boot_boundary`; current bytes alone
68
+ cannot prove what a previously booted Rails process consumed. A fresh launcher
69
+ full run replaces that uncertainty. Hooks continue their existing refresh flow;
70
+ enabling hooks does not implicitly restart or replace a daemon.
71
+
72
+ ## Scope and partial extraction
73
+
74
+ Coverage follows the shared file/whole-app dispatch rules and reload policy:
75
+ application and lib Ruby, known views/locales/tests/packages/schedules/schema and
76
+ boot configuration. Source-only boot coverage also includes `Rakefile`,
77
+ `config.ru`, root gemspecs and otherwise-unclassified Ruby helpers under `config/`.
78
+ Normal generated/hidden directories are pruned before traversal, using the watch
79
+ scanner's exclusions. Explicit source roots override generic exclusions; the
80
+ index output is always excluded. Contained file symlinks are checked for stable
81
+ resolution; directory symlinks and escaping/unreadable inputs leave uncertainty.
82
+
83
+ Every consuming scope keeps its own identities. An events scan can reread a
84
+ service file while its service unit remains untouched; refreshing events does
85
+ not certify the retained service unit. Successful file/whole-extractor work
86
+ updates only its scopes, including negative results and confirmed deletion.
87
+ Unchanged scopes keep their earlier baseline. Boot inputs advance only on a full
88
+ run. Partial runtime changes retain an explicit `runtime_consumption` uncertainty
89
+ when Woods cannot prove every retained reflected fact was re-serialized. Named
90
+ framework refreshes do not certify unrelated application inputs. A handled
91
+ extractor error retains an explicit `extractor:<name>` uncertainty even if the
92
+ extractor returns an empty result. Successful consumers keep their own evidence;
93
+ a later full run without that failure can replace the uncertainty.
94
+
95
+ Custom loader source outside captured roots, missing eager-load coverage and
96
+ uncaptured application-owned unit paths remain unknown. This is application
97
+ source evidence: it does not certify external database schemas/data, remote
98
+ configuration, installed gem bytes, provider state or live runtime services.
99
+
100
+ ## Containers and hooks
101
+
102
+ Run the launcher inside the application container when that is where the bundle
103
+ and source exist, for example `docker compose exec -T app bundle exec woods-extract full`.
104
+ An MCP reader must see the source and original private key; otherwise it reports
105
+ unknown. `woods:source_status` has no Rails environment prerequisite and uses the
106
+ same verifier without Rails initialization or provider work. Its optional
107
+ Base64-encoded JSON transport supports `output`, `root` (an explicit reader-side
108
+ source mapping), and `mode` (`quick` or `deep`).
109
+
110
+ The opt-in SessionStart hook uses `WOODS_HOOK_RAKE` and `WOODS_OUTPUT`, including a
111
+ Docker command prefix without requiring a host application bundle. It prints
112
+ an actionable drift or unknown warning and stays quiet for current evidence.
113
+ Its ten-second process deadline includes command startup; the scan uses quick
114
+ mode. Missing older tasks and failed/timed-out commands report unknown. Cancelling
115
+ Docker exec does not itself prove the process inside the container stopped.
116
+ A quiet session hook does not acknowledge deferred PostToolUse queue entries.
117
+
118
+ ## Artifact and cost
119
+
120
+ Each atomic payload contains versioned `source_inputs.json`. It records a compact
121
+ identity table shared by per-consumer scope/path maps, root, rules fingerprint,
122
+ generation, coverage and scan metrics. All file identities use HMAC-SHA256 with
123
+ `<output>/.source-inputs.key`, a 32-byte, owner-only private file outside payloads.
124
+ Low-entropy secret-bearing input files receive the same protection as ordinary
125
+ source; no source bytes or raw secret hashes are added to this artifact. Do not
126
+ publish the key with an index. Missing, insecure or mismatched keys produce
127
+ unknown; Woods does not silently repair permissions or rotate keys. Independent
128
+ outputs have different identities and must be compared using their own keys and
129
+ consumer semantics, not raw manifest equality.
130
+
131
+ Failed/no-op extraction does not advance the artifact's published generation.
132
+ Flat fallback and older indexes lack verified atomic source evidence.
133
+ `WOODS_PROFILE=1` reports `source capture` and `source verification` separately.
134
+ Capture/recheck each allow up to ten seconds with the same file/byte caps.
135
+
136
+ September 2026 fixture measurements: a pinned Writebook source tree (456 visited
137
+ files, 223 hashed, 233KB) completed quick scans in median 36ms native / 45ms on a
138
+ Linux container bind mount. Discourse (26,133 visited, 6,209 hashed, 56.4MB) needed
139
+ about 1.6–2 seconds; quick scans returned unknown, while five-second scans completed
140
+ traversal and still reported an opaque directory symlink. Native Ruby 4.0 and
141
+ container Ruby 3.4 differed, so these are a budget envelope, not a filesystem
142
+ speed comparison or a macOS virtiofs benchmark. Measure your own application;
143
+ a large or slow source tree may need a full extraction rather than a longer read.