woods 2.0.0.beta1 → 2.0.0.beta3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (221) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +400 -1
  3. data/CONTRIBUTING.md +224 -9
  4. data/README.md +7 -3
  5. data/SECURITY.md +9 -6
  6. data/docs/AGENT_GUIDE.md +83 -4
  7. data/docs/AGENT_SETUP.md +82 -1
  8. data/docs/BACKEND_MATRIX.md +20 -0
  9. data/docs/CLIENT_HOOKS.md +111 -0
  10. data/docs/CONFIGURATION_REFERENCE.md +233 -13
  11. data/docs/CONSOLE_MCP_SETUP.md +35 -5
  12. data/docs/DOCKER_SETUP.md +21 -2
  13. data/docs/EVALUATION.md +464 -1
  14. data/docs/EXTRACTOR_REFERENCE.md +36 -5
  15. data/docs/FAQ.md +11 -12
  16. data/docs/GETTING_STARTED.md +17 -5
  17. data/docs/INCREMENTAL_EXTRACTION.md +158 -2
  18. data/docs/INDEX_LAYOUT.md +382 -0
  19. data/docs/INTERNALS.md +15 -7
  20. data/docs/MCP_SERVERS.md +221 -5
  21. data/docs/MCP_TOOL_COOKBOOK.md +33 -18
  22. data/docs/NOTION_INTEGRATION.md +13 -0
  23. data/docs/OBSIDIAN_INTEGRATION.md +57 -9
  24. data/docs/PUBLISHED_INDEX.md +71 -0
  25. data/docs/README.md +7 -0
  26. data/docs/RETRIEVAL_GUIDE.md +253 -11
  27. data/docs/RUNTIME_TRACING.md +71 -0
  28. data/docs/SOURCE_FRESHNESS.md +143 -0
  29. data/docs/TROUBLESHOOTING.md +117 -5
  30. data/docs/UNBLOCKED_INTEGRATION.md +25 -0
  31. data/docs/UPGRADING_TO_2.md +44 -22
  32. data/docs/WATCH_DAEMON.md +259 -59
  33. data/exe/woods-agent-config +6 -0
  34. data/exe/woods-extract +5 -0
  35. data/exe/woods-hook-context +6 -0
  36. data/lib/generators/woods/templates/woods.rb.tt +1 -3
  37. data/lib/tasks/woods.rake +47 -397
  38. data/lib/woods/agent_configuration/applier.rb +133 -0
  39. data/lib/woods/agent_configuration/cli.rb +101 -0
  40. data/lib/woods/agent_configuration/cli_options.rb +29 -0
  41. data/lib/woods/agent_configuration/document.rb +105 -0
  42. data/lib/woods/agent_configuration/error.rb +7 -0
  43. data/lib/woods/agent_configuration/launcher.rb +75 -0
  44. data/lib/woods/agent_configuration/layout.rb +59 -0
  45. data/lib/woods/agent_configuration/managed_section.rb +62 -0
  46. data/lib/woods/agent_configuration/plan.rb +98 -0
  47. data/lib/woods/agent_configuration/plan_diff.rb +38 -0
  48. data/lib/woods/agent_configuration/planned_files.rb +61 -0
  49. data/lib/woods/agent_configuration/planner.rb +63 -0
  50. data/lib/woods/agent_configuration/planner_validation.rb +77 -0
  51. data/lib/woods/agent_configuration/preflight.rb +100 -0
  52. data/lib/woods/agent_configuration/recovery.rb +49 -0
  53. data/lib/woods/ast/node.rb +2 -0
  54. data/lib/woods/ast/parser.rb +38 -5
  55. data/lib/woods/atomic_file.rb +133 -3
  56. data/lib/woods/builder.rb +21 -5
  57. data/lib/woods/cache/cache_middleware.rb +28 -7
  58. data/lib/woods/cache/cache_store.rb +4 -5
  59. data/lib/woods/change_set.rb +5 -4
  60. data/lib/woods/console/credential_index.rb +20 -2
  61. data/lib/woods/console/credential_scanner.rb +14 -14
  62. data/lib/woods/console/credential_scanner_registry.rb +36 -0
  63. data/lib/woods/console/embedded_executor.rb +1 -1
  64. data/lib/woods/console/encrypted_credential_snapshot.rb +16 -0
  65. data/lib/woods/console/rack_middleware.rb +22 -13
  66. data/lib/woods/console/server.rb +18 -16
  67. data/lib/woods/dependency_graph.rb +65 -13
  68. data/lib/woods/embedding/corpus.rb +94 -0
  69. data/lib/woods/embedding/indexer.rb +90 -46
  70. data/lib/woods/embedding/openai.rb +17 -6
  71. data/lib/woods/evaluation/ablation_executor.rb +6 -1
  72. data/lib/woods/evaluation/ablation_timed_executor.rb +22 -4
  73. data/lib/woods/export/typed_reader.rb +56 -0
  74. data/lib/woods/extractor.rb +557 -228
  75. data/lib/woods/extractors/action_cable_extractor.rb +3 -1
  76. data/lib/woods/extractors/behavioral_profile.rb +9 -7
  77. data/lib/woods/extractors/caching_extractor.rb +3 -1
  78. data/lib/woods/extractors/concern_extractor.rb +64 -6
  79. data/lib/woods/extractors/configuration_extractor.rb +7 -3
  80. data/lib/woods/extractors/controller_extractor.rb +13 -4
  81. data/lib/woods/extractors/database_view_extractor.rb +3 -1
  82. data/lib/woods/extractors/decorator_extractor.rb +3 -1
  83. data/lib/woods/extractors/engine_extractor.rb +3 -1
  84. data/lib/woods/extractors/event_extractor.rb +4 -2
  85. data/lib/woods/extractors/factory_extractor.rb +3 -1
  86. data/lib/woods/extractors/graphql_extractor.rb +8 -2
  87. data/lib/woods/extractors/i18n_extractor.rb +3 -1
  88. data/lib/woods/extractors/job_extractor.rb +6 -19
  89. data/lib/woods/extractors/lib_extractor.rb +3 -1
  90. data/lib/woods/extractors/mailer_extractor.rb +20 -5
  91. data/lib/woods/extractors/manager_extractor.rb +3 -1
  92. data/lib/woods/extractors/method_parameters.rb +53 -0
  93. data/lib/woods/extractors/middleware_argument.rb +65 -0
  94. data/lib/woods/extractors/middleware_extractor.rb +9 -3
  95. data/lib/woods/extractors/migration_extractor.rb +3 -1
  96. data/lib/woods/extractors/model_extractor.rb +39 -33
  97. data/lib/woods/extractors/package_extractor.rb +24 -4
  98. data/lib/woods/extractors/phlex_extractor.rb +3 -1
  99. data/lib/woods/extractors/policy_extractor.rb +3 -1
  100. data/lib/woods/extractors/poro_extractor.rb +3 -1
  101. data/lib/woods/extractors/pundit_extractor.rb +3 -1
  102. data/lib/woods/extractors/rails_source_extractor.rb +4 -2
  103. data/lib/woods/extractors/rake_task_extractor.rb +4 -2
  104. data/lib/woods/extractors/route_extractor.rb +3 -1
  105. data/lib/woods/extractors/route_helper_resolver.rb +10 -33
  106. data/lib/woods/extractors/scheduled_job_extractor.rb +41 -15
  107. data/lib/woods/extractors/serializer_extractor.rb +4 -2
  108. data/lib/woods/extractors/service_extractor.rb +3 -1
  109. data/lib/woods/extractors/shared_dependency_scanner.rb +2 -2
  110. data/lib/woods/extractors/shared_utility_methods.rb +27 -15
  111. data/lib/woods/extractors/source_nesting.rb +1 -1
  112. data/lib/woods/extractors/state_machine_extractor.rb +3 -1
  113. data/lib/woods/extractors/test_mapping_extractor.rb +3 -1
  114. data/lib/woods/extractors/validator_extractor.rb +3 -1
  115. data/lib/woods/extractors/view_component_extractor.rb +3 -1
  116. data/lib/woods/extractors/view_template_extractor.rb +3 -1
  117. data/lib/woods/flow_assembler.rb +87 -8
  118. data/lib/woods/flow_precomputer.rb +44 -7
  119. data/lib/woods/gem_mapper.rb +2 -0
  120. data/lib/woods/git_history.rb +116 -0
  121. data/lib/woods/graph_analyzer.rb +195 -63
  122. data/lib/woods/hooks/context_cli.rb +54 -0
  123. data/lib/woods/hooks/context_event.rb +88 -0
  124. data/lib/woods/hooks/context_hint.rb +73 -0
  125. data/lib/woods/hooks/context_impact.rb +77 -0
  126. data/lib/woods/hooks/context_output.rb +47 -0
  127. data/lib/woods/hooks/context_state.rb +102 -0
  128. data/lib/woods/hooks/refresh.rb +79 -0
  129. data/lib/woods/hooks/rule_projection.rb +78 -0
  130. data/lib/woods/input_rules.rb +19 -0
  131. data/lib/woods/mcp/bearer_auth.rb +20 -12
  132. data/lib/woods/mcp/bootstrapper.rb +62 -0
  133. data/lib/woods/mcp/index_reader.rb +323 -160
  134. data/lib/woods/mcp/initialization_guidance.rb +27 -0
  135. data/lib/woods/mcp/origin_guard.rb +17 -9
  136. data/lib/woods/mcp/published_lexical_retriever.rb +115 -0
  137. data/lib/woods/mcp/renderers/markdown_renderer.rb +8 -1
  138. data/lib/woods/mcp/renderers/plain_renderer.rb +7 -1
  139. data/lib/woods/mcp/search_results.rb +74 -0
  140. data/lib/woods/mcp/server.rb +158 -37
  141. data/lib/woods/mcp/tool_contract.rb +2 -0
  142. data/lib/woods/mcp/tool_response_renderer.rb +25 -0
  143. data/lib/woods/mcp/traversal_evidence.rb +113 -0
  144. data/lib/woods/mcp/traversal_evidence_index.rb +100 -0
  145. data/lib/woods/mcp/traversal_evidence_page.rb +41 -0
  146. data/lib/woods/mcp/traversal_evidence_text.rb +52 -0
  147. data/lib/woods/notion/exporter.rb +56 -17
  148. data/lib/woods/obsidian/destination_plan.rb +98 -0
  149. data/lib/woods/obsidian/name_mapper.rb +19 -3
  150. data/lib/woods/obsidian/note_builder.rb +19 -10
  151. data/lib/woods/obsidian/vault_exporter.rb +88 -32
  152. data/lib/woods/operator/pipeline_guard.rb +18 -13
  153. data/lib/woods/path_dispatcher.rb +7 -1
  154. data/lib/woods/payload_store.rb +29 -15
  155. data/lib/woods/railtie.rb +3 -3
  156. data/lib/woods/railtie_support.rb +12 -12
  157. data/lib/woods/rake_helpers.rb +392 -0
  158. data/lib/woods/resilience/graph_invariant_validator/membership_checks.rb +71 -0
  159. data/lib/woods/resilience/graph_invariant_validator/node_checks.rb +61 -0
  160. data/lib/woods/resilience/graph_invariant_validator/reverse_relationship_checks.rb +46 -0
  161. data/lib/woods/resilience/graph_invariant_validator.rb +119 -0
  162. data/lib/woods/resilience/index_validator/graph_checks.rb +80 -0
  163. data/lib/woods/resilience/index_validator.rb +112 -23
  164. data/lib/woods/retrieval/context_assembler.rb +50 -15
  165. data/lib/woods/retrieval/lexical_assembler.rb +73 -0
  166. data/lib/woods/retrieval/lexical_index.rb +119 -0
  167. data/lib/woods/retrieval/ranker.rb +4 -2
  168. data/lib/woods/retrieval/scope.rb +108 -0
  169. data/lib/woods/retrieval/scoped_graph_store.rb +32 -0
  170. data/lib/woods/retrieval/scoped_vector_store.rb +55 -0
  171. data/lib/woods/retrieval/search_executor.rb +86 -27
  172. data/lib/woods/retrieval/source_evidence.rb +200 -0
  173. data/lib/woods/retriever.rb +98 -22
  174. data/lib/woods/ruby_analyzer/trace_enricher.rb +80 -38
  175. data/lib/woods/session_tracer/middleware.rb +10 -12
  176. data/lib/woods/session_tracer/redis_store.rb +22 -6
  177. data/lib/woods/session_tracer/session_flow_assembler.rb +23 -17
  178. data/lib/woods/session_tracer/solid_cache_coordination.rb +6 -4
  179. data/lib/woods/session_tracer/unit_resolver.rb +63 -0
  180. data/lib/woods/source_inputs/consumer_errors.rb +27 -0
  181. data/lib/woods/source_inputs/handoff.rb +102 -0
  182. data/lib/woods/source_inputs/launcher.rb +157 -0
  183. data/lib/woods/source_inputs/manifest.rb +124 -0
  184. data/lib/woods/source_inputs/private_key.rb +55 -0
  185. data/lib/woods/source_inputs/scanner.rb +171 -0
  186. data/lib/woods/source_inputs/scopes.rb +71 -0
  187. data/lib/woods/source_inputs/session.rb +214 -0
  188. data/lib/woods/source_inputs/status.rb +84 -0
  189. data/lib/woods/source_inputs/verifier.rb +107 -0
  190. data/lib/woods/storage/metadata_store.rb +25 -25
  191. data/lib/woods/storage/pgvector.rb +29 -8
  192. data/lib/woods/storage/qdrant.rb +17 -7
  193. data/lib/woods/storage/vector_store.rb +18 -6
  194. data/lib/woods/tasks.rb +3 -2
  195. data/lib/woods/temporal/json_snapshot_store.rb +29 -8
  196. data/lib/woods/unblocked/exporter.rb +59 -70
  197. data/lib/woods/version.rb +1 -1
  198. data/lib/woods/watch/boot_snapshot.rb +52 -0
  199. data/lib/woods/watch/daemon.rb +136 -28
  200. data/lib/woods/watch/listen_watcher.rb +4 -0
  201. data/lib/woods/watch/polling_watcher.rb +5 -1
  202. data/lib/woods/watch/status.rb +20 -15
  203. data/lib/woods/watch/tree_scan.rb +21 -13
  204. data/lib/woods/watch/watcher.rb +4 -1
  205. data/lib/woods.rb +135 -11
  206. data/plugin/.claude-plugin/plugin.json +1 -1
  207. data/plugin/hooks/adapters/normalize.jq +15 -0
  208. data/plugin/hooks/adapters/normalize.rb +63 -0
  209. data/plugin/hooks/hooks.json +20 -0
  210. data/plugin/hooks/woods-context.sh +50 -0
  211. data/plugin/hooks/woods-input-rules.sh +159 -0
  212. data/plugin/hooks/woods-opencode.mjs +65 -0
  213. data/plugin/hooks/woods-post-edit.sh +2 -225
  214. data/plugin/hooks/woods-refresh.sh +260 -0
  215. data/plugin/hooks/woods-session-start.sh +47 -55
  216. data/plugin/skills/woods-agent-enable/SKILL.md +13 -0
  217. data/plugin/skills/woods-diagnose/SKILL.md +288 -1
  218. data/plugin/skills/woods-investigate/SKILL.md +106 -0
  219. data/plugin/skills/woods-mcp-config/SKILL.md +89 -1
  220. data/plugin/skills/woods-setup/SKILL.md +107 -6
  221. metadata +84 -5
@@ -60,9 +60,13 @@ Every extractor returns `Array<ExtractedUnit>`. An `ExtractedUnit` is a self-con
60
60
 
61
61
  **Key details:**
62
62
  - Uses `ActiveRecord::Base.descendants` for discovery (runtime introspection, not static parsing)
63
+ - Named, source-defined app model mixins (for example `Card::Pinnable` in `app/models/card/pinnable.rb`) resolve through runtime source locations, with conventional concern paths as fallbacks. Included mixins also receive `:concern` units, so their actual files map to the includer through dependency edges. Gem-owned modules stay outside this discovery. Conventional concern files retain their existing identity even when nested helpers share the same file. Outside those directories, each included runtime mixin receives its own concern identity even when several share a source file; editing that file refreshes every includer.
63
64
  - Inlines concerns: all `include FooConcern` references are resolved and the concern source is appended to `source_code`. Inlined concern names are recorded in `metadata[:inlined_concerns]`
64
- - Extracts all 19 callback types: `before_validation`, `after_validation`, `before_save`, `after_save`, `around_save`, `before_create`, `after_create`, `around_create`, `before_update`, `after_update`, `around_update`, `before_destroy`, `after_destroy`, `around_destroy`, `after_commit`, `after_rollback`, `after_initialize`, `after_find`, `after_touch`
65
+ - Reads Rails' per-event callback chains (`_save_callbacks`, `_create_callbacks`, and the other lifecycle events), preserving each chain's order and each entry's `kind`, filter and conditions. The public `type` combines kind and event, such as `before_save` or `after_create`; Rails' separate `before_commit` event is reported as `before_commit`, not `before_before_commit`. `callback_count` equals the emitted callback list's length. The list includes framework-registered callbacks; it is runtime metadata, not an application-only filter or a cross-event execution trace. Regenerate the index after upgrading to pick up corrected callback metadata.
66
+ - Proc/lambda filters, including Rails-generated association callbacks, use stable source-site labels in both metadata and callback chunks: `#<Proc app/models/post.rb:12>` (or `lambda`). App paths are relative to `Rails.root`; external paths are retained and native procs use `native`. Rails 6's numeric filter identity is resolved through `raw_filter`. These labels describe location and callable kind, not captured closure state; callbacks are never executed. Model condition labels retain their existing format.
67
+ - Default callback-object representations omit process addresses: an instance becomes `#<CleanupCallback>`, an anonymous class becomes `#<Class>`, and its instance becomes `#<#<Class>>`. Anonymous namespace prefixes are normalized too (for example, `#<Module>::CleanupCallback`). Named classes and custom `to_s` labels retain their text. These labels do not distinguish arbitrary object state; separate registered callbacks remain separate entries even when their descriptive labels match. Controller object-filter formatting is unchanged.
65
68
  - Callback side-effects are analyzed via `CallbackAnalyzer`: detects columns written (`self.col =`), jobs enqueued (`perform_later`), and services called
69
+ - Reflects model class and instance methods after reading the schema, so Rails schema-loading optimizations produce the same method metadata in cold and warmed runs. Application-defined constructors remain visible; Rails versions that install an optimized singleton `new` during schema loading consistently include it in `class_methods`.
66
70
  - Automatically skips HABTM join models and anonymous classes
67
71
  - Chunks every model into semantic sections: `:summary`, `:associations`, `:callbacks`, `:validations`, `:scopes`, `:methods`
68
72
  - **Runtime-generated method detection:** Because extraction runs inside a booted Rails process, `instance_methods(false)` captures every method Rails generates dynamically, enum predicates (`status_active?`, `status_pending?`), association builders (`build_profile`, `create_line_item!`), attribute accessors, and dynamically registered scopes. Static analysis tools cannot see these methods because they only exist after Rails processes the DSL declarations at boot time
@@ -158,6 +162,8 @@ class PageView < AnalyticsRecord; end # metadata[:database] => "analytics"
158
162
  - Route context is inlined in `source_code` as a comment header, not just in metadata
159
163
  - Chunks per-action: each action becomes a `:action` chunk with its applicable filters and route
160
164
  - Metadata includes permitted params (strong parameters), response formats, and applied filters per action
165
+ - Inline callbacks use stable source-site labels in filter metadata, controller annotations and action chunks: `#<Proc app/controllers/posts_controller.rb:12>` (or `lambda`). The controller filter metadata and annotations use the same labels for `if`/`unless` procs. App paths are relative to `Rails.root`; external paths are retained and native procs use `native`. Labels describe the callable location and kind, not captured closure state, and never execute callbacks.
166
+ - Route helper resolution accepts every live named controller/action route, including `file_path`, `image_url`, `download_path`, and `root_path`. Unknown filesystem/asset helpers produce no edge; matching names are conservative source references, not proof a call executes.
161
167
  - Extracts `redirect_to` navigation edges: named route helpers (`posts_path`, `users_url`) are resolved to controller targets via `RouteHelperResolver`, producing `:redirect_to` dependency edges (gated by `extract_navigation_edges` config)
162
168
 
163
169
  **Edge cases:**
@@ -194,6 +200,7 @@ class PageView < AnalyticsRecord; end # metadata[:database] => "analytics"
194
200
  - Scans: `app/services`, `app/interactors`, `app/operations`, `app/commands`, `app/use_cases`
195
201
  - Extracts public entry points (`call`, `perform`, `execute`, `run`), custom error classes, and dependency references
196
202
  - File-based discovery (not class introspection), so it catches services with non-standard superclasses
203
+ - `initialize_params` describes declared names, default presence and keyword status from Ruby syntax. Nested/comma-bearing defaults are not evaluated or treated as parameters; named rest, keyword-rest and block parameters retain their names. Anonymous forwarding has no name to report; malformed source produces an empty parameter list.
197
204
 
198
205
  **Example output (abbreviated):**
199
206
 
@@ -218,6 +225,7 @@ class PageView < AnalyticsRecord; end # metadata[:database] => "analytics"
218
225
  **Key details:**
219
226
  - Scans: `app/jobs`, `app/workers`, `app/sidekiq`
220
227
  - Extracts queue name, retry configuration, concurrency options, perform method arguments, and callbacks
228
+ - `perform_params` uses the same syntax-aware signature parsing as service initializers and preserves its `name`, `splat` (`single`/`double`/null), and `has_default` fields. Keyword defaults do not invent additional argument names.
221
229
  - Records what triggers this job (reverse lookup via dependency graph after extraction)
222
230
  - Supports both ActiveJob and Sidekiq native workers
223
231
 
@@ -243,7 +251,8 @@ class PageView < AnalyticsRecord; end # metadata[:database] => "analytics"
243
251
  **What it captures:** ActionMailer classes with their mailer actions, defaults, template paths, callbacks, and helper usage.
244
252
 
245
253
  **Key details:**
246
- - Discovers via class introspection (`ActionMailer::Base.descendants`)
254
+ - Discovers `ApplicationMailer.descendants` when that class exists, otherwise `ActionMailer::Base.descendants`; an app without ActionMailer contributes no mailer units.
255
+ - Discovery and direct extraction accept only mailers backed by an existing app-owned source file, excluding dependency mailers and fabricated convention paths.
247
256
  - Each mailer action corresponds to an email template, template paths are recorded in metadata
248
257
  - Extracts `default from:`, `layout`, and per-action subject patterns
249
258
 
@@ -294,7 +303,8 @@ class PageView < AnalyticsRecord; end # metadata[:database] => "analytics"
294
303
 
295
304
  **Key details:**
296
305
  - Extracts the entire stack as one unit (not one per middleware)
297
- - Records middleware class names, insertion order, and any initialization arguments
306
+ - Records middleware class names, insertion order, and initialization arguments as readable strings
307
+ - Argument rendering preserves literal strings and nested array/hash configuration. Procs use source locations; anonymous classes (including Ruby temporary names used by Rails executors/reloaders) use parent names and method source locations. Opaque objects using Ruby's default `to_s` are represented by class, without walking private runtime state. Custom `to_s` output is preserved, so application-defined nondeterministic renderers can still vary. Closure captures and opaque object internals are not serialized.
298
308
  - No per-file mapping, so incremental re-extraction re-runs `MiddlewareExtractor` wholesale when `config/application.rb`, `Gemfile.lock`, or a file under `config/initializers`/`config/environments` changes
299
309
 
300
310
  ---
@@ -331,6 +341,7 @@ class PageView < AnalyticsRecord; end # metadata[:database] => "analytics"
331
341
  **Key details:**
332
342
  - File-based scanning, no Rails boot needed for the actual file reading
333
343
  - Records which partials a template renders and which instance variables it expects
344
+ - Loads the runtime route collection before caching named helpers, including Rails lazy route sets. Fresh-process incremental view extraction resolves the same navigation targets as full extraction.
334
345
  - Extracts navigation dependencies: `link_to` and `form_with`/`form_for` calls using `_path`/`_url` route helpers are resolved to controller targets via `RouteHelperResolver`
335
346
  - Navigation edges use `:link_to` and `:form_action` via types in the dependency array
336
347
  - Gated by `extract_navigation_edges` config (default: true)
@@ -469,6 +480,7 @@ class PageView < AnalyticsRecord; end # metadata[:database] => "analytics"
469
480
  **Key details:**
470
481
  - Identifier is the package directory relative to `Rails.root` (`.` for the root package), the same name Packwerk uses
471
482
  - Honors `packwerk.yml` `package_paths` and `exclude`; without one, `**/` with the Packwerk default excludes (`bin`, `node_modules`, `script`, `tmp`, `vendor`)
483
+ - With those exact defaults, excluded top-level directories are pruned before discovery descends into them, so an index or snapshots under `tmp/` do not add package-scan work. Custom patterns or exclusions retain their configured glob behavior. Hidden directories and symlink directories are not recursively followed by default.
472
484
  - `metadata`: `name`, `dependencies` (sorted), `enforce_dependencies` (`true`, `false`, or `"strict"`), `enforce_privacy`, `layer` (pks), `public_path`, `owner`
473
485
  - Each declared dependency becomes a `{ type: :package, target: <name>, via: :package_dependency }` edge
474
486
  - Package membership on other units (`metadata[:package]`, below) does not depend on how a unit was discovered: any registered unit with a file path under a package root is annotated. The undeclared cross-package edge report remains a follow-up, not this extractor. Discovery is the separate open gap: a pack-resident file-based unit is not yet found by `PathDispatcher` when only its `package.yml` changes (follow-up B-175), so it carries no membership only because it has no unit at all yet, not because membership skips it
@@ -530,7 +542,17 @@ Every app-owned unit under a package root carries `metadata[:package]` with the
530
542
  **Key details:**
531
543
  - Reads: `config/recurring.yml` (Solid Queue), `config/sidekiq_cron.yml` (Sidekiq Cron), `config/schedule.rb` (Whenever)
532
544
  - Extracts job class name, cron expression, queue, and any arguments
533
- - File-based (static read, no Rails introspection needed)
545
+ - Resolves trusted application `recurring.yml` through Rails' configuration loader,
546
+ including ERB, filename-relative `require_relative`, and YAML aliases. On Rails
547
+ 6.0 (before that loader existed), evaluates ERB with its filename and retains
548
+ safe YAML loading of scalars, hashes, arrays and symbols. ERB runs application
549
+ code in the extraction process; index only applications you trust.
550
+ - Environment-wrapped task maps select the current Rails environment, including
551
+ custom names; an absent environment falls back to the first section, while an
552
+ explicitly empty section stays empty. Flat task maps remain supported.
553
+ - Sidekiq-Cron remains safe-loaded YAML; Whenever remains a static DSL scan.
554
+ Invalid YAML/ERB, missing required files and runtime configuration errors are
555
+ logged and omit that schedule file. Source remains the original file text.
534
556
  - No per-file mapping, so incremental re-extraction re-runs `ScheduledJobExtractor` wholesale whenever one of the schedule files above changes
535
557
 
536
558
  ---
@@ -711,7 +733,16 @@ When written to disk, units also include:
711
733
 
712
734
  ### Git enrichment fields (`metadata[:git]`)
713
735
 
714
- If the host app is a git repo, the following are added to `metadata[:git]` after extraction:
736
+ If the host app is a git repo, the following are added to `metadata[:git]` after extraction.
737
+ History is limited to commits reachable from `HEAD` in the past 365 days,
738
+ including merged branch history. Unmerged branches, remote refs, and tool
739
+ checkpoint refs do not contribute. Commands run against the application root;
740
+ when `WOODS_GIT_DIR` is set, `HEAD` belongs to that explicitly selected git
741
+ directory, which may differ from a linked worktree's HEAD.
742
+
743
+ After upgrading from a version that included all refs, run a full
744
+ `woods:extract` to replace previously published git metadata. Incremental
745
+ extraction refreshes only the units it rewrites.
715
746
 
716
747
  | Field | Description |
717
748
  |-------|-------------|
data/docs/FAQ.md CHANGED
@@ -274,7 +274,13 @@ Verify the active v2 generation with `docker compose exec app bundle exec rake w
274
274
 
275
275
  ### How do I configure the Console Server with Docker?
276
276
 
277
- First set `config.console_mcp_enabled = true` in the Rails initializer after reviewing the live-data trust boundary. Stdio does not send a bearer token, but production Rails boot still requires a configured `console_mcp_token` of at least 32 characters whenever Console is enabled. Supply `WOODS_CONSOLE_MCP_TOKEN` through the container's secret mechanism; see [Console MCP setup](CONSOLE_MCP_SETUP.md#option-a-stdio-via-rake-recommended). Then, for the embedded mode (9 Tier 1 tools), point the MCP client at `docker compose exec -T` so Compose does not allocate a pseudo-TTY:
277
+ First enable the master `console_mcp_enabled` switch after reviewing the
278
+ live-data trust boundary. For stdio-only use, explicitly set
279
+ `console_mcp_http_enabled = false`; HTTP remains enabled by default for
280
+ compatibility and requires its bearer token while enabled. Follow
281
+ [Console MCP setup](CONSOLE_MCP_SETUP.md#option-a-stdio-via-rake-recommended)
282
+ for the configuration. Then point the embedded-mode client (9 Tier 1 tools)
283
+ at `docker compose exec -T` so Compose does not allocate a pseudo-TTY:
278
284
 
279
285
  ```json
280
286
  {
@@ -414,7 +420,7 @@ When you run `rake woods:embed`, Woods generates embedding vectors for each extr
414
420
  Several options for tuning retrieval:
415
421
 
416
422
  - **Increase `max_context_tokens`** to include more units per query (at the cost of larger LLM context).
417
- - **Lower `similarity_threshold`** (default 0.7) to include less similar results.
423
+ - **Use explicit retrieval scopes** and inspect ranking evidence. `similarity_threshold` is deprecated and does not filter results; see [retrieval tuning](RETRIEVAL_GUIDE.md#tuning).
418
424
  - **Enable framework sources** (`include_framework_sources: true`) if Rails internals are relevant to your queries.
419
425
  - **Use retrieval feedback only in a custom embedded server** that wires a feedback store. The normal packaged executable does not register feedback tools.
420
426
 
@@ -442,16 +448,9 @@ Snapshots prefer their own SQLite database (`woods.sqlite3` in the output direct
442
448
 
443
449
  The session tracer is middleware that records which Rails actions are invoked during a browser session, assembles the relevant extracted units, and makes that context available via the `session_trace` MCP tool. It is useful for giving an AI tool accurate context about what code path was active during a specific user interaction.
444
450
 
445
- Session tracing is disabled by default. To enable it:
446
-
447
- ```ruby
448
- config.session_tracer_enabled = true
449
- config.session_store = Woods::SessionTracer::FileStore.new(
450
- Rails.root.join('tmp/session_traces')
451
- )
452
- ```
453
-
454
- The `session_store` option is required, there is no default store.
451
+ Session tracing is disabled by default and requires an explicit `session_store`.
452
+ Follow the [canonical configuration example](CONFIGURATION_REFERENCE.md#session-tracer-options)
453
+ for store construction and review trace retention and access controls before enabling it.
455
454
 
456
455
  ---
457
456
 
@@ -12,7 +12,19 @@ If an agent will perform the installation, use the safety and handoff checklist
12
12
 
13
13
  ## 1. Install the gem
14
14
 
15
- Add Woods to the development group:
15
+ Use the [README release table](../README.md) to choose a version, then confirm
16
+ that **exact version is published** on the [RubyGems versions page](https://rubygems.org/gems/woods/versions)
17
+ before editing the Gemfile. A prepared release checkout can update the README
18
+ before its gem is published; if the version is absent, choose an available
19
+ version or wait for publication.
20
+
21
+ If the published 2.x line has only beta or release-candidate versions, use an
22
+ exact pin to the published prerelease, following the README's prerelease
23
+ instructions; `~> 2.0` does not select prereleases. Follow the selected version's
24
+ tag documentation. The `main` guides may describe features absent from the
25
+ published gem.
26
+
27
+ Once a stable 2.x release is published, add Woods to the development group with:
16
28
 
17
29
  ```ruby
18
30
  # Gemfile
@@ -108,13 +120,13 @@ For example:
108
120
 
109
121
  > Use Woods to find `Order`, inspect its resolved callbacks and associations, and list the first two levels of code that depend on it. Cite the Woods identifiers you used.
110
122
 
111
- The Index schema inventory totals 29 schemas. Fourteen register as tools in a normal packaged launch; `codebase_retrieve` is among them but returns a configuration error until embeddings are enabled. The other structural tools work immediately. See [Agent guide](AGENT_GUIDE.md) for a reliable query workflow.
123
+ The Index schema inventory totals 29 schemas. Fourteen register as tools in a normal packaged launch; `codebase_retrieve` is among them but needs embeddings in the default semantic mode, or explicit [lexical retrieval](RETRIEVAL_GUIDE.md#embedding-free-lexical-retrieval) over extraction output. The other structural tools work immediately. See [Agent guide](AGENT_GUIDE.md) for a reliable query workflow.
112
124
 
113
125
  ## Optional next steps
114
126
 
115
127
  ### Add semantic search
116
128
 
117
- Structural search, exact lookup, dependency traversal, graph analysis, and flow tracing do not need embeddings. Add embeddings only when agents need natural-language retrieval.
129
+ Structural search, exact lookup, dependency traversal, graph analysis, and flow tracing do not need embeddings. For ranked natural-language retrieval, choose explicit [lexical mode](RETRIEVAL_GUIDE.md#embedding-free-lexical-retrieval) without providers, or configure embeddings for semantic matching.
118
130
 
119
131
  The local preset uses SQLite metadata, persisted in-memory vectors, and a local Ollama service. Add `gem "sqlite3"` to the application bundle if it is not already present. MySQL/PostgreSQL applications that do not want that dependency can use the `:shared_filesystem` preset instead; it still uses Ollama but persists all stores beneath the Woods output directory.
120
132
 
@@ -154,7 +166,7 @@ When dependencies, initializers, database configuration, credentials, or schema
154
166
 
155
167
  The watcher maintains the structural index. If semantic retrieval is enabled, also run `bin/rails woods:embed_incremental` to update vectors. Without a resident watcher, run `bin/rails woods:incremental` after changes. Use a full `woods:extract` after major upgrades or when validation reports drift. CI and shared-artifact patterns are covered in [Incremental extraction](INCREMENTAL_EXTRACTION.md).
156
168
 
157
- On Rails 8.1, `config/ci.rb` can refresh the index before any gate that reads it: `step "Woods: refresh", "bin/rails woods:incremental"`. With the Claude Code plugin installed, an opt-in `PostToolUse` hook refreshes the index after graph-changing edits and an opt-in `SessionStart` hook warns when it predates the last commit; set `WOODS_HOOKS_ENABLED=1` to turn them on. See [Watch daemon](WATCH_DAEMON.md#hooks-for-agent-sessions).
169
+ On Rails 8.1, `config/ci.rb` can refresh the index before any gate that reads it: `step "Woods: refresh", "bin/rails woods:incremental"`. With the Claude Code plugin installed, an opt-in `PostToolUse` hook refreshes the index after graph-changing edits and an opt-in `SessionStart` hook warns about source-content drift or unknown evidence; set `WOODS_HOOKS_ENABLED=1` to turn them on. See [Watch daemon](WATCH_DAEMON.md#hooks-for-agent-sessions).
158
170
 
159
171
  ### Enable the Console Server
160
172
 
@@ -171,7 +183,7 @@ If live-data queries are necessary, review its allowlists, blocked tables, crede
171
183
  | Rails fails during extraction | Boot and eager-load Rails with the same environment variables | [Troubleshooting](TROUBLESHOOTING.md) |
172
184
  | Validation reports missing or stale units | Run a full extraction, then validate again | [Incremental extraction](INCREMENTAL_EXTRACTION.md) |
173
185
  | MCP reports no index or zero units | Confirm `cwd`, the host-visible `tmp/woods` path, and `woods:stats` output | [MCP servers](MCP_SERVERS.md) |
174
- | `codebase_retrieve` says it is disabled | Configure an embedding provider and run `woods:embed`, or use `search` | [Retrieval guide](RETRIEVAL_GUIDE.md) |
186
+ | `codebase_retrieve` says it is disabled | Choose lexical mode or configure embeddings and run `woods:embed`; `search` also works | [Retrieval guide](RETRIEVAL_GUIDE.md) |
175
187
  | Docker extraction succeeds but MCP cannot see it | Translate the container output path to its host-mounted path | [Docker setup](DOCKER_SETUP.md) |
176
188
 
177
189
  ## Where to go next
@@ -28,12 +28,24 @@ Three differences are tolerated, and nothing else:
28
28
  | Ordering inside a unit's `dependents` | Full extraction appends in extractor order, incremental in graph order. Same multiset. |
29
29
  | PageRank beyond six decimal places | Iterative floating point accumulated in each run's registration order. Scores are compared as values; only the last bits are forgiven. |
30
30
 
31
+ The unit-file write skip ignores only Woods' top-level `extracted_at` stamp.
32
+ A nested metadata field with the same name is application data: changing it
33
+ rewrites the unit in both compact and pretty JSON output.
34
+
31
35
  `graph_analysis.json` used to be a fourth row, tolerating list ordering. It no
32
36
  longer is: the analyzer is order-independent and the oracle compares the file
33
37
  exactly. Tolerating the ordering there meant the harness, the only test that
34
38
  compares a full run against an incremental one, could not see the very
35
39
  dependence the analyzer's determinism work existed to remove.
36
40
 
41
+ Graph targets use string identifiers even when an extractor emits a symbolic
42
+ external target such as `:http_api`. Full and incremental runs retain every
43
+ reverse dependency across JSON restoration and re-registration, including
44
+ contributions from units sharing an identifier under different types. Unit
45
+ dependency metadata retains its extractor-provided values. If an older version
46
+ already lost reverse dependencies on these targets, run a full extraction once
47
+ to restore them; loading the damaged graph cannot recover discarded entries.
48
+
37
49
  This matters most for **incremental CI chains**: restore the previous graph,
38
50
  run `woods:incremental` per merge. There, a unit that goes missing propagates
39
51
  forward run over run instead of being erased by the next full rebuild.
@@ -52,6 +64,10 @@ failure a CI chain cannot afford:
52
64
  | The range fails otherwise | Actionable error naming the range, **exit 1**. |
53
65
  | There is no `git` binary at all | Same two rows as above: the failure reads `git unavailable: …` and takes the daemon-coverage decision, rather than dying with an `Errno::ENOENT` backtrace. |
54
66
 
67
+ Changed paths are normalized lexically before dispatch: trailing root slashes,
68
+ duplicate separators and `.`/`..` segments do not create separate changes or
69
+ bypass matching. Missing files remain representable; symlinks are not resolved.
70
+
55
71
  The range comes from `CI_COMMIT_BEFORE_SHA..CI_COMMIT_SHA` (GitLab),
56
72
  `origin/$GITHUB_BASE_REF...HEAD` (GitHub Actions), or `HEAD~1` (default). An
57
73
  unresolvable range — a GitLab zero-SHA on a new branch, an unfetched base ref,
@@ -76,13 +92,20 @@ The diff itself is rooted at the extracted application (`git -C Rails.root`),
76
92
  so it cannot read whatever checkout the process happened to start in — the
77
93
  same rooting rule the manifest's git provenance follows.
78
94
 
95
+ Named, source-defined app modules included by runtime models are tracked as concern units even
96
+ outside `concerns/` directories. Changing their source refreshes their includers,
97
+ including inlined code and callback analysis. Multiple runtime mixins sharing a source
98
+ file retain separate identities and refresh all their includers. Run a full extraction after upgrading
99
+ to populate these previously missing source mappings.
100
+
79
101
  ## What a run does, in order
80
102
 
81
103
  `Extractor#extract_changed` is order-sensitive; each step exists because of the
82
104
  step before it.
83
105
 
84
106
  1. **Blast radius** from the *pre-change* graph, so dependents of a file that
85
- just disappeared still get re-extracted.
107
+ just disappeared still get re-extracted. Unbounded by default; see
108
+ [Bounding the blast radius](#bounding-the-blast-radius).
86
109
  2. **Reconcile changed paths.** Every changed path that still exists is handed
87
110
  to the file-based extractors that claim it (`PathDispatcher`), and units the
88
111
  path no longer produces are dropped. This is what indexes a file the index
@@ -119,6 +142,11 @@ step before it.
119
142
  string literal, and an unrelated addition in the same batch do not.
120
143
  Idempotent when nothing was pruned.
121
144
 
145
+ Git enrichment uses the same eligibility checks in full and incremental runs:
146
+ existing app-owned files under `Rails.root`, excluding `vendor/`, `node_modules/`,
147
+ and framework/gem source units. Each typed unit resolves its own file history,
148
+ even when its identifier is shared by another type.
149
+
122
150
  Then the second pass: `dependents` and `metadata.git` are refreshed on every
123
151
  touched unit (the incremental equivalents of full extraction's phases 2 and 4),
124
152
  type indexes are regenerated, the graph, `graph_analysis.json` and the
@@ -130,6 +158,33 @@ A run that changed nothing **does not rewrite the manifest**. The manifest
130
158
  timestamp drives `woods_status.staleness_seconds`, and touching it after a no-op
131
159
  would report the index as freshly synced when nothing was re-read.
132
160
 
161
+ ## Bounding the blast radius
162
+
163
+ `incremental_blast_radius_depth` caps how many reverse hops step 1 walks.
164
+ `nil`, the default, keeps the unbounded transitive closure: every unit that
165
+ reaches the changed file, at any depth, is re-extracted.
166
+
167
+ The reason to cap it is cost. A unit's extracted content is mostly a function
168
+ of its own source and its own reflection, so on most graphs the deep half of
169
+ the closure re-derives bytes that do not change. On a 200-service chain in the
170
+ dummy app, one leaf edit re-extracts 200 units unbounded and 2 at a depth
171
+ of 1.
172
+
173
+ The reason the default is not capped is that "mostly" is not "always". An STI
174
+ grandchild reads its grandparent's reflection: `SportsCar < Car < Vehicle`
175
+ inherits `Vehicle`'s associations, validations and callback chain, and a
176
+ nested `has_many :through` resolves through the same kind of chain. The graph
177
+ records only the one-hop superclass reference each source file mentions, so
178
+ that grandchild sits two hops out while its content depends on hop zero. Set
179
+ the key on a tree you know has neither shape.
180
+
181
+ What the cap never affects is `dependents`. Edges change only when a unit is
182
+ re-extracted, and the run marks every target of a re-extracted unit's edges,
183
+ before and after registration, so a unit that gains or loses an inbound edge
184
+ is rewritten by the second pass whether or not the walk reached it.
185
+ `spec/integration/incremental_equivalence_spec.rb` holds a depth of 1 to
186
+ full-extraction equivalence, including that case.
187
+
133
188
  ## Dispatch inventory
134
189
 
135
190
  ### Per-file
@@ -146,7 +201,8 @@ automatically.
146
201
  | `app/decorators`, `app/presenters`, `app/form_objects` | decorators |
147
202
  | `app/managers` / `app/policies` / `app/validators` | managers / policies + pundit_policies / validators |
148
203
  | `app/**/concerns/**/*.rb` | concerns |
149
- | `app/models/**/*.rb` (outside `concerns/`) | poros, caching |
204
+ | `app/models/**/*.rb` (outside `concerns/`) | poros, caching; concerns when runtime model inclusion confirms a mixin |
205
+ | `app/**/*.rb`, `lib/**/*.rb` (outside `concerns/`) | runtime model mixins also dispatch to concerns |
150
206
  | `app/controllers/**/*.rb` | caching |
151
207
  | `app/views/**/*.erb` | view_templates, caching |
152
208
  | `config/locales/**/*.yml` | i18n |
@@ -230,6 +286,34 @@ oracle compares against emits it too, both sides agree, wrongly. The coverage
230
286
  is in `spec/extractor_spec.rb`, driving the reconciler with a shrinking
231
287
  discovery set.
232
288
 
289
+ ### Runtime removals and bundle updates
290
+
291
+ Jobs discovered through `ApplicationJob.descendants` supplement the job-file
292
+ scan, but jobs are not part of `CLASS_BASED_DISCOVERY` removal reconciliation.
293
+ If a dynamically defined or gem-owned job disappears without a tracked source
294
+ path changing, its unit can survive subsequent incremental runs. A full
295
+ extraction in a fresh Rails process removes it; an in-process full extraction
296
+ can still see an old constant retained by that process (B-165).
297
+
298
+ After adding, removing, or updating bundled gems, boot the updated bundle in a
299
+ fresh process and run:
300
+
301
+ ```bash
302
+ bundle exec rake woods:extract woods:validate
303
+ ```
304
+
305
+ A `Gemfile.lock` change refreshes engines, middleware, and optional framework
306
+ sources. It does not refresh every gem-owned model, job, or other runtime unit.
307
+ Their recorded paths or metadata can remain stale, including absolute paths to
308
+ a removed gem version and paths under `vendor/`. A full extraction rebuilds
309
+ those units against the installed bundle (B-166).
310
+
311
+ An absent external source path can also mean the validator runs on a different
312
+ host or mount from extraction. Confirm the bundle and filesystem context before
313
+ rebuilding; a full run in one container does not make its gem paths visible on
314
+ another host. Validation warnings identify missing paths, but do not prove a
315
+ retained unit matches the currently installed gem when its path still exists.
316
+
233
317
  ### Deletion
234
318
 
235
319
  - Paths named in the change set that no longer exist are **authoritative** for
@@ -274,6 +358,18 @@ them for its **delta**:
274
358
  generation, which payload seeding hardlinks into the run's payload
275
359
  directory.
276
360
 
361
+ Re-assembly is scoped further, because a flow document only reaches
362
+ `FlowPrecomputer::DEFAULT_MAX_DEPTH` units. The run walks the pre-change graph
363
+ to that same depth from its changed files; a re-extracted controller inside
364
+ that radius is re-assembled, and one outside it takes its
365
+ `metadata[:flow_paths]` back from the previous index without paying for the
366
+ assembly. Three cases opt out and re-assemble every re-extracted controller: a
367
+ targeted `Extractor#refresh`, which has no change set; a routes re-run, which
368
+ replaces every controller and moves the route a flow document carries without
369
+ touching any dependency edge; and a controller whose action set no longer
370
+ matches the previous index, which is how an action inherited from further up a
371
+ controller chain than the radius reaches still lands.
372
+
277
373
  After the index is rewritten, a **dedicated flow-artifact sweep** removes
278
374
  every `flows/` document no index entry references. It validates against
279
375
  `flow_index.json` and is deliberately separate from the unit sweep: flows/
@@ -383,6 +479,66 @@ owns the definition of "the two indexes agree" and documents every exclusion.
383
479
 
384
480
  **Run it before and after any change to the incremental path.**
385
481
 
482
+ ## Profiling fixed costs
483
+
484
+ Set `WOODS_PROFILE=1` to time extraction phases. Git enrichment and unit JSON
485
+ finalization are separate from incremental re-extraction; runtime discovery,
486
+ whole-app reruns and pruning appear under `reconciliation`. `payload sync`,
487
+ `publish` (the generation pointer write), and `payload prune` (retention) are
488
+ separate, additive phases. Older versions included sync and retention inside
489
+ `publish`, so do not sum those older lines without subtracting nested sync.
490
+
491
+ `[profile total]` reports time inside the extraction call, including setup and
492
+ failed runs. It is not another phase to sum. Compare it with phase durations
493
+ to find unaccounted work; each line rounds to hundredths of a second. Rails
494
+ boot, Bundler and watch reload work outside the call require separate wall
495
+ measurements. Do not infer that all unaccounted time is boot.
496
+
497
+ Both full and incremental runs currently seed the prior payload. Cloning uses
498
+ per-file hardlinks (or copies where links are unsupported); updated files are
499
+ replaced atomically, preserving previous generations. Full-run carry-forward
500
+ and generation retention remain unchanged. The seed walker classifies each entry
501
+ once, avoiding duplicate file metadata lookups; it still links each file
502
+ separately. Generation directories remain independent, so pruning an old generation cannot remove a newer one's files.
503
+ Compare repeated runs on the actual index filesystem before attributing latency
504
+ to extraction or changing the index layout.
505
+
506
+ For a repeatable component comparison from a source checkout:
507
+
508
+ ```bash
509
+ # BENCH_ROOT is an existing scratch parent on the index filesystem.
510
+ # The script creates and removes only its own temporary directory there.
511
+ BENCH_ROOT=/path/on/index/filesystem WOODS_SOURCE=/path/to/baseline \
512
+ ruby bench/payload_seed.rb
513
+ BENCH_ROOT=/path/on/index/filesystem WOODS_SOURCE=/path/to/candidate \
514
+ ruby bench/payload_seed.rb
515
+ ```
516
+
517
+ The benchmark clones 8,335 synthetic 2 KiB files across 35 directories, checks
518
+ file counts and bytes outside timing, and reports seven samples plus medians.
519
+ This measures the seed component only; it does not boot Rails or establish an
520
+ end-to-end host improvement. Filesystem metadata latency and the copy fallback
521
+ can dominate differently from local hardlink results. A first full extraction
522
+ has no previous payload, so include a repeated full run when measuring seed cost.
523
+ Full seeding also preserves on-demand framework units when framework extraction
524
+ is disabled, and non-JSON files in existing type directories; dropping the seed
525
+ wholesale would change that behavior.
526
+
527
+ ### Choosing full versus incremental for CI
528
+
529
+ Measure repeated full and representative-day incremental runs with
530
+ `WOODS_PROFILE=1` on the same resulting application tree, configuration and index
531
+ filesystem. Restore the same baseline index before each incremental trial;
532
+ otherwise a second trial may be a no-op. Include Rails boot in both wall times
533
+ when comparing separate task invocations, and compare resident watcher cycles
534
+ separately. Validate each resulting index with `woods:validate`.
535
+
536
+ Use full extraction for that workload when its median wall time is no greater
537
+ than the representative incremental run. There is no universal changed-file
538
+ threshold: shared dependencies and whole-app extractor triggers change the work
539
+ per file. Re-measure after substantial application or Woods changes. A fast leaf
540
+ edit does not establish that a day of commits is below the crossover.
541
+
386
542
  ## Boundaries and open work
387
543
 
388
544
  - **Reloaded deletion is supported.** The resident watcher reloads changed