woods 2.0.0.beta3 → 2.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (120) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +500 -420
  3. data/CONTRIBUTING.md +29 -17
  4. data/README.md +78 -178
  5. data/docs/AGENT_GUIDE.md +52 -11
  6. data/docs/AGENT_SETUP.md +34 -17
  7. data/docs/AUTOMATIC_MAINTENANCE.md +222 -0
  8. data/docs/BACKEND_MATRIX.md +18 -7
  9. data/docs/CLIENT_HOOKS.md +1 -1
  10. data/docs/CONFIGURATION_REFERENCE.md +105 -29
  11. data/docs/CONSOLE_MCP_SETUP.md +54 -9
  12. data/docs/DOCKER_SETUP.md +16 -1
  13. data/docs/EVALUATION.md +10 -4
  14. data/docs/EXTRACTOR_REFERENCE.md +23 -3
  15. data/docs/FAQ.md +14 -3
  16. data/docs/GETTING_STARTED.md +18 -17
  17. data/docs/INCREMENTAL_EXTRACTION.md +37 -8
  18. data/docs/INDEX_LAYOUT.md +2 -2
  19. data/docs/MCP_SERVERS.md +79 -7
  20. data/docs/MCP_TOOL_COOKBOOK.md +5 -5
  21. data/docs/MCP_WORKTREE_SETUP.md +55 -83
  22. data/docs/PUBLISHED_INDEX.md +17 -0
  23. data/docs/README.md +2 -1
  24. data/docs/RETRIEVAL_GUIDE.md +81 -13
  25. data/docs/SOURCE_FRESHNESS.md +1 -1
  26. data/docs/TOKEN_BENCHMARK.md +16 -10
  27. data/docs/TROUBLESHOOTING.md +142 -47
  28. data/docs/UPGRADING_TO_2.md +12 -6
  29. data/docs/WATCH_DAEMON.md +189 -24
  30. data/docs/WHY_WOODS.md +9 -5
  31. data/exe/woods-console +13 -11
  32. data/exe/woods-mcp-start +14 -9
  33. data/exe/woods-watch +5 -0
  34. data/lib/generators/woods/pgvector_generator.rb +8 -2
  35. data/lib/generators/woods/watch_generator.rb +53 -0
  36. data/lib/puma/plugin/woods.rb +10 -0
  37. data/lib/tasks/woods.rake +14 -0
  38. data/lib/woods/agent_configuration/applier.rb +5 -3
  39. data/lib/woods/agent_configuration/cli.rb +2 -2
  40. data/lib/woods/agent_configuration/layout.rb +13 -0
  41. data/lib/woods/cache/cache_middleware.rb +6 -0
  42. data/lib/woods/console/credential_scanner.rb +4 -3
  43. data/lib/woods/console/dispatch_pipeline.rb +7 -0
  44. data/lib/woods/console/embedded_executor.rb +31 -9
  45. data/lib/woods/console/sql_noise_stripper.rb +9 -7
  46. data/lib/woods/console/sql_table_scanner.rb +47 -7
  47. data/lib/woods/console/sql_validator.rb +49 -9
  48. data/lib/woods/console/sqlite_read_guard.rb +46 -0
  49. data/lib/woods/console/stdio_transport.rb +27 -0
  50. data/lib/woods/coordination/pipeline_lock.rb +3 -2
  51. data/lib/woods/embedding/indexer.rb +24 -14
  52. data/lib/woods/extractor.rb +70 -19
  53. data/lib/woods/extractors/declared_parent.rb +55 -0
  54. data/lib/woods/extractors/graphql_extractor.rb +2 -11
  55. data/lib/woods/extractors/lib_extractor.rb +10 -8
  56. data/lib/woods/extractors/mailer_extractor.rb +6 -10
  57. data/lib/woods/extractors/model_extractor.rb +1 -15
  58. data/lib/woods/extractors/poro_extractor.rb +10 -8
  59. data/lib/woods/extractors/shared_utility_methods.rb +22 -5
  60. data/lib/woods/git_command.rb +6 -7
  61. data/lib/woods/git_provenance.rb +4 -6
  62. data/lib/woods/mcp/bearer_auth.rb +2 -1
  63. data/lib/woods/mcp/bootstrapper.rb +20 -5
  64. data/lib/woods/mcp/config_resolver.rb +2 -1
  65. data/lib/woods/mcp/index_reader.rb +11 -2
  66. data/lib/woods/mcp/initialization_guidance.rb +1 -1
  67. data/lib/woods/mcp/renderers/markdown_renderer.rb +14 -8
  68. data/lib/woods/mcp/renderers/plain_renderer.rb +11 -7
  69. data/lib/woods/mcp/server.rb +63 -37
  70. data/lib/woods/mcp/tool_contract.rb +1 -1
  71. data/lib/woods/mcp/tool_response_renderer.rb +16 -0
  72. data/lib/woods/mcp/traversal_evidence_text.rb +1 -1
  73. data/lib/woods/mcp/traversal_response.rb +22 -0
  74. data/lib/woods/path_dispatcher.rb +6 -5
  75. data/lib/woods/published_index/typed_unit_reader.rb +40 -3
  76. data/lib/woods/published_index.rb +2 -2
  77. data/lib/woods/rake_helpers.rb +2 -12
  78. data/lib/woods/retrieval/corpus_status.rb +46 -0
  79. data/lib/woods/retrieval/lexical_assembler.rb +14 -3
  80. data/lib/woods/retrieval/lexical_index.rb +2 -1
  81. data/lib/woods/retriever.rb +19 -7
  82. data/lib/woods/session_tracer/file_store.rb +6 -1
  83. data/lib/woods/source_inputs/consumer_errors.rb +4 -0
  84. data/lib/woods/storage/local_corpus_stats.rb +32 -0
  85. data/lib/woods/storage/metadata_store.rb +20 -0
  86. data/lib/woods/storage/pgvector.rb +6 -2
  87. data/lib/woods/storage/vector_store.rb +10 -0
  88. data/lib/woods/temporal/json_snapshot_store.rb +35 -7
  89. data/lib/woods/version.rb +1 -1
  90. data/lib/woods/watch/child_environment.rb +30 -0
  91. data/lib/woods/watch/cli.rb +91 -0
  92. data/lib/woods/watch/daemon.rb +73 -11
  93. data/lib/woods/watch/event_stream.rb +70 -0
  94. data/lib/woods/watch/guardian.rb +142 -0
  95. data/lib/woods/watch/installation/layout.rb +70 -0
  96. data/lib/woods/watch/installation/options.rb +128 -0
  97. data/lib/woods/watch/installation/planner.rb +128 -0
  98. data/lib/woods/watch/installation/probe.rb +101 -0
  99. data/lib/woods/watch/installation/receipt.rb +77 -0
  100. data/lib/woods/watch/installation/recovery.rb +64 -0
  101. data/lib/woods/watch/installation/templates.rb +58 -0
  102. data/lib/woods/watch/installation.rb +56 -0
  103. data/lib/woods/watch/lifecycle.rb +182 -0
  104. data/lib/woods/watch/managed_child.rb +113 -0
  105. data/lib/woods/watch/managed_cleanup.rb +48 -0
  106. data/lib/woods/watch/managed_process.rb +144 -0
  107. data/lib/woods/watch/puma_adapter.rb +87 -0
  108. data/lib/woods/watch/puma_child.rb +66 -0
  109. data/lib/woods/watch/supervision_records.rb +95 -0
  110. data/lib/woods/watch/supervision_status.rb +104 -0
  111. data/lib/woods/watch/supervisor.rb +161 -0
  112. data/lib/woods/watch/supervisor_reporting.rb +46 -0
  113. data/plugin/.claude-plugin/plugin.json +1 -1
  114. data/plugin/hooks/woods-input-rules.sh +4 -4
  115. data/plugin/skills/woods-agent-enable/SKILL.md +7 -1
  116. data/plugin/skills/woods-diagnose/SKILL.md +134 -34
  117. data/plugin/skills/woods-investigate/SKILL.md +54 -15
  118. data/plugin/skills/woods-mcp-config/SKILL.md +38 -11
  119. data/plugin/skills/woods-setup/SKILL.md +72 -15
  120. metadata +38 -5
data/docs/AGENT_GUIDE.md CHANGED
@@ -5,7 +5,7 @@ This guide is for coding agents using an already connected Woods MCP server. Woo
5
5
  Supporting servers also send a concise version of this workflow in MCP
6
6
  initialization/discovery instructions, without requiring an installed plugin.
7
7
  Check the connected server's version and registered tools; this feature is
8
- unreleased after `2.0.0.beta2`, and protocol `2024-11-05` omits the field.
8
+ included in Woods `2.0.0`, and protocol `2024-11-05` omits the field.
9
9
  The [initialization contract](MCP_SERVERS.md#initialization-guidance) describes
10
10
  availability. This guide remains the detailed reference when instructions are
11
11
  absent or the client does not display them.
@@ -16,7 +16,7 @@ Call `woods_status` before relying on the index. Check:
16
16
 
17
17
  - the index is ready and has a current generation;
18
18
  - unit counts are non-zero for relevant types;
19
- - retrieval is enabled before choosing `codebase_retrieve`;
19
+ - the retrieval mode and its data are usable before choosing `codebase_retrieve`;
20
20
  - warnings do not indicate a stale or partial index.
21
21
 
22
22
  If status is unhealthy, report the evidence and ask the owner to extract or refresh. Do not fill gaps by asserting that Woods found nothing.
@@ -35,7 +35,9 @@ Use this four-step loop for most codebase questions:
35
35
  3. **Traverse** from that identifier with `dependencies`, `dependents`, or `trace_flow`.
36
36
  4. **Verify** important claims against the returned source paths and current repository files.
37
37
 
38
- Identifiers are namespaced and typed. Never invent one from a filename when `search` can return the exact value.
38
+ Identifiers are namespaced and typed. Never invent one from a filename when
39
+ `search` can return the exact value. Carry both the returned `identifier` and
40
+ `type` into `lookup`; the same identifier can belong to more than one unit type.
39
41
 
40
42
  ## Pick the smallest useful tool
41
43
 
@@ -54,7 +56,7 @@ Identifiers are namespaced and typed. Never invent one from a filename when `sea
54
56
  | Find dependencies that change faster than their dependents | `graph_analysis` with `analysis: "volatile_dependencies"` | `recent_changes` |
55
57
  | See a Packwerk boundary before calling across it | `graph_analysis` with `analysis: "undeclared_package_edges"` | `lookup` on the package unit |
56
58
  | Find central or high-impact units | `pagerank` | `dependents` |
57
- | Ask a conceptual question | `codebase_retrieve` if status says ready | `lookup` and graph tools |
59
+ | Ask a conceptual question | `codebase_retrieve` after checking its retrieval mode and data | `lookup` and graph tools |
58
60
  | Refresh after a published extraction | `reload` | `woods_status` |
59
61
 
60
62
  Do not start with a broad graph or semantic query when an exact search will answer the question with less noise.
@@ -64,7 +66,7 @@ Do not start with a broad graph or semantic query when an exact search will answ
64
66
  ### Understand a model
65
67
 
66
68
  1. `search(query: "^Order$", types: ["model"])`
67
- 2. `lookup(identifier: <returned identifier>)`
69
+ 2. `lookup(identifier: <returned identifier>, type: <returned type>)`
68
70
  3. Read resolved schema, associations, validations, scopes, enums, callbacks, and included concerns.
69
71
  4. `dependencies(identifier: ..., depth: 1)` for collaborators.
70
72
  5. `dependents(identifier: ..., depth: 1)` for callers and affected features.
@@ -74,11 +76,21 @@ Woods may inline concern behavior beside the owning model. Distinguish the resol
74
76
  ### Trace a feature flow
75
77
 
76
78
  1. Search for the route, controller action, job, mailer, or service at the user-visible entry point.
77
- 2. Call `trace_flow` on the exact identifier.
79
+ 2. Call `trace_flow(entry_point: "UnitIdentifier#method")` for a method on that exact indexed unit, or use `UnitIdentifier` for the whole unit. For example, use `CheckoutService#order` for its `order` method.
78
80
  3. Inspect important or ambiguous nodes with `lookup`.
79
81
  4. Follow missing branches with `dependencies` and a narrow `via` filter when useful.
80
82
  5. Verify behavior that depends on conditions, dynamic dispatch, or runtime data in source and tests.
81
83
 
84
+ Bare names identify units, not methods across the application. For example,
85
+ `trace_flow(entry_point: "order")` selects the indexed `order` unit, which may
86
+ be a FactoryBot factory in `spec/factories/`. Find the owning class with
87
+ `search` and `lookup`, then pass its identifier with `#order`.
88
+
89
+ Flow assembly derives operations from source. A receiverless local call such
90
+ as `order` inside `CheckoutService#call` may remain visible without expanding
91
+ the local method body. Follow it in source or trace `CheckoutService#order`
92
+ explicitly. A flow is not proof of runtime execution or exhaustive call coverage.
93
+
82
94
  ### Assess change impact
83
95
 
84
96
  1. Search and look up the unit being changed.
@@ -123,24 +135,44 @@ an exact total within the requested index/query domain. Narrow types, literal
123
135
  prefix/suffix filters, or deep fields when `partial` is true. A detected artifact
124
136
  failure remains an error with unknown completeness, never proof of no matches.
125
137
 
126
- This metadata is unreleased after `2.0.0.beta2`; older servers may omit it.
138
+ This metadata is included in Woods `2.0.0`; older servers may omit it.
127
139
  Do not infer completeness from a full page or missing metadata. See the
128
140
  [search response contract](MCP_SERVERS.md#search-completeness).
129
141
 
130
142
  ## Traverse deliberately
131
143
 
132
- `dependencies` means “what this unit uses.” `dependents` means “what uses this unit.” Both default to bounded breadth-first traversal and accept type or relationship filters.
144
+ `dependencies` shows recorded relationships from this unit; `dependents` shows
145
+ recorded relationships to it. Both default to bounded breadth-first traversal and
146
+ accept type or relationship filters. They are not exhaustive source-reference or
147
+ call graphs: selective scanning can miss arbitrary method-body constant references,
148
+ including generic PORO and library targets. No dependents or test-only dependents
149
+ do not establish absence of production callers. Check source before claiming absence.
150
+ The `graph_coverage` notice makes this scope explicit in supporting responses;
151
+ that metadata is included in Woods `2.0.0`.
133
152
 
134
153
  Start at depth 1 or 2. A deeper unfiltered traversal can obscure the direct evidence that matters. Common relationship values include associations (`belongs_to`, `has_many`, `has_one`), code references, renders, redirects, form actions, and navigation links.
135
154
 
136
155
  Both return at most 50 nodes by default and say so with a `Showing N of M (truncated)`
137
- line. Narrow with `depth`, `types` and `via` before paging with `limit` and
156
+ line for completed walks. Budget-limited answers in supporting versions instead
157
+ say `Showing N of at least M (total unknown: <budget reason>)`. Narrow with `depth`, `types` and `via` before paging with `limit` and
138
158
  `offset`: narrowing answers the question, paging only splits the same answer
139
159
  across turns. In a multi-database app each row names the unit's database.
140
160
 
161
+ On supporting servers, inspect `structuredContent.data` for traversal nodes,
162
+ `graph_coverage`, exactness, budgets and optional explanation witnesses, regardless
163
+ of the text renderer. This packaged stdio/HTTP payload is included in Woods
164
+ `2.0.0`; check the actual response and use its text when structured data is
165
+ absent. There is no traversal `format` argument. See the
166
+ [response contract](MCP_SERVERS.md#dependency-graph-coverage).
167
+
141
168
  A traversal can also stop at its independent node or edge budget. Treat
142
169
  `partial`/`partial_reason` as incomplete graph evidence even on the final page;
143
- paging cannot recover nodes the walk never reached. Check the connected schema
170
+ paging cannot recover nodes the walk never reached. Supporting responses include
171
+ `total_is_exact: false` for a cutoff and true for a finished walk, independently of
172
+ pagination. This field is included in Woods `2.0.0`; on older servers inspect
173
+ `partial` directly. `nodes_total` remains the root-inclusive admitted prefix count,
174
+ not the full reachable total when partial. Even an exact count covers only the
175
+ requested root, depth, filters and published graph generation. Check the connected schema
144
176
  before using `max_nodes`/`max_edges`, and follow the
145
177
  [budget contract](MCP_SERVERS.md#dependency-traversal-budgets).
146
178
 
@@ -149,7 +181,9 @@ recorded source-to-target relationships and a shared shortest witness to each
149
181
  row. Report `direct` relationships separately from `transitive` inferred impact.
150
182
  Follow `parent`/`edge_id` references; `context: true` ancestors are outside the
151
183
  current result page. Unknown labels and ambiguous candidate types stay unknown;
152
- `typed_path_complete: false` does not establish a uniquely typed path. See the
184
+ `typed_path_complete: false` does not establish a uniquely typed path. True means
185
+ only unambiguous witness types, not complete source coverage. Supporting text
186
+ responses label this `witness types unambiguous` (included in Woods `2.0.0`). See the
153
187
  [explanation contract](MCP_SERVERS.md#traversal-explanations).
154
188
 
155
189
  Use recorded relationship labels as evidence. Do not infer execution or call
@@ -159,6 +193,13 @@ order from a dependency edge alone.
159
193
 
160
194
  `codebase_retrieve` answers natural-language questions with token-budgeted context. Use it when `woods_status` reports explicit lexical mode over a current published index, or a configured embedding provider and current vector data in semantic mode. Lexical mode explains matching terms/fields and does not infer synonyms absent from the text; a no-match response is not proof of missing behavior.
161
195
 
196
+ Top-level `ready` describes the structural index. Bootstrap `hydrated` does not
197
+ prove that semantic stores contain data. When supported, inspect
198
+ `retriever.corpus`; missing or unknown counts require checking embedding
199
+ artifacts. Known-empty stores need embedding or an explicit switch to lexical
200
+ mode. Positive record counts do not prove complete application coverage. See
201
+ [semantic corpus diagnostics](RETRIEVAL_GUIDE.md#semantic-corpus-diagnostics).
202
+
162
203
  Important parameters:
163
204
 
164
205
  - `query`: the conceptual question;
data/docs/AGENT_SETUP.md CHANGED
@@ -40,7 +40,9 @@ If the worktree contains unrelated changes, preserve them. Do not overwrite an e
40
40
 
41
41
  Use structural-only setup when the user wants code navigation, runtime Rails structure, dependencies, flows, or blast-radius analysis. Fourteen tools register in the normal packaged launch without an embedding provider.
42
42
 
43
- Discuss semantic retrieval only if the user needs natural-language `codebase_retrieve`. The choice depends on whether they prefer local Ollama or hosted OpenAI and which vector store fits their environment. See [Backend matrix](BACKEND_MATRIX.md).
43
+ If the user wants ranked discovery through `codebase_retrieve`, offer [explicit lexical mode](RETRIEVAL_GUIDE.md#embedding-free-lexical-retrieval) over the published index without a provider or embeddings. Check that the installed version supports it, set `WOODS_RETRIEVAL_MODE=lexical` in the MCP process environment, restart that server, and verify `woods_status.retriever.mode`. Keep structural-only setup as the default unless this mode is requested.
44
+
45
+ For semantic matching, discuss local Ollama or hosted OpenAI and the appropriate vector store separately; adding a provider still requires authorization. See [Backend matrix](BACKEND_MATRIX.md).
44
46
 
45
47
  Do not infer permission to configure Console MCP from a request to “set up Woods” or “set up MCP.” The Index Server reads generated code context; the Console Server can read live data.
46
48
 
@@ -49,7 +51,7 @@ Do not infer permission to configure Console MCP from a request to “set up Woo
49
51
  Create or switch to the branch requested by the repository owner. Select the
50
52
  published version using the [installation guide](GETTING_STARTED.md#1-install-the-gem).
51
53
  Before stable 2.x is published, use the exact published prerelease constraint
52
- from the README release table; `~> 2.0` will not select a beta or release candidate.
54
+ from RubyGems; `~> 2.0` will not select a beta or release candidate.
53
55
  Use the selected version's tag documentation and verify its capabilities before
54
56
  configuring features described on `main`.
55
57
 
@@ -130,7 +132,7 @@ Reconnect the client and call `woods_status`. Confirm a current generation and n
130
132
 
131
133
  ### Managed Claude Code configuration
132
134
 
133
- The development command `woods-agent-config` is unreleased after 2.0.0.beta2.
135
+ `woods-agent-config` is available from Woods `2.0.0.beta3`.
134
136
  Check `bundle exec woods-agent-config --help` in the selected application bundle;
135
137
  use the manual client configuration below when it is absent. The supported
136
138
  client format is Claude Code (tested with 2.1.267).
@@ -188,8 +190,16 @@ receipt for future update/removal. Unrelated servers, hooks, settings,
188
190
  instruction text, permissions, and line-ending conventions are retained;
189
191
  changing JSON may reformat its whitespace.
190
192
 
193
+ Included in Woods `2.0.0`: apply and recovery coordinate on the actual
194
+ managed file paths, including user configuration and shared instruction files.
195
+ Two application roots sharing those files cannot apply overlapping plans at the
196
+ same time. A competing operation reports a conflict; after it finishes, create a
197
+ fresh preview if the saved plan's snapshots changed. Both applications keep
198
+ their own ownership receipts. Do not delete an active coordination lock.
199
+
191
200
  Writes use atomic replacement per file and a private recovery journal beside
192
- the receipt. The plan summary names the `.lock` and `.pending` runtime paths;
201
+ the receipt. The plan summary names all adjacent `.woods.lock` files, the
202
+ receipt `.lock`, and the `.pending` journal;
193
203
  a lock file may remain after completion. Multiple files are not one atomic
194
204
  transaction. An ordinary write failure restores original files when safe; an
195
205
  interruption or concurrent edit can retain the journal. Resolve reported
@@ -203,26 +213,29 @@ is changed; reduce the selected configuration before applying.
203
213
 
204
214
  Use a class known to exist in the application:
205
215
 
206
- 1. Call `search` to obtain its exact identifier.
207
- 2. Call `lookup` to confirm source and metadata are present.
216
+ 1. Call `search` to obtain its exact identifier and type.
217
+ 2. Call `lookup` with that identifier and type to confirm source and metadata are present.
208
218
  3. Call `dependents` with depth 1 or 2 to confirm graph edges are queryable.
209
219
 
210
- If `codebase_retrieve` reports that semantic search is disabled, that is expected for structural-only setup. Do not configure credentials merely to remove the message.
220
+ If `codebase_retrieve` reports that semantic search is disabled, that is expected for structural-only setup. Do not configure credentials merely to remove the message. If lexical retrieval was requested, verify its mode with `woods_status` and make one `codebase_retrieve` call against the published index.
211
221
 
212
222
  ## 8. Offer automatic index maintenance
213
223
 
214
- Ask whether the owner wants Woods added to the development process manager. If authorized, use the repository's existing Procfile or equivalent convention:
215
-
216
- ```text
217
- web: bin/rails server
218
- woods: bundle exec rake woods:watch
219
- ```
224
+ Within the owner's setup authorization, select one development startup owner.
225
+ Follow [managed startup](WATCH_DAEMON.md#managed-development-startup): Puma for
226
+ simple Rails startup, a verified existing Foreman command/Procfile, or the existing
227
+ external Docker/Grove supervisor. The launcher/generator are **included in Woods
228
+ `2.0.0`**; check installed `woods-watch --help` and generator help first.
229
+ Preserve `bin/dev`; a file that only starts Rails does not consume a Procfile.
230
+ For older gems use their raw task with a restart-capable external supervisor,
231
+ not a bare Foreman entry.
220
232
 
221
233
  The watcher catches up missed changes, maintains the structural index as files change, and publishes generations the Index MCP server detects automatically. Ordinary edits then need no manual re-extraction or MCP restart. It should run in development, not production.
222
234
 
223
235
  Report these boundaries in the handoff:
224
236
 
225
- - boot-captured changes make the watcher exit 75 and require supervisor restart;
237
+ - raw tasks exit 75 for boot-captured changes; managed launchers absorb that restart;
238
+ - managed mode rejects idle TTL and parks ownership conflicts without taking over;
226
239
  - container bind mounts may require `WOODS_WATCH_POLL=1`;
227
240
  - semantic vectors still require `woods:embed_incremental`;
228
241
  - without a resident watcher, the fallback is `woods:incremental` after changes.
@@ -263,9 +276,13 @@ Verified capabilities:
263
276
  - Index Server connected: yes/no
264
277
  - woods_status current: yes/no
265
278
  - search/lookup/dependents checked: yes/no
266
- - semantic retrieval: disabled/enabled (provider)
279
+ - retrieval: disabled/lexical/semantic (provider when semantic)
267
280
  - Console MCP: disabled/enabled (authorization)
268
- - automatic structural updates: disabled/enabled (process manager)
281
+ - automatic structural updates: disabled/enabled (owner and actual startup command)
282
+ - served root/index and completed startup catch-up:
283
+ - edit and planned restart observed through existing MCP connection:
284
+ - worktree/Grove switching verified (if applicable):
285
+ - post-edit hooks / session freshness checks / context hints: separate enablement
269
286
 
270
287
  Follow-up or unresolved risk:
271
288
  ```
@@ -274,7 +291,7 @@ Never report a capability as enabled solely because its schema exists in source.
274
291
 
275
292
  ## Copyable prompt for an installation agent
276
293
 
277
- > Install Woods 2.x in this Rails repository using `docs/AGENT_SETUP.md`. Start with read-only preflight and preserve unrelated changes. Default to the structural Index Server; do not enable embeddings, Console MCP, HTTP transport, secrets, or purge overrides without asking me. Inspect generated files before migrating, run extraction and validation in the app's normal execution environment, configure a project-scoped MCP server in the same filesystem context as the application bundle and index, and verify `woods_status`, `search`, `lookup`, and `dependents`. Finish with the runbook's handoff report.
294
+ > Install Woods 2.x in this Rails repository using https://github.com/lost-in-the/woods/blob/main/docs/AGENT_SETUP.md. Select a published version and follow that version's tag documentation and supported capabilities. Start with read-only preflight and preserve unrelated changes. Default to the structural Index Server; do not enable embeddings, Console MCP, HTTP transport, secrets, or purge overrides without asking me. Inspect generated files before migrating, run extraction and validation in the app's normal execution environment, configure a project-scoped MCP server in the same filesystem context as the application bundle and index, and verify `woods_status`, `search`, `lookup` with the discovered identifier and type, and `dependents`. Finish with the runbook's handoff report.
278
295
 
279
296
  ## Related guides
280
297
 
@@ -0,0 +1,222 @@
1
+ # Woods with minimal manual maintenance
2
+
3
+ **Recommended development setup:** create a baseline index, run one supervised
4
+ watcher per active application/worktree index, and let your MCP client launch the
5
+ Index Server. The watcher publishes changes; an already connected Index Server
6
+ reads the new generation on its next call. Ordinary edits then need no manual
7
+ extraction or MCP restart.
8
+
9
+ This page maps that workflow to its canonical documentation. The raw watcher and
10
+ hooks are available in Woods `2.0.0.beta4`; the managed launcher, installation
11
+ generator, and Puma adapter are **included in Woods `2.0.0`**. Check the
12
+ installed gem/revision and plugin separately. Installing the gem or registering
13
+ an MCP connection does **not** enable automatic index maintenance.
14
+
15
+ ## 1. Where each part is documented
16
+
17
+ | What you need | Documentation section | What it defines |
18
+ |---|---|---|
19
+ | First extraction and validation | [Getting started: extract](GETTING_STARTED.md#3-extract-the-application) and [validate](GETTING_STARTED.md#4-validate-and-inspect-the-index) | Establish a usable baseline before connecting tools. |
20
+ | The simplest automatic workflow | [Getting started: keep the index current](GETTING_STARTED.md#keep-the-index-current) | Watcher beside Rails, process-manager example, automatic reader refresh. |
21
+ | Agent-operated installation | [Agent setup: automatic maintenance](AGENT_SETUP.md#8-offer-automatic-index-maintenance) | Add a watcher using the application's existing development process convention. |
22
+ | Starting and supervising the watcher | [Managed development startup](WATCH_DAEMON.md#managed-development-startup), [raw task](WATCH_DAEMON.md#running-it), and [restart triggers](WATCH_DAEMON.md#restart-triggers) | Native/Puma installation, external supervision, and restart after exit 75. |
23
+ | Updating or removing automatic startup | [Owned installation](WATCH_DAEMON.md#ownership-updates-and-removal) | Portable receipt, preserved application files, explicit updates/removal, and worktree-safe ownership. |
24
+ | Catching changes made while stopped | [Watch daemon: startup reconciliation](WATCH_DAEMON.md#startup-is-not-a-clean-slate) | Catch-up before normal watching, including missed changes and deletions. |
25
+ | Running inside Docker | [Docker: extraction](DOCKER_SETUP.md#extraction) and [index persistence](DOCKER_SETUP.md#index-persistence) | Container-side execution, polling, shared paths, and persistent output. |
26
+ | Connecting the reader | [MCP: stdio client](MCP_SERVERS.md#configure-a-stdio-client) and [Docker process/path rule](MCP_SERVERS.md#docker-process-and-path-rule) | Client-owned process startup and paths visible to that process. |
27
+ | Seeing new generations without reconnecting | [Watch daemon: freshness contract](WATCH_DAEMON.md#the-freshness-contract) | Publication and automatic reader refresh. |
28
+ | Optional refresh after agent edits | [Edit client adapters](CLIENT_HOOKS.md) and [hook operation](WATCH_DAEMON.md#hooks-for-agent-sessions) | Registration, supported edit events, durable queue, retries, and Docker execution. |
29
+ | Session checks and optional context hints | [Source freshness: containers and hooks](SOURCE_FRESHNESS.md#containers-and-hooks) and [bounded context hints](WATCH_DAEMON.md#optional-bounded-context-hints) | Freshness warnings and separately enabled context delivery. |
30
+ | Defaults and environment variables | [Watch settings](CONFIGURATION_REFERENCE.md#watch-daemon-woodswatch) and [plugin hook settings](CONFIGURATION_REFERENCE.md#opt-in-plugin-refresh-hooks) | Authoritative variable names, defaults, and enablement rules. |
31
+ | Worktrees and separate containers | [Multiple worktrees](WATCH_DAEMON.md#multiple-worktrees), [cross-host liveness](WATCH_DAEMON.md#cross-host-liveness), and [MCP worktree setup](MCP_WORKTREE_SETUP.md) | Separate index ownership, reader registration, and heartbeat trust. |
32
+ | Grove-managed worktree switching | [Grove integration below](#grove-coordinate-the-watcher-with-worktree-switches) and [Grove's external Docker integration](https://github.com/lost-in-the/grove/blob/cf833e65c6d2525b0fafa91ebeea9be6455decdb/plugins/docker/README.md#external-mode) | Include the watcher in the services Grove starts against the selected worktree. |
33
+ | Diagnosing stale or uncertain answers | [Source freshness: read the result](SOURCE_FRESHNESS.md#read-the-result) and [troubleshooting](TROUBLESHOOTING.md#source-freshness-is-unknown-or-drifted) | Distinguish content drift, incomplete evidence, age, and daemon health. |
34
+
35
+ ## 2. What starts what
36
+
37
+ | Component | Started by | Responsibility |
38
+ |---|---|---|
39
+ | Rails application | Your development process manager or Compose | Runs the application; a configured development Puma adapter also starts Woods' separate launcher. |
40
+ | Managed `woods-watch` | Selected Foreman command or opt-in Puma adapter | Owns a fresh extraction child and handles planned restarts without stopping Rails. |
41
+ | `woods:watch` | A separately configured supervisor/service | Boots Rails, catches up missed changes, watches files, and publishes structural generations. |
42
+ | Index MCP over stdio | Your editor or agent's MCP client | Reads the published index. It does not boot Rails or launch a watcher. |
43
+ | Post-edit refresh hook | A supported client edit event, after opt-in | Requests extraction for relevant edits; an active watcher can cause it to defer. |
44
+ | Session-start hook | A supported client session event, after opt-in | Checks source freshness and warns. It does not extract or revive a stopped watcher. |
45
+ | Optional context hook | Supported client events, after separate opt-in | Supplies bounded orientation or impact hints. It does not enable refresh. |
46
+
47
+ ```mermaid
48
+ flowchart LR
49
+ Supervisor[Development supervisor] --> Watcher[Woods watcher]
50
+ Files[Application file changes] --> Watcher
51
+ Watcher --> Index[Published generation]
52
+ Client[MCP client session] --> Reader[Index MCP process]
53
+ Index -->|Read on subsequent calls| Reader
54
+ ```
55
+
56
+ Console MCP is a separate, optional live-data tool. Its HTTP transport can be
57
+ mounted in Rails when enabled, but that does not maintain the structural index.
58
+ See [the server distinction](MCP_SERVERS.md#choose-a-server).
59
+
60
+ ## 3. Recommended setup for daily work
61
+
62
+ 1. **Create and validate the initial index.** Follow the first two links in the
63
+ table. If verified fresh-boot provenance matters, use the
64
+ [fresh-process extraction launcher](SOURCE_FRESHNESS.md#establish-a-fresh-baseline).
65
+ 2. **Add a resident watcher to development startup.** Follow the canonical
66
+ [mode selection and preview](WATCH_DAEMON.md#managed-development-startup): Puma
67
+ for simple Rails startup, verified Foreman for an existing Procfile workflow,
68
+ or the existing external supervisor for Docker/Grove. Preserve `bin/dev` and
69
+ record the actual startup command. Keep catch-up enabled and idle TTL unset.
70
+ 3. **Make writer and reader share the intended index.** In Docker, preserve the
71
+ application's required source, bundle, configuration, and git mounts. Persist
72
+ the index across container replacement. Use polling when bind-mount events
73
+ are unreliable. Follow the linked Docker guide instead of copying a generic
74
+ service that might replace required inherited settings.
75
+ 4. **Register the Index MCP client once for the correct worktree.** The client
76
+ starts the reader for its session. A container-based launcher requires its
77
+ target container to be running; registering MCP does not start Docker.
78
+ 5. **Prove one automatic update.** Make a reversible edit, observe successful
79
+ publication, and query the changed unit through the existing MCP connection.
80
+ Repeat once across watcher restart and worktree switch if those are routine.
81
+
82
+ For sustained work, the resident watcher detects changes to supported application
83
+ files regardless of which editor or agent made them. Do not enable post-edit
84
+ refresh hooks merely to make the watcher faster: hooks can defer to it and retain
85
+ queued events.
86
+
87
+ The watcher maintains the structural index. Semantic vectors need their own
88
+ `woods:embed_incremental` workflow. For provider-free ranked retrieval, consider
89
+ [explicit lexical mode](RETRIEVAL_GUIDE.md#embedding-free-lexical-retrieval), which
90
+ uses published extraction units without a separate embedding pipeline.
91
+
92
+ ### Docker: verify the resolved service
93
+
94
+ Inspect the result of `docker compose config` using the same files, environment,
95
+ and active worktree as normal startup. Keep resolved secrets out of shared logs.
96
+
97
+ - **Preserve inherited settings.** Within a YAML anchor merge, an explicit
98
+ `volumes:` or `environment:` key replaces that inherited value. Across Compose
99
+ files, volume entries merge by container target. A separate override can add
100
+ the index mount while retaining the base service's application and bundle
101
+ mounts. Verify the resolved result, including any inherited ports or web-only
102
+ healthcheck. See Docker's [YAML fragments](https://docs.docker.com/reference/compose-file/fragments/)
103
+ and [Compose-file merge rules](https://docs.docker.com/reference/compose-file/merge/).
104
+ - **Choose restart behavior deliberately.** `on-failure` restarts exit 75, but
105
+ does not restart the container after the Docker daemon restarts. For recovery
106
+ after Docker/OrbStack restarts while respecting an intentional stop, use
107
+ `unless-stopped`. See [Docker restart policies](https://docs.docker.com/engine/containers/start-containers-automatically/).
108
+ - **Check database readiness.** The short `depends_on` form orders startup; it
109
+ does not wait for the database to accept connections. Use the application's
110
+ existing readiness convention, such as `service_healthy` with a database
111
+ healthcheck, and reference the Compose service key. See
112
+ [Compose startup order](https://docs.docker.com/compose/how-tos/startup-order/).
113
+ - **Verify catch-up and worktree switching.** A running container does not prove
114
+ its first extraction finished. Confirm successful publication and the intended
115
+ source/index paths before calling the setup current. Recreate the watcher with
116
+ the other application services when the active worktree changes.
117
+
118
+ Use the application's actual Rails task entrypoint. An application whose root
119
+ `Rakefile` wraps Docker may require `bundle exec rails woods:watch` instead of
120
+ `bundle exec rake woods:watch`; verify the command inside the application container.
121
+
122
+ ### Grove: coordinate the watcher with worktree switches
123
+
124
+ When using [Grove](https://github.com/lost-in-the/grove) with an external Compose
125
+ stack, Grove supplies the worktree-switch lifecycle. Its Docker integration
126
+ persists the selected worktree in the configured environment file and runs
127
+ `docker compose up -d` for the configured services. Compose recreates services
128
+ whose resolved configuration changed, including changed bind-mount paths.
129
+
130
+ - Add the watcher service to the existing
131
+ `[plugins.docker.external].services` list in `.grove/config.toml`, preserving
132
+ the other entries. The service must also exist in the Compose configuration.
133
+ - Keep the application's source mount and index mount tied to the same
134
+ worktree variable configured by `env_var`. Confirm the MCP launcher reads
135
+ that worktree's index too.
136
+ - Check `plugins.docker.enabled`, `auto_start`, `auto_stop`, and
137
+ `switch.container_switch`. With automatic lifecycle enabled, a normal
138
+ `grove to <worktree>` switch stops the configured old services and brings them
139
+ up against the selected tree. Updating the environment file alone does not
140
+ reconfigure an already running container.
141
+ - If using isolated agent stacks, check their separate
142
+ `[plugins.docker.external.agent].services` list and Compose template as well;
143
+ the shared stack's watcher configuration is not sufficient for those slots.
144
+
145
+ The ownership chain is **Grove selects the worktree and starts its services →
146
+ Docker supervises the watcher → Woods catches up and publishes file changes**.
147
+ The MCP client starts the reader, which reloads published generations. This setup
148
+ needs no additional Woods-starting Grove hook when its Docker integration already
149
+ manages the watcher. Verify a switch by observing an edit in the new tree while
150
+ the previous tree's index stays unchanged.
151
+
152
+ See Grove's [Docker lifecycle documentation](https://github.com/lost-in-the/grove/blob/cf833e65c6d2525b0fafa91ebeea9be6455decdb/plugins/docker/README.md#hook-integration)
153
+ and [external-stack configuration examples](https://github.com/lost-in-the/grove/blob/cf833e65c6d2525b0fafa91ebeea9be6455decdb/docs/CONFIGURATION_REFERENCE.md#rails-project-with-external-docker).
154
+ These links pin the Grove revision checked for this guide; compare with
155
+ `grove version` when diagnosing a different installation.
156
+
157
+ ### Cross-container status is different from an extraction lock
158
+
159
+ When a watcher and a status-checking process share an index but have different
160
+ hostnames, `WOODS_WATCH_TRUST_FOREIGN_HOST=1` lets that process use the watcher's
161
+ fresh heartbeat. Set it inside the relevant MCP/task container or pass it through
162
+ the launcher; setting it only in the host shell or watcher does not configure
163
+ other containers. This is bounded heartbeat trust, not proof the remote process
164
+ still exists: a stopped daemon can remain believable for up to 15 minutes.
165
+
166
+ | Operation | Effect of the running watcher |
167
+ |---|---|
168
+ | Ordinary Index MCP reads | Read published generations without acquiring the extraction writer lock. Foreign-host trust is not required to read or reload an index. |
169
+ | `woods_status` / `woods:watch_status` | Foreign-host trust affects daemon liveness reporting. A live but degraded daemon is not evidence that updates are succeeding. |
170
+ | `woods:incremental` | Can stand down when a trusted running daemon covers the index; a degraded daemon does not provide that coverage. |
171
+ | Post-edit hook refresh | Can defer with exit 75 while retaining its queued work. |
172
+ | Manual `woods:extract` or named refresh | Acquires the shared writer lock. An idle watcher does not hold it; an active extraction can make the manual task wait. |
173
+
174
+ The default writer-lock wait is up to 600 seconds, configurable with
175
+ `WOODS_LOCK_WAIT`; failure to acquire it exits nonzero. It is not a mandatory
176
+ delay whenever a watcher exists. Heartbeat trust does not bypass the lock.
177
+ See [writer coordination](WATCH_DAEMON.md#within-one-worktree-writers-serialize)
178
+ and [cross-host liveness](WATCH_DAEMON.md#cross-host-liveness).
179
+
180
+ ## 4. When hooks are useful instead
181
+
182
+ Post-edit hooks suit occasional agent edits when keeping another Rails process
183
+ resident is undesirable. They cover the registered client events, not arbitrary
184
+ shell commands, git operations, or edits from another application. They are not
185
+ a substitute for a watcher when all filesystem changes need automatic coverage.
186
+
187
+ Follow [client registration](CLIENT_HOOKS.md), then verify:
188
+
189
+ - `WOODS_HOOKS_ENABLED=1` reaches the process launching the client;
190
+ `WOODS_HOOKS_DISABLED=1` overrides it.
191
+ - The installed gem supports the hook task and a baseline index already exists.
192
+ - `WOODS_HOOK_RAKE` runs in the application's real environment; Docker-only
193
+ bundles need a container command prefix or wrapper.
194
+ - The hook can see the intended index and map edits to the correct application.
195
+ - A supported edit produces a successful publication, not just a hook callback.
196
+ Inspect `hook.log` and `hook-pending/` when work is deferred or fails.
197
+
198
+ `WOODS_HOOK_CONTEXT_ENABLED` is a separate opt-in. Neither context hints nor a
199
+ quiet session-start check establish that anything is refreshing the index.
200
+ Use the settings and recovery links above for the full contract.
201
+
202
+ ## 5. A low-interaction setup is complete when
203
+
204
+ - Development startup starts the intended watcher through its supervisor.
205
+ - Missed changes are reconciled when the watcher starts again.
206
+ - Edits, creates, and deletes appear through the existing MCP connection.
207
+ - A worktree switch changes both the watched source and the served index.
208
+ - A failed edit retains the last good generation and reports the degraded state;
209
+ correction allows recovery without discarding the index.
210
+ - The handoff separately records watcher supervision, post-edit hooks, session
211
+ checks, context hints, MCP registration, and any embedding refresh workflow.
212
+
213
+ Check `woods_status` before relying on indexed facts. Daemon liveness and an old
214
+ publication timestamp do not by themselves establish source freshness. Even a
215
+ working watcher can report `unknown` source evidence, including an unverified
216
+ boot boundary; read the reason before deciding what to refresh. See
217
+ [source-freshness scope](SOURCE_FRESHNESS.md#scope-and-partial-extraction).
218
+
219
+ Automatic revival after a raw task's idle shutdown requires a separately
220
+ configured supervisor or startup hook; the shipped SessionStart hook does not
221
+ do it. Managed mode rejects idle TTL. Confirm the selected startup integration
222
+ and a successful publication before claiming automatic maintenance is active.
@@ -35,14 +35,20 @@ The shape determines the capability matrix:
35
35
 
36
36
  ### Database compatibility
37
37
 
38
- The vector store you can use depends on the primary database your Rails app uses. MySQL stacks **must** pair with an external vector backend; PostgreSQL stacks have the option of running pgvector inside the same database.
38
+ Vector storage is independent of the application's primary database. Structural
39
+ MCP tools and explicit lexical retrieval need no vector backend. For semantic
40
+ retrieval, any supported application database can use `:in_memory`, Qdrant, or
41
+ pgvector through a live PostgreSQL connection.
39
42
 
40
- | Primary database | Supported vector stores | Required? |
43
+ | Primary database | Supported vector stores | PostgreSQL connection for pgvector |
41
44
  |---|---|---|
42
- | **MySQL / Percona / MariaDB / Aurora MySQL** | `:qdrant` (external); `:in_memory` (local dev only) | Yes. MySQL has no native vector extension |
43
- | **PostgreSQL / Aurora PostgreSQL** | `:pgvector` (in-database), `:qdrant`; `:in_memory` (local dev only) | No, `:pgvector` runs inside the same database |
45
+ | **MySQL / Percona / MariaDB / Aurora MySQL** | `:pgvector`, `:qdrant`, `:in_memory` | Separate PostgreSQL database |
46
+ | **PostgreSQL / Aurora PostgreSQL** | `:pgvector`, `:qdrant`, `:in_memory` | Application connection or a separate PostgreSQL database |
44
47
 
45
- **Why MySQL needs an external backend.** MySQL ships no equivalent of the `pgvector` extension. Approximate-nearest-neighbour search over arbitrary float vectors is not part of InnoDB / MyISAM and cannot be added via plugin. Woods does not emulate vector search in MySQL, the gem only ships adapters that delegate to a real vector engine. The shipped pairing for MySQL apps is `:qdrant` for vectors with Woods' own `:sqlite` metadata store; Woods never stores metadata in your application database.
48
+ Woods has no MySQL vector adapter. A MySQL application can use the `:local`
49
+ preset, Qdrant, or a separate PostgreSQL connection for pgvector. Woods' metadata
50
+ store remains a separate choice: SQLite or in-memory. The pgvector row's JSON
51
+ metadata does not replace that store.
46
52
 
47
53
  ### pgvector (PostgreSQL extension)
48
54
 
@@ -52,7 +58,7 @@ The vector store you can use depends on the primary database your Rails app uses
52
58
 
53
59
  **Strengths:**
54
60
  - Zero additional infrastructure if you're on PostgreSQL
55
- - Transactional consistency with metadata (same database)
61
+ - Stores each vector and its per-vector JSON metadata in one PostgreSQL row
56
62
  - Familiar SQL interface, works with ActiveRecord
57
63
  - Supports HNSW indexing
58
64
  - Backed by strong open-source community
@@ -96,6 +102,11 @@ CREATE INDEX IF NOT EXISTS idx_woods_vectors_embedding_hnsw
96
102
  ON woods_vectors USING hnsw (embedding vector_cosine_ops);
97
103
  ```
98
104
 
105
+ **Dimension limit:** Woods uses `vector_cosine_ops` HNSW, limited to 2,000 dimensions.
106
+ The default 3,072-dimensional `text-embedding-3-large` output needs an explicit
107
+ smaller provider output width or another backend. See the
108
+ [pgvector configuration contract](CONFIGURATION_REFERENCE.md#pgvector-postgresql).
109
+
99
110
  **Performance notes:**
100
111
  - HNSW: ~5ms search at 10K vectors, ~20ms at 100K. Memory: ~1.5x vector size.
101
112
  - For codebase indexing (~1000-5000 units, potentially 5000-20000 chunks), HNSW is appropriate.
@@ -103,7 +114,7 @@ CREATE INDEX IF NOT EXISTS idx_woods_vectors_embedding_hnsw
103
114
 
104
115
  **When to use:** PostgreSQL is your primary database, you value simplicity, and scale is under ~50K vectors.
105
116
 
106
- **When to avoid:** MySQL is your primary database (can't use pgvector), you need sub-millisecond search, or you're indexing multiple large codebases.
117
+ **When to avoid:** You do not want to operate PostgreSQL, you need sub-millisecond search, or you're indexing multiple large codebases.
107
118
 
108
119
  ---
109
120
 
data/docs/CLIENT_HOOKS.md CHANGED
@@ -2,7 +2,7 @@
2
2
 
3
3
  Edit hooks are optional. MCP reads and `woods:watch` work independently of them.
4
4
  Check the installed gem exposes `woods:hook_refresh` before enabling these
5
- unreleased adapters; updating the plugin does not update the application gem.
5
+ adapters included in Woods `2.0.0`; updating the plugin does not update the application gem.
6
6
  Start the client from the Rails application root, with an existing index.
7
7
 
8
8
  ## Supported client contracts