woods 2.0.0.beta3 → 2.0.0.beta4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (73) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +77 -0
  3. data/CONTRIBUTING.md +29 -17
  4. data/README.md +92 -177
  5. data/docs/AGENT_GUIDE.md +26 -4
  6. data/docs/AGENT_SETUP.md +18 -8
  7. data/docs/BACKEND_MATRIX.md +5 -0
  8. data/docs/CONFIGURATION_REFERENCE.md +74 -8
  9. data/docs/CONSOLE_MCP_SETUP.md +45 -2
  10. data/docs/DOCKER_SETUP.md +1 -1
  11. data/docs/EXTRACTOR_REFERENCE.md +9 -1
  12. data/docs/INCREMENTAL_EXTRACTION.md +30 -6
  13. data/docs/MCP_SERVERS.md +57 -2
  14. data/docs/MCP_TOOL_COOKBOOK.md +4 -4
  15. data/docs/MCP_WORKTREE_SETUP.md +43 -83
  16. data/docs/PUBLISHED_INDEX.md +17 -0
  17. data/docs/RETRIEVAL_GUIDE.md +24 -5
  18. data/docs/TROUBLESHOOTING.md +12 -13
  19. data/docs/UPGRADING_TO_2.md +6 -2
  20. data/docs/WATCH_DAEMON.md +18 -8
  21. data/exe/woods-mcp-start +14 -9
  22. data/lib/generators/woods/pgvector_generator.rb +8 -2
  23. data/lib/woods/agent_configuration/applier.rb +5 -3
  24. data/lib/woods/agent_configuration/cli.rb +2 -2
  25. data/lib/woods/agent_configuration/layout.rb +13 -0
  26. data/lib/woods/console/credential_scanner.rb +4 -3
  27. data/lib/woods/console/dispatch_pipeline.rb +7 -0
  28. data/lib/woods/console/embedded_executor.rb +31 -9
  29. data/lib/woods/console/sql_noise_stripper.rb +9 -7
  30. data/lib/woods/console/sql_table_scanner.rb +47 -7
  31. data/lib/woods/console/sql_validator.rb +49 -9
  32. data/lib/woods/console/sqlite_read_guard.rb +46 -0
  33. data/lib/woods/coordination/pipeline_lock.rb +3 -2
  34. data/lib/woods/embedding/indexer.rb +24 -14
  35. data/lib/woods/extractor.rb +45 -12
  36. data/lib/woods/extractors/declared_parent.rb +55 -0
  37. data/lib/woods/extractors/graphql_extractor.rb +2 -11
  38. data/lib/woods/extractors/lib_extractor.rb +10 -8
  39. data/lib/woods/extractors/mailer_extractor.rb +6 -10
  40. data/lib/woods/extractors/model_extractor.rb +1 -15
  41. data/lib/woods/extractors/poro_extractor.rb +10 -8
  42. data/lib/woods/extractors/shared_utility_methods.rb +22 -5
  43. data/lib/woods/mcp/bearer_auth.rb +2 -1
  44. data/lib/woods/mcp/bootstrapper.rb +17 -4
  45. data/lib/woods/mcp/config_resolver.rb +2 -1
  46. data/lib/woods/mcp/index_reader.rb +11 -2
  47. data/lib/woods/mcp/renderers/markdown_renderer.rb +14 -8
  48. data/lib/woods/mcp/renderers/plain_renderer.rb +11 -7
  49. data/lib/woods/mcp/server.rb +22 -28
  50. data/lib/woods/mcp/tool_contract.rb +1 -1
  51. data/lib/woods/mcp/tool_response_renderer.rb +16 -0
  52. data/lib/woods/mcp/traversal_evidence_text.rb +1 -1
  53. data/lib/woods/mcp/traversal_response.rb +22 -0
  54. data/lib/woods/path_dispatcher.rb +6 -5
  55. data/lib/woods/published_index/typed_unit_reader.rb +40 -3
  56. data/lib/woods/published_index.rb +2 -2
  57. data/lib/woods/rake_helpers.rb +2 -12
  58. data/lib/woods/retrieval/lexical_assembler.rb +14 -3
  59. data/lib/woods/retrieval/lexical_index.rb +2 -1
  60. data/lib/woods/session_tracer/file_store.rb +6 -1
  61. data/lib/woods/source_inputs/consumer_errors.rb +4 -0
  62. data/lib/woods/storage/pgvector.rb +6 -2
  63. data/lib/woods/temporal/json_snapshot_store.rb +35 -7
  64. data/lib/woods/version.rb +1 -1
  65. data/lib/woods/watch/daemon.rb +18 -4
  66. data/plugin/.claude-plugin/plugin.json +1 -1
  67. data/plugin/hooks/woods-input-rules.sh +4 -4
  68. data/plugin/skills/woods-agent-enable/SKILL.md +7 -1
  69. data/plugin/skills/woods-diagnose/SKILL.md +64 -33
  70. data/plugin/skills/woods-investigate/SKILL.md +51 -12
  71. data/plugin/skills/woods-mcp-config/SKILL.md +11 -11
  72. data/plugin/skills/woods-setup/SKILL.md +14 -11
  73. metadata +8 -5
@@ -1,127 +1,87 @@
1
1
  # MCP Registration in Git Worktrees
2
2
 
3
- > **Claude Code specific.** This page covers Claude Code's MCP registration model (`/mcp`, `~/.claude/plugins/`); other MCP clients manage per-directory registration their own way.
3
+ > **Claude Code specific.** Other MCP clients manage registration and subagent tool access differently. Use the client's current documentation alongside the Woods [MCP setup guide](MCP_SERVERS.md).
4
4
 
5
- When you work in a git worktree, a separate directory checked out from the same repository, your MCP tools may not be available to subagents running in that directory. This page explains why, how to fix it, and how to confirm the registration took effect.
5
+ A git worktree is a separate checkout. For Woods, verify both which servers a session can access and which checkout's index those servers serve.
6
6
 
7
- ## Why Worktree Subagents May Not See Woods Tools
7
+ ## Separate sessions and inherited subagents
8
8
 
9
- MCP server registration in Claude Code is controlled by `.mcp.json` files. Claude Code discovers these files by walking up the directory tree from the working directory. It stops at the first `.mcp.json` it finds (or at `~/.claude/settings.json` for global registrations).
9
+ A **new Claude Code session launched in a worktree** can have different MCP registrations from a session in the main checkout:
10
10
 
11
- A git worktree has its own root directory separate from the main repository checkout. When a subagent starts inside the worktree root, it walks up from that path, not from the main repository root. Unless a `.mcp.json` exists inside the worktree directory tree or in an ancestor shared with both checkouts, the subagent sees no MCP servers.
11
+ | Registration scope | Location | Availability |
12
+ |---|---|---|
13
+ | Local | Project entry in `~/.claude.json` | The project where the server was registered |
14
+ | Project | `.mcp.json` in the project root | That project, subject to client approval/settings |
15
+ | User | `mcpServers` in `~/.claude.json` | Across the user's projects |
12
16
 
13
- Example directory layout:
17
+ A tracked `.mcp.json` may already be present in another worktree; a local, untracked file will not be copied by Git. User-level registration can make a server available across checkouts, but its configured paths still determine which index it serves.
14
18
 
15
- ```
16
- ~/work/my-app/ ← main checkout, has .mcp.json here
17
- ~/work/my-app-feature/ ← worktree, no .mcp.json, so MCP tools are missing
18
- ```
19
+ An **inherited subagent** receives the parent conversation's MCP tools, subject to tool restrictions. Changing the subagent's working directory does not by itself launch another Woods server or switch the served index. Distinguish this from starting an independent client process in that directory.
20
+
21
+ See Claude Code's [MCP installation scopes](https://code.claude.com/docs/en/mcp#mcp-installation-scopes) and [subagent tool access](https://code.claude.com/docs/en/sub-agents#available-tools). Do not diagnose registration by assuming the client searches ancestor directories for the nearest `.mcp.json`.
19
22
 
20
- A subagent spawned in `~/work/my-app-feature/` will not find the `.mcp.json` from `~/work/my-app/`.
23
+ ## Configure a server for the intended checkout
21
24
 
22
- ## Fix: Add a `.mcp.json` to the Worktree Root
25
+ First check `/mcp` in the session that needs Woods. If a suitable server is already connected, verify its index before adding a duplicate registration.
23
26
 
24
- Create a `.mcp.json` in the worktree's root directory with the same woods server entries you use in the main checkout:
27
+ For a host with the application's Ruby bundle installed, a project-root `.mcp.json` can select the bundle and index explicitly:
25
28
 
26
29
  ```json
27
30
  {
28
31
  "mcpServers": {
29
32
  "woods": {
30
- "command": "woods-mcp-start",
31
- "args": ["./tmp/woods"]
32
- },
33
- "woods-console": {
34
- "command": "docker",
35
- "args": [
36
- "compose", "exec", "-T", "app",
37
- "bundle", "exec", "rake", "woods:console"
38
- ],
39
- "cwd": "/absolute/host/path/to/worktree"
33
+ "command": "bundle",
34
+ "args": ["exec", "woods-mcp-start", "/absolute/path/to/worktree/tmp/woods"],
35
+ "env": {
36
+ "BUNDLE_GEMFILE": "/absolute/path/to/worktree/Gemfile"
37
+ }
40
38
  }
41
39
  }
42
40
  }
43
41
  ```
44
42
 
45
- Adjust the paths and arguments to match your project's setup. In particular:
46
-
47
- - `./tmp/woods` is a relative path, it resolves against the worktree root, which is correct if extraction output is written into each worktree separately.
48
- - If you share a single extraction output directory between checkouts, use the absolute path to the shared output: `"/absolute/path/to/main-checkout/tmp/woods"`.
49
- - For Docker projects, the `docker compose exec` command works the same from any host path.
50
-
51
- ## How Plugin Discovery Works
52
-
53
- Claude Code also discovers MCP servers registered in plugin manifests. A plugin at `~/.claude/plugins/<plugin-name>/admin-tools/.mcp.json` is loaded globally, its servers are available in any session regardless of working directory.
43
+ Replace both paths with the intended Rails worktree. Preserve other server entries. These absolute paths are machine-specific; do not commit them as shared team defaults. Follow the client's approval/reconnection steps, then verify the connection below.
54
44
 
55
- If your team distributes the woods MCP registration through a shared plugin, the worktree problem does not apply. Check whether woods is already registered this way:
45
+ For Docker, use the [container-first setup](DOCKER_SETUP.md#default-run-it-through-the-application-container). Select the intended Compose project and service, and verify that its mounted application directory is the worktree you mean to index. `docker compose exec` targets an existing container; running the command from a different checkout does not change that container's mounts. The index path must be visible inside that container.
56
46
 
57
- ```bash
58
- ls ~/.claude/plugins/
59
- # look for a directory containing admin-tools/.mcp.json
60
- cat ~/.claude/plugins/<plugin-name>/admin-tools/.mcp.json
61
- ```
62
-
63
- If you find the woods servers registered there, subagents in any worktree will have access automatically, you do not need a per-worktree `.mcp.json`.
64
-
65
- ## Verifying MCP Registration for a Subagent
47
+ Keep a separate output directory for each checkout when you need checkout-specific answers. If you intentionally point at another checkout's index, Woods serves that published content; registration alone does not make it describe the current worktree.
66
48
 
67
- ### Option 1: List tools from a claude session in the worktree
49
+ ## Plugin-provided servers
68
50
 
69
- Open a new Claude Code session with the worktree as the working directory and run:
51
+ An enabled plugin can supply MCP server definitions, but installing a plugin does not establish that a Woods server is connected to the intended checkout. Availability depends on the plugin's configuration, enabled scope, and the client's tool restrictions. Check `/mcp` rather than assuming a fixed path under `~/.claude/plugins/` or global availability.
70
52
 
71
- ```
72
- /mcp
73
- ```
53
+ The Woods setup/configuration skills help configure a server; their presence alone is not verification that the Index Server is running. See [agent setup](AGENT_SETUP.md).
74
54
 
75
- This lists all connected MCP servers and their tools. If `woods` and/or `woods-console` appear, registration is working.
55
+ ## Verify registration and the served index
76
56
 
77
- ### Option 2: Check via the woods status tool
57
+ 1. Open `/mcp` in the relevant Claude Code session and inspect the Woods connection and available tools.
58
+ 2. Call the Index Server's `woods_status`. Check `index_dir`, generation, and available freshness/provenance information against the intended extraction. A successful connection to the wrong index is still the wrong setup.
59
+ 3. Use `search` and typed `lookup` for a known unit from that checkout. When checking a subagent, verify that it can call the inherited tools and is using the same intended index.
78
60
 
79
- Ask Claude to call the status tool:
80
-
81
- ```
82
- Use woods-console console_status to check what models are available.
83
- ```
84
-
85
- If the tool runs successfully, MCP is registered and the console server is reachable.
86
-
87
- ### Option 3: Inspect the MCP config that Claude Code loaded
88
-
89
- From the worktree directory, run:
90
-
91
- ```bash
92
- cat .mcp.json
93
- ```
94
-
95
- If the file exists and contains the woods entries, Claude Code will use it. If the file is missing, check parent directories up to your home directory for any `.mcp.json` that would be discovered.
61
+ The optional Console Server is a separate connection to a booted Rails application. If deliberately enabled, check it with `console_status` and verify its application/container separately. Console access is not required to verify the Index Server.
96
62
 
97
63
  ## Extraction Provenance in Worktrees (`git_branch` / `git_sha`)
98
64
 
99
- `manifest.json` records the `git_branch` and `git_sha` the extraction ran against. In a linked worktree, `.git` is a **file** containing a `gitdir:` pointer to the real git directory, frequently an absolute host path, rather than a `.git` directory.
65
+ The published payload's `manifest.json` records the extraction's `git_branch` and `git_sha`. In a linked worktree, `.git` is a file containing a `gitdir:` pointer, often to an absolute host path. That private worktree directory also refers to the parent repository's shared Git data.
100
66
 
101
- Woods resolves provenance with worktree-aware git plumbing (`git -C <root> rev-parse`), so an ordinary worktree reports the correct branch and SHA. When a `.git` is present but the pointed-to git directory **cannot be resolved**: most commonly a worktree extracted inside a container where the host path isn't mounted. Woods records `git_branch: "unknown"` / `git_sha: "unknown"` rather than a stale, misleading value: a baked `GIT_BRANCH`/`GIT_SHA` build arg is **not** trusted here (it could be stale). The env vars are honored only when there is no `.git` at the root at all (a non-repo checkout, e.g. a Docker `COPY` that excludes `.git`) or git is unavailable.
67
+ Woods uses worktree-aware Git commands. If a present `.git` cannot be resolved, provenance is `"unknown"`; stale `GIT_BRANCH`/`GIT_SHA` values are not substituted. Those environment variables are fallbacks only when the root has no `.git` or Git is unavailable. Temporal snapshots skip an unknown SHA.
102
68
 
103
- To get correct provenance from a containerized worktree, mount the directory the `gitdir:` pointer references (the parent repository's `.git`) into the container so git can resolve it. Temporal snapshots skip an `"unknown"` SHA, so misleading provenance never keys a snapshot.
69
+ For extraction in a container, make the canonical Git directory and the worktree's pointer resolvable there. Mounting only the private worktree Git directory can leave its shared object store unreachable. Follow the [Git provenance troubleshooting guide](TROUBLESHOOTING.md) for mount and `WOODS_GIT_DIR` guidance, and the [published index layout](INDEX_LAYOUT.md) when locating the manifest.
104
70
 
105
71
  ## Troubleshooting
106
72
 
107
- **"Unknown tool" or "no MCP server named woods"**
108
-
109
- The woods MCP server is not registered for this session. Add a `.mcp.json` to the worktree root as shown above, then restart the session.
73
+ **Woods or an expected tool is absent**
110
74
 
111
- **Console server starts but returns "unsupported: Not yet implemented in embedded mode" for console_sql / console_query**
75
+ Check `/mcp` for connection errors, enabled registration scopes, and project approvals. For a subagent, also check its tool restrictions. Compare the requested tool with the [supported tool surface](MCP_SERVERS.md#conditional-index-capabilities); some tools require optional collaborators and are not registered by the packaged server. Do not add duplicate registrations before establishing which case applies.
112
76
 
113
- The console server is registered and reachable, but `embedded_read_tools` is disabled (the default). To enable console_sql and console_query in embedded mode, see the [Console MCP Setup guide](CONSOLE_MCP_SETUP.md), specifically the `embedded_read_tools: true` option for the Rack middleware.
77
+ **The server connects but shows another checkout**
114
78
 
115
- **Extraction output is empty or stale in the worktree**
79
+ Inspect its configured bundle, index path, Compose project/service, and container mounts. Re-extract from the intended Rails checkout into its own output directory, then reconnect to that index and repeat `woods_status`. See [Docker setup](DOCKER_SETUP.md).
116
80
 
117
- Extraction writes to `tmp/woods/` relative to the Rails application root (inside the container). If the worktree's volume mount points to a different host path than the main checkout, run extraction again from the worktree:
118
-
119
- ```bash
120
- docker compose exec app bundle exec rake woods:extract
121
- ```
81
+ **Two registrations use the same server name**
122
82
 
123
- See [DOCKER_SETUP.md](DOCKER_SETUP.md) for the full Docker workflow.
83
+ Check the client's [scope precedence](https://code.claude.com/docs/en/mcp#scope-hierarchy-and-precedence) and the effective connection shown by `/mcp`. Do not assume the two definitions merge or that a project file overrides every other scope.
124
84
 
125
- **Worktree `.mcp.json` conflicts with main checkout `.mcp.json`**
85
+ **Console SQL/query tools are absent or unsupported**
126
86
 
127
- Each file is independent. Claude Code loads the one closest to the working directory. There is no inheritance or merging between them. Keep both files in sync manually, or move the shared configuration into a global plugin manifest.
87
+ Console read-tool availability is configured separately from registration. Follow [Console MCP setup](CONSOLE_MCP_SETUP.md) for the supported 9-tool default and optional 11-tool mode.
@@ -109,6 +109,23 @@ The reader wraps `Woods::MCP::IndexReader` with `auto_refresh: false`; the unit
109
109
 
110
110
  `Woods::MCP::IndexReader#find_unit` keys its identifier map on identifier alone. If two type directories both list the same identifier (a model and a service both named `Foo`, for example), whichever type sorts last in `Woods::MCP::IndexReader::TYPE_DIRS` silently wins, and `unit(identifier)` returns that one. Pass `type:` to read a specific type's unit file directly and skip the collision entirely; `#table_database_map` always does this internally (`type: 'model'`), so a same-named non-model unit can never shadow a model's `table_name`/`database`.
111
111
 
112
+ ### Actual unit types and directory families
113
+
114
+ Unreleased after `2.0.0.beta3`: `unit` and `units` accept actual published
115
+ `graphql_type`, `graphql_mutation`, `graphql_resolver`, `graphql_query`, and
116
+ `gem_source` types. Enumeration preserves each unit's actual type rather than
117
+ labeling every GraphQL member `graphql` or every gem source `rails_source`.
118
+ Check the loaded Git revision when testing a development checkout.
119
+
120
+ The existing `graphql` and `rails_source` filters remain directory-family
121
+ aliases: `graphql` selects all four GraphQL types; `rails_source` selects both
122
+ Rails and gem sources. Returned records keep their actual type. Use
123
+ `units(type: 'gem_source')` for gem sources only; to select Rails sources only,
124
+ filter `units(type: 'rails_source')` on each entry's `"type" == "rails_source"`.
125
+ Unknown types return no records, and an explicit GraphQL subtype never matches
126
+ another subtype. Enumeration continues to return index-entry fields, not full
127
+ unit source bodies.
128
+
112
129
  ### `available_generations`: published means published
113
130
 
114
131
  A generation is listed only when both hold:
@@ -63,7 +63,7 @@ Keyword results are scored by how many distinct fields matched (identifier, sour
63
63
 
64
64
  ## Configuring Retrieval
65
65
 
66
- Retrieval requires an embedding provider and a vector store. Set these in `config/initializers/woods.rb`.
66
+ Semantic retrieval requires an embedding provider and a vector store. Configure these in `config/initializers/woods.rb` before embedding. For provider-free ranked retrieval over extraction output, use [explicit lexical mode](#embedding-free-lexical-retrieval) in the MCP process environment instead.
67
67
 
68
68
  ### Presets (recommended)
69
69
 
@@ -180,7 +180,12 @@ WOODS_RETRIEVAL_MODE=lexical bundle exec woods-mcp-start ./tmp/woods
180
180
  For a Ruby-built retriever, set `config.retrieval_mode = :lexical` and supply a
181
181
  populated metadata store to `Builder#build_retriever`. The packaged MCP server
182
182
  loads published unit JSON itself; it does not boot Rails or read current source
183
- files. Neither path constructs an embedding provider or vector adapter. The
183
+ files. Neither path constructs an embedding provider or vector adapter. If a
184
+ no-provider error suggests only OpenAI, Ollama or `search`, explicit lexical
185
+ mode is still available from `2.0.0.beta3`: set the variable in the MCP client
186
+ configuration and restart that server. Confirm `woods_status.retriever.mode`
187
+ is `lexical`; setting it only in a Rails initializer does not configure a
188
+ separate MCP process. The
184
189
  existing `:semantic` mode remains the default; provider failures never switch
185
190
  modes automatically. A Rails initializer is not loaded by the standalone MCP
186
191
  process, so set the environment variable in that process's client configuration.
@@ -191,15 +196,25 @@ and validations). Exact full identifiers rank first, with ambiguous typed owners
191
196
  retained. Other ties are deterministic. Responses name the lexical mode and
192
197
  matching fields/terms; runtime-field hits include the selected published runtime
193
198
  values. Lexical Ruby results leave the semantic-only `type_rank_context` table
194
- `nil`; they do not report a global vector rank or vector fallback. The top 20 eligible positive matches are considered for the
195
- output budget; this is ranked discovery, not an exhaustive match listing. Explicit
199
+ `nil`; they do not report a global vector rank or vector fallback. The top 20
200
+ eligible positive matches form the candidate shortlist for the output budget.
201
+ The lexical header reports `sources included`, `candidates considered`, and
202
+ `candidate limit: 20`. Included sources count the actual returned entries;
203
+ considered candidates count the shortlist after filtering and the limit, not
204
+ all matches or all documents examined. This is ranked discovery, not an
205
+ exhaustive match listing. Explicit
196
206
  `types` filters override default exclusions, as in semantic retrieval, and apply
197
207
  before that limit. A query with no lexical evidence returns no matches; unrelated
198
208
  graph hubs are never added. Query-seeded graph ranking is evaluation-only.
199
209
 
200
210
  The budget covers headers, matching explanations and truncation notices using a
201
211
  labelled character-based estimate, not an exact provider tokenizer. Full source
202
- remains available through `lookup`. Very small budgets can omit all sources. The
212
+ remains available through `lookup`. Count text is charged before source selection;
213
+ final counts do not trigger a second selection pass. Very small budgets can omit
214
+ all sources or clip the header itself. Zero candidates means no lexical matches;
215
+ positive candidates with zero included sources means no source entry fit the
216
+ available budget. The same counts apply to full, compact, outline and scoped
217
+ retrieval, agreeing with returned source attribution. The
203
218
  reader pins one published generation for building and querying its immutable
204
219
  lexical snapshot, rebuilding after publication. Corrupt units fail explicitly;
205
220
  they cannot quietly become a successful partial index. Older flat indexes rebuild
@@ -353,6 +368,10 @@ bundle exec rake woods:embed
353
368
  | `text-embedding-3-small` (default) | 1536 |
354
369
  | `text-embedding-3-large` | 3072 |
355
370
 
371
+ Woods' pgvector HNSW adapter supports at most 2,000 dimensions. For the large
372
+ model, request a supported output width explicitly or choose another backend;
373
+ see [pgvector configuration](CONFIGURATION_REFERENCE.md#pgvector-postgresql).
374
+
356
375
  **Ollama default model:** `nomic-embed-text`. Dimensions are detected dynamically on first embed.
357
376
 
358
377
  ---
@@ -8,7 +8,7 @@ This guide covers the most common problems encountered when installing, extracti
8
8
 
9
9
  | Error message | Cause | Fix |
10
10
  |---------------|-------|-----|
11
- | `No manifest.json found` | Wrong index path or no published generation | Use the path visible to the server process; run `woods:validate` |
11
+ | `Could not resolve a published Woods index` (older versions: `No manifest.json found`) | Wrong index path or unresolved published generation | Select the existing index using a path visible to the server process; see [startup diagnostics](#index-cannot-be-resolved-at-startup) |
12
12
  | `uninitialized constant Rails` | Not running inside Rails app | Run via `bundle exec rake` in Rails root |
13
13
  | `type "vector" does not exist` | pgvector not installed | `CREATE EXTENSION vector` in PostgreSQL |
14
14
  | `Connection refused (localhost:11434)` | Ollama not running | `ollama serve` |
@@ -23,6 +23,7 @@ This guide covers the most common problems encountered when installing, extracti
23
23
  | `No such container` | Wrong container name | Check with `docker ps --format '{{.Names}}'` |
24
24
  | `JSON parse errors` (MCP) | Rails boot noise on stdout | Remove `puts` calls from initializers |
25
25
  | Query timeout | Large table, no scope | Add scope conditions to narrow results |
26
+ | `Extraction failed for …; the previous generation remains active` | A consumer handled a source error during incremental extraction or refresh (unreleased after `2.0.0.beta3`) | Fix the logged source error and retry the [complete batch](INCREMENTAL_EXTRACTION.md#handled-source-errors-and-retry); watch keeps it pending |
26
27
  | Empty extraction output | `eager_load!` failure | Check for `NameError` in boot output |
27
28
  | Git metadata missing | Shallow clone in CI | Use `fetch-depth: 0` for complete history |
28
29
  | Parallel tool calls all fail | MCP client batches calls | Send calls sequentially, validate params first |
@@ -390,13 +391,17 @@ relative and resolves outside the mount.
390
391
 
391
392
  ## MCP Server Problems
392
393
 
393
- ### "No manifest.json" error when starting the Index Server
394
+ <a id="no-manifestjson-error-when-starting-the-index-server"></a>
394
395
 
395
- **Symptom:** `woods-mcp-start` exits with an error like `No manifest.json found at /path/to/...` even though extraction completed.
396
+ ### Index cannot be resolved at startup
396
397
 
397
- **Cause:** The Index Server is using the container-internal path rather than the host-side path to the volume-mounted output. The server runs on the host and cannot access container filesystem paths.
398
+ **Symptom:** An Index MCP executable exits with `Could not resolve a published Woods index in: /path/to/...` even though extraction completed. This headline is unreleased after `2.0.0.beta3`; older versions say `No manifest.json found`. Both mean the selected index could not resolve its manifest, not that an atomic index needs a root manifest.
398
399
 
399
- **Fix:** Use the host path in your `.mcp.json`:
400
+ Embedded Index MCP startup through `IndexReader` also raises an `ArgumentError` with the selected directory and layout guidance when the marker cannot resolve a manifest, including malformed marker shapes such as `[]` or a numeric `payload` (unreleased after `2.0.0.beta3`). Earlier builds may expose a raw `TypeError` or `NoMethodError` for those shapes. Inspect the marker and preserve the failing index before attempting recovery.
401
+
402
+ **Cause:** The selected directory is not the published index root, the published generation cannot be resolved, or the path is not visible to the MCP process. A container path is appropriate for a container process; a host process needs the host-visible path.
403
+
404
+ **Fix:** Point at the existing index before extracting again. Check the examined directory in the error and the [MCP path precedence](CONFIGURATION_REFERENCE.md#environment-variables). For a host-side launch whose working directory contains `tmp/woods`, for example:
400
405
 
401
406
  ```json
402
407
  {
@@ -409,13 +414,7 @@ relative and resolves outside the mount.
409
414
  }
410
415
  ```
411
416
 
412
- Verify the output is accessible from the host:
413
-
414
- ```bash
415
- ls ./tmp/woods/manifest.json
416
- ```
417
-
418
- **Since Woods 2.0, a healthy index may not have `manifest.json` at the output root at all.** Extraction publishes each generation into an immutable `payloads/gen-<N>/` directory and points to it from `generation.json`. If the flat path is missing, check the payload path instead before assuming extraction failed:
417
+ **Since Woods 2.0, a healthy index may not have `manifest.json` at the output root at all.** Extraction publishes each generation into an immutable `payloads/gen-<N>/` directory and points to it from `generation.json`. Inspect the marker and its payload in the MCP process's filesystem before assuming extraction failed (use the generation named by your marker):
419
418
 
420
419
  ```bash
421
420
  cat ./tmp/woods/generation.json # {"number": 42, "payload": "payloads/gen-42", ...}
@@ -427,7 +426,7 @@ Update their gate using the [filesystem layout contract](INDEX_LAYOUT.md), which
427
426
  includes Bash/jq and Python readers. An upload must pin and copy one complete
428
427
  payload before publishing its captured pointer; keep a failed copy unpublished.
429
428
 
430
- `woods-mcp-start` and `IndexReader` already resolve this automatically, this is only for manual inspection. If neither path has a manifest, your Docker volume mount is not configured correctly. See [DOCKER_SETUP.md](DOCKER_SETUP.md).
429
+ `woods-mcp-start` and `IndexReader` resolve this automatically; these commands are for manual inspection. Legacy flat indexes use a root `manifest.json`. If neither layout resolves, check the selected path, pointer, payload and any volume mount. See [DOCKER_SETUP.md](DOCKER_SETUP.md) for container launches.
431
430
 
432
431
  ---
433
432
 
@@ -5,8 +5,8 @@ Woods 2.0 changes observable index identifiers, publication layout, vector-store
5
5
  This guide assumes the last v1 release, 1.6.1, and targets 2.0.0.
6
6
 
7
7
  <!-- release-state:upgrade-availability -->
8
- > RubyGems lists 2.0.0.beta3 as a prerelease. Pin it explicitly with
9
- > `gem "woods", "2.0.0.beta3"`; `~> 2.0` resolves only once
8
+ > This tree declares 2.0.0.beta4 as a prerelease. After RubyGems lists it, pin it with
9
+ > `gem "woods", "2.0.0.beta4"`; `~> 2.0` resolves only once
10
10
  > 2.0.0 is published.
11
11
  <!-- release-state:end -->
12
12
 
@@ -142,6 +142,10 @@ bin/rails woods:validate
142
142
  bin/rails woods:stats
143
143
  ```
144
144
 
145
+ Unreleased after `2.0.0.beta3`: `woods:clean` removes index artifacts but keeps
146
+ the output directory and its hidden extraction guard. This stable guard lets
147
+ concurrent writers coordinate safely; its presence does not mean an index remains.
148
+
145
149
  The clean extract is required for corrected identifier shapes. Do not use an incremental run as the first v2 extraction: after `woods:clean` there is no baseline, and v2 `woods:incremental` refuses that state rather than publishing a near-empty index as the application's complete truth.
146
150
 
147
151
  An interrupted extraction leaves readers on the last complete generation because Woods publishes `generation.json` only after the payload is complete. Re-run the task; do not delete a partial directory speculatively. A run that completes its payload but cannot publish the marker now fails loudly instead of reporting success, so treat a non-zero exit as work to redo rather than as a partial success.
data/docs/WATCH_DAEMON.md CHANGED
@@ -96,7 +96,7 @@ partial write:
96
96
 
97
97
  | Failure | What happens |
98
98
  |---|---|
99
- | Reload raises (`SyntaxError`, `NameError`) | Degraded status naming the reason; index intact at generation N; retried on the next event |
99
+ | Reload raises (`SyntaxError`, `NameError`) | Degraded status naming the reason; index intact at generation N; pending paths retried on the next file event or heartbeat |
100
100
  | Extraction raises | Degraded status; generation not advanced |
101
101
  | Payload directory can't be opened, over a payload-born index | Degraded status; generation not advanced. An incremental run only writes the units it touched, so there is no complete flat index it could fall back to publishing, see [Payload publishing](#payload-publishing) |
102
102
  | Index written but the generation bump failed | Degraded status; paths carried forward. The extractor deliberately does not fail an otherwise-good extraction over an unwritable marker, but the marker *is* what readers refresh on, so the daemon cross-checks that the number moved rather than reporting `running` over an index nothing can see |
@@ -119,17 +119,23 @@ at a known generation, reason attached), `stopped` (nothing is maintaining this
119
119
  index). A stale answer is only dangerous when nothing says so.
120
120
 
121
121
  The file is written world-readable (0644) by design: host-side hooks read it
122
- through a bind mount. Every other artifact Woods writes stays at 0600.
122
+ through a bind mount. Writes through `Woods::AtomicFile` default to owner-only
123
+ 0600 unless the caller supplies another mode. This is not a guarantee for every
124
+ Woods artifact: the SQLite metadata store does not enforce 0600, and a newly
125
+ created database uses 0644 under umask 022. Restrict access to the output
126
+ directory according to the source and metadata it contains.
123
127
 
124
128
  Note that `SyntaxError` is a `ScriptError`, not a `StandardError`. Rescuing
125
129
  only the latter would let a half-typed file kill the daemon.
126
130
 
127
131
  A cycle that fails to land its work never loses its paths. Lock contention, a
128
132
  failed reload, and a raising extraction all carry the batch into `@pending`, and
129
- the next cycle folds it back in, the files really did change, and no later
130
- event will mention them again. The retry is not a tight loop: a degraded cycle
131
- ends the drain and waits for the next event, because the cause needs an edit to
132
- clear.
133
+ the next cycle folds it back in even if no new event mentions those files.
134
+ A degraded cycle ends the current drain to avoid a tight retry loop. Pending
135
+ paths are retried on the next file event or [heartbeat](#the-heartbeat), so a
136
+ finished contending writer does not require another edit to trigger recovery.
137
+ Heartbeat retries use a separate worker so status updates and lock refresh
138
+ continue while extraction runs.
133
139
 
134
140
  ### The heartbeat
135
141
 
@@ -163,6 +169,9 @@ is alive, so *alive has to mean covered*.
163
169
  The standalone `woods:watch` task snapshots reload/restart inputs before invoking
164
170
  Rails' `environment` task. Inputs unchanged across that boundary, including
165
171
  carried paths that remain deleted, may be reconciled by a full extraction.
172
+ Unreleased after `2.0.0.beta3`: registered restart inputs deleted while the
173
+ daemon was stopped also trigger a full extraction after a fresh environment
174
+ boot. Nominal framework paths still use the bounded deletion sweep.
166
175
  Changes during environment initialization still require restart. Lock contention,
167
176
  extraction failure, and publication failure retain the full-reconciliation
168
177
  obligation for retry; a successful publish clears it.
@@ -202,8 +211,9 @@ already tolerate the duplicate paths this produces against whatever catch-up
202
211
  finds on its own via the tree scan.
203
212
 
204
213
  Deletions need one extra step, because a deleted file leaves no mtime to scan:
205
- if any path the index attributes a unit to is gone from disk, the daemon runs
206
- one cycle with an *empty* change set, which reaches the ghost units through the
214
+ registered restart inputs follow the full-reconciliation rule above. For other
215
+ registered paths gone from disk, a deletion-only startup runs one cycle with an
216
+ *empty* change set, which reaches the ghost units through the
207
217
  extractor's bounded deletion sweep. Deliberately empty, naming the paths would
208
218
  make the deletions authoritative for every unit type, and some registered paths
209
219
  are nominal (on Rails < 7.1, `ActiveRecord::SchemaMigration` registers a
data/exe/woods-mcp-start CHANGED
@@ -7,7 +7,7 @@
7
7
  # same as when launched directly. RubyGems loads gem executables as Ruby, so
8
8
  # this wrapper must remain a Ruby program when packaged.
9
9
 
10
- index_dir = ARGV[0] || ENV.fetch('WOODS_DIR', nil)
10
+ index_dir = ARGV[0] || ENV.fetch('WOODS_DIR', nil) || ENV.fetch('WOODS_OUTPUT', nil)
11
11
 
12
12
  if index_dir.nil? || index_dir.empty?
13
13
  warn 'Error: No index directory specified.'
@@ -15,9 +15,13 @@ if index_dir.nil? || index_dir.empty?
15
15
  exit 1
16
16
  end
17
17
 
18
+ index_dir = File.expand_path(index_dir)
19
+ path_remedy = 'Point at the existing index with an explicit path, WOODS_DIR, or WOODS_OUTPUT. ' \
20
+ 'If no index exists, run `bundle exec rake woods:extract` in your Rails app.'
21
+
18
22
  unless File.directory?(index_dir)
19
23
  warn "Error: Index directory does not exist: #{index_dir}"
20
- warn 'Run extraction first: bundle exec rake woods:extract'
24
+ warn path_remedy
21
25
  exit 1
22
26
  end
23
27
 
@@ -27,7 +31,7 @@ end
27
31
  # whole library just to check one file is wasted work on every boot. A
28
32
  # payload-born index has no manifest.json at the root — it lives under the
29
33
  # directory generation.json's `payload` pointer names — so the pointer is
30
- # followed here too, with the same escape guard, before concluding the
34
+ # followed here too, with the same realpath containment check, before concluding the
31
35
  # directory holds no index. woods-mcp re-checks this properly through
32
36
  # Bootstrapper regardless; this is just an early, friendlier exit.
33
37
  def manifest_present?(index_dir)
@@ -39,20 +43,21 @@ def manifest_present?(index_dir)
39
43
  require 'json'
40
44
  require_relative '../lib/woods/atomic_file'
41
45
  payload_name = JSON.parse(Woods::AtomicFile.read(generation_path))['payload']
42
- return false if payload_name.nil? || payload_name.empty?
46
+ return false unless payload_name.is_a?(String) && !payload_name.empty?
43
47
 
44
- root = File.expand_path(index_dir)
45
- candidate = File.expand_path(File.join(root, payload_name))
48
+ root = File.realpath(index_dir)
49
+ candidate = File.realpath(payload_name, root)
46
50
  return false unless candidate.start_with?("#{root}#{File::SEPARATOR}")
47
51
 
48
52
  File.file?(File.join(candidate, 'manifest.json'))
49
- rescue JSON::ParserError, SystemCallError
53
+ rescue JSON::ParserError, SystemCallError, TypeError, NoMethodError
50
54
  false
51
55
  end
52
56
 
53
57
  unless manifest_present?(index_dir)
54
- warn "Error: No manifest.json in: #{index_dir}"
55
- warn 'Run extraction first: bundle exec rake woods:extract'
58
+ warn "Error: Could not resolve a published Woods index in: #{index_dir}"
59
+ warn 'Expected generation.json pointing to a payload manifest.json, or a legacy flat manifest.json.'
60
+ warn path_remedy
56
61
  exit 1
57
62
  end
58
63
 
@@ -2,6 +2,7 @@
2
2
 
3
3
  require 'rails/generators'
4
4
  require 'rails/generators/active_record'
5
+ require 'woods/storage/pgvector'
5
6
 
6
7
  module Woods
7
8
  module Generators
@@ -15,7 +16,7 @@ module Woods
15
16
  #
16
17
  # Usage:
17
18
  # rails generate woods:pgvector
18
- # rails generate woods:pgvector --dimensions 3072
19
+ # rails generate woods:pgvector --dimensions 768
19
20
  #
20
21
  class PgvectorGenerator < Rails::Generators::Base
21
22
  include ActiveRecord::Generators::Migration
@@ -25,11 +26,16 @@ module Woods
25
26
  desc 'Creates the woods_vectors table (pgvector column + HNSW index) used by the Woods vector store'
26
27
 
27
28
  class_option :dimensions, type: :numeric, default: 1536,
28
- desc: 'Vector dimensions (1536 for text-embedding-3-small, 3072 for large)'
29
+ desc: 'Vector dimensions (1-2000; default 1536 for text-embedding-3-small)'
29
30
 
30
31
  # @return [void]
31
32
  def create_migration_file
32
33
  @dimensions = options[:dimensions]
34
+ maximum = Woods::Storage::VectorStore::Pgvector::MAX_HNSW_DIMENSIONS
35
+ unless @dimensions.is_a?(Integer) && @dimensions.between?(1, maximum)
36
+ raise ArgumentError, "dimensions must be a positive Integer no greater than #{maximum} for pgvector HNSW"
37
+ end
38
+
33
39
  migration_template(
34
40
  'add_pgvector_to_woods.rb.erb',
35
41
  'db/migrate/add_pgvector_to_woods.rb'
@@ -50,8 +50,10 @@ module Woods
50
50
  "#{@layout.receipt_path}.pending"
51
51
  end
52
52
 
53
- def with_lock
54
- path = "#{@layout.receipt_path}.lock"
53
+ def with_lock(paths = @layout.lock_paths, &operation)
54
+ return operation.call if paths.empty?
55
+
56
+ path, *remaining = paths
55
57
  Document.validate_path!(path)
56
58
  FileUtils.mkdir_p(File.dirname(path), mode: 0o700)
57
59
  File.open(path, File::RDWR | File::CREAT | File::NOFOLLOW | File::NONBLOCK, 0o600) do |lock|
@@ -61,7 +63,7 @@ module Woods
61
63
  'Another Woods configuration operation is active; retry after it finishes'
62
64
  end
63
65
 
64
- yield
66
+ with_lock(remaining, &operation)
65
67
  ensure
66
68
  lock&.flock(File::LOCK_UN)
67
69
  end
@@ -78,12 +78,12 @@ module Woods
78
78
  write_plan(path, plan, layout)
79
79
  PlanDiff.show(plan, @stderr) if options[:diff]
80
80
  plan.summary.merge('plan_file' => File.expand_path(path),
81
- 'runtime_files' => ["#{layout.receipt_path}.lock", "#{layout.receipt_path}.pending"])
81
+ 'runtime_files' => layout.runtime_paths)
82
82
  end
83
83
 
84
84
  def write_plan(path, plan, layout)
85
85
  target = File.expand_path(path)
86
- protected_paths = layout.allowed_paths + ["#{layout.receipt_path}.lock", "#{layout.receipt_path}.pending"]
86
+ protected_paths = layout.allowed_paths + layout.runtime_paths
87
87
  raise Conflict, 'Plan output must differ from every managed/runtime target' if protected_paths.include?(target)
88
88
 
89
89
  Document.validate_path!(target)
@@ -50,6 +50,19 @@ module Woods
50
50
  end
51
51
  end
52
52
 
53
+ # Lock actual managed targets, since user-scoped paths can be shared
54
+ # by different applications. Keep the receipt lock name for existing
55
+ # same-application callers; receipts and recovery journals stay separate.
56
+ def lock_paths
57
+ allowed_paths.map do |path|
58
+ path == receipt_path ? "#{path}.lock" : "#{path}.woods.lock"
59
+ end.sort
60
+ end
61
+
62
+ def runtime_paths
63
+ lock_paths + ["#{receipt_path}.pending"]
64
+ end
65
+
53
66
  def identity
54
67
  { 'client' => 'claude', 'scope' => scope, 'root' => root, 'config_dir' => config_dir,
55
68
  'config_path' => config_path, 'receipt_path' => receipt_path }
@@ -145,9 +145,9 @@ module Woods
145
145
 
146
146
  # Scan a value (String, Hash, Array, or any other object) for credentials.
147
147
  #
148
- # Strings are gsub'd against every active pattern. Hash values and Array
149
- # elements are walked recursively; keys and non-string scalars
150
- # (Integer, Float, true/false, nil) pass through untouched.
148
+ # Strings and Symbols are scanned against every active pattern. Hash
149
+ # keys/values and Array elements are walked recursively; numeric values,
150
+ # booleans, and nil pass through untouched. Symbols retain their type.
151
151
  #
152
152
  # @param value [Object]
153
153
  # @return [Array(Object, Hash{Symbol=>Integer})] two-tuple of the scanned
@@ -164,6 +164,7 @@ module Woods
164
164
  def walk(value, counts, index)
165
165
  case value
166
166
  when String then scan_string(value, counts, index)
167
+ when Symbol then scan_string(value.to_s, counts, index).to_sym
167
168
  when Hash then walk_hash(value, counts, index)
168
169
  when Array then value.map { |item| walk(item, counts, index) }
169
170
  else value
@@ -103,7 +103,14 @@ module Woods
103
103
  response = @conn_mgr.send_request(request)
104
104
  return error_from_response(response, request) unless response['ok']
105
105
 
106
+ # Materialize the wire representation before policy checks. Otherwise
107
+ # Symbols and custom serializers can introduce unscanned strings later.
108
+ # Redact first so protected serializers never run; redact again after
109
+ # normalization to cover fields introduced by a custom serializer.
110
+ # Both renderers consume this same JSON-compatible data tree.
106
111
  result = @ctx.redact(response['result'])
112
+ result = JSON.parse(JSON.generate(result))
113
+ result = @ctx.redact(result)
107
114
  result = scan_for_credentials(result, request)
108
115
  text = @renderer ? @renderer.render_default(result) : JSON.pretty_generate(result)
109
116
  success_response(text)