woods 2.0.1 → 2.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (154) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +94 -7
  3. data/CONTRIBUTING.md +134 -19
  4. data/README.md +1 -1
  5. data/docs/AGENT_GUIDE.md +19 -0
  6. data/docs/AGENT_SETUP.md +22 -2
  7. data/docs/BACKEND_MATRIX.md +7 -0
  8. data/docs/CLIENT_HOOKS.md +6 -0
  9. data/docs/CONFIGURATION_REFERENCE.md +133 -25
  10. data/docs/CONSOLE_MCP_SETUP.md +82 -30
  11. data/docs/EMBEDDING_MODELS.md +16 -19
  12. data/docs/EXTRACTOR_REFERENCE.md +219 -21
  13. data/docs/FAQ.md +11 -25
  14. data/docs/GETTING_STARTED.md +7 -1
  15. data/docs/INCREMENTAL_EXTRACTION.md +261 -19
  16. data/docs/INDEX_LAYOUT.md +5 -0
  17. data/docs/INTERNALS.md +9 -0
  18. data/docs/MCP_HTTP_TRANSPORT.md +20 -15
  19. data/docs/MCP_SERVERS.md +87 -8
  20. data/docs/MCP_TOOL_COOKBOOK.md +13 -55
  21. data/docs/NOTION_INTEGRATION.md +7 -1
  22. data/docs/PUBLISHED_INDEX.md +6 -0
  23. data/docs/README.md +6 -1
  24. data/docs/RETRIEVAL_GUIDE.md +17 -0
  25. data/docs/SOURCE_FRESHNESS.md +157 -5
  26. data/docs/TOKEN_BENCHMARK.md +10 -18
  27. data/docs/TROUBLESHOOTING.md +70 -14
  28. data/docs/UNBLOCKED_INTEGRATION.md +60 -8
  29. data/docs/UPGRADING_TO_2.md +153 -38
  30. data/docs/WATCH_DAEMON.md +97 -14
  31. data/exe/woods-console-mcp +2 -2
  32. data/lib/generators/woods/templates/woods.rb.tt +2 -1
  33. data/lib/tasks/woods.rake +23 -7
  34. data/lib/tasks/woods_checks.rake +2 -2
  35. data/lib/woods/agent_configuration/cli.rb +1 -1
  36. data/lib/woods/agent_configuration/layout.rb +16 -2
  37. data/lib/woods/agent_configuration/plan.rb +13 -3
  38. data/lib/woods/agent_configuration/planner_validation.rb +4 -2
  39. data/lib/woods/agent_configuration/preflight.rb +5 -3
  40. data/lib/woods/builder.rb +17 -57
  41. data/lib/woods/cache/cache_middleware.rb +56 -30
  42. data/lib/woods/chunking/contributor_chunks.rb +119 -0
  43. data/lib/woods/chunking/semantic_chunker.rb +44 -21
  44. data/lib/woods/console/connection_manager.rb +56 -3
  45. data/lib/woods/console/embedded_executor.rb +30 -5
  46. data/lib/woods/console/rack_middleware.rb +29 -1
  47. data/lib/woods/dependency_graph.rb +34 -10
  48. data/lib/woods/embedding/fake.rb +12 -0
  49. data/lib/woods/embedding/indexer.rb +195 -98
  50. data/lib/woods/embedding/input_budget.rb +67 -0
  51. data/lib/woods/embedding/openai.rb +70 -20
  52. data/lib/woods/embedding/provider.rb +37 -25
  53. data/lib/woods/embedding/text_preparer.rb +76 -32
  54. data/lib/woods/embedding/token_counter.rb +18 -81
  55. data/lib/woods/embedding/vector_configuration.rb +48 -0
  56. data/lib/woods/extraction_identities.rb +175 -0
  57. data/lib/woods/extractor.rb +304 -107
  58. data/lib/woods/extractors/action_cable_extractor.rb +8 -3
  59. data/lib/woods/extractors/assigned_value_discovery.rb +74 -0
  60. data/lib/woods/extractors/class_declarations.rb +121 -0
  61. data/lib/woods/extractors/configuration_extractor.rb +11 -3
  62. data/lib/woods/extractors/declaration_ancestry.rb +92 -0
  63. data/lib/woods/extractors/event_extractor.rb +8 -0
  64. data/lib/woods/extractors/graphql_extractor.rb +134 -77
  65. data/lib/woods/extractors/job_extractor.rb +5 -1
  66. data/lib/woods/extractors/lib_extractor.rb +132 -15
  67. data/lib/woods/extractors/mailer_extractor.rb +3 -5
  68. data/lib/woods/extractors/manager_extractor.rb +7 -21
  69. data/lib/woods/extractors/migration_declaration.rb +87 -0
  70. data/lib/woods/extractors/migration_extractor.rb +5 -39
  71. data/lib/woods/extractors/phlex_extractor.rb +6 -2
  72. data/lib/woods/extractors/policy_extractor.rb +9 -5
  73. data/lib/woods/extractors/poro_extractor.rb +112 -53
  74. data/lib/woods/extractors/pundit_extractor.rb +11 -6
  75. data/lib/woods/extractors/scheduled_job_extractor.rb +45 -4
  76. data/lib/woods/extractors/serializer_extractor.rb +34 -22
  77. data/lib/woods/extractors/shared_utility_methods.rb +18 -1
  78. data/lib/woods/extractors/source_nesting.rb +142 -106
  79. data/lib/woods/extractors/standalone_module_discovery.rb +123 -0
  80. data/lib/woods/extractors/state_machine_extractor.rb +46 -40
  81. data/lib/woods/extractors/view_component_extractor.rb +9 -7
  82. data/lib/woods/flow_assembler.rb +4 -1
  83. data/lib/woods/generation.rb +25 -0
  84. data/lib/woods/hooks/context_hint.rb +7 -2
  85. data/lib/woods/mcp/bootstrapper.rb +33 -7
  86. data/lib/woods/mcp/config_resolver.rb +26 -7
  87. data/lib/woods/mcp/index_reader.rb +125 -24
  88. data/lib/woods/mcp/index_reader_pinning.rb +16 -0
  89. data/lib/woods/mcp/renderers/markdown_renderer.rb +7 -1
  90. data/lib/woods/mcp/renderers/plain_renderer.rb +3 -1
  91. data/lib/woods/mcp/search_results.rb +7 -1
  92. data/lib/woods/mcp/server.rb +24 -4
  93. data/lib/woods/module_reconciliation.rb +151 -0
  94. data/lib/woods/path_dispatcher.rb +7 -2
  95. data/lib/woods/rake_helpers.rb +43 -11
  96. data/lib/woods/release.rb +1 -1
  97. data/lib/woods/resilience/index_validator.rb +8 -3
  98. data/lib/woods/resilience/retryable_provider.rb +18 -1
  99. data/lib/woods/resolved_config.rb +68 -8
  100. data/lib/woods/retrieval/context_assembler.rb +3 -3
  101. data/lib/woods/retrieval/lexical_assembler.rb +3 -2
  102. data/lib/woods/retrieval/scope.rb +18 -2
  103. data/lib/woods/retrieval/source_evidence.rb +14 -2
  104. data/lib/woods/source_contributor_validation.rb +78 -0
  105. data/lib/woods/source_contributors.rb +116 -0
  106. data/lib/woods/source_inputs/handoff.rb +37 -0
  107. data/lib/woods/source_inputs/launcher.rb +53 -13
  108. data/lib/woods/source_inputs/manifest.rb +84 -3
  109. data/lib/woods/source_inputs/private_key.rb +44 -12
  110. data/lib/woods/source_inputs/scanner.rb +98 -27
  111. data/lib/woods/source_inputs/scopes.rb +1 -1
  112. data/lib/woods/source_inputs/session.rb +147 -15
  113. data/lib/woods/source_inputs/stable_reader.rb +127 -0
  114. data/lib/woods/source_inputs/status.rb +40 -8
  115. data/lib/woods/source_inputs/verifier.rb +28 -5
  116. data/lib/woods/source_path_encoding.rb +33 -0
  117. data/lib/woods/source_references/cache.rb +284 -0
  118. data/lib/woods/source_references/collector.rb +120 -0
  119. data/lib/woods/source_references/extraction.rb +185 -0
  120. data/lib/woods/source_references/inputs.rb +134 -0
  121. data/lib/woods/source_references/parser_adapter.rb +134 -0
  122. data/lib/woods/source_references/pass.rb +152 -0
  123. data/lib/woods/source_references/prism_adapter.rb +116 -0
  124. data/lib/woods/source_references/registry.rb +178 -0
  125. data/lib/woods/source_references/runtime_lookup.rb +127 -0
  126. data/lib/woods/source_references/value_class.rb +82 -0
  127. data/lib/woods/storage/metadata_store.rb +4 -1
  128. data/lib/woods/storage/qdrant.rb +2 -2
  129. data/lib/woods/unblocked/client.rb +12 -7
  130. data/lib/woods/unblocked/document_builder.rb +4 -1
  131. data/lib/woods/unblocked/exporter.rb +127 -37
  132. data/lib/woods/unblocked/sync_manifest.rb +137 -21
  133. data/lib/woods/unblocked/uri_migration.rb +105 -0
  134. data/lib/woods/util/host_guard.rb +3 -2
  135. data/lib/woods/version.rb +1 -1
  136. data/lib/woods/watch/catch_up.rb +138 -0
  137. data/lib/woods/watch/claim_lease.rb +150 -0
  138. data/lib/woods/watch/cli.rb +26 -2
  139. data/lib/woods/watch/daemon.rb +80 -59
  140. data/lib/woods/watch/installation/options.rb +1 -1
  141. data/lib/woods/watch/installation/receipt.rb +6 -1
  142. data/lib/woods/watch/managed_child.rb +1 -1
  143. data/lib/woods/watch/supervisor.rb +1 -1
  144. data/lib/woods/watch/tree_scan.rb +14 -2
  145. data/plugin/.claude-plugin/plugin.json +1 -1
  146. data/plugin/hooks/adapters/normalize.rb +3 -2
  147. data/plugin/hooks/woods-input-rules.sh +4 -0
  148. data/plugin/hooks/woods-refresh.sh +15 -7
  149. data/plugin/hooks/woods-session-start.sh +60 -3
  150. data/plugin/skills/woods-diagnose/SKILL.md +334 -11
  151. data/plugin/skills/woods-investigate/SKILL.md +11 -0
  152. data/plugin/skills/woods-mcp-config/SKILL.md +79 -8
  153. data/plugin/skills/woods-setup/SKILL.md +53 -6
  154. metadata +32 -5
@@ -241,11 +241,64 @@ document id of everything last pushed. On each run the exporter:
241
241
  - **pushes** only new or changed documents,
242
242
  - **deletes** documents whose source unit has disappeared.
243
243
 
244
- Documents are upserted by URI, so the sync is always safe to re-run. If the
245
- manifest is missing (first run, or a CI cache miss) the exporter reconciles
246
- document ids from the remote collection and falls back to a full re-push,
247
- rebuilding the manifest, correct, just more API calls than one run. In steady
248
- state an unchanged codebase costs ~0 calls.
244
+ Documents are upserted by URI. **Included in Woods 2.1:** the manifest keeps
245
+ separate repository/ref scopes; switching branches preserves the other branch's
246
+ documents. Ref segments are URL-encoded, including `#`, `?`, and `%`.
247
+
248
+ Every sync first inventories remote documents through all pages. A short page
249
+ is not proof of completion; malformed pages or repeated cursors stop before
250
+ writes. Individual remote documents without a non-empty string URI are skipped;
251
+ they provide no URI ownership evidence. This adds listing calls even when all
252
+ local documents are unchanged.
253
+ URIs are unique across the remote organization: Woods refuses a URI already
254
+ assigned to another collection instead of moving it.
255
+
256
+ A missing manifest rebuilds receipts for current documents but does **not**
257
+ adopt unrelated remote documents for deletion. Keep the manifest in a durable
258
+ CI cache. An unreadable, corrupt, or wrong-collection manifest requires recovery
259
+ from a known-good copy; setting force flags cannot recover ownership. Historical
260
+ entries adopted remotely with a null content hash are not deletion authority.
261
+
262
+ ### Explicit ref migration
263
+
264
+ **Included in Woods 2.1:** unresolved owned receipts from a legacy manifest
265
+ leave sync incomplete. A URI prefix cannot prove the old ref: a slash can belong
266
+ to either a ref name or a source path. Woods never adopts or purges those
267
+ receipts by prefix. Select the known source ref with
268
+ `UNBLOCKED_MIGRATE_FROM_REF`, or review obsolete remote documents and their
269
+ legacy receipts for manual cleanup when there is no current replacement.
270
+ `UNBLOCKED_FORCE_PURGE` cannot bypass this ownership requirement.
271
+
272
+ Use migration only when retiring the old export scope, rather than publishing
273
+ two branches independently. Keep one writer per local manifest **and** remote
274
+ repository/ref scope; coordinate writers across worktrees and CI jobs.
275
+
276
+ ```bash
277
+ UNBLOCKED_MIGRATE_FROM_REF=old-branch UNBLOCKED_DRY_RUN=1 bin/rails woods:unblocked_sync
278
+ UNBLOCKED_MIGRATE_FROM_REF=old-branch bin/rails woods:unblocked_sync
279
+ ```
280
+
281
+ Preview reads the pinned index and local receipts, makes no remote requests or
282
+ local writes, and lists the proposed URI pairs. Review that list before applying.
283
+ Use the same source and target ref to migrate formerly unescaped special-character
284
+ URIs. Woods uploads replacements, persists successful receipts, then deletes
285
+ only recorded old IDs with unambiguous current replacements. This bounded
286
+ one-for-one cleanup does not require `UNBLOCKED_FORCE_PURGE`.
287
+
288
+ Interrupted migration resumes on the next sync of the target scope, even without
289
+ the environment variable. Missing ownership, changed remote IDs, missing
290
+ replacements, failed writes, and incomplete cleanup remain visible as incomplete;
291
+ Woods retains old copies until it can verify the replacement. It never infers
292
+ ownership from the collection listing. Records without matching replacement
293
+ units require operator review and manual cleanup. A historical bare URI shared
294
+ by multiple current units is ambiguous and is also retained for review; Woods
295
+ does not guess its old owner from the new lexically first identifier. Proven
296
+ one-to-one migrations also upload replacements outside the ordinary top-N cap. This is not an API-level
297
+ rename: remote IDs can change, so annotations attached to the old ID may not move.
298
+
299
+ Do not run an older Woods exporter against the upgraded version-2 manifest.
300
+ For rollback, stop sync and restore its pre-migration manifest together with the
301
+ corresponding remote state; restoring just the file cannot undo remote deletions.
249
302
 
250
303
  Pair with `woods:incremental` to re-extract only changed files; the sync then
251
304
  pushes only the documents whose content actually changed.
@@ -273,10 +326,9 @@ enumeration or complete per-bucket listings plus strict typed lookup.
273
326
  is a small subset and an unguarded purge would wipe the collection.
274
327
 
275
328
  The guard also fires on *intentional* large removals: dropping a unit type
276
- from the sync set, changing `unblocked_repo_url` (every URI changes), or a
277
- big codebase deletion can all legitimately exceed 30%. The refusal warning
329
+ from the sync set or a big codebase deletion can all legitimately exceed 30%. The refusal warning
278
330
  names the counts, if the deletions are expected, re-run once with
279
- `UNBLOCKED_FORCE_PURGE=1`.
331
+ `UNBLOCKED_FORCE_PURGE=1`. It does not bypass migration ownership checks.
280
332
 
281
333
  ## Troubleshooting
282
334
 
@@ -8,35 +8,6 @@ published 1.6.x security patch as the rollback version.
8
8
  <!-- release-state:upgrade-availability -->
9
9
  <!-- release-state:end -->
10
10
 
11
- ## 2.0.1 security maintenance update
12
-
13
- This maintenance line adds Console request/output policy corrections and isolates
14
- retrieval contexts between retriever instances. It does not add the graph or
15
- extraction features under development for 2.1. Confirm the installed package
16
- version and use its matching tag documentation.
17
-
18
- No index or database schema migration is required. Restart Console/MCP processes
19
- after upgrading. Explicit malformed HTTP origin entries now fail at boot with the
20
- offending entry named; fix the entry rather than weakening authentication.
21
- Automatic and manual Console mounts raise `Woods::ConfigurationError` for these
22
- settings, including invalidly encoded entries. With
23
- no allowlist configured, the existing loopback defaults remain unchanged. See
24
- [HTTP origin matching](MCP_HTTP_TRANSPORT.md#browser-origins-dns-rebinding-defense).
25
-
26
- Raw `console_sql` now refuses ambiguous protected results and genuinely unknown
27
- adapter families. PostgreSQL-subclass adapters retain PostgreSQL handling;
28
- structured Console tools remain available with other adapters. Prefer explicit
29
- unaliased scalar projections or structured reads when a query is refused.
30
- [Console setup](CONSOLE_MCP_SETUP.md#maintenance-policy-corrections) describes the
31
- compatibility boundary. Context caches refill after restart; retired entries
32
- follow the configured TTL or backend eviction. See
33
- [retrieval cache options](CONFIGURATION_REFERENCE.md#retrieval-cache-options).
34
- Rolling back restores the affected behavior.
35
-
36
- Polymorphic `belongs_to` association counts remain unsupported in 2.0.1 and
37
- 1.6.4 and return a generic execution error; the 2.1 functional correction is
38
- not backported.
39
-
40
11
  ## Upgrade outcome
41
12
 
42
13
  After this runbook you will have:
@@ -48,6 +19,21 @@ After this runbook you will have:
48
19
  - an MCP client connected to the v2 packaged tool surface;
49
20
  - a documented way back to v1 if verification fails.
50
21
 
22
+ ### HTTP configuration checks
23
+
24
+ Supporting revisions after 2.0.0 validate HTTP origin allowlists at boot.
25
+ Enabled automatic and manual Console mounts use the same
26
+ `Woods::ConfigurationError` diagnostic for malformed origin entries. An invalid
27
+ entry now prevents enabled HTTP startup and the diagnostic names that entry.
28
+ Use `http(s)://host[:port]` entries without paths, credentials, or query strings,
29
+ then restart. With no explicit Index allowlist, the default policy is unchanged.
30
+ An explicit list replaces browser-origin defaults, including loopback. HTTP(S)
31
+ default ports are equivalent to their omitted form. Stdio-only Console users can
32
+ keep `console_mcp_http_enabled = false`.
33
+
34
+ See [HTTP origins](MCP_HTTP_TRANSPORT.md#browser-origins-dns-rebinding-defense)
35
+ and [Console setup](CONSOLE_MCP_SETUP.md) for the configuration contract.
36
+
51
37
  ### Console read compatibility
52
38
 
53
39
  For supporting security-patch revisions, any nonempty column or EAV redaction
@@ -90,8 +76,9 @@ eviction; disabling both can retain them indefinitely. See
90
76
  | `mcp >= 1.2, < 2.0` and protocol negotiation | Old lockfiles or manually pinned protocol versions can fail | Bundle update Woods/MCP; normally leave protocol version unset |
91
77
  | Index MCP surface aligned to executable wiring | Agents may ask for tools that only exist as conditional schemas | Update agent instructions to the 14-tool default |
92
78
  | Console surface tightened to 9 or 11 tools | Agents may ask for Tier 2/3 or eval schemas that do not execute | Use registered default/read tools only |
93
- | Missing-token behavior changed outside production | An enabled Console HTTP endpoint now stays mounted but returns 401 without a valid token; production still refuses to boot without one | Preserve or configure a secret token of at least 32 characters; send it only to the HTTP transport |
79
+ | Console HTTP validation uses finalized application configuration | Flags set in a 1.x initializer could miss early boot validation; v2 refuses a missing token in production and a configured short token in any environment | Configure a secret token of at least 32 characters, or explicitly disable HTTP for stdio-only use |
94
80
  | Durable-store reconciliation and a 30% purge guard | The first v2 embed may refuse a legitimate rename-heavy deletion | Back up, inspect the deletion, then use the one-run override only if correct |
81
+ | Explicit `config.embedding_model` selects the embedding model | An old assignment that embedding ignored can now change the produced vectors; `embedding_options[:model]` takes precedence | Verify the effective model and rebuild embeddings when it changes, even if the dimension stays the same |
95
82
  | Embedding dimension preflight | A previously tolerated model/store mismatch now fails before writing | Rebuild into a store with the configured dimension |
96
83
  | Export reconciliation guards | Obsidian or Unblocked can refuse a rename-heavy stale-document sweep | Back up and use exporter-specific override only after review |
97
84
  | Notion column pages are grouped by physical table | Models sharing a table (STI, a shared `self.table_name`) previously rewrote each other's column pages on every run | Re-sync once after re-extraction; the shared pages settle and the churn stops |
@@ -101,8 +88,25 @@ eviction; disabling both can retain them indefinitely. See
101
88
  | `woods:embed`, `woods:embed_incremental`, and `woods:notion_sync` exit 1 on reported errors | CI jobs that were green while every unit or page failed now fail | Read the printed errors, fix the cause, re-run; completed work is durable |
102
89
  | The Index MCP `reload` tool needs write access to the index directory | A read-only index mount can serve structural reads but cannot reload in place | Grant write access, or restart the MCP process after publishing |
103
90
  | `config.extractors` and `config.add_gem` warn as unimplemented | Old config may imply filtering that never occurred | Remove or comment the settings; do not rely on them |
91
+ | Legacy `config.log_level=` removed | The old no-op assignment raises `NoMethodError` | Remove it; configure the application logger instead |
92
+ | Legacy unsafe-eval opt-in removed | `WOODS_CONSOLE_UNSAFE_EVAL=true`, an enabled `console_unsafe_eval_enabled`, or legacy confirmation/audit options refuse Console server construction | Remove these settings; use supported read tools or the application’s normal console for deliberate code execution |
104
93
  | New watch, refresh, and evaluation tasks | New operational options become available | Optional; no migration action |
105
94
 
95
+ ## Updating an existing 2.0 installation to 2.1
96
+
97
+ After selecting Woods 2.1 in the application bundle, keep the last good index and
98
+ run one full `bin/rails woods:extract`, followed by `bin/rails woods:validate`.
99
+ This establishes source-reference cache format 3 and repairs affected discovery
100
+ identities. Incremental extraction requires that compatible baseline. Rebuild
101
+ embeddings and exports if the identifiers they store changed; restart MCP to load
102
+ the new gem. A reader upgrade alone cannot add relationships to an older index.
103
+
104
+ The expanded references remain conservative: component callers and unresolved
105
+ runtime scopes are not exhaustively covered. Full and incremental results can
106
+ still differ in `dependents` presentation order without changing membership.
107
+ Large affected sets may cost as much as a full extraction; choose full extraction
108
+ for those workloads rather than assuming every incremental run is faster.
109
+
106
110
  ## Before changing the bundle
107
111
 
108
112
  ### Check the loader for wrapper-nested classes
@@ -124,6 +128,39 @@ extraction. Rebuild embeddings and exports if identifiers change. A genuine
124
128
  duplicate under a supported loader still needs distinct constants or one source
125
129
  file. Woods does not provide a classic-mode naming fallback for this case.
126
130
 
131
+ Woods 2.0.0 can also misidentify valid Struct/Data assignments inside namespace
132
+ wrappers on supported loaders ([#559](https://github.com/lost-in-the/woods/issues/559)).
133
+ **Included in Woods 2.1:** supporting writers verify the loaded
134
+ assignment's identity and source ownership for PORO and library units. Check
135
+ the exact writer revision, then follow the [assigned value-class rules](EXTRACTOR_REFERENCE.md#assigned-value-classes).
136
+ Run a full extraction to repair old wrapper identities and rebuild reference-cache
137
+ format 3; incremental extraction refuses older formats. Keep the previous
138
+ generation until that rebuild succeeds. Do not rename valid application constants
139
+ to work around an older writer's inference.
140
+
141
+ **Included in Woods 2.1:** the collision guard also applies to incremental
142
+ extraction and targeted refresh ([#561](https://github.com/lost-in-the/woods/issues/561)).
143
+ Those paths previously could publish a conflicting source that a full extraction
144
+ would reject. Refusal leaves the last published generation readable. If an older
145
+ run already replaced a unit's owner, fixing the guard alone cannot reconstruct
146
+ the lost unit: correct the producer/source issue and run a successful full
147
+ extraction, then validate it. Keep the previous generation until the rebuild
148
+ succeeds; do not delete the index to bypass the collision.
149
+
150
+ ### Refresh GraphQL discovery on a supporting writer
151
+
152
+ **Included in Woods 2.1:** the discovery fixes for
153
+ [#558](https://github.com/lost-in-the/woods/issues/558),
154
+ [#562](https://github.com/lost-in-the/woods/issues/562) and
155
+ [#563](https://github.com/lost-in-the/woods/issues/563) add schema-class units,
156
+ loaded resolver subclasses and types from every application schema. Schemas use
157
+ the existing `graphql_type` category with `metadata.graphql_kind: "schema"`.
158
+ Verify the loaded writer revision, boot/eager-load the intended application,
159
+ then run a full extraction and validate to fill gaps in older indexes. Readers
160
+ can continue serving the previous generation until publication succeeds.
161
+ See [GraphQL extraction](EXTRACTOR_REFERENCE.md#graphqlextractor) for runtime
162
+ ownership, query-root classification, handled failures and source-fallback limits.
163
+
127
164
  ### 1. Record the current installation
128
165
 
129
166
  Run in the same environment that boots Rails:
@@ -139,7 +176,13 @@ Record the current Woods version, output directory, storage preset/providers, em
139
176
 
140
177
  ### 2. Back up durable data
141
178
 
142
- The generated structural index can be recreated, but its location may also hold local vector dumps and exporter manifests. Copy or snapshot the complete configured output directory before cleaning it.
179
+ The generated structural index can be recreated, but its location may also hold
180
+ `woods.sqlite3` snapshot history, JSON `snapshots/`, local vector dumps and exporter
181
+ manifests. Stop Woods writers/readers, including the watcher, before taking an
182
+ offline copy or consistent filesystem snapshot of the complete output directory.
183
+ For SQLite, use a database-aware backup or close its users before copying; a live
184
+ main-file-only copy can omit data still in the WAL. Back up explicitly configured
185
+ stores outside the output directory separately.
143
186
 
144
187
  Back up external vector stores separately:
145
188
 
@@ -189,26 +232,73 @@ Pay particular attention to:
189
232
  - the configured embedding model/dimension;
190
233
  - `console_mcp_enabled`, `console_mcp_http_enabled`, the HTTP `console_mcp_token` secret source, allowed origins, path, and embedded read-tool flags;
191
234
  - snapshot, session, Notion, Obsidian, and Unblocked settings;
192
- - old `config.extractors` or `config.add_gem` calls, which are not implemented selectors.
235
+ - old `config.extractors` or `config.add_gem` calls, which are not implemented selectors;
236
+ - removed `config.log_level=` and unsafe-eval settings;
237
+ - early middleware configuration: a custom `console_mcp_path` and session tracer
238
+ setup belong in `config/application.rb`, before Railtie initializers run. See
239
+ [Console configuration](CONSOLE_MCP_SETUP.md#configuration-options) and
240
+ [session tracer options](CONFIGURATION_REFERENCE.md#session-tracer-options).
193
241
 
194
242
  Review existing Woods migrations and tables before accepting any newly generated migration. Do not create duplicate `woods_units`, `woods_edges`, or `woods_embeddings` tables.
195
243
 
196
244
  ### 3. Clean and re-extract
197
245
 
198
- After the backup is verified:
246
+ After the backup is verified and Woods processes are stopped:
199
247
 
200
248
  ```bash
201
249
  bin/rails woods:clean
250
+ ```
251
+
252
+ **Cleaning deletes durable history inside this directory too:** `woods.sqlite3`,
253
+ JSON `snapshots/`, vector dumps and export manifests. Before running extraction,
254
+ choose how to retain history:
255
+
256
+ - Keep the complete backup as a separate historical/rollback copy and start a new
257
+ v2 history; or
258
+ - Restore only the closed, consistent `woods.sqlite3` backup and/or `snapshots/`
259
+ directory into the cleaned output. Woods migrates the snapshot database on
260
+ its next use. Do not restore the old structural payload, `generation.json`,
261
+ vector dumps or export manifests into this new baseline.
262
+
263
+ Restoring JSON files does not import them into SQLite. If SQLite has become
264
+ available since JSON capture, extraction and packaged MCP prefer it; keep an
265
+ explicit JSON historical reader for the restored files. `WOODS_SNAPSHOTS=true`
266
+ enables snapshot construction without forcing that backend. See
267
+ [snapshot store selection](MCP_SERVERS.md#conditional-index-capabilities).
268
+
269
+ Then run:
270
+
271
+ ```bash
202
272
  bin/rails woods:extract
203
273
  bin/rails woods:validate
204
274
  bin/rails woods:stats
205
275
  ```
206
276
 
277
+ If you carried history forward, verify an old snapshot remains readable before
278
+ restarting writers. A full capture at an existing Git SHA replaces that SHA's
279
+ snapshot, and configured retention can prune old entries. Keep the untouched
280
+ backup through the rollback window.
281
+
207
282
  Included in Woods `2.0.0`: `woods:clean` removes index artifacts but keeps
208
283
  the output directory and its hidden extraction guard. This stable guard lets
209
284
  concurrent writers coordinate safely; its presence does not mean an index remains.
210
285
 
211
- The clean extract is required for corrected identifier shapes. Do not use an incremental run as the first v2 extraction: after `woods:clean` there is no baseline, and v2 `woods:incremental` refuses that state rather than publishing a near-empty index as the application's complete truth.
286
+ The clean extract is required for corrected identifier shapes. Do not use
287
+ incremental extraction, targeted refresh or an embedded pipeline's incremental
288
+ operation as the first v2 run. A flat index being readable, a successful
289
+ `woods:validate`, or a newer manifest `woods_version` does not certify that all
290
+ retained v1 units were migrated. After cleaning there is no structural baseline,
291
+ and incremental extraction refuses that state. **Included in Woods 2.1:**
292
+ incremental extraction and targeted refresh also refuse a flat index with a
293
+ manifest, or a generation whose manifest names a writer major version below 2,
294
+ before creating a new payload. Run a full `woods:extract` to rebuild it. A
295
+ missing writer field in a generation manifest is allowed for early v2 betas;
296
+ readers retain legacy compatibility. This guard cannot detect v1 units already
297
+ retained by an older incremental writer that relabeled its manifest as v2.
298
+ Supporting source-reference writers also refuse incompatible reference caches.
299
+ CI caches must distinguish Woods major versions
300
+ and restore the exact source baseline described in the
301
+ [incremental CI recipe](INCREMENTAL_EXTRACTION.md#github-actions-with-an-exact-baseline).
212
302
 
213
303
  An interrupted extraction leaves readers on the last complete generation because Woods publishes `generation.json` only after the payload is complete. Re-run the task; do not delete a partial directory speculatively. A run that completes its payload but cannot publish the marker now fails loudly instead of reporting success, so treat a non-zero exit as work to redo rather than as a partial success.
214
304
 
@@ -237,7 +327,13 @@ WOODS_ALLOW_PURGE=1 bin/rails woods:embed
237
327
 
238
328
  The override permits deletion; it is not a repair command. Do not set it permanently.
239
329
 
240
- If Woods reports a dimension mismatch, verify the configured embedding model. Rebuild into a store created for the new dimension. Vectors cannot be converted in place.
330
+ An explicitly assigned `config.embedding_model` now selects the model used to
331
+ embed, unless `embedding_options[:model]` overrides it. Earlier embedding could
332
+ ignore that assignment even though query configuration recorded it. Verify the
333
+ effective model before the first embed. A model change requires rebuilding vectors
334
+ even when both models have the same width. A dimension mismatch additionally
335
+ requires a store created for the new dimension; vectors cannot be converted in
336
+ place.
241
337
 
242
338
  An interrupted embed is safe to re-run; durable checkpoints resume or repair the missing unit.
243
339
 
@@ -253,6 +349,14 @@ Re-run every export after extraction and embeddings are verified. Renamed identi
253
349
 
254
350
  Review the target and backup before any force-purge override. Use `WOODS_NOTION_FORCE=1` only when you intentionally want Notion to re-check unchanged content hashes.
255
351
 
352
+ **Included in Woods 2.1:** Unblocked keeps separate repository/ref scopes.
353
+ Legacy receipts that cannot be assigned safely leave sync incomplete, including
354
+ old exports from a non-`main` ref. Preview an explicit
355
+ `UNBLOCKED_MIGRATE_FROM_REF=<old-ref>` migration; review obsolete documents with
356
+ no current replacement for manual remote and receipt cleanup. Woods does not
357
+ infer ownership from a URI prefix, and `UNBLOCKED_FORCE_PURGE` cannot resolve it.
358
+ Follow [explicit ref migration](UNBLOCKED_INTEGRATION.md#explicit-ref-migration).
359
+
256
360
  **Notion needs one settling re-sync.** v2 groups column pages by physical table instead of by model, so models that share a table write one page per physical column, with the `Table` relation listing every owning model and their validations unioned. The page titles, and therefore the manifest keys, are unchanged, but the content hash of every shared-table column changes once. Expect the first post-upgrade `woods:notion_sync` to update those pages; subsequent runs skip them. This also ends the v1 behavior where two models sharing a table rewrote the same column page back and forth on every run. See [Notion integration](NOTION_INTEGRATION.md).
257
361
 
258
362
  ## Update CI and scheduled automation
@@ -311,7 +415,8 @@ names and count same-named typed variants once, matching name-based ground truth
311
415
  Older dumps remain readable; re-embed to recover variants
312
416
  that an older writer had already overwritten.
313
417
 
314
- SQLite migration 007 preserves snapshot rows and permits one row per
418
+ SQLite migration 007 preserves rows **in a database that was retained or restored**;
419
+ it cannot recover a database deleted by `woods:clean`. It permits one row per
315
420
  `(snapshot_id, identifier, unit_type)`. JSON snapshot readers accept older untyped
316
421
  records, while new records preserve both names and types. Lost historical variants
317
422
  cannot be reconstructed from old snapshots. Back up `woods.sqlite3` and the whole
@@ -346,6 +451,10 @@ Update agent prompts that refer to the old inventory. Standard Index launch prov
346
451
 
347
452
  Structural reads still work from a read-only index mount, but the `reload` tool does not: its transactional refresh takes the same on-disk writer lock as extraction and embedding, so the MCP process needs write access to the index directory. Without it, `reload` returns a typed degraded error and keeps serving the previous aligned generation rather than swapping in a partial one. Grant write access, or restart the MCP process after publishing. [MCP servers](MCP_SERVERS.md) owns the detail.
348
453
 
454
+ **Included in Woods 2.1:** `:local` snapshot vectors and SQLite metadata cannot
455
+ reload atomically in the running server, even with write access. Restart
456
+ `woods-mcp` after `woods:embed` when using that preset.
457
+
349
458
  ### Console users: preserve or configure the HTTP token
350
459
 
351
460
  If HTTP Console is enabled, preserve or configure a secret token of at least
@@ -358,8 +467,11 @@ Stdio does not use a bearer token. On versions supporting
358
467
  `console_mcp_http_enabled`, set it to `false` for stdio-only use without HTTP
359
468
  boot validation, while keeping the master `console_mcp_enabled` flag on.
360
469
  The HTTP flag defaults to `true` to preserve existing deployments; choosing a
361
- stdio client alone does not turn HTTP off. Older versions without this flag
362
- still require a token at production boot whenever Console is enabled.
470
+ stdio client alone does not turn HTTP off. In some 1.x paths, flags assigned in
471
+ `config/initializers/woods.rb` were read too early to trigger boot validation.
472
+ v2 validates the settled configuration after initializers: the same application
473
+ can now fail boot until its token is corrected or HTTP is explicitly disabled.
474
+ Do not rely on the earlier validation timing as a supported configuration.
363
475
 
364
476
  Follow [Console MCP setup](CONSOLE_MCP_SETUP.md) for transport-specific setup
365
477
  and the [Configuration reference](CONFIGURATION_REFERENCE.md) for defaults.
@@ -378,6 +490,9 @@ Complete every applicable check:
378
490
  - [ ] `search`, `lookup`, and `dependents` work with v2 identifiers.
379
491
  - [ ] Semantic retrieval works after re-embedding, if enabled.
380
492
  - [ ] Console exposes only the authorized 9/11 tools, if enabled.
493
+ - [ ] Legacy `config.log_level=` and unsafe-eval settings were removed.
494
+ - [ ] Custom Console path and session tracing settings are configured before Railtie initialization.
495
+ - [ ] Retained snapshot history is readable, and its untouched backup remains available.
381
496
  - [ ] An enabled Console reads a token of at least 32 characters from a secret source; its value was not printed or committed.
382
497
  - [ ] Console HTTP rejects a request without the bearer token with 401 and accepts the configured client, if HTTP is used.
383
498
  - [ ] Console stdio starts through the application bundle, if stdio is used.
data/docs/WATCH_DAEMON.md CHANGED
@@ -116,9 +116,17 @@ supported. Do not prepend an `environment` task or an already booted Rails runne
116
116
  The generator records relative owned paths and fingerprints in `.woods-watch.json`.
117
117
  Commit it with the generated configuration so a new clone/worktree can update or
118
118
  remove that setup. Runtime transaction state stays under `tmp/woods-watch-install/`;
119
- keep that directory ignored. Changes to owned content or executable permissions
120
- cause a conflict rather than an overwrite. Review the conflict and restore or
121
- adapt the owned setup explicitly; do not delete the receipt to force an overwrite.
119
+ keep that directory ignored. Changes to owned content or removal of the owner's
120
+ executable bit cause a conflict rather than an overwrite. Review the conflict
121
+ and restore or adapt the owned setup explicitly; do not delete the receipt to
122
+ force an overwrite.
123
+
124
+ Included in Woods 2.1: ownership compares the executable bit Git records,
125
+ so ordinary checkout permissions such as `0755` and `0775` both retain ownership.
126
+ Older builds compare the entire mode and can refuse setup in a clone made under
127
+ a different umask. Record the loaded revision before relying on this fix.
128
+ Saved preview/apply snapshots still check exact permissions: a chmod after a
129
+ preview of that file requires a fresh preview.
122
130
 
123
131
  ```bash
124
132
  # Change modes or update owned setup (include the selected mode's options).
@@ -178,6 +186,56 @@ There is no automatic ownership takeover. Managed duplicate prevention is scoped
178
186
  to one host/process namespace; do not mix raw and managed writers across different
179
187
  containers sharing an index. Use one external owner for that arrangement.
180
188
 
189
+ **Included in Woods 2.1 ([#591](https://github.com/lost-in-the/woods/issues/591)):**
190
+ empty and whitespace-only idle-timeout values behave as unset in the task,
191
+ launcher and installer. Earlier builds accepted an empty value in managed
192
+ validation but crashed during task startup; unset it when using those builds.
193
+
194
+ ### Recovering an abandoned managed claim
195
+
196
+ **Included in Woods 2.1; verify `woods-watch --help` before using this command.**
197
+ New managed daemons hold a lifetime filesystem lock in `watch_claim.json.lease`.
198
+ The Rails child holds it throughout startup, extraction and shutdown, independently
199
+ of its launcher. Its `watch_claim.json` records a fresh `token` and lease identity.
200
+ After the original owner/container has stopped, inspect the claim at the exact
201
+ index used by the application, then select its token explicitly:
202
+
203
+ ```bash
204
+ bundle exec woods-watch --recover-claim /app/tmp/woods --claim-token TOKEN_FROM_CLAIM
205
+ ```
206
+
207
+ Run this inside the environment that can access that index and its sidecars.
208
+ Recovery does not boot Rails or start a watcher. It holds both the startup
209
+ coordination lock and lifetime lease, checks the token and lease inode, and
210
+ removes only the selected abandoned claim. Restart the normal Foreman/Puma/external
211
+ owner afterward and verify startup reconciliation. Tokens select a claim; they
212
+ are not credentials. A changed token means another owner claimed the index:
213
+ inspect it rather than repeatedly substituting tokens.
214
+
215
+ Recovery preserves the old status record. Keep the reader-only
216
+ `WOODS_WATCH_TRUST_FOREIGN_HOST` override unset in the owner startup environment;
217
+ otherwise its existing status precheck can honor the retired container's last
218
+ heartbeat until the normal 15-minute freshness bound expires. Lifetime lease
219
+ protection remains active without that override.
220
+
221
+ A live local or foreign owner keeps its lease even if its claim is old. Recovery
222
+ refuses a held lock, unknown protocol, missing/replaced lease, malformed record,
223
+ or unavailable locking. Claim age, heartbeat age and a PID in a different container
224
+ never prove abandonment. The filesystem must support cooperative `flock` across
225
+ every process sharing the index; this is not a distributed lease service. Do not
226
+ mix old/raw writers with managed ownership across containers. Never unlink,
227
+ replace or recreate the `.lease` or `.lock` sidecars while writers may exist.
228
+ `woods:clean` preserves these sidecars and the claim, including during forced
229
+ cleanup; it does not reset watcher ownership.
230
+
231
+ Legacy claims have no lifetime-lease proof and cannot use this command. Stop
232
+ their original owner through its process manager. If the original namespace is
233
+ gone and ownership cannot be verified, choose a new empty output directory,
234
+ configure exactly one writer and its readers to use it, and run a fresh full
235
+ extraction. Keep the old index until the replacement is verified. This recovery
236
+ does not require deleting the unverifiable claim or guessing whether its PID is
237
+ dead. Upgrading Woods alone does not rewrite a live or abandoned legacy claim.
238
+
181
239
  Daemon liveness, completed startup reconciliation, and source freshness remain
182
240
  different facts. Managed supervision records under `watch_supervisors/` expose
183
241
  starting/retrying/parked state separately in `woods_status`; an alive supervisor
@@ -329,15 +387,27 @@ against a settled boot configuration. If Rails is already initialized or the
329
387
  handling. Use `bundle exec rake woods:watch` as a separate process, rather than
330
388
  `bundle exec rake environment woods:watch`.
331
389
 
332
- So `run` reconciles before it waits. The watermark is `generation.json`'s mtime, written last on every successful run, so it means "when this index was last
333
- known good", and everything modified since is uncovered, whoever changed it.
334
- With no generation file there is no index, every file is uncovered, and the
335
- storm threshold correctly turns that into one full extraction. A marker whose
336
- payload pointer no longer resolves counts as no index too: the marker can
337
- outlive the directory it names (a partial restore from a CI artifact, an
338
- external cleanup targeting the large directories), and readers deliberately
339
- degrade a dangling pointer to the index root, so trusting the mtime there would
340
- report "current at startup" over a directory holding nothing.
390
+ So `run` reconciles before it waits. Supporting Git builds record source
391
+ capture start as optional `captured_at` metadata in the generation's
392
+ `source_inputs.json`. Startup considers files modified from that boundary,
393
+ rounded down to include the entire second on coarse filesystems, plus recorded
394
+ source-change error paths. Publication time is too late: a file can change
395
+ between capture and publication, including after final source verification.
396
+
397
+ A bounded content scan excludes candidates only when their recorded consumer
398
+ identities all match the current bytes. Unchanged files in the capture second
399
+ therefore do not repeatedly trigger extraction. Missing identity evidence keeps
400
+ the candidate uncovered. This startup check is not a full source-freshness
401
+ certification; use [`source_freshness`](SOURCE_FRESHNESS.md) for that evidence.
402
+
403
+ An older index without this capture boundary, or one with invalid/mismatched
404
+ metadata or a dangling payload pointer, receives a full reconciliation under
405
+ the same reload/restart rules. A successful full publication records the new
406
+ boundary even when no incremental units would change, so the migration does
407
+ not repeat on every startup. With no generation at all, every file is uncovered
408
+ and the normal storm threshold applies. On older installed versions that still
409
+ use the marker mtime, recover by completing a full extraction against settled
410
+ source before restarting watch.
341
411
 
342
412
  **The built-in watcher establishes detection before reconciliation runs.**
343
413
  Polling signals readiness after its baseline scan; native watching signals after
@@ -350,8 +420,8 @@ minutes on a storm-triggered full run) used to be lost twice: no watcher
350
420
  existed yet to see it, and the polling watcher takes its baseline snapshot
351
421
  inside `start`, after the save, so its first diff already excluded it. Worse,
352
422
  the save's mtime predates the generation bump catch-up publishes at the end, so
353
- a future restart's watermark check would read the file as already covered,
354
- permanently. Starting the watcher first closes that window; `enqueue`/`drain`
423
+ older marker-based catch-up could miss it again on restart. Starting the watcher
424
+ first closes the live-event window; `enqueue`/`drain`
355
425
  already tolerate the duplicate paths this produces against whatever catch-up
356
426
  finds on its own via the tree scan.
357
427
 
@@ -366,6 +436,19 @@ convention path no app has), which authoritative deletion would wrongly remove.
366
436
  The sweep carries the bounds that make reconciliation safe; the daemon only
367
437
  supplies the trigger.
368
438
 
439
+ **Included in Woods 2.1:** if that deletion-only cycle fails or cannot take the
440
+ extraction lock, the daemon keeps a reconciliation obligation even though its
441
+ path list is empty. The next event or heartbeat retries the bounded sweep;
442
+ successful reconciliation clears the obligation, including a genuine no-op
443
+ for a nominal framework path. A restart rediscovers outstanding deletions from
444
+ the unchanged published graph.
445
+
446
+ The daemon also checks the extractor's publication error before accepting an
447
+ empty touched-unit list. Removing disabled precomputed flow artifacts changes
448
+ the payload without necessarily changing any units. If publication fails, the
449
+ previous generation remains active, status stays degraded, and the event is
450
+ retained for retry.
451
+
369
452
  This is what makes the documented hook pattern safe:
370
453
 
371
454
  ```bash
@@ -7,7 +7,7 @@
7
7
  # Configuration precedence:
8
8
  # 1. WOODS_CONSOLE_CONFIG, when set (the file must exist)
9
9
  # 2. ~/.woods/console.yml, when present
10
- # 3. direct mode with `bundle exec rake woods:console`
10
+ # 3. direct mode with executable app bin/rake, otherwise bundle exec rake
11
11
 
12
12
  require 'yaml'
13
13
  require_relative '../lib/woods/console/connection_manager'
@@ -22,7 +22,7 @@ end
22
22
 
23
23
  begin
24
24
  config = File.file?(config_path) ? YAML.safe_load_file(config_path, aliases: false) : {}
25
- config ||= {}
25
+ config = {} if config.nil?
26
26
  raise Woods::Console::ConnectionError, "#{config_path} must contain a YAML mapping" unless config.is_a?(Hash)
27
27
 
28
28
  Woods::Console::ConnectionManager.new(config: config).replace_process!
@@ -122,7 +122,8 @@ Woods.configure do |config|
122
122
 
123
123
  # config.console_mcp_enabled = false
124
124
  # config.console_mcp_http_enabled = true # set false for stdio-only Console use
125
- # config.console_mcp_path = '/mcp/console'
125
+ # A custom console_mcp_path belongs in config/application.rb before Railtie
126
+ # initialization, not this initializer. The default is '/mcp/console'.
126
127
 
127
128
  # Console HTTP requires a strong bearer token. Its Origin/Host guard is
128
129
  # loopback-only by default; list the public MCP host and any browser client