woods 2.0.0 → 2.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (167) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +108 -0
  3. data/CONTRIBUTING.md +135 -10
  4. data/README.md +1 -1
  5. data/docs/AGENT_GUIDE.md +19 -0
  6. data/docs/AGENT_SETUP.md +22 -2
  7. data/docs/BACKEND_MATRIX.md +7 -0
  8. data/docs/CLIENT_HOOKS.md +6 -0
  9. data/docs/CONFIGURATION_REFERENCE.md +133 -18
  10. data/docs/CONSOLE_MCP_SETUP.md +141 -12
  11. data/docs/EMBEDDING_MODELS.md +16 -19
  12. data/docs/EXTRACTOR_REFERENCE.md +219 -21
  13. data/docs/FAQ.md +11 -25
  14. data/docs/GETTING_STARTED.md +7 -1
  15. data/docs/INCREMENTAL_EXTRACTION.md +261 -19
  16. data/docs/INDEX_LAYOUT.md +5 -0
  17. data/docs/INTERNALS.md +9 -0
  18. data/docs/MCP_HTTP_TRANSPORT.md +59 -2
  19. data/docs/MCP_SERVERS.md +87 -8
  20. data/docs/MCP_TOOL_COOKBOOK.md +13 -55
  21. data/docs/NOTION_INTEGRATION.md +7 -1
  22. data/docs/PUBLISHED_INDEX.md +6 -0
  23. data/docs/README.md +6 -1
  24. data/docs/RETRIEVAL_GUIDE.md +17 -0
  25. data/docs/SOURCE_FRESHNESS.md +157 -5
  26. data/docs/TOKEN_BENCHMARK.md +10 -18
  27. data/docs/TROUBLESHOOTING.md +70 -14
  28. data/docs/UNBLOCKED_INTEGRATION.md +60 -8
  29. data/docs/UPGRADING_TO_2.md +184 -9
  30. data/docs/WATCH_DAEMON.md +97 -14
  31. data/exe/woods-console-mcp +2 -2
  32. data/exe/woods-mcp-http +16 -9
  33. data/lib/generators/woods/templates/woods.rb.tt +2 -1
  34. data/lib/tasks/woods.rake +23 -7
  35. data/lib/tasks/woods_checks.rake +2 -2
  36. data/lib/woods/agent_configuration/cli.rb +1 -1
  37. data/lib/woods/agent_configuration/layout.rb +16 -2
  38. data/lib/woods/agent_configuration/plan.rb +13 -3
  39. data/lib/woods/agent_configuration/planner_validation.rb +4 -2
  40. data/lib/woods/agent_configuration/preflight.rb +5 -3
  41. data/lib/woods/builder.rb +17 -57
  42. data/lib/woods/cache/cache_middleware.rb +68 -41
  43. data/lib/woods/chunking/contributor_chunks.rb +119 -0
  44. data/lib/woods/chunking/semantic_chunker.rb +44 -21
  45. data/lib/woods/console/adapter_family.rb +39 -0
  46. data/lib/woods/console/connection_manager.rb +56 -3
  47. data/lib/woods/console/credential_index.rb +33 -3
  48. data/lib/woods/console/embedded_executor.rb +431 -48
  49. data/lib/woods/console/model_validator.rb +8 -0
  50. data/lib/woods/console/rack_middleware.rb +68 -11
  51. data/lib/woods/console/redactor.rb +24 -10
  52. data/lib/woods/console/safe_context.rb +44 -7
  53. data/lib/woods/console/sql_noise_stripper.rb +41 -12
  54. data/lib/woods/console/sql_table_scanner.rb +45 -34
  55. data/lib/woods/console/sql_validator.rb +37 -2
  56. data/lib/woods/dependency_graph.rb +34 -10
  57. data/lib/woods/embedding/fake.rb +12 -0
  58. data/lib/woods/embedding/indexer.rb +195 -98
  59. data/lib/woods/embedding/input_budget.rb +67 -0
  60. data/lib/woods/embedding/openai.rb +70 -20
  61. data/lib/woods/embedding/provider.rb +37 -25
  62. data/lib/woods/embedding/text_preparer.rb +76 -32
  63. data/lib/woods/embedding/token_counter.rb +18 -81
  64. data/lib/woods/embedding/vector_configuration.rb +48 -0
  65. data/lib/woods/extraction_identities.rb +175 -0
  66. data/lib/woods/extractor.rb +304 -107
  67. data/lib/woods/extractors/action_cable_extractor.rb +8 -3
  68. data/lib/woods/extractors/assigned_value_discovery.rb +74 -0
  69. data/lib/woods/extractors/class_declarations.rb +121 -0
  70. data/lib/woods/extractors/configuration_extractor.rb +11 -3
  71. data/lib/woods/extractors/declaration_ancestry.rb +92 -0
  72. data/lib/woods/extractors/event_extractor.rb +8 -0
  73. data/lib/woods/extractors/graphql_extractor.rb +134 -77
  74. data/lib/woods/extractors/job_extractor.rb +5 -1
  75. data/lib/woods/extractors/lib_extractor.rb +132 -15
  76. data/lib/woods/extractors/mailer_extractor.rb +3 -5
  77. data/lib/woods/extractors/manager_extractor.rb +7 -21
  78. data/lib/woods/extractors/migration_declaration.rb +87 -0
  79. data/lib/woods/extractors/migration_extractor.rb +5 -39
  80. data/lib/woods/extractors/phlex_extractor.rb +6 -2
  81. data/lib/woods/extractors/policy_extractor.rb +9 -5
  82. data/lib/woods/extractors/poro_extractor.rb +112 -53
  83. data/lib/woods/extractors/pundit_extractor.rb +11 -6
  84. data/lib/woods/extractors/scheduled_job_extractor.rb +45 -4
  85. data/lib/woods/extractors/serializer_extractor.rb +34 -22
  86. data/lib/woods/extractors/shared_utility_methods.rb +18 -1
  87. data/lib/woods/extractors/source_nesting.rb +142 -106
  88. data/lib/woods/extractors/standalone_module_discovery.rb +123 -0
  89. data/lib/woods/extractors/state_machine_extractor.rb +46 -40
  90. data/lib/woods/extractors/view_component_extractor.rb +9 -7
  91. data/lib/woods/flow_assembler.rb +4 -1
  92. data/lib/woods/generation.rb +25 -0
  93. data/lib/woods/hooks/context_hint.rb +7 -2
  94. data/lib/woods/mcp/bearer_auth.rb +1 -1
  95. data/lib/woods/mcp/bootstrapper.rb +33 -7
  96. data/lib/woods/mcp/config_resolver.rb +26 -7
  97. data/lib/woods/mcp/index_reader.rb +125 -24
  98. data/lib/woods/mcp/index_reader_pinning.rb +16 -0
  99. data/lib/woods/mcp/origin_guard.rb +24 -77
  100. data/lib/woods/mcp/origin_policy.rb +124 -0
  101. data/lib/woods/mcp/renderers/markdown_renderer.rb +7 -1
  102. data/lib/woods/mcp/renderers/plain_renderer.rb +3 -1
  103. data/lib/woods/mcp/search_results.rb +7 -1
  104. data/lib/woods/mcp/server.rb +24 -4
  105. data/lib/woods/module_reconciliation.rb +151 -0
  106. data/lib/woods/path_dispatcher.rb +7 -2
  107. data/lib/woods/railtie_support.rb +8 -0
  108. data/lib/woods/rake_helpers.rb +43 -11
  109. data/lib/woods/release.rb +1 -1
  110. data/lib/woods/resilience/index_validator.rb +8 -3
  111. data/lib/woods/resilience/retryable_provider.rb +18 -1
  112. data/lib/woods/resolved_config.rb +68 -8
  113. data/lib/woods/retrieval/context_assembler.rb +3 -3
  114. data/lib/woods/retrieval/lexical_assembler.rb +3 -2
  115. data/lib/woods/retrieval/scope.rb +18 -2
  116. data/lib/woods/retrieval/source_evidence.rb +14 -2
  117. data/lib/woods/source_contributor_validation.rb +78 -0
  118. data/lib/woods/source_contributors.rb +116 -0
  119. data/lib/woods/source_inputs/handoff.rb +37 -0
  120. data/lib/woods/source_inputs/launcher.rb +53 -13
  121. data/lib/woods/source_inputs/manifest.rb +84 -3
  122. data/lib/woods/source_inputs/private_key.rb +44 -12
  123. data/lib/woods/source_inputs/scanner.rb +98 -27
  124. data/lib/woods/source_inputs/scopes.rb +1 -1
  125. data/lib/woods/source_inputs/session.rb +147 -15
  126. data/lib/woods/source_inputs/stable_reader.rb +127 -0
  127. data/lib/woods/source_inputs/status.rb +40 -8
  128. data/lib/woods/source_inputs/verifier.rb +28 -5
  129. data/lib/woods/source_path_encoding.rb +33 -0
  130. data/lib/woods/source_references/cache.rb +284 -0
  131. data/lib/woods/source_references/collector.rb +120 -0
  132. data/lib/woods/source_references/extraction.rb +185 -0
  133. data/lib/woods/source_references/inputs.rb +134 -0
  134. data/lib/woods/source_references/parser_adapter.rb +134 -0
  135. data/lib/woods/source_references/pass.rb +152 -0
  136. data/lib/woods/source_references/prism_adapter.rb +116 -0
  137. data/lib/woods/source_references/registry.rb +178 -0
  138. data/lib/woods/source_references/runtime_lookup.rb +127 -0
  139. data/lib/woods/source_references/value_class.rb +82 -0
  140. data/lib/woods/storage/metadata_store.rb +4 -1
  141. data/lib/woods/storage/qdrant.rb +2 -2
  142. data/lib/woods/unblocked/client.rb +12 -7
  143. data/lib/woods/unblocked/document_builder.rb +4 -1
  144. data/lib/woods/unblocked/exporter.rb +127 -37
  145. data/lib/woods/unblocked/sync_manifest.rb +137 -21
  146. data/lib/woods/unblocked/uri_migration.rb +105 -0
  147. data/lib/woods/util/host_guard.rb +3 -2
  148. data/lib/woods/version.rb +1 -1
  149. data/lib/woods/watch/catch_up.rb +138 -0
  150. data/lib/woods/watch/claim_lease.rb +150 -0
  151. data/lib/woods/watch/cli.rb +26 -2
  152. data/lib/woods/watch/daemon.rb +80 -59
  153. data/lib/woods/watch/installation/options.rb +1 -1
  154. data/lib/woods/watch/installation/receipt.rb +6 -1
  155. data/lib/woods/watch/managed_child.rb +1 -1
  156. data/lib/woods/watch/supervisor.rb +1 -1
  157. data/lib/woods/watch/tree_scan.rb +14 -2
  158. data/plugin/.claude-plugin/plugin.json +1 -1
  159. data/plugin/hooks/adapters/normalize.rb +3 -2
  160. data/plugin/hooks/woods-input-rules.sh +4 -0
  161. data/plugin/hooks/woods-refresh.sh +15 -7
  162. data/plugin/hooks/woods-session-start.sh +60 -3
  163. data/plugin/skills/woods-diagnose/SKILL.md +336 -2
  164. data/plugin/skills/woods-investigate/SKILL.md +11 -0
  165. data/plugin/skills/woods-mcp-config/SKILL.md +86 -0
  166. data/plugin/skills/woods-setup/SKILL.md +56 -1
  167. metadata +34 -5
data/lib/tasks/woods.rake CHANGED
@@ -58,6 +58,7 @@ namespace :woods do
58
58
  desc 'Incremental extraction based on git changes'
59
59
  task incremental: :environment do
60
60
  require 'woods/extractor'
61
+ require 'woods/input_rules'
61
62
 
62
63
  output_dir = ENV.fetch('WOODS_OUTPUT', Woods.configuration.output_dir)
63
64
 
@@ -71,8 +72,9 @@ namespace :woods do
71
72
  # PathDispatcher alongside the dispatch itself — a second hand-maintained
72
73
  # pattern list here would drift, and a path this filter drops never
73
74
  # reaches the index however good the dispatch behind it is (#164).
74
- dispatcher = Woods::PathDispatcher.new
75
- changed_files = changed_files.reject(&:empty?).select { |f| dispatcher.relevant?(f) }
75
+ input_rules = Woods::InputRules.new
76
+ changed_files = changed_files.reject(&:empty?).reject { |path| input_rules.action(path) == :ignore }
77
+ full = changed_files.any? { |path| input_rules.action(path) == :full }
76
78
 
77
79
  if changed_files.empty?
78
80
  puts 'No relevant files changed. Skipping extraction.'
@@ -91,13 +93,17 @@ namespace :woods do
91
93
  puts 'Warning: a watch daemon is alive but degraded — extracting anyway rather than assuming coverage.'
92
94
  end
93
95
 
94
- puts "Incremental extraction for #{changed_files.size} changed files..."
96
+ puts "#{full ? 'Full' : 'Incremental'} extraction for #{changed_files.size} changed files..."
95
97
  changed_files.each { |f| puts " - #{f}" }
96
98
  puts
97
99
 
98
100
  extractor = Woods::Extractor.new(output_dir: output_dir)
99
101
  affected = Woods::RakeHelpers.woods_with_extraction_lock(output_dir) do
100
- extracted = extractor.extract_changed(changed_files)
102
+ extracted = if full
103
+ extractor.extract_all.values.flatten(1).map(&:identifier).uniq
104
+ else
105
+ extractor.extract_changed(changed_files)
106
+ end
101
107
  extractor.raise_on_publication_failure!
102
108
  extracted
103
109
  end
@@ -168,7 +174,7 @@ namespace :woods do
168
174
  # FS events do not propagate and the daemon would sit silent. Nothing
169
175
  # exposed it before, which made the advice unfollowable.
170
176
  force_polling: ENV['WOODS_WATCH_POLL'] == '1', # container autodetect also applies; see Watcher.containerized?
171
- idle_timeout: ENV.fetch('WOODS_WATCH_IDLE_TIMEOUT', nil) && Float(ENV.fetch('WOODS_WATCH_IDLE_TIMEOUT')),
177
+ idle_timeout: Woods::RakeHelpers.woods_watch_idle_timeout,
172
178
  catch_up: ENV['WOODS_WATCH_CATCH_UP'] != '0',
173
179
  boot_snapshot: boot_snapshot,
174
180
  lifecycle: managed_child,
@@ -724,6 +730,8 @@ namespace :woods do
724
730
  env_flag = ->(name) { %w[1 true yes].include?(ENV.fetch(name, '').strip.downcase) }
725
731
  force_full = env_flag.call('UNBLOCKED_FORCE_FULL_SYNC')
726
732
  force_purge = env_flag.call('UNBLOCKED_FORCE_PURGE')
733
+ dry_run = env_flag.call('UNBLOCKED_DRY_RUN')
734
+ migrate_from_ref = ENV.fetch('UNBLOCKED_MIGRATE_FROM_REF', nil)
727
735
 
728
736
  puts 'Syncing extraction data to Unblocked...'
729
737
  puts " Output dir: #{output_dir}"
@@ -735,12 +743,19 @@ namespace :woods do
735
743
  exporter = Woods::Unblocked::Exporter.new(
736
744
  index_dir: output_dir,
737
745
  force_full: force_full,
738
- force_purge: force_purge
746
+ force_purge: force_purge,
747
+ dry_run: dry_run,
748
+ migrate_from_ref: migrate_from_ref
739
749
  )
740
750
  stats = exporter.sync_all
741
751
 
742
752
  puts
743
- puts 'Sync complete!'
753
+ if stats[:dry_run]
754
+ puts 'Sync preview (no remote requests or local writes):'
755
+ puts JSON.pretty_generate(stats)
756
+ next
757
+ end
758
+ puts(stats[:complete] == false ? 'Sync incomplete; re-run after resolving the reported condition.' : 'Sync complete!')
744
759
  puts " Documents synced: #{stats[:synced]}"
745
760
  puts " Documents skipped: #{stats[:skipped]}"
746
761
  puts " Documents deleted: #{stats[:deleted]}"
@@ -765,6 +780,7 @@ namespace :woods do
765
780
  exit 1
766
781
  end
767
782
  end
783
+ exit 1 if stats[:complete] == false && stats[:errors].empty?
768
784
  end
769
785
 
770
786
  desc 'Relay findings to Unblocked (alias for unblocked_sync)'
@@ -11,6 +11,7 @@
11
11
  # WOODS_CHECK_STRICT=1 ... # exit 1 on findings (still heuristic; see below)
12
12
 
13
13
  require 'json'
14
+ require 'woods/rake_helpers'
14
15
 
15
16
  namespace :woods do
16
17
  namespace :check do
@@ -62,8 +63,7 @@ module Woods
62
63
  # @return [String]
63
64
  def check_index_dir
64
65
  ENV.fetch('WOODS_OUTPUT') do
65
- root = respond_to?(:woods_task_root, true) ? woods_task_root : Rake.application.original_dir
66
- File.join(root, 'tmp/woods')
66
+ File.join(Woods::RakeHelpers.woods_task_root, 'tmp/woods')
67
67
  end
68
68
  end
69
69
 
@@ -86,7 +86,7 @@ module Woods
86
86
  protected_paths = layout.allowed_paths + layout.runtime_paths
87
87
  raise Conflict, 'Plan output must differ from every managed/runtime target' if protected_paths.include?(target)
88
88
 
89
- Document.validate_path!(target)
89
+ Plan.validate_path!(target)
90
90
  content = "#{JSON.pretty_generate(plan.data)}\n"
91
91
  raise Conflict, 'Plan exceeds the supported size' if content.bytesize > Document::MAX_BYTES
92
92
 
@@ -64,8 +64,22 @@ module Woods
64
64
  end
65
65
 
66
66
  def identity
67
- { 'client' => 'claude', 'scope' => scope, 'root' => root, 'config_dir' => config_dir,
68
- 'config_path' => config_path, 'receipt_path' => receipt_path }
67
+ { 'client' => 'claude', 'scope' => scope, 'root' => root,
68
+ 'config_path' => config_path, 'receipt_path' => receipt_path }.tap do |value|
69
+ value['config_dir'] = config_dir if scope == 'user'
70
+ end
71
+ end
72
+
73
+ # Older project receipts/plans included an unused user configuration path.
74
+ # Only that field is irrelevant; every actual target and scope stays exact.
75
+ def self.validate_identity!(record, expected, message:)
76
+ raise Conflict, "#{message} (identity field: layout)" unless record.is_a?(Hash)
77
+
78
+ record = record.except('config_dir') if expected['client'] == 'claude' && expected['scope'] == 'project'
79
+ return if record == expected
80
+
81
+ fields = (record.keys | expected.keys).reject { |key| record.slice(key) == expected.slice(key) }
82
+ raise Conflict, "#{message} (identity fields: #{fields.join(', ')})"
69
83
  end
70
84
  end
71
85
  end
@@ -3,6 +3,7 @@
3
3
  require 'base64'
4
4
  require 'json'
5
5
  require_relative 'document'
6
+ require_relative 'layout'
6
7
 
7
8
  module Woods
8
9
  module AgentConfiguration
@@ -19,11 +20,19 @@ module Woods
19
20
  end
20
21
 
21
22
  def self.load(path)
23
+ validate_path!(path)
22
24
  object = allocate
23
25
  object.instance_variable_set(:@data, Document.new(path).json)
24
26
  object
25
27
  end
26
28
 
29
+ def self.validate_path!(path)
30
+ Document.validate_path!(path)
31
+ rescue Conflict => e
32
+ raise Conflict, "#{e.message}. Use a regular plan file under a real parent directory; " \
33
+ 'for temporary plans, create a private directory and resolve it with File.realpath.'
34
+ end
35
+
27
36
  def add(document, content, description:)
28
37
  return if content == document.content
29
38
 
@@ -67,12 +76,13 @@ module Woods
67
76
  private
68
77
 
69
78
  def validate_format!(layout)
70
- valid = data.is_a?(Hash) && data['schema_version'] == SCHEMA_VERSION && data['layout'] == layout.identity &&
79
+ valid = data.is_a?(Hash) && data['schema_version'] == SCHEMA_VERSION &&
71
80
  %w[setup update remove].include?(data['operation']) && data['changes'].is_a?(Array) &&
72
81
  data['changes'].all?(Hash)
73
- return if valid
82
+ message = 'Plan format or selected application/client/scope differs; create a fresh preview'
83
+ raise Conflict, message unless valid
74
84
 
75
- raise Conflict, 'Plan format or selected application/client/scope differs; create a fresh preview'
85
+ Layout.validate_identity!(data['layout'], layout.identity, message: message)
76
86
  end
77
87
 
78
88
  def validate_paths!(layout)
@@ -19,11 +19,13 @@ module Woods
19
19
  end
20
20
 
21
21
  def validate_receipt_identity!
22
- valid = @previous['schema_version'] == 1 && @previous['layout'] == @layout.identity &&
22
+ valid = @previous['schema_version'] == 1 &&
23
23
  @previous['server_name'] == @name && @previous['sections'].is_a?(Hash) &&
24
24
  @previous['created_files'].is_a?(Array)
25
- raise Conflict, 'Installation receipt does not match this application/client/scope/server' unless valid
25
+ message = 'Installation receipt does not match this application/client/scope/server'
26
+ raise Conflict, message unless valid
26
27
 
28
+ Layout.validate_identity!(@previous['layout'], @layout.identity, message: message)
27
29
  validate_receipt_content!
28
30
  end
29
31
 
@@ -3,8 +3,8 @@
3
3
  require 'json'
4
4
  require 'open3'
5
5
  require 'timeout'
6
- require 'bundler'
7
6
  require_relative 'error'
7
+ require_relative '../watch/child_environment'
8
8
 
9
9
  module Woods
10
10
  module AgentConfiguration
@@ -32,8 +32,9 @@ module Woods
32
32
  end
33
33
  RUBY
34
34
 
35
- def initialize(timeout: 30)
35
+ def initialize(timeout: 30, environment: ENV)
36
36
  @timeout = timeout
37
+ @environment = environment
37
38
  end
38
39
 
39
40
  def call(launcher)
@@ -54,7 +55,8 @@ module Woods
54
55
 
55
56
  def run(launcher)
56
57
  command = launcher.probe_command(SCRIPT)
57
- environment = Bundler.unbundled_env.merge(launcher.probe_environment)
58
+ environment = Watch::ChildEnvironment.build(@environment.to_h, root: launcher.root)
59
+ .merge(launcher.probe_environment)
58
60
  Open3.popen3(environment, *command, chdir: launcher.root, unsetenv_others: true,
59
61
  pgroup: true) do |stdin, out, err, wait|
60
62
  stdin.close
data/lib/woods/builder.rb CHANGED
@@ -191,10 +191,9 @@ module Woods
191
191
  # their own implementation without patching the Builder. It flows
192
192
  # through {#build_resilient_embedding_provider} like the built-ins.
193
193
  #
194
- # Strips `embedding_options` keys that belong to the ResolvedConfig layer
195
- # (like `:dimension`) before splatting into the provider's constructor —
196
- # those keys are useful for the Snapshotter's schema header but
197
- # aren't part of the provider's API.
194
+ # The legacy :dimension option aliases the explicit :dimensions request.
195
+ # Stored observed widths use :expected_dimensions so restoring an index
196
+ # does not accidentally change the provider request.
198
197
  #
199
198
  # @return [Embedding::Provider::Interface] Embedding provider instance
200
199
  # @raise [ArgumentError] if the configured type is not recognized
@@ -244,8 +243,8 @@ module Woods
244
243
  end
245
244
 
246
245
  PROVIDER_OPTION_KEYS = {
247
- openai: %i[api_key model dimension dimensions],
248
- ollama: %i[model host num_ctx read_timeout dimension dimensions],
246
+ openai: %i[api_key model dimension dimensions expected_dimensions],
247
+ ollama: %i[model host num_ctx read_timeout dimension dimensions expected_dimensions],
249
248
  fake: %i[model dims dimension dimensions]
250
249
  }.freeze
251
250
  private_constant :PROVIDER_OPTION_KEYS
@@ -328,10 +327,9 @@ module Woods
328
327
  # Build the deterministic fake provider (#178).
329
328
  #
330
329
  # Dimension resolution: `embedding_options[:dims]` maps directly onto
331
- # the {Embedding::Provider::Fake} constructor; failing that, the
332
- # ResolvedConfig-level `embedding_options[:dimension]` key — normally
333
- # snapshot-only bookkeeping stripped by {#provider_kwargs} — is
334
- # honoured, so hosts that declare their dimension there (and the MCP
330
+ # the {Embedding::Provider::Fake} constructor; failing that, the legacy
331
+ # `embedding_options[:dimension]` alias is honoured, so hosts that
332
+ # declare their dimension there (and the MCP
335
333
  # boot path, which restores exactly that key from woods.json) get
336
334
  # vectors of the recorded dimension.
337
335
  #
@@ -341,54 +339,20 @@ module Woods
341
339
  end
342
340
  private :build_fake_provider
343
341
 
344
- # Build a {Embedding::TextPreparer} calibrated to a given provider.
345
- #
346
- # OpenAI embedders use tiktoken (cl100k_base) — 4.0 chars/token is a
347
- # good conservative average. Ollama BERT/WordPiece tokenizers
348
- # (nomic-embed-text, bge-*) run much hotter on dense Ruby/Rails
349
- # source — long CamelCase constants, docstrings, callback DSLs, and
350
- # heavy symbol use all sit below 2.0 chars/token in practice.
351
- # Empirically, a 16 KB chunk of `ActionMailer::Base` still blows the
352
- # 8192-token budget at 2.0 chars/token, so we budget at 1.5 to stay
353
- # clear of tokenizer surprises even on the densest Rails internals.
354
- #
355
- # `max_tokens` tracks the provider's actual input budget when it
356
- # reports one, falling back to the TextPreparer default otherwise.
357
- #
358
- # @param provider [Embedding::Provider::Interface]
359
- # @return [Embedding::TextPreparer]
342
+ # Build a preparer with the provider's explicit input-counting policy.
343
+ # Ratios remain estimates; known OpenAI models use a byte-BPE upper bound.
360
344
  def build_text_preparer(provider)
361
345
  chars_per_token = chars_per_token_for(provider)
362
346
  budget = safe_max_input_tokens(provider)
363
347
  max_tokens = budget || Embedding::TextPreparer::DEFAULT_MAX_TOKENS
364
348
 
365
- Embedding::TextPreparer.new(max_tokens: max_tokens, chars_per_token: chars_per_token)
349
+ input_budget = Embedding::InputBudget.for(provider, limit: max_tokens, chars_per_token: chars_per_token)
350
+ Embedding::TextPreparer.new(max_tokens: max_tokens, chars_per_token: chars_per_token, input_budget: input_budget)
366
351
  end
367
352
 
368
- # Build a {Chunking::SemanticChunker} sized to a given provider.
369
- #
370
- # `max_chars` is derived from the provider's input budget and the
371
- # matching chars-per-token ratio, minus the context-prefix
372
- # allowance the Indexer accounts for separately. Units that exceed
373
- # this ceiling get sliced so no single chunk can blow the provider's
374
- # input cap.
375
- #
376
- # For Ollama (and other BERT/WordPiece-backed models), char-based
377
- # estimation is unreliable — CamelCase, `::` separators, and symbol
378
- # literals tokenize much denser than chars/token averages suggest.
379
- # When the optional `tokenizers` gem is installed, pass a
380
- # {Embedding::TokenCounter} and `max_tokens` so the chunker can
381
- # verify every slice with the real tokenizer and re-split any piece
382
- # that still exceeds `num_ctx`. See docs/EMBEDDING_MODELS.md.
383
- #
384
- # Ollama v0.13.5+ stopped honouring `truncate: true` on `/api/embed`
385
- # (ollama/ollama#14186), so any chunk that exceeds `num_ctx` returns
386
- # a 400 rather than being silently truncated. Exact client-side
387
- # sizing is the only reliable path until the regression is fixed
388
- # upstream.
389
- #
390
- # @param provider [Embedding::Provider::Interface]
391
- # @return [Chunking::SemanticChunker]
353
+ # Initial semantic sizing is approximate. TextPreparer subsequently checks
354
+ # every complete prefix plus source against the provider policy and splits
355
+ # losslessly; no fixed prefix allowance establishes final admission.
392
356
  def build_chunker(provider)
393
357
  budget = safe_max_input_tokens(provider)
394
358
  max_chars = ((budget * chars_per_token_for(provider)).floor - CHUNKER_PREFIX_ALLOWANCE if budget)
@@ -424,12 +388,8 @@ module Woods
424
388
 
425
389
  private
426
390
 
427
- # Return a TokenCounter for providers that benefit from exact token
428
- # counting. OpenAI's tiktoken ratios are already stable at 4.0
429
- # chars/token on code, so it doesn't need this.
430
- #
431
- # @param provider [Embedding::Provider::Interface]
432
- # @return [Embedding::TokenCounter, nil]
391
+ # Ollama has no universal local tokenizer. This counter is an explicitly
392
+ # approximate sizing hint; truncate:false enforces the server's actual cap.
433
393
  def token_counter_for(provider)
434
394
  return unless unwrap_provider(provider).is_a?(Embedding::Provider::Ollama)
435
395
 
@@ -1,6 +1,7 @@
1
1
  # frozen_string_literal: true
2
2
 
3
3
  require 'digest'
4
+ require 'securerandom'
4
5
  require_relative 'cache_store'
5
6
  # CachedEmbeddingProvider includes Embedding::Provider::Interface at load
6
7
  # time, so the interface must be defined before this file's class bodies run
@@ -108,6 +109,9 @@ module Woods
108
109
  class CachedEmbeddingProvider
109
110
  include Embedding::Provider::Interface
110
111
 
112
+ # @return [Object] underlying provider, including any resilience wrapper
113
+ attr_reader :provider
114
+
111
115
  # @param provider [Embedding::Provider::Interface] The real embedding provider
112
116
  # @param cache_store [CacheStore] Cache backend instance
113
117
  # @param ttl [Integer] TTL for cached embeddings in seconds
@@ -117,6 +121,10 @@ module Woods
117
121
  @ttl = ttl
118
122
  @inflight = {}
119
123
  @inflight_mutex = Mutex.new
124
+ identity = @provider.cache_identity if @provider.respond_to?(:cache_identity)
125
+ # Unknown custom providers cannot prove cross-instance compatibility.
126
+ # Hash every identity so endpoint credentials never enter backend keys.
127
+ @embedding_identity = Digest::SHA256.hexdigest(JSON.generate(identity || SecureRandom.hex(16)))
120
128
  end
121
129
 
122
130
  # Embed a single text, returning a cached vector when available.
@@ -128,7 +136,7 @@ module Woods
128
136
  # @param text [String] Text to embed
129
137
  # @return [Array<Float>] Embedding vector
130
138
  def embed(text)
131
- cached = @cache_store.read(embedding_key(text))
139
+ cached = read_cached(text)
132
140
  return cached unless cached.nil?
133
141
 
134
142
  with_single_flight(text) { @provider.embed(text) }
@@ -147,13 +155,17 @@ module Woods
147
155
  # @return [Array<Array<Float>>] Embedding vectors (same order as input)
148
156
  def embed_batch(texts)
149
157
  results, misses, miss_indices = partition_cached(texts)
150
- return results if misses.empty?
158
+ if misses.empty?
159
+ validate_vectors!(results, texts.size)
160
+ return results
161
+ end
151
162
 
152
163
  to_fetch, to_fetch_positions, our_entries, awaiting = claim_inflight(misses)
153
164
 
154
165
  fetch_and_fulfill(to_fetch, to_fetch_positions, our_entries, results, miss_indices)
155
166
  await_others(awaiting, results, miss_indices)
156
167
 
168
+ validate_vectors!(results, texts.size)
157
169
  results
158
170
  end
159
171
 
@@ -171,6 +183,23 @@ module Woods
171
183
  @provider.model_name
172
184
  end
173
185
 
186
+ # Preserve pure configuration declarations through this wrapper.
187
+ def cache_identity
188
+ @provider.cache_identity if @provider.respond_to?(:cache_identity)
189
+ end
190
+
191
+ def input_budget
192
+ @provider.input_budget if @provider.respond_to?(:input_budget)
193
+ end
194
+
195
+ def configured_dimensions
196
+ @provider.configured_dimensions if @provider.respond_to?(:configured_dimensions)
197
+ end
198
+
199
+ def requested_dimensions
200
+ @provider.requested_dimensions if @provider.respond_to?(:requested_dimensions)
201
+ end
202
+
174
203
  # Delegate the per-provider input cap so Builder's chunker / text
175
204
  # preparer wiring keeps working when the cache wrapper is in front
176
205
  # of the provider. Without this, `respond_to?(:max_input_tokens)`
@@ -200,6 +229,7 @@ module Woods
200
229
 
201
230
  begin
202
231
  vector = yield
232
+ validate_vectors!([vector], 1)
203
233
  write_cache(text, vector)
204
234
  entry.fulfill(vector)
205
235
  vector
@@ -283,13 +313,7 @@ module Woods
283
313
 
284
314
  begin
285
315
  fresh_vectors = @provider.embed_batch(to_fetch)
286
- # Reject a malformed provider response up-front rather than silently
287
- # fulfilling waiters with `nil` (or masking a missing tail vector by
288
- # under-writing the cache).
289
- if fresh_vectors.size != to_fetch.size
290
- raise ArgumentError,
291
- "provider returned #{fresh_vectors.size} vectors for #{to_fetch.size} texts"
292
- end
316
+ validate_fresh_vectors!(fresh_vectors, to_fetch.size)
293
317
  rescue StandardError => e
294
318
  our_entries.each { |entry| entry.reject(e) }
295
319
  raise
@@ -350,7 +374,7 @@ module Woods
350
374
  miss_indices = []
351
375
 
352
376
  texts.each_with_index do |text, idx|
353
- cached = @cache_store.read(embedding_key(text))
377
+ cached = read_cached(text)
354
378
  if cached
355
379
  results[idx] = cached
356
380
  else
@@ -362,29 +386,32 @@ module Woods
362
386
  [results, misses, miss_indices]
363
387
  end
364
388
 
365
- # Build a cache key for an embedding text.
366
- #
367
- # The provider's model_name is folded into the key: a cached embedding
368
- # is only valid for the exact model that produced it. Without this, a
369
- # persistent shared backend (Redis/SolidCache) returns the previous
370
- # model's vector after a model switch or upgrade — different dimensions
371
- # error mid-batch, same dimensions silently corrupt similarity scores
372
- # (which the provider-vs-store dimension check can't detect, since it
373
- # compares declared widths, not cache contents).
374
- #
375
- # model_name (a plain attribute) is used rather than dimensions on
376
- # purpose: for every supported provider the model uniquely determines
377
- # the vector dimensionality, and `Provider#dimensions` can force a live
378
- # network probe (Ollama memoizes `embed('test').length`; OpenAI probes
379
- # for unknown models). Keying on dimensions made every cache lookup —
380
- # including hits — depend on the provider being reachable, defeating the
381
- # cache exactly when the backend is down. model_name distinguishes
382
- # models without any I/O.
383
- #
389
+ def read_cached(text)
390
+ vector = @cache_store.read(embedding_key(text))
391
+ validate_vectors!([vector], 1) unless vector.nil?
392
+ vector
393
+ end
394
+
395
+ # Validate the complete batch before writing or fulfilling any entry.
396
+ def validate_fresh_vectors!(vectors, count)
397
+ raise ArgumentError, "provider returned #{vectors.size} vectors for #{count} texts" if vectors.size != count
398
+
399
+ validate_vectors!(vectors, count)
400
+ end
401
+
402
+ def validate_vectors!(vectors, count)
403
+ width = @provider.configured_dimensions if @provider.respond_to?(:configured_dimensions)
404
+ Embedding::Provider::VectorValidation.validate!(
405
+ vectors, expected_count: count, provider: 'CachedEmbeddingProvider', expected_dimensions: width
406
+ )
407
+ end
408
+
409
+ # Configuration identity is captured without probing dimensions. Version
410
+ # the domain to prevent reuse of old model-only keys after an upgrade.
384
411
  # @param text [String]
385
412
  # @return [String]
386
413
  def embedding_key(text)
387
- Cache.cache_key(:embeddings, @provider.model_name.to_s, Digest::SHA256.hexdigest(text))
414
+ Cache.cache_key(:embeddings, 'configuration-v1', @embedding_identity, Digest::SHA256.hexdigest(text))
388
415
  end
389
416
  end
390
417
 
@@ -408,6 +435,8 @@ module Woods
408
435
  @retriever = retriever
409
436
  @cache_store = cache_store
410
437
  @context_ttl = context_ttl
438
+ @context_namespace = SecureRandom.hex(16)
439
+ @context_mutex = Mutex.new
411
440
  end
412
441
 
413
442
  # Expose the wrapped stores so the MCP +reload+ tool and
@@ -427,20 +456,17 @@ module Woods
427
456
  @retriever.corpus_status(include_types: include_types) if @retriever.respond_to?(:corpus_status)
428
457
  end
429
458
 
430
- # Invalidate every cached context result. Called from the MCP +reload+
431
- # tool after the retriever's stores have been re-hydrated from a fresh
432
- # embed — otherwise cached results from the old embedding run would
433
- # linger until their TTL expires and contradict the new stores.
459
+ # Retire this retriever's cached contexts after its corpus reloads.
434
460
  #
435
- # Embedding caches (query → vector) are NOT cleared: the query-vector
436
- # mapping is deterministic for a given provider+model and survives any
437
- # index reload. Only context results (query → ranked units) go stale.
461
+ # Each retriever has its own unpredictable namespace, even when several
462
+ # applications share a backend. Rotating it also prevents an in-flight
463
+ # request from repopulating the active cache with a pre-reload result.
464
+ # Old entries expire by their configured TTL; no global backend deletion
465
+ # is needed. Embedding caches are independent and remain available.
438
466
  #
439
467
  # @return [void]
440
468
  def invalidate_context_cache!
441
- @cache_store.clear(namespace: :context)
442
- rescue StandardError => e
443
- warn("[Woods] CachedRetriever context-cache invalidation failed: #{e.message}")
469
+ @context_mutex.synchronize { @context_namespace = SecureRandom.hex(16) }
444
470
  end
445
471
 
446
472
  # Execute the retrieval pipeline with context-level caching.
@@ -492,7 +518,8 @@ module Woods
492
518
  # @param exclude_types [Array<String, Symbol>, nil]
493
519
  # @return [String]
494
520
  def context_key(query, budget, types: nil, exclude_types: nil, packages: nil, source_paths: nil, evidence: 'full') # rubocop:disable Metrics/ParameterLists
495
- parts = [query, budget.to_s, fingerprint(types), fingerprint(exclude_types)]
521
+ namespace = @context_mutex.synchronize { @context_namespace }
522
+ parts = [namespace, query, budget.to_s, fingerprint(types), fingerprint(exclude_types)]
496
523
  parts << 'lexical' if mode == :lexical
497
524
  parts << JSON.generate(evidence: evidence) unless evidence == 'full'
498
525
  if Retrieval::Scope.requested?(packages: packages, source_paths: source_paths)