woods 2.0.1 → 2.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (154) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +94 -7
  3. data/CONTRIBUTING.md +134 -19
  4. data/README.md +1 -1
  5. data/docs/AGENT_GUIDE.md +19 -0
  6. data/docs/AGENT_SETUP.md +22 -2
  7. data/docs/BACKEND_MATRIX.md +7 -0
  8. data/docs/CLIENT_HOOKS.md +6 -0
  9. data/docs/CONFIGURATION_REFERENCE.md +133 -25
  10. data/docs/CONSOLE_MCP_SETUP.md +82 -30
  11. data/docs/EMBEDDING_MODELS.md +16 -19
  12. data/docs/EXTRACTOR_REFERENCE.md +219 -21
  13. data/docs/FAQ.md +11 -25
  14. data/docs/GETTING_STARTED.md +7 -1
  15. data/docs/INCREMENTAL_EXTRACTION.md +261 -19
  16. data/docs/INDEX_LAYOUT.md +5 -0
  17. data/docs/INTERNALS.md +9 -0
  18. data/docs/MCP_HTTP_TRANSPORT.md +20 -15
  19. data/docs/MCP_SERVERS.md +87 -8
  20. data/docs/MCP_TOOL_COOKBOOK.md +13 -55
  21. data/docs/NOTION_INTEGRATION.md +7 -1
  22. data/docs/PUBLISHED_INDEX.md +6 -0
  23. data/docs/README.md +6 -1
  24. data/docs/RETRIEVAL_GUIDE.md +17 -0
  25. data/docs/SOURCE_FRESHNESS.md +157 -5
  26. data/docs/TOKEN_BENCHMARK.md +10 -18
  27. data/docs/TROUBLESHOOTING.md +70 -14
  28. data/docs/UNBLOCKED_INTEGRATION.md +60 -8
  29. data/docs/UPGRADING_TO_2.md +153 -38
  30. data/docs/WATCH_DAEMON.md +97 -14
  31. data/exe/woods-console-mcp +2 -2
  32. data/lib/generators/woods/templates/woods.rb.tt +2 -1
  33. data/lib/tasks/woods.rake +23 -7
  34. data/lib/tasks/woods_checks.rake +2 -2
  35. data/lib/woods/agent_configuration/cli.rb +1 -1
  36. data/lib/woods/agent_configuration/layout.rb +16 -2
  37. data/lib/woods/agent_configuration/plan.rb +13 -3
  38. data/lib/woods/agent_configuration/planner_validation.rb +4 -2
  39. data/lib/woods/agent_configuration/preflight.rb +5 -3
  40. data/lib/woods/builder.rb +17 -57
  41. data/lib/woods/cache/cache_middleware.rb +56 -30
  42. data/lib/woods/chunking/contributor_chunks.rb +119 -0
  43. data/lib/woods/chunking/semantic_chunker.rb +44 -21
  44. data/lib/woods/console/connection_manager.rb +56 -3
  45. data/lib/woods/console/embedded_executor.rb +30 -5
  46. data/lib/woods/console/rack_middleware.rb +29 -1
  47. data/lib/woods/dependency_graph.rb +34 -10
  48. data/lib/woods/embedding/fake.rb +12 -0
  49. data/lib/woods/embedding/indexer.rb +195 -98
  50. data/lib/woods/embedding/input_budget.rb +67 -0
  51. data/lib/woods/embedding/openai.rb +70 -20
  52. data/lib/woods/embedding/provider.rb +37 -25
  53. data/lib/woods/embedding/text_preparer.rb +76 -32
  54. data/lib/woods/embedding/token_counter.rb +18 -81
  55. data/lib/woods/embedding/vector_configuration.rb +48 -0
  56. data/lib/woods/extraction_identities.rb +175 -0
  57. data/lib/woods/extractor.rb +304 -107
  58. data/lib/woods/extractors/action_cable_extractor.rb +8 -3
  59. data/lib/woods/extractors/assigned_value_discovery.rb +74 -0
  60. data/lib/woods/extractors/class_declarations.rb +121 -0
  61. data/lib/woods/extractors/configuration_extractor.rb +11 -3
  62. data/lib/woods/extractors/declaration_ancestry.rb +92 -0
  63. data/lib/woods/extractors/event_extractor.rb +8 -0
  64. data/lib/woods/extractors/graphql_extractor.rb +134 -77
  65. data/lib/woods/extractors/job_extractor.rb +5 -1
  66. data/lib/woods/extractors/lib_extractor.rb +132 -15
  67. data/lib/woods/extractors/mailer_extractor.rb +3 -5
  68. data/lib/woods/extractors/manager_extractor.rb +7 -21
  69. data/lib/woods/extractors/migration_declaration.rb +87 -0
  70. data/lib/woods/extractors/migration_extractor.rb +5 -39
  71. data/lib/woods/extractors/phlex_extractor.rb +6 -2
  72. data/lib/woods/extractors/policy_extractor.rb +9 -5
  73. data/lib/woods/extractors/poro_extractor.rb +112 -53
  74. data/lib/woods/extractors/pundit_extractor.rb +11 -6
  75. data/lib/woods/extractors/scheduled_job_extractor.rb +45 -4
  76. data/lib/woods/extractors/serializer_extractor.rb +34 -22
  77. data/lib/woods/extractors/shared_utility_methods.rb +18 -1
  78. data/lib/woods/extractors/source_nesting.rb +142 -106
  79. data/lib/woods/extractors/standalone_module_discovery.rb +123 -0
  80. data/lib/woods/extractors/state_machine_extractor.rb +46 -40
  81. data/lib/woods/extractors/view_component_extractor.rb +9 -7
  82. data/lib/woods/flow_assembler.rb +4 -1
  83. data/lib/woods/generation.rb +25 -0
  84. data/lib/woods/hooks/context_hint.rb +7 -2
  85. data/lib/woods/mcp/bootstrapper.rb +33 -7
  86. data/lib/woods/mcp/config_resolver.rb +26 -7
  87. data/lib/woods/mcp/index_reader.rb +125 -24
  88. data/lib/woods/mcp/index_reader_pinning.rb +16 -0
  89. data/lib/woods/mcp/renderers/markdown_renderer.rb +7 -1
  90. data/lib/woods/mcp/renderers/plain_renderer.rb +3 -1
  91. data/lib/woods/mcp/search_results.rb +7 -1
  92. data/lib/woods/mcp/server.rb +24 -4
  93. data/lib/woods/module_reconciliation.rb +151 -0
  94. data/lib/woods/path_dispatcher.rb +7 -2
  95. data/lib/woods/rake_helpers.rb +43 -11
  96. data/lib/woods/release.rb +1 -1
  97. data/lib/woods/resilience/index_validator.rb +8 -3
  98. data/lib/woods/resilience/retryable_provider.rb +18 -1
  99. data/lib/woods/resolved_config.rb +68 -8
  100. data/lib/woods/retrieval/context_assembler.rb +3 -3
  101. data/lib/woods/retrieval/lexical_assembler.rb +3 -2
  102. data/lib/woods/retrieval/scope.rb +18 -2
  103. data/lib/woods/retrieval/source_evidence.rb +14 -2
  104. data/lib/woods/source_contributor_validation.rb +78 -0
  105. data/lib/woods/source_contributors.rb +116 -0
  106. data/lib/woods/source_inputs/handoff.rb +37 -0
  107. data/lib/woods/source_inputs/launcher.rb +53 -13
  108. data/lib/woods/source_inputs/manifest.rb +84 -3
  109. data/lib/woods/source_inputs/private_key.rb +44 -12
  110. data/lib/woods/source_inputs/scanner.rb +98 -27
  111. data/lib/woods/source_inputs/scopes.rb +1 -1
  112. data/lib/woods/source_inputs/session.rb +147 -15
  113. data/lib/woods/source_inputs/stable_reader.rb +127 -0
  114. data/lib/woods/source_inputs/status.rb +40 -8
  115. data/lib/woods/source_inputs/verifier.rb +28 -5
  116. data/lib/woods/source_path_encoding.rb +33 -0
  117. data/lib/woods/source_references/cache.rb +284 -0
  118. data/lib/woods/source_references/collector.rb +120 -0
  119. data/lib/woods/source_references/extraction.rb +185 -0
  120. data/lib/woods/source_references/inputs.rb +134 -0
  121. data/lib/woods/source_references/parser_adapter.rb +134 -0
  122. data/lib/woods/source_references/pass.rb +152 -0
  123. data/lib/woods/source_references/prism_adapter.rb +116 -0
  124. data/lib/woods/source_references/registry.rb +178 -0
  125. data/lib/woods/source_references/runtime_lookup.rb +127 -0
  126. data/lib/woods/source_references/value_class.rb +82 -0
  127. data/lib/woods/storage/metadata_store.rb +4 -1
  128. data/lib/woods/storage/qdrant.rb +2 -2
  129. data/lib/woods/unblocked/client.rb +12 -7
  130. data/lib/woods/unblocked/document_builder.rb +4 -1
  131. data/lib/woods/unblocked/exporter.rb +127 -37
  132. data/lib/woods/unblocked/sync_manifest.rb +137 -21
  133. data/lib/woods/unblocked/uri_migration.rb +105 -0
  134. data/lib/woods/util/host_guard.rb +3 -2
  135. data/lib/woods/version.rb +1 -1
  136. data/lib/woods/watch/catch_up.rb +138 -0
  137. data/lib/woods/watch/claim_lease.rb +150 -0
  138. data/lib/woods/watch/cli.rb +26 -2
  139. data/lib/woods/watch/daemon.rb +80 -59
  140. data/lib/woods/watch/installation/options.rb +1 -1
  141. data/lib/woods/watch/installation/receipt.rb +6 -1
  142. data/lib/woods/watch/managed_child.rb +1 -1
  143. data/lib/woods/watch/supervisor.rb +1 -1
  144. data/lib/woods/watch/tree_scan.rb +14 -2
  145. data/plugin/.claude-plugin/plugin.json +1 -1
  146. data/plugin/hooks/adapters/normalize.rb +3 -2
  147. data/plugin/hooks/woods-input-rules.sh +4 -0
  148. data/plugin/hooks/woods-refresh.sh +15 -7
  149. data/plugin/hooks/woods-session-start.sh +60 -3
  150. data/plugin/skills/woods-diagnose/SKILL.md +334 -11
  151. data/plugin/skills/woods-investigate/SKILL.md +11 -0
  152. data/plugin/skills/woods-mcp-config/SKILL.md +79 -8
  153. data/plugin/skills/woods-setup/SKILL.md +53 -6
  154. metadata +32 -5
data/lib/tasks/woods.rake CHANGED
@@ -58,6 +58,7 @@ namespace :woods do
58
58
  desc 'Incremental extraction based on git changes'
59
59
  task incremental: :environment do
60
60
  require 'woods/extractor'
61
+ require 'woods/input_rules'
61
62
 
62
63
  output_dir = ENV.fetch('WOODS_OUTPUT', Woods.configuration.output_dir)
63
64
 
@@ -71,8 +72,9 @@ namespace :woods do
71
72
  # PathDispatcher alongside the dispatch itself — a second hand-maintained
72
73
  # pattern list here would drift, and a path this filter drops never
73
74
  # reaches the index however good the dispatch behind it is (#164).
74
- dispatcher = Woods::PathDispatcher.new
75
- changed_files = changed_files.reject(&:empty?).select { |f| dispatcher.relevant?(f) }
75
+ input_rules = Woods::InputRules.new
76
+ changed_files = changed_files.reject(&:empty?).reject { |path| input_rules.action(path) == :ignore }
77
+ full = changed_files.any? { |path| input_rules.action(path) == :full }
76
78
 
77
79
  if changed_files.empty?
78
80
  puts 'No relevant files changed. Skipping extraction.'
@@ -91,13 +93,17 @@ namespace :woods do
91
93
  puts 'Warning: a watch daemon is alive but degraded — extracting anyway rather than assuming coverage.'
92
94
  end
93
95
 
94
- puts "Incremental extraction for #{changed_files.size} changed files..."
96
+ puts "#{full ? 'Full' : 'Incremental'} extraction for #{changed_files.size} changed files..."
95
97
  changed_files.each { |f| puts " - #{f}" }
96
98
  puts
97
99
 
98
100
  extractor = Woods::Extractor.new(output_dir: output_dir)
99
101
  affected = Woods::RakeHelpers.woods_with_extraction_lock(output_dir) do
100
- extracted = extractor.extract_changed(changed_files)
102
+ extracted = if full
103
+ extractor.extract_all.values.flatten(1).map(&:identifier).uniq
104
+ else
105
+ extractor.extract_changed(changed_files)
106
+ end
101
107
  extractor.raise_on_publication_failure!
102
108
  extracted
103
109
  end
@@ -168,7 +174,7 @@ namespace :woods do
168
174
  # FS events do not propagate and the daemon would sit silent. Nothing
169
175
  # exposed it before, which made the advice unfollowable.
170
176
  force_polling: ENV['WOODS_WATCH_POLL'] == '1', # container autodetect also applies; see Watcher.containerized?
171
- idle_timeout: ENV.fetch('WOODS_WATCH_IDLE_TIMEOUT', nil) && Float(ENV.fetch('WOODS_WATCH_IDLE_TIMEOUT')),
177
+ idle_timeout: Woods::RakeHelpers.woods_watch_idle_timeout,
172
178
  catch_up: ENV['WOODS_WATCH_CATCH_UP'] != '0',
173
179
  boot_snapshot: boot_snapshot,
174
180
  lifecycle: managed_child,
@@ -724,6 +730,8 @@ namespace :woods do
724
730
  env_flag = ->(name) { %w[1 true yes].include?(ENV.fetch(name, '').strip.downcase) }
725
731
  force_full = env_flag.call('UNBLOCKED_FORCE_FULL_SYNC')
726
732
  force_purge = env_flag.call('UNBLOCKED_FORCE_PURGE')
733
+ dry_run = env_flag.call('UNBLOCKED_DRY_RUN')
734
+ migrate_from_ref = ENV.fetch('UNBLOCKED_MIGRATE_FROM_REF', nil)
727
735
 
728
736
  puts 'Syncing extraction data to Unblocked...'
729
737
  puts " Output dir: #{output_dir}"
@@ -735,12 +743,19 @@ namespace :woods do
735
743
  exporter = Woods::Unblocked::Exporter.new(
736
744
  index_dir: output_dir,
737
745
  force_full: force_full,
738
- force_purge: force_purge
746
+ force_purge: force_purge,
747
+ dry_run: dry_run,
748
+ migrate_from_ref: migrate_from_ref
739
749
  )
740
750
  stats = exporter.sync_all
741
751
 
742
752
  puts
743
- puts 'Sync complete!'
753
+ if stats[:dry_run]
754
+ puts 'Sync preview (no remote requests or local writes):'
755
+ puts JSON.pretty_generate(stats)
756
+ next
757
+ end
758
+ puts(stats[:complete] == false ? 'Sync incomplete; re-run after resolving the reported condition.' : 'Sync complete!')
744
759
  puts " Documents synced: #{stats[:synced]}"
745
760
  puts " Documents skipped: #{stats[:skipped]}"
746
761
  puts " Documents deleted: #{stats[:deleted]}"
@@ -765,6 +780,7 @@ namespace :woods do
765
780
  exit 1
766
781
  end
767
782
  end
783
+ exit 1 if stats[:complete] == false && stats[:errors].empty?
768
784
  end
769
785
 
770
786
  desc 'Relay findings to Unblocked (alias for unblocked_sync)'
@@ -11,6 +11,7 @@
11
11
  # WOODS_CHECK_STRICT=1 ... # exit 1 on findings (still heuristic; see below)
12
12
 
13
13
  require 'json'
14
+ require 'woods/rake_helpers'
14
15
 
15
16
  namespace :woods do
16
17
  namespace :check do
@@ -62,8 +63,7 @@ module Woods
62
63
  # @return [String]
63
64
  def check_index_dir
64
65
  ENV.fetch('WOODS_OUTPUT') do
65
- root = respond_to?(:woods_task_root, true) ? woods_task_root : Rake.application.original_dir
66
- File.join(root, 'tmp/woods')
66
+ File.join(Woods::RakeHelpers.woods_task_root, 'tmp/woods')
67
67
  end
68
68
  end
69
69
 
@@ -86,7 +86,7 @@ module Woods
86
86
  protected_paths = layout.allowed_paths + layout.runtime_paths
87
87
  raise Conflict, 'Plan output must differ from every managed/runtime target' if protected_paths.include?(target)
88
88
 
89
- Document.validate_path!(target)
89
+ Plan.validate_path!(target)
90
90
  content = "#{JSON.pretty_generate(plan.data)}\n"
91
91
  raise Conflict, 'Plan exceeds the supported size' if content.bytesize > Document::MAX_BYTES
92
92
 
@@ -64,8 +64,22 @@ module Woods
64
64
  end
65
65
 
66
66
  def identity
67
- { 'client' => 'claude', 'scope' => scope, 'root' => root, 'config_dir' => config_dir,
68
- 'config_path' => config_path, 'receipt_path' => receipt_path }
67
+ { 'client' => 'claude', 'scope' => scope, 'root' => root,
68
+ 'config_path' => config_path, 'receipt_path' => receipt_path }.tap do |value|
69
+ value['config_dir'] = config_dir if scope == 'user'
70
+ end
71
+ end
72
+
73
+ # Older project receipts/plans included an unused user configuration path.
74
+ # Only that field is irrelevant; every actual target and scope stays exact.
75
+ def self.validate_identity!(record, expected, message:)
76
+ raise Conflict, "#{message} (identity field: layout)" unless record.is_a?(Hash)
77
+
78
+ record = record.except('config_dir') if expected['client'] == 'claude' && expected['scope'] == 'project'
79
+ return if record == expected
80
+
81
+ fields = (record.keys | expected.keys).reject { |key| record.slice(key) == expected.slice(key) }
82
+ raise Conflict, "#{message} (identity fields: #{fields.join(', ')})"
69
83
  end
70
84
  end
71
85
  end
@@ -3,6 +3,7 @@
3
3
  require 'base64'
4
4
  require 'json'
5
5
  require_relative 'document'
6
+ require_relative 'layout'
6
7
 
7
8
  module Woods
8
9
  module AgentConfiguration
@@ -19,11 +20,19 @@ module Woods
19
20
  end
20
21
 
21
22
  def self.load(path)
23
+ validate_path!(path)
22
24
  object = allocate
23
25
  object.instance_variable_set(:@data, Document.new(path).json)
24
26
  object
25
27
  end
26
28
 
29
+ def self.validate_path!(path)
30
+ Document.validate_path!(path)
31
+ rescue Conflict => e
32
+ raise Conflict, "#{e.message}. Use a regular plan file under a real parent directory; " \
33
+ 'for temporary plans, create a private directory and resolve it with File.realpath.'
34
+ end
35
+
27
36
  def add(document, content, description:)
28
37
  return if content == document.content
29
38
 
@@ -67,12 +76,13 @@ module Woods
67
76
  private
68
77
 
69
78
  def validate_format!(layout)
70
- valid = data.is_a?(Hash) && data['schema_version'] == SCHEMA_VERSION && data['layout'] == layout.identity &&
79
+ valid = data.is_a?(Hash) && data['schema_version'] == SCHEMA_VERSION &&
71
80
  %w[setup update remove].include?(data['operation']) && data['changes'].is_a?(Array) &&
72
81
  data['changes'].all?(Hash)
73
- return if valid
82
+ message = 'Plan format or selected application/client/scope differs; create a fresh preview'
83
+ raise Conflict, message unless valid
74
84
 
75
- raise Conflict, 'Plan format or selected application/client/scope differs; create a fresh preview'
85
+ Layout.validate_identity!(data['layout'], layout.identity, message: message)
76
86
  end
77
87
 
78
88
  def validate_paths!(layout)
@@ -19,11 +19,13 @@ module Woods
19
19
  end
20
20
 
21
21
  def validate_receipt_identity!
22
- valid = @previous['schema_version'] == 1 && @previous['layout'] == @layout.identity &&
22
+ valid = @previous['schema_version'] == 1 &&
23
23
  @previous['server_name'] == @name && @previous['sections'].is_a?(Hash) &&
24
24
  @previous['created_files'].is_a?(Array)
25
- raise Conflict, 'Installation receipt does not match this application/client/scope/server' unless valid
25
+ message = 'Installation receipt does not match this application/client/scope/server'
26
+ raise Conflict, message unless valid
26
27
 
28
+ Layout.validate_identity!(@previous['layout'], @layout.identity, message: message)
27
29
  validate_receipt_content!
28
30
  end
29
31
 
@@ -3,8 +3,8 @@
3
3
  require 'json'
4
4
  require 'open3'
5
5
  require 'timeout'
6
- require 'bundler'
7
6
  require_relative 'error'
7
+ require_relative '../watch/child_environment'
8
8
 
9
9
  module Woods
10
10
  module AgentConfiguration
@@ -32,8 +32,9 @@ module Woods
32
32
  end
33
33
  RUBY
34
34
 
35
- def initialize(timeout: 30)
35
+ def initialize(timeout: 30, environment: ENV)
36
36
  @timeout = timeout
37
+ @environment = environment
37
38
  end
38
39
 
39
40
  def call(launcher)
@@ -54,7 +55,8 @@ module Woods
54
55
 
55
56
  def run(launcher)
56
57
  command = launcher.probe_command(SCRIPT)
57
- environment = Bundler.unbundled_env.merge(launcher.probe_environment)
58
+ environment = Watch::ChildEnvironment.build(@environment.to_h, root: launcher.root)
59
+ .merge(launcher.probe_environment)
58
60
  Open3.popen3(environment, *command, chdir: launcher.root, unsetenv_others: true,
59
61
  pgroup: true) do |stdin, out, err, wait|
60
62
  stdin.close
data/lib/woods/builder.rb CHANGED
@@ -191,10 +191,9 @@ module Woods
191
191
  # their own implementation without patching the Builder. It flows
192
192
  # through {#build_resilient_embedding_provider} like the built-ins.
193
193
  #
194
- # Strips `embedding_options` keys that belong to the ResolvedConfig layer
195
- # (like `:dimension`) before splatting into the provider's constructor —
196
- # those keys are useful for the Snapshotter's schema header but
197
- # aren't part of the provider's API.
194
+ # The legacy :dimension option aliases the explicit :dimensions request.
195
+ # Stored observed widths use :expected_dimensions so restoring an index
196
+ # does not accidentally change the provider request.
198
197
  #
199
198
  # @return [Embedding::Provider::Interface] Embedding provider instance
200
199
  # @raise [ArgumentError] if the configured type is not recognized
@@ -244,8 +243,8 @@ module Woods
244
243
  end
245
244
 
246
245
  PROVIDER_OPTION_KEYS = {
247
- openai: %i[api_key model dimension dimensions],
248
- ollama: %i[model host num_ctx read_timeout dimension dimensions],
246
+ openai: %i[api_key model dimension dimensions expected_dimensions],
247
+ ollama: %i[model host num_ctx read_timeout dimension dimensions expected_dimensions],
249
248
  fake: %i[model dims dimension dimensions]
250
249
  }.freeze
251
250
  private_constant :PROVIDER_OPTION_KEYS
@@ -328,10 +327,9 @@ module Woods
328
327
  # Build the deterministic fake provider (#178).
329
328
  #
330
329
  # Dimension resolution: `embedding_options[:dims]` maps directly onto
331
- # the {Embedding::Provider::Fake} constructor; failing that, the
332
- # ResolvedConfig-level `embedding_options[:dimension]` key — normally
333
- # snapshot-only bookkeeping stripped by {#provider_kwargs} — is
334
- # honoured, so hosts that declare their dimension there (and the MCP
330
+ # the {Embedding::Provider::Fake} constructor; failing that, the legacy
331
+ # `embedding_options[:dimension]` alias is honoured, so hosts that
332
+ # declare their dimension there (and the MCP
335
333
  # boot path, which restores exactly that key from woods.json) get
336
334
  # vectors of the recorded dimension.
337
335
  #
@@ -341,54 +339,20 @@ module Woods
341
339
  end
342
340
  private :build_fake_provider
343
341
 
344
- # Build a {Embedding::TextPreparer} calibrated to a given provider.
345
- #
346
- # OpenAI embedders use tiktoken (cl100k_base) — 4.0 chars/token is a
347
- # good conservative average. Ollama BERT/WordPiece tokenizers
348
- # (nomic-embed-text, bge-*) run much hotter on dense Ruby/Rails
349
- # source — long CamelCase constants, docstrings, callback DSLs, and
350
- # heavy symbol use all sit below 2.0 chars/token in practice.
351
- # Empirically, a 16 KB chunk of `ActionMailer::Base` still blows the
352
- # 8192-token budget at 2.0 chars/token, so we budget at 1.5 to stay
353
- # clear of tokenizer surprises even on the densest Rails internals.
354
- #
355
- # `max_tokens` tracks the provider's actual input budget when it
356
- # reports one, falling back to the TextPreparer default otherwise.
357
- #
358
- # @param provider [Embedding::Provider::Interface]
359
- # @return [Embedding::TextPreparer]
342
+ # Build a preparer with the provider's explicit input-counting policy.
343
+ # Ratios remain estimates; known OpenAI models use a byte-BPE upper bound.
360
344
  def build_text_preparer(provider)
361
345
  chars_per_token = chars_per_token_for(provider)
362
346
  budget = safe_max_input_tokens(provider)
363
347
  max_tokens = budget || Embedding::TextPreparer::DEFAULT_MAX_TOKENS
364
348
 
365
- Embedding::TextPreparer.new(max_tokens: max_tokens, chars_per_token: chars_per_token)
349
+ input_budget = Embedding::InputBudget.for(provider, limit: max_tokens, chars_per_token: chars_per_token)
350
+ Embedding::TextPreparer.new(max_tokens: max_tokens, chars_per_token: chars_per_token, input_budget: input_budget)
366
351
  end
367
352
 
368
- # Build a {Chunking::SemanticChunker} sized to a given provider.
369
- #
370
- # `max_chars` is derived from the provider's input budget and the
371
- # matching chars-per-token ratio, minus the context-prefix
372
- # allowance the Indexer accounts for separately. Units that exceed
373
- # this ceiling get sliced so no single chunk can blow the provider's
374
- # input cap.
375
- #
376
- # For Ollama (and other BERT/WordPiece-backed models), char-based
377
- # estimation is unreliable — CamelCase, `::` separators, and symbol
378
- # literals tokenize much denser than chars/token averages suggest.
379
- # When the optional `tokenizers` gem is installed, pass a
380
- # {Embedding::TokenCounter} and `max_tokens` so the chunker can
381
- # verify every slice with the real tokenizer and re-split any piece
382
- # that still exceeds `num_ctx`. See docs/EMBEDDING_MODELS.md.
383
- #
384
- # Ollama v0.13.5+ stopped honouring `truncate: true` on `/api/embed`
385
- # (ollama/ollama#14186), so any chunk that exceeds `num_ctx` returns
386
- # a 400 rather than being silently truncated. Exact client-side
387
- # sizing is the only reliable path until the regression is fixed
388
- # upstream.
389
- #
390
- # @param provider [Embedding::Provider::Interface]
391
- # @return [Chunking::SemanticChunker]
353
+ # Initial semantic sizing is approximate. TextPreparer subsequently checks
354
+ # every complete prefix plus source against the provider policy and splits
355
+ # losslessly; no fixed prefix allowance establishes final admission.
392
356
  def build_chunker(provider)
393
357
  budget = safe_max_input_tokens(provider)
394
358
  max_chars = ((budget * chars_per_token_for(provider)).floor - CHUNKER_PREFIX_ALLOWANCE if budget)
@@ -424,12 +388,8 @@ module Woods
424
388
 
425
389
  private
426
390
 
427
- # Return a TokenCounter for providers that benefit from exact token
428
- # counting. OpenAI's tiktoken ratios are already stable at 4.0
429
- # chars/token on code, so it doesn't need this.
430
- #
431
- # @param provider [Embedding::Provider::Interface]
432
- # @return [Embedding::TokenCounter, nil]
391
+ # Ollama has no universal local tokenizer. This counter is an explicitly
392
+ # approximate sizing hint; truncate:false enforces the server's actual cap.
433
393
  def token_counter_for(provider)
434
394
  return unless unwrap_provider(provider).is_a?(Embedding::Provider::Ollama)
435
395
 
@@ -109,6 +109,9 @@ module Woods
109
109
  class CachedEmbeddingProvider
110
110
  include Embedding::Provider::Interface
111
111
 
112
+ # @return [Object] underlying provider, including any resilience wrapper
113
+ attr_reader :provider
114
+
112
115
  # @param provider [Embedding::Provider::Interface] The real embedding provider
113
116
  # @param cache_store [CacheStore] Cache backend instance
114
117
  # @param ttl [Integer] TTL for cached embeddings in seconds
@@ -118,6 +121,10 @@ module Woods
118
121
  @ttl = ttl
119
122
  @inflight = {}
120
123
  @inflight_mutex = Mutex.new
124
+ identity = @provider.cache_identity if @provider.respond_to?(:cache_identity)
125
+ # Unknown custom providers cannot prove cross-instance compatibility.
126
+ # Hash every identity so endpoint credentials never enter backend keys.
127
+ @embedding_identity = Digest::SHA256.hexdigest(JSON.generate(identity || SecureRandom.hex(16)))
121
128
  end
122
129
 
123
130
  # Embed a single text, returning a cached vector when available.
@@ -129,7 +136,7 @@ module Woods
129
136
  # @param text [String] Text to embed
130
137
  # @return [Array<Float>] Embedding vector
131
138
  def embed(text)
132
- cached = @cache_store.read(embedding_key(text))
139
+ cached = read_cached(text)
133
140
  return cached unless cached.nil?
134
141
 
135
142
  with_single_flight(text) { @provider.embed(text) }
@@ -148,13 +155,17 @@ module Woods
148
155
  # @return [Array<Array<Float>>] Embedding vectors (same order as input)
149
156
  def embed_batch(texts)
150
157
  results, misses, miss_indices = partition_cached(texts)
151
- return results if misses.empty?
158
+ if misses.empty?
159
+ validate_vectors!(results, texts.size)
160
+ return results
161
+ end
152
162
 
153
163
  to_fetch, to_fetch_positions, our_entries, awaiting = claim_inflight(misses)
154
164
 
155
165
  fetch_and_fulfill(to_fetch, to_fetch_positions, our_entries, results, miss_indices)
156
166
  await_others(awaiting, results, miss_indices)
157
167
 
168
+ validate_vectors!(results, texts.size)
158
169
  results
159
170
  end
160
171
 
@@ -172,6 +183,23 @@ module Woods
172
183
  @provider.model_name
173
184
  end
174
185
 
186
+ # Preserve pure configuration declarations through this wrapper.
187
+ def cache_identity
188
+ @provider.cache_identity if @provider.respond_to?(:cache_identity)
189
+ end
190
+
191
+ def input_budget
192
+ @provider.input_budget if @provider.respond_to?(:input_budget)
193
+ end
194
+
195
+ def configured_dimensions
196
+ @provider.configured_dimensions if @provider.respond_to?(:configured_dimensions)
197
+ end
198
+
199
+ def requested_dimensions
200
+ @provider.requested_dimensions if @provider.respond_to?(:requested_dimensions)
201
+ end
202
+
175
203
  # Delegate the per-provider input cap so Builder's chunker / text
176
204
  # preparer wiring keeps working when the cache wrapper is in front
177
205
  # of the provider. Without this, `respond_to?(:max_input_tokens)`
@@ -201,6 +229,7 @@ module Woods
201
229
 
202
230
  begin
203
231
  vector = yield
232
+ validate_vectors!([vector], 1)
204
233
  write_cache(text, vector)
205
234
  entry.fulfill(vector)
206
235
  vector
@@ -284,13 +313,7 @@ module Woods
284
313
 
285
314
  begin
286
315
  fresh_vectors = @provider.embed_batch(to_fetch)
287
- # Reject a malformed provider response up-front rather than silently
288
- # fulfilling waiters with `nil` (or masking a missing tail vector by
289
- # under-writing the cache).
290
- if fresh_vectors.size != to_fetch.size
291
- raise ArgumentError,
292
- "provider returned #{fresh_vectors.size} vectors for #{to_fetch.size} texts"
293
- end
316
+ validate_fresh_vectors!(fresh_vectors, to_fetch.size)
294
317
  rescue StandardError => e
295
318
  our_entries.each { |entry| entry.reject(e) }
296
319
  raise
@@ -351,7 +374,7 @@ module Woods
351
374
  miss_indices = []
352
375
 
353
376
  texts.each_with_index do |text, idx|
354
- cached = @cache_store.read(embedding_key(text))
377
+ cached = read_cached(text)
355
378
  if cached
356
379
  results[idx] = cached
357
380
  else
@@ -363,29 +386,32 @@ module Woods
363
386
  [results, misses, miss_indices]
364
387
  end
365
388
 
366
- # Build a cache key for an embedding text.
367
- #
368
- # The provider's model_name is folded into the key: a cached embedding
369
- # is only valid for the exact model that produced it. Without this, a
370
- # persistent shared backend (Redis/SolidCache) returns the previous
371
- # model's vector after a model switch or upgrade — different dimensions
372
- # error mid-batch, same dimensions silently corrupt similarity scores
373
- # (which the provider-vs-store dimension check can't detect, since it
374
- # compares declared widths, not cache contents).
375
- #
376
- # model_name (a plain attribute) is used rather than dimensions on
377
- # purpose: for every supported provider the model uniquely determines
378
- # the vector dimensionality, and `Provider#dimensions` can force a live
379
- # network probe (Ollama memoizes `embed('test').length`; OpenAI probes
380
- # for unknown models). Keying on dimensions made every cache lookup —
381
- # including hits — depend on the provider being reachable, defeating the
382
- # cache exactly when the backend is down. model_name distinguishes
383
- # models without any I/O.
384
- #
389
+ def read_cached(text)
390
+ vector = @cache_store.read(embedding_key(text))
391
+ validate_vectors!([vector], 1) unless vector.nil?
392
+ vector
393
+ end
394
+
395
+ # Validate the complete batch before writing or fulfilling any entry.
396
+ def validate_fresh_vectors!(vectors, count)
397
+ raise ArgumentError, "provider returned #{vectors.size} vectors for #{count} texts" if vectors.size != count
398
+
399
+ validate_vectors!(vectors, count)
400
+ end
401
+
402
+ def validate_vectors!(vectors, count)
403
+ width = @provider.configured_dimensions if @provider.respond_to?(:configured_dimensions)
404
+ Embedding::Provider::VectorValidation.validate!(
405
+ vectors, expected_count: count, provider: 'CachedEmbeddingProvider', expected_dimensions: width
406
+ )
407
+ end
408
+
409
+ # Configuration identity is captured without probing dimensions. Version
410
+ # the domain to prevent reuse of old model-only keys after an upgrade.
385
411
  # @param text [String]
386
412
  # @return [String]
387
413
  def embedding_key(text)
388
- Cache.cache_key(:embeddings, @provider.model_name.to_s, Digest::SHA256.hexdigest(text))
414
+ Cache.cache_key(:embeddings, 'configuration-v1', @embedding_identity, Digest::SHA256.hexdigest(text))
389
415
  end
390
416
  end
391
417
 
@@ -0,0 +1,119 @@
1
+ # frozen_string_literal: true
2
+
3
+ require_relative 'chunk'
4
+ require_relative '../source_contributors'
5
+
6
+ module Woods
7
+ module Chunking
8
+ # Exact physical attribution for contiguous slices of validated contributors.
9
+ # Generated composite headers and cross-file content never get a physical span.
10
+ module ContributorChunks
11
+ module_function
12
+
13
+ def chunks(unit)
14
+ SourceContributors.records(unit).each_with_index.map do |record, index|
15
+ first = record.fetch('published_start_byte')
16
+ last = record.fetch('published_end_byte')
17
+ Chunk.new(content: unit.source_code.byteslice(first...last), chunk_type: :"contributor_#{index}",
18
+ parent_identifier: unit.identifier, parent_type: unit.type,
19
+ metadata: location(unit, first, last))
20
+ end
21
+ end
22
+
23
+ # Preserve existing bounded chunks only if they cover every raw byte once,
24
+ # in contributor order. Caller-supplied ranges never authorize citations.
25
+ def ensure!(unit)
26
+ records = SourceContributors.records(unit)
27
+ return if records.empty?
28
+
29
+ unit.chunks = chunks(unit).map(&:to_h) unless complete_coverage?(unit, records)
30
+ unit.chunks.each do |chunk|
31
+ range = published_range(chunk)
32
+ chunk[:metadata] = (chunk[:metadata] || {}).merge(location(unit, range[:start_byte], range[:end_byte]))
33
+ end
34
+ end
35
+
36
+ def complete_coverage?(unit, records)
37
+ remaining = unit.chunks.dup
38
+ records.all? do |record|
39
+ cursor = record['published_start_byte']
40
+ while cursor < record['published_end_byte']
41
+ chunk = remaining.shift
42
+ return false unless chunk
43
+
44
+ span = published_range(chunk)
45
+ return false unless contiguous?(unit, chunk, span, cursor, record['published_end_byte'])
46
+
47
+ cursor = span[:end_byte]
48
+ end
49
+ true
50
+ end && remaining.empty?
51
+ end
52
+
53
+ def contiguous?(unit, chunk, span, cursor, last)
54
+ span[:start_byte] == cursor && span[:end_byte].is_a?(Integer) &&
55
+ span[:end_byte] > cursor && span[:end_byte] <= last &&
56
+ unit.source_code.byteslice(cursor...span[:end_byte]) == chunk[:content]
57
+ end
58
+
59
+ def published_range(chunk)
60
+ metadata = SourceContributors.field(chunk, :metadata) || {}
61
+ span = SourceContributors.field(metadata, :published_location)
62
+ span.is_a?(Hash) ? span.transform_keys(&:to_sym) : {}
63
+ end
64
+
65
+ def location(unit, first, last)
66
+ physical = SourceContributors.physical_span(unit, start_byte: first, end_byte: last)
67
+ return {} unless physical
68
+
69
+ record = SourceContributors.records(unit).find { |entry| entry['file_path'] == physical[:file_path] }
70
+ physical = physical.merge(start_byte: first - record['published_start_byte'],
71
+ end_byte: last - record['published_start_byte'])
72
+ { physical_location: physical, published_location: { start_byte: first, end_byte: last } }
73
+ end
74
+
75
+ # Offsets are relative to this chunk, including for a second splitting pass.
76
+ def slice_metadata(metadata, source, first, last)
77
+ metadata = symbol_keys(metadata)
78
+ physical = symbol_keys(metadata[:physical_location])
79
+ published = symbol_keys(metadata[:published_location])
80
+ clean = metadata.except(:physical_location, :published_location)
81
+ return clean unless mapped_slice?(physical, published, source, first, last)
82
+
83
+ clean.merge(physical_location: physical_slice(physical, source, first, last),
84
+ published_location: { start_byte: published[:start_byte] + first,
85
+ end_byte: published[:start_byte] + last })
86
+ end
87
+
88
+ def physical_slice(physical, source, first, last)
89
+ start_line = physical[:start_line] + source.byteslice(0...first).count("\n")
90
+ end_line = start_line + source.byteslice(first...last).delete_suffix("\n").count("\n")
91
+ physical.merge(start_byte: physical[:start_byte] + first, end_byte: physical[:start_byte] + last,
92
+ start_line: start_line, end_line: end_line)
93
+ end
94
+
95
+ def symbol_keys(value)
96
+ value.is_a?(Hash) ? value.transform_keys(&:to_sym) : {}
97
+ end
98
+
99
+ def mapped_slice?(physical, published, source, first, last)
100
+ integers = [physical[:start_byte], physical[:end_byte], physical[:start_line],
101
+ published[:start_byte], published[:end_byte]]
102
+ integers.all?(Integer) &&
103
+ [physical, published].all? { |span| span[:end_byte] - span[:start_byte] == source.bytesize } &&
104
+ first >= 0 && last > first && last <= source.bytesize
105
+ end
106
+
107
+ # Re-derive spans from the published source when hydrating a metadata dump.
108
+ def vector_metadata(unit, chunk_metadata)
109
+ return {} unless SourceContributors.multiple?(unit)
110
+
111
+ span = published_range(metadata: chunk_metadata)
112
+ facts = location(unit, span[:start_byte], span[:end_byte])
113
+ { source_paths: SourceContributors.paths(unit),
114
+ file_path: facts.dig(:physical_location, :file_path) }.merge(facts)
115
+ end
116
+ private_class_method :complete_coverage?, :contiguous?, :mapped_slice?, :physical_slice, :symbol_keys
117
+ end
118
+ end
119
+ end