woods 2.0.1 → 2.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (154) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +94 -7
  3. data/CONTRIBUTING.md +134 -19
  4. data/README.md +1 -1
  5. data/docs/AGENT_GUIDE.md +19 -0
  6. data/docs/AGENT_SETUP.md +22 -2
  7. data/docs/BACKEND_MATRIX.md +7 -0
  8. data/docs/CLIENT_HOOKS.md +6 -0
  9. data/docs/CONFIGURATION_REFERENCE.md +133 -25
  10. data/docs/CONSOLE_MCP_SETUP.md +82 -30
  11. data/docs/EMBEDDING_MODELS.md +16 -19
  12. data/docs/EXTRACTOR_REFERENCE.md +219 -21
  13. data/docs/FAQ.md +11 -25
  14. data/docs/GETTING_STARTED.md +7 -1
  15. data/docs/INCREMENTAL_EXTRACTION.md +261 -19
  16. data/docs/INDEX_LAYOUT.md +5 -0
  17. data/docs/INTERNALS.md +9 -0
  18. data/docs/MCP_HTTP_TRANSPORT.md +20 -15
  19. data/docs/MCP_SERVERS.md +87 -8
  20. data/docs/MCP_TOOL_COOKBOOK.md +13 -55
  21. data/docs/NOTION_INTEGRATION.md +7 -1
  22. data/docs/PUBLISHED_INDEX.md +6 -0
  23. data/docs/README.md +6 -1
  24. data/docs/RETRIEVAL_GUIDE.md +17 -0
  25. data/docs/SOURCE_FRESHNESS.md +157 -5
  26. data/docs/TOKEN_BENCHMARK.md +10 -18
  27. data/docs/TROUBLESHOOTING.md +70 -14
  28. data/docs/UNBLOCKED_INTEGRATION.md +60 -8
  29. data/docs/UPGRADING_TO_2.md +153 -38
  30. data/docs/WATCH_DAEMON.md +97 -14
  31. data/exe/woods-console-mcp +2 -2
  32. data/lib/generators/woods/templates/woods.rb.tt +2 -1
  33. data/lib/tasks/woods.rake +23 -7
  34. data/lib/tasks/woods_checks.rake +2 -2
  35. data/lib/woods/agent_configuration/cli.rb +1 -1
  36. data/lib/woods/agent_configuration/layout.rb +16 -2
  37. data/lib/woods/agent_configuration/plan.rb +13 -3
  38. data/lib/woods/agent_configuration/planner_validation.rb +4 -2
  39. data/lib/woods/agent_configuration/preflight.rb +5 -3
  40. data/lib/woods/builder.rb +17 -57
  41. data/lib/woods/cache/cache_middleware.rb +56 -30
  42. data/lib/woods/chunking/contributor_chunks.rb +119 -0
  43. data/lib/woods/chunking/semantic_chunker.rb +44 -21
  44. data/lib/woods/console/connection_manager.rb +56 -3
  45. data/lib/woods/console/embedded_executor.rb +30 -5
  46. data/lib/woods/console/rack_middleware.rb +29 -1
  47. data/lib/woods/dependency_graph.rb +34 -10
  48. data/lib/woods/embedding/fake.rb +12 -0
  49. data/lib/woods/embedding/indexer.rb +195 -98
  50. data/lib/woods/embedding/input_budget.rb +67 -0
  51. data/lib/woods/embedding/openai.rb +70 -20
  52. data/lib/woods/embedding/provider.rb +37 -25
  53. data/lib/woods/embedding/text_preparer.rb +76 -32
  54. data/lib/woods/embedding/token_counter.rb +18 -81
  55. data/lib/woods/embedding/vector_configuration.rb +48 -0
  56. data/lib/woods/extraction_identities.rb +175 -0
  57. data/lib/woods/extractor.rb +304 -107
  58. data/lib/woods/extractors/action_cable_extractor.rb +8 -3
  59. data/lib/woods/extractors/assigned_value_discovery.rb +74 -0
  60. data/lib/woods/extractors/class_declarations.rb +121 -0
  61. data/lib/woods/extractors/configuration_extractor.rb +11 -3
  62. data/lib/woods/extractors/declaration_ancestry.rb +92 -0
  63. data/lib/woods/extractors/event_extractor.rb +8 -0
  64. data/lib/woods/extractors/graphql_extractor.rb +134 -77
  65. data/lib/woods/extractors/job_extractor.rb +5 -1
  66. data/lib/woods/extractors/lib_extractor.rb +132 -15
  67. data/lib/woods/extractors/mailer_extractor.rb +3 -5
  68. data/lib/woods/extractors/manager_extractor.rb +7 -21
  69. data/lib/woods/extractors/migration_declaration.rb +87 -0
  70. data/lib/woods/extractors/migration_extractor.rb +5 -39
  71. data/lib/woods/extractors/phlex_extractor.rb +6 -2
  72. data/lib/woods/extractors/policy_extractor.rb +9 -5
  73. data/lib/woods/extractors/poro_extractor.rb +112 -53
  74. data/lib/woods/extractors/pundit_extractor.rb +11 -6
  75. data/lib/woods/extractors/scheduled_job_extractor.rb +45 -4
  76. data/lib/woods/extractors/serializer_extractor.rb +34 -22
  77. data/lib/woods/extractors/shared_utility_methods.rb +18 -1
  78. data/lib/woods/extractors/source_nesting.rb +142 -106
  79. data/lib/woods/extractors/standalone_module_discovery.rb +123 -0
  80. data/lib/woods/extractors/state_machine_extractor.rb +46 -40
  81. data/lib/woods/extractors/view_component_extractor.rb +9 -7
  82. data/lib/woods/flow_assembler.rb +4 -1
  83. data/lib/woods/generation.rb +25 -0
  84. data/lib/woods/hooks/context_hint.rb +7 -2
  85. data/lib/woods/mcp/bootstrapper.rb +33 -7
  86. data/lib/woods/mcp/config_resolver.rb +26 -7
  87. data/lib/woods/mcp/index_reader.rb +125 -24
  88. data/lib/woods/mcp/index_reader_pinning.rb +16 -0
  89. data/lib/woods/mcp/renderers/markdown_renderer.rb +7 -1
  90. data/lib/woods/mcp/renderers/plain_renderer.rb +3 -1
  91. data/lib/woods/mcp/search_results.rb +7 -1
  92. data/lib/woods/mcp/server.rb +24 -4
  93. data/lib/woods/module_reconciliation.rb +151 -0
  94. data/lib/woods/path_dispatcher.rb +7 -2
  95. data/lib/woods/rake_helpers.rb +43 -11
  96. data/lib/woods/release.rb +1 -1
  97. data/lib/woods/resilience/index_validator.rb +8 -3
  98. data/lib/woods/resilience/retryable_provider.rb +18 -1
  99. data/lib/woods/resolved_config.rb +68 -8
  100. data/lib/woods/retrieval/context_assembler.rb +3 -3
  101. data/lib/woods/retrieval/lexical_assembler.rb +3 -2
  102. data/lib/woods/retrieval/scope.rb +18 -2
  103. data/lib/woods/retrieval/source_evidence.rb +14 -2
  104. data/lib/woods/source_contributor_validation.rb +78 -0
  105. data/lib/woods/source_contributors.rb +116 -0
  106. data/lib/woods/source_inputs/handoff.rb +37 -0
  107. data/lib/woods/source_inputs/launcher.rb +53 -13
  108. data/lib/woods/source_inputs/manifest.rb +84 -3
  109. data/lib/woods/source_inputs/private_key.rb +44 -12
  110. data/lib/woods/source_inputs/scanner.rb +98 -27
  111. data/lib/woods/source_inputs/scopes.rb +1 -1
  112. data/lib/woods/source_inputs/session.rb +147 -15
  113. data/lib/woods/source_inputs/stable_reader.rb +127 -0
  114. data/lib/woods/source_inputs/status.rb +40 -8
  115. data/lib/woods/source_inputs/verifier.rb +28 -5
  116. data/lib/woods/source_path_encoding.rb +33 -0
  117. data/lib/woods/source_references/cache.rb +284 -0
  118. data/lib/woods/source_references/collector.rb +120 -0
  119. data/lib/woods/source_references/extraction.rb +185 -0
  120. data/lib/woods/source_references/inputs.rb +134 -0
  121. data/lib/woods/source_references/parser_adapter.rb +134 -0
  122. data/lib/woods/source_references/pass.rb +152 -0
  123. data/lib/woods/source_references/prism_adapter.rb +116 -0
  124. data/lib/woods/source_references/registry.rb +178 -0
  125. data/lib/woods/source_references/runtime_lookup.rb +127 -0
  126. data/lib/woods/source_references/value_class.rb +82 -0
  127. data/lib/woods/storage/metadata_store.rb +4 -1
  128. data/lib/woods/storage/qdrant.rb +2 -2
  129. data/lib/woods/unblocked/client.rb +12 -7
  130. data/lib/woods/unblocked/document_builder.rb +4 -1
  131. data/lib/woods/unblocked/exporter.rb +127 -37
  132. data/lib/woods/unblocked/sync_manifest.rb +137 -21
  133. data/lib/woods/unblocked/uri_migration.rb +105 -0
  134. data/lib/woods/util/host_guard.rb +3 -2
  135. data/lib/woods/version.rb +1 -1
  136. data/lib/woods/watch/catch_up.rb +138 -0
  137. data/lib/woods/watch/claim_lease.rb +150 -0
  138. data/lib/woods/watch/cli.rb +26 -2
  139. data/lib/woods/watch/daemon.rb +80 -59
  140. data/lib/woods/watch/installation/options.rb +1 -1
  141. data/lib/woods/watch/installation/receipt.rb +6 -1
  142. data/lib/woods/watch/managed_child.rb +1 -1
  143. data/lib/woods/watch/supervisor.rb +1 -1
  144. data/lib/woods/watch/tree_scan.rb +14 -2
  145. data/plugin/.claude-plugin/plugin.json +1 -1
  146. data/plugin/hooks/adapters/normalize.rb +3 -2
  147. data/plugin/hooks/woods-input-rules.sh +4 -0
  148. data/plugin/hooks/woods-refresh.sh +15 -7
  149. data/plugin/hooks/woods-session-start.sh +60 -3
  150. data/plugin/skills/woods-diagnose/SKILL.md +334 -11
  151. data/plugin/skills/woods-investigate/SKILL.md +11 -0
  152. data/plugin/skills/woods-mcp-config/SKILL.md +79 -8
  153. data/plugin/skills/woods-setup/SKILL.md +53 -6
  154. metadata +32 -5
@@ -3,6 +3,7 @@
3
3
  require 'net/http'
4
4
  require 'json'
5
5
  require_relative 'provider'
6
+ require_relative 'input_budget'
6
7
 
7
8
  module Woods
8
9
  module Embedding
@@ -19,18 +20,19 @@ module Woods
19
20
  class OpenAI # rubocop:disable Metrics/ClassLength
20
21
  include Interface
21
22
  include DiscardableClient
23
+ include VectorConfiguration
22
24
 
23
25
  ENDPOINT = URI('https://api.openai.com/v1/embeddings')
24
26
  DEFAULT_MODEL = 'text-embedding-3-small'
25
27
  DIMENSIONS = {
26
28
  'text-embedding-3-small' => 1536,
27
- 'text-embedding-3-large' => 3072
29
+ 'text-embedding-3-large' => 3072,
30
+ 'text-embedding-ada-002' => 1536
28
31
  }.freeze
29
32
  # Conservatively chunk below OpenAI's 8192-token input limit across
30
33
  # text-embedding-3-small / -3-large / ada-002. The chunker uses
31
- # 8191 as its ceiling — the actual chunk size lands well
32
- # below it once chars-per-token estimation and the prefix
33
- # allowance are factored in (see Builder#build_chunker).
34
+ # 8191 as its ceiling, checked against complete input bytes for known
35
+ # byte-BPE models (including every context-prefix character).
34
36
  MAX_INPUT_TOKENS = 8191
35
37
 
36
38
  # The API accepts at most 2048 inputs and 300,000 total tokens per
@@ -43,10 +45,12 @@ module Woods
43
45
  # @param api_key [String] OpenAI API key
44
46
  # @param model [String] OpenAI embedding model name (default: text-embedding-3-small)
45
47
  # @param dimensions [Integer, nil] Requested output size for text-embedding-3 models
46
- def initialize(api_key:, model: DEFAULT_MODEL, dimensions: nil)
48
+ # @param expected_dimensions [Integer, nil] Expected width; never sent to the API
49
+ def initialize(api_key:, model: DEFAULT_MODEL, dimensions: nil, expected_dimensions: nil)
47
50
  @api_key = api_key
48
51
  @model = model
49
- @dimensions = normalize_dimensions(dimensions)
52
+ configure_dimensions(dimensions: dimensions, expected_dimensions: expected_dimensions)
53
+ validate_model_dimensions!
50
54
  end
51
55
 
52
56
  # Embed a single text string.
@@ -58,9 +62,10 @@ module Woods
58
62
  def embed(text)
59
63
  raise ArgumentError, 'embed(text) requires a non-empty string' if text.nil? || text.to_s.strip.empty?
60
64
 
65
+ input_budget.validate!(text)
61
66
  response = post_request(request_body(text))
62
67
  vectors = Array(response['data']).map { |item| item['embedding'] }
63
- VectorValidation.validate!(vectors, expected_count: 1, provider: 'OpenAI')
68
+ validate_vectors!(vectors, expected_count: 1, provider: 'OpenAI')
64
69
  vectors.first
65
70
  end
66
71
 
@@ -73,16 +78,13 @@ module Woods
73
78
  # @raise [Woods::Error] if the API returns an error
74
79
  # @raise [ArgumentError] if the array is empty or any element is nil/empty
75
80
  def embed_batch(texts)
76
- raise ArgumentError, 'embed_batch(texts) requires a non-empty array' if texts.nil? || texts.empty?
77
- if texts.any? { |t| t.nil? || t.to_s.strip.empty? }
78
- raise ArgumentError, 'embed_batch(texts) rejects nil/empty entries (OpenAI returns 400)'
79
- end
81
+ validate_inputs!(texts)
80
82
 
81
83
  vectors = texts.each_slice(MAX_REQUEST_INPUTS).flat_map do |slice|
82
84
  response = post_request(request_body(slice))
83
85
  extract_validated_batch(response, slice.size)
84
86
  end
85
- VectorValidation.validate!(vectors, expected_count: texts.size, provider: 'OpenAI')
87
+ validate_vectors!(vectors, expected_count: texts.size, provider: 'OpenAI')
86
88
  vectors
87
89
  end
88
90
 
@@ -93,7 +95,24 @@ module Woods
93
95
  #
94
96
  # @return [Integer] number of dimensions
95
97
  def dimensions
96
- @dimensions || DIMENSIONS[@model] || embed('test').length
98
+ configured_dimensions || @observed_dimensions || embed('test').length
99
+ end
100
+
101
+ # @return [Integer, nil] expected width without a network probe
102
+ def configured_dimensions
103
+ super || DIMENSIONS[@model]
104
+ end
105
+
106
+ # API keys do not change embeddings and are deliberately excluded.
107
+ # @return [Array]
108
+ def cache_identity
109
+ [self.class.name, ENDPOINT.to_s, @model, requested_dimensions, configured_dimensions]
110
+ end
111
+
112
+ # Pure constructor settings; credentials are supplied by the reader's host.
113
+ def configuration_options
114
+ { model: @model, dimensions: requested_dimensions,
115
+ expected_dimensions: configured_dimensions || @observed_dimensions }
97
116
  end
98
117
 
99
118
  # Return the model name.
@@ -111,8 +130,25 @@ module Woods
111
130
  MAX_INPUT_TOKENS
112
131
  end
113
132
 
133
+ # Known cl100k byte-BPE models have a dependency-free UTF-8 byte bound.
134
+ # Unknown future/custom model names retain honest sizing estimates;
135
+ # the API remains authoritative and rejects over-limit input.
136
+ def input_budget
137
+ @input_budget ||= InputBudget.new(limit: max_input_tokens, model: @model,
138
+ method: DIMENSIONS.key?(@model) ? 'utf8_bytes_bound' : 'estimate')
139
+ end
140
+
114
141
  private
115
142
 
143
+ def validate_inputs!(texts)
144
+ raise ArgumentError, 'embed_batch(texts) requires a non-empty array' if texts.nil? || texts.empty?
145
+ if texts.any? { |t| t.nil? || t.to_s.strip.empty? }
146
+ raise ArgumentError, 'embed_batch(texts) rejects nil/empty entries (OpenAI returns 400)'
147
+ end
148
+
149
+ texts.each { |text| input_budget.validate!(text) }
150
+ end
151
+
116
152
  # Validate the batch response's shape/cardinality/indexes and
117
153
  # return vectors reordered to match the input order.
118
154
  #
@@ -121,7 +157,7 @@ module Woods
121
157
  # @return [Array<Array<Float>>]
122
158
  def extract_validated_batch(response, expected_count)
123
159
  data = Array(response['data'])
124
- VectorValidation.validate!(
160
+ validate_vectors!(
125
161
  data.map { |item| item['embedding'] },
126
162
  expected_count: expected_count,
127
163
  provider: 'OpenAI',
@@ -132,17 +168,31 @@ module Woods
132
168
 
133
169
  def request_body(input)
134
170
  { model: @model, input: input }.tap do |body|
135
- body[:dimensions] = @dimensions if @dimensions
171
+ body[:dimensions] = requested_dimensions if requested_dimensions
136
172
  end
137
173
  end
138
174
 
139
- def normalize_dimensions(value)
140
- return if value.nil?
175
+ def validate_model_dimensions!
176
+ native = DIMENSIONS[@model]
177
+ return unless native
178
+
179
+ width = configured_dimensions
180
+ raise ArgumentError, "#{@model} supports at most #{native} dimensions" if width > native
181
+
182
+ return validate_ada_dimensions!(width, native) if @model == 'text-embedding-ada-002'
183
+
184
+ return unless !requested_dimensions && width != native
185
+
186
+ raise ArgumentError, "expected dimensions #{width} do not match #{@model} output #{native}; " \
187
+ 'set dimensions explicitly to request a reduction'
188
+ end
141
189
 
142
- dimensions = Integer(value)
143
- raise ArgumentError, "dimensions must be positive, got #{value.inspect}" unless dimensions.positive?
190
+ def validate_ada_dimensions!(width, native)
191
+ raise ArgumentError, "#{@model} requires dimensions #{native}" unless width == native
144
192
 
145
- dimensions
193
+ # ada has a fixed width and does not accept the dimensions parameter,
194
+ # even when a caller explicitly declares the matching native width.
195
+ @requested_dimensions = nil
146
196
  end
147
197
 
148
198
  # Cap interpolated response bodies so misconfigured API errors
@@ -2,6 +2,8 @@
2
2
 
3
3
  require 'net/http'
4
4
  require 'json'
5
+ require_relative 'vector_configuration'
6
+ require_relative 'input_budget'
5
7
 
6
8
  module Woods
7
9
  # Standalone-load guard — keeps `require 'woods/embedding/provider'`
@@ -158,7 +160,7 @@ module Woods
158
160
  # check is skipped.
159
161
  # @raise [InvalidEmbeddingResponse] on any violation
160
162
  # @return [void]
161
- def validate!(vectors, expected_count:, provider:, indexes: nil)
163
+ def validate!(vectors, expected_count:, provider:, indexes: nil, expected_dimensions: nil)
162
164
  fail_with = lambda do |msg|
163
165
  raise InvalidEmbeddingResponse.new(msg, provider: provider, batch_size: expected_count)
164
166
  end
@@ -169,7 +171,7 @@ module Woods
169
171
 
170
172
  validate_indexes!(indexes, expected_count, fail_with) if indexes
171
173
 
172
- validate_vector_shapes!(vectors, fail_with)
174
+ validate_vector_shapes!(vectors, fail_with, expected_dimensions)
173
175
  end
174
176
 
175
177
  # @api private
@@ -187,8 +189,7 @@ module Woods
187
189
  private_class_method :validate_indexes!
188
190
 
189
191
  # @api private
190
- def validate_vector_shapes!(vectors, fail_with)
191
- dimension = nil
192
+ def validate_vector_shapes!(vectors, fail_with, dimension)
192
193
  vectors.each_with_index do |vector, i|
193
194
  check_vector_shape!(vector, i, fail_with)
194
195
  dimension ||= vector.size
@@ -220,9 +221,10 @@ module Woods
220
221
  # provider = Woods::Embedding::Provider::Ollama.new
221
222
  # vector = provider.embed("class User < ApplicationRecord; end")
222
223
  # vectors = provider.embed_batch(["text1", "text2"])
223
- class Ollama
224
+ class Ollama # rubocop:disable Metrics/ClassLength
224
225
  include Interface
225
226
  include DiscardableClient
227
+ include VectorConfiguration
226
228
 
227
229
  DEFAULT_MODEL = 'nomic-embed-text'
228
230
  DEFAULT_HOST = 'http://localhost:11434'
@@ -270,14 +272,15 @@ module Woods
270
272
  # unknown models. Set explicitly only if running a model with a
271
273
  # known-larger native context that isn't in the registry yet.
272
274
  # @param dimensions [Integer, nil] Requested output vector size.
275
+ # @param expected_dimensions [Integer, nil] Expected width; never sent to the API.
273
276
  # @param read_timeout [Integer] HTTP read timeout in seconds.
274
277
  # Bump this for slow / cold-start hosts or very large batches.
275
- def initialize(model: DEFAULT_MODEL, host: DEFAULT_HOST, num_ctx: nil,
276
- dimensions: nil, read_timeout: DEFAULT_READ_TIMEOUT)
278
+ def initialize(model: DEFAULT_MODEL, host: DEFAULT_HOST, num_ctx: nil, # rubocop:disable Metrics/ParameterLists
279
+ dimensions: nil, expected_dimensions: nil, read_timeout: DEFAULT_READ_TIMEOUT)
277
280
  @model = model
278
281
  @host = host
279
282
  @num_ctx = num_ctx || MODEL_CONTEXT_LENGTHS.fetch(model, FALLBACK_NUM_CTX)
280
- @dimensions = normalize_dimensions(dimensions)
283
+ configure_dimensions(dimensions: dimensions, expected_dimensions: expected_dimensions)
281
284
  @read_timeout = read_timeout
282
285
  @uri = URI("#{host}/api/embed")
283
286
  end
@@ -293,7 +296,7 @@ module Woods
293
296
 
294
297
  response = post_request(build_body(text))
295
298
  vectors = Array(response['embeddings'])
296
- VectorValidation.validate!(vectors, expected_count: 1, provider: 'Ollama')
299
+ validate_vectors!(vectors, expected_count: 1, provider: 'Ollama')
297
300
  vectors.first
298
301
  end
299
302
 
@@ -311,7 +314,7 @@ module Woods
311
314
 
312
315
  response = post_request(build_body(texts))
313
316
  vectors = Array(response['embeddings'])
314
- VectorValidation.validate!(vectors, expected_count: texts.size, provider: 'Ollama')
317
+ validate_vectors!(vectors, expected_count: texts.size, provider: 'Ollama')
315
318
  vectors
316
319
  end
317
320
 
@@ -321,7 +324,20 @@ module Woods
321
324
  #
322
325
  # @return [Integer] number of dimensions
323
326
  def dimensions
324
- @dimensions ||= embed('test').length
327
+ configured_dimensions || @observed_dimensions || embed('test').length
328
+ end
329
+
330
+ # Pure configuration identity; probes never change request semantics or keys.
331
+ # @return [Array]
332
+ def cache_identity
333
+ [self.class.name, @host, @model, @num_ctx, requested_dimensions, configured_dimensions, 'truncate:false']
334
+ end
335
+
336
+ # Pure constructor settings for ResolvedConfig's allowlisted serialization.
337
+ # The snapshot layer checks the endpoint before persisting it.
338
+ def configuration_options
339
+ { model: @model, host: @host, num_ctx: @num_ctx, read_timeout: @read_timeout,
340
+ dimensions: requested_dimensions, expected_dimensions: configured_dimensions || @observed_dimensions }
325
341
  end
326
342
 
327
343
  # Return the model name.
@@ -341,17 +357,14 @@ module Woods
341
357
  @num_ctx
342
358
  end
343
359
 
344
- private
345
-
346
- def normalize_dimensions(value)
347
- return if value.nil?
348
-
349
- dimensions = Integer(value)
350
- raise ArgumentError, "dimensions must be positive, got #{value.inspect}" unless dimensions.positive?
351
-
352
- dimensions
360
+ # A model-specific tokenizer is not available locally. This is a sizing
361
+ # estimate; truncate:false makes the server reject an actual overflow.
362
+ def input_budget
363
+ @input_budget ||= InputBudget.new(limit: max_input_tokens, model: @model)
353
364
  end
354
365
 
366
+ private
367
+
355
368
  # Cap interpolated response bodies so misconfigured Ollama responses
356
369
  # (e.g. proxied HTML error pages) don't unbounded-leak into logs or
357
370
  # re-raised error messages.
@@ -365,12 +378,11 @@ module Woods
365
378
  s.length > 500 ? "#{s[0, 500]}... [truncated]" : s
366
379
  end
367
380
 
368
- # Build the JSON body for an `/api/embed` call. Adds `options.num_ctx`
369
- # when configured — without it, Ollama silently truncates to 2048
370
- # tokens and returns 400 when the input exceeds that default.
381
+ # Request the configured context and explicitly refuse server truncation.
382
+ # Unknown model tokenization remains server-authoritative.
371
383
  def build_body(input)
372
- body = { model: @model, input: input }
373
- body[:dimensions] = @dimensions if @dimensions
384
+ body = { model: @model, input: input, truncate: false }
385
+ body[:dimensions] = requested_dimensions if requested_dimensions
374
386
  body[:options] = { num_ctx: @num_ctx } if @num_ctx
375
387
  body
376
388
  end
@@ -1,6 +1,8 @@
1
1
  # frozen_string_literal: true
2
2
 
3
3
  require_relative '../token_utils'
4
+ require_relative 'input_budget'
5
+ require_relative '../chunking/contributor_chunks'
4
6
 
5
7
  module Woods
6
8
  module Embedding
@@ -12,8 +14,8 @@ module Woods
12
14
  # file: ...
13
15
  # dependencies: dep1, dep2, ...
14
16
  #
15
- # Handles token limit enforcement by truncating text that exceeds the
16
- # embedding model's context window.
17
+ # Refuses oversized direct inputs; the indexing path splits complete inputs
18
+ # without truncation before submitting them to the provider.
17
19
  #
18
20
  # @example
19
21
  # preparer = Woods::Embedding::TextPreparer.new(max_tokens: 8192)
@@ -29,9 +31,10 @@ module Woods
29
31
 
30
32
  # @param max_tokens [Integer] maximum token budget for prepared text
31
33
  # @param chars_per_token [Float] tokenizer-calibrated char/token ratio
32
- def initialize(max_tokens: DEFAULT_MAX_TOKENS, chars_per_token: DEFAULT_CHARS_PER_TOKEN)
34
+ def initialize(max_tokens: DEFAULT_MAX_TOKENS, chars_per_token: DEFAULT_CHARS_PER_TOKEN, input_budget: nil)
33
35
  @max_tokens = max_tokens
34
36
  @chars_per_token = chars_per_token
37
+ @input_budget = input_budget || InputBudget.new(limit: max_tokens, chars_per_token: chars_per_token)
35
38
  end
36
39
 
37
40
  # @return [Float] configured chars-per-token ratio
@@ -43,15 +46,15 @@ module Woods
43
46
  # Prepare text for embedding from an ExtractedUnit.
44
47
  #
45
48
  # Builds a context prefix and appends the unit's source code (or first
46
- # chunk content for chunked units). Enforces token limits via truncation.
49
+ # chunk content for chunked units). Refuses inputs exceeding the configured limit.
47
50
  #
48
51
  # @param unit [Woods::ExtractedUnit] the unit to prepare
49
52
  # @return [String] context-prefixed text ready for embedding
50
- def prepare(unit)
51
- prefix = build_prefix(unit)
53
+ def prepare(unit, budget: @input_budget)
54
+ prefix = build_prefix(unit, unit.chunks.first)
52
55
  content = select_content(unit)
53
56
  text = "#{prefix}\n#{content}"
54
- enforce_token_limit(text)
57
+ budget.validate!(text)
55
58
  end
56
59
 
57
60
  # Prepare text for each chunk of an ExtractedUnit.
@@ -62,31 +65,89 @@ module Woods
62
65
  #
63
66
  # @param unit [Woods::ExtractedUnit] the unit to prepare
64
67
  # @return [Array<String>] array of context-prefixed texts
65
- def prepare_chunks(unit)
66
- return [prepare(unit)] unless unit.chunks&.any?
68
+ def prepare_chunks(unit, budget: @input_budget)
69
+ Chunking::ContributorChunks.ensure!(unit)
70
+ return [prepare(unit, budget: budget)] unless unit.chunks&.any?
67
71
 
68
- prefix = build_prefix(unit)
69
72
  unit.chunks.map do |chunk|
70
- text = "#{prefix}\n#{chunk[:content]}"
71
- enforce_token_limit(text)
73
+ text = "#{build_prefix(unit, chunk)}\n#{chunk[:content]}"
74
+ budget.validate!(text)
72
75
  end
73
76
  end
74
77
 
78
+ # Fit complete prefixed inputs without removing any source characters.
79
+ # Existing chunk attributes survive; embedding_slice records byte offsets
80
+ # relative to the original chunk for downstream physical-source attribution.
81
+ def prepare_for_embedding(unit, budget: @input_budget)
82
+ Chunking::ContributorChunks.ensure!(unit)
83
+ originals = unit.chunks.any? ? unit.chunks : [{ content: unit.source_code || '', chunk_type: :whole }]
84
+ fitted = originals.flat_map do |chunk|
85
+ prefix = "#{build_prefix(unit, chunk)}\n"
86
+ unless budget.fits?(prefix)
87
+ raise InputLimitError, "Embedding prefix exceeds input limit (#{budget.method}: #{budget.count(prefix)})"
88
+ end
89
+
90
+ fit_chunk(chunk, prefix, budget)
91
+ end
92
+ unit.chunks = fitted if unit.chunks.any? || fitted.size > 1
93
+ prepare_chunks(unit, budget: budget)
94
+ end
95
+
96
+ def preparation_identity
97
+ { 'class' => self.class.name, 'version' => 2, 'budget' => @input_budget.identity }
98
+ end
99
+
75
100
  private
76
101
 
102
+ def fit_chunk(chunk, prefix, budget)
103
+ content = chunk[:content].to_s
104
+ return [chunk] if budget.fits?(prefix + content)
105
+
106
+ offset = SourceContributors.field(chunk[:embedding_slice], :start_byte) || 0
107
+ initial_offset = offset
108
+ split_content(content, prefix, budget).map do |part|
109
+ first = offset
110
+ offset += part.bytesize
111
+ metadata = Chunking::ContributorChunks.slice_metadata(chunk[:metadata], content, first - initial_offset,
112
+ offset - initial_offset)
113
+ chunk.merge(content: part, metadata: metadata, embedding_slice: { start_byte: first, end_byte: offset })
114
+ end
115
+ end
116
+
117
+ def split_content(content, prefix, budget)
118
+ return [content] if budget.fits?(prefix + content)
119
+ if content.length <= 1
120
+ raise InputLimitError, 'Embedding input limit leaves no room for one source character after the prefix'
121
+ end
122
+
123
+ middle = content.length / 2
124
+ split_content(content[0...middle], prefix, budget) + split_content(content[middle..], prefix, budget)
125
+ end
126
+
77
127
  # Build the context prefix for a unit.
78
128
  #
79
129
  # @param unit [Woods::ExtractedUnit] the unit
80
130
  # @return [String] formatted prefix lines
81
- def build_prefix(unit)
131
+ def build_prefix(unit, chunk = nil)
82
132
  lines = []
83
133
  lines << "[#{unit.type}] #{unit.identifier}"
84
134
  lines << "namespace: #{unit.namespace}" if unit.namespace
85
- lines << "file: #{unit.file_path}" if unit.file_path
135
+ path = chunk&.dig(:metadata, :physical_location, :file_path)
136
+ if SourceContributors.multiple?(unit)
137
+ lines.concat(contributor_prefix(unit, path))
138
+ elsif unit.file_path
139
+ lines << "file: #{unit.file_path}"
140
+ end
86
141
  append_dependency_line(lines, unit.dependencies)
87
142
  lines.join("\n")
88
143
  end
89
144
 
145
+ def contributor_prefix(unit, path)
146
+ return ["file: #{path}"] if path
147
+
148
+ ["primary file: #{unit.file_path}", "contributing files: #{SourceContributors.paths(unit).join(', ')}"]
149
+ end
150
+
90
151
  # Append a formatted dependency line if dependencies exist.
91
152
  #
92
153
  # @param lines [Array<String>] lines to append to
@@ -100,7 +161,7 @@ module Woods
100
161
  # reads JSON and does not symbolize dependency keys, unlike chunks).
101
162
  # Read both forms or the whole "dependencies:" prefix silently
102
163
  # vanishes from every embedded document on the indexing path.
103
- dep_names = dependencies.filter_map { |d| d[:target] || d['target'] }.first(10)
164
+ dep_names = dependencies.filter_map { |d| d.is_a?(String) ? d : d[:target] || d['target'] }.first(10)
104
165
  lines << "dependencies: #{dep_names.join(', ')}" if dep_names.any?
105
166
  end
106
167
 
@@ -115,23 +176,6 @@ module Woods
115
176
  unit.source_code || ''
116
177
  end
117
178
  end
118
-
119
- # Truncate text to fit within the token budget.
120
- #
121
- # Uses the configured `chars_per_token` ratio to estimate both the
122
- # token count and the safe character cap. Truncation is a last
123
- # resort — by the time text reaches here the chunker should have
124
- # already split oversize units into pieces that fit.
125
- #
126
- # @param text [String] the text to truncate
127
- # @return [String] text within token limits
128
- def enforce_token_limit(text)
129
- estimated = (text.length / @chars_per_token).ceil
130
- return text if estimated <= @max_tokens
131
-
132
- max_chars = (@max_tokens * @chars_per_token).floor
133
- text[0...max_chars]
134
- end
135
179
  end
136
180
  end
137
181
  end
@@ -4,114 +4,51 @@ require 'set'
4
4
 
5
5
  module Woods
6
6
  module Embedding
7
- # Exact or estimated token counts for embedding inputs.
8
- #
9
- # When the optional `tokenizers` gem (ankane) is installed, loads the
10
- # `bert-base-uncased` WordPiece tokenizer that nomic-embed-text is
11
- # built on and returns exact token counts. Otherwise falls back to a
12
- # conservative chars/token ratio and warns once.
13
- #
14
- # Exact counting is strictly preferred for the Ollama path — Ollama
15
- # v0.13.5+ stopped honouring the `truncate: true` flag on
16
- # `/api/embed` (see ollama/ollama#14186), so chunks that exceed
17
- # `num_ctx` return a 400 instead of being truncated. Client-side
18
- # sizing is the only reliable option until the regression is fixed
19
- # upstream, and chars/token ratios vary too widely across Rails
20
- # internals to cover every case with a fixed number.
21
- #
22
- # @example
23
- # counter = Woods::Embedding::TokenCounter.new
24
- # counter.count("ActionController::Metal::ConditionalGet") # => 13
7
+ # Counts with an explicitly supplied local tokenizer, otherwise estimates.
8
+ # The caller owns the tokenizer/model match. No model is downloaded here.
25
9
  class TokenCounter
26
- # HuggingFace tokenizer id shared by every nomic-embed-text variant.
10
+ # Legacy constants/keyword remain accepted; tokenizer_id no longer loads
11
+ # a remote tokenizer or implies that BERT matches the configured model.
27
12
  BERT_MODEL = 'bert-base-uncased'
28
-
29
- # Conservative floor for when the tokenizer gem isn't installed.
30
- # Lower than any ratio we've observed failing in the testbed
31
- # against dense Rails source. Still approximate — install
32
- # `tokenizers` for exact counts.
33
13
  CONSERVATIVE_CHARS_PER_TOKEN = 1.2
34
14
 
35
- # @param chars_per_token [Float] fallback ratio when the tokenizer
36
- # is unavailable
37
- # @param tokenizer_id [String] HuggingFace model id passed to
38
- # `Tokenizers.from_pretrained`
39
- def initialize(chars_per_token: CONSERVATIVE_CHARS_PER_TOKEN, tokenizer_id: BERT_MODEL)
15
+ attr_reader :chars_per_token
16
+
17
+ def initialize(chars_per_token: CONSERVATIVE_CHARS_PER_TOKEN, tokenizer_id: nil, tokenizer: nil)
18
+ raise ArgumentError, 'chars_per_token must be positive' unless chars_per_token.positive?
19
+
40
20
  @chars_per_token = chars_per_token
21
+ @tokenizer = tokenizer
41
22
  @tokenizer_id = tokenizer_id
42
- @load_attempted = false
43
- @load_mutex = Mutex.new
44
23
  end
45
24
 
46
- # @return [Float] fallback chars-per-token ratio
47
- attr_reader :chars_per_token
48
-
49
- # Exact token count when the tokenizer is loaded, chars/token
50
- # estimate otherwise.
51
- #
52
- # @param text [String, nil]
53
- # @return [Integer]
54
25
  def count(text)
55
26
  return 0 if text.nil? || text.empty?
27
+ return @tokenizer.encode(text).ids.length if @tokenizer
56
28
 
57
- tok = tokenizer
58
- tok ? tok.encode(text).ids.length : estimate(text)
59
- end
60
-
61
- private
62
-
63
- def estimate(text)
29
+ warn_once('Embedding token counts are estimates: no model-matched local tokenizer was supplied. ' \
30
+ 'Ollama uses truncate:false and rejects inputs beyond its actual context window.')
64
31
  (text.length / @chars_per_token).ceil
65
32
  end
66
33
 
67
- # Lazy-load the tokenizer under a mutex so concurrent first-calls
68
- # don't each trigger a separate download. After the first attempt
69
- # (successful or not) we memoize the result and skip the load path.
70
- def tokenizer
71
- @load_mutex.synchronize do
72
- return @tokenizer if @load_attempted
73
-
74
- @load_attempted = true
75
- @tokenizer = try_load
76
- end
77
- end
78
-
79
- def try_load
80
- require 'tokenizers'
81
- Tokenizers.from_pretrained(@tokenizer_id)
82
- rescue LoadError
83
- warn_once(
84
- 'Exact token counting disabled: `tokenizers` gem not installed. ' \
85
- "Falling back to #{@chars_per_token} chars/token estimation. " \
86
- "Add `gem 'tokenizers', '~> 0.5'` to your Gemfile for exact sizing on Ollama."
87
- )
88
- nil
89
- rescue StandardError => e
90
- warn_once(
91
- "Could not load tokenizer #{@tokenizer_id.inspect} " \
92
- "(#{e.class}: #{e.message}). Falling back to chars/token estimate."
93
- )
94
- nil
34
+ def exact?
35
+ !@tokenizer.nil?
95
36
  end
96
37
 
97
- # Per-process dedup so multiple TokenCounter instances (one per
98
- # retriever build, plus one per chunker, plus tests) don't each
99
- # spam the same fallback warning. The mutex keeps the dedup set
100
- # consistent under the same concurrent-first-call pattern that
101
- # the per-instance load mutex protects against.
102
38
  @warned_messages = Set.new
103
39
  @warned_mutex = Mutex.new
104
40
 
105
41
  class << self
106
42
  attr_reader :warned_messages, :warned_mutex
107
43
 
108
- # Reset the per-process warning dedup. For tests only — production
109
- # callers should never need to clear it.
44
+ # Test seam for process-wide diagnostic deduplication.
110
45
  def reset_warned!
111
46
  @warned_mutex.synchronize { @warned_messages.clear }
112
47
  end
113
48
  end
114
49
 
50
+ private
51
+
115
52
  def warn_once(message)
116
53
  full = "[woods] #{message}"
117
54
  self.class.warned_mutex.synchronize do