woods 2.0.1 → 2.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +94 -7
- data/CONTRIBUTING.md +134 -19
- data/README.md +1 -1
- data/docs/AGENT_GUIDE.md +19 -0
- data/docs/AGENT_SETUP.md +22 -2
- data/docs/BACKEND_MATRIX.md +7 -0
- data/docs/CLIENT_HOOKS.md +6 -0
- data/docs/CONFIGURATION_REFERENCE.md +133 -25
- data/docs/CONSOLE_MCP_SETUP.md +82 -30
- data/docs/EMBEDDING_MODELS.md +16 -19
- data/docs/EXTRACTOR_REFERENCE.md +219 -21
- data/docs/FAQ.md +11 -25
- data/docs/GETTING_STARTED.md +7 -1
- data/docs/INCREMENTAL_EXTRACTION.md +261 -19
- data/docs/INDEX_LAYOUT.md +5 -0
- data/docs/INTERNALS.md +9 -0
- data/docs/MCP_HTTP_TRANSPORT.md +20 -15
- data/docs/MCP_SERVERS.md +87 -8
- data/docs/MCP_TOOL_COOKBOOK.md +13 -55
- data/docs/NOTION_INTEGRATION.md +7 -1
- data/docs/PUBLISHED_INDEX.md +6 -0
- data/docs/README.md +6 -1
- data/docs/RETRIEVAL_GUIDE.md +17 -0
- data/docs/SOURCE_FRESHNESS.md +157 -5
- data/docs/TOKEN_BENCHMARK.md +10 -18
- data/docs/TROUBLESHOOTING.md +70 -14
- data/docs/UNBLOCKED_INTEGRATION.md +60 -8
- data/docs/UPGRADING_TO_2.md +153 -38
- data/docs/WATCH_DAEMON.md +97 -14
- data/exe/woods-console-mcp +2 -2
- data/lib/generators/woods/templates/woods.rb.tt +2 -1
- data/lib/tasks/woods.rake +23 -7
- data/lib/tasks/woods_checks.rake +2 -2
- data/lib/woods/agent_configuration/cli.rb +1 -1
- data/lib/woods/agent_configuration/layout.rb +16 -2
- data/lib/woods/agent_configuration/plan.rb +13 -3
- data/lib/woods/agent_configuration/planner_validation.rb +4 -2
- data/lib/woods/agent_configuration/preflight.rb +5 -3
- data/lib/woods/builder.rb +17 -57
- data/lib/woods/cache/cache_middleware.rb +56 -30
- data/lib/woods/chunking/contributor_chunks.rb +119 -0
- data/lib/woods/chunking/semantic_chunker.rb +44 -21
- data/lib/woods/console/connection_manager.rb +56 -3
- data/lib/woods/console/embedded_executor.rb +30 -5
- data/lib/woods/console/rack_middleware.rb +29 -1
- data/lib/woods/dependency_graph.rb +34 -10
- data/lib/woods/embedding/fake.rb +12 -0
- data/lib/woods/embedding/indexer.rb +195 -98
- data/lib/woods/embedding/input_budget.rb +67 -0
- data/lib/woods/embedding/openai.rb +70 -20
- data/lib/woods/embedding/provider.rb +37 -25
- data/lib/woods/embedding/text_preparer.rb +76 -32
- data/lib/woods/embedding/token_counter.rb +18 -81
- data/lib/woods/embedding/vector_configuration.rb +48 -0
- data/lib/woods/extraction_identities.rb +175 -0
- data/lib/woods/extractor.rb +304 -107
- data/lib/woods/extractors/action_cable_extractor.rb +8 -3
- data/lib/woods/extractors/assigned_value_discovery.rb +74 -0
- data/lib/woods/extractors/class_declarations.rb +121 -0
- data/lib/woods/extractors/configuration_extractor.rb +11 -3
- data/lib/woods/extractors/declaration_ancestry.rb +92 -0
- data/lib/woods/extractors/event_extractor.rb +8 -0
- data/lib/woods/extractors/graphql_extractor.rb +134 -77
- data/lib/woods/extractors/job_extractor.rb +5 -1
- data/lib/woods/extractors/lib_extractor.rb +132 -15
- data/lib/woods/extractors/mailer_extractor.rb +3 -5
- data/lib/woods/extractors/manager_extractor.rb +7 -21
- data/lib/woods/extractors/migration_declaration.rb +87 -0
- data/lib/woods/extractors/migration_extractor.rb +5 -39
- data/lib/woods/extractors/phlex_extractor.rb +6 -2
- data/lib/woods/extractors/policy_extractor.rb +9 -5
- data/lib/woods/extractors/poro_extractor.rb +112 -53
- data/lib/woods/extractors/pundit_extractor.rb +11 -6
- data/lib/woods/extractors/scheduled_job_extractor.rb +45 -4
- data/lib/woods/extractors/serializer_extractor.rb +34 -22
- data/lib/woods/extractors/shared_utility_methods.rb +18 -1
- data/lib/woods/extractors/source_nesting.rb +142 -106
- data/lib/woods/extractors/standalone_module_discovery.rb +123 -0
- data/lib/woods/extractors/state_machine_extractor.rb +46 -40
- data/lib/woods/extractors/view_component_extractor.rb +9 -7
- data/lib/woods/flow_assembler.rb +4 -1
- data/lib/woods/generation.rb +25 -0
- data/lib/woods/hooks/context_hint.rb +7 -2
- data/lib/woods/mcp/bootstrapper.rb +33 -7
- data/lib/woods/mcp/config_resolver.rb +26 -7
- data/lib/woods/mcp/index_reader.rb +125 -24
- data/lib/woods/mcp/index_reader_pinning.rb +16 -0
- data/lib/woods/mcp/renderers/markdown_renderer.rb +7 -1
- data/lib/woods/mcp/renderers/plain_renderer.rb +3 -1
- data/lib/woods/mcp/search_results.rb +7 -1
- data/lib/woods/mcp/server.rb +24 -4
- data/lib/woods/module_reconciliation.rb +151 -0
- data/lib/woods/path_dispatcher.rb +7 -2
- data/lib/woods/rake_helpers.rb +43 -11
- data/lib/woods/release.rb +1 -1
- data/lib/woods/resilience/index_validator.rb +8 -3
- data/lib/woods/resilience/retryable_provider.rb +18 -1
- data/lib/woods/resolved_config.rb +68 -8
- data/lib/woods/retrieval/context_assembler.rb +3 -3
- data/lib/woods/retrieval/lexical_assembler.rb +3 -2
- data/lib/woods/retrieval/scope.rb +18 -2
- data/lib/woods/retrieval/source_evidence.rb +14 -2
- data/lib/woods/source_contributor_validation.rb +78 -0
- data/lib/woods/source_contributors.rb +116 -0
- data/lib/woods/source_inputs/handoff.rb +37 -0
- data/lib/woods/source_inputs/launcher.rb +53 -13
- data/lib/woods/source_inputs/manifest.rb +84 -3
- data/lib/woods/source_inputs/private_key.rb +44 -12
- data/lib/woods/source_inputs/scanner.rb +98 -27
- data/lib/woods/source_inputs/scopes.rb +1 -1
- data/lib/woods/source_inputs/session.rb +147 -15
- data/lib/woods/source_inputs/stable_reader.rb +127 -0
- data/lib/woods/source_inputs/status.rb +40 -8
- data/lib/woods/source_inputs/verifier.rb +28 -5
- data/lib/woods/source_path_encoding.rb +33 -0
- data/lib/woods/source_references/cache.rb +284 -0
- data/lib/woods/source_references/collector.rb +120 -0
- data/lib/woods/source_references/extraction.rb +185 -0
- data/lib/woods/source_references/inputs.rb +134 -0
- data/lib/woods/source_references/parser_adapter.rb +134 -0
- data/lib/woods/source_references/pass.rb +152 -0
- data/lib/woods/source_references/prism_adapter.rb +116 -0
- data/lib/woods/source_references/registry.rb +178 -0
- data/lib/woods/source_references/runtime_lookup.rb +127 -0
- data/lib/woods/source_references/value_class.rb +82 -0
- data/lib/woods/storage/metadata_store.rb +4 -1
- data/lib/woods/storage/qdrant.rb +2 -2
- data/lib/woods/unblocked/client.rb +12 -7
- data/lib/woods/unblocked/document_builder.rb +4 -1
- data/lib/woods/unblocked/exporter.rb +127 -37
- data/lib/woods/unblocked/sync_manifest.rb +137 -21
- data/lib/woods/unblocked/uri_migration.rb +105 -0
- data/lib/woods/util/host_guard.rb +3 -2
- data/lib/woods/version.rb +1 -1
- data/lib/woods/watch/catch_up.rb +138 -0
- data/lib/woods/watch/claim_lease.rb +150 -0
- data/lib/woods/watch/cli.rb +26 -2
- data/lib/woods/watch/daemon.rb +80 -59
- data/lib/woods/watch/installation/options.rb +1 -1
- data/lib/woods/watch/installation/receipt.rb +6 -1
- data/lib/woods/watch/managed_child.rb +1 -1
- data/lib/woods/watch/supervisor.rb +1 -1
- data/lib/woods/watch/tree_scan.rb +14 -2
- data/plugin/.claude-plugin/plugin.json +1 -1
- data/plugin/hooks/adapters/normalize.rb +3 -2
- data/plugin/hooks/woods-input-rules.sh +4 -0
- data/plugin/hooks/woods-refresh.sh +15 -7
- data/plugin/hooks/woods-session-start.sh +60 -3
- data/plugin/skills/woods-diagnose/SKILL.md +334 -11
- data/plugin/skills/woods-investigate/SKILL.md +11 -0
- data/plugin/skills/woods-mcp-config/SKILL.md +79 -8
- data/plugin/skills/woods-setup/SKILL.md +53 -6
- metadata +32 -5
|
@@ -3,6 +3,7 @@
|
|
|
3
3
|
require 'net/http'
|
|
4
4
|
require 'json'
|
|
5
5
|
require_relative 'provider'
|
|
6
|
+
require_relative 'input_budget'
|
|
6
7
|
|
|
7
8
|
module Woods
|
|
8
9
|
module Embedding
|
|
@@ -19,18 +20,19 @@ module Woods
|
|
|
19
20
|
class OpenAI # rubocop:disable Metrics/ClassLength
|
|
20
21
|
include Interface
|
|
21
22
|
include DiscardableClient
|
|
23
|
+
include VectorConfiguration
|
|
22
24
|
|
|
23
25
|
ENDPOINT = URI('https://api.openai.com/v1/embeddings')
|
|
24
26
|
DEFAULT_MODEL = 'text-embedding-3-small'
|
|
25
27
|
DIMENSIONS = {
|
|
26
28
|
'text-embedding-3-small' => 1536,
|
|
27
|
-
'text-embedding-3-large' => 3072
|
|
29
|
+
'text-embedding-3-large' => 3072,
|
|
30
|
+
'text-embedding-ada-002' => 1536
|
|
28
31
|
}.freeze
|
|
29
32
|
# Conservatively chunk below OpenAI's 8192-token input limit across
|
|
30
33
|
# text-embedding-3-small / -3-large / ada-002. The chunker uses
|
|
31
|
-
# 8191 as its ceiling
|
|
32
|
-
#
|
|
33
|
-
# allowance are factored in (see Builder#build_chunker).
|
|
34
|
+
# 8191 as its ceiling, checked against complete input bytes for known
|
|
35
|
+
# byte-BPE models (including every context-prefix character).
|
|
34
36
|
MAX_INPUT_TOKENS = 8191
|
|
35
37
|
|
|
36
38
|
# The API accepts at most 2048 inputs and 300,000 total tokens per
|
|
@@ -43,10 +45,12 @@ module Woods
|
|
|
43
45
|
# @param api_key [String] OpenAI API key
|
|
44
46
|
# @param model [String] OpenAI embedding model name (default: text-embedding-3-small)
|
|
45
47
|
# @param dimensions [Integer, nil] Requested output size for text-embedding-3 models
|
|
46
|
-
|
|
48
|
+
# @param expected_dimensions [Integer, nil] Expected width; never sent to the API
|
|
49
|
+
def initialize(api_key:, model: DEFAULT_MODEL, dimensions: nil, expected_dimensions: nil)
|
|
47
50
|
@api_key = api_key
|
|
48
51
|
@model = model
|
|
49
|
-
|
|
52
|
+
configure_dimensions(dimensions: dimensions, expected_dimensions: expected_dimensions)
|
|
53
|
+
validate_model_dimensions!
|
|
50
54
|
end
|
|
51
55
|
|
|
52
56
|
# Embed a single text string.
|
|
@@ -58,9 +62,10 @@ module Woods
|
|
|
58
62
|
def embed(text)
|
|
59
63
|
raise ArgumentError, 'embed(text) requires a non-empty string' if text.nil? || text.to_s.strip.empty?
|
|
60
64
|
|
|
65
|
+
input_budget.validate!(text)
|
|
61
66
|
response = post_request(request_body(text))
|
|
62
67
|
vectors = Array(response['data']).map { |item| item['embedding'] }
|
|
63
|
-
|
|
68
|
+
validate_vectors!(vectors, expected_count: 1, provider: 'OpenAI')
|
|
64
69
|
vectors.first
|
|
65
70
|
end
|
|
66
71
|
|
|
@@ -73,16 +78,13 @@ module Woods
|
|
|
73
78
|
# @raise [Woods::Error] if the API returns an error
|
|
74
79
|
# @raise [ArgumentError] if the array is empty or any element is nil/empty
|
|
75
80
|
def embed_batch(texts)
|
|
76
|
-
|
|
77
|
-
if texts.any? { |t| t.nil? || t.to_s.strip.empty? }
|
|
78
|
-
raise ArgumentError, 'embed_batch(texts) rejects nil/empty entries (OpenAI returns 400)'
|
|
79
|
-
end
|
|
81
|
+
validate_inputs!(texts)
|
|
80
82
|
|
|
81
83
|
vectors = texts.each_slice(MAX_REQUEST_INPUTS).flat_map do |slice|
|
|
82
84
|
response = post_request(request_body(slice))
|
|
83
85
|
extract_validated_batch(response, slice.size)
|
|
84
86
|
end
|
|
85
|
-
|
|
87
|
+
validate_vectors!(vectors, expected_count: texts.size, provider: 'OpenAI')
|
|
86
88
|
vectors
|
|
87
89
|
end
|
|
88
90
|
|
|
@@ -93,7 +95,24 @@ module Woods
|
|
|
93
95
|
#
|
|
94
96
|
# @return [Integer] number of dimensions
|
|
95
97
|
def dimensions
|
|
96
|
-
|
|
98
|
+
configured_dimensions || @observed_dimensions || embed('test').length
|
|
99
|
+
end
|
|
100
|
+
|
|
101
|
+
# @return [Integer, nil] expected width without a network probe
|
|
102
|
+
def configured_dimensions
|
|
103
|
+
super || DIMENSIONS[@model]
|
|
104
|
+
end
|
|
105
|
+
|
|
106
|
+
# API keys do not change embeddings and are deliberately excluded.
|
|
107
|
+
# @return [Array]
|
|
108
|
+
def cache_identity
|
|
109
|
+
[self.class.name, ENDPOINT.to_s, @model, requested_dimensions, configured_dimensions]
|
|
110
|
+
end
|
|
111
|
+
|
|
112
|
+
# Pure constructor settings; credentials are supplied by the reader's host.
|
|
113
|
+
def configuration_options
|
|
114
|
+
{ model: @model, dimensions: requested_dimensions,
|
|
115
|
+
expected_dimensions: configured_dimensions || @observed_dimensions }
|
|
97
116
|
end
|
|
98
117
|
|
|
99
118
|
# Return the model name.
|
|
@@ -111,8 +130,25 @@ module Woods
|
|
|
111
130
|
MAX_INPUT_TOKENS
|
|
112
131
|
end
|
|
113
132
|
|
|
133
|
+
# Known cl100k byte-BPE models have a dependency-free UTF-8 byte bound.
|
|
134
|
+
# Unknown future/custom model names retain honest sizing estimates;
|
|
135
|
+
# the API remains authoritative and rejects over-limit input.
|
|
136
|
+
def input_budget
|
|
137
|
+
@input_budget ||= InputBudget.new(limit: max_input_tokens, model: @model,
|
|
138
|
+
method: DIMENSIONS.key?(@model) ? 'utf8_bytes_bound' : 'estimate')
|
|
139
|
+
end
|
|
140
|
+
|
|
114
141
|
private
|
|
115
142
|
|
|
143
|
+
def validate_inputs!(texts)
|
|
144
|
+
raise ArgumentError, 'embed_batch(texts) requires a non-empty array' if texts.nil? || texts.empty?
|
|
145
|
+
if texts.any? { |t| t.nil? || t.to_s.strip.empty? }
|
|
146
|
+
raise ArgumentError, 'embed_batch(texts) rejects nil/empty entries (OpenAI returns 400)'
|
|
147
|
+
end
|
|
148
|
+
|
|
149
|
+
texts.each { |text| input_budget.validate!(text) }
|
|
150
|
+
end
|
|
151
|
+
|
|
116
152
|
# Validate the batch response's shape/cardinality/indexes and
|
|
117
153
|
# return vectors reordered to match the input order.
|
|
118
154
|
#
|
|
@@ -121,7 +157,7 @@ module Woods
|
|
|
121
157
|
# @return [Array<Array<Float>>]
|
|
122
158
|
def extract_validated_batch(response, expected_count)
|
|
123
159
|
data = Array(response['data'])
|
|
124
|
-
|
|
160
|
+
validate_vectors!(
|
|
125
161
|
data.map { |item| item['embedding'] },
|
|
126
162
|
expected_count: expected_count,
|
|
127
163
|
provider: 'OpenAI',
|
|
@@ -132,17 +168,31 @@ module Woods
|
|
|
132
168
|
|
|
133
169
|
def request_body(input)
|
|
134
170
|
{ model: @model, input: input }.tap do |body|
|
|
135
|
-
body[:dimensions] =
|
|
171
|
+
body[:dimensions] = requested_dimensions if requested_dimensions
|
|
136
172
|
end
|
|
137
173
|
end
|
|
138
174
|
|
|
139
|
-
def
|
|
140
|
-
|
|
175
|
+
def validate_model_dimensions!
|
|
176
|
+
native = DIMENSIONS[@model]
|
|
177
|
+
return unless native
|
|
178
|
+
|
|
179
|
+
width = configured_dimensions
|
|
180
|
+
raise ArgumentError, "#{@model} supports at most #{native} dimensions" if width > native
|
|
181
|
+
|
|
182
|
+
return validate_ada_dimensions!(width, native) if @model == 'text-embedding-ada-002'
|
|
183
|
+
|
|
184
|
+
return unless !requested_dimensions && width != native
|
|
185
|
+
|
|
186
|
+
raise ArgumentError, "expected dimensions #{width} do not match #{@model} output #{native}; " \
|
|
187
|
+
'set dimensions explicitly to request a reduction'
|
|
188
|
+
end
|
|
141
189
|
|
|
142
|
-
|
|
143
|
-
raise ArgumentError, "
|
|
190
|
+
def validate_ada_dimensions!(width, native)
|
|
191
|
+
raise ArgumentError, "#{@model} requires dimensions #{native}" unless width == native
|
|
144
192
|
|
|
145
|
-
dimensions
|
|
193
|
+
# ada has a fixed width and does not accept the dimensions parameter,
|
|
194
|
+
# even when a caller explicitly declares the matching native width.
|
|
195
|
+
@requested_dimensions = nil
|
|
146
196
|
end
|
|
147
197
|
|
|
148
198
|
# Cap interpolated response bodies so misconfigured API errors
|
|
@@ -2,6 +2,8 @@
|
|
|
2
2
|
|
|
3
3
|
require 'net/http'
|
|
4
4
|
require 'json'
|
|
5
|
+
require_relative 'vector_configuration'
|
|
6
|
+
require_relative 'input_budget'
|
|
5
7
|
|
|
6
8
|
module Woods
|
|
7
9
|
# Standalone-load guard — keeps `require 'woods/embedding/provider'`
|
|
@@ -158,7 +160,7 @@ module Woods
|
|
|
158
160
|
# check is skipped.
|
|
159
161
|
# @raise [InvalidEmbeddingResponse] on any violation
|
|
160
162
|
# @return [void]
|
|
161
|
-
def validate!(vectors, expected_count:, provider:, indexes: nil)
|
|
163
|
+
def validate!(vectors, expected_count:, provider:, indexes: nil, expected_dimensions: nil)
|
|
162
164
|
fail_with = lambda do |msg|
|
|
163
165
|
raise InvalidEmbeddingResponse.new(msg, provider: provider, batch_size: expected_count)
|
|
164
166
|
end
|
|
@@ -169,7 +171,7 @@ module Woods
|
|
|
169
171
|
|
|
170
172
|
validate_indexes!(indexes, expected_count, fail_with) if indexes
|
|
171
173
|
|
|
172
|
-
validate_vector_shapes!(vectors, fail_with)
|
|
174
|
+
validate_vector_shapes!(vectors, fail_with, expected_dimensions)
|
|
173
175
|
end
|
|
174
176
|
|
|
175
177
|
# @api private
|
|
@@ -187,8 +189,7 @@ module Woods
|
|
|
187
189
|
private_class_method :validate_indexes!
|
|
188
190
|
|
|
189
191
|
# @api private
|
|
190
|
-
def validate_vector_shapes!(vectors, fail_with)
|
|
191
|
-
dimension = nil
|
|
192
|
+
def validate_vector_shapes!(vectors, fail_with, dimension)
|
|
192
193
|
vectors.each_with_index do |vector, i|
|
|
193
194
|
check_vector_shape!(vector, i, fail_with)
|
|
194
195
|
dimension ||= vector.size
|
|
@@ -220,9 +221,10 @@ module Woods
|
|
|
220
221
|
# provider = Woods::Embedding::Provider::Ollama.new
|
|
221
222
|
# vector = provider.embed("class User < ApplicationRecord; end")
|
|
222
223
|
# vectors = provider.embed_batch(["text1", "text2"])
|
|
223
|
-
class Ollama
|
|
224
|
+
class Ollama # rubocop:disable Metrics/ClassLength
|
|
224
225
|
include Interface
|
|
225
226
|
include DiscardableClient
|
|
227
|
+
include VectorConfiguration
|
|
226
228
|
|
|
227
229
|
DEFAULT_MODEL = 'nomic-embed-text'
|
|
228
230
|
DEFAULT_HOST = 'http://localhost:11434'
|
|
@@ -270,14 +272,15 @@ module Woods
|
|
|
270
272
|
# unknown models. Set explicitly only if running a model with a
|
|
271
273
|
# known-larger native context that isn't in the registry yet.
|
|
272
274
|
# @param dimensions [Integer, nil] Requested output vector size.
|
|
275
|
+
# @param expected_dimensions [Integer, nil] Expected width; never sent to the API.
|
|
273
276
|
# @param read_timeout [Integer] HTTP read timeout in seconds.
|
|
274
277
|
# Bump this for slow / cold-start hosts or very large batches.
|
|
275
|
-
def initialize(model: DEFAULT_MODEL, host: DEFAULT_HOST, num_ctx: nil,
|
|
276
|
-
dimensions: nil, read_timeout: DEFAULT_READ_TIMEOUT)
|
|
278
|
+
def initialize(model: DEFAULT_MODEL, host: DEFAULT_HOST, num_ctx: nil, # rubocop:disable Metrics/ParameterLists
|
|
279
|
+
dimensions: nil, expected_dimensions: nil, read_timeout: DEFAULT_READ_TIMEOUT)
|
|
277
280
|
@model = model
|
|
278
281
|
@host = host
|
|
279
282
|
@num_ctx = num_ctx || MODEL_CONTEXT_LENGTHS.fetch(model, FALLBACK_NUM_CTX)
|
|
280
|
-
|
|
283
|
+
configure_dimensions(dimensions: dimensions, expected_dimensions: expected_dimensions)
|
|
281
284
|
@read_timeout = read_timeout
|
|
282
285
|
@uri = URI("#{host}/api/embed")
|
|
283
286
|
end
|
|
@@ -293,7 +296,7 @@ module Woods
|
|
|
293
296
|
|
|
294
297
|
response = post_request(build_body(text))
|
|
295
298
|
vectors = Array(response['embeddings'])
|
|
296
|
-
|
|
299
|
+
validate_vectors!(vectors, expected_count: 1, provider: 'Ollama')
|
|
297
300
|
vectors.first
|
|
298
301
|
end
|
|
299
302
|
|
|
@@ -311,7 +314,7 @@ module Woods
|
|
|
311
314
|
|
|
312
315
|
response = post_request(build_body(texts))
|
|
313
316
|
vectors = Array(response['embeddings'])
|
|
314
|
-
|
|
317
|
+
validate_vectors!(vectors, expected_count: texts.size, provider: 'Ollama')
|
|
315
318
|
vectors
|
|
316
319
|
end
|
|
317
320
|
|
|
@@ -321,7 +324,20 @@ module Woods
|
|
|
321
324
|
#
|
|
322
325
|
# @return [Integer] number of dimensions
|
|
323
326
|
def dimensions
|
|
324
|
-
@
|
|
327
|
+
configured_dimensions || @observed_dimensions || embed('test').length
|
|
328
|
+
end
|
|
329
|
+
|
|
330
|
+
# Pure configuration identity; probes never change request semantics or keys.
|
|
331
|
+
# @return [Array]
|
|
332
|
+
def cache_identity
|
|
333
|
+
[self.class.name, @host, @model, @num_ctx, requested_dimensions, configured_dimensions, 'truncate:false']
|
|
334
|
+
end
|
|
335
|
+
|
|
336
|
+
# Pure constructor settings for ResolvedConfig's allowlisted serialization.
|
|
337
|
+
# The snapshot layer checks the endpoint before persisting it.
|
|
338
|
+
def configuration_options
|
|
339
|
+
{ model: @model, host: @host, num_ctx: @num_ctx, read_timeout: @read_timeout,
|
|
340
|
+
dimensions: requested_dimensions, expected_dimensions: configured_dimensions || @observed_dimensions }
|
|
325
341
|
end
|
|
326
342
|
|
|
327
343
|
# Return the model name.
|
|
@@ -341,17 +357,14 @@ module Woods
|
|
|
341
357
|
@num_ctx
|
|
342
358
|
end
|
|
343
359
|
|
|
344
|
-
|
|
345
|
-
|
|
346
|
-
def
|
|
347
|
-
|
|
348
|
-
|
|
349
|
-
dimensions = Integer(value)
|
|
350
|
-
raise ArgumentError, "dimensions must be positive, got #{value.inspect}" unless dimensions.positive?
|
|
351
|
-
|
|
352
|
-
dimensions
|
|
360
|
+
# A model-specific tokenizer is not available locally. This is a sizing
|
|
361
|
+
# estimate; truncate:false makes the server reject an actual overflow.
|
|
362
|
+
def input_budget
|
|
363
|
+
@input_budget ||= InputBudget.new(limit: max_input_tokens, model: @model)
|
|
353
364
|
end
|
|
354
365
|
|
|
366
|
+
private
|
|
367
|
+
|
|
355
368
|
# Cap interpolated response bodies so misconfigured Ollama responses
|
|
356
369
|
# (e.g. proxied HTML error pages) don't unbounded-leak into logs or
|
|
357
370
|
# re-raised error messages.
|
|
@@ -365,12 +378,11 @@ module Woods
|
|
|
365
378
|
s.length > 500 ? "#{s[0, 500]}... [truncated]" : s
|
|
366
379
|
end
|
|
367
380
|
|
|
368
|
-
#
|
|
369
|
-
#
|
|
370
|
-
# tokens and returns 400 when the input exceeds that default.
|
|
381
|
+
# Request the configured context and explicitly refuse server truncation.
|
|
382
|
+
# Unknown model tokenization remains server-authoritative.
|
|
371
383
|
def build_body(input)
|
|
372
|
-
body = { model: @model, input: input }
|
|
373
|
-
body[:dimensions] =
|
|
384
|
+
body = { model: @model, input: input, truncate: false }
|
|
385
|
+
body[:dimensions] = requested_dimensions if requested_dimensions
|
|
374
386
|
body[:options] = { num_ctx: @num_ctx } if @num_ctx
|
|
375
387
|
body
|
|
376
388
|
end
|
|
@@ -1,6 +1,8 @@
|
|
|
1
1
|
# frozen_string_literal: true
|
|
2
2
|
|
|
3
3
|
require_relative '../token_utils'
|
|
4
|
+
require_relative 'input_budget'
|
|
5
|
+
require_relative '../chunking/contributor_chunks'
|
|
4
6
|
|
|
5
7
|
module Woods
|
|
6
8
|
module Embedding
|
|
@@ -12,8 +14,8 @@ module Woods
|
|
|
12
14
|
# file: ...
|
|
13
15
|
# dependencies: dep1, dep2, ...
|
|
14
16
|
#
|
|
15
|
-
#
|
|
16
|
-
#
|
|
17
|
+
# Refuses oversized direct inputs; the indexing path splits complete inputs
|
|
18
|
+
# without truncation before submitting them to the provider.
|
|
17
19
|
#
|
|
18
20
|
# @example
|
|
19
21
|
# preparer = Woods::Embedding::TextPreparer.new(max_tokens: 8192)
|
|
@@ -29,9 +31,10 @@ module Woods
|
|
|
29
31
|
|
|
30
32
|
# @param max_tokens [Integer] maximum token budget for prepared text
|
|
31
33
|
# @param chars_per_token [Float] tokenizer-calibrated char/token ratio
|
|
32
|
-
def initialize(max_tokens: DEFAULT_MAX_TOKENS, chars_per_token: DEFAULT_CHARS_PER_TOKEN)
|
|
34
|
+
def initialize(max_tokens: DEFAULT_MAX_TOKENS, chars_per_token: DEFAULT_CHARS_PER_TOKEN, input_budget: nil)
|
|
33
35
|
@max_tokens = max_tokens
|
|
34
36
|
@chars_per_token = chars_per_token
|
|
37
|
+
@input_budget = input_budget || InputBudget.new(limit: max_tokens, chars_per_token: chars_per_token)
|
|
35
38
|
end
|
|
36
39
|
|
|
37
40
|
# @return [Float] configured chars-per-token ratio
|
|
@@ -43,15 +46,15 @@ module Woods
|
|
|
43
46
|
# Prepare text for embedding from an ExtractedUnit.
|
|
44
47
|
#
|
|
45
48
|
# Builds a context prefix and appends the unit's source code (or first
|
|
46
|
-
# chunk content for chunked units).
|
|
49
|
+
# chunk content for chunked units). Refuses inputs exceeding the configured limit.
|
|
47
50
|
#
|
|
48
51
|
# @param unit [Woods::ExtractedUnit] the unit to prepare
|
|
49
52
|
# @return [String] context-prefixed text ready for embedding
|
|
50
|
-
def prepare(unit)
|
|
51
|
-
prefix = build_prefix(unit)
|
|
53
|
+
def prepare(unit, budget: @input_budget)
|
|
54
|
+
prefix = build_prefix(unit, unit.chunks.first)
|
|
52
55
|
content = select_content(unit)
|
|
53
56
|
text = "#{prefix}\n#{content}"
|
|
54
|
-
|
|
57
|
+
budget.validate!(text)
|
|
55
58
|
end
|
|
56
59
|
|
|
57
60
|
# Prepare text for each chunk of an ExtractedUnit.
|
|
@@ -62,31 +65,89 @@ module Woods
|
|
|
62
65
|
#
|
|
63
66
|
# @param unit [Woods::ExtractedUnit] the unit to prepare
|
|
64
67
|
# @return [Array<String>] array of context-prefixed texts
|
|
65
|
-
def prepare_chunks(unit)
|
|
66
|
-
|
|
68
|
+
def prepare_chunks(unit, budget: @input_budget)
|
|
69
|
+
Chunking::ContributorChunks.ensure!(unit)
|
|
70
|
+
return [prepare(unit, budget: budget)] unless unit.chunks&.any?
|
|
67
71
|
|
|
68
|
-
prefix = build_prefix(unit)
|
|
69
72
|
unit.chunks.map do |chunk|
|
|
70
|
-
text = "#{
|
|
71
|
-
|
|
73
|
+
text = "#{build_prefix(unit, chunk)}\n#{chunk[:content]}"
|
|
74
|
+
budget.validate!(text)
|
|
72
75
|
end
|
|
73
76
|
end
|
|
74
77
|
|
|
78
|
+
# Fit complete prefixed inputs without removing any source characters.
|
|
79
|
+
# Existing chunk attributes survive; embedding_slice records byte offsets
|
|
80
|
+
# relative to the original chunk for downstream physical-source attribution.
|
|
81
|
+
def prepare_for_embedding(unit, budget: @input_budget)
|
|
82
|
+
Chunking::ContributorChunks.ensure!(unit)
|
|
83
|
+
originals = unit.chunks.any? ? unit.chunks : [{ content: unit.source_code || '', chunk_type: :whole }]
|
|
84
|
+
fitted = originals.flat_map do |chunk|
|
|
85
|
+
prefix = "#{build_prefix(unit, chunk)}\n"
|
|
86
|
+
unless budget.fits?(prefix)
|
|
87
|
+
raise InputLimitError, "Embedding prefix exceeds input limit (#{budget.method}: #{budget.count(prefix)})"
|
|
88
|
+
end
|
|
89
|
+
|
|
90
|
+
fit_chunk(chunk, prefix, budget)
|
|
91
|
+
end
|
|
92
|
+
unit.chunks = fitted if unit.chunks.any? || fitted.size > 1
|
|
93
|
+
prepare_chunks(unit, budget: budget)
|
|
94
|
+
end
|
|
95
|
+
|
|
96
|
+
def preparation_identity
|
|
97
|
+
{ 'class' => self.class.name, 'version' => 2, 'budget' => @input_budget.identity }
|
|
98
|
+
end
|
|
99
|
+
|
|
75
100
|
private
|
|
76
101
|
|
|
102
|
+
def fit_chunk(chunk, prefix, budget)
|
|
103
|
+
content = chunk[:content].to_s
|
|
104
|
+
return [chunk] if budget.fits?(prefix + content)
|
|
105
|
+
|
|
106
|
+
offset = SourceContributors.field(chunk[:embedding_slice], :start_byte) || 0
|
|
107
|
+
initial_offset = offset
|
|
108
|
+
split_content(content, prefix, budget).map do |part|
|
|
109
|
+
first = offset
|
|
110
|
+
offset += part.bytesize
|
|
111
|
+
metadata = Chunking::ContributorChunks.slice_metadata(chunk[:metadata], content, first - initial_offset,
|
|
112
|
+
offset - initial_offset)
|
|
113
|
+
chunk.merge(content: part, metadata: metadata, embedding_slice: { start_byte: first, end_byte: offset })
|
|
114
|
+
end
|
|
115
|
+
end
|
|
116
|
+
|
|
117
|
+
def split_content(content, prefix, budget)
|
|
118
|
+
return [content] if budget.fits?(prefix + content)
|
|
119
|
+
if content.length <= 1
|
|
120
|
+
raise InputLimitError, 'Embedding input limit leaves no room for one source character after the prefix'
|
|
121
|
+
end
|
|
122
|
+
|
|
123
|
+
middle = content.length / 2
|
|
124
|
+
split_content(content[0...middle], prefix, budget) + split_content(content[middle..], prefix, budget)
|
|
125
|
+
end
|
|
126
|
+
|
|
77
127
|
# Build the context prefix for a unit.
|
|
78
128
|
#
|
|
79
129
|
# @param unit [Woods::ExtractedUnit] the unit
|
|
80
130
|
# @return [String] formatted prefix lines
|
|
81
|
-
def build_prefix(unit)
|
|
131
|
+
def build_prefix(unit, chunk = nil)
|
|
82
132
|
lines = []
|
|
83
133
|
lines << "[#{unit.type}] #{unit.identifier}"
|
|
84
134
|
lines << "namespace: #{unit.namespace}" if unit.namespace
|
|
85
|
-
|
|
135
|
+
path = chunk&.dig(:metadata, :physical_location, :file_path)
|
|
136
|
+
if SourceContributors.multiple?(unit)
|
|
137
|
+
lines.concat(contributor_prefix(unit, path))
|
|
138
|
+
elsif unit.file_path
|
|
139
|
+
lines << "file: #{unit.file_path}"
|
|
140
|
+
end
|
|
86
141
|
append_dependency_line(lines, unit.dependencies)
|
|
87
142
|
lines.join("\n")
|
|
88
143
|
end
|
|
89
144
|
|
|
145
|
+
def contributor_prefix(unit, path)
|
|
146
|
+
return ["file: #{path}"] if path
|
|
147
|
+
|
|
148
|
+
["primary file: #{unit.file_path}", "contributing files: #{SourceContributors.paths(unit).join(', ')}"]
|
|
149
|
+
end
|
|
150
|
+
|
|
90
151
|
# Append a formatted dependency line if dependencies exist.
|
|
91
152
|
#
|
|
92
153
|
# @param lines [Array<String>] lines to append to
|
|
@@ -100,7 +161,7 @@ module Woods
|
|
|
100
161
|
# reads JSON and does not symbolize dependency keys, unlike chunks).
|
|
101
162
|
# Read both forms or the whole "dependencies:" prefix silently
|
|
102
163
|
# vanishes from every embedded document on the indexing path.
|
|
103
|
-
dep_names = dependencies.filter_map { |d| d[:target] || d['target'] }.first(10)
|
|
164
|
+
dep_names = dependencies.filter_map { |d| d.is_a?(String) ? d : d[:target] || d['target'] }.first(10)
|
|
104
165
|
lines << "dependencies: #{dep_names.join(', ')}" if dep_names.any?
|
|
105
166
|
end
|
|
106
167
|
|
|
@@ -115,23 +176,6 @@ module Woods
|
|
|
115
176
|
unit.source_code || ''
|
|
116
177
|
end
|
|
117
178
|
end
|
|
118
|
-
|
|
119
|
-
# Truncate text to fit within the token budget.
|
|
120
|
-
#
|
|
121
|
-
# Uses the configured `chars_per_token` ratio to estimate both the
|
|
122
|
-
# token count and the safe character cap. Truncation is a last
|
|
123
|
-
# resort — by the time text reaches here the chunker should have
|
|
124
|
-
# already split oversize units into pieces that fit.
|
|
125
|
-
#
|
|
126
|
-
# @param text [String] the text to truncate
|
|
127
|
-
# @return [String] text within token limits
|
|
128
|
-
def enforce_token_limit(text)
|
|
129
|
-
estimated = (text.length / @chars_per_token).ceil
|
|
130
|
-
return text if estimated <= @max_tokens
|
|
131
|
-
|
|
132
|
-
max_chars = (@max_tokens * @chars_per_token).floor
|
|
133
|
-
text[0...max_chars]
|
|
134
|
-
end
|
|
135
179
|
end
|
|
136
180
|
end
|
|
137
181
|
end
|
|
@@ -4,114 +4,51 @@ require 'set'
|
|
|
4
4
|
|
|
5
5
|
module Woods
|
|
6
6
|
module Embedding
|
|
7
|
-
#
|
|
8
|
-
#
|
|
9
|
-
# When the optional `tokenizers` gem (ankane) is installed, loads the
|
|
10
|
-
# `bert-base-uncased` WordPiece tokenizer that nomic-embed-text is
|
|
11
|
-
# built on and returns exact token counts. Otherwise falls back to a
|
|
12
|
-
# conservative chars/token ratio and warns once.
|
|
13
|
-
#
|
|
14
|
-
# Exact counting is strictly preferred for the Ollama path — Ollama
|
|
15
|
-
# v0.13.5+ stopped honouring the `truncate: true` flag on
|
|
16
|
-
# `/api/embed` (see ollama/ollama#14186), so chunks that exceed
|
|
17
|
-
# `num_ctx` return a 400 instead of being truncated. Client-side
|
|
18
|
-
# sizing is the only reliable option until the regression is fixed
|
|
19
|
-
# upstream, and chars/token ratios vary too widely across Rails
|
|
20
|
-
# internals to cover every case with a fixed number.
|
|
21
|
-
#
|
|
22
|
-
# @example
|
|
23
|
-
# counter = Woods::Embedding::TokenCounter.new
|
|
24
|
-
# counter.count("ActionController::Metal::ConditionalGet") # => 13
|
|
7
|
+
# Counts with an explicitly supplied local tokenizer, otherwise estimates.
|
|
8
|
+
# The caller owns the tokenizer/model match. No model is downloaded here.
|
|
25
9
|
class TokenCounter
|
|
26
|
-
#
|
|
10
|
+
# Legacy constants/keyword remain accepted; tokenizer_id no longer loads
|
|
11
|
+
# a remote tokenizer or implies that BERT matches the configured model.
|
|
27
12
|
BERT_MODEL = 'bert-base-uncased'
|
|
28
|
-
|
|
29
|
-
# Conservative floor for when the tokenizer gem isn't installed.
|
|
30
|
-
# Lower than any ratio we've observed failing in the testbed
|
|
31
|
-
# against dense Rails source. Still approximate — install
|
|
32
|
-
# `tokenizers` for exact counts.
|
|
33
13
|
CONSERVATIVE_CHARS_PER_TOKEN = 1.2
|
|
34
14
|
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
15
|
+
attr_reader :chars_per_token
|
|
16
|
+
|
|
17
|
+
def initialize(chars_per_token: CONSERVATIVE_CHARS_PER_TOKEN, tokenizer_id: nil, tokenizer: nil)
|
|
18
|
+
raise ArgumentError, 'chars_per_token must be positive' unless chars_per_token.positive?
|
|
19
|
+
|
|
40
20
|
@chars_per_token = chars_per_token
|
|
21
|
+
@tokenizer = tokenizer
|
|
41
22
|
@tokenizer_id = tokenizer_id
|
|
42
|
-
@load_attempted = false
|
|
43
|
-
@load_mutex = Mutex.new
|
|
44
23
|
end
|
|
45
24
|
|
|
46
|
-
# @return [Float] fallback chars-per-token ratio
|
|
47
|
-
attr_reader :chars_per_token
|
|
48
|
-
|
|
49
|
-
# Exact token count when the tokenizer is loaded, chars/token
|
|
50
|
-
# estimate otherwise.
|
|
51
|
-
#
|
|
52
|
-
# @param text [String, nil]
|
|
53
|
-
# @return [Integer]
|
|
54
25
|
def count(text)
|
|
55
26
|
return 0 if text.nil? || text.empty?
|
|
27
|
+
return @tokenizer.encode(text).ids.length if @tokenizer
|
|
56
28
|
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
end
|
|
60
|
-
|
|
61
|
-
private
|
|
62
|
-
|
|
63
|
-
def estimate(text)
|
|
29
|
+
warn_once('Embedding token counts are estimates: no model-matched local tokenizer was supplied. ' \
|
|
30
|
+
'Ollama uses truncate:false and rejects inputs beyond its actual context window.')
|
|
64
31
|
(text.length / @chars_per_token).ceil
|
|
65
32
|
end
|
|
66
33
|
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
# (successful or not) we memoize the result and skip the load path.
|
|
70
|
-
def tokenizer
|
|
71
|
-
@load_mutex.synchronize do
|
|
72
|
-
return @tokenizer if @load_attempted
|
|
73
|
-
|
|
74
|
-
@load_attempted = true
|
|
75
|
-
@tokenizer = try_load
|
|
76
|
-
end
|
|
77
|
-
end
|
|
78
|
-
|
|
79
|
-
def try_load
|
|
80
|
-
require 'tokenizers'
|
|
81
|
-
Tokenizers.from_pretrained(@tokenizer_id)
|
|
82
|
-
rescue LoadError
|
|
83
|
-
warn_once(
|
|
84
|
-
'Exact token counting disabled: `tokenizers` gem not installed. ' \
|
|
85
|
-
"Falling back to #{@chars_per_token} chars/token estimation. " \
|
|
86
|
-
"Add `gem 'tokenizers', '~> 0.5'` to your Gemfile for exact sizing on Ollama."
|
|
87
|
-
)
|
|
88
|
-
nil
|
|
89
|
-
rescue StandardError => e
|
|
90
|
-
warn_once(
|
|
91
|
-
"Could not load tokenizer #{@tokenizer_id.inspect} " \
|
|
92
|
-
"(#{e.class}: #{e.message}). Falling back to chars/token estimate."
|
|
93
|
-
)
|
|
94
|
-
nil
|
|
34
|
+
def exact?
|
|
35
|
+
!@tokenizer.nil?
|
|
95
36
|
end
|
|
96
37
|
|
|
97
|
-
# Per-process dedup so multiple TokenCounter instances (one per
|
|
98
|
-
# retriever build, plus one per chunker, plus tests) don't each
|
|
99
|
-
# spam the same fallback warning. The mutex keeps the dedup set
|
|
100
|
-
# consistent under the same concurrent-first-call pattern that
|
|
101
|
-
# the per-instance load mutex protects against.
|
|
102
38
|
@warned_messages = Set.new
|
|
103
39
|
@warned_mutex = Mutex.new
|
|
104
40
|
|
|
105
41
|
class << self
|
|
106
42
|
attr_reader :warned_messages, :warned_mutex
|
|
107
43
|
|
|
108
|
-
#
|
|
109
|
-
# callers should never need to clear it.
|
|
44
|
+
# Test seam for process-wide diagnostic deduplication.
|
|
110
45
|
def reset_warned!
|
|
111
46
|
@warned_mutex.synchronize { @warned_messages.clear }
|
|
112
47
|
end
|
|
113
48
|
end
|
|
114
49
|
|
|
50
|
+
private
|
|
51
|
+
|
|
115
52
|
def warn_once(message)
|
|
116
53
|
full = "[woods] #{message}"
|
|
117
54
|
self.class.warned_mutex.synchronize do
|