woods 2.0.1 → 2.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +94 -7
- data/CONTRIBUTING.md +134 -19
- data/README.md +1 -1
- data/docs/AGENT_GUIDE.md +19 -0
- data/docs/AGENT_SETUP.md +22 -2
- data/docs/BACKEND_MATRIX.md +7 -0
- data/docs/CLIENT_HOOKS.md +6 -0
- data/docs/CONFIGURATION_REFERENCE.md +133 -25
- data/docs/CONSOLE_MCP_SETUP.md +82 -30
- data/docs/EMBEDDING_MODELS.md +16 -19
- data/docs/EXTRACTOR_REFERENCE.md +219 -21
- data/docs/FAQ.md +11 -25
- data/docs/GETTING_STARTED.md +7 -1
- data/docs/INCREMENTAL_EXTRACTION.md +261 -19
- data/docs/INDEX_LAYOUT.md +5 -0
- data/docs/INTERNALS.md +9 -0
- data/docs/MCP_HTTP_TRANSPORT.md +20 -15
- data/docs/MCP_SERVERS.md +87 -8
- data/docs/MCP_TOOL_COOKBOOK.md +13 -55
- data/docs/NOTION_INTEGRATION.md +7 -1
- data/docs/PUBLISHED_INDEX.md +6 -0
- data/docs/README.md +6 -1
- data/docs/RETRIEVAL_GUIDE.md +17 -0
- data/docs/SOURCE_FRESHNESS.md +157 -5
- data/docs/TOKEN_BENCHMARK.md +10 -18
- data/docs/TROUBLESHOOTING.md +70 -14
- data/docs/UNBLOCKED_INTEGRATION.md +60 -8
- data/docs/UPGRADING_TO_2.md +153 -38
- data/docs/WATCH_DAEMON.md +97 -14
- data/exe/woods-console-mcp +2 -2
- data/lib/generators/woods/templates/woods.rb.tt +2 -1
- data/lib/tasks/woods.rake +23 -7
- data/lib/tasks/woods_checks.rake +2 -2
- data/lib/woods/agent_configuration/cli.rb +1 -1
- data/lib/woods/agent_configuration/layout.rb +16 -2
- data/lib/woods/agent_configuration/plan.rb +13 -3
- data/lib/woods/agent_configuration/planner_validation.rb +4 -2
- data/lib/woods/agent_configuration/preflight.rb +5 -3
- data/lib/woods/builder.rb +17 -57
- data/lib/woods/cache/cache_middleware.rb +56 -30
- data/lib/woods/chunking/contributor_chunks.rb +119 -0
- data/lib/woods/chunking/semantic_chunker.rb +44 -21
- data/lib/woods/console/connection_manager.rb +56 -3
- data/lib/woods/console/embedded_executor.rb +30 -5
- data/lib/woods/console/rack_middleware.rb +29 -1
- data/lib/woods/dependency_graph.rb +34 -10
- data/lib/woods/embedding/fake.rb +12 -0
- data/lib/woods/embedding/indexer.rb +195 -98
- data/lib/woods/embedding/input_budget.rb +67 -0
- data/lib/woods/embedding/openai.rb +70 -20
- data/lib/woods/embedding/provider.rb +37 -25
- data/lib/woods/embedding/text_preparer.rb +76 -32
- data/lib/woods/embedding/token_counter.rb +18 -81
- data/lib/woods/embedding/vector_configuration.rb +48 -0
- data/lib/woods/extraction_identities.rb +175 -0
- data/lib/woods/extractor.rb +304 -107
- data/lib/woods/extractors/action_cable_extractor.rb +8 -3
- data/lib/woods/extractors/assigned_value_discovery.rb +74 -0
- data/lib/woods/extractors/class_declarations.rb +121 -0
- data/lib/woods/extractors/configuration_extractor.rb +11 -3
- data/lib/woods/extractors/declaration_ancestry.rb +92 -0
- data/lib/woods/extractors/event_extractor.rb +8 -0
- data/lib/woods/extractors/graphql_extractor.rb +134 -77
- data/lib/woods/extractors/job_extractor.rb +5 -1
- data/lib/woods/extractors/lib_extractor.rb +132 -15
- data/lib/woods/extractors/mailer_extractor.rb +3 -5
- data/lib/woods/extractors/manager_extractor.rb +7 -21
- data/lib/woods/extractors/migration_declaration.rb +87 -0
- data/lib/woods/extractors/migration_extractor.rb +5 -39
- data/lib/woods/extractors/phlex_extractor.rb +6 -2
- data/lib/woods/extractors/policy_extractor.rb +9 -5
- data/lib/woods/extractors/poro_extractor.rb +112 -53
- data/lib/woods/extractors/pundit_extractor.rb +11 -6
- data/lib/woods/extractors/scheduled_job_extractor.rb +45 -4
- data/lib/woods/extractors/serializer_extractor.rb +34 -22
- data/lib/woods/extractors/shared_utility_methods.rb +18 -1
- data/lib/woods/extractors/source_nesting.rb +142 -106
- data/lib/woods/extractors/standalone_module_discovery.rb +123 -0
- data/lib/woods/extractors/state_machine_extractor.rb +46 -40
- data/lib/woods/extractors/view_component_extractor.rb +9 -7
- data/lib/woods/flow_assembler.rb +4 -1
- data/lib/woods/generation.rb +25 -0
- data/lib/woods/hooks/context_hint.rb +7 -2
- data/lib/woods/mcp/bootstrapper.rb +33 -7
- data/lib/woods/mcp/config_resolver.rb +26 -7
- data/lib/woods/mcp/index_reader.rb +125 -24
- data/lib/woods/mcp/index_reader_pinning.rb +16 -0
- data/lib/woods/mcp/renderers/markdown_renderer.rb +7 -1
- data/lib/woods/mcp/renderers/plain_renderer.rb +3 -1
- data/lib/woods/mcp/search_results.rb +7 -1
- data/lib/woods/mcp/server.rb +24 -4
- data/lib/woods/module_reconciliation.rb +151 -0
- data/lib/woods/path_dispatcher.rb +7 -2
- data/lib/woods/rake_helpers.rb +43 -11
- data/lib/woods/release.rb +1 -1
- data/lib/woods/resilience/index_validator.rb +8 -3
- data/lib/woods/resilience/retryable_provider.rb +18 -1
- data/lib/woods/resolved_config.rb +68 -8
- data/lib/woods/retrieval/context_assembler.rb +3 -3
- data/lib/woods/retrieval/lexical_assembler.rb +3 -2
- data/lib/woods/retrieval/scope.rb +18 -2
- data/lib/woods/retrieval/source_evidence.rb +14 -2
- data/lib/woods/source_contributor_validation.rb +78 -0
- data/lib/woods/source_contributors.rb +116 -0
- data/lib/woods/source_inputs/handoff.rb +37 -0
- data/lib/woods/source_inputs/launcher.rb +53 -13
- data/lib/woods/source_inputs/manifest.rb +84 -3
- data/lib/woods/source_inputs/private_key.rb +44 -12
- data/lib/woods/source_inputs/scanner.rb +98 -27
- data/lib/woods/source_inputs/scopes.rb +1 -1
- data/lib/woods/source_inputs/session.rb +147 -15
- data/lib/woods/source_inputs/stable_reader.rb +127 -0
- data/lib/woods/source_inputs/status.rb +40 -8
- data/lib/woods/source_inputs/verifier.rb +28 -5
- data/lib/woods/source_path_encoding.rb +33 -0
- data/lib/woods/source_references/cache.rb +284 -0
- data/lib/woods/source_references/collector.rb +120 -0
- data/lib/woods/source_references/extraction.rb +185 -0
- data/lib/woods/source_references/inputs.rb +134 -0
- data/lib/woods/source_references/parser_adapter.rb +134 -0
- data/lib/woods/source_references/pass.rb +152 -0
- data/lib/woods/source_references/prism_adapter.rb +116 -0
- data/lib/woods/source_references/registry.rb +178 -0
- data/lib/woods/source_references/runtime_lookup.rb +127 -0
- data/lib/woods/source_references/value_class.rb +82 -0
- data/lib/woods/storage/metadata_store.rb +4 -1
- data/lib/woods/storage/qdrant.rb +2 -2
- data/lib/woods/unblocked/client.rb +12 -7
- data/lib/woods/unblocked/document_builder.rb +4 -1
- data/lib/woods/unblocked/exporter.rb +127 -37
- data/lib/woods/unblocked/sync_manifest.rb +137 -21
- data/lib/woods/unblocked/uri_migration.rb +105 -0
- data/lib/woods/util/host_guard.rb +3 -2
- data/lib/woods/version.rb +1 -1
- data/lib/woods/watch/catch_up.rb +138 -0
- data/lib/woods/watch/claim_lease.rb +150 -0
- data/lib/woods/watch/cli.rb +26 -2
- data/lib/woods/watch/daemon.rb +80 -59
- data/lib/woods/watch/installation/options.rb +1 -1
- data/lib/woods/watch/installation/receipt.rb +6 -1
- data/lib/woods/watch/managed_child.rb +1 -1
- data/lib/woods/watch/supervisor.rb +1 -1
- data/lib/woods/watch/tree_scan.rb +14 -2
- data/plugin/.claude-plugin/plugin.json +1 -1
- data/plugin/hooks/adapters/normalize.rb +3 -2
- data/plugin/hooks/woods-input-rules.sh +4 -0
- data/plugin/hooks/woods-refresh.sh +15 -7
- data/plugin/hooks/woods-session-start.sh +60 -3
- data/plugin/skills/woods-diagnose/SKILL.md +334 -11
- data/plugin/skills/woods-investigate/SKILL.md +11 -0
- data/plugin/skills/woods-mcp-config/SKILL.md +79 -8
- data/plugin/skills/woods-setup/SKILL.md +53 -6
- metadata +32 -5
data/lib/tasks/woods.rake
CHANGED
|
@@ -58,6 +58,7 @@ namespace :woods do
|
|
|
58
58
|
desc 'Incremental extraction based on git changes'
|
|
59
59
|
task incremental: :environment do
|
|
60
60
|
require 'woods/extractor'
|
|
61
|
+
require 'woods/input_rules'
|
|
61
62
|
|
|
62
63
|
output_dir = ENV.fetch('WOODS_OUTPUT', Woods.configuration.output_dir)
|
|
63
64
|
|
|
@@ -71,8 +72,9 @@ namespace :woods do
|
|
|
71
72
|
# PathDispatcher alongside the dispatch itself — a second hand-maintained
|
|
72
73
|
# pattern list here would drift, and a path this filter drops never
|
|
73
74
|
# reaches the index however good the dispatch behind it is (#164).
|
|
74
|
-
|
|
75
|
-
changed_files = changed_files.reject(&:empty?).
|
|
75
|
+
input_rules = Woods::InputRules.new
|
|
76
|
+
changed_files = changed_files.reject(&:empty?).reject { |path| input_rules.action(path) == :ignore }
|
|
77
|
+
full = changed_files.any? { |path| input_rules.action(path) == :full }
|
|
76
78
|
|
|
77
79
|
if changed_files.empty?
|
|
78
80
|
puts 'No relevant files changed. Skipping extraction.'
|
|
@@ -91,13 +93,17 @@ namespace :woods do
|
|
|
91
93
|
puts 'Warning: a watch daemon is alive but degraded — extracting anyway rather than assuming coverage.'
|
|
92
94
|
end
|
|
93
95
|
|
|
94
|
-
puts "Incremental extraction for #{changed_files.size} changed files..."
|
|
96
|
+
puts "#{full ? 'Full' : 'Incremental'} extraction for #{changed_files.size} changed files..."
|
|
95
97
|
changed_files.each { |f| puts " - #{f}" }
|
|
96
98
|
puts
|
|
97
99
|
|
|
98
100
|
extractor = Woods::Extractor.new(output_dir: output_dir)
|
|
99
101
|
affected = Woods::RakeHelpers.woods_with_extraction_lock(output_dir) do
|
|
100
|
-
extracted =
|
|
102
|
+
extracted = if full
|
|
103
|
+
extractor.extract_all.values.flatten(1).map(&:identifier).uniq
|
|
104
|
+
else
|
|
105
|
+
extractor.extract_changed(changed_files)
|
|
106
|
+
end
|
|
101
107
|
extractor.raise_on_publication_failure!
|
|
102
108
|
extracted
|
|
103
109
|
end
|
|
@@ -168,7 +174,7 @@ namespace :woods do
|
|
|
168
174
|
# FS events do not propagate and the daemon would sit silent. Nothing
|
|
169
175
|
# exposed it before, which made the advice unfollowable.
|
|
170
176
|
force_polling: ENV['WOODS_WATCH_POLL'] == '1', # container autodetect also applies; see Watcher.containerized?
|
|
171
|
-
idle_timeout:
|
|
177
|
+
idle_timeout: Woods::RakeHelpers.woods_watch_idle_timeout,
|
|
172
178
|
catch_up: ENV['WOODS_WATCH_CATCH_UP'] != '0',
|
|
173
179
|
boot_snapshot: boot_snapshot,
|
|
174
180
|
lifecycle: managed_child,
|
|
@@ -724,6 +730,8 @@ namespace :woods do
|
|
|
724
730
|
env_flag = ->(name) { %w[1 true yes].include?(ENV.fetch(name, '').strip.downcase) }
|
|
725
731
|
force_full = env_flag.call('UNBLOCKED_FORCE_FULL_SYNC')
|
|
726
732
|
force_purge = env_flag.call('UNBLOCKED_FORCE_PURGE')
|
|
733
|
+
dry_run = env_flag.call('UNBLOCKED_DRY_RUN')
|
|
734
|
+
migrate_from_ref = ENV.fetch('UNBLOCKED_MIGRATE_FROM_REF', nil)
|
|
727
735
|
|
|
728
736
|
puts 'Syncing extraction data to Unblocked...'
|
|
729
737
|
puts " Output dir: #{output_dir}"
|
|
@@ -735,12 +743,19 @@ namespace :woods do
|
|
|
735
743
|
exporter = Woods::Unblocked::Exporter.new(
|
|
736
744
|
index_dir: output_dir,
|
|
737
745
|
force_full: force_full,
|
|
738
|
-
force_purge: force_purge
|
|
746
|
+
force_purge: force_purge,
|
|
747
|
+
dry_run: dry_run,
|
|
748
|
+
migrate_from_ref: migrate_from_ref
|
|
739
749
|
)
|
|
740
750
|
stats = exporter.sync_all
|
|
741
751
|
|
|
742
752
|
puts
|
|
743
|
-
|
|
753
|
+
if stats[:dry_run]
|
|
754
|
+
puts 'Sync preview (no remote requests or local writes):'
|
|
755
|
+
puts JSON.pretty_generate(stats)
|
|
756
|
+
next
|
|
757
|
+
end
|
|
758
|
+
puts(stats[:complete] == false ? 'Sync incomplete; re-run after resolving the reported condition.' : 'Sync complete!')
|
|
744
759
|
puts " Documents synced: #{stats[:synced]}"
|
|
745
760
|
puts " Documents skipped: #{stats[:skipped]}"
|
|
746
761
|
puts " Documents deleted: #{stats[:deleted]}"
|
|
@@ -765,6 +780,7 @@ namespace :woods do
|
|
|
765
780
|
exit 1
|
|
766
781
|
end
|
|
767
782
|
end
|
|
783
|
+
exit 1 if stats[:complete] == false && stats[:errors].empty?
|
|
768
784
|
end
|
|
769
785
|
|
|
770
786
|
desc 'Relay findings to Unblocked (alias for unblocked_sync)'
|
data/lib/tasks/woods_checks.rake
CHANGED
|
@@ -11,6 +11,7 @@
|
|
|
11
11
|
# WOODS_CHECK_STRICT=1 ... # exit 1 on findings (still heuristic; see below)
|
|
12
12
|
|
|
13
13
|
require 'json'
|
|
14
|
+
require 'woods/rake_helpers'
|
|
14
15
|
|
|
15
16
|
namespace :woods do
|
|
16
17
|
namespace :check do
|
|
@@ -62,8 +63,7 @@ module Woods
|
|
|
62
63
|
# @return [String]
|
|
63
64
|
def check_index_dir
|
|
64
65
|
ENV.fetch('WOODS_OUTPUT') do
|
|
65
|
-
|
|
66
|
-
File.join(root, 'tmp/woods')
|
|
66
|
+
File.join(Woods::RakeHelpers.woods_task_root, 'tmp/woods')
|
|
67
67
|
end
|
|
68
68
|
end
|
|
69
69
|
|
|
@@ -86,7 +86,7 @@ module Woods
|
|
|
86
86
|
protected_paths = layout.allowed_paths + layout.runtime_paths
|
|
87
87
|
raise Conflict, 'Plan output must differ from every managed/runtime target' if protected_paths.include?(target)
|
|
88
88
|
|
|
89
|
-
|
|
89
|
+
Plan.validate_path!(target)
|
|
90
90
|
content = "#{JSON.pretty_generate(plan.data)}\n"
|
|
91
91
|
raise Conflict, 'Plan exceeds the supported size' if content.bytesize > Document::MAX_BYTES
|
|
92
92
|
|
|
@@ -64,8 +64,22 @@ module Woods
|
|
|
64
64
|
end
|
|
65
65
|
|
|
66
66
|
def identity
|
|
67
|
-
{ 'client' => 'claude', 'scope' => scope, 'root' => root,
|
|
68
|
-
'config_path' => config_path, 'receipt_path' => receipt_path }
|
|
67
|
+
{ 'client' => 'claude', 'scope' => scope, 'root' => root,
|
|
68
|
+
'config_path' => config_path, 'receipt_path' => receipt_path }.tap do |value|
|
|
69
|
+
value['config_dir'] = config_dir if scope == 'user'
|
|
70
|
+
end
|
|
71
|
+
end
|
|
72
|
+
|
|
73
|
+
# Older project receipts/plans included an unused user configuration path.
|
|
74
|
+
# Only that field is irrelevant; every actual target and scope stays exact.
|
|
75
|
+
def self.validate_identity!(record, expected, message:)
|
|
76
|
+
raise Conflict, "#{message} (identity field: layout)" unless record.is_a?(Hash)
|
|
77
|
+
|
|
78
|
+
record = record.except('config_dir') if expected['client'] == 'claude' && expected['scope'] == 'project'
|
|
79
|
+
return if record == expected
|
|
80
|
+
|
|
81
|
+
fields = (record.keys | expected.keys).reject { |key| record.slice(key) == expected.slice(key) }
|
|
82
|
+
raise Conflict, "#{message} (identity fields: #{fields.join(', ')})"
|
|
69
83
|
end
|
|
70
84
|
end
|
|
71
85
|
end
|
|
@@ -3,6 +3,7 @@
|
|
|
3
3
|
require 'base64'
|
|
4
4
|
require 'json'
|
|
5
5
|
require_relative 'document'
|
|
6
|
+
require_relative 'layout'
|
|
6
7
|
|
|
7
8
|
module Woods
|
|
8
9
|
module AgentConfiguration
|
|
@@ -19,11 +20,19 @@ module Woods
|
|
|
19
20
|
end
|
|
20
21
|
|
|
21
22
|
def self.load(path)
|
|
23
|
+
validate_path!(path)
|
|
22
24
|
object = allocate
|
|
23
25
|
object.instance_variable_set(:@data, Document.new(path).json)
|
|
24
26
|
object
|
|
25
27
|
end
|
|
26
28
|
|
|
29
|
+
def self.validate_path!(path)
|
|
30
|
+
Document.validate_path!(path)
|
|
31
|
+
rescue Conflict => e
|
|
32
|
+
raise Conflict, "#{e.message}. Use a regular plan file under a real parent directory; " \
|
|
33
|
+
'for temporary plans, create a private directory and resolve it with File.realpath.'
|
|
34
|
+
end
|
|
35
|
+
|
|
27
36
|
def add(document, content, description:)
|
|
28
37
|
return if content == document.content
|
|
29
38
|
|
|
@@ -67,12 +76,13 @@ module Woods
|
|
|
67
76
|
private
|
|
68
77
|
|
|
69
78
|
def validate_format!(layout)
|
|
70
|
-
valid = data.is_a?(Hash) && data['schema_version'] == SCHEMA_VERSION &&
|
|
79
|
+
valid = data.is_a?(Hash) && data['schema_version'] == SCHEMA_VERSION &&
|
|
71
80
|
%w[setup update remove].include?(data['operation']) && data['changes'].is_a?(Array) &&
|
|
72
81
|
data['changes'].all?(Hash)
|
|
73
|
-
|
|
82
|
+
message = 'Plan format or selected application/client/scope differs; create a fresh preview'
|
|
83
|
+
raise Conflict, message unless valid
|
|
74
84
|
|
|
75
|
-
|
|
85
|
+
Layout.validate_identity!(data['layout'], layout.identity, message: message)
|
|
76
86
|
end
|
|
77
87
|
|
|
78
88
|
def validate_paths!(layout)
|
|
@@ -19,11 +19,13 @@ module Woods
|
|
|
19
19
|
end
|
|
20
20
|
|
|
21
21
|
def validate_receipt_identity!
|
|
22
|
-
valid = @previous['schema_version'] == 1 &&
|
|
22
|
+
valid = @previous['schema_version'] == 1 &&
|
|
23
23
|
@previous['server_name'] == @name && @previous['sections'].is_a?(Hash) &&
|
|
24
24
|
@previous['created_files'].is_a?(Array)
|
|
25
|
-
|
|
25
|
+
message = 'Installation receipt does not match this application/client/scope/server'
|
|
26
|
+
raise Conflict, message unless valid
|
|
26
27
|
|
|
28
|
+
Layout.validate_identity!(@previous['layout'], @layout.identity, message: message)
|
|
27
29
|
validate_receipt_content!
|
|
28
30
|
end
|
|
29
31
|
|
|
@@ -3,8 +3,8 @@
|
|
|
3
3
|
require 'json'
|
|
4
4
|
require 'open3'
|
|
5
5
|
require 'timeout'
|
|
6
|
-
require 'bundler'
|
|
7
6
|
require_relative 'error'
|
|
7
|
+
require_relative '../watch/child_environment'
|
|
8
8
|
|
|
9
9
|
module Woods
|
|
10
10
|
module AgentConfiguration
|
|
@@ -32,8 +32,9 @@ module Woods
|
|
|
32
32
|
end
|
|
33
33
|
RUBY
|
|
34
34
|
|
|
35
|
-
def initialize(timeout: 30)
|
|
35
|
+
def initialize(timeout: 30, environment: ENV)
|
|
36
36
|
@timeout = timeout
|
|
37
|
+
@environment = environment
|
|
37
38
|
end
|
|
38
39
|
|
|
39
40
|
def call(launcher)
|
|
@@ -54,7 +55,8 @@ module Woods
|
|
|
54
55
|
|
|
55
56
|
def run(launcher)
|
|
56
57
|
command = launcher.probe_command(SCRIPT)
|
|
57
|
-
environment =
|
|
58
|
+
environment = Watch::ChildEnvironment.build(@environment.to_h, root: launcher.root)
|
|
59
|
+
.merge(launcher.probe_environment)
|
|
58
60
|
Open3.popen3(environment, *command, chdir: launcher.root, unsetenv_others: true,
|
|
59
61
|
pgroup: true) do |stdin, out, err, wait|
|
|
60
62
|
stdin.close
|
data/lib/woods/builder.rb
CHANGED
|
@@ -191,10 +191,9 @@ module Woods
|
|
|
191
191
|
# their own implementation without patching the Builder. It flows
|
|
192
192
|
# through {#build_resilient_embedding_provider} like the built-ins.
|
|
193
193
|
#
|
|
194
|
-
#
|
|
195
|
-
#
|
|
196
|
-
#
|
|
197
|
-
# aren't part of the provider's API.
|
|
194
|
+
# The legacy :dimension option aliases the explicit :dimensions request.
|
|
195
|
+
# Stored observed widths use :expected_dimensions so restoring an index
|
|
196
|
+
# does not accidentally change the provider request.
|
|
198
197
|
#
|
|
199
198
|
# @return [Embedding::Provider::Interface] Embedding provider instance
|
|
200
199
|
# @raise [ArgumentError] if the configured type is not recognized
|
|
@@ -244,8 +243,8 @@ module Woods
|
|
|
244
243
|
end
|
|
245
244
|
|
|
246
245
|
PROVIDER_OPTION_KEYS = {
|
|
247
|
-
openai: %i[api_key model dimension dimensions],
|
|
248
|
-
ollama: %i[model host num_ctx read_timeout dimension dimensions],
|
|
246
|
+
openai: %i[api_key model dimension dimensions expected_dimensions],
|
|
247
|
+
ollama: %i[model host num_ctx read_timeout dimension dimensions expected_dimensions],
|
|
249
248
|
fake: %i[model dims dimension dimensions]
|
|
250
249
|
}.freeze
|
|
251
250
|
private_constant :PROVIDER_OPTION_KEYS
|
|
@@ -328,10 +327,9 @@ module Woods
|
|
|
328
327
|
# Build the deterministic fake provider (#178).
|
|
329
328
|
#
|
|
330
329
|
# Dimension resolution: `embedding_options[:dims]` maps directly onto
|
|
331
|
-
# the {Embedding::Provider::Fake} constructor; failing that, the
|
|
332
|
-
#
|
|
333
|
-
#
|
|
334
|
-
# honoured, so hosts that declare their dimension there (and the MCP
|
|
330
|
+
# the {Embedding::Provider::Fake} constructor; failing that, the legacy
|
|
331
|
+
# `embedding_options[:dimension]` alias is honoured, so hosts that
|
|
332
|
+
# declare their dimension there (and the MCP
|
|
335
333
|
# boot path, which restores exactly that key from woods.json) get
|
|
336
334
|
# vectors of the recorded dimension.
|
|
337
335
|
#
|
|
@@ -341,54 +339,20 @@ module Woods
|
|
|
341
339
|
end
|
|
342
340
|
private :build_fake_provider
|
|
343
341
|
|
|
344
|
-
# Build a
|
|
345
|
-
#
|
|
346
|
-
# OpenAI embedders use tiktoken (cl100k_base) — 4.0 chars/token is a
|
|
347
|
-
# good conservative average. Ollama BERT/WordPiece tokenizers
|
|
348
|
-
# (nomic-embed-text, bge-*) run much hotter on dense Ruby/Rails
|
|
349
|
-
# source — long CamelCase constants, docstrings, callback DSLs, and
|
|
350
|
-
# heavy symbol use all sit below 2.0 chars/token in practice.
|
|
351
|
-
# Empirically, a 16 KB chunk of `ActionMailer::Base` still blows the
|
|
352
|
-
# 8192-token budget at 2.0 chars/token, so we budget at 1.5 to stay
|
|
353
|
-
# clear of tokenizer surprises even on the densest Rails internals.
|
|
354
|
-
#
|
|
355
|
-
# `max_tokens` tracks the provider's actual input budget when it
|
|
356
|
-
# reports one, falling back to the TextPreparer default otherwise.
|
|
357
|
-
#
|
|
358
|
-
# @param provider [Embedding::Provider::Interface]
|
|
359
|
-
# @return [Embedding::TextPreparer]
|
|
342
|
+
# Build a preparer with the provider's explicit input-counting policy.
|
|
343
|
+
# Ratios remain estimates; known OpenAI models use a byte-BPE upper bound.
|
|
360
344
|
def build_text_preparer(provider)
|
|
361
345
|
chars_per_token = chars_per_token_for(provider)
|
|
362
346
|
budget = safe_max_input_tokens(provider)
|
|
363
347
|
max_tokens = budget || Embedding::TextPreparer::DEFAULT_MAX_TOKENS
|
|
364
348
|
|
|
365
|
-
Embedding::
|
|
349
|
+
input_budget = Embedding::InputBudget.for(provider, limit: max_tokens, chars_per_token: chars_per_token)
|
|
350
|
+
Embedding::TextPreparer.new(max_tokens: max_tokens, chars_per_token: chars_per_token, input_budget: input_budget)
|
|
366
351
|
end
|
|
367
352
|
|
|
368
|
-
#
|
|
369
|
-
#
|
|
370
|
-
#
|
|
371
|
-
# matching chars-per-token ratio, minus the context-prefix
|
|
372
|
-
# allowance the Indexer accounts for separately. Units that exceed
|
|
373
|
-
# this ceiling get sliced so no single chunk can blow the provider's
|
|
374
|
-
# input cap.
|
|
375
|
-
#
|
|
376
|
-
# For Ollama (and other BERT/WordPiece-backed models), char-based
|
|
377
|
-
# estimation is unreliable — CamelCase, `::` separators, and symbol
|
|
378
|
-
# literals tokenize much denser than chars/token averages suggest.
|
|
379
|
-
# When the optional `tokenizers` gem is installed, pass a
|
|
380
|
-
# {Embedding::TokenCounter} and `max_tokens` so the chunker can
|
|
381
|
-
# verify every slice with the real tokenizer and re-split any piece
|
|
382
|
-
# that still exceeds `num_ctx`. See docs/EMBEDDING_MODELS.md.
|
|
383
|
-
#
|
|
384
|
-
# Ollama v0.13.5+ stopped honouring `truncate: true` on `/api/embed`
|
|
385
|
-
# (ollama/ollama#14186), so any chunk that exceeds `num_ctx` returns
|
|
386
|
-
# a 400 rather than being silently truncated. Exact client-side
|
|
387
|
-
# sizing is the only reliable path until the regression is fixed
|
|
388
|
-
# upstream.
|
|
389
|
-
#
|
|
390
|
-
# @param provider [Embedding::Provider::Interface]
|
|
391
|
-
# @return [Chunking::SemanticChunker]
|
|
353
|
+
# Initial semantic sizing is approximate. TextPreparer subsequently checks
|
|
354
|
+
# every complete prefix plus source against the provider policy and splits
|
|
355
|
+
# losslessly; no fixed prefix allowance establishes final admission.
|
|
392
356
|
def build_chunker(provider)
|
|
393
357
|
budget = safe_max_input_tokens(provider)
|
|
394
358
|
max_chars = ((budget * chars_per_token_for(provider)).floor - CHUNKER_PREFIX_ALLOWANCE if budget)
|
|
@@ -424,12 +388,8 @@ module Woods
|
|
|
424
388
|
|
|
425
389
|
private
|
|
426
390
|
|
|
427
|
-
#
|
|
428
|
-
#
|
|
429
|
-
# chars/token on code, so it doesn't need this.
|
|
430
|
-
#
|
|
431
|
-
# @param provider [Embedding::Provider::Interface]
|
|
432
|
-
# @return [Embedding::TokenCounter, nil]
|
|
391
|
+
# Ollama has no universal local tokenizer. This counter is an explicitly
|
|
392
|
+
# approximate sizing hint; truncate:false enforces the server's actual cap.
|
|
433
393
|
def token_counter_for(provider)
|
|
434
394
|
return unless unwrap_provider(provider).is_a?(Embedding::Provider::Ollama)
|
|
435
395
|
|
|
@@ -109,6 +109,9 @@ module Woods
|
|
|
109
109
|
class CachedEmbeddingProvider
|
|
110
110
|
include Embedding::Provider::Interface
|
|
111
111
|
|
|
112
|
+
# @return [Object] underlying provider, including any resilience wrapper
|
|
113
|
+
attr_reader :provider
|
|
114
|
+
|
|
112
115
|
# @param provider [Embedding::Provider::Interface] The real embedding provider
|
|
113
116
|
# @param cache_store [CacheStore] Cache backend instance
|
|
114
117
|
# @param ttl [Integer] TTL for cached embeddings in seconds
|
|
@@ -118,6 +121,10 @@ module Woods
|
|
|
118
121
|
@ttl = ttl
|
|
119
122
|
@inflight = {}
|
|
120
123
|
@inflight_mutex = Mutex.new
|
|
124
|
+
identity = @provider.cache_identity if @provider.respond_to?(:cache_identity)
|
|
125
|
+
# Unknown custom providers cannot prove cross-instance compatibility.
|
|
126
|
+
# Hash every identity so endpoint credentials never enter backend keys.
|
|
127
|
+
@embedding_identity = Digest::SHA256.hexdigest(JSON.generate(identity || SecureRandom.hex(16)))
|
|
121
128
|
end
|
|
122
129
|
|
|
123
130
|
# Embed a single text, returning a cached vector when available.
|
|
@@ -129,7 +136,7 @@ module Woods
|
|
|
129
136
|
# @param text [String] Text to embed
|
|
130
137
|
# @return [Array<Float>] Embedding vector
|
|
131
138
|
def embed(text)
|
|
132
|
-
cached =
|
|
139
|
+
cached = read_cached(text)
|
|
133
140
|
return cached unless cached.nil?
|
|
134
141
|
|
|
135
142
|
with_single_flight(text) { @provider.embed(text) }
|
|
@@ -148,13 +155,17 @@ module Woods
|
|
|
148
155
|
# @return [Array<Array<Float>>] Embedding vectors (same order as input)
|
|
149
156
|
def embed_batch(texts)
|
|
150
157
|
results, misses, miss_indices = partition_cached(texts)
|
|
151
|
-
|
|
158
|
+
if misses.empty?
|
|
159
|
+
validate_vectors!(results, texts.size)
|
|
160
|
+
return results
|
|
161
|
+
end
|
|
152
162
|
|
|
153
163
|
to_fetch, to_fetch_positions, our_entries, awaiting = claim_inflight(misses)
|
|
154
164
|
|
|
155
165
|
fetch_and_fulfill(to_fetch, to_fetch_positions, our_entries, results, miss_indices)
|
|
156
166
|
await_others(awaiting, results, miss_indices)
|
|
157
167
|
|
|
168
|
+
validate_vectors!(results, texts.size)
|
|
158
169
|
results
|
|
159
170
|
end
|
|
160
171
|
|
|
@@ -172,6 +183,23 @@ module Woods
|
|
|
172
183
|
@provider.model_name
|
|
173
184
|
end
|
|
174
185
|
|
|
186
|
+
# Preserve pure configuration declarations through this wrapper.
|
|
187
|
+
def cache_identity
|
|
188
|
+
@provider.cache_identity if @provider.respond_to?(:cache_identity)
|
|
189
|
+
end
|
|
190
|
+
|
|
191
|
+
def input_budget
|
|
192
|
+
@provider.input_budget if @provider.respond_to?(:input_budget)
|
|
193
|
+
end
|
|
194
|
+
|
|
195
|
+
def configured_dimensions
|
|
196
|
+
@provider.configured_dimensions if @provider.respond_to?(:configured_dimensions)
|
|
197
|
+
end
|
|
198
|
+
|
|
199
|
+
def requested_dimensions
|
|
200
|
+
@provider.requested_dimensions if @provider.respond_to?(:requested_dimensions)
|
|
201
|
+
end
|
|
202
|
+
|
|
175
203
|
# Delegate the per-provider input cap so Builder's chunker / text
|
|
176
204
|
# preparer wiring keeps working when the cache wrapper is in front
|
|
177
205
|
# of the provider. Without this, `respond_to?(:max_input_tokens)`
|
|
@@ -201,6 +229,7 @@ module Woods
|
|
|
201
229
|
|
|
202
230
|
begin
|
|
203
231
|
vector = yield
|
|
232
|
+
validate_vectors!([vector], 1)
|
|
204
233
|
write_cache(text, vector)
|
|
205
234
|
entry.fulfill(vector)
|
|
206
235
|
vector
|
|
@@ -284,13 +313,7 @@ module Woods
|
|
|
284
313
|
|
|
285
314
|
begin
|
|
286
315
|
fresh_vectors = @provider.embed_batch(to_fetch)
|
|
287
|
-
|
|
288
|
-
# fulfilling waiters with `nil` (or masking a missing tail vector by
|
|
289
|
-
# under-writing the cache).
|
|
290
|
-
if fresh_vectors.size != to_fetch.size
|
|
291
|
-
raise ArgumentError,
|
|
292
|
-
"provider returned #{fresh_vectors.size} vectors for #{to_fetch.size} texts"
|
|
293
|
-
end
|
|
316
|
+
validate_fresh_vectors!(fresh_vectors, to_fetch.size)
|
|
294
317
|
rescue StandardError => e
|
|
295
318
|
our_entries.each { |entry| entry.reject(e) }
|
|
296
319
|
raise
|
|
@@ -351,7 +374,7 @@ module Woods
|
|
|
351
374
|
miss_indices = []
|
|
352
375
|
|
|
353
376
|
texts.each_with_index do |text, idx|
|
|
354
|
-
cached =
|
|
377
|
+
cached = read_cached(text)
|
|
355
378
|
if cached
|
|
356
379
|
results[idx] = cached
|
|
357
380
|
else
|
|
@@ -363,29 +386,32 @@ module Woods
|
|
|
363
386
|
[results, misses, miss_indices]
|
|
364
387
|
end
|
|
365
388
|
|
|
366
|
-
|
|
367
|
-
|
|
368
|
-
|
|
369
|
-
|
|
370
|
-
|
|
371
|
-
|
|
372
|
-
#
|
|
373
|
-
|
|
374
|
-
|
|
375
|
-
|
|
376
|
-
|
|
377
|
-
|
|
378
|
-
|
|
379
|
-
|
|
380
|
-
|
|
381
|
-
|
|
382
|
-
|
|
383
|
-
|
|
384
|
-
|
|
389
|
+
def read_cached(text)
|
|
390
|
+
vector = @cache_store.read(embedding_key(text))
|
|
391
|
+
validate_vectors!([vector], 1) unless vector.nil?
|
|
392
|
+
vector
|
|
393
|
+
end
|
|
394
|
+
|
|
395
|
+
# Validate the complete batch before writing or fulfilling any entry.
|
|
396
|
+
def validate_fresh_vectors!(vectors, count)
|
|
397
|
+
raise ArgumentError, "provider returned #{vectors.size} vectors for #{count} texts" if vectors.size != count
|
|
398
|
+
|
|
399
|
+
validate_vectors!(vectors, count)
|
|
400
|
+
end
|
|
401
|
+
|
|
402
|
+
def validate_vectors!(vectors, count)
|
|
403
|
+
width = @provider.configured_dimensions if @provider.respond_to?(:configured_dimensions)
|
|
404
|
+
Embedding::Provider::VectorValidation.validate!(
|
|
405
|
+
vectors, expected_count: count, provider: 'CachedEmbeddingProvider', expected_dimensions: width
|
|
406
|
+
)
|
|
407
|
+
end
|
|
408
|
+
|
|
409
|
+
# Configuration identity is captured without probing dimensions. Version
|
|
410
|
+
# the domain to prevent reuse of old model-only keys after an upgrade.
|
|
385
411
|
# @param text [String]
|
|
386
412
|
# @return [String]
|
|
387
413
|
def embedding_key(text)
|
|
388
|
-
Cache.cache_key(:embeddings, @
|
|
414
|
+
Cache.cache_key(:embeddings, 'configuration-v1', @embedding_identity, Digest::SHA256.hexdigest(text))
|
|
389
415
|
end
|
|
390
416
|
end
|
|
391
417
|
|
|
@@ -0,0 +1,119 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require_relative 'chunk'
|
|
4
|
+
require_relative '../source_contributors'
|
|
5
|
+
|
|
6
|
+
module Woods
|
|
7
|
+
module Chunking
|
|
8
|
+
# Exact physical attribution for contiguous slices of validated contributors.
|
|
9
|
+
# Generated composite headers and cross-file content never get a physical span.
|
|
10
|
+
module ContributorChunks
|
|
11
|
+
module_function
|
|
12
|
+
|
|
13
|
+
def chunks(unit)
|
|
14
|
+
SourceContributors.records(unit).each_with_index.map do |record, index|
|
|
15
|
+
first = record.fetch('published_start_byte')
|
|
16
|
+
last = record.fetch('published_end_byte')
|
|
17
|
+
Chunk.new(content: unit.source_code.byteslice(first...last), chunk_type: :"contributor_#{index}",
|
|
18
|
+
parent_identifier: unit.identifier, parent_type: unit.type,
|
|
19
|
+
metadata: location(unit, first, last))
|
|
20
|
+
end
|
|
21
|
+
end
|
|
22
|
+
|
|
23
|
+
# Preserve existing bounded chunks only if they cover every raw byte once,
|
|
24
|
+
# in contributor order. Caller-supplied ranges never authorize citations.
|
|
25
|
+
def ensure!(unit)
|
|
26
|
+
records = SourceContributors.records(unit)
|
|
27
|
+
return if records.empty?
|
|
28
|
+
|
|
29
|
+
unit.chunks = chunks(unit).map(&:to_h) unless complete_coverage?(unit, records)
|
|
30
|
+
unit.chunks.each do |chunk|
|
|
31
|
+
range = published_range(chunk)
|
|
32
|
+
chunk[:metadata] = (chunk[:metadata] || {}).merge(location(unit, range[:start_byte], range[:end_byte]))
|
|
33
|
+
end
|
|
34
|
+
end
|
|
35
|
+
|
|
36
|
+
def complete_coverage?(unit, records)
|
|
37
|
+
remaining = unit.chunks.dup
|
|
38
|
+
records.all? do |record|
|
|
39
|
+
cursor = record['published_start_byte']
|
|
40
|
+
while cursor < record['published_end_byte']
|
|
41
|
+
chunk = remaining.shift
|
|
42
|
+
return false unless chunk
|
|
43
|
+
|
|
44
|
+
span = published_range(chunk)
|
|
45
|
+
return false unless contiguous?(unit, chunk, span, cursor, record['published_end_byte'])
|
|
46
|
+
|
|
47
|
+
cursor = span[:end_byte]
|
|
48
|
+
end
|
|
49
|
+
true
|
|
50
|
+
end && remaining.empty?
|
|
51
|
+
end
|
|
52
|
+
|
|
53
|
+
def contiguous?(unit, chunk, span, cursor, last)
|
|
54
|
+
span[:start_byte] == cursor && span[:end_byte].is_a?(Integer) &&
|
|
55
|
+
span[:end_byte] > cursor && span[:end_byte] <= last &&
|
|
56
|
+
unit.source_code.byteslice(cursor...span[:end_byte]) == chunk[:content]
|
|
57
|
+
end
|
|
58
|
+
|
|
59
|
+
def published_range(chunk)
|
|
60
|
+
metadata = SourceContributors.field(chunk, :metadata) || {}
|
|
61
|
+
span = SourceContributors.field(metadata, :published_location)
|
|
62
|
+
span.is_a?(Hash) ? span.transform_keys(&:to_sym) : {}
|
|
63
|
+
end
|
|
64
|
+
|
|
65
|
+
def location(unit, first, last)
|
|
66
|
+
physical = SourceContributors.physical_span(unit, start_byte: first, end_byte: last)
|
|
67
|
+
return {} unless physical
|
|
68
|
+
|
|
69
|
+
record = SourceContributors.records(unit).find { |entry| entry['file_path'] == physical[:file_path] }
|
|
70
|
+
physical = physical.merge(start_byte: first - record['published_start_byte'],
|
|
71
|
+
end_byte: last - record['published_start_byte'])
|
|
72
|
+
{ physical_location: physical, published_location: { start_byte: first, end_byte: last } }
|
|
73
|
+
end
|
|
74
|
+
|
|
75
|
+
# Offsets are relative to this chunk, including for a second splitting pass.
|
|
76
|
+
def slice_metadata(metadata, source, first, last)
|
|
77
|
+
metadata = symbol_keys(metadata)
|
|
78
|
+
physical = symbol_keys(metadata[:physical_location])
|
|
79
|
+
published = symbol_keys(metadata[:published_location])
|
|
80
|
+
clean = metadata.except(:physical_location, :published_location)
|
|
81
|
+
return clean unless mapped_slice?(physical, published, source, first, last)
|
|
82
|
+
|
|
83
|
+
clean.merge(physical_location: physical_slice(physical, source, first, last),
|
|
84
|
+
published_location: { start_byte: published[:start_byte] + first,
|
|
85
|
+
end_byte: published[:start_byte] + last })
|
|
86
|
+
end
|
|
87
|
+
|
|
88
|
+
def physical_slice(physical, source, first, last)
|
|
89
|
+
start_line = physical[:start_line] + source.byteslice(0...first).count("\n")
|
|
90
|
+
end_line = start_line + source.byteslice(first...last).delete_suffix("\n").count("\n")
|
|
91
|
+
physical.merge(start_byte: physical[:start_byte] + first, end_byte: physical[:start_byte] + last,
|
|
92
|
+
start_line: start_line, end_line: end_line)
|
|
93
|
+
end
|
|
94
|
+
|
|
95
|
+
def symbol_keys(value)
|
|
96
|
+
value.is_a?(Hash) ? value.transform_keys(&:to_sym) : {}
|
|
97
|
+
end
|
|
98
|
+
|
|
99
|
+
def mapped_slice?(physical, published, source, first, last)
|
|
100
|
+
integers = [physical[:start_byte], physical[:end_byte], physical[:start_line],
|
|
101
|
+
published[:start_byte], published[:end_byte]]
|
|
102
|
+
integers.all?(Integer) &&
|
|
103
|
+
[physical, published].all? { |span| span[:end_byte] - span[:start_byte] == source.bytesize } &&
|
|
104
|
+
first >= 0 && last > first && last <= source.bytesize
|
|
105
|
+
end
|
|
106
|
+
|
|
107
|
+
# Re-derive spans from the published source when hydrating a metadata dump.
|
|
108
|
+
def vector_metadata(unit, chunk_metadata)
|
|
109
|
+
return {} unless SourceContributors.multiple?(unit)
|
|
110
|
+
|
|
111
|
+
span = published_range(metadata: chunk_metadata)
|
|
112
|
+
facts = location(unit, span[:start_byte], span[:end_byte])
|
|
113
|
+
{ source_paths: SourceContributors.paths(unit),
|
|
114
|
+
file_path: facts.dig(:physical_location, :file_path) }.merge(facts)
|
|
115
|
+
end
|
|
116
|
+
private_class_method :complete_coverage?, :contiguous?, :mapped_slice?, :physical_slice, :symbol_keys
|
|
117
|
+
end
|
|
118
|
+
end
|
|
119
|
+
end
|