woods 2.0.1 → 2.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +94 -7
- data/CONTRIBUTING.md +134 -19
- data/README.md +1 -1
- data/docs/AGENT_GUIDE.md +19 -0
- data/docs/AGENT_SETUP.md +22 -2
- data/docs/BACKEND_MATRIX.md +7 -0
- data/docs/CLIENT_HOOKS.md +6 -0
- data/docs/CONFIGURATION_REFERENCE.md +133 -25
- data/docs/CONSOLE_MCP_SETUP.md +82 -30
- data/docs/EMBEDDING_MODELS.md +16 -19
- data/docs/EXTRACTOR_REFERENCE.md +219 -21
- data/docs/FAQ.md +11 -25
- data/docs/GETTING_STARTED.md +7 -1
- data/docs/INCREMENTAL_EXTRACTION.md +261 -19
- data/docs/INDEX_LAYOUT.md +5 -0
- data/docs/INTERNALS.md +9 -0
- data/docs/MCP_HTTP_TRANSPORT.md +20 -15
- data/docs/MCP_SERVERS.md +87 -8
- data/docs/MCP_TOOL_COOKBOOK.md +13 -55
- data/docs/NOTION_INTEGRATION.md +7 -1
- data/docs/PUBLISHED_INDEX.md +6 -0
- data/docs/README.md +6 -1
- data/docs/RETRIEVAL_GUIDE.md +17 -0
- data/docs/SOURCE_FRESHNESS.md +157 -5
- data/docs/TOKEN_BENCHMARK.md +10 -18
- data/docs/TROUBLESHOOTING.md +70 -14
- data/docs/UNBLOCKED_INTEGRATION.md +60 -8
- data/docs/UPGRADING_TO_2.md +153 -38
- data/docs/WATCH_DAEMON.md +97 -14
- data/exe/woods-console-mcp +2 -2
- data/lib/generators/woods/templates/woods.rb.tt +2 -1
- data/lib/tasks/woods.rake +23 -7
- data/lib/tasks/woods_checks.rake +2 -2
- data/lib/woods/agent_configuration/cli.rb +1 -1
- data/lib/woods/agent_configuration/layout.rb +16 -2
- data/lib/woods/agent_configuration/plan.rb +13 -3
- data/lib/woods/agent_configuration/planner_validation.rb +4 -2
- data/lib/woods/agent_configuration/preflight.rb +5 -3
- data/lib/woods/builder.rb +17 -57
- data/lib/woods/cache/cache_middleware.rb +56 -30
- data/lib/woods/chunking/contributor_chunks.rb +119 -0
- data/lib/woods/chunking/semantic_chunker.rb +44 -21
- data/lib/woods/console/connection_manager.rb +56 -3
- data/lib/woods/console/embedded_executor.rb +30 -5
- data/lib/woods/console/rack_middleware.rb +29 -1
- data/lib/woods/dependency_graph.rb +34 -10
- data/lib/woods/embedding/fake.rb +12 -0
- data/lib/woods/embedding/indexer.rb +195 -98
- data/lib/woods/embedding/input_budget.rb +67 -0
- data/lib/woods/embedding/openai.rb +70 -20
- data/lib/woods/embedding/provider.rb +37 -25
- data/lib/woods/embedding/text_preparer.rb +76 -32
- data/lib/woods/embedding/token_counter.rb +18 -81
- data/lib/woods/embedding/vector_configuration.rb +48 -0
- data/lib/woods/extraction_identities.rb +175 -0
- data/lib/woods/extractor.rb +304 -107
- data/lib/woods/extractors/action_cable_extractor.rb +8 -3
- data/lib/woods/extractors/assigned_value_discovery.rb +74 -0
- data/lib/woods/extractors/class_declarations.rb +121 -0
- data/lib/woods/extractors/configuration_extractor.rb +11 -3
- data/lib/woods/extractors/declaration_ancestry.rb +92 -0
- data/lib/woods/extractors/event_extractor.rb +8 -0
- data/lib/woods/extractors/graphql_extractor.rb +134 -77
- data/lib/woods/extractors/job_extractor.rb +5 -1
- data/lib/woods/extractors/lib_extractor.rb +132 -15
- data/lib/woods/extractors/mailer_extractor.rb +3 -5
- data/lib/woods/extractors/manager_extractor.rb +7 -21
- data/lib/woods/extractors/migration_declaration.rb +87 -0
- data/lib/woods/extractors/migration_extractor.rb +5 -39
- data/lib/woods/extractors/phlex_extractor.rb +6 -2
- data/lib/woods/extractors/policy_extractor.rb +9 -5
- data/lib/woods/extractors/poro_extractor.rb +112 -53
- data/lib/woods/extractors/pundit_extractor.rb +11 -6
- data/lib/woods/extractors/scheduled_job_extractor.rb +45 -4
- data/lib/woods/extractors/serializer_extractor.rb +34 -22
- data/lib/woods/extractors/shared_utility_methods.rb +18 -1
- data/lib/woods/extractors/source_nesting.rb +142 -106
- data/lib/woods/extractors/standalone_module_discovery.rb +123 -0
- data/lib/woods/extractors/state_machine_extractor.rb +46 -40
- data/lib/woods/extractors/view_component_extractor.rb +9 -7
- data/lib/woods/flow_assembler.rb +4 -1
- data/lib/woods/generation.rb +25 -0
- data/lib/woods/hooks/context_hint.rb +7 -2
- data/lib/woods/mcp/bootstrapper.rb +33 -7
- data/lib/woods/mcp/config_resolver.rb +26 -7
- data/lib/woods/mcp/index_reader.rb +125 -24
- data/lib/woods/mcp/index_reader_pinning.rb +16 -0
- data/lib/woods/mcp/renderers/markdown_renderer.rb +7 -1
- data/lib/woods/mcp/renderers/plain_renderer.rb +3 -1
- data/lib/woods/mcp/search_results.rb +7 -1
- data/lib/woods/mcp/server.rb +24 -4
- data/lib/woods/module_reconciliation.rb +151 -0
- data/lib/woods/path_dispatcher.rb +7 -2
- data/lib/woods/rake_helpers.rb +43 -11
- data/lib/woods/release.rb +1 -1
- data/lib/woods/resilience/index_validator.rb +8 -3
- data/lib/woods/resilience/retryable_provider.rb +18 -1
- data/lib/woods/resolved_config.rb +68 -8
- data/lib/woods/retrieval/context_assembler.rb +3 -3
- data/lib/woods/retrieval/lexical_assembler.rb +3 -2
- data/lib/woods/retrieval/scope.rb +18 -2
- data/lib/woods/retrieval/source_evidence.rb +14 -2
- data/lib/woods/source_contributor_validation.rb +78 -0
- data/lib/woods/source_contributors.rb +116 -0
- data/lib/woods/source_inputs/handoff.rb +37 -0
- data/lib/woods/source_inputs/launcher.rb +53 -13
- data/lib/woods/source_inputs/manifest.rb +84 -3
- data/lib/woods/source_inputs/private_key.rb +44 -12
- data/lib/woods/source_inputs/scanner.rb +98 -27
- data/lib/woods/source_inputs/scopes.rb +1 -1
- data/lib/woods/source_inputs/session.rb +147 -15
- data/lib/woods/source_inputs/stable_reader.rb +127 -0
- data/lib/woods/source_inputs/status.rb +40 -8
- data/lib/woods/source_inputs/verifier.rb +28 -5
- data/lib/woods/source_path_encoding.rb +33 -0
- data/lib/woods/source_references/cache.rb +284 -0
- data/lib/woods/source_references/collector.rb +120 -0
- data/lib/woods/source_references/extraction.rb +185 -0
- data/lib/woods/source_references/inputs.rb +134 -0
- data/lib/woods/source_references/parser_adapter.rb +134 -0
- data/lib/woods/source_references/pass.rb +152 -0
- data/lib/woods/source_references/prism_adapter.rb +116 -0
- data/lib/woods/source_references/registry.rb +178 -0
- data/lib/woods/source_references/runtime_lookup.rb +127 -0
- data/lib/woods/source_references/value_class.rb +82 -0
- data/lib/woods/storage/metadata_store.rb +4 -1
- data/lib/woods/storage/qdrant.rb +2 -2
- data/lib/woods/unblocked/client.rb +12 -7
- data/lib/woods/unblocked/document_builder.rb +4 -1
- data/lib/woods/unblocked/exporter.rb +127 -37
- data/lib/woods/unblocked/sync_manifest.rb +137 -21
- data/lib/woods/unblocked/uri_migration.rb +105 -0
- data/lib/woods/util/host_guard.rb +3 -2
- data/lib/woods/version.rb +1 -1
- data/lib/woods/watch/catch_up.rb +138 -0
- data/lib/woods/watch/claim_lease.rb +150 -0
- data/lib/woods/watch/cli.rb +26 -2
- data/lib/woods/watch/daemon.rb +80 -59
- data/lib/woods/watch/installation/options.rb +1 -1
- data/lib/woods/watch/installation/receipt.rb +6 -1
- data/lib/woods/watch/managed_child.rb +1 -1
- data/lib/woods/watch/supervisor.rb +1 -1
- data/lib/woods/watch/tree_scan.rb +14 -2
- data/plugin/.claude-plugin/plugin.json +1 -1
- data/plugin/hooks/adapters/normalize.rb +3 -2
- data/plugin/hooks/woods-input-rules.sh +4 -0
- data/plugin/hooks/woods-refresh.sh +15 -7
- data/plugin/hooks/woods-session-start.sh +60 -3
- data/plugin/skills/woods-diagnose/SKILL.md +334 -11
- data/plugin/skills/woods-investigate/SKILL.md +11 -0
- data/plugin/skills/woods-mcp-config/SKILL.md +79 -8
- data/plugin/skills/woods-setup/SKILL.md +53 -6
- metadata +32 -5
|
@@ -5,6 +5,7 @@ require 'digest'
|
|
|
5
5
|
require 'fileutils'
|
|
6
6
|
require 'set'
|
|
7
7
|
|
|
8
|
+
require_relative 'input_budget'
|
|
8
9
|
require_relative '../atomic_file'
|
|
9
10
|
require_relative '../storage_identity'
|
|
10
11
|
require_relative '../generation'
|
|
@@ -64,9 +65,10 @@ module Woods
|
|
|
64
65
|
# into semantically coherent chunks before embedding. +nil+ disables
|
|
65
66
|
# chunking — units go to the provider whole (useful in tests).
|
|
66
67
|
# @param checkpoint_interval [Integer] Save checkpoint every N batches (default: 10)
|
|
67
|
-
# @param metadata_store [
|
|
68
|
-
#
|
|
69
|
-
#
|
|
68
|
+
# @param metadata_store [Storage::MetadataStore::Interface, nil] Optional metadata store.
|
|
69
|
+
# Existing identities participate in snapshot-store reconciliation.
|
|
70
|
+
# Stores with #each_entry and #bulk_load are persisted alongside vectors;
|
|
71
|
+
# SQLite retains its records in the configured database instead.
|
|
70
72
|
# @param resolved_config [Woods::ResolvedConfig, nil] Captured config for
|
|
71
73
|
# +woods.json+ — written to +output_dir+ on {#index_all} completion.
|
|
72
74
|
# @param dump_retention_count [Integer] Number of completed dump directories
|
|
@@ -152,6 +154,7 @@ module Woods
|
|
|
152
154
|
prepare_run(incremental: incremental)
|
|
153
155
|
checkpoint = incremental ? load_checkpoint : {}
|
|
154
156
|
units = assign_storage_identities(units, checkpoint: checkpoint)
|
|
157
|
+
preflight_inputs(units, checkpoint, incremental: incremental)
|
|
155
158
|
stats = { processed: 0, skipped: 0, errors: 0 }
|
|
156
159
|
|
|
157
160
|
embed_batches(units, checkpoint, stats, incremental: incremental)
|
|
@@ -170,6 +173,15 @@ module Woods
|
|
|
170
173
|
stats
|
|
171
174
|
end
|
|
172
175
|
|
|
176
|
+
# Reject deterministic input failures before metadata or durable writes.
|
|
177
|
+
def preflight_inputs(units, checkpoint, incremental:)
|
|
178
|
+
units.each { |unit| prepared_fingerprint(unit) }
|
|
179
|
+
return unless reconcilable? && @durable_ids.nil?
|
|
180
|
+
return if incremental && units.all? { |unit| checkpoint_satisfied?(unit, checkpoint) }
|
|
181
|
+
|
|
182
|
+
raise InputLimitError, 'Cannot replace embedding inputs: existing durable vector IDs could not be read'
|
|
183
|
+
end
|
|
184
|
+
|
|
173
185
|
# Unambiguous existing keys stay stable. A collision uses reversible typed keys.
|
|
174
186
|
def assign_storage_identities(units, checkpoint:)
|
|
175
187
|
counts = units.group_by { |unit| unit['identifier'] }.transform_values(&:size)
|
|
@@ -474,6 +486,11 @@ module Woods
|
|
|
474
486
|
@metadata_changed = false
|
|
475
487
|
@vectors_changed = false
|
|
476
488
|
@empty_units = {}
|
|
489
|
+
@prepared_texts = {}
|
|
490
|
+
@prepared_chunks = {}
|
|
491
|
+
@prepared_inputs = {}
|
|
492
|
+
@checkpoint_inputs = {}
|
|
493
|
+
@unknown_reconciliation_warned = false
|
|
477
494
|
@persisted_metadata = nil
|
|
478
495
|
prepare_snapshot_stores(incremental: incremental)
|
|
479
496
|
retain_metadata_identities
|
|
@@ -492,14 +509,17 @@ module Woods
|
|
|
492
509
|
entries = []
|
|
493
510
|
@vector_store.each_entry { |id, _vector, _metadata| entries << { id: id } }
|
|
494
511
|
@persisted_ids = index_ids_by_identifier(entries)
|
|
495
|
-
retain_existing_metadata_identities
|
|
496
512
|
end
|
|
513
|
+
retain_existing_metadata_identities
|
|
497
514
|
end
|
|
498
515
|
|
|
499
|
-
def retain_existing_metadata_identities
|
|
500
|
-
|
|
501
|
-
|
|
502
|
-
@metadata_store.each_entry
|
|
516
|
+
def retain_existing_metadata_identities(identities = @persisted_ids)
|
|
517
|
+
if implements_own?(@metadata_store, :all_identifiers)
|
|
518
|
+
@metadata_store.all_identifiers.each { |identifier| identities[identifier] ||= [] }
|
|
519
|
+
elsif @metadata_store.respond_to?(:each_entry)
|
|
520
|
+
# Compatibility for custom snapshot-only stores.
|
|
521
|
+
@metadata_store.each_entry { |identifier, _unit| identities[identifier] ||= [] }
|
|
522
|
+
end
|
|
503
523
|
end
|
|
504
524
|
|
|
505
525
|
# Source-empty units retain metadata but intentionally have no vectors.
|
|
@@ -524,6 +544,7 @@ module Woods
|
|
|
524
544
|
def load_durable_store_ids
|
|
525
545
|
@durable_ids = Hash.new { |hash, key| hash[key] = [] }
|
|
526
546
|
@vector_store.each_id { |id| @durable_ids[base_identifier(id)] << id }
|
|
547
|
+
retain_existing_metadata_identities(@durable_ids)
|
|
527
548
|
rescue StandardError => e
|
|
528
549
|
# A store that cannot be enumerated must not take the embed run down
|
|
529
550
|
# with it. Reconciliation and the presence check both degrade to their
|
|
@@ -609,11 +630,9 @@ module Woods
|
|
|
609
630
|
|
|
610
631
|
# May this unit's embedding be skipped?
|
|
611
632
|
#
|
|
612
|
-
# Incremental skip
|
|
613
|
-
#
|
|
614
|
-
#
|
|
615
|
-
# unit_data JSON — so key ordering or whitespace in the _index.json
|
|
616
|
-
# does not invalidate checkpoints across Ruby-minor upgrades.
|
|
633
|
+
# Incremental skip requires both the source hash and the complete ordered
|
|
634
|
+
# prepared-input fingerprint. Dependency/path/namespace/chunk changes
|
|
635
|
+
# invalidate the latter even if source_code did not change.
|
|
617
636
|
#
|
|
618
637
|
# A matching hash is necessary but not sufficient: the vector must also
|
|
619
638
|
# actually exist. On the dump-backed path that means present in what we
|
|
@@ -626,17 +645,17 @@ module Woods
|
|
|
626
645
|
# unit permanently: checkpoint.json said "embedded", the new store held
|
|
627
646
|
# nothing, and no subsequent incremental run ever disagreed.
|
|
628
647
|
def checkpoint_satisfied?(unit_data, checkpoint)
|
|
629
|
-
|
|
648
|
+
identifier = storage_id(unit_data)
|
|
649
|
+
return false unless checkpoint[identifier] == unit_data['source_hash']
|
|
650
|
+
return false unless @checkpoint_inputs[identifier] == prepared_fingerprint(unit_data)
|
|
630
651
|
|
|
631
652
|
known_ids = persistable? ? @persisted_ids : @durable_ids
|
|
632
653
|
# No durable view to check against (an adapter with no #each_id, or an
|
|
633
654
|
# enumeration that failed) — fall back to trusting the checkpoint.
|
|
634
655
|
return true if known_ids.nil?
|
|
635
656
|
|
|
636
|
-
|
|
637
|
-
|
|
638
|
-
# again rather than treating a missing nonempty vector as a cache hit.
|
|
639
|
-
return true if prepare_texts(unit_data).empty?
|
|
657
|
+
expected = embedding_ids(identifier, prepared_texts(unit_data).size)
|
|
658
|
+
return true if Array(known_ids[identifier]).sort == expected.sort
|
|
640
659
|
|
|
641
660
|
@checkpoint_misses += 1
|
|
642
661
|
false
|
|
@@ -654,8 +673,10 @@ module Woods
|
|
|
654
673
|
return unless @metadata_store
|
|
655
674
|
|
|
656
675
|
id = storage_id(unit_data)
|
|
657
|
-
|
|
658
|
-
|
|
676
|
+
chunks = @prepared_chunks[id]
|
|
677
|
+
data = chunks ? unit_data.merge('embedding_chunks' => JSON.parse(JSON.generate(chunks))) : unit_data
|
|
678
|
+
@metadata_changed ||= @persisted_metadata && @persisted_metadata.find(id) != data
|
|
679
|
+
@metadata_store.store(id, data)
|
|
659
680
|
end
|
|
660
681
|
|
|
661
682
|
# Compare with the promoted artifact, not a fresh metadata store or the
|
|
@@ -684,21 +705,22 @@ module Woods
|
|
|
684
705
|
end
|
|
685
706
|
|
|
686
707
|
def collect_embed_items(unit_data, items)
|
|
687
|
-
texts =
|
|
708
|
+
texts = prepared_texts(unit_data)
|
|
688
709
|
identifier = storage_id(unit_data)
|
|
689
710
|
@empty_units[identifier] = unit_data['source_hash'] if texts.empty?
|
|
690
711
|
|
|
691
712
|
texts.each_with_index do |text, idx|
|
|
692
713
|
embed_id = texts.length > 1 ? "#{identifier}#chunk_#{idx}" : identifier
|
|
693
714
|
items << { id: embed_id, text: text, unit_data: unit_data,
|
|
694
|
-
source_hash: unit_data['source_hash'], identifier: identifier
|
|
715
|
+
source_hash: unit_data['source_hash'], identifier: identifier,
|
|
716
|
+
chunk_metadata: @prepared_chunks[identifier]&.[](idx) }
|
|
695
717
|
end
|
|
696
718
|
end
|
|
697
719
|
|
|
698
720
|
# Defer zero-text deletion until every batch has prepared/embedded
|
|
699
721
|
# successfully. A later provider failure must not retire earlier vectors.
|
|
700
|
-
#
|
|
701
|
-
#
|
|
722
|
+
# The no-vector checkpoint also records the complete-input fingerprint;
|
|
723
|
+
# it advances only after obsolete vectors have been reconciled.
|
|
702
724
|
def reconcile_empty_units(checkpoint)
|
|
703
725
|
verify_empty_reconciliation!
|
|
704
726
|
@empty_units.each do |identifier, source_hash|
|
|
@@ -706,6 +728,7 @@ module Woods
|
|
|
706
728
|
prune_identifier(identifier, []) if implements_own?(@vector_store, :delete)
|
|
707
729
|
prune_durable_identifier(identifier, []) if @durable_ids && implements_own?(@vector_store, :delete)
|
|
708
730
|
checkpoint[identifier] = source_hash
|
|
731
|
+
@checkpoint_inputs[identifier] = @prepared_inputs.fetch(identifier)
|
|
709
732
|
end
|
|
710
733
|
end
|
|
711
734
|
|
|
@@ -715,18 +738,92 @@ module Woods
|
|
|
715
738
|
raise Woods::Error, 'Cannot reconcile source-empty units: existing vector IDs could not be read'
|
|
716
739
|
end
|
|
717
740
|
|
|
718
|
-
def
|
|
741
|
+
def input_budget
|
|
742
|
+
@input_budget ||= InputBudget.for(
|
|
743
|
+
@provider, limit: [safe_max_input_tokens, preparer_option(:max_tokens, 8192)].compact.min,
|
|
744
|
+
chars_per_token: preparer_option(:chars_per_token, 4.0)
|
|
745
|
+
)
|
|
746
|
+
end
|
|
747
|
+
|
|
748
|
+
def preparer_option(name, fallback)
|
|
749
|
+
@text_preparer.respond_to?(name) ? @text_preparer.public_send(name) : fallback
|
|
750
|
+
end
|
|
751
|
+
|
|
752
|
+
def prepared_texts(unit_data)
|
|
753
|
+
@prepared_texts[storage_id(unit_data)] ||= prepare_texts(unit_data)
|
|
754
|
+
rescue InputLimitError, ArgumentError => e
|
|
755
|
+
detail = e.is_a?(InputLimitError) ? e.message : 'Cannot split source within the input limit'
|
|
756
|
+
raise InputLimitError, "#{detail}; #{unit_diagnostic(unit_data)}; #{budget_diagnostic}"
|
|
757
|
+
end
|
|
758
|
+
|
|
759
|
+
def embedding_ids(identifier, count)
|
|
760
|
+
Array.new(count) { |index| count > 1 ? "#{identifier}#chunk_#{index}" : identifier }
|
|
761
|
+
end
|
|
762
|
+
|
|
763
|
+
def prepared_fingerprint(unit_data)
|
|
764
|
+
identifier = storage_id(unit_data)
|
|
765
|
+
@prepared_inputs[identifier] ||= begin
|
|
766
|
+
texts = prepared_texts(unit_data)
|
|
767
|
+
Digest::SHA256.hexdigest(JSON.generate(embedding_ids(identifier, texts.size).zip(texts)))
|
|
768
|
+
end
|
|
769
|
+
end
|
|
770
|
+
|
|
771
|
+
def unit_diagnostic(unit_data)
|
|
772
|
+
%w[type identifier file_path].map { |key| "#{key}=#{unit_data[key].to_s[0, 160].inspect}" }.join(' ')
|
|
773
|
+
end
|
|
774
|
+
|
|
775
|
+
def budget_diagnostic
|
|
776
|
+
"model=#{input_budget.model.to_s[0, 100].inspect} counting=#{input_budget.method} limit=#{input_budget.limit}"
|
|
777
|
+
end
|
|
778
|
+
|
|
779
|
+
def prepare_texts(unit_data) # rubocop:disable Metrics/CyclomaticComplexity
|
|
719
780
|
unit = build_unit(unit_data)
|
|
781
|
+
return [] if unit.chunks.empty? && unit.source_code.to_s.strip.empty?
|
|
782
|
+
|
|
783
|
+
Chunking::ContributorChunks.ensure!(unit)
|
|
720
784
|
apply_chunking(unit) if @chunker && unit.chunks.empty? && needs_chunking?(unit)
|
|
721
785
|
# Extraction may have emitted chunks larger than the provider's
|
|
722
786
|
# budget (rails_source in particular). Enforce the ceiling on
|
|
723
787
|
# whatever chunks we have before handing off to the provider.
|
|
724
788
|
@chunker&.enforce_chunk_limits!(unit) if unit.chunks.any?
|
|
725
|
-
texts =
|
|
789
|
+
texts = prepare_unit_texts(unit)
|
|
726
790
|
# Drop empty/whitespace-only texts — embedding providers reject
|
|
727
791
|
# them with 400 and retrying never succeeds. Unit is effectively
|
|
728
792
|
# skipped when every text is empty (zero-source unit).
|
|
729
|
-
|
|
793
|
+
select_prepared_texts(unit_data, unit, texts)
|
|
794
|
+
end
|
|
795
|
+
|
|
796
|
+
def select_prepared_texts(unit_data, unit, texts)
|
|
797
|
+
selected = texts.each_index.reject do |index|
|
|
798
|
+
text = texts[index]
|
|
799
|
+
text.nil? || text.strip.empty? || content_portion_empty?(text, unit)
|
|
800
|
+
end
|
|
801
|
+
if SourceContributors.multiple?(unit)
|
|
802
|
+
@prepared_chunks[storage_id(unit_data)] = selected.map do |index|
|
|
803
|
+
prepared_chunk_metadata(unit, texts, index)
|
|
804
|
+
end
|
|
805
|
+
end
|
|
806
|
+
selected.map { |index| texts[index] }
|
|
807
|
+
end
|
|
808
|
+
|
|
809
|
+
def prepared_chunk_metadata(unit, texts, index)
|
|
810
|
+
return {} unless texts.size == unit.chunks.size
|
|
811
|
+
|
|
812
|
+
chunk = unit.chunks.fetch(index)
|
|
813
|
+
content = chunk.fetch(:content).to_s
|
|
814
|
+
return {} if content.empty? || !texts[index].to_s.end_with?(content)
|
|
815
|
+
|
|
816
|
+
chunk.fetch(:metadata, {})
|
|
817
|
+
end
|
|
818
|
+
|
|
819
|
+
def prepare_unit_texts(unit)
|
|
820
|
+
texts = if @text_preparer.respond_to?(:prepare_for_embedding)
|
|
821
|
+
@text_preparer.prepare_for_embedding(unit, budget: input_budget)
|
|
822
|
+
else
|
|
823
|
+
unit.chunks.any? ? @text_preparer.prepare_chunks(unit) : [@text_preparer.prepare(unit)]
|
|
824
|
+
end
|
|
825
|
+
texts.compact.each { |text| input_budget.validate!(text) unless text.strip.empty? }
|
|
826
|
+
texts
|
|
730
827
|
end
|
|
731
828
|
|
|
732
829
|
# True when a prepared text is just the metadata prefix with no
|
|
@@ -777,9 +874,7 @@ module Woods
|
|
|
777
874
|
# own +max_chars+ safety net is what guarantees each chunk fits,
|
|
778
875
|
# so we pass the same char budget through here.
|
|
779
876
|
def apply_chunking(unit)
|
|
780
|
-
unit.chunks = @chunker.chunk(unit).map
|
|
781
|
-
{ content: chunk.content, chunk_type: chunk.chunk_type }
|
|
782
|
-
end
|
|
877
|
+
unit.chunks = @chunker.chunk(unit).map(&:to_h)
|
|
783
878
|
end
|
|
784
879
|
|
|
785
880
|
def build_unit(data)
|
|
@@ -787,6 +882,7 @@ module Woods
|
|
|
787
882
|
file_path: data['file_path'])
|
|
788
883
|
unit.namespace = data['namespace']
|
|
789
884
|
unit.source_code = data['source_code']
|
|
885
|
+
unit.metadata = data['metadata'] || {}
|
|
790
886
|
unit.dependencies = data['dependencies'] || []
|
|
791
887
|
unit.chunks = (data['chunks'] || []).map { |c| c.transform_keys(&:to_sym) }
|
|
792
888
|
unit
|
|
@@ -801,18 +897,46 @@ module Woods
|
|
|
801
897
|
def embed_and_store(items, checkpoint, stats)
|
|
802
898
|
return if items.empty?
|
|
803
899
|
|
|
804
|
-
|
|
900
|
+
warn_unreconciled_adapter
|
|
901
|
+
vectors = @provider.embed_batch(items.map { |item| item[:text] })
|
|
805
902
|
store_vectors(items, vectors, checkpoint, stats)
|
|
806
903
|
rescue StandardError => e
|
|
807
904
|
stats[:errors] += items.size
|
|
808
|
-
raise
|
|
905
|
+
raise if e.is_a?(InputLimitError)
|
|
906
|
+
|
|
907
|
+
raise Woods::Error, embedding_failure(e, items), cause: nil
|
|
908
|
+
end
|
|
909
|
+
|
|
910
|
+
def warn_unreconciled_adapter
|
|
911
|
+
return if persistable? || reconcilable? || @unknown_reconciliation_warned
|
|
912
|
+
|
|
913
|
+
warn '[woods] custom vector store cannot enumerate and reconcile obsolete chunk IDs; cleanup is adapter-owned.'
|
|
914
|
+
@unknown_reconciliation_warned = true
|
|
915
|
+
end
|
|
916
|
+
|
|
917
|
+
def embedding_failure(error, items)
|
|
918
|
+
labels = items.lazy.map { |item| unit_diagnostic(item[:unit_data]) }.uniq.first(5).join('; ')
|
|
919
|
+
status = error.respond_to?(:http_status) ? error.http_status : nil
|
|
920
|
+
"Embedding failed (#{safe_failure_detail(error)}, HTTP #{status || 'unknown'}); " \
|
|
921
|
+
"#{labels}; #{items.size} input(s), largest=#{items.map { |item| input_budget.count(item[:text]) }.max}; " \
|
|
922
|
+
"#{budget_diagnostic}. " \
|
|
923
|
+
'No failed unit was checkpointed; check provider availability and input limits.'
|
|
924
|
+
end
|
|
925
|
+
|
|
926
|
+
def safe_failure_detail(error)
|
|
927
|
+
# Preserve numeric shape diagnostics, never source-bearing HTTP bodies
|
|
928
|
+
# or malformed response indexes (which can contain arbitrary strings).
|
|
929
|
+
if defined?(Provider::InvalidEmbeddingResponse) && error.is_a?(Provider::InvalidEmbeddingResponse)
|
|
930
|
+
return error.message[/vector at position \d+ has dimension \d+, expected \d+/] || error.class.name
|
|
931
|
+
end
|
|
932
|
+
|
|
933
|
+
error.class.name
|
|
809
934
|
end
|
|
810
935
|
|
|
811
936
|
def store_vectors(items, vectors, checkpoint, stats)
|
|
812
937
|
entries = items.each_with_index.map do |item, idx|
|
|
813
938
|
{ id: item[:id], vector: vectors[idx],
|
|
814
|
-
metadata:
|
|
815
|
-
file_path: item[:unit_data]['file_path'] } }
|
|
939
|
+
metadata: vector_metadata(item) }
|
|
816
940
|
end
|
|
817
941
|
|
|
818
942
|
@vector_store.store_batch(entries)
|
|
@@ -821,10 +945,17 @@ module Woods
|
|
|
821
945
|
|
|
822
946
|
items.each do |item|
|
|
823
947
|
checkpoint[item[:identifier]] = item[:source_hash]
|
|
948
|
+
@checkpoint_inputs[item[:identifier]] = @prepared_inputs.fetch(item[:identifier])
|
|
824
949
|
stats[:processed] += 1
|
|
825
950
|
end
|
|
826
951
|
end
|
|
827
952
|
|
|
953
|
+
def vector_metadata(item)
|
|
954
|
+
data = item.fetch(:unit_data)
|
|
955
|
+
{ type: data['type'], identifier: data['identifier'], file_path: data['file_path'] }
|
|
956
|
+
.merge(Chunking::ContributorChunks.vector_metadata(data, item[:chunk_metadata]))
|
|
957
|
+
end
|
|
958
|
+
|
|
828
959
|
# Suffix {#collect_embed_items} appends when a unit is split across
|
|
829
960
|
# several vectors. Mirrors the pattern in {Retriever},
|
|
830
961
|
# {Retrieval::ContextAssembler} and {MCP::Bootstrapper}.
|
|
@@ -938,79 +1069,43 @@ module Woods
|
|
|
938
1069
|
AtomicFile.write(File.join(@output_dir, 'checkpoint.json'), JSON.generate(checkpoint_payload(checkpoint)))
|
|
939
1070
|
end
|
|
940
1071
|
|
|
941
|
-
#
|
|
942
|
-
#
|
|
943
|
-
CHECKPOINT_SCHEMA_VERSION =
|
|
1072
|
+
# v2 adds a preparation-policy stamp and complete-input fingerprints.
|
|
1073
|
+
# Source hashes alone cannot detect path/namespace/dependency/chunk edits.
|
|
1074
|
+
CHECKPOINT_SCHEMA_VERSION = 2
|
|
944
1075
|
private_constant :CHECKPOINT_SCHEMA_VERSION
|
|
945
1076
|
|
|
946
|
-
# The provider/model/dimension triple checkpoint.json is stamped with,
|
|
947
|
-
# or +nil+ when this indexer was built without a +resolved_config+ (no
|
|
948
|
-
# identity to stamp or compare against — see {#checkpoint_payload} and
|
|
949
|
-
# {#checkpoint_hashes}, both of which treat +nil+ as "skip identity
|
|
950
|
-
# tracking entirely" for full backward compatibility with callers that
|
|
951
|
-
# never pass one).
|
|
952
|
-
#
|
|
953
|
-
# Reads {ResolvedConfig#to_snapshot_json} rather than calling
|
|
954
|
-
# +#embedding_provider+/+#dimension+ directly so a test double only
|
|
955
|
-
# needs to stub the one method the WVF1 header path already requires.
|
|
956
|
-
#
|
|
957
|
-
# @return [Hash, nil]
|
|
958
1077
|
def current_checkpoint_identity
|
|
959
|
-
|
|
960
|
-
|
|
961
|
-
provider = @resolved_config.to_snapshot_json['embedding_provider'] || {}
|
|
1078
|
+
provider = @resolved_config ? @resolved_config.to_snapshot_json['embedding_provider'] || {} : {}
|
|
962
1079
|
provider.transform_keys(&:to_s).slice('class', 'model', 'dimension')
|
|
963
1080
|
end
|
|
964
1081
|
|
|
965
|
-
|
|
966
|
-
|
|
967
|
-
|
|
968
|
-
identity = current_checkpoint_identity
|
|
969
|
-
return checkpoint if identity.nil?
|
|
970
|
-
|
|
971
|
-
{ 'schema_version' => CHECKPOINT_SCHEMA_VERSION, 'identity' => identity, 'hashes' => checkpoint }
|
|
1082
|
+
def preparation_identity
|
|
1083
|
+
{ 'version' => 1, 'budget' => input_budget.identity,
|
|
1084
|
+
'preparer' => collaborator_identity(@text_preparer), 'chunker' => collaborator_identity(@chunker) }
|
|
972
1085
|
end
|
|
973
1086
|
|
|
974
|
-
|
|
975
|
-
|
|
976
|
-
|
|
977
|
-
# - versioned (carries a top-level "hashes" key): written by this gem
|
|
978
|
-
# version, stamped with the provider/model/dimension identity that
|
|
979
|
-
# produced it (see #checkpoint_payload). A stamped identity that
|
|
980
|
-
# disagrees with {#current_checkpoint_identity} — a same-dimension
|
|
981
|
-
# model switch, the P1 finding this exists to close — means nothing
|
|
982
|
-
# here can say which individual hits are still good, so the *whole*
|
|
983
|
-
# checkpoint is discarded rather than trusted per-unit.
|
|
984
|
-
# - flat (every checkpoint written before this gem version): carries no
|
|
985
|
-
# identity at all. When this run tracks identity (a resolved_config
|
|
986
|
-
# was given), "no identity recorded" is indistinguishable from "the
|
|
987
|
-
# identity that produced this changed" — so it is discarded the same
|
|
988
|
-
# way: one full re-embed, after which every checkpoint this gem
|
|
989
|
-
# writes is stamped and can be trusted again. When this run has no
|
|
990
|
-
# resolved_config either there is nothing to compare against, and the
|
|
991
|
-
# flat map is trusted exactly as every prior gem version did.
|
|
992
|
-
def checkpoint_hashes(data)
|
|
993
|
-
return data unless data.is_a?(Hash)
|
|
994
|
-
|
|
995
|
-
current = current_checkpoint_identity
|
|
996
|
-
return checkpoint_hashes_versioned(data, current) if data.key?('hashes')
|
|
997
|
-
return data if current.nil?
|
|
1087
|
+
def collaborator_identity(object)
|
|
1088
|
+
object.respond_to?(:preparation_identity) ? object.preparation_identity : object&.class&.name
|
|
1089
|
+
end
|
|
998
1090
|
|
|
999
|
-
|
|
1000
|
-
|
|
1001
|
-
|
|
1002
|
-
|
|
1091
|
+
def checkpoint_payload(checkpoint)
|
|
1092
|
+
{ 'schema_version' => CHECKPOINT_SCHEMA_VERSION, 'identity' => current_checkpoint_identity,
|
|
1093
|
+
'preparation' => preparation_identity, 'hashes' => checkpoint,
|
|
1094
|
+
'prepared_inputs' => @checkpoint_inputs.slice(*checkpoint.keys) }
|
|
1003
1095
|
end
|
|
1004
1096
|
|
|
1005
|
-
def
|
|
1006
|
-
|
|
1007
|
-
|
|
1097
|
+
def checkpoint_hashes(data)
|
|
1098
|
+
valid = data.is_a?(Hash) && data['schema_version'] == CHECKPOINT_SCHEMA_VERSION &&
|
|
1099
|
+
data['identity'] == current_checkpoint_identity && data['preparation'] == preparation_identity &&
|
|
1100
|
+
data['hashes'].is_a?(Hash) && data['prepared_inputs'].is_a?(Hash)
|
|
1101
|
+
unless valid
|
|
1102
|
+
warn '[woods] checkpoint.json has no matching embedding identity and preparation policy; ' \
|
|
1103
|
+
'discarding the checkpoint and re-embedding every unit once.'
|
|
1104
|
+
return {}
|
|
1105
|
+
end
|
|
1008
1106
|
|
|
1009
|
-
|
|
1010
|
-
|
|
1011
|
-
'changed since the last run. Discarding the checkpoint and re-embedding every ' \
|
|
1012
|
-
'unit so no stale-model vector survives.'
|
|
1013
|
-
{}
|
|
1107
|
+
@checkpoint_inputs = data['prepared_inputs']
|
|
1108
|
+
data['hashes']
|
|
1014
1109
|
end
|
|
1015
1110
|
|
|
1016
1111
|
# Returns true when the vector store can actually be dumped to
|
|
@@ -1055,16 +1150,18 @@ module Woods
|
|
|
1055
1150
|
end
|
|
1056
1151
|
|
|
1057
1152
|
# Does +object+ define +method_name+ itself, rather than inheriting the
|
|
1058
|
-
#
|
|
1153
|
+
# storage interfaces' default stubs?
|
|
1059
1154
|
#
|
|
1060
1155
|
# @param object [Object] the adapter under test
|
|
1061
1156
|
# @param method_name [Symbol]
|
|
1062
1157
|
# @return [Boolean]
|
|
1063
1158
|
def implements_own?(object, method_name)
|
|
1064
1159
|
return false unless object.respond_to?(method_name)
|
|
1065
|
-
return true unless defined?(Storage::VectorStore::Interface)
|
|
1066
1160
|
|
|
1067
|
-
|
|
1161
|
+
interfaces = []
|
|
1162
|
+
interfaces << Storage::VectorStore::Interface if defined?(Storage::VectorStore::Interface)
|
|
1163
|
+
interfaces << Storage::MetadataStore::Interface if defined?(Storage::MetadataStore::Interface)
|
|
1164
|
+
!interfaces.include?(object.method(method_name).owner)
|
|
1068
1165
|
end
|
|
1069
1166
|
|
|
1070
1167
|
# Persist stores to a timestamped dump directory, write +woods.json+,
|
|
@@ -0,0 +1,67 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Woods
|
|
4
|
+
class Error < StandardError; end unless defined?(Woods::Error)
|
|
5
|
+
|
|
6
|
+
module Embedding
|
|
7
|
+
# A refusal, never a request for the retry wrapper to repeat the same input.
|
|
8
|
+
class InputLimitError < Woods::Error; end
|
|
9
|
+
|
|
10
|
+
# Counts used for admission are explicitly distinguished from sizing estimates.
|
|
11
|
+
# Byte BPE starts with UTF-8 bytes and only merges them: its token count cannot
|
|
12
|
+
# exceed bytesize. This bound applies to the known OpenAI embedding encodings,
|
|
13
|
+
# not arbitrary tokenizers (whose normalization or special tokens may differ).
|
|
14
|
+
# Encoding mapping and token-byte examples:
|
|
15
|
+
# https://developers.openai.com/cookbook/examples/how_to_count_tokens_with_tiktoken
|
|
16
|
+
class InputBudget
|
|
17
|
+
attr_reader :limit, :method, :model
|
|
18
|
+
|
|
19
|
+
def initialize(limit:, model: nil, method: 'estimate', chars_per_token: 1.2)
|
|
20
|
+
raise ArgumentError, 'input limit must be positive' unless limit.is_a?(Integer) && limit.positive?
|
|
21
|
+
raise ArgumentError, 'chars_per_token must be positive' unless chars_per_token.positive?
|
|
22
|
+
|
|
23
|
+
@limit = limit
|
|
24
|
+
@model = model
|
|
25
|
+
@method = method
|
|
26
|
+
@chars_per_token = chars_per_token
|
|
27
|
+
end
|
|
28
|
+
|
|
29
|
+
def self.for(provider, limit:, chars_per_token:)
|
|
30
|
+
supplied = provider.input_budget if provider.respond_to?(:input_budget)
|
|
31
|
+
return new(limit: limit, chars_per_token: chars_per_token) unless supplied
|
|
32
|
+
return supplied unless limit < supplied.limit
|
|
33
|
+
|
|
34
|
+
supplied.with_limit(limit)
|
|
35
|
+
end
|
|
36
|
+
|
|
37
|
+
# Preserve the provider's counting policy under a stricter caller cap.
|
|
38
|
+
def with_limit(value)
|
|
39
|
+
self.class.new(limit: [limit, value].min, model: model, method: method, chars_per_token: @chars_per_token)
|
|
40
|
+
end
|
|
41
|
+
|
|
42
|
+
def count(text)
|
|
43
|
+
utf8 = text.encode(Encoding::UTF_8)
|
|
44
|
+
raise InputLimitError, 'Embedding input requires valid UTF-8 text' unless utf8.valid_encoding?
|
|
45
|
+
|
|
46
|
+
method == 'utf8_bytes_bound' ? utf8.bytesize : (utf8.length / @chars_per_token).ceil
|
|
47
|
+
rescue EncodingError
|
|
48
|
+
raise InputLimitError, 'Embedding input requires valid UTF-8 text'
|
|
49
|
+
end
|
|
50
|
+
|
|
51
|
+
def fits?(text)
|
|
52
|
+
count(text) <= limit
|
|
53
|
+
end
|
|
54
|
+
|
|
55
|
+
def validate!(text)
|
|
56
|
+
size = count(text)
|
|
57
|
+
return text if size <= limit
|
|
58
|
+
|
|
59
|
+
raise InputLimitError, "Embedding input limit exceeded (#{method}: #{size}, limit: #{limit}); split the input"
|
|
60
|
+
end
|
|
61
|
+
|
|
62
|
+
def identity
|
|
63
|
+
{ 'method' => method, 'limit' => limit, 'model' => model, 'chars_per_token' => @chars_per_token }
|
|
64
|
+
end
|
|
65
|
+
end
|
|
66
|
+
end
|
|
67
|
+
end
|