woods 2.0.1 → 2.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (154) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +94 -7
  3. data/CONTRIBUTING.md +134 -19
  4. data/README.md +1 -1
  5. data/docs/AGENT_GUIDE.md +19 -0
  6. data/docs/AGENT_SETUP.md +22 -2
  7. data/docs/BACKEND_MATRIX.md +7 -0
  8. data/docs/CLIENT_HOOKS.md +6 -0
  9. data/docs/CONFIGURATION_REFERENCE.md +133 -25
  10. data/docs/CONSOLE_MCP_SETUP.md +82 -30
  11. data/docs/EMBEDDING_MODELS.md +16 -19
  12. data/docs/EXTRACTOR_REFERENCE.md +219 -21
  13. data/docs/FAQ.md +11 -25
  14. data/docs/GETTING_STARTED.md +7 -1
  15. data/docs/INCREMENTAL_EXTRACTION.md +261 -19
  16. data/docs/INDEX_LAYOUT.md +5 -0
  17. data/docs/INTERNALS.md +9 -0
  18. data/docs/MCP_HTTP_TRANSPORT.md +20 -15
  19. data/docs/MCP_SERVERS.md +87 -8
  20. data/docs/MCP_TOOL_COOKBOOK.md +13 -55
  21. data/docs/NOTION_INTEGRATION.md +7 -1
  22. data/docs/PUBLISHED_INDEX.md +6 -0
  23. data/docs/README.md +6 -1
  24. data/docs/RETRIEVAL_GUIDE.md +17 -0
  25. data/docs/SOURCE_FRESHNESS.md +157 -5
  26. data/docs/TOKEN_BENCHMARK.md +10 -18
  27. data/docs/TROUBLESHOOTING.md +70 -14
  28. data/docs/UNBLOCKED_INTEGRATION.md +60 -8
  29. data/docs/UPGRADING_TO_2.md +153 -38
  30. data/docs/WATCH_DAEMON.md +97 -14
  31. data/exe/woods-console-mcp +2 -2
  32. data/lib/generators/woods/templates/woods.rb.tt +2 -1
  33. data/lib/tasks/woods.rake +23 -7
  34. data/lib/tasks/woods_checks.rake +2 -2
  35. data/lib/woods/agent_configuration/cli.rb +1 -1
  36. data/lib/woods/agent_configuration/layout.rb +16 -2
  37. data/lib/woods/agent_configuration/plan.rb +13 -3
  38. data/lib/woods/agent_configuration/planner_validation.rb +4 -2
  39. data/lib/woods/agent_configuration/preflight.rb +5 -3
  40. data/lib/woods/builder.rb +17 -57
  41. data/lib/woods/cache/cache_middleware.rb +56 -30
  42. data/lib/woods/chunking/contributor_chunks.rb +119 -0
  43. data/lib/woods/chunking/semantic_chunker.rb +44 -21
  44. data/lib/woods/console/connection_manager.rb +56 -3
  45. data/lib/woods/console/embedded_executor.rb +30 -5
  46. data/lib/woods/console/rack_middleware.rb +29 -1
  47. data/lib/woods/dependency_graph.rb +34 -10
  48. data/lib/woods/embedding/fake.rb +12 -0
  49. data/lib/woods/embedding/indexer.rb +195 -98
  50. data/lib/woods/embedding/input_budget.rb +67 -0
  51. data/lib/woods/embedding/openai.rb +70 -20
  52. data/lib/woods/embedding/provider.rb +37 -25
  53. data/lib/woods/embedding/text_preparer.rb +76 -32
  54. data/lib/woods/embedding/token_counter.rb +18 -81
  55. data/lib/woods/embedding/vector_configuration.rb +48 -0
  56. data/lib/woods/extraction_identities.rb +175 -0
  57. data/lib/woods/extractor.rb +304 -107
  58. data/lib/woods/extractors/action_cable_extractor.rb +8 -3
  59. data/lib/woods/extractors/assigned_value_discovery.rb +74 -0
  60. data/lib/woods/extractors/class_declarations.rb +121 -0
  61. data/lib/woods/extractors/configuration_extractor.rb +11 -3
  62. data/lib/woods/extractors/declaration_ancestry.rb +92 -0
  63. data/lib/woods/extractors/event_extractor.rb +8 -0
  64. data/lib/woods/extractors/graphql_extractor.rb +134 -77
  65. data/lib/woods/extractors/job_extractor.rb +5 -1
  66. data/lib/woods/extractors/lib_extractor.rb +132 -15
  67. data/lib/woods/extractors/mailer_extractor.rb +3 -5
  68. data/lib/woods/extractors/manager_extractor.rb +7 -21
  69. data/lib/woods/extractors/migration_declaration.rb +87 -0
  70. data/lib/woods/extractors/migration_extractor.rb +5 -39
  71. data/lib/woods/extractors/phlex_extractor.rb +6 -2
  72. data/lib/woods/extractors/policy_extractor.rb +9 -5
  73. data/lib/woods/extractors/poro_extractor.rb +112 -53
  74. data/lib/woods/extractors/pundit_extractor.rb +11 -6
  75. data/lib/woods/extractors/scheduled_job_extractor.rb +45 -4
  76. data/lib/woods/extractors/serializer_extractor.rb +34 -22
  77. data/lib/woods/extractors/shared_utility_methods.rb +18 -1
  78. data/lib/woods/extractors/source_nesting.rb +142 -106
  79. data/lib/woods/extractors/standalone_module_discovery.rb +123 -0
  80. data/lib/woods/extractors/state_machine_extractor.rb +46 -40
  81. data/lib/woods/extractors/view_component_extractor.rb +9 -7
  82. data/lib/woods/flow_assembler.rb +4 -1
  83. data/lib/woods/generation.rb +25 -0
  84. data/lib/woods/hooks/context_hint.rb +7 -2
  85. data/lib/woods/mcp/bootstrapper.rb +33 -7
  86. data/lib/woods/mcp/config_resolver.rb +26 -7
  87. data/lib/woods/mcp/index_reader.rb +125 -24
  88. data/lib/woods/mcp/index_reader_pinning.rb +16 -0
  89. data/lib/woods/mcp/renderers/markdown_renderer.rb +7 -1
  90. data/lib/woods/mcp/renderers/plain_renderer.rb +3 -1
  91. data/lib/woods/mcp/search_results.rb +7 -1
  92. data/lib/woods/mcp/server.rb +24 -4
  93. data/lib/woods/module_reconciliation.rb +151 -0
  94. data/lib/woods/path_dispatcher.rb +7 -2
  95. data/lib/woods/rake_helpers.rb +43 -11
  96. data/lib/woods/release.rb +1 -1
  97. data/lib/woods/resilience/index_validator.rb +8 -3
  98. data/lib/woods/resilience/retryable_provider.rb +18 -1
  99. data/lib/woods/resolved_config.rb +68 -8
  100. data/lib/woods/retrieval/context_assembler.rb +3 -3
  101. data/lib/woods/retrieval/lexical_assembler.rb +3 -2
  102. data/lib/woods/retrieval/scope.rb +18 -2
  103. data/lib/woods/retrieval/source_evidence.rb +14 -2
  104. data/lib/woods/source_contributor_validation.rb +78 -0
  105. data/lib/woods/source_contributors.rb +116 -0
  106. data/lib/woods/source_inputs/handoff.rb +37 -0
  107. data/lib/woods/source_inputs/launcher.rb +53 -13
  108. data/lib/woods/source_inputs/manifest.rb +84 -3
  109. data/lib/woods/source_inputs/private_key.rb +44 -12
  110. data/lib/woods/source_inputs/scanner.rb +98 -27
  111. data/lib/woods/source_inputs/scopes.rb +1 -1
  112. data/lib/woods/source_inputs/session.rb +147 -15
  113. data/lib/woods/source_inputs/stable_reader.rb +127 -0
  114. data/lib/woods/source_inputs/status.rb +40 -8
  115. data/lib/woods/source_inputs/verifier.rb +28 -5
  116. data/lib/woods/source_path_encoding.rb +33 -0
  117. data/lib/woods/source_references/cache.rb +284 -0
  118. data/lib/woods/source_references/collector.rb +120 -0
  119. data/lib/woods/source_references/extraction.rb +185 -0
  120. data/lib/woods/source_references/inputs.rb +134 -0
  121. data/lib/woods/source_references/parser_adapter.rb +134 -0
  122. data/lib/woods/source_references/pass.rb +152 -0
  123. data/lib/woods/source_references/prism_adapter.rb +116 -0
  124. data/lib/woods/source_references/registry.rb +178 -0
  125. data/lib/woods/source_references/runtime_lookup.rb +127 -0
  126. data/lib/woods/source_references/value_class.rb +82 -0
  127. data/lib/woods/storage/metadata_store.rb +4 -1
  128. data/lib/woods/storage/qdrant.rb +2 -2
  129. data/lib/woods/unblocked/client.rb +12 -7
  130. data/lib/woods/unblocked/document_builder.rb +4 -1
  131. data/lib/woods/unblocked/exporter.rb +127 -37
  132. data/lib/woods/unblocked/sync_manifest.rb +137 -21
  133. data/lib/woods/unblocked/uri_migration.rb +105 -0
  134. data/lib/woods/util/host_guard.rb +3 -2
  135. data/lib/woods/version.rb +1 -1
  136. data/lib/woods/watch/catch_up.rb +138 -0
  137. data/lib/woods/watch/claim_lease.rb +150 -0
  138. data/lib/woods/watch/cli.rb +26 -2
  139. data/lib/woods/watch/daemon.rb +80 -59
  140. data/lib/woods/watch/installation/options.rb +1 -1
  141. data/lib/woods/watch/installation/receipt.rb +6 -1
  142. data/lib/woods/watch/managed_child.rb +1 -1
  143. data/lib/woods/watch/supervisor.rb +1 -1
  144. data/lib/woods/watch/tree_scan.rb +14 -2
  145. data/plugin/.claude-plugin/plugin.json +1 -1
  146. data/plugin/hooks/adapters/normalize.rb +3 -2
  147. data/plugin/hooks/woods-input-rules.sh +4 -0
  148. data/plugin/hooks/woods-refresh.sh +15 -7
  149. data/plugin/hooks/woods-session-start.sh +60 -3
  150. data/plugin/skills/woods-diagnose/SKILL.md +334 -11
  151. data/plugin/skills/woods-investigate/SKILL.md +11 -0
  152. data/plugin/skills/woods-mcp-config/SKILL.md +79 -8
  153. data/plugin/skills/woods-setup/SKILL.md +53 -6
  154. metadata +32 -5
@@ -5,6 +5,7 @@ require 'digest'
5
5
  require 'fileutils'
6
6
  require 'set'
7
7
 
8
+ require_relative 'input_budget'
8
9
  require_relative '../atomic_file'
9
10
  require_relative '../storage_identity'
10
11
  require_relative '../generation'
@@ -64,9 +65,10 @@ module Woods
64
65
  # into semantically coherent chunks before embedding. +nil+ disables
65
66
  # chunking — units go to the provider whole (useful in tests).
66
67
  # @param checkpoint_interval [Integer] Save checkpoint every N batches (default: 10)
67
- # @param metadata_store [#each_entry, #bulk_load, nil] Optional metadata store.
68
- # When present alongside an in-memory vector store, both are persisted
69
- # at the end of a successful {#index_all} run.
68
+ # @param metadata_store [Storage::MetadataStore::Interface, nil] Optional metadata store.
69
+ # Existing identities participate in snapshot-store reconciliation.
70
+ # Stores with #each_entry and #bulk_load are persisted alongside vectors;
71
+ # SQLite retains its records in the configured database instead.
70
72
  # @param resolved_config [Woods::ResolvedConfig, nil] Captured config for
71
73
  # +woods.json+ — written to +output_dir+ on {#index_all} completion.
72
74
  # @param dump_retention_count [Integer] Number of completed dump directories
@@ -152,6 +154,7 @@ module Woods
152
154
  prepare_run(incremental: incremental)
153
155
  checkpoint = incremental ? load_checkpoint : {}
154
156
  units = assign_storage_identities(units, checkpoint: checkpoint)
157
+ preflight_inputs(units, checkpoint, incremental: incremental)
155
158
  stats = { processed: 0, skipped: 0, errors: 0 }
156
159
 
157
160
  embed_batches(units, checkpoint, stats, incremental: incremental)
@@ -170,6 +173,15 @@ module Woods
170
173
  stats
171
174
  end
172
175
 
176
+ # Reject deterministic input failures before metadata or durable writes.
177
+ def preflight_inputs(units, checkpoint, incremental:)
178
+ units.each { |unit| prepared_fingerprint(unit) }
179
+ return unless reconcilable? && @durable_ids.nil?
180
+ return if incremental && units.all? { |unit| checkpoint_satisfied?(unit, checkpoint) }
181
+
182
+ raise InputLimitError, 'Cannot replace embedding inputs: existing durable vector IDs could not be read'
183
+ end
184
+
173
185
  # Unambiguous existing keys stay stable. A collision uses reversible typed keys.
174
186
  def assign_storage_identities(units, checkpoint:)
175
187
  counts = units.group_by { |unit| unit['identifier'] }.transform_values(&:size)
@@ -474,6 +486,11 @@ module Woods
474
486
  @metadata_changed = false
475
487
  @vectors_changed = false
476
488
  @empty_units = {}
489
+ @prepared_texts = {}
490
+ @prepared_chunks = {}
491
+ @prepared_inputs = {}
492
+ @checkpoint_inputs = {}
493
+ @unknown_reconciliation_warned = false
477
494
  @persisted_metadata = nil
478
495
  prepare_snapshot_stores(incremental: incremental)
479
496
  retain_metadata_identities
@@ -492,14 +509,17 @@ module Woods
492
509
  entries = []
493
510
  @vector_store.each_entry { |id, _vector, _metadata| entries << { id: id } }
494
511
  @persisted_ids = index_ids_by_identifier(entries)
495
- retain_existing_metadata_identities
496
512
  end
513
+ retain_existing_metadata_identities
497
514
  end
498
515
 
499
- def retain_existing_metadata_identities
500
- return unless @metadata_store.respond_to?(:each_entry)
501
-
502
- @metadata_store.each_entry { |identifier, _unit| @persisted_ids[identifier] ||= [] }
516
+ def retain_existing_metadata_identities(identities = @persisted_ids)
517
+ if implements_own?(@metadata_store, :all_identifiers)
518
+ @metadata_store.all_identifiers.each { |identifier| identities[identifier] ||= [] }
519
+ elsif @metadata_store.respond_to?(:each_entry)
520
+ # Compatibility for custom snapshot-only stores.
521
+ @metadata_store.each_entry { |identifier, _unit| identities[identifier] ||= [] }
522
+ end
503
523
  end
504
524
 
505
525
  # Source-empty units retain metadata but intentionally have no vectors.
@@ -524,6 +544,7 @@ module Woods
524
544
  def load_durable_store_ids
525
545
  @durable_ids = Hash.new { |hash, key| hash[key] = [] }
526
546
  @vector_store.each_id { |id| @durable_ids[base_identifier(id)] << id }
547
+ retain_existing_metadata_identities(@durable_ids)
527
548
  rescue StandardError => e
528
549
  # A store that cannot be enumerated must not take the embed run down
529
550
  # with it. Reconciliation and the presence check both degrade to their
@@ -609,11 +630,9 @@ module Woods
609
630
 
610
631
  # May this unit's embedding be skipped?
611
632
  #
612
- # Incremental skip uses `source_hash`, which the extractor derives
613
- # from the unit's *source_code string only* (see ExtractedUnit#to_h
614
- # and Extractor#dump_units). It is NOT a hash of the serialized
615
- # unit_data JSON — so key ordering or whitespace in the _index.json
616
- # does not invalidate checkpoints across Ruby-minor upgrades.
633
+ # Incremental skip requires both the source hash and the complete ordered
634
+ # prepared-input fingerprint. Dependency/path/namespace/chunk changes
635
+ # invalidate the latter even if source_code did not change.
617
636
  #
618
637
  # A matching hash is necessary but not sufficient: the vector must also
619
638
  # actually exist. On the dump-backed path that means present in what we
@@ -626,17 +645,17 @@ module Woods
626
645
  # unit permanently: checkpoint.json said "embedded", the new store held
627
646
  # nothing, and no subsequent incremental run ever disagreed.
628
647
  def checkpoint_satisfied?(unit_data, checkpoint)
629
- return false unless checkpoint[storage_id(unit_data)] == unit_data['source_hash']
648
+ identifier = storage_id(unit_data)
649
+ return false unless checkpoint[identifier] == unit_data['source_hash']
650
+ return false unless @checkpoint_inputs[identifier] == prepared_fingerprint(unit_data)
630
651
 
631
652
  known_ids = persistable? ? @persisted_ids : @durable_ids
632
653
  # No durable view to check against (an adapter with no #each_id, or an
633
654
  # enumeration that failed) — fall back to trusting the checkpoint.
634
655
  return true if known_ids.nil?
635
656
 
636
- return true if known_ids[storage_id(unit_data)]&.any?
637
- # A source-empty unit intentionally has no vector. Verify that state
638
- # again rather than treating a missing nonempty vector as a cache hit.
639
- return true if prepare_texts(unit_data).empty?
657
+ expected = embedding_ids(identifier, prepared_texts(unit_data).size)
658
+ return true if Array(known_ids[identifier]).sort == expected.sort
640
659
 
641
660
  @checkpoint_misses += 1
642
661
  false
@@ -654,8 +673,10 @@ module Woods
654
673
  return unless @metadata_store
655
674
 
656
675
  id = storage_id(unit_data)
657
- @metadata_changed ||= @persisted_metadata && @persisted_metadata.find(id) != unit_data
658
- @metadata_store.store(id, unit_data)
676
+ chunks = @prepared_chunks[id]
677
+ data = chunks ? unit_data.merge('embedding_chunks' => JSON.parse(JSON.generate(chunks))) : unit_data
678
+ @metadata_changed ||= @persisted_metadata && @persisted_metadata.find(id) != data
679
+ @metadata_store.store(id, data)
659
680
  end
660
681
 
661
682
  # Compare with the promoted artifact, not a fresh metadata store or the
@@ -684,21 +705,22 @@ module Woods
684
705
  end
685
706
 
686
707
  def collect_embed_items(unit_data, items)
687
- texts = prepare_texts(unit_data)
708
+ texts = prepared_texts(unit_data)
688
709
  identifier = storage_id(unit_data)
689
710
  @empty_units[identifier] = unit_data['source_hash'] if texts.empty?
690
711
 
691
712
  texts.each_with_index do |text, idx|
692
713
  embed_id = texts.length > 1 ? "#{identifier}#chunk_#{idx}" : identifier
693
714
  items << { id: embed_id, text: text, unit_data: unit_data,
694
- source_hash: unit_data['source_hash'], identifier: identifier }
715
+ source_hash: unit_data['source_hash'], identifier: identifier,
716
+ chunk_metadata: @prepared_chunks[identifier]&.[](idx) }
695
717
  end
696
718
  end
697
719
 
698
720
  # Defer zero-text deletion until every batch has prepared/embedded
699
721
  # successfully. A later provider failure must not retire earlier vectors.
700
- # Checkpoints retain their source-hash format; a matching no-vector hash
701
- # is trusted only after verifying the current input still prepares empty.
722
+ # The no-vector checkpoint also records the complete-input fingerprint;
723
+ # it advances only after obsolete vectors have been reconciled.
702
724
  def reconcile_empty_units(checkpoint)
703
725
  verify_empty_reconciliation!
704
726
  @empty_units.each do |identifier, source_hash|
@@ -706,6 +728,7 @@ module Woods
706
728
  prune_identifier(identifier, []) if implements_own?(@vector_store, :delete)
707
729
  prune_durable_identifier(identifier, []) if @durable_ids && implements_own?(@vector_store, :delete)
708
730
  checkpoint[identifier] = source_hash
731
+ @checkpoint_inputs[identifier] = @prepared_inputs.fetch(identifier)
709
732
  end
710
733
  end
711
734
 
@@ -715,18 +738,92 @@ module Woods
715
738
  raise Woods::Error, 'Cannot reconcile source-empty units: existing vector IDs could not be read'
716
739
  end
717
740
 
718
- def prepare_texts(unit_data) # rubocop:disable Metrics/CyclomaticComplexity, Metrics/PerceivedComplexity
741
+ def input_budget
742
+ @input_budget ||= InputBudget.for(
743
+ @provider, limit: [safe_max_input_tokens, preparer_option(:max_tokens, 8192)].compact.min,
744
+ chars_per_token: preparer_option(:chars_per_token, 4.0)
745
+ )
746
+ end
747
+
748
+ def preparer_option(name, fallback)
749
+ @text_preparer.respond_to?(name) ? @text_preparer.public_send(name) : fallback
750
+ end
751
+
752
+ def prepared_texts(unit_data)
753
+ @prepared_texts[storage_id(unit_data)] ||= prepare_texts(unit_data)
754
+ rescue InputLimitError, ArgumentError => e
755
+ detail = e.is_a?(InputLimitError) ? e.message : 'Cannot split source within the input limit'
756
+ raise InputLimitError, "#{detail}; #{unit_diagnostic(unit_data)}; #{budget_diagnostic}"
757
+ end
758
+
759
+ def embedding_ids(identifier, count)
760
+ Array.new(count) { |index| count > 1 ? "#{identifier}#chunk_#{index}" : identifier }
761
+ end
762
+
763
+ def prepared_fingerprint(unit_data)
764
+ identifier = storage_id(unit_data)
765
+ @prepared_inputs[identifier] ||= begin
766
+ texts = prepared_texts(unit_data)
767
+ Digest::SHA256.hexdigest(JSON.generate(embedding_ids(identifier, texts.size).zip(texts)))
768
+ end
769
+ end
770
+
771
+ def unit_diagnostic(unit_data)
772
+ %w[type identifier file_path].map { |key| "#{key}=#{unit_data[key].to_s[0, 160].inspect}" }.join(' ')
773
+ end
774
+
775
+ def budget_diagnostic
776
+ "model=#{input_budget.model.to_s[0, 100].inspect} counting=#{input_budget.method} limit=#{input_budget.limit}"
777
+ end
778
+
779
+ def prepare_texts(unit_data) # rubocop:disable Metrics/CyclomaticComplexity
719
780
  unit = build_unit(unit_data)
781
+ return [] if unit.chunks.empty? && unit.source_code.to_s.strip.empty?
782
+
783
+ Chunking::ContributorChunks.ensure!(unit)
720
784
  apply_chunking(unit) if @chunker && unit.chunks.empty? && needs_chunking?(unit)
721
785
  # Extraction may have emitted chunks larger than the provider's
722
786
  # budget (rails_source in particular). Enforce the ceiling on
723
787
  # whatever chunks we have before handing off to the provider.
724
788
  @chunker&.enforce_chunk_limits!(unit) if unit.chunks.any?
725
- texts = unit.chunks.any? ? @text_preparer.prepare_chunks(unit) : [@text_preparer.prepare(unit)]
789
+ texts = prepare_unit_texts(unit)
726
790
  # Drop empty/whitespace-only texts — embedding providers reject
727
791
  # them with 400 and retrying never succeeds. Unit is effectively
728
792
  # skipped when every text is empty (zero-source unit).
729
- texts.reject { |t| t.nil? || t.strip.empty? || content_portion_empty?(t, unit) }
793
+ select_prepared_texts(unit_data, unit, texts)
794
+ end
795
+
796
+ def select_prepared_texts(unit_data, unit, texts)
797
+ selected = texts.each_index.reject do |index|
798
+ text = texts[index]
799
+ text.nil? || text.strip.empty? || content_portion_empty?(text, unit)
800
+ end
801
+ if SourceContributors.multiple?(unit)
802
+ @prepared_chunks[storage_id(unit_data)] = selected.map do |index|
803
+ prepared_chunk_metadata(unit, texts, index)
804
+ end
805
+ end
806
+ selected.map { |index| texts[index] }
807
+ end
808
+
809
+ def prepared_chunk_metadata(unit, texts, index)
810
+ return {} unless texts.size == unit.chunks.size
811
+
812
+ chunk = unit.chunks.fetch(index)
813
+ content = chunk.fetch(:content).to_s
814
+ return {} if content.empty? || !texts[index].to_s.end_with?(content)
815
+
816
+ chunk.fetch(:metadata, {})
817
+ end
818
+
819
+ def prepare_unit_texts(unit)
820
+ texts = if @text_preparer.respond_to?(:prepare_for_embedding)
821
+ @text_preparer.prepare_for_embedding(unit, budget: input_budget)
822
+ else
823
+ unit.chunks.any? ? @text_preparer.prepare_chunks(unit) : [@text_preparer.prepare(unit)]
824
+ end
825
+ texts.compact.each { |text| input_budget.validate!(text) unless text.strip.empty? }
826
+ texts
730
827
  end
731
828
 
732
829
  # True when a prepared text is just the metadata prefix with no
@@ -777,9 +874,7 @@ module Woods
777
874
  # own +max_chars+ safety net is what guarantees each chunk fits,
778
875
  # so we pass the same char budget through here.
779
876
  def apply_chunking(unit)
780
- unit.chunks = @chunker.chunk(unit).map do |chunk|
781
- { content: chunk.content, chunk_type: chunk.chunk_type }
782
- end
877
+ unit.chunks = @chunker.chunk(unit).map(&:to_h)
783
878
  end
784
879
 
785
880
  def build_unit(data)
@@ -787,6 +882,7 @@ module Woods
787
882
  file_path: data['file_path'])
788
883
  unit.namespace = data['namespace']
789
884
  unit.source_code = data['source_code']
885
+ unit.metadata = data['metadata'] || {}
790
886
  unit.dependencies = data['dependencies'] || []
791
887
  unit.chunks = (data['chunks'] || []).map { |c| c.transform_keys(&:to_sym) }
792
888
  unit
@@ -801,18 +897,46 @@ module Woods
801
897
  def embed_and_store(items, checkpoint, stats)
802
898
  return if items.empty?
803
899
 
804
- vectors = @provider.embed_batch(items.map { |i| i[:text] })
900
+ warn_unreconciled_adapter
901
+ vectors = @provider.embed_batch(items.map { |item| item[:text] })
805
902
  store_vectors(items, vectors, checkpoint, stats)
806
903
  rescue StandardError => e
807
904
  stats[:errors] += items.size
808
- raise Woods::Error, "Embedding failed: #{e.message}"
905
+ raise if e.is_a?(InputLimitError)
906
+
907
+ raise Woods::Error, embedding_failure(e, items), cause: nil
908
+ end
909
+
910
+ def warn_unreconciled_adapter
911
+ return if persistable? || reconcilable? || @unknown_reconciliation_warned
912
+
913
+ warn '[woods] custom vector store cannot enumerate and reconcile obsolete chunk IDs; cleanup is adapter-owned.'
914
+ @unknown_reconciliation_warned = true
915
+ end
916
+
917
+ def embedding_failure(error, items)
918
+ labels = items.lazy.map { |item| unit_diagnostic(item[:unit_data]) }.uniq.first(5).join('; ')
919
+ status = error.respond_to?(:http_status) ? error.http_status : nil
920
+ "Embedding failed (#{safe_failure_detail(error)}, HTTP #{status || 'unknown'}); " \
921
+ "#{labels}; #{items.size} input(s), largest=#{items.map { |item| input_budget.count(item[:text]) }.max}; " \
922
+ "#{budget_diagnostic}. " \
923
+ 'No failed unit was checkpointed; check provider availability and input limits.'
924
+ end
925
+
926
+ def safe_failure_detail(error)
927
+ # Preserve numeric shape diagnostics, never source-bearing HTTP bodies
928
+ # or malformed response indexes (which can contain arbitrary strings).
929
+ if defined?(Provider::InvalidEmbeddingResponse) && error.is_a?(Provider::InvalidEmbeddingResponse)
930
+ return error.message[/vector at position \d+ has dimension \d+, expected \d+/] || error.class.name
931
+ end
932
+
933
+ error.class.name
809
934
  end
810
935
 
811
936
  def store_vectors(items, vectors, checkpoint, stats)
812
937
  entries = items.each_with_index.map do |item, idx|
813
938
  { id: item[:id], vector: vectors[idx],
814
- metadata: { type: item[:unit_data]['type'], identifier: item[:unit_data]['identifier'],
815
- file_path: item[:unit_data]['file_path'] } }
939
+ metadata: vector_metadata(item) }
816
940
  end
817
941
 
818
942
  @vector_store.store_batch(entries)
@@ -821,10 +945,17 @@ module Woods
821
945
 
822
946
  items.each do |item|
823
947
  checkpoint[item[:identifier]] = item[:source_hash]
948
+ @checkpoint_inputs[item[:identifier]] = @prepared_inputs.fetch(item[:identifier])
824
949
  stats[:processed] += 1
825
950
  end
826
951
  end
827
952
 
953
+ def vector_metadata(item)
954
+ data = item.fetch(:unit_data)
955
+ { type: data['type'], identifier: data['identifier'], file_path: data['file_path'] }
956
+ .merge(Chunking::ContributorChunks.vector_metadata(data, item[:chunk_metadata]))
957
+ end
958
+
828
959
  # Suffix {#collect_embed_items} appends when a unit is split across
829
960
  # several vectors. Mirrors the pattern in {Retriever},
830
961
  # {Retrieval::ContextAssembler} and {MCP::Bootstrapper}.
@@ -938,79 +1069,43 @@ module Woods
938
1069
  AtomicFile.write(File.join(@output_dir, 'checkpoint.json'), JSON.generate(checkpoint_payload(checkpoint)))
939
1070
  end
940
1071
 
941
- # Schema of the on-disk checkpoint payload when {#resolved_config} is
942
- # tracked. Bump only alongside a reader change in {#checkpoint_hashes}.
943
- CHECKPOINT_SCHEMA_VERSION = 1
1072
+ # v2 adds a preparation-policy stamp and complete-input fingerprints.
1073
+ # Source hashes alone cannot detect path/namespace/dependency/chunk edits.
1074
+ CHECKPOINT_SCHEMA_VERSION = 2
944
1075
  private_constant :CHECKPOINT_SCHEMA_VERSION
945
1076
 
946
- # The provider/model/dimension triple checkpoint.json is stamped with,
947
- # or +nil+ when this indexer was built without a +resolved_config+ (no
948
- # identity to stamp or compare against — see {#checkpoint_payload} and
949
- # {#checkpoint_hashes}, both of which treat +nil+ as "skip identity
950
- # tracking entirely" for full backward compatibility with callers that
951
- # never pass one).
952
- #
953
- # Reads {ResolvedConfig#to_snapshot_json} rather than calling
954
- # +#embedding_provider+/+#dimension+ directly so a test double only
955
- # needs to stub the one method the WVF1 header path already requires.
956
- #
957
- # @return [Hash, nil]
958
1077
  def current_checkpoint_identity
959
- return nil unless @resolved_config
960
-
961
- provider = @resolved_config.to_snapshot_json['embedding_provider'] || {}
1078
+ provider = @resolved_config ? @resolved_config.to_snapshot_json['embedding_provider'] || {} : {}
962
1079
  provider.transform_keys(&:to_s).slice('class', 'model', 'dimension')
963
1080
  end
964
1081
 
965
- # Wrap the flat identifier=>source_hash map with its identity stamp for
966
- # writing, or leave it flat when this run tracks no identity.
967
- def checkpoint_payload(checkpoint)
968
- identity = current_checkpoint_identity
969
- return checkpoint if identity.nil?
970
-
971
- { 'schema_version' => CHECKPOINT_SCHEMA_VERSION, 'identity' => identity, 'hashes' => checkpoint }
1082
+ def preparation_identity
1083
+ { 'version' => 1, 'budget' => input_budget.identity,
1084
+ 'preparer' => collaborator_identity(@text_preparer), 'chunker' => collaborator_identity(@chunker) }
972
1085
  end
973
1086
 
974
- # Recover the flat identifier=>source_hash map {#checkpoint_satisfied?}
975
- # consumes from whichever on-disk shape was parsed. Two shapes:
976
- #
977
- # - versioned (carries a top-level "hashes" key): written by this gem
978
- # version, stamped with the provider/model/dimension identity that
979
- # produced it (see #checkpoint_payload). A stamped identity that
980
- # disagrees with {#current_checkpoint_identity} — a same-dimension
981
- # model switch, the P1 finding this exists to close — means nothing
982
- # here can say which individual hits are still good, so the *whole*
983
- # checkpoint is discarded rather than trusted per-unit.
984
- # - flat (every checkpoint written before this gem version): carries no
985
- # identity at all. When this run tracks identity (a resolved_config
986
- # was given), "no identity recorded" is indistinguishable from "the
987
- # identity that produced this changed" — so it is discarded the same
988
- # way: one full re-embed, after which every checkpoint this gem
989
- # writes is stamped and can be trusted again. When this run has no
990
- # resolved_config either there is nothing to compare against, and the
991
- # flat map is trusted exactly as every prior gem version did.
992
- def checkpoint_hashes(data)
993
- return data unless data.is_a?(Hash)
994
-
995
- current = current_checkpoint_identity
996
- return checkpoint_hashes_versioned(data, current) if data.key?('hashes')
997
- return data if current.nil?
1087
+ def collaborator_identity(object)
1088
+ object.respond_to?(:preparation_identity) ? object.preparation_identity : object&.class&.name
1089
+ end
998
1090
 
999
- warn '[woods] checkpoint.json predates embedding-identity tracking and cannot be ' \
1000
- 'verified against the current provider/model — discarding it and re-embedding ' \
1001
- 'every unit once so future checkpoints are stamped and can be trusted safely.'
1002
- {}
1091
+ def checkpoint_payload(checkpoint)
1092
+ { 'schema_version' => CHECKPOINT_SCHEMA_VERSION, 'identity' => current_checkpoint_identity,
1093
+ 'preparation' => preparation_identity, 'hashes' => checkpoint,
1094
+ 'prepared_inputs' => @checkpoint_inputs.slice(*checkpoint.keys) }
1003
1095
  end
1004
1096
 
1005
- def checkpoint_hashes_versioned(data, current)
1006
- stamped = data['identity']
1007
- return data['hashes'] || {} if current.nil? || stamped == current
1097
+ def checkpoint_hashes(data)
1098
+ valid = data.is_a?(Hash) && data['schema_version'] == CHECKPOINT_SCHEMA_VERSION &&
1099
+ data['identity'] == current_checkpoint_identity && data['preparation'] == preparation_identity &&
1100
+ data['hashes'].is_a?(Hash) && data['prepared_inputs'].is_a?(Hash)
1101
+ unless valid
1102
+ warn '[woods] checkpoint.json has no matching embedding identity and preparation policy; ' \
1103
+ 'discarding the checkpoint and re-embedding every unit once.'
1104
+ return {}
1105
+ end
1008
1106
 
1009
- warn '[woods] checkpoint.json was stamped for a different embedding identity ' \
1010
- "(#{stamped.inspect} vs current #{current.inspect}) — the provider or model " \
1011
- 'changed since the last run. Discarding the checkpoint and re-embedding every ' \
1012
- 'unit so no stale-model vector survives.'
1013
- {}
1107
+ @checkpoint_inputs = data['prepared_inputs']
1108
+ data['hashes']
1014
1109
  end
1015
1110
 
1016
1111
  # Returns true when the vector store can actually be dumped to
@@ -1055,16 +1150,18 @@ module Woods
1055
1150
  end
1056
1151
 
1057
1152
  # Does +object+ define +method_name+ itself, rather than inheriting the
1058
- # vector-store interface's default stub?
1153
+ # storage interfaces' default stubs?
1059
1154
  #
1060
1155
  # @param object [Object] the adapter under test
1061
1156
  # @param method_name [Symbol]
1062
1157
  # @return [Boolean]
1063
1158
  def implements_own?(object, method_name)
1064
1159
  return false unless object.respond_to?(method_name)
1065
- return true unless defined?(Storage::VectorStore::Interface)
1066
1160
 
1067
- object.method(method_name).owner != Storage::VectorStore::Interface
1161
+ interfaces = []
1162
+ interfaces << Storage::VectorStore::Interface if defined?(Storage::VectorStore::Interface)
1163
+ interfaces << Storage::MetadataStore::Interface if defined?(Storage::MetadataStore::Interface)
1164
+ !interfaces.include?(object.method(method_name).owner)
1068
1165
  end
1069
1166
 
1070
1167
  # Persist stores to a timestamped dump directory, write +woods.json+,
@@ -0,0 +1,67 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Woods
4
+ class Error < StandardError; end unless defined?(Woods::Error)
5
+
6
+ module Embedding
7
+ # A refusal, never a request for the retry wrapper to repeat the same input.
8
+ class InputLimitError < Woods::Error; end
9
+
10
+ # Counts used for admission are explicitly distinguished from sizing estimates.
11
+ # Byte BPE starts with UTF-8 bytes and only merges them: its token count cannot
12
+ # exceed bytesize. This bound applies to the known OpenAI embedding encodings,
13
+ # not arbitrary tokenizers (whose normalization or special tokens may differ).
14
+ # Encoding mapping and token-byte examples:
15
+ # https://developers.openai.com/cookbook/examples/how_to_count_tokens_with_tiktoken
16
+ class InputBudget
17
+ attr_reader :limit, :method, :model
18
+
19
+ def initialize(limit:, model: nil, method: 'estimate', chars_per_token: 1.2)
20
+ raise ArgumentError, 'input limit must be positive' unless limit.is_a?(Integer) && limit.positive?
21
+ raise ArgumentError, 'chars_per_token must be positive' unless chars_per_token.positive?
22
+
23
+ @limit = limit
24
+ @model = model
25
+ @method = method
26
+ @chars_per_token = chars_per_token
27
+ end
28
+
29
+ def self.for(provider, limit:, chars_per_token:)
30
+ supplied = provider.input_budget if provider.respond_to?(:input_budget)
31
+ return new(limit: limit, chars_per_token: chars_per_token) unless supplied
32
+ return supplied unless limit < supplied.limit
33
+
34
+ supplied.with_limit(limit)
35
+ end
36
+
37
+ # Preserve the provider's counting policy under a stricter caller cap.
38
+ def with_limit(value)
39
+ self.class.new(limit: [limit, value].min, model: model, method: method, chars_per_token: @chars_per_token)
40
+ end
41
+
42
+ def count(text)
43
+ utf8 = text.encode(Encoding::UTF_8)
44
+ raise InputLimitError, 'Embedding input requires valid UTF-8 text' unless utf8.valid_encoding?
45
+
46
+ method == 'utf8_bytes_bound' ? utf8.bytesize : (utf8.length / @chars_per_token).ceil
47
+ rescue EncodingError
48
+ raise InputLimitError, 'Embedding input requires valid UTF-8 text'
49
+ end
50
+
51
+ def fits?(text)
52
+ count(text) <= limit
53
+ end
54
+
55
+ def validate!(text)
56
+ size = count(text)
57
+ return text if size <= limit
58
+
59
+ raise InputLimitError, "Embedding input limit exceeded (#{method}: #{size}, limit: #{limit}); split the input"
60
+ end
61
+
62
+ def identity
63
+ { 'method' => method, 'limit' => limit, 'model' => model, 'chars_per_token' => @chars_per_token }
64
+ end
65
+ end
66
+ end
67
+ end