woods 2.0.0.beta1 → 2.0.0.beta3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (221) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +400 -1
  3. data/CONTRIBUTING.md +224 -9
  4. data/README.md +7 -3
  5. data/SECURITY.md +9 -6
  6. data/docs/AGENT_GUIDE.md +83 -4
  7. data/docs/AGENT_SETUP.md +82 -1
  8. data/docs/BACKEND_MATRIX.md +20 -0
  9. data/docs/CLIENT_HOOKS.md +111 -0
  10. data/docs/CONFIGURATION_REFERENCE.md +233 -13
  11. data/docs/CONSOLE_MCP_SETUP.md +35 -5
  12. data/docs/DOCKER_SETUP.md +21 -2
  13. data/docs/EVALUATION.md +464 -1
  14. data/docs/EXTRACTOR_REFERENCE.md +36 -5
  15. data/docs/FAQ.md +11 -12
  16. data/docs/GETTING_STARTED.md +17 -5
  17. data/docs/INCREMENTAL_EXTRACTION.md +158 -2
  18. data/docs/INDEX_LAYOUT.md +382 -0
  19. data/docs/INTERNALS.md +15 -7
  20. data/docs/MCP_SERVERS.md +221 -5
  21. data/docs/MCP_TOOL_COOKBOOK.md +33 -18
  22. data/docs/NOTION_INTEGRATION.md +13 -0
  23. data/docs/OBSIDIAN_INTEGRATION.md +57 -9
  24. data/docs/PUBLISHED_INDEX.md +71 -0
  25. data/docs/README.md +7 -0
  26. data/docs/RETRIEVAL_GUIDE.md +253 -11
  27. data/docs/RUNTIME_TRACING.md +71 -0
  28. data/docs/SOURCE_FRESHNESS.md +143 -0
  29. data/docs/TROUBLESHOOTING.md +117 -5
  30. data/docs/UNBLOCKED_INTEGRATION.md +25 -0
  31. data/docs/UPGRADING_TO_2.md +44 -22
  32. data/docs/WATCH_DAEMON.md +259 -59
  33. data/exe/woods-agent-config +6 -0
  34. data/exe/woods-extract +5 -0
  35. data/exe/woods-hook-context +6 -0
  36. data/lib/generators/woods/templates/woods.rb.tt +1 -3
  37. data/lib/tasks/woods.rake +47 -397
  38. data/lib/woods/agent_configuration/applier.rb +133 -0
  39. data/lib/woods/agent_configuration/cli.rb +101 -0
  40. data/lib/woods/agent_configuration/cli_options.rb +29 -0
  41. data/lib/woods/agent_configuration/document.rb +105 -0
  42. data/lib/woods/agent_configuration/error.rb +7 -0
  43. data/lib/woods/agent_configuration/launcher.rb +75 -0
  44. data/lib/woods/agent_configuration/layout.rb +59 -0
  45. data/lib/woods/agent_configuration/managed_section.rb +62 -0
  46. data/lib/woods/agent_configuration/plan.rb +98 -0
  47. data/lib/woods/agent_configuration/plan_diff.rb +38 -0
  48. data/lib/woods/agent_configuration/planned_files.rb +61 -0
  49. data/lib/woods/agent_configuration/planner.rb +63 -0
  50. data/lib/woods/agent_configuration/planner_validation.rb +77 -0
  51. data/lib/woods/agent_configuration/preflight.rb +100 -0
  52. data/lib/woods/agent_configuration/recovery.rb +49 -0
  53. data/lib/woods/ast/node.rb +2 -0
  54. data/lib/woods/ast/parser.rb +38 -5
  55. data/lib/woods/atomic_file.rb +133 -3
  56. data/lib/woods/builder.rb +21 -5
  57. data/lib/woods/cache/cache_middleware.rb +28 -7
  58. data/lib/woods/cache/cache_store.rb +4 -5
  59. data/lib/woods/change_set.rb +5 -4
  60. data/lib/woods/console/credential_index.rb +20 -2
  61. data/lib/woods/console/credential_scanner.rb +14 -14
  62. data/lib/woods/console/credential_scanner_registry.rb +36 -0
  63. data/lib/woods/console/embedded_executor.rb +1 -1
  64. data/lib/woods/console/encrypted_credential_snapshot.rb +16 -0
  65. data/lib/woods/console/rack_middleware.rb +22 -13
  66. data/lib/woods/console/server.rb +18 -16
  67. data/lib/woods/dependency_graph.rb +65 -13
  68. data/lib/woods/embedding/corpus.rb +94 -0
  69. data/lib/woods/embedding/indexer.rb +90 -46
  70. data/lib/woods/embedding/openai.rb +17 -6
  71. data/lib/woods/evaluation/ablation_executor.rb +6 -1
  72. data/lib/woods/evaluation/ablation_timed_executor.rb +22 -4
  73. data/lib/woods/export/typed_reader.rb +56 -0
  74. data/lib/woods/extractor.rb +557 -228
  75. data/lib/woods/extractors/action_cable_extractor.rb +3 -1
  76. data/lib/woods/extractors/behavioral_profile.rb +9 -7
  77. data/lib/woods/extractors/caching_extractor.rb +3 -1
  78. data/lib/woods/extractors/concern_extractor.rb +64 -6
  79. data/lib/woods/extractors/configuration_extractor.rb +7 -3
  80. data/lib/woods/extractors/controller_extractor.rb +13 -4
  81. data/lib/woods/extractors/database_view_extractor.rb +3 -1
  82. data/lib/woods/extractors/decorator_extractor.rb +3 -1
  83. data/lib/woods/extractors/engine_extractor.rb +3 -1
  84. data/lib/woods/extractors/event_extractor.rb +4 -2
  85. data/lib/woods/extractors/factory_extractor.rb +3 -1
  86. data/lib/woods/extractors/graphql_extractor.rb +8 -2
  87. data/lib/woods/extractors/i18n_extractor.rb +3 -1
  88. data/lib/woods/extractors/job_extractor.rb +6 -19
  89. data/lib/woods/extractors/lib_extractor.rb +3 -1
  90. data/lib/woods/extractors/mailer_extractor.rb +20 -5
  91. data/lib/woods/extractors/manager_extractor.rb +3 -1
  92. data/lib/woods/extractors/method_parameters.rb +53 -0
  93. data/lib/woods/extractors/middleware_argument.rb +65 -0
  94. data/lib/woods/extractors/middleware_extractor.rb +9 -3
  95. data/lib/woods/extractors/migration_extractor.rb +3 -1
  96. data/lib/woods/extractors/model_extractor.rb +39 -33
  97. data/lib/woods/extractors/package_extractor.rb +24 -4
  98. data/lib/woods/extractors/phlex_extractor.rb +3 -1
  99. data/lib/woods/extractors/policy_extractor.rb +3 -1
  100. data/lib/woods/extractors/poro_extractor.rb +3 -1
  101. data/lib/woods/extractors/pundit_extractor.rb +3 -1
  102. data/lib/woods/extractors/rails_source_extractor.rb +4 -2
  103. data/lib/woods/extractors/rake_task_extractor.rb +4 -2
  104. data/lib/woods/extractors/route_extractor.rb +3 -1
  105. data/lib/woods/extractors/route_helper_resolver.rb +10 -33
  106. data/lib/woods/extractors/scheduled_job_extractor.rb +41 -15
  107. data/lib/woods/extractors/serializer_extractor.rb +4 -2
  108. data/lib/woods/extractors/service_extractor.rb +3 -1
  109. data/lib/woods/extractors/shared_dependency_scanner.rb +2 -2
  110. data/lib/woods/extractors/shared_utility_methods.rb +27 -15
  111. data/lib/woods/extractors/source_nesting.rb +1 -1
  112. data/lib/woods/extractors/state_machine_extractor.rb +3 -1
  113. data/lib/woods/extractors/test_mapping_extractor.rb +3 -1
  114. data/lib/woods/extractors/validator_extractor.rb +3 -1
  115. data/lib/woods/extractors/view_component_extractor.rb +3 -1
  116. data/lib/woods/extractors/view_template_extractor.rb +3 -1
  117. data/lib/woods/flow_assembler.rb +87 -8
  118. data/lib/woods/flow_precomputer.rb +44 -7
  119. data/lib/woods/gem_mapper.rb +2 -0
  120. data/lib/woods/git_history.rb +116 -0
  121. data/lib/woods/graph_analyzer.rb +195 -63
  122. data/lib/woods/hooks/context_cli.rb +54 -0
  123. data/lib/woods/hooks/context_event.rb +88 -0
  124. data/lib/woods/hooks/context_hint.rb +73 -0
  125. data/lib/woods/hooks/context_impact.rb +77 -0
  126. data/lib/woods/hooks/context_output.rb +47 -0
  127. data/lib/woods/hooks/context_state.rb +102 -0
  128. data/lib/woods/hooks/refresh.rb +79 -0
  129. data/lib/woods/hooks/rule_projection.rb +78 -0
  130. data/lib/woods/input_rules.rb +19 -0
  131. data/lib/woods/mcp/bearer_auth.rb +20 -12
  132. data/lib/woods/mcp/bootstrapper.rb +62 -0
  133. data/lib/woods/mcp/index_reader.rb +323 -160
  134. data/lib/woods/mcp/initialization_guidance.rb +27 -0
  135. data/lib/woods/mcp/origin_guard.rb +17 -9
  136. data/lib/woods/mcp/published_lexical_retriever.rb +115 -0
  137. data/lib/woods/mcp/renderers/markdown_renderer.rb +8 -1
  138. data/lib/woods/mcp/renderers/plain_renderer.rb +7 -1
  139. data/lib/woods/mcp/search_results.rb +74 -0
  140. data/lib/woods/mcp/server.rb +158 -37
  141. data/lib/woods/mcp/tool_contract.rb +2 -0
  142. data/lib/woods/mcp/tool_response_renderer.rb +25 -0
  143. data/lib/woods/mcp/traversal_evidence.rb +113 -0
  144. data/lib/woods/mcp/traversal_evidence_index.rb +100 -0
  145. data/lib/woods/mcp/traversal_evidence_page.rb +41 -0
  146. data/lib/woods/mcp/traversal_evidence_text.rb +52 -0
  147. data/lib/woods/notion/exporter.rb +56 -17
  148. data/lib/woods/obsidian/destination_plan.rb +98 -0
  149. data/lib/woods/obsidian/name_mapper.rb +19 -3
  150. data/lib/woods/obsidian/note_builder.rb +19 -10
  151. data/lib/woods/obsidian/vault_exporter.rb +88 -32
  152. data/lib/woods/operator/pipeline_guard.rb +18 -13
  153. data/lib/woods/path_dispatcher.rb +7 -1
  154. data/lib/woods/payload_store.rb +29 -15
  155. data/lib/woods/railtie.rb +3 -3
  156. data/lib/woods/railtie_support.rb +12 -12
  157. data/lib/woods/rake_helpers.rb +392 -0
  158. data/lib/woods/resilience/graph_invariant_validator/membership_checks.rb +71 -0
  159. data/lib/woods/resilience/graph_invariant_validator/node_checks.rb +61 -0
  160. data/lib/woods/resilience/graph_invariant_validator/reverse_relationship_checks.rb +46 -0
  161. data/lib/woods/resilience/graph_invariant_validator.rb +119 -0
  162. data/lib/woods/resilience/index_validator/graph_checks.rb +80 -0
  163. data/lib/woods/resilience/index_validator.rb +112 -23
  164. data/lib/woods/retrieval/context_assembler.rb +50 -15
  165. data/lib/woods/retrieval/lexical_assembler.rb +73 -0
  166. data/lib/woods/retrieval/lexical_index.rb +119 -0
  167. data/lib/woods/retrieval/ranker.rb +4 -2
  168. data/lib/woods/retrieval/scope.rb +108 -0
  169. data/lib/woods/retrieval/scoped_graph_store.rb +32 -0
  170. data/lib/woods/retrieval/scoped_vector_store.rb +55 -0
  171. data/lib/woods/retrieval/search_executor.rb +86 -27
  172. data/lib/woods/retrieval/source_evidence.rb +200 -0
  173. data/lib/woods/retriever.rb +98 -22
  174. data/lib/woods/ruby_analyzer/trace_enricher.rb +80 -38
  175. data/lib/woods/session_tracer/middleware.rb +10 -12
  176. data/lib/woods/session_tracer/redis_store.rb +22 -6
  177. data/lib/woods/session_tracer/session_flow_assembler.rb +23 -17
  178. data/lib/woods/session_tracer/solid_cache_coordination.rb +6 -4
  179. data/lib/woods/session_tracer/unit_resolver.rb +63 -0
  180. data/lib/woods/source_inputs/consumer_errors.rb +27 -0
  181. data/lib/woods/source_inputs/handoff.rb +102 -0
  182. data/lib/woods/source_inputs/launcher.rb +157 -0
  183. data/lib/woods/source_inputs/manifest.rb +124 -0
  184. data/lib/woods/source_inputs/private_key.rb +55 -0
  185. data/lib/woods/source_inputs/scanner.rb +171 -0
  186. data/lib/woods/source_inputs/scopes.rb +71 -0
  187. data/lib/woods/source_inputs/session.rb +214 -0
  188. data/lib/woods/source_inputs/status.rb +84 -0
  189. data/lib/woods/source_inputs/verifier.rb +107 -0
  190. data/lib/woods/storage/metadata_store.rb +25 -25
  191. data/lib/woods/storage/pgvector.rb +29 -8
  192. data/lib/woods/storage/qdrant.rb +17 -7
  193. data/lib/woods/storage/vector_store.rb +18 -6
  194. data/lib/woods/tasks.rb +3 -2
  195. data/lib/woods/temporal/json_snapshot_store.rb +29 -8
  196. data/lib/woods/unblocked/exporter.rb +59 -70
  197. data/lib/woods/version.rb +1 -1
  198. data/lib/woods/watch/boot_snapshot.rb +52 -0
  199. data/lib/woods/watch/daemon.rb +136 -28
  200. data/lib/woods/watch/listen_watcher.rb +4 -0
  201. data/lib/woods/watch/polling_watcher.rb +5 -1
  202. data/lib/woods/watch/status.rb +20 -15
  203. data/lib/woods/watch/tree_scan.rb +21 -13
  204. data/lib/woods/watch/watcher.rb +4 -1
  205. data/lib/woods.rb +135 -11
  206. data/plugin/.claude-plugin/plugin.json +1 -1
  207. data/plugin/hooks/adapters/normalize.jq +15 -0
  208. data/plugin/hooks/adapters/normalize.rb +63 -0
  209. data/plugin/hooks/hooks.json +20 -0
  210. data/plugin/hooks/woods-context.sh +50 -0
  211. data/plugin/hooks/woods-input-rules.sh +159 -0
  212. data/plugin/hooks/woods-opencode.mjs +65 -0
  213. data/plugin/hooks/woods-post-edit.sh +2 -225
  214. data/plugin/hooks/woods-refresh.sh +260 -0
  215. data/plugin/hooks/woods-session-start.sh +47 -55
  216. data/plugin/skills/woods-agent-enable/SKILL.md +13 -0
  217. data/plugin/skills/woods-diagnose/SKILL.md +288 -1
  218. data/plugin/skills/woods-investigate/SKILL.md +106 -0
  219. data/plugin/skills/woods-mcp-config/SKILL.md +89 -1
  220. data/plugin/skills/woods-setup/SKILL.md +107 -6
  221. metadata +84 -5
@@ -32,9 +32,14 @@ module Woods
32
32
  # enforce_dependencies: package units (Task 7)
33
33
  # commit_count, change_frequency: git enrichment (Task 5)
34
34
  NODE_ATTRIBUTE_KEYS = %i[
35
- database table foreign_key_tables package enforce_dependencies commit_count change_frequency
35
+ database table foreign_key_tables package enforce_dependencies commit_count change_frequency kind
36
36
  ].freeze
37
37
 
38
+ # These extractors describe a whole source file rather than one constant.
39
+ # Other types remain unclassified: a path-like identifier is not evidence
40
+ # that a unit is a profile, and an unmarked unit need not name a constant.
41
+ FILE_PROFILE_TYPES = %i[caching configuration test_mapping rails_source gem_source].freeze
42
+
38
43
  # Optional edge keys beyond target and via. `through` is the through
39
44
  # association name; `through_db` is the through model's resolved
40
45
  # database, emitted only when present; `disable_joins` is emitted only
@@ -80,14 +85,14 @@ module Woods
80
85
  }.merge(node_attributes_from(unit))
81
86
 
82
87
  (@edges[unit.identifier] ||= {})[unit.type] =
83
- unit.dependencies.map { |d| { target: d[:target], via: d[:via] }.merge(self.class.edge_attributes(d)) }
88
+ self.class.normalize_edges(unit.dependencies, strict: true)
84
89
  (@file_map[unit.file_path] ||= Set.new).add(unit.identifier) if unit.file_path
85
90
 
86
91
  # Type index for filtering (Set-based for O(1) insert)
87
92
  (@type_index[unit.type] ||= Set.new).add(unit.identifier)
88
93
 
89
94
  # Build reverse edges (Set-based for O(1) insert)
90
- unit.dependencies.each do |dep|
95
+ @edges[unit.identifier][unit.type].each do |dep|
91
96
  (@reverse[dep[:target]] ||= Set.new).add(unit.identifier)
92
97
  (@reverse_via[[dep[:target], dep[:via]]] ||= Set.new).add(unit.identifier)
93
98
  end
@@ -645,7 +650,7 @@ module Woods
645
650
  # @return [Hash{Symbol => Object}] normalized, nil-free
646
651
  def node_attributes_from(unit)
647
652
  metadata = unit.respond_to?(:metadata) && unit.metadata.is_a?(Hash) ? unit.metadata : {}
648
- attrs = {}
653
+ attrs = self.class.file_profile_attributes(unit.type)
649
654
  attrs[:database] = metadata[:database] unless metadata[:database].nil?
650
655
  attrs[:table] = metadata[:table_name] if unit.type == :model && !metadata[:table_name].nil?
651
656
  tables = Array(metadata[:foreign_keys]).filter_map { |fk| fk[:to_table] || fk['to_table'] if fk.is_a?(Hash) }
@@ -729,14 +734,15 @@ module Woods
729
734
  # `nodes` and `edges` stay keyed on the bare identifier, holding the
730
735
  # **primary** node — the type that sorts first. Identifiers registered
731
736
  # under more than one type put their remaining nodes in `variants`, a flat
732
- # array that is **omitted entirely when empty**. So the serialized form of
733
- # a graph with no collisions is byte-for-byte what it has always been, and
734
- # a `dependency_graph.json` written before this change loads through
737
+ # array that is **omitted entirely when empty**. A
738
+ # `dependency_graph.json` written before this change loads through
735
739
  # {.from_h} unmodified: it simply carries no `variants`.
736
740
  #
737
741
  # `reverse`, `file_map` and `type_index` need no new shape. They are
738
742
  # already identifier-valued sets, so a Scenic view `reports` and a factory
739
743
  # `reports` each contribute to their own type bucket and their own path.
744
+ # The additive `reverse_via` index preserves typed source identities and
745
+ # complete relationship records without changing the legacy `reverse` map.
740
746
  #
741
747
  # @return [Hash] Complete graph data
742
748
  def to_h
@@ -747,6 +753,7 @@ module Woods
747
753
  nodes: key_sorted(self.class.relativize_nodes(primary_nodes, root)),
748
754
  edges: key_sorted(primary_edges),
749
755
  reverse: key_sorted(@reverse.transform_values { |ids| ids.to_a.sort }),
756
+ reverse_via: published_reverse_via,
750
757
  file_map: key_sorted(self.class.relativize_file_map(@file_map, root)),
751
758
  type_index: key_sorted(@type_index.transform_values { |ids| ids.to_a.sort }),
752
759
  stats: {
@@ -770,6 +777,23 @@ module Woods
770
777
  end
771
778
  private :key_sorted
772
779
 
780
+ # Reverse buckets are derived from every typed forward registration, rather
781
+ # than the bare-name in-memory filter index, which cannot preserve variants
782
+ # or distinguish association attributes. Legacy nil relationships stay nil.
783
+ def published_reverse_via
784
+ buckets = {}
785
+ @edges.each do |source, by_type|
786
+ by_type.each do |type, edges|
787
+ edges.each do |edge|
788
+ record = { source: source, source_type: type }.merge(edge.except(:target))
789
+ (buckets[edge[:target]] ||= []) << record
790
+ end
791
+ end
792
+ end
793
+ key_sorted(buckets.transform_values { |records| records.sort_by { |record| JSON.generate(record) } })
794
+ end
795
+ private :published_reverse_via
796
+
773
797
  # A copy whose edge containers are the caller's alone.
774
798
  #
775
799
  # `dup` copies only the top level, so `to_h[:edges][id]` used to be the
@@ -783,6 +807,9 @@ module Woods
783
807
  def detached_snapshot(memo)
784
808
  snapshot = memo.dup
785
809
  snapshot[:edges] = memo[:edges].transform_values { |list| list.map(&:dup) }
810
+ snapshot[:reverse_via] = memo[:reverse_via].transform_values do |records|
811
+ records.map { |record| record.transform_values { |value| value.is_a?(String) ? value.dup : value } }
812
+ end
786
813
  if memo.key?(:variants)
787
814
  snapshot[:variants] = memo[:variants].map do |record|
788
815
  record.merge(edges: record[:edges].map(&:dup))
@@ -794,7 +821,12 @@ module Woods
794
821
 
795
822
  # @return [Hash{String => Hash}] identifier => primary node
796
823
  def primary_nodes
797
- @nodes.transform_values { |nodes| primary_of(nodes) }
824
+ # Late enrichment and JSON loading can insert optional fields in a
825
+ # different order. Canonicalize like variant_records so a graph's JSON
826
+ # fingerprint is stable across full and incremental publication.
827
+ @nodes.transform_values do |nodes|
828
+ primary_of(nodes).slice(:type, :file_path, :namespace, *NODE_ATTRIBUTE_KEYS)
829
+ end
798
830
  end
799
831
  private :primary_nodes
800
832
 
@@ -969,10 +1001,17 @@ module Woods
969
1001
  # @param node [Hash]
970
1002
  # @return [Hash{Symbol => Object}]
971
1003
  def self.persisted_node_attributes(node)
972
- NODE_ATTRIBUTE_KEYS.each_with_object({}) do |key, attrs|
1004
+ attributes = NODE_ATTRIBUTE_KEYS.each_with_object({}) do |key, attrs|
973
1005
  value = node.key?(key) ? node[key] : node[key.to_s]
974
1006
  attrs[key] = normalize_node_attribute(key, value) unless value.nil?
975
1007
  end
1008
+ attributes.merge(file_profile_attributes(node[:type] || node['type']))
1009
+ end
1010
+
1011
+ # Backfill the additive marker when an older graph is republished, so
1012
+ # unchanged nodes in an incremental run agree with a fresh full graph.
1013
+ def self.file_profile_attributes(type)
1014
+ FILE_PROFILE_TYPES.include?(type&.to_sym) ? { kind: 'file_profile' } : {}
976
1015
  end
977
1016
 
978
1017
  # Edge attributes present in a dependency or persisted edge hash.
@@ -1032,7 +1071,10 @@ module Woods
1032
1071
  graph.instance_variable_set(:@edges, edges)
1033
1072
 
1034
1073
  raw_reverse = data[:reverse] || data['reverse'] || {}
1035
- graph.instance_variable_set(:@reverse, raw_reverse.transform_values { |v| v.is_a?(Set) ? v : Set.new(v) })
1074
+ reverse = raw_reverse.each_with_object({}) do |(target, sources), index|
1075
+ (index[target.to_s] ||= Set.new).merge(sources)
1076
+ end
1077
+ graph.instance_variable_set(:@reverse, reverse)
1036
1078
 
1037
1079
  raw_file_map = data[:file_map] || data['file_map'] || {}
1038
1080
  graph.instance_variable_set(:@file_map, absolutize_file_map(normalize_file_map(raw_file_map), root))
@@ -1126,7 +1168,12 @@ module Woods
1126
1168
  }.merge(persisted_node_attributes(node))
1127
1169
  end
1128
1170
 
1129
- # Normalize edge data from either old format (bare strings) or new format (hashes).
1171
+ # Normalize fresh dependencies and persisted edges to the same identity:
1172
+ # string targets and symbol relationship labels. Extractors may emit
1173
+ # symbolic external targets (such as :http_api); retaining those symbols
1174
+ # after a JSON-loaded baseline splits reverse indexes on re-registration.
1175
+ # Unit dependency metadata stays unchanged; only graph records normalize.
1176
+ # Also accepts the old bare-string edge format.
1130
1177
  #
1131
1178
  # ROUND-TRIP INVARIANT (do not break when refactoring):
1132
1179
  # DependencyGraph#to_h -> JSON.generate -> JSON.parse -> DependencyGraph.from_h
@@ -1147,15 +1194,20 @@ module Woods
1147
1194
  # explicit data conversion.
1148
1195
  #
1149
1196
  # @param edges [Array] Edge entries — either strings or hashes
1197
+ # @param strict [Boolean] require hashes for fresh units; legacy loads stay tolerant
1150
1198
  # @return [Array<Hash>] Normalized edges with :target and :via keys
1151
- def self.normalize_edges(edges)
1199
+ def self.normalize_edges(edges, strict: false)
1200
+ raise ArgumentError, 'Fresh graph dependencies must be an array' if strict && !edges.is_a?(Array)
1201
+
1152
1202
  return [] unless edges.is_a?(Array)
1153
1203
 
1154
1204
  edges.map do |edge|
1205
+ raise ArgumentError, 'Fresh graph dependencies must be hashes' if strict && !edge.is_a?(Hash)
1206
+
1155
1207
  if edge.is_a?(String)
1156
1208
  { target: edge, via: nil }
1157
1209
  elsif edge.is_a?(Hash)
1158
- { target: edge[:target] || edge['target'], via: (edge[:via] || edge['via'])&.to_sym }
1210
+ { target: (edge[:target] || edge['target'])&.to_s, via: (edge[:via] || edge['via'])&.to_sym }
1159
1211
  .merge(edge_attributes(edge))
1160
1212
  else
1161
1213
  { target: edge.to_s, via: nil }
@@ -0,0 +1,94 @@
1
+ # frozen_string_literal: true
2
+
3
+ require 'json'
4
+ require_relative '../atomic_file'
5
+ require_relative '../generation'
6
+
7
+ module Woods
8
+ class Error < StandardError; end unless defined?(Woods::Error)
9
+
10
+ module Embedding
11
+ # Collect the complete extraction input before an embedding run can mutate
12
+ # stores. Native publications use their authoritative listings, never a glob
13
+ # that could mistake a missing/corrupt payload for intentional deletion.
14
+ class Corpus
15
+ class Incomplete < Woods::Error; end
16
+
17
+ def initialize(output_dir)
18
+ @output_dir = output_dir.to_s
19
+ end
20
+
21
+ def load
22
+ native? ? published_units : legacy_units
23
+ rescue IOError, SystemCallError, JSON::ParserError, EncodingError, ArgumentError => e
24
+ raise Incomplete, "Embedding input incomplete: #{e.message}. Repair or rebuild extraction before embedding."
25
+ end
26
+
27
+ private
28
+
29
+ def native?
30
+ path = File.join(@output_dir, Generation::FILENAME)
31
+ if File.exist?(path)
32
+ marker = JSON.parse(AtomicFile.read(path))
33
+ raise IOError, 'invalid generation marker' unless marker.is_a?(Hash)
34
+ end
35
+
36
+ if !marker || marker['payload'].nil?
37
+ if File.directory?(File.join(@output_dir, 'payloads'))
38
+ raise IOError, 'missing publication pointer beside payloads'
39
+ end
40
+
41
+ return false
42
+ end
43
+
44
+ validate_marker(marker)
45
+ true
46
+ end
47
+
48
+ def validate_marker(marker)
49
+ strings_valid = %w[payload token].all? { |key| marker[key].is_a?(String) && !marker[key].empty? }
50
+ return if strings_valid && marker['number'].is_a?(Integer) && marker['number'].positive?
51
+
52
+ raise IOError, 'invalid published generation marker'
53
+ end
54
+
55
+ def published_units
56
+ require_relative '../mcp/index_reader'
57
+
58
+ reader = MCP::IndexReader.new(@output_dir)
59
+ reader.with_pinned_generation do
60
+ # Generation's compatibility fallback to the root is useful for
61
+ # readers, but cannot prove completeness for destructive reconciliation.
62
+ if reader.payload_dir.expand_path == Pathname.new(@output_dir).expand_path
63
+ raise IOError,
64
+ 'published payload missing or invalid'
65
+ end
66
+
67
+ validate_counts(reader.manifest)
68
+ reader.each_unit.to_a
69
+ end
70
+ end
71
+
72
+ def validate_counts(manifest)
73
+ counts = manifest.is_a?(Hash) && manifest['counts']
74
+ valid = counts.is_a?(Hash) && counts.all? do |dir, count|
75
+ MCP::IndexReader::TYPE_DIRS.include?(dir) && count.is_a?(Integer) && count >= 0
76
+ end
77
+ raise IOError, 'invalid published manifest counts' unless valid
78
+ end
79
+
80
+ # Pre-pointer indexes may use arbitrary filenames and have no manifest
81
+ # or authoritative listing. Keep that input contract, but never silently
82
+ # omit unreadable JSON before reconciling persisted identities.
83
+ def legacy_units
84
+ Dir.glob(File.join(@output_dir, '**', '*.json')).filter_map do |path|
85
+ relative = path.delete_prefix("#{@output_dir}/")
86
+ next if relative.start_with?('dumps/', 'payloads/') || File.basename(path) == 'checkpoint.json'
87
+
88
+ data = JSON.parse(AtomicFile.read(path))
89
+ data if data.is_a?(Hash) && data.key?('type') && data.key?('identifier')
90
+ end
91
+ end
92
+ end
93
+ end
94
+ end
@@ -119,40 +119,14 @@ module Woods
119
119
 
120
120
  private
121
121
 
122
- # Where this run's units are read from — the published generation's
123
- # payload, or the output root for a flat index.
124
- #
125
- # Globbing the output root would walk `payloads/` and ingest every
126
- # retained generation, so each unit would be embedded once per retained
127
- # payload. It also picks up `dumps/`, which holds no unit JSON but is
128
- # needlessly large to walk.
129
- #
130
- # @return [String]
131
- def units_dir
132
- Woods::Generation.new(output_dir: @output_dir).payload_dir.to_s
133
- end
134
-
135
122
  def load_units
136
- Dir.glob(File.join(units_dir, '**', '*.json')).filter_map do |path|
137
- next if File.basename(path) == 'checkpoint.json'
138
-
139
- # AtomicFile.read, not File.read: a bare read tags the bytes with the
140
- # process's default external encoding (US-ASCII under LANG=C), so a
141
- # single multibyte character in one unit raised EncodingError out of
142
- # JSON.parse and aborted the whole embed run.
143
- data = JSON.parse(AtomicFile.read(path))
144
- # Extraction output also contains index listings (_index.json arrays) and
145
- # summary files (manifest.json, dependency_graph.json, graph_analysis.json)
146
- # that live alongside per-unit JSON. Filter to the unit shape.
147
- data if data.is_a?(Hash) && data.key?('type') && data.key?('identifier')
148
- rescue JSON::ParserError, EncodingError => e
149
- warn "[woods] skipping unreadable unit file #{path} (#{e.class}: #{e.message})"
150
- nil
151
- end
123
+ require_relative 'corpus'
124
+
125
+ Corpus.new(@output_dir).load
152
126
  end
153
127
 
154
- # The invariant: **checkpoint.json never advances over a unit whose
155
- # vector was not durably stored.** Two things uphold it here.
128
+ # The invariant: **checkpoint.json advances only after the intended
129
+ # vector state (including an empty set) is durable.** Two things uphold it here.
156
130
  #
157
131
  # 1. Ordering. For a store whose only durable copy is the dump
158
132
  # (+persistable?+), the checkpoint is written *after*
@@ -170,17 +144,19 @@ module Woods
170
144
  #
171
145
  # 2. Trust, verified. {#checkpoint_satisfied?} honours a checkpoint hit
172
146
  # only when the durable artifact actually holds a vector for that
173
- # unit. A checkpoint that ran ahead of its dump — an older gem with
147
+ # unit, or preparation verifies that it intentionally has no text.
148
+ # A checkpoint that ran ahead of its dump — an older gem with
174
149
  # this bug, an interrupted promote, a store swap — self-heals into a
175
150
  # re-embed instead of stranding the unit forever.
176
151
  def process_units(units, incremental:)
177
152
  prepare_run(incremental: incremental)
178
- units = assign_storage_identities(units)
179
153
  checkpoint = incremental ? load_checkpoint : {}
154
+ units = assign_storage_identities(units, checkpoint: checkpoint)
180
155
  stats = { processed: 0, skipped: 0, errors: 0 }
181
156
 
182
157
  embed_batches(units, checkpoint, stats, incremental: incremental)
183
158
 
159
+ reconcile_empty_units(checkpoint)
184
160
  retire_legacy_identities
185
161
  report_checkpoint_misses
186
162
  vanished = incremental && persistable? ? drop_vanished_units : 0
@@ -196,12 +172,12 @@ module Woods
196
172
  end
197
173
 
198
174
  # Unambiguous existing keys stay stable. A collision uses reversible typed keys.
199
- def assign_storage_identities(units)
175
+ def assign_storage_identities(units, checkpoint:)
200
176
  counts = units.group_by { |unit| unit['identifier'] }.transform_values(&:size)
201
177
  units.map do |unit|
202
178
  id = unit['identifier']
203
179
  typed = StorageIdentity.key(id, unit['type'])
204
- existing = known_storage_key?(typed)
180
+ existing = known_storage_key?(typed, checkpoint: checkpoint)
205
181
  next unit unless counts[id] > 1 || existing || id.start_with?(StorageIdentity::PREFIX)
206
182
 
207
183
  unit.merge('storage_id' => typed)
@@ -224,8 +200,8 @@ module Woods
224
200
  end
225
201
  end
226
202
 
227
- def known_storage_key?(key)
228
- (@persisted_ids || {}).key?(key) || (@durable_ids || {}).key?(key)
203
+ def known_storage_key?(key, checkpoint:)
204
+ (@persisted_ids || {}).key?(key) || (@durable_ids || {}).key?(key) || checkpoint.key?(key)
229
205
  end
230
206
 
231
207
  def retire_legacy_key(legacy)
@@ -243,11 +219,9 @@ module Woods
243
219
  # `woods:embed_incremental` runs evict every genuinely older dump in
244
220
  # favour of copies of the same state.
245
221
  #
246
- # Safe because nothing else mutates the *vector* store on a zero-processed
247
- # run: `prune_superseded_vectors` is reached only from `store_vectors`,
248
- # which runs only for items that were actually embedded. A checkpoint
249
- # self-heal counts as processed, so a run that re-embeds a stranded unit
250
- # still dumps.
222
+ # A zero-text transition can retire vectors without a provider call;
223
+ # @vectors_changed captures that case. Checkpoint self-heals still count
224
+ # as processed, so a run re-embedding a stranded unit also dumps.
251
225
  #
252
226
  # But "nothing embedded" is not "nothing changed" (B-069). `persist_snapshot`
253
227
  # writes the vector dump, the *metadata* dump and the config into one
@@ -266,6 +240,11 @@ module Woods
266
240
  # `@current_identifiers` is what this run saw. Anything in the first and
267
241
  # not the second is stale, and the dump has to be rewritten to drop it.
268
242
  #
243
+ # Existing units can also change metadata without changing source (B-119).
244
+ # Compare their complete records with the promoted metadata snapshot;
245
+ # publishing those changes must not require a provider call or invalidate
246
+ # the source-hash checkpoint contract.
247
+ #
269
248
  # Full runs always dump: they rebuild the store from scratch, so "nothing
270
249
  # processed" there means the store is genuinely empty and the dump must
271
250
  # say so rather than leave a stale one promoted.
@@ -274,7 +253,7 @@ module Woods
274
253
  def snapshot_worth_writing?(stats, vanished, incremental:)
275
254
  return true unless incremental
276
255
 
277
- stats[:processed].positive? || vanished.positive?
256
+ stats[:processed].positive? || vanished.positive? || @metadata_changed || @vectors_changed
278
257
  end
279
258
 
280
259
  # Fraction of the persisted units the vanished-unit sweep may remove
@@ -489,10 +468,36 @@ module Woods
489
468
  @current_identifiers = Set.new
490
469
  @durable_ids = nil
491
470
  @checkpoint_misses = 0
492
- hydrate_persisted_vectors if incremental && persistable?
471
+ @metadata_changed = false
472
+ @vectors_changed = false
473
+ @empty_units = {}
474
+ @persisted_metadata = nil
475
+ prepare_snapshot_stores(incremental: incremental)
476
+ retain_metadata_identities
493
477
  load_durable_store_ids if reconcilable?
494
478
  end
495
479
 
480
+ def prepare_snapshot_stores(incremental:)
481
+ return unless persistable?
482
+
483
+ if incremental
484
+ hydrate_persisted_metadata
485
+ hydrate_persisted_vectors
486
+ else
487
+ # A direct caller may reuse an in-memory adapter for a full run.
488
+ # Its old chunks still need replacement, including by an empty set.
489
+ entries = []
490
+ @vector_store.each_entry { |id, _vector, _metadata| entries << { id: id } }
491
+ @persisted_ids = index_ids_by_identifier(entries)
492
+ end
493
+ end
494
+
495
+ # Source-empty units retain metadata but intentionally have no vectors.
496
+ # Keep those identities in the same deletion and typed-key accounting.
497
+ def retain_metadata_identities
498
+ @persisted_metadata&.each_entry { |identifier, _unit| @persisted_ids[identifier] ||= [] }
499
+ end
500
+
496
501
  # Read back what the durable store currently holds, as base identifiers.
497
502
  #
498
503
  # One enumeration serves both halves of #211:
@@ -618,7 +623,10 @@ module Woods
618
623
  # enumeration that failed) — fall back to trusting the checkpoint.
619
624
  return true if known_ids.nil?
620
625
 
621
- return true if known_ids.key?(storage_id(unit_data))
626
+ return true if known_ids[storage_id(unit_data)]&.any?
627
+ # A source-empty unit intentionally has no vector. Verify that state
628
+ # again rather than treating a missing nonempty vector as a cache hit.
629
+ return true if prepare_texts(unit_data).empty?
622
630
 
623
631
  @checkpoint_misses += 1
624
632
  false
@@ -635,7 +643,22 @@ module Woods
635
643
  def persist_unit_metadata(unit_data)
636
644
  return unless @metadata_store
637
645
 
638
- @metadata_store.store(storage_id(unit_data), unit_data)
646
+ id = storage_id(unit_data)
647
+ @metadata_changed ||= @persisted_metadata && @persisted_metadata.find(id) != unit_data
648
+ @metadata_store.store(id, unit_data)
649
+ end
650
+
651
+ # Compare with the promoted artifact, not a fresh metadata store or the
652
+ # source-only checkpoint. Retain old records until the guarded vanished
653
+ # sweep permits their deletion, just as vector hydration does.
654
+ def hydrate_persisted_metadata
655
+ return unless @metadata_store.respond_to?(:each_entry) && @metadata_store.respond_to?(:bulk_load)
656
+
657
+ require_relative '../index_artifact'
658
+ require_relative '../storage/snapshotter'
659
+
660
+ @persisted_metadata = Storage::Snapshotter::Metadata.load_or_empty(IndexArtifact.new(@output_dir))
661
+ @metadata_store.bulk_load(@persisted_metadata.each_entry)
639
662
  end
640
663
 
641
664
  # Refuse to index a unit whose real identifier already matches the
@@ -653,6 +676,7 @@ module Woods
653
676
  def collect_embed_items(unit_data, items)
654
677
  texts = prepare_texts(unit_data)
655
678
  identifier = storage_id(unit_data)
679
+ @empty_units[identifier] = unit_data['source_hash'] if texts.empty?
656
680
 
657
681
  texts.each_with_index do |text, idx|
658
682
  embed_id = texts.length > 1 ? "#{identifier}#chunk_#{idx}" : identifier
@@ -661,6 +685,26 @@ module Woods
661
685
  end
662
686
  end
663
687
 
688
+ # Defer zero-text deletion until every batch has prepared/embedded
689
+ # successfully. A later provider failure must not retire earlier vectors.
690
+ # Checkpoints retain their source-hash format; a matching no-vector hash
691
+ # is trusted only after verifying the current input still prepares empty.
692
+ def reconcile_empty_units(checkpoint)
693
+ verify_empty_reconciliation!
694
+ @empty_units.each do |identifier, source_hash|
695
+ @vectors_changed ||= @persisted_ids[identifier]&.any?
696
+ prune_identifier(identifier, []) if implements_own?(@vector_store, :delete)
697
+ prune_durable_identifier(identifier, []) if @durable_ids && implements_own?(@vector_store, :delete)
698
+ checkpoint[identifier] = source_hash
699
+ end
700
+ end
701
+
702
+ def verify_empty_reconciliation!
703
+ return unless @empty_units.any? && reconcilable? && @durable_ids.nil?
704
+
705
+ raise Woods::Error, 'Cannot reconcile source-empty units: existing vector IDs could not be read'
706
+ end
707
+
664
708
  def prepare_texts(unit_data) # rubocop:disable Metrics/CyclomaticComplexity, Metrics/PerceivedComplexity
665
709
  unit = build_unit(unit_data)
666
710
  apply_chunking(unit) if @chunker && unit.chunks.empty? && needs_chunking?(unit)
@@ -16,7 +16,7 @@ module Woods
16
16
  # provider = Woods::Embedding::Provider::OpenAI.new(api_key: ENV['OPENAI_API_KEY'])
17
17
  # vector = provider.embed("class User < ApplicationRecord; end")
18
18
  # vectors = provider.embed_batch(["text1", "text2"])
19
- class OpenAI
19
+ class OpenAI # rubocop:disable Metrics/ClassLength
20
20
  include Interface
21
21
  include DiscardableClient
22
22
 
@@ -26,13 +26,20 @@ module Woods
26
26
  'text-embedding-3-small' => 1536,
27
27
  'text-embedding-3-large' => 3072
28
28
  }.freeze
29
- # OpenAI embedding models share an 8191-token input cap across
29
+ # Conservatively chunk below OpenAI's 8192-token input limit across
30
30
  # text-embedding-3-small / -3-large / ada-002. The chunker uses
31
- # this as a hard ceiling — the actual chunk size lands well
31
+ # 8191 as its ceiling — the actual chunk size lands well
32
32
  # below it once chars-per-token estimation and the prefix
33
33
  # allowance are factored in (see Builder#build_chunker).
34
34
  MAX_INPUT_TOKENS = 8191
35
35
 
36
+ # The API accepts at most 2048 inputs and 300,000 total tokens per
37
+ # request, with 8192 tokens per valid input. At most 36 inputs keeps
38
+ # even maximum-size texts below both limits without token estimates.
39
+ # https://developers.openai.com/api/reference/resources/embeddings/methods/create
40
+ MAX_REQUEST_INPUTS = 36
41
+ private_constant :MAX_REQUEST_INPUTS
42
+
36
43
  # @param api_key [String] OpenAI API key
37
44
  # @param model [String] OpenAI embedding model name (default: text-embedding-3-small)
38
45
  # @param dimensions [Integer, nil] Requested output size for text-embedding-3 models
@@ -57,7 +64,7 @@ module Woods
57
64
  vectors.first
58
65
  end
59
66
 
60
- # Embed multiple texts in a single request.
67
+ # Embed multiple texts in bounded requests, preserving input order.
61
68
  #
62
69
  # Sorts results by the index field to guarantee ordering matches input.
63
70
  #
@@ -71,8 +78,12 @@ module Woods
71
78
  raise ArgumentError, 'embed_batch(texts) rejects nil/empty entries (OpenAI returns 400)'
72
79
  end
73
80
 
74
- response = post_request(request_body(texts))
75
- extract_validated_batch(response, texts.size)
81
+ vectors = texts.each_slice(MAX_REQUEST_INPUTS).flat_map do |slice|
82
+ response = post_request(request_body(slice))
83
+ extract_validated_batch(response, slice.size)
84
+ end
85
+ VectorValidation.validate!(vectors, expected_count: texts.size, provider: 'OpenAI')
86
+ vectors
76
87
  end
77
88
 
78
89
  # Return the dimensionality of vectors produced by this model.
@@ -14,10 +14,14 @@ module Woods
14
14
  # @return [Integer, nil] the pid of the most recently spawned process
15
15
  attr_reader :pid
16
16
 
17
+ # @return [Integer, nil] process group owned by the current command
18
+ attr_reader :process_group_id
19
+
17
20
  # @param command [String] a full shell command line
18
21
  # @param chdir [String]
19
22
  # @return [Array(String, String, Boolean)] stdout, stderr, success
20
23
  def call(command, chdir:)
24
+ @pid = @process_group_id = nil
21
25
  stdout_r, stdout_w = IO.pipe
22
26
  stderr_r, stderr_w = IO.pipe
23
27
  spawn_child(command, chdir, stdout_w, stderr_w)
@@ -59,7 +63,8 @@ module Woods
59
63
  # terminated on timeout.
60
64
  def spawn_child(command, chdir, stdout_w, stderr_w)
61
65
  Thread.handle_interrupt(Timeout::Error => :never) do
62
- @pid = Process.spawn(command, chdir: chdir, in: File::NULL, out: stdout_w, err: stderr_w)
66
+ @pid = Process.spawn(command, chdir: chdir, in: File::NULL, out: stdout_w, err: stderr_w, pgroup: true)
67
+ @process_group_id = @pid
63
68
  end
64
69
  end
65
70
  end
@@ -16,7 +16,10 @@ module Woods
16
16
  # subprocess started by the wrapped executor keeps running unless
17
17
  # something kills it. When the wrapped executor exposes its most recent
18
18
  # pid (see {AblationExecutor}), this sends TERM, waits briefly, then KILL
19
- # if the process is still alive, before reporting the call as a timed-out
19
+ # if still alive. The default executor owns a process group, so cleanup
20
+ # includes ordinary descendants even if their parent already exited.
21
+ # Custom executors exposing only a pid retain single-process cleanup.
22
+ # Cleanup finishes before reporting the call as a timed-out
20
23
  # failure. An executor that does not expose a pid (for example a fake
21
24
  # executor in a spec) still gets the timeout, just without a process to
22
25
  # terminate.
@@ -47,15 +50,30 @@ module Woods
47
50
  pid = executor_pid
48
51
  return '' unless pid
49
52
 
50
- Process.kill('TERM', pid)
51
- Process.kill('KILL', pid) unless process_exited?(pid, within: TERM_GRACE_SECONDS)
53
+ target = signal_target(pid)
54
+ signal('TERM', target)
55
+ signal('KILL', target) unless process_exited?(target, within: TERM_GRACE_SECONDS)
52
56
  reap(pid, within: REAP_GRACE_SECONDS) ? '' : " (pid #{pid} did not reap within #{REAP_GRACE_SECONDS}s)"
53
57
  rescue Errno::ESRCH, Errno::ECHILD
54
58
  ''
55
59
  end
56
60
 
57
61
  def executor_pid
58
- @executor.pid if @executor.respond_to?(:pid)
62
+ pid = @executor.pid if @executor.respond_to?(:pid)
63
+ pid if pid.is_a?(Integer) && pid.positive?
64
+ end
65
+
66
+ # A group created by spawn(pgroup: true) is named after that child.
67
+ # Never infer a group from a custom pid or signal our own group.
68
+ def signal_target(pid)
69
+ group = @executor.process_group_id if @executor.respond_to?(:process_group_id)
70
+ group == pid && group != Process.getpgrp ? -group : pid
71
+ end
72
+
73
+ def signal(name, target)
74
+ Process.kill(name, target)
75
+ rescue Errno::ESRCH
76
+ nil
59
77
  end
60
78
 
61
79
  def process_exited?(pid, within:)