woods 2.0.0.beta1 → 2.0.0.beta3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +400 -1
- data/CONTRIBUTING.md +224 -9
- data/README.md +7 -3
- data/SECURITY.md +9 -6
- data/docs/AGENT_GUIDE.md +83 -4
- data/docs/AGENT_SETUP.md +82 -1
- data/docs/BACKEND_MATRIX.md +20 -0
- data/docs/CLIENT_HOOKS.md +111 -0
- data/docs/CONFIGURATION_REFERENCE.md +233 -13
- data/docs/CONSOLE_MCP_SETUP.md +35 -5
- data/docs/DOCKER_SETUP.md +21 -2
- data/docs/EVALUATION.md +464 -1
- data/docs/EXTRACTOR_REFERENCE.md +36 -5
- data/docs/FAQ.md +11 -12
- data/docs/GETTING_STARTED.md +17 -5
- data/docs/INCREMENTAL_EXTRACTION.md +158 -2
- data/docs/INDEX_LAYOUT.md +382 -0
- data/docs/INTERNALS.md +15 -7
- data/docs/MCP_SERVERS.md +221 -5
- data/docs/MCP_TOOL_COOKBOOK.md +33 -18
- data/docs/NOTION_INTEGRATION.md +13 -0
- data/docs/OBSIDIAN_INTEGRATION.md +57 -9
- data/docs/PUBLISHED_INDEX.md +71 -0
- data/docs/README.md +7 -0
- data/docs/RETRIEVAL_GUIDE.md +253 -11
- data/docs/RUNTIME_TRACING.md +71 -0
- data/docs/SOURCE_FRESHNESS.md +143 -0
- data/docs/TROUBLESHOOTING.md +117 -5
- data/docs/UNBLOCKED_INTEGRATION.md +25 -0
- data/docs/UPGRADING_TO_2.md +44 -22
- data/docs/WATCH_DAEMON.md +259 -59
- data/exe/woods-agent-config +6 -0
- data/exe/woods-extract +5 -0
- data/exe/woods-hook-context +6 -0
- data/lib/generators/woods/templates/woods.rb.tt +1 -3
- data/lib/tasks/woods.rake +47 -397
- data/lib/woods/agent_configuration/applier.rb +133 -0
- data/lib/woods/agent_configuration/cli.rb +101 -0
- data/lib/woods/agent_configuration/cli_options.rb +29 -0
- data/lib/woods/agent_configuration/document.rb +105 -0
- data/lib/woods/agent_configuration/error.rb +7 -0
- data/lib/woods/agent_configuration/launcher.rb +75 -0
- data/lib/woods/agent_configuration/layout.rb +59 -0
- data/lib/woods/agent_configuration/managed_section.rb +62 -0
- data/lib/woods/agent_configuration/plan.rb +98 -0
- data/lib/woods/agent_configuration/plan_diff.rb +38 -0
- data/lib/woods/agent_configuration/planned_files.rb +61 -0
- data/lib/woods/agent_configuration/planner.rb +63 -0
- data/lib/woods/agent_configuration/planner_validation.rb +77 -0
- data/lib/woods/agent_configuration/preflight.rb +100 -0
- data/lib/woods/agent_configuration/recovery.rb +49 -0
- data/lib/woods/ast/node.rb +2 -0
- data/lib/woods/ast/parser.rb +38 -5
- data/lib/woods/atomic_file.rb +133 -3
- data/lib/woods/builder.rb +21 -5
- data/lib/woods/cache/cache_middleware.rb +28 -7
- data/lib/woods/cache/cache_store.rb +4 -5
- data/lib/woods/change_set.rb +5 -4
- data/lib/woods/console/credential_index.rb +20 -2
- data/lib/woods/console/credential_scanner.rb +14 -14
- data/lib/woods/console/credential_scanner_registry.rb +36 -0
- data/lib/woods/console/embedded_executor.rb +1 -1
- data/lib/woods/console/encrypted_credential_snapshot.rb +16 -0
- data/lib/woods/console/rack_middleware.rb +22 -13
- data/lib/woods/console/server.rb +18 -16
- data/lib/woods/dependency_graph.rb +65 -13
- data/lib/woods/embedding/corpus.rb +94 -0
- data/lib/woods/embedding/indexer.rb +90 -46
- data/lib/woods/embedding/openai.rb +17 -6
- data/lib/woods/evaluation/ablation_executor.rb +6 -1
- data/lib/woods/evaluation/ablation_timed_executor.rb +22 -4
- data/lib/woods/export/typed_reader.rb +56 -0
- data/lib/woods/extractor.rb +557 -228
- data/lib/woods/extractors/action_cable_extractor.rb +3 -1
- data/lib/woods/extractors/behavioral_profile.rb +9 -7
- data/lib/woods/extractors/caching_extractor.rb +3 -1
- data/lib/woods/extractors/concern_extractor.rb +64 -6
- data/lib/woods/extractors/configuration_extractor.rb +7 -3
- data/lib/woods/extractors/controller_extractor.rb +13 -4
- data/lib/woods/extractors/database_view_extractor.rb +3 -1
- data/lib/woods/extractors/decorator_extractor.rb +3 -1
- data/lib/woods/extractors/engine_extractor.rb +3 -1
- data/lib/woods/extractors/event_extractor.rb +4 -2
- data/lib/woods/extractors/factory_extractor.rb +3 -1
- data/lib/woods/extractors/graphql_extractor.rb +8 -2
- data/lib/woods/extractors/i18n_extractor.rb +3 -1
- data/lib/woods/extractors/job_extractor.rb +6 -19
- data/lib/woods/extractors/lib_extractor.rb +3 -1
- data/lib/woods/extractors/mailer_extractor.rb +20 -5
- data/lib/woods/extractors/manager_extractor.rb +3 -1
- data/lib/woods/extractors/method_parameters.rb +53 -0
- data/lib/woods/extractors/middleware_argument.rb +65 -0
- data/lib/woods/extractors/middleware_extractor.rb +9 -3
- data/lib/woods/extractors/migration_extractor.rb +3 -1
- data/lib/woods/extractors/model_extractor.rb +39 -33
- data/lib/woods/extractors/package_extractor.rb +24 -4
- data/lib/woods/extractors/phlex_extractor.rb +3 -1
- data/lib/woods/extractors/policy_extractor.rb +3 -1
- data/lib/woods/extractors/poro_extractor.rb +3 -1
- data/lib/woods/extractors/pundit_extractor.rb +3 -1
- data/lib/woods/extractors/rails_source_extractor.rb +4 -2
- data/lib/woods/extractors/rake_task_extractor.rb +4 -2
- data/lib/woods/extractors/route_extractor.rb +3 -1
- data/lib/woods/extractors/route_helper_resolver.rb +10 -33
- data/lib/woods/extractors/scheduled_job_extractor.rb +41 -15
- data/lib/woods/extractors/serializer_extractor.rb +4 -2
- data/lib/woods/extractors/service_extractor.rb +3 -1
- data/lib/woods/extractors/shared_dependency_scanner.rb +2 -2
- data/lib/woods/extractors/shared_utility_methods.rb +27 -15
- data/lib/woods/extractors/source_nesting.rb +1 -1
- data/lib/woods/extractors/state_machine_extractor.rb +3 -1
- data/lib/woods/extractors/test_mapping_extractor.rb +3 -1
- data/lib/woods/extractors/validator_extractor.rb +3 -1
- data/lib/woods/extractors/view_component_extractor.rb +3 -1
- data/lib/woods/extractors/view_template_extractor.rb +3 -1
- data/lib/woods/flow_assembler.rb +87 -8
- data/lib/woods/flow_precomputer.rb +44 -7
- data/lib/woods/gem_mapper.rb +2 -0
- data/lib/woods/git_history.rb +116 -0
- data/lib/woods/graph_analyzer.rb +195 -63
- data/lib/woods/hooks/context_cli.rb +54 -0
- data/lib/woods/hooks/context_event.rb +88 -0
- data/lib/woods/hooks/context_hint.rb +73 -0
- data/lib/woods/hooks/context_impact.rb +77 -0
- data/lib/woods/hooks/context_output.rb +47 -0
- data/lib/woods/hooks/context_state.rb +102 -0
- data/lib/woods/hooks/refresh.rb +79 -0
- data/lib/woods/hooks/rule_projection.rb +78 -0
- data/lib/woods/input_rules.rb +19 -0
- data/lib/woods/mcp/bearer_auth.rb +20 -12
- data/lib/woods/mcp/bootstrapper.rb +62 -0
- data/lib/woods/mcp/index_reader.rb +323 -160
- data/lib/woods/mcp/initialization_guidance.rb +27 -0
- data/lib/woods/mcp/origin_guard.rb +17 -9
- data/lib/woods/mcp/published_lexical_retriever.rb +115 -0
- data/lib/woods/mcp/renderers/markdown_renderer.rb +8 -1
- data/lib/woods/mcp/renderers/plain_renderer.rb +7 -1
- data/lib/woods/mcp/search_results.rb +74 -0
- data/lib/woods/mcp/server.rb +158 -37
- data/lib/woods/mcp/tool_contract.rb +2 -0
- data/lib/woods/mcp/tool_response_renderer.rb +25 -0
- data/lib/woods/mcp/traversal_evidence.rb +113 -0
- data/lib/woods/mcp/traversal_evidence_index.rb +100 -0
- data/lib/woods/mcp/traversal_evidence_page.rb +41 -0
- data/lib/woods/mcp/traversal_evidence_text.rb +52 -0
- data/lib/woods/notion/exporter.rb +56 -17
- data/lib/woods/obsidian/destination_plan.rb +98 -0
- data/lib/woods/obsidian/name_mapper.rb +19 -3
- data/lib/woods/obsidian/note_builder.rb +19 -10
- data/lib/woods/obsidian/vault_exporter.rb +88 -32
- data/lib/woods/operator/pipeline_guard.rb +18 -13
- data/lib/woods/path_dispatcher.rb +7 -1
- data/lib/woods/payload_store.rb +29 -15
- data/lib/woods/railtie.rb +3 -3
- data/lib/woods/railtie_support.rb +12 -12
- data/lib/woods/rake_helpers.rb +392 -0
- data/lib/woods/resilience/graph_invariant_validator/membership_checks.rb +71 -0
- data/lib/woods/resilience/graph_invariant_validator/node_checks.rb +61 -0
- data/lib/woods/resilience/graph_invariant_validator/reverse_relationship_checks.rb +46 -0
- data/lib/woods/resilience/graph_invariant_validator.rb +119 -0
- data/lib/woods/resilience/index_validator/graph_checks.rb +80 -0
- data/lib/woods/resilience/index_validator.rb +112 -23
- data/lib/woods/retrieval/context_assembler.rb +50 -15
- data/lib/woods/retrieval/lexical_assembler.rb +73 -0
- data/lib/woods/retrieval/lexical_index.rb +119 -0
- data/lib/woods/retrieval/ranker.rb +4 -2
- data/lib/woods/retrieval/scope.rb +108 -0
- data/lib/woods/retrieval/scoped_graph_store.rb +32 -0
- data/lib/woods/retrieval/scoped_vector_store.rb +55 -0
- data/lib/woods/retrieval/search_executor.rb +86 -27
- data/lib/woods/retrieval/source_evidence.rb +200 -0
- data/lib/woods/retriever.rb +98 -22
- data/lib/woods/ruby_analyzer/trace_enricher.rb +80 -38
- data/lib/woods/session_tracer/middleware.rb +10 -12
- data/lib/woods/session_tracer/redis_store.rb +22 -6
- data/lib/woods/session_tracer/session_flow_assembler.rb +23 -17
- data/lib/woods/session_tracer/solid_cache_coordination.rb +6 -4
- data/lib/woods/session_tracer/unit_resolver.rb +63 -0
- data/lib/woods/source_inputs/consumer_errors.rb +27 -0
- data/lib/woods/source_inputs/handoff.rb +102 -0
- data/lib/woods/source_inputs/launcher.rb +157 -0
- data/lib/woods/source_inputs/manifest.rb +124 -0
- data/lib/woods/source_inputs/private_key.rb +55 -0
- data/lib/woods/source_inputs/scanner.rb +171 -0
- data/lib/woods/source_inputs/scopes.rb +71 -0
- data/lib/woods/source_inputs/session.rb +214 -0
- data/lib/woods/source_inputs/status.rb +84 -0
- data/lib/woods/source_inputs/verifier.rb +107 -0
- data/lib/woods/storage/metadata_store.rb +25 -25
- data/lib/woods/storage/pgvector.rb +29 -8
- data/lib/woods/storage/qdrant.rb +17 -7
- data/lib/woods/storage/vector_store.rb +18 -6
- data/lib/woods/tasks.rb +3 -2
- data/lib/woods/temporal/json_snapshot_store.rb +29 -8
- data/lib/woods/unblocked/exporter.rb +59 -70
- data/lib/woods/version.rb +1 -1
- data/lib/woods/watch/boot_snapshot.rb +52 -0
- data/lib/woods/watch/daemon.rb +136 -28
- data/lib/woods/watch/listen_watcher.rb +4 -0
- data/lib/woods/watch/polling_watcher.rb +5 -1
- data/lib/woods/watch/status.rb +20 -15
- data/lib/woods/watch/tree_scan.rb +21 -13
- data/lib/woods/watch/watcher.rb +4 -1
- data/lib/woods.rb +135 -11
- data/plugin/.claude-plugin/plugin.json +1 -1
- data/plugin/hooks/adapters/normalize.jq +15 -0
- data/plugin/hooks/adapters/normalize.rb +63 -0
- data/plugin/hooks/hooks.json +20 -0
- data/plugin/hooks/woods-context.sh +50 -0
- data/plugin/hooks/woods-input-rules.sh +159 -0
- data/plugin/hooks/woods-opencode.mjs +65 -0
- data/plugin/hooks/woods-post-edit.sh +2 -225
- data/plugin/hooks/woods-refresh.sh +260 -0
- data/plugin/hooks/woods-session-start.sh +47 -55
- data/plugin/skills/woods-agent-enable/SKILL.md +13 -0
- data/plugin/skills/woods-diagnose/SKILL.md +288 -1
- data/plugin/skills/woods-investigate/SKILL.md +106 -0
- data/plugin/skills/woods-mcp-config/SKILL.md +89 -1
- data/plugin/skills/woods-setup/SKILL.md +107 -6
- metadata +84 -5
|
@@ -32,9 +32,14 @@ module Woods
|
|
|
32
32
|
# enforce_dependencies: package units (Task 7)
|
|
33
33
|
# commit_count, change_frequency: git enrichment (Task 5)
|
|
34
34
|
NODE_ATTRIBUTE_KEYS = %i[
|
|
35
|
-
database table foreign_key_tables package enforce_dependencies commit_count change_frequency
|
|
35
|
+
database table foreign_key_tables package enforce_dependencies commit_count change_frequency kind
|
|
36
36
|
].freeze
|
|
37
37
|
|
|
38
|
+
# These extractors describe a whole source file rather than one constant.
|
|
39
|
+
# Other types remain unclassified: a path-like identifier is not evidence
|
|
40
|
+
# that a unit is a profile, and an unmarked unit need not name a constant.
|
|
41
|
+
FILE_PROFILE_TYPES = %i[caching configuration test_mapping rails_source gem_source].freeze
|
|
42
|
+
|
|
38
43
|
# Optional edge keys beyond target and via. `through` is the through
|
|
39
44
|
# association name; `through_db` is the through model's resolved
|
|
40
45
|
# database, emitted only when present; `disable_joins` is emitted only
|
|
@@ -80,14 +85,14 @@ module Woods
|
|
|
80
85
|
}.merge(node_attributes_from(unit))
|
|
81
86
|
|
|
82
87
|
(@edges[unit.identifier] ||= {})[unit.type] =
|
|
83
|
-
unit.dependencies
|
|
88
|
+
self.class.normalize_edges(unit.dependencies, strict: true)
|
|
84
89
|
(@file_map[unit.file_path] ||= Set.new).add(unit.identifier) if unit.file_path
|
|
85
90
|
|
|
86
91
|
# Type index for filtering (Set-based for O(1) insert)
|
|
87
92
|
(@type_index[unit.type] ||= Set.new).add(unit.identifier)
|
|
88
93
|
|
|
89
94
|
# Build reverse edges (Set-based for O(1) insert)
|
|
90
|
-
unit.
|
|
95
|
+
@edges[unit.identifier][unit.type].each do |dep|
|
|
91
96
|
(@reverse[dep[:target]] ||= Set.new).add(unit.identifier)
|
|
92
97
|
(@reverse_via[[dep[:target], dep[:via]]] ||= Set.new).add(unit.identifier)
|
|
93
98
|
end
|
|
@@ -645,7 +650,7 @@ module Woods
|
|
|
645
650
|
# @return [Hash{Symbol => Object}] normalized, nil-free
|
|
646
651
|
def node_attributes_from(unit)
|
|
647
652
|
metadata = unit.respond_to?(:metadata) && unit.metadata.is_a?(Hash) ? unit.metadata : {}
|
|
648
|
-
attrs =
|
|
653
|
+
attrs = self.class.file_profile_attributes(unit.type)
|
|
649
654
|
attrs[:database] = metadata[:database] unless metadata[:database].nil?
|
|
650
655
|
attrs[:table] = metadata[:table_name] if unit.type == :model && !metadata[:table_name].nil?
|
|
651
656
|
tables = Array(metadata[:foreign_keys]).filter_map { |fk| fk[:to_table] || fk['to_table'] if fk.is_a?(Hash) }
|
|
@@ -729,14 +734,15 @@ module Woods
|
|
|
729
734
|
# `nodes` and `edges` stay keyed on the bare identifier, holding the
|
|
730
735
|
# **primary** node — the type that sorts first. Identifiers registered
|
|
731
736
|
# under more than one type put their remaining nodes in `variants`, a flat
|
|
732
|
-
# array that is **omitted entirely when empty**.
|
|
733
|
-
#
|
|
734
|
-
# a `dependency_graph.json` written before this change loads through
|
|
737
|
+
# array that is **omitted entirely when empty**. A
|
|
738
|
+
# `dependency_graph.json` written before this change loads through
|
|
735
739
|
# {.from_h} unmodified: it simply carries no `variants`.
|
|
736
740
|
#
|
|
737
741
|
# `reverse`, `file_map` and `type_index` need no new shape. They are
|
|
738
742
|
# already identifier-valued sets, so a Scenic view `reports` and a factory
|
|
739
743
|
# `reports` each contribute to their own type bucket and their own path.
|
|
744
|
+
# The additive `reverse_via` index preserves typed source identities and
|
|
745
|
+
# complete relationship records without changing the legacy `reverse` map.
|
|
740
746
|
#
|
|
741
747
|
# @return [Hash] Complete graph data
|
|
742
748
|
def to_h
|
|
@@ -747,6 +753,7 @@ module Woods
|
|
|
747
753
|
nodes: key_sorted(self.class.relativize_nodes(primary_nodes, root)),
|
|
748
754
|
edges: key_sorted(primary_edges),
|
|
749
755
|
reverse: key_sorted(@reverse.transform_values { |ids| ids.to_a.sort }),
|
|
756
|
+
reverse_via: published_reverse_via,
|
|
750
757
|
file_map: key_sorted(self.class.relativize_file_map(@file_map, root)),
|
|
751
758
|
type_index: key_sorted(@type_index.transform_values { |ids| ids.to_a.sort }),
|
|
752
759
|
stats: {
|
|
@@ -770,6 +777,23 @@ module Woods
|
|
|
770
777
|
end
|
|
771
778
|
private :key_sorted
|
|
772
779
|
|
|
780
|
+
# Reverse buckets are derived from every typed forward registration, rather
|
|
781
|
+
# than the bare-name in-memory filter index, which cannot preserve variants
|
|
782
|
+
# or distinguish association attributes. Legacy nil relationships stay nil.
|
|
783
|
+
def published_reverse_via
|
|
784
|
+
buckets = {}
|
|
785
|
+
@edges.each do |source, by_type|
|
|
786
|
+
by_type.each do |type, edges|
|
|
787
|
+
edges.each do |edge|
|
|
788
|
+
record = { source: source, source_type: type }.merge(edge.except(:target))
|
|
789
|
+
(buckets[edge[:target]] ||= []) << record
|
|
790
|
+
end
|
|
791
|
+
end
|
|
792
|
+
end
|
|
793
|
+
key_sorted(buckets.transform_values { |records| records.sort_by { |record| JSON.generate(record) } })
|
|
794
|
+
end
|
|
795
|
+
private :published_reverse_via
|
|
796
|
+
|
|
773
797
|
# A copy whose edge containers are the caller's alone.
|
|
774
798
|
#
|
|
775
799
|
# `dup` copies only the top level, so `to_h[:edges][id]` used to be the
|
|
@@ -783,6 +807,9 @@ module Woods
|
|
|
783
807
|
def detached_snapshot(memo)
|
|
784
808
|
snapshot = memo.dup
|
|
785
809
|
snapshot[:edges] = memo[:edges].transform_values { |list| list.map(&:dup) }
|
|
810
|
+
snapshot[:reverse_via] = memo[:reverse_via].transform_values do |records|
|
|
811
|
+
records.map { |record| record.transform_values { |value| value.is_a?(String) ? value.dup : value } }
|
|
812
|
+
end
|
|
786
813
|
if memo.key?(:variants)
|
|
787
814
|
snapshot[:variants] = memo[:variants].map do |record|
|
|
788
815
|
record.merge(edges: record[:edges].map(&:dup))
|
|
@@ -794,7 +821,12 @@ module Woods
|
|
|
794
821
|
|
|
795
822
|
# @return [Hash{String => Hash}] identifier => primary node
|
|
796
823
|
def primary_nodes
|
|
797
|
-
|
|
824
|
+
# Late enrichment and JSON loading can insert optional fields in a
|
|
825
|
+
# different order. Canonicalize like variant_records so a graph's JSON
|
|
826
|
+
# fingerprint is stable across full and incremental publication.
|
|
827
|
+
@nodes.transform_values do |nodes|
|
|
828
|
+
primary_of(nodes).slice(:type, :file_path, :namespace, *NODE_ATTRIBUTE_KEYS)
|
|
829
|
+
end
|
|
798
830
|
end
|
|
799
831
|
private :primary_nodes
|
|
800
832
|
|
|
@@ -969,10 +1001,17 @@ module Woods
|
|
|
969
1001
|
# @param node [Hash]
|
|
970
1002
|
# @return [Hash{Symbol => Object}]
|
|
971
1003
|
def self.persisted_node_attributes(node)
|
|
972
|
-
NODE_ATTRIBUTE_KEYS.each_with_object({}) do |key, attrs|
|
|
1004
|
+
attributes = NODE_ATTRIBUTE_KEYS.each_with_object({}) do |key, attrs|
|
|
973
1005
|
value = node.key?(key) ? node[key] : node[key.to_s]
|
|
974
1006
|
attrs[key] = normalize_node_attribute(key, value) unless value.nil?
|
|
975
1007
|
end
|
|
1008
|
+
attributes.merge(file_profile_attributes(node[:type] || node['type']))
|
|
1009
|
+
end
|
|
1010
|
+
|
|
1011
|
+
# Backfill the additive marker when an older graph is republished, so
|
|
1012
|
+
# unchanged nodes in an incremental run agree with a fresh full graph.
|
|
1013
|
+
def self.file_profile_attributes(type)
|
|
1014
|
+
FILE_PROFILE_TYPES.include?(type&.to_sym) ? { kind: 'file_profile' } : {}
|
|
976
1015
|
end
|
|
977
1016
|
|
|
978
1017
|
# Edge attributes present in a dependency or persisted edge hash.
|
|
@@ -1032,7 +1071,10 @@ module Woods
|
|
|
1032
1071
|
graph.instance_variable_set(:@edges, edges)
|
|
1033
1072
|
|
|
1034
1073
|
raw_reverse = data[:reverse] || data['reverse'] || {}
|
|
1035
|
-
|
|
1074
|
+
reverse = raw_reverse.each_with_object({}) do |(target, sources), index|
|
|
1075
|
+
(index[target.to_s] ||= Set.new).merge(sources)
|
|
1076
|
+
end
|
|
1077
|
+
graph.instance_variable_set(:@reverse, reverse)
|
|
1036
1078
|
|
|
1037
1079
|
raw_file_map = data[:file_map] || data['file_map'] || {}
|
|
1038
1080
|
graph.instance_variable_set(:@file_map, absolutize_file_map(normalize_file_map(raw_file_map), root))
|
|
@@ -1126,7 +1168,12 @@ module Woods
|
|
|
1126
1168
|
}.merge(persisted_node_attributes(node))
|
|
1127
1169
|
end
|
|
1128
1170
|
|
|
1129
|
-
# Normalize
|
|
1171
|
+
# Normalize fresh dependencies and persisted edges to the same identity:
|
|
1172
|
+
# string targets and symbol relationship labels. Extractors may emit
|
|
1173
|
+
# symbolic external targets (such as :http_api); retaining those symbols
|
|
1174
|
+
# after a JSON-loaded baseline splits reverse indexes on re-registration.
|
|
1175
|
+
# Unit dependency metadata stays unchanged; only graph records normalize.
|
|
1176
|
+
# Also accepts the old bare-string edge format.
|
|
1130
1177
|
#
|
|
1131
1178
|
# ROUND-TRIP INVARIANT (do not break when refactoring):
|
|
1132
1179
|
# DependencyGraph#to_h -> JSON.generate -> JSON.parse -> DependencyGraph.from_h
|
|
@@ -1147,15 +1194,20 @@ module Woods
|
|
|
1147
1194
|
# explicit data conversion.
|
|
1148
1195
|
#
|
|
1149
1196
|
# @param edges [Array] Edge entries — either strings or hashes
|
|
1197
|
+
# @param strict [Boolean] require hashes for fresh units; legacy loads stay tolerant
|
|
1150
1198
|
# @return [Array<Hash>] Normalized edges with :target and :via keys
|
|
1151
|
-
def self.normalize_edges(edges)
|
|
1199
|
+
def self.normalize_edges(edges, strict: false)
|
|
1200
|
+
raise ArgumentError, 'Fresh graph dependencies must be an array' if strict && !edges.is_a?(Array)
|
|
1201
|
+
|
|
1152
1202
|
return [] unless edges.is_a?(Array)
|
|
1153
1203
|
|
|
1154
1204
|
edges.map do |edge|
|
|
1205
|
+
raise ArgumentError, 'Fresh graph dependencies must be hashes' if strict && !edge.is_a?(Hash)
|
|
1206
|
+
|
|
1155
1207
|
if edge.is_a?(String)
|
|
1156
1208
|
{ target: edge, via: nil }
|
|
1157
1209
|
elsif edge.is_a?(Hash)
|
|
1158
|
-
{ target: edge[:target] || edge['target'], via: (edge[:via] || edge['via'])&.to_sym }
|
|
1210
|
+
{ target: (edge[:target] || edge['target'])&.to_s, via: (edge[:via] || edge['via'])&.to_sym }
|
|
1159
1211
|
.merge(edge_attributes(edge))
|
|
1160
1212
|
else
|
|
1161
1213
|
{ target: edge.to_s, via: nil }
|
|
@@ -0,0 +1,94 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require 'json'
|
|
4
|
+
require_relative '../atomic_file'
|
|
5
|
+
require_relative '../generation'
|
|
6
|
+
|
|
7
|
+
module Woods
|
|
8
|
+
class Error < StandardError; end unless defined?(Woods::Error)
|
|
9
|
+
|
|
10
|
+
module Embedding
|
|
11
|
+
# Collect the complete extraction input before an embedding run can mutate
|
|
12
|
+
# stores. Native publications use their authoritative listings, never a glob
|
|
13
|
+
# that could mistake a missing/corrupt payload for intentional deletion.
|
|
14
|
+
class Corpus
|
|
15
|
+
class Incomplete < Woods::Error; end
|
|
16
|
+
|
|
17
|
+
def initialize(output_dir)
|
|
18
|
+
@output_dir = output_dir.to_s
|
|
19
|
+
end
|
|
20
|
+
|
|
21
|
+
def load
|
|
22
|
+
native? ? published_units : legacy_units
|
|
23
|
+
rescue IOError, SystemCallError, JSON::ParserError, EncodingError, ArgumentError => e
|
|
24
|
+
raise Incomplete, "Embedding input incomplete: #{e.message}. Repair or rebuild extraction before embedding."
|
|
25
|
+
end
|
|
26
|
+
|
|
27
|
+
private
|
|
28
|
+
|
|
29
|
+
def native?
|
|
30
|
+
path = File.join(@output_dir, Generation::FILENAME)
|
|
31
|
+
if File.exist?(path)
|
|
32
|
+
marker = JSON.parse(AtomicFile.read(path))
|
|
33
|
+
raise IOError, 'invalid generation marker' unless marker.is_a?(Hash)
|
|
34
|
+
end
|
|
35
|
+
|
|
36
|
+
if !marker || marker['payload'].nil?
|
|
37
|
+
if File.directory?(File.join(@output_dir, 'payloads'))
|
|
38
|
+
raise IOError, 'missing publication pointer beside payloads'
|
|
39
|
+
end
|
|
40
|
+
|
|
41
|
+
return false
|
|
42
|
+
end
|
|
43
|
+
|
|
44
|
+
validate_marker(marker)
|
|
45
|
+
true
|
|
46
|
+
end
|
|
47
|
+
|
|
48
|
+
def validate_marker(marker)
|
|
49
|
+
strings_valid = %w[payload token].all? { |key| marker[key].is_a?(String) && !marker[key].empty? }
|
|
50
|
+
return if strings_valid && marker['number'].is_a?(Integer) && marker['number'].positive?
|
|
51
|
+
|
|
52
|
+
raise IOError, 'invalid published generation marker'
|
|
53
|
+
end
|
|
54
|
+
|
|
55
|
+
def published_units
|
|
56
|
+
require_relative '../mcp/index_reader'
|
|
57
|
+
|
|
58
|
+
reader = MCP::IndexReader.new(@output_dir)
|
|
59
|
+
reader.with_pinned_generation do
|
|
60
|
+
# Generation's compatibility fallback to the root is useful for
|
|
61
|
+
# readers, but cannot prove completeness for destructive reconciliation.
|
|
62
|
+
if reader.payload_dir.expand_path == Pathname.new(@output_dir).expand_path
|
|
63
|
+
raise IOError,
|
|
64
|
+
'published payload missing or invalid'
|
|
65
|
+
end
|
|
66
|
+
|
|
67
|
+
validate_counts(reader.manifest)
|
|
68
|
+
reader.each_unit.to_a
|
|
69
|
+
end
|
|
70
|
+
end
|
|
71
|
+
|
|
72
|
+
def validate_counts(manifest)
|
|
73
|
+
counts = manifest.is_a?(Hash) && manifest['counts']
|
|
74
|
+
valid = counts.is_a?(Hash) && counts.all? do |dir, count|
|
|
75
|
+
MCP::IndexReader::TYPE_DIRS.include?(dir) && count.is_a?(Integer) && count >= 0
|
|
76
|
+
end
|
|
77
|
+
raise IOError, 'invalid published manifest counts' unless valid
|
|
78
|
+
end
|
|
79
|
+
|
|
80
|
+
# Pre-pointer indexes may use arbitrary filenames and have no manifest
|
|
81
|
+
# or authoritative listing. Keep that input contract, but never silently
|
|
82
|
+
# omit unreadable JSON before reconciling persisted identities.
|
|
83
|
+
def legacy_units
|
|
84
|
+
Dir.glob(File.join(@output_dir, '**', '*.json')).filter_map do |path|
|
|
85
|
+
relative = path.delete_prefix("#{@output_dir}/")
|
|
86
|
+
next if relative.start_with?('dumps/', 'payloads/') || File.basename(path) == 'checkpoint.json'
|
|
87
|
+
|
|
88
|
+
data = JSON.parse(AtomicFile.read(path))
|
|
89
|
+
data if data.is_a?(Hash) && data.key?('type') && data.key?('identifier')
|
|
90
|
+
end
|
|
91
|
+
end
|
|
92
|
+
end
|
|
93
|
+
end
|
|
94
|
+
end
|
|
@@ -119,40 +119,14 @@ module Woods
|
|
|
119
119
|
|
|
120
120
|
private
|
|
121
121
|
|
|
122
|
-
# Where this run's units are read from — the published generation's
|
|
123
|
-
# payload, or the output root for a flat index.
|
|
124
|
-
#
|
|
125
|
-
# Globbing the output root would walk `payloads/` and ingest every
|
|
126
|
-
# retained generation, so each unit would be embedded once per retained
|
|
127
|
-
# payload. It also picks up `dumps/`, which holds no unit JSON but is
|
|
128
|
-
# needlessly large to walk.
|
|
129
|
-
#
|
|
130
|
-
# @return [String]
|
|
131
|
-
def units_dir
|
|
132
|
-
Woods::Generation.new(output_dir: @output_dir).payload_dir.to_s
|
|
133
|
-
end
|
|
134
|
-
|
|
135
122
|
def load_units
|
|
136
|
-
|
|
137
|
-
|
|
138
|
-
|
|
139
|
-
# AtomicFile.read, not File.read: a bare read tags the bytes with the
|
|
140
|
-
# process's default external encoding (US-ASCII under LANG=C), so a
|
|
141
|
-
# single multibyte character in one unit raised EncodingError out of
|
|
142
|
-
# JSON.parse and aborted the whole embed run.
|
|
143
|
-
data = JSON.parse(AtomicFile.read(path))
|
|
144
|
-
# Extraction output also contains index listings (_index.json arrays) and
|
|
145
|
-
# summary files (manifest.json, dependency_graph.json, graph_analysis.json)
|
|
146
|
-
# that live alongside per-unit JSON. Filter to the unit shape.
|
|
147
|
-
data if data.is_a?(Hash) && data.key?('type') && data.key?('identifier')
|
|
148
|
-
rescue JSON::ParserError, EncodingError => e
|
|
149
|
-
warn "[woods] skipping unreadable unit file #{path} (#{e.class}: #{e.message})"
|
|
150
|
-
nil
|
|
151
|
-
end
|
|
123
|
+
require_relative 'corpus'
|
|
124
|
+
|
|
125
|
+
Corpus.new(@output_dir).load
|
|
152
126
|
end
|
|
153
127
|
|
|
154
|
-
# The invariant: **checkpoint.json
|
|
155
|
-
# vector
|
|
128
|
+
# The invariant: **checkpoint.json advances only after the intended
|
|
129
|
+
# vector state (including an empty set) is durable.** Two things uphold it here.
|
|
156
130
|
#
|
|
157
131
|
# 1. Ordering. For a store whose only durable copy is the dump
|
|
158
132
|
# (+persistable?+), the checkpoint is written *after*
|
|
@@ -170,17 +144,19 @@ module Woods
|
|
|
170
144
|
#
|
|
171
145
|
# 2. Trust, verified. {#checkpoint_satisfied?} honours a checkpoint hit
|
|
172
146
|
# only when the durable artifact actually holds a vector for that
|
|
173
|
-
# unit
|
|
147
|
+
# unit, or preparation verifies that it intentionally has no text.
|
|
148
|
+
# A checkpoint that ran ahead of its dump — an older gem with
|
|
174
149
|
# this bug, an interrupted promote, a store swap — self-heals into a
|
|
175
150
|
# re-embed instead of stranding the unit forever.
|
|
176
151
|
def process_units(units, incremental:)
|
|
177
152
|
prepare_run(incremental: incremental)
|
|
178
|
-
units = assign_storage_identities(units)
|
|
179
153
|
checkpoint = incremental ? load_checkpoint : {}
|
|
154
|
+
units = assign_storage_identities(units, checkpoint: checkpoint)
|
|
180
155
|
stats = { processed: 0, skipped: 0, errors: 0 }
|
|
181
156
|
|
|
182
157
|
embed_batches(units, checkpoint, stats, incremental: incremental)
|
|
183
158
|
|
|
159
|
+
reconcile_empty_units(checkpoint)
|
|
184
160
|
retire_legacy_identities
|
|
185
161
|
report_checkpoint_misses
|
|
186
162
|
vanished = incremental && persistable? ? drop_vanished_units : 0
|
|
@@ -196,12 +172,12 @@ module Woods
|
|
|
196
172
|
end
|
|
197
173
|
|
|
198
174
|
# Unambiguous existing keys stay stable. A collision uses reversible typed keys.
|
|
199
|
-
def assign_storage_identities(units)
|
|
175
|
+
def assign_storage_identities(units, checkpoint:)
|
|
200
176
|
counts = units.group_by { |unit| unit['identifier'] }.transform_values(&:size)
|
|
201
177
|
units.map do |unit|
|
|
202
178
|
id = unit['identifier']
|
|
203
179
|
typed = StorageIdentity.key(id, unit['type'])
|
|
204
|
-
existing = known_storage_key?(typed)
|
|
180
|
+
existing = known_storage_key?(typed, checkpoint: checkpoint)
|
|
205
181
|
next unit unless counts[id] > 1 || existing || id.start_with?(StorageIdentity::PREFIX)
|
|
206
182
|
|
|
207
183
|
unit.merge('storage_id' => typed)
|
|
@@ -224,8 +200,8 @@ module Woods
|
|
|
224
200
|
end
|
|
225
201
|
end
|
|
226
202
|
|
|
227
|
-
def known_storage_key?(key)
|
|
228
|
-
(@persisted_ids || {}).key?(key) || (@durable_ids || {}).key?(key)
|
|
203
|
+
def known_storage_key?(key, checkpoint:)
|
|
204
|
+
(@persisted_ids || {}).key?(key) || (@durable_ids || {}).key?(key) || checkpoint.key?(key)
|
|
229
205
|
end
|
|
230
206
|
|
|
231
207
|
def retire_legacy_key(legacy)
|
|
@@ -243,11 +219,9 @@ module Woods
|
|
|
243
219
|
# `woods:embed_incremental` runs evict every genuinely older dump in
|
|
244
220
|
# favour of copies of the same state.
|
|
245
221
|
#
|
|
246
|
-
#
|
|
247
|
-
#
|
|
248
|
-
#
|
|
249
|
-
# self-heal counts as processed, so a run that re-embeds a stranded unit
|
|
250
|
-
# still dumps.
|
|
222
|
+
# A zero-text transition can retire vectors without a provider call;
|
|
223
|
+
# @vectors_changed captures that case. Checkpoint self-heals still count
|
|
224
|
+
# as processed, so a run re-embedding a stranded unit also dumps.
|
|
251
225
|
#
|
|
252
226
|
# But "nothing embedded" is not "nothing changed" (B-069). `persist_snapshot`
|
|
253
227
|
# writes the vector dump, the *metadata* dump and the config into one
|
|
@@ -266,6 +240,11 @@ module Woods
|
|
|
266
240
|
# `@current_identifiers` is what this run saw. Anything in the first and
|
|
267
241
|
# not the second is stale, and the dump has to be rewritten to drop it.
|
|
268
242
|
#
|
|
243
|
+
# Existing units can also change metadata without changing source (B-119).
|
|
244
|
+
# Compare their complete records with the promoted metadata snapshot;
|
|
245
|
+
# publishing those changes must not require a provider call or invalidate
|
|
246
|
+
# the source-hash checkpoint contract.
|
|
247
|
+
#
|
|
269
248
|
# Full runs always dump: they rebuild the store from scratch, so "nothing
|
|
270
249
|
# processed" there means the store is genuinely empty and the dump must
|
|
271
250
|
# say so rather than leave a stale one promoted.
|
|
@@ -274,7 +253,7 @@ module Woods
|
|
|
274
253
|
def snapshot_worth_writing?(stats, vanished, incremental:)
|
|
275
254
|
return true unless incremental
|
|
276
255
|
|
|
277
|
-
stats[:processed].positive? || vanished.positive?
|
|
256
|
+
stats[:processed].positive? || vanished.positive? || @metadata_changed || @vectors_changed
|
|
278
257
|
end
|
|
279
258
|
|
|
280
259
|
# Fraction of the persisted units the vanished-unit sweep may remove
|
|
@@ -489,10 +468,36 @@ module Woods
|
|
|
489
468
|
@current_identifiers = Set.new
|
|
490
469
|
@durable_ids = nil
|
|
491
470
|
@checkpoint_misses = 0
|
|
492
|
-
|
|
471
|
+
@metadata_changed = false
|
|
472
|
+
@vectors_changed = false
|
|
473
|
+
@empty_units = {}
|
|
474
|
+
@persisted_metadata = nil
|
|
475
|
+
prepare_snapshot_stores(incremental: incremental)
|
|
476
|
+
retain_metadata_identities
|
|
493
477
|
load_durable_store_ids if reconcilable?
|
|
494
478
|
end
|
|
495
479
|
|
|
480
|
+
def prepare_snapshot_stores(incremental:)
|
|
481
|
+
return unless persistable?
|
|
482
|
+
|
|
483
|
+
if incremental
|
|
484
|
+
hydrate_persisted_metadata
|
|
485
|
+
hydrate_persisted_vectors
|
|
486
|
+
else
|
|
487
|
+
# A direct caller may reuse an in-memory adapter for a full run.
|
|
488
|
+
# Its old chunks still need replacement, including by an empty set.
|
|
489
|
+
entries = []
|
|
490
|
+
@vector_store.each_entry { |id, _vector, _metadata| entries << { id: id } }
|
|
491
|
+
@persisted_ids = index_ids_by_identifier(entries)
|
|
492
|
+
end
|
|
493
|
+
end
|
|
494
|
+
|
|
495
|
+
# Source-empty units retain metadata but intentionally have no vectors.
|
|
496
|
+
# Keep those identities in the same deletion and typed-key accounting.
|
|
497
|
+
def retain_metadata_identities
|
|
498
|
+
@persisted_metadata&.each_entry { |identifier, _unit| @persisted_ids[identifier] ||= [] }
|
|
499
|
+
end
|
|
500
|
+
|
|
496
501
|
# Read back what the durable store currently holds, as base identifiers.
|
|
497
502
|
#
|
|
498
503
|
# One enumeration serves both halves of #211:
|
|
@@ -618,7 +623,10 @@ module Woods
|
|
|
618
623
|
# enumeration that failed) — fall back to trusting the checkpoint.
|
|
619
624
|
return true if known_ids.nil?
|
|
620
625
|
|
|
621
|
-
return true if known_ids
|
|
626
|
+
return true if known_ids[storage_id(unit_data)]&.any?
|
|
627
|
+
# A source-empty unit intentionally has no vector. Verify that state
|
|
628
|
+
# again rather than treating a missing nonempty vector as a cache hit.
|
|
629
|
+
return true if prepare_texts(unit_data).empty?
|
|
622
630
|
|
|
623
631
|
@checkpoint_misses += 1
|
|
624
632
|
false
|
|
@@ -635,7 +643,22 @@ module Woods
|
|
|
635
643
|
def persist_unit_metadata(unit_data)
|
|
636
644
|
return unless @metadata_store
|
|
637
645
|
|
|
638
|
-
|
|
646
|
+
id = storage_id(unit_data)
|
|
647
|
+
@metadata_changed ||= @persisted_metadata && @persisted_metadata.find(id) != unit_data
|
|
648
|
+
@metadata_store.store(id, unit_data)
|
|
649
|
+
end
|
|
650
|
+
|
|
651
|
+
# Compare with the promoted artifact, not a fresh metadata store or the
|
|
652
|
+
# source-only checkpoint. Retain old records until the guarded vanished
|
|
653
|
+
# sweep permits their deletion, just as vector hydration does.
|
|
654
|
+
def hydrate_persisted_metadata
|
|
655
|
+
return unless @metadata_store.respond_to?(:each_entry) && @metadata_store.respond_to?(:bulk_load)
|
|
656
|
+
|
|
657
|
+
require_relative '../index_artifact'
|
|
658
|
+
require_relative '../storage/snapshotter'
|
|
659
|
+
|
|
660
|
+
@persisted_metadata = Storage::Snapshotter::Metadata.load_or_empty(IndexArtifact.new(@output_dir))
|
|
661
|
+
@metadata_store.bulk_load(@persisted_metadata.each_entry)
|
|
639
662
|
end
|
|
640
663
|
|
|
641
664
|
# Refuse to index a unit whose real identifier already matches the
|
|
@@ -653,6 +676,7 @@ module Woods
|
|
|
653
676
|
def collect_embed_items(unit_data, items)
|
|
654
677
|
texts = prepare_texts(unit_data)
|
|
655
678
|
identifier = storage_id(unit_data)
|
|
679
|
+
@empty_units[identifier] = unit_data['source_hash'] if texts.empty?
|
|
656
680
|
|
|
657
681
|
texts.each_with_index do |text, idx|
|
|
658
682
|
embed_id = texts.length > 1 ? "#{identifier}#chunk_#{idx}" : identifier
|
|
@@ -661,6 +685,26 @@ module Woods
|
|
|
661
685
|
end
|
|
662
686
|
end
|
|
663
687
|
|
|
688
|
+
# Defer zero-text deletion until every batch has prepared/embedded
|
|
689
|
+
# successfully. A later provider failure must not retire earlier vectors.
|
|
690
|
+
# Checkpoints retain their source-hash format; a matching no-vector hash
|
|
691
|
+
# is trusted only after verifying the current input still prepares empty.
|
|
692
|
+
def reconcile_empty_units(checkpoint)
|
|
693
|
+
verify_empty_reconciliation!
|
|
694
|
+
@empty_units.each do |identifier, source_hash|
|
|
695
|
+
@vectors_changed ||= @persisted_ids[identifier]&.any?
|
|
696
|
+
prune_identifier(identifier, []) if implements_own?(@vector_store, :delete)
|
|
697
|
+
prune_durable_identifier(identifier, []) if @durable_ids && implements_own?(@vector_store, :delete)
|
|
698
|
+
checkpoint[identifier] = source_hash
|
|
699
|
+
end
|
|
700
|
+
end
|
|
701
|
+
|
|
702
|
+
def verify_empty_reconciliation!
|
|
703
|
+
return unless @empty_units.any? && reconcilable? && @durable_ids.nil?
|
|
704
|
+
|
|
705
|
+
raise Woods::Error, 'Cannot reconcile source-empty units: existing vector IDs could not be read'
|
|
706
|
+
end
|
|
707
|
+
|
|
664
708
|
def prepare_texts(unit_data) # rubocop:disable Metrics/CyclomaticComplexity, Metrics/PerceivedComplexity
|
|
665
709
|
unit = build_unit(unit_data)
|
|
666
710
|
apply_chunking(unit) if @chunker && unit.chunks.empty? && needs_chunking?(unit)
|
|
@@ -16,7 +16,7 @@ module Woods
|
|
|
16
16
|
# provider = Woods::Embedding::Provider::OpenAI.new(api_key: ENV['OPENAI_API_KEY'])
|
|
17
17
|
# vector = provider.embed("class User < ApplicationRecord; end")
|
|
18
18
|
# vectors = provider.embed_batch(["text1", "text2"])
|
|
19
|
-
class OpenAI
|
|
19
|
+
class OpenAI # rubocop:disable Metrics/ClassLength
|
|
20
20
|
include Interface
|
|
21
21
|
include DiscardableClient
|
|
22
22
|
|
|
@@ -26,13 +26,20 @@ module Woods
|
|
|
26
26
|
'text-embedding-3-small' => 1536,
|
|
27
27
|
'text-embedding-3-large' => 3072
|
|
28
28
|
}.freeze
|
|
29
|
-
#
|
|
29
|
+
# Conservatively chunk below OpenAI's 8192-token input limit across
|
|
30
30
|
# text-embedding-3-small / -3-large / ada-002. The chunker uses
|
|
31
|
-
#
|
|
31
|
+
# 8191 as its ceiling — the actual chunk size lands well
|
|
32
32
|
# below it once chars-per-token estimation and the prefix
|
|
33
33
|
# allowance are factored in (see Builder#build_chunker).
|
|
34
34
|
MAX_INPUT_TOKENS = 8191
|
|
35
35
|
|
|
36
|
+
# The API accepts at most 2048 inputs and 300,000 total tokens per
|
|
37
|
+
# request, with 8192 tokens per valid input. At most 36 inputs keeps
|
|
38
|
+
# even maximum-size texts below both limits without token estimates.
|
|
39
|
+
# https://developers.openai.com/api/reference/resources/embeddings/methods/create
|
|
40
|
+
MAX_REQUEST_INPUTS = 36
|
|
41
|
+
private_constant :MAX_REQUEST_INPUTS
|
|
42
|
+
|
|
36
43
|
# @param api_key [String] OpenAI API key
|
|
37
44
|
# @param model [String] OpenAI embedding model name (default: text-embedding-3-small)
|
|
38
45
|
# @param dimensions [Integer, nil] Requested output size for text-embedding-3 models
|
|
@@ -57,7 +64,7 @@ module Woods
|
|
|
57
64
|
vectors.first
|
|
58
65
|
end
|
|
59
66
|
|
|
60
|
-
# Embed multiple texts in
|
|
67
|
+
# Embed multiple texts in bounded requests, preserving input order.
|
|
61
68
|
#
|
|
62
69
|
# Sorts results by the index field to guarantee ordering matches input.
|
|
63
70
|
#
|
|
@@ -71,8 +78,12 @@ module Woods
|
|
|
71
78
|
raise ArgumentError, 'embed_batch(texts) rejects nil/empty entries (OpenAI returns 400)'
|
|
72
79
|
end
|
|
73
80
|
|
|
74
|
-
|
|
75
|
-
|
|
81
|
+
vectors = texts.each_slice(MAX_REQUEST_INPUTS).flat_map do |slice|
|
|
82
|
+
response = post_request(request_body(slice))
|
|
83
|
+
extract_validated_batch(response, slice.size)
|
|
84
|
+
end
|
|
85
|
+
VectorValidation.validate!(vectors, expected_count: texts.size, provider: 'OpenAI')
|
|
86
|
+
vectors
|
|
76
87
|
end
|
|
77
88
|
|
|
78
89
|
# Return the dimensionality of vectors produced by this model.
|
|
@@ -14,10 +14,14 @@ module Woods
|
|
|
14
14
|
# @return [Integer, nil] the pid of the most recently spawned process
|
|
15
15
|
attr_reader :pid
|
|
16
16
|
|
|
17
|
+
# @return [Integer, nil] process group owned by the current command
|
|
18
|
+
attr_reader :process_group_id
|
|
19
|
+
|
|
17
20
|
# @param command [String] a full shell command line
|
|
18
21
|
# @param chdir [String]
|
|
19
22
|
# @return [Array(String, String, Boolean)] stdout, stderr, success
|
|
20
23
|
def call(command, chdir:)
|
|
24
|
+
@pid = @process_group_id = nil
|
|
21
25
|
stdout_r, stdout_w = IO.pipe
|
|
22
26
|
stderr_r, stderr_w = IO.pipe
|
|
23
27
|
spawn_child(command, chdir, stdout_w, stderr_w)
|
|
@@ -59,7 +63,8 @@ module Woods
|
|
|
59
63
|
# terminated on timeout.
|
|
60
64
|
def spawn_child(command, chdir, stdout_w, stderr_w)
|
|
61
65
|
Thread.handle_interrupt(Timeout::Error => :never) do
|
|
62
|
-
@pid = Process.spawn(command, chdir: chdir, in: File::NULL, out: stdout_w, err: stderr_w)
|
|
66
|
+
@pid = Process.spawn(command, chdir: chdir, in: File::NULL, out: stdout_w, err: stderr_w, pgroup: true)
|
|
67
|
+
@process_group_id = @pid
|
|
63
68
|
end
|
|
64
69
|
end
|
|
65
70
|
end
|
|
@@ -16,7 +16,10 @@ module Woods
|
|
|
16
16
|
# subprocess started by the wrapped executor keeps running unless
|
|
17
17
|
# something kills it. When the wrapped executor exposes its most recent
|
|
18
18
|
# pid (see {AblationExecutor}), this sends TERM, waits briefly, then KILL
|
|
19
|
-
# if
|
|
19
|
+
# if still alive. The default executor owns a process group, so cleanup
|
|
20
|
+
# includes ordinary descendants even if their parent already exited.
|
|
21
|
+
# Custom executors exposing only a pid retain single-process cleanup.
|
|
22
|
+
# Cleanup finishes before reporting the call as a timed-out
|
|
20
23
|
# failure. An executor that does not expose a pid (for example a fake
|
|
21
24
|
# executor in a spec) still gets the timeout, just without a process to
|
|
22
25
|
# terminate.
|
|
@@ -47,15 +50,30 @@ module Woods
|
|
|
47
50
|
pid = executor_pid
|
|
48
51
|
return '' unless pid
|
|
49
52
|
|
|
50
|
-
|
|
51
|
-
|
|
53
|
+
target = signal_target(pid)
|
|
54
|
+
signal('TERM', target)
|
|
55
|
+
signal('KILL', target) unless process_exited?(target, within: TERM_GRACE_SECONDS)
|
|
52
56
|
reap(pid, within: REAP_GRACE_SECONDS) ? '' : " (pid #{pid} did not reap within #{REAP_GRACE_SECONDS}s)"
|
|
53
57
|
rescue Errno::ESRCH, Errno::ECHILD
|
|
54
58
|
''
|
|
55
59
|
end
|
|
56
60
|
|
|
57
61
|
def executor_pid
|
|
58
|
-
@executor.pid if @executor.respond_to?(:pid)
|
|
62
|
+
pid = @executor.pid if @executor.respond_to?(:pid)
|
|
63
|
+
pid if pid.is_a?(Integer) && pid.positive?
|
|
64
|
+
end
|
|
65
|
+
|
|
66
|
+
# A group created by spawn(pgroup: true) is named after that child.
|
|
67
|
+
# Never infer a group from a custom pid or signal our own group.
|
|
68
|
+
def signal_target(pid)
|
|
69
|
+
group = @executor.process_group_id if @executor.respond_to?(:process_group_id)
|
|
70
|
+
group == pid && group != Process.getpgrp ? -group : pid
|
|
71
|
+
end
|
|
72
|
+
|
|
73
|
+
def signal(name, target)
|
|
74
|
+
Process.kill(name, target)
|
|
75
|
+
rescue Errno::ESRCH
|
|
76
|
+
nil
|
|
59
77
|
end
|
|
60
78
|
|
|
61
79
|
def process_exited?(pid, within:)
|