woods 2.0.0.beta2 → 2.0.0.beta3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +262 -1
- data/CONTRIBUTING.md +173 -9
- data/README.md +7 -3
- data/SECURITY.md +9 -6
- data/docs/AGENT_GUIDE.md +83 -4
- data/docs/AGENT_SETUP.md +82 -1
- data/docs/BACKEND_MATRIX.md +20 -0
- data/docs/CLIENT_HOOKS.md +111 -0
- data/docs/CONFIGURATION_REFERENCE.md +199 -14
- data/docs/CONSOLE_MCP_SETUP.md +35 -5
- data/docs/DOCKER_SETUP.md +21 -2
- data/docs/EVALUATION.md +464 -1
- data/docs/EXTRACTOR_REFERENCE.md +36 -5
- data/docs/FAQ.md +11 -12
- data/docs/GETTING_STARTED.md +17 -5
- data/docs/INCREMENTAL_EXTRACTION.md +117 -1
- data/docs/INDEX_LAYOUT.md +382 -0
- data/docs/INTERNALS.md +7 -2
- data/docs/MCP_SERVERS.md +221 -5
- data/docs/MCP_TOOL_COOKBOOK.md +33 -18
- data/docs/NOTION_INTEGRATION.md +13 -0
- data/docs/OBSIDIAN_INTEGRATION.md +57 -9
- data/docs/PUBLISHED_INDEX.md +55 -0
- data/docs/README.md +7 -0
- data/docs/RETRIEVAL_GUIDE.md +253 -11
- data/docs/RUNTIME_TRACING.md +71 -0
- data/docs/SOURCE_FRESHNESS.md +143 -0
- data/docs/TROUBLESHOOTING.md +117 -5
- data/docs/UNBLOCKED_INTEGRATION.md +25 -0
- data/docs/UPGRADING_TO_2.md +44 -22
- data/docs/WATCH_DAEMON.md +259 -59
- data/exe/woods-agent-config +6 -0
- data/exe/woods-extract +5 -0
- data/exe/woods-hook-context +6 -0
- data/lib/generators/woods/templates/woods.rb.tt +1 -3
- data/lib/tasks/woods.rake +47 -397
- data/lib/woods/agent_configuration/applier.rb +133 -0
- data/lib/woods/agent_configuration/cli.rb +101 -0
- data/lib/woods/agent_configuration/cli_options.rb +29 -0
- data/lib/woods/agent_configuration/document.rb +105 -0
- data/lib/woods/agent_configuration/error.rb +7 -0
- data/lib/woods/agent_configuration/launcher.rb +75 -0
- data/lib/woods/agent_configuration/layout.rb +59 -0
- data/lib/woods/agent_configuration/managed_section.rb +62 -0
- data/lib/woods/agent_configuration/plan.rb +98 -0
- data/lib/woods/agent_configuration/plan_diff.rb +38 -0
- data/lib/woods/agent_configuration/planned_files.rb +61 -0
- data/lib/woods/agent_configuration/planner.rb +63 -0
- data/lib/woods/agent_configuration/planner_validation.rb +77 -0
- data/lib/woods/agent_configuration/preflight.rb +100 -0
- data/lib/woods/agent_configuration/recovery.rb +49 -0
- data/lib/woods/ast/node.rb +2 -0
- data/lib/woods/ast/parser.rb +38 -5
- data/lib/woods/builder.rb +21 -5
- data/lib/woods/cache/cache_middleware.rb +28 -7
- data/lib/woods/cache/cache_store.rb +4 -5
- data/lib/woods/change_set.rb +5 -4
- data/lib/woods/console/credential_index.rb +20 -2
- data/lib/woods/console/credential_scanner.rb +14 -14
- data/lib/woods/console/credential_scanner_registry.rb +36 -0
- data/lib/woods/console/embedded_executor.rb +1 -1
- data/lib/woods/console/encrypted_credential_snapshot.rb +16 -0
- data/lib/woods/console/rack_middleware.rb +22 -13
- data/lib/woods/console/server.rb +18 -16
- data/lib/woods/dependency_graph.rb +65 -13
- data/lib/woods/embedding/corpus.rb +94 -0
- data/lib/woods/embedding/indexer.rb +90 -46
- data/lib/woods/embedding/openai.rb +17 -6
- data/lib/woods/evaluation/ablation_executor.rb +6 -1
- data/lib/woods/evaluation/ablation_timed_executor.rb +22 -4
- data/lib/woods/export/typed_reader.rb +56 -0
- data/lib/woods/extractor.rb +232 -137
- data/lib/woods/extractors/action_cable_extractor.rb +3 -1
- data/lib/woods/extractors/behavioral_profile.rb +9 -7
- data/lib/woods/extractors/caching_extractor.rb +3 -1
- data/lib/woods/extractors/concern_extractor.rb +64 -6
- data/lib/woods/extractors/configuration_extractor.rb +7 -3
- data/lib/woods/extractors/controller_extractor.rb +13 -4
- data/lib/woods/extractors/database_view_extractor.rb +3 -1
- data/lib/woods/extractors/decorator_extractor.rb +3 -1
- data/lib/woods/extractors/engine_extractor.rb +3 -1
- data/lib/woods/extractors/event_extractor.rb +4 -2
- data/lib/woods/extractors/factory_extractor.rb +3 -1
- data/lib/woods/extractors/graphql_extractor.rb +8 -2
- data/lib/woods/extractors/i18n_extractor.rb +3 -1
- data/lib/woods/extractors/job_extractor.rb +6 -19
- data/lib/woods/extractors/lib_extractor.rb +3 -1
- data/lib/woods/extractors/mailer_extractor.rb +20 -5
- data/lib/woods/extractors/manager_extractor.rb +3 -1
- data/lib/woods/extractors/method_parameters.rb +53 -0
- data/lib/woods/extractors/middleware_argument.rb +65 -0
- data/lib/woods/extractors/middleware_extractor.rb +9 -3
- data/lib/woods/extractors/migration_extractor.rb +3 -1
- data/lib/woods/extractors/model_extractor.rb +39 -33
- data/lib/woods/extractors/package_extractor.rb +24 -4
- data/lib/woods/extractors/phlex_extractor.rb +3 -1
- data/lib/woods/extractors/policy_extractor.rb +3 -1
- data/lib/woods/extractors/poro_extractor.rb +3 -1
- data/lib/woods/extractors/pundit_extractor.rb +3 -1
- data/lib/woods/extractors/rails_source_extractor.rb +4 -2
- data/lib/woods/extractors/rake_task_extractor.rb +4 -2
- data/lib/woods/extractors/route_extractor.rb +3 -1
- data/lib/woods/extractors/route_helper_resolver.rb +10 -33
- data/lib/woods/extractors/scheduled_job_extractor.rb +41 -15
- data/lib/woods/extractors/serializer_extractor.rb +4 -2
- data/lib/woods/extractors/service_extractor.rb +3 -1
- data/lib/woods/extractors/shared_dependency_scanner.rb +2 -2
- data/lib/woods/extractors/shared_utility_methods.rb +27 -15
- data/lib/woods/extractors/source_nesting.rb +1 -1
- data/lib/woods/extractors/state_machine_extractor.rb +3 -1
- data/lib/woods/extractors/test_mapping_extractor.rb +3 -1
- data/lib/woods/extractors/validator_extractor.rb +3 -1
- data/lib/woods/extractors/view_component_extractor.rb +3 -1
- data/lib/woods/extractors/view_template_extractor.rb +3 -1
- data/lib/woods/gem_mapper.rb +2 -0
- data/lib/woods/git_history.rb +116 -0
- data/lib/woods/graph_analyzer.rb +35 -6
- data/lib/woods/hooks/context_cli.rb +54 -0
- data/lib/woods/hooks/context_event.rb +88 -0
- data/lib/woods/hooks/context_hint.rb +73 -0
- data/lib/woods/hooks/context_impact.rb +77 -0
- data/lib/woods/hooks/context_output.rb +47 -0
- data/lib/woods/hooks/context_state.rb +102 -0
- data/lib/woods/hooks/refresh.rb +79 -0
- data/lib/woods/hooks/rule_projection.rb +78 -0
- data/lib/woods/input_rules.rb +19 -0
- data/lib/woods/mcp/bearer_auth.rb +20 -12
- data/lib/woods/mcp/bootstrapper.rb +62 -0
- data/lib/woods/mcp/index_reader.rb +323 -160
- data/lib/woods/mcp/initialization_guidance.rb +27 -0
- data/lib/woods/mcp/origin_guard.rb +17 -9
- data/lib/woods/mcp/published_lexical_retriever.rb +115 -0
- data/lib/woods/mcp/renderers/markdown_renderer.rb +8 -1
- data/lib/woods/mcp/renderers/plain_renderer.rb +7 -1
- data/lib/woods/mcp/search_results.rb +74 -0
- data/lib/woods/mcp/server.rb +158 -37
- data/lib/woods/mcp/tool_contract.rb +2 -0
- data/lib/woods/mcp/tool_response_renderer.rb +25 -0
- data/lib/woods/mcp/traversal_evidence.rb +113 -0
- data/lib/woods/mcp/traversal_evidence_index.rb +100 -0
- data/lib/woods/mcp/traversal_evidence_page.rb +41 -0
- data/lib/woods/mcp/traversal_evidence_text.rb +52 -0
- data/lib/woods/notion/exporter.rb +56 -17
- data/lib/woods/obsidian/destination_plan.rb +98 -0
- data/lib/woods/obsidian/name_mapper.rb +19 -3
- data/lib/woods/obsidian/note_builder.rb +19 -10
- data/lib/woods/obsidian/vault_exporter.rb +88 -32
- data/lib/woods/operator/pipeline_guard.rb +18 -13
- data/lib/woods/path_dispatcher.rb +7 -1
- data/lib/woods/payload_store.rb +27 -26
- data/lib/woods/railtie.rb +3 -3
- data/lib/woods/railtie_support.rb +12 -12
- data/lib/woods/rake_helpers.rb +392 -0
- data/lib/woods/resilience/graph_invariant_validator/membership_checks.rb +71 -0
- data/lib/woods/resilience/graph_invariant_validator/node_checks.rb +61 -0
- data/lib/woods/resilience/graph_invariant_validator/reverse_relationship_checks.rb +46 -0
- data/lib/woods/resilience/graph_invariant_validator.rb +119 -0
- data/lib/woods/resilience/index_validator/graph_checks.rb +80 -0
- data/lib/woods/resilience/index_validator.rb +112 -23
- data/lib/woods/retrieval/context_assembler.rb +50 -15
- data/lib/woods/retrieval/lexical_assembler.rb +73 -0
- data/lib/woods/retrieval/lexical_index.rb +119 -0
- data/lib/woods/retrieval/ranker.rb +4 -2
- data/lib/woods/retrieval/scope.rb +108 -0
- data/lib/woods/retrieval/scoped_graph_store.rb +32 -0
- data/lib/woods/retrieval/scoped_vector_store.rb +55 -0
- data/lib/woods/retrieval/search_executor.rb +86 -27
- data/lib/woods/retrieval/source_evidence.rb +200 -0
- data/lib/woods/retriever.rb +98 -22
- data/lib/woods/ruby_analyzer/trace_enricher.rb +77 -38
- data/lib/woods/session_tracer/middleware.rb +10 -12
- data/lib/woods/session_tracer/redis_store.rb +22 -6
- data/lib/woods/session_tracer/session_flow_assembler.rb +23 -17
- data/lib/woods/session_tracer/solid_cache_coordination.rb +6 -4
- data/lib/woods/session_tracer/unit_resolver.rb +63 -0
- data/lib/woods/source_inputs/consumer_errors.rb +27 -0
- data/lib/woods/source_inputs/handoff.rb +102 -0
- data/lib/woods/source_inputs/launcher.rb +157 -0
- data/lib/woods/source_inputs/manifest.rb +124 -0
- data/lib/woods/source_inputs/private_key.rb +55 -0
- data/lib/woods/source_inputs/scanner.rb +171 -0
- data/lib/woods/source_inputs/scopes.rb +71 -0
- data/lib/woods/source_inputs/session.rb +214 -0
- data/lib/woods/source_inputs/status.rb +84 -0
- data/lib/woods/source_inputs/verifier.rb +107 -0
- data/lib/woods/storage/metadata_store.rb +25 -25
- data/lib/woods/storage/pgvector.rb +29 -8
- data/lib/woods/storage/qdrant.rb +17 -7
- data/lib/woods/storage/vector_store.rb +18 -6
- data/lib/woods/tasks.rb +3 -2
- data/lib/woods/temporal/json_snapshot_store.rb +29 -8
- data/lib/woods/unblocked/exporter.rb +59 -70
- data/lib/woods/version.rb +1 -1
- data/lib/woods/watch/boot_snapshot.rb +52 -0
- data/lib/woods/watch/daemon.rb +136 -28
- data/lib/woods/watch/listen_watcher.rb +4 -0
- data/lib/woods/watch/polling_watcher.rb +5 -1
- data/lib/woods/watch/status.rb +20 -15
- data/lib/woods/watch/tree_scan.rb +21 -13
- data/lib/woods/watch/watcher.rb +4 -1
- data/lib/woods.rb +50 -11
- data/plugin/.claude-plugin/plugin.json +1 -1
- data/plugin/hooks/adapters/normalize.jq +15 -0
- data/plugin/hooks/adapters/normalize.rb +63 -0
- data/plugin/hooks/hooks.json +20 -0
- data/plugin/hooks/woods-context.sh +50 -0
- data/plugin/hooks/woods-input-rules.sh +159 -0
- data/plugin/hooks/woods-opencode.mjs +65 -0
- data/plugin/hooks/woods-post-edit.sh +2 -225
- data/plugin/hooks/woods-refresh.sh +260 -0
- data/plugin/hooks/woods-session-start.sh +47 -55
- data/plugin/skills/woods-agent-enable/SKILL.md +13 -0
- data/plugin/skills/woods-diagnose/SKILL.md +288 -1
- data/plugin/skills/woods-investigate/SKILL.md +106 -0
- data/plugin/skills/woods-mcp-config/SKILL.md +89 -1
- data/plugin/skills/woods-setup/SKILL.md +107 -6
- metadata +84 -5
|
@@ -119,40 +119,14 @@ module Woods
|
|
|
119
119
|
|
|
120
120
|
private
|
|
121
121
|
|
|
122
|
-
# Where this run's units are read from — the published generation's
|
|
123
|
-
# payload, or the output root for a flat index.
|
|
124
|
-
#
|
|
125
|
-
# Globbing the output root would walk `payloads/` and ingest every
|
|
126
|
-
# retained generation, so each unit would be embedded once per retained
|
|
127
|
-
# payload. It also picks up `dumps/`, which holds no unit JSON but is
|
|
128
|
-
# needlessly large to walk.
|
|
129
|
-
#
|
|
130
|
-
# @return [String]
|
|
131
|
-
def units_dir
|
|
132
|
-
Woods::Generation.new(output_dir: @output_dir).payload_dir.to_s
|
|
133
|
-
end
|
|
134
|
-
|
|
135
122
|
def load_units
|
|
136
|
-
|
|
137
|
-
|
|
138
|
-
|
|
139
|
-
# AtomicFile.read, not File.read: a bare read tags the bytes with the
|
|
140
|
-
# process's default external encoding (US-ASCII under LANG=C), so a
|
|
141
|
-
# single multibyte character in one unit raised EncodingError out of
|
|
142
|
-
# JSON.parse and aborted the whole embed run.
|
|
143
|
-
data = JSON.parse(AtomicFile.read(path))
|
|
144
|
-
# Extraction output also contains index listings (_index.json arrays) and
|
|
145
|
-
# summary files (manifest.json, dependency_graph.json, graph_analysis.json)
|
|
146
|
-
# that live alongside per-unit JSON. Filter to the unit shape.
|
|
147
|
-
data if data.is_a?(Hash) && data.key?('type') && data.key?('identifier')
|
|
148
|
-
rescue JSON::ParserError, EncodingError => e
|
|
149
|
-
warn "[woods] skipping unreadable unit file #{path} (#{e.class}: #{e.message})"
|
|
150
|
-
nil
|
|
151
|
-
end
|
|
123
|
+
require_relative 'corpus'
|
|
124
|
+
|
|
125
|
+
Corpus.new(@output_dir).load
|
|
152
126
|
end
|
|
153
127
|
|
|
154
|
-
# The invariant: **checkpoint.json
|
|
155
|
-
# vector
|
|
128
|
+
# The invariant: **checkpoint.json advances only after the intended
|
|
129
|
+
# vector state (including an empty set) is durable.** Two things uphold it here.
|
|
156
130
|
#
|
|
157
131
|
# 1. Ordering. For a store whose only durable copy is the dump
|
|
158
132
|
# (+persistable?+), the checkpoint is written *after*
|
|
@@ -170,17 +144,19 @@ module Woods
|
|
|
170
144
|
#
|
|
171
145
|
# 2. Trust, verified. {#checkpoint_satisfied?} honours a checkpoint hit
|
|
172
146
|
# only when the durable artifact actually holds a vector for that
|
|
173
|
-
# unit
|
|
147
|
+
# unit, or preparation verifies that it intentionally has no text.
|
|
148
|
+
# A checkpoint that ran ahead of its dump — an older gem with
|
|
174
149
|
# this bug, an interrupted promote, a store swap — self-heals into a
|
|
175
150
|
# re-embed instead of stranding the unit forever.
|
|
176
151
|
def process_units(units, incremental:)
|
|
177
152
|
prepare_run(incremental: incremental)
|
|
178
|
-
units = assign_storage_identities(units)
|
|
179
153
|
checkpoint = incremental ? load_checkpoint : {}
|
|
154
|
+
units = assign_storage_identities(units, checkpoint: checkpoint)
|
|
180
155
|
stats = { processed: 0, skipped: 0, errors: 0 }
|
|
181
156
|
|
|
182
157
|
embed_batches(units, checkpoint, stats, incremental: incremental)
|
|
183
158
|
|
|
159
|
+
reconcile_empty_units(checkpoint)
|
|
184
160
|
retire_legacy_identities
|
|
185
161
|
report_checkpoint_misses
|
|
186
162
|
vanished = incremental && persistable? ? drop_vanished_units : 0
|
|
@@ -196,12 +172,12 @@ module Woods
|
|
|
196
172
|
end
|
|
197
173
|
|
|
198
174
|
# Unambiguous existing keys stay stable. A collision uses reversible typed keys.
|
|
199
|
-
def assign_storage_identities(units)
|
|
175
|
+
def assign_storage_identities(units, checkpoint:)
|
|
200
176
|
counts = units.group_by { |unit| unit['identifier'] }.transform_values(&:size)
|
|
201
177
|
units.map do |unit|
|
|
202
178
|
id = unit['identifier']
|
|
203
179
|
typed = StorageIdentity.key(id, unit['type'])
|
|
204
|
-
existing = known_storage_key?(typed)
|
|
180
|
+
existing = known_storage_key?(typed, checkpoint: checkpoint)
|
|
205
181
|
next unit unless counts[id] > 1 || existing || id.start_with?(StorageIdentity::PREFIX)
|
|
206
182
|
|
|
207
183
|
unit.merge('storage_id' => typed)
|
|
@@ -224,8 +200,8 @@ module Woods
|
|
|
224
200
|
end
|
|
225
201
|
end
|
|
226
202
|
|
|
227
|
-
def known_storage_key?(key)
|
|
228
|
-
(@persisted_ids || {}).key?(key) || (@durable_ids || {}).key?(key)
|
|
203
|
+
def known_storage_key?(key, checkpoint:)
|
|
204
|
+
(@persisted_ids || {}).key?(key) || (@durable_ids || {}).key?(key) || checkpoint.key?(key)
|
|
229
205
|
end
|
|
230
206
|
|
|
231
207
|
def retire_legacy_key(legacy)
|
|
@@ -243,11 +219,9 @@ module Woods
|
|
|
243
219
|
# `woods:embed_incremental` runs evict every genuinely older dump in
|
|
244
220
|
# favour of copies of the same state.
|
|
245
221
|
#
|
|
246
|
-
#
|
|
247
|
-
#
|
|
248
|
-
#
|
|
249
|
-
# self-heal counts as processed, so a run that re-embeds a stranded unit
|
|
250
|
-
# still dumps.
|
|
222
|
+
# A zero-text transition can retire vectors without a provider call;
|
|
223
|
+
# @vectors_changed captures that case. Checkpoint self-heals still count
|
|
224
|
+
# as processed, so a run re-embedding a stranded unit also dumps.
|
|
251
225
|
#
|
|
252
226
|
# But "nothing embedded" is not "nothing changed" (B-069). `persist_snapshot`
|
|
253
227
|
# writes the vector dump, the *metadata* dump and the config into one
|
|
@@ -266,6 +240,11 @@ module Woods
|
|
|
266
240
|
# `@current_identifiers` is what this run saw. Anything in the first and
|
|
267
241
|
# not the second is stale, and the dump has to be rewritten to drop it.
|
|
268
242
|
#
|
|
243
|
+
# Existing units can also change metadata without changing source (B-119).
|
|
244
|
+
# Compare their complete records with the promoted metadata snapshot;
|
|
245
|
+
# publishing those changes must not require a provider call or invalidate
|
|
246
|
+
# the source-hash checkpoint contract.
|
|
247
|
+
#
|
|
269
248
|
# Full runs always dump: they rebuild the store from scratch, so "nothing
|
|
270
249
|
# processed" there means the store is genuinely empty and the dump must
|
|
271
250
|
# say so rather than leave a stale one promoted.
|
|
@@ -274,7 +253,7 @@ module Woods
|
|
|
274
253
|
def snapshot_worth_writing?(stats, vanished, incremental:)
|
|
275
254
|
return true unless incremental
|
|
276
255
|
|
|
277
|
-
stats[:processed].positive? || vanished.positive?
|
|
256
|
+
stats[:processed].positive? || vanished.positive? || @metadata_changed || @vectors_changed
|
|
278
257
|
end
|
|
279
258
|
|
|
280
259
|
# Fraction of the persisted units the vanished-unit sweep may remove
|
|
@@ -489,10 +468,36 @@ module Woods
|
|
|
489
468
|
@current_identifiers = Set.new
|
|
490
469
|
@durable_ids = nil
|
|
491
470
|
@checkpoint_misses = 0
|
|
492
|
-
|
|
471
|
+
@metadata_changed = false
|
|
472
|
+
@vectors_changed = false
|
|
473
|
+
@empty_units = {}
|
|
474
|
+
@persisted_metadata = nil
|
|
475
|
+
prepare_snapshot_stores(incremental: incremental)
|
|
476
|
+
retain_metadata_identities
|
|
493
477
|
load_durable_store_ids if reconcilable?
|
|
494
478
|
end
|
|
495
479
|
|
|
480
|
+
def prepare_snapshot_stores(incremental:)
|
|
481
|
+
return unless persistable?
|
|
482
|
+
|
|
483
|
+
if incremental
|
|
484
|
+
hydrate_persisted_metadata
|
|
485
|
+
hydrate_persisted_vectors
|
|
486
|
+
else
|
|
487
|
+
# A direct caller may reuse an in-memory adapter for a full run.
|
|
488
|
+
# Its old chunks still need replacement, including by an empty set.
|
|
489
|
+
entries = []
|
|
490
|
+
@vector_store.each_entry { |id, _vector, _metadata| entries << { id: id } }
|
|
491
|
+
@persisted_ids = index_ids_by_identifier(entries)
|
|
492
|
+
end
|
|
493
|
+
end
|
|
494
|
+
|
|
495
|
+
# Source-empty units retain metadata but intentionally have no vectors.
|
|
496
|
+
# Keep those identities in the same deletion and typed-key accounting.
|
|
497
|
+
def retain_metadata_identities
|
|
498
|
+
@persisted_metadata&.each_entry { |identifier, _unit| @persisted_ids[identifier] ||= [] }
|
|
499
|
+
end
|
|
500
|
+
|
|
496
501
|
# Read back what the durable store currently holds, as base identifiers.
|
|
497
502
|
#
|
|
498
503
|
# One enumeration serves both halves of #211:
|
|
@@ -618,7 +623,10 @@ module Woods
|
|
|
618
623
|
# enumeration that failed) — fall back to trusting the checkpoint.
|
|
619
624
|
return true if known_ids.nil?
|
|
620
625
|
|
|
621
|
-
return true if known_ids
|
|
626
|
+
return true if known_ids[storage_id(unit_data)]&.any?
|
|
627
|
+
# A source-empty unit intentionally has no vector. Verify that state
|
|
628
|
+
# again rather than treating a missing nonempty vector as a cache hit.
|
|
629
|
+
return true if prepare_texts(unit_data).empty?
|
|
622
630
|
|
|
623
631
|
@checkpoint_misses += 1
|
|
624
632
|
false
|
|
@@ -635,7 +643,22 @@ module Woods
|
|
|
635
643
|
def persist_unit_metadata(unit_data)
|
|
636
644
|
return unless @metadata_store
|
|
637
645
|
|
|
638
|
-
|
|
646
|
+
id = storage_id(unit_data)
|
|
647
|
+
@metadata_changed ||= @persisted_metadata && @persisted_metadata.find(id) != unit_data
|
|
648
|
+
@metadata_store.store(id, unit_data)
|
|
649
|
+
end
|
|
650
|
+
|
|
651
|
+
# Compare with the promoted artifact, not a fresh metadata store or the
|
|
652
|
+
# source-only checkpoint. Retain old records until the guarded vanished
|
|
653
|
+
# sweep permits their deletion, just as vector hydration does.
|
|
654
|
+
def hydrate_persisted_metadata
|
|
655
|
+
return unless @metadata_store.respond_to?(:each_entry) && @metadata_store.respond_to?(:bulk_load)
|
|
656
|
+
|
|
657
|
+
require_relative '../index_artifact'
|
|
658
|
+
require_relative '../storage/snapshotter'
|
|
659
|
+
|
|
660
|
+
@persisted_metadata = Storage::Snapshotter::Metadata.load_or_empty(IndexArtifact.new(@output_dir))
|
|
661
|
+
@metadata_store.bulk_load(@persisted_metadata.each_entry)
|
|
639
662
|
end
|
|
640
663
|
|
|
641
664
|
# Refuse to index a unit whose real identifier already matches the
|
|
@@ -653,6 +676,7 @@ module Woods
|
|
|
653
676
|
def collect_embed_items(unit_data, items)
|
|
654
677
|
texts = prepare_texts(unit_data)
|
|
655
678
|
identifier = storage_id(unit_data)
|
|
679
|
+
@empty_units[identifier] = unit_data['source_hash'] if texts.empty?
|
|
656
680
|
|
|
657
681
|
texts.each_with_index do |text, idx|
|
|
658
682
|
embed_id = texts.length > 1 ? "#{identifier}#chunk_#{idx}" : identifier
|
|
@@ -661,6 +685,26 @@ module Woods
|
|
|
661
685
|
end
|
|
662
686
|
end
|
|
663
687
|
|
|
688
|
+
# Defer zero-text deletion until every batch has prepared/embedded
|
|
689
|
+
# successfully. A later provider failure must not retire earlier vectors.
|
|
690
|
+
# Checkpoints retain their source-hash format; a matching no-vector hash
|
|
691
|
+
# is trusted only after verifying the current input still prepares empty.
|
|
692
|
+
def reconcile_empty_units(checkpoint)
|
|
693
|
+
verify_empty_reconciliation!
|
|
694
|
+
@empty_units.each do |identifier, source_hash|
|
|
695
|
+
@vectors_changed ||= @persisted_ids[identifier]&.any?
|
|
696
|
+
prune_identifier(identifier, []) if implements_own?(@vector_store, :delete)
|
|
697
|
+
prune_durable_identifier(identifier, []) if @durable_ids && implements_own?(@vector_store, :delete)
|
|
698
|
+
checkpoint[identifier] = source_hash
|
|
699
|
+
end
|
|
700
|
+
end
|
|
701
|
+
|
|
702
|
+
def verify_empty_reconciliation!
|
|
703
|
+
return unless @empty_units.any? && reconcilable? && @durable_ids.nil?
|
|
704
|
+
|
|
705
|
+
raise Woods::Error, 'Cannot reconcile source-empty units: existing vector IDs could not be read'
|
|
706
|
+
end
|
|
707
|
+
|
|
664
708
|
def prepare_texts(unit_data) # rubocop:disable Metrics/CyclomaticComplexity, Metrics/PerceivedComplexity
|
|
665
709
|
unit = build_unit(unit_data)
|
|
666
710
|
apply_chunking(unit) if @chunker && unit.chunks.empty? && needs_chunking?(unit)
|
|
@@ -16,7 +16,7 @@ module Woods
|
|
|
16
16
|
# provider = Woods::Embedding::Provider::OpenAI.new(api_key: ENV['OPENAI_API_KEY'])
|
|
17
17
|
# vector = provider.embed("class User < ApplicationRecord; end")
|
|
18
18
|
# vectors = provider.embed_batch(["text1", "text2"])
|
|
19
|
-
class OpenAI
|
|
19
|
+
class OpenAI # rubocop:disable Metrics/ClassLength
|
|
20
20
|
include Interface
|
|
21
21
|
include DiscardableClient
|
|
22
22
|
|
|
@@ -26,13 +26,20 @@ module Woods
|
|
|
26
26
|
'text-embedding-3-small' => 1536,
|
|
27
27
|
'text-embedding-3-large' => 3072
|
|
28
28
|
}.freeze
|
|
29
|
-
#
|
|
29
|
+
# Conservatively chunk below OpenAI's 8192-token input limit across
|
|
30
30
|
# text-embedding-3-small / -3-large / ada-002. The chunker uses
|
|
31
|
-
#
|
|
31
|
+
# 8191 as its ceiling — the actual chunk size lands well
|
|
32
32
|
# below it once chars-per-token estimation and the prefix
|
|
33
33
|
# allowance are factored in (see Builder#build_chunker).
|
|
34
34
|
MAX_INPUT_TOKENS = 8191
|
|
35
35
|
|
|
36
|
+
# The API accepts at most 2048 inputs and 300,000 total tokens per
|
|
37
|
+
# request, with 8192 tokens per valid input. At most 36 inputs keeps
|
|
38
|
+
# even maximum-size texts below both limits without token estimates.
|
|
39
|
+
# https://developers.openai.com/api/reference/resources/embeddings/methods/create
|
|
40
|
+
MAX_REQUEST_INPUTS = 36
|
|
41
|
+
private_constant :MAX_REQUEST_INPUTS
|
|
42
|
+
|
|
36
43
|
# @param api_key [String] OpenAI API key
|
|
37
44
|
# @param model [String] OpenAI embedding model name (default: text-embedding-3-small)
|
|
38
45
|
# @param dimensions [Integer, nil] Requested output size for text-embedding-3 models
|
|
@@ -57,7 +64,7 @@ module Woods
|
|
|
57
64
|
vectors.first
|
|
58
65
|
end
|
|
59
66
|
|
|
60
|
-
# Embed multiple texts in
|
|
67
|
+
# Embed multiple texts in bounded requests, preserving input order.
|
|
61
68
|
#
|
|
62
69
|
# Sorts results by the index field to guarantee ordering matches input.
|
|
63
70
|
#
|
|
@@ -71,8 +78,12 @@ module Woods
|
|
|
71
78
|
raise ArgumentError, 'embed_batch(texts) rejects nil/empty entries (OpenAI returns 400)'
|
|
72
79
|
end
|
|
73
80
|
|
|
74
|
-
|
|
75
|
-
|
|
81
|
+
vectors = texts.each_slice(MAX_REQUEST_INPUTS).flat_map do |slice|
|
|
82
|
+
response = post_request(request_body(slice))
|
|
83
|
+
extract_validated_batch(response, slice.size)
|
|
84
|
+
end
|
|
85
|
+
VectorValidation.validate!(vectors, expected_count: texts.size, provider: 'OpenAI')
|
|
86
|
+
vectors
|
|
76
87
|
end
|
|
77
88
|
|
|
78
89
|
# Return the dimensionality of vectors produced by this model.
|
|
@@ -14,10 +14,14 @@ module Woods
|
|
|
14
14
|
# @return [Integer, nil] the pid of the most recently spawned process
|
|
15
15
|
attr_reader :pid
|
|
16
16
|
|
|
17
|
+
# @return [Integer, nil] process group owned by the current command
|
|
18
|
+
attr_reader :process_group_id
|
|
19
|
+
|
|
17
20
|
# @param command [String] a full shell command line
|
|
18
21
|
# @param chdir [String]
|
|
19
22
|
# @return [Array(String, String, Boolean)] stdout, stderr, success
|
|
20
23
|
def call(command, chdir:)
|
|
24
|
+
@pid = @process_group_id = nil
|
|
21
25
|
stdout_r, stdout_w = IO.pipe
|
|
22
26
|
stderr_r, stderr_w = IO.pipe
|
|
23
27
|
spawn_child(command, chdir, stdout_w, stderr_w)
|
|
@@ -59,7 +63,8 @@ module Woods
|
|
|
59
63
|
# terminated on timeout.
|
|
60
64
|
def spawn_child(command, chdir, stdout_w, stderr_w)
|
|
61
65
|
Thread.handle_interrupt(Timeout::Error => :never) do
|
|
62
|
-
@pid = Process.spawn(command, chdir: chdir, in: File::NULL, out: stdout_w, err: stderr_w)
|
|
66
|
+
@pid = Process.spawn(command, chdir: chdir, in: File::NULL, out: stdout_w, err: stderr_w, pgroup: true)
|
|
67
|
+
@process_group_id = @pid
|
|
63
68
|
end
|
|
64
69
|
end
|
|
65
70
|
end
|
|
@@ -16,7 +16,10 @@ module Woods
|
|
|
16
16
|
# subprocess started by the wrapped executor keeps running unless
|
|
17
17
|
# something kills it. When the wrapped executor exposes its most recent
|
|
18
18
|
# pid (see {AblationExecutor}), this sends TERM, waits briefly, then KILL
|
|
19
|
-
# if
|
|
19
|
+
# if still alive. The default executor owns a process group, so cleanup
|
|
20
|
+
# includes ordinary descendants even if their parent already exited.
|
|
21
|
+
# Custom executors exposing only a pid retain single-process cleanup.
|
|
22
|
+
# Cleanup finishes before reporting the call as a timed-out
|
|
20
23
|
# failure. An executor that does not expose a pid (for example a fake
|
|
21
24
|
# executor in a spec) still gets the timeout, just without a process to
|
|
22
25
|
# terminate.
|
|
@@ -47,15 +50,30 @@ module Woods
|
|
|
47
50
|
pid = executor_pid
|
|
48
51
|
return '' unless pid
|
|
49
52
|
|
|
50
|
-
|
|
51
|
-
|
|
53
|
+
target = signal_target(pid)
|
|
54
|
+
signal('TERM', target)
|
|
55
|
+
signal('KILL', target) unless process_exited?(target, within: TERM_GRACE_SECONDS)
|
|
52
56
|
reap(pid, within: REAP_GRACE_SECONDS) ? '' : " (pid #{pid} did not reap within #{REAP_GRACE_SECONDS}s)"
|
|
53
57
|
rescue Errno::ESRCH, Errno::ECHILD
|
|
54
58
|
''
|
|
55
59
|
end
|
|
56
60
|
|
|
57
61
|
def executor_pid
|
|
58
|
-
@executor.pid if @executor.respond_to?(:pid)
|
|
62
|
+
pid = @executor.pid if @executor.respond_to?(:pid)
|
|
63
|
+
pid if pid.is_a?(Integer) && pid.positive?
|
|
64
|
+
end
|
|
65
|
+
|
|
66
|
+
# A group created by spawn(pgroup: true) is named after that child.
|
|
67
|
+
# Never infer a group from a custom pid or signal our own group.
|
|
68
|
+
def signal_target(pid)
|
|
69
|
+
group = @executor.process_group_id if @executor.respond_to?(:process_group_id)
|
|
70
|
+
group == pid && group != Process.getpgrp ? -group : pid
|
|
71
|
+
end
|
|
72
|
+
|
|
73
|
+
def signal(name, target)
|
|
74
|
+
Process.kill(name, target)
|
|
75
|
+
rescue Errno::ESRCH
|
|
76
|
+
nil
|
|
59
77
|
end
|
|
60
78
|
|
|
61
79
|
def process_exited?(pid, within:)
|
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require 'woods/mcp/index_reader'
|
|
4
|
+
|
|
5
|
+
module Woods
|
|
6
|
+
module Export
|
|
7
|
+
# Typed reads for export consumers. Published identifiers remain unchanged.
|
|
8
|
+
class TypedReader
|
|
9
|
+
def initialize(reader)
|
|
10
|
+
@reader = reader
|
|
11
|
+
end
|
|
12
|
+
|
|
13
|
+
# Refuse missing or mismatched identities instead of exporting a sibling.
|
|
14
|
+
def find(identifier, type)
|
|
15
|
+
unit = @reader.find_unit(identifier, type: type)
|
|
16
|
+
unless unit.is_a?(Hash) && unit['identifier'] == identifier && unit['type'] == type
|
|
17
|
+
raise ExtractionError, "export unit missing or mismatched: #{type}:#{identifier}"
|
|
18
|
+
end
|
|
19
|
+
|
|
20
|
+
unit
|
|
21
|
+
rescue ArgumentError => e
|
|
22
|
+
raise ExtractionError, "export reader requires typed lookup: #{e.message}"
|
|
23
|
+
rescue IOError, SystemCallError, JSON::ParserError => e
|
|
24
|
+
raise ExtractionError, "export unit unreadable: #{type}:#{identifier}: #{e.message}"
|
|
25
|
+
end
|
|
26
|
+
|
|
27
|
+
# IndexReader validates every published artifact, including mixed buckets.
|
|
28
|
+
# The fallback supports injected readers with the documented list/find API.
|
|
29
|
+
def all(only: nil)
|
|
30
|
+
units = @reader.respond_to?(:each_unit) ? @reader.each_unit : injected_units(only)
|
|
31
|
+
units.select { |unit| !only || only.include?(unit['type']) }
|
|
32
|
+
rescue IOError, SystemCallError, JSON::ParserError => e
|
|
33
|
+
raise ExtractionError, "export index incomplete: #{e.message}"
|
|
34
|
+
end
|
|
35
|
+
|
|
36
|
+
private
|
|
37
|
+
|
|
38
|
+
def injected_units(only)
|
|
39
|
+
MCP::IndexReader::UNIT_TYPES_BY_DIR.flat_map do |dir, types|
|
|
40
|
+
next [] if only && (types & only).empty?
|
|
41
|
+
|
|
42
|
+
@reader.list_units(type: MCP::IndexReader::DIR_TO_TYPE.fetch(dir)).map do |entry|
|
|
43
|
+
find(entry['identifier'], entry_type(entry, types))
|
|
44
|
+
end
|
|
45
|
+
end
|
|
46
|
+
end
|
|
47
|
+
|
|
48
|
+
def entry_type(entry, types)
|
|
49
|
+
type = entry['type'] || (types.first if types.one?)
|
|
50
|
+
return type if types.include?(type)
|
|
51
|
+
|
|
52
|
+
raise ExtractionError, "export entry requires actual type: #{entry['identifier']}"
|
|
53
|
+
end
|
|
54
|
+
end
|
|
55
|
+
end
|
|
56
|
+
end
|