woods 2.0.0.beta2 → 2.0.0.beta4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +339 -1
- data/CONTRIBUTING.md +188 -12
- data/README.md +93 -174
- data/SECURITY.md +9 -6
- data/docs/AGENT_GUIDE.md +109 -8
- data/docs/AGENT_SETUP.md +98 -7
- data/docs/BACKEND_MATRIX.md +25 -0
- data/docs/CLIENT_HOOKS.md +111 -0
- data/docs/CONFIGURATION_REFERENCE.md +267 -16
- data/docs/CONSOLE_MCP_SETUP.md +80 -7
- data/docs/DOCKER_SETUP.md +22 -3
- data/docs/EVALUATION.md +464 -1
- data/docs/EXTRACTOR_REFERENCE.md +45 -6
- data/docs/FAQ.md +11 -12
- data/docs/GETTING_STARTED.md +17 -5
- data/docs/INCREMENTAL_EXTRACTION.md +147 -7
- data/docs/INDEX_LAYOUT.md +382 -0
- data/docs/INTERNALS.md +7 -2
- data/docs/MCP_SERVERS.md +276 -5
- data/docs/MCP_TOOL_COOKBOOK.md +37 -22
- data/docs/MCP_WORKTREE_SETUP.md +43 -83
- data/docs/NOTION_INTEGRATION.md +13 -0
- data/docs/OBSIDIAN_INTEGRATION.md +57 -9
- data/docs/PUBLISHED_INDEX.md +72 -0
- data/docs/README.md +7 -0
- data/docs/RETRIEVAL_GUIDE.md +273 -12
- data/docs/RUNTIME_TRACING.md +71 -0
- data/docs/SOURCE_FRESHNESS.md +143 -0
- data/docs/TROUBLESHOOTING.md +129 -18
- data/docs/UNBLOCKED_INTEGRATION.md +25 -0
- data/docs/UPGRADING_TO_2.md +48 -22
- data/docs/WATCH_DAEMON.md +277 -67
- data/exe/woods-agent-config +6 -0
- data/exe/woods-extract +5 -0
- data/exe/woods-hook-context +6 -0
- data/exe/woods-mcp-start +14 -9
- data/lib/generators/woods/pgvector_generator.rb +8 -2
- data/lib/generators/woods/templates/woods.rb.tt +1 -3
- data/lib/tasks/woods.rake +47 -397
- data/lib/woods/agent_configuration/applier.rb +135 -0
- data/lib/woods/agent_configuration/cli.rb +101 -0
- data/lib/woods/agent_configuration/cli_options.rb +29 -0
- data/lib/woods/agent_configuration/document.rb +105 -0
- data/lib/woods/agent_configuration/error.rb +7 -0
- data/lib/woods/agent_configuration/launcher.rb +75 -0
- data/lib/woods/agent_configuration/layout.rb +72 -0
- data/lib/woods/agent_configuration/managed_section.rb +62 -0
- data/lib/woods/agent_configuration/plan.rb +98 -0
- data/lib/woods/agent_configuration/plan_diff.rb +38 -0
- data/lib/woods/agent_configuration/planned_files.rb +61 -0
- data/lib/woods/agent_configuration/planner.rb +63 -0
- data/lib/woods/agent_configuration/planner_validation.rb +77 -0
- data/lib/woods/agent_configuration/preflight.rb +100 -0
- data/lib/woods/agent_configuration/recovery.rb +49 -0
- data/lib/woods/ast/node.rb +2 -0
- data/lib/woods/ast/parser.rb +38 -5
- data/lib/woods/builder.rb +21 -5
- data/lib/woods/cache/cache_middleware.rb +28 -7
- data/lib/woods/cache/cache_store.rb +4 -5
- data/lib/woods/change_set.rb +5 -4
- data/lib/woods/console/credential_index.rb +20 -2
- data/lib/woods/console/credential_scanner.rb +18 -17
- data/lib/woods/console/credential_scanner_registry.rb +36 -0
- data/lib/woods/console/dispatch_pipeline.rb +7 -0
- data/lib/woods/console/embedded_executor.rb +32 -10
- data/lib/woods/console/encrypted_credential_snapshot.rb +16 -0
- data/lib/woods/console/rack_middleware.rb +22 -13
- data/lib/woods/console/server.rb +18 -16
- data/lib/woods/console/sql_noise_stripper.rb +9 -7
- data/lib/woods/console/sql_table_scanner.rb +47 -7
- data/lib/woods/console/sql_validator.rb +49 -9
- data/lib/woods/console/sqlite_read_guard.rb +46 -0
- data/lib/woods/coordination/pipeline_lock.rb +3 -2
- data/lib/woods/dependency_graph.rb +65 -13
- data/lib/woods/embedding/corpus.rb +94 -0
- data/lib/woods/embedding/indexer.rb +114 -60
- data/lib/woods/embedding/openai.rb +17 -6
- data/lib/woods/evaluation/ablation_executor.rb +6 -1
- data/lib/woods/evaluation/ablation_timed_executor.rb +22 -4
- data/lib/woods/export/typed_reader.rb +56 -0
- data/lib/woods/extractor.rb +277 -149
- data/lib/woods/extractors/action_cable_extractor.rb +3 -1
- data/lib/woods/extractors/behavioral_profile.rb +9 -7
- data/lib/woods/extractors/caching_extractor.rb +3 -1
- data/lib/woods/extractors/concern_extractor.rb +64 -6
- data/lib/woods/extractors/configuration_extractor.rb +7 -3
- data/lib/woods/extractors/controller_extractor.rb +13 -4
- data/lib/woods/extractors/database_view_extractor.rb +3 -1
- data/lib/woods/extractors/declared_parent.rb +55 -0
- data/lib/woods/extractors/decorator_extractor.rb +3 -1
- data/lib/woods/extractors/engine_extractor.rb +3 -1
- data/lib/woods/extractors/event_extractor.rb +4 -2
- data/lib/woods/extractors/factory_extractor.rb +3 -1
- data/lib/woods/extractors/graphql_extractor.rb +10 -13
- data/lib/woods/extractors/i18n_extractor.rb +3 -1
- data/lib/woods/extractors/job_extractor.rb +6 -19
- data/lib/woods/extractors/lib_extractor.rb +13 -9
- data/lib/woods/extractors/mailer_extractor.rb +26 -15
- data/lib/woods/extractors/manager_extractor.rb +3 -1
- data/lib/woods/extractors/method_parameters.rb +53 -0
- data/lib/woods/extractors/middleware_argument.rb +65 -0
- data/lib/woods/extractors/middleware_extractor.rb +9 -3
- data/lib/woods/extractors/migration_extractor.rb +3 -1
- data/lib/woods/extractors/model_extractor.rb +26 -34
- data/lib/woods/extractors/package_extractor.rb +24 -4
- data/lib/woods/extractors/phlex_extractor.rb +3 -1
- data/lib/woods/extractors/policy_extractor.rb +3 -1
- data/lib/woods/extractors/poro_extractor.rb +13 -9
- data/lib/woods/extractors/pundit_extractor.rb +3 -1
- data/lib/woods/extractors/rails_source_extractor.rb +4 -2
- data/lib/woods/extractors/rake_task_extractor.rb +4 -2
- data/lib/woods/extractors/route_extractor.rb +3 -1
- data/lib/woods/extractors/route_helper_resolver.rb +10 -33
- data/lib/woods/extractors/scheduled_job_extractor.rb +41 -15
- data/lib/woods/extractors/serializer_extractor.rb +4 -2
- data/lib/woods/extractors/service_extractor.rb +3 -1
- data/lib/woods/extractors/shared_dependency_scanner.rb +2 -2
- data/lib/woods/extractors/shared_utility_methods.rb +48 -19
- data/lib/woods/extractors/source_nesting.rb +1 -1
- data/lib/woods/extractors/state_machine_extractor.rb +3 -1
- data/lib/woods/extractors/test_mapping_extractor.rb +3 -1
- data/lib/woods/extractors/validator_extractor.rb +3 -1
- data/lib/woods/extractors/view_component_extractor.rb +3 -1
- data/lib/woods/extractors/view_template_extractor.rb +3 -1
- data/lib/woods/gem_mapper.rb +2 -0
- data/lib/woods/git_history.rb +116 -0
- data/lib/woods/graph_analyzer.rb +35 -6
- data/lib/woods/hooks/context_cli.rb +54 -0
- data/lib/woods/hooks/context_event.rb +88 -0
- data/lib/woods/hooks/context_hint.rb +73 -0
- data/lib/woods/hooks/context_impact.rb +77 -0
- data/lib/woods/hooks/context_output.rb +47 -0
- data/lib/woods/hooks/context_state.rb +102 -0
- data/lib/woods/hooks/refresh.rb +79 -0
- data/lib/woods/hooks/rule_projection.rb +78 -0
- data/lib/woods/input_rules.rb +19 -0
- data/lib/woods/mcp/bearer_auth.rb +22 -13
- data/lib/woods/mcp/bootstrapper.rb +79 -4
- data/lib/woods/mcp/config_resolver.rb +2 -1
- data/lib/woods/mcp/index_reader.rb +334 -162
- data/lib/woods/mcp/initialization_guidance.rb +27 -0
- data/lib/woods/mcp/origin_guard.rb +17 -9
- data/lib/woods/mcp/published_lexical_retriever.rb +115 -0
- data/lib/woods/mcp/renderers/markdown_renderer.rb +22 -9
- data/lib/woods/mcp/renderers/plain_renderer.rb +18 -8
- data/lib/woods/mcp/search_results.rb +74 -0
- data/lib/woods/mcp/server.rb +178 -63
- data/lib/woods/mcp/tool_contract.rb +3 -1
- data/lib/woods/mcp/tool_response_renderer.rb +41 -0
- data/lib/woods/mcp/traversal_evidence.rb +113 -0
- data/lib/woods/mcp/traversal_evidence_index.rb +100 -0
- data/lib/woods/mcp/traversal_evidence_page.rb +41 -0
- data/lib/woods/mcp/traversal_evidence_text.rb +52 -0
- data/lib/woods/mcp/traversal_response.rb +22 -0
- data/lib/woods/notion/exporter.rb +56 -17
- data/lib/woods/obsidian/destination_plan.rb +98 -0
- data/lib/woods/obsidian/name_mapper.rb +19 -3
- data/lib/woods/obsidian/note_builder.rb +19 -10
- data/lib/woods/obsidian/vault_exporter.rb +88 -32
- data/lib/woods/operator/pipeline_guard.rb +18 -13
- data/lib/woods/path_dispatcher.rb +13 -6
- data/lib/woods/payload_store.rb +27 -26
- data/lib/woods/published_index/typed_unit_reader.rb +40 -3
- data/lib/woods/published_index.rb +2 -2
- data/lib/woods/railtie.rb +3 -3
- data/lib/woods/railtie_support.rb +12 -12
- data/lib/woods/rake_helpers.rb +382 -0
- data/lib/woods/resilience/graph_invariant_validator/membership_checks.rb +71 -0
- data/lib/woods/resilience/graph_invariant_validator/node_checks.rb +61 -0
- data/lib/woods/resilience/graph_invariant_validator/reverse_relationship_checks.rb +46 -0
- data/lib/woods/resilience/graph_invariant_validator.rb +119 -0
- data/lib/woods/resilience/index_validator/graph_checks.rb +80 -0
- data/lib/woods/resilience/index_validator.rb +112 -23
- data/lib/woods/retrieval/context_assembler.rb +50 -15
- data/lib/woods/retrieval/lexical_assembler.rb +84 -0
- data/lib/woods/retrieval/lexical_index.rb +120 -0
- data/lib/woods/retrieval/ranker.rb +4 -2
- data/lib/woods/retrieval/scope.rb +108 -0
- data/lib/woods/retrieval/scoped_graph_store.rb +32 -0
- data/lib/woods/retrieval/scoped_vector_store.rb +55 -0
- data/lib/woods/retrieval/search_executor.rb +86 -27
- data/lib/woods/retrieval/source_evidence.rb +200 -0
- data/lib/woods/retriever.rb +98 -22
- data/lib/woods/ruby_analyzer/trace_enricher.rb +77 -38
- data/lib/woods/session_tracer/file_store.rb +6 -1
- data/lib/woods/session_tracer/middleware.rb +10 -12
- data/lib/woods/session_tracer/redis_store.rb +22 -6
- data/lib/woods/session_tracer/session_flow_assembler.rb +23 -17
- data/lib/woods/session_tracer/solid_cache_coordination.rb +6 -4
- data/lib/woods/session_tracer/unit_resolver.rb +63 -0
- data/lib/woods/source_inputs/consumer_errors.rb +31 -0
- data/lib/woods/source_inputs/handoff.rb +102 -0
- data/lib/woods/source_inputs/launcher.rb +157 -0
- data/lib/woods/source_inputs/manifest.rb +124 -0
- data/lib/woods/source_inputs/private_key.rb +55 -0
- data/lib/woods/source_inputs/scanner.rb +171 -0
- data/lib/woods/source_inputs/scopes.rb +71 -0
- data/lib/woods/source_inputs/session.rb +214 -0
- data/lib/woods/source_inputs/status.rb +84 -0
- data/lib/woods/source_inputs/verifier.rb +107 -0
- data/lib/woods/storage/metadata_store.rb +25 -25
- data/lib/woods/storage/pgvector.rb +35 -10
- data/lib/woods/storage/qdrant.rb +17 -7
- data/lib/woods/storage/vector_store.rb +18 -6
- data/lib/woods/tasks.rb +3 -2
- data/lib/woods/temporal/json_snapshot_store.rb +58 -9
- data/lib/woods/unblocked/exporter.rb +59 -70
- data/lib/woods/version.rb +1 -1
- data/lib/woods/watch/boot_snapshot.rb +52 -0
- data/lib/woods/watch/daemon.rb +154 -32
- data/lib/woods/watch/listen_watcher.rb +4 -0
- data/lib/woods/watch/polling_watcher.rb +5 -1
- data/lib/woods/watch/status.rb +20 -15
- data/lib/woods/watch/tree_scan.rb +21 -13
- data/lib/woods/watch/watcher.rb +4 -1
- data/lib/woods.rb +50 -11
- data/plugin/.claude-plugin/plugin.json +1 -1
- data/plugin/hooks/adapters/normalize.jq +15 -0
- data/plugin/hooks/adapters/normalize.rb +63 -0
- data/plugin/hooks/hooks.json +20 -0
- data/plugin/hooks/woods-context.sh +50 -0
- data/plugin/hooks/woods-input-rules.sh +159 -0
- data/plugin/hooks/woods-opencode.mjs +65 -0
- data/plugin/hooks/woods-post-edit.sh +2 -225
- data/plugin/hooks/woods-refresh.sh +260 -0
- data/plugin/hooks/woods-session-start.sh +47 -55
- data/plugin/skills/woods-agent-enable/SKILL.md +19 -0
- data/plugin/skills/woods-diagnose/SKILL.md +319 -1
- data/plugin/skills/woods-investigate/SKILL.md +145 -0
- data/plugin/skills/woods-mcp-config/SKILL.md +90 -2
- data/plugin/skills/woods-setup/SKILL.md +110 -6
- metadata +87 -5
|
@@ -0,0 +1,84 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require 'base64'
|
|
4
|
+
require 'woods/generation'
|
|
5
|
+
require 'woods/source_inputs/verifier'
|
|
6
|
+
|
|
7
|
+
module Woods
|
|
8
|
+
module SourceInputs
|
|
9
|
+
# Shared CLI/hook/MCP reader. A caller serving an older immutable payload
|
|
10
|
+
# passes its own directory and generation, never the newest marker's proof.
|
|
11
|
+
class Status
|
|
12
|
+
BUDGETS = { 'quick' => 0.25, 'deep' => 5.0 }.freeze
|
|
13
|
+
|
|
14
|
+
def self.from_transport(encoded)
|
|
15
|
+
raise ArgumentError, 'source status options are too large' if encoded.to_s.bytesize > 16_384
|
|
16
|
+
|
|
17
|
+
options = encoded.nil? ? {} : JSON.parse(Base64.strict_decode64(encoded))
|
|
18
|
+
unless options.is_a?(Hash) && (options.keys - %w[output root mode]).empty? && options.values.all?(String)
|
|
19
|
+
raise ArgumentError, 'source status options must contain only output, root and mode strings'
|
|
20
|
+
end
|
|
21
|
+
|
|
22
|
+
new(output_dir: options.fetch('output', ENV.fetch('WOODS_OUTPUT', 'tmp/woods')),
|
|
23
|
+
root: options['root'], mode: options.fetch('mode', 'quick')).call
|
|
24
|
+
rescue JSON::ParserError
|
|
25
|
+
raise ArgumentError, 'invalid source status options'
|
|
26
|
+
end
|
|
27
|
+
|
|
28
|
+
def initialize(output_dir:, payload_dir: nil, generation: nil, root: nil, mode: 'quick')
|
|
29
|
+
raise ArgumentError, 'source status mode must be quick or deep' unless BUDGETS.key?(mode)
|
|
30
|
+
|
|
31
|
+
@output = File.expand_path(output_dir.to_s)
|
|
32
|
+
@root = root
|
|
33
|
+
@mode = mode
|
|
34
|
+
@generation = generation
|
|
35
|
+
@payload = payload_dir
|
|
36
|
+
end
|
|
37
|
+
|
|
38
|
+
def call
|
|
39
|
+
return unknown(@resolution_error || 'atomic_source_manifest_unavailable') unless resolve_payload
|
|
40
|
+
|
|
41
|
+
flags = File::RDONLY | File::NONBLOCK
|
|
42
|
+
flags |= File::NOFOLLOW if defined?(File::NOFOLLOW)
|
|
43
|
+
manifest = File.open(File.join(@payload, Manifest::FILE_NAME), flags) do |file|
|
|
44
|
+
return unknown('invalid_source_manifest') unless file.stat.file?
|
|
45
|
+
return unknown('source_manifest_too_large') if file.stat.size > Manifest::MAX_BYTES
|
|
46
|
+
|
|
47
|
+
Manifest.parse(file.read(Manifest::MAX_BYTES + 1))
|
|
48
|
+
end
|
|
49
|
+
Verifier.new(manifest: manifest, output_dir: @output, root: @root,
|
|
50
|
+
generation: @generation, max_seconds: BUDGETS.fetch(@mode)).call.merge('check' => @mode)
|
|
51
|
+
rescue Errno::ENOENT
|
|
52
|
+
unknown('source_manifest_unavailable')
|
|
53
|
+
rescue Manifest::Invalid, SystemCallError, IOError
|
|
54
|
+
unknown('invalid_source_manifest')
|
|
55
|
+
end
|
|
56
|
+
|
|
57
|
+
private
|
|
58
|
+
|
|
59
|
+
def resolve_payload
|
|
60
|
+
generation = Generation.new(output_dir: @output)
|
|
61
|
+
marker = if @payload
|
|
62
|
+
Generation::Marker.new(payload: @payload.to_s)
|
|
63
|
+
else
|
|
64
|
+
generation.current.tap { |current| @generation = current.number }
|
|
65
|
+
end
|
|
66
|
+
return false unless marker.payload.is_a?(String)
|
|
67
|
+
|
|
68
|
+
resolved = generation.payload_dir(marker)
|
|
69
|
+
return false if resolved == generation.root
|
|
70
|
+
|
|
71
|
+
@payload = resolved
|
|
72
|
+
true
|
|
73
|
+
rescue TypeError, NoMethodError
|
|
74
|
+
@resolution_error = 'invalid_generation'
|
|
75
|
+
false
|
|
76
|
+
end
|
|
77
|
+
|
|
78
|
+
def unknown(reason)
|
|
79
|
+
{ 'state' => 'unknown', 'mode' => 'content', 'check' => @mode, 'generation' => @generation,
|
|
80
|
+
'checked_at' => Time.now.utc.iso8601, 'complete' => false, 'reasons' => [reason] }
|
|
81
|
+
end
|
|
82
|
+
end
|
|
83
|
+
end
|
|
84
|
+
end
|
|
@@ -0,0 +1,107 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require 'time'
|
|
4
|
+
require 'woods/source_inputs/scanner'
|
|
5
|
+
require 'woods/source_inputs/manifest'
|
|
6
|
+
|
|
7
|
+
module Woods
|
|
8
|
+
module SourceInputs
|
|
9
|
+
# Bounded, byte-verifying source evidence. No stat-only result is called
|
|
10
|
+
# current: the same-size/same-mtime case is deliberately read and hashed.
|
|
11
|
+
class Verifier
|
|
12
|
+
DEFAULT_SECONDS = 0.25
|
|
13
|
+
SUMMARY_LIMIT = 30
|
|
14
|
+
|
|
15
|
+
# rubocop:disable Metrics/ParameterLists -- caller chooses identity, served generation and bounded scan
|
|
16
|
+
def initialize(manifest:, output_dir:, root: nil, generation: nil, max_seconds: DEFAULT_SECONDS, **limits)
|
|
17
|
+
@manifest = manifest.is_a?(Manifest) ? manifest : Manifest.new(manifest)
|
|
18
|
+
@output_dir = output_dir
|
|
19
|
+
@root = root || @manifest.data.fetch('root')
|
|
20
|
+
@generation = generation
|
|
21
|
+
@limits = limits.merge(max_seconds: max_seconds)
|
|
22
|
+
end
|
|
23
|
+
|
|
24
|
+
# rubocop:enable Metrics/ParameterLists
|
|
25
|
+
|
|
26
|
+
def call # rubocop:disable Metrics/AbcSize -- cheap refusal checks precede the only source scan
|
|
27
|
+
return unknown('generation_mismatch') if @generation && @generation != @manifest.data['generation']
|
|
28
|
+
return unknown('source_root_unavailable') unless File.directory?(@root)
|
|
29
|
+
|
|
30
|
+
key = PrivateKey.new(output_dir: @output_dir)
|
|
31
|
+
return unknown('identity_key_mismatch') unless key.identifier == @manifest.data.fetch('key_id')
|
|
32
|
+
|
|
33
|
+
scopes = Scopes.new(extra_roots: @manifest.data.fetch('extra_roots'))
|
|
34
|
+
return unknown('input_rules_changed') unless scopes.fingerprint == @manifest.data.fetch('rules')
|
|
35
|
+
|
|
36
|
+
current = Scanner.new(root: @root, output_dir: @output_dir, key: key, scopes: scopes, **@limits).call
|
|
37
|
+
compare(current)
|
|
38
|
+
rescue PrivateKey::Unavailable => e
|
|
39
|
+
unknown(e.message)
|
|
40
|
+
rescue SystemCallError, IOError
|
|
41
|
+
unknown('source_root_unavailable')
|
|
42
|
+
end
|
|
43
|
+
|
|
44
|
+
private
|
|
45
|
+
|
|
46
|
+
def compare(current)
|
|
47
|
+
previous = @manifest.expanded
|
|
48
|
+
changes = { 'added' => [], 'changed' => [], 'removed' => [] }
|
|
49
|
+
previous.each { |scope, paths| compare_scope(scope, paths, changes, current) }
|
|
50
|
+
add_new_scopes(previous, changes, current) if complete?(current)
|
|
51
|
+
reasons = coverage_reasons(current)
|
|
52
|
+
summarized(changes, reasons, current)
|
|
53
|
+
end
|
|
54
|
+
|
|
55
|
+
def complete?(current)
|
|
56
|
+
@manifest.data['complete'] && current['complete']
|
|
57
|
+
end
|
|
58
|
+
|
|
59
|
+
def compare_scope(scope, paths, changes, current)
|
|
60
|
+
paths.each do |path, identity|
|
|
61
|
+
now = current.fetch('files')[path]
|
|
62
|
+
changes['changed'] << path if now && now != identity
|
|
63
|
+
changes['removed'] << path if now.nil? && current['complete']
|
|
64
|
+
end
|
|
65
|
+
return unless complete?(current)
|
|
66
|
+
|
|
67
|
+
changes['added'].concat(current.fetch('scope_paths').fetch(scope, []) - paths.keys)
|
|
68
|
+
end
|
|
69
|
+
|
|
70
|
+
def add_new_scopes(previous, changes, current)
|
|
71
|
+
new_scopes = current.fetch('scope_paths').keys - previous.keys
|
|
72
|
+
new_scopes.each { |scope| changes['added'].concat(current.fetch('scope_paths').fetch(scope)) }
|
|
73
|
+
end
|
|
74
|
+
|
|
75
|
+
def coverage_reasons(current)
|
|
76
|
+
reasons = (@manifest.data.fetch('errors') + current.fetch('errors')).map { |error| error['reason'] }.compact
|
|
77
|
+
reasons << 'incomplete_capture' unless @manifest.data['complete']
|
|
78
|
+
reasons << 'incomplete_verification' unless current['complete']
|
|
79
|
+
reasons << 'unverified_boot_boundary' unless @manifest.data['boot_verified']
|
|
80
|
+
reasons << 'unverified_consumption_scopes' unless @manifest.data.fetch('unverified_scopes').empty?
|
|
81
|
+
reasons.uniq
|
|
82
|
+
end
|
|
83
|
+
|
|
84
|
+
def summarized(changes, reasons, current)
|
|
85
|
+
changes.transform_values! { |paths| paths.uniq.sort }
|
|
86
|
+
counts = changes.transform_values(&:size)
|
|
87
|
+
{ 'state' => state_for(counts, reasons), 'mode' => 'content', 'generation' => @manifest.data['generation'],
|
|
88
|
+
'checked_at' => Time.now.utc.iso8601, 'reasons' => reasons,
|
|
89
|
+
'complete' => current['complete'] && @manifest.data['complete'],
|
|
90
|
+
'counts' => counts, 'truncated' => counts.values.any? { |count| count > SUMMARY_LIMIT },
|
|
91
|
+
'changes' => changes.transform_values { |paths| paths.first(SUMMARY_LIMIT) },
|
|
92
|
+
'metrics' => current.fetch('metrics') }
|
|
93
|
+
end
|
|
94
|
+
|
|
95
|
+
def state_for(counts, reasons)
|
|
96
|
+
return 'drifted' if counts.values.any?(&:positive?)
|
|
97
|
+
|
|
98
|
+
reasons.empty? ? 'current' : 'unknown'
|
|
99
|
+
end
|
|
100
|
+
|
|
101
|
+
def unknown(reason)
|
|
102
|
+
{ 'state' => 'unknown', 'mode' => 'content', 'generation' => @manifest.data['generation'],
|
|
103
|
+
'checked_at' => Time.now.utc.iso8601, 'reasons' => [reason], 'complete' => false }
|
|
104
|
+
end
|
|
105
|
+
end
|
|
106
|
+
end
|
|
107
|
+
end
|
|
@@ -84,16 +84,19 @@ module Woods
|
|
|
84
84
|
# adapter — they were substitutable in name only until the semantics
|
|
85
85
|
# were written down):
|
|
86
86
|
#
|
|
87
|
-
# - **
|
|
87
|
+
# - **Literal substring match**, including embedded NUL, not word or prefix match.
|
|
88
88
|
# - **Case-insensitive.** InMemory used a case-sensitive
|
|
89
89
|
# `String#include?` while SQLite used `LIKE`, so the same query
|
|
90
90
|
# returned different results depending on the configured backend.
|
|
91
91
|
# Case-insensitive is both the search-like expectation and what the
|
|
92
|
-
# durable adapter already did. (SQLite's `
|
|
92
|
+
# durable adapter already did. (SQLite's `lower` folds ASCII only, so
|
|
93
93
|
# non-ASCII case folding remains backend-specific — do not rely on
|
|
94
94
|
# it either way.)
|
|
95
95
|
# - **`fields: nil` searches the whole record**, including keys, as
|
|
96
96
|
# serialized JSON. A query can therefore match a field *name*.
|
|
97
|
+
# - **Field-scoped values use JSON spellings.** Strings are unquoted,
|
|
98
|
+
# objects/arrays use JSON text, Booleans use `true`/`false`, and
|
|
99
|
+
# numbers remain numeric text. Null and absent fields never match.
|
|
97
100
|
# - **LIKE metacharacters are literal.** `%` and `_` in a query match
|
|
98
101
|
# themselves rather than acting as wildcards.
|
|
99
102
|
# - Field names are validated against {SEARCH_FIELD_NAME} by every
|
|
@@ -218,15 +221,14 @@ module Woods
|
|
|
218
221
|
# @see Interface#search
|
|
219
222
|
#
|
|
220
223
|
# Matching is literal substring inclusion — `%` and `_` in the query
|
|
221
|
-
# have no special meaning here
|
|
222
|
-
# so the two adapters agree.
|
|
224
|
+
# have no special meaning here or in the SQLite adapter.
|
|
223
225
|
#
|
|
224
226
|
# @raise [ArgumentError] if a field name fails {SEARCH_FIELD_NAME}
|
|
225
227
|
def search(query, fields: nil)
|
|
226
228
|
fields = validate_search_fields!(fields)
|
|
227
229
|
return [] if fields == []
|
|
228
230
|
|
|
229
|
-
# Case-insensitive, matching the SQLite adapter's `
|
|
231
|
+
# Case-insensitive, matching the SQLite adapter's `lower` (see the
|
|
230
232
|
# contract note on {Interface#search}). This used to be a
|
|
231
233
|
# case-sensitive `String#include?`, so the same query returned
|
|
232
234
|
# different results depending on which backend a host had configured.
|
|
@@ -312,9 +314,9 @@ module Woods
|
|
|
312
314
|
JSON.parse(JSON.generate(metadata))
|
|
313
315
|
end
|
|
314
316
|
|
|
315
|
-
# The text
|
|
316
|
-
#
|
|
317
|
-
#
|
|
317
|
+
# The searchable text for one field value: strings come back raw,
|
|
318
|
+
# structured values as JSON text, Booleans as true/false, and numbers
|
|
319
|
+
# as their decimal form. A Ruby
|
|
318
320
|
# +Hash#to_s+ haystack used to leak `=>` and `:sym` syntax that no
|
|
319
321
|
# JSON document contains (STO-8).
|
|
320
322
|
#
|
|
@@ -425,23 +427,22 @@ module Woods
|
|
|
425
427
|
# Field names are interpolated into a `json_extract` JSON-path
|
|
426
428
|
# literal, so they are validated against {SEARCH_FIELD_NAME} first —
|
|
427
429
|
# a crafted name could otherwise break out of the literal and alter
|
|
428
|
-
# the SQL shape.
|
|
429
|
-
#
|
|
430
|
-
#
|
|
431
|
-
# silently broadening matches.
|
|
430
|
+
# the SQL shape. instr and lower perform literal, ASCII-case-insensitive
|
|
431
|
+
# substring matching without LIKE's truncation at embedded NUL bytes.
|
|
432
|
+
# Wildcard characters need no escaping.
|
|
432
433
|
#
|
|
433
434
|
# @raise [ArgumentError] if a field name fails the whitelist
|
|
434
435
|
def search(query, fields: nil)
|
|
435
436
|
fields = validate_search_fields!(fields)
|
|
436
437
|
return [] if fields == []
|
|
437
438
|
|
|
438
|
-
|
|
439
|
+
needle = query.to_s
|
|
439
440
|
if fields
|
|
440
|
-
conditions = fields.map { "
|
|
441
|
-
params = Array.new(fields.size,
|
|
441
|
+
conditions = fields.map { "instr(lower(#{field_haystack_sql(_1)}), lower(?)) > 0" }.join(' OR ')
|
|
442
|
+
params = Array.new(fields.size, needle)
|
|
442
443
|
rows = @db.execute("SELECT id, data FROM units WHERE #{conditions}", params)
|
|
443
444
|
else
|
|
444
|
-
rows = @db.execute(
|
|
445
|
+
rows = @db.execute('SELECT id, data FROM units WHERE instr(lower(data), lower(?)) > 0', [needle])
|
|
445
446
|
end
|
|
446
447
|
|
|
447
448
|
rows.map { |row| parse_row(row) }
|
|
@@ -489,15 +490,14 @@ module Woods
|
|
|
489
490
|
defined?(SQLite3::BusyException) && error.is_a?(SQLite3::BusyException)
|
|
490
491
|
end
|
|
491
492
|
|
|
492
|
-
#
|
|
493
|
-
#
|
|
494
|
-
#
|
|
495
|
-
|
|
496
|
-
|
|
497
|
-
|
|
498
|
-
|
|
499
|
-
|
|
500
|
-
query.to_s.gsub(/[\\%_]/) { |ch| "\\#{ch}" }
|
|
493
|
+
# JSON1 extracts Boolean scalars as integers; use their JSON type to
|
|
494
|
+
# preserve true/false without conflating them with numeric 1/0.
|
|
495
|
+
# Field names have already passed validate_search_fields!.
|
|
496
|
+
def field_haystack_sql(field)
|
|
497
|
+
path = "$.#{field}"
|
|
498
|
+
"CASE json_type(data, '#{path}') " \
|
|
499
|
+
"WHEN 'true' THEN 'true' WHEN 'false' THEN 'false' " \
|
|
500
|
+
"ELSE json_extract(data, '#{path}') END"
|
|
501
501
|
end
|
|
502
502
|
|
|
503
503
|
# Parse a database row into a metadata hash with the id field injected.
|
|
@@ -26,6 +26,7 @@ module Woods
|
|
|
26
26
|
class Pgvector # rubocop:disable Metrics/ClassLength
|
|
27
27
|
include Interface
|
|
28
28
|
|
|
29
|
+
MAX_HNSW_DIMENSIONS = 2000
|
|
29
30
|
TABLE = 'woods_vectors'
|
|
30
31
|
TABLE_NAME_PATTERN = /\A[a-z_][a-z0-9_]*\z/
|
|
31
32
|
|
|
@@ -122,6 +123,9 @@ module Woods
|
|
|
122
123
|
SQL
|
|
123
124
|
end
|
|
124
125
|
|
|
126
|
+
# Native raw-ID eligibility is applied before ranking and the limit.
|
|
127
|
+
def supports_id_filter? = true
|
|
128
|
+
|
|
125
129
|
# Search for similar vectors using cosine distance.
|
|
126
130
|
#
|
|
127
131
|
# The query vector is dimension-checked before SQL runs, mirroring the
|
|
@@ -136,19 +140,14 @@ module Woods
|
|
|
136
140
|
# @raise [Woods::Error] if the query vector's length disagrees with the
|
|
137
141
|
# configured dimension
|
|
138
142
|
# @see Interface#search
|
|
139
|
-
def search(query_vector, limit: 10, filters: {})
|
|
143
|
+
def search(query_vector, limit: 10, filters: {}, ids: nil)
|
|
140
144
|
validate_vector!(query_vector)
|
|
141
145
|
validate_dimensions!(query_vector) if @dimensions
|
|
142
146
|
vector_literal = build_vector_literal(query_vector)
|
|
143
147
|
where_clause = build_where(filters)
|
|
148
|
+
where_clause = append_id_filter(where_clause, ids) if ids
|
|
144
149
|
|
|
145
|
-
sql =
|
|
146
|
-
SELECT id, embedding <=> '#{vector_literal}' AS distance, metadata
|
|
147
|
-
FROM #{qualified_table}
|
|
148
|
-
#{where_clause}
|
|
149
|
-
ORDER BY distance ASC
|
|
150
|
-
LIMIT #{limit.to_i}
|
|
151
|
-
SQL
|
|
150
|
+
sql = search_sql(vector_literal, where_clause, limit, ids)
|
|
152
151
|
|
|
153
152
|
rows = @connection.execute(sql)
|
|
154
153
|
rows.map { |row| row_to_result(row) }
|
|
@@ -221,9 +220,12 @@ module Woods
|
|
|
221
220
|
private
|
|
222
221
|
|
|
223
222
|
def normalize_dimensions(value)
|
|
224
|
-
|
|
223
|
+
raise ArgumentError, 'dimensions must be a positive Integer' unless value.is_a?(Integer) && value.positive?
|
|
224
|
+
return value if value <= MAX_HNSW_DIMENSIONS
|
|
225
225
|
|
|
226
|
-
raise ArgumentError,
|
|
226
|
+
raise ArgumentError,
|
|
227
|
+
"pgvector HNSW vector dimensions must be at most #{MAX_HNSW_DIMENSIONS}, got #{value}. " \
|
|
228
|
+
'Request a supported provider output width or choose another vector backend.'
|
|
227
229
|
end
|
|
228
230
|
|
|
229
231
|
def validate_identifier!(name, value, original)
|
|
@@ -345,6 +347,29 @@ module Woods
|
|
|
345
347
|
)
|
|
346
348
|
end
|
|
347
349
|
|
|
350
|
+
def search_sql(vector_literal, where_clause, limit, ids)
|
|
351
|
+
source = ids ? 'eligible' : qualified_table
|
|
352
|
+
prefix = if ids
|
|
353
|
+
'WITH eligible AS MATERIALIZED (SELECT id, embedding, metadata ' \
|
|
354
|
+
"FROM #{qualified_table} #{where_clause})\n"
|
|
355
|
+
else
|
|
356
|
+
''
|
|
357
|
+
end
|
|
358
|
+
prefix + <<~SQL
|
|
359
|
+
SELECT id, embedding <=> '#{vector_literal}' AS distance, metadata
|
|
360
|
+
FROM #{source}
|
|
361
|
+
#{where_clause unless ids}
|
|
362
|
+
ORDER BY distance ASC#{', id ASC' if ids}
|
|
363
|
+
LIMIT #{limit.to_i}
|
|
364
|
+
SQL
|
|
365
|
+
end
|
|
366
|
+
|
|
367
|
+
# Raw IDs work with old vectors that have no identifier/package payload.
|
|
368
|
+
def append_id_filter(where_clause, ids)
|
|
369
|
+
condition = ids.empty? ? 'FALSE' : "id IN (#{ids.map { |id| @connection.quote(id.to_s) }.join(', ')})"
|
|
370
|
+
where_clause.empty? ? "WHERE #{condition}" : "#{where_clause} AND #{condition}"
|
|
371
|
+
end
|
|
372
|
+
|
|
348
373
|
# Build a WHERE clause from metadata filters.
|
|
349
374
|
#
|
|
350
375
|
# @param filters [Hash] Metadata key-value pairs
|
data/lib/woods/storage/qdrant.rb
CHANGED
|
@@ -332,6 +332,9 @@ module Woods
|
|
|
332
332
|
request(:put, "/collections/#{@collection}/points#{WAIT_FOR_WRITE}", body)
|
|
333
333
|
end
|
|
334
334
|
|
|
335
|
+
# Native raw-ID eligibility is applied before ranking and the limit.
|
|
336
|
+
def supports_id_filter? = true
|
|
337
|
+
|
|
335
338
|
# Search for similar vectors.
|
|
336
339
|
#
|
|
337
340
|
# The query vector is dimension-checked before the request, mirroring
|
|
@@ -348,14 +351,11 @@ module Woods
|
|
|
348
351
|
# @raise [Woods::Error] if the query vector's length disagrees with the
|
|
349
352
|
# configured dimension
|
|
350
353
|
# @see Interface#search
|
|
351
|
-
def search(query_vector, limit: 10, filters: {})
|
|
354
|
+
def search(query_vector, limit: 10, filters: {}, ids: nil)
|
|
355
|
+
return [] if ids == []
|
|
356
|
+
|
|
352
357
|
validate_dimensions!(query_vector) if @dimensions
|
|
353
|
-
body =
|
|
354
|
-
vector: query_vector,
|
|
355
|
-
limit: limit,
|
|
356
|
-
with_payload: true
|
|
357
|
-
}
|
|
358
|
-
body[:filter] = build_filter(filters) unless filters.empty?
|
|
358
|
+
body = search_body(query_vector, limit, filters, ids)
|
|
359
359
|
|
|
360
360
|
response = request(:post, "/collections/#{@collection}/points/search", body)
|
|
361
361
|
results = response['result'] || []
|
|
@@ -579,6 +579,16 @@ module Woods
|
|
|
579
579
|
"Vector dimension mismatch#{where}: got #{got}, expected #{@dimensions}"
|
|
580
580
|
end
|
|
581
581
|
|
|
582
|
+
def search_body(query_vector, limit, filters, ids)
|
|
583
|
+
body = { vector: query_vector, limit: limit, with_payload: true }
|
|
584
|
+
body[:filter] = build_filter(filters) unless filters.empty? && ids.nil?
|
|
585
|
+
if ids
|
|
586
|
+
body[:filter][:must] << { key: IDENTIFIER_KEY, match: { any: ids } }
|
|
587
|
+
body[:params] = { exact: true }
|
|
588
|
+
end
|
|
589
|
+
body
|
|
590
|
+
end
|
|
591
|
+
|
|
582
592
|
# Build a Qdrant filter from metadata key-value pairs.
|
|
583
593
|
#
|
|
584
594
|
# @param filters [Hash] Metadata filters
|
|
@@ -90,12 +90,17 @@ module Woods
|
|
|
90
90
|
# @param limit [Integer] Maximum number of results to return
|
|
91
91
|
# @param filters [Hash] Optional metadata filters — values may be
|
|
92
92
|
# scalars or Arrays
|
|
93
|
+
# @param ids [Array<String>, nil] Raw vector IDs eligible before ranking;
|
|
94
|
+
# optional capability advertised by #supports_id_filter?. Empty matches none.
|
|
93
95
|
# @return [Array<SearchResult>] Results sorted by descending similarity
|
|
94
96
|
# @raise [NotImplementedError] if not implemented by adapter
|
|
95
|
-
def search(query_vector, limit: 10, filters: {})
|
|
97
|
+
def search(query_vector, limit: 10, filters: {}, ids: nil)
|
|
96
98
|
raise NotImplementedError
|
|
97
99
|
end
|
|
98
100
|
|
|
101
|
+
# Whether native raw-ID eligibility is supported before the result limit.
|
|
102
|
+
def supports_id_filter? = false
|
|
103
|
+
|
|
99
104
|
# Delete a vector by ID.
|
|
100
105
|
#
|
|
101
106
|
# @param id [String] The identifier to delete
|
|
@@ -220,8 +225,10 @@ module Woods
|
|
|
220
225
|
end
|
|
221
226
|
end
|
|
222
227
|
|
|
228
|
+
def supports_id_filter? = true
|
|
229
|
+
|
|
223
230
|
# @see Interface#search
|
|
224
|
-
def search(query_vector, limit: 10, filters: {})
|
|
231
|
+
def search(query_vector, limit: 10, filters: {}, ids: nil)
|
|
225
232
|
return [] if @dim.nil?
|
|
226
233
|
|
|
227
234
|
unless query_vector.length == @dim
|
|
@@ -229,8 +236,8 @@ module Woods
|
|
|
229
236
|
"Vector dimension mismatch (#{query_vector.length} vs #{@dim})"
|
|
230
237
|
end
|
|
231
238
|
|
|
232
|
-
scored = gather_candidates(query_vector, filters)
|
|
233
|
-
scored.sort_by! { |r| -r.score }
|
|
239
|
+
scored = gather_candidates(query_vector, filters, ids)
|
|
240
|
+
scored.sort_by! { |r| ids ? [-r.score, r.id] : [-r.score] }
|
|
234
241
|
scored.first(limit)
|
|
235
242
|
end
|
|
236
243
|
|
|
@@ -300,15 +307,20 @@ module Woods
|
|
|
300
307
|
@metadata[idx] = metadata
|
|
301
308
|
end
|
|
302
309
|
|
|
310
|
+
def excluded_index?(idx, allowed_ids)
|
|
311
|
+
@tombstones.include?(idx) || (allowed_ids && !allowed_ids.include?(@ids[idx]))
|
|
312
|
+
end
|
|
313
|
+
|
|
303
314
|
# Walk every non-tombstoned index, apply filters, score survivors.
|
|
304
315
|
# Filter check runs BEFORE the cosine kernel — avoids computing
|
|
305
316
|
# 12k dot products only to discard most of them.
|
|
306
|
-
def gather_candidates(query_vector, filters)
|
|
317
|
+
def gather_candidates(query_vector, filters, ids)
|
|
307
318
|
scored = []
|
|
319
|
+
allowed_ids = ids&.to_set
|
|
308
320
|
len = @ids.size
|
|
309
321
|
idx = 0
|
|
310
322
|
while idx < len
|
|
311
|
-
if
|
|
323
|
+
if excluded_index?(idx, allowed_ids)
|
|
312
324
|
idx += 1
|
|
313
325
|
next
|
|
314
326
|
end
|
data/lib/woods/tasks.rb
CHANGED
|
@@ -27,6 +27,7 @@ module Woods
|
|
|
27
27
|
# @return [Embedding::Indexer]
|
|
28
28
|
def build_embed_indexer
|
|
29
29
|
config = Woods.configuration
|
|
30
|
+
output_dir = ENV.fetch('WOODS_OUTPUT', config.output_dir)
|
|
30
31
|
builder = Builder.new(config)
|
|
31
32
|
provider = builder.build_embedding_provider
|
|
32
33
|
|
|
@@ -56,11 +57,11 @@ module Woods
|
|
|
56
57
|
provider: resilient_provider,
|
|
57
58
|
text_preparer: builder.build_text_preparer(provider),
|
|
58
59
|
vector_store: vector_store,
|
|
59
|
-
metadata_store: config.metadata_store ? builder.build_metadata_store : nil,
|
|
60
|
+
metadata_store: config.metadata_store ? builder.build_metadata_store(output_dir: output_dir) : nil,
|
|
60
61
|
resolved_config: build_resolved_config(config, provider: provider),
|
|
61
62
|
chunker: builder.build_chunker(provider),
|
|
62
63
|
dump_retention_count: config.dump_retention_count,
|
|
63
|
-
output_dir:
|
|
64
|
+
output_dir: output_dir
|
|
64
65
|
)
|
|
65
66
|
end
|
|
66
67
|
|
|
@@ -22,7 +22,8 @@ module Woods
|
|
|
22
22
|
#
|
|
23
23
|
# Implements the same public interface as SnapshotStore so the MCP server
|
|
24
24
|
# tools work identically.
|
|
25
|
-
# Malformed
|
|
25
|
+
# Malformed, invalidly shaped, or unreadable retained snapshots (including files
|
|
26
|
+
# removed during retention) are warned about and treated as absent:
|
|
26
27
|
# +find+ returns nil, +diff+ returns an empty result, and history/list scans
|
|
27
28
|
# omit the corrupt file.
|
|
28
29
|
#
|
|
@@ -94,7 +95,6 @@ module Woods
|
|
|
94
95
|
snapshots = load_all_with_units
|
|
95
96
|
.sort_by { |s| s[:extracted_at] || '' }
|
|
96
97
|
.reverse
|
|
97
|
-
.first(limit)
|
|
98
98
|
|
|
99
99
|
entries = snapshots.flat_map do |snap|
|
|
100
100
|
snap[:units].values.select { |unit| unit[:identifier] == identifier }.map do |unit|
|
|
@@ -122,9 +122,9 @@ module Woods
|
|
|
122
122
|
value.positive? ? value : PayloadStore::DEFAULT_RETENTION
|
|
123
123
|
end
|
|
124
124
|
|
|
125
|
-
# Delete snapshots beyond the retention count,
|
|
126
|
-
# first. +protect+ names the snapshot just
|
|
127
|
-
#
|
|
125
|
+
# Delete snapshots beyond the retention count, corrupt/unreadable files
|
|
126
|
+
# first, then oldest by extracted_at. +protect+ names the snapshot just
|
|
127
|
+
# captured; even an older timestamp must not cause its deletion.
|
|
128
128
|
#
|
|
129
129
|
# Failures are non-fatal: retention is housekeeping, and a failed
|
|
130
130
|
# prune leaves the store growing as it did before rather than
|
|
@@ -133,7 +133,7 @@ module Woods
|
|
|
133
133
|
# @param protect [String] git SHA of the just-captured snapshot
|
|
134
134
|
# @return [void]
|
|
135
135
|
def prune_snapshots(protect:)
|
|
136
|
-
summaries =
|
|
136
|
+
summaries = retention_summaries
|
|
137
137
|
overflow = summaries.size - @retention
|
|
138
138
|
return unless overflow.positive?
|
|
139
139
|
|
|
@@ -142,7 +142,19 @@ module Woods
|
|
|
142
142
|
warn "[Woods] Snapshot retention failed: #{e.message}"
|
|
143
143
|
end
|
|
144
144
|
|
|
145
|
-
#
|
|
145
|
+
# Include unreadable snapshots in the bound. Only files named like a
|
|
146
|
+
# snapshot are eligible; the filename owns identity, not JSON content.
|
|
147
|
+
def retention_summaries
|
|
148
|
+
Dir.glob(File.join(@dir, '*.json')).filter_map do |path|
|
|
149
|
+
sha = File.basename(path, '.json')
|
|
150
|
+
next unless sha.match?(/\A[0-9a-f]+\z/i) && File.file?(path)
|
|
151
|
+
|
|
152
|
+
data = read_snapshot(path)
|
|
153
|
+
{ git_sha: sha, extracted_at: data&.[]('extracted_at') || '', corrupt: data.nil? }
|
|
154
|
+
end
|
|
155
|
+
end
|
|
156
|
+
|
|
157
|
+
# Corrupt files, then oldest `overflow` snapshots, never the protected one.
|
|
146
158
|
# The protected SHA is rejected before sorting and slicing: it is
|
|
147
159
|
# captured last, so its extracted_at can tie or precede older entries,
|
|
148
160
|
# and letting it occupy the victim slice would leave the store one file
|
|
@@ -154,7 +166,7 @@ module Woods
|
|
|
154
166
|
# @return [Array<String>]
|
|
155
167
|
def retention_victims(summaries, overflow, protect)
|
|
156
168
|
summaries.reject { |summary| summary[:git_sha] == protect }
|
|
157
|
-
.sort_by { |summary| summary[:extracted_at]
|
|
169
|
+
.sort_by { |summary| [summary[:corrupt] ? 0 : 1, summary[:extracted_at], summary[:git_sha]] }
|
|
158
170
|
.first(overflow)
|
|
159
171
|
.map { |summary| summary[:git_sha] }
|
|
160
172
|
end
|
|
@@ -264,10 +276,47 @@ module Woods
|
|
|
264
276
|
end
|
|
265
277
|
|
|
266
278
|
def read_snapshot(path)
|
|
267
|
-
JSON.parse(AtomicFile.read(path))
|
|
279
|
+
data = JSON.parse(AtomicFile.read(path))
|
|
280
|
+
validate_snapshot_shape!(data)
|
|
281
|
+
unless data['git_sha'] == File.basename(path, '.json')
|
|
282
|
+
raise JSON::ParserError, 'git_sha does not match the snapshot filename'
|
|
283
|
+
end
|
|
284
|
+
|
|
285
|
+
data
|
|
268
286
|
rescue JSON::ParserError => e
|
|
269
287
|
warn "[Woods] Skipping corrupt snapshot #{File.basename(path)}: #{e.message}"
|
|
270
288
|
nil
|
|
289
|
+
rescue SystemCallError => e
|
|
290
|
+
warn "[Woods] Skipping unreadable snapshot #{File.basename(path)}: #{e.message}"
|
|
291
|
+
nil
|
|
292
|
+
end
|
|
293
|
+
|
|
294
|
+
# Guard only shapes consumed by conversion, sorting, and the next capture.
|
|
295
|
+
# Legacy files can omit units/timestamps and optional per-unit hashes;
|
|
296
|
+
# bare unit keys still supply identifiers when the record does not.
|
|
297
|
+
def validate_snapshot_shape!(data)
|
|
298
|
+
raise JSON::ParserError, 'expected a JSON object' unless data.is_a?(Hash)
|
|
299
|
+
|
|
300
|
+
sha = data['git_sha']
|
|
301
|
+
unless sha.is_a?(String) && sha.match?(/\A[0-9a-f]+\z/i)
|
|
302
|
+
raise JSON::ParserError, 'expected git_sha to be a hexadecimal string'
|
|
303
|
+
end
|
|
304
|
+
|
|
305
|
+
timestamp = data['extracted_at']
|
|
306
|
+
unless timestamp.nil? || timestamp.is_a?(String)
|
|
307
|
+
raise JSON::ParserError, 'expected extracted_at to be a string or null'
|
|
308
|
+
end
|
|
309
|
+
|
|
310
|
+
validate_snapshot_units!(data['units'])
|
|
311
|
+
end
|
|
312
|
+
|
|
313
|
+
def validate_snapshot_units!(units)
|
|
314
|
+
return if units.nil?
|
|
315
|
+
|
|
316
|
+
raise JSON::ParserError, 'expected units to be an object or null' unless units.is_a?(Hash)
|
|
317
|
+
return if units.each_value.all?(Hash)
|
|
318
|
+
|
|
319
|
+
raise JSON::ParserError, 'expected every unit record to be an object'
|
|
271
320
|
end
|
|
272
321
|
|
|
273
322
|
# @param exclude_sha [String, nil] SHA to leave out of the result
|