woods 2.0.0.beta2 → 2.0.0.beta4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +339 -1
- data/CONTRIBUTING.md +188 -12
- data/README.md +93 -174
- data/SECURITY.md +9 -6
- data/docs/AGENT_GUIDE.md +109 -8
- data/docs/AGENT_SETUP.md +98 -7
- data/docs/BACKEND_MATRIX.md +25 -0
- data/docs/CLIENT_HOOKS.md +111 -0
- data/docs/CONFIGURATION_REFERENCE.md +267 -16
- data/docs/CONSOLE_MCP_SETUP.md +80 -7
- data/docs/DOCKER_SETUP.md +22 -3
- data/docs/EVALUATION.md +464 -1
- data/docs/EXTRACTOR_REFERENCE.md +45 -6
- data/docs/FAQ.md +11 -12
- data/docs/GETTING_STARTED.md +17 -5
- data/docs/INCREMENTAL_EXTRACTION.md +147 -7
- data/docs/INDEX_LAYOUT.md +382 -0
- data/docs/INTERNALS.md +7 -2
- data/docs/MCP_SERVERS.md +276 -5
- data/docs/MCP_TOOL_COOKBOOK.md +37 -22
- data/docs/MCP_WORKTREE_SETUP.md +43 -83
- data/docs/NOTION_INTEGRATION.md +13 -0
- data/docs/OBSIDIAN_INTEGRATION.md +57 -9
- data/docs/PUBLISHED_INDEX.md +72 -0
- data/docs/README.md +7 -0
- data/docs/RETRIEVAL_GUIDE.md +273 -12
- data/docs/RUNTIME_TRACING.md +71 -0
- data/docs/SOURCE_FRESHNESS.md +143 -0
- data/docs/TROUBLESHOOTING.md +129 -18
- data/docs/UNBLOCKED_INTEGRATION.md +25 -0
- data/docs/UPGRADING_TO_2.md +48 -22
- data/docs/WATCH_DAEMON.md +277 -67
- data/exe/woods-agent-config +6 -0
- data/exe/woods-extract +5 -0
- data/exe/woods-hook-context +6 -0
- data/exe/woods-mcp-start +14 -9
- data/lib/generators/woods/pgvector_generator.rb +8 -2
- data/lib/generators/woods/templates/woods.rb.tt +1 -3
- data/lib/tasks/woods.rake +47 -397
- data/lib/woods/agent_configuration/applier.rb +135 -0
- data/lib/woods/agent_configuration/cli.rb +101 -0
- data/lib/woods/agent_configuration/cli_options.rb +29 -0
- data/lib/woods/agent_configuration/document.rb +105 -0
- data/lib/woods/agent_configuration/error.rb +7 -0
- data/lib/woods/agent_configuration/launcher.rb +75 -0
- data/lib/woods/agent_configuration/layout.rb +72 -0
- data/lib/woods/agent_configuration/managed_section.rb +62 -0
- data/lib/woods/agent_configuration/plan.rb +98 -0
- data/lib/woods/agent_configuration/plan_diff.rb +38 -0
- data/lib/woods/agent_configuration/planned_files.rb +61 -0
- data/lib/woods/agent_configuration/planner.rb +63 -0
- data/lib/woods/agent_configuration/planner_validation.rb +77 -0
- data/lib/woods/agent_configuration/preflight.rb +100 -0
- data/lib/woods/agent_configuration/recovery.rb +49 -0
- data/lib/woods/ast/node.rb +2 -0
- data/lib/woods/ast/parser.rb +38 -5
- data/lib/woods/builder.rb +21 -5
- data/lib/woods/cache/cache_middleware.rb +28 -7
- data/lib/woods/cache/cache_store.rb +4 -5
- data/lib/woods/change_set.rb +5 -4
- data/lib/woods/console/credential_index.rb +20 -2
- data/lib/woods/console/credential_scanner.rb +18 -17
- data/lib/woods/console/credential_scanner_registry.rb +36 -0
- data/lib/woods/console/dispatch_pipeline.rb +7 -0
- data/lib/woods/console/embedded_executor.rb +32 -10
- data/lib/woods/console/encrypted_credential_snapshot.rb +16 -0
- data/lib/woods/console/rack_middleware.rb +22 -13
- data/lib/woods/console/server.rb +18 -16
- data/lib/woods/console/sql_noise_stripper.rb +9 -7
- data/lib/woods/console/sql_table_scanner.rb +47 -7
- data/lib/woods/console/sql_validator.rb +49 -9
- data/lib/woods/console/sqlite_read_guard.rb +46 -0
- data/lib/woods/coordination/pipeline_lock.rb +3 -2
- data/lib/woods/dependency_graph.rb +65 -13
- data/lib/woods/embedding/corpus.rb +94 -0
- data/lib/woods/embedding/indexer.rb +114 -60
- data/lib/woods/embedding/openai.rb +17 -6
- data/lib/woods/evaluation/ablation_executor.rb +6 -1
- data/lib/woods/evaluation/ablation_timed_executor.rb +22 -4
- data/lib/woods/export/typed_reader.rb +56 -0
- data/lib/woods/extractor.rb +277 -149
- data/lib/woods/extractors/action_cable_extractor.rb +3 -1
- data/lib/woods/extractors/behavioral_profile.rb +9 -7
- data/lib/woods/extractors/caching_extractor.rb +3 -1
- data/lib/woods/extractors/concern_extractor.rb +64 -6
- data/lib/woods/extractors/configuration_extractor.rb +7 -3
- data/lib/woods/extractors/controller_extractor.rb +13 -4
- data/lib/woods/extractors/database_view_extractor.rb +3 -1
- data/lib/woods/extractors/declared_parent.rb +55 -0
- data/lib/woods/extractors/decorator_extractor.rb +3 -1
- data/lib/woods/extractors/engine_extractor.rb +3 -1
- data/lib/woods/extractors/event_extractor.rb +4 -2
- data/lib/woods/extractors/factory_extractor.rb +3 -1
- data/lib/woods/extractors/graphql_extractor.rb +10 -13
- data/lib/woods/extractors/i18n_extractor.rb +3 -1
- data/lib/woods/extractors/job_extractor.rb +6 -19
- data/lib/woods/extractors/lib_extractor.rb +13 -9
- data/lib/woods/extractors/mailer_extractor.rb +26 -15
- data/lib/woods/extractors/manager_extractor.rb +3 -1
- data/lib/woods/extractors/method_parameters.rb +53 -0
- data/lib/woods/extractors/middleware_argument.rb +65 -0
- data/lib/woods/extractors/middleware_extractor.rb +9 -3
- data/lib/woods/extractors/migration_extractor.rb +3 -1
- data/lib/woods/extractors/model_extractor.rb +26 -34
- data/lib/woods/extractors/package_extractor.rb +24 -4
- data/lib/woods/extractors/phlex_extractor.rb +3 -1
- data/lib/woods/extractors/policy_extractor.rb +3 -1
- data/lib/woods/extractors/poro_extractor.rb +13 -9
- data/lib/woods/extractors/pundit_extractor.rb +3 -1
- data/lib/woods/extractors/rails_source_extractor.rb +4 -2
- data/lib/woods/extractors/rake_task_extractor.rb +4 -2
- data/lib/woods/extractors/route_extractor.rb +3 -1
- data/lib/woods/extractors/route_helper_resolver.rb +10 -33
- data/lib/woods/extractors/scheduled_job_extractor.rb +41 -15
- data/lib/woods/extractors/serializer_extractor.rb +4 -2
- data/lib/woods/extractors/service_extractor.rb +3 -1
- data/lib/woods/extractors/shared_dependency_scanner.rb +2 -2
- data/lib/woods/extractors/shared_utility_methods.rb +48 -19
- data/lib/woods/extractors/source_nesting.rb +1 -1
- data/lib/woods/extractors/state_machine_extractor.rb +3 -1
- data/lib/woods/extractors/test_mapping_extractor.rb +3 -1
- data/lib/woods/extractors/validator_extractor.rb +3 -1
- data/lib/woods/extractors/view_component_extractor.rb +3 -1
- data/lib/woods/extractors/view_template_extractor.rb +3 -1
- data/lib/woods/gem_mapper.rb +2 -0
- data/lib/woods/git_history.rb +116 -0
- data/lib/woods/graph_analyzer.rb +35 -6
- data/lib/woods/hooks/context_cli.rb +54 -0
- data/lib/woods/hooks/context_event.rb +88 -0
- data/lib/woods/hooks/context_hint.rb +73 -0
- data/lib/woods/hooks/context_impact.rb +77 -0
- data/lib/woods/hooks/context_output.rb +47 -0
- data/lib/woods/hooks/context_state.rb +102 -0
- data/lib/woods/hooks/refresh.rb +79 -0
- data/lib/woods/hooks/rule_projection.rb +78 -0
- data/lib/woods/input_rules.rb +19 -0
- data/lib/woods/mcp/bearer_auth.rb +22 -13
- data/lib/woods/mcp/bootstrapper.rb +79 -4
- data/lib/woods/mcp/config_resolver.rb +2 -1
- data/lib/woods/mcp/index_reader.rb +334 -162
- data/lib/woods/mcp/initialization_guidance.rb +27 -0
- data/lib/woods/mcp/origin_guard.rb +17 -9
- data/lib/woods/mcp/published_lexical_retriever.rb +115 -0
- data/lib/woods/mcp/renderers/markdown_renderer.rb +22 -9
- data/lib/woods/mcp/renderers/plain_renderer.rb +18 -8
- data/lib/woods/mcp/search_results.rb +74 -0
- data/lib/woods/mcp/server.rb +178 -63
- data/lib/woods/mcp/tool_contract.rb +3 -1
- data/lib/woods/mcp/tool_response_renderer.rb +41 -0
- data/lib/woods/mcp/traversal_evidence.rb +113 -0
- data/lib/woods/mcp/traversal_evidence_index.rb +100 -0
- data/lib/woods/mcp/traversal_evidence_page.rb +41 -0
- data/lib/woods/mcp/traversal_evidence_text.rb +52 -0
- data/lib/woods/mcp/traversal_response.rb +22 -0
- data/lib/woods/notion/exporter.rb +56 -17
- data/lib/woods/obsidian/destination_plan.rb +98 -0
- data/lib/woods/obsidian/name_mapper.rb +19 -3
- data/lib/woods/obsidian/note_builder.rb +19 -10
- data/lib/woods/obsidian/vault_exporter.rb +88 -32
- data/lib/woods/operator/pipeline_guard.rb +18 -13
- data/lib/woods/path_dispatcher.rb +13 -6
- data/lib/woods/payload_store.rb +27 -26
- data/lib/woods/published_index/typed_unit_reader.rb +40 -3
- data/lib/woods/published_index.rb +2 -2
- data/lib/woods/railtie.rb +3 -3
- data/lib/woods/railtie_support.rb +12 -12
- data/lib/woods/rake_helpers.rb +382 -0
- data/lib/woods/resilience/graph_invariant_validator/membership_checks.rb +71 -0
- data/lib/woods/resilience/graph_invariant_validator/node_checks.rb +61 -0
- data/lib/woods/resilience/graph_invariant_validator/reverse_relationship_checks.rb +46 -0
- data/lib/woods/resilience/graph_invariant_validator.rb +119 -0
- data/lib/woods/resilience/index_validator/graph_checks.rb +80 -0
- data/lib/woods/resilience/index_validator.rb +112 -23
- data/lib/woods/retrieval/context_assembler.rb +50 -15
- data/lib/woods/retrieval/lexical_assembler.rb +84 -0
- data/lib/woods/retrieval/lexical_index.rb +120 -0
- data/lib/woods/retrieval/ranker.rb +4 -2
- data/lib/woods/retrieval/scope.rb +108 -0
- data/lib/woods/retrieval/scoped_graph_store.rb +32 -0
- data/lib/woods/retrieval/scoped_vector_store.rb +55 -0
- data/lib/woods/retrieval/search_executor.rb +86 -27
- data/lib/woods/retrieval/source_evidence.rb +200 -0
- data/lib/woods/retriever.rb +98 -22
- data/lib/woods/ruby_analyzer/trace_enricher.rb +77 -38
- data/lib/woods/session_tracer/file_store.rb +6 -1
- data/lib/woods/session_tracer/middleware.rb +10 -12
- data/lib/woods/session_tracer/redis_store.rb +22 -6
- data/lib/woods/session_tracer/session_flow_assembler.rb +23 -17
- data/lib/woods/session_tracer/solid_cache_coordination.rb +6 -4
- data/lib/woods/session_tracer/unit_resolver.rb +63 -0
- data/lib/woods/source_inputs/consumer_errors.rb +31 -0
- data/lib/woods/source_inputs/handoff.rb +102 -0
- data/lib/woods/source_inputs/launcher.rb +157 -0
- data/lib/woods/source_inputs/manifest.rb +124 -0
- data/lib/woods/source_inputs/private_key.rb +55 -0
- data/lib/woods/source_inputs/scanner.rb +171 -0
- data/lib/woods/source_inputs/scopes.rb +71 -0
- data/lib/woods/source_inputs/session.rb +214 -0
- data/lib/woods/source_inputs/status.rb +84 -0
- data/lib/woods/source_inputs/verifier.rb +107 -0
- data/lib/woods/storage/metadata_store.rb +25 -25
- data/lib/woods/storage/pgvector.rb +35 -10
- data/lib/woods/storage/qdrant.rb +17 -7
- data/lib/woods/storage/vector_store.rb +18 -6
- data/lib/woods/tasks.rb +3 -2
- data/lib/woods/temporal/json_snapshot_store.rb +58 -9
- data/lib/woods/unblocked/exporter.rb +59 -70
- data/lib/woods/version.rb +1 -1
- data/lib/woods/watch/boot_snapshot.rb +52 -0
- data/lib/woods/watch/daemon.rb +154 -32
- data/lib/woods/watch/listen_watcher.rb +4 -0
- data/lib/woods/watch/polling_watcher.rb +5 -1
- data/lib/woods/watch/status.rb +20 -15
- data/lib/woods/watch/tree_scan.rb +21 -13
- data/lib/woods/watch/watcher.rb +4 -1
- data/lib/woods.rb +50 -11
- data/plugin/.claude-plugin/plugin.json +1 -1
- data/plugin/hooks/adapters/normalize.jq +15 -0
- data/plugin/hooks/adapters/normalize.rb +63 -0
- data/plugin/hooks/hooks.json +20 -0
- data/plugin/hooks/woods-context.sh +50 -0
- data/plugin/hooks/woods-input-rules.sh +159 -0
- data/plugin/hooks/woods-opencode.mjs +65 -0
- data/plugin/hooks/woods-post-edit.sh +2 -225
- data/plugin/hooks/woods-refresh.sh +260 -0
- data/plugin/hooks/woods-session-start.sh +47 -55
- data/plugin/skills/woods-agent-enable/SKILL.md +19 -0
- data/plugin/skills/woods-diagnose/SKILL.md +319 -1
- data/plugin/skills/woods-investigate/SKILL.md +145 -0
- data/plugin/skills/woods-mcp-config/SKILL.md +90 -2
- data/plugin/skills/woods-setup/SKILL.md +110 -6
- metadata +87 -5
|
@@ -15,11 +15,9 @@ module Woods
|
|
|
15
15
|
# exercise the request-time procs and boot-time verification without
|
|
16
16
|
# loading Rails::Railtie.
|
|
17
17
|
module RailtieSupport
|
|
18
|
-
# Boot-time message
|
|
19
|
-
#
|
|
20
|
-
#
|
|
21
|
-
# path); the stdio server never sees a bearer header, so for it a missing
|
|
22
|
-
# token is a production-boot failure, not a request failure.
|
|
18
|
+
# Boot-time message when HTTP Console is enabled without a token.
|
|
19
|
+
# Production refuses to boot; other environments warn. Explicitly
|
|
20
|
+
# disabling HTTP leaves stdio available without HTTP token validation.
|
|
23
21
|
MISSING_TOKEN_MESSAGE =
|
|
24
22
|
'[Woods Console] console_mcp_token is not set — Console MCP is a high-privilege ' \
|
|
25
23
|
'endpoint that runs SQL and model introspection against the live database. ' \
|
|
@@ -37,7 +35,8 @@ module Woods
|
|
|
37
35
|
'HTTP Console MCP requests will be refused (401) until one is set. The stdio ' \
|
|
38
36
|
'transport (rake woods:console) does not transmit or use a bearer token, so a ' \
|
|
39
37
|
'stdio-only setup still works — set the token before exposing the HTTP endpoint. ' \
|
|
40
|
-
'
|
|
38
|
+
'Set console_mcp_http_enabled = false for stdio-only use. ' \
|
|
39
|
+
'Production boot will raise while HTTP Console MCP is enabled without a token.'
|
|
41
40
|
|
|
42
41
|
class << self
|
|
43
42
|
# @return [String, nil] path the console stack was mounted at (captured
|
|
@@ -53,12 +52,13 @@ module Woods
|
|
|
53
52
|
|
|
54
53
|
# Request-time predicate handed to the console guards and consulted by
|
|
55
54
|
# {Woods::Console::RackMiddleware}. Reads the live configuration on
|
|
56
|
-
# every call
|
|
55
|
+
# every call. Both the Console master switch and HTTP switch must be on,
|
|
56
|
+
# so a flag set after the middleware was inserted (e.g. in
|
|
57
57
|
# config/initializers) still takes effect.
|
|
58
58
|
#
|
|
59
59
|
# @return [Proc] arity-0 proc returning the current enabled flag
|
|
60
60
|
def console_enabled_proc
|
|
61
|
-
-> { config.console_mcp_enabled }
|
|
61
|
+
-> { config.console_mcp_enabled && config.console_mcp_http_enabled }
|
|
62
62
|
end
|
|
63
63
|
|
|
64
64
|
# Deferred bearer token for {Woods::MCP::BearerAuth}, resolved on each
|
|
@@ -95,12 +95,12 @@ module Woods
|
|
|
95
95
|
end
|
|
96
96
|
|
|
97
97
|
# Boot-time console validation, run from `after_initialize` once the
|
|
98
|
-
# final configuration is known. With
|
|
98
|
+
# final configuration is known. With HTTP Console enabled and no usable
|
|
99
99
|
# token, every HTTP request at the console path will be refused with
|
|
100
100
|
# 401 — production refuses to boot instead (matching the pre-#183
|
|
101
101
|
# fail-closed posture), other environments warn loudly. The stdio
|
|
102
|
-
# server carries no bearer check
|
|
103
|
-
#
|
|
102
|
+
# server carries no bearer check; explicit stdio-only configuration
|
|
103
|
+
# skips token validation entirely. A token shorter than
|
|
104
104
|
# {Woods::MCP::BearerAuth::MIN_TOKEN_LENGTH} is a misconfiguration and
|
|
105
105
|
# raises in every environment (as the eager BearerAuth constructor
|
|
106
106
|
# used to). Also warns when `console_mcp_path` changed after the stack
|
|
@@ -112,7 +112,7 @@ module Woods
|
|
|
112
112
|
# or a too-short token anywhere
|
|
113
113
|
def verify_console_configuration!(production:)
|
|
114
114
|
warn_late_console_path(config)
|
|
115
|
-
return unless config.console_mcp_enabled
|
|
115
|
+
return unless config.console_mcp_enabled && config.console_mcp_http_enabled
|
|
116
116
|
|
|
117
117
|
require 'woods/mcp/bearer_auth'
|
|
118
118
|
token = config.console_mcp_token.to_s
|
|
@@ -0,0 +1,382 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require 'woods/atomic_file'
|
|
4
|
+
require 'woods/generation'
|
|
5
|
+
require 'woods/git_command'
|
|
6
|
+
|
|
7
|
+
module Woods
|
|
8
|
+
# Shared rake implementation, kept off the host application's Object.
|
|
9
|
+
module RakeHelpers # rubocop:disable Metrics/ModuleLength -- existing task helpers grouped without changing behavior
|
|
10
|
+
module_function
|
|
11
|
+
|
|
12
|
+
# ── Multi-instance helpers (#164 phase 4) ────────────────────────────────
|
|
13
|
+
#
|
|
14
|
+
# Worktrees are disjoint by construction (each has its own Rails.root and
|
|
15
|
+
# its own output dir), so these only ever mediate writers against the *same*
|
|
16
|
+
# index: a manual rake run, a hook-triggered sync, and the watch daemon.
|
|
17
|
+
|
|
18
|
+
# Run a block holding the extraction lock, waiting for another writer to
|
|
19
|
+
# finish.
|
|
20
|
+
#
|
|
21
|
+
# This used to proceed *without* the lock after 30s, on the reasoning that a
|
|
22
|
+
# daemon cycle is milliseconds so a longer wait meant something unusual. That
|
|
23
|
+
# reasoning was wrong in the case that matters: a cycle includes a
|
|
24
|
+
# storm-triggered `extract_all`, which on a large host app runs for minutes.
|
|
25
|
+
# Proceeding then means two writers load `dependency_graph.json`, mutate
|
|
26
|
+
# divergent copies, and the last one silently discards the other's work — then
|
|
27
|
+
# bumps the generation, marking the clobbered graph fresh. Per-file atomic
|
|
28
|
+
# writes do not help, because the file *set* is not atomic.
|
|
29
|
+
#
|
|
30
|
+
# So the wait is now generous and the failure explicit. `WOODS_LOCK_WAIT`
|
|
31
|
+
# overrides it; exceeding it exits non-zero rather than corrupting the index,
|
|
32
|
+
# which is the outcome a CI job or a developer can actually act on.
|
|
33
|
+
# Fail the task when a run reported per-item errors (INF-4 / EXP-4).
|
|
34
|
+
#
|
|
35
|
+
# A printed-but-green run is invisible in the CI/cron pipelines these tasks
|
|
36
|
+
# target: a revoked API key, a full vector store, or a 401 on every page
|
|
37
|
+
# prints `Errors: N` and the job stays green while the index or the export
|
|
38
|
+
# target rots. `woods:unblocked_sync` and `woods:obsidian` already exit 1 on
|
|
39
|
+
# the same shape, and #270 fixed it for the extraction family; the embedding
|
|
40
|
+
# and Notion tasks carry identical CI exposure.
|
|
41
|
+
#
|
|
42
|
+
# Partial progress is deliberately *not* a carve-out here. The unblocked
|
|
43
|
+
# exporter has one (budget exhaustion with progress is the expected
|
|
44
|
+
# cold-start shape and converges on the next run); an embedding or Notion
|
|
45
|
+
# error is a genuine failure per unit and does not self-heal, so any
|
|
46
|
+
# non-zero count fails.
|
|
47
|
+
#
|
|
48
|
+
# @param errors [Integer, Array, nil] the run's error count or error list
|
|
49
|
+
# @return [void]
|
|
50
|
+
def woods_exit_on_reported_errors(errors)
|
|
51
|
+
# The embed indexer reports a count; the exporters report a list.
|
|
52
|
+
# (`Integer#size` exists and answers 8, so this must branch on the type.)
|
|
53
|
+
count = errors.is_a?(Numeric) ? errors.to_i : Array(errors).size
|
|
54
|
+
return if count.zero?
|
|
55
|
+
|
|
56
|
+
puts
|
|
57
|
+
puts 'Run completed with errors — failing so CI surfaces it.'
|
|
58
|
+
exit 1
|
|
59
|
+
end
|
|
60
|
+
|
|
61
|
+
def woods_with_extraction_lock(output_dir, wait: nil, raise_on_timeout: false, &block)
|
|
62
|
+
# Requires first. The default wait reads a constant from the daemon, so
|
|
63
|
+
# resolving it above these lines NameError'd every write task — the same
|
|
64
|
+
# load-order bug as the missing require in `woods:watch`, reintroduced one
|
|
65
|
+
# method over by the fix for it.
|
|
66
|
+
require 'woods/coordination/pipeline_lock'
|
|
67
|
+
require 'woods/coordination/lock_heartbeat'
|
|
68
|
+
require 'woods/watch/daemon'
|
|
69
|
+
|
|
70
|
+
wait ||= Float(ENV.fetch('WOODS_LOCK_WAIT', Woods::Watch::Daemon::LOCK_STALE_TIMEOUT))
|
|
71
|
+
|
|
72
|
+
lock = Woods::Coordination::PipelineLock.new(
|
|
73
|
+
lock_dir: output_dir.to_s,
|
|
74
|
+
name: Woods::Watch::Daemon::LOCK_NAME,
|
|
75
|
+
stale_timeout: Woods::Watch::Daemon::LOCK_STALE_TIMEOUT
|
|
76
|
+
)
|
|
77
|
+
|
|
78
|
+
woods_abort_on_lock_timeout(wait, raise_error: raise_on_timeout) unless woods_acquire_within(lock, wait)
|
|
79
|
+
|
|
80
|
+
begin
|
|
81
|
+
Woods::Coordination::LockHeartbeat.run(lock, &block)
|
|
82
|
+
ensure
|
|
83
|
+
lock.release
|
|
84
|
+
end
|
|
85
|
+
end
|
|
86
|
+
|
|
87
|
+
# Poll for the lock until `wait` seconds have elapsed.
|
|
88
|
+
#
|
|
89
|
+
# Monotonic, so a clock adjustment mid-wait cannot cut the window short or
|
|
90
|
+
# extend it indefinitely.
|
|
91
|
+
#
|
|
92
|
+
# @return [Boolean] whether the lock was acquired
|
|
93
|
+
def woods_acquire_within(lock, wait)
|
|
94
|
+
deadline = Process.clock_gettime(Process::CLOCK_MONOTONIC) + wait
|
|
95
|
+
acquired = lock.acquire
|
|
96
|
+
until acquired || Process.clock_gettime(Process::CLOCK_MONOTONIC) > deadline
|
|
97
|
+
sleep 0.25
|
|
98
|
+
acquired = lock.acquire
|
|
99
|
+
end
|
|
100
|
+
acquired
|
|
101
|
+
end
|
|
102
|
+
|
|
103
|
+
def woods_abort_on_lock_timeout(wait, raise_error: false)
|
|
104
|
+
if raise_error
|
|
105
|
+
raise Coordination::LockError,
|
|
106
|
+
"Another writer holds the extraction lock after #{wait}s; set WOODS_LOCK_WAIT to wait longer"
|
|
107
|
+
end
|
|
108
|
+
|
|
109
|
+
warn "ERROR: another writer has held the extraction lock for #{wait.round}s."
|
|
110
|
+
warn 'Refusing to write: two concurrent writers rewrite the dependency graph from divergent'
|
|
111
|
+
warn 'copies, and the loser\'s work is discarded under a generation that says "fresh".'
|
|
112
|
+
warn 'Set WOODS_LOCK_WAIT to wait longer, or stop the other writer.'
|
|
113
|
+
exit 1
|
|
114
|
+
end
|
|
115
|
+
|
|
116
|
+
# Is a watch daemon already maintaining this index?
|
|
117
|
+
#
|
|
118
|
+
# A session-start or worktree hook that fires `woods:incremental` on a tree
|
|
119
|
+
# a daemon is already watching is pure duplicated work — and it contends for
|
|
120
|
+
# the lock the daemon needs. Set WOODS_IGNORE_WATCH=1 to run anyway.
|
|
121
|
+
#
|
|
122
|
+
# Liveness alone is not coverage, and the difference matters. A `:running`
|
|
123
|
+
# daemon reconciles everything modified since the index's last successful
|
|
124
|
+
# publish when it starts (Daemon#catch_up), so changes that predate it are
|
|
125
|
+
# covered whether or not it witnessed them — that is what makes standing down
|
|
126
|
+
# safe. A `:degraded` daemon is alive but *cannot* currently update, so
|
|
127
|
+
# standing down for it would report success over work nothing is doing.
|
|
128
|
+
#
|
|
129
|
+
# @return [Symbol] `:none`, `:running`, or `:degraded`
|
|
130
|
+
def woods_daemon_coverage(output_dir)
|
|
131
|
+
return :none if ENV['WOODS_IGNORE_WATCH'] == '1'
|
|
132
|
+
|
|
133
|
+
require 'woods/watch/status'
|
|
134
|
+
status = Woods::Watch::Status.new(output_dir: output_dir)
|
|
135
|
+
return :none unless status.alive?
|
|
136
|
+
|
|
137
|
+
status.read['state'] == 'degraded' ? :degraded : :running
|
|
138
|
+
rescue StandardError
|
|
139
|
+
:none
|
|
140
|
+
end
|
|
141
|
+
|
|
142
|
+
# Delete the index without yanking it out from under another writer (#170).
|
|
143
|
+
#
|
|
144
|
+
# The old `woods:clean` body was a bare rm_rf. Run mid-daemon-cycle it
|
|
145
|
+
# deleted the daemon's own extraction lock, so the "writers serialize on
|
|
146
|
+
# PipelineLock" invariant evaporated at the exact moment a writer was
|
|
147
|
+
# mid-graph-rewrite. Now: refuse while a daemon is alive (same stand-down
|
|
148
|
+
# check as `woods:incremental`; WOODS_IGNORE_WATCH=1 overrides), then take
|
|
149
|
+
# the extraction lock like every other writer, delete everything EXCEPT the
|
|
150
|
+
# lock file itself, and let release remove the lock last.
|
|
151
|
+
#
|
|
152
|
+
# The directory the published generation's payload lives in.
|
|
153
|
+
#
|
|
154
|
+
# Read-side tasks resolve through this rather than the output root: an index
|
|
155
|
+
# publishing per-generation payloads keeps only `generation.json`, `dumps/`,
|
|
156
|
+
# `payloads/` and the lock files at the root. A flat index resolves to the
|
|
157
|
+
# root unchanged.
|
|
158
|
+
#
|
|
159
|
+
# @param output_dir [Pathname, String] index directory
|
|
160
|
+
# @return [Pathname]
|
|
161
|
+
def woods_payload_dir(output_dir)
|
|
162
|
+
Woods::Generation.new(output_dir: output_dir).payload_dir
|
|
163
|
+
end
|
|
164
|
+
|
|
165
|
+
# @param output_dir [Pathname, String] index directory
|
|
166
|
+
# @param wait [Numeric, nil] seconds to wait for the lock (default: the
|
|
167
|
+
# shared writer wait; injectable so a spec need not sit out the window)
|
|
168
|
+
# @return [Symbol] `:cleaned`, or `:refused` when a live daemon is
|
|
169
|
+
# maintaining this index
|
|
170
|
+
def woods_clean_index(output_dir, wait: nil)
|
|
171
|
+
require 'woods/watch/daemon'
|
|
172
|
+
|
|
173
|
+
output_dir = Pathname.new(output_dir)
|
|
174
|
+
|
|
175
|
+
unless woods_daemon_coverage(output_dir) == :none
|
|
176
|
+
warn 'ERROR: a watch daemon is maintaining this index — refusing to delete it out from under it.'
|
|
177
|
+
warn 'Stop the daemon first, or set WOODS_IGNORE_WATCH=1 to clean anyway.'
|
|
178
|
+
return :refused
|
|
179
|
+
end
|
|
180
|
+
|
|
181
|
+
lock_name = Woods::Watch::Daemon::LOCK_NAME
|
|
182
|
+
woods_with_extraction_lock(output_dir, wait: wait) do
|
|
183
|
+
woods_sweep_index_dir(output_dir, lock_name)
|
|
184
|
+
end
|
|
185
|
+
|
|
186
|
+
# Retain the stable guard and directory: another writer may already hold
|
|
187
|
+
# the guard after release, before it has created its extraction.lock.
|
|
188
|
+
:cleaned
|
|
189
|
+
end
|
|
190
|
+
|
|
191
|
+
# Delete every index artifact except the lock file and its transaction
|
|
192
|
+
# guard. The guard lives inside the lock directory and a contender may hold
|
|
193
|
+
# its flock right now — deleting it mid-sweep would split the flock across
|
|
194
|
+
# two inodes and defeat the mutual exclusion it provides.
|
|
195
|
+
def woods_sweep_index_dir(output_dir, lock_name)
|
|
196
|
+
preserved = [
|
|
197
|
+
output_dir.join("#{lock_name}.lock"),
|
|
198
|
+
output_dir.join(Woods::Coordination::PipelineLock.guard_filename(lock_name))
|
|
199
|
+
]
|
|
200
|
+
output_dir.children.each do |entry|
|
|
201
|
+
FileUtils.rm_rf(entry) unless preserved.include?(entry)
|
|
202
|
+
end
|
|
203
|
+
end
|
|
204
|
+
|
|
205
|
+
# The root containing the Rakefile that loaded this task file.
|
|
206
|
+
#
|
|
207
|
+
# `woods:watch_status` intentionally avoids Rails boot, so it cannot ask
|
|
208
|
+
# `Rails.root` for the conventional output path. `Dir.pwd` is not a stable
|
|
209
|
+
# substitute: `rake -f /app/Rakefile` and worktree launchers may invoke the
|
|
210
|
+
# task from somewhere else. Rake retains the selected Rakefile even when it
|
|
211
|
+
# does not chdir, which gives the same application root without loading the
|
|
212
|
+
# environment.
|
|
213
|
+
#
|
|
214
|
+
# @return [String] absolute directory containing the active Rakefile
|
|
215
|
+
def woods_task_root
|
|
216
|
+
rakefile = Rake.application.rakefile
|
|
217
|
+
return Rake.application.original_dir if rakefile.nil? || rakefile.empty?
|
|
218
|
+
|
|
219
|
+
# Rake may keep this relative (`Rakefile`) after searching upward from a
|
|
220
|
+
# nested invocation. At task execution its cwd is the directory it loaded
|
|
221
|
+
# that relative file from. An explicit absolute `-f` is already complete.
|
|
222
|
+
File.dirname(File.expand_path(rakefile))
|
|
223
|
+
end
|
|
224
|
+
|
|
225
|
+
# Changed paths across a git range, for `woods:incremental`'s CI branches.
|
|
226
|
+
#
|
|
227
|
+
# `git diff --name-only` split on lines corrupted three things at once: a
|
|
228
|
+
# path containing a newline split into two entries; a non-ASCII path came
|
|
229
|
+
# back octal-escaped inside quotes under git's default `core.quotePath`,
|
|
230
|
+
# which the dispatcher then can't match to any rule; and a rename reported
|
|
231
|
+
# only the new path, so the old path's unit was never pruned. `-z` +
|
|
232
|
+
# `--name-status` + `--no-renames` fixes all three: NUL-delimited records,
|
|
233
|
+
# `core.quotePath=false` unescaped, and a rename decomposed by git itself
|
|
234
|
+
# into a separate `A <new>` and `D <old>` record rather than one `R` record
|
|
235
|
+
# naming both.
|
|
236
|
+
#
|
|
237
|
+
# The diff is rooted at the extracted application (`git -C Rails.root`),
|
|
238
|
+
# consistent with {Woods::GitProvenance} (#262): extraction launched from
|
|
239
|
+
# another checkout must not diff that checkout's history.
|
|
240
|
+
#
|
|
241
|
+
# The child status is carried out, not discarded (M1): `Open3.capture2`
|
|
242
|
+
# turned an unresolvable range — a GitLab zero-SHA, an unfetched GitHub base
|
|
243
|
+
# ref, garbage — into an empty change set the caller could not tell apart
|
|
244
|
+
# from "nothing changed", and `woods:incremental` exited 0 over a sync that
|
|
245
|
+
# never ran.
|
|
246
|
+
#
|
|
247
|
+
# @param range [String] a git diff range/revision expression
|
|
248
|
+
# @param root [Pathname, String] repository root the diff runs against
|
|
249
|
+
# @return [Array(Array<String>, String, nil)] the changed paths (both halves
|
|
250
|
+
# of any rename included), or nil paths plus a human-readable failure when
|
|
251
|
+
# the range could not be resolved
|
|
252
|
+
def woods_changed_paths_for_range(range, root: Rails.root)
|
|
253
|
+
require 'open3'
|
|
254
|
+
output, error, status = Open3.capture3(
|
|
255
|
+
*Woods::GitCommand.argv(
|
|
256
|
+
root, '-c', 'core.quotePath=false',
|
|
257
|
+
'diff', '--name-status', '-z', '--no-renames', range
|
|
258
|
+
)
|
|
259
|
+
)
|
|
260
|
+
return [woods_parse_git_diff_name_status(output), nil] if status.success?
|
|
261
|
+
|
|
262
|
+
[nil, "#{error.strip} (git exited #{status.exitstatus})"]
|
|
263
|
+
rescue SystemCallError => e
|
|
264
|
+
# No git binary at all (a slim container image, git removed after
|
|
265
|
+
# checkout). Errno::ENOENT out of `Open3.capture3` used to kill the task
|
|
266
|
+
# with a backtrace: non-zero, so safe, but it bypassed the decision below
|
|
267
|
+
# — the `:running`-daemon stand-down branch was unreachable, and the
|
|
268
|
+
# operator got a stack trace instead of the remediation text. An absent
|
|
269
|
+
# binary is an unresolvable range like any other (INF-12).
|
|
270
|
+
[nil, "git unavailable: #{e.message}"]
|
|
271
|
+
end
|
|
272
|
+
|
|
273
|
+
# The changed-path set `woods:incremental` will process, or a stand-down.
|
|
274
|
+
#
|
|
275
|
+
# An explicit `CHANGED_FILES` list bypasses git entirely. Otherwise the
|
|
276
|
+
# range comes from the CI environment (GitLab's before-SHA, GitHub's base
|
|
277
|
+
# ref) or defaults to the last commit.
|
|
278
|
+
#
|
|
279
|
+
# A failed range is a decision, not a skip (M1): the changed-file set is
|
|
280
|
+
# unknown, so silently extracting nothing would exit 0 over work that never
|
|
281
|
+
# happened and leave CI drift unbounded. When a `:running` daemon maintains
|
|
282
|
+
# the index, its start-up catch-up covers whatever changed, so standing down
|
|
283
|
+
# with a printed reason is safe; a degraded daemon covers nothing, so the
|
|
284
|
+
# run fails like any other uncovered case. The decision runs BEFORE the
|
|
285
|
+
# task's empty-range exit — a failed range must never be mistaken for an
|
|
286
|
+
# empty one.
|
|
287
|
+
#
|
|
288
|
+
# @param output_dir [Pathname, String] index directory, for daemon coverage
|
|
289
|
+
# @return [Array<String>] changed paths
|
|
290
|
+
def woods_incremental_changed_paths(output_dir)
|
|
291
|
+
explicit = ENV.fetch('CHANGED_FILES', nil)
|
|
292
|
+
return explicit.split(',').map(&:strip) if explicit
|
|
293
|
+
|
|
294
|
+
range = woods_incremental_range
|
|
295
|
+
changed_files, failure = woods_changed_paths_for_range(range)
|
|
296
|
+
return changed_files unless failure
|
|
297
|
+
|
|
298
|
+
if woods_daemon_coverage(output_dir) == :running
|
|
299
|
+
puts "Could not resolve the git diff range #{range.inspect} (#{failure})."
|
|
300
|
+
puts 'A watch daemon is maintaining this index, so its catch-up covers the changed paths — standing down.'
|
|
301
|
+
puts 'To extract now anyway, repair or provide the range (check the CI env refs),'
|
|
302
|
+
puts 'set CHANGED_FILES explicitly, or run a full woods:extract.'
|
|
303
|
+
exit 0
|
|
304
|
+
end
|
|
305
|
+
|
|
306
|
+
warn "ERROR: could not resolve the git diff range #{range.inspect}: #{failure}"
|
|
307
|
+
warn 'The changed-file set is unknown, so incremental extraction would silently index nothing.'
|
|
308
|
+
warn 'Fix the range (a zero SHA or an unfetched base ref resolve to nothing), or run a full woods:extract.'
|
|
309
|
+
exit 1
|
|
310
|
+
end
|
|
311
|
+
|
|
312
|
+
# The git range `woods:incremental` diffs, from the CI environment or the
|
|
313
|
+
# last-commit default.
|
|
314
|
+
#
|
|
315
|
+
# @return [String]
|
|
316
|
+
def woods_incremental_range
|
|
317
|
+
if ENV['CI_COMMIT_BEFORE_SHA']
|
|
318
|
+
# GitLab CI
|
|
319
|
+
"#{ENV['CI_COMMIT_BEFORE_SHA']}..#{ENV.fetch('CI_COMMIT_SHA', nil)}"
|
|
320
|
+
elsif ENV['GITHUB_BASE_REF']
|
|
321
|
+
# GitHub Actions PR
|
|
322
|
+
"origin/#{ENV['GITHUB_BASE_REF']}...HEAD"
|
|
323
|
+
else
|
|
324
|
+
# Default: changes since last commit
|
|
325
|
+
'HEAD~1'
|
|
326
|
+
end
|
|
327
|
+
end
|
|
328
|
+
|
|
329
|
+
# Parse NUL-delimited `git diff --name-status -z --no-renames` output.
|
|
330
|
+
#
|
|
331
|
+
# Each record is `<status>\0<path>\0` — `--no-renames` guarantees a single
|
|
332
|
+
# path per record, since it is what stops git emitting a two-path `R`/`C`
|
|
333
|
+
# record in the first place. Status letters are not inspected beyond "did
|
|
334
|
+
# git report anything at all"; a deleted path still needs to reach the
|
|
335
|
+
# change set so its unit can be pruned.
|
|
336
|
+
#
|
|
337
|
+
# A bare `Open3.capture2` read is tagged with the process's default
|
|
338
|
+
# external encoding — US-ASCII under `LANG=C`, this daemon's usual
|
|
339
|
+
# environment (see `AtomicFile.read`'s gotcha) — so a UTF-8 path is a
|
|
340
|
+
# US-ASCII string containing invalid bytes until re-tagged.
|
|
341
|
+
#
|
|
342
|
+
# @param output [String] raw NUL-delimited git output
|
|
343
|
+
# @return [Array<String>] changed paths
|
|
344
|
+
def woods_parse_git_diff_name_status(output)
|
|
345
|
+
fields = output.dup.force_encoding(Encoding::UTF_8).split("\x00")
|
|
346
|
+
paths = []
|
|
347
|
+
fields.each_slice(2) do |status, path|
|
|
348
|
+
break if path.nil?
|
|
349
|
+
|
|
350
|
+
paths << path unless status.nil? || status.empty?
|
|
351
|
+
end
|
|
352
|
+
paths
|
|
353
|
+
end
|
|
354
|
+
|
|
355
|
+
# Resolve the configured retrieval stack and run one ad-hoc query (#178).
|
|
356
|
+
#
|
|
357
|
+
# Every backend — embedding provider, vector store, metadata store, graph
|
|
358
|
+
# store — is resolved through Woods::Builder from Woods.configuration,
|
|
359
|
+
# the same wiring `woods:embed` writes through
|
|
360
|
+
# (Woods::Tasks.build_embed_indexer). The old task body hardcoded
|
|
361
|
+
# Ollama + InMemory + SQLite + Memory, so on any other configured stack it
|
|
362
|
+
# queried backends the embed run never wrote to and silently returned
|
|
363
|
+
# nothing.
|
|
364
|
+
#
|
|
365
|
+
# In-memory stores start empty in a fresh process; hosts on the :local /
|
|
366
|
+
# :shared_filesystem presets should query through woods-mcp, which
|
|
367
|
+
# hydrates them from the dumps on disk.
|
|
368
|
+
#
|
|
369
|
+
# @param query [String] natural-language retrieval query
|
|
370
|
+
# @return [String] human-formatted retrieval output
|
|
371
|
+
def woods_run_retrieval(query)
|
|
372
|
+
require 'woods'
|
|
373
|
+
require 'woods/formatting/human_adapter'
|
|
374
|
+
|
|
375
|
+
config = Woods.configuration
|
|
376
|
+
retriever = Woods::Builder.new(config).build_retriever
|
|
377
|
+
result = retriever.retrieve(query, budget: config.max_context_tokens)
|
|
378
|
+
|
|
379
|
+
Woods::Formatting::HumanAdapter.new.format(result)
|
|
380
|
+
end
|
|
381
|
+
end
|
|
382
|
+
end
|
|
@@ -0,0 +1,71 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Woods
|
|
4
|
+
module Resilience
|
|
5
|
+
class GraphInvariantValidator
|
|
6
|
+
# Set-membership and index agreement checks, sharing the validator
|
|
7
|
+
# diagnostics and typed nodes. Keeping these separate leaves graph
|
|
8
|
+
# edge/variant validation in the main validator.
|
|
9
|
+
module MembershipChecks
|
|
10
|
+
private
|
|
11
|
+
|
|
12
|
+
def validate_membership(section, expected, scalar: false)
|
|
13
|
+
actual = object_section(section).each_with_object({}) do |(key, entries), result|
|
|
14
|
+
entries = [entries] if scalar && entries.is_a?(String)
|
|
15
|
+
values = membership_entries(key, entries, section)
|
|
16
|
+
result[key] = values if values
|
|
17
|
+
end
|
|
18
|
+
(expected.keys | actual.keys).each do |key|
|
|
19
|
+
compare_membership(section, key, expected.fetch(key, Set.new), actual.fetch(key, Set.new))
|
|
20
|
+
end
|
|
21
|
+
end
|
|
22
|
+
|
|
23
|
+
def compare_membership(section, key, wanted, recorded)
|
|
24
|
+
(wanted - recorded).each { |id| error("#{section}[#{key.inspect}]", "missing #{id.inspect}") }
|
|
25
|
+
(recorded - wanted).each { |id| error("#{section}[#{key.inspect}]", "unexpected #{id.inspect}") }
|
|
26
|
+
end
|
|
27
|
+
|
|
28
|
+
def membership_entries(key, entries, section)
|
|
29
|
+
return entries.to_set if name?(key) && entries.is_a?(Array) && entries.all? { |entry| name?(entry) }
|
|
30
|
+
|
|
31
|
+
error("#{section}[#{key.inspect}]", 'expected a named bucket containing identifiers')
|
|
32
|
+
nil
|
|
33
|
+
end
|
|
34
|
+
|
|
35
|
+
def validate_index_agreement
|
|
36
|
+
unless @index_entries.is_a?(Array)
|
|
37
|
+
error('unit indexes', 'expected typed index entries')
|
|
38
|
+
return
|
|
39
|
+
end
|
|
40
|
+
|
|
41
|
+
entries = {}
|
|
42
|
+
@index_entries.each { |entry| collect_index_entry(entry, entries) }
|
|
43
|
+
(@typed_nodes.keys - entries.keys).each do |identifier, type|
|
|
44
|
+
error('unit indexes', "graph node #{type}:#{identifier} is not indexed")
|
|
45
|
+
end
|
|
46
|
+
(entries.keys - @typed_nodes.keys).each do |identifier, type|
|
|
47
|
+
error('nodes', "missing indexed unit #{type}:#{identifier}")
|
|
48
|
+
end
|
|
49
|
+
end
|
|
50
|
+
|
|
51
|
+
def collect_index_entry(entry, entries)
|
|
52
|
+
unless typed_entry?(entry)
|
|
53
|
+
error('unit indexes', 'entry requires a nonempty identifier and type')
|
|
54
|
+
return
|
|
55
|
+
end
|
|
56
|
+
key = [entry['identifier'], entry['type']]
|
|
57
|
+
error('unit indexes', "duplicate typed entry #{key.last}:#{key.first}") if entries.key?(key)
|
|
58
|
+
entries[key] = entry
|
|
59
|
+
node = @typed_nodes[key]
|
|
60
|
+
return unless node && entry.key?('file_path') && node['file_path'] != entry['file_path']
|
|
61
|
+
|
|
62
|
+
error('unit indexes', "file_path differs for #{key.last}:#{key.first}")
|
|
63
|
+
end
|
|
64
|
+
|
|
65
|
+
def typed_entry?(entry)
|
|
66
|
+
entry.is_a?(Hash) && name?(entry['identifier']) && name?(entry['type'])
|
|
67
|
+
end
|
|
68
|
+
end
|
|
69
|
+
end
|
|
70
|
+
end
|
|
71
|
+
end
|
|
@@ -0,0 +1,61 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Woods
|
|
4
|
+
module Resilience
|
|
5
|
+
class GraphInvariantValidator
|
|
6
|
+
# Collect typed primary/variant identities without normalizing malformed
|
|
7
|
+
# records away; derive type and file membership from those identities.
|
|
8
|
+
module NodeChecks
|
|
9
|
+
private
|
|
10
|
+
|
|
11
|
+
def collect_nodes
|
|
12
|
+
object_section('nodes').each do |identifier, node|
|
|
13
|
+
type = register_node(identifier, node, "nodes[#{identifier.inspect}]")
|
|
14
|
+
@primary_types[identifier] = type if type
|
|
15
|
+
end
|
|
16
|
+
@variants = @graph.fetch('variants', [])
|
|
17
|
+
unless @variants.is_a?(Array)
|
|
18
|
+
error('variants', 'expected an array')
|
|
19
|
+
@variants = []
|
|
20
|
+
end
|
|
21
|
+
@variants.each_with_index { |record, index| collect_variant(record, index) }
|
|
22
|
+
end
|
|
23
|
+
|
|
24
|
+
def collect_variant(record, index)
|
|
25
|
+
label = "variants[#{index}]"
|
|
26
|
+
unless record.is_a?(Hash)
|
|
27
|
+
error(label, 'expected an object')
|
|
28
|
+
return
|
|
29
|
+
end
|
|
30
|
+
|
|
31
|
+
identifier = record['identifier']
|
|
32
|
+
error(label, "missing primary node for #{identifier.inspect}") unless @primary_types.key?(identifier)
|
|
33
|
+
register_node(identifier, record, label)
|
|
34
|
+
end
|
|
35
|
+
|
|
36
|
+
def register_node(identifier, node, label)
|
|
37
|
+
unless name?(identifier) && node.is_a?(Hash) && name?(node['type'])
|
|
38
|
+
error(label, 'expected a nonempty identifier and an object with a nonempty type')
|
|
39
|
+
return
|
|
40
|
+
end
|
|
41
|
+
type = node['type']
|
|
42
|
+
key = [identifier, type]
|
|
43
|
+
if @typed_nodes.key?(key)
|
|
44
|
+
error(label, "duplicate typed node #{type}:#{identifier}")
|
|
45
|
+
return
|
|
46
|
+
end
|
|
47
|
+
|
|
48
|
+
@typed_nodes[key] = node
|
|
49
|
+
@expected_types[type].add(identifier)
|
|
50
|
+
path = node['file_path']
|
|
51
|
+
if path.is_a?(String)
|
|
52
|
+
@expected_files[path].add(identifier)
|
|
53
|
+
elsif !path.nil?
|
|
54
|
+
error(label, 'file_path must be a string or null')
|
|
55
|
+
end
|
|
56
|
+
type
|
|
57
|
+
end
|
|
58
|
+
end
|
|
59
|
+
end
|
|
60
|
+
end
|
|
61
|
+
end
|
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Woods
|
|
4
|
+
module Resilience
|
|
5
|
+
class GraphInvariantValidator
|
|
6
|
+
# The additive reverse relationship index must preserve every typed
|
|
7
|
+
# forward record. Legacy graphs can omit it entirely.
|
|
8
|
+
module ReverseRelationshipChecks
|
|
9
|
+
RELATIONSHIP_KEYS = %w[via through through_db disable_joins].freeze
|
|
10
|
+
|
|
11
|
+
private
|
|
12
|
+
|
|
13
|
+
def record_reverse_relationship(target, source, source_type, edge)
|
|
14
|
+
return unless name?(source) && name?(source_type)
|
|
15
|
+
|
|
16
|
+
attributes = edge.is_a?(Hash) ? edge.slice(*RELATIONSHIP_KEYS) : {}
|
|
17
|
+
record = { 'source' => source, 'source_type' => source_type, 'via' => nil }.merge(attributes)
|
|
18
|
+
@expected_reverse_via[target] << record
|
|
19
|
+
end
|
|
20
|
+
|
|
21
|
+
def validate_reverse_relationships
|
|
22
|
+
return unless @graph.key?('reverse_via')
|
|
23
|
+
|
|
24
|
+
actual = object_section('reverse_via')
|
|
25
|
+
(@expected_reverse_via.keys | actual.keys).each do |target|
|
|
26
|
+
compare_reverse_relationships(target, actual.fetch(target, []))
|
|
27
|
+
end
|
|
28
|
+
end
|
|
29
|
+
|
|
30
|
+
def compare_reverse_relationships(target, rows)
|
|
31
|
+
label = "reverse_via[#{target.inspect}]"
|
|
32
|
+
unless name?(target) && rows.is_a?(Array) && rows.all?(Hash)
|
|
33
|
+
error(label, 'expected a named bucket containing relationship objects')
|
|
34
|
+
return
|
|
35
|
+
end
|
|
36
|
+
expected = @expected_reverse_via.fetch(target, [])
|
|
37
|
+
keys = %w[source source_type] + RELATIONSHIP_KEYS
|
|
38
|
+
recorded = rows.map { |row| row.slice(*keys) }
|
|
39
|
+
return if expected.tally == recorded.tally
|
|
40
|
+
|
|
41
|
+
error(label, 'relationship records differ from typed forward edges')
|
|
42
|
+
end
|
|
43
|
+
end
|
|
44
|
+
end
|
|
45
|
+
end
|
|
46
|
+
end
|