woods 2.0.0.beta2 → 2.0.0.beta4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +339 -1
- data/CONTRIBUTING.md +188 -12
- data/README.md +93 -174
- data/SECURITY.md +9 -6
- data/docs/AGENT_GUIDE.md +109 -8
- data/docs/AGENT_SETUP.md +98 -7
- data/docs/BACKEND_MATRIX.md +25 -0
- data/docs/CLIENT_HOOKS.md +111 -0
- data/docs/CONFIGURATION_REFERENCE.md +267 -16
- data/docs/CONSOLE_MCP_SETUP.md +80 -7
- data/docs/DOCKER_SETUP.md +22 -3
- data/docs/EVALUATION.md +464 -1
- data/docs/EXTRACTOR_REFERENCE.md +45 -6
- data/docs/FAQ.md +11 -12
- data/docs/GETTING_STARTED.md +17 -5
- data/docs/INCREMENTAL_EXTRACTION.md +147 -7
- data/docs/INDEX_LAYOUT.md +382 -0
- data/docs/INTERNALS.md +7 -2
- data/docs/MCP_SERVERS.md +276 -5
- data/docs/MCP_TOOL_COOKBOOK.md +37 -22
- data/docs/MCP_WORKTREE_SETUP.md +43 -83
- data/docs/NOTION_INTEGRATION.md +13 -0
- data/docs/OBSIDIAN_INTEGRATION.md +57 -9
- data/docs/PUBLISHED_INDEX.md +72 -0
- data/docs/README.md +7 -0
- data/docs/RETRIEVAL_GUIDE.md +273 -12
- data/docs/RUNTIME_TRACING.md +71 -0
- data/docs/SOURCE_FRESHNESS.md +143 -0
- data/docs/TROUBLESHOOTING.md +129 -18
- data/docs/UNBLOCKED_INTEGRATION.md +25 -0
- data/docs/UPGRADING_TO_2.md +48 -22
- data/docs/WATCH_DAEMON.md +277 -67
- data/exe/woods-agent-config +6 -0
- data/exe/woods-extract +5 -0
- data/exe/woods-hook-context +6 -0
- data/exe/woods-mcp-start +14 -9
- data/lib/generators/woods/pgvector_generator.rb +8 -2
- data/lib/generators/woods/templates/woods.rb.tt +1 -3
- data/lib/tasks/woods.rake +47 -397
- data/lib/woods/agent_configuration/applier.rb +135 -0
- data/lib/woods/agent_configuration/cli.rb +101 -0
- data/lib/woods/agent_configuration/cli_options.rb +29 -0
- data/lib/woods/agent_configuration/document.rb +105 -0
- data/lib/woods/agent_configuration/error.rb +7 -0
- data/lib/woods/agent_configuration/launcher.rb +75 -0
- data/lib/woods/agent_configuration/layout.rb +72 -0
- data/lib/woods/agent_configuration/managed_section.rb +62 -0
- data/lib/woods/agent_configuration/plan.rb +98 -0
- data/lib/woods/agent_configuration/plan_diff.rb +38 -0
- data/lib/woods/agent_configuration/planned_files.rb +61 -0
- data/lib/woods/agent_configuration/planner.rb +63 -0
- data/lib/woods/agent_configuration/planner_validation.rb +77 -0
- data/lib/woods/agent_configuration/preflight.rb +100 -0
- data/lib/woods/agent_configuration/recovery.rb +49 -0
- data/lib/woods/ast/node.rb +2 -0
- data/lib/woods/ast/parser.rb +38 -5
- data/lib/woods/builder.rb +21 -5
- data/lib/woods/cache/cache_middleware.rb +28 -7
- data/lib/woods/cache/cache_store.rb +4 -5
- data/lib/woods/change_set.rb +5 -4
- data/lib/woods/console/credential_index.rb +20 -2
- data/lib/woods/console/credential_scanner.rb +18 -17
- data/lib/woods/console/credential_scanner_registry.rb +36 -0
- data/lib/woods/console/dispatch_pipeline.rb +7 -0
- data/lib/woods/console/embedded_executor.rb +32 -10
- data/lib/woods/console/encrypted_credential_snapshot.rb +16 -0
- data/lib/woods/console/rack_middleware.rb +22 -13
- data/lib/woods/console/server.rb +18 -16
- data/lib/woods/console/sql_noise_stripper.rb +9 -7
- data/lib/woods/console/sql_table_scanner.rb +47 -7
- data/lib/woods/console/sql_validator.rb +49 -9
- data/lib/woods/console/sqlite_read_guard.rb +46 -0
- data/lib/woods/coordination/pipeline_lock.rb +3 -2
- data/lib/woods/dependency_graph.rb +65 -13
- data/lib/woods/embedding/corpus.rb +94 -0
- data/lib/woods/embedding/indexer.rb +114 -60
- data/lib/woods/embedding/openai.rb +17 -6
- data/lib/woods/evaluation/ablation_executor.rb +6 -1
- data/lib/woods/evaluation/ablation_timed_executor.rb +22 -4
- data/lib/woods/export/typed_reader.rb +56 -0
- data/lib/woods/extractor.rb +277 -149
- data/lib/woods/extractors/action_cable_extractor.rb +3 -1
- data/lib/woods/extractors/behavioral_profile.rb +9 -7
- data/lib/woods/extractors/caching_extractor.rb +3 -1
- data/lib/woods/extractors/concern_extractor.rb +64 -6
- data/lib/woods/extractors/configuration_extractor.rb +7 -3
- data/lib/woods/extractors/controller_extractor.rb +13 -4
- data/lib/woods/extractors/database_view_extractor.rb +3 -1
- data/lib/woods/extractors/declared_parent.rb +55 -0
- data/lib/woods/extractors/decorator_extractor.rb +3 -1
- data/lib/woods/extractors/engine_extractor.rb +3 -1
- data/lib/woods/extractors/event_extractor.rb +4 -2
- data/lib/woods/extractors/factory_extractor.rb +3 -1
- data/lib/woods/extractors/graphql_extractor.rb +10 -13
- data/lib/woods/extractors/i18n_extractor.rb +3 -1
- data/lib/woods/extractors/job_extractor.rb +6 -19
- data/lib/woods/extractors/lib_extractor.rb +13 -9
- data/lib/woods/extractors/mailer_extractor.rb +26 -15
- data/lib/woods/extractors/manager_extractor.rb +3 -1
- data/lib/woods/extractors/method_parameters.rb +53 -0
- data/lib/woods/extractors/middleware_argument.rb +65 -0
- data/lib/woods/extractors/middleware_extractor.rb +9 -3
- data/lib/woods/extractors/migration_extractor.rb +3 -1
- data/lib/woods/extractors/model_extractor.rb +26 -34
- data/lib/woods/extractors/package_extractor.rb +24 -4
- data/lib/woods/extractors/phlex_extractor.rb +3 -1
- data/lib/woods/extractors/policy_extractor.rb +3 -1
- data/lib/woods/extractors/poro_extractor.rb +13 -9
- data/lib/woods/extractors/pundit_extractor.rb +3 -1
- data/lib/woods/extractors/rails_source_extractor.rb +4 -2
- data/lib/woods/extractors/rake_task_extractor.rb +4 -2
- data/lib/woods/extractors/route_extractor.rb +3 -1
- data/lib/woods/extractors/route_helper_resolver.rb +10 -33
- data/lib/woods/extractors/scheduled_job_extractor.rb +41 -15
- data/lib/woods/extractors/serializer_extractor.rb +4 -2
- data/lib/woods/extractors/service_extractor.rb +3 -1
- data/lib/woods/extractors/shared_dependency_scanner.rb +2 -2
- data/lib/woods/extractors/shared_utility_methods.rb +48 -19
- data/lib/woods/extractors/source_nesting.rb +1 -1
- data/lib/woods/extractors/state_machine_extractor.rb +3 -1
- data/lib/woods/extractors/test_mapping_extractor.rb +3 -1
- data/lib/woods/extractors/validator_extractor.rb +3 -1
- data/lib/woods/extractors/view_component_extractor.rb +3 -1
- data/lib/woods/extractors/view_template_extractor.rb +3 -1
- data/lib/woods/gem_mapper.rb +2 -0
- data/lib/woods/git_history.rb +116 -0
- data/lib/woods/graph_analyzer.rb +35 -6
- data/lib/woods/hooks/context_cli.rb +54 -0
- data/lib/woods/hooks/context_event.rb +88 -0
- data/lib/woods/hooks/context_hint.rb +73 -0
- data/lib/woods/hooks/context_impact.rb +77 -0
- data/lib/woods/hooks/context_output.rb +47 -0
- data/lib/woods/hooks/context_state.rb +102 -0
- data/lib/woods/hooks/refresh.rb +79 -0
- data/lib/woods/hooks/rule_projection.rb +78 -0
- data/lib/woods/input_rules.rb +19 -0
- data/lib/woods/mcp/bearer_auth.rb +22 -13
- data/lib/woods/mcp/bootstrapper.rb +79 -4
- data/lib/woods/mcp/config_resolver.rb +2 -1
- data/lib/woods/mcp/index_reader.rb +334 -162
- data/lib/woods/mcp/initialization_guidance.rb +27 -0
- data/lib/woods/mcp/origin_guard.rb +17 -9
- data/lib/woods/mcp/published_lexical_retriever.rb +115 -0
- data/lib/woods/mcp/renderers/markdown_renderer.rb +22 -9
- data/lib/woods/mcp/renderers/plain_renderer.rb +18 -8
- data/lib/woods/mcp/search_results.rb +74 -0
- data/lib/woods/mcp/server.rb +178 -63
- data/lib/woods/mcp/tool_contract.rb +3 -1
- data/lib/woods/mcp/tool_response_renderer.rb +41 -0
- data/lib/woods/mcp/traversal_evidence.rb +113 -0
- data/lib/woods/mcp/traversal_evidence_index.rb +100 -0
- data/lib/woods/mcp/traversal_evidence_page.rb +41 -0
- data/lib/woods/mcp/traversal_evidence_text.rb +52 -0
- data/lib/woods/mcp/traversal_response.rb +22 -0
- data/lib/woods/notion/exporter.rb +56 -17
- data/lib/woods/obsidian/destination_plan.rb +98 -0
- data/lib/woods/obsidian/name_mapper.rb +19 -3
- data/lib/woods/obsidian/note_builder.rb +19 -10
- data/lib/woods/obsidian/vault_exporter.rb +88 -32
- data/lib/woods/operator/pipeline_guard.rb +18 -13
- data/lib/woods/path_dispatcher.rb +13 -6
- data/lib/woods/payload_store.rb +27 -26
- data/lib/woods/published_index/typed_unit_reader.rb +40 -3
- data/lib/woods/published_index.rb +2 -2
- data/lib/woods/railtie.rb +3 -3
- data/lib/woods/railtie_support.rb +12 -12
- data/lib/woods/rake_helpers.rb +382 -0
- data/lib/woods/resilience/graph_invariant_validator/membership_checks.rb +71 -0
- data/lib/woods/resilience/graph_invariant_validator/node_checks.rb +61 -0
- data/lib/woods/resilience/graph_invariant_validator/reverse_relationship_checks.rb +46 -0
- data/lib/woods/resilience/graph_invariant_validator.rb +119 -0
- data/lib/woods/resilience/index_validator/graph_checks.rb +80 -0
- data/lib/woods/resilience/index_validator.rb +112 -23
- data/lib/woods/retrieval/context_assembler.rb +50 -15
- data/lib/woods/retrieval/lexical_assembler.rb +84 -0
- data/lib/woods/retrieval/lexical_index.rb +120 -0
- data/lib/woods/retrieval/ranker.rb +4 -2
- data/lib/woods/retrieval/scope.rb +108 -0
- data/lib/woods/retrieval/scoped_graph_store.rb +32 -0
- data/lib/woods/retrieval/scoped_vector_store.rb +55 -0
- data/lib/woods/retrieval/search_executor.rb +86 -27
- data/lib/woods/retrieval/source_evidence.rb +200 -0
- data/lib/woods/retriever.rb +98 -22
- data/lib/woods/ruby_analyzer/trace_enricher.rb +77 -38
- data/lib/woods/session_tracer/file_store.rb +6 -1
- data/lib/woods/session_tracer/middleware.rb +10 -12
- data/lib/woods/session_tracer/redis_store.rb +22 -6
- data/lib/woods/session_tracer/session_flow_assembler.rb +23 -17
- data/lib/woods/session_tracer/solid_cache_coordination.rb +6 -4
- data/lib/woods/session_tracer/unit_resolver.rb +63 -0
- data/lib/woods/source_inputs/consumer_errors.rb +31 -0
- data/lib/woods/source_inputs/handoff.rb +102 -0
- data/lib/woods/source_inputs/launcher.rb +157 -0
- data/lib/woods/source_inputs/manifest.rb +124 -0
- data/lib/woods/source_inputs/private_key.rb +55 -0
- data/lib/woods/source_inputs/scanner.rb +171 -0
- data/lib/woods/source_inputs/scopes.rb +71 -0
- data/lib/woods/source_inputs/session.rb +214 -0
- data/lib/woods/source_inputs/status.rb +84 -0
- data/lib/woods/source_inputs/verifier.rb +107 -0
- data/lib/woods/storage/metadata_store.rb +25 -25
- data/lib/woods/storage/pgvector.rb +35 -10
- data/lib/woods/storage/qdrant.rb +17 -7
- data/lib/woods/storage/vector_store.rb +18 -6
- data/lib/woods/tasks.rb +3 -2
- data/lib/woods/temporal/json_snapshot_store.rb +58 -9
- data/lib/woods/unblocked/exporter.rb +59 -70
- data/lib/woods/version.rb +1 -1
- data/lib/woods/watch/boot_snapshot.rb +52 -0
- data/lib/woods/watch/daemon.rb +154 -32
- data/lib/woods/watch/listen_watcher.rb +4 -0
- data/lib/woods/watch/polling_watcher.rb +5 -1
- data/lib/woods/watch/status.rb +20 -15
- data/lib/woods/watch/tree_scan.rb +21 -13
- data/lib/woods/watch/watcher.rb +4 -1
- data/lib/woods.rb +50 -11
- data/plugin/.claude-plugin/plugin.json +1 -1
- data/plugin/hooks/adapters/normalize.jq +15 -0
- data/plugin/hooks/adapters/normalize.rb +63 -0
- data/plugin/hooks/hooks.json +20 -0
- data/plugin/hooks/woods-context.sh +50 -0
- data/plugin/hooks/woods-input-rules.sh +159 -0
- data/plugin/hooks/woods-opencode.mjs +65 -0
- data/plugin/hooks/woods-post-edit.sh +2 -225
- data/plugin/hooks/woods-refresh.sh +260 -0
- data/plugin/hooks/woods-session-start.sh +47 -55
- data/plugin/skills/woods-agent-enable/SKILL.md +19 -0
- data/plugin/skills/woods-diagnose/SKILL.md +319 -1
- data/plugin/skills/woods-investigate/SKILL.md +145 -0
- data/plugin/skills/woods-mcp-config/SKILL.md +90 -2
- data/plugin/skills/woods-setup/SKILL.md +110 -6
- metadata +87 -5
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
# frozen_string_literal: true
|
|
2
2
|
|
|
3
3
|
require 'woods/console/sql_noise_stripper'
|
|
4
|
+
require 'woods/console/sqlite_read_guard'
|
|
4
5
|
|
|
5
6
|
# @see Woods
|
|
6
7
|
module Woods
|
|
@@ -28,11 +29,11 @@ module Woods
|
|
|
28
29
|
# named constants), not imperative logic.
|
|
29
30
|
class SqlValidator # rubocop:disable Metrics/ClassLength
|
|
30
31
|
# SQL dialects whose normalization the lock-clause check can run over.
|
|
31
|
-
KNOWN_DIALECTS = %i[postgres mysql].freeze
|
|
32
|
+
KNOWN_DIALECTS = %i[postgres mysql sqlite].freeze
|
|
32
33
|
|
|
33
34
|
# The dialect the statement will execute under, when the caller knows
|
|
34
35
|
# it. `nil` (the default) keeps the conservative union: every check
|
|
35
|
-
# runs against
|
|
36
|
+
# runs against all supported dialect normalizations, which can reject a
|
|
36
37
|
# statement that is valid under one dialect's quote grammar (a MySQL
|
|
37
38
|
# `\'` escape hides prose that the PostgreSQL view reads as SQL). When
|
|
38
39
|
# the execution boundary knows the adapter, passing the matching
|
|
@@ -206,9 +207,9 @@ module Woods
|
|
|
206
207
|
# function calls.
|
|
207
208
|
FUNCTION_SCAN_EXCLUDED_KEYWORDS = %w[
|
|
208
209
|
IN EXISTS NOT AND OR VALUES WHERE HAVING ON IS BETWEEN CASE WHEN
|
|
209
|
-
THEN ELSE
|
|
210
|
-
UNION INTERSECT EXCEPT ORDER GROUP BY
|
|
211
|
-
OVER
|
|
210
|
+
THEN ELSE FROM JOIN USING SELECT DISTINCT ALL ANY SOME
|
|
211
|
+
UNION INTERSECT EXCEPT ORDER GROUP BY LIMIT OFFSET AS INTO
|
|
212
|
+
OVER FILTER RETURNING EXPLAIN
|
|
212
213
|
].freeze
|
|
213
214
|
|
|
214
215
|
# EXPLAIN is a statement leader, so `EXPLAIN (FORMAT JSON) SELECT` is an
|
|
@@ -318,6 +319,7 @@ module Woods
|
|
|
318
319
|
|
|
319
320
|
return validate_dialect_variants!(sql) if unknown_grammar?
|
|
320
321
|
|
|
322
|
+
SqliteReadGuard.validate!(sql) if dialect == :sqlite
|
|
321
323
|
normalized = sql.strip
|
|
322
324
|
|
|
323
325
|
# Reject multiple statements (semicolons not inside string literals)
|
|
@@ -582,10 +584,9 @@ module Woods
|
|
|
582
584
|
match = Regexp.last_match
|
|
583
585
|
quoted = !(match[1] || match[2]).nil?
|
|
584
586
|
identifier = match[1] || match[2] || match[3]
|
|
585
|
-
#
|
|
586
|
-
#
|
|
587
|
-
|
|
588
|
-
next if !quoted && FUNCTION_SCAN_EXCLUDED_KEYWORDS.include?(identifier.upcase)
|
|
587
|
+
# Some engines also allow keyword spellings as function names.
|
|
588
|
+
# Only exempt those names in their supported grammar positions.
|
|
589
|
+
next if !quoted && function_keyword_grammar?(identifier.upcase, stripped[0...match.begin(0)])
|
|
589
590
|
next if ALLOWED_FUNCTIONS.include?(identifier.downcase)
|
|
590
591
|
|
|
591
592
|
raise SqlValidationError,
|
|
@@ -594,6 +595,45 @@ module Woods
|
|
|
594
595
|
end
|
|
595
596
|
end
|
|
596
597
|
|
|
598
|
+
# Recognize keyword grammar without exempting identically named calls.
|
|
599
|
+
# The prefix is already stripped of comments and literal contents.
|
|
600
|
+
#
|
|
601
|
+
# @param keyword [String] Uppercase bare identifier before `(`.
|
|
602
|
+
# @param prefix [String] SQL preceding the identifier.
|
|
603
|
+
# @return [Boolean]
|
|
604
|
+
def function_keyword_grammar?(keyword, prefix)
|
|
605
|
+
return false unless FUNCTION_SCAN_EXCLUDED_KEYWORDS.include?(keyword)
|
|
606
|
+
|
|
607
|
+
case keyword
|
|
608
|
+
when 'EXPLAIN' then prefix.strip.empty?
|
|
609
|
+
when 'BY' then prefix.match?(/\b(?:ORDER|GROUP|PARTITION)\s*\z/i)
|
|
610
|
+
when 'OVER', 'FILTER' then prefix.match?(/\)\s*\z/)
|
|
611
|
+
when 'ANY', 'SOME' then comparison_quantifier?(prefix)
|
|
612
|
+
when 'OFFSET' then offset_grammar?(prefix)
|
|
613
|
+
else true
|
|
614
|
+
end
|
|
615
|
+
end
|
|
616
|
+
|
|
617
|
+
# SQLite has no quantified ANY/SOME comparison grammar; both spellings
|
|
618
|
+
# can instead invoke application-defined functions after an operator.
|
|
619
|
+
def comparison_quantifier?(prefix)
|
|
620
|
+
return true if dialect == :postgres # Reserved quantifiers, including LIKE ANY.
|
|
621
|
+
|
|
622
|
+
dialect != :sqlite && prefix.match?(/[=<>]\s*\z/)
|
|
623
|
+
end
|
|
624
|
+
|
|
625
|
+
# PostgreSQL reserves bare OFFSET (it cannot name a function), so its
|
|
626
|
+
# standalone OFFSET clause remains supported. Other dialects only accept
|
|
627
|
+
# parenthesized offsets after a completed LIMIT operand. Calls
|
|
628
|
+
# at an expression's start or after an operator are never offset syntax.
|
|
629
|
+
# Keep the supported operand boundary conservative: literals, bind
|
|
630
|
+
# numbers, parenthesized expressions, or PostgreSQL's LIMIT ALL.
|
|
631
|
+
def offset_grammar?(prefix)
|
|
632
|
+
return true if dialect == :postgres
|
|
633
|
+
|
|
634
|
+
prefix.match?(/(?:[0-9)'"`]|\bLIMIT\s+ALL)\s*\z/i)
|
|
635
|
+
end
|
|
636
|
+
|
|
597
637
|
# Check if the SQL contains a forbidden keyword at a statement-leader
|
|
598
638
|
# position after stripping comments and string literals. This catches
|
|
599
639
|
# comment-hidden injections like "SELECT 1 --;\nDELETE FROM users",
|
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require 'woods/console/sql_noise_stripper'
|
|
4
|
+
require 'woods/console/sql_table_scanner'
|
|
5
|
+
|
|
6
|
+
module Woods
|
|
7
|
+
module Console
|
|
8
|
+
# Refuse SQLite-specific syntax outside the scanner's supported grammar.
|
|
9
|
+
# This is a conservative read boundary, not a general-purpose SQL parser.
|
|
10
|
+
module SqliteReadGuard
|
|
11
|
+
SIMPLE_IDENTIFIER = /\A[A-Za-z_][A-Za-z0-9_]*\z/
|
|
12
|
+
UNSUPPORTED_SYNTAX = /[^\x00-\x7F]|\$|\[|''\s*\(|\b(?:FROM|JOIN)(?=['"`(])/i
|
|
13
|
+
QUOTED_IDENTIFIER = /"(?:[^"]|"")*"|`(?:[^`]|``)*`/
|
|
14
|
+
|
|
15
|
+
# @param sql [String] SQL to execute on SQLite
|
|
16
|
+
# @raise [SqlValidationError] when an identifier or table factor cannot be checked
|
|
17
|
+
# @return [void]
|
|
18
|
+
def self.validate!(sql)
|
|
19
|
+
view = SqlNoiseStripper.strip_noise(sql, dialect: :sqlite)
|
|
20
|
+
refuse! if view.match?(UNSUPPORTED_SYNTAX)
|
|
21
|
+
view.scan(QUOTED_IDENTIFIER) { |quoted| refuse! unless quoted[1...-1].match?(SIMPLE_IDENTIFIER) }
|
|
22
|
+
SqlTableScanner.relation_factors(view).each do |factor|
|
|
23
|
+
refuse! unless supported_factor?(factor.strip)
|
|
24
|
+
end
|
|
25
|
+
end
|
|
26
|
+
|
|
27
|
+
# A SELECT/WITH subquery has its own independently scanned table factors.
|
|
28
|
+
# Parenthesized table groups and string-quoted names are refused rather
|
|
29
|
+
# than silently omitted from the blocked-table scan.
|
|
30
|
+
def self.supported_factor?(factor)
|
|
31
|
+
return true if factor.match?(/\A\(\s*(?:SELECT|WITH)\b/i)
|
|
32
|
+
|
|
33
|
+
match = SqlTableScanner::LEAD_IDENT.match(factor)
|
|
34
|
+
match && factor[match.end(0)..].match?(/\A(?:\s|,|\)|\z)/)
|
|
35
|
+
end
|
|
36
|
+
private_class_method :supported_factor?
|
|
37
|
+
|
|
38
|
+
def self.refuse!
|
|
39
|
+
raise SqlValidationError,
|
|
40
|
+
'Rejected: unsupported SQLite identifier or table-reference syntax. ' \
|
|
41
|
+
'Use simple bare or double-quoted identifiers and SELECT subqueries.'
|
|
42
|
+
end
|
|
43
|
+
private_class_method :refuse!
|
|
44
|
+
end
|
|
45
|
+
end
|
|
46
|
+
end
|
|
@@ -30,8 +30,9 @@ module Woods
|
|
|
30
30
|
# The transaction-guard filename for a lock of +name+, exposed so
|
|
31
31
|
# cleanup code that empties a lock directory (woods:clean) can skip the
|
|
32
32
|
# guard during its sweep — deleting a flock'd guard out from under a
|
|
33
|
-
# contender's critical section would split the flock across two inodes
|
|
34
|
-
#
|
|
33
|
+
# contender's critical section would split the flock across two inodes.
|
|
34
|
+
# The guard stays after cleanup and lock release: a new contender may
|
|
35
|
+
# already be holding it before its lock file exists.
|
|
35
36
|
#
|
|
36
37
|
# @param name [String]
|
|
37
38
|
# @return [String]
|
|
@@ -32,9 +32,14 @@ module Woods
|
|
|
32
32
|
# enforce_dependencies: package units (Task 7)
|
|
33
33
|
# commit_count, change_frequency: git enrichment (Task 5)
|
|
34
34
|
NODE_ATTRIBUTE_KEYS = %i[
|
|
35
|
-
database table foreign_key_tables package enforce_dependencies commit_count change_frequency
|
|
35
|
+
database table foreign_key_tables package enforce_dependencies commit_count change_frequency kind
|
|
36
36
|
].freeze
|
|
37
37
|
|
|
38
|
+
# These extractors describe a whole source file rather than one constant.
|
|
39
|
+
# Other types remain unclassified: a path-like identifier is not evidence
|
|
40
|
+
# that a unit is a profile, and an unmarked unit need not name a constant.
|
|
41
|
+
FILE_PROFILE_TYPES = %i[caching configuration test_mapping rails_source gem_source].freeze
|
|
42
|
+
|
|
38
43
|
# Optional edge keys beyond target and via. `through` is the through
|
|
39
44
|
# association name; `through_db` is the through model's resolved
|
|
40
45
|
# database, emitted only when present; `disable_joins` is emitted only
|
|
@@ -80,14 +85,14 @@ module Woods
|
|
|
80
85
|
}.merge(node_attributes_from(unit))
|
|
81
86
|
|
|
82
87
|
(@edges[unit.identifier] ||= {})[unit.type] =
|
|
83
|
-
unit.dependencies
|
|
88
|
+
self.class.normalize_edges(unit.dependencies, strict: true)
|
|
84
89
|
(@file_map[unit.file_path] ||= Set.new).add(unit.identifier) if unit.file_path
|
|
85
90
|
|
|
86
91
|
# Type index for filtering (Set-based for O(1) insert)
|
|
87
92
|
(@type_index[unit.type] ||= Set.new).add(unit.identifier)
|
|
88
93
|
|
|
89
94
|
# Build reverse edges (Set-based for O(1) insert)
|
|
90
|
-
unit.
|
|
95
|
+
@edges[unit.identifier][unit.type].each do |dep|
|
|
91
96
|
(@reverse[dep[:target]] ||= Set.new).add(unit.identifier)
|
|
92
97
|
(@reverse_via[[dep[:target], dep[:via]]] ||= Set.new).add(unit.identifier)
|
|
93
98
|
end
|
|
@@ -645,7 +650,7 @@ module Woods
|
|
|
645
650
|
# @return [Hash{Symbol => Object}] normalized, nil-free
|
|
646
651
|
def node_attributes_from(unit)
|
|
647
652
|
metadata = unit.respond_to?(:metadata) && unit.metadata.is_a?(Hash) ? unit.metadata : {}
|
|
648
|
-
attrs =
|
|
653
|
+
attrs = self.class.file_profile_attributes(unit.type)
|
|
649
654
|
attrs[:database] = metadata[:database] unless metadata[:database].nil?
|
|
650
655
|
attrs[:table] = metadata[:table_name] if unit.type == :model && !metadata[:table_name].nil?
|
|
651
656
|
tables = Array(metadata[:foreign_keys]).filter_map { |fk| fk[:to_table] || fk['to_table'] if fk.is_a?(Hash) }
|
|
@@ -729,14 +734,15 @@ module Woods
|
|
|
729
734
|
# `nodes` and `edges` stay keyed on the bare identifier, holding the
|
|
730
735
|
# **primary** node — the type that sorts first. Identifiers registered
|
|
731
736
|
# under more than one type put their remaining nodes in `variants`, a flat
|
|
732
|
-
# array that is **omitted entirely when empty**.
|
|
733
|
-
#
|
|
734
|
-
# a `dependency_graph.json` written before this change loads through
|
|
737
|
+
# array that is **omitted entirely when empty**. A
|
|
738
|
+
# `dependency_graph.json` written before this change loads through
|
|
735
739
|
# {.from_h} unmodified: it simply carries no `variants`.
|
|
736
740
|
#
|
|
737
741
|
# `reverse`, `file_map` and `type_index` need no new shape. They are
|
|
738
742
|
# already identifier-valued sets, so a Scenic view `reports` and a factory
|
|
739
743
|
# `reports` each contribute to their own type bucket and their own path.
|
|
744
|
+
# The additive `reverse_via` index preserves typed source identities and
|
|
745
|
+
# complete relationship records without changing the legacy `reverse` map.
|
|
740
746
|
#
|
|
741
747
|
# @return [Hash] Complete graph data
|
|
742
748
|
def to_h
|
|
@@ -747,6 +753,7 @@ module Woods
|
|
|
747
753
|
nodes: key_sorted(self.class.relativize_nodes(primary_nodes, root)),
|
|
748
754
|
edges: key_sorted(primary_edges),
|
|
749
755
|
reverse: key_sorted(@reverse.transform_values { |ids| ids.to_a.sort }),
|
|
756
|
+
reverse_via: published_reverse_via,
|
|
750
757
|
file_map: key_sorted(self.class.relativize_file_map(@file_map, root)),
|
|
751
758
|
type_index: key_sorted(@type_index.transform_values { |ids| ids.to_a.sort }),
|
|
752
759
|
stats: {
|
|
@@ -770,6 +777,23 @@ module Woods
|
|
|
770
777
|
end
|
|
771
778
|
private :key_sorted
|
|
772
779
|
|
|
780
|
+
# Reverse buckets are derived from every typed forward registration, rather
|
|
781
|
+
# than the bare-name in-memory filter index, which cannot preserve variants
|
|
782
|
+
# or distinguish association attributes. Legacy nil relationships stay nil.
|
|
783
|
+
def published_reverse_via
|
|
784
|
+
buckets = {}
|
|
785
|
+
@edges.each do |source, by_type|
|
|
786
|
+
by_type.each do |type, edges|
|
|
787
|
+
edges.each do |edge|
|
|
788
|
+
record = { source: source, source_type: type }.merge(edge.except(:target))
|
|
789
|
+
(buckets[edge[:target]] ||= []) << record
|
|
790
|
+
end
|
|
791
|
+
end
|
|
792
|
+
end
|
|
793
|
+
key_sorted(buckets.transform_values { |records| records.sort_by { |record| JSON.generate(record) } })
|
|
794
|
+
end
|
|
795
|
+
private :published_reverse_via
|
|
796
|
+
|
|
773
797
|
# A copy whose edge containers are the caller's alone.
|
|
774
798
|
#
|
|
775
799
|
# `dup` copies only the top level, so `to_h[:edges][id]` used to be the
|
|
@@ -783,6 +807,9 @@ module Woods
|
|
|
783
807
|
def detached_snapshot(memo)
|
|
784
808
|
snapshot = memo.dup
|
|
785
809
|
snapshot[:edges] = memo[:edges].transform_values { |list| list.map(&:dup) }
|
|
810
|
+
snapshot[:reverse_via] = memo[:reverse_via].transform_values do |records|
|
|
811
|
+
records.map { |record| record.transform_values { |value| value.is_a?(String) ? value.dup : value } }
|
|
812
|
+
end
|
|
786
813
|
if memo.key?(:variants)
|
|
787
814
|
snapshot[:variants] = memo[:variants].map do |record|
|
|
788
815
|
record.merge(edges: record[:edges].map(&:dup))
|
|
@@ -794,7 +821,12 @@ module Woods
|
|
|
794
821
|
|
|
795
822
|
# @return [Hash{String => Hash}] identifier => primary node
|
|
796
823
|
def primary_nodes
|
|
797
|
-
|
|
824
|
+
# Late enrichment and JSON loading can insert optional fields in a
|
|
825
|
+
# different order. Canonicalize like variant_records so a graph's JSON
|
|
826
|
+
# fingerprint is stable across full and incremental publication.
|
|
827
|
+
@nodes.transform_values do |nodes|
|
|
828
|
+
primary_of(nodes).slice(:type, :file_path, :namespace, *NODE_ATTRIBUTE_KEYS)
|
|
829
|
+
end
|
|
798
830
|
end
|
|
799
831
|
private :primary_nodes
|
|
800
832
|
|
|
@@ -969,10 +1001,17 @@ module Woods
|
|
|
969
1001
|
# @param node [Hash]
|
|
970
1002
|
# @return [Hash{Symbol => Object}]
|
|
971
1003
|
def self.persisted_node_attributes(node)
|
|
972
|
-
NODE_ATTRIBUTE_KEYS.each_with_object({}) do |key, attrs|
|
|
1004
|
+
attributes = NODE_ATTRIBUTE_KEYS.each_with_object({}) do |key, attrs|
|
|
973
1005
|
value = node.key?(key) ? node[key] : node[key.to_s]
|
|
974
1006
|
attrs[key] = normalize_node_attribute(key, value) unless value.nil?
|
|
975
1007
|
end
|
|
1008
|
+
attributes.merge(file_profile_attributes(node[:type] || node['type']))
|
|
1009
|
+
end
|
|
1010
|
+
|
|
1011
|
+
# Backfill the additive marker when an older graph is republished, so
|
|
1012
|
+
# unchanged nodes in an incremental run agree with a fresh full graph.
|
|
1013
|
+
def self.file_profile_attributes(type)
|
|
1014
|
+
FILE_PROFILE_TYPES.include?(type&.to_sym) ? { kind: 'file_profile' } : {}
|
|
976
1015
|
end
|
|
977
1016
|
|
|
978
1017
|
# Edge attributes present in a dependency or persisted edge hash.
|
|
@@ -1032,7 +1071,10 @@ module Woods
|
|
|
1032
1071
|
graph.instance_variable_set(:@edges, edges)
|
|
1033
1072
|
|
|
1034
1073
|
raw_reverse = data[:reverse] || data['reverse'] || {}
|
|
1035
|
-
|
|
1074
|
+
reverse = raw_reverse.each_with_object({}) do |(target, sources), index|
|
|
1075
|
+
(index[target.to_s] ||= Set.new).merge(sources)
|
|
1076
|
+
end
|
|
1077
|
+
graph.instance_variable_set(:@reverse, reverse)
|
|
1036
1078
|
|
|
1037
1079
|
raw_file_map = data[:file_map] || data['file_map'] || {}
|
|
1038
1080
|
graph.instance_variable_set(:@file_map, absolutize_file_map(normalize_file_map(raw_file_map), root))
|
|
@@ -1126,7 +1168,12 @@ module Woods
|
|
|
1126
1168
|
}.merge(persisted_node_attributes(node))
|
|
1127
1169
|
end
|
|
1128
1170
|
|
|
1129
|
-
# Normalize
|
|
1171
|
+
# Normalize fresh dependencies and persisted edges to the same identity:
|
|
1172
|
+
# string targets and symbol relationship labels. Extractors may emit
|
|
1173
|
+
# symbolic external targets (such as :http_api); retaining those symbols
|
|
1174
|
+
# after a JSON-loaded baseline splits reverse indexes on re-registration.
|
|
1175
|
+
# Unit dependency metadata stays unchanged; only graph records normalize.
|
|
1176
|
+
# Also accepts the old bare-string edge format.
|
|
1130
1177
|
#
|
|
1131
1178
|
# ROUND-TRIP INVARIANT (do not break when refactoring):
|
|
1132
1179
|
# DependencyGraph#to_h -> JSON.generate -> JSON.parse -> DependencyGraph.from_h
|
|
@@ -1147,15 +1194,20 @@ module Woods
|
|
|
1147
1194
|
# explicit data conversion.
|
|
1148
1195
|
#
|
|
1149
1196
|
# @param edges [Array] Edge entries — either strings or hashes
|
|
1197
|
+
# @param strict [Boolean] require hashes for fresh units; legacy loads stay tolerant
|
|
1150
1198
|
# @return [Array<Hash>] Normalized edges with :target and :via keys
|
|
1151
|
-
def self.normalize_edges(edges)
|
|
1199
|
+
def self.normalize_edges(edges, strict: false)
|
|
1200
|
+
raise ArgumentError, 'Fresh graph dependencies must be an array' if strict && !edges.is_a?(Array)
|
|
1201
|
+
|
|
1152
1202
|
return [] unless edges.is_a?(Array)
|
|
1153
1203
|
|
|
1154
1204
|
edges.map do |edge|
|
|
1205
|
+
raise ArgumentError, 'Fresh graph dependencies must be hashes' if strict && !edge.is_a?(Hash)
|
|
1206
|
+
|
|
1155
1207
|
if edge.is_a?(String)
|
|
1156
1208
|
{ target: edge, via: nil }
|
|
1157
1209
|
elsif edge.is_a?(Hash)
|
|
1158
|
-
{ target: edge[:target] || edge['target'], via: (edge[:via] || edge['via'])&.to_sym }
|
|
1210
|
+
{ target: (edge[:target] || edge['target'])&.to_s, via: (edge[:via] || edge['via'])&.to_sym }
|
|
1159
1211
|
.merge(edge_attributes(edge))
|
|
1160
1212
|
else
|
|
1161
1213
|
{ target: edge.to_s, via: nil }
|
|
@@ -0,0 +1,94 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require 'json'
|
|
4
|
+
require_relative '../atomic_file'
|
|
5
|
+
require_relative '../generation'
|
|
6
|
+
|
|
7
|
+
module Woods
|
|
8
|
+
class Error < StandardError; end unless defined?(Woods::Error)
|
|
9
|
+
|
|
10
|
+
module Embedding
|
|
11
|
+
# Collect the complete extraction input before an embedding run can mutate
|
|
12
|
+
# stores. Native publications use their authoritative listings, never a glob
|
|
13
|
+
# that could mistake a missing/corrupt payload for intentional deletion.
|
|
14
|
+
class Corpus
|
|
15
|
+
class Incomplete < Woods::Error; end
|
|
16
|
+
|
|
17
|
+
def initialize(output_dir)
|
|
18
|
+
@output_dir = output_dir.to_s
|
|
19
|
+
end
|
|
20
|
+
|
|
21
|
+
def load
|
|
22
|
+
native? ? published_units : legacy_units
|
|
23
|
+
rescue IOError, SystemCallError, JSON::ParserError, EncodingError, ArgumentError => e
|
|
24
|
+
raise Incomplete, "Embedding input incomplete: #{e.message}. Repair or rebuild extraction before embedding."
|
|
25
|
+
end
|
|
26
|
+
|
|
27
|
+
private
|
|
28
|
+
|
|
29
|
+
def native?
|
|
30
|
+
path = File.join(@output_dir, Generation::FILENAME)
|
|
31
|
+
if File.exist?(path)
|
|
32
|
+
marker = JSON.parse(AtomicFile.read(path))
|
|
33
|
+
raise IOError, 'invalid generation marker' unless marker.is_a?(Hash)
|
|
34
|
+
end
|
|
35
|
+
|
|
36
|
+
if !marker || marker['payload'].nil?
|
|
37
|
+
if File.directory?(File.join(@output_dir, 'payloads'))
|
|
38
|
+
raise IOError, 'missing publication pointer beside payloads'
|
|
39
|
+
end
|
|
40
|
+
|
|
41
|
+
return false
|
|
42
|
+
end
|
|
43
|
+
|
|
44
|
+
validate_marker(marker)
|
|
45
|
+
true
|
|
46
|
+
end
|
|
47
|
+
|
|
48
|
+
def validate_marker(marker)
|
|
49
|
+
strings_valid = %w[payload token].all? { |key| marker[key].is_a?(String) && !marker[key].empty? }
|
|
50
|
+
return if strings_valid && marker['number'].is_a?(Integer) && marker['number'].positive?
|
|
51
|
+
|
|
52
|
+
raise IOError, 'invalid published generation marker'
|
|
53
|
+
end
|
|
54
|
+
|
|
55
|
+
def published_units
|
|
56
|
+
require_relative '../mcp/index_reader'
|
|
57
|
+
|
|
58
|
+
reader = MCP::IndexReader.new(@output_dir)
|
|
59
|
+
reader.with_pinned_generation do
|
|
60
|
+
# Generation's compatibility fallback to the root is useful for
|
|
61
|
+
# readers, but cannot prove completeness for destructive reconciliation.
|
|
62
|
+
if reader.payload_dir.expand_path == Pathname.new(@output_dir).expand_path
|
|
63
|
+
raise IOError,
|
|
64
|
+
'published payload missing or invalid'
|
|
65
|
+
end
|
|
66
|
+
|
|
67
|
+
validate_counts(reader.manifest)
|
|
68
|
+
reader.each_unit.to_a
|
|
69
|
+
end
|
|
70
|
+
end
|
|
71
|
+
|
|
72
|
+
def validate_counts(manifest)
|
|
73
|
+
counts = manifest.is_a?(Hash) && manifest['counts']
|
|
74
|
+
valid = counts.is_a?(Hash) && counts.all? do |dir, count|
|
|
75
|
+
MCP::IndexReader::TYPE_DIRS.include?(dir) && count.is_a?(Integer) && count >= 0
|
|
76
|
+
end
|
|
77
|
+
raise IOError, 'invalid published manifest counts' unless valid
|
|
78
|
+
end
|
|
79
|
+
|
|
80
|
+
# Pre-pointer indexes may use arbitrary filenames and have no manifest
|
|
81
|
+
# or authoritative listing. Keep that input contract, but never silently
|
|
82
|
+
# omit unreadable JSON before reconciling persisted identities.
|
|
83
|
+
def legacy_units
|
|
84
|
+
Dir.glob(File.join(@output_dir, '**', '*.json')).filter_map do |path|
|
|
85
|
+
relative = path.delete_prefix("#{@output_dir}/")
|
|
86
|
+
next if relative.start_with?('dumps/', 'payloads/') || File.basename(path) == 'checkpoint.json'
|
|
87
|
+
|
|
88
|
+
data = JSON.parse(AtomicFile.read(path))
|
|
89
|
+
data if data.is_a?(Hash) && data.key?('type') && data.key?('identifier')
|
|
90
|
+
end
|
|
91
|
+
end
|
|
92
|
+
end
|
|
93
|
+
end
|
|
94
|
+
end
|