woods 2.0.0.beta4 → 2.0.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (95) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +495 -471
  3. data/CONTRIBUTING.md +12 -2
  4. data/README.md +11 -26
  5. data/docs/AGENT_GUIDE.md +31 -12
  6. data/docs/AGENT_SETUP.md +17 -10
  7. data/docs/AUTOMATIC_MAINTENANCE.md +222 -0
  8. data/docs/BACKEND_MATRIX.md +13 -7
  9. data/docs/CLIENT_HOOKS.md +1 -1
  10. data/docs/CONFIGURATION_REFERENCE.md +44 -27
  11. data/docs/CONSOLE_MCP_SETUP.md +95 -16
  12. data/docs/DOCKER_SETUP.md +15 -0
  13. data/docs/EVALUATION.md +10 -4
  14. data/docs/EXTRACTOR_REFERENCE.md +14 -2
  15. data/docs/FAQ.md +14 -3
  16. data/docs/GETTING_STARTED.md +18 -17
  17. data/docs/INCREMENTAL_EXTRACTION.md +8 -3
  18. data/docs/INDEX_LAYOUT.md +2 -2
  19. data/docs/MCP_HTTP_TRANSPORT.md +54 -2
  20. data/docs/MCP_SERVERS.md +28 -11
  21. data/docs/MCP_TOOL_COOKBOOK.md +1 -1
  22. data/docs/MCP_WORKTREE_SETUP.md +13 -1
  23. data/docs/PUBLISHED_INDEX.md +1 -1
  24. data/docs/README.md +2 -1
  25. data/docs/RETRIEVAL_GUIDE.md +57 -8
  26. data/docs/SOURCE_FRESHNESS.md +1 -1
  27. data/docs/TOKEN_BENCHMARK.md +16 -10
  28. data/docs/TROUBLESHOOTING.md +133 -37
  29. data/docs/UPGRADING_TO_2.md +69 -7
  30. data/docs/WATCH_DAEMON.md +172 -17
  31. data/docs/WHY_WOODS.md +9 -5
  32. data/exe/woods-console +13 -11
  33. data/exe/woods-mcp-http +16 -9
  34. data/exe/woods-watch +5 -0
  35. data/lib/generators/woods/watch_generator.rb +53 -0
  36. data/lib/puma/plugin/woods.rb +10 -0
  37. data/lib/tasks/woods.rake +14 -0
  38. data/lib/woods/cache/cache_middleware.rb +18 -11
  39. data/lib/woods/console/adapter_family.rb +39 -0
  40. data/lib/woods/console/credential_index.rb +33 -3
  41. data/lib/woods/console/embedded_executor.rb +401 -43
  42. data/lib/woods/console/model_validator.rb +8 -0
  43. data/lib/woods/console/rack_middleware.rb +39 -10
  44. data/lib/woods/console/redactor.rb +24 -10
  45. data/lib/woods/console/safe_context.rb +44 -7
  46. data/lib/woods/console/sql_noise_stripper.rb +41 -12
  47. data/lib/woods/console/sql_table_scanner.rb +45 -34
  48. data/lib/woods/console/sql_validator.rb +37 -2
  49. data/lib/woods/console/stdio_transport.rb +27 -0
  50. data/lib/woods/extractor.rb +25 -7
  51. data/lib/woods/git_command.rb +6 -7
  52. data/lib/woods/git_provenance.rb +4 -6
  53. data/lib/woods/mcp/bearer_auth.rb +1 -1
  54. data/lib/woods/mcp/bootstrapper.rb +3 -1
  55. data/lib/woods/mcp/initialization_guidance.rb +1 -1
  56. data/lib/woods/mcp/origin_guard.rb +24 -77
  57. data/lib/woods/mcp/origin_policy.rb +124 -0
  58. data/lib/woods/mcp/server.rb +41 -9
  59. data/lib/woods/railtie_support.rb +8 -0
  60. data/lib/woods/retrieval/corpus_status.rb +46 -0
  61. data/lib/woods/retriever.rb +19 -7
  62. data/lib/woods/storage/local_corpus_stats.rb +32 -0
  63. data/lib/woods/storage/metadata_store.rb +20 -0
  64. data/lib/woods/storage/vector_store.rb +10 -0
  65. data/lib/woods/version.rb +1 -1
  66. data/lib/woods/watch/child_environment.rb +30 -0
  67. data/lib/woods/watch/cli.rb +91 -0
  68. data/lib/woods/watch/daemon.rb +55 -7
  69. data/lib/woods/watch/event_stream.rb +70 -0
  70. data/lib/woods/watch/guardian.rb +142 -0
  71. data/lib/woods/watch/installation/layout.rb +70 -0
  72. data/lib/woods/watch/installation/options.rb +128 -0
  73. data/lib/woods/watch/installation/planner.rb +128 -0
  74. data/lib/woods/watch/installation/probe.rb +101 -0
  75. data/lib/woods/watch/installation/receipt.rb +77 -0
  76. data/lib/woods/watch/installation/recovery.rb +64 -0
  77. data/lib/woods/watch/installation/templates.rb +58 -0
  78. data/lib/woods/watch/installation.rb +56 -0
  79. data/lib/woods/watch/lifecycle.rb +182 -0
  80. data/lib/woods/watch/managed_child.rb +113 -0
  81. data/lib/woods/watch/managed_cleanup.rb +48 -0
  82. data/lib/woods/watch/managed_process.rb +144 -0
  83. data/lib/woods/watch/puma_adapter.rb +87 -0
  84. data/lib/woods/watch/puma_child.rb +66 -0
  85. data/lib/woods/watch/supervision_records.rb +95 -0
  86. data/lib/woods/watch/supervision_status.rb +104 -0
  87. data/lib/woods/watch/supervisor.rb +161 -0
  88. data/lib/woods/watch/supervisor_reporting.rb +46 -0
  89. data/plugin/.claude-plugin/plugin.json +1 -1
  90. data/plugin/skills/woods-agent-enable/SKILL.md +1 -1
  91. data/plugin/skills/woods-diagnose/SKILL.md +88 -8
  92. data/plugin/skills/woods-investigate/SKILL.md +6 -6
  93. data/plugin/skills/woods-mcp-config/SKILL.md +43 -1
  94. data/plugin/skills/woods-setup/SKILL.md +66 -4
  95. metadata +37 -5
data/docs/WHY_WOODS.md CHANGED
@@ -56,9 +56,11 @@ It describes what the service does, but misses that `order.save!` triggers `afte
56
56
  :send_confirmation_email` on `Order`, which itself enqueues `InventoryJob` via
57
57
  `after_save :reserve_stock` on `LineItem`.
58
58
 
59
- With Woods, the dependency graph links `CheckoutService` → `Order` → `LineItem` →
60
- `InventoryJob`. A single retrieval call assembles the full execution picture: the service,
61
- the models it touches, the callbacks those models fire, and the jobs those callbacks enqueue.
59
+ With Woods, published graph relationships can connect related services, models,
60
+ callbacks, and jobs for retrieval. Arbitrary method-body constant references are
61
+ not exhaustively indexed, so the whole `CheckoutService` → `Order` → `LineItem` →
62
+ `InventoryJob` chain may not be recorded. Inspect the source and any flow evidence,
63
+ including reported ambiguity and traversal limits, before concluding what runs.
62
64
 
63
65
  ---
64
66
 
@@ -175,11 +177,13 @@ Three tools answered "what is in this Rails app" for coding agents in 2026. Wood
175
177
  | Rubydex (Shopify) | 0.4.1, announced 2026-05-12 | Rust static index of declarations, references, ancestors; experimental `rdx mcp` | Symbol references across a large tree, fast re-index, reported 15 to 80 percent token reduction | Resolved callbacks, inlined concerns, routes as Rails builds them, database partition, churn |
176
178
  | rails-mcp-server | 2.0.0 | Boots the app; `analyze_models`, `get_routes`, `get_schema` | Live model, route, and schema listings over MCP | Callback side effects, request flows, git churn, graph reports, persistent index with generations |
177
179
  | ruby-lsp-rails | 0.5.0.beta1 | Runtime server over `rails runner` for the editor | Model columns, association targets, route info at the cursor | A persistent index other tools can read, graph analysis, multi-database facts |
178
- | Woods | 2.0 | Boots the app once, publishes an atomic JSON generation, serves it without Rails | Resolved runtime behavior on top of structure: inlined concerns, callback side effects, flows, churn, PageRank; database partition and Packwerk boundaries are in progress on this branch | Symbol-level references inside method bodies (Rubydex is the better fit and is complementary) |
180
+ | Woods | 2.0 | Boots the app once, publishes an atomic JSON generation, serves it without Rails | Resolved runtime behavior on top of structure: inlined concerns, callback side effects, flows, churn, PageRank; recorded database partitions and Packwerk boundaries | Symbol-level references inside method bodies (Rubydex is the better fit and is complementary) |
179
181
 
180
182
  Woods and Rubydex are complementary. Rubydex answers "where is this symbol referenced". Woods answers "what happens when this runs, and what does it touch". An agent can use both: Rubydex for references, Woods for behavior, boundaries, and blast radius.
181
183
 
182
- The database-partition layer, in progress on this branch, is the one place Woods will be alone. Rubydex is static, and the other two resolve associations without saying which database each side lives on. See [Extractor reference](EXTRACTOR_REFERENCE.md#modelextractor) for the fields.
184
+ Woods records database-partition metadata on models and Packwerk package ownership
185
+ and boundaries. See [Extractor reference](EXTRACTOR_REFERENCE.md#modelextractor)
186
+ for the fields and their coverage limits.
183
187
 
184
188
  ---
185
189
 
data/exe/woods-console CHANGED
@@ -21,14 +21,14 @@
21
21
  # Check if the rake task already captured stdout for us.
22
22
  protocol_out = $woods_protocol_out # rubocop:disable Style/GlobalVars
23
23
 
24
- unless protocol_out
25
- # Running via rails runner — capture stdout ourselves.
26
- protocol_out = $stdout.dup
27
- $stdout.reopen($stderr)
28
- end
24
+ # Running via rails runner — capture stdout ourselves.
25
+ protocol_out ||= $stdout.dup
26
+ # Keep application logs and writes on stderr for the entire server lifetime.
27
+ $stdout.reopen($stderr)
29
28
 
30
29
  require 'woods'
31
30
  require 'woods/console/server'
31
+ require 'woods/console/stdio_transport'
32
32
 
33
33
  unless Woods.configuration.console_mcp_enabled
34
34
  warn 'Woods Console MCP is disabled. Set ' \
@@ -132,9 +132,11 @@ server = Woods::Console::Server.build_embedded(
132
132
  model_reflections: model_reflections
133
133
  )
134
134
 
135
- # Restore the protocol output for MCP transport.
136
- $stdout.reopen(protocol_out)
137
- protocol_out.close unless protocol_out.closed?
138
-
139
- transport = MCP::Server::Transports::StdioTransport.new(server)
140
- transport.open
135
+ # Only the transport writes to the saved pipe. Restoring process stdout here
136
+ # would also redirect Rails loggers and runtime puts back into the protocol.
137
+ transport = Woods::Console::StdioTransport.new(server, output: protocol_out)
138
+ begin
139
+ transport.open
140
+ ensure
141
+ protocol_out.close unless protocol_out.closed?
142
+ end
data/exe/woods-mcp-http CHANGED
@@ -46,6 +46,20 @@ require_relative '../lib/woods/embedding/text_preparer'
46
46
  require_relative '../lib/woods/embedding/indexer'
47
47
 
48
48
  begin
49
+ raw_origins = ENV.fetch('WOODS_MCP_HTTP_ALLOWED_ORIGINS', '')
50
+ unless raw_origins.valid_encoding?
51
+ invalid_entry = raw_origins.b.split(',').find do |entry|
52
+ !entry.dup.force_encoding(raw_origins.encoding).valid_encoding?
53
+ end
54
+ label = invalid_entry.to_s.b.byteslice(0, 160).inspect
55
+ raise Woods::ConfigurationError, "Invalid WOODS_MCP_HTTP_ALLOWED_ORIGINS entry #{label}: invalid encoding"
56
+ end
57
+ allowed_origins = raw_origins.split(',').map(&:strip).reject(&:empty?)
58
+ begin
59
+ origin_policy = Woods::MCP::OriginPolicy.new(allowed_origins: allowed_origins)
60
+ rescue ArgumentError => e
61
+ raise Woods::ConfigurationError, "#{e.message} (WOODS_MCP_HTTP_ALLOWED_ORIGINS)"
62
+ end
49
63
  index_dir = Woods::MCP::Bootstrapper.resolve_index_dir(ARGV)
50
64
  retriever, bootstrap_state = Woods::MCP::Bootstrapper.build_retriever(index_dir: index_dir)
51
65
  snapshot_store = Woods::MCP::Bootstrapper.build_snapshot_store(index_dir)
@@ -96,17 +110,10 @@ server = Woods::MCP::Server.build(
96
110
  # hatch is transitional — the spec has removed all three. See
97
111
  # docs/MCP_HTTP_TRANSPORT.md#statelessness.
98
112
  stateless = !%w[0 false no].include?(ENV.fetch('WOODS_MCP_HTTP_STATELESS', '1').strip.downcase)
99
- allowed_origins = ENV.fetch('WOODS_MCP_HTTP_ALLOWED_ORIGINS', '').split(',').map(&:strip).reject(&:empty?)
100
- allowed_hosts = allowed_origins.filter_map do |origin|
101
- URI.parse(origin).host
102
- rescue URI::InvalidURIError
103
- nil
104
- end
105
113
  transport = MCP::Server::Transports::StreamableHTTPTransport.new(
106
114
  server,
107
115
  stateless: stateless,
108
- allowed_origins: allowed_origins,
109
- allowed_hosts: allowed_hosts
116
+ **origin_policy.transport_options
110
117
  )
111
118
  server.transport = transport
112
119
 
@@ -116,7 +123,7 @@ inner = proc do |env|
116
123
  transport.handle_request(Rack::Request.new(env))
117
124
  end
118
125
  app = token ? Woods::MCP::BearerAuth.new(inner, token: token) : inner
119
- app = Woods::MCP::OriginGuard.new(app, allowed_origins: allowed_origins)
126
+ app = Woods::MCP::OriginGuard.new(app, policy: origin_policy)
120
127
 
121
128
  origin_summary = allowed_origins.empty? ? 'loopback' : allowed_origins.join(',')
122
129
  auth_mode = token ? 'bearer' : 'none'
data/exe/woods-watch ADDED
@@ -0,0 +1,5 @@
1
+ #!/usr/bin/env ruby
2
+ # frozen_string_literal: true
3
+
4
+ require 'woods/watch/cli'
5
+ exit Woods::Watch::CLI.new.run(ARGV)
@@ -0,0 +1,53 @@
1
+ # frozen_string_literal: true
2
+
3
+ require 'rails/generators'
4
+ require 'woods/watch/installation'
5
+
6
+ module Woods
7
+ module Generators
8
+ # Reversible opt-in startup integration; never rewrites an application's bin/dev.
9
+ class WatchGenerator < Rails::Generators::Base
10
+ desc 'Install, update, or remove owned Woods watcher startup configuration'
11
+
12
+ class_option :mode, type: :string, desc: 'Explicit startup mode: procfile, puma, or external'
13
+ class_option :operation, type: :string, default: 'setup', desc: 'setup, update, remove, or recover'
14
+ class_option :procfile, type: :string, default: 'Procfile.dev', desc: 'Existing root-level Foreman Procfile'
15
+ class_option :manager_command, type: :string,
16
+ desc: 'Normal Foreman startup argv, e.g. foreman start -f Procfile.dev'
17
+ class_option :child_command, type: :string, default: 'bin/rails woods:watch',
18
+ desc: 'Application task command, parsed as argv (no shell evaluation)'
19
+
20
+ # Rails generators otherwise print Thor errors and return a successful status.
21
+ # @return [Boolean] whether a refused installation fails the CLI command
22
+ def self.exit_on_failure?
23
+ true
24
+ end
25
+
26
+ # @return [void]
27
+ def configure_watcher
28
+ operation = behavior == :revoke ? 'remove' : options[:operation]
29
+ installation = build_installation(operation)
30
+ if operation == 'recover'
31
+ say installation.recover(pretend: options[:pretend])
32
+ return
33
+ end
34
+ plan = installation.plan
35
+ say JSON.pretty_generate(plan.summary)
36
+ return if options[:pretend]
37
+
38
+ say installation.apply(plan)
39
+ say installation.handoff unless operation == 'remove'
40
+ rescue Woods::Watch::Installation::Conflict => e
41
+ raise Thor::Error, e.message
42
+ end
43
+
44
+ private
45
+
46
+ def build_installation(operation)
47
+ Woods::Watch::Installation.new(root: destination_root, mode: options[:mode], operation: operation,
48
+ procfile: options[:procfile], manager_command: options[:manager_command],
49
+ child_command: options[:child_command])
50
+ end
51
+ end
52
+ end
53
+ end
@@ -0,0 +1,10 @@
1
+ # frozen_string_literal: true
2
+
3
+ require_relative '../../woods/watch/puma_adapter'
4
+
5
+ Puma::Plugin.create do
6
+ def start(launcher)
7
+ @woods_adapter ||= Woods::Watch::PumaAdapter.new(launcher)
8
+ @woods_adapter.install
9
+ end
10
+ end
data/lib/tasks/woods.rake CHANGED
@@ -123,6 +123,9 @@ namespace :woods do
123
123
 
124
124
  desc 'Watch the app and keep the index current (resident daemon)'
125
125
  task :watch do
126
+ require 'woods/watch/managed_child'
127
+ managed_child = Woods::Watch::ManagedChild.from_env
128
+ managed_child&.call(:task_loaded, woods_version: Woods::VERSION)
126
129
  # Observe inputs before Rails initializes. A snapshot taken in Daemon.new
127
130
  # would silently bless edits made while initializers were running.
128
131
  require 'woods/watch/boot_snapshot'
@@ -130,6 +133,11 @@ namespace :woods do
130
133
  fresh_environment = !environment.already_invoked && !Rails.application.initialized?
131
134
  boot_snapshot = Woods::Watch::BootSnapshot.new(root: Rails.root) if fresh_environment
132
135
  environment.invoke
136
+ if managed_child && !Rails.env.development?
137
+ managed_child.call(:terminal, reason: 'unsupported_environment')
138
+ warn 'Managed Woods watching requires the finalized Rails development environment.'
139
+ next
140
+ end
133
141
  # Both, and the extractor is not optional. The daemon's default
134
142
  # extractor_factory names Woods::Extractor lazily, so omitting this require
135
143
  # loaded and started cleanly and then NameError'd on the first real cycle —
@@ -140,6 +148,7 @@ namespace :woods do
140
148
  require 'woods/watch/daemon'
141
149
 
142
150
  output_dir = ENV.fetch('WOODS_OUTPUT', Woods.configuration.output_dir)
151
+ managed_child&.call(:identity, root: File.expand_path(Rails.root.to_s), index: File.expand_path(output_dir.to_s))
143
152
 
144
153
  poll_interval = begin
145
154
  Float(ENV.fetch('WOODS_WATCH_POLL_INTERVAL', Woods::Watch::Watcher::DEFAULT_POLL_INTERVAL))
@@ -162,6 +171,8 @@ namespace :woods do
162
171
  idle_timeout: ENV.fetch('WOODS_WATCH_IDLE_TIMEOUT', nil) && Float(ENV.fetch('WOODS_WATCH_IDLE_TIMEOUT')),
163
172
  catch_up: ENV['WOODS_WATCH_CATCH_UP'] != '0',
164
173
  boot_snapshot: boot_snapshot,
174
+ lifecycle: managed_child,
175
+ conservative_claims: !managed_child.nil?,
165
176
  logger: Rails.logger
166
177
  )
167
178
 
@@ -182,6 +193,7 @@ namespace :woods do
182
193
  puts
183
194
 
184
195
  reason = daemon.run
196
+ managed_child&.call(:terminal, reason: reason.to_s)
185
197
 
186
198
  if reason == :restart_required
187
199
  # Boot-captured state changed; Rails cannot reload it. Exit non-zero so
@@ -192,6 +204,8 @@ namespace :woods do
192
204
  end
193
205
 
194
206
  puts 'Watcher stopped.'
207
+ ensure
208
+ managed_child&.close
195
209
  end
196
210
 
197
211
  desc 'Keep watch over the woods — resident index daemon (alias for watch)'
@@ -1,6 +1,7 @@
1
1
  # frozen_string_literal: true
2
2
 
3
3
  require 'digest'
4
+ require 'securerandom'
4
5
  require_relative 'cache_store'
5
6
  # CachedEmbeddingProvider includes Embedding::Provider::Interface at load
6
7
  # time, so the interface must be defined before this file's class bodies run
@@ -408,6 +409,8 @@ module Woods
408
409
  @retriever = retriever
409
410
  @cache_store = cache_store
410
411
  @context_ttl = context_ttl
412
+ @context_namespace = SecureRandom.hex(16)
413
+ @context_mutex = Mutex.new
411
414
  end
412
415
 
413
416
  # Expose the wrapped stores so the MCP +reload+ tool and
@@ -421,20 +424,23 @@ module Woods
421
424
  def mode = @retriever.respond_to?(:mode) ? @retriever.mode : :semantic
422
425
  def default_budget = @retriever.respond_to?(:default_budget) ? @retriever.default_budget : 8000
423
426
 
424
- # Invalidate every cached context result. Called from the MCP +reload+
425
- # tool after the retriever's stores have been re-hydrated from a fresh
426
- # embed — otherwise cached results from the old embedding run would
427
- # linger until their TTL expires and contradict the new stores.
427
+ # Read the live corpus rather than caching diagnostics across reloads.
428
+ # @return [Hash, nil] local semantic corpus statistics, when supported
429
+ def corpus_status(include_types: true)
430
+ @retriever.corpus_status(include_types: include_types) if @retriever.respond_to?(:corpus_status)
431
+ end
432
+
433
+ # Retire this retriever's cached contexts after its corpus reloads.
428
434
  #
429
- # Embedding caches (query → vector) are NOT cleared: the query-vector
430
- # mapping is deterministic for a given provider+model and survives any
431
- # index reload. Only context results (query → ranked units) go stale.
435
+ # Each retriever has its own unpredictable namespace, even when several
436
+ # applications share a backend. Rotating it also prevents an in-flight
437
+ # request from repopulating the active cache with a pre-reload result.
438
+ # Old entries expire by their configured TTL; no global backend deletion
439
+ # is needed. Embedding caches are independent and remain available.
432
440
  #
433
441
  # @return [void]
434
442
  def invalidate_context_cache!
435
- @cache_store.clear(namespace: :context)
436
- rescue StandardError => e
437
- warn("[Woods] CachedRetriever context-cache invalidation failed: #{e.message}")
443
+ @context_mutex.synchronize { @context_namespace = SecureRandom.hex(16) }
438
444
  end
439
445
 
440
446
  # Execute the retrieval pipeline with context-level caching.
@@ -486,7 +492,8 @@ module Woods
486
492
  # @param exclude_types [Array<String, Symbol>, nil]
487
493
  # @return [String]
488
494
  def context_key(query, budget, types: nil, exclude_types: nil, packages: nil, source_paths: nil, evidence: 'full') # rubocop:disable Metrics/ParameterLists
489
- parts = [query, budget.to_s, fingerprint(types), fingerprint(exclude_types)]
495
+ namespace = @context_mutex.synchronize { @context_namespace }
496
+ parts = [namespace, query, budget.to_s, fingerprint(types), fingerprint(exclude_types)]
490
497
  parts << 'lexical' if mode == :lexical
491
498
  parts << JSON.generate(evidence: evidence) unless evidence == 'full'
492
499
  if Retrieval::Scope.requested?(packages: packages, source_paths: source_paths)
@@ -0,0 +1,39 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Woods
4
+ module Console
5
+ # Adapter names differ from the SQL grammar and session settings they use.
6
+ module AdapterFamily
7
+ # Classify a live connection by adapter ancestry, database configuration,
8
+ # then its display name. Unknown families remain nil: SQL callers must
9
+ # refuse rather than silently omit dialect-specific safeguards.
10
+ # @param connection [Object] Active Record connection or compatible adapter
11
+ # @return [Symbol, nil] :postgres, :mysql, :sqlite, or unknown
12
+ def self.for(connection)
13
+ ancestry = connection.class.ancestors.filter_map(&:name)
14
+ return :postgres if ancestry.include?('ActiveRecord::ConnectionAdapters::PostgreSQLAdapter')
15
+ return :mysql if ancestry.any? { |name| name.match?(/::(?:AbstractMysql|Mysql2|Trilogy)Adapter\z/) }
16
+ return :sqlite if ancestry.include?('ActiveRecord::ConnectionAdapters::SQLite3Adapter')
17
+
18
+ from_name(configured_adapter(connection)) || from_name(connection.adapter_name)
19
+ end
20
+
21
+ def self.configured_adapter(connection)
22
+ return unless connection.respond_to?(:pool) && connection.pool.respond_to?(:db_config)
23
+
24
+ connection.pool.db_config.adapter
25
+ end
26
+ private_class_method :configured_adapter
27
+
28
+ def self.from_name(value)
29
+ name = value.to_s.downcase
30
+ return :mysql if name.include?('mysql') || %w[trilogy mariadb].include?(name)
31
+ return :postgres if name.include?('postgre') || %w[postgis cockroachdb redshift].include?(name)
32
+ return :sqlite if name.include?('sqlite')
33
+
34
+ nil
35
+ end
36
+ private_class_method :from_name
37
+ end
38
+ end
39
+ end
@@ -46,7 +46,7 @@ module Woods
46
46
  # index.redact("token: sk_live_actual_secret_value")
47
47
  # # => "token: [REDACTED:credential]"
48
48
  #
49
- class CredentialIndex
49
+ class CredentialIndex # rubocop:disable Metrics/ClassLength
50
50
  # Captured at require time so the mtime-check warning has a stable
51
51
  # reference point even if the clock skews later. Frozen immediately
52
52
  # to prevent accidental mutation.
@@ -187,7 +187,10 @@ module Woods
187
187
  def initialize(secrets:)
188
188
  filtered = Array(secrets).select { |s| s.is_a?(String) && s.length >= MIN_LENGTH }
189
189
  @secrets = filtered.to_set.freeze
190
- @pattern = @secrets.empty? ? nil : Regexp.union(@secrets.to_a)
190
+ # Regexp alternatives match in order: a shorter prefix must not consume
191
+ # only the start of a longer indexed credential and expose its suffix.
192
+ longest_first = @secrets.each_with_index.sort_by { |secret, idx| [-secret.length, idx] }.map(&:first)
193
+ @pattern = @secrets.empty? ? nil : Regexp.union(longest_first)
191
194
  end
192
195
 
193
196
  # @return [Boolean] true when no secrets were collected (missing key,
@@ -212,7 +215,34 @@ module Woods
212
215
  def redact(str)
213
216
  return str if empty? || !str.is_a?(String) || !@pattern.match?(str)
214
217
 
215
- str.gsub(@pattern, REDACTED)
218
+ offset = 0
219
+ parts = []
220
+ redaction_spans(str).each do |start, finish|
221
+ parts << str[offset...start] << REDACTED
222
+ offset = finish
223
+ end
224
+ parts << str[offset..]
225
+ parts.join
226
+ end
227
+
228
+ private
229
+
230
+ # Search from each match's start, rather than its end, so an overlapping
231
+ # credential cannot expose its suffix after an earlier replacement.
232
+ # Union the covered spans before changing the original string.
233
+ def redaction_spans(str)
234
+ spans = []
235
+ offset = 0
236
+ while (match = @pattern.match(str, offset))
237
+ start, finish = match.offset(0)
238
+ if spans.last && start < spans.last[1]
239
+ spans.last[1] = [spans.last[1], finish].max
240
+ else
241
+ spans << [start, finish]
242
+ end
243
+ offset = start + 1
244
+ end
245
+ spans
216
246
  end
217
247
  end
218
248
  end