woods 2.0.0 → 2.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (167) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +108 -0
  3. data/CONTRIBUTING.md +135 -10
  4. data/README.md +1 -1
  5. data/docs/AGENT_GUIDE.md +19 -0
  6. data/docs/AGENT_SETUP.md +22 -2
  7. data/docs/BACKEND_MATRIX.md +7 -0
  8. data/docs/CLIENT_HOOKS.md +6 -0
  9. data/docs/CONFIGURATION_REFERENCE.md +133 -18
  10. data/docs/CONSOLE_MCP_SETUP.md +141 -12
  11. data/docs/EMBEDDING_MODELS.md +16 -19
  12. data/docs/EXTRACTOR_REFERENCE.md +219 -21
  13. data/docs/FAQ.md +11 -25
  14. data/docs/GETTING_STARTED.md +7 -1
  15. data/docs/INCREMENTAL_EXTRACTION.md +261 -19
  16. data/docs/INDEX_LAYOUT.md +5 -0
  17. data/docs/INTERNALS.md +9 -0
  18. data/docs/MCP_HTTP_TRANSPORT.md +59 -2
  19. data/docs/MCP_SERVERS.md +87 -8
  20. data/docs/MCP_TOOL_COOKBOOK.md +13 -55
  21. data/docs/NOTION_INTEGRATION.md +7 -1
  22. data/docs/PUBLISHED_INDEX.md +6 -0
  23. data/docs/README.md +6 -1
  24. data/docs/RETRIEVAL_GUIDE.md +17 -0
  25. data/docs/SOURCE_FRESHNESS.md +157 -5
  26. data/docs/TOKEN_BENCHMARK.md +10 -18
  27. data/docs/TROUBLESHOOTING.md +70 -14
  28. data/docs/UNBLOCKED_INTEGRATION.md +60 -8
  29. data/docs/UPGRADING_TO_2.md +184 -9
  30. data/docs/WATCH_DAEMON.md +97 -14
  31. data/exe/woods-console-mcp +2 -2
  32. data/exe/woods-mcp-http +16 -9
  33. data/lib/generators/woods/templates/woods.rb.tt +2 -1
  34. data/lib/tasks/woods.rake +23 -7
  35. data/lib/tasks/woods_checks.rake +2 -2
  36. data/lib/woods/agent_configuration/cli.rb +1 -1
  37. data/lib/woods/agent_configuration/layout.rb +16 -2
  38. data/lib/woods/agent_configuration/plan.rb +13 -3
  39. data/lib/woods/agent_configuration/planner_validation.rb +4 -2
  40. data/lib/woods/agent_configuration/preflight.rb +5 -3
  41. data/lib/woods/builder.rb +17 -57
  42. data/lib/woods/cache/cache_middleware.rb +68 -41
  43. data/lib/woods/chunking/contributor_chunks.rb +119 -0
  44. data/lib/woods/chunking/semantic_chunker.rb +44 -21
  45. data/lib/woods/console/adapter_family.rb +39 -0
  46. data/lib/woods/console/connection_manager.rb +56 -3
  47. data/lib/woods/console/credential_index.rb +33 -3
  48. data/lib/woods/console/embedded_executor.rb +431 -48
  49. data/lib/woods/console/model_validator.rb +8 -0
  50. data/lib/woods/console/rack_middleware.rb +68 -11
  51. data/lib/woods/console/redactor.rb +24 -10
  52. data/lib/woods/console/safe_context.rb +44 -7
  53. data/lib/woods/console/sql_noise_stripper.rb +41 -12
  54. data/lib/woods/console/sql_table_scanner.rb +45 -34
  55. data/lib/woods/console/sql_validator.rb +37 -2
  56. data/lib/woods/dependency_graph.rb +34 -10
  57. data/lib/woods/embedding/fake.rb +12 -0
  58. data/lib/woods/embedding/indexer.rb +195 -98
  59. data/lib/woods/embedding/input_budget.rb +67 -0
  60. data/lib/woods/embedding/openai.rb +70 -20
  61. data/lib/woods/embedding/provider.rb +37 -25
  62. data/lib/woods/embedding/text_preparer.rb +76 -32
  63. data/lib/woods/embedding/token_counter.rb +18 -81
  64. data/lib/woods/embedding/vector_configuration.rb +48 -0
  65. data/lib/woods/extraction_identities.rb +175 -0
  66. data/lib/woods/extractor.rb +304 -107
  67. data/lib/woods/extractors/action_cable_extractor.rb +8 -3
  68. data/lib/woods/extractors/assigned_value_discovery.rb +74 -0
  69. data/lib/woods/extractors/class_declarations.rb +121 -0
  70. data/lib/woods/extractors/configuration_extractor.rb +11 -3
  71. data/lib/woods/extractors/declaration_ancestry.rb +92 -0
  72. data/lib/woods/extractors/event_extractor.rb +8 -0
  73. data/lib/woods/extractors/graphql_extractor.rb +134 -77
  74. data/lib/woods/extractors/job_extractor.rb +5 -1
  75. data/lib/woods/extractors/lib_extractor.rb +132 -15
  76. data/lib/woods/extractors/mailer_extractor.rb +3 -5
  77. data/lib/woods/extractors/manager_extractor.rb +7 -21
  78. data/lib/woods/extractors/migration_declaration.rb +87 -0
  79. data/lib/woods/extractors/migration_extractor.rb +5 -39
  80. data/lib/woods/extractors/phlex_extractor.rb +6 -2
  81. data/lib/woods/extractors/policy_extractor.rb +9 -5
  82. data/lib/woods/extractors/poro_extractor.rb +112 -53
  83. data/lib/woods/extractors/pundit_extractor.rb +11 -6
  84. data/lib/woods/extractors/scheduled_job_extractor.rb +45 -4
  85. data/lib/woods/extractors/serializer_extractor.rb +34 -22
  86. data/lib/woods/extractors/shared_utility_methods.rb +18 -1
  87. data/lib/woods/extractors/source_nesting.rb +142 -106
  88. data/lib/woods/extractors/standalone_module_discovery.rb +123 -0
  89. data/lib/woods/extractors/state_machine_extractor.rb +46 -40
  90. data/lib/woods/extractors/view_component_extractor.rb +9 -7
  91. data/lib/woods/flow_assembler.rb +4 -1
  92. data/lib/woods/generation.rb +25 -0
  93. data/lib/woods/hooks/context_hint.rb +7 -2
  94. data/lib/woods/mcp/bearer_auth.rb +1 -1
  95. data/lib/woods/mcp/bootstrapper.rb +33 -7
  96. data/lib/woods/mcp/config_resolver.rb +26 -7
  97. data/lib/woods/mcp/index_reader.rb +125 -24
  98. data/lib/woods/mcp/index_reader_pinning.rb +16 -0
  99. data/lib/woods/mcp/origin_guard.rb +24 -77
  100. data/lib/woods/mcp/origin_policy.rb +124 -0
  101. data/lib/woods/mcp/renderers/markdown_renderer.rb +7 -1
  102. data/lib/woods/mcp/renderers/plain_renderer.rb +3 -1
  103. data/lib/woods/mcp/search_results.rb +7 -1
  104. data/lib/woods/mcp/server.rb +24 -4
  105. data/lib/woods/module_reconciliation.rb +151 -0
  106. data/lib/woods/path_dispatcher.rb +7 -2
  107. data/lib/woods/railtie_support.rb +8 -0
  108. data/lib/woods/rake_helpers.rb +43 -11
  109. data/lib/woods/release.rb +1 -1
  110. data/lib/woods/resilience/index_validator.rb +8 -3
  111. data/lib/woods/resilience/retryable_provider.rb +18 -1
  112. data/lib/woods/resolved_config.rb +68 -8
  113. data/lib/woods/retrieval/context_assembler.rb +3 -3
  114. data/lib/woods/retrieval/lexical_assembler.rb +3 -2
  115. data/lib/woods/retrieval/scope.rb +18 -2
  116. data/lib/woods/retrieval/source_evidence.rb +14 -2
  117. data/lib/woods/source_contributor_validation.rb +78 -0
  118. data/lib/woods/source_contributors.rb +116 -0
  119. data/lib/woods/source_inputs/handoff.rb +37 -0
  120. data/lib/woods/source_inputs/launcher.rb +53 -13
  121. data/lib/woods/source_inputs/manifest.rb +84 -3
  122. data/lib/woods/source_inputs/private_key.rb +44 -12
  123. data/lib/woods/source_inputs/scanner.rb +98 -27
  124. data/lib/woods/source_inputs/scopes.rb +1 -1
  125. data/lib/woods/source_inputs/session.rb +147 -15
  126. data/lib/woods/source_inputs/stable_reader.rb +127 -0
  127. data/lib/woods/source_inputs/status.rb +40 -8
  128. data/lib/woods/source_inputs/verifier.rb +28 -5
  129. data/lib/woods/source_path_encoding.rb +33 -0
  130. data/lib/woods/source_references/cache.rb +284 -0
  131. data/lib/woods/source_references/collector.rb +120 -0
  132. data/lib/woods/source_references/extraction.rb +185 -0
  133. data/lib/woods/source_references/inputs.rb +134 -0
  134. data/lib/woods/source_references/parser_adapter.rb +134 -0
  135. data/lib/woods/source_references/pass.rb +152 -0
  136. data/lib/woods/source_references/prism_adapter.rb +116 -0
  137. data/lib/woods/source_references/registry.rb +178 -0
  138. data/lib/woods/source_references/runtime_lookup.rb +127 -0
  139. data/lib/woods/source_references/value_class.rb +82 -0
  140. data/lib/woods/storage/metadata_store.rb +4 -1
  141. data/lib/woods/storage/qdrant.rb +2 -2
  142. data/lib/woods/unblocked/client.rb +12 -7
  143. data/lib/woods/unblocked/document_builder.rb +4 -1
  144. data/lib/woods/unblocked/exporter.rb +127 -37
  145. data/lib/woods/unblocked/sync_manifest.rb +137 -21
  146. data/lib/woods/unblocked/uri_migration.rb +105 -0
  147. data/lib/woods/util/host_guard.rb +3 -2
  148. data/lib/woods/version.rb +1 -1
  149. data/lib/woods/watch/catch_up.rb +138 -0
  150. data/lib/woods/watch/claim_lease.rb +150 -0
  151. data/lib/woods/watch/cli.rb +26 -2
  152. data/lib/woods/watch/daemon.rb +80 -59
  153. data/lib/woods/watch/installation/options.rb +1 -1
  154. data/lib/woods/watch/installation/receipt.rb +6 -1
  155. data/lib/woods/watch/managed_child.rb +1 -1
  156. data/lib/woods/watch/supervisor.rb +1 -1
  157. data/lib/woods/watch/tree_scan.rb +14 -2
  158. data/plugin/.claude-plugin/plugin.json +1 -1
  159. data/plugin/hooks/adapters/normalize.rb +3 -2
  160. data/plugin/hooks/woods-input-rules.sh +4 -0
  161. data/plugin/hooks/woods-refresh.sh +15 -7
  162. data/plugin/hooks/woods-session-start.sh +60 -3
  163. data/plugin/skills/woods-diagnose/SKILL.md +336 -2
  164. data/plugin/skills/woods-investigate/SKILL.md +11 -0
  165. data/plugin/skills/woods-mcp-config/SKILL.md +86 -0
  166. data/plugin/skills/woods-setup/SKILL.md +56 -1
  167. metadata +34 -5
@@ -0,0 +1,119 @@
1
+ # frozen_string_literal: true
2
+
3
+ require_relative 'chunk'
4
+ require_relative '../source_contributors'
5
+
6
+ module Woods
7
+ module Chunking
8
+ # Exact physical attribution for contiguous slices of validated contributors.
9
+ # Generated composite headers and cross-file content never get a physical span.
10
+ module ContributorChunks
11
+ module_function
12
+
13
+ def chunks(unit)
14
+ SourceContributors.records(unit).each_with_index.map do |record, index|
15
+ first = record.fetch('published_start_byte')
16
+ last = record.fetch('published_end_byte')
17
+ Chunk.new(content: unit.source_code.byteslice(first...last), chunk_type: :"contributor_#{index}",
18
+ parent_identifier: unit.identifier, parent_type: unit.type,
19
+ metadata: location(unit, first, last))
20
+ end
21
+ end
22
+
23
+ # Preserve existing bounded chunks only if they cover every raw byte once,
24
+ # in contributor order. Caller-supplied ranges never authorize citations.
25
+ def ensure!(unit)
26
+ records = SourceContributors.records(unit)
27
+ return if records.empty?
28
+
29
+ unit.chunks = chunks(unit).map(&:to_h) unless complete_coverage?(unit, records)
30
+ unit.chunks.each do |chunk|
31
+ range = published_range(chunk)
32
+ chunk[:metadata] = (chunk[:metadata] || {}).merge(location(unit, range[:start_byte], range[:end_byte]))
33
+ end
34
+ end
35
+
36
+ def complete_coverage?(unit, records)
37
+ remaining = unit.chunks.dup
38
+ records.all? do |record|
39
+ cursor = record['published_start_byte']
40
+ while cursor < record['published_end_byte']
41
+ chunk = remaining.shift
42
+ return false unless chunk
43
+
44
+ span = published_range(chunk)
45
+ return false unless contiguous?(unit, chunk, span, cursor, record['published_end_byte'])
46
+
47
+ cursor = span[:end_byte]
48
+ end
49
+ true
50
+ end && remaining.empty?
51
+ end
52
+
53
+ def contiguous?(unit, chunk, span, cursor, last)
54
+ span[:start_byte] == cursor && span[:end_byte].is_a?(Integer) &&
55
+ span[:end_byte] > cursor && span[:end_byte] <= last &&
56
+ unit.source_code.byteslice(cursor...span[:end_byte]) == chunk[:content]
57
+ end
58
+
59
+ def published_range(chunk)
60
+ metadata = SourceContributors.field(chunk, :metadata) || {}
61
+ span = SourceContributors.field(metadata, :published_location)
62
+ span.is_a?(Hash) ? span.transform_keys(&:to_sym) : {}
63
+ end
64
+
65
+ def location(unit, first, last)
66
+ physical = SourceContributors.physical_span(unit, start_byte: first, end_byte: last)
67
+ return {} unless physical
68
+
69
+ record = SourceContributors.records(unit).find { |entry| entry['file_path'] == physical[:file_path] }
70
+ physical = physical.merge(start_byte: first - record['published_start_byte'],
71
+ end_byte: last - record['published_start_byte'])
72
+ { physical_location: physical, published_location: { start_byte: first, end_byte: last } }
73
+ end
74
+
75
+ # Offsets are relative to this chunk, including for a second splitting pass.
76
+ def slice_metadata(metadata, source, first, last)
77
+ metadata = symbol_keys(metadata)
78
+ physical = symbol_keys(metadata[:physical_location])
79
+ published = symbol_keys(metadata[:published_location])
80
+ clean = metadata.except(:physical_location, :published_location)
81
+ return clean unless mapped_slice?(physical, published, source, first, last)
82
+
83
+ clean.merge(physical_location: physical_slice(physical, source, first, last),
84
+ published_location: { start_byte: published[:start_byte] + first,
85
+ end_byte: published[:start_byte] + last })
86
+ end
87
+
88
+ def physical_slice(physical, source, first, last)
89
+ start_line = physical[:start_line] + source.byteslice(0...first).count("\n")
90
+ end_line = start_line + source.byteslice(first...last).delete_suffix("\n").count("\n")
91
+ physical.merge(start_byte: physical[:start_byte] + first, end_byte: physical[:start_byte] + last,
92
+ start_line: start_line, end_line: end_line)
93
+ end
94
+
95
+ def symbol_keys(value)
96
+ value.is_a?(Hash) ? value.transform_keys(&:to_sym) : {}
97
+ end
98
+
99
+ def mapped_slice?(physical, published, source, first, last)
100
+ integers = [physical[:start_byte], physical[:end_byte], physical[:start_line],
101
+ published[:start_byte], published[:end_byte]]
102
+ integers.all?(Integer) &&
103
+ [physical, published].all? { |span| span[:end_byte] - span[:start_byte] == source.bytesize } &&
104
+ first >= 0 && last > first && last <= source.bytesize
105
+ end
106
+
107
+ # Re-derive spans from the published source when hydrating a metadata dump.
108
+ def vector_metadata(unit, chunk_metadata)
109
+ return {} unless SourceContributors.multiple?(unit)
110
+
111
+ span = published_range(metadata: chunk_metadata)
112
+ facts = location(unit, span[:start_byte], span[:end_byte])
113
+ { source_paths: SourceContributors.paths(unit),
114
+ file_path: facts.dig(:physical_location, :file_path) }.merge(facts)
115
+ end
116
+ private_class_method :complete_coverage?, :contiguous?, :mapped_slice?, :physical_slice, :symbol_keys
117
+ end
118
+ end
119
+ end
@@ -1,6 +1,7 @@
1
1
  # frozen_string_literal: true
2
2
 
3
3
  require_relative 'chunk'
4
+ require_relative 'contributor_chunks'
4
5
 
5
6
  module Woods
6
7
  module Chunking
@@ -181,11 +182,8 @@ module Woods
181
182
  # Default token threshold below which units stay whole.
182
183
  DEFAULT_THRESHOLD = 200
183
184
 
184
- # Minimum chars-per-slice budget during tokenizer-driven recursive
185
- # splitting. Prevents unbounded halving on pathological content
186
- # (e.g., a single 2000-char regex line that tokenizes into 6000
187
- # tokens because BERT WordPiece fragments every `\w+` boundary).
188
- MIN_SLICE_CHARS = 256
185
+ # Stop at one Unicode character and refuse if even that cannot fit.
186
+ MIN_SLICE_CHARS = 1
189
187
  private_constant :MIN_SLICE_CHARS
190
188
 
191
189
  # @param threshold [Integer] Token count threshold for chunking
@@ -222,7 +220,8 @@ module Woods
222
220
  # @return [Array<Chunk>] Ordered list of chunks
223
221
  def chunk(unit)
224
222
  return [] if unit.source_code.nil? || unit.source_code.strip.empty?
225
- return [build_whole_chunk(unit)] if unit.estimated_tokens <= @threshold
223
+ return enforce_char_limit(ContributorChunks.chunks(unit), unit) if SourceContributors.multiple?(unit)
224
+ return enforce_char_limit([build_whole_chunk(unit)], unit) if unit.estimated_tokens <= @threshold
226
225
 
227
226
  enforce_char_limit(chunks_for(unit), unit)
228
227
  end
@@ -247,6 +246,11 @@ module Woods
247
246
  unit.chunks = unit.chunks.flat_map { |chunk| split_oversize_hash_chunk(chunk) }
248
247
  end
249
248
 
249
+ def preparation_identity
250
+ { 'class' => self.class.name, 'contributors_version' => 1, 'threshold' => @threshold, 'max_chars' => @max_chars,
251
+ 'max_tokens' => @max_tokens, 'counter' => @token_counter&.class&.name }
252
+ end
253
+
250
254
  private
251
255
 
252
256
  # True when either the char ceiling or the token-based verifier is
@@ -270,12 +274,20 @@ module Woods
270
274
  # @param chunk [Hash] a unit-chunk hash (symbol or string keys)
271
275
  # @return [Array<Hash>]
272
276
  def split_oversize_hash_chunk(chunk)
273
- content = chunk[:content] || chunk['content']
277
+ chunk = chunk.transform_keys(&:to_sym)
278
+ content = chunk[:content]
274
279
  return [chunk] if content.nil? || !oversize?(content)
275
280
 
276
- chunk_type = chunk[:chunk_type] || chunk['chunk_type'] || :whole
281
+ chunk_type = chunk[:chunk_type] || :whole
282
+ offset = SourceContributors.field(chunk[:embedding_slice], :start_byte) || 0
283
+ initial_offset = offset
277
284
  verified_slices(content).each_with_index.map do |slice, idx|
278
- { content: slice, chunk_type: :"#{chunk_type}_part_#{idx}" }
285
+ first = offset
286
+ offset += slice.bytesize
287
+ metadata = ContributorChunks.slice_metadata(chunk[:metadata], content, first - initial_offset,
288
+ offset - initial_offset)
289
+ chunk.merge(content: slice, chunk_type: :"#{chunk_type}_part_#{idx}", metadata: metadata,
290
+ embedding_slice: { start_byte: first, end_byte: offset })
279
291
  end
280
292
  end
281
293
 
@@ -332,12 +344,16 @@ module Woods
332
344
  def split_oversize_chunk(chunk, unit)
333
345
  return [chunk] unless oversize?(chunk.content)
334
346
 
347
+ offset = 0
335
348
  verified_slices(chunk.content).each_with_index.map do |slice, idx|
349
+ first = offset
350
+ offset += slice.bytesize
336
351
  Chunk.new(
337
352
  content: slice,
338
353
  chunk_type: :"#{chunk.chunk_type}_part_#{idx}",
339
354
  parent_identifier: unit.identifier,
340
- parent_type: unit.type
355
+ parent_type: unit.type,
356
+ metadata: ContributorChunks.slice_metadata(chunk.metadata, chunk.content, first, offset)
341
357
  )
342
358
  end
343
359
  end
@@ -359,9 +375,8 @@ module Woods
359
375
  end
360
376
 
361
377
  # Ensure a single post-line-split slice fits the token budget.
362
- # Halves the char limit and reslices if it doesn't. Stops at
363
- # {MIN_SLICE_CHARS} to avoid unbounded recursion on content that
364
- # cannot be split line-wise (minified output, huge regex literals).
378
+ # Halves the character limit and reslices if necessary. A minimum-sized
379
+ # slice that still exceeds the counter's limit is refused, never emitted.
365
380
  #
366
381
  # @param slice [String]
367
382
  # @param char_limit [Integer]
@@ -370,7 +385,9 @@ module Woods
370
385
  return [slice] unless @token_counter.count(slice) > @max_tokens
371
386
 
372
387
  smaller = char_limit / 2
373
- return [slice] if smaller < MIN_SLICE_CHARS
388
+ if smaller < MIN_SLICE_CHARS || slice.length <= 1
389
+ raise ArgumentError, 'Chunk input limit cannot fit one source character'
390
+ end
374
391
 
375
392
  slice_by_lines(slice, smaller).flat_map { |sub| verify_slice(sub, smaller) }
376
393
  end
@@ -385,13 +402,15 @@ module Woods
385
402
  end
386
403
 
387
404
  # Greedy line-based slicing that respects a supplied `limit`.
388
- # Lines longer than `limit` are hard-cut (lossy — but such lines
389
- # are already pathological: minified JSON dumps, long regexes).
405
+ # Lines longer than `limit` are cut at Unicode character boundaries;
406
+ # concatenating the slices preserves every source character.
390
407
  #
391
408
  # @param content [String]
392
409
  # @param limit [Integer]
393
410
  # @return [Array<String>]
394
411
  def slice_by_lines(content, limit = @max_chars)
412
+ raise ArgumentError, 'Chunk size must be positive' unless limit.positive?
413
+
395
414
  slices = []
396
415
  current = String.new
397
416
  content.each_line do |line|
@@ -658,6 +677,8 @@ module Woods
658
677
  include ChunkBuilder
659
678
  include LineDepthTracking
660
679
 
680
+ METHOD_NAME = /\A\s*def\s+((?:self\s*\.\s*)?(?:#{OPERATOR_METHOD_NAMES}|[[:word:]]+[?!=]?))/
681
+
661
682
  # @param unit [ExtractedUnit]
662
683
  def initialize(unit)
663
684
  @unit = unit
@@ -677,7 +698,7 @@ module Woods
677
698
  # @return [Hash]
678
699
  def parse_lines(lines)
679
700
  state = {
680
- summary: [], methods: {}, private_methods: [],
701
+ summary: [], methods: {}, method_occurrences: Hash.new(0), private_methods: [],
681
702
  current_method: nil, depth: 0, in_private: false
682
703
  }
683
704
  lines.each do |line|
@@ -725,14 +746,16 @@ module Woods
725
746
  end
726
747
 
727
748
  def start_method(state, line)
728
- method_name = line[/def\s+(?:self\.)?(\w+)/, 1] || operator_def_name(line)
749
+ method_name = line[METHOD_NAME, 1]&.delete(" \t")
729
750
 
730
751
  if state[:in_private]
731
752
  state[:private_methods] << line
732
753
  else
733
- # Preserve insertion order — Hash does this by default, but we
734
- # initialize the entry here so `build_chunks` below walks
735
- # methods in source order.
754
+ # Repeated definitions (nested owners, singleton bodies, or inlined
755
+ # overrides) must never overwrite earlier source. Preserve existing
756
+ # keys for unique methods and number later occurrences in source order.
757
+ occurrence = state[:method_occurrences][method_name] += 1
758
+ method_name = "#{method_name}@#{occurrence}" if occurrence > 1
736
759
  state[:methods][method_name] = [line]
737
760
  end
738
761
 
@@ -0,0 +1,39 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Woods
4
+ module Console
5
+ # Adapter names differ from the SQL grammar and session settings they use.
6
+ module AdapterFamily
7
+ # Classify a live connection by adapter ancestry, database configuration,
8
+ # then its display name. Unknown families remain nil: SQL callers must
9
+ # refuse rather than silently omit dialect-specific safeguards.
10
+ # @param connection [Object] Active Record connection or compatible adapter
11
+ # @return [Symbol, nil] :postgres, :mysql, :sqlite, or unknown
12
+ def self.for(connection)
13
+ ancestry = connection.class.ancestors.filter_map(&:name)
14
+ return :postgres if ancestry.include?('ActiveRecord::ConnectionAdapters::PostgreSQLAdapter')
15
+ return :mysql if ancestry.any? { |name| name.match?(/::(?:AbstractMysql|Mysql2|Trilogy)Adapter\z/) }
16
+ return :sqlite if ancestry.include?('ActiveRecord::ConnectionAdapters::SQLite3Adapter')
17
+
18
+ from_name(configured_adapter(connection)) || from_name(connection.adapter_name)
19
+ end
20
+
21
+ def self.configured_adapter(connection)
22
+ return unless connection.respond_to?(:pool) && connection.pool.respond_to?(:db_config)
23
+
24
+ connection.pool.db_config.adapter
25
+ end
26
+ private_class_method :configured_adapter
27
+
28
+ def self.from_name(value)
29
+ name = value.to_s.downcase
30
+ return :mysql if name.include?('mysql') || %w[trilogy mariadb].include?(name)
31
+ return :postgres if name.include?('postgre') || %w[postgis cockroachdb redshift].include?(name)
32
+ return :sqlite if name.include?('sqlite')
33
+
34
+ nil
35
+ end
36
+ private_class_method :from_name
37
+ end
38
+ end
39
+ end
@@ -17,6 +17,12 @@ module Woods
17
17
  # process with a direct, Docker, or SSH command.
18
18
  class ConnectionManager
19
19
  DEFAULT_COMMAND = 'bundle exec rake woods:console'
20
+ MODE_OPTIONS = {
21
+ 'direct' => %w[mode command directory],
22
+ 'docker' => %w[mode command container],
23
+ 'ssh' => %w[mode command host user]
24
+ }.freeze
25
+ CONFIG_KEYS = MODE_OPTIONS.values.flatten.uniq.freeze
20
26
 
21
27
  # @param config [Hash] Process-launch configuration
22
28
  # @option config [String] 'mode' direct, docker, or ssh (default: direct)
@@ -27,8 +33,9 @@ module Woods
27
33
  # @option config [String] 'user' Optional SSH user
28
34
  def initialize(config:)
29
35
  @config = config
36
+ validate_config!
30
37
  @mode = config.fetch('mode', 'direct')
31
- @embedded_command = config.fetch('command', DEFAULT_COMMAND)
38
+ @embedded_command = config['command']
32
39
  end
33
40
 
34
41
  # Return the argv used to launch the embedded MCP server.
@@ -36,6 +43,7 @@ module Woods
36
43
  # @return [Array<String>]
37
44
  # @raise [ConnectionError] when the mode is invalid or incomplete
38
45
  def command
46
+ validate_mode_options!
39
47
  case @mode
40
48
  when 'direct' then embedded_argv
41
49
  when 'docker' then docker_command
@@ -53,10 +61,11 @@ module Woods
53
61
  # @return [void]
54
62
  # @raise [ConnectionError] when the process cannot be launched
55
63
  def replace_process!
64
+ argv = command
56
65
  if @mode == 'direct' && @config['directory']
57
- Dir.chdir(@config['directory']) { exec(*command) }
66
+ Dir.chdir(@config['directory']) { exec(*argv) }
58
67
  else
59
- exec(*command)
68
+ exec(*argv)
60
69
  end
61
70
  rescue SystemCallError, ArgumentError => e
62
71
  raise ConnectionError, "Failed to launch embedded Console MCP (#{@mode}): #{e.message}"
@@ -64,7 +73,43 @@ module Woods
64
73
 
65
74
  private
66
75
 
76
+ def validate_config!
77
+ raise ConnectionError, 'Console configuration must be a YAML mapping' unless @config.is_a?(Hash)
78
+ raise ConnectionError, 'Console configuration keys must be strings' unless @config.keys.all?(String)
79
+
80
+ unknown = @config.keys - CONFIG_KEYS
81
+ unless unknown.empty?
82
+ raise ConnectionError,
83
+ "Unsupported console.yml keys: #{unknown.map(&:inspect).join(', ')}. " \
84
+ "Use top-level #{CONFIG_KEYS.join(', ')} launch options; " \
85
+ 'configure access controls and redaction in the Rails initializer.'
86
+ end
87
+
88
+ @config.each do |key, value|
89
+ next if valid_launch_string?(value)
90
+
91
+ raise ConnectionError, "Console #{key} must be a non-empty string without NUL bytes"
92
+ end
93
+ end
94
+
95
+ def valid_launch_string?(value)
96
+ value.is_a?(String) && !value.strip.empty? && !value.include?("\0")
97
+ end
98
+
99
+ def validate_mode_options!
100
+ accepted = MODE_OPTIONS[@mode]
101
+ return unless accepted
102
+
103
+ unused = @config.keys - accepted
104
+ return if unused.empty?
105
+
106
+ raise ConnectionError,
107
+ "Console options #{unused.join(', ')} are not used in #{@mode} mode; select the intended mode"
108
+ end
109
+
67
110
  def embedded_argv
111
+ return default_argv unless @embedded_command
112
+
68
113
  argv = @embedded_command.to_s.shellsplit
69
114
  raise ConnectionError, 'Console command must not be empty' if argv.empty?
70
115
 
@@ -73,6 +118,14 @@ module Woods
73
118
  raise ConnectionError, "Invalid console command: #{e.message}"
74
119
  end
75
120
 
121
+ def default_argv
122
+ if @mode == 'direct'
123
+ binstub = File.expand_path('bin/rake', @config.fetch('directory', Dir.pwd))
124
+ return [binstub, 'woods:console'] if File.file?(binstub) && File.executable?(binstub)
125
+ end
126
+ DEFAULT_COMMAND.shellsplit
127
+ end
128
+
76
129
  def docker_command
77
130
  container = @config['container'] || raise(ConnectionError, 'Docker mode requires container name')
78
131
  ['docker', 'exec', '-i', container] + embedded_argv
@@ -46,7 +46,7 @@ module Woods
46
46
  # index.redact("token: sk_live_actual_secret_value")
47
47
  # # => "token: [REDACTED:credential]"
48
48
  #
49
- class CredentialIndex
49
+ class CredentialIndex # rubocop:disable Metrics/ClassLength
50
50
  # Captured at require time so the mtime-check warning has a stable
51
51
  # reference point even if the clock skews later. Frozen immediately
52
52
  # to prevent accidental mutation.
@@ -187,7 +187,10 @@ module Woods
187
187
  def initialize(secrets:)
188
188
  filtered = Array(secrets).select { |s| s.is_a?(String) && s.length >= MIN_LENGTH }
189
189
  @secrets = filtered.to_set.freeze
190
- @pattern = @secrets.empty? ? nil : Regexp.union(@secrets.to_a)
190
+ # Regexp alternatives match in order: a shorter prefix must not consume
191
+ # only the start of a longer indexed credential and expose its suffix.
192
+ longest_first = @secrets.each_with_index.sort_by { |secret, idx| [-secret.length, idx] }.map(&:first)
193
+ @pattern = @secrets.empty? ? nil : Regexp.union(longest_first)
191
194
  end
192
195
 
193
196
  # @return [Boolean] true when no secrets were collected (missing key,
@@ -212,7 +215,34 @@ module Woods
212
215
  def redact(str)
213
216
  return str if empty? || !str.is_a?(String) || !@pattern.match?(str)
214
217
 
215
- str.gsub(@pattern, REDACTED)
218
+ offset = 0
219
+ parts = []
220
+ redaction_spans(str).each do |start, finish|
221
+ parts << str[offset...start] << REDACTED
222
+ offset = finish
223
+ end
224
+ parts << str[offset..]
225
+ parts.join
226
+ end
227
+
228
+ private
229
+
230
+ # Search from each match's start, rather than its end, so an overlapping
231
+ # credential cannot expose its suffix after an earlier replacement.
232
+ # Union the covered spans before changing the original string.
233
+ def redaction_spans(str)
234
+ spans = []
235
+ offset = 0
236
+ while (match = @pattern.match(str, offset))
237
+ start, finish = match.offset(0)
238
+ if spans.last && start < spans.last[1]
239
+ spans.last[1] = [spans.last[1], finish].max
240
+ else
241
+ spans << [start, finish]
242
+ end
243
+ offset = start + 1
244
+ end
245
+ spans
216
246
  end
217
247
  end
218
248
  end