woods 2.0.0 → 2.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +108 -0
- data/CONTRIBUTING.md +135 -10
- data/README.md +1 -1
- data/docs/AGENT_GUIDE.md +19 -0
- data/docs/AGENT_SETUP.md +22 -2
- data/docs/BACKEND_MATRIX.md +7 -0
- data/docs/CLIENT_HOOKS.md +6 -0
- data/docs/CONFIGURATION_REFERENCE.md +133 -18
- data/docs/CONSOLE_MCP_SETUP.md +141 -12
- data/docs/EMBEDDING_MODELS.md +16 -19
- data/docs/EXTRACTOR_REFERENCE.md +219 -21
- data/docs/FAQ.md +11 -25
- data/docs/GETTING_STARTED.md +7 -1
- data/docs/INCREMENTAL_EXTRACTION.md +261 -19
- data/docs/INDEX_LAYOUT.md +5 -0
- data/docs/INTERNALS.md +9 -0
- data/docs/MCP_HTTP_TRANSPORT.md +59 -2
- data/docs/MCP_SERVERS.md +87 -8
- data/docs/MCP_TOOL_COOKBOOK.md +13 -55
- data/docs/NOTION_INTEGRATION.md +7 -1
- data/docs/PUBLISHED_INDEX.md +6 -0
- data/docs/README.md +6 -1
- data/docs/RETRIEVAL_GUIDE.md +17 -0
- data/docs/SOURCE_FRESHNESS.md +157 -5
- data/docs/TOKEN_BENCHMARK.md +10 -18
- data/docs/TROUBLESHOOTING.md +70 -14
- data/docs/UNBLOCKED_INTEGRATION.md +60 -8
- data/docs/UPGRADING_TO_2.md +184 -9
- data/docs/WATCH_DAEMON.md +97 -14
- data/exe/woods-console-mcp +2 -2
- data/exe/woods-mcp-http +16 -9
- data/lib/generators/woods/templates/woods.rb.tt +2 -1
- data/lib/tasks/woods.rake +23 -7
- data/lib/tasks/woods_checks.rake +2 -2
- data/lib/woods/agent_configuration/cli.rb +1 -1
- data/lib/woods/agent_configuration/layout.rb +16 -2
- data/lib/woods/agent_configuration/plan.rb +13 -3
- data/lib/woods/agent_configuration/planner_validation.rb +4 -2
- data/lib/woods/agent_configuration/preflight.rb +5 -3
- data/lib/woods/builder.rb +17 -57
- data/lib/woods/cache/cache_middleware.rb +68 -41
- data/lib/woods/chunking/contributor_chunks.rb +119 -0
- data/lib/woods/chunking/semantic_chunker.rb +44 -21
- data/lib/woods/console/adapter_family.rb +39 -0
- data/lib/woods/console/connection_manager.rb +56 -3
- data/lib/woods/console/credential_index.rb +33 -3
- data/lib/woods/console/embedded_executor.rb +431 -48
- data/lib/woods/console/model_validator.rb +8 -0
- data/lib/woods/console/rack_middleware.rb +68 -11
- data/lib/woods/console/redactor.rb +24 -10
- data/lib/woods/console/safe_context.rb +44 -7
- data/lib/woods/console/sql_noise_stripper.rb +41 -12
- data/lib/woods/console/sql_table_scanner.rb +45 -34
- data/lib/woods/console/sql_validator.rb +37 -2
- data/lib/woods/dependency_graph.rb +34 -10
- data/lib/woods/embedding/fake.rb +12 -0
- data/lib/woods/embedding/indexer.rb +195 -98
- data/lib/woods/embedding/input_budget.rb +67 -0
- data/lib/woods/embedding/openai.rb +70 -20
- data/lib/woods/embedding/provider.rb +37 -25
- data/lib/woods/embedding/text_preparer.rb +76 -32
- data/lib/woods/embedding/token_counter.rb +18 -81
- data/lib/woods/embedding/vector_configuration.rb +48 -0
- data/lib/woods/extraction_identities.rb +175 -0
- data/lib/woods/extractor.rb +304 -107
- data/lib/woods/extractors/action_cable_extractor.rb +8 -3
- data/lib/woods/extractors/assigned_value_discovery.rb +74 -0
- data/lib/woods/extractors/class_declarations.rb +121 -0
- data/lib/woods/extractors/configuration_extractor.rb +11 -3
- data/lib/woods/extractors/declaration_ancestry.rb +92 -0
- data/lib/woods/extractors/event_extractor.rb +8 -0
- data/lib/woods/extractors/graphql_extractor.rb +134 -77
- data/lib/woods/extractors/job_extractor.rb +5 -1
- data/lib/woods/extractors/lib_extractor.rb +132 -15
- data/lib/woods/extractors/mailer_extractor.rb +3 -5
- data/lib/woods/extractors/manager_extractor.rb +7 -21
- data/lib/woods/extractors/migration_declaration.rb +87 -0
- data/lib/woods/extractors/migration_extractor.rb +5 -39
- data/lib/woods/extractors/phlex_extractor.rb +6 -2
- data/lib/woods/extractors/policy_extractor.rb +9 -5
- data/lib/woods/extractors/poro_extractor.rb +112 -53
- data/lib/woods/extractors/pundit_extractor.rb +11 -6
- data/lib/woods/extractors/scheduled_job_extractor.rb +45 -4
- data/lib/woods/extractors/serializer_extractor.rb +34 -22
- data/lib/woods/extractors/shared_utility_methods.rb +18 -1
- data/lib/woods/extractors/source_nesting.rb +142 -106
- data/lib/woods/extractors/standalone_module_discovery.rb +123 -0
- data/lib/woods/extractors/state_machine_extractor.rb +46 -40
- data/lib/woods/extractors/view_component_extractor.rb +9 -7
- data/lib/woods/flow_assembler.rb +4 -1
- data/lib/woods/generation.rb +25 -0
- data/lib/woods/hooks/context_hint.rb +7 -2
- data/lib/woods/mcp/bearer_auth.rb +1 -1
- data/lib/woods/mcp/bootstrapper.rb +33 -7
- data/lib/woods/mcp/config_resolver.rb +26 -7
- data/lib/woods/mcp/index_reader.rb +125 -24
- data/lib/woods/mcp/index_reader_pinning.rb +16 -0
- data/lib/woods/mcp/origin_guard.rb +24 -77
- data/lib/woods/mcp/origin_policy.rb +124 -0
- data/lib/woods/mcp/renderers/markdown_renderer.rb +7 -1
- data/lib/woods/mcp/renderers/plain_renderer.rb +3 -1
- data/lib/woods/mcp/search_results.rb +7 -1
- data/lib/woods/mcp/server.rb +24 -4
- data/lib/woods/module_reconciliation.rb +151 -0
- data/lib/woods/path_dispatcher.rb +7 -2
- data/lib/woods/railtie_support.rb +8 -0
- data/lib/woods/rake_helpers.rb +43 -11
- data/lib/woods/release.rb +1 -1
- data/lib/woods/resilience/index_validator.rb +8 -3
- data/lib/woods/resilience/retryable_provider.rb +18 -1
- data/lib/woods/resolved_config.rb +68 -8
- data/lib/woods/retrieval/context_assembler.rb +3 -3
- data/lib/woods/retrieval/lexical_assembler.rb +3 -2
- data/lib/woods/retrieval/scope.rb +18 -2
- data/lib/woods/retrieval/source_evidence.rb +14 -2
- data/lib/woods/source_contributor_validation.rb +78 -0
- data/lib/woods/source_contributors.rb +116 -0
- data/lib/woods/source_inputs/handoff.rb +37 -0
- data/lib/woods/source_inputs/launcher.rb +53 -13
- data/lib/woods/source_inputs/manifest.rb +84 -3
- data/lib/woods/source_inputs/private_key.rb +44 -12
- data/lib/woods/source_inputs/scanner.rb +98 -27
- data/lib/woods/source_inputs/scopes.rb +1 -1
- data/lib/woods/source_inputs/session.rb +147 -15
- data/lib/woods/source_inputs/stable_reader.rb +127 -0
- data/lib/woods/source_inputs/status.rb +40 -8
- data/lib/woods/source_inputs/verifier.rb +28 -5
- data/lib/woods/source_path_encoding.rb +33 -0
- data/lib/woods/source_references/cache.rb +284 -0
- data/lib/woods/source_references/collector.rb +120 -0
- data/lib/woods/source_references/extraction.rb +185 -0
- data/lib/woods/source_references/inputs.rb +134 -0
- data/lib/woods/source_references/parser_adapter.rb +134 -0
- data/lib/woods/source_references/pass.rb +152 -0
- data/lib/woods/source_references/prism_adapter.rb +116 -0
- data/lib/woods/source_references/registry.rb +178 -0
- data/lib/woods/source_references/runtime_lookup.rb +127 -0
- data/lib/woods/source_references/value_class.rb +82 -0
- data/lib/woods/storage/metadata_store.rb +4 -1
- data/lib/woods/storage/qdrant.rb +2 -2
- data/lib/woods/unblocked/client.rb +12 -7
- data/lib/woods/unblocked/document_builder.rb +4 -1
- data/lib/woods/unblocked/exporter.rb +127 -37
- data/lib/woods/unblocked/sync_manifest.rb +137 -21
- data/lib/woods/unblocked/uri_migration.rb +105 -0
- data/lib/woods/util/host_guard.rb +3 -2
- data/lib/woods/version.rb +1 -1
- data/lib/woods/watch/catch_up.rb +138 -0
- data/lib/woods/watch/claim_lease.rb +150 -0
- data/lib/woods/watch/cli.rb +26 -2
- data/lib/woods/watch/daemon.rb +80 -59
- data/lib/woods/watch/installation/options.rb +1 -1
- data/lib/woods/watch/installation/receipt.rb +6 -1
- data/lib/woods/watch/managed_child.rb +1 -1
- data/lib/woods/watch/supervisor.rb +1 -1
- data/lib/woods/watch/tree_scan.rb +14 -2
- data/plugin/.claude-plugin/plugin.json +1 -1
- data/plugin/hooks/adapters/normalize.rb +3 -2
- data/plugin/hooks/woods-input-rules.sh +4 -0
- data/plugin/hooks/woods-refresh.sh +15 -7
- data/plugin/hooks/woods-session-start.sh +60 -3
- data/plugin/skills/woods-diagnose/SKILL.md +336 -2
- data/plugin/skills/woods-investigate/SKILL.md +11 -0
- data/plugin/skills/woods-mcp-config/SKILL.md +86 -0
- data/plugin/skills/woods-setup/SKILL.md +56 -1
- metadata +34 -5
|
@@ -0,0 +1,119 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require_relative 'chunk'
|
|
4
|
+
require_relative '../source_contributors'
|
|
5
|
+
|
|
6
|
+
module Woods
|
|
7
|
+
module Chunking
|
|
8
|
+
# Exact physical attribution for contiguous slices of validated contributors.
|
|
9
|
+
# Generated composite headers and cross-file content never get a physical span.
|
|
10
|
+
module ContributorChunks
|
|
11
|
+
module_function
|
|
12
|
+
|
|
13
|
+
def chunks(unit)
|
|
14
|
+
SourceContributors.records(unit).each_with_index.map do |record, index|
|
|
15
|
+
first = record.fetch('published_start_byte')
|
|
16
|
+
last = record.fetch('published_end_byte')
|
|
17
|
+
Chunk.new(content: unit.source_code.byteslice(first...last), chunk_type: :"contributor_#{index}",
|
|
18
|
+
parent_identifier: unit.identifier, parent_type: unit.type,
|
|
19
|
+
metadata: location(unit, first, last))
|
|
20
|
+
end
|
|
21
|
+
end
|
|
22
|
+
|
|
23
|
+
# Preserve existing bounded chunks only if they cover every raw byte once,
|
|
24
|
+
# in contributor order. Caller-supplied ranges never authorize citations.
|
|
25
|
+
def ensure!(unit)
|
|
26
|
+
records = SourceContributors.records(unit)
|
|
27
|
+
return if records.empty?
|
|
28
|
+
|
|
29
|
+
unit.chunks = chunks(unit).map(&:to_h) unless complete_coverage?(unit, records)
|
|
30
|
+
unit.chunks.each do |chunk|
|
|
31
|
+
range = published_range(chunk)
|
|
32
|
+
chunk[:metadata] = (chunk[:metadata] || {}).merge(location(unit, range[:start_byte], range[:end_byte]))
|
|
33
|
+
end
|
|
34
|
+
end
|
|
35
|
+
|
|
36
|
+
def complete_coverage?(unit, records)
|
|
37
|
+
remaining = unit.chunks.dup
|
|
38
|
+
records.all? do |record|
|
|
39
|
+
cursor = record['published_start_byte']
|
|
40
|
+
while cursor < record['published_end_byte']
|
|
41
|
+
chunk = remaining.shift
|
|
42
|
+
return false unless chunk
|
|
43
|
+
|
|
44
|
+
span = published_range(chunk)
|
|
45
|
+
return false unless contiguous?(unit, chunk, span, cursor, record['published_end_byte'])
|
|
46
|
+
|
|
47
|
+
cursor = span[:end_byte]
|
|
48
|
+
end
|
|
49
|
+
true
|
|
50
|
+
end && remaining.empty?
|
|
51
|
+
end
|
|
52
|
+
|
|
53
|
+
def contiguous?(unit, chunk, span, cursor, last)
|
|
54
|
+
span[:start_byte] == cursor && span[:end_byte].is_a?(Integer) &&
|
|
55
|
+
span[:end_byte] > cursor && span[:end_byte] <= last &&
|
|
56
|
+
unit.source_code.byteslice(cursor...span[:end_byte]) == chunk[:content]
|
|
57
|
+
end
|
|
58
|
+
|
|
59
|
+
def published_range(chunk)
|
|
60
|
+
metadata = SourceContributors.field(chunk, :metadata) || {}
|
|
61
|
+
span = SourceContributors.field(metadata, :published_location)
|
|
62
|
+
span.is_a?(Hash) ? span.transform_keys(&:to_sym) : {}
|
|
63
|
+
end
|
|
64
|
+
|
|
65
|
+
def location(unit, first, last)
|
|
66
|
+
physical = SourceContributors.physical_span(unit, start_byte: first, end_byte: last)
|
|
67
|
+
return {} unless physical
|
|
68
|
+
|
|
69
|
+
record = SourceContributors.records(unit).find { |entry| entry['file_path'] == physical[:file_path] }
|
|
70
|
+
physical = physical.merge(start_byte: first - record['published_start_byte'],
|
|
71
|
+
end_byte: last - record['published_start_byte'])
|
|
72
|
+
{ physical_location: physical, published_location: { start_byte: first, end_byte: last } }
|
|
73
|
+
end
|
|
74
|
+
|
|
75
|
+
# Offsets are relative to this chunk, including for a second splitting pass.
|
|
76
|
+
def slice_metadata(metadata, source, first, last)
|
|
77
|
+
metadata = symbol_keys(metadata)
|
|
78
|
+
physical = symbol_keys(metadata[:physical_location])
|
|
79
|
+
published = symbol_keys(metadata[:published_location])
|
|
80
|
+
clean = metadata.except(:physical_location, :published_location)
|
|
81
|
+
return clean unless mapped_slice?(physical, published, source, first, last)
|
|
82
|
+
|
|
83
|
+
clean.merge(physical_location: physical_slice(physical, source, first, last),
|
|
84
|
+
published_location: { start_byte: published[:start_byte] + first,
|
|
85
|
+
end_byte: published[:start_byte] + last })
|
|
86
|
+
end
|
|
87
|
+
|
|
88
|
+
def physical_slice(physical, source, first, last)
|
|
89
|
+
start_line = physical[:start_line] + source.byteslice(0...first).count("\n")
|
|
90
|
+
end_line = start_line + source.byteslice(first...last).delete_suffix("\n").count("\n")
|
|
91
|
+
physical.merge(start_byte: physical[:start_byte] + first, end_byte: physical[:start_byte] + last,
|
|
92
|
+
start_line: start_line, end_line: end_line)
|
|
93
|
+
end
|
|
94
|
+
|
|
95
|
+
def symbol_keys(value)
|
|
96
|
+
value.is_a?(Hash) ? value.transform_keys(&:to_sym) : {}
|
|
97
|
+
end
|
|
98
|
+
|
|
99
|
+
def mapped_slice?(physical, published, source, first, last)
|
|
100
|
+
integers = [physical[:start_byte], physical[:end_byte], physical[:start_line],
|
|
101
|
+
published[:start_byte], published[:end_byte]]
|
|
102
|
+
integers.all?(Integer) &&
|
|
103
|
+
[physical, published].all? { |span| span[:end_byte] - span[:start_byte] == source.bytesize } &&
|
|
104
|
+
first >= 0 && last > first && last <= source.bytesize
|
|
105
|
+
end
|
|
106
|
+
|
|
107
|
+
# Re-derive spans from the published source when hydrating a metadata dump.
|
|
108
|
+
def vector_metadata(unit, chunk_metadata)
|
|
109
|
+
return {} unless SourceContributors.multiple?(unit)
|
|
110
|
+
|
|
111
|
+
span = published_range(metadata: chunk_metadata)
|
|
112
|
+
facts = location(unit, span[:start_byte], span[:end_byte])
|
|
113
|
+
{ source_paths: SourceContributors.paths(unit),
|
|
114
|
+
file_path: facts.dig(:physical_location, :file_path) }.merge(facts)
|
|
115
|
+
end
|
|
116
|
+
private_class_method :complete_coverage?, :contiguous?, :mapped_slice?, :physical_slice, :symbol_keys
|
|
117
|
+
end
|
|
118
|
+
end
|
|
119
|
+
end
|
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
# frozen_string_literal: true
|
|
2
2
|
|
|
3
3
|
require_relative 'chunk'
|
|
4
|
+
require_relative 'contributor_chunks'
|
|
4
5
|
|
|
5
6
|
module Woods
|
|
6
7
|
module Chunking
|
|
@@ -181,11 +182,8 @@ module Woods
|
|
|
181
182
|
# Default token threshold below which units stay whole.
|
|
182
183
|
DEFAULT_THRESHOLD = 200
|
|
183
184
|
|
|
184
|
-
#
|
|
185
|
-
|
|
186
|
-
# (e.g., a single 2000-char regex line that tokenizes into 6000
|
|
187
|
-
# tokens because BERT WordPiece fragments every `\w+` boundary).
|
|
188
|
-
MIN_SLICE_CHARS = 256
|
|
185
|
+
# Stop at one Unicode character and refuse if even that cannot fit.
|
|
186
|
+
MIN_SLICE_CHARS = 1
|
|
189
187
|
private_constant :MIN_SLICE_CHARS
|
|
190
188
|
|
|
191
189
|
# @param threshold [Integer] Token count threshold for chunking
|
|
@@ -222,7 +220,8 @@ module Woods
|
|
|
222
220
|
# @return [Array<Chunk>] Ordered list of chunks
|
|
223
221
|
def chunk(unit)
|
|
224
222
|
return [] if unit.source_code.nil? || unit.source_code.strip.empty?
|
|
225
|
-
return
|
|
223
|
+
return enforce_char_limit(ContributorChunks.chunks(unit), unit) if SourceContributors.multiple?(unit)
|
|
224
|
+
return enforce_char_limit([build_whole_chunk(unit)], unit) if unit.estimated_tokens <= @threshold
|
|
226
225
|
|
|
227
226
|
enforce_char_limit(chunks_for(unit), unit)
|
|
228
227
|
end
|
|
@@ -247,6 +246,11 @@ module Woods
|
|
|
247
246
|
unit.chunks = unit.chunks.flat_map { |chunk| split_oversize_hash_chunk(chunk) }
|
|
248
247
|
end
|
|
249
248
|
|
|
249
|
+
def preparation_identity
|
|
250
|
+
{ 'class' => self.class.name, 'contributors_version' => 1, 'threshold' => @threshold, 'max_chars' => @max_chars,
|
|
251
|
+
'max_tokens' => @max_tokens, 'counter' => @token_counter&.class&.name }
|
|
252
|
+
end
|
|
253
|
+
|
|
250
254
|
private
|
|
251
255
|
|
|
252
256
|
# True when either the char ceiling or the token-based verifier is
|
|
@@ -270,12 +274,20 @@ module Woods
|
|
|
270
274
|
# @param chunk [Hash] a unit-chunk hash (symbol or string keys)
|
|
271
275
|
# @return [Array<Hash>]
|
|
272
276
|
def split_oversize_hash_chunk(chunk)
|
|
273
|
-
|
|
277
|
+
chunk = chunk.transform_keys(&:to_sym)
|
|
278
|
+
content = chunk[:content]
|
|
274
279
|
return [chunk] if content.nil? || !oversize?(content)
|
|
275
280
|
|
|
276
|
-
chunk_type = chunk[:chunk_type] ||
|
|
281
|
+
chunk_type = chunk[:chunk_type] || :whole
|
|
282
|
+
offset = SourceContributors.field(chunk[:embedding_slice], :start_byte) || 0
|
|
283
|
+
initial_offset = offset
|
|
277
284
|
verified_slices(content).each_with_index.map do |slice, idx|
|
|
278
|
-
|
|
285
|
+
first = offset
|
|
286
|
+
offset += slice.bytesize
|
|
287
|
+
metadata = ContributorChunks.slice_metadata(chunk[:metadata], content, first - initial_offset,
|
|
288
|
+
offset - initial_offset)
|
|
289
|
+
chunk.merge(content: slice, chunk_type: :"#{chunk_type}_part_#{idx}", metadata: metadata,
|
|
290
|
+
embedding_slice: { start_byte: first, end_byte: offset })
|
|
279
291
|
end
|
|
280
292
|
end
|
|
281
293
|
|
|
@@ -332,12 +344,16 @@ module Woods
|
|
|
332
344
|
def split_oversize_chunk(chunk, unit)
|
|
333
345
|
return [chunk] unless oversize?(chunk.content)
|
|
334
346
|
|
|
347
|
+
offset = 0
|
|
335
348
|
verified_slices(chunk.content).each_with_index.map do |slice, idx|
|
|
349
|
+
first = offset
|
|
350
|
+
offset += slice.bytesize
|
|
336
351
|
Chunk.new(
|
|
337
352
|
content: slice,
|
|
338
353
|
chunk_type: :"#{chunk.chunk_type}_part_#{idx}",
|
|
339
354
|
parent_identifier: unit.identifier,
|
|
340
|
-
parent_type: unit.type
|
|
355
|
+
parent_type: unit.type,
|
|
356
|
+
metadata: ContributorChunks.slice_metadata(chunk.metadata, chunk.content, first, offset)
|
|
341
357
|
)
|
|
342
358
|
end
|
|
343
359
|
end
|
|
@@ -359,9 +375,8 @@ module Woods
|
|
|
359
375
|
end
|
|
360
376
|
|
|
361
377
|
# Ensure a single post-line-split slice fits the token budget.
|
|
362
|
-
# Halves the
|
|
363
|
-
#
|
|
364
|
-
# cannot be split line-wise (minified output, huge regex literals).
|
|
378
|
+
# Halves the character limit and reslices if necessary. A minimum-sized
|
|
379
|
+
# slice that still exceeds the counter's limit is refused, never emitted.
|
|
365
380
|
#
|
|
366
381
|
# @param slice [String]
|
|
367
382
|
# @param char_limit [Integer]
|
|
@@ -370,7 +385,9 @@ module Woods
|
|
|
370
385
|
return [slice] unless @token_counter.count(slice) > @max_tokens
|
|
371
386
|
|
|
372
387
|
smaller = char_limit / 2
|
|
373
|
-
|
|
388
|
+
if smaller < MIN_SLICE_CHARS || slice.length <= 1
|
|
389
|
+
raise ArgumentError, 'Chunk input limit cannot fit one source character'
|
|
390
|
+
end
|
|
374
391
|
|
|
375
392
|
slice_by_lines(slice, smaller).flat_map { |sub| verify_slice(sub, smaller) }
|
|
376
393
|
end
|
|
@@ -385,13 +402,15 @@ module Woods
|
|
|
385
402
|
end
|
|
386
403
|
|
|
387
404
|
# Greedy line-based slicing that respects a supplied `limit`.
|
|
388
|
-
# Lines longer than `limit` are
|
|
389
|
-
#
|
|
405
|
+
# Lines longer than `limit` are cut at Unicode character boundaries;
|
|
406
|
+
# concatenating the slices preserves every source character.
|
|
390
407
|
#
|
|
391
408
|
# @param content [String]
|
|
392
409
|
# @param limit [Integer]
|
|
393
410
|
# @return [Array<String>]
|
|
394
411
|
def slice_by_lines(content, limit = @max_chars)
|
|
412
|
+
raise ArgumentError, 'Chunk size must be positive' unless limit.positive?
|
|
413
|
+
|
|
395
414
|
slices = []
|
|
396
415
|
current = String.new
|
|
397
416
|
content.each_line do |line|
|
|
@@ -658,6 +677,8 @@ module Woods
|
|
|
658
677
|
include ChunkBuilder
|
|
659
678
|
include LineDepthTracking
|
|
660
679
|
|
|
680
|
+
METHOD_NAME = /\A\s*def\s+((?:self\s*\.\s*)?(?:#{OPERATOR_METHOD_NAMES}|[[:word:]]+[?!=]?))/
|
|
681
|
+
|
|
661
682
|
# @param unit [ExtractedUnit]
|
|
662
683
|
def initialize(unit)
|
|
663
684
|
@unit = unit
|
|
@@ -677,7 +698,7 @@ module Woods
|
|
|
677
698
|
# @return [Hash]
|
|
678
699
|
def parse_lines(lines)
|
|
679
700
|
state = {
|
|
680
|
-
summary: [], methods: {}, private_methods: [],
|
|
701
|
+
summary: [], methods: {}, method_occurrences: Hash.new(0), private_methods: [],
|
|
681
702
|
current_method: nil, depth: 0, in_private: false
|
|
682
703
|
}
|
|
683
704
|
lines.each do |line|
|
|
@@ -725,14 +746,16 @@ module Woods
|
|
|
725
746
|
end
|
|
726
747
|
|
|
727
748
|
def start_method(state, line)
|
|
728
|
-
method_name = line[
|
|
749
|
+
method_name = line[METHOD_NAME, 1]&.delete(" \t")
|
|
729
750
|
|
|
730
751
|
if state[:in_private]
|
|
731
752
|
state[:private_methods] << line
|
|
732
753
|
else
|
|
733
|
-
#
|
|
734
|
-
#
|
|
735
|
-
# methods in source order.
|
|
754
|
+
# Repeated definitions (nested owners, singleton bodies, or inlined
|
|
755
|
+
# overrides) must never overwrite earlier source. Preserve existing
|
|
756
|
+
# keys for unique methods and number later occurrences in source order.
|
|
757
|
+
occurrence = state[:method_occurrences][method_name] += 1
|
|
758
|
+
method_name = "#{method_name}@#{occurrence}" if occurrence > 1
|
|
736
759
|
state[:methods][method_name] = [line]
|
|
737
760
|
end
|
|
738
761
|
|
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Woods
|
|
4
|
+
module Console
|
|
5
|
+
# Adapter names differ from the SQL grammar and session settings they use.
|
|
6
|
+
module AdapterFamily
|
|
7
|
+
# Classify a live connection by adapter ancestry, database configuration,
|
|
8
|
+
# then its display name. Unknown families remain nil: SQL callers must
|
|
9
|
+
# refuse rather than silently omit dialect-specific safeguards.
|
|
10
|
+
# @param connection [Object] Active Record connection or compatible adapter
|
|
11
|
+
# @return [Symbol, nil] :postgres, :mysql, :sqlite, or unknown
|
|
12
|
+
def self.for(connection)
|
|
13
|
+
ancestry = connection.class.ancestors.filter_map(&:name)
|
|
14
|
+
return :postgres if ancestry.include?('ActiveRecord::ConnectionAdapters::PostgreSQLAdapter')
|
|
15
|
+
return :mysql if ancestry.any? { |name| name.match?(/::(?:AbstractMysql|Mysql2|Trilogy)Adapter\z/) }
|
|
16
|
+
return :sqlite if ancestry.include?('ActiveRecord::ConnectionAdapters::SQLite3Adapter')
|
|
17
|
+
|
|
18
|
+
from_name(configured_adapter(connection)) || from_name(connection.adapter_name)
|
|
19
|
+
end
|
|
20
|
+
|
|
21
|
+
def self.configured_adapter(connection)
|
|
22
|
+
return unless connection.respond_to?(:pool) && connection.pool.respond_to?(:db_config)
|
|
23
|
+
|
|
24
|
+
connection.pool.db_config.adapter
|
|
25
|
+
end
|
|
26
|
+
private_class_method :configured_adapter
|
|
27
|
+
|
|
28
|
+
def self.from_name(value)
|
|
29
|
+
name = value.to_s.downcase
|
|
30
|
+
return :mysql if name.include?('mysql') || %w[trilogy mariadb].include?(name)
|
|
31
|
+
return :postgres if name.include?('postgre') || %w[postgis cockroachdb redshift].include?(name)
|
|
32
|
+
return :sqlite if name.include?('sqlite')
|
|
33
|
+
|
|
34
|
+
nil
|
|
35
|
+
end
|
|
36
|
+
private_class_method :from_name
|
|
37
|
+
end
|
|
38
|
+
end
|
|
39
|
+
end
|
|
@@ -17,6 +17,12 @@ module Woods
|
|
|
17
17
|
# process with a direct, Docker, or SSH command.
|
|
18
18
|
class ConnectionManager
|
|
19
19
|
DEFAULT_COMMAND = 'bundle exec rake woods:console'
|
|
20
|
+
MODE_OPTIONS = {
|
|
21
|
+
'direct' => %w[mode command directory],
|
|
22
|
+
'docker' => %w[mode command container],
|
|
23
|
+
'ssh' => %w[mode command host user]
|
|
24
|
+
}.freeze
|
|
25
|
+
CONFIG_KEYS = MODE_OPTIONS.values.flatten.uniq.freeze
|
|
20
26
|
|
|
21
27
|
# @param config [Hash] Process-launch configuration
|
|
22
28
|
# @option config [String] 'mode' direct, docker, or ssh (default: direct)
|
|
@@ -27,8 +33,9 @@ module Woods
|
|
|
27
33
|
# @option config [String] 'user' Optional SSH user
|
|
28
34
|
def initialize(config:)
|
|
29
35
|
@config = config
|
|
36
|
+
validate_config!
|
|
30
37
|
@mode = config.fetch('mode', 'direct')
|
|
31
|
-
@embedded_command = config
|
|
38
|
+
@embedded_command = config['command']
|
|
32
39
|
end
|
|
33
40
|
|
|
34
41
|
# Return the argv used to launch the embedded MCP server.
|
|
@@ -36,6 +43,7 @@ module Woods
|
|
|
36
43
|
# @return [Array<String>]
|
|
37
44
|
# @raise [ConnectionError] when the mode is invalid or incomplete
|
|
38
45
|
def command
|
|
46
|
+
validate_mode_options!
|
|
39
47
|
case @mode
|
|
40
48
|
when 'direct' then embedded_argv
|
|
41
49
|
when 'docker' then docker_command
|
|
@@ -53,10 +61,11 @@ module Woods
|
|
|
53
61
|
# @return [void]
|
|
54
62
|
# @raise [ConnectionError] when the process cannot be launched
|
|
55
63
|
def replace_process!
|
|
64
|
+
argv = command
|
|
56
65
|
if @mode == 'direct' && @config['directory']
|
|
57
|
-
Dir.chdir(@config['directory']) { exec(*
|
|
66
|
+
Dir.chdir(@config['directory']) { exec(*argv) }
|
|
58
67
|
else
|
|
59
|
-
exec(*
|
|
68
|
+
exec(*argv)
|
|
60
69
|
end
|
|
61
70
|
rescue SystemCallError, ArgumentError => e
|
|
62
71
|
raise ConnectionError, "Failed to launch embedded Console MCP (#{@mode}): #{e.message}"
|
|
@@ -64,7 +73,43 @@ module Woods
|
|
|
64
73
|
|
|
65
74
|
private
|
|
66
75
|
|
|
76
|
+
def validate_config!
|
|
77
|
+
raise ConnectionError, 'Console configuration must be a YAML mapping' unless @config.is_a?(Hash)
|
|
78
|
+
raise ConnectionError, 'Console configuration keys must be strings' unless @config.keys.all?(String)
|
|
79
|
+
|
|
80
|
+
unknown = @config.keys - CONFIG_KEYS
|
|
81
|
+
unless unknown.empty?
|
|
82
|
+
raise ConnectionError,
|
|
83
|
+
"Unsupported console.yml keys: #{unknown.map(&:inspect).join(', ')}. " \
|
|
84
|
+
"Use top-level #{CONFIG_KEYS.join(', ')} launch options; " \
|
|
85
|
+
'configure access controls and redaction in the Rails initializer.'
|
|
86
|
+
end
|
|
87
|
+
|
|
88
|
+
@config.each do |key, value|
|
|
89
|
+
next if valid_launch_string?(value)
|
|
90
|
+
|
|
91
|
+
raise ConnectionError, "Console #{key} must be a non-empty string without NUL bytes"
|
|
92
|
+
end
|
|
93
|
+
end
|
|
94
|
+
|
|
95
|
+
def valid_launch_string?(value)
|
|
96
|
+
value.is_a?(String) && !value.strip.empty? && !value.include?("\0")
|
|
97
|
+
end
|
|
98
|
+
|
|
99
|
+
def validate_mode_options!
|
|
100
|
+
accepted = MODE_OPTIONS[@mode]
|
|
101
|
+
return unless accepted
|
|
102
|
+
|
|
103
|
+
unused = @config.keys - accepted
|
|
104
|
+
return if unused.empty?
|
|
105
|
+
|
|
106
|
+
raise ConnectionError,
|
|
107
|
+
"Console options #{unused.join(', ')} are not used in #{@mode} mode; select the intended mode"
|
|
108
|
+
end
|
|
109
|
+
|
|
67
110
|
def embedded_argv
|
|
111
|
+
return default_argv unless @embedded_command
|
|
112
|
+
|
|
68
113
|
argv = @embedded_command.to_s.shellsplit
|
|
69
114
|
raise ConnectionError, 'Console command must not be empty' if argv.empty?
|
|
70
115
|
|
|
@@ -73,6 +118,14 @@ module Woods
|
|
|
73
118
|
raise ConnectionError, "Invalid console command: #{e.message}"
|
|
74
119
|
end
|
|
75
120
|
|
|
121
|
+
def default_argv
|
|
122
|
+
if @mode == 'direct'
|
|
123
|
+
binstub = File.expand_path('bin/rake', @config.fetch('directory', Dir.pwd))
|
|
124
|
+
return [binstub, 'woods:console'] if File.file?(binstub) && File.executable?(binstub)
|
|
125
|
+
end
|
|
126
|
+
DEFAULT_COMMAND.shellsplit
|
|
127
|
+
end
|
|
128
|
+
|
|
76
129
|
def docker_command
|
|
77
130
|
container = @config['container'] || raise(ConnectionError, 'Docker mode requires container name')
|
|
78
131
|
['docker', 'exec', '-i', container] + embedded_argv
|
|
@@ -46,7 +46,7 @@ module Woods
|
|
|
46
46
|
# index.redact("token: sk_live_actual_secret_value")
|
|
47
47
|
# # => "token: [REDACTED:credential]"
|
|
48
48
|
#
|
|
49
|
-
class CredentialIndex
|
|
49
|
+
class CredentialIndex # rubocop:disable Metrics/ClassLength
|
|
50
50
|
# Captured at require time so the mtime-check warning has a stable
|
|
51
51
|
# reference point even if the clock skews later. Frozen immediately
|
|
52
52
|
# to prevent accidental mutation.
|
|
@@ -187,7 +187,10 @@ module Woods
|
|
|
187
187
|
def initialize(secrets:)
|
|
188
188
|
filtered = Array(secrets).select { |s| s.is_a?(String) && s.length >= MIN_LENGTH }
|
|
189
189
|
@secrets = filtered.to_set.freeze
|
|
190
|
-
|
|
190
|
+
# Regexp alternatives match in order: a shorter prefix must not consume
|
|
191
|
+
# only the start of a longer indexed credential and expose its suffix.
|
|
192
|
+
longest_first = @secrets.each_with_index.sort_by { |secret, idx| [-secret.length, idx] }.map(&:first)
|
|
193
|
+
@pattern = @secrets.empty? ? nil : Regexp.union(longest_first)
|
|
191
194
|
end
|
|
192
195
|
|
|
193
196
|
# @return [Boolean] true when no secrets were collected (missing key,
|
|
@@ -212,7 +215,34 @@ module Woods
|
|
|
212
215
|
def redact(str)
|
|
213
216
|
return str if empty? || !str.is_a?(String) || !@pattern.match?(str)
|
|
214
217
|
|
|
215
|
-
|
|
218
|
+
offset = 0
|
|
219
|
+
parts = []
|
|
220
|
+
redaction_spans(str).each do |start, finish|
|
|
221
|
+
parts << str[offset...start] << REDACTED
|
|
222
|
+
offset = finish
|
|
223
|
+
end
|
|
224
|
+
parts << str[offset..]
|
|
225
|
+
parts.join
|
|
226
|
+
end
|
|
227
|
+
|
|
228
|
+
private
|
|
229
|
+
|
|
230
|
+
# Search from each match's start, rather than its end, so an overlapping
|
|
231
|
+
# credential cannot expose its suffix after an earlier replacement.
|
|
232
|
+
# Union the covered spans before changing the original string.
|
|
233
|
+
def redaction_spans(str)
|
|
234
|
+
spans = []
|
|
235
|
+
offset = 0
|
|
236
|
+
while (match = @pattern.match(str, offset))
|
|
237
|
+
start, finish = match.offset(0)
|
|
238
|
+
if spans.last && start < spans.last[1]
|
|
239
|
+
spans.last[1] = [spans.last[1], finish].max
|
|
240
|
+
else
|
|
241
|
+
spans << [start, finish]
|
|
242
|
+
end
|
|
243
|
+
offset = start + 1
|
|
244
|
+
end
|
|
245
|
+
spans
|
|
216
246
|
end
|
|
217
247
|
end
|
|
218
248
|
end
|