woods 2.0.0.beta3 → 2.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +500 -420
- data/CONTRIBUTING.md +29 -17
- data/README.md +78 -178
- data/docs/AGENT_GUIDE.md +52 -11
- data/docs/AGENT_SETUP.md +34 -17
- data/docs/AUTOMATIC_MAINTENANCE.md +222 -0
- data/docs/BACKEND_MATRIX.md +18 -7
- data/docs/CLIENT_HOOKS.md +1 -1
- data/docs/CONFIGURATION_REFERENCE.md +105 -29
- data/docs/CONSOLE_MCP_SETUP.md +54 -9
- data/docs/DOCKER_SETUP.md +16 -1
- data/docs/EVALUATION.md +10 -4
- data/docs/EXTRACTOR_REFERENCE.md +23 -3
- data/docs/FAQ.md +14 -3
- data/docs/GETTING_STARTED.md +18 -17
- data/docs/INCREMENTAL_EXTRACTION.md +37 -8
- data/docs/INDEX_LAYOUT.md +2 -2
- data/docs/MCP_SERVERS.md +79 -7
- data/docs/MCP_TOOL_COOKBOOK.md +5 -5
- data/docs/MCP_WORKTREE_SETUP.md +55 -83
- data/docs/PUBLISHED_INDEX.md +17 -0
- data/docs/README.md +2 -1
- data/docs/RETRIEVAL_GUIDE.md +81 -13
- data/docs/SOURCE_FRESHNESS.md +1 -1
- data/docs/TOKEN_BENCHMARK.md +16 -10
- data/docs/TROUBLESHOOTING.md +142 -47
- data/docs/UPGRADING_TO_2.md +12 -6
- data/docs/WATCH_DAEMON.md +189 -24
- data/docs/WHY_WOODS.md +9 -5
- data/exe/woods-console +13 -11
- data/exe/woods-mcp-start +14 -9
- data/exe/woods-watch +5 -0
- data/lib/generators/woods/pgvector_generator.rb +8 -2
- data/lib/generators/woods/watch_generator.rb +53 -0
- data/lib/puma/plugin/woods.rb +10 -0
- data/lib/tasks/woods.rake +14 -0
- data/lib/woods/agent_configuration/applier.rb +5 -3
- data/lib/woods/agent_configuration/cli.rb +2 -2
- data/lib/woods/agent_configuration/layout.rb +13 -0
- data/lib/woods/cache/cache_middleware.rb +6 -0
- data/lib/woods/console/credential_scanner.rb +4 -3
- data/lib/woods/console/dispatch_pipeline.rb +7 -0
- data/lib/woods/console/embedded_executor.rb +31 -9
- data/lib/woods/console/sql_noise_stripper.rb +9 -7
- data/lib/woods/console/sql_table_scanner.rb +47 -7
- data/lib/woods/console/sql_validator.rb +49 -9
- data/lib/woods/console/sqlite_read_guard.rb +46 -0
- data/lib/woods/console/stdio_transport.rb +27 -0
- data/lib/woods/coordination/pipeline_lock.rb +3 -2
- data/lib/woods/embedding/indexer.rb +24 -14
- data/lib/woods/extractor.rb +70 -19
- data/lib/woods/extractors/declared_parent.rb +55 -0
- data/lib/woods/extractors/graphql_extractor.rb +2 -11
- data/lib/woods/extractors/lib_extractor.rb +10 -8
- data/lib/woods/extractors/mailer_extractor.rb +6 -10
- data/lib/woods/extractors/model_extractor.rb +1 -15
- data/lib/woods/extractors/poro_extractor.rb +10 -8
- data/lib/woods/extractors/shared_utility_methods.rb +22 -5
- data/lib/woods/git_command.rb +6 -7
- data/lib/woods/git_provenance.rb +4 -6
- data/lib/woods/mcp/bearer_auth.rb +2 -1
- data/lib/woods/mcp/bootstrapper.rb +20 -5
- data/lib/woods/mcp/config_resolver.rb +2 -1
- data/lib/woods/mcp/index_reader.rb +11 -2
- data/lib/woods/mcp/initialization_guidance.rb +1 -1
- data/lib/woods/mcp/renderers/markdown_renderer.rb +14 -8
- data/lib/woods/mcp/renderers/plain_renderer.rb +11 -7
- data/lib/woods/mcp/server.rb +63 -37
- data/lib/woods/mcp/tool_contract.rb +1 -1
- data/lib/woods/mcp/tool_response_renderer.rb +16 -0
- data/lib/woods/mcp/traversal_evidence_text.rb +1 -1
- data/lib/woods/mcp/traversal_response.rb +22 -0
- data/lib/woods/path_dispatcher.rb +6 -5
- data/lib/woods/published_index/typed_unit_reader.rb +40 -3
- data/lib/woods/published_index.rb +2 -2
- data/lib/woods/rake_helpers.rb +2 -12
- data/lib/woods/retrieval/corpus_status.rb +46 -0
- data/lib/woods/retrieval/lexical_assembler.rb +14 -3
- data/lib/woods/retrieval/lexical_index.rb +2 -1
- data/lib/woods/retriever.rb +19 -7
- data/lib/woods/session_tracer/file_store.rb +6 -1
- data/lib/woods/source_inputs/consumer_errors.rb +4 -0
- data/lib/woods/storage/local_corpus_stats.rb +32 -0
- data/lib/woods/storage/metadata_store.rb +20 -0
- data/lib/woods/storage/pgvector.rb +6 -2
- data/lib/woods/storage/vector_store.rb +10 -0
- data/lib/woods/temporal/json_snapshot_store.rb +35 -7
- data/lib/woods/version.rb +1 -1
- data/lib/woods/watch/child_environment.rb +30 -0
- data/lib/woods/watch/cli.rb +91 -0
- data/lib/woods/watch/daemon.rb +73 -11
- data/lib/woods/watch/event_stream.rb +70 -0
- data/lib/woods/watch/guardian.rb +142 -0
- data/lib/woods/watch/installation/layout.rb +70 -0
- data/lib/woods/watch/installation/options.rb +128 -0
- data/lib/woods/watch/installation/planner.rb +128 -0
- data/lib/woods/watch/installation/probe.rb +101 -0
- data/lib/woods/watch/installation/receipt.rb +77 -0
- data/lib/woods/watch/installation/recovery.rb +64 -0
- data/lib/woods/watch/installation/templates.rb +58 -0
- data/lib/woods/watch/installation.rb +56 -0
- data/lib/woods/watch/lifecycle.rb +182 -0
- data/lib/woods/watch/managed_child.rb +113 -0
- data/lib/woods/watch/managed_cleanup.rb +48 -0
- data/lib/woods/watch/managed_process.rb +144 -0
- data/lib/woods/watch/puma_adapter.rb +87 -0
- data/lib/woods/watch/puma_child.rb +66 -0
- data/lib/woods/watch/supervision_records.rb +95 -0
- data/lib/woods/watch/supervision_status.rb +104 -0
- data/lib/woods/watch/supervisor.rb +161 -0
- data/lib/woods/watch/supervisor_reporting.rb +46 -0
- data/plugin/.claude-plugin/plugin.json +1 -1
- data/plugin/hooks/woods-input-rules.sh +4 -4
- data/plugin/skills/woods-agent-enable/SKILL.md +7 -1
- data/plugin/skills/woods-diagnose/SKILL.md +134 -34
- data/plugin/skills/woods-investigate/SKILL.md +54 -15
- data/plugin/skills/woods-mcp-config/SKILL.md +38 -11
- data/plugin/skills/woods-setup/SKILL.md +72 -15
- metadata +38 -5
|
@@ -421,6 +421,12 @@ module Woods
|
|
|
421
421
|
def mode = @retriever.respond_to?(:mode) ? @retriever.mode : :semantic
|
|
422
422
|
def default_budget = @retriever.respond_to?(:default_budget) ? @retriever.default_budget : 8000
|
|
423
423
|
|
|
424
|
+
# Read the live corpus rather than caching diagnostics across reloads.
|
|
425
|
+
# @return [Hash, nil] local semantic corpus statistics, when supported
|
|
426
|
+
def corpus_status(include_types: true)
|
|
427
|
+
@retriever.corpus_status(include_types: include_types) if @retriever.respond_to?(:corpus_status)
|
|
428
|
+
end
|
|
429
|
+
|
|
424
430
|
# Invalidate every cached context result. Called from the MCP +reload+
|
|
425
431
|
# tool after the retriever's stores have been re-hydrated from a fresh
|
|
426
432
|
# embed — otherwise cached results from the old embedding run would
|
|
@@ -145,9 +145,9 @@ module Woods
|
|
|
145
145
|
|
|
146
146
|
# Scan a value (String, Hash, Array, or any other object) for credentials.
|
|
147
147
|
#
|
|
148
|
-
# Strings are
|
|
149
|
-
# elements are walked recursively;
|
|
150
|
-
#
|
|
148
|
+
# Strings and Symbols are scanned against every active pattern. Hash
|
|
149
|
+
# keys/values and Array elements are walked recursively; numeric values,
|
|
150
|
+
# booleans, and nil pass through untouched. Symbols retain their type.
|
|
151
151
|
#
|
|
152
152
|
# @param value [Object]
|
|
153
153
|
# @return [Array(Object, Hash{Symbol=>Integer})] two-tuple of the scanned
|
|
@@ -164,6 +164,7 @@ module Woods
|
|
|
164
164
|
def walk(value, counts, index)
|
|
165
165
|
case value
|
|
166
166
|
when String then scan_string(value, counts, index)
|
|
167
|
+
when Symbol then scan_string(value.to_s, counts, index).to_sym
|
|
167
168
|
when Hash then walk_hash(value, counts, index)
|
|
168
169
|
when Array then value.map { |item| walk(item, counts, index) }
|
|
169
170
|
else value
|
|
@@ -103,7 +103,14 @@ module Woods
|
|
|
103
103
|
response = @conn_mgr.send_request(request)
|
|
104
104
|
return error_from_response(response, request) unless response['ok']
|
|
105
105
|
|
|
106
|
+
# Materialize the wire representation before policy checks. Otherwise
|
|
107
|
+
# Symbols and custom serializers can introduce unscanned strings later.
|
|
108
|
+
# Redact first so protected serializers never run; redact again after
|
|
109
|
+
# normalization to cover fields introduced by a custom serializer.
|
|
110
|
+
# Both renderers consume this same JSON-compatible data tree.
|
|
106
111
|
result = @ctx.redact(response['result'])
|
|
112
|
+
result = JSON.parse(JSON.generate(result))
|
|
113
|
+
result = @ctx.redact(result)
|
|
107
114
|
result = scan_for_credentials(result, request)
|
|
108
115
|
text = @renderer ? @renderer.render_default(result) : JSON.pretty_generate(result)
|
|
109
116
|
success_response(text)
|
|
@@ -479,7 +479,7 @@ module Woods
|
|
|
479
479
|
def handle_count(params)
|
|
480
480
|
model = resolve_model(params['model'])
|
|
481
481
|
scope = apply_scope(model, params['scope'], model_name: params['model'])
|
|
482
|
-
{ 'count' => scope.count }
|
|
482
|
+
{ 'count' => checked_relation(scope).count }
|
|
483
483
|
end
|
|
484
484
|
|
|
485
485
|
def handle_sample(params)
|
|
@@ -488,7 +488,7 @@ module Woods
|
|
|
488
488
|
limit = params.fetch('limit', 5)
|
|
489
489
|
scope = apply_scope(model, params['scope'], model_name: params['model'])
|
|
490
490
|
scope = apply_columns(scope, params['columns'])
|
|
491
|
-
records = scope.order(random_function).limit(limit)
|
|
491
|
+
records = checked_relation(scope.order(random_function).limit(limit))
|
|
492
492
|
{ 'records' => serialize_records(records, params['columns']) }
|
|
493
493
|
end
|
|
494
494
|
|
|
@@ -501,7 +501,8 @@ module Woods
|
|
|
501
501
|
end
|
|
502
502
|
validate_select_columns!(params)
|
|
503
503
|
model = resolve_model(params['model'])
|
|
504
|
-
|
|
504
|
+
scope = checked_relation(model)
|
|
505
|
+
record = params['id'] ? scope.find_by(id: params['id']) : scope.find_by(params['by'])
|
|
505
506
|
{ 'record' => record ? serialize_record(record, params['columns']) : nil }
|
|
506
507
|
end
|
|
507
508
|
|
|
@@ -534,7 +535,7 @@ module Woods
|
|
|
534
535
|
limit = params.fetch('limit', 100)
|
|
535
536
|
scope = apply_scope(model, params['scope'], model_name: params['model'])
|
|
536
537
|
scope = scope.distinct if params['distinct']
|
|
537
|
-
values = scope.limit(limit).pluck(*columns.map(&:to_sym))
|
|
538
|
+
values = checked_relation(scope.limit(limit)).pluck(*columns.map(&:to_sym))
|
|
538
539
|
{ 'columns' => Array(columns), 'values' => values }
|
|
539
540
|
end
|
|
540
541
|
|
|
@@ -554,6 +555,7 @@ module Woods
|
|
|
554
555
|
|
|
555
556
|
model = resolve_model(params['model'])
|
|
556
557
|
scope = apply_scope(model, params['scope'], model_name: params['model'])
|
|
558
|
+
scope = checked_relation(scope)
|
|
557
559
|
|
|
558
560
|
value = if function == 'count'
|
|
559
561
|
column ? scope.count(column.to_sym) : scope.count
|
|
@@ -566,6 +568,7 @@ module Woods
|
|
|
566
568
|
def handle_association_count(params)
|
|
567
569
|
model = resolve_model(params['model'])
|
|
568
570
|
association_name = params['association']
|
|
571
|
+
requested_scope = params['scope']
|
|
569
572
|
reflection = model.reflect_on_association(association_name.to_sym)
|
|
570
573
|
|
|
571
574
|
raise ValidationError, "Unknown association '#{association_name}' on #{params['model']}" unless reflection
|
|
@@ -581,15 +584,33 @@ module Woods
|
|
|
581
584
|
# association's own model before any database I/O runs (not just
|
|
582
585
|
# before the association is read): `model.find` below is itself a
|
|
583
586
|
# query, and a request with a bad scope should never reach it.
|
|
584
|
-
validate_scope_columns!(
|
|
587
|
+
validate_scope_columns!(requested_scope, reflection.klass.name) if requested_scope
|
|
585
588
|
|
|
586
|
-
record = model.find(params['id'])
|
|
589
|
+
record = checked_relation(model).find(params['id'])
|
|
587
590
|
scope = record.public_send(association_name)
|
|
588
|
-
scope = apply_scope(scope,
|
|
591
|
+
scope = apply_scope(scope, requested_scope, model_name: reflection.klass.name) if requested_scope
|
|
589
592
|
gate_association_sql!(scope)
|
|
590
593
|
{ 'count' => scope.count }
|
|
591
594
|
end
|
|
592
595
|
|
|
596
|
+
# Materialize a model's default scope once, inspect the resolved SQL,
|
|
597
|
+
# and return that same relation for execution. Checking model.table_name
|
|
598
|
+
# alone misses default scopes that change FROM or introduce joins. Do
|
|
599
|
+
# not check model.all then issue a fresh query through the model: a
|
|
600
|
+
# dynamic default scope could differ between the check and execution.
|
|
601
|
+
# With no TableGate collaborator, retain the ungated executor contract.
|
|
602
|
+
#
|
|
603
|
+
# @param scope [Class, ActiveRecord::Relation] Pending model read
|
|
604
|
+
# @return [Class, ActiveRecord::Relation] Relation authorized for reading
|
|
605
|
+
# @raise [ValidationError] if its SQL references a blocked table
|
|
606
|
+
def checked_relation(scope)
|
|
607
|
+
return scope unless @table_gate
|
|
608
|
+
|
|
609
|
+
relation = scope.all
|
|
610
|
+
gate_sql!(relation.to_sql)
|
|
611
|
+
relation
|
|
612
|
+
end
|
|
613
|
+
|
|
593
614
|
# Defense-in-depth: gate_association! (called by
|
|
594
615
|
# {#handle_association_count} before this) only proves the
|
|
595
616
|
# association's OWN target table isn't blocked. A through-association,
|
|
@@ -738,7 +759,7 @@ module Woods
|
|
|
738
759
|
|
|
739
760
|
scope = apply_scope(model, params['scope'], model_name: params['model'])
|
|
740
761
|
scope = apply_columns(scope, params['columns'])
|
|
741
|
-
records = scope.order(order_by => direction.to_sym).limit(limit)
|
|
762
|
+
records = checked_relation(scope.order(order_by => direction.to_sym).limit(limit))
|
|
742
763
|
{ 'records' => serialize_records(records, params['columns']) }
|
|
743
764
|
end
|
|
744
765
|
|
|
@@ -798,13 +819,14 @@ module Woods
|
|
|
798
819
|
# and PostgreSQL quote/comment grammars differ (`\'` escapes, `#`
|
|
799
820
|
# comments); validating with the matching dialect accepts dialect-valid
|
|
800
821
|
# literals and still rejects every known bypass form. Unknown adapters
|
|
801
|
-
# return nil and get the conservative
|
|
822
|
+
# return nil and get the conservative union of supported dialects.
|
|
802
823
|
#
|
|
803
824
|
# @return [Symbol, nil]
|
|
804
825
|
def sql_dialect
|
|
805
826
|
adapter = active_connection.adapter_name.to_s.downcase
|
|
806
827
|
return :mysql if adapter.include?('mysql')
|
|
807
828
|
return :postgres if adapter.include?('postgre')
|
|
829
|
+
return :sqlite if adapter.include?('sqlite')
|
|
808
830
|
|
|
809
831
|
nil
|
|
810
832
|
end
|
|
@@ -53,20 +53,21 @@ module Woods
|
|
|
53
53
|
# single-quote scanner.
|
|
54
54
|
#
|
|
55
55
|
# @param sql [String] the SQL string to process
|
|
56
|
-
# @param dialect [Symbol] `:postgres` (default) or `:
|
|
56
|
+
# @param dialect [Symbol] `:postgres` (default), `:mysql`, or `:sqlite`.
|
|
57
57
|
# - `:postgres` — single-quoted strings support `''` as an apostrophe
|
|
58
58
|
# escape. Backslash is treated literally and does not escape quotes.
|
|
59
59
|
# Dollar-quoted strings (`$$...$$`, `$tag$...$tag$`) are also stripped.
|
|
60
60
|
# - `:mysql` — single-quoted strings support both `\'` (backslash-escape)
|
|
61
61
|
# and `''` (doubled-quote) as apostrophe escapes. Dollar-quoted strings
|
|
62
62
|
# are not recognized by the combined MySQL security scanner.
|
|
63
|
+
# - `:sqlite` — doubled apostrophes escape strings; backslashes and dollar signs are literal.
|
|
63
64
|
# @return [String] a new string with all string literals replaced by `''`
|
|
64
65
|
# @raise [ArgumentError] if an unsupported dialect is provided
|
|
65
66
|
DOLLAR_QUOTED = /\$(\w*)\$.*?\$\1\$/m
|
|
66
67
|
SINGLE_QUOTED_POSTGRES = /'(?:''|[^'])*'/m
|
|
67
68
|
SINGLE_QUOTED_MYSQL = /'(?:\\.|''|[^'])*'/m
|
|
68
69
|
|
|
69
|
-
SUPPORTED_DIALECTS = %i[postgres mysql].freeze
|
|
70
|
+
SUPPORTED_DIALECTS = %i[postgres mysql sqlite].freeze
|
|
70
71
|
private_constant :SUPPORTED_DIALECTS
|
|
71
72
|
|
|
72
73
|
def self.strip_literals(sql, dialect: :postgres)
|
|
@@ -76,7 +77,7 @@ module Woods
|
|
|
76
77
|
|
|
77
78
|
# Strip dollar-quoted strings first so stray apostrophes inside them
|
|
78
79
|
# do not interfere with the single-quote scanner.
|
|
79
|
-
out = sql.gsub(DOLLAR_QUOTED, "''")
|
|
80
|
+
out = dialect == :sqlite ? sql : sql.gsub(DOLLAR_QUOTED, "''")
|
|
80
81
|
|
|
81
82
|
pattern = dialect == :mysql ? SINGLE_QUOTED_MYSQL : SINGLE_QUOTED_POSTGRES
|
|
82
83
|
out.gsub(pattern, "''")
|
|
@@ -112,7 +113,7 @@ module Woods
|
|
|
112
113
|
# still reads as following a boundary once comments are gone.
|
|
113
114
|
#
|
|
114
115
|
# @param sql [String] the SQL string to process
|
|
115
|
-
# @param dialect [Symbol] `:postgres` (default) or `:
|
|
116
|
+
# @param dialect [Symbol] `:postgres` (default), `:mysql`, or `:sqlite` — controls
|
|
116
117
|
# single-quote escape rules (see {.strip_literals}) and whether `#`
|
|
117
118
|
# opens a line comment. MySQL quote flags reflect session sql_mode.
|
|
118
119
|
# @return [String] a new string with comments removed and every string
|
|
@@ -134,7 +135,8 @@ module Woods
|
|
|
134
135
|
if ch == "'"
|
|
135
136
|
close = single_quote_end(
|
|
136
137
|
sql, i,
|
|
137
|
-
backslash_escapes: (mysql && !no_backslash_escapes) ||
|
|
138
|
+
backslash_escapes: (mysql && !no_backslash_escapes) ||
|
|
139
|
+
(dialect == :postgres && postgres_escape_string?(sql, i))
|
|
138
140
|
)
|
|
139
141
|
if close
|
|
140
142
|
out << "''"
|
|
@@ -158,7 +160,7 @@ module Woods
|
|
|
158
160
|
out << ch
|
|
159
161
|
i += 1
|
|
160
162
|
end
|
|
161
|
-
elsif mysql && ch == '`'
|
|
163
|
+
elsif (mysql || dialect == :sqlite) && ch == '`'
|
|
162
164
|
close = quoted_span_end(sql, i, quote: '`', backslash_escapes: false)
|
|
163
165
|
if close
|
|
164
166
|
# Backticks delimit identifiers. Preserve the token for table
|
|
@@ -170,7 +172,7 @@ module Woods
|
|
|
170
172
|
out << ch
|
|
171
173
|
i += 1
|
|
172
174
|
end
|
|
173
|
-
elsif
|
|
175
|
+
elsif dialect == :postgres && ch == '$' && !preceded_by_word_char?(sql, i) && (tag = dollar_tag_at(sql, i))
|
|
174
176
|
close = sql.index(tag, i + tag.length)
|
|
175
177
|
if close
|
|
176
178
|
out << "''"
|
|
@@ -167,6 +167,49 @@ module Woods
|
|
|
167
167
|
results.uniq
|
|
168
168
|
end
|
|
169
169
|
|
|
170
|
+
# Table-factor prefixes, retaining commas after balanced subqueries and
|
|
171
|
+
# JOIN predicates. Each nested FROM/JOIN is also scanned independently.
|
|
172
|
+
# @param sql [String] noise-stripped SQL
|
|
173
|
+
# @return [Array<String>]
|
|
174
|
+
def self.relation_factors(sql)
|
|
175
|
+
sql.to_enum(:scan, /\b(?:FROM|JOIN)\s+/i).flat_map do
|
|
176
|
+
suffix = sql[Regexp.last_match.end(0)..]
|
|
177
|
+
split_top_level_commas(relation_clause(suffix))
|
|
178
|
+
end
|
|
179
|
+
end
|
|
180
|
+
|
|
181
|
+
# Quoted tokens shield punctuation and clause words; a closing parenthesis
|
|
182
|
+
# at depth zero ends this query's clause, not a nested table expression.
|
|
183
|
+
def self.relation_clause(suffix)
|
|
184
|
+
depth = 0
|
|
185
|
+
tokens = /
|
|
186
|
+
"(?:[^"]|"")*"|`(?:[^`]|``)*`|''|[()]|
|
|
187
|
+
\b(?:WHERE|GROUP|HAVING|ORDER|LIMIT|OFFSET|UNION|INTERSECT|EXCEPT|WINDOW)\b
|
|
188
|
+
/ix
|
|
189
|
+
suffix.to_enum(:scan, tokens).each do
|
|
190
|
+
token = Regexp.last_match[0]
|
|
191
|
+
start = Regexp.last_match.begin(0)
|
|
192
|
+
finish = Regexp.last_match.end(0)
|
|
193
|
+
boundary = token == ')' || relation_keyword_boundary?(suffix[0...start], suffix[finish..], token)
|
|
194
|
+
return suffix[0...start] if depth.zero? && boundary
|
|
195
|
+
|
|
196
|
+
depth += 1 if token == '('
|
|
197
|
+
depth -= 1 if token == ')'
|
|
198
|
+
end
|
|
199
|
+
suffix
|
|
200
|
+
end
|
|
201
|
+
private_class_method :relation_clause
|
|
202
|
+
|
|
203
|
+
def self.relation_keyword_boundary?(prefix, rest, token)
|
|
204
|
+
return false unless token.match?(/\A[A-Za-z]/)
|
|
205
|
+
return false if prefix.strip.empty? || prefix.match?(/(?:,|\bAS)\s*\z/i) || rest.lstrip.start_with?(',')
|
|
206
|
+
return rest.match?(/\A\s+BY\b/i) if %w[GROUP ORDER].include?(token.upcase)
|
|
207
|
+
return rest.match?(/\A\s+\w+\s+AS\b/i) if token.casecmp?('WINDOW')
|
|
208
|
+
|
|
209
|
+
true
|
|
210
|
+
end
|
|
211
|
+
private_class_method :relation_keyword_boundary?
|
|
212
|
+
|
|
170
213
|
# @api private
|
|
171
214
|
# Every view of the stripped SQL the FROM/JOIN scans must consider for
|
|
172
215
|
# MySQL executable comments (`/*! ... */`). {SqlNoiseStripper} leaves
|
|
@@ -203,12 +246,9 @@ module Woods
|
|
|
203
246
|
|
|
204
247
|
# @api private
|
|
205
248
|
def self.collect_from_identifiers(sql, results)
|
|
206
|
-
sql.
|
|
207
|
-
|
|
208
|
-
|
|
209
|
-
ident = lead_identifier(chunk)
|
|
210
|
-
results << ident if ident
|
|
211
|
-
end
|
|
249
|
+
relation_factors(sql).each do |chunk|
|
|
250
|
+
ident = lead_identifier(chunk)
|
|
251
|
+
results << ident if ident
|
|
212
252
|
end
|
|
213
253
|
end
|
|
214
254
|
private_class_method :collect_from_identifiers
|
|
@@ -229,7 +269,7 @@ module Woods
|
|
|
229
269
|
depth = 0
|
|
230
270
|
buf = +''
|
|
231
271
|
parts = []
|
|
232
|
-
clause.
|
|
272
|
+
clause.scan(/"(?:[^"]|"")*"|`(?:[^`]|``)*`|''|./m).each do |ch|
|
|
233
273
|
case ch
|
|
234
274
|
when '('
|
|
235
275
|
depth += 1
|
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
# frozen_string_literal: true
|
|
2
2
|
|
|
3
3
|
require 'woods/console/sql_noise_stripper'
|
|
4
|
+
require 'woods/console/sqlite_read_guard'
|
|
4
5
|
|
|
5
6
|
# @see Woods
|
|
6
7
|
module Woods
|
|
@@ -28,11 +29,11 @@ module Woods
|
|
|
28
29
|
# named constants), not imperative logic.
|
|
29
30
|
class SqlValidator # rubocop:disable Metrics/ClassLength
|
|
30
31
|
# SQL dialects whose normalization the lock-clause check can run over.
|
|
31
|
-
KNOWN_DIALECTS = %i[postgres mysql].freeze
|
|
32
|
+
KNOWN_DIALECTS = %i[postgres mysql sqlite].freeze
|
|
32
33
|
|
|
33
34
|
# The dialect the statement will execute under, when the caller knows
|
|
34
35
|
# it. `nil` (the default) keeps the conservative union: every check
|
|
35
|
-
# runs against
|
|
36
|
+
# runs against all supported dialect normalizations, which can reject a
|
|
36
37
|
# statement that is valid under one dialect's quote grammar (a MySQL
|
|
37
38
|
# `\'` escape hides prose that the PostgreSQL view reads as SQL). When
|
|
38
39
|
# the execution boundary knows the adapter, passing the matching
|
|
@@ -206,9 +207,9 @@ module Woods
|
|
|
206
207
|
# function calls.
|
|
207
208
|
FUNCTION_SCAN_EXCLUDED_KEYWORDS = %w[
|
|
208
209
|
IN EXISTS NOT AND OR VALUES WHERE HAVING ON IS BETWEEN CASE WHEN
|
|
209
|
-
THEN ELSE
|
|
210
|
-
UNION INTERSECT EXCEPT ORDER GROUP BY
|
|
211
|
-
OVER
|
|
210
|
+
THEN ELSE FROM JOIN USING SELECT DISTINCT ALL ANY SOME
|
|
211
|
+
UNION INTERSECT EXCEPT ORDER GROUP BY LIMIT OFFSET AS INTO
|
|
212
|
+
OVER FILTER RETURNING EXPLAIN
|
|
212
213
|
].freeze
|
|
213
214
|
|
|
214
215
|
# EXPLAIN is a statement leader, so `EXPLAIN (FORMAT JSON) SELECT` is an
|
|
@@ -318,6 +319,7 @@ module Woods
|
|
|
318
319
|
|
|
319
320
|
return validate_dialect_variants!(sql) if unknown_grammar?
|
|
320
321
|
|
|
322
|
+
SqliteReadGuard.validate!(sql) if dialect == :sqlite
|
|
321
323
|
normalized = sql.strip
|
|
322
324
|
|
|
323
325
|
# Reject multiple statements (semicolons not inside string literals)
|
|
@@ -582,10 +584,9 @@ module Woods
|
|
|
582
584
|
match = Regexp.last_match
|
|
583
585
|
quoted = !(match[1] || match[2]).nil?
|
|
584
586
|
identifier = match[1] || match[2] || match[3]
|
|
585
|
-
#
|
|
586
|
-
#
|
|
587
|
-
|
|
588
|
-
next if !quoted && FUNCTION_SCAN_EXCLUDED_KEYWORDS.include?(identifier.upcase)
|
|
587
|
+
# Some engines also allow keyword spellings as function names.
|
|
588
|
+
# Only exempt those names in their supported grammar positions.
|
|
589
|
+
next if !quoted && function_keyword_grammar?(identifier.upcase, stripped[0...match.begin(0)])
|
|
589
590
|
next if ALLOWED_FUNCTIONS.include?(identifier.downcase)
|
|
590
591
|
|
|
591
592
|
raise SqlValidationError,
|
|
@@ -594,6 +595,45 @@ module Woods
|
|
|
594
595
|
end
|
|
595
596
|
end
|
|
596
597
|
|
|
598
|
+
# Recognize keyword grammar without exempting identically named calls.
|
|
599
|
+
# The prefix is already stripped of comments and literal contents.
|
|
600
|
+
#
|
|
601
|
+
# @param keyword [String] Uppercase bare identifier before `(`.
|
|
602
|
+
# @param prefix [String] SQL preceding the identifier.
|
|
603
|
+
# @return [Boolean]
|
|
604
|
+
def function_keyword_grammar?(keyword, prefix)
|
|
605
|
+
return false unless FUNCTION_SCAN_EXCLUDED_KEYWORDS.include?(keyword)
|
|
606
|
+
|
|
607
|
+
case keyword
|
|
608
|
+
when 'EXPLAIN' then prefix.strip.empty?
|
|
609
|
+
when 'BY' then prefix.match?(/\b(?:ORDER|GROUP|PARTITION)\s*\z/i)
|
|
610
|
+
when 'OVER', 'FILTER' then prefix.match?(/\)\s*\z/)
|
|
611
|
+
when 'ANY', 'SOME' then comparison_quantifier?(prefix)
|
|
612
|
+
when 'OFFSET' then offset_grammar?(prefix)
|
|
613
|
+
else true
|
|
614
|
+
end
|
|
615
|
+
end
|
|
616
|
+
|
|
617
|
+
# SQLite has no quantified ANY/SOME comparison grammar; both spellings
|
|
618
|
+
# can instead invoke application-defined functions after an operator.
|
|
619
|
+
def comparison_quantifier?(prefix)
|
|
620
|
+
return true if dialect == :postgres # Reserved quantifiers, including LIKE ANY.
|
|
621
|
+
|
|
622
|
+
dialect != :sqlite && prefix.match?(/[=<>]\s*\z/)
|
|
623
|
+
end
|
|
624
|
+
|
|
625
|
+
# PostgreSQL reserves bare OFFSET (it cannot name a function), so its
|
|
626
|
+
# standalone OFFSET clause remains supported. Other dialects only accept
|
|
627
|
+
# parenthesized offsets after a completed LIMIT operand. Calls
|
|
628
|
+
# at an expression's start or after an operator are never offset syntax.
|
|
629
|
+
# Keep the supported operand boundary conservative: literals, bind
|
|
630
|
+
# numbers, parenthesized expressions, or PostgreSQL's LIMIT ALL.
|
|
631
|
+
def offset_grammar?(prefix)
|
|
632
|
+
return true if dialect == :postgres
|
|
633
|
+
|
|
634
|
+
prefix.match?(/(?:[0-9)'"`]|\bLIMIT\s+ALL)\s*\z/i)
|
|
635
|
+
end
|
|
636
|
+
|
|
597
637
|
# Check if the SQL contains a forbidden keyword at a statement-leader
|
|
598
638
|
# position after stripping comments and string literals. This catches
|
|
599
639
|
# comment-hidden injections like "SELECT 1 --;\nDELETE FROM users",
|
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require 'woods/console/sql_noise_stripper'
|
|
4
|
+
require 'woods/console/sql_table_scanner'
|
|
5
|
+
|
|
6
|
+
module Woods
|
|
7
|
+
module Console
|
|
8
|
+
# Refuse SQLite-specific syntax outside the scanner's supported grammar.
|
|
9
|
+
# This is a conservative read boundary, not a general-purpose SQL parser.
|
|
10
|
+
module SqliteReadGuard
|
|
11
|
+
SIMPLE_IDENTIFIER = /\A[A-Za-z_][A-Za-z0-9_]*\z/
|
|
12
|
+
UNSUPPORTED_SYNTAX = /[^\x00-\x7F]|\$|\[|''\s*\(|\b(?:FROM|JOIN)(?=['"`(])/i
|
|
13
|
+
QUOTED_IDENTIFIER = /"(?:[^"]|"")*"|`(?:[^`]|``)*`/
|
|
14
|
+
|
|
15
|
+
# @param sql [String] SQL to execute on SQLite
|
|
16
|
+
# @raise [SqlValidationError] when an identifier or table factor cannot be checked
|
|
17
|
+
# @return [void]
|
|
18
|
+
def self.validate!(sql)
|
|
19
|
+
view = SqlNoiseStripper.strip_noise(sql, dialect: :sqlite)
|
|
20
|
+
refuse! if view.match?(UNSUPPORTED_SYNTAX)
|
|
21
|
+
view.scan(QUOTED_IDENTIFIER) { |quoted| refuse! unless quoted[1...-1].match?(SIMPLE_IDENTIFIER) }
|
|
22
|
+
SqlTableScanner.relation_factors(view).each do |factor|
|
|
23
|
+
refuse! unless supported_factor?(factor.strip)
|
|
24
|
+
end
|
|
25
|
+
end
|
|
26
|
+
|
|
27
|
+
# A SELECT/WITH subquery has its own independently scanned table factors.
|
|
28
|
+
# Parenthesized table groups and string-quoted names are refused rather
|
|
29
|
+
# than silently omitted from the blocked-table scan.
|
|
30
|
+
def self.supported_factor?(factor)
|
|
31
|
+
return true if factor.match?(/\A\(\s*(?:SELECT|WITH)\b/i)
|
|
32
|
+
|
|
33
|
+
match = SqlTableScanner::LEAD_IDENT.match(factor)
|
|
34
|
+
match && factor[match.end(0)..].match?(/\A(?:\s|,|\)|\z)/)
|
|
35
|
+
end
|
|
36
|
+
private_class_method :supported_factor?
|
|
37
|
+
|
|
38
|
+
def self.refuse!
|
|
39
|
+
raise SqlValidationError,
|
|
40
|
+
'Rejected: unsupported SQLite identifier or table-reference syntax. ' \
|
|
41
|
+
'Use simple bare or double-quoted identifiers and SELECT subqueries.'
|
|
42
|
+
end
|
|
43
|
+
private_class_method :refuse!
|
|
44
|
+
end
|
|
45
|
+
end
|
|
46
|
+
end
|
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require 'mcp'
|
|
4
|
+
|
|
5
|
+
module Woods
|
|
6
|
+
module Console
|
|
7
|
+
# Keeps protocol writes separate from the host application's stdout.
|
|
8
|
+
# The SDK still owns framing, negotiation, notifications and shutdown.
|
|
9
|
+
class StdioTransport < ::MCP::Server::Transports::StdioTransport
|
|
10
|
+
# @param server [::MCP::Server] Console server
|
|
11
|
+
# @param output [IO] Original stdout saved before redirecting Rails output
|
|
12
|
+
def initialize(server, output:)
|
|
13
|
+
super(server)
|
|
14
|
+
@output = output
|
|
15
|
+
@output.set_encoding(Encoding::UTF_8)
|
|
16
|
+
end
|
|
17
|
+
|
|
18
|
+
# SDK responses, notifications and server requests share this writer.
|
|
19
|
+
# @param message [String, Hash] Encoded JSON or a JSON-compatible message
|
|
20
|
+
# @return [IO] Flushed protocol output
|
|
21
|
+
def send_response(message)
|
|
22
|
+
@output.puts(message.is_a?(String) ? message : JSON.generate(message))
|
|
23
|
+
@output.flush
|
|
24
|
+
end
|
|
25
|
+
end
|
|
26
|
+
end
|
|
27
|
+
end
|
|
@@ -30,8 +30,9 @@ module Woods
|
|
|
30
30
|
# The transaction-guard filename for a lock of +name+, exposed so
|
|
31
31
|
# cleanup code that empties a lock directory (woods:clean) can skip the
|
|
32
32
|
# guard during its sweep — deleting a flock'd guard out from under a
|
|
33
|
-
# contender's critical section would split the flock across two inodes
|
|
34
|
-
#
|
|
33
|
+
# contender's critical section would split the flock across two inodes.
|
|
34
|
+
# The guard stays after cleanup and lock release: a new contender may
|
|
35
|
+
# already be holding it before its lock file exists.
|
|
35
36
|
#
|
|
36
37
|
# @param name [String]
|
|
37
38
|
# @return [String]
|
|
@@ -159,12 +159,11 @@ module Woods
|
|
|
159
159
|
reconcile_empty_units(checkpoint)
|
|
160
160
|
retire_legacy_identities
|
|
161
161
|
report_checkpoint_misses
|
|
162
|
-
vanished =
|
|
162
|
+
vanished = persistable? ? drop_vanished_units(incremental: incremental) : 0
|
|
163
163
|
persist_snapshot if persistable? && snapshot_worth_writing?(stats, vanished, incremental: incremental)
|
|
164
164
|
# Durable backends have no dump to rewrite, so staleness has to be
|
|
165
|
-
# removed from the store itself — on full runs too, since
|
|
166
|
-
#
|
|
167
|
-
# the in-memory path does (#211).
|
|
165
|
+
# removed from the store itself — on full runs too, since pgvector/
|
|
166
|
+
# Qdrant retain rows instead of replacing a published dump (#211).
|
|
168
167
|
reconcile_durable_store if reconcilable?
|
|
169
168
|
save_checkpoint(checkpoint)
|
|
170
169
|
|
|
@@ -275,20 +274,24 @@ module Woods
|
|
|
275
274
|
# this costs no IO) and `@current_identifiers` is what this run saw.
|
|
276
275
|
# Pruning with an empty fresh-id list removes every chunk of the unit.
|
|
277
276
|
#
|
|
278
|
-
#
|
|
279
|
-
# prune returns 0, which reads as "nothing vanished" to
|
|
277
|
+
# Incremental runs use {#vanished_prune_permitted?} (B-079 / #191).
|
|
278
|
+
# A refused prune returns 0, which reads as "nothing vanished" to
|
|
280
279
|
# {#snapshot_worth_writing?}, so a run that also embedded nothing writes
|
|
281
280
|
# no dump and the retention window is not rotated over the good dumps.
|
|
282
281
|
# The warn precedes the deletes: if the prune raises partway, the
|
|
283
282
|
# operator still learns what it was doing.
|
|
284
283
|
#
|
|
284
|
+
# Full rebuilds deliberately replace the complete corpus, including an
|
|
285
|
+
# empty corpus, even when callers reuse their in-memory stores.
|
|
286
|
+
#
|
|
287
|
+
# @param incremental [Boolean] whether to apply the incremental purge guard
|
|
285
288
|
# @return [Integer] how many units were dropped
|
|
286
|
-
def drop_vanished_units
|
|
287
|
-
return 0 if @persisted_ids.
|
|
289
|
+
def drop_vanished_units(incremental:)
|
|
290
|
+
return 0 if @persisted_ids.empty?
|
|
288
291
|
|
|
289
292
|
vanished = @persisted_ids.keys.reject { |identifier| @current_identifiers.include?(identifier) }
|
|
290
293
|
return 0 if vanished.empty?
|
|
291
|
-
return 0
|
|
294
|
+
return 0 if incremental && !vanished_prune_permitted?(vanished)
|
|
292
295
|
|
|
293
296
|
warn "[woods] dropping #{vanished.size} unit(s) from the vector index that the " \
|
|
294
297
|
'extraction no longer holds; rewriting the dump.'
|
|
@@ -321,8 +324,8 @@ module Woods
|
|
|
321
324
|
# Refusal never loses data: the stale vectors stay hydrated in the
|
|
322
325
|
# store, so any dump this run does write (for freshly embedded work)
|
|
323
326
|
# still carries them. Full runs never reach this guard —
|
|
324
|
-
#
|
|
325
|
-
#
|
|
327
|
+
# a full run replaces the complete corpus and must publish even an empty
|
|
328
|
+
# result. Its vanished-unit reconciliation bypasses this guard.
|
|
326
329
|
#
|
|
327
330
|
# @param vanished [Array<String>] identifiers about to be pruned
|
|
328
331
|
# @return [Boolean] true when the prune may proceed
|
|
@@ -357,9 +360,9 @@ module Woods
|
|
|
357
360
|
|
|
358
361
|
# Delete vectors a durable store holds for units the index no longer has.
|
|
359
362
|
#
|
|
360
|
-
# The dump-backed path
|
|
361
|
-
#
|
|
362
|
-
#
|
|
363
|
+
# The dump-backed path removes vanished units from its hydrated or
|
|
364
|
+
# reused stores before publishing a replacement dump. A durable backend
|
|
365
|
+
# has no such rewrite — rows in
|
|
363
366
|
# `woods_vectors` and points in Qdrant survive until something deletes
|
|
364
367
|
# them, which nothing did. A unit deleted from the codebase therefore
|
|
365
368
|
# stayed retrievable through `codebase_retrieve` indefinitely, *including
|
|
@@ -489,9 +492,16 @@ module Woods
|
|
|
489
492
|
entries = []
|
|
490
493
|
@vector_store.each_entry { |id, _vector, _metadata| entries << { id: id } }
|
|
491
494
|
@persisted_ids = index_ids_by_identifier(entries)
|
|
495
|
+
retain_existing_metadata_identities
|
|
492
496
|
end
|
|
493
497
|
end
|
|
494
498
|
|
|
499
|
+
def retain_existing_metadata_identities
|
|
500
|
+
return unless @metadata_store.respond_to?(:each_entry)
|
|
501
|
+
|
|
502
|
+
@metadata_store.each_entry { |identifier, _unit| @persisted_ids[identifier] ||= [] }
|
|
503
|
+
end
|
|
504
|
+
|
|
495
505
|
# Source-empty units retain metadata but intentionally have no vectors.
|
|
496
506
|
# Keep those identities in the same deletion and typed-key accounting.
|
|
497
507
|
def retain_metadata_identities
|