tree_haver 7.0.0 → 7.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- checksums.yaml.gz.sig +0 -0
- data/LICENSE.md +13 -0
- data/README.md +1959 -0
- data/lib/tree_haver/backend_api.rb +392 -0
- data/lib/tree_haver/backend_registry.rb +153 -3
- data/lib/tree_haver/backends/citrus.rb +489 -0
- data/lib/tree_haver/backends/ffi.rb +1013 -0
- data/lib/tree_haver/backends/java.rb +909 -0
- data/lib/tree_haver/backends/mri.rb +367 -0
- data/lib/tree_haver/backends/parslet.rb +565 -0
- data/lib/tree_haver/backends/prism.rb +568 -0
- data/lib/tree_haver/backends/psych.rb +379 -0
- data/lib/tree_haver/backends/rust.rb +243 -0
- data/lib/tree_haver/backends/tslp.rb +274 -0
- data/lib/tree_haver/base/comment.rb +320 -0
- data/lib/tree_haver/base/language.rb +98 -0
- data/lib/tree_haver/base/node.rb +330 -0
- data/lib/tree_haver/base/parser.rb +28 -0
- data/lib/tree_haver/base/point.rb +48 -0
- data/lib/tree_haver/base/tree.rb +128 -0
- data/lib/tree_haver/citrus_grammar_finder.rb +213 -0
- data/lib/tree_haver/contracts.rb +661 -96
- data/lib/tree_haver/grammar_finder.rb +429 -0
- data/lib/tree_haver/kaitai_backend.rb +2 -2
- data/lib/tree_haver/language.rb +294 -0
- data/lib/tree_haver/language_pack.rb +17 -166
- data/lib/tree_haver/language_registry.rb +221 -0
- data/lib/tree_haver/library_path_utils.rb +80 -0
- data/lib/tree_haver/node.rb +588 -0
- data/lib/tree_haver/parser.rb +445 -0
- data/lib/tree_haver/parslet_grammar_finder.rb +217 -0
- data/lib/tree_haver/path_validator.rb +356 -0
- data/lib/tree_haver/peg_backends.rb +7 -7
- data/lib/tree_haver/point.rb +27 -0
- data/lib/tree_haver/rspec/dependency_tags.rb +52 -0
- data/lib/tree_haver/rspec.rb +3 -0
- data/lib/tree_haver/tree.rb +267 -0
- data/lib/tree_haver/version.rb +5 -3
- data/lib/tree_haver.rb +613 -8
- data/sig/tree_haver.rbs +6 -0
- data.tar.gz.sig +0 -0
- metadata +314 -13
- metadata.gz.sig +0 -0
|
@@ -0,0 +1,274 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module TreeHaver
|
|
4
|
+
module Backends
|
|
5
|
+
# TSLP backend using tree_sitter_language_pack's on-demand parser API.
|
|
6
|
+
#
|
|
7
|
+
# This backend intentionally requires the real parser API. The language-pack
|
|
8
|
+
# process API is not a TreeHaver parser backend and must not be used as a
|
|
9
|
+
# merge-gem integration surface.
|
|
10
|
+
module Tslp
|
|
11
|
+
@load_attempted = false
|
|
12
|
+
@loaded = false
|
|
13
|
+
@unavailable_reason = nil
|
|
14
|
+
@language_availability = {}
|
|
15
|
+
@language_unavailable_reasons = {}
|
|
16
|
+
PARSER_SMOKE_SOURCES = {
|
|
17
|
+
'json' => '{}',
|
|
18
|
+
'json5' => '{}',
|
|
19
|
+
'bash' => "echo tree_haver\n",
|
|
20
|
+
'go' => "package main\nfunc main() {}\n",
|
|
21
|
+
'html' => "<!doctype html>\n<title>TreeHaver</title>\n",
|
|
22
|
+
'markdown' => "# TreeHaver\n",
|
|
23
|
+
'ruby' => "class TreeHaverSmoke\nend\n",
|
|
24
|
+
'rust' => "fn main() {}\n",
|
|
25
|
+
'typescript' => "export const treeHaver = true;\n",
|
|
26
|
+
'toml' => "title = \"tree_haver\"\n",
|
|
27
|
+
'yaml' => "tree_haver: true\n"
|
|
28
|
+
}.freeze
|
|
29
|
+
DEFAULT_PARSER_SMOKE_SOURCE = ''
|
|
30
|
+
|
|
31
|
+
class << self
|
|
32
|
+
attr_reader :unavailable_reason
|
|
33
|
+
|
|
34
|
+
def available?
|
|
35
|
+
return @loaded if @load_attempted
|
|
36
|
+
|
|
37
|
+
@load_attempted = true
|
|
38
|
+
begin
|
|
39
|
+
require 'tree_sitter_language_pack' unless defined?(::TreeSitterLanguagePack)
|
|
40
|
+
@loaded = parser_api_available?
|
|
41
|
+
if !@loaded && @unavailable_reason.to_s.empty?
|
|
42
|
+
@unavailable_reason = 'tree_sitter_language_pack parser API is not exposed'
|
|
43
|
+
end
|
|
44
|
+
rescue LoadError => e
|
|
45
|
+
@loaded = false
|
|
46
|
+
@unavailable_reason = e.message
|
|
47
|
+
rescue StandardError => e
|
|
48
|
+
@loaded = false
|
|
49
|
+
@unavailable_reason = e.message
|
|
50
|
+
end
|
|
51
|
+
@loaded
|
|
52
|
+
end
|
|
53
|
+
|
|
54
|
+
def reset!
|
|
55
|
+
@load_attempted = false
|
|
56
|
+
@loaded = false
|
|
57
|
+
@unavailable_reason = nil
|
|
58
|
+
@language_availability = {}
|
|
59
|
+
@language_unavailable_reasons = {}
|
|
60
|
+
end
|
|
61
|
+
|
|
62
|
+
def capabilities
|
|
63
|
+
return {} unless available?
|
|
64
|
+
|
|
65
|
+
{
|
|
66
|
+
backend: :tslp,
|
|
67
|
+
query: false,
|
|
68
|
+
bytes_field: true,
|
|
69
|
+
incremental: false,
|
|
70
|
+
comment_support: :nodes_only,
|
|
71
|
+
language_pack: true
|
|
72
|
+
}
|
|
73
|
+
end
|
|
74
|
+
|
|
75
|
+
def parser_available_for?(language_name)
|
|
76
|
+
return false unless available?
|
|
77
|
+
|
|
78
|
+
name = language_name.to_s
|
|
79
|
+
return @language_availability.fetch(name) if @language_availability.key?(name)
|
|
80
|
+
|
|
81
|
+
@language_availability[name] = smoke_parse_language(name)
|
|
82
|
+
@language_unavailable_reasons[name] = @unavailable_reason unless @language_availability.fetch(name)
|
|
83
|
+
@language_availability.fetch(name)
|
|
84
|
+
rescue StandardError => e
|
|
85
|
+
@unavailable_reason = e.message
|
|
86
|
+
@language_unavailable_reasons[language_name.to_s] = e.message
|
|
87
|
+
false
|
|
88
|
+
end
|
|
89
|
+
|
|
90
|
+
private
|
|
91
|
+
|
|
92
|
+
def parser_api_available?
|
|
93
|
+
return false unless ::TreeSitterLanguagePack.respond_to?(:get_parser)
|
|
94
|
+
return false unless defined?(::TreeSitterLanguagePack::Parser)
|
|
95
|
+
return false unless ::TreeSitterLanguagePack::Parser.instance_methods.include?(:parse)
|
|
96
|
+
|
|
97
|
+
parser_api_smoke_test
|
|
98
|
+
rescue StandardError => e
|
|
99
|
+
@unavailable_reason = e.message
|
|
100
|
+
false
|
|
101
|
+
end
|
|
102
|
+
|
|
103
|
+
def parser_api_smoke_test
|
|
104
|
+
language_name, source = PARSER_SMOKE_SOURCES.find do |name, _smoke_source|
|
|
105
|
+
!::TreeSitterLanguagePack.respond_to?(:has_language) ||
|
|
106
|
+
::TreeSitterLanguagePack.has_language(name)
|
|
107
|
+
end
|
|
108
|
+
return false unless language_name
|
|
109
|
+
|
|
110
|
+
smoke_parse_language(language_name, source: source)
|
|
111
|
+
end
|
|
112
|
+
|
|
113
|
+
def smoke_parse_language(language_name, source: smoke_source_for(language_name))
|
|
114
|
+
name = language_name.to_s
|
|
115
|
+
if ::TreeSitterLanguagePack.respond_to?(:has_language) &&
|
|
116
|
+
!::TreeSitterLanguagePack.has_language(name)
|
|
117
|
+
@unavailable_reason = "tree_sitter_language_pack does not publish #{name}"
|
|
118
|
+
return false
|
|
119
|
+
end
|
|
120
|
+
|
|
121
|
+
parser = ::TreeSitterLanguagePack.get_parser(name)
|
|
122
|
+
return false unless parser
|
|
123
|
+
|
|
124
|
+
tree = parser.parse(source)
|
|
125
|
+
return false unless tree&.respond_to?(:root_node)
|
|
126
|
+
|
|
127
|
+
root = tree.root_node
|
|
128
|
+
root && !node_has_error?(root)
|
|
129
|
+
end
|
|
130
|
+
|
|
131
|
+
def smoke_source_for(language_name)
|
|
132
|
+
PARSER_SMOKE_SOURCES.fetch(language_name.to_s, DEFAULT_PARSER_SMOKE_SOURCE)
|
|
133
|
+
end
|
|
134
|
+
|
|
135
|
+
def node_has_error?(node)
|
|
136
|
+
if node.respond_to?(:has_error)
|
|
137
|
+
node.has_error
|
|
138
|
+
elsif node.respond_to?(:has_error?)
|
|
139
|
+
node.has_error?
|
|
140
|
+
else
|
|
141
|
+
false
|
|
142
|
+
end
|
|
143
|
+
end
|
|
144
|
+
end
|
|
145
|
+
|
|
146
|
+
class Language < TreeHaver::Base::Language
|
|
147
|
+
def initialize(name)
|
|
148
|
+
super(name.to_sym, backend: :tslp, options: {})
|
|
149
|
+
end
|
|
150
|
+
|
|
151
|
+
class << self
|
|
152
|
+
def from_library(_path = nil, symbol: nil, name: nil) # rubocop:disable Lint/UnusedMethodArgument
|
|
153
|
+
new(name || :unknown)
|
|
154
|
+
end
|
|
155
|
+
end
|
|
156
|
+
end
|
|
157
|
+
|
|
158
|
+
class Parser < TreeHaver::Base::Parser
|
|
159
|
+
def parse(source)
|
|
160
|
+
raise TreeHaver::NotAvailable, unavailable_message unless Tslp.available?
|
|
161
|
+
raise TreeHaver::NotAvailable, 'TSLP language is not set' unless language
|
|
162
|
+
|
|
163
|
+
parser = ::TreeSitterLanguagePack.get_parser(language.name.to_s)
|
|
164
|
+
raise TreeHaver::NotAvailable, "TSLP did not return a parser for #{language.name}" unless parser
|
|
165
|
+
|
|
166
|
+
raw_tree = parser.parse(source)
|
|
167
|
+
raise TreeHaver::NotAvailable, "TSLP did not return a parse tree for #{language.name}" unless raw_tree
|
|
168
|
+
|
|
169
|
+
Tree.new(raw_tree, source: source, language: language.name)
|
|
170
|
+
end
|
|
171
|
+
|
|
172
|
+
private
|
|
173
|
+
|
|
174
|
+
def unavailable_message
|
|
175
|
+
reason = Tslp.unavailable_reason
|
|
176
|
+
detail = reason.to_s.empty? ? 'unknown reason' : reason
|
|
177
|
+
"tree_sitter_language_pack parser API is unavailable: #{detail}"
|
|
178
|
+
end
|
|
179
|
+
end
|
|
180
|
+
|
|
181
|
+
class Tree < TreeHaver::Base::Tree
|
|
182
|
+
attr_reader :language
|
|
183
|
+
|
|
184
|
+
def initialize(inner_tree = nil, source: nil, lines: nil, language: nil)
|
|
185
|
+
super(inner_tree, source: source, lines: lines)
|
|
186
|
+
@language = language.to_s
|
|
187
|
+
end
|
|
188
|
+
|
|
189
|
+
def root_node
|
|
190
|
+
Node.new(inner_tree.root_node, source: source, lines: lines, language: language)
|
|
191
|
+
end
|
|
192
|
+
end
|
|
193
|
+
|
|
194
|
+
class Node < TreeHaver::Base::Node
|
|
195
|
+
NODE_TYPE_ALIASES = {
|
|
196
|
+
'json5' => {
|
|
197
|
+
'file' => 'document',
|
|
198
|
+
'member' => 'pair'
|
|
199
|
+
}
|
|
200
|
+
}.freeze
|
|
201
|
+
|
|
202
|
+
attr_reader :language
|
|
203
|
+
|
|
204
|
+
def initialize(node, source: nil, lines: nil, language: nil)
|
|
205
|
+
super(node, source: source, lines: lines)
|
|
206
|
+
@language = language.to_s
|
|
207
|
+
end
|
|
208
|
+
|
|
209
|
+
def type
|
|
210
|
+
NODE_TYPE_ALIASES.fetch(language, {}).fetch(native_type, native_type)
|
|
211
|
+
end
|
|
212
|
+
|
|
213
|
+
def native_type
|
|
214
|
+
inner_node.kind
|
|
215
|
+
end
|
|
216
|
+
|
|
217
|
+
def start_byte
|
|
218
|
+
inner_node.start_byte
|
|
219
|
+
end
|
|
220
|
+
|
|
221
|
+
def end_byte
|
|
222
|
+
inner_node.end_byte
|
|
223
|
+
end
|
|
224
|
+
|
|
225
|
+
def start_point
|
|
226
|
+
point = inner_node.start_position
|
|
227
|
+
{ row: point.row, column: point.column }
|
|
228
|
+
end
|
|
229
|
+
|
|
230
|
+
def end_point
|
|
231
|
+
point = inner_node.end_position
|
|
232
|
+
{ row: point.row, column: point.column }
|
|
233
|
+
end
|
|
234
|
+
|
|
235
|
+
def children
|
|
236
|
+
Array.new(inner_node.child_count) do |index|
|
|
237
|
+
child = inner_node.child(index)
|
|
238
|
+
child && self.class.new(child, source: source, lines: lines, language: language)
|
|
239
|
+
end.compact
|
|
240
|
+
end
|
|
241
|
+
|
|
242
|
+
def child_by_field_name(name)
|
|
243
|
+
child = inner_node.child_by_field_name(name.to_s) if inner_node.respond_to?(:child_by_field_name)
|
|
244
|
+
child && self.class.new(child, source: source, lines: lines, language: language)
|
|
245
|
+
end
|
|
246
|
+
|
|
247
|
+
def parent
|
|
248
|
+
parent = inner_node.parent if inner_node.respond_to?(:parent)
|
|
249
|
+
parent && self.class.new(parent, source: source, lines: lines, language: language)
|
|
250
|
+
end
|
|
251
|
+
|
|
252
|
+
def named?
|
|
253
|
+
inner_node.is_named
|
|
254
|
+
end
|
|
255
|
+
|
|
256
|
+
def has_error?
|
|
257
|
+
inner_node.has_error
|
|
258
|
+
end
|
|
259
|
+
|
|
260
|
+
def error?
|
|
261
|
+
inner_node.is_error
|
|
262
|
+
end
|
|
263
|
+
|
|
264
|
+
def missing?
|
|
265
|
+
inner_node.is_missing
|
|
266
|
+
end
|
|
267
|
+
|
|
268
|
+
def extra?
|
|
269
|
+
inner_node.is_extra
|
|
270
|
+
end
|
|
271
|
+
end
|
|
272
|
+
end
|
|
273
|
+
end
|
|
274
|
+
end
|
|
@@ -0,0 +1,320 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module TreeHaver
|
|
4
|
+
module Base
|
|
5
|
+
# Base class for backend comment wrappers.
|
|
6
|
+
#
|
|
7
|
+
# This defines the parser-facing contract for normalized comment wrappers in
|
|
8
|
+
# TreeHaver. Backends that can expose comment objects should subclass this and
|
|
9
|
+
# implement text/type/location accessors using their native parser data.
|
|
10
|
+
class Comment
|
|
11
|
+
ATTACHMENT_HINTS = %i[leading inline trailing].freeze
|
|
12
|
+
|
|
13
|
+
# The underlying backend-specific comment object.
|
|
14
|
+
# @return [Object]
|
|
15
|
+
attr_reader :inner_comment
|
|
16
|
+
|
|
17
|
+
# The source text used for fallback range extraction.
|
|
18
|
+
# @return [String, nil]
|
|
19
|
+
attr_reader :source
|
|
20
|
+
|
|
21
|
+
# Optional parser-provided attachment hint.
|
|
22
|
+
# @return [Symbol, nil]
|
|
23
|
+
attr_reader :attachment_hint
|
|
24
|
+
|
|
25
|
+
def initialize(comment, source: nil, attachment_hint: nil)
|
|
26
|
+
@inner_comment = comment
|
|
27
|
+
@source = source
|
|
28
|
+
@attachment_hint = normalize_attachment_hint(attachment_hint)
|
|
29
|
+
end
|
|
30
|
+
|
|
31
|
+
# Get the normalized comment type.
|
|
32
|
+
# Examples: "inline_comment", "block_comment".
|
|
33
|
+
# @return [String]
|
|
34
|
+
def type
|
|
35
|
+
raise NotImplementedError, "#{self.class}#type must be implemented"
|
|
36
|
+
end
|
|
37
|
+
|
|
38
|
+
# Get the comment text including delimiters when appropriate.
|
|
39
|
+
# @return [String]
|
|
40
|
+
def text
|
|
41
|
+
raise NotImplementedError, "#{self.class}#text must be implemented"
|
|
42
|
+
end
|
|
43
|
+
|
|
44
|
+
# Get the start byte offset of the comment.
|
|
45
|
+
# @return [Integer]
|
|
46
|
+
def start_byte
|
|
47
|
+
raise NotImplementedError, "#{self.class}#start_byte must be implemented"
|
|
48
|
+
end
|
|
49
|
+
|
|
50
|
+
# Get the end byte offset of the comment.
|
|
51
|
+
# @return [Integer]
|
|
52
|
+
def end_byte
|
|
53
|
+
raise NotImplementedError, "#{self.class}#end_byte must be implemented"
|
|
54
|
+
end
|
|
55
|
+
|
|
56
|
+
# Get the start position (row/column, 0-based).
|
|
57
|
+
# @return [Hash{Symbol => Integer}]
|
|
58
|
+
def start_point
|
|
59
|
+
{ row: 0, column: 0 }
|
|
60
|
+
end
|
|
61
|
+
|
|
62
|
+
# Get the end position (row/column, 0-based).
|
|
63
|
+
# @return [Hash{Symbol => Integer}]
|
|
64
|
+
def end_point
|
|
65
|
+
{ row: 0, column: 0 }
|
|
66
|
+
end
|
|
67
|
+
|
|
68
|
+
# Get the normalized delimiter style.
|
|
69
|
+
#
|
|
70
|
+
# @return [Symbol, nil]
|
|
71
|
+
def style
|
|
72
|
+
nil
|
|
73
|
+
end
|
|
74
|
+
|
|
75
|
+
# Get the opening delimiter for the comment, when the backend can provide it.
|
|
76
|
+
#
|
|
77
|
+
# Examples:
|
|
78
|
+
# - `#` for hash comments
|
|
79
|
+
# - `//` for C-style line comments
|
|
80
|
+
# - `<!--` for HTML/XML block comments
|
|
81
|
+
#
|
|
82
|
+
# @return [String, nil]
|
|
83
|
+
def opening_delimiter
|
|
84
|
+
nil
|
|
85
|
+
end
|
|
86
|
+
|
|
87
|
+
# Get the closing delimiter for the comment, when applicable.
|
|
88
|
+
#
|
|
89
|
+
# This is typically `nil` for line comments.
|
|
90
|
+
#
|
|
91
|
+
# @return [String, nil]
|
|
92
|
+
def closing_delimiter
|
|
93
|
+
nil
|
|
94
|
+
end
|
|
95
|
+
|
|
96
|
+
# Get the comment body with outer delimiters removed when possible.
|
|
97
|
+
#
|
|
98
|
+
# Backends with richer native comment models should override this when they
|
|
99
|
+
# can provide a more semantically accurate body than simple delimiter
|
|
100
|
+
# trimming.
|
|
101
|
+
#
|
|
102
|
+
# @return [String]
|
|
103
|
+
def body_text
|
|
104
|
+
extract_body_text(text.to_s)
|
|
105
|
+
end
|
|
106
|
+
|
|
107
|
+
# Get a matching-friendly normalized comment body.
|
|
108
|
+
#
|
|
109
|
+
# This strips leading and trailing whitespace from {#body_text} while
|
|
110
|
+
# preserving the raw rendered form in {#text}.
|
|
111
|
+
#
|
|
112
|
+
# @return [String]
|
|
113
|
+
def normalized_text
|
|
114
|
+
body_text.strip
|
|
115
|
+
end
|
|
116
|
+
|
|
117
|
+
# Get delimiter/body metadata in one normalized hash.
|
|
118
|
+
#
|
|
119
|
+
# @return [Hash{Symbol => String, nil}]
|
|
120
|
+
def delimiter_metadata
|
|
121
|
+
{
|
|
122
|
+
opening: opening_delimiter,
|
|
123
|
+
closing: closing_delimiter,
|
|
124
|
+
body: body_text
|
|
125
|
+
}
|
|
126
|
+
end
|
|
127
|
+
|
|
128
|
+
# Whether this comment uses a line-comment style.
|
|
129
|
+
#
|
|
130
|
+
# @return [Boolean]
|
|
131
|
+
def line?
|
|
132
|
+
style == :line
|
|
133
|
+
end
|
|
134
|
+
|
|
135
|
+
# Whether this comment uses a block-comment style.
|
|
136
|
+
#
|
|
137
|
+
# @return [Boolean]
|
|
138
|
+
def block?
|
|
139
|
+
style == :block
|
|
140
|
+
end
|
|
141
|
+
|
|
142
|
+
# Whether this comment spans multiple source lines.
|
|
143
|
+
#
|
|
144
|
+
# @return [Boolean]
|
|
145
|
+
def multiline?
|
|
146
|
+
start_line != end_line
|
|
147
|
+
end
|
|
148
|
+
|
|
149
|
+
# Whether this comment has a parser-provided attachment hint.
|
|
150
|
+
#
|
|
151
|
+
# @return [Boolean]
|
|
152
|
+
def attachment_hint?
|
|
153
|
+
!attachment_hint.nil?
|
|
154
|
+
end
|
|
155
|
+
|
|
156
|
+
# Whether this comment is hinted as leading its owner.
|
|
157
|
+
#
|
|
158
|
+
# @return [Boolean]
|
|
159
|
+
def leading?
|
|
160
|
+
attachment_hint == :leading
|
|
161
|
+
end
|
|
162
|
+
|
|
163
|
+
# Whether this comment is hinted as inline with its owner.
|
|
164
|
+
#
|
|
165
|
+
# @return [Boolean]
|
|
166
|
+
def inline?
|
|
167
|
+
attachment_hint == :inline
|
|
168
|
+
end
|
|
169
|
+
|
|
170
|
+
# Whether this comment is hinted as trailing its owner.
|
|
171
|
+
#
|
|
172
|
+
# @return [Boolean]
|
|
173
|
+
def trailing?
|
|
174
|
+
attachment_hint == :trailing
|
|
175
|
+
end
|
|
176
|
+
|
|
177
|
+
# Get the 1-based start line.
|
|
178
|
+
#
|
|
179
|
+
# @return [Integer]
|
|
180
|
+
def start_line
|
|
181
|
+
start_point[:row] + 1
|
|
182
|
+
end
|
|
183
|
+
|
|
184
|
+
# Get the 1-based end line.
|
|
185
|
+
#
|
|
186
|
+
# @return [Integer]
|
|
187
|
+
def end_line
|
|
188
|
+
end_point[:row] + 1
|
|
189
|
+
end
|
|
190
|
+
|
|
191
|
+
# Get a normalized source-position hash.
|
|
192
|
+
#
|
|
193
|
+
# @return [Hash{Symbol => Integer}]
|
|
194
|
+
def source_position
|
|
195
|
+
{
|
|
196
|
+
start_line: start_line,
|
|
197
|
+
end_line: end_line,
|
|
198
|
+
start_column: start_point[:column],
|
|
199
|
+
end_column: end_point[:column]
|
|
200
|
+
}
|
|
201
|
+
end
|
|
202
|
+
|
|
203
|
+
# Whether this comment starts at the first line of the available source.
|
|
204
|
+
#
|
|
205
|
+
# @return [Boolean]
|
|
206
|
+
def at_file_start?
|
|
207
|
+
start_line == 1
|
|
208
|
+
end
|
|
209
|
+
|
|
210
|
+
# Whether this comment ends at the final line of the available source.
|
|
211
|
+
#
|
|
212
|
+
# @return [Boolean]
|
|
213
|
+
def at_file_end?
|
|
214
|
+
return false if source_lines.none?
|
|
215
|
+
|
|
216
|
+
end_line >= source_lines.length
|
|
217
|
+
end
|
|
218
|
+
|
|
219
|
+
# Return the contiguous blank source lines immediately before this comment.
|
|
220
|
+
#
|
|
221
|
+
# This is parser-facing layout metadata. It does not decide ownership; it
|
|
222
|
+
# simply exposes spacing adjacent to the comment so merge layers can reason
|
|
223
|
+
# about floating vs attached behavior without rescanning source text.
|
|
224
|
+
#
|
|
225
|
+
# @return [Array<String>]
|
|
226
|
+
def blank_lines_before
|
|
227
|
+
return [] if at_file_start?
|
|
228
|
+
|
|
229
|
+
line_num = start_line - 1
|
|
230
|
+
blanks = []
|
|
231
|
+
|
|
232
|
+
while line_num >= 1
|
|
233
|
+
line = source_line(line_num)
|
|
234
|
+
break unless line && line.strip.empty?
|
|
235
|
+
|
|
236
|
+
blanks.unshift(line)
|
|
237
|
+
line_num -= 1
|
|
238
|
+
end
|
|
239
|
+
|
|
240
|
+
blanks
|
|
241
|
+
end
|
|
242
|
+
|
|
243
|
+
# Return the contiguous blank source lines immediately after this comment.
|
|
244
|
+
#
|
|
245
|
+
# @return [Array<String>]
|
|
246
|
+
def blank_lines_after
|
|
247
|
+
return [] if at_file_end?
|
|
248
|
+
|
|
249
|
+
line_num = end_line + 1
|
|
250
|
+
blanks = []
|
|
251
|
+
|
|
252
|
+
while line_num <= source_lines.length
|
|
253
|
+
line = source_line(line_num)
|
|
254
|
+
break unless line && line.strip.empty?
|
|
255
|
+
|
|
256
|
+
blanks << line
|
|
257
|
+
line_num += 1
|
|
258
|
+
end
|
|
259
|
+
|
|
260
|
+
blanks
|
|
261
|
+
end
|
|
262
|
+
|
|
263
|
+
# Return the count of immediately preceding blank source lines.
|
|
264
|
+
#
|
|
265
|
+
# @return [Integer]
|
|
266
|
+
def blank_line_count_before
|
|
267
|
+
blank_lines_before.length
|
|
268
|
+
end
|
|
269
|
+
|
|
270
|
+
# Return the count of immediately following blank source lines.
|
|
271
|
+
#
|
|
272
|
+
# @return [Integer]
|
|
273
|
+
def blank_line_count_after
|
|
274
|
+
blank_lines_after.length
|
|
275
|
+
end
|
|
276
|
+
|
|
277
|
+
def inspect
|
|
278
|
+
"#<#{self.class} type=#{type.inspect} range=#{start_byte}...#{end_byte}>"
|
|
279
|
+
end
|
|
280
|
+
|
|
281
|
+
private
|
|
282
|
+
|
|
283
|
+
def source_lines
|
|
284
|
+
@source_lines ||= begin
|
|
285
|
+
values = source.to_s.split("\n", -1)
|
|
286
|
+
values.pop if values.last&.empty? && source.to_s.end_with?("\n")
|
|
287
|
+
values
|
|
288
|
+
end
|
|
289
|
+
end
|
|
290
|
+
|
|
291
|
+
def source_line(line_number)
|
|
292
|
+
return if line_number < 1 || line_number > source_lines.length
|
|
293
|
+
|
|
294
|
+
source_lines[line_number - 1]
|
|
295
|
+
end
|
|
296
|
+
|
|
297
|
+
def extract_body_text(raw_text)
|
|
298
|
+
text_without_opening = if opening_delimiter
|
|
299
|
+
raw_text.sub(/\A#{Regexp.escape(opening_delimiter)}[ \t]?/, '')
|
|
300
|
+
else
|
|
301
|
+
raw_text
|
|
302
|
+
end
|
|
303
|
+
|
|
304
|
+
return text_without_opening unless closing_delimiter
|
|
305
|
+
|
|
306
|
+
text_without_opening.sub(/[ \t]*#{Regexp.escape(closing_delimiter)}\z/m, '')
|
|
307
|
+
end
|
|
308
|
+
|
|
309
|
+
def normalize_attachment_hint(hint)
|
|
310
|
+
return if hint.nil?
|
|
311
|
+
|
|
312
|
+
normalized = hint.to_sym
|
|
313
|
+
return normalized if ATTACHMENT_HINTS.include?(normalized)
|
|
314
|
+
|
|
315
|
+
raise ArgumentError,
|
|
316
|
+
"Unknown comment attachment hint: #{hint.inspect}. Expected one of: #{ATTACHMENT_HINTS.join(', ')}"
|
|
317
|
+
end
|
|
318
|
+
end
|
|
319
|
+
end
|
|
320
|
+
end
|
|
@@ -0,0 +1,98 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module TreeHaver
|
|
4
|
+
module Base
|
|
5
|
+
# Base class for backend Language implementations
|
|
6
|
+
#
|
|
7
|
+
# This class defines the API contract for all language implementations.
|
|
8
|
+
# Backend-specific Language classes should inherit from this and implement
|
|
9
|
+
# the required interface.
|
|
10
|
+
#
|
|
11
|
+
# @abstract Subclasses must implement #name and #backend at minimum
|
|
12
|
+
class Language
|
|
13
|
+
include Comparable
|
|
14
|
+
|
|
15
|
+
# The language name (e.g., :markdown, :ruby, :json)
|
|
16
|
+
# @return [Symbol] Language name
|
|
17
|
+
attr_reader :name
|
|
18
|
+
|
|
19
|
+
# The backend this language is for
|
|
20
|
+
# @return [Symbol] Backend identifier (e.g., :commonmarker, :markly, :prism)
|
|
21
|
+
attr_reader :backend
|
|
22
|
+
|
|
23
|
+
# Language-specific options
|
|
24
|
+
# @return [Hash] Options hash
|
|
25
|
+
attr_reader :options
|
|
26
|
+
|
|
27
|
+
# Create a new Language instance
|
|
28
|
+
#
|
|
29
|
+
# @param name [Symbol, String] Language name
|
|
30
|
+
# @param backend [Symbol] Backend identifier
|
|
31
|
+
# @param options [Hash] Backend-specific options
|
|
32
|
+
def initialize(name, backend:, options: {})
|
|
33
|
+
@name = name.to_sym
|
|
34
|
+
@backend = backend.to_sym
|
|
35
|
+
@options = options
|
|
36
|
+
end
|
|
37
|
+
|
|
38
|
+
# Alias for name (tree-sitter compatibility)
|
|
39
|
+
alias language_name name
|
|
40
|
+
|
|
41
|
+
# -- Shared Implementation ------------------------------------------------
|
|
42
|
+
|
|
43
|
+
# Comparison based on backend then name
|
|
44
|
+
# @param other [Object]
|
|
45
|
+
# @return [Integer, nil]
|
|
46
|
+
def <=>(other)
|
|
47
|
+
return unless other.is_a?(Language)
|
|
48
|
+
return unless other.respond_to?(:backend) && other.backend == backend
|
|
49
|
+
|
|
50
|
+
name <=> other.name
|
|
51
|
+
end
|
|
52
|
+
|
|
53
|
+
# Hash value for use in Sets/Hashes
|
|
54
|
+
# @return [Integer]
|
|
55
|
+
def hash
|
|
56
|
+
[backend, name, options.to_a.sort].hash
|
|
57
|
+
end
|
|
58
|
+
|
|
59
|
+
# Equality check for Hash keys
|
|
60
|
+
# @param other [Object]
|
|
61
|
+
# @return [Boolean]
|
|
62
|
+
def eql?(other)
|
|
63
|
+
return false unless other.is_a?(Language)
|
|
64
|
+
|
|
65
|
+
backend == other.backend && name == other.name && options == other.options
|
|
66
|
+
end
|
|
67
|
+
|
|
68
|
+
# Human-readable representation
|
|
69
|
+
# @return [String]
|
|
70
|
+
def inspect
|
|
71
|
+
opts = options.empty? ? '' : " options=#{options}"
|
|
72
|
+
class_name = self.class.name || "#{self.class.superclass.name}(anonymous)"
|
|
73
|
+
"#<#{class_name} name=#{name} backend=#{backend}#{opts}>"
|
|
74
|
+
end
|
|
75
|
+
|
|
76
|
+
# -- Class Methods --------------------------------------------------------
|
|
77
|
+
|
|
78
|
+
class << self
|
|
79
|
+
# Load a language from a library path (factory method)
|
|
80
|
+
#
|
|
81
|
+
# For pure-Ruby backends (Commonmarker, Markly, Prism, Psych), this
|
|
82
|
+
# typically ignores the path and returns the single supported language.
|
|
83
|
+
#
|
|
84
|
+
# For tree-sitter backends (MRI, Rust, FFI, Java), this loads the
|
|
85
|
+
# language from the shared library file.
|
|
86
|
+
#
|
|
87
|
+
# @param _path [String, nil] Path to shared library (optional for pure-Ruby)
|
|
88
|
+
# @param symbol [String, nil] Symbol name to load (optional)
|
|
89
|
+
# @param name [String, nil] Language name hint (optional)
|
|
90
|
+
# @return [Language] Loaded language instance
|
|
91
|
+
# @raise [NotImplementedError] If not implemented by subclass
|
|
92
|
+
def from_library(_path = nil, symbol: nil, name: nil)
|
|
93
|
+
raise NotImplementedError, "#{self}.from_library must be implemented"
|
|
94
|
+
end
|
|
95
|
+
end
|
|
96
|
+
end
|
|
97
|
+
end
|
|
98
|
+
end
|