markbridge 0.2.1 → 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
checksums.yaml CHANGED
@@ -1,7 +1,7 @@
1
1
  ---
2
2
  SHA256:
3
- metadata.gz: 0cee12c41cea526db53cf389e2cc2bc49310568c0ebe9979d03daa7a4c2d23ac
4
- data.tar.gz: a6d541954eea2357b6bb31bcfb685a634383de405dd1f505b257c78007e087d4
3
+ metadata.gz: dbebafc60a3665c98544f3cfd9496c3c5cdc2c0940339c998791b2df1f04aa44
4
+ data.tar.gz: 2be35c042f803e37302fa149af772897326241488a9429a525ddc714af0623d1
5
5
  SHA512:
6
- metadata.gz: b7ddd72dd5d0aa197a7b2477f1bfddd72e9cf7a01f26b4def8da341c8edd739503c04a23a501175adbcc23e23064578a2a4bcb17f3518b543f2716059636e3e7
7
- data.tar.gz: 409dcbed4de6b6551c420ca51a00ed998abf11aeac8bd3bce9522b49a5ca07d3e524d68737b9461ddb6e3949485c1c84123c30c8e6e44cb57d7a99d873c7c3af
6
+ metadata.gz: 6b3e3dc46705d4f7cf52aabaac8b272cbc9241e57473caf11a6c53b1d1355ec2e7874ae485732ba526f67445144f12daccc521dae40a6405994e409a34e00ede
7
+ data.tar.gz: 89385e8ec7942af10b79b267e6400bfb209af7871775e0891f86153e792e3ef61b15b08eb025f9a6f6d5673c3aaf9bcbb6ec26200ac3a62440b426121fcba1fa
@@ -4,42 +4,72 @@ module Markbridge
4
4
  module AST
5
5
  # Represents a quote/blockquote element.
6
6
  #
7
+ # The attribution fields follow Discourse's quote format
8
+ # (`[quote="username, post:2, topic:456"]`): +post_number+ is the
9
+ # position of the quoted post *within its topic* (not a database
10
+ # id), +topic_id+ is the topic's id. Sources that attribute quotes
11
+ # by database id instead (phpBB, XenForo) use +post_id+ and
12
+ # +user_id+ — those never feed a Discourse +post:+ reference and
13
+ # are carried for consumers to remap.
14
+ #
7
15
  # @example Basic quote
8
16
  # quote = AST::Quote.new
9
17
  # quote << AST::Text.new("quoted text")
10
18
  #
11
19
  # @example Quote with author attribution
12
- # quote = AST::Quote.new(author: "John")
20
+ # quote = AST::Quote.new(author: "alice")
13
21
  # quote << AST::Text.new("quoted text")
14
22
  #
15
23
  # @example Quote with full Discourse context
16
- # quote = AST::Quote.new(author: "John", post: "123", topic: "456", username: "john123")
24
+ # quote = AST::Quote.new(author: "alice", post_number: 123, topic_id: 456, username: "alice_b")
17
25
  # quote << AST::Text.new("quoted text")
18
26
  class Quote < Element
19
- # @return [String, nil] the author/username of the quote
27
+ # @return [String, nil] the author attribution (display name)
20
28
  attr_reader :author
21
29
 
22
- # @return [String, nil] the post ID for Discourse quotes
23
- attr_reader :post
30
+ # @return [Integer, nil] the quoted post's number within its topic
31
+ # (Discourse semantics — not a post id)
32
+ attr_reader :post_number
33
+
34
+ # @return [Integer, nil] the quoted post's database id, for sources
35
+ # that attribute quotes by id (phpBB's +post_id+, XenForo's
36
+ # +post:+). Distinct from {#post_number}; parsers map whichever
37
+ # their dialect provides.
38
+ attr_reader :post_id
24
39
 
25
- # @return [String, nil] the topic ID for Discourse quotes
26
- attr_reader :topic
40
+ # @return [Integer, nil] the topic id the quoted post belongs to
41
+ attr_reader :topic_id
27
42
 
28
- # @return [String, nil] the username for Discourse quotes
43
+ # @return [String, nil] the quoted user's username
29
44
  attr_reader :username
30
45
 
46
+ # @return [Integer, nil] the quoted user's id, for sources that
47
+ # attribute quotes by id (usernames may be unknown or stale)
48
+ attr_reader :user_id
49
+
31
50
  # Create a new Quote element.
32
51
  #
33
- # @param author [String, nil] the author attribution
34
- # @param post [String, nil] the post ID (Discourse-specific)
35
- # @param topic [String, nil] the topic ID (Discourse-specific)
36
- # @param username [String, nil] the username (Discourse-specific)
37
- def initialize(author: nil, post: nil, topic: nil, username: nil)
52
+ # @param author [String, nil] the author attribution (display name)
53
+ # @param post_number [Integer, nil] the quoted post's number within its topic
54
+ # @param post_id [Integer, nil] the quoted post's database id
55
+ # @param topic_id [Integer, nil] the topic id
56
+ # @param username [String, nil] the quoted user's username
57
+ # @param user_id [Integer, nil] the quoted user's id
58
+ def initialize(
59
+ author: nil,
60
+ post_number: nil,
61
+ post_id: nil,
62
+ topic_id: nil,
63
+ username: nil,
64
+ user_id: nil
65
+ )
38
66
  super()
39
67
  @author = author
40
- @post = post
41
- @topic = topic
68
+ @post_number = post_number
69
+ @post_id = post_id
70
+ @topic_id = topic_id
42
71
  @username = username
72
+ @user_id = user_id
43
73
  end
44
74
  end
45
75
  end
@@ -19,13 +19,15 @@ module Markbridge
19
19
 
20
20
  # Create a new text node with the given content.
21
21
  #
22
- # The text is copied into a fresh mutable buffer so that subsequent
23
- # in-place mutations (e.g. via {#merge}) do not leak back to the caller's
24
- # original string.
22
+ # Frozen input is shared as-is (copy-on-write: {#merge} dups before
23
+ # its first append) the parser hot path always passes frozen token
24
+ # text, so no copy is made per text node. Mutable input is copied so
25
+ # that in-place mutations cannot leak in either direction between
26
+ # the caller's string and this node.
25
27
  #
26
28
  # @param text [String] the text content
27
29
  def initialize(text)
28
- @text = text.dup
30
+ @text = text.frozen? ? text : text.dup
29
31
  end
30
32
 
31
33
  # Merge another text node's content into this one.
@@ -34,6 +36,7 @@ module Markbridge
34
36
  # @param other [Text] the text node to merge from
35
37
  # @return [Text] self for method chaining
36
38
  def merge(other)
39
+ @text = @text.dup if @text.frozen?
37
40
  @text << other.text
38
41
  self
39
42
  end
@@ -22,6 +22,21 @@ module Markbridge
22
22
  super()
23
23
  @href = href
24
24
  end
25
+
26
+ # Whether this link is "bare" — it has no link text of its own:
27
+ # either no children, or a single Text child whose content is
28
+ # exactly the href (how parsers model a URL that stands for
29
+ # itself). Consumers that record or rewrite links need this
30
+ # judgment too (a bare URL must stay bare to keep autolinking
31
+ # and oneboxing working), so it lives on the node rather than
32
+ # in the renderer.
33
+ #
34
+ # @return [Boolean]
35
+ def bare?
36
+ return true if children.empty?
37
+
38
+ children.size == 1 && children.first.instance_of?(Text) && children.first.text == href
39
+ end
25
40
  end
26
41
  end
27
42
  end
@@ -62,7 +62,11 @@ module Markbridge
62
62
  # @param tag_name [String]
63
63
  # @return [BaseHandler, nil]
64
64
  def [](tag_name)
65
- @handlers[tag_name.to_s.downcase]
65
+ # Keys are always downcased strings and the scanner emits
66
+ # downcased tags, so the fetch hits directly on the hot path;
67
+ # the block normalizes only unusual lookups (mixed case,
68
+ # symbols) instead of allocating a downcased copy per token.
69
+ @handlers.fetch(tag_name) { @handlers[tag_name.to_s.downcase] }
66
70
  end
67
71
 
68
72
  # Get handler for an element instance
@@ -176,6 +180,28 @@ module Markbridge
176
180
  registry
177
181
  end
178
182
 
183
+ # Shared, deep-frozen default registry for the no-customization
184
+ # fast path. Built once per process; {Parser} falls back to it
185
+ # when no +handlers:+ are given, skipping the full registry
186
+ # construction on every parse. Handlers and closing strategies
187
+ # are stateless after construction, so sharing is safe across
188
+ # parsers and threads.
189
+ #
190
+ # @return [HandlerRegistry] the same frozen instance on every call
191
+ def self.shared_default
192
+ @shared_default ||= default.freeze
193
+ end
194
+
195
+ # Freeze the registry together with its internal collections so
196
+ # that registration on a shared instance fails loudly instead of
197
+ # silently mutating state visible to every parser.
198
+ def freeze
199
+ @handlers.freeze
200
+ @element_handlers.freeze
201
+ @auto_closeable_elements.freeze
202
+ super
203
+ end
204
+
179
205
  # Build a registry from the default configuration with optional customization
180
206
  # @yield [HandlerRegistry] the registry to customize
181
207
  # @return [HandlerRegistry]
@@ -10,7 +10,14 @@ module Markbridge
10
10
  # - [quote=author]text[/quote]
11
11
  # - [quote="author"]text[/quote]
12
12
  # - [quote author=username]text[/quote]
13
- # - [quote="username, post:123, topic:456"]text[/quote] (Discourse format)
13
+ # - [quote="username, post:2, topic:456"]text[/quote] (Discourse format)
14
+ #
15
+ # Attribution semantics are Discourse's: `post:` is the post's
16
+ # number within its topic and `topic:` is the topic id. Beware
17
+ # when feeding other dialects through this handler — XenForo
18
+ # attributions ("name, post: 12345, member: 678") also match the
19
+ # `post:` pattern, but there the value is a database post id, not
20
+ # a post number.
14
21
  class QuoteHandler < BaseHandler
15
22
  def initialize
16
23
  @element_class = AST::Quote
@@ -27,13 +34,13 @@ module Markbridge
27
34
  private
28
35
 
29
36
  def extract_quote_attrs(token)
30
- author, post, topic, username = extract_from_option(token)
37
+ author, post_number, topic_id, username = extract_from_option(token)
31
38
  author ||= token.attrs[:author]
32
39
 
33
40
  {
34
41
  author:,
35
- post: token.attrs[:post] || post,
36
- topic: token.attrs[:topic] || topic,
42
+ post_number: integer_or_nil(token.attrs[:post]) || post_number,
43
+ topic_id: integer_or_nil(token.attrs[:topic]) || topic_id,
37
44
  username: token.attrs[:username] || username,
38
45
  }
39
46
  end
@@ -42,15 +49,31 @@ module Markbridge
42
49
  option = token.attrs[:option]
43
50
  return nil, nil, nil, nil unless option
44
51
 
45
- post = option[/,\s*post:(\d+)/, 1]
46
- return option, nil, nil, nil unless post
52
+ post_number = option[/,\s*post:(\d+)/, 1]
53
+ return option, nil, nil, nil unless post_number
47
54
 
48
- # Discourse format: "username, post:123, topic:456" (topic optional,
55
+ # Discourse format: "username, post:2, topic:456" (topic optional,
49
56
  # order irrelevant between post: and topic:).
50
57
  username = option.split(",").first.strip
51
- topic = option[/,\s*topic:(\d+)/, 1]
58
+ topic_id = option[/,\s*topic:(\d+)/, 1]
52
59
 
53
- [username, post, topic, username]
60
+ # The regex captures are guaranteed digit runs, so strict
61
+ # Integer() applies. Base 10 is explicit: without it a
62
+ # leading zero switches Integer() to octal and "099" raises.
63
+ [username, Integer(post_number, 10), topic_id && Integer(topic_id, 10), username]
64
+ end
65
+
66
+ # Coerce an attribute value to Integer; nil and non-numeric
67
+ # values (which would make a useless attribution anyway)
68
+ # become nil. Base 10 is explicit so leading zeros ("099")
69
+ # don't trip Integer()'s octal mode and prefix forms ("0x1A")
70
+ # don't smuggle in surprise values. The result is an unbounded
71
+ # Ruby Integer — a runaway digit run like
72
+ # "post:77777777777777777789999" parses to a bignum, so
73
+ # consumers binding these into fixed-width storage (int64
74
+ # columns) need their own bounds check.
75
+ def integer_or_nil(value)
76
+ Integer(value, 10, exception: false)
54
77
  end
55
78
  end
56
79
  end
@@ -11,6 +11,13 @@ module Markbridge
11
11
  def initialize(element_class, collector: RawContentCollector.new)
12
12
  @element_class = element_class
13
13
  @collector = collector
14
+ # Computed once here: handlers are long-lived, elements are
15
+ # per-tag, and the reflection is too costly to repeat per element.
16
+ @accepts_language =
17
+ element_class
18
+ .instance_method(:initialize)
19
+ .parameters
20
+ .any? { |_kind, name| name == :language }
14
21
  end
15
22
 
16
23
  def on_open(token:, context:, registry:, tokens:)
@@ -43,10 +50,7 @@ module Markbridge
43
50
  end
44
51
 
45
52
  def accepts_language?
46
- @element_class
47
- .instance_method(:initialize)
48
- .parameters
49
- .any? { |_kind, name| name == :language }
53
+ @accepts_language
50
54
  end
51
55
  end
52
56
  end
@@ -26,7 +26,7 @@ module Markbridge
26
26
  if block_given?
27
27
  HandlerRegistry.build_from_default(&block)
28
28
  else
29
- handlers || HandlerRegistry.default
29
+ handlers || HandlerRegistry.shared_default
30
30
  end
31
31
  @unknown_tags = Hash.new(0)
32
32
  end
@@ -53,11 +53,14 @@ module Markbridge
53
53
 
54
54
  private
55
55
 
56
+ LINE_ENDING_RE = /\r\n?|[\u2028\u2029]+/
57
+ private_constant :LINE_ENDING_RE
58
+
56
59
  # Normalize line endings (CR, CRLF, and Unicode separators)
57
60
  # @param input [String]
58
- # @return [String]
61
+ # @return [String] the input itself when already normalized (LF-only)
59
62
  def normalize_line_endings(input)
60
- input.gsub(/\r\n?|[\u2028\u2029]+/, "\n")
63
+ input.match?(LINE_ENDING_RE) ? input.gsub(LINE_ENDING_RE, "\n") : input
61
64
  end
62
65
 
63
66
  # Parse tokens using scanner
@@ -3,26 +3,39 @@
3
3
  module Markbridge
4
4
  module Parsers
5
5
  module BBCode
6
- # High-performance character-by-character BBCode scanner
6
+ # High-performance BBCode scanner
7
7
  # Tokenizes BBCode in O(n) time with minimal allocations and bounded backtracking
8
+ #
9
+ # The scanner works entirely in *byte* offsets (byteindex/byteslice/
10
+ # getbyte). CRuby has no character-index cache, so character-index
11
+ # operations like `str[pos]` walk the string from the start on any
12
+ # input containing a multibyte character — O(pos) per call and
13
+ # superlinear per document. Byte offsets keep multibyte input at
14
+ # near-ASCII cost. Invariant: `@current_pos` always sits on a
15
+ # character boundary — jumps land on matches of ASCII-only patterns
16
+ # and advances step over ASCII bytes or whole matches. Token
17
+ # positions are therefore byte offsets into the input.
8
18
  class Scanner
9
19
  def initialize(input)
10
20
  @input = input
11
- @length = input.length
21
+ @length = input.bytesize
12
22
  @current_pos = 0
13
23
  end
14
24
 
15
25
  def next_token
16
26
  return nil if end_of_input?
17
27
  start_pos = @current_pos
18
- bracket_index = @input.index("[", @current_pos)
28
+ # `byteindex` returns 0 for a match at the start — nil-check, never
29
+ # truthiness-check. Can't be 0 here though: callers guarantee
30
+ # @current_pos <= bracket_index.
31
+ bracket_index = @input.byteindex("[", @current_pos)
19
32
 
20
33
  if bracket_index.nil?
21
- text = @input[@current_pos..]
34
+ text = @input.byteslice(@current_pos, @length - @current_pos)
22
35
  @current_pos = @length
23
36
  TextToken.new(text:, pos: start_pos)
24
37
  elsif bracket_index > @current_pos
25
- text = @input[@current_pos...bracket_index]
38
+ text = @input.byteslice(@current_pos, bracket_index - @current_pos)
26
39
  @current_pos = bracket_index
27
40
  TextToken.new(text:, pos: start_pos)
28
41
  elsif (tag_token = parse_tag_at_cursor)
@@ -35,31 +48,65 @@ module Markbridge
35
48
 
36
49
  private
37
50
 
38
- TAG_INITIAL_CHAR = /[a-z*]/i
39
- TAG_NAME_CHAR = /[a-z0-9]/i
40
- UID_HEX_CHAR = /[0-9a-f]/i
41
- ATTR_NAME_CHAR = /\w/
42
- WHITESPACE_CHAR = /\s/
51
+ # Byte constants for tag scanning. A byte >= 0x80 (multibyte
52
+ # lead/continuation) never satisfies any of the ASCII predicates
53
+ # below, so no encoding-awareness is needed at probe sites.
54
+ TAB = 9 # \t
55
+ CR = 13 # \r (\s == [ \t\n\v\f\r] == 0x20, 0x09..0x0D)
56
+ SPACE = 32
57
+ DOUBLE_QUOTE = 34 # "
58
+ SINGLE_QUOTE = 39 # '
59
+ STAR = 42 # *
60
+ SLASH = 47 # /
61
+ DIGIT_0 = 48
62
+ DIGIT_9 = 57
63
+ COLON = 58 # :
64
+ EQUALS = 61 # =
65
+ UPPER_A = 65
66
+ UPPER_F = 70
67
+ UPPER_Z = 90
68
+ BRACKET_CLOSE = 93 # ]
69
+ UNDERSCORE = 95 # _
70
+ LOWER_A = 97
71
+ LOWER_F = 102
72
+ LOWER_Z = 122
73
+
74
+ # Characters an unquoted attribute value stops at. ASCII-only, so a
75
+ # byteindex match always lands on a character boundary.
43
76
  UNQUOTED_VALUE_STOP = /[\[\]\s]/
44
77
 
45
- private_constant :TAG_INITIAL_CHAR,
46
- :TAG_NAME_CHAR,
47
- :UID_HEX_CHAR,
48
- :ATTR_NAME_CHAR,
49
- :WHITESPACE_CHAR,
78
+ private_constant :TAB,
79
+ :CR,
80
+ :SPACE,
81
+ :DOUBLE_QUOTE,
82
+ :SINGLE_QUOTE,
83
+ :STAR,
84
+ :SLASH,
85
+ :DIGIT_0,
86
+ :DIGIT_9,
87
+ :COLON,
88
+ :EQUALS,
89
+ :UPPER_A,
90
+ :UPPER_F,
91
+ :UPPER_Z,
92
+ :BRACKET_CLOSE,
93
+ :UNDERSCORE,
94
+ :LOWER_A,
95
+ :LOWER_F,
96
+ :LOWER_Z,
50
97
  :UNQUOTED_VALUE_STOP
51
98
 
52
99
  # @return [Token, nil] tag token or nil if not a valid tag (caller rolls back)
53
- # Precondition: caller has verified current_char == "[".
100
+ # Precondition: caller has verified the byte at the cursor is "[".
54
101
  def parse_tag_at_cursor
55
102
  tag_start_pos = @current_pos
56
103
  @current_pos += 1 # skip '['
57
- closing = consume("/")
104
+ closing = consume(SLASH)
58
105
  tag_name = scan_tag_name
59
106
  attrs = (closing || tag_name.nil?) ? {} : scan_attributes
60
- return rollback(tag_start_pos) unless tag_name && consume("]")
107
+ return rollback(tag_start_pos) unless tag_name && consume(BRACKET_CLOSE)
61
108
 
62
- source = @input[tag_start_pos...@current_pos]
109
+ source = @input.byteslice(tag_start_pos, @current_pos - tag_start_pos)
63
110
  build_token(closing:, tag: tag_name.downcase, attrs:, pos: tag_start_pos, source:)
64
111
  end
65
112
 
@@ -76,27 +123,26 @@ module Markbridge
76
123
  nil
77
124
  end
78
125
 
79
- # Scan a tag name: [a-z*][a-z0-9]*(:hex*)?
126
+ # Scan a tag name: [a-z*][a-z0-9]*(:hex*)? (case-insensitive)
80
127
  #
81
- # Char-by-char rather than a single regex over `@input[pos..]`
128
+ # Byte-predicate loop rather than a regex over `@input[pos..]`
82
129
  # because the regex form allocates a substring for every tag,
83
- # which is a dominant cost on tag-heavy input. The char-based
84
- # loop is ~3x faster under YJIT.
130
+ # which is a dominant cost on tag-heavy input.
85
131
  # @return [String, nil]
86
132
  def scan_tag_name
87
133
  start = @current_pos
88
134
 
89
- return nil unless current_char&.match?(TAG_INITIAL_CHAR)
135
+ return nil unless tag_initial_byte?(current_byte)
90
136
  @current_pos += 1
91
137
 
92
- @current_pos += 1 while current_char&.match?(TAG_NAME_CHAR)
138
+ @current_pos += 1 while tag_name_byte?(current_byte)
93
139
 
94
- if current_char == ":"
140
+ if current_byte == COLON
95
141
  @current_pos += 1
96
- @current_pos += 1 while current_char&.match?(UID_HEX_CHAR)
142
+ @current_pos += 1 while uid_hex_byte?(current_byte)
97
143
  end
98
144
 
99
- @input[start...@current_pos]
145
+ @input.byteslice(start, @current_pos - start)
100
146
  end
101
147
 
102
148
  # Scan tag attributes
@@ -107,7 +153,7 @@ module Markbridge
107
153
  attrs = {}
108
154
  skip_whitespace
109
155
 
110
- if current_char == "="
156
+ if current_byte == EQUALS
111
157
  @current_pos += 1
112
158
  skip_whitespace
113
159
  if (val = scan_attribute_value)
@@ -116,9 +162,9 @@ module Markbridge
116
162
  skip_whitespace
117
163
  end
118
164
 
119
- while (name = scan_while(ATTR_NAME_CHAR))
165
+ while (name = scan_attr_name)
120
166
  skip_whitespace
121
- break unless consume("=")
167
+ break unless consume(EQUALS)
122
168
 
123
169
  skip_whitespace
124
170
  value = scan_attribute_value
@@ -129,16 +175,16 @@ module Markbridge
129
175
  attrs
130
176
  end
131
177
 
132
- def consume(char)
133
- return false if current_char != char
178
+ def consume(byte)
179
+ return false if current_byte != byte
134
180
 
135
181
  @current_pos += 1
136
182
  true
137
183
  end
138
184
 
139
185
  def scan_attribute_value
140
- char = current_char
141
- if char == '"' || char == "'"
186
+ byte = current_byte
187
+ if byte == DOUBLE_QUOTE || byte == SINGLE_QUOTE
142
188
  scan_quoted_string
143
189
  else
144
190
  scan_unquoted_value
@@ -161,52 +207,86 @@ module Markbridge
161
207
  #
162
208
  # @return [String, nil] the unescaped attribute value, or nil if unterminated
163
209
  def scan_quoted_string
164
- quote_char = current_char
210
+ quote = current_byte == DOUBLE_QUOTE ? "\"" : "'"
165
211
  start = (@current_pos += 1) # skip opening quote
166
- closing_index = @input.index(quote_char, start)
212
+ closing_index = @input.byteindex(quote, start)
167
213
  return nil unless closing_index
168
214
 
169
215
  @current_pos = closing_index + 1
170
- @input[start...closing_index]
216
+ @input.byteslice(start, closing_index - start)
171
217
  end
172
218
 
173
219
  def scan_unquoted_value
174
- scan_until(UNQUOTED_VALUE_STOP)
220
+ consume_range(@input.byteindex(UNQUOTED_VALUE_STOP, @current_pos) || @length)
175
221
  end
176
222
 
177
- # Consumes characters matching +pattern+; returns substring or nil if empty
178
- def scan_while(pattern)
223
+ # Consumes attribute-name characters (\w == [A-Za-z0-9_]); returns
224
+ # substring or nil if empty
225
+ def scan_attr_name
179
226
  stop_index = @current_pos
180
- stop_index += 1 while stop_index < @length && @input[stop_index].match?(pattern)
227
+ stop_index += 1 while stop_index < @length && attr_name_byte?(@input.getbyte(stop_index))
181
228
  consume_range(stop_index)
182
229
  end
183
230
 
184
- # Consumes characters until +pattern+ matches (or end of input); returns substring or nil if empty
185
- def scan_until(pattern)
186
- consume_range(@input.index(pattern, @current_pos) || @length)
187
- end
188
-
189
231
  # Slice [@current_pos, stop_index), advance the cursor, or return nil for empty.
190
232
  def consume_range(stop_index)
191
233
  return nil if stop_index == @current_pos
192
234
 
193
- value = @input[@current_pos...stop_index]
235
+ value = @input.byteslice(@current_pos, stop_index - @current_pos)
194
236
  @current_pos = stop_index
195
237
  value
196
238
  end
197
239
 
198
- def current_char
199
- @input[@current_pos]
240
+ # @return [Integer, nil] byte at the cursor, nil at end of input
241
+ def current_byte
242
+ @input.getbyte(@current_pos)
200
243
  end
201
244
 
202
245
  def skip_whitespace
203
246
  @current_pos += 1 while @current_pos < @length &&
204
- @input[@current_pos].match?(WHITESPACE_CHAR)
247
+ whitespace_byte?(@input.getbyte(@current_pos))
248
+ end
249
+
250
+ # [a-z*]/i — first character of a tag name; nil (end of input) is
251
+ # not a tag byte
252
+ def tag_initial_byte?(byte)
253
+ byte &&
254
+ (
255
+ (byte >= LOWER_A && byte <= LOWER_Z) || (byte >= UPPER_A && byte <= UPPER_Z) ||
256
+ byte == STAR
257
+ )
258
+ end
259
+
260
+ # [a-z0-9]/i — rest of a tag name
261
+ def tag_name_byte?(byte)
262
+ return false if byte.nil?
263
+
264
+ (byte >= LOWER_A && byte <= LOWER_Z) || (byte >= UPPER_A && byte <= UPPER_Z) ||
265
+ (byte >= DIGIT_0 && byte <= DIGIT_9)
266
+ end
267
+
268
+ # [0-9a-f]/i — uid suffix after ':'
269
+ def uid_hex_byte?(byte)
270
+ return false if byte.nil?
271
+
272
+ (byte >= DIGIT_0 && byte <= DIGIT_9) || (byte >= LOWER_A && byte <= LOWER_F) ||
273
+ (byte >= UPPER_A && byte <= UPPER_F)
274
+ end
275
+
276
+ # \w — attribute name character
277
+ def attr_name_byte?(byte)
278
+ (byte >= LOWER_A && byte <= LOWER_Z) || (byte >= UPPER_A && byte <= UPPER_Z) ||
279
+ (byte >= DIGIT_0 && byte <= DIGIT_9) || byte == UNDERSCORE
280
+ end
281
+
282
+ # \s — exactly [ \t\n\v\f\r]
283
+ def whitespace_byte?(byte)
284
+ byte == SPACE || (byte >= TAB && byte <= CR)
205
285
  end
206
286
 
207
287
  def end_of_input?
208
- # All callers maintain @current_pos <= @length (scan_while
209
- # bounds on @length; scan_until uses `index || @length`;
288
+ # All callers maintain @current_pos <= @length (scan_attr_name
289
+ # bounds on @length; scan_unquoted_value uses `byteindex || @length`;
210
290
  # consume is a no-op at EOF); `==` and `>=` are observably
211
291
  # identical here.
212
292
  @current_pos == @length
@@ -112,7 +112,10 @@ module Markbridge
112
112
  # @param tag_name [String]
113
113
  # @return [BaseHandler, nil]
114
114
  def [](tag_name)
115
- @handlers[tag_name.to_s.downcase]
115
+ # Keys are always downcased strings and Nokogiri's HTML parser
116
+ # yields downcased element names, so the fetch hits directly on
117
+ # the hot path; the block normalizes only unusual lookups.
118
+ @handlers.fetch(tag_name) { @handlers[tag_name.to_s.downcase] }
116
119
  end
117
120
 
118
121
  # Create the default handler registry with common HTML tags
@@ -149,6 +152,29 @@ module Markbridge
149
152
  yield(registry) if block_given?
150
153
  registry
151
154
  end
155
+
156
+ # Shared, deep-frozen default registry for the no-customization
157
+ # fast path. Built once per process; {Parser} falls back to it
158
+ # when no +handlers:+ are given, skipping the full registry
159
+ # construction on every parse. Handlers are stateless after
160
+ # construction, so sharing is safe across parsers and threads.
161
+ # Callers who want to mutate the tag sets must build their own
162
+ # {.default}.
163
+ #
164
+ # @return [HandlerRegistry] the same frozen instance on every call
165
+ def self.shared_default
166
+ @shared_default ||= default.freeze
167
+ end
168
+
169
+ # Freeze the registry together with its internal collections so
170
+ # that mutation of a shared instance fails loudly instead of
171
+ # silently changing state visible to every parser.
172
+ def freeze
173
+ @handlers.freeze
174
+ @block_level_tags.freeze
175
+ @whitespace_preserving_tags.freeze
176
+ super
177
+ end
152
178
  end
153
179
  end
154
180
  end