polytypo 1.7.0 → 1.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -41,9 +41,11 @@ module Polytypo
41
41
  end
42
42
 
43
43
  # Block/inline node types that are never processable, in full (modes.md 3.7.3): fenced and
44
- # indented code (comrak represents both as :code_block), inline code spans, thematic
45
- # breaks (no prose content), and frontmatter (skipped whole including its delimiters).
46
- SKIP_WHOLE_TYPES = %i[code code_block thematic_break frontmatter].freeze
44
+ # indented code (comrak represents both as :code_block), inline code spans, and thematic
45
+ # breaks (no prose content). Frontmatter is NOT in this list and is not a node type here:
46
+ # comrak's front_matter_delimiter extension is never enabled, because since spec 1.8.0 the
47
+ # block's extent is modes.md 3.7.3a's scan and not a parser's opinion of it.
48
+ SKIP_WHOLE_TYPES = %i[code code_block thematic_break].freeze
47
49
 
48
50
  # A leaf whose Segment/source_position directly delineates literal, processable prose --
49
51
  # comrak's AST is segment-based like goldmark's: structural bytes (emphasis delimiters,
@@ -55,17 +57,169 @@ module Polytypo
55
57
  # port's goldmark-based adapter).
56
58
  TEXT_LEAF_TYPES = %i[text].freeze
57
59
 
58
- # detect_frontmatter_delimiter: comrak's frontmatter extension takes one literal delimiter
59
- # string and requires the opening AND closing lines to match it exactly -- unlike the Go
60
- # port (which had to hand-roll frontmatter detection entirely), here only the *which
61
- # delimiter* choice needs to be made upfront, from the document's own leading bytes.
62
- def self.detect_frontmatter_delimiter(source)
63
- return "+++" if source.start_with?("+++")
64
- return "---" if source.start_with?("---")
60
+ SPACE = " "
61
+ TAB = "\t"
62
+ CR = "\r"
63
+ LF = "\n"
64
+ BOM = "\uFEFF"
65
+ FRONTMATTER_DELIMITERS = ["---", "+++"].freeze
65
66
 
67
+ # What detect_frontmatter found: +end+, one past the closing delimiter line's terminator;
68
+ # the delimiter it used; and the content range -- from after the opening delimiter line's
69
+ # terminator to the code point that begins the closing delimiter line (modes.md 3.7.4).
70
+ # Both delimiter lines and every line terminator lie outside that content range. The block
71
+ # always begins at the document's first code point, or at the second when a byte-order mark
72
+ # was stepped over, and both are masked, so its start needs no member of its own.
73
+ FrontmatterBlock = Struct.new(:end, :delimiter, :content_start, :content_end)
74
+
75
+ # modes.md 3.7.3a (spec 1.8.0). WHERE THE BLOCK BEGINS AND ENDS IS DECIDED HERE, not by
76
+ # comrak's front_matter_delimiter extension: that extension requires both delimiter lines
77
+ # to match the literal string exactly, so a fence carrying one trailing space was not a
78
+ # block and its metadata was typeset as prose (polytypo/polytypo#58) -- the damage 3.7.3
79
+ # exists to prevent, reached by an invisible character no author typed on purpose.
80
+ #
81
+ # ONE SCAN, ONE ANSWER: this is the authority for both the body's skip (3.7.3) and the
82
+ # content range frontmatter_keys names (3.7.4). Two locators is how a skip and an option
83
+ # come to disagree about the same three lines, and it cost this runtime a second parse of
84
+ # the whole document as well (polytypo/polytypo#59).
85
+ #
86
+ # The steps, in the section's own order: the document must BEGIN with the delimiter, with
87
+ # no leading blank line and no indentation, a single leading U+FEFF stepped over first (a
88
+ # byte-order mark is not content, and reading it as content denies the block to every file
89
+ # some Windows editors write); the rest of that line may be U+0020 and U+0009 and nothing
90
+ # else, so "--- yaml" is not an opener and "----" is not a delimiter; the closing line is
91
+ # the first later line whose FIRST code point begins the same delimiter, followed by only
92
+ # U+0020 and U+0009 -- indentation disqualifies it exactly as it disqualifies the opener,
93
+ # "..." closes nothing, and neither does a delimiter of the other kind; with no such line
94
+ # there is no block.
95
+ #
96
+ # Offsets are code-point indices into +chars+, like every Span in this runtime.
97
+ def self.detect_frontmatter(chars)
98
+ start = chars[0] == BOM ? 1 : 0
99
+ delimiter = FRONTMATTER_DELIMITERS.find { |d| delimiter_at?(chars, start, d) }
100
+ return nil if delimiter.nil?
101
+
102
+ line_end, next_start = line_bounds(chars, start)
103
+ return nil unless spaces_and_tabs_only?(chars, start + delimiter.length, line_end)
104
+
105
+ find_closing_line(chars, delimiter, next_start)
106
+ end
107
+ private_class_method :detect_frontmatter
108
+
109
+ def self.find_closing_line(chars, delimiter, content_start)
110
+ cursor = content_start
111
+ while cursor < chars.length
112
+ line_end, next_start = line_bounds(chars, cursor)
113
+ if delimiter_at?(chars, cursor, delimiter) &&
114
+ spaces_and_tabs_only?(chars, cursor + delimiter.length, line_end)
115
+ return FrontmatterBlock.new(next_start, delimiter, content_start, cursor)
116
+ end
117
+ cursor = next_start
118
+ end
66
119
  nil
67
120
  end
68
- private_class_method :detect_frontmatter_delimiter
121
+ private_class_method :find_closing_line
122
+
123
+ def self.delimiter_at?(chars, i, delimiter)
124
+ delimiter.each_char.with_index.all? { |ch, k| chars[i + k] == ch }
125
+ end
126
+ private_class_method :delimiter_at?
127
+
128
+ # modes.md 3.7.3a step 5. A LINE ENDS AS COMMONMARK ENDS ONE -- at U+000A, at a U+000D not
129
+ # followed by U+000A, or at the end of input -- and the terminator is never part of the
130
+ # line. Returns [line_end, next_line_start]: a CRLF document gives the same block as the
131
+ # same bytes with LF, a final "---\r" with no U+000A still closes its block, and a document
132
+ # written with lone U+000D endings has lines at all.
133
+ #
134
+ # DELIBERATELY NOT 3.8.4's LF-only model, which Yaml.yaml_spans keeps for the block's
135
+ # content: the block is a Markdown construct and ends its lines the way the language around
136
+ # it does. 3.8.4 is a deliberate simplification rather than what YAML says (1.2.2 5.4 admits
137
+ # a lone U+000D too), and widening it here would change "yaml" mode for every caller, so the
138
+ # two models stay different on purpose -- the cost is modes.md 7.13's: inside a lone-U+000D
139
+ # block frontmatter_keys yields nothing at all. Harmonising them silently changes one.
140
+ def self.line_bounds(chars, from)
141
+ i = from
142
+ i += 1 while i < chars.length && chars[i] != LF && chars[i] != CR
143
+ return [i, i] if i >= chars.length
144
+
145
+ [i, chars[i] == CR && chars[i + 1] == LF ? i + 2 : i + 1]
146
+ end
147
+ private_class_method :line_bounds
148
+
149
+ def self.spaces_and_tabs_only?(chars, from, to)
150
+ (from...to).all? { |i| chars[i] == SPACE || chars[i] == TAB }
151
+ end
152
+ private_class_method :spaces_and_tabs_only?
153
+
154
+ # modes.md 3.7.3a: THE PARSER IS HANDED THE BLOCK MASKED OUT -- the block, both delimiter
155
+ # lines included and the leading U+FEFF of step 1 with them, replaced by U+0020 with line
156
+ # terminators kept as they are, so nothing inside it can form or close a construct in the
157
+ # body. The mask runs from 0: the only code point that can precede the delimiter is that
158
+ # mark.
159
+ #
160
+ # THE INVARIANT IS POSITIONAL ALIGNMENT IN THE UNIT THIS RUNTIME MAPS PARSER OFFSETS BACK
161
+ # THROUGH, which is not the same as "one U+0020 per index unit". comrak reports line and
162
+ # column in CODE POINTS under `sourcepos_chars: true`, and build_line_starts resolves them
163
+ # against the original document, so code-point alignment is all this runtime owes -- one
164
+ # space per character gives it, and in Ruby, whose strings are sequences of code points,
165
+ # byte-length preservation cannot even be expressed. A runtime that hands its parser's BYTE
166
+ # offsets straight through owes byte-length preservation instead, and masking an astral
167
+ # character to a single space would shorten its source by three.
168
+ #
169
+ # Suppressing the block's spans after the parse is NOT a substitute for masking, and is
170
+ # measured wrong in the spec's own table: a fenced-code line inside a metadata value pairs
171
+ # with the body's own fence, and the body's code block and its prose swap roles. No
172
+ # span-level check can see that, because the damage is in what the parser concluded before
173
+ # any span existed. The two are not alternatives -- 3.7.3a requires both, and the clip in
174
+ # emit_span is the other half.
175
+ def self.mask_frontmatter(source, block)
176
+ return source if block.nil?
177
+
178
+ segment = source[0, block.end].each_char.map { |ch| ch == LF || ch == CR ? ch : SPACE }
179
+ masked = source.dup
180
+ masked[0, block.end] = segment.join
181
+ masked
182
+ end
183
+ private_class_method :mask_frontmatter
184
+
185
+ # A LOCAL WORKAROUND FOR A MEASURED commonmarker 2.10.0 DEFECT, not a spec rule. With
186
+ # `sourcepos_chars: true` comrak converts its byte columns to code-point columns, and on a
187
+ # document whose lines end in a LONE U+000D the conversion does not reach them: measured on
188
+ # line 3 of such a document, "Body has “quotes” here." comes back with end_column 27, its
189
+ # byte length, instead of 23 -- the very number the option exists to remove. The same
190
+ # document with LF or CRLF endings is converted correctly.
191
+ #
192
+ # U+000D not followed by U+000A is replaced by U+000A for the parser's copy only. The three
193
+ # line endings are the same construct in CommonMark, the replacement is one code point for
194
+ # one, so every source position still addresses the original document -- which is what the
195
+ # output is emitted from, carriage returns and all.
196
+ # Does the 3.8.4 line containing +index+ carry a U+000D that is not followed by U+000A? The
197
+ # line is LF-delimited, because this is the content's own line model (3.8.4), not 3.7.3a's.
198
+ def self.lone_cr_line?(content, index)
199
+ from = (content.rindex(LF, index) || -1) + 1
200
+ to = content.index(LF, index) || content.length
201
+ i = content.index(CR, from)
202
+ while i && i < to
203
+ return true if content[i + 1] != LF
204
+
205
+ i = content.index(CR, i + 1)
206
+ end
207
+ false
208
+ end
209
+ private_class_method :lone_cr_line?
210
+
211
+ def self.normalize_lone_cr(source)
212
+ i = source.index(CR)
213
+ return source if i.nil?
214
+
215
+ normalized = source.dup
216
+ while i
217
+ normalized[i] = LF if normalized[i + 1] != LF
218
+ i = normalized.index(CR, i + 1)
219
+ end
220
+ normalized
221
+ end
222
+ private_class_method :normalize_lone_cr
69
223
 
70
224
  # modes.md 3.7.4, spec 1.7.0. The frontmatter block's own spans, which form a SECOND TEXT
71
225
  # UNIT: the pipeline runs over them separately from the body's, so an unbalanced mark in a
@@ -74,76 +228,82 @@ module Polytypo
74
228
  #
75
229
  # Spans come from the scan of modes.md 3.8 -- frontmatter IS YAML, and implementing that
76
230
  # grammar twice is how two implementations of one spec drift -- with +keys+ as step 8's key
77
- # predicate. The block is the construct markdown_spans skips (3.7.3, the :frontmatter node),
78
- # so the option only ever adds spans where the skip removed them: no source position belongs
79
- # to both units. A TOML block yields nothing, with the option or without it -- its quoting is
80
- # a second grammar this scan does not claim (modes.md 7.13).
231
+ # predicate. The block is the construct markdown_spans masks out (3.7.3, the same
232
+ # detect_frontmatter scan), so the option only ever adds spans where the skip removed them:
233
+ # no source position belongs to both units. A TOML block yields nothing, with the option or
234
+ # without it -- its quoting is a second grammar this scan does not claim (modes.md 7.13).
235
+ #
236
+ # The content range comes from detect_frontmatter, the same scan markdown_spans skips by
237
+ # (3.7.3a, spec 1.8.0). Locating the block needs no parser at all, which is why this method
238
+ # no longer parses the document a second time.
81
239
  #
82
- # The content range comes from the node rather than from a second scan of the text: comrak
83
- # reports the block from its opening delimiter line through its closing one, so the content
84
- # is every line between them, and both delimiters and every line terminator stay outside
85
- # every span.
240
+ # A CONTENT LINE CARRYING A U+000D NOT FOLLOWED BY U+000A YIELDS NO SPANS (3.7.4), per line
241
+ # and not per block, exactly as 3.8.4 step 1 declines a line containing U+0009: a stray
242
+ # U+000D inside one quoted value costs that value and not the whole block. The two line
243
+ # models meet here and do not compose -- 3.7.3a step 5 finds the block in a lone-U+000D
244
+ # document and 3.8.4's LF-only scan then reads the whole of it as ONE line, so such a
245
+ # document loses every span, which is the price of not widening 3.8.4 (that would change
246
+ # "yaml" mode for every caller). Measured, letting it through is not merely inert: marks
247
+ # pair across mapping lines, an unlisted line inside a listed key's scalar takes the
248
+ # locale's spacing, and the U+000D lands inside a span, which 3.8.4 forbids.
86
249
  def self.frontmatter_spans(source, keys)
87
- require "commonmarker"
88
250
  return [] if keys.empty?
89
- return [] unless detect_frontmatter_delimiter(source) == "---"
90
251
 
91
252
  cp = Polytypo::Engine::Codepoints.to_codepoints(source)
92
253
  chars = cp.map { |c| [c].pack("U") }
93
- line_starts = build_line_starts(chars)
254
+ block = detect_frontmatter(chars)
255
+ return [] if block.nil? || block.delimiter != "---"
256
+ return [] if block.content_end <= block.content_start
94
257
 
95
- spans = []
96
- ParseError.wrap do
97
- doc = Commonmarker.parse(
98
- source,
99
- options: { parse: { sourcepos_chars: true },
100
- extension: { front_matter_delimiter: "---" } },
101
- )
102
- doc.each do |node|
103
- next unless node.type == :frontmatter
104
-
105
- pos = node.source_position
106
- content_start = line_starts[pos[:start_line]]
107
- content_end = line_starts[pos[:end_line] - 1]
108
- break if content_start.nil? || content_end.nil? || content_end <= content_start
109
-
110
- content = chars[content_start...content_end].join
111
- Yaml.yaml_spans(content, keys).each do |span|
112
- spans << Spans::Span.new(span.start + content_start, span.end + content_start)
113
- end
114
- break
115
- end
258
+ content = chars[block.content_start...block.content_end].join
259
+ spans = Yaml.yaml_spans(content, keys)
260
+ spans = spans.reject { |span| lone_cr_line?(content, span.start) } if content.include?(CR)
261
+ spans.map do |span|
262
+ Spans::Span.new(span.start + block.content_start, span.end + block.content_start)
116
263
  end
117
- spans
118
264
  end
119
265
 
120
266
  # Locates the processable spans of a Markdown document. dialect must already be validated
121
267
  # via resolve_dialect (== "commonmark"); this function does not re-check it.
268
+ #
269
+ # The frontmatter block is located by detect_frontmatter and masked out of the source the
270
+ # parser sees (3.7.3a), so comrak needs no frontmatter concept and is never given its
271
+ # front_matter_delimiter option: a masked document has no frontmatter left to find, and
272
+ # the extent is this runtime's scan rather than an extension's exact-match opinion of it.
273
+ #
274
+ # frontmatter_end carries 3.7.3a's second requirement: NO SPAN MAY LIE INSIDE THE BLOCK,
275
+ # whatever the parser did with the masked text. Masking is what makes that true for comrak
276
+ # -- measured, it emits no node inside the masked range -- and the spec keeps the rule
277
+ # anyway because it is not true of every parser: tree-sitter-markdown reads a final
278
+ # all-space line with no terminator as a paragraph and would emit a span over the closing
279
+ # delimiter itself.
122
280
  def self.markdown_spans(source)
123
281
  require "commonmarker"
124
282
 
125
283
  cp = Polytypo::Engine::Codepoints.to_codepoints(source)
126
284
  chars = cp.map { |c| [c].pack("U") }
127
285
  line_starts = build_line_starts(chars)
128
-
129
- delimiter = detect_frontmatter_delimiter(source)
130
- options = { parse: { sourcepos_chars: true } }
131
- options[:extension] = { front_matter_delimiter: delimiter } if delimiter
286
+ block = detect_frontmatter(chars)
287
+ parser_source = normalize_lone_cr(mask_frontmatter(source, block))
132
288
 
133
289
  spans = []
134
290
  ParseError.wrap do
135
- doc = Commonmarker.parse(source, options: options)
136
- html_stack_holder = { stack: [] }
137
- walk(doc, chars, line_starts, spans, html_stack_holder)
291
+ doc = Commonmarker.parse(parser_source, options: { parse: { sourcepos_chars: true } })
292
+ walk(doc, chars, line_starts, spans, { stack: [], frontmatter_end: block.nil? ? 0 : block.end })
138
293
  end
139
294
  spans
140
295
  end
141
296
 
142
297
  # line_starts[i] is the 0-based code-point index at which comrak's 1-based line i begins.
298
+ # comrak counts CommonMark line endings, which are U+000A, U+000D, and U+000D U+000A --
299
+ # measured, not assumed: a document with lone U+000D endings reports its body on line 5,
300
+ # and an LF-only table would have no entry to map that to.
143
301
  def self.build_line_starts(chars)
144
302
  starts = [0]
145
303
  chars.each_with_index do |ch, i|
146
- starts << (i + 1) if ch == "\n"
304
+ next unless ch == LF || (ch == CR && chars[i + 1] != LF)
305
+
306
+ starts << (i + 1)
147
307
  end
148
308
  starts
149
309
  end
@@ -154,17 +314,33 @@ module Polytypo
154
314
  end
155
315
  private_class_method :offset_of
156
316
 
157
- def self.walk(node, chars, line_starts, spans, html_holder)
317
+ # modes.md 3.7.3a's second requirement: NO SPAN MAY LIE INSIDE THE BLOCK, whatever the parser
318
+ # did with the masked text. A span that overlaps the block is CLIPPED to the part outside it
319
+ # and dropped when nothing is left -- the block's own characters must not reach the rules,
320
+ # and body prose past the block must not be lost to a parser's mistake about where the block
321
+ # ended. It never fires with comrak, which emits no node inside the masked range -- measured
322
+ # by counting clips over every markdown fixture's input and output, 216 spans, none clipped
323
+ # -- and it is implemented anyway because the spec states it of every runtime:
324
+ # tree-sitter-markdown reads a final all-space line with no terminator as a paragraph.
325
+ def self.emit_span(spans, start_offset, end_offset, state)
326
+ start_offset = state[:frontmatter_end] if start_offset < state[:frontmatter_end]
327
+ return if end_offset <= start_offset
328
+
329
+ spans << Spans::Span.new(start_offset, end_offset)
330
+ end
331
+ private_class_method :emit_span
332
+
333
+ def self.walk(node, chars, line_starts, spans, state)
158
334
  type = node.type
159
335
  return if SKIP_WHOLE_TYPES.include?(type)
160
336
 
161
337
  if type == :html_block
162
- handle_html_block(node, chars, line_starts, spans)
338
+ handle_html_block(node, chars, line_starts, spans, state)
163
339
  return
164
340
  end
165
341
 
166
342
  if type == :html_inline
167
- handle_html_inline(node, chars, line_starts, html_holder)
343
+ handle_html_inline(node, chars, line_starts, state)
168
344
  return
169
345
  end
170
346
 
@@ -186,23 +362,21 @@ module Polytypo
186
362
  sp = node.source_position
187
363
  start_offset = offset_of(line_starts, sp[:start_line], sp[:start_column])
188
364
  end_offset = offset_of(line_starts, sp[:end_line], sp[:end_column]) + 1
189
- if html_holder[:stack].empty? && end_offset > start_offset
190
- spans << Spans::Span.new(start_offset, end_offset)
191
- end
365
+ emit_span(spans, start_offset, end_offset, state) if state[:stack].empty?
192
366
  return
193
367
  end
194
368
 
195
369
  if %i[paragraph heading table_cell].include?(type)
196
370
  # A fresh top-level inline-bearing container: the raw-HTML skip stack resets here (an
197
371
  # unclosed skipped start tag skips to the end of the block, modes.md 3.7.3).
198
- html_holder[:stack] = []
372
+ state[:stack] = []
199
373
  end
200
374
 
201
- node.each { |child| walk(child, chars, line_starts, spans, html_holder) }
375
+ node.each { |child| walk(child, chars, line_starts, spans, state) }
202
376
  end
203
377
  private_class_method :walk
204
378
 
205
- def self.handle_html_block(node, chars, line_starts, spans)
379
+ def self.handle_html_block(node, chars, line_starts, spans, state)
206
380
  sp = node.source_position
207
381
  start_offset = offset_of(line_starts, sp[:start_line], sp[:start_column])
208
382
  end_offset = offset_of(line_starts, sp[:end_line], sp[:end_column]) + 1
@@ -210,12 +384,12 @@ module Polytypo
210
384
 
211
385
  text = chars[start_offset...end_offset].join
212
386
  Html.html_spans(text).each do |s|
213
- spans << Spans::Span.new(start_offset + s.start, start_offset + s.end)
387
+ emit_span(spans, start_offset + s.start, start_offset + s.end, state)
214
388
  end
215
389
  end
216
390
  private_class_method :handle_html_block
217
391
 
218
- def self.handle_html_inline(node, chars, line_starts, html_holder)
392
+ def self.handle_html_inline(node, chars, line_starts, state)
219
393
  sp = node.source_position
220
394
  start_offset = offset_of(line_starts, sp[:start_line], sp[:start_column])
221
395
  end_offset = offset_of(line_starts, sp[:end_line], sp[:end_column]) + 1
@@ -223,7 +397,7 @@ module Polytypo
223
397
  name, closing, self_closing = Html.read_tag(raw) || [nil, nil, nil]
224
398
  return if name.nil?
225
399
 
226
- stack = html_holder[:stack]
400
+ stack = state[:stack]
227
401
  if closing
228
402
  if !stack.empty? && stack[-1] == name
229
403
  stack.pop
@@ -1,5 +1,5 @@
1
1
  # frozen_string_literal: true
2
2
 
3
3
  module Polytypo
4
- VERSION = "1.7.0"
4
+ VERSION = "1.8.0"
5
5
  end
metadata CHANGED
@@ -1,7 +1,7 @@
1
1
  --- !ruby/object:Gem::Specification
2
2
  name: polytypo
3
3
  version: !ruby/object:Gem::Version
4
- version: 1.7.0
4
+ version: 1.8.0
5
5
  platform: ruby
6
6
  authors:
7
7
  - Iurii Rogulia