polytypo 1.6.3 → 1.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (54) hide show
  1. checksums.yaml +4 -4
  2. data/README.md +25 -0
  3. data/lib/polytypo/data/.not-vendored +11 -0
  4. data/lib/polytypo/data/README.md +16 -11
  5. data/lib/polytypo/data/VERSION +1 -1
  6. data/lib/polytypo/data/fixtures/cs.json +1 -1
  7. data/lib/polytypo/data/fixtures/de-CH.json +1 -1
  8. data/lib/polytypo/data/fixtures/de-DE.json +14 -1
  9. data/lib/polytypo/data/fixtures/el.json +1 -1
  10. data/lib/polytypo/data/fixtures/en-GB.json +1 -1
  11. data/lib/polytypo/data/fixtures/en-US.json +521 -1
  12. data/lib/polytypo/data/fixtures/es.json +1 -1
  13. data/lib/polytypo/data/fixtures/fi.json +1 -1
  14. data/lib/polytypo/data/fixtures/fr-CA.json +1 -1
  15. data/lib/polytypo/data/fixtures/fr.json +68 -1
  16. data/lib/polytypo/data/fixtures/it.json +1 -1
  17. data/lib/polytypo/data/fixtures/locale-resolution.json +1 -1
  18. data/lib/polytypo/data/fixtures/nl.json +1 -1
  19. data/lib/polytypo/data/fixtures/pl.json +1 -1
  20. data/lib/polytypo/data/fixtures/pt-BR.json +1 -1
  21. data/lib/polytypo/data/fixtures/pt-PT.json +1 -1
  22. data/lib/polytypo/data/fixtures/ru.json +1 -1
  23. data/lib/polytypo/data/fixtures/sv.json +1 -1
  24. data/lib/polytypo/data/fixtures/tr.json +1 -1
  25. data/lib/polytypo/data/fixtures/uk.json +1 -1
  26. data/lib/polytypo/data/locales/cs.json +56 -37
  27. data/lib/polytypo/data/locales/de-CH.json +43 -34
  28. data/lib/polytypo/data/locales/de-DE.json +47 -32
  29. data/lib/polytypo/data/locales/el.json +51 -1
  30. data/lib/polytypo/data/locales/en-GB.json +56 -4
  31. data/lib/polytypo/data/locales/en-US.json +68 -4
  32. data/lib/polytypo/data/locales/es.json +66 -29
  33. data/lib/polytypo/data/locales/fi.json +77 -7
  34. data/lib/polytypo/data/locales/fr-CA.json +63 -36
  35. data/lib/polytypo/data/locales/fr.json +70 -41
  36. data/lib/polytypo/data/locales/it.json +65 -22
  37. data/lib/polytypo/data/locales/nl.json +61 -29
  38. data/lib/polytypo/data/locales/pl.json +61 -28
  39. data/lib/polytypo/data/locales/pt-BR.json +53 -28
  40. data/lib/polytypo/data/locales/pt-PT.json +55 -28
  41. data/lib/polytypo/data/locales/registry.json +1 -1
  42. data/lib/polytypo/data/locales/ru.json +45 -30
  43. data/lib/polytypo/data/locales/sv.json +63 -4
  44. data/lib/polytypo/data/locales/tr.json +72 -32
  45. data/lib/polytypo/data/locales/uk.json +55 -27
  46. data/lib/polytypo/data/rules/modes.md +430 -11
  47. data/lib/polytypo/data/rules/order.json +1 -1
  48. data/lib/polytypo/data/schema/fixtures.schema.json +12 -1
  49. data/lib/polytypo/engine/pipeline.rb +29 -0
  50. data/lib/polytypo/modes/markdown.rb +256 -31
  51. data/lib/polytypo/modes/runner.rb +33 -3
  52. data/lib/polytypo/version.rb +1 -1
  53. data/lib/polytypo.rb +27 -12
  54. metadata +2 -1
@@ -2,6 +2,7 @@
2
2
 
3
3
  require_relative "spans"
4
4
  require_relative "html"
5
+ require_relative "yaml"
5
6
  require_relative "parse_error"
6
7
  require_relative "../errors"
7
8
 
@@ -40,9 +41,11 @@ module Polytypo
40
41
  end
41
42
 
42
43
  # Block/inline node types that are never processable, in full (modes.md 3.7.3): fenced and
43
- # indented code (comrak represents both as :code_block), inline code spans, thematic
44
- # breaks (no prose content), and frontmatter (skipped whole including its delimiters).
45
- SKIP_WHOLE_TYPES = %i[code code_block thematic_break frontmatter].freeze
44
+ # indented code (comrak represents both as :code_block), inline code spans, and thematic
45
+ # breaks (no prose content). Frontmatter is NOT in this list and is not a node type here:
46
+ # comrak's front_matter_delimiter extension is never enabled, because since spec 1.8.0 the
47
+ # block's extent is modes.md 3.7.3a's scan and not a parser's opinion of it.
48
+ SKIP_WHOLE_TYPES = %i[code code_block thematic_break].freeze
46
49
 
47
50
  # A leaf whose Segment/source_position directly delineates literal, processable prose --
48
51
  # comrak's AST is segment-based like goldmark's: structural bytes (emphasis delimiters,
@@ -54,45 +57,253 @@ module Polytypo
54
57
  # port's goldmark-based adapter).
55
58
  TEXT_LEAF_TYPES = %i[text].freeze
56
59
 
57
- # detect_frontmatter_delimiter: comrak's frontmatter extension takes one literal delimiter
58
- # string and requires the opening AND closing lines to match it exactly -- unlike the Go
59
- # port (which had to hand-roll frontmatter detection entirely), here only the *which
60
- # delimiter* choice needs to be made upfront, from the document's own leading bytes.
61
- def self.detect_frontmatter_delimiter(source)
62
- return "+++" if source.start_with?("+++")
63
- return "---" if source.start_with?("---")
60
+ SPACE = " "
61
+ TAB = "\t"
62
+ CR = "\r"
63
+ LF = "\n"
64
+ BOM = "\uFEFF"
65
+ FRONTMATTER_DELIMITERS = ["---", "+++"].freeze
64
66
 
67
+ # What detect_frontmatter found: +end+, one past the closing delimiter line's terminator;
68
+ # the delimiter it used; and the content range -- from after the opening delimiter line's
69
+ # terminator to the code point that begins the closing delimiter line (modes.md 3.7.4).
70
+ # Both delimiter lines and every line terminator lie outside that content range. The block
71
+ # always begins at the document's first code point, or at the second when a byte-order mark
72
+ # was stepped over, and both are masked, so its start needs no member of its own.
73
+ FrontmatterBlock = Struct.new(:end, :delimiter, :content_start, :content_end)
74
+
75
+ # modes.md 3.7.3a (spec 1.8.0). WHERE THE BLOCK BEGINS AND ENDS IS DECIDED HERE, not by
76
+ # comrak's front_matter_delimiter extension: that extension requires both delimiter lines
77
+ # to match the literal string exactly, so a fence carrying one trailing space was not a
78
+ # block and its metadata was typeset as prose (polytypo/polytypo#58) -- the damage 3.7.3
79
+ # exists to prevent, reached by an invisible character no author typed on purpose.
80
+ #
81
+ # ONE SCAN, ONE ANSWER: this is the authority for both the body's skip (3.7.3) and the
82
+ # content range frontmatter_keys names (3.7.4). Two locators is how a skip and an option
83
+ # come to disagree about the same three lines, and it cost this runtime a second parse of
84
+ # the whole document as well (polytypo/polytypo#59).
85
+ #
86
+ # The steps, in the section's own order: the document must BEGIN with the delimiter, with
87
+ # no leading blank line and no indentation, a single leading U+FEFF stepped over first (a
88
+ # byte-order mark is not content, and reading it as content denies the block to every file
89
+ # some Windows editors write); the rest of that line may be U+0020 and U+0009 and nothing
90
+ # else, so "--- yaml" is not an opener and "----" is not a delimiter; the closing line is
91
+ # the first later line whose FIRST code point begins the same delimiter, followed by only
92
+ # U+0020 and U+0009 -- indentation disqualifies it exactly as it disqualifies the opener,
93
+ # "..." closes nothing, and neither does a delimiter of the other kind; with no such line
94
+ # there is no block.
95
+ #
96
+ # Offsets are code-point indices into +chars+, like every Span in this runtime.
97
+ def self.detect_frontmatter(chars)
98
+ start = chars[0] == BOM ? 1 : 0
99
+ delimiter = FRONTMATTER_DELIMITERS.find { |d| delimiter_at?(chars, start, d) }
100
+ return nil if delimiter.nil?
101
+
102
+ line_end, next_start = line_bounds(chars, start)
103
+ return nil unless spaces_and_tabs_only?(chars, start + delimiter.length, line_end)
104
+
105
+ find_closing_line(chars, delimiter, next_start)
106
+ end
107
+ private_class_method :detect_frontmatter
108
+
109
+ def self.find_closing_line(chars, delimiter, content_start)
110
+ cursor = content_start
111
+ while cursor < chars.length
112
+ line_end, next_start = line_bounds(chars, cursor)
113
+ if delimiter_at?(chars, cursor, delimiter) &&
114
+ spaces_and_tabs_only?(chars, cursor + delimiter.length, line_end)
115
+ return FrontmatterBlock.new(next_start, delimiter, content_start, cursor)
116
+ end
117
+ cursor = next_start
118
+ end
65
119
  nil
66
120
  end
67
- private_class_method :detect_frontmatter_delimiter
121
+ private_class_method :find_closing_line
122
+
123
+ def self.delimiter_at?(chars, i, delimiter)
124
+ delimiter.each_char.with_index.all? { |ch, k| chars[i + k] == ch }
125
+ end
126
+ private_class_method :delimiter_at?
127
+
128
+ # modes.md 3.7.3a step 5. A LINE ENDS AS COMMONMARK ENDS ONE -- at U+000A, at a U+000D not
129
+ # followed by U+000A, or at the end of input -- and the terminator is never part of the
130
+ # line. Returns [line_end, next_line_start]: a CRLF document gives the same block as the
131
+ # same bytes with LF, a final "---\r" with no U+000A still closes its block, and a document
132
+ # written with lone U+000D endings has lines at all.
133
+ #
134
+ # DELIBERATELY NOT 3.8.4's LF-only model, which Yaml.yaml_spans keeps for the block's
135
+ # content: the block is a Markdown construct and ends its lines the way the language around
136
+ # it does. 3.8.4 is a deliberate simplification rather than what YAML says (1.2.2 5.4 admits
137
+ # a lone U+000D too), and widening it here would change "yaml" mode for every caller, so the
138
+ # two models stay different on purpose -- the cost is modes.md 7.13's: inside a lone-U+000D
139
+ # block frontmatter_keys yields nothing at all. Harmonising them silently changes one.
140
+ def self.line_bounds(chars, from)
141
+ i = from
142
+ i += 1 while i < chars.length && chars[i] != LF && chars[i] != CR
143
+ return [i, i] if i >= chars.length
144
+
145
+ [i, chars[i] == CR && chars[i + 1] == LF ? i + 2 : i + 1]
146
+ end
147
+ private_class_method :line_bounds
148
+
149
+ def self.spaces_and_tabs_only?(chars, from, to)
150
+ (from...to).all? { |i| chars[i] == SPACE || chars[i] == TAB }
151
+ end
152
+ private_class_method :spaces_and_tabs_only?
153
+
154
+ # modes.md 3.7.3a: THE PARSER IS HANDED THE BLOCK MASKED OUT -- the block, both delimiter
155
+ # lines included and the leading U+FEFF of step 1 with them, replaced by U+0020 with line
156
+ # terminators kept as they are, so nothing inside it can form or close a construct in the
157
+ # body. The mask runs from 0: the only code point that can precede the delimiter is that
158
+ # mark.
159
+ #
160
+ # THE INVARIANT IS POSITIONAL ALIGNMENT IN THE UNIT THIS RUNTIME MAPS PARSER OFFSETS BACK
161
+ # THROUGH, which is not the same as "one U+0020 per index unit". comrak reports line and
162
+ # column in CODE POINTS under `sourcepos_chars: true`, and build_line_starts resolves them
163
+ # against the original document, so code-point alignment is all this runtime owes -- one
164
+ # space per character gives it, and in Ruby, whose strings are sequences of code points,
165
+ # byte-length preservation cannot even be expressed. A runtime that hands its parser's BYTE
166
+ # offsets straight through owes byte-length preservation instead, and masking an astral
167
+ # character to a single space would shorten its source by three.
168
+ #
169
+ # Suppressing the block's spans after the parse is NOT a substitute for masking, and is
170
+ # measured wrong in the spec's own table: a fenced-code line inside a metadata value pairs
171
+ # with the body's own fence, and the body's code block and its prose swap roles. No
172
+ # span-level check can see that, because the damage is in what the parser concluded before
173
+ # any span existed. The two are not alternatives -- 3.7.3a requires both, and the clip in
174
+ # emit_span is the other half.
175
+ def self.mask_frontmatter(source, block)
176
+ return source if block.nil?
177
+
178
+ segment = source[0, block.end].each_char.map { |ch| ch == LF || ch == CR ? ch : SPACE }
179
+ masked = source.dup
180
+ masked[0, block.end] = segment.join
181
+ masked
182
+ end
183
+ private_class_method :mask_frontmatter
184
+
185
+ # A LOCAL WORKAROUND FOR A MEASURED commonmarker 2.10.0 DEFECT, not a spec rule. With
186
+ # `sourcepos_chars: true` comrak converts its byte columns to code-point columns, and on a
187
+ # document whose lines end in a LONE U+000D the conversion does not reach them: measured on
188
+ # line 3 of such a document, "Body has “quotes” here." comes back with end_column 27, its
189
+ # byte length, instead of 23 -- the very number the option exists to remove. The same
190
+ # document with LF or CRLF endings is converted correctly.
191
+ #
192
+ # U+000D not followed by U+000A is replaced by U+000A for the parser's copy only. The three
193
+ # line endings are the same construct in CommonMark, the replacement is one code point for
194
+ # one, so every source position still addresses the original document -- which is what the
195
+ # output is emitted from, carriage returns and all.
196
+ # Does the 3.8.4 line containing +index+ carry a U+000D that is not followed by U+000A? The
197
+ # line is LF-delimited, because this is the content's own line model (3.8.4), not 3.7.3a's.
198
+ def self.lone_cr_line?(content, index)
199
+ from = (content.rindex(LF, index) || -1) + 1
200
+ to = content.index(LF, index) || content.length
201
+ i = content.index(CR, from)
202
+ while i && i < to
203
+ return true if content[i + 1] != LF
204
+
205
+ i = content.index(CR, i + 1)
206
+ end
207
+ false
208
+ end
209
+ private_class_method :lone_cr_line?
210
+
211
+ def self.normalize_lone_cr(source)
212
+ i = source.index(CR)
213
+ return source if i.nil?
214
+
215
+ normalized = source.dup
216
+ while i
217
+ normalized[i] = LF if normalized[i + 1] != LF
218
+ i = normalized.index(CR, i + 1)
219
+ end
220
+ normalized
221
+ end
222
+ private_class_method :normalize_lone_cr
223
+
224
+ # modes.md 3.7.4, spec 1.7.0. The frontmatter block's own spans, which form a SECOND TEXT
225
+ # UNIT: the pipeline runs over them separately from the body's, so an unbalanced mark in a
226
+ # metadata field can never pair with one in the first paragraph, and the option cannot
227
+ # change a byte outside the block.
228
+ #
229
+ # Spans come from the scan of modes.md 3.8 -- frontmatter IS YAML, and implementing that
230
+ # grammar twice is how two implementations of one spec drift -- with +keys+ as step 8's key
231
+ # predicate. The block is the construct markdown_spans masks out (3.7.3, the same
232
+ # detect_frontmatter scan), so the option only ever adds spans where the skip removed them:
233
+ # no source position belongs to both units. A TOML block yields nothing, with the option or
234
+ # without it -- its quoting is a second grammar this scan does not claim (modes.md 7.13).
235
+ #
236
+ # The content range comes from detect_frontmatter, the same scan markdown_spans skips by
237
+ # (3.7.3a, spec 1.8.0). Locating the block needs no parser at all, which is why this method
238
+ # no longer parses the document a second time.
239
+ #
240
+ # A CONTENT LINE CARRYING A U+000D NOT FOLLOWED BY U+000A YIELDS NO SPANS (3.7.4), per line
241
+ # and not per block, exactly as 3.8.4 step 1 declines a line containing U+0009: a stray
242
+ # U+000D inside one quoted value costs that value and not the whole block. The two line
243
+ # models meet here and do not compose -- 3.7.3a step 5 finds the block in a lone-U+000D
244
+ # document and 3.8.4's LF-only scan then reads the whole of it as ONE line, so such a
245
+ # document loses every span, which is the price of not widening 3.8.4 (that would change
246
+ # "yaml" mode for every caller). Measured, letting it through is not merely inert: marks
247
+ # pair across mapping lines, an unlisted line inside a listed key's scalar takes the
248
+ # locale's spacing, and the U+000D lands inside a span, which 3.8.4 forbids.
249
+ def self.frontmatter_spans(source, keys)
250
+ return [] if keys.empty?
251
+
252
+ cp = Polytypo::Engine::Codepoints.to_codepoints(source)
253
+ chars = cp.map { |c| [c].pack("U") }
254
+ block = detect_frontmatter(chars)
255
+ return [] if block.nil? || block.delimiter != "---"
256
+ return [] if block.content_end <= block.content_start
257
+
258
+ content = chars[block.content_start...block.content_end].join
259
+ spans = Yaml.yaml_spans(content, keys)
260
+ spans = spans.reject { |span| lone_cr_line?(content, span.start) } if content.include?(CR)
261
+ spans.map do |span|
262
+ Spans::Span.new(span.start + block.content_start, span.end + block.content_start)
263
+ end
264
+ end
68
265
 
69
266
  # Locates the processable spans of a Markdown document. dialect must already be validated
70
267
  # via resolve_dialect (== "commonmark"); this function does not re-check it.
268
+ #
269
+ # The frontmatter block is located by detect_frontmatter and masked out of the source the
270
+ # parser sees (3.7.3a), so comrak needs no frontmatter concept and is never given its
271
+ # front_matter_delimiter option: a masked document has no frontmatter left to find, and
272
+ # the extent is this runtime's scan rather than an extension's exact-match opinion of it.
273
+ #
274
+ # frontmatter_end carries 3.7.3a's second requirement: NO SPAN MAY LIE INSIDE THE BLOCK,
275
+ # whatever the parser did with the masked text. Masking is what makes that true for comrak
276
+ # -- measured, it emits no node inside the masked range -- and the spec keeps the rule
277
+ # anyway because it is not true of every parser: tree-sitter-markdown reads a final
278
+ # all-space line with no terminator as a paragraph and would emit a span over the closing
279
+ # delimiter itself.
71
280
  def self.markdown_spans(source)
72
281
  require "commonmarker"
73
282
 
74
283
  cp = Polytypo::Engine::Codepoints.to_codepoints(source)
75
284
  chars = cp.map { |c| [c].pack("U") }
76
285
  line_starts = build_line_starts(chars)
77
-
78
- delimiter = detect_frontmatter_delimiter(source)
79
- options = { parse: { sourcepos_chars: true } }
80
- options[:extension] = { front_matter_delimiter: delimiter } if delimiter
286
+ block = detect_frontmatter(chars)
287
+ parser_source = normalize_lone_cr(mask_frontmatter(source, block))
81
288
 
82
289
  spans = []
83
290
  ParseError.wrap do
84
- doc = Commonmarker.parse(source, options: options)
85
- html_stack_holder = { stack: [] }
86
- walk(doc, chars, line_starts, spans, html_stack_holder)
291
+ doc = Commonmarker.parse(parser_source, options: { parse: { sourcepos_chars: true } })
292
+ walk(doc, chars, line_starts, spans, { stack: [], frontmatter_end: block.nil? ? 0 : block.end })
87
293
  end
88
294
  spans
89
295
  end
90
296
 
91
297
  # line_starts[i] is the 0-based code-point index at which comrak's 1-based line i begins.
298
+ # comrak counts CommonMark line endings, which are U+000A, U+000D, and U+000D U+000A --
299
+ # measured, not assumed: a document with lone U+000D endings reports its body on line 5,
300
+ # and an LF-only table would have no entry to map that to.
92
301
  def self.build_line_starts(chars)
93
302
  starts = [0]
94
303
  chars.each_with_index do |ch, i|
95
- starts << (i + 1) if ch == "\n"
304
+ next unless ch == LF || (ch == CR && chars[i + 1] != LF)
305
+
306
+ starts << (i + 1)
96
307
  end
97
308
  starts
98
309
  end
@@ -103,17 +314,33 @@ module Polytypo
103
314
  end
104
315
  private_class_method :offset_of
105
316
 
106
- def self.walk(node, chars, line_starts, spans, html_holder)
317
+ # modes.md 3.7.3a's second requirement: NO SPAN MAY LIE INSIDE THE BLOCK, whatever the parser
318
+ # did with the masked text. A span that overlaps the block is CLIPPED to the part outside it
319
+ # and dropped when nothing is left -- the block's own characters must not reach the rules,
320
+ # and body prose past the block must not be lost to a parser's mistake about where the block
321
+ # ended. It never fires with comrak, which emits no node inside the masked range -- measured
322
+ # by counting clips over every markdown fixture's input and output, 216 spans, none clipped
323
+ # -- and it is implemented anyway because the spec states it of every runtime:
324
+ # tree-sitter-markdown reads a final all-space line with no terminator as a paragraph.
325
+ def self.emit_span(spans, start_offset, end_offset, state)
326
+ start_offset = state[:frontmatter_end] if start_offset < state[:frontmatter_end]
327
+ return if end_offset <= start_offset
328
+
329
+ spans << Spans::Span.new(start_offset, end_offset)
330
+ end
331
+ private_class_method :emit_span
332
+
333
+ def self.walk(node, chars, line_starts, spans, state)
107
334
  type = node.type
108
335
  return if SKIP_WHOLE_TYPES.include?(type)
109
336
 
110
337
  if type == :html_block
111
- handle_html_block(node, chars, line_starts, spans)
338
+ handle_html_block(node, chars, line_starts, spans, state)
112
339
  return
113
340
  end
114
341
 
115
342
  if type == :html_inline
116
- handle_html_inline(node, chars, line_starts, html_holder)
343
+ handle_html_inline(node, chars, line_starts, state)
117
344
  return
118
345
  end
119
346
 
@@ -135,23 +362,21 @@ module Polytypo
135
362
  sp = node.source_position
136
363
  start_offset = offset_of(line_starts, sp[:start_line], sp[:start_column])
137
364
  end_offset = offset_of(line_starts, sp[:end_line], sp[:end_column]) + 1
138
- if html_holder[:stack].empty? && end_offset > start_offset
139
- spans << Spans::Span.new(start_offset, end_offset)
140
- end
365
+ emit_span(spans, start_offset, end_offset, state) if state[:stack].empty?
141
366
  return
142
367
  end
143
368
 
144
369
  if %i[paragraph heading table_cell].include?(type)
145
370
  # A fresh top-level inline-bearing container: the raw-HTML skip stack resets here (an
146
371
  # unclosed skipped start tag skips to the end of the block, modes.md 3.7.3).
147
- html_holder[:stack] = []
372
+ state[:stack] = []
148
373
  end
149
374
 
150
- node.each { |child| walk(child, chars, line_starts, spans, html_holder) }
375
+ node.each { |child| walk(child, chars, line_starts, spans, state) }
151
376
  end
152
377
  private_class_method :walk
153
378
 
154
- def self.handle_html_block(node, chars, line_starts, spans)
379
+ def self.handle_html_block(node, chars, line_starts, spans, state)
155
380
  sp = node.source_position
156
381
  start_offset = offset_of(line_starts, sp[:start_line], sp[:start_column])
157
382
  end_offset = offset_of(line_starts, sp[:end_line], sp[:end_column]) + 1
@@ -159,12 +384,12 @@ module Polytypo
159
384
 
160
385
  text = chars[start_offset...end_offset].join
161
386
  Html.html_spans(text).each do |s|
162
- spans << Spans::Span.new(start_offset + s.start, start_offset + s.end)
387
+ emit_span(spans, start_offset + s.start, start_offset + s.end, state)
163
388
  end
164
389
  end
165
390
  private_class_method :handle_html_block
166
391
 
167
- def self.handle_html_inline(node, chars, line_starts, html_holder)
392
+ def self.handle_html_inline(node, chars, line_starts, state)
168
393
  sp = node.source_position
169
394
  start_offset = offset_of(line_starts, sp[:start_line], sp[:start_column])
170
395
  end_offset = offset_of(line_starts, sp[:end_line], sp[:end_column]) + 1
@@ -172,7 +397,7 @@ module Polytypo
172
397
  name, closing, self_closing = Html.read_tag(raw) || [nil, nil, nil]
173
398
  return if name.nil?
174
399
 
175
- stack = html_holder[:stack]
400
+ stack = state[:stack]
176
401
  if closing
177
402
  if !stack.empty? && stack[-1] == name
178
403
  stack.pop
@@ -34,17 +34,46 @@ module Polytypo
34
34
  # source_cp is the whole input already converted to code points; spans address it by
35
35
  # code-point index.
36
36
  def self.run_over_spans(source_cp, spans, plan, locale_data, ctx)
37
+ emit(source_cp, replacements_of_unit(source_cp, spans, plan, locale_data, ctx))
38
+ end
39
+
40
+ # modes.md 3.1 and 3.5 step 3 (spec 1.7.0). A document has one text unit, except in
41
+ # "markdown" with frontmatter_keys, where the frontmatter block's spans form a unit of their
42
+ # own. The pipeline runs once per unit and the two edit sets are disjoint, because no span of
43
+ # one unit lies inside the other -- which is what the body's walk skipping the block
44
+ # guarantees. Only step 5 is shared: the source is emitted once, in document order.
45
+ def self.run_over_units(source_cp, units, plan, locale_data, ctx)
46
+ replacements = units.flat_map do |spans|
47
+ replacements_of_unit(source_cp, spans, plan, locale_data, ctx)
48
+ end
49
+ emit(source_cp, replacements.sort_by { |span, _piece| span.start })
50
+ end
51
+
52
+ # analyze_over_spans per text unit (modes.md 3.1), reported in document order.
53
+ def self.analyze_over_units(source_cp, units, plan, locale_data, ctx)
54
+ units.flat_map { |spans| analyze_over_spans(source_cp, spans, plan, locale_data, ctx) }
55
+ .sort_by(&:start)
56
+ end
57
+
58
+ # One text unit: the marker-separated concatenation, the pipeline, and the pieces it
59
+ # produced, paired with the spans they replace.
60
+ def self.replacements_of_unit(source_cp, spans, plan, locale_data, ctx)
37
61
  normalized = Spans.normalize_spans(spans)
38
- return source_cp.pack("U*") if normalized.empty?
62
+ return [] if normalized.empty?
39
63
 
40
64
  concatenated = Spans.concatenate_spans(source_cp, normalized)
41
65
  transformed = run_rules_over_spans(concatenated, plan, locale_data, ctx)
42
66
  pieces = Spans.split_on_marker(transformed, normalized.length)
67
+ normalized.each_with_index.map { |span, i| [span, pieces[i]] }
68
+ end
69
+ private_class_method :replacements_of_unit
43
70
 
71
+ # modes.md 4: the source with disjoint replacements applied at recorded offsets, and nothing
72
+ # else changed.
73
+ def self.emit(source_cp, replacements)
44
74
  out = []
45
75
  cursor = 0
46
- normalized.each_with_index do |span, i|
47
- piece = pieces[i]
76
+ replacements.each do |span, piece|
48
77
  original = source_cp[span.start...span.end]
49
78
  out.concat(source_cp[cursor...span.start])
50
79
  out.concat(piece == original ? original : piece)
@@ -53,6 +82,7 @@ module Polytypo
53
82
  out.concat(source_cp[cursor..])
54
83
  out.pack("U*")
55
84
  end
85
+ private_class_method :emit
56
86
 
57
87
  # run_over_spans, reporting instead of applying (analyze.md section 1). The span table
58
88
  # supplies the origin map, so every change comes back in DOCUMENT coordinates -- analyze.md
@@ -1,5 +1,5 @@
1
1
  # frozen_string_literal: true
2
2
 
3
3
  module Polytypo
4
- VERSION = "1.6.3"
4
+ VERSION = "1.8.0"
5
5
  end
data/lib/polytypo.rb CHANGED
@@ -32,7 +32,8 @@ module Polytypo
32
32
  #
33
33
  # Only `mode: "text"`/`"html"` ever touch this file's own requires; `mode: "markdown"` lazily
34
34
  # requires "commonmarker" from within Modes::Markdown, never at load time of this file.
35
- def self.transform(input, locale:, mode: "text", dialect: nil, keys: nil, rules: nil, narrow_nbsp: nil)
35
+ def self.transform(input, locale:, mode: "text", dialect: nil, keys: nil, rules: nil, narrow_nbsp: nil,
36
+ frontmatter_keys: nil)
36
37
  resolved_mode = resolve_mode(mode)
37
38
  narrow_target = Engine.resolve_narrow_target(narrow_nbsp)
38
39
 
@@ -53,7 +54,7 @@ module Polytypo
53
54
  end
54
55
  transform_yaml(input, locale, keys, rules, narrow_target)
55
56
  else # "markdown"
56
- transform_markdown(input, locale, dialect, rules, narrow_target)
57
+ transform_markdown(input, locale, dialect, rules, narrow_target, frontmatter_keys)
57
58
  end
58
59
  end
59
60
 
@@ -69,7 +70,8 @@ module Polytypo
69
70
  # for the text (analyze.md sections 4 and 5).
70
71
  #
71
72
  # Pure and thread-safe on the same terms as .transform.
72
- def self.analyze(input, locale:, mode: "text", dialect: nil, keys: nil, rules: nil, narrow_nbsp: nil)
73
+ def self.analyze(input, locale:, mode: "text", dialect: nil, keys: nil, rules: nil, narrow_nbsp: nil,
74
+ frontmatter_keys: nil)
73
75
  resolved_mode = resolve_mode(mode)
74
76
  narrow_target = Engine.resolve_narrow_target(narrow_nbsp)
75
77
 
@@ -90,7 +92,7 @@ module Polytypo
90
92
  end
91
93
  analyze_yaml(input, locale, keys, rules, narrow_target)
92
94
  else # "markdown"
93
- analyze_markdown(input, locale, dialect, rules, narrow_target)
95
+ analyze_markdown(input, locale, dialect, rules, narrow_target, frontmatter_keys)
94
96
  end
95
97
  end
96
98
 
@@ -143,18 +145,30 @@ module Polytypo
143
145
  end
144
146
  private_class_method :transform_yaml
145
147
 
146
- def self.transform_markdown(input, locale, dialect, rules, narrow_target)
148
+ def self.transform_markdown(input, locale, dialect, rules, narrow_target, frontmatter_keys = nil)
147
149
  # Validation order is public, tested behaviour, identical across every runtime: rules (an
148
- # unknown rule id), then locale (an unknown locale), then dialect/parsing.
150
+ # unknown rule id), then locale (an unknown locale), then dialect, then frontmatter_keys, then
151
+ # parsing (modes.md 3.7.4).
149
152
  resolved_locale, locale_data, plan = Engine::Pipeline.prepare(locale, rules)
150
153
  require_relative "polytypo/modes/markdown"
151
154
  Modes::Markdown.resolve_dialect(dialect)
152
- spans = Modes::Markdown.markdown_spans(input)
155
+ resolved_keys = Engine.resolve_frontmatter_keys(frontmatter_keys)
153
156
  ctx = Engine::RuleContext.new(mode: "markdown", dialect: dialect, locale: resolved_locale,
154
157
  narrow_target: narrow_target)
155
158
  cp = Engine::Codepoints.to_codepoints(input)
156
- Modes::Runner.run_over_spans(cp, spans, plan, locale_data, ctx)
159
+ Modes::Runner.run_over_units(cp, markdown_units(input, resolved_keys), plan, locale_data, ctx)
157
160
  end
161
+
162
+ # modes.md 3.7.4: the body, and -- only when the caller named frontmatter keys -- the
163
+ # frontmatter block as a second text unit. With frontmatter_keys nil this is exactly the single
164
+ # unit every document had before spec 1.7.0, which is why no released output can move.
165
+ def self.markdown_units(input, resolved_keys)
166
+ body = Modes::Markdown.markdown_spans(input)
167
+ return [body] if resolved_keys.nil?
168
+
169
+ [Modes::Markdown.frontmatter_spans(input, resolved_keys), body]
170
+ end
171
+ private_class_method :markdown_units
158
172
  private_class_method :transform_markdown
159
173
 
160
174
  def self.analyze_text(input, locale, rules, narrow_target)
@@ -187,16 +201,17 @@ module Polytypo
187
201
  end
188
202
  private_class_method :analyze_yaml
189
203
 
190
- def self.analyze_markdown(input, locale, dialect, rules, narrow_target)
204
+ def self.analyze_markdown(input, locale, dialect, rules, narrow_target, frontmatter_keys = nil)
191
205
  # Validation order is public, tested behaviour and is shared with .transform: rules, then
192
- # locale, then dialect/parsing (analyze.md section 4, A1).
206
+ # locale, then dialect, then frontmatter_keys, then parsing (analyze.md section 4, A1).
193
207
  resolved_locale, locale_data, plan = Engine::Pipeline.prepare(locale, rules)
194
208
  require_relative "polytypo/modes/markdown"
195
209
  Modes::Markdown.resolve_dialect(dialect)
196
- spans = Modes::Markdown.markdown_spans(input)
210
+ resolved_keys = Engine.resolve_frontmatter_keys(frontmatter_keys)
197
211
  ctx = Engine::RuleContext.new(mode: "markdown", dialect: dialect, locale: resolved_locale,
198
212
  narrow_target: narrow_target)
199
- Modes::Runner.analyze_over_spans(Engine::Codepoints.to_codepoints(input), spans, plan, locale_data, ctx)
213
+ Modes::Runner.analyze_over_units(Engine::Codepoints.to_codepoints(input),
214
+ markdown_units(input, resolved_keys), plan, locale_data, ctx)
200
215
  end
201
216
  private_class_method :analyze_markdown
202
217
  end
metadata CHANGED
@@ -1,7 +1,7 @@
1
1
  --- !ruby/object:Gem::Specification
2
2
  name: polytypo
3
3
  version: !ruby/object:Gem::Version
4
- version: 1.6.3
4
+ version: 1.8.0
5
5
  platform: ruby
6
6
  authors:
7
7
  - Iurii Rogulia
@@ -36,6 +36,7 @@ files:
36
36
  - README.md
37
37
  - lib/polytypo.rb
38
38
  - lib/polytypo/data/.not-canonical
39
+ - lib/polytypo/data/.not-vendored
39
40
  - lib/polytypo/data/README.md
40
41
  - lib/polytypo/data/UNICODE
41
42
  - lib/polytypo/data/VERSION