polytypo 1.6.3 → 1.8.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/README.md +25 -0
- data/lib/polytypo/data/.not-vendored +11 -0
- data/lib/polytypo/data/README.md +16 -11
- data/lib/polytypo/data/VERSION +1 -1
- data/lib/polytypo/data/fixtures/cs.json +1 -1
- data/lib/polytypo/data/fixtures/de-CH.json +1 -1
- data/lib/polytypo/data/fixtures/de-DE.json +14 -1
- data/lib/polytypo/data/fixtures/el.json +1 -1
- data/lib/polytypo/data/fixtures/en-GB.json +1 -1
- data/lib/polytypo/data/fixtures/en-US.json +521 -1
- data/lib/polytypo/data/fixtures/es.json +1 -1
- data/lib/polytypo/data/fixtures/fi.json +1 -1
- data/lib/polytypo/data/fixtures/fr-CA.json +1 -1
- data/lib/polytypo/data/fixtures/fr.json +68 -1
- data/lib/polytypo/data/fixtures/it.json +1 -1
- data/lib/polytypo/data/fixtures/locale-resolution.json +1 -1
- data/lib/polytypo/data/fixtures/nl.json +1 -1
- data/lib/polytypo/data/fixtures/pl.json +1 -1
- data/lib/polytypo/data/fixtures/pt-BR.json +1 -1
- data/lib/polytypo/data/fixtures/pt-PT.json +1 -1
- data/lib/polytypo/data/fixtures/ru.json +1 -1
- data/lib/polytypo/data/fixtures/sv.json +1 -1
- data/lib/polytypo/data/fixtures/tr.json +1 -1
- data/lib/polytypo/data/fixtures/uk.json +1 -1
- data/lib/polytypo/data/locales/cs.json +56 -37
- data/lib/polytypo/data/locales/de-CH.json +43 -34
- data/lib/polytypo/data/locales/de-DE.json +47 -32
- data/lib/polytypo/data/locales/el.json +51 -1
- data/lib/polytypo/data/locales/en-GB.json +56 -4
- data/lib/polytypo/data/locales/en-US.json +68 -4
- data/lib/polytypo/data/locales/es.json +66 -29
- data/lib/polytypo/data/locales/fi.json +77 -7
- data/lib/polytypo/data/locales/fr-CA.json +63 -36
- data/lib/polytypo/data/locales/fr.json +70 -41
- data/lib/polytypo/data/locales/it.json +65 -22
- data/lib/polytypo/data/locales/nl.json +61 -29
- data/lib/polytypo/data/locales/pl.json +61 -28
- data/lib/polytypo/data/locales/pt-BR.json +53 -28
- data/lib/polytypo/data/locales/pt-PT.json +55 -28
- data/lib/polytypo/data/locales/registry.json +1 -1
- data/lib/polytypo/data/locales/ru.json +45 -30
- data/lib/polytypo/data/locales/sv.json +63 -4
- data/lib/polytypo/data/locales/tr.json +72 -32
- data/lib/polytypo/data/locales/uk.json +55 -27
- data/lib/polytypo/data/rules/modes.md +430 -11
- data/lib/polytypo/data/rules/order.json +1 -1
- data/lib/polytypo/data/schema/fixtures.schema.json +12 -1
- data/lib/polytypo/engine/pipeline.rb +29 -0
- data/lib/polytypo/modes/markdown.rb +256 -31
- data/lib/polytypo/modes/runner.rb +33 -3
- data/lib/polytypo/version.rb +1 -1
- data/lib/polytypo.rb +27 -12
- metadata +2 -1
|
@@ -2,6 +2,7 @@
|
|
|
2
2
|
|
|
3
3
|
require_relative "spans"
|
|
4
4
|
require_relative "html"
|
|
5
|
+
require_relative "yaml"
|
|
5
6
|
require_relative "parse_error"
|
|
6
7
|
require_relative "../errors"
|
|
7
8
|
|
|
@@ -40,9 +41,11 @@ module Polytypo
|
|
|
40
41
|
end
|
|
41
42
|
|
|
42
43
|
# Block/inline node types that are never processable, in full (modes.md 3.7.3): fenced and
|
|
43
|
-
# indented code (comrak represents both as :code_block), inline code spans, thematic
|
|
44
|
-
# breaks (no prose content)
|
|
45
|
-
|
|
44
|
+
# indented code (comrak represents both as :code_block), inline code spans, and thematic
|
|
45
|
+
# breaks (no prose content). Frontmatter is NOT in this list and is not a node type here:
|
|
46
|
+
# comrak's front_matter_delimiter extension is never enabled, because since spec 1.8.0 the
|
|
47
|
+
# block's extent is modes.md 3.7.3a's scan and not a parser's opinion of it.
|
|
48
|
+
SKIP_WHOLE_TYPES = %i[code code_block thematic_break].freeze
|
|
46
49
|
|
|
47
50
|
# A leaf whose Segment/source_position directly delineates literal, processable prose --
|
|
48
51
|
# comrak's AST is segment-based like goldmark's: structural bytes (emphasis delimiters,
|
|
@@ -54,45 +57,253 @@ module Polytypo
|
|
|
54
57
|
# port's goldmark-based adapter).
|
|
55
58
|
TEXT_LEAF_TYPES = %i[text].freeze
|
|
56
59
|
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
return "---" if source.start_with?("---")
|
|
60
|
+
SPACE = " "
|
|
61
|
+
TAB = "\t"
|
|
62
|
+
CR = "\r"
|
|
63
|
+
LF = "\n"
|
|
64
|
+
BOM = "\uFEFF"
|
|
65
|
+
FRONTMATTER_DELIMITERS = ["---", "+++"].freeze
|
|
64
66
|
|
|
67
|
+
# What detect_frontmatter found: +end+, one past the closing delimiter line's terminator;
|
|
68
|
+
# the delimiter it used; and the content range -- from after the opening delimiter line's
|
|
69
|
+
# terminator to the code point that begins the closing delimiter line (modes.md 3.7.4).
|
|
70
|
+
# Both delimiter lines and every line terminator lie outside that content range. The block
|
|
71
|
+
# always begins at the document's first code point, or at the second when a byte-order mark
|
|
72
|
+
# was stepped over, and both are masked, so its start needs no member of its own.
|
|
73
|
+
FrontmatterBlock = Struct.new(:end, :delimiter, :content_start, :content_end)
|
|
74
|
+
|
|
75
|
+
# modes.md 3.7.3a (spec 1.8.0). WHERE THE BLOCK BEGINS AND ENDS IS DECIDED HERE, not by
|
|
76
|
+
# comrak's front_matter_delimiter extension: that extension requires both delimiter lines
|
|
77
|
+
# to match the literal string exactly, so a fence carrying one trailing space was not a
|
|
78
|
+
# block and its metadata was typeset as prose (polytypo/polytypo#58) -- the damage 3.7.3
|
|
79
|
+
# exists to prevent, reached by an invisible character no author typed on purpose.
|
|
80
|
+
#
|
|
81
|
+
# ONE SCAN, ONE ANSWER: this is the authority for both the body's skip (3.7.3) and the
|
|
82
|
+
# content range frontmatter_keys names (3.7.4). Two locators is how a skip and an option
|
|
83
|
+
# come to disagree about the same three lines, and it cost this runtime a second parse of
|
|
84
|
+
# the whole document as well (polytypo/polytypo#59).
|
|
85
|
+
#
|
|
86
|
+
# The steps, in the section's own order: the document must BEGIN with the delimiter, with
|
|
87
|
+
# no leading blank line and no indentation, a single leading U+FEFF stepped over first (a
|
|
88
|
+
# byte-order mark is not content, and reading it as content denies the block to every file
|
|
89
|
+
# some Windows editors write); the rest of that line may be U+0020 and U+0009 and nothing
|
|
90
|
+
# else, so "--- yaml" is not an opener and "----" is not a delimiter; the closing line is
|
|
91
|
+
# the first later line whose FIRST code point begins the same delimiter, followed by only
|
|
92
|
+
# U+0020 and U+0009 -- indentation disqualifies it exactly as it disqualifies the opener,
|
|
93
|
+
# "..." closes nothing, and neither does a delimiter of the other kind; with no such line
|
|
94
|
+
# there is no block.
|
|
95
|
+
#
|
|
96
|
+
# Offsets are code-point indices into +chars+, like every Span in this runtime.
|
|
97
|
+
def self.detect_frontmatter(chars)
|
|
98
|
+
start = chars[0] == BOM ? 1 : 0
|
|
99
|
+
delimiter = FRONTMATTER_DELIMITERS.find { |d| delimiter_at?(chars, start, d) }
|
|
100
|
+
return nil if delimiter.nil?
|
|
101
|
+
|
|
102
|
+
line_end, next_start = line_bounds(chars, start)
|
|
103
|
+
return nil unless spaces_and_tabs_only?(chars, start + delimiter.length, line_end)
|
|
104
|
+
|
|
105
|
+
find_closing_line(chars, delimiter, next_start)
|
|
106
|
+
end
|
|
107
|
+
private_class_method :detect_frontmatter
|
|
108
|
+
|
|
109
|
+
def self.find_closing_line(chars, delimiter, content_start)
|
|
110
|
+
cursor = content_start
|
|
111
|
+
while cursor < chars.length
|
|
112
|
+
line_end, next_start = line_bounds(chars, cursor)
|
|
113
|
+
if delimiter_at?(chars, cursor, delimiter) &&
|
|
114
|
+
spaces_and_tabs_only?(chars, cursor + delimiter.length, line_end)
|
|
115
|
+
return FrontmatterBlock.new(next_start, delimiter, content_start, cursor)
|
|
116
|
+
end
|
|
117
|
+
cursor = next_start
|
|
118
|
+
end
|
|
65
119
|
nil
|
|
66
120
|
end
|
|
67
|
-
private_class_method :
|
|
121
|
+
private_class_method :find_closing_line
|
|
122
|
+
|
|
123
|
+
def self.delimiter_at?(chars, i, delimiter)
|
|
124
|
+
delimiter.each_char.with_index.all? { |ch, k| chars[i + k] == ch }
|
|
125
|
+
end
|
|
126
|
+
private_class_method :delimiter_at?
|
|
127
|
+
|
|
128
|
+
# modes.md 3.7.3a step 5. A LINE ENDS AS COMMONMARK ENDS ONE -- at U+000A, at a U+000D not
|
|
129
|
+
# followed by U+000A, or at the end of input -- and the terminator is never part of the
|
|
130
|
+
# line. Returns [line_end, next_line_start]: a CRLF document gives the same block as the
|
|
131
|
+
# same bytes with LF, a final "---\r" with no U+000A still closes its block, and a document
|
|
132
|
+
# written with lone U+000D endings has lines at all.
|
|
133
|
+
#
|
|
134
|
+
# DELIBERATELY NOT 3.8.4's LF-only model, which Yaml.yaml_spans keeps for the block's
|
|
135
|
+
# content: the block is a Markdown construct and ends its lines the way the language around
|
|
136
|
+
# it does. 3.8.4 is a deliberate simplification rather than what YAML says (1.2.2 5.4 admits
|
|
137
|
+
# a lone U+000D too), and widening it here would change "yaml" mode for every caller, so the
|
|
138
|
+
# two models stay different on purpose -- the cost is modes.md 7.13's: inside a lone-U+000D
|
|
139
|
+
# block frontmatter_keys yields nothing at all. Harmonising them silently changes one.
|
|
140
|
+
def self.line_bounds(chars, from)
|
|
141
|
+
i = from
|
|
142
|
+
i += 1 while i < chars.length && chars[i] != LF && chars[i] != CR
|
|
143
|
+
return [i, i] if i >= chars.length
|
|
144
|
+
|
|
145
|
+
[i, chars[i] == CR && chars[i + 1] == LF ? i + 2 : i + 1]
|
|
146
|
+
end
|
|
147
|
+
private_class_method :line_bounds
|
|
148
|
+
|
|
149
|
+
def self.spaces_and_tabs_only?(chars, from, to)
|
|
150
|
+
(from...to).all? { |i| chars[i] == SPACE || chars[i] == TAB }
|
|
151
|
+
end
|
|
152
|
+
private_class_method :spaces_and_tabs_only?
|
|
153
|
+
|
|
154
|
+
# modes.md 3.7.3a: THE PARSER IS HANDED THE BLOCK MASKED OUT -- the block, both delimiter
|
|
155
|
+
# lines included and the leading U+FEFF of step 1 with them, replaced by U+0020 with line
|
|
156
|
+
# terminators kept as they are, so nothing inside it can form or close a construct in the
|
|
157
|
+
# body. The mask runs from 0: the only code point that can precede the delimiter is that
|
|
158
|
+
# mark.
|
|
159
|
+
#
|
|
160
|
+
# THE INVARIANT IS POSITIONAL ALIGNMENT IN THE UNIT THIS RUNTIME MAPS PARSER OFFSETS BACK
|
|
161
|
+
# THROUGH, which is not the same as "one U+0020 per index unit". comrak reports line and
|
|
162
|
+
# column in CODE POINTS under `sourcepos_chars: true`, and build_line_starts resolves them
|
|
163
|
+
# against the original document, so code-point alignment is all this runtime owes -- one
|
|
164
|
+
# space per character gives it, and in Ruby, whose strings are sequences of code points,
|
|
165
|
+
# byte-length preservation cannot even be expressed. A runtime that hands its parser's BYTE
|
|
166
|
+
# offsets straight through owes byte-length preservation instead, and masking an astral
|
|
167
|
+
# character to a single space would shorten its source by three.
|
|
168
|
+
#
|
|
169
|
+
# Suppressing the block's spans after the parse is NOT a substitute for masking, and is
|
|
170
|
+
# measured wrong in the spec's own table: a fenced-code line inside a metadata value pairs
|
|
171
|
+
# with the body's own fence, and the body's code block and its prose swap roles. No
|
|
172
|
+
# span-level check can see that, because the damage is in what the parser concluded before
|
|
173
|
+
# any span existed. The two are not alternatives -- 3.7.3a requires both, and the clip in
|
|
174
|
+
# emit_span is the other half.
|
|
175
|
+
def self.mask_frontmatter(source, block)
|
|
176
|
+
return source if block.nil?
|
|
177
|
+
|
|
178
|
+
segment = source[0, block.end].each_char.map { |ch| ch == LF || ch == CR ? ch : SPACE }
|
|
179
|
+
masked = source.dup
|
|
180
|
+
masked[0, block.end] = segment.join
|
|
181
|
+
masked
|
|
182
|
+
end
|
|
183
|
+
private_class_method :mask_frontmatter
|
|
184
|
+
|
|
185
|
+
# A LOCAL WORKAROUND FOR A MEASURED commonmarker 2.10.0 DEFECT, not a spec rule. With
|
|
186
|
+
# `sourcepos_chars: true` comrak converts its byte columns to code-point columns, and on a
|
|
187
|
+
# document whose lines end in a LONE U+000D the conversion does not reach them: measured on
|
|
188
|
+
# line 3 of such a document, "Body has “quotes” here." comes back with end_column 27, its
|
|
189
|
+
# byte length, instead of 23 -- the very number the option exists to remove. The same
|
|
190
|
+
# document with LF or CRLF endings is converted correctly.
|
|
191
|
+
#
|
|
192
|
+
# U+000D not followed by U+000A is replaced by U+000A for the parser's copy only. The three
|
|
193
|
+
# line endings are the same construct in CommonMark, the replacement is one code point for
|
|
194
|
+
# one, so every source position still addresses the original document -- which is what the
|
|
195
|
+
# output is emitted from, carriage returns and all.
|
|
196
|
+
# Does the 3.8.4 line containing +index+ carry a U+000D that is not followed by U+000A? The
|
|
197
|
+
# line is LF-delimited, because this is the content's own line model (3.8.4), not 3.7.3a's.
|
|
198
|
+
def self.lone_cr_line?(content, index)
|
|
199
|
+
from = (content.rindex(LF, index) || -1) + 1
|
|
200
|
+
to = content.index(LF, index) || content.length
|
|
201
|
+
i = content.index(CR, from)
|
|
202
|
+
while i && i < to
|
|
203
|
+
return true if content[i + 1] != LF
|
|
204
|
+
|
|
205
|
+
i = content.index(CR, i + 1)
|
|
206
|
+
end
|
|
207
|
+
false
|
|
208
|
+
end
|
|
209
|
+
private_class_method :lone_cr_line?
|
|
210
|
+
|
|
211
|
+
def self.normalize_lone_cr(source)
|
|
212
|
+
i = source.index(CR)
|
|
213
|
+
return source if i.nil?
|
|
214
|
+
|
|
215
|
+
normalized = source.dup
|
|
216
|
+
while i
|
|
217
|
+
normalized[i] = LF if normalized[i + 1] != LF
|
|
218
|
+
i = normalized.index(CR, i + 1)
|
|
219
|
+
end
|
|
220
|
+
normalized
|
|
221
|
+
end
|
|
222
|
+
private_class_method :normalize_lone_cr
|
|
223
|
+
|
|
224
|
+
# modes.md 3.7.4, spec 1.7.0. The frontmatter block's own spans, which form a SECOND TEXT
|
|
225
|
+
# UNIT: the pipeline runs over them separately from the body's, so an unbalanced mark in a
|
|
226
|
+
# metadata field can never pair with one in the first paragraph, and the option cannot
|
|
227
|
+
# change a byte outside the block.
|
|
228
|
+
#
|
|
229
|
+
# Spans come from the scan of modes.md 3.8 -- frontmatter IS YAML, and implementing that
|
|
230
|
+
# grammar twice is how two implementations of one spec drift -- with +keys+ as step 8's key
|
|
231
|
+
# predicate. The block is the construct markdown_spans masks out (3.7.3, the same
|
|
232
|
+
# detect_frontmatter scan), so the option only ever adds spans where the skip removed them:
|
|
233
|
+
# no source position belongs to both units. A TOML block yields nothing, with the option or
|
|
234
|
+
# without it -- its quoting is a second grammar this scan does not claim (modes.md 7.13).
|
|
235
|
+
#
|
|
236
|
+
# The content range comes from detect_frontmatter, the same scan markdown_spans skips by
|
|
237
|
+
# (3.7.3a, spec 1.8.0). Locating the block needs no parser at all, which is why this method
|
|
238
|
+
# no longer parses the document a second time.
|
|
239
|
+
#
|
|
240
|
+
# A CONTENT LINE CARRYING A U+000D NOT FOLLOWED BY U+000A YIELDS NO SPANS (3.7.4), per line
|
|
241
|
+
# and not per block, exactly as 3.8.4 step 1 declines a line containing U+0009: a stray
|
|
242
|
+
# U+000D inside one quoted value costs that value and not the whole block. The two line
|
|
243
|
+
# models meet here and do not compose -- 3.7.3a step 5 finds the block in a lone-U+000D
|
|
244
|
+
# document and 3.8.4's LF-only scan then reads the whole of it as ONE line, so such a
|
|
245
|
+
# document loses every span, which is the price of not widening 3.8.4 (that would change
|
|
246
|
+
# "yaml" mode for every caller). Measured, letting it through is not merely inert: marks
|
|
247
|
+
# pair across mapping lines, an unlisted line inside a listed key's scalar takes the
|
|
248
|
+
# locale's spacing, and the U+000D lands inside a span, which 3.8.4 forbids.
|
|
249
|
+
def self.frontmatter_spans(source, keys)
|
|
250
|
+
return [] if keys.empty?
|
|
251
|
+
|
|
252
|
+
cp = Polytypo::Engine::Codepoints.to_codepoints(source)
|
|
253
|
+
chars = cp.map { |c| [c].pack("U") }
|
|
254
|
+
block = detect_frontmatter(chars)
|
|
255
|
+
return [] if block.nil? || block.delimiter != "---"
|
|
256
|
+
return [] if block.content_end <= block.content_start
|
|
257
|
+
|
|
258
|
+
content = chars[block.content_start...block.content_end].join
|
|
259
|
+
spans = Yaml.yaml_spans(content, keys)
|
|
260
|
+
spans = spans.reject { |span| lone_cr_line?(content, span.start) } if content.include?(CR)
|
|
261
|
+
spans.map do |span|
|
|
262
|
+
Spans::Span.new(span.start + block.content_start, span.end + block.content_start)
|
|
263
|
+
end
|
|
264
|
+
end
|
|
68
265
|
|
|
69
266
|
# Locates the processable spans of a Markdown document. dialect must already be validated
|
|
70
267
|
# via resolve_dialect (== "commonmark"); this function does not re-check it.
|
|
268
|
+
#
|
|
269
|
+
# The frontmatter block is located by detect_frontmatter and masked out of the source the
|
|
270
|
+
# parser sees (3.7.3a), so comrak needs no frontmatter concept and is never given its
|
|
271
|
+
# front_matter_delimiter option: a masked document has no frontmatter left to find, and
|
|
272
|
+
# the extent is this runtime's scan rather than an extension's exact-match opinion of it.
|
|
273
|
+
#
|
|
274
|
+
# frontmatter_end carries 3.7.3a's second requirement: NO SPAN MAY LIE INSIDE THE BLOCK,
|
|
275
|
+
# whatever the parser did with the masked text. Masking is what makes that true for comrak
|
|
276
|
+
# -- measured, it emits no node inside the masked range -- and the spec keeps the rule
|
|
277
|
+
# anyway because it is not true of every parser: tree-sitter-markdown reads a final
|
|
278
|
+
# all-space line with no terminator as a paragraph and would emit a span over the closing
|
|
279
|
+
# delimiter itself.
|
|
71
280
|
def self.markdown_spans(source)
|
|
72
281
|
require "commonmarker"
|
|
73
282
|
|
|
74
283
|
cp = Polytypo::Engine::Codepoints.to_codepoints(source)
|
|
75
284
|
chars = cp.map { |c| [c].pack("U") }
|
|
76
285
|
line_starts = build_line_starts(chars)
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
options = { parse: { sourcepos_chars: true } }
|
|
80
|
-
options[:extension] = { front_matter_delimiter: delimiter } if delimiter
|
|
286
|
+
block = detect_frontmatter(chars)
|
|
287
|
+
parser_source = normalize_lone_cr(mask_frontmatter(source, block))
|
|
81
288
|
|
|
82
289
|
spans = []
|
|
83
290
|
ParseError.wrap do
|
|
84
|
-
doc = Commonmarker.parse(
|
|
85
|
-
|
|
86
|
-
walk(doc, chars, line_starts, spans, html_stack_holder)
|
|
291
|
+
doc = Commonmarker.parse(parser_source, options: { parse: { sourcepos_chars: true } })
|
|
292
|
+
walk(doc, chars, line_starts, spans, { stack: [], frontmatter_end: block.nil? ? 0 : block.end })
|
|
87
293
|
end
|
|
88
294
|
spans
|
|
89
295
|
end
|
|
90
296
|
|
|
91
297
|
# line_starts[i] is the 0-based code-point index at which comrak's 1-based line i begins.
|
|
298
|
+
# comrak counts CommonMark line endings, which are U+000A, U+000D, and U+000D U+000A --
|
|
299
|
+
# measured, not assumed: a document with lone U+000D endings reports its body on line 5,
|
|
300
|
+
# and an LF-only table would have no entry to map that to.
|
|
92
301
|
def self.build_line_starts(chars)
|
|
93
302
|
starts = [0]
|
|
94
303
|
chars.each_with_index do |ch, i|
|
|
95
|
-
|
|
304
|
+
next unless ch == LF || (ch == CR && chars[i + 1] != LF)
|
|
305
|
+
|
|
306
|
+
starts << (i + 1)
|
|
96
307
|
end
|
|
97
308
|
starts
|
|
98
309
|
end
|
|
@@ -103,17 +314,33 @@ module Polytypo
|
|
|
103
314
|
end
|
|
104
315
|
private_class_method :offset_of
|
|
105
316
|
|
|
106
|
-
|
|
317
|
+
# modes.md 3.7.3a's second requirement: NO SPAN MAY LIE INSIDE THE BLOCK, whatever the parser
|
|
318
|
+
# did with the masked text. A span that overlaps the block is CLIPPED to the part outside it
|
|
319
|
+
# and dropped when nothing is left -- the block's own characters must not reach the rules,
|
|
320
|
+
# and body prose past the block must not be lost to a parser's mistake about where the block
|
|
321
|
+
# ended. It never fires with comrak, which emits no node inside the masked range -- measured
|
|
322
|
+
# by counting clips over every markdown fixture's input and output, 216 spans, none clipped
|
|
323
|
+
# -- and it is implemented anyway because the spec states it of every runtime:
|
|
324
|
+
# tree-sitter-markdown reads a final all-space line with no terminator as a paragraph.
|
|
325
|
+
def self.emit_span(spans, start_offset, end_offset, state)
|
|
326
|
+
start_offset = state[:frontmatter_end] if start_offset < state[:frontmatter_end]
|
|
327
|
+
return if end_offset <= start_offset
|
|
328
|
+
|
|
329
|
+
spans << Spans::Span.new(start_offset, end_offset)
|
|
330
|
+
end
|
|
331
|
+
private_class_method :emit_span
|
|
332
|
+
|
|
333
|
+
def self.walk(node, chars, line_starts, spans, state)
|
|
107
334
|
type = node.type
|
|
108
335
|
return if SKIP_WHOLE_TYPES.include?(type)
|
|
109
336
|
|
|
110
337
|
if type == :html_block
|
|
111
|
-
handle_html_block(node, chars, line_starts, spans)
|
|
338
|
+
handle_html_block(node, chars, line_starts, spans, state)
|
|
112
339
|
return
|
|
113
340
|
end
|
|
114
341
|
|
|
115
342
|
if type == :html_inline
|
|
116
|
-
handle_html_inline(node, chars, line_starts,
|
|
343
|
+
handle_html_inline(node, chars, line_starts, state)
|
|
117
344
|
return
|
|
118
345
|
end
|
|
119
346
|
|
|
@@ -135,23 +362,21 @@ module Polytypo
|
|
|
135
362
|
sp = node.source_position
|
|
136
363
|
start_offset = offset_of(line_starts, sp[:start_line], sp[:start_column])
|
|
137
364
|
end_offset = offset_of(line_starts, sp[:end_line], sp[:end_column]) + 1
|
|
138
|
-
if
|
|
139
|
-
spans << Spans::Span.new(start_offset, end_offset)
|
|
140
|
-
end
|
|
365
|
+
emit_span(spans, start_offset, end_offset, state) if state[:stack].empty?
|
|
141
366
|
return
|
|
142
367
|
end
|
|
143
368
|
|
|
144
369
|
if %i[paragraph heading table_cell].include?(type)
|
|
145
370
|
# A fresh top-level inline-bearing container: the raw-HTML skip stack resets here (an
|
|
146
371
|
# unclosed skipped start tag skips to the end of the block, modes.md 3.7.3).
|
|
147
|
-
|
|
372
|
+
state[:stack] = []
|
|
148
373
|
end
|
|
149
374
|
|
|
150
|
-
node.each { |child| walk(child, chars, line_starts, spans,
|
|
375
|
+
node.each { |child| walk(child, chars, line_starts, spans, state) }
|
|
151
376
|
end
|
|
152
377
|
private_class_method :walk
|
|
153
378
|
|
|
154
|
-
def self.handle_html_block(node, chars, line_starts, spans)
|
|
379
|
+
def self.handle_html_block(node, chars, line_starts, spans, state)
|
|
155
380
|
sp = node.source_position
|
|
156
381
|
start_offset = offset_of(line_starts, sp[:start_line], sp[:start_column])
|
|
157
382
|
end_offset = offset_of(line_starts, sp[:end_line], sp[:end_column]) + 1
|
|
@@ -159,12 +384,12 @@ module Polytypo
|
|
|
159
384
|
|
|
160
385
|
text = chars[start_offset...end_offset].join
|
|
161
386
|
Html.html_spans(text).each do |s|
|
|
162
|
-
spans
|
|
387
|
+
emit_span(spans, start_offset + s.start, start_offset + s.end, state)
|
|
163
388
|
end
|
|
164
389
|
end
|
|
165
390
|
private_class_method :handle_html_block
|
|
166
391
|
|
|
167
|
-
def self.handle_html_inline(node, chars, line_starts,
|
|
392
|
+
def self.handle_html_inline(node, chars, line_starts, state)
|
|
168
393
|
sp = node.source_position
|
|
169
394
|
start_offset = offset_of(line_starts, sp[:start_line], sp[:start_column])
|
|
170
395
|
end_offset = offset_of(line_starts, sp[:end_line], sp[:end_column]) + 1
|
|
@@ -172,7 +397,7 @@ module Polytypo
|
|
|
172
397
|
name, closing, self_closing = Html.read_tag(raw) || [nil, nil, nil]
|
|
173
398
|
return if name.nil?
|
|
174
399
|
|
|
175
|
-
stack =
|
|
400
|
+
stack = state[:stack]
|
|
176
401
|
if closing
|
|
177
402
|
if !stack.empty? && stack[-1] == name
|
|
178
403
|
stack.pop
|
|
@@ -34,17 +34,46 @@ module Polytypo
|
|
|
34
34
|
# source_cp is the whole input already converted to code points; spans address it by
|
|
35
35
|
# code-point index.
|
|
36
36
|
def self.run_over_spans(source_cp, spans, plan, locale_data, ctx)
|
|
37
|
+
emit(source_cp, replacements_of_unit(source_cp, spans, plan, locale_data, ctx))
|
|
38
|
+
end
|
|
39
|
+
|
|
40
|
+
# modes.md 3.1 and 3.5 step 3 (spec 1.7.0). A document has one text unit, except in
|
|
41
|
+
# "markdown" with frontmatter_keys, where the frontmatter block's spans form a unit of their
|
|
42
|
+
# own. The pipeline runs once per unit and the two edit sets are disjoint, because no span of
|
|
43
|
+
# one unit lies inside the other -- which is what the body's walk skipping the block
|
|
44
|
+
# guarantees. Only step 5 is shared: the source is emitted once, in document order.
|
|
45
|
+
def self.run_over_units(source_cp, units, plan, locale_data, ctx)
|
|
46
|
+
replacements = units.flat_map do |spans|
|
|
47
|
+
replacements_of_unit(source_cp, spans, plan, locale_data, ctx)
|
|
48
|
+
end
|
|
49
|
+
emit(source_cp, replacements.sort_by { |span, _piece| span.start })
|
|
50
|
+
end
|
|
51
|
+
|
|
52
|
+
# analyze_over_spans per text unit (modes.md 3.1), reported in document order.
|
|
53
|
+
def self.analyze_over_units(source_cp, units, plan, locale_data, ctx)
|
|
54
|
+
units.flat_map { |spans| analyze_over_spans(source_cp, spans, plan, locale_data, ctx) }
|
|
55
|
+
.sort_by(&:start)
|
|
56
|
+
end
|
|
57
|
+
|
|
58
|
+
# One text unit: the marker-separated concatenation, the pipeline, and the pieces it
|
|
59
|
+
# produced, paired with the spans they replace.
|
|
60
|
+
def self.replacements_of_unit(source_cp, spans, plan, locale_data, ctx)
|
|
37
61
|
normalized = Spans.normalize_spans(spans)
|
|
38
|
-
return
|
|
62
|
+
return [] if normalized.empty?
|
|
39
63
|
|
|
40
64
|
concatenated = Spans.concatenate_spans(source_cp, normalized)
|
|
41
65
|
transformed = run_rules_over_spans(concatenated, plan, locale_data, ctx)
|
|
42
66
|
pieces = Spans.split_on_marker(transformed, normalized.length)
|
|
67
|
+
normalized.each_with_index.map { |span, i| [span, pieces[i]] }
|
|
68
|
+
end
|
|
69
|
+
private_class_method :replacements_of_unit
|
|
43
70
|
|
|
71
|
+
# modes.md 4: the source with disjoint replacements applied at recorded offsets, and nothing
|
|
72
|
+
# else changed.
|
|
73
|
+
def self.emit(source_cp, replacements)
|
|
44
74
|
out = []
|
|
45
75
|
cursor = 0
|
|
46
|
-
|
|
47
|
-
piece = pieces[i]
|
|
76
|
+
replacements.each do |span, piece|
|
|
48
77
|
original = source_cp[span.start...span.end]
|
|
49
78
|
out.concat(source_cp[cursor...span.start])
|
|
50
79
|
out.concat(piece == original ? original : piece)
|
|
@@ -53,6 +82,7 @@ module Polytypo
|
|
|
53
82
|
out.concat(source_cp[cursor..])
|
|
54
83
|
out.pack("U*")
|
|
55
84
|
end
|
|
85
|
+
private_class_method :emit
|
|
56
86
|
|
|
57
87
|
# run_over_spans, reporting instead of applying (analyze.md section 1). The span table
|
|
58
88
|
# supplies the origin map, so every change comes back in DOCUMENT coordinates -- analyze.md
|
data/lib/polytypo/version.rb
CHANGED
data/lib/polytypo.rb
CHANGED
|
@@ -32,7 +32,8 @@ module Polytypo
|
|
|
32
32
|
#
|
|
33
33
|
# Only `mode: "text"`/`"html"` ever touch this file's own requires; `mode: "markdown"` lazily
|
|
34
34
|
# requires "commonmarker" from within Modes::Markdown, never at load time of this file.
|
|
35
|
-
def self.transform(input, locale:, mode: "text", dialect: nil, keys: nil, rules: nil, narrow_nbsp: nil
|
|
35
|
+
def self.transform(input, locale:, mode: "text", dialect: nil, keys: nil, rules: nil, narrow_nbsp: nil,
|
|
36
|
+
frontmatter_keys: nil)
|
|
36
37
|
resolved_mode = resolve_mode(mode)
|
|
37
38
|
narrow_target = Engine.resolve_narrow_target(narrow_nbsp)
|
|
38
39
|
|
|
@@ -53,7 +54,7 @@ module Polytypo
|
|
|
53
54
|
end
|
|
54
55
|
transform_yaml(input, locale, keys, rules, narrow_target)
|
|
55
56
|
else # "markdown"
|
|
56
|
-
transform_markdown(input, locale, dialect, rules, narrow_target)
|
|
57
|
+
transform_markdown(input, locale, dialect, rules, narrow_target, frontmatter_keys)
|
|
57
58
|
end
|
|
58
59
|
end
|
|
59
60
|
|
|
@@ -69,7 +70,8 @@ module Polytypo
|
|
|
69
70
|
# for the text (analyze.md sections 4 and 5).
|
|
70
71
|
#
|
|
71
72
|
# Pure and thread-safe on the same terms as .transform.
|
|
72
|
-
def self.analyze(input, locale:, mode: "text", dialect: nil, keys: nil, rules: nil, narrow_nbsp: nil
|
|
73
|
+
def self.analyze(input, locale:, mode: "text", dialect: nil, keys: nil, rules: nil, narrow_nbsp: nil,
|
|
74
|
+
frontmatter_keys: nil)
|
|
73
75
|
resolved_mode = resolve_mode(mode)
|
|
74
76
|
narrow_target = Engine.resolve_narrow_target(narrow_nbsp)
|
|
75
77
|
|
|
@@ -90,7 +92,7 @@ module Polytypo
|
|
|
90
92
|
end
|
|
91
93
|
analyze_yaml(input, locale, keys, rules, narrow_target)
|
|
92
94
|
else # "markdown"
|
|
93
|
-
analyze_markdown(input, locale, dialect, rules, narrow_target)
|
|
95
|
+
analyze_markdown(input, locale, dialect, rules, narrow_target, frontmatter_keys)
|
|
94
96
|
end
|
|
95
97
|
end
|
|
96
98
|
|
|
@@ -143,18 +145,30 @@ module Polytypo
|
|
|
143
145
|
end
|
|
144
146
|
private_class_method :transform_yaml
|
|
145
147
|
|
|
146
|
-
def self.transform_markdown(input, locale, dialect, rules, narrow_target)
|
|
148
|
+
def self.transform_markdown(input, locale, dialect, rules, narrow_target, frontmatter_keys = nil)
|
|
147
149
|
# Validation order is public, tested behaviour, identical across every runtime: rules (an
|
|
148
|
-
# unknown rule id), then locale (an unknown locale), then dialect
|
|
150
|
+
# unknown rule id), then locale (an unknown locale), then dialect, then frontmatter_keys, then
|
|
151
|
+
# parsing (modes.md 3.7.4).
|
|
149
152
|
resolved_locale, locale_data, plan = Engine::Pipeline.prepare(locale, rules)
|
|
150
153
|
require_relative "polytypo/modes/markdown"
|
|
151
154
|
Modes::Markdown.resolve_dialect(dialect)
|
|
152
|
-
|
|
155
|
+
resolved_keys = Engine.resolve_frontmatter_keys(frontmatter_keys)
|
|
153
156
|
ctx = Engine::RuleContext.new(mode: "markdown", dialect: dialect, locale: resolved_locale,
|
|
154
157
|
narrow_target: narrow_target)
|
|
155
158
|
cp = Engine::Codepoints.to_codepoints(input)
|
|
156
|
-
Modes::Runner.
|
|
159
|
+
Modes::Runner.run_over_units(cp, markdown_units(input, resolved_keys), plan, locale_data, ctx)
|
|
157
160
|
end
|
|
161
|
+
|
|
162
|
+
# modes.md 3.7.4: the body, and -- only when the caller named frontmatter keys -- the
|
|
163
|
+
# frontmatter block as a second text unit. With frontmatter_keys nil this is exactly the single
|
|
164
|
+
# unit every document had before spec 1.7.0, which is why no released output can move.
|
|
165
|
+
def self.markdown_units(input, resolved_keys)
|
|
166
|
+
body = Modes::Markdown.markdown_spans(input)
|
|
167
|
+
return [body] if resolved_keys.nil?
|
|
168
|
+
|
|
169
|
+
[Modes::Markdown.frontmatter_spans(input, resolved_keys), body]
|
|
170
|
+
end
|
|
171
|
+
private_class_method :markdown_units
|
|
158
172
|
private_class_method :transform_markdown
|
|
159
173
|
|
|
160
174
|
def self.analyze_text(input, locale, rules, narrow_target)
|
|
@@ -187,16 +201,17 @@ module Polytypo
|
|
|
187
201
|
end
|
|
188
202
|
private_class_method :analyze_yaml
|
|
189
203
|
|
|
190
|
-
def self.analyze_markdown(input, locale, dialect, rules, narrow_target)
|
|
204
|
+
def self.analyze_markdown(input, locale, dialect, rules, narrow_target, frontmatter_keys = nil)
|
|
191
205
|
# Validation order is public, tested behaviour and is shared with .transform: rules, then
|
|
192
|
-
# locale, then dialect
|
|
206
|
+
# locale, then dialect, then frontmatter_keys, then parsing (analyze.md section 4, A1).
|
|
193
207
|
resolved_locale, locale_data, plan = Engine::Pipeline.prepare(locale, rules)
|
|
194
208
|
require_relative "polytypo/modes/markdown"
|
|
195
209
|
Modes::Markdown.resolve_dialect(dialect)
|
|
196
|
-
|
|
210
|
+
resolved_keys = Engine.resolve_frontmatter_keys(frontmatter_keys)
|
|
197
211
|
ctx = Engine::RuleContext.new(mode: "markdown", dialect: dialect, locale: resolved_locale,
|
|
198
212
|
narrow_target: narrow_target)
|
|
199
|
-
Modes::Runner.
|
|
213
|
+
Modes::Runner.analyze_over_units(Engine::Codepoints.to_codepoints(input),
|
|
214
|
+
markdown_units(input, resolved_keys), plan, locale_data, ctx)
|
|
200
215
|
end
|
|
201
216
|
private_class_method :analyze_markdown
|
|
202
217
|
end
|
metadata
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
--- !ruby/object:Gem::Specification
|
|
2
2
|
name: polytypo
|
|
3
3
|
version: !ruby/object:Gem::Version
|
|
4
|
-
version: 1.
|
|
4
|
+
version: 1.8.0
|
|
5
5
|
platform: ruby
|
|
6
6
|
authors:
|
|
7
7
|
- Iurii Rogulia
|
|
@@ -36,6 +36,7 @@ files:
|
|
|
36
36
|
- README.md
|
|
37
37
|
- lib/polytypo.rb
|
|
38
38
|
- lib/polytypo/data/.not-canonical
|
|
39
|
+
- lib/polytypo/data/.not-vendored
|
|
39
40
|
- lib/polytypo/data/README.md
|
|
40
41
|
- lib/polytypo/data/UNICODE
|
|
41
42
|
- lib/polytypo/data/VERSION
|