polytypo 1.1.0 → 1.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (66) hide show
  1. checksums.yaml +4 -4
  2. data/README.md +33 -1
  3. data/lib/polytypo/data/VERSION +1 -1
  4. data/lib/polytypo/data/fixtures/cs.json +161 -0
  5. data/lib/polytypo/data/fixtures/de-CH.json +9 -1
  6. data/lib/polytypo/data/fixtures/de-DE.json +227 -6
  7. data/lib/polytypo/data/fixtures/el.json +9 -1
  8. data/lib/polytypo/data/fixtures/en-GB.json +28 -1
  9. data/lib/polytypo/data/fixtures/en-US.json +690 -1
  10. data/lib/polytypo/data/fixtures/es.json +193 -0
  11. data/lib/polytypo/data/fixtures/fi.json +9 -1
  12. data/lib/polytypo/data/fixtures/fr-CA.json +50 -1
  13. data/lib/polytypo/data/fixtures/fr.json +282 -1
  14. data/lib/polytypo/data/fixtures/it.json +161 -0
  15. data/lib/polytypo/data/fixtures/locale-resolution.json +76 -4
  16. data/lib/polytypo/data/fixtures/nl.json +121 -0
  17. data/lib/polytypo/data/fixtures/pl.json +137 -0
  18. data/lib/polytypo/data/fixtures/pt-BR.json +156 -0
  19. data/lib/polytypo/data/fixtures/pt-PT.json +156 -0
  20. data/lib/polytypo/data/fixtures/ru.json +47 -1
  21. data/lib/polytypo/data/fixtures/sv.json +9 -1
  22. data/lib/polytypo/data/fixtures/uk.json +153 -0
  23. data/lib/polytypo/data/locales/cs.json +90 -0
  24. data/lib/polytypo/data/locales/de-DE.json +7 -2
  25. data/lib/polytypo/data/locales/en-US.json +3 -3
  26. data/lib/polytypo/data/locales/es.json +111 -0
  27. data/lib/polytypo/data/locales/fr-CA.json +7 -1
  28. data/lib/polytypo/data/locales/fr.json +7 -1
  29. data/lib/polytypo/data/locales/it.json +95 -0
  30. data/lib/polytypo/data/locales/nl.json +84 -0
  31. data/lib/polytypo/data/locales/pl.json +96 -0
  32. data/lib/polytypo/data/locales/pt-BR.json +82 -0
  33. data/lib/polytypo/data/locales/pt-PT.json +84 -0
  34. data/lib/polytypo/data/locales/registry.json +23 -3
  35. data/lib/polytypo/data/locales/ru.json +2 -2
  36. data/lib/polytypo/data/locales/uk.json +130 -0
  37. data/lib/polytypo/data/rules/analyze.md +157 -0
  38. data/lib/polytypo/data/rules/apostrophe.md +432 -0
  39. data/lib/polytypo/data/rules/dashes.md +128 -37
  40. data/lib/polytypo/data/rules/ellipsis.md +271 -0
  41. data/lib/polytypo/data/rules/hyphen.md +353 -0
  42. data/lib/polytypo/data/rules/locale-resolution.md +239 -0
  43. data/lib/polytypo/data/rules/modes.md +1281 -0
  44. data/lib/polytypo/data/rules/nbsp.md +1157 -0
  45. data/lib/polytypo/data/rules/order.json +11 -11
  46. data/lib/polytypo/data/rules/pipeline-idempotency.md +605 -0
  47. data/lib/polytypo/data/rules/quotes.md +1324 -0
  48. data/lib/polytypo/data/rules/ranges.md +489 -0
  49. data/lib/polytypo/data/rules/spaces.md +649 -0
  50. data/lib/polytypo/data/rules/symbols.md +540 -0
  51. data/lib/polytypo/data/schema/fixtures.schema.json +18 -3
  52. data/lib/polytypo/engine/origin.rb +75 -0
  53. data/lib/polytypo/engine/pipeline.rb +72 -1
  54. data/lib/polytypo/engine/rules/apostrophe.rb +10 -1
  55. data/lib/polytypo/engine/rules/dash_shared.rb +85 -3
  56. data/lib/polytypo/engine/rules/dashes.rb +4 -1
  57. data/lib/polytypo/engine/rules/nbsp.rb +53 -20
  58. data/lib/polytypo/engine/rules/ranges.rb +24 -20
  59. data/lib/polytypo/engine/rules/spaces.rb +8 -1
  60. data/lib/polytypo/errors.rb +3 -0
  61. data/lib/polytypo/modes/runner.rb +17 -0
  62. data/lib/polytypo/modes/spans.rb +30 -2
  63. data/lib/polytypo/modes/yaml.rb +312 -0
  64. data/lib/polytypo/version.rb +1 -1
  65. data/lib/polytypo.rb +126 -15
  66. metadata +31 -1
@@ -1,5 +1,6 @@
1
1
  # frozen_string_literal: true
2
2
 
3
+ require_relative "../engine/origin"
3
4
  require_relative "../engine/sentinels"
4
5
  require_relative "../errors"
5
6
 
@@ -11,6 +12,9 @@ module Polytypo
11
12
  module Spans
12
13
  LINE_TERMINATORS = [0x0A, 0x0D, 0x0B, 0x0C, 0x85, 0x2028, 0x2029].freeze
13
14
 
15
+ # U+0020, the one emitted code point whose meaning is positional (modes.md 3.4, 5 item 2).
16
+ SPACE = 0x20
17
+
14
18
  # A processable span, identified by its offsets in the original source, addressed as
15
19
  # code-point indices (ARCHITECTURE.md section 4.2) -- Ruby String indices are already
16
20
  # code-point indices for a UTF-8 String, no separate byte/char distinction to manage here.
@@ -68,6 +72,19 @@ module Polytypo
68
72
  cp
69
73
  end
70
74
 
75
+ # The origin map for concatenate_spans (analyze.md section 2): for every code point of the
76
+ # joined array, the code-point offset of the character it came from IN THE DOCUMENT, and
77
+ # NO_ORIGIN for the markers, which came from nowhere. A Span's bounds are already
78
+ # code-point offsets, so no coordinate conversion belongs here.
79
+ def self.origin_of_spans(spans)
80
+ origin = []
81
+ spans.each_with_index do |span, i|
82
+ origin << Engine::Origin::NO_ORIGIN unless i.zero?
83
+ origin.concat((span.start...span.end).to_a)
84
+ end
85
+ origin
86
+ end
87
+
71
88
  # The span extents of the array as it stands. Recomputed after every rule, because applying
72
89
  # edits shifts every index after the first one -- the markers themselves always survive,
73
90
  # since no edit may contain one.
@@ -94,7 +111,13 @@ module Polytypo
94
111
  # 1. No edit may contain a marker -- one that does is a bug, discarded rather than
95
112
  # redistributed.
96
113
  # 2. The edge-growth rule: an edit is discarded if it would place code points at an
97
- # extremity of its span that were not there before.
114
+ # extremity of its span that were not there before. That sentence is the rule;
115
+ # "r > d" alone is an incorrect formalisation of it and misses r == d. dashes P3
116
+ # admits a run of THREE dashes, so "---" -> U+0020 en-dash U+0020 is 3 -> 3: the
117
+ # length test sees nothing while U+0020 lands on both extremities anyway. The second
118
+ # clause tests the CHARACTER, and only U+0020 needs testing -- it is the one code
119
+ # point any rule emits whose meaning comes from its position rather than from itself
120
+ # (modes.md 5 item 2).
98
121
  def self.filter_boundary_edits(cp, edits, ranges)
99
122
  edits.select do |edit|
100
123
  next false if (edit.start...edit.end).any? { |i| Engine.marker?(cp[i]) }
@@ -104,7 +127,12 @@ module Polytypo
104
127
  d = edit.end - edit.start
105
128
  r = edit.replacement.length
106
129
  span = span_containing(ranges, p)
107
- !(span && r > d && (p == span.first || q == span.last))
130
+ next true unless span && (p == span.first || q == span.last)
131
+ next false if r > d
132
+ next false if r.positive? && p == span.first && edit.replacement[0] == SPACE && cp[p] != SPACE
133
+ next false if r.positive? && q == span.last && edit.replacement[r - 1] == SPACE && cp[q] != SPACE
134
+
135
+ true
108
136
  end
109
137
  end
110
138
 
@@ -0,0 +1,312 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Polytypo
4
+ module Modes
5
+ # spec/rules/modes.md 3.8. "yaml" differs from the other two document modes twice over.
6
+ #
7
+ # It uses NO PARSER (3.8.1): two of the five ecosystems' YAML libraries cannot report the
8
+ # source offsets the round-trip guarantee needs, so the scan below is specified rather than
9
+ # delegated and is written the same way in every runtime.
10
+ #
11
+ # And the caller names the keys (3.8.2). YAML is a data format with islands of prose in it --
12
+ # the inverse of HTML and Markdown -- and nothing in its syntax separates "description:" from
13
+ # "run:". A keyless draft of this file rewrote "if !" as "if!" inside a workflow's shell
14
+ # script; there is no content test that would not, because shell and template expressions are
15
+ # written in words.
16
+ #
17
+ # It also SKIPS BY DEFAULT -- the inverse of 3.6's closed skip list. A construct this scan
18
+ # does not recognise with certainty yields no spans, so the worst outcome of a gap in it is
19
+ # prose left untypeset, never a changed byte.
20
+ module Yaml
21
+ SPACE = " "
22
+ TAB = "\t"
23
+ LF = "\n"
24
+ CR = "\r"
25
+ DECLINING_KEY_CHARS = ['"', "'", "{", "[", "&", "*", "!", "#"].freeze
26
+ NON_SCALAR_VALUE_CHARS = ["#", "&", "*", "!", "{", "["].freeze
27
+
28
+ Line = Struct.new(:start, :end)
29
+
30
+ # Offsets are code-point indices, like every Span in this runtime (ARCHITECTURE.md 4.2).
31
+ def self.yaml_spans(source, keys)
32
+ chars = source.chars
33
+ lines = split_lines(chars)
34
+ spans = []
35
+ li = 0
36
+ li = scan_line(chars, lines, li, keys, spans) while li < lines.length
37
+ spans
38
+ end
39
+
40
+ # 3.8.4: a line ends at U+000A, and A U+000D IMMEDIATELY BEFORE IT IS NOT PART OF THE LINE
41
+ # -- it is a terminator like the U+000A itself, so it lies outside every span and comes back
42
+ # untouched. Without that clause a CRLF file behaves differently from the same bytes with
43
+ # LF: the block header reads as "|\r" and is unrecognised, and a plain scalar carries the
44
+ # carriage return inside its span. Five runtimes split lines with five different standard
45
+ # library calls, so the treatment has to be stated rather than inherited.
46
+ def self.split_lines(chars)
47
+ lines = []
48
+ start = 0
49
+ end_of = ->(i) { i > start && chars[i - 1] == CR ? i - 1 : i }
50
+ chars.each_with_index do |ch, i|
51
+ next unless ch == LF
52
+
53
+ lines << Line.new(start, end_of.call(i))
54
+ start = i + 1
55
+ end
56
+ lines << Line.new(start, end_of.call(chars.length)) if start < chars.length
57
+ lines
58
+ end
59
+ private_class_method :split_lines
60
+
61
+ def self.first_non_space(chars, line)
62
+ i = line.start
63
+ i += 1 while i < line.end && chars[i] == SPACE
64
+ i
65
+ end
66
+ private_class_method :first_non_space
67
+
68
+ def self.blank?(chars, line)
69
+ first_non_space(chars, line) == line.end
70
+ end
71
+ private_class_method :blank?
72
+
73
+ def self.tab?(chars, line)
74
+ (line.start...line.end).any? { |i| chars[i] == TAB }
75
+ end
76
+ private_class_method :tab?
77
+
78
+ # 3.8.4 step 3. "---" and "..." at the head of a line, bare or introducing a node: the
79
+ # trailing-content form is declined too, so "--- key: value" never yields a key of
80
+ # "--- key".
81
+ def self.document_marker?(chars, from, to)
82
+ return false if to - from < 3
83
+
84
+ c = chars[from]
85
+ return false unless ["-", "."].include?(c)
86
+ return false unless chars[from + 1] == c && chars[from + 2] == c
87
+
88
+ from + 3 == to || chars[from + 3] == SPACE
89
+ end
90
+ private_class_method :document_marker?
91
+
92
+ # A colon that ends the line or is followed by U+0020 -- the only colon YAML reads as an
93
+ # indicator.
94
+ def self.indicator_colon?(chars, j, to)
95
+ return false unless chars[j] == ":"
96
+
97
+ j + 1 == to || chars[j + 1] == SPACE
98
+ end
99
+ private_class_method :indicator_colon?
100
+
101
+ def self.sequence_dash?(chars, j, to)
102
+ return false unless chars[j] == "-"
103
+
104
+ j + 1 == to || chars[j + 1] == SPACE
105
+ end
106
+ private_class_method :sequence_dash?
107
+
108
+ # The value run of 3.8.4 step 7: every following line that is blank or indented more than
109
+ # the key line. THOSE LINES ARE NEVER SCANNED AGAIN -- without that, a multi-line quoted
110
+ # scalar, a multi-line flow collection and a folded plain scalar all leak their continuation
111
+ # lines back into the scan as if they were mappings, and a span can end up holding a
112
+ # scalar's own closing delimiter.
113
+ def self.value_run_end(chars, lines, li, indent)
114
+ k = li + 1
115
+ while k < lines.length
116
+ nxt = lines[k]
117
+ break if !blank?(chars, nxt) && first_non_space(chars, nxt) - nxt.start <= indent
118
+
119
+ k += 1
120
+ end
121
+ k
122
+ end
123
+ private_class_method :value_run_end
124
+
125
+ # One step of 3.8.4. Returns the index of the next line to scan.
126
+ def self.scan_line(chars, lines, li, keys, spans)
127
+ line = lines[li]
128
+ skip_line = li + 1
129
+
130
+ # step 1 -- blank, or a tab anywhere, which makes indentation undecidable.
131
+ start = first_non_space(chars, line)
132
+ return skip_line if start == line.end || tab?(chars, line)
133
+
134
+ indent = start - line.start
135
+
136
+ # steps 2 and 3 -- comment, directive, document marker.
137
+ return skip_line if ["#", "%"].include?(chars[start])
138
+ return skip_line if document_marker?(chars, start, line.end)
139
+
140
+ # step 4 -- block sequence entries are consumed, not skipped; "- - key: v" nests.
141
+ i = start
142
+ while i < line.end && sequence_dash?(chars, i, line.end)
143
+ i += 2
144
+ i += 1 while i < line.end && chars[i] == SPACE
145
+ end
146
+ return skip_line if i >= line.end
147
+
148
+ # step 5 -- find the key. A colon NOT followed by U+0020 or the line end is an ordinary
149
+ # key character, so "a:b: v" has the key "a:b"; stating that is what keeps five scanners
150
+ # agreeing.
151
+ key_start = i
152
+ colon = -1
153
+ (i...line.end).each do |j|
154
+ return skip_line if DECLINING_KEY_CHARS.include?(chars[j])
155
+
156
+ if indicator_colon?(chars, j, line.end)
157
+ colon = j
158
+ break
159
+ end
160
+ end
161
+ return skip_line if colon.negative?
162
+
163
+ key_end = colon
164
+ key_end -= 1 while key_end > key_start && chars[key_end - 1] == SPACE
165
+ return skip_line if key_end <= key_start
166
+
167
+ # step 6 -- an empty value means a nested node, whose lines ARE scanned on their own.
168
+ v = colon + 1
169
+ v += 1 while v < line.end && chars[v] == SPACE
170
+ return skip_line if v >= line.end
171
+
172
+ # step 7 -- the line carries an inline value, so its continuation lines belong to it.
173
+ nxt = value_run_end(chars, lines, li, indent)
174
+
175
+ # step 8 -- the key must be listed. Checked before the value's form, so an unlisted key
176
+ # costs nothing to decline: this is what makes "run:", "if:" and "image:" unreachable.
177
+ return nxt unless keys.include?(chars[key_start...key_end].join)
178
+
179
+ # step 9 -- the scalar form.
180
+ value = chars[v]
181
+ return nxt if NON_SCALAR_VALUE_CHARS.include?(value)
182
+
183
+ if ["|", ">"].include?(value)
184
+ block_scalar(chars, line, lines[(li + 1)...nxt], indent, v, spans)
185
+ elsif ['"', "'"].include?(value)
186
+ quoted_scalar(chars, line, v, spans) if nxt == li + 1
187
+ elsif nxt == li + 1
188
+ plain_scalar(chars, line, v, spans)
189
+ end
190
+ nxt
191
+ end
192
+ private_class_method :scan_line
193
+
194
+ # 3.8.5. One span per non-blank content line, starting after the block's own indentation.
195
+ # The header, the indentation and every line terminator lie outside every span -- including
196
+ # the run of line terminators at the end that the chomping indicator governs, which is why
197
+ # "|", "|-", "|+", ">", ">-" and ">+" are handled identically here.
198
+ def self.block_scalar(chars, line, run_lines, indent, v, spans)
199
+ # The header: at most one chomping indicator and at most one indentation indicator, in
200
+ # either order, then optional spaces and an optional comment. Anything else is
201
+ # unrecognised.
202
+ h = v + 1
203
+ explicit_indent = 0
204
+ chomping = false
205
+ while h < line.end
206
+ c = chars[h]
207
+ if ["-", "+"].include?(c) && !chomping
208
+ chomping = true
209
+ h += 1
210
+ elsif c >= "1" && c <= "9" && explicit_indent.zero?
211
+ explicit_indent = c.ord - "0".ord
212
+ h += 1
213
+ else
214
+ break
215
+ end
216
+ end
217
+ h += 1 while h < line.end && chars[h] == SPACE
218
+ return if h < line.end && chars[h] != "#"
219
+
220
+ # One definition of the run, and three conditions that make the whole block yield no
221
+ # spans.
222
+ content = []
223
+ content_indent = -1
224
+ run_lines.each do |nxt|
225
+ next if blank?(chars, nxt) # blank lines belong to the block and yield no span
226
+
227
+ # A tab makes indentation undecidable; an explicit indicator that disagrees with the
228
+ # block as written, or a later line dedented inside it, is ambiguous rather than
229
+ # guessable. All three make the WHOLE block yield no spans -- bail rather than choose.
230
+ break content_indent = -1 if tab?(chars, nxt)
231
+
232
+ next_indent = first_non_space(chars, nxt) - nxt.start
233
+ if content_indent.negative?
234
+ content_indent = explicit_indent.positive? ? indent + explicit_indent : next_indent
235
+ end
236
+ break content_indent = -1 if next_indent < content_indent
237
+
238
+ content << nxt
239
+ end
240
+ return if content_indent <= 0
241
+
242
+ content.each do |c_line|
243
+ from = c_line.start + content_indent
244
+ spans << Spans::Span.new(from, c_line.end) if c_line.end > from
245
+ end
246
+ end
247
+ private_class_method :block_scalar
248
+
249
+ # 3.8.6. The span is the content between the quotes. Both bails exist so that source
250
+ # characters and content characters are the same thing, which the offset model of 3.1
251
+ # requires -- the same constraint that makes an HTML character reference an opaque unit in
252
+ # 3.6. No colon test applies here: quoting neutralises the colon, and applying the
253
+ # plain-scalar test would decline `title: "Chapter 1: the beginning"`.
254
+ def self.quoted_scalar(chars, line, v, spans)
255
+ quote = chars[v]
256
+ close = -1
257
+ j = v + 1
258
+ while j < line.end
259
+ c = chars[j]
260
+ return if quote == '"' && c == "\\"
261
+ return if quote == "'" && c == "'" && j + 1 < line.end && chars[j + 1] == "'"
262
+
263
+ if c == quote
264
+ close = j
265
+ break
266
+ end
267
+ j += 1
268
+ end
269
+ return if close.negative?
270
+
271
+ after = close + 1
272
+ after += 1 while after < line.end && chars[after] == SPACE
273
+ return if after < line.end && chars[after] != "#"
274
+
275
+ spans << Spans::Span.new(v + 1, close) if close > v + 1
276
+ end
277
+ private_class_method :quoted_scalar
278
+
279
+ # 3.8.6. In a plain scalar ":" and "#" are still live: U+0020 beside either of them is what
280
+ # turns a scalar into a mapping indicator or a comment, and `dashes` emits U+0020 in every
281
+ # "-spaced" locale. Lifting both out as opaque units puts the dash token at a span
282
+ # extremity, where the edge-growth rule of 3.4 discards the replacement that emits one.
283
+ def self.plain_scalar(chars, line, v, spans)
284
+ # The scalar ends before a trailing comment, so a colon inside that comment is not the
285
+ # scalar's and must not decline it.
286
+ to = line.end
287
+ (v...line.end).each do |j|
288
+ if chars[j] == "#" && j > v && chars[j - 1] == SPACE
289
+ to = j - 1
290
+ break
291
+ end
292
+ end
293
+ to -= 1 while to > v && chars[to - 1] == SPACE
294
+ return if to <= v
295
+
296
+ # Compact nesting is not a value: "key: - item" opens a sequence, and "key: a .:" is a
297
+ # mapping whose key is "a ." -- a plain scalar can never contain a colon in that position.
298
+ return if sequence_dash?(chars, v, line.end)
299
+ return if (v...to).any? { |j| indicator_colon?(chars, j, line.end) }
300
+
301
+ segment = v
302
+ (v..to).each do |j|
303
+ next unless j == to || [":", "#"].include?(chars[j])
304
+
305
+ spans << Spans::Span.new(segment, j) if j > segment
306
+ segment = j + 1
307
+ end
308
+ end
309
+ private_class_method :plain_scalar
310
+ end
311
+ end
312
+ end
@@ -1,5 +1,5 @@
1
1
  # frozen_string_literal: true
2
2
 
3
3
  module Polytypo
4
- VERSION = "1.1.0"
4
+ VERSION = "1.3.0"
5
5
  end
data/lib/polytypo.rb CHANGED
@@ -3,11 +3,13 @@
3
3
  require_relative "polytypo/version"
4
4
  require_relative "polytypo/errors"
5
5
  require_relative "polytypo/engine/codepoints"
6
+ require_relative "polytypo/engine/origin"
6
7
  require_relative "polytypo/engine/pipeline"
7
8
  require_relative "polytypo/engine/rules" # side effect: registers all 9 rules
8
9
  require_relative "polytypo/modes/spans"
9
10
  require_relative "polytypo/modes/runner"
10
11
  require_relative "polytypo/modes/html"
12
+ require_relative "polytypo/modes/yaml"
11
13
 
12
14
  # polytypo normalizes typography across languages: locale-correct quotes, dashes, ellipses,
13
15
  # apostrophes, symbols and no-break spaces, from a spec shared across every polytypo runtime
@@ -17,9 +19,12 @@ module Polytypo
17
19
  # Applies polytypo's rule pipeline to +input+ and returns the result.
18
20
  #
19
21
  # +locale+ is required, with no default -- an unknown locale raises Polytypo::Error with
20
- # CODE_UNKNOWN_LOCALE; there is never a fallback to English. +mode+ is "text" (default), "html"
21
- # or "markdown". +dialect+ is required iff mode is "markdown" ("commonmark"; "mdx" is not
22
- # implemented by this runtime). +rules+ is an opt-out Hash keyed by rule id.
22
+ # CODE_UNKNOWN_LOCALE; there is never a fallback to English. +mode+ is "text" (default),
23
+ # "html", "markdown" or "yaml". +dialect+ is required iff mode is "markdown" ("commonmark";
24
+ # "mdx" is not implemented by this runtime). +keys+ is required iff mode is "yaml" and names
25
+ # the mapping keys whose scalar values are prose (modes.md 3.8.2) -- it has no default, because
26
+ # nothing in YAML's syntax separates "description:" from "run:". +rules+ is an opt-out Hash
27
+ # keyed by rule id.
23
28
  #
24
29
  # Pure: no I/O, no environment, no clock, no globals, no class-level mutable state --
25
30
  # thread-safe and reentrant, callable concurrently from any number of Threads with no external
@@ -27,22 +32,65 @@ module Polytypo
27
32
  #
28
33
  # Only `mode: "text"`/`"html"` ever touch this file's own requires; `mode: "markdown"` lazily
29
34
  # requires "commonmarker" from within Modes::Markdown, never at load time of this file.
30
- def self.transform(input, locale:, mode: "text", dialect: nil, rules: nil)
35
+ def self.transform(input, locale:, mode: "text", dialect: nil, keys: nil, rules: nil, narrow_nbsp: nil)
31
36
  resolved_mode = resolve_mode(mode)
37
+ narrow_target = Engine.resolve_narrow_target(narrow_nbsp)
32
38
 
33
39
  case resolved_mode
34
40
  when "text"
35
41
  if dialect
36
42
  raise Error.new(CODE_INVALID_DIALECT, '"dialect" is only valid when mode is "markdown"')
37
43
  end
38
- transform_text(input, locale, rules)
44
+ transform_text(input, locale, rules, narrow_target)
39
45
  when "html"
40
46
  if dialect
41
47
  raise Error.new(CODE_INVALID_DIALECT, '"dialect" is only valid when mode is "markdown"')
42
48
  end
43
- transform_html(input, locale, rules)
49
+ transform_html(input, locale, rules, narrow_target)
50
+ when "yaml"
51
+ if dialect
52
+ raise Error.new(CODE_INVALID_DIALECT, '"dialect" is only valid when mode is "markdown"')
53
+ end
54
+ transform_yaml(input, locale, keys, rules, narrow_target)
55
+ else # "markdown"
56
+ transform_markdown(input, locale, dialect, rules, narrow_target)
57
+ end
58
+ end
59
+
60
+ # Runs the same pipeline as .transform and reports what it would do instead of doing it
61
+ # (spec/rules/analyze.md), returning an Array of Polytypo::Engine::Change. Offsets are
62
+ # code-point offsets into +input+ in every mode -- into the document, in "html" and "markdown"
63
+ # mode, not into a span.
64
+ #
65
+ # What it guarantees: the list is empty exactly when .transform would return the input
66
+ # unchanged, every +rule_id+ was enabled for the call, and every offset is inside the input.
67
+ # What it does not: the list is a report, not a patch -- two rules may touch the same original
68
+ # range, so replaying it is not guaranteed to reproduce .transform's output. Call .transform
69
+ # for the text (analyze.md sections 4 and 5).
70
+ #
71
+ # Pure and thread-safe on the same terms as .transform.
72
+ def self.analyze(input, locale:, mode: "text", dialect: nil, keys: nil, rules: nil, narrow_nbsp: nil)
73
+ resolved_mode = resolve_mode(mode)
74
+ narrow_target = Engine.resolve_narrow_target(narrow_nbsp)
75
+
76
+ case resolved_mode
77
+ when "text"
78
+ if dialect
79
+ raise Error.new(CODE_INVALID_DIALECT, '"dialect" is only valid when mode is "markdown"')
80
+ end
81
+ analyze_text(input, locale, rules, narrow_target)
82
+ when "html"
83
+ if dialect
84
+ raise Error.new(CODE_INVALID_DIALECT, '"dialect" is only valid when mode is "markdown"')
85
+ end
86
+ analyze_html(input, locale, rules, narrow_target)
87
+ when "yaml"
88
+ if dialect
89
+ raise Error.new(CODE_INVALID_DIALECT, '"dialect" is only valid when mode is "markdown"')
90
+ end
91
+ analyze_yaml(input, locale, keys, rules, narrow_target)
44
92
  else # "markdown"
45
- transform_markdown(input, locale, dialect, rules)
93
+ analyze_markdown(input, locale, dialect, rules, narrow_target)
46
94
  end
47
95
  end
48
96
 
@@ -50,42 +98,105 @@ module Polytypo
50
98
  case mode
51
99
  when nil, "text"
52
100
  "text"
53
- when "html", "markdown"
101
+ when "html", "markdown", "yaml"
54
102
  mode
55
103
  else
56
- raise Error.new(CODE_INVALID_MODE, "unknown mode #{mode.inspect}. Expected \"text\", \"html\" or \"markdown\"")
104
+ raise Error.new(CODE_INVALID_MODE,
105
+ "unknown mode #{mode.inspect}. Expected \"text\", \"html\", \"markdown\" or \"yaml\"")
57
106
  end
58
107
  end
59
108
  private_class_method :resolve_mode
60
109
 
61
- def self.transform_text(input, locale, rules)
110
+ def self.transform_text(input, locale, rules, narrow_target)
62
111
  _resolved, locale_data, plan = Engine::Pipeline.prepare(locale, rules)
63
112
  cp = Engine::Codepoints.to_codepoints(input)
64
- ctx = Engine::RuleContext.new(mode: "text", dialect: nil, locale: locale)
113
+ ctx = Engine::RuleContext.new(mode: "text", dialect: nil, locale: locale,
114
+ narrow_target: narrow_target)
65
115
  result = Engine::Pipeline.run_rules(cp, plan, locale_data, ctx)
66
116
  Engine::Codepoints.from_codepoints(result)
67
117
  end
68
118
  private_class_method :transform_text
69
119
 
70
- def self.transform_html(input, locale, rules)
120
+ def self.transform_html(input, locale, rules, narrow_target)
71
121
  resolved_locale, locale_data, plan = Engine::Pipeline.prepare(locale, rules)
72
122
  spans = Modes::Html.html_spans(input)
73
- ctx = Engine::RuleContext.new(mode: "html", dialect: nil, locale: resolved_locale)
123
+ ctx = Engine::RuleContext.new(mode: "html", dialect: nil, locale: resolved_locale,
124
+ narrow_target: narrow_target)
74
125
  cp = Engine::Codepoints.to_codepoints(input)
75
126
  Modes::Runner.run_over_spans(cp, spans, plan, locale_data, ctx)
76
127
  end
77
128
  private_class_method :transform_html
78
129
 
79
- def self.transform_markdown(input, locale, dialect, rules)
130
+ # "yaml" mode: the only pipeline here with no parser dependency at all -- span selection is the
131
+ # specified scan of modes.md 3.8, not a library. There is likewise no CODE_MALFORMED_INPUT
132
+ # counterpart: with no declared grammar to violate, a file that is not YAML yields few spans or
133
+ # none and comes back byte for byte (modes.md 3.8.3). The only raise this mode adds is +keys+,
134
+ # which is about the call and not the input.
135
+ def self.transform_yaml(input, locale, keys, rules, narrow_target)
136
+ resolved_locale, locale_data, plan = Engine::Pipeline.prepare(locale, rules)
137
+ resolved_keys = Engine.resolve_yaml_keys(keys)
138
+ spans = Modes::Yaml.yaml_spans(input, resolved_keys)
139
+ ctx = Engine::RuleContext.new(mode: "yaml", dialect: nil, locale: resolved_locale,
140
+ narrow_target: narrow_target)
141
+ cp = Engine::Codepoints.to_codepoints(input)
142
+ Modes::Runner.run_over_spans(cp, spans, plan, locale_data, ctx)
143
+ end
144
+ private_class_method :transform_yaml
145
+
146
+ def self.transform_markdown(input, locale, dialect, rules, narrow_target)
80
147
  # Validation order is public, tested behaviour, identical across every runtime: rules (an
81
148
  # unknown rule id), then locale (an unknown locale), then dialect/parsing.
82
149
  resolved_locale, locale_data, plan = Engine::Pipeline.prepare(locale, rules)
83
150
  require_relative "polytypo/modes/markdown"
84
151
  Modes::Markdown.resolve_dialect(dialect)
85
152
  spans = Modes::Markdown.markdown_spans(input)
86
- ctx = Engine::RuleContext.new(mode: "markdown", dialect: dialect, locale: resolved_locale)
153
+ ctx = Engine::RuleContext.new(mode: "markdown", dialect: dialect, locale: resolved_locale,
154
+ narrow_target: narrow_target)
87
155
  cp = Engine::Codepoints.to_codepoints(input)
88
156
  Modes::Runner.run_over_spans(cp, spans, plan, locale_data, ctx)
89
157
  end
90
158
  private_class_method :transform_markdown
159
+
160
+ def self.analyze_text(input, locale, rules, narrow_target)
161
+ _resolved, locale_data, plan = Engine::Pipeline.prepare(locale, rules)
162
+ cp = Engine::Codepoints.to_codepoints(input)
163
+ ctx = Engine::RuleContext.new(mode: "text", dialect: nil, locale: locale,
164
+ narrow_target: narrow_target)
165
+ Engine::Pipeline.run_rules_recording(cp, plan, locale_data, ctx, (0...cp.length).to_a, cp.length)
166
+ end
167
+ private_class_method :analyze_text
168
+
169
+ def self.analyze_html(input, locale, rules, narrow_target)
170
+ resolved_locale, locale_data, plan = Engine::Pipeline.prepare(locale, rules)
171
+ spans = Modes::Html.html_spans(input)
172
+ ctx = Engine::RuleContext.new(mode: "html", dialect: nil, locale: resolved_locale,
173
+ narrow_target: narrow_target)
174
+ Modes::Runner.analyze_over_spans(Engine::Codepoints.to_codepoints(input), spans, plan, locale_data, ctx)
175
+ end
176
+ private_class_method :analyze_html
177
+
178
+ # analyze.md section 1, "yaml" mode: offsets are into the document, not into a span
179
+ # (analyze.md section 6).
180
+ def self.analyze_yaml(input, locale, keys, rules, narrow_target)
181
+ resolved_locale, locale_data, plan = Engine::Pipeline.prepare(locale, rules)
182
+ resolved_keys = Engine.resolve_yaml_keys(keys)
183
+ spans = Modes::Yaml.yaml_spans(input, resolved_keys)
184
+ ctx = Engine::RuleContext.new(mode: "yaml", dialect: nil, locale: resolved_locale,
185
+ narrow_target: narrow_target)
186
+ Modes::Runner.analyze_over_spans(Engine::Codepoints.to_codepoints(input), spans, plan, locale_data, ctx)
187
+ end
188
+ private_class_method :analyze_yaml
189
+
190
+ def self.analyze_markdown(input, locale, dialect, rules, narrow_target)
191
+ # Validation order is public, tested behaviour and is shared with .transform: rules, then
192
+ # locale, then dialect/parsing (analyze.md section 4, A1).
193
+ resolved_locale, locale_data, plan = Engine::Pipeline.prepare(locale, rules)
194
+ require_relative "polytypo/modes/markdown"
195
+ Modes::Markdown.resolve_dialect(dialect)
196
+ spans = Modes::Markdown.markdown_spans(input)
197
+ ctx = Engine::RuleContext.new(mode: "markdown", dialect: dialect, locale: resolved_locale,
198
+ narrow_target: narrow_target)
199
+ Modes::Runner.analyze_over_spans(Engine::Codepoints.to_codepoints(input), spans, plan, locale_data, ctx)
200
+ end
201
+ private_class_method :analyze_markdown
91
202
  end