polytypo 1.2.0 → 1.3.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (64) hide show
  1. checksums.yaml +4 -4
  2. data/README.md +33 -1
  3. data/lib/polytypo/data/VERSION +1 -1
  4. data/lib/polytypo/data/fixtures/cs.json +161 -0
  5. data/lib/polytypo/data/fixtures/de-CH.json +1 -1
  6. data/lib/polytypo/data/fixtures/de-DE.json +195 -6
  7. data/lib/polytypo/data/fixtures/el.json +1 -1
  8. data/lib/polytypo/data/fixtures/en-GB.json +12 -1
  9. data/lib/polytypo/data/fixtures/en-US.json +648 -1
  10. data/lib/polytypo/data/fixtures/es.json +193 -0
  11. data/lib/polytypo/data/fixtures/fi.json +1 -1
  12. data/lib/polytypo/data/fixtures/fr-CA.json +25 -1
  13. data/lib/polytypo/data/fixtures/fr.json +176 -1
  14. data/lib/polytypo/data/fixtures/it.json +161 -0
  15. data/lib/polytypo/data/fixtures/locale-resolution.json +76 -4
  16. data/lib/polytypo/data/fixtures/nl.json +121 -0
  17. data/lib/polytypo/data/fixtures/pl.json +137 -0
  18. data/lib/polytypo/data/fixtures/pt-BR.json +156 -0
  19. data/lib/polytypo/data/fixtures/pt-PT.json +156 -0
  20. data/lib/polytypo/data/fixtures/ru.json +23 -1
  21. data/lib/polytypo/data/fixtures/sv.json +1 -1
  22. data/lib/polytypo/data/fixtures/uk.json +153 -0
  23. data/lib/polytypo/data/locales/cs.json +90 -0
  24. data/lib/polytypo/data/locales/de-DE.json +7 -2
  25. data/lib/polytypo/data/locales/en-US.json +3 -3
  26. data/lib/polytypo/data/locales/es.json +111 -0
  27. data/lib/polytypo/data/locales/fr-CA.json +7 -1
  28. data/lib/polytypo/data/locales/fr.json +7 -1
  29. data/lib/polytypo/data/locales/it.json +95 -0
  30. data/lib/polytypo/data/locales/nl.json +84 -0
  31. data/lib/polytypo/data/locales/pl.json +96 -0
  32. data/lib/polytypo/data/locales/pt-BR.json +82 -0
  33. data/lib/polytypo/data/locales/pt-PT.json +84 -0
  34. data/lib/polytypo/data/locales/registry.json +23 -3
  35. data/lib/polytypo/data/locales/ru.json +2 -2
  36. data/lib/polytypo/data/locales/uk.json +130 -0
  37. data/lib/polytypo/data/rules/analyze.md +157 -0
  38. data/lib/polytypo/data/rules/apostrophe.md +432 -0
  39. data/lib/polytypo/data/rules/dashes.md +128 -37
  40. data/lib/polytypo/data/rules/ellipsis.md +271 -0
  41. data/lib/polytypo/data/rules/hyphen.md +353 -0
  42. data/lib/polytypo/data/rules/locale-resolution.md +239 -0
  43. data/lib/polytypo/data/rules/modes.md +1281 -0
  44. data/lib/polytypo/data/rules/nbsp.md +1157 -0
  45. data/lib/polytypo/data/rules/order.json +11 -11
  46. data/lib/polytypo/data/rules/pipeline-idempotency.md +605 -0
  47. data/lib/polytypo/data/rules/quotes.md +1324 -0
  48. data/lib/polytypo/data/rules/ranges.md +489 -0
  49. data/lib/polytypo/data/rules/spaces.md +649 -0
  50. data/lib/polytypo/data/rules/symbols.md +540 -0
  51. data/lib/polytypo/data/schema/fixtures.schema.json +18 -3
  52. data/lib/polytypo/engine/origin.rb +75 -0
  53. data/lib/polytypo/engine/pipeline.rb +72 -1
  54. data/lib/polytypo/engine/rules/dash_shared.rb +85 -3
  55. data/lib/polytypo/engine/rules/dashes.rb +4 -1
  56. data/lib/polytypo/engine/rules/nbsp.rb +43 -7
  57. data/lib/polytypo/engine/rules/ranges.rb +24 -20
  58. data/lib/polytypo/errors.rb +3 -0
  59. data/lib/polytypo/modes/runner.rb +17 -0
  60. data/lib/polytypo/modes/spans.rb +30 -2
  61. data/lib/polytypo/modes/yaml.rb +312 -0
  62. data/lib/polytypo/version.rb +1 -1
  63. data/lib/polytypo.rb +126 -15
  64. metadata +31 -1
@@ -0,0 +1,75 @@
1
+ # frozen_string_literal: true
2
+
3
+ require_relative "codepoints"
4
+
5
+ module Polytypo
6
+ # One entry of Polytypo.analyze's result, in input coordinates (analyze.md section 2):
7
+ # +rule_id+ is a rule id from spec/rules/order.json, +start+/+end+ are code-point offsets into
8
+ # the input (inclusive/exclusive), and +before+/+after+ are the text on either side of that one
9
+ # edit. +start+ == +end+ is a pure insertion, and +before+ is then empty.
10
+ #
11
+ # Public API, so it is named in this namespace rather than in Engine, where it is built.
12
+ Change = Data.define(:rule_id, :start, :end, :before, :after)
13
+
14
+ module Engine
15
+ # analyze.md section 2: every reported offset is a code-point offset into the input the
16
+ # caller passed, in every mode. The rules, however, run over an array that is not the input
17
+ # -- in "text" mode edits from earlier rules have already shifted it, and in "html"/"markdown"
18
+ # mode it is the marker-joined concatenation of the processable spans (modes.md 3.5). This
19
+ # module carries the one structure that bridges the two: an origin map, parallel to the
20
+ # current code-point array, holding the input offset each code point came from, or NO_ORIGIN
21
+ # for one the pipeline itself produced. Mirrors polytypo-js's src/engine/origin.ts.
22
+ module Origin
23
+ NO_ORIGIN = -1
24
+
25
+ # The input offset an edit boundary at +index+ addresses. Synthetic code points have no
26
+ # origin of their own, so the scan runs forward to the first that has one -- an insertion
27
+ # between two earlier insertions still lands where the next real character is. Falling off
28
+ # the end means the boundary is at the end of the input.
29
+ def self.origin_at(origin, index, input_length)
30
+ (index...origin.length).each do |i|
31
+ return origin[i] unless origin[i] == NO_ORIGIN
32
+ end
33
+ input_length
34
+ end
35
+
36
+ # The origin map for the array Edits.apply_edits is about to produce. A replacement of
37
+ # equal length keeps its origins position by position, which is what makes a conversion
38
+ # (U+0020 -> U+00A0) still point at the character it converted; anything longer is
39
+ # synthetic beyond the positions it covers.
40
+ def self.apply_edits_to_origin(origin, edits)
41
+ return origin.dup if edits.empty?
42
+
43
+ out = []
44
+ cursor = 0
45
+ edits.each do |edit|
46
+ out.concat(origin[cursor...edit.start])
47
+ edit.replacement.each_index do |k|
48
+ source = edit.start + k
49
+ out << (source < edit.end ? origin[source] : NO_ORIGIN)
50
+ end
51
+ cursor = edit.end
52
+ end
53
+ out.concat(origin[cursor..])
54
+ out
55
+ end
56
+
57
+ # One rule's edits, in the coordinates that rule saw, rendered as Changes in input
58
+ # coordinates. +before+ is the text this rule replaced and +after+ what it replaced it with
59
+ # (analyze.md section 2), so on a text two rules have both touched, +before+ is what the
60
+ # second rule saw rather than what the caller typed -- section 5 says so and shows the
61
+ # French case where it matters.
62
+ def self.record_changes(cp, edits, origin, input_length, rule_id)
63
+ edits.map do |edit|
64
+ Polytypo::Change.new(
65
+ rule_id: rule_id,
66
+ start: origin_at(origin, edit.start, input_length),
67
+ end: origin_at(origin, edit.end, input_length),
68
+ before: Codepoints.from_codepoints(cp[edit.start...edit.end]),
69
+ after: Codepoints.from_codepoints(edit.replacement)
70
+ )
71
+ end
72
+ end
73
+ end
74
+ end
75
+ end
@@ -2,13 +2,62 @@
2
2
 
3
3
  require_relative "edits"
4
4
  require_relative "locale"
5
+ require_relative "origin"
5
6
  require_relative "registry"
6
7
  require_relative "../errors"
7
8
 
8
9
  module Polytypo
9
10
  module Engine
10
11
  # Per-call context every rule reads, alongside the code-point array and the locale data.
11
- RuleContext = Struct.new(:mode, :dialect, :locale, keyword_init: true)
12
+ # :narrow_target is nbsp.md 3.1a's NARROW-TARGET, already resolved to a code point: U+202F
13
+ # by default, U+00A0 when the caller passed narrow_nbsp: "nbsp". A rule reads a code point
14
+ # and never the option, so the string never reaches the pipeline.
15
+ RuleContext = Struct.new(:mode, :dialect, :locale, :narrow_target, keyword_init: true)
16
+
17
+ # nbsp.md 3.1a: resolve the narrow_nbsp option to the code point the rule writes. Checked
18
+ # immediately after mode and before rules (ARCHITECTURE.md 7) -- the two checks that read
19
+ # nothing but the call itself come first -- and it runs whether or not `nbsp` is enabled, so
20
+ # a misspelled value still raises rather than being silently ignored.
21
+ NARROW_NO_BREAK_SPACE = 0x202F
22
+ NO_BREAK_SPACE = 0x00A0
23
+
24
+ def self.resolve_narrow_target(value)
25
+ return NARROW_NO_BREAK_SPACE if value.nil? || value == "narrow"
26
+ return NO_BREAK_SPACE if value == "nbsp"
27
+
28
+ raise Polytypo::Error.new(
29
+ Polytypo::CODE_INVALID_OPTION,
30
+ "Unknown narrow_nbsp #{value.inspect}. Expected \"narrow\" or \"nbsp\".",
31
+ )
32
+ end
33
+
34
+ # modes.md 3.8.2: "yaml" mode's `keys` option, resolved once at the call boundary.
35
+ #
36
+ # It has no default, for the reason `dialect` has none. YAML is a data format with islands of
37
+ # prose in it, and nothing in its syntax marks them -- `description` holds a sentence and
38
+ # `run` holds a shell script, spelled identically -- so a default would be a guess about the
39
+ # schema above the document. An empty list is legal: "process nothing" is a choice a caller
40
+ # may make, not an error. Checked after locale, last of the option checks, alongside dialect.
41
+ def self.resolve_yaml_keys(value)
42
+ unless value.is_a?(Array)
43
+ raise Polytypo::Error.new(
44
+ Polytypo::CODE_INVALID_OPTION,
45
+ '"keys" is required when mode is "yaml" and must be an Array of Strings ' \
46
+ "(modes.md 3.8.2). Received #{value.inspect}.",
47
+ )
48
+ end
49
+
50
+ value.each do |key|
51
+ next if key.is_a?(String)
52
+
53
+ raise Polytypo::Error.new(
54
+ Polytypo::CODE_INVALID_OPTION,
55
+ "\"keys\" must contain only Strings; received #{key.inspect}.",
56
+ )
57
+ end
58
+
59
+ value.to_set
60
+ end
12
61
 
13
62
  # Resolves the locale, builds the rule plan, and runs each enabled rule in
14
63
  # spec/rules/order.json order over a code-point array, applying its edits before the next
@@ -56,6 +105,28 @@ module Polytypo
56
105
  end
57
106
  current
58
107
  end
108
+
109
+ # run_rules, keeping the edits instead of discarding them (analyze.md section 1: same
110
+ # pipeline, same order, reporting rather than applying). The origin map travels alongside
111
+ # the array so every change comes back in input coordinates, and +filter_edits+ is the hook
112
+ # the span-runner needs for modes.md 3.4's boundary filters, passed as a block -- text mode
113
+ # passes none and gets the identity.
114
+ def self.run_rules_recording(cp, plan, locale_data, ctx, origin, input_length)
115
+ current = cp
116
+ current_origin = origin
117
+ changes = []
118
+ plan.each do |rule_id|
119
+ fn = Registry.rule(rule_id)
120
+ produced = fn.call(current, locale_data, ctx)
121
+ edits = block_given? ? yield(current, produced) : produced
122
+ next if edits.empty?
123
+
124
+ changes.concat(Origin.record_changes(current, edits, current_origin, input_length, rule_id))
125
+ current_origin = Origin.apply_edits_to_origin(current_origin, edits)
126
+ current = Edits.apply_edits(current, edits, rule_id)
127
+ end
128
+ changes
129
+ end
59
130
  end
60
131
  end
61
132
  end
@@ -98,6 +98,21 @@ module Polytypo
98
98
  cp >= DIGIT_ZERO && cp <= DIGIT_NINE
99
99
  end
100
100
 
101
+ # ranges.md 3.2a CLOSED-SYMBOL (spec 1.3.0): the symbols conventionally written closed up
102
+ # to a number. A literal code-point set, never a Unicode category test -- a category makes
103
+ # the verdict depend on which Unicode version a runtime was built against, and the five
104
+ # runtimes must agree. The currency part is the U+20A0-U+20CF block by its own bounds, not
105
+ # the subset assigned in some Unicode version: the assigned subset drifts between
106
+ # releases, block bounds do not.
107
+ def closed_up_symbol?(cp)
108
+ return true if cp == 0x0024
109
+ return true if cp >= 0x00A2 && cp <= 0x00A5
110
+ return true if cp >= 0x20A0 && cp <= 0x20CF
111
+ return true if cp == 0x0025 || cp == 0x2030 || cp == 0x2031
112
+
113
+ cp == 0x00B0
114
+ end
115
+
101
116
  # BREAK, including Engine::LINE_MARKER: a member of BREAK for every rule everywhere
102
117
  # (modes.md 3.2).
103
118
  def break?(cp)
@@ -207,6 +222,63 @@ module Polytypo
207
222
  cp[i]
208
223
  end
209
224
 
225
+ # ranges.md 3.2a: a token's flanks after the closed-up-symbol walk, with the digit runs
226
+ # the guards and the replacement then read. `left`/`right` are L\u2032/R\u2032;
227
+ # `outer_left`/`outer_right` are the matched outer symbols' indices, or nil.
228
+ RangeFlanks = Data.define(:left, :right, :a, :b, :outer_left, :outer_right)
229
+
230
+ # ranges.md 3.2 and 3.2a -- is this token a range candidate, and where are its digit runs?
231
+ # nil means it is not one, which is the signal that the token belongs to `dashes`.
232
+ #
233
+ # Both sides are decided from the ORIGINAL left/right, simultaneously; a side consumes a
234
+ # closed-up symbol only when the opposite member repeats the same code point. An unmatched
235
+ # symbol leaves the flank a non-DIGIT, so "$15-\u20ac20" and "15-$20" are not candidates
236
+ # and do not change hands.
237
+ def range_flanks(cp, left, right)
238
+ n = cp.length
239
+ inner_right = if right >= 0 && right < n && closed_up_symbol?(cp[right]) &&
240
+ right + 1 < n && digit?(cp[right + 1])
241
+ cp[right]
242
+ end
243
+ inner_left = if left.positive? && left < n && closed_up_symbol?(cp[left]) &&
244
+ digit?(cp[left - 1])
245
+ cp[left]
246
+ end
247
+
248
+ l = inner_left.nil? ? left : left - 1
249
+ r = inner_right.nil? ? right : right + 1
250
+ return nil if l.negative? || l >= n || r.negative? || r >= n
251
+ return nil unless digit?(cp[l]) && digit?(cp[r])
252
+
253
+ a = l
254
+ a -= 1 while a.positive? && digit?(cp[a - 1])
255
+ b = r
256
+ b += 1 while b + 1 < n && digit?(cp[b + 1])
257
+
258
+ outer_left = nil
259
+ unless inner_right.nil?
260
+ outer_left = effective_index(cp, a - 1, -1)
261
+ return nil if outer_left.nil? || cp[outer_left] != inner_right
262
+ end
263
+ outer_right = nil
264
+ unless inner_left.nil?
265
+ outer_right = effective_index(cp, b + 1, 1)
266
+ return nil if outer_right.nil? || cp[outer_right] != inner_left
267
+ end
268
+
269
+ RangeFlanks.new(left: l, right: r, a: a, b: b, outer_left: outer_left,
270
+ outer_right: outer_right)
271
+ end
272
+
273
+ # T1's reach is transparent to one CLOSED-SYMBOL on either end of a digit run (spec
274
+ # 1.3.0): the run it protects may be a range member carrying an outer symbol.
275
+ def skip_closed_up_symbol(cp, from, step)
276
+ i = effective_index(cp, from, step)
277
+ return from if i.nil? || !closed_up_symbol?(cp[i])
278
+
279
+ i + step
280
+ end
281
+
210
282
  # dashes.md 3.2 step 8 (T1) -- the spacing-transition guard. A tight token must not become
211
283
  # spaced when doing so would insert a U+0020 between itself and a digit run that has
212
284
  # another dash on its far side, read through effective neighbours (3.2b). Shared because a
@@ -215,10 +287,17 @@ module Polytypo
215
287
  def spacing_transition_blocked?(cp, left, right)
216
288
  n = cp.length
217
289
 
290
+ # Spec 1.3.0, position p1: step over a CLOSED-SYMBOL between the token and the run.
291
+ left -= 1 if left >= 0 && left < n && closed_up_symbol?(cp[left]) && left.positive? &&
292
+ digit?(cp[left - 1])
293
+ right += 1 if right >= 0 && right < n && closed_up_symbol?(cp[right]) &&
294
+ right + 1 < n && digit?(cp[right + 1])
295
+
218
296
  if left >= 0 && left < n && digit?(cp[left])
219
297
  d = left
220
298
  d -= 1 while d.positive? && digit?(cp[d - 1])
221
- i1 = effective_index(cp, d - 1, -1)
299
+ # Position p2: and over one at the far end of the run.
300
+ i1 = effective_index(cp, skip_closed_up_symbol(cp, d - 1, -1), -1)
222
301
  one = i1.nil? ? Polytypo::Engine::NONE : cp[i1]
223
302
  two = i1.nil? ? Polytypo::Engine::NONE : effective_neighbour(cp, i1 - 1, -1)
224
303
  return true if dash_union?(one)
@@ -228,7 +307,7 @@ module Polytypo
228
307
  if right >= 0 && right < n && digit?(cp[right])
229
308
  d = right
230
309
  d += 1 while d + 1 < n && digit?(cp[d + 1])
231
- i1 = effective_index(cp, d + 1, 1)
310
+ i1 = effective_index(cp, skip_closed_up_symbol(cp, d + 1, 1), 1)
232
311
  one = i1.nil? ? Polytypo::Engine::NONE : cp[i1]
233
312
  two = i1.nil? ? Polytypo::Engine::NONE : effective_neighbour(cp, i1 + 1, 1)
234
313
  return true if dash_union?(one)
@@ -309,7 +388,10 @@ module Polytypo
309
388
  left_cp = cp[left]
310
389
  right_cp = cp[right]
311
390
 
312
- next if crossed_joiner && !(digit?(left_cp) && digit?(right_cp))
391
+ # dashes.md 3.2a: re-entry across a joiner is only ever a bound range `ranges`
392
+ # produced on an earlier pass. Spec 1.3.0 reads that condition after the
393
+ # closed-up-symbol walk, so a bound range carrying symbols re-enters the same way.
394
+ next if crossed_joiner && range_flanks(cp, left, right).nil?
313
395
  next if break?(left_cp) || break?(right_cp)
314
396
 
315
397
  # dashes.md 3.2 step 6 -- isolation guard.
@@ -72,7 +72,10 @@ module Polytypo
72
72
  DashShared.find_tokens(cp).each do |token|
73
73
  # A digit-flanked token is `ranges`' territory, never `dashes`' -- declined
74
74
  # unconditionally, whether or not `ranges` is enabled (operator decision, spec 0.5.0).
75
- next if DashShared.digit?(token.left_cp) && DashShared.digit?(token.right_cp)
75
+ # A range candidate is `ranges`' territory, never `dashes`'. Since spec 1.3.0 a
76
+ # candidate may carry a matched closed-up symbol on a flank (ranges.md 3.2a), which
77
+ # is why this is range_flanks rather than a digit test on both flanks.
78
+ next unless DashShared.range_flanks(cp, token.left, token.right).nil?
76
79
 
77
80
  # dashes.md 3.4 P5 -- authored en-dash mark-identity veto (spec 0.6.0). A run
78
81
  # consisting of exactly one U+2013 is declined unconditionally: every locale, tight or
@@ -34,10 +34,34 @@ module Polytypo
34
34
  SQUARE_CLOSE = 0x5D
35
35
  BRACE_CLOSE = 0x7D
36
36
 
37
+ SEMICOLON = 0x3B
38
+ AMPERSAND = 0x26
39
+ HASH = 0x23
40
+ # The longest HTML named reference is 31 code points ("CounterClockwiseContourIntegral").
41
+ MAX_CHARACTER_REFERENCE_NAME = 32
42
+
37
43
  EN_DASH = 0x2013
38
44
  EM_DASH = 0x2014
39
45
  ELLIPSIS = 0x2026
40
46
 
47
+ # nbsp.md 3.3 step 4: do the code points left of this ";" have the shape of a character
48
+ # reference? A bounded left walk over ASCII alphanumerics, optionally one "#", then "&".
49
+ # Shape, not the HTML named-reference table -- declining on "&notaname;" costs nothing,
50
+ # and no runtime carries thousands of entries for it.
51
+ def self.ends_character_reference?(cp, index)
52
+ j = index - 1
53
+ j -= 1 while j >= 0 && ascii_alnum?(cp[j])
54
+ length = index - 1 - j
55
+ return false if length < 1 || length > MAX_CHARACTER_REFERENCE_NAME
56
+
57
+ j -= 1 if j >= 0 && cp[j] == HASH
58
+ j >= 0 && cp[j] == AMPERSAND
59
+ end
60
+
61
+ def self.ascii_alnum?(cp)
62
+ (cp >= 0x30 && cp <= 0x39) || (cp >= 0x41 && cp <= 0x5A) || (cp >= 0x61 && cp <= 0x7A)
63
+ end
64
+
41
65
  # cp[i], or NONE if i is out of bounds -- the spec's own boundary value.
42
66
  def self.at(cp, i)
43
67
  return NONE if i.negative? || i >= cp.length
@@ -98,6 +122,8 @@ module Polytypo
98
122
  :before_punctuation, :narrow_before_punctuation, :short_words, :abbreviations,
99
123
  :units, :before_number, :before_word, :symbols, :initial_binding, :opens, :closes,
100
124
  :quote_pairs,
125
+ # nbsp.md 3.1a NARROW-TARGET: what N2 writes, and what N8 writes for a narrow-nbsp pair.
126
+ :narrow_target,
101
127
  keyword_init: true,
102
128
  )
103
129
 
@@ -121,7 +147,7 @@ module Polytypo
121
147
  # Resolves the locale's nbsp and quotes fields to code points once. nbsp.md 2 lists the
122
148
  # fields; 2.1 explains why the mechanism (U+00A0 vs U+202F, convert-only vs insert)
123
149
  # lives here and not in the locale file.
124
- def self.prepare(locale_data)
150
+ def self.prepare(locale_data, narrow_target = NNBSP)
125
151
  data = locale_data["nbsp"]
126
152
  before_punctuation = data["beforePunctuation"].map { |e| single_code_point(e, "beforePunctuation") }
127
153
  narrow_before_punctuation =
@@ -167,7 +193,7 @@ module Polytypo
167
193
  # (7.5).
168
194
  next if open_cp == close_cp
169
195
 
170
- target = pair["innerSpace"] == "nbsp" ? NBSP : NNBSP
196
+ target = pair["innerSpace"] == "nbsp" ? NBSP : narrow_target
171
197
  quote_pairs << QuoteTarget.new(open_cp, close_cp, target)
172
198
  end
173
199
 
@@ -184,6 +210,7 @@ module Polytypo
184
210
  opens: opens,
185
211
  closes: closes,
186
212
  quote_pairs: quote_pairs,
213
+ narrow_target: narrow_target,
187
214
  )
188
215
  end
189
216
 
@@ -295,7 +322,12 @@ module Polytypo
295
322
  next if openish?(prep, left)
296
323
  next if left != NONE && space_like?(left) && openish?(prep, at(cp, i - 2))
297
324
 
298
- # Step 4.
325
+ # Step 4 (spec 1.3.0) -- character-reference guard. text mode has no markup concept,
326
+ # so a locale listing ";" used to insert before the ";" that *ends* a reference and
327
+ # "Bonjour&#160;: oui" stopped being what it was (nbsp.md 3.3 step 4).
328
+ next if cp[i] == SEMICOLON && ends_character_reference?(cp, i)
329
+
330
+ # Step 5.
299
331
  next if left == target
300
332
 
301
333
  if left == SPACE || left == other
@@ -581,12 +613,16 @@ module Polytypo
581
613
  # yielding the index to a lower-priority sub-rule. Sub-rules wanting *different* code
582
614
  # points at a shared index are therefore made disjoint by construction elsewhere
583
615
  # (N1/N2's quote-glyph guard, nbsp.md 3.10.1) rather than relying on ordering alone.
584
- def self.scan(cp, locale_data, _ctx)
585
- prep = prepare(locale_data)
616
+ def self.scan(cp, locale_data, ctx)
617
+ prep = prepare(locale_data, ctx.narrow_target)
586
618
  claims = Array.new(cp.length + 1)
587
619
 
588
- punctuation_sub_rule(cp, prep, claims, prep.before_punctuation, NBSP, NNBSP) # N1
589
- punctuation_sub_rule(cp, prep, claims, prep.narrow_before_punctuation, NNBSP, NBSP) # N2
620
+ punctuation_sub_rule(cp, prep, claims, prep.before_punctuation, NBSP, NNBSP) # N1
621
+ # N2's target is NARROW-TARGET (nbsp.md 3.1a); `other` is the NOBREAK member that is
622
+ # not the target, which is what the sub-rule converts. With the substitution on, N2 and
623
+ # N1 want the same character -- never different ones.
624
+ other = prep.narrow_target == NBSP ? NNBSP : NBSP
625
+ punctuation_sub_rule(cp, prep, claims, prep.narrow_before_punctuation, prep.narrow_target, other) # N2
590
626
  short_words_sub_rule(cp, prep, claims) # N3
591
627
  abbreviations_sub_rule(cp, prep, claims) # N4
592
628
  units_sub_rule(cp, prep, claims) # N5
@@ -38,17 +38,19 @@ module Polytypo
38
38
  true
39
39
  end
40
40
 
41
- # ranges.md 3.3, G1-G5. left/right are the token's post-joiner-walk flank indices (both
42
- # already known to be DIGIT by the caller).
43
- def guards_pass?(cp, left, right)
44
- n = cp.length
45
- a = left
46
- a -= 1 while a.positive? && DashShared.digit?(cp[a - 1])
47
- b = right
48
- b += 1 while b + 1 < n && DashShared.digit?(cp[b + 1])
49
-
50
- before = DashShared.effective_neighbour(cp, a - 1, -1)
51
- after = DashShared.effective_neighbour(cp, b + 1, 1)
41
+ # ranges.md 3.2, G1-G5, over the flanks and digit runs 3.2a's walk produced. before/after
42
+ # read past a matched outer closed-up symbol, so G1-G3 judge the text in front of the
43
+ # whole member rather than the symbol itself -- which is what declines "US$15-$20" on G1.
44
+ def guards_pass?(cp, flanks)
45
+ left = flanks.left
46
+ right = flanks.right
47
+ a = flanks.a
48
+ b = flanks.b
49
+
50
+ before_from = flanks.outer_left || a
51
+ after_from = flanks.outer_right || b
52
+ before = DashShared.effective_neighbour(cp, before_from - 1, -1)
53
+ after = DashShared.effective_neighbour(cp, after_from + 1, 1)
52
54
 
53
55
  # G1 -- no letter adjacency.
54
56
  return false if Polytypo::Engine::UnicodeUtil.letter?(before)
@@ -78,13 +80,13 @@ module Polytypo
78
80
  style = locale_data["dash"]["range"]
79
81
 
80
82
  DashShared.find_tokens(cp).each do |token|
81
- # ranges.md 3.2 -- a range candidate iff both flanks are DIGIT. `ranges` never
82
- # processes any other token shape; that is `dashes`' territory, and `dashes` declines
83
- # a digit-flanked token unconditionally too (operator decision, spec 0.5.0) -- neither
84
- # rule reinterprets the other's shape, whether or not `ranges` is enabled.
85
- next unless DashShared.digit?(token.left_cp) && DashShared.digit?(token.right_cp)
83
+ # ranges.md 3.2, 3.2a -- a candidate iff both flanks are DIGIT once a matched
84
+ # closed-up symbol has been walked over. Anything else is `dashes`' territory, and
85
+ # `dashes` declines a candidate unconditionally too (operator decision, spec 0.5.0).
86
+ flanks = DashShared.range_flanks(cp, token.left, token.right)
87
+ next if flanks.nil?
86
88
 
87
- next unless guards_pass?(cp, token.left, token.right)
89
+ next unless guards_pass?(cp, flanks)
88
90
 
89
91
  # "none": the locale has no verified range convention, so nothing is substituted --
90
92
  # not a fallback to dash.parenthetical, nothing (ranges.md 2).
@@ -92,12 +94,14 @@ module Polytypo
92
94
 
93
95
  if DashShared.spaced_style?(style)
94
96
  # T1: a tight token may not become spaced across a digit run that has a far dash.
97
+ # T1/T2 read the walked flanks: ranges.md 3.2a makes cp[L']/cp[R'] what every
98
+ # shared guard sees once a closed-up symbol has been consumed.
95
99
  next if token.lsp.zero? && token.rsp.zero? &&
96
- DashShared.spacing_transition_blocked?(cp, token.left, token.right)
100
+ DashShared.spacing_transition_blocked?(cp, flanks.left, flanks.right)
97
101
 
98
102
  # T2: the emitted U+0020 must not land where `spaces` (order 10) would delete it.
99
- next if DashShared.strip_before_or_close_bracket?(token.right_cp)
100
- next if DashShared.open_bracket?(token.left_cp)
103
+ next if DashShared.strip_before_or_close_bracket?(cp[flanks.right])
104
+ next if DashShared.open_bracket?(cp[flanks.left])
101
105
  end
102
106
 
103
107
  # ranges.md 3.3.1: never make an edit whose entire content is invisible. Try the
@@ -10,6 +10,9 @@ module Polytypo
10
10
  CODE_MALFORMED_LOCALE_DATA = "POLYTYPO_MALFORMED_LOCALE_DATA"
11
11
  CODE_RULE_CONTRACT = "POLYTYPO_RULE_CONTRACT"
12
12
  CODE_MALFORMED_INPUT = "POLYTYPO_MALFORMED_INPUT"
13
+ # Spec 1.3.0. Deliberately general: every option added from 1.3.0 on shares this code, while
14
+ # `mode` and `dialect` keep their own because callers branch on them.
15
+ CODE_INVALID_OPTION = "POLYTYPO_INVALID_OPTION"
13
16
 
14
17
  # The only error type Transform ever raises. One class is enough: the stable machine-readable
15
18
  # +code+ is the contract, not the exception type or the message text.
@@ -2,6 +2,7 @@
2
2
 
3
3
  require_relative "spans"
4
4
  require_relative "../engine/edits"
5
+ require_relative "../engine/pipeline"
5
6
  require_relative "../engine/registry"
6
7
 
7
8
  module Polytypo
@@ -52,6 +53,22 @@ module Polytypo
52
53
  out.concat(source_cp[cursor..])
53
54
  out.pack("U*")
54
55
  end
56
+
57
+ # run_over_spans, reporting instead of applying (analyze.md section 1). The span table
58
+ # supplies the origin map, so every change comes back in DOCUMENT coordinates -- analyze.md
59
+ # section 6 names a runtime that reports span-local offsets here as the mistake that passes
60
+ # every text-mode test.
61
+ def self.analyze_over_spans(source_cp, spans, plan, locale_data, ctx)
62
+ normalized = Spans.normalize_spans(spans)
63
+ return [] if normalized.empty?
64
+
65
+ Engine::Pipeline.run_rules_recording(
66
+ Spans.concatenate_spans(source_cp, normalized),
67
+ plan, locale_data, ctx,
68
+ Spans.origin_of_spans(normalized),
69
+ source_cp.length
70
+ ) { |current, edits| Spans.filter_boundary_edits(current, edits, Spans.span_ranges_of(current)) }
71
+ end
55
72
  end
56
73
  end
57
74
  end
@@ -1,5 +1,6 @@
1
1
  # frozen_string_literal: true
2
2
 
3
+ require_relative "../engine/origin"
3
4
  require_relative "../engine/sentinels"
4
5
  require_relative "../errors"
5
6
 
@@ -11,6 +12,9 @@ module Polytypo
11
12
  module Spans
12
13
  LINE_TERMINATORS = [0x0A, 0x0D, 0x0B, 0x0C, 0x85, 0x2028, 0x2029].freeze
13
14
 
15
+ # U+0020, the one emitted code point whose meaning is positional (modes.md 3.4, 5 item 2).
16
+ SPACE = 0x20
17
+
14
18
  # A processable span, identified by its offsets in the original source, addressed as
15
19
  # code-point indices (ARCHITECTURE.md section 4.2) -- Ruby String indices are already
16
20
  # code-point indices for a UTF-8 String, no separate byte/char distinction to manage here.
@@ -68,6 +72,19 @@ module Polytypo
68
72
  cp
69
73
  end
70
74
 
75
+ # The origin map for concatenate_spans (analyze.md section 2): for every code point of the
76
+ # joined array, the code-point offset of the character it came from IN THE DOCUMENT, and
77
+ # NO_ORIGIN for the markers, which came from nowhere. A Span's bounds are already
78
+ # code-point offsets, so no coordinate conversion belongs here.
79
+ def self.origin_of_spans(spans)
80
+ origin = []
81
+ spans.each_with_index do |span, i|
82
+ origin << Engine::Origin::NO_ORIGIN unless i.zero?
83
+ origin.concat((span.start...span.end).to_a)
84
+ end
85
+ origin
86
+ end
87
+
71
88
  # The span extents of the array as it stands. Recomputed after every rule, because applying
72
89
  # edits shifts every index after the first one -- the markers themselves always survive,
73
90
  # since no edit may contain one.
@@ -94,7 +111,13 @@ module Polytypo
94
111
  # 1. No edit may contain a marker -- one that does is a bug, discarded rather than
95
112
  # redistributed.
96
113
  # 2. The edge-growth rule: an edit is discarded if it would place code points at an
97
- # extremity of its span that were not there before.
114
+ # extremity of its span that were not there before. That sentence is the rule;
115
+ # "r > d" alone is an incorrect formalisation of it and misses r == d. dashes P3
116
+ # admits a run of THREE dashes, so "---" -> U+0020 en-dash U+0020 is 3 -> 3: the
117
+ # length test sees nothing while U+0020 lands on both extremities anyway. The second
118
+ # clause tests the CHARACTER, and only U+0020 needs testing -- it is the one code
119
+ # point any rule emits whose meaning comes from its position rather than from itself
120
+ # (modes.md 5 item 2).
98
121
  def self.filter_boundary_edits(cp, edits, ranges)
99
122
  edits.select do |edit|
100
123
  next false if (edit.start...edit.end).any? { |i| Engine.marker?(cp[i]) }
@@ -104,7 +127,12 @@ module Polytypo
104
127
  d = edit.end - edit.start
105
128
  r = edit.replacement.length
106
129
  span = span_containing(ranges, p)
107
- !(span && r > d && (p == span.first || q == span.last))
130
+ next true unless span && (p == span.first || q == span.last)
131
+ next false if r > d
132
+ next false if r.positive? && p == span.first && edit.replacement[0] == SPACE && cp[p] != SPACE
133
+ next false if r.positive? && q == span.last && edit.replacement[r - 1] == SPACE && cp[q] != SPACE
134
+
135
+ true
108
136
  end
109
137
  end
110
138