polytypo 1.2.0 → 1.3.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/README.md +33 -1
- data/lib/polytypo/data/VERSION +1 -1
- data/lib/polytypo/data/fixtures/cs.json +161 -0
- data/lib/polytypo/data/fixtures/de-CH.json +1 -1
- data/lib/polytypo/data/fixtures/de-DE.json +195 -6
- data/lib/polytypo/data/fixtures/el.json +1 -1
- data/lib/polytypo/data/fixtures/en-GB.json +12 -1
- data/lib/polytypo/data/fixtures/en-US.json +648 -1
- data/lib/polytypo/data/fixtures/es.json +193 -0
- data/lib/polytypo/data/fixtures/fi.json +1 -1
- data/lib/polytypo/data/fixtures/fr-CA.json +25 -1
- data/lib/polytypo/data/fixtures/fr.json +176 -1
- data/lib/polytypo/data/fixtures/it.json +161 -0
- data/lib/polytypo/data/fixtures/locale-resolution.json +76 -4
- data/lib/polytypo/data/fixtures/nl.json +121 -0
- data/lib/polytypo/data/fixtures/pl.json +137 -0
- data/lib/polytypo/data/fixtures/pt-BR.json +156 -0
- data/lib/polytypo/data/fixtures/pt-PT.json +156 -0
- data/lib/polytypo/data/fixtures/ru.json +23 -1
- data/lib/polytypo/data/fixtures/sv.json +1 -1
- data/lib/polytypo/data/fixtures/uk.json +153 -0
- data/lib/polytypo/data/locales/cs.json +90 -0
- data/lib/polytypo/data/locales/de-DE.json +7 -2
- data/lib/polytypo/data/locales/en-US.json +3 -3
- data/lib/polytypo/data/locales/es.json +111 -0
- data/lib/polytypo/data/locales/fr-CA.json +7 -1
- data/lib/polytypo/data/locales/fr.json +7 -1
- data/lib/polytypo/data/locales/it.json +95 -0
- data/lib/polytypo/data/locales/nl.json +84 -0
- data/lib/polytypo/data/locales/pl.json +96 -0
- data/lib/polytypo/data/locales/pt-BR.json +82 -0
- data/lib/polytypo/data/locales/pt-PT.json +84 -0
- data/lib/polytypo/data/locales/registry.json +23 -3
- data/lib/polytypo/data/locales/ru.json +2 -2
- data/lib/polytypo/data/locales/uk.json +130 -0
- data/lib/polytypo/data/rules/analyze.md +157 -0
- data/lib/polytypo/data/rules/apostrophe.md +432 -0
- data/lib/polytypo/data/rules/dashes.md +128 -37
- data/lib/polytypo/data/rules/ellipsis.md +271 -0
- data/lib/polytypo/data/rules/hyphen.md +353 -0
- data/lib/polytypo/data/rules/locale-resolution.md +239 -0
- data/lib/polytypo/data/rules/modes.md +1281 -0
- data/lib/polytypo/data/rules/nbsp.md +1157 -0
- data/lib/polytypo/data/rules/order.json +11 -11
- data/lib/polytypo/data/rules/pipeline-idempotency.md +605 -0
- data/lib/polytypo/data/rules/quotes.md +1324 -0
- data/lib/polytypo/data/rules/ranges.md +489 -0
- data/lib/polytypo/data/rules/spaces.md +649 -0
- data/lib/polytypo/data/rules/symbols.md +540 -0
- data/lib/polytypo/data/schema/fixtures.schema.json +18 -3
- data/lib/polytypo/engine/origin.rb +75 -0
- data/lib/polytypo/engine/pipeline.rb +72 -1
- data/lib/polytypo/engine/rules/dash_shared.rb +85 -3
- data/lib/polytypo/engine/rules/dashes.rb +4 -1
- data/lib/polytypo/engine/rules/nbsp.rb +43 -7
- data/lib/polytypo/engine/rules/ranges.rb +24 -20
- data/lib/polytypo/errors.rb +3 -0
- data/lib/polytypo/modes/runner.rb +17 -0
- data/lib/polytypo/modes/spans.rb +30 -2
- data/lib/polytypo/modes/yaml.rb +312 -0
- data/lib/polytypo/version.rb +1 -1
- data/lib/polytypo.rb +126 -15
- metadata +31 -1
|
@@ -0,0 +1,75 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require_relative "codepoints"
|
|
4
|
+
|
|
5
|
+
module Polytypo
|
|
6
|
+
# One entry of Polytypo.analyze's result, in input coordinates (analyze.md section 2):
|
|
7
|
+
# +rule_id+ is a rule id from spec/rules/order.json, +start+/+end+ are code-point offsets into
|
|
8
|
+
# the input (inclusive/exclusive), and +before+/+after+ are the text on either side of that one
|
|
9
|
+
# edit. +start+ == +end+ is a pure insertion, and +before+ is then empty.
|
|
10
|
+
#
|
|
11
|
+
# Public API, so it is named in this namespace rather than in Engine, where it is built.
|
|
12
|
+
Change = Data.define(:rule_id, :start, :end, :before, :after)
|
|
13
|
+
|
|
14
|
+
module Engine
|
|
15
|
+
# analyze.md section 2: every reported offset is a code-point offset into the input the
|
|
16
|
+
# caller passed, in every mode. The rules, however, run over an array that is not the input
|
|
17
|
+
# -- in "text" mode edits from earlier rules have already shifted it, and in "html"/"markdown"
|
|
18
|
+
# mode it is the marker-joined concatenation of the processable spans (modes.md 3.5). This
|
|
19
|
+
# module carries the one structure that bridges the two: an origin map, parallel to the
|
|
20
|
+
# current code-point array, holding the input offset each code point came from, or NO_ORIGIN
|
|
21
|
+
# for one the pipeline itself produced. Mirrors polytypo-js's src/engine/origin.ts.
|
|
22
|
+
module Origin
|
|
23
|
+
NO_ORIGIN = -1
|
|
24
|
+
|
|
25
|
+
# The input offset an edit boundary at +index+ addresses. Synthetic code points have no
|
|
26
|
+
# origin of their own, so the scan runs forward to the first that has one -- an insertion
|
|
27
|
+
# between two earlier insertions still lands where the next real character is. Falling off
|
|
28
|
+
# the end means the boundary is at the end of the input.
|
|
29
|
+
def self.origin_at(origin, index, input_length)
|
|
30
|
+
(index...origin.length).each do |i|
|
|
31
|
+
return origin[i] unless origin[i] == NO_ORIGIN
|
|
32
|
+
end
|
|
33
|
+
input_length
|
|
34
|
+
end
|
|
35
|
+
|
|
36
|
+
# The origin map for the array Edits.apply_edits is about to produce. A replacement of
|
|
37
|
+
# equal length keeps its origins position by position, which is what makes a conversion
|
|
38
|
+
# (U+0020 -> U+00A0) still point at the character it converted; anything longer is
|
|
39
|
+
# synthetic beyond the positions it covers.
|
|
40
|
+
def self.apply_edits_to_origin(origin, edits)
|
|
41
|
+
return origin.dup if edits.empty?
|
|
42
|
+
|
|
43
|
+
out = []
|
|
44
|
+
cursor = 0
|
|
45
|
+
edits.each do |edit|
|
|
46
|
+
out.concat(origin[cursor...edit.start])
|
|
47
|
+
edit.replacement.each_index do |k|
|
|
48
|
+
source = edit.start + k
|
|
49
|
+
out << (source < edit.end ? origin[source] : NO_ORIGIN)
|
|
50
|
+
end
|
|
51
|
+
cursor = edit.end
|
|
52
|
+
end
|
|
53
|
+
out.concat(origin[cursor..])
|
|
54
|
+
out
|
|
55
|
+
end
|
|
56
|
+
|
|
57
|
+
# One rule's edits, in the coordinates that rule saw, rendered as Changes in input
|
|
58
|
+
# coordinates. +before+ is the text this rule replaced and +after+ what it replaced it with
|
|
59
|
+
# (analyze.md section 2), so on a text two rules have both touched, +before+ is what the
|
|
60
|
+
# second rule saw rather than what the caller typed -- section 5 says so and shows the
|
|
61
|
+
# French case where it matters.
|
|
62
|
+
def self.record_changes(cp, edits, origin, input_length, rule_id)
|
|
63
|
+
edits.map do |edit|
|
|
64
|
+
Polytypo::Change.new(
|
|
65
|
+
rule_id: rule_id,
|
|
66
|
+
start: origin_at(origin, edit.start, input_length),
|
|
67
|
+
end: origin_at(origin, edit.end, input_length),
|
|
68
|
+
before: Codepoints.from_codepoints(cp[edit.start...edit.end]),
|
|
69
|
+
after: Codepoints.from_codepoints(edit.replacement)
|
|
70
|
+
)
|
|
71
|
+
end
|
|
72
|
+
end
|
|
73
|
+
end
|
|
74
|
+
end
|
|
75
|
+
end
|
|
@@ -2,13 +2,62 @@
|
|
|
2
2
|
|
|
3
3
|
require_relative "edits"
|
|
4
4
|
require_relative "locale"
|
|
5
|
+
require_relative "origin"
|
|
5
6
|
require_relative "registry"
|
|
6
7
|
require_relative "../errors"
|
|
7
8
|
|
|
8
9
|
module Polytypo
|
|
9
10
|
module Engine
|
|
10
11
|
# Per-call context every rule reads, alongside the code-point array and the locale data.
|
|
11
|
-
|
|
12
|
+
# :narrow_target is nbsp.md 3.1a's NARROW-TARGET, already resolved to a code point: U+202F
|
|
13
|
+
# by default, U+00A0 when the caller passed narrow_nbsp: "nbsp". A rule reads a code point
|
|
14
|
+
# and never the option, so the string never reaches the pipeline.
|
|
15
|
+
RuleContext = Struct.new(:mode, :dialect, :locale, :narrow_target, keyword_init: true)
|
|
16
|
+
|
|
17
|
+
# nbsp.md 3.1a: resolve the narrow_nbsp option to the code point the rule writes. Checked
|
|
18
|
+
# immediately after mode and before rules (ARCHITECTURE.md 7) -- the two checks that read
|
|
19
|
+
# nothing but the call itself come first -- and it runs whether or not `nbsp` is enabled, so
|
|
20
|
+
# a misspelled value still raises rather than being silently ignored.
|
|
21
|
+
NARROW_NO_BREAK_SPACE = 0x202F
|
|
22
|
+
NO_BREAK_SPACE = 0x00A0
|
|
23
|
+
|
|
24
|
+
def self.resolve_narrow_target(value)
|
|
25
|
+
return NARROW_NO_BREAK_SPACE if value.nil? || value == "narrow"
|
|
26
|
+
return NO_BREAK_SPACE if value == "nbsp"
|
|
27
|
+
|
|
28
|
+
raise Polytypo::Error.new(
|
|
29
|
+
Polytypo::CODE_INVALID_OPTION,
|
|
30
|
+
"Unknown narrow_nbsp #{value.inspect}. Expected \"narrow\" or \"nbsp\".",
|
|
31
|
+
)
|
|
32
|
+
end
|
|
33
|
+
|
|
34
|
+
# modes.md 3.8.2: "yaml" mode's `keys` option, resolved once at the call boundary.
|
|
35
|
+
#
|
|
36
|
+
# It has no default, for the reason `dialect` has none. YAML is a data format with islands of
|
|
37
|
+
# prose in it, and nothing in its syntax marks them -- `description` holds a sentence and
|
|
38
|
+
# `run` holds a shell script, spelled identically -- so a default would be a guess about the
|
|
39
|
+
# schema above the document. An empty list is legal: "process nothing" is a choice a caller
|
|
40
|
+
# may make, not an error. Checked after locale, last of the option checks, alongside dialect.
|
|
41
|
+
def self.resolve_yaml_keys(value)
|
|
42
|
+
unless value.is_a?(Array)
|
|
43
|
+
raise Polytypo::Error.new(
|
|
44
|
+
Polytypo::CODE_INVALID_OPTION,
|
|
45
|
+
'"keys" is required when mode is "yaml" and must be an Array of Strings ' \
|
|
46
|
+
"(modes.md 3.8.2). Received #{value.inspect}.",
|
|
47
|
+
)
|
|
48
|
+
end
|
|
49
|
+
|
|
50
|
+
value.each do |key|
|
|
51
|
+
next if key.is_a?(String)
|
|
52
|
+
|
|
53
|
+
raise Polytypo::Error.new(
|
|
54
|
+
Polytypo::CODE_INVALID_OPTION,
|
|
55
|
+
"\"keys\" must contain only Strings; received #{key.inspect}.",
|
|
56
|
+
)
|
|
57
|
+
end
|
|
58
|
+
|
|
59
|
+
value.to_set
|
|
60
|
+
end
|
|
12
61
|
|
|
13
62
|
# Resolves the locale, builds the rule plan, and runs each enabled rule in
|
|
14
63
|
# spec/rules/order.json order over a code-point array, applying its edits before the next
|
|
@@ -56,6 +105,28 @@ module Polytypo
|
|
|
56
105
|
end
|
|
57
106
|
current
|
|
58
107
|
end
|
|
108
|
+
|
|
109
|
+
# run_rules, keeping the edits instead of discarding them (analyze.md section 1: same
|
|
110
|
+
# pipeline, same order, reporting rather than applying). The origin map travels alongside
|
|
111
|
+
# the array so every change comes back in input coordinates, and +filter_edits+ is the hook
|
|
112
|
+
# the span-runner needs for modes.md 3.4's boundary filters, passed as a block -- text mode
|
|
113
|
+
# passes none and gets the identity.
|
|
114
|
+
def self.run_rules_recording(cp, plan, locale_data, ctx, origin, input_length)
|
|
115
|
+
current = cp
|
|
116
|
+
current_origin = origin
|
|
117
|
+
changes = []
|
|
118
|
+
plan.each do |rule_id|
|
|
119
|
+
fn = Registry.rule(rule_id)
|
|
120
|
+
produced = fn.call(current, locale_data, ctx)
|
|
121
|
+
edits = block_given? ? yield(current, produced) : produced
|
|
122
|
+
next if edits.empty?
|
|
123
|
+
|
|
124
|
+
changes.concat(Origin.record_changes(current, edits, current_origin, input_length, rule_id))
|
|
125
|
+
current_origin = Origin.apply_edits_to_origin(current_origin, edits)
|
|
126
|
+
current = Edits.apply_edits(current, edits, rule_id)
|
|
127
|
+
end
|
|
128
|
+
changes
|
|
129
|
+
end
|
|
59
130
|
end
|
|
60
131
|
end
|
|
61
132
|
end
|
|
@@ -98,6 +98,21 @@ module Polytypo
|
|
|
98
98
|
cp >= DIGIT_ZERO && cp <= DIGIT_NINE
|
|
99
99
|
end
|
|
100
100
|
|
|
101
|
+
# ranges.md 3.2a CLOSED-SYMBOL (spec 1.3.0): the symbols conventionally written closed up
|
|
102
|
+
# to a number. A literal code-point set, never a Unicode category test -- a category makes
|
|
103
|
+
# the verdict depend on which Unicode version a runtime was built against, and the five
|
|
104
|
+
# runtimes must agree. The currency part is the U+20A0-U+20CF block by its own bounds, not
|
|
105
|
+
# the subset assigned in some Unicode version: the assigned subset drifts between
|
|
106
|
+
# releases, block bounds do not.
|
|
107
|
+
def closed_up_symbol?(cp)
|
|
108
|
+
return true if cp == 0x0024
|
|
109
|
+
return true if cp >= 0x00A2 && cp <= 0x00A5
|
|
110
|
+
return true if cp >= 0x20A0 && cp <= 0x20CF
|
|
111
|
+
return true if cp == 0x0025 || cp == 0x2030 || cp == 0x2031
|
|
112
|
+
|
|
113
|
+
cp == 0x00B0
|
|
114
|
+
end
|
|
115
|
+
|
|
101
116
|
# BREAK, including Engine::LINE_MARKER: a member of BREAK for every rule everywhere
|
|
102
117
|
# (modes.md 3.2).
|
|
103
118
|
def break?(cp)
|
|
@@ -207,6 +222,63 @@ module Polytypo
|
|
|
207
222
|
cp[i]
|
|
208
223
|
end
|
|
209
224
|
|
|
225
|
+
# ranges.md 3.2a: a token's flanks after the closed-up-symbol walk, with the digit runs
|
|
226
|
+
# the guards and the replacement then read. `left`/`right` are L\u2032/R\u2032;
|
|
227
|
+
# `outer_left`/`outer_right` are the matched outer symbols' indices, or nil.
|
|
228
|
+
RangeFlanks = Data.define(:left, :right, :a, :b, :outer_left, :outer_right)
|
|
229
|
+
|
|
230
|
+
# ranges.md 3.2 and 3.2a -- is this token a range candidate, and where are its digit runs?
|
|
231
|
+
# nil means it is not one, which is the signal that the token belongs to `dashes`.
|
|
232
|
+
#
|
|
233
|
+
# Both sides are decided from the ORIGINAL left/right, simultaneously; a side consumes a
|
|
234
|
+
# closed-up symbol only when the opposite member repeats the same code point. An unmatched
|
|
235
|
+
# symbol leaves the flank a non-DIGIT, so "$15-\u20ac20" and "15-$20" are not candidates
|
|
236
|
+
# and do not change hands.
|
|
237
|
+
def range_flanks(cp, left, right)
|
|
238
|
+
n = cp.length
|
|
239
|
+
inner_right = if right >= 0 && right < n && closed_up_symbol?(cp[right]) &&
|
|
240
|
+
right + 1 < n && digit?(cp[right + 1])
|
|
241
|
+
cp[right]
|
|
242
|
+
end
|
|
243
|
+
inner_left = if left.positive? && left < n && closed_up_symbol?(cp[left]) &&
|
|
244
|
+
digit?(cp[left - 1])
|
|
245
|
+
cp[left]
|
|
246
|
+
end
|
|
247
|
+
|
|
248
|
+
l = inner_left.nil? ? left : left - 1
|
|
249
|
+
r = inner_right.nil? ? right : right + 1
|
|
250
|
+
return nil if l.negative? || l >= n || r.negative? || r >= n
|
|
251
|
+
return nil unless digit?(cp[l]) && digit?(cp[r])
|
|
252
|
+
|
|
253
|
+
a = l
|
|
254
|
+
a -= 1 while a.positive? && digit?(cp[a - 1])
|
|
255
|
+
b = r
|
|
256
|
+
b += 1 while b + 1 < n && digit?(cp[b + 1])
|
|
257
|
+
|
|
258
|
+
outer_left = nil
|
|
259
|
+
unless inner_right.nil?
|
|
260
|
+
outer_left = effective_index(cp, a - 1, -1)
|
|
261
|
+
return nil if outer_left.nil? || cp[outer_left] != inner_right
|
|
262
|
+
end
|
|
263
|
+
outer_right = nil
|
|
264
|
+
unless inner_left.nil?
|
|
265
|
+
outer_right = effective_index(cp, b + 1, 1)
|
|
266
|
+
return nil if outer_right.nil? || cp[outer_right] != inner_left
|
|
267
|
+
end
|
|
268
|
+
|
|
269
|
+
RangeFlanks.new(left: l, right: r, a: a, b: b, outer_left: outer_left,
|
|
270
|
+
outer_right: outer_right)
|
|
271
|
+
end
|
|
272
|
+
|
|
273
|
+
# T1's reach is transparent to one CLOSED-SYMBOL on either end of a digit run (spec
|
|
274
|
+
# 1.3.0): the run it protects may be a range member carrying an outer symbol.
|
|
275
|
+
def skip_closed_up_symbol(cp, from, step)
|
|
276
|
+
i = effective_index(cp, from, step)
|
|
277
|
+
return from if i.nil? || !closed_up_symbol?(cp[i])
|
|
278
|
+
|
|
279
|
+
i + step
|
|
280
|
+
end
|
|
281
|
+
|
|
210
282
|
# dashes.md 3.2 step 8 (T1) -- the spacing-transition guard. A tight token must not become
|
|
211
283
|
# spaced when doing so would insert a U+0020 between itself and a digit run that has
|
|
212
284
|
# another dash on its far side, read through effective neighbours (3.2b). Shared because a
|
|
@@ -215,10 +287,17 @@ module Polytypo
|
|
|
215
287
|
def spacing_transition_blocked?(cp, left, right)
|
|
216
288
|
n = cp.length
|
|
217
289
|
|
|
290
|
+
# Spec 1.3.0, position p1: step over a CLOSED-SYMBOL between the token and the run.
|
|
291
|
+
left -= 1 if left >= 0 && left < n && closed_up_symbol?(cp[left]) && left.positive? &&
|
|
292
|
+
digit?(cp[left - 1])
|
|
293
|
+
right += 1 if right >= 0 && right < n && closed_up_symbol?(cp[right]) &&
|
|
294
|
+
right + 1 < n && digit?(cp[right + 1])
|
|
295
|
+
|
|
218
296
|
if left >= 0 && left < n && digit?(cp[left])
|
|
219
297
|
d = left
|
|
220
298
|
d -= 1 while d.positive? && digit?(cp[d - 1])
|
|
221
|
-
|
|
299
|
+
# Position p2: and over one at the far end of the run.
|
|
300
|
+
i1 = effective_index(cp, skip_closed_up_symbol(cp, d - 1, -1), -1)
|
|
222
301
|
one = i1.nil? ? Polytypo::Engine::NONE : cp[i1]
|
|
223
302
|
two = i1.nil? ? Polytypo::Engine::NONE : effective_neighbour(cp, i1 - 1, -1)
|
|
224
303
|
return true if dash_union?(one)
|
|
@@ -228,7 +307,7 @@ module Polytypo
|
|
|
228
307
|
if right >= 0 && right < n && digit?(cp[right])
|
|
229
308
|
d = right
|
|
230
309
|
d += 1 while d + 1 < n && digit?(cp[d + 1])
|
|
231
|
-
i1 = effective_index(cp, d + 1, 1)
|
|
310
|
+
i1 = effective_index(cp, skip_closed_up_symbol(cp, d + 1, 1), 1)
|
|
232
311
|
one = i1.nil? ? Polytypo::Engine::NONE : cp[i1]
|
|
233
312
|
two = i1.nil? ? Polytypo::Engine::NONE : effective_neighbour(cp, i1 + 1, 1)
|
|
234
313
|
return true if dash_union?(one)
|
|
@@ -309,7 +388,10 @@ module Polytypo
|
|
|
309
388
|
left_cp = cp[left]
|
|
310
389
|
right_cp = cp[right]
|
|
311
390
|
|
|
312
|
-
|
|
391
|
+
# dashes.md 3.2a: re-entry across a joiner is only ever a bound range `ranges`
|
|
392
|
+
# produced on an earlier pass. Spec 1.3.0 reads that condition after the
|
|
393
|
+
# closed-up-symbol walk, so a bound range carrying symbols re-enters the same way.
|
|
394
|
+
next if crossed_joiner && range_flanks(cp, left, right).nil?
|
|
313
395
|
next if break?(left_cp) || break?(right_cp)
|
|
314
396
|
|
|
315
397
|
# dashes.md 3.2 step 6 -- isolation guard.
|
|
@@ -72,7 +72,10 @@ module Polytypo
|
|
|
72
72
|
DashShared.find_tokens(cp).each do |token|
|
|
73
73
|
# A digit-flanked token is `ranges`' territory, never `dashes`' -- declined
|
|
74
74
|
# unconditionally, whether or not `ranges` is enabled (operator decision, spec 0.5.0).
|
|
75
|
-
|
|
75
|
+
# A range candidate is `ranges`' territory, never `dashes`'. Since spec 1.3.0 a
|
|
76
|
+
# candidate may carry a matched closed-up symbol on a flank (ranges.md 3.2a), which
|
|
77
|
+
# is why this is range_flanks rather than a digit test on both flanks.
|
|
78
|
+
next unless DashShared.range_flanks(cp, token.left, token.right).nil?
|
|
76
79
|
|
|
77
80
|
# dashes.md 3.4 P5 -- authored en-dash mark-identity veto (spec 0.6.0). A run
|
|
78
81
|
# consisting of exactly one U+2013 is declined unconditionally: every locale, tight or
|
|
@@ -34,10 +34,34 @@ module Polytypo
|
|
|
34
34
|
SQUARE_CLOSE = 0x5D
|
|
35
35
|
BRACE_CLOSE = 0x7D
|
|
36
36
|
|
|
37
|
+
SEMICOLON = 0x3B
|
|
38
|
+
AMPERSAND = 0x26
|
|
39
|
+
HASH = 0x23
|
|
40
|
+
# The longest HTML named reference is 31 code points ("CounterClockwiseContourIntegral").
|
|
41
|
+
MAX_CHARACTER_REFERENCE_NAME = 32
|
|
42
|
+
|
|
37
43
|
EN_DASH = 0x2013
|
|
38
44
|
EM_DASH = 0x2014
|
|
39
45
|
ELLIPSIS = 0x2026
|
|
40
46
|
|
|
47
|
+
# nbsp.md 3.3 step 4: do the code points left of this ";" have the shape of a character
|
|
48
|
+
# reference? A bounded left walk over ASCII alphanumerics, optionally one "#", then "&".
|
|
49
|
+
# Shape, not the HTML named-reference table -- declining on "¬aname;" costs nothing,
|
|
50
|
+
# and no runtime carries thousands of entries for it.
|
|
51
|
+
def self.ends_character_reference?(cp, index)
|
|
52
|
+
j = index - 1
|
|
53
|
+
j -= 1 while j >= 0 && ascii_alnum?(cp[j])
|
|
54
|
+
length = index - 1 - j
|
|
55
|
+
return false if length < 1 || length > MAX_CHARACTER_REFERENCE_NAME
|
|
56
|
+
|
|
57
|
+
j -= 1 if j >= 0 && cp[j] == HASH
|
|
58
|
+
j >= 0 && cp[j] == AMPERSAND
|
|
59
|
+
end
|
|
60
|
+
|
|
61
|
+
def self.ascii_alnum?(cp)
|
|
62
|
+
(cp >= 0x30 && cp <= 0x39) || (cp >= 0x41 && cp <= 0x5A) || (cp >= 0x61 && cp <= 0x7A)
|
|
63
|
+
end
|
|
64
|
+
|
|
41
65
|
# cp[i], or NONE if i is out of bounds -- the spec's own boundary value.
|
|
42
66
|
def self.at(cp, i)
|
|
43
67
|
return NONE if i.negative? || i >= cp.length
|
|
@@ -98,6 +122,8 @@ module Polytypo
|
|
|
98
122
|
:before_punctuation, :narrow_before_punctuation, :short_words, :abbreviations,
|
|
99
123
|
:units, :before_number, :before_word, :symbols, :initial_binding, :opens, :closes,
|
|
100
124
|
:quote_pairs,
|
|
125
|
+
# nbsp.md 3.1a NARROW-TARGET: what N2 writes, and what N8 writes for a narrow-nbsp pair.
|
|
126
|
+
:narrow_target,
|
|
101
127
|
keyword_init: true,
|
|
102
128
|
)
|
|
103
129
|
|
|
@@ -121,7 +147,7 @@ module Polytypo
|
|
|
121
147
|
# Resolves the locale's nbsp and quotes fields to code points once. nbsp.md 2 lists the
|
|
122
148
|
# fields; 2.1 explains why the mechanism (U+00A0 vs U+202F, convert-only vs insert)
|
|
123
149
|
# lives here and not in the locale file.
|
|
124
|
-
def self.prepare(locale_data)
|
|
150
|
+
def self.prepare(locale_data, narrow_target = NNBSP)
|
|
125
151
|
data = locale_data["nbsp"]
|
|
126
152
|
before_punctuation = data["beforePunctuation"].map { |e| single_code_point(e, "beforePunctuation") }
|
|
127
153
|
narrow_before_punctuation =
|
|
@@ -167,7 +193,7 @@ module Polytypo
|
|
|
167
193
|
# (7.5).
|
|
168
194
|
next if open_cp == close_cp
|
|
169
195
|
|
|
170
|
-
target = pair["innerSpace"] == "nbsp" ? NBSP :
|
|
196
|
+
target = pair["innerSpace"] == "nbsp" ? NBSP : narrow_target
|
|
171
197
|
quote_pairs << QuoteTarget.new(open_cp, close_cp, target)
|
|
172
198
|
end
|
|
173
199
|
|
|
@@ -184,6 +210,7 @@ module Polytypo
|
|
|
184
210
|
opens: opens,
|
|
185
211
|
closes: closes,
|
|
186
212
|
quote_pairs: quote_pairs,
|
|
213
|
+
narrow_target: narrow_target,
|
|
187
214
|
)
|
|
188
215
|
end
|
|
189
216
|
|
|
@@ -295,7 +322,12 @@ module Polytypo
|
|
|
295
322
|
next if openish?(prep, left)
|
|
296
323
|
next if left != NONE && space_like?(left) && openish?(prep, at(cp, i - 2))
|
|
297
324
|
|
|
298
|
-
# Step 4.
|
|
325
|
+
# Step 4 (spec 1.3.0) -- character-reference guard. text mode has no markup concept,
|
|
326
|
+
# so a locale listing ";" used to insert before the ";" that *ends* a reference and
|
|
327
|
+
# "Bonjour : oui" stopped being what it was (nbsp.md 3.3 step 4).
|
|
328
|
+
next if cp[i] == SEMICOLON && ends_character_reference?(cp, i)
|
|
329
|
+
|
|
330
|
+
# Step 5.
|
|
299
331
|
next if left == target
|
|
300
332
|
|
|
301
333
|
if left == SPACE || left == other
|
|
@@ -581,12 +613,16 @@ module Polytypo
|
|
|
581
613
|
# yielding the index to a lower-priority sub-rule. Sub-rules wanting *different* code
|
|
582
614
|
# points at a shared index are therefore made disjoint by construction elsewhere
|
|
583
615
|
# (N1/N2's quote-glyph guard, nbsp.md 3.10.1) rather than relying on ordering alone.
|
|
584
|
-
def self.scan(cp, locale_data,
|
|
585
|
-
prep = prepare(locale_data)
|
|
616
|
+
def self.scan(cp, locale_data, ctx)
|
|
617
|
+
prep = prepare(locale_data, ctx.narrow_target)
|
|
586
618
|
claims = Array.new(cp.length + 1)
|
|
587
619
|
|
|
588
|
-
punctuation_sub_rule(cp, prep, claims, prep.before_punctuation, NBSP, NNBSP)
|
|
589
|
-
|
|
620
|
+
punctuation_sub_rule(cp, prep, claims, prep.before_punctuation, NBSP, NNBSP) # N1
|
|
621
|
+
# N2's target is NARROW-TARGET (nbsp.md 3.1a); `other` is the NOBREAK member that is
|
|
622
|
+
# not the target, which is what the sub-rule converts. With the substitution on, N2 and
|
|
623
|
+
# N1 want the same character -- never different ones.
|
|
624
|
+
other = prep.narrow_target == NBSP ? NNBSP : NBSP
|
|
625
|
+
punctuation_sub_rule(cp, prep, claims, prep.narrow_before_punctuation, prep.narrow_target, other) # N2
|
|
590
626
|
short_words_sub_rule(cp, prep, claims) # N3
|
|
591
627
|
abbreviations_sub_rule(cp, prep, claims) # N4
|
|
592
628
|
units_sub_rule(cp, prep, claims) # N5
|
|
@@ -38,17 +38,19 @@ module Polytypo
|
|
|
38
38
|
true
|
|
39
39
|
end
|
|
40
40
|
|
|
41
|
-
# ranges.md 3.
|
|
42
|
-
#
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
b
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
41
|
+
# ranges.md 3.2, G1-G5, over the flanks and digit runs 3.2a's walk produced. before/after
|
|
42
|
+
# read past a matched outer closed-up symbol, so G1-G3 judge the text in front of the
|
|
43
|
+
# whole member rather than the symbol itself -- which is what declines "US$15-$20" on G1.
|
|
44
|
+
def guards_pass?(cp, flanks)
|
|
45
|
+
left = flanks.left
|
|
46
|
+
right = flanks.right
|
|
47
|
+
a = flanks.a
|
|
48
|
+
b = flanks.b
|
|
49
|
+
|
|
50
|
+
before_from = flanks.outer_left || a
|
|
51
|
+
after_from = flanks.outer_right || b
|
|
52
|
+
before = DashShared.effective_neighbour(cp, before_from - 1, -1)
|
|
53
|
+
after = DashShared.effective_neighbour(cp, after_from + 1, 1)
|
|
52
54
|
|
|
53
55
|
# G1 -- no letter adjacency.
|
|
54
56
|
return false if Polytypo::Engine::UnicodeUtil.letter?(before)
|
|
@@ -78,13 +80,13 @@ module Polytypo
|
|
|
78
80
|
style = locale_data["dash"]["range"]
|
|
79
81
|
|
|
80
82
|
DashShared.find_tokens(cp).each do |token|
|
|
81
|
-
# ranges.md 3.2 -- a
|
|
82
|
-
#
|
|
83
|
-
# a
|
|
84
|
-
|
|
85
|
-
next
|
|
83
|
+
# ranges.md 3.2, 3.2a -- a candidate iff both flanks are DIGIT once a matched
|
|
84
|
+
# closed-up symbol has been walked over. Anything else is `dashes`' territory, and
|
|
85
|
+
# `dashes` declines a candidate unconditionally too (operator decision, spec 0.5.0).
|
|
86
|
+
flanks = DashShared.range_flanks(cp, token.left, token.right)
|
|
87
|
+
next if flanks.nil?
|
|
86
88
|
|
|
87
|
-
next unless guards_pass?(cp,
|
|
89
|
+
next unless guards_pass?(cp, flanks)
|
|
88
90
|
|
|
89
91
|
# "none": the locale has no verified range convention, so nothing is substituted --
|
|
90
92
|
# not a fallback to dash.parenthetical, nothing (ranges.md 2).
|
|
@@ -92,12 +94,14 @@ module Polytypo
|
|
|
92
94
|
|
|
93
95
|
if DashShared.spaced_style?(style)
|
|
94
96
|
# T1: a tight token may not become spaced across a digit run that has a far dash.
|
|
97
|
+
# T1/T2 read the walked flanks: ranges.md 3.2a makes cp[L']/cp[R'] what every
|
|
98
|
+
# shared guard sees once a closed-up symbol has been consumed.
|
|
95
99
|
next if token.lsp.zero? && token.rsp.zero? &&
|
|
96
|
-
DashShared.spacing_transition_blocked?(cp,
|
|
100
|
+
DashShared.spacing_transition_blocked?(cp, flanks.left, flanks.right)
|
|
97
101
|
|
|
98
102
|
# T2: the emitted U+0020 must not land where `spaces` (order 10) would delete it.
|
|
99
|
-
next if DashShared.strip_before_or_close_bracket?(
|
|
100
|
-
next if DashShared.open_bracket?(
|
|
103
|
+
next if DashShared.strip_before_or_close_bracket?(cp[flanks.right])
|
|
104
|
+
next if DashShared.open_bracket?(cp[flanks.left])
|
|
101
105
|
end
|
|
102
106
|
|
|
103
107
|
# ranges.md 3.3.1: never make an edit whose entire content is invisible. Try the
|
data/lib/polytypo/errors.rb
CHANGED
|
@@ -10,6 +10,9 @@ module Polytypo
|
|
|
10
10
|
CODE_MALFORMED_LOCALE_DATA = "POLYTYPO_MALFORMED_LOCALE_DATA"
|
|
11
11
|
CODE_RULE_CONTRACT = "POLYTYPO_RULE_CONTRACT"
|
|
12
12
|
CODE_MALFORMED_INPUT = "POLYTYPO_MALFORMED_INPUT"
|
|
13
|
+
# Spec 1.3.0. Deliberately general: every option added from 1.3.0 on shares this code, while
|
|
14
|
+
# `mode` and `dialect` keep their own because callers branch on them.
|
|
15
|
+
CODE_INVALID_OPTION = "POLYTYPO_INVALID_OPTION"
|
|
13
16
|
|
|
14
17
|
# The only error type Transform ever raises. One class is enough: the stable machine-readable
|
|
15
18
|
# +code+ is the contract, not the exception type or the message text.
|
|
@@ -2,6 +2,7 @@
|
|
|
2
2
|
|
|
3
3
|
require_relative "spans"
|
|
4
4
|
require_relative "../engine/edits"
|
|
5
|
+
require_relative "../engine/pipeline"
|
|
5
6
|
require_relative "../engine/registry"
|
|
6
7
|
|
|
7
8
|
module Polytypo
|
|
@@ -52,6 +53,22 @@ module Polytypo
|
|
|
52
53
|
out.concat(source_cp[cursor..])
|
|
53
54
|
out.pack("U*")
|
|
54
55
|
end
|
|
56
|
+
|
|
57
|
+
# run_over_spans, reporting instead of applying (analyze.md section 1). The span table
|
|
58
|
+
# supplies the origin map, so every change comes back in DOCUMENT coordinates -- analyze.md
|
|
59
|
+
# section 6 names a runtime that reports span-local offsets here as the mistake that passes
|
|
60
|
+
# every text-mode test.
|
|
61
|
+
def self.analyze_over_spans(source_cp, spans, plan, locale_data, ctx)
|
|
62
|
+
normalized = Spans.normalize_spans(spans)
|
|
63
|
+
return [] if normalized.empty?
|
|
64
|
+
|
|
65
|
+
Engine::Pipeline.run_rules_recording(
|
|
66
|
+
Spans.concatenate_spans(source_cp, normalized),
|
|
67
|
+
plan, locale_data, ctx,
|
|
68
|
+
Spans.origin_of_spans(normalized),
|
|
69
|
+
source_cp.length
|
|
70
|
+
) { |current, edits| Spans.filter_boundary_edits(current, edits, Spans.span_ranges_of(current)) }
|
|
71
|
+
end
|
|
55
72
|
end
|
|
56
73
|
end
|
|
57
74
|
end
|
data/lib/polytypo/modes/spans.rb
CHANGED
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
# frozen_string_literal: true
|
|
2
2
|
|
|
3
|
+
require_relative "../engine/origin"
|
|
3
4
|
require_relative "../engine/sentinels"
|
|
4
5
|
require_relative "../errors"
|
|
5
6
|
|
|
@@ -11,6 +12,9 @@ module Polytypo
|
|
|
11
12
|
module Spans
|
|
12
13
|
LINE_TERMINATORS = [0x0A, 0x0D, 0x0B, 0x0C, 0x85, 0x2028, 0x2029].freeze
|
|
13
14
|
|
|
15
|
+
# U+0020, the one emitted code point whose meaning is positional (modes.md 3.4, 5 item 2).
|
|
16
|
+
SPACE = 0x20
|
|
17
|
+
|
|
14
18
|
# A processable span, identified by its offsets in the original source, addressed as
|
|
15
19
|
# code-point indices (ARCHITECTURE.md section 4.2) -- Ruby String indices are already
|
|
16
20
|
# code-point indices for a UTF-8 String, no separate byte/char distinction to manage here.
|
|
@@ -68,6 +72,19 @@ module Polytypo
|
|
|
68
72
|
cp
|
|
69
73
|
end
|
|
70
74
|
|
|
75
|
+
# The origin map for concatenate_spans (analyze.md section 2): for every code point of the
|
|
76
|
+
# joined array, the code-point offset of the character it came from IN THE DOCUMENT, and
|
|
77
|
+
# NO_ORIGIN for the markers, which came from nowhere. A Span's bounds are already
|
|
78
|
+
# code-point offsets, so no coordinate conversion belongs here.
|
|
79
|
+
def self.origin_of_spans(spans)
|
|
80
|
+
origin = []
|
|
81
|
+
spans.each_with_index do |span, i|
|
|
82
|
+
origin << Engine::Origin::NO_ORIGIN unless i.zero?
|
|
83
|
+
origin.concat((span.start...span.end).to_a)
|
|
84
|
+
end
|
|
85
|
+
origin
|
|
86
|
+
end
|
|
87
|
+
|
|
71
88
|
# The span extents of the array as it stands. Recomputed after every rule, because applying
|
|
72
89
|
# edits shifts every index after the first one -- the markers themselves always survive,
|
|
73
90
|
# since no edit may contain one.
|
|
@@ -94,7 +111,13 @@ module Polytypo
|
|
|
94
111
|
# 1. No edit may contain a marker -- one that does is a bug, discarded rather than
|
|
95
112
|
# redistributed.
|
|
96
113
|
# 2. The edge-growth rule: an edit is discarded if it would place code points at an
|
|
97
|
-
# extremity of its span that were not there before.
|
|
114
|
+
# extremity of its span that were not there before. That sentence is the rule;
|
|
115
|
+
# "r > d" alone is an incorrect formalisation of it and misses r == d. dashes P3
|
|
116
|
+
# admits a run of THREE dashes, so "---" -> U+0020 en-dash U+0020 is 3 -> 3: the
|
|
117
|
+
# length test sees nothing while U+0020 lands on both extremities anyway. The second
|
|
118
|
+
# clause tests the CHARACTER, and only U+0020 needs testing -- it is the one code
|
|
119
|
+
# point any rule emits whose meaning comes from its position rather than from itself
|
|
120
|
+
# (modes.md 5 item 2).
|
|
98
121
|
def self.filter_boundary_edits(cp, edits, ranges)
|
|
99
122
|
edits.select do |edit|
|
|
100
123
|
next false if (edit.start...edit.end).any? { |i| Engine.marker?(cp[i]) }
|
|
@@ -104,7 +127,12 @@ module Polytypo
|
|
|
104
127
|
d = edit.end - edit.start
|
|
105
128
|
r = edit.replacement.length
|
|
106
129
|
span = span_containing(ranges, p)
|
|
107
|
-
|
|
130
|
+
next true unless span && (p == span.first || q == span.last)
|
|
131
|
+
next false if r > d
|
|
132
|
+
next false if r.positive? && p == span.first && edit.replacement[0] == SPACE && cp[p] != SPACE
|
|
133
|
+
next false if r.positive? && q == span.last && edit.replacement[r - 1] == SPACE && cp[q] != SPACE
|
|
134
|
+
|
|
135
|
+
true
|
|
108
136
|
end
|
|
109
137
|
end
|
|
110
138
|
|