polytypo 1.2.0 → 1.3.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (64) hide show
  1. checksums.yaml +4 -4
  2. data/README.md +33 -1
  3. data/lib/polytypo/data/VERSION +1 -1
  4. data/lib/polytypo/data/fixtures/cs.json +161 -0
  5. data/lib/polytypo/data/fixtures/de-CH.json +1 -1
  6. data/lib/polytypo/data/fixtures/de-DE.json +195 -6
  7. data/lib/polytypo/data/fixtures/el.json +1 -1
  8. data/lib/polytypo/data/fixtures/en-GB.json +12 -1
  9. data/lib/polytypo/data/fixtures/en-US.json +648 -1
  10. data/lib/polytypo/data/fixtures/es.json +193 -0
  11. data/lib/polytypo/data/fixtures/fi.json +1 -1
  12. data/lib/polytypo/data/fixtures/fr-CA.json +25 -1
  13. data/lib/polytypo/data/fixtures/fr.json +176 -1
  14. data/lib/polytypo/data/fixtures/it.json +161 -0
  15. data/lib/polytypo/data/fixtures/locale-resolution.json +76 -4
  16. data/lib/polytypo/data/fixtures/nl.json +121 -0
  17. data/lib/polytypo/data/fixtures/pl.json +137 -0
  18. data/lib/polytypo/data/fixtures/pt-BR.json +156 -0
  19. data/lib/polytypo/data/fixtures/pt-PT.json +156 -0
  20. data/lib/polytypo/data/fixtures/ru.json +23 -1
  21. data/lib/polytypo/data/fixtures/sv.json +1 -1
  22. data/lib/polytypo/data/fixtures/uk.json +153 -0
  23. data/lib/polytypo/data/locales/cs.json +90 -0
  24. data/lib/polytypo/data/locales/de-DE.json +7 -2
  25. data/lib/polytypo/data/locales/en-US.json +3 -3
  26. data/lib/polytypo/data/locales/es.json +111 -0
  27. data/lib/polytypo/data/locales/fr-CA.json +7 -1
  28. data/lib/polytypo/data/locales/fr.json +7 -1
  29. data/lib/polytypo/data/locales/it.json +95 -0
  30. data/lib/polytypo/data/locales/nl.json +84 -0
  31. data/lib/polytypo/data/locales/pl.json +96 -0
  32. data/lib/polytypo/data/locales/pt-BR.json +82 -0
  33. data/lib/polytypo/data/locales/pt-PT.json +84 -0
  34. data/lib/polytypo/data/locales/registry.json +23 -3
  35. data/lib/polytypo/data/locales/ru.json +2 -2
  36. data/lib/polytypo/data/locales/uk.json +130 -0
  37. data/lib/polytypo/data/rules/analyze.md +157 -0
  38. data/lib/polytypo/data/rules/apostrophe.md +432 -0
  39. data/lib/polytypo/data/rules/dashes.md +128 -37
  40. data/lib/polytypo/data/rules/ellipsis.md +271 -0
  41. data/lib/polytypo/data/rules/hyphen.md +353 -0
  42. data/lib/polytypo/data/rules/locale-resolution.md +239 -0
  43. data/lib/polytypo/data/rules/modes.md +1281 -0
  44. data/lib/polytypo/data/rules/nbsp.md +1157 -0
  45. data/lib/polytypo/data/rules/order.json +11 -11
  46. data/lib/polytypo/data/rules/pipeline-idempotency.md +605 -0
  47. data/lib/polytypo/data/rules/quotes.md +1324 -0
  48. data/lib/polytypo/data/rules/ranges.md +489 -0
  49. data/lib/polytypo/data/rules/spaces.md +649 -0
  50. data/lib/polytypo/data/rules/symbols.md +540 -0
  51. data/lib/polytypo/data/schema/fixtures.schema.json +18 -3
  52. data/lib/polytypo/engine/origin.rb +75 -0
  53. data/lib/polytypo/engine/pipeline.rb +72 -1
  54. data/lib/polytypo/engine/rules/dash_shared.rb +85 -3
  55. data/lib/polytypo/engine/rules/dashes.rb +4 -1
  56. data/lib/polytypo/engine/rules/nbsp.rb +43 -7
  57. data/lib/polytypo/engine/rules/ranges.rb +24 -20
  58. data/lib/polytypo/errors.rb +3 -0
  59. data/lib/polytypo/modes/runner.rb +17 -0
  60. data/lib/polytypo/modes/spans.rb +30 -2
  61. data/lib/polytypo/modes/yaml.rb +312 -0
  62. data/lib/polytypo/version.rb +1 -1
  63. data/lib/polytypo.rb +126 -15
  64. metadata +31 -1
@@ -0,0 +1,540 @@
1
+ # Rule: `symbols`
2
+
3
+ **Order:** 60. **Default:** on. **Modes:** text, html, markdown, yaml.
4
+ **Spec version:** 0.1.0.
5
+
6
+ ---
7
+
8
+ ## 1. Purpose
9
+
10
+ `symbols` performs two narrow, unrelated substitutions that share only their risk profile:
11
+ the ASCII stand-ins `(c)`, `(r)` and `(tm)` become U+00A9 (©), U+00AE (®) and U+2122 (™); and
12
+ the letter `x` standing between two numbers becomes U+00D7 (×), the multiplication sign. Both
13
+ are the sort of thing that looks trivial and then eats a codebase: `(c)` appears in function
14
+ calls and in Markdown footnote-ish constructs, and `x` appears in `0x1F`, in variable names,
15
+ in `1080x` shorthand and in the middle of every third English word. The rule is therefore
16
+ built out of literal-string matching plus hard adjacency guards, with no case folding
17
+ anywhere — the accepted spellings are enumerated explicitly, because
18
+ ARCHITECTURE.md §4.4 forbids `toLowerCase()` semantics that vary by host locale (Turkish
19
+ dotless ı would otherwise make `(TM)` behave differently in a Turkish process).
20
+
21
+ ---
22
+
23
+ ## 2. Locale data consumed
24
+
25
+ **None.** `order.json` declares `"localeData": []`. `©`, `®`, `™` and `×` are the same code
26
+ points in every locale polytypo supports, and the spacing around `×` is `nbsp`'s decision,
27
+ not this rule's.
28
+
29
+ ---
30
+
31
+ ## 3. Algorithm
32
+
33
+ Input is a code-point array `cp[0 … n-1]`. The rule performs one left-to-right scan; at each
34
+ index it tries the trademark table first, then the multiplication case.
35
+
36
+ ### 3.1 Character classes and tables
37
+
38
+ | Class | Members |
39
+ | --------------- | --------------------------------------------------------------- |
40
+ | `DIGIT` | U+0030–U+0039 only. **ASCII digits only** — see §7.1 |
41
+ | `LETTER` | general category `Lu`, `Ll`, `Lt`, `Lm`, `Lo`, `Mn`, `Mc`, `Me` |
42
+ | `ALNUM` | `LETTER` ∪ `DIGIT` |
43
+ | `SPACE` | U+0020 only |
44
+ | `NOBREAK-SPACE` | U+00A0, U+202F |
45
+ | `BREAK` | U+000A, U+000D, U+000B, U+000C, U+0085, U+2028, U+2029 |
46
+
47
+ **Trademark table.** Exhaustive and case-explicit; nothing else matches.
48
+
49
+ | Literal (code points) | Replacement |
50
+ | ------------------------------------ | ----------- |
51
+ | `(c)` = U+0028 U+0063 U+0029 | U+00A9 (©) |
52
+ | `(C)` = U+0028 U+0043 U+0029 | U+00A9 (©) |
53
+ | `(r)` = U+0028 U+0072 U+0029 | U+00AE (®) |
54
+ | `(R)` = U+0028 U+0052 U+0029 | U+00AE (®) |
55
+ | `(tm)` = U+0028 U+0074 U+006D U+0029 | U+2122 (™) |
56
+ | `(TM)` = U+0028 U+0054 U+004D U+0029 | U+2122 (™) |
57
+ | `(Tm)` = U+0028 U+0054 U+006D U+0029 | U+2122 (™) |
58
+ | `(tM)` = U+0028 U+0074 U+004D U+0029 | U+2122 (™) |
59
+
60
+ Comparison is code-point equality. No normalisation, no folding, no locale.
61
+
62
+ **Unicode version.** The general categories and case mappings this rule reads are those of the UCD version pinned in `spec/UNICODE` (`17.0`). The pin is normative for the **derived tables**, not for the host runtime — see [pipeline-idempotency.md](pipeline-idempotency.md) §6a, which also specifies the canary fixtures that make the pin detectable.
63
+
64
+ **Multiplication letters — `MUL-LETTER`.** Exactly four code points, enumerated, not
65
+ case-folded and not derived from any locale field:
66
+
67
+ | Code point | Glyph | |
68
+ | ---------- | ----- | ----------------------- |
69
+ | U+0078 | `x` | Latin small x |
70
+ | U+0058 | `X` | Latin capital X |
71
+ | U+0445 | `х` | **Cyrillic** small ha |
72
+ | U+0425 | `Х` | **Cyrillic** capital Ha |
73
+
74
+ **The Cyrillic pair is in this set unconditionally, and this rule reads no locale data.** That
75
+ is the right call rather than a convenient one: the risk is carried entirely by the _numeric
76
+ context_, not by the language. U+0445 standing between two ASCII digits is either a Russian
77
+ dimension typed on a Cyrillic layout (`Размер 5х4 см`) or keyboard-layout debris — there is no
78
+ third reading — and the sequence does not occur in German, Finnish or English text at all, so
79
+ gating it on locale would buy no safety and would add a `localeData` dependency that
80
+ `order.json` deliberately does not give this rule.
81
+
82
+ **Why this substitution in particular must be done by algorithm.** `5х4` and `5x4` are
83
+ **visually identical in every font**. This is the one conversion in the whole rule set that is
84
+ _invisible in a diff_, so the author cannot catch it by eye at the M4 dogfooding gate
85
+ (PLAN.md §8) — the review that is the ship criterion for everything else does not work here. A
86
+ human proof-reader will never find it; a machine finds it every time. That asymmetry is the
87
+ argument for doing it at all.
88
+
89
+ ### 3.2 Trademark substitution
90
+
91
+ At index `i`, if `cp[i]` is U+0028 `(`:
92
+
93
+ 1. Try each row of the trademark table in order of decreasing literal length (the 4-code-point
94
+ rows before the 3-code-point rows), comparing code point by code point against
95
+ `cp[i … i+len-1]`. If none matches, this branch does nothing.
96
+ 2. Let `m` be the matched span `[i, i+len)`. Compute
97
+ - `before = cp[i-1]` if `i > 0`, else `NONE`;
98
+ - `after = cp[i+len]` if `i+len < n`, else `NONE`.
99
+ 3. **Guard S1 — left adjacency. Applies to the `(c)` and `(r)` rows only.** If the matched
100
+ row is `(c)`, `(C)`, `(r)` or `(R)`, and `before` is in `ALNUM`, or is U+0029 `)`, or is
101
+ U+005D `]`, or is one of U+00A9 (©), U+00AE (®), U+2122 (™), emit nothing. This rejects
102
+ `f(c)`, `a[i](c)`, `foo(c)` — anything that looks like a call or an index.
103
+ **The `(tm)` rows are exempt from S1**: `Acme(tm)` with no space is how the trademark sign
104
+ is actually typed, and `(tm)` is not a plausible argument list — a single-letter argument
105
+ `c` or `r` is common, a two-letter argument spelled exactly `tm` immediately after an
106
+ identifier is not. This is the operator-accepted resolution of what was §7.2 option (b).
107
+ The three replacement code points are in the list for an idempotency reason argued in §5;
108
+ without them `(c)(r)` needs two runs to converge.
109
+ 4. **Guard S2 — right adjacency.** If `after` is in `ALNUM`, emit nothing. This rejects
110
+ `(r)evolution`, `(c)ompiler`, `(tm)odel` — the "optional first letter" idiom.
111
+ 5. **Guard S3 — no nesting.** If `before` is U+0028 `(`, emit nothing. `((c))` is almost
112
+ always deliberate ASCII art or code.
113
+ 6. Otherwise emit one edit replacing `cp[i … i+len-1]` with the single replacement code
114
+ point. Advance `i` past the matched span.
115
+
116
+ ### 3.3 Multiplication sign — the chain
117
+
118
+ The branch is specified as a scan of a **whole chain**, `DIGIT+ (X DIGIT+)+`, not as a test on
119
+ one letter between two numbers. That shape is what makes `5x4x3` convert **in a single pass**,
120
+ and it is deliberately not a patched version of the pairwise form: patching it would mean
121
+ converting one operator, leaving the next for a later pass, and reproducing exactly the
122
+ non-idempotency recorded in §7.10.
123
+
124
+ At index `i`, if `cp[i]` is in `DIGIT` and (`i = 0` or `cp[i-1]` is not in `DIGIT`) — the start
125
+ of a maximal digit run — attempt to read a chain:
126
+
127
+ 1. **Read the chain.** Let `a = i`. Consume the maximal `DIGIT` run. Then repeatedly attempt to
128
+ consume one **link**: an optional single code point from `SPACE` ∪ `NOBREAK-SPACE`, one
129
+ `MUL-LETTER`, an optional single code point from `SPACE` ∪ `NOBREAK-SPACE`, and a non-empty
130
+ `DIGIT` run.
131
+ **`NOBREAK-SPACE` is included deliberately**, matching the pairwise form this replaces: a
132
+ U+00A0 that a previous `nbsp` run placed between a number and its multiplication sign must
133
+ still read as spacing, or re-processing already-typeset text would stop recognising the
134
+ chain. Step 8 then carries that exact code point across the edit, and §6 case 10 pins it. Stop at the first position where a link cannot be completed.
135
+ Let the chain be `cp[a … b]`, with `m` links and `m + 1` digit runs.
136
+ 2. If `m = 0`, there is no chain here. Emit nothing and continue the scan from `b + 1`.
137
+ 3. **Guard M1 — spacing is symmetric and uniform.** For each link, the space-like code point
138
+ before its `MUL-LETTER` and the one after it must both be present or both be absent; and
139
+ **every link in the chain must agree**. Presence is what must match, not the exact code
140
+ point: a link spaced `U+00A0 × U+0020` is symmetric. Let `sp` be that common value, `0` or `1`. If any link is
141
+ asymmetric, or two links disagree, emit nothing for the entire chain.
142
+ `1080x` and `x 5` are not multiplications; `5x4 x 3` is ambiguous input and is declined
143
+ whole rather than half-converted.
144
+ 4. **Guard M2 — the chain must start at a number.** Let `before = cp[a-1]` (or `NONE`). If
145
+ `before` is in `LETTER`, emit nothing. (`H2x4`, `ax3`.)
146
+ 5. **Guard M3 — the chain must end at a number.** Let `after = cp[b+1]` (or `NONE`). If `after`
147
+ is in `LETTER`, emit nothing. This is what rejects hexadecimal: in `0x1F` the chain is `0x1`
148
+ and `after` is `F`.
149
+ **M2 and M3 apply to the outer boundaries of the chain, not to each link.** In the pairwise
150
+ form they applied per pair, which is what made a chain reject itself — the letter on the far
151
+ side of the middle digit run was another `MUL-LETTER`.
152
+ 6. **Guard M4 — hexadecimal literal.** If `sp = 0`, the **first** link's letter is U+0078
153
+ (Latin lowercase — a hex literal is never written with Cyrillic, so this guard stays
154
+ Latin-only), and the first digit run is the single code point U+0030 (`0`), emit nothing.
155
+ This catches `0x10`, `0x24` and every hex literal whose digits happen to be decimal, which
156
+ M3 alone misses.
157
+ 7. **There is no M5.** The former chain guard existed only to decline the shape this branch now
158
+ converts, and it was **removed rather than extended** — see §7.3. Its stated redundancy with
159
+ M2/M3 was correct for the pairwise form and is the reason the pairwise form could never have
160
+ been patched into working: M2/M3 _are_ what rejected a chain.
161
+ 8. Otherwise emit **one edit per link**. For link `k` whose letter sits at index `j`, replace
162
+ `cp[j-sp … j+sp]` with
163
+ - `sp = 0` → the single code point U+00D7;
164
+ - `sp = 1` → the three code points `cp[j-1]` U+00D7 `cp[j+1]`, i.e. **each side keeps the
165
+ exact space code point it had in the input**, so a U+00A0 placed by a previous `nbsp` run
166
+ is preserved rather than downgraded to U+0020.
167
+
168
+ The edits cannot overlap: consecutive links are separated by a digit run of at least one
169
+ code point. If a computed replacement is identical to the span it replaces, emit nothing for
170
+ that link. Continue the scan from `b + 1`.
171
+
172
+ **Mixed alphabets convert.** `5x4х3` — one Latin `x`, one Cyrillic `х` — is a single chain and
173
+ both links convert. The four members of `MUL-LETTER` are interchangeable here, and no guard
174
+ distinguishes them (M4 is the sole exception and it inspects only the first link). This is the
175
+ right answer rather than a permissive one: mixed input is precisely the keyboard-layout debris
176
+ the Cyrillic addition was made for (§3.1), the numeric context is identical whichever letter
177
+ appears, and declining on mixture would make the rule's behaviour depend on **which layout the
178
+ author's finger slipped to** — the least predictable criterion available, and one no author
179
+ could ever discover from the output, since the two letters are visually identical.
180
+
181
+ **An already-converted operator ends a chain.** U+00D7 is not in `MUL-LETTER`, so `5×4x3` reads
182
+ as the digit run `5`, then U+00D7 (no chain), then the chain `4x3`, which converts. The result
183
+ is `5×4×3` either way; sub-chains do not need to be joined.
184
+
185
+ ### 3.4 Plus-minus
186
+
187
+ The literal three-code-point sequence **`+/-`** — U+002B, U+002F, U+002D — becomes U+00B1 (`±`).
188
+
189
+ **Only `+/-`. The bare sequence `+-` is never converted**, and that asymmetry is the whole
190
+ design. `+/-` is unambiguous: no language, format or notation uses it for anything else. `+-`
191
+ is not — it occurs in diff and patch listings, in ASCII table rules and borders, in regular
192
+ expression character classes such as `[+-]`, and as two adjacent operators in source code. None
193
+ of those is protected in `text` mode, where nothing is skipped. Converting `+-` would be a false
194
+ positive in exactly the content this package is used on. See §7.11 so it is not added later "for
195
+ symmetry".
196
+
197
+ At index `i`, if `cp[i]` is U+002B and `cp[i+1]` is U+002F and `cp[i+2]` is U+002D:
198
+
199
+ 1. **Guard F1 — not a character class.** Let `before = cp[i-1]` (or `NONE`). If `before` is
200
+ U+005B `[`, emit nothing. `[+/-]` is a regular expression, and this is the one code-like
201
+ context that survives into prose about code.
202
+ 2. **Guard F2 — numeric context.** Let `j = i + 3`. If `cp[j]` is U+0020, advance `j` by one.
203
+ If `cp[j]` is not in `DIGIT`, emit nothing.
204
+ 3. Otherwise emit one edit replacing `cp[i … i+2]` with the single code point U+00B1. Any space
205
+ between the sign and the digit is left exactly as it was.
206
+
207
+ **Why a following digit is required.** This was the one genuinely open choice, and I took the
208
+ narrow side. The permissive form — convert `+/-` wherever it appears — is more useful in the
209
+ rare standalone reading ("плюс-минус", "the error is +/-"), and it is wrong in a case that is
210
+ not rare at all: **prose that names the characters rather than using them**. A changelog reading
211
+ `lines marked +/- were edited`, or a legend `+/- indicates added and removed rows`, becomes
212
+ `lines marked ± were edited` — a silent corruption of a sentence that was about the ASCII
213
+ symbols themselves. Requiring a digit declines every one of those, because what follows is a
214
+ letter.
215
+
216
+ The cost is precisely measurable and small: standalone `±` in running prose is uncommon, it is
217
+ usually spelled out in words when it is meant, and an author who wants it can type it. The
218
+ benefit is that the guard is _structural_ — it keys on the numeric context, exactly as the
219
+ multiplication branch does — rather than on a list of contexts someone has to remember to
220
+ extend. One intervening U+0020 is allowed so that both `+/-5` and `+/- 5` work, which covers
221
+ the way the sequence is actually written.
222
+
223
+ ### 3.5 Scan order
224
+
225
+ The three branches key on different code points — U+0028, a `DIGIT`, and U+002B — so no two can
226
+ match at the same index. The scan visits each index once; on a successful edit it continues from
227
+ the index after the matched span, so a replacement can never be re-examined within the same
228
+ pass.
229
+
230
+ ---
231
+
232
+ ## 4. Must not touch
233
+
234
+ **Scope.** Per [pipeline-idempotency.md](pipeline-idempotency.md) §5.2 each bullet is **[P]** —
235
+ a guarantee of `transform` as a whole — or **[R]** — true of this rule in isolation but capable
236
+ of being falsified by another rule, which is then named.
237
+
238
+ - **[P] `(s)`, `(a)`, `(e)`, `(i)`, `(n)`** and every other parenthesised single letter. The
239
+ table is exhaustive.
240
+ - **[P] `(c)` and `(r)` in a call or index position:** `f(c)`, `arr[i](c)`. Guard S1. Note that
241
+ `(tm)` is **not** protected this way and `f(tm)` does become `f™` — deliberate, §3.2 step 3.
242
+ - **[P] `(r)evolution`, `(c)ompiler`, `(s)he`.** Guard S2.
243
+ - **[P] Existing ©, ®, ™, ×.** Not candidates; the rule reads `(` and `x`/`X` only.
244
+ - **[P] Cyrillic `х` outside a numeric context.** `хорошо`, `их`, `по-моему х` — M2/M3 decline
245
+ a letter neighbour in either alphabet, and step 2 requires an ASCII digit on each side.
246
+ - **[P] `x` as a word or a variable:** `x = 5`, `the x axis`, `Malcolm X`. Guard M1 or step 2
247
+ (no digit neighbour).
248
+ - **[P] `1080x`, `x264`, `2x` alone.** Guard M1 (asymmetric spacing) or step 2.
249
+ - **[P] Hexadecimal literals `0x1F`, `0xFF`, `0x10`.** Guards M3 and M4.
250
+ - **[P] The bare sequence `+-`.** Never converted, in any context — §3.4 and §7.11.
251
+ - **[P] `[+/-]` in a regular expression.** Guard F1.
252
+ - **[P] `+/-` followed by a letter.** `+/- indicates added rows` keeps its ASCII sequence.
253
+ Guard F2.
254
+ - **[R] Spacing.** The rule never inserts or deletes a space; it only carries the existing one
255
+ across the edit (§3.3 step 8).
256
+ - **[P] Anything inside a skipped region.** `(c)` in a code span and `0x1F` in a fenced block are
257
+ removed by the mode adapter before this rule runs.
258
+ - **[P] Emoticons and ASCII art.** `(x)`, `:-)`, `(^_^)` contain no table entry and no numeral
259
+ context.
260
+
261
+ ---
262
+
263
+ ## 5. Idempotency argument
264
+
265
+ **Trademark branch.** Every edit replaces a span beginning with U+0028 by a single code
266
+ point that is not U+0028, so a second run never re-enters the branch at that position, and no
267
+ new U+0028 is ever created (the edit only removes them). The classification of a surviving
268
+ `(`-candidate depends on `before`, `after` and the literal span. `after` can only have been
269
+ changed if the _following_ span was edited — impossible, because the following span would
270
+ then have to start at `after`'s position with U+0028, and a candidate whose `after` is U+0028
271
+ fails nothing and is edited itself, removing the adjacency. `before` can have been changed by
272
+ an immediately preceding edit: `(c)(r)`. On run 1 the second match has `before` = U+0029 and
273
+ is rejected by S1; after the first edit the array reads `©(r)`, and without the amendment
274
+ `before` = U+00A9 would satisfy S1 on run 2 and produce a new edit — a two-run convergence,
275
+ i.e. a violated invariant. **S1 lists U+00A9, U+00AE and U+2122 for exactly this reason**, so
276
+ `(c)(r)` converges to `©(r)` on run 1 and is stable.
277
+
278
+ The `(tm)` exemption from S1 does not weaken this. A `(tm)` match is admitted regardless of
279
+ `before`, so it is edited on run 1 whatever its left neighbour is; there is no left-neighbour
280
+ condition left to flip on run 2. `(c)(tm)` converges to `©™` in a single run, and `(tm)(c)`
281
+ to `™(c)` — the trailing `(c)` being rejected by S1 on both runs, on `)` first and on U+2122
282
+ after.
283
+
284
+ **Multiplication branch.** Every edit replaces `[space] X [space]` with `[space] × [space]`
285
+ (or a bare multiplication letter with `×`). U+00D7 is not in `MUL-LETTER`, so no edited position
286
+ is a candidate again — and because the branch converts a **whole chain in one pass**, no
287
+ _unedited_ `MUL-LETTER` is left inside a chain either. A second run reading `5×4×3` finds digit
288
+ runs separated by U+00D7, cannot complete a single link, and emits nothing.
289
+
290
+ This is the property the chain form buys, and it is worth stating as the contrast it is: the
291
+ pairwise form this replaced could only ever convert one operator at a time, so `5x4x3` would
292
+ have gone to `5×4x3` and then to `5×4×3` — two passes, two different answers, which is the
293
+ defect §7.10 records in another implementation. Idempotency here is **by construction of the
294
+ scan**, not by a guard that has to be kept symmetric.
295
+
296
+ The surrounding digits are untouched, so `before` and `after` are unchanged for every other
297
+ chain in the text; and chains are separated by at least one non-digit, non-link code point, so
298
+ one chain's edits can never alter another's boundaries.
299
+
300
+ **Plus-minus branch.** The edit replaces three code points beginning with U+002B by one U+00B1,
301
+ which is not U+002B, so the position is not a candidate again. No new U+002B is created — the
302
+ edit only removes one — so no neighbouring candidate can be created. F1 reads `cp[i-1]` and F2
303
+ reads forward past the match; neither position can have been changed by another `+/-` edit,
304
+ since two matches cannot overlap and an edit never writes a digit or a `[`.
305
+
306
+ Both branches therefore leave no candidate at an edited position and no changed
307
+ classification at a surviving one, so `T(T(x)) = T(x)`.
308
+
309
+ **What had to be fixed**, summarised: (a) Guard S1 had to be extended to the three
310
+ replacement code points, or `(c)(r)` needs two passes; (b) the multiplication branch had to
311
+ carry the existing space code point across the edit rather than emitting U+0020, or text that
312
+ `nbsp` had already processed would oscillate between U+00A0 and U+0020 on alternate runs;
313
+ (c) _(superseded.)_ This clause required "the chain guard M5 to be symmetric, rejecting on
314
+ _either_ side, or `2x3x4` would converge to `2×3x4` on run 1 and `2×3×4` on run 2". **There is
315
+ no M5** — §3.3 step 7 removed it rather than extending it, because M5's redundancy with M2/M3
316
+ was precisely what made the pairwise form reject chains. The defect it describes is now
317
+ prevented by the chain scan itself: `2x3x4` converts in one pass, so there is no run 2 for it to
318
+ converge on. Kept as a record of what the pairwise form required, and marked, rather than
319
+ deleted silently — the sentence is the only surviving trace of why M5 could not simply be
320
+ patched.
321
+
322
+ ---
323
+
324
+ ### Composition obligation
325
+
326
+ Per [pipeline-idempotency.md](pipeline-idempotency.md) §5. This rule is **R₇**; the obligation
327
+ runs against `spaces`, `ellipsis`, `dashes`, `hyphen`, `quotes` and `apostrophe`.
328
+
329
+ **What this rule emits.** U+00A9, U+00AE or U+2122 replacing a three- or four-code-point span
330
+ beginning with U+0028; and U+00D7 replacing one `MUL-LETTER`, carrying the adjacent space code
331
+ points across unchanged (§3.3 step 8). It never emits a space, a dot, a dash, a quotation mark
332
+ or a letter.
333
+
334
+ **Against `I₁` (`spaces`).** Discharged. No U+0020 is emitted. The trademark replacement
335
+ _shortens_ a span, which could in principle bring two code points together — but both
336
+ neighbours are guarded to be non-`ALNUM` on the left and non-`ALNUM` on the right, and neither
337
+ guard admits a U+0020 on both sides simultaneously in a way that creates a run: the span
338
+ replaced is bounded by `(` and `)`, so a space on each side of it stays one space on each side
339
+ of the resulting sign. `×` is not in `STRIP-BEFORE`, so a preserved space before it is not
340
+ something `spaces` would delete.
341
+
342
+ **Against `I₂` (`ellipsis`).** Discharged: no dots emitted, and the deleted `(`…`)` span cannot
343
+ separate two dot runs, because a `DOTLIKE` run adjacent to `(` on one side and `)` on the other
344
+ is not brought into contact — the sign replaces them.
345
+
346
+ **Against `I₃` (`dashes`).** Discharged: no dash is emitted and no spacing changes. A dash
347
+ token adjacent to a replaced span sees its `cp[L]`/`cp[R]` change from `(`/`)` to a sign; both
348
+ are ordinary content to `dashes` and neither is in `DASH`, `INERT-DASH`, `DIGIT` or a space
349
+ class, so no verdict moves.
350
+
351
+ **Against `I₄` (`hyphen`).** Discharged: the replaced span is bounded by non-`ALNUM` code
352
+ points and the emitted signs are not in `WORDISH`, so no listed hyphen form's word boundary
353
+ changes.
354
+
355
+ **Against `I₅`/`I₆` (`quotes`, `apostrophe`).** Discharged: this rule emits nothing in
356
+ `STRAIGHT`, and the signs it does emit are not in `SPACELIKE` or `ALNUM`, so a surviving
357
+ straight mark adjacent to one keeps every capability it had — and by `quotes` Claim 3, which
358
+ needs only that capabilities do not _increase_, that is sufficient.
359
+
360
+ ---
361
+
362
+ ## 6. Worked examples
363
+
364
+ `␣` = U+0020, `⍽` = U+00A0, `⟶` = no change. Locale-independent.
365
+
366
+ **A standing note for anyone adding a row here.** These rows become conformance fixtures
367
+ verbatim, so each one must exercise **only** the rule it documents. Choose the surrounding words
368
+ so that no _other_ rule fires: no unit or unit-like token (`mm`, `cm`, `см`, `km`, `%`, `°C`),
369
+ which `nbsp.beforeUnits` binds to a preceding number with U+00A0; no abbreviation ending in a
370
+ full stop, which `nbsp.beforeNumber` / `beforeWord` bind; and no short function word, which
371
+ `nbsp.afterShortWords` binds. A row reading `Tolerance ±5 mm` teaches a reader that `symbols`
372
+ produces the space it does not produce — the true pipeline output carries a U+00A0 that belongs
373
+ to `nbsp`. Prose elsewhere in this document may use realistic units freely; only §6 is
374
+ constrained.
375
+
376
+ | # | Input | Output | Why |
377
+ | --- | ---------------------------------------- | -------------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
378
+ | 1 | `Copyright (c) 2026 Iurii Rogulia` | `Copyright © 2026 Iurii Rogulia` | table hit, S1/S2/S3 pass |
379
+ | 2 | `Acme(tm) and Acme (R)` | `Acme™ and Acme ®` | `(tm)` is exempt from S1 (§3.2 step 3), so the letter `e` at `before` does not reject it; `(R)` passes S1 because `before` is a space. S2 passes for both (`after` is a space and end-of-text) |
380
+ | 2b | `f(c) and f(tm)` | `f(c) and f™` | the S1 asymmetry, stated as one case: S1 applies to the `(c)`/`(r)` rows and rejects the first (`before` is `f`); the `(tm)` rows are exempt and the second converts. Deliberate — §3.2 step 3, §7.2 |
381
+ | 3 | `The (r)evolution will not be televised` | ⟶ | S2: `after` is `e` |
382
+ | 4 | `f(c) returns c` | ⟶ | S1: `before` is `f` |
383
+ | 5 | `A 3x5 card, 10 x 20 grid` | `A 3×5 card, 10 × 20 grid` | tight and symmetric-spaced forms |
384
+ | 5a | `Размер 5х4` | `Размер 5×4` | **Cyrillic U+0445 between ASCII digits.** Identical in appearance to case 5 and invisible in a diff — §3.1 |
385
+ | 5b | `Размер 5Х4` | `Размер 5×4` | Cyrillic capital U+0425 |
386
+ | 5c | `хорошо и их` | ⟶ | Cyrillic `х` with letter neighbours: step 2 requires an ASCII digit on each side |
387
+ | 5d | `5х4х3` | `5×4×3` | an all-Cyrillic chain, both links converted in **one pass** (§3.3) |
388
+ | 5e | `A 2x3x4 box` | `A 2×3×4 box` | the Latin chain. Previously declined in its entirety |
389
+ | 5f | `5x4х3` | `5×4×3` | **mixed alphabets** — one Latin `x`, one Cyrillic `х`, one chain, both links convert |
390
+ | 5g | `Стол 120х80х75` | `Стол 120×80×75` | three-link chain, multi-digit runs |
391
+ | 5h | `10 x 20 x 30` | `10 × 20 × 30` | spaced chain; `sp = 1` uniformly, and each side keeps its own space code point |
392
+ | 5i | `5x4 x 3` | ⟶ | **M1**: link 1 is tight, link 2 is spaced. Ambiguous input, declined whole rather than half-converted |
393
+ | 5j | `5×4x3` | `5×4×3` | an already-converted operator ends a chain; the remaining sub-chain converts |
394
+ | 5k | `Tolerance +/-5 points` | `Tolerance ±5 points` | `+/-` with an immediately following digit |
395
+ | 5l | `Tolerance +/- 5 points` | `Tolerance ± 5 points` | one intervening U+0020 is allowed; the space itself is untouched |
396
+ | 5m | `Погрешность 5+/-3` | `Погрешность 5±3` | a digit on the left is not an obstacle — F1 rejects only `[` |
397
+ | 5n | `lines marked +/- were edited` | ⟶ | **F2**: what follows is a letter. Prose _about_ the characters is not converted |
398
+ | 5o | `match [+/-] once` | ⟶ | **F1**: a regular-expression character class |
399
+ | 5p | `Range 5+-3` | ⟶ | bare `+-` is never converted, in any context (§3.4, §7.11) |
400
+ | 6 | `Colour 0x1F, mask 0xFF, offset 0x10` | ⟶ | M3 (`0x1F`, `0xFF`) and M4 (`0x10`) |
401
+ | 7 | `Resolution 1920x1080` | `Resolution 1920×1080` | digits both sides, no letters adjacent |
402
+ | 8 | `Let x = 5 and solve for x` | ⟶ | no digit neighbour |
403
+ | 9 | `0x1F, 0xFF, 0x10 stay` | ⟶ | M3 rejects the first two (`after` is a letter), M4 the third |
404
+ | 10 | `10⍽×␣20` | ⟶ | already converted, and the U+00A0 is preserved (§3.3 step 8) |
405
+ | 11 | `((c))` | ⟶ | S3 |
406
+ | 12 | `Version 1080x` | ⟶ | M1, asymmetric |
407
+ | 13 | `© 2026, ® and ™` | ⟶ | no candidates at all |
408
+ | 14 | `2*3` | ⟶ | `*` is never a multiplication sign here — §7.8 |
409
+
410
+ Cases 3, 4, 5c, 5i, 5n, 5o, 5p, 6, 8, 9, 10, 11, 12, 13 and 14 are "no change" cases.
411
+
412
+ ---
413
+
414
+ ## 7. Open questions
415
+
416
+ 1. **`DIGIT` is ASCII-only**, as in `dashes`. `١٠ x ٢٠` is not recognised. Deliberate, for the
417
+ same portability reason (no Unicode `Nd` value table required in five runtimes).
418
+ 2. _(Settled.)_ S1 used to apply to all three signs, which missed `Acme(tm)` — the commonest
419
+ real spelling. Option (b), restricting S1 to `(c)`/`(r)`, was accepted by the operator and
420
+ is written into §3.2 step 3. The residual cost is that `f(tm)` in prose becomes `f™`; in
421
+ `html`/`markdown` mode a real code span is removed by the mode adapter before this rule
422
+ runs, so the exposure is `text` mode only. Fixtures: §6 cases 2 and 2b.
423
+ 3. _(Settled — implemented.)_ Chains convert. §3.3 is a scan of `DIGIT+ (X DIGIT+)+` with
424
+ M2/M3 applied to the chain's outer boundaries, which is the respecification this item asked
425
+ for; M5 was **removed rather than patched**, because M5's redundancy with M2/M3 was exactly
426
+ what made the pairwise form reject chains. `5x4x3`, `5х4х3` and the mixed `5x4х3` all
427
+ convert in one pass.
428
+ 4. **`Pixel 4x5` and similar product names are converted** to `Pixel 4×5`. No guard can
429
+ distinguish that from a genuine dimension. Accepted false positive; it should be checked
430
+ against real content at the M4 gate.
431
+ 5. **No `(p)` → `℗`.** `order.json`'s summary for this rule names the copyright, registered and
432
+ trademark signs, the multiplication sign, and — since spec 0.1.0 — plus-minus. Adding more is
433
+ a spec change, not an implementation detail.
434
+ 6. **Vulgar fractions are refused, in every form.** Neither the precomposed characters (`½`,
435
+ `¼`, `⅜`) nor the U+2044 fraction-slash construction. This is a **decision, not an open
436
+ question**, and it is the only candidate in the review refused on grounds of principle rather
437
+ than of risk. The argument, in the order that decided it:
438
+
439
+ - **The Unicode inventory is incomplete by construction, so the rule cannot be consistent
440
+ even when it is correct.** `½ ¼ ¾ ⅓ ⅔ ⅕ ⅛` exist; `3/16`, `5/32`, `1/12`, `7/10` do not.
441
+ So a paragraph containing `1/2 дюйма` and `3/16 дюйма` comes out as `½ дюйма` beside
442
+ `3/16 дюйма` — two notations for the same kind of quantity, in adjacent sentences,
443
+ **where the input had one**. The rule leaves the text less internally consistent than it
444
+ found it. That is not a rule that is right most of the time; it is a rule that is wrong
445
+ _when it fires correctly_, and no guard can fix it, because the defect is in the existence
446
+ of the mapping table rather than in its edges.
447
+ - **In `text` mode nothing is skipped, and a path is ordinary text.** `/usr/1/2/bin` becomes
448
+ `/usr/½/bin`. `modes.md` skips code and URLs in `html` and `markdown`, but `text` mode has
449
+ no adapter, and a substitution whose safety depends on the mode is not one this spec should
450
+ carry.
451
+ - **It does not survive downstream normalisation.** U+00BD has a **compatibility
452
+ decomposition** to `1` U+2044 `2`, so any consumer applying NFKC — a search indexer, a
453
+ database collation, a template engine — silently undoes the work. ARCHITECTURE.md §4.3
454
+ forbids _us_ from normalising; it cannot forbid anyone downstream. A conversion that a
455
+ common downstream pass reverses is not worth any false-positive budget at all.
456
+ - **It is invisible in a diff**, so the M4 gate cannot catch it — the same class as the
457
+ Cyrillic `х` (§3.1). But the Cyrillic case is one where the machine is right every time and
458
+ the human cannot see it; this is one where the machine introduces the inconsistency and the
459
+ human still cannot see it. The asymmetry is what makes the same property an argument _for_
460
+ one and _against_ the other.
461
+ - **No authority asks for it.** Chicago, the Unicode Standard and Мильчин all describe how a
462
+ fraction should _look_ — raised numerator, proper solidus, correct kerning — which is the
463
+ job of a font's `frac` feature or a typesetter, not of a character substitution. PLAN.md
464
+ §6.1 settles disagreements by citation, and there is no citation to be had.
465
+ - The ambiguity cases — `1/2/2026`, `and/or`, a URL path, `5 1/2` versus `5/2` — are real but
466
+ were **not decisive**. Guards could be written for them. Nothing can be written for the
467
+ first point.
468
+
469
+ **The single condition under which this reopens:** a rendering context where polytypo knows
470
+ the target font provides `frac`, so a fraction could be marked up rather than substituted.
471
+ That is a different product (PLAN.md §9), it requires emitting markup, which §7.9 refuses on
472
+ architectural grounds, and it is **not authorized**.
473
+
474
+ 7. _(Settled — implemented.)_ Cyrillic U+0445/U+0425 are members of `MUL-LETTER` (§3.1). The
475
+ premise of the original question was wrong: this is **not** locale-dependent data. The
476
+ numeric context carries the whole risk, the sequence does not occur in the other v1 locales,
477
+ and no schema field or `localeData` entry is needed. Confirmed empirically against Lebedev's
478
+ live service, which performs the same substitution.
479
+ 8. **Decided refusals.** Recorded so they are not re-litigated. Each was checked against
480
+ Lebedev's live service (Kovodstvo §62, Typograf) rather than reconstructed from memory:
481
+ - **`*` → `×`.** Refused, and Lebedev does not do it either. `2*3` is indistinguishable from
482
+ Markdown emphasis, a shell glob, a footnote marker and multiplication in source. No
483
+ numeric-context guard separates them, because the numeric context is present in all four
484
+ readings.
485
+ - **Arrows (`->`, `=>`) and comparison operators (`!=`, `<=`, `>=`).** Refused: source code
486
+ far more often than prose in the content this package targets, and `->` collides with
487
+ `dashes`.
488
+ - **Degree insertion** (`5 C` → `5 °C`), **digit-group binding** (`100 000`), **`m2` → `м²`**,
489
+ **currency substitution** (`1 руб.` → `1 ₽`), **ordinal repair** (`10-ый` → `10-й`).
490
+ Refused: each inserts or rewrites _content_ rather than normalising typography, and several
491
+ are locale-specific editorial conventions rather than typographic cleanup. Digit-group binding
492
+ would belong to `nbsp` if it were ever wanted, not here.
493
+ - **ISO date reformatting.** Refused outright. `dashes` goes to considerable trouble to leave
494
+ `2026-08-15` byte-identical (§3.2 step 7 there); rewriting it elsewhere would be incoherent.
495
+ - **Greek compatibility punctuation.** No mapping U+037E → U+003B (ερωτηματικό) and none
496
+ U+0387 → U+00B7 (άνω τελεία), in either direction. Both compatibility characters decompose
497
+ **canonically**, so rewriting one is normalisation under a different name — forbidden by
498
+ ARCHITECTURE.md §4.3 — and redundant besides, since any downstream NFC pass performs it.
499
+ The converse mapping, "spell the Greek question mark U+037E because it is the Greek one",
500
+ is refused on the Unicode Standard's own advice (ch. 7 §7.2.1: their use "is not generally
501
+ encouraged"). This rule therefore has no Greek branch at all, and `spaces.md` §3.5 carries
502
+ the full argument, including why the ambiguity costs the pipeline nothing.
503
+ - **Any insertion of markup into the result** — see §7.9.
504
+ 9. **No markup is ever inserted, by this rule or any other.** Lebedev's service wraps its
505
+ output: `212-85-06` becomes `<nobr class="phone">…</nobr>`, and `(r)` becomes
506
+ `<sup class="reg">®</sup>`. polytypo will not, and the reason is architectural rather than
507
+ aesthetic: ARCHITECTURE.md §2 puts the rule engine at L1, which knows nothing about HTML, and
508
+ `modes.md` §4 guarantees the output is the input with disjoint substring replacements and
509
+ nothing else. A rule that emitted a tag would be unimplementable in `text` mode and would
510
+ break the round-trip guarantee in the other two.
511
+ It is also, empirically, that approach's worst failure: `Дата 2026-08-15 отчёт` comes back
512
+ with the ISO date wrapped as a telephone number. A rule that decides "this is a phone number"
513
+ and then _encodes that decision in the document_ converts a silent miss into permanent
514
+ damage — which is the strongest available argument for `dashes`' cluster guard leaving
515
+ digit-dash chains untouched instead of trying to classify them.
516
+ 10. **Lebedev's own output is not idempotent**, and this is an observation from the live
517
+ service rather than a claim about it: `5х4х3` → `5×4х3` → `5×4×3`. Two passes give two
518
+ different answers, and a third gives a third. For content that is re-processed on every
519
+ save — the CMS case PLAN.md §1 names — that is a document that never converges.
520
+ This is the most concrete evidence available for why PLAN.md §3.4 makes
521
+ `transform(transform(x)) == transform(x)` a **release blocker** rather than a quality goal,
522
+ and it belongs in the README comparison table that PLAN.md §8 M5 requires.
523
+
524
+ Our behaviour on the same input is now the direct contrast: `5х4х3` → `5×4×3` in **one
525
+ pass**, and the second pass is a no-op (§6 case 5d). That is not because we were more
526
+ careful with the same design — it is because the branch is specified as a chain scan rather
527
+ than as a pairwise test (§3.3), and a pairwise test _cannot_ be made idempotent here by any
528
+ amount of guarding. The comparison is therefore about the shape of the specification, not
529
+ about implementation quality, which is the more useful thing to say in a README.
530
+
531
+ 11. **`+-` is not converted and must not be added later.** Only the literal `+/-` becomes U+00B1
532
+ (§3.4). The temptation to add `+-` "for symmetry" is exactly why this entry exists: `+-`
533
+ occurs in diff and patch listings, in ASCII table borders, in regex character classes like
534
+ `[+-]`, and as two adjacent operators in source code — none of which is skipped in `text`
535
+ mode. `+/-` has no such collisions. The asymmetry is the point, not an oversight.
536
+ 12. **`+/-` requires a following digit** (guard F2), so standalone `±` in prose is a miss:
537
+ `плюс-минус`, or `the tolerance is +/-`, keeps its ASCII form. The alternative was refused
538
+ because prose that _names_ the characters — `lines marked +/- were edited` — is common in
539
+ changelogs and documentation and would be silently corrupted. If real content shows the miss
540
+ matters, the safe widening is a short list of following words, not the removal of F2.
@@ -2,7 +2,7 @@
2
2
  "$schema": "https://json-schema.org/draft/2020-12/schema",
3
3
  "$id": "https://polytypo.org/schema/fixtures.schema.json",
4
4
  "title": "polytypo conformance fixtures",
5
- "description": "Executable form of the rule semantics. Every case is also an idempotency case: a runner must additionally assert transform(out) == out.",
5
+ "description": "Executable form of the rule semantics. Every case is also an idempotency case: a runner must additionally assert transform(out) == out, passing the case's own options (mode, dialect, keys, rules, narrowNbsp) on that second call exactly as on the first — a case is a fixed point under its own options, not under the defaults.",
6
6
  "type": "object",
7
7
  "additionalProperties": false,
8
8
  "required": ["spec", "locale", "cases"],
@@ -24,6 +24,11 @@
24
24
  "if": { "properties": { "mode": { "const": "markdown" } }, "required": ["mode"] },
25
25
  "then": { "required": ["dialect"] },
26
26
  "else": { "not": { "required": ["dialect"] } }
27
+ },
28
+ {
29
+ "if": { "properties": { "mode": { "const": "yaml" } }, "required": ["mode"] },
30
+ "then": { "required": ["keys"] },
31
+ "else": { "not": { "required": ["keys"] } }
27
32
  }
28
33
  ],
29
34
  "oneOf": [{ "required": ["out"] }, { "required": ["throws"] }],
@@ -47,11 +52,16 @@
47
52
  ],
48
53
  "description": "The rule under test. Required so a runtime can report partial conformance honestly."
49
54
  },
50
- "mode": { "enum": ["text", "html", "markdown"] },
55
+ "mode": { "enum": ["text", "html", "markdown", "yaml"] },
51
56
  "dialect": {
52
57
  "enum": ["commonmark", "mdx"],
53
58
  "description": "Required when mode is \"markdown\" — spec/rules/modes.md §3.7.1 forbids detection, so a fixture must name the dialect exactly as a caller would. Meaningless for the other modes."
54
59
  },
60
+ "keys": {
61
+ "type": "array",
62
+ "items": { "type": "string" },
63
+ "description": "Required when mode is \"yaml\" — spec/rules/modes.md §3.8.2 gives the option no default, so a fixture must name the processable keys exactly as a caller would. An empty array is legal and processes nothing. Meaningless for the other modes."
64
+ },
55
65
  "in": { "type": "string" },
56
66
  "out": { "type": "string" },
57
67
  "throws": {
@@ -62,7 +72,8 @@
62
72
  "POLYTYPO_UNKNOWN_RULE",
63
73
  "POLYTYPO_MALFORMED_LOCALE_DATA",
64
74
  "POLYTYPO_RULE_CONTRACT",
65
- "POLYTYPO_MALFORMED_INPUT"
75
+ "POLYTYPO_MALFORMED_INPUT",
76
+ "POLYTYPO_INVALID_OPTION"
66
77
  ],
67
78
  "description": "Expected error code instead of an output string. Messages are not part of the contract; codes are."
68
79
  },
@@ -71,6 +82,10 @@
71
82
  "type": "object",
72
83
  "additionalProperties": { "type": "boolean" },
73
84
  "description": "Optional rule overrides for this case, same shape as the public `rules` option."
85
+ },
86
+ "narrowNbsp": {
87
+ "enum": ["narrow", "nbsp"],
88
+ "description": "Optional, spec 1.3.0. Same shape as the public `narrowNbsp` option (spec/rules/nbsp.md §3.1a): \"nbsp\" makes the engine emit U+00A0 wherever it would emit U+202F. Omitted means \"narrow\", the default, so every pre-1.3.0 case keeps its meaning. A runner passes this through to transform() exactly as it passes `rules`."
74
89
  }
75
90
  }
76
91
  }