polytypo 1.0.1 → 1.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/lib/polytypo/data/VERSION +1 -1
- data/lib/polytypo/data/fixtures/de-CH.json +11 -3
- data/lib/polytypo/data/fixtures/de-DE.json +35 -3
- data/lib/polytypo/data/fixtures/el.json +15 -7
- data/lib/polytypo/data/fixtures/en-GB.json +21 -5
- data/lib/polytypo/data/fixtures/en-US.json +101 -37
- data/lib/polytypo/data/fixtures/fi.json +19 -3
- data/lib/polytypo/data/fixtures/fr-CA.json +30 -5
- data/lib/polytypo/data/fixtures/fr.json +111 -5
- data/lib/polytypo/data/fixtures/locale-resolution.json +1 -1
- data/lib/polytypo/data/fixtures/ru.json +37 -5
- data/lib/polytypo/data/fixtures/sv.json +11 -3
- data/lib/polytypo/data/locales/registry.json +1 -1
- data/lib/polytypo/data/rules/order.json +4 -4
- data/lib/polytypo/engine/rules/apostrophe.rb +15 -13
- data/lib/polytypo/engine/rules/nbsp.rb +10 -13
- data/lib/polytypo/engine/rules/quote_ambiguity.rb +22 -39
- data/lib/polytypo/engine/rules/spaces.rb +8 -1
- data/lib/polytypo/version.rb +1 -1
- metadata +1 -1
checksums.yaml
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
---
|
|
2
2
|
SHA256:
|
|
3
|
-
metadata.gz:
|
|
4
|
-
data.tar.gz:
|
|
3
|
+
metadata.gz: 83d0c7e30bb5525df1fcde26cfc166534b11b92e83a68490b2fdb468af059a93
|
|
4
|
+
data.tar.gz: fba93f881b2fffc39b572cabf93dd5d0a0af8073ffe3179c9594b76c17e6d052
|
|
5
5
|
SHA512:
|
|
6
|
-
metadata.gz:
|
|
7
|
-
data.tar.gz:
|
|
6
|
+
metadata.gz: 736115dda633084592307b6402ce41c40183c2b71d04b62ad95512f7ad8d62c5f43a29ea4380e718f8f1ee6d7b83da5af527f24a049e947b389b4b0e56b83980
|
|
7
|
+
data.tar.gz: dd20929d2101f1151bfd004bb9cdcfb89e3a7793442864fb35e897bf9295a21fbca7224f55efc3077ade7ed9d2079487493c8f26c051b70f6b0f39cfe701709c
|
data/lib/polytypo/data/VERSION
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
1.
|
|
1
|
+
1.2.0
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
{
|
|
2
|
-
"spec": "1.
|
|
2
|
+
"spec": "1.2.0",
|
|
3
3
|
"locale": "de-CH",
|
|
4
4
|
"cases": [
|
|
5
5
|
{
|
|
@@ -494,8 +494,16 @@
|
|
|
494
494
|
"rule": "quotes",
|
|
495
495
|
"mode": "text",
|
|
496
496
|
"in": "rock 'n' roll",
|
|
497
|
-
"out": "rock
|
|
498
|
-
"note": "spec
|
|
497
|
+
"out": "rock ’n’ roll",
|
|
498
|
+
"note": "spec 1.1.0, quotes.md §3.2's universal medial-n elision veto: by operator decision the idiom is international, so `quotes` declines the pairing in every locale rather than only where a cited quotes.elisionIdioms entry exists, and `apostrophe` (order 50) converts each mark by its own case ladder — leading elision (apostrophe.md §3.3 case 4), then trailing elision (case 3). Same output and same two case decisions as en-us-quotes-rock-n-roll, which reaches them through the cited idiom instead. History: spec 0.5.0's general ambiguous-medial-span veto preserved both marks here as literal U+0027; before 0.5.0 `quotes` paired them as an ordinary depth-1 quotation in this locale's primary glyphs."
|
|
499
|
+
},
|
|
500
|
+
{
|
|
501
|
+
"id": "de-ch-apostrophe-elision-before-guillemet",
|
|
502
|
+
"rule": "apostrophe",
|
|
503
|
+
"mode": "text",
|
|
504
|
+
"in": "die Zeitung l'\"Express\"",
|
|
505
|
+
"out": "die Zeitung l’«Express»",
|
|
506
|
+
"note": "apostrophe.md §3.3 case 3a (spec 1.2.0): `quotes` gives de-CH's « », and the elision before « converts."
|
|
499
507
|
}
|
|
500
508
|
]
|
|
501
509
|
}
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
{
|
|
2
|
-
"spec": "1.
|
|
2
|
+
"spec": "1.2.0",
|
|
3
3
|
"locale": "de-DE",
|
|
4
4
|
"cases": [
|
|
5
5
|
{
|
|
@@ -540,8 +540,40 @@
|
|
|
540
540
|
"rule": "quotes",
|
|
541
541
|
"mode": "text",
|
|
542
542
|
"in": "rock 'n' roll",
|
|
543
|
-
"out": "rock
|
|
544
|
-
"note": "spec
|
|
543
|
+
"out": "rock ’n’ roll",
|
|
544
|
+
"note": "spec 1.1.0, quotes.md §3.2's universal medial-n elision veto: by operator decision the idiom is international, so `quotes` declines the pairing in every locale rather than only where a cited quotes.elisionIdioms entry exists, and `apostrophe` (order 50) converts each mark by its own case ladder — leading elision (apostrophe.md §3.3 case 4), then trailing elision (case 3). Same output and same two case decisions as en-us-quotes-rock-n-roll, which reaches them through the cited idiom instead. History: spec 0.5.0's general ambiguous-medial-span veto preserved both marks here as literal U+0027; before 0.5.0 `quotes` paired them as an ordinary depth-1 quotation in this locale's primary glyphs."
|
|
545
|
+
},
|
|
546
|
+
{
|
|
547
|
+
"id": "de-de-spaces-dot-word-start-dotfile",
|
|
548
|
+
"rule": "spaces",
|
|
549
|
+
"mode": "text",
|
|
550
|
+
"in": "Die Datei .gitignore ist leer .",
|
|
551
|
+
"out": "Die Datei .gitignore ist leer.",
|
|
552
|
+
"note": "spaces.md §3.4 word-start clause (spec 1.2.0): the space before `.gitignore` survives because a letter follows the dot; the space before the final dot is deleted because nothing follows it."
|
|
553
|
+
},
|
|
554
|
+
{
|
|
555
|
+
"id": "de-de-apostrophe-possessive-before-closing-quote",
|
|
556
|
+
"rule": "apostrophe",
|
|
557
|
+
"mode": "text",
|
|
558
|
+
"in": "Er sagte „Hans'“ und ging.",
|
|
559
|
+
"out": "Er sagte „Hans’“ und ging.",
|
|
560
|
+
"note": "apostrophe.md §3.3 case 3a (spec 1.2.0), §6 row 17: U+201C closes a de-DE quotation but is not in CLOSEISH, so case 3 missed the possessive; case 3a reaches the same U+2019 without reading locale data."
|
|
561
|
+
},
|
|
562
|
+
{
|
|
563
|
+
"id": "de-de-apostrophe-closing-single-before-closing-double-not-elision",
|
|
564
|
+
"rule": "apostrophe",
|
|
565
|
+
"mode": "text",
|
|
566
|
+
"in": "\"Er sagte 'Hans'\" und ging.",
|
|
567
|
+
"out": "„Er sagte ‚Hans‘“ und ging.",
|
|
568
|
+
"note": "apostrophe.md §3.3 case 3a (spec 1.2.0), §7 item 7: in de-DE the closing double quote U+201C is in OPENQUOTE, so a closing single quote directly before it has exactly case 3a's shape. `quotes` (order 40) pairs both levels first, so no U+0027 reaches this rule and the nested quotation keeps its de-DE glyphs."
|
|
569
|
+
},
|
|
570
|
+
{
|
|
571
|
+
"id": "de-de-apostrophe-mixed-curly-and-straight-closing-single",
|
|
572
|
+
"rule": "apostrophe",
|
|
573
|
+
"mode": "text",
|
|
574
|
+
"in": "„Er sagte ‚Hans'“ und ging.",
|
|
575
|
+
"out": "„Er sagte ‚Hans‘“ und ging.",
|
|
576
|
+
"note": "apostrophe.md §3.3 case 3a (spec 1.2.0): the same shape with the outer pair and the opening single quote already curly. `quotes` still pairs the straight closing single quote with ‚, so case 3a never sees it."
|
|
545
577
|
}
|
|
546
578
|
]
|
|
547
579
|
}
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
{
|
|
2
|
-
"spec": "1.
|
|
2
|
+
"spec": "1.2.0",
|
|
3
3
|
"locale": "el",
|
|
4
4
|
"cases": [
|
|
5
5
|
{
|
|
@@ -133,8 +133,8 @@
|
|
|
133
133
|
"rule": "quotes",
|
|
134
134
|
"mode": "text",
|
|
135
135
|
"in": "\"Είπε 'όχι' σε μένα\", σημείωσε.",
|
|
136
|
-
"out": "«Είπε
|
|
137
|
-
"note": "quotes.md §6: one stack, inner pair at depth 2 takes the secondary pair U+201C/U+201D. Greek combines guillemets with curly double quotes, which no other v1 locale does. spec 0.5.0
|
|
136
|
+
"out": "«Είπε “όχι” σε μένα», σημείωσε.",
|
|
137
|
+
"note": "quotes.md §6: one stack, inner pair at depth 2 takes the secondary pair U+201C/U+201D. Greek combines guillemets with curly double quotes, which no other v1 locale does. spec 1.1.0 restores the nesting this case was written for: 0.5.0's general ambiguous-medial-span veto had preserved the inner 'όχι' as literal U+0027 because it was 3 letters, space-flanked; the veto that replaced it matches only a medial n (quotes.md §3.2), so an ordinary short quotation pairs again."
|
|
138
138
|
},
|
|
139
139
|
{
|
|
140
140
|
"id": "el-quotes-already-guillemets",
|
|
@@ -205,8 +205,8 @@
|
|
|
205
205
|
"rule": "quotes",
|
|
206
206
|
"mode": "text",
|
|
207
207
|
"in": "\"Είπε 'όχι' σε μένα\", σημείωσε.",
|
|
208
|
-
"out": "«Είπε
|
|
209
|
-
"note": "quotes.md §6 row 16: width drift with no gate intervention — el's secondary pair is WIDE (“ ”), so the inner NARROW pair drifts to WIDE on render and all four marks land on one stack in the gate's re-scan; the certification gate certifies without declining anything, evidence the gate's cost is near zero for the common case. spec
|
|
208
|
+
"out": "«Είπε “όχι” σε μένα», σημείωσε.",
|
|
209
|
+
"note": "quotes.md §6 row 16: width drift with no gate intervention — el's secondary pair is WIDE (“ ”), so the inner NARROW pair drifts to WIDE on render and all four marks land on one stack in the gate's re-scan; the certification gate certifies without declining anything, evidence the gate's cost is near zero for the common case. spec 1.1.0: same as el-quotes-nested — the width-drift path this case exists to exercise is reachable again now that a 3-letter medial span is no longer vetoed (quotes.md §3.2). Under 0.5.0 the inner marks never paired, so nothing drifted and the case proved nothing about the gate."
|
|
210
210
|
},
|
|
211
211
|
{
|
|
212
212
|
"id": "el-quotes-030-adversarial-interleaved",
|
|
@@ -221,8 +221,8 @@
|
|
|
221
221
|
"rule": "quotes",
|
|
222
222
|
"mode": "text",
|
|
223
223
|
"in": "rock 'n' roll",
|
|
224
|
-
"out": "rock
|
|
225
|
-
"note": "spec 0.
|
|
224
|
+
"out": "rock ’n’ roll",
|
|
225
|
+
"note": "spec 1.1.0, quotes.md §3.2's universal medial-n elision veto: by operator decision the idiom is international, so `quotes` declines the pairing in every locale rather than only where a cited quotes.elisionIdioms entry exists, and `apostrophe` (order 50) converts each mark by its own case ladder — leading elision (apostrophe.md §3.3 case 4), then trailing elision (case 3). Same output and same two case decisions as en-us-quotes-rock-n-roll, which reaches them through the cited idiom instead. History: spec 0.5.0's general ambiguous-medial-span veto preserved both marks here as literal U+0027; before 0.5.0 `quotes` paired them as an ordinary depth-1 quotation in this locale's primary glyphs."
|
|
226
226
|
},
|
|
227
227
|
{
|
|
228
228
|
"id": "el-ranges-none",
|
|
@@ -234,6 +234,14 @@
|
|
|
234
234
|
"ranges": true
|
|
235
235
|
},
|
|
236
236
|
"note": "spec 0.5.0: el's dash.range is \"none\" (no verified convention), so `ranges` substitutes nothing even when explicitly enabled (ranges.md §2, §3.3) — the same invariant the pre-0.5.0 `dashes`-tagged no-op fixtures for this locale already proved for the combined rule."
|
|
237
|
+
},
|
|
238
|
+
{
|
|
239
|
+
"id": "el-apostrophe-elision-before-guillemet",
|
|
240
|
+
"rule": "apostrophe",
|
|
241
|
+
"mode": "text",
|
|
242
|
+
"in": "η εφημερίδα l'\"Humanité\"",
|
|
243
|
+
"out": "η εφημερίδα l’«Humanité»",
|
|
244
|
+
"note": "apostrophe.md §3.3 case 3a (spec 1.2.0): `quotes` gives el's « », and the elision before « converts."
|
|
237
245
|
}
|
|
238
246
|
]
|
|
239
247
|
}
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
{
|
|
2
|
-
"spec": "1.
|
|
2
|
+
"spec": "1.2.0",
|
|
3
3
|
"locale": "en-GB",
|
|
4
4
|
"cases": [
|
|
5
5
|
{
|
|
@@ -632,8 +632,8 @@
|
|
|
632
632
|
"rule": "quotes",
|
|
633
633
|
"mode": "text",
|
|
634
634
|
"in": "\"He said 'no' to me,\" she noted.",
|
|
635
|
-
"out": "‘He said
|
|
636
|
-
"note": "quotes.md §6 case 2, inverted: depth 1 → single, depth 2 → double. Compare en-
|
|
635
|
+
"out": "‘He said “no” to me,’ she noted.",
|
|
636
|
+
"note": "quotes.md §6 case 2, inverted: depth 1 → single, depth 2 → double. Compare en-us-quotes-nested, same input. spec 1.1.0: the inner 'no' pairs as a nested quotation again — 0.5.0's general ambiguous-medial-span veto had preserved it as literal U+0027, and the universal medial-n veto that replaced it does not match a two-letter span (quotes.md §3.2)."
|
|
637
637
|
},
|
|
638
638
|
{
|
|
639
639
|
"id": "en-gb-quotes-nested-inverted-input",
|
|
@@ -712,8 +712,8 @@
|
|
|
712
712
|
"rule": "quotes",
|
|
713
713
|
"mode": "text",
|
|
714
714
|
"in": "rock 'n' roll",
|
|
715
|
-
"out": "rock
|
|
716
|
-
"note": "quotes.md §
|
|
715
|
+
"out": "rock ’n’ roll",
|
|
716
|
+
"note": "spec 1.1.0, quotes.md §3.2's universal medial-n elision veto: by operator decision the idiom is international, so `quotes` declines the pairing in every locale rather than only where a cited quotes.elisionIdioms entry exists, and `apostrophe` (order 50) converts each mark by its own case ladder — leading elision (apostrophe.md §3.3 case 4), then trailing elision (case 3). Same output and same two case decisions as en-us-quotes-rock-n-roll, which reaches them through the cited idiom instead. History: spec 0.5.0's general ambiguous-medial-span veto preserved both marks here as literal U+0027; before 0.5.0 `quotes` paired them as an ordinary depth-1 quotation in this locale's primary glyphs. en-GB is the locale quotes.md §7 item 8 flags as carrying the highest genuine-dialogue collision risk, because its primary pair is itself the single quote — that risk is why no elisionIdioms entry was ever cited here, and 1.1.0's veto deliberately does not depend on one."
|
|
717
717
|
},
|
|
718
718
|
{
|
|
719
719
|
"id": "en-gb-quotes-elision-inside-quotation",
|
|
@@ -1269,6 +1269,22 @@
|
|
|
1269
1269
|
"in": "'90s were fun,' he said",
|
|
1270
1270
|
"out": "‘90s were fun,’ he said",
|
|
1271
1271
|
"note": "quotes.md §6 row E2s: the straight-input control, confirming the corruption in E2 predates mandate 1."
|
|
1272
|
+
},
|
|
1273
|
+
{
|
|
1274
|
+
"id": "en-gb-spaces-dot-word-start-file-extensions",
|
|
1275
|
+
"rule": "spaces",
|
|
1276
|
+
"mode": "text",
|
|
1277
|
+
"in": "CAD files (.DWG, .STEP) on request",
|
|
1278
|
+
"out": "CAD files (.DWG, .STEP) on request",
|
|
1279
|
+
"note": "spaces.md §3.4 word-start clause (spec 1.2.0), §6 row 10g: `.STEP` starts a token, so the space after the comma survives. Spec 1.1.0 produced `(.DWG,.STEP)`."
|
|
1280
|
+
},
|
|
1281
|
+
{
|
|
1282
|
+
"id": "en-gb-apostrophe-elision-before-curly-quote",
|
|
1283
|
+
"rule": "apostrophe",
|
|
1284
|
+
"mode": "text",
|
|
1285
|
+
"in": "dell'“arte” moderna",
|
|
1286
|
+
"out": "dell’‘arte’ moderna",
|
|
1287
|
+
"note": "apostrophe.md §3.3 case 3a (spec 1.2.0), §6 row 16: `quotes` re-typesets the curly pair as en-GB's primary single quotes, and the elision before the opening glyph converts to U+2019."
|
|
1272
1288
|
}
|
|
1273
1289
|
]
|
|
1274
1290
|
}
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
{
|
|
2
|
-
"spec": "1.
|
|
2
|
+
"spec": "1.2.0",
|
|
3
3
|
"locale": "en-US",
|
|
4
4
|
"cases": [
|
|
5
5
|
{
|
|
@@ -755,8 +755,8 @@
|
|
|
755
755
|
"rule": "quotes",
|
|
756
756
|
"mode": "text",
|
|
757
757
|
"in": "\"He said 'no' to me,\" she noted.",
|
|
758
|
-
"out": "“He said
|
|
759
|
-
"note": "quotes.md §6 case 2: one stack, inner pair at depth 2 → secondary. spec
|
|
758
|
+
"out": "“He said ‘no’ to me,” she noted.",
|
|
759
|
+
"note": "quotes.md §6 case 2: one stack, inner pair at depth 2 → secondary. spec 1.1.0 restores the nesting this case was written for: 0.5.0's general ambiguous-medial-span veto had preserved the inner 'no' as literal U+0027 because it was 2 letters and space-flanked, and the veto that replaced it matches only a medial n (quotes.md §3.2), so an ordinary short quotation pairs again. Compare en-gb-quotes-nested, same input, inverted glyphs."
|
|
760
760
|
},
|
|
761
761
|
{
|
|
762
762
|
"id": "en-us-quotes-nested-inverted-input",
|
|
@@ -836,7 +836,7 @@
|
|
|
836
836
|
"mode": "text",
|
|
837
837
|
"in": "rock 'n' roll",
|
|
838
838
|
"out": "rock ’n’ roll",
|
|
839
|
-
"note": "quotes.md §3.2's listed elision veto (spec 0.4.0): quotes.elisionIdioms = [{ left: \"rock\", elided: \"n\", right: \"roll\" }] (en-US.json sources: Chicago Manual of Style Online on the mark's function, American Heritage Dictionary on the spaced variant) makes both marks decline pairing when the full left/elided/right context matches, so they survive to apostrophe (order 50), which independently renders each as U+2019 — leading elision (case 4) then trailing elision (case 3). apostrophe.md §6 row 11a.
|
|
839
|
+
"note": "quotes.md §3.2's listed elision veto (spec 0.4.0): quotes.elisionIdioms = [{ left: \"rock\", elided: \"n\", right: \"roll\" }] (en-US.json sources: Chicago Manual of Style Online on the mark's function, American Heritage Dictionary on the spaced variant) makes both marks decline pairing when the full left/elided/right context matches, so they survive to apostrophe (order 50), which independently renders each as U+2019 — leading elision (case 4) then trailing elision (case 3). apostrophe.md §6 row 11a. This output has been constant since 0.4.0 and is the model every other locale was brought into line with in spec 1.1.0, whose universal medial-n veto (quotes.md §3.2) now vetoes a superset of these positions — see ru-quotes-rock-n-roll for the same output reached without a citation. Matching elided content alone (a bare word list) was tried and withdrawn in 0.4.0; 1.1.0 readmits that cost knowingly and only for a single code point — see en-us-quotes-letter-n-elided."
|
|
840
840
|
},
|
|
841
841
|
{
|
|
842
842
|
"id": "en-us-quotes-elision-inside-quotation",
|
|
@@ -844,7 +844,7 @@
|
|
|
844
844
|
"mode": "text",
|
|
845
845
|
"in": "She said \"it's the '90s\" and left.",
|
|
846
846
|
"out": "She said “it’s the ’90s” and left.",
|
|
847
|
-
"note": "quotes.md §3.3 and §5.5: pass 2 keeps TWO STACKS, one per kind, and a candidate only ever touches the stack for its own kind — no reader opens with `\"` and closes with `'`, in any language. So the two DQ marks pair with each other, and the SQ elision marks find no SQ partner, survive unmatched, and `apostrophe` renders them U+2019. A PORT WITH ONE SHARED STACK PRODUCES `She said \"it’s the “90s” and left.` — the `'` of `'90s` is canClose, it pops the opening `\"`, and the real closing `\"` is stranded. That is damage rather than a miss (§5.5 calls it a release blocker). Contrast the `rock 'n' roll` case, which is a MEDIAL elision between two SQ marks, not this rule's leading/trailing single-mark elision shape — see en-us-quotes-rock-n-roll for how spec
|
|
847
|
+
"note": "quotes.md §3.3 and §5.5: pass 2 keeps TWO STACKS, one per kind, and a candidate only ever touches the stack for its own kind — no reader opens with `\"` and closes with `'`, in any language. So the two DQ marks pair with each other, and the SQ elision marks find no SQ partner, survive unmatched, and `apostrophe` renders them U+2019. A PORT WITH ONE SHARED STACK PRODUCES `She said \"it’s the “90s” and left.` — the `'` of `'90s` is canClose, it pops the opening `\"`, and the real closing `\"` is stranded. That is damage rather than a miss (§5.5 calls it a release blocker). Contrast the `rock 'n' roll` case, which is a MEDIAL elision between two SQ marks, not this rule's leading/trailing single-mark elision shape — see en-us-quotes-rock-n-roll for how spec 1.1.0's universal medial-n veto handles that different shape. History: before the per-kind stack split landed (pre-0.3.0), this exact case's canonical fixture asserted the single-shared-stack damage shown above (the stranded-quote output) as its expected `out` — a previous implementation defect, not a documented limitation; the two-stack design fixed it, and the `out` actually asserted above (the correct, two-stack form) has been correct ever since."
|
|
848
848
|
},
|
|
849
849
|
{
|
|
850
850
|
"id": "en-us-quotes-adjacent-to-punctuation",
|
|
@@ -1653,47 +1653,47 @@
|
|
|
1653
1653
|
"mode": "text",
|
|
1654
1654
|
"in": "rock ’n’ roll",
|
|
1655
1655
|
"out": "rock ’n’ roll",
|
|
1656
|
-
"note": "quotes.md §3.2/§5
|
|
1656
|
+
"note": "quotes.md §3.2/§5: U+2019 is NARROW, so the veto fires identically on already-correct input — a fixed point. This is the witness for spec 1.1.0's widening of the predicate from straight ASCII to the whole NARROW class, which is an idempotency obligation rather than a preference: the veto now lets apostrophe CONVERT both marks, so a straight-ASCII-only predicate would not recognise its own output and pass 2 would pair rock ’n’ roll as an ordinary NARROW quotation on the next pipeline run. See ru-quotes-rock-n-roll-already-curly and fi-quotes-rock-n-roll-already-curly, which fail loudly under a straight-only predicate where this en-US case would not."
|
|
1657
1657
|
},
|
|
1658
1658
|
{
|
|
1659
|
-
"id": "en-us-quotes-letter-n-
|
|
1659
|
+
"id": "en-us-quotes-letter-n-elided",
|
|
1660
1660
|
"rule": "quotes",
|
|
1661
1661
|
"mode": "text",
|
|
1662
1662
|
"in": "The letter 'n' is common.",
|
|
1663
|
-
"out": "The letter
|
|
1664
|
-
"note": "quotes.md §3.2: no
|
|
1663
|
+
"out": "The letter ’n’ is common.",
|
|
1664
|
+
"note": "quotes.md §3.2, §6 row N1: no left/right context matches (\"letter\" is not \"rock\"), so the listed-idiom veto does not fire — but the universal medial-n veto does, and apostrophe converts both marks. **This is spec 1.1.0's accepted false positive**: a genuine quotation of the letter n comes out as an elision. It is pinned precisely because it is the input that sank the context-free `elisionForms` design in 0.4.0; 1.1.0 accepts it deliberately, against the measured alternative recorded in quotes.md §7 item 8 — spec 0.5.0 avoided this one case by declining every short quotation in every locale, a far larger error class. Through spec 0.5.0 this case asserted the unconverted input and was named en-us-quotes-letter-n-not-elided."
|
|
1665
1665
|
},
|
|
1666
1666
|
{
|
|
1667
|
-
"id": "en-us-quotes-nested-quote-n-
|
|
1667
|
+
"id": "en-us-quotes-nested-quote-n-elided",
|
|
1668
1668
|
"rule": "quotes",
|
|
1669
1669
|
"mode": "text",
|
|
1670
1670
|
"in": "He said \"press 'n' now\".",
|
|
1671
|
-
"out": "He said “press
|
|
1672
|
-
"note": "quotes.md §3.2: same
|
|
1671
|
+
"out": "He said “press ’n’ now”.",
|
|
1672
|
+
"note": "quotes.md §3.2, §6 row N2: the same shape as en-us-quotes-letter-n-elided, one level deeper inside a resolved outer quotation. §3.2's predicate reads a mark's immediate neighbours and knows nothing of enclosing pairs, so nesting depth never reaches it and the inner marks are converted while the outer pair resolves normally. The second half of spec 1.1.0's accepted false positive, pinned for the same reason. Through spec 0.5.0 this case asserted the unconverted inner marks and was named en-us-quotes-nested-quote-n-not-elided."
|
|
1673
1673
|
},
|
|
1674
1674
|
{
|
|
1675
|
-
"id": "en-us-quotes-rock-n-pop-
|
|
1675
|
+
"id": "en-us-quotes-rock-n-pop-elided",
|
|
1676
1676
|
"rule": "quotes",
|
|
1677
1677
|
"mode": "text",
|
|
1678
1678
|
"in": "rock 'n' pop",
|
|
1679
|
-
"out": "rock
|
|
1680
|
-
"note": "quotes.md §3.2, §6 row N3: `right` is \"pop\", not \"roll\"
|
|
1679
|
+
"out": "rock ’n’ pop",
|
|
1680
|
+
"note": "quotes.md §3.2, §6 row N3: `right` is \"pop\", not \"roll\", so no configured idiom matches — and since spec 1.1.0 none is needed. The universal medial-n veto fires on the bare medial n and apostrophe converts both marks. Named en-us-quotes-rock-n-pop-not-elided through spec 0.5.0, when the same input was preserved unconverted."
|
|
1681
1681
|
},
|
|
1682
1682
|
{
|
|
1683
|
-
"id": "en-us-quotes-fish-n-chips-
|
|
1683
|
+
"id": "en-us-quotes-fish-n-chips-elided",
|
|
1684
1684
|
"rule": "quotes",
|
|
1685
1685
|
"mode": "text",
|
|
1686
1686
|
"in": "fish 'n' chips",
|
|
1687
|
-
"out": "fish
|
|
1688
|
-
"note": "quotes.md §3.2, §6 row N4:
|
|
1687
|
+
"out": "fish ’n’ chips",
|
|
1688
|
+
"note": "quotes.md §3.2, §6 row N4: en-US.json's AHD citation attests only rock/n/roll, and no { left: \"fish\", elided: \"n\", right: \"chips\" } entry exists — the universal medial-n veto covers it anyway, which is the practical reason spec 1.1.0 stopped requiring a citation per phrase: the long tail of X 'n' Y borrowings cannot be enumerated one citation at a time. Named en-us-quotes-fish-n-chips-not-elided through spec 0.5.0."
|
|
1689
1689
|
},
|
|
1690
1690
|
{
|
|
1691
|
-
"id": "en-us-quotes-rock-cap-n-roll-
|
|
1691
|
+
"id": "en-us-quotes-rock-cap-n-roll-elided",
|
|
1692
1692
|
"rule": "quotes",
|
|
1693
1693
|
"mode": "text",
|
|
1694
1694
|
"in": "rock 'N' roll",
|
|
1695
|
-
"out": "rock
|
|
1696
|
-
"note": "quotes.md §3.2, §6 row N5: `elided` is matched exactly, with no case leniency at all — the first-code-point leniency applies only to
|
|
1695
|
+
"out": "rock ’N’ roll",
|
|
1696
|
+
"note": "quotes.md §3.2, §6 row N5: `elided` is matched exactly by the LISTED mechanism, with no case leniency at all — the first-code-point leniency applies only to left/right — so `N` ≠ `n` still fails the cited en-US idiom. The universal medial-n veto matches U+004E as well as U+006E and converts both marks. This fixture is the witness that idiom-matching stays case-sensitive even though the veto covering the same bytes is not. Named en-us-quotes-rock-cap-n-roll-not-elided through spec 0.5.0."
|
|
1697
1697
|
},
|
|
1698
1698
|
{
|
|
1699
1699
|
"id": "en-us-quotes-rock-n-roll-html",
|
|
@@ -1708,16 +1708,16 @@
|
|
|
1708
1708
|
"rule": "quotes",
|
|
1709
1709
|
"mode": "html",
|
|
1710
1710
|
"in": "<p><em>rock</em> 'n' roll</p>",
|
|
1711
|
-
"out": "<p><em>rock</em>
|
|
1712
|
-
"note": "quotes.md §3.2, §6 row H1: modes.md §3.2's boundary marker is not INLINE-SPACE, so `left`'s word is unreachable across the <em> span boundary and the
|
|
1711
|
+
"out": "<p><em>rock</em> ’n’ roll</p>",
|
|
1712
|
+
"note": "quotes.md §3.2, §6 row H1: modes.md §3.2's boundary marker is not INLINE-SPACE, so `left`'s word is unreachable across the <em> span boundary and the LISTED idiom does not match. The universal medial-n veto (spec 1.1.0) reads no context word at all — only the enclosed code point and the single space immediately outside each mark, both of which sit in the same prose span as the marks — so it fires here and apostrophe converts both marks, converging with the unsplit en-us-quotes-rock-n-roll-html. Through spec 0.5.0 this case asserted the unconverted input."
|
|
1713
1713
|
},
|
|
1714
1714
|
{
|
|
1715
1715
|
"id": "en-us-quotes-rock-n-roll-html-right-split",
|
|
1716
1716
|
"rule": "quotes",
|
|
1717
1717
|
"mode": "html",
|
|
1718
1718
|
"in": "<p>rock 'n' <em>roll</em></p>",
|
|
1719
|
-
"out": "<p>rock
|
|
1720
|
-
"note": "quotes.md §3.2, §6 row H2: mirror
|
|
1719
|
+
"out": "<p>rock ’n’ <em>roll</em></p>",
|
|
1720
|
+
"note": "quotes.md §3.2, §6 row H2: the mirror of en-us-quotes-rock-n-roll-html-left-split, with `right`'s word inside the following <em> span. modes.md §3.2's boundary marker is not INLINE-SPACE, so `right` is unreachable and the LISTED idiom does not match. The universal medial-n veto (spec 1.1.0) reads no context word at all — only the enclosed code point and the single space immediately outside each mark, both in the same prose span as the marks — so it fires and apostrophe converts both, converging with the unsplit en-us-quotes-rock-n-roll-html. Through spec 0.5.0 this case asserted the unconverted input."
|
|
1721
1721
|
},
|
|
1722
1722
|
{
|
|
1723
1723
|
"id": "en-us-quotes-rock-n-roll-markdown-left-split",
|
|
@@ -1725,32 +1725,32 @@
|
|
|
1725
1725
|
"mode": "markdown",
|
|
1726
1726
|
"dialect": "commonmark",
|
|
1727
1727
|
"in": "*rock* 'n' roll\n",
|
|
1728
|
-
"out": "*rock*
|
|
1729
|
-
"note": "quotes.md §3.2, §6 row H3: `left`'s word sits inside an emphasis span
|
|
1728
|
+
"out": "*rock* ’n’ roll\n",
|
|
1729
|
+
"note": "quotes.md §3.2, §6 row H3: `left`'s word sits inside an emphasis span, so the LISTED idiom's word-boundary test cannot reach it across the span marker (modes.md §3.2) and no idiom matches. The universal medial-n veto (spec 1.1.0) needs no context word — the mark's own immediately-preceding INLINE-SPACE is in the same span as the mark — so it fires and apostrophe converts both marks, matching text mode on the same characters split the same way. Through spec 0.5.0 this case asserted the unconverted input."
|
|
1730
1730
|
},
|
|
1731
1731
|
{
|
|
1732
1732
|
"id": "en-us-quotes-ambiguous-single-letter",
|
|
1733
1733
|
"rule": "quotes",
|
|
1734
1734
|
"mode": "text",
|
|
1735
1735
|
"in": "She chose 'A' today",
|
|
1736
|
-
"out": "She chose
|
|
1737
|
-
"note": "spec
|
|
1736
|
+
"out": "She chose “A” today",
|
|
1737
|
+
"note": "spec 1.1.0, quotes.md §3.2: 'A' is a single quoted letter and pairs as an ordinary quotation. Under spec 0.5.0 the general ambiguous-medial-span veto preserved it as literal U+0027, because its shape test admitted any 1 to 3 LETTER code points and could not tell this from rock 'n' roll. This fixture flipped role in 1.1.0: it pinned that veto's reach, and now pins that the replacement veto does NOT reach a span whose enclosed code point is not n or N."
|
|
1738
1738
|
},
|
|
1739
1739
|
{
|
|
1740
1740
|
"id": "en-us-quotes-ambiguous-two-letter",
|
|
1741
1741
|
"rule": "quotes",
|
|
1742
1742
|
"mode": "text",
|
|
1743
1743
|
"in": "They said 'no' yesterday",
|
|
1744
|
-
"out": "They said
|
|
1745
|
-
"note": "spec
|
|
1744
|
+
"out": "They said “no” yesterday",
|
|
1745
|
+
"note": "spec 1.1.0, quotes.md §3.2: 2 letters enclosed — not a medial n, so the universal veto does not fire and the span pairs as an ordinary quotation. Same reasoning and same role reversal as en-us-quotes-ambiguous-single-letter."
|
|
1746
1746
|
},
|
|
1747
1747
|
{
|
|
1748
1748
|
"id": "en-us-quotes-ambiguous-three-letter",
|
|
1749
1749
|
"rule": "quotes",
|
|
1750
1750
|
"mode": "text",
|
|
1751
1751
|
"in": "say 'yes' now",
|
|
1752
|
-
"out": "say
|
|
1753
|
-
"note": "spec
|
|
1752
|
+
"out": "say “yes” now",
|
|
1753
|
+
"note": "spec 1.1.0, quotes.md §3.2: 3 letters enclosed — the upper bound of spec 0.5.0's withdrawn shape test, kept as a fixture precisely because it was that test's boundary case. The universal medial-n veto does not fire, and the span pairs as an ordinary quotation."
|
|
1754
1754
|
},
|
|
1755
1755
|
{
|
|
1756
1756
|
"id": "en-us-quotes-ambiguous-four-letter-not-caught",
|
|
@@ -1765,8 +1765,8 @@
|
|
|
1765
1765
|
"rule": "quotes",
|
|
1766
1766
|
"mode": "text",
|
|
1767
1767
|
"in": "say '𐐀' now",
|
|
1768
|
-
"out": "say
|
|
1769
|
-
"note": "spec 0.
|
|
1768
|
+
"out": "say “𐐀” now",
|
|
1769
|
+
"note": "spec 1.1.0, quotes.md §3.2: a single astral-plane LETTER (U+10400, two UTF-16 units, one code point). It pairs as an ordinary quotation because it is not n or N — the veto compares code points, so an astral letter is neither specially admitted nor specially excluded. Retained from spec 0.5.0, where it pinned that the withdrawn shape test counted code points rather than UTF-16 units; the code-point discipline it tests (ARCHITECTURE.md §4.2) is unchanged."
|
|
1770
1770
|
},
|
|
1771
1771
|
{
|
|
1772
1772
|
"id": "en-us-quotes-ambiguous-digit-not-letter",
|
|
@@ -1774,7 +1774,7 @@
|
|
|
1774
1774
|
"mode": "text",
|
|
1775
1775
|
"in": "say '12' now",
|
|
1776
1776
|
"out": "say “12” now",
|
|
1777
|
-
"note": "spec 0.
|
|
1777
|
+
"note": "spec 1.1.0, quotes.md §3.2: the enclosed code points are digits, not n or N, so the veto does not fire and the span pairs as an ordinary quotation. This output has been constant across 0.5.0 and 1.1.0 — under 0.5.0 because digits were not LETTER, now because they are not the one code point the veto matches."
|
|
1778
1778
|
},
|
|
1779
1779
|
{
|
|
1780
1780
|
"id": "en-us-quotes-ambiguous-punctuation-not-letter",
|
|
@@ -1790,7 +1790,7 @@
|
|
|
1790
1790
|
"mode": "text",
|
|
1791
1791
|
"in": "She chose ’A’ today",
|
|
1792
1792
|
"out": "She chose “A” today",
|
|
1793
|
-
"note": "spec 0.
|
|
1793
|
+
"note": "spec 1.1.0, quotes.md §3.2: 'A' is not n or N, so the veto does not fire and the already-curly pair is re-typeset as an ordinary quotation. **The reason this output is unchanged is not the reason it was unchanged in 0.5.0.** 0.5.0's veto matched straight ASCII U+0027 only, so a curly pair was out of scope by construction; 1.1.0's matches the whole NARROW class — an idempotency obligation, since the veto now produces U+2019 — and declines this span on its enclosed code point instead. See en-us-quotes-rock-n-roll-already-curly for a curly pair the veto does match."
|
|
1794
1794
|
},
|
|
1795
1795
|
{
|
|
1796
1796
|
"id": "en-us-ranges-decimal-and-path-rejected",
|
|
@@ -1802,6 +1802,70 @@
|
|
|
1802
1802
|
"ranges": true
|
|
1803
1803
|
},
|
|
1804
1804
|
"note": "ranges.md §3.2, G3 — not part of a decimal or a path. `before` is U+002E (.) for the first candidate and U+002F (/) for the second; `after` is U+002F (/) for the second. This is currently the only canonical fixture exercising G3 — previously untested by any canonical case even before the spec 0.5.0 ownership correction (dashes.test.ts's engine-level 'must not touch' list covered it, but no fixture did)."
|
|
1805
|
+
},
|
|
1806
|
+
{
|
|
1807
|
+
"id": "en-us-spaces-dot-word-start-dotnet",
|
|
1808
|
+
"rule": "spaces",
|
|
1809
|
+
"mode": "text",
|
|
1810
|
+
"in": "Use .NET, .NET Core or .NET Framework",
|
|
1811
|
+
"out": "Use .NET, .NET Core or .NET Framework",
|
|
1812
|
+
"note": "spaces.md §3.4 word-start clause (spec 1.2.0), §6 row 10f: each dot is followed by a letter, so it starts a token and the space before it is not deleted. Spec 1.1.0 produced `Use.NET,.NET Core or.NET Framework`."
|
|
1813
|
+
},
|
|
1814
|
+
{
|
|
1815
|
+
"id": "en-us-spaces-dot-word-start-decimal",
|
|
1816
|
+
"rule": "spaces",
|
|
1817
|
+
"mode": "text",
|
|
1818
|
+
"in": "the values .5 and .9 apply",
|
|
1819
|
+
"out": "the values .5 and .9 apply",
|
|
1820
|
+
"note": "spaces.md §3.4 word-start clause (spec 1.2.0), §6 row 10h: an ASCII digit after the dot counts the same as a letter."
|
|
1821
|
+
},
|
|
1822
|
+
{
|
|
1823
|
+
"id": "en-us-spaces-dot-word-start-misplaced-full-stop",
|
|
1824
|
+
"rule": "spaces",
|
|
1825
|
+
"mode": "text",
|
|
1826
|
+
"in": "the end .Next sentence",
|
|
1827
|
+
"out": "the end .Next sentence",
|
|
1828
|
+
"note": "spaces.md §3.4, §6 row 10i: the accepted cost of the word-start clause. A misplaced full stop followed directly by a letter keeps its stray space, because deleting it would glue two words that a later pass cannot separate again."
|
|
1829
|
+
},
|
|
1830
|
+
{
|
|
1831
|
+
"id": "en-us-spaces-dot-terminal-still-strips",
|
|
1832
|
+
"rule": "spaces",
|
|
1833
|
+
"mode": "text",
|
|
1834
|
+
"in": "See p. 12 . Next",
|
|
1835
|
+
"out": "See p. 12. Next",
|
|
1836
|
+
"note": "spaces.md §3.4, §6 row 10j: a dot followed by a space is terminal punctuation, so the space before it is still deleted under spec 1.2.0."
|
|
1837
|
+
},
|
|
1838
|
+
{
|
|
1839
|
+
"id": "en-us-nbsp-span-boundary-no-punctuation-data",
|
|
1840
|
+
"rule": "nbsp",
|
|
1841
|
+
"mode": "html",
|
|
1842
|
+
"in": "<strong>Note :</strong> read this",
|
|
1843
|
+
"out": "<strong>Note:</strong> read this",
|
|
1844
|
+
"note": "nbsp.md §3.3 step 2 (spec 1.2.0): accepting the marker changes nothing in a locale whose punctuation lists are empty. `spaces` deletes the space and nothing is put back, the same result as in text mode."
|
|
1845
|
+
},
|
|
1846
|
+
{
|
|
1847
|
+
"id": "en-us-apostrophe-elision-before-quote",
|
|
1848
|
+
"rule": "apostrophe",
|
|
1849
|
+
"mode": "text",
|
|
1850
|
+
"in": "the magazine l'\"Uomo\" today",
|
|
1851
|
+
"out": "the magazine l’“Uomo” today",
|
|
1852
|
+
"note": "apostrophe.md §3.3 case 3a (spec 1.2.0): `quotes` converts the straight double quotes to U+201C/U+201D, and the elision before U+201C converts to U+2019."
|
|
1853
|
+
},
|
|
1854
|
+
{
|
|
1855
|
+
"id": "en-us-apostrophe-prime-before-bracket-untouched",
|
|
1856
|
+
"rule": "apostrophe",
|
|
1857
|
+
"mode": "text",
|
|
1858
|
+
"in": "f'(x) = 2x",
|
|
1859
|
+
"out": "f'(x) = 2x",
|
|
1860
|
+
"note": "apostrophe.md §3.3 case 3a (spec 1.2.0), §6 row 18: `(` is in OPENISH but not in OPENQUOTE, so a prime on a function name is left alone."
|
|
1861
|
+
},
|
|
1862
|
+
{
|
|
1863
|
+
"id": "en-us-spaces-dot-before-span-boundary-is-terminal",
|
|
1864
|
+
"rule": "spaces",
|
|
1865
|
+
"mode": "html",
|
|
1866
|
+
"in": "Use .<b>NET</b> today",
|
|
1867
|
+
"out": "Use.<b>NET</b> today",
|
|
1868
|
+
"note": "spaces.md §3.4 word-start clause (spec 1.2.0): the code point after the dot is the −1 span boundary marker, which is neither a LETTER nor an ASCII digit (modes.md §3.3), so the dot counts as terminal and the space before it is deleted."
|
|
1805
1869
|
}
|
|
1806
1870
|
]
|
|
1807
1871
|
}
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
{
|
|
2
|
-
"spec": "1.
|
|
2
|
+
"spec": "1.2.0",
|
|
3
3
|
"locale": "fi",
|
|
4
4
|
"cases": [
|
|
5
5
|
{
|
|
@@ -752,8 +752,16 @@
|
|
|
752
752
|
"rule": "quotes",
|
|
753
753
|
"mode": "text",
|
|
754
754
|
"in": "rock 'n' roll",
|
|
755
|
-
"out": "rock
|
|
756
|
-
"note": "quotes.md §
|
|
755
|
+
"out": "rock ’n’ roll",
|
|
756
|
+
"note": "spec 1.1.0, quotes.md §3.2's universal medial-n elision veto: by operator decision the idiom is international, so `quotes` declines the pairing in every locale rather than only where a cited quotes.elisionIdioms entry exists, and `apostrophe` (order 50) converts each mark by its own case ladder — leading elision (apostrophe.md §3.3 case 4), then trailing elision (case 3). Same output and same two case decisions as en-us-quotes-rock-n-roll, which reaches them through the cited idiom instead. History: spec 0.5.0's general ambiguous-medial-span veto preserved both marks here as literal U+0027; before 0.5.0 `quotes` paired them as an ordinary depth-1 quotation in this locale's primary glyphs. fi's own dictionary writes this loanword closed up (rock'n'roll), so a citation for the spaced form would have been evidenced-inert and none was ever added — quotes.md §7 item 8. The universal veto covers the spaced form regardless, which is the point of moving the mechanism out of locale data. See fi-quotes-rock-n-roll-already-curly for the second-pass fixed point this output depends on."
|
|
757
|
+
},
|
|
758
|
+
{
|
|
759
|
+
"id": "fi-quotes-rock-n-roll-already-curly",
|
|
760
|
+
"rule": "quotes",
|
|
761
|
+
"mode": "text",
|
|
762
|
+
"in": "rock ’n’ roll",
|
|
763
|
+
"out": "rock ’n’ roll",
|
|
764
|
+
"note": "quotes.md §3.2, the sharpest regression witness for spec 1.1.0's widening of the veto to the whole NARROW class: fi's SECONDARY pair is U+2019 on both sides, so the marks this rule's own output leaves behind are indistinguishable from fi's own prescribed nested-quotation glyphs. A port that matches U+0027 only does not recognise them, pairs them, and turns the output of fi-quotes-rock-n-roll into `rock ”n” roll` — fi's primary pair at depth 1. Companion to ru-quotes-rock-n-roll-already-curly, which fails the same way in a locale whose primary is WIDE; both are measured, not hypothesised."
|
|
757
765
|
},
|
|
758
766
|
{
|
|
759
767
|
"id": "fi-quotes-already-correct",
|
|
@@ -1301,6 +1309,14 @@
|
|
|
1301
1309
|
"in": "'90s were fun,' he said",
|
|
1302
1310
|
"out": "”90s were fun,” he said",
|
|
1303
1311
|
"note": "quotes.md §6 row E3s: the straight-input control for E3."
|
|
1312
|
+
},
|
|
1313
|
+
{
|
|
1314
|
+
"id": "fi-apostrophe-elision-before-shared-quote-glyph",
|
|
1315
|
+
"rule": "apostrophe",
|
|
1316
|
+
"mode": "text",
|
|
1317
|
+
"in": "lehti l'\"Humanité\"",
|
|
1318
|
+
"out": "lehti l’”Humanité”",
|
|
1319
|
+
"note": "apostrophe.md §3.3 cases 3 and 3a (spec 1.2.0): fi opens and closes with U+201D, which is in CLOSEISH, so case 3 already converted this before 1.2.0. Pinned so that case 3a is not read as changing a shared-glyph locale."
|
|
1304
1320
|
}
|
|
1305
1321
|
]
|
|
1306
1322
|
}
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
{
|
|
2
|
-
"spec": "1.
|
|
2
|
+
"spec": "1.2.0",
|
|
3
3
|
"locale": "fr-CA",
|
|
4
4
|
"cases": [
|
|
5
5
|
{
|
|
@@ -80,8 +80,8 @@
|
|
|
80
80
|
"rule": "quotes",
|
|
81
81
|
"mode": "text",
|
|
82
82
|
"in": "\"Il a dit 'oui' hier.\"",
|
|
83
|
-
"out": "« Il a dit
|
|
84
|
-
"note": "spec
|
|
83
|
+
"out": "« Il a dit “oui” hier. »",
|
|
84
|
+
"note": "spec 1.1.0: same as fr-quotes-nested — the inner 'oui' pairs as a nested quotation again now that the veto matches only a medial n (quotes.md §3.2), where 0.5.0's shape test preserved every 1-to-3-letter span."
|
|
85
85
|
},
|
|
86
86
|
{
|
|
87
87
|
"id": "fr-ca-apostrophe-elision",
|
|
@@ -250,8 +250,8 @@
|
|
|
250
250
|
"rule": "quotes",
|
|
251
251
|
"mode": "text",
|
|
252
252
|
"in": "rock 'n' roll",
|
|
253
|
-
"out": "rock
|
|
254
|
-
"note": "spec
|
|
253
|
+
"out": "rock ’n’ roll",
|
|
254
|
+
"note": "spec 1.1.0, quotes.md §3.2's universal medial-n elision veto: by operator decision the idiom is international, so `quotes` declines the pairing in every locale rather than only where a cited quotes.elisionIdioms entry exists, and `apostrophe` (order 50) converts each mark by its own case ladder — leading elision (apostrophe.md §3.3 case 4), then trailing elision (case 3). Same output and same two case decisions as en-us-quotes-rock-n-roll, which reaches them through the cited idiom instead. History: spec 0.5.0's general ambiguous-medial-span veto preserved both marks here as literal U+0027; before 0.5.0 `quotes` paired them as an ordinary depth-1 quotation in this locale's primary glyphs."
|
|
255
255
|
},
|
|
256
256
|
{
|
|
257
257
|
"id": "fr-ca-ranges-none",
|
|
@@ -263,6 +263,31 @@
|
|
|
263
263
|
"ranges": true
|
|
264
264
|
},
|
|
265
265
|
"note": "spec 0.5.0: fr-CA's dash.range is \"none\" (no verified convention), so `ranges` substitutes nothing even when explicitly enabled (ranges.md §2, §3.3) — the same invariant the pre-0.5.0 `dashes`-tagged no-op fixtures for this locale already proved for the combined rule."
|
|
266
|
+
},
|
|
267
|
+
{
|
|
268
|
+
"id": "fr-ca-nbsp-span-boundary-strong-colon",
|
|
269
|
+
"rule": "nbsp",
|
|
270
|
+
"mode": "html",
|
|
271
|
+
"in": "<strong>Résistant au gel :</strong> il reste souple",
|
|
272
|
+
"out": "<strong>Résistant au gel :</strong> il reste souple",
|
|
273
|
+
"note": "nbsp.md §3.3 step 2 (spec 1.2.0): same as the `fr` case; `fr-CA` lists `:` in its own punctuation data."
|
|
274
|
+
},
|
|
275
|
+
{
|
|
276
|
+
"id": "fr-ca-nbsp-span-boundary-markdown-strong-colon",
|
|
277
|
+
"rule": "nbsp",
|
|
278
|
+
"mode": "markdown",
|
|
279
|
+
"dialect": "commonmark",
|
|
280
|
+
"in": "**Note :** voir plus bas\n",
|
|
281
|
+
"out": "**Note :** voir plus bas\n",
|
|
282
|
+
"note": "nbsp.md §3.3 step 2 (spec 1.2.0): the colon ends the strong span; the −1 marker after it is in CLOSEISH, so N1 restores the U+00A0 that `spaces` deleted. (fr-CA lists no punctuation for N2, so `?` would take nothing back in any mode.)"
|
|
283
|
+
},
|
|
284
|
+
{
|
|
285
|
+
"id": "fr-ca-apostrophe-elision-before-straight-quote",
|
|
286
|
+
"rule": "apostrophe",
|
|
287
|
+
"mode": "text",
|
|
288
|
+
"in": "l'\"idée\" est simple",
|
|
289
|
+
"out": "l’« idée » est simple",
|
|
290
|
+
"note": "apostrophe.md §3.3 case 3a (spec 1.2.0): same as `fr`."
|
|
266
291
|
}
|
|
267
292
|
]
|
|
268
293
|
}
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
{
|
|
2
|
-
"spec": "1.
|
|
2
|
+
"spec": "1.2.0",
|
|
3
3
|
"locale": "fr",
|
|
4
4
|
"cases": [
|
|
5
5
|
{
|
|
@@ -235,8 +235,8 @@
|
|
|
235
235
|
"rule": "quotes",
|
|
236
236
|
"mode": "text",
|
|
237
237
|
"in": "\"Il a dit 'oui' hier.\"",
|
|
238
|
-
"out": "« Il a dit
|
|
239
|
-
"note": "Depth 1 primary « », depth 2 secondary “ ” whose innerSpace is none. spec
|
|
238
|
+
"out": "« Il a dit “oui” hier. »",
|
|
239
|
+
"note": "Depth 1 primary « », depth 2 secondary “ ” whose innerSpace is none. spec 1.1.0: the inner 'oui' pairs again — 0.5.0's general ambiguous-medial-span veto had preserved it as literal U+0027 for being 3 letters and space-flanked, and the universal medial-n veto that replaced it matches only a single n (quotes.md §3.2)."
|
|
240
240
|
},
|
|
241
241
|
{
|
|
242
242
|
"id": "fr-quotes-unbalanced",
|
|
@@ -585,8 +585,8 @@
|
|
|
585
585
|
"rule": "quotes",
|
|
586
586
|
"mode": "text",
|
|
587
587
|
"in": "rock 'n' roll",
|
|
588
|
-
"out": "rock
|
|
589
|
-
"note": "spec 0.
|
|
588
|
+
"out": "rock ’n’ roll",
|
|
589
|
+
"note": "spec 1.1.0, quotes.md §3.2's universal medial-n elision veto: by operator decision the idiom is international, so `quotes` declines the pairing in every locale rather than only where a cited quotes.elisionIdioms entry exists, and `apostrophe` (order 50) converts each mark by its own case ladder — leading elision (apostrophe.md §3.3 case 4), then trailing elision (case 3). Same output and same two case decisions as en-us-quotes-rock-n-roll, which reaches them through the cited idiom instead. History: spec 0.5.0's general ambiguous-medial-span veto preserved both marks here as literal U+0027; before 0.5.0 `quotes` paired them as an ordinary depth-1 quotation in this locale's primary glyphs."
|
|
590
590
|
},
|
|
591
591
|
{
|
|
592
592
|
"id": "fr-ranges-none",
|
|
@@ -598,6 +598,112 @@
|
|
|
598
598
|
"ranges": true
|
|
599
599
|
},
|
|
600
600
|
"note": "spec 0.5.0: fr's dash.range is \"none\" (no verified convention), so `ranges` substitutes nothing even when explicitly enabled (ranges.md §2, §3.3) — the same invariant the pre-0.5.0 `dashes`-tagged no-op fixtures for this locale already proved for the combined rule."
|
|
601
|
+
},
|
|
602
|
+
{
|
|
603
|
+
"id": "fr-spaces-dot-word-start-dotfile",
|
|
604
|
+
"rule": "spaces",
|
|
605
|
+
"mode": "text",
|
|
606
|
+
"in": "Voir le fichier .htaccess .",
|
|
607
|
+
"out": "Voir le fichier .htaccess.",
|
|
608
|
+
"note": "spaces.md §3.4 word-start clause (spec 1.2.0): the space before `.htaccess` survives, the space before the terminal dot is deleted. `.` is in no `nbsp` list for `fr`, so nothing is put back."
|
|
609
|
+
},
|
|
610
|
+
{
|
|
611
|
+
"id": "fr-nbsp-span-boundary-strong-colon",
|
|
612
|
+
"rule": "nbsp",
|
|
613
|
+
"mode": "html",
|
|
614
|
+
"in": "<strong>Résistant au gel :</strong> il reste souple",
|
|
615
|
+
"out": "<strong>Résistant au gel :</strong> il reste souple",
|
|
616
|
+
"note": "nbsp.md §3.3 step 2 (spec 1.2.0), §6 row 6c; modes.md §3.3: `spaces` deletes the typed space before `:`, and the −1 span boundary marker after the mark is in `nbsp`'s CLOSEISH, so N1 inserts U+00A0 at the mark's index, inside the span. Spec 1.1.0 runtimes produced `gel:</strong>`."
|
|
617
|
+
},
|
|
618
|
+
{
|
|
619
|
+
"id": "fr-nbsp-span-boundary-br-question",
|
|
620
|
+
"rule": "nbsp",
|
|
621
|
+
"mode": "html",
|
|
622
|
+
"in": "Et la réglementation ?<br>Oui",
|
|
623
|
+
"out": "Et la réglementation ?<br>Oui",
|
|
624
|
+
"note": "nbsp.md §3.3 step 2 (spec 1.2.0), §6 row 6d: `<br>` leaves no line terminator in the gap, so the marker is −1; it is in CLOSEISH and N2 restores U+202F."
|
|
625
|
+
},
|
|
626
|
+
{
|
|
627
|
+
"id": "fr-nbsp-span-boundary-link-question",
|
|
628
|
+
"rule": "nbsp",
|
|
629
|
+
"mode": "html",
|
|
630
|
+
"in": "<a href=\"/faq\">Comment ?</a> Oui",
|
|
631
|
+
"out": "<a href=\"/faq\">Comment ?</a> Oui",
|
|
632
|
+
"note": "nbsp.md §3.3 step 2 (spec 1.2.0): the mark ends the link text, the marker follows it, and N2 restores U+202F instead of leaving `Comment?`."
|
|
633
|
+
},
|
|
634
|
+
{
|
|
635
|
+
"id": "fr-nbsp-span-boundary-markdown-strong-colon",
|
|
636
|
+
"rule": "nbsp",
|
|
637
|
+
"mode": "markdown",
|
|
638
|
+
"dialect": "commonmark",
|
|
639
|
+
"in": "**Résistant au gel :** il reste souple\n",
|
|
640
|
+
"out": "**Résistant au gel :** il reste souple\n",
|
|
641
|
+
"note": "nbsp.md §3.3 step 2 (spec 1.2.0): the Markdown form of the French definition pattern; the closing `**` is a skipped region with no line terminator, so the marker is −1 and N1 restores U+00A0."
|
|
642
|
+
},
|
|
643
|
+
{
|
|
644
|
+
"id": "fr-nbsp-span-boundary-not-openish-em",
|
|
645
|
+
"rule": "nbsp",
|
|
646
|
+
"mode": "html",
|
|
647
|
+
"in": "Il dit <em>non</em> ! Oui",
|
|
648
|
+
"out": "Il dit <em>non</em> ! Oui",
|
|
649
|
+
"note": "nbsp.md §3.3 step 3 (spec 1.2.0), §6 row 6e; modes.md §3.3: the marker is not in `nbsp`'s OPENISH, so the quote-glyph guard does not decline and N2 converts the space. A port that applied the quote-glyph guard to the marker would leave a plain U+0020 here."
|
|
650
|
+
},
|
|
651
|
+
{
|
|
652
|
+
"id": "fr-nbsp-span-boundary-not-openish-link",
|
|
653
|
+
"rule": "nbsp",
|
|
654
|
+
"mode": "markdown",
|
|
655
|
+
"dialect": "commonmark",
|
|
656
|
+
"in": "Voir [ceci](http://example.org) : oui\n",
|
|
657
|
+
"out": "Voir [ceci](http://example.org) : oui\n",
|
|
658
|
+
"note": "nbsp.md §3.3 step 3 (spec 1.2.0): the second witness that the quote-glyph guard does not decline at a span boundary. The link text ends two places left of `:`, and N1 still converts the space. The marker's absence from the rest of `nbsp`'s OPENISH is pinned by ru-nbsp-span-boundary-not-openish-short-word."
|
|
659
|
+
},
|
|
660
|
+
{
|
|
661
|
+
"id": "fr-nbsp-span-boundary-mark-alone-in-span",
|
|
662
|
+
"rule": "nbsp",
|
|
663
|
+
"mode": "html",
|
|
664
|
+
"in": "mot<em>!</em>",
|
|
665
|
+
"out": "mot<em>!</em>",
|
|
666
|
+
"note": "nbsp.md §3.3 step 2; modes.md §3.4 edge-growth rule: the mark is alone in its span, so the insertion point is the span edge and the insertion is discarded, as it was before spec 1.2.0."
|
|
667
|
+
},
|
|
668
|
+
{
|
|
669
|
+
"id": "fr-nbsp-span-boundary-colon-before-digit-span",
|
|
670
|
+
"rule": "nbsp",
|
|
671
|
+
"mode": "html",
|
|
672
|
+
"in": "Il est 12:<b>30</b>",
|
|
673
|
+
"out": "Il est 12 :<b>30</b>",
|
|
674
|
+
"note": "nbsp.md §3.3 step 2 (spec 1.2.0), §6 row 6f: the stated cost of accepting the marker. Step 2 cannot see past the span boundary, so a colon that ends one span while a digit starts the next is treated as sentence punctuation."
|
|
675
|
+
},
|
|
676
|
+
{
|
|
677
|
+
"id": "fr-apostrophe-elision-before-guillemet",
|
|
678
|
+
"rule": "apostrophe",
|
|
679
|
+
"mode": "text",
|
|
680
|
+
"in": "un mélange d'« urine diluée »",
|
|
681
|
+
"out": "un mélange d’« urine diluée »",
|
|
682
|
+
"note": "apostrophe.md §3.3 case 3a (spec 1.2.0), §6 row 15: letter left, opening quotation glyph right. Spec 1.1.0 left the apostrophe straight, because no case matched a letter followed by an OPENISH-only glyph."
|
|
683
|
+
},
|
|
684
|
+
{
|
|
685
|
+
"id": "fr-apostrophe-elision-before-straight-quote",
|
|
686
|
+
"rule": "apostrophe",
|
|
687
|
+
"mode": "text",
|
|
688
|
+
"in": "l'\"idée\" est simple",
|
|
689
|
+
"out": "l’« idée » est simple",
|
|
690
|
+
"note": "apostrophe.md §3.2, §3.3 case 3a (spec 1.2.0): `quotes` (order 40) turns the straight `\"` into `«` first, so the mark reaches this rule next to an opening glyph and case 3a converts it."
|
|
691
|
+
},
|
|
692
|
+
{
|
|
693
|
+
"id": "fr-apostrophe-elision-qu-before-guillemet",
|
|
694
|
+
"rule": "apostrophe",
|
|
695
|
+
"mode": "text",
|
|
696
|
+
"in": "Il faut qu'« il vienne »",
|
|
697
|
+
"out": "Il faut qu’« il vienne »",
|
|
698
|
+
"note": "apostrophe.md §3.3 case 3a (spec 1.2.0): `qu'` before a quotation."
|
|
699
|
+
},
|
|
700
|
+
{
|
|
701
|
+
"id": "fr-nbsp-span-boundary-colon-before-url-span",
|
|
702
|
+
"rule": "nbsp",
|
|
703
|
+
"mode": "html",
|
|
704
|
+
"in": "Voir http:<a href=\"//x.org\">//x.org</a>",
|
|
705
|
+
"out": "Voir http :<a href=\"//x.org\">//x.org</a>",
|
|
706
|
+
"note": "nbsp.md §3.3 step 2 (spec 1.2.0): the second stated cost of accepting the marker. Step 2 cannot see the `/` in the next span, so the colon is treated as sentence punctuation and N1 inserts U+00A0."
|
|
601
707
|
}
|
|
602
708
|
]
|
|
603
709
|
}
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
{
|
|
2
|
-
"spec": "1.
|
|
2
|
+
"spec": "1.2.0",
|
|
3
3
|
"locale": "ru",
|
|
4
4
|
"cases": [
|
|
5
5
|
{
|
|
@@ -323,8 +323,8 @@
|
|
|
323
323
|
"rule": "quotes",
|
|
324
324
|
"mode": "text",
|
|
325
325
|
"in": "Он сказал: \"это 'моё' дело\".",
|
|
326
|
-
"out": "Он сказал: «это
|
|
327
|
-
"note": "quotes.md §6 case 14: depth 1 takes the ёлочки « », depth 2 the лапки „ “ — Лопатин 2006. Note the secondary pair's closing glyph is U+201C, which is in OPENISH; the class says nothing about the role. spec
|
|
326
|
+
"out": "Он сказал: «это „моё“ дело».",
|
|
327
|
+
"note": "quotes.md §6 case 14: depth 1 takes the ёлочки « », depth 2 the лапки „ “ — Лопатин 2006. Note the secondary pair's closing glyph is U+201C, which is in OPENISH; the class says nothing about the role. spec 1.1.0: the inner 'моё' pairs as a nested quotation again. Under 0.5.0 the general ambiguous-medial-span veto preserved it as literal U+0027 for being 3 LETTER code points and space-flanked — a count that also depended on the input's normalization form, since LETTER includes Mn and the decomposed ё is two code points. The universal medial-n veto that replaced it compares one code point against a fixed pair (quotes.md §3.2) and has neither problem."
|
|
328
328
|
},
|
|
329
329
|
{
|
|
330
330
|
"id": "ru-quotes-roundtrip",
|
|
@@ -681,8 +681,40 @@
|
|
|
681
681
|
"rule": "quotes",
|
|
682
682
|
"mode": "text",
|
|
683
683
|
"in": "rock 'n' roll",
|
|
684
|
-
"out": "rock
|
|
685
|
-
"note": "spec 0.
|
|
684
|
+
"out": "rock ’n’ roll",
|
|
685
|
+
"note": "spec 1.1.0, quotes.md §3.2's universal medial-n elision veto: by operator decision the idiom is international, so `quotes` declines the pairing in every locale rather than only where a cited quotes.elisionIdioms entry exists, and `apostrophe` (order 50) converts each mark by its own case ladder — leading elision (apostrophe.md §3.3 case 4), then trailing elision (case 3). Same output and same two case decisions as en-us-quotes-rock-n-roll, which reaches them through the cited idiom instead. History: spec 0.5.0's general ambiguous-medial-span veto preserved both marks here as literal U+0027; before 0.5.0 `quotes` paired them as an ordinary depth-1 quotation in this locale's primary glyphs. See ru-quotes-rock-n-roll-already-curly for the second-pass fixed point this output depends on."
|
|
686
|
+
},
|
|
687
|
+
{
|
|
688
|
+
"id": "ru-quotes-rock-n-roll-already-curly",
|
|
689
|
+
"rule": "quotes",
|
|
690
|
+
"mode": "text",
|
|
691
|
+
"in": "rock ’n’ roll",
|
|
692
|
+
"out": "rock ’n’ roll",
|
|
693
|
+
"note": "quotes.md §3.2, the regression witness for spec 1.1.0's widening of the veto from straight ASCII to the whole NARROW class. This is the output of ru-quotes-rock-n-roll fed back in, so it is that case's idempotency obligation made explicit — and it is the case a naive implementation gets wrong. A port that matches U+0027 only (as spec 0.5.0's withdrawn predicate did, correctly for its own preserve-then-stop outcome) does not recognise the U+2019 marks it just produced: the pair falls through to ordinary NARROW quotation and this input becomes `rock «n» roll`, ru's primary pair at depth 1. Measured, not hypothesised. `apostrophe` emits U+2019 only in place of U+0027 (apostrophe.md §1), so once the veto matches, nothing edits these marks and the form is stable."
|
|
694
|
+
},
|
|
695
|
+
{
|
|
696
|
+
"id": "ru-spaces-dot-word-start-dotfile",
|
|
697
|
+
"rule": "spaces",
|
|
698
|
+
"mode": "text",
|
|
699
|
+
"in": "Проверьте файл .env сейчас",
|
|
700
|
+
"out": "Проверьте файл .env сейчас",
|
|
701
|
+
"note": "spaces.md §3.4 word-start clause (spec 1.2.0): a dot followed by a letter starts a token in every locale; `spaces` reads no locale data (§2)."
|
|
702
|
+
},
|
|
703
|
+
{
|
|
704
|
+
"id": "ru-apostrophe-elision-before-guillemet",
|
|
705
|
+
"rule": "apostrophe",
|
|
706
|
+
"mode": "text",
|
|
707
|
+
"in": "газета l'\"Humanité\"",
|
|
708
|
+
"out": "газета l’«Humanité»",
|
|
709
|
+
"note": "apostrophe.md §3.3 case 3a (spec 1.2.0): a French title in Russian text; `quotes` gives « », and the elision before « converts."
|
|
710
|
+
},
|
|
711
|
+
{
|
|
712
|
+
"id": "ru-nbsp-span-boundary-not-openish-short-word",
|
|
713
|
+
"rule": "nbsp",
|
|
714
|
+
"mode": "html",
|
|
715
|
+
"in": "Он живёт в <em>Москве</em>",
|
|
716
|
+
"out": "Он живёт в <em>Москве</em>",
|
|
717
|
+
"note": "nbsp.md §3.1, §7 item 12 (spec 1.2.0); modes.md §3.3: the −1 span boundary marker is not in `nbsp`'s OPENISH, so N3's following-token guard does not accept it and the space after `в` stays U+0020. This pins the membership split itself: a port that also put the marker in OPENISH would bind here."
|
|
686
718
|
}
|
|
687
719
|
]
|
|
688
720
|
}
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
{
|
|
2
|
-
"spec": "1.
|
|
2
|
+
"spec": "1.2.0",
|
|
3
3
|
"locale": "sv",
|
|
4
4
|
"cases": [
|
|
5
5
|
{
|
|
@@ -752,8 +752,8 @@
|
|
|
752
752
|
"rule": "quotes",
|
|
753
753
|
"mode": "text",
|
|
754
754
|
"in": "rock 'n' roll",
|
|
755
|
-
"out": "rock
|
|
756
|
-
"note": "quotes.md §
|
|
755
|
+
"out": "rock ’n’ roll",
|
|
756
|
+
"note": "spec 1.1.0, quotes.md §3.2's universal medial-n elision veto: by operator decision the idiom is international, so `quotes` declines the pairing in every locale rather than only where a cited quotes.elisionIdioms entry exists, and `apostrophe` (order 50) converts each mark by its own case ladder — leading elision (apostrophe.md §3.3 case 4), then trailing elision (case 3). Same output and same two case decisions as en-us-quotes-rock-n-roll, which reaches them through the cited idiom instead. History: spec 0.5.0's general ambiguous-medial-span veto preserved both marks here as literal U+0027; before 0.5.0 `quotes` paired them as an ordinary depth-1 quotation in this locale's primary glyphs. sv writes the loanword closed up in its own dictionaries, exactly as fi does, so no spaced-form citation exists here either — see fi-quotes-rock-n-roll."
|
|
757
757
|
},
|
|
758
758
|
{
|
|
759
759
|
"id": "sv-quotes-already-correct",
|
|
@@ -1285,6 +1285,14 @@
|
|
|
1285
1285
|
"in": "'90s were fun,' he said",
|
|
1286
1286
|
"out": "”90s were fun,” he said",
|
|
1287
1287
|
"note": "quotes.md §6 row E4s: the straight-input control for E4."
|
|
1288
|
+
},
|
|
1289
|
+
{
|
|
1290
|
+
"id": "sv-apostrophe-elision-before-shared-quote-glyph",
|
|
1291
|
+
"rule": "apostrophe",
|
|
1292
|
+
"mode": "text",
|
|
1293
|
+
"in": "tidningen l'\"Humanité\"",
|
|
1294
|
+
"out": "tidningen l’”Humanité”",
|
|
1295
|
+
"note": "apostrophe.md §3.3 cases 3 and 3a (spec 1.2.0): same as fi; U+201D is in CLOSEISH and the output is unchanged from spec 1.1.0."
|
|
1288
1296
|
}
|
|
1289
1297
|
]
|
|
1290
1298
|
}
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
{
|
|
2
|
-
"spec": "1.
|
|
2
|
+
"spec": "1.2.0",
|
|
3
3
|
"$comment": "Locale resolution input. Algorithm is specified in spec/rules/locale-resolution.md and is identical in every runtime — never delegate it to a platform locale-negotiation library. \"spec\" here must track spec/VERSION exactly — it is not itself the global version source; scripts/validate-spec.mjs enforces the match.",
|
|
4
4
|
"locales": ["en-US", "en-GB", "de-DE", "de-CH", "fr", "fr-CA", "ru", "fi", "sv", "el"],
|
|
5
5
|
"aliases": {
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
{
|
|
2
|
-
"spec": "1.
|
|
2
|
+
"spec": "1.2.0",
|
|
3
3
|
"$comment": "Single source of truth for pipeline order. Rules run in ascending `order`. Disabling a rule removes it from the sequence and never reorders the rest. Rule ids are public API (see docs/ARCHITECTURE.md section 5). \"spec\" here must track spec/VERSION exactly — it is not itself the global version source; scripts/validate-spec.mjs enforces the match.",
|
|
4
4
|
"rules": [
|
|
5
5
|
{
|
|
@@ -48,15 +48,15 @@
|
|
|
48
48
|
"default": "on",
|
|
49
49
|
"modes": ["text", "html", "markdown"],
|
|
50
50
|
"localeData": ["quotes"],
|
|
51
|
-
"summary": "Straight quotes to locale primary/secondary pairs with nesting resolution. Runs before `apostrophe` so that a straight apostrophe is still available as quote evidence. Declines to pair
|
|
51
|
+
"summary": "Straight quotes to locale primary/secondary pairs with nesting resolution. Runs before `apostrophe` so that a straight apostrophe is still available as quote evidence. Declines to pair a medial single-quoted `n` (spec 1.1.0, every locale) or a span matching a cited `quotes.elisionIdioms` entry, so that `apostrophe` converts both marks instead."
|
|
52
52
|
},
|
|
53
53
|
{
|
|
54
54
|
"id": "apostrophe",
|
|
55
55
|
"order": 50,
|
|
56
56
|
"default": "on",
|
|
57
57
|
"modes": ["text", "html", "markdown"],
|
|
58
|
-
"localeData": [
|
|
59
|
-
"summary": "Remaining straight apostrophes to U+2019 without corrupting contractions.
|
|
58
|
+
"localeData": [],
|
|
59
|
+
"summary": "Remaining straight apostrophes to U+2019 without corrupting contractions. Reads no locale data: spec 0.5.0's `quotes.elisionIdioms` lookup existed only for the preserve set withdrawn in spec 1.1.0, so every mark `quotes` declined to claim now reaches this rule's ordinary structural case ladder."
|
|
60
60
|
},
|
|
61
61
|
{
|
|
62
62
|
"id": "symbols",
|
|
@@ -9,7 +9,7 @@ require_relative "../registry"
|
|
|
9
9
|
module Polytypo
|
|
10
10
|
module Engine
|
|
11
11
|
module Rules
|
|
12
|
-
# spec/rules/apostrophe.md (spec
|
|
12
|
+
# spec/rules/apostrophe.md (spec 1.2.0), order 50.
|
|
13
13
|
#
|
|
14
14
|
# Converts a straight U+0027 to U+2019 where it is genuinely an apostrophe: a contraction,
|
|
15
15
|
# an elision, a possessive, or a decade elision. Runs immediately after `quotes` (order 40)
|
|
@@ -17,9 +17,10 @@ module Polytypo
|
|
|
17
17
|
# replacing one code point; the rule never inserts, never deletes, and never touches U+2019
|
|
18
18
|
# itself.
|
|
19
19
|
#
|
|
20
|
-
# As of spec
|
|
21
|
-
#
|
|
22
|
-
#
|
|
20
|
+
# As of spec 1.1.0 this rule reads no locale data and skips no position (apostrophe.md
|
|
21
|
+
# 3.4). Spec 0.5.0's preserve set existed to stop the case ladder from converting the marks
|
|
22
|
+
# quotes had vetoed; conversion is now the specified outcome for exactly those marks --
|
|
23
|
+
# cases 4 and 3 are what turn `rock 'n' roll` into `rock ’n’ roll`, in every locale.
|
|
23
24
|
module Apostrophe
|
|
24
25
|
SQ = 0x27
|
|
25
26
|
RIGHT_SINGLE = 0x2019
|
|
@@ -46,6 +47,10 @@ module Polytypo
|
|
|
46
47
|
0x2D, 0x2011, 0x2013, 0x2014
|
|
47
48
|
].freeze
|
|
48
49
|
|
|
50
|
+
# OPENQUOTE is apostrophe.md 3.1 OPENQUOTE (spec 1.2.0): the quotation glyphs of OPENISH,
|
|
51
|
+
# without its brackets, its dashes and Engine::MARKER. Read only by case 3a.
|
|
52
|
+
OPENQUOTE = [0xAB, 0x2018, 0x201A, 0x201B, 0x201C, 0x201E, 0x201F, 0x2039].freeze
|
|
53
|
+
|
|
49
54
|
BREAK = [0x0A, 0x0D, 0x0B, 0x0C, 0x85, 0x2028, 0x2029, Engine::LINE_MARKER].freeze
|
|
50
55
|
|
|
51
56
|
# SPACELIKE, including Engine::LINE_MARKER as a member of BREAK for every rule everywhere
|
|
@@ -80,6 +85,11 @@ module Polytypo
|
|
|
80
85
|
return true
|
|
81
86
|
end
|
|
82
87
|
|
|
88
|
+
# Case 3a (spec 1.2.0) -- elision before a quotation: `d'« urine »`, `l'“idea”`. Case 3
|
|
89
|
+
# misses it because an opening quotation glyph is not in CLOSEISH; `f'(x)` stays straight
|
|
90
|
+
# because a bracket is not in OPENQUOTE.
|
|
91
|
+
return true if UnicodeUtil.letter?(left) && OPENQUOTE.include?(right)
|
|
92
|
+
|
|
83
93
|
# Case 4 -- leading elision: `'90s`, `'tis`, `'em`, `'n'` (the leading mark). The
|
|
84
94
|
# replacement is U+2019, never U+2018 -- a leading elision is a raised comma, not an
|
|
85
95
|
# opening quotation mark, and `quotes` has already had its chance to claim the mark as a
|
|
@@ -92,20 +102,12 @@ module Polytypo
|
|
|
92
102
|
false
|
|
93
103
|
end
|
|
94
104
|
|
|
95
|
-
def self.scan(cp,
|
|
105
|
+
def self.scan(cp, _locale_data, _ctx)
|
|
96
106
|
n = cp.length
|
|
97
|
-
# apostrophe.md 3.4, spec 0.5.0: positions in the shared ambiguous-medial-span preserve
|
|
98
|
-
# set (an ambiguous shape with no cited elisionIdioms match) are skipped entirely,
|
|
99
|
-
# before left/right are even read -- this rule's own case ladder would otherwise curl
|
|
100
|
-
# both marks of e.g. `rock 'n' roll` independently and silently defeat quotes'
|
|
101
|
-
# deliberate veto.
|
|
102
|
-
idioms = (locale_data["quotes"] || {})["elisionIdioms"] || []
|
|
103
|
-
preserve = QuoteAmbiguity.compute_preserve_indices(cp, idioms)
|
|
104
107
|
|
|
105
108
|
edits = []
|
|
106
109
|
(0...n).each do |i|
|
|
107
110
|
next unless cp[i] == SQ
|
|
108
|
-
next if preserve.key?(i)
|
|
109
111
|
|
|
110
112
|
left = QuoteAmbiguity.at(cp, i - 1)
|
|
111
113
|
right = QuoteAmbiguity.at(cp, i + 1)
|
|
@@ -9,7 +9,7 @@ require_relative "../../errors"
|
|
|
9
9
|
module Polytypo
|
|
10
10
|
module Engine
|
|
11
11
|
module Rules
|
|
12
|
-
# `nbsp` -- spec/rules/nbsp.md (spec
|
|
12
|
+
# `nbsp` -- spec/rules/nbsp.md (spec 1.2.0), order 70 (last).
|
|
13
13
|
#
|
|
14
14
|
# Ten sub-rules, N1 through N10, evaluated in that fixed order (3.2); each produces
|
|
15
15
|
# candidate edits keyed by the index of the space (or insertion point) it claims, and the
|
|
@@ -54,16 +54,7 @@ module Polytypo
|
|
|
54
54
|
end
|
|
55
55
|
|
|
56
56
|
# BREAK (nbsp.md 3.1), including LINE_MARKER -- a member of BREAK for every rule,
|
|
57
|
-
# everywhere (modes.md 3.2).
|
|
58
|
-
#
|
|
59
|
-
# MARKER is deliberately NOT a member here, and openish?/closeish? below do not add it
|
|
60
|
-
# either. modes.md 3.3's table says nbsp's OPENISH/CLOSEISH include the span-boundary
|
|
61
|
-
# MARKER, the same way quotes' and apostrophe's do -- but the JS reference
|
|
62
|
-
# implementation's own isBreak/isOpenish/isCloseish never test for MARKER, only
|
|
63
|
-
# LINE_MARKER via isBreak. That is a documented, preserved discrepancy (tracked, not
|
|
64
|
-
# resolved, in the roadmap), carried forward here unchanged rather than "corrected"
|
|
65
|
-
# against the table, for cross-runtime consistency with the already-shipped JS, Python
|
|
66
|
-
# and Go ports.
|
|
57
|
+
# everywhere (modes.md 3.2). MARKER is not a member of BREAK; see closeish? below.
|
|
67
58
|
def self.break?(cp)
|
|
68
59
|
cp == 0x0A || cp == 0x0D || cp == 0x0B || cp == 0x0C || cp == 0x85 ||
|
|
69
60
|
cp == 0x2028 || cp == 0x2029 || cp == LINE_MARKER
|
|
@@ -197,13 +188,19 @@ module Polytypo
|
|
|
197
188
|
end
|
|
198
189
|
|
|
199
190
|
# OPENISH / CLOSEISH (nbsp.md 3.1): the ASCII brackets plus every locale quote glyph.
|
|
200
|
-
#
|
|
191
|
+
#
|
|
192
|
+
# Since spec 1.2.0 the span boundary MARKER is a member of CLOSEISH and not of OPENISH
|
|
193
|
+
# (nbsp.md 3.1 and 7 item 12, modes.md 3.3). CLOSEISH is read only by N1/N2's right-context
|
|
194
|
+
# guard, where membership lets French `<strong>gel :</strong>` keep its no-break space
|
|
195
|
+
# after `spaces` deletes the U+0020. OPENISH stays marker-free: with the marker in it,
|
|
196
|
+
# N1/N2's quote-glyph guard would decline `<em>non</em> ! Oui`, and the left-boundary
|
|
197
|
+
# tests of N3, N7, N9, N10 would widen at span edges.
|
|
201
198
|
def self.openish?(prep, cp)
|
|
202
199
|
prep.opens.include?(cp)
|
|
203
200
|
end
|
|
204
201
|
|
|
205
202
|
def self.closeish?(prep, cp)
|
|
206
|
-
prep.closes.include?(cp)
|
|
203
|
+
cp == MARKER || prep.closes.include?(cp)
|
|
207
204
|
end
|
|
208
205
|
|
|
209
206
|
def self.mark?(prep, cp)
|
|
@@ -26,14 +26,17 @@ module Polytypo
|
|
|
26
26
|
# the right one as a trailing possessive/elision, apostrophe.md 3.3 cases 3/4) must not
|
|
27
27
|
# convert them either.
|
|
28
28
|
module QuoteAmbiguity
|
|
29
|
-
#
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
#
|
|
35
|
-
#
|
|
36
|
-
#
|
|
29
|
+
# LOWER_N / UPPER_N -- quotes.md 3.2's one enclosed code point, in either case.
|
|
30
|
+
LOWER_N = 0x6E
|
|
31
|
+
UPPER_N = 0x4E
|
|
32
|
+
|
|
33
|
+
# NARROW -- quotes.md 3.1 NARROW -- every glyph an elision mark may appear as across
|
|
34
|
+
# pipeline passes (straight, or already curled by an earlier pass). Shared with quotes.rb
|
|
35
|
+
# so the two rules cannot define two slightly different NARROW sets. For the universal
|
|
36
|
+
# medial-n veto, matching the whole class is an IDEMPOTENCY obligation rather than a
|
|
37
|
+
# preference: its marks are converted to U+2019 by apostrophe, so a straight-ASCII-only
|
|
38
|
+
# predicate would not recognise its own output and pass 2 would pair `rock ’n’ roll`
|
|
39
|
+
# as an ordinary NARROW quotation on the next run.
|
|
37
40
|
NARROW = [0x27, 0x2018, 0x2019, 0x201A, 0x201B, 0x2039, 0x203A].freeze
|
|
38
41
|
|
|
39
42
|
# INLINE_SPACE -- quotes.md 3.1 INLINE-SPACE, deliberately excluding BREAK/MARKER/
|
|
@@ -41,9 +44,6 @@ module Polytypo
|
|
|
41
44
|
# same anchor the elisionIdioms matcher already uses.
|
|
42
45
|
INLINE_SPACE = [0x20, 0x09, 0xA0, 0x202F, 0x2007, 0x2009, 0x200A].freeze
|
|
43
46
|
|
|
44
|
-
MIN_ENCLOSED = 1
|
|
45
|
-
MAX_ENCLOSED = 3
|
|
46
|
-
|
|
47
47
|
def self.narrow?(cp)
|
|
48
48
|
NARROW.include?(cp)
|
|
49
49
|
end
|
|
@@ -187,28 +187,27 @@ module Polytypo
|
|
|
187
187
|
|
|
188
188
|
# compute_ambiguous_shape_indices is the general ambiguous-medial-span shape,
|
|
189
189
|
# locale-independent (quotes.md 3.2, spec 0.5.0): a pair of straight ASCII single quotes
|
|
190
|
-
# (
|
|
191
|
-
# point immediately outside each mark. Both
|
|
192
|
-
#
|
|
193
|
-
#
|
|
194
|
-
#
|
|
195
|
-
# short.
|
|
190
|
+
# (spec 1.1.0) a pair of NARROW marks enclosing exactly one code point, U+006E or
|
|
191
|
+
# U+004E, with at least one INLINE-SPACE code point immediately outside each mark. Both
|
|
192
|
+
# mark positions are returned for every match. A superset of
|
|
193
|
+
# compute_idiom_matched_indices's output for every idiom whose elided field is a single
|
|
194
|
+
# n (true of every idiom shipped so far), but computed independently rather than assumed,
|
|
195
|
+
# since a future idiom's elided field is not required to be that short.
|
|
196
196
|
def self.compute_ambiguous_shape_indices(cp)
|
|
197
197
|
ambiguous = {}
|
|
198
198
|
n = cp.length
|
|
199
199
|
|
|
200
200
|
(0...n).each do |i|
|
|
201
|
-
next unless at(cp, i)
|
|
201
|
+
next unless narrow?(at(cp, i))
|
|
202
202
|
|
|
203
203
|
l_lit = at(cp, i - 1)
|
|
204
204
|
next if l_lit == Engine::NONE || !inline_space?(l_lit)
|
|
205
205
|
|
|
206
|
-
|
|
207
|
-
|
|
208
|
-
next if k < MIN_ENCLOSED
|
|
206
|
+
enclosed = at(cp, i + 1)
|
|
207
|
+
next unless [LOWER_N, UPPER_N].include?(enclosed)
|
|
209
208
|
|
|
210
|
-
j = i +
|
|
211
|
-
next unless at(cp, j)
|
|
209
|
+
j = i + 2
|
|
210
|
+
next unless narrow?(at(cp, j))
|
|
212
211
|
|
|
213
212
|
r_lit = at(cp, j + 1)
|
|
214
213
|
next if r_lit == Engine::NONE || !inline_space?(r_lit)
|
|
@@ -219,22 +218,6 @@ module Polytypo
|
|
|
219
218
|
|
|
220
219
|
ambiguous
|
|
221
220
|
end
|
|
222
|
-
|
|
223
|
-
# compute_preserve_indices is the set of straight-ASCII-quote index positions
|
|
224
|
-
# apostrophe.md 3.4 requires `apostrophe` to skip byte-identically: ambiguous-shaped, but
|
|
225
|
-
# with no matching cited idiom. A position with a matching idiom is not in this set --
|
|
226
|
-
# apostrophe's ordinary case ladder still curls it, exactly as spec 0.4.0-0.4.1 did.
|
|
227
|
-
def self.compute_preserve_indices(cp, idioms)
|
|
228
|
-
ambiguous = compute_ambiguous_shape_indices(cp)
|
|
229
|
-
return ambiguous if ambiguous.empty?
|
|
230
|
-
|
|
231
|
-
idiom_matched = compute_idiom_matched_indices(cp, idioms)
|
|
232
|
-
return ambiguous if idiom_matched.empty?
|
|
233
|
-
|
|
234
|
-
ambiguous.each_with_object({}) do |(idx, _), preserve|
|
|
235
|
-
preserve[idx] = true unless idiom_matched.key?(idx)
|
|
236
|
-
end
|
|
237
|
-
end
|
|
238
221
|
end
|
|
239
222
|
end
|
|
240
223
|
end
|
|
@@ -107,10 +107,17 @@ module Polytypo
|
|
|
107
107
|
# after any edit had been applied would break the Chicago spaced ellipsis
|
|
108
108
|
# "Hello . . .", where every dot is a lone dot at decision time and all three spaces
|
|
109
109
|
# must still strip.
|
|
110
|
+
#
|
|
111
|
+
# Spec 1.2.0's word-start clause: a single dot directly followed by a letter or an ASCII
|
|
112
|
+
# digit starts a token (".NET", ".env", ".5"), so "Use .NET" keeps its space. A span
|
|
113
|
+
# boundary marker after the dot is neither, so "a .</em>" still strips.
|
|
110
114
|
def self.lone_dot?(cp, e)
|
|
111
115
|
return true if at(cp, e) != FULL_STOP
|
|
112
116
|
|
|
113
|
-
|
|
117
|
+
after = at(cp, e + 1)
|
|
118
|
+
return false if UnicodeUtil.letter?(after) || digit_ascii?(after)
|
|
119
|
+
|
|
120
|
+
!dotlike?(after)
|
|
114
121
|
end
|
|
115
122
|
|
|
116
123
|
def self.digit_ascii?(value)
|
data/lib/polytypo/version.rb
CHANGED