polytypo 1.1.0 → 1.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (66) hide show
  1. checksums.yaml +4 -4
  2. data/README.md +33 -1
  3. data/lib/polytypo/data/VERSION +1 -1
  4. data/lib/polytypo/data/fixtures/cs.json +161 -0
  5. data/lib/polytypo/data/fixtures/de-CH.json +9 -1
  6. data/lib/polytypo/data/fixtures/de-DE.json +227 -6
  7. data/lib/polytypo/data/fixtures/el.json +9 -1
  8. data/lib/polytypo/data/fixtures/en-GB.json +28 -1
  9. data/lib/polytypo/data/fixtures/en-US.json +690 -1
  10. data/lib/polytypo/data/fixtures/es.json +193 -0
  11. data/lib/polytypo/data/fixtures/fi.json +9 -1
  12. data/lib/polytypo/data/fixtures/fr-CA.json +50 -1
  13. data/lib/polytypo/data/fixtures/fr.json +282 -1
  14. data/lib/polytypo/data/fixtures/it.json +161 -0
  15. data/lib/polytypo/data/fixtures/locale-resolution.json +76 -4
  16. data/lib/polytypo/data/fixtures/nl.json +121 -0
  17. data/lib/polytypo/data/fixtures/pl.json +137 -0
  18. data/lib/polytypo/data/fixtures/pt-BR.json +156 -0
  19. data/lib/polytypo/data/fixtures/pt-PT.json +156 -0
  20. data/lib/polytypo/data/fixtures/ru.json +47 -1
  21. data/lib/polytypo/data/fixtures/sv.json +9 -1
  22. data/lib/polytypo/data/fixtures/uk.json +153 -0
  23. data/lib/polytypo/data/locales/cs.json +90 -0
  24. data/lib/polytypo/data/locales/de-DE.json +7 -2
  25. data/lib/polytypo/data/locales/en-US.json +3 -3
  26. data/lib/polytypo/data/locales/es.json +111 -0
  27. data/lib/polytypo/data/locales/fr-CA.json +7 -1
  28. data/lib/polytypo/data/locales/fr.json +7 -1
  29. data/lib/polytypo/data/locales/it.json +95 -0
  30. data/lib/polytypo/data/locales/nl.json +84 -0
  31. data/lib/polytypo/data/locales/pl.json +96 -0
  32. data/lib/polytypo/data/locales/pt-BR.json +82 -0
  33. data/lib/polytypo/data/locales/pt-PT.json +84 -0
  34. data/lib/polytypo/data/locales/registry.json +23 -3
  35. data/lib/polytypo/data/locales/ru.json +2 -2
  36. data/lib/polytypo/data/locales/uk.json +130 -0
  37. data/lib/polytypo/data/rules/analyze.md +157 -0
  38. data/lib/polytypo/data/rules/apostrophe.md +432 -0
  39. data/lib/polytypo/data/rules/dashes.md +128 -37
  40. data/lib/polytypo/data/rules/ellipsis.md +271 -0
  41. data/lib/polytypo/data/rules/hyphen.md +353 -0
  42. data/lib/polytypo/data/rules/locale-resolution.md +239 -0
  43. data/lib/polytypo/data/rules/modes.md +1281 -0
  44. data/lib/polytypo/data/rules/nbsp.md +1157 -0
  45. data/lib/polytypo/data/rules/order.json +11 -11
  46. data/lib/polytypo/data/rules/pipeline-idempotency.md +605 -0
  47. data/lib/polytypo/data/rules/quotes.md +1324 -0
  48. data/lib/polytypo/data/rules/ranges.md +489 -0
  49. data/lib/polytypo/data/rules/spaces.md +649 -0
  50. data/lib/polytypo/data/rules/symbols.md +540 -0
  51. data/lib/polytypo/data/schema/fixtures.schema.json +18 -3
  52. data/lib/polytypo/engine/origin.rb +75 -0
  53. data/lib/polytypo/engine/pipeline.rb +72 -1
  54. data/lib/polytypo/engine/rules/apostrophe.rb +10 -1
  55. data/lib/polytypo/engine/rules/dash_shared.rb +85 -3
  56. data/lib/polytypo/engine/rules/dashes.rb +4 -1
  57. data/lib/polytypo/engine/rules/nbsp.rb +53 -20
  58. data/lib/polytypo/engine/rules/ranges.rb +24 -20
  59. data/lib/polytypo/engine/rules/spaces.rb +8 -1
  60. data/lib/polytypo/errors.rb +3 -0
  61. data/lib/polytypo/modes/runner.rb +17 -0
  62. data/lib/polytypo/modes/spans.rb +30 -2
  63. data/lib/polytypo/modes/yaml.rb +312 -0
  64. data/lib/polytypo/version.rb +1 -1
  65. data/lib/polytypo.rb +126 -15
  66. metadata +31 -1
@@ -0,0 +1,156 @@
1
+ {
2
+ "spec": "1.3.0",
3
+ "locale": "pt-BR",
4
+ "cases": [
5
+ {
6
+ "id": "pt-br-quotes-primary",
7
+ "rule": "quotes",
8
+ "mode": "text",
9
+ "in": "Ele disse \"bom dia\" e saiu.",
10
+ "out": "Ele disse “bom dia” e saiu.",
11
+ "note": "quotes.primary com innerSpace \"none\". O uso brasileiro atestado pelo Manual de Redação da Presidência da República compõe as citações com aspas curvas duplas e nunca com angulares. É a diferença central em relação a pt-PT."
12
+ },
13
+ {
14
+ "id": "pt-br-quotes-nested",
15
+ "rule": "quotes",
16
+ "mode": "text",
17
+ "in": "Escreveu: \"Disse-me 'chego já' e desapareceu.\"",
18
+ "out": "Escreveu: “Disse-me ‘chego já’ e desapareceu.”",
19
+ "note": "Primeiro e segundo níveis de aspas. O terceiro nível que as fontes descrevem não é exprimível no esquema, que só tem primary e secondary."
20
+ },
21
+ {
22
+ "id": "pt-br-dashes-parenthetical",
23
+ "rule": "dashes",
24
+ "mode": "text",
25
+ "in": "As condições - ordenado e subvenções - eram boas.",
26
+ "out": "As condições – ordenado e subvenções – eram boas.",
27
+ "note": "dash.parenthetical = \"en-spaced\": o travessão substitui parênteses e leva um espaço ordinário de cada lado. A largura foi verificada extraindo a camada de texto do PDF da fonte, não julgada a olho sobre a página rasterizada."
28
+ },
29
+ {
30
+ "id": "pt-br-ranges-hyphen-kept",
31
+ "rule": "ranges",
32
+ "mode": "text",
33
+ "rules": {
34
+ "ranges": true
35
+ },
36
+ "in": "o programa para 1996-1997",
37
+ "out": "o programa para 1996-1997",
38
+ "note": "ranges.md §2: com dash.range = \"none\" a regra não emite nada — nem sequer os U+2060 que um intervalo convertido carrega noutras locales — e o U+002D escrito pelo autor sobrevive byte a byte."
39
+ },
40
+ {
41
+ "id": "pt-br-ranges-slash-untouched",
42
+ "rule": "ranges",
43
+ "mode": "text",
44
+ "rules": {
45
+ "ranges": true
46
+ },
47
+ "in": "o ano letivo de 1990/1991",
48
+ "out": "o ano letivo de 1990/1991",
49
+ "note": "ranges.md §3: a barra não é um token de intervalo para esta regra. Fica fixado que o outro ramo da convenção portuguesa — período que não abrange os dois anos completos — também não é tocado."
50
+ },
51
+ {
52
+ "id": "pt-br-ellipsis-three-dots",
53
+ "rule": "ellipsis",
54
+ "mode": "text",
55
+ "in": "Espere... o quê?",
56
+ "out": "Espere… o quê?",
57
+ "note": "ellipsis.md §6: três U+002E passam a U+2026. Nada acontece antes de «?», porque beforePunctuation e narrowBeforePunctuation estão vazios — ao contrário de fr."
58
+ },
59
+ {
60
+ "id": "pt-br-ellipsis-not-abbreviated-after-terminal",
61
+ "rule": "ellipsis",
62
+ "mode": "text",
63
+ "in": "É o dianho!..",
64
+ "out": "É o dianho!…",
65
+ "note": "abbreviatedAfterTerminal = false: a forma abreviada de dois pontos que o russo usa não existe em português, pelo que «!..» é um lapso por «!…». A fonte europeia imprime exatamente «É o dianho!…»."
66
+ },
67
+ {
68
+ "id": "pt-br-hyphen-noop-compounds",
69
+ "rule": "hyphen",
70
+ "mode": "text",
71
+ "in": "Uma reunião na segunda-feira sobre o salmão-do-atlântico.",
72
+ "out": "Uma reunião na segunda-feira sobre o salmão-do-atlântico.",
73
+ "note": "hyphen.md §2: as três listas estão vazias porque o Acordo Ortográfico (Base XX) manda QUEBRAR a linha no hífen do composto e repeti-lo na linha seguinte. Um port que ligue compostos portugueses com U+2011 falha este caso."
74
+ },
75
+ {
76
+ "id": "pt-br-hyphen-noop-enclisis",
77
+ "rule": "hyphen",
78
+ "mode": "text",
79
+ "in": "far-se-á e exigem-lhe",
80
+ "out": "far-se-á e exigem-lhe",
81
+ "note": "Ênclise e mesóclise, exemplos literais do Manual de Redação da Presidência da República, que prescreve a mesma repetição do hífen na translineação."
82
+ },
83
+ {
84
+ "id": "pt-br-nbsp-percent",
85
+ "rule": "nbsp",
86
+ "mode": "text",
87
+ "in": "7 % do volume de negócios",
88
+ "out": "7 % do volume de negócios",
89
+ "note": "N5 converte um espaço já escrito. O português escreve «7 %» com espaço, ao contrário do grego, que o escreve colado."
90
+ },
91
+ {
92
+ "id": "pt-br-nbsp-celsius",
93
+ "rule": "nbsp",
94
+ "mode": "text",
95
+ "in": "a temperatura atingiu hoje os 39 °C",
96
+ "out": "a temperatura atingiu hoje os 39 °C",
97
+ "note": "Exemplo literal do ponto 10.9.1 do Código de Redação Interinstitucional."
98
+ },
99
+ {
100
+ "id": "pt-br-nbsp-hour-is-not-a-unit",
101
+ "rule": "nbsp",
102
+ "mode": "text",
103
+ "in": "eram as 18h30",
104
+ "out": "eram as 18h30",
105
+ "note": "O reparo citado: «h» escreve-se «sempre sem ponto, sem espaços» (ponto 10.9.1), pelo que está deliberadamente ausente de beforeUnits. Este caso fixa essa ausência para que ninguém a preencha por analogia."
106
+ },
107
+ {
108
+ "id": "pt-br-nbsp-abbreviation",
109
+ "rule": "nbsp",
110
+ "mode": "text",
111
+ "in": "p. ex. isto",
112
+ "out": "p. ex. isto",
113
+ "note": "N4 liga o espaço interno da abreviatura listada «p. ex.», que o anexo A3 manda preferir a «e. g.»."
114
+ },
115
+ {
116
+ "id": "pt-br-nbsp-before-number",
117
+ "rule": "nbsp",
118
+ "mode": "text",
119
+ "in": "ver p. 24",
120
+ "out": "ver p. 24",
121
+ "note": "N9: «p. 24» é o exemplo impresso. «n.º» NÃO está na lista, e por citação e não por esquecimento — a nota (3) do ponto 6.4 prescreve um «o» sobrescrito e recusa expressamente U+00BA e U+00B0."
122
+ },
123
+ {
124
+ "id": "pt-br-nbsp-no-space-before-colon",
125
+ "rule": "nbsp",
126
+ "mode": "text",
127
+ "in": "As principais cidades de Portugal são: Lisboa, Porto e Coimbra.",
128
+ "out": "As principais cidades de Portugal são: Lisboa, Porto e Coimbra.",
129
+ "note": "beforePunctuation e narrowBeforePunctuation vazios: nada é inserido antes de «:». Frase literal do ponto 10.4.4. Um port que aplique a prática francesa a todas as locales falha aqui."
130
+ },
131
+ {
132
+ "id": "pt-br-spaces-brackets-and-final-stop",
133
+ "rule": "spaces",
134
+ "mode": "text",
135
+ "in": "O texto ( corrente ) não muda .",
136
+ "out": "O texto (corrente) não muda.",
137
+ "note": "spaces.md §3: os espaços interiores dos parênteses desaparecem e a série anterior ao ponto final reduz-se. Regra independente da locale, presente para que esta tenha um caso por cada regra canónica."
138
+ },
139
+ {
140
+ "id": "pt-br-symbols-copyright-and-dimensions",
141
+ "rule": "symbols",
142
+ "mode": "text",
143
+ "in": "Copyright (c) 2026, formato 40x60 cm",
144
+ "out": "Copyright © 2026, formato 40×60 cm",
145
+ "note": "symbols.md: «(c)» passa a U+00A9 e o «x» entre algarismos a U+00D7. O U+00A0 antes de «cm» é obra de nbsp N5, não de symbols."
146
+ },
147
+ {
148
+ "id": "pt-br-apostrophe-elision",
149
+ "rule": "apostrophe",
150
+ "mode": "text",
151
+ "in": "pinga d'água",
152
+ "out": "pinga d’água",
153
+ "note": "U+0027 passa a U+2019. A regra não lê dados de locale; o caso certifica-a para esta locale."
154
+ }
155
+ ]
156
+ }
@@ -0,0 +1,156 @@
1
+ {
2
+ "spec": "1.3.0",
3
+ "locale": "pt-PT",
4
+ "cases": [
5
+ {
6
+ "id": "pt-pt-quotes-primary",
7
+ "rule": "quotes",
8
+ "mode": "text",
9
+ "in": "Ele disse \"bom dia\" e saiu.",
10
+ "out": "Ele disse «bom dia» e saiu.",
11
+ "note": "quotes.primary com innerSpace \"none\". O primeiro nível é o das aspas angulares (ponto 10.4.10 do Código de Redação Interinstitucional). É a diferença central em relação a pt-BR, que usa aspas curvas duplas."
12
+ },
13
+ {
14
+ "id": "pt-pt-quotes-nested",
15
+ "rule": "quotes",
16
+ "mode": "text",
17
+ "in": "Escreveu: \"Disse-me 'chego já' e desapareceu.\"",
18
+ "out": "Escreveu: «Disse-me “chego já” e desapareceu.»",
19
+ "note": "Primeiro e segundo níveis de aspas. O terceiro nível que as fontes descrevem não é exprimível no esquema, que só tem primary e secondary."
20
+ },
21
+ {
22
+ "id": "pt-pt-dashes-parenthetical",
23
+ "rule": "dashes",
24
+ "mode": "text",
25
+ "in": "As condições - ordenado e subvenções - eram boas.",
26
+ "out": "As condições — ordenado e subvenções — eram boas.",
27
+ "note": "dash.parenthetical = \"em-spaced\": o travessão substitui parênteses e leva um espaço ordinário de cada lado. A largura foi verificada extraindo a camada de texto do PDF da fonte, não julgada a olho sobre a página rasterizada."
28
+ },
29
+ {
30
+ "id": "pt-pt-ranges-hyphen-kept",
31
+ "rule": "ranges",
32
+ "mode": "text",
33
+ "rules": {
34
+ "ranges": true
35
+ },
36
+ "in": "o programa para 1996-1997",
37
+ "out": "o programa para 1996-1997",
38
+ "note": "ranges.md §2: com dash.range = \"none\" a regra não emite nada — nem sequer os U+2060 que um intervalo convertido carrega noutras locales — e o U+002D escrito pelo autor sobrevive byte a byte."
39
+ },
40
+ {
41
+ "id": "pt-pt-ranges-slash-untouched",
42
+ "rule": "ranges",
43
+ "mode": "text",
44
+ "rules": {
45
+ "ranges": true
46
+ },
47
+ "in": "o ano letivo de 1990/1991",
48
+ "out": "o ano letivo de 1990/1991",
49
+ "note": "ranges.md §3: a barra não é um token de intervalo para esta regra. Fica fixado que o outro ramo da convenção portuguesa — período que não abrange os dois anos completos — também não é tocado."
50
+ },
51
+ {
52
+ "id": "pt-pt-ellipsis-three-dots",
53
+ "rule": "ellipsis",
54
+ "mode": "text",
55
+ "in": "Espere... o quê?",
56
+ "out": "Espere… o quê?",
57
+ "note": "ellipsis.md §6: três U+002E passam a U+2026. Nada acontece antes de «?», porque beforePunctuation e narrowBeforePunctuation estão vazios — ao contrário de fr."
58
+ },
59
+ {
60
+ "id": "pt-pt-ellipsis-not-abbreviated-after-terminal",
61
+ "rule": "ellipsis",
62
+ "mode": "text",
63
+ "in": "É o dianho!..",
64
+ "out": "É o dianho!…",
65
+ "note": "abbreviatedAfterTerminal = false: a forma abreviada de dois pontos que o russo usa não existe em português, pelo que «!..» é um lapso por «!…». A fonte europeia imprime exatamente «É o dianho!…»."
66
+ },
67
+ {
68
+ "id": "pt-pt-hyphen-noop-compounds",
69
+ "rule": "hyphen",
70
+ "mode": "text",
71
+ "in": "Uma reunião na segunda-feira sobre o salmão-do-atlântico.",
72
+ "out": "Uma reunião na segunda-feira sobre o salmão-do-atlântico.",
73
+ "note": "hyphen.md §2: as três listas estão vazias porque o Acordo Ortográfico (Base XX) manda QUEBRAR a linha no hífen do composto e repeti-lo na linha seguinte. Um port que ligue compostos portugueses com U+2011 falha este caso."
74
+ },
75
+ {
76
+ "id": "pt-pt-hyphen-noop-enclisis",
77
+ "rule": "hyphen",
78
+ "mode": "text",
79
+ "in": "far-se-á e exigem-lhe",
80
+ "out": "far-se-á e exigem-lhe",
81
+ "note": "Ênclise e mesóclise, exemplos literais do Manual de Redação da Presidência da República, que prescreve a mesma repetição do hífen na translineação."
82
+ },
83
+ {
84
+ "id": "pt-pt-nbsp-percent",
85
+ "rule": "nbsp",
86
+ "mode": "text",
87
+ "in": "7 % do volume de negócios",
88
+ "out": "7 % do volume de negócios",
89
+ "note": "N5 converte um espaço já escrito. O português escreve «7 %» com espaço, ao contrário do grego, que o escreve colado."
90
+ },
91
+ {
92
+ "id": "pt-pt-nbsp-celsius",
93
+ "rule": "nbsp",
94
+ "mode": "text",
95
+ "in": "a temperatura atingiu hoje os 39 °C",
96
+ "out": "a temperatura atingiu hoje os 39 °C",
97
+ "note": "Exemplo literal do ponto 10.9.1 do Código de Redação Interinstitucional."
98
+ },
99
+ {
100
+ "id": "pt-pt-nbsp-hour-is-not-a-unit",
101
+ "rule": "nbsp",
102
+ "mode": "text",
103
+ "in": "eram as 18h30",
104
+ "out": "eram as 18h30",
105
+ "note": "O reparo citado: «h» escreve-se «sempre sem ponto, sem espaços» (ponto 10.9.1), pelo que está deliberadamente ausente de beforeUnits. Este caso fixa essa ausência para que ninguém a preencha por analogia."
106
+ },
107
+ {
108
+ "id": "pt-pt-nbsp-abbreviation",
109
+ "rule": "nbsp",
110
+ "mode": "text",
111
+ "in": "p. ex. isto",
112
+ "out": "p. ex. isto",
113
+ "note": "N4 liga o espaço interno da abreviatura listada «p. ex.», que o anexo A3 manda preferir a «e. g.»."
114
+ },
115
+ {
116
+ "id": "pt-pt-nbsp-before-number",
117
+ "rule": "nbsp",
118
+ "mode": "text",
119
+ "in": "ver p. 24",
120
+ "out": "ver p. 24",
121
+ "note": "N9: «p. 24» é o exemplo impresso. «n.º» NÃO está na lista, e por citação e não por esquecimento — a nota (3) do ponto 6.4 prescreve um «o» sobrescrito e recusa expressamente U+00BA e U+00B0."
122
+ },
123
+ {
124
+ "id": "pt-pt-nbsp-no-space-before-colon",
125
+ "rule": "nbsp",
126
+ "mode": "text",
127
+ "in": "As principais cidades de Portugal são: Lisboa, Porto e Coimbra.",
128
+ "out": "As principais cidades de Portugal são: Lisboa, Porto e Coimbra.",
129
+ "note": "beforePunctuation e narrowBeforePunctuation vazios: nada é inserido antes de «:». Frase literal do ponto 10.4.4. Um port que aplique a prática francesa a todas as locales falha aqui."
130
+ },
131
+ {
132
+ "id": "pt-pt-spaces-brackets-and-final-stop",
133
+ "rule": "spaces",
134
+ "mode": "text",
135
+ "in": "O texto ( corrente ) não muda .",
136
+ "out": "O texto (corrente) não muda.",
137
+ "note": "spaces.md §3: os espaços interiores dos parênteses desaparecem e a série anterior ao ponto final reduz-se. Regra independente da locale, presente para que esta tenha um caso por cada regra canónica."
138
+ },
139
+ {
140
+ "id": "pt-pt-symbols-copyright-and-dimensions",
141
+ "rule": "symbols",
142
+ "mode": "text",
143
+ "in": "Copyright (c) 2026, formato 40x60 cm",
144
+ "out": "Copyright © 2026, formato 40×60 cm",
145
+ "note": "symbols.md: «(c)» passa a U+00A9 e o «x» entre algarismos a U+00D7. O U+00A0 antes de «cm» é obra de nbsp N5, não de symbols."
146
+ },
147
+ {
148
+ "id": "pt-pt-apostrophe-elision",
149
+ "rule": "apostrophe",
150
+ "mode": "text",
151
+ "in": "pinga d'água",
152
+ "out": "pinga d’água",
153
+ "note": "U+0027 passa a U+2019. A regra não lê dados de locale; o caso certifica-a para esta locale."
154
+ }
155
+ ]
156
+ }
@@ -1,5 +1,5 @@
1
1
  {
2
- "spec": "1.1.0",
2
+ "spec": "1.3.0",
3
3
  "locale": "ru",
4
4
  "cases": [
5
5
  {
@@ -691,6 +691,52 @@
691
691
  "in": "rock ’n’ roll",
692
692
  "out": "rock ’n’ roll",
693
693
  "note": "quotes.md §3.2, the regression witness for spec 1.1.0's widening of the veto from straight ASCII to the whole NARROW class. This is the output of ru-quotes-rock-n-roll fed back in, so it is that case's idempotency obligation made explicit — and it is the case a naive implementation gets wrong. A port that matches U+0027 only (as spec 0.5.0's withdrawn predicate did, correctly for its own preserve-then-stop outcome) does not recognise the U+2019 marks it just produced: the pair falls through to ordinary NARROW quotation and this input becomes `rock «n» roll`, ru's primary pair at depth 1. Measured, not hypothesised. `apostrophe` emits U+2019 only in place of U+0027 (apostrophe.md §1), so once the veto matches, nothing edits these marks and the form is stable."
694
+ },
695
+ {
696
+ "id": "ru-spaces-dot-word-start-dotfile",
697
+ "rule": "spaces",
698
+ "mode": "text",
699
+ "in": "Проверьте файл .env сейчас",
700
+ "out": "Проверьте файл .env сейчас",
701
+ "note": "spaces.md §3.4 word-start clause (spec 1.2.0): a dot followed by a letter starts a token in every locale; `spaces` reads no locale data (§2)."
702
+ },
703
+ {
704
+ "id": "ru-apostrophe-elision-before-guillemet",
705
+ "rule": "apostrophe",
706
+ "mode": "text",
707
+ "in": "газета l'\"Humanité\"",
708
+ "out": "газета l’«Humanité»",
709
+ "note": "apostrophe.md §3.3 case 3a (spec 1.2.0): a French title in Russian text; `quotes` gives « », and the elision before « converts."
710
+ },
711
+ {
712
+ "id": "ru-nbsp-span-boundary-not-openish-short-word",
713
+ "rule": "nbsp",
714
+ "mode": "html",
715
+ "in": "Он живёт в <em>Москве</em>",
716
+ "out": "Он живёт в <em>Москве</em>",
717
+ "note": "nbsp.md §3.1, §7 item 12 (spec 1.2.0); modes.md §3.3: the −1 span boundary marker is not in `nbsp`'s OPENISH, so N3's following-token guard does not accept it and the space after `в` stays U+0020. This pins the membership split itself: a port that also put the marker in OPENISH would bind here."
718
+ },
719
+ {
720
+ "id": "ru-ranges-closed-up-percent",
721
+ "rule": "ranges",
722
+ "mode": "text",
723
+ "in": "35%-50%",
724
+ "out": "35%⁠—⁠50%",
725
+ "rules": {
726
+ "ranges": true
727
+ },
728
+ "note": "ranges.md §3.2a (spec 1.3.0): the suffix form in a locale whose `dash.range` is em-tight, so the same token produces U+2014 here and U+2013 in `en-US`. The closed-up walk decides candidacy; the locale still decides the glyph."
729
+ },
730
+ {
731
+ "id": "ru-dashes-t1-closed-up-symbol-suffix",
732
+ "rule": "dashes",
733
+ "mode": "text",
734
+ "in": "35%-50%—b",
735
+ "out": "35%-50%—b",
736
+ "rules": {
737
+ "ranges": true
738
+ },
739
+ "note": "ranges.md §3.2a, via dashes.md §3.2 step 8 (spec 1.3.0): the same witness in the suffix direction and in an em-dash locale — the disturbed guard is the range's `after` rather than its `before`. Pinned separately because the two directions are separate code paths in every port. To be precise about \"before the amendment\": the drift is a property of the INTERMEDIATE state — ranges.md §3.2a's widened candidacy without T1's transparency — which never shipped. Under spec 1.2.0 this input was stable, because the token was not a range candidate at all."
694
740
  }
695
741
  ]
696
742
  }
@@ -1,5 +1,5 @@
1
1
  {
2
- "spec": "1.1.0",
2
+ "spec": "1.3.0",
3
3
  "locale": "sv",
4
4
  "cases": [
5
5
  {
@@ -1285,6 +1285,14 @@
1285
1285
  "in": "'90s were fun,' he said",
1286
1286
  "out": "”90s were fun,” he said",
1287
1287
  "note": "quotes.md §6 row E4s: the straight-input control for E4."
1288
+ },
1289
+ {
1290
+ "id": "sv-apostrophe-elision-before-shared-quote-glyph",
1291
+ "rule": "apostrophe",
1292
+ "mode": "text",
1293
+ "in": "tidningen l'\"Humanité\"",
1294
+ "out": "tidningen l’”Humanité”",
1295
+ "note": "apostrophe.md §3.3 cases 3 and 3a (spec 1.2.0): same as fi; U+201D is in CLOSEISH and the output is unchanged from spec 1.1.0."
1288
1296
  }
1289
1297
  ]
1290
1298
  }
@@ -0,0 +1,153 @@
1
+ {
2
+ "spec": "1.3.0",
3
+ "locale": "uk",
4
+ "cases": [
5
+ {
6
+ "id": "uk-quotes-primary",
7
+ "rule": "quotes",
8
+ "mode": "text",
9
+ "in": "Він сказав: \"Це мій Кобзар\".",
10
+ "out": "Він сказав: «Це мій Кобзар».",
11
+ "note": "§ 164 п. 3 правопису: зовнішні лапки — «ялинки» U+00AB/U+00BB, innerSpace = none."
12
+ },
13
+ {
14
+ "id": "uk-quotes-nested",
15
+ "rule": "quotes",
16
+ "mode": "text",
17
+ "in": "\"Це мій 'Кобзар'\", — сказав він.",
18
+ "out": "«Це мій “Кобзар”», — сказав він.",
19
+ "note": "Приклад із самого § 164 п. 3. Внутрішні лапки — U+201C/U+201D, а НЕ „…“: останню той-таки пункт відносить до рукописних текстів. Фікстура існує саме для того, щоб форму не «виправили» на російську за аналогією."
20
+ },
21
+ {
22
+ "id": "uk-ellipsis-abbreviated-after-question",
23
+ "rule": "ellipsis",
24
+ "mode": "text",
25
+ "in": "а щоб селяни?...",
26
+ "out": "а щоб селяни?..",
27
+ "note": "abbreviatedAfterTerminal = true. § 162, Примітка: після знака питання ставимо ДВІ крапки. Приклад із того ж параграфа (К. Гордієнко). Українська — друга після російської локаль із цим значенням."
28
+ },
29
+ {
30
+ "id": "uk-ellipsis-abbreviated-after-exclamation",
31
+ "rule": "ellipsis",
32
+ "mode": "text",
33
+ "in": "Рокочуть ріки ясноводі!...",
34
+ "out": "Рокочуть ріки ясноводі!..",
35
+ "note": "Другий приклад тієї самої Примітки (М. Рильський): після знака оклику так само дві крапки."
36
+ },
37
+ {
38
+ "id": "uk-ellipsis-ordinary",
39
+ "rule": "ellipsis",
40
+ "mode": "text",
41
+ "in": "Літак... Димки...",
42
+ "out": "Літак… Димки…",
43
+ "note": "Поза постпозицією до ?/! три крапки стають U+2026, як в усіх локалях."
44
+ },
45
+ {
46
+ "id": "uk-dashes-parenthetical",
47
+ "rule": "dashes",
48
+ "mode": "text",
49
+ "in": "Це - наш дім.",
50
+ "out": "Це — наш дім.",
51
+ "note": "dash.parenthetical = em-spaced: U+2014 з відбивкою. Довжина з заголовка § 161 «ТИРЕ (—)»; відбивка — з протиставлення в Примітці п. 14, і це в файлі позначено як висновок, а не як цитата."
52
+ },
53
+ {
54
+ "id": "uk-ranges-em-tight",
55
+ "rule": "ranges",
56
+ "mode": "text",
57
+ "rules": {
58
+ "ranges": true
59
+ },
60
+ "in": "на сторінках 1-10",
61
+ "out": "на сторінках 1⁠—⁠10",
62
+ "note": "ranges.md §3: dash.range = em-tight дає U+2014 без відступів, загорнуте в U+2060. Приклад із § 161 I п. 14, Примітка: «на сторінках 1—10»."
63
+ },
64
+ {
65
+ "id": "uk-hyphen-compound-preposition",
66
+ "rule": "hyphen",
67
+ "mode": "text",
68
+ "in": "з-під столу і будь-хто",
69
+ "out": "з‑під столу і будь‑хто",
70
+ "note": "hyphen: складений прийменник (§ 42 п. 2) і частка (§ 44 п. 3) зв'язуються через U+2011. Для ЦИХ класів нерозривність — рішення проєкту, а не норма: §§ 63—64 про них мовчать. Пор. наступний випадок, де норма пряма."
71
+ },
72
+ {
73
+ "id": "uk-hyphen-graphic-shortening",
74
+ "rule": "hyphen",
75
+ "mode": "text",
76
+ "in": "вид-во та ін-т",
77
+ "out": "вид‑во та ін‑т",
78
+ "note": "Тут нерозривність САМА Є НОРМОЮ: § 64 п. 4 прямо забороняє розривати графічні скорочення, а § 62 п. 2 закриває клас переліком. Цим українська відрізняється від польської, де поділ у місці дефіса приписаний."
79
+ },
80
+ {
81
+ "id": "uk-nbsp-units-and-year",
82
+ "rule": "nbsp",
83
+ "mode": "text",
84
+ "in": "150 га і 2008 р.",
85
+ "out": "150 га і 2008 р.",
86
+ "note": "N5: обидва приклади дослівні з § 64 п. 2. «р.» безпечне там, де російське «г.» не було: українське «р.» означає тільки «рік» і зв'язується вліво."
87
+ },
88
+ {
89
+ "id": "uk-nbsp-initials-chain",
90
+ "rule": "nbsp",
91
+ "mode": "text",
92
+ "in": "Т. Г. Шевченко",
93
+ "out": "Т. Г. Шевченко",
94
+ "note": "initialBinding = chain, приклад із § 64 п. 1. Значення взято з того, що пункт наводить послідовність ДВОХ ініціалів і не дає прикладу одного."
95
+ },
96
+ {
97
+ "id": "uk-nbsp-before-word",
98
+ "rule": "nbsp",
99
+ "mode": "text",
100
+ "in": "проф. Гончаренко",
101
+ "out": "проф. Гончаренко",
102
+ "note": "N10 beforeWord із § 64 п. 1. «п.» і «гр.» з того самого пункту свідомо НЕ внесено: «п.» — також «пункт», «гр.» — також «градус»."
103
+ },
104
+ {
105
+ "id": "uk-nbsp-abbreviation",
106
+ "rule": "nbsp",
107
+ "mode": "text",
108
+ "in": "і т. д.",
109
+ "out": "і т. д.",
110
+ "note": "N4: обидва внутрішні пробіли скорочення стають U+00A0 (§ 64 п. 4). Колізії з N3 немає, бо afterShortWords для української порожній."
111
+ },
112
+ {
113
+ "id": "uk-nbsp-short-word-inert",
114
+ "rule": "nbsp",
115
+ "mode": "text",
116
+ "in": "у Києві",
117
+ "out": "у Києві",
118
+ "note": "НЕГАТИВНА ФІКСТУРА, і вона тут найважливіша. afterShortWords порожній: § 64 — закритий перелік із п'яти пунктів, і правила про однобуквені прийменники в ньому немає. Випадок стоїть на сторожі, щоб список із ru.json не перенесли сюди за аналогією спорідненої мови."
119
+ },
120
+ {
121
+ "id": "uk-nbsp-percent",
122
+ "rule": "nbsp",
123
+ "mode": "text",
124
+ "in": "зростання на 5 %",
125
+ "out": "зростання на 5 %",
126
+ "note": "«%» внесено як РІШЕННЯ, а не як цитата: його немає ні в § 62, ні в § 64, ні в скороченому викладі BIPM, але відбивка відсотка є в усіх інших локалях проєкту. Див. запис sources."
127
+ },
128
+ {
129
+ "id": "uk-spaces-brackets-and-final-stop",
130
+ "rule": "spaces",
131
+ "mode": "text",
132
+ "in": "Текст ( поточний ) не змінюється .",
133
+ "out": "Текст (поточний) не змінюється.",
134
+ "note": "spaces.md §3: правило не залежить від локалі, випадок потрібен, щоб uk мала по кейсу на кожне канонічне правило."
135
+ },
136
+ {
137
+ "id": "uk-symbols-copyright-and-dimensions",
138
+ "rule": "symbols",
139
+ "mode": "text",
140
+ "in": "Copyright (c) 2026, формат 40x60 см",
141
+ "out": "Copyright © 2026, формат 40×60 см",
142
+ "note": "symbols.md: «(c)» стає U+00A9, «x» між цифрами — U+00D7. U+00A0 перед «см» ставить nbsp N5."
143
+ },
144
+ {
145
+ "id": "uk-apostrophe-orthographic",
146
+ "rule": "apostrophe",
147
+ "mode": "text",
148
+ "in": "п'ять об'єктів",
149
+ "out": "п’ять об’єктів",
150
+ "note": "ВАЖЛИВИЙ ВИПАДОК для української, і перевірений на рушії, а не припущений: апостроф у «п'ять», «об'єкт» — частина орфографії слова, а не пунктуація, і правило apostrophe перетворює прямий U+0027 на U+2019 саме там, де він і має бути в друкованому тексті. Спершу цей випадок був написаний як no-op — рушій показав протилежне, і показав правильно."
151
+ }
152
+ ]
153
+ }