polytypo 1.2.0 → 1.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/README.md +33 -1
- data/lib/polytypo/data/VERSION +1 -1
- data/lib/polytypo/data/fixtures/cs.json +161 -0
- data/lib/polytypo/data/fixtures/de-CH.json +1 -1
- data/lib/polytypo/data/fixtures/de-DE.json +195 -6
- data/lib/polytypo/data/fixtures/el.json +1 -1
- data/lib/polytypo/data/fixtures/en-GB.json +12 -1
- data/lib/polytypo/data/fixtures/en-US.json +626 -1
- data/lib/polytypo/data/fixtures/es.json +193 -0
- data/lib/polytypo/data/fixtures/fi.json +1 -1
- data/lib/polytypo/data/fixtures/fr-CA.json +25 -1
- data/lib/polytypo/data/fixtures/fr.json +176 -1
- data/lib/polytypo/data/fixtures/it.json +161 -0
- data/lib/polytypo/data/fixtures/locale-resolution.json +76 -4
- data/lib/polytypo/data/fixtures/nl.json +121 -0
- data/lib/polytypo/data/fixtures/pl.json +137 -0
- data/lib/polytypo/data/fixtures/pt-BR.json +156 -0
- data/lib/polytypo/data/fixtures/pt-PT.json +156 -0
- data/lib/polytypo/data/fixtures/ru.json +23 -1
- data/lib/polytypo/data/fixtures/sv.json +1 -1
- data/lib/polytypo/data/fixtures/uk.json +153 -0
- data/lib/polytypo/data/locales/cs.json +90 -0
- data/lib/polytypo/data/locales/de-DE.json +7 -2
- data/lib/polytypo/data/locales/en-US.json +3 -3
- data/lib/polytypo/data/locales/es.json +111 -0
- data/lib/polytypo/data/locales/fr-CA.json +7 -1
- data/lib/polytypo/data/locales/fr.json +7 -1
- data/lib/polytypo/data/locales/it.json +95 -0
- data/lib/polytypo/data/locales/nl.json +84 -0
- data/lib/polytypo/data/locales/pl.json +96 -0
- data/lib/polytypo/data/locales/pt-BR.json +82 -0
- data/lib/polytypo/data/locales/pt-PT.json +84 -0
- data/lib/polytypo/data/locales/registry.json +23 -3
- data/lib/polytypo/data/locales/ru.json +2 -2
- data/lib/polytypo/data/locales/uk.json +130 -0
- data/lib/polytypo/data/rules/analyze.md +157 -0
- data/lib/polytypo/data/rules/apostrophe.md +432 -0
- data/lib/polytypo/data/rules/dashes.md +128 -37
- data/lib/polytypo/data/rules/ellipsis.md +271 -0
- data/lib/polytypo/data/rules/hyphen.md +353 -0
- data/lib/polytypo/data/rules/locale-resolution.md +239 -0
- data/lib/polytypo/data/rules/modes.md +1281 -0
- data/lib/polytypo/data/rules/nbsp.md +1157 -0
- data/lib/polytypo/data/rules/order.json +11 -11
- data/lib/polytypo/data/rules/pipeline-idempotency.md +605 -0
- data/lib/polytypo/data/rules/quotes.md +1324 -0
- data/lib/polytypo/data/rules/ranges.md +489 -0
- data/lib/polytypo/data/rules/spaces.md +649 -0
- data/lib/polytypo/data/rules/symbols.md +540 -0
- data/lib/polytypo/data/schema/fixtures.schema.json +18 -3
- data/lib/polytypo/engine/origin.rb +75 -0
- data/lib/polytypo/engine/pipeline.rb +72 -1
- data/lib/polytypo/engine/rules/dash_shared.rb +85 -3
- data/lib/polytypo/engine/rules/dashes.rb +4 -1
- data/lib/polytypo/engine/rules/nbsp.rb +43 -7
- data/lib/polytypo/engine/rules/ranges.rb +24 -20
- data/lib/polytypo/errors.rb +3 -0
- data/lib/polytypo/modes/runner.rb +17 -0
- data/lib/polytypo/modes/spans.rb +30 -2
- data/lib/polytypo/modes/yaml.rb +312 -0
- data/lib/polytypo/version.rb +1 -1
- data/lib/polytypo.rb +126 -15
- metadata +31 -1
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
{
|
|
2
|
-
"spec": "1.
|
|
2
|
+
"spec": "1.3.0",
|
|
3
3
|
"cases": [
|
|
4
4
|
{
|
|
5
5
|
"id": "exact-en-us",
|
|
@@ -152,10 +152,10 @@
|
|
|
152
152
|
"note": "section 4: nb must not resolve to the nearest language sv"
|
|
153
153
|
},
|
|
154
154
|
{
|
|
155
|
-
"id": "unknown-
|
|
156
|
-
"tag": "
|
|
155
|
+
"id": "unknown-ja",
|
|
156
|
+
"tag": "ja",
|
|
157
157
|
"throws": "POLYTYPO_UNKNOWN_LOCALE",
|
|
158
|
-
"note": "a language polytypo does not carry yet is an error, not a default"
|
|
158
|
+
"note": "a language polytypo does not carry yet is an error, not a default. This case named `es` until spec 1.3.0 added that locale; a fixture asserting a registry member is unknown contradicts section 3.4 step 1 and would fail every runtime the moment the registry was vendored."
|
|
159
159
|
},
|
|
160
160
|
{
|
|
161
161
|
"id": "invalid-three-letter",
|
|
@@ -204,6 +204,78 @@
|
|
|
204
204
|
"tag": "en-US-POSIX",
|
|
205
205
|
"throws": "POLYTYPO_UNKNOWN_LOCALE",
|
|
206
206
|
"note": "section 3.3: length 11. Section 7 open question 7: region is the only supported variation, so extra subtags are rejected rather than ignored."
|
|
207
|
+
},
|
|
208
|
+
{
|
|
209
|
+
"id": "exact-es",
|
|
210
|
+
"tag": "es",
|
|
211
|
+
"resolves": "es",
|
|
212
|
+
"note": "section 3.4 step 1. Added with the locale in spec 1.3.0 — the eight locales carried since 1.2.0 had no resolution case at all, so nothing proved a registry member resolved rather than threw."
|
|
213
|
+
},
|
|
214
|
+
{
|
|
215
|
+
"id": "exact-it",
|
|
216
|
+
"tag": "it",
|
|
217
|
+
"resolves": "it",
|
|
218
|
+
"note": "section 3.4 step 1."
|
|
219
|
+
},
|
|
220
|
+
{
|
|
221
|
+
"id": "exact-nl",
|
|
222
|
+
"tag": "nl",
|
|
223
|
+
"resolves": "nl",
|
|
224
|
+
"note": "section 3.4 step 1."
|
|
225
|
+
},
|
|
226
|
+
{
|
|
227
|
+
"id": "exact-pl",
|
|
228
|
+
"tag": "pl",
|
|
229
|
+
"resolves": "pl",
|
|
230
|
+
"note": "section 3.4 step 1."
|
|
231
|
+
},
|
|
232
|
+
{
|
|
233
|
+
"id": "exact-uk",
|
|
234
|
+
"tag": "uk",
|
|
235
|
+
"resolves": "uk",
|
|
236
|
+
"note": "section 3.4 step 1."
|
|
237
|
+
},
|
|
238
|
+
{
|
|
239
|
+
"id": "exact-cs",
|
|
240
|
+
"tag": "cs",
|
|
241
|
+
"resolves": "cs",
|
|
242
|
+
"note": "section 3.4 step 1."
|
|
243
|
+
},
|
|
244
|
+
{
|
|
245
|
+
"id": "exact-pt-pt",
|
|
246
|
+
"tag": "pt-PT",
|
|
247
|
+
"resolves": "pt-PT",
|
|
248
|
+
"note": "section 3.4 step 1: a language+region member is matched exactly, before any stripping."
|
|
249
|
+
},
|
|
250
|
+
{
|
|
251
|
+
"id": "exact-pt-br",
|
|
252
|
+
"tag": "pt-BR",
|
|
253
|
+
"resolves": "pt-BR",
|
|
254
|
+
"note": "section 3.4 step 1: pt-BR is its own locale, not a region strip of pt-PT."
|
|
255
|
+
},
|
|
256
|
+
{
|
|
257
|
+
"id": "alias-pt",
|
|
258
|
+
"tag": "pt",
|
|
259
|
+
"resolves": "pt-PT",
|
|
260
|
+
"note": "section 3.4 step 2: the third declared alias, alongside en and de."
|
|
261
|
+
},
|
|
262
|
+
{
|
|
263
|
+
"id": "strip-es-mx",
|
|
264
|
+
"tag": "es-MX",
|
|
265
|
+
"resolves": "es",
|
|
266
|
+
"note": "section 3.4 step 3a: an unlisted region strips to the base language once."
|
|
267
|
+
},
|
|
268
|
+
{
|
|
269
|
+
"id": "strip-nl-be",
|
|
270
|
+
"tag": "nl-BE",
|
|
271
|
+
"resolves": "nl",
|
|
272
|
+
"note": "section 3.4 step 3a."
|
|
273
|
+
},
|
|
274
|
+
{
|
|
275
|
+
"id": "strip-pt-ao",
|
|
276
|
+
"tag": "pt-AO",
|
|
277
|
+
"resolves": "pt-PT",
|
|
278
|
+
"note": "section 3.4 step 3b: the strip lands on an alias key, so the alias is followed — the one path through step 3 that steps 1 and 2 cannot reach."
|
|
207
279
|
}
|
|
208
280
|
]
|
|
209
281
|
}
|
|
@@ -0,0 +1,121 @@
|
|
|
1
|
+
{
|
|
2
|
+
"spec": "1.3.0",
|
|
3
|
+
"locale": "nl",
|
|
4
|
+
"cases": [
|
|
5
|
+
{
|
|
6
|
+
"id": "nl-quotes-primary",
|
|
7
|
+
"rule": "quotes",
|
|
8
|
+
"mode": "text",
|
|
9
|
+
"in": "Hij zei \"dag\" en vertrok.",
|
|
10
|
+
"out": "Hij zei “dag” en vertrok.",
|
|
11
|
+
"note": "quotes.primary = U+201C/U+201D. De Taalunie rangschikt de vormen NIET («Er zijn geen vaste regels»); de keuze is een beslissing van de operator, vastgelegd in de sources-ingang, en volgt de variant die de Taalunie «traditioneel» aan het letterlijke citaat verbindt."
|
|
12
|
+
},
|
|
13
|
+
{
|
|
14
|
+
"id": "nl-quotes-nested",
|
|
15
|
+
"rule": "quotes",
|
|
16
|
+
"mode": "text",
|
|
17
|
+
"in": "De koning zei: \"Ik hoorde iemand 'hoera' roepen.\"",
|
|
18
|
+
"out": "De koning zei: “Ik hoorde iemand ‘hoera’ roepen.”",
|
|
19
|
+
"note": "Voorbeeld (4a) van de Taalunie zelf: dubbele tekens op het eerste niveau, enkele op het tweede."
|
|
20
|
+
},
|
|
21
|
+
{
|
|
22
|
+
"id": "nl-quotes-apostrophe-not-a-quote",
|
|
23
|
+
"rule": "apostrophe",
|
|
24
|
+
"mode": "text",
|
|
25
|
+
"in": "Hij kocht twee auto's.",
|
|
26
|
+
"out": "Hij kocht twee auto’s.",
|
|
27
|
+
"note": "Het Nederlands staat vol apostroffen (auto's, 's-Hertogenbosch, A4'tje) en die worden U+2019 — hetzelfde codepunt als het sluitende enkele aanhalingsteken. Gecontroleerd vóór de keuze van de aanhalingstekens: het maakt voor deze zinnen geen verschil welke variant het eerste niveau krijgt."
|
|
28
|
+
},
|
|
29
|
+
{
|
|
30
|
+
"id": "nl-dashes-parenthetical",
|
|
31
|
+
"rule": "dashes",
|
|
32
|
+
"mode": "text",
|
|
33
|
+
"in": "Het nieuws - volgens de kranten - is niet bekend.",
|
|
34
|
+
"out": "Het nieuws – volgens de kranten – is niet bekend.",
|
|
35
|
+
"note": "dash.parenthetical = «en-spaced»: het halve kastlijntje U+2013 met een spatie aan weerszijden, precies het voorbeeld van de Taalunie."
|
|
36
|
+
},
|
|
37
|
+
{
|
|
38
|
+
"id": "nl-dashes-already-correct",
|
|
39
|
+
"rule": "dashes",
|
|
40
|
+
"mode": "text",
|
|
41
|
+
"in": "Het nieuws – volgens de kranten – is niet bekend.",
|
|
42
|
+
"out": "Het nieuws – volgens de kranten – is niet bekend.",
|
|
43
|
+
"note": "dashes.md §3.4 P5 laat een reeks van precies één U+2013 met rust, dus correct gezet Nederlands blijft ongemoeid."
|
|
44
|
+
},
|
|
45
|
+
{
|
|
46
|
+
"id": "nl-ranges-hyphen-kept",
|
|
47
|
+
"rule": "ranges",
|
|
48
|
+
"mode": "text",
|
|
49
|
+
"rules": {
|
|
50
|
+
"ranges": true
|
|
51
|
+
},
|
|
52
|
+
"in": "de categorie 30-45 jaar",
|
|
53
|
+
"out": "de categorie 30-45 jaar",
|
|
54
|
+
"note": "ranges.md §2: met dash.range = «none» emitteert de regel niets — ook niet de U+2060 die een omgezet bereik in andere locales meekrijgt. Het Nederlandse bereikteken is het koppelteken U+002D dat er al staat (Onze Taal: «1940-1945, de categorie 30-45 jaar»). Het geval draait met ranges expliciet aan, want de regel staat standaard uit."
|
|
55
|
+
},
|
|
56
|
+
{
|
|
57
|
+
"id": "nl-ellipsis-three-dots",
|
|
58
|
+
"rule": "ellipsis",
|
|
59
|
+
"mode": "text",
|
|
60
|
+
"in": "Hou je van jazz, blues, soul ...?",
|
|
61
|
+
"out": "Hou je van jazz, blues, soul …?",
|
|
62
|
+
"note": "Het voorbeeld van de Taalunie zelf. Let op de spatie VÓÓR het beletselteken: die hoort in het Nederlands en spaces.md §3.4 laat hem in elk locale staan — hier zijn bron en motor het eens."
|
|
63
|
+
},
|
|
64
|
+
{
|
|
65
|
+
"id": "nl-hyphen-noop",
|
|
66
|
+
"rule": "hyphen",
|
|
67
|
+
"mode": "text",
|
|
68
|
+
"in": "een auto-ongeluk en mee-eten",
|
|
69
|
+
"out": "een auto-ongeluk en mee-eten",
|
|
70
|
+
"note": "hyphen.md §2: drie lege lijsten, dus een aantoonbare no-op. Bij afbreking valt het Nederlandse koppelteken juist wég, het omgekeerde van een claim op U+2011."
|
|
71
|
+
},
|
|
72
|
+
{
|
|
73
|
+
"id": "nl-nbsp-currency",
|
|
74
|
+
"rule": "nbsp",
|
|
75
|
+
"mode": "text",
|
|
76
|
+
"in": "De vaas kostte € 179.",
|
|
77
|
+
"out": "De vaas kostte € 179.",
|
|
78
|
+
"note": "N6 afterSymbols: in het Nederlands gaat het valutateken aan het getal VOORAF, dus het hoort in afterSymbols en niet in beforeUnits zoals in fr, waar «2 €» geschreven wordt."
|
|
79
|
+
},
|
|
80
|
+
{
|
|
81
|
+
"id": "nl-nbsp-unit",
|
|
82
|
+
"rule": "nbsp",
|
|
83
|
+
"mode": "text",
|
|
84
|
+
"in": "De afstand was 15 cm.",
|
|
85
|
+
"out": "De afstand was 15 cm.",
|
|
86
|
+
"note": "N5 beforeUnits, voorbeeld (28) van de Taalunie."
|
|
87
|
+
},
|
|
88
|
+
{
|
|
89
|
+
"id": "nl-nbsp-compound-untouched",
|
|
90
|
+
"rule": "nbsp",
|
|
91
|
+
"mode": "text",
|
|
92
|
+
"in": "Ze hanteerde de 15cm-norm.",
|
|
93
|
+
"out": "Ze hanteerde de 15cm-norm.",
|
|
94
|
+
"note": "Voorbeeld (29) van dezelfde pagina: in een samenstelling staat geen spatie. N5 zet alleen een bestaande spatie om en voegt er nooit een in, dus de samenstelling is veilig."
|
|
95
|
+
},
|
|
96
|
+
{
|
|
97
|
+
"id": "nl-nbsp-abbreviation",
|
|
98
|
+
"rule": "nbsp",
|
|
99
|
+
"mode": "text",
|
|
100
|
+
"in": "prof. dr. Jansen",
|
|
101
|
+
"out": "prof. dr. Jansen",
|
|
102
|
+
"note": "N4 abbreviations op «prof. dr.». De spatie vóór de achternaam blijft gewoon: beforeWord is leeg, en initialBinding is «none»."
|
|
103
|
+
},
|
|
104
|
+
{
|
|
105
|
+
"id": "nl-spaces-brackets-and-final-stop",
|
|
106
|
+
"rule": "spaces",
|
|
107
|
+
"mode": "text",
|
|
108
|
+
"in": "De tekst ( gewoon ) verandert niet .",
|
|
109
|
+
"out": "De tekst (gewoon) verandert niet.",
|
|
110
|
+
"note": "spaces.md §3: locale-onafhankelijke regel, hier aanwezig zodat nl voor elke canonieke regel een geval heeft."
|
|
111
|
+
},
|
|
112
|
+
{
|
|
113
|
+
"id": "nl-symbols-copyright-and-dimensions",
|
|
114
|
+
"rule": "symbols",
|
|
115
|
+
"mode": "text",
|
|
116
|
+
"in": "Copyright (c) 2026, formaat 40x60 cm",
|
|
117
|
+
"out": "Copyright © 2026, formaat 40×60 cm",
|
|
118
|
+
"note": "symbols.md: «(c)» wordt U+00A9 en de «x» tussen cijfers U+00D7. De U+00A0 vóór «cm» is het werk van nbsp N5."
|
|
119
|
+
}
|
|
120
|
+
]
|
|
121
|
+
}
|
|
@@ -0,0 +1,137 @@
|
|
|
1
|
+
{
|
|
2
|
+
"spec": "1.3.0",
|
|
3
|
+
"locale": "pl",
|
|
4
|
+
"cases": [
|
|
5
|
+
{
|
|
6
|
+
"id": "pl-quotes-primary",
|
|
7
|
+
"rule": "quotes",
|
|
8
|
+
"mode": "text",
|
|
9
|
+
"in": "Powiedział \"dzień dobry\".",
|
|
10
|
+
"out": "Powiedział „dzień dobry”.",
|
|
11
|
+
"note": "quotes.primary = U+201E otwierający i U+201D zamykający (PWN, Zasady, §98). Uwaga: zamykający to U+201D, nie U+201C jak w niemieckim — różnica w składzie prawie niewidoczna, dlatego fixture."
|
|
12
|
+
},
|
|
13
|
+
{
|
|
14
|
+
"id": "pl-quotes-nested",
|
|
15
|
+
"rule": "quotes",
|
|
16
|
+
"mode": "text",
|
|
17
|
+
"in": "Powiedział \"cytat 'w cytacie' koniec\".",
|
|
18
|
+
"out": "Powiedział „cytat »w cytacie« koniec”.",
|
|
19
|
+
"note": "Drugi stopień to para o ostrzach skierowanych do środka: U+00BB otwierający, U+00AB zamykający — kierunek odwrotny niż francuski. W tym samym przypadku widać N3: jednoliterowe «w» wiąże się z następnym wyrazem przez U+00A0 (afterShortWords)."
|
|
20
|
+
},
|
|
21
|
+
{
|
|
22
|
+
"id": "pl-dashes-parenthetical",
|
|
23
|
+
"rule": "dashes",
|
|
24
|
+
"mode": "text",
|
|
25
|
+
"in": "Wtrącenie - tak jest - kończy zdanie.",
|
|
26
|
+
"out": "Wtrącenie – tak jest – kończy zdanie.",
|
|
27
|
+
"note": "dash.parenthetical = «en-spaced»: półpauza U+2013 ze spacjami po obu stronach. Wybór długości kreski jest decyzją operatora (18.09.2026) — PWN normuje odstępy, nie długość; zob. wpis sources."
|
|
28
|
+
},
|
|
29
|
+
{
|
|
30
|
+
"id": "pl-dashes-already-correct",
|
|
31
|
+
"rule": "dashes",
|
|
32
|
+
"mode": "text",
|
|
33
|
+
"in": "Wtrącenie – tak jest – kończy zdanie.",
|
|
34
|
+
"out": "Wtrącenie – tak jest – kończy zdanie.",
|
|
35
|
+
"note": "dashes.md §3.4 P5 odrzuca ciąg dokładnie jednej U+2013, więc tekst już złożony półpauzą pozostaje nietknięty. To jest ta połowa decyzji o «en-spaced», która nic nie psuje w istniejących tekstach."
|
|
36
|
+
},
|
|
37
|
+
{
|
|
38
|
+
"id": "pl-ranges-en-tight",
|
|
39
|
+
"rule": "ranges",
|
|
40
|
+
"mode": "text",
|
|
41
|
+
"rules": {
|
|
42
|
+
"ranges": true
|
|
43
|
+
},
|
|
44
|
+
"in": "W latach 1756-1763 toczyła się wojna.",
|
|
45
|
+
"out": "W latach 1756–1763 toczyła się wojna.",
|
|
46
|
+
"note": "ranges.md §3: dash.range = «en-tight» daje półpauzę bez spacji, owiniętą w U+2060 (łączniki wyrazów), zgodnie z [408] UWAGA: «1914–1918». Wielka litera «W» wiąże się przez N3, bo dopasowanie pierwszej litery jest niewrażliwe na wielkość."
|
|
47
|
+
},
|
|
48
|
+
{
|
|
49
|
+
"id": "pl-ellipsis-three-dots",
|
|
50
|
+
"rule": "ellipsis",
|
|
51
|
+
"mode": "text",
|
|
52
|
+
"in": "Czekaj... co?",
|
|
53
|
+
"out": "Czekaj… co?",
|
|
54
|
+
"note": "ellipsis.md §6: trzy U+002E stają się U+2026."
|
|
55
|
+
},
|
|
56
|
+
{
|
|
57
|
+
"id": "pl-ellipsis-before-terminal",
|
|
58
|
+
"rule": "ellipsis",
|
|
59
|
+
"mode": "text",
|
|
60
|
+
"in": "Jak strasznie gorąco…!",
|
|
61
|
+
"out": "Jak strasznie gorąco…!",
|
|
62
|
+
"note": "[396] 92.4: wykrzyknik stoi PO wielokropku i oba się zachowują. abbreviatedAfterTerminal = false, więc rosyjska forma dwukropkowa nie powstaje."
|
|
63
|
+
},
|
|
64
|
+
{
|
|
65
|
+
"id": "pl-hyphen-noop",
|
|
66
|
+
"rule": "hyphen",
|
|
67
|
+
"mode": "text",
|
|
68
|
+
"in": "biało-czerwona flaga",
|
|
69
|
+
"out": "biało-czerwona flaga",
|
|
70
|
+
"note": "hyphen.md §2: trzy puste listy czynią regułę dowodliwym no-opem. [196] każe dzielić wiersz w miejscu łącznika, czyli dokładnie odwrotnie niż U+2011."
|
|
71
|
+
},
|
|
72
|
+
{
|
|
73
|
+
"id": "pl-nbsp-short-words",
|
|
74
|
+
"rule": "nbsp",
|
|
75
|
+
"mode": "text",
|
|
76
|
+
"in": "Mieszkam w Warszawie i pracuję.",
|
|
77
|
+
"out": "Mieszkam w Warszawie i pracuję.",
|
|
78
|
+
"note": "N3 afterShortWords: jednoliterowe przyimki i spójniki (a, i, o, u, w, z) nie zostają na końcu wiersza. Podstawa doradcza (Wolański), przyjęta decyzją operatora 18.09.2026 — «Zasady» [204] są łagodniejsze i uzależniają regułę od szerokości łamu, czego silnik nie widzi."
|
|
79
|
+
},
|
|
80
|
+
{
|
|
81
|
+
"id": "pl-nbsp-paragraph-sign",
|
|
82
|
+
"rule": "nbsp",
|
|
83
|
+
"mode": "text",
|
|
84
|
+
"in": "Zgodnie z § 5 ustawy.",
|
|
85
|
+
"out": "Zgodnie z § 5 ustawy.",
|
|
86
|
+
"note": "Dwie subreguły w jednym zdaniu. N6 wiąże «§» z cyfrą (afterSymbols). N3 NIE wiąże «z», bo po nim stoi «§», które nie należy ani do ALNUM, ani do OPENISH (nbsp.md §3.5 krok 4) — przypadek istnieje właśnie po to, żeby to pokazać."
|
|
87
|
+
},
|
|
88
|
+
{
|
|
89
|
+
"id": "pl-nbsp-unit",
|
|
90
|
+
"rule": "nbsp",
|
|
91
|
+
"mode": "text",
|
|
92
|
+
"in": "Waży 5 kg.",
|
|
93
|
+
"out": "Waży 5 kg.",
|
|
94
|
+
"note": "N5 beforeUnits: istniejąca spacja staje się U+00A0."
|
|
95
|
+
},
|
|
96
|
+
{
|
|
97
|
+
"id": "pl-nbsp-unit-tight-untouched",
|
|
98
|
+
"rule": "nbsp",
|
|
99
|
+
"mode": "text",
|
|
100
|
+
"in": "Wzrost o 5%.",
|
|
101
|
+
"out": "Wzrost o 5%.",
|
|
102
|
+
"note": "N5 nigdy nie WSTAWIA spacji, więc «5%» — forma zgodna z polską tradycją ortotypograficzną — zostaje nietknięta, mimo że «%» jest na liście. Wiąże się natomiast «o» przez N3. Ten przypadek pokazuje, że jedna lista obsługuje obie poświadczone formy («5 %» i «5%»)."
|
|
103
|
+
},
|
|
104
|
+
{
|
|
105
|
+
"id": "pl-nbsp-abbreviation",
|
|
106
|
+
"rule": "nbsp",
|
|
107
|
+
"mode": "text",
|
|
108
|
+
"in": "lek. med. Kowalski",
|
|
109
|
+
"out": "lek. med. Kowalski",
|
|
110
|
+
"note": "N4 abbreviations: wewnętrzna spacja skrótu wielowyrazowego staje się U+00A0 (21.1.3). Spacja przed nazwiskiem pozostaje zwykła — beforeWord jest puste."
|
|
111
|
+
},
|
|
112
|
+
{
|
|
113
|
+
"id": "pl-spaces-brackets-and-final-stop",
|
|
114
|
+
"rule": "spaces",
|
|
115
|
+
"mode": "text",
|
|
116
|
+
"in": "Tekst ( bieżący ) nie zmienia się .",
|
|
117
|
+
"out": "Tekst (bieżący) nie zmienia się.",
|
|
118
|
+
"note": "spaces.md §3: reguła niezależna od locale, obecna po to, by pl miało przypadek dla każdej reguły kanonicznej."
|
|
119
|
+
},
|
|
120
|
+
{
|
|
121
|
+
"id": "pl-symbols-copyright-and-dimensions",
|
|
122
|
+
"rule": "symbols",
|
|
123
|
+
"mode": "text",
|
|
124
|
+
"in": "Copyright (c) 2026, format 40x60 cm",
|
|
125
|
+
"out": "Copyright © 2026, format 40×60 cm",
|
|
126
|
+
"note": "symbols.md: «(c)» staje się U+00A9, a «x» między cyframi U+00D7. U+00A0 przed «cm» to robota nbsp N5, nie symbols."
|
|
127
|
+
},
|
|
128
|
+
{
|
|
129
|
+
"id": "pl-apostrophe-foreign-name",
|
|
130
|
+
"rule": "apostrophe",
|
|
131
|
+
"mode": "text",
|
|
132
|
+
"in": "d'Artagnan",
|
|
133
|
+
"out": "d’Artagnan",
|
|
134
|
+
"note": "U+0027 staje się U+2019. Reguła nie czyta danych locale; przypadek ją dla polskiego poświadcza."
|
|
135
|
+
}
|
|
136
|
+
]
|
|
137
|
+
}
|
|
@@ -0,0 +1,156 @@
|
|
|
1
|
+
{
|
|
2
|
+
"spec": "1.3.0",
|
|
3
|
+
"locale": "pt-BR",
|
|
4
|
+
"cases": [
|
|
5
|
+
{
|
|
6
|
+
"id": "pt-br-quotes-primary",
|
|
7
|
+
"rule": "quotes",
|
|
8
|
+
"mode": "text",
|
|
9
|
+
"in": "Ele disse \"bom dia\" e saiu.",
|
|
10
|
+
"out": "Ele disse “bom dia” e saiu.",
|
|
11
|
+
"note": "quotes.primary com innerSpace \"none\". O uso brasileiro atestado pelo Manual de Redação da Presidência da República compõe as citações com aspas curvas duplas e nunca com angulares. É a diferença central em relação a pt-PT."
|
|
12
|
+
},
|
|
13
|
+
{
|
|
14
|
+
"id": "pt-br-quotes-nested",
|
|
15
|
+
"rule": "quotes",
|
|
16
|
+
"mode": "text",
|
|
17
|
+
"in": "Escreveu: \"Disse-me 'chego já' e desapareceu.\"",
|
|
18
|
+
"out": "Escreveu: “Disse-me ‘chego já’ e desapareceu.”",
|
|
19
|
+
"note": "Primeiro e segundo níveis de aspas. O terceiro nível que as fontes descrevem não é exprimível no esquema, que só tem primary e secondary."
|
|
20
|
+
},
|
|
21
|
+
{
|
|
22
|
+
"id": "pt-br-dashes-parenthetical",
|
|
23
|
+
"rule": "dashes",
|
|
24
|
+
"mode": "text",
|
|
25
|
+
"in": "As condições - ordenado e subvenções - eram boas.",
|
|
26
|
+
"out": "As condições – ordenado e subvenções – eram boas.",
|
|
27
|
+
"note": "dash.parenthetical = \"en-spaced\": o travessão substitui parênteses e leva um espaço ordinário de cada lado. A largura foi verificada extraindo a camada de texto do PDF da fonte, não julgada a olho sobre a página rasterizada."
|
|
28
|
+
},
|
|
29
|
+
{
|
|
30
|
+
"id": "pt-br-ranges-hyphen-kept",
|
|
31
|
+
"rule": "ranges",
|
|
32
|
+
"mode": "text",
|
|
33
|
+
"rules": {
|
|
34
|
+
"ranges": true
|
|
35
|
+
},
|
|
36
|
+
"in": "o programa para 1996-1997",
|
|
37
|
+
"out": "o programa para 1996-1997",
|
|
38
|
+
"note": "ranges.md §2: com dash.range = \"none\" a regra não emite nada — nem sequer os U+2060 que um intervalo convertido carrega noutras locales — e o U+002D escrito pelo autor sobrevive byte a byte."
|
|
39
|
+
},
|
|
40
|
+
{
|
|
41
|
+
"id": "pt-br-ranges-slash-untouched",
|
|
42
|
+
"rule": "ranges",
|
|
43
|
+
"mode": "text",
|
|
44
|
+
"rules": {
|
|
45
|
+
"ranges": true
|
|
46
|
+
},
|
|
47
|
+
"in": "o ano letivo de 1990/1991",
|
|
48
|
+
"out": "o ano letivo de 1990/1991",
|
|
49
|
+
"note": "ranges.md §3: a barra não é um token de intervalo para esta regra. Fica fixado que o outro ramo da convenção portuguesa — período que não abrange os dois anos completos — também não é tocado."
|
|
50
|
+
},
|
|
51
|
+
{
|
|
52
|
+
"id": "pt-br-ellipsis-three-dots",
|
|
53
|
+
"rule": "ellipsis",
|
|
54
|
+
"mode": "text",
|
|
55
|
+
"in": "Espere... o quê?",
|
|
56
|
+
"out": "Espere… o quê?",
|
|
57
|
+
"note": "ellipsis.md §6: três U+002E passam a U+2026. Nada acontece antes de «?», porque beforePunctuation e narrowBeforePunctuation estão vazios — ao contrário de fr."
|
|
58
|
+
},
|
|
59
|
+
{
|
|
60
|
+
"id": "pt-br-ellipsis-not-abbreviated-after-terminal",
|
|
61
|
+
"rule": "ellipsis",
|
|
62
|
+
"mode": "text",
|
|
63
|
+
"in": "É o dianho!..",
|
|
64
|
+
"out": "É o dianho!…",
|
|
65
|
+
"note": "abbreviatedAfterTerminal = false: a forma abreviada de dois pontos que o russo usa não existe em português, pelo que «!..» é um lapso por «!…». A fonte europeia imprime exatamente «É o dianho!…»."
|
|
66
|
+
},
|
|
67
|
+
{
|
|
68
|
+
"id": "pt-br-hyphen-noop-compounds",
|
|
69
|
+
"rule": "hyphen",
|
|
70
|
+
"mode": "text",
|
|
71
|
+
"in": "Uma reunião na segunda-feira sobre o salmão-do-atlântico.",
|
|
72
|
+
"out": "Uma reunião na segunda-feira sobre o salmão-do-atlântico.",
|
|
73
|
+
"note": "hyphen.md §2: as três listas estão vazias porque o Acordo Ortográfico (Base XX) manda QUEBRAR a linha no hífen do composto e repeti-lo na linha seguinte. Um port que ligue compostos portugueses com U+2011 falha este caso."
|
|
74
|
+
},
|
|
75
|
+
{
|
|
76
|
+
"id": "pt-br-hyphen-noop-enclisis",
|
|
77
|
+
"rule": "hyphen",
|
|
78
|
+
"mode": "text",
|
|
79
|
+
"in": "far-se-á e exigem-lhe",
|
|
80
|
+
"out": "far-se-á e exigem-lhe",
|
|
81
|
+
"note": "Ênclise e mesóclise, exemplos literais do Manual de Redação da Presidência da República, que prescreve a mesma repetição do hífen na translineação."
|
|
82
|
+
},
|
|
83
|
+
{
|
|
84
|
+
"id": "pt-br-nbsp-percent",
|
|
85
|
+
"rule": "nbsp",
|
|
86
|
+
"mode": "text",
|
|
87
|
+
"in": "7 % do volume de negócios",
|
|
88
|
+
"out": "7 % do volume de negócios",
|
|
89
|
+
"note": "N5 converte um espaço já escrito. O português escreve «7 %» com espaço, ao contrário do grego, que o escreve colado."
|
|
90
|
+
},
|
|
91
|
+
{
|
|
92
|
+
"id": "pt-br-nbsp-celsius",
|
|
93
|
+
"rule": "nbsp",
|
|
94
|
+
"mode": "text",
|
|
95
|
+
"in": "a temperatura atingiu hoje os 39 °C",
|
|
96
|
+
"out": "a temperatura atingiu hoje os 39 °C",
|
|
97
|
+
"note": "Exemplo literal do ponto 10.9.1 do Código de Redação Interinstitucional."
|
|
98
|
+
},
|
|
99
|
+
{
|
|
100
|
+
"id": "pt-br-nbsp-hour-is-not-a-unit",
|
|
101
|
+
"rule": "nbsp",
|
|
102
|
+
"mode": "text",
|
|
103
|
+
"in": "eram as 18h30",
|
|
104
|
+
"out": "eram as 18h30",
|
|
105
|
+
"note": "O reparo citado: «h» escreve-se «sempre sem ponto, sem espaços» (ponto 10.9.1), pelo que está deliberadamente ausente de beforeUnits. Este caso fixa essa ausência para que ninguém a preencha por analogia."
|
|
106
|
+
},
|
|
107
|
+
{
|
|
108
|
+
"id": "pt-br-nbsp-abbreviation",
|
|
109
|
+
"rule": "nbsp",
|
|
110
|
+
"mode": "text",
|
|
111
|
+
"in": "p. ex. isto",
|
|
112
|
+
"out": "p. ex. isto",
|
|
113
|
+
"note": "N4 liga o espaço interno da abreviatura listada «p. ex.», que o anexo A3 manda preferir a «e. g.»."
|
|
114
|
+
},
|
|
115
|
+
{
|
|
116
|
+
"id": "pt-br-nbsp-before-number",
|
|
117
|
+
"rule": "nbsp",
|
|
118
|
+
"mode": "text",
|
|
119
|
+
"in": "ver p. 24",
|
|
120
|
+
"out": "ver p. 24",
|
|
121
|
+
"note": "N9: «p. 24» é o exemplo impresso. «n.º» NÃO está na lista, e por citação e não por esquecimento — a nota (3) do ponto 6.4 prescreve um «o» sobrescrito e recusa expressamente U+00BA e U+00B0."
|
|
122
|
+
},
|
|
123
|
+
{
|
|
124
|
+
"id": "pt-br-nbsp-no-space-before-colon",
|
|
125
|
+
"rule": "nbsp",
|
|
126
|
+
"mode": "text",
|
|
127
|
+
"in": "As principais cidades de Portugal são: Lisboa, Porto e Coimbra.",
|
|
128
|
+
"out": "As principais cidades de Portugal são: Lisboa, Porto e Coimbra.",
|
|
129
|
+
"note": "beforePunctuation e narrowBeforePunctuation vazios: nada é inserido antes de «:». Frase literal do ponto 10.4.4. Um port que aplique a prática francesa a todas as locales falha aqui."
|
|
130
|
+
},
|
|
131
|
+
{
|
|
132
|
+
"id": "pt-br-spaces-brackets-and-final-stop",
|
|
133
|
+
"rule": "spaces",
|
|
134
|
+
"mode": "text",
|
|
135
|
+
"in": "O texto ( corrente ) não muda .",
|
|
136
|
+
"out": "O texto (corrente) não muda.",
|
|
137
|
+
"note": "spaces.md §3: os espaços interiores dos parênteses desaparecem e a série anterior ao ponto final reduz-se. Regra independente da locale, presente para que esta tenha um caso por cada regra canónica."
|
|
138
|
+
},
|
|
139
|
+
{
|
|
140
|
+
"id": "pt-br-symbols-copyright-and-dimensions",
|
|
141
|
+
"rule": "symbols",
|
|
142
|
+
"mode": "text",
|
|
143
|
+
"in": "Copyright (c) 2026, formato 40x60 cm",
|
|
144
|
+
"out": "Copyright © 2026, formato 40×60 cm",
|
|
145
|
+
"note": "symbols.md: «(c)» passa a U+00A9 e o «x» entre algarismos a U+00D7. O U+00A0 antes de «cm» é obra de nbsp N5, não de symbols."
|
|
146
|
+
},
|
|
147
|
+
{
|
|
148
|
+
"id": "pt-br-apostrophe-elision",
|
|
149
|
+
"rule": "apostrophe",
|
|
150
|
+
"mode": "text",
|
|
151
|
+
"in": "pinga d'água",
|
|
152
|
+
"out": "pinga d’água",
|
|
153
|
+
"note": "U+0027 passa a U+2019. A regra não lê dados de locale; o caso certifica-a para esta locale."
|
|
154
|
+
}
|
|
155
|
+
]
|
|
156
|
+
}
|