polytypo 1.1.0 → 1.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/README.md +33 -1
- data/lib/polytypo/data/VERSION +1 -1
- data/lib/polytypo/data/fixtures/cs.json +161 -0
- data/lib/polytypo/data/fixtures/de-CH.json +9 -1
- data/lib/polytypo/data/fixtures/de-DE.json +227 -6
- data/lib/polytypo/data/fixtures/el.json +9 -1
- data/lib/polytypo/data/fixtures/en-GB.json +28 -1
- data/lib/polytypo/data/fixtures/en-US.json +690 -1
- data/lib/polytypo/data/fixtures/es.json +193 -0
- data/lib/polytypo/data/fixtures/fi.json +9 -1
- data/lib/polytypo/data/fixtures/fr-CA.json +50 -1
- data/lib/polytypo/data/fixtures/fr.json +282 -1
- data/lib/polytypo/data/fixtures/it.json +161 -0
- data/lib/polytypo/data/fixtures/locale-resolution.json +76 -4
- data/lib/polytypo/data/fixtures/nl.json +121 -0
- data/lib/polytypo/data/fixtures/pl.json +137 -0
- data/lib/polytypo/data/fixtures/pt-BR.json +156 -0
- data/lib/polytypo/data/fixtures/pt-PT.json +156 -0
- data/lib/polytypo/data/fixtures/ru.json +47 -1
- data/lib/polytypo/data/fixtures/sv.json +9 -1
- data/lib/polytypo/data/fixtures/uk.json +153 -0
- data/lib/polytypo/data/locales/cs.json +90 -0
- data/lib/polytypo/data/locales/de-DE.json +7 -2
- data/lib/polytypo/data/locales/en-US.json +3 -3
- data/lib/polytypo/data/locales/es.json +111 -0
- data/lib/polytypo/data/locales/fr-CA.json +7 -1
- data/lib/polytypo/data/locales/fr.json +7 -1
- data/lib/polytypo/data/locales/it.json +95 -0
- data/lib/polytypo/data/locales/nl.json +84 -0
- data/lib/polytypo/data/locales/pl.json +96 -0
- data/lib/polytypo/data/locales/pt-BR.json +82 -0
- data/lib/polytypo/data/locales/pt-PT.json +84 -0
- data/lib/polytypo/data/locales/registry.json +23 -3
- data/lib/polytypo/data/locales/ru.json +2 -2
- data/lib/polytypo/data/locales/uk.json +130 -0
- data/lib/polytypo/data/rules/analyze.md +157 -0
- data/lib/polytypo/data/rules/apostrophe.md +432 -0
- data/lib/polytypo/data/rules/dashes.md +128 -37
- data/lib/polytypo/data/rules/ellipsis.md +271 -0
- data/lib/polytypo/data/rules/hyphen.md +353 -0
- data/lib/polytypo/data/rules/locale-resolution.md +239 -0
- data/lib/polytypo/data/rules/modes.md +1281 -0
- data/lib/polytypo/data/rules/nbsp.md +1157 -0
- data/lib/polytypo/data/rules/order.json +11 -11
- data/lib/polytypo/data/rules/pipeline-idempotency.md +605 -0
- data/lib/polytypo/data/rules/quotes.md +1324 -0
- data/lib/polytypo/data/rules/ranges.md +489 -0
- data/lib/polytypo/data/rules/spaces.md +649 -0
- data/lib/polytypo/data/rules/symbols.md +540 -0
- data/lib/polytypo/data/schema/fixtures.schema.json +18 -3
- data/lib/polytypo/engine/origin.rb +75 -0
- data/lib/polytypo/engine/pipeline.rb +72 -1
- data/lib/polytypo/engine/rules/apostrophe.rb +10 -1
- data/lib/polytypo/engine/rules/dash_shared.rb +85 -3
- data/lib/polytypo/engine/rules/dashes.rb +4 -1
- data/lib/polytypo/engine/rules/nbsp.rb +53 -20
- data/lib/polytypo/engine/rules/ranges.rb +24 -20
- data/lib/polytypo/engine/rules/spaces.rb +8 -1
- data/lib/polytypo/errors.rb +3 -0
- data/lib/polytypo/modes/runner.rb +17 -0
- data/lib/polytypo/modes/spans.rb +30 -2
- data/lib/polytypo/modes/yaml.rb +312 -0
- data/lib/polytypo/version.rb +1 -1
- data/lib/polytypo.rb +126 -15
- metadata +31 -1
|
@@ -0,0 +1,161 @@
|
|
|
1
|
+
{
|
|
2
|
+
"spec": "1.3.0",
|
|
3
|
+
"locale": "it",
|
|
4
|
+
"cases": [
|
|
5
|
+
{
|
|
6
|
+
"id": "it-quotes-primary",
|
|
7
|
+
"rule": "quotes",
|
|
8
|
+
"mode": "text",
|
|
9
|
+
"in": "Ha detto \"buongiorno\" ed è uscito.",
|
|
10
|
+
"out": "Ha detto «buongiorno» ed è uscito.",
|
|
11
|
+
"note": "quotes.primary = U+00AB/U+00BB, innerSpace \"none\" (Manuale interistituzionale 4.2.3 livello 1; Parte quarta 10.1.7 «Le virgolette non richiedono spazio né in apertura né in chiusura»). Contrasto diretto con fr, che qui inserisce U+00A0 all'interno."
|
|
12
|
+
},
|
|
13
|
+
{
|
|
14
|
+
"id": "it-quotes-nested",
|
|
15
|
+
"rule": "quotes",
|
|
16
|
+
"mode": "text",
|
|
17
|
+
"in": "Scrisse: \"Mi disse 'arrivo subito' e sparì.\"",
|
|
18
|
+
"out": "Scrisse: «Mi disse “arrivo subito” e sparì.»",
|
|
19
|
+
"note": "Livello 1 caporali, livello 2 virgolette alte. Il livello 3 del Manuale non è rappresentabile nello schema."
|
|
20
|
+
},
|
|
21
|
+
{
|
|
22
|
+
"id": "it-quotes-inner-space-deleted",
|
|
23
|
+
"rule": "quotes",
|
|
24
|
+
"mode": "text",
|
|
25
|
+
"in": "Ha detto « buongiorno » ed è uscito.",
|
|
26
|
+
"out": "Ha detto «buongiorno» ed è uscito.",
|
|
27
|
+
"note": "Con innerSpace \"none\" la regola CANCELLA lo spazio interno: un testo italiano copiato da una fonte francese si normalizza."
|
|
28
|
+
},
|
|
29
|
+
{
|
|
30
|
+
"id": "it-quotes-elision-before-quotation",
|
|
31
|
+
"rule": "quotes",
|
|
32
|
+
"mode": "text",
|
|
33
|
+
"in": "il titolo dell'\"amico\" ritrovato",
|
|
34
|
+
"out": "il titolo dell’«amico» ritrovato",
|
|
35
|
+
"note": "Il caso di maggiore impatto reale per l'italiano, che elide molto più del francese: l'apostrofo di elisione diventa U+2019 (apostrophe.md caso 3a, spec 1.2.0) e non viene rivendicato come segno di apertura, mentre le virgolette restano prive di spazio interno."
|
|
36
|
+
},
|
|
37
|
+
{
|
|
38
|
+
"id": "it-dashes-parenthetical-em-spaced",
|
|
39
|
+
"rule": "dashes",
|
|
40
|
+
"mode": "text",
|
|
41
|
+
"in": "La casa - se casa si poteva definire quel rudere - sorgeva ai piedi del colle.",
|
|
42
|
+
"out": "La casa — se casa si poteva definire quel rudere — sorgeva ai piedi del colle.",
|
|
43
|
+
"note": "dash.parenthetical = \"em-spaced\": U+2014 con U+0020 su ciascun lato. Frase dell'esempio stampato al punto 10.1.10 del Manuale."
|
|
44
|
+
},
|
|
45
|
+
{
|
|
46
|
+
"id": "it-dashes-compound-hyphen",
|
|
47
|
+
"rule": "dashes",
|
|
48
|
+
"mode": "text",
|
|
49
|
+
"in": "un trattato italo-francese",
|
|
50
|
+
"out": "un trattato italo-francese",
|
|
51
|
+
"note": "Guardia P1 di dashes.md §3.2: un trattino nudo e non spaziato è una parola composta. Esempio stampato al punto 10.1.11."
|
|
52
|
+
},
|
|
53
|
+
{
|
|
54
|
+
"id": "it-dashes-date-range-accepted-cost",
|
|
55
|
+
"rule": "dashes",
|
|
56
|
+
"mode": "text",
|
|
57
|
+
"in": "25 maggio - 10 giugno 1985",
|
|
58
|
+
"out": "25 maggio — 10 giugno 1985",
|
|
59
|
+
"note": "COSTO ACCETTATO, verificato sul motore e fissato qui perché non venga riscoperto come bug. Il punto 10.1.11 del Manuale prescrive il trattino spaziato per collegare due date di mesi diversi, ma la guardia P1 converte qualunque trattino isolato e spaziato fra parole. Non è una particolarità italiana: fr, ru, de-DE e en-GB, già pubblicate, fanno lo stesso su questa identica stringa."
|
|
60
|
+
},
|
|
61
|
+
{
|
|
62
|
+
"id": "it-ranges-none-years",
|
|
63
|
+
"rule": "ranges",
|
|
64
|
+
"mode": "text",
|
|
65
|
+
"in": "anno accademico 2010-2011",
|
|
66
|
+
"out": "anno accademico 2010-2011",
|
|
67
|
+
"note": "ranges.md §3: il token non viene convertito perché dash.range è \"none\" — il punto 10.3.1 del Manuale prescrive il trattino U+002D, già presente, e l'enum non ha un valore per esso. Il caso è generato con ranges attivato esplicitamente, dato che la regola è disattivata per impostazione predefinita.",
|
|
68
|
+
"rules": {
|
|
69
|
+
"ranges": true
|
|
70
|
+
}
|
|
71
|
+
},
|
|
72
|
+
{
|
|
73
|
+
"id": "it-ellipsis-three-dots",
|
|
74
|
+
"rule": "ellipsis",
|
|
75
|
+
"mode": "text",
|
|
76
|
+
"in": "Ti chiami Leone... ma sei un coniglio.",
|
|
77
|
+
"out": "Ti chiami Leone… ma sei un coniglio.",
|
|
78
|
+
"note": "Tre U+002E diventano U+2026. Frase dell'esempio stampato al punto 10.1.8."
|
|
79
|
+
},
|
|
80
|
+
{
|
|
81
|
+
"id": "it-ellipsis-not-abbreviated-after-terminal",
|
|
82
|
+
"rule": "ellipsis",
|
|
83
|
+
"mode": "text",
|
|
84
|
+
"in": "Davvero?..",
|
|
85
|
+
"out": "Davvero?…",
|
|
86
|
+
"note": "abbreviatedAfterTerminal = false: la forma a due punti del russo non esiste in italiano (10.1.8 «costituito da tre punti»; Crusca «sempre nel numero di tre»)."
|
|
87
|
+
},
|
|
88
|
+
{
|
|
89
|
+
"id": "it-hyphen-noop",
|
|
90
|
+
"rule": "hyphen",
|
|
91
|
+
"mode": "text",
|
|
92
|
+
"in": "un porta-finestra e un maxi-schermo",
|
|
93
|
+
"out": "un porta-finestra e un maxi-schermo",
|
|
94
|
+
"note": "Le tre liste di hyphen sono vuote, quindi la regola è un no-op totale e dimostrabile. Composti tratti da Treccani, «Trattino [prontuario]»."
|
|
95
|
+
},
|
|
96
|
+
{
|
|
97
|
+
"id": "it-nbsp-no-space-before-punctuation",
|
|
98
|
+
"rule": "nbsp",
|
|
99
|
+
"mode": "text",
|
|
100
|
+
"in": "Che cosa dici ? Nulla ! Davvero ; sì : nulla.",
|
|
101
|
+
"out": "Che cosa dici? Nulla! Davvero; sì: nulla.",
|
|
102
|
+
"note": "IL CASO CHE FISSA LA DIFFERENZA CON IL FRANCESE: spaces elimina lo spazio ordinario prima dei segni e nbsp non lo reinserisce, perché entrambe le liste sono vuote. Lo stesso input in fr produce U+202F prima di ? ! ; e U+00A0 prima di :."
|
|
103
|
+
},
|
|
104
|
+
{
|
|
105
|
+
"id": "it-nbsp-percent",
|
|
106
|
+
"rule": "nbsp",
|
|
107
|
+
"mode": "text",
|
|
108
|
+
"in": "un aumento del 45 % quest'anno",
|
|
109
|
+
"out": "un aumento del 45 % quest’anno",
|
|
110
|
+
"note": "N5 beforeUnits: lo spazio diventa U+00A0 (BIPM). L'apostrofo è opera della regola apostrophe, non di nbsp."
|
|
111
|
+
},
|
|
112
|
+
{
|
|
113
|
+
"id": "it-nbsp-celsius",
|
|
114
|
+
"rule": "nbsp",
|
|
115
|
+
"mode": "text",
|
|
116
|
+
"in": "una temperatura di 20 °C",
|
|
117
|
+
"out": "una temperatura di 20 °C",
|
|
118
|
+
"note": "Stessa subrregola su «°C», che è U+00B0 seguito da U+0043."
|
|
119
|
+
},
|
|
120
|
+
{
|
|
121
|
+
"id": "it-nbsp-before-number",
|
|
122
|
+
"rule": "nbsp",
|
|
123
|
+
"mode": "text",
|
|
124
|
+
"in": "il regolamento n. 3600 e cfr. pag. 2064",
|
|
125
|
+
"out": "il regolamento n. 3600 e cfr. pag. 2064",
|
|
126
|
+
"note": "N9 beforeNumber su «n.» e «pag.». Riga accettata come INFERENZA DICHIARATA e non come citazione piena: il riquadro «Spazi vuoti protetti» del punto 4.2.3 non è marcato come italiano, ma «pag.» è forma italiana. Si noti che «cfr.» non è in elenco e il suo spazio resta U+0020."
|
|
127
|
+
},
|
|
128
|
+
{
|
|
129
|
+
"id": "it-apostrophe-elision",
|
|
130
|
+
"rule": "apostrophe",
|
|
131
|
+
"mode": "text",
|
|
132
|
+
"in": "un'utopia, un'assurda aspirazione dell'Unione",
|
|
133
|
+
"out": "un’utopia, un’assurda aspirazione dell’Unione",
|
|
134
|
+
"note": "U+0027 diventa U+2019. Parole tratte dalla prosa del Manuale. La regola non legge dati di locale, ma l'elisione è la forma che il motore incontra più spesso in italiano."
|
|
135
|
+
},
|
|
136
|
+
{
|
|
137
|
+
"id": "it-roundtrip-noop",
|
|
138
|
+
"rule": "quotes",
|
|
139
|
+
"mode": "text",
|
|
140
|
+
"in": "Ha detto «buongiorno» ed è uscito…",
|
|
141
|
+
"out": "Ha detto «buongiorno» ed è uscito…",
|
|
142
|
+
"note": "modes.md §4: un documento già composto torna byte per byte."
|
|
143
|
+
},
|
|
144
|
+
{
|
|
145
|
+
"id": "it-spaces-brackets-and-final-stop",
|
|
146
|
+
"rule": "spaces",
|
|
147
|
+
"mode": "text",
|
|
148
|
+
"in": "Disse ( senza dubbio ) di sì .",
|
|
149
|
+
"out": "Disse (senza dubbio) di sì.",
|
|
150
|
+
"note": "spaces.md §3: gli spazi interni alle parentesi spariscono e la serie prima del punto finale si riduce. Regola indipendente dalla locale, presente qui perché it abbia un caso per ogni regola canonica."
|
|
151
|
+
},
|
|
152
|
+
{
|
|
153
|
+
"id": "it-symbols-copyright-and-dimensions",
|
|
154
|
+
"rule": "symbols",
|
|
155
|
+
"mode": "text",
|
|
156
|
+
"in": "Copyright (c) 2026, formato 40x60 cm",
|
|
157
|
+
"out": "Copyright © 2026, formato 40×60 cm",
|
|
158
|
+
"note": "symbols.md: «(c)» diventa U+00A9 e la «x» fra cifre U+00D7. L'U+00A0 davanti a «cm» è opera di nbsp N5, non di symbols: i due casi stanno insieme per mostrare che le regole si compongono."
|
|
159
|
+
}
|
|
160
|
+
]
|
|
161
|
+
}
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
{
|
|
2
|
-
"spec": "1.
|
|
2
|
+
"spec": "1.3.0",
|
|
3
3
|
"cases": [
|
|
4
4
|
{
|
|
5
5
|
"id": "exact-en-us",
|
|
@@ -152,10 +152,10 @@
|
|
|
152
152
|
"note": "section 4: nb must not resolve to the nearest language sv"
|
|
153
153
|
},
|
|
154
154
|
{
|
|
155
|
-
"id": "unknown-
|
|
156
|
-
"tag": "
|
|
155
|
+
"id": "unknown-ja",
|
|
156
|
+
"tag": "ja",
|
|
157
157
|
"throws": "POLYTYPO_UNKNOWN_LOCALE",
|
|
158
|
-
"note": "a language polytypo does not carry yet is an error, not a default"
|
|
158
|
+
"note": "a language polytypo does not carry yet is an error, not a default. This case named `es` until spec 1.3.0 added that locale; a fixture asserting a registry member is unknown contradicts section 3.4 step 1 and would fail every runtime the moment the registry was vendored."
|
|
159
159
|
},
|
|
160
160
|
{
|
|
161
161
|
"id": "invalid-three-letter",
|
|
@@ -204,6 +204,78 @@
|
|
|
204
204
|
"tag": "en-US-POSIX",
|
|
205
205
|
"throws": "POLYTYPO_UNKNOWN_LOCALE",
|
|
206
206
|
"note": "section 3.3: length 11. Section 7 open question 7: region is the only supported variation, so extra subtags are rejected rather than ignored."
|
|
207
|
+
},
|
|
208
|
+
{
|
|
209
|
+
"id": "exact-es",
|
|
210
|
+
"tag": "es",
|
|
211
|
+
"resolves": "es",
|
|
212
|
+
"note": "section 3.4 step 1. Added with the locale in spec 1.3.0 — the eight locales carried since 1.2.0 had no resolution case at all, so nothing proved a registry member resolved rather than threw."
|
|
213
|
+
},
|
|
214
|
+
{
|
|
215
|
+
"id": "exact-it",
|
|
216
|
+
"tag": "it",
|
|
217
|
+
"resolves": "it",
|
|
218
|
+
"note": "section 3.4 step 1."
|
|
219
|
+
},
|
|
220
|
+
{
|
|
221
|
+
"id": "exact-nl",
|
|
222
|
+
"tag": "nl",
|
|
223
|
+
"resolves": "nl",
|
|
224
|
+
"note": "section 3.4 step 1."
|
|
225
|
+
},
|
|
226
|
+
{
|
|
227
|
+
"id": "exact-pl",
|
|
228
|
+
"tag": "pl",
|
|
229
|
+
"resolves": "pl",
|
|
230
|
+
"note": "section 3.4 step 1."
|
|
231
|
+
},
|
|
232
|
+
{
|
|
233
|
+
"id": "exact-uk",
|
|
234
|
+
"tag": "uk",
|
|
235
|
+
"resolves": "uk",
|
|
236
|
+
"note": "section 3.4 step 1."
|
|
237
|
+
},
|
|
238
|
+
{
|
|
239
|
+
"id": "exact-cs",
|
|
240
|
+
"tag": "cs",
|
|
241
|
+
"resolves": "cs",
|
|
242
|
+
"note": "section 3.4 step 1."
|
|
243
|
+
},
|
|
244
|
+
{
|
|
245
|
+
"id": "exact-pt-pt",
|
|
246
|
+
"tag": "pt-PT",
|
|
247
|
+
"resolves": "pt-PT",
|
|
248
|
+
"note": "section 3.4 step 1: a language+region member is matched exactly, before any stripping."
|
|
249
|
+
},
|
|
250
|
+
{
|
|
251
|
+
"id": "exact-pt-br",
|
|
252
|
+
"tag": "pt-BR",
|
|
253
|
+
"resolves": "pt-BR",
|
|
254
|
+
"note": "section 3.4 step 1: pt-BR is its own locale, not a region strip of pt-PT."
|
|
255
|
+
},
|
|
256
|
+
{
|
|
257
|
+
"id": "alias-pt",
|
|
258
|
+
"tag": "pt",
|
|
259
|
+
"resolves": "pt-PT",
|
|
260
|
+
"note": "section 3.4 step 2: the third declared alias, alongside en and de."
|
|
261
|
+
},
|
|
262
|
+
{
|
|
263
|
+
"id": "strip-es-mx",
|
|
264
|
+
"tag": "es-MX",
|
|
265
|
+
"resolves": "es",
|
|
266
|
+
"note": "section 3.4 step 3a: an unlisted region strips to the base language once."
|
|
267
|
+
},
|
|
268
|
+
{
|
|
269
|
+
"id": "strip-nl-be",
|
|
270
|
+
"tag": "nl-BE",
|
|
271
|
+
"resolves": "nl",
|
|
272
|
+
"note": "section 3.4 step 3a."
|
|
273
|
+
},
|
|
274
|
+
{
|
|
275
|
+
"id": "strip-pt-ao",
|
|
276
|
+
"tag": "pt-AO",
|
|
277
|
+
"resolves": "pt-PT",
|
|
278
|
+
"note": "section 3.4 step 3b: the strip lands on an alias key, so the alias is followed — the one path through step 3 that steps 1 and 2 cannot reach."
|
|
207
279
|
}
|
|
208
280
|
]
|
|
209
281
|
}
|
|
@@ -0,0 +1,121 @@
|
|
|
1
|
+
{
|
|
2
|
+
"spec": "1.3.0",
|
|
3
|
+
"locale": "nl",
|
|
4
|
+
"cases": [
|
|
5
|
+
{
|
|
6
|
+
"id": "nl-quotes-primary",
|
|
7
|
+
"rule": "quotes",
|
|
8
|
+
"mode": "text",
|
|
9
|
+
"in": "Hij zei \"dag\" en vertrok.",
|
|
10
|
+
"out": "Hij zei “dag” en vertrok.",
|
|
11
|
+
"note": "quotes.primary = U+201C/U+201D. De Taalunie rangschikt de vormen NIET («Er zijn geen vaste regels»); de keuze is een beslissing van de operator, vastgelegd in de sources-ingang, en volgt de variant die de Taalunie «traditioneel» aan het letterlijke citaat verbindt."
|
|
12
|
+
},
|
|
13
|
+
{
|
|
14
|
+
"id": "nl-quotes-nested",
|
|
15
|
+
"rule": "quotes",
|
|
16
|
+
"mode": "text",
|
|
17
|
+
"in": "De koning zei: \"Ik hoorde iemand 'hoera' roepen.\"",
|
|
18
|
+
"out": "De koning zei: “Ik hoorde iemand ‘hoera’ roepen.”",
|
|
19
|
+
"note": "Voorbeeld (4a) van de Taalunie zelf: dubbele tekens op het eerste niveau, enkele op het tweede."
|
|
20
|
+
},
|
|
21
|
+
{
|
|
22
|
+
"id": "nl-quotes-apostrophe-not-a-quote",
|
|
23
|
+
"rule": "apostrophe",
|
|
24
|
+
"mode": "text",
|
|
25
|
+
"in": "Hij kocht twee auto's.",
|
|
26
|
+
"out": "Hij kocht twee auto’s.",
|
|
27
|
+
"note": "Het Nederlands staat vol apostroffen (auto's, 's-Hertogenbosch, A4'tje) en die worden U+2019 — hetzelfde codepunt als het sluitende enkele aanhalingsteken. Gecontroleerd vóór de keuze van de aanhalingstekens: het maakt voor deze zinnen geen verschil welke variant het eerste niveau krijgt."
|
|
28
|
+
},
|
|
29
|
+
{
|
|
30
|
+
"id": "nl-dashes-parenthetical",
|
|
31
|
+
"rule": "dashes",
|
|
32
|
+
"mode": "text",
|
|
33
|
+
"in": "Het nieuws - volgens de kranten - is niet bekend.",
|
|
34
|
+
"out": "Het nieuws – volgens de kranten – is niet bekend.",
|
|
35
|
+
"note": "dash.parenthetical = «en-spaced»: het halve kastlijntje U+2013 met een spatie aan weerszijden, precies het voorbeeld van de Taalunie."
|
|
36
|
+
},
|
|
37
|
+
{
|
|
38
|
+
"id": "nl-dashes-already-correct",
|
|
39
|
+
"rule": "dashes",
|
|
40
|
+
"mode": "text",
|
|
41
|
+
"in": "Het nieuws – volgens de kranten – is niet bekend.",
|
|
42
|
+
"out": "Het nieuws – volgens de kranten – is niet bekend.",
|
|
43
|
+
"note": "dashes.md §3.4 P5 laat een reeks van precies één U+2013 met rust, dus correct gezet Nederlands blijft ongemoeid."
|
|
44
|
+
},
|
|
45
|
+
{
|
|
46
|
+
"id": "nl-ranges-hyphen-kept",
|
|
47
|
+
"rule": "ranges",
|
|
48
|
+
"mode": "text",
|
|
49
|
+
"rules": {
|
|
50
|
+
"ranges": true
|
|
51
|
+
},
|
|
52
|
+
"in": "de categorie 30-45 jaar",
|
|
53
|
+
"out": "de categorie 30-45 jaar",
|
|
54
|
+
"note": "ranges.md §2: met dash.range = «none» emitteert de regel niets — ook niet de U+2060 die een omgezet bereik in andere locales meekrijgt. Het Nederlandse bereikteken is het koppelteken U+002D dat er al staat (Onze Taal: «1940-1945, de categorie 30-45 jaar»). Het geval draait met ranges expliciet aan, want de regel staat standaard uit."
|
|
55
|
+
},
|
|
56
|
+
{
|
|
57
|
+
"id": "nl-ellipsis-three-dots",
|
|
58
|
+
"rule": "ellipsis",
|
|
59
|
+
"mode": "text",
|
|
60
|
+
"in": "Hou je van jazz, blues, soul ...?",
|
|
61
|
+
"out": "Hou je van jazz, blues, soul …?",
|
|
62
|
+
"note": "Het voorbeeld van de Taalunie zelf. Let op de spatie VÓÓR het beletselteken: die hoort in het Nederlands en spaces.md §3.4 laat hem in elk locale staan — hier zijn bron en motor het eens."
|
|
63
|
+
},
|
|
64
|
+
{
|
|
65
|
+
"id": "nl-hyphen-noop",
|
|
66
|
+
"rule": "hyphen",
|
|
67
|
+
"mode": "text",
|
|
68
|
+
"in": "een auto-ongeluk en mee-eten",
|
|
69
|
+
"out": "een auto-ongeluk en mee-eten",
|
|
70
|
+
"note": "hyphen.md §2: drie lege lijsten, dus een aantoonbare no-op. Bij afbreking valt het Nederlandse koppelteken juist wég, het omgekeerde van een claim op U+2011."
|
|
71
|
+
},
|
|
72
|
+
{
|
|
73
|
+
"id": "nl-nbsp-currency",
|
|
74
|
+
"rule": "nbsp",
|
|
75
|
+
"mode": "text",
|
|
76
|
+
"in": "De vaas kostte € 179.",
|
|
77
|
+
"out": "De vaas kostte € 179.",
|
|
78
|
+
"note": "N6 afterSymbols: in het Nederlands gaat het valutateken aan het getal VOORAF, dus het hoort in afterSymbols en niet in beforeUnits zoals in fr, waar «2 €» geschreven wordt."
|
|
79
|
+
},
|
|
80
|
+
{
|
|
81
|
+
"id": "nl-nbsp-unit",
|
|
82
|
+
"rule": "nbsp",
|
|
83
|
+
"mode": "text",
|
|
84
|
+
"in": "De afstand was 15 cm.",
|
|
85
|
+
"out": "De afstand was 15 cm.",
|
|
86
|
+
"note": "N5 beforeUnits, voorbeeld (28) van de Taalunie."
|
|
87
|
+
},
|
|
88
|
+
{
|
|
89
|
+
"id": "nl-nbsp-compound-untouched",
|
|
90
|
+
"rule": "nbsp",
|
|
91
|
+
"mode": "text",
|
|
92
|
+
"in": "Ze hanteerde de 15cm-norm.",
|
|
93
|
+
"out": "Ze hanteerde de 15cm-norm.",
|
|
94
|
+
"note": "Voorbeeld (29) van dezelfde pagina: in een samenstelling staat geen spatie. N5 zet alleen een bestaande spatie om en voegt er nooit een in, dus de samenstelling is veilig."
|
|
95
|
+
},
|
|
96
|
+
{
|
|
97
|
+
"id": "nl-nbsp-abbreviation",
|
|
98
|
+
"rule": "nbsp",
|
|
99
|
+
"mode": "text",
|
|
100
|
+
"in": "prof. dr. Jansen",
|
|
101
|
+
"out": "prof. dr. Jansen",
|
|
102
|
+
"note": "N4 abbreviations op «prof. dr.». De spatie vóór de achternaam blijft gewoon: beforeWord is leeg, en initialBinding is «none»."
|
|
103
|
+
},
|
|
104
|
+
{
|
|
105
|
+
"id": "nl-spaces-brackets-and-final-stop",
|
|
106
|
+
"rule": "spaces",
|
|
107
|
+
"mode": "text",
|
|
108
|
+
"in": "De tekst ( gewoon ) verandert niet .",
|
|
109
|
+
"out": "De tekst (gewoon) verandert niet.",
|
|
110
|
+
"note": "spaces.md §3: locale-onafhankelijke regel, hier aanwezig zodat nl voor elke canonieke regel een geval heeft."
|
|
111
|
+
},
|
|
112
|
+
{
|
|
113
|
+
"id": "nl-symbols-copyright-and-dimensions",
|
|
114
|
+
"rule": "symbols",
|
|
115
|
+
"mode": "text",
|
|
116
|
+
"in": "Copyright (c) 2026, formaat 40x60 cm",
|
|
117
|
+
"out": "Copyright © 2026, formaat 40×60 cm",
|
|
118
|
+
"note": "symbols.md: «(c)» wordt U+00A9 en de «x» tussen cijfers U+00D7. De U+00A0 vóór «cm» is het werk van nbsp N5."
|
|
119
|
+
}
|
|
120
|
+
]
|
|
121
|
+
}
|
|
@@ -0,0 +1,137 @@
|
|
|
1
|
+
{
|
|
2
|
+
"spec": "1.3.0",
|
|
3
|
+
"locale": "pl",
|
|
4
|
+
"cases": [
|
|
5
|
+
{
|
|
6
|
+
"id": "pl-quotes-primary",
|
|
7
|
+
"rule": "quotes",
|
|
8
|
+
"mode": "text",
|
|
9
|
+
"in": "Powiedział \"dzień dobry\".",
|
|
10
|
+
"out": "Powiedział „dzień dobry”.",
|
|
11
|
+
"note": "quotes.primary = U+201E otwierający i U+201D zamykający (PWN, Zasady, §98). Uwaga: zamykający to U+201D, nie U+201C jak w niemieckim — różnica w składzie prawie niewidoczna, dlatego fixture."
|
|
12
|
+
},
|
|
13
|
+
{
|
|
14
|
+
"id": "pl-quotes-nested",
|
|
15
|
+
"rule": "quotes",
|
|
16
|
+
"mode": "text",
|
|
17
|
+
"in": "Powiedział \"cytat 'w cytacie' koniec\".",
|
|
18
|
+
"out": "Powiedział „cytat »w cytacie« koniec”.",
|
|
19
|
+
"note": "Drugi stopień to para o ostrzach skierowanych do środka: U+00BB otwierający, U+00AB zamykający — kierunek odwrotny niż francuski. W tym samym przypadku widać N3: jednoliterowe «w» wiąże się z następnym wyrazem przez U+00A0 (afterShortWords)."
|
|
20
|
+
},
|
|
21
|
+
{
|
|
22
|
+
"id": "pl-dashes-parenthetical",
|
|
23
|
+
"rule": "dashes",
|
|
24
|
+
"mode": "text",
|
|
25
|
+
"in": "Wtrącenie - tak jest - kończy zdanie.",
|
|
26
|
+
"out": "Wtrącenie – tak jest – kończy zdanie.",
|
|
27
|
+
"note": "dash.parenthetical = «en-spaced»: półpauza U+2013 ze spacjami po obu stronach. Wybór długości kreski jest decyzją operatora (18.09.2026) — PWN normuje odstępy, nie długość; zob. wpis sources."
|
|
28
|
+
},
|
|
29
|
+
{
|
|
30
|
+
"id": "pl-dashes-already-correct",
|
|
31
|
+
"rule": "dashes",
|
|
32
|
+
"mode": "text",
|
|
33
|
+
"in": "Wtrącenie – tak jest – kończy zdanie.",
|
|
34
|
+
"out": "Wtrącenie – tak jest – kończy zdanie.",
|
|
35
|
+
"note": "dashes.md §3.4 P5 odrzuca ciąg dokładnie jednej U+2013, więc tekst już złożony półpauzą pozostaje nietknięty. To jest ta połowa decyzji o «en-spaced», która nic nie psuje w istniejących tekstach."
|
|
36
|
+
},
|
|
37
|
+
{
|
|
38
|
+
"id": "pl-ranges-en-tight",
|
|
39
|
+
"rule": "ranges",
|
|
40
|
+
"mode": "text",
|
|
41
|
+
"rules": {
|
|
42
|
+
"ranges": true
|
|
43
|
+
},
|
|
44
|
+
"in": "W latach 1756-1763 toczyła się wojna.",
|
|
45
|
+
"out": "W latach 1756–1763 toczyła się wojna.",
|
|
46
|
+
"note": "ranges.md §3: dash.range = «en-tight» daje półpauzę bez spacji, owiniętą w U+2060 (łączniki wyrazów), zgodnie z [408] UWAGA: «1914–1918». Wielka litera «W» wiąże się przez N3, bo dopasowanie pierwszej litery jest niewrażliwe na wielkość."
|
|
47
|
+
},
|
|
48
|
+
{
|
|
49
|
+
"id": "pl-ellipsis-three-dots",
|
|
50
|
+
"rule": "ellipsis",
|
|
51
|
+
"mode": "text",
|
|
52
|
+
"in": "Czekaj... co?",
|
|
53
|
+
"out": "Czekaj… co?",
|
|
54
|
+
"note": "ellipsis.md §6: trzy U+002E stają się U+2026."
|
|
55
|
+
},
|
|
56
|
+
{
|
|
57
|
+
"id": "pl-ellipsis-before-terminal",
|
|
58
|
+
"rule": "ellipsis",
|
|
59
|
+
"mode": "text",
|
|
60
|
+
"in": "Jak strasznie gorąco…!",
|
|
61
|
+
"out": "Jak strasznie gorąco…!",
|
|
62
|
+
"note": "[396] 92.4: wykrzyknik stoi PO wielokropku i oba się zachowują. abbreviatedAfterTerminal = false, więc rosyjska forma dwukropkowa nie powstaje."
|
|
63
|
+
},
|
|
64
|
+
{
|
|
65
|
+
"id": "pl-hyphen-noop",
|
|
66
|
+
"rule": "hyphen",
|
|
67
|
+
"mode": "text",
|
|
68
|
+
"in": "biało-czerwona flaga",
|
|
69
|
+
"out": "biało-czerwona flaga",
|
|
70
|
+
"note": "hyphen.md §2: trzy puste listy czynią regułę dowodliwym no-opem. [196] każe dzielić wiersz w miejscu łącznika, czyli dokładnie odwrotnie niż U+2011."
|
|
71
|
+
},
|
|
72
|
+
{
|
|
73
|
+
"id": "pl-nbsp-short-words",
|
|
74
|
+
"rule": "nbsp",
|
|
75
|
+
"mode": "text",
|
|
76
|
+
"in": "Mieszkam w Warszawie i pracuję.",
|
|
77
|
+
"out": "Mieszkam w Warszawie i pracuję.",
|
|
78
|
+
"note": "N3 afterShortWords: jednoliterowe przyimki i spójniki (a, i, o, u, w, z) nie zostają na końcu wiersza. Podstawa doradcza (Wolański), przyjęta decyzją operatora 18.09.2026 — «Zasady» [204] są łagodniejsze i uzależniają regułę od szerokości łamu, czego silnik nie widzi."
|
|
79
|
+
},
|
|
80
|
+
{
|
|
81
|
+
"id": "pl-nbsp-paragraph-sign",
|
|
82
|
+
"rule": "nbsp",
|
|
83
|
+
"mode": "text",
|
|
84
|
+
"in": "Zgodnie z § 5 ustawy.",
|
|
85
|
+
"out": "Zgodnie z § 5 ustawy.",
|
|
86
|
+
"note": "Dwie subreguły w jednym zdaniu. N6 wiąże «§» z cyfrą (afterSymbols). N3 NIE wiąże «z», bo po nim stoi «§», które nie należy ani do ALNUM, ani do OPENISH (nbsp.md §3.5 krok 4) — przypadek istnieje właśnie po to, żeby to pokazać."
|
|
87
|
+
},
|
|
88
|
+
{
|
|
89
|
+
"id": "pl-nbsp-unit",
|
|
90
|
+
"rule": "nbsp",
|
|
91
|
+
"mode": "text",
|
|
92
|
+
"in": "Waży 5 kg.",
|
|
93
|
+
"out": "Waży 5 kg.",
|
|
94
|
+
"note": "N5 beforeUnits: istniejąca spacja staje się U+00A0."
|
|
95
|
+
},
|
|
96
|
+
{
|
|
97
|
+
"id": "pl-nbsp-unit-tight-untouched",
|
|
98
|
+
"rule": "nbsp",
|
|
99
|
+
"mode": "text",
|
|
100
|
+
"in": "Wzrost o 5%.",
|
|
101
|
+
"out": "Wzrost o 5%.",
|
|
102
|
+
"note": "N5 nigdy nie WSTAWIA spacji, więc «5%» — forma zgodna z polską tradycją ortotypograficzną — zostaje nietknięta, mimo że «%» jest na liście. Wiąże się natomiast «o» przez N3. Ten przypadek pokazuje, że jedna lista obsługuje obie poświadczone formy («5 %» i «5%»)."
|
|
103
|
+
},
|
|
104
|
+
{
|
|
105
|
+
"id": "pl-nbsp-abbreviation",
|
|
106
|
+
"rule": "nbsp",
|
|
107
|
+
"mode": "text",
|
|
108
|
+
"in": "lek. med. Kowalski",
|
|
109
|
+
"out": "lek. med. Kowalski",
|
|
110
|
+
"note": "N4 abbreviations: wewnętrzna spacja skrótu wielowyrazowego staje się U+00A0 (21.1.3). Spacja przed nazwiskiem pozostaje zwykła — beforeWord jest puste."
|
|
111
|
+
},
|
|
112
|
+
{
|
|
113
|
+
"id": "pl-spaces-brackets-and-final-stop",
|
|
114
|
+
"rule": "spaces",
|
|
115
|
+
"mode": "text",
|
|
116
|
+
"in": "Tekst ( bieżący ) nie zmienia się .",
|
|
117
|
+
"out": "Tekst (bieżący) nie zmienia się.",
|
|
118
|
+
"note": "spaces.md §3: reguła niezależna od locale, obecna po to, by pl miało przypadek dla każdej reguły kanonicznej."
|
|
119
|
+
},
|
|
120
|
+
{
|
|
121
|
+
"id": "pl-symbols-copyright-and-dimensions",
|
|
122
|
+
"rule": "symbols",
|
|
123
|
+
"mode": "text",
|
|
124
|
+
"in": "Copyright (c) 2026, format 40x60 cm",
|
|
125
|
+
"out": "Copyright © 2026, format 40×60 cm",
|
|
126
|
+
"note": "symbols.md: «(c)» staje się U+00A9, a «x» między cyframi U+00D7. U+00A0 przed «cm» to robota nbsp N5, nie symbols."
|
|
127
|
+
},
|
|
128
|
+
{
|
|
129
|
+
"id": "pl-apostrophe-foreign-name",
|
|
130
|
+
"rule": "apostrophe",
|
|
131
|
+
"mode": "text",
|
|
132
|
+
"in": "d'Artagnan",
|
|
133
|
+
"out": "d’Artagnan",
|
|
134
|
+
"note": "U+0027 staje się U+2019. Reguła nie czyta danych locale; przypadek ją dla polskiego poświadcza."
|
|
135
|
+
}
|
|
136
|
+
]
|
|
137
|
+
}
|