polytypo 1.2.0 → 1.3.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/README.md +33 -1
- data/lib/polytypo/data/VERSION +1 -1
- data/lib/polytypo/data/fixtures/cs.json +161 -0
- data/lib/polytypo/data/fixtures/de-CH.json +1 -1
- data/lib/polytypo/data/fixtures/de-DE.json +195 -6
- data/lib/polytypo/data/fixtures/el.json +1 -1
- data/lib/polytypo/data/fixtures/en-GB.json +12 -1
- data/lib/polytypo/data/fixtures/en-US.json +648 -1
- data/lib/polytypo/data/fixtures/es.json +193 -0
- data/lib/polytypo/data/fixtures/fi.json +1 -1
- data/lib/polytypo/data/fixtures/fr-CA.json +25 -1
- data/lib/polytypo/data/fixtures/fr.json +176 -1
- data/lib/polytypo/data/fixtures/it.json +161 -0
- data/lib/polytypo/data/fixtures/locale-resolution.json +76 -4
- data/lib/polytypo/data/fixtures/nl.json +121 -0
- data/lib/polytypo/data/fixtures/pl.json +137 -0
- data/lib/polytypo/data/fixtures/pt-BR.json +156 -0
- data/lib/polytypo/data/fixtures/pt-PT.json +156 -0
- data/lib/polytypo/data/fixtures/ru.json +23 -1
- data/lib/polytypo/data/fixtures/sv.json +1 -1
- data/lib/polytypo/data/fixtures/uk.json +153 -0
- data/lib/polytypo/data/locales/cs.json +90 -0
- data/lib/polytypo/data/locales/de-DE.json +7 -2
- data/lib/polytypo/data/locales/en-US.json +3 -3
- data/lib/polytypo/data/locales/es.json +111 -0
- data/lib/polytypo/data/locales/fr-CA.json +7 -1
- data/lib/polytypo/data/locales/fr.json +7 -1
- data/lib/polytypo/data/locales/it.json +95 -0
- data/lib/polytypo/data/locales/nl.json +84 -0
- data/lib/polytypo/data/locales/pl.json +96 -0
- data/lib/polytypo/data/locales/pt-BR.json +82 -0
- data/lib/polytypo/data/locales/pt-PT.json +84 -0
- data/lib/polytypo/data/locales/registry.json +23 -3
- data/lib/polytypo/data/locales/ru.json +2 -2
- data/lib/polytypo/data/locales/uk.json +130 -0
- data/lib/polytypo/data/rules/analyze.md +157 -0
- data/lib/polytypo/data/rules/apostrophe.md +432 -0
- data/lib/polytypo/data/rules/dashes.md +128 -37
- data/lib/polytypo/data/rules/ellipsis.md +271 -0
- data/lib/polytypo/data/rules/hyphen.md +353 -0
- data/lib/polytypo/data/rules/locale-resolution.md +239 -0
- data/lib/polytypo/data/rules/modes.md +1281 -0
- data/lib/polytypo/data/rules/nbsp.md +1157 -0
- data/lib/polytypo/data/rules/order.json +11 -11
- data/lib/polytypo/data/rules/pipeline-idempotency.md +605 -0
- data/lib/polytypo/data/rules/quotes.md +1324 -0
- data/lib/polytypo/data/rules/ranges.md +489 -0
- data/lib/polytypo/data/rules/spaces.md +649 -0
- data/lib/polytypo/data/rules/symbols.md +540 -0
- data/lib/polytypo/data/schema/fixtures.schema.json +18 -3
- data/lib/polytypo/engine/origin.rb +75 -0
- data/lib/polytypo/engine/pipeline.rb +72 -1
- data/lib/polytypo/engine/rules/dash_shared.rb +85 -3
- data/lib/polytypo/engine/rules/dashes.rb +4 -1
- data/lib/polytypo/engine/rules/nbsp.rb +43 -7
- data/lib/polytypo/engine/rules/ranges.rb +24 -20
- data/lib/polytypo/errors.rb +3 -0
- data/lib/polytypo/modes/runner.rb +17 -0
- data/lib/polytypo/modes/spans.rb +30 -2
- data/lib/polytypo/modes/yaml.rb +312 -0
- data/lib/polytypo/version.rb +1 -1
- data/lib/polytypo.rb +126 -15
- metadata +31 -1
|
@@ -0,0 +1,193 @@
|
|
|
1
|
+
{
|
|
2
|
+
"spec": "1.3.1",
|
|
3
|
+
"locale": "es",
|
|
4
|
+
"cases": [
|
|
5
|
+
{
|
|
6
|
+
"id": "es-quotes-primary",
|
|
7
|
+
"rule": "quotes",
|
|
8
|
+
"mode": "text",
|
|
9
|
+
"in": "Ha dicho \"buenos días\" y salió.",
|
|
10
|
+
"out": "Ha dicho «buenos días» y salió.",
|
|
11
|
+
"note": "quotes.primary = U+00AB/U+00BB con innerSpace \"none\" (DPD «comillas» 1: «se recomienda utilizar en primera instancia las comillas angulares»; «pegadas a la primera y la última palabra»)."
|
|
12
|
+
},
|
|
13
|
+
{
|
|
14
|
+
"id": "es-quotes-nested",
|
|
15
|
+
"rule": "quotes",
|
|
16
|
+
"mode": "text",
|
|
17
|
+
"in": "Escribió: \"Me dijo 'llego enseguida' y desapareció.\"",
|
|
18
|
+
"out": "Escribió: «Me dijo “llego enseguida” y desapareció.»",
|
|
19
|
+
"note": "Nivel 1 angulares, nivel 2 inglesas, según el orden que fija el DPD. El tercer nivel del DPD (simples) no es expresable en el esquema."
|
|
20
|
+
},
|
|
21
|
+
{
|
|
22
|
+
"id": "es-quotes-inner-space-deleted",
|
|
23
|
+
"rule": "quotes",
|
|
24
|
+
"mode": "text",
|
|
25
|
+
"in": "Ha dicho « buenos días » y salió.",
|
|
26
|
+
"out": "Ha dicho «buenos días» y salió.",
|
|
27
|
+
"note": "innerSpace \"none\": N8 elimina el espacio interior que un texto copiado de una fuente francesa trae consigo."
|
|
28
|
+
},
|
|
29
|
+
{
|
|
30
|
+
"id": "es-ellipsis-three-dots",
|
|
31
|
+
"rule": "ellipsis",
|
|
32
|
+
"mode": "text",
|
|
33
|
+
"in": "Espera... ¿qué?",
|
|
34
|
+
"out": "Espera… ¿qué?",
|
|
35
|
+
"note": "Tres U+002E pasan a U+2026 (DPD «puntos suspensivos» 1: «tres puntos consecutivos ―y solo tres―»)."
|
|
36
|
+
},
|
|
37
|
+
{
|
|
38
|
+
"id": "es-ellipsis-not-abbreviated-after-terminal",
|
|
39
|
+
"rule": "ellipsis",
|
|
40
|
+
"mode": "text",
|
|
41
|
+
"in": "¿De verdad?..",
|
|
42
|
+
"out": "¿De verdad?…",
|
|
43
|
+
"note": "abbreviatedAfterTerminal = false: la forma de dos puntos tras «?» que usa el ruso no existe en español, así que los dos puntos se completan a U+2026 en vez de conservarse."
|
|
44
|
+
},
|
|
45
|
+
{
|
|
46
|
+
"id": "es-dashes-parenthetical-none",
|
|
47
|
+
"rule": "dashes",
|
|
48
|
+
"mode": "text",
|
|
49
|
+
"in": "El plan - si existe - fracasa.",
|
|
50
|
+
"out": "El plan - si existe - fracasa.",
|
|
51
|
+
"note": "dash.parenthetical = \"none\": el DPD prescribe raya con espacios FUERA del par y ninguno dentro, asimetría que el enum no expresa; mientras tanto la regla no toca nada. Véase la voz dashes de es.json."
|
|
52
|
+
},
|
|
53
|
+
{
|
|
54
|
+
"id": "es-ranges-guion-none",
|
|
55
|
+
"rule": "ranges",
|
|
56
|
+
"mode": "text",
|
|
57
|
+
"in": "las páginas 23-45",
|
|
58
|
+
"out": "las páginas 23-45",
|
|
59
|
+
"note": "ranges.md §3: el token no se convierte porque dash.range es \"none\" — el signo español del intervalo es el guion U+002D, que ya está escrito (DPD «guion» 3 b), y el enum no tiene un valor para él. El caso se genera con ranges activado explícitamente, ya que la regla está desactivada por defecto.",
|
|
60
|
+
"rules": {
|
|
61
|
+
"ranges": true
|
|
62
|
+
}
|
|
63
|
+
},
|
|
64
|
+
{
|
|
65
|
+
"id": "es-hyphen-noop",
|
|
66
|
+
"rule": "hyphen",
|
|
67
|
+
"mode": "text",
|
|
68
|
+
"in": "un estudio teórico-práctico",
|
|
69
|
+
"out": "un estudio teórico-práctico",
|
|
70
|
+
"note": "Las tres listas de hyphen están vacías, así que la regla es un no-op total demostrable para el español (hyphen.md §2)."
|
|
71
|
+
},
|
|
72
|
+
{
|
|
73
|
+
"id": "es-nbsp-inverted-marks-untouched",
|
|
74
|
+
"rule": "nbsp",
|
|
75
|
+
"mode": "text",
|
|
76
|
+
"in": "¿Qué hora es? ¡Vaya!",
|
|
77
|
+
"out": "¿Qué hora es? ¡Vaya!",
|
|
78
|
+
"note": "beforePunctuation y narrowBeforePunctuation vacías: el español no pone espacio antes de «?» ni de «!», y U+00BF/U+00A1 no son miembros de CLOSEISH, así que la restricción Q-P impediría expresarlo aunque la fuente lo pidiera."
|
|
79
|
+
},
|
|
80
|
+
{
|
|
81
|
+
"id": "es-nbsp-degree-tight",
|
|
82
|
+
"rule": "nbsp",
|
|
83
|
+
"mode": "text",
|
|
84
|
+
"in": "Hacía 27° a la sombra",
|
|
85
|
+
"out": "Hacía 27° a la sombra",
|
|
86
|
+
"note": "El DPD «símbolo» 5.4 pega «°» a la cifra cuando no se especifica la escala, por eso «°» a secas NO está en beforeUnits y aquí no se inserta nada."
|
|
87
|
+
},
|
|
88
|
+
{
|
|
89
|
+
"id": "es-nbsp-percent",
|
|
90
|
+
"rule": "nbsp",
|
|
91
|
+
"mode": "text",
|
|
92
|
+
"in": "Aprobó un 8 % de los alumnos",
|
|
93
|
+
"out": "Aprobó un 8 % de los alumnos",
|
|
94
|
+
"note": "N5 beforeUnits: el espacio pasa a U+00A0. RAE, Duda lingüística: «El símbolo del porcentaje se separa con espacio de la cifra que lo precede: 50 %»; la indivisibilidad la impone el DPD «símbolo» 5.6."
|
|
95
|
+
},
|
|
96
|
+
{
|
|
97
|
+
"id": "es-nbsp-celsius",
|
|
98
|
+
"rule": "nbsp",
|
|
99
|
+
"mode": "text",
|
|
100
|
+
"in": "La temperatura es de 27 °C",
|
|
101
|
+
"out": "La temperatura es de 27 °C",
|
|
102
|
+
"note": "«°C» sí está en beforeUnits, citado literalmente por el DPD «símbolo» 5.4 («27° […] pero 27 °C»). Contrasta con es-nbsp-degree-tight."
|
|
103
|
+
},
|
|
104
|
+
{
|
|
105
|
+
"id": "es-nbsp-abbreviation-internal-space",
|
|
106
|
+
"rule": "nbsp",
|
|
107
|
+
"mode": "text",
|
|
108
|
+
"in": "Viajó a los EE. UU. el lunes",
|
|
109
|
+
"out": "Viajó a los EE. UU. el lunes",
|
|
110
|
+
"note": "N4 abbreviations: el espacio interior de una abreviatura de varios elementos pasa a U+00A0 (DPD «abreviatura»: «Cuando la abreviatura se compone de varios elementos, estos no deben separarse en líneas diferentes»)."
|
|
111
|
+
},
|
|
112
|
+
{
|
|
113
|
+
"id": "es-nbsp-abbreviation-p-ej",
|
|
114
|
+
"rule": "nbsp",
|
|
115
|
+
"mode": "text",
|
|
116
|
+
"in": "algunos casos, p. ej. este",
|
|
117
|
+
"out": "algunos casos, p. ej. este",
|
|
118
|
+
"note": "La misma subrregla sobre «p. ej.», forma atestiguada en la Lista de abreviaturas de El buen uso del español."
|
|
119
|
+
},
|
|
120
|
+
{
|
|
121
|
+
"id": "es-nbsp-before-number",
|
|
122
|
+
"rule": "nbsp",
|
|
123
|
+
"mode": "text",
|
|
124
|
+
"in": "véase la pág. 23",
|
|
125
|
+
"out": "véase la pág. 23",
|
|
126
|
+
"note": "N9 beforeNumber: «⊗15 / págs.» es el ejemplo del propio DPD «abreviatura» para este caso."
|
|
127
|
+
},
|
|
128
|
+
{
|
|
129
|
+
"id": "es-nbsp-before-word",
|
|
130
|
+
"rule": "nbsp",
|
|
131
|
+
"mode": "text",
|
|
132
|
+
"in": "El Sr. Pérez llegó tarde",
|
|
133
|
+
"out": "El Sr. Pérez llegó tarde",
|
|
134
|
+
"note": "N10 beforeWord: «⊗Sr. / Pérez» es el segundo ejemplo de la misma entrada. La lista se limita a los tratamientos de cortesía, por el riesgo de falsos positivos de esta subrregla."
|
|
135
|
+
},
|
|
136
|
+
{
|
|
137
|
+
"id": "es-nbsp-initials-chain",
|
|
138
|
+
"rule": "nbsp",
|
|
139
|
+
"mode": "text",
|
|
140
|
+
"in": "Lo escribió J. A. Pérez ayer",
|
|
141
|
+
"out": "Lo escribió J. A. Pérez ayer",
|
|
142
|
+
"note": "initialBinding = \"chain\": la cadena completa se liga, incluido el espacio que precede a la primera inicial, igual que en en-US y ru. Decisión del operador (18.09.2026) ante una fuente que no distingue entre una inicial y dos; véase la voz nbsp correspondiente."
|
|
143
|
+
},
|
|
144
|
+
{
|
|
145
|
+
"id": "es-nbsp-lone-initial-not-bound",
|
|
146
|
+
"rule": "nbsp",
|
|
147
|
+
"mode": "text",
|
|
148
|
+
"in": "La ley N. Roma decidió",
|
|
149
|
+
"out": "La ley N. Roma decidió",
|
|
150
|
+
"note": "El reverso del caso anterior y la razón de elegir \"chain\": una sola mayúscula con punto seguida de un nombre propio NO se liga, de modo que un final de frase no se confunde con una inicial."
|
|
151
|
+
},
|
|
152
|
+
{
|
|
153
|
+
"id": "es-nbsp-ac-guard",
|
|
154
|
+
"rule": "nbsp",
|
|
155
|
+
"mode": "text",
|
|
156
|
+
"in": "en el año 33 a. C. Roma era una república",
|
|
157
|
+
"out": "en el año 33 a. C. Roma era una república",
|
|
158
|
+
"note": "N4 liga el espacio interior de «a. C.» y la guarda C1-a (nbsp.md §3.9) impide que C1 lea «C.» como inicial y «Roma» como apellido. Equivalente español del alemán «z. B. Berlin»."
|
|
159
|
+
},
|
|
160
|
+
{
|
|
161
|
+
"id": "es-apostrophe-foreign-name",
|
|
162
|
+
"rule": "apostrophe",
|
|
163
|
+
"mode": "text",
|
|
164
|
+
"in": "L'Hospitalet de Llobregat",
|
|
165
|
+
"out": "L’Hospitalet de Llobregat",
|
|
166
|
+
"note": "U+0027 pasa a U+2019. El DPD «apóstrofo» limita el signo en español a textos antiguos, habla reproducida y nombres de otras lenguas: este es el tercer caso, y es el que un texto español encuentra de verdad."
|
|
167
|
+
},
|
|
168
|
+
{
|
|
169
|
+
"id": "es-roundtrip-noop",
|
|
170
|
+
"rule": "quotes",
|
|
171
|
+
"mode": "text",
|
|
172
|
+
"in": "Ha dicho «buenos días» y salió…",
|
|
173
|
+
"out": "Ha dicho «buenos días» y salió…",
|
|
174
|
+
"note": "modes.md §4: un texto ya compuesto vuelve byte a byte."
|
|
175
|
+
},
|
|
176
|
+
{
|
|
177
|
+
"id": "es-spaces-brackets-and-final-stop",
|
|
178
|
+
"rule": "spaces",
|
|
179
|
+
"mode": "text",
|
|
180
|
+
"in": "Dijo ( sin dudar ) que sí .",
|
|
181
|
+
"out": "Dijo (sin dudar) que sí.",
|
|
182
|
+
"note": "spaces.md §3: los espacios interiores de los paréntesis se eliminan y la serie anterior al punto final se colapsa. Regla independiente de la locale, presente aquí para que es tenga un caso por cada regla canónica."
|
|
183
|
+
},
|
|
184
|
+
{
|
|
185
|
+
"id": "es-symbols-copyright-and-dimensions",
|
|
186
|
+
"rule": "symbols",
|
|
187
|
+
"mode": "text",
|
|
188
|
+
"in": "Copyright (c) 2026, tamaño 40x60 cm",
|
|
189
|
+
"out": "Copyright © 2026, tamaño 40×60 cm",
|
|
190
|
+
"note": "symbols.md: «(c)» pasa a U+00A9 y la «x» entre cifras a U+00D7. El U+00A0 ante «cm» lo pone nbsp N5, no symbols; se deja en el mismo caso porque así se ve que las dos reglas componen."
|
|
191
|
+
}
|
|
192
|
+
]
|
|
193
|
+
}
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
{
|
|
2
|
-
"spec": "1.
|
|
2
|
+
"spec": "1.3.1",
|
|
3
3
|
"locale": "fr-CA",
|
|
4
4
|
"cases": [
|
|
5
5
|
{
|
|
@@ -288,6 +288,30 @@
|
|
|
288
288
|
"in": "l'\"idée\" est simple",
|
|
289
289
|
"out": "l’« idée » est simple",
|
|
290
290
|
"note": "apostrophe.md §3.3 case 3a (spec 1.2.0): same as `fr`."
|
|
291
|
+
},
|
|
292
|
+
{
|
|
293
|
+
"id": "fr-ca-nbsp-numero-binds-number",
|
|
294
|
+
"rule": "nbsp",
|
|
295
|
+
"mode": "text",
|
|
296
|
+
"in": "voir n° 5",
|
|
297
|
+
"out": "voir n° 5",
|
|
298
|
+
"note": "nbsp.md §3.11 N9 : « n° » figure dans nbsp.beforeNumber depuis la spec 1.3.0 et se lie au nombre qui suit. La forme retenue est « n » + U+00B0, l'usage le plus répandu ; voir l'entrée sources de la locale, la source (OQLF) prescrivant un tracé et non un code point."
|
|
299
|
+
},
|
|
300
|
+
{
|
|
301
|
+
"id": "fr-ca-nbsp-numero-capital-is-its-own-entry",
|
|
302
|
+
"rule": "nbsp",
|
|
303
|
+
"mode": "text",
|
|
304
|
+
"in": "N° 5 de la revue",
|
|
305
|
+
"out": "N° 5 de la revue",
|
|
306
|
+
"note": "nbsp.md §3.11 N9 étape 1 : aucune tolérance de casse, « N° » est donc une entrée distincte de « n° » dans nbsp.beforeNumber."
|
|
307
|
+
},
|
|
308
|
+
{
|
|
309
|
+
"id": "fr-ca-nbsp-numero-needs-a-digit",
|
|
310
|
+
"rule": "nbsp",
|
|
311
|
+
"mode": "text",
|
|
312
|
+
"in": "le n° cinq",
|
|
313
|
+
"out": "le n° cinq",
|
|
314
|
+
"note": "nbsp.md §3.11 N9 étape 4 : il faut un DIGIT après le séparateur. Un nombre écrit en toutes lettres n'en est pas un, l'espace reste sécable."
|
|
291
315
|
}
|
|
292
316
|
]
|
|
293
317
|
}
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
{
|
|
2
|
-
"spec": "1.
|
|
2
|
+
"spec": "1.3.1",
|
|
3
3
|
"locale": "fr",
|
|
4
4
|
"cases": [
|
|
5
5
|
{
|
|
@@ -704,6 +704,181 @@
|
|
|
704
704
|
"in": "Voir http:<a href=\"//x.org\">//x.org</a>",
|
|
705
705
|
"out": "Voir http :<a href=\"//x.org\">//x.org</a>",
|
|
706
706
|
"note": "nbsp.md §3.3 step 2 (spec 1.2.0): the second stated cost of accepting the marker. Step 2 cannot see the `/` in the next span, so the colon is treated as sentence punctuation and N1 inserts U+00A0."
|
|
707
|
+
},
|
|
708
|
+
{
|
|
709
|
+
"id": "fr-nbsp-numero-binds-number",
|
|
710
|
+
"rule": "nbsp",
|
|
711
|
+
"mode": "text",
|
|
712
|
+
"in": "voir n° 5",
|
|
713
|
+
"out": "voir n° 5",
|
|
714
|
+
"note": "nbsp.md §3.11 N9 : « n° » figure dans nbsp.beforeNumber depuis la spec 1.3.0 et se lie au nombre qui suit. La forme retenue est « n » + U+00B0, l'usage le plus répandu ; voir l'entrée sources de la locale, la source (OQLF) prescrivant un tracé et non un code point."
|
|
715
|
+
},
|
|
716
|
+
{
|
|
717
|
+
"id": "fr-nbsp-numero-capital-is-its-own-entry",
|
|
718
|
+
"rule": "nbsp",
|
|
719
|
+
"mode": "text",
|
|
720
|
+
"in": "N° 5 de la revue",
|
|
721
|
+
"out": "N° 5 de la revue",
|
|
722
|
+
"note": "nbsp.md §3.11 N9 étape 1 : aucune tolérance de casse, « N° » est donc une entrée distincte de « n° » dans nbsp.beforeNumber."
|
|
723
|
+
},
|
|
724
|
+
{
|
|
725
|
+
"id": "fr-nbsp-numero-needs-a-digit",
|
|
726
|
+
"rule": "nbsp",
|
|
727
|
+
"mode": "text",
|
|
728
|
+
"in": "le n° cinq",
|
|
729
|
+
"out": "le n° cinq",
|
|
730
|
+
"note": "nbsp.md §3.11 N9 étape 4 : il faut un DIGIT après le séparateur. Un nombre écrit en toutes lettres n'en est pas un, l'espace reste sécable."
|
|
731
|
+
},
|
|
732
|
+
{
|
|
733
|
+
"id": "fr-markdown-commonmark-frontmatter-nbsp",
|
|
734
|
+
"rule": "nbsp",
|
|
735
|
+
"mode": "markdown",
|
|
736
|
+
"dialect": "commonmark",
|
|
737
|
+
"in": "---\ntitle: Une note ; suite\n---\n\nEst-ce vrai ?\n",
|
|
738
|
+
"out": "---\ntitle: Une note ; suite\n---\n\nEst-ce vrai ?\n",
|
|
739
|
+
"note": "modes.md §3.7.3 gives this exact reason for skipping frontmatter: without it `fr` would put U+00A0 before the colon of a machine-read field and U+202F before its semicolon. The body sentence shows the rule is otherwise active in the same document, so the untouched field proves the skip rather than an inactive rule."
|
|
740
|
+
},
|
|
741
|
+
{
|
|
742
|
+
"id": "fr-markdown-mdx-frontmatter-nbsp",
|
|
743
|
+
"rule": "nbsp",
|
|
744
|
+
"mode": "markdown",
|
|
745
|
+
"dialect": "mdx",
|
|
746
|
+
"in": "---\ntitle: Une note ; suite\n---\n\n<Callout>Est-ce vrai ?</Callout>\n",
|
|
747
|
+
"out": "---\ntitle: Une note ; suite\n---\n\n<Callout>Est-ce vrai ?</Callout>\n",
|
|
748
|
+
"note": "modes.md §3.7.3: the same guarantee under \"mdx\", where frontmatter plus JSX is the ordinary document shape. This is the case the polytypo-js source comment above its frontmatter skip entry described as a spec gap."
|
|
749
|
+
},
|
|
750
|
+
{
|
|
751
|
+
"id": "fr-nbsp-character-reference-numeric",
|
|
752
|
+
"rule": "nbsp",
|
|
753
|
+
"mode": "text",
|
|
754
|
+
"in": "Bonjour : oui",
|
|
755
|
+
"out": "Bonjour : oui",
|
|
756
|
+
"note": "nbsp.md §3.3 step 4 (spec 1.3.0): the `;` closes ` `, so N2 emits nothing and the reference survives. The `:` takes nothing either — step 1's run guard sees a mark to its left. Before 1.3.0 this produced `Bonjour ` + U+202F + `: oui`, which is no longer a character reference."
|
|
757
|
+
},
|
|
758
|
+
{
|
|
759
|
+
"id": "fr-nbsp-character-reference-named",
|
|
760
|
+
"rule": "nbsp",
|
|
761
|
+
"mode": "text",
|
|
762
|
+
"in": "Tom & Jerry",
|
|
763
|
+
"out": "Tom & Jerry",
|
|
764
|
+
"note": "nbsp.md §3.3 step 4 (spec 1.3.0): the same guard on a named reference. Before 1.3.0 `text` mode destroyed every character reference in French input, and only in locales whose nbsp data puts a space before `;`."
|
|
765
|
+
},
|
|
766
|
+
{
|
|
767
|
+
"id": "fr-nbsp-character-reference-shape-not-table",
|
|
768
|
+
"rule": "nbsp",
|
|
769
|
+
"mode": "text",
|
|
770
|
+
"in": "a ¬aname; b",
|
|
771
|
+
"out": "a ¬aname; b",
|
|
772
|
+
"note": "nbsp.md §3.3 step 4 (spec 1.3.0): the guard tests the shape of a reference, not membership of the HTML named-reference table, which five runtimes would otherwise have to carry identically. Declining here costs nothing."
|
|
773
|
+
},
|
|
774
|
+
{
|
|
775
|
+
"id": "fr-nbsp-ordinary-semicolon-still-binds",
|
|
776
|
+
"rule": "nbsp",
|
|
777
|
+
"mode": "text",
|
|
778
|
+
"in": "Oui ; non",
|
|
779
|
+
"out": "Oui ; non",
|
|
780
|
+
"note": "nbsp.md §3.3 step 4 (spec 1.3.0) is not a blanket refusal of `;`: an ordinary semicolon still takes U+202F under N2. Pinned next to the reference cases so the guard cannot be widened into one."
|
|
781
|
+
},
|
|
782
|
+
{
|
|
783
|
+
"id": "fr-nbsp-semicolon-after-digit-binds",
|
|
784
|
+
"rule": "nbsp",
|
|
785
|
+
"mode": "text",
|
|
786
|
+
"in": "Section 4; suite",
|
|
787
|
+
"out": "Section 4 ; suite",
|
|
788
|
+
"note": "nbsp.md §3.3 step 4 (spec 1.3.0): the left walk finds the digit run but no `&` before it, so the guard does not fire."
|
|
789
|
+
},
|
|
790
|
+
{
|
|
791
|
+
"id": "fr-nbsp-narrow-substituted",
|
|
792
|
+
"rule": "nbsp",
|
|
793
|
+
"mode": "text",
|
|
794
|
+
"narrowNbsp": "nbsp",
|
|
795
|
+
"in": "Un délai ? Vraiment ! Et puis ; voilà.",
|
|
796
|
+
"out": "Un délai ? Vraiment ! Et puis ; voilà.",
|
|
797
|
+
"note": "nbsp.md §3.1a (spec 1.3.0): `narrowNbsp: \"nbsp\"` moves N2's target from U+202F to U+00A0. The same indices are claimed by the same sub-rule under the same guards — only the character written changes. Compare fr-nbsp-narrow-default, which is this input with the option absent."
|
|
798
|
+
},
|
|
799
|
+
{
|
|
800
|
+
"id": "fr-nbsp-narrow-default",
|
|
801
|
+
"rule": "nbsp",
|
|
802
|
+
"mode": "text",
|
|
803
|
+
"in": "Un délai ? Vraiment ! Et puis ; voilà.",
|
|
804
|
+
"out": "Un délai ? Vraiment ! Et puis ; voilà.",
|
|
805
|
+
"note": "nbsp.md §3.1a: the control for fr-nbsp-narrow-substituted. No `narrowNbsp` key means \"narrow\", so every case written before spec 1.3.0 keeps its meaning and a runtime that ignores the option unknowingly still passes this one — which is why the substituted case above is the one that proves the feature."
|
|
806
|
+
},
|
|
807
|
+
{
|
|
808
|
+
"id": "fr-nbsp-narrow-substituted-authored",
|
|
809
|
+
"rule": "nbsp",
|
|
810
|
+
"mode": "text",
|
|
811
|
+
"narrowNbsp": "nbsp",
|
|
812
|
+
"in": "Oui ? Non !",
|
|
813
|
+
"out": "Oui ? Non !",
|
|
814
|
+
"note": "nbsp.md §3.1a: an authored U+202F at an index N2 claims is normalised to the substituted target, exactly as an authored U+00A0 is normalised to U+202F in the default configuration — the rule normalises a claimed index to its target, and the target has moved. This is the case that separates a target rewrite from post-processing: a caller's replaceAll would also produce this output, but would not be a fixed point when its own output is fed back in."
|
|
815
|
+
},
|
|
816
|
+
{
|
|
817
|
+
"id": "fr-nbsp-narrow-substituted-authored-default",
|
|
818
|
+
"rule": "nbsp",
|
|
819
|
+
"mode": "text",
|
|
820
|
+
"in": "Oui ? Non !",
|
|
821
|
+
"out": "Oui ? Non !",
|
|
822
|
+
"note": "nbsp.md §3.1a: the control for the case above — with the default target an authored U+202F is already correct and nothing is emitted."
|
|
823
|
+
},
|
|
824
|
+
{
|
|
825
|
+
"id": "fr-nbsp-narrow-substituted-with-quotes",
|
|
826
|
+
"rule": "nbsp",
|
|
827
|
+
"mode": "text",
|
|
828
|
+
"narrowNbsp": "nbsp",
|
|
829
|
+
"in": "Il a dit : « oui » ; puis ?",
|
|
830
|
+
"out": "Il a dit : « oui » ; puis ?",
|
|
831
|
+
"note": "nbsp.md §3.1a with N1, N2 and N8 all claiming indices in one string. N1 (the colon) and N8 (`fr`'s primary pair, whose innerSpace is \"nbsp\") already wrote U+00A0 and are untouched by the option; N2 (`;` and `?`) moves. With the substitution on, all three want the same character — the option can only ever make N2's target EQUAL to N1's, never different, which is why §5's clause 0 needs no new case."
|
|
832
|
+
},
|
|
833
|
+
{
|
|
834
|
+
"id": "fr-nbsp-narrow-substituted-guards-unchanged",
|
|
835
|
+
"rule": "nbsp",
|
|
836
|
+
"mode": "text",
|
|
837
|
+
"narrowNbsp": "nbsp",
|
|
838
|
+
"in": "12:30 et http://x ; oui",
|
|
839
|
+
"out": "12:30 et http://x ; oui",
|
|
840
|
+
"note": "nbsp.md §3.1a: the option changes what the rule writes, never what it reads or which indices it claims. N1's right-context guard still protects the time and the URL — the colon in `12:30` and in `http://` takes nothing, under the substitution exactly as under the default."
|
|
841
|
+
},
|
|
842
|
+
{
|
|
843
|
+
"id": "fr-nbsp-quote-inner-space-authored-narrow",
|
|
844
|
+
"rule": "nbsp",
|
|
845
|
+
"mode": "text",
|
|
846
|
+
"in": "Il a dit « mot ».",
|
|
847
|
+
"out": "Il a dit « mot ».",
|
|
848
|
+
"note": "nbsp.md §3.10 and §6 rows 3-4, corrected in spec 1.3.0. `fr.json` sets the primary pair's innerSpace to \"nbsp\", so N8's target is U+00A0 and an authored U+202F inside the quotation marks is CONVERTED, not already correct. Through spec 1.2.0 the worked-example table claimed both the narrow space and a fixed point here — a claim about a locale file the file never supported, and nothing pinned it. This case pins it."
|
|
849
|
+
},
|
|
850
|
+
{
|
|
851
|
+
"id": "fr-yaml-narrow-space-in-quoted",
|
|
852
|
+
"rule": "nbsp",
|
|
853
|
+
"mode": "yaml",
|
|
854
|
+
"keys": [
|
|
855
|
+
"description"
|
|
856
|
+
],
|
|
857
|
+
"in": "description: \"Une note !\"\n",
|
|
858
|
+
"out": "description: \"Une note !\"\n",
|
|
859
|
+
"note": "modes.md §3.8.6: a quoted scalar's content is one span, and French spacing applies inside it exactly as in text mode."
|
|
860
|
+
},
|
|
861
|
+
{
|
|
862
|
+
"id": "fr-yaml-no-space-across-colon-split",
|
|
863
|
+
"rule": "nbsp",
|
|
864
|
+
"mode": "yaml",
|
|
865
|
+
"keys": [
|
|
866
|
+
"description"
|
|
867
|
+
],
|
|
868
|
+
"in": "description: prose:more prose\n",
|
|
869
|
+
"out": "description: prose:more prose\n",
|
|
870
|
+
"note": "modes.md §3.8.6: the `:` is outside every span, so the insertion `nbsp` would make before it lands on a span edge and is discarded by §3.4. This is the accepted cost of the split, and it is what keeps the scalar parseable."
|
|
871
|
+
},
|
|
872
|
+
{
|
|
873
|
+
"id": "fr-yaml-guillemets-across-block-lines",
|
|
874
|
+
"rule": "quotes",
|
|
875
|
+
"mode": "yaml",
|
|
876
|
+
"keys": [
|
|
877
|
+
"description"
|
|
878
|
+
],
|
|
879
|
+
"in": "description: |\n Il a dit \"bonjour\n monsieur\" et il est parti\n",
|
|
880
|
+
"out": "description: |\n Il a dit « bonjour\n monsieur » et il est parti\n",
|
|
881
|
+
"note": "modes.md §3.2 model C in a locale whose quotation output is guillemets plus U+00A0 rather than curly quotes — structurally different from the en-US case of the same shape."
|
|
707
882
|
}
|
|
708
883
|
]
|
|
709
884
|
}
|
|
@@ -0,0 +1,161 @@
|
|
|
1
|
+
{
|
|
2
|
+
"spec": "1.3.1",
|
|
3
|
+
"locale": "it",
|
|
4
|
+
"cases": [
|
|
5
|
+
{
|
|
6
|
+
"id": "it-quotes-primary",
|
|
7
|
+
"rule": "quotes",
|
|
8
|
+
"mode": "text",
|
|
9
|
+
"in": "Ha detto \"buongiorno\" ed è uscito.",
|
|
10
|
+
"out": "Ha detto «buongiorno» ed è uscito.",
|
|
11
|
+
"note": "quotes.primary = U+00AB/U+00BB, innerSpace \"none\" (Manuale interistituzionale 4.2.3 livello 1; Parte quarta 10.1.7 «Le virgolette non richiedono spazio né in apertura né in chiusura»). Contrasto diretto con fr, che qui inserisce U+00A0 all'interno."
|
|
12
|
+
},
|
|
13
|
+
{
|
|
14
|
+
"id": "it-quotes-nested",
|
|
15
|
+
"rule": "quotes",
|
|
16
|
+
"mode": "text",
|
|
17
|
+
"in": "Scrisse: \"Mi disse 'arrivo subito' e sparì.\"",
|
|
18
|
+
"out": "Scrisse: «Mi disse “arrivo subito” e sparì.»",
|
|
19
|
+
"note": "Livello 1 caporali, livello 2 virgolette alte. Il livello 3 del Manuale non è rappresentabile nello schema."
|
|
20
|
+
},
|
|
21
|
+
{
|
|
22
|
+
"id": "it-quotes-inner-space-deleted",
|
|
23
|
+
"rule": "quotes",
|
|
24
|
+
"mode": "text",
|
|
25
|
+
"in": "Ha detto « buongiorno » ed è uscito.",
|
|
26
|
+
"out": "Ha detto «buongiorno» ed è uscito.",
|
|
27
|
+
"note": "Con innerSpace \"none\" la regola CANCELLA lo spazio interno: un testo italiano copiato da una fonte francese si normalizza."
|
|
28
|
+
},
|
|
29
|
+
{
|
|
30
|
+
"id": "it-quotes-elision-before-quotation",
|
|
31
|
+
"rule": "quotes",
|
|
32
|
+
"mode": "text",
|
|
33
|
+
"in": "il titolo dell'\"amico\" ritrovato",
|
|
34
|
+
"out": "il titolo dell’«amico» ritrovato",
|
|
35
|
+
"note": "Il caso di maggiore impatto reale per l'italiano, che elide molto più del francese: l'apostrofo di elisione diventa U+2019 (apostrophe.md caso 3a, spec 1.2.0) e non viene rivendicato come segno di apertura, mentre le virgolette restano prive di spazio interno."
|
|
36
|
+
},
|
|
37
|
+
{
|
|
38
|
+
"id": "it-dashes-parenthetical-em-spaced",
|
|
39
|
+
"rule": "dashes",
|
|
40
|
+
"mode": "text",
|
|
41
|
+
"in": "La casa - se casa si poteva definire quel rudere - sorgeva ai piedi del colle.",
|
|
42
|
+
"out": "La casa — se casa si poteva definire quel rudere — sorgeva ai piedi del colle.",
|
|
43
|
+
"note": "dash.parenthetical = \"em-spaced\": U+2014 con U+0020 su ciascun lato. Frase dell'esempio stampato al punto 10.1.10 del Manuale."
|
|
44
|
+
},
|
|
45
|
+
{
|
|
46
|
+
"id": "it-dashes-compound-hyphen",
|
|
47
|
+
"rule": "dashes",
|
|
48
|
+
"mode": "text",
|
|
49
|
+
"in": "un trattato italo-francese",
|
|
50
|
+
"out": "un trattato italo-francese",
|
|
51
|
+
"note": "Guardia P1 di dashes.md §3.2: un trattino nudo e non spaziato è una parola composta. Esempio stampato al punto 10.1.11."
|
|
52
|
+
},
|
|
53
|
+
{
|
|
54
|
+
"id": "it-dashes-date-range-accepted-cost",
|
|
55
|
+
"rule": "dashes",
|
|
56
|
+
"mode": "text",
|
|
57
|
+
"in": "25 maggio - 10 giugno 1985",
|
|
58
|
+
"out": "25 maggio — 10 giugno 1985",
|
|
59
|
+
"note": "COSTO ACCETTATO, verificato sul motore e fissato qui perché non venga riscoperto come bug. Il punto 10.1.11 del Manuale prescrive il trattino spaziato per collegare due date di mesi diversi, ma la guardia P1 converte qualunque trattino isolato e spaziato fra parole. Non è una particolarità italiana: fr, ru, de-DE e en-GB, già pubblicate, fanno lo stesso su questa identica stringa."
|
|
60
|
+
},
|
|
61
|
+
{
|
|
62
|
+
"id": "it-ranges-none-years",
|
|
63
|
+
"rule": "ranges",
|
|
64
|
+
"mode": "text",
|
|
65
|
+
"in": "anno accademico 2010-2011",
|
|
66
|
+
"out": "anno accademico 2010-2011",
|
|
67
|
+
"note": "ranges.md §3: il token non viene convertito perché dash.range è \"none\" — il punto 10.3.1 del Manuale prescrive il trattino U+002D, già presente, e l'enum non ha un valore per esso. Il caso è generato con ranges attivato esplicitamente, dato che la regola è disattivata per impostazione predefinita.",
|
|
68
|
+
"rules": {
|
|
69
|
+
"ranges": true
|
|
70
|
+
}
|
|
71
|
+
},
|
|
72
|
+
{
|
|
73
|
+
"id": "it-ellipsis-three-dots",
|
|
74
|
+
"rule": "ellipsis",
|
|
75
|
+
"mode": "text",
|
|
76
|
+
"in": "Ti chiami Leone... ma sei un coniglio.",
|
|
77
|
+
"out": "Ti chiami Leone… ma sei un coniglio.",
|
|
78
|
+
"note": "Tre U+002E diventano U+2026. Frase dell'esempio stampato al punto 10.1.8."
|
|
79
|
+
},
|
|
80
|
+
{
|
|
81
|
+
"id": "it-ellipsis-not-abbreviated-after-terminal",
|
|
82
|
+
"rule": "ellipsis",
|
|
83
|
+
"mode": "text",
|
|
84
|
+
"in": "Davvero?..",
|
|
85
|
+
"out": "Davvero?…",
|
|
86
|
+
"note": "abbreviatedAfterTerminal = false: la forma a due punti del russo non esiste in italiano (10.1.8 «costituito da tre punti»; Crusca «sempre nel numero di tre»)."
|
|
87
|
+
},
|
|
88
|
+
{
|
|
89
|
+
"id": "it-hyphen-noop",
|
|
90
|
+
"rule": "hyphen",
|
|
91
|
+
"mode": "text",
|
|
92
|
+
"in": "un porta-finestra e un maxi-schermo",
|
|
93
|
+
"out": "un porta-finestra e un maxi-schermo",
|
|
94
|
+
"note": "Le tre liste di hyphen sono vuote, quindi la regola è un no-op totale e dimostrabile. Composti tratti da Treccani, «Trattino [prontuario]»."
|
|
95
|
+
},
|
|
96
|
+
{
|
|
97
|
+
"id": "it-nbsp-no-space-before-punctuation",
|
|
98
|
+
"rule": "nbsp",
|
|
99
|
+
"mode": "text",
|
|
100
|
+
"in": "Che cosa dici ? Nulla ! Davvero ; sì : nulla.",
|
|
101
|
+
"out": "Che cosa dici? Nulla! Davvero; sì: nulla.",
|
|
102
|
+
"note": "IL CASO CHE FISSA LA DIFFERENZA CON IL FRANCESE: spaces elimina lo spazio ordinario prima dei segni e nbsp non lo reinserisce, perché entrambe le liste sono vuote. Lo stesso input in fr produce U+202F prima di ? ! ; e U+00A0 prima di :."
|
|
103
|
+
},
|
|
104
|
+
{
|
|
105
|
+
"id": "it-nbsp-percent",
|
|
106
|
+
"rule": "nbsp",
|
|
107
|
+
"mode": "text",
|
|
108
|
+
"in": "un aumento del 45 % quest'anno",
|
|
109
|
+
"out": "un aumento del 45 % quest’anno",
|
|
110
|
+
"note": "N5 beforeUnits: lo spazio diventa U+00A0 (BIPM). L'apostrofo è opera della regola apostrophe, non di nbsp."
|
|
111
|
+
},
|
|
112
|
+
{
|
|
113
|
+
"id": "it-nbsp-celsius",
|
|
114
|
+
"rule": "nbsp",
|
|
115
|
+
"mode": "text",
|
|
116
|
+
"in": "una temperatura di 20 °C",
|
|
117
|
+
"out": "una temperatura di 20 °C",
|
|
118
|
+
"note": "Stessa subrregola su «°C», che è U+00B0 seguito da U+0043."
|
|
119
|
+
},
|
|
120
|
+
{
|
|
121
|
+
"id": "it-nbsp-before-number",
|
|
122
|
+
"rule": "nbsp",
|
|
123
|
+
"mode": "text",
|
|
124
|
+
"in": "il regolamento n. 3600 e cfr. pag. 2064",
|
|
125
|
+
"out": "il regolamento n. 3600 e cfr. pag. 2064",
|
|
126
|
+
"note": "N9 beforeNumber su «n.» e «pag.». Riga accettata come INFERENZA DICHIARATA e non come citazione piena: il riquadro «Spazi vuoti protetti» del punto 4.2.3 non è marcato come italiano, ma «pag.» è forma italiana. Si noti che «cfr.» non è in elenco e il suo spazio resta U+0020."
|
|
127
|
+
},
|
|
128
|
+
{
|
|
129
|
+
"id": "it-apostrophe-elision",
|
|
130
|
+
"rule": "apostrophe",
|
|
131
|
+
"mode": "text",
|
|
132
|
+
"in": "un'utopia, un'assurda aspirazione dell'Unione",
|
|
133
|
+
"out": "un’utopia, un’assurda aspirazione dell’Unione",
|
|
134
|
+
"note": "U+0027 diventa U+2019. Parole tratte dalla prosa del Manuale. La regola non legge dati di locale, ma l'elisione è la forma che il motore incontra più spesso in italiano."
|
|
135
|
+
},
|
|
136
|
+
{
|
|
137
|
+
"id": "it-roundtrip-noop",
|
|
138
|
+
"rule": "quotes",
|
|
139
|
+
"mode": "text",
|
|
140
|
+
"in": "Ha detto «buongiorno» ed è uscito…",
|
|
141
|
+
"out": "Ha detto «buongiorno» ed è uscito…",
|
|
142
|
+
"note": "modes.md §4: un documento già composto torna byte per byte."
|
|
143
|
+
},
|
|
144
|
+
{
|
|
145
|
+
"id": "it-spaces-brackets-and-final-stop",
|
|
146
|
+
"rule": "spaces",
|
|
147
|
+
"mode": "text",
|
|
148
|
+
"in": "Disse ( senza dubbio ) di sì .",
|
|
149
|
+
"out": "Disse (senza dubbio) di sì.",
|
|
150
|
+
"note": "spaces.md §3: gli spazi interni alle parentesi spariscono e la serie prima del punto finale si riduce. Regola indipendente dalla locale, presente qui perché it abbia un caso per ogni regola canonica."
|
|
151
|
+
},
|
|
152
|
+
{
|
|
153
|
+
"id": "it-symbols-copyright-and-dimensions",
|
|
154
|
+
"rule": "symbols",
|
|
155
|
+
"mode": "text",
|
|
156
|
+
"in": "Copyright (c) 2026, formato 40x60 cm",
|
|
157
|
+
"out": "Copyright © 2026, formato 40×60 cm",
|
|
158
|
+
"note": "symbols.md: «(c)» diventa U+00A9 e la «x» fra cifre U+00D7. L'U+00A0 davanti a «cm» è opera di nbsp N5, non di symbols: i due casi stanno insieme per mostrare che le regole si compongono."
|
|
159
|
+
}
|
|
160
|
+
]
|
|
161
|
+
}
|