polytypo 1.1.0 → 1.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/README.md +33 -1
- data/lib/polytypo/data/VERSION +1 -1
- data/lib/polytypo/data/fixtures/cs.json +161 -0
- data/lib/polytypo/data/fixtures/de-CH.json +9 -1
- data/lib/polytypo/data/fixtures/de-DE.json +227 -6
- data/lib/polytypo/data/fixtures/el.json +9 -1
- data/lib/polytypo/data/fixtures/en-GB.json +28 -1
- data/lib/polytypo/data/fixtures/en-US.json +690 -1
- data/lib/polytypo/data/fixtures/es.json +193 -0
- data/lib/polytypo/data/fixtures/fi.json +9 -1
- data/lib/polytypo/data/fixtures/fr-CA.json +50 -1
- data/lib/polytypo/data/fixtures/fr.json +282 -1
- data/lib/polytypo/data/fixtures/it.json +161 -0
- data/lib/polytypo/data/fixtures/locale-resolution.json +76 -4
- data/lib/polytypo/data/fixtures/nl.json +121 -0
- data/lib/polytypo/data/fixtures/pl.json +137 -0
- data/lib/polytypo/data/fixtures/pt-BR.json +156 -0
- data/lib/polytypo/data/fixtures/pt-PT.json +156 -0
- data/lib/polytypo/data/fixtures/ru.json +47 -1
- data/lib/polytypo/data/fixtures/sv.json +9 -1
- data/lib/polytypo/data/fixtures/uk.json +153 -0
- data/lib/polytypo/data/locales/cs.json +90 -0
- data/lib/polytypo/data/locales/de-DE.json +7 -2
- data/lib/polytypo/data/locales/en-US.json +3 -3
- data/lib/polytypo/data/locales/es.json +111 -0
- data/lib/polytypo/data/locales/fr-CA.json +7 -1
- data/lib/polytypo/data/locales/fr.json +7 -1
- data/lib/polytypo/data/locales/it.json +95 -0
- data/lib/polytypo/data/locales/nl.json +84 -0
- data/lib/polytypo/data/locales/pl.json +96 -0
- data/lib/polytypo/data/locales/pt-BR.json +82 -0
- data/lib/polytypo/data/locales/pt-PT.json +84 -0
- data/lib/polytypo/data/locales/registry.json +23 -3
- data/lib/polytypo/data/locales/ru.json +2 -2
- data/lib/polytypo/data/locales/uk.json +130 -0
- data/lib/polytypo/data/rules/analyze.md +157 -0
- data/lib/polytypo/data/rules/apostrophe.md +432 -0
- data/lib/polytypo/data/rules/dashes.md +128 -37
- data/lib/polytypo/data/rules/ellipsis.md +271 -0
- data/lib/polytypo/data/rules/hyphen.md +353 -0
- data/lib/polytypo/data/rules/locale-resolution.md +239 -0
- data/lib/polytypo/data/rules/modes.md +1281 -0
- data/lib/polytypo/data/rules/nbsp.md +1157 -0
- data/lib/polytypo/data/rules/order.json +11 -11
- data/lib/polytypo/data/rules/pipeline-idempotency.md +605 -0
- data/lib/polytypo/data/rules/quotes.md +1324 -0
- data/lib/polytypo/data/rules/ranges.md +489 -0
- data/lib/polytypo/data/rules/spaces.md +649 -0
- data/lib/polytypo/data/rules/symbols.md +540 -0
- data/lib/polytypo/data/schema/fixtures.schema.json +18 -3
- data/lib/polytypo/engine/origin.rb +75 -0
- data/lib/polytypo/engine/pipeline.rb +72 -1
- data/lib/polytypo/engine/rules/apostrophe.rb +10 -1
- data/lib/polytypo/engine/rules/dash_shared.rb +85 -3
- data/lib/polytypo/engine/rules/dashes.rb +4 -1
- data/lib/polytypo/engine/rules/nbsp.rb +53 -20
- data/lib/polytypo/engine/rules/ranges.rb +24 -20
- data/lib/polytypo/engine/rules/spaces.rb +8 -1
- data/lib/polytypo/errors.rb +3 -0
- data/lib/polytypo/modes/runner.rb +17 -0
- data/lib/polytypo/modes/spans.rb +30 -2
- data/lib/polytypo/modes/yaml.rb +312 -0
- data/lib/polytypo/version.rb +1 -1
- data/lib/polytypo.rb +126 -15
- metadata +31 -1
|
@@ -0,0 +1,193 @@
|
|
|
1
|
+
{
|
|
2
|
+
"spec": "1.3.0",
|
|
3
|
+
"locale": "es",
|
|
4
|
+
"cases": [
|
|
5
|
+
{
|
|
6
|
+
"id": "es-quotes-primary",
|
|
7
|
+
"rule": "quotes",
|
|
8
|
+
"mode": "text",
|
|
9
|
+
"in": "Ha dicho \"buenos días\" y salió.",
|
|
10
|
+
"out": "Ha dicho «buenos días» y salió.",
|
|
11
|
+
"note": "quotes.primary = U+00AB/U+00BB con innerSpace \"none\" (DPD «comillas» 1: «se recomienda utilizar en primera instancia las comillas angulares»; «pegadas a la primera y la última palabra»)."
|
|
12
|
+
},
|
|
13
|
+
{
|
|
14
|
+
"id": "es-quotes-nested",
|
|
15
|
+
"rule": "quotes",
|
|
16
|
+
"mode": "text",
|
|
17
|
+
"in": "Escribió: \"Me dijo 'llego enseguida' y desapareció.\"",
|
|
18
|
+
"out": "Escribió: «Me dijo “llego enseguida” y desapareció.»",
|
|
19
|
+
"note": "Nivel 1 angulares, nivel 2 inglesas, según el orden que fija el DPD. El tercer nivel del DPD (simples) no es expresable en el esquema."
|
|
20
|
+
},
|
|
21
|
+
{
|
|
22
|
+
"id": "es-quotes-inner-space-deleted",
|
|
23
|
+
"rule": "quotes",
|
|
24
|
+
"mode": "text",
|
|
25
|
+
"in": "Ha dicho « buenos días » y salió.",
|
|
26
|
+
"out": "Ha dicho «buenos días» y salió.",
|
|
27
|
+
"note": "innerSpace \"none\": N8 elimina el espacio interior que un texto copiado de una fuente francesa trae consigo."
|
|
28
|
+
},
|
|
29
|
+
{
|
|
30
|
+
"id": "es-ellipsis-three-dots",
|
|
31
|
+
"rule": "ellipsis",
|
|
32
|
+
"mode": "text",
|
|
33
|
+
"in": "Espera... ¿qué?",
|
|
34
|
+
"out": "Espera… ¿qué?",
|
|
35
|
+
"note": "Tres U+002E pasan a U+2026 (DPD «puntos suspensivos» 1: «tres puntos consecutivos ―y solo tres―»)."
|
|
36
|
+
},
|
|
37
|
+
{
|
|
38
|
+
"id": "es-ellipsis-not-abbreviated-after-terminal",
|
|
39
|
+
"rule": "ellipsis",
|
|
40
|
+
"mode": "text",
|
|
41
|
+
"in": "¿De verdad?..",
|
|
42
|
+
"out": "¿De verdad?…",
|
|
43
|
+
"note": "abbreviatedAfterTerminal = false: la forma de dos puntos tras «?» que usa el ruso no existe en español, así que los dos puntos se completan a U+2026 en vez de conservarse."
|
|
44
|
+
},
|
|
45
|
+
{
|
|
46
|
+
"id": "es-dashes-parenthetical-none",
|
|
47
|
+
"rule": "dashes",
|
|
48
|
+
"mode": "text",
|
|
49
|
+
"in": "El plan - si existe - fracasa.",
|
|
50
|
+
"out": "El plan - si existe - fracasa.",
|
|
51
|
+
"note": "dash.parenthetical = \"none\": el DPD prescribe raya con espacios FUERA del par y ninguno dentro, asimetría que el enum no expresa; mientras tanto la regla no toca nada. Véase la voz dashes de es.json."
|
|
52
|
+
},
|
|
53
|
+
{
|
|
54
|
+
"id": "es-ranges-guion-none",
|
|
55
|
+
"rule": "ranges",
|
|
56
|
+
"mode": "text",
|
|
57
|
+
"in": "las páginas 23-45",
|
|
58
|
+
"out": "las páginas 23-45",
|
|
59
|
+
"note": "ranges.md §3: el token no se convierte porque dash.range es \"none\" — el signo español del intervalo es el guion U+002D, que ya está escrito (DPD «guion» 3 b), y el enum no tiene un valor para él. El caso se genera con ranges activado explícitamente, ya que la regla está desactivada por defecto.",
|
|
60
|
+
"rules": {
|
|
61
|
+
"ranges": true
|
|
62
|
+
}
|
|
63
|
+
},
|
|
64
|
+
{
|
|
65
|
+
"id": "es-hyphen-noop",
|
|
66
|
+
"rule": "hyphen",
|
|
67
|
+
"mode": "text",
|
|
68
|
+
"in": "un estudio teórico-práctico",
|
|
69
|
+
"out": "un estudio teórico-práctico",
|
|
70
|
+
"note": "Las tres listas de hyphen están vacías, así que la regla es un no-op total demostrable para el español (hyphen.md §2)."
|
|
71
|
+
},
|
|
72
|
+
{
|
|
73
|
+
"id": "es-nbsp-inverted-marks-untouched",
|
|
74
|
+
"rule": "nbsp",
|
|
75
|
+
"mode": "text",
|
|
76
|
+
"in": "¿Qué hora es? ¡Vaya!",
|
|
77
|
+
"out": "¿Qué hora es? ¡Vaya!",
|
|
78
|
+
"note": "beforePunctuation y narrowBeforePunctuation vacías: el español no pone espacio antes de «?» ni de «!», y U+00BF/U+00A1 no son miembros de CLOSEISH, así que la restricción Q-P impediría expresarlo aunque la fuente lo pidiera."
|
|
79
|
+
},
|
|
80
|
+
{
|
|
81
|
+
"id": "es-nbsp-degree-tight",
|
|
82
|
+
"rule": "nbsp",
|
|
83
|
+
"mode": "text",
|
|
84
|
+
"in": "Hacía 27° a la sombra",
|
|
85
|
+
"out": "Hacía 27° a la sombra",
|
|
86
|
+
"note": "El DPD «símbolo» 5.4 pega «°» a la cifra cuando no se especifica la escala, por eso «°» a secas NO está en beforeUnits y aquí no se inserta nada."
|
|
87
|
+
},
|
|
88
|
+
{
|
|
89
|
+
"id": "es-nbsp-percent",
|
|
90
|
+
"rule": "nbsp",
|
|
91
|
+
"mode": "text",
|
|
92
|
+
"in": "Aprobó un 8 % de los alumnos",
|
|
93
|
+
"out": "Aprobó un 8 % de los alumnos",
|
|
94
|
+
"note": "N5 beforeUnits: el espacio pasa a U+00A0. RAE, Duda lingüística: «El símbolo del porcentaje se separa con espacio de la cifra que lo precede: 50 %»; la indivisibilidad la impone el DPD «símbolo» 5.6."
|
|
95
|
+
},
|
|
96
|
+
{
|
|
97
|
+
"id": "es-nbsp-celsius",
|
|
98
|
+
"rule": "nbsp",
|
|
99
|
+
"mode": "text",
|
|
100
|
+
"in": "La temperatura es de 27 °C",
|
|
101
|
+
"out": "La temperatura es de 27 °C",
|
|
102
|
+
"note": "«°C» sí está en beforeUnits, citado literalmente por el DPD «símbolo» 5.4 («27° […] pero 27 °C»). Contrasta con es-nbsp-degree-tight."
|
|
103
|
+
},
|
|
104
|
+
{
|
|
105
|
+
"id": "es-nbsp-abbreviation-internal-space",
|
|
106
|
+
"rule": "nbsp",
|
|
107
|
+
"mode": "text",
|
|
108
|
+
"in": "Viajó a los EE. UU. el lunes",
|
|
109
|
+
"out": "Viajó a los EE. UU. el lunes",
|
|
110
|
+
"note": "N4 abbreviations: el espacio interior de una abreviatura de varios elementos pasa a U+00A0 (DPD «abreviatura»: «Cuando la abreviatura se compone de varios elementos, estos no deben separarse en líneas diferentes»)."
|
|
111
|
+
},
|
|
112
|
+
{
|
|
113
|
+
"id": "es-nbsp-abbreviation-p-ej",
|
|
114
|
+
"rule": "nbsp",
|
|
115
|
+
"mode": "text",
|
|
116
|
+
"in": "algunos casos, p. ej. este",
|
|
117
|
+
"out": "algunos casos, p. ej. este",
|
|
118
|
+
"note": "La misma subrregla sobre «p. ej.», forma atestiguada en la Lista de abreviaturas de El buen uso del español."
|
|
119
|
+
},
|
|
120
|
+
{
|
|
121
|
+
"id": "es-nbsp-before-number",
|
|
122
|
+
"rule": "nbsp",
|
|
123
|
+
"mode": "text",
|
|
124
|
+
"in": "véase la pág. 23",
|
|
125
|
+
"out": "véase la pág. 23",
|
|
126
|
+
"note": "N9 beforeNumber: «⊗15 / págs.» es el ejemplo del propio DPD «abreviatura» para este caso."
|
|
127
|
+
},
|
|
128
|
+
{
|
|
129
|
+
"id": "es-nbsp-before-word",
|
|
130
|
+
"rule": "nbsp",
|
|
131
|
+
"mode": "text",
|
|
132
|
+
"in": "El Sr. Pérez llegó tarde",
|
|
133
|
+
"out": "El Sr. Pérez llegó tarde",
|
|
134
|
+
"note": "N10 beforeWord: «⊗Sr. / Pérez» es el segundo ejemplo de la misma entrada. La lista se limita a los tratamientos de cortesía, por el riesgo de falsos positivos de esta subrregla."
|
|
135
|
+
},
|
|
136
|
+
{
|
|
137
|
+
"id": "es-nbsp-initials-chain",
|
|
138
|
+
"rule": "nbsp",
|
|
139
|
+
"mode": "text",
|
|
140
|
+
"in": "Lo escribió J. A. Pérez ayer",
|
|
141
|
+
"out": "Lo escribió J. A. Pérez ayer",
|
|
142
|
+
"note": "initialBinding = \"chain\": la cadena completa se liga, incluido el espacio que precede a la primera inicial, igual que en en-US y ru. Decisión del operador (18.09.2026) ante una fuente que no distingue entre una inicial y dos; véase la voz nbsp correspondiente."
|
|
143
|
+
},
|
|
144
|
+
{
|
|
145
|
+
"id": "es-nbsp-lone-initial-not-bound",
|
|
146
|
+
"rule": "nbsp",
|
|
147
|
+
"mode": "text",
|
|
148
|
+
"in": "La ley N. Roma decidió",
|
|
149
|
+
"out": "La ley N. Roma decidió",
|
|
150
|
+
"note": "El reverso del caso anterior y la razón de elegir \"chain\": una sola mayúscula con punto seguida de un nombre propio NO se liga, de modo que un final de frase no se confunde con una inicial."
|
|
151
|
+
},
|
|
152
|
+
{
|
|
153
|
+
"id": "es-nbsp-ac-guard",
|
|
154
|
+
"rule": "nbsp",
|
|
155
|
+
"mode": "text",
|
|
156
|
+
"in": "en el año 33 a. C. Roma era una república",
|
|
157
|
+
"out": "en el año 33 a. C. Roma era una república",
|
|
158
|
+
"note": "N4 liga el espacio interior de «a. C.» y la guarda C1-a (nbsp.md §3.9) impide que C1 lea «C.» como inicial y «Roma» como apellido. Equivalente español del alemán «z. B. Berlin»."
|
|
159
|
+
},
|
|
160
|
+
{
|
|
161
|
+
"id": "es-apostrophe-foreign-name",
|
|
162
|
+
"rule": "apostrophe",
|
|
163
|
+
"mode": "text",
|
|
164
|
+
"in": "L'Hospitalet de Llobregat",
|
|
165
|
+
"out": "L’Hospitalet de Llobregat",
|
|
166
|
+
"note": "U+0027 pasa a U+2019. El DPD «apóstrofo» limita el signo en español a textos antiguos, habla reproducida y nombres de otras lenguas: este es el tercer caso, y es el que un texto español encuentra de verdad."
|
|
167
|
+
},
|
|
168
|
+
{
|
|
169
|
+
"id": "es-roundtrip-noop",
|
|
170
|
+
"rule": "quotes",
|
|
171
|
+
"mode": "text",
|
|
172
|
+
"in": "Ha dicho «buenos días» y salió…",
|
|
173
|
+
"out": "Ha dicho «buenos días» y salió…",
|
|
174
|
+
"note": "modes.md §4: un texto ya compuesto vuelve byte a byte."
|
|
175
|
+
},
|
|
176
|
+
{
|
|
177
|
+
"id": "es-spaces-brackets-and-final-stop",
|
|
178
|
+
"rule": "spaces",
|
|
179
|
+
"mode": "text",
|
|
180
|
+
"in": "Dijo ( sin dudar ) que sí .",
|
|
181
|
+
"out": "Dijo (sin dudar) que sí.",
|
|
182
|
+
"note": "spaces.md §3: los espacios interiores de los paréntesis se eliminan y la serie anterior al punto final se colapsa. Regla independiente de la locale, presente aquí para que es tenga un caso por cada regla canónica."
|
|
183
|
+
},
|
|
184
|
+
{
|
|
185
|
+
"id": "es-symbols-copyright-and-dimensions",
|
|
186
|
+
"rule": "symbols",
|
|
187
|
+
"mode": "text",
|
|
188
|
+
"in": "Copyright (c) 2026, tamaño 40x60 cm",
|
|
189
|
+
"out": "Copyright © 2026, tamaño 40×60 cm",
|
|
190
|
+
"note": "symbols.md: «(c)» pasa a U+00A9 y la «x» entre cifras a U+00D7. El U+00A0 ante «cm» lo pone nbsp N5, no symbols; se deja en el mismo caso porque así se ve que las dos reglas componen."
|
|
191
|
+
}
|
|
192
|
+
]
|
|
193
|
+
}
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
{
|
|
2
|
-
"spec": "1.
|
|
2
|
+
"spec": "1.3.0",
|
|
3
3
|
"locale": "fi",
|
|
4
4
|
"cases": [
|
|
5
5
|
{
|
|
@@ -1309,6 +1309,14 @@
|
|
|
1309
1309
|
"in": "'90s were fun,' he said",
|
|
1310
1310
|
"out": "”90s were fun,” he said",
|
|
1311
1311
|
"note": "quotes.md §6 row E3s: the straight-input control for E3."
|
|
1312
|
+
},
|
|
1313
|
+
{
|
|
1314
|
+
"id": "fi-apostrophe-elision-before-shared-quote-glyph",
|
|
1315
|
+
"rule": "apostrophe",
|
|
1316
|
+
"mode": "text",
|
|
1317
|
+
"in": "lehti l'\"Humanité\"",
|
|
1318
|
+
"out": "lehti l’”Humanité”",
|
|
1319
|
+
"note": "apostrophe.md §3.3 cases 3 and 3a (spec 1.2.0): fi opens and closes with U+201D, which is in CLOSEISH, so case 3 already converted this before 1.2.0. Pinned so that case 3a is not read as changing a shared-glyph locale."
|
|
1312
1320
|
}
|
|
1313
1321
|
]
|
|
1314
1322
|
}
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
{
|
|
2
|
-
"spec": "1.
|
|
2
|
+
"spec": "1.3.0",
|
|
3
3
|
"locale": "fr-CA",
|
|
4
4
|
"cases": [
|
|
5
5
|
{
|
|
@@ -263,6 +263,55 @@
|
|
|
263
263
|
"ranges": true
|
|
264
264
|
},
|
|
265
265
|
"note": "spec 0.5.0: fr-CA's dash.range is \"none\" (no verified convention), so `ranges` substitutes nothing even when explicitly enabled (ranges.md §2, §3.3) — the same invariant the pre-0.5.0 `dashes`-tagged no-op fixtures for this locale already proved for the combined rule."
|
|
266
|
+
},
|
|
267
|
+
{
|
|
268
|
+
"id": "fr-ca-nbsp-span-boundary-strong-colon",
|
|
269
|
+
"rule": "nbsp",
|
|
270
|
+
"mode": "html",
|
|
271
|
+
"in": "<strong>Résistant au gel :</strong> il reste souple",
|
|
272
|
+
"out": "<strong>Résistant au gel :</strong> il reste souple",
|
|
273
|
+
"note": "nbsp.md §3.3 step 2 (spec 1.2.0): same as the `fr` case; `fr-CA` lists `:` in its own punctuation data."
|
|
274
|
+
},
|
|
275
|
+
{
|
|
276
|
+
"id": "fr-ca-nbsp-span-boundary-markdown-strong-colon",
|
|
277
|
+
"rule": "nbsp",
|
|
278
|
+
"mode": "markdown",
|
|
279
|
+
"dialect": "commonmark",
|
|
280
|
+
"in": "**Note :** voir plus bas\n",
|
|
281
|
+
"out": "**Note :** voir plus bas\n",
|
|
282
|
+
"note": "nbsp.md §3.3 step 2 (spec 1.2.0): the colon ends the strong span; the −1 marker after it is in CLOSEISH, so N1 restores the U+00A0 that `spaces` deleted. (fr-CA lists no punctuation for N2, so `?` would take nothing back in any mode.)"
|
|
283
|
+
},
|
|
284
|
+
{
|
|
285
|
+
"id": "fr-ca-apostrophe-elision-before-straight-quote",
|
|
286
|
+
"rule": "apostrophe",
|
|
287
|
+
"mode": "text",
|
|
288
|
+
"in": "l'\"idée\" est simple",
|
|
289
|
+
"out": "l’« idée » est simple",
|
|
290
|
+
"note": "apostrophe.md §3.3 case 3a (spec 1.2.0): same as `fr`."
|
|
291
|
+
},
|
|
292
|
+
{
|
|
293
|
+
"id": "fr-ca-nbsp-numero-binds-number",
|
|
294
|
+
"rule": "nbsp",
|
|
295
|
+
"mode": "text",
|
|
296
|
+
"in": "voir n° 5",
|
|
297
|
+
"out": "voir n° 5",
|
|
298
|
+
"note": "nbsp.md §3.11 N9 : « n° » figure dans nbsp.beforeNumber depuis la spec 1.3.0 et se lie au nombre qui suit. La forme retenue est « n » + U+00B0, l'usage le plus répandu ; voir l'entrée sources de la locale, la source (OQLF) prescrivant un tracé et non un code point."
|
|
299
|
+
},
|
|
300
|
+
{
|
|
301
|
+
"id": "fr-ca-nbsp-numero-capital-is-its-own-entry",
|
|
302
|
+
"rule": "nbsp",
|
|
303
|
+
"mode": "text",
|
|
304
|
+
"in": "N° 5 de la revue",
|
|
305
|
+
"out": "N° 5 de la revue",
|
|
306
|
+
"note": "nbsp.md §3.11 N9 étape 1 : aucune tolérance de casse, « N° » est donc une entrée distincte de « n° » dans nbsp.beforeNumber."
|
|
307
|
+
},
|
|
308
|
+
{
|
|
309
|
+
"id": "fr-ca-nbsp-numero-needs-a-digit",
|
|
310
|
+
"rule": "nbsp",
|
|
311
|
+
"mode": "text",
|
|
312
|
+
"in": "le n° cinq",
|
|
313
|
+
"out": "le n° cinq",
|
|
314
|
+
"note": "nbsp.md §3.11 N9 étape 4 : il faut un DIGIT après le séparateur. Un nombre écrit en toutes lettres n'en est pas un, l'espace reste sécable."
|
|
266
315
|
}
|
|
267
316
|
]
|
|
268
317
|
}
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
{
|
|
2
|
-
"spec": "1.
|
|
2
|
+
"spec": "1.3.0",
|
|
3
3
|
"locale": "fr",
|
|
4
4
|
"cases": [
|
|
5
5
|
{
|
|
@@ -598,6 +598,287 @@
|
|
|
598
598
|
"ranges": true
|
|
599
599
|
},
|
|
600
600
|
"note": "spec 0.5.0: fr's dash.range is \"none\" (no verified convention), so `ranges` substitutes nothing even when explicitly enabled (ranges.md §2, §3.3) — the same invariant the pre-0.5.0 `dashes`-tagged no-op fixtures for this locale already proved for the combined rule."
|
|
601
|
+
},
|
|
602
|
+
{
|
|
603
|
+
"id": "fr-spaces-dot-word-start-dotfile",
|
|
604
|
+
"rule": "spaces",
|
|
605
|
+
"mode": "text",
|
|
606
|
+
"in": "Voir le fichier .htaccess .",
|
|
607
|
+
"out": "Voir le fichier .htaccess.",
|
|
608
|
+
"note": "spaces.md §3.4 word-start clause (spec 1.2.0): the space before `.htaccess` survives, the space before the terminal dot is deleted. `.` is in no `nbsp` list for `fr`, so nothing is put back."
|
|
609
|
+
},
|
|
610
|
+
{
|
|
611
|
+
"id": "fr-nbsp-span-boundary-strong-colon",
|
|
612
|
+
"rule": "nbsp",
|
|
613
|
+
"mode": "html",
|
|
614
|
+
"in": "<strong>Résistant au gel :</strong> il reste souple",
|
|
615
|
+
"out": "<strong>Résistant au gel :</strong> il reste souple",
|
|
616
|
+
"note": "nbsp.md §3.3 step 2 (spec 1.2.0), §6 row 6c; modes.md §3.3: `spaces` deletes the typed space before `:`, and the −1 span boundary marker after the mark is in `nbsp`'s CLOSEISH, so N1 inserts U+00A0 at the mark's index, inside the span. Spec 1.1.0 runtimes produced `gel:</strong>`."
|
|
617
|
+
},
|
|
618
|
+
{
|
|
619
|
+
"id": "fr-nbsp-span-boundary-br-question",
|
|
620
|
+
"rule": "nbsp",
|
|
621
|
+
"mode": "html",
|
|
622
|
+
"in": "Et la réglementation ?<br>Oui",
|
|
623
|
+
"out": "Et la réglementation ?<br>Oui",
|
|
624
|
+
"note": "nbsp.md §3.3 step 2 (spec 1.2.0), §6 row 6d: `<br>` leaves no line terminator in the gap, so the marker is −1; it is in CLOSEISH and N2 restores U+202F."
|
|
625
|
+
},
|
|
626
|
+
{
|
|
627
|
+
"id": "fr-nbsp-span-boundary-link-question",
|
|
628
|
+
"rule": "nbsp",
|
|
629
|
+
"mode": "html",
|
|
630
|
+
"in": "<a href=\"/faq\">Comment ?</a> Oui",
|
|
631
|
+
"out": "<a href=\"/faq\">Comment ?</a> Oui",
|
|
632
|
+
"note": "nbsp.md §3.3 step 2 (spec 1.2.0): the mark ends the link text, the marker follows it, and N2 restores U+202F instead of leaving `Comment?`."
|
|
633
|
+
},
|
|
634
|
+
{
|
|
635
|
+
"id": "fr-nbsp-span-boundary-markdown-strong-colon",
|
|
636
|
+
"rule": "nbsp",
|
|
637
|
+
"mode": "markdown",
|
|
638
|
+
"dialect": "commonmark",
|
|
639
|
+
"in": "**Résistant au gel :** il reste souple\n",
|
|
640
|
+
"out": "**Résistant au gel :** il reste souple\n",
|
|
641
|
+
"note": "nbsp.md §3.3 step 2 (spec 1.2.0): the Markdown form of the French definition pattern; the closing `**` is a skipped region with no line terminator, so the marker is −1 and N1 restores U+00A0."
|
|
642
|
+
},
|
|
643
|
+
{
|
|
644
|
+
"id": "fr-nbsp-span-boundary-not-openish-em",
|
|
645
|
+
"rule": "nbsp",
|
|
646
|
+
"mode": "html",
|
|
647
|
+
"in": "Il dit <em>non</em> ! Oui",
|
|
648
|
+
"out": "Il dit <em>non</em> ! Oui",
|
|
649
|
+
"note": "nbsp.md §3.3 step 3 (spec 1.2.0), §6 row 6e; modes.md §3.3: the marker is not in `nbsp`'s OPENISH, so the quote-glyph guard does not decline and N2 converts the space. A port that applied the quote-glyph guard to the marker would leave a plain U+0020 here."
|
|
650
|
+
},
|
|
651
|
+
{
|
|
652
|
+
"id": "fr-nbsp-span-boundary-not-openish-link",
|
|
653
|
+
"rule": "nbsp",
|
|
654
|
+
"mode": "markdown",
|
|
655
|
+
"dialect": "commonmark",
|
|
656
|
+
"in": "Voir [ceci](http://example.org) : oui\n",
|
|
657
|
+
"out": "Voir [ceci](http://example.org) : oui\n",
|
|
658
|
+
"note": "nbsp.md §3.3 step 3 (spec 1.2.0): the second witness that the quote-glyph guard does not decline at a span boundary. The link text ends two places left of `:`, and N1 still converts the space. The marker's absence from the rest of `nbsp`'s OPENISH is pinned by ru-nbsp-span-boundary-not-openish-short-word."
|
|
659
|
+
},
|
|
660
|
+
{
|
|
661
|
+
"id": "fr-nbsp-span-boundary-mark-alone-in-span",
|
|
662
|
+
"rule": "nbsp",
|
|
663
|
+
"mode": "html",
|
|
664
|
+
"in": "mot<em>!</em>",
|
|
665
|
+
"out": "mot<em>!</em>",
|
|
666
|
+
"note": "nbsp.md §3.3 step 2; modes.md §3.4 edge-growth rule: the mark is alone in its span, so the insertion point is the span edge and the insertion is discarded, as it was before spec 1.2.0."
|
|
667
|
+
},
|
|
668
|
+
{
|
|
669
|
+
"id": "fr-nbsp-span-boundary-colon-before-digit-span",
|
|
670
|
+
"rule": "nbsp",
|
|
671
|
+
"mode": "html",
|
|
672
|
+
"in": "Il est 12:<b>30</b>",
|
|
673
|
+
"out": "Il est 12 :<b>30</b>",
|
|
674
|
+
"note": "nbsp.md §3.3 step 2 (spec 1.2.0), §6 row 6f: the stated cost of accepting the marker. Step 2 cannot see past the span boundary, so a colon that ends one span while a digit starts the next is treated as sentence punctuation."
|
|
675
|
+
},
|
|
676
|
+
{
|
|
677
|
+
"id": "fr-apostrophe-elision-before-guillemet",
|
|
678
|
+
"rule": "apostrophe",
|
|
679
|
+
"mode": "text",
|
|
680
|
+
"in": "un mélange d'« urine diluée »",
|
|
681
|
+
"out": "un mélange d’« urine diluée »",
|
|
682
|
+
"note": "apostrophe.md §3.3 case 3a (spec 1.2.0), §6 row 15: letter left, opening quotation glyph right. Spec 1.1.0 left the apostrophe straight, because no case matched a letter followed by an OPENISH-only glyph."
|
|
683
|
+
},
|
|
684
|
+
{
|
|
685
|
+
"id": "fr-apostrophe-elision-before-straight-quote",
|
|
686
|
+
"rule": "apostrophe",
|
|
687
|
+
"mode": "text",
|
|
688
|
+
"in": "l'\"idée\" est simple",
|
|
689
|
+
"out": "l’« idée » est simple",
|
|
690
|
+
"note": "apostrophe.md §3.2, §3.3 case 3a (spec 1.2.0): `quotes` (order 40) turns the straight `\"` into `«` first, so the mark reaches this rule next to an opening glyph and case 3a converts it."
|
|
691
|
+
},
|
|
692
|
+
{
|
|
693
|
+
"id": "fr-apostrophe-elision-qu-before-guillemet",
|
|
694
|
+
"rule": "apostrophe",
|
|
695
|
+
"mode": "text",
|
|
696
|
+
"in": "Il faut qu'« il vienne »",
|
|
697
|
+
"out": "Il faut qu’« il vienne »",
|
|
698
|
+
"note": "apostrophe.md §3.3 case 3a (spec 1.2.0): `qu'` before a quotation."
|
|
699
|
+
},
|
|
700
|
+
{
|
|
701
|
+
"id": "fr-nbsp-span-boundary-colon-before-url-span",
|
|
702
|
+
"rule": "nbsp",
|
|
703
|
+
"mode": "html",
|
|
704
|
+
"in": "Voir http:<a href=\"//x.org\">//x.org</a>",
|
|
705
|
+
"out": "Voir http :<a href=\"//x.org\">//x.org</a>",
|
|
706
|
+
"note": "nbsp.md §3.3 step 2 (spec 1.2.0): the second stated cost of accepting the marker. Step 2 cannot see the `/` in the next span, so the colon is treated as sentence punctuation and N1 inserts U+00A0."
|
|
707
|
+
},
|
|
708
|
+
{
|
|
709
|
+
"id": "fr-nbsp-numero-binds-number",
|
|
710
|
+
"rule": "nbsp",
|
|
711
|
+
"mode": "text",
|
|
712
|
+
"in": "voir n° 5",
|
|
713
|
+
"out": "voir n° 5",
|
|
714
|
+
"note": "nbsp.md §3.11 N9 : « n° » figure dans nbsp.beforeNumber depuis la spec 1.3.0 et se lie au nombre qui suit. La forme retenue est « n » + U+00B0, l'usage le plus répandu ; voir l'entrée sources de la locale, la source (OQLF) prescrivant un tracé et non un code point."
|
|
715
|
+
},
|
|
716
|
+
{
|
|
717
|
+
"id": "fr-nbsp-numero-capital-is-its-own-entry",
|
|
718
|
+
"rule": "nbsp",
|
|
719
|
+
"mode": "text",
|
|
720
|
+
"in": "N° 5 de la revue",
|
|
721
|
+
"out": "N° 5 de la revue",
|
|
722
|
+
"note": "nbsp.md §3.11 N9 étape 1 : aucune tolérance de casse, « N° » est donc une entrée distincte de « n° » dans nbsp.beforeNumber."
|
|
723
|
+
},
|
|
724
|
+
{
|
|
725
|
+
"id": "fr-nbsp-numero-needs-a-digit",
|
|
726
|
+
"rule": "nbsp",
|
|
727
|
+
"mode": "text",
|
|
728
|
+
"in": "le n° cinq",
|
|
729
|
+
"out": "le n° cinq",
|
|
730
|
+
"note": "nbsp.md §3.11 N9 étape 4 : il faut un DIGIT après le séparateur. Un nombre écrit en toutes lettres n'en est pas un, l'espace reste sécable."
|
|
731
|
+
},
|
|
732
|
+
{
|
|
733
|
+
"id": "fr-markdown-commonmark-frontmatter-nbsp",
|
|
734
|
+
"rule": "nbsp",
|
|
735
|
+
"mode": "markdown",
|
|
736
|
+
"dialect": "commonmark",
|
|
737
|
+
"in": "---\ntitle: Une note ; suite\n---\n\nEst-ce vrai ?\n",
|
|
738
|
+
"out": "---\ntitle: Une note ; suite\n---\n\nEst-ce vrai ?\n",
|
|
739
|
+
"note": "modes.md §3.7.3 gives this exact reason for skipping frontmatter: without it `fr` would put U+00A0 before the colon of a machine-read field and U+202F before its semicolon. The body sentence shows the rule is otherwise active in the same document, so the untouched field proves the skip rather than an inactive rule."
|
|
740
|
+
},
|
|
741
|
+
{
|
|
742
|
+
"id": "fr-markdown-mdx-frontmatter-nbsp",
|
|
743
|
+
"rule": "nbsp",
|
|
744
|
+
"mode": "markdown",
|
|
745
|
+
"dialect": "mdx",
|
|
746
|
+
"in": "---\ntitle: Une note ; suite\n---\n\n<Callout>Est-ce vrai ?</Callout>\n",
|
|
747
|
+
"out": "---\ntitle: Une note ; suite\n---\n\n<Callout>Est-ce vrai ?</Callout>\n",
|
|
748
|
+
"note": "modes.md §3.7.3: the same guarantee under \"mdx\", where frontmatter plus JSX is the ordinary document shape. This is the case the polytypo-js source comment above its frontmatter skip entry described as a spec gap."
|
|
749
|
+
},
|
|
750
|
+
{
|
|
751
|
+
"id": "fr-nbsp-character-reference-numeric",
|
|
752
|
+
"rule": "nbsp",
|
|
753
|
+
"mode": "text",
|
|
754
|
+
"in": "Bonjour : oui",
|
|
755
|
+
"out": "Bonjour : oui",
|
|
756
|
+
"note": "nbsp.md §3.3 step 4 (spec 1.3.0): the `;` closes ` `, so N2 emits nothing and the reference survives. The `:` takes nothing either — step 1's run guard sees a mark to its left. Before 1.3.0 this produced `Bonjour ` + U+202F + `: oui`, which is no longer a character reference."
|
|
757
|
+
},
|
|
758
|
+
{
|
|
759
|
+
"id": "fr-nbsp-character-reference-named",
|
|
760
|
+
"rule": "nbsp",
|
|
761
|
+
"mode": "text",
|
|
762
|
+
"in": "Tom & Jerry",
|
|
763
|
+
"out": "Tom & Jerry",
|
|
764
|
+
"note": "nbsp.md §3.3 step 4 (spec 1.3.0): the same guard on a named reference. Before 1.3.0 `text` mode destroyed every character reference in French input, and only in locales whose nbsp data puts a space before `;`."
|
|
765
|
+
},
|
|
766
|
+
{
|
|
767
|
+
"id": "fr-nbsp-character-reference-shape-not-table",
|
|
768
|
+
"rule": "nbsp",
|
|
769
|
+
"mode": "text",
|
|
770
|
+
"in": "a ¬aname; b",
|
|
771
|
+
"out": "a ¬aname; b",
|
|
772
|
+
"note": "nbsp.md §3.3 step 4 (spec 1.3.0): the guard tests the shape of a reference, not membership of the HTML named-reference table, which five runtimes would otherwise have to carry identically. Declining here costs nothing."
|
|
773
|
+
},
|
|
774
|
+
{
|
|
775
|
+
"id": "fr-nbsp-ordinary-semicolon-still-binds",
|
|
776
|
+
"rule": "nbsp",
|
|
777
|
+
"mode": "text",
|
|
778
|
+
"in": "Oui ; non",
|
|
779
|
+
"out": "Oui ; non",
|
|
780
|
+
"note": "nbsp.md §3.3 step 4 (spec 1.3.0) is not a blanket refusal of `;`: an ordinary semicolon still takes U+202F under N2. Pinned next to the reference cases so the guard cannot be widened into one."
|
|
781
|
+
},
|
|
782
|
+
{
|
|
783
|
+
"id": "fr-nbsp-semicolon-after-digit-binds",
|
|
784
|
+
"rule": "nbsp",
|
|
785
|
+
"mode": "text",
|
|
786
|
+
"in": "Section 4; suite",
|
|
787
|
+
"out": "Section 4 ; suite",
|
|
788
|
+
"note": "nbsp.md §3.3 step 4 (spec 1.3.0): the left walk finds the digit run but no `&` before it, so the guard does not fire."
|
|
789
|
+
},
|
|
790
|
+
{
|
|
791
|
+
"id": "fr-nbsp-narrow-substituted",
|
|
792
|
+
"rule": "nbsp",
|
|
793
|
+
"mode": "text",
|
|
794
|
+
"narrowNbsp": "nbsp",
|
|
795
|
+
"in": "Un délai ? Vraiment ! Et puis ; voilà.",
|
|
796
|
+
"out": "Un délai ? Vraiment ! Et puis ; voilà.",
|
|
797
|
+
"note": "nbsp.md §3.1a (spec 1.3.0): `narrowNbsp: \"nbsp\"` moves N2's target from U+202F to U+00A0. The same indices are claimed by the same sub-rule under the same guards — only the character written changes. Compare fr-nbsp-narrow-default, which is this input with the option absent."
|
|
798
|
+
},
|
|
799
|
+
{
|
|
800
|
+
"id": "fr-nbsp-narrow-default",
|
|
801
|
+
"rule": "nbsp",
|
|
802
|
+
"mode": "text",
|
|
803
|
+
"in": "Un délai ? Vraiment ! Et puis ; voilà.",
|
|
804
|
+
"out": "Un délai ? Vraiment ! Et puis ; voilà.",
|
|
805
|
+
"note": "nbsp.md §3.1a: the control for fr-nbsp-narrow-substituted. No `narrowNbsp` key means \"narrow\", so every case written before spec 1.3.0 keeps its meaning and a runtime that ignores the option unknowingly still passes this one — which is why the substituted case above is the one that proves the feature."
|
|
806
|
+
},
|
|
807
|
+
{
|
|
808
|
+
"id": "fr-nbsp-narrow-substituted-authored",
|
|
809
|
+
"rule": "nbsp",
|
|
810
|
+
"mode": "text",
|
|
811
|
+
"narrowNbsp": "nbsp",
|
|
812
|
+
"in": "Oui ? Non !",
|
|
813
|
+
"out": "Oui ? Non !",
|
|
814
|
+
"note": "nbsp.md §3.1a: an authored U+202F at an index N2 claims is normalised to the substituted target, exactly as an authored U+00A0 is normalised to U+202F in the default configuration — the rule normalises a claimed index to its target, and the target has moved. This is the case that separates a target rewrite from post-processing: a caller's replaceAll would also produce this output, but would not be a fixed point when its own output is fed back in."
|
|
815
|
+
},
|
|
816
|
+
{
|
|
817
|
+
"id": "fr-nbsp-narrow-substituted-authored-default",
|
|
818
|
+
"rule": "nbsp",
|
|
819
|
+
"mode": "text",
|
|
820
|
+
"in": "Oui ? Non !",
|
|
821
|
+
"out": "Oui ? Non !",
|
|
822
|
+
"note": "nbsp.md §3.1a: the control for the case above — with the default target an authored U+202F is already correct and nothing is emitted."
|
|
823
|
+
},
|
|
824
|
+
{
|
|
825
|
+
"id": "fr-nbsp-narrow-substituted-with-quotes",
|
|
826
|
+
"rule": "nbsp",
|
|
827
|
+
"mode": "text",
|
|
828
|
+
"narrowNbsp": "nbsp",
|
|
829
|
+
"in": "Il a dit : « oui » ; puis ?",
|
|
830
|
+
"out": "Il a dit : « oui » ; puis ?",
|
|
831
|
+
"note": "nbsp.md §3.1a with N1, N2 and N8 all claiming indices in one string. N1 (the colon) and N8 (`fr`'s primary pair, whose innerSpace is \"nbsp\") already wrote U+00A0 and are untouched by the option; N2 (`;` and `?`) moves. With the substitution on, all three want the same character — the option can only ever make N2's target EQUAL to N1's, never different, which is why §5's clause 0 needs no new case."
|
|
832
|
+
},
|
|
833
|
+
{
|
|
834
|
+
"id": "fr-nbsp-narrow-substituted-guards-unchanged",
|
|
835
|
+
"rule": "nbsp",
|
|
836
|
+
"mode": "text",
|
|
837
|
+
"narrowNbsp": "nbsp",
|
|
838
|
+
"in": "12:30 et http://x ; oui",
|
|
839
|
+
"out": "12:30 et http://x ; oui",
|
|
840
|
+
"note": "nbsp.md §3.1a: the option changes what the rule writes, never what it reads or which indices it claims. N1's right-context guard still protects the time and the URL — the colon in `12:30` and in `http://` takes nothing, under the substitution exactly as under the default."
|
|
841
|
+
},
|
|
842
|
+
{
|
|
843
|
+
"id": "fr-nbsp-quote-inner-space-authored-narrow",
|
|
844
|
+
"rule": "nbsp",
|
|
845
|
+
"mode": "text",
|
|
846
|
+
"in": "Il a dit « mot ».",
|
|
847
|
+
"out": "Il a dit « mot ».",
|
|
848
|
+
"note": "nbsp.md §3.10 and §6 rows 3-4, corrected in spec 1.3.0. `fr.json` sets the primary pair's innerSpace to \"nbsp\", so N8's target is U+00A0 and an authored U+202F inside the quotation marks is CONVERTED, not already correct. Through spec 1.2.0 the worked-example table claimed both the narrow space and a fixed point here — a claim about a locale file the file never supported, and nothing pinned it. This case pins it."
|
|
849
|
+
},
|
|
850
|
+
{
|
|
851
|
+
"id": "fr-yaml-narrow-space-in-quoted",
|
|
852
|
+
"rule": "nbsp",
|
|
853
|
+
"mode": "yaml",
|
|
854
|
+
"keys": [
|
|
855
|
+
"description"
|
|
856
|
+
],
|
|
857
|
+
"in": "description: \"Une note !\"\n",
|
|
858
|
+
"out": "description: \"Une note !\"\n",
|
|
859
|
+
"note": "modes.md §3.8.6: a quoted scalar's content is one span, and French spacing applies inside it exactly as in text mode."
|
|
860
|
+
},
|
|
861
|
+
{
|
|
862
|
+
"id": "fr-yaml-no-space-across-colon-split",
|
|
863
|
+
"rule": "nbsp",
|
|
864
|
+
"mode": "yaml",
|
|
865
|
+
"keys": [
|
|
866
|
+
"description"
|
|
867
|
+
],
|
|
868
|
+
"in": "description: prose:more prose\n",
|
|
869
|
+
"out": "description: prose:more prose\n",
|
|
870
|
+
"note": "modes.md §3.8.6: the `:` is outside every span, so the insertion `nbsp` would make before it lands on a span edge and is discarded by §3.4. This is the accepted cost of the split, and it is what keeps the scalar parseable."
|
|
871
|
+
},
|
|
872
|
+
{
|
|
873
|
+
"id": "fr-yaml-guillemets-across-block-lines",
|
|
874
|
+
"rule": "quotes",
|
|
875
|
+
"mode": "yaml",
|
|
876
|
+
"keys": [
|
|
877
|
+
"description"
|
|
878
|
+
],
|
|
879
|
+
"in": "description: |\n Il a dit \"bonjour\n monsieur\" et il est parti\n",
|
|
880
|
+
"out": "description: |\n Il a dit « bonjour\n monsieur » et il est parti\n",
|
|
881
|
+
"note": "modes.md §3.2 model C in a locale whose quotation output is guillemets plus U+00A0 rather than curly quotes — structurally different from the en-US case of the same shape."
|
|
601
882
|
}
|
|
602
883
|
]
|
|
603
884
|
}
|