polytypo 1.3.0 → 1.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (46) hide show
  1. checksums.yaml +4 -4
  2. data/lib/polytypo/data/VERSION +1 -1
  3. data/lib/polytypo/data/fixtures/cs.json +1 -1
  4. data/lib/polytypo/data/fixtures/de-CH.json +1 -1
  5. data/lib/polytypo/data/fixtures/de-DE.json +1 -1
  6. data/lib/polytypo/data/fixtures/el.json +1 -1
  7. data/lib/polytypo/data/fixtures/en-GB.json +44 -1
  8. data/lib/polytypo/data/fixtures/en-US.json +40 -1
  9. data/lib/polytypo/data/fixtures/es.json +1 -1
  10. data/lib/polytypo/data/fixtures/fi.json +1 -1
  11. data/lib/polytypo/data/fixtures/fr-CA.json +9 -1
  12. data/lib/polytypo/data/fixtures/fr.json +50 -1
  13. data/lib/polytypo/data/fixtures/it.json +17 -1
  14. data/lib/polytypo/data/fixtures/locale-resolution.json +1 -1
  15. data/lib/polytypo/data/fixtures/nl.json +17 -1
  16. data/lib/polytypo/data/fixtures/pl.json +1 -1
  17. data/lib/polytypo/data/fixtures/pt-BR.json +9 -1
  18. data/lib/polytypo/data/fixtures/pt-PT.json +9 -1
  19. data/lib/polytypo/data/fixtures/ru.json +1 -1
  20. data/lib/polytypo/data/fixtures/sv.json +1 -1
  21. data/lib/polytypo/data/fixtures/uk.json +1 -1
  22. data/lib/polytypo/data/locales/cs.json +5 -1
  23. data/lib/polytypo/data/locales/de-CH.json +5 -1
  24. data/lib/polytypo/data/locales/de-DE.json +5 -1
  25. data/lib/polytypo/data/locales/el.json +5 -1
  26. data/lib/polytypo/data/locales/en-GB.json +11 -1
  27. data/lib/polytypo/data/locales/en-US.json +11 -1
  28. data/lib/polytypo/data/locales/es.json +5 -1
  29. data/lib/polytypo/data/locales/fi.json +5 -1
  30. data/lib/polytypo/data/locales/fr-CA.json +31 -1
  31. data/lib/polytypo/data/locales/fr.json +31 -1
  32. data/lib/polytypo/data/locales/it.json +11 -1
  33. data/lib/polytypo/data/locales/nl.json +17 -1
  34. data/lib/polytypo/data/locales/pl.json +5 -1
  35. data/lib/polytypo/data/locales/pt-BR.json +11 -1
  36. data/lib/polytypo/data/locales/pt-PT.json +11 -1
  37. data/lib/polytypo/data/locales/registry.json +1 -1
  38. data/lib/polytypo/data/locales/ru.json +5 -1
  39. data/lib/polytypo/data/locales/sv.json +5 -1
  40. data/lib/polytypo/data/locales/uk.json +5 -1
  41. data/lib/polytypo/data/rules/order.json +2 -2
  42. data/lib/polytypo/data/schema/locale.schema.json +23 -2
  43. data/lib/polytypo/engine/rules/quote_ambiguity.rb +68 -0
  44. data/lib/polytypo/engine/rules/quotes.rb +10 -6
  45. data/lib/polytypo/version.rb +1 -1
  46. metadata +1 -1
checksums.yaml CHANGED
@@ -1,7 +1,7 @@
1
1
  ---
2
2
  SHA256:
3
- metadata.gz: 9513225b38e76d5b36886680653ad3d9e859393c9dd04e8002012d63353248ad
4
- data.tar.gz: f2773c497ad977079d67b7c15874a3fe0417d0bbbdca8605228ca7904642ca6c
3
+ metadata.gz: b64ad8e4514e383316d10438c15b6fcf49e4bbeeb9317240bc01164848afc04c
4
+ data.tar.gz: 58d3629dbbfd627ebce163640856bc07e1ace94feb995e3d124f419d560c4ae9
5
5
  SHA512:
6
- metadata.gz: 3cf1ec3e8e150fa736798e680aaf3e62636d122c2a5bd55e1f14ffda00382d4963bf909dd2fab0da9b05c6794b8f3772d391939b15fba37b7ee2cd61404ab0c4
7
- data.tar.gz: 7f9b44747c309f42fd6c0a9262421d06345ea498222218eed5161ae78d3805560716d20d8639b587ad249fc7fece4f96387e9f4aa897a36abd2129aa68b13964
6
+ metadata.gz: 267cba6c58b7e020ee7fbfefccf4cec03eaf6eb7411188014df5cb363a7588f306dc69b9484cb20b3820ed403cc8e217157be3e133ea129bb5652c37e14de87e
7
+ data.tar.gz: 9b0b5365d9165b32652da6cf263ef0dba552d902964b32609684cbf26ab78b862288bda665d7094be8afd6a9b6336842dad9cd38fa15dd8097f9f9e33b024d97
@@ -1 +1 @@
1
- 1.3.0
1
+ 1.4.0
@@ -1,5 +1,5 @@
1
1
  {
2
- "spec": "1.3.0",
2
+ "spec": "1.4.0",
3
3
  "locale": "cs",
4
4
  "cases": [
5
5
  {
@@ -1,5 +1,5 @@
1
1
  {
2
- "spec": "1.3.0",
2
+ "spec": "1.4.0",
3
3
  "locale": "de-CH",
4
4
  "cases": [
5
5
  {
@@ -1,5 +1,5 @@
1
1
  {
2
- "spec": "1.3.0",
2
+ "spec": "1.4.0",
3
3
  "locale": "de-DE",
4
4
  "cases": [
5
5
  {
@@ -1,5 +1,5 @@
1
1
  {
2
- "spec": "1.3.0",
2
+ "spec": "1.4.0",
3
3
  "locale": "el",
4
4
  "cases": [
5
5
  {
@@ -1,5 +1,5 @@
1
1
  {
2
- "spec": "1.3.0",
2
+ "spec": "1.4.0",
3
3
  "locale": "en-GB",
4
4
  "cases": [
5
5
  {
@@ -1296,6 +1296,49 @@
1296
1296
  "ranges": true
1297
1297
  },
1298
1298
  "note": "ranges.md §3.2a (spec 1.3.0): U+00A3, pinned in the locale that actually writes it. The closed-up walk is locale-independent shape logic; `en-GB` shares `en-US`'s `range: \"en-tight\"`."
1299
+ },
1300
+ {
1301
+ "id": "en-gb-markdown-commonmark-span-boundary-possessive-declined",
1302
+ "rule": "quotes",
1303
+ "mode": "markdown",
1304
+ "dialect": "commonmark",
1305
+ "in": "He says 'avoid `x`'s printer.' Done.\n",
1306
+ "out": "He says ‘avoid `x`’s printer.’ Done.\n",
1307
+ "note": "Canonical issue #53's headline case. quotes.md §3.2's span-boundary elision veto (spec 1.4.0): the possessive sits flush against a code span, so modes.md §3.2's marker occupies the position the x would hold and the medial-elision veto cannot fire. The cited quotes.elisionClitics.after entry s declines the mark, apostrophe (order 50) renders it U+2019 by its case 4 across the marker, and the author's own pair keeps the primary glyphs. Through spec 1.3.1 this produced ’avoid `x`‘s printer.’ — the possessive opening the quotation and the real opening mark demoted to a closing glyph — and it was a fixed point, so a second pass never repaired it."
1308
+ },
1309
+ {
1310
+ "id": "en-gb-html-span-boundary-possessive-declined",
1311
+ "rule": "quotes",
1312
+ "mode": "html",
1313
+ "in": "He says 'avoid <code>x</code>'s printer.' Done.",
1314
+ "out": "He says ‘avoid <code>x</code>’s printer.’ Done.",
1315
+ "note": "The same shape in html rather than markdown. modes.md §3.2's marker model is mode-independent, so the two agree on identical prose split the same way (modes.md §7.4)."
1316
+ },
1317
+ {
1318
+ "id": "en-gb-markdown-commonmark-span-boundary-possessive-strong",
1319
+ "rule": "quotes",
1320
+ "mode": "markdown",
1321
+ "dialect": "commonmark",
1322
+ "in": "He says 'avoid **x**'s printer.' Done.\n",
1323
+ "out": "He says ‘avoid **x**’s printer.’ Done.\n",
1324
+ "note": "Strong emphasis behaves as a code span does: the veto reads the marker, not what kind of span produced it."
1325
+ },
1326
+ {
1327
+ "id": "en-gb-markdown-commonmark-span-boundary-plural-possessive-not-closed",
1328
+ "rule": "quotes",
1329
+ "mode": "markdown",
1330
+ "dialect": "commonmark",
1331
+ "in": "He says 'avoid `xs`' printer.' Done.\n",
1332
+ "out": "He says ‘avoid `xs`’ printer.' Done.\n",
1333
+ "note": "Known non-closure, pinned so it stays visible: a PLURAL possessive after a span has the marker on its left and a space on its right, so its attaching run is empty and no entry can match. The same three code points are also a quotation's closing mark after a span, which is a modes.md §3.3 row, so no test over these neighbours could separate them. The second mark takes the pairing and the third is abandoned as U+0027. quotes.md §4 and §7 item 10; tracked as issue #54."
1334
+ },
1335
+ {
1336
+ "id": "en-gb-html-span-boundary-quotation-not-declined",
1337
+ "rule": "quotes",
1338
+ "mode": "html",
1339
+ "in": "'a'<code>x</code>'b'",
1340
+ "out": "‘a’<code>x</code>‘b’",
1341
+ "note": "The negative control, and the reason the veto reads a cited fragment rather than the marker itself. modes.md §3.3 puts the marker in OPENISH precisely so the second pair here can open flush after a span; letting the marker satisfy the medial-elision veto's ALNUM test would have broken this row and en-us-markdown-commonmark-boundary-nested-quotes with it. The run b is not a listed fragment, so the quotation stands."
1299
1342
  }
1300
1343
  ]
1301
1344
  }
@@ -1,5 +1,5 @@
1
1
  {
2
- "spec": "1.3.0",
2
+ "spec": "1.4.0",
3
3
  "locale": "en-US",
4
4
  "cases": [
5
5
  {
@@ -2148,6 +2148,28 @@
2148
2148
  "out": "title: \"Chapter 1: one…two\"\n",
2149
2149
  "note": "modes.md §3.8.6: the plain-scalar colon test does not apply to a quoted scalar — quoting neutralises the colon. A port that applies it declines this line."
2150
2150
  },
2151
+ {
2152
+ "id": "en-us-yaml-plain-colon-declined",
2153
+ "rule": "ellipsis",
2154
+ "mode": "yaml",
2155
+ "keys": [
2156
+ "description"
2157
+ ],
2158
+ "in": "description: She asked: one...two\n",
2159
+ "out": "description: She asked: one...two\n",
2160
+ "note": "modes.md §3.8.6 \"compact nesting is not a value\": a plain scalar containing a colon followed by U+0020 yields no spans, even though its key is listed. The line is not what it looks like — YAML reads it as a nested mapping and a real parser rejects the document outright (\"Nested mappings are not allowed in compact mappings\"), so a port that processes it is rewriting text it cannot read back. The pair with en-us-yaml-block-scalar-takes-the-same-colon is what makes the distinction visible: identical prose, opposite outcome, decided only by the value's form."
2161
+ },
2162
+ {
2163
+ "id": "en-us-yaml-block-scalar-takes-the-same-colon",
2164
+ "rule": "ellipsis",
2165
+ "mode": "yaml",
2166
+ "keys": [
2167
+ "description"
2168
+ ],
2169
+ "in": "description: |\n She asked: one...two\n",
2170
+ "out": "description: |\n She asked: one…two\n",
2171
+ "note": "modes.md §3.8.5: the compact-nesting bail of §3.8.6 belongs to the plain form alone and does not leak into a block scalar, so this line is processed. Paired with en-us-yaml-plain-colon-declined: identical prose, opposite outcome, decided only by the value's form. This case does NOT discriminate whether a port wrongly applies §3.8.6's colon split to block-scalar content — `one...two` is not adjacent to the colon, so such a port emits the same output; §8 item 3a's `-spaced` locale requirement is what covers that."
2172
+ },
2151
2173
  {
2152
2174
  "id": "en-us-yaml-block-literal-lines",
2153
2175
  "rule": "ellipsis",
@@ -2491,6 +2513,23 @@
2491
2513
  "in": "description: one...two\n three...four\n",
2492
2514
  "out": "description: one...two\n three...four\n",
2493
2515
  "note": "modes.md §3.8.6: a multi-line plain scalar's folding is a construct this scan does not claim, so the whole value yields no spans — including its first line."
2516
+ },
2517
+ {
2518
+ "id": "en-us-markdown-commonmark-span-boundary-possessive-declined",
2519
+ "rule": "quotes",
2520
+ "mode": "markdown",
2521
+ "dialect": "commonmark",
2522
+ "in": "He says 'avoid `x`'s printer.' Done.\n",
2523
+ "out": "He says “avoid `x`’s printer.” Done.\n",
2524
+ "note": "Canonical issue #53's headline case in a locale whose primary pair is WIDE, where the inversion was loudest: through spec 1.3.1 the possessive opened a double-quoted quotation, giving ’avoid `x`“s printer.”. quotes.md §3.2's span-boundary elision veto (spec 1.4.0) declines the mark on the cited quotes.elisionClitics.after entry s."
2525
+ },
2526
+ {
2527
+ "id": "en-us-html-span-boundary-listed-fragment-quoted",
2528
+ "rule": "quotes",
2529
+ "mode": "html",
2530
+ "in": "He said <em>'s'</em> loudly.",
2531
+ "out": "He said <em>’s’</em> loudly.",
2532
+ "note": "Accepted false positive, recorded in the suite rather than left in someone's content: a quotation inside an inline span whose content is EXACTLY a listed fragment is read as an elision. Strictly narrower than, and the same family as, the universal medial-n veto's accepted The letter 'n' is common. — it needs the span boundary as well as the single-fragment content. '<em>x</em>', '<em>no</em>' and '<em>fine</em>' in the same position are unaffected. quotes.md §3.2."
2494
2533
  }
2495
2534
  ]
2496
2535
  }
@@ -1,5 +1,5 @@
1
1
  {
2
- "spec": "1.3.0",
2
+ "spec": "1.4.0",
3
3
  "locale": "es",
4
4
  "cases": [
5
5
  {
@@ -1,5 +1,5 @@
1
1
  {
2
- "spec": "1.3.0",
2
+ "spec": "1.4.0",
3
3
  "locale": "fi",
4
4
  "cases": [
5
5
  {
@@ -1,5 +1,5 @@
1
1
  {
2
- "spec": "1.3.0",
2
+ "spec": "1.4.0",
3
3
  "locale": "fr-CA",
4
4
  "cases": [
5
5
  {
@@ -312,6 +312,14 @@
312
312
  "in": "le n° cinq",
313
313
  "out": "le n° cinq",
314
314
  "note": "nbsp.md §3.11 N9 étape 4 : il faut un DIGIT après le séparateur. Un nombre écrit en toutes lettres n'en est pas un, l'espace reste sécable."
315
+ },
316
+ {
317
+ "id": "fr-ca-quotes-span-boundary-elision-declined",
318
+ "rule": "quotes",
319
+ "mode": "html",
320
+ "in": "Il dit 'l'<em>idée</em> est bonne.' Fin.",
321
+ "out": "Il dit « l’<em>idée</em> est bonne. » Fin.",
322
+ "note": "quotes.md §3.2's span-boundary elision veto (spec 1.4.0), on the identical OQLF citation fr carries — the OQLF is Quebec's own authority, so for fr-CA the evidence is direct rather than a substitute. No retrieved source states any France/Quebec divergence on elision, so the two lists are identical."
315
323
  }
316
324
  ]
317
325
  }
@@ -1,5 +1,5 @@
1
1
  {
2
- "spec": "1.3.0",
2
+ "spec": "1.4.0",
3
3
  "locale": "fr",
4
4
  "cases": [
5
5
  {
@@ -879,6 +879,55 @@
879
879
  "in": "description: |\n Il a dit \"bonjour\n monsieur\" et il est parti\n",
880
880
  "out": "description: |\n Il a dit « bonjour\n monsieur » et il est parti\n",
881
881
  "note": "modes.md §3.2 model C in a locale whose quotation output is guillemets plus U+00A0 rather than curly quotes — structurally different from the en-US case of the same shape."
882
+ },
883
+ {
884
+ "id": "fr-quotes-span-boundary-elision-declined",
885
+ "rule": "quotes",
886
+ "mode": "html",
887
+ "in": "Il dit 'l'<em>idée</em> est bonne.' Fin.",
888
+ "out": "Il dit « l’<em>idée</em> est bonne. » Fin.",
889
+ "note": "quotes.md §3.2's span-boundary elision veto (spec 1.4.0): the elided l' stands flush against the emphasis boundary, so modes.md §3.2's marker occupies the position the attaching word would hold and the medial-elision veto cannot fire. The cited quotes.elisionClitics.before entry l declines the mark, apostrophe (order 50) renders it U+2019 by its case 3 across the marker, and the author's own pair keeps the guillemets. Before spec 1.4.0 this produced « l » around the single letter and abandoned the real closing mark as U+0027 (canonical issue #53)."
890
+ },
891
+ {
892
+ "id": "fr-quotes-span-boundary-elision-multi-letter-fragment",
893
+ "rule": "quotes",
894
+ "mode": "html",
895
+ "in": "Il dit 'jusqu'<em>ici</em> tout va bien.' Fin.",
896
+ "out": "Il dit « jusqu’<em>ici</em> tout va bien. » Fin.",
897
+ "note": "Why quotes.elisionClitics carries jusqu and not only qu: the veto compares the MAXIMAL LETTER run, which is jusqu here, so an entry qu could not match this at all. lorsqu, puisqu and quoiqu are listed for the same reason."
898
+ },
899
+ {
900
+ "id": "fr-quotes-span-boundary-elision-caps-not-folded",
901
+ "rule": "quotes",
902
+ "mode": "html",
903
+ "in": "Il dit 'QU'<em>il</em> vienne.' Fin.",
904
+ "out": "Il dit « QU »<em>il</em> vienne.' Fin.",
905
+ "note": "Accepted limitation, pinned rather than left to be rediscovered: the veto folds only the run's FIRST code point, and only ASCII A-Z, so the all-caps QU' does not match the entry qu and the inversion survives. Same leniency, and same limit, as the listed elision veto's left/right words. quotes.md §7 item 10."
906
+ },
907
+ {
908
+ "id": "fr-quotes-span-boundary-quotation-not-declined",
909
+ "rule": "quotes",
910
+ "mode": "html",
911
+ "in": "Il dit <em>'oui'</em> ici.",
912
+ "out": "Il dit <em>« oui »</em> ici.",
913
+ "note": "The negative control the whole mechanism is shaped around. A quotation wholly inside an emphasis span has the marker flush against its opening mark too, exactly as an elision does — what separates them is that the LETTER run oui is not a listed fragment. This is why the marker itself must not be allowed to satisfy the medial-elision veto's ALNUM test, and why the run is compared whole rather than as a prefix."
914
+ },
915
+ {
916
+ "id": "fr-quotes-span-boundary-elision-markdown",
917
+ "rule": "quotes",
918
+ "mode": "markdown",
919
+ "dialect": "commonmark",
920
+ "in": "Il dit 'l'*idée* est bonne.' Fin.\n",
921
+ "out": "Il dit « l’*idée* est bonne. » Fin.\n",
922
+ "note": "The same shape in markdown emphasis rather than an HTML element: modes.md §3.2's marker model is mode-independent, so the two agree on identical characters split the same way (modes.md §7.4)."
923
+ },
924
+ {
925
+ "id": "fr-quotes-span-boundary-listed-fragment-quoted",
926
+ "rule": "quotes",
927
+ "mode": "html",
928
+ "in": "Il dit <em>'l'</em> ici.",
929
+ "out": "Il dit <em>’l’</em> ici.",
930
+ "note": "The accepted false positive in the locale with the widest exposure to it: fr lists thirteen fragments, eight of them a single letter, so a quotation whose whole content is one of them is read as an elision here more readily than anywhere else. Same class as en-us-html-span-boundary-listed-fragment-quoted and as the universal medial-n veto's accepted The letter 'n' is common. — pinned in both locales so the exposure sits in the suite rather than in someone's content. quotes.md §3.2, §6 row S7."
882
931
  }
883
932
  ]
884
933
  }
@@ -1,5 +1,5 @@
1
1
  {
2
- "spec": "1.3.0",
2
+ "spec": "1.4.0",
3
3
  "locale": "it",
4
4
  "cases": [
5
5
  {
@@ -156,6 +156,22 @@
156
156
  "in": "Copyright (c) 2026, formato 40x60 cm",
157
157
  "out": "Copyright © 2026, formato 40×60 cm",
158
158
  "note": "symbols.md: «(c)» diventa U+00A9 e la «x» fra cifre U+00D7. L'U+00A0 davanti a «cm» è opera di nbsp N5, non di symbols: i due casi stanno insieme per mostrare che le regole si compongono."
159
+ },
160
+ {
161
+ "id": "it-quotes-span-boundary-elision-declined",
162
+ "rule": "quotes",
163
+ "mode": "html",
164
+ "in": "Lui dice 'l'<em>idea</em> è buona.' Fine.",
165
+ "out": "Lui dice «l’<em>idea</em> è buona.» Fine.",
166
+ "note": "quotes.md §3.2's span-boundary elision veto (spec 1.4.0), on the Crusca-cited entry l. Italian elision is not a closed set, so this field evidences membership per fragment rather than the completeness of the list — the same discipline quotes.md §3.2 states for each elisionIdioms triple."
167
+ },
168
+ {
169
+ "id": "it-quotes-span-boundary-elision-feminine-article",
170
+ "rule": "quotes",
171
+ "mode": "html",
172
+ "in": "Lui dice 'un'<em>idea</em> è buona.' Fine.",
173
+ "out": "Lui dice «un’<em>idea</em> è buona.» Fine.",
174
+ "note": "The entry un covers the feminine indefinite un' only; Crusca is categorical that the masculine un takes no apostrophe. The mechanism cannot tell the two apart, which is harmless here because it only ever declines a pairing."
159
175
  }
160
176
  ]
161
177
  }
@@ -1,5 +1,5 @@
1
1
  {
2
- "spec": "1.3.0",
2
+ "spec": "1.4.0",
3
3
  "cases": [
4
4
  {
5
5
  "id": "exact-en-us",
@@ -1,5 +1,5 @@
1
1
  {
2
- "spec": "1.3.0",
2
+ "spec": "1.4.0",
3
3
  "locale": "nl",
4
4
  "cases": [
5
5
  {
@@ -116,6 +116,22 @@
116
116
  "in": "Copyright (c) 2026, formaat 40x60 cm",
117
117
  "out": "Copyright © 2026, formaat 40×60 cm",
118
118
  "note": "symbols.md: «(c)» wordt U+00A9 en de «x» tussen cijfers U+00D7. De U+00A0 vóór «cm» is het werk van nbsp N5."
119
+ },
120
+ {
121
+ "id": "nl-quotes-span-boundary-elision-declined",
122
+ "rule": "quotes",
123
+ "mode": "html",
124
+ "in": "Hij zegt 'ik kom <em>vroeg </em>'s avonds terug.' Klaar.",
125
+ "out": "Hij zegt “ik kom <em>vroeg </em>’s avonds terug.” Klaar.",
126
+ "note": "quotes.md §3.2's span-boundary elision veto (spec 1.4.0), read from quotes.elisionClitics.after rather than before: 's is a word-INITIAL omission (des avonds), so the letter run sits to the mark's right. Both lists are positional (quotes.md §2) — the fragment attaches forward to what follows, which is exactly why an attachment-based reading of the field would have missed it."
127
+ },
128
+ {
129
+ "id": "nl-quotes-span-boundary-elision-t",
130
+ "rule": "quotes",
131
+ "mode": "html",
132
+ "in": "Hij zegt 'ik kom <em>vroeg </em>'t is laat.' Klaar.",
133
+ "out": "Hij zegt “ik kom <em>vroeg </em>’t is laat.” Klaar.",
134
+ "note": "The same mechanism on the entry t ('t was), attested by the same Taalunie pair of pages."
119
135
  }
120
136
  ]
121
137
  }
@@ -1,5 +1,5 @@
1
1
  {
2
- "spec": "1.3.0",
2
+ "spec": "1.4.0",
3
3
  "locale": "pl",
4
4
  "cases": [
5
5
  {
@@ -1,5 +1,5 @@
1
1
  {
2
- "spec": "1.3.0",
2
+ "spec": "1.4.0",
3
3
  "locale": "pt-BR",
4
4
  "cases": [
5
5
  {
@@ -151,6 +151,14 @@
151
151
  "in": "pinga d'água",
152
152
  "out": "pinga d’água",
153
153
  "note": "U+0027 passa a U+2019. A regra não lê dados de locale; o caso certifica-a para esta locale."
154
+ },
155
+ {
156
+ "id": "pt-br-quotes-span-boundary-elision-declined",
157
+ "rule": "quotes",
158
+ "mode": "html",
159
+ "in": "Ele diz 'd'<em>Os Lusíadas</em> é bom.' Fim.",
160
+ "out": "Ele diz “d’<em>Os Lusíadas</em> é bom.” Fim.",
161
+ "note": "Same input and same Base XVIII citation as pt-PT, different primary pair — pt-BR opens with U+201C where pt-PT opens with a guillemet. Pinned in both files so a runtime cannot pass one and fail the other."
154
162
  }
155
163
  ]
156
164
  }
@@ -1,5 +1,5 @@
1
1
  {
2
- "spec": "1.3.0",
2
+ "spec": "1.4.0",
3
3
  "locale": "pt-PT",
4
4
  "cases": [
5
5
  {
@@ -151,6 +151,14 @@
151
151
  "in": "pinga d'água",
152
152
  "out": "pinga d’água",
153
153
  "note": "U+0027 passa a U+2019. A regra não lê dados de locale; o caso certifica-a para esta locale."
154
+ },
155
+ {
156
+ "id": "pt-pt-quotes-span-boundary-elision-declined",
157
+ "rule": "quotes",
158
+ "mode": "html",
159
+ "in": "Ele diz 'd'<em>Os Lusíadas</em> é bom.' Fim.",
160
+ "out": "Ele diz «d’<em>Os Lusíadas</em> é bom.» Fim.",
161
+ "note": "quotes.md §3.2's span-boundary elision veto (spec 1.4.0), on Acordo Ortográfico Base XVIII 1.º a) — whose own examples are book titles, which is what makes this list load-bearing: a title set in emphasis rather than italics puts the span boundary immediately after the mark."
154
162
  }
155
163
  ]
156
164
  }
@@ -1,5 +1,5 @@
1
1
  {
2
- "spec": "1.3.0",
2
+ "spec": "1.4.0",
3
3
  "locale": "ru",
4
4
  "cases": [
5
5
  {
@@ -1,5 +1,5 @@
1
1
  {
2
- "spec": "1.3.0",
2
+ "spec": "1.4.0",
3
3
  "locale": "sv",
4
4
  "cases": [
5
5
  {
@@ -1,5 +1,5 @@
1
1
  {
2
- "spec": "1.3.0",
2
+ "spec": "1.4.0",
3
3
  "locale": "uk",
4
4
  "cases": [
5
5
  {
@@ -12,7 +12,11 @@
12
12
  "close": "‘",
13
13
  "innerSpace": "none"
14
14
  },
15
- "elisionIdioms": []
15
+ "elisionIdioms": [],
16
+ "elisionClitics": {
17
+ "before": [],
18
+ "after": []
19
+ }
16
20
  },
17
21
  "dash": {
18
22
  "parenthetical": "en-spaced",
@@ -12,7 +12,11 @@
12
12
  "close": "›",
13
13
  "innerSpace": "none"
14
14
  },
15
- "elisionIdioms": []
15
+ "elisionIdioms": [],
16
+ "elisionClitics": {
17
+ "before": [],
18
+ "after": []
19
+ }
16
20
  },
17
21
  "dash": {
18
22
  "parenthetical": "en-spaced",
@@ -12,7 +12,11 @@
12
12
  "close": "‘",
13
13
  "innerSpace": "none"
14
14
  },
15
- "elisionIdioms": []
15
+ "elisionIdioms": [],
16
+ "elisionClitics": {
17
+ "before": [],
18
+ "after": []
19
+ }
16
20
  },
17
21
  "dash": {
18
22
  "parenthetical": "en-spaced",
@@ -12,7 +12,11 @@
12
12
  "close": "”",
13
13
  "innerSpace": "none"
14
14
  },
15
- "elisionIdioms": []
15
+ "elisionIdioms": [],
16
+ "elisionClitics": {
17
+ "before": [],
18
+ "after": []
19
+ }
16
20
  },
17
21
  "dash": {
18
22
  "parenthetical": "none",
@@ -12,7 +12,11 @@
12
12
  "close": "”",
13
13
  "innerSpace": "none"
14
14
  },
15
- "elisionIdioms": []
15
+ "elisionIdioms": [],
16
+ "elisionClitics": {
17
+ "before": [],
18
+ "after": ["s"]
19
+ }
16
20
  },
17
21
  "dash": {
18
22
  "parenthetical": "en-spaced",
@@ -110,6 +114,12 @@
110
114
  "rule": "hyphen",
111
115
  "cite": "No English authority prescribes non-breaking hyphens for particular word forms; lists deliberately empty",
112
116
  "note": "`prefixes`, `suffixes` and `compounds` are empty. The field exists for languages with a closed, normative set of hyphenated morphological forms (Russian кое-, -таки, из-под). The Oxford guide’s hyphen section describes when to hyphenate (adjectival phrases before a noun, verb participles) and never restricts breaking at the resulting hyphen. An English list here would be an invented sample."
117
+ },
118
+ {
119
+ "rule": "quotes",
120
+ "cite": "The Chicago Manual of Style, 7.16 “Possessive form of most nouns” (section number and title confirmed against the publisher's own chapter-7 table of contents; cited as 7.16–19 and 7.22 by the CMOS Q&A on the 18th-edition site). Verbatim, as reproduced by CMOS Shop Talk, “Section 7.16 in the Spotlight”, 12 May 2015: “The possessive of most singular nouns is formed by adding an apostrophe and an s.” Reconfirmed for the current edition by CMOS Shop Talk, 31 August 2021: “Chicago adds an apostrophe and an s to form the possessive of a singular noun, including singular nouns ending in s” (see CMOS 7.16)",
121
+ "url": "https://cmosshoptalk.com/2021/08/31/introducing-the-chicago-manual-of-style-for-perfectit/",
122
+ "note": "Supports quotes.elisionClitics.after = [\"s\"], identical to en-US. LABELLED GAP: en-GB rests on an American authority because no British one was reachable. New Hart's Rules: The Oxford Style Guide (ed. Anne Waddingham, OUP) is print and paywalled; ox.ac.uk's style guide returned HTTP 403, OUP's own free companion site for New Hart's Rules returned an empty body, and Oxford Reference is paywalled. Every British statement of the rule found was a third-party paraphrase and therefore not citable. The gap is narrow rather than papered over: the single point on which US and UK possessive practice actually diverges is whether a second s follows a sibilant (Dickens's, Chicago's preference per 7.16–19, versus Dickens', the alternative practice per 7.22), and that divergence cannot reach this entry, because the veto matches the maximal LETTER run to the RIGHT of the mark — Burns's fires on s, Burns' has no letter there and fires nothing. So the two locales take the identical entry not by assumption but because the two practices coincide wherever this field is consulted. Replace with New Hart's Rules if the book is obtained. Note also that en-GB's primary close is U+2019, so this is the locale where a wrong entry could collide with genuine quotation — a one-element list is the minimum that closes issue #53."
113
123
  }
114
124
  ]
115
125
  }
@@ -18,7 +18,11 @@
18
18
  "elided": "n",
19
19
  "right": "roll"
20
20
  }
21
- ]
21
+ ],
22
+ "elisionClitics": {
23
+ "before": [],
24
+ "after": ["s"]
25
+ }
22
26
  },
23
27
  "dash": {
24
28
  "parenthetical": "em-tight",
@@ -128,6 +132,12 @@
128
132
  "cite": "The Chicago Manual of Style Online, Q&A, topic “Word Division”: “we don’t include any recommendation that says you must carry an abbreviation that ends in a period over to the next line when the sentence continues beyond the abbreviation. So a line is allowed to end with an ‘a.m.’ or ‘Jr.’ or the like that occurs in the middle of a sentence.”",
129
133
  "url": "https://www.chicagomanualofstyle.org/qanda/data/faq/topics/WordDivision/faq0007.html",
130
134
  "note": "`beforeNumber` and `beforeWord` are empty, and Chicago does not merely omit reference abbreviations from its nonbreaking-space guidance — it states positively that no such rule exists, so `No. 5`, `p. 12` and `pp. 12–14` are refused on the source’s own words rather than on an argument from silence (issue #30). The same answer attributes the practice to typesetters rather than to Chicago: “some typesetters will use a nonbreaking space to prevent a compound like ‘St. Louis’ or ‘Dr. Smith’ from breaking at the end of a line” — which is why `beforeWord` stays empty too; the named exception is initials in a name, carried by `initialBinding`. CMOS Shop Talk, “Navigating Spaces in Manuscripts and Beyond” (15 June 2021), cited above for `initialBinding` and `beforeUnits`, introduces its four contexts with “can be helpful in the following contexts, among others”, so it must not be described as an exhaustive list."
135
+ },
136
+ {
137
+ "rule": "quotes",
138
+ "cite": "The Chicago Manual of Style, 7.16 “Possessive form of most nouns” (section number and title confirmed against the publisher's own chapter-7 table of contents; cited as 7.16–19 and 7.22 by the CMOS Q&A on the 18th-edition site). Verbatim, as reproduced by CMOS Shop Talk, “Section 7.16 in the Spotlight”, 12 May 2015: “The possessive of most singular nouns is formed by adding an apostrophe and an s.” Reconfirmed for the current edition by CMOS Shop Talk, 31 August 2021: “Chicago adds an apostrophe and an s to form the possessive of a singular noun, including singular nouns ending in s” (see CMOS 7.16)",
139
+ "url": "https://cmosshoptalk.com/2021/08/31/introducing-the-chicago-manual-of-style-for-perfectit/",
140
+ "note": "Supports the single entry quotes.elisionClitics.after = [\"s\"] — the possessive s, not the contraction enclitics. t, ll, re, ve, d and m are deliberately absent: nobody writes `won`'t, so they have no attested occurrence in the shape this veto is scoped to, whereas possessivising a set-off identifier (`x`'s printer) is everyday technical prose and is the shape reported in canonical issue #53. Chicago states the rule, not a code point; U+2019 follows from apostrophe.md §3.3, which is unchanged. Edition and section provenance is mixed deliberately and should not be tidied into a single number: the numbered paragraph itself is paywalled, the verbatim wording above is the 16th edition's (the 2015 post carries an editor's note saying so), and the 2021 post plus the 18th-edition Q&A are what establish that 7.16 is still the general rule. CMOS 7.22's alternative practice (Burns' rather than Burns's) does not affect this entry: the veto matches the maximal LETTER run to the RIGHT of the mark, so Burns's fires on s while Burns' has no letter there and fires nothing. CMOS 7.29 “Possessive with italicized or quoted terms” is a closer fit to the reported shape than 7.16 and would strengthen this citation, but its text is paywalled and was not read — only its title, from the table of contents."
131
141
  }
132
142
  ]
133
143
  }
@@ -12,7 +12,11 @@
12
12
  "close": "”",
13
13
  "innerSpace": "none"
14
14
  },
15
- "elisionIdioms": []
15
+ "elisionIdioms": [],
16
+ "elisionClitics": {
17
+ "before": [],
18
+ "after": []
19
+ }
16
20
  },
17
21
  "dash": {
18
22
  "parenthetical": "none",
@@ -12,7 +12,11 @@
12
12
  "close": "’",
13
13
  "innerSpace": "none"
14
14
  },
15
- "elisionIdioms": []
15
+ "elisionIdioms": [],
16
+ "elisionClitics": {
17
+ "before": [],
18
+ "after": []
19
+ }
16
20
  },
17
21
  "dash": {
18
22
  "parenthetical": "en-spaced",
@@ -12,7 +12,25 @@
12
12
  "close": "”",
13
13
  "innerSpace": "none"
14
14
  },
15
- "elisionIdioms": []
15
+ "elisionIdioms": [],
16
+ "elisionClitics": {
17
+ "before": [
18
+ "l",
19
+ "j",
20
+ "m",
21
+ "t",
22
+ "s",
23
+ "c",
24
+ "qu",
25
+ "jusqu",
26
+ "d",
27
+ "n",
28
+ "lorsqu",
29
+ "puisqu",
30
+ "quoiqu"
31
+ ],
32
+ "after": []
33
+ }
16
34
  },
17
35
  "dash": {
18
36
  "parenthetical": "em-spaced",
@@ -79,6 +97,18 @@
79
97
  "cite": "Office québécois de la langue française, Banque de dépannage linguistique, « Abréviation de numéro » : « L'abréviation courante du nom numéro est no ou No, avec la lettre o minuscule en exposant » ; « On n'abrège le mot numéro que s'il suit immédiatement le nom qu'il détermine » ; exemple : « C'est le billet no 852410 qui a valu le gros lot à sa détentrice »",
80
98
  "url": "https://vitrinelinguistique.oqlf.gouv.qc.ca/25450/les-abreviations-et-les-symboles/les-abreviations/cas-particuliers-dabreviations/abreviation-de-numero",
81
99
  "note": "Deux choses distinctes, et il faut les tenir séparées. Ce que la source établit : « n° » est une abréviation, pas un symbole, d'où beforeNumber et non afterSymbols — l'argument est aussi mécanique, un « ° » seul placé dans afterSymbols serait inerte puisque N6 exige que le code point précédent ne soit pas ALNUM, or c'est la lettre « n » (nbsp.md §3.8). Ce que la source n'établit pas : la séquence exacte de code points. L'OQLF prescrit un o minuscule en exposant, un tracé et non un code point, et ne nomme nulle part le signe de degré ni la ligature « numéro » (U+2116). Pour la France, rien : le § 5.1.3 de Jacques André, relu intégralement, ne mentionne ni « numéro » ni « no », et le Lexique de l'Imprimerie nationale est un ouvrage imprimé qui n'a pas pu être consulté. Décision de l'opérateur (18.09.2026) : quand la règle est derrière une source payante, polytypo retient l'usage le plus répandu et l'applique de la même façon partout, l'uniformité primant la conformité au canon, et la décision est signalée comme telle au lieu d'être présentée comme une citation. L'usage le plus répandu est « n » + U+00B0, c'est ce que l'on tape et ce que l'on trouve dans les textes en ligne ; N9 n'ayant aucune tolérance de casse, « n° » et « N° » sont deux entrées. La phrase « on ne sépare pas les numéros par des espacements » de la même page vise les tranches de chiffres à l'intérieur du nombre (« billet no 852410 »), pas l'espace entre l'abréviation et le nombre."
100
+ },
101
+ {
102
+ "rule": "quotes",
103
+ "cite": "Office québécois de la langue française, Banque de dépannage linguistique, « Cas où l'élision est obligatoire » (dernière mise à jour : 2015) : « L'élision ne touche que des mots grammaticaux, habituellement courts, et que les voyelles a, e et i. » L'élision est obligatoire pour le déterminant et le pronom la et le, pour les pronoms je, me, te, se et ce placés devant le verbe, pour si devant il(s), pour que et les conjonctions qui le contiennent, pour jusque, pour la préposition de et pour l'adverbe de négation ne",
104
+ "url": "https://vitrinelinguistique.oqlf.gouv.qc.ca/21737/lorthographe/elision-et-apostrophe/elision-obligatoire",
105
+ "note": "Justifie dix des treize entrées de quotes.elisionClitics.before : l (la, le), j (je), m (me), t (te), s (se et si), c (ce), qu (que), jusqu (jusque), d (de), n (ne). Chaque entrée est la plage de lettres maximale à gauche de la marque, d'où jusqu et non qu pour « jusqu'à » : sous une comparaison par plage maximale, une entrée qu ne peut pas apparier « jusqu' ». presqu (presqu'île) et quelqu (quelqu'un) sont volontairement exclus, faute d'attestation dans les articles relus, et l'exclusion est sans coût : dans les deux mots la marque a une lettre de chaque côté. La liste after est vide — la BDL définit l'élision comme l'effacement de la voyelle finale d'un mot, donc la marque suit toujours le mot élidé, et l'enclise française prend le trait d'union (donne-moi, y a-t-il), jamais l'apostrophe. Le cas porté par cette liste est l'élision devant un élément en ligne, l'<em>idée</em> : sans elle, quotes appariait la marque comme une citation et invertissait la paire englobante (issue #53). L'OQLF est l'autorité du Québec : pour fr-CA la citation est directe, pour fr elle tient lieu de substitut, le Lexique de l'Imprimerie nationale et Le Bon Usage n'étant pas consultables en ligne — même lacune que celle déjà signalée ailleurs dans ce fichier. quotes.md §3.2's span-boundary elision veto (spec 1.4.0) reads this list only when the mark's other literal neighbour is modes.md §3.2's inline span boundary, so every entry is matched against the maximal LETTER run on one side of the mark and nothing else. The list is decline-only: the worst it can do is refuse a pairing quotes would otherwise have formed."
106
+ },
107
+ {
108
+ "rule": "quotes",
109
+ "cite": "Office québécois de la langue française, Banque de dépannage linguistique, « Élision de lorsque, puisque et quoique » (2020) : « les conjonctions lorsque, puisque et quoique s'élident obligatoirement devant il(s), elle(s), on, un et une, et aussi devant la préposition en » ; l'élision généralisée devant tout mot commençant par une voyelle ou un h muet « demeure facultative »",
110
+ "url": "https://vitrinelinguistique.oqlf.gouv.qc.ca/23623/lorthographe/elision-et-apostrophe/elision-de-lorsque-puisque-et-quoique",
111
+ "note": "Justifie les trois entrées lorsqu, puisqu, quoiqu. L'élision obligatoire devant il(s), elle(s), on, un, une et en suffit à l'appartenance : la liste étant uniquement restrictive, le caractère facultatif de l'élision généralisée ne change rien. Là encore la plage maximale impose des entrées distinctes : qu n'apparie pas « lorsqu' ». quotes.md §3.2's span-boundary elision veto (spec 1.4.0) reads this list only when the mark's other literal neighbour is modes.md §3.2's inline span boundary, so every entry is matched against the maximal LETTER run on one side of the mark and nothing else. The list is decline-only: the worst it can do is refuse a pairing quotes would otherwise have formed."
82
112
  }
83
113
  ]
84
114
  }
@@ -12,7 +12,25 @@
12
12
  "close": "”",
13
13
  "innerSpace": "none"
14
14
  },
15
- "elisionIdioms": []
15
+ "elisionIdioms": [],
16
+ "elisionClitics": {
17
+ "before": [
18
+ "l",
19
+ "j",
20
+ "m",
21
+ "t",
22
+ "s",
23
+ "c",
24
+ "qu",
25
+ "jusqu",
26
+ "d",
27
+ "n",
28
+ "lorsqu",
29
+ "puisqu",
30
+ "quoiqu"
31
+ ],
32
+ "after": []
33
+ }
16
34
  },
17
35
  "dash": {
18
36
  "parenthetical": "em-spaced",
@@ -85,6 +103,18 @@
85
103
  "cite": "Office québécois de la langue française, Banque de dépannage linguistique, « Abréviation de numéro » : « L'abréviation courante du nom numéro est no ou No, avec la lettre o minuscule en exposant » ; « On n'abrège le mot numéro que s'il suit immédiatement le nom qu'il détermine » ; exemple : « C'est le billet no 852410 qui a valu le gros lot à sa détentrice »",
86
104
  "url": "https://vitrinelinguistique.oqlf.gouv.qc.ca/25450/les-abreviations-et-les-symboles/les-abreviations/cas-particuliers-dabreviations/abreviation-de-numero",
87
105
  "note": "Deux choses distinctes, et il faut les tenir séparées. Ce que la source établit : « n° » est une abréviation, pas un symbole, d'où beforeNumber et non afterSymbols — l'argument est aussi mécanique, un « ° » seul placé dans afterSymbols serait inerte puisque N6 exige que le code point précédent ne soit pas ALNUM, or c'est la lettre « n » (nbsp.md §3.8). Ce que la source n'établit pas : la séquence exacte de code points. L'OQLF prescrit un o minuscule en exposant, un tracé et non un code point, et ne nomme nulle part le signe de degré ni la ligature « numéro » (U+2116). Pour la France, rien : le § 5.1.3 de Jacques André, relu intégralement, ne mentionne ni « numéro » ni « no », et le Lexique de l'Imprimerie nationale est un ouvrage imprimé qui n'a pas pu être consulté. Décision de l'opérateur (18.09.2026) : quand la règle est derrière une source payante, polytypo retient l'usage le plus répandu et l'applique de la même façon partout, l'uniformité primant la conformité au canon, et la décision est signalée comme telle au lieu d'être présentée comme une citation. L'usage le plus répandu est « n » + U+00B0, c'est ce que l'on tape et ce que l'on trouve dans les textes en ligne ; N9 n'ayant aucune tolérance de casse, « n° » et « N° » sont deux entrées. La phrase « on ne sépare pas les numéros par des espacements » de la même page vise les tranches de chiffres à l'intérieur du nombre (« billet no 852410 »), pas l'espace entre l'abréviation et le nombre."
106
+ },
107
+ {
108
+ "rule": "quotes",
109
+ "cite": "Office québécois de la langue française, Banque de dépannage linguistique, « Cas où l'élision est obligatoire » (dernière mise à jour : 2015) : « L'élision ne touche que des mots grammaticaux, habituellement courts, et que les voyelles a, e et i. » L'élision est obligatoire pour le déterminant et le pronom la et le, pour les pronoms je, me, te, se et ce placés devant le verbe, pour si devant il(s), pour que et les conjonctions qui le contiennent, pour jusque, pour la préposition de et pour l'adverbe de négation ne",
110
+ "url": "https://vitrinelinguistique.oqlf.gouv.qc.ca/21737/lorthographe/elision-et-apostrophe/elision-obligatoire",
111
+ "note": "Justifie dix des treize entrées de quotes.elisionClitics.before : l (la, le), j (je), m (me), t (te), s (se et si), c (ce), qu (que), jusqu (jusque), d (de), n (ne). Chaque entrée est la plage de lettres maximale à gauche de la marque, d'où jusqu et non qu pour « jusqu'à » : sous une comparaison par plage maximale, une entrée qu ne peut pas apparier « jusqu' ». presqu (presqu'île) et quelqu (quelqu'un) sont volontairement exclus, faute d'attestation dans les articles relus, et l'exclusion est sans coût : dans les deux mots la marque a une lettre de chaque côté. La liste after est vide — la BDL définit l'élision comme l'effacement de la voyelle finale d'un mot, donc la marque suit toujours le mot élidé, et l'enclise française prend le trait d'union (donne-moi, y a-t-il), jamais l'apostrophe. Le cas porté par cette liste est l'élision devant un élément en ligne, l'<em>idée</em> : sans elle, quotes appariait la marque comme une citation et invertissait la paire englobante (issue #53). L'OQLF est l'autorité du Québec : pour fr-CA la citation est directe, pour fr elle tient lieu de substitut, le Lexique de l'Imprimerie nationale et Le Bon Usage n'étant pas consultables en ligne — même lacune que celle déjà signalée ailleurs dans ce fichier. quotes.md §3.2's span-boundary elision veto (spec 1.4.0) reads this list only when the mark's other literal neighbour is modes.md §3.2's inline span boundary, so every entry is matched against the maximal LETTER run on one side of the mark and nothing else. The list is decline-only: the worst it can do is refuse a pairing quotes would otherwise have formed."
112
+ },
113
+ {
114
+ "rule": "quotes",
115
+ "cite": "Office québécois de la langue française, Banque de dépannage linguistique, « Élision de lorsque, puisque et quoique » (2020) : « les conjonctions lorsque, puisque et quoique s'élident obligatoirement devant il(s), elle(s), on, un et une, et aussi devant la préposition en » ; l'élision généralisée devant tout mot commençant par une voyelle ou un h muet « demeure facultative »",
116
+ "url": "https://vitrinelinguistique.oqlf.gouv.qc.ca/23623/lorthographe/elision-et-apostrophe/elision-de-lorsque-puisque-et-quoique",
117
+ "note": "Justifie les trois entrées lorsqu, puisqu, quoiqu. L'élision obligatoire devant il(s), elle(s), on, un, une et en suffit à l'appartenance : la liste étant uniquement restrictive, le caractère facultatif de l'élision généralisée ne change rien. Là encore la plage maximale impose des entrées distinctes : qu n'apparie pas « lorsqu' ». quotes.md §3.2's span-boundary elision veto (spec 1.4.0) reads this list only when the mark's other literal neighbour is modes.md §3.2's inline span boundary, so every entry is matched against the maximal LETTER run on one side of the mark and nothing else. The list is decline-only: the worst it can do is refuse a pairing quotes would otherwise have formed."
88
118
  }
89
119
  ]
90
120
  }
@@ -12,7 +12,11 @@
12
12
  "close": "”",
13
13
  "innerSpace": "none"
14
14
  },
15
- "elisionIdioms": []
15
+ "elisionIdioms": [],
16
+ "elisionClitics": {
17
+ "before": ["l", "un", "d"],
18
+ "after": []
19
+ }
16
20
  },
17
21
  "dash": {
18
22
  "parenthetical": "em-spaced",
@@ -90,6 +94,12 @@
90
94
  "rule": "nbsp",
91
95
  "cite": "Nessuna fonte normativa italiana esaminata prescrive uno spazio prima di un segno d'interpunzione, né un legame per abbreviazioni con spazio interno, simboli o parole seguenti; liste deliberatamente vuote",
92
96
  "note": "beforePunctuation e narrowBeforePunctuation sono vuote, e va detto con esattezza su che cosa poggiano: sull'ASSENZA di una fonte italiana che richieda quello spazio, NON su una fonte che lo neghi. A differenza del capitolo greco, che respinge la prassi francese nominandola, la Parte quarta italiana non la nomina mai: i punti 10.1.3 e 10.1.4 descrivono la funzione dei segni e tacciono sulla spaziatura. Il vincolo Q-P rende comunque la lista vuota la scelta strutturalmente sicura. afterShortWords, abbreviations, beforeWord e afterSymbols sono vuote per la stessa ragione: nessuna fonte esaminata le attesta per l'italiano. SEGNALAZIONE ALL'OPERATORE, non risolta qui: la Parte quarta si apre, al punto 10.1, con «Per la spaziatura dei segni d'interpunzione, cfr. punto 6.4», cioè il capitolo squalificato come prassi editoriale in el.json; se quel rinvio si leggesse come un'adozione, la tabella diventerebbe evidenza italiana e rafforzerebbe queste liste vuote. Consultati il 18.09.2026."
97
+ },
98
+ {
99
+ "rule": "quotes",
100
+ "cite": "Accademia della Crusca, Consulenza linguistica : Vera Gheno, « L'articolo indeterminativo » (30 settembre 2002), sulla forma femminile « un' (nei casi in cui si userebbe l'articolo determinativo l', ma l'elisione non è obbligatoria) [...] va usata davanti a parole che iniziano per vocale » ; Raffaella Setti, « Elisione e troncamento nell'italiano contemporaneo » (14 marzo 2008) : con di l'elisione è spesso applicata e in alcuni casi obbligatoria (d'accordo, d'oro), mentre con mi, ti, la, vi, si « l'apostrofo è del tutto facoltativo » e « non c'è nessun obbligo di elidere, ma è opportuno valutare caso per caso »",
101
+ "url": "https://accademiadellacrusca.it/it/consulenza/elisione-e-troncamento-nellitaliano-contemporaneo/174",
102
+ "note": "Italian elision is not a closed set — the Accademia states in terms that it is optional and to be judged case by case — so this entry evidences membership per fragment, exactly as quotes.md §3.2 requires of each elisionIdioms triple, and not the completeness of the list. Three fragments are named literally and are all that is listed: l (definite article l'), un (feminine indefinite un' — Crusca is categorical that masculine un takes no apostrophe, and the mechanism cannot tell the two apart, which is harmless because it only ever declines a pairing), and d (di, with d'accordo and d'oro given as obligatory). dell, nell, all, sull and dall are deliberately absent, being unattested by the consulenze read here; under maximal-run matching an l entry does not cover dell'arte either way, since the run there is dell. after is empty: Italian has no enclitic written with an apostrophe. quotes.md §3.2's span-boundary elision veto (spec 1.4.0) reads this list only when the mark's other literal neighbour is modes.md §3.2's inline span boundary, so every entry is matched against the maximal LETTER run on one side of the mark and nothing else. The list is decline-only: the worst it can do is refuse a pairing quotes would otherwise have formed."
93
103
  }
94
104
  ]
95
105
  }
@@ -12,7 +12,11 @@
12
12
  "close": "’",
13
13
  "innerSpace": "none"
14
14
  },
15
- "elisionIdioms": []
15
+ "elisionIdioms": [],
16
+ "elisionClitics": {
17
+ "before": [],
18
+ "after": ["s", "t", "ns", "k", "m", "em", "r", "et"]
19
+ }
16
20
  },
17
21
  "dash": {
18
22
  "parenthetical": "en-spaced",
@@ -79,6 +83,18 @@
79
83
  "cite": "Nederlandse Taalunie, Taaladvies.net, «Prof.( )dr.» (12 mei 2021): «Ja, tussen bijvoorbeeld prof. dr. komt een spatie»; «Alleen als de afgekorte woorden tezamen een afkorting vormen, vervalt de spatie: a.s., a.u.b., m.b.v., n.a.v., z.o.z.»; «In verzorgd zetwerk wordt tussen voorletters en in afkortingen als v.d. bij voorkeur een halve spatie (of dunspatie) geplaatst»",
80
84
  "url": "http://taaladvies.net/legacylink.php?id=vraag/678",
81
85
  "note": "Onderbouwt abbreviations = [«prof. dr.»] — de enige Nederlandse afkorting met een interne U+0020 die letterlijk is aangetroffen. De gesloten vormen (a.s., a.u.b.) hebben geen interne spatie, zodat N4 daar niets te bevorderen heeft; dáárom telt de lijst één ingang en is hij niet leeg. initialBinding = «none»: de enige Nederlandse uitspraak hierover vraagt om een DUNNE spatie, een waarde die het schema niet kent terwijl N7 U+00A0 invoegt — en, belangrijker, geen geraadpleegde bron zegt dat een voorletter niet door een regelovergang van de achternaam gescheiden mag worden, en dát is wat N7 werkelijk doet. Geraadpleegd op 18.09.2026."
86
+ },
87
+ {
88
+ "rule": "quotes",
89
+ "cite": "Nederlandse Taalunie, Taaladvies.net, « Wat-ie wil / wat ie wil » (12 mei 2021) : « Gereduceerde vormen van persoonlijke voornaamwoorden komen voor in de gesproken taal. Bij de weergave daarvan in schrift gebruiken we bij vormen als 'k, '(e)m, 'r, d'r en '(e)t een weglatingsteken of apostrof. »",
90
+ "url": "https://taaladvies.net/wat-ie-wil/",
91
+ "note": "Attests five entries of quotes.elisionClitics.after: k ('k), m and em ('(e)m — the Taalunie's own notation covers both 'm and 'em), r ('r), and t with et ('(e)t). These are word-initial omissions: the mark stands at the start of the reduced form, so the letter run is on the mark's right, which is why they belong in after and not before even though the fragment attaches forward rather than back — both lists are positional, per quotes.md §2. before is empty: m'n, z'n, d'r, zo'n and foto's all have a letter on each side of the mark, so the medial-elision veto already declines them and any before entry would be evidenced-inert; keeping it empty also avoids having to adjudicate m, which occurs on both sides of the mark across the language ('m = hem, m'n = mijn). 'n (een) is not listed — it was not attested by a qualifying Taalunie page in this pass. quotes.md §3.2's span-boundary elision veto (spec 1.4.0) reads this list only when the mark's other literal neighbour is modes.md §3.2's inline span boundary, so every entry is matched against the maximal LETTER run on one side of the mark and nothing else. The list is decline-only: the worst it can do is refuse a pairing quotes would otherwise have formed."
92
+ },
93
+ {
94
+ "rule": "quotes",
95
+ "cite": "Nederlandse Taalunie, Taaladvies.net, « 'S avonds / 's Avonds (hoofdletter?) » (12 mei 2021, op grond van de Woordenlijst 2015) : « Als de zin met een apostrof begint, krijgt het eerstvolgende volledige woord in de zin de hoofdletter », met de voorbeelden « 's Avonds werkt John in een café », « 't Was een mooie dag » en « 'ns Kijken hoe we dat doen »",
96
+ "url": "https://taaladvies.net/s-avonds-of-s-avonds-hoofdletter/",
97
+ "note": "Attests s ('s avonds, from the older genitive des avonds), t ('t was) and ns ('ns kijken). The same entries cover 's-Gravenhage and 's-Hertogenbosch, where a hyphen rather than a space follows the letter run."
82
98
  }
83
99
  ]
84
100
  }
@@ -12,7 +12,11 @@
12
12
  "close": "«",
13
13
  "innerSpace": "none"
14
14
  },
15
- "elisionIdioms": []
15
+ "elisionIdioms": [],
16
+ "elisionClitics": {
17
+ "before": [],
18
+ "after": []
19
+ }
16
20
  },
17
21
  "dash": {
18
22
  "parenthetical": "en-spaced",
@@ -12,7 +12,11 @@
12
12
  "close": "’",
13
13
  "innerSpace": "none"
14
14
  },
15
- "elisionIdioms": []
15
+ "elisionIdioms": [],
16
+ "elisionClitics": {
17
+ "before": ["d", "n", "pel", "m", "t", "lh", "sant"],
18
+ "after": []
19
+ }
16
20
  },
17
21
  "dash": {
18
22
  "parenthetical": "en-spaced",
@@ -77,6 +81,12 @@
77
81
  "rule": "nbsp",
78
82
  "cite": "Nenhuma fonte brasileira consultada exige espaço antes de pontuação dupla, nem ligação de abreviaturas a palavras ou iniciais; listas deliberadamente vazias",
79
83
  "note": "Justifica as listas vazias e initialBinding = \"none\", e diz com exatidão em que se apoiam: ao contrário do guia grego, a parte portuguesa NÃO contém nenhuma frase que negue pelo nome o uso francês do espaço antes da pontuação dupla. A posição é «nenhuma fonte o exige», e não «uma fonte o proíbe»; o que se observa é que nenhum exemplo do capítulo o escreve, e isso é uma observação, assinalada como tal. beforeNumber contém apenas «p.»: o ponto 10.9.1 atesta a ligação («p. 24», «Lei n.º 123»), mas «n.º» NÃO pode ser listado, porque a nota (3) do ponto 6.4 do mesmo guia prescreve um «o» sobrescrito e RECUSA expressamente tanto U+00BA como U+00B0 — listar «n.º» seria citar uma fonte que proíbe esse mesmo carácter. A decisão do operador de 18.9.2026 sobre o «n°» francês cobre uma fonte SILENCIOSA e não se estende a uma que fala e recusa. beforeWord fica vazio: o anexo A3 enumera «Sr.», «Dr.», «Prof.», mas limita-se a expandi-los e nenhuma fonte os mostra ligados ao nome seguinte. GAP REGISTADO, não resolvido: o ponto 10.9.1 exige espaço protegido DENTRO dos grupos de algarismos («um total de 12 345 euros»; «este espaço é protegido»), e locale.schema.json não tem campo para isso. Consultado em 18.9.2026."
84
+ },
85
+ {
86
+ "rule": "quotes",
87
+ "cite": "Acordo Ortográfico da Língua Portuguesa (1990), Base XVIII « Do apóstrofo », 1.º : a) « Faz-se uso do apóstrofo para cindir graficamente uma contração ou aglutinação vocabular, quando um elemento ou fração respetiva pertence propriamente a um conjunto vocabular distinto: d'Os Lusíadas, d'Os Sertões; n'Os Lusíadas, n'Os Sertões; pel'Os Lusíadas, pel'Os Sertões » ; b) « d'Ele, n'Ele, d'Aquele, n'Aquele, d'O, n'O, pel'O, m'O, t'O, lh'O » e a série feminina correspondente ; c) « Sant'Ana, Sant'Iago » ; d) « borda-d'água, copo-d'água, estrela-d'alva, mãe-d'água, pau-d'alho, pau-d'arco »",
88
+ "url": "https://www.priberam.pt/docs/AcOrtog90.pdf",
89
+ "note": "Base XVIII is a primary normative text common to both orthographies, which is why pt-PT and pt-BR carry the identical list: d, n, pel (1.º a and b), m, t, lh (1.º b) and sant (1.º c). Base XVIII 2.º bounds the set from the other side — the apóstrofo is not admissible in the combinations of de and em with the definite-article forms (do, da, dele, no, na, nele, num and the rest) — so this is an enumerated set rather than an argument from silence. The archaic anthroponymic forms of 1.º c) (Nun'Álvares, Pedr'Eanes) are cited by the same clause but deliberately not listed: both have a letter on each side of the mark, so quotes.md §3.2's medial-elision veto already declines them and an entry would be evidenced-inert. after is empty: Portuguese enclisis takes a hyphen (Base XVII: amá-lo, dá-se), never an apostrophe. The clause that makes this list load-bearing is 1.º a), which is about book titles: a title set in emphasis rather than italics puts a span boundary immediately after the mark, d'<em>Os Lusíadas</em>, which is precisely the shape the veto reads. Retrieved from a faithful reproduction hosted by Priberam/FLiP, including the official text's own errata footnotes, rather than from the official gazette; the Academia Brasileira de Letras PDF has no extractable text layer. quotes.md §3.2's span-boundary elision veto (spec 1.4.0) reads this list only when the mark's other literal neighbour is modes.md §3.2's inline span boundary, so every entry is matched against the maximal LETTER run on one side of the mark and nothing else. The list is decline-only: the worst it can do is refuse a pairing quotes would otherwise have formed."
80
90
  }
81
91
  ]
82
92
  }
@@ -12,7 +12,11 @@
12
12
  "close": "”",
13
13
  "innerSpace": "none"
14
14
  },
15
- "elisionIdioms": []
15
+ "elisionIdioms": [],
16
+ "elisionClitics": {
17
+ "before": ["d", "n", "pel", "m", "t", "lh", "sant"],
18
+ "after": []
19
+ }
16
20
  },
17
21
  "dash": {
18
22
  "parenthetical": "em-spaced",
@@ -79,6 +83,12 @@
79
83
  "cite": "Código de Redação Interinstitucional (PT, 2022), anexo A3, ponto 4 «Obras»: «p. ex. por exemplo»; «e. g. exempli gratia (por exemplo). Utilizar de preferência p. ex.»; Quarta parte, ponto 10.9.1: «paginação corrente, parágrafos, artigos: artigo 2.º, terceiro parágrafo, alínea b), p. 24»; nota (3) do ponto 6.4: «Para o «º» ordinal (1.º, 2.º…) ou em «n.º», utilizar a terminação «o» após o ponto e em posição superior à linha [não utilizar o sinal «º» do teclado Azerty nem a sequência Alt 0176 (símbolo do grau «°»)]»",
80
84
  "url": "https://op.europa.eu/pt/publication-detail/-/publication/01ed788a-d266-11ec-a95f-01aa75ed71a1",
81
85
  "note": "Justifica as listas vazias e initialBinding = \"none\", e diz com exatidão em que se apoiam: ao contrário do guia grego, a parte portuguesa NÃO contém nenhuma frase que negue pelo nome o uso francês do espaço antes da pontuação dupla. A posição é «nenhuma fonte o exige», e não «uma fonte o proíbe»; o que se observa é que nenhum exemplo do capítulo o escreve, e isso é uma observação, assinalada como tal. beforeNumber contém apenas «p.»: o ponto 10.9.1 atesta a ligação («p. 24», «Lei n.º 123»), mas «n.º» NÃO pode ser listado, porque a nota (3) do ponto 6.4 do mesmo guia prescreve um «o» sobrescrito e RECUSA expressamente tanto U+00BA como U+00B0 — listar «n.º» seria citar uma fonte que proíbe esse mesmo carácter. A decisão do operador de 18.9.2026 sobre o «n°» francês cobre uma fonte SILENCIOSA e não se estende a uma que fala e recusa. beforeWord fica vazio: o anexo A3 enumera «Sr.», «Dr.», «Prof.», mas limita-se a expandi-los e nenhuma fonte os mostra ligados ao nome seguinte. GAP REGISTADO, não resolvido: o ponto 10.9.1 exige espaço protegido DENTRO dos grupos de algarismos («um total de 12 345 euros»; «este espaço é protegido»), e locale.schema.json não tem campo para isso. Consultado em 18.9.2026."
86
+ },
87
+ {
88
+ "rule": "quotes",
89
+ "cite": "Acordo Ortográfico da Língua Portuguesa (1990), Base XVIII « Do apóstrofo », 1.º : a) « Faz-se uso do apóstrofo para cindir graficamente uma contração ou aglutinação vocabular, quando um elemento ou fração respetiva pertence propriamente a um conjunto vocabular distinto: d'Os Lusíadas, d'Os Sertões; n'Os Lusíadas, n'Os Sertões; pel'Os Lusíadas, pel'Os Sertões » ; b) « d'Ele, n'Ele, d'Aquele, n'Aquele, d'O, n'O, pel'O, m'O, t'O, lh'O » e a série feminina correspondente ; c) « Sant'Ana, Sant'Iago » ; d) « borda-d'água, copo-d'água, estrela-d'alva, mãe-d'água, pau-d'alho, pau-d'arco »",
90
+ "url": "https://www.priberam.pt/docs/AcOrtog90.pdf",
91
+ "note": "Base XVIII is a primary normative text common to both orthographies, which is why pt-PT and pt-BR carry the identical list: d, n, pel (1.º a and b), m, t, lh (1.º b) and sant (1.º c). Base XVIII 2.º bounds the set from the other side — the apóstrofo is not admissible in the combinations of de and em with the definite-article forms (do, da, dele, no, na, nele, num and the rest) — so this is an enumerated set rather than an argument from silence. The archaic anthroponymic forms of 1.º c) (Nun'Álvares, Pedr'Eanes) are cited by the same clause but deliberately not listed: both have a letter on each side of the mark, so quotes.md §3.2's medial-elision veto already declines them and an entry would be evidenced-inert. after is empty: Portuguese enclisis takes a hyphen (Base XVII: amá-lo, dá-se), never an apostrophe. The clause that makes this list load-bearing is 1.º a), which is about book titles: a title set in emphasis rather than italics puts a span boundary immediately after the mark, d'<em>Os Lusíadas</em>, which is precisely the shape the veto reads. Retrieved from a faithful reproduction hosted by Priberam/FLiP, including the official text's own errata footnotes, rather than from the official gazette; the Academia Brasileira de Letras PDF has no extractable text layer. quotes.md §3.2's span-boundary elision veto (spec 1.4.0) reads this list only when the mark's other literal neighbour is modes.md §3.2's inline span boundary, so every entry is matched against the maximal LETTER run on one side of the mark and nothing else. The list is decline-only: the worst it can do is refuse a pairing quotes would otherwise have formed."
82
92
  }
83
93
  ]
84
94
  }
@@ -1,5 +1,5 @@
1
1
  {
2
- "spec": "1.3.0",
2
+ "spec": "1.4.0",
3
3
  "$comment": "Locale resolution input. Algorithm is specified in spec/rules/locale-resolution.md and is identical in every runtime — never delegate it to a platform locale-negotiation library. \"spec\" here must track spec/VERSION exactly — it is not itself the global version source; scripts/validate-spec.mjs enforces the match.",
4
4
  "locales": [
5
5
  "en-US",
@@ -12,7 +12,11 @@
12
12
  "close": "“",
13
13
  "innerSpace": "none"
14
14
  },
15
- "elisionIdioms": []
15
+ "elisionIdioms": [],
16
+ "elisionClitics": {
17
+ "before": [],
18
+ "after": []
19
+ }
16
20
  },
17
21
  "dash": {
18
22
  "parenthetical": "em-spaced",
@@ -12,7 +12,11 @@
12
12
  "close": "’",
13
13
  "innerSpace": "none"
14
14
  },
15
- "elisionIdioms": []
15
+ "elisionIdioms": [],
16
+ "elisionClitics": {
17
+ "before": [],
18
+ "after": []
19
+ }
16
20
  },
17
21
  "dash": {
18
22
  "parenthetical": "en-spaced",
@@ -12,7 +12,11 @@
12
12
  "close": "”",
13
13
  "innerSpace": "none"
14
14
  },
15
- "elisionIdioms": []
15
+ "elisionIdioms": [],
16
+ "elisionClitics": {
17
+ "before": [],
18
+ "after": []
19
+ }
16
20
  },
17
21
  "dash": {
18
22
  "parenthetical": "em-spaced",
@@ -1,5 +1,5 @@
1
1
  {
2
- "spec": "1.3.0",
2
+ "spec": "1.4.0",
3
3
  "$comment": "Single source of truth for pipeline order. Rules run in ascending `order`. Disabling a rule removes it from the sequence and never reorders the rest. Rule ids are public API (see docs/ARCHITECTURE.md section 5). \"spec\" here must track spec/VERSION exactly — it is not itself the global version source; scripts/validate-spec.mjs enforces the match.",
4
4
  "rules": [
5
5
  {
@@ -48,7 +48,7 @@
48
48
  "default": "on",
49
49
  "modes": ["text", "html", "markdown", "yaml"],
50
50
  "localeData": ["quotes"],
51
- "summary": "Straight quotes to locale primary/secondary pairs with nesting resolution. Runs before `apostrophe` so that a straight apostrophe is still available as quote evidence. Declines to pair a medial single-quoted `n` (spec 1.1.0, every locale) or a span matching a cited `quotes.elisionIdioms` entry, so that `apostrophe` converts both marks instead."
51
+ "summary": "Straight quotes to locale primary/secondary pairs with nesting resolution. Runs before `apostrophe` so that a straight apostrophe is still available as quote evidence. Declines to pair a medial single-quoted `n` (spec 1.1.0, every locale) or a span matching a cited `quotes.elisionIdioms` entry, so that `apostrophe` converts both marks instead, and declines a mark written flush against an inline span boundary whose attaching word fragment is a cited `quotes.elisionClitics` entry (spec 1.4.0), so that a possessive or elision cannot take the pairing from the author's own quotation mark."
52
52
  },
53
53
  {
54
54
  "id": "apostrophe",
@@ -19,7 +19,7 @@
19
19
  "quotes": {
20
20
  "type": "object",
21
21
  "additionalProperties": false,
22
- "required": ["primary", "secondary", "elisionIdioms"],
22
+ "required": ["primary", "secondary", "elisionIdioms", "elisionClitics"],
23
23
  "properties": {
24
24
  "primary": { "$ref": "#/$defs/quotePair" },
25
25
  "secondary": { "$ref": "#/$defs/quotePair" },
@@ -28,7 +28,8 @@
28
28
  "items": { "$ref": "#/$defs/elisionIdiom" },
29
29
  "uniqueItems": true,
30
30
  "description": "Literal, closed-set elision idioms consulted only to decline pairing two NARROW quote marks as a quotation when the full context — the word immediately before the opening mark and the word immediately after the closing mark, not merely the elided content between them — matches (quotes.md 3.2's listed elision veto), e.g. English { left: \"rock\", elided: \"n\", right: \"roll\" } for \"rock 'n' roll\". Empty is the default and means no verified idiom of this shape — a missing citation is never a licence to guess. Matching the elided content alone (a bare word list) is insufficient and is not this field's shape: it cannot distinguish the idiom from an arbitrary quoted letter, e.g. \"The letter 'n' is common.\", which is exactly the false positive this three-part contract exists to avoid."
31
- }
31
+ },
32
+ "elisionClitics": { "$ref": "#/$defs/elisionClitics" }
32
33
  }
33
34
  },
34
35
  "dash": {
@@ -208,6 +209,26 @@
208
209
  "items": { "$ref": "#/$defs/singleChar" },
209
210
  "uniqueItems": true
210
211
  },
212
+ "elisionClitics": {
213
+ "type": "object",
214
+ "additionalProperties": false,
215
+ "required": ["before", "after"],
216
+ "description": "quotes.md 3.2's span-boundary elision veto (spec 1.4.0). Consulted ONLY when one literal neighbour of a NARROW mark is modes.md 3.2's inline boundary marker, so it is vacuous in text mode, which produces no marker. The trigger is the marker, never the mode: an implementation must not gate this on a mode test (quotes.md 3.2). Decline-only: it can never widen what quotes pairs, only narrow it. Both lists empty is the default and a total no-op, and is the correct content for a locale with no citable closed set of such fragments \u2014 a missing citation is never a licence to guess.",
217
+ "properties": {
218
+ "before": {
219
+ "type": "array",
220
+ "items": { "type": "string", "minLength": 1 },
221
+ "uniqueItems": true,
222
+ "description": "Word fragments that stand immediately BEFORE an elision apostrophe and attach forward to the following word: French l', d', qu', n' give \"l\", \"d\", \"qu\", \"n\". Authored lowercase \u2014 the veto folds only the first code point, and only ASCII A-Z, so \"L'\" matches an entry \"l\" and \"QU'\" does not match \"qu\". Every code point must be LETTER; JSON Schema cannot express a Unicode-category constraint, so that is enforced by scripts/validate-spec.mjs against the same pinned table the engine uses, exactly as for elisionIdiom's left/right."
223
+ },
224
+ "after": {
225
+ "type": "array",
226
+ "items": { "type": "string", "minLength": 1 },
227
+ "uniqueItems": true,
228
+ "description": "Word fragments that stand immediately AFTER an elision or possessive apostrophe and attach back to the preceding word, or forward from a word-initial omission \u2014 the lists are positional, not attachment claims (quotes.md 3.2). English ships the possessive \"s\" alone (x's); the contraction fragments t, ll, re, ve, d and m are deliberately absent, having no attested occurrence flush against a span boundary. Dutch ships \"s\", \"t\", \"ns\", \"k\", \"m\", \"em\", \"r\", \"et\" ('s morgens, 't was, 'ns kijken). Same lowercase authoring rule and same LETTER-only constraint as \"before\"."
229
+ }
230
+ }
231
+ },
211
232
  "elisionIdiom": {
212
233
  "type": "object",
213
234
  "additionalProperties": false,
@@ -144,6 +144,74 @@ module Polytypo
144
144
  #
145
145
  # idioms is the locale's quotes.elisionIdioms array: a list of Hashes with String keys
146
146
  # "left"/"elided"/"right", each a literal String (locale.schema.json).
147
+ # Span-boundary elision veto (quotes.md 3.2, spec 1.4.0, canonical issue #53 1).
148
+ #
149
+ # Fires only where one literal neighbour of a NARROW mark IS modes.md 3.2's inline
150
+ # MARKER -- the marker stands exactly where the attaching word would be, which is why
151
+ # the medial-elision veto cannot see the shape and why a possessive or elision written
152
+ # flush against a span was classified as a quotation candidate and inverted the
153
+ # enclosing pair.
154
+ #
155
+ # The attaching side is read as a MAXIMAL LETTER run bounded by a non-ALNUM code point
156
+ # and compared whole: a prefix test would match the entry "s" inside "sure" and eat
157
+ # <em>'sure'</em>, a genuine quotation. The comparison folds the RUN's first code point,
158
+ # ASCII A-Z only, and is exact thereafter -- never a locale-dependent case mapping
159
+ # (ARCHITECTURE.md 4.4), so String#downcase must not appear here.
160
+ #
161
+ # Keyed off the marker and never off the mode: a mode conditional is forbidden
162
+ # (modes.md 7.4), which is also why text mode needs no separate path.
163
+ def self.compute_span_boundary_veto_indices(cp, clitics)
164
+ before = clitics["before"] || []
165
+ after = clitics["after"] || []
166
+ return {} if before.empty? && after.empty?
167
+
168
+ before_cps = before.map(&:codepoints)
169
+ after_cps = after.map(&:codepoints)
170
+ vetoed = {}
171
+
172
+ (0...cp.length).each do |i|
173
+ next unless narrow?(cp[i])
174
+
175
+ if !after_cps.empty? && at(cp, i - 1) == Engine::MARKER &&
176
+ run_matches?(cp, i, 1, after_cps)
177
+ vetoed[i] = true
178
+ next
179
+ end
180
+ if !before_cps.empty? && at(cp, i + 1) == Engine::MARKER &&
181
+ run_matches?(cp, i, -1, before_cps)
182
+ vetoed[i] = true
183
+ end
184
+ end
185
+ vetoed
186
+ end
187
+
188
+ # The maximal LETTER run adjacent to the mark at `i`, growing in `dir`, compared against
189
+ # `entries`. Declines an empty run, and one an ALNUM code point continues past -- that
190
+ # bound is what makes the run the WHOLE fragment rather than a prefix of one.
191
+ def self.run_matches?(cp, i, dir, entries)
192
+ j = i + dir
193
+ j += dir while j >= 0 && j < cp.length && UnicodeUtil.letter?(cp[j])
194
+ return false if j == i + dir
195
+
196
+ outer = at(cp, j)
197
+ return false if outer != Engine::NONE && alnum?(outer)
198
+
199
+ start = dir == -1 ? j + 1 : i + 1
200
+ length = dir == -1 ? i - 1 - start + 1 : j - 1 - start + 1
201
+
202
+ entries.any? do |entry|
203
+ next false unless entry.length == length
204
+
205
+ same = ascii_lower(cp[start]) == entry[0]
206
+ (1...entry.length).each do |k|
207
+ break unless same
208
+
209
+ same = cp[start + k] == entry[k]
210
+ end
211
+ same
212
+ end
213
+ end
214
+
147
215
  def self.compute_idiom_matched_indices(cp, idioms)
148
216
  vetoed = {}
149
217
  return vetoed if idioms.nil? || idioms.empty?
@@ -141,7 +141,7 @@ module Polytypo
141
141
  # can_open always skips right and can_close always skips left (mandate 2's inner-side
142
142
  # skip); the outer side skips only when nbsp can reach it (the locale-derived
143
143
  # space_right/space_left sets), which is what keeps every verdict inert to nbsp (Lemma B).
144
- def self.collect_candidates(cp, skip_sets, idioms)
144
+ def self.collect_candidates(cp, skip_sets, idioms, clitics)
145
145
  n = cp.length
146
146
  candidates = []
147
147
 
@@ -149,9 +149,12 @@ module Polytypo
149
149
  # and the general ambiguous-medial-span shape (quotes.md 3.2a) -- quotes must decline
150
150
  # pairing for both, so apostrophe's own case ladder never independently "fixes" a shape
151
151
  # quotes left alone.
152
+ # spec 1.4.0 adds a third member to the same union: the span-boundary elision veto,
153
+ # which fires only where one literal neighbour is the inline MARKER (quotes.md 3.2).
152
154
  idiom_matched = QuoteAmbiguity.compute_idiom_matched_indices(cp, idioms)
153
155
  ambiguous_shape = QuoteAmbiguity.compute_ambiguous_shape_indices(cp)
154
- elision_vetoed = idiom_matched.merge(ambiguous_shape)
156
+ span_boundary = QuoteAmbiguity.compute_span_boundary_veto_indices(cp, clitics)
157
+ elision_vetoed = idiom_matched.merge(ambiguous_shape).merge(span_boundary)
155
158
 
156
159
  (0...n).each do |i|
157
160
  g = cp[i]
@@ -326,7 +329,7 @@ module Polytypo
326
329
  # simultaneous per round, and when the intersection fails to shrink A, the pair with the
327
330
  # greatest open index is forced out -- both clauses are normative, so two ports cannot
328
331
  # disagree.
329
- def self.certify(cp, initial_pairs, quotes_data, skip_sets, idioms)
332
+ def self.certify(cp, initial_pairs, quotes_data, skip_sets, idioms, clitics)
330
333
  accepted = initial_pairs.dup
331
334
  # Each round accepts or strictly shrinks `accepted`; it is finite and the empty set
332
335
  # accepts unconditionally, so the loop runs at most |A0| + 1 times (quotes.md 3.5). The
@@ -338,7 +341,7 @@ module Polytypo
338
341
 
339
342
  plan = compute_render_plan(cp, accepted, quotes_data)
340
343
  y, m = apply_render_plan(cp, plan)
341
- rederived = pair_candidates(y, collect_candidates(y, skip_sets, idioms))
344
+ rederived = pair_candidates(y, collect_candidates(y, skip_sets, idioms, clitics))
342
345
 
343
346
  b_set = Set.new(rederived)
344
347
  projected = accepted.map { |p| Pair.new(m[p.open], m[p.close]) }
@@ -396,15 +399,16 @@ module Polytypo
396
399
  def self.scan(cp, locale_data, _ctx)
397
400
  quotes_data = locale_data["quotes"]
398
401
  idioms = quotes_data["elisionIdioms"] || []
402
+ clitics = quotes_data["elisionClitics"] || {}
399
403
  skip_sets = compute_skip_sets(quotes_data)
400
404
 
401
- candidates = collect_candidates(cp, skip_sets, idioms)
405
+ candidates = collect_candidates(cp, skip_sets, idioms, clitics)
402
406
  return [] if candidates.empty?
403
407
 
404
408
  initial_pairs = pair_candidates(cp, candidates)
405
409
  return [] if initial_pairs.empty?
406
410
 
407
- accepted = certify(cp, initial_pairs, quotes_data, skip_sets, idioms)
411
+ accepted = certify(cp, initial_pairs, quotes_data, skip_sets, idioms, clitics)
408
412
  return [] if accepted.empty?
409
413
 
410
414
  emit(cp, accepted, quotes_data)
@@ -1,5 +1,5 @@
1
1
  # frozen_string_literal: true
2
2
 
3
3
  module Polytypo
4
- VERSION = "1.3.0"
4
+ VERSION = "1.4.0"
5
5
  end
metadata CHANGED
@@ -1,7 +1,7 @@
1
1
  --- !ruby/object:Gem::Specification
2
2
  name: polytypo
3
3
  version: !ruby/object:Gem::Version
4
- version: 1.3.0
4
+ version: 1.4.0
5
5
  platform: ruby
6
6
  authors:
7
7
  - Iurii Rogulia