polytypo 1.3.1 → 1.4.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/lib/polytypo/data/README.md +15 -0
- data/lib/polytypo/data/VERSION +1 -1
- data/lib/polytypo/data/fixtures/cs.json +1 -1
- data/lib/polytypo/data/fixtures/de-CH.json +1 -1
- data/lib/polytypo/data/fixtures/de-DE.json +1 -1
- data/lib/polytypo/data/fixtures/el.json +1 -1
- data/lib/polytypo/data/fixtures/en-GB.json +44 -1
- data/lib/polytypo/data/fixtures/en-US.json +18 -1
- data/lib/polytypo/data/fixtures/es.json +1 -1
- data/lib/polytypo/data/fixtures/fi.json +1 -1
- data/lib/polytypo/data/fixtures/fr-CA.json +9 -1
- data/lib/polytypo/data/fixtures/fr.json +50 -1
- data/lib/polytypo/data/fixtures/it.json +17 -1
- data/lib/polytypo/data/fixtures/locale-resolution.json +1 -1
- data/lib/polytypo/data/fixtures/nl.json +17 -1
- data/lib/polytypo/data/fixtures/pl.json +1 -1
- data/lib/polytypo/data/fixtures/pt-BR.json +9 -1
- data/lib/polytypo/data/fixtures/pt-PT.json +9 -1
- data/lib/polytypo/data/fixtures/ru.json +1 -1
- data/lib/polytypo/data/fixtures/sv.json +1 -1
- data/lib/polytypo/data/fixtures/uk.json +1 -1
- data/lib/polytypo/data/locales/cs.json +42 -57
- data/lib/polytypo/data/locales/de-CH.json +39 -44
- data/lib/polytypo/data/locales/de-DE.json +37 -48
- data/lib/polytypo/data/locales/el.json +6 -52
- data/lib/polytypo/data/locales/en-GB.json +8 -50
- data/lib/polytypo/data/locales/en-US.json +8 -62
- data/lib/polytypo/data/locales/es.json +34 -67
- data/lib/polytypo/data/locales/fi.json +12 -78
- data/lib/polytypo/data/locales/fr-CA.json +55 -52
- data/lib/polytypo/data/locales/fr.json +60 -59
- data/lib/polytypo/data/locales/it.json +26 -59
- data/lib/polytypo/data/locales/nl.json +33 -49
- data/lib/polytypo/data/locales/pl.json +33 -62
- data/lib/polytypo/data/locales/pt-BR.json +32 -47
- data/lib/polytypo/data/locales/pt-PT.json +32 -49
- data/lib/polytypo/data/locales/registry.json +1 -1
- data/lib/polytypo/data/locales/ru.json +35 -46
- data/lib/polytypo/data/locales/sv.json +9 -64
- data/lib/polytypo/data/locales/uk.json +32 -56
- data/lib/polytypo/data/rules/order.json +2 -2
- data/lib/polytypo/data/schema/locale.schema.json +23 -2
- data/lib/polytypo/engine/rules/quote_ambiguity.rb +68 -0
- data/lib/polytypo/engine/rules/quotes.rb +10 -6
- data/lib/polytypo/version.rb +1 -1
- metadata +1 -1
|
@@ -12,7 +12,11 @@
|
|
|
12
12
|
"close": "”",
|
|
13
13
|
"innerSpace": "none"
|
|
14
14
|
},
|
|
15
|
-
"elisionIdioms": []
|
|
15
|
+
"elisionIdioms": [],
|
|
16
|
+
"elisionClitics": {
|
|
17
|
+
"before": [],
|
|
18
|
+
"after": []
|
|
19
|
+
}
|
|
16
20
|
},
|
|
17
21
|
"dash": {
|
|
18
22
|
"parenthetical": "em-spaced",
|
|
@@ -22,8 +26,21 @@
|
|
|
22
26
|
"abbreviatedAfterTerminal": true
|
|
23
27
|
},
|
|
24
28
|
"hyphen": {
|
|
25
|
-
"prefixes": [
|
|
26
|
-
|
|
29
|
+
"prefixes": [
|
|
30
|
+
"будь-",
|
|
31
|
+
"казна-",
|
|
32
|
+
"хтозна-",
|
|
33
|
+
"бозна-"
|
|
34
|
+
],
|
|
35
|
+
"suffixes": [
|
|
36
|
+
"-бо",
|
|
37
|
+
"-но",
|
|
38
|
+
"-от",
|
|
39
|
+
"-то",
|
|
40
|
+
"-таки",
|
|
41
|
+
"-будь",
|
|
42
|
+
"-небудь"
|
|
43
|
+
],
|
|
27
44
|
"compounds": [
|
|
28
45
|
"вид-во",
|
|
29
46
|
"гр-н",
|
|
@@ -50,7 +67,12 @@
|
|
|
50
67
|
"beforePunctuation": [],
|
|
51
68
|
"narrowBeforePunctuation": [],
|
|
52
69
|
"afterShortWords": [],
|
|
53
|
-
"abbreviations": [
|
|
70
|
+
"abbreviations": [
|
|
71
|
+
"і т. д.",
|
|
72
|
+
"і т. ін.",
|
|
73
|
+
"та ін.",
|
|
74
|
+
"куб. см"
|
|
75
|
+
],
|
|
54
76
|
"beforeUnits": [
|
|
55
77
|
"%",
|
|
56
78
|
"га",
|
|
@@ -73,58 +95,12 @@
|
|
|
73
95
|
"р."
|
|
74
96
|
],
|
|
75
97
|
"beforeNumber": [],
|
|
76
|
-
"beforeWord": [
|
|
98
|
+
"beforeWord": [
|
|
99
|
+
"акад.",
|
|
100
|
+
"доц.",
|
|
101
|
+
"проф."
|
|
102
|
+
],
|
|
77
103
|
"afterSymbols": [],
|
|
78
104
|
"initialBinding": "chain"
|
|
79
|
-
}
|
|
80
|
-
"sources": [
|
|
81
|
-
{
|
|
82
|
-
"rule": "quotes",
|
|
83
|
-
"cite": "Український правопис (2019), схвалений Кабінетом Міністрів України (Постанова № 437 від 22.05.2019), § 164 «ЛАПКИ (« », “ ”, „ “, рідше „ ”)», п. 3: «У функції перших рекомендовано вживати кутові лапки, або «лапки-ялинки» («…»), у функції внутрішніх — «лапки-лапки» (“…” та ін.): «Це мій “Кобзар”», — сказав він»; там само: «На письмі (у рукописних текстах) «лапки-лапки» традиційно використовують у формі „…“»",
|
|
84
|
-
"url": "https://www.ulif.org.ua/system/files/pravopus-new.pdf",
|
|
85
|
-
"note": "Зовнішні лапки — U+00AB і U+00BB, внутрішні — U+201C і U+201D. Код кожного знака встановлено за друкованим зображенням с. 247 офіційного видання, а не за текстовою конверсією: у прикладі § 164 п. 3 обидва внутрішні знаки підняті до верху рядка, відкривальний має форму 6, закривальний — форму 9. ЦЕ СПРОСТОВУЄ поширене припущення, що українська бере всередину „…“: заголовок § 164 справді дозволяє чотири пари, але саме п. 3 відносить „…“ до РУКОПИСНИХ текстів, а для друкованого рекомендує “…”. innerSpace = «none»: жодне джерело не вимагає відступу всередині лапок. Третій рівень вкладення в locale.schema.json невиразний. Звірено 18.09.2026."
|
|
86
|
-
},
|
|
87
|
-
{
|
|
88
|
-
"rule": "dashes",
|
|
89
|
-
"cite": "Український правопис (2019), § 161 «ТИРЕ (—)», I, п. 10—11 і Примітка 2: тире ставимо перед відокремленим зворотом або вставленою конструкцією — «Топольський — молодий чоловік, але — на думку пана посла — незвичайно талановитий і солідний» (О. Маковей)",
|
|
90
|
-
"url": "https://www.ulif.org.ua/system/files/pravopus-new.pdf",
|
|
91
|
-
"note": "Обґрунтовує dash.parenthetical = «em-spaced». ДОВЖИНА встановлена прямо: заголовок параграфа — «ТИРЕ (—)», тобто U+2014, і цей самий знак стоїть у кожному прикладі §§ 161, 166, 167. ВІДБИВКА встановлена протиставленням, а не прозовим формулюванням: правопис ніде не пише «тире відбивається пробілами», але Примітка до п. 14 каже «тире ставимо без відступів між знаками» саме для випадку між цифрами, а «без відступів» має сенс лише як відхилення від відбитого за замовчуванням. Це єдиний запис цього файлу, що спирається на висновок із тексту джерела, і він позначений як такий свідомо. Звірено 18.09.2026."
|
|
92
|
-
},
|
|
93
|
-
{
|
|
94
|
-
"rule": "ranges",
|
|
95
|
-
"cite": "Український правопис (2019), § 161 «ТИРЕ (—)», I, п. 14, Примітка: «Між цифрами в таких випадках тире ставимо без відступів між знаками: у 2010—2018 роках; пам'ятки української мови XVI—XVIII ст.; на сторінках 1—10; у 1—4 томах, але, напр.: наприкінці XX — на початку XXI ст.»",
|
|
96
|
-
"url": "https://www.ulif.org.ua/system/files/pravopus-new.pdf",
|
|
97
|
-
"note": "Обґрунтовує dash.range = «em-tight»: довге тире U+2014 без відбивки. Протиставлення в самій Примітці («на сторінках 1—10» без відступів, але «наприкінці XX — на початку XXI ст.» з відступами) показує, що безвідступна форма стосується рівно того, що робить правило ranges. Правило вимкнене за замовчуванням (spec 0.5.0). Звірено 18.09.2026."
|
|
98
|
-
},
|
|
99
|
-
{
|
|
100
|
-
"rule": "ellipsis",
|
|
101
|
-
"cite": "Український правопис (2019), § 162 «ТРИ КРАПКИ, АБО КРАПКИ (…)», Примітка: «у постпозиції — після знака питання і знака оклику — ставимо дві крапки: Стражники на людей стріляли, це відомо, а щоб селяни?.. (К. Гордієнко); Встає народ, гудуть мости, Рокочуть ріки ясноводі!.. (М. Рильський)»; § 166 «КОМБІНОВАНЕ ВЖИВАННЯ РОЗДІЛОВИХ ЗНАКІВ», п. 2, що перелічує допустимі поєднання як «…?; …!; ?..; !..»",
|
|
102
|
-
"url": "https://www.ulif.org.ua/system/files/pravopus-new.pdf",
|
|
103
|
-
"note": "Обґрунтовує abbreviatedAfterTerminal = true. Українська — друга після російської локаль із цим значенням, і це перевірено окремо за двома параграфами, а не перенесено за аналогією зі спорідненої мови. Звірено 18.09.2026."
|
|
104
|
-
},
|
|
105
|
-
{
|
|
106
|
-
"rule": "hyphen",
|
|
107
|
-
"cite": "Український правопис (2019), § 64 «Технічні правила переносу», п. 4: «Не можна розривати умовні (графічні) скорочення на зразок вид-во, і т. д., і т. ін., та ін., т-во тощо»; § 62, п. 2: «У графічних скороченнях пропущену середню частину слова позначаємо дефісом: вид-во (видавництво), гр-н (громадянин), ін-т (інститут), р-н (район), ун-т (університет), ф-ка (фабрика)»",
|
|
108
|
-
"url": "https://www.ulif.org.ua/system/files/pravopus-new.pdf",
|
|
109
|
-
"note": "Це єдина частина hyphen для української, де НЕРОЗРИВНІСТЬ САМА Є НОРМОЮ, а не рішенням проєкту: § 64 п. 4 прямо забороняє розривати цей клас, а § 62 п. 2 закриває його літеральним переліком, бо «тощо» в § 64 залишає клас відкритим. Українська цим відрізняється від польської, де PWN [196] поділ у місці дефіса саме ПРИПИСУЄ, через що pl.json має три порожні списки. Звірено 18.09.2026."
|
|
110
|
-
},
|
|
111
|
-
{
|
|
112
|
-
"rule": "hyphen",
|
|
113
|
-
"cite": "Український правопис (2019), § 42 «Прийменники», п. 2: «З дефісом пишемо складені прийменники, утворені з простих прийменників з, із та інших прийменників: з-за (із-за), з-над, з-перед, з-під (із-під), з-поза, з-поміж, з-понад, з-попід, з-посеред, з-проміж»; § 44 «Частки», п. 3: «З дефісом пишемо: 1) частки -бо, -но, -от, -то, -таки … 2) частки будь-, -будь, -небудь, казна-, хтозна-, бозна- із займенниками і прислівниками»",
|
|
114
|
-
"url": "https://www.ulif.org.ua/system/files/pravopus-new.pdf",
|
|
115
|
-
"note": "Ці параграфи нормативні для СКЛАДУ форм, але не для заборони переносу, і змішувати це не можна — так само, як у ru.json. §§ 63 і 64 прочитано повністю: § 63 нормує поділ за складами, § 64 забороняє розривати ініціали, назви мір, нарощення й графічні скорочення, але про дефіс у формах §§ 42 і 44 не каже нічого. Отже, для цих класів зв'язування через U+2011 — рішення проєкту (docs/PLAN.md §3.3), а не орфографічна норма, і це записано прямо. НЕ ВНЕСЕНО, свідомо: § 64 п. 3 («Граматичні закінчення, поєднані із цифрами дефісом, не можна відривати й переносити: 2-й, 4-го, 10-му») нормативно нерозривний, але це форма «цифра-дефіс-літера»; чи є це членством у hyphen.suffixes, чи вже алгоритмом, має вирішити spec-guardian. Звірено 18.09.2026."
|
|
116
|
-
},
|
|
117
|
-
{
|
|
118
|
-
"rule": "nbsp",
|
|
119
|
-
"cite": "Український правопис (2019), § 64 «Технічні правила переносу», п. 1: «Не можна переносити прізвища, залишаючи в кінці попереднього рядка ініціали або інші умовні скорочення, що належать до них: Т. Г. Шевченко (не Т. Г. // Шевченко), гр. Іваненко, акад. (доц., проф.) Гончаренко, п. Гнатюк»; п. 4: «Не можна розривати умовні (графічні) скорочення на зразок вид-во, і т. д., і т. ін., та ін., т-во тощо»; п. 5: «Не можна переносити в наступний рядок розділові знаки (крім тире), дужку або лапки, що закривають попередній рядок»",
|
|
120
|
-
"url": "https://www.ulif.org.ua/system/files/pravopus-new.pdf",
|
|
121
|
-
"note": "§ 64 — ЗАКРИТИЙ перелік із п'яти пунктів, і це важливо не лише тим, що він містить, а й тим, чого в ньому немає. (1) initialBinding = «chain»: п. 1 наводить послідовність ДВОХ ініціалів і не дає прикладу одного ініціала перед прізвищем. (2) beforeWord: з п. 1 взято лише «акад.», «доц.», «проф.». «п.» і «гр.» свідомо НЕ внесено за тією самою дисципліною, з якої ru.json виключає «г.»: «п.» — це також «пункт» («п. 3»), «гр.» — також «градус» і «графа», а літеральний список цих напрямків не розрізняє. (3) beforePunctuation і narrowBeforePunctuation порожні: п. 5 прив'язує розділовий знак до ПОПЕРЕДНЬОГО слова — це заперечення, а не мовчання. (4) afterShortWords ПОРОЖНІЙ, і це підтверджена відсутність норми: перелік § 64 закритий, правила про однобуквені прийменники в ньому немає, а в самому правописі «і», «у», «в», «з» регулярно стоять у кінці рядка. Список ru.json сюди НЕ переноситься за аналогією, хоча мови споріднені. (5) afterSymbols і beforeNumber порожні: конструкцій «символ + число» і «скорочення + число» § 64 не містить. Звірено 18.09.2026."
|
|
122
|
-
},
|
|
123
|
-
{
|
|
124
|
-
"rule": "nbsp",
|
|
125
|
-
"cite": "Український правопис (2019), § 64, п. 2: «Не можна відривати скорочені назви мір від цифр, до яких вони належать: 2008 р. (не 2008 // р.), 150 га (не 150 // га), 20 см³ або 20 куб. см, 5 г (не 5 // г)»; § 62: «Скорочені назви одиниць вимірювання пишемо без крапок: Б — байт, Вт — ват, г — грам, га — гектар, год — година, дм — дециметр, кБ — кілобайт, кВт — кіловат, кг — кілограм, км — кілометр, л — літр, м — метр, мм — міліметр, с — секунда, см — сантиметр, т — тонна, хв — хвилина, ц — центнер»; BIPM, The International System of Units (SI), 9th ed., concise summary: «A single space is always left between the number and the unit»",
|
|
126
|
-
"url": "https://www.bipm.org/documents/20126/41483022/SI-Brochure-9-concise-EN.pdf",
|
|
127
|
-
"note": "Розподіл ролей тут ІНШИЙ, ніж у pl, fi, sv і en-US: там BIPM дає перелік символів, а національне джерело — конвенцію відступу; для української національне джерело дає і те, і те, бо § 64 п. 2 нормує саме невідривність. BIPM наведено як підтвердження загального принципу, а не як несуча цитата. Однобуквені позначення (г, л, м, с, т, ц, Б) свідомо пропущено: правий кордон N5 вимагає лише, щоб cp[a+k] не належав ALNUM, і пробіл цю умову задовольняє, тож однобуквений запис зв'язував би звичайну прозу. «р.» внесено: на відміну від російського «г.», українське «р.» означає тільки «рік» і зв'язується вліво після числа. «%» внесено ЯК РІШЕННЯ, а не як цитата: його немає ні в § 62, ні в § 64, ні в скороченому викладі брошури BIPM (правило про % — у § 5.4.7 повного видання, яке не діставалося), але відбивка відсотка від числа є і в російській локалі, і в усіх інших локалях цього проєкту, а розбіжність тут дала б українській вужчу поведінку без жодної підстави в джерелі. Звірено 18.09.2026."
|
|
128
|
-
}
|
|
129
|
-
]
|
|
105
|
+
}
|
|
130
106
|
}
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
{
|
|
2
|
-
"spec": "1.
|
|
2
|
+
"spec": "1.4.0",
|
|
3
3
|
"$comment": "Single source of truth for pipeline order. Rules run in ascending `order`. Disabling a rule removes it from the sequence and never reorders the rest. Rule ids are public API (see docs/ARCHITECTURE.md section 5). \"spec\" here must track spec/VERSION exactly — it is not itself the global version source; scripts/validate-spec.mjs enforces the match.",
|
|
4
4
|
"rules": [
|
|
5
5
|
{
|
|
@@ -48,7 +48,7 @@
|
|
|
48
48
|
"default": "on",
|
|
49
49
|
"modes": ["text", "html", "markdown", "yaml"],
|
|
50
50
|
"localeData": ["quotes"],
|
|
51
|
-
"summary": "Straight quotes to locale primary/secondary pairs with nesting resolution. Runs before `apostrophe` so that a straight apostrophe is still available as quote evidence. Declines to pair a medial single-quoted `n` (spec 1.1.0, every locale) or a span matching a cited `quotes.elisionIdioms` entry, so that `apostrophe` converts both marks instead."
|
|
51
|
+
"summary": "Straight quotes to locale primary/secondary pairs with nesting resolution. Runs before `apostrophe` so that a straight apostrophe is still available as quote evidence. Declines to pair a medial single-quoted `n` (spec 1.1.0, every locale) or a span matching a cited `quotes.elisionIdioms` entry, so that `apostrophe` converts both marks instead, and declines a mark written flush against an inline span boundary whose attaching word fragment is a cited `quotes.elisionClitics` entry (spec 1.4.0), so that a possessive or elision cannot take the pairing from the author's own quotation mark."
|
|
52
52
|
},
|
|
53
53
|
{
|
|
54
54
|
"id": "apostrophe",
|
|
@@ -19,7 +19,7 @@
|
|
|
19
19
|
"quotes": {
|
|
20
20
|
"type": "object",
|
|
21
21
|
"additionalProperties": false,
|
|
22
|
-
"required": ["primary", "secondary", "elisionIdioms"],
|
|
22
|
+
"required": ["primary", "secondary", "elisionIdioms", "elisionClitics"],
|
|
23
23
|
"properties": {
|
|
24
24
|
"primary": { "$ref": "#/$defs/quotePair" },
|
|
25
25
|
"secondary": { "$ref": "#/$defs/quotePair" },
|
|
@@ -28,7 +28,8 @@
|
|
|
28
28
|
"items": { "$ref": "#/$defs/elisionIdiom" },
|
|
29
29
|
"uniqueItems": true,
|
|
30
30
|
"description": "Literal, closed-set elision idioms consulted only to decline pairing two NARROW quote marks as a quotation when the full context — the word immediately before the opening mark and the word immediately after the closing mark, not merely the elided content between them — matches (quotes.md 3.2's listed elision veto), e.g. English { left: \"rock\", elided: \"n\", right: \"roll\" } for \"rock 'n' roll\". Empty is the default and means no verified idiom of this shape — a missing citation is never a licence to guess. Matching the elided content alone (a bare word list) is insufficient and is not this field's shape: it cannot distinguish the idiom from an arbitrary quoted letter, e.g. \"The letter 'n' is common.\", which is exactly the false positive this three-part contract exists to avoid."
|
|
31
|
-
}
|
|
31
|
+
},
|
|
32
|
+
"elisionClitics": { "$ref": "#/$defs/elisionClitics" }
|
|
32
33
|
}
|
|
33
34
|
},
|
|
34
35
|
"dash": {
|
|
@@ -208,6 +209,26 @@
|
|
|
208
209
|
"items": { "$ref": "#/$defs/singleChar" },
|
|
209
210
|
"uniqueItems": true
|
|
210
211
|
},
|
|
212
|
+
"elisionClitics": {
|
|
213
|
+
"type": "object",
|
|
214
|
+
"additionalProperties": false,
|
|
215
|
+
"required": ["before", "after"],
|
|
216
|
+
"description": "quotes.md 3.2's span-boundary elision veto (spec 1.4.0). Consulted ONLY when one literal neighbour of a NARROW mark is modes.md 3.2's inline boundary marker, so it is vacuous in text mode, which produces no marker. The trigger is the marker, never the mode: an implementation must not gate this on a mode test (quotes.md 3.2). Decline-only: it can never widen what quotes pairs, only narrow it. Both lists empty is the default and a total no-op, and is the correct content for a locale with no citable closed set of such fragments \u2014 a missing citation is never a licence to guess.",
|
|
217
|
+
"properties": {
|
|
218
|
+
"before": {
|
|
219
|
+
"type": "array",
|
|
220
|
+
"items": { "type": "string", "minLength": 1 },
|
|
221
|
+
"uniqueItems": true,
|
|
222
|
+
"description": "Word fragments that stand immediately BEFORE an elision apostrophe and attach forward to the following word: French l', d', qu', n' give \"l\", \"d\", \"qu\", \"n\". Authored lowercase \u2014 the veto folds only the first code point, and only ASCII A-Z, so \"L'\" matches an entry \"l\" and \"QU'\" does not match \"qu\". Every code point must be LETTER; JSON Schema cannot express a Unicode-category constraint, so that is enforced by scripts/validate-spec.mjs against the same pinned table the engine uses, exactly as for elisionIdiom's left/right."
|
|
223
|
+
},
|
|
224
|
+
"after": {
|
|
225
|
+
"type": "array",
|
|
226
|
+
"items": { "type": "string", "minLength": 1 },
|
|
227
|
+
"uniqueItems": true,
|
|
228
|
+
"description": "Word fragments that stand immediately AFTER an elision or possessive apostrophe and attach back to the preceding word, or forward from a word-initial omission \u2014 the lists are positional, not attachment claims (quotes.md 3.2). English ships the possessive \"s\" alone (x's); the contraction fragments t, ll, re, ve, d and m are deliberately absent, having no attested occurrence flush against a span boundary. Dutch ships \"s\", \"t\", \"ns\", \"k\", \"m\", \"em\", \"r\", \"et\" ('s morgens, 't was, 'ns kijken). Same lowercase authoring rule and same LETTER-only constraint as \"before\"."
|
|
229
|
+
}
|
|
230
|
+
}
|
|
231
|
+
},
|
|
211
232
|
"elisionIdiom": {
|
|
212
233
|
"type": "object",
|
|
213
234
|
"additionalProperties": false,
|
|
@@ -144,6 +144,74 @@ module Polytypo
|
|
|
144
144
|
#
|
|
145
145
|
# idioms is the locale's quotes.elisionIdioms array: a list of Hashes with String keys
|
|
146
146
|
# "left"/"elided"/"right", each a literal String (locale.schema.json).
|
|
147
|
+
# Span-boundary elision veto (quotes.md 3.2, spec 1.4.0, canonical issue #53 1).
|
|
148
|
+
#
|
|
149
|
+
# Fires only where one literal neighbour of a NARROW mark IS modes.md 3.2's inline
|
|
150
|
+
# MARKER -- the marker stands exactly where the attaching word would be, which is why
|
|
151
|
+
# the medial-elision veto cannot see the shape and why a possessive or elision written
|
|
152
|
+
# flush against a span was classified as a quotation candidate and inverted the
|
|
153
|
+
# enclosing pair.
|
|
154
|
+
#
|
|
155
|
+
# The attaching side is read as a MAXIMAL LETTER run bounded by a non-ALNUM code point
|
|
156
|
+
# and compared whole: a prefix test would match the entry "s" inside "sure" and eat
|
|
157
|
+
# <em>'sure'</em>, a genuine quotation. The comparison folds the RUN's first code point,
|
|
158
|
+
# ASCII A-Z only, and is exact thereafter -- never a locale-dependent case mapping
|
|
159
|
+
# (ARCHITECTURE.md 4.4), so String#downcase must not appear here.
|
|
160
|
+
#
|
|
161
|
+
# Keyed off the marker and never off the mode: a mode conditional is forbidden
|
|
162
|
+
# (modes.md 7.4), which is also why text mode needs no separate path.
|
|
163
|
+
def self.compute_span_boundary_veto_indices(cp, clitics)
|
|
164
|
+
before = clitics["before"] || []
|
|
165
|
+
after = clitics["after"] || []
|
|
166
|
+
return {} if before.empty? && after.empty?
|
|
167
|
+
|
|
168
|
+
before_cps = before.map(&:codepoints)
|
|
169
|
+
after_cps = after.map(&:codepoints)
|
|
170
|
+
vetoed = {}
|
|
171
|
+
|
|
172
|
+
(0...cp.length).each do |i|
|
|
173
|
+
next unless narrow?(cp[i])
|
|
174
|
+
|
|
175
|
+
if !after_cps.empty? && at(cp, i - 1) == Engine::MARKER &&
|
|
176
|
+
run_matches?(cp, i, 1, after_cps)
|
|
177
|
+
vetoed[i] = true
|
|
178
|
+
next
|
|
179
|
+
end
|
|
180
|
+
if !before_cps.empty? && at(cp, i + 1) == Engine::MARKER &&
|
|
181
|
+
run_matches?(cp, i, -1, before_cps)
|
|
182
|
+
vetoed[i] = true
|
|
183
|
+
end
|
|
184
|
+
end
|
|
185
|
+
vetoed
|
|
186
|
+
end
|
|
187
|
+
|
|
188
|
+
# The maximal LETTER run adjacent to the mark at `i`, growing in `dir`, compared against
|
|
189
|
+
# `entries`. Declines an empty run, and one an ALNUM code point continues past -- that
|
|
190
|
+
# bound is what makes the run the WHOLE fragment rather than a prefix of one.
|
|
191
|
+
def self.run_matches?(cp, i, dir, entries)
|
|
192
|
+
j = i + dir
|
|
193
|
+
j += dir while j >= 0 && j < cp.length && UnicodeUtil.letter?(cp[j])
|
|
194
|
+
return false if j == i + dir
|
|
195
|
+
|
|
196
|
+
outer = at(cp, j)
|
|
197
|
+
return false if outer != Engine::NONE && alnum?(outer)
|
|
198
|
+
|
|
199
|
+
start = dir == -1 ? j + 1 : i + 1
|
|
200
|
+
length = dir == -1 ? i - 1 - start + 1 : j - 1 - start + 1
|
|
201
|
+
|
|
202
|
+
entries.any? do |entry|
|
|
203
|
+
next false unless entry.length == length
|
|
204
|
+
|
|
205
|
+
same = ascii_lower(cp[start]) == entry[0]
|
|
206
|
+
(1...entry.length).each do |k|
|
|
207
|
+
break unless same
|
|
208
|
+
|
|
209
|
+
same = cp[start + k] == entry[k]
|
|
210
|
+
end
|
|
211
|
+
same
|
|
212
|
+
end
|
|
213
|
+
end
|
|
214
|
+
|
|
147
215
|
def self.compute_idiom_matched_indices(cp, idioms)
|
|
148
216
|
vetoed = {}
|
|
149
217
|
return vetoed if idioms.nil? || idioms.empty?
|
|
@@ -141,7 +141,7 @@ module Polytypo
|
|
|
141
141
|
# can_open always skips right and can_close always skips left (mandate 2's inner-side
|
|
142
142
|
# skip); the outer side skips only when nbsp can reach it (the locale-derived
|
|
143
143
|
# space_right/space_left sets), which is what keeps every verdict inert to nbsp (Lemma B).
|
|
144
|
-
def self.collect_candidates(cp, skip_sets, idioms)
|
|
144
|
+
def self.collect_candidates(cp, skip_sets, idioms, clitics)
|
|
145
145
|
n = cp.length
|
|
146
146
|
candidates = []
|
|
147
147
|
|
|
@@ -149,9 +149,12 @@ module Polytypo
|
|
|
149
149
|
# and the general ambiguous-medial-span shape (quotes.md 3.2a) -- quotes must decline
|
|
150
150
|
# pairing for both, so apostrophe's own case ladder never independently "fixes" a shape
|
|
151
151
|
# quotes left alone.
|
|
152
|
+
# spec 1.4.0 adds a third member to the same union: the span-boundary elision veto,
|
|
153
|
+
# which fires only where one literal neighbour is the inline MARKER (quotes.md 3.2).
|
|
152
154
|
idiom_matched = QuoteAmbiguity.compute_idiom_matched_indices(cp, idioms)
|
|
153
155
|
ambiguous_shape = QuoteAmbiguity.compute_ambiguous_shape_indices(cp)
|
|
154
|
-
|
|
156
|
+
span_boundary = QuoteAmbiguity.compute_span_boundary_veto_indices(cp, clitics)
|
|
157
|
+
elision_vetoed = idiom_matched.merge(ambiguous_shape).merge(span_boundary)
|
|
155
158
|
|
|
156
159
|
(0...n).each do |i|
|
|
157
160
|
g = cp[i]
|
|
@@ -326,7 +329,7 @@ module Polytypo
|
|
|
326
329
|
# simultaneous per round, and when the intersection fails to shrink A, the pair with the
|
|
327
330
|
# greatest open index is forced out -- both clauses are normative, so two ports cannot
|
|
328
331
|
# disagree.
|
|
329
|
-
def self.certify(cp, initial_pairs, quotes_data, skip_sets, idioms)
|
|
332
|
+
def self.certify(cp, initial_pairs, quotes_data, skip_sets, idioms, clitics)
|
|
330
333
|
accepted = initial_pairs.dup
|
|
331
334
|
# Each round accepts or strictly shrinks `accepted`; it is finite and the empty set
|
|
332
335
|
# accepts unconditionally, so the loop runs at most |A0| + 1 times (quotes.md 3.5). The
|
|
@@ -338,7 +341,7 @@ module Polytypo
|
|
|
338
341
|
|
|
339
342
|
plan = compute_render_plan(cp, accepted, quotes_data)
|
|
340
343
|
y, m = apply_render_plan(cp, plan)
|
|
341
|
-
rederived = pair_candidates(y, collect_candidates(y, skip_sets, idioms))
|
|
344
|
+
rederived = pair_candidates(y, collect_candidates(y, skip_sets, idioms, clitics))
|
|
342
345
|
|
|
343
346
|
b_set = Set.new(rederived)
|
|
344
347
|
projected = accepted.map { |p| Pair.new(m[p.open], m[p.close]) }
|
|
@@ -396,15 +399,16 @@ module Polytypo
|
|
|
396
399
|
def self.scan(cp, locale_data, _ctx)
|
|
397
400
|
quotes_data = locale_data["quotes"]
|
|
398
401
|
idioms = quotes_data["elisionIdioms"] || []
|
|
402
|
+
clitics = quotes_data["elisionClitics"] || {}
|
|
399
403
|
skip_sets = compute_skip_sets(quotes_data)
|
|
400
404
|
|
|
401
|
-
candidates = collect_candidates(cp, skip_sets, idioms)
|
|
405
|
+
candidates = collect_candidates(cp, skip_sets, idioms, clitics)
|
|
402
406
|
return [] if candidates.empty?
|
|
403
407
|
|
|
404
408
|
initial_pairs = pair_candidates(cp, candidates)
|
|
405
409
|
return [] if initial_pairs.empty?
|
|
406
410
|
|
|
407
|
-
accepted = certify(cp, initial_pairs, quotes_data, skip_sets, idioms)
|
|
411
|
+
accepted = certify(cp, initial_pairs, quotes_data, skip_sets, idioms, clitics)
|
|
408
412
|
return [] if accepted.empty?
|
|
409
413
|
|
|
410
414
|
emit(cp, accepted, quotes_data)
|
data/lib/polytypo/version.rb
CHANGED