polytypo 1.6.3 → 1.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (54) hide show
  1. checksums.yaml +4 -4
  2. data/README.md +25 -0
  3. data/lib/polytypo/data/.not-vendored +11 -0
  4. data/lib/polytypo/data/README.md +16 -11
  5. data/lib/polytypo/data/VERSION +1 -1
  6. data/lib/polytypo/data/fixtures/cs.json +1 -1
  7. data/lib/polytypo/data/fixtures/de-CH.json +1 -1
  8. data/lib/polytypo/data/fixtures/de-DE.json +14 -1
  9. data/lib/polytypo/data/fixtures/el.json +1 -1
  10. data/lib/polytypo/data/fixtures/en-GB.json +1 -1
  11. data/lib/polytypo/data/fixtures/en-US.json +143 -1
  12. data/lib/polytypo/data/fixtures/es.json +1 -1
  13. data/lib/polytypo/data/fixtures/fi.json +1 -1
  14. data/lib/polytypo/data/fixtures/fr-CA.json +1 -1
  15. data/lib/polytypo/data/fixtures/fr.json +13 -1
  16. data/lib/polytypo/data/fixtures/it.json +1 -1
  17. data/lib/polytypo/data/fixtures/locale-resolution.json +1 -1
  18. data/lib/polytypo/data/fixtures/nl.json +1 -1
  19. data/lib/polytypo/data/fixtures/pl.json +1 -1
  20. data/lib/polytypo/data/fixtures/pt-BR.json +1 -1
  21. data/lib/polytypo/data/fixtures/pt-PT.json +1 -1
  22. data/lib/polytypo/data/fixtures/ru.json +1 -1
  23. data/lib/polytypo/data/fixtures/sv.json +1 -1
  24. data/lib/polytypo/data/fixtures/tr.json +1 -1
  25. data/lib/polytypo/data/fixtures/uk.json +1 -1
  26. data/lib/polytypo/data/locales/cs.json +56 -37
  27. data/lib/polytypo/data/locales/de-CH.json +43 -34
  28. data/lib/polytypo/data/locales/de-DE.json +47 -32
  29. data/lib/polytypo/data/locales/el.json +51 -1
  30. data/lib/polytypo/data/locales/en-GB.json +56 -4
  31. data/lib/polytypo/data/locales/en-US.json +68 -4
  32. data/lib/polytypo/data/locales/es.json +66 -29
  33. data/lib/polytypo/data/locales/fi.json +77 -7
  34. data/lib/polytypo/data/locales/fr-CA.json +63 -36
  35. data/lib/polytypo/data/locales/fr.json +70 -41
  36. data/lib/polytypo/data/locales/it.json +65 -22
  37. data/lib/polytypo/data/locales/nl.json +61 -29
  38. data/lib/polytypo/data/locales/pl.json +61 -28
  39. data/lib/polytypo/data/locales/pt-BR.json +53 -28
  40. data/lib/polytypo/data/locales/pt-PT.json +55 -28
  41. data/lib/polytypo/data/locales/registry.json +1 -1
  42. data/lib/polytypo/data/locales/ru.json +45 -30
  43. data/lib/polytypo/data/locales/sv.json +63 -4
  44. data/lib/polytypo/data/locales/tr.json +72 -32
  45. data/lib/polytypo/data/locales/uk.json +55 -27
  46. data/lib/polytypo/data/rules/modes.md +243 -11
  47. data/lib/polytypo/data/rules/order.json +1 -1
  48. data/lib/polytypo/data/schema/fixtures.schema.json +12 -1
  49. data/lib/polytypo/engine/pipeline.rb +29 -0
  50. data/lib/polytypo/modes/markdown.rb +51 -0
  51. data/lib/polytypo/modes/runner.rb +33 -3
  52. data/lib/polytypo/version.rb +1 -1
  53. data/lib/polytypo.rb +27 -12
  54. metadata +2 -1
@@ -15,21 +15,7 @@
15
15
  "elisionIdioms": [],
16
16
  "elisionClitics": {
17
17
  "before": [],
18
- "after": [
19
- "nin",
20
- "nın",
21
- "de",
22
- "da",
23
- "te",
24
- "ye",
25
- "yle",
26
- "nı",
27
- "dan",
28
- "den",
29
- "lik",
30
- "nci",
31
- "üm"
32
- ]
18
+ "after": ["nin", "nın", "de", "da", "te", "ye", "yle", "nı", "dan", "den", "lik", "nci", "üm"]
33
19
  }
34
20
  },
35
21
  "dash": {
@@ -48,25 +34,79 @@
48
34
  "beforePunctuation": [],
49
35
  "narrowBeforePunctuation": [],
50
36
  "afterShortWords": [],
51
- "abbreviations": [
52
- "Kur. Bşk.",
53
- "Nö. Sb."
54
- ],
55
- "beforeUnits": [
56
- "mm",
57
- "cm",
58
- "km",
59
- "kg",
60
- "mg",
61
- "hl",
62
- "m²",
63
- "cm²",
64
- "°C",
65
- "ton"
66
- ],
37
+ "abbreviations": ["Kur. Bşk.", "Nö. Sb."],
38
+ "beforeUnits": ["mm", "cm", "km", "kg", "mg", "hl", "m²", "cm²", "°C", "ton"],
67
39
  "beforeNumber": [],
68
40
  "beforeWord": [],
69
41
  "afterSymbols": [],
70
42
  "initialBinding": "none"
71
- }
43
+ },
44
+ "sources": [
45
+ {
46
+ "rule": "quotes",
47
+ "cite": "Türk Dil Kurumu, Yazım Kuralları, «Noktalama İşaretleri (Açıklamalar)», «Tek Tırnak İşareti»: «Tırnak içinde verilen cümlenin içinde yeniden tırnağa alınması gereken bir sözü, ibareyi belirtmek için kullanılır: Edebiyat öğretmeni \"Şiirler içinde 'Han Duvarları' gibisi var mı?\" dedi…»",
48
+ "url": "https://tdk.gov.tr/icerik/yazim-kurallari/noktalama-isaretleri-aciklamalar/",
49
+ "note": "Kuralın kendisi: birinci derece çift tırnak, ikinci derece (tırnak içinde tırnak) tek tırnaktır. Bu sayfa işaretleri düz ASCII olarak sunduğu için kod noktalarını vermez; onlar ayrı kayıtta, Türk Dili'nin metin katmanından okunmuştur. BAĞIMSIZ DOĞRULAMA: Unicode Consortium, CLDR 'tr' delimiters — quotationStart U+201C, quotationEnd U+201D, alternateQuotationStart U+2018, alternateQuotationEnd U+2019 (https://github.com/unicode-org/cldr-json, cldr-misc-full/main/tr/delimiters.json). İki kaynak aynı dört kod noktasında birleşiyor. elisionIdioms BOŞ: Türkçede kesme işaretiyle yazılan, iki yana boşluk alan bir elizyon deyimi (rock 'n' roll biçimi) yoktur. elisionClitics.before BOŞ: Türkçe ekleri sona getirir, kesme işaretinden ÖNCEki dizi her zaman sözcüğün kendisidir. elisionClitics.after DOLU ve gerekçesi ayrı kayıttadır. Erişim: 23.09.2026."
50
+ },
51
+ {
52
+ "rule": "quotes",
53
+ "cite": "Şener Mete, «Tırnak İşareti Üzerine», Türk Dili (Türk Dil Kurumu dergisi), s. 31, dipnot 1 «TDK Yazım Kılavuzu»: «TDK Yazım Kılavuzu’na göre tırnak işaretinin ( “ ” ) kullanıldığı yerler şöyle sıralanmıştır»; «Bir de Tek Tırnak İşareti ( ‘ ’ ) vardır ki o da tırnak içinde verilen ve yeniden tırnağa alınması gereken bir sözü belirtmek için kullanılır: “Atatürk henüz ‘Gazi Mustafa Kemal Paşa’ idi.”»",
54
+ "url": "https://tdk.gov.tr/wp-content/uploads/2012/01/31-41.pdf",
55
+ "note": "KOD NOKTALARI DOSYANIN KENDİ BAYTLARINDAN OKUNDU, GÖZLE DEĞİL: PDF'in metin katmanı pdftotext ile çıkarıldı ve parantez içindeki işaretler U+201C U+201D ile U+2018 U+2019 olarak sayıldı; aynı paragrafın örneğinde de çift tırnak U+201C…U+201D, iç içe tek tırnak U+2018…U+2019 olarak geçiyor. Bu, yazarın kendi tercihi değil: pasajın dipnotu doğrudan «TDK Yazım Kılavuzu»dur, yani makale Kılavuz'un kuralını işaretleriyle birlikte aktarmaktadır. KAYNAK RÜTBESİ, açıkça: Türk Dili TDK'nin kendi dergisidir ama bir yazar makalesidir; burada yüklenen iş yalnızca Kılavuz'un işaretlerinin HANGİ kod noktaları olduğunu tespit etmektir, kuralın kendisini kurmak değil — kural TDK'nin kendi Yazım Kuralları sayfasından alınmıştır (ayrı kayıt). tdk.gov.tr sayfaları düz ASCII işaretlerle (U+0022, U+0027) sunulduğu ve locale.schema.json'ın singleChar tanımı bu ikisini açıkça yasakladığı için web sayfası ilke olarak bu değerin kaynağı olamaz. #51'DE BİLDİRİLEN «…» / »…« ÇÜRÜTÜLDÜ: aynı makale s. 32'de yan tırnağı «Bazı yazarlar, çift tırnak işaretini üstten değil yandan kullanmayı uygun görmüşlerdir» diye tanıtır — kurumun normu değil, bazı yazarların tercihidir. Metin katmanında U+00AB ile U+00BB toplam birer kez geçerken U+201C 60, U+201D 56 kez geçmektedir. innerSpace = «none»: hiçbir kaynak tırnak içinde boşluk istemiyor ve Kılavuz'un örnekleri bitişiktir. Erişim: 23.09.2026."
56
+ },
57
+ {
58
+ "rule": "quotes",
59
+ "cite": "Türk Dil Kurumu, Yazım Kuralları, «Kesme İşareti»: «Özel adlara getirilen iyelik, durum ve bildirme ekleri kesme işaretiyle ayrılır: Kurtuluş Savaşı'nı, Atatürk'üm, Türkiye'mizin»; «Kısaltmalara getirilen ekleri ayırmak için konur: TBMM'nin, TDK'nin, BM'de, ABD'de, TV'ye»; «Sayılara getirilen ekleri ayırmak için konur: 1985'te, 8'inci madde, 2'nci kat; 7,65'lik, 9,65'lik, 657'yle»; aynı bölümün örneklerinden ayrıca: «Batı'da», «Sultan Ana'nın», «Çanakkale Boğazı'nın», «Yurdakul'dan», «Cebrail'den»",
60
+ "url": "https://tdk.gov.tr/icerik/yazim-kurallari/kesme-isareti/",
61
+ "note": "quotes.elisionClitics.after listesinin gerekçesi. Bu giriş listenin TAM OLDUĞUNU değil, HER PARÇANIN AYRI AYRI belgelendiğini kanıtlar — it.json'daki ölçütün aynısı ve quotes.md §3.2'nin her girişten istediği de budur. Türkçe ek zincirleri kapalı bir küme değildir; bu alan zaten kapalı bir küme istemez. ÖLÇÜT, tek biçimde uygulandı: bir parça ancak TDK'nin BASTIĞI bir örnekte kesme işaretinden sonraki EN UZUN LETTER dizisinin TAMAMI ise alınmıştır, paradigma yeterli değildir — bu yüzden «TBMM'nin» örneği «nin»i kanıtlar ama «in»i kanıtlamaz, ve «Atatürk'üm» örneği «üm»ü verir, «ü»yü değil. KABUL EDİLEN YANLIŞ POZİTİF, İngilizce «s» ile eşit sayılmadan: listelenen bir parça, içeriği TAMAMEN o parça olan ve satır içi bir span sınırına DAYANAN bir alıntıyı ek sanar — «<em>'de'</em>» biçimi, bir fikstürle sabitlenmiştir. Risk Türkçede İngilizceden YÜKSEKTİR, çünkü «da/de» ayrıca ayrı yazılan bir bağlaçtır; iki ölçülen hafifletici kaydedilir: TDK'nin kendi anma biçimi başta kısa çizgi taşır («-lık'la»; «Bulunma Durumu Eki -da / -de / -ta / -te'nin Yazılışı» başlığı) ve kısa çizgi LETTER olmadığı için dizi boş kalır ve veto çalışamaz; ayrıca «Bağlaç Olan da, de'nin Yazılışı» sayfası bağlacı tırnaksız, düz «da / de» yazar. BÜYÜK HARF KAYBI, adıyla ve fikstürle: «TBMM'NİN» eşleşmez, çünkü katlama yalnızca ilk kod noktasını ve yalnızca ASCII A-Z'yi kapsar — «N»→«n» katlanır ama kuyrukta U+0130 ≠ U+0069 olur; «TBMM'NIN» da U+0049 ≠ U+0069 ile düşer. Katlamayı genişletmek ARCHITECTURE.md §4.4'ün açıkça yasakladığı yerdir (noktasız ı, U+0131), bu yüzden kayıp giderilmez, KABUL EDİLİR. BİLEREK DIŞARIDA BIRAKILANLAR, kanıtlı oldukları hâlde: «a», «e», «i» (Samsun'a, Fatih Sultan Mehmet'e, Kâzım Karabekir'i) — tek harfler harf ADLARIdır ve TDK'nin kendi «a'dan z'ye» örneği harf alıntılama biçimini basar; bedeli açıkça söylenir, ünsüzle biten adlarda yönelme eki span sınırında onarılmadan kalır. «ya», «un», «in», «inci», «tan» da kanıtlıdır ama çıplak sözcük olarak çakışırlar. «dan», «den», «lik», «nci» ve «üm» LİSTEYE ALINDI: hepsi yukarıdaki cite'ta basılı örneklerle kanıtlıdır (Yurdakul'dan, Cebrail'den, 7,65'lik, 2'nci, Atatürk'üm), hiçbiri çıplak bir Türkçe sözcükle çakışmaz, dolayısıyla maliyetleri zaten kabul edilmiş «de/da» ikilisinden DÜŞÜKTÜR. Böylece liste, ölçüte göre kanıtlı olup çakışmayan her parçayı içerir ve dışarıda kalanların hepsi aşağıda gerekçesiyle adlandırılmıştır — keyfî bir alt küme değildir. KAPSAM BOŞLUĞU, tek tek: «ta», «ten», «nun», «nün», «ler», «lar» okunan sayfalarda tam dizi olarak basılmamıştır, dolayısıyla «Irak'ta», «Oslo'nun», «TBMM'ler» span sınırında onarılmaz; zincirli ekler («Türkiye'mizin» → dizi «mizin», «Devleti'ndeki» → «ndeki») hiçbir kısa girişle eşleşmez. Liste YALNIZCA REDDEDER: quotes'un kuracağı bir eşleştirmeyi engellemekten başka bir şey yapamaz. Erişim: 23.09.2026."
62
+ },
63
+ {
64
+ "rule": "dashes",
65
+ "cite": "Türk Dil Kurumu, Yazım Kuralları, «Noktalama İşaretleri (Açıklamalar)», «Kısa Çizgi» 2: «Cümle içinde ara sözleri veya ara cümleleri ayırmak için ara sözlerin veya ara cümlelerin başına ve sonuna konur, bitişik yazılır: Küçük bir sürü -dört inekle birkaç koyun- köye giren geniş yolun ağzında durmuştu.»; «Uzun Çizgi»: «Yazıda satır başına alınan konuşmaları göstermek için kullanılır. Buna konuşma çizgisi de denir.» UYARI: «Konuşmalar tırnak içinde verildiğinde uzun çizgi kullanılmaz.»",
66
+ "url": "https://tdk.gov.tr/icerik/yazim-kurallari/noktalama-isaretleri-aciklamalar/",
67
+ "note": "dash.parenthetical = «none» gerekçesi, ve bu bir SUSKUNLUK DEĞİL, İFADE EDİLEMEYEN BİR NORMDUR. İki iş ayrı ayrı sınandı: uzun çizgi YALNIZCA konuşma çizgisidir — bölümün tek kuralı ve tek UYARI'sı budur, ara sözle ilgili hiçbir hükmü yoktur. Ara sözü ayıran işaret KISA ÇİZGİ'dir ve kaynak onu «bitişik yazılır» diye niteler. locale.schema.json'ın enum'u yalnızca em/en uzunluklarını tanır; kısa çizgi ne U+2014 ne U+2013'tür, dolayısıyla Türkçenin gerçek geleneği bu alanda İFADE EDİLEMEZ. «en-tight» yazmak TDK'ye aykırı bir dönüşüm üretirdi — yaklaşık bir değer yerine kuralı sessiz bırakmak seçildi. Sonuç, kısa çizginin U+002D mi U+2010 mu olduğundan bağımsızdır: hiçbiri en/em değildir. tr, her iki alanı da «none» olan ÜÇÜNCÜ locale'dir — el ve es'ten sonra; dashes kuralı Türkçe için onlarda olduğu gibi kanıtlanabilir bir total no-op'tur (dashes.md §6, «el» bölümü). Erişim: 23.09.2026."
68
+ },
69
+ {
70
+ "rule": "ranges",
71
+ "cite": "Türk Dil Kurumu, Yazım Kuralları, «Noktalama İşaretleri (Açıklamalar)», «Kısa Çizgi» 7: «Arasında, ve, ile, ila, …-den …-e anlamlarını vermek için kelimeler veya sayılar arasında kullanılır: Aydın-İzmir yolu, Türk-Alman ilişkileri, Ural-Altay dil grubu, 09.30-10.30, 1914-1918 Birinci Dünya Savaşı, Türkçe-Fransızca Sözlük vb.»",
72
+ "url": "https://tdk.gov.tr/icerik/yazim-kurallari/noktalama-isaretleri-aciklamalar/",
73
+ "note": "dash.range = «none» gerekçesi. Aralık normludur — ve bu, alanın boş olmasının nedenidir değil, tam tersine nedeniyle birlikte kaydedilmesinin nedenidir: işaret KISA ÇİZGİ'dir, TDK hiçbir yerde sayılar arasında yarım veya tam kare çizgi istemez. Doğrulanmış bir en/em geleneği YOKTUR, dolayısıyla «none» doğru içeriktir (ranges.md §2). «ranges» kuralı zaten varsayılan olarak kapalıdır (spec 0.5.0), bu alan bir çağıran açıkça açmadıkça atıldır. Erişim: 23.09.2026."
74
+ },
75
+ {
76
+ "rule": "ellipsis",
77
+ "cite": "Türk Dil Kurumu, Yazım Kuralları, «Noktalama İşaretleri (Açıklamalar)», «Üç Nokta», UYARI: «Ünlem ve soru işaretinden sonra üç nokta yerine iki nokta konulması yeterlidir: Gök ekini biçer gibi!.. Başaklar daha dolmadan. (Tarık Buğra) / Nasıl da akşam oldu?.. Nasıl da yavrucaklar sustu?.. Nasıl da serçecikler yuvalarına sığındı?.. (Necip Fazıl Kısakürek)»",
78
+ "url": "https://tdk.gov.tr/icerik/yazim-kurallari/noktalama-isaretleri-aciklamalar/",
79
+ "note": "abbreviatedAfterTerminal = true. Türkçe, Rusça ve Ukraynacadan sonra bu değeri alan üçüncü locale'dir; araştırma bunun TERSİNİ beklediği için özellikle sınandı. SINIFLANDIRMA, açıkça: «yeterlidir» «yazılır» demek değildir, yani izin verilen bir biçimdir, mutlak bir buyruk değil. Buna rağmen «true» seçilmiştir ve gerekçe tercih değil ÖLÇÜMDÜR: bu alan iki yönlüdür, «false» tarafsız değildir. polytypo 1.5.0 ile ölçüldü — abbreviatedAfterTerminal'ı false olan bir locale'de «Nasıl da akşam oldu?..» çıktısı «Nasıl da akşam oldu?…» olur, yani Kılavuz'un kendi bastığı biçim bozulur; true olan bir locale'de aynı girdi sabit noktadır ve «?...» de ona normalleştirilir. Kılavuz'un örneklerinin tamamı iki noktalıdır ve «?...» biçiminde tek bir TDK örneği yoktur, dolayısıyla iki yönden yalnızca biri kaynakla uyumludur. BEDEL, gizlenmeden: ellipsis.md §3 adım 4→6 «?...» girdisini «?..» olarak YENİDEN YAZAR, yani UYARI'nın istediğinden bir adım ileri gider. Erişim: 23.09.2026."
80
+ },
81
+ {
82
+ "rule": "hyphen",
83
+ "cite": "Türk Dil Kurumu, Yazım Kuralları, «Hece Yapısı ve Satır Sonunda Kelimelerin Bölünmesi»: «Türkçede satır sonunda kelimeler bölünebilir fakat heceler bölünemez.»; «Satıra sığmayan kelimeler bölünürken satır sonuna kısa çizgi (-) konur.»; «Ayırmada satır sonunda ve satır başında tek harf bırakılmaz.»; «Kesme işareti satır sonuna geldiğinde yalnız kesme işareti kullanılır; ayrıca çizgi kullanılmaz.»",
84
+ "url": "https://tdk.gov.tr/icerik/yazim-kurallari/hece-yapisi-ve-satir-sonunda-kelimelerin-bolunmesi/",
85
+ "note": "Üç boş listenin gerekçesi, ve bu OLUMLU bir bulgudur, bir eksiklik değil. Türkçe satır sonunda bölmeyi yalnızca hoş görmez, NORMLAR: bölme noktasına kısa çizgi konur. Bu, U+2011 (bölünmez kısa çizgi) talebinin tam tersidir. Sayfanın tamamı ile «Kısa Çizgi» bölümünün bütün UYARI'ları okundu; hâlihazırda kısa çizgi taşıyan bir biçimin satır sonunda bölünmesini yasaklayan tek bir kural yoktur. Boş listeler kuralı Türkçe için kanıtlanabilir bir no-op yapar (hyphen.md §2). Erişim: 23.09.2026."
86
+ },
87
+ {
88
+ "rule": "nbsp",
89
+ "cite": "Türk Dil Kurumu, Yazım Kuralları, «Noktalama İşaretleri (Açıklamalar)», giriş hükmü: «Noktalama işaretlerinden nokta, virgül, noktalı virgül, iki nokta, üç nokta, soru, ünlem, tırnak, ayraç ve kesme işaretleri ait oldukları kelimelere bitişik olarak yazılır ve kesme dışındaki işaretlerden sonra bir harf boşluğu ara verilir.»",
90
+ "url": "https://tdk.gov.tr/icerik/yazim-kurallari/noktalama-isaretleri-aciklamalar/",
91
+ "note": "beforePunctuation ve narrowBeforePunctuation boş. Bu SESSİZLİK DEĞİL, AÇIK BİR REDDİR: işaretler ait oldukları kelimeye bitişiktir ve boşluk işaretten SONRA gelir, dolayısıyla Fransızca «mot !» geleneği Türkçede ne U+00A0 ne U+202F olarak vardır. Aynı hüküm kesme işaretinin ayrıcalığını da adlandırır: ondan sonra boşluk bırakılmaz. initialBinding = «none»: «Kısaltmalar» ve «Büyük Harflerin Kullanıldığı Yerler» sayfaları okundu, kişi adı ve soyadı baş harflerinin nasıl aralanacağına dair TDK'de bir kural bulunamadı — bu, olumlu bir dayanağa değil kaynağın suskunluğuna dayanan bir «none»'dur ve aradaki fark burada belirtilmektedir. afterShortWords boş: Türkçede tek harfli bağlaç veya edat yoktur ve TDK satır sonunda asılı kalan kısa sözcüklerle ilgili bir kural vermez; «satır sonunda ve satır başında tek harf bırakılmaz» kuralı HECE bölmeye ilişkindir, sözcüklere değil, ve aynı olgu değildir. beforeNumber ve beforeWord boş: hiçbir kısaltmayı ardından gelen bir sayıya veya kelimeye bağlayan kural bulunamadı. Erişim: 23.09.2026."
92
+ },
93
+ {
94
+ "rule": "nbsp",
95
+ "cite": "Türk Dil Kurumu, Kısaltmalar Dizini: «Kur. Bşk. Kurmay Başkanı, Başkanlığı»; «Nö. Sb. Nöbetçi subayı»",
96
+ "url": "https://tdk.gov.tr/icerik/yazim-kurallari/kisaltmalar-dizini/",
97
+ "note": "abbreviations listesi. İki giriş, dizinin PDF'inden iç boşluklarıyla birlikte harfi harfine alındı. ÖLÇÜT, tek biçimde uygulandı: yalnızca HER İKİ parçası da noktalı kısaltma olan girişler. Bilerek dışarıda bırakıldı: «Osm. T.», «SEFD Bşk.», «TOBB ETÜ», «TÜRKİYE KOOP» — her birinde en az bir parça akronimdir, noktalı kısaltma değil. KAPSAM UYARISI: dizin baştan sona okunmadı, bu yüzden liste doğrulanmış bir ALT KÜMEDİR, tam sayım değildir; eksik bir giriş yanlış bir giriş değildir, N4 yalnızca listedekini bağlar. TDK'nin kendi yazımıyla ilgili, dizinden doğrudan okunan iki düzeltme: «MÖ» ve «MS» noktasız yazılır, «T.C.» ise iç boşluksuzdur — «M.Ö.» ve «M. Ö.» biçimleri TDK'nin biçimleri değildir ve bu alana girmezler. N4 tam eşleşme yapar, hiçbir büyük-küçük harf esnekliği yoktur (nbsp.md §3.6); N1, N2 ve N3 bu locale'de boş olduğundan «önce talep eden kazanır» çatışması da yoktur. Erişim: 23.09.2026."
98
+ },
99
+ {
100
+ "rule": "nbsp",
101
+ "cite": "Türk Dil Kurumu, Sıkça Sorulan Sorular, «°C \"santigrat derece\" işaretiyle sayı arasında boşluk bırakılır mı?»: «Sayı ve ölçü birimi kısaltmaları aralarında boşluk bırakılır. Tıpkı 5 m, 20 kg, 350 ton örneklerinde olduğu gibi, °C \"santigrat derece\" işaretiyle sayı arasında boşluk bırakılır: 15 °C gibi.»; Türk Dil Kurumu, Yazım Kuralları, «Kısaltmalar» 2: «Ölçü birimlerinin uluslararası kısaltmaları kullanılır: m (metre), mm (milimetre), cm (santimetre), km (kilometre), g (gram), kg (kilogram), l (litre), hl (hektolitre), mg (miligram), m² (metrekare), cm² (santimetrekare) vb.»",
102
+ "url": "https://tdk.gov.tr/icerik/sikca-sorulan-sorular/c-derece-santigrat-isaretiyle-sayi-arasinda-bosluk-birakilir-mi/",
103
+ "note": "beforeUnits. Rol paylaşımı başka locale'lerdeki gibi iki kuruma değil tek kuruma düşüyor: hangi dizgilerin uluslararası birim simgesi olduğunu «Kısaltmalar» 2, aralarında boşluk bırakıldığını SSS söylüyor. KAYNAK RÜTBESİ açıkça belirtiliyor: Sıkça Sorulan Sorular, Yazım Kılavuzu'nun altındadır ve boşluk hükmünün tek dayanağı odur. Tek harfli simgeler (m, g, l) fr.json ve pl.json uygulamasına uyularak atlandı. «ton» ayrıca açıklanıyor: uluslararası bir simge değil bir kelimedir ve yalnızca SSS'nin düzyazı örneğinden («350 ton») gelir — listede tutulmasının nedeni N5'in sağ sınır testinin onu «tonluk» içinde eşleştirmemesi ve hiçbir şey uydurulmamış olmasıdır. «cm» ile «cm²» birlikte listelenebilir, çünkü aynı konumda EN UZUN EŞLEŞME KAZANIR (nbsp.md §3.7) — «15 cm²» girdisinde «cm» değil «cm²» eşleşir. «%» BİLEREK YOK: TDK yüzde işaretini sayıdan ÖNCE ve bitişik yazar (ayrı kayıt). N5 yalnızca var olan bir boşluğu DÖNÜŞTÜRÜR, asla eklemez (nbsp.md §3.7 madde 3), bu yüzden «20kg» dokunulmadan kalır. Erişim: 23.09.2026."
104
+ },
105
+ {
106
+ "rule": "nbsp",
107
+ "cite": "Türk Dil Kurumu, Yazım Kuralları, «Sayıların Yazılışı»: «Yüzde ve binde işaretleri yazılırken sayılarla işaret arasında boşluk bırakılmaz: %25, ‰50»",
108
+ "url": "https://tdk.gov.tr/icerik/yazim-kurallari/sayilarin-yazilisi/",
109
+ "note": "afterSymbols boş, ve gerekçesi OLUMSUZ BİR KANITTIR, suskunluk değil. Türkçede yüzde işareti sayıdan ÖNCE gelir ve bitişik yazılır; bu, pl.json'un «%» girişinin hem konum hem boşluk bakımından tam tersidir. beforeUnits'e giremez, çünkü orada işaret sayıdan SONRA beklenir; afterSymbols'e de giremez, çünkü N6 sayıdan önce VAR OLAN bir boşluk arar ve norm o boşluğu yasaklar (nbsp.md §3.8). N5 ve N6 yalnızca dönüştürür, hiç eklemez, dolayısıyla «%25» biçimi her iki durumda da dokunulmadan kalır ve boş liste hiçbir şey kaybettirmez. «§» de bu listede yok: TDK bu işaret için bir kural vermiyor. Erişim: 23.09.2026."
110
+ }
111
+ ]
72
112
  }
@@ -26,21 +26,8 @@
26
26
  "abbreviatedAfterTerminal": true
27
27
  },
28
28
  "hyphen": {
29
- "prefixes": [
30
- "будь-",
31
- "казна-",
32
- "хтозна-",
33
- "бозна-"
34
- ],
35
- "suffixes": [
36
- "-бо",
37
- "-но",
38
- "-от",
39
- "-то",
40
- "-таки",
41
- "-будь",
42
- "-небудь"
43
- ],
29
+ "prefixes": ["будь-", "казна-", "хтозна-", "бозна-"],
30
+ "suffixes": ["-бо", "-но", "-от", "-то", "-таки", "-будь", "-небудь"],
44
31
  "compounds": [
45
32
  "вид-во",
46
33
  "гр-н",
@@ -67,12 +54,7 @@
67
54
  "beforePunctuation": [],
68
55
  "narrowBeforePunctuation": [],
69
56
  "afterShortWords": [],
70
- "abbreviations": [
71
- "і т. д.",
72
- "і т. ін.",
73
- "та ін.",
74
- "куб. см"
75
- ],
57
+ "abbreviations": ["і т. д.", "і т. ін.", "та ін.", "куб. см"],
76
58
  "beforeUnits": [
77
59
  "%",
78
60
  "га",
@@ -95,12 +77,58 @@
95
77
  "р."
96
78
  ],
97
79
  "beforeNumber": [],
98
- "beforeWord": [
99
- "акад.",
100
- "доц.",
101
- "проф."
102
- ],
80
+ "beforeWord": ["акад.", "доц.", "проф."],
103
81
  "afterSymbols": [],
104
82
  "initialBinding": "chain"
105
- }
83
+ },
84
+ "sources": [
85
+ {
86
+ "rule": "quotes",
87
+ "cite": "Український правопис (2019), схвалений Кабінетом Міністрів України (Постанова № 437 від 22.05.2019), § 164 «ЛАПКИ (« », “ ”, „ “, рідше „ ”)», п. 3: «У функції перших рекомендовано вживати кутові лапки, або «лапки-ялинки» («…»), у функції внутрішніх — «лапки-лапки» (“…” та ін.): «Це мій “Кобзар”», — сказав він»; там само: «На письмі (у рукописних текстах) «лапки-лапки» традиційно використовують у формі „…“»",
88
+ "url": "https://www.ulif.org.ua/system/files/pravopus-new.pdf",
89
+ "note": "Зовнішні лапки — U+00AB і U+00BB, внутрішні — U+201C і U+201D. Код кожного знака встановлено за друкованим зображенням с. 247 офіційного видання, а не за текстовою конверсією: у прикладі § 164 п. 3 обидва внутрішні знаки підняті до верху рядка, відкривальний має форму 6, закривальний — форму 9. ЦЕ СПРОСТОВУЄ поширене припущення, що українська бере всередину „…“: заголовок § 164 справді дозволяє чотири пари, але саме п. 3 відносить „…“ до РУКОПИСНИХ текстів, а для друкованого рекомендує “…”. innerSpace = «none»: жодне джерело не вимагає відступу всередині лапок. Третій рівень вкладення в locale.schema.json невиразний. Звірено 18.09.2026."
90
+ },
91
+ {
92
+ "rule": "dashes",
93
+ "cite": "Український правопис (2019), § 161 «ТИРЕ (—)», I, п. 10—11 і Примітка 2: тире ставимо перед відокремленим зворотом або вставленою конструкцією — «Топольський — молодий чоловік, але — на думку пана посла — незвичайно талановитий і солідний» (О. Маковей)",
94
+ "url": "https://www.ulif.org.ua/system/files/pravopus-new.pdf",
95
+ "note": "Обґрунтовує dash.parenthetical = «em-spaced». ДОВЖИНА встановлена прямо: заголовок параграфа — «ТИРЕ (—)», тобто U+2014, і цей самий знак стоїть у кожному прикладі §§ 161, 166, 167. ВІДБИВКА встановлена протиставленням, а не прозовим формулюванням: правопис ніде не пише «тире відбивається пробілами», але Примітка до п. 14 каже «тире ставимо без відступів між знаками» саме для випадку між цифрами, а «без відступів» має сенс лише як відхилення від відбитого за замовчуванням. Це єдиний запис цього файлу, що спирається на висновок із тексту джерела, і він позначений як такий свідомо. Звірено 18.09.2026."
96
+ },
97
+ {
98
+ "rule": "ranges",
99
+ "cite": "Український правопис (2019), § 161 «ТИРЕ (—)», I, п. 14, Примітка: «Між цифрами в таких випадках тире ставимо без відступів між знаками: у 2010—2018 роках; пам'ятки української мови XVI—XVIII ст.; на сторінках 1—10; у 1—4 томах, але, напр.: наприкінці XX — на початку XXI ст.»",
100
+ "url": "https://www.ulif.org.ua/system/files/pravopus-new.pdf",
101
+ "note": "Обґрунтовує dash.range = «em-tight»: довге тире U+2014 без відбивки. Протиставлення в самій Примітці («на сторінках 1—10» без відступів, але «наприкінці XX — на початку XXI ст.» з відступами) показує, що безвідступна форма стосується рівно того, що робить правило ranges. Правило вимкнене за замовчуванням (spec 0.5.0). Звірено 18.09.2026."
102
+ },
103
+ {
104
+ "rule": "ellipsis",
105
+ "cite": "Український правопис (2019), § 162 «ТРИ КРАПКИ, АБО КРАПКИ (…)», Примітка: «у постпозиції — після знака питання і знака оклику — ставимо дві крапки: Стражники на людей стріляли, це відомо, а щоб селяни?.. (К. Гордієнко); Встає народ, гудуть мости, Рокочуть ріки ясноводі!.. (М. Рильський)»; § 166 «КОМБІНОВАНЕ ВЖИВАННЯ РОЗДІЛОВИХ ЗНАКІВ», п. 2, що перелічує допустимі поєднання як «…?; …!; ?..; !..»",
106
+ "url": "https://www.ulif.org.ua/system/files/pravopus-new.pdf",
107
+ "note": "Обґрунтовує abbreviatedAfterTerminal = true. Українська — друга після російської локаль із цим значенням, і це перевірено окремо за двома параграфами, а не перенесено за аналогією зі спорідненої мови. Звірено 18.09.2026."
108
+ },
109
+ {
110
+ "rule": "hyphen",
111
+ "cite": "Український правопис (2019), § 64 «Технічні правила переносу», п. 4: «Не можна розривати умовні (графічні) скорочення на зразок вид-во, і т. д., і т. ін., та ін., т-во тощо»; § 62, п. 2: «У графічних скороченнях пропущену середню частину слова позначаємо дефісом: вид-во (видавництво), гр-н (громадянин), ін-т (інститут), р-н (район), ун-т (університет), ф-ка (фабрика)»",
112
+ "url": "https://www.ulif.org.ua/system/files/pravopus-new.pdf",
113
+ "note": "Це єдина частина hyphen для української, де НЕРОЗРИВНІСТЬ САМА Є НОРМОЮ, а не рішенням проєкту: § 64 п. 4 прямо забороняє розривати цей клас, а § 62 п. 2 закриває його літеральним переліком, бо «тощо» в § 64 залишає клас відкритим. Українська цим відрізняється від польської, де PWN [196] поділ у місці дефіса саме ПРИПИСУЄ, через що pl.json має три порожні списки. Звірено 18.09.2026."
114
+ },
115
+ {
116
+ "rule": "hyphen",
117
+ "cite": "Український правопис (2019), § 42 «Прийменники», п. 2: «З дефісом пишемо складені прийменники, утворені з простих прийменників з, із та інших прийменників: з-за (із-за), з-над, з-перед, з-під (із-під), з-поза, з-поміж, з-понад, з-попід, з-посеред, з-проміж»; § 44 «Частки», п. 3: «З дефісом пишемо: 1) частки -бо, -но, -от, -то, -таки … 2) частки будь-, -будь, -небудь, казна-, хтозна-, бозна- із займенниками і прислівниками»",
118
+ "url": "https://www.ulif.org.ua/system/files/pravopus-new.pdf",
119
+ "note": "Ці параграфи нормативні для СКЛАДУ форм, але не для заборони переносу, і змішувати це не можна — так само, як у ru.json. §§ 63 і 64 прочитано повністю: § 63 нормує поділ за складами, § 64 забороняє розривати ініціали, назви мір, нарощення й графічні скорочення, але про дефіс у формах §§ 42 і 44 не каже нічого. Отже, для цих класів зв'язування через U+2011 — рішення проєкту (docs/PLAN.md §3.3), а не орфографічна норма, і це записано прямо. НЕ ВНЕСЕНО, свідомо: § 64 п. 3 («Граматичні закінчення, поєднані із цифрами дефісом, не можна відривати й переносити: 2-й, 4-го, 10-му») нормативно нерозривний, але це форма «цифра-дефіс-літера»; чи є це членством у hyphen.suffixes, чи вже алгоритмом, має вирішити spec-guardian. Звірено 18.09.2026."
120
+ },
121
+ {
122
+ "rule": "nbsp",
123
+ "cite": "Український правопис (2019), § 64 «Технічні правила переносу», п. 1: «Не можна переносити прізвища, залишаючи в кінці попереднього рядка ініціали або інші умовні скорочення, що належать до них: Т. Г. Шевченко (не Т. Г. // Шевченко), гр. Іваненко, акад. (доц., проф.) Гончаренко, п. Гнатюк»; п. 4: «Не можна розривати умовні (графічні) скорочення на зразок вид-во, і т. д., і т. ін., та ін., т-во тощо»; п. 5: «Не можна переносити в наступний рядок розділові знаки (крім тире), дужку або лапки, що закривають попередній рядок»",
124
+ "url": "https://www.ulif.org.ua/system/files/pravopus-new.pdf",
125
+ "note": "§ 64 — ЗАКРИТИЙ перелік із п'яти пунктів, і це важливо не лише тим, що він містить, а й тим, чого в ньому немає. (1) initialBinding = «chain»: п. 1 наводить послідовність ДВОХ ініціалів і не дає прикладу одного ініціала перед прізвищем. (2) beforeWord: з п. 1 взято лише «акад.», «доц.», «проф.». «п.» і «гр.» свідомо НЕ внесено за тією самою дисципліною, з якої ru.json виключає «г.»: «п.» — це також «пункт» («п. 3»), «гр.» — також «градус» і «графа», а літеральний список цих напрямків не розрізняє. (3) beforePunctuation і narrowBeforePunctuation порожні: п. 5 прив'язує розділовий знак до ПОПЕРЕДНЬОГО слова — це заперечення, а не мовчання. (4) afterShortWords ПОРОЖНІЙ, і це підтверджена відсутність норми: перелік § 64 закритий, правила про однобуквені прийменники в ньому немає, а в самому правописі «і», «у», «в», «з» регулярно стоять у кінці рядка. Список ru.json сюди НЕ переноситься за аналогією, хоча мови споріднені. (5) afterSymbols і beforeNumber порожні: конструкцій «символ + число» і «скорочення + число» § 64 не містить. Звірено 18.09.2026."
126
+ },
127
+ {
128
+ "rule": "nbsp",
129
+ "cite": "Український правопис (2019), § 64, п. 2: «Не можна відривати скорочені назви мір від цифр, до яких вони належать: 2008 р. (не 2008 // р.), 150 га (не 150 // га), 20 см³ або 20 куб. см, 5 г (не 5 // г)»; § 62: «Скорочені назви одиниць вимірювання пишемо без крапок: Б — байт, Вт — ват, г — грам, га — гектар, год — година, дм — дециметр, кБ — кілобайт, кВт — кіловат, кг — кілограм, км — кілометр, л — літр, м — метр, мм — міліметр, с — секунда, см — сантиметр, т — тонна, хв — хвилина, ц — центнер»; BIPM, The International System of Units (SI), 9th ed., concise summary: «A single space is always left between the number and the unit»",
130
+ "url": "https://www.bipm.org/documents/20126/41483022/SI-Brochure-9-concise-EN.pdf",
131
+ "note": "Розподіл ролей тут ІНШИЙ, ніж у pl, fi, sv і en-US: там BIPM дає перелік символів, а національне джерело — конвенцію відступу; для української національне джерело дає і те, і те, бо § 64 п. 2 нормує саме невідривність. BIPM наведено як підтвердження загального принципу, а не як несуча цитата. Однобуквені позначення (г, л, м, с, т, ц, Б) свідомо пропущено: правий кордон N5 вимагає лише, щоб cp[a+k] не належав ALNUM, і пробіл цю умову задовольняє, тож однобуквений запис зв'язував би звичайну прозу. «р.» внесено: на відміну від російського «г.», українське «р.» означає тільки «рік» і зв'язується вліво після числа. «%» внесено ЯК РІШЕННЯ, а не як цитата: його немає ні в § 62, ні в § 64, ні в скороченому викладі брошури BIPM (правило про % — у § 5.4.7 повного видання, яке не діставалося), але відбивка відсотка від числа є і в російській локалі, і в усіх інших локалях цього проєкту, а розбіжність тут дала б українській вужчу поведінку без жодної підстави в джерелі. Звірено 18.09.2026."
132
+ }
133
+ ]
106
134
  }
@@ -7,10 +7,12 @@ for all five runtimes and is parser-agnostic by construction: `parse5`, `nokogir
7
7
  `golang.org/x/net/html` and PHP's DOM disagree about almost everything this document does not
8
8
  forbid them from doing. `yaml` mode is parser-**free** rather than parser-agnostic, for the
9
9
  reason §3.8.1 measures.
10
- **Spec version:** 1.5.0 (0.1.0 for everything except §3.3's class-membership table rows for
10
+ **Spec version:** 1.7.0 (0.1.0 for everything except §3.3's class-membership table rows for
11
11
  `nbsp` and `apostrophe`, split in 1.2.0, §3.8, added in 1.3.0, §3.3's note on `quotes`'
12
12
  span-boundary elision veto reading the marker as a trigger, added in 1.4.0 — which changes no
13
- row of the table it follows — and §3.3's `CLOSEDELIM` entry, added in 1.5.0).
13
+ row of the table it follows — §3.3's `CLOSEDELIM` entry, added in 1.5.0, and §3.7.4 with the
14
+ amendments it carries to §3.1's definitions, §3.2's Model C, §3.5 step 3, §3.7.3's frontmatter
15
+ bullet, §3.8.6's accepted-cost paragraph, §5, §6 and §7, added in 1.7.0).
14
16
 
15
17
  ---
16
18
 
@@ -49,6 +51,10 @@ sees; the pipeline decides what to do with them.
49
51
  identified by its offsets in the **original source**, and those offsets are the only handle
50
52
  the mode layer keeps.
51
53
  - The **span sequence** `S₁ … Sₘ` is the processable spans in document order.
54
+ - A **text unit** is one span sequence and the array built from it (§3.5 step 2). A document has
55
+ exactly one, **except** in `markdown` with `frontmatterKeys` (§3.7.4, spec 1.7.0), where the
56
+ frontmatter block's spans form a second unit and the pipeline runs once per unit. The term is
57
+ defined here because §4 and §6 already used it informally for "the concatenation".
52
58
  - `text` mode is the degenerate case: one span covering the whole input, no skipped regions.
53
59
  Every statement below holds for it trivially.
54
60
 
@@ -88,7 +94,9 @@ each rule is behaving exactly as specified on the input it was given.
88
94
 
89
95
  **Model C — concatenation with an explicit boundary marker. Adopted.** The spans are
90
96
  concatenated with a **boundary marker** between each adjacent pair. The pipeline runs once, over
91
- the whole marker-separated array. Edits are then redistributed to spans by offset.
97
+ the whole marker-separated array — once per **text unit** (§3.1), which for every mode but
98
+ `markdown` with `frontmatterKeys` (§3.7.4) means once per document. Edits are then
99
+ redistributed to spans by offset.
92
100
 
93
101
  The marker gives the rules what Model A denies them — the knowledge that `'hi'` sits inside a
94
102
  larger quotation — while denying them what Model B wrongly grants: the belief that the last
@@ -365,7 +373,9 @@ reaches the edge-growth rule. `--` is the live carrier.)
365
373
  gives flow collections no spans — see `quotes.md` §3.2, which depends on the answer for
366
374
  nothing but states the obligation it creates: a rule must key off the marker, never off the
367
375
  mode.
368
- 3. Run the pipeline **once**, in `order.json` order, over that array.
376
+ 3. Run the pipeline **once**, in `order.json` order, over that array. A document with a second
377
+ text unit — `markdown` with `frontmatterKeys`, and only that (§3.1, §3.7.4) — repeats steps 1
378
+ to 4 for it. The two edit sets are disjoint, because no span of one unit lies inside the other.
369
379
  4. Each edit lies wholly within one span (§3.4). Map it back to source offsets.
370
380
  5. Emit the **original source bytes**, with those replacements applied and nothing else changed.
371
381
 
@@ -477,7 +487,9 @@ Two constraints on the throw:
477
487
  Skipped, exhaustively:
478
488
 
479
489
  - **frontmatter** — a metadata block at the very start of the document, delimited by `---`
480
- (YAML) or `+++` (TOML), skipped whole including its delimiters. Without it the closing `---`
490
+ (YAML) or `+++` (TOML), skipped whole including its delimiters. **Since 1.7.0 a caller may
491
+ name keys inside a YAML block to process — §3.7.4 — and the skip stands unchanged when they
492
+ do not.** Without it the closing `---`
481
493
  reads as a setext underline, `title: Une note` becomes a paragraph, and `fr` inserts a narrow
482
494
  no-break space before the colon of a machine-read field. That is a guaranteed false positive
483
495
  on the M4 corpus (PLAN.md §8), where every file opens with frontmatter;
@@ -512,6 +524,174 @@ lower-cases JSX names before matching will skip a component's children.
512
524
  Nesting follows the same rule as `html`: a skipped construct is skipped whole, including
513
525
  anything that looks processable inside it.
514
526
 
527
+ #### 3.7.4 Frontmatter by named keys — the one opt-out from §3.7.3
528
+
529
+ **Spec 1.7.0.** §3.7.3 skips the frontmatter block whole and its reason for doing so still holds.
530
+ It is also the one place where the default hides the sentence most readers of the page will see:
531
+ where frontmatter carries `title`, `description` and `summary`, those strings are the heading,
532
+ the `<title>`, the meta description and the card text, so the single most visible string on the
533
+ page is the one polytypo will not touch. Both reports that asked for this — polytypo/polytypo#13,
534
+ and the production integration in #26 over 196 MDX posts, independently — hand-rolled it instead,
535
+ and the first of them corrupted its own content doing so: its extractor knew double-quoted YAML
536
+ scalars and silently skipped every single-quoted one.
537
+
538
+ > **`markdown` mode takes an optional `frontmatterKeys` option: the mapping keys in the
539
+ > document's YAML frontmatter block whose scalar values are processable.** Absent means
540
+ > §3.7.3 unchanged — the block is skipped whole, delimiters included. An **empty list is
541
+ > legal** and yields no spans, as it does for `keys` (§3.8.2). A value that is not a list of
542
+ > strings throws `POLYTYPO_INVALID_OPTION`.
543
+
544
+ **Ratified by the operator as public contract, 2026-09-25**, under this name rather than by
545
+ widening `keys`. Three reasons, in the order that decided it: `keys` is **required** in `yaml`
546
+ mode and optional here, so one name would carry two requiredness contracts; `polytypo` 1.6.3
547
+ **accepts a `keys` it ignores** in `markdown` mode — measured, not assumed — so reusing the name
548
+ would silently begin typesetting frontmatter for any caller who passes one options object to both
549
+ modes; and the name says which block it governs, which a bare `keys` on a mode whose body is also
550
+ full of keys does not.
551
+
552
+ **What the option governs, in source offsets.** The block is the frontmatter construct §3.7.3
553
+ already recognises, and the option only changes what happens inside it:
554
+
555
+ - the **content** is the source from the code point after the opening delimiter line's
556
+ terminator to the code point that begins the closing delimiter line. A U+000D before that
557
+ terminator belongs to the terminator, exactly as in §3.8.4, so a CRLF document and the same
558
+ bytes with LF give the same content;
559
+ - **both delimiter lines stay outside every span**, as does every line terminator, so no edit
560
+ can reach `---` itself and §3.7.3's setext-underline hazard is unreachable;
561
+ - an **unterminated** block is not a block — §3.7.3 already yields no frontmatter construct
562
+ there, and the option adds no spans to it;
563
+ - a **TOML block (`+++`) yields no spans, with the option given or not.** TOML's quoting is a
564
+ second grammar — basic strings escape with U+005C, literal strings do not escape at all, and
565
+ both have multi-line forms — and §3.8.3's posture applies: a construct the scan cannot claim
566
+ with certainty yields no spans. Neither report asked for TOML and no corpus measured here
567
+ contains one, so specifying a TOML locator would be scope taken on speculation. Recorded as an
568
+ accepted miss in §7.13.
569
+
570
+ **The option adds spans only where §3.7.3's skip removed them.** If the mode did not recognise a
571
+ frontmatter construct — an unterminated block, a `---` that is not at the start of the document,
572
+ anything a given parser's frontmatter support declines — there is no block, the text is ordinary
573
+ prose in the body's own unit, and the option contributes nothing. That coupling is what makes
574
+ double processing unreachable: no source position can belong to both units. It also means the
575
+ option inherits whatever variance the five parsers already have in recognising the construct, which
576
+ is a pre-existing property of §3.7.3 rather than a new one, and §7.13 records it.
577
+
578
+ **Key matching is §3.8.2's, which means bare names at any depth.** `title` is processable wherever
579
+ it occurs in the block, `seo.title` included — measured: `seo:` then an indented `title:` is
580
+ processed under `frontmatterKeys: ["title"]` — with the cost §7.12 already accepts for `keys`: no
581
+ paths, no globs, so a caller with a machine-read `title` nested somewhere must either take it too
582
+ or name none. Matching is exact, code point for code point, with no case folding.
583
+
584
+ **The scan is §3.8's, unchanged.** §3.8.4 steps 1–9, §3.8.5 and §3.8.6 apply to the block content
585
+ verbatim, with `frontmatterKeys` as step 8's key predicate in place of `keys`. Nothing about YAML
586
+ is specified twice: frontmatter **is** YAML, and the argument that made §3.8 a hand-written scan
587
+ rather than a parser call (§3.8.1 — two of the five ecosystems' libraries cannot report an end
588
+ offset) applies here for the same reason and with the same measurements. Every miss §7.11 lists
589
+ is inherited with it, the quoted-escape bails included, and §7.13 gives what that costs on a real
590
+ corpus.
591
+
592
+ **The frontmatter block is its own text unit, and this is the part that is not obvious.** §3.2
593
+ concatenates a document's spans into one array and runs the pipeline once over it, and §7.10
594
+ records that a quotation opened in one paragraph can still pair with a mark in the next, because
595
+ the stack is not reset at a −2 marker. Measured on `polytypo` 1.6.3 in `yaml` mode, which has
596
+ exactly this shape:
597
+
598
+ ```
599
+ a: he said "hello → a: he said ‘hello
600
+ b: world" she said b: world’ she said
601
+ ```
602
+
603
+ Two spans, a −2 marker between them, and the marks paired across it. If frontmatter spans joined
604
+ the body's array, the same mechanism would let an unbalanced mark in `title` pair with one in the
605
+ first paragraph — and the body's output would then depend on the document's metadata.
606
+
607
+ > **The frontmatter block's spans form a text unit of their own.** The pipeline runs over
608
+ > that array and over the body's array separately; the two edit sets are disjoint by
609
+ > construction, since no span of one lies inside the other.
610
+
611
+ That is one more pipeline run per document and it buys a claim worth having, which a port can
612
+ test directly: **`frontmatterKeys` cannot change a byte outside the frontmatter block**, so every
613
+ case released before 1.7.0 keeps its recorded output — those cases set no option, and the body is
614
+ not reachable from one. The narrower claim is the true one: a released case's *block* would
615
+ convert if the option named a key in it. Measured, `fr-markdown-commonmark-frontmatter-nbsp`:
616
+ `title: Une note ; suite` takes its narrow no-break space under `frontmatterKeys: ["title"]`,
617
+ which is the whole point of the option and not a change to that case. The
618
+ asymmetry with `yaml` mode, where one document's keys do pair across each other, is deliberate:
619
+ there the whole document is data with prose in it, while here the block is metadata *about* a
620
+ document whose prose is the body, and the two are not one sentence in any document.
621
+
622
+ **The option applies to both dialects**, `commonmark` and `mdx`. YAML frontmatter is the same
623
+ construct in both, and §3.7.3 already lists it once for both.
624
+
625
+ **Validation.** `nbsp.md` §3.1a fixes the order through `dialect` — `mode` → `narrowNbsp` →
626
+ `rules` → `locale` → `dialect` — and `ARCHITECTURE.md`'s options table carries the whole of it.
627
+ `frontmatterKeys` joins that chain **after `dialect`**, and, like every option, is checked **before
628
+ the parse**. Both halves decide a case that is otherwise ambiguous. A call naming neither a valid
629
+ `dialect` nor a valid `frontmatterKeys` raises `POLYTYPO_INVALID_DIALECT`, because `dialect` is
630
+ first — and unlike `dialect` and `keys`, which belong to different modes and so never both apply,
631
+ these two do, which is what makes their order observable at all. A document that does not parse in
632
+ its dialect, called with a `frontmatterKeys` that is not a list of strings, raises
633
+ `POLYTYPO_INVALID_OPTION` and not `POLYTYPO_MALFORMED_INPUT`.
634
+
635
+ In `text`, `html` and `yaml` modes the option is **ignored and not validated**, exactly as
636
+ `dialect` is ignored in `text` and `html` (§3.7.1). That is the weaker of
637
+ the two choices and it is taken for consistency: a mode-specific option that throws in one mode and
638
+ is ignored in another teaches a caller nothing they can act on, and `dialect` set the precedent
639
+ before this option existed.
640
+
641
+ **What a fixture cannot express here**, the same gap `nbsp.md` §3.1a records for `narrowNbsp`: the
642
+ schema admits only a list of strings and only on a `markdown` case, so neither the throw above nor
643
+ the ignored-elsewhere rule has a fixture. Both are each runtime's own unit test, and the order in
644
+ this paragraph is what those tests assert.
645
+
646
+ ##### 3.7.4.1 What this was tested against
647
+
648
+ The corpus is **187 `.mdx` files** — the author's own site content, all of `content/`, of which the
649
+ 107 under `content/blog/` are the M4 corpus proper (PLAN.md §8). Every one of them opens with YAML
650
+ frontmatter and none with TOML. Keys `title`, `summary`,
651
+ `description`, `subtitle`, `quote`: **447 listed scalars**. Eight locales (`en-GB`, `en-US`,
652
+ `de-DE`, `fr`, `ru`, `es`, `sv`, `tr`), so 1496 file/locale cases per configuration.
653
+
654
+ Measured with `polytypo` 1.6.3's **`yaml` mode over the extracted block** — the scan this section
655
+ reuses — because `markdown` mode with the option exists in no runtime yet. The separate-unit rule
656
+ above is what makes that a faithful proxy rather than an approximation: the body cannot
657
+ participate.
658
+
659
+ Three configurations, 4488 cases — the corpus as authored, the corpus de-typeset, and the corpus
660
+ de-typeset with **every one of its 39 frontmatter keys** listed: **no byte changed outside a
661
+ listed scalar, no parse failure, no structural change, no change to an unlisted leaf, no
662
+ idempotency failure.** The de-typeset corpus is derived, and reported as derived: the site's
663
+ content is already typeset, so the characters polytypo inserts were folded back to their ASCII
664
+ originals to obtain input that has something to convert. It is a weaker corpus than found text,
665
+ and it is the only way this corpus can show a conversion at all.
666
+
667
+ - **The M4 bar holds as written.** As authored, in `en-GB` — the site's own locale — **0 of 187
668
+ files change**. Each of the other six changes 13 files and `fr` changes 117, every one of them
669
+ inside a listed scalar and from that locale's own conventions rather than from anything missed.
670
+ - **De-typeset, 247 of the 447 listed scalars convert** in `en-GB`, and none of the other 200
671
+ is a miss: the pipeline leaves them alone in `text` mode too.
672
+ - **The measurement is discriminating, which was checked rather than assumed.** The naive thing
673
+ a caller hand-rolls — the same blocks through `text` mode, no scan and no key list — damages
674
+ **every single case**: of 748 (187 files in `en-GB`, `de-DE`, `fr` and `ru`), **420 no longer
675
+ parse as YAML at all, 82 come back with different mapping keys, 246 with different values, and
676
+ none comes back unchanged.** The witnesses are ordinary: `fr` turns the key `name:` into
677
+ `name :`, and `en-GB` turns `slug: "pierre-moreau-architecture"` into
678
+ `slug: "‘pierre-moreau-architecture’"`. A harness reporting clean for both runs would prove
679
+ nothing; this one separates them completely.
680
+ - **What the key list protects here.** Listing all 39 keys converts four values the recommended
681
+ five do not, and two of them are damage rather than coverage: in `fr`, `seoTitle` gains a
682
+ narrow no-break space before the colon of a string written for a search engine, and a client
683
+ name `"HTPBE?"` becomes `"HTPBE ?"`. The other two are improvements a caller might well want
684
+ (`"Niamh O'Sullivan"` → `"Niamh O’Sullivan"`), which is the point: only the caller can tell
685
+ those apart, and that is the same argument §3.8.2 makes.
686
+ - **The inherited escape bail, measured.** Re-emitting each of the 247 convertible values in
687
+ one quoting style and running the scan over it: **single-quoted, 130 of 247 yield no spans**;
688
+ double-quoted, none do. The reason is §3.8.6's — an apostrophe inside a single-quoted scalar
689
+ is written `''`, which spells content with more characters than it has. Every listed scalar in
690
+ this corpus as written is double-quoted, so the bail never fires on it; a caller whose YAML
691
+ style is single quotes gets nothing on half of their prose, silently. That number is the
692
+ accepted cost of reusing §3.8's scan rather than a defect of this section, and §7.13 records it
693
+ where a reader will look for it.
694
+
515
695
  ### 3.8 Skip list — `yaml`
516
696
 
517
697
  **Spec 1.3.0.** `yaml` is the fourth mode id, and PLAN.md §3.2's "exactly three modes" is amended
@@ -785,7 +965,9 @@ The length test of §3.4 separates exactly the two, with no knowledge of YAML
785
965
  makes it bind rules not yet written, while a per-rule observation would not.
786
966
 
787
967
  The accepted cost is that a conversion whose replacement would **grow** against a colon or a hash
788
- is missed: `fr` inserts no narrow no-break space before a colon inside a YAML scalar, and
968
+ is missed: `fr` inserts no narrow no-break space before a colon inside a **plain** YAML scalar —
969
+ inside a quoted one the colon is content, no split applies and the space is inserted, which
970
+ `fr-markdown-commonmark-frontmatter-keys-colon` pins — and
789
971
  `one--two` takes its spaced dash only where the split leaves it interior to a span. That is the
790
972
  same trade §7.3 already made for `mot<em>!</em>`, and in the same direction: a miss is visible to
791
973
  the author and fixable in the source; a corrupted document is neither.
@@ -982,6 +1164,16 @@ they do not carry it over for free either. Write `M` for the whole mode transfor
982
1164
  > §3.8.2's `keys` option is the repair: the predicate reads the **key**, which lies outside
983
1165
  > every span and which no rule can reach, so the partition is a function of the source alone.
984
1166
 
1167
+ **`markdown` with `frontmatterKeys` (§3.7.4) inherits both**, and adds one obligation of its
1168
+ own that is discharged by the same observation. The block's spans are selected by §3.8's scan
1169
+ over YAML, so the four parts and the fifth hold verbatim — the predicate is `frontmatterKeys`
1170
+ against a key, and a key is outside every span. What is new is that the document now has **two
1171
+ units** rather than one, and `M` must partition it into the same two on the second run: the
1172
+ boundary between them is the frontmatter construct's own delimiter lines, which lie outside
1173
+ every span in either unit, so no edit can move, create or destroy one. A document whose
1174
+ frontmatter is processed therefore has a partition that is a function of the source alone,
1175
+ exactly as one whose frontmatter is skipped does.
1176
+
985
1177
  A content-dependent predicate is not merely risky here, it is unarguable: item 2's whole
986
1178
  method is to show that the characters rules may write and the positions they may write to are
987
1179
  disjoint from what decides structure. A predicate over span content puts the rules' own output
@@ -1004,7 +1196,9 @@ tests none of this document. In `yaml` the sweep alphabet must include `:`, `#`,
1004
1196
  as one token**, since those are the characters whose adjacency the argument above turns on and a
1005
1197
  three-dash run is not reachable from single dashes at a bounded payload length; and it must
1006
1198
  include U+0022 and U+005C, without which the sweep cannot reach a quoted scalar's delimiters at
1007
- all. §3.8.7 records what each run did and did not establish — including that a sweep comparing
1199
+ all. Since 1.7.0 the `markdown` run must also carry a template with a **frontmatter block and
1200
+ `frontmatterKeys` naming a key in it** — a document with two text units (§3.1) tests a composition
1201
+ the single-unit templates cannot reach. §3.8.7 records what each run did and did not establish — including that a sweep comparing
1008
1202
  structure and types is blind to a string whose content is damaged, and that an alphabet without
1009
1203
  `---` reported clean twice while §3.4's `r = d` hole was open.
1010
1204
 
@@ -1020,7 +1214,11 @@ means the claim holds in those modes only.
1020
1214
  **[P: html, markdown, yaml]** and is now _defined_ rather than assumed — §3.6, §3.7 and §3.8
1021
1215
  are what those bullets refer to. In `text` mode there are no skipped regions and the bullet is
1022
1216
  vacuous. In `yaml` the definition runs the other way round (§3.8.2): the skipped region is
1023
- everything the scan did not claim.
1217
+ everything the scan did not claim. Since 1.7.0 one region in `markdown` is skipped
1218
+ **conditionally** — the frontmatter block, whenever `frontmatterKeys` names a key in it
1219
+ (§3.7.4) — so the bullet is read against the options the call was made with. No rule's §4
1220
+ changes: the block is either skipped whole, as before, or decomposed by §3.8's scan, whose
1221
+ own claims §3.8 already states.
1024
1222
  - **`dashes` §4 "URLs, code spans, fenced code, HTML attributes"** — **[P: html, markdown]**.
1025
1223
  In `text` mode a URL is ordinary text. The rule is nevertheless safe there, but by its own
1026
1224
  guards (P1 declines the tight hyphens in `a-b`, the cluster guard declines `2026-08-15`), not
@@ -1040,9 +1238,13 @@ means the claim holds in those modes only.
1040
1238
  mark that would have been unbalanced within its own span may now pair across an element, which
1041
1239
  converts more, never less.
1042
1240
  - **`nbsp` §4 "The start or end of a text unit"** — **[P] in all modes, and strengthened.** A
1043
- "text unit" is the concatenation, and §3.4 additionally refuses any insertion at a span
1044
- boundary, so no span ever begins or ends with a character `nbsp` put there. The guarantee is
1045
- now about element boundaries as well as document boundaries.
1241
+ "text unit" is §3.1's — the concatenation — and §3.4 additionally refuses any insertion at a
1242
+ span boundary, so no span ever begins or ends with a character `nbsp` put there. The guarantee
1243
+ is now about element boundaries as well as document boundaries. Since 1.7.0 a document may have
1244
+ two units (§3.7.4), and the guarantee is read **per unit**: the frontmatter block's first and
1245
+ last positions are unit extremities of their own, which refuses more than one unit would, never
1246
+ less. That is also what makes `-separate-unit` deterministic rather than a coincidence — the
1247
+ same reading `quotes` gives a unit edge.
1046
1248
 
1047
1249
  - **`dashes` §4, the `-spaced` forms** — a new **[R]** consequence in `html`/`markdown`, not a
1048
1250
  change to any existing bullet. A `-spaced` locale converts a dash only where the replacement
@@ -1223,6 +1425,36 @@ rule-local.
1223
1425
  document has yet needed it. If one does, this is where it reopens, and the extension is
1224
1426
  additive — a caller passing bare names keeps today's behaviour.
1225
1427
 
1428
+ 13. **What `frontmatterKeys` deliberately does not claim (spec 1.7.0).** Three entries, and the
1429
+ last is a number rather than a construct:
1430
+
1431
+ - **TOML frontmatter (`+++`) yields no spans**, with the option given or not. Its quoting is a
1432
+ second grammar — U+005C escapes in basic strings, none in literal strings, multi-line forms
1433
+ of both — and §3.8.3's posture is to claim only what the scan has proved. Neither report
1434
+ behind the option (polytypo/polytypo#13, #26) asked for TOML, and the 187-file corpus of
1435
+ §3.7.4.1 contains none. Widening to TOML is additive: a caller passing keys today keeps
1436
+ today's behaviour.
1437
+ - **The block itself is recognised by the mode, not by a scan specified here.** §3.7.4's
1438
+ content range is exact once there is a block, but frontmatter is in neither CommonMark nor
1439
+ GFM — every runtime reaches it through its parser's own frontmatter support, and those
1440
+ disagree at the edges: a trailing space on either delimiter is a block in this repository's
1441
+ reference runtime, a `...` closer is not, a `---` after a blank line is not. That variance
1442
+ predates 1.7.0 and already changes output, since a parser that does not recognise the block
1443
+ typesets the metadata as prose; what 1.7.0 adds is a second way for it to show. Specifying
1444
+ the locator belongs with §3.7.3's construct recognition, which governs the skip for every
1445
+ caller, not with an option only some callers pass.
1446
+ - **§3.8.6's single-quoted bail costs more here than anywhere it has been measured before.**
1447
+ Of the 247 corpus values a locale would convert, **130 yield no spans when written as a
1448
+ single-quoted scalar**, because an apostrophe inside one is spelled `''` — and an apostrophe
1449
+ is exactly what `apostrophe` and `quotes` convert, so the bail falls hardest on the values
1450
+ the option exists for. Double-quoted, none bail. The corpus as authored is entirely
1451
+ double-quoted, so its own author never meets this; a caller whose YAML style is single
1452
+ quotes gets nothing on half their prose, and gets it silently. The repair is not this
1453
+ section's to make: §3.8.6 is ratified and measured, and the obvious extension — treating
1454
+ `''` as an opaque two-code-point unit, exactly as §3.8.6 already treats `:` and `#` inside a
1455
+ plain scalar — changes `yaml` mode for every caller and needs its own measurement and its
1456
+ own sign-off.
1457
+
1226
1458
  ## 8. Fixture coverage strategy (non-normative)
1227
1459
 
1228
1460
  This section records why `spec/fixtures/` does not carry the full every-locale × four-modes
@@ -1,5 +1,5 @@
1
1
  {
2
- "spec": "1.6.3",
2
+ "spec": "1.7.0",
3
3
  "$comment": "Single source of truth for pipeline order. Rules run in ascending `order`. Disabling a rule removes it from the sequence and never reorders the rest. Rule ids are public API (see docs/ARCHITECTURE.md section 5). \"spec\" here must track spec/VERSION exactly — it is not itself the global version source; scripts/validate-spec.mjs enforces the match.",
4
4
  "rules": [
5
5
  {