@kirigami/php-prepros 1.9.3 → 3.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,1257 @@
1
+ <?php
2
+
3
+ /**
4
+ * MD_LEGACY — hand-written, pure-PHP CommonMark/GFM-ish Markdown renderer.
5
+ *
6
+ * Retired in favor of MD:: (native `mdhtml` extension, cmark-gfm 0.29.0.gfm.13
7
+ * — see docs/STATUS.md and docs/DECISIONS.md in the repo root). Kept here as
8
+ * a reference implementation and an easy rollback path; not autoloaded or
9
+ * used anywhere by default, and its default plugins ({% codepen %},
10
+ * {% checklist %}, {% callout %}, {% img-asset %}) are not registered on it —
11
+ * see md.plugins.php, which now targets MD:: only.
12
+ */
13
+ class MD_LEGACY {
14
+
15
+ // ========================================================================
16
+ // PLUGIN SYSTEM
17
+ //
18
+ // INLINE SYNTAX (args on the same line):
19
+ // {% plugin_name arg1 arg2 "arg with spaces" %}
20
+ //
21
+ // BLOCK SYNTAX (multi-line content):
22
+ // {% plugin_name arg1 arg2
23
+ // content line 1
24
+ // content line 2
25
+ // %}
26
+ //
27
+ // The callback always receives (array $args, string $body):
28
+ // - $args : array of the arguments passed on the opening line
29
+ // - $body : multi-line content (empty "" for inline tags)
30
+ //
31
+ // Examples:
32
+ // GithubReadmeParser::registerPlugin('codepen', function(array $args, string $body): string {
33
+ // $id = htmlspecialchars($args[0] ?? '', ENT_QUOTES, 'UTF-8');
34
+ // return "<iframe src=\"https://codepen.io/embed/{$id}\"></iframe>";
35
+ // });
36
+ //
37
+ // GithubReadmeParser::registerPlugin('checklist', function(array $args, string $body): string {
38
+ // $items = array_filter(explode("\n", trim($body)));
39
+ // $html = '<ul class="checklist">';
40
+ // foreach ($items as $item) {
41
+ // $html .= '<li><input type="checkbox" /> ' . htmlspecialchars(trim($item), ENT_QUOTES, 'UTF-8') . '</li>';
42
+ // }
43
+ // return $html . '</ul>';
44
+ // });
45
+ // ========================================================================
46
+
47
+ /** @var array<string, callable(string[]): string> */
48
+ private static array $plugins = [];
49
+
50
+ /**
51
+ * Registers a plugin by name.
52
+ *
53
+ * @param string $name Tag name, e.g. "codepen"
54
+ * @param callable $callback function(array $args): string
55
+ * $args[0] = first argument, $args[1] = second, etc.
56
+ */
57
+ public static function registerPlugin(string $name, callable $callback): void {
58
+ self::$plugins[strtolower(trim($name))] = $callback;
59
+ }
60
+
61
+ /**
62
+ * Removes a registered plugin.
63
+ */
64
+ public static function unregisterPlugin(string $name): void {
65
+ unset(self::$plugins[strtolower(trim($name))]);
66
+ }
67
+
68
+ /**
69
+ * Returns the list of registered plugins.
70
+ *
71
+ * @return string[]
72
+ */
73
+ public static function getRegisteredPlugins(): array {
74
+ return array_keys(self::$plugins);
75
+ }
76
+
77
+ // ========================================================================
78
+ // Generates a "slug"-style id for heading anchors (ATX and Setext).
79
+ // ========================================================================
80
+ private static function slugify(string $text): string {
81
+ $id = strtolower(preg_replace('/[^\w\- ]/u', '', $text));
82
+ return preg_replace('/\s+/', '-', trim($id));
83
+ }
84
+
85
+ // ========================================================================
86
+ // Converts an indentation width (spaces/tabs) into a column count, a tab
87
+ // counting as 4 spaces.
88
+ // ========================================================================
89
+ private static function indentWidth(string $whitespace): int {
90
+ return strlen(str_replace("\t", ' ', $whitespace));
91
+ }
92
+
93
+ /**
94
+ * Recursively builds a (nested) <ol>/<ul> list from a flat array of items
95
+ * { indent, type, text }. $i is advanced as items are consumed.
96
+ *
97
+ * @param array<int, array{indent:int, type:string, text:string}> $items
98
+ */
99
+ private static function buildListTree(array $items, int &$i, int $count): string {
100
+ $type = $items[$i]['type'];
101
+ $baseIndent = $items[$i]['indent'];
102
+ $out = "<{$type}>\n";
103
+
104
+ while ($i < $count && $items[$i]['indent'] === $baseIndent && $items[$i]['type'] === $type) {
105
+ $text = $items[$i]['text'];
106
+ $i++;
107
+
108
+ $nested = '';
109
+ if ($i < $count && $items[$i]['indent'] > $baseIndent) {
110
+ $nested = "\n" . self::buildListTree($items, $i, $count);
111
+ }
112
+
113
+ $out .= " <li>{$text}{$nested}</li>\n";
114
+ }
115
+
116
+ return $out . "</{$type}>";
117
+ }
118
+
119
+ // ========================================================================
120
+ // EMOJIS (extended syntax): :shortcode: → unicode character.
121
+ // Non-exhaustive table but covering the most common shortcuts; extensible
122
+ // via registerEmoji().
123
+ // ========================================================================
124
+ /** @var array<string, string> */
125
+ private static array $extraEmoji = [];
126
+
127
+ private static array $emojiMap = [
128
+ 'smile' => '😄', 'smiley' => '😃', 'grin' => '😁', 'joy' => '😂', 'rofl' => '🤣',
129
+ 'blush' => '😊', 'wink' => '😉', 'relaxed' => '☺️', 'slight_smile' => '🙂',
130
+ 'upside_down_face' => '🙃', 'innocent' => '😇', 'heart_eyes' => '😍', 'kissing_heart' => '😘',
131
+ 'thinking' => '🤔', 'neutral_face' => '😐', 'expressionless' => '😑', 'no_mouth' => '😶',
132
+ 'roll_eyes' => '🙄', 'smirk' => '😏', 'unamused' => '😒', 'grimacing' => '😬',
133
+ 'lying_face' => '🤥', 'relieved' => '😌', 'pensive' => '😔', 'sleepy' => '😪',
134
+ 'drooling_face' => '🤤', 'sleeping' => '😴', 'mask' => '😷', 'sunglasses' => '😎',
135
+ 'star_struck' => '🤩', 'partying_face' => '🥳', 'worried' => '😟', 'frowning' => '☹️',
136
+ 'confused' => '😕', 'slightly_frowning_face' => '🙁', 'cry' => '😢', 'sob' => '😭',
137
+ 'scream' => '😱', 'confounded' => '😖', 'persevere' => '😣', 'disappointed' => '😞',
138
+ 'sweat' => '😓', 'weary' => '😩', 'tired_face' => '😫', 'yawning_face' => '🥱',
139
+ 'triumph' => '😤', 'rage' => '😡', 'angry' => '😠', 'cursing_face' => '🤬',
140
+ 'exploding_head' => '🤯', 'flushed' => '😳', 'hot_face' => '🥵', 'cold_face' => '🥶',
141
+ 'scream_cat' => '🙀', 'nerd_face' => '🤓', 'monocle_face' => '🧐', 'zany_face' => '🤪',
142
+ 'raised_eyebrow' => '🤨', 'shushing_face' => '🤫', 'zipper_mouth_face' => '🤐',
143
+ 'heart' => '❤️', 'orange_heart' => '🧡', 'yellow_heart' => '💛', 'green_heart' => '💚',
144
+ 'blue_heart' => '💙', 'purple_heart' => '💜', 'black_heart' => '🖤', 'white_heart' => '🤍',
145
+ 'broken_heart' => '💔', 'two_hearts' => '💕', 'sparkling_heart' => '💖', 'heartbeat' => '💓',
146
+ 'thumbsup' => '👍', '+1' => '👍', 'thumbsdown' => '👎', '-1' => '👎',
147
+ 'clap' => '👏', 'raised_hands' => '🙌', 'pray' => '🙏', 'wave' => '👋',
148
+ 'ok_hand' => '👌', 'v' => '✌️', 'crossed_fingers' => '🤞', 'muscle' => '💪',
149
+ 'point_up' => '☝️', 'point_down' => '👇', 'point_left' => '👈', 'point_right' => '👉',
150
+ 'handshake' => '🤝', 'writing_hand' => '✍️', 'fire' => '🔥', 'star' => '⭐',
151
+ 'star2' => '🌟', 'sparkles' => '✨', 'zap' => '⚡', 'boom' => '💥', 'collision' => '💥',
152
+ 'rocket' => '🚀', 'tada' => '🎉', 'confetti_ball' => '🎊', 'gift' => '🎁',
153
+ 'balloon' => '🎈', 'trophy' => '🏆', 'medal' => '🏅', 'crown' => '👑',
154
+ 'gem' => '💎', 'moneybag' => '💰', 'dollar' => '💵', '100' => '💯',
155
+ 'warning' => '⚠️', 'no_entry' => '⛔', 'stop_sign' => '🛑', 'checkered_flag' => '🏁',
156
+ 'white_check_mark' => '✅', 'heavy_check_mark' => '✔️', 'x' => '❌', 'negative_squared_cross_mark' => '❎',
157
+ 'question' => '❓', 'grey_question' => '❔', 'exclamation' => '❗', 'bangbang' => '‼️',
158
+ 'interrobang' => '⁉️', 'bulb' => '💡', 'bell' => '🔔', 'no_bell' => '🔕',
159
+ 'lock' => '🔒', 'unlock' => '🔓', 'key' => '🔑', 'mag' => '🔍', 'link' => '🔗',
160
+ 'pushpin' => '📌', 'paperclip' => '📎', 'calendar' => '📅', 'clock' => '🕐',
161
+ 'hourglass' => '⌛', 'alarm_clock' => '⏰', 'memo' => '📝', 'pencil2' => '✏️',
162
+ 'book' => '📖', 'books' => '📚', 'newspaper' => '📰', 'email' => '📧',
163
+ 'envelope' => '✉️', 'inbox_tray' => '📥', 'outbox_tray' => '📤', 'package' => '📦',
164
+ 'file_folder' => '📁', 'open_file_folder' => '📂', 'clipboard' => '📋',
165
+ 'chart_with_upwards_trend' => '📈', 'chart_with_downwards_trend' => '📉', 'bar_chart' => '📊',
166
+ 'computer' => '💻', 'desktop_computer' => '🖥️', 'keyboard' => '⌨️', 'printer' => '🖨️',
167
+ 'phone' => '📱', 'iphone' => '📱', 'camera' => '📷', 'video_camera' => '📹',
168
+ 'tv' => '📺', 'radio' => '📻', 'battery' => '🔋', 'electric_plug' => '🔌',
169
+ 'bug' => '🐛', 'beetle' => '🪲', 'gear' => '⚙️', 'wrench' => '🔧', 'hammer' => '🔨',
170
+ 'nut_and_bolt' => '🔩', 'toolbox' => '🧰', 'test_tube' => '🧪', 'microscope' => '🔬',
171
+ 'satellite' => '🛰️', 'globe_with_meridians' => '🌐', 'earth_americas' => '🌎',
172
+ 'sun' => '☀️', 'sunny' => '☀️', 'partly_sunny' => '⛅', 'cloud' => '☁️',
173
+ 'rainbow' => '🌈', 'umbrella' => '☂️', 'snowflake' => '❄️', 'droplet' => '💧',
174
+ 'ocean' => '🌊', 'tent' => '⛺', 'camping' => '🏕️', 'mountain' => '⛰️',
175
+ 'evergreen_tree' => '🌲', 'deciduous_tree' => '🌳', 'palm_tree' => '🌴',
176
+ 'cactus' => '🌵', 'seedling' => '🌱', 'four_leaf_clover' => '🍀', 'maple_leaf' => '🍁',
177
+ 'dog' => '🐶', 'cat' => '🐱', 'mouse' => '🐭', 'rabbit' => '🐰', 'fox_face' => '🦊',
178
+ 'bear' => '🐻', 'panda_face' => '🐼', 'koala' => '🐨', 'tiger' => '🐯', 'lion' => '🦁',
179
+ 'cow' => '🐮', 'pig' => '🐷', 'frog' => '🐸', 'monkey_face' => '🐵', 'chicken' => '🐔',
180
+ 'penguin' => '🐧', 'bird' => '🐦', 'baby_chick' => '🐤', 'owl' => '🦉',
181
+ 'horse' => '🐴', 'unicorn' => '🦄', 'bee' => '🐝', 'butterfly' => '🦋', 'snail' => '🐌',
182
+ 'octopus' => '🐙', 'fish' => '🐟', 'dolphin' => '🐬', 'whale' => '🐳',
183
+ 'pizza' => '🍕', 'hamburger' => '🍔', 'fries' => '🍟', 'hotdog' => '🌭',
184
+ 'taco' => '🌮', 'sushi' => '🍣', 'ramen' => '🍜', 'spaghetti' => '🍝',
185
+ 'bread' => '🍞', 'cheese' => '🧀', 'egg' => '🥚', 'popcorn' => '🍿',
186
+ 'cookie' => '🍪', 'doughnut' => '🍩', 'cake' => '🍰', 'birthday' => '🎂',
187
+ 'candy' => '🍬', 'chocolate_bar' => '🍫', 'icecream' => '🍦', 'apple' => '🍎',
188
+ 'banana' => '🍌', 'grapes' => '🍇', 'watermelon' => '🍉', 'strawberry' => '🍓',
189
+ 'lemon' => '🍋', 'peach' => '🍑', 'coffee' => '☕', 'tea' => '🍵', 'beer' => '🍺',
190
+ 'beers' => '🍻', 'wine_glass' => '🍷', 'cocktail' => '🍸', 'tropical_drink' => '🍹',
191
+ 'champagne' => '🍾', 'soccer' => '⚽', 'basketball' => '🏀', 'football' => '🏈',
192
+ 'baseball' => '⚾', 'tennis' => '🎾', 'volleyball' => '🏐', 'rugby_football' => '🏉',
193
+ '8ball' => '🎱', 'golf' => '⛳', 'dart' => '🎯', 'video_game' => '🎮',
194
+ 'game_die' => '🎲', 'jigsaw' => '🧩', 'car' => '🚗', 'taxi' => '🚕', 'bus' => '🚌',
195
+ 'ambulance' => '🚑', 'fire_engine' => '🚒', 'police_car' => '🚓', 'bike' => '🚲',
196
+ 'airplane' => '✈️', 'helicopter' => '🚁', 'train' => '🚆', 'ship' => '🚢',
197
+ 'house' => '🏠', 'office' => '🏢', 'hospital' => '🏥', 'school' => '🏫',
198
+ 'church' => '⛪', 'castle' => '🏰', 'world_map' => '🗺️', 'flag_white' => '🏳️',
199
+ 'flag_black' => '🏴', 'checkered_flag2' => '🏁', 'eyes' => '👀', 'eye' => '👁️',
200
+ 'speech_balloon' => '💬', 'thought_balloon' => '💭', 'zzz' => '💤', 'boom2' => '💥',
201
+ 'sos' => '🆘', 'new' => '🆕', 'ok' => '🆗', 'up' => '🆙', 'cool' => '🆒',
202
+ 'free' => '🆓', 'id' => '🆔', 'ng' => '🆖',
203
+
204
+ // -- Faces and emotions (continued) --------------------------------
205
+ 'smiling_face_with_three_hearts' => '🥰', 'kissing' => '😗', 'kissing_closed_eyes' => '😚',
206
+ 'kissing_smiling_eyes' => '😙', 'yum' => '😋', 'stuck_out_tongue' => '😛',
207
+ 'stuck_out_tongue_winking_eye' => '😜', 'stuck_out_tongue_closed_eyes' => '😝',
208
+ 'money_mouth_face' => '🤑', 'hugs' => '🤗', 'disappointed_relieved' => '😥',
209
+ 'dizzy_face' => '😵', 'astonished' => '😲', 'open_mouth' => '😮', 'hushed' => '😯',
210
+ 'fearful' => '😨', 'cold_sweat' => '😰', 'nauseated_face' => '🤢', 'vomiting_face' => '🤮',
211
+ 'sneezing_face' => '🤧', 'face_with_thermometer' => '🤒', 'face_with_head_bandage' => '🤕',
212
+ 'woozy_face' => '🥴', 'smiling_imp' => '😈', 'imp' => '👿', 'japanese_ogre' => '👹',
213
+ 'japanese_goblin' => '👺', 'skull' => '💀', 'skull_and_crossbones' => '☠️',
214
+ 'ghost' => '👻', 'alien' => '👽', 'space_invader' => '👾', 'robot' => '🤖',
215
+ 'poop' => '💩', 'clown_face' => '🤡', 'smiley_cat' => '😺', 'smile_cat' => '😸',
216
+ 'joy_cat' => '😹', 'heart_eyes_cat' => '😻', 'smirk_cat' => '😼', 'kissing_cat' => '😽',
217
+ 'pouting_cat' => '😾', 'crying_cat_face' => '😿',
218
+
219
+ // -- Corps, gestes, personnages --------------------------------
220
+ 'raised_hand' => '✋', 'raised_back_of_hand' => '🤚', 'vulcan_salute' => '🖖',
221
+ 'pinching_hand' => '🤏', 'fist' => '✊', 'punch' => '👊', 'left_facing_fist' => '🤛',
222
+ 'right_facing_fist' => '🤜', 'open_hands' => '👐', 'palms_up_together' => '🤲',
223
+ 'nail_care' => '💅', 'selfie' => '🤳', 'ear' => '👂', 'nose' => '👃', 'brain' => '🧠',
224
+ 'tongue' => '👅', 'lips' => '👄', 'tooth' => '🦷', 'bone' => '🦴',
225
+ 'baby' => '👶', 'child' => '🧒', 'boy' => '👦', 'girl' => '👧', 'adult' => '🧑',
226
+ 'man' => '👨', 'woman' => '👩', 'older_adult' => '🧓', 'older_man' => '👴', 'older_woman' => '👵',
227
+ 'mage' => '🧙', 'superhero' => '🦸', 'supervillain' => '🦹', 'vampire' => '🧛',
228
+ 'zombie' => '🧟', 'genie' => '🧞', 'merperson' => '🧜', 'elf' => '🧝', 'fairy' => '🧚',
229
+
230
+ // -- Animaux (suite) ---------------------------------------------
231
+ 'wolf' => '🐺', 'boar' => '🐗', 'racehorse' => '🐎', 'zebra' => '🦓', 'deer' => '🦌',
232
+ 'cow2' => '🐄', 'ox' => '🐂', 'water_buffalo' => '🐃', 'pig2' => '🐖', 'ram' => '🐏',
233
+ 'sheep' => '🐑', 'goat' => '🐐', 'camel' => '🐫', 'dromedary_camel' => '🐪',
234
+ 'llama' => '🦙', 'giraffe' => '🦒', 'elephant' => '🐘', 'rhinoceros' => '🦏',
235
+ 'hippopotamus' => '🦛', 'mouse2' => '🐁', 'rat' => '🐀', 'hamster' => '🐹',
236
+ 'chipmunk' => '🐿️', 'hedgehog' => '🦔', 'bat' => '🦇', 'duck' => '🦆', 'eagle' => '🦅',
237
+ 'flamingo' => '🦩', 'peacock' => '🦚', 'parrot' => '🦜', 'swan' => '🦢',
238
+ 'turkey' => '🦃', 'dove' => '🕊️', 'rooster' => '🐓', 'crocodile' => '🐊',
239
+ 'turtle' => '🐢', 'lizard' => '🦎', 'snake' => '🐍', 'dragon_face' => '🐲',
240
+ 'dragon' => '🐉', 'sauropod' => '🦕', 't-rex' => '🦖', 'whale2' => '🐋',
241
+ 'shark' => '🦈', 'seal' => '🦭', 'squid' => '🦑', 'shrimp' => '🦐', 'lobster' => '🦞',
242
+ 'crab' => '🦀', 'blowfish' => '🐡', 'tropical_fish' => '🐠', 'oyster' => '🦪',
243
+ 'ant' => '🐜', 'spider' => '🕷️', 'spider_web' => '🕸️', 'scorpion' => '🦂',
244
+ 'mosquito' => '🦟', 'microbe' => '🦠', 'paw_prints' => '🐾',
245
+
246
+ // -- Nature, plants, weather (continued) --------------------------------
247
+ 'cherry_blossom' => '🌸', 'blossom' => '🌼', 'rose' => '🌹', 'wilted_flower' => '🥀',
248
+ 'hibiscus' => '🌺', 'sunflower' => '🌻', 'tulip' => '🌷', 'herb' => '🌿',
249
+ 'shamrock' => '☘️', 'fallen_leaf' => '🍂', 'leaves' => '🍃', 'mushroom' => '🍄',
250
+ 'chestnut' => '🌰', 'crescent_moon' => '🌙', 'full_moon' => '🌕', 'new_moon' => '🌑',
251
+ 'milky_way' => '🌌', 'stars' => '🌠', 'cyclone' => '🌀', 'fog' => '🌫️',
252
+ 'wind_face' => '🌬️', 'tornado' => '🌪️', 'thunder_cloud_and_rain' => '⛈️',
253
+ 'sweat_drops' => '💦', 'snowman' => '⛄', 'snowman_with_snow' => '☃️', 'comet' => '☄️',
254
+
255
+ // -- Nourriture (suite) --------------------------------------------
256
+ 'tomato' => '🍅', 'eggplant' => '🍆', 'avocado' => '🥑', 'broccoli' => '🥦',
257
+ 'carrot' => '🥕', 'corn' => '🌽', 'hot_pepper' => '🌶️', 'cucumber' => '🥒',
258
+ 'potato' => '🥔', 'sweet_potato' => '🍠', 'peanuts' => '🥜', 'honey_pot' => '🍯',
259
+ 'croissant' => '🥐', 'bagel' => '🥯', 'pretzel' => '🥨', 'pancakes' => '🥞',
260
+ 'waffle' => '🧇', 'meat_on_bone' => '🍖', 'poultry_leg' => '🍗', 'bacon' => '🥓',
261
+ 'sandwich' => '🥪', 'stuffed_flatbread' => '🥙', 'burrito' => '🌯', 'salad' => '🥗',
262
+ 'shallow_pan_of_food' => '🥘', 'canned_food' => '🥫', 'bento' => '🍱',
263
+ 'rice_ball' => '🍙', 'rice' => '🍚', 'curry' => '🍛', 'stew' => '🍲', 'oden' => '🍢',
264
+ 'dango' => '🍡', 'shaved_ice' => '🍧', 'ice_cream' => '🍨', 'pie' => '🥧',
265
+ 'cupcake' => '🧁', 'moon_cake' => '🥮', 'lollipop' => '🍭', 'custard' => '🍮',
266
+ 'milk_glass' => '🥛', 'baby_bottle' => '🍼', 'mate' => '🧉', 'ice_cube' => '🧊',
267
+ 'tumbler_glass' => '🥃', 'cup_with_straw' => '🥤', 'chopsticks' => '🥢',
268
+ 'fork_and_knife' => '🍴', 'spoon' => '🥄', 'plate_with_cutlery' => '🍽️',
269
+
270
+ // -- Activities, sport, leisure --------------------------------------
271
+ 'running' => '🏃', 'walking' => '🚶', 'swimming' => '🏊', 'surfing' => '🏄',
272
+ 'skateboard' => '🛹', 'snowboarder' => '🏂', 'weight_lifting' => '🏋️',
273
+ 'cyclist' => '🚴', 'medal_military' => '🎖️', 'ticket' => '🎫', 'circus_tent' => '🎪',
274
+ 'performing_arts' => '🎭', 'art' => '🎨', 'clapper' => '🎬', 'microphone' => '🎤',
275
+ 'headphones' => '🎧', 'musical_note' => '🎵', 'musical_score' => '🎼', 'guitar' => '🎸',
276
+ 'violin' => '🎻', 'drum' => '🥁', 'trumpet' => '🎺', 'saxophone' => '🎷',
277
+ 'musical_keyboard' => '🎹', 'chess_pawn' => '♟️', 'bowling' => '🎳',
278
+ 'ice_skate' => '⛸️', 'ski' => '🎿', 'fishing_pole_and_fish' => '🎣',
279
+ 'boxing_glove' => '🥊', 'martial_arts_uniform' => '🥋', 'goal_net' => '🥅',
280
+ 'flying_disc' => '🥏', 'yo_yo' => '🪀', 'kite' => '🪁',
281
+
282
+ // -- Voyages et lieux (suite) ----------------------------------------
283
+ 'airplane_departure' => '🛫', 'airplane_arriving' => '🛬', 'flying_saucer' => '🛸',
284
+ 'motorcycle' => '🏍️', 'scooter' => '🛴', 'tractor' => '🚜', 'truck' => '🚚',
285
+ 'articulated_lorry' => '🚛', 'trolleybus' => '🚎', 'minibus' => '🚐', 'metro' => '🚇',
286
+ 'station' => '🚉', 'monorail' => '🚝', 'bullettrain_front' => '🚄',
287
+ 'steam_locomotive' => '🚂', 'anchor' => '⚓', 'sailboat' => '⛵', 'canoe' => '🛶',
288
+ 'speedboat' => '🚤', 'ferry' => '⛴️', 'passport_control' => '🛂', 'customs' => '🛃',
289
+ 'baggage_claim' => '🛄', 'left_luggage' => '🛅', 'vertical_traffic_light' => '🚦',
290
+ 'construction' => '🚧', 'fuelpump' => '⛽', 'busstop' => '🚏', 'moyai' => '🗿',
291
+ 'statue_of_liberty' => '🗽', 'tokyo_tower' => '🗼', 'fountain' => '⛲',
292
+ 'stadium' => '🏟️', 'ferris_wheel' => '🎡', 'roller_coaster' => '🎢',
293
+ 'carousel_horse' => '🎠', 'beach_umbrella' => '🏖️', 'desert' => '🏜️',
294
+ 'desert_island' => '🏝️', 'national_park' => '🏞️', 'sunrise' => '🌅',
295
+ 'sunrise_over_mountains' => '🌄', 'sparkler' => '🎇', 'fireworks' => '🎆',
296
+ 'city_sunset' => '🌇', 'bridge_at_night' => '🌉', 'houses' => '🏘️',
297
+ 'derelict_house' => '🏚️', 'classical_building' => '🏛️', 'department_store' => '🏬',
298
+ 'post_office' => '🏣', 'hotel' => '🏨', 'convenience_store' => '🏪', 'bank' => '🏦',
299
+ 'factory' => '🏭',
300
+
301
+ // -- Objets (suite) ---------------------------------------------------
302
+ 'watch' => '⌚', 'stopwatch' => '⏱️', 'timer_clock' => '⏲️', 'joystick' => '🕹️',
303
+ 'floppy_disk' => '💾', 'cd' => '💿', 'dvd' => '📀', 'movie_camera' => '🎥',
304
+ 'projector' => '📽️', 'telephone' => '☎️', 'pager' => '📟', 'fax' => '📠',
305
+ 'candle' => '🕯️', 'fire_extinguisher' => '🧯', 'oil_drum' => '🛢️',
306
+ 'money_with_wings' => '💸', 'credit_card' => '💳', 'yen' => '💴', 'euro' => '💶',
307
+ 'pound' => '💷', 'briefcase' => '💼', 'balance_scale' => '⚖️', 'compass' => '🧭',
308
+ 'triangular_ruler' => '📐', 'straight_ruler' => '📏', 'round_pushpin' => '📍',
309
+ 'scissors' => '✂️', 'thread' => '🧵', 'yarn' => '🧶', 'safety_pin' => '🧷',
310
+ 'basket' => '🧺', 'hourglass_flowing_sand' => '⏳', 'notebook' => '📓',
311
+ 'notebook_with_decorative_cover' => '📔', 'page_facing_up' => '📄',
312
+ 'page_with_curl' => '📃', 'bookmark_tabs' => '📑', 'bookmark' => '🔖',
313
+ 'label' => '🏷️', 'receipt' => '🧾', 'card_index' => '📇', 'wastebasket' => '🗑️',
314
+ 'old_key' => '🗝️', 'hammer_and_wrench' => '🛠️', 'pick' => '⛏️', 'shield' => '🛡️',
315
+ 'syringe' => '💉', 'pill' => '💊', 'thermometer' => '🌡️', 'soap' => '🧼',
316
+ 'broom' => '🧹',
317
+
318
+ // -- Symboles (suite) -----------------------------------------------
319
+ 'heavy_multiplication_x' => '✖️', 'heavy_plus_sign' => '➕', 'heavy_minus_sign' => '➖',
320
+ 'heavy_division_sign' => '➗', 'infinity' => '♾️', 'recycle' => '♻️', 'trident' => '🔱',
321
+ 'atom_symbol' => '⚛️', 'om' => '🕉️', 'peace_symbol' => '☮️', 'yin_yang' => '☯️',
322
+ 'wheel_of_dharma' => '☸️', 'star_of_david' => '✡️', 'star_and_crescent' => '☪️',
323
+ 'cross' => '✝️', 'menorah' => '🕎', 'radioactive' => '☢️', 'biohazard' => '☣️',
324
+ 'arrow_up' => '⬆️', 'arrow_down' => '⬇️', 'arrow_left' => '⬅️', 'arrow_right' => '➡️',
325
+ 'arrow_upper_right' => '↗️', 'arrow_lower_right' => '↘️', 'arrow_lower_left' => '↙️',
326
+ 'arrow_upper_left' => '↖️', 'arrows_clockwise' => '🔃', 'arrows_counterclockwise' => '🔄',
327
+ 'back' => '🔙', 'end' => '🔚', 'on' => '🔛', 'soon' => '🔜', 'top' => '🔝',
328
+ 'radio_button' => '🔘', 'red_circle' => '🔴', 'orange_circle' => '🟠',
329
+ 'yellow_circle' => '🟡', 'green_circle' => '🟢', 'blue_circle' => '🔵',
330
+ 'purple_circle' => '🟣', 'brown_circle' => '🟤', 'white_circle' => '⚪',
331
+ 'black_circle' => '⚫',
332
+
333
+ // -- Drapeaux (suite) -------------------------------------------------
334
+ 'triangular_flag_on_post' => '🚩', 'crossed_flags' => '🎌',
335
+ 'us' => '🇺🇸', 'gb' => '🇬🇧', 'fr' => '🇫🇷', 'de' => '🇩🇪', 'es' => '🇪🇸',
336
+ 'it' => '🇮🇹', 'jp' => '🇯🇵', 'cn' => '🇨🇳', 'kr' => '🇰🇷', 'ca' => '🇨🇦',
337
+ 'au' => '🇦🇺', 'br' => '🇧🇷', 'in' => '🇮🇳', 'ru' => '🇷🇺', 'eu' => '🇪🇺',
338
+ ];
339
+
340
+ /**
341
+ * Registers (or replaces) a custom emoji shortcut.
342
+ */
343
+ public static function registerEmoji(string $shortcode, string $char): void {
344
+ self::$extraEmoji[strtolower(trim($shortcode, ':'))] = $char;
345
+ }
346
+
347
+ private static function emojiFor(string $shortcode): ?string {
348
+ $key = strtolower($shortcode);
349
+ return self::$extraEmoji[$key] ?? self::$emojiMap[$key] ?? null;
350
+ }
351
+
352
+ // ========================================================================
353
+ // DEFINITION LISTS (extended syntax)
354
+ // Term
355
+ // : Definition
356
+ // Procedural line-by-line analysis (safer than a single big regex for
357
+ // grouping several term/definition pairs into one <dl>, separated by a
358
+ // blank line or not).
359
+ // ========================================================================
360
+ private static function isDefinitionColonLine(string $line): bool {
361
+ return (bool) preg_match('/^[ \t]*:[ \t]+.+$/', $line);
362
+ }
363
+
364
+ private static function looksLikeOtherBlock(string $line): bool {
365
+ $t = ltrim($line);
366
+ if ($t === '') return true;
367
+ if (str_starts_with($t, '<')) return true;
368
+ return (bool) preg_match('/^(?:#{1,6}[ \t]|>|```|\||[-*+][ \t]|\d+\.[ \t])/', $t);
369
+ }
370
+
371
+ private static function extractDefinitionLists(string $html): string {
372
+ $lines = explode("\n", $html);
373
+ $n = count($lines);
374
+ $out = [];
375
+ $i = 0;
376
+
377
+ while ($i < $n) {
378
+ $isTermStart = $i + 1 < $n
379
+ && !self::looksLikeOtherBlock($lines[$i])
380
+ && !self::isDefinitionColonLine($lines[$i])
381
+ && self::isDefinitionColonLine($lines[$i + 1]);
382
+
383
+ if (!$isTermStart) {
384
+ $out[] = $lines[$i];
385
+ $i++;
386
+ continue;
387
+ }
388
+
389
+ $dl = "<dl>\n";
390
+ while (true) {
391
+ $term = trim($lines[$i]);
392
+ $dl .= " <dt>{$term}</dt>\n";
393
+ $i++;
394
+ while ($i < $n && preg_match('/^[ \t]*:[ \t]+(.*)$/', $lines[$i], $m)) {
395
+ $dl .= " <dd>{$m[1]}</dd>\n";
396
+ $i++;
397
+ }
398
+
399
+ // A single blank line between two groups stays in the same <dl>
400
+ // if the next group really is a new term.
401
+ if ($i < $n && trim($lines[$i]) === '') {
402
+ $j = $i;
403
+ while ($j < $n && trim($lines[$j]) === '') $j++;
404
+ if ($j + 1 < $n
405
+ && !self::looksLikeOtherBlock($lines[$j])
406
+ && !self::isDefinitionColonLine($lines[$j])
407
+ && self::isDefinitionColonLine($lines[$j + 1])
408
+ ) {
409
+ $i = $j;
410
+ continue;
411
+ }
412
+ }
413
+ break;
414
+ }
415
+ $dl .= "</dl>";
416
+ $out[] = $dl;
417
+ }
418
+
419
+ return implode("\n", $out);
420
+ }
421
+
422
+ // ========================================================================
423
+ // RAW HTML (safe subset, GitHub README style)
424
+ //
425
+ // Writing HTML tags directly (e.g. <div align="center">, <img>, <sub>,
426
+ // <br>, HTML tables...) is allowed ONLY if:
427
+ // - the tag is part of the HTML_ALLOWED_TAGS whitelist;
428
+ // - every attribute is part of the whitelist for that tag (or of the
429
+ // global HTML_GLOBAL_ATTRS attributes);
430
+ // - no attribute starts with "on" (onclick, onerror, ...);
431
+ // - URLs (href/src) use a safe scheme (isSafeUrl).
432
+ //
433
+ // Any unknown or dangerous tag (script/style/iframe/...), or any
434
+ // non-whitelisted attribute, is silently stripped. The text content
435
+ // between the tags is NOT swallowed: it keeps being processed as normal
436
+ // markdown (that's what lets you have markdown headings, badges and images
437
+ // inside a <div align="center">...</div>).
438
+ // ========================================================================
439
+
440
+ private const HTML_ALLOWED_TAGS = [
441
+ 'div', 'span', 'p', 'br', 'hr', 'wbr',
442
+ 'b', 'strong', 'i', 'em', 'u', 's', 'strike', 'del', 'ins',
443
+ 'mark', 'small', 'sub', 'sup', 'kbd', 'code', 'pre', 'abbr', 'q', 'cite',
444
+ 'ul', 'ol', 'li', 'dl', 'dt', 'dd',
445
+ 'table', 'thead', 'tbody', 'tfoot', 'tr', 'td', 'th', 'caption', 'colgroup', 'col',
446
+ 'blockquote',
447
+ 'a', 'img', 'picture', 'source', 'figure', 'figcaption',
448
+ 'h1', 'h2', 'h3', 'h4', 'h5', 'h6',
449
+ 'details', 'summary', 'center',
450
+ ];
451
+
452
+ /** Attributes allowed on any whitelisted tag. */
453
+ private const HTML_GLOBAL_ATTRS = ['id', 'class', 'title', 'align', 'valign', 'width', 'height', 'dir', 'lang'];
454
+
455
+ /** Extra attributes allowed, per tag. */
456
+ private const HTML_TAG_ATTRS = [
457
+ 'a' => ['href', 'name', 'target', 'rel'],
458
+ 'img' => ['src', 'alt', 'loading', 'srcset', 'sizes'],
459
+ 'source' => ['src', 'srcset', 'type', 'media'],
460
+ 'td' => ['colspan', 'rowspan'],
461
+ 'th' => ['colspan', 'rowspan', 'scope'],
462
+ 'col' => ['span'],
463
+ 'ol' => ['start', 'type'],
464
+ 'details' => ['open'],
465
+ ];
466
+
467
+ /** Self-closing tags (no closing tag expected). */
468
+ private const HTML_VOID_TAGS = ['img', 'br', 'hr', 'wbr', 'source', 'col'];
469
+
470
+ /**
471
+ * Checks that a URL (href/src) uses a safe scheme: relative links, anchors,
472
+ * http(s), mailto, tel, or base64-encoded images (png/gif/jpeg/webp only —
473
+ * not svg+xml, which can embed a <script>).
474
+ * Notably rejects javascript:, vbscript:, data:text/html.
475
+ */
476
+ private static function isSafeUrl(string $url): bool {
477
+ $url = trim($url);
478
+ if ($url === '') return true;
479
+ // A path with no explicit scheme ("assets/x.png", "../x", "#anchor",
480
+ // "/x", "x") is a relative link or an anchor: always safe.
481
+ if (!preg_match('~^([a-zA-Z][a-zA-Z0-9+.\-]*):~', $url, $m)) return true;
482
+ $scheme = strtolower($m[1]);
483
+ if (in_array($scheme, ['http', 'https', 'mailto', 'tel'], true)) return true;
484
+ if ($scheme === 'data') {
485
+ // Base64-encoded images only — no data:image/svg+xml, which can
486
+ // embed a <script>, and no data:text/html.
487
+ return (bool) preg_match('~^data:image/(png|gif|jpe?g|webp);base64,~i', $url);
488
+ }
489
+ return false; // javascript:, vbscript:, file:, etc. → rejected
490
+ }
491
+
492
+ /**
493
+ * Sanitizes a single raw HTML tag (e.g. '<div align="center">', '</div>',
494
+ * '<img src="..." onerror="...">').
495
+ *
496
+ * @return string|null The cleaned tag to keep, an empty string to strip it
497
+ * silently, or null if it doesn't look like a valid
498
+ * HTML tag (in which case the caller strips it too, to
499
+ * be safe).
500
+ */
501
+ private static function sanitizeHtmlTag(string $tag): ?string {
502
+ if (!preg_match(
503
+ '/^<(\/)?([a-zA-Z][a-zA-Z0-9-]*)((?:\s+[a-zA-Z_:][a-zA-Z0-9_:.-]*(?:\s*=\s*(?:"[^"]*"|\'[^\']*\'|[^\s"\'>]+))?)*)\s*(\/)?>$/s',
504
+ $tag,
505
+ $m
506
+ )) {
507
+ return null;
508
+ }
509
+
510
+ $closing = $m[1] === '/';
511
+ $tagName = strtolower($m[2]);
512
+ $attrsRaw = $m[3];
513
+
514
+ if (!in_array($tagName, self::HTML_ALLOWED_TAGS, true)) {
515
+ return null;
516
+ }
517
+
518
+ if ($closing) {
519
+ return "</{$tagName}>";
520
+ }
521
+
522
+ $allowedAttrs = array_merge(self::HTML_GLOBAL_ATTRS, self::HTML_TAG_ATTRS[$tagName] ?? []);
523
+ $safeAttrs = '';
524
+
525
+ if (preg_match_all(
526
+ '/([a-zA-Z_:][a-zA-Z0-9_:.-]*)(?:\s*=\s*("([^"]*)"|\'([^\']*)\'|([^\s"\'>]+)))?/',
527
+ $attrsRaw,
528
+ $am,
529
+ PREG_SET_ORDER
530
+ )) {
531
+ foreach ($am as $a) {
532
+ $attrName = strtolower($a[1]);
533
+ if ($attrName === '') continue;
534
+ if (str_starts_with($attrName, 'on')) continue; // safety net against JS handlers
535
+ if (!in_array($attrName, $allowedAttrs, true)) continue;
536
+
537
+ if ($tagName === 'details' && $attrName === 'open') {
538
+ $safeAttrs .= ' open';
539
+ continue;
540
+ }
541
+
542
+ $attrVal = $a[3] ?? ($a[4] ?? ($a[5] ?? ''));
543
+
544
+ if (in_array($attrName, ['href', 'src'], true) && !self::isSafeUrl($attrVal)) {
545
+ continue;
546
+ }
547
+
548
+ $safeAttrs .= ' ' . $attrName . '="' . htmlspecialchars($attrVal, ENT_QUOTES, 'UTF-8') . '"';
549
+ }
550
+ }
551
+
552
+ $close = in_array($tagName, self::HTML_VOID_TAGS, true) ? ' /' : '';
553
+ return "<{$tagName}{$safeAttrs}{$close}>";
554
+ }
555
+
556
+ // ========================================================================
557
+
558
+ public static function toHtml(string $markdown): string {
559
+
560
+ // ====================================================================
561
+ // STEP 1: Line-ending normalization
562
+ // ====================================================================
563
+ $html = str_replace(["\r\n", "\r"], "\n", $markdown);
564
+
565
+
566
+ // ====================================================================
567
+ // STEP 2: PLUGINS
568
+ // Two forms supported:
569
+ //
570
+ // INLINE: {% name arg1 "arg 2" %}
571
+ // → $args = ['arg1', 'arg 2'], $body = ''
572
+ //
573
+ // BLOCK : {% name arg1\ncontent\nover\nseveral lines\n%}
574
+ // → $args = ['arg1'], $body = "content\nover\nseveral lines"
575
+ //
576
+ // Both are captured by a single regex that tells apart the presence of
577
+ // a newline after the args (block) or not (inline).
578
+ // Processed before XSS encoding — re-injected as the very last step.
579
+ // ====================================================================
580
+ $pluginBlocks = [];
581
+ // Literal, XSS-escaped source of each captured tag, keyed by the same
582
+ // placeholder. Used to restore a `{% tag %}` that turns out to sit
583
+ // inside a code span / code block as verbatim text instead of expanding it.
584
+ $pluginLiterals = [];
585
+
586
+ /**
587
+ * Parses an argument string into an array.
588
+ * Supports bare words, "double quotes" and 'single quotes'.
589
+ */
590
+ $parseArgs = static function (string $rawArgs): array {
591
+ $args = [];
592
+ if (trim($rawArgs) === '') return $args;
593
+ preg_match_all(
594
+ '/"([^"\\\\]*(?:\\\\.[^"\\\\]*)*)"|\'([^\'\\\\]*(?:\\\\.[^\'\\\\]*)*)\'|(\S+)/',
595
+ $rawArgs,
596
+ $m
597
+ );
598
+ foreach ($m[0] as $i => $_) {
599
+ $args[] = $m[1][$i] !== ''
600
+ ? stripslashes($m[1][$i])
601
+ : ($m[2][$i] !== ''
602
+ ? stripslashes($m[2][$i])
603
+ : $m[3][$i]);
604
+ }
605
+ return $args;
606
+ };
607
+
608
+ $html = preg_replace_callback(
609
+ // Group 1: plugin name
610
+ // Group 2: inline args (everything on the first line after the name)
611
+ // Group 3: multi-line body (present only for block tags)
612
+ '/\{%\s*([a-zA-Z0-9_-]+)([^\n%]*?)(?:\n([\s\S]*?))?\s*%\}/m',
613
+ function ($matches) use (&$pluginBlocks, &$pluginLiterals, $parseArgs): string {
614
+ $name = strtolower(trim($matches[1]));
615
+ $args = $parseArgs(trim($matches[2] ?? ''));
616
+ // $matches[3] exists only if the tag is multi-line
617
+ $body = isset($matches[3]) ? trim($matches[3]) : '';
618
+
619
+ if (!isset(self::$plugins[$name])) {
620
+ // Unknown plugin: kept encoded rather than silently removed
621
+ return htmlspecialchars($matches[0], ENT_QUOTES, 'UTF-8');
622
+ }
623
+
624
+ $output = (self::$plugins[$name])($args, $body);
625
+ $placeholder = "\x02PLG" . count($pluginBlocks) . "\x03";
626
+ $pluginBlocks[$placeholder] = $output;
627
+ $pluginLiterals[$placeholder] = htmlspecialchars($matches[0], ENT_QUOTES, 'UTF-8');
628
+ return $placeholder;
629
+ },
630
+ $html
631
+ );
632
+
633
+
634
+ // ====================================================================
635
+ // STEP 2a: FOOTNOTE DEFINITIONS
636
+ // [^1]: Note text.
637
+ // [^bignote]: First line.
638
+ //
639
+ // Following paragraph, indented by 4 spaces or 1 tab.
640
+ //
641
+ // `{ some code }`
642
+ // Extracted (and removed from the text) BEFORE reference link
643
+ // definitions, since [^label]: would otherwise match their regex too.
644
+ // Each note's content is rendered via a recursive call to toHtml() to
645
+ // support multiple paragraphs, code, etc.
646
+ // ====================================================================
647
+ $footnoteDefs = [];
648
+ $html = preg_replace_callback(
649
+ '/^\[\^([^\]\s]+)\]:[ \t]?([^\n]*)((?:\n(?:[ \t]{4}[^\n]*|[ \t]*))*)/m',
650
+ function ($m) use (&$footnoteDefs): string {
651
+ $label = strtolower(trim($m[1]));
652
+ $first = $m[2];
653
+ $rest = $m[3] ?? '';
654
+ $restLines = $rest !== '' ? explode("\n", $rest) : [];
655
+ $restLines = array_map(static function (string $l): string {
656
+ return preg_replace('/^(?:[ ]{4}|\t)/', '', $l);
657
+ }, $restLines);
658
+ $content = trim($first . "\n" . implode("\n", $restLines));
659
+ $footnoteDefs[$label] = self::toHtml($content);
660
+ return '';
661
+ },
662
+ $html
663
+ );
664
+
665
+
666
+ // ====================================================================
667
+ // STEP 2b: REFERENCE LINK DEFINITIONS
668
+ // [label]: https://example.com "Optional title"
669
+ // [label]: <https://example.com> 'Optional title'
670
+ // [label]: https://example.com (Optional title)
671
+ // Extracted (and removed from the text) before everything else; used
672
+ // later by the [text][label] / [text][] links.
673
+ // ====================================================================
674
+ $refDefs = [];
675
+ $html = preg_replace_callback(
676
+ '/^[ \t]{0,3}\[([^\]]+)\]:[ \t]*<?([^\s>]+)>?(?:[ \t]+(?:"([^"]*)"|\'([^\']*)\'|\(([^)]*)\)))?[ \t]*$/m',
677
+ function ($m) use (&$refDefs): string {
678
+ $label = strtolower(trim($m[1]));
679
+ $title = $m[3] !== '' ? $m[3] : ($m[4] !== '' ? $m[4] : ($m[5] ?? ''));
680
+ $refDefs[$label] = ['url' => $m[2], 'title' => $title];
681
+ return '';
682
+ },
683
+ $html
684
+ );
685
+
686
+
687
+ // ====================================================================
688
+ // STEP 3: CODE BLOCKS (```lang ... ```)
689
+ // ====================================================================
690
+ $codeBlocks = [];
691
+ $html = preg_replace_callback('/^```([a-zA-Z0-9_+-]*)\n([\s\S]*?)\n^```/m', function ($matches) use (&$codeBlocks) {
692
+ $lang = !empty($matches[1]) ? ' class="language-' . htmlspecialchars($matches[1], ENT_QUOTES, 'UTF-8') . '"' : '';
693
+ $code = htmlspecialchars($matches[2], ENT_QUOTES, 'UTF-8');
694
+ $placeholder = "\x02CB" . count($codeBlocks) . "\x03";
695
+ $codeBlocks[$placeholder] = "<pre><code{$lang}>{$code}</code></pre>";
696
+ return $placeholder;
697
+ }, $html);
698
+
699
+ // ====================================================================
700
+ // STEP 3a: INDENTED CODE BLOCKS (4 spaces or 1 tab)
701
+ // Recognized only when preceded by a blank line (or the start of the
702
+ // document) and followed by a blank line (or the end of the document),
703
+ // to avoid conflicts with the indentation of nested lists.
704
+ // ====================================================================
705
+ $html = preg_replace_callback(
706
+ '/(?<=\n\n|^)((?:[ ]{4}|\t)[^\n]*(?:\n(?:[ ]{4}|\t)[^\n]*)*)(?=\n\n|\n*$)/',
707
+ function ($matches) use (&$codeBlocks) {
708
+ $lines = explode("\n", $matches[1]);
709
+ $stripped = array_map(static function (string $l): string {
710
+ return preg_replace('/^(?:[ ]{4}|\t)/', '', $l);
711
+ }, $lines);
712
+ $code = htmlspecialchars(implode("\n", $stripped), ENT_QUOTES, 'UTF-8');
713
+ $placeholder = "\x02CB" . count($codeBlocks) . "\x03";
714
+ $codeBlocks[$placeholder] = "<pre><code>{$code}</code></pre>";
715
+ return $placeholder;
716
+ },
717
+ $html
718
+ );
719
+
720
+ // Inline code with double backticks (lets you include a literal backtick)
721
+ $inlineCodes = [];
722
+ $html = preg_replace_callback('/``(.+?)``/s', function ($matches) use (&$inlineCodes) {
723
+ $content = $matches[1];
724
+ // Standard convention: if the content starts and ends with a space
725
+ // (and isn't only spaces), strip one space on each side — handy for
726
+ // wrapping a ` at the edge.
727
+ if (preg_match('/^ (.*[^ ]) $/s', $content, $trim)) {
728
+ $content = $trim[1];
729
+ }
730
+ $code = htmlspecialchars($content, ENT_QUOTES, 'UTF-8');
731
+ $placeholder = "\x02IC" . count($inlineCodes) . "\x03";
732
+ $inlineCodes[$placeholder] = "<code>{$code}</code>";
733
+ return $placeholder;
734
+ }, $html);
735
+
736
+ // Inline code (`...`)
737
+ $html = preg_replace_callback('/`([^`\n]+)`/', function ($matches) use (&$inlineCodes) {
738
+ $code = htmlspecialchars($matches[1], ENT_QUOTES, 'UTF-8');
739
+ $placeholder = "\x02IC" . count($inlineCodes) . "\x03";
740
+ $inlineCodes[$placeholder] = "<code>{$code}</code>";
741
+ return $placeholder;
742
+ }, $html);
743
+
744
+ // A `{% tag %}` sitting inside a code span or code block was captured
745
+ // by STEP 2 and is now a plugin placeholder embedded in the stored
746
+ // code. Swap those back for the literal (escaped) tag source so code
747
+ // shows `{% tag %}` verbatim instead of its rendered output — or a
748
+ // stray control-char placeholder that never gets restored.
749
+ if ($pluginLiterals) {
750
+ foreach ($codeBlocks as $k => $v) $codeBlocks[$k] = strtr($v, $pluginLiterals);
751
+ foreach ($inlineCodes as $k => $v) $inlineCodes[$k] = strtr($v, $pluginLiterals);
752
+ }
753
+
754
+
755
+ // ====================================================================
756
+ // STEP 3b: CHARACTER ESCAPING (\* \_ \# etc.)
757
+ // Processed after code extraction (code stays literal) and before
758
+ // everything else, so that \* doesn't open emphasis, \# doesn't create
759
+ // a heading, \- doesn't create a list, etc.
760
+ // ====================================================================
761
+ $escapes = [];
762
+ $html = preg_replace_callback(
763
+ '/\\\\([\\\\`*_{}\[\]<>()#+\-.!|])/',
764
+ function ($m) use (&$escapes): string {
765
+ $placeholder = "\x02ESC" . count($escapes) . "\x03";
766
+ $escapes[$placeholder] = htmlspecialchars($m[1], ENT_QUOTES, 'UTF-8');
767
+ return $placeholder;
768
+ },
769
+ $html
770
+ );
771
+ // &#124; is the documented convention (Markdown Extra / PHP Markdown)
772
+ // for showing a literal pipe in a table cell without it being
773
+ // interpreted as a column separator.
774
+ $html = preg_replace_callback(
775
+ '/&#124;/i',
776
+ function () use (&$escapes): string {
777
+ $placeholder = "\x02ESC" . count($escapes) . "\x03";
778
+ $escapes[$placeholder] = '|';
779
+ return $placeholder;
780
+ },
781
+ $html
782
+ );
783
+
784
+
785
+ // ====================================================================
786
+ // STEP 3d: AUTOMATIC LINKS <https://...> and <email@example.com>
787
+ // Processed before XSS encoding because the < > characters would be
788
+ // encoded to &lt; &gt; and the regex would no longer match.
789
+ // ====================================================================
790
+ $autolinks = [];
791
+ $html = preg_replace_callback('/<(https?:\/\/[^\s<>]+)>/', function ($m) use (&$autolinks): string {
792
+ $url = htmlspecialchars($m[1], ENT_QUOTES, 'UTF-8');
793
+ $placeholder = "\x02AL" . count($autolinks) . "\x03";
794
+ $autolinks[$placeholder] = "<a href=\"{$url}\" target=\"_blank\" rel=\"noopener noreferrer\">{$url}</a>";
795
+ return $placeholder;
796
+ }, $html);
797
+ $html = preg_replace_callback('/<([^\s<>]+@[^\s<>]+\.[^\s<>]+)>/', function ($m) use (&$autolinks): string {
798
+ $email = htmlspecialchars($m[1], ENT_QUOTES, 'UTF-8');
799
+ $placeholder = "\x02AL" . count($autolinks) . "\x03";
800
+ $autolinks[$placeholder] = "<a href=\"mailto:{$email}\">{$email}</a>";
801
+ return $placeholder;
802
+ }, $html);
803
+
804
+
805
+ // ====================================================================
806
+ // STEP 3e: GFM ALERTS AND BLOCKQUOTES
807
+ // Processed before XSS encoding because the > character would be
808
+ // encoded to &gt; and the regexes would no longer match.
809
+ // ====================================================================
810
+ $blockquotes = [];
811
+
812
+ // GFM alerts (> [!NOTE], etc.) — more specific, processed first
813
+ $html = preg_replace_callback(
814
+ '/^(>\s*\[!(NOTE|TIP|IMPORTANT|WARNING|CAUTION)\]\n(?:>[ \t]?[^\n]*\n?)*)/m',
815
+ function ($matches) use (&$blockquotes): string {
816
+ $type = strtolower($matches[2]);
817
+ $label = htmlspecialchars($matches[2], ENT_QUOTES, 'UTF-8');
818
+ $content = preg_replace('/^>\s?\[!(?:NOTE|TIP|IMPORTANT|WARNING|CAUTION)\]\n?/m', '', $matches[1]);
819
+ $content = preg_replace('/^>[ \t]?/m', '', $content);
820
+ $content = self::toHtml(trim($content));
821
+ $placeholder = "\x02BQ" . count($blockquotes) . "\x03";
822
+ $blockquotes[$placeholder] = "<div class=\"markdown-alert markdown-alert-{$type}\">"
823
+ . "<p class=\"markdown-alert-title\">{$label}</p>"
824
+ . "{$content}</div>";
825
+ // The trailing \n consumed by the regex is re-injected after
826
+ // the placeholder so the following blank line doesn't merge
827
+ // with the placeholder's line (which would break, for example,
828
+ // detecting a Setext heading right after).
829
+ return $placeholder . (str_ends_with($matches[1], "\n") ? "\n" : '');
830
+ },
831
+ $html
832
+ );
833
+
834
+ // Standard blockquotes (nesting handled by recursion through toHtml,
835
+ // which re-applies this same rule to the content already stripped of
836
+ // one ">" level)
837
+ $html = preg_replace_callback('/^((?:>[ \t]?[^\n]*\n?)+)/m', function ($matches) use (&$blockquotes): string {
838
+ $content = preg_replace('/^>[ \t]?/m', '', $matches[1]);
839
+ // The two trailing spaces are left as-is: toHtml() handles them itself
840
+ $inner = self::toHtml(trim($content));
841
+ $placeholder = "\x02BQ" . count($blockquotes) . "\x03";
842
+ $blockquotes[$placeholder] = "<blockquote>{$inner}</blockquote>";
843
+ // See the comment above: preserve the trailing \n that was consumed.
844
+ return $placeholder . (str_ends_with($matches[1], "\n") ? "\n" : '');
845
+ }, $html);
846
+
847
+
848
+ // ====================================================================
849
+ // STEP 3f: RAW HTML (safe subset, GitHub README style)
850
+ // Processed before XSS encoding because the < > characters would be
851
+ // encoded to &lt; &gt; and no longer recognized as tags.
852
+ // The content between the tags is not swallowed: it stays in the
853
+ // stream and keeps being processed as normal markdown.
854
+ // ====================================================================
855
+
856
+ // Intrinsically dangerous elements: removed along with their content
857
+ // (script/style/iframe can embed JS or load a third-party page;
858
+ // form/button/textarea/select/option have no place in markdown
859
+ // content).
860
+ $html = preg_replace(
861
+ '/<(script|style|iframe|object|embed|noscript|template|form|button|textarea|select|option)\b[^>]*>[\s\S]*?<\/\1>/i',
862
+ '',
863
+ $html
864
+ );
865
+
866
+ $rawHtml = [];
867
+ $html = preg_replace_callback(
868
+ '/<!--[\s\S]*?-->|<\/?[a-zA-Z][a-zA-Z0-9-]*(?:\s+[a-zA-Z_:][a-zA-Z0-9_:.-]*(?:\s*=\s*(?:"[^"]*"|\'[^\']*\'|[^\s"\'>]+))?)*\s*\/?>/',
869
+ function ($m) use (&$rawHtml): string {
870
+ $tag = $m[0];
871
+ // HTML comment: invisible, safe to remove.
872
+ if (str_starts_with($tag, '<!--')) return '';
873
+
874
+ $sanitized = self::sanitizeHtmlTag($tag);
875
+ if ($sanitized === null || $sanitized === '') return '';
876
+
877
+ $placeholder = "\x02HT" . count($rawHtml) . "\x03";
878
+ $rawHtml[$placeholder] = $sanitized;
879
+ return $placeholder;
880
+ },
881
+ $html
882
+ );
883
+
884
+
885
+ // ====================================================================
886
+ // STEP 4: Global XSS encoding
887
+ // ====================================================================
888
+ $html = htmlspecialchars($html, ENT_NOQUOTES, 'UTF-8');
889
+
890
+
891
+ // ====================================================================
892
+ // STEP 5: GFM TABLES
893
+ // Supports rows with or without a trailing pipe (| col | or | col)
894
+ // ====================================================================
895
+ $html = preg_replace_callback(
896
+ '/^(\|[^\n]+\|?\n)([ \t]*\|[ \t]*:?-+:?[ \t]*(?:\|[ \t]*:?-+:?[ \t]*)*\|?\n)((?:\|[^\n]+\|?\n?)+)/m',
897
+ function ($matches) {
898
+ $parseRow = function (string $line): array {
899
+ return array_values(array_filter(
900
+ array_map('trim', explode('|', trim($line, "| \t\n")))
901
+ ));
902
+ };
903
+
904
+ $headers = $parseRow($matches[1]);
905
+ $alignments = [];
906
+ $sepCells = $parseRow($matches[2]);
907
+ foreach ($sepCells as $sep) {
908
+ $left = str_starts_with(trim($sep), ':');
909
+ $right = str_ends_with(trim($sep), ':');
910
+ if ($left && $right) $alignments[] = ' style="text-align:center"';
911
+ elseif ($right) $alignments[] = ' style="text-align:right"';
912
+ elseif ($left) $alignments[] = ' style="text-align:left"';
913
+ else $alignments[] = '';
914
+ }
915
+
916
+ $out = "<table>\n <thead>\n <tr>\n";
917
+ foreach ($headers as $i => $header) {
918
+ $align = $alignments[$i] ?? '';
919
+ $out .= " <th{$align}>{$header}</th>\n";
920
+ }
921
+ $out .= " </tr>\n </thead>\n <tbody>\n";
922
+
923
+ $bodyLines = array_filter(explode("\n", trim($matches[3])));
924
+ foreach ($bodyLines as $line) {
925
+ $cells = $parseRow($line);
926
+ $out .= " <tr>\n";
927
+ foreach ($cells as $i => $cell) {
928
+ $align = $alignments[$i] ?? '';
929
+ $out .= " <td{$align}>{$cell}</td>\n";
930
+ }
931
+ $out .= " </tr>\n";
932
+ }
933
+ $out .= " </tbody>\n</table>";
934
+ return $out;
935
+ },
936
+ $html
937
+ );
938
+
939
+
940
+ // ====================================================================
941
+ // STEP 6: (GFM alerts and blockquotes handled in step 3e)
942
+ // ====================================================================
943
+
944
+
945
+ // ====================================================================
946
+ // STEP 7: TASK LISTS (GFM checkboxes)
947
+ // ====================================================================
948
+ $html = preg_replace('/^[ \t]*[-*+] \[ \] (.+)$/m', '<li class="task-item"><input type="checkbox" disabled /> $1</li>', $html);
949
+ $html = preg_replace('/^[ \t]*[-*+] \[[xX]\] (.+)$/m', '<li class="task-item"><input type="checkbox" checked disabled /> $1</li>', $html);
950
+
951
+
952
+ // ====================================================================
953
+ // STEP 7b: SETEXT HEADINGS (alternative == / -- syntax)
954
+ // Title
955
+ // ===== → <h1>
956
+ //
957
+ // Title
958
+ // ----- → <h2>
959
+ // Processed before ATX headings and before horizontal rules (a line of
960
+ // dashes right after a line of text is a heading, not an <hr>).
961
+ // ====================================================================
962
+ $html = preg_replace_callback(
963
+ '/^(?![ \t]*(?:#{1,6}[ \t]|>|```|\||[-*+][ \t]|\d+\.[ \t]))[ \t]*(\S.*?)[ \t]*(?:\{#([a-zA-Z0-9_\-:.]+)\}[ \t]*)?\n[ \t]*=+[ \t]*$/m',
964
+ function ($matches) {
965
+ $text = trim($matches[1]);
966
+ $id = !empty($matches[2]) ? $matches[2] : self::slugify($text);
967
+ return "<h1 id=\"{$id}\">{$text}</h1>";
968
+ },
969
+ $html
970
+ );
971
+ $html = preg_replace_callback(
972
+ '/^(?![ \t]*(?:#{1,6}[ \t]|>|```|\||[-*+][ \t]|\d+\.[ \t]))[ \t]*(\S.*?)[ \t]*(?:\{#([a-zA-Z0-9_\-:.]+)\}[ \t]*)?\n[ \t]*-+[ \t]*$/m',
973
+ function ($matches) {
974
+ $text = trim($matches[1]);
975
+ $id = !empty($matches[2]) ? $matches[2] : self::slugify($text);
976
+ return "<h2 id=\"{$id}\">{$text}</h2>";
977
+ },
978
+ $html
979
+ );
980
+
981
+
982
+ // ====================================================================
983
+ // STEP 8: HEADINGS (ATX: # to ######)
984
+ // ====================================================================
985
+ $html = preg_replace_callback(
986
+ '/^(#{1,6})[ \t]+(.+?)[ \t]*(?:\{#([a-zA-Z0-9_\-:.]+)\}[ \t]*)?(?:[ \t]+#+)?$/m',
987
+ function ($matches) {
988
+ $level = strlen($matches[1]);
989
+ $text = trim($matches[2]);
990
+ $id = !empty($matches[3]) ? $matches[3] : self::slugify($text);
991
+ return "<h{$level} id=\"{$id}\">{$text}</h{$level}>";
992
+ },
993
+ $html
994
+ );
995
+
996
+
997
+ // ====================================================================
998
+ // STEP 9: LISTS (bullets and ordered, with nesting)
999
+ // A single pass detects a contiguous block of lines that are either a
1000
+ // bullet (-,*,+) or a numbered item, whatever their indentation level;
1001
+ // the block is then rebuilt recursively into nested <ol>/<ul>
1002
+ // according to the relative indentation depth. An indented line with
1003
+ // no marker of its own is a *continuation* of the previous item's
1004
+ // text (a soft-wrapped source line) and is appended to it, not
1005
+ // treated as the block's end — a marker-less continuation line used
1006
+ // to fall outside the match entirely, splitting one list into two
1007
+ // with the continuation text stranded as a stray <p> in between.
1008
+ // Task items (already converted to <li class="task-item">) no longer
1009
+ // match this pattern and are therefore not re-wrapped here.
1010
+ // ====================================================================
1011
+ $html = preg_replace_callback(
1012
+ '/^([ \t]*(?:\d+\.|[-*+])[ \t]+.+(?:\n(?:[ \t]*(?:\d+\.|[-*+])[ \t]+.+|[ \t]+\S.*))*)/m',
1013
+ function ($matches) {
1014
+ $lines = explode("\n", $matches[1]);
1015
+ $items = [];
1016
+ foreach ($lines as $line) {
1017
+ if (preg_match('/^([ \t]*)(\d+)\.[ \t]+(.*)$/', $line, $m)) {
1018
+ $items[] = ['indent' => self::indentWidth($m[1]), 'type' => 'ol', 'text' => $m[3]];
1019
+ } elseif (preg_match('/^([ \t]*)[-*+][ \t]+(.*)$/', $line, $m)) {
1020
+ $items[] = ['indent' => self::indentWidth($m[1]), 'type' => 'ul', 'text' => $m[2]];
1021
+ } elseif (!empty($items) && preg_match('/^[ \t]+(\S.*)$/', $line, $m)) {
1022
+ $items[count($items) - 1]['text'] .= ' ' . $m[1];
1023
+ }
1024
+ }
1025
+ if (empty($items)) return $matches[1];
1026
+ // Normalize the lowest indentation level to 0
1027
+ $minIndent = min(array_column($items, 'indent'));
1028
+ foreach ($items as &$it) $it['indent'] -= $minIndent;
1029
+ unset($it);
1030
+
1031
+ $i = 0;
1032
+ return self::buildListTree($items, $i, count($items));
1033
+ },
1034
+ $html
1035
+ );
1036
+
1037
+ $html = preg_replace_callback(
1038
+ '/(?:<li class="task-item">.*<\/li>\n?)+/s',
1039
+ function ($matches) {
1040
+ return "<ul class=\"task-list\">\n" . $matches[0] . "</ul>\n";
1041
+ },
1042
+ $html
1043
+ );
1044
+
1045
+
1046
+ // ====================================================================
1047
+ // STEP 9b: DEFINITION LISTS (extended syntax)
1048
+ // Term
1049
+ // : Definition
1050
+ // ====================================================================
1051
+ $html = self::extractDefinitionLists($html);
1052
+
1053
+
1054
+ // ====================================================================
1055
+ // STEP 9c: FOOTNOTE REFERENCES [^label]
1056
+ // Converted BEFORE emphasis so they don't collide with the new
1057
+ // superscript ^text^ (a [^1] followed later by a [^2] on the same line
1058
+ // could otherwise be read as ^1] ... [^2^).
1059
+ // Numbering is sequential, in order of first appearance in the text
1060
+ // (as documented).
1061
+ // ====================================================================
1062
+ $footnoteOrder = [];
1063
+ $html = preg_replace_callback('/\[\^([^\]\s]+)\]/', function ($m) use (&$footnoteOrder, &$footnoteDefs): string {
1064
+ $label = strtolower(trim($m[1]));
1065
+ if (!isset($footnoteDefs[$label])) {
1066
+ // Reference to an undefined note: left as-is.
1067
+ return $m[0];
1068
+ }
1069
+ if (!isset($footnoteOrder[$label])) {
1070
+ $footnoteOrder[$label] = count($footnoteOrder) + 1;
1071
+ }
1072
+ $num = $footnoteOrder[$label];
1073
+ return "<sup id=\"fnref:{$label}\"><a href=\"#fn:{$label}\">{$num}</a></sup>";
1074
+ }, $html);
1075
+
1076
+
1077
+ // ====================================================================
1078
+ // STEP 10: INLINE TEXT (Bold, Italic, Strikethrough, Highlight,
1079
+ // Subscript/Superscript, Emoji)
1080
+ // ====================================================================
1081
+ $html = preg_replace('/\*\*\*(.+?)\*\*\*/s', '<strong><em>$1</em></strong>', $html);
1082
+ $html = preg_replace('/___(.+?)___/s', '<strong><em>$1</em></strong>', $html);
1083
+ $html = preg_replace('/\*\*(.+?)\*\*/s', '<strong>$1</strong>', $html);
1084
+ $html = preg_replace('/__(.+?)__/s', '<strong>$1</strong>', $html);
1085
+ $html = preg_replace('/\*(.+?)\*/s', '<em>$1</em>', $html);
1086
+ // Italic _ must only match at word boundaries so it doesn't capture
1087
+ // snake_case, package names (@php-wasm/node), etc.
1088
+ $html = preg_replace('/(?<!\w)_([^_\n]+)_(?!\w)/', '<em>$1</em>', $html);
1089
+ // Highlight ==text== (extended syntax)
1090
+ $html = preg_replace('/==(.+?)==/s', '<mark>$1</mark>', $html);
1091
+ // Strikethrough ~~text~~ — processed BEFORE subscript (single ~) so the
1092
+ // latter doesn't match half of a double-tilde pair.
1093
+ $html = preg_replace('/~~(.+?)~~/s', '<del>$1</del>', $html);
1094
+ // Superscript ^text^ (extended syntax) — placing it before the note
1095
+ // reference escaping ([^label]) is not a problem: those are wrapped in
1096
+ // brackets and so don't form an isolated ^...^ pair.
1097
+ $html = preg_replace('/\^([^\^\n]+)\^/', '<sup>$1</sup>', $html);
1098
+ // Subscript ~text~ (a single tilde; the ~~ were already consumed just
1099
+ // above by strikethrough).
1100
+ $html = preg_replace('/~([^~\n]+)~/', '<sub>$1</sub>', $html);
1101
+
1102
+ // Emojis :shortcode: (extended syntax) — unknown shortcuts are left
1103
+ // as-is rather than silently removed.
1104
+ $html = preg_replace_callback('/:([a-zA-Z0-9_+\-]+):/', function ($m): string {
1105
+ $emoji = self::emojiFor($m[1]);
1106
+ return $emoji ?? $m[0];
1107
+ }, $html);
1108
+
1109
+
1110
+ // ====================================================================
1111
+ // STEP 11: LINKS & IMAGES
1112
+ // External links (https?://) get target="_blank" + rel="noopener noreferrer".
1113
+ // Internal links (/page, #anchor, ../thing) don't.
1114
+ // ====================================================================
1115
+ $html = preg_replace(
1116
+ '/!\[([^\]]*)\]\(([^)\s]+)(?:\s+"([^"]*)")?\)/',
1117
+ '<img src="$2" alt="$1" title="$3" loading="lazy" />',
1118
+ $html
1119
+ );
1120
+
1121
+ $buildLink = static function (string $text, string $href, string $title): string {
1122
+ $titleAttr = $title !== '' ? ' title="' . $title . '"' : '';
1123
+ $extern = preg_match('/^https?:\/\//i', $href)
1124
+ ? ' target="_blank" rel="noopener noreferrer"'
1125
+ : '';
1126
+ return "<a href=\"{$href}\"{$titleAttr}{$extern}>{$text}</a>";
1127
+ };
1128
+
1129
+ // Reference links [text][label] and [text][] (shortcut = label = text)
1130
+ $html = preg_replace_callback(
1131
+ '/\[([^\]]+)\]\[([^\]]*)\]/',
1132
+ function ($m) use (&$refDefs, $buildLink): string {
1133
+ $text = $m[1];
1134
+ $label = strtolower(trim($m[2] !== '' ? $m[2] : $m[1]));
1135
+ if (!isset($refDefs[$label])) return $m[0];
1136
+ $def = $refDefs[$label];
1137
+ return $buildLink($text, $def['url'], $def['title']);
1138
+ },
1139
+ $html
1140
+ );
1141
+
1142
+ // Markdown links [text](url "optional title")
1143
+ $html = preg_replace_callback(
1144
+ '/\[([^\]]+)\]\(([^)\s]+)(?:\s+"([^"]*)")?\)/',
1145
+ function ($m) use ($buildLink): string {
1146
+ return $buildLink($m[1], $m[2], $m[3] ?? '');
1147
+ },
1148
+ $html
1149
+ );
1150
+
1151
+ // Bare URLs https://... (extended syntax: auto-link without brackets).
1152
+ // Excludes those already inside quotes/attributes (href="...") or
1153
+ // already turned into a link, so they don't get doubled.
1154
+ $html = preg_replace(
1155
+ '/(?<!["\'=>])\b(https?:\/\/[^\s<>"\')\]]+)/',
1156
+ '<a href="$1" target="_blank" rel="noopener noreferrer">$1</a>',
1157
+ $html
1158
+ );
1159
+
1160
+
1161
+ // ====================================================================
1162
+ // STEP 12: HORIZONTAL RULES
1163
+ // ====================================================================
1164
+ $html = preg_replace('/^(?:[-*_][ \t]*){3,}$/m', '<hr />', $html);
1165
+
1166
+
1167
+ // ====================================================================
1168
+ // STEP 13: PARAGRAPHS
1169
+ // Strategy: process line by line. Lines that start with a block-level
1170
+ // tag or a placeholder are left as-is. Consecutive raw-text lines are
1171
+ // accumulated then wrapped in a <p> when a block line or a blank line
1172
+ // is reached.
1173
+ // ====================================================================
1174
+ $blockStartTags = ['<h', '<pre', '<ul', '<ol', '<li', '<table', '<thead', '<tbody',
1175
+ '<tr', '<td', '<th', '<blockquote', '<div', '<hr', '<img',
1176
+ '<dl', '<dt', '<dd',
1177
+ "\x02CB", "\x02PLG", "\x02BQ", "\x02HT"];
1178
+
1179
+ $isBlockLine = static function (string $line) use ($blockStartTags): bool {
1180
+ $t = ltrim($line);
1181
+ if ($t === '') return false;
1182
+ // Any closing tag (</...>) is always treated as a "block" line:
1183
+ // this keeps a closing </table>, </thead>, </tr>, etc. from being
1184
+ // absorbed into a surrounding <p>.
1185
+ if (str_starts_with($t, '</')) return true;
1186
+ foreach ($blockStartTags as $tag) {
1187
+ if (str_starts_with($t, $tag)) return true;
1188
+ }
1189
+ return false;
1190
+ };
1191
+
1192
+ $lines = explode("\n", $html);
1193
+ $output = [];
1194
+ $textBuffer = [];
1195
+
1196
+ $flushBuffer = static function () use (&$textBuffer, &$output): void {
1197
+ if (empty($textBuffer)) return;
1198
+ $content = implode("\n", $textBuffer);
1199
+ if (trim($content) !== '') {
1200
+ // Two trailing spaces → <br> (standard markdown convention)
1201
+ $content = preg_replace('/ $/m', '<br>', $content);
1202
+ // Single line break → space (GitHub behavior)
1203
+ // Unless already converted to <br> above
1204
+ $content = preg_replace('/(?<!r>)\n/', ' ', $content);
1205
+ $output[] = '<p>' . trim($content) . '</p>';
1206
+ }
1207
+ $textBuffer = [];
1208
+ };
1209
+
1210
+ foreach ($lines as $line) {
1211
+ if ($isBlockLine($line)) {
1212
+ $flushBuffer();
1213
+ $output[] = $line;
1214
+ } elseif (trim($line) === '') {
1215
+ // Blank line = paragraph separator
1216
+ $flushBuffer();
1217
+ } else {
1218
+ $textBuffer[] = $line;
1219
+ }
1220
+ }
1221
+ $flushBuffer();
1222
+
1223
+ $html = implode("\n", $output);
1224
+
1225
+
1226
+ // ====================================================================
1227
+ // STEP 14: Re-inject the placeholders
1228
+ // ====================================================================
1229
+ $html = strtr($html, $pluginBlocks);
1230
+ $html = strtr($html, $blockquotes);
1231
+ $html = strtr($html, $rawHtml);
1232
+ $html = strtr($html, $codeBlocks);
1233
+ $html = strtr($html, $inlineCodes);
1234
+ $html = strtr($html, $autolinks);
1235
+ // The escapes are re-injected last, once no Markdown regex can
1236
+ // interpret them anymore.
1237
+ $html = strtr($html, $escapes);
1238
+
1239
+
1240
+ // ====================================================================
1241
+ // STEP 15: FOOTNOTES BLOCK
1242
+ // Appended at the end of the document, only if at least one note was
1243
+ // referenced (notes that are defined but never referenced are
1244
+ // silently ignored).
1245
+ // ====================================================================
1246
+ if (!empty($footnoteOrder)) {
1247
+ $html .= "\n<div class=\"footnotes\">\n<ol>\n";
1248
+ foreach ($footnoteOrder as $label => $num) {
1249
+ $content = $footnoteDefs[$label];
1250
+ $html .= " <li id=\"fn:{$label}\">{$content} <a href=\"#fnref:{$label}\" class=\"footnote-backref\">↩</a></li>\n";
1251
+ }
1252
+ $html .= "</ol>\n</div>";
1253
+ }
1254
+
1255
+ return $html;
1256
+ }
1257
+ }