@kirigami/php-prepros 1.1.0 → 1.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -3,22 +3,22 @@
3
3
  class MD {
4
4
 
5
5
  // ========================================================================
6
- // SYSTÈME DE PLUGINS
6
+ // PLUGIN SYSTEM
7
7
  //
8
- // SYNTAXE INLINE (args sur la même ligne) :
9
- // {% nom_plugin arg1 arg2 "arg avec espaces" %}
8
+ // INLINE SYNTAX (args on the same line):
9
+ // {% plugin_name arg1 arg2 "arg with spaces" %}
10
10
  //
11
- // SYNTAXE BLOC (contenu multi-ligne) :
12
- // {% nom_plugin arg1 arg2
13
- // ligne de contenu 1
14
- // ligne de contenu 2
11
+ // BLOCK SYNTAX (multi-line content):
12
+ // {% plugin_name arg1 arg2
13
+ // content line 1
14
+ // content line 2
15
15
  // %}
16
16
  //
17
- // Le callback reçoit toujours (array $args, string $body) :
18
- // - $args : tableau des arguments passés sur la ligne d'ouverture
19
- // - $body : contenu multi-ligne (vide "" pour les tags inline)
17
+ // The callback always receives (array $args, string $body):
18
+ // - $args : array of the arguments passed on the opening line
19
+ // - $body : multi-line content (empty "" for inline tags)
20
20
  //
21
- // Exemples :
21
+ // Examples:
22
22
  // GithubReadmeParser::registerPlugin('codepen', function(array $args, string $body): string {
23
23
  // $id = htmlspecialchars($args[0] ?? '', ENT_QUOTES, 'UTF-8');
24
24
  // return "<iframe src=\"https://codepen.io/embed/{$id}\"></iframe>";
@@ -38,25 +38,25 @@ class MD {
38
38
  private static array $plugins = [];
39
39
 
40
40
  /**
41
- * Enregistre un plugin par son nom.
41
+ * Registers a plugin by name.
42
42
  *
43
- * @param string $name Nom du tag, ex: "codepen"
43
+ * @param string $name Tag name, e.g. "codepen"
44
44
  * @param callable $callback function(array $args): string
45
- * $args[0] = premier argument, $args[1] = second, etc.
45
+ * $args[0] = first argument, $args[1] = second, etc.
46
46
  */
47
47
  public static function registerPlugin(string $name, callable $callback): void {
48
48
  self::$plugins[strtolower(trim($name))] = $callback;
49
49
  }
50
50
 
51
51
  /**
52
- * Supprime un plugin enregistré.
52
+ * Removes a registered plugin.
53
53
  */
54
54
  public static function unregisterPlugin(string $name): void {
55
55
  unset(self::$plugins[strtolower(trim($name))]);
56
56
  }
57
57
 
58
58
  /**
59
- * Retourne la liste des plugins enregistrés.
59
+ * Returns the list of registered plugins.
60
60
  *
61
61
  * @return string[]
62
62
  */
@@ -65,7 +65,7 @@ class MD {
65
65
  }
66
66
 
67
67
  // ========================================================================
68
- // Génère un id de type "slug" pour les ancres de titres (ATX et Setext).
68
+ // Generates a "slug"-style id for heading anchors (ATX and Setext).
69
69
  // ========================================================================
70
70
  private static function slugify(string $text): string {
71
71
  $id = strtolower(preg_replace('/[^\w\- ]/u', '', $text));
@@ -73,17 +73,16 @@ class MD {
73
73
  }
74
74
 
75
75
  // ========================================================================
76
- // Convertit une largeur d'indentation (espaces/tabs) en nombre de colonnes,
77
- // une tabulation comptant pour 4 espaces.
76
+ // Converts an indentation width (spaces/tabs) into a column count, a tab
77
+ // counting as 4 spaces.
78
78
  // ========================================================================
79
79
  private static function indentWidth(string $whitespace): int {
80
80
  return strlen(str_replace("\t", ' ', $whitespace));
81
81
  }
82
82
 
83
83
  /**
84
- * Construit récursivement une liste (imbriquée) <ol>/<ul> à partir d'un
85
- * tableau plat d'items { indent, type, text }. $i est avancé au fur et à
86
- * mesure de la consommation des items.
84
+ * Recursively builds a (nested) <ol>/<ul> list from a flat array of items
85
+ * { indent, type, text }. $i is advanced as items are consumed.
87
86
  *
88
87
  * @param array<int, array{indent:int, type:string, text:string}> $items
89
88
  */
@@ -108,9 +107,9 @@ class MD {
108
107
  }
109
108
 
110
109
  // ========================================================================
111
- // ÉMOJIS (syntaxe étendue) : :shortcode: → caractère unicode.
112
- // Table non exhaustive mais couvrant les raccourcis les plus courants ;
113
- // extensible via registerEmoji().
110
+ // EMOJIS (extended syntax): :shortcode: → unicode character.
111
+ // Non-exhaustive table but covering the most common shortcuts; extensible
112
+ // via registerEmoji().
114
113
  // ========================================================================
115
114
  /** @var array<string, string> */
116
115
  private static array $extraEmoji = [];
@@ -191,10 +190,145 @@ class MD {
191
190
  'speech_balloon' => '💬', 'thought_balloon' => '💭', 'zzz' => '💤', 'boom2' => '💥',
192
191
  'sos' => '🆘', 'new' => '🆕', 'ok' => '🆗', 'up' => '🆙', 'cool' => '🆒',
193
192
  'free' => '🆓', 'id' => '🆔', 'ng' => '🆖',
193
+
194
+ // -- Faces and emotions (continued) --------------------------------
195
+ 'smiling_face_with_three_hearts' => '🥰', 'kissing' => '😗', 'kissing_closed_eyes' => '😚',
196
+ 'kissing_smiling_eyes' => '😙', 'yum' => '😋', 'stuck_out_tongue' => '😛',
197
+ 'stuck_out_tongue_winking_eye' => '😜', 'stuck_out_tongue_closed_eyes' => '😝',
198
+ 'money_mouth_face' => '🤑', 'hugs' => '🤗', 'disappointed_relieved' => '😥',
199
+ 'dizzy_face' => '😵', 'astonished' => '😲', 'open_mouth' => '😮', 'hushed' => '😯',
200
+ 'fearful' => '😨', 'cold_sweat' => '😰', 'nauseated_face' => '🤢', 'vomiting_face' => '🤮',
201
+ 'sneezing_face' => '🤧', 'face_with_thermometer' => '🤒', 'face_with_head_bandage' => '🤕',
202
+ 'woozy_face' => '🥴', 'smiling_imp' => '😈', 'imp' => '👿', 'japanese_ogre' => '👹',
203
+ 'japanese_goblin' => '👺', 'skull' => '💀', 'skull_and_crossbones' => '☠️',
204
+ 'ghost' => '👻', 'alien' => '👽', 'space_invader' => '👾', 'robot' => '🤖',
205
+ 'poop' => '💩', 'clown_face' => '🤡', 'smiley_cat' => '😺', 'smile_cat' => '😸',
206
+ 'joy_cat' => '😹', 'heart_eyes_cat' => '😻', 'smirk_cat' => '😼', 'kissing_cat' => '😽',
207
+ 'pouting_cat' => '😾', 'crying_cat_face' => '😿',
208
+
209
+ // -- Corps, gestes, personnages --------------------------------
210
+ 'raised_hand' => '✋', 'raised_back_of_hand' => '🤚', 'vulcan_salute' => '🖖',
211
+ 'pinching_hand' => '🤏', 'fist' => '✊', 'punch' => '👊', 'left_facing_fist' => '🤛',
212
+ 'right_facing_fist' => '🤜', 'open_hands' => '👐', 'palms_up_together' => '🤲',
213
+ 'nail_care' => '💅', 'selfie' => '🤳', 'ear' => '👂', 'nose' => '👃', 'brain' => '🧠',
214
+ 'tongue' => '👅', 'lips' => '👄', 'tooth' => '🦷', 'bone' => '🦴',
215
+ 'baby' => '👶', 'child' => '🧒', 'boy' => '👦', 'girl' => '👧', 'adult' => '🧑',
216
+ 'man' => '👨', 'woman' => '👩', 'older_adult' => '🧓', 'older_man' => '👴', 'older_woman' => '👵',
217
+ 'mage' => '🧙', 'superhero' => '🦸', 'supervillain' => '🦹', 'vampire' => '🧛',
218
+ 'zombie' => '🧟', 'genie' => '🧞', 'merperson' => '🧜', 'elf' => '🧝', 'fairy' => '🧚',
219
+
220
+ // -- Animaux (suite) ---------------------------------------------
221
+ 'wolf' => '🐺', 'boar' => '🐗', 'racehorse' => '🐎', 'zebra' => '🦓', 'deer' => '🦌',
222
+ 'cow2' => '🐄', 'ox' => '🐂', 'water_buffalo' => '🐃', 'pig2' => '🐖', 'ram' => '🐏',
223
+ 'sheep' => '🐑', 'goat' => '🐐', 'camel' => '🐫', 'dromedary_camel' => '🐪',
224
+ 'llama' => '🦙', 'giraffe' => '🦒', 'elephant' => '🐘', 'rhinoceros' => '🦏',
225
+ 'hippopotamus' => '🦛', 'mouse2' => '🐁', 'rat' => '🐀', 'hamster' => '🐹',
226
+ 'chipmunk' => '🐿️', 'hedgehog' => '🦔', 'bat' => '🦇', 'duck' => '🦆', 'eagle' => '🦅',
227
+ 'flamingo' => '🦩', 'peacock' => '🦚', 'parrot' => '🦜', 'swan' => '🦢',
228
+ 'turkey' => '🦃', 'dove' => '🕊️', 'rooster' => '🐓', 'crocodile' => '🐊',
229
+ 'turtle' => '🐢', 'lizard' => '🦎', 'snake' => '🐍', 'dragon_face' => '🐲',
230
+ 'dragon' => '🐉', 'sauropod' => '🦕', 't-rex' => '🦖', 'whale2' => '🐋',
231
+ 'shark' => '🦈', 'seal' => '🦭', 'squid' => '🦑', 'shrimp' => '🦐', 'lobster' => '🦞',
232
+ 'crab' => '🦀', 'blowfish' => '🐡', 'tropical_fish' => '🐠', 'oyster' => '🦪',
233
+ 'ant' => '🐜', 'spider' => '🕷️', 'spider_web' => '🕸️', 'scorpion' => '🦂',
234
+ 'mosquito' => '🦟', 'microbe' => '🦠', 'paw_prints' => '🐾',
235
+
236
+ // -- Nature, plants, weather (continued) --------------------------------
237
+ 'cherry_blossom' => '🌸', 'blossom' => '🌼', 'rose' => '🌹', 'wilted_flower' => '🥀',
238
+ 'hibiscus' => '🌺', 'sunflower' => '🌻', 'tulip' => '🌷', 'herb' => '🌿',
239
+ 'shamrock' => '☘️', 'fallen_leaf' => '🍂', 'leaves' => '🍃', 'mushroom' => '🍄',
240
+ 'chestnut' => '🌰', 'crescent_moon' => '🌙', 'full_moon' => '🌕', 'new_moon' => '🌑',
241
+ 'milky_way' => '🌌', 'stars' => '🌠', 'cyclone' => '🌀', 'fog' => '🌫️',
242
+ 'wind_face' => '🌬️', 'tornado' => '🌪️', 'thunder_cloud_and_rain' => '⛈️',
243
+ 'sweat_drops' => '💦', 'snowman' => '⛄', 'snowman_with_snow' => '☃️', 'comet' => '☄️',
244
+
245
+ // -- Nourriture (suite) --------------------------------------------
246
+ 'tomato' => '🍅', 'eggplant' => '🍆', 'avocado' => '🥑', 'broccoli' => '🥦',
247
+ 'carrot' => '🥕', 'corn' => '🌽', 'hot_pepper' => '🌶️', 'cucumber' => '🥒',
248
+ 'potato' => '🥔', 'sweet_potato' => '🍠', 'peanuts' => '🥜', 'honey_pot' => '🍯',
249
+ 'croissant' => '🥐', 'bagel' => '🥯', 'pretzel' => '🥨', 'pancakes' => '🥞',
250
+ 'waffle' => '🧇', 'meat_on_bone' => '🍖', 'poultry_leg' => '🍗', 'bacon' => '🥓',
251
+ 'sandwich' => '🥪', 'stuffed_flatbread' => '🥙', 'burrito' => '🌯', 'salad' => '🥗',
252
+ 'shallow_pan_of_food' => '🥘', 'canned_food' => '🥫', 'bento' => '🍱',
253
+ 'rice_ball' => '🍙', 'rice' => '🍚', 'curry' => '🍛', 'stew' => '🍲', 'oden' => '🍢',
254
+ 'dango' => '🍡', 'shaved_ice' => '🍧', 'ice_cream' => '🍨', 'pie' => '🥧',
255
+ 'cupcake' => '🧁', 'moon_cake' => '🥮', 'lollipop' => '🍭', 'custard' => '🍮',
256
+ 'milk_glass' => '🥛', 'baby_bottle' => '🍼', 'mate' => '🧉', 'ice_cube' => '🧊',
257
+ 'tumbler_glass' => '🥃', 'cup_with_straw' => '🥤', 'chopsticks' => '🥢',
258
+ 'fork_and_knife' => '🍴', 'spoon' => '🥄', 'plate_with_cutlery' => '🍽️',
259
+
260
+ // -- Activities, sport, leisure --------------------------------------
261
+ 'running' => '🏃', 'walking' => '🚶', 'swimming' => '🏊', 'surfing' => '🏄',
262
+ 'skateboard' => '🛹', 'snowboarder' => '🏂', 'weight_lifting' => '🏋️',
263
+ 'cyclist' => '🚴', 'medal_military' => '🎖️', 'ticket' => '🎫', 'circus_tent' => '🎪',
264
+ 'performing_arts' => '🎭', 'art' => '🎨', 'clapper' => '🎬', 'microphone' => '🎤',
265
+ 'headphones' => '🎧', 'musical_note' => '🎵', 'musical_score' => '🎼', 'guitar' => '🎸',
266
+ 'violin' => '🎻', 'drum' => '🥁', 'trumpet' => '🎺', 'saxophone' => '🎷',
267
+ 'musical_keyboard' => '🎹', 'chess_pawn' => '♟️', 'bowling' => '🎳',
268
+ 'ice_skate' => '⛸️', 'ski' => '🎿', 'fishing_pole_and_fish' => '🎣',
269
+ 'boxing_glove' => '🥊', 'martial_arts_uniform' => '🥋', 'goal_net' => '🥅',
270
+ 'flying_disc' => '🥏', 'yo_yo' => '🪀', 'kite' => '🪁',
271
+
272
+ // -- Voyages et lieux (suite) ----------------------------------------
273
+ 'airplane_departure' => '🛫', 'airplane_arriving' => '🛬', 'flying_saucer' => '🛸',
274
+ 'motorcycle' => '🏍️', 'scooter' => '🛴', 'tractor' => '🚜', 'truck' => '🚚',
275
+ 'articulated_lorry' => '🚛', 'trolleybus' => '🚎', 'minibus' => '🚐', 'metro' => '🚇',
276
+ 'station' => '🚉', 'monorail' => '🚝', 'bullettrain_front' => '🚄',
277
+ 'steam_locomotive' => '🚂', 'anchor' => '⚓', 'sailboat' => '⛵', 'canoe' => '🛶',
278
+ 'speedboat' => '🚤', 'ferry' => '⛴️', 'passport_control' => '🛂', 'customs' => '🛃',
279
+ 'baggage_claim' => '🛄', 'left_luggage' => '🛅', 'vertical_traffic_light' => '🚦',
280
+ 'construction' => '🚧', 'fuelpump' => '⛽', 'busstop' => '🚏', 'moyai' => '🗿',
281
+ 'statue_of_liberty' => '🗽', 'tokyo_tower' => '🗼', 'fountain' => '⛲',
282
+ 'stadium' => '🏟️', 'ferris_wheel' => '🎡', 'roller_coaster' => '🎢',
283
+ 'carousel_horse' => '🎠', 'beach_umbrella' => '🏖️', 'desert' => '🏜️',
284
+ 'desert_island' => '🏝️', 'national_park' => '🏞️', 'sunrise' => '🌅',
285
+ 'sunrise_over_mountains' => '🌄', 'sparkler' => '🎇', 'fireworks' => '🎆',
286
+ 'city_sunset' => '🌇', 'bridge_at_night' => '🌉', 'houses' => '🏘️',
287
+ 'derelict_house' => '🏚️', 'classical_building' => '🏛️', 'department_store' => '🏬',
288
+ 'post_office' => '🏣', 'hotel' => '🏨', 'convenience_store' => '🏪', 'bank' => '🏦',
289
+ 'factory' => '🏭',
290
+
291
+ // -- Objets (suite) ---------------------------------------------------
292
+ 'watch' => '⌚', 'stopwatch' => '⏱️', 'timer_clock' => '⏲️', 'joystick' => '🕹️',
293
+ 'floppy_disk' => '💾', 'cd' => '💿', 'dvd' => '📀', 'movie_camera' => '🎥',
294
+ 'projector' => '📽️', 'telephone' => '☎️', 'pager' => '📟', 'fax' => '📠',
295
+ 'candle' => '🕯️', 'fire_extinguisher' => '🧯', 'oil_drum' => '🛢️',
296
+ 'money_with_wings' => '💸', 'credit_card' => '💳', 'yen' => '💴', 'euro' => '💶',
297
+ 'pound' => '💷', 'briefcase' => '💼', 'balance_scale' => '⚖️', 'compass' => '🧭',
298
+ 'triangular_ruler' => '📐', 'straight_ruler' => '📏', 'round_pushpin' => '📍',
299
+ 'scissors' => '✂️', 'thread' => '🧵', 'yarn' => '🧶', 'safety_pin' => '🧷',
300
+ 'basket' => '🧺', 'hourglass_flowing_sand' => '⏳', 'notebook' => '📓',
301
+ 'notebook_with_decorative_cover' => '📔', 'page_facing_up' => '📄',
302
+ 'page_with_curl' => '📃', 'bookmark_tabs' => '📑', 'bookmark' => '🔖',
303
+ 'label' => '🏷️', 'receipt' => '🧾', 'card_index' => '📇', 'wastebasket' => '🗑️',
304
+ 'old_key' => '🗝️', 'hammer_and_wrench' => '🛠️', 'pick' => '⛏️', 'shield' => '🛡️',
305
+ 'syringe' => '💉', 'pill' => '💊', 'thermometer' => '🌡️', 'soap' => '🧼',
306
+ 'broom' => '🧹',
307
+
308
+ // -- Symboles (suite) -----------------------------------------------
309
+ 'heavy_multiplication_x' => '✖️', 'heavy_plus_sign' => '➕', 'heavy_minus_sign' => '➖',
310
+ 'heavy_division_sign' => '➗', 'infinity' => '♾️', 'recycle' => '♻️', 'trident' => '🔱',
311
+ 'atom_symbol' => '⚛️', 'om' => '🕉️', 'peace_symbol' => '☮️', 'yin_yang' => '☯️',
312
+ 'wheel_of_dharma' => '☸️', 'star_of_david' => '✡️', 'star_and_crescent' => '☪️',
313
+ 'cross' => '✝️', 'menorah' => '🕎', 'radioactive' => '☢️', 'biohazard' => '☣️',
314
+ 'arrow_up' => '⬆️', 'arrow_down' => '⬇️', 'arrow_left' => '⬅️', 'arrow_right' => '➡️',
315
+ 'arrow_upper_right' => '↗️', 'arrow_lower_right' => '↘️', 'arrow_lower_left' => '↙️',
316
+ 'arrow_upper_left' => '↖️', 'arrows_clockwise' => '🔃', 'arrows_counterclockwise' => '🔄',
317
+ 'back' => '🔙', 'end' => '🔚', 'on' => '🔛', 'soon' => '🔜', 'top' => '🔝',
318
+ 'radio_button' => '🔘', 'red_circle' => '🔴', 'orange_circle' => '🟠',
319
+ 'yellow_circle' => '🟡', 'green_circle' => '🟢', 'blue_circle' => '🔵',
320
+ 'purple_circle' => '🟣', 'brown_circle' => '🟤', 'white_circle' => '⚪',
321
+ 'black_circle' => '⚫',
322
+
323
+ // -- Drapeaux (suite) -------------------------------------------------
324
+ 'triangular_flag_on_post' => '🚩', 'crossed_flags' => '🎌',
325
+ 'us' => '🇺🇸', 'gb' => '🇬🇧', 'fr' => '🇫🇷', 'de' => '🇩🇪', 'es' => '🇪🇸',
326
+ 'it' => '🇮🇹', 'jp' => '🇯🇵', 'cn' => '🇨🇳', 'kr' => '🇰🇷', 'ca' => '🇨🇦',
327
+ 'au' => '🇦🇺', 'br' => '🇧🇷', 'in' => '🇮🇳', 'ru' => '🇷🇺', 'eu' => '🇪🇺',
194
328
  ];
195
329
 
196
330
  /**
197
- * Enregistre (ou remplace) un raccourci emoji personnalisé.
331
+ * Registers (or replaces) a custom emoji shortcut.
198
332
  */
199
333
  public static function registerEmoji(string $shortcode, string $char): void {
200
334
  self::$extraEmoji[strtolower(trim($shortcode, ':'))] = $char;
@@ -206,12 +340,12 @@ class MD {
206
340
  }
207
341
 
208
342
  // ========================================================================
209
- // LISTES DE DÉFINITION (syntaxe étendue)
210
- // Terme
211
- // : Définition
212
- // Analyse procédurale ligne par ligne (plus sûre qu'une seule grosse
213
- // regex pour regrouper plusieurs paires terme/définitions dans un même
214
- // <dl>, séparées ou non par une ligne vide).
343
+ // DEFINITION LISTS (extended syntax)
344
+ // Term
345
+ // : Definition
346
+ // Procedural line-by-line analysis (safer than a single big regex for
347
+ // grouping several term/definition pairs into one <dl>, separated by a
348
+ // blank line or not).
215
349
  // ========================================================================
216
350
  private static function isDefinitionColonLine(string $line): bool {
217
351
  return (bool) preg_match('/^[ \t]*:[ \t]+.+$/', $line);
@@ -252,8 +386,8 @@ class MD {
252
386
  $i++;
253
387
  }
254
388
 
255
- // Une seule ligne vide entre deux groupes reste dans le même <dl>
256
- // si le groupe suivant est bien un nouveau terme.
389
+ // A single blank line between two groups stays in the same <dl>
390
+ // if the next group really is a new term.
257
391
  if ($i < $n && trim($lines[$i]) === '') {
258
392
  $j = $i;
259
393
  while ($j < $n && trim($lines[$j]) === '') $j++;
@@ -275,35 +409,169 @@ class MD {
275
409
  return implode("\n", $out);
276
410
  }
277
411
 
412
+ // ========================================================================
413
+ // RAW HTML (safe subset, GitHub README style)
414
+ //
415
+ // Writing HTML tags directly (e.g. <div align="center">, <img>, <sub>,
416
+ // <br>, HTML tables...) is allowed ONLY if:
417
+ // - the tag is part of the HTML_ALLOWED_TAGS whitelist;
418
+ // - every attribute is part of the whitelist for that tag (or of the
419
+ // global HTML_GLOBAL_ATTRS attributes);
420
+ // - no attribute starts with "on" (onclick, onerror, ...);
421
+ // - URLs (href/src) use a safe scheme (isSafeUrl).
422
+ //
423
+ // Any unknown or dangerous tag (script/style/iframe/...), or any
424
+ // non-whitelisted attribute, is silently stripped. The text content
425
+ // between the tags is NOT swallowed: it keeps being processed as normal
426
+ // markdown (that's what lets you have markdown headings, badges and images
427
+ // inside a <div align="center">...</div>).
428
+ // ========================================================================
429
+
430
+ private const HTML_ALLOWED_TAGS = [
431
+ 'div', 'span', 'p', 'br', 'hr', 'wbr',
432
+ 'b', 'strong', 'i', 'em', 'u', 's', 'strike', 'del', 'ins',
433
+ 'mark', 'small', 'sub', 'sup', 'kbd', 'code', 'pre', 'abbr', 'q', 'cite',
434
+ 'ul', 'ol', 'li', 'dl', 'dt', 'dd',
435
+ 'table', 'thead', 'tbody', 'tfoot', 'tr', 'td', 'th', 'caption', 'colgroup', 'col',
436
+ 'blockquote',
437
+ 'a', 'img', 'picture', 'source', 'figure', 'figcaption',
438
+ 'h1', 'h2', 'h3', 'h4', 'h5', 'h6',
439
+ 'details', 'summary', 'center',
440
+ ];
441
+
442
+ /** Attributes allowed on any whitelisted tag. */
443
+ private const HTML_GLOBAL_ATTRS = ['id', 'class', 'title', 'align', 'valign', 'width', 'height', 'dir', 'lang'];
444
+
445
+ /** Extra attributes allowed, per tag. */
446
+ private const HTML_TAG_ATTRS = [
447
+ 'a' => ['href', 'name', 'target', 'rel'],
448
+ 'img' => ['src', 'alt', 'loading', 'srcset', 'sizes'],
449
+ 'source' => ['src', 'srcset', 'type', 'media'],
450
+ 'td' => ['colspan', 'rowspan'],
451
+ 'th' => ['colspan', 'rowspan', 'scope'],
452
+ 'col' => ['span'],
453
+ 'ol' => ['start', 'type'],
454
+ 'details' => ['open'],
455
+ ];
456
+
457
+ /** Self-closing tags (no closing tag expected). */
458
+ private const HTML_VOID_TAGS = ['img', 'br', 'hr', 'wbr', 'source', 'col'];
459
+
460
+ /**
461
+ * Checks that a URL (href/src) uses a safe scheme: relative links, anchors,
462
+ * http(s), mailto, tel, or base64-encoded images (png/gif/jpeg/webp only —
463
+ * not svg+xml, which can embed a <script>).
464
+ * Notably rejects javascript:, vbscript:, data:text/html.
465
+ */
466
+ private static function isSafeUrl(string $url): bool {
467
+ $url = trim($url);
468
+ if ($url === '') return true;
469
+ // A path with no explicit scheme ("assets/x.png", "../x", "#anchor",
470
+ // "/x", "x") is a relative link or an anchor: always safe.
471
+ if (!preg_match('~^([a-zA-Z][a-zA-Z0-9+.\-]*):~', $url, $m)) return true;
472
+ $scheme = strtolower($m[1]);
473
+ if (in_array($scheme, ['http', 'https', 'mailto', 'tel'], true)) return true;
474
+ if ($scheme === 'data') {
475
+ // Base64-encoded images only — no data:image/svg+xml, which can
476
+ // embed a <script>, and no data:text/html.
477
+ return (bool) preg_match('~^data:image/(png|gif|jpe?g|webp);base64,~i', $url);
478
+ }
479
+ return false; // javascript:, vbscript:, file:, etc. → rejected
480
+ }
481
+
482
+ /**
483
+ * Sanitizes a single raw HTML tag (e.g. '<div align="center">', '</div>',
484
+ * '<img src="..." onerror="...">').
485
+ *
486
+ * @return string|null The cleaned tag to keep, an empty string to strip it
487
+ * silently, or null if it doesn't look like a valid
488
+ * HTML tag (in which case the caller strips it too, to
489
+ * be safe).
490
+ */
491
+ private static function sanitizeHtmlTag(string $tag): ?string {
492
+ if (!preg_match(
493
+ '/^<(\/)?([a-zA-Z][a-zA-Z0-9-]*)((?:\s+[a-zA-Z_:][a-zA-Z0-9_:.-]*(?:\s*=\s*(?:"[^"]*"|\'[^\']*\'|[^\s"\'>]+))?)*)\s*(\/)?>$/s',
494
+ $tag,
495
+ $m
496
+ )) {
497
+ return null;
498
+ }
499
+
500
+ $closing = $m[1] === '/';
501
+ $tagName = strtolower($m[2]);
502
+ $attrsRaw = $m[3];
503
+
504
+ if (!in_array($tagName, self::HTML_ALLOWED_TAGS, true)) {
505
+ return null;
506
+ }
507
+
508
+ if ($closing) {
509
+ return "</{$tagName}>";
510
+ }
511
+
512
+ $allowedAttrs = array_merge(self::HTML_GLOBAL_ATTRS, self::HTML_TAG_ATTRS[$tagName] ?? []);
513
+ $safeAttrs = '';
514
+
515
+ if (preg_match_all(
516
+ '/([a-zA-Z_:][a-zA-Z0-9_:.-]*)(?:\s*=\s*("([^"]*)"|\'([^\']*)\'|([^\s"\'>]+)))?/',
517
+ $attrsRaw,
518
+ $am,
519
+ PREG_SET_ORDER
520
+ )) {
521
+ foreach ($am as $a) {
522
+ $attrName = strtolower($a[1]);
523
+ if ($attrName === '') continue;
524
+ if (str_starts_with($attrName, 'on')) continue; // safety net against JS handlers
525
+ if (!in_array($attrName, $allowedAttrs, true)) continue;
526
+
527
+ if ($tagName === 'details' && $attrName === 'open') {
528
+ $safeAttrs .= ' open';
529
+ continue;
530
+ }
531
+
532
+ $attrVal = $a[3] ?? ($a[4] ?? ($a[5] ?? ''));
533
+
534
+ if (in_array($attrName, ['href', 'src'], true) && !self::isSafeUrl($attrVal)) {
535
+ continue;
536
+ }
537
+
538
+ $safeAttrs .= ' ' . $attrName . '="' . htmlspecialchars($attrVal, ENT_QUOTES, 'UTF-8') . '"';
539
+ }
540
+ }
541
+
542
+ $close = in_array($tagName, self::HTML_VOID_TAGS, true) ? ' /' : '';
543
+ return "<{$tagName}{$safeAttrs}{$close}>";
544
+ }
545
+
278
546
  // ========================================================================
279
547
 
280
548
  public static function toHtml(string $markdown): string {
281
549
 
282
550
  // ====================================================================
283
- // ÉTAPE 1 : Normalisation des fins de ligne
551
+ // STEP 1: Line-ending normalization
284
552
  // ====================================================================
285
553
  $html = str_replace(["\r\n", "\r"], "\n", $markdown);
286
554
 
287
555
 
288
556
  // ====================================================================
289
- // ÉTAPE 2 : PLUGINS
290
- // Deux formes supportées :
557
+ // STEP 2: PLUGINS
558
+ // Two forms supported:
291
559
  //
292
- // INLINE : {% nom arg1 "arg 2" %}
560
+ // INLINE: {% name arg1 "arg 2" %}
293
561
  // → $args = ['arg1', 'arg 2'], $body = ''
294
562
  //
295
- // BLOC : {% nom arg1\ncontenu\nsur\nplusieurs lignes\n%}
296
- // → $args = ['arg1'], $body = "contenu\nsur\nplusieurs lignes"
563
+ // BLOCK : {% name arg1\ncontent\nover\nseveral lines\n%}
564
+ // → $args = ['arg1'], $body = "content\nover\nseveral lines"
297
565
  //
298
- // Les deux sont capturés par une seule regex qui distingue la présence
299
- // d'un saut de ligne après les args (bloc) ou non (inline).
300
- // Traités avant l'encodage XSS — réinjectés en toute dernière étape.
566
+ // Both are captured by a single regex that tells apart the presence of
567
+ // a newline after the args (block) or not (inline).
568
+ // Processed before XSS encoding — re-injected as the very last step.
301
569
  // ====================================================================
302
570
  $pluginBlocks = [];
303
571
 
304
572
  /**
305
- * Parse une chaîne d'arguments en tableau.
306
- * Supporte les mots simples, "guillemets doubles" et 'simples'.
573
+ * Parses an argument string into an array.
574
+ * Supports bare words, "double quotes" and 'single quotes'.
307
575
  */
308
576
  $parseArgs = static function (string $rawArgs): array {
309
577
  $args = [];
@@ -324,18 +592,18 @@ class MD {
324
592
  };
325
593
 
326
594
  $html = preg_replace_callback(
327
- // Groupe 1 : nom du plugin
328
- // Groupe 2 : args inline (tout ce qui est sur la première ligne après le nom)
329
- // Groupe 3 : corps multi-ligne (présent seulement pour les tags blocs)
595
+ // Group 1: plugin name
596
+ // Group 2: inline args (everything on the first line after the name)
597
+ // Group 3: multi-line body (present only for block tags)
330
598
  '/\{%\s*([a-zA-Z0-9_-]+)([^\n%]*?)(?:\n([\s\S]*?))?\s*%\}/m',
331
599
  function ($matches) use (&$pluginBlocks, $parseArgs): string {
332
600
  $name = strtolower(trim($matches[1]));
333
601
  $args = $parseArgs(trim($matches[2] ?? ''));
334
- // $matches[3] existe uniquement si le tag est multi-ligne
602
+ // $matches[3] exists only if the tag is multi-line
335
603
  $body = isset($matches[3]) ? trim($matches[3]) : '';
336
604
 
337
605
  if (!isset(self::$plugins[$name])) {
338
- // Plugin inconnu : préservé encodé plutôt que silencieusement supprimé
606
+ // Unknown plugin: kept encoded rather than silently removed
339
607
  return htmlspecialchars($matches[0], ENT_QUOTES, 'UTF-8');
340
608
  }
341
609
 
@@ -349,17 +617,17 @@ class MD {
349
617
 
350
618
 
351
619
  // ====================================================================
352
- // ÉTAPE 2a : DÉFINITIONS DE NOTES DE BAS DE PAGE (footnotes)
353
- // [^1]: Texte de la note.
354
- // [^bignote]: Première ligne.
620
+ // STEP 2a: FOOTNOTE DEFINITIONS
621
+ // [^1]: Note text.
622
+ // [^bignote]: First line.
355
623
  //
356
- // Paragraphe suivant, indenté de 4 espaces ou 1 tabulation.
624
+ // Following paragraph, indented by 4 spaces or 1 tab.
357
625
  //
358
- // `{ du code }`
359
- // Extraites (et retirées du texte) AVANT les définitions de liens par
360
- // référence, car [^label]: matcherait aussi leur regex sinon.
361
- // Le contenu de chaque note est rendu via un appel récursif à
362
- // toHtml() pour supporter plusieurs paragraphes, du code, etc.
626
+ // `{ some code }`
627
+ // Extracted (and removed from the text) BEFORE reference link
628
+ // definitions, since [^label]: would otherwise match their regex too.
629
+ // Each note's content is rendered via a recursive call to toHtml() to
630
+ // support multiple paragraphs, code, etc.
363
631
  // ====================================================================
364
632
  $footnoteDefs = [];
365
633
  $html = preg_replace_callback(
@@ -381,12 +649,12 @@ class MD {
381
649
 
382
650
 
383
651
  // ====================================================================
384
- // ÉTAPE 2b : DÉFINITIONS DE LIENS PAR RÉFÉRENCE
385
- // [label]: https://example.com "Titre optionnel"
386
- // [label]: <https://example.com> 'Titre optionnel'
387
- // [label]: https://example.com (Titre optionnel)
388
- // Extraites (et retirées du texte) avant tout le reste ; utilisées
389
- // plus loin par les liens [texte][label] / [texte][].
652
+ // STEP 2b: REFERENCE LINK DEFINITIONS
653
+ // [label]: https://example.com "Optional title"
654
+ // [label]: <https://example.com> 'Optional title'
655
+ // [label]: https://example.com (Optional title)
656
+ // Extracted (and removed from the text) before everything else; used
657
+ // later by the [text][label] / [text][] links.
390
658
  // ====================================================================
391
659
  $refDefs = [];
392
660
  $html = preg_replace_callback(
@@ -402,7 +670,7 @@ class MD {
402
670
 
403
671
 
404
672
  // ====================================================================
405
- // ÉTAPE 3 : BLOCS DE CODE (```lang ... ```)
673
+ // STEP 3: CODE BLOCKS (```lang ... ```)
406
674
  // ====================================================================
407
675
  $codeBlocks = [];
408
676
  $html = preg_replace_callback('/^```([a-zA-Z0-9_+-]*)\n([\s\S]*?)\n^```/m', function ($matches) use (&$codeBlocks) {
@@ -414,10 +682,10 @@ class MD {
414
682
  }, $html);
415
683
 
416
684
  // ====================================================================
417
- // ÉTAPE 3a : BLOCS DE CODE INDENTÉS (4 espaces ou 1 tabulation)
418
- // Reconnu seulement quand précédé d'une ligne vide (ou du début du
419
- // document) et suivi d'une ligne vide (ou de la fin du document), afin
420
- // d'éviter les conflits avec l'indentation des listes imbriquées.
685
+ // STEP 3a: INDENTED CODE BLOCKS (4 spaces or 1 tab)
686
+ // Recognized only when preceded by a blank line (or the start of the
687
+ // document) and followed by a blank line (or the end of the document),
688
+ // to avoid conflicts with the indentation of nested lists.
421
689
  // ====================================================================
422
690
  $html = preg_replace_callback(
423
691
  '/(?<=\n\n|^)((?:[ ]{4}|\t)[^\n]*(?:\n(?:[ ]{4}|\t)[^\n]*)*)(?=\n\n|\n*$)/',
@@ -434,13 +702,13 @@ class MD {
434
702
  $html
435
703
  );
436
704
 
437
- // Code inline avec double backticks (permet d'inclure un backtick littéral)
705
+ // Inline code with double backticks (lets you include a literal backtick)
438
706
  $inlineCodes = [];
439
707
  $html = preg_replace_callback('/``(.+?)``/s', function ($matches) use (&$inlineCodes) {
440
708
  $content = $matches[1];
441
- // Convention standard : si le contenu commence et finit par un
442
- // espace (et n'est pas uniquement des espaces), on retire un
443
- // espace de chaque côté — utile pour englober un ` en bordure.
709
+ // Standard convention: if the content starts and ends with a space
710
+ // (and isn't only spaces), strip one space on each side — handy for
711
+ // wrapping a ` at the edge.
444
712
  if (preg_match('/^ (.*[^ ]) $/s', $content, $trim)) {
445
713
  $content = $trim[1];
446
714
  }
@@ -450,7 +718,7 @@ class MD {
450
718
  return $placeholder;
451
719
  }, $html);
452
720
 
453
- // Code inline (`...`)
721
+ // Inline code (`...`)
454
722
  $html = preg_replace_callback('/`([^`\n]+)`/', function ($matches) use (&$inlineCodes) {
455
723
  $code = htmlspecialchars($matches[1], ENT_QUOTES, 'UTF-8');
456
724
  $placeholder = "\x02IC" . count($inlineCodes) . "\x03";
@@ -460,10 +728,10 @@ class MD {
460
728
 
461
729
 
462
730
  // ====================================================================
463
- // ÉTAPE 3b : ÉCHAPPEMENT DES CARACTÈRES (\* \_ \# etc.)
464
- // Traité après l'extraction du code (le code reste littéral) et avant
465
- // tout le reste, pour que \* n'ouvre pas une emphase, \# ne crée pas
466
- // un titre, \- ne crée pas de liste, etc.
731
+ // STEP 3b: CHARACTER ESCAPING (\* \_ \# etc.)
732
+ // Processed after code extraction (code stays literal) and before
733
+ // everything else, so that \* doesn't open emphasis, \# doesn't create
734
+ // a heading, \- doesn't create a list, etc.
467
735
  // ====================================================================
468
736
  $escapes = [];
469
737
  $html = preg_replace_callback(
@@ -475,9 +743,9 @@ class MD {
475
743
  },
476
744
  $html
477
745
  );
478
- // &#124; est la convention documentée (Markdown Extra / PHP Markdown)
479
- // pour afficher un pipe littéral dans une cellule de tableau sans
480
- // qu'il soit interprété comme séparateur de colonnes.
746
+ // &#124; is the documented convention (Markdown Extra / PHP Markdown)
747
+ // for showing a literal pipe in a table cell without it being
748
+ // interpreted as a column separator.
481
749
  $html = preg_replace_callback(
482
750
  '/&#124;/i',
483
751
  function () use (&$escapes): string {
@@ -490,9 +758,9 @@ class MD {
490
758
 
491
759
 
492
760
  // ====================================================================
493
- // ÉTAPE 3d : LIENS AUTOMATIQUES <https://...> et <email@example.com>
494
- // Traités avant l'encodage XSS car les caractères < > seraient encodés
495
- // en &lt; &gt; et la regex ne matcherait plus.
761
+ // STEP 3d: AUTOMATIC LINKS <https://...> and <email@example.com>
762
+ // Processed before XSS encoding because the < > characters would be
763
+ // encoded to &lt; &gt; and the regex would no longer match.
496
764
  // ====================================================================
497
765
  $autolinks = [];
498
766
  $html = preg_replace_callback('/<(https?:\/\/[^\s<>]+)>/', function ($m) use (&$autolinks): string {
@@ -510,13 +778,13 @@ class MD {
510
778
 
511
779
 
512
780
  // ====================================================================
513
- // ÉTAPE 3e : ALERTES GFM ET BLOCKQUOTES
514
- // Traités avant l'encodage XSS car le caractère > serait encodé en &gt;
515
- // et les regex ne matcheraient plus.
781
+ // STEP 3e: GFM ALERTS AND BLOCKQUOTES
782
+ // Processed before XSS encoding because the > character would be
783
+ // encoded to &gt; and the regexes would no longer match.
516
784
  // ====================================================================
517
785
  $blockquotes = [];
518
786
 
519
- // Alertes GFM (> [!NOTE], etc.) — plus spécifique, traité en premier
787
+ // GFM alerts (> [!NOTE], etc.) — more specific, processed first
520
788
  $html = preg_replace_callback(
521
789
  '/^(>\s*\[!(NOTE|TIP|IMPORTANT|WARNING|CAUTION)\]\n(?:>[ \t]?[^\n]*\n?)*)/m',
522
790
  function ($matches) use (&$blockquotes): string {
@@ -529,38 +797,75 @@ class MD {
529
797
  $blockquotes[$placeholder] = "<div class=\"markdown-alert markdown-alert-{$type}\">"
530
798
  . "<p class=\"markdown-alert-title\">{$label}</p>"
531
799
  . "<p>{$content}</p></div>";
532
- // Le \n final consommé par la regex est réinjecté après le
533
- // placeholder pour ne pas fusionner la ligne vide suivante
534
- // avec celle du placeholder (ce qui fausserait par exemple
535
- // la détection d'un titre Setext juste après).
800
+ // The trailing \n consumed by the regex is re-injected after
801
+ // the placeholder so the following blank line doesn't merge
802
+ // with the placeholder's line (which would break, for example,
803
+ // detecting a Setext heading right after).
536
804
  return $placeholder . (str_ends_with($matches[1], "\n") ? "\n" : '');
537
805
  },
538
806
  $html
539
807
  );
540
808
 
541
- // Blockquotes standards (imbrication gérée par récursion sur toHtml,
542
- // qui ré-applique cette même règle sur le contenu déjà dé-préfixé
543
- // d'un niveau de ">")
809
+ // Standard blockquotes (nesting handled by recursion through toHtml,
810
+ // which re-applies this same rule to the content already stripped of
811
+ // one ">" level)
544
812
  $html = preg_replace_callback('/^((?:>[ \t]?[^\n]*\n?)+)/m', function ($matches) use (&$blockquotes): string {
545
813
  $content = preg_replace('/^>[ \t]?/m', '', $matches[1]);
546
- // Les deux espaces trailing sont laissés tels quels : toHtml() les gère lui-même
814
+ // The two trailing spaces are left as-is: toHtml() handles them itself
547
815
  $inner = self::toHtml(trim($content));
548
816
  $placeholder = "\x02BQ" . count($blockquotes) . "\x03";
549
817
  $blockquotes[$placeholder] = "<blockquote>{$inner}</blockquote>";
550
- // Voir commentaire ci-dessus : on préserve le \n final consommé.
818
+ // See the comment above: preserve the trailing \n that was consumed.
551
819
  return $placeholder . (str_ends_with($matches[1], "\n") ? "\n" : '');
552
820
  }, $html);
553
821
 
554
822
 
555
823
  // ====================================================================
556
- // ÉTAPE 4 : Encodage XSS global
824
+ // STEP 3f: RAW HTML (safe subset, GitHub README style)
825
+ // Processed before XSS encoding because the < > characters would be
826
+ // encoded to &lt; &gt; and no longer recognized as tags.
827
+ // The content between the tags is not swallowed: it stays in the
828
+ // stream and keeps being processed as normal markdown.
829
+ // ====================================================================
830
+
831
+ // Intrinsically dangerous elements: removed along with their content
832
+ // (script/style/iframe can embed JS or load a third-party page;
833
+ // form/button/textarea/select/option have no place in markdown
834
+ // content).
835
+ $html = preg_replace(
836
+ '/<(script|style|iframe|object|embed|noscript|template|form|button|textarea|select|option)\b[^>]*>[\s\S]*?<\/\1>/i',
837
+ '',
838
+ $html
839
+ );
840
+
841
+ $rawHtml = [];
842
+ $html = preg_replace_callback(
843
+ '/<!--[\s\S]*?-->|<\/?[a-zA-Z][a-zA-Z0-9-]*(?:\s+[a-zA-Z_:][a-zA-Z0-9_:.-]*(?:\s*=\s*(?:"[^"]*"|\'[^\']*\'|[^\s"\'>]+))?)*\s*\/?>/',
844
+ function ($m) use (&$rawHtml): string {
845
+ $tag = $m[0];
846
+ // HTML comment: invisible, safe to remove.
847
+ if (str_starts_with($tag, '<!--')) return '';
848
+
849
+ $sanitized = self::sanitizeHtmlTag($tag);
850
+ if ($sanitized === null || $sanitized === '') return '';
851
+
852
+ $placeholder = "\x02HT" . count($rawHtml) . "\x03";
853
+ $rawHtml[$placeholder] = $sanitized;
854
+ return $placeholder;
855
+ },
856
+ $html
857
+ );
858
+
859
+
860
+ // ====================================================================
861
+ // STEP 4: Global XSS encoding
557
862
  // ====================================================================
558
863
  $html = htmlspecialchars($html, ENT_NOQUOTES, 'UTF-8');
559
864
 
560
865
 
561
866
  // ====================================================================
562
- // ÉTAPE 5 : TABLEAUX GFM
563
- // Supporte les lignes avec ou sans pipe final (| col | ou | col)
867
+ // STEP 5: GFM TABLES
868
+ // Supports rows with or without a trailing pipe (| col | or | col)
564
869
  // ====================================================================
565
870
  $html = preg_replace_callback(
566
871
  '/^(\|[^\n]+\|?\n)([ \t]*\|[ \t]*:?-+:?[ \t]*(?:\|[ \t]*:?-+:?[ \t]*)*\|?\n)((?:\|[^\n]+\|?\n?)+)/m',
@@ -608,26 +913,26 @@ class MD {
608
913
 
609
914
 
610
915
  // ====================================================================
611
- // ÉTAPE 6 : (Alertes GFM et blockquotes traités à l'étape 3e)
916
+ // STEP 6: (GFM alerts and blockquotes handled in step 3e)
612
917
  // ====================================================================
613
918
 
614
919
 
615
920
  // ====================================================================
616
- // ÉTAPE 7 : LISTES DE TÂCHES (GFM checkboxes)
921
+ // STEP 7: TASK LISTS (GFM checkboxes)
617
922
  // ====================================================================
618
923
  $html = preg_replace('/^[ \t]*[-*+] \[ \] (.+)$/m', '<li class="task-item"><input type="checkbox" disabled /> $1</li>', $html);
619
924
  $html = preg_replace('/^[ \t]*[-*+] \[[xX]\] (.+)$/m', '<li class="task-item"><input type="checkbox" checked disabled /> $1</li>', $html);
620
925
 
621
926
 
622
927
  // ====================================================================
623
- // ÉTAPE 7b : TITRES SETEXT (syntaxe alternative == / --)
624
- // Titre
928
+ // STEP 7b: SETEXT HEADINGS (alternative == / -- syntax)
929
+ // Title
625
930
  // ===== → <h1>
626
931
  //
627
- // Titre
932
+ // Title
628
933
  // ----- → <h2>
629
- // Traité avant les titres ATX et avant les lignes séparatrices (une
630
- // ligne de tirets juste après une ligne de texte est un titre, pas un <hr>).
934
+ // Processed before ATX headings and before horizontal rules (a line of
935
+ // dashes right after a line of text is a heading, not an <hr>).
631
936
  // ====================================================================
632
937
  $html = preg_replace_callback(
633
938
  '/^(?![ \t]*(?:#{1,6}[ \t]|>|```|\||[-*+][ \t]|\d+\.[ \t]))[ \t]*(\S.*?)[ \t]*(?:\{#([a-zA-Z0-9_\-:.]+)\}[ \t]*)?\n[ \t]*=+[ \t]*$/m',
@@ -650,7 +955,7 @@ class MD {
650
955
 
651
956
 
652
957
  // ====================================================================
653
- // ÉTAPE 8 : TITRES (ATX : # à ######)
958
+ // STEP 8: HEADINGS (ATX: # to ######)
654
959
  // ====================================================================
655
960
  $html = preg_replace_callback(
656
961
  '/^(#{1,6})[ \t]+(.+?)[ \t]*(?:\{#([a-zA-Z0-9_\-:.]+)\}[ \t]*)?(?:[ \t]+#+)?$/m',
@@ -665,13 +970,13 @@ class MD {
665
970
 
666
971
 
667
972
  // ====================================================================
668
- // ÉTAPE 9 : LISTES (puces et ordonnées, avec imbrication)
669
- // Une seule passe détecte un bloc contigu de lignes qui sont soit une
670
- // puce (-,*,+) soit un item numéroté, quel que soit leur niveau
671
- // d'indentation ; le bloc est ensuite reconstruit récursivement en
672
- // <ol>/<ul> imbriqués selon la profondeur d'indentation relative.
673
- // Les items de tâches (déjà convertis en <li class="task-item">) ne
674
- // matchent plus ce motif et ne sont donc pas ré-englobés ici.
973
+ // STEP 9: LISTS (bullets and ordered, with nesting)
974
+ // A single pass detects a contiguous block of lines that are either a
975
+ // bullet (-,*,+) or a numbered item, whatever their indentation level;
976
+ // the block is then rebuilt recursively into nested <ol>/<ul>
977
+ // according to the relative indentation depth.
978
+ // Task items (already converted to <li class="task-item">) no longer
979
+ // match this pattern and are therefore not re-wrapped here.
675
980
  // ====================================================================
676
981
  $html = preg_replace_callback(
677
982
  '/^([ \t]*(?:\d+\.|[-*+])[ \t]+.+(?:\n[ \t]*(?:\d+\.|[-*+])[ \t]+.+)*)/m',
@@ -686,7 +991,7 @@ class MD {
686
991
  }
687
992
  }
688
993
  if (empty($items)) return $matches[1];
689
- // Normalise le niveau d'indentation le plus bas à 0
994
+ // Normalize the lowest indentation level to 0
690
995
  $minIndent = min(array_column($items, 'indent'));
691
996
  foreach ($items as &$it) $it['indent'] -= $minIndent;
692
997
  unset($it);
@@ -707,26 +1012,26 @@ class MD {
707
1012
 
708
1013
 
709
1014
  // ====================================================================
710
- // ÉTAPE 9b : LISTES DE DÉFINITION (syntaxe étendue)
711
- // Terme
712
- // : Définition
1015
+ // STEP 9b: DEFINITION LISTS (extended syntax)
1016
+ // Term
1017
+ // : Definition
713
1018
  // ====================================================================
714
1019
  $html = self::extractDefinitionLists($html);
715
1020
 
716
1021
 
717
1022
  // ====================================================================
718
- // ÉTAPE 9c : RÉFÉRENCES DE NOTES DE BAS DE PAGE [^label]
719
- // Converties AVANT l'emphase pour ne pas entrer en collision avec le
720
- // nouvel exposant ^texte^ (un [^1] suivi plus loin d'un [^2] sur la
721
- // même ligne pourrait sinon être interprété comme ^1] ... [^2^).
722
- // La numérotation est séquentielle, dans l'ordre de première
723
- // apparition dans le texte (comme documenté).
1023
+ // STEP 9c: FOOTNOTE REFERENCES [^label]
1024
+ // Converted BEFORE emphasis so they don't collide with the new
1025
+ // superscript ^text^ (a [^1] followed later by a [^2] on the same line
1026
+ // could otherwise be read as ^1] ... [^2^).
1027
+ // Numbering is sequential, in order of first appearance in the text
1028
+ // (as documented).
724
1029
  // ====================================================================
725
1030
  $footnoteOrder = [];
726
1031
  $html = preg_replace_callback('/\[\^([^\]\s]+)\]/', function ($m) use (&$footnoteOrder, &$footnoteDefs): string {
727
1032
  $label = strtolower(trim($m[1]));
728
1033
  if (!isset($footnoteDefs[$label])) {
729
- // Référence vers une note non définie : laissée telle quelle.
1034
+ // Reference to an undefined note: left as-is.
730
1035
  return $m[0];
731
1036
  }
732
1037
  if (!isset($footnoteOrder[$label])) {
@@ -738,32 +1043,32 @@ class MD {
738
1043
 
739
1044
 
740
1045
  // ====================================================================
741
- // ÉTAPE 10 : TEXTE EN LIGNE (Gras, Italique, Barré, Surlignage,
742
- // Indice/Exposant, Emoji)
1046
+ // STEP 10: INLINE TEXT (Bold, Italic, Strikethrough, Highlight,
1047
+ // Subscript/Superscript, Emoji)
743
1048
  // ====================================================================
744
1049
  $html = preg_replace('/\*\*\*(.+?)\*\*\*/s', '<strong><em>$1</em></strong>', $html);
745
1050
  $html = preg_replace('/___(.+?)___/s', '<strong><em>$1</em></strong>', $html);
746
1051
  $html = preg_replace('/\*\*(.+?)\*\*/s', '<strong>$1</strong>', $html);
747
1052
  $html = preg_replace('/__(.+?)__/s', '<strong>$1</strong>', $html);
748
1053
  $html = preg_replace('/\*(.+?)\*/s', '<em>$1</em>', $html);
749
- // Le _ italique ne doit matcher qu'aux frontières de mots pour ne pas
750
- // capturer les snake_case, noms de packages (@php-wasm/node), etc.
1054
+ // Italic _ must only match at word boundaries so it doesn't capture
1055
+ // snake_case, package names (@php-wasm/node), etc.
751
1056
  $html = preg_replace('/(?<!\w)_([^_\n]+)_(?!\w)/', '<em>$1</em>', $html);
752
- // Surlignage ==texte== (syntaxe étendue)
1057
+ // Highlight ==text== (extended syntax)
753
1058
  $html = preg_replace('/==(.+?)==/s', '<mark>$1</mark>', $html);
754
- // Barré ~~texte~~ — traité AVANT le sous-script (simple ~) pour que
755
- // celui-ci ne matche pas la moitié d'une paire de tildes doubles.
1059
+ // Strikethrough ~~text~~ — processed BEFORE subscript (single ~) so the
1060
+ // latter doesn't match half of a double-tilde pair.
756
1061
  $html = preg_replace('/~~(.+?)~~/s', '<del>$1</del>', $html);
757
- // Exposant ^texte^ (syntaxe étendue) — placé avant l'échappement des
758
- // références de notes ([^label]) n'est pas un souci : celles-ci sont
759
- // encadrées de crochets et ne forment donc pas de paire ^...^ isolée.
1062
+ // Superscript ^text^ (extended syntax) — placing it before the note
1063
+ // reference escaping ([^label]) is not a problem: those are wrapped in
1064
+ // brackets and so don't form an isolated ^...^ pair.
760
1065
  $html = preg_replace('/\^([^\^\n]+)\^/', '<sup>$1</sup>', $html);
761
- // Sous-script ~texte~ (un seul tilde ; les ~~ ont déjà été consommés
762
- // juste au-dessus par le barré).
1066
+ // Subscript ~text~ (a single tilde; the ~~ were already consumed just
1067
+ // above by strikethrough).
763
1068
  $html = preg_replace('/~([^~\n]+)~/', '<sub>$1</sub>', $html);
764
1069
 
765
- // Émojis :shortcode: (syntaxe étendue) — les raccourcis inconnus sont
766
- // laissés tels quels plutôt que silencieusement supprimés.
1070
+ // Emojis :shortcode: (extended syntax) — unknown shortcuts are left
1071
+ // as-is rather than silently removed.
767
1072
  $html = preg_replace_callback('/:([a-zA-Z0-9_+\-]+):/', function ($m): string {
768
1073
  $emoji = self::emojiFor($m[1]);
769
1074
  return $emoji ?? $m[0];
@@ -771,9 +1076,9 @@ class MD {
771
1076
 
772
1077
 
773
1078
  // ====================================================================
774
- // ÉTAPE 11 : LIENS & IMAGES
775
- // Les liens externes (https?://) reçoivent target="_blank" + rel="noopener noreferrer".
776
- // Les liens internes (/page, #anchor, ../truc) n'en reçoivent pas.
1079
+ // STEP 11: LINKS & IMAGES
1080
+ // External links (https?://) get target="_blank" + rel="noopener noreferrer".
1081
+ // Internal links (/page, #anchor, ../thing) don't.
777
1082
  // ====================================================================
778
1083
  $html = preg_replace(
779
1084
  '/!\[([^\]]*)\]\(([^)\s]+)(?:\s+"([^"]*)")?\)/',
@@ -789,7 +1094,7 @@ class MD {
789
1094
  return "<a href=\"{$href}\"{$titleAttr}{$extern}>{$text}</a>";
790
1095
  };
791
1096
 
792
- // Liens par référence [texte][label] et [texte][] (raccourci = label = texte)
1097
+ // Reference links [text][label] and [text][] (shortcut = label = text)
793
1098
  $html = preg_replace_callback(
794
1099
  '/\[([^\]]+)\]\[([^\]]*)\]/',
795
1100
  function ($m) use (&$refDefs, $buildLink): string {
@@ -802,7 +1107,7 @@ class MD {
802
1107
  $html
803
1108
  );
804
1109
 
805
- // Liens markdown [texte](url "titre optionnel")
1110
+ // Markdown links [text](url "optional title")
806
1111
  $html = preg_replace_callback(
807
1112
  '/\[([^\]]+)\]\(([^)\s]+)(?:\s+"([^"]*)")?\)/',
808
1113
  function ($m) use ($buildLink): string {
@@ -811,9 +1116,9 @@ class MD {
811
1116
  $html
812
1117
  );
813
1118
 
814
- // URL nues https://... (syntaxe étendue : auto-link sans crochets).
815
- // Exclut celles déjà entre guillemets/attributs (href="...") ou déjà
816
- // transformées en lien pour ne pas les doubler.
1119
+ // Bare URLs https://... (extended syntax: auto-link without brackets).
1120
+ // Excludes those already inside quotes/attributes (href="...") or
1121
+ // already turned into a link, so they don't get doubled.
817
1122
  $html = preg_replace(
818
1123
  '/(?<!["\'=>])\b(https?:\/\/[^\s<>"\')\]]+)/',
819
1124
  '<a href="$1" target="_blank" rel="noopener noreferrer">$1</a>',
@@ -822,29 +1127,29 @@ class MD {
822
1127
 
823
1128
 
824
1129
  // ====================================================================
825
- // ÉTAPE 12 : LIGNES SÉPARATRICES
1130
+ // STEP 12: HORIZONTAL RULES
826
1131
  // ====================================================================
827
1132
  $html = preg_replace('/^(?:[-*_][ \t]*){3,}$/m', '<hr />', $html);
828
1133
 
829
1134
 
830
1135
  // ====================================================================
831
- // ÉTAPE 13 : PARAGRAPHES
832
- // Stratégie : on traite ligne par ligne. Les lignes qui commencent par
833
- // une balise block-level ou un placeholder sont laissées telles quelles.
834
- // Les lignes de texte brut consécutives sont accumulées puis wrappées
835
- // dans un <p> quand on rencontre une ligne block ou une ligne vide.
1136
+ // STEP 13: PARAGRAPHS
1137
+ // Strategy: process line by line. Lines that start with a block-level
1138
+ // tag or a placeholder are left as-is. Consecutive raw-text lines are
1139
+ // accumulated then wrapped in a <p> when a block line or a blank line
1140
+ // is reached.
836
1141
  // ====================================================================
837
1142
  $blockStartTags = ['<h', '<pre', '<ul', '<ol', '<li', '<table', '<thead', '<tbody',
838
1143
  '<tr', '<td', '<th', '<blockquote', '<div', '<hr', '<img',
839
1144
  '<dl', '<dt', '<dd',
840
- "\x02CB", "\x02PLG", "\x02BQ"];
1145
+ "\x02CB", "\x02PLG", "\x02BQ", "\x02HT"];
841
1146
 
842
1147
  $isBlockLine = static function (string $line) use ($blockStartTags): bool {
843
1148
  $t = ltrim($line);
844
1149
  if ($t === '') return false;
845
- // Toute balise fermante (</...>) est toujours considérée comme une
846
- // ligne "bloc" : ça évite qu'une fermeture de <table>, <thead>,
847
- // <tr>, etc. finisse absorbée dans un <p> environnant.
1150
+ // Any closing tag (</...>) is always treated as a "block" line:
1151
+ // this keeps a closing </table>, </thead>, </tr>, etc. from being
1152
+ // absorbed into a surrounding <p>.
848
1153
  if (str_starts_with($t, '</')) return true;
849
1154
  foreach ($blockStartTags as $tag) {
850
1155
  if (str_starts_with($t, $tag)) return true;
@@ -860,10 +1165,10 @@ class MD {
860
1165
  if (empty($textBuffer)) return;
861
1166
  $content = implode("\n", $textBuffer);
862
1167
  if (trim($content) !== '') {
863
- // Deux espaces en fin de ligne → <br> (convention markdown standard)
1168
+ // Two trailing spaces → <br> (standard markdown convention)
864
1169
  $content = preg_replace('/ $/m', '<br>', $content);
865
- // Saut de ligne simple → espace (comportement GitHub)
866
- // Sauf si déjà converti en <br> ci-dessus
1170
+ // Single line break → space (GitHub behavior)
1171
+ // Unless already converted to <br> above
867
1172
  $content = preg_replace('/(?<!r>)\n/', ' ', $content);
868
1173
  $output[] = '<p>' . trim($content) . '</p>';
869
1174
  }
@@ -875,7 +1180,7 @@ class MD {
875
1180
  $flushBuffer();
876
1181
  $output[] = $line;
877
1182
  } elseif (trim($line) === '') {
878
- // Ligne vide = séparateur de paragraphe
1183
+ // Blank line = paragraph separator
879
1184
  $flushBuffer();
880
1185
  } else {
881
1186
  $textBuffer[] = $line;
@@ -887,23 +1192,24 @@ class MD {
887
1192
 
888
1193
 
889
1194
  // ====================================================================
890
- // ÉTAPE 14 : Réinjecter les placeholders
1195
+ // STEP 14: Re-inject the placeholders
891
1196
  // ====================================================================
892
1197
  $html = strtr($html, $pluginBlocks);
893
1198
  $html = strtr($html, $blockquotes);
1199
+ $html = strtr($html, $rawHtml);
894
1200
  $html = strtr($html, $codeBlocks);
895
1201
  $html = strtr($html, $inlineCodes);
896
1202
  $html = strtr($html, $autolinks);
897
- // Les échappements sont réinjectés en tout dernier, une fois que plus
898
- // aucune regex Markdown ne peut les interpréter.
1203
+ // The escapes are re-injected last, once no Markdown regex can
1204
+ // interpret them anymore.
899
1205
  $html = strtr($html, $escapes);
900
1206
 
901
1207
 
902
1208
  // ====================================================================
903
- // ÉTAPE 15 : BLOC DES NOTES DE BAS DE PAGE
904
- // Ajouté en fin de document, uniquement si au moins une note a été
905
- // référencée (les notes définies mais jamais référencées sont
906
- // silencieusement ignorées).
1209
+ // STEP 15: FOOTNOTES BLOCK
1210
+ // Appended at the end of the document, only if at least one note was
1211
+ // referenced (notes that are defined but never referenced are
1212
+ // silently ignored).
907
1213
  // ====================================================================
908
1214
  if (!empty($footnoteOrder)) {
909
1215
  $html .= "\n<div class=\"footnotes\">\n<ol>\n";