@kirigami/php-prepros 1.1.0 → 1.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1114 -869
- package/index.d.ts +75 -2
- package/index.js +1 -1
- package/package.json +4 -4
- package/src/libraries/curl.class.php +5 -0
- package/src/libraries/html.class.php +23 -22
- package/src/libraries/img.class.php +356 -16
- package/src/libraries/md.class.php +476 -170
- package/src/libraries/md.plugins.php +19 -19
- package/src/libraries/normalizer.class.php +310 -0
- package/src/libraries/prepros.class.php +36 -7
- package/src/libraries/prepros.plugins.php +24 -0
- package/src/libraries/schema.class.php +550 -0
- package/src/libraries/scraper.class.php +64 -59
- package/src/libraries/str.class.php +49 -15
- package/src/libraries/yaml.class.php +66 -66
- package/src/phpjs/fstat.js +9 -0
- package/src/prepros.js +51 -44
- package/src/utils/getfilestats.js +30 -0
- package/src/utils/isbinary.js +14 -14
- package/src/utils.inc.php +9 -8
|
@@ -3,22 +3,22 @@
|
|
|
3
3
|
class MD {
|
|
4
4
|
|
|
5
5
|
// ========================================================================
|
|
6
|
-
//
|
|
6
|
+
// PLUGIN SYSTEM
|
|
7
7
|
//
|
|
8
|
-
//
|
|
9
|
-
// {%
|
|
8
|
+
// INLINE SYNTAX (args on the same line):
|
|
9
|
+
// {% plugin_name arg1 arg2 "arg with spaces" %}
|
|
10
10
|
//
|
|
11
|
-
//
|
|
12
|
-
// {%
|
|
13
|
-
//
|
|
14
|
-
//
|
|
11
|
+
// BLOCK SYNTAX (multi-line content):
|
|
12
|
+
// {% plugin_name arg1 arg2
|
|
13
|
+
// content line 1
|
|
14
|
+
// content line 2
|
|
15
15
|
// %}
|
|
16
16
|
//
|
|
17
|
-
//
|
|
18
|
-
// - $args :
|
|
19
|
-
// - $body :
|
|
17
|
+
// The callback always receives (array $args, string $body):
|
|
18
|
+
// - $args : array of the arguments passed on the opening line
|
|
19
|
+
// - $body : multi-line content (empty "" for inline tags)
|
|
20
20
|
//
|
|
21
|
-
//
|
|
21
|
+
// Examples:
|
|
22
22
|
// GithubReadmeParser::registerPlugin('codepen', function(array $args, string $body): string {
|
|
23
23
|
// $id = htmlspecialchars($args[0] ?? '', ENT_QUOTES, 'UTF-8');
|
|
24
24
|
// return "<iframe src=\"https://codepen.io/embed/{$id}\"></iframe>";
|
|
@@ -38,25 +38,25 @@ class MD {
|
|
|
38
38
|
private static array $plugins = [];
|
|
39
39
|
|
|
40
40
|
/**
|
|
41
|
-
*
|
|
41
|
+
* Registers a plugin by name.
|
|
42
42
|
*
|
|
43
|
-
* @param string $name
|
|
43
|
+
* @param string $name Tag name, e.g. "codepen"
|
|
44
44
|
* @param callable $callback function(array $args): string
|
|
45
|
-
* $args[0] =
|
|
45
|
+
* $args[0] = first argument, $args[1] = second, etc.
|
|
46
46
|
*/
|
|
47
47
|
public static function registerPlugin(string $name, callable $callback): void {
|
|
48
48
|
self::$plugins[strtolower(trim($name))] = $callback;
|
|
49
49
|
}
|
|
50
50
|
|
|
51
51
|
/**
|
|
52
|
-
*
|
|
52
|
+
* Removes a registered plugin.
|
|
53
53
|
*/
|
|
54
54
|
public static function unregisterPlugin(string $name): void {
|
|
55
55
|
unset(self::$plugins[strtolower(trim($name))]);
|
|
56
56
|
}
|
|
57
57
|
|
|
58
58
|
/**
|
|
59
|
-
*
|
|
59
|
+
* Returns the list of registered plugins.
|
|
60
60
|
*
|
|
61
61
|
* @return string[]
|
|
62
62
|
*/
|
|
@@ -65,7 +65,7 @@ class MD {
|
|
|
65
65
|
}
|
|
66
66
|
|
|
67
67
|
// ========================================================================
|
|
68
|
-
//
|
|
68
|
+
// Generates a "slug"-style id for heading anchors (ATX and Setext).
|
|
69
69
|
// ========================================================================
|
|
70
70
|
private static function slugify(string $text): string {
|
|
71
71
|
$id = strtolower(preg_replace('/[^\w\- ]/u', '', $text));
|
|
@@ -73,17 +73,16 @@ class MD {
|
|
|
73
73
|
}
|
|
74
74
|
|
|
75
75
|
// ========================================================================
|
|
76
|
-
//
|
|
77
|
-
//
|
|
76
|
+
// Converts an indentation width (spaces/tabs) into a column count, a tab
|
|
77
|
+
// counting as 4 spaces.
|
|
78
78
|
// ========================================================================
|
|
79
79
|
private static function indentWidth(string $whitespace): int {
|
|
80
80
|
return strlen(str_replace("\t", ' ', $whitespace));
|
|
81
81
|
}
|
|
82
82
|
|
|
83
83
|
/**
|
|
84
|
-
*
|
|
85
|
-
*
|
|
86
|
-
* mesure de la consommation des items.
|
|
84
|
+
* Recursively builds a (nested) <ol>/<ul> list from a flat array of items
|
|
85
|
+
* { indent, type, text }. $i is advanced as items are consumed.
|
|
87
86
|
*
|
|
88
87
|
* @param array<int, array{indent:int, type:string, text:string}> $items
|
|
89
88
|
*/
|
|
@@ -108,9 +107,9 @@ class MD {
|
|
|
108
107
|
}
|
|
109
108
|
|
|
110
109
|
// ========================================================================
|
|
111
|
-
//
|
|
112
|
-
//
|
|
113
|
-
//
|
|
110
|
+
// EMOJIS (extended syntax): :shortcode: → unicode character.
|
|
111
|
+
// Non-exhaustive table but covering the most common shortcuts; extensible
|
|
112
|
+
// via registerEmoji().
|
|
114
113
|
// ========================================================================
|
|
115
114
|
/** @var array<string, string> */
|
|
116
115
|
private static array $extraEmoji = [];
|
|
@@ -191,10 +190,145 @@ class MD {
|
|
|
191
190
|
'speech_balloon' => '💬', 'thought_balloon' => '💭', 'zzz' => '💤', 'boom2' => '💥',
|
|
192
191
|
'sos' => '🆘', 'new' => '🆕', 'ok' => '🆗', 'up' => '🆙', 'cool' => '🆒',
|
|
193
192
|
'free' => '🆓', 'id' => '🆔', 'ng' => '🆖',
|
|
193
|
+
|
|
194
|
+
// -- Faces and emotions (continued) --------------------------------
|
|
195
|
+
'smiling_face_with_three_hearts' => '🥰', 'kissing' => '😗', 'kissing_closed_eyes' => '😚',
|
|
196
|
+
'kissing_smiling_eyes' => '😙', 'yum' => '😋', 'stuck_out_tongue' => '😛',
|
|
197
|
+
'stuck_out_tongue_winking_eye' => '😜', 'stuck_out_tongue_closed_eyes' => '😝',
|
|
198
|
+
'money_mouth_face' => '🤑', 'hugs' => '🤗', 'disappointed_relieved' => '😥',
|
|
199
|
+
'dizzy_face' => '😵', 'astonished' => '😲', 'open_mouth' => '😮', 'hushed' => '😯',
|
|
200
|
+
'fearful' => '😨', 'cold_sweat' => '😰', 'nauseated_face' => '🤢', 'vomiting_face' => '🤮',
|
|
201
|
+
'sneezing_face' => '🤧', 'face_with_thermometer' => '🤒', 'face_with_head_bandage' => '🤕',
|
|
202
|
+
'woozy_face' => '🥴', 'smiling_imp' => '😈', 'imp' => '👿', 'japanese_ogre' => '👹',
|
|
203
|
+
'japanese_goblin' => '👺', 'skull' => '💀', 'skull_and_crossbones' => '☠️',
|
|
204
|
+
'ghost' => '👻', 'alien' => '👽', 'space_invader' => '👾', 'robot' => '🤖',
|
|
205
|
+
'poop' => '💩', 'clown_face' => '🤡', 'smiley_cat' => '😺', 'smile_cat' => '😸',
|
|
206
|
+
'joy_cat' => '😹', 'heart_eyes_cat' => '😻', 'smirk_cat' => '😼', 'kissing_cat' => '😽',
|
|
207
|
+
'pouting_cat' => '😾', 'crying_cat_face' => '😿',
|
|
208
|
+
|
|
209
|
+
// -- Corps, gestes, personnages --------------------------------
|
|
210
|
+
'raised_hand' => '✋', 'raised_back_of_hand' => '🤚', 'vulcan_salute' => '🖖',
|
|
211
|
+
'pinching_hand' => '🤏', 'fist' => '✊', 'punch' => '👊', 'left_facing_fist' => '🤛',
|
|
212
|
+
'right_facing_fist' => '🤜', 'open_hands' => '👐', 'palms_up_together' => '🤲',
|
|
213
|
+
'nail_care' => '💅', 'selfie' => '🤳', 'ear' => '👂', 'nose' => '👃', 'brain' => '🧠',
|
|
214
|
+
'tongue' => '👅', 'lips' => '👄', 'tooth' => '🦷', 'bone' => '🦴',
|
|
215
|
+
'baby' => '👶', 'child' => '🧒', 'boy' => '👦', 'girl' => '👧', 'adult' => '🧑',
|
|
216
|
+
'man' => '👨', 'woman' => '👩', 'older_adult' => '🧓', 'older_man' => '👴', 'older_woman' => '👵',
|
|
217
|
+
'mage' => '🧙', 'superhero' => '🦸', 'supervillain' => '🦹', 'vampire' => '🧛',
|
|
218
|
+
'zombie' => '🧟', 'genie' => '🧞', 'merperson' => '🧜', 'elf' => '🧝', 'fairy' => '🧚',
|
|
219
|
+
|
|
220
|
+
// -- Animaux (suite) ---------------------------------------------
|
|
221
|
+
'wolf' => '🐺', 'boar' => '🐗', 'racehorse' => '🐎', 'zebra' => '🦓', 'deer' => '🦌',
|
|
222
|
+
'cow2' => '🐄', 'ox' => '🐂', 'water_buffalo' => '🐃', 'pig2' => '🐖', 'ram' => '🐏',
|
|
223
|
+
'sheep' => '🐑', 'goat' => '🐐', 'camel' => '🐫', 'dromedary_camel' => '🐪',
|
|
224
|
+
'llama' => '🦙', 'giraffe' => '🦒', 'elephant' => '🐘', 'rhinoceros' => '🦏',
|
|
225
|
+
'hippopotamus' => '🦛', 'mouse2' => '🐁', 'rat' => '🐀', 'hamster' => '🐹',
|
|
226
|
+
'chipmunk' => '🐿️', 'hedgehog' => '🦔', 'bat' => '🦇', 'duck' => '🦆', 'eagle' => '🦅',
|
|
227
|
+
'flamingo' => '🦩', 'peacock' => '🦚', 'parrot' => '🦜', 'swan' => '🦢',
|
|
228
|
+
'turkey' => '🦃', 'dove' => '🕊️', 'rooster' => '🐓', 'crocodile' => '🐊',
|
|
229
|
+
'turtle' => '🐢', 'lizard' => '🦎', 'snake' => '🐍', 'dragon_face' => '🐲',
|
|
230
|
+
'dragon' => '🐉', 'sauropod' => '🦕', 't-rex' => '🦖', 'whale2' => '🐋',
|
|
231
|
+
'shark' => '🦈', 'seal' => '🦭', 'squid' => '🦑', 'shrimp' => '🦐', 'lobster' => '🦞',
|
|
232
|
+
'crab' => '🦀', 'blowfish' => '🐡', 'tropical_fish' => '🐠', 'oyster' => '🦪',
|
|
233
|
+
'ant' => '🐜', 'spider' => '🕷️', 'spider_web' => '🕸️', 'scorpion' => '🦂',
|
|
234
|
+
'mosquito' => '🦟', 'microbe' => '🦠', 'paw_prints' => '🐾',
|
|
235
|
+
|
|
236
|
+
// -- Nature, plants, weather (continued) --------------------------------
|
|
237
|
+
'cherry_blossom' => '🌸', 'blossom' => '🌼', 'rose' => '🌹', 'wilted_flower' => '🥀',
|
|
238
|
+
'hibiscus' => '🌺', 'sunflower' => '🌻', 'tulip' => '🌷', 'herb' => '🌿',
|
|
239
|
+
'shamrock' => '☘️', 'fallen_leaf' => '🍂', 'leaves' => '🍃', 'mushroom' => '🍄',
|
|
240
|
+
'chestnut' => '🌰', 'crescent_moon' => '🌙', 'full_moon' => '🌕', 'new_moon' => '🌑',
|
|
241
|
+
'milky_way' => '🌌', 'stars' => '🌠', 'cyclone' => '🌀', 'fog' => '🌫️',
|
|
242
|
+
'wind_face' => '🌬️', 'tornado' => '🌪️', 'thunder_cloud_and_rain' => '⛈️',
|
|
243
|
+
'sweat_drops' => '💦', 'snowman' => '⛄', 'snowman_with_snow' => '☃️', 'comet' => '☄️',
|
|
244
|
+
|
|
245
|
+
// -- Nourriture (suite) --------------------------------------------
|
|
246
|
+
'tomato' => '🍅', 'eggplant' => '🍆', 'avocado' => '🥑', 'broccoli' => '🥦',
|
|
247
|
+
'carrot' => '🥕', 'corn' => '🌽', 'hot_pepper' => '🌶️', 'cucumber' => '🥒',
|
|
248
|
+
'potato' => '🥔', 'sweet_potato' => '🍠', 'peanuts' => '🥜', 'honey_pot' => '🍯',
|
|
249
|
+
'croissant' => '🥐', 'bagel' => '🥯', 'pretzel' => '🥨', 'pancakes' => '🥞',
|
|
250
|
+
'waffle' => '🧇', 'meat_on_bone' => '🍖', 'poultry_leg' => '🍗', 'bacon' => '🥓',
|
|
251
|
+
'sandwich' => '🥪', 'stuffed_flatbread' => '🥙', 'burrito' => '🌯', 'salad' => '🥗',
|
|
252
|
+
'shallow_pan_of_food' => '🥘', 'canned_food' => '🥫', 'bento' => '🍱',
|
|
253
|
+
'rice_ball' => '🍙', 'rice' => '🍚', 'curry' => '🍛', 'stew' => '🍲', 'oden' => '🍢',
|
|
254
|
+
'dango' => '🍡', 'shaved_ice' => '🍧', 'ice_cream' => '🍨', 'pie' => '🥧',
|
|
255
|
+
'cupcake' => '🧁', 'moon_cake' => '🥮', 'lollipop' => '🍭', 'custard' => '🍮',
|
|
256
|
+
'milk_glass' => '🥛', 'baby_bottle' => '🍼', 'mate' => '🧉', 'ice_cube' => '🧊',
|
|
257
|
+
'tumbler_glass' => '🥃', 'cup_with_straw' => '🥤', 'chopsticks' => '🥢',
|
|
258
|
+
'fork_and_knife' => '🍴', 'spoon' => '🥄', 'plate_with_cutlery' => '🍽️',
|
|
259
|
+
|
|
260
|
+
// -- Activities, sport, leisure --------------------------------------
|
|
261
|
+
'running' => '🏃', 'walking' => '🚶', 'swimming' => '🏊', 'surfing' => '🏄',
|
|
262
|
+
'skateboard' => '🛹', 'snowboarder' => '🏂', 'weight_lifting' => '🏋️',
|
|
263
|
+
'cyclist' => '🚴', 'medal_military' => '🎖️', 'ticket' => '🎫', 'circus_tent' => '🎪',
|
|
264
|
+
'performing_arts' => '🎭', 'art' => '🎨', 'clapper' => '🎬', 'microphone' => '🎤',
|
|
265
|
+
'headphones' => '🎧', 'musical_note' => '🎵', 'musical_score' => '🎼', 'guitar' => '🎸',
|
|
266
|
+
'violin' => '🎻', 'drum' => '🥁', 'trumpet' => '🎺', 'saxophone' => '🎷',
|
|
267
|
+
'musical_keyboard' => '🎹', 'chess_pawn' => '♟️', 'bowling' => '🎳',
|
|
268
|
+
'ice_skate' => '⛸️', 'ski' => '🎿', 'fishing_pole_and_fish' => '🎣',
|
|
269
|
+
'boxing_glove' => '🥊', 'martial_arts_uniform' => '🥋', 'goal_net' => '🥅',
|
|
270
|
+
'flying_disc' => '🥏', 'yo_yo' => '🪀', 'kite' => '🪁',
|
|
271
|
+
|
|
272
|
+
// -- Voyages et lieux (suite) ----------------------------------------
|
|
273
|
+
'airplane_departure' => '🛫', 'airplane_arriving' => '🛬', 'flying_saucer' => '🛸',
|
|
274
|
+
'motorcycle' => '🏍️', 'scooter' => '🛴', 'tractor' => '🚜', 'truck' => '🚚',
|
|
275
|
+
'articulated_lorry' => '🚛', 'trolleybus' => '🚎', 'minibus' => '🚐', 'metro' => '🚇',
|
|
276
|
+
'station' => '🚉', 'monorail' => '🚝', 'bullettrain_front' => '🚄',
|
|
277
|
+
'steam_locomotive' => '🚂', 'anchor' => '⚓', 'sailboat' => '⛵', 'canoe' => '🛶',
|
|
278
|
+
'speedboat' => '🚤', 'ferry' => '⛴️', 'passport_control' => '🛂', 'customs' => '🛃',
|
|
279
|
+
'baggage_claim' => '🛄', 'left_luggage' => '🛅', 'vertical_traffic_light' => '🚦',
|
|
280
|
+
'construction' => '🚧', 'fuelpump' => '⛽', 'busstop' => '🚏', 'moyai' => '🗿',
|
|
281
|
+
'statue_of_liberty' => '🗽', 'tokyo_tower' => '🗼', 'fountain' => '⛲',
|
|
282
|
+
'stadium' => '🏟️', 'ferris_wheel' => '🎡', 'roller_coaster' => '🎢',
|
|
283
|
+
'carousel_horse' => '🎠', 'beach_umbrella' => '🏖️', 'desert' => '🏜️',
|
|
284
|
+
'desert_island' => '🏝️', 'national_park' => '🏞️', 'sunrise' => '🌅',
|
|
285
|
+
'sunrise_over_mountains' => '🌄', 'sparkler' => '🎇', 'fireworks' => '🎆',
|
|
286
|
+
'city_sunset' => '🌇', 'bridge_at_night' => '🌉', 'houses' => '🏘️',
|
|
287
|
+
'derelict_house' => '🏚️', 'classical_building' => '🏛️', 'department_store' => '🏬',
|
|
288
|
+
'post_office' => '🏣', 'hotel' => '🏨', 'convenience_store' => '🏪', 'bank' => '🏦',
|
|
289
|
+
'factory' => '🏭',
|
|
290
|
+
|
|
291
|
+
// -- Objets (suite) ---------------------------------------------------
|
|
292
|
+
'watch' => '⌚', 'stopwatch' => '⏱️', 'timer_clock' => '⏲️', 'joystick' => '🕹️',
|
|
293
|
+
'floppy_disk' => '💾', 'cd' => '💿', 'dvd' => '📀', 'movie_camera' => '🎥',
|
|
294
|
+
'projector' => '📽️', 'telephone' => '☎️', 'pager' => '📟', 'fax' => '📠',
|
|
295
|
+
'candle' => '🕯️', 'fire_extinguisher' => '🧯', 'oil_drum' => '🛢️',
|
|
296
|
+
'money_with_wings' => '💸', 'credit_card' => '💳', 'yen' => '💴', 'euro' => '💶',
|
|
297
|
+
'pound' => '💷', 'briefcase' => '💼', 'balance_scale' => '⚖️', 'compass' => '🧭',
|
|
298
|
+
'triangular_ruler' => '📐', 'straight_ruler' => '📏', 'round_pushpin' => '📍',
|
|
299
|
+
'scissors' => '✂️', 'thread' => '🧵', 'yarn' => '🧶', 'safety_pin' => '🧷',
|
|
300
|
+
'basket' => '🧺', 'hourglass_flowing_sand' => '⏳', 'notebook' => '📓',
|
|
301
|
+
'notebook_with_decorative_cover' => '📔', 'page_facing_up' => '📄',
|
|
302
|
+
'page_with_curl' => '📃', 'bookmark_tabs' => '📑', 'bookmark' => '🔖',
|
|
303
|
+
'label' => '🏷️', 'receipt' => '🧾', 'card_index' => '📇', 'wastebasket' => '🗑️',
|
|
304
|
+
'old_key' => '🗝️', 'hammer_and_wrench' => '🛠️', 'pick' => '⛏️', 'shield' => '🛡️',
|
|
305
|
+
'syringe' => '💉', 'pill' => '💊', 'thermometer' => '🌡️', 'soap' => '🧼',
|
|
306
|
+
'broom' => '🧹',
|
|
307
|
+
|
|
308
|
+
// -- Symboles (suite) -----------------------------------------------
|
|
309
|
+
'heavy_multiplication_x' => '✖️', 'heavy_plus_sign' => '➕', 'heavy_minus_sign' => '➖',
|
|
310
|
+
'heavy_division_sign' => '➗', 'infinity' => '♾️', 'recycle' => '♻️', 'trident' => '🔱',
|
|
311
|
+
'atom_symbol' => '⚛️', 'om' => '🕉️', 'peace_symbol' => '☮️', 'yin_yang' => '☯️',
|
|
312
|
+
'wheel_of_dharma' => '☸️', 'star_of_david' => '✡️', 'star_and_crescent' => '☪️',
|
|
313
|
+
'cross' => '✝️', 'menorah' => '🕎', 'radioactive' => '☢️', 'biohazard' => '☣️',
|
|
314
|
+
'arrow_up' => '⬆️', 'arrow_down' => '⬇️', 'arrow_left' => '⬅️', 'arrow_right' => '➡️',
|
|
315
|
+
'arrow_upper_right' => '↗️', 'arrow_lower_right' => '↘️', 'arrow_lower_left' => '↙️',
|
|
316
|
+
'arrow_upper_left' => '↖️', 'arrows_clockwise' => '🔃', 'arrows_counterclockwise' => '🔄',
|
|
317
|
+
'back' => '🔙', 'end' => '🔚', 'on' => '🔛', 'soon' => '🔜', 'top' => '🔝',
|
|
318
|
+
'radio_button' => '🔘', 'red_circle' => '🔴', 'orange_circle' => '🟠',
|
|
319
|
+
'yellow_circle' => '🟡', 'green_circle' => '🟢', 'blue_circle' => '🔵',
|
|
320
|
+
'purple_circle' => '🟣', 'brown_circle' => '🟤', 'white_circle' => '⚪',
|
|
321
|
+
'black_circle' => '⚫',
|
|
322
|
+
|
|
323
|
+
// -- Drapeaux (suite) -------------------------------------------------
|
|
324
|
+
'triangular_flag_on_post' => '🚩', 'crossed_flags' => '🎌',
|
|
325
|
+
'us' => '🇺🇸', 'gb' => '🇬🇧', 'fr' => '🇫🇷', 'de' => '🇩🇪', 'es' => '🇪🇸',
|
|
326
|
+
'it' => '🇮🇹', 'jp' => '🇯🇵', 'cn' => '🇨🇳', 'kr' => '🇰🇷', 'ca' => '🇨🇦',
|
|
327
|
+
'au' => '🇦🇺', 'br' => '🇧🇷', 'in' => '🇮🇳', 'ru' => '🇷🇺', 'eu' => '🇪🇺',
|
|
194
328
|
];
|
|
195
329
|
|
|
196
330
|
/**
|
|
197
|
-
*
|
|
331
|
+
* Registers (or replaces) a custom emoji shortcut.
|
|
198
332
|
*/
|
|
199
333
|
public static function registerEmoji(string $shortcode, string $char): void {
|
|
200
334
|
self::$extraEmoji[strtolower(trim($shortcode, ':'))] = $char;
|
|
@@ -206,12 +340,12 @@ class MD {
|
|
|
206
340
|
}
|
|
207
341
|
|
|
208
342
|
// ========================================================================
|
|
209
|
-
//
|
|
210
|
-
//
|
|
211
|
-
// :
|
|
212
|
-
//
|
|
213
|
-
//
|
|
214
|
-
//
|
|
343
|
+
// DEFINITION LISTS (extended syntax)
|
|
344
|
+
// Term
|
|
345
|
+
// : Definition
|
|
346
|
+
// Procedural line-by-line analysis (safer than a single big regex for
|
|
347
|
+
// grouping several term/definition pairs into one <dl>, separated by a
|
|
348
|
+
// blank line or not).
|
|
215
349
|
// ========================================================================
|
|
216
350
|
private static function isDefinitionColonLine(string $line): bool {
|
|
217
351
|
return (bool) preg_match('/^[ \t]*:[ \t]+.+$/', $line);
|
|
@@ -252,8 +386,8 @@ class MD {
|
|
|
252
386
|
$i++;
|
|
253
387
|
}
|
|
254
388
|
|
|
255
|
-
//
|
|
256
|
-
//
|
|
389
|
+
// A single blank line between two groups stays in the same <dl>
|
|
390
|
+
// if the next group really is a new term.
|
|
257
391
|
if ($i < $n && trim($lines[$i]) === '') {
|
|
258
392
|
$j = $i;
|
|
259
393
|
while ($j < $n && trim($lines[$j]) === '') $j++;
|
|
@@ -275,35 +409,169 @@ class MD {
|
|
|
275
409
|
return implode("\n", $out);
|
|
276
410
|
}
|
|
277
411
|
|
|
412
|
+
// ========================================================================
|
|
413
|
+
// RAW HTML (safe subset, GitHub README style)
|
|
414
|
+
//
|
|
415
|
+
// Writing HTML tags directly (e.g. <div align="center">, <img>, <sub>,
|
|
416
|
+
// <br>, HTML tables...) is allowed ONLY if:
|
|
417
|
+
// - the tag is part of the HTML_ALLOWED_TAGS whitelist;
|
|
418
|
+
// - every attribute is part of the whitelist for that tag (or of the
|
|
419
|
+
// global HTML_GLOBAL_ATTRS attributes);
|
|
420
|
+
// - no attribute starts with "on" (onclick, onerror, ...);
|
|
421
|
+
// - URLs (href/src) use a safe scheme (isSafeUrl).
|
|
422
|
+
//
|
|
423
|
+
// Any unknown or dangerous tag (script/style/iframe/...), or any
|
|
424
|
+
// non-whitelisted attribute, is silently stripped. The text content
|
|
425
|
+
// between the tags is NOT swallowed: it keeps being processed as normal
|
|
426
|
+
// markdown (that's what lets you have markdown headings, badges and images
|
|
427
|
+
// inside a <div align="center">...</div>).
|
|
428
|
+
// ========================================================================
|
|
429
|
+
|
|
430
|
+
private const HTML_ALLOWED_TAGS = [
|
|
431
|
+
'div', 'span', 'p', 'br', 'hr', 'wbr',
|
|
432
|
+
'b', 'strong', 'i', 'em', 'u', 's', 'strike', 'del', 'ins',
|
|
433
|
+
'mark', 'small', 'sub', 'sup', 'kbd', 'code', 'pre', 'abbr', 'q', 'cite',
|
|
434
|
+
'ul', 'ol', 'li', 'dl', 'dt', 'dd',
|
|
435
|
+
'table', 'thead', 'tbody', 'tfoot', 'tr', 'td', 'th', 'caption', 'colgroup', 'col',
|
|
436
|
+
'blockquote',
|
|
437
|
+
'a', 'img', 'picture', 'source', 'figure', 'figcaption',
|
|
438
|
+
'h1', 'h2', 'h3', 'h4', 'h5', 'h6',
|
|
439
|
+
'details', 'summary', 'center',
|
|
440
|
+
];
|
|
441
|
+
|
|
442
|
+
/** Attributes allowed on any whitelisted tag. */
|
|
443
|
+
private const HTML_GLOBAL_ATTRS = ['id', 'class', 'title', 'align', 'valign', 'width', 'height', 'dir', 'lang'];
|
|
444
|
+
|
|
445
|
+
/** Extra attributes allowed, per tag. */
|
|
446
|
+
private const HTML_TAG_ATTRS = [
|
|
447
|
+
'a' => ['href', 'name', 'target', 'rel'],
|
|
448
|
+
'img' => ['src', 'alt', 'loading', 'srcset', 'sizes'],
|
|
449
|
+
'source' => ['src', 'srcset', 'type', 'media'],
|
|
450
|
+
'td' => ['colspan', 'rowspan'],
|
|
451
|
+
'th' => ['colspan', 'rowspan', 'scope'],
|
|
452
|
+
'col' => ['span'],
|
|
453
|
+
'ol' => ['start', 'type'],
|
|
454
|
+
'details' => ['open'],
|
|
455
|
+
];
|
|
456
|
+
|
|
457
|
+
/** Self-closing tags (no closing tag expected). */
|
|
458
|
+
private const HTML_VOID_TAGS = ['img', 'br', 'hr', 'wbr', 'source', 'col'];
|
|
459
|
+
|
|
460
|
+
/**
|
|
461
|
+
* Checks that a URL (href/src) uses a safe scheme: relative links, anchors,
|
|
462
|
+
* http(s), mailto, tel, or base64-encoded images (png/gif/jpeg/webp only —
|
|
463
|
+
* not svg+xml, which can embed a <script>).
|
|
464
|
+
* Notably rejects javascript:, vbscript:, data:text/html.
|
|
465
|
+
*/
|
|
466
|
+
private static function isSafeUrl(string $url): bool {
|
|
467
|
+
$url = trim($url);
|
|
468
|
+
if ($url === '') return true;
|
|
469
|
+
// A path with no explicit scheme ("assets/x.png", "../x", "#anchor",
|
|
470
|
+
// "/x", "x") is a relative link or an anchor: always safe.
|
|
471
|
+
if (!preg_match('~^([a-zA-Z][a-zA-Z0-9+.\-]*):~', $url, $m)) return true;
|
|
472
|
+
$scheme = strtolower($m[1]);
|
|
473
|
+
if (in_array($scheme, ['http', 'https', 'mailto', 'tel'], true)) return true;
|
|
474
|
+
if ($scheme === 'data') {
|
|
475
|
+
// Base64-encoded images only — no data:image/svg+xml, which can
|
|
476
|
+
// embed a <script>, and no data:text/html.
|
|
477
|
+
return (bool) preg_match('~^data:image/(png|gif|jpe?g|webp);base64,~i', $url);
|
|
478
|
+
}
|
|
479
|
+
return false; // javascript:, vbscript:, file:, etc. → rejected
|
|
480
|
+
}
|
|
481
|
+
|
|
482
|
+
/**
|
|
483
|
+
* Sanitizes a single raw HTML tag (e.g. '<div align="center">', '</div>',
|
|
484
|
+
* '<img src="..." onerror="...">').
|
|
485
|
+
*
|
|
486
|
+
* @return string|null The cleaned tag to keep, an empty string to strip it
|
|
487
|
+
* silently, or null if it doesn't look like a valid
|
|
488
|
+
* HTML tag (in which case the caller strips it too, to
|
|
489
|
+
* be safe).
|
|
490
|
+
*/
|
|
491
|
+
private static function sanitizeHtmlTag(string $tag): ?string {
|
|
492
|
+
if (!preg_match(
|
|
493
|
+
'/^<(\/)?([a-zA-Z][a-zA-Z0-9-]*)((?:\s+[a-zA-Z_:][a-zA-Z0-9_:.-]*(?:\s*=\s*(?:"[^"]*"|\'[^\']*\'|[^\s"\'>]+))?)*)\s*(\/)?>$/s',
|
|
494
|
+
$tag,
|
|
495
|
+
$m
|
|
496
|
+
)) {
|
|
497
|
+
return null;
|
|
498
|
+
}
|
|
499
|
+
|
|
500
|
+
$closing = $m[1] === '/';
|
|
501
|
+
$tagName = strtolower($m[2]);
|
|
502
|
+
$attrsRaw = $m[3];
|
|
503
|
+
|
|
504
|
+
if (!in_array($tagName, self::HTML_ALLOWED_TAGS, true)) {
|
|
505
|
+
return null;
|
|
506
|
+
}
|
|
507
|
+
|
|
508
|
+
if ($closing) {
|
|
509
|
+
return "</{$tagName}>";
|
|
510
|
+
}
|
|
511
|
+
|
|
512
|
+
$allowedAttrs = array_merge(self::HTML_GLOBAL_ATTRS, self::HTML_TAG_ATTRS[$tagName] ?? []);
|
|
513
|
+
$safeAttrs = '';
|
|
514
|
+
|
|
515
|
+
if (preg_match_all(
|
|
516
|
+
'/([a-zA-Z_:][a-zA-Z0-9_:.-]*)(?:\s*=\s*("([^"]*)"|\'([^\']*)\'|([^\s"\'>]+)))?/',
|
|
517
|
+
$attrsRaw,
|
|
518
|
+
$am,
|
|
519
|
+
PREG_SET_ORDER
|
|
520
|
+
)) {
|
|
521
|
+
foreach ($am as $a) {
|
|
522
|
+
$attrName = strtolower($a[1]);
|
|
523
|
+
if ($attrName === '') continue;
|
|
524
|
+
if (str_starts_with($attrName, 'on')) continue; // safety net against JS handlers
|
|
525
|
+
if (!in_array($attrName, $allowedAttrs, true)) continue;
|
|
526
|
+
|
|
527
|
+
if ($tagName === 'details' && $attrName === 'open') {
|
|
528
|
+
$safeAttrs .= ' open';
|
|
529
|
+
continue;
|
|
530
|
+
}
|
|
531
|
+
|
|
532
|
+
$attrVal = $a[3] ?? ($a[4] ?? ($a[5] ?? ''));
|
|
533
|
+
|
|
534
|
+
if (in_array($attrName, ['href', 'src'], true) && !self::isSafeUrl($attrVal)) {
|
|
535
|
+
continue;
|
|
536
|
+
}
|
|
537
|
+
|
|
538
|
+
$safeAttrs .= ' ' . $attrName . '="' . htmlspecialchars($attrVal, ENT_QUOTES, 'UTF-8') . '"';
|
|
539
|
+
}
|
|
540
|
+
}
|
|
541
|
+
|
|
542
|
+
$close = in_array($tagName, self::HTML_VOID_TAGS, true) ? ' /' : '';
|
|
543
|
+
return "<{$tagName}{$safeAttrs}{$close}>";
|
|
544
|
+
}
|
|
545
|
+
|
|
278
546
|
// ========================================================================
|
|
279
547
|
|
|
280
548
|
public static function toHtml(string $markdown): string {
|
|
281
549
|
|
|
282
550
|
// ====================================================================
|
|
283
|
-
//
|
|
551
|
+
// STEP 1: Line-ending normalization
|
|
284
552
|
// ====================================================================
|
|
285
553
|
$html = str_replace(["\r\n", "\r"], "\n", $markdown);
|
|
286
554
|
|
|
287
555
|
|
|
288
556
|
// ====================================================================
|
|
289
|
-
//
|
|
290
|
-
//
|
|
557
|
+
// STEP 2: PLUGINS
|
|
558
|
+
// Two forms supported:
|
|
291
559
|
//
|
|
292
|
-
// INLINE
|
|
560
|
+
// INLINE: {% name arg1 "arg 2" %}
|
|
293
561
|
// → $args = ['arg1', 'arg 2'], $body = ''
|
|
294
562
|
//
|
|
295
|
-
//
|
|
296
|
-
// → $args = ['arg1'], $body = "
|
|
563
|
+
// BLOCK : {% name arg1\ncontent\nover\nseveral lines\n%}
|
|
564
|
+
// → $args = ['arg1'], $body = "content\nover\nseveral lines"
|
|
297
565
|
//
|
|
298
|
-
//
|
|
299
|
-
//
|
|
300
|
-
//
|
|
566
|
+
// Both are captured by a single regex that tells apart the presence of
|
|
567
|
+
// a newline after the args (block) or not (inline).
|
|
568
|
+
// Processed before XSS encoding — re-injected as the very last step.
|
|
301
569
|
// ====================================================================
|
|
302
570
|
$pluginBlocks = [];
|
|
303
571
|
|
|
304
572
|
/**
|
|
305
|
-
*
|
|
306
|
-
*
|
|
573
|
+
* Parses an argument string into an array.
|
|
574
|
+
* Supports bare words, "double quotes" and 'single quotes'.
|
|
307
575
|
*/
|
|
308
576
|
$parseArgs = static function (string $rawArgs): array {
|
|
309
577
|
$args = [];
|
|
@@ -324,18 +592,18 @@ class MD {
|
|
|
324
592
|
};
|
|
325
593
|
|
|
326
594
|
$html = preg_replace_callback(
|
|
327
|
-
//
|
|
328
|
-
//
|
|
329
|
-
//
|
|
595
|
+
// Group 1: plugin name
|
|
596
|
+
// Group 2: inline args (everything on the first line after the name)
|
|
597
|
+
// Group 3: multi-line body (present only for block tags)
|
|
330
598
|
'/\{%\s*([a-zA-Z0-9_-]+)([^\n%]*?)(?:\n([\s\S]*?))?\s*%\}/m',
|
|
331
599
|
function ($matches) use (&$pluginBlocks, $parseArgs): string {
|
|
332
600
|
$name = strtolower(trim($matches[1]));
|
|
333
601
|
$args = $parseArgs(trim($matches[2] ?? ''));
|
|
334
|
-
// $matches[3]
|
|
602
|
+
// $matches[3] exists only if the tag is multi-line
|
|
335
603
|
$body = isset($matches[3]) ? trim($matches[3]) : '';
|
|
336
604
|
|
|
337
605
|
if (!isset(self::$plugins[$name])) {
|
|
338
|
-
//
|
|
606
|
+
// Unknown plugin: kept encoded rather than silently removed
|
|
339
607
|
return htmlspecialchars($matches[0], ENT_QUOTES, 'UTF-8');
|
|
340
608
|
}
|
|
341
609
|
|
|
@@ -349,17 +617,17 @@ class MD {
|
|
|
349
617
|
|
|
350
618
|
|
|
351
619
|
// ====================================================================
|
|
352
|
-
//
|
|
353
|
-
// [^1]:
|
|
354
|
-
// [^bignote]:
|
|
620
|
+
// STEP 2a: FOOTNOTE DEFINITIONS
|
|
621
|
+
// [^1]: Note text.
|
|
622
|
+
// [^bignote]: First line.
|
|
355
623
|
//
|
|
356
|
-
//
|
|
624
|
+
// Following paragraph, indented by 4 spaces or 1 tab.
|
|
357
625
|
//
|
|
358
|
-
// `{
|
|
359
|
-
//
|
|
360
|
-
//
|
|
361
|
-
//
|
|
362
|
-
//
|
|
626
|
+
// `{ some code }`
|
|
627
|
+
// Extracted (and removed from the text) BEFORE reference link
|
|
628
|
+
// definitions, since [^label]: would otherwise match their regex too.
|
|
629
|
+
// Each note's content is rendered via a recursive call to toHtml() to
|
|
630
|
+
// support multiple paragraphs, code, etc.
|
|
363
631
|
// ====================================================================
|
|
364
632
|
$footnoteDefs = [];
|
|
365
633
|
$html = preg_replace_callback(
|
|
@@ -381,12 +649,12 @@ class MD {
|
|
|
381
649
|
|
|
382
650
|
|
|
383
651
|
// ====================================================================
|
|
384
|
-
//
|
|
385
|
-
// [label]: https://example.com "
|
|
386
|
-
// [label]: <https://example.com> '
|
|
387
|
-
// [label]: https://example.com (
|
|
388
|
-
//
|
|
389
|
-
//
|
|
652
|
+
// STEP 2b: REFERENCE LINK DEFINITIONS
|
|
653
|
+
// [label]: https://example.com "Optional title"
|
|
654
|
+
// [label]: <https://example.com> 'Optional title'
|
|
655
|
+
// [label]: https://example.com (Optional title)
|
|
656
|
+
// Extracted (and removed from the text) before everything else; used
|
|
657
|
+
// later by the [text][label] / [text][] links.
|
|
390
658
|
// ====================================================================
|
|
391
659
|
$refDefs = [];
|
|
392
660
|
$html = preg_replace_callback(
|
|
@@ -402,7 +670,7 @@ class MD {
|
|
|
402
670
|
|
|
403
671
|
|
|
404
672
|
// ====================================================================
|
|
405
|
-
//
|
|
673
|
+
// STEP 3: CODE BLOCKS (```lang ... ```)
|
|
406
674
|
// ====================================================================
|
|
407
675
|
$codeBlocks = [];
|
|
408
676
|
$html = preg_replace_callback('/^```([a-zA-Z0-9_+-]*)\n([\s\S]*?)\n^```/m', function ($matches) use (&$codeBlocks) {
|
|
@@ -414,10 +682,10 @@ class MD {
|
|
|
414
682
|
}, $html);
|
|
415
683
|
|
|
416
684
|
// ====================================================================
|
|
417
|
-
//
|
|
418
|
-
//
|
|
419
|
-
// document)
|
|
420
|
-
//
|
|
685
|
+
// STEP 3a: INDENTED CODE BLOCKS (4 spaces or 1 tab)
|
|
686
|
+
// Recognized only when preceded by a blank line (or the start of the
|
|
687
|
+
// document) and followed by a blank line (or the end of the document),
|
|
688
|
+
// to avoid conflicts with the indentation of nested lists.
|
|
421
689
|
// ====================================================================
|
|
422
690
|
$html = preg_replace_callback(
|
|
423
691
|
'/(?<=\n\n|^)((?:[ ]{4}|\t)[^\n]*(?:\n(?:[ ]{4}|\t)[^\n]*)*)(?=\n\n|\n*$)/',
|
|
@@ -434,13 +702,13 @@ class MD {
|
|
|
434
702
|
$html
|
|
435
703
|
);
|
|
436
704
|
|
|
437
|
-
//
|
|
705
|
+
// Inline code with double backticks (lets you include a literal backtick)
|
|
438
706
|
$inlineCodes = [];
|
|
439
707
|
$html = preg_replace_callback('/``(.+?)``/s', function ($matches) use (&$inlineCodes) {
|
|
440
708
|
$content = $matches[1];
|
|
441
|
-
//
|
|
442
|
-
//
|
|
443
|
-
//
|
|
709
|
+
// Standard convention: if the content starts and ends with a space
|
|
710
|
+
// (and isn't only spaces), strip one space on each side — handy for
|
|
711
|
+
// wrapping a ` at the edge.
|
|
444
712
|
if (preg_match('/^ (.*[^ ]) $/s', $content, $trim)) {
|
|
445
713
|
$content = $trim[1];
|
|
446
714
|
}
|
|
@@ -450,7 +718,7 @@ class MD {
|
|
|
450
718
|
return $placeholder;
|
|
451
719
|
}, $html);
|
|
452
720
|
|
|
453
|
-
//
|
|
721
|
+
// Inline code (`...`)
|
|
454
722
|
$html = preg_replace_callback('/`([^`\n]+)`/', function ($matches) use (&$inlineCodes) {
|
|
455
723
|
$code = htmlspecialchars($matches[1], ENT_QUOTES, 'UTF-8');
|
|
456
724
|
$placeholder = "\x02IC" . count($inlineCodes) . "\x03";
|
|
@@ -460,10 +728,10 @@ class MD {
|
|
|
460
728
|
|
|
461
729
|
|
|
462
730
|
// ====================================================================
|
|
463
|
-
//
|
|
464
|
-
//
|
|
465
|
-
//
|
|
466
|
-
//
|
|
731
|
+
// STEP 3b: CHARACTER ESCAPING (\* \_ \# etc.)
|
|
732
|
+
// Processed after code extraction (code stays literal) and before
|
|
733
|
+
// everything else, so that \* doesn't open emphasis, \# doesn't create
|
|
734
|
+
// a heading, \- doesn't create a list, etc.
|
|
467
735
|
// ====================================================================
|
|
468
736
|
$escapes = [];
|
|
469
737
|
$html = preg_replace_callback(
|
|
@@ -475,9 +743,9 @@ class MD {
|
|
|
475
743
|
},
|
|
476
744
|
$html
|
|
477
745
|
);
|
|
478
|
-
// |
|
|
479
|
-
//
|
|
480
|
-
//
|
|
746
|
+
// | is the documented convention (Markdown Extra / PHP Markdown)
|
|
747
|
+
// for showing a literal pipe in a table cell without it being
|
|
748
|
+
// interpreted as a column separator.
|
|
481
749
|
$html = preg_replace_callback(
|
|
482
750
|
'/|/i',
|
|
483
751
|
function () use (&$escapes): string {
|
|
@@ -490,9 +758,9 @@ class MD {
|
|
|
490
758
|
|
|
491
759
|
|
|
492
760
|
// ====================================================================
|
|
493
|
-
//
|
|
494
|
-
//
|
|
495
|
-
//
|
|
761
|
+
// STEP 3d: AUTOMATIC LINKS <https://...> and <email@example.com>
|
|
762
|
+
// Processed before XSS encoding because the < > characters would be
|
|
763
|
+
// encoded to < > and the regex would no longer match.
|
|
496
764
|
// ====================================================================
|
|
497
765
|
$autolinks = [];
|
|
498
766
|
$html = preg_replace_callback('/<(https?:\/\/[^\s<>]+)>/', function ($m) use (&$autolinks): string {
|
|
@@ -510,13 +778,13 @@ class MD {
|
|
|
510
778
|
|
|
511
779
|
|
|
512
780
|
// ====================================================================
|
|
513
|
-
//
|
|
514
|
-
//
|
|
515
|
-
//
|
|
781
|
+
// STEP 3e: GFM ALERTS AND BLOCKQUOTES
|
|
782
|
+
// Processed before XSS encoding because the > character would be
|
|
783
|
+
// encoded to > and the regexes would no longer match.
|
|
516
784
|
// ====================================================================
|
|
517
785
|
$blockquotes = [];
|
|
518
786
|
|
|
519
|
-
//
|
|
787
|
+
// GFM alerts (> [!NOTE], etc.) — more specific, processed first
|
|
520
788
|
$html = preg_replace_callback(
|
|
521
789
|
'/^(>\s*\[!(NOTE|TIP|IMPORTANT|WARNING|CAUTION)\]\n(?:>[ \t]?[^\n]*\n?)*)/m',
|
|
522
790
|
function ($matches) use (&$blockquotes): string {
|
|
@@ -529,38 +797,75 @@ class MD {
|
|
|
529
797
|
$blockquotes[$placeholder] = "<div class=\"markdown-alert markdown-alert-{$type}\">"
|
|
530
798
|
. "<p class=\"markdown-alert-title\">{$label}</p>"
|
|
531
799
|
. "<p>{$content}</p></div>";
|
|
532
|
-
//
|
|
533
|
-
// placeholder
|
|
534
|
-
//
|
|
535
|
-
//
|
|
800
|
+
// The trailing \n consumed by the regex is re-injected after
|
|
801
|
+
// the placeholder so the following blank line doesn't merge
|
|
802
|
+
// with the placeholder's line (which would break, for example,
|
|
803
|
+
// detecting a Setext heading right after).
|
|
536
804
|
return $placeholder . (str_ends_with($matches[1], "\n") ? "\n" : '');
|
|
537
805
|
},
|
|
538
806
|
$html
|
|
539
807
|
);
|
|
540
808
|
|
|
541
|
-
//
|
|
542
|
-
//
|
|
543
|
-
//
|
|
809
|
+
// Standard blockquotes (nesting handled by recursion through toHtml,
|
|
810
|
+
// which re-applies this same rule to the content already stripped of
|
|
811
|
+
// one ">" level)
|
|
544
812
|
$html = preg_replace_callback('/^((?:>[ \t]?[^\n]*\n?)+)/m', function ($matches) use (&$blockquotes): string {
|
|
545
813
|
$content = preg_replace('/^>[ \t]?/m', '', $matches[1]);
|
|
546
|
-
//
|
|
814
|
+
// The two trailing spaces are left as-is: toHtml() handles them itself
|
|
547
815
|
$inner = self::toHtml(trim($content));
|
|
548
816
|
$placeholder = "\x02BQ" . count($blockquotes) . "\x03";
|
|
549
817
|
$blockquotes[$placeholder] = "<blockquote>{$inner}</blockquote>";
|
|
550
|
-
//
|
|
818
|
+
// See the comment above: preserve the trailing \n that was consumed.
|
|
551
819
|
return $placeholder . (str_ends_with($matches[1], "\n") ? "\n" : '');
|
|
552
820
|
}, $html);
|
|
553
821
|
|
|
554
822
|
|
|
555
823
|
// ====================================================================
|
|
556
|
-
//
|
|
824
|
+
// STEP 3f: RAW HTML (safe subset, GitHub README style)
|
|
825
|
+
// Processed before XSS encoding because the < > characters would be
|
|
826
|
+
// encoded to < > and no longer recognized as tags.
|
|
827
|
+
// The content between the tags is not swallowed: it stays in the
|
|
828
|
+
// stream and keeps being processed as normal markdown.
|
|
829
|
+
// ====================================================================
|
|
830
|
+
|
|
831
|
+
// Intrinsically dangerous elements: removed along with their content
|
|
832
|
+
// (script/style/iframe can embed JS or load a third-party page;
|
|
833
|
+
// form/button/textarea/select/option have no place in markdown
|
|
834
|
+
// content).
|
|
835
|
+
$html = preg_replace(
|
|
836
|
+
'/<(script|style|iframe|object|embed|noscript|template|form|button|textarea|select|option)\b[^>]*>[\s\S]*?<\/\1>/i',
|
|
837
|
+
'',
|
|
838
|
+
$html
|
|
839
|
+
);
|
|
840
|
+
|
|
841
|
+
$rawHtml = [];
|
|
842
|
+
$html = preg_replace_callback(
|
|
843
|
+
'/<!--[\s\S]*?-->|<\/?[a-zA-Z][a-zA-Z0-9-]*(?:\s+[a-zA-Z_:][a-zA-Z0-9_:.-]*(?:\s*=\s*(?:"[^"]*"|\'[^\']*\'|[^\s"\'>]+))?)*\s*\/?>/',
|
|
844
|
+
function ($m) use (&$rawHtml): string {
|
|
845
|
+
$tag = $m[0];
|
|
846
|
+
// HTML comment: invisible, safe to remove.
|
|
847
|
+
if (str_starts_with($tag, '<!--')) return '';
|
|
848
|
+
|
|
849
|
+
$sanitized = self::sanitizeHtmlTag($tag);
|
|
850
|
+
if ($sanitized === null || $sanitized === '') return '';
|
|
851
|
+
|
|
852
|
+
$placeholder = "\x02HT" . count($rawHtml) . "\x03";
|
|
853
|
+
$rawHtml[$placeholder] = $sanitized;
|
|
854
|
+
return $placeholder;
|
|
855
|
+
},
|
|
856
|
+
$html
|
|
857
|
+
);
|
|
858
|
+
|
|
859
|
+
|
|
860
|
+
// ====================================================================
|
|
861
|
+
// STEP 4: Global XSS encoding
|
|
557
862
|
// ====================================================================
|
|
558
863
|
$html = htmlspecialchars($html, ENT_NOQUOTES, 'UTF-8');
|
|
559
864
|
|
|
560
865
|
|
|
561
866
|
// ====================================================================
|
|
562
|
-
//
|
|
563
|
-
//
|
|
867
|
+
// STEP 5: GFM TABLES
|
|
868
|
+
// Supports rows with or without a trailing pipe (| col | or | col)
|
|
564
869
|
// ====================================================================
|
|
565
870
|
$html = preg_replace_callback(
|
|
566
871
|
'/^(\|[^\n]+\|?\n)([ \t]*\|[ \t]*:?-+:?[ \t]*(?:\|[ \t]*:?-+:?[ \t]*)*\|?\n)((?:\|[^\n]+\|?\n?)+)/m',
|
|
@@ -608,26 +913,26 @@ class MD {
|
|
|
608
913
|
|
|
609
914
|
|
|
610
915
|
// ====================================================================
|
|
611
|
-
//
|
|
916
|
+
// STEP 6: (GFM alerts and blockquotes handled in step 3e)
|
|
612
917
|
// ====================================================================
|
|
613
918
|
|
|
614
919
|
|
|
615
920
|
// ====================================================================
|
|
616
|
-
//
|
|
921
|
+
// STEP 7: TASK LISTS (GFM checkboxes)
|
|
617
922
|
// ====================================================================
|
|
618
923
|
$html = preg_replace('/^[ \t]*[-*+] \[ \] (.+)$/m', '<li class="task-item"><input type="checkbox" disabled /> $1</li>', $html);
|
|
619
924
|
$html = preg_replace('/^[ \t]*[-*+] \[[xX]\] (.+)$/m', '<li class="task-item"><input type="checkbox" checked disabled /> $1</li>', $html);
|
|
620
925
|
|
|
621
926
|
|
|
622
927
|
// ====================================================================
|
|
623
|
-
//
|
|
624
|
-
//
|
|
928
|
+
// STEP 7b: SETEXT HEADINGS (alternative == / -- syntax)
|
|
929
|
+
// Title
|
|
625
930
|
// ===== → <h1>
|
|
626
931
|
//
|
|
627
|
-
//
|
|
932
|
+
// Title
|
|
628
933
|
// ----- → <h2>
|
|
629
|
-
//
|
|
630
|
-
//
|
|
934
|
+
// Processed before ATX headings and before horizontal rules (a line of
|
|
935
|
+
// dashes right after a line of text is a heading, not an <hr>).
|
|
631
936
|
// ====================================================================
|
|
632
937
|
$html = preg_replace_callback(
|
|
633
938
|
'/^(?![ \t]*(?:#{1,6}[ \t]|>|```|\||[-*+][ \t]|\d+\.[ \t]))[ \t]*(\S.*?)[ \t]*(?:\{#([a-zA-Z0-9_\-:.]+)\}[ \t]*)?\n[ \t]*=+[ \t]*$/m',
|
|
@@ -650,7 +955,7 @@ class MD {
|
|
|
650
955
|
|
|
651
956
|
|
|
652
957
|
// ====================================================================
|
|
653
|
-
//
|
|
958
|
+
// STEP 8: HEADINGS (ATX: # to ######)
|
|
654
959
|
// ====================================================================
|
|
655
960
|
$html = preg_replace_callback(
|
|
656
961
|
'/^(#{1,6})[ \t]+(.+?)[ \t]*(?:\{#([a-zA-Z0-9_\-:.]+)\}[ \t]*)?(?:[ \t]+#+)?$/m',
|
|
@@ -665,13 +970,13 @@ class MD {
|
|
|
665
970
|
|
|
666
971
|
|
|
667
972
|
// ====================================================================
|
|
668
|
-
//
|
|
669
|
-
//
|
|
670
|
-
//
|
|
671
|
-
//
|
|
672
|
-
//
|
|
673
|
-
//
|
|
674
|
-
//
|
|
973
|
+
// STEP 9: LISTS (bullets and ordered, with nesting)
|
|
974
|
+
// A single pass detects a contiguous block of lines that are either a
|
|
975
|
+
// bullet (-,*,+) or a numbered item, whatever their indentation level;
|
|
976
|
+
// the block is then rebuilt recursively into nested <ol>/<ul>
|
|
977
|
+
// according to the relative indentation depth.
|
|
978
|
+
// Task items (already converted to <li class="task-item">) no longer
|
|
979
|
+
// match this pattern and are therefore not re-wrapped here.
|
|
675
980
|
// ====================================================================
|
|
676
981
|
$html = preg_replace_callback(
|
|
677
982
|
'/^([ \t]*(?:\d+\.|[-*+])[ \t]+.+(?:\n[ \t]*(?:\d+\.|[-*+])[ \t]+.+)*)/m',
|
|
@@ -686,7 +991,7 @@ class MD {
|
|
|
686
991
|
}
|
|
687
992
|
}
|
|
688
993
|
if (empty($items)) return $matches[1];
|
|
689
|
-
//
|
|
994
|
+
// Normalize the lowest indentation level to 0
|
|
690
995
|
$minIndent = min(array_column($items, 'indent'));
|
|
691
996
|
foreach ($items as &$it) $it['indent'] -= $minIndent;
|
|
692
997
|
unset($it);
|
|
@@ -707,26 +1012,26 @@ class MD {
|
|
|
707
1012
|
|
|
708
1013
|
|
|
709
1014
|
// ====================================================================
|
|
710
|
-
//
|
|
711
|
-
//
|
|
712
|
-
// :
|
|
1015
|
+
// STEP 9b: DEFINITION LISTS (extended syntax)
|
|
1016
|
+
// Term
|
|
1017
|
+
// : Definition
|
|
713
1018
|
// ====================================================================
|
|
714
1019
|
$html = self::extractDefinitionLists($html);
|
|
715
1020
|
|
|
716
1021
|
|
|
717
1022
|
// ====================================================================
|
|
718
|
-
//
|
|
719
|
-
//
|
|
720
|
-
//
|
|
721
|
-
//
|
|
722
|
-
//
|
|
723
|
-
//
|
|
1023
|
+
// STEP 9c: FOOTNOTE REFERENCES [^label]
|
|
1024
|
+
// Converted BEFORE emphasis so they don't collide with the new
|
|
1025
|
+
// superscript ^text^ (a [^1] followed later by a [^2] on the same line
|
|
1026
|
+
// could otherwise be read as ^1] ... [^2^).
|
|
1027
|
+
// Numbering is sequential, in order of first appearance in the text
|
|
1028
|
+
// (as documented).
|
|
724
1029
|
// ====================================================================
|
|
725
1030
|
$footnoteOrder = [];
|
|
726
1031
|
$html = preg_replace_callback('/\[\^([^\]\s]+)\]/', function ($m) use (&$footnoteOrder, &$footnoteDefs): string {
|
|
727
1032
|
$label = strtolower(trim($m[1]));
|
|
728
1033
|
if (!isset($footnoteDefs[$label])) {
|
|
729
|
-
//
|
|
1034
|
+
// Reference to an undefined note: left as-is.
|
|
730
1035
|
return $m[0];
|
|
731
1036
|
}
|
|
732
1037
|
if (!isset($footnoteOrder[$label])) {
|
|
@@ -738,32 +1043,32 @@ class MD {
|
|
|
738
1043
|
|
|
739
1044
|
|
|
740
1045
|
// ====================================================================
|
|
741
|
-
//
|
|
742
|
-
//
|
|
1046
|
+
// STEP 10: INLINE TEXT (Bold, Italic, Strikethrough, Highlight,
|
|
1047
|
+
// Subscript/Superscript, Emoji)
|
|
743
1048
|
// ====================================================================
|
|
744
1049
|
$html = preg_replace('/\*\*\*(.+?)\*\*\*/s', '<strong><em>$1</em></strong>', $html);
|
|
745
1050
|
$html = preg_replace('/___(.+?)___/s', '<strong><em>$1</em></strong>', $html);
|
|
746
1051
|
$html = preg_replace('/\*\*(.+?)\*\*/s', '<strong>$1</strong>', $html);
|
|
747
1052
|
$html = preg_replace('/__(.+?)__/s', '<strong>$1</strong>', $html);
|
|
748
1053
|
$html = preg_replace('/\*(.+?)\*/s', '<em>$1</em>', $html);
|
|
749
|
-
//
|
|
750
|
-
//
|
|
1054
|
+
// Italic _ must only match at word boundaries so it doesn't capture
|
|
1055
|
+
// snake_case, package names (@php-wasm/node), etc.
|
|
751
1056
|
$html = preg_replace('/(?<!\w)_([^_\n]+)_(?!\w)/', '<em>$1</em>', $html);
|
|
752
|
-
//
|
|
1057
|
+
// Highlight ==text== (extended syntax)
|
|
753
1058
|
$html = preg_replace('/==(.+?)==/s', '<mark>$1</mark>', $html);
|
|
754
|
-
//
|
|
755
|
-
//
|
|
1059
|
+
// Strikethrough ~~text~~ — processed BEFORE subscript (single ~) so the
|
|
1060
|
+
// latter doesn't match half of a double-tilde pair.
|
|
756
1061
|
$html = preg_replace('/~~(.+?)~~/s', '<del>$1</del>', $html);
|
|
757
|
-
//
|
|
758
|
-
//
|
|
759
|
-
//
|
|
1062
|
+
// Superscript ^text^ (extended syntax) — placing it before the note
|
|
1063
|
+
// reference escaping ([^label]) is not a problem: those are wrapped in
|
|
1064
|
+
// brackets and so don't form an isolated ^...^ pair.
|
|
760
1065
|
$html = preg_replace('/\^([^\^\n]+)\^/', '<sup>$1</sup>', $html);
|
|
761
|
-
//
|
|
762
|
-
//
|
|
1066
|
+
// Subscript ~text~ (a single tilde; the ~~ were already consumed just
|
|
1067
|
+
// above by strikethrough).
|
|
763
1068
|
$html = preg_replace('/~([^~\n]+)~/', '<sub>$1</sub>', $html);
|
|
764
1069
|
|
|
765
|
-
//
|
|
766
|
-
//
|
|
1070
|
+
// Emojis :shortcode: (extended syntax) — unknown shortcuts are left
|
|
1071
|
+
// as-is rather than silently removed.
|
|
767
1072
|
$html = preg_replace_callback('/:([a-zA-Z0-9_+\-]+):/', function ($m): string {
|
|
768
1073
|
$emoji = self::emojiFor($m[1]);
|
|
769
1074
|
return $emoji ?? $m[0];
|
|
@@ -771,9 +1076,9 @@ class MD {
|
|
|
771
1076
|
|
|
772
1077
|
|
|
773
1078
|
// ====================================================================
|
|
774
|
-
//
|
|
775
|
-
//
|
|
776
|
-
//
|
|
1079
|
+
// STEP 11: LINKS & IMAGES
|
|
1080
|
+
// External links (https?://) get target="_blank" + rel="noopener noreferrer".
|
|
1081
|
+
// Internal links (/page, #anchor, ../thing) don't.
|
|
777
1082
|
// ====================================================================
|
|
778
1083
|
$html = preg_replace(
|
|
779
1084
|
'/!\[([^\]]*)\]\(([^)\s]+)(?:\s+"([^"]*)")?\)/',
|
|
@@ -789,7 +1094,7 @@ class MD {
|
|
|
789
1094
|
return "<a href=\"{$href}\"{$titleAttr}{$extern}>{$text}</a>";
|
|
790
1095
|
};
|
|
791
1096
|
|
|
792
|
-
//
|
|
1097
|
+
// Reference links [text][label] and [text][] (shortcut = label = text)
|
|
793
1098
|
$html = preg_replace_callback(
|
|
794
1099
|
'/\[([^\]]+)\]\[([^\]]*)\]/',
|
|
795
1100
|
function ($m) use (&$refDefs, $buildLink): string {
|
|
@@ -802,7 +1107,7 @@ class MD {
|
|
|
802
1107
|
$html
|
|
803
1108
|
);
|
|
804
1109
|
|
|
805
|
-
//
|
|
1110
|
+
// Markdown links [text](url "optional title")
|
|
806
1111
|
$html = preg_replace_callback(
|
|
807
1112
|
'/\[([^\]]+)\]\(([^)\s]+)(?:\s+"([^"]*)")?\)/',
|
|
808
1113
|
function ($m) use ($buildLink): string {
|
|
@@ -811,9 +1116,9 @@ class MD {
|
|
|
811
1116
|
$html
|
|
812
1117
|
);
|
|
813
1118
|
|
|
814
|
-
//
|
|
815
|
-
//
|
|
816
|
-
//
|
|
1119
|
+
// Bare URLs https://... (extended syntax: auto-link without brackets).
|
|
1120
|
+
// Excludes those already inside quotes/attributes (href="...") or
|
|
1121
|
+
// already turned into a link, so they don't get doubled.
|
|
817
1122
|
$html = preg_replace(
|
|
818
1123
|
'/(?<!["\'=>])\b(https?:\/\/[^\s<>"\')\]]+)/',
|
|
819
1124
|
'<a href="$1" target="_blank" rel="noopener noreferrer">$1</a>',
|
|
@@ -822,29 +1127,29 @@ class MD {
|
|
|
822
1127
|
|
|
823
1128
|
|
|
824
1129
|
// ====================================================================
|
|
825
|
-
//
|
|
1130
|
+
// STEP 12: HORIZONTAL RULES
|
|
826
1131
|
// ====================================================================
|
|
827
1132
|
$html = preg_replace('/^(?:[-*_][ \t]*){3,}$/m', '<hr />', $html);
|
|
828
1133
|
|
|
829
1134
|
|
|
830
1135
|
// ====================================================================
|
|
831
|
-
//
|
|
832
|
-
//
|
|
833
|
-
//
|
|
834
|
-
//
|
|
835
|
-
//
|
|
1136
|
+
// STEP 13: PARAGRAPHS
|
|
1137
|
+
// Strategy: process line by line. Lines that start with a block-level
|
|
1138
|
+
// tag or a placeholder are left as-is. Consecutive raw-text lines are
|
|
1139
|
+
// accumulated then wrapped in a <p> when a block line or a blank line
|
|
1140
|
+
// is reached.
|
|
836
1141
|
// ====================================================================
|
|
837
1142
|
$blockStartTags = ['<h', '<pre', '<ul', '<ol', '<li', '<table', '<thead', '<tbody',
|
|
838
1143
|
'<tr', '<td', '<th', '<blockquote', '<div', '<hr', '<img',
|
|
839
1144
|
'<dl', '<dt', '<dd',
|
|
840
|
-
"\x02CB", "\x02PLG", "\x02BQ"];
|
|
1145
|
+
"\x02CB", "\x02PLG", "\x02BQ", "\x02HT"];
|
|
841
1146
|
|
|
842
1147
|
$isBlockLine = static function (string $line) use ($blockStartTags): bool {
|
|
843
1148
|
$t = ltrim($line);
|
|
844
1149
|
if ($t === '') return false;
|
|
845
|
-
//
|
|
846
|
-
//
|
|
847
|
-
//
|
|
1150
|
+
// Any closing tag (</...>) is always treated as a "block" line:
|
|
1151
|
+
// this keeps a closing </table>, </thead>, </tr>, etc. from being
|
|
1152
|
+
// absorbed into a surrounding <p>.
|
|
848
1153
|
if (str_starts_with($t, '</')) return true;
|
|
849
1154
|
foreach ($blockStartTags as $tag) {
|
|
850
1155
|
if (str_starts_with($t, $tag)) return true;
|
|
@@ -860,10 +1165,10 @@ class MD {
|
|
|
860
1165
|
if (empty($textBuffer)) return;
|
|
861
1166
|
$content = implode("\n", $textBuffer);
|
|
862
1167
|
if (trim($content) !== '') {
|
|
863
|
-
//
|
|
1168
|
+
// Two trailing spaces → <br> (standard markdown convention)
|
|
864
1169
|
$content = preg_replace('/ $/m', '<br>', $content);
|
|
865
|
-
//
|
|
866
|
-
//
|
|
1170
|
+
// Single line break → space (GitHub behavior)
|
|
1171
|
+
// Unless already converted to <br> above
|
|
867
1172
|
$content = preg_replace('/(?<!r>)\n/', ' ', $content);
|
|
868
1173
|
$output[] = '<p>' . trim($content) . '</p>';
|
|
869
1174
|
}
|
|
@@ -875,7 +1180,7 @@ class MD {
|
|
|
875
1180
|
$flushBuffer();
|
|
876
1181
|
$output[] = $line;
|
|
877
1182
|
} elseif (trim($line) === '') {
|
|
878
|
-
//
|
|
1183
|
+
// Blank line = paragraph separator
|
|
879
1184
|
$flushBuffer();
|
|
880
1185
|
} else {
|
|
881
1186
|
$textBuffer[] = $line;
|
|
@@ -887,23 +1192,24 @@ class MD {
|
|
|
887
1192
|
|
|
888
1193
|
|
|
889
1194
|
// ====================================================================
|
|
890
|
-
//
|
|
1195
|
+
// STEP 14: Re-inject the placeholders
|
|
891
1196
|
// ====================================================================
|
|
892
1197
|
$html = strtr($html, $pluginBlocks);
|
|
893
1198
|
$html = strtr($html, $blockquotes);
|
|
1199
|
+
$html = strtr($html, $rawHtml);
|
|
894
1200
|
$html = strtr($html, $codeBlocks);
|
|
895
1201
|
$html = strtr($html, $inlineCodes);
|
|
896
1202
|
$html = strtr($html, $autolinks);
|
|
897
|
-
//
|
|
898
|
-
//
|
|
1203
|
+
// The escapes are re-injected last, once no Markdown regex can
|
|
1204
|
+
// interpret them anymore.
|
|
899
1205
|
$html = strtr($html, $escapes);
|
|
900
1206
|
|
|
901
1207
|
|
|
902
1208
|
// ====================================================================
|
|
903
|
-
//
|
|
904
|
-
//
|
|
905
|
-
//
|
|
906
|
-
//
|
|
1209
|
+
// STEP 15: FOOTNOTES BLOCK
|
|
1210
|
+
// Appended at the end of the document, only if at least one note was
|
|
1211
|
+
// referenced (notes that are defined but never referenced are
|
|
1212
|
+
// silently ignored).
|
|
907
1213
|
// ====================================================================
|
|
908
1214
|
if (!empty($footnoteOrder)) {
|
|
909
1215
|
$html .= "\n<div class=\"footnotes\">\n<ol>\n";
|