@kirigami/php-prepros 1.9.3 → 3.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +502 -21
- package/README.md +1988 -1817
- package/index.d.ts +62 -26
- package/index.js +1 -1
- package/package.json +59 -59
- package/src/imagebatch.php +68 -67
- package/src/libraries/aliases.inc.php +879 -879
- package/src/libraries/curl.class.php +7 -7
- package/src/libraries/ld.class.php +803 -804
- package/src/libraries/md-legacy.class.php +1257 -0
- package/src/libraries/md.class.php +51 -1249
- package/src/libraries/meta.class.php +496 -506
- package/src/libraries/{normalizer.class.php → normalizer-legacy.class.php} +11 -1
- package/src/libraries/prepros.class.php +85 -8
- package/src/libraries/schema-legacy.class.php +549 -0
- package/src/libraries/schema.class.php +99 -472
- package/src/libraries/yaml-legacy.class.php +598 -0
- package/src/libraries/yaml.class.php +41 -406
- package/src/prepros.js +152 -48
- package/src/prepros.php +4 -5
- package/src/runenv.php +2 -2
- package/src/utils.inc.php +12 -2
|
@@ -1,1255 +1,57 @@
|
|
|
1
1
|
<?php
|
|
2
2
|
|
|
3
|
-
|
|
4
|
-
|
|
5
|
-
|
|
6
|
-
|
|
7
|
-
|
|
8
|
-
|
|
9
|
-
|
|
10
|
-
|
|
11
|
-
|
|
12
|
-
|
|
13
|
-
|
|
14
|
-
|
|
15
|
-
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
self::$plugins[strtolower(trim($name))] = $callback;
|
|
49
|
-
}
|
|
50
|
-
|
|
51
|
-
/**
|
|
52
|
-
* Removes a registered plugin.
|
|
53
|
-
*/
|
|
54
|
-
public static function unregisterPlugin(string $name): void {
|
|
55
|
-
unset(self::$plugins[strtolower(trim($name))]);
|
|
56
|
-
}
|
|
57
|
-
|
|
58
|
-
/**
|
|
59
|
-
* Returns the list of registered plugins.
|
|
60
|
-
*
|
|
61
|
-
* @return string[]
|
|
62
|
-
*/
|
|
63
|
-
public static function getRegisteredPlugins(): array {
|
|
64
|
-
return array_keys(self::$plugins);
|
|
65
|
-
}
|
|
66
|
-
|
|
67
|
-
// ========================================================================
|
|
68
|
-
// Generates a "slug"-style id for heading anchors (ATX and Setext).
|
|
69
|
-
// ========================================================================
|
|
70
|
-
private static function slugify(string $text): string {
|
|
71
|
-
$id = strtolower(preg_replace('/[^\w\- ]/u', '', $text));
|
|
72
|
-
return preg_replace('/\s+/', '-', trim($id));
|
|
73
|
-
}
|
|
74
|
-
|
|
75
|
-
// ========================================================================
|
|
76
|
-
// Converts an indentation width (spaces/tabs) into a column count, a tab
|
|
77
|
-
// counting as 4 spaces.
|
|
78
|
-
// ========================================================================
|
|
79
|
-
private static function indentWidth(string $whitespace): int {
|
|
80
|
-
return strlen(str_replace("\t", ' ', $whitespace));
|
|
81
|
-
}
|
|
82
|
-
|
|
83
|
-
/**
|
|
84
|
-
* Recursively builds a (nested) <ol>/<ul> list from a flat array of items
|
|
85
|
-
* { indent, type, text }. $i is advanced as items are consumed.
|
|
86
|
-
*
|
|
87
|
-
* @param array<int, array{indent:int, type:string, text:string}> $items
|
|
88
|
-
*/
|
|
89
|
-
private static function buildListTree(array $items, int &$i, int $count): string {
|
|
90
|
-
$type = $items[$i]['type'];
|
|
91
|
-
$baseIndent = $items[$i]['indent'];
|
|
92
|
-
$out = "<{$type}>\n";
|
|
93
|
-
|
|
94
|
-
while ($i < $count && $items[$i]['indent'] === $baseIndent && $items[$i]['type'] === $type) {
|
|
95
|
-
$text = $items[$i]['text'];
|
|
96
|
-
$i++;
|
|
97
|
-
|
|
98
|
-
$nested = '';
|
|
99
|
-
if ($i < $count && $items[$i]['indent'] > $baseIndent) {
|
|
100
|
-
$nested = "\n" . self::buildListTree($items, $i, $count);
|
|
101
|
-
}
|
|
102
|
-
|
|
103
|
-
$out .= " <li>{$text}{$nested}</li>\n";
|
|
104
|
-
}
|
|
105
|
-
|
|
106
|
-
return $out . "</{$type}>";
|
|
107
|
-
}
|
|
108
|
-
|
|
109
|
-
// ========================================================================
|
|
110
|
-
// EMOJIS (extended syntax): :shortcode: → unicode character.
|
|
111
|
-
// Non-exhaustive table but covering the most common shortcuts; extensible
|
|
112
|
-
// via registerEmoji().
|
|
113
|
-
// ========================================================================
|
|
114
|
-
/** @var array<string, string> */
|
|
115
|
-
private static array $extraEmoji = [];
|
|
116
|
-
|
|
117
|
-
private static array $emojiMap = [
|
|
118
|
-
'smile' => '😄', 'smiley' => '😃', 'grin' => '😁', 'joy' => '😂', 'rofl' => '🤣',
|
|
119
|
-
'blush' => '😊', 'wink' => '😉', 'relaxed' => '☺️', 'slight_smile' => '🙂',
|
|
120
|
-
'upside_down_face' => '🙃', 'innocent' => '😇', 'heart_eyes' => '😍', 'kissing_heart' => '😘',
|
|
121
|
-
'thinking' => '🤔', 'neutral_face' => '😐', 'expressionless' => '😑', 'no_mouth' => '😶',
|
|
122
|
-
'roll_eyes' => '🙄', 'smirk' => '😏', 'unamused' => '😒', 'grimacing' => '😬',
|
|
123
|
-
'lying_face' => '🤥', 'relieved' => '😌', 'pensive' => '😔', 'sleepy' => '😪',
|
|
124
|
-
'drooling_face' => '🤤', 'sleeping' => '😴', 'mask' => '😷', 'sunglasses' => '😎',
|
|
125
|
-
'star_struck' => '🤩', 'partying_face' => '🥳', 'worried' => '😟', 'frowning' => '☹️',
|
|
126
|
-
'confused' => '😕', 'slightly_frowning_face' => '🙁', 'cry' => '😢', 'sob' => '😭',
|
|
127
|
-
'scream' => '😱', 'confounded' => '😖', 'persevere' => '😣', 'disappointed' => '😞',
|
|
128
|
-
'sweat' => '😓', 'weary' => '😩', 'tired_face' => '😫', 'yawning_face' => '🥱',
|
|
129
|
-
'triumph' => '😤', 'rage' => '😡', 'angry' => '😠', 'cursing_face' => '🤬',
|
|
130
|
-
'exploding_head' => '🤯', 'flushed' => '😳', 'hot_face' => '🥵', 'cold_face' => '🥶',
|
|
131
|
-
'scream_cat' => '🙀', 'nerd_face' => '🤓', 'monocle_face' => '🧐', 'zany_face' => '🤪',
|
|
132
|
-
'raised_eyebrow' => '🤨', 'shushing_face' => '🤫', 'zipper_mouth_face' => '🤐',
|
|
133
|
-
'heart' => '❤️', 'orange_heart' => '🧡', 'yellow_heart' => '💛', 'green_heart' => '💚',
|
|
134
|
-
'blue_heart' => '💙', 'purple_heart' => '💜', 'black_heart' => '🖤', 'white_heart' => '🤍',
|
|
135
|
-
'broken_heart' => '💔', 'two_hearts' => '💕', 'sparkling_heart' => '💖', 'heartbeat' => '💓',
|
|
136
|
-
'thumbsup' => '👍', '+1' => '👍', 'thumbsdown' => '👎', '-1' => '👎',
|
|
137
|
-
'clap' => '👏', 'raised_hands' => '🙌', 'pray' => '🙏', 'wave' => '👋',
|
|
138
|
-
'ok_hand' => '👌', 'v' => '✌️', 'crossed_fingers' => '🤞', 'muscle' => '💪',
|
|
139
|
-
'point_up' => '☝️', 'point_down' => '👇', 'point_left' => '👈', 'point_right' => '👉',
|
|
140
|
-
'handshake' => '🤝', 'writing_hand' => '✍️', 'fire' => '🔥', 'star' => '⭐',
|
|
141
|
-
'star2' => '🌟', 'sparkles' => '✨', 'zap' => '⚡', 'boom' => '💥', 'collision' => '💥',
|
|
142
|
-
'rocket' => '🚀', 'tada' => '🎉', 'confetti_ball' => '🎊', 'gift' => '🎁',
|
|
143
|
-
'balloon' => '🎈', 'trophy' => '🏆', 'medal' => '🏅', 'crown' => '👑',
|
|
144
|
-
'gem' => '💎', 'moneybag' => '💰', 'dollar' => '💵', '100' => '💯',
|
|
145
|
-
'warning' => '⚠️', 'no_entry' => '⛔', 'stop_sign' => '🛑', 'checkered_flag' => '🏁',
|
|
146
|
-
'white_check_mark' => '✅', 'heavy_check_mark' => '✔️', 'x' => '❌', 'negative_squared_cross_mark' => '❎',
|
|
147
|
-
'question' => '❓', 'grey_question' => '❔', 'exclamation' => '❗', 'bangbang' => '‼️',
|
|
148
|
-
'interrobang' => '⁉️', 'bulb' => '💡', 'bell' => '🔔', 'no_bell' => '🔕',
|
|
149
|
-
'lock' => '🔒', 'unlock' => '🔓', 'key' => '🔑', 'mag' => '🔍', 'link' => '🔗',
|
|
150
|
-
'pushpin' => '📌', 'paperclip' => '📎', 'calendar' => '📅', 'clock' => '🕐',
|
|
151
|
-
'hourglass' => '⌛', 'alarm_clock' => '⏰', 'memo' => '📝', 'pencil2' => '✏️',
|
|
152
|
-
'book' => '📖', 'books' => '📚', 'newspaper' => '📰', 'email' => '📧',
|
|
153
|
-
'envelope' => '✉️', 'inbox_tray' => '📥', 'outbox_tray' => '📤', 'package' => '📦',
|
|
154
|
-
'file_folder' => '📁', 'open_file_folder' => '📂', 'clipboard' => '📋',
|
|
155
|
-
'chart_with_upwards_trend' => '📈', 'chart_with_downwards_trend' => '📉', 'bar_chart' => '📊',
|
|
156
|
-
'computer' => '💻', 'desktop_computer' => '🖥️', 'keyboard' => '⌨️', 'printer' => '🖨️',
|
|
157
|
-
'phone' => '📱', 'iphone' => '📱', 'camera' => '📷', 'video_camera' => '📹',
|
|
158
|
-
'tv' => '📺', 'radio' => '📻', 'battery' => '🔋', 'electric_plug' => '🔌',
|
|
159
|
-
'bug' => '🐛', 'beetle' => '🪲', 'gear' => '⚙️', 'wrench' => '🔧', 'hammer' => '🔨',
|
|
160
|
-
'nut_and_bolt' => '🔩', 'toolbox' => '🧰', 'test_tube' => '🧪', 'microscope' => '🔬',
|
|
161
|
-
'satellite' => '🛰️', 'globe_with_meridians' => '🌐', 'earth_americas' => '🌎',
|
|
162
|
-
'sun' => '☀️', 'sunny' => '☀️', 'partly_sunny' => '⛅', 'cloud' => '☁️',
|
|
163
|
-
'rainbow' => '🌈', 'umbrella' => '☂️', 'snowflake' => '❄️', 'droplet' => '💧',
|
|
164
|
-
'ocean' => '🌊', 'tent' => '⛺', 'camping' => '🏕️', 'mountain' => '⛰️',
|
|
165
|
-
'evergreen_tree' => '🌲', 'deciduous_tree' => '🌳', 'palm_tree' => '🌴',
|
|
166
|
-
'cactus' => '🌵', 'seedling' => '🌱', 'four_leaf_clover' => '🍀', 'maple_leaf' => '🍁',
|
|
167
|
-
'dog' => '🐶', 'cat' => '🐱', 'mouse' => '🐭', 'rabbit' => '🐰', 'fox_face' => '🦊',
|
|
168
|
-
'bear' => '🐻', 'panda_face' => '🐼', 'koala' => '🐨', 'tiger' => '🐯', 'lion' => '🦁',
|
|
169
|
-
'cow' => '🐮', 'pig' => '🐷', 'frog' => '🐸', 'monkey_face' => '🐵', 'chicken' => '🐔',
|
|
170
|
-
'penguin' => '🐧', 'bird' => '🐦', 'baby_chick' => '🐤', 'owl' => '🦉',
|
|
171
|
-
'horse' => '🐴', 'unicorn' => '🦄', 'bee' => '🐝', 'butterfly' => '🦋', 'snail' => '🐌',
|
|
172
|
-
'octopus' => '🐙', 'fish' => '🐟', 'dolphin' => '🐬', 'whale' => '🐳',
|
|
173
|
-
'pizza' => '🍕', 'hamburger' => '🍔', 'fries' => '🍟', 'hotdog' => '🌭',
|
|
174
|
-
'taco' => '🌮', 'sushi' => '🍣', 'ramen' => '🍜', 'spaghetti' => '🍝',
|
|
175
|
-
'bread' => '🍞', 'cheese' => '🧀', 'egg' => '🥚', 'popcorn' => '🍿',
|
|
176
|
-
'cookie' => '🍪', 'doughnut' => '🍩', 'cake' => '🍰', 'birthday' => '🎂',
|
|
177
|
-
'candy' => '🍬', 'chocolate_bar' => '🍫', 'icecream' => '🍦', 'apple' => '🍎',
|
|
178
|
-
'banana' => '🍌', 'grapes' => '🍇', 'watermelon' => '🍉', 'strawberry' => '🍓',
|
|
179
|
-
'lemon' => '🍋', 'peach' => '🍑', 'coffee' => '☕', 'tea' => '🍵', 'beer' => '🍺',
|
|
180
|
-
'beers' => '🍻', 'wine_glass' => '🍷', 'cocktail' => '🍸', 'tropical_drink' => '🍹',
|
|
181
|
-
'champagne' => '🍾', 'soccer' => '⚽', 'basketball' => '🏀', 'football' => '🏈',
|
|
182
|
-
'baseball' => '⚾', 'tennis' => '🎾', 'volleyball' => '🏐', 'rugby_football' => '🏉',
|
|
183
|
-
'8ball' => '🎱', 'golf' => '⛳', 'dart' => '🎯', 'video_game' => '🎮',
|
|
184
|
-
'game_die' => '🎲', 'jigsaw' => '🧩', 'car' => '🚗', 'taxi' => '🚕', 'bus' => '🚌',
|
|
185
|
-
'ambulance' => '🚑', 'fire_engine' => '🚒', 'police_car' => '🚓', 'bike' => '🚲',
|
|
186
|
-
'airplane' => '✈️', 'helicopter' => '🚁', 'train' => '🚆', 'ship' => '🚢',
|
|
187
|
-
'house' => '🏠', 'office' => '🏢', 'hospital' => '🏥', 'school' => '🏫',
|
|
188
|
-
'church' => '⛪', 'castle' => '🏰', 'world_map' => '🗺️', 'flag_white' => '🏳️',
|
|
189
|
-
'flag_black' => '🏴', 'checkered_flag2' => '🏁', 'eyes' => '👀', 'eye' => '👁️',
|
|
190
|
-
'speech_balloon' => '💬', 'thought_balloon' => '💭', 'zzz' => '💤', 'boom2' => '💥',
|
|
191
|
-
'sos' => '🆘', 'new' => '🆕', 'ok' => '🆗', 'up' => '🆙', 'cool' => '🆒',
|
|
192
|
-
'free' => '🆓', 'id' => '🆔', 'ng' => '🆖',
|
|
193
|
-
|
|
194
|
-
// -- Faces and emotions (continued) --------------------------------
|
|
195
|
-
'smiling_face_with_three_hearts' => '🥰', 'kissing' => '😗', 'kissing_closed_eyes' => '😚',
|
|
196
|
-
'kissing_smiling_eyes' => '😙', 'yum' => '😋', 'stuck_out_tongue' => '😛',
|
|
197
|
-
'stuck_out_tongue_winking_eye' => '😜', 'stuck_out_tongue_closed_eyes' => '😝',
|
|
198
|
-
'money_mouth_face' => '🤑', 'hugs' => '🤗', 'disappointed_relieved' => '😥',
|
|
199
|
-
'dizzy_face' => '😵', 'astonished' => '😲', 'open_mouth' => '😮', 'hushed' => '😯',
|
|
200
|
-
'fearful' => '😨', 'cold_sweat' => '😰', 'nauseated_face' => '🤢', 'vomiting_face' => '🤮',
|
|
201
|
-
'sneezing_face' => '🤧', 'face_with_thermometer' => '🤒', 'face_with_head_bandage' => '🤕',
|
|
202
|
-
'woozy_face' => '🥴', 'smiling_imp' => '😈', 'imp' => '👿', 'japanese_ogre' => '👹',
|
|
203
|
-
'japanese_goblin' => '👺', 'skull' => '💀', 'skull_and_crossbones' => '☠️',
|
|
204
|
-
'ghost' => '👻', 'alien' => '👽', 'space_invader' => '👾', 'robot' => '🤖',
|
|
205
|
-
'poop' => '💩', 'clown_face' => '🤡', 'smiley_cat' => '😺', 'smile_cat' => '😸',
|
|
206
|
-
'joy_cat' => '😹', 'heart_eyes_cat' => '😻', 'smirk_cat' => '😼', 'kissing_cat' => '😽',
|
|
207
|
-
'pouting_cat' => '😾', 'crying_cat_face' => '😿',
|
|
208
|
-
|
|
209
|
-
// -- Corps, gestes, personnages --------------------------------
|
|
210
|
-
'raised_hand' => '✋', 'raised_back_of_hand' => '🤚', 'vulcan_salute' => '🖖',
|
|
211
|
-
'pinching_hand' => '🤏', 'fist' => '✊', 'punch' => '👊', 'left_facing_fist' => '🤛',
|
|
212
|
-
'right_facing_fist' => '🤜', 'open_hands' => '👐', 'palms_up_together' => '🤲',
|
|
213
|
-
'nail_care' => '💅', 'selfie' => '🤳', 'ear' => '👂', 'nose' => '👃', 'brain' => '🧠',
|
|
214
|
-
'tongue' => '👅', 'lips' => '👄', 'tooth' => '🦷', 'bone' => '🦴',
|
|
215
|
-
'baby' => '👶', 'child' => '🧒', 'boy' => '👦', 'girl' => '👧', 'adult' => '🧑',
|
|
216
|
-
'man' => '👨', 'woman' => '👩', 'older_adult' => '🧓', 'older_man' => '👴', 'older_woman' => '👵',
|
|
217
|
-
'mage' => '🧙', 'superhero' => '🦸', 'supervillain' => '🦹', 'vampire' => '🧛',
|
|
218
|
-
'zombie' => '🧟', 'genie' => '🧞', 'merperson' => '🧜', 'elf' => '🧝', 'fairy' => '🧚',
|
|
219
|
-
|
|
220
|
-
// -- Animaux (suite) ---------------------------------------------
|
|
221
|
-
'wolf' => '🐺', 'boar' => '🐗', 'racehorse' => '🐎', 'zebra' => '🦓', 'deer' => '🦌',
|
|
222
|
-
'cow2' => '🐄', 'ox' => '🐂', 'water_buffalo' => '🐃', 'pig2' => '🐖', 'ram' => '🐏',
|
|
223
|
-
'sheep' => '🐑', 'goat' => '🐐', 'camel' => '🐫', 'dromedary_camel' => '🐪',
|
|
224
|
-
'llama' => '🦙', 'giraffe' => '🦒', 'elephant' => '🐘', 'rhinoceros' => '🦏',
|
|
225
|
-
'hippopotamus' => '🦛', 'mouse2' => '🐁', 'rat' => '🐀', 'hamster' => '🐹',
|
|
226
|
-
'chipmunk' => '🐿️', 'hedgehog' => '🦔', 'bat' => '🦇', 'duck' => '🦆', 'eagle' => '🦅',
|
|
227
|
-
'flamingo' => '🦩', 'peacock' => '🦚', 'parrot' => '🦜', 'swan' => '🦢',
|
|
228
|
-
'turkey' => '🦃', 'dove' => '🕊️', 'rooster' => '🐓', 'crocodile' => '🐊',
|
|
229
|
-
'turtle' => '🐢', 'lizard' => '🦎', 'snake' => '🐍', 'dragon_face' => '🐲',
|
|
230
|
-
'dragon' => '🐉', 'sauropod' => '🦕', 't-rex' => '🦖', 'whale2' => '🐋',
|
|
231
|
-
'shark' => '🦈', 'seal' => '🦭', 'squid' => '🦑', 'shrimp' => '🦐', 'lobster' => '🦞',
|
|
232
|
-
'crab' => '🦀', 'blowfish' => '🐡', 'tropical_fish' => '🐠', 'oyster' => '🦪',
|
|
233
|
-
'ant' => '🐜', 'spider' => '🕷️', 'spider_web' => '🕸️', 'scorpion' => '🦂',
|
|
234
|
-
'mosquito' => '🦟', 'microbe' => '🦠', 'paw_prints' => '🐾',
|
|
235
|
-
|
|
236
|
-
// -- Nature, plants, weather (continued) --------------------------------
|
|
237
|
-
'cherry_blossom' => '🌸', 'blossom' => '🌼', 'rose' => '🌹', 'wilted_flower' => '🥀',
|
|
238
|
-
'hibiscus' => '🌺', 'sunflower' => '🌻', 'tulip' => '🌷', 'herb' => '🌿',
|
|
239
|
-
'shamrock' => '☘️', 'fallen_leaf' => '🍂', 'leaves' => '🍃', 'mushroom' => '🍄',
|
|
240
|
-
'chestnut' => '🌰', 'crescent_moon' => '🌙', 'full_moon' => '🌕', 'new_moon' => '🌑',
|
|
241
|
-
'milky_way' => '🌌', 'stars' => '🌠', 'cyclone' => '🌀', 'fog' => '🌫️',
|
|
242
|
-
'wind_face' => '🌬️', 'tornado' => '🌪️', 'thunder_cloud_and_rain' => '⛈️',
|
|
243
|
-
'sweat_drops' => '💦', 'snowman' => '⛄', 'snowman_with_snow' => '☃️', 'comet' => '☄️',
|
|
244
|
-
|
|
245
|
-
// -- Nourriture (suite) --------------------------------------------
|
|
246
|
-
'tomato' => '🍅', 'eggplant' => '🍆', 'avocado' => '🥑', 'broccoli' => '🥦',
|
|
247
|
-
'carrot' => '🥕', 'corn' => '🌽', 'hot_pepper' => '🌶️', 'cucumber' => '🥒',
|
|
248
|
-
'potato' => '🥔', 'sweet_potato' => '🍠', 'peanuts' => '🥜', 'honey_pot' => '🍯',
|
|
249
|
-
'croissant' => '🥐', 'bagel' => '🥯', 'pretzel' => '🥨', 'pancakes' => '🥞',
|
|
250
|
-
'waffle' => '🧇', 'meat_on_bone' => '🍖', 'poultry_leg' => '🍗', 'bacon' => '🥓',
|
|
251
|
-
'sandwich' => '🥪', 'stuffed_flatbread' => '🥙', 'burrito' => '🌯', 'salad' => '🥗',
|
|
252
|
-
'shallow_pan_of_food' => '🥘', 'canned_food' => '🥫', 'bento' => '🍱',
|
|
253
|
-
'rice_ball' => '🍙', 'rice' => '🍚', 'curry' => '🍛', 'stew' => '🍲', 'oden' => '🍢',
|
|
254
|
-
'dango' => '🍡', 'shaved_ice' => '🍧', 'ice_cream' => '🍨', 'pie' => '🥧',
|
|
255
|
-
'cupcake' => '🧁', 'moon_cake' => '🥮', 'lollipop' => '🍭', 'custard' => '🍮',
|
|
256
|
-
'milk_glass' => '🥛', 'baby_bottle' => '🍼', 'mate' => '🧉', 'ice_cube' => '🧊',
|
|
257
|
-
'tumbler_glass' => '🥃', 'cup_with_straw' => '🥤', 'chopsticks' => '🥢',
|
|
258
|
-
'fork_and_knife' => '🍴', 'spoon' => '🥄', 'plate_with_cutlery' => '🍽️',
|
|
259
|
-
|
|
260
|
-
// -- Activities, sport, leisure --------------------------------------
|
|
261
|
-
'running' => '🏃', 'walking' => '🚶', 'swimming' => '🏊', 'surfing' => '🏄',
|
|
262
|
-
'skateboard' => '🛹', 'snowboarder' => '🏂', 'weight_lifting' => '🏋️',
|
|
263
|
-
'cyclist' => '🚴', 'medal_military' => '🎖️', 'ticket' => '🎫', 'circus_tent' => '🎪',
|
|
264
|
-
'performing_arts' => '🎭', 'art' => '🎨', 'clapper' => '🎬', 'microphone' => '🎤',
|
|
265
|
-
'headphones' => '🎧', 'musical_note' => '🎵', 'musical_score' => '🎼', 'guitar' => '🎸',
|
|
266
|
-
'violin' => '🎻', 'drum' => '🥁', 'trumpet' => '🎺', 'saxophone' => '🎷',
|
|
267
|
-
'musical_keyboard' => '🎹', 'chess_pawn' => '♟️', 'bowling' => '🎳',
|
|
268
|
-
'ice_skate' => '⛸️', 'ski' => '🎿', 'fishing_pole_and_fish' => '🎣',
|
|
269
|
-
'boxing_glove' => '🥊', 'martial_arts_uniform' => '🥋', 'goal_net' => '🥅',
|
|
270
|
-
'flying_disc' => '🥏', 'yo_yo' => '🪀', 'kite' => '🪁',
|
|
271
|
-
|
|
272
|
-
// -- Voyages et lieux (suite) ----------------------------------------
|
|
273
|
-
'airplane_departure' => '🛫', 'airplane_arriving' => '🛬', 'flying_saucer' => '🛸',
|
|
274
|
-
'motorcycle' => '🏍️', 'scooter' => '🛴', 'tractor' => '🚜', 'truck' => '🚚',
|
|
275
|
-
'articulated_lorry' => '🚛', 'trolleybus' => '🚎', 'minibus' => '🚐', 'metro' => '🚇',
|
|
276
|
-
'station' => '🚉', 'monorail' => '🚝', 'bullettrain_front' => '🚄',
|
|
277
|
-
'steam_locomotive' => '🚂', 'anchor' => '⚓', 'sailboat' => '⛵', 'canoe' => '🛶',
|
|
278
|
-
'speedboat' => '🚤', 'ferry' => '⛴️', 'passport_control' => '🛂', 'customs' => '🛃',
|
|
279
|
-
'baggage_claim' => '🛄', 'left_luggage' => '🛅', 'vertical_traffic_light' => '🚦',
|
|
280
|
-
'construction' => '🚧', 'fuelpump' => '⛽', 'busstop' => '🚏', 'moyai' => '🗿',
|
|
281
|
-
'statue_of_liberty' => '🗽', 'tokyo_tower' => '🗼', 'fountain' => '⛲',
|
|
282
|
-
'stadium' => '🏟️', 'ferris_wheel' => '🎡', 'roller_coaster' => '🎢',
|
|
283
|
-
'carousel_horse' => '🎠', 'beach_umbrella' => '🏖️', 'desert' => '🏜️',
|
|
284
|
-
'desert_island' => '🏝️', 'national_park' => '🏞️', 'sunrise' => '🌅',
|
|
285
|
-
'sunrise_over_mountains' => '🌄', 'sparkler' => '🎇', 'fireworks' => '🎆',
|
|
286
|
-
'city_sunset' => '🌇', 'bridge_at_night' => '🌉', 'houses' => '🏘️',
|
|
287
|
-
'derelict_house' => '🏚️', 'classical_building' => '🏛️', 'department_store' => '🏬',
|
|
288
|
-
'post_office' => '🏣', 'hotel' => '🏨', 'convenience_store' => '🏪', 'bank' => '🏦',
|
|
289
|
-
'factory' => '🏭',
|
|
290
|
-
|
|
291
|
-
// -- Objets (suite) ---------------------------------------------------
|
|
292
|
-
'watch' => '⌚', 'stopwatch' => '⏱️', 'timer_clock' => '⏲️', 'joystick' => '🕹️',
|
|
293
|
-
'floppy_disk' => '💾', 'cd' => '💿', 'dvd' => '📀', 'movie_camera' => '🎥',
|
|
294
|
-
'projector' => '📽️', 'telephone' => '☎️', 'pager' => '📟', 'fax' => '📠',
|
|
295
|
-
'candle' => '🕯️', 'fire_extinguisher' => '🧯', 'oil_drum' => '🛢️',
|
|
296
|
-
'money_with_wings' => '💸', 'credit_card' => '💳', 'yen' => '💴', 'euro' => '💶',
|
|
297
|
-
'pound' => '💷', 'briefcase' => '💼', 'balance_scale' => '⚖️', 'compass' => '🧭',
|
|
298
|
-
'triangular_ruler' => '📐', 'straight_ruler' => '📏', 'round_pushpin' => '📍',
|
|
299
|
-
'scissors' => '✂️', 'thread' => '🧵', 'yarn' => '🧶', 'safety_pin' => '🧷',
|
|
300
|
-
'basket' => '🧺', 'hourglass_flowing_sand' => '⏳', 'notebook' => '📓',
|
|
301
|
-
'notebook_with_decorative_cover' => '📔', 'page_facing_up' => '📄',
|
|
302
|
-
'page_with_curl' => '📃', 'bookmark_tabs' => '📑', 'bookmark' => '🔖',
|
|
303
|
-
'label' => '🏷️', 'receipt' => '🧾', 'card_index' => '📇', 'wastebasket' => '🗑️',
|
|
304
|
-
'old_key' => '🗝️', 'hammer_and_wrench' => '🛠️', 'pick' => '⛏️', 'shield' => '🛡️',
|
|
305
|
-
'syringe' => '💉', 'pill' => '💊', 'thermometer' => '🌡️', 'soap' => '🧼',
|
|
306
|
-
'broom' => '🧹',
|
|
307
|
-
|
|
308
|
-
// -- Symboles (suite) -----------------------------------------------
|
|
309
|
-
'heavy_multiplication_x' => '✖️', 'heavy_plus_sign' => '➕', 'heavy_minus_sign' => '➖',
|
|
310
|
-
'heavy_division_sign' => '➗', 'infinity' => '♾️', 'recycle' => '♻️', 'trident' => '🔱',
|
|
311
|
-
'atom_symbol' => '⚛️', 'om' => '🕉️', 'peace_symbol' => '☮️', 'yin_yang' => '☯️',
|
|
312
|
-
'wheel_of_dharma' => '☸️', 'star_of_david' => '✡️', 'star_and_crescent' => '☪️',
|
|
313
|
-
'cross' => '✝️', 'menorah' => '🕎', 'radioactive' => '☢️', 'biohazard' => '☣️',
|
|
314
|
-
'arrow_up' => '⬆️', 'arrow_down' => '⬇️', 'arrow_left' => '⬅️', 'arrow_right' => '➡️',
|
|
315
|
-
'arrow_upper_right' => '↗️', 'arrow_lower_right' => '↘️', 'arrow_lower_left' => '↙️',
|
|
316
|
-
'arrow_upper_left' => '↖️', 'arrows_clockwise' => '🔃', 'arrows_counterclockwise' => '🔄',
|
|
317
|
-
'back' => '🔙', 'end' => '🔚', 'on' => '🔛', 'soon' => '🔜', 'top' => '🔝',
|
|
318
|
-
'radio_button' => '🔘', 'red_circle' => '🔴', 'orange_circle' => '🟠',
|
|
319
|
-
'yellow_circle' => '🟡', 'green_circle' => '🟢', 'blue_circle' => '🔵',
|
|
320
|
-
'purple_circle' => '🟣', 'brown_circle' => '🟤', 'white_circle' => '⚪',
|
|
321
|
-
'black_circle' => '⚫',
|
|
322
|
-
|
|
323
|
-
// -- Drapeaux (suite) -------------------------------------------------
|
|
324
|
-
'triangular_flag_on_post' => '🚩', 'crossed_flags' => '🎌',
|
|
325
|
-
'us' => '🇺🇸', 'gb' => '🇬🇧', 'fr' => '🇫🇷', 'de' => '🇩🇪', 'es' => '🇪🇸',
|
|
326
|
-
'it' => '🇮🇹', 'jp' => '🇯🇵', 'cn' => '🇨🇳', 'kr' => '🇰🇷', 'ca' => '🇨🇦',
|
|
327
|
-
'au' => '🇦🇺', 'br' => '🇧🇷', 'in' => '🇮🇳', 'ru' => '🇷🇺', 'eu' => '🇪🇺',
|
|
328
|
-
];
|
|
329
|
-
|
|
330
|
-
/**
|
|
331
|
-
* Registers (or replaces) a custom emoji shortcut.
|
|
332
|
-
*/
|
|
333
|
-
public static function registerEmoji(string $shortcode, string $char): void {
|
|
334
|
-
self::$extraEmoji[strtolower(trim($shortcode, ':'))] = $char;
|
|
335
|
-
}
|
|
336
|
-
|
|
337
|
-
private static function emojiFor(string $shortcode): ?string {
|
|
338
|
-
$key = strtolower($shortcode);
|
|
339
|
-
return self::$extraEmoji[$key] ?? self::$emojiMap[$key] ?? null;
|
|
340
|
-
}
|
|
341
|
-
|
|
342
|
-
// ========================================================================
|
|
343
|
-
// DEFINITION LISTS (extended syntax)
|
|
344
|
-
// Term
|
|
345
|
-
// : Definition
|
|
346
|
-
// Procedural line-by-line analysis (safer than a single big regex for
|
|
347
|
-
// grouping several term/definition pairs into one <dl>, separated by a
|
|
348
|
-
// blank line or not).
|
|
349
|
-
// ========================================================================
|
|
350
|
-
private static function isDefinitionColonLine(string $line): bool {
|
|
351
|
-
return (bool) preg_match('/^[ \t]*:[ \t]+.+$/', $line);
|
|
352
|
-
}
|
|
353
|
-
|
|
354
|
-
private static function looksLikeOtherBlock(string $line): bool {
|
|
355
|
-
$t = ltrim($line);
|
|
356
|
-
if ($t === '') return true;
|
|
357
|
-
if (str_starts_with($t, '<')) return true;
|
|
358
|
-
return (bool) preg_match('/^(?:#{1,6}[ \t]|>|```|\||[-*+][ \t]|\d+\.[ \t])/', $t);
|
|
359
|
-
}
|
|
360
|
-
|
|
361
|
-
private static function extractDefinitionLists(string $html): string {
|
|
362
|
-
$lines = explode("\n", $html);
|
|
363
|
-
$n = count($lines);
|
|
364
|
-
$out = [];
|
|
365
|
-
$i = 0;
|
|
366
|
-
|
|
367
|
-
while ($i < $n) {
|
|
368
|
-
$isTermStart = $i + 1 < $n
|
|
369
|
-
&& !self::looksLikeOtherBlock($lines[$i])
|
|
370
|
-
&& !self::isDefinitionColonLine($lines[$i])
|
|
371
|
-
&& self::isDefinitionColonLine($lines[$i + 1]);
|
|
372
|
-
|
|
373
|
-
if (!$isTermStart) {
|
|
374
|
-
$out[] = $lines[$i];
|
|
375
|
-
$i++;
|
|
376
|
-
continue;
|
|
377
|
-
}
|
|
378
|
-
|
|
379
|
-
$dl = "<dl>\n";
|
|
380
|
-
while (true) {
|
|
381
|
-
$term = trim($lines[$i]);
|
|
382
|
-
$dl .= " <dt>{$term}</dt>\n";
|
|
383
|
-
$i++;
|
|
384
|
-
while ($i < $n && preg_match('/^[ \t]*:[ \t]+(.*)$/', $lines[$i], $m)) {
|
|
385
|
-
$dl .= " <dd>{$m[1]}</dd>\n";
|
|
386
|
-
$i++;
|
|
387
|
-
}
|
|
388
|
-
|
|
389
|
-
// A single blank line between two groups stays in the same <dl>
|
|
390
|
-
// if the next group really is a new term.
|
|
391
|
-
if ($i < $n && trim($lines[$i]) === '') {
|
|
392
|
-
$j = $i;
|
|
393
|
-
while ($j < $n && trim($lines[$j]) === '') $j++;
|
|
394
|
-
if ($j + 1 < $n
|
|
395
|
-
&& !self::looksLikeOtherBlock($lines[$j])
|
|
396
|
-
&& !self::isDefinitionColonLine($lines[$j])
|
|
397
|
-
&& self::isDefinitionColonLine($lines[$j + 1])
|
|
398
|
-
) {
|
|
399
|
-
$i = $j;
|
|
400
|
-
continue;
|
|
401
|
-
}
|
|
402
|
-
}
|
|
403
|
-
break;
|
|
404
|
-
}
|
|
405
|
-
$dl .= "</dl>";
|
|
406
|
-
$out[] = $dl;
|
|
407
|
-
}
|
|
408
|
-
|
|
409
|
-
return implode("\n", $out);
|
|
410
|
-
}
|
|
411
|
-
|
|
412
|
-
// ========================================================================
|
|
413
|
-
// RAW HTML (safe subset, GitHub README style)
|
|
414
|
-
//
|
|
415
|
-
// Writing HTML tags directly (e.g. <div align="center">, <img>, <sub>,
|
|
416
|
-
// <br>, HTML tables...) is allowed ONLY if:
|
|
417
|
-
// - the tag is part of the HTML_ALLOWED_TAGS whitelist;
|
|
418
|
-
// - every attribute is part of the whitelist for that tag (or of the
|
|
419
|
-
// global HTML_GLOBAL_ATTRS attributes);
|
|
420
|
-
// - no attribute starts with "on" (onclick, onerror, ...);
|
|
421
|
-
// - URLs (href/src) use a safe scheme (isSafeUrl).
|
|
422
|
-
//
|
|
423
|
-
// Any unknown or dangerous tag (script/style/iframe/...), or any
|
|
424
|
-
// non-whitelisted attribute, is silently stripped. The text content
|
|
425
|
-
// between the tags is NOT swallowed: it keeps being processed as normal
|
|
426
|
-
// markdown (that's what lets you have markdown headings, badges and images
|
|
427
|
-
// inside a <div align="center">...</div>).
|
|
428
|
-
// ========================================================================
|
|
429
|
-
|
|
430
|
-
private const HTML_ALLOWED_TAGS = [
|
|
431
|
-
'div', 'span', 'p', 'br', 'hr', 'wbr',
|
|
432
|
-
'b', 'strong', 'i', 'em', 'u', 's', 'strike', 'del', 'ins',
|
|
433
|
-
'mark', 'small', 'sub', 'sup', 'kbd', 'code', 'pre', 'abbr', 'q', 'cite',
|
|
434
|
-
'ul', 'ol', 'li', 'dl', 'dt', 'dd',
|
|
435
|
-
'table', 'thead', 'tbody', 'tfoot', 'tr', 'td', 'th', 'caption', 'colgroup', 'col',
|
|
436
|
-
'blockquote',
|
|
437
|
-
'a', 'img', 'picture', 'source', 'figure', 'figcaption',
|
|
438
|
-
'h1', 'h2', 'h3', 'h4', 'h5', 'h6',
|
|
439
|
-
'details', 'summary', 'center',
|
|
440
|
-
];
|
|
441
|
-
|
|
442
|
-
/** Attributes allowed on any whitelisted tag. */
|
|
443
|
-
private const HTML_GLOBAL_ATTRS = ['id', 'class', 'title', 'align', 'valign', 'width', 'height', 'dir', 'lang'];
|
|
444
|
-
|
|
445
|
-
/** Extra attributes allowed, per tag. */
|
|
446
|
-
private const HTML_TAG_ATTRS = [
|
|
447
|
-
'a' => ['href', 'name', 'target', 'rel'],
|
|
448
|
-
'img' => ['src', 'alt', 'loading', 'srcset', 'sizes'],
|
|
449
|
-
'source' => ['src', 'srcset', 'type', 'media'],
|
|
450
|
-
'td' => ['colspan', 'rowspan'],
|
|
451
|
-
'th' => ['colspan', 'rowspan', 'scope'],
|
|
452
|
-
'col' => ['span'],
|
|
453
|
-
'ol' => ['start', 'type'],
|
|
454
|
-
'details' => ['open'],
|
|
455
|
-
];
|
|
456
|
-
|
|
457
|
-
/** Self-closing tags (no closing tag expected). */
|
|
458
|
-
private const HTML_VOID_TAGS = ['img', 'br', 'hr', 'wbr', 'source', 'col'];
|
|
459
|
-
|
|
460
|
-
/**
|
|
461
|
-
* Checks that a URL (href/src) uses a safe scheme: relative links, anchors,
|
|
462
|
-
* http(s), mailto, tel, or base64-encoded images (png/gif/jpeg/webp only —
|
|
463
|
-
* not svg+xml, which can embed a <script>).
|
|
464
|
-
* Notably rejects javascript:, vbscript:, data:text/html.
|
|
465
|
-
*/
|
|
466
|
-
private static function isSafeUrl(string $url): bool {
|
|
467
|
-
$url = trim($url);
|
|
468
|
-
if ($url === '') return true;
|
|
469
|
-
// A path with no explicit scheme ("assets/x.png", "../x", "#anchor",
|
|
470
|
-
// "/x", "x") is a relative link or an anchor: always safe.
|
|
471
|
-
if (!preg_match('~^([a-zA-Z][a-zA-Z0-9+.\-]*):~', $url, $m)) return true;
|
|
472
|
-
$scheme = strtolower($m[1]);
|
|
473
|
-
if (in_array($scheme, ['http', 'https', 'mailto', 'tel'], true)) return true;
|
|
474
|
-
if ($scheme === 'data') {
|
|
475
|
-
// Base64-encoded images only — no data:image/svg+xml, which can
|
|
476
|
-
// embed a <script>, and no data:text/html.
|
|
477
|
-
return (bool) preg_match('~^data:image/(png|gif|jpe?g|webp);base64,~i', $url);
|
|
478
|
-
}
|
|
479
|
-
return false; // javascript:, vbscript:, file:, etc. → rejected
|
|
480
|
-
}
|
|
481
|
-
|
|
482
|
-
/**
|
|
483
|
-
* Sanitizes a single raw HTML tag (e.g. '<div align="center">', '</div>',
|
|
484
|
-
* '<img src="..." onerror="...">').
|
|
485
|
-
*
|
|
486
|
-
* @return string|null The cleaned tag to keep, an empty string to strip it
|
|
487
|
-
* silently, or null if it doesn't look like a valid
|
|
488
|
-
* HTML tag (in which case the caller strips it too, to
|
|
489
|
-
* be safe).
|
|
490
|
-
*/
|
|
491
|
-
private static function sanitizeHtmlTag(string $tag): ?string {
|
|
492
|
-
if (!preg_match(
|
|
493
|
-
'/^<(\/)?([a-zA-Z][a-zA-Z0-9-]*)((?:\s+[a-zA-Z_:][a-zA-Z0-9_:.-]*(?:\s*=\s*(?:"[^"]*"|\'[^\']*\'|[^\s"\'>]+))?)*)\s*(\/)?>$/s',
|
|
494
|
-
$tag,
|
|
495
|
-
$m
|
|
496
|
-
)) {
|
|
497
|
-
return null;
|
|
498
|
-
}
|
|
499
|
-
|
|
500
|
-
$closing = $m[1] === '/';
|
|
501
|
-
$tagName = strtolower($m[2]);
|
|
502
|
-
$attrsRaw = $m[3];
|
|
503
|
-
|
|
504
|
-
if (!in_array($tagName, self::HTML_ALLOWED_TAGS, true)) {
|
|
505
|
-
return null;
|
|
506
|
-
}
|
|
507
|
-
|
|
508
|
-
if ($closing) {
|
|
509
|
-
return "</{$tagName}>";
|
|
510
|
-
}
|
|
511
|
-
|
|
512
|
-
$allowedAttrs = array_merge(self::HTML_GLOBAL_ATTRS, self::HTML_TAG_ATTRS[$tagName] ?? []);
|
|
513
|
-
$safeAttrs = '';
|
|
514
|
-
|
|
515
|
-
if (preg_match_all(
|
|
516
|
-
'/([a-zA-Z_:][a-zA-Z0-9_:.-]*)(?:\s*=\s*("([^"]*)"|\'([^\']*)\'|([^\s"\'>]+)))?/',
|
|
517
|
-
$attrsRaw,
|
|
518
|
-
$am,
|
|
519
|
-
PREG_SET_ORDER
|
|
520
|
-
)) {
|
|
521
|
-
foreach ($am as $a) {
|
|
522
|
-
$attrName = strtolower($a[1]);
|
|
523
|
-
if ($attrName === '') continue;
|
|
524
|
-
if (str_starts_with($attrName, 'on')) continue; // safety net against JS handlers
|
|
525
|
-
if (!in_array($attrName, $allowedAttrs, true)) continue;
|
|
526
|
-
|
|
527
|
-
if ($tagName === 'details' && $attrName === 'open') {
|
|
528
|
-
$safeAttrs .= ' open';
|
|
529
|
-
continue;
|
|
530
|
-
}
|
|
531
|
-
|
|
532
|
-
$attrVal = $a[3] ?? ($a[4] ?? ($a[5] ?? ''));
|
|
533
|
-
|
|
534
|
-
if (in_array($attrName, ['href', 'src'], true) && !self::isSafeUrl($attrVal)) {
|
|
535
|
-
continue;
|
|
536
|
-
}
|
|
537
|
-
|
|
538
|
-
$safeAttrs .= ' ' . $attrName . '="' . htmlspecialchars($attrVal, ENT_QUOTES, 'UTF-8') . '"';
|
|
539
|
-
}
|
|
540
|
-
}
|
|
541
|
-
|
|
542
|
-
$close = in_array($tagName, self::HTML_VOID_TAGS, true) ? ' /' : '';
|
|
543
|
-
return "<{$tagName}{$safeAttrs}{$close}>";
|
|
544
|
-
}
|
|
545
|
-
|
|
546
|
-
// ========================================================================
|
|
547
|
-
|
|
548
|
-
public static function toHtml(string $markdown): string {
|
|
549
|
-
|
|
550
|
-
// ====================================================================
|
|
551
|
-
// STEP 1: Line-ending normalization
|
|
552
|
-
// ====================================================================
|
|
553
|
-
$html = str_replace(["\r\n", "\r"], "\n", $markdown);
|
|
554
|
-
|
|
555
|
-
|
|
556
|
-
// ====================================================================
|
|
557
|
-
// STEP 2: PLUGINS
|
|
558
|
-
// Two forms supported:
|
|
559
|
-
//
|
|
560
|
-
// INLINE: {% name arg1 "arg 2" %}
|
|
561
|
-
// → $args = ['arg1', 'arg 2'], $body = ''
|
|
562
|
-
//
|
|
563
|
-
// BLOCK : {% name arg1\ncontent\nover\nseveral lines\n%}
|
|
564
|
-
// → $args = ['arg1'], $body = "content\nover\nseveral lines"
|
|
565
|
-
//
|
|
566
|
-
// Both are captured by a single regex that tells apart the presence of
|
|
567
|
-
// a newline after the args (block) or not (inline).
|
|
568
|
-
// Processed before XSS encoding — re-injected as the very last step.
|
|
569
|
-
// ====================================================================
|
|
570
|
-
$pluginBlocks = [];
|
|
571
|
-
// Literal, XSS-escaped source of each captured tag, keyed by the same
|
|
572
|
-
// placeholder. Used to restore a `{% tag %}` that turns out to sit
|
|
573
|
-
// inside a code span / code block as verbatim text instead of expanding it.
|
|
574
|
-
$pluginLiterals = [];
|
|
575
|
-
|
|
576
|
-
/**
|
|
577
|
-
* Parses an argument string into an array.
|
|
578
|
-
* Supports bare words, "double quotes" and 'single quotes'.
|
|
579
|
-
*/
|
|
580
|
-
$parseArgs = static function (string $rawArgs): array {
|
|
581
|
-
$args = [];
|
|
582
|
-
if (trim($rawArgs) === '') return $args;
|
|
583
|
-
preg_match_all(
|
|
584
|
-
'/"([^"\\\\]*(?:\\\\.[^"\\\\]*)*)"|\'([^\'\\\\]*(?:\\\\.[^\'\\\\]*)*)\'|(\S+)/',
|
|
585
|
-
$rawArgs,
|
|
586
|
-
$m
|
|
587
|
-
);
|
|
588
|
-
foreach ($m[0] as $i => $_) {
|
|
589
|
-
$args[] = $m[1][$i] !== ''
|
|
590
|
-
? stripslashes($m[1][$i])
|
|
591
|
-
: ($m[2][$i] !== ''
|
|
592
|
-
? stripslashes($m[2][$i])
|
|
593
|
-
: $m[3][$i]);
|
|
594
|
-
}
|
|
595
|
-
return $args;
|
|
596
|
-
};
|
|
597
|
-
|
|
598
|
-
$html = preg_replace_callback(
|
|
599
|
-
// Group 1: plugin name
|
|
600
|
-
// Group 2: inline args (everything on the first line after the name)
|
|
601
|
-
// Group 3: multi-line body (present only for block tags)
|
|
602
|
-
'/\{%\s*([a-zA-Z0-9_-]+)([^\n%]*?)(?:\n([\s\S]*?))?\s*%\}/m',
|
|
603
|
-
function ($matches) use (&$pluginBlocks, &$pluginLiterals, $parseArgs): string {
|
|
604
|
-
$name = strtolower(trim($matches[1]));
|
|
605
|
-
$args = $parseArgs(trim($matches[2] ?? ''));
|
|
606
|
-
// $matches[3] exists only if the tag is multi-line
|
|
607
|
-
$body = isset($matches[3]) ? trim($matches[3]) : '';
|
|
608
|
-
|
|
609
|
-
if (!isset(self::$plugins[$name])) {
|
|
610
|
-
// Unknown plugin: kept encoded rather than silently removed
|
|
611
|
-
return htmlspecialchars($matches[0], ENT_QUOTES, 'UTF-8');
|
|
612
|
-
}
|
|
613
|
-
|
|
614
|
-
$output = (self::$plugins[$name])($args, $body);
|
|
615
|
-
$placeholder = "\x02PLG" . count($pluginBlocks) . "\x03";
|
|
616
|
-
$pluginBlocks[$placeholder] = $output;
|
|
617
|
-
$pluginLiterals[$placeholder] = htmlspecialchars($matches[0], ENT_QUOTES, 'UTF-8');
|
|
618
|
-
return $placeholder;
|
|
619
|
-
},
|
|
620
|
-
$html
|
|
621
|
-
);
|
|
622
|
-
|
|
623
|
-
|
|
624
|
-
// ====================================================================
|
|
625
|
-
// STEP 2a: FOOTNOTE DEFINITIONS
|
|
626
|
-
// [^1]: Note text.
|
|
627
|
-
// [^bignote]: First line.
|
|
628
|
-
//
|
|
629
|
-
// Following paragraph, indented by 4 spaces or 1 tab.
|
|
630
|
-
//
|
|
631
|
-
// `{ some code }`
|
|
632
|
-
// Extracted (and removed from the text) BEFORE reference link
|
|
633
|
-
// definitions, since [^label]: would otherwise match their regex too.
|
|
634
|
-
// Each note's content is rendered via a recursive call to toHtml() to
|
|
635
|
-
// support multiple paragraphs, code, etc.
|
|
636
|
-
// ====================================================================
|
|
637
|
-
$footnoteDefs = [];
|
|
638
|
-
$html = preg_replace_callback(
|
|
639
|
-
'/^\[\^([^\]\s]+)\]:[ \t]?([^\n]*)((?:\n(?:[ \t]{4}[^\n]*|[ \t]*))*)/m',
|
|
640
|
-
function ($m) use (&$footnoteDefs): string {
|
|
641
|
-
$label = strtolower(trim($m[1]));
|
|
642
|
-
$first = $m[2];
|
|
643
|
-
$rest = $m[3] ?? '';
|
|
644
|
-
$restLines = $rest !== '' ? explode("\n", $rest) : [];
|
|
645
|
-
$restLines = array_map(static function (string $l): string {
|
|
646
|
-
return preg_replace('/^(?:[ ]{4}|\t)/', '', $l);
|
|
647
|
-
}, $restLines);
|
|
648
|
-
$content = trim($first . "\n" . implode("\n", $restLines));
|
|
649
|
-
$footnoteDefs[$label] = self::toHtml($content);
|
|
650
|
-
return '';
|
|
651
|
-
},
|
|
652
|
-
$html
|
|
653
|
-
);
|
|
654
|
-
|
|
655
|
-
|
|
656
|
-
// ====================================================================
|
|
657
|
-
// STEP 2b: REFERENCE LINK DEFINITIONS
|
|
658
|
-
// [label]: https://example.com "Optional title"
|
|
659
|
-
// [label]: <https://example.com> 'Optional title'
|
|
660
|
-
// [label]: https://example.com (Optional title)
|
|
661
|
-
// Extracted (and removed from the text) before everything else; used
|
|
662
|
-
// later by the [text][label] / [text][] links.
|
|
663
|
-
// ====================================================================
|
|
664
|
-
$refDefs = [];
|
|
665
|
-
$html = preg_replace_callback(
|
|
666
|
-
'/^[ \t]{0,3}\[([^\]]+)\]:[ \t]*<?([^\s>]+)>?(?:[ \t]+(?:"([^"]*)"|\'([^\']*)\'|\(([^)]*)\)))?[ \t]*$/m',
|
|
667
|
-
function ($m) use (&$refDefs): string {
|
|
668
|
-
$label = strtolower(trim($m[1]));
|
|
669
|
-
$title = $m[3] !== '' ? $m[3] : ($m[4] !== '' ? $m[4] : ($m[5] ?? ''));
|
|
670
|
-
$refDefs[$label] = ['url' => $m[2], 'title' => $title];
|
|
671
|
-
return '';
|
|
672
|
-
},
|
|
673
|
-
$html
|
|
674
|
-
);
|
|
675
|
-
|
|
676
|
-
|
|
677
|
-
// ====================================================================
|
|
678
|
-
// STEP 3: CODE BLOCKS (```lang ... ```)
|
|
679
|
-
// ====================================================================
|
|
680
|
-
$codeBlocks = [];
|
|
681
|
-
$html = preg_replace_callback('/^```([a-zA-Z0-9_+-]*)\n([\s\S]*?)\n^```/m', function ($matches) use (&$codeBlocks) {
|
|
682
|
-
$lang = !empty($matches[1]) ? ' class="language-' . htmlspecialchars($matches[1], ENT_QUOTES, 'UTF-8') . '"' : '';
|
|
683
|
-
$code = htmlspecialchars($matches[2], ENT_QUOTES, 'UTF-8');
|
|
684
|
-
$placeholder = "\x02CB" . count($codeBlocks) . "\x03";
|
|
685
|
-
$codeBlocks[$placeholder] = "<pre><code{$lang}>{$code}</code></pre>";
|
|
686
|
-
return $placeholder;
|
|
687
|
-
}, $html);
|
|
688
|
-
|
|
689
|
-
// ====================================================================
|
|
690
|
-
// STEP 3a: INDENTED CODE BLOCKS (4 spaces or 1 tab)
|
|
691
|
-
// Recognized only when preceded by a blank line (or the start of the
|
|
692
|
-
// document) and followed by a blank line (or the end of the document),
|
|
693
|
-
// to avoid conflicts with the indentation of nested lists.
|
|
694
|
-
// ====================================================================
|
|
695
|
-
$html = preg_replace_callback(
|
|
696
|
-
'/(?<=\n\n|^)((?:[ ]{4}|\t)[^\n]*(?:\n(?:[ ]{4}|\t)[^\n]*)*)(?=\n\n|\n*$)/',
|
|
697
|
-
function ($matches) use (&$codeBlocks) {
|
|
698
|
-
$lines = explode("\n", $matches[1]);
|
|
699
|
-
$stripped = array_map(static function (string $l): string {
|
|
700
|
-
return preg_replace('/^(?:[ ]{4}|\t)/', '', $l);
|
|
701
|
-
}, $lines);
|
|
702
|
-
$code = htmlspecialchars(implode("\n", $stripped), ENT_QUOTES, 'UTF-8');
|
|
703
|
-
$placeholder = "\x02CB" . count($codeBlocks) . "\x03";
|
|
704
|
-
$codeBlocks[$placeholder] = "<pre><code>{$code}</code></pre>";
|
|
705
|
-
return $placeholder;
|
|
706
|
-
},
|
|
707
|
-
$html
|
|
708
|
-
);
|
|
709
|
-
|
|
710
|
-
// Inline code with double backticks (lets you include a literal backtick)
|
|
711
|
-
$inlineCodes = [];
|
|
712
|
-
$html = preg_replace_callback('/``(.+?)``/s', function ($matches) use (&$inlineCodes) {
|
|
713
|
-
$content = $matches[1];
|
|
714
|
-
// Standard convention: if the content starts and ends with a space
|
|
715
|
-
// (and isn't only spaces), strip one space on each side — handy for
|
|
716
|
-
// wrapping a ` at the edge.
|
|
717
|
-
if (preg_match('/^ (.*[^ ]) $/s', $content, $trim)) {
|
|
718
|
-
$content = $trim[1];
|
|
719
|
-
}
|
|
720
|
-
$code = htmlspecialchars($content, ENT_QUOTES, 'UTF-8');
|
|
721
|
-
$placeholder = "\x02IC" . count($inlineCodes) . "\x03";
|
|
722
|
-
$inlineCodes[$placeholder] = "<code>{$code}</code>";
|
|
723
|
-
return $placeholder;
|
|
724
|
-
}, $html);
|
|
725
|
-
|
|
726
|
-
// Inline code (`...`)
|
|
727
|
-
$html = preg_replace_callback('/`([^`\n]+)`/', function ($matches) use (&$inlineCodes) {
|
|
728
|
-
$code = htmlspecialchars($matches[1], ENT_QUOTES, 'UTF-8');
|
|
729
|
-
$placeholder = "\x02IC" . count($inlineCodes) . "\x03";
|
|
730
|
-
$inlineCodes[$placeholder] = "<code>{$code}</code>";
|
|
731
|
-
return $placeholder;
|
|
732
|
-
}, $html);
|
|
733
|
-
|
|
734
|
-
// A `{% tag %}` sitting inside a code span or code block was captured
|
|
735
|
-
// by STEP 2 and is now a plugin placeholder embedded in the stored
|
|
736
|
-
// code. Swap those back for the literal (escaped) tag source so code
|
|
737
|
-
// shows `{% tag %}` verbatim instead of its rendered output — or a
|
|
738
|
-
// stray control-char placeholder that never gets restored.
|
|
739
|
-
if ($pluginLiterals) {
|
|
740
|
-
foreach ($codeBlocks as $k => $v) $codeBlocks[$k] = strtr($v, $pluginLiterals);
|
|
741
|
-
foreach ($inlineCodes as $k => $v) $inlineCodes[$k] = strtr($v, $pluginLiterals);
|
|
742
|
-
}
|
|
743
|
-
|
|
744
|
-
|
|
745
|
-
// ====================================================================
|
|
746
|
-
// STEP 3b: CHARACTER ESCAPING (\* \_ \# etc.)
|
|
747
|
-
// Processed after code extraction (code stays literal) and before
|
|
748
|
-
// everything else, so that \* doesn't open emphasis, \# doesn't create
|
|
749
|
-
// a heading, \- doesn't create a list, etc.
|
|
750
|
-
// ====================================================================
|
|
751
|
-
$escapes = [];
|
|
752
|
-
$html = preg_replace_callback(
|
|
753
|
-
'/\\\\([\\\\`*_{}\[\]<>()#+\-.!|])/',
|
|
754
|
-
function ($m) use (&$escapes): string {
|
|
755
|
-
$placeholder = "\x02ESC" . count($escapes) . "\x03";
|
|
756
|
-
$escapes[$placeholder] = htmlspecialchars($m[1], ENT_QUOTES, 'UTF-8');
|
|
757
|
-
return $placeholder;
|
|
758
|
-
},
|
|
759
|
-
$html
|
|
760
|
-
);
|
|
761
|
-
// | is the documented convention (Markdown Extra / PHP Markdown)
|
|
762
|
-
// for showing a literal pipe in a table cell without it being
|
|
763
|
-
// interpreted as a column separator.
|
|
764
|
-
$html = preg_replace_callback(
|
|
765
|
-
'/|/i',
|
|
766
|
-
function () use (&$escapes): string {
|
|
767
|
-
$placeholder = "\x02ESC" . count($escapes) . "\x03";
|
|
768
|
-
$escapes[$placeholder] = '|';
|
|
769
|
-
return $placeholder;
|
|
770
|
-
},
|
|
771
|
-
$html
|
|
772
|
-
);
|
|
773
|
-
|
|
774
|
-
|
|
775
|
-
// ====================================================================
|
|
776
|
-
// STEP 3d: AUTOMATIC LINKS <https://...> and <email@example.com>
|
|
777
|
-
// Processed before XSS encoding because the < > characters would be
|
|
778
|
-
// encoded to < > and the regex would no longer match.
|
|
779
|
-
// ====================================================================
|
|
780
|
-
$autolinks = [];
|
|
781
|
-
$html = preg_replace_callback('/<(https?:\/\/[^\s<>]+)>/', function ($m) use (&$autolinks): string {
|
|
782
|
-
$url = htmlspecialchars($m[1], ENT_QUOTES, 'UTF-8');
|
|
783
|
-
$placeholder = "\x02AL" . count($autolinks) . "\x03";
|
|
784
|
-
$autolinks[$placeholder] = "<a href=\"{$url}\" target=\"_blank\" rel=\"noopener noreferrer\">{$url}</a>";
|
|
785
|
-
return $placeholder;
|
|
786
|
-
}, $html);
|
|
787
|
-
$html = preg_replace_callback('/<([^\s<>]+@[^\s<>]+\.[^\s<>]+)>/', function ($m) use (&$autolinks): string {
|
|
788
|
-
$email = htmlspecialchars($m[1], ENT_QUOTES, 'UTF-8');
|
|
789
|
-
$placeholder = "\x02AL" . count($autolinks) . "\x03";
|
|
790
|
-
$autolinks[$placeholder] = "<a href=\"mailto:{$email}\">{$email}</a>";
|
|
791
|
-
return $placeholder;
|
|
792
|
-
}, $html);
|
|
793
|
-
|
|
794
|
-
|
|
795
|
-
// ====================================================================
|
|
796
|
-
// STEP 3e: GFM ALERTS AND BLOCKQUOTES
|
|
797
|
-
// Processed before XSS encoding because the > character would be
|
|
798
|
-
// encoded to > and the regexes would no longer match.
|
|
799
|
-
// ====================================================================
|
|
800
|
-
$blockquotes = [];
|
|
801
|
-
|
|
802
|
-
// GFM alerts (> [!NOTE], etc.) — more specific, processed first
|
|
803
|
-
$html = preg_replace_callback(
|
|
804
|
-
'/^(>\s*\[!(NOTE|TIP|IMPORTANT|WARNING|CAUTION)\]\n(?:>[ \t]?[^\n]*\n?)*)/m',
|
|
805
|
-
function ($matches) use (&$blockquotes): string {
|
|
806
|
-
$type = strtolower($matches[2]);
|
|
807
|
-
$label = htmlspecialchars($matches[2], ENT_QUOTES, 'UTF-8');
|
|
808
|
-
$content = preg_replace('/^>\s?\[!(?:NOTE|TIP|IMPORTANT|WARNING|CAUTION)\]\n?/m', '', $matches[1]);
|
|
809
|
-
$content = preg_replace('/^>[ \t]?/m', '', $content);
|
|
810
|
-
$content = self::toHtml(trim($content));
|
|
811
|
-
$placeholder = "\x02BQ" . count($blockquotes) . "\x03";
|
|
812
|
-
$blockquotes[$placeholder] = "<div class=\"markdown-alert markdown-alert-{$type}\">"
|
|
813
|
-
. "<p class=\"markdown-alert-title\">{$label}</p>"
|
|
814
|
-
. "{$content}</div>";
|
|
815
|
-
// The trailing \n consumed by the regex is re-injected after
|
|
816
|
-
// the placeholder so the following blank line doesn't merge
|
|
817
|
-
// with the placeholder's line (which would break, for example,
|
|
818
|
-
// detecting a Setext heading right after).
|
|
819
|
-
return $placeholder . (str_ends_with($matches[1], "\n") ? "\n" : '');
|
|
820
|
-
},
|
|
821
|
-
$html
|
|
822
|
-
);
|
|
823
|
-
|
|
824
|
-
// Standard blockquotes (nesting handled by recursion through toHtml,
|
|
825
|
-
// which re-applies this same rule to the content already stripped of
|
|
826
|
-
// one ">" level)
|
|
827
|
-
$html = preg_replace_callback('/^((?:>[ \t]?[^\n]*\n?)+)/m', function ($matches) use (&$blockquotes): string {
|
|
828
|
-
$content = preg_replace('/^>[ \t]?/m', '', $matches[1]);
|
|
829
|
-
// The two trailing spaces are left as-is: toHtml() handles them itself
|
|
830
|
-
$inner = self::toHtml(trim($content));
|
|
831
|
-
$placeholder = "\x02BQ" . count($blockquotes) . "\x03";
|
|
832
|
-
$blockquotes[$placeholder] = "<blockquote>{$inner}</blockquote>";
|
|
833
|
-
// See the comment above: preserve the trailing \n that was consumed.
|
|
834
|
-
return $placeholder . (str_ends_with($matches[1], "\n") ? "\n" : '');
|
|
835
|
-
}, $html);
|
|
836
|
-
|
|
837
|
-
|
|
838
|
-
// ====================================================================
|
|
839
|
-
// STEP 3f: RAW HTML (safe subset, GitHub README style)
|
|
840
|
-
// Processed before XSS encoding because the < > characters would be
|
|
841
|
-
// encoded to < > and no longer recognized as tags.
|
|
842
|
-
// The content between the tags is not swallowed: it stays in the
|
|
843
|
-
// stream and keeps being processed as normal markdown.
|
|
844
|
-
// ====================================================================
|
|
845
|
-
|
|
846
|
-
// Intrinsically dangerous elements: removed along with their content
|
|
847
|
-
// (script/style/iframe can embed JS or load a third-party page;
|
|
848
|
-
// form/button/textarea/select/option have no place in markdown
|
|
849
|
-
// content).
|
|
850
|
-
$html = preg_replace(
|
|
851
|
-
'/<(script|style|iframe|object|embed|noscript|template|form|button|textarea|select|option)\b[^>]*>[\s\S]*?<\/\1>/i',
|
|
852
|
-
'',
|
|
853
|
-
$html
|
|
854
|
-
);
|
|
855
|
-
|
|
856
|
-
$rawHtml = [];
|
|
857
|
-
$html = preg_replace_callback(
|
|
858
|
-
'/<!--[\s\S]*?-->|<\/?[a-zA-Z][a-zA-Z0-9-]*(?:\s+[a-zA-Z_:][a-zA-Z0-9_:.-]*(?:\s*=\s*(?:"[^"]*"|\'[^\']*\'|[^\s"\'>]+))?)*\s*\/?>/',
|
|
859
|
-
function ($m) use (&$rawHtml): string {
|
|
860
|
-
$tag = $m[0];
|
|
861
|
-
// HTML comment: invisible, safe to remove.
|
|
862
|
-
if (str_starts_with($tag, '<!--')) return '';
|
|
863
|
-
|
|
864
|
-
$sanitized = self::sanitizeHtmlTag($tag);
|
|
865
|
-
if ($sanitized === null || $sanitized === '') return '';
|
|
866
|
-
|
|
867
|
-
$placeholder = "\x02HT" . count($rawHtml) . "\x03";
|
|
868
|
-
$rawHtml[$placeholder] = $sanitized;
|
|
869
|
-
return $placeholder;
|
|
870
|
-
},
|
|
871
|
-
$html
|
|
872
|
-
);
|
|
873
|
-
|
|
874
|
-
|
|
875
|
-
// ====================================================================
|
|
876
|
-
// STEP 4: Global XSS encoding
|
|
877
|
-
// ====================================================================
|
|
878
|
-
$html = htmlspecialchars($html, ENT_NOQUOTES, 'UTF-8');
|
|
879
|
-
|
|
880
|
-
|
|
881
|
-
// ====================================================================
|
|
882
|
-
// STEP 5: GFM TABLES
|
|
883
|
-
// Supports rows with or without a trailing pipe (| col | or | col)
|
|
884
|
-
// ====================================================================
|
|
885
|
-
$html = preg_replace_callback(
|
|
886
|
-
'/^(\|[^\n]+\|?\n)([ \t]*\|[ \t]*:?-+:?[ \t]*(?:\|[ \t]*:?-+:?[ \t]*)*\|?\n)((?:\|[^\n]+\|?\n?)+)/m',
|
|
887
|
-
function ($matches) {
|
|
888
|
-
$parseRow = function (string $line): array {
|
|
889
|
-
return array_values(array_filter(
|
|
890
|
-
array_map('trim', explode('|', trim($line, "| \t\n")))
|
|
891
|
-
));
|
|
892
|
-
};
|
|
893
|
-
|
|
894
|
-
$headers = $parseRow($matches[1]);
|
|
895
|
-
$alignments = [];
|
|
896
|
-
$sepCells = $parseRow($matches[2]);
|
|
897
|
-
foreach ($sepCells as $sep) {
|
|
898
|
-
$left = str_starts_with(trim($sep), ':');
|
|
899
|
-
$right = str_ends_with(trim($sep), ':');
|
|
900
|
-
if ($left && $right) $alignments[] = ' style="text-align:center"';
|
|
901
|
-
elseif ($right) $alignments[] = ' style="text-align:right"';
|
|
902
|
-
elseif ($left) $alignments[] = ' style="text-align:left"';
|
|
903
|
-
else $alignments[] = '';
|
|
904
|
-
}
|
|
905
|
-
|
|
906
|
-
$out = "<table>\n <thead>\n <tr>\n";
|
|
907
|
-
foreach ($headers as $i => $header) {
|
|
908
|
-
$align = $alignments[$i] ?? '';
|
|
909
|
-
$out .= " <th{$align}>{$header}</th>\n";
|
|
910
|
-
}
|
|
911
|
-
$out .= " </tr>\n </thead>\n <tbody>\n";
|
|
912
|
-
|
|
913
|
-
$bodyLines = array_filter(explode("\n", trim($matches[3])));
|
|
914
|
-
foreach ($bodyLines as $line) {
|
|
915
|
-
$cells = $parseRow($line);
|
|
916
|
-
$out .= " <tr>\n";
|
|
917
|
-
foreach ($cells as $i => $cell) {
|
|
918
|
-
$align = $alignments[$i] ?? '';
|
|
919
|
-
$out .= " <td{$align}>{$cell}</td>\n";
|
|
920
|
-
}
|
|
921
|
-
$out .= " </tr>\n";
|
|
922
|
-
}
|
|
923
|
-
$out .= " </tbody>\n</table>";
|
|
924
|
-
return $out;
|
|
925
|
-
},
|
|
926
|
-
$html
|
|
927
|
-
);
|
|
928
|
-
|
|
929
|
-
|
|
930
|
-
// ====================================================================
|
|
931
|
-
// STEP 6: (GFM alerts and blockquotes handled in step 3e)
|
|
932
|
-
// ====================================================================
|
|
933
|
-
|
|
934
|
-
|
|
935
|
-
// ====================================================================
|
|
936
|
-
// STEP 7: TASK LISTS (GFM checkboxes)
|
|
937
|
-
// ====================================================================
|
|
938
|
-
$html = preg_replace('/^[ \t]*[-*+] \[ \] (.+)$/m', '<li class="task-item"><input type="checkbox" disabled /> $1</li>', $html);
|
|
939
|
-
$html = preg_replace('/^[ \t]*[-*+] \[[xX]\] (.+)$/m', '<li class="task-item"><input type="checkbox" checked disabled /> $1</li>', $html);
|
|
940
|
-
|
|
941
|
-
|
|
942
|
-
// ====================================================================
|
|
943
|
-
// STEP 7b: SETEXT HEADINGS (alternative == / -- syntax)
|
|
944
|
-
// Title
|
|
945
|
-
// ===== → <h1>
|
|
946
|
-
//
|
|
947
|
-
// Title
|
|
948
|
-
// ----- → <h2>
|
|
949
|
-
// Processed before ATX headings and before horizontal rules (a line of
|
|
950
|
-
// dashes right after a line of text is a heading, not an <hr>).
|
|
951
|
-
// ====================================================================
|
|
952
|
-
$html = preg_replace_callback(
|
|
953
|
-
'/^(?![ \t]*(?:#{1,6}[ \t]|>|```|\||[-*+][ \t]|\d+\.[ \t]))[ \t]*(\S.*?)[ \t]*(?:\{#([a-zA-Z0-9_\-:.]+)\}[ \t]*)?\n[ \t]*=+[ \t]*$/m',
|
|
954
|
-
function ($matches) {
|
|
955
|
-
$text = trim($matches[1]);
|
|
956
|
-
$id = !empty($matches[2]) ? $matches[2] : self::slugify($text);
|
|
957
|
-
return "<h1 id=\"{$id}\">{$text}</h1>";
|
|
958
|
-
},
|
|
959
|
-
$html
|
|
960
|
-
);
|
|
961
|
-
$html = preg_replace_callback(
|
|
962
|
-
'/^(?![ \t]*(?:#{1,6}[ \t]|>|```|\||[-*+][ \t]|\d+\.[ \t]))[ \t]*(\S.*?)[ \t]*(?:\{#([a-zA-Z0-9_\-:.]+)\}[ \t]*)?\n[ \t]*-+[ \t]*$/m',
|
|
963
|
-
function ($matches) {
|
|
964
|
-
$text = trim($matches[1]);
|
|
965
|
-
$id = !empty($matches[2]) ? $matches[2] : self::slugify($text);
|
|
966
|
-
return "<h2 id=\"{$id}\">{$text}</h2>";
|
|
967
|
-
},
|
|
968
|
-
$html
|
|
969
|
-
);
|
|
970
|
-
|
|
971
|
-
|
|
972
|
-
// ====================================================================
|
|
973
|
-
// STEP 8: HEADINGS (ATX: # to ######)
|
|
974
|
-
// ====================================================================
|
|
975
|
-
$html = preg_replace_callback(
|
|
976
|
-
'/^(#{1,6})[ \t]+(.+?)[ \t]*(?:\{#([a-zA-Z0-9_\-:.]+)\}[ \t]*)?(?:[ \t]+#+)?$/m',
|
|
977
|
-
function ($matches) {
|
|
978
|
-
$level = strlen($matches[1]);
|
|
979
|
-
$text = trim($matches[2]);
|
|
980
|
-
$id = !empty($matches[3]) ? $matches[3] : self::slugify($text);
|
|
981
|
-
return "<h{$level} id=\"{$id}\">{$text}</h{$level}>";
|
|
982
|
-
},
|
|
983
|
-
$html
|
|
984
|
-
);
|
|
985
|
-
|
|
986
|
-
|
|
987
|
-
// ====================================================================
|
|
988
|
-
// STEP 9: LISTS (bullets and ordered, with nesting)
|
|
989
|
-
// A single pass detects a contiguous block of lines that are either a
|
|
990
|
-
// bullet (-,*,+) or a numbered item, whatever their indentation level;
|
|
991
|
-
// the block is then rebuilt recursively into nested <ol>/<ul>
|
|
992
|
-
// according to the relative indentation depth. An indented line with
|
|
993
|
-
// no marker of its own is a *continuation* of the previous item's
|
|
994
|
-
// text (a soft-wrapped source line) and is appended to it, not
|
|
995
|
-
// treated as the block's end — a marker-less continuation line used
|
|
996
|
-
// to fall outside the match entirely, splitting one list into two
|
|
997
|
-
// with the continuation text stranded as a stray <p> in between.
|
|
998
|
-
// Task items (already converted to <li class="task-item">) no longer
|
|
999
|
-
// match this pattern and are therefore not re-wrapped here.
|
|
1000
|
-
// ====================================================================
|
|
1001
|
-
$html = preg_replace_callback(
|
|
1002
|
-
'/^([ \t]*(?:\d+\.|[-*+])[ \t]+.+(?:\n(?:[ \t]*(?:\d+\.|[-*+])[ \t]+.+|[ \t]+\S.*))*)/m',
|
|
1003
|
-
function ($matches) {
|
|
1004
|
-
$lines = explode("\n", $matches[1]);
|
|
1005
|
-
$items = [];
|
|
1006
|
-
foreach ($lines as $line) {
|
|
1007
|
-
if (preg_match('/^([ \t]*)(\d+)\.[ \t]+(.*)$/', $line, $m)) {
|
|
1008
|
-
$items[] = ['indent' => self::indentWidth($m[1]), 'type' => 'ol', 'text' => $m[3]];
|
|
1009
|
-
} elseif (preg_match('/^([ \t]*)[-*+][ \t]+(.*)$/', $line, $m)) {
|
|
1010
|
-
$items[] = ['indent' => self::indentWidth($m[1]), 'type' => 'ul', 'text' => $m[2]];
|
|
1011
|
-
} elseif (!empty($items) && preg_match('/^[ \t]+(\S.*)$/', $line, $m)) {
|
|
1012
|
-
$items[count($items) - 1]['text'] .= ' ' . $m[1];
|
|
1013
|
-
}
|
|
1014
|
-
}
|
|
1015
|
-
if (empty($items)) return $matches[1];
|
|
1016
|
-
// Normalize the lowest indentation level to 0
|
|
1017
|
-
$minIndent = min(array_column($items, 'indent'));
|
|
1018
|
-
foreach ($items as &$it) $it['indent'] -= $minIndent;
|
|
1019
|
-
unset($it);
|
|
1020
|
-
|
|
1021
|
-
$i = 0;
|
|
1022
|
-
return self::buildListTree($items, $i, count($items));
|
|
1023
|
-
},
|
|
1024
|
-
$html
|
|
1025
|
-
);
|
|
1026
|
-
|
|
1027
|
-
$html = preg_replace_callback(
|
|
1028
|
-
'/(?:<li class="task-item">.*<\/li>\n?)+/s',
|
|
1029
|
-
function ($matches) {
|
|
1030
|
-
return "<ul class=\"task-list\">\n" . $matches[0] . "</ul>\n";
|
|
1031
|
-
},
|
|
1032
|
-
$html
|
|
1033
|
-
);
|
|
1034
|
-
|
|
1035
|
-
|
|
1036
|
-
// ====================================================================
|
|
1037
|
-
// STEP 9b: DEFINITION LISTS (extended syntax)
|
|
1038
|
-
// Term
|
|
1039
|
-
// : Definition
|
|
1040
|
-
// ====================================================================
|
|
1041
|
-
$html = self::extractDefinitionLists($html);
|
|
1042
|
-
|
|
1043
|
-
|
|
1044
|
-
// ====================================================================
|
|
1045
|
-
// STEP 9c: FOOTNOTE REFERENCES [^label]
|
|
1046
|
-
// Converted BEFORE emphasis so they don't collide with the new
|
|
1047
|
-
// superscript ^text^ (a [^1] followed later by a [^2] on the same line
|
|
1048
|
-
// could otherwise be read as ^1] ... [^2^).
|
|
1049
|
-
// Numbering is sequential, in order of first appearance in the text
|
|
1050
|
-
// (as documented).
|
|
1051
|
-
// ====================================================================
|
|
1052
|
-
$footnoteOrder = [];
|
|
1053
|
-
$html = preg_replace_callback('/\[\^([^\]\s]+)\]/', function ($m) use (&$footnoteOrder, &$footnoteDefs): string {
|
|
1054
|
-
$label = strtolower(trim($m[1]));
|
|
1055
|
-
if (!isset($footnoteDefs[$label])) {
|
|
1056
|
-
// Reference to an undefined note: left as-is.
|
|
1057
|
-
return $m[0];
|
|
1058
|
-
}
|
|
1059
|
-
if (!isset($footnoteOrder[$label])) {
|
|
1060
|
-
$footnoteOrder[$label] = count($footnoteOrder) + 1;
|
|
1061
|
-
}
|
|
1062
|
-
$num = $footnoteOrder[$label];
|
|
1063
|
-
return "<sup id=\"fnref:{$label}\"><a href=\"#fn:{$label}\">{$num}</a></sup>";
|
|
1064
|
-
}, $html);
|
|
1065
|
-
|
|
1066
|
-
|
|
1067
|
-
// ====================================================================
|
|
1068
|
-
// STEP 10: INLINE TEXT (Bold, Italic, Strikethrough, Highlight,
|
|
1069
|
-
// Subscript/Superscript, Emoji)
|
|
1070
|
-
// ====================================================================
|
|
1071
|
-
$html = preg_replace('/\*\*\*(.+?)\*\*\*/s', '<strong><em>$1</em></strong>', $html);
|
|
1072
|
-
$html = preg_replace('/___(.+?)___/s', '<strong><em>$1</em></strong>', $html);
|
|
1073
|
-
$html = preg_replace('/\*\*(.+?)\*\*/s', '<strong>$1</strong>', $html);
|
|
1074
|
-
$html = preg_replace('/__(.+?)__/s', '<strong>$1</strong>', $html);
|
|
1075
|
-
$html = preg_replace('/\*(.+?)\*/s', '<em>$1</em>', $html);
|
|
1076
|
-
// Italic _ must only match at word boundaries so it doesn't capture
|
|
1077
|
-
// snake_case, package names (@php-wasm/node), etc.
|
|
1078
|
-
$html = preg_replace('/(?<!\w)_([^_\n]+)_(?!\w)/', '<em>$1</em>', $html);
|
|
1079
|
-
// Highlight ==text== (extended syntax)
|
|
1080
|
-
$html = preg_replace('/==(.+?)==/s', '<mark>$1</mark>', $html);
|
|
1081
|
-
// Strikethrough ~~text~~ — processed BEFORE subscript (single ~) so the
|
|
1082
|
-
// latter doesn't match half of a double-tilde pair.
|
|
1083
|
-
$html = preg_replace('/~~(.+?)~~/s', '<del>$1</del>', $html);
|
|
1084
|
-
// Superscript ^text^ (extended syntax) — placing it before the note
|
|
1085
|
-
// reference escaping ([^label]) is not a problem: those are wrapped in
|
|
1086
|
-
// brackets and so don't form an isolated ^...^ pair.
|
|
1087
|
-
$html = preg_replace('/\^([^\^\n]+)\^/', '<sup>$1</sup>', $html);
|
|
1088
|
-
// Subscript ~text~ (a single tilde; the ~~ were already consumed just
|
|
1089
|
-
// above by strikethrough).
|
|
1090
|
-
$html = preg_replace('/~([^~\n]+)~/', '<sub>$1</sub>', $html);
|
|
1091
|
-
|
|
1092
|
-
// Emojis :shortcode: (extended syntax) — unknown shortcuts are left
|
|
1093
|
-
// as-is rather than silently removed.
|
|
1094
|
-
$html = preg_replace_callback('/:([a-zA-Z0-9_+\-]+):/', function ($m): string {
|
|
1095
|
-
$emoji = self::emojiFor($m[1]);
|
|
1096
|
-
return $emoji ?? $m[0];
|
|
1097
|
-
}, $html);
|
|
1098
|
-
|
|
1099
|
-
|
|
1100
|
-
// ====================================================================
|
|
1101
|
-
// STEP 11: LINKS & IMAGES
|
|
1102
|
-
// External links (https?://) get target="_blank" + rel="noopener noreferrer".
|
|
1103
|
-
// Internal links (/page, #anchor, ../thing) don't.
|
|
1104
|
-
// ====================================================================
|
|
1105
|
-
$html = preg_replace(
|
|
1106
|
-
'/!\[([^\]]*)\]\(([^)\s]+)(?:\s+"([^"]*)")?\)/',
|
|
1107
|
-
'<img src="$2" alt="$1" title="$3" loading="lazy" />',
|
|
1108
|
-
$html
|
|
1109
|
-
);
|
|
1110
|
-
|
|
1111
|
-
$buildLink = static function (string $text, string $href, string $title): string {
|
|
1112
|
-
$titleAttr = $title !== '' ? ' title="' . $title . '"' : '';
|
|
1113
|
-
$extern = preg_match('/^https?:\/\//i', $href)
|
|
1114
|
-
? ' target="_blank" rel="noopener noreferrer"'
|
|
1115
|
-
: '';
|
|
1116
|
-
return "<a href=\"{$href}\"{$titleAttr}{$extern}>{$text}</a>";
|
|
1117
|
-
};
|
|
1118
|
-
|
|
1119
|
-
// Reference links [text][label] and [text][] (shortcut = label = text)
|
|
1120
|
-
$html = preg_replace_callback(
|
|
1121
|
-
'/\[([^\]]+)\]\[([^\]]*)\]/',
|
|
1122
|
-
function ($m) use (&$refDefs, $buildLink): string {
|
|
1123
|
-
$text = $m[1];
|
|
1124
|
-
$label = strtolower(trim($m[2] !== '' ? $m[2] : $m[1]));
|
|
1125
|
-
if (!isset($refDefs[$label])) return $m[0];
|
|
1126
|
-
$def = $refDefs[$label];
|
|
1127
|
-
return $buildLink($text, $def['url'], $def['title']);
|
|
1128
|
-
},
|
|
1129
|
-
$html
|
|
1130
|
-
);
|
|
1131
|
-
|
|
1132
|
-
// Markdown links [text](url "optional title")
|
|
1133
|
-
$html = preg_replace_callback(
|
|
1134
|
-
'/\[([^\]]+)\]\(([^)\s]+)(?:\s+"([^"]*)")?\)/',
|
|
1135
|
-
function ($m) use ($buildLink): string {
|
|
1136
|
-
return $buildLink($m[1], $m[2], $m[3] ?? '');
|
|
1137
|
-
},
|
|
1138
|
-
$html
|
|
1139
|
-
);
|
|
1140
|
-
|
|
1141
|
-
// Bare URLs https://... (extended syntax: auto-link without brackets).
|
|
1142
|
-
// Excludes those already inside quotes/attributes (href="...") or
|
|
1143
|
-
// already turned into a link, so they don't get doubled.
|
|
1144
|
-
$html = preg_replace(
|
|
1145
|
-
'/(?<!["\'=>])\b(https?:\/\/[^\s<>"\')\]]+)/',
|
|
1146
|
-
'<a href="$1" target="_blank" rel="noopener noreferrer">$1</a>',
|
|
1147
|
-
$html
|
|
1148
|
-
);
|
|
1149
|
-
|
|
1150
|
-
|
|
1151
|
-
// ====================================================================
|
|
1152
|
-
// STEP 12: HORIZONTAL RULES
|
|
1153
|
-
// ====================================================================
|
|
1154
|
-
$html = preg_replace('/^(?:[-*_][ \t]*){3,}$/m', '<hr />', $html);
|
|
1155
|
-
|
|
1156
|
-
|
|
1157
|
-
// ====================================================================
|
|
1158
|
-
// STEP 13: PARAGRAPHS
|
|
1159
|
-
// Strategy: process line by line. Lines that start with a block-level
|
|
1160
|
-
// tag or a placeholder are left as-is. Consecutive raw-text lines are
|
|
1161
|
-
// accumulated then wrapped in a <p> when a block line or a blank line
|
|
1162
|
-
// is reached.
|
|
1163
|
-
// ====================================================================
|
|
1164
|
-
$blockStartTags = ['<h', '<pre', '<ul', '<ol', '<li', '<table', '<thead', '<tbody',
|
|
1165
|
-
'<tr', '<td', '<th', '<blockquote', '<div', '<hr', '<img',
|
|
1166
|
-
'<dl', '<dt', '<dd',
|
|
1167
|
-
"\x02CB", "\x02PLG", "\x02BQ", "\x02HT"];
|
|
1168
|
-
|
|
1169
|
-
$isBlockLine = static function (string $line) use ($blockStartTags): bool {
|
|
1170
|
-
$t = ltrim($line);
|
|
1171
|
-
if ($t === '') return false;
|
|
1172
|
-
// Any closing tag (</...>) is always treated as a "block" line:
|
|
1173
|
-
// this keeps a closing </table>, </thead>, </tr>, etc. from being
|
|
1174
|
-
// absorbed into a surrounding <p>.
|
|
1175
|
-
if (str_starts_with($t, '</')) return true;
|
|
1176
|
-
foreach ($blockStartTags as $tag) {
|
|
1177
|
-
if (str_starts_with($t, $tag)) return true;
|
|
1178
|
-
}
|
|
1179
|
-
return false;
|
|
1180
|
-
};
|
|
1181
|
-
|
|
1182
|
-
$lines = explode("\n", $html);
|
|
1183
|
-
$output = [];
|
|
1184
|
-
$textBuffer = [];
|
|
1185
|
-
|
|
1186
|
-
$flushBuffer = static function () use (&$textBuffer, &$output): void {
|
|
1187
|
-
if (empty($textBuffer)) return;
|
|
1188
|
-
$content = implode("\n", $textBuffer);
|
|
1189
|
-
if (trim($content) !== '') {
|
|
1190
|
-
// Two trailing spaces → <br> (standard markdown convention)
|
|
1191
|
-
$content = preg_replace('/ $/m', '<br>', $content);
|
|
1192
|
-
// Single line break → space (GitHub behavior)
|
|
1193
|
-
// Unless already converted to <br> above
|
|
1194
|
-
$content = preg_replace('/(?<!r>)\n/', ' ', $content);
|
|
1195
|
-
$output[] = '<p>' . trim($content) . '</p>';
|
|
1196
|
-
}
|
|
1197
|
-
$textBuffer = [];
|
|
1198
|
-
};
|
|
1199
|
-
|
|
1200
|
-
foreach ($lines as $line) {
|
|
1201
|
-
if ($isBlockLine($line)) {
|
|
1202
|
-
$flushBuffer();
|
|
1203
|
-
$output[] = $line;
|
|
1204
|
-
} elseif (trim($line) === '') {
|
|
1205
|
-
// Blank line = paragraph separator
|
|
1206
|
-
$flushBuffer();
|
|
1207
|
-
} else {
|
|
1208
|
-
$textBuffer[] = $line;
|
|
1209
|
-
}
|
|
1210
|
-
}
|
|
1211
|
-
$flushBuffer();
|
|
1212
|
-
|
|
1213
|
-
$html = implode("\n", $output);
|
|
1214
|
-
|
|
1215
|
-
|
|
1216
|
-
// ====================================================================
|
|
1217
|
-
// STEP 14: Re-inject the placeholders
|
|
1218
|
-
// ====================================================================
|
|
1219
|
-
$html = strtr($html, $pluginBlocks);
|
|
1220
|
-
$html = strtr($html, $blockquotes);
|
|
1221
|
-
$html = strtr($html, $rawHtml);
|
|
1222
|
-
$html = strtr($html, $codeBlocks);
|
|
1223
|
-
$html = strtr($html, $inlineCodes);
|
|
1224
|
-
$html = strtr($html, $autolinks);
|
|
1225
|
-
// The escapes are re-injected last, once no Markdown regex can
|
|
1226
|
-
// interpret them anymore.
|
|
1227
|
-
$html = strtr($html, $escapes);
|
|
1228
|
-
|
|
1229
|
-
|
|
1230
|
-
// ====================================================================
|
|
1231
|
-
// STEP 15: FOOTNOTES BLOCK
|
|
1232
|
-
// Appended at the end of the document, only if at least one note was
|
|
1233
|
-
// referenced (notes that are defined but never referenced are
|
|
1234
|
-
// silently ignored).
|
|
1235
|
-
// ====================================================================
|
|
1236
|
-
if (!empty($footnoteOrder)) {
|
|
1237
|
-
$html .= "\n<div class=\"footnotes\">\n<ol>\n";
|
|
1238
|
-
foreach ($footnoteOrder as $label => $num) {
|
|
1239
|
-
$content = $footnoteDefs[$label];
|
|
1240
|
-
$html .= " <li id=\"fn:{$label}\">{$content} <a href=\"#fnref:{$label}\" class=\"footnote-backref\">↩</a></li>\n";
|
|
1241
|
-
}
|
|
1242
|
-
$html .= "</ol>\n</div>";
|
|
1243
|
-
}
|
|
1244
|
-
|
|
1245
|
-
return $html;
|
|
3
|
+
/**
|
|
4
|
+
* MD — Markdown renderer backed by the native `mdhtml` extension (CommonMark
|
|
5
|
+
* + GFM via cmark-gfm 0.29.0.gfm.13, statically built into @kirigami/php-wasm
|
|
6
|
+
* — see php-kirigami/php-mdhtml).
|
|
7
|
+
*
|
|
8
|
+
* Switched from the hand-written recursive-descent/regex renderer to this
|
|
9
|
+
* native extension — see docs/DECISIONS.md and docs/STATUS.md in the repo
|
|
10
|
+
* root for the full reasoning. Every MD::toHtml() extension (plugins, emoji,
|
|
11
|
+
* GFM alerts, definition lists, heading anchors, ==highlight==/^sup^/~sub~)
|
|
12
|
+
* is reimplemented in C, not just the CommonMark/GFM core — see php-mdhtml's
|
|
13
|
+
* README for exactly what's covered.
|
|
14
|
+
*
|
|
15
|
+
* Kept the old implementation as MD_LEGACY (md-legacy.class.php) — untouched,
|
|
16
|
+
* still autoloadable — as a rollback path.
|
|
17
|
+
*
|
|
18
|
+
* Usage:
|
|
19
|
+
* $html = MD::toHtml($markdown);
|
|
20
|
+
* MD::registerPlugin('name', function (array $args, string $body): string { ... });
|
|
21
|
+
* MD::registerEmoji('kirigami', '📐');
|
|
22
|
+
*/
|
|
23
|
+
class MD
|
|
24
|
+
{
|
|
25
|
+
public static function registerPlugin(string $name, callable $callback): void
|
|
26
|
+
{
|
|
27
|
+
\MDHtml\RegisterPlugin($name, $callback);
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
public static function unregisterPlugin(string $name): void
|
|
31
|
+
{
|
|
32
|
+
\MDHtml\UnregisterPlugin($name);
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
public static function getRegisteredPlugins(): array
|
|
36
|
+
{
|
|
37
|
+
return \MDHtml\GetRegisteredPlugins();
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
public static function registerEmoji(string $shortcode, string $char): void
|
|
41
|
+
{
|
|
42
|
+
\MDHtml\RegisterEmoji($shortcode, $char);
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
public static function toHtml(string $markdown): string
|
|
46
|
+
{
|
|
47
|
+
return \MDHtml\Render($markdown);
|
|
1246
48
|
}
|
|
1247
49
|
}
|
|
1248
50
|
|
|
1249
51
|
|
|
1250
|
-
// Default Markdown plugins ({% codepen %}, {%
|
|
1251
|
-
// {%
|
|
1252
|
-
// every entrypoint (prepros.php, runenv.php, imagebatch.php) gets them
|
|
1253
|
-
// an explicit include. A project can still MD::unregisterPlugin() any
|
|
1254
|
-
// or MD::registerPlugin() its own with the same name to override.
|
|
1255
|
-
include_once(__DIR__ . '/md.plugins.php');
|
|
52
|
+
// Default Markdown plugins ({% codepen %}, {% checklist %}, {% callout %},
|
|
53
|
+
// {% img-asset %}) — "registered out of the box" per the README. Loaded here
|
|
54
|
+
// so every entrypoint (prepros.php, runenv.php, imagebatch.php) gets them
|
|
55
|
+
// without an explicit include. A project can still MD::unregisterPlugin() any
|
|
56
|
+
// of them, or MD::registerPlugin() its own with the same name to override.
|
|
57
|
+
include_once(__DIR__ . '/md.plugins.php');
|