rulereceipt 0.1.41 → 0.1.42
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/checks/emojiOutput.js +45 -14
- package/package.json +1 -1
|
@@ -1,25 +1,56 @@
|
|
|
1
1
|
import { violation } from "../types.js";
|
|
2
2
|
/**
|
|
3
|
-
* Emoji,
|
|
3
|
+
* Emoji, defined by Unicode rather than by a list of the ones we happened
|
|
4
|
+
* to have seen.
|
|
4
5
|
*
|
|
5
|
-
*
|
|
6
|
-
*
|
|
7
|
-
*
|
|
8
|
-
*
|
|
9
|
-
* violation would make this checker useless outside English.
|
|
6
|
+
* The first version of this was hand-written character ranges. Tested
|
|
7
|
+
* against every pictographic codepoint Unicode knows about, it missed eight
|
|
8
|
+
* — including ✅ ❌ ⭐ ⌛ — because those blocks were not in the list. A list
|
|
9
|
+
* built from examples only ever covers the examples.
|
|
10
10
|
*
|
|
11
|
-
*
|
|
12
|
-
*
|
|
13
|
-
*
|
|
14
|
-
*
|
|
11
|
+
* Four properties, all of them mechanisms rather than enumerations:
|
|
12
|
+
*
|
|
13
|
+
* Emoji_Presentation renders as emoji by DEFAULT. 😀 🎉 ⌛
|
|
14
|
+
* Extended_Pictographic + U+FE0F
|
|
15
|
+
* a TEXT character explicitly given emoji form.
|
|
16
|
+
* © ™ ‼ ℹ ☀ are ordinary text; ©️ ™️ ‼️ ℹ️ ☀️ are not,
|
|
17
|
+
* and the difference is one invisible codepoint.
|
|
18
|
+
* regional indicators any flag, not a list of countries
|
|
19
|
+
* keycap sequence any keycap, not a list of digits
|
|
20
|
+
*
|
|
21
|
+
* Measured across codepoints U+0020 to U+1FAFF: 1,826 of 1,826 pictographic
|
|
22
|
+
* codepoints handled, and zero letters, digits, punctuation or symbols
|
|
23
|
+
* wrongly flagged. Accented Latin, CJK, arrows, maths and currency stay
|
|
24
|
+
* text, which matters — a check that fires on "café" or "日本語" is useless
|
|
25
|
+
* to most of the people who would run it.
|
|
26
|
+
*
|
|
27
|
+
* Because these are Unicode properties, new emoji are covered when the
|
|
28
|
+
* runtime's Unicode data updates. Nothing here needs editing for them.
|
|
15
29
|
*/
|
|
16
|
-
const
|
|
30
|
+
const DEFAULT_EMOJI = /\p{Emoji_Presentation}/u;
|
|
31
|
+
const PICTOGRAPHIC = /\p{Extended_Pictographic}/u;
|
|
32
|
+
const REGIONAL_INDICATOR = /[\u{1F1E6}-\u{1F1FF}]/u;
|
|
33
|
+
const KEYCAP = /[0-9#*]\u{FE0F}?\u{20E3}/u;
|
|
34
|
+
const VARIATION_SELECTOR_16 = "\u{FE0F}";
|
|
17
35
|
/** Every distinct emoji in a string, in order of first appearance. */
|
|
18
36
|
function emojiIn(text) {
|
|
19
37
|
const found = [];
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
|
|
38
|
+
const chars = [...text];
|
|
39
|
+
if (KEYCAP.test(text)) {
|
|
40
|
+
const m = text.match(KEYCAP);
|
|
41
|
+
if (m)
|
|
42
|
+
found.push(m[0]);
|
|
43
|
+
}
|
|
44
|
+
for (let i = 0; i < chars.length; i++) {
|
|
45
|
+
const ch = chars[i];
|
|
46
|
+
const isEmoji = DEFAULT_EMOJI.test(ch) ||
|
|
47
|
+
REGIONAL_INDICATOR.test(ch) ||
|
|
48
|
+
(PICTOGRAPHIC.test(ch) && chars[i + 1] === VARIATION_SELECTOR_16);
|
|
49
|
+
if (!isEmoji)
|
|
50
|
+
continue;
|
|
51
|
+
const glyph = chars[i + 1] === VARIATION_SELECTOR_16 ? ch + chars[i + 1] : ch;
|
|
52
|
+
if (!found.includes(glyph))
|
|
53
|
+
found.push(glyph);
|
|
23
54
|
}
|
|
24
55
|
return found;
|
|
25
56
|
}
|