extract-webpage 1.2.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +212 -0
- package/dist/config/env.d.ts +6 -0
- package/dist/config/index.d.ts +23 -0
- package/dist/config/serverRegistry.d.ts +7 -0
- package/dist/config/types.d.ts +4 -0
- package/dist/extract-webpage.cjs.js +2 -0
- package/dist/extract-webpage.cjs.js.map +1 -0
- package/dist/extract-webpage.es.js +5 -0
- package/dist/extract-webpage.es.js.map +1 -0
- package/dist/html-to-cite/extract-author.d.ts +11 -0
- package/dist/html-to-cite/extract-cite.d.ts +33 -0
- package/dist/html-to-cite/extract-date/date-extractors.d.ts +40 -0
- package/dist/html-to-cite/extract-date/date-validators.d.ts +15 -0
- package/dist/html-to-cite/extract-date/extract-date-quick.d.ts +8 -0
- package/dist/html-to-cite/extract-date/extract-date.d.ts +26 -0
- package/dist/html-to-cite/extract-source.d.ts +7 -0
- package/dist/html-to-cite/extract-title.d.ts +11 -0
- package/dist/html-to-cite/human-names-recognize.d.ts +16 -0
- package/dist/html-to-cite/metadata-to-cite.d.ts +12 -0
- package/dist/html-to-cite/url-to-domain.d.ts +20 -0
- package/dist/html-to-content/extract-content/extract-content-mercury-utils.d.ts +27 -0
- package/dist/html-to-content/extract-content/extract-content-mercury.d.ts +61 -0
- package/dist/html-to-content/extract-content/extract-content-readability.d.ts +101 -0
- package/dist/html-to-content/html-to-basic-html.d.ts +36 -0
- package/dist/html-to-content/html-to-content.d.ts +51 -0
- package/dist/html-to-content/html-utils.d.ts +76 -0
- package/dist/index.d.ts +26 -0
- package/dist/search/index.d.ts +14 -0
- package/dist/search/meta-search-agent-reexport.d.ts +8 -0
- package/dist/search/public-searxng.d.ts +47 -0
- package/dist/search/search-web.d.ts +33 -0
- package/dist/search/tavily.d.ts +20 -0
- package/dist/search/url-to-html.d.ts +62 -0
- package/dist/seektopic/fold-keyphrases.d.ts +28 -0
- package/dist/seektopic/ngrams.d.ts +27 -0
- package/dist/seektopic/rank-sentences-keyphrases.d.ts +28 -0
- package/dist/seektopic/seektopic-keyphrases.d.ts +53 -0
- package/dist/seektopic/types.d.ts +86 -0
- package/dist/seektopic/vector-search.d.ts +89 -0
- package/dist/seektopic/weight-keyphrases.d.ts +22 -0
- package/dist/suggest-next-words/autocomplete-ai.d.ts +0 -0
- package/dist/suggest-next-words/autocomplete-search-engines.d.ts +64 -0
- package/dist/tokenize/suggest-complete-word.d.ts +48 -0
- package/dist/tokenize/text-to-chunks.d.ts +48 -0
- package/dist/tokenize/text-to-sentences.d.ts +35 -0
- package/dist/tokenize/text-to-topic-tokens.d.ts +51 -0
- package/dist/tokenize/word-is-ignored.d.ts +12 -0
- package/dist/tokenize/word-to-root-stem.d.ts +16 -0
- package/dist/url-to-content/docx-to-content.d.ts +22 -0
- package/dist/url-to-content/is-url-adult.d.ts +26 -0
- package/dist/url-to-content/url-to-content.d.ts +127 -0
- package/dist/url-to-content/url-to-html.d.ts +60 -0
- package/dist/url-to-content/youtube-helpers.d.ts +23 -0
- package/dist/url-to-content/youtube-to-text.d.ts +70 -0
- package/dist/utils/documents.d.ts +4 -0
- package/dist/utils/grab.d.ts +18 -0
- package/package.json +109 -0
- package/src/config/env.ts +8 -0
- package/src/config/index.ts +233 -0
- package/src/config/serverRegistry.ts +24 -0
- package/src/config/types.ts +17 -0
- package/src/fs-mock.js +22 -0
- package/src/global.d.ts +8 -0
- package/src/html-to-cite/extract-author.ts +125 -0
- package/src/html-to-cite/extract-cite.ts +97 -0
- package/src/html-to-cite/extract-date/date-extractors.ts +484 -0
- package/src/html-to-cite/extract-date/date-validators.ts +191 -0
- package/src/html-to-cite/extract-date/extract-date-quick.ts +184 -0
- package/src/html-to-cite/extract-date/extract-date.ts +1049 -0
- package/src/html-to-cite/extract-source.ts +30 -0
- package/src/html-to-cite/extract-title.ts +78 -0
- package/src/html-to-cite/human-names-92k.json +1 -0
- package/src/html-to-cite/human-names-recognize.ts +396 -0
- package/src/html-to-cite/metadata-to-cite.ts +73 -0
- package/src/html-to-cite/url-to-domain.ts +50 -0
- package/src/html-to-content/extract-content/extract-content-mercury-utils.ts +696 -0
- package/src/html-to-content/extract-content/extract-content-mercury.ts +830 -0
- package/src/html-to-content/extract-content/extract-content-readability.ts +432 -0
- package/src/html-to-content/extract-content/extract-selectors-per-domain.json +3453 -0
- package/src/html-to-content/html-to-basic-html.ts +282 -0
- package/src/html-to-content/html-to-content.ts +97 -0
- package/src/html-to-content/html-utils.ts +398 -0
- package/src/index.ts +29 -0
- package/src/search/__tests__/public-searxng.test.ts +529 -0
- package/src/search/index.ts +43 -0
- package/src/search/meta-search-agent-reexport.ts +38 -0
- package/src/search/public-searxng.ts +470 -0
- package/src/search/search-web.ts +668 -0
- package/src/search/tavily.ts +106 -0
- package/src/search/url-to-html.ts +278 -0
- package/src/seektopic/fold-keyphrases.ts +87 -0
- package/src/seektopic/ngrams.ts +64 -0
- package/src/seektopic/rank-sentences-keyphrases.ts +132 -0
- package/src/seektopic/seektopic-keyphrases.ts +279 -0
- package/src/seektopic/types.ts +92 -0
- package/src/seektopic/vector-search.ts +232 -0
- package/src/seektopic/weight-keyphrases.ts +59 -0
- package/src/suggest-next-words/autocomplete-ai.ts +38 -0
- package/src/suggest-next-words/autocomplete-search-engines.ts +435 -0
- package/src/tokenize/suggest-complete-word.ts +137 -0
- package/src/tokenize/text-to-chunks.ts +150 -0
- package/src/tokenize/text-to-sentences.ts +614 -0
- package/src/tokenize/text-to-topic-tokens.ts +175 -0
- package/src/tokenize/word-is-ignored.ts +53 -0
- package/src/tokenize/word-to-root-stem.ts +151 -0
- package/src/types.d.ts +130 -0
- package/src/url-to-content/.fuse_hidden003bd28a0000000d +332 -0
- package/src/url-to-content/__tests__/url-to-content.test.ts +368 -0
- package/src/url-to-content/__tests__/url-to-html.test.ts +301 -0
- package/src/url-to-content/docx-to-content.ts +702 -0
- package/src/url-to-content/is-url-adult.ts +318 -0
- package/src/url-to-content/url-to-content.ts +367 -0
- package/src/url-to-content/url-to-html.ts +436 -0
- package/src/url-to-content/youtube-helpers.ts +64 -0
- package/src/url-to-content/youtube-to-text.ts +468 -0
- package/src/utils/documents.ts +71 -0
- package/src/utils/grab.ts +51 -0
|
@@ -0,0 +1,150 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @fileoverview Utility for splitting text into semantic chunks for RAG and NLP tasks.
|
|
3
|
+
* Uses complex regex patterns to identify structural elements like lists, tables, and code.
|
|
4
|
+
*
|
|
5
|
+
* ### Split Text by Semantic Characters
|
|
6
|
+
* <img width="350px" src="https://i.imgur.com/RpXf5as.png" />
|
|
7
|
+
*
|
|
8
|
+
*
|
|
9
|
+
* Splits document text into semantic chunks based on various textual and structural
|
|
10
|
+
* elements like HTML, markdown, and paragraphs.
|
|
11
|
+
*
|
|
12
|
+
* This function performs a comprehensive tokenization of the input text, considering a wide range
|
|
13
|
+
* of semantic elements and structural patterns commonly found in documents.It uses regular
|
|
14
|
+
* expressions to identify and separate the following elements:
|
|
15
|
+
*
|
|
16
|
+
* 1. Headings(Setext - style, Markdown, and HTML - style)
|
|
17
|
+
* 2. Citations(e.g., [1])
|
|
18
|
+
* 3. List items(bulleted, numbered, lettered, or task lists, including nested up to three levels)
|
|
19
|
+
* 4. Block quotes(including nested quotes and citations, up to three levels)
|
|
20
|
+
* 5. Code blocks(fenced, indented, or HTML pre / code tags)
|
|
21
|
+
* 6. Tables(Markdown, grid tables, and HTML tables)
|
|
22
|
+
* 7. Horizontal rules(Markdown and HTML hr tag)
|
|
23
|
+
* 8. Standalone lines or phrases(including single - line blocks and HTML elements)
|
|
24
|
+
* 9. Sentences or phrases ending with punctuation(including ellipsis and Unicode punctuation)
|
|
25
|
+
* 10. Quoted text, parenthetical phrases, or bracketed content
|
|
26
|
+
* 11. Paragraphs
|
|
27
|
+
* 12. HTML - like tags and their content(including self - closing tags and attributes)
|
|
28
|
+
* 13. LaTeX - style math expressions(inline and block)
|
|
29
|
+
* 14. Any remaining content(fallback)
|
|
30
|
+
*
|
|
31
|
+
* The function applies various length constraints to each type of element to ensure reasonable
|
|
32
|
+
* chunk sizes.It also handles nested structures and special cases like code blocks and math
|
|
33
|
+
* expressions.
|
|
34
|
+
*
|
|
35
|
+
* [Sentence RAG Benchmarks](https://superlinked.com/vectorhub/articles/evaluation-rag-retrieval-chunking-methods)
|
|
36
|
+
*
|
|
37
|
+
* @author[Jina AI(2024)](https://gist.github.com/hanxiao/3f60354cf6dc5ac698bc9154163b4e6a)
|
|
38
|
+
* @param { string } text - The input text to be split into semantic chunks.
|
|
39
|
+
* @param { Object }[options = {}] - Optional configuration options(currently unused).
|
|
40
|
+
* @returns { Array.<string> } An array of text chunks, each representing a semantic unit of the document.
|
|
41
|
+
* @category Topics
|
|
42
|
+
* @example
|
|
43
|
+
* const text = "# Heading\n\nThis is a paragraph.\n\n- List item 1\n- List item 2\n\n";
|
|
44
|
+
* const chunks = splitTextSemanticChars(text);
|
|
45
|
+
* console.log(chunks);
|
|
46
|
+
* // Output: ['# Heading', 'This is a paragraph.', '- List item 1', '- List item 2']
|
|
47
|
+
*/
|
|
48
|
+
export function splitTextSemanticChars(text, options = {}) {
|
|
49
|
+
|
|
50
|
+
const MAX_SENTENCE_LENGTH = 400;
|
|
51
|
+
const MAX_HEADING_LENGTH = 7;
|
|
52
|
+
const MAX_HEADING_CONTENT_LENGTH = 200;
|
|
53
|
+
const MAX_HEADING_UNDERLINE_LENGTH = 200;
|
|
54
|
+
const MAX_HTML_HEADING_ATTRIBUTES_LENGTH = 100;
|
|
55
|
+
const MAX_LIST_ITEM_LENGTH = 200;
|
|
56
|
+
const MAX_NESTED_LIST_ITEMS = 6;
|
|
57
|
+
const MAX_LIST_INDENT_SPACES = 7;
|
|
58
|
+
const MAX_BLOCKQUOTE_LINE_LENGTH = 200;
|
|
59
|
+
const MAX_BLOCKQUOTE_LINES = 15;
|
|
60
|
+
const MAX_CODE_BLOCK_LENGTH = 1500;
|
|
61
|
+
const MAX_CODE_LANGUAGE_LENGTH = 20;
|
|
62
|
+
const MAX_INDENTED_CODE_LINES = 20;
|
|
63
|
+
const MAX_TABLE_CELL_LENGTH = 200;
|
|
64
|
+
const MAX_TABLE_ROWS = 20;
|
|
65
|
+
const MAX_HTML_TABLE_LENGTH = 2000;
|
|
66
|
+
const MIN_HORIZONTAL_RULE_LENGTH = 3;
|
|
67
|
+
const MAX_QUOTED_TEXT_LENGTH = 300;
|
|
68
|
+
const MAX_PARENTHETICAL_CONTENT_LENGTH = 200;
|
|
69
|
+
const MAX_NESTED_PARENTHESES = 5;
|
|
70
|
+
const MAX_MATH_INLINE_LENGTH = 100;
|
|
71
|
+
const MAX_MATH_BLOCK_LENGTH = 500;
|
|
72
|
+
const MAX_PARAGRAPH_LENGTH = 1000;
|
|
73
|
+
const MAX_STANDALONE_LINE_LENGTH = 800;
|
|
74
|
+
const MAX_HTML_TAG_ATTRIBUTES_LENGTH = 100;
|
|
75
|
+
const MAX_HTML_TAG_CONTENT_LENGTH = 1000;
|
|
76
|
+
const LOOKAHEAD_RANGE = 100; // Number of characters to look ahead for a sentence boundary
|
|
77
|
+
|
|
78
|
+
const AVOID_AT_START = `[\\s\\]})>,']`;
|
|
79
|
+
const PUNCTUATION = `[.!?\u2026]|\\.{3}|[\\u2026\\u2047-\\u2049]|[\\p{Emoji_Presentation}\\p{Extended_Pictographic}]`;
|
|
80
|
+
const QUOTE_END = `(?:'(?=\`)|''(?=\`\`))`;
|
|
81
|
+
const SENTENCE_END = `(?:${PUNCTUATION}(?<!${AVOID_AT_START}(?=${PUNCTUATION}))|${QUOTE_END})(?=\\S|$)`;
|
|
82
|
+
const SENTENCE_BOUNDARY = `(?:${SENTENCE_END}|(?=[\\r\\n]|$))`;
|
|
83
|
+
const LOOKAHEAD_PATTERN = `(?:(?!${SENTENCE_END}).){1,${LOOKAHEAD_RANGE}}${SENTENCE_END}`;
|
|
84
|
+
const NOT_PUNCTUATION_SPACE = `(?!${PUNCTUATION}\\s)`;
|
|
85
|
+
const SENTENCE_PATTERN = `${NOT_PUNCTUATION_SPACE}(?:[^\\r\\n]{1,{MAX_LENGTH}}${SENTENCE_BOUNDARY}|[^\\r\\n]{1,{MAX_LENGTH}}(?=${PUNCTUATION}|${QUOTE_END})(?:${LOOKAHEAD_PATTERN})?)${AVOID_AT_START}*`;
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
var arrayTextChunks = Array.from(text?.match(new RegExp(
|
|
89
|
+
"(" +
|
|
90
|
+
// 1. Headings (Setext-style, Markdown, and HTML-style, with length constraints)
|
|
91
|
+
`(?:^(?:[#*=-]{1,${MAX_HEADING_LENGTH}}|\\w[^\\r\\n]{0,${MAX_HEADING_CONTENT_LENGTH}}\\r?\\n[-=]{2,${MAX_HEADING_UNDERLINE_LENGTH}}|<h[1-6][^>]{0,${MAX_HTML_HEADING_ATTRIBUTES_LENGTH}}>)[^\\r\\n]{1,${MAX_HEADING_CONTENT_LENGTH}}(?:</h[1-6]>)?(?:\\r?\\n|$))` +
|
|
92
|
+
"|" +
|
|
93
|
+
// New pattern for citations
|
|
94
|
+
`(?:\\[[0-9]+\\][^\\r\\n]{1,${MAX_STANDALONE_LINE_LENGTH}})` +
|
|
95
|
+
"|" +
|
|
96
|
+
// 2. List items (bulleted, numbered, lettered, or task lists, including nested, up to three levels, with length constraints)
|
|
97
|
+
`(?:(?:^|\\r?\\n)[ \\t]{0,3}(?:[-*+\u2022]|\\d{1,3}\\.\\w\\.|\\[[ xX]\\])[ \\t]+${SENTENCE_PATTERN.replace(/{MAX_LENGTH}/g, String(MAX_LIST_ITEM_LENGTH))}` +
|
|
98
|
+
`(?:(?:\\r?\\n[ \\t]{2,5}(?:[-*+\u2022]|\\d{1,3}\\.\\w\\.|\\[[ xX]\\])[ \\t]+${SENTENCE_PATTERN.replace(/{MAX_LENGTH}/g, String(MAX_LIST_ITEM_LENGTH))}){0,${MAX_NESTED_LIST_ITEMS}}` +
|
|
99
|
+
`(?:\\r?\\n[ \\t]{4,${MAX_LIST_INDENT_SPACES}}(?:[-*+\u2022]|\\d{1,3}\\.\\w\\.|\\[[ xX]\\])[ \\t]+${SENTENCE_PATTERN.replace(/{MAX_LENGTH}/g, String(MAX_LIST_ITEM_LENGTH))}){0,${MAX_NESTED_LIST_ITEMS}})?)` +
|
|
100
|
+
"|" +
|
|
101
|
+
// 3. Block quotes (including nested quotes and citations, up to three levels, with length constraints)
|
|
102
|
+
`(?:(?:^>(?:>|\\s{2,}){0,2}${SENTENCE_PATTERN.replace(/{MAX_LENGTH}/g, String(MAX_BLOCKQUOTE_LINE_LENGTH))}\\r?\\n?){1,${MAX_BLOCKQUOTE_LINES}})` +
|
|
103
|
+
"|" +
|
|
104
|
+
// 4. Code blocks (fenced, indented, or HTML pre/code tags, with length constraints)
|
|
105
|
+
`(?:(?:^|\\r?\\n)(?:\`\`\`|~~~)(?:\\w{0,${MAX_CODE_LANGUAGE_LENGTH}})?\\r?\\n[\\s\\S]{0,${MAX_CODE_BLOCK_LENGTH}}?(?:\`\`\`|~~~)\\r?\\n?` +
|
|
106
|
+
`|(?:(?:^|\\r?\\n)(?: {4}|\\t)[^\\r\\n]{0,${MAX_LIST_ITEM_LENGTH}}(?:\\r?\\n(?: {4}|\\t)[^\\r\\n]{0,${MAX_LIST_ITEM_LENGTH}}){0,${MAX_INDENTED_CODE_LINES}}\\r?\\n?)` +
|
|
107
|
+
`|(?:<pre>(?:<code>)?[\\s\\S]{0,${MAX_CODE_BLOCK_LENGTH}}?(?:</code>)?</pre>))` +
|
|
108
|
+
"|" +
|
|
109
|
+
// 5. Tables (Markdown, grid tables, and HTML tables, with length constraints)
|
|
110
|
+
`(?:(?:^|\\r?\\n)(?:\\|[^\\r\\n]{0,${MAX_TABLE_CELL_LENGTH}}\\|(?:\\r?\\n\\|[-:]{1,${MAX_TABLE_CELL_LENGTH}}\\|){0,1}(?:\\r?\\n\\|[^\\r\\n]{0,${MAX_TABLE_CELL_LENGTH}}\\|){0,${MAX_TABLE_ROWS}}` +
|
|
111
|
+
`|<table>[\\s\\S]{0,${MAX_HTML_TABLE_LENGTH}}?</table>))` +
|
|
112
|
+
"|" +
|
|
113
|
+
// 6. Horizontal rules (Markdown and HTML hr tag)
|
|
114
|
+
`(?:^(?:[-*_]){${MIN_HORIZONTAL_RULE_LENGTH},}\\s*$|<hr\\s*/?>)` +
|
|
115
|
+
"|" +
|
|
116
|
+
// 10. Standalone lines or phrases (including single-line blocks and HTML elements, with length constraints)
|
|
117
|
+
`(?!${AVOID_AT_START})(?:^(?:<[a-zA-Z][^>]{0,${MAX_HTML_TAG_ATTRIBUTES_LENGTH}}>)?${SENTENCE_PATTERN.replace(/{MAX_LENGTH}/g, String(MAX_STANDALONE_LINE_LENGTH))}(?:</[a-zA-Z]+>)?(?:\\r?\\n|$))` +
|
|
118
|
+
"|" +
|
|
119
|
+
// 7. Sentences or phrases ending with punctuation (including ellipsis and Unicode punctuation)
|
|
120
|
+
`(?!${AVOID_AT_START})${SENTENCE_PATTERN.replace(/{MAX_LENGTH}/g, String(MAX_SENTENCE_LENGTH))}` +
|
|
121
|
+
"|" +
|
|
122
|
+
// 8. Quoted text, parenthetical phrases, or bracketed content (with length constraints)
|
|
123
|
+
"(?:" +
|
|
124
|
+
`(?<!\\w)\"\"\"[^\"]{0,${MAX_QUOTED_TEXT_LENGTH}}\"\"\"(?!\\w)` +
|
|
125
|
+
`|(?<!\\w)(?:['\"\`'"])[^\\r\\n]{0,${MAX_QUOTED_TEXT_LENGTH}}\\1(?!\\w)` +
|
|
126
|
+
`|(?<!\\w)\`[^\\r\\n]{0,${MAX_QUOTED_TEXT_LENGTH}}'(?!\\w)` +
|
|
127
|
+
`|(?<!\\w)\`\`[^\\r\\n]{0,${MAX_QUOTED_TEXT_LENGTH}}''(?!\\w)` +
|
|
128
|
+
`|\\([^\\r\\n()]{0,${MAX_PARENTHETICAL_CONTENT_LENGTH}}(?:\\([^\\r\\n()]{0,${MAX_PARENTHETICAL_CONTENT_LENGTH}}\\)[^\\r\\n()]{0,${MAX_PARENTHETICAL_CONTENT_LENGTH}}){0,${MAX_NESTED_PARENTHESES}}\\)` +
|
|
129
|
+
`|\\[[^\\r\\n\\[\\]]{0,${MAX_PARENTHETICAL_CONTENT_LENGTH}}(?:\\[[^\\r\\n\\[\\]]{0,${MAX_PARENTHETICAL_CONTENT_LENGTH}}\\][^\\r\\n\\[\\]]{0,${MAX_PARENTHETICAL_CONTENT_LENGTH}}){0,${MAX_NESTED_PARENTHESES}}\\]` +
|
|
130
|
+
`|\\$[^\\r\\n$]{0,${MAX_MATH_INLINE_LENGTH}}\\$` +
|
|
131
|
+
`|\`[^\`\\r\\n]{0,${MAX_MATH_INLINE_LENGTH}}\`` +
|
|
132
|
+
")" +
|
|
133
|
+
"|" +
|
|
134
|
+
// 9. Paragraphs (with length constraints)
|
|
135
|
+
`(?!${AVOID_AT_START})(?:(?:^|\\r?\\n\\r?\\n)(?:<p>)?${SENTENCE_PATTERN.replace(/{MAX_LENGTH}/g, String(MAX_PARAGRAPH_LENGTH))}(?:</p>)?(?=\\r?\\n\\r?\\n|$))` +
|
|
136
|
+
"|" +
|
|
137
|
+
// 11. HTML-like tags and their content (including self-closing tags and attributes, with length constraints)
|
|
138
|
+
`(?:<[a-zA-Z][^>]{0,${MAX_HTML_TAG_ATTRIBUTES_LENGTH}}(?:>[\\s\\S]{0,${MAX_HTML_TAG_CONTENT_LENGTH}}?</[a-zA-Z]+>|\\s*/>))` +
|
|
139
|
+
"|" +
|
|
140
|
+
// 12. LaTeX-style math expressions (inline and block, with length constraints)
|
|
141
|
+
`(?:(?:\\$\\$[\\s\\S]{0,${MAX_MATH_BLOCK_LENGTH}}?\\$\\$)|(?:\\$[^\\$\\r\\n]{0,${MAX_MATH_INLINE_LENGTH}}\\$))` +
|
|
142
|
+
"|" +
|
|
143
|
+
// 14. Fallback for any remaining content (with length constraints)
|
|
144
|
+
`(?!${AVOID_AT_START})${SENTENCE_PATTERN.replace(/{MAX_LENGTH}/g, String(MAX_STANDALONE_LINE_LENGTH))}` +
|
|
145
|
+
")",
|
|
146
|
+
"gmu"
|
|
147
|
+
)));
|
|
148
|
+
|
|
149
|
+
return arrayTextChunks;
|
|
150
|
+
}
|