extract-webpage 1.2.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (117) hide show
  1. package/README.md +212 -0
  2. package/dist/config/env.d.ts +6 -0
  3. package/dist/config/index.d.ts +23 -0
  4. package/dist/config/serverRegistry.d.ts +7 -0
  5. package/dist/config/types.d.ts +4 -0
  6. package/dist/extract-webpage.cjs.js +2 -0
  7. package/dist/extract-webpage.cjs.js.map +1 -0
  8. package/dist/extract-webpage.es.js +5 -0
  9. package/dist/extract-webpage.es.js.map +1 -0
  10. package/dist/html-to-cite/extract-author.d.ts +11 -0
  11. package/dist/html-to-cite/extract-cite.d.ts +33 -0
  12. package/dist/html-to-cite/extract-date/date-extractors.d.ts +40 -0
  13. package/dist/html-to-cite/extract-date/date-validators.d.ts +15 -0
  14. package/dist/html-to-cite/extract-date/extract-date-quick.d.ts +8 -0
  15. package/dist/html-to-cite/extract-date/extract-date.d.ts +26 -0
  16. package/dist/html-to-cite/extract-source.d.ts +7 -0
  17. package/dist/html-to-cite/extract-title.d.ts +11 -0
  18. package/dist/html-to-cite/human-names-recognize.d.ts +16 -0
  19. package/dist/html-to-cite/metadata-to-cite.d.ts +12 -0
  20. package/dist/html-to-cite/url-to-domain.d.ts +20 -0
  21. package/dist/html-to-content/extract-content/extract-content-mercury-utils.d.ts +27 -0
  22. package/dist/html-to-content/extract-content/extract-content-mercury.d.ts +61 -0
  23. package/dist/html-to-content/extract-content/extract-content-readability.d.ts +101 -0
  24. package/dist/html-to-content/html-to-basic-html.d.ts +36 -0
  25. package/dist/html-to-content/html-to-content.d.ts +51 -0
  26. package/dist/html-to-content/html-utils.d.ts +76 -0
  27. package/dist/index.d.ts +26 -0
  28. package/dist/search/index.d.ts +14 -0
  29. package/dist/search/meta-search-agent-reexport.d.ts +8 -0
  30. package/dist/search/public-searxng.d.ts +47 -0
  31. package/dist/search/search-web.d.ts +33 -0
  32. package/dist/search/tavily.d.ts +20 -0
  33. package/dist/search/url-to-html.d.ts +62 -0
  34. package/dist/seektopic/fold-keyphrases.d.ts +28 -0
  35. package/dist/seektopic/ngrams.d.ts +27 -0
  36. package/dist/seektopic/rank-sentences-keyphrases.d.ts +28 -0
  37. package/dist/seektopic/seektopic-keyphrases.d.ts +53 -0
  38. package/dist/seektopic/types.d.ts +86 -0
  39. package/dist/seektopic/vector-search.d.ts +89 -0
  40. package/dist/seektopic/weight-keyphrases.d.ts +22 -0
  41. package/dist/suggest-next-words/autocomplete-ai.d.ts +0 -0
  42. package/dist/suggest-next-words/autocomplete-search-engines.d.ts +64 -0
  43. package/dist/tokenize/suggest-complete-word.d.ts +48 -0
  44. package/dist/tokenize/text-to-chunks.d.ts +48 -0
  45. package/dist/tokenize/text-to-sentences.d.ts +35 -0
  46. package/dist/tokenize/text-to-topic-tokens.d.ts +51 -0
  47. package/dist/tokenize/word-is-ignored.d.ts +12 -0
  48. package/dist/tokenize/word-to-root-stem.d.ts +16 -0
  49. package/dist/url-to-content/docx-to-content.d.ts +22 -0
  50. package/dist/url-to-content/is-url-adult.d.ts +26 -0
  51. package/dist/url-to-content/url-to-content.d.ts +127 -0
  52. package/dist/url-to-content/url-to-html.d.ts +60 -0
  53. package/dist/url-to-content/youtube-helpers.d.ts +23 -0
  54. package/dist/url-to-content/youtube-to-text.d.ts +70 -0
  55. package/dist/utils/documents.d.ts +4 -0
  56. package/dist/utils/grab.d.ts +18 -0
  57. package/package.json +109 -0
  58. package/src/config/env.ts +8 -0
  59. package/src/config/index.ts +233 -0
  60. package/src/config/serverRegistry.ts +24 -0
  61. package/src/config/types.ts +17 -0
  62. package/src/fs-mock.js +22 -0
  63. package/src/global.d.ts +8 -0
  64. package/src/html-to-cite/extract-author.ts +125 -0
  65. package/src/html-to-cite/extract-cite.ts +97 -0
  66. package/src/html-to-cite/extract-date/date-extractors.ts +484 -0
  67. package/src/html-to-cite/extract-date/date-validators.ts +191 -0
  68. package/src/html-to-cite/extract-date/extract-date-quick.ts +184 -0
  69. package/src/html-to-cite/extract-date/extract-date.ts +1049 -0
  70. package/src/html-to-cite/extract-source.ts +30 -0
  71. package/src/html-to-cite/extract-title.ts +78 -0
  72. package/src/html-to-cite/human-names-92k.json +1 -0
  73. package/src/html-to-cite/human-names-recognize.ts +396 -0
  74. package/src/html-to-cite/metadata-to-cite.ts +73 -0
  75. package/src/html-to-cite/url-to-domain.ts +50 -0
  76. package/src/html-to-content/extract-content/extract-content-mercury-utils.ts +696 -0
  77. package/src/html-to-content/extract-content/extract-content-mercury.ts +830 -0
  78. package/src/html-to-content/extract-content/extract-content-readability.ts +432 -0
  79. package/src/html-to-content/extract-content/extract-selectors-per-domain.json +3453 -0
  80. package/src/html-to-content/html-to-basic-html.ts +282 -0
  81. package/src/html-to-content/html-to-content.ts +97 -0
  82. package/src/html-to-content/html-utils.ts +398 -0
  83. package/src/index.ts +29 -0
  84. package/src/search/__tests__/public-searxng.test.ts +529 -0
  85. package/src/search/index.ts +43 -0
  86. package/src/search/meta-search-agent-reexport.ts +38 -0
  87. package/src/search/public-searxng.ts +470 -0
  88. package/src/search/search-web.ts +668 -0
  89. package/src/search/tavily.ts +106 -0
  90. package/src/search/url-to-html.ts +278 -0
  91. package/src/seektopic/fold-keyphrases.ts +87 -0
  92. package/src/seektopic/ngrams.ts +64 -0
  93. package/src/seektopic/rank-sentences-keyphrases.ts +132 -0
  94. package/src/seektopic/seektopic-keyphrases.ts +279 -0
  95. package/src/seektopic/types.ts +92 -0
  96. package/src/seektopic/vector-search.ts +232 -0
  97. package/src/seektopic/weight-keyphrases.ts +59 -0
  98. package/src/suggest-next-words/autocomplete-ai.ts +38 -0
  99. package/src/suggest-next-words/autocomplete-search-engines.ts +435 -0
  100. package/src/tokenize/suggest-complete-word.ts +137 -0
  101. package/src/tokenize/text-to-chunks.ts +150 -0
  102. package/src/tokenize/text-to-sentences.ts +614 -0
  103. package/src/tokenize/text-to-topic-tokens.ts +175 -0
  104. package/src/tokenize/word-is-ignored.ts +53 -0
  105. package/src/tokenize/word-to-root-stem.ts +151 -0
  106. package/src/types.d.ts +130 -0
  107. package/src/url-to-content/.fuse_hidden003bd28a0000000d +332 -0
  108. package/src/url-to-content/__tests__/url-to-content.test.ts +368 -0
  109. package/src/url-to-content/__tests__/url-to-html.test.ts +301 -0
  110. package/src/url-to-content/docx-to-content.ts +702 -0
  111. package/src/url-to-content/is-url-adult.ts +318 -0
  112. package/src/url-to-content/url-to-content.ts +367 -0
  113. package/src/url-to-content/url-to-html.ts +436 -0
  114. package/src/url-to-content/youtube-helpers.ts +64 -0
  115. package/src/url-to-content/youtube-to-text.ts +468 -0
  116. package/src/utils/documents.ts +71 -0
  117. package/src/utils/grab.ts +51 -0
@@ -0,0 +1,468 @@
1
+ // @ts-nocheck
2
+ import { convertURLSafeHTMLToHTML } from "../html-to-content/html-utils";
3
+ import { scrapeURL } from "./url-to-html";
4
+ import grab from "../utils/grab";
5
+ import { generate } from "youtube-po-token-generator";
6
+ import { decode, encode } from "html-entities";
7
+ /**
8
+ * Fetch youtube.com video's webpage HTML for embedded transcript.
9
+ * If blocked, use scraper of alternative sites providing transcripts.
10
+ * @param {string} videoUrl
11
+ * @param {Object} [options]
12
+ * @param {boolean} options.addTimestamps default=true -
13
+ * true to return timestamps, default true
14
+ * @param {boolean} options.timeout default=5 - http request timeout
15
+ * @return {{content: string, timestamps: string, word_count: number}}
16
+ * where content is the full text of the transcript,
17
+ * timestamps is a string of comma-separated [characterIndex, timeSeconds] pairs,
18
+ * and word_count is the number of words in the transcript.
19
+ * @category Extract
20
+ * @author [vtempest (2025)](https://github.com/vtempest)
21
+ */
22
+ export async function convertYoutubeToText(videoUrl, options = {}) {
23
+ const {
24
+ addTimestamps = true,
25
+ addPlayer = true,
26
+ timeout = 10,
27
+ proxy = null,
28
+ } = options;
29
+
30
+ var videoId = getURLYoutubeVideo(videoUrl),
31
+ res = {};
32
+
33
+ // res = await fetchTranscriptTactiq(videoId, options);
34
+
35
+ if (!res.content || res.error)
36
+ res = await fetchTranscriptOfficialYoutube(videoId, options);
37
+
38
+ console.log(res);
39
+
40
+ // if (!res.content || res.error)
41
+ // res = await fetchViaYoutubeToTranscriptCom(videoId, options);
42
+
43
+ // if (!res.content || res.error)
44
+ // res = await fetchViaYoutubeTranscript(videoId, options);
45
+
46
+ // var { date, title, author_cite, length } = await extractYouTubeInfo(
47
+ // videoId,
48
+ // options,
49
+ // );
50
+
51
+ // if (!res.content || res.error) return { error: 1 };
52
+ // var { content, timestamps } = res;
53
+
54
+ // var word_count = content.split(" ").length;
55
+
56
+ // content = convertURLSafeHTMLToHTML(content);
57
+
58
+ // console.log(content);
59
+
60
+ return;
61
+
62
+ //timestamp to track characters per second speed at each interval
63
+ var speedsEveryCharPeriod = {};
64
+ const valueCharPeriod = 100;
65
+
66
+ for (var timestamp of timestamps) {
67
+ var [char, time] = timestamp;
68
+
69
+ var speed = Math.floor(char / time) - 10;
70
+ speedsEveryCharPeriod[Math.floor(char / valueCharPeriod)] = speed;
71
+ }
72
+
73
+ var speeds = Object.keys(speedsEveryCharPeriod).map(
74
+ (timeKey) => speedsEveryCharPeriod[timeKey],
75
+ );
76
+
77
+ let compressed = [];
78
+ let compressedCount = [];
79
+ let currentNum = speeds[0];
80
+ let count = 1;
81
+
82
+ for (let i = 1; i < speeds.length; i++) {
83
+ if (speeds[i] === currentNum) {
84
+ count++;
85
+ } else {
86
+ compressed.push(currentNum);
87
+ compressedCount.push(count);
88
+ currentNum = speeds[i];
89
+ count = 1;
90
+ }
91
+ }
92
+ compressed.push(currentNum);
93
+ compressedCount.push(count);
94
+
95
+ var total = 0;
96
+ compressedCount = compressedCount.map((c) => {
97
+ total += c;
98
+ return total;
99
+ });
100
+
101
+ //remove extra spaces
102
+ content = content.replace(/\s+/g, " ");
103
+
104
+ speeds = compressed.join(",") + " " + compressedCount.join(",");
105
+
106
+ if (addPlayer)
107
+ content = `<iframe width="100%" height="315px" data-timestamps="${speeds}"
108
+ src="https://www.youtube.com/embed/${videoId}" frameborder="0"
109
+ allow="accelerometer; autoplay; clipboard-write; encrypted-media;
110
+ gyroscope; picture-in-picture" allowfullscreen></iframe>${content}`;
111
+
112
+ var source = "YouTube";
113
+
114
+ return {
115
+ html: content,
116
+ word_count,
117
+ source,
118
+ date,
119
+ title,
120
+ author_cite,
121
+ length,
122
+ };
123
+ }
124
+
125
+ function decompressTimestampsArray(compressedStr) {
126
+ let decompressed = [];
127
+ let parts = compressedStr.split(",");
128
+
129
+ for (let part of parts) {
130
+ let [num, count] = part.split("x");
131
+ num = parseInt(num);
132
+ count = parseInt(count);
133
+ decompressed.push(...Array(count).fill(num));
134
+ }
135
+
136
+ return decompressed;
137
+ }
138
+
139
+ /**
140
+ * Test if URL is to youtube video and return video id if true
141
+ * @param {string} url - youtube video URL
142
+ * @returns {string|boolean} video ID or false
143
+ * @private
144
+ */
145
+ export function getURLYoutubeVideo(url) {
146
+ var match = url?.match(
147
+ /(?:\/embed\/|v=|v\/|vi\/|youtu\.be\/|\/v\/|^https?:\/\/(?:www\.)?youtube\.com\/(?:(?:watch)?\?.*v=|(?:embed|v|vi|user)\/))([^#\&\?]*).*/,
148
+ );
149
+ return match ? match[1] : false;
150
+ }
151
+
152
+ /**
153
+ * Fetch-based scraper of youtubetotranscript.com
154
+ * @returns {Object} content, timestamps - where content is the full text of
155
+ * the transcript, and timestamps is an array of [characterIndex, timeSeconds]
156
+ */
157
+ export async function fetchViaYoutubeToTranscriptCom(videoId, options = {}) {
158
+ // try {
159
+ const url = `https://youtubetotranscript.com/transcript?v=${videoId}&current_language_code=en`;
160
+
161
+ var html = await scrapeURL(url, options);
162
+ if (html.error) return { error: 1 };
163
+
164
+ if (!html || html.error) return { error: 1 };
165
+
166
+ //remove line breaks
167
+ html = html?.replace(/[\r\n]/gi, " ");
168
+ // Title regex
169
+ const titleRegex = /<h1[^>]*>([^<]+)<\/h1>/gi;
170
+ var title = html?.match(titleRegex)?.[1];
171
+ //extract title between h1 tags
172
+ title = title
173
+ ?.replace(/<[^>]*>/g, "")
174
+ ?.replace("Transcript of ", "")
175
+ ?.trim();
176
+
177
+ // Author regex with "Author :" prefix
178
+
179
+ const authorRegex = /Author\s*:\s*<a[\s\S]*?>\s*(.*?)\s*<\/a\s*>/;
180
+ var author_cite = html.match(authorRegex)?.[1];
181
+
182
+ const transcriptRegex =
183
+ /<span[^>]*?data-start="([\d.]+)"[^>]*?class="transcript-segment"[^>]*?>[\s\n]*((?:(?!<\/span>).|\n)*?)[\s\n]*<\/span>/gms;
184
+
185
+ const matches = Array.from(html.matchAll(transcriptRegex));
186
+
187
+ const transcript = matches.map((match) => ({
188
+ text: match[2]?.replace(/<br\s*\/?>/gi, " ")?.trim(),
189
+ offset: parseFloat(match[1]),
190
+ }));
191
+
192
+ const content = transcript.map((item) => item.text).join(" ");
193
+ let timestamps = [];
194
+ let charIndex = 0;
195
+
196
+ transcript.forEach((item) => {
197
+ timestamps.push([charIndex, Math.floor(item.offset)]);
198
+ charIndex += item.text.length + 1; // +1 for the space we added
199
+ });
200
+
201
+ return { content, title, author_cite, timestamps };
202
+ // } catch (e) {
203
+ // return { error: 1 };
204
+ // }
205
+ }
206
+
207
+ /**
208
+ * Fetches via tactiq api
209
+ * @param {string} videoId
210
+ * @returns
211
+ */
212
+ async function fetchTranscriptTactiq(videoId, options = {}) {
213
+ try {
214
+ const data = await grab("https://tactiq-apps-prod.tactiq.io/transcript", {
215
+ method: "POST",
216
+ body: JSON.stringify({
217
+ videoUrl: "https://www.youtube.com/watch?v=" + videoId,
218
+ }),
219
+ });
220
+
221
+ if (!data.captions || data.captions.length === 0) {
222
+ return { error: true };
223
+ }
224
+
225
+ let content = "";
226
+ let timestamps = [];
227
+ let currentLength = 0;
228
+
229
+ data.captions.forEach(({ start, dur, text }) => {
230
+ timestamps.push([currentLength, Math.floor(parseFloat(start))]);
231
+ content += text + " ";
232
+ currentLength = content.length;
233
+ });
234
+
235
+ return { content, timestamps };
236
+ } catch (error) {
237
+ console.error("Error fetching transcript:", error);
238
+ return { error: true };
239
+ }
240
+ }
241
+
242
+ /** ========== NOT WORKING ========== */
243
+
244
+ /**
245
+ * Get YouTube transcript of most YouTube videos,
246
+ * except if disabled by uploader
247
+ * fetch-based scraper of youtubetranscript.com
248
+ *
249
+ * @param {string} videoUrl
250
+ * @returns {Object} where content is the full text of
251
+ * the transcript, and timestamps is an array of [characterIndex, timeSeconds]
252
+ * @private
253
+ */
254
+ export async function fetchViaYoutubeTranscript(videoId, options = {}) {
255
+ const url = "https://youtubetranscript.com/?server_vid2=" + videoId;
256
+
257
+ const html = await grab(url, { responseType: "text" });
258
+
259
+ if (html.error) return { error: 1 };
260
+
261
+ const transcriptRegex =
262
+ /<text start="([\d.]+)" dur="[\d.]+">((?:(?!<\/text>).|\n)*?)<\/text>/gms;
263
+ const matches = Array.from(html.matchAll(transcriptRegex));
264
+
265
+ const transcript = matches.map((match) => ({
266
+ text: match[2],
267
+ offset: parseFloat(match[1]),
268
+ }));
269
+
270
+ const content = transcript.map((item) => item.text).join(" ");
271
+ let timestamps = [];
272
+ let charIndex = 0;
273
+
274
+ if (content.includes("YouTube is currently blocking us from fetching"))
275
+ return { error: 1 };
276
+
277
+ transcript.forEach((item) => {
278
+ timestamps.push([charIndex, Math.floor(item.offset)]);
279
+ charIndex += item.text.length + 1; // +1 for the space we added
280
+ });
281
+
282
+ return { content, timestamps };
283
+ }
284
+
285
+ export async function extractYouTubeInfo(videoId, options = {}) {
286
+ var htmlString = await scrapeURL(
287
+ `https://www.youtube.com/watch?v=${videoId}`,
288
+ options,
289
+ )?.data;
290
+
291
+ // youtube bot limiting
292
+ if (
293
+ htmlString?.error ||
294
+ htmlString?.data.includes('class="g-recaptcha"') ||
295
+ !htmlString?.data.includes('"playabilityStatus":')
296
+ )
297
+ return { error: 1 };
298
+
299
+ // Remove newlines for easier regex matching
300
+ htmlString = htmlString?.replace(/\n/g, "");
301
+
302
+ const result = {};
303
+
304
+ // Extract date
305
+ const datePattern =
306
+ /id="info"[^>]*>(?:.*?)(?:(?:Jan|Feb|Mar|Apr|May|Jun|Jul|Aug|Sep|Oct|Nov|Dec)\s+\d{1,2},\s+\d{4}|\d+\s+(?:second|minute|hour|day|week|month|year)s?\s+ago)/gi;
307
+ const dateMatch = htmlString.match(datePattern);
308
+
309
+ if (dateMatch) {
310
+ // Extract just the date part using two possible patterns
311
+ const absoluteDatePattern =
312
+ /(?:Jan|Feb|Mar|Apr|May|Jun|Jul|Aug|Sep|Oct|Nov|Dec)\s+\d{1,2},\s+\d{4}/;
313
+ const relativeDatePattern =
314
+ /\d+\s+(?:second|minute|hour|day|week|month|year)s?\s+ago/i;
315
+
316
+ const absoluteMatch = dateMatch[0].match(absoluteDatePattern);
317
+ const relativeMatch = dateMatch[0].match(relativeDatePattern);
318
+
319
+ if (absoluteMatch) {
320
+ result.date = absoluteMatch[0];
321
+ } else if (relativeMatch) {
322
+ const relative = relativeMatch[0];
323
+ const [amount, unit] = relative.split(" ");
324
+
325
+ // Convert relative date to absolute date
326
+ const now = new Date();
327
+ const date = new Date(now);
328
+
329
+ switch (unit?.toLowerCase()) {
330
+ case "second":
331
+ case "seconds":
332
+ date.setSeconds(now.getSeconds() - parseInt(amount));
333
+ break;
334
+ case "minute":
335
+ case "minutes":
336
+ date.setMinutes(now.getMinutes() - parseInt(amount));
337
+ break;
338
+ case "hour":
339
+ case "hours":
340
+ date.setHours(now.getHours() - parseInt(amount));
341
+ break;
342
+ case "day":
343
+ case "days":
344
+ date.setDate(now.getDate() - parseInt(amount));
345
+ break;
346
+ case "week":
347
+ case "weeks":
348
+ date.setDate(now.getDate() - parseInt(amount) * 7);
349
+ break;
350
+ case "month":
351
+ case "months":
352
+ date.setMonth(now.getMonth() - parseInt(amount));
353
+ break;
354
+ case "year":
355
+ case "years":
356
+ date.setFullYear(now.getFullYear() - parseInt(amount));
357
+ break;
358
+ }
359
+
360
+ // Format the date in YouTube style (MMM DD, YYYY)
361
+ const months = [
362
+ "Jan",
363
+ "Feb",
364
+ "Mar",
365
+ "Apr",
366
+ "May",
367
+ "Jun",
368
+ "Jul",
369
+ "Aug",
370
+ "Sep",
371
+ "Oct",
372
+ "Nov",
373
+ "Dec",
374
+ ];
375
+ const formatted = `${months[date.getMonth()]} ${date.getDate()}, ${date.getFullYear()}`;
376
+ result.date = formatted;
377
+ }
378
+ } else {
379
+ result.date = null;
380
+ }
381
+
382
+ // Extract title
383
+ const titlePattern = /"title":"([^"]+)"/;
384
+ const titleMatch = htmlString.match(titlePattern);
385
+ result.title = titleMatch ? titleMatch[1] : null;
386
+
387
+ // Extract author/channel name
388
+ const authorPattern = /"author":"([^"]+)"/;
389
+ const authorMatch = htmlString.match(authorPattern);
390
+ result.author_cite = authorMatch ? authorMatch[1] : null;
391
+
392
+ // Extract length in seconds (bonus)
393
+ const lengthPattern = /"lengthSeconds":"(\d+)"/;
394
+ const lengthMatch = htmlString.match(lengthPattern);
395
+ result.length = lengthMatch ? parseInt(lengthMatch[1]) : null;
396
+
397
+ return result;
398
+ }
399
+
400
+ async function fetchTranscriptOfficialYoutube(videoId, options = {}) {
401
+ let videoPageBody = await scrapeURL(
402
+ `https://www.youtube.com/watch?v=${videoId}`,
403
+ options,
404
+ );
405
+
406
+ videoPageBody = videoPageBody?.data;
407
+ // if (videoPageBody?.error) return { error: 1 };
408
+ // if (
409
+ // videoPageBody?.includes('class="g-recaptcha"') ||
410
+ // !videoPageBody?.includes('"playabilityStatus":')
411
+ // )
412
+ // return { error: 1 };
413
+
414
+ var videoObj = videoPageBody
415
+ .replace("\n", "")
416
+ .split('"captions":')?.[1]
417
+ ?.split(',"videoDetails')[0];
418
+
419
+ if (!videoObj) return { error: 2 };
420
+
421
+ const captions = JSON.parse(videoObj)?.playerCaptionsTracklistRenderer;
422
+
423
+ if (!captions?.captionTracks) return { error: 3 };
424
+
425
+ const track = captions.captionTracks.find(
426
+ (track) => track.languageCode === "en",
427
+ );
428
+
429
+ if (!track) return { error: 4 };
430
+
431
+ const { poToken } = await generate();
432
+
433
+ let transcriptURL =
434
+ track.baseUrl.replaceAll(",", "%2C") +
435
+ "&potc=1&pot=" +
436
+ encodeURIComponent(poToken.replace(/=/g, "%3D")) +
437
+ "&fmt=json3&xorb=2&xobt=3&xovt=3&cbr=Chrome&cbrver=144.0.0.0&c=WEB&cver=2.20260312.08.00&cplayer=UNIPLAYER&cos=X11&cplatform=DESKTOP";
438
+ console.log(transcriptURL);
439
+
440
+ const transcriptBody = await scrapeURL(transcriptURL, options);
441
+
442
+ console.log(transcriptBody);
443
+ return;
444
+ if (transcriptBody.error) return { error: true };
445
+
446
+ const results = [
447
+ ...transcriptBody.data.matchAll(
448
+ /<text start="([^"]*)" dur="([^"]*)">([^<]*)<\/text>/g,
449
+ ),
450
+ ];
451
+
452
+ var transcript = results.map(([, start, duration, text]) => ({
453
+ text,
454
+ duration: parseFloat(duration),
455
+ offset: parseFloat(start),
456
+ lang: track.languageCode,
457
+ }));
458
+
459
+ var content = "";
460
+ var timestamps = [];
461
+ transcript.forEach(({ offset, text }) => {
462
+ timestamps.push([content.length, Math.floor(offset, 0)]);
463
+
464
+ content += text + " ";
465
+ });
466
+
467
+ return { content, timestamps };
468
+ }
@@ -0,0 +1,71 @@
1
+ import axios from 'axios';
2
+ import { splitTextIntoChunks, type Document } from 'chat-agent-toolkit';
3
+
4
+ /** Strip HTML tags and decode entities \u2014 works in Cloudflare edge runtime */
5
+ function htmlToText(html: string): string {
6
+ return html
7
+ .replace(/<(script|style)[^>]*>[\s\S]*?<\/(script|style)>/gi, ' ')
8
+ .replace(/<[^>]+>/g, ' ')
9
+ .replace(/&nbsp;/g, ' ')
10
+ .replace(/&amp;/g, '&')
11
+ .replace(/&lt;/g, '<')
12
+ .replace(/&gt;/g, '>')
13
+ .replace(/&quot;/g, '"')
14
+ .replace(/&#39;/g, "'")
15
+ .replace(/&[a-z#][a-z0-9]+;/gi, ' ')
16
+ .replace(/\s+/g, ' ')
17
+ .trim();
18
+ }
19
+
20
+ export const getDocumentsFromLinks = async ({ links }: { links: string[] }) => {
21
+ let docs: Document[] = [];
22
+
23
+ await Promise.all(
24
+ links.map(async (link) => {
25
+ link =
26
+ link.startsWith('http://') || link.startsWith('https://')
27
+ ? link
28
+ : `https://${link}`;
29
+
30
+ try {
31
+ const res = await axios.get(link, {
32
+ responseType: 'arraybuffer',
33
+ });
34
+
35
+ const parsedText = htmlToText(res.data.toString('utf8'))
36
+ .replace(/(\r\n|\n|\r)/gm, ' ')
37
+ .replace(/\s+/g, ' ')
38
+ .trim();
39
+
40
+ const splittedText = splitTextIntoChunks(parsedText);
41
+ const title = res.data
42
+ .toString('utf8')
43
+ .match(/<title.*>(.*?)<\/title>/)?.[1];
44
+
45
+ const linkDocs: Document[] = splittedText.map((text) => ({
46
+ pageContent: text,
47
+ metadata: {
48
+ title: title || link,
49
+ url: link,
50
+ },
51
+ }));
52
+
53
+ docs.push(...linkDocs);
54
+ } catch (err) {
55
+ console.error(
56
+ 'An error occurred while getting documents from links: ',
57
+ err,
58
+ );
59
+ docs.push({
60
+ pageContent: `Failed to retrieve content from the link: ${err}`,
61
+ metadata: {
62
+ title: 'Failed to retrieve content',
63
+ url: link,
64
+ },
65
+ });
66
+ }
67
+ }),
68
+ );
69
+
70
+ return docs;
71
+ };
@@ -0,0 +1,51 @@
1
+ /**
2
+ * Fetch wrapper for grabbing binary content
3
+ * Replacement for grab-url package using standard fetch API
4
+ */
5
+ export interface GrabOptions {
6
+ responseType?: "text" | "arraybuffer";
7
+ /** Timeout in seconds */
8
+ timeout?: number;
9
+ method?: string;
10
+ headers?: Record<string, string>;
11
+ body?: string;
12
+ }
13
+
14
+ export default async function grab(
15
+ url: string,
16
+ options?: GrabOptions & { responseType?: "text" }
17
+ ): Promise<string>;
18
+ export default async function grab(
19
+ url: string,
20
+ options: GrabOptions & { responseType: "arraybuffer" }
21
+ ): Promise<ArrayBuffer>;
22
+ export default async function grab(
23
+ url: string,
24
+ options: GrabOptions = {}
25
+ ): Promise<string | ArrayBuffer> {
26
+ const timeoutMs = options.timeout ? options.timeout * 1000 : 10000;
27
+ const controller = new AbortController();
28
+ const timeoutId = setTimeout(() => controller.abort(), timeoutMs);
29
+
30
+ try {
31
+ const response = await fetch(url, {
32
+ method: options.method,
33
+ headers: options.headers,
34
+ body: options.body,
35
+ signal: controller.signal,
36
+ });
37
+ clearTimeout(timeoutId);
38
+
39
+ if (!response.ok) {
40
+ throw new Error(`HTTP ${response.status}`);
41
+ }
42
+
43
+ if (options.responseType === "arraybuffer") {
44
+ return await response.arrayBuffer();
45
+ }
46
+ return await response.text();
47
+ } catch (error) {
48
+ clearTimeout(timeoutId);
49
+ throw error;
50
+ }
51
+ }