extract-webpage 1.2.47 → 1.2.49

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.d.ts CHANGED
@@ -16,7 +16,7 @@ export * from './tokenize/text-to-chunks';
16
16
  export * from './url-to-content/url-to-content';
17
17
  export * from './url-to-content/url-to-html';
18
18
  export * from './html-to-cite/url-to-domain';
19
- export * from './url-to-content/youtube-to-text';
19
+ export * from './url-to-content/youtube-helpers';
20
20
  export * from './url-to-content/docx-to-content';
21
21
  export * from './html-to-content/html-to-content';
22
22
  export * from './html-to-content/extract-content/extract-content-readability';
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "extract-webpage",
3
- "version": "1.2.47",
3
+ "version": "1.2.49",
4
4
  "module": "./dist/extract-webpage.es.js",
5
5
  "description": "Search, extract, cite, and outline the web for a topic with AI Research Agent.",
6
6
  "author": "vtempest <grokthiscontact@gmail.com>",
@@ -78,11 +78,11 @@
78
78
  "dependencies": {
79
79
  "@huggingface/transformers": "^3.8.1",
80
80
  "ai": "^5.0.0",
81
- "chat-agent-toolkit": "^1.2.46",
81
+ "chat-agent-toolkit": "^1.2.48",
82
82
  "chrono-node": "^2.9.0",
83
83
  "drizzle-orm": "^0.45.1",
84
- "extract-pdf": "^0.1.34",
85
- "extract-youtube": "^1.0.33",
84
+ "extract-pdf": "^0.1.36",
85
+ "extract-youtube": "^1.0.36",
86
86
  "highlight.js": "^11.11.1",
87
87
  "html-entities": "^2.6.0",
88
88
  "js-yaml": "^4.1.1",
@@ -93,7 +93,6 @@
93
93
  "node-fetch": "^3.3.2",
94
94
  "qwksearch-api-client": "^0.0.12",
95
95
  "tldts": "^7.0.25",
96
- "youtube-po-token-generator": "^0.6.0",
97
96
  "zod": "^4.3.6"
98
97
  },
99
98
  "keywords": [
package/src/index.ts CHANGED
@@ -17,7 +17,7 @@ export * from "./tokenize/text-to-chunks";
17
17
  export * from "./url-to-content/url-to-content";
18
18
  export * from "./url-to-content/url-to-html";
19
19
  export * from "./html-to-cite/url-to-domain";
20
- export * from "./url-to-content/youtube-to-text";
20
+ export * from "./url-to-content/youtube-helpers";
21
21
  // PDF export removed from main index to prevent pdfjs-serverless from being evaluated at build time
22
22
  // Import directly from "./pdf-to-html/pdfToHtml" when needed
23
23
  export * from "./url-to-content/docx-to-content";
@@ -1,70 +0,0 @@
1
- /**
2
- * Fetch youtube.com video's webpage HTML for embedded transcript.
3
- * If blocked, use scraper of alternative sites providing transcripts.
4
- * @param {string} videoUrl
5
- * @param {Object} [options]
6
- * @param {boolean} options.addTimestamps default=true -
7
- * true to return timestamps, default true
8
- * @param {boolean} options.timeout default=5 - http request timeout
9
- * @return {{content: string, timestamps: string, word_count: number}}
10
- * where content is the full text of the transcript,
11
- * timestamps is a string of comma-separated [characterIndex, timeSeconds] pairs,
12
- * and word_count is the number of words in the transcript.
13
- * @category Extract
14
- * @author [vtempest (2025)](https://github.com/vtempest)
15
- */
16
- export declare function convertYoutubeToText(videoUrl: any, options?: {}): Promise<{
17
- html: any;
18
- word_count: any;
19
- source: string;
20
- date: any;
21
- title: any;
22
- author_cite: any;
23
- length: number;
24
- }>;
25
- /**
26
- * Test if URL is to youtube video and return video id if true
27
- * @param {string} url - youtube video URL
28
- * @returns {string|boolean} video ID or false
29
- * @private
30
- */
31
- export declare function getURLYoutubeVideo(url: any): any;
32
- /**
33
- * Fetch-based scraper of youtubetotranscript.com
34
- * @returns {Object} content, timestamps - where content is the full text of
35
- * the transcript, and timestamps is an array of [characterIndex, timeSeconds]
36
- */
37
- export declare function fetchViaYoutubeToTranscriptCom(videoId: any, options?: {}): Promise<{
38
- error: number;
39
- content?: undefined;
40
- title?: undefined;
41
- author_cite?: undefined;
42
- timestamps?: undefined;
43
- } | {
44
- content: string;
45
- title: any;
46
- author_cite: any;
47
- timestamps: any[];
48
- error?: undefined;
49
- }>;
50
- /** ========== NOT WORKING ========== */
51
- /**
52
- * Get YouTube transcript of most YouTube videos,
53
- * except if disabled by uploader
54
- * fetch-based scraper of youtubetranscript.com
55
- *
56
- * @param {string} videoUrl
57
- * @returns {Object} where content is the full text of
58
- * the transcript, and timestamps is an array of [characterIndex, timeSeconds]
59
- * @private
60
- */
61
- export declare function fetchViaYoutubeTranscript(videoId: any, options?: {}): Promise<{
62
- error: number;
63
- content?: undefined;
64
- timestamps?: undefined;
65
- } | {
66
- content: string;
67
- timestamps: any[];
68
- error?: undefined;
69
- }>;
70
- export declare function extractYouTubeInfo(videoId: any, options?: {}): Promise<{}>;
@@ -1,468 +0,0 @@
1
- // @ts-nocheck
2
- import { convertURLSafeHTMLToHTML } from "../html-to-content/html-utils";
3
- import { scrapeURL } from "./url-to-html";
4
- import grab from "../utils/grab";
5
- import { generate } from "youtube-po-token-generator";
6
- import { decode, encode } from "html-entities";
7
- /**
8
- * Fetch youtube.com video's webpage HTML for embedded transcript.
9
- * If blocked, use scraper of alternative sites providing transcripts.
10
- * @param {string} videoUrl
11
- * @param {Object} [options]
12
- * @param {boolean} options.addTimestamps default=true -
13
- * true to return timestamps, default true
14
- * @param {boolean} options.timeout default=5 - http request timeout
15
- * @return {{content: string, timestamps: string, word_count: number}}
16
- * where content is the full text of the transcript,
17
- * timestamps is a string of comma-separated [characterIndex, timeSeconds] pairs,
18
- * and word_count is the number of words in the transcript.
19
- * @category Extract
20
- * @author [vtempest (2025)](https://github.com/vtempest)
21
- */
22
- export async function convertYoutubeToText(videoUrl, options = {}) {
23
- const {
24
- addTimestamps = true,
25
- addPlayer = true,
26
- timeout = 10,
27
- proxy = null,
28
- } = options;
29
-
30
- var videoId = getURLYoutubeVideo(videoUrl),
31
- res = {};
32
-
33
- // res = await fetchTranscriptTactiq(videoId, options);
34
-
35
- if (!res.content || res.error)
36
- res = await fetchTranscriptOfficialYoutube(videoId, options);
37
-
38
- console.log(res);
39
-
40
- // if (!res.content || res.error)
41
- // res = await fetchViaYoutubeToTranscriptCom(videoId, options);
42
-
43
- // if (!res.content || res.error)
44
- // res = await fetchViaYoutubeTranscript(videoId, options);
45
-
46
- // var { date, title, author_cite, length } = await extractYouTubeInfo(
47
- // videoId,
48
- // options,
49
- // );
50
-
51
- // if (!res.content || res.error) return { error: 1 };
52
- // var { content, timestamps } = res;
53
-
54
- // var word_count = content.split(" ").length;
55
-
56
- // content = convertURLSafeHTMLToHTML(content);
57
-
58
- // console.log(content);
59
-
60
- return;
61
-
62
- //timestamp to track characters per second speed at each interval
63
- var speedsEveryCharPeriod = {};
64
- const valueCharPeriod = 100;
65
-
66
- for (var timestamp of timestamps) {
67
- var [char, time] = timestamp;
68
-
69
- var speed = Math.floor(char / time) - 10;
70
- speedsEveryCharPeriod[Math.floor(char / valueCharPeriod)] = speed;
71
- }
72
-
73
- var speeds = Object.keys(speedsEveryCharPeriod).map(
74
- (timeKey) => speedsEveryCharPeriod[timeKey],
75
- );
76
-
77
- let compressed = [];
78
- let compressedCount = [];
79
- let currentNum = speeds[0];
80
- let count = 1;
81
-
82
- for (let i = 1; i < speeds.length; i++) {
83
- if (speeds[i] === currentNum) {
84
- count++;
85
- } else {
86
- compressed.push(currentNum);
87
- compressedCount.push(count);
88
- currentNum = speeds[i];
89
- count = 1;
90
- }
91
- }
92
- compressed.push(currentNum);
93
- compressedCount.push(count);
94
-
95
- var total = 0;
96
- compressedCount = compressedCount.map((c) => {
97
- total += c;
98
- return total;
99
- });
100
-
101
- //remove extra spaces
102
- content = content.replace(/\s+/g, " ");
103
-
104
- speeds = compressed.join(",") + " " + compressedCount.join(",");
105
-
106
- if (addPlayer)
107
- content = `<iframe width="100%" height="315px" data-timestamps="${speeds}"
108
- src="https://www.youtube.com/embed/${videoId}" frameborder="0"
109
- allow="accelerometer; autoplay; clipboard-write; encrypted-media;
110
- gyroscope; picture-in-picture" allowfullscreen></iframe>${content}`;
111
-
112
- var source = "YouTube";
113
-
114
- return {
115
- html: content,
116
- word_count,
117
- source,
118
- date,
119
- title,
120
- author_cite,
121
- length,
122
- };
123
- }
124
-
125
- function decompressTimestampsArray(compressedStr) {
126
- let decompressed = [];
127
- let parts = compressedStr.split(",");
128
-
129
- for (let part of parts) {
130
- let [num, count] = part.split("x");
131
- num = parseInt(num);
132
- count = parseInt(count);
133
- decompressed.push(...Array(count).fill(num));
134
- }
135
-
136
- return decompressed;
137
- }
138
-
139
- /**
140
- * Test if URL is to youtube video and return video id if true
141
- * @param {string} url - youtube video URL
142
- * @returns {string|boolean} video ID or false
143
- * @private
144
- */
145
- export function getURLYoutubeVideo(url) {
146
- var match = url?.match(
147
- /(?:\/embed\/|v=|v\/|vi\/|youtu\.be\/|\/v\/|^https?:\/\/(?:www\.)?youtube\.com\/(?:(?:watch)?\?.*v=|(?:embed|v|vi|user)\/))([^#\&\?]*).*/,
148
- );
149
- return match ? match[1] : false;
150
- }
151
-
152
- /**
153
- * Fetch-based scraper of youtubetotranscript.com
154
- * @returns {Object} content, timestamps - where content is the full text of
155
- * the transcript, and timestamps is an array of [characterIndex, timeSeconds]
156
- */
157
- export async function fetchViaYoutubeToTranscriptCom(videoId, options = {}) {
158
- // try {
159
- const url = `https://youtubetotranscript.com/transcript?v=${videoId}&current_language_code=en`;
160
-
161
- var html = await scrapeURL(url, options);
162
- if (html.error) return { error: 1 };
163
-
164
- if (!html || html.error) return { error: 1 };
165
-
166
- //remove line breaks
167
- html = html?.replace(/[\r\n]/gi, " ");
168
- // Title regex
169
- const titleRegex = /<h1[^>]*>([^<]+)<\/h1>/gi;
170
- var title = html?.match(titleRegex)?.[1];
171
- //extract title between h1 tags
172
- title = title
173
- ?.replace(/<[^>]*>/g, "")
174
- ?.replace("Transcript of ", "")
175
- ?.trim();
176
-
177
- // Author regex with "Author :" prefix
178
-
179
- const authorRegex = /Author\s*:\s*<a[\s\S]*?>\s*(.*?)\s*<\/a\s*>/;
180
- var author_cite = html.match(authorRegex)?.[1];
181
-
182
- const transcriptRegex =
183
- /<span[^>]*?data-start="([\d.]+)"[^>]*?class="transcript-segment"[^>]*?>[\s\n]*((?:(?!<\/span>).|\n)*?)[\s\n]*<\/span>/gms;
184
-
185
- const matches = Array.from(html.matchAll(transcriptRegex));
186
-
187
- const transcript = matches.map((match) => ({
188
- text: match[2]?.replace(/<br\s*\/?>/gi, " ")?.trim(),
189
- offset: parseFloat(match[1]),
190
- }));
191
-
192
- const content = transcript.map((item) => item.text).join(" ");
193
- let timestamps = [];
194
- let charIndex = 0;
195
-
196
- transcript.forEach((item) => {
197
- timestamps.push([charIndex, Math.floor(item.offset)]);
198
- charIndex += item.text.length + 1; // +1 for the space we added
199
- });
200
-
201
- return { content, title, author_cite, timestamps };
202
- // } catch (e) {
203
- // return { error: 1 };
204
- // }
205
- }
206
-
207
- /**
208
- * Fetches via tactiq api
209
- * @param {string} videoId
210
- * @returns
211
- */
212
- async function fetchTranscriptTactiq(videoId, options = {}) {
213
- try {
214
- const data = await grab("https://tactiq-apps-prod.tactiq.io/transcript", {
215
- method: "POST",
216
- body: JSON.stringify({
217
- videoUrl: "https://www.youtube.com/watch?v=" + videoId,
218
- }),
219
- });
220
-
221
- if (!data.captions || data.captions.length === 0) {
222
- return { error: true };
223
- }
224
-
225
- let content = "";
226
- let timestamps = [];
227
- let currentLength = 0;
228
-
229
- data.captions.forEach(({ start, dur, text }) => {
230
- timestamps.push([currentLength, Math.floor(parseFloat(start))]);
231
- content += text + " ";
232
- currentLength = content.length;
233
- });
234
-
235
- return { content, timestamps };
236
- } catch (error) {
237
- console.error("Error fetching transcript:", error);
238
- return { error: true };
239
- }
240
- }
241
-
242
- /** ========== NOT WORKING ========== */
243
-
244
- /**
245
- * Get YouTube transcript of most YouTube videos,
246
- * except if disabled by uploader
247
- * fetch-based scraper of youtubetranscript.com
248
- *
249
- * @param {string} videoUrl
250
- * @returns {Object} where content is the full text of
251
- * the transcript, and timestamps is an array of [characterIndex, timeSeconds]
252
- * @private
253
- */
254
- export async function fetchViaYoutubeTranscript(videoId, options = {}) {
255
- const url = "https://youtubetranscript.com/?server_vid2=" + videoId;
256
-
257
- const html = await grab(url, { responseType: "text" });
258
-
259
- if (html.error) return { error: 1 };
260
-
261
- const transcriptRegex =
262
- /<text start="([\d.]+)" dur="[\d.]+">((?:(?!<\/text>).|\n)*?)<\/text>/gms;
263
- const matches = Array.from(html.matchAll(transcriptRegex));
264
-
265
- const transcript = matches.map((match) => ({
266
- text: match[2],
267
- offset: parseFloat(match[1]),
268
- }));
269
-
270
- const content = transcript.map((item) => item.text).join(" ");
271
- let timestamps = [];
272
- let charIndex = 0;
273
-
274
- if (content.includes("YouTube is currently blocking us from fetching"))
275
- return { error: 1 };
276
-
277
- transcript.forEach((item) => {
278
- timestamps.push([charIndex, Math.floor(item.offset)]);
279
- charIndex += item.text.length + 1; // +1 for the space we added
280
- });
281
-
282
- return { content, timestamps };
283
- }
284
-
285
- export async function extractYouTubeInfo(videoId, options = {}) {
286
- var htmlString = await scrapeURL(
287
- `https://www.youtube.com/watch?v=${videoId}`,
288
- options,
289
- )?.data;
290
-
291
- // youtube bot limiting
292
- if (
293
- htmlString?.error ||
294
- htmlString?.data.includes('class="g-recaptcha"') ||
295
- !htmlString?.data.includes('"playabilityStatus":')
296
- )
297
- return { error: 1 };
298
-
299
- // Remove newlines for easier regex matching
300
- htmlString = htmlString?.replace(/\n/g, "");
301
-
302
- const result = {};
303
-
304
- // Extract date
305
- const datePattern =
306
- /id="info"[^>]*>(?:.*?)(?:(?:Jan|Feb|Mar|Apr|May|Jun|Jul|Aug|Sep|Oct|Nov|Dec)\s+\d{1,2},\s+\d{4}|\d+\s+(?:second|minute|hour|day|week|month|year)s?\s+ago)/gi;
307
- const dateMatch = htmlString.match(datePattern);
308
-
309
- if (dateMatch) {
310
- // Extract just the date part using two possible patterns
311
- const absoluteDatePattern =
312
- /(?:Jan|Feb|Mar|Apr|May|Jun|Jul|Aug|Sep|Oct|Nov|Dec)\s+\d{1,2},\s+\d{4}/;
313
- const relativeDatePattern =
314
- /\d+\s+(?:second|minute|hour|day|week|month|year)s?\s+ago/i;
315
-
316
- const absoluteMatch = dateMatch[0].match(absoluteDatePattern);
317
- const relativeMatch = dateMatch[0].match(relativeDatePattern);
318
-
319
- if (absoluteMatch) {
320
- result.date = absoluteMatch[0];
321
- } else if (relativeMatch) {
322
- const relative = relativeMatch[0];
323
- const [amount, unit] = relative.split(" ");
324
-
325
- // Convert relative date to absolute date
326
- const now = new Date();
327
- const date = new Date(now);
328
-
329
- switch (unit?.toLowerCase()) {
330
- case "second":
331
- case "seconds":
332
- date.setSeconds(now.getSeconds() - parseInt(amount));
333
- break;
334
- case "minute":
335
- case "minutes":
336
- date.setMinutes(now.getMinutes() - parseInt(amount));
337
- break;
338
- case "hour":
339
- case "hours":
340
- date.setHours(now.getHours() - parseInt(amount));
341
- break;
342
- case "day":
343
- case "days":
344
- date.setDate(now.getDate() - parseInt(amount));
345
- break;
346
- case "week":
347
- case "weeks":
348
- date.setDate(now.getDate() - parseInt(amount) * 7);
349
- break;
350
- case "month":
351
- case "months":
352
- date.setMonth(now.getMonth() - parseInt(amount));
353
- break;
354
- case "year":
355
- case "years":
356
- date.setFullYear(now.getFullYear() - parseInt(amount));
357
- break;
358
- }
359
-
360
- // Format the date in YouTube style (MMM DD, YYYY)
361
- const months = [
362
- "Jan",
363
- "Feb",
364
- "Mar",
365
- "Apr",
366
- "May",
367
- "Jun",
368
- "Jul",
369
- "Aug",
370
- "Sep",
371
- "Oct",
372
- "Nov",
373
- "Dec",
374
- ];
375
- const formatted = `${months[date.getMonth()]} ${date.getDate()}, ${date.getFullYear()}`;
376
- result.date = formatted;
377
- }
378
- } else {
379
- result.date = null;
380
- }
381
-
382
- // Extract title
383
- const titlePattern = /"title":"([^"]+)"/;
384
- const titleMatch = htmlString.match(titlePattern);
385
- result.title = titleMatch ? titleMatch[1] : null;
386
-
387
- // Extract author/channel name
388
- const authorPattern = /"author":"([^"]+)"/;
389
- const authorMatch = htmlString.match(authorPattern);
390
- result.author_cite = authorMatch ? authorMatch[1] : null;
391
-
392
- // Extract length in seconds (bonus)
393
- const lengthPattern = /"lengthSeconds":"(\d+)"/;
394
- const lengthMatch = htmlString.match(lengthPattern);
395
- result.length = lengthMatch ? parseInt(lengthMatch[1]) : null;
396
-
397
- return result;
398
- }
399
-
400
- async function fetchTranscriptOfficialYoutube(videoId, options = {}) {
401
- let videoPageBody = await scrapeURL(
402
- `https://www.youtube.com/watch?v=${videoId}`,
403
- options,
404
- );
405
-
406
- videoPageBody = videoPageBody?.data;
407
- // if (videoPageBody?.error) return { error: 1 };
408
- // if (
409
- // videoPageBody?.includes('class="g-recaptcha"') ||
410
- // !videoPageBody?.includes('"playabilityStatus":')
411
- // )
412
- // return { error: 1 };
413
-
414
- var videoObj = videoPageBody
415
- .replace("\n", "")
416
- .split('"captions":')?.[1]
417
- ?.split(',"videoDetails')[0];
418
-
419
- if (!videoObj) return { error: 2 };
420
-
421
- const captions = JSON.parse(videoObj)?.playerCaptionsTracklistRenderer;
422
-
423
- if (!captions?.captionTracks) return { error: 3 };
424
-
425
- const track = captions.captionTracks.find(
426
- (track) => track.languageCode === "en",
427
- );
428
-
429
- if (!track) return { error: 4 };
430
-
431
- const { poToken } = await generate();
432
-
433
- let transcriptURL =
434
- track.baseUrl.replaceAll(",", "%2C") +
435
- "&potc=1&pot=" +
436
- encodeURIComponent(poToken.replace(/=/g, "%3D")) +
437
- "&fmt=json3&xorb=2&xobt=3&xovt=3&cbr=Chrome&cbrver=144.0.0.0&c=WEB&cver=2.20260312.08.00&cplayer=UNIPLAYER&cos=X11&cplatform=DESKTOP";
438
- console.log(transcriptURL);
439
-
440
- const transcriptBody = await scrapeURL(transcriptURL, options);
441
-
442
- console.log(transcriptBody);
443
- return;
444
- if (transcriptBody.error) return { error: true };
445
-
446
- const results = [
447
- ...transcriptBody.data.matchAll(
448
- /<text start="([^"]*)" dur="([^"]*)">([^<]*)<\/text>/g,
449
- ),
450
- ];
451
-
452
- var transcript = results.map(([, start, duration, text]) => ({
453
- text,
454
- duration: parseFloat(duration),
455
- offset: parseFloat(start),
456
- lang: track.languageCode,
457
- }));
458
-
459
- var content = "";
460
- var timestamps = [];
461
- transcript.forEach(({ offset, text }) => {
462
- timestamps.push([content.length, Math.floor(offset, 0)]);
463
-
464
- content += text + " ";
465
- });
466
-
467
- return { content, timestamps };
468
- }