extract-webpage 1.2.48 → 1.2.49

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.d.ts CHANGED
@@ -16,7 +16,7 @@ export * from './tokenize/text-to-chunks';
16
16
  export * from './url-to-content/url-to-content';
17
17
  export * from './url-to-content/url-to-html';
18
18
  export * from './html-to-cite/url-to-domain';
19
- export * from './url-to-content/youtube-to-text';
19
+ export * from './url-to-content/youtube-helpers';
20
20
  export * from './url-to-content/docx-to-content';
21
21
  export * from './html-to-content/html-to-content';
22
22
  export * from './html-to-content/extract-content/extract-content-readability';
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "extract-webpage",
3
- "version": "1.2.48",
3
+ "version": "1.2.49",
4
4
  "module": "./dist/extract-webpage.es.js",
5
5
  "description": "Search, extract, cite, and outline the web for a topic with AI Research Agent.",
6
6
  "author": "vtempest <grokthiscontact@gmail.com>",
@@ -78,11 +78,11 @@
78
78
  "dependencies": {
79
79
  "@huggingface/transformers": "^3.8.1",
80
80
  "ai": "^5.0.0",
81
- "chat-agent-toolkit": "^1.2.47",
81
+ "chat-agent-toolkit": "^1.2.48",
82
82
  "chrono-node": "^2.9.0",
83
83
  "drizzle-orm": "^0.45.1",
84
- "extract-pdf": "^0.1.35",
85
- "extract-youtube": "^1.0.35",
84
+ "extract-pdf": "^0.1.36",
85
+ "extract-youtube": "^1.0.36",
86
86
  "highlight.js": "^11.11.1",
87
87
  "html-entities": "^2.6.0",
88
88
  "js-yaml": "^4.1.1",
package/src/index.ts CHANGED
@@ -17,7 +17,7 @@ export * from "./tokenize/text-to-chunks";
17
17
  export * from "./url-to-content/url-to-content";
18
18
  export * from "./url-to-content/url-to-html";
19
19
  export * from "./html-to-cite/url-to-domain";
20
- export * from "./url-to-content/youtube-to-text";
20
+ export * from "./url-to-content/youtube-helpers";
21
21
  // PDF export removed from main index to prevent pdfjs-serverless from being evaluated at build time
22
22
  // Import directly from "./pdf-to-html/pdfToHtml" when needed
23
23
  export * from "./url-to-content/docx-to-content";
@@ -1,70 +0,0 @@
1
- /**
2
- * Fetch youtube.com video's webpage HTML for embedded transcript.
3
- * If blocked, use scraper of alternative sites providing transcripts.
4
- * @param {string} videoUrl
5
- * @param {Object} [options]
6
- * @param {boolean} options.addTimestamps default=true -
7
- * true to return timestamps, default true
8
- * @param {boolean} options.timeout default=5 - http request timeout
9
- * @return {{content: string, timestamps: string, word_count: number}}
10
- * where content is the full text of the transcript,
11
- * timestamps is a string of comma-separated [characterIndex, timeSeconds] pairs,
12
- * and word_count is the number of words in the transcript.
13
- * @category Extract
14
- * @author [vtempest (2025)](https://github.com/vtempest)
15
- */
16
- export declare function convertYoutubeToText(videoUrl: any, options?: {}): Promise<{
17
- html: any;
18
- word_count: any;
19
- source: string;
20
- date: any;
21
- title: any;
22
- author_cite: any;
23
- length: number;
24
- }>;
25
- /**
26
- * Test if URL is to youtube video and return video id if true
27
- * @param {string} url - youtube video URL
28
- * @returns {string|boolean} video ID or false
29
- * @private
30
- */
31
- export declare function getURLYoutubeVideo(url: any): any;
32
- /**
33
- * Fetch-based scraper of youtubetotranscript.com
34
- * @returns {Object} content, timestamps - where content is the full text of
35
- * the transcript, and timestamps is an array of [characterIndex, timeSeconds]
36
- */
37
- export declare function fetchViaYoutubeToTranscriptCom(videoId: any, options?: {}): Promise<{
38
- error: number;
39
- content?: undefined;
40
- title?: undefined;
41
- author_cite?: undefined;
42
- timestamps?: undefined;
43
- } | {
44
- content: string;
45
- title: any;
46
- author_cite: any;
47
- timestamps: any[];
48
- error?: undefined;
49
- }>;
50
- /** ========== NOT WORKING ========== */
51
- /**
52
- * Get YouTube transcript of most YouTube videos,
53
- * except if disabled by uploader
54
- * fetch-based scraper of youtubetranscript.com
55
- *
56
- * @param {string} videoUrl
57
- * @returns {Object} where content is the full text of
58
- * the transcript, and timestamps is an array of [characterIndex, timeSeconds]
59
- * @private
60
- */
61
- export declare function fetchViaYoutubeTranscript(videoId: any, options?: {}): Promise<{
62
- error: number;
63
- content?: undefined;
64
- timestamps?: undefined;
65
- } | {
66
- content: string;
67
- timestamps: any[];
68
- error?: undefined;
69
- }>;
70
- export declare function extractYouTubeInfo(videoId: any, options?: {}): Promise<{}>;
@@ -1,468 +0,0 @@
1
- // @ts-nocheck
2
- import { convertURLSafeHTMLToHTML } from "../html-to-content/html-utils";
3
- import { scrapeURL } from "./url-to-html";
4
- import grab from "../utils/grab";
5
- import { decode, encode } from "html-entities";
6
- /**
7
- * Fetch youtube.com video's webpage HTML for embedded transcript.
8
- * If blocked, use scraper of alternative sites providing transcripts.
9
- * @param {string} videoUrl
10
- * @param {Object} [options]
11
- * @param {boolean} options.addTimestamps default=true -
12
- * true to return timestamps, default true
13
- * @param {boolean} options.timeout default=5 - http request timeout
14
- * @return {{content: string, timestamps: string, word_count: number}}
15
- * where content is the full text of the transcript,
16
- * timestamps is a string of comma-separated [characterIndex, timeSeconds] pairs,
17
- * and word_count is the number of words in the transcript.
18
- * @category Extract
19
- * @author [vtempest (2025)](https://github.com/vtempest)
20
- */
21
- export async function convertYoutubeToText(videoUrl, options = {}) {
22
- const {
23
- addTimestamps = true,
24
- addPlayer = true,
25
- timeout = 10,
26
- proxy = null,
27
- } = options;
28
-
29
- var videoId = getURLYoutubeVideo(videoUrl),
30
- res = {};
31
-
32
- // res = await fetchTranscriptTactiq(videoId, options);
33
-
34
- if (!res.content || res.error)
35
- res = await fetchTranscriptOfficialYoutube(videoId, options);
36
-
37
- console.log(res);
38
-
39
- // if (!res.content || res.error)
40
- // res = await fetchViaYoutubeToTranscriptCom(videoId, options);
41
-
42
- // if (!res.content || res.error)
43
- // res = await fetchViaYoutubeTranscript(videoId, options);
44
-
45
- // var { date, title, author_cite, length } = await extractYouTubeInfo(
46
- // videoId,
47
- // options,
48
- // );
49
-
50
- // if (!res.content || res.error) return { error: 1 };
51
- // var { content, timestamps } = res;
52
-
53
- // var word_count = content.split(" ").length;
54
-
55
- // content = convertURLSafeHTMLToHTML(content);
56
-
57
- // console.log(content);
58
-
59
- return;
60
-
61
- //timestamp to track characters per second speed at each interval
62
- var speedsEveryCharPeriod = {};
63
- const valueCharPeriod = 100;
64
-
65
- for (var timestamp of timestamps) {
66
- var [char, time] = timestamp;
67
-
68
- var speed = Math.floor(char / time) - 10;
69
- speedsEveryCharPeriod[Math.floor(char / valueCharPeriod)] = speed;
70
- }
71
-
72
- var speeds = Object.keys(speedsEveryCharPeriod).map(
73
- (timeKey) => speedsEveryCharPeriod[timeKey],
74
- );
75
-
76
- let compressed = [];
77
- let compressedCount = [];
78
- let currentNum = speeds[0];
79
- let count = 1;
80
-
81
- for (let i = 1; i < speeds.length; i++) {
82
- if (speeds[i] === currentNum) {
83
- count++;
84
- } else {
85
- compressed.push(currentNum);
86
- compressedCount.push(count);
87
- currentNum = speeds[i];
88
- count = 1;
89
- }
90
- }
91
- compressed.push(currentNum);
92
- compressedCount.push(count);
93
-
94
- var total = 0;
95
- compressedCount = compressedCount.map((c) => {
96
- total += c;
97
- return total;
98
- });
99
-
100
- //remove extra spaces
101
- content = content.replace(/\s+/g, " ");
102
-
103
- speeds = compressed.join(",") + " " + compressedCount.join(",");
104
-
105
- if (addPlayer)
106
- content = `<iframe width="100%" height="315px" data-timestamps="${speeds}"
107
- src="https://www.youtube.com/embed/${videoId}" frameborder="0"
108
- allow="accelerometer; autoplay; clipboard-write; encrypted-media;
109
- gyroscope; picture-in-picture" allowfullscreen></iframe>${content}`;
110
-
111
- var source = "YouTube";
112
-
113
- return {
114
- html: content,
115
- word_count,
116
- source,
117
- date,
118
- title,
119
- author_cite,
120
- length,
121
- };
122
- }
123
-
124
- function decompressTimestampsArray(compressedStr) {
125
- let decompressed = [];
126
- let parts = compressedStr.split(",");
127
-
128
- for (let part of parts) {
129
- let [num, count] = part.split("x");
130
- num = parseInt(num);
131
- count = parseInt(count);
132
- decompressed.push(...Array(count).fill(num));
133
- }
134
-
135
- return decompressed;
136
- }
137
-
138
- /**
139
- * Test if URL is to youtube video and return video id if true
140
- * @param {string} url - youtube video URL
141
- * @returns {string|boolean} video ID or false
142
- * @private
143
- */
144
- export function getURLYoutubeVideo(url) {
145
- var match = url?.match(
146
- /(?:\/embed\/|v=|v\/|vi\/|youtu\.be\/|\/v\/|^https?:\/\/(?:www\.)?youtube\.com\/(?:(?:watch)?\?.*v=|(?:embed|v|vi|user)\/))([^#\&\?]*).*/,
147
- );
148
- return match ? match[1] : false;
149
- }
150
-
151
- /**
152
- * Fetch-based scraper of youtubetotranscript.com
153
- * @returns {Object} content, timestamps - where content is the full text of
154
- * the transcript, and timestamps is an array of [characterIndex, timeSeconds]
155
- */
156
- export async function fetchViaYoutubeToTranscriptCom(videoId, options = {}) {
157
- // try {
158
- const url = `https://youtubetotranscript.com/transcript?v=${videoId}&current_language_code=en`;
159
-
160
- var html = await scrapeURL(url, options);
161
- if (html.error) return { error: 1 };
162
-
163
- if (!html || html.error) return { error: 1 };
164
-
165
- //remove line breaks
166
- html = html?.replace(/[\r\n]/gi, " ");
167
- // Title regex
168
- const titleRegex = /<h1[^>]*>([^<]+)<\/h1>/gi;
169
- var title = html?.match(titleRegex)?.[1];
170
- //extract title between h1 tags
171
- title = title
172
- ?.replace(/<[^>]*>/g, "")
173
- ?.replace("Transcript of ", "")
174
- ?.trim();
175
-
176
- // Author regex with "Author :" prefix
177
-
178
- const authorRegex = /Author\s*:\s*<a[\s\S]*?>\s*(.*?)\s*<\/a\s*>/;
179
- var author_cite = html.match(authorRegex)?.[1];
180
-
181
- const transcriptRegex =
182
- /<span[^>]*?data-start="([\d.]+)"[^>]*?class="transcript-segment"[^>]*?>[\s\n]*((?:(?!<\/span>).|\n)*?)[\s\n]*<\/span>/gms;
183
-
184
- const matches = Array.from(html.matchAll(transcriptRegex));
185
-
186
- const transcript = matches.map((match) => ({
187
- text: match[2]?.replace(/<br\s*\/?>/gi, " ")?.trim(),
188
- offset: parseFloat(match[1]),
189
- }));
190
-
191
- const content = transcript.map((item) => item.text).join(" ");
192
- let timestamps = [];
193
- let charIndex = 0;
194
-
195
- transcript.forEach((item) => {
196
- timestamps.push([charIndex, Math.floor(item.offset)]);
197
- charIndex += item.text.length + 1; // +1 for the space we added
198
- });
199
-
200
- return { content, title, author_cite, timestamps };
201
- // } catch (e) {
202
- // return { error: 1 };
203
- // }
204
- }
205
-
206
- /**
207
- * Fetches via tactiq api
208
- * @param {string} videoId
209
- * @returns
210
- */
211
- async function fetchTranscriptTactiq(videoId, options = {}) {
212
- try {
213
- const data = await grab("https://tactiq-apps-prod.tactiq.io/transcript", {
214
- method: "POST",
215
- body: JSON.stringify({
216
- videoUrl: "https://www.youtube.com/watch?v=" + videoId,
217
- }),
218
- });
219
-
220
- if (!data.captions || data.captions.length === 0) {
221
- return { error: true };
222
- }
223
-
224
- let content = "";
225
- let timestamps = [];
226
- let currentLength = 0;
227
-
228
- data.captions.forEach(({ start, dur, text }) => {
229
- timestamps.push([currentLength, Math.floor(parseFloat(start))]);
230
- content += text + " ";
231
- currentLength = content.length;
232
- });
233
-
234
- return { content, timestamps };
235
- } catch (error) {
236
- console.error("Error fetching transcript:", error);
237
- return { error: true };
238
- }
239
- }
240
-
241
- /** ========== NOT WORKING ========== */
242
-
243
- /**
244
- * Get YouTube transcript of most YouTube videos,
245
- * except if disabled by uploader
246
- * fetch-based scraper of youtubetranscript.com
247
- *
248
- * @param {string} videoUrl
249
- * @returns {Object} where content is the full text of
250
- * the transcript, and timestamps is an array of [characterIndex, timeSeconds]
251
- * @private
252
- */
253
- export async function fetchViaYoutubeTranscript(videoId, options = {}) {
254
- const url = "https://youtubetranscript.com/?server_vid2=" + videoId;
255
-
256
- const html = await grab(url, { responseType: "text" });
257
-
258
- if (html.error) return { error: 1 };
259
-
260
- const transcriptRegex =
261
- /<text start="([\d.]+)" dur="[\d.]+">((?:(?!<\/text>).|\n)*?)<\/text>/gms;
262
- const matches = Array.from(html.matchAll(transcriptRegex));
263
-
264
- const transcript = matches.map((match) => ({
265
- text: match[2],
266
- offset: parseFloat(match[1]),
267
- }));
268
-
269
- const content = transcript.map((item) => item.text).join(" ");
270
- let timestamps = [];
271
- let charIndex = 0;
272
-
273
- if (content.includes("YouTube is currently blocking us from fetching"))
274
- return { error: 1 };
275
-
276
- transcript.forEach((item) => {
277
- timestamps.push([charIndex, Math.floor(item.offset)]);
278
- charIndex += item.text.length + 1; // +1 for the space we added
279
- });
280
-
281
- return { content, timestamps };
282
- }
283
-
284
- export async function extractYouTubeInfo(videoId, options = {}) {
285
- var htmlString = await scrapeURL(
286
- `https://www.youtube.com/watch?v=${videoId}`,
287
- options,
288
- )?.data;
289
-
290
- // youtube bot limiting
291
- if (
292
- htmlString?.error ||
293
- htmlString?.data.includes('class="g-recaptcha"') ||
294
- !htmlString?.data.includes('"playabilityStatus":')
295
- )
296
- return { error: 1 };
297
-
298
- // Remove newlines for easier regex matching
299
- htmlString = htmlString?.replace(/\n/g, "");
300
-
301
- const result = {};
302
-
303
- // Extract date
304
- const datePattern =
305
- /id="info"[^>]*>(?:.*?)(?:(?:Jan|Feb|Mar|Apr|May|Jun|Jul|Aug|Sep|Oct|Nov|Dec)\s+\d{1,2},\s+\d{4}|\d+\s+(?:second|minute|hour|day|week|month|year)s?\s+ago)/gi;
306
- const dateMatch = htmlString.match(datePattern);
307
-
308
- if (dateMatch) {
309
- // Extract just the date part using two possible patterns
310
- const absoluteDatePattern =
311
- /(?:Jan|Feb|Mar|Apr|May|Jun|Jul|Aug|Sep|Oct|Nov|Dec)\s+\d{1,2},\s+\d{4}/;
312
- const relativeDatePattern =
313
- /\d+\s+(?:second|minute|hour|day|week|month|year)s?\s+ago/i;
314
-
315
- const absoluteMatch = dateMatch[0].match(absoluteDatePattern);
316
- const relativeMatch = dateMatch[0].match(relativeDatePattern);
317
-
318
- if (absoluteMatch) {
319
- result.date = absoluteMatch[0];
320
- } else if (relativeMatch) {
321
- const relative = relativeMatch[0];
322
- const [amount, unit] = relative.split(" ");
323
-
324
- // Convert relative date to absolute date
325
- const now = new Date();
326
- const date = new Date(now);
327
-
328
- switch (unit?.toLowerCase()) {
329
- case "second":
330
- case "seconds":
331
- date.setSeconds(now.getSeconds() - parseInt(amount));
332
- break;
333
- case "minute":
334
- case "minutes":
335
- date.setMinutes(now.getMinutes() - parseInt(amount));
336
- break;
337
- case "hour":
338
- case "hours":
339
- date.setHours(now.getHours() - parseInt(amount));
340
- break;
341
- case "day":
342
- case "days":
343
- date.setDate(now.getDate() - parseInt(amount));
344
- break;
345
- case "week":
346
- case "weeks":
347
- date.setDate(now.getDate() - parseInt(amount) * 7);
348
- break;
349
- case "month":
350
- case "months":
351
- date.setMonth(now.getMonth() - parseInt(amount));
352
- break;
353
- case "year":
354
- case "years":
355
- date.setFullYear(now.getFullYear() - parseInt(amount));
356
- break;
357
- }
358
-
359
- // Format the date in YouTube style (MMM DD, YYYY)
360
- const months = [
361
- "Jan",
362
- "Feb",
363
- "Mar",
364
- "Apr",
365
- "May",
366
- "Jun",
367
- "Jul",
368
- "Aug",
369
- "Sep",
370
- "Oct",
371
- "Nov",
372
- "Dec",
373
- ];
374
- const formatted = `${months[date.getMonth()]} ${date.getDate()}, ${date.getFullYear()}`;
375
- result.date = formatted;
376
- }
377
- } else {
378
- result.date = null;
379
- }
380
-
381
- // Extract title
382
- const titlePattern = /"title":"([^"]+)"/;
383
- const titleMatch = htmlString.match(titlePattern);
384
- result.title = titleMatch ? titleMatch[1] : null;
385
-
386
- // Extract author/channel name
387
- const authorPattern = /"author":"([^"]+)"/;
388
- const authorMatch = htmlString.match(authorPattern);
389
- result.author_cite = authorMatch ? authorMatch[1] : null;
390
-
391
- // Extract length in seconds (bonus)
392
- const lengthPattern = /"lengthSeconds":"(\d+)"/;
393
- const lengthMatch = htmlString.match(lengthPattern);
394
- result.length = lengthMatch ? parseInt(lengthMatch[1]) : null;
395
-
396
- return result;
397
- }
398
-
399
- async function fetchTranscriptOfficialYoutube(videoId, options = {}) {
400
- let videoPageBody = await scrapeURL(
401
- `https://www.youtube.com/watch?v=${videoId}`,
402
- options,
403
- );
404
-
405
- videoPageBody = videoPageBody?.data;
406
- // if (videoPageBody?.error) return { error: 1 };
407
- // if (
408
- // videoPageBody?.includes('class="g-recaptcha"') ||
409
- // !videoPageBody?.includes('"playabilityStatus":')
410
- // )
411
- // return { error: 1 };
412
-
413
- var videoObj = videoPageBody
414
- .replace("\n", "")
415
- .split('"captions":')?.[1]
416
- ?.split(',"videoDetails')[0];
417
-
418
- if (!videoObj) return { error: 2 };
419
-
420
- const captions = JSON.parse(videoObj)?.playerCaptionsTracklistRenderer;
421
-
422
- if (!captions?.captionTracks) return { error: 3 };
423
-
424
- const track = captions.captionTracks.find(
425
- (track) => track.languageCode === "en",
426
- );
427
-
428
- if (!track) return { error: 4 };
429
-
430
- //TODO: implement youtube-po-token-generator";
431
- // const { poToken } = await generate();
432
- const { poToken } = 1;
433
- let transcriptURL =
434
- track.baseUrl.replaceAll(",", "%2C") +
435
- "&potc=1&pot=" +
436
- encodeURIComponent(poToken.replace(/=/g, "%3D")) +
437
- "&fmt=json3&xorb=2&xobt=3&xovt=3&cbr=Chrome&cbrver=144.0.0.0&c=WEB&cver=2.20260312.08.00&cplayer=UNIPLAYER&cos=X11&cplatform=DESKTOP";
438
- console.log(transcriptURL);
439
-
440
- const transcriptBody = await scrapeURL(transcriptURL, options);
441
-
442
- console.log(transcriptBody);
443
- return;
444
- if (transcriptBody.error) return { error: true };
445
-
446
- const results = [
447
- ...transcriptBody.data.matchAll(
448
- /<text start="([^"]*)" dur="([^"]*)">([^<]*)<\/text>/g,
449
- ),
450
- ];
451
-
452
- var transcript = results.map(([, start, duration, text]) => ({
453
- text,
454
- duration: parseFloat(duration),
455
- offset: parseFloat(start),
456
- lang: track.languageCode,
457
- }));
458
-
459
- var content = "";
460
- var timestamps = [];
461
- transcript.forEach(({ offset, text }) => {
462
- timestamps.push([content.length, Math.floor(offset, 0)]);
463
-
464
- content += text + " ";
465
- });
466
-
467
- return { content, timestamps };
468
- }