extract-webpage 1.2.48 → 1.2.50
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/extract-webpage.cjs.js +1 -1
- package/dist/extract-webpage.cjs.js.map +1 -1
- package/dist/extract-webpage.es.js +3 -3
- package/dist/extract-webpage.es.js.map +1 -1
- package/dist/index.d.ts +1 -1
- package/package.json +4 -4
- package/src/index.ts +1 -1
- package/dist/url-to-content/youtube-to-text.d.ts +0 -70
- package/src/url-to-content/youtube-to-text.ts +0 -468
package/dist/index.d.ts
CHANGED
|
@@ -16,7 +16,7 @@ export * from './tokenize/text-to-chunks';
|
|
|
16
16
|
export * from './url-to-content/url-to-content';
|
|
17
17
|
export * from './url-to-content/url-to-html';
|
|
18
18
|
export * from './html-to-cite/url-to-domain';
|
|
19
|
-
export * from './url-to-content/youtube-
|
|
19
|
+
export * from './url-to-content/youtube-helpers';
|
|
20
20
|
export * from './url-to-content/docx-to-content';
|
|
21
21
|
export * from './html-to-content/html-to-content';
|
|
22
22
|
export * from './html-to-content/extract-content/extract-content-readability';
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "extract-webpage",
|
|
3
|
-
"version": "1.2.
|
|
3
|
+
"version": "1.2.50",
|
|
4
4
|
"module": "./dist/extract-webpage.es.js",
|
|
5
5
|
"description": "Search, extract, cite, and outline the web for a topic with AI Research Agent.",
|
|
6
6
|
"author": "vtempest <grokthiscontact@gmail.com>",
|
|
@@ -78,11 +78,11 @@
|
|
|
78
78
|
"dependencies": {
|
|
79
79
|
"@huggingface/transformers": "^3.8.1",
|
|
80
80
|
"ai": "^5.0.0",
|
|
81
|
-
"chat-agent-toolkit": "^1.2.
|
|
81
|
+
"chat-agent-toolkit": "^1.2.49",
|
|
82
82
|
"chrono-node": "^2.9.0",
|
|
83
83
|
"drizzle-orm": "^0.45.1",
|
|
84
|
-
"extract-pdf": "^0.1.
|
|
85
|
-
"extract-youtube": "^1.0.
|
|
84
|
+
"extract-pdf": "^0.1.37",
|
|
85
|
+
"extract-youtube": "^1.0.36",
|
|
86
86
|
"highlight.js": "^11.11.1",
|
|
87
87
|
"html-entities": "^2.6.0",
|
|
88
88
|
"js-yaml": "^4.1.1",
|
package/src/index.ts
CHANGED
|
@@ -17,7 +17,7 @@ export * from "./tokenize/text-to-chunks";
|
|
|
17
17
|
export * from "./url-to-content/url-to-content";
|
|
18
18
|
export * from "./url-to-content/url-to-html";
|
|
19
19
|
export * from "./html-to-cite/url-to-domain";
|
|
20
|
-
export * from "./url-to-content/youtube-
|
|
20
|
+
export * from "./url-to-content/youtube-helpers";
|
|
21
21
|
// PDF export removed from main index to prevent pdfjs-serverless from being evaluated at build time
|
|
22
22
|
// Import directly from "./pdf-to-html/pdfToHtml" when needed
|
|
23
23
|
export * from "./url-to-content/docx-to-content";
|
|
@@ -1,70 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* Fetch youtube.com video's webpage HTML for embedded transcript.
|
|
3
|
-
* If blocked, use scraper of alternative sites providing transcripts.
|
|
4
|
-
* @param {string} videoUrl
|
|
5
|
-
* @param {Object} [options]
|
|
6
|
-
* @param {boolean} options.addTimestamps default=true -
|
|
7
|
-
* true to return timestamps, default true
|
|
8
|
-
* @param {boolean} options.timeout default=5 - http request timeout
|
|
9
|
-
* @return {{content: string, timestamps: string, word_count: number}}
|
|
10
|
-
* where content is the full text of the transcript,
|
|
11
|
-
* timestamps is a string of comma-separated [characterIndex, timeSeconds] pairs,
|
|
12
|
-
* and word_count is the number of words in the transcript.
|
|
13
|
-
* @category Extract
|
|
14
|
-
* @author [vtempest (2025)](https://github.com/vtempest)
|
|
15
|
-
*/
|
|
16
|
-
export declare function convertYoutubeToText(videoUrl: any, options?: {}): Promise<{
|
|
17
|
-
html: any;
|
|
18
|
-
word_count: any;
|
|
19
|
-
source: string;
|
|
20
|
-
date: any;
|
|
21
|
-
title: any;
|
|
22
|
-
author_cite: any;
|
|
23
|
-
length: number;
|
|
24
|
-
}>;
|
|
25
|
-
/**
|
|
26
|
-
* Test if URL is to youtube video and return video id if true
|
|
27
|
-
* @param {string} url - youtube video URL
|
|
28
|
-
* @returns {string|boolean} video ID or false
|
|
29
|
-
* @private
|
|
30
|
-
*/
|
|
31
|
-
export declare function getURLYoutubeVideo(url: any): any;
|
|
32
|
-
/**
|
|
33
|
-
* Fetch-based scraper of youtubetotranscript.com
|
|
34
|
-
* @returns {Object} content, timestamps - where content is the full text of
|
|
35
|
-
* the transcript, and timestamps is an array of [characterIndex, timeSeconds]
|
|
36
|
-
*/
|
|
37
|
-
export declare function fetchViaYoutubeToTranscriptCom(videoId: any, options?: {}): Promise<{
|
|
38
|
-
error: number;
|
|
39
|
-
content?: undefined;
|
|
40
|
-
title?: undefined;
|
|
41
|
-
author_cite?: undefined;
|
|
42
|
-
timestamps?: undefined;
|
|
43
|
-
} | {
|
|
44
|
-
content: string;
|
|
45
|
-
title: any;
|
|
46
|
-
author_cite: any;
|
|
47
|
-
timestamps: any[];
|
|
48
|
-
error?: undefined;
|
|
49
|
-
}>;
|
|
50
|
-
/** ========== NOT WORKING ========== */
|
|
51
|
-
/**
|
|
52
|
-
* Get YouTube transcript of most YouTube videos,
|
|
53
|
-
* except if disabled by uploader
|
|
54
|
-
* fetch-based scraper of youtubetranscript.com
|
|
55
|
-
*
|
|
56
|
-
* @param {string} videoUrl
|
|
57
|
-
* @returns {Object} where content is the full text of
|
|
58
|
-
* the transcript, and timestamps is an array of [characterIndex, timeSeconds]
|
|
59
|
-
* @private
|
|
60
|
-
*/
|
|
61
|
-
export declare function fetchViaYoutubeTranscript(videoId: any, options?: {}): Promise<{
|
|
62
|
-
error: number;
|
|
63
|
-
content?: undefined;
|
|
64
|
-
timestamps?: undefined;
|
|
65
|
-
} | {
|
|
66
|
-
content: string;
|
|
67
|
-
timestamps: any[];
|
|
68
|
-
error?: undefined;
|
|
69
|
-
}>;
|
|
70
|
-
export declare function extractYouTubeInfo(videoId: any, options?: {}): Promise<{}>;
|
|
@@ -1,468 +0,0 @@
|
|
|
1
|
-
// @ts-nocheck
|
|
2
|
-
import { convertURLSafeHTMLToHTML } from "../html-to-content/html-utils";
|
|
3
|
-
import { scrapeURL } from "./url-to-html";
|
|
4
|
-
import grab from "../utils/grab";
|
|
5
|
-
import { decode, encode } from "html-entities";
|
|
6
|
-
/**
|
|
7
|
-
* Fetch youtube.com video's webpage HTML for embedded transcript.
|
|
8
|
-
* If blocked, use scraper of alternative sites providing transcripts.
|
|
9
|
-
* @param {string} videoUrl
|
|
10
|
-
* @param {Object} [options]
|
|
11
|
-
* @param {boolean} options.addTimestamps default=true -
|
|
12
|
-
* true to return timestamps, default true
|
|
13
|
-
* @param {boolean} options.timeout default=5 - http request timeout
|
|
14
|
-
* @return {{content: string, timestamps: string, word_count: number}}
|
|
15
|
-
* where content is the full text of the transcript,
|
|
16
|
-
* timestamps is a string of comma-separated [characterIndex, timeSeconds] pairs,
|
|
17
|
-
* and word_count is the number of words in the transcript.
|
|
18
|
-
* @category Extract
|
|
19
|
-
* @author [vtempest (2025)](https://github.com/vtempest)
|
|
20
|
-
*/
|
|
21
|
-
export async function convertYoutubeToText(videoUrl, options = {}) {
|
|
22
|
-
const {
|
|
23
|
-
addTimestamps = true,
|
|
24
|
-
addPlayer = true,
|
|
25
|
-
timeout = 10,
|
|
26
|
-
proxy = null,
|
|
27
|
-
} = options;
|
|
28
|
-
|
|
29
|
-
var videoId = getURLYoutubeVideo(videoUrl),
|
|
30
|
-
res = {};
|
|
31
|
-
|
|
32
|
-
// res = await fetchTranscriptTactiq(videoId, options);
|
|
33
|
-
|
|
34
|
-
if (!res.content || res.error)
|
|
35
|
-
res = await fetchTranscriptOfficialYoutube(videoId, options);
|
|
36
|
-
|
|
37
|
-
console.log(res);
|
|
38
|
-
|
|
39
|
-
// if (!res.content || res.error)
|
|
40
|
-
// res = await fetchViaYoutubeToTranscriptCom(videoId, options);
|
|
41
|
-
|
|
42
|
-
// if (!res.content || res.error)
|
|
43
|
-
// res = await fetchViaYoutubeTranscript(videoId, options);
|
|
44
|
-
|
|
45
|
-
// var { date, title, author_cite, length } = await extractYouTubeInfo(
|
|
46
|
-
// videoId,
|
|
47
|
-
// options,
|
|
48
|
-
// );
|
|
49
|
-
|
|
50
|
-
// if (!res.content || res.error) return { error: 1 };
|
|
51
|
-
// var { content, timestamps } = res;
|
|
52
|
-
|
|
53
|
-
// var word_count = content.split(" ").length;
|
|
54
|
-
|
|
55
|
-
// content = convertURLSafeHTMLToHTML(content);
|
|
56
|
-
|
|
57
|
-
// console.log(content);
|
|
58
|
-
|
|
59
|
-
return;
|
|
60
|
-
|
|
61
|
-
//timestamp to track characters per second speed at each interval
|
|
62
|
-
var speedsEveryCharPeriod = {};
|
|
63
|
-
const valueCharPeriod = 100;
|
|
64
|
-
|
|
65
|
-
for (var timestamp of timestamps) {
|
|
66
|
-
var [char, time] = timestamp;
|
|
67
|
-
|
|
68
|
-
var speed = Math.floor(char / time) - 10;
|
|
69
|
-
speedsEveryCharPeriod[Math.floor(char / valueCharPeriod)] = speed;
|
|
70
|
-
}
|
|
71
|
-
|
|
72
|
-
var speeds = Object.keys(speedsEveryCharPeriod).map(
|
|
73
|
-
(timeKey) => speedsEveryCharPeriod[timeKey],
|
|
74
|
-
);
|
|
75
|
-
|
|
76
|
-
let compressed = [];
|
|
77
|
-
let compressedCount = [];
|
|
78
|
-
let currentNum = speeds[0];
|
|
79
|
-
let count = 1;
|
|
80
|
-
|
|
81
|
-
for (let i = 1; i < speeds.length; i++) {
|
|
82
|
-
if (speeds[i] === currentNum) {
|
|
83
|
-
count++;
|
|
84
|
-
} else {
|
|
85
|
-
compressed.push(currentNum);
|
|
86
|
-
compressedCount.push(count);
|
|
87
|
-
currentNum = speeds[i];
|
|
88
|
-
count = 1;
|
|
89
|
-
}
|
|
90
|
-
}
|
|
91
|
-
compressed.push(currentNum);
|
|
92
|
-
compressedCount.push(count);
|
|
93
|
-
|
|
94
|
-
var total = 0;
|
|
95
|
-
compressedCount = compressedCount.map((c) => {
|
|
96
|
-
total += c;
|
|
97
|
-
return total;
|
|
98
|
-
});
|
|
99
|
-
|
|
100
|
-
//remove extra spaces
|
|
101
|
-
content = content.replace(/\s+/g, " ");
|
|
102
|
-
|
|
103
|
-
speeds = compressed.join(",") + " " + compressedCount.join(",");
|
|
104
|
-
|
|
105
|
-
if (addPlayer)
|
|
106
|
-
content = `<iframe width="100%" height="315px" data-timestamps="${speeds}"
|
|
107
|
-
src="https://www.youtube.com/embed/${videoId}" frameborder="0"
|
|
108
|
-
allow="accelerometer; autoplay; clipboard-write; encrypted-media;
|
|
109
|
-
gyroscope; picture-in-picture" allowfullscreen></iframe>${content}`;
|
|
110
|
-
|
|
111
|
-
var source = "YouTube";
|
|
112
|
-
|
|
113
|
-
return {
|
|
114
|
-
html: content,
|
|
115
|
-
word_count,
|
|
116
|
-
source,
|
|
117
|
-
date,
|
|
118
|
-
title,
|
|
119
|
-
author_cite,
|
|
120
|
-
length,
|
|
121
|
-
};
|
|
122
|
-
}
|
|
123
|
-
|
|
124
|
-
function decompressTimestampsArray(compressedStr) {
|
|
125
|
-
let decompressed = [];
|
|
126
|
-
let parts = compressedStr.split(",");
|
|
127
|
-
|
|
128
|
-
for (let part of parts) {
|
|
129
|
-
let [num, count] = part.split("x");
|
|
130
|
-
num = parseInt(num);
|
|
131
|
-
count = parseInt(count);
|
|
132
|
-
decompressed.push(...Array(count).fill(num));
|
|
133
|
-
}
|
|
134
|
-
|
|
135
|
-
return decompressed;
|
|
136
|
-
}
|
|
137
|
-
|
|
138
|
-
/**
|
|
139
|
-
* Test if URL is to youtube video and return video id if true
|
|
140
|
-
* @param {string} url - youtube video URL
|
|
141
|
-
* @returns {string|boolean} video ID or false
|
|
142
|
-
* @private
|
|
143
|
-
*/
|
|
144
|
-
export function getURLYoutubeVideo(url) {
|
|
145
|
-
var match = url?.match(
|
|
146
|
-
/(?:\/embed\/|v=|v\/|vi\/|youtu\.be\/|\/v\/|^https?:\/\/(?:www\.)?youtube\.com\/(?:(?:watch)?\?.*v=|(?:embed|v|vi|user)\/))([^#\&\?]*).*/,
|
|
147
|
-
);
|
|
148
|
-
return match ? match[1] : false;
|
|
149
|
-
}
|
|
150
|
-
|
|
151
|
-
/**
|
|
152
|
-
* Fetch-based scraper of youtubetotranscript.com
|
|
153
|
-
* @returns {Object} content, timestamps - where content is the full text of
|
|
154
|
-
* the transcript, and timestamps is an array of [characterIndex, timeSeconds]
|
|
155
|
-
*/
|
|
156
|
-
export async function fetchViaYoutubeToTranscriptCom(videoId, options = {}) {
|
|
157
|
-
// try {
|
|
158
|
-
const url = `https://youtubetotranscript.com/transcript?v=${videoId}¤t_language_code=en`;
|
|
159
|
-
|
|
160
|
-
var html = await scrapeURL(url, options);
|
|
161
|
-
if (html.error) return { error: 1 };
|
|
162
|
-
|
|
163
|
-
if (!html || html.error) return { error: 1 };
|
|
164
|
-
|
|
165
|
-
//remove line breaks
|
|
166
|
-
html = html?.replace(/[\r\n]/gi, " ");
|
|
167
|
-
// Title regex
|
|
168
|
-
const titleRegex = /<h1[^>]*>([^<]+)<\/h1>/gi;
|
|
169
|
-
var title = html?.match(titleRegex)?.[1];
|
|
170
|
-
//extract title between h1 tags
|
|
171
|
-
title = title
|
|
172
|
-
?.replace(/<[^>]*>/g, "")
|
|
173
|
-
?.replace("Transcript of ", "")
|
|
174
|
-
?.trim();
|
|
175
|
-
|
|
176
|
-
// Author regex with "Author :" prefix
|
|
177
|
-
|
|
178
|
-
const authorRegex = /Author\s*:\s*<a[\s\S]*?>\s*(.*?)\s*<\/a\s*>/;
|
|
179
|
-
var author_cite = html.match(authorRegex)?.[1];
|
|
180
|
-
|
|
181
|
-
const transcriptRegex =
|
|
182
|
-
/<span[^>]*?data-start="([\d.]+)"[^>]*?class="transcript-segment"[^>]*?>[\s\n]*((?:(?!<\/span>).|\n)*?)[\s\n]*<\/span>/gms;
|
|
183
|
-
|
|
184
|
-
const matches = Array.from(html.matchAll(transcriptRegex));
|
|
185
|
-
|
|
186
|
-
const transcript = matches.map((match) => ({
|
|
187
|
-
text: match[2]?.replace(/<br\s*\/?>/gi, " ")?.trim(),
|
|
188
|
-
offset: parseFloat(match[1]),
|
|
189
|
-
}));
|
|
190
|
-
|
|
191
|
-
const content = transcript.map((item) => item.text).join(" ");
|
|
192
|
-
let timestamps = [];
|
|
193
|
-
let charIndex = 0;
|
|
194
|
-
|
|
195
|
-
transcript.forEach((item) => {
|
|
196
|
-
timestamps.push([charIndex, Math.floor(item.offset)]);
|
|
197
|
-
charIndex += item.text.length + 1; // +1 for the space we added
|
|
198
|
-
});
|
|
199
|
-
|
|
200
|
-
return { content, title, author_cite, timestamps };
|
|
201
|
-
// } catch (e) {
|
|
202
|
-
// return { error: 1 };
|
|
203
|
-
// }
|
|
204
|
-
}
|
|
205
|
-
|
|
206
|
-
/**
|
|
207
|
-
* Fetches via tactiq api
|
|
208
|
-
* @param {string} videoId
|
|
209
|
-
* @returns
|
|
210
|
-
*/
|
|
211
|
-
async function fetchTranscriptTactiq(videoId, options = {}) {
|
|
212
|
-
try {
|
|
213
|
-
const data = await grab("https://tactiq-apps-prod.tactiq.io/transcript", {
|
|
214
|
-
method: "POST",
|
|
215
|
-
body: JSON.stringify({
|
|
216
|
-
videoUrl: "https://www.youtube.com/watch?v=" + videoId,
|
|
217
|
-
}),
|
|
218
|
-
});
|
|
219
|
-
|
|
220
|
-
if (!data.captions || data.captions.length === 0) {
|
|
221
|
-
return { error: true };
|
|
222
|
-
}
|
|
223
|
-
|
|
224
|
-
let content = "";
|
|
225
|
-
let timestamps = [];
|
|
226
|
-
let currentLength = 0;
|
|
227
|
-
|
|
228
|
-
data.captions.forEach(({ start, dur, text }) => {
|
|
229
|
-
timestamps.push([currentLength, Math.floor(parseFloat(start))]);
|
|
230
|
-
content += text + " ";
|
|
231
|
-
currentLength = content.length;
|
|
232
|
-
});
|
|
233
|
-
|
|
234
|
-
return { content, timestamps };
|
|
235
|
-
} catch (error) {
|
|
236
|
-
console.error("Error fetching transcript:", error);
|
|
237
|
-
return { error: true };
|
|
238
|
-
}
|
|
239
|
-
}
|
|
240
|
-
|
|
241
|
-
/** ========== NOT WORKING ========== */
|
|
242
|
-
|
|
243
|
-
/**
|
|
244
|
-
* Get YouTube transcript of most YouTube videos,
|
|
245
|
-
* except if disabled by uploader
|
|
246
|
-
* fetch-based scraper of youtubetranscript.com
|
|
247
|
-
*
|
|
248
|
-
* @param {string} videoUrl
|
|
249
|
-
* @returns {Object} where content is the full text of
|
|
250
|
-
* the transcript, and timestamps is an array of [characterIndex, timeSeconds]
|
|
251
|
-
* @private
|
|
252
|
-
*/
|
|
253
|
-
export async function fetchViaYoutubeTranscript(videoId, options = {}) {
|
|
254
|
-
const url = "https://youtubetranscript.com/?server_vid2=" + videoId;
|
|
255
|
-
|
|
256
|
-
const html = await grab(url, { responseType: "text" });
|
|
257
|
-
|
|
258
|
-
if (html.error) return { error: 1 };
|
|
259
|
-
|
|
260
|
-
const transcriptRegex =
|
|
261
|
-
/<text start="([\d.]+)" dur="[\d.]+">((?:(?!<\/text>).|\n)*?)<\/text>/gms;
|
|
262
|
-
const matches = Array.from(html.matchAll(transcriptRegex));
|
|
263
|
-
|
|
264
|
-
const transcript = matches.map((match) => ({
|
|
265
|
-
text: match[2],
|
|
266
|
-
offset: parseFloat(match[1]),
|
|
267
|
-
}));
|
|
268
|
-
|
|
269
|
-
const content = transcript.map((item) => item.text).join(" ");
|
|
270
|
-
let timestamps = [];
|
|
271
|
-
let charIndex = 0;
|
|
272
|
-
|
|
273
|
-
if (content.includes("YouTube is currently blocking us from fetching"))
|
|
274
|
-
return { error: 1 };
|
|
275
|
-
|
|
276
|
-
transcript.forEach((item) => {
|
|
277
|
-
timestamps.push([charIndex, Math.floor(item.offset)]);
|
|
278
|
-
charIndex += item.text.length + 1; // +1 for the space we added
|
|
279
|
-
});
|
|
280
|
-
|
|
281
|
-
return { content, timestamps };
|
|
282
|
-
}
|
|
283
|
-
|
|
284
|
-
export async function extractYouTubeInfo(videoId, options = {}) {
|
|
285
|
-
var htmlString = await scrapeURL(
|
|
286
|
-
`https://www.youtube.com/watch?v=${videoId}`,
|
|
287
|
-
options,
|
|
288
|
-
)?.data;
|
|
289
|
-
|
|
290
|
-
// youtube bot limiting
|
|
291
|
-
if (
|
|
292
|
-
htmlString?.error ||
|
|
293
|
-
htmlString?.data.includes('class="g-recaptcha"') ||
|
|
294
|
-
!htmlString?.data.includes('"playabilityStatus":')
|
|
295
|
-
)
|
|
296
|
-
return { error: 1 };
|
|
297
|
-
|
|
298
|
-
// Remove newlines for easier regex matching
|
|
299
|
-
htmlString = htmlString?.replace(/\n/g, "");
|
|
300
|
-
|
|
301
|
-
const result = {};
|
|
302
|
-
|
|
303
|
-
// Extract date
|
|
304
|
-
const datePattern =
|
|
305
|
-
/id="info"[^>]*>(?:.*?)(?:(?:Jan|Feb|Mar|Apr|May|Jun|Jul|Aug|Sep|Oct|Nov|Dec)\s+\d{1,2},\s+\d{4}|\d+\s+(?:second|minute|hour|day|week|month|year)s?\s+ago)/gi;
|
|
306
|
-
const dateMatch = htmlString.match(datePattern);
|
|
307
|
-
|
|
308
|
-
if (dateMatch) {
|
|
309
|
-
// Extract just the date part using two possible patterns
|
|
310
|
-
const absoluteDatePattern =
|
|
311
|
-
/(?:Jan|Feb|Mar|Apr|May|Jun|Jul|Aug|Sep|Oct|Nov|Dec)\s+\d{1,2},\s+\d{4}/;
|
|
312
|
-
const relativeDatePattern =
|
|
313
|
-
/\d+\s+(?:second|minute|hour|day|week|month|year)s?\s+ago/i;
|
|
314
|
-
|
|
315
|
-
const absoluteMatch = dateMatch[0].match(absoluteDatePattern);
|
|
316
|
-
const relativeMatch = dateMatch[0].match(relativeDatePattern);
|
|
317
|
-
|
|
318
|
-
if (absoluteMatch) {
|
|
319
|
-
result.date = absoluteMatch[0];
|
|
320
|
-
} else if (relativeMatch) {
|
|
321
|
-
const relative = relativeMatch[0];
|
|
322
|
-
const [amount, unit] = relative.split(" ");
|
|
323
|
-
|
|
324
|
-
// Convert relative date to absolute date
|
|
325
|
-
const now = new Date();
|
|
326
|
-
const date = new Date(now);
|
|
327
|
-
|
|
328
|
-
switch (unit?.toLowerCase()) {
|
|
329
|
-
case "second":
|
|
330
|
-
case "seconds":
|
|
331
|
-
date.setSeconds(now.getSeconds() - parseInt(amount));
|
|
332
|
-
break;
|
|
333
|
-
case "minute":
|
|
334
|
-
case "minutes":
|
|
335
|
-
date.setMinutes(now.getMinutes() - parseInt(amount));
|
|
336
|
-
break;
|
|
337
|
-
case "hour":
|
|
338
|
-
case "hours":
|
|
339
|
-
date.setHours(now.getHours() - parseInt(amount));
|
|
340
|
-
break;
|
|
341
|
-
case "day":
|
|
342
|
-
case "days":
|
|
343
|
-
date.setDate(now.getDate() - parseInt(amount));
|
|
344
|
-
break;
|
|
345
|
-
case "week":
|
|
346
|
-
case "weeks":
|
|
347
|
-
date.setDate(now.getDate() - parseInt(amount) * 7);
|
|
348
|
-
break;
|
|
349
|
-
case "month":
|
|
350
|
-
case "months":
|
|
351
|
-
date.setMonth(now.getMonth() - parseInt(amount));
|
|
352
|
-
break;
|
|
353
|
-
case "year":
|
|
354
|
-
case "years":
|
|
355
|
-
date.setFullYear(now.getFullYear() - parseInt(amount));
|
|
356
|
-
break;
|
|
357
|
-
}
|
|
358
|
-
|
|
359
|
-
// Format the date in YouTube style (MMM DD, YYYY)
|
|
360
|
-
const months = [
|
|
361
|
-
"Jan",
|
|
362
|
-
"Feb",
|
|
363
|
-
"Mar",
|
|
364
|
-
"Apr",
|
|
365
|
-
"May",
|
|
366
|
-
"Jun",
|
|
367
|
-
"Jul",
|
|
368
|
-
"Aug",
|
|
369
|
-
"Sep",
|
|
370
|
-
"Oct",
|
|
371
|
-
"Nov",
|
|
372
|
-
"Dec",
|
|
373
|
-
];
|
|
374
|
-
const formatted = `${months[date.getMonth()]} ${date.getDate()}, ${date.getFullYear()}`;
|
|
375
|
-
result.date = formatted;
|
|
376
|
-
}
|
|
377
|
-
} else {
|
|
378
|
-
result.date = null;
|
|
379
|
-
}
|
|
380
|
-
|
|
381
|
-
// Extract title
|
|
382
|
-
const titlePattern = /"title":"([^"]+)"/;
|
|
383
|
-
const titleMatch = htmlString.match(titlePattern);
|
|
384
|
-
result.title = titleMatch ? titleMatch[1] : null;
|
|
385
|
-
|
|
386
|
-
// Extract author/channel name
|
|
387
|
-
const authorPattern = /"author":"([^"]+)"/;
|
|
388
|
-
const authorMatch = htmlString.match(authorPattern);
|
|
389
|
-
result.author_cite = authorMatch ? authorMatch[1] : null;
|
|
390
|
-
|
|
391
|
-
// Extract length in seconds (bonus)
|
|
392
|
-
const lengthPattern = /"lengthSeconds":"(\d+)"/;
|
|
393
|
-
const lengthMatch = htmlString.match(lengthPattern);
|
|
394
|
-
result.length = lengthMatch ? parseInt(lengthMatch[1]) : null;
|
|
395
|
-
|
|
396
|
-
return result;
|
|
397
|
-
}
|
|
398
|
-
|
|
399
|
-
async function fetchTranscriptOfficialYoutube(videoId, options = {}) {
|
|
400
|
-
let videoPageBody = await scrapeURL(
|
|
401
|
-
`https://www.youtube.com/watch?v=${videoId}`,
|
|
402
|
-
options,
|
|
403
|
-
);
|
|
404
|
-
|
|
405
|
-
videoPageBody = videoPageBody?.data;
|
|
406
|
-
// if (videoPageBody?.error) return { error: 1 };
|
|
407
|
-
// if (
|
|
408
|
-
// videoPageBody?.includes('class="g-recaptcha"') ||
|
|
409
|
-
// !videoPageBody?.includes('"playabilityStatus":')
|
|
410
|
-
// )
|
|
411
|
-
// return { error: 1 };
|
|
412
|
-
|
|
413
|
-
var videoObj = videoPageBody
|
|
414
|
-
.replace("\n", "")
|
|
415
|
-
.split('"captions":')?.[1]
|
|
416
|
-
?.split(',"videoDetails')[0];
|
|
417
|
-
|
|
418
|
-
if (!videoObj) return { error: 2 };
|
|
419
|
-
|
|
420
|
-
const captions = JSON.parse(videoObj)?.playerCaptionsTracklistRenderer;
|
|
421
|
-
|
|
422
|
-
if (!captions?.captionTracks) return { error: 3 };
|
|
423
|
-
|
|
424
|
-
const track = captions.captionTracks.find(
|
|
425
|
-
(track) => track.languageCode === "en",
|
|
426
|
-
);
|
|
427
|
-
|
|
428
|
-
if (!track) return { error: 4 };
|
|
429
|
-
|
|
430
|
-
//TODO: implement youtube-po-token-generator";
|
|
431
|
-
// const { poToken } = await generate();
|
|
432
|
-
const { poToken } = 1;
|
|
433
|
-
let transcriptURL =
|
|
434
|
-
track.baseUrl.replaceAll(",", "%2C") +
|
|
435
|
-
"&potc=1&pot=" +
|
|
436
|
-
encodeURIComponent(poToken.replace(/=/g, "%3D")) +
|
|
437
|
-
"&fmt=json3&xorb=2&xobt=3&xovt=3&cbr=Chrome&cbrver=144.0.0.0&c=WEB&cver=2.20260312.08.00&cplayer=UNIPLAYER&cos=X11&cplatform=DESKTOP";
|
|
438
|
-
console.log(transcriptURL);
|
|
439
|
-
|
|
440
|
-
const transcriptBody = await scrapeURL(transcriptURL, options);
|
|
441
|
-
|
|
442
|
-
console.log(transcriptBody);
|
|
443
|
-
return;
|
|
444
|
-
if (transcriptBody.error) return { error: true };
|
|
445
|
-
|
|
446
|
-
const results = [
|
|
447
|
-
...transcriptBody.data.matchAll(
|
|
448
|
-
/<text start="([^"]*)" dur="([^"]*)">([^<]*)<\/text>/g,
|
|
449
|
-
),
|
|
450
|
-
];
|
|
451
|
-
|
|
452
|
-
var transcript = results.map(([, start, duration, text]) => ({
|
|
453
|
-
text,
|
|
454
|
-
duration: parseFloat(duration),
|
|
455
|
-
offset: parseFloat(start),
|
|
456
|
-
lang: track.languageCode,
|
|
457
|
-
}));
|
|
458
|
-
|
|
459
|
-
var content = "";
|
|
460
|
-
var timestamps = [];
|
|
461
|
-
transcript.forEach(({ offset, text }) => {
|
|
462
|
-
timestamps.push([content.length, Math.floor(offset, 0)]);
|
|
463
|
-
|
|
464
|
-
content += text + " ";
|
|
465
|
-
});
|
|
466
|
-
|
|
467
|
-
return { content, timestamps };
|
|
468
|
-
}
|