extract-webpage 1.2.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +212 -0
- package/dist/config/env.d.ts +6 -0
- package/dist/config/index.d.ts +23 -0
- package/dist/config/serverRegistry.d.ts +7 -0
- package/dist/config/types.d.ts +4 -0
- package/dist/extract-webpage.cjs.js +2 -0
- package/dist/extract-webpage.cjs.js.map +1 -0
- package/dist/extract-webpage.es.js +5 -0
- package/dist/extract-webpage.es.js.map +1 -0
- package/dist/html-to-cite/extract-author.d.ts +11 -0
- package/dist/html-to-cite/extract-cite.d.ts +33 -0
- package/dist/html-to-cite/extract-date/date-extractors.d.ts +40 -0
- package/dist/html-to-cite/extract-date/date-validators.d.ts +15 -0
- package/dist/html-to-cite/extract-date/extract-date-quick.d.ts +8 -0
- package/dist/html-to-cite/extract-date/extract-date.d.ts +26 -0
- package/dist/html-to-cite/extract-source.d.ts +7 -0
- package/dist/html-to-cite/extract-title.d.ts +11 -0
- package/dist/html-to-cite/human-names-recognize.d.ts +16 -0
- package/dist/html-to-cite/metadata-to-cite.d.ts +12 -0
- package/dist/html-to-cite/url-to-domain.d.ts +20 -0
- package/dist/html-to-content/extract-content/extract-content-mercury-utils.d.ts +27 -0
- package/dist/html-to-content/extract-content/extract-content-mercury.d.ts +61 -0
- package/dist/html-to-content/extract-content/extract-content-readability.d.ts +101 -0
- package/dist/html-to-content/html-to-basic-html.d.ts +36 -0
- package/dist/html-to-content/html-to-content.d.ts +51 -0
- package/dist/html-to-content/html-utils.d.ts +76 -0
- package/dist/index.d.ts +26 -0
- package/dist/search/index.d.ts +14 -0
- package/dist/search/meta-search-agent-reexport.d.ts +8 -0
- package/dist/search/public-searxng.d.ts +47 -0
- package/dist/search/search-web.d.ts +33 -0
- package/dist/search/tavily.d.ts +20 -0
- package/dist/search/url-to-html.d.ts +62 -0
- package/dist/seektopic/fold-keyphrases.d.ts +28 -0
- package/dist/seektopic/ngrams.d.ts +27 -0
- package/dist/seektopic/rank-sentences-keyphrases.d.ts +28 -0
- package/dist/seektopic/seektopic-keyphrases.d.ts +53 -0
- package/dist/seektopic/types.d.ts +86 -0
- package/dist/seektopic/vector-search.d.ts +89 -0
- package/dist/seektopic/weight-keyphrases.d.ts +22 -0
- package/dist/suggest-next-words/autocomplete-ai.d.ts +0 -0
- package/dist/suggest-next-words/autocomplete-search-engines.d.ts +64 -0
- package/dist/tokenize/suggest-complete-word.d.ts +48 -0
- package/dist/tokenize/text-to-chunks.d.ts +48 -0
- package/dist/tokenize/text-to-sentences.d.ts +35 -0
- package/dist/tokenize/text-to-topic-tokens.d.ts +51 -0
- package/dist/tokenize/word-is-ignored.d.ts +12 -0
- package/dist/tokenize/word-to-root-stem.d.ts +16 -0
- package/dist/url-to-content/docx-to-content.d.ts +22 -0
- package/dist/url-to-content/is-url-adult.d.ts +26 -0
- package/dist/url-to-content/url-to-content.d.ts +127 -0
- package/dist/url-to-content/url-to-html.d.ts +60 -0
- package/dist/url-to-content/youtube-helpers.d.ts +23 -0
- package/dist/url-to-content/youtube-to-text.d.ts +70 -0
- package/dist/utils/documents.d.ts +4 -0
- package/dist/utils/grab.d.ts +18 -0
- package/package.json +109 -0
- package/src/config/env.ts +8 -0
- package/src/config/index.ts +233 -0
- package/src/config/serverRegistry.ts +24 -0
- package/src/config/types.ts +17 -0
- package/src/fs-mock.js +22 -0
- package/src/global.d.ts +8 -0
- package/src/html-to-cite/extract-author.ts +125 -0
- package/src/html-to-cite/extract-cite.ts +97 -0
- package/src/html-to-cite/extract-date/date-extractors.ts +484 -0
- package/src/html-to-cite/extract-date/date-validators.ts +191 -0
- package/src/html-to-cite/extract-date/extract-date-quick.ts +184 -0
- package/src/html-to-cite/extract-date/extract-date.ts +1049 -0
- package/src/html-to-cite/extract-source.ts +30 -0
- package/src/html-to-cite/extract-title.ts +78 -0
- package/src/html-to-cite/human-names-92k.json +1 -0
- package/src/html-to-cite/human-names-recognize.ts +396 -0
- package/src/html-to-cite/metadata-to-cite.ts +73 -0
- package/src/html-to-cite/url-to-domain.ts +50 -0
- package/src/html-to-content/extract-content/extract-content-mercury-utils.ts +696 -0
- package/src/html-to-content/extract-content/extract-content-mercury.ts +830 -0
- package/src/html-to-content/extract-content/extract-content-readability.ts +432 -0
- package/src/html-to-content/extract-content/extract-selectors-per-domain.json +3453 -0
- package/src/html-to-content/html-to-basic-html.ts +282 -0
- package/src/html-to-content/html-to-content.ts +97 -0
- package/src/html-to-content/html-utils.ts +398 -0
- package/src/index.ts +29 -0
- package/src/search/__tests__/public-searxng.test.ts +529 -0
- package/src/search/index.ts +43 -0
- package/src/search/meta-search-agent-reexport.ts +38 -0
- package/src/search/public-searxng.ts +470 -0
- package/src/search/search-web.ts +668 -0
- package/src/search/tavily.ts +106 -0
- package/src/search/url-to-html.ts +278 -0
- package/src/seektopic/fold-keyphrases.ts +87 -0
- package/src/seektopic/ngrams.ts +64 -0
- package/src/seektopic/rank-sentences-keyphrases.ts +132 -0
- package/src/seektopic/seektopic-keyphrases.ts +279 -0
- package/src/seektopic/types.ts +92 -0
- package/src/seektopic/vector-search.ts +232 -0
- package/src/seektopic/weight-keyphrases.ts +59 -0
- package/src/suggest-next-words/autocomplete-ai.ts +38 -0
- package/src/suggest-next-words/autocomplete-search-engines.ts +435 -0
- package/src/tokenize/suggest-complete-word.ts +137 -0
- package/src/tokenize/text-to-chunks.ts +150 -0
- package/src/tokenize/text-to-sentences.ts +614 -0
- package/src/tokenize/text-to-topic-tokens.ts +175 -0
- package/src/tokenize/word-is-ignored.ts +53 -0
- package/src/tokenize/word-to-root-stem.ts +151 -0
- package/src/types.d.ts +130 -0
- package/src/url-to-content/.fuse_hidden003bd28a0000000d +332 -0
- package/src/url-to-content/__tests__/url-to-content.test.ts +368 -0
- package/src/url-to-content/__tests__/url-to-html.test.ts +301 -0
- package/src/url-to-content/docx-to-content.ts +702 -0
- package/src/url-to-content/is-url-adult.ts +318 -0
- package/src/url-to-content/url-to-content.ts +367 -0
- package/src/url-to-content/url-to-html.ts +436 -0
- package/src/url-to-content/youtube-helpers.ts +64 -0
- package/src/url-to-content/youtube-to-text.ts +468 -0
- package/src/utils/documents.ts +71 -0
- package/src/utils/grab.ts +51 -0
|
@@ -0,0 +1,468 @@
|
|
|
1
|
+
// @ts-nocheck
|
|
2
|
+
import { convertURLSafeHTMLToHTML } from "../html-to-content/html-utils";
|
|
3
|
+
import { scrapeURL } from "./url-to-html";
|
|
4
|
+
import grab from "../utils/grab";
|
|
5
|
+
import { generate } from "youtube-po-token-generator";
|
|
6
|
+
import { decode, encode } from "html-entities";
|
|
7
|
+
/**
|
|
8
|
+
* Fetch youtube.com video's webpage HTML for embedded transcript.
|
|
9
|
+
* If blocked, use scraper of alternative sites providing transcripts.
|
|
10
|
+
* @param {string} videoUrl
|
|
11
|
+
* @param {Object} [options]
|
|
12
|
+
* @param {boolean} options.addTimestamps default=true -
|
|
13
|
+
* true to return timestamps, default true
|
|
14
|
+
* @param {boolean} options.timeout default=5 - http request timeout
|
|
15
|
+
* @return {{content: string, timestamps: string, word_count: number}}
|
|
16
|
+
* where content is the full text of the transcript,
|
|
17
|
+
* timestamps is a string of comma-separated [characterIndex, timeSeconds] pairs,
|
|
18
|
+
* and word_count is the number of words in the transcript.
|
|
19
|
+
* @category Extract
|
|
20
|
+
* @author [vtempest (2025)](https://github.com/vtempest)
|
|
21
|
+
*/
|
|
22
|
+
export async function convertYoutubeToText(videoUrl, options = {}) {
|
|
23
|
+
const {
|
|
24
|
+
addTimestamps = true,
|
|
25
|
+
addPlayer = true,
|
|
26
|
+
timeout = 10,
|
|
27
|
+
proxy = null,
|
|
28
|
+
} = options;
|
|
29
|
+
|
|
30
|
+
var videoId = getURLYoutubeVideo(videoUrl),
|
|
31
|
+
res = {};
|
|
32
|
+
|
|
33
|
+
// res = await fetchTranscriptTactiq(videoId, options);
|
|
34
|
+
|
|
35
|
+
if (!res.content || res.error)
|
|
36
|
+
res = await fetchTranscriptOfficialYoutube(videoId, options);
|
|
37
|
+
|
|
38
|
+
console.log(res);
|
|
39
|
+
|
|
40
|
+
// if (!res.content || res.error)
|
|
41
|
+
// res = await fetchViaYoutubeToTranscriptCom(videoId, options);
|
|
42
|
+
|
|
43
|
+
// if (!res.content || res.error)
|
|
44
|
+
// res = await fetchViaYoutubeTranscript(videoId, options);
|
|
45
|
+
|
|
46
|
+
// var { date, title, author_cite, length } = await extractYouTubeInfo(
|
|
47
|
+
// videoId,
|
|
48
|
+
// options,
|
|
49
|
+
// );
|
|
50
|
+
|
|
51
|
+
// if (!res.content || res.error) return { error: 1 };
|
|
52
|
+
// var { content, timestamps } = res;
|
|
53
|
+
|
|
54
|
+
// var word_count = content.split(" ").length;
|
|
55
|
+
|
|
56
|
+
// content = convertURLSafeHTMLToHTML(content);
|
|
57
|
+
|
|
58
|
+
// console.log(content);
|
|
59
|
+
|
|
60
|
+
return;
|
|
61
|
+
|
|
62
|
+
//timestamp to track characters per second speed at each interval
|
|
63
|
+
var speedsEveryCharPeriod = {};
|
|
64
|
+
const valueCharPeriod = 100;
|
|
65
|
+
|
|
66
|
+
for (var timestamp of timestamps) {
|
|
67
|
+
var [char, time] = timestamp;
|
|
68
|
+
|
|
69
|
+
var speed = Math.floor(char / time) - 10;
|
|
70
|
+
speedsEveryCharPeriod[Math.floor(char / valueCharPeriod)] = speed;
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
var speeds = Object.keys(speedsEveryCharPeriod).map(
|
|
74
|
+
(timeKey) => speedsEveryCharPeriod[timeKey],
|
|
75
|
+
);
|
|
76
|
+
|
|
77
|
+
let compressed = [];
|
|
78
|
+
let compressedCount = [];
|
|
79
|
+
let currentNum = speeds[0];
|
|
80
|
+
let count = 1;
|
|
81
|
+
|
|
82
|
+
for (let i = 1; i < speeds.length; i++) {
|
|
83
|
+
if (speeds[i] === currentNum) {
|
|
84
|
+
count++;
|
|
85
|
+
} else {
|
|
86
|
+
compressed.push(currentNum);
|
|
87
|
+
compressedCount.push(count);
|
|
88
|
+
currentNum = speeds[i];
|
|
89
|
+
count = 1;
|
|
90
|
+
}
|
|
91
|
+
}
|
|
92
|
+
compressed.push(currentNum);
|
|
93
|
+
compressedCount.push(count);
|
|
94
|
+
|
|
95
|
+
var total = 0;
|
|
96
|
+
compressedCount = compressedCount.map((c) => {
|
|
97
|
+
total += c;
|
|
98
|
+
return total;
|
|
99
|
+
});
|
|
100
|
+
|
|
101
|
+
//remove extra spaces
|
|
102
|
+
content = content.replace(/\s+/g, " ");
|
|
103
|
+
|
|
104
|
+
speeds = compressed.join(",") + " " + compressedCount.join(",");
|
|
105
|
+
|
|
106
|
+
if (addPlayer)
|
|
107
|
+
content = `<iframe width="100%" height="315px" data-timestamps="${speeds}"
|
|
108
|
+
src="https://www.youtube.com/embed/${videoId}" frameborder="0"
|
|
109
|
+
allow="accelerometer; autoplay; clipboard-write; encrypted-media;
|
|
110
|
+
gyroscope; picture-in-picture" allowfullscreen></iframe>${content}`;
|
|
111
|
+
|
|
112
|
+
var source = "YouTube";
|
|
113
|
+
|
|
114
|
+
return {
|
|
115
|
+
html: content,
|
|
116
|
+
word_count,
|
|
117
|
+
source,
|
|
118
|
+
date,
|
|
119
|
+
title,
|
|
120
|
+
author_cite,
|
|
121
|
+
length,
|
|
122
|
+
};
|
|
123
|
+
}
|
|
124
|
+
|
|
125
|
+
function decompressTimestampsArray(compressedStr) {
|
|
126
|
+
let decompressed = [];
|
|
127
|
+
let parts = compressedStr.split(",");
|
|
128
|
+
|
|
129
|
+
for (let part of parts) {
|
|
130
|
+
let [num, count] = part.split("x");
|
|
131
|
+
num = parseInt(num);
|
|
132
|
+
count = parseInt(count);
|
|
133
|
+
decompressed.push(...Array(count).fill(num));
|
|
134
|
+
}
|
|
135
|
+
|
|
136
|
+
return decompressed;
|
|
137
|
+
}
|
|
138
|
+
|
|
139
|
+
/**
|
|
140
|
+
* Test if URL is to youtube video and return video id if true
|
|
141
|
+
* @param {string} url - youtube video URL
|
|
142
|
+
* @returns {string|boolean} video ID or false
|
|
143
|
+
* @private
|
|
144
|
+
*/
|
|
145
|
+
export function getURLYoutubeVideo(url) {
|
|
146
|
+
var match = url?.match(
|
|
147
|
+
/(?:\/embed\/|v=|v\/|vi\/|youtu\.be\/|\/v\/|^https?:\/\/(?:www\.)?youtube\.com\/(?:(?:watch)?\?.*v=|(?:embed|v|vi|user)\/))([^#\&\?]*).*/,
|
|
148
|
+
);
|
|
149
|
+
return match ? match[1] : false;
|
|
150
|
+
}
|
|
151
|
+
|
|
152
|
+
/**
|
|
153
|
+
* Fetch-based scraper of youtubetotranscript.com
|
|
154
|
+
* @returns {Object} content, timestamps - where content is the full text of
|
|
155
|
+
* the transcript, and timestamps is an array of [characterIndex, timeSeconds]
|
|
156
|
+
*/
|
|
157
|
+
export async function fetchViaYoutubeToTranscriptCom(videoId, options = {}) {
|
|
158
|
+
// try {
|
|
159
|
+
const url = `https://youtubetotranscript.com/transcript?v=${videoId}¤t_language_code=en`;
|
|
160
|
+
|
|
161
|
+
var html = await scrapeURL(url, options);
|
|
162
|
+
if (html.error) return { error: 1 };
|
|
163
|
+
|
|
164
|
+
if (!html || html.error) return { error: 1 };
|
|
165
|
+
|
|
166
|
+
//remove line breaks
|
|
167
|
+
html = html?.replace(/[\r\n]/gi, " ");
|
|
168
|
+
// Title regex
|
|
169
|
+
const titleRegex = /<h1[^>]*>([^<]+)<\/h1>/gi;
|
|
170
|
+
var title = html?.match(titleRegex)?.[1];
|
|
171
|
+
//extract title between h1 tags
|
|
172
|
+
title = title
|
|
173
|
+
?.replace(/<[^>]*>/g, "")
|
|
174
|
+
?.replace("Transcript of ", "")
|
|
175
|
+
?.trim();
|
|
176
|
+
|
|
177
|
+
// Author regex with "Author :" prefix
|
|
178
|
+
|
|
179
|
+
const authorRegex = /Author\s*:\s*<a[\s\S]*?>\s*(.*?)\s*<\/a\s*>/;
|
|
180
|
+
var author_cite = html.match(authorRegex)?.[1];
|
|
181
|
+
|
|
182
|
+
const transcriptRegex =
|
|
183
|
+
/<span[^>]*?data-start="([\d.]+)"[^>]*?class="transcript-segment"[^>]*?>[\s\n]*((?:(?!<\/span>).|\n)*?)[\s\n]*<\/span>/gms;
|
|
184
|
+
|
|
185
|
+
const matches = Array.from(html.matchAll(transcriptRegex));
|
|
186
|
+
|
|
187
|
+
const transcript = matches.map((match) => ({
|
|
188
|
+
text: match[2]?.replace(/<br\s*\/?>/gi, " ")?.trim(),
|
|
189
|
+
offset: parseFloat(match[1]),
|
|
190
|
+
}));
|
|
191
|
+
|
|
192
|
+
const content = transcript.map((item) => item.text).join(" ");
|
|
193
|
+
let timestamps = [];
|
|
194
|
+
let charIndex = 0;
|
|
195
|
+
|
|
196
|
+
transcript.forEach((item) => {
|
|
197
|
+
timestamps.push([charIndex, Math.floor(item.offset)]);
|
|
198
|
+
charIndex += item.text.length + 1; // +1 for the space we added
|
|
199
|
+
});
|
|
200
|
+
|
|
201
|
+
return { content, title, author_cite, timestamps };
|
|
202
|
+
// } catch (e) {
|
|
203
|
+
// return { error: 1 };
|
|
204
|
+
// }
|
|
205
|
+
}
|
|
206
|
+
|
|
207
|
+
/**
|
|
208
|
+
* Fetches via tactiq api
|
|
209
|
+
* @param {string} videoId
|
|
210
|
+
* @returns
|
|
211
|
+
*/
|
|
212
|
+
async function fetchTranscriptTactiq(videoId, options = {}) {
|
|
213
|
+
try {
|
|
214
|
+
const data = await grab("https://tactiq-apps-prod.tactiq.io/transcript", {
|
|
215
|
+
method: "POST",
|
|
216
|
+
body: JSON.stringify({
|
|
217
|
+
videoUrl: "https://www.youtube.com/watch?v=" + videoId,
|
|
218
|
+
}),
|
|
219
|
+
});
|
|
220
|
+
|
|
221
|
+
if (!data.captions || data.captions.length === 0) {
|
|
222
|
+
return { error: true };
|
|
223
|
+
}
|
|
224
|
+
|
|
225
|
+
let content = "";
|
|
226
|
+
let timestamps = [];
|
|
227
|
+
let currentLength = 0;
|
|
228
|
+
|
|
229
|
+
data.captions.forEach(({ start, dur, text }) => {
|
|
230
|
+
timestamps.push([currentLength, Math.floor(parseFloat(start))]);
|
|
231
|
+
content += text + " ";
|
|
232
|
+
currentLength = content.length;
|
|
233
|
+
});
|
|
234
|
+
|
|
235
|
+
return { content, timestamps };
|
|
236
|
+
} catch (error) {
|
|
237
|
+
console.error("Error fetching transcript:", error);
|
|
238
|
+
return { error: true };
|
|
239
|
+
}
|
|
240
|
+
}
|
|
241
|
+
|
|
242
|
+
/** ========== NOT WORKING ========== */
|
|
243
|
+
|
|
244
|
+
/**
|
|
245
|
+
* Get YouTube transcript of most YouTube videos,
|
|
246
|
+
* except if disabled by uploader
|
|
247
|
+
* fetch-based scraper of youtubetranscript.com
|
|
248
|
+
*
|
|
249
|
+
* @param {string} videoUrl
|
|
250
|
+
* @returns {Object} where content is the full text of
|
|
251
|
+
* the transcript, and timestamps is an array of [characterIndex, timeSeconds]
|
|
252
|
+
* @private
|
|
253
|
+
*/
|
|
254
|
+
export async function fetchViaYoutubeTranscript(videoId, options = {}) {
|
|
255
|
+
const url = "https://youtubetranscript.com/?server_vid2=" + videoId;
|
|
256
|
+
|
|
257
|
+
const html = await grab(url, { responseType: "text" });
|
|
258
|
+
|
|
259
|
+
if (html.error) return { error: 1 };
|
|
260
|
+
|
|
261
|
+
const transcriptRegex =
|
|
262
|
+
/<text start="([\d.]+)" dur="[\d.]+">((?:(?!<\/text>).|\n)*?)<\/text>/gms;
|
|
263
|
+
const matches = Array.from(html.matchAll(transcriptRegex));
|
|
264
|
+
|
|
265
|
+
const transcript = matches.map((match) => ({
|
|
266
|
+
text: match[2],
|
|
267
|
+
offset: parseFloat(match[1]),
|
|
268
|
+
}));
|
|
269
|
+
|
|
270
|
+
const content = transcript.map((item) => item.text).join(" ");
|
|
271
|
+
let timestamps = [];
|
|
272
|
+
let charIndex = 0;
|
|
273
|
+
|
|
274
|
+
if (content.includes("YouTube is currently blocking us from fetching"))
|
|
275
|
+
return { error: 1 };
|
|
276
|
+
|
|
277
|
+
transcript.forEach((item) => {
|
|
278
|
+
timestamps.push([charIndex, Math.floor(item.offset)]);
|
|
279
|
+
charIndex += item.text.length + 1; // +1 for the space we added
|
|
280
|
+
});
|
|
281
|
+
|
|
282
|
+
return { content, timestamps };
|
|
283
|
+
}
|
|
284
|
+
|
|
285
|
+
export async function extractYouTubeInfo(videoId, options = {}) {
|
|
286
|
+
var htmlString = await scrapeURL(
|
|
287
|
+
`https://www.youtube.com/watch?v=${videoId}`,
|
|
288
|
+
options,
|
|
289
|
+
)?.data;
|
|
290
|
+
|
|
291
|
+
// youtube bot limiting
|
|
292
|
+
if (
|
|
293
|
+
htmlString?.error ||
|
|
294
|
+
htmlString?.data.includes('class="g-recaptcha"') ||
|
|
295
|
+
!htmlString?.data.includes('"playabilityStatus":')
|
|
296
|
+
)
|
|
297
|
+
return { error: 1 };
|
|
298
|
+
|
|
299
|
+
// Remove newlines for easier regex matching
|
|
300
|
+
htmlString = htmlString?.replace(/\n/g, "");
|
|
301
|
+
|
|
302
|
+
const result = {};
|
|
303
|
+
|
|
304
|
+
// Extract date
|
|
305
|
+
const datePattern =
|
|
306
|
+
/id="info"[^>]*>(?:.*?)(?:(?:Jan|Feb|Mar|Apr|May|Jun|Jul|Aug|Sep|Oct|Nov|Dec)\s+\d{1,2},\s+\d{4}|\d+\s+(?:second|minute|hour|day|week|month|year)s?\s+ago)/gi;
|
|
307
|
+
const dateMatch = htmlString.match(datePattern);
|
|
308
|
+
|
|
309
|
+
if (dateMatch) {
|
|
310
|
+
// Extract just the date part using two possible patterns
|
|
311
|
+
const absoluteDatePattern =
|
|
312
|
+
/(?:Jan|Feb|Mar|Apr|May|Jun|Jul|Aug|Sep|Oct|Nov|Dec)\s+\d{1,2},\s+\d{4}/;
|
|
313
|
+
const relativeDatePattern =
|
|
314
|
+
/\d+\s+(?:second|minute|hour|day|week|month|year)s?\s+ago/i;
|
|
315
|
+
|
|
316
|
+
const absoluteMatch = dateMatch[0].match(absoluteDatePattern);
|
|
317
|
+
const relativeMatch = dateMatch[0].match(relativeDatePattern);
|
|
318
|
+
|
|
319
|
+
if (absoluteMatch) {
|
|
320
|
+
result.date = absoluteMatch[0];
|
|
321
|
+
} else if (relativeMatch) {
|
|
322
|
+
const relative = relativeMatch[0];
|
|
323
|
+
const [amount, unit] = relative.split(" ");
|
|
324
|
+
|
|
325
|
+
// Convert relative date to absolute date
|
|
326
|
+
const now = new Date();
|
|
327
|
+
const date = new Date(now);
|
|
328
|
+
|
|
329
|
+
switch (unit?.toLowerCase()) {
|
|
330
|
+
case "second":
|
|
331
|
+
case "seconds":
|
|
332
|
+
date.setSeconds(now.getSeconds() - parseInt(amount));
|
|
333
|
+
break;
|
|
334
|
+
case "minute":
|
|
335
|
+
case "minutes":
|
|
336
|
+
date.setMinutes(now.getMinutes() - parseInt(amount));
|
|
337
|
+
break;
|
|
338
|
+
case "hour":
|
|
339
|
+
case "hours":
|
|
340
|
+
date.setHours(now.getHours() - parseInt(amount));
|
|
341
|
+
break;
|
|
342
|
+
case "day":
|
|
343
|
+
case "days":
|
|
344
|
+
date.setDate(now.getDate() - parseInt(amount));
|
|
345
|
+
break;
|
|
346
|
+
case "week":
|
|
347
|
+
case "weeks":
|
|
348
|
+
date.setDate(now.getDate() - parseInt(amount) * 7);
|
|
349
|
+
break;
|
|
350
|
+
case "month":
|
|
351
|
+
case "months":
|
|
352
|
+
date.setMonth(now.getMonth() - parseInt(amount));
|
|
353
|
+
break;
|
|
354
|
+
case "year":
|
|
355
|
+
case "years":
|
|
356
|
+
date.setFullYear(now.getFullYear() - parseInt(amount));
|
|
357
|
+
break;
|
|
358
|
+
}
|
|
359
|
+
|
|
360
|
+
// Format the date in YouTube style (MMM DD, YYYY)
|
|
361
|
+
const months = [
|
|
362
|
+
"Jan",
|
|
363
|
+
"Feb",
|
|
364
|
+
"Mar",
|
|
365
|
+
"Apr",
|
|
366
|
+
"May",
|
|
367
|
+
"Jun",
|
|
368
|
+
"Jul",
|
|
369
|
+
"Aug",
|
|
370
|
+
"Sep",
|
|
371
|
+
"Oct",
|
|
372
|
+
"Nov",
|
|
373
|
+
"Dec",
|
|
374
|
+
];
|
|
375
|
+
const formatted = `${months[date.getMonth()]} ${date.getDate()}, ${date.getFullYear()}`;
|
|
376
|
+
result.date = formatted;
|
|
377
|
+
}
|
|
378
|
+
} else {
|
|
379
|
+
result.date = null;
|
|
380
|
+
}
|
|
381
|
+
|
|
382
|
+
// Extract title
|
|
383
|
+
const titlePattern = /"title":"([^"]+)"/;
|
|
384
|
+
const titleMatch = htmlString.match(titlePattern);
|
|
385
|
+
result.title = titleMatch ? titleMatch[1] : null;
|
|
386
|
+
|
|
387
|
+
// Extract author/channel name
|
|
388
|
+
const authorPattern = /"author":"([^"]+)"/;
|
|
389
|
+
const authorMatch = htmlString.match(authorPattern);
|
|
390
|
+
result.author_cite = authorMatch ? authorMatch[1] : null;
|
|
391
|
+
|
|
392
|
+
// Extract length in seconds (bonus)
|
|
393
|
+
const lengthPattern = /"lengthSeconds":"(\d+)"/;
|
|
394
|
+
const lengthMatch = htmlString.match(lengthPattern);
|
|
395
|
+
result.length = lengthMatch ? parseInt(lengthMatch[1]) : null;
|
|
396
|
+
|
|
397
|
+
return result;
|
|
398
|
+
}
|
|
399
|
+
|
|
400
|
+
async function fetchTranscriptOfficialYoutube(videoId, options = {}) {
|
|
401
|
+
let videoPageBody = await scrapeURL(
|
|
402
|
+
`https://www.youtube.com/watch?v=${videoId}`,
|
|
403
|
+
options,
|
|
404
|
+
);
|
|
405
|
+
|
|
406
|
+
videoPageBody = videoPageBody?.data;
|
|
407
|
+
// if (videoPageBody?.error) return { error: 1 };
|
|
408
|
+
// if (
|
|
409
|
+
// videoPageBody?.includes('class="g-recaptcha"') ||
|
|
410
|
+
// !videoPageBody?.includes('"playabilityStatus":')
|
|
411
|
+
// )
|
|
412
|
+
// return { error: 1 };
|
|
413
|
+
|
|
414
|
+
var videoObj = videoPageBody
|
|
415
|
+
.replace("\n", "")
|
|
416
|
+
.split('"captions":')?.[1]
|
|
417
|
+
?.split(',"videoDetails')[0];
|
|
418
|
+
|
|
419
|
+
if (!videoObj) return { error: 2 };
|
|
420
|
+
|
|
421
|
+
const captions = JSON.parse(videoObj)?.playerCaptionsTracklistRenderer;
|
|
422
|
+
|
|
423
|
+
if (!captions?.captionTracks) return { error: 3 };
|
|
424
|
+
|
|
425
|
+
const track = captions.captionTracks.find(
|
|
426
|
+
(track) => track.languageCode === "en",
|
|
427
|
+
);
|
|
428
|
+
|
|
429
|
+
if (!track) return { error: 4 };
|
|
430
|
+
|
|
431
|
+
const { poToken } = await generate();
|
|
432
|
+
|
|
433
|
+
let transcriptURL =
|
|
434
|
+
track.baseUrl.replaceAll(",", "%2C") +
|
|
435
|
+
"&potc=1&pot=" +
|
|
436
|
+
encodeURIComponent(poToken.replace(/=/g, "%3D")) +
|
|
437
|
+
"&fmt=json3&xorb=2&xobt=3&xovt=3&cbr=Chrome&cbrver=144.0.0.0&c=WEB&cver=2.20260312.08.00&cplayer=UNIPLAYER&cos=X11&cplatform=DESKTOP";
|
|
438
|
+
console.log(transcriptURL);
|
|
439
|
+
|
|
440
|
+
const transcriptBody = await scrapeURL(transcriptURL, options);
|
|
441
|
+
|
|
442
|
+
console.log(transcriptBody);
|
|
443
|
+
return;
|
|
444
|
+
if (transcriptBody.error) return { error: true };
|
|
445
|
+
|
|
446
|
+
const results = [
|
|
447
|
+
...transcriptBody.data.matchAll(
|
|
448
|
+
/<text start="([^"]*)" dur="([^"]*)">([^<]*)<\/text>/g,
|
|
449
|
+
),
|
|
450
|
+
];
|
|
451
|
+
|
|
452
|
+
var transcript = results.map(([, start, duration, text]) => ({
|
|
453
|
+
text,
|
|
454
|
+
duration: parseFloat(duration),
|
|
455
|
+
offset: parseFloat(start),
|
|
456
|
+
lang: track.languageCode,
|
|
457
|
+
}));
|
|
458
|
+
|
|
459
|
+
var content = "";
|
|
460
|
+
var timestamps = [];
|
|
461
|
+
transcript.forEach(({ offset, text }) => {
|
|
462
|
+
timestamps.push([content.length, Math.floor(offset, 0)]);
|
|
463
|
+
|
|
464
|
+
content += text + " ";
|
|
465
|
+
});
|
|
466
|
+
|
|
467
|
+
return { content, timestamps };
|
|
468
|
+
}
|
|
@@ -0,0 +1,71 @@
|
|
|
1
|
+
import axios from 'axios';
|
|
2
|
+
import { splitTextIntoChunks, type Document } from 'chat-agent-toolkit';
|
|
3
|
+
|
|
4
|
+
/** Strip HTML tags and decode entities \u2014 works in Cloudflare edge runtime */
|
|
5
|
+
function htmlToText(html: string): string {
|
|
6
|
+
return html
|
|
7
|
+
.replace(/<(script|style)[^>]*>[\s\S]*?<\/(script|style)>/gi, ' ')
|
|
8
|
+
.replace(/<[^>]+>/g, ' ')
|
|
9
|
+
.replace(/ /g, ' ')
|
|
10
|
+
.replace(/&/g, '&')
|
|
11
|
+
.replace(/</g, '<')
|
|
12
|
+
.replace(/>/g, '>')
|
|
13
|
+
.replace(/"/g, '"')
|
|
14
|
+
.replace(/'/g, "'")
|
|
15
|
+
.replace(/&[a-z#][a-z0-9]+;/gi, ' ')
|
|
16
|
+
.replace(/\s+/g, ' ')
|
|
17
|
+
.trim();
|
|
18
|
+
}
|
|
19
|
+
|
|
20
|
+
export const getDocumentsFromLinks = async ({ links }: { links: string[] }) => {
|
|
21
|
+
let docs: Document[] = [];
|
|
22
|
+
|
|
23
|
+
await Promise.all(
|
|
24
|
+
links.map(async (link) => {
|
|
25
|
+
link =
|
|
26
|
+
link.startsWith('http://') || link.startsWith('https://')
|
|
27
|
+
? link
|
|
28
|
+
: `https://${link}`;
|
|
29
|
+
|
|
30
|
+
try {
|
|
31
|
+
const res = await axios.get(link, {
|
|
32
|
+
responseType: 'arraybuffer',
|
|
33
|
+
});
|
|
34
|
+
|
|
35
|
+
const parsedText = htmlToText(res.data.toString('utf8'))
|
|
36
|
+
.replace(/(\r\n|\n|\r)/gm, ' ')
|
|
37
|
+
.replace(/\s+/g, ' ')
|
|
38
|
+
.trim();
|
|
39
|
+
|
|
40
|
+
const splittedText = splitTextIntoChunks(parsedText);
|
|
41
|
+
const title = res.data
|
|
42
|
+
.toString('utf8')
|
|
43
|
+
.match(/<title.*>(.*?)<\/title>/)?.[1];
|
|
44
|
+
|
|
45
|
+
const linkDocs: Document[] = splittedText.map((text) => ({
|
|
46
|
+
pageContent: text,
|
|
47
|
+
metadata: {
|
|
48
|
+
title: title || link,
|
|
49
|
+
url: link,
|
|
50
|
+
},
|
|
51
|
+
}));
|
|
52
|
+
|
|
53
|
+
docs.push(...linkDocs);
|
|
54
|
+
} catch (err) {
|
|
55
|
+
console.error(
|
|
56
|
+
'An error occurred while getting documents from links: ',
|
|
57
|
+
err,
|
|
58
|
+
);
|
|
59
|
+
docs.push({
|
|
60
|
+
pageContent: `Failed to retrieve content from the link: ${err}`,
|
|
61
|
+
metadata: {
|
|
62
|
+
title: 'Failed to retrieve content',
|
|
63
|
+
url: link,
|
|
64
|
+
},
|
|
65
|
+
});
|
|
66
|
+
}
|
|
67
|
+
}),
|
|
68
|
+
);
|
|
69
|
+
|
|
70
|
+
return docs;
|
|
71
|
+
};
|
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Fetch wrapper for grabbing binary content
|
|
3
|
+
* Replacement for grab-url package using standard fetch API
|
|
4
|
+
*/
|
|
5
|
+
export interface GrabOptions {
|
|
6
|
+
responseType?: "text" | "arraybuffer";
|
|
7
|
+
/** Timeout in seconds */
|
|
8
|
+
timeout?: number;
|
|
9
|
+
method?: string;
|
|
10
|
+
headers?: Record<string, string>;
|
|
11
|
+
body?: string;
|
|
12
|
+
}
|
|
13
|
+
|
|
14
|
+
export default async function grab(
|
|
15
|
+
url: string,
|
|
16
|
+
options?: GrabOptions & { responseType?: "text" }
|
|
17
|
+
): Promise<string>;
|
|
18
|
+
export default async function grab(
|
|
19
|
+
url: string,
|
|
20
|
+
options: GrabOptions & { responseType: "arraybuffer" }
|
|
21
|
+
): Promise<ArrayBuffer>;
|
|
22
|
+
export default async function grab(
|
|
23
|
+
url: string,
|
|
24
|
+
options: GrabOptions = {}
|
|
25
|
+
): Promise<string | ArrayBuffer> {
|
|
26
|
+
const timeoutMs = options.timeout ? options.timeout * 1000 : 10000;
|
|
27
|
+
const controller = new AbortController();
|
|
28
|
+
const timeoutId = setTimeout(() => controller.abort(), timeoutMs);
|
|
29
|
+
|
|
30
|
+
try {
|
|
31
|
+
const response = await fetch(url, {
|
|
32
|
+
method: options.method,
|
|
33
|
+
headers: options.headers,
|
|
34
|
+
body: options.body,
|
|
35
|
+
signal: controller.signal,
|
|
36
|
+
});
|
|
37
|
+
clearTimeout(timeoutId);
|
|
38
|
+
|
|
39
|
+
if (!response.ok) {
|
|
40
|
+
throw new Error(`HTTP ${response.status}`);
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
if (options.responseType === "arraybuffer") {
|
|
44
|
+
return await response.arrayBuffer();
|
|
45
|
+
}
|
|
46
|
+
return await response.text();
|
|
47
|
+
} catch (error) {
|
|
48
|
+
clearTimeout(timeoutId);
|
|
49
|
+
throw error;
|
|
50
|
+
}
|
|
51
|
+
}
|