umedia 0.3.0 → 0.4.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,177 @@
1
+ import { fetchText, hostMatches } from "../util.js";
2
+ import { Codes, UMediaError } from "../errors.js";
3
+
4
+ /**
5
+ * Snapchat — Spotlight videos and public story snap lists.
6
+ * Primary: the page's __NEXT_DATA__ -> spotlightFeed.spotlightStories, matched
7
+ * by storyId (the exact structure yt-dlp's snapchat extractor reads) ->
8
+ * videoMetadata.contentUrl (direct sc-cdn mp4). Fallback: the preload video
9
+ * link (cobalt). Stories: pageProps.story.snapList — ALL snaps, in order.
10
+ */
11
+ export const name = "snapchat";
12
+
13
+ const NEXT_DATA = /<script id="__NEXT_DATA__" type="application\/json">({.+?})<\/script>/;
14
+ const PRELOAD_VIDEO = /<link data-react-helmet="true" rel="preload" href="([^"]+)" as="video"\/>/;
15
+
16
+ export function canHandle(url) {
17
+ try {
18
+ return hostMatches(new URL(String(url)).hostname, ["snapchat.com", "t.snapchat.com"]);
19
+ } catch { return false; }
20
+ }
21
+
22
+ export function spotlightId(url) {
23
+ const m = String(url).match(/\/spotlight\/([\w-]{10,120})/);
24
+ return m ? m[1] : null;
25
+ }
26
+
27
+ function nextData(html) {
28
+ const m = html.match(NEXT_DATA);
29
+ if (!m) return null;
30
+ try { return JSON.parse(m[1]); } catch { return null; }
31
+ }
32
+
33
+ function pushStorySnap(snap, media) {
34
+ const isPhoto = snap.snapMediaType === 0;
35
+ const u = snap.snapUrls?.mediaUrl;
36
+ if (!u) return;
37
+ media.push({
38
+ type: isPhoto ? "image" : "video", index: media.length, url: u,
39
+ thumbnail: snap.snapUrls?.mediaPreviewUrl?.value || u,
40
+ mimeType: isPhoto ? "image/jpeg" : "video/mp4",
41
+ width: null, height: null,
42
+ duration: snap.durationInMs ? Math.round(snap.durationInMs / 1000) : (snap.duration ? Math.round(snap.duration / 1000) : null),
43
+ size: null, quality: null,
44
+ hasAudio: !isPhoto, hasVideo: !isPhoto, source: name, variants: [],
45
+ });
46
+ }
47
+
48
+ export async function resolve(url, ctx = {}) {
49
+ const attempt = { provider: name, status: "skipped" };
50
+ ctx.attempts?.push(attempt);
51
+ const t0 = Date.now();
52
+
53
+ let target = String(url);
54
+ if (/t\.snapchat\.com\//i.test(target)) {
55
+ try {
56
+ const res = await fetch(target, { redirect: "follow", headers: { "user-agent": "Mozilla/5.0" } });
57
+ target = res.url || target;
58
+ } catch { /* keep original */ }
59
+ }
60
+
61
+ const media = [];
62
+ let title = null;
63
+ let author = null;
64
+ let duration = null;
65
+
66
+ const sid = spotlightId(target);
67
+ const story = target.match(/\/add\/([\w.-]+)(?:\/([\w-]+))?/i);
68
+
69
+ if (sid) {
70
+ const r = await fetchText(`https://www.snapchat.com/spotlight/${sid}`, { headers: { "user-agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/126.0 Safari/537.36" } }, 20_000).catch(() => null);
71
+ const html = r?.text || "";
72
+
73
+ // primary: __NEXT_DATA__ spotlight feed entry matched by storyId (yt-dlp structure)
74
+ const d = nextData(html);
75
+ const stories = d?.props?.pageProps?.spotlightFeed?.spotlightStories || [];
76
+ const entry = stories.find((s) => s?.story?.storyId?.value === sid);
77
+ const vm = entry?.metadata?.videoMetadata;
78
+ if (vm?.contentUrl) {
79
+ media.push({
80
+ type: "video", index: 0, url: vm.contentUrl,
81
+ thumbnail: vm.thumbnailUrl || null,
82
+ mimeType: "video/mp4",
83
+ width: vm.width ?? null, height: vm.height ?? null,
84
+ duration: vm.durationMs ? Math.round(vm.durationMs / 1000) : null,
85
+ size: null, quality: vm.height ? `${vm.height}p` : null,
86
+ hasAudio: true, hasVideo: true, source: name, variants: [],
87
+ });
88
+ title = vm.name || null;
89
+ author = vm.creator?.personCreator?.username || vm.creator?.username || null;
90
+ duration = vm.durationMs ? Math.round(vm.durationMs / 1000) : null;
91
+ }
92
+
93
+ // fallback: the preload video link (cobalt's route)
94
+ if (!media.length) {
95
+ const v = html.match(PRELOAD_VIDEO)?.[1];
96
+ if (v && /^https?:/.test(v) && new URL(v).hostname.endsWith("sc-cdn.net")) {
97
+ media.push({
98
+ type: "video", index: 0, url: v, thumbnail: null,
99
+ mimeType: "video/mp4", width: null, height: null,
100
+ duration: null, size: null, quality: null,
101
+ hasAudio: true, hasVideo: true, source: name, variants: [],
102
+ });
103
+ }
104
+ }
105
+ } else if (story) {
106
+ const r = await fetchText(`https://www.snapchat.com/add/${story[1]}${story[2] ? "/" + story[2] : ""}`,
107
+ { headers: { "user-agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/126.0 Safari/537.36" } }, 20_000).catch(() => null);
108
+ const d = nextData(r?.text || "");
109
+ const pp = d?.props?.pageProps || {};
110
+ author = story[1];
111
+
112
+ const activeStory = pp.story?.snapList || [];
113
+ const highlights = pp.spotlightHighlights || [];
114
+ const curated = pp.curatedHighlights || [];
115
+
116
+ if (story[2]) {
117
+ // a single snap (by snapId) or one highlight reel (by storyId/highlightId)
118
+ const wanted = story[2];
119
+ const allLists = [activeStory, ...curated.map((c) => c.snapList || []), ...highlights.map((h) => h.snapList || [])];
120
+ for (const list of allLists) {
121
+ const snap = list.find((s) => s.snapId?.value === wanted || s.snapId === wanted);
122
+ if (snap) { pushStorySnap(snap, media); break; }
123
+ }
124
+ if (!media.length) {
125
+ const hl = highlights.find((h) => h.storyId === wanted || h.highlightId === wanted || h.storyShareId === wanted);
126
+ if (hl) {
127
+ title = hl.storyTitle || null;
128
+ (hl.snapList || []).forEach((snap) => pushStorySnap(snap, media));
129
+ }
130
+ }
131
+ }
132
+ if (!media.length) {
133
+ // bare /add/{user}: the current story if it exists, else their public
134
+ // highlight reels — ALL snaps, in order, nothing silently dropped
135
+ const list = activeStory.length
136
+ ? activeStory
137
+ : (curated[0]?.snapList?.length ? curated[0].snapList : highlights.flatMap((h) => h.snapList || []));
138
+ if (!activeStory.length && highlights.length && !curated[0]?.snapList?.length) {
139
+ title = `${author} — ${highlights.length} public highlight${highlights.length > 1 ? "s" : ""}`;
140
+ }
141
+ list.forEach((snap) => pushStorySnap(snap, media));
142
+ }
143
+ }
144
+
145
+ if (!media.length) {
146
+ attempt.status = "failed";
147
+ attempt.error = "no snap media exposed (removed, private, or an expired spotlight)";
148
+ throw new UMediaError(Codes.MEDIA_NOT_FOUND,
149
+ `Snapchat exposes no media for this URL (${attempt.error}) — public spotlight links and stories work; removed snaps are gone from Snapchat itself.`, { attempts: [attempt] });
150
+ }
151
+
152
+ attempt.status = "success";
153
+ attempt.latencyMs = Date.now() - t0;
154
+ return {
155
+ result: {
156
+ sourceUrl: url,
157
+ platform: name,
158
+ title: title || (author ? `Snapchat story @${author}` : null),
159
+ author,
160
+ thumbnail: media[0].thumbnail,
161
+ itemCount: media.length,
162
+ counts: {
163
+ images: media.filter((m) => m.type === "image").length,
164
+ videos: media.filter((m) => m.type === "video").length,
165
+ audios: 0, other: 0,
166
+ },
167
+ truncated: false,
168
+ media,
169
+ streamUrlsAvailable: true,
170
+ },
171
+ engine: { attempts: ctx.attempts || [attempt], finalProvider: name, totalLatencyMs: Date.now() - t0 },
172
+ };
173
+ }
174
+
175
+ export async function search() {
176
+ return [];
177
+ }
@@ -0,0 +1,103 @@
1
+ import { fetchText, hostMatches } from "../util.js";
2
+ import { Codes, UMediaError } from "../errors.js";
3
+
4
+ /**
5
+ * Threads (Meta) — the share page's server-rendered state (`video_url` /
6
+ * `image_urls` in the post JSON) plus official og: meta. Best-effort like all
7
+ * Meta surfaces (login-gated on some networks); yt-dlp tier covers gated ones.
8
+ */
9
+ export const name = "threads";
10
+
11
+ export function canHandle(url) {
12
+ try {
13
+ return hostMatches(new URL(String(url)).hostname, ["threads.net", "threads.com"]);
14
+ } catch { return false; }
15
+ }
16
+
17
+ export async function resolve(url, ctx = {}) {
18
+ const attempt = { provider: name, status: "skipped" };
19
+ ctx.attempts?.push(attempt);
20
+ const t0 = Date.now();
21
+
22
+ let html;
23
+ try {
24
+ const r = await fetchText(String(url), {
25
+ headers: {
26
+ "user-agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/126.0 Safari/537.36",
27
+ accept: "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
28
+ "accept-language": "en-US,en;q=0.9",
29
+ },
30
+ }, 18_000);
31
+ html = r.text;
32
+ } catch (e) {
33
+ attempt.status = "failed";
34
+ attempt.error = e.message;
35
+ throw new UMediaError(Codes.PROVIDER_UNAVAILABLE, `Threads page not retrievable. ${e.message}`, { attempts: [attempt] });
36
+ }
37
+
38
+ const un = (s) => { try { return JSON.parse(`"${s}"`); } catch { return s.replace(/\\\//g, "/"); } };
39
+ const media = [];
40
+
41
+ // post state: video_url + display_url (escaped JSON strings inside the SSR payload)
42
+ const video = html.match(/\\"video_url\\":\\"((?:[^"\\]|\\.)+?)\\"/) || html.match(/"video_url":"((?:[^"\\]|\\.)+?)"/);
43
+ const image = html.match(/\\"display_url\\":\\"((?:[^"\\]|\\.)+?)\\"/) || html.match(/"display_url":"((?:[^"\\]|\\.)+?)"/);
44
+ if (video?.[1]) {
45
+ const u = un(video[1]);
46
+ if (/^https?:/.test(u)) media.push({
47
+ type: "video", index: 0, url: u, thumbnail: image ? un(image[1]) : u,
48
+ mimeType: "video/mp4", width: null, height: null, duration: null, size: null,
49
+ quality: null, hasAudio: true, hasVideo: true, source: name, variants: [],
50
+ });
51
+ }
52
+ if (!media.length && image?.[1]) {
53
+ const u = un(image[1]);
54
+ if (/^https?:/.test(u)) media.push({
55
+ type: "image", index: 0, url: u, thumbnail: u,
56
+ mimeType: "image/jpeg", width: null, height: null, duration: null, size: null,
57
+ quality: null, hasAudio: false, hasVideo: false, source: name, variants: [],
58
+ });
59
+ }
60
+
61
+ // og: meta fallback
62
+ if (!media.length) {
63
+ const ogv = (html.match(/property="og:video"[^>]+content="([^"]+)"/) || html.match(/content="([^"]+)"[^>]+property="og:video"/) || [])[1];
64
+ const ogi = (html.match(/property="og:image"[^>]+content="([^"]+)"/) || html.match(/content="([^"]+)"[^>]+property="og:image"/) || [])[1];
65
+ const u = ogv || ogi;
66
+ if (u && /^https?:/.test(u)) media.push({
67
+ type: ogv ? "video" : "image", index: 0, url: u.replace(/&amp;/g, "&"), thumbnail: (ogi || u).replace(/&amp;/g, "&"),
68
+ mimeType: ogv ? "video/mp4" : "image/jpeg", width: null, height: null,
69
+ duration: null, size: null, quality: null,
70
+ hasAudio: Boolean(ogv), hasVideo: Boolean(ogv), source: name, variants: [],
71
+ });
72
+ }
73
+
74
+ if (!media.length) {
75
+ attempt.status = "failed";
76
+ attempt.error = "no media in the public page (login-gated or removed)";
77
+ throw new UMediaError(Codes.MEDIA_NOT_FOUND,
78
+ `Threads exposes no media for this URL (${attempt.error}) — the yt-dlp tier covers gated posts.`, { attempts: [attempt] });
79
+ }
80
+
81
+ const title = (html.match(/"caption":"((?:[^"\\]|\\.){1,200})"/) || html.match(/<title[^>]*>([^<]*)/i) || [])[1] || null;
82
+ attempt.status = "success";
83
+ attempt.latencyMs = Date.now() - t0;
84
+ return {
85
+ result: {
86
+ sourceUrl: url,
87
+ platform: name,
88
+ title: title ? un(title).slice(0, 200) : null,
89
+ author: (String(url).match(/@([\w.]+)/) || [])[1] || null,
90
+ thumbnail: media[0].thumbnail,
91
+ itemCount: media.length,
92
+ counts: { images: media.filter((m) => m.type === "image").length, videos: media.filter((m) => m.type === "video").length, audios: 0, other: 0 },
93
+ truncated: false,
94
+ media,
95
+ streamUrlsAvailable: true,
96
+ },
97
+ engine: { attempts: ctx.attempts || [attempt], finalProvider: name, totalLatencyMs: Date.now() - t0 },
98
+ };
99
+ }
100
+
101
+ export async function search() {
102
+ return [];
103
+ }
@@ -0,0 +1,153 @@
1
+ import { fetchJson, hostMatches } from "../util.js";
2
+ import { Codes, UMediaError } from "../errors.js";
3
+
4
+ /**
5
+ * Tumblr — the public mobile API (api-http2.tumblr.com, the same endpoint the
6
+ * official iPhone app calls with its public api_key). Videos, audio, photos —
7
+ * including reblog trails. Reference: cobalt tumblr service.
8
+ */
9
+ export const name = "tumblr";
10
+
11
+ const API_KEY = "jrsCWX1XDuVxAFO4GkK147syAoN8BJZ5voz8tS80bPcj26Vc5Z";
12
+ const API_BASE = "https://api-http2.tumblr.com";
13
+ const MOBILE = {
14
+ "user-agent": "Tumblr/iPhone/33.3/333010/17.3.1/tumblr",
15
+ "x-version": "iPhone/33.3/333010/17.3.1/tumblr",
16
+ };
17
+
18
+ export function canHandle(url) {
19
+ try {
20
+ return hostMatches(new URL(String(url)).hostname, ["tumblr.com"]);
21
+ } catch { return false; }
22
+ }
23
+
24
+ export function parseUrl(url) {
25
+ const u = new URL(String(url));
26
+ const host = u.hostname.replace(/^www\./, "");
27
+ let m;
28
+ // https://{user}.tumblr.com/post/{id}/...
29
+ m = u.pathname.match(/^\/post\/(\d+)/);
30
+ if (m && host !== "tumblr.com") return { user: host.split(".")[0], id: m[1] };
31
+ // https://www.tumblr.com/{user}/{id} or /blog/view/{user}/{id}
32
+ m = u.pathname.match(/^\/(?:blog\/view\/)?([\w.-]+)\/(\d+)/);
33
+ if (m) return { user: m[1], id: m[2] };
34
+ return null;
35
+ }
36
+
37
+ export async function resolve(url, ctx = {}) {
38
+ const attempt = { provider: name, status: "skipped" };
39
+ ctx.attempts?.push(attempt);
40
+ const t0 = Date.now();
41
+
42
+ const parsed = parseUrl(url);
43
+ if (!parsed) {
44
+ attempt.status = "failed";
45
+ throw new UMediaError(Codes.INVALID_URL, "Not a Tumblr post URL (expected {user}.tumblr.com/post/{id} or tumblr.com/{user}/{id})", { attempts: [attempt] });
46
+ }
47
+
48
+ let data;
49
+ try {
50
+ data = await fetchJson(
51
+ `${API_BASE}/v2/blog/${encodeURIComponent(parsed.user)}.tumblr.com/posts/${parsed.id}/permalink?api_key=${API_KEY}`, { headers: MOBILE }, 20_000);
52
+ } catch (e) {
53
+ attempt.status = "failed";
54
+ attempt.error = e.message;
55
+ throw new UMediaError(Codes.PROVIDER_UNAVAILABLE, `Tumblr API unavailable. ${e.message}`, { attempts: [attempt] });
56
+ }
57
+
58
+ const element = data?.response?.timeline?.elements?.[0];
59
+ if (!element) {
60
+ attempt.status = "failed";
61
+ attempt.error = data?.meta?.msg || "post not found";
62
+ throw new UMediaError(Codes.MEDIA_NOT_FOUND, `Tumblr: ${attempt.error}`, { attempts: [attempt] });
63
+ }
64
+
65
+ // contents: the post itself + any reblog trail, in order — ALL of it
66
+ const contents = [
67
+ ...(element.content || []),
68
+ ...(element.trail || []).flatMap((t) => t.content || []),
69
+ ];
70
+
71
+ const media = [];
72
+ for (const c of contents) {
73
+ if (c.type === "video" && (c.provider === "tumblr" || !c.provider) && c.media?.url) {
74
+ media.push({
75
+ type: "video", index: media.length, url: c.media.url,
76
+ thumbnail: c.poster?.[0]?.url || null,
77
+ mimeType: "video/mp4", width: c.width ?? null, height: c.height ?? null,
78
+ duration: null, size: null, quality: null,
79
+ hasAudio: true, hasVideo: true, source: name, variants: [],
80
+ });
81
+ } else if (c.type === "audio" && c.media?.url) {
82
+ media.push({
83
+ type: "audio", index: media.length, url: c.media.url,
84
+ thumbnail: null, mimeType: "audio/mpeg", width: null, height: null,
85
+ duration: null, size: null, quality: null,
86
+ hasAudio: true, hasVideo: false, source: name, variants: [],
87
+ });
88
+ } else if (c.type === "image" || c.type === "photo") {
89
+ // image content carries media as an ARRAY of renditions — take the best
90
+ // one as the item, keep the rest as variants. never inflate items.
91
+ const items = (Array.isArray(c.media) ? c.media : (c.media?.url ? [c.media] : []))
92
+ .filter((mm) => mm?.url);
93
+ const uniq = [...new Map(items.map((mm) => [mm.url, mm])).values()]
94
+ .sort((a, b) => ((b.width || 0) * (b.height || 0)) - ((a.width || 0) * (a.height || 0)));
95
+ const best = uniq[0];
96
+ if (best) {
97
+ media.push({
98
+ type: "image", index: media.length, url: best.url,
99
+ thumbnail: best.url, mimeType: "image/jpeg",
100
+ width: best.width ?? null, height: best.height ?? null,
101
+ duration: null, size: null, quality: best.width ? `${best.width}px` : null,
102
+ hasAudio: false, hasVideo: false, source: name,
103
+ variants: uniq.map((x) => ({
104
+ height: x.height ?? null, width: x.width ?? null, url: x.url,
105
+ hasAudio: false, hasVideo: false, quality: x.width ? `${x.width}px` : null, mimeType: "image/jpeg",
106
+ })),
107
+ });
108
+ }
109
+ }
110
+ }
111
+
112
+ // reblog trails repeat media from earlier hops — dedupe by URL, keep first order
113
+ const seen = new Set();
114
+ const deduped = [];
115
+ for (const mm of media) {
116
+ if (seen.has(mm.url)) continue;
117
+ seen.add(mm.url);
118
+ deduped.push({ ...mm, index: deduped.length });
119
+ }
120
+ media.length = 0;
121
+ media.push(...deduped);
122
+
123
+ if (!media.length) {
124
+ attempt.status = "failed";
125
+ throw new UMediaError(Codes.MEDIA_NOT_FOUND, "Tumblr post contains no downloadable media (text/link post, or third-party embed — use the yt-dlp tier for YouTube/Vimeo embeds)", { attempts: [attempt] });
126
+ }
127
+
128
+ attempt.status = "success";
129
+ attempt.latencyMs = Date.now() - t0;
130
+ return {
131
+ result: {
132
+ sourceUrl: url,
133
+ platform: name,
134
+ title: (element.summary || element.let || "").slice(0, 200) || null,
135
+ author: element.blog_name || parsed.user,
136
+ thumbnail: media[0].thumbnail,
137
+ itemCount: media.length,
138
+ counts: {
139
+ images: media.filter((m) => m.type === "image").length,
140
+ videos: media.filter((m) => m.type === "video").length,
141
+ audios: media.filter((m) => m.type === "audio").length, other: 0,
142
+ },
143
+ truncated: false,
144
+ media,
145
+ streamUrlsAvailable: true,
146
+ },
147
+ engine: { attempts: ctx.attempts || [attempt], finalProvider: name, totalLatencyMs: Date.now() - t0 },
148
+ };
149
+ }
150
+
151
+ export async function search() {
152
+ return [];
153
+ }