umedia 0.3.0 → 0.4.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -8,7 +8,7 @@
8
8
 
9
9
  No hosted API. No API keys. No rate limits. **No server bill.**
10
10
 
11
- YouTube · TikTok · Pinterest · SoundCloud · Dailymotion · Vimeo · Streamable · Twitch · Bluesky · Instagram · Reddit · X · Facebook · iTunes · 1800+ sites (via the optional yt-dlp tier)
11
+ YouTube · TikTok · Pinterest · SoundCloud · Dailymotion · Vimeo · Streamable · Twitch · Bluesky · Instagram · Reddit · X · Facebook · Threads · Snapchat · Rumble · Tumblr · iTunes · 1800+ sites (via the optional yt-dlp tier)
12
12
 
13
13
  [![npm version](https://img.shields.io/npm/v/umedia?color=67e8f9&label=npm)](https://www.npmjs.com/package/umedia)
14
14
  [![node](https://img.shields.io/badge/node-%E2%89%A5%2018-brightgreen)](https://nodejs.org)
@@ -83,10 +83,14 @@ trailers, social posts with official embeds. `previewKind: "none"` beats a fake
83
83
  | **Streamable** | — | ✅ | ✅ | direct mp4 renditions |
84
84
  | **Twitch** | — | ✅ | ✅ | clips = direct CloudFront mp4s; VODs/livestreams via the yt-dlp tier |
85
85
  | **Bluesky** | — | ✅ | ✅ | public AppView API — full-size images + video playlists |
86
- | **Instagram** | — | ✅ best-effort | ✅ | public embeds (gated on some networks — the yt-dlp tier is the reliable path) |
87
- | **Reddit** | — | ✅ best-effort | ✅ | public JSON across reddit edges (datacenter IPs often 403 — typed honest error) |
88
- | **X** | — | ✅ best-effort | ✅ | public syndication (media login-gated on some networks — typed honest error) |
89
- | **Facebook · Threads · Snapchat · Rumble · Tumblr** | — | ✅ best-effort | ✅ best-effort | official Open Graph media on their share pages; often login-gated on datacenter IPs |
86
+ | **Instagram** | — | ✅ | ✅ | oEmbed→mobile info API: photos, videos, full carousels |
87
+ | **Reddit** | — | ✅ | ✅ | posts + galleries + video with its audio sidecar (DC IPs may 403 — typed honest error) |
88
+ | **X** | — | ✅ | ✅ | syndication token + guest GraphQL: photos/videos/gifs at real bitrates |
89
+ | **Facebook** | — | ✅ | ✅ | reels/watch/videos/shares/fb.watch via the web player JSON |
90
+ | **Threads** | — | ✅ | ✅ | post SSR state + og media (Meta gates some networks — typed honest error) |
91
+ | **Snapchat** | — | ✅ | ✅ | spotlight videos (direct sc-cdn mp4) + public stories/highlights — ALL snaps in order |
92
+ | **Rumble** | — | ✅ best-effort | ✅ best-effort | embed API direct mp4s (site blocks datacenter IPs) |
93
+ | **Tumblr** | — | ✅ | ✅ | mobile API: video/audio/photo posts + reblog trails, all in order |
90
94
  | **iTunes** | ✅ | — | ✅ previews | real 30s clips, rock solid |
91
95
  | **1800+ sites** | — | ✅ | ✅ | optional tier: `pip install yt-dlp` — labeled `source: "ytdlp"` |
92
96
 
package/lib/download.js CHANGED
@@ -1,5 +1,5 @@
1
1
  import { createWriteStream } from "node:fs";
2
- import { mkdir, stat } from "node:fs/promises";
2
+ import { mkdir, stat, rename } from "node:fs/promises";
3
3
  import { Readable } from "node:stream";
4
4
  import { pipeline } from "node:stream/promises";
5
5
  import path from "node:path";
@@ -42,6 +42,7 @@ async function downloadHls(url, dest, onProgress) {
42
42
  let manifest = await getText(url);
43
43
  let base = url;
44
44
  let pickedHeight = null;
45
+ let fmp4 = false;
45
46
 
46
47
  if (manifest.includes("#EXT-X-STREAM-INF")) {
47
48
  const lines = manifest.split(/\r?\n/);
@@ -65,14 +66,21 @@ async function downloadHls(url, dest, onProgress) {
65
66
  manifest = await getText(base);
66
67
  }
67
68
  if (manifest.includes("#EXT-X-MAP")) {
68
- throw new UMediaError(Codes.QUALITY_UNAVAILABLE,
69
- "fMP4 HLS stream — single-file muxing needs the yt-dlp tier: downloadWithYtdlp()");
69
+ // fMP4: init segment + media fragments concat to a valid fragmented MP4
70
+ fmp4 = true;
70
71
  }
72
+ const mapUri = /#EXT-X-MAP:URI="([^"]+)"/.exec(manifest)?.[1] || null;
71
73
  const segs = manifest.split(/\r?\n/).filter((l) => l && !l.startsWith("#")).map((l) => new URL(l, base).href);
72
74
  if (!segs.length) throw new UMediaError(Codes.PROVIDER_ERROR, "HLS media playlist has no segments");
73
75
 
74
76
  await mkdir(path.dirname(dest), { recursive: true });
75
77
  const ws = createWriteStream(dest);
78
+ if (fmp4 && mapUri) {
79
+ const initRes = await fetch(new URL(mapUri, base).href, { headers: { "user-agent": UA }, redirect: "follow" });
80
+ if (!initRes.ok) throw new UMediaError(Codes.PROVIDER_ERROR, `HTTP ${initRes.status} fetching fMP4 init segment`);
81
+ const initBuf = Buffer.from(await initRes.arrayBuffer());
82
+ await new Promise((r, j) => ws.write(initBuf, (e) => (e ? j(e) : r())));
83
+ }
76
84
  let done = 0;
77
85
  for (const seg of segs) {
78
86
  const res = await fetch(seg, { headers: { "user-agent": UA }, redirect: "follow" });
@@ -84,7 +92,7 @@ async function downloadHls(url, dest, onProgress) {
84
92
  }
85
93
  await new Promise((r) => ws.end(r));
86
94
  const st = await stat(dest);
87
- return { size: st.size, height: pickedHeight };
95
+ return { size: st.size, height: pickedHeight, ext: fmp4 ? "mp4" : "ts" };
88
96
  }
89
97
 
90
98
  /**
@@ -146,15 +154,22 @@ export async function download({ url, quality = "best", type = "video", dir = ".
146
154
  if (!selectedQuality) { selectedQuality = pick.selectedQuality; fallback = pick.fallback; }
147
155
 
148
156
  const isHls = /\.m3u8$/i.test(chosenUrl.split(/[?#]/)[0]);
149
- const ext = isHls ? "ts" : ((chosenUrl.split("?")[0].match(/\.(mp4|webm|m4a|mp3|jpg|jpeg|png|gif|webp)$/) || [])[1]
157
+ let ext = isHls ? "ts" : ((chosenUrl.split("?")[0].match(/\.(mp4|webm|m4a|mp3|jpg|jpeg|png|gif|webp)$/) || [])[1]
150
158
  || (item.type === "image" ? "jpg" : item.type === "audio" ? "m4a" : "mp4"));
151
159
  const base = cleanName(resolved.title);
152
- const name = `${String(item.index + 1).padStart(3, "0")} - ${base.toLowerCase().endsWith(`.${ext}`) ? base.slice(0, -(ext.length + 1)) : base}.${ext}`;
153
- const dest = path.join(dir, name);
154
- let size, hlsHeight = null;
160
+ let name = `${String(item.index + 1).padStart(3, "0")} - ${base.toLowerCase().endsWith(`.${ext}`) ? base.slice(0, -(ext.length + 1)) : base}.${ext}`;
161
+ let dest = path.join(dir, name);
162
+ let size, hlsHeight = null, hlsExt = null;
155
163
  if (isHls) {
156
- ({ size, height: hlsHeight } = await downloadHls(chosenUrl, dest, onProgress));
164
+ ({ size, height: hlsHeight, ext: hlsExt } = await downloadHls(chosenUrl, dest, onProgress));
157
165
  if (hlsHeight) selectedQuality = `${hlsHeight}p`; // the REAL rendition downloaded
166
+ if (hlsExt && hlsExt !== ext) {
167
+ // fMP4 concat produces a real .mp4 — rename so the file never lies about its container
168
+ ext = hlsExt;
169
+ name = name.replace(/\.ts$/, `.${ext}`);
170
+ dest = path.join(dir, name);
171
+ await rename(`${dest.replace(new RegExp(`\\.${ext}$`), ".ts")}`, dest).catch(() => {});
172
+ }
158
173
  } else {
159
174
  size = await fetchToFile(chosenUrl, dest, onProgress);
160
175
  }
@@ -0,0 +1,126 @@
1
+ import { fetchText, hostMatches } from "../util.js";
2
+ import { Codes, UMediaError } from "../errors.js";
3
+
4
+ /**
5
+ * Facebook — the platform's own player JSON embedded in the page (the exact
6
+ * fields Facebook's web player reads: browser_native_hd/sd_url,
7
+ * playable_url_quality_hd/playable_url). Reels, watch, videos, shares, fb.watch.
8
+ * Reference: yt-dlp facebook extractor + cobalt facebook service.
9
+ */
10
+ export const name = "facebook";
11
+
12
+ const HEADERS = {
13
+ accept: "text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,*/*;q=0.8",
14
+ "accept-language": "en-US,en;q=0.5",
15
+ "sec-fetch-mode": "navigate",
16
+ "sec-fetch-site": "none",
17
+ };
18
+
19
+ export function canHandle(url) {
20
+ try {
21
+ return hostMatches(new URL(String(url)).hostname, ["facebook.com", "fb.watch", "fb.com"]);
22
+ } catch { return false; }
23
+ }
24
+
25
+ export function videoId(url) {
26
+ const u = String(url);
27
+ return (u.match(/\/(?:reel|videos|watch)\/(?:\?v=)?(\d+)/) || u.match(/[?&]v=(\d+)/) || u.match(/\/share\/[^/]+\/([\w-]+)/) || [])[1] || null;
28
+ }
29
+
30
+ function pickJsonStr(html, key) {
31
+ const m = html.match(new RegExp(`"${key}":("(?:[^"\\\\]|\\\\.)*")`));
32
+ if (!m) return null;
33
+ try { return JSON.parse(m[1]); } catch { return null; }
34
+ }
35
+
36
+ async function resolveShort(url, ctx) {
37
+ // fb.watch / facebook.com/share short links -> canonical page (Location or Link header)
38
+ const res = await fetch(url, { headers: HEADERS, redirect: "follow" }).catch(() => null);
39
+ if (!res) return url;
40
+ return res.url || url;
41
+ }
42
+
43
+ export async function resolve(url, ctx = {}) {
44
+ const attempt = { provider: name, status: "skipped" };
45
+ ctx.attempts?.push(attempt);
46
+ const t0 = Date.now();
47
+
48
+ let target = String(url);
49
+ if (/fb\.watch\//i.test(target) || /facebook\.com\/share\//i.test(target)) {
50
+ try { target = await resolveShort(target, ctx); } catch { /* use original */ }
51
+ }
52
+ // normalize watch/video.php/reel forms onto a fetchable page
53
+ const id = videoId(target) || videoId(url);
54
+ if (/\/watch\/?\?/i.test(target) && id) target = `https://www.facebook.com/watch/?v=${id}`;
55
+ if (/video\.php/i.test(target) && id) target = `https://www.facebook.com/video.php?v=${id}`;
56
+ if (/\/reel\//i.test(target) && id) target = `https://www.facebook.com/reel/${id}`;
57
+
58
+ let html;
59
+ try {
60
+ const r = await fetchText(target, { headers: HEADERS }, 20_000);
61
+ html = r.text;
62
+ } catch (e) {
63
+ attempt.status = "failed";
64
+ attempt.error = e.message;
65
+ throw new UMediaError(Codes.PROVIDER_UNAVAILABLE, `Facebook page not retrievable. ${e.message}`, { attempts: [attempt] });
66
+ }
67
+
68
+ const hd = pickJsonStr(html, "browser_native_hd_url") || pickJsonStr(html, "playable_url_quality_hd");
69
+ const sd = pickJsonStr(html, "browser_native_sd_url") || pickJsonStr(html, "playable_url");
70
+
71
+ const media = [];
72
+ const variants = [];
73
+ for (const [u, q] of [[hd, "hd"], [sd, "sd"]]) {
74
+ if (u && /^https?:/.test(u)) variants.push({ url: u, quality: q, height: null, width: null, hasAudio: true, hasVideo: true, mimeType: "video/mp4" });
75
+ }
76
+ if (variants.length) {
77
+ media.push({
78
+ type: "video", index: 0, url: variants[0].url,
79
+ thumbnail: pickJsonStr(html, "playable_url_thumbnail") || null,
80
+ mimeType: "video/mp4", width: null, height: null, duration: null, size: null,
81
+ quality: variants[0].quality, hasAudio: true, hasVideo: true, source: name, variants,
82
+ });
83
+ } else {
84
+ // og fallback (public photos / gated networks)
85
+ const og = (html.match(/property="og:(?:video|image)"[^>]+content="([^"]+)"/) || html.match(/content="([^"]+)"[^>]+property="og:(?:video|image)"/) || [])[1];
86
+ const isVideo = /property="og:video"/.test(html);
87
+ if (og && /^https?:/.test(og.replace(/&/g, "&"))) {
88
+ media.push({
89
+ type: isVideo ? "video" : "image", index: 0, url: og.replace(/&/g, "&"), thumbnail: og,
90
+ mimeType: isVideo ? "video/mp4" : "image/jpeg", width: null, height: null,
91
+ duration: null, size: null, quality: null,
92
+ hasAudio: isVideo, hasVideo: isVideo, source: name, variants: [],
93
+ });
94
+ }
95
+ }
96
+
97
+ if (!media.length) {
98
+ attempt.status = "failed";
99
+ attempt.error = "no player JSON or og media in the page (login-gated, private, or removed)";
100
+ throw new UMediaError(Codes.MEDIA_NOT_FOUND,
101
+ `Facebook exposes no media for this URL (${attempt.error}) — public videos work from most networks; the yt-dlp tier covers gated ones.`, { attempts: [attempt] });
102
+ }
103
+
104
+ const title = (html.match(/"title":"((?:[^"\\]|\\.)*)"/) || html.match(/<title[^>]*>([^<]*)/i) || [])[1] || null;
105
+ attempt.status = "success";
106
+ attempt.latencyMs = Date.now() - t0;
107
+ return {
108
+ result: {
109
+ sourceUrl: url,
110
+ platform: name,
111
+ title: title ? title.replace(/\\u0025/g, "%").slice(0, 200) : null,
112
+ author: null,
113
+ thumbnail: media[0].thumbnail,
114
+ itemCount: media.length,
115
+ counts: { images: media.filter((m) => m.type === "image").length, videos: media.filter((m) => m.type === "video").length, audios: 0, other: 0 },
116
+ truncated: false,
117
+ media,
118
+ streamUrlsAvailable: true,
119
+ },
120
+ engine: { attempts: ctx.attempts || [attempt], finalProvider: name, totalLatencyMs: Date.now() - t0 },
121
+ };
122
+ }
123
+
124
+ export async function search() {
125
+ return [];
126
+ }
@@ -1,67 +1,189 @@
1
- import { fetchText, hostMatches } from "../util.js";
1
+ import { fetchJson, fetchText, hostMatches } from "../util.js";
2
2
  import { Codes, UMediaError } from "../errors.js";
3
3
 
4
- /** Instagram — public embed page parse. Best-effort: IG gates aggressively; failures are typed. */
4
+ /**
5
+ * Instagram — the proven logged-out chain (yt-dlp instagram + cobalt instagram):
6
+ * 1. i.instagram.com oEmbed -> numeric media_id
7
+ * 2. i.instagram.com mobile media info API -> ALL carousel items, best video/photo
8
+ * 3. the /embed/captioned/ page's embedded contextJSON
9
+ * 4. official og: meta on the share page
10
+ */
5
11
  export const name = "instagram";
6
12
 
13
+ const MOBILE_HEADERS = {
14
+ "x-ig-app-locale": "en_US",
15
+ "x-ig-device-locale": "en_US",
16
+ "x-ig-mapped-locale": "en_US",
17
+ "user-agent": "Instagram 275.0.0.27.98 Android (33/13; 280dpi; 720x1423; Xiaomi; Redmi 7; onclite; qcom; en_US; 458229237)",
18
+ "accept-language": "en-US",
19
+ "x-fb-http-engine": "Liger",
20
+ "x-fb-client-ip": "True",
21
+ "x-fb-server-cluster": "True",
22
+ };
23
+
24
+ const EMBED_HEADERS = {
25
+ accept: "text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,image/apng,*/*;q=0.8",
26
+ "accept-language": "en-GB,en;q=0.9",
27
+ "sec-fetch-dest": "document",
28
+ "sec-fetch-mode": "navigate",
29
+ "sec-fetch-site": "none",
30
+ "upgrade-insecure-requests": "1",
31
+ "user-agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/126.0 Safari/537.36",
32
+ };
33
+
7
34
  export function canHandle(url) {
8
35
  try {
9
- return hostMatches(new URL(String(url)).hostname, ["instagram.com", "instagr.am"]) && /\/(p|reel|tv)\//i.test(String(url));
36
+ return hostMatches(new URL(String(url)).hostname, ["instagram.com", "instagr.am"]) && /\/(p|reel|tv)\/[\w-]+/i.test(String(url));
10
37
  } catch { return false; }
11
38
  }
12
39
 
40
+ export function shortcode(url) {
41
+ const m = String(url).match(/\/(?:p|reel|tv)\/([\w-]+)/);
42
+ return m ? m[1] : null;
43
+ }
44
+
45
+ function pushMobileItem(item, media) {
46
+ const isVideo = Array.isArray(item.video_versions) && item.video_versions.length > 0;
47
+ const img = item.image_versions2?.candidates?.[0]?.url || null;
48
+ if (isVideo) {
49
+ const best = item.video_versions.reduce((a, b) => (a.width * a.height < b.width * b.height ? b : a), item.video_versions[0]);
50
+ media.push({
51
+ type: "video", index: media.length, url: best.url, thumbnail: img,
52
+ mimeType: "video/mp4", width: best.width ?? null, height: best.height ?? null,
53
+ duration: item.video_duration ? Math.round(item.video_duration) : null, size: null,
54
+ quality: best.height ? `${best.height}p` : null,
55
+ hasAudio: true, hasVideo: true, source: name,
56
+ variants: item.video_versions.map((v) => ({
57
+ height: v.height ?? null, width: v.width ?? null, url: v.url,
58
+ hasAudio: true, hasVideo: true, quality: v.height ? `${v.height}p` : null, mimeType: "video/mp4",
59
+ })),
60
+ });
61
+ } else if (img) {
62
+ media.push({
63
+ type: "image", index: media.length, url: img, thumbnail: img,
64
+ mimeType: "image/jpeg", width: item.original_width ?? null, height: item.original_height ?? null,
65
+ duration: null, size: null, quality: "original",
66
+ hasAudio: false, hasVideo: false, source: name,
67
+ variants: (item.image_versions2?.candidates || []).map((c) => ({
68
+ height: c.height ?? null, width: c.width ?? null, url: c.url,
69
+ hasAudio: false, hasVideo: false, quality: c.height ? `${c.height}p` : null, mimeType: "image/jpeg",
70
+ })),
71
+ });
72
+ }
73
+ }
74
+
13
75
  export async function resolve(url, ctx = {}) {
14
76
  const attempt = { provider: name, status: "skipped" };
15
77
  ctx.attempts?.push(attempt);
16
78
  const t0 = Date.now();
17
- const clean = String(url).split("?")[0].replace(/\/$/, "");
79
+ const sc = shortcode(url);
80
+ if (!sc) {
81
+ attempt.status = "failed";
82
+ throw new UMediaError(Codes.INVALID_URL, "Not an Instagram post URL (expected instagram.com/p|reel|tv/{code})", { attempts: [attempt] });
83
+ }
84
+
85
+ const media = [];
86
+ let title = null;
87
+ let author = null;
88
+
89
+ // 1+2) oEmbed -> mobile media info (the complete path: carousels included)
18
90
  try {
19
- const { status, text } = await fetchText(`${clean}/embed/captioned/`, {}, 15_000);
20
- if (status !== 200) throw new UMediaError(Codes.PROVIDER_ERROR, `Instagram embed returned HTTP ${status}`);
21
- const media = [];
22
- // video first (reels), then display images
23
- for (const m of text.matchAll(/"video_url":"(https:[^"]+?)"/g)) {
24
- const u = m[1].replace(/\\u0026/g, "&").replace(/\\\//g, "/");
25
- media.push({
26
- type: "video", index: media.length, url: u, thumbnail: null, mimeType: "video/mp4",
27
- width: null, height: null, duration: null, size: null, quality: null,
28
- hasAudio: true, hasVideo: true, source: name, variants: [],
29
- });
30
- break; // first video per embed page
91
+ const o = await fetchJson(`https://i.instagram.com/api/v1/oembed/?url=${encodeURIComponent(`https://www.instagram.com/p/${sc}/`)}`,
92
+ { headers: MOBILE_HEADERS }, 15_000);
93
+ if (o?.media_id) {
94
+ author = o.author_name || null;
95
+ title = o.title || null;
96
+ const info = await fetchJson(`https://i.instagram.com/api/v1/media/${o.media_id}/info/`,
97
+ { headers: MOBILE_HEADERS }, 15_000);
98
+ const item = info?.items?.[0];
99
+ if (item) {
100
+ if (Array.isArray(item.carousel_media) && item.carousel_media.length) {
101
+ // ALL slides, in post order — never truncate
102
+ item.carousel_media.forEach((c) => pushMobileItem(c, media));
103
+ } else {
104
+ pushMobileItem(item, media);
105
+ }
106
+ title = title || item.caption?.text?.slice(0, 200) || null;
107
+ }
31
108
  }
32
- if (!media.length) {
33
- for (const m of text.matchAll(/"display_url":"(https:[^"]+?)"/g)) {
34
- const u = m[1].replace(/\\u0026/g, "&").replace(/\\\//g, "/");
35
- if (!media.some((x) => x.url === u)) media.push({
36
- type: "image", index: media.length, url: u, thumbnail: u, mimeType: "image/jpeg",
37
- width: null, height: null, duration: null, size: null, quality: null,
38
- hasAudio: false, hasVideo: false, source: name, variants: [],
39
- });
109
+ } catch (e) {
110
+ attempt.note = `mobile: ${String(e.message).slice(0, 60)}`;
111
+ }
112
+
113
+ // 3) embed page's embedded contextJSON
114
+ if (!media.length) {
115
+ try {
116
+ const r = await fetchText(`https://www.instagram.com/p/${sc}/embed/captioned/`, { headers: EMBED_HEADERS }, 18_000);
117
+ const init = r.text.match(/"init",\[\],\[(.*?)\]\],/);
118
+ if (init) {
119
+ const arr = JSON.parse(`[${init[1]}]`);
120
+ const ctxJson = arr.map((x) => x?.contextJSON).find(Boolean);
121
+ if (ctxJson) {
122
+ const d = JSON.parse(ctxJson);
123
+ const m = d?.media || d;
124
+ const isVideo = Boolean(m.video_url);
125
+ const u = m.video_url || m.display_url;
126
+ if (u) media.push({
127
+ type: isVideo ? "video" : "image", index: 0, url: u, thumbnail: m.display_url || u,
128
+ mimeType: isVideo ? "video/mp4" : "image/jpeg", width: null, height: null,
129
+ duration: null, size: null, quality: null,
130
+ hasAudio: isVideo, hasVideo: isVideo, source: name, variants: [],
131
+ });
132
+ title = title || m.caption?.text?.slice(0, 200) || null;
133
+ author = author || m.owner?.username || null;
134
+ }
40
135
  }
136
+ } catch (e) {
137
+ attempt.note = `${attempt.note ? attempt.note + " · " : ""}embed: ${String(e.message).slice(0, 60)}`;
41
138
  }
42
- if (!media.length) {
43
- throw new UMediaError(Codes.MEDIA_NOT_FOUND, "No media exposed in the public embed (post may be private)");
139
+ }
140
+
141
+ // 4) og: meta (official share tags)
142
+ if (!media.length) {
143
+ try {
144
+ const r = await fetchText(`https://www.instagram.com/p/${sc}/`, { headers: EMBED_HEADERS }, 18_000);
145
+ const ogv = (r.text.match(/property="og:video"[^>]+content="([^"]+)"/) || r.text.match(/content="([^"]+)"[^>]+property="og:video"/) || [])[1];
146
+ const ogi = (r.text.match(/property="og:image"[^>]+content="([^"]+)"/) || r.text.match(/content="([^"]+)"[^>]+property="og:image"/) || [])[1];
147
+ const u = ogv || ogi;
148
+ if (u) media.push({
149
+ type: ogv ? "video" : "image", index: 0, url: u.replace(/&amp;/g, "&"), thumbnail: (ogi || u).replace(/&amp;/g, "&"),
150
+ mimeType: ogv ? "video/mp4" : "image/jpeg", width: null, height: null,
151
+ duration: null, size: null, quality: null,
152
+ hasAudio: Boolean(ogv), hasVideo: Boolean(ogv), source: name, variants: [],
153
+ });
154
+ } catch (e) {
155
+ attempt.note = `${attempt.note ? attempt.note + " · " : ""}og: ${String(e.message).slice(0, 60)}`;
44
156
  }
45
- attempt.status = "success";
46
- attempt.latencyMs = Date.now() - t0;
47
- return {
48
- result: {
49
- sourceUrl: url, platform: "instagram", title: null, author: null,
50
- thumbnail: media[0].thumbnail, itemCount: media.length,
51
- counts: {
52
- images: media.filter((m) => m.type === "image").length,
53
- videos: media.filter((m) => m.type === "video").length, audios: 0, other: 0,
54
- },
55
- truncated: false, media,
56
- },
57
- engine: { attempts: ctx.attempts || [attempt], finalProvider: name, totalLatencyMs: Date.now() - t0 },
58
- };
59
- } catch (e) {
157
+ }
158
+
159
+ if (!media.length) {
60
160
  attempt.status = "failed";
61
- attempt.error = e.message;
62
- if (e instanceof UMediaError) throw e;
63
- throw new UMediaError(Codes.PROVIDER_UNAVAILABLE, `Instagram unavailable: ${e.message}`, { attempts: [attempt] });
161
+ attempt.error = attempt.note || "no media exposed (private, removed, or login-gated)";
162
+ throw new UMediaError(Codes.MEDIA_NOT_FOUND,
163
+ `Instagram exposes no media for this URL (${attempt.error}) — the yt-dlp tier covers gated posts with cookies.`, { attempts: [attempt] });
64
164
  }
165
+
166
+ attempt.status = "success";
167
+ attempt.latencyMs = Date.now() - t0;
168
+ return {
169
+ result: {
170
+ sourceUrl: url,
171
+ platform: name,
172
+ title: title || null,
173
+ author: author || null,
174
+ thumbnail: media[0].thumbnail,
175
+ itemCount: media.length,
176
+ counts: {
177
+ images: media.filter((m) => m.type === "image").length,
178
+ videos: media.filter((m) => m.type === "video").length,
179
+ audios: 0, other: 0,
180
+ },
181
+ truncated: false,
182
+ media,
183
+ streamUrlsAvailable: true,
184
+ },
185
+ engine: { attempts: ctx.attempts || [attempt], finalProvider: name, totalLatencyMs: Date.now() - t0 },
186
+ };
65
187
  }
66
188
 
67
189
  export async function search() {
@@ -14,7 +14,15 @@ export async function resolve(url, ctx = {}) {
14
14
  const attempt = { provider: name, status: "skipped" };
15
15
  ctx.attempts?.push(attempt);
16
16
  const t0 = Date.now();
17
- const base = String(url).split("?")[0].replace(/\/$/, "");
17
+ let base = String(url).split("?")[0].replace(/\/$/, "");
18
+
19
+ // short links: /video/{id} and /r/{sub}/s/{shareId} redirect to the full permalink
20
+ if (/reddit\.com\/video\/[\w-]+/i.test(base) || /\/s\/[\w-]+/i.test(base)) {
21
+ try {
22
+ const res = await fetch(base, { headers: { "user-agent": UA + " umedia-sdk" }, redirect: "follow" });
23
+ if (res.url && /comments\//.test(res.url)) base = res.url.split("?")[0].replace(/\/$/, "");
24
+ } catch { /* keep original */ }
25
+ }
18
26
 
19
27
  // Reddit edge-gates datacenter IPs per-host — try the alternate public edges, honestly.
20
28
  async function fetchRedditJson() {
@@ -45,11 +53,29 @@ export async function resolve(url, ctx = {}) {
45
53
  });
46
54
 
47
55
  const rv = post.secure_media?.reddit_video || post.media?.reddit_video;
48
- if (rv) add("video", rv.fallback_url, {
49
- width: rv.width, height: rv.height, duration: rv.duration,
50
- quality: rv.height ? `${rv.height}p` : null, mimeType: "video/mp4",
51
- thumbnail: post.preview?.images?.[0]?.source?.url?.replace(/&amp;/g, "&") ?? null,
52
- });
56
+ if (rv) {
57
+ add("video", rv.fallback_url, {
58
+ width: rv.width, height: rv.height, duration: rv.duration,
59
+ quality: rv.height ? `${rv.height}p` : null, mimeType: "video/mp4",
60
+ thumbnail: post.preview?.images?.[0]?.source?.url?.replace(/&amp;/g, "&") ?? null,
61
+ });
62
+ // reddit serves audio as a SIDE CAR file — fetch it too so downloads are complete
63
+ const vidUrl = rv.fallback_url.split("?")[0];
64
+ const audioCandidates = [
65
+ vidUrl.includes(".mp4") ? `${vidUrl.split("_")[0]}_audio.mp4` : null,
66
+ vidUrl.replace(/DASH[^/]*\.mp4/, "DASH_audio"),
67
+ `${vidUrl.split("_")[0]}_AUDIO_128.mp4`,
68
+ ].filter(Boolean);
69
+ for (const au of audioCandidates) {
70
+ try {
71
+ const head = await fetch(au, { method: "HEAD", headers: { "user-agent": UA + " umedia-sdk" }, redirect: "follow" });
72
+ if (head.ok && Number(head.headers.get("content-length") || 1) > 0) {
73
+ add("audio", au, { mimeType: "audio/mp4", quality: "audio" });
74
+ break;
75
+ }
76
+ } catch { /* try the next candidate */ }
77
+ }
78
+ }
53
79
 
54
80
  if (post.gallery_data?.items && post.media_metadata) {
55
81
  for (const it of post.gallery_data.items) {
@@ -0,0 +1,110 @@
1
+ import { fetchJson, hostMatches } from "../util.js";
2
+ import { Codes, UMediaError } from "../errors.js";
3
+
4
+ /**
5
+ * Rumble — the public embed JSON (rumble.com/embedJS/u3/?request=video&ver=2&v={id}),
6
+ * the same route Rumble's own embed player calls: direct mp4 renditions with real
7
+ * w/h/bitrate + HLS + audio. Reference: yt-dlp rumble extractor.
8
+ */
9
+ export const name = "rumble";
10
+
11
+ export function canHandle(url) {
12
+ try {
13
+ return hostMatches(new URL(String(url)).hostname, ["rumble.com"]);
14
+ } catch { return false; }
15
+ }
16
+
17
+ export function videoId(url) {
18
+ const u = String(url);
19
+ // /embed/{id} | /v{id}-{slug}.html | /v{id} | /{id}.html (embed ids)
20
+ const m = u.match(/\/embed\/(?:[\w-]+\.)?([\w-]+)/) || u.match(/\/(v[\w]+)[-.]/) || u.match(/\/(v[\w]+)\/?$/) || u.match(/\/([\w]+)\.html/);
21
+ return m ? m[1] : null;
22
+ }
23
+
24
+ export async function resolve(url, ctx = {}) {
25
+ const attempt = { provider: name, status: "skipped" };
26
+ ctx.attempts?.push(attempt);
27
+ const t0 = Date.now();
28
+
29
+ const id = videoId(url);
30
+ if (!id) {
31
+ attempt.status = "failed";
32
+ throw new UMediaError(Codes.INVALID_URL, "Not a Rumble video URL", { attempts: [attempt] });
33
+ }
34
+
35
+ let v;
36
+ try {
37
+ v = await fetchJson(`https://rumble.com/embedJS/u3/?request=video&ver=2&v=${encodeURIComponent(id)}`, {}, 15_000);
38
+ } catch (e) {
39
+ attempt.status = "failed";
40
+ attempt.error = e.message;
41
+ throw new UMediaError(Codes.PROVIDER_UNAVAILABLE, `Rumble embed API unavailable. ${e.message}`, { attempts: [attempt] });
42
+ }
43
+ if (v?.sys?.msg || !v?.ua) {
44
+ attempt.status = "failed";
45
+ attempt.error = v?.sys?.msg || "video not found";
46
+ throw new UMediaError(Codes.MEDIA_NOT_FOUND, `Rumble: ${attempt.error}`, { attempts: [attempt] });
47
+ }
48
+
49
+ // ua = { mp4: {...height-keyed or list}, hls: [...], audio: [...], timeline: [...] }
50
+ const variants = [];
51
+ const addList = (type, list, heightKeyed) => {
52
+ const entries = heightKeyed ? Object.entries(list).map(([h, info]) => [h, info]) : (list || []).map((info) => [info?.meta?.h, info]);
53
+ for (const [h, info] of entries) {
54
+ if (!info?.url) continue;
55
+ if (type === "tar" || type === "timeline") continue;
56
+ const isHls = type === "hls" || /\.m3u8/.test(info.url);
57
+ variants.push({
58
+ height: parseInt(h ?? info?.meta?.h, 10) || info?.meta?.h || null,
59
+ width: info?.meta?.w ?? null,
60
+ url: info.url,
61
+ hasAudio: type !== "audio", hasVideo: type !== "audio",
62
+ quality: h ? `${h}p` : (type === "audio" ? "audio" : type),
63
+ mimeType: type === "audio" ? "audio/mpeg" : isHls ? "application/x-mpegURL" : "video/mp4",
64
+ bitrate: info?.meta?.bitrate ?? null,
65
+ });
66
+ }
67
+ };
68
+ for (const [type, formatInfo] of Object.entries(v.ua || {})) {
69
+ if (type === "tar") continue;
70
+ addList(type, formatInfo, !Array.isArray(formatInfo));
71
+ }
72
+
73
+ const progressive = variants.filter((x) => /mp4/.test(x.mimeType || "") || (!/m3u8|mpegurl/i.test(x.url + (x.mimeType || "")) && x.hasVideo));
74
+ const chosen = progressive.sort((a, b) => (b.height || 0) - (a.height || 0))[0] || variants[0];
75
+ if (!chosen) {
76
+ attempt.status = "failed";
77
+ throw new UMediaError(Codes.MEDIA_NOT_FOUND, "Rumble: no downloadable renditions", { attempts: [attempt] });
78
+ }
79
+
80
+ attempt.status = "success";
81
+ attempt.latencyMs = Date.now() - t0;
82
+ return {
83
+ result: {
84
+ sourceUrl: url,
85
+ platform: name,
86
+ title: v.title || null,
87
+ author: v.author?.name || null,
88
+ thumbnail: Array.isArray(v.t) ? v.t[v.t.length - 1]?.i : v.t?.i || null,
89
+ itemCount: 1,
90
+ counts: { images: 0, videos: 1, audios: 0, other: 0 },
91
+ truncated: false,
92
+ media: [{
93
+ type: chosen.hasVideo ? "video" : "audio", index: 0, url: chosen.url,
94
+ thumbnail: Array.isArray(v.t) ? v.t[v.t.length - 1]?.i : null,
95
+ mimeType: chosen.mimeType,
96
+ width: chosen.width ?? null, height: chosen.height ?? null,
97
+ duration: v.duration ?? null, size: null,
98
+ quality: chosen.quality,
99
+ hasAudio: true, hasVideo: chosen.hasVideo !== false, source: name,
100
+ variants: variants.filter((x) => x.url),
101
+ }],
102
+ streamUrlsAvailable: true,
103
+ },
104
+ engine: { attempts: ctx.attempts || [attempt], finalProvider: name, totalLatencyMs: Date.now() - t0 },
105
+ };
106
+ }
107
+
108
+ export async function search() {
109
+ return [];
110
+ }