umedia 0.2.0 → 0.2.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -75,10 +75,10 @@ trailers, social posts with official embeds. `previewKind: "none"` beats a fake
75
75
  | Platform | Search | Resolve | Download | How |
76
76
  |---|---|---|---|---|
77
77
  | **YouTube** | ✅ | ✅ | ✅ * | InnerTube — search & metadata always; stream URLs depend on your network (see below) |
78
- | **TikTok** | — | ✅ | ✅ | photo galleries + video posts |
79
- | **Instagram** | — | ✅ best-effort | ✅ | public embeds |
80
- | **Reddit** | — | ✅ | ✅ | posts + galleries |
81
- | **X** | — | ✅ best-effort | ✅ | public posts |
78
+ | **TikTok** | — | ✅ | ✅ | video posts = direct CDN mp4; photo galleries = every slide the platform publicly exposes, `truncated: true` when it degrades galleries |
79
+ | **Instagram** | — | ✅ best-effort | ✅ | public embeds (gated on some networks — the yt-dlp tier is the reliable path) |
80
+ | **Reddit** | — | ✅ best-effort | ✅ | public JSON across reddit edges (datacenter IPs often 403 — typed honest error) |
81
+ | **X** | — | ✅ best-effort | ✅ | public syndication (media login-gated on some networks — typed honest error) |
82
82
  | **iTunes** | ✅ | — | ✅ previews | real 30s clips, rock solid |
83
83
  | **1800+ sites** | — | ✅ | ✅ | optional tier: `pip install yt-dlp` — labeled `source: "ytdlp"` |
84
84
 
@@ -88,6 +88,9 @@ On typical home connections `download()` just works; on hard networks you'll get
88
88
  `downloadWithYtdlp()` / `umedia download <url> --ytdlp` which handles PO tokens, cookies and muxing.
89
89
  We'd rather tell you the truth than fake a download.
90
90
 
91
+ **Direct media URLs always work:** give `resolve()`/`download()` any direct `.mp4/.m4a/.jpg/…` link and it
92
+ goes straight to disk — no platform needed.
93
+
91
94
  ```js
92
95
  // the power tier — one function, 1800+ sites, hard networks
93
96
  await media.downloadWithYtdlp({ url: "https://example.com/anything", quality: "720p", dir: "./out" });
package/lib/download.js CHANGED
@@ -5,7 +5,7 @@ import { pipeline } from "node:stream/promises";
5
5
  import path from "node:path";
6
6
  import { zipSync } from "fflate";
7
7
  import { UMediaError, Codes } from "./errors.js";
8
- import { UA, cleanName, rid } from "./util.js";
8
+ import { UA, cleanName, rid, validateUrl } from "./util.js";
9
9
  import { pickFormat } from "./quality.js";
10
10
  import * as registry from "./registry.js";
11
11
  import * as ytdlp from "./providers/ytdlp.js";
@@ -38,6 +38,9 @@ export async function download({ url, quality = "best", type = "video", dir = ".
38
38
  const requestId = rid();
39
39
  const t0 = Date.now();
40
40
 
41
+ // SSRF hygiene FIRST — user-supplied URLs must never reach private/internal hosts
42
+ validateUrl(url);
43
+
41
44
  let resolved, engine;
42
45
  if (MEDIA_EXT.test(String(url).split("#")[0])) {
43
46
  const ext = (String(url).match(MEDIA_EXT) || [])[1].toLowerCase();
@@ -79,6 +82,7 @@ export async function download({ url, quality = "best", type = "video", dir = ".
79
82
  throw new UMediaError(Codes.PROVIDER_UNAVAILABLE,
80
83
  "no direct media URL available on this network (bot-gated) — install yt-dlp for hard networks");
81
84
  }
85
+ validateUrl(chosenUrl); // never fetch private/internal targets, even from a provider response
82
86
  if (!selectedQuality) { selectedQuality = pick.selectedQuality; fallback = pick.fallback; }
83
87
 
84
88
  const ext = (chosenUrl.split("?")[0].match(/\.(mp4|webm|m4a|mp3|jpg|jpeg|png|gif|webp)$/) || [])[1]
@@ -132,6 +136,7 @@ export async function download({ url, quality = "best", type = "video", dir = ".
132
136
  /** yt-dlp escape hatch for downloads on hard networks / 1800+ sites. */
133
137
  export async function downloadWithYtdlp({ url, quality = "best", dir = "./umedia-downloads", onProgress } = {}) {
134
138
  const requestId = rid();
139
+ validateUrl(url);
135
140
  if (!ytdlp.binary()) throw new UMediaError(Codes.PROVIDER_UNAVAILABLE, "yt-dlp not installed — `pip install yt-dlp`");
136
141
  await mkdir(dir, { recursive: true });
137
142
  const { spawn } = await import("node:child_process");
@@ -1,11 +1,13 @@
1
- import { fetchText } from "../util.js";
1
+ import { fetchText, hostMatches } from "../util.js";
2
2
  import { Codes, UMediaError } from "../errors.js";
3
3
 
4
4
  /** Instagram — public embed page parse. Best-effort: IG gates aggressively; failures are typed. */
5
5
  export const name = "instagram";
6
6
 
7
7
  export function canHandle(url) {
8
- return /(?:^|\.)(instagram\.com|instagr\.am)\//i.test(String(url)) && /\/(p|reel|tv)\//i.test(String(url));
8
+ try {
9
+ return hostMatches(new URL(String(url)).hostname, ["instagram.com", "instagr.am"]) && /\/(p|reel|tv)\//i.test(String(url));
10
+ } catch { return false; }
9
11
  }
10
12
 
11
13
  export async function resolve(url, ctx = {}) {
@@ -1,11 +1,13 @@
1
- import { fetchJson, UA } from "../util.js";
1
+ import { fetchJson, UA, hostMatches } from "../util.js";
2
2
  import { Codes, UMediaError } from "../errors.js";
3
3
 
4
4
  /** Reddit — public JSON endpoints. Datacenter IPs may get 403; the error is honest. */
5
5
  export const name = "reddit";
6
6
 
7
7
  export function canHandle(url) {
8
- return /(?:^|\.)(reddit\.com|redd\.it)\//i.test(String(url));
8
+ try {
9
+ return hostMatches(new URL(String(url)).hostname, ["reddit.com", "redd.it"]);
10
+ } catch { return false; }
9
11
  }
10
12
 
11
13
  export async function resolve(url, ctx = {}) {
@@ -13,8 +15,25 @@ export async function resolve(url, ctx = {}) {
13
15
  ctx.attempts?.push(attempt);
14
16
  const t0 = Date.now();
15
17
  const base = String(url).split("?")[0].replace(/\/$/, "");
18
+
19
+ // Reddit edge-gates datacenter IPs per-host — try the alternate public edges, honestly.
20
+ async function fetchRedditJson() {
21
+ const path = base.replace(/^https?:\/\/[^/]+/, "");
22
+ const hosts = [...new Set([
23
+ new URL(base).host,
24
+ "old.reddit.com", "api.reddit.com", "www.reddit.com",
25
+ ])];
26
+ let lastErr = null;
27
+ for (const host of hosts) {
28
+ try {
29
+ return await fetchJson(`https://${host}${path}.json?raw_json=1`, { headers: { "user-agent": UA + " umedia-sdk" } }, 15_000);
30
+ } catch (e) { lastErr = e; }
31
+ }
32
+ throw lastErr;
33
+ }
34
+
16
35
  try {
17
- const json = await fetchJson(`${base}.json?raw_json=1`, { headers: { "user-agent": UA + " umedia-sdk" } }, 15_000);
36
+ const json = await fetchRedditJson();
18
37
  const post = json?.[0]?.data?.children?.[0]?.data;
19
38
  if (!post) throw new UMediaError(Codes.MEDIA_NOT_FOUND, "Reddit post not found");
20
39
  const media = [];
@@ -42,8 +42,24 @@ export async function resolve(url, ctx = {}) {
42
42
  payload = { data: JSON.parse(raw) };
43
43
  if (id) _cache.set(id, { at: Date.now(), data: payload.data });
44
44
  } else if (status === 200) {
45
- const m = text.match(/<script id="[^"]*" type="application\/json">(.*?)<\/script>/s);
46
- if (m) { payload = { data: JSON.parse(m[1]) }; if (id) _cache.set(id, { at: Date.now(), data: payload.data }); }
45
+ // Frontity state (current embed/v2 shape) first, then any JSON script that carries media data
46
+ const frontity = text.split('<script id="__FRONTITY_CONNECT_STATE__" type="application/json">')[1]?.split("</script>")[0];
47
+ if (frontity) {
48
+ payload = { data: JSON.parse(frontity) };
49
+ if (id) _cache.set(id, { at: Date.now(), data: payload.data });
50
+ } else {
51
+ for (const m of text.matchAll(/<script[^>]*type="application\/json"[^>]*>([\s\S]*?)<\/script>/g)) {
52
+ try {
53
+ const parsed = JSON.parse(m[1]);
54
+ const rawStr = JSON.stringify(parsed);
55
+ if (rawStr.includes("urlList") || rawStr.includes("playAddr") || rawStr.includes("imagePost")) {
56
+ payload = { data: parsed };
57
+ if (id) _cache.set(id, { at: Date.now(), data: payload.data });
58
+ break;
59
+ }
60
+ } catch { /* not state JSON */ }
61
+ }
62
+ }
47
63
  }
48
64
  if (!payload) attempts.push(`embed fetch ${status}`);
49
65
  } catch (e) {
@@ -58,8 +74,20 @@ export async function resolve(url, ctx = {}) {
58
74
  || payload?.data?.itemInfo?.itemStruct
59
75
  || null;
60
76
 
61
- if (scope?.imagePost?.images?.length) {
62
- scope.imagePost.images.forEach((img, i) => {
77
+ // Frontity state — the CURRENT embed/v2 shape (Sept 2026+): source.data[<route>].videoData
78
+ let frontity = null;
79
+ const sourceData = payload?.data?.source?.data;
80
+ if (sourceData && typeof sourceData === "object") {
81
+ for (const node of Object.values(sourceData)) {
82
+ if (node && typeof node === "object" && node.videoData) { frontity = node.videoData; break; }
83
+ }
84
+ }
85
+
86
+ // images: canonical imagePost.images[] (complete) OR displayImages[] (public surface — possibly partial)
87
+ let galleryDegraded = false;
88
+ const imageList = scope?.imagePost?.images;
89
+ if (imageList?.length) {
90
+ imageList.forEach((img, i) => {
63
91
  const u = img.imageURL?.urlList?.[0] || img.imageURL?.url;
64
92
  if (u) media.push({
65
93
  type: "image", index: i, url: u, thumbnail: u, mimeType: "image/jpeg",
@@ -68,13 +96,35 @@ export async function resolve(url, ctx = {}) {
68
96
  hasAudio: false, hasVideo: false, source: name, variants: [],
69
97
  });
70
98
  });
71
- } else if (scope?.video) {
72
- const u = scope.video.playAddr || scope.video.downloadAddr || scope.video.bitrateInfo?.[0]?.PlayAddr?.UrlList?.[0];
73
- if (u) media.push({
74
- type: "video", index: 0, url: u, thumbnail: scope.video.cover || meta.thumbnail_url || null,
75
- mimeType: "video/mp4", width: scope.video.width ?? null, height: scope.video.height ?? null,
76
- duration: scope.video.duration ? scope.video.duration / 1000 : null, size: null,
77
- quality: scope.video.height ? `${scope.video.height}p` : null,
99
+ } else if (frontity?.imagePostInfo?.displayImages?.length) {
100
+ galleryDegraded = true; // platform now serves a partial gallery publicly — honest truncated flag
101
+ frontity.imagePostInfo.displayImages.forEach((img, i) => {
102
+ const u = img?.urlList?.[0];
103
+ if (u) media.push({
104
+ type: "image", index: i, url: u, thumbnail: u, mimeType: "image/jpeg",
105
+ width: img.width ?? null, height: img.height ?? null, duration: null, size: null,
106
+ quality: img.width && img.height ? `${img.width}x${img.height}` : null,
107
+ hasAudio: false, hasVideo: false, source: name, variants: [],
108
+ });
109
+ });
110
+ }
111
+
112
+ // video: classic playAddr OR Frontity itemInfos.video.urls (direct CDN mp4)
113
+ const fv = frontity?.itemInfos?.video;
114
+ const videoUrl = scope?.video?.playAddr || scope?.video?.downloadAddr
115
+ || scope?.video?.bitrateInfo?.[0]?.PlayAddr?.UrlList?.[0]
116
+ || (fv?.urls?.length ? fv.urls[0] : null);
117
+ if (videoUrl) {
118
+ const meta = scope?.video || {};
119
+ const fmeta = fv?.videoMeta || {};
120
+ media.push({
121
+ type: "video", index: media.length, url: videoUrl,
122
+ thumbnail: meta.cover || frontity?.itemInfos?.covers?.[0] || null,
123
+ mimeType: "video/mp4",
124
+ width: meta.width ?? fmeta.width ?? null, height: meta.height ?? fmeta.height ?? null,
125
+ duration: meta.duration ? meta.duration / 1000 : (fmeta.duration ?? null),
126
+ size: null,
127
+ quality: (meta.height || fmeta.height) ? `${meta.height || fmeta.height}p` : null,
78
128
  hasAudio: true, hasVideo: true, source: name, variants: [],
79
129
  });
80
130
  }
@@ -92,8 +142,8 @@ export async function resolve(url, ctx = {}) {
92
142
  result: {
93
143
  sourceUrl: url,
94
144
  platform: "tiktok",
95
- title: meta.title ?? scope?.desc ?? null,
96
- author: meta.author_name ?? scope?.author?.nickname ?? null,
145
+ title: meta.title ?? scope?.desc ?? frontity?.itemInfos?.text ?? null,
146
+ author: meta.author_name ?? scope?.author?.nickname ?? frontity?.authorInfos?.nickName ?? frontity?.authorInfos?.uniqueId ?? null,
97
147
  thumbnail: meta.thumbnail_url ?? media[0].thumbnail,
98
148
  itemCount: media.length,
99
149
  counts: {
@@ -101,7 +151,9 @@ export async function resolve(url, ctx = {}) {
101
151
  videos: media.filter((m) => m.type === "video").length,
102
152
  audios: 0, other: 0,
103
153
  },
104
- truncated: false,
154
+ // honest: true when the public surface exposed only a partial gallery —
155
+ // the full slides need the yt-dlp tier (TikTok degrades embeds since 2026-09)
156
+ truncated: galleryDegraded,
105
157
  media,
106
158
  },
107
159
  engine: { attempts: ctx.attempts || [attempt], finalProvider: name, totalLatencyMs: Date.now() - t0 },
@@ -1,11 +1,13 @@
1
- import { fetchJson } from "../util.js";
1
+ import { fetchJson, hostMatches } from "../util.js";
2
2
  import { Codes, UMediaError } from "../errors.js";
3
3
 
4
4
  /** X (Twitter) — syndication endpoint for public posts. Best-effort, honestly labeled. */
5
5
  export const name = "x";
6
6
 
7
7
  export function canHandle(url) {
8
- return /(?:^|\.)(twitter\.com|x\.com)\//i.test(String(url));
8
+ try {
9
+ return hostMatches(new URL(String(url)).hostname, ["twitter.com", "x.com"]);
10
+ } catch { return false; }
9
11
  }
10
12
 
11
13
  export async function resolve(url, ctx = {}) {
@@ -1,4 +1,4 @@
1
- import { parseDuration } from "../util.js";
1
+ import { parseDuration, hostMatches } from "../util.js";
2
2
  import { Codes, UMediaError } from "../errors.js";
3
3
 
4
4
  /** YouTube via InnerTube (youtubei.js) — search + metadata always; stream URLs when the network allows. */
@@ -14,7 +14,9 @@ async function engine() {
14
14
  }
15
15
 
16
16
  export function canHandle(url) {
17
- return /(?:^|\.)(youtube\.com|youtu\.be|youtube-nocookie\.com)\/|^(www\.)?youtu\.be\//i.test(String(url));
17
+ try {
18
+ return hostMatches(new URL(String(url)).hostname, ["youtube.com", "youtu.be", "youtube-nocookie.com"]);
19
+ } catch { return false; }
18
20
  }
19
21
 
20
22
  export function videoId(url) {
package/lib/registry.js CHANGED
@@ -6,16 +6,48 @@ import * as ig from "./providers/instagram.js";
6
6
  import * as ytdlp from "./providers/ytdlp.js";
7
7
  import * as itunes from "./providers/itunes.js";
8
8
  import { Codes, UMediaError } from "./errors.js";
9
+ import { validateUrl } from "./util.js";
9
10
 
10
11
  const NATIVE = [yt, tiktok, reddit, xprov, ig];
11
12
 
12
13
  export function detectPlatform(url) {
13
- for (const p of NATIVE) if (p.canHandle(url)) return p.name;
14
+ for (const p of NATIVE) {
15
+ try { if (p.canHandle(url)) return p.name; } catch { /* bad URL — keep scanning */ }
16
+ }
14
17
  return ytdlp.binary() ? "ytdlp" : "unknown";
15
18
  }
16
19
 
20
+ const DIRECT_MEDIA = /\.(mp4|webm|m4a|mp3|aac|wav|jpg|jpeg|png|gif|webp|mov|mkv)(\?|#|$)/i;
21
+
17
22
  /** Resolve a URL: native adapter first, optional yt-dlp fallback. attempts[] is the honest trail. */
18
23
  export async function resolve(url) {
24
+ // SSRF hygiene FIRST — never dispatch or fetch private/internal targets (bots pass user URLs!)
25
+ validateUrl(url);
26
+
27
+ // direct media URL (an image/video/audio file) -> single-item result, no platform needed
28
+ const direct = String(url).split("#")[0].match(DIRECT_MEDIA);
29
+ if (direct) {
30
+ const ext = direct[1].toLowerCase();
31
+ const kind = ["mp4", "webm", "mov", "mkv"].includes(ext) ? "video"
32
+ : ["jpg", "jpeg", "png", "gif", "webp"].includes(ext) ? "image" : "audio";
33
+ return {
34
+ data: {
35
+ sourceUrl: url, platform: "direct",
36
+ title: String(url).split("/").pop() || "Direct media",
37
+ author: null, thumbnail: kind === "video" ? null : url,
38
+ itemCount: 1,
39
+ counts: { images: kind === "image" ? 1 : 0, videos: kind === "video" ? 1 : 0, audios: kind === "audio" ? 1 : 0, other: 0 },
40
+ truncated: false,
41
+ media: [{
42
+ type: kind, index: 0, url: String(url), thumbnail: null, mimeType: null,
43
+ width: null, height: null, duration: null, size: null, quality: "original",
44
+ hasAudio: kind !== "image", hasVideo: kind !== "audio", source: "direct", variants: [],
45
+ }],
46
+ },
47
+ engine: { attempts: [{ provider: "direct", status: "success" }], finalProvider: "direct", totalLatencyMs: 0 },
48
+ };
49
+ }
50
+
19
51
  const ctx = { attempts: [] };
20
52
  for (const p of NATIVE) {
21
53
  if (!p.canHandle(url)) continue;
package/lib/util.js CHANGED
@@ -7,6 +7,12 @@ export function rid() {
7
7
  return "req_" + Math.random().toString(16).slice(2, 10) + Date.now().toString(16).slice(-8);
8
8
  }
9
9
 
10
+ /** Hostname match: exact or dot-suffix (www-insensitive). */
11
+ export function hostMatches(hostname, list) {
12
+ const h = String(hostname || "").toLowerCase().replace(/^www\./, "");
13
+ return list.some((d) => h === d || h.endsWith("." + d));
14
+ }
15
+
10
16
  /** http(s) only, no private/loopback/link-local targets (SSRF hygiene). */
11
17
  export function validateUrl(raw) {
12
18
  let u;
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "umedia",
3
- "version": "0.2.0",
3
+ "version": "0.2.2",
4
4
  "description": "Standalone media engine — search, resolve and download media from YouTube, TikTok, Instagram, Reddit, X, iTunes (and 1800+ sites via the yt-dlp tier) entirely on your machine. No hosted API, no keys, no server bills. Honest quality labels, complete galleries, typed errors.",
5
5
  "type": "module",
6
6
  "main": "./index.js",
@@ -12,7 +12,7 @@
12
12
  }
13
13
  },
14
14
  "bin": {
15
- "umedia": "./bin/umedia.js"
15
+ "umedia": "bin/umedia.js"
16
16
  },
17
17
  "files": [
18
18
  "index.js",