umedia 0.2.0 → 0.2.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -75,10 +75,10 @@ trailers, social posts with official embeds. `previewKind: "none"` beats a fake
75
75
  | Platform | Search | Resolve | Download | How |
76
76
  |---|---|---|---|---|
77
77
  | **YouTube** | ✅ | ✅ | ✅ * | InnerTube — search & metadata always; stream URLs depend on your network (see below) |
78
- | **TikTok** | — | ✅ | ✅ | photo galleries + video posts |
79
- | **Instagram** | — | ✅ best-effort | ✅ | public embeds |
80
- | **Reddit** | — | ✅ | ✅ | posts + galleries |
81
- | **X** | — | ✅ best-effort | ✅ | public posts |
78
+ | **TikTok** | — | ✅ | ✅ | video posts = direct CDN mp4; photo galleries = every slide the platform publicly exposes, `truncated: true` when it degrades galleries |
79
+ | **Instagram** | — | ✅ best-effort | ✅ | public embeds (gated on some networks — the yt-dlp tier is the reliable path) |
80
+ | **Reddit** | — | ✅ best-effort | ✅ | public JSON across reddit edges (datacenter IPs often 403 — typed honest error) |
81
+ | **X** | — | ✅ best-effort | ✅ | public syndication (media login-gated on some networks — typed honest error) |
82
82
  | **iTunes** | ✅ | — | ✅ previews | real 30s clips, rock solid |
83
83
  | **1800+ sites** | — | ✅ | ✅ | optional tier: `pip install yt-dlp` — labeled `source: "ytdlp"` |
84
84
 
@@ -88,6 +88,9 @@ On typical home connections `download()` just works; on hard networks you'll get
88
88
  `downloadWithYtdlp()` / `umedia download <url> --ytdlp` which handles PO tokens, cookies and muxing.
89
89
  We'd rather tell you the truth than fake a download.
90
90
 
91
+ **Direct media URLs always work:** give `resolve()`/`download()` any direct `.mp4/.m4a/.jpg/…` link and it
92
+ goes straight to disk — no platform needed.
93
+
91
94
  ```js
92
95
  // the power tier — one function, 1800+ sites, hard networks
93
96
  await media.downloadWithYtdlp({ url: "https://example.com/anything", quality: "720p", dir: "./out" });
@@ -1,11 +1,13 @@
1
- import { fetchText } from "../util.js";
1
+ import { fetchText, hostMatches } from "../util.js";
2
2
  import { Codes, UMediaError } from "../errors.js";
3
3
 
4
4
  /** Instagram — public embed page parse. Best-effort: IG gates aggressively; failures are typed. */
5
5
  export const name = "instagram";
6
6
 
7
7
  export function canHandle(url) {
8
- return /(?:^|\.)(instagram\.com|instagr\.am)\//i.test(String(url)) && /\/(p|reel|tv)\//i.test(String(url));
8
+ try {
9
+ return hostMatches(new URL(String(url)).hostname, ["instagram.com", "instagr.am"]) && /\/(p|reel|tv)\//i.test(String(url));
10
+ } catch { return false; }
9
11
  }
10
12
 
11
13
  export async function resolve(url, ctx = {}) {
@@ -1,11 +1,13 @@
1
- import { fetchJson, UA } from "../util.js";
1
+ import { fetchJson, UA, hostMatches } from "../util.js";
2
2
  import { Codes, UMediaError } from "../errors.js";
3
3
 
4
4
  /** Reddit — public JSON endpoints. Datacenter IPs may get 403; the error is honest. */
5
5
  export const name = "reddit";
6
6
 
7
7
  export function canHandle(url) {
8
- return /(?:^|\.)(reddit\.com|redd\.it)\//i.test(String(url));
8
+ try {
9
+ return hostMatches(new URL(String(url)).hostname, ["reddit.com", "redd.it"]);
10
+ } catch { return false; }
9
11
  }
10
12
 
11
13
  export async function resolve(url, ctx = {}) {
@@ -13,8 +15,25 @@ export async function resolve(url, ctx = {}) {
13
15
  ctx.attempts?.push(attempt);
14
16
  const t0 = Date.now();
15
17
  const base = String(url).split("?")[0].replace(/\/$/, "");
18
+
19
+ // Reddit edge-gates datacenter IPs per-host — try the alternate public edges, honestly.
20
+ async function fetchRedditJson() {
21
+ const path = base.replace(/^https?:\/\/[^/]+/, "");
22
+ const hosts = [...new Set([
23
+ new URL(base).host,
24
+ "old.reddit.com", "api.reddit.com", "www.reddit.com",
25
+ ])];
26
+ let lastErr = null;
27
+ for (const host of hosts) {
28
+ try {
29
+ return await fetchJson(`https://${host}${path}.json?raw_json=1`, { headers: { "user-agent": UA + " umedia-sdk" } }, 15_000);
30
+ } catch (e) { lastErr = e; }
31
+ }
32
+ throw lastErr;
33
+ }
34
+
16
35
  try {
17
- const json = await fetchJson(`${base}.json?raw_json=1`, { headers: { "user-agent": UA + " umedia-sdk" } }, 15_000);
36
+ const json = await fetchRedditJson();
18
37
  const post = json?.[0]?.data?.children?.[0]?.data;
19
38
  if (!post) throw new UMediaError(Codes.MEDIA_NOT_FOUND, "Reddit post not found");
20
39
  const media = [];
@@ -42,8 +42,24 @@ export async function resolve(url, ctx = {}) {
42
42
  payload = { data: JSON.parse(raw) };
43
43
  if (id) _cache.set(id, { at: Date.now(), data: payload.data });
44
44
  } else if (status === 200) {
45
- const m = text.match(/<script id="[^"]*" type="application\/json">(.*?)<\/script>/s);
46
- if (m) { payload = { data: JSON.parse(m[1]) }; if (id) _cache.set(id, { at: Date.now(), data: payload.data }); }
45
+ // Frontity state (current embed/v2 shape) first, then any JSON script that carries media data
46
+ const frontity = text.split('<script id="__FRONTITY_CONNECT_STATE__" type="application/json">')[1]?.split("</script>")[0];
47
+ if (frontity) {
48
+ payload = { data: JSON.parse(frontity) };
49
+ if (id) _cache.set(id, { at: Date.now(), data: payload.data });
50
+ } else {
51
+ for (const m of text.matchAll(/<script[^>]*type="application\/json"[^>]*>([\s\S]*?)<\/script>/g)) {
52
+ try {
53
+ const parsed = JSON.parse(m[1]);
54
+ const rawStr = JSON.stringify(parsed);
55
+ if (rawStr.includes("urlList") || rawStr.includes("playAddr") || rawStr.includes("imagePost")) {
56
+ payload = { data: parsed };
57
+ if (id) _cache.set(id, { at: Date.now(), data: payload.data });
58
+ break;
59
+ }
60
+ } catch { /* not state JSON */ }
61
+ }
62
+ }
47
63
  }
48
64
  if (!payload) attempts.push(`embed fetch ${status}`);
49
65
  } catch (e) {
@@ -58,8 +74,20 @@ export async function resolve(url, ctx = {}) {
58
74
  || payload?.data?.itemInfo?.itemStruct
59
75
  || null;
60
76
 
61
- if (scope?.imagePost?.images?.length) {
62
- scope.imagePost.images.forEach((img, i) => {
77
+ // Frontity state — the CURRENT embed/v2 shape (Sept 2026+): source.data[<route>].videoData
78
+ let frontity = null;
79
+ const sourceData = payload?.data?.source?.data;
80
+ if (sourceData && typeof sourceData === "object") {
81
+ for (const node of Object.values(sourceData)) {
82
+ if (node && typeof node === "object" && node.videoData) { frontity = node.videoData; break; }
83
+ }
84
+ }
85
+
86
+ // images: canonical imagePost.images[] (complete) OR displayImages[] (public surface — possibly partial)
87
+ let galleryDegraded = false;
88
+ const imageList = scope?.imagePost?.images;
89
+ if (imageList?.length) {
90
+ imageList.forEach((img, i) => {
63
91
  const u = img.imageURL?.urlList?.[0] || img.imageURL?.url;
64
92
  if (u) media.push({
65
93
  type: "image", index: i, url: u, thumbnail: u, mimeType: "image/jpeg",
@@ -68,13 +96,35 @@ export async function resolve(url, ctx = {}) {
68
96
  hasAudio: false, hasVideo: false, source: name, variants: [],
69
97
  });
70
98
  });
71
- } else if (scope?.video) {
72
- const u = scope.video.playAddr || scope.video.downloadAddr || scope.video.bitrateInfo?.[0]?.PlayAddr?.UrlList?.[0];
73
- if (u) media.push({
74
- type: "video", index: 0, url: u, thumbnail: scope.video.cover || meta.thumbnail_url || null,
75
- mimeType: "video/mp4", width: scope.video.width ?? null, height: scope.video.height ?? null,
76
- duration: scope.video.duration ? scope.video.duration / 1000 : null, size: null,
77
- quality: scope.video.height ? `${scope.video.height}p` : null,
99
+ } else if (frontity?.imagePostInfo?.displayImages?.length) {
100
+ galleryDegraded = true; // platform now serves a partial gallery publicly — honest truncated flag
101
+ frontity.imagePostInfo.displayImages.forEach((img, i) => {
102
+ const u = img?.urlList?.[0];
103
+ if (u) media.push({
104
+ type: "image", index: i, url: u, thumbnail: u, mimeType: "image/jpeg",
105
+ width: img.width ?? null, height: img.height ?? null, duration: null, size: null,
106
+ quality: img.width && img.height ? `${img.width}x${img.height}` : null,
107
+ hasAudio: false, hasVideo: false, source: name, variants: [],
108
+ });
109
+ });
110
+ }
111
+
112
+ // video: classic playAddr OR Frontity itemInfos.video.urls (direct CDN mp4)
113
+ const fv = frontity?.itemInfos?.video;
114
+ const videoUrl = scope?.video?.playAddr || scope?.video?.downloadAddr
115
+ || scope?.video?.bitrateInfo?.[0]?.PlayAddr?.UrlList?.[0]
116
+ || (fv?.urls?.length ? fv.urls[0] : null);
117
+ if (videoUrl) {
118
+ const meta = scope?.video || {};
119
+ const fmeta = fv?.videoMeta || {};
120
+ media.push({
121
+ type: "video", index: media.length, url: videoUrl,
122
+ thumbnail: meta.cover || frontity?.itemInfos?.covers?.[0] || null,
123
+ mimeType: "video/mp4",
124
+ width: meta.width ?? fmeta.width ?? null, height: meta.height ?? fmeta.height ?? null,
125
+ duration: meta.duration ? meta.duration / 1000 : (fmeta.duration ?? null),
126
+ size: null,
127
+ quality: (meta.height || fmeta.height) ? `${meta.height || fmeta.height}p` : null,
78
128
  hasAudio: true, hasVideo: true, source: name, variants: [],
79
129
  });
80
130
  }
@@ -92,8 +142,8 @@ export async function resolve(url, ctx = {}) {
92
142
  result: {
93
143
  sourceUrl: url,
94
144
  platform: "tiktok",
95
- title: meta.title ?? scope?.desc ?? null,
96
- author: meta.author_name ?? scope?.author?.nickname ?? null,
145
+ title: meta.title ?? scope?.desc ?? frontity?.itemInfos?.text ?? null,
146
+ author: meta.author_name ?? scope?.author?.nickname ?? frontity?.authorInfos?.nickName ?? frontity?.authorInfos?.uniqueId ?? null,
97
147
  thumbnail: meta.thumbnail_url ?? media[0].thumbnail,
98
148
  itemCount: media.length,
99
149
  counts: {
@@ -101,7 +151,9 @@ export async function resolve(url, ctx = {}) {
101
151
  videos: media.filter((m) => m.type === "video").length,
102
152
  audios: 0, other: 0,
103
153
  },
104
- truncated: false,
154
+ // honest: true when the public surface exposed only a partial gallery —
155
+ // the full slides need the yt-dlp tier (TikTok degrades embeds since 2026-09)
156
+ truncated: galleryDegraded,
105
157
  media,
106
158
  },
107
159
  engine: { attempts: ctx.attempts || [attempt], finalProvider: name, totalLatencyMs: Date.now() - t0 },
@@ -1,11 +1,13 @@
1
- import { fetchJson } from "../util.js";
1
+ import { fetchJson, hostMatches } from "../util.js";
2
2
  import { Codes, UMediaError } from "../errors.js";
3
3
 
4
4
  /** X (Twitter) — syndication endpoint for public posts. Best-effort, honestly labeled. */
5
5
  export const name = "x";
6
6
 
7
7
  export function canHandle(url) {
8
- return /(?:^|\.)(twitter\.com|x\.com)\//i.test(String(url));
8
+ try {
9
+ return hostMatches(new URL(String(url)).hostname, ["twitter.com", "x.com"]);
10
+ } catch { return false; }
9
11
  }
10
12
 
11
13
  export async function resolve(url, ctx = {}) {
@@ -1,4 +1,4 @@
1
- import { parseDuration } from "../util.js";
1
+ import { parseDuration, hostMatches } from "../util.js";
2
2
  import { Codes, UMediaError } from "../errors.js";
3
3
 
4
4
  /** YouTube via InnerTube (youtubei.js) — search + metadata always; stream URLs when the network allows. */
@@ -14,7 +14,9 @@ async function engine() {
14
14
  }
15
15
 
16
16
  export function canHandle(url) {
17
- return /(?:^|\.)(youtube\.com|youtu\.be|youtube-nocookie\.com)\/|^(www\.)?youtu\.be\//i.test(String(url));
17
+ try {
18
+ return hostMatches(new URL(String(url)).hostname, ["youtube.com", "youtu.be", "youtube-nocookie.com"]);
19
+ } catch { return false; }
18
20
  }
19
21
 
20
22
  export function videoId(url) {
package/lib/registry.js CHANGED
@@ -10,12 +10,40 @@ import { Codes, UMediaError } from "./errors.js";
10
10
  const NATIVE = [yt, tiktok, reddit, xprov, ig];
11
11
 
12
12
  export function detectPlatform(url) {
13
- for (const p of NATIVE) if (p.canHandle(url)) return p.name;
13
+ for (const p of NATIVE) {
14
+ try { if (p.canHandle(url)) return p.name; } catch { /* bad URL — keep scanning */ }
15
+ }
14
16
  return ytdlp.binary() ? "ytdlp" : "unknown";
15
17
  }
16
18
 
19
+ const DIRECT_MEDIA = /\.(mp4|webm|m4a|mp3|aac|wav|jpg|jpeg|png|gif|webp|mov|mkv)(\?|#|$)/i;
20
+
17
21
  /** Resolve a URL: native adapter first, optional yt-dlp fallback. attempts[] is the honest trail. */
18
22
  export async function resolve(url) {
23
+ // direct media URL (an image/video/audio file) -> single-item result, no platform needed
24
+ const direct = String(url).split("#")[0].match(DIRECT_MEDIA);
25
+ if (direct) {
26
+ const ext = direct[1].toLowerCase();
27
+ const kind = ["mp4", "webm", "mov", "mkv"].includes(ext) ? "video"
28
+ : ["jpg", "jpeg", "png", "gif", "webp"].includes(ext) ? "image" : "audio";
29
+ return {
30
+ data: {
31
+ sourceUrl: url, platform: "direct",
32
+ title: String(url).split("/").pop() || "Direct media",
33
+ author: null, thumbnail: kind === "video" ? null : url,
34
+ itemCount: 1,
35
+ counts: { images: kind === "image" ? 1 : 0, videos: kind === "video" ? 1 : 0, audios: kind === "audio" ? 1 : 0, other: 0 },
36
+ truncated: false,
37
+ media: [{
38
+ type: kind, index: 0, url: String(url), thumbnail: null, mimeType: null,
39
+ width: null, height: null, duration: null, size: null, quality: "original",
40
+ hasAudio: kind !== "image", hasVideo: kind !== "audio", source: "direct", variants: [],
41
+ }],
42
+ },
43
+ engine: { attempts: [{ provider: "direct", status: "success" }], finalProvider: "direct", totalLatencyMs: 0 },
44
+ };
45
+ }
46
+
19
47
  const ctx = { attempts: [] };
20
48
  for (const p of NATIVE) {
21
49
  if (!p.canHandle(url)) continue;
package/lib/util.js CHANGED
@@ -7,6 +7,12 @@ export function rid() {
7
7
  return "req_" + Math.random().toString(16).slice(2, 10) + Date.now().toString(16).slice(-8);
8
8
  }
9
9
 
10
+ /** Hostname match: exact or dot-suffix (www-insensitive). */
11
+ export function hostMatches(hostname, list) {
12
+ const h = String(hostname || "").toLowerCase().replace(/^www\./, "");
13
+ return list.some((d) => h === d || h.endsWith("." + d));
14
+ }
15
+
10
16
  /** http(s) only, no private/loopback/link-local targets (SSRF hygiene). */
11
17
  export function validateUrl(raw) {
12
18
  let u;
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "umedia",
3
- "version": "0.2.0",
3
+ "version": "0.2.1",
4
4
  "description": "Standalone media engine — search, resolve and download media from YouTube, TikTok, Instagram, Reddit, X, iTunes (and 1800+ sites via the yt-dlp tier) entirely on your machine. No hosted API, no keys, no server bills. Honest quality labels, complete galleries, typed errors.",
5
5
  "type": "module",
6
6
  "main": "./index.js",