@dxos/plugin-magazine 0.9.1-staging.ee54ba693a → 0.11.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/lib/MagazinePlugin.mjs +36 -0
- package/dist/lib/MagazinePlugin.mjs.map +1 -0
- package/dist/lib/MagazinePlugin.workerd.mjs +26 -0
- package/dist/lib/MagazinePlugin.workerd.mjs.map +1 -0
- package/dist/lib/atoms.mjs +175 -0
- package/dist/lib/atoms.mjs.map +1 -0
- package/dist/lib/capabilities.mjs +12 -0
- package/dist/lib/capabilities.mjs.map +1 -0
- package/dist/lib/chunk-FeedArticle.mjs +70 -0
- package/dist/lib/chunk-FeedArticle.mjs.map +1 -0
- package/dist/lib/chunk-FeedProperties.mjs +78 -0
- package/dist/lib/chunk-FeedProperties.mjs.map +1 -0
- package/dist/lib/chunk-MagazineArticle.mjs +267 -0
- package/dist/lib/chunk-MagazineArticle.mjs.map +1 -0
- package/dist/lib/chunk-PostArticle.mjs +174 -0
- package/dist/lib/chunk-PostArticle.mjs.map +1 -0
- package/dist/lib/chunk-PostCard.mjs +62 -0
- package/dist/lib/chunk-PostCard.mjs.map +1 -0
- package/dist/lib/chunk-app-graph-builder.mjs +91 -0
- package/dist/lib/chunk-app-graph-builder.mjs.map +1 -0
- package/dist/lib/chunk-clear-magazine.mjs +37 -0
- package/dist/lib/chunk-clear-magazine.mjs.map +1 -0
- package/dist/lib/chunk-create-object.mjs +59 -0
- package/dist/lib/chunk-create-object.mjs.map +1 -0
- package/dist/lib/chunk-curate-magazine.mjs +164 -0
- package/dist/lib/chunk-curate-magazine.mjs.map +1 -0
- package/dist/lib/chunk-date.mjs +25 -0
- package/dist/lib/chunk-date.mjs.map +1 -0
- package/dist/lib/chunk-fetch-article-content.mjs +26 -0
- package/dist/lib/chunk-fetch-article-content.mjs.map +1 -0
- package/dist/lib/chunk-load-post-content.mjs +57 -0
- package/dist/lib/chunk-load-post-content.mjs.map +1 -0
- package/dist/lib/{neutral/chunk-JFPR6OV4.mjs → chunk-meta.mjs} +19 -26
- package/dist/lib/chunk-meta.mjs.map +1 -0
- package/dist/lib/chunk-operation-handler.mjs +13 -0
- package/dist/lib/chunk-operation-handler.mjs.map +1 -0
- package/dist/lib/chunk-paths.mjs +8 -0
- package/dist/lib/chunk-paths.mjs.map +1 -0
- package/dist/lib/chunk-post-content.mjs +14 -0
- package/dist/lib/chunk-post-content.mjs.map +1 -0
- package/dist/lib/chunk-react-surface.mjs +60 -0
- package/dist/lib/chunk-react-surface.mjs.map +1 -0
- package/dist/lib/chunk-routine-templates.mjs +52 -0
- package/dist/lib/chunk-routine-templates.mjs.map +1 -0
- package/dist/lib/chunk-skill-definition.mjs +12 -0
- package/dist/lib/chunk-skill-definition.mjs.map +1 -0
- package/dist/lib/chunk-sources.mjs +576 -0
- package/dist/lib/chunk-sources.mjs.map +1 -0
- package/dist/lib/chunk-sync-feed.mjs +86 -0
- package/dist/lib/chunk-sync-feed.mjs.map +1 -0
- package/dist/lib/chunk-text.mjs +52 -0
- package/dist/lib/chunk-text.mjs.map +1 -0
- package/dist/lib/chunk-types.mjs +607 -0
- package/dist/lib/chunk-types.mjs.map +1 -0
- package/dist/lib/components.mjs +323 -0
- package/dist/lib/components.mjs.map +1 -0
- package/dist/lib/containers.mjs +11 -0
- package/dist/lib/containers.mjs.map +1 -0
- package/dist/lib/index.mjs +3 -0
- package/dist/lib/meta.mjs +2 -0
- package/dist/lib/operations.mjs +7 -0
- package/dist/lib/operations.mjs.map +1 -0
- package/dist/lib/plugin.mjs +9 -0
- package/dist/lib/plugin.mjs.map +1 -0
- package/dist/lib/plugin.workerd.mjs +3 -0
- package/dist/lib/{neutral/skills/index.mjs → skills.mjs} +15 -28
- package/dist/lib/skills.mjs.map +1 -0
- package/dist/lib/testing.mjs +71 -0
- package/dist/lib/testing.mjs.map +1 -0
- package/dist/lib/translations.mjs +74 -0
- package/dist/lib/translations.mjs.map +1 -0
- package/dist/lib/types.mjs +2 -0
- package/dist/types/src/MagazinePlugin.d.ts.map +1 -1
- package/dist/types/src/capabilities/app-graph-builder.d.ts +1 -1
- package/dist/types/src/capabilities/create-object.d.ts +1 -1
- package/dist/types/src/capabilities/index.d.ts +0 -1
- package/dist/types/src/capabilities/index.d.ts.map +1 -1
- package/dist/types/src/capabilities/operation-handler.d.ts +1 -1
- package/dist/types/src/capabilities/react-surface.d.ts +1 -1
- package/dist/types/src/capabilities/react-surface.d.ts.map +1 -1
- package/dist/types/src/capabilities/routine-templates.d.ts +1 -1
- package/dist/types/src/components/PostContent/PostContent.d.ts +1 -13
- package/dist/types/src/components/PostContent/PostContent.d.ts.map +1 -1
- package/dist/types/src/components/PostContent/dedupe-images.d.ts +13 -0
- package/dist/types/src/components/PostContent/dedupe-images.d.ts.map +1 -0
- package/dist/types/src/components/PostContent/index.d.ts +1 -0
- package/dist/types/src/components/PostContent/index.d.ts.map +1 -1
- package/dist/types/src/components/PostStack/PostStack.d.ts +1 -1
- package/dist/types/src/components/PostStack/PostStack.d.ts.map +1 -1
- package/dist/types/src/components/SubscriptionStack/SubscriptionStack.d.ts +1 -1
- package/dist/types/src/components/SubscriptionStack/SubscriptionStack.d.ts.map +1 -1
- package/dist/types/src/containers/FeedArticle/FeedArticle.d.ts +4 -1
- package/dist/types/src/containers/FeedArticle/FeedArticle.d.ts.map +1 -1
- package/dist/types/src/containers/FeedProperties/FeedProperties.d.ts +4 -1
- package/dist/types/src/containers/FeedProperties/FeedProperties.d.ts.map +1 -1
- package/dist/types/src/containers/MagazineArticle/MagazineArticle.d.ts +4 -1
- package/dist/types/src/containers/MagazineArticle/MagazineArticle.d.ts.map +1 -1
- package/dist/types/src/containers/MagazineArticle/MagazineArticle.stories.d.ts.map +1 -1
- package/dist/types/src/containers/MagazineArticle/MagazineTile.d.ts +4 -1
- package/dist/types/src/containers/MagazineArticle/MagazineTile.d.ts.map +1 -1
- package/dist/types/src/containers/MagazineArticle/useToolbar.d.ts +4 -8
- package/dist/types/src/containers/MagazineArticle/useToolbar.d.ts.map +1 -1
- package/dist/types/src/containers/PostArticle/PostArticle.d.ts +4 -1
- package/dist/types/src/containers/PostArticle/PostArticle.d.ts.map +1 -1
- package/dist/types/src/containers/PostArticle/PostArticle.stories.d.ts.map +1 -1
- package/dist/types/src/containers/PostArticle/PostToolbar.d.ts +1 -1
- package/dist/types/src/containers/PostArticle/PostToolbar.d.ts.map +1 -1
- package/dist/types/src/containers/PostCard/PostCard.d.ts +4 -6
- package/dist/types/src/containers/PostCard/PostCard.d.ts.map +1 -1
- package/dist/types/src/containers/SubscriptionsArticle/SubscriptionsArticle.d.ts +4 -1
- package/dist/types/src/containers/SubscriptionsArticle/SubscriptionsArticle.d.ts.map +1 -1
- package/dist/types/src/containers/SubscriptionsArticle/SubscriptionsArticle.stories.d.ts.map +1 -1
- package/dist/types/src/stories/ArticleExtractor.stories.d.ts +1 -0
- package/dist/types/src/stories/ArticleExtractor.stories.d.ts.map +1 -1
- package/dist/types/src/stories/MagazineCurate.stories.d.ts.map +1 -1
- package/dist/types/src/templates/magazine-curation.d.ts.map +1 -1
- package/dist/types/src/translations.d.ts +1 -0
- package/dist/types/src/translations.d.ts.map +1 -1
- package/dist/types/tsconfig.tsbuildinfo +1 -1
- package/dx.config.ts +1 -1
- package/package.json +57 -60
- package/src/MagazinePlugin.tsx +0 -3
- package/src/atoms/atoms.test.ts +5 -4
- package/src/capabilities/app-graph-builder.ts +8 -6
- package/src/capabilities/index.ts +0 -1
- package/src/capabilities/{react-surface.tsx → react-surface.ts} +10 -12
- package/src/components/PostContent/PostContent.test.ts +1 -1
- package/src/components/PostContent/PostContent.tsx +1 -41
- package/src/components/PostContent/dedupe-images.ts +48 -0
- package/src/components/PostContent/index.ts +1 -0
- package/src/containers/FeedArticle/FeedArticle.stories.tsx +2 -2
- package/src/containers/FeedArticle/FeedArticle.tsx +4 -2
- package/src/containers/FeedProperties/FeedProperties.tsx +6 -4
- package/src/containers/MagazineArticle/MagazineArticle.stories.tsx +2 -1
- package/src/containers/MagazineArticle/MagazineArticle.tsx +8 -6
- package/src/containers/MagazineArticle/MagazineTile.tsx +2 -0
- package/src/containers/MagazineArticle/useToolbar.tsx +2 -0
- package/src/containers/PostArticle/PostArticle.stories.tsx +2 -1
- package/src/containers/PostArticle/PostArticle.tsx +6 -3
- package/src/containers/PostArticle/PostToolbar.tsx +1 -1
- package/src/containers/PostCard/PostCard.tsx +3 -1
- package/src/containers/SubscriptionsArticle/SubscriptionsArticle.stories.tsx +2 -1
- package/src/containers/SubscriptionsArticle/SubscriptionsArticle.tsx +5 -3
- package/src/operations/curate-magazine.test.ts +108 -4
- package/src/operations/sync-feed.test.ts +1 -1
- package/src/operations/sync-feed.ts +1 -1
- package/src/paths.ts +3 -3
- package/src/stories/ArticleExtractor.stories.tsx +1 -1
- package/src/stories/MagazineCurate.stories.tsx +3 -2
- package/src/templates/magazine-curation.ts +3 -2
- package/src/types/Magazine.test.ts +1 -1
- package/src/types/Subscription.ts +1 -1
- package/dist/lib/neutral/FeedArticle-PEWCFFSP.mjs +0 -90
- package/dist/lib/neutral/FeedArticle-PEWCFFSP.mjs.map +0 -7
- package/dist/lib/neutral/FeedProperties-EIO7IZOA.mjs +0 -97
- package/dist/lib/neutral/FeedProperties-EIO7IZOA.mjs.map +0 -7
- package/dist/lib/neutral/MagazineArticle-5T7RTIPS.mjs +0 -325
- package/dist/lib/neutral/MagazineArticle-5T7RTIPS.mjs.map +0 -7
- package/dist/lib/neutral/MagazinePlugin.mjs +0 -62
- package/dist/lib/neutral/MagazinePlugin.mjs.map +0 -7
- package/dist/lib/neutral/MagazinePlugin.workerd.mjs +0 -12
- package/dist/lib/neutral/MagazinePlugin.workerd.mjs.map +0 -7
- package/dist/lib/neutral/PostArticle-BGIPU7XA.mjs +0 -230
- package/dist/lib/neutral/PostArticle-BGIPU7XA.mjs.map +0 -7
- package/dist/lib/neutral/PostCard-VDGN7BBA.mjs +0 -60
- package/dist/lib/neutral/PostCard-VDGN7BBA.mjs.map +0 -7
- package/dist/lib/neutral/app-graph-builder-JDPRBVHL.mjs +0 -121
- package/dist/lib/neutral/app-graph-builder-JDPRBVHL.mjs.map +0 -7
- package/dist/lib/neutral/capabilities/index.mjs +0 -21
- package/dist/lib/neutral/capabilities/index.mjs.map +0 -7
- package/dist/lib/neutral/chunk-3K4AQKBZ.mjs +0 -12
- package/dist/lib/neutral/chunk-3K4AQKBZ.mjs.map +0 -7
- package/dist/lib/neutral/chunk-AFPBWBTK.mjs +0 -591
- package/dist/lib/neutral/chunk-AFPBWBTK.mjs.map +0 -7
- package/dist/lib/neutral/chunk-APQRTYT3.mjs +0 -57
- package/dist/lib/neutral/chunk-APQRTYT3.mjs.map +0 -7
- package/dist/lib/neutral/chunk-GU725ZF4.mjs +0 -11
- package/dist/lib/neutral/chunk-GU725ZF4.mjs.map +0 -7
- package/dist/lib/neutral/chunk-J5LGTIGS.mjs +0 -10
- package/dist/lib/neutral/chunk-J5LGTIGS.mjs.map +0 -7
- package/dist/lib/neutral/chunk-JFPR6OV4.mjs.map +0 -7
- package/dist/lib/neutral/chunk-KTYLUOUG.mjs +0 -36
- package/dist/lib/neutral/chunk-KTYLUOUG.mjs.map +0 -7
- package/dist/lib/neutral/chunk-LFSQMLPK.mjs +0 -8
- package/dist/lib/neutral/chunk-LFSQMLPK.mjs.map +0 -7
- package/dist/lib/neutral/chunk-MBPX7RPI.mjs +0 -14
- package/dist/lib/neutral/chunk-MBPX7RPI.mjs.map +0 -7
- package/dist/lib/neutral/chunk-OYCDRSWB.mjs +0 -624
- package/dist/lib/neutral/chunk-OYCDRSWB.mjs.map +0 -7
- package/dist/lib/neutral/chunk-SAWKLGJU.mjs +0 -28
- package/dist/lib/neutral/chunk-SAWKLGJU.mjs.map +0 -7
- package/dist/lib/neutral/chunk-XCNCK47N.mjs +0 -15
- package/dist/lib/neutral/chunk-XCNCK47N.mjs.map +0 -7
- package/dist/lib/neutral/clear-magazine-KNY4OWY4.mjs +0 -59
- package/dist/lib/neutral/clear-magazine-KNY4OWY4.mjs.map +0 -7
- package/dist/lib/neutral/components/index.mjs +0 -338
- package/dist/lib/neutral/components/index.mjs.map +0 -7
- package/dist/lib/neutral/containers/index.mjs +0 -17
- package/dist/lib/neutral/containers/index.mjs.map +0 -7
- package/dist/lib/neutral/create-object-SYZMRR4H.mjs +0 -98
- package/dist/lib/neutral/create-object-SYZMRR4H.mjs.map +0 -7
- package/dist/lib/neutral/curate-magazine-7RMKSJC5.mjs +0 -184
- package/dist/lib/neutral/curate-magazine-7RMKSJC5.mjs.map +0 -7
- package/dist/lib/neutral/fetch-article-content-CXPKXIMR.mjs +0 -31
- package/dist/lib/neutral/fetch-article-content-CXPKXIMR.mjs.map +0 -7
- package/dist/lib/neutral/index.mjs +0 -18
- package/dist/lib/neutral/index.mjs.map +0 -7
- package/dist/lib/neutral/load-post-content-H3S4TKVM.mjs +0 -65
- package/dist/lib/neutral/load-post-content-H3S4TKVM.mjs.map +0 -7
- package/dist/lib/neutral/meta.json +0 -1
- package/dist/lib/neutral/meta.mjs +0 -8
- package/dist/lib/neutral/meta.mjs.map +0 -7
- package/dist/lib/neutral/navigation-resolver-IN4OMCDR.mjs +0 -18
- package/dist/lib/neutral/navigation-resolver-IN4OMCDR.mjs.map +0 -7
- package/dist/lib/neutral/operation-handler-JE3743BU.mjs +0 -8
- package/dist/lib/neutral/operation-handler-JE3743BU.mjs.map +0 -7
- package/dist/lib/neutral/operations/index.mjs +0 -8
- package/dist/lib/neutral/operations/index.mjs.map +0 -7
- package/dist/lib/neutral/plugin.mjs +0 -16
- package/dist/lib/neutral/plugin.mjs.map +0 -7
- package/dist/lib/neutral/plugin.workerd.mjs +0 -16
- package/dist/lib/neutral/plugin.workerd.mjs.map +0 -7
- package/dist/lib/neutral/react-surface-TZESAMG2.mjs +0 -60
- package/dist/lib/neutral/react-surface-TZESAMG2.mjs.map +0 -7
- package/dist/lib/neutral/routine-templates-D5A2L4KL.mjs +0 -52
- package/dist/lib/neutral/routine-templates-D5A2L4KL.mjs.map +0 -7
- package/dist/lib/neutral/skill-definition-3KLMUEC5.mjs +0 -8
- package/dist/lib/neutral/skill-definition-3KLMUEC5.mjs.map +0 -7
- package/dist/lib/neutral/skills/index.mjs.map +0 -7
- package/dist/lib/neutral/sync-feed-SDXYF7HU.mjs +0 -99
- package/dist/lib/neutral/sync-feed-SDXYF7HU.mjs.map +0 -7
- package/dist/lib/neutral/translations.mjs +0 -81
- package/dist/lib/neutral/translations.mjs.map +0 -7
- package/dist/lib/neutral/types/index.mjs +0 -14
- package/dist/lib/neutral/types/index.mjs.map +0 -7
- package/dist/types/src/capabilities/navigation-resolver.d.ts +0 -5
- package/dist/types/src/capabilities/navigation-resolver.d.ts.map +0 -1
- package/dist/types/src/operations/curate-magazine.skill.test.d.ts +0 -2
- package/dist/types/src/operations/curate-magazine.skill.test.d.ts.map +0 -1
- package/src/capabilities/navigation-resolver.ts +0 -21
- package/src/operations/curate-magazine.skill.conversations.json +0 -1
- package/src/operations/curate-magazine.skill.test.ts +0 -167
|
@@ -0,0 +1,576 @@
|
|
|
1
|
+
import { r as makeSnippet, t as decodeEntities } from "./chunk-text.mjs";
|
|
2
|
+
import * as Schema from "effect/Schema";
|
|
3
|
+
import * as Effect from "effect/Effect";
|
|
4
|
+
import { Subscription } from "#types";
|
|
5
|
+
import * as Data from "effect/Data";
|
|
6
|
+
import Defuddle from "defuddle/full";
|
|
7
|
+
import * as FetchHttpClient from "@effect/platform/FetchHttpClient";
|
|
8
|
+
import { XMLParser } from "fast-xml-parser";
|
|
9
|
+
import { normalizeText } from "@dxos/markdown";
|
|
10
|
+
import * as HttpClient from "@effect/platform/HttpClient";
|
|
11
|
+
import * as HttpClientRequest from "@effect/platform/HttpClientRequest";
|
|
12
|
+
import * as HttpClientResponse from "@effect/platform/HttpClientResponse";
|
|
13
|
+
import * as Schedule from "effect/Schedule";
|
|
14
|
+
//#region src/operations/extraction/article.ts
|
|
15
|
+
var isDocumentParserAvailable = () => typeof DOMParser !== "undefined";
|
|
16
|
+
var collectImageUrls = (lead, contentHtml) => {
|
|
17
|
+
const seen = /* @__PURE__ */ new Set();
|
|
18
|
+
const urls = [];
|
|
19
|
+
const push = (url) => {
|
|
20
|
+
if (!url || url.startsWith("data:") || seen.has(url)) return;
|
|
21
|
+
seen.add(url);
|
|
22
|
+
urls.push(url);
|
|
23
|
+
};
|
|
24
|
+
push(lead);
|
|
25
|
+
const imgRegex = /<img\b[^>]+src=["']([^"']+)["']/gi;
|
|
26
|
+
let match;
|
|
27
|
+
while ((match = imgRegex.exec(contentHtml)) != null) push(match[1]);
|
|
28
|
+
return urls;
|
|
29
|
+
};
|
|
30
|
+
/** Substring patterns on class/id that mark a block as chrome. */
|
|
31
|
+
var CHROME_CLASS_PATTERN = /(?:^|[\s_-])(?:tag|topic|related|recirc|footer|sidebar|widget|share|social|comments?|disqus|recommend|more[-_]from|read[-_]next|outbrain|taboola)(?:[\s_-]|$)/i;
|
|
32
|
+
/** Hrefs that point to tag/category/related routes. */
|
|
33
|
+
var CHROME_HREF_PATTERN = /\/(?:tags?|topics?|categor(?:y|ies)|authors?|related|recommended)\//i;
|
|
34
|
+
var isHeading = (el) => /^H[1-6]$/i.test(el.tagName);
|
|
35
|
+
/** A `<ul>` / `<ol>` whose items are essentially a single link each. */
|
|
36
|
+
var isLinkOnlyList = (el) => {
|
|
37
|
+
if (el.tagName !== "UL" && el.tagName !== "OL") return false;
|
|
38
|
+
const items = Array.from(el.children).filter((child) => child.tagName === "LI");
|
|
39
|
+
if (items.length === 0) return false;
|
|
40
|
+
return items.every((li) => {
|
|
41
|
+
const links = li.querySelectorAll("a");
|
|
42
|
+
if (links.length !== 1) return false;
|
|
43
|
+
const text = (li.textContent ?? "").trim();
|
|
44
|
+
const linkText = (links[0].textContent ?? "").trim();
|
|
45
|
+
return text.length > 0 && Math.abs(text.length - linkText.length) <= 4;
|
|
46
|
+
});
|
|
47
|
+
};
|
|
48
|
+
/** Block where the majority of links point to tag/category/related routes. */
|
|
49
|
+
var isTagURLBlock = (el) => {
|
|
50
|
+
const links = Array.from(el.querySelectorAll("a[href]"));
|
|
51
|
+
if (links.length < 2) return false;
|
|
52
|
+
return links.filter((anchor) => CHROME_HREF_PATTERN.test(anchor.getAttribute("href") ?? "")).length / links.length > .6;
|
|
53
|
+
};
|
|
54
|
+
/**
|
|
55
|
+
* True when the element has no body-content descendants. Body content is
|
|
56
|
+
* paragraphs, blockquotes, code/pre, figures, images, and tables — the
|
|
57
|
+
* signal that a block contributes article substance, not chrome. (Lists and
|
|
58
|
+
* raw `<a>` tags don't count: tag clouds and link rails are entirely those.)
|
|
59
|
+
*/
|
|
60
|
+
var hasNoBodyContent = (el) => el.querySelector("p, blockquote, pre, code, figure, img, table") == null;
|
|
61
|
+
var isChromeElement = (el) => {
|
|
62
|
+
const tag = el.tagName;
|
|
63
|
+
if (tag === "ASIDE" || tag === "NAV" || tag === "FOOTER") return true;
|
|
64
|
+
const role = el.getAttribute("role");
|
|
65
|
+
if (role === "navigation" || role === "complementary" || role === "contentinfo") return true;
|
|
66
|
+
if (isLinkOnlyList(el)) return true;
|
|
67
|
+
if (!hasNoBodyContent(el)) return false;
|
|
68
|
+
if (isTagURLBlock(el)) return true;
|
|
69
|
+
const classNameAndId = `${(el.getAttribute("class") ?? "").toString()} ${el.id ?? ""}`;
|
|
70
|
+
if (CHROME_CLASS_PATTERN.test(classNameAndId)) return true;
|
|
71
|
+
return false;
|
|
72
|
+
};
|
|
73
|
+
/**
|
|
74
|
+
* Find the element defuddle is most likely to treat as the main content
|
|
75
|
+
* root, so we prune at the same scope it'll later extract. Mirrors the
|
|
76
|
+
* common selector chain article > main > [role=main] > body.
|
|
77
|
+
*/
|
|
78
|
+
var findContentRoot = (doc) => doc.querySelector("article") ?? doc.querySelector("main") ?? doc.querySelector("[role=\"main\"]") ?? doc.body;
|
|
79
|
+
/**
|
|
80
|
+
* First pass: scan all descendants of the content root and remove elements
|
|
81
|
+
* that look like chrome wherever they appear. An element qualifies when:
|
|
82
|
+
* - its class/id contains a chrome keyword
|
|
83
|
+
* (`tag`, `topic`, `related`, `comments`, `share`, `widget`, `sidebar`,
|
|
84
|
+
* `footer`, `recommend`, etc.), AND
|
|
85
|
+
* - it has no body-content descendants (paragraphs, blockquotes, figures,
|
|
86
|
+
* tables) — i.e. it's a self-contained widget, not an article wrapper
|
|
87
|
+
* that happens to share a keyword.
|
|
88
|
+
* Plus elements that are structurally just a list of links (tag clouds,
|
|
89
|
+
* "related" rails) regardless of class.
|
|
90
|
+
*
|
|
91
|
+
* This catches chrome buried alongside body content in real-world layouts
|
|
92
|
+
* (e.g. theregister's `<div class="similar_topics">` and `<div class="comments">`
|
|
93
|
+
* nested as siblings of the article body inside `<div id="article-wrapper">`).
|
|
94
|
+
*/
|
|
95
|
+
var pruneChromeDescendants = (root) => {
|
|
96
|
+
const candidates = Array.from(root.querySelectorAll("*"));
|
|
97
|
+
for (const el of candidates) {
|
|
98
|
+
if (!root.contains(el)) continue;
|
|
99
|
+
if (isChromeElement(el)) el.remove();
|
|
100
|
+
}
|
|
101
|
+
};
|
|
102
|
+
/**
|
|
103
|
+
* Second pass: when the descendant scan removed an element that used to sit
|
|
104
|
+
* at the trailing edge of the article, the heading that titled it is now
|
|
105
|
+
* dangling. Strip it.
|
|
106
|
+
*
|
|
107
|
+
* Naturally-trailing headings on legitimate articles (essays ending with
|
|
108
|
+
* `<h2>Conclusion</h2>`) are preserved because we only fire when the
|
|
109
|
+
* original trailing edge was actually severed.
|
|
110
|
+
*/
|
|
111
|
+
var trimDanglingTrailingHeading = (root) => {
|
|
112
|
+
let cursor = root.lastElementChild;
|
|
113
|
+
while (cursor && isHeading(cursor) && cursor.nextElementSibling == null) {
|
|
114
|
+
const previous = cursor.previousElementSibling;
|
|
115
|
+
cursor.remove();
|
|
116
|
+
cursor = previous;
|
|
117
|
+
}
|
|
118
|
+
};
|
|
119
|
+
/** @internal exported for unit testing the chrome-pruning rules in isolation. */
|
|
120
|
+
var pruneTrailingChrome = (doc) => {
|
|
121
|
+
const root = findContentRoot(doc);
|
|
122
|
+
if (!root) return;
|
|
123
|
+
const originalLastChild = root.lastElementChild;
|
|
124
|
+
pruneChromeDescendants(root);
|
|
125
|
+
if (originalLastChild && !root.contains(originalLastChild)) trimDanglingTrailingHeading(root);
|
|
126
|
+
};
|
|
127
|
+
var mapResult = (result) => {
|
|
128
|
+
const lead = result.image || void 0;
|
|
129
|
+
return {
|
|
130
|
+
markdown: result.contentMarkdown ?? "",
|
|
131
|
+
html: result.content ?? "",
|
|
132
|
+
title: result.title || void 0,
|
|
133
|
+
author: result.author || void 0,
|
|
134
|
+
description: result.description || void 0,
|
|
135
|
+
published: result.published || void 0,
|
|
136
|
+
image: lead,
|
|
137
|
+
domain: result.domain || void 0,
|
|
138
|
+
imageUrls: collectImageUrls(lead, result.content ?? ""),
|
|
139
|
+
wordCount: result.wordCount
|
|
140
|
+
};
|
|
141
|
+
};
|
|
142
|
+
/**
|
|
143
|
+
* Extracts the main article from a web page's HTML using `defuddle`.
|
|
144
|
+
* Discards navigation, comments, ads, and other chrome; returns the body
|
|
145
|
+
* as Markdown plus structured metadata.
|
|
146
|
+
*
|
|
147
|
+
* Works in both browser/worker (uses `DOMParser`) and Node (delegates to
|
|
148
|
+
* `defuddle/node`, which uses linkedom). The Node path is loaded lazily so
|
|
149
|
+
* the linkedom dependency stays out of browser bundles.
|
|
150
|
+
*/
|
|
151
|
+
var extractArticle = async (html, url) => {
|
|
152
|
+
if (!html) return {
|
|
153
|
+
markdown: "",
|
|
154
|
+
html: "",
|
|
155
|
+
imageUrls: []
|
|
156
|
+
};
|
|
157
|
+
if (isDocumentParserAvailable()) {
|
|
158
|
+
const doc = new DOMParser().parseFromString(html, "text/html");
|
|
159
|
+
pruneTrailingChrome(doc);
|
|
160
|
+
return mapResult(new Defuddle(doc, {
|
|
161
|
+
url,
|
|
162
|
+
separateMarkdown: true
|
|
163
|
+
}).parse());
|
|
164
|
+
}
|
|
165
|
+
const { Defuddle: DefuddleNode } = await import("defuddle/node");
|
|
166
|
+
return mapResult(await DefuddleNode(html, url, { separateMarkdown: true }));
|
|
167
|
+
};
|
|
168
|
+
//#endregion
|
|
169
|
+
//#region src/operations/sources/cors.ts
|
|
170
|
+
/**
|
|
171
|
+
* Cross-origin fetch helpers. Feed/article URLs are arbitrary third-party origins, so in the browser
|
|
172
|
+
* they must be routed through the dev/edge RSS proxy; server-side (no `window`) they're fetched
|
|
173
|
+
* directly.
|
|
174
|
+
*/
|
|
175
|
+
/** Browser proxy path; the target URL is appended URL-encoded. */
|
|
176
|
+
var CORS_PROXY = "/api/rss?url=";
|
|
177
|
+
/** The proxy to use in the current environment: the browser proxy, or undefined server-side. */
|
|
178
|
+
var browserCorsProxy = () => typeof window !== "undefined" ? CORS_PROXY : void 0;
|
|
179
|
+
/** Wraps `url` with `proxy` (URL-encoded) when a proxy is set, else returns `url` unchanged. */
|
|
180
|
+
var applyCorsProxy = (url, proxy) => proxy ? `${proxy}${encodeURIComponent(url)}` : url;
|
|
181
|
+
//#endregion
|
|
182
|
+
//#region src/operations/sources/article.ts
|
|
183
|
+
var FETCH_TIMEOUT_MS = 1e4;
|
|
184
|
+
var MAX_RESPONSE_BYTES = 2e6;
|
|
185
|
+
var MAX_RESPONSE_BYTES_HEADER = 5e6;
|
|
186
|
+
/**
|
|
187
|
+
* Exact-match hostname denylist. Defense-in-depth against trivial SSRF when
|
|
188
|
+
* this runs in a trusted/worker context. Not a substitute for DNS-level
|
|
189
|
+
* egress filtering when available.
|
|
190
|
+
*/
|
|
191
|
+
var BLOCKED_HOSTS = /* @__PURE__ */ new Set([
|
|
192
|
+
"localhost",
|
|
193
|
+
"127.0.0.1",
|
|
194
|
+
"0.0.0.0",
|
|
195
|
+
"::1",
|
|
196
|
+
"169.254.169.254",
|
|
197
|
+
"metadata.google.internal"
|
|
198
|
+
]);
|
|
199
|
+
/** Throws unless `link` is an http(s) URL targeting a non-loopback, non-metadata host. */
|
|
200
|
+
var validateUrl = (link) => {
|
|
201
|
+
const url = new URL(link);
|
|
202
|
+
if (url.protocol !== "http:" && url.protocol !== "https:") throw new Error(`Unsupported protocol: ${url.protocol}`);
|
|
203
|
+
const host = url.hostname.toLowerCase().replace(/^\[|\]$/g, "");
|
|
204
|
+
if (BLOCKED_HOSTS.has(host)) throw new Error(`Blocked host: ${host}`);
|
|
205
|
+
return url;
|
|
206
|
+
};
|
|
207
|
+
/**
|
|
208
|
+
* Read the body with a hard byte cap; prevents unbounded memory use on
|
|
209
|
+
* adversarial responses. Throws if the stream API is unavailable so we never
|
|
210
|
+
* silently bypass the cap.
|
|
211
|
+
*/
|
|
212
|
+
var readCapped = async (response, limit) => {
|
|
213
|
+
const reader = response.body?.getReader();
|
|
214
|
+
if (!reader) throw new Error("Response body stream unavailable.");
|
|
215
|
+
const decoder = new TextDecoder("utf-8");
|
|
216
|
+
let received = 0;
|
|
217
|
+
let out = "";
|
|
218
|
+
while (received < limit) {
|
|
219
|
+
const { value, done } = await reader.read();
|
|
220
|
+
if (done) break;
|
|
221
|
+
if (value) {
|
|
222
|
+
received += value.byteLength;
|
|
223
|
+
out += decoder.decode(value, { stream: true });
|
|
224
|
+
}
|
|
225
|
+
}
|
|
226
|
+
try {
|
|
227
|
+
await reader.cancel();
|
|
228
|
+
} catch {}
|
|
229
|
+
return (out + decoder.decode()).slice(0, limit);
|
|
230
|
+
};
|
|
231
|
+
/**
|
|
232
|
+
* Fetches a post's article page over HTTP and returns extracted plain text
|
|
233
|
+
* plus any image URLs found. Applies protocol validation, a fetch timeout,
|
|
234
|
+
* a Content-Length rejection, and a streamed byte cap.
|
|
235
|
+
*
|
|
236
|
+
* Wraps the original error as `cause` so callers can distinguish
|
|
237
|
+
* AbortError (timeout) from network/fetch failures.
|
|
238
|
+
*/
|
|
239
|
+
var fetchArticle = async (link, options = {}) => {
|
|
240
|
+
try {
|
|
241
|
+
const url = validateUrl(link);
|
|
242
|
+
const fetchTarget = applyCorsProxy(url.toString(), options.corsProxy);
|
|
243
|
+
const response = await fetch(fetchTarget, {
|
|
244
|
+
signal: AbortSignal.timeout(FETCH_TIMEOUT_MS),
|
|
245
|
+
redirect: "follow"
|
|
246
|
+
});
|
|
247
|
+
if (!response.ok) throw new Error(`Fetch failed: ${response.status} ${response.statusText}`);
|
|
248
|
+
const contentLength = response.headers.get("content-length");
|
|
249
|
+
if (contentLength && Number(contentLength) > MAX_RESPONSE_BYTES_HEADER) throw new Error(`Response too large: ${contentLength} bytes`);
|
|
250
|
+
const article = await extractArticle(await readCapped(response, MAX_RESPONSE_BYTES), url.toString());
|
|
251
|
+
return {
|
|
252
|
+
text: article.markdown,
|
|
253
|
+
imageUrls: article.imageUrls
|
|
254
|
+
};
|
|
255
|
+
} catch (error) {
|
|
256
|
+
throw new Error(`Failed to fetch article: ${String(error)}`, { cause: error instanceof Error ? error : void 0 });
|
|
257
|
+
}
|
|
258
|
+
};
|
|
259
|
+
//#endregion
|
|
260
|
+
//#region src/operations/sources/feed-fetcher.ts
|
|
261
|
+
/** Failure fetching or decoding a feed (network error, non-2xx response, or malformed body). */
|
|
262
|
+
var FeedFetchError = class extends Data.TaggedError("FeedFetchError") {};
|
|
263
|
+
//#endregion
|
|
264
|
+
//#region src/operations/sources/http.ts
|
|
265
|
+
var retryPolicy = Schedule.exponential("500 millis").pipe(Schedule.compose(Schedule.recurs(2)));
|
|
266
|
+
/** GETs a URL (through the optional CORS proxy) and decodes the JSON body against `schema`. */
|
|
267
|
+
var getJson = (schema, url, proxy) => HttpClientRequest.get(applyCorsProxy(url, proxy)).pipe(HttpClient.execute, Effect.flatMap(HttpClientResponse.schemaBodyJson(schema)), Effect.timeout("10 seconds"), Effect.retry(retryPolicy), Effect.scoped, Effect.mapError((cause) => new FeedFetchError({
|
|
268
|
+
message: `Fetch failed: ${url}`,
|
|
269
|
+
cause
|
|
270
|
+
})));
|
|
271
|
+
/** GETs a URL (through the optional CORS proxy) and returns the response body as text. */
|
|
272
|
+
var getText = (url, proxy) => HttpClientRequest.get(applyCorsProxy(url, proxy)).pipe(HttpClient.execute, Effect.flatMap((response) => response.text), Effect.timeout("10 seconds"), Effect.retry(retryPolicy), Effect.scoped, Effect.mapError((cause) => new FeedFetchError({
|
|
273
|
+
message: `Fetch failed: ${url}`,
|
|
274
|
+
cause
|
|
275
|
+
})));
|
|
276
|
+
//#endregion
|
|
277
|
+
//#region src/operations/sources/rss.ts
|
|
278
|
+
/**
|
|
279
|
+
* Unwrap `<![CDATA[ ... ]]>` sections, returning the inner content.
|
|
280
|
+
* fast-xml-parser returns stopNode'd content (description / summary / content / content:encoded)
|
|
281
|
+
* verbatim, so CDATA-wrapped HTML — which feeds use pervasively — arrives as a literal
|
|
282
|
+
* `<![CDATA[ ... ]]>` string. Strip the wrapper(s) so the inner HTML reaches downstream
|
|
283
|
+
* conversion. A no-op for values that contain no CDATA section.
|
|
284
|
+
*/
|
|
285
|
+
var stripCdata = (value) => value.replace(/<!\[CDATA\[([\s\S]*?)\]\]>/g, "$1");
|
|
286
|
+
/**
|
|
287
|
+
* Normalize a fast-xml-parser value to a plain string.
|
|
288
|
+
* fast-xml-parser yields objects like `{'#text': 'foo', '@_type': 'html'}` for
|
|
289
|
+
* elements with attributes, and plain strings/numbers for text-only elements.
|
|
290
|
+
* Returns undefined for nullish values.
|
|
291
|
+
*
|
|
292
|
+
* Entities are decoded explicitly: stopNode'd nodes (description / content /
|
|
293
|
+
* summary / content:encoded) are returned as raw text by fast-xml-parser,
|
|
294
|
+
* skipping its built-in entity decoding and CDATA unwrapping. Decoding and
|
|
295
|
+
* CDATA stripping here keep callers from having to special-case those fields.
|
|
296
|
+
* For non-stopped nodes both are no-ops (the parser has already decoded
|
|
297
|
+
* entities and there is no CDATA wrapper).
|
|
298
|
+
*/
|
|
299
|
+
var text = (value) => {
|
|
300
|
+
if (value == null) return;
|
|
301
|
+
if (typeof value === "string") return decodeEntities(stripCdata(value));
|
|
302
|
+
if (typeof value === "number" || typeof value === "boolean") return String(value);
|
|
303
|
+
if (typeof value === "object") {
|
|
304
|
+
const t = value["#text"];
|
|
305
|
+
if (typeof t === "string") return decodeEntities(stripCdata(t));
|
|
306
|
+
if (typeof t === "number" || typeof t === "boolean") return String(t);
|
|
307
|
+
}
|
|
308
|
+
};
|
|
309
|
+
/**
|
|
310
|
+
* Convert a feed text value to Markdown. Feed description/content/summary fields routinely carry
|
|
311
|
+
* embedded HTML (escaped or raw); normalizeText converts it to Markdown and passes plaintext
|
|
312
|
+
* through unchanged. Returns undefined for nullish values.
|
|
313
|
+
*/
|
|
314
|
+
var markdown = (value) => value != null ? normalizeText(value) : void 0;
|
|
315
|
+
/** Fetches and parses an RSS/Atom feed URL into Subscription objects. */
|
|
316
|
+
var fetchRss = (url, options) => Effect.gen(function* () {
|
|
317
|
+
const xml = yield* getText(url, options?.corsProxy);
|
|
318
|
+
return yield* Effect.try({
|
|
319
|
+
try: () => parseFeed(url, xml),
|
|
320
|
+
catch: (cause) => new FeedFetchError({
|
|
321
|
+
message: `Unrecognized feed format: ${url}`,
|
|
322
|
+
cause
|
|
323
|
+
})
|
|
324
|
+
});
|
|
325
|
+
}).pipe(Effect.provide(FetchHttpClient.layer));
|
|
326
|
+
/** Parses RSS/Atom XML into a normalized {@link FetchResult}; throws on an unrecognized shape. */
|
|
327
|
+
var parseFeed = (url, xml) => {
|
|
328
|
+
const parsed = new XMLParser({
|
|
329
|
+
ignoreAttributes: false,
|
|
330
|
+
attributeNamePrefix: "@_",
|
|
331
|
+
maxNestedTags: 1e4,
|
|
332
|
+
stopNodes: [
|
|
333
|
+
"*.description",
|
|
334
|
+
"*.summary",
|
|
335
|
+
"*.content",
|
|
336
|
+
"*.content:encoded"
|
|
337
|
+
]
|
|
338
|
+
}).parse(xml);
|
|
339
|
+
const channel = parsed.rss?.channel ?? parsed.feed;
|
|
340
|
+
if (!channel) throw new Error("Unrecognized feed format");
|
|
341
|
+
const isAtom = !parsed.rss;
|
|
342
|
+
const feedName = text(channel.title) ?? "";
|
|
343
|
+
const feedDescription = markdown(text(isAtom ? channel.subtitle : channel.description)) ?? "";
|
|
344
|
+
const items = (isAtom ? channel.entry : channel.item) ?? [];
|
|
345
|
+
const posts = (Array.isArray(items) ? items : [items]).map((item) => {
|
|
346
|
+
const link = isAtom ? (Array.isArray(item.link) ? item.link.find((l) => l["@_rel"] === "alternate")?.["@_href"] : item.link?.["@_href"] ?? text(item.link)) ?? "" : text(item.link) ?? "";
|
|
347
|
+
const author = isAtom ? text(item.author?.name) ?? text(item.author) : text(item["dc:creator"]) ?? text(item.author);
|
|
348
|
+
const description = markdown(text(item.description));
|
|
349
|
+
const content = isAtom ? markdown(text(item.summary)) ?? markdown(text(item.content)) : markdown(text(item["content:encoded"]));
|
|
350
|
+
return Subscription.makePost({
|
|
351
|
+
title: text(item.title),
|
|
352
|
+
link,
|
|
353
|
+
description,
|
|
354
|
+
content,
|
|
355
|
+
author,
|
|
356
|
+
published: text(item.pubDate) ?? text(item.published) ?? text(item.updated),
|
|
357
|
+
guid: (isAtom ? text(item.id) : text(item.guid)) ?? link
|
|
358
|
+
});
|
|
359
|
+
});
|
|
360
|
+
return {
|
|
361
|
+
feed: Subscription.makeSubscription({
|
|
362
|
+
name: feedName,
|
|
363
|
+
url,
|
|
364
|
+
description: feedDescription
|
|
365
|
+
}),
|
|
366
|
+
posts
|
|
367
|
+
};
|
|
368
|
+
};
|
|
369
|
+
//#endregion
|
|
370
|
+
//#region src/operations/sources/standard-site.ts
|
|
371
|
+
var BSKY_PUBLIC_API = "https://public.api.bsky.app/xrpc";
|
|
372
|
+
var PLC_DIRECTORY = "https://plc.directory";
|
|
373
|
+
var DOCUMENT_COLLECTION = "site.standard.document";
|
|
374
|
+
var MARKDOWN_CONTENT_TYPE = "site.standard.content.markdown";
|
|
375
|
+
/** URL builders for the AT Protocol XRPC endpoints used by this module. */
|
|
376
|
+
var endpoints = {
|
|
377
|
+
resolveHandle: (handle) => `${BSKY_PUBLIC_API}/com.atproto.identity.resolveHandle?handle=${encodeURIComponent(handle)}`,
|
|
378
|
+
getProfile: (actor) => `${BSKY_PUBLIC_API}/app.bsky.actor.getProfile?actor=${encodeURIComponent(actor)}`,
|
|
379
|
+
searchActors: (query, limit) => `${BSKY_PUBLIC_API}/app.bsky.actor.searchActorsTypeahead?q=${encodeURIComponent(query)}&limit=${limit}`,
|
|
380
|
+
plcDoc: (did) => `${PLC_DIRECTORY}/${did}`,
|
|
381
|
+
listRecords: (pds, did, collection, limit) => `${pds}/xrpc/com.atproto.repo.listRecords?repo=${encodeURIComponent(did)}&collection=${encodeURIComponent(collection)}&limit=${limit}`,
|
|
382
|
+
getRecord: (pds, did, collection, rkey) => `${pds}/xrpc/com.atproto.repo.getRecord?repo=${encodeURIComponent(did)}&collection=${encodeURIComponent(collection)}&rkey=${encodeURIComponent(rkey)}`
|
|
383
|
+
};
|
|
384
|
+
/**
|
|
385
|
+
* Extracts an atproto handle or DID from a URL or raw identifier.
|
|
386
|
+
* Supports: bare handles (`dxos.org`), `@handle`, `did:plc:…`/`did:web:…`, or a
|
|
387
|
+
* `https://bsky.app/profile/{actor}` URL.
|
|
388
|
+
*/
|
|
389
|
+
var parseStandardSiteActor = (url) => {
|
|
390
|
+
const match = url.match(/bsky\.app\/profile\/([^/?#]+)/);
|
|
391
|
+
if (match) return match[1];
|
|
392
|
+
return url.replace(/^@/, "").trim();
|
|
393
|
+
};
|
|
394
|
+
/** Lists the publications a handle publishes under (deduped by `site`), for publication selection. */
|
|
395
|
+
var listStandardSitePublications = (actorOrUrl, options) => Effect.gen(function* () {
|
|
396
|
+
const proxy = options?.corsProxy;
|
|
397
|
+
const did = yield* resolveDid(parseStandardSiteActor(actorOrUrl), proxy);
|
|
398
|
+
const sites = distinct((yield* listDocuments(yield* resolvePds(did, proxy), did, proxy)).map((record) => record.value.site).filter(isString));
|
|
399
|
+
return yield* Effect.forEach(sites, (site) => resolvePublication(site, proxy).pipe(Effect.catchAll(() => Effect.succeed({ site }))), { concurrency: "unbounded" });
|
|
400
|
+
}).pipe(Effect.provide(FetchHttpClient.layer));
|
|
401
|
+
/** Searches atproto handles by prefix (typeahead), for combobox suggestions; empty on blank/failed query. */
|
|
402
|
+
var searchStandardSiteHandles = (query, options) => query.trim().length === 0 ? Effect.succeed([]) : getJson(SearchActorsResponse, endpoints.searchActors(query.trim(), 8), options?.corsProxy).pipe(Effect.map((response) => (response.actors ?? []).map((actor) => ({
|
|
403
|
+
handle: actor.handle,
|
|
404
|
+
displayName: actor.displayName
|
|
405
|
+
}))), Effect.orElseSucceed(() => []), Effect.provide(FetchHttpClient.layer));
|
|
406
|
+
/**
|
|
407
|
+
* Fetches a Standard.site feed. `url` is the publication's `site` reference — either an `at://` URI
|
|
408
|
+
* (DID extracted directly) or an `https://` URL (DID resolved via `/.well-known/atproto-did`). In both
|
|
409
|
+
* cases, all documents in the author's repo filtered to that publication are fetched.
|
|
410
|
+
*/
|
|
411
|
+
var fetchStandardSite = (url, options) => Effect.gen(function* () {
|
|
412
|
+
const proxy = options?.corsProxy;
|
|
413
|
+
const did = yield* resolveDidFromSite(url, proxy);
|
|
414
|
+
const records = (yield* listDocuments(yield* resolvePds(did, proxy), did, proxy)).filter((record) => record.value.site === url);
|
|
415
|
+
const profile = yield* fetchProfile(did, proxy);
|
|
416
|
+
const authorName = profile?.displayName ?? profile?.handle ?? did;
|
|
417
|
+
const publication = yield* resolvePublication(url, proxy).pipe(Effect.catchAll(() => Effect.succeed({ site: url })));
|
|
418
|
+
const posts = records.map((record) => {
|
|
419
|
+
const value = record.value;
|
|
420
|
+
const content = value.content?.$type === MARKDOWN_CONTENT_TYPE ? value.content.text : void 0;
|
|
421
|
+
const description = value.description ?? (value.textContent ? makeSnippet(value.textContent) : void 0);
|
|
422
|
+
return Subscription.makePost({
|
|
423
|
+
title: value.title,
|
|
424
|
+
link: joinUrl(publication.url, value.path),
|
|
425
|
+
description,
|
|
426
|
+
content,
|
|
427
|
+
author: authorName,
|
|
428
|
+
published: value.publishedAt,
|
|
429
|
+
guid: record.uri
|
|
430
|
+
});
|
|
431
|
+
});
|
|
432
|
+
posts.sort((postA, postB) => (postB.published ?? "").localeCompare(postA.published ?? ""));
|
|
433
|
+
return {
|
|
434
|
+
feed: Subscription.makeSubscription({
|
|
435
|
+
name: publication.name ?? authorName,
|
|
436
|
+
url,
|
|
437
|
+
description: profile?.description ?? `Standard.site articles from @${profile?.handle ?? did}`,
|
|
438
|
+
iconUrl: profile?.avatar,
|
|
439
|
+
type: "standard-site"
|
|
440
|
+
}),
|
|
441
|
+
posts
|
|
442
|
+
};
|
|
443
|
+
}).pipe(Effect.provide(FetchHttpClient.layer));
|
|
444
|
+
var ResolveHandleResponse = Schema.Struct({ did: Schema.optional(Schema.String) });
|
|
445
|
+
var SearchActorsResponse = Schema.Struct({ actors: Schema.optional(Schema.Array(Schema.Struct({
|
|
446
|
+
handle: Schema.String,
|
|
447
|
+
displayName: Schema.optional(Schema.String)
|
|
448
|
+
}))) });
|
|
449
|
+
var DidDocument = Schema.Struct({ service: Schema.optional(Schema.Array(Schema.Struct({
|
|
450
|
+
id: Schema.optional(Schema.String),
|
|
451
|
+
type: Schema.optional(Schema.String),
|
|
452
|
+
serviceEndpoint: Schema.optional(Schema.String)
|
|
453
|
+
}))) });
|
|
454
|
+
var Profile = Schema.Struct({
|
|
455
|
+
handle: Schema.optional(Schema.String),
|
|
456
|
+
displayName: Schema.optional(Schema.String),
|
|
457
|
+
avatar: Schema.optional(Schema.String),
|
|
458
|
+
description: Schema.optional(Schema.String)
|
|
459
|
+
});
|
|
460
|
+
var StandardSiteDocument = Schema.Struct({
|
|
461
|
+
site: Schema.optional(Schema.String),
|
|
462
|
+
title: Schema.optional(Schema.String),
|
|
463
|
+
publishedAt: Schema.optional(Schema.String),
|
|
464
|
+
path: Schema.optional(Schema.String),
|
|
465
|
+
description: Schema.optional(Schema.String),
|
|
466
|
+
content: Schema.optional(Schema.Struct({
|
|
467
|
+
$type: Schema.optional(Schema.String),
|
|
468
|
+
text: Schema.optional(Schema.String)
|
|
469
|
+
})),
|
|
470
|
+
textContent: Schema.optional(Schema.String)
|
|
471
|
+
});
|
|
472
|
+
var ListRecordsResponse = Schema.Struct({
|
|
473
|
+
records: Schema.optional(Schema.Array(Schema.Struct({
|
|
474
|
+
uri: Schema.String,
|
|
475
|
+
value: StandardSiteDocument
|
|
476
|
+
}))),
|
|
477
|
+
cursor: Schema.optional(Schema.String)
|
|
478
|
+
});
|
|
479
|
+
var PublicationRecord = Schema.Struct({ value: Schema.optional(Schema.Struct({
|
|
480
|
+
url: Schema.optional(Schema.String),
|
|
481
|
+
name: Schema.optional(Schema.String)
|
|
482
|
+
})) });
|
|
483
|
+
/**
|
|
484
|
+
* Resolves a publication `site` reference to its author's DID:
|
|
485
|
+
* - `at://did:…/…` → DID extracted directly from the URI.
|
|
486
|
+
* - `https://…` → DID fetched from `/.well-known/atproto-did` on that domain.
|
|
487
|
+
*/
|
|
488
|
+
var resolveDidFromSite = (site, proxy) => {
|
|
489
|
+
if (site.startsWith("at://")) {
|
|
490
|
+
const parsed = parseAtUri(site);
|
|
491
|
+
return parsed ? Effect.succeed(parsed.did) : Effect.fail(new FeedFetchError({ message: `Malformed at:// site reference: ${site}` }));
|
|
492
|
+
}
|
|
493
|
+
if (site.startsWith("https://")) {
|
|
494
|
+
const { hostname } = new URL(site);
|
|
495
|
+
return getJson(Schema.Struct({ did: Schema.optional(Schema.String) }), `https://${hostname}/.well-known/atproto-did`, proxy).pipe(Effect.flatMap((result) => result.did ? Effect.succeed(result.did) : Effect.fail(new FeedFetchError({ message: `No atproto DID found at ${hostname}/.well-known/atproto-did` }))));
|
|
496
|
+
}
|
|
497
|
+
return Effect.fail(new FeedFetchError({ message: `Cannot resolve DID from site reference: ${site}` }));
|
|
498
|
+
};
|
|
499
|
+
/** Resolves a handle to a DID (no-op when already a DID) via the public `resolveHandle` XRPC. */
|
|
500
|
+
var resolveDid = (actor, proxy) => actor.startsWith("did:") ? Effect.succeed(actor) : getJson(ResolveHandleResponse, endpoints.resolveHandle(actor), proxy).pipe(Effect.flatMap((resolved) => resolved.did ? Effect.succeed(resolved.did) : Effect.fail(new FeedFetchError({ message: `Could not resolve handle to DID: ${actor}` }))));
|
|
501
|
+
/**
|
|
502
|
+
* Resolves a DID to its PDS endpoint: `did:plc` via the PLC directory DID-doc, `did:web` via the
|
|
503
|
+
* domain's `/.well-known/did.json`. Mirrors `plugin-bluesky`'s `BlueskyApi.resolvePds` endpoints.
|
|
504
|
+
*/
|
|
505
|
+
var resolvePds = (did, proxy) => {
|
|
506
|
+
if (did.startsWith("did:plc:")) return getJson(DidDocument, endpoints.plcDoc(did), proxy).pipe(Effect.flatMap((doc) => extractPds(doc, did)));
|
|
507
|
+
if (did.startsWith("did:web:")) {
|
|
508
|
+
const [host, ...segments] = did.slice(8).split(":").map((segment) => decodeURIComponent(segment));
|
|
509
|
+
if (!host) return Effect.fail(new FeedFetchError({ message: `Invalid did:web identifier: ${did}` }));
|
|
510
|
+
return getJson(DidDocument, `https://${host}${segments.length > 0 ? `/${segments.join("/")}/did.json` : "/.well-known/did.json"}`, proxy).pipe(Effect.flatMap((doc) => extractPds(doc, did)));
|
|
511
|
+
}
|
|
512
|
+
return Effect.fail(new FeedFetchError({ message: `Unsupported DID method: ${did}` }));
|
|
513
|
+
};
|
|
514
|
+
var extractPds = (doc, did) => {
|
|
515
|
+
const endpoint = (doc.service?.find((entry) => entry.id === "#atproto_pds" || entry.type === "AtprotoPersonalDataServer"))?.serviceEndpoint;
|
|
516
|
+
return typeof endpoint === "string" && endpoint.length > 0 ? Effect.succeed(endpoint.replace(/\/$/, "")) : Effect.fail(new FeedFetchError({ message: `No PDS endpoint found for DID: ${did}` }));
|
|
517
|
+
};
|
|
518
|
+
/** Lists ALL of the actor's `site.standard.document` records by following pagination cursors. */
|
|
519
|
+
var listDocuments = (pds, did, proxy) => Effect.gen(function* () {
|
|
520
|
+
const all = [];
|
|
521
|
+
let cursor;
|
|
522
|
+
do {
|
|
523
|
+
const page = yield* getJson(ListRecordsResponse, endpoints.listRecords(pds, did, DOCUMENT_COLLECTION, 100) + (cursor ? `&cursor=${encodeURIComponent(cursor)}` : ""), proxy);
|
|
524
|
+
all.push(...page.records ?? []);
|
|
525
|
+
cursor = page.cursor;
|
|
526
|
+
} while (cursor);
|
|
527
|
+
return all;
|
|
528
|
+
});
|
|
529
|
+
/** Best-effort author profile lookup (display name / avatar / bio); undefined on any failure. */
|
|
530
|
+
var fetchProfile = (actor, proxy) => getJson(Profile, endpoints.getProfile(actor), proxy).pipe(Effect.catchAll(() => Effect.succeed(void 0)));
|
|
531
|
+
/**
|
|
532
|
+
* Resolves a publication `site` reference to its {@link Publication} metadata:
|
|
533
|
+
* - `https://` → canonical URL directly (no record fetch needed).
|
|
534
|
+
* - `at://` → `getRecord` the publication record for `url`/`name`.
|
|
535
|
+
*/
|
|
536
|
+
var resolvePublication = (site, proxy) => {
|
|
537
|
+
if (site.startsWith("http")) return Effect.succeed({
|
|
538
|
+
site,
|
|
539
|
+
url: site.replace(/\/$/, "")
|
|
540
|
+
});
|
|
541
|
+
if (site.startsWith("at://")) {
|
|
542
|
+
const parsed = parseAtUri(site);
|
|
543
|
+
if (!parsed) return Effect.succeed({ site });
|
|
544
|
+
return Effect.gen(function* () {
|
|
545
|
+
const pds = yield* resolvePds(parsed.did, proxy);
|
|
546
|
+
const value = (yield* getJson(PublicationRecord, endpoints.getRecord(pds, parsed.did, parsed.collection, parsed.rkey), proxy)).value ?? {};
|
|
547
|
+
return {
|
|
548
|
+
site,
|
|
549
|
+
url: value.url ? value.url.replace(/\/$/, "") : void 0,
|
|
550
|
+
name: value.name
|
|
551
|
+
};
|
|
552
|
+
});
|
|
553
|
+
}
|
|
554
|
+
return Effect.succeed({ site });
|
|
555
|
+
};
|
|
556
|
+
/** Splits an `at://{did}/{collection}/{rkey}` URI into its parts; undefined when malformed. */
|
|
557
|
+
var parseAtUri = (uri) => {
|
|
558
|
+
const match = uri.match(/^at:\/\/([^/]+)\/([^/]+)\/([^/]+)$/);
|
|
559
|
+
return match ? {
|
|
560
|
+
did: match[1],
|
|
561
|
+
collection: match[2],
|
|
562
|
+
rkey: match[3]
|
|
563
|
+
} : void 0;
|
|
564
|
+
};
|
|
565
|
+
/** Joins a publication base URL with a document path into a canonical link. */
|
|
566
|
+
var joinUrl = (base, path) => {
|
|
567
|
+
if (!base) return;
|
|
568
|
+
if (!path) return base;
|
|
569
|
+
return `${base}${path.startsWith("/") ? "" : "/"}${path}`;
|
|
570
|
+
};
|
|
571
|
+
var isString = (value) => typeof value === "string";
|
|
572
|
+
var distinct = (values) => [...new Set(values)];
|
|
573
|
+
//#endregion
|
|
574
|
+
export { fetchArticle as a, fetchRss as i, listStandardSitePublications as n, browserCorsProxy as o, searchStandardSiteHandles as r, fetchStandardSite as t };
|
|
575
|
+
|
|
576
|
+
//# sourceMappingURL=chunk-sources.mjs.map
|