mcp-scraper 0.49.0 → 0.51.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (59) hide show
  1. package/README.md +11 -3
  2. package/dist/{analytics-repository-WWXVU3VW.js → analytics-repository-IKE53S3E.js} +10 -2
  3. package/dist/bin/api-server.cjs +840 -177
  4. package/dist/bin/api-server.cjs.map +1 -1
  5. package/dist/bin/api-server.js +4 -4
  6. package/dist/bin/mcp-scraper-cli.cjs +1 -1
  7. package/dist/bin/mcp-scraper-cli.cjs.map +1 -1
  8. package/dist/bin/mcp-scraper-cli.js +1 -1
  9. package/dist/bin/mcp-scraper-install.cjs +1 -1
  10. package/dist/bin/mcp-scraper-install.cjs.map +1 -1
  11. package/dist/bin/mcp-scraper-install.js +1 -1
  12. package/dist/bin/mcp-stdio-server.cjs +141 -35
  13. package/dist/bin/mcp-stdio-server.cjs.map +1 -1
  14. package/dist/bin/mcp-stdio-server.js +4 -4
  15. package/dist/bin/paa-harvest.cjs.map +1 -1
  16. package/dist/bin/paa-harvest.js +3 -3
  17. package/dist/{chunk-MFNGUM4L.js → chunk-2XTYLJQK.js} +15 -1
  18. package/dist/chunk-2XTYLJQK.js.map +1 -0
  19. package/dist/{chunk-YW2LXLDU.js → chunk-5BEPOFKG.js} +160 -15
  20. package/dist/chunk-5BEPOFKG.js.map +1 -0
  21. package/dist/chunk-73MUQSWC.js +7 -0
  22. package/dist/chunk-73MUQSWC.js.map +1 -0
  23. package/dist/{chunk-MFE6RMIA.js → chunk-GOZIG6HD.js} +2 -2
  24. package/dist/{chunk-R66PJOZW.js → chunk-NNZXZTJ2.js} +2 -2
  25. package/dist/{chunk-64WBUDPC.js → chunk-O63VVCCE.js} +143 -37
  26. package/dist/chunk-O63VVCCE.js.map +1 -0
  27. package/dist/{chunk-6DPIE262.js → chunk-QHVFKFME.js} +2 -2
  28. package/dist/{chunk-PODASLGT.js → chunk-QXAY44SA.js} +2 -2
  29. package/dist/{chunk-ZEMHL36J.js → chunk-YMSZO62N.js} +2 -2
  30. package/dist/{db-A4YUPV4Z.js → db-UPSS5BGM.js} +2 -2
  31. package/dist/editorial-reading-room/assets/app.js +62 -0
  32. package/dist/editorial-reading-room/assets/index.html +8 -0
  33. package/dist/editorial-reading-room/assets/styles.css +39 -0
  34. package/dist/{extract-bundle-2HONVC5T.js → extract-bundle-RCTNANCH.js} +3 -3
  35. package/dist/index.cjs.map +1 -1
  36. package/dist/index.js +3 -3
  37. package/dist/{location-data-repository-YWFJGNIP.js → location-data-repository-UUSHH5M2.js} +3 -3
  38. package/dist/{server-JE7SLZLJ.js → server-IGIIEHXY.js} +518 -135
  39. package/dist/server-IGIIEHXY.js.map +1 -0
  40. package/dist/{site-extract-repository-STNOI5OH.js → site-extract-repository-3VQAALNR.js} +3 -3
  41. package/dist/{worker-4MHZS3OB.js → worker-KQN673JF.js} +5 -5
  42. package/package.json +1 -1
  43. package/dist/chunk-64WBUDPC.js.map +0 -1
  44. package/dist/chunk-MFNGUM4L.js.map +0 -1
  45. package/dist/chunk-XLEDHV3J.js +0 -7
  46. package/dist/chunk-XLEDHV3J.js.map +0 -1
  47. package/dist/chunk-YW2LXLDU.js.map +0 -1
  48. package/dist/server-JE7SLZLJ.js.map +0 -1
  49. /package/dist/{analytics-repository-WWXVU3VW.js.map → analytics-repository-IKE53S3E.js.map} +0 -0
  50. /package/dist/{chunk-MFE6RMIA.js.map → chunk-GOZIG6HD.js.map} +0 -0
  51. /package/dist/{chunk-R66PJOZW.js.map → chunk-NNZXZTJ2.js.map} +0 -0
  52. /package/dist/{chunk-6DPIE262.js.map → chunk-QHVFKFME.js.map} +0 -0
  53. /package/dist/{chunk-PODASLGT.js.map → chunk-QXAY44SA.js.map} +0 -0
  54. /package/dist/{chunk-ZEMHL36J.js.map → chunk-YMSZO62N.js.map} +0 -0
  55. /package/dist/{db-A4YUPV4Z.js.map → db-UPSS5BGM.js.map} +0 -0
  56. /package/dist/{extract-bundle-2HONVC5T.js.map → extract-bundle-RCTNANCH.js.map} +0 -0
  57. /package/dist/{location-data-repository-YWFJGNIP.js.map → location-data-repository-UUSHH5M2.js.map} +0 -0
  58. /package/dist/{site-extract-repository-STNOI5OH.js.map → site-extract-repository-3VQAALNR.js.map} +0 -0
  59. /package/dist/{worker-4MHZS3OB.js.map → worker-KQN673JF.js.map} +0 -0
@@ -18,10 +18,10 @@ import {
18
18
  browserServiceProfileSaveChanges,
19
19
  recordVendorUsage,
20
20
  vendorCostUsd
21
- } from "./chunk-PODASLGT.js";
21
+ } from "./chunk-QXAY44SA.js";
22
22
  import {
23
23
  PACKAGE_VERSION
24
- } from "./chunk-XLEDHV3J.js";
24
+ } from "./chunk-73MUQSWC.js";
25
25
  import {
26
26
  MC_PER_CREDIT
27
27
  } from "./chunk-SOEYDWJU.js";
@@ -33,6 +33,7 @@ import {
33
33
 
34
34
  // src/harvest-timeout.ts
35
35
  var VERCEL_FUNCTION_MAX_MS = 3e5;
36
+ var DEFAULT_TOOL_CLIENT_TIMEOUT_MS = 3e5;
36
37
  var CLIENT_OVER_SERVER_MARGIN_MS = 15e3;
37
38
  function harvestTimeoutBudget(maxQuestions, serpOnly = false) {
38
39
  const requested = Number.isFinite(maxQuestions) && maxQuestions > 0 ? Math.trunc(maxQuestions) : 30;
@@ -451,7 +452,10 @@ seam is noted so you can chain them.
451
452
  handles Reddit's bot wall itself (no login needed). Find threads first with \`search_serp\` or \`reddit_trending\`.
452
453
  - DISCOVER what a niche is talking about (topic, no known thread) -> **reddit_trending** (takes a topic, optional
453
454
  subreddit; returns the last 30 days' top threads ranked by engagement plus the questions people asked \u2014
454
- feed winning \`rankedThreads[].url\` values into \`reddit_thread\` for the full comment tree).
455
+ feed winning \`rankedThreads[].url\` values into \`reddit_thread\` for the full comment tree). It uses bounded
456
+ direct Reddit discovery when the primary SERP is empty. Before interpreting an empty result, inspect
457
+ \`resultQuality\`, \`discoverySource\`, \`degradationReasons\`, \`retryRecommended\`, and \`billingRefunded\`;
458
+ a degraded empty result is not evidence that the topic has no Reddit discussion.
455
459
 
456
460
  ## Other sites & logins (browser agent)
457
461
  For an arbitrary site or a logged-in dashboard with no dedicated tool, use the browser_* agent. **First
@@ -501,6 +505,12 @@ Multi-step orchestrations \u2014 prefer these over hand-chaining primitives when
501
505
  versioning reusable website templates; use the saved-template tools above for those jobs.
502
506
  - The creation tool is a renderer, not a research or writing model. Preserve source truth in each
503
507
  \`sourceLabel\`; do not hand it raw source material and expect it to invent the editorial architecture.
508
+ - One edition may contain up to 100 articles. Images may appear as structured article/card heroes or as
509
+ Markdown body images. Use \`site.ogImage\` for the collection social preview and \`article.ogImage\` for an
510
+ article override; otherwise the article hero, then collection image, is reused. Every image requires useful
511
+ alt text. Preserve caption, credit, source URL, and rights context when known; a reachable URL does not prove
512
+ reuse rights. Static artifacts include collection OG tags and update article tags in-browser, while crawler-
513
+ perfect per-article unfurls require the publishing host to serve article-specific metadata.
504
514
  - ${savesReportsLocally ? "This local stdio server writes one self-contained HTML file and returns its localPath so the user can open it." : "This hosted server creates a private seven-day HTML artifact and returns a signed download URL; use renew_editorial_reading_room_download after the URL expires."}
505
515
 
506
516
  ## Local Sourcebook listings
@@ -592,6 +602,17 @@ deliberate edits and migrations. Use
592
602
  they'll filter/sort by exact value. **memory-search** is meaning-based over full content (embeds, slower);
593
603
  **memory-list** filters one vault by kind/tags (fast, exact).
594
604
 
605
+ Choose the read surface by completeness, not convenience. **list-vaults** is the account inventory and
606
+ reports each vault's note count. **memory-list** returns the complete metadata inventory for one vault but
607
+ no note bodies, plus a sorted complete folder inventory derived from those paths; use it to identify every
608
+ stable \`vault + path\` address. **list-memory-tags** returns the
609
+ complete account-wide canonical tag inventory, including aliases, usage counts, and per-vault distribution.
610
+ **memory-search** returns ranked
611
+ content chunks, not an exhaustive inventory or complete documents: deduplicate hits by vault/path and call
612
+ **memory-get** for every note whose full meaning matters. When the user explicitly needs every full note for
613
+ backup, migration, or corpus-wide audit, use **memory-export** one vault at a time; do not pull the entire
614
+ corpus merely to edit one note.
615
+
595
616
  For People, Deals, Projects, Tasks, and Communications, use the returned contract rather than guessing:
596
617
  People only holds a real person or organization hub. Its contact card uses \`phone\`, \`text_phone\`, and
597
618
  \`email\` for Call/Text/Email; \`memories\` for durable person context; and linked Deals, Projects,
@@ -650,19 +671,35 @@ Tags are live vocabulary, not improvised labels: **list-memory-tags** shows what
650
671
  reusable, and has no exact, alias, or near-equivalent. Use **memory-backlinks**,
651
672
  **memory-graph-universe**, and **memory-graph-path** to trace the linked universe across vaults.
652
673
 
653
- **Always inspect the complete tag inventory and related notes first.** Use hybrid Smart RAG by default:
674
+ **Always inspect the complete tag inventory and related notes first.** When the user wants to find, recall,
675
+ understand, or connect something and does not already provide an exact vault/path, start with hybrid Smart
676
+ RAG through **memory-search**. The exceptions are exhaustive inventory (**memory-list**), title-only lookup
677
+ (**memory-suggest**), an exact known note (**memory-get**), and an explicit full-vault export (**memory-export**).
678
+ For hybrid retrieval,
654
679
  form 3 focused queries (2\u20134 when useful), fuse exact tag/metadata/vault/date and semantic matches into 50
655
680
  candidates, expand the top 8 seeds by one link/backlink hop with at most 5 neighbors each, then Jina-rerank
656
681
  the combined pool to the best 30. Graph neighbors are candidates, never automatic links; add only links
657
- supported by the note contents.
682
+ supported by the note contents. A search hit is a discovery excerpt, not the note: call **memory-get** and read
683
+ the complete note before relying on it for an answer, summary, edit, relationship, or durable write. Read the
684
+ strong candidates, not every low-ranked hit.
658
685
 
659
686
  Scrape deposits are raw evidence and therefore go to **Library** through **library-ingest**, with the full
660
687
  Library template and source metadata. If the source contains durable applicable guidance, create a separate
661
688
  Knowledge companion through prepare-memory-write + memory-capture and link it to the Library source with
662
689
  derived_from; do not replace the raw source with the guide.
663
690
 
664
- For an update, call **memory-get** first and pass its revision as baseRevision. Preserve the existing
665
- title, source, capture time, content context, and props unless the request deliberately changes them.
691
+ For an update, identify the existing note by its stable \`vault + path\`, call **memory-get** first, and pass
692
+ its revision as \`baseRevision\`. **memory-put replaces the entire content body; it is not a text patch.**
693
+ Merge the requested change into the full body returned by memory-get, then send that complete merged body.
694
+ Its supplied \`props\` patch existing metadata, so omit unchanged props and use an empty array only when the
695
+ user deliberately wants to clear a link list. Preserve the existing title, source, capture time, surrounding
696
+ content, links, template sections, and unsupported uncertainty unless the request deliberately changes them.
697
+ A title change does not create a new note identity; keep the path unless the user explicitly asks to move it.
698
+ Never convert an edit into a new path just because search returned a similar title. If \`baseRevision\`
699
+ conflicts, reconcile the returned current body with the requested change and retry against the new revision;
700
+ do not resend the stale full body unchanged. Treat Agent Inbox, optimizer rollups, channel messages, and other
701
+ system-managed notes through their purpose-built tools when available rather than rewriting their backing
702
+ notes with memory-put.
666
703
  If a legacy backend does not return props, do not overwrite an existing linked note through that surface;
667
704
  report that link preservation cannot be proved. After the write, read it back and search a distinctive
668
705
  phrase to verify persistence and indexing.
@@ -2665,7 +2702,8 @@ ${t.topQuestions.map((q) => `- ${q}`).join("\n")}` : ""
2665
2702
  ${q.threadUrl}` : ""}`).join("\n");
2666
2703
  const full = [
2667
2704
  `# Reddit Trending: "${d.topic || input.topic}"${subreddit ? ` in r/${subreddit}` : ""}`,
2668
- `**${totals.threads} threads \xB7 ${totals.upvotes} upvotes \xB7 ${totals.comments} comments** \xB7 last ${d.window === "7d" ? "week" : "month"} \xB7 ${d.threadsScraped ?? 0} of ${d.candidatesFound ?? threads.length} discovered scraped${d.partial ? " (partial \u2014 hit the time limit)" : ""}`,
2705
+ `**${totals.threads} threads \xB7 ${totals.upvotes} upvotes \xB7 ${totals.comments} comments** \xB7 last ${d.window === "7d" ? "week" : "month"} \xB7 ${d.threadsScraped ?? 0} of ${d.candidatesFound ?? threads.length} discovered scraped${d.partial ? " (partial)" : ""}`,
2706
+ d.degradedResult ? `**Discovery degraded:** ${(d.degradationReasons ?? []).join(", ") || "no usable discovery surface"}${d.retryRecommended ? " \xB7 retry recommended" : ""}${d.billingRefunded ? " \xB7 discovery charge refunded" : ""}` : d.discoverySource === "reddit_search_fallback" ? "**Discovery fallback:** direct Reddit search was used after the primary SERP returned no usable threads." : "",
2669
2707
  `
2670
2708
  ## Ranked threads
2671
2709
  ${threadBlocks || "_No threads found._"}`,
@@ -2697,7 +2735,13 @@ ${questionList || "_No questions extracted._"}`,
2697
2735
  threadsScraped: Number(d.threadsScraped ?? 0),
2698
2736
  candidatesFound: Number(d.candidatesFound ?? threads.length),
2699
2737
  partial: Boolean(d.partial),
2700
- searchQuery: d.searchQuery ?? ""
2738
+ searchQuery: d.searchQuery ?? "",
2739
+ discoverySource: d.discoverySource ?? "google_serp",
2740
+ resultQuality: d.resultQuality ?? "complete",
2741
+ degradedResult: Boolean(d.degradedResult),
2742
+ degradationReasons: d.degradationReasons ?? [],
2743
+ retryRecommended: Boolean(d.retryRecommended),
2744
+ billingRefunded: Boolean(d.billingRefunded)
2701
2745
  }
2702
2746
  };
2703
2747
  }
@@ -5607,7 +5651,22 @@ var EditorialReadingRoomSiteSchema = z4.object({
5607
5651
  issueLabel: z4.string().trim().min(1).max(100).default("Current edition").describe("Issue, date, or collection label in the home-page issue line."),
5608
5652
  eyebrow: z4.string().trim().min(1).max(120).default("A guided collection").describe("Short editorial eyebrow above the home-page headline."),
5609
5653
  heroTitle: z4.string().trim().min(1).max(180).describe("Outcome-led home-page headline for the whole reading room."),
5610
- startLabel: z4.string().trim().min(1).max(60).default("Start reading").describe("Label for the primary start-reading button.")
5654
+ startLabel: z4.string().trim().min(1).max(60).default("Start reading").describe("Label for the primary start-reading button."),
5655
+ ogImage: z4.object({
5656
+ url: z4.string().url().refine((value) => /^https?:\/\//i.test(value), "Image URL must use HTTP or HTTPS.").describe("Public HTTP(S) image URL used for the reading-room home page social preview."),
5657
+ alt: z4.string().trim().min(1).max(500).describe("Accessible description and og:image:alt text."),
5658
+ width: z4.number().int().positive().optional().describe("Optional intrinsic width in pixels."),
5659
+ height: z4.number().int().positive().optional().describe("Optional intrinsic height in pixels.")
5660
+ }).strict().optional().describe("Optional collection-level Open Graph image. Individual articles may override it.")
5661
+ }).strict();
5662
+ var EditorialReadingRoomImageSchema = z4.object({
5663
+ url: z4.string().url().refine((value) => /^https?:\/\//i.test(value), "Image URL must use HTTP or HTTPS.").describe("Public HTTP(S) image URL."),
5664
+ alt: z4.string().trim().min(1).max(500).describe("Required accessible description of the image."),
5665
+ caption: z4.string().trim().max(1e3).optional().describe("Optional visible caption."),
5666
+ credit: z4.string().trim().max(500).optional().describe("Optional visible creator, publisher, or rights credit."),
5667
+ sourceUrl: z4.string().url().refine((value) => /^https?:\/\//i.test(value), "Source URL must use HTTP or HTTPS.").optional().describe("Optional public HTTP(S) source page for provenance or rights context."),
5668
+ width: z4.number().int().positive().optional().describe("Optional intrinsic width in pixels."),
5669
+ height: z4.number().int().positive().optional().describe("Optional intrinsic height in pixels.")
5611
5670
  }).strict();
5612
5671
  var EditorialReadingRoomArticleSchema = z4.object({
5613
5672
  slug: z4.string().trim().regex(/^[a-z0-9]+(?:-[a-z0-9]+)*$/).max(80).describe("Unique kebab-case article identifier."),
@@ -5620,7 +5679,9 @@ var EditorialReadingRoomArticleSchema = z4.object({
5620
5679
  sourceLabel: z4.string().trim().min(1).max(500).describe("Visible provenance label naming the material this article was derived from. Do not invent a source."),
5621
5680
  revision: z4.string().trim().min(1).max(80).optional().describe("Optional revision identifier or version label."),
5622
5681
  updatedAt: z4.string().trim().min(1).max(80).optional().describe("Optional human-readable source update date."),
5623
- markdown: z4.string().min(1).max(1e5).describe("Complete article body in Markdown. Use H2/H3 headings for jump links, short paragraphs, concrete examples, and tables only where they improve comparison.")
5682
+ image: EditorialReadingRoomImageSchema.optional().describe("Optional article image shown on cards and above the article body. Markdown images remain supported inside the body."),
5683
+ ogImage: EditorialReadingRoomImageSchema.optional().describe("Optional article-specific social preview image. Defaults to article.image, then site.ogImage."),
5684
+ markdown: z4.string().min(1).max(1e5).describe("Complete article body in Markdown. Standard Markdown images are allowed with descriptive alt text. Use H2/H3 headings for jump links, short paragraphs, concrete examples, and tables only where they improve comparison.")
5624
5685
  }).strict();
5625
5686
  var EditorialReadingRoomGuideInputSchema = {
5626
5687
  focus: z4.enum(["workflow", "content_contract", "example"]).default("workflow").describe("Which part of the reusable editorial-reading-room guide to return. Start with workflow; fetch the content contract or compact example only when needed.")
@@ -5633,7 +5694,7 @@ var EditorialReadingRoomGuideOutputSchema = {
5633
5694
  var CreateEditorialReadingRoomInputSchema = {
5634
5695
  site: EditorialReadingRoomSiteSchema,
5635
5696
  deck: z4.string().trim().min(1).max(1e3).describe("Two or three sentences that explain the collection\u2019s value and scope without generic marketing language."),
5636
- articles: z4.array(EditorialReadingRoomArticleSchema).min(1).max(40).describe("One to forty fully authored articles, with no more than 2,000,000 Markdown bytes combined. Read all in-scope source material before composing them; preserve distinctions, uncertainty, and provenance instead of flattening the corpus."),
5697
+ articles: z4.array(EditorialReadingRoomArticleSchema).min(1).max(100).describe("One to one hundred fully authored articles, with no more than 2,000,000 Markdown bytes combined. Articles may include structured card/hero images, article-specific Open Graph images, and Markdown body images. Read all in-scope source material before composing them; preserve distinctions, uncertainty, image provenance, and rights context instead of flattening the corpus."),
5637
5698
  filename: z4.string().trim().regex(/^[a-zA-Z0-9][a-zA-Z0-9._-]*$/).max(120).optional().describe("Optional download filename. The server always normalizes it to a safe .html filename.")
5638
5699
  };
5639
5700
  var EditorialReadingRoomArtifactSchema = z4.object({
@@ -6460,7 +6521,13 @@ var RedditTrendingOutputSchema = {
6460
6521
  threadsScraped: z4.number().int().min(0),
6461
6522
  candidatesFound: z4.number().int().min(0),
6462
6523
  partial: z4.boolean(),
6463
- searchQuery: z4.string()
6524
+ searchQuery: z4.string(),
6525
+ discoverySource: z4.enum(["google_serp", "reddit_search_fallback", "none"]),
6526
+ resultQuality: z4.enum(["complete", "partial", "degraded"]),
6527
+ degradedResult: z4.boolean(),
6528
+ degradationReasons: z4.array(z4.string()),
6529
+ retryRecommended: z4.boolean(),
6530
+ billingRefunded: z4.boolean()
6464
6531
  };
6465
6532
  var FacebookPageIntelOutputSchema = {
6466
6533
  advertiserName: NullableString,
@@ -7626,10 +7693,11 @@ var WORKFLOW_GUIDE = `Create an editorial reading room only after doing the inte
7626
7693
  2. Read every in-scope source before designing the page. Build a source inventory with purpose, authority, overlap, and provenance.
7627
7694
  3. Architect the corpus into a small editorial edition. Prefer one article per real question, decision, lesson, or reusable pattern. Merge duplicate material; preserve meaningful distinctions and uncertainty.
7628
7695
  4. Write the collection-level promise: a specific hero title and deck that explain what the reader will understand or be able to do.
7629
- 5. Author complete articles in Markdown. Use H2/H3 headings as a useful table of contents, short paragraphs, concrete examples, and tables only when they materially improve comparison.
7630
- 6. Preserve provenance in sourceLabel. Never invent a source, fact, result, quote, or certainty that the source material does not support.
7631
- 7. Call create_editorial_reading_room with the finished site metadata, deck, and ordered articles. Do not ask the renderer to discover, research, or write the content for you.
7632
- 8. Inspect the returned page at both mobile and desktop sizes. Verify navigation, jump links, article order, typography, overflow, source labels, and that the page opens without a build step.
7696
+ 5. Author complete articles in Markdown. Use H2/H3 headings as a useful table of contents, short paragraphs, concrete examples, and tables only when they materially improve comparison. Markdown images are supported inside the body.
7697
+ 6. Choose images only when they help the reader identify, understand, compare, or remember the material. Use article.image for the card/article hero, article.ogImage for a distinct social preview, and site.ogImage for the collection default. Every image needs descriptive alt text; preserve caption, credit, source URL, and rights context when known. Never invent attribution or imply usage rights.
7698
+ 7. Preserve provenance in sourceLabel. Never invent a source, fact, result, quote, or certainty that the source material does not support.
7699
+ 8. Call create_editorial_reading_room with the finished site metadata, deck, and ordered articles. Do not ask the renderer to discover, research, or write the content for you.
7700
+ 9. Inspect the returned page at both mobile and desktop sizes. Verify navigation, jump links, article order, images, alt text, typography, overflow, source labels, social metadata, and that the page opens without a build step.
7633
7701
 
7634
7702
  The renderer owns the reusable New York Times-inspired reading surface, responsive hamburger navigation, search, article jump links, progress, text-size controls, evening mode, and portable single-file HTML delivery.`;
7635
7703
  var CONTENT_CONTRACT = `Editorial reading-room content contract
@@ -7642,13 +7710,16 @@ var CONTENT_CONTRACT = `Editorial reading-room content contract
7642
7710
  - site.eyebrow: collection framing, not a duplicate headline.
7643
7711
  - site.heroTitle: specific outcome or understanding promised by the collection.
7644
7712
  - site.startLabel: concise reading CTA.
7713
+ - site.ogImage: optional collection-level Open Graph image with required alt text.
7645
7714
  - deck: two or three concrete sentences covering value and scope.
7646
- - articles: 1-40 complete editorial pieces, each with a unique slug and order.
7715
+ - articles: 1-100 complete editorial pieces, each with a unique slug and order.
7647
7716
  - article.category: a repeated grouping label when multiple pieces belong together.
7648
7717
  - article.kicker: short framing line.
7649
7718
  - article.title: the question, decision, lesson, or reusable pattern.
7650
7719
  - article.summary: what the reader will understand.
7651
7720
  - article.sourceLabel: visible, truthful provenance.
7721
+ - article.image: optional card/article hero image with alt text and optional caption, credit, source URL, and dimensions.
7722
+ - article.ogImage: optional article-specific social image; otherwise article.image, then site.ogImage is used.
7652
7723
  - article.markdown: full body with useful H2/H3 headings.
7653
7724
 
7654
7725
  Quality gates:
@@ -7657,6 +7728,7 @@ Quality gates:
7657
7728
  - Do not use placeholder copy.
7658
7729
  - Keep headings descriptive enough to work as jump links.
7659
7730
  - Prefer a coherent reading sequence over source-file order.
7731
+ - Use only public HTTP(S) image URLs. Preserve provenance and rights context; an available URL is not proof of reuse rights.
7660
7732
  - If the corpus is too thin for multiple articles, create one strong article instead of padding the edition.`;
7661
7733
  var EXAMPLE = `Compact example:
7662
7734
 
@@ -7670,7 +7742,11 @@ var EXAMPLE = `Compact example:
7670
7742
  "issueLabel": "Pattern library",
7671
7743
  "eyebrow": "A practical field guide",
7672
7744
  "heroTitle": "Make every saved note easier to find and reuse",
7673
- "startLabel": "Begin with the pattern"
7745
+ "startLabel": "Begin with the pattern",
7746
+ "ogImage": {
7747
+ "url": "https://example.com/reading-room-social.jpg",
7748
+ "alt": "Connected notes arranged as a navigable editorial collection"
7749
+ }
7674
7750
  },
7675
7751
  "deck": "A concise guide to placing new material in the right vault, connecting it to natural neighbors, and preserving source evidence without turning memory into a generic notes bucket.",
7676
7752
  "articles": [{
@@ -7682,6 +7758,12 @@ var EXAMPLE = `Compact example:
7682
7758
  "summary": "A useful note is not only stored; it is classified, connected, and made retrievable.",
7683
7759
  "sourceType": "Workflow synthesis",
7684
7760
  "sourceLabel": "Derived from the MCP Memory capture contract supplied by the user.",
7761
+ "image": {
7762
+ "url": "https://example.com/graph-operation.jpg",
7763
+ "alt": "A source note connected to related project and knowledge notes",
7764
+ "caption": "Useful captures become connected retrieval objects.",
7765
+ "sourceUrl": "https://example.com/source"
7766
+ },
7685
7767
  "markdown": "## Start with the destination\\n\\nChoose the vault whose job matches the material...\\n\\n## Connect natural neighbors\\n\\nLink only relationships supported by the notes..."
7686
7768
  }]
7687
7769
  }`;
@@ -8532,7 +8614,7 @@ function registerPaaExtractorMcpTools(server, executor, options = {}) {
8532
8614
  }, async (input) => formatRedditThread(await executor.redditThread(input), input));
8533
8615
  server.registerTool("reddit_trending", {
8534
8616
  title: "Reddit Trending",
8535
- description: "Discover the top Reddit conversations about a topic from the last week or month: finds relevant recent threads via a Google site:reddit.com search (optionally scoped to one subreddit), scrapes them for real upvotes, comments, and the questions people asked, and ranks by engagement (upvotes + 2x comments). Scraping runs in parallel across the discovered threads; set includeComments:false for a fast, cheap discovery-only sweep (relevant thread list, no engagement stats, no per-thread billing) and then read the ones you want with reddit_thread. Not for reading one known thread URL \u2014 use reddit_thread for that.",
8617
+ description: "Discover top Reddit conversations from the last week or month. It tries Google site:reddit.com discovery, falls back to a bounded direct Reddit search when that SERP is empty or unavailable, then optionally scrapes threads for real upvotes, comments, questions, and engagement ranking. Inspect resultQuality, discoverySource, degradationReasons, retryRecommended, and billingRefunded before treating an empty result as a genuine lack of discussion. Set includeComments:false for a cheap discovery-only sweep; use reddit_thread for one known URL.",
8536
8618
  inputSchema: RedditTrendingInputSchema,
8537
8619
  outputSchema: recordOutputSchema("reddit_trending", RedditTrendingOutputSchema),
8538
8620
  annotations: liveWebToolAnnotations("Reddit Trending")
@@ -8771,7 +8853,7 @@ function registerPaaExtractorMcpTools(server, executor, options = {}) {
8771
8853
  }, async (input) => executor.commonsPreparePublication(input));
8772
8854
  server.registerTool("commons_validate_publication", {
8773
8855
  title: "Validate Transparent Commons Publication",
8774
- description: "Validate a publication name claim or a complete source-grounded editorial edition without writing. Use operation claim before commons_claim_publication and operation publish before commons_publish_editorial. This uses the editorial reading-room contract and checks ownership plus revision conflicts.",
8856
+ description: "Validate a publication name claim or a complete source-grounded editorial edition without writing. Publish validation accepts up to 100 articles plus structured article/card images, Markdown body images, and collection/article Open Graph images under the editorial reading-room contract. Use operation claim before commons_claim_publication and operation publish before commons_publish_editorial; ownership and revision conflicts are checked.",
8775
8857
  inputSchema: CommonsValidatePublicationInputSchema,
8776
8858
  outputSchema: recordOutputSchema("commons_validate_publication", CommonsGenericOutputSchema),
8777
8859
  annotations: { title: "Validate Transparent Commons Publication", readOnlyHint: true, destructiveHint: false, idempotentHint: true, openWorldHint: false }
@@ -8785,7 +8867,7 @@ function registerPaaExtractorMcpTools(server, executor, options = {}) {
8785
8867
  }, async (input) => executor.commonsClaimPublication(input));
8786
8868
  server.registerTool("commons_publish_editorial", {
8787
8869
  title: "Publish Transparent Commons Editorial Edition",
8788
- description: "Publish a fully authored editorial reading-room edition to the caller-owned Transparent Commons subdomain and return permanent root, archive, and edition URLs. The calling AI must research and author the source-grounded edition first; this tool validates, renders, and persists it. For an existing edition, pass its current baseRevision. Requires an idempotencyKey; this is not the neutral wiki write tool.",
8870
+ description: "Publish up to 100 fully authored editorial pieces, including optional article/card images, Markdown body images, and collection/article Open Graph images, to the caller-owned Transparent Commons subdomain. The calling AI must research and author the source-grounded edition first; image URLs need alt text and preserved provenance/rights context. The tool validates, renders, persists, and returns permanent root, archive, and edition URLs. Existing editions require current baseRevision. Requires idempotencyKey; this is not the neutral wiki write tool.",
8789
8871
  inputSchema: CommonsPublishEditorialInputSchema,
8790
8872
  outputSchema: recordOutputSchema("commons_publish_editorial", CommonsGenericOutputSchema),
8791
8873
  annotations: { title: "Publish Transparent Commons Editorial Edition", readOnlyHint: false, destructiveHint: false, idempotentHint: true, openWorldHint: true }
@@ -8922,7 +9004,7 @@ function registerPaaExtractorMcpTools(server, executor, options = {}) {
8922
9004
  }, async (input) => formatWorkflowArtifactRead(await executor.workflowArtifactRead(input), input));
8923
9005
  server.registerTool("editorial_reading_room_guide", {
8924
9006
  title: "Editorial Reading Room Guide",
8925
- description: 'Read the reusable composition contract before creating an editorial reading room. It tells the calling AI how to inventory the supplied corpus, preserve source truth, architect a coherent edition, write useful articles, and verify the finished page. Start with focus "workflow"; fetch "content_contract" or "example" only when needed. This does not research, write, or create a page.',
9007
+ description: 'Read the reusable composition contract before creating an editorial reading room. It covers corpus inventory, source truth, coherent architecture, up to 100 articles, purposeful images, image provenance, collection/article Open Graph images, and finished-page verification. Start with focus "workflow"; fetch "content_contract" or "example" only when needed. This does not research, write, or create a page.',
8926
9008
  inputSchema: EditorialReadingRoomGuideInputSchema,
8927
9009
  outputSchema: recordOutputSchema("editorial_reading_room_guide", EditorialReadingRoomGuideOutputSchema),
8928
9010
  annotations: localPlanningToolAnnotations("Editorial Reading Room Guide")
@@ -8938,7 +9020,7 @@ function registerPaaExtractorMcpTools(server, executor, options = {}) {
8938
9020
  }));
8939
9021
  server.registerTool("create_editorial_reading_room", {
8940
9022
  title: "Create Editorial Reading Room",
8941
- description: `Turn fully authored, source-grounded articles into one polished mobile-first editorial report with contents, search, hamburger navigation, article jump links, reading progress, text sizing, evening mode, and provenance. Do not use this tool to discover, preview, save, or version reusable website templates; use list_artifact_templates and get_artifact_template_example for that workflow. The calling AI must first read all in-scope material and use editorial_reading_room_guide when it has not already internalized the workflow; this renderer does not perform research or invent copy. ${fileBehavior("Local stdio clients save one self-contained HTML file under the MCP Scraper output directory and return localPath.", "Hosted/app clients receive an owner-scoped private HTML artifact retained for seven days with a renewable signed download URL.")}`,
9023
+ description: `Render up to 100 fully authored, source-grounded articles into one polished mobile-first editorial report with contents, search, navigation, jump links, progress, text sizing, evening mode, provenance, structured article/card images, Markdown body images, and collection/article Open Graph metadata. Public HTTP(S) image URLs require alt text; preserve caption, credit, source, and rights context when known. The static artifact carries collection OG tags and updates article tags in-browser; a publishing host must serve article-specific metadata for crawler-perfect per-article unfurls. This renderer does not research or invent copy. For reusable website templates use list_artifact_templates instead. ${fileBehavior("Local stdio clients save one self-contained HTML file under the MCP Scraper output directory and return localPath.", "Hosted/app clients receive an owner-scoped private HTML artifact retained for seven days with a renewable signed download URL.")}`,
8942
9024
  inputSchema: CreateEditorialReadingRoomInputSchema,
8943
9025
  outputSchema: recordOutputSchema("create_editorial_reading_room", CreateEditorialReadingRoomOutputSchema),
8944
9026
  annotations: {
@@ -9138,6 +9220,29 @@ function registerPaaExtractorMcpTools(server, executor, options = {}) {
9138
9220
 
9139
9221
  // src/mcp/http-mcp-tool-executor.ts
9140
9222
  import { createHash as createHash4, randomUUID as randomUUID2 } from "crypto";
9223
+ function unclassifiedTransportFailure(path, err) {
9224
+ const name = err instanceof Error ? err.name : typeof err;
9225
+ const detail = err instanceof Error ? err.message : String(err);
9226
+ console.error(JSON.stringify({
9227
+ event: "mcp_executor_transport_failed",
9228
+ path,
9229
+ error_name: name,
9230
+ error_message: sanitizeVendorName(detail).slice(0, 400)
9231
+ }));
9232
+ return {
9233
+ content: [{
9234
+ type: "text",
9235
+ text: JSON.stringify({
9236
+ error: "service_unavailable",
9237
+ error_type: name,
9238
+ retryable: true,
9239
+ path,
9240
+ message: publicErrorMessage("service_unavailable")
9241
+ })
9242
+ }],
9243
+ isError: true
9244
+ };
9245
+ }
9141
9246
  function youtubeVideoIdFromUrl(url) {
9142
9247
  if (!url) return null;
9143
9248
  try {
@@ -9238,7 +9343,7 @@ var HttpMcpToolExecutor = class {
9238
9343
  const rawOverride = process.env.MCP_SCRAPER_HTTP_TIMEOUT_MS;
9239
9344
  const parsedOverride = rawOverride === void 0 ? NaN : Number(rawOverride);
9240
9345
  this.httpTimeoutOverrideMs = Number.isFinite(parsedOverride) && parsedOverride > 0 ? parsedOverride : null;
9241
- this.timeoutMs = this.httpTimeoutOverrideMs ?? 11e4;
9346
+ this.timeoutMs = this.httpTimeoutOverrideMs ?? DEFAULT_TOOL_CLIENT_TIMEOUT_MS;
9242
9347
  const configuredSerpIntelligenceTimeoutMs = Number(process.env.MCP_SCRAPER_SERP_INTELLIGENCE_HTTP_TIMEOUT_MS ?? this.timeoutMs);
9243
9348
  this.serpIntelligenceTimeoutMs = Number.isFinite(configuredSerpIntelligenceTimeoutMs) && configuredSerpIntelligenceTimeoutMs > 0 ? configuredSerpIntelligenceTimeoutMs : this.timeoutMs;
9244
9349
  }
@@ -9277,7 +9382,7 @@ var HttpMcpToolExecutor = class {
9277
9382
  isError: true
9278
9383
  };
9279
9384
  }
9280
- return { content: [{ type: "text", text: publicErrorMessage("service_unavailable") }], isError: true };
9385
+ return unclassifiedTransportFailure(path, err);
9281
9386
  }
9282
9387
  }
9283
9388
  async callConnectedMutation(path, body, idempotencyKey, timeoutMs = this.timeoutMs) {
@@ -9311,8 +9416,8 @@ var HttpMcpToolExecutor = class {
9311
9416
  return { content: [{ type: "text", text: JSON.stringify(httpErrorPayload(path, res, data)) }], isError: true };
9312
9417
  }
9313
9418
  return { content: [{ type: "text", text: JSON.stringify(data) }] };
9314
- } catch {
9315
- return { content: [{ type: "text", text: publicErrorMessage("service_unavailable") }], isError: true };
9419
+ } catch (err) {
9420
+ return unclassifiedTransportFailure(path, err);
9316
9421
  }
9317
9422
  }
9318
9423
  async getTextArtifact(path, maxBytes, timeoutMs = this.timeoutMs) {
@@ -9349,7 +9454,7 @@ var HttpMcpToolExecutor = class {
9349
9454
  }]
9350
9455
  };
9351
9456
  } catch (err) {
9352
- return { content: [{ type: "text", text: publicErrorMessage("service_unavailable") }], isError: true };
9457
+ return unclassifiedTransportFailure(path, err);
9353
9458
  }
9354
9459
  }
9355
9460
  harvestPaa(input) {
@@ -12559,7 +12664,7 @@ var DeleteNoteSchema = {
12559
12664
  var ExportSchema = {
12560
12665
  id: "memory-export",
12561
12666
  upstreamName: "exportTool",
12562
- description: "Export every note in a vault as a full dump for backup, migration, or bulk download \u2014 path, title, full content, kind, and last-updated per note, plus a count. Defaults to the active (or first entitled) vault. Requires export scope; the export is logged to provenance.",
12667
+ description: "Export every full note in one vault for an explicitly requested backup, migration, or corpus-wide audit \u2014 path, title, content, kind, last-updated, and count. This can be large and is not the normal way to find or edit one note: use memory-list for complete metadata inventory, memory-search for ranked recall, and memory-get for exact content. Defaults to the active or first entitled vault. Requires export scope; the export is logged to provenance.",
12563
12668
  input: {
12564
12669
  vault: z9.string().optional().describe(
12565
12670
  "Vault to export. Optional; defaults to the session active vault, then the first vault the caller is entitled to."
@@ -12611,7 +12716,7 @@ var getTool_notePropsSchema = z9.object({
12611
12716
  var GetSchema = {
12612
12717
  id: "memory-get",
12613
12718
  upstreamName: "getTool",
12614
- description: "Read a single note from a vault by its exact path, or by shareId for a note shared with you and accepted. Owned notes include their stored Obsidian props so edits can preserve links and template metadata. Returns a revision number \u2014 pass it as baseRevision on a later memory-put/delete-note to detect a concurrent edit instead of silently overwriting it. Requires read scope.",
12719
+ description: "Read one complete note by exact vault+path, or by accepted shareId. After memory-search identifies a strong candidate, use this before relying on it for an answer, summary, edit, link, or durable write: search results are excerpts, not complete notes. Owned notes include stored Obsidian props so edits preserve links and template metadata. Returns the revision required as baseRevision on later edits/deletes. Requires read scope.",
12615
12720
  input: {
12616
12721
  vault: z9.string().optional().describe(
12617
12722
  "Vault to read from. Optional; defaults to the session active vault, then the first vault the caller is entitled to. Ignored when shareId is given."
@@ -12648,7 +12753,7 @@ var GetSchema = {
12648
12753
  var ListSchema = {
12649
12754
  id: "memory-list",
12650
12755
  upstreamName: "listTool",
12651
- description: "List notes in a vault \u2014 path, title, kind, tags, last-updated \u2014 optionally filtered by kind and/or tags (matches ANY given tag). Defaults to the active or first entitled vault; also returns vaults the caller is entitled to. Requires read scope.",
12756
+ description: "Return every note in one vault as a complete metadata inventory \u2014 stable path, title, kind, tags, and last-updated \u2014 plus a sorted list of every unique folder and nested folder represented by those paths. It contains no bodies. Use memory-get for exact full content, memory-search for ranked semantic recall, or memory-export only when explicitly asked for every full note. Defaults to the active or first entitled vault; also returns entitled vaults. Requires read scope.",
12652
12757
  input: {
12653
12758
  vault: z9.string().optional().describe(
12654
12759
  "Vault to list. Optional; defaults to the session active vault, then the first vault the caller is entitled to."
@@ -12668,6 +12773,7 @@ var ListSchema = {
12668
12773
  updatedAt: z9.string().describe("ISO-8601 timestamp of the note last update.")
12669
12774
  })
12670
12775
  ).optional().describe("The notes in the vault (metadata only, no content). Present when ok is true."),
12776
+ folders: z9.array(z9.string()).optional().describe("Sorted complete folder inventory derived from note paths, including represented nested parent folders. Present when ok is true."),
12671
12777
  vaults: z9.array(z9.string()).optional().describe("All vaults the caller is entitled to, for choosing a different vault to list."),
12672
12778
  error: z9.string().optional().describe("Human-readable failure reason when ok is false.")
12673
12779
  },
@@ -12702,7 +12808,7 @@ var putTool_notePropsSchema = z9.object({
12702
12808
  var PutSchema = {
12703
12809
  id: "memory-put",
12704
12810
  upstreamName: "putTool",
12705
- description: "Create or deliberately edit one note at a path in a memory vault; content is persisted and indexed for search. For normal new People, Organizations, Deals, Projects, Tasks, or Communication records, use prepare-memory-write then memory-capture: People must be real people, Organizations must be real organizations, Deals need a known party, Projects need a supported kind, Tasks can be independent Inbox todos or use a verified Project when linked, and draft emails need pending approval. For row-shaped datasets you'll filter/sort by exact value, use table-create/table-insert-rows/table-query instead. Ordinary vaults are indexed and shareable \u2014 never store real secrets there; use a secure vault (create-secure-vault) instead, which is never indexed or shareable and is encrypted at rest. Requires write scope.",
12811
+ description: "Create or deliberately edit one note at a stable vault+path. On an existing note, first call memory-get, merge the requested change into its full body, and pass that complete body plus baseRevision: content is replace-all, while supplied props patch the stored props. A conflict returns the current body for reconciliation; never retry the stale full body unchanged. Use purpose-built tools instead of rewriting system-managed Agent Inbox, optimizer, or channel backing notes. For normal new People, Organizations, Deals, Projects, Tasks, or Communication records, use prepare-memory-write then memory-capture. For row-shaped datasets use table tools. Ordinary vaults are indexed and shareable; store real secrets only in a secure vault. Requires write scope.",
12706
12812
  input: {
12707
12813
  vault: z9.string().optional().describe(
12708
12814
  "Vault to write to. Optional; defaults to the session active vault, then the first vault the caller is entitled to. On a default-provisioned account, pick the vault whose job matches the content (see the server instructions for the full 16-vault guide) rather than defaulting blindly \u2014 e.g. a lesson learned goes in Knowledge, the raw source it came from goes in Library, a broken feature goes in Issues, a named real-world initiative goes in Projects. Do not use this low-level tool to create ordinary People, Organizations, Deals, Projects, Tasks, or Communication records: first use prepare-memory-write then memory-capture so relationships and approval state are validated."
@@ -12710,9 +12816,9 @@ var PutSchema = {
12710
12816
  path: z9.string().optional().describe("Vault-relative note path to create or overwrite, e.g. projects/q3-plan. Writing an existing path replaces it. Required unless shareId is given."),
12711
12817
  shareId: z9.string().optional().describe("Edit a note someone individually shared with you and you accepted (accept-share), by its shareId, instead of vault+path. Requires the share to grant edit permission, and baseRevision is mandatory (get the current revision first) since you are editing alongside the owner and possibly others."),
12712
12818
  title: z9.string().optional().describe("Optional human-readable title; defaults are derived from the path when omitted."),
12713
- content: z9.string().min(1).describe("The full note body to store and index for semantic search. Must be non-empty."),
12819
+ content: z9.string().min(1).describe("The complete note body to store and index. On edit this replaces the prior body, so merge the requested change into the full memory-get content before calling; never pass only the changed fragment."),
12714
12820
  props: putTool_notePropsSchema.optional().describe("Obsidian note primitives plus vault-specific template fields. On edits, supplied fields patch the stored props instead of replacing the whole object; pass an empty array to deliberately clear a link list. Type/domain/folder also steer routing when no vault is given."),
12715
- baseRevision: z9.number().optional().describe("Revision the edit is based on (from a prior get/put). When provided, the write only applies if the note is still at this revision; otherwise it is rejected as a conflict instead of silently overwriting a concurrent edit. Omit for last-write-wins (fine for solo notes)."),
12821
+ baseRevision: z9.number().optional().describe("Revision the edit is based on (from memory-get/put). Always supply it when an AI edits an existing note; a mismatch rejects the write and returns current content for reconciliation. Omit only for an intentional new note or explicit last-write-wins migration."),
12716
12822
  tagDescriptions: z9.record(z9.string(), z9.string()).optional().describe("One-line meaning for any tag in props.tags that is new to the account, keyed by tag. Tags resolve against the account's existing vocabulary; new tags require a one-line description.")
12717
12823
  },
12718
12824
  output: {
@@ -12746,7 +12852,7 @@ var searchTool_primitiveValue = z9.union([z9.string(), z9.number(), z9.boolean()
12746
12852
  var SearchSchema = {
12747
12853
  id: "memory-search",
12748
12854
  upstreamName: "searchTool",
12749
- description: "Default Smart RAG search across accessible memory. Form 2-4 focused query variants, combine semantic matches with exact vault/tag/date/kind/type/metadata filters, expand one bounded hop of outgoing links and backlinks around strong seeds, then rerank. Defaults: retrieve/fuse 50 candidates, 8 graph seeds, 5 neighbors per seed, rerank to 30. Graph neighbors are candidates, never automatic winners or links. Before tagging or writing, also call list-memory-tags to inspect the complete vocabulary and reuse existing tags; read strong related notes before selecting links.",
12855
+ description: "Default first tool whenever the user wants to find, recall, understand, or connect Memory content without an exact vault+path. Hybrid Smart RAG combines 2-4 semantic query variants with exact vault/tag/date/kind/type/metadata matches, expands one bounded graph hop, then reranks. Results are ranked excerpts, never an exhaustive inventory or complete notes: deduplicate by vault/path and call memory-get on strong candidates before answering, summarizing, editing, linking, or writing. Use memory-list for every note, memory-suggest for title-only lookup, and memory-export only for explicit full-vault dumps. Before tagging or writing, also inspect list-memory-tags.",
12750
12856
  input: {
12751
12857
  vault: z9.string().optional().describe("Exact logical vault handle to search. Omit to search every entitled vault."),
12752
12858
  query: z9.string().min(1).describe("A focused semantic reformulation of the request."),
@@ -14467,4 +14573,4 @@ export {
14467
14573
  ScheduledResultsMcpExecutor,
14468
14574
  registerScheduledResultsMcpTools
14469
14575
  };
14470
- //# sourceMappingURL=chunk-64WBUDPC.js.map
14576
+ //# sourceMappingURL=chunk-O63VVCCE.js.map