crawlableseo 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- crawlableseo/__init__.py +42 -0
- crawlableseo/head.py +58 -0
- crawlableseo/indexnow.py +111 -0
- crawlableseo/integrations/__init__.py +0 -0
- crawlableseo/integrations/fastapi.py +83 -0
- crawlableseo/llmstxt.py +38 -0
- crawlableseo/page.py +82 -0
- crawlableseo/py.typed +0 -0
- crawlableseo/robots.py +60 -0
- crawlableseo/shell.py +134 -0
- crawlableseo/shell_marker.py +7 -0
- crawlableseo/site.py +270 -0
- crawlableseo/sitemap.py +45 -0
- crawlableseo/urls.py +36 -0
- crawlableseo-0.1.0.dist-info/METADATA +208 -0
- crawlableseo-0.1.0.dist-info/RECORD +18 -0
- crawlableseo-0.1.0.dist-info/WHEEL +4 -0
- crawlableseo-0.1.0.dist-info/licenses/LICENSE +21 -0
crawlableseo/__init__.py
ADDED
|
@@ -0,0 +1,42 @@
|
|
|
1
|
+
"""Make a client-rendered SPA crawlable from its own HTML shell.
|
|
2
|
+
|
|
3
|
+
No headless browser, no SSR framework, no third-party rendering service:
|
|
4
|
+
the server fills in the shell's ``<head>`` and mount node per URL, and the
|
|
5
|
+
same declaration produces robots.txt, sitemap.xml and llms.txt.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
from .head import head_block
|
|
11
|
+
from .indexnow import IndexNow
|
|
12
|
+
from .llmstxt import LlmsSection, llms_txt
|
|
13
|
+
from .page import MAX_DESCRIPTION, MAX_TITLE, MIN_DESCRIPTION, NotFound, Page
|
|
14
|
+
from .robots import DEFAULT_AI_CRAWLERS, robots_txt
|
|
15
|
+
from .shell import MARKER, render_shell
|
|
16
|
+
from .site import DynamicUrl, Site
|
|
17
|
+
from .sitemap import SitemapEntry, sitemap_xml
|
|
18
|
+
from .urls import page_url
|
|
19
|
+
|
|
20
|
+
__version__ = "0.1.0"
|
|
21
|
+
|
|
22
|
+
__all__ = [
|
|
23
|
+
"DEFAULT_AI_CRAWLERS",
|
|
24
|
+
"MARKER",
|
|
25
|
+
"MAX_DESCRIPTION",
|
|
26
|
+
"MAX_TITLE",
|
|
27
|
+
"MIN_DESCRIPTION",
|
|
28
|
+
"DynamicUrl",
|
|
29
|
+
"IndexNow",
|
|
30
|
+
"LlmsSection",
|
|
31
|
+
"NotFound",
|
|
32
|
+
"Page",
|
|
33
|
+
"Site",
|
|
34
|
+
"SitemapEntry",
|
|
35
|
+
"__version__",
|
|
36
|
+
"head_block",
|
|
37
|
+
"llms_txt",
|
|
38
|
+
"page_url",
|
|
39
|
+
"render_shell",
|
|
40
|
+
"robots_txt",
|
|
41
|
+
"sitemap_xml",
|
|
42
|
+
]
|
crawlableseo/head.py
ADDED
|
@@ -0,0 +1,58 @@
|
|
|
1
|
+
"""The block of tags written into <head> for one page."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import html
|
|
6
|
+
import json
|
|
7
|
+
|
|
8
|
+
from .page import Page
|
|
9
|
+
from .shell_marker import MARKER
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def head_block(
|
|
13
|
+
page: Page,
|
|
14
|
+
*,
|
|
15
|
+
canonical: str,
|
|
16
|
+
base_url: str,
|
|
17
|
+
site_name: str = "",
|
|
18
|
+
twitter_site: str = "",
|
|
19
|
+
default_og_image: str | None = None,
|
|
20
|
+
) -> str:
|
|
21
|
+
e = html.escape
|
|
22
|
+
img = page.og_image or default_og_image
|
|
23
|
+
if img and img.startswith("/"):
|
|
24
|
+
img = base_url.rstrip("/") + img
|
|
25
|
+
|
|
26
|
+
lines = [
|
|
27
|
+
MARKER,
|
|
28
|
+
f'<link rel="canonical" href="{e(canonical)}" />',
|
|
29
|
+
f'<meta name="robots" content="{page.robots_value()}" />',
|
|
30
|
+
'<meta property="og:type" content="website" />',
|
|
31
|
+
f'<meta property="og:title" content="{e(page.title)}" />',
|
|
32
|
+
f'<meta property="og:description" content="{e(page.description)}" />',
|
|
33
|
+
f'<meta property="og:url" content="{e(canonical)}" />',
|
|
34
|
+
'<meta name="twitter:card" content="summary_large_image" />',
|
|
35
|
+
f'<meta name="twitter:title" content="{e(page.title)}" />',
|
|
36
|
+
f'<meta name="twitter:description" content="{e(page.description)}" />',
|
|
37
|
+
]
|
|
38
|
+
if site_name:
|
|
39
|
+
lines.insert(4, f'<meta property="og:site_name" content="{e(site_name)}" />')
|
|
40
|
+
if twitter_site:
|
|
41
|
+
lines.append(f'<meta name="twitter:site" content="{e(twitter_site)}" />')
|
|
42
|
+
if img:
|
|
43
|
+
lines += [
|
|
44
|
+
f'<meta property="og:image" content="{e(img)}" />',
|
|
45
|
+
'<meta property="og:image:width" content="1200" />',
|
|
46
|
+
'<meta property="og:image:height" content="630" />',
|
|
47
|
+
f'<meta name="twitter:image" content="{e(img)}" />',
|
|
48
|
+
]
|
|
49
|
+
lines += [_jsonld_tag(obj) for obj in page.jsonld]
|
|
50
|
+
return "\n".join(lines)
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def _jsonld_tag(obj: dict[str, object]) -> str:
|
|
54
|
+
# A "</" anywhere in the data would close the script element early, so the
|
|
55
|
+
# rest of the document becomes the browser's problem. Escaping the slash
|
|
56
|
+
# is valid JSON and parses back to the same string.
|
|
57
|
+
payload = json.dumps(obj, separators=(",", ":"), ensure_ascii=False).replace("</", "<\\/")
|
|
58
|
+
return '<script type="application/ld+json">' + payload + "</script>"
|
crawlableseo/indexnow.py
ADDED
|
@@ -0,0 +1,111 @@
|
|
|
1
|
+
"""IndexNow: tell search engines a URL changed, instead of waiting to be crawled.
|
|
2
|
+
|
|
3
|
+
The protocol is small. The site publishes a key at ``/<key>.txt`` whose body
|
|
4
|
+
is the key itself, then POSTs changed URLs with that key. Bing, Yandex,
|
|
5
|
+
Seznam and Naver share one endpoint; Bing matters most because it feeds
|
|
6
|
+
ChatGPT search and Copilot.
|
|
7
|
+
|
|
8
|
+
The key is public by design - anyone can read the key file - and grants
|
|
9
|
+
nothing beyond submitting URLs for that one host.
|
|
10
|
+
|
|
11
|
+
IndexNow is an accelerator, never a dependency: engines re-crawl on their own
|
|
12
|
+
schedule regardless, so a failed submission is logged and dropped, never
|
|
13
|
+
raised into the caller's request.
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
from __future__ import annotations
|
|
17
|
+
|
|
18
|
+
import logging
|
|
19
|
+
import re
|
|
20
|
+
from collections.abc import Sequence
|
|
21
|
+
|
|
22
|
+
logger = logging.getLogger("crawlableseo.indexnow")
|
|
23
|
+
|
|
24
|
+
ENDPOINT = "https://api.indexnow.org/indexnow"
|
|
25
|
+
# The protocol caps one submission at 10,000 URLs. Batching well below that
|
|
26
|
+
# keeps a whole-sitemap resubmit to a handful of requests.
|
|
27
|
+
BATCH = 500
|
|
28
|
+
_KEY_RE = re.compile(r"^[A-Za-z0-9-]{8,128}$")
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
class IndexNow:
|
|
32
|
+
"""Client for one host.
|
|
33
|
+
|
|
34
|
+
``httpx`` is an optional dependency; install ``crawlableseo[indexnow]``
|
|
35
|
+
to use this class.
|
|
36
|
+
"""
|
|
37
|
+
|
|
38
|
+
def __init__(self, base_url: str, key: str | None, *, endpoint: str = ENDPOINT) -> None:
|
|
39
|
+
self.base_url = base_url.rstrip("/")
|
|
40
|
+
self.endpoint = endpoint
|
|
41
|
+
self.key = self._validated(key)
|
|
42
|
+
|
|
43
|
+
@staticmethod
|
|
44
|
+
def _validated(key: str | None) -> str | None:
|
|
45
|
+
k = (key or "").strip()
|
|
46
|
+
if not k:
|
|
47
|
+
return None
|
|
48
|
+
if not _KEY_RE.match(k):
|
|
49
|
+
logger.warning(
|
|
50
|
+
"indexnow: key must be 8-128 characters of [A-Za-z0-9-]; disabled"
|
|
51
|
+
)
|
|
52
|
+
return None
|
|
53
|
+
return k
|
|
54
|
+
|
|
55
|
+
@property
|
|
56
|
+
def enabled(self) -> bool:
|
|
57
|
+
return self.key is not None
|
|
58
|
+
|
|
59
|
+
@property
|
|
60
|
+
def key_path(self) -> str | None:
|
|
61
|
+
"""Path the key file must be served at, e.g. ``/abc123.txt``."""
|
|
62
|
+
return f"/{self.key}.txt" if self.key else None
|
|
63
|
+
|
|
64
|
+
def absolute(self, urls: Sequence[str]) -> list[str]:
|
|
65
|
+
out = []
|
|
66
|
+
for u in urls:
|
|
67
|
+
if not u:
|
|
68
|
+
continue
|
|
69
|
+
out.append(u if u.startswith("http") else self.base_url + u)
|
|
70
|
+
return out
|
|
71
|
+
|
|
72
|
+
async def submit(self, urls: Sequence[str]) -> bool:
|
|
73
|
+
"""POST the URLs. Returns False when disabled or when the call failed."""
|
|
74
|
+
if not self.key:
|
|
75
|
+
logger.debug("indexnow: no key configured, skipping %d url(s)", len(urls))
|
|
76
|
+
return False
|
|
77
|
+
try:
|
|
78
|
+
import httpx
|
|
79
|
+
except ImportError: # pragma: no cover - exercised by the extras test
|
|
80
|
+
logger.warning("indexnow: httpx is not installed; install crawlableseo[indexnow]")
|
|
81
|
+
return False
|
|
82
|
+
|
|
83
|
+
host = self.base_url.split("://", 1)[-1].split("/", 1)[0]
|
|
84
|
+
absolute = self.absolute(urls)
|
|
85
|
+
ok = True
|
|
86
|
+
async with httpx.AsyncClient(timeout=10.0) as client:
|
|
87
|
+
for start in range(0, len(absolute), BATCH):
|
|
88
|
+
chunk = absolute[start : start + BATCH]
|
|
89
|
+
payload = {
|
|
90
|
+
"host": host,
|
|
91
|
+
"key": self.key,
|
|
92
|
+
"keyLocation": f"{self.base_url}{self.key_path}",
|
|
93
|
+
"urlList": chunk,
|
|
94
|
+
}
|
|
95
|
+
try:
|
|
96
|
+
response = await client.post(self.endpoint, json=payload)
|
|
97
|
+
except Exception as exc:
|
|
98
|
+
logger.warning("indexnow: submission failed (%s)", exc)
|
|
99
|
+
ok = False
|
|
100
|
+
continue
|
|
101
|
+
if response.status_code >= 400:
|
|
102
|
+
logger.warning(
|
|
103
|
+
"indexnow: %s rejected %d url(s): %s",
|
|
104
|
+
response.status_code,
|
|
105
|
+
len(chunk),
|
|
106
|
+
response.text[:200],
|
|
107
|
+
)
|
|
108
|
+
ok = False
|
|
109
|
+
else:
|
|
110
|
+
logger.info("indexnow: submitted %d url(s)", len(chunk))
|
|
111
|
+
return ok
|
|
File without changes
|
|
@@ -0,0 +1,83 @@
|
|
|
1
|
+
"""Wire a :class:`~crawlableseo.site.Site` into FastAPI or Starlette.
|
|
2
|
+
|
|
3
|
+
``router(site)`` returns the crawler-facing endpoints plus the SPA catch-all.
|
|
4
|
+
Mount it last: the catch-all answers every path the rest of the app did not
|
|
5
|
+
claim.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
from collections.abc import Callable, Sequence
|
|
11
|
+
|
|
12
|
+
# Imported at module scope on purpose. FastAPI resolves a handler's
|
|
13
|
+
# annotations against its module's globals, so a Request imported inside the
|
|
14
|
+
# function is invisible to it: every request then fails validation with
|
|
15
|
+
# "query.request field required" instead of being served. This module is only
|
|
16
|
+
# imported by callers who already depend on FastAPI.
|
|
17
|
+
from fastapi import APIRouter, Request, Response
|
|
18
|
+
from fastapi.responses import HTMLResponse, PlainTextResponse
|
|
19
|
+
|
|
20
|
+
from ..indexnow import IndexNow
|
|
21
|
+
from ..llmstxt import LlmsSection
|
|
22
|
+
from ..robots import DEFAULT_AI_CRAWLERS
|
|
23
|
+
from ..site import Site
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def router(
|
|
27
|
+
site: Site,
|
|
28
|
+
*,
|
|
29
|
+
disallow: Sequence[str] = (),
|
|
30
|
+
ai_crawlers: Sequence[str] | None = DEFAULT_AI_CRAWLERS,
|
|
31
|
+
llms: Callable[[], str] | None = None,
|
|
32
|
+
indexnow: IndexNow | None = None,
|
|
33
|
+
shell_cache_seconds: int = 0,
|
|
34
|
+
text_cache_seconds: int = 3600,
|
|
35
|
+
catch_all: bool = True,
|
|
36
|
+
) -> APIRouter:
|
|
37
|
+
"""Build an ``APIRouter`` serving robots.txt, sitemap.xml, llms.txt and the SPA."""
|
|
38
|
+
api = APIRouter()
|
|
39
|
+
text_headers = {"Cache-Control": f"public, max-age={text_cache_seconds}"}
|
|
40
|
+
|
|
41
|
+
@api.get("/robots.txt", include_in_schema=False)
|
|
42
|
+
async def robots() -> Response:
|
|
43
|
+
return PlainTextResponse(
|
|
44
|
+
site.robots(disallow=disallow, ai_crawlers=ai_crawlers), headers=text_headers
|
|
45
|
+
)
|
|
46
|
+
|
|
47
|
+
@api.get("/sitemap.xml", include_in_schema=False)
|
|
48
|
+
async def sitemap() -> Response:
|
|
49
|
+
return Response(
|
|
50
|
+
site.sitemap(), media_type="application/xml", headers=text_headers
|
|
51
|
+
)
|
|
52
|
+
|
|
53
|
+
if llms is not None:
|
|
54
|
+
@api.get("/llms.txt", include_in_schema=False)
|
|
55
|
+
async def llms_route() -> Response:
|
|
56
|
+
return PlainTextResponse(llms(), headers=text_headers)
|
|
57
|
+
|
|
58
|
+
if indexnow is not None and indexnow.key_path:
|
|
59
|
+
key = indexnow.key
|
|
60
|
+
|
|
61
|
+
@api.get(indexnow.key_path, include_in_schema=False)
|
|
62
|
+
async def indexnow_key() -> Response:
|
|
63
|
+
# The body must be exactly the key: that is how the engine
|
|
64
|
+
# verifies the submitter controls this host.
|
|
65
|
+
return PlainTextResponse(
|
|
66
|
+
key or "", headers={"Cache-Control": "public, max-age=86400"}
|
|
67
|
+
)
|
|
68
|
+
|
|
69
|
+
if catch_all:
|
|
70
|
+
@api.get("/{path:path}", include_in_schema=False)
|
|
71
|
+
async def spa(path: str, request: Request) -> Response:
|
|
72
|
+
html, status = site.render("/" + path, dict(request.query_params))
|
|
73
|
+
headers = (
|
|
74
|
+
{"Cache-Control": f"public, max-age={shell_cache_seconds}"}
|
|
75
|
+
if shell_cache_seconds
|
|
76
|
+
else {"Cache-Control": "no-cache"}
|
|
77
|
+
)
|
|
78
|
+
return HTMLResponse(html, status_code=status, headers=headers)
|
|
79
|
+
|
|
80
|
+
return api
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
__all__ = ["LlmsSection", "router"]
|
crawlableseo/llmstxt.py
ADDED
|
@@ -0,0 +1,38 @@
|
|
|
1
|
+
"""llms.txt: a plain-language map of the site for models and agents.
|
|
2
|
+
|
|
3
|
+
The format is Markdown by convention: an H1 with the site's name, a
|
|
4
|
+
blockquote summarising it in one paragraph, then sections of links with a
|
|
5
|
+
sentence each. Keep it honest and specific; a model quoting this file will
|
|
6
|
+
quote whatever it says, including the parts that oversell.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
from collections.abc import Iterable, Sequence
|
|
12
|
+
from dataclasses import dataclass
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
@dataclass(frozen=True)
|
|
16
|
+
class LlmsSection:
|
|
17
|
+
title: str
|
|
18
|
+
links: Sequence[tuple[str, str, str]]
|
|
19
|
+
"""``(label, url, one-line description)`` per link."""
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def llms_txt(
|
|
23
|
+
name: str,
|
|
24
|
+
summary: str,
|
|
25
|
+
sections: Iterable[LlmsSection],
|
|
26
|
+
*,
|
|
27
|
+
notes: Sequence[str] = (),
|
|
28
|
+
) -> str:
|
|
29
|
+
out = [f"# {name}", "", f"> {summary}", ""]
|
|
30
|
+
for note in notes:
|
|
31
|
+
out += [note, ""]
|
|
32
|
+
for section in sections:
|
|
33
|
+
out.append(f"## {section.title}")
|
|
34
|
+
out.append("")
|
|
35
|
+
for label, url, description in section.links:
|
|
36
|
+
out.append(f"- [{label}]({url}): {description}")
|
|
37
|
+
out.append("")
|
|
38
|
+
return "\n".join(out).rstrip() + "\n"
|
crawlableseo/page.py
ADDED
|
@@ -0,0 +1,82 @@
|
|
|
1
|
+
"""The one description of a page, shared by every output the library makes."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from dataclasses import dataclass, field, replace
|
|
6
|
+
|
|
7
|
+
# Bing's site scanner reports a description outside this range as an SEO
|
|
8
|
+
# error, and Google truncates at roughly the same point.
|
|
9
|
+
MIN_DESCRIPTION = 50
|
|
10
|
+
MAX_DESCRIPTION = 160
|
|
11
|
+
# Google renders about this much of a title before the ellipsis.
|
|
12
|
+
MAX_TITLE = 60
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
@dataclass(frozen=True)
|
|
16
|
+
class Page:
|
|
17
|
+
"""Everything the shell, the sitemap and llms.txt need about one URL.
|
|
18
|
+
|
|
19
|
+
``body`` is crawlable HTML placed inside the mount node. The framework
|
|
20
|
+
replaces the node's children when it mounts, so users never see it; a
|
|
21
|
+
crawler that does not run JavaScript reads it as the page's content.
|
|
22
|
+
"""
|
|
23
|
+
|
|
24
|
+
title: str
|
|
25
|
+
description: str
|
|
26
|
+
body: str = ""
|
|
27
|
+
# Extra query parameters that belong in the canonical URL. A parameter
|
|
28
|
+
# that changes the content belongs here; a tracking parameter does not.
|
|
29
|
+
params: dict[str, str] = field(default_factory=dict)
|
|
30
|
+
jsonld: tuple[dict[str, object], ...] = ()
|
|
31
|
+
og_image: str | None = None
|
|
32
|
+
index: bool = True
|
|
33
|
+
# Kept separate from ``index``: a page can be worth excluding from the
|
|
34
|
+
# index while its outgoing links are still worth crawling.
|
|
35
|
+
follow: bool = True
|
|
36
|
+
status: int = 200
|
|
37
|
+
# Sitemap hints. ``in_sitemap=False`` keeps a real page out of it.
|
|
38
|
+
in_sitemap: bool = True
|
|
39
|
+
changefreq: str = "weekly"
|
|
40
|
+
priority: float = 0.5
|
|
41
|
+
lastmod: str | None = None
|
|
42
|
+
|
|
43
|
+
def robots_value(self) -> str:
|
|
44
|
+
if self.index:
|
|
45
|
+
return "index, follow" if self.follow else "index, nofollow"
|
|
46
|
+
return "noindex, follow" if self.follow else "noindex, nofollow"
|
|
47
|
+
|
|
48
|
+
def clipped(self) -> Page:
|
|
49
|
+
"""Title and description cut to the lengths search engines keep.
|
|
50
|
+
|
|
51
|
+
Applied once, where the tags are written, rather than trusted to
|
|
52
|
+
every caller: pages built from database strings have no length limit
|
|
53
|
+
of their own.
|
|
54
|
+
"""
|
|
55
|
+
return replace(
|
|
56
|
+
self,
|
|
57
|
+
title=_clip(self.title, MAX_TITLE),
|
|
58
|
+
description=_clip(self.description, MAX_DESCRIPTION),
|
|
59
|
+
)
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def _clip(text: str, limit: int) -> str:
|
|
63
|
+
text = " ".join(text.split())
|
|
64
|
+
if len(text) <= limit:
|
|
65
|
+
return text
|
|
66
|
+
cut = text[: limit - 1].rstrip()
|
|
67
|
+
space = cut.rfind(" ")
|
|
68
|
+
if space > limit // 2:
|
|
69
|
+
cut = cut[:space]
|
|
70
|
+
return cut.rstrip(" ,.;:-") + "…"
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
@dataclass(frozen=True)
|
|
74
|
+
class NotFound:
|
|
75
|
+
"""Returned by a resolver for a URL that looks valid but has no content.
|
|
76
|
+
|
|
77
|
+
Answering 200 with an "isn't available" screen is what search engines
|
|
78
|
+
file as a soft 404: the page is counted, judged empty, and the crawl
|
|
79
|
+
budget spent on it is gone. Say 404 instead.
|
|
80
|
+
"""
|
|
81
|
+
|
|
82
|
+
canonical_to: str = "/"
|
crawlableseo/py.typed
ADDED
|
File without changes
|
crawlableseo/robots.py
ADDED
|
@@ -0,0 +1,60 @@
|
|
|
1
|
+
"""robots.txt, with the two rules that cost the most to learn."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from collections.abc import Iterable, Sequence
|
|
6
|
+
|
|
7
|
+
# Crawlers that feed AI answers. Being quotable by them is a distribution
|
|
8
|
+
# channel, and the crawlable body this library injects is written for them.
|
|
9
|
+
DEFAULT_AI_CRAWLERS: tuple[str, ...] = (
|
|
10
|
+
"GPTBot",
|
|
11
|
+
"OAI-SearchBot",
|
|
12
|
+
"ChatGPT-User",
|
|
13
|
+
"PerplexityBot",
|
|
14
|
+
"ClaudeBot",
|
|
15
|
+
"Claude-SearchBot",
|
|
16
|
+
"Google-Extended",
|
|
17
|
+
)
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def robots_txt(
|
|
21
|
+
base_url: str,
|
|
22
|
+
*,
|
|
23
|
+
disallow: Sequence[str] = (),
|
|
24
|
+
allow: Sequence[str] = ("/",),
|
|
25
|
+
sitemap: str | Iterable[str] | None = "/sitemap.xml",
|
|
26
|
+
ai_crawlers: Sequence[str] | None = DEFAULT_AI_CRAWLERS,
|
|
27
|
+
extra_lines: Sequence[str] = (),
|
|
28
|
+
) -> str:
|
|
29
|
+
"""Build robots.txt.
|
|
30
|
+
|
|
31
|
+
Two warnings, both of them expensive in practice:
|
|
32
|
+
|
|
33
|
+
Do not disallow the JSON your own app fetches. A modern crawler renders
|
|
34
|
+
the page like a browser; blocking the API it calls leaves every route
|
|
35
|
+
empty at render time, and empty routes are filed as soft 404s. Disallow
|
|
36
|
+
the write and admin surfaces, not the read-only data.
|
|
37
|
+
|
|
38
|
+
There is no Content-Signal line and no option to add one. Lighthouse's
|
|
39
|
+
robots.txt validator does not know the directive, reports it as an
|
|
40
|
+
unknown directive and takes points off the SEO score of every page on
|
|
41
|
+
the site. Absence already means "no restriction".
|
|
42
|
+
"""
|
|
43
|
+
base_url = base_url.rstrip("/")
|
|
44
|
+
lines: list[str] = ["User-agent: *"]
|
|
45
|
+
lines += [f"Allow: {p}" for p in allow]
|
|
46
|
+
lines += [f"Disallow: {p}" for p in disallow]
|
|
47
|
+
|
|
48
|
+
for agent in ai_crawlers or ():
|
|
49
|
+
lines += ["", f"User-agent: {agent}", "Allow: /"]
|
|
50
|
+
|
|
51
|
+
lines += list(extra_lines)
|
|
52
|
+
|
|
53
|
+
if sitemap:
|
|
54
|
+
maps = [sitemap] if isinstance(sitemap, str) else list(sitemap)
|
|
55
|
+
lines.append("")
|
|
56
|
+
for m in maps:
|
|
57
|
+
lines.append(f"Sitemap: {m if m.startswith('http') else base_url + m}")
|
|
58
|
+
|
|
59
|
+
lines.append("")
|
|
60
|
+
return "\n".join(lines)
|
crawlableseo/shell.py
ADDED
|
@@ -0,0 +1,134 @@
|
|
|
1
|
+
"""Inject per-URL metadata and crawlable content into a built SPA shell."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import html
|
|
6
|
+
import re
|
|
7
|
+
|
|
8
|
+
from .head import head_block
|
|
9
|
+
from .page import Page
|
|
10
|
+
from .shell_marker import MARKER
|
|
11
|
+
|
|
12
|
+
__all__ = ["MARKER", "render_shell"]
|
|
13
|
+
|
|
14
|
+
_COMMENT_RE = re.compile(r"<!--.*?-->", re.DOTALL)
|
|
15
|
+
_TITLE_RE = re.compile(r"<title\b[^>]*>.*?</title>", re.IGNORECASE | re.DOTALL)
|
|
16
|
+
_DESC_RE = re.compile(
|
|
17
|
+
r"""<meta\s+name=["']description["'][^>]*>""", re.IGNORECASE
|
|
18
|
+
)
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def _comment_spans(template: str) -> list[tuple[int, int]]:
|
|
22
|
+
return [m.span() for m in _COMMENT_RE.finditer(template)]
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def _search_outside_comments(
|
|
26
|
+
pattern: re.Pattern[str], template: str, spans: list[tuple[int, int]]
|
|
27
|
+
) -> re.Match[str] | None:
|
|
28
|
+
"""First match that is not inside an HTML comment.
|
|
29
|
+
|
|
30
|
+
A build-time comment that merely mentions ``<title>`` is not the title.
|
|
31
|
+
Matching it and replacing it destroys the comment's closing marker, which
|
|
32
|
+
comments out the rest of the head: the stylesheet and the bundle script
|
|
33
|
+
become comment text, and the site serves a blank page with no console
|
|
34
|
+
error and no failed request.
|
|
35
|
+
"""
|
|
36
|
+
pos = 0
|
|
37
|
+
while pos <= len(template):
|
|
38
|
+
m = pattern.search(template, pos)
|
|
39
|
+
if m is None:
|
|
40
|
+
return None
|
|
41
|
+
inside = next((b for a, b in spans if a <= m.start() < b), None)
|
|
42
|
+
if inside is None:
|
|
43
|
+
return m
|
|
44
|
+
# Resume after the comment, not after the match: this match started
|
|
45
|
+
# inside the comment and swallowed the real tag that follows it.
|
|
46
|
+
pos = inside
|
|
47
|
+
return None
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def render_shell(
|
|
51
|
+
template: str,
|
|
52
|
+
page: Page,
|
|
53
|
+
*,
|
|
54
|
+
canonical: str,
|
|
55
|
+
base_url: str,
|
|
56
|
+
site_name: str = "",
|
|
57
|
+
twitter_site: str = "",
|
|
58
|
+
default_og_image: str | None = None,
|
|
59
|
+
mount_id: str = "root",
|
|
60
|
+
) -> str:
|
|
61
|
+
"""Return the shell with this page's tags and crawlable body in place."""
|
|
62
|
+
if MARKER in template:
|
|
63
|
+
return template
|
|
64
|
+
|
|
65
|
+
page = page.clipped()
|
|
66
|
+
e = html.escape
|
|
67
|
+
|
|
68
|
+
# Both tags are located in one pass over the original document, then the
|
|
69
|
+
# edits are applied back to front so the earlier offsets stay valid.
|
|
70
|
+
# Slicing, never re.sub with a replacement string: a title built from a
|
|
71
|
+
# query parameter can contain a backslash escape, which re would read as
|
|
72
|
+
# a group reference and raise on, turning an ordinary URL into a 500.
|
|
73
|
+
spans = _comment_spans(template)
|
|
74
|
+
title_tag = f"<title>{e(page.title)}</title>"
|
|
75
|
+
desc_tag = f'<meta name="description" content="{e(page.description)}" />'
|
|
76
|
+
edits: list[tuple[int, int, str]] = []
|
|
77
|
+
missing: list[str] = []
|
|
78
|
+
|
|
79
|
+
for pattern, tag in ((_TITLE_RE, title_tag), (_DESC_RE, desc_tag)):
|
|
80
|
+
m = _search_outside_comments(pattern, template, spans)
|
|
81
|
+
if m:
|
|
82
|
+
edits.append((m.start(), m.end(), tag))
|
|
83
|
+
else:
|
|
84
|
+
missing.append(tag)
|
|
85
|
+
|
|
86
|
+
out = template
|
|
87
|
+
for start, end, tag in sorted(edits, reverse=True):
|
|
88
|
+
out = out[:start] + tag + out[end:]
|
|
89
|
+
for tag in missing:
|
|
90
|
+
out = _insert_into_head(out, tag)
|
|
91
|
+
|
|
92
|
+
out = _insert_into_head(
|
|
93
|
+
out,
|
|
94
|
+
head_block(
|
|
95
|
+
page,
|
|
96
|
+
canonical=canonical,
|
|
97
|
+
base_url=base_url,
|
|
98
|
+
site_name=site_name,
|
|
99
|
+
twitter_site=twitter_site,
|
|
100
|
+
default_og_image=default_og_image,
|
|
101
|
+
),
|
|
102
|
+
)
|
|
103
|
+
|
|
104
|
+
if page.body:
|
|
105
|
+
out = _fill_mount_node(out, mount_id, page.body)
|
|
106
|
+
return out
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
def _insert_into_head(template: str, block: str) -> str:
|
|
110
|
+
idx = template.lower().find("</head>")
|
|
111
|
+
if idx == -1:
|
|
112
|
+
return template + "\n" + block
|
|
113
|
+
return template[:idx] + block + "\n" + template[idx:]
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
_MOUNT_RE_CACHE: dict[str, re.Pattern[str]] = {}
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
def _mount_re(mount_id: str) -> re.Pattern[str]:
|
|
120
|
+
cached = _MOUNT_RE_CACHE.get(mount_id)
|
|
121
|
+
if cached is None:
|
|
122
|
+
cached = re.compile(
|
|
123
|
+
r"(<div\b[^>]*\bid=[\"']?" + re.escape(mount_id) + r"[\"']?[^>]*>)\s*</div>",
|
|
124
|
+
re.IGNORECASE,
|
|
125
|
+
)
|
|
126
|
+
_MOUNT_RE_CACHE[mount_id] = cached
|
|
127
|
+
return cached
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
def _fill_mount_node(template: str, mount_id: str, body: str) -> str:
|
|
131
|
+
m = _mount_re(mount_id).search(template)
|
|
132
|
+
if not m:
|
|
133
|
+
return template
|
|
134
|
+
return template[: m.start()] + m.group(1) + body + "</div>" + template[m.end() :]
|
|
@@ -0,0 +1,7 @@
|
|
|
1
|
+
"""The injection marker, in its own module so head and shell can share it."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
# Marks a shell that has already been injected, so a second pass is a no-op
|
|
6
|
+
# rather than a document carrying two canonicals.
|
|
7
|
+
MARKER = "<!-- crawlableseo -->"
|
crawlableseo/site.py
ADDED
|
@@ -0,0 +1,270 @@
|
|
|
1
|
+
"""One declaration of a site; every output is derived from it."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from collections.abc import Callable, Iterable, Sequence
|
|
6
|
+
from dataclasses import dataclass
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
|
|
9
|
+
from .llmstxt import LlmsSection, llms_txt
|
|
10
|
+
from .page import NotFound, Page
|
|
11
|
+
from .robots import DEFAULT_AI_CRAWLERS, robots_txt
|
|
12
|
+
from .shell import render_shell
|
|
13
|
+
from .sitemap import SitemapEntry, sitemap_xml
|
|
14
|
+
from .urls import page_url
|
|
15
|
+
|
|
16
|
+
Resolver = Callable[[str, dict[str, str]], "Page | NotFound | None"]
|
|
17
|
+
UrlSupplier = Callable[[], Iterable["DynamicUrl"]]
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
@dataclass(frozen=True)
|
|
21
|
+
class DynamicUrl:
|
|
22
|
+
"""One URL of a dynamic route, for the sitemap.
|
|
23
|
+
|
|
24
|
+
``params`` must be the same parameters the resolver puts on the page it
|
|
25
|
+
returns, because both the canonical tag and this entry are built from
|
|
26
|
+
them by :func:`page_url`.
|
|
27
|
+
"""
|
|
28
|
+
|
|
29
|
+
path: str
|
|
30
|
+
params: dict[str, str] | None = None
|
|
31
|
+
lastmod: str | None = None
|
|
32
|
+
changefreq: str = "weekly"
|
|
33
|
+
priority: float = 0.5
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
@dataclass
|
|
37
|
+
class _Dynamic:
|
|
38
|
+
prefix: str
|
|
39
|
+
resolver: Resolver
|
|
40
|
+
urls: UrlSupplier | None
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
class Site:
|
|
44
|
+
"""The site's crawlable surface.
|
|
45
|
+
|
|
46
|
+
A static page is declared once with :meth:`page` and appears in the
|
|
47
|
+
shell's tags and in the sitemap. A dynamic route is declared with
|
|
48
|
+
:meth:`dynamic`: a resolver builds the page for a request, and an
|
|
49
|
+
optional supplier lists that route's URLs for the sitemap.
|
|
50
|
+
"""
|
|
51
|
+
|
|
52
|
+
def __init__(
|
|
53
|
+
self,
|
|
54
|
+
base_url: str,
|
|
55
|
+
*,
|
|
56
|
+
shell: str | Path | None = None,
|
|
57
|
+
shell_html: str | None = None,
|
|
58
|
+
name: str = "",
|
|
59
|
+
twitter_site: str = "",
|
|
60
|
+
default_og_image: str | None = None,
|
|
61
|
+
mount_id: str = "root",
|
|
62
|
+
default_title: str = "",
|
|
63
|
+
default_description: str = "",
|
|
64
|
+
) -> None:
|
|
65
|
+
self.base_url = base_url.rstrip("/")
|
|
66
|
+
self.name = name
|
|
67
|
+
self.twitter_site = twitter_site
|
|
68
|
+
self.default_og_image = default_og_image
|
|
69
|
+
self.mount_id = mount_id
|
|
70
|
+
self.default_title = default_title or name
|
|
71
|
+
self.default_description = default_description
|
|
72
|
+
if (shell is None) == (shell_html is None):
|
|
73
|
+
raise ValueError("pass exactly one of shell (a path) or shell_html (markup)")
|
|
74
|
+
self._shell_path = Path(shell) if shell is not None else None
|
|
75
|
+
self._shell_cache: str | None = shell_html
|
|
76
|
+
self._pages: dict[str, Page] = {}
|
|
77
|
+
self._dynamic: list[_Dynamic] = []
|
|
78
|
+
self._noindex_prefixes: list[str] = []
|
|
79
|
+
|
|
80
|
+
# -- declaration --------------------------------------------------
|
|
81
|
+
|
|
82
|
+
def page(
|
|
83
|
+
self,
|
|
84
|
+
path: str,
|
|
85
|
+
*,
|
|
86
|
+
title: str,
|
|
87
|
+
description: str,
|
|
88
|
+
body: str = "",
|
|
89
|
+
params: dict[str, str] | None = None,
|
|
90
|
+
jsonld: tuple[dict[str, object], ...] = (),
|
|
91
|
+
og_image: str | None = None,
|
|
92
|
+
index: bool = True,
|
|
93
|
+
follow: bool = True,
|
|
94
|
+
in_sitemap: bool = True,
|
|
95
|
+
changefreq: str = "weekly",
|
|
96
|
+
priority: float = 0.5,
|
|
97
|
+
lastmod: str | None = None,
|
|
98
|
+
) -> Page:
|
|
99
|
+
"""Declare a static page.
|
|
100
|
+
|
|
101
|
+
The arguments are :class:`Page`'s fields, spelled out rather than
|
|
102
|
+
forwarded as ``**kwargs``: this is the call people write most often,
|
|
103
|
+
and it should complete and type-check in an editor.
|
|
104
|
+
"""
|
|
105
|
+
page = Page(
|
|
106
|
+
title=title,
|
|
107
|
+
description=description,
|
|
108
|
+
body=body,
|
|
109
|
+
params=params or {},
|
|
110
|
+
jsonld=jsonld,
|
|
111
|
+
og_image=og_image,
|
|
112
|
+
index=index,
|
|
113
|
+
follow=follow,
|
|
114
|
+
in_sitemap=in_sitemap,
|
|
115
|
+
changefreq=changefreq,
|
|
116
|
+
priority=priority,
|
|
117
|
+
lastmod=lastmod,
|
|
118
|
+
)
|
|
119
|
+
self._pages[_normalise(path)] = page
|
|
120
|
+
return page
|
|
121
|
+
|
|
122
|
+
def add(self, path: str, page: Page) -> Page:
|
|
123
|
+
"""Declare a static page from a :class:`Page` you built yourself."""
|
|
124
|
+
self._pages[_normalise(path)] = page
|
|
125
|
+
return page
|
|
126
|
+
|
|
127
|
+
def dynamic(
|
|
128
|
+
self, prefix: str, *, urls: UrlSupplier | None = None
|
|
129
|
+
) -> Callable[[Resolver], Resolver]:
|
|
130
|
+
"""Register a resolver for every path under ``prefix``.
|
|
131
|
+
|
|
132
|
+
The resolver returns a :class:`Page`, or :class:`NotFound` for a URL
|
|
133
|
+
that has no content. Returning None falls through to the next
|
|
134
|
+
resolver, then to the site defaults.
|
|
135
|
+
"""
|
|
136
|
+
|
|
137
|
+
def decorate(resolver: Resolver) -> Resolver:
|
|
138
|
+
self._dynamic.append(_Dynamic(_normalise(prefix), resolver, urls))
|
|
139
|
+
return resolver
|
|
140
|
+
|
|
141
|
+
return decorate
|
|
142
|
+
|
|
143
|
+
def noindex(self, *prefixes: str) -> None:
|
|
144
|
+
"""Mark private or utility route prefixes as never indexable."""
|
|
145
|
+
self._noindex_prefixes.extend(_normalise(p) for p in prefixes)
|
|
146
|
+
|
|
147
|
+
# -- resolution ---------------------------------------------------
|
|
148
|
+
|
|
149
|
+
def resolve(self, path: str, query: dict[str, str] | None = None) -> Page:
|
|
150
|
+
path = _normalise(path)
|
|
151
|
+
query = query or {}
|
|
152
|
+
|
|
153
|
+
for prefix in self._noindex_prefixes:
|
|
154
|
+
if path == prefix or path.startswith(prefix.rstrip("/") + "/"):
|
|
155
|
+
return self._hidden()
|
|
156
|
+
|
|
157
|
+
declared = self._pages.get(path)
|
|
158
|
+
if declared is not None:
|
|
159
|
+
return declared
|
|
160
|
+
|
|
161
|
+
for entry in self._dynamic:
|
|
162
|
+
if path == entry.prefix or path.startswith(entry.prefix.rstrip("/") + "/"):
|
|
163
|
+
result = entry.resolver(path, query)
|
|
164
|
+
if isinstance(result, NotFound):
|
|
165
|
+
return self._hidden(status=404)
|
|
166
|
+
if result is not None:
|
|
167
|
+
return result
|
|
168
|
+
|
|
169
|
+
# An unknown path is not a second copy of the home page. Answering
|
|
170
|
+
# 200 there turns every typo and every stale link into an indexable
|
|
171
|
+
# duplicate of the site's most important page.
|
|
172
|
+
return self._hidden(status=404)
|
|
173
|
+
|
|
174
|
+
def _hidden(self, status: int = 200) -> Page:
|
|
175
|
+
"""The site's fallback card for a page no crawler should index."""
|
|
176
|
+
return Page(
|
|
177
|
+
title=self.default_title,
|
|
178
|
+
description=self.default_description,
|
|
179
|
+
index=False,
|
|
180
|
+
follow=False,
|
|
181
|
+
in_sitemap=False,
|
|
182
|
+
status=status,
|
|
183
|
+
)
|
|
184
|
+
|
|
185
|
+
def canonical(self, path: str, page: Page) -> str:
|
|
186
|
+
return page_url(self.base_url, path, page.params)
|
|
187
|
+
|
|
188
|
+
# -- outputs ------------------------------------------------------
|
|
189
|
+
|
|
190
|
+
def shell(self) -> str:
|
|
191
|
+
"""The built shell, read once and kept in memory.
|
|
192
|
+
|
|
193
|
+
Call :meth:`reload_shell` after a new frontend build, or restart the
|
|
194
|
+
process: a long-lived server otherwise serves the previous bundle's
|
|
195
|
+
script tags for as long as it runs.
|
|
196
|
+
"""
|
|
197
|
+
if self._shell_cache is None:
|
|
198
|
+
assert self._shell_path is not None
|
|
199
|
+
self._shell_cache = self._shell_path.read_text(encoding="utf-8")
|
|
200
|
+
return self._shell_cache
|
|
201
|
+
|
|
202
|
+
def reload_shell(self) -> None:
|
|
203
|
+
if self._shell_path is not None:
|
|
204
|
+
self._shell_cache = None
|
|
205
|
+
|
|
206
|
+
def render(self, path: str, query: dict[str, str] | None = None) -> tuple[str, int]:
|
|
207
|
+
"""Return ``(html, status)`` for a request path."""
|
|
208
|
+
page = self.resolve(path, query)
|
|
209
|
+
html = render_shell(
|
|
210
|
+
self.shell(),
|
|
211
|
+
page,
|
|
212
|
+
canonical=self.canonical(_normalise(path), page),
|
|
213
|
+
base_url=self.base_url,
|
|
214
|
+
site_name=self.name,
|
|
215
|
+
twitter_site=self.twitter_site,
|
|
216
|
+
default_og_image=self.default_og_image,
|
|
217
|
+
mount_id=self.mount_id,
|
|
218
|
+
)
|
|
219
|
+
return html, page.status
|
|
220
|
+
|
|
221
|
+
def sitemap_entries(self) -> list[SitemapEntry]:
|
|
222
|
+
entries = [
|
|
223
|
+
SitemapEntry(
|
|
224
|
+
loc=page_url(self.base_url, path, page.params),
|
|
225
|
+
lastmod=page.lastmod,
|
|
226
|
+
changefreq=page.changefreq,
|
|
227
|
+
priority=page.priority,
|
|
228
|
+
)
|
|
229
|
+
for path, page in sorted(self._pages.items())
|
|
230
|
+
if page.in_sitemap and page.index
|
|
231
|
+
]
|
|
232
|
+
for entry in self._dynamic:
|
|
233
|
+
for url in entry.urls() if entry.urls else ():
|
|
234
|
+
entries.append(
|
|
235
|
+
SitemapEntry(
|
|
236
|
+
loc=page_url(self.base_url, url.path, url.params),
|
|
237
|
+
lastmod=url.lastmod,
|
|
238
|
+
changefreq=url.changefreq,
|
|
239
|
+
priority=url.priority,
|
|
240
|
+
)
|
|
241
|
+
)
|
|
242
|
+
return entries
|
|
243
|
+
|
|
244
|
+
def sitemap(self) -> str:
|
|
245
|
+
return sitemap_xml(self.sitemap_entries())
|
|
246
|
+
|
|
247
|
+
def robots(
|
|
248
|
+
self,
|
|
249
|
+
*,
|
|
250
|
+
disallow: Sequence[str] = (),
|
|
251
|
+
ai_crawlers: Sequence[str] | None = DEFAULT_AI_CRAWLERS,
|
|
252
|
+
extra_lines: Sequence[str] = (),
|
|
253
|
+
) -> str:
|
|
254
|
+
return robots_txt(
|
|
255
|
+
self.base_url,
|
|
256
|
+
disallow=list(disallow) + self._noindex_prefixes,
|
|
257
|
+
ai_crawlers=ai_crawlers,
|
|
258
|
+
extra_lines=extra_lines,
|
|
259
|
+
)
|
|
260
|
+
|
|
261
|
+
def llms(
|
|
262
|
+
self, summary: str, sections: Iterable[LlmsSection], *, notes: Sequence[str] = ()
|
|
263
|
+
) -> str:
|
|
264
|
+
return llms_txt(self.name, summary, sections, notes=notes)
|
|
265
|
+
|
|
266
|
+
|
|
267
|
+
def _normalise(path: str) -> str:
|
|
268
|
+
if not path.startswith("/"):
|
|
269
|
+
path = "/" + path
|
|
270
|
+
return path.rstrip("/") or "/"
|
crawlableseo/sitemap.py
ADDED
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
"""sitemap.xml, built from the same URL function as the canonical tag."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from collections.abc import Iterable
|
|
6
|
+
from dataclasses import dataclass
|
|
7
|
+
|
|
8
|
+
from .urls import xml_escape
|
|
9
|
+
|
|
10
|
+
# The protocol's own ceiling. Past it a sitemap must be split and listed in
|
|
11
|
+
# an index; this library raises rather than shipping a file crawlers reject.
|
|
12
|
+
MAX_URLS = 50_000
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
@dataclass(frozen=True)
|
|
16
|
+
class SitemapEntry:
|
|
17
|
+
loc: str
|
|
18
|
+
lastmod: str | None = None
|
|
19
|
+
changefreq: str = "weekly"
|
|
20
|
+
priority: float = 0.5
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def sitemap_xml(entries: Iterable[SitemapEntry]) -> str:
|
|
24
|
+
entries = list(entries)
|
|
25
|
+
if len(entries) > MAX_URLS:
|
|
26
|
+
raise ValueError(
|
|
27
|
+
f"a sitemap holds at most {MAX_URLS} URLs, got {len(entries)}; "
|
|
28
|
+
"split it and publish a sitemap index"
|
|
29
|
+
)
|
|
30
|
+
body = "\n".join(_url_tag(e) for e in entries)
|
|
31
|
+
return (
|
|
32
|
+
'<?xml version="1.0" encoding="UTF-8"?>\n'
|
|
33
|
+
'<urlset xmlns="http://www.sitemaps.org/schemas/sitemap/0.9">\n'
|
|
34
|
+
f"{body}\n"
|
|
35
|
+
"</urlset>\n"
|
|
36
|
+
)
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def _url_tag(entry: SitemapEntry) -> str:
|
|
40
|
+
parts = [f" <url><loc>{xml_escape(entry.loc)}</loc>"]
|
|
41
|
+
if entry.lastmod:
|
|
42
|
+
parts.append(f"<lastmod>{xml_escape(entry.lastmod)}</lastmod>")
|
|
43
|
+
parts.append(f"<changefreq>{entry.changefreq}</changefreq>")
|
|
44
|
+
parts.append(f"<priority>{entry.priority:.1f}</priority></url>")
|
|
45
|
+
return "".join(parts)
|
crawlableseo/urls.py
ADDED
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
"""The single place a page's URL is built.
|
|
2
|
+
|
|
3
|
+
Every URL this library emits - the canonical tag, the sitemap entry, the
|
|
4
|
+
IndexNow submission, the llms.txt link - comes from :func:`page_url`. When a
|
|
5
|
+
canonical and its sitemap entry are built by two different pieces of code
|
|
6
|
+
they drift, usually over one query parameter, and a search engine treats the
|
|
7
|
+
two spellings as two pages with identical content.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
from urllib.parse import quote
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def page_url(base_url: str, path: str, params: dict[str, str] | None = None) -> str:
|
|
16
|
+
base_url = base_url.rstrip("/")
|
|
17
|
+
if not path.startswith("/"):
|
|
18
|
+
path = "/" + path
|
|
19
|
+
if path != "/":
|
|
20
|
+
path = path.rstrip("/") or "/"
|
|
21
|
+
url = base_url + quote(path, safe="/-._~")
|
|
22
|
+
if params:
|
|
23
|
+
query = "&".join(
|
|
24
|
+
f"{quote(k, safe='')}={quote(v, safe='')}" for k, v in sorted(params.items())
|
|
25
|
+
)
|
|
26
|
+
url = f"{url}?{query}"
|
|
27
|
+
return url
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def xml_escape(value: str) -> str:
|
|
31
|
+
return (
|
|
32
|
+
value.replace("&", "&")
|
|
33
|
+
.replace("<", "<")
|
|
34
|
+
.replace(">", ">")
|
|
35
|
+
.replace('"', """)
|
|
36
|
+
)
|
|
@@ -0,0 +1,208 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: crawlableseo
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Make a client-rendered SPA crawlable from its own HTML shell: meta tags, JSON-LD, robots.txt, sitemap.xml and llms.txt from one declaration.
|
|
5
|
+
Project-URL: Homepage, https://github.com/kulykivska/crawlableseo
|
|
6
|
+
Project-URL: Issues, https://github.com/kulykivska/crawlableseo/issues
|
|
7
|
+
Project-URL: Changelog, https://github.com/kulykivska/crawlableseo/blob/main/CHANGELOG.md
|
|
8
|
+
Author: Yuliia Kulykivska
|
|
9
|
+
License: MIT License
|
|
10
|
+
|
|
11
|
+
Copyright (c) 2026 Yuliia Kulykivska
|
|
12
|
+
|
|
13
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
14
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
15
|
+
in the Software without restriction, including without limitation the rights
|
|
16
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
17
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
18
|
+
furnished to do so, subject to the following conditions:
|
|
19
|
+
|
|
20
|
+
The above copyright notice and this permission notice shall be included in all
|
|
21
|
+
copies or substantial portions of the Software.
|
|
22
|
+
|
|
23
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
24
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
25
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
26
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
27
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
28
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
29
|
+
SOFTWARE.
|
|
30
|
+
License-File: LICENSE
|
|
31
|
+
Keywords: crawler,fastapi,indexnow,json-ld,llms-txt,robots,seo,sitemap,spa,starlette
|
|
32
|
+
Classifier: Development Status :: 4 - Beta
|
|
33
|
+
Classifier: Intended Audience :: Developers
|
|
34
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
35
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
36
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
37
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
38
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
39
|
+
Classifier: Topic :: Internet :: WWW/HTTP
|
|
40
|
+
Classifier: Topic :: Internet :: WWW/HTTP :: Indexing/Search
|
|
41
|
+
Classifier: Typing :: Typed
|
|
42
|
+
Requires-Python: >=3.10
|
|
43
|
+
Provides-Extra: dev
|
|
44
|
+
Requires-Dist: fastapi>=0.100; extra == 'dev'
|
|
45
|
+
Requires-Dist: httpx>=0.24; extra == 'dev'
|
|
46
|
+
Requires-Dist: mypy>=1.10; extra == 'dev'
|
|
47
|
+
Requires-Dist: pytest-asyncio>=0.23; extra == 'dev'
|
|
48
|
+
Requires-Dist: pytest>=8; extra == 'dev'
|
|
49
|
+
Requires-Dist: ruff>=0.6; extra == 'dev'
|
|
50
|
+
Provides-Extra: fastapi
|
|
51
|
+
Requires-Dist: fastapi>=0.100; extra == 'fastapi'
|
|
52
|
+
Provides-Extra: indexnow
|
|
53
|
+
Requires-Dist: httpx>=0.24; extra == 'indexnow'
|
|
54
|
+
Description-Content-Type: text/markdown
|
|
55
|
+
|
|
56
|
+
# crawlableseo
|
|
57
|
+
|
|
58
|
+
Make a client-rendered single-page app crawlable from its own HTML shell.
|
|
59
|
+
|
|
60
|
+
No headless browser, no SSR framework, no third-party prerendering service. Your
|
|
61
|
+
Python server fills in the shell's `<head>` and mount node for each URL, and the
|
|
62
|
+
same declaration produces `robots.txt`, `sitemap.xml` and `llms.txt`.
|
|
63
|
+
|
|
64
|
+
```bash
|
|
65
|
+
pip install crawlableseo
|
|
66
|
+
```
|
|
67
|
+
|
|
68
|
+
## The problem
|
|
69
|
+
|
|
70
|
+
A Vite or CRA build ships one `index.html` with an empty `<div id="root">`. Every
|
|
71
|
+
URL on the site returns the same document: the same title, the same description,
|
|
72
|
+
no content. Google renders JavaScript and will often cope; Bing is slower to; and
|
|
73
|
+
the crawlers behind AI answers — GPTBot, PerplexityBot, ClaudeBot — mostly read
|
|
74
|
+
the HTML they are given. What they are given is an empty div.
|
|
75
|
+
|
|
76
|
+
The usual answers are to adopt a meta-framework, run a headless browser per
|
|
77
|
+
request, or pay a prerendering service. This library takes the fourth option:
|
|
78
|
+
serve the same static shell, but write the real title, description, canonical,
|
|
79
|
+
Open Graph, JSON-LD and a block of readable HTML into it before it goes out.
|
|
80
|
+
|
|
81
|
+
## Quickstart
|
|
82
|
+
|
|
83
|
+
```python
|
|
84
|
+
from fastapi import FastAPI
|
|
85
|
+
from crawlableseo import DynamicUrl, NotFound, Page, Site
|
|
86
|
+
from crawlableseo.integrations.fastapi import router
|
|
87
|
+
|
|
88
|
+
site = Site(
|
|
89
|
+
"https://example.com",
|
|
90
|
+
shell="frontend/dist/index.html",
|
|
91
|
+
name="Example",
|
|
92
|
+
default_title="Example",
|
|
93
|
+
default_description="What this site is, in one sentence.",
|
|
94
|
+
)
|
|
95
|
+
|
|
96
|
+
site.page(
|
|
97
|
+
"/pricing",
|
|
98
|
+
title="Pricing | Example",
|
|
99
|
+
description="Three plans, what each one includes, and what they cost.",
|
|
100
|
+
body="<h1>Pricing</h1><p>Free, Pro and Team...</p>",
|
|
101
|
+
priority=0.8,
|
|
102
|
+
)
|
|
103
|
+
|
|
104
|
+
site.noindex("/admin", "/login", "/settings")
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
@site.dynamic("/product", urls=lambda: [DynamicUrl("/product", {"id": p.id}) for p in catalogue()])
|
|
108
|
+
def product(path: str, query: dict[str, str]) -> Page | NotFound | None:
|
|
109
|
+
item = lookup(query.get("id"))
|
|
110
|
+
if item is None:
|
|
111
|
+
return NotFound()
|
|
112
|
+
return Page(
|
|
113
|
+
title=f"{item.name} | Example",
|
|
114
|
+
description=item.summary,
|
|
115
|
+
params={"id": item.id},
|
|
116
|
+
body=f"<h1>{item.name}</h1><p>{item.summary}</p>",
|
|
117
|
+
jsonld=({"@context": "https://schema.org", "@type": "Product", "name": item.name},),
|
|
118
|
+
)
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
app = FastAPI()
|
|
122
|
+
# ... your API routes ...
|
|
123
|
+
app.include_router(router(site, disallow=["/api/admin"])) # mount last
|
|
124
|
+
```
|
|
125
|
+
|
|
126
|
+
That serves `/robots.txt`, `/sitemap.xml` and the SPA catch-all. Add `llms=` for
|
|
127
|
+
`/llms.txt` and `indexnow=` to publish an IndexNow key file.
|
|
128
|
+
|
|
129
|
+
Not using FastAPI? Everything underneath is a plain function:
|
|
130
|
+
|
|
131
|
+
```python
|
|
132
|
+
from crawlableseo import Page, render_shell, robots_txt, sitemap_xml
|
|
133
|
+
```
|
|
134
|
+
|
|
135
|
+
## What it does
|
|
136
|
+
|
|
137
|
+
- **Per-URL `<head>`** — title, description, canonical, robots, Open Graph,
|
|
138
|
+
Twitter cards, and any number of JSON-LD objects.
|
|
139
|
+
- **Crawlable content** — HTML written into the mount node. Your framework
|
|
140
|
+
replaces the node's children when it mounts, so users never see it; a crawler
|
|
141
|
+
that does not run JavaScript reads it as the page.
|
|
142
|
+
- **`robots.txt`, `sitemap.xml`, `llms.txt`** — from the same declaration, so a
|
|
143
|
+
page cannot be in one and missing from another.
|
|
144
|
+
- **IndexNow** — submit changed URLs to Bing (and through it, ChatGPT search)
|
|
145
|
+
instead of waiting for the next crawl.
|
|
146
|
+
- **Honest status codes** — a URL with no content answers 404, not a 200 with an
|
|
147
|
+
empty screen.
|
|
148
|
+
|
|
149
|
+
## What it does not do
|
|
150
|
+
|
|
151
|
+
- It never writes your copy. You supply the text; the library places it.
|
|
152
|
+
- It does not render your JavaScript. If a page's content exists only after a
|
|
153
|
+
client-side fetch, give the resolver access to the same data on the server.
|
|
154
|
+
- It is not a meta-framework and will not become one.
|
|
155
|
+
|
|
156
|
+
## Bugs this prevents
|
|
157
|
+
|
|
158
|
+
Each of these was shipped to production on a live site before it was understood.
|
|
159
|
+
They are the reason the library exists, and every one has a test.
|
|
160
|
+
|
|
161
|
+
**A `<title>` mentioned in a build comment is not the title.** A shell carried a
|
|
162
|
+
comment explaining that the server rewrites `<title>`. A naive search-and-replace
|
|
163
|
+
matched that mention; the replacement ate the comment's closing `-->`, which
|
|
164
|
+
commented out the rest of `<head>` — stylesheet and bundle script included. The
|
|
165
|
+
site served a blank page with no console error and no failed request.
|
|
166
|
+
|
|
167
|
+
**Canonical and sitemap drift.** When the canonical tag and the sitemap entry are
|
|
168
|
+
built by two pieces of code, they disagree over a query parameter sooner or later,
|
|
169
|
+
and the two spellings become two pages with identical content. Here both come from
|
|
170
|
+
one function, and there is a test that fails if they ever differ.
|
|
171
|
+
|
|
172
|
+
**Blocking your own API starves the renderer.** `Disallow: /api/` in `robots.txt`
|
|
173
|
+
looks tidy. A crawler renders the page like a browser, so blocking the JSON the
|
|
174
|
+
app fetches leaves every route empty at render time. On one site that produced 22
|
|
175
|
+
soft 404s and 225 URLs stuck at "Discovered – currently not indexed". Disallow the
|
|
176
|
+
write and admin surfaces; leave the read-only data alone.
|
|
177
|
+
|
|
178
|
+
**`Content-Signal` costs more than it gives.** Lighthouse's `robots.txt` validator
|
|
179
|
+
does not know the directive, reports it as unknown, and takes points off the SEO
|
|
180
|
+
score of every page. Its absence already means no restriction, so this library has
|
|
181
|
+
no option to emit it.
|
|
182
|
+
|
|
183
|
+
**A `</` in your data closes the script tag.** One product name with a slash in it
|
|
184
|
+
and the JSON-LD block ends early, taking the rest of the document with it. Escaped
|
|
185
|
+
here, once.
|
|
186
|
+
|
|
187
|
+
**A backslash in a title raises a 500.** Titles built from query parameters can
|
|
188
|
+
contain anything; `re.sub` reads `\1` in a replacement string as a group
|
|
189
|
+
reference. An ordinary crafted URL becomes a server error.
|
|
190
|
+
|
|
191
|
+
**A 200 on an unknown path is a duplicate home page.** Every typo and stale link
|
|
192
|
+
becomes an indexable copy of your most important page. Unknown paths answer 404.
|
|
193
|
+
|
|
194
|
+
**`noindex` is not `nofollow`.** A page can be worth keeping out of the index while
|
|
195
|
+
its outgoing links are still worth crawling. They are separate flags.
|
|
196
|
+
|
|
197
|
+
**Do not detach the crawlable block early.** If you remove it before your framework
|
|
198
|
+
mounts, the page paints, empties, then repaints: on one site that measured a
|
|
199
|
+
cumulative layout shift of 0.28. Let the framework replace it.
|
|
200
|
+
|
|
201
|
+
## Compatibility
|
|
202
|
+
|
|
203
|
+
Python 3.10+. No required dependencies. `crawlableseo[indexnow]` adds `httpx`;
|
|
204
|
+
`crawlableseo[fastapi]` adds FastAPI for the router.
|
|
205
|
+
|
|
206
|
+
## License
|
|
207
|
+
|
|
208
|
+
MIT
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
crawlableseo/__init__.py,sha256=FFUM8__7X8zs3swBAUOQ5rKNwalZCHkD_2NdZPYamYA,1082
|
|
2
|
+
crawlableseo/head.py,sha256=2BOodVIwYq3BBMwV3V6AhFW4OxSkeQl6t2rRozaSM4w,2169
|
|
3
|
+
crawlableseo/indexnow.py,sha256=R0gRG-EOysDymV1MLwcx6r-VZM7tzdADEFTIk13RRLk,4008
|
|
4
|
+
crawlableseo/llmstxt.py,sha256=loamGNHLVlR68fx9MKTsyPgwPJxd51MvWpG7KGQbCso,1134
|
|
5
|
+
crawlableseo/page.py,sha256=g8B91Dbzo9B_PNdcWbwl_O6cM6BQzHjZiQwIgzsxxIY,2795
|
|
6
|
+
crawlableseo/py.typed,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
7
|
+
crawlableseo/robots.py,sha256=pcmQhTENGrG6dLcehWJo54hTEP1TZlDyfkGP1e4-irw,1980
|
|
8
|
+
crawlableseo/shell.py,sha256=iWU5iOIV4vu6J-5yXZ90e7N8j1faBRTKFG8HRNVl3Ec,4250
|
|
9
|
+
crawlableseo/shell_marker.py,sha256=Jg7NdSRGHmMVfDjN-YK0oiEjGuIXfciog15VlVNJvBY,274
|
|
10
|
+
crawlableseo/site.py,sha256=Jzzwv0SDEGqr7121DwwcrY1xLPsbo86zP4c4a5MpBR8,9164
|
|
11
|
+
crawlableseo/sitemap.py,sha256=bA4GS_xdOl5hi2slsYf1i9wcnj69hVxT9-7fO3dERNI,1407
|
|
12
|
+
crawlableseo/urls.py,sha256=vj4aqQiafxtImw_UHh994QYo3m9bs-cJtJIimLLFpV8,1127
|
|
13
|
+
crawlableseo/integrations/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
14
|
+
crawlableseo/integrations/fastapi.py,sha256=i9M3ablddBMivtzcxDGHpemgAQkpficWD4n2YACo-ZA,3047
|
|
15
|
+
crawlableseo-0.1.0.dist-info/METADATA,sha256=fpxF4_zc1A3Kn6FNtwHy_LDRd50I0e3wH513G3Ib4Xk,9082
|
|
16
|
+
crawlableseo-0.1.0.dist-info/WHEEL,sha256=THafob7ofN-NsuMN7Mg4qZyHaQI7KkD-QlcQatYhXPo,87
|
|
17
|
+
crawlableseo-0.1.0.dist-info/licenses/LICENSE,sha256=Q8_f8lqxmXMni_cH9slkOwVhwcaztOi8BIwK6nSEC9k,1074
|
|
18
|
+
crawlableseo-0.1.0.dist-info/RECORD,,
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Yuliia Kulykivska
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|