aegis-stack-crawl4ai 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- aegis_stack_crawl4ai/__init__.py +1 -0
- aegis_stack_crawl4ai/plugin.py +144 -0
- aegis_stack_crawl4ai/templates/{{ project_slug }}/app/cli/crawl.py.jinja +460 -0
- aegis_stack_crawl4ai/templates/{{ project_slug }}/app/components/backend/api/crawler/__init__.py.jinja +8 -0
- aegis_stack_crawl4ai/templates/{{ project_slug }}/app/components/backend/api/crawler/router.py.jinja +122 -0
- aegis_stack_crawl4ai/templates/{{ project_slug }}/app/components/frontend/dashboard/cards/crawler_card.py.jinja +84 -0
- aegis_stack_crawl4ai/templates/{{ project_slug }}/app/components/frontend/dashboard/crawler_ui.py.jinja +42 -0
- aegis_stack_crawl4ai/templates/{{ project_slug }}/app/components/frontend/dashboard/modals/crawler_crawl_tab.py.jinja +337 -0
- aegis_stack_crawl4ai/templates/{{ project_slug }}/app/components/frontend/dashboard/modals/crawler_modal.py.jinja +199 -0
- aegis_stack_crawl4ai/templates/{{ project_slug }}/app/components/frontend/dashboard/modals/crawler_pages_tab.py.jinja +255 -0
- aegis_stack_crawl4ai/templates/{{ project_slug }}/app/services/crawler/__init__.py.jinja +15 -0
- aegis_stack_crawl4ai/templates/{{ project_slug }}/app/services/crawler/deps.py.jinja +34 -0
- aegis_stack_crawl4ai/templates/{{ project_slug }}/app/services/crawler/dispatch.py.jinja +46 -0
- aegis_stack_crawl4ai/templates/{{ project_slug }}/app/services/crawler/health.py.jinja +61 -0
- aegis_stack_crawl4ai/templates/{{ project_slug }}/app/services/crawler/jobs.py.jinja +79 -0
- aegis_stack_crawl4ai/templates/{{ project_slug }}/app/services/crawler/models.py.jinja +48 -0
- aegis_stack_crawl4ai/templates/{{ project_slug }}/app/services/crawler/options.py.jinja +84 -0
- aegis_stack_crawl4ai/templates/{{ project_slug }}/app/services/crawler/queries.py.jinja +112 -0
- aegis_stack_crawl4ai/templates/{{ project_slug }}/app/services/crawler/schemas.py.jinja +137 -0
- aegis_stack_crawl4ai/templates/{{ project_slug }}/app/services/crawler/service.py.jinja +440 -0
- aegis_stack_crawl4ai/templates/{{ project_slug }}/app/services/crawler/settings.py.jinja +35 -0
- aegis_stack_crawl4ai/templates/{{ project_slug }}/app/services/crawler/urls.py.jinja +27 -0
- aegis_stack_crawl4ai/templates/{{ project_slug }}/tests/components/frontend/test_crawler_tab.py.jinja +230 -0
- aegis_stack_crawl4ai/templates/{{ project_slug }}/tests/services/test_crawler_jobs.py.jinja +234 -0
- aegis_stack_crawl4ai/templates/{{ project_slug }}/tests/services/test_crawler_layout.py.jinja +49 -0
- aegis_stack_crawl4ai/templates/{{ project_slug }}/tests/services/test_crawler_options.py.jinja +197 -0
- aegis_stack_crawl4ai/templates/{{ project_slug }}/tests/services/test_crawler_sites.py.jinja +110 -0
- aegis_stack_crawl4ai/templates/{{ project_slug }}/tests/services/test_crawler_wiring.py.jinja +48 -0
- aegis_stack_crawl4ai-0.1.0.dist-info/METADATA +317 -0
- aegis_stack_crawl4ai-0.1.0.dist-info/RECORD +33 -0
- aegis_stack_crawl4ai-0.1.0.dist-info/WHEEL +4 -0
- aegis_stack_crawl4ai-0.1.0.dist-info/entry_points.txt +2 -0
- aegis_stack_crawl4ai-0.1.0.dist-info/licenses/LICENSE +202 -0
|
@@ -0,0 +1 @@
|
|
|
1
|
+
"""Aegis Stack plugin: crawl4ai"""
|
|
@@ -0,0 +1,144 @@
|
|
|
1
|
+
"""PluginSpec for ``aegis-stack-crawl4ai``.
|
|
2
|
+
|
|
3
|
+
Declares a service-flavoured plugin that adds web-crawling-to-markdown
|
|
4
|
+
capability to an Aegis Stack project. Wraps the upstream ``crawl4ai``
|
|
5
|
+
library (Apache 2.0). Aegis discovers this spec via the
|
|
6
|
+
``aegis.plugins`` entry point in ``pyproject.toml`` (see
|
|
7
|
+
``aegis.core.plugins.discovery``).
|
|
8
|
+
|
|
9
|
+
Design notes:
|
|
10
|
+
|
|
11
|
+
- ``MigrationSpec(schema="crawler")`` — the plugin's tables live in
|
|
12
|
+
their own Postgres schema, isolated from the project's other services.
|
|
13
|
+
SQLite has no schemas; there the table lands unqualified.
|
|
14
|
+
- ``required_components=['database']`` — the resolver auto-installs
|
|
15
|
+
the database component if the target project doesn't have it.
|
|
16
|
+
- ``crawled_page`` table is generic by design (URL + content + metadata
|
|
17
|
+
JSONB) so future ingestion plugins (RSS, PDF, sitemap) can either
|
|
18
|
+
share the schema or follow the same shape.
|
|
19
|
+
"""
|
|
20
|
+
|
|
21
|
+
from aegis.core.file_manifest import FileManifest
|
|
22
|
+
from aegis.core.migration_spec import MigrationSpec
|
|
23
|
+
from aegis.core.plugins.spec import (
|
|
24
|
+
FrontendWidgetWiring,
|
|
25
|
+
HealthCheckWiring,
|
|
26
|
+
PluginKind,
|
|
27
|
+
PluginSpec,
|
|
28
|
+
PluginWiring,
|
|
29
|
+
RouterWiring,
|
|
30
|
+
SymbolWiring,
|
|
31
|
+
)
|
|
32
|
+
|
|
33
|
+
# The name the health check reports under. The dashboard groups services
|
|
34
|
+
# by it, the card factory dispatches on it, and the modal registers under
|
|
35
|
+
# ``service_<label>``; the rendered code mirrors it in
|
|
36
|
+
# ``app/components/frontend/dashboard/crawler_ui.py``, which derives it
|
|
37
|
+
# from ``SERVICE_NAME`` in the service package. One spelling, because a
|
|
38
|
+
# mismatch is a dead click rather than an error.
|
|
39
|
+
HEALTH_LABEL = "Crawler"
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def get_spec() -> PluginSpec:
|
|
43
|
+
"""Return the plugin spec."""
|
|
44
|
+
return PluginSpec(
|
|
45
|
+
name="crawl4ai",
|
|
46
|
+
kind=PluginKind.SERVICE,
|
|
47
|
+
description="Web crawling and scraping via Crawl4AI",
|
|
48
|
+
version="0.1.0",
|
|
49
|
+
verified=False,
|
|
50
|
+
# PEP 440 specifier — the spec declares no tables (a revision is
|
|
51
|
+
# derived from the model), which the CLI only understands from
|
|
52
|
+
# 0.12 on. 0.13.1 is where the generated card-render test stops
|
|
53
|
+
# failing on a plugin's card, so anything older installs and then
|
|
54
|
+
# hands the user a red ``make check``.
|
|
55
|
+
aegis_version=">=0.13.1",
|
|
56
|
+
# CLI verb the plugin exposes in the generated project. Decoupled
|
|
57
|
+
# from the install identifier (``crawl4ai``) so users type the
|
|
58
|
+
# natural ``<project> crawl ...`` instead of the package name.
|
|
59
|
+
cli_name="crawl",
|
|
60
|
+
# Resolver auto-installs database if missing.
|
|
61
|
+
required_components=["database"],
|
|
62
|
+
# Pinned to >=0.8.6 because 0.8.6 is the supply-chain hotfix
|
|
63
|
+
# that swapped the compromised ``litellm`` package out — earlier
|
|
64
|
+
# 0.8.x builds pull the bad transitive dep.
|
|
65
|
+
# ``alembic`` is required because this plugin ships migrations;
|
|
66
|
+
# database-only projects (no auth / no insights) don't pull it
|
|
67
|
+
# transitively, so we declare it here.
|
|
68
|
+
pyproject_deps=["crawl4ai>=0.8.6", "alembic>=1.13"],
|
|
69
|
+
# Generic ``pages`` table — URL + content + JSONB metadata.
|
|
70
|
+
# Indexed on source_url, content_hash, and fetched_at for the
|
|
71
|
+
# three common query shapes (last fetch, dedup, time-window).
|
|
72
|
+
# ``doc_metadata`` rather than ``metadata`` because the latter
|
|
73
|
+
# is reserved on SQLAlchemy's ``DeclarativeBase``.
|
|
74
|
+
migrations=[
|
|
75
|
+
MigrationSpec(
|
|
76
|
+
service_name="crawler",
|
|
77
|
+
description="Crawled page store",
|
|
78
|
+
# Proof it ran: the startup hook stamps instead of
|
|
79
|
+
# replaying DDL on a database that already has this.
|
|
80
|
+
stamp_signature=("table", "crawler.crawled_page"),
|
|
81
|
+
# Crawler tables live in their own ``crawler`` Postgres
|
|
82
|
+
# schema. SQLite has no schemas: the generator drops the
|
|
83
|
+
# qualifier there and the model gates ``__table_args__``
|
|
84
|
+
# on the engine, so one declaration serves both.
|
|
85
|
+
schema="crawler",
|
|
86
|
+
),
|
|
87
|
+
],
|
|
88
|
+
# All files this plugin owns inside the target project.
|
|
89
|
+
# ``aegis remove`` walks this list to clean up.
|
|
90
|
+
files=FileManifest(
|
|
91
|
+
primary=[
|
|
92
|
+
"app/services/crawler",
|
|
93
|
+
"app/components/backend/api/crawler",
|
|
94
|
+
"app/components/frontend/dashboard/crawler_ui.py",
|
|
95
|
+
"app/components/frontend/dashboard/cards/crawler_card.py",
|
|
96
|
+
"app/components/frontend/dashboard/modals/crawler_modal.py",
|
|
97
|
+
"app/components/frontend/dashboard/modals/crawler_pages_tab.py",
|
|
98
|
+
"app/components/frontend/dashboard/modals/crawler_crawl_tab.py",
|
|
99
|
+
"app/cli/crawl.py",
|
|
100
|
+
]
|
|
101
|
+
),
|
|
102
|
+
wiring=PluginWiring(
|
|
103
|
+
routers=[
|
|
104
|
+
RouterWiring(
|
|
105
|
+
module="app.components.backend.api.crawler.router",
|
|
106
|
+
symbol="router",
|
|
107
|
+
prefix="/api/v1/crawler",
|
|
108
|
+
tags=["crawler"],
|
|
109
|
+
),
|
|
110
|
+
],
|
|
111
|
+
settings_mixins=[
|
|
112
|
+
SymbolWiring(
|
|
113
|
+
module="app.services.crawler.settings",
|
|
114
|
+
symbol="CrawlerSettingsMixin",
|
|
115
|
+
),
|
|
116
|
+
],
|
|
117
|
+
health_checks=[
|
|
118
|
+
HealthCheckWiring(
|
|
119
|
+
module="app.services.crawler.health",
|
|
120
|
+
symbol="crawler_health",
|
|
121
|
+
label=HEALTH_LABEL,
|
|
122
|
+
),
|
|
123
|
+
],
|
|
124
|
+
dashboard_cards=[
|
|
125
|
+
FrontendWidgetWiring(
|
|
126
|
+
module="app.components.frontend.dashboard.cards.crawler_card",
|
|
127
|
+
symbol="CrawlerCard",
|
|
128
|
+
# ``modal_id`` here doubles as the dispatch key the
|
|
129
|
+
# frontend uses to pick this card, so it is the health
|
|
130
|
+
# label itself.
|
|
131
|
+
modal_id=HEALTH_LABEL,
|
|
132
|
+
),
|
|
133
|
+
],
|
|
134
|
+
dashboard_modals=[
|
|
135
|
+
FrontendWidgetWiring(
|
|
136
|
+
module="app.components.frontend.dashboard.modals.crawler_modal",
|
|
137
|
+
symbol="CrawlerDetailDialog",
|
|
138
|
+
# The key the dashboard's click handler emits, and the
|
|
139
|
+
# one ``dashboard.crawler_ui.MODAL_KEY`` gives the card.
|
|
140
|
+
modal_id=f"service_{HEALTH_LABEL}",
|
|
141
|
+
),
|
|
142
|
+
],
|
|
143
|
+
),
|
|
144
|
+
)
|
|
@@ -0,0 +1,460 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Crawler service CLI commands.
|
|
3
|
+
|
|
4
|
+
Talks to the running webserver via the FastAPI routes — no raw DB
|
|
5
|
+
calls. Same shape as ``app/cli/health.py``: build an ``httpx`` client
|
|
6
|
+
against the configured ``API_BASE_URL`` and hit the documented
|
|
7
|
+
endpoints (``POST /crawl``, ``GET /pages``, ``GET /stats``).
|
|
8
|
+
|
|
9
|
+
Usage::
|
|
10
|
+
|
|
11
|
+
<project> crawlstats
|
|
12
|
+
<project> crawlfetch https://example.com
|
|
13
|
+
<project> crawlfetch https://example.com -m campaign=march -m source=manual
|
|
14
|
+
<project> crawllist --limit 5
|
|
15
|
+
<project> crawllist --failed
|
|
16
|
+
<project> crawlretry-failed
|
|
17
|
+
"""
|
|
18
|
+
|
|
19
|
+
from __future__ import annotations
|
|
20
|
+
|
|
21
|
+
import asyncio
|
|
22
|
+
import json
|
|
23
|
+
import sys
|
|
24
|
+
from typing import Any
|
|
25
|
+
|
|
26
|
+
import httpx
|
|
27
|
+
from rich import box
|
|
28
|
+
from rich.console import Console
|
|
29
|
+
from rich.markdown import Markdown
|
|
30
|
+
from rich.panel import Panel
|
|
31
|
+
from rich.syntax import Syntax
|
|
32
|
+
from rich.table import Table
|
|
33
|
+
import typer
|
|
34
|
+
|
|
35
|
+
from app.components.backend.api.crawler import API_PREFIX
|
|
36
|
+
from app.core.config import settings
|
|
37
|
+
from app.core.constants import Defaults
|
|
38
|
+
from app.core.formatting import format_bytes, format_relative_time
|
|
39
|
+
from app.services.crawler.options import MAX_DEPTH, CrawlOptions
|
|
40
|
+
from app.services.crawler.schemas import CrawlSummary
|
|
41
|
+
|
|
42
|
+
app = typer.Typer(
|
|
43
|
+
name="crawl",
|
|
44
|
+
help="Web crawling and scraping (Crawl4AI plugin).",
|
|
45
|
+
no_args_is_help=True,
|
|
46
|
+
)
|
|
47
|
+
console = Console()
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def _api_base() -> str:
|
|
51
|
+
# Fall back to the port this project actually serves on, not a
|
|
52
|
+
# hardcoded 8000: whatever else is listening there answers first,
|
|
53
|
+
# and its 401 looks like ours.
|
|
54
|
+
port = getattr(settings, "PORT", 8000)
|
|
55
|
+
return getattr(settings, "API_BASE_URL", None) or f"http://localhost:{port}"
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def _client() -> httpx.AsyncClient:
|
|
59
|
+
return httpx.AsyncClient(
|
|
60
|
+
base_url=_api_base(),
|
|
61
|
+
timeout=httpx.Timeout(getattr(Defaults, "API_TIMEOUT", 30.0)),
|
|
62
|
+
)
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def _status(status_code: Any) -> str:
|
|
66
|
+
"""A status code in the one colour every command agrees on: green
|
|
67
|
+
when the fetch landed, red when it did not, a dash when unknown."""
|
|
68
|
+
code = status_code or 0
|
|
69
|
+
colour = "green" if 0 < code < 400 else "red"
|
|
70
|
+
return f"[{colour}]{code or '—'}[/{colour}]"
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def _parse_metadata(pairs: list[str]) -> dict[str, Any]:
|
|
74
|
+
"""Parse ``-m key=value`` pairs into a dict.
|
|
75
|
+
|
|
76
|
+
Values that parse as JSON are decoded (so ``-m count=5`` → int);
|
|
77
|
+
otherwise treated as strings.
|
|
78
|
+
"""
|
|
79
|
+
out: dict[str, Any] = {}
|
|
80
|
+
for pair in pairs:
|
|
81
|
+
if "=" not in pair:
|
|
82
|
+
raise typer.BadParameter(
|
|
83
|
+
f"--metadata expects key=value, got {pair!r}"
|
|
84
|
+
)
|
|
85
|
+
key, raw = pair.split("=", 1)
|
|
86
|
+
try:
|
|
87
|
+
out[key] = json.loads(raw)
|
|
88
|
+
except json.JSONDecodeError:
|
|
89
|
+
out[key] = raw
|
|
90
|
+
return out
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
# ---------------------------------------------------------------------
|
|
94
|
+
# stats
|
|
95
|
+
# ---------------------------------------------------------------------
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
@app.command(help="Show document counts + last fetch timestamp.")
|
|
99
|
+
async def stats() -> None:
|
|
100
|
+
async with _client() as client:
|
|
101
|
+
resp = await client.get(f"{API_PREFIX}/stats")
|
|
102
|
+
resp.raise_for_status()
|
|
103
|
+
data = resp.json()
|
|
104
|
+
|
|
105
|
+
failed = data.get("failed", 0) or 0
|
|
106
|
+
failed_color = "red" if failed else "green"
|
|
107
|
+
|
|
108
|
+
table = Table(title="Crawler stats", show_header=False, title_style="bold magenta")
|
|
109
|
+
table.add_column("metric", style="cyan", no_wrap=True)
|
|
110
|
+
table.add_column("value", style="bold")
|
|
111
|
+
table.add_row("Pages", f"[green]{data.get('pages', 0):,}[/green]")
|
|
112
|
+
table.add_row("Failed", f"[{failed_color}]{failed:,}[/{failed_color}]")
|
|
113
|
+
table.add_row(
|
|
114
|
+
"Last fetch",
|
|
115
|
+
f"[yellow]{format_relative_time(data.get('last_fetched_at'))}[/yellow]",
|
|
116
|
+
)
|
|
117
|
+
console.print(table)
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
# ---------------------------------------------------------------------
|
|
121
|
+
# fetch
|
|
122
|
+
# ---------------------------------------------------------------------
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
@app.command(help="Fetch one URL and persist the result.")
|
|
126
|
+
async def fetch(
|
|
127
|
+
url: str = typer.Argument(..., help="URL to fetch."),
|
|
128
|
+
metadata: list[str] = typer.Option(
|
|
129
|
+
[],
|
|
130
|
+
"--metadata",
|
|
131
|
+
"-m",
|
|
132
|
+
help=(
|
|
133
|
+
"Repeatable ``key=value`` pairs attached to the stored "
|
|
134
|
+
"document. Values are JSON-decoded if possible "
|
|
135
|
+
"(``-m count=5`` → int)."
|
|
136
|
+
),
|
|
137
|
+
),
|
|
138
|
+
source_kind: str = typer.Option(
|
|
139
|
+
"crawl4ai",
|
|
140
|
+
"--source-kind",
|
|
141
|
+
help="Producer label written to the document's source_kind.",
|
|
142
|
+
),
|
|
143
|
+
fresh: bool = typer.Option(
|
|
144
|
+
False,
|
|
145
|
+
"--fresh",
|
|
146
|
+
help=(
|
|
147
|
+
"Bypass both crawl4ai's network cache and row-level dedup. "
|
|
148
|
+
"Default uses the cache (no network on repeat URLs) and "
|
|
149
|
+
"skips inserts when content hash matches an existing row. "
|
|
150
|
+
"Pass ``--fresh`` for a true re-fetch."
|
|
151
|
+
),
|
|
152
|
+
),
|
|
153
|
+
) -> None:
|
|
154
|
+
# Snapshot the most recent id so we can tell whether the API wrote
|
|
155
|
+
# a new row or returned an existing one (the fast-path dedup case).
|
|
156
|
+
# A returned id ≤ the snapshot means it was deduped (no insert).
|
|
157
|
+
async with _client() as client:
|
|
158
|
+
prior_resp = await client.get(
|
|
159
|
+
f"{API_PREFIX}/pages",
|
|
160
|
+
params={"limit": 1},
|
|
161
|
+
)
|
|
162
|
+
prior_resp.raise_for_status()
|
|
163
|
+
prior_rows = prior_resp.json()
|
|
164
|
+
prior_max_id = prior_rows[0]["id"] if prior_rows else 0
|
|
165
|
+
|
|
166
|
+
payload = {
|
|
167
|
+
"url": url,
|
|
168
|
+
"doc_metadata": _parse_metadata(metadata),
|
|
169
|
+
"source_kind": source_kind,
|
|
170
|
+
"fresh": fresh,
|
|
171
|
+
}
|
|
172
|
+
resp = await client.post(f"{API_PREFIX}/crawl", json=payload)
|
|
173
|
+
if resp.status_code >= 400:
|
|
174
|
+
console.print(
|
|
175
|
+
f"[red]Fetch failed:[/red] {resp.status_code} {resp.text}",
|
|
176
|
+
)
|
|
177
|
+
raise typer.Exit(1)
|
|
178
|
+
doc = resp.json()
|
|
179
|
+
|
|
180
|
+
status_code = doc.get("status_code") or 0
|
|
181
|
+
deduped = (not fresh) and doc["id"] <= prior_max_id
|
|
182
|
+
|
|
183
|
+
if deduped:
|
|
184
|
+
marker, suffix = "[cyan]↻[/cyan]", " [cyan](cached • hash match)[/cyan]"
|
|
185
|
+
elif status_code < 400:
|
|
186
|
+
marker, suffix = "[green]✓[/green]", ""
|
|
187
|
+
else:
|
|
188
|
+
marker, suffix = "[red]✗[/red]", ""
|
|
189
|
+
|
|
190
|
+
console.print(
|
|
191
|
+
f"{marker} "
|
|
192
|
+
f"[cyan]id={doc['id']}[/cyan] "
|
|
193
|
+
f"status={_status(status_code)} "
|
|
194
|
+
f"[yellow]type={doc.get('content_type')}[/yellow] "
|
|
195
|
+
f"[dim]bytes={format_bytes(len(doc.get('content') or ''))}[/dim]"
|
|
196
|
+
f"{suffix}"
|
|
197
|
+
)
|
|
198
|
+
|
|
199
|
+
|
|
200
|
+
# ---------------------------------------------------------------------
|
|
201
|
+
# list
|
|
202
|
+
# ---------------------------------------------------------------------
|
|
203
|
+
|
|
204
|
+
|
|
205
|
+
@app.command(help="Crawl one or more URLs as a background job.")
|
|
206
|
+
async def site(
|
|
207
|
+
urls: list[str] = typer.Argument(..., help="Seed URLs. A bare domain is fine."),
|
|
208
|
+
depth: int = typer.Option(
|
|
209
|
+
0, "--depth", "-d", min=0, max=MAX_DEPTH, help="How many links deep to follow."
|
|
210
|
+
),
|
|
211
|
+
max_pages: int = typer.Option(
|
|
212
|
+
25, "--max-pages", min=1, max=500, help="Ceiling on pages per seed URL."
|
|
213
|
+
),
|
|
214
|
+
selector: str = typer.Option(
|
|
215
|
+
"", "--selector", help="CSS selector: keep only this part of each page."
|
|
216
|
+
),
|
|
217
|
+
min_words: int = typer.Option(
|
|
218
|
+
0, "--min-words", min=0, help="Drop blocks shorter than this many words."
|
|
219
|
+
),
|
|
220
|
+
external: bool = typer.Option(
|
|
221
|
+
False, "--external", help="Follow links off the seed URL's site."
|
|
222
|
+
),
|
|
223
|
+
fresh: bool = typer.Option(False, "--fresh", help="Bypass cache and dedup."),
|
|
224
|
+
source_kind: str = typer.Option("crawl4ai", "--source-kind"),
|
|
225
|
+
) -> None:
|
|
226
|
+
"""The same job the dashboard starts, from a terminal."""
|
|
227
|
+
options = CrawlOptions(
|
|
228
|
+
depth=depth,
|
|
229
|
+
max_pages=max_pages,
|
|
230
|
+
selector=selector or None,
|
|
231
|
+
min_words=min_words,
|
|
232
|
+
stay_on_domain=not external,
|
|
233
|
+
fresh=fresh,
|
|
234
|
+
)
|
|
235
|
+
|
|
236
|
+
async with _client() as client:
|
|
237
|
+
resp = await client.post(
|
|
238
|
+
f"{API_PREFIX}/crawl-batch",
|
|
239
|
+
json={
|
|
240
|
+
"urls": urls,
|
|
241
|
+
"options": options.model_dump(),
|
|
242
|
+
"source_kind": source_kind,
|
|
243
|
+
},
|
|
244
|
+
)
|
|
245
|
+
resp.raise_for_status()
|
|
246
|
+
job_id = resp.json()["job_id"]
|
|
247
|
+
|
|
248
|
+
console.print(f"[dim]job {job_id}[/dim]")
|
|
249
|
+
with console.status("Crawling...") as status:
|
|
250
|
+
while True:
|
|
251
|
+
await asyncio.sleep(1.0)
|
|
252
|
+
job = (await client.get(f"/api/v1/jobs/{job_id}")).json()
|
|
253
|
+
status.update(job.get("label") or "Crawling...")
|
|
254
|
+
if job.get("status") != "running":
|
|
255
|
+
break
|
|
256
|
+
|
|
257
|
+
if job.get("status") != "done":
|
|
258
|
+
console.print(f"[red]{job.get('error') or 'The crawl failed.'}[/red]")
|
|
259
|
+
raise typer.Exit(1)
|
|
260
|
+
|
|
261
|
+
summary = CrawlSummary.model_validate(job.get("result") or {})
|
|
262
|
+
table = Table(title=f"Crawled {summary.fetched} page(s)", box=box.SIMPLE)
|
|
263
|
+
table.add_column("status", justify="right")
|
|
264
|
+
table.add_column("size", justify="right")
|
|
265
|
+
table.add_column("url")
|
|
266
|
+
for page in summary.pages:
|
|
267
|
+
table.add_row(_status(page.status), format_bytes(page.size), page.url)
|
|
268
|
+
console.print(table)
|
|
269
|
+
if summary.failed:
|
|
270
|
+
console.print(f"[yellow]{summary.failed} failed[/yellow]")
|
|
271
|
+
|
|
272
|
+
|
|
273
|
+
@app.command("list", help="List recent pages.")
|
|
274
|
+
async def list_pages(
|
|
275
|
+
limit: int = typer.Option(20, "--limit", "-n", help="Max rows to show."),
|
|
276
|
+
failed: bool = typer.Option(
|
|
277
|
+
False, "--failed", help="Only show fetches with status >= 400."
|
|
278
|
+
),
|
|
279
|
+
) -> None:
|
|
280
|
+
params: dict[str, Any] = {"limit": limit}
|
|
281
|
+
if failed:
|
|
282
|
+
params["failed_only"] = "true"
|
|
283
|
+
|
|
284
|
+
async with _client() as client:
|
|
285
|
+
resp = await client.get(f"{API_PREFIX}/pages", params=params)
|
|
286
|
+
resp.raise_for_status()
|
|
287
|
+
rows = resp.json()
|
|
288
|
+
|
|
289
|
+
if not rows:
|
|
290
|
+
console.print("[yellow]No pages yet.[/yellow]")
|
|
291
|
+
return
|
|
292
|
+
|
|
293
|
+
table = Table(
|
|
294
|
+
title=f"Crawled pages ({len(rows)})",
|
|
295
|
+
show_header=True,
|
|
296
|
+
header_style="bold magenta",
|
|
297
|
+
title_style="bold magenta",
|
|
298
|
+
)
|
|
299
|
+
table.add_column("id", justify="right", style="cyan", no_wrap=True)
|
|
300
|
+
table.add_column("status", justify="right", style="bold")
|
|
301
|
+
table.add_column("type", style="yellow")
|
|
302
|
+
table.add_column("bytes", justify="right", style="dim")
|
|
303
|
+
table.add_column("fetched_at", style="dim")
|
|
304
|
+
table.add_column("url", style="green", overflow="fold")
|
|
305
|
+
for r in rows:
|
|
306
|
+
table.add_row(
|
|
307
|
+
str(r["id"]),
|
|
308
|
+
_status(r.get("status_code")),
|
|
309
|
+
r.get("content_type") or "",
|
|
310
|
+
format_bytes(len(r.get("content") or "")),
|
|
311
|
+
format_relative_time(r.get("fetched_at")),
|
|
312
|
+
r.get("source_url") or "",
|
|
313
|
+
)
|
|
314
|
+
console.print(table)
|
|
315
|
+
console.print(f"\n[cyan]Total pages shown:[/cyan] {len(rows)}")
|
|
316
|
+
|
|
317
|
+
|
|
318
|
+
# ---------------------------------------------------------------------
|
|
319
|
+
# show
|
|
320
|
+
# ---------------------------------------------------------------------
|
|
321
|
+
|
|
322
|
+
|
|
323
|
+
@app.command(help="Show one document's body + metadata.")
|
|
324
|
+
async def show(
|
|
325
|
+
page_id: int = typer.Argument(..., help="CrawledPage id to render."),
|
|
326
|
+
raw: bool = typer.Option(
|
|
327
|
+
False,
|
|
328
|
+
"--raw",
|
|
329
|
+
help="Print body as plain text instead of rendered Markdown / "
|
|
330
|
+
"syntax-highlighted HTML.",
|
|
331
|
+
),
|
|
332
|
+
head: int = typer.Option(
|
|
333
|
+
0,
|
|
334
|
+
"--head",
|
|
335
|
+
"-h",
|
|
336
|
+
help="Print only the first N characters of the body. 0 = full.",
|
|
337
|
+
),
|
|
338
|
+
) -> None:
|
|
339
|
+
async with _client() as client:
|
|
340
|
+
resp = await client.get(f"{API_PREFIX}/pages/{page_id}")
|
|
341
|
+
if resp.status_code == 404:
|
|
342
|
+
console.print(f"[red]No document with id={page_id}[/red]")
|
|
343
|
+
raise typer.Exit(1)
|
|
344
|
+
resp.raise_for_status()
|
|
345
|
+
doc = resp.json()
|
|
346
|
+
|
|
347
|
+
status_code = doc.get("status_code") or 0
|
|
348
|
+
|
|
349
|
+
# Header table — id, url, status, content_type, fetched_at, metadata.
|
|
350
|
+
header = Table(
|
|
351
|
+
title=f"CrawledPage #{doc['id']}",
|
|
352
|
+
show_header=False,
|
|
353
|
+
box=None,
|
|
354
|
+
pad_edge=False,
|
|
355
|
+
title_style="bold magenta",
|
|
356
|
+
)
|
|
357
|
+
header.add_column("k", style="cyan", no_wrap=True)
|
|
358
|
+
header.add_column("v", overflow="fold")
|
|
359
|
+
header.add_row("source_url", f"[green]{doc.get('source_url') or ''}[/green]")
|
|
360
|
+
header.add_row("status", _status(status_code))
|
|
361
|
+
header.add_row("content_type", f"[yellow]{doc.get('content_type') or ''}[/yellow]")
|
|
362
|
+
header.add_row("source_kind", f"[dim]{doc.get('source_kind') or ''}[/dim]")
|
|
363
|
+
header.add_row(
|
|
364
|
+
"fetched_at", f"[dim]{format_relative_time(doc.get('fetched_at'))}[/dim]"
|
|
365
|
+
)
|
|
366
|
+
header.add_row("hash", f"[dim]{doc.get('content_hash') or ''}[/dim]")
|
|
367
|
+
if doc.get("doc_metadata"):
|
|
368
|
+
header.add_row("metadata", json.dumps(doc["doc_metadata"]))
|
|
369
|
+
console.print(header)
|
|
370
|
+
console.print()
|
|
371
|
+
|
|
372
|
+
body = doc.get("content") or ""
|
|
373
|
+
if not body:
|
|
374
|
+
console.print("[dim](no body)[/dim]")
|
|
375
|
+
return
|
|
376
|
+
if head and head > 0:
|
|
377
|
+
body = body[:head]
|
|
378
|
+
|
|
379
|
+
if raw:
|
|
380
|
+
console.print(body)
|
|
381
|
+
return
|
|
382
|
+
|
|
383
|
+
content_type = (doc.get("content_type") or "").lower()
|
|
384
|
+
if content_type == "markdown":
|
|
385
|
+
console.print(Panel(Markdown(body), title="content", border_style="dim"))
|
|
386
|
+
elif content_type in ("html", "json"):
|
|
387
|
+
console.print(
|
|
388
|
+
Panel(
|
|
389
|
+
Syntax(body, content_type, theme="ansi_dark", word_wrap=True),
|
|
390
|
+
title=f"content ({content_type})",
|
|
391
|
+
border_style="dim",
|
|
392
|
+
)
|
|
393
|
+
)
|
|
394
|
+
else:
|
|
395
|
+
console.print(Panel(body, title="content", border_style="dim"))
|
|
396
|
+
|
|
397
|
+
|
|
398
|
+
# ---------------------------------------------------------------------
|
|
399
|
+
# retry-failed
|
|
400
|
+
# ---------------------------------------------------------------------
|
|
401
|
+
|
|
402
|
+
|
|
403
|
+
@app.command("retry-failed", help="Re-fetch every document with status >= 400.")
|
|
404
|
+
async def retry_failed(
|
|
405
|
+
limit: int = typer.Option(50, "--limit", "-n", help="Max retries."),
|
|
406
|
+
) -> None:
|
|
407
|
+
async with _client() as client:
|
|
408
|
+
resp = await client.get(
|
|
409
|
+
f"{API_PREFIX}/pages",
|
|
410
|
+
params={"limit": limit, "failed_only": "true"},
|
|
411
|
+
)
|
|
412
|
+
resp.raise_for_status()
|
|
413
|
+
failed_rows = resp.json()
|
|
414
|
+
|
|
415
|
+
if not failed_rows:
|
|
416
|
+
console.print("[green]✓ Nothing to retry.[/green]")
|
|
417
|
+
return
|
|
418
|
+
|
|
419
|
+
console.print(
|
|
420
|
+
f"[bold yellow]Retrying {len(failed_rows)} failed fetches…[/bold yellow]"
|
|
421
|
+
)
|
|
422
|
+
successes = 0
|
|
423
|
+
for row in failed_rows:
|
|
424
|
+
url = row.get("source_url")
|
|
425
|
+
if not url:
|
|
426
|
+
continue
|
|
427
|
+
payload = {
|
|
428
|
+
"url": url,
|
|
429
|
+
"doc_metadata": row.get("doc_metadata") or {},
|
|
430
|
+
"source_kind": row.get("source_kind") or "crawl4ai",
|
|
431
|
+
}
|
|
432
|
+
try:
|
|
433
|
+
r = await client.post(f"{API_PREFIX}/crawl", json=payload)
|
|
434
|
+
if r.status_code < 400:
|
|
435
|
+
new_doc = r.json()
|
|
436
|
+
new_status = new_doc.get("status_code") or 0
|
|
437
|
+
if new_status < 400:
|
|
438
|
+
successes += 1
|
|
439
|
+
console.print(f" [green]✓[/green] [dim]{url}[/dim]")
|
|
440
|
+
else:
|
|
441
|
+
console.print(
|
|
442
|
+
f" [red]✗[/red] [dim]{url}[/dim] "
|
|
443
|
+
f"[red](status {new_status})[/red]"
|
|
444
|
+
)
|
|
445
|
+
except httpx.HTTPError as exc:
|
|
446
|
+
console.print(
|
|
447
|
+
f" [red]✗[/red] [dim]{url}[/dim]: [red]{exc}[/red]",
|
|
448
|
+
markup=True,
|
|
449
|
+
)
|
|
450
|
+
|
|
451
|
+
fail_count = len(failed_rows) - successes
|
|
452
|
+
console.print(
|
|
453
|
+
f"\n[bold]Retried[/bold] [cyan]{len(failed_rows)}[/cyan] • "
|
|
454
|
+
f"[green]{successes} succeeded[/green] • "
|
|
455
|
+
f"[red]{fail_count} still failing[/red]"
|
|
456
|
+
)
|
|
457
|
+
|
|
458
|
+
|
|
459
|
+
if __name__ == "__main__":
|
|
460
|
+
sys.exit(app())
|
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
"""Crawler service API package.
|
|
2
|
+
|
|
3
|
+
``API_PREFIX`` is the one place the crawler's mount point is spelled.
|
|
4
|
+
The plugin spec mounts the router there, the dashboard builds its
|
|
5
|
+
endpoints from it and the CLI calls it, so none of them can drift.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
API_PREFIX = "/api/v1/crawler"
|