aegis-stack-crawl4ai 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (33) hide show
  1. aegis_stack_crawl4ai/__init__.py +1 -0
  2. aegis_stack_crawl4ai/plugin.py +144 -0
  3. aegis_stack_crawl4ai/templates/{{ project_slug }}/app/cli/crawl.py.jinja +460 -0
  4. aegis_stack_crawl4ai/templates/{{ project_slug }}/app/components/backend/api/crawler/__init__.py.jinja +8 -0
  5. aegis_stack_crawl4ai/templates/{{ project_slug }}/app/components/backend/api/crawler/router.py.jinja +122 -0
  6. aegis_stack_crawl4ai/templates/{{ project_slug }}/app/components/frontend/dashboard/cards/crawler_card.py.jinja +84 -0
  7. aegis_stack_crawl4ai/templates/{{ project_slug }}/app/components/frontend/dashboard/crawler_ui.py.jinja +42 -0
  8. aegis_stack_crawl4ai/templates/{{ project_slug }}/app/components/frontend/dashboard/modals/crawler_crawl_tab.py.jinja +337 -0
  9. aegis_stack_crawl4ai/templates/{{ project_slug }}/app/components/frontend/dashboard/modals/crawler_modal.py.jinja +199 -0
  10. aegis_stack_crawl4ai/templates/{{ project_slug }}/app/components/frontend/dashboard/modals/crawler_pages_tab.py.jinja +255 -0
  11. aegis_stack_crawl4ai/templates/{{ project_slug }}/app/services/crawler/__init__.py.jinja +15 -0
  12. aegis_stack_crawl4ai/templates/{{ project_slug }}/app/services/crawler/deps.py.jinja +34 -0
  13. aegis_stack_crawl4ai/templates/{{ project_slug }}/app/services/crawler/dispatch.py.jinja +46 -0
  14. aegis_stack_crawl4ai/templates/{{ project_slug }}/app/services/crawler/health.py.jinja +61 -0
  15. aegis_stack_crawl4ai/templates/{{ project_slug }}/app/services/crawler/jobs.py.jinja +79 -0
  16. aegis_stack_crawl4ai/templates/{{ project_slug }}/app/services/crawler/models.py.jinja +48 -0
  17. aegis_stack_crawl4ai/templates/{{ project_slug }}/app/services/crawler/options.py.jinja +84 -0
  18. aegis_stack_crawl4ai/templates/{{ project_slug }}/app/services/crawler/queries.py.jinja +112 -0
  19. aegis_stack_crawl4ai/templates/{{ project_slug }}/app/services/crawler/schemas.py.jinja +137 -0
  20. aegis_stack_crawl4ai/templates/{{ project_slug }}/app/services/crawler/service.py.jinja +440 -0
  21. aegis_stack_crawl4ai/templates/{{ project_slug }}/app/services/crawler/settings.py.jinja +35 -0
  22. aegis_stack_crawl4ai/templates/{{ project_slug }}/app/services/crawler/urls.py.jinja +27 -0
  23. aegis_stack_crawl4ai/templates/{{ project_slug }}/tests/components/frontend/test_crawler_tab.py.jinja +230 -0
  24. aegis_stack_crawl4ai/templates/{{ project_slug }}/tests/services/test_crawler_jobs.py.jinja +234 -0
  25. aegis_stack_crawl4ai/templates/{{ project_slug }}/tests/services/test_crawler_layout.py.jinja +49 -0
  26. aegis_stack_crawl4ai/templates/{{ project_slug }}/tests/services/test_crawler_options.py.jinja +197 -0
  27. aegis_stack_crawl4ai/templates/{{ project_slug }}/tests/services/test_crawler_sites.py.jinja +110 -0
  28. aegis_stack_crawl4ai/templates/{{ project_slug }}/tests/services/test_crawler_wiring.py.jinja +48 -0
  29. aegis_stack_crawl4ai-0.1.0.dist-info/METADATA +317 -0
  30. aegis_stack_crawl4ai-0.1.0.dist-info/RECORD +33 -0
  31. aegis_stack_crawl4ai-0.1.0.dist-info/WHEEL +4 -0
  32. aegis_stack_crawl4ai-0.1.0.dist-info/entry_points.txt +2 -0
  33. aegis_stack_crawl4ai-0.1.0.dist-info/licenses/LICENSE +202 -0
@@ -0,0 +1 @@
1
+ """Aegis Stack plugin: crawl4ai"""
@@ -0,0 +1,144 @@
1
+ """PluginSpec for ``aegis-stack-crawl4ai``.
2
+
3
+ Declares a service-flavoured plugin that adds web-crawling-to-markdown
4
+ capability to an Aegis Stack project. Wraps the upstream ``crawl4ai``
5
+ library (Apache 2.0). Aegis discovers this spec via the
6
+ ``aegis.plugins`` entry point in ``pyproject.toml`` (see
7
+ ``aegis.core.plugins.discovery``).
8
+
9
+ Design notes:
10
+
11
+ - ``MigrationSpec(schema="crawler")`` — the plugin's tables live in
12
+ their own Postgres schema, isolated from the project's other services.
13
+ SQLite has no schemas; there the table lands unqualified.
14
+ - ``required_components=['database']`` — the resolver auto-installs
15
+ the database component if the target project doesn't have it.
16
+ - ``crawled_page`` table is generic by design (URL + content + metadata
17
+ JSONB) so future ingestion plugins (RSS, PDF, sitemap) can either
18
+ share the schema or follow the same shape.
19
+ """
20
+
21
+ from aegis.core.file_manifest import FileManifest
22
+ from aegis.core.migration_spec import MigrationSpec
23
+ from aegis.core.plugins.spec import (
24
+ FrontendWidgetWiring,
25
+ HealthCheckWiring,
26
+ PluginKind,
27
+ PluginSpec,
28
+ PluginWiring,
29
+ RouterWiring,
30
+ SymbolWiring,
31
+ )
32
+
33
+ # The name the health check reports under. The dashboard groups services
34
+ # by it, the card factory dispatches on it, and the modal registers under
35
+ # ``service_<label>``; the rendered code mirrors it in
36
+ # ``app/components/frontend/dashboard/crawler_ui.py``, which derives it
37
+ # from ``SERVICE_NAME`` in the service package. One spelling, because a
38
+ # mismatch is a dead click rather than an error.
39
+ HEALTH_LABEL = "Crawler"
40
+
41
+
42
+ def get_spec() -> PluginSpec:
43
+ """Return the plugin spec."""
44
+ return PluginSpec(
45
+ name="crawl4ai",
46
+ kind=PluginKind.SERVICE,
47
+ description="Web crawling and scraping via Crawl4AI",
48
+ version="0.1.0",
49
+ verified=False,
50
+ # PEP 440 specifier — the spec declares no tables (a revision is
51
+ # derived from the model), which the CLI only understands from
52
+ # 0.12 on. 0.13.1 is where the generated card-render test stops
53
+ # failing on a plugin's card, so anything older installs and then
54
+ # hands the user a red ``make check``.
55
+ aegis_version=">=0.13.1",
56
+ # CLI verb the plugin exposes in the generated project. Decoupled
57
+ # from the install identifier (``crawl4ai``) so users type the
58
+ # natural ``<project> crawl ...`` instead of the package name.
59
+ cli_name="crawl",
60
+ # Resolver auto-installs database if missing.
61
+ required_components=["database"],
62
+ # Pinned to >=0.8.6 because 0.8.6 is the supply-chain hotfix
63
+ # that swapped the compromised ``litellm`` package out — earlier
64
+ # 0.8.x builds pull the bad transitive dep.
65
+ # ``alembic`` is required because this plugin ships migrations;
66
+ # database-only projects (no auth / no insights) don't pull it
67
+ # transitively, so we declare it here.
68
+ pyproject_deps=["crawl4ai>=0.8.6", "alembic>=1.13"],
69
+ # Generic ``pages`` table — URL + content + JSONB metadata.
70
+ # Indexed on source_url, content_hash, and fetched_at for the
71
+ # three common query shapes (last fetch, dedup, time-window).
72
+ # ``doc_metadata`` rather than ``metadata`` because the latter
73
+ # is reserved on SQLAlchemy's ``DeclarativeBase``.
74
+ migrations=[
75
+ MigrationSpec(
76
+ service_name="crawler",
77
+ description="Crawled page store",
78
+ # Proof it ran: the startup hook stamps instead of
79
+ # replaying DDL on a database that already has this.
80
+ stamp_signature=("table", "crawler.crawled_page"),
81
+ # Crawler tables live in their own ``crawler`` Postgres
82
+ # schema. SQLite has no schemas: the generator drops the
83
+ # qualifier there and the model gates ``__table_args__``
84
+ # on the engine, so one declaration serves both.
85
+ schema="crawler",
86
+ ),
87
+ ],
88
+ # All files this plugin owns inside the target project.
89
+ # ``aegis remove`` walks this list to clean up.
90
+ files=FileManifest(
91
+ primary=[
92
+ "app/services/crawler",
93
+ "app/components/backend/api/crawler",
94
+ "app/components/frontend/dashboard/crawler_ui.py",
95
+ "app/components/frontend/dashboard/cards/crawler_card.py",
96
+ "app/components/frontend/dashboard/modals/crawler_modal.py",
97
+ "app/components/frontend/dashboard/modals/crawler_pages_tab.py",
98
+ "app/components/frontend/dashboard/modals/crawler_crawl_tab.py",
99
+ "app/cli/crawl.py",
100
+ ]
101
+ ),
102
+ wiring=PluginWiring(
103
+ routers=[
104
+ RouterWiring(
105
+ module="app.components.backend.api.crawler.router",
106
+ symbol="router",
107
+ prefix="/api/v1/crawler",
108
+ tags=["crawler"],
109
+ ),
110
+ ],
111
+ settings_mixins=[
112
+ SymbolWiring(
113
+ module="app.services.crawler.settings",
114
+ symbol="CrawlerSettingsMixin",
115
+ ),
116
+ ],
117
+ health_checks=[
118
+ HealthCheckWiring(
119
+ module="app.services.crawler.health",
120
+ symbol="crawler_health",
121
+ label=HEALTH_LABEL,
122
+ ),
123
+ ],
124
+ dashboard_cards=[
125
+ FrontendWidgetWiring(
126
+ module="app.components.frontend.dashboard.cards.crawler_card",
127
+ symbol="CrawlerCard",
128
+ # ``modal_id`` here doubles as the dispatch key the
129
+ # frontend uses to pick this card, so it is the health
130
+ # label itself.
131
+ modal_id=HEALTH_LABEL,
132
+ ),
133
+ ],
134
+ dashboard_modals=[
135
+ FrontendWidgetWiring(
136
+ module="app.components.frontend.dashboard.modals.crawler_modal",
137
+ symbol="CrawlerDetailDialog",
138
+ # The key the dashboard's click handler emits, and the
139
+ # one ``dashboard.crawler_ui.MODAL_KEY`` gives the card.
140
+ modal_id=f"service_{HEALTH_LABEL}",
141
+ ),
142
+ ],
143
+ ),
144
+ )
@@ -0,0 +1,460 @@
1
+ """
2
+ Crawler service CLI commands.
3
+
4
+ Talks to the running webserver via the FastAPI routes — no raw DB
5
+ calls. Same shape as ``app/cli/health.py``: build an ``httpx`` client
6
+ against the configured ``API_BASE_URL`` and hit the documented
7
+ endpoints (``POST /crawl``, ``GET /pages``, ``GET /stats``).
8
+
9
+ Usage::
10
+
11
+ <project> crawlstats
12
+ <project> crawlfetch https://example.com
13
+ <project> crawlfetch https://example.com -m campaign=march -m source=manual
14
+ <project> crawllist --limit 5
15
+ <project> crawllist --failed
16
+ <project> crawlretry-failed
17
+ """
18
+
19
+ from __future__ import annotations
20
+
21
+ import asyncio
22
+ import json
23
+ import sys
24
+ from typing import Any
25
+
26
+ import httpx
27
+ from rich import box
28
+ from rich.console import Console
29
+ from rich.markdown import Markdown
30
+ from rich.panel import Panel
31
+ from rich.syntax import Syntax
32
+ from rich.table import Table
33
+ import typer
34
+
35
+ from app.components.backend.api.crawler import API_PREFIX
36
+ from app.core.config import settings
37
+ from app.core.constants import Defaults
38
+ from app.core.formatting import format_bytes, format_relative_time
39
+ from app.services.crawler.options import MAX_DEPTH, CrawlOptions
40
+ from app.services.crawler.schemas import CrawlSummary
41
+
42
+ app = typer.Typer(
43
+ name="crawl",
44
+ help="Web crawling and scraping (Crawl4AI plugin).",
45
+ no_args_is_help=True,
46
+ )
47
+ console = Console()
48
+
49
+
50
+ def _api_base() -> str:
51
+ # Fall back to the port this project actually serves on, not a
52
+ # hardcoded 8000: whatever else is listening there answers first,
53
+ # and its 401 looks like ours.
54
+ port = getattr(settings, "PORT", 8000)
55
+ return getattr(settings, "API_BASE_URL", None) or f"http://localhost:{port}"
56
+
57
+
58
+ def _client() -> httpx.AsyncClient:
59
+ return httpx.AsyncClient(
60
+ base_url=_api_base(),
61
+ timeout=httpx.Timeout(getattr(Defaults, "API_TIMEOUT", 30.0)),
62
+ )
63
+
64
+
65
+ def _status(status_code: Any) -> str:
66
+ """A status code in the one colour every command agrees on: green
67
+ when the fetch landed, red when it did not, a dash when unknown."""
68
+ code = status_code or 0
69
+ colour = "green" if 0 < code < 400 else "red"
70
+ return f"[{colour}]{code or '—'}[/{colour}]"
71
+
72
+
73
+ def _parse_metadata(pairs: list[str]) -> dict[str, Any]:
74
+ """Parse ``-m key=value`` pairs into a dict.
75
+
76
+ Values that parse as JSON are decoded (so ``-m count=5`` → int);
77
+ otherwise treated as strings.
78
+ """
79
+ out: dict[str, Any] = {}
80
+ for pair in pairs:
81
+ if "=" not in pair:
82
+ raise typer.BadParameter(
83
+ f"--metadata expects key=value, got {pair!r}"
84
+ )
85
+ key, raw = pair.split("=", 1)
86
+ try:
87
+ out[key] = json.loads(raw)
88
+ except json.JSONDecodeError:
89
+ out[key] = raw
90
+ return out
91
+
92
+
93
+ # ---------------------------------------------------------------------
94
+ # stats
95
+ # ---------------------------------------------------------------------
96
+
97
+
98
+ @app.command(help="Show document counts + last fetch timestamp.")
99
+ async def stats() -> None:
100
+ async with _client() as client:
101
+ resp = await client.get(f"{API_PREFIX}/stats")
102
+ resp.raise_for_status()
103
+ data = resp.json()
104
+
105
+ failed = data.get("failed", 0) or 0
106
+ failed_color = "red" if failed else "green"
107
+
108
+ table = Table(title="Crawler stats", show_header=False, title_style="bold magenta")
109
+ table.add_column("metric", style="cyan", no_wrap=True)
110
+ table.add_column("value", style="bold")
111
+ table.add_row("Pages", f"[green]{data.get('pages', 0):,}[/green]")
112
+ table.add_row("Failed", f"[{failed_color}]{failed:,}[/{failed_color}]")
113
+ table.add_row(
114
+ "Last fetch",
115
+ f"[yellow]{format_relative_time(data.get('last_fetched_at'))}[/yellow]",
116
+ )
117
+ console.print(table)
118
+
119
+
120
+ # ---------------------------------------------------------------------
121
+ # fetch
122
+ # ---------------------------------------------------------------------
123
+
124
+
125
+ @app.command(help="Fetch one URL and persist the result.")
126
+ async def fetch(
127
+ url: str = typer.Argument(..., help="URL to fetch."),
128
+ metadata: list[str] = typer.Option(
129
+ [],
130
+ "--metadata",
131
+ "-m",
132
+ help=(
133
+ "Repeatable ``key=value`` pairs attached to the stored "
134
+ "document. Values are JSON-decoded if possible "
135
+ "(``-m count=5`` → int)."
136
+ ),
137
+ ),
138
+ source_kind: str = typer.Option(
139
+ "crawl4ai",
140
+ "--source-kind",
141
+ help="Producer label written to the document's source_kind.",
142
+ ),
143
+ fresh: bool = typer.Option(
144
+ False,
145
+ "--fresh",
146
+ help=(
147
+ "Bypass both crawl4ai's network cache and row-level dedup. "
148
+ "Default uses the cache (no network on repeat URLs) and "
149
+ "skips inserts when content hash matches an existing row. "
150
+ "Pass ``--fresh`` for a true re-fetch."
151
+ ),
152
+ ),
153
+ ) -> None:
154
+ # Snapshot the most recent id so we can tell whether the API wrote
155
+ # a new row or returned an existing one (the fast-path dedup case).
156
+ # A returned id ≤ the snapshot means it was deduped (no insert).
157
+ async with _client() as client:
158
+ prior_resp = await client.get(
159
+ f"{API_PREFIX}/pages",
160
+ params={"limit": 1},
161
+ )
162
+ prior_resp.raise_for_status()
163
+ prior_rows = prior_resp.json()
164
+ prior_max_id = prior_rows[0]["id"] if prior_rows else 0
165
+
166
+ payload = {
167
+ "url": url,
168
+ "doc_metadata": _parse_metadata(metadata),
169
+ "source_kind": source_kind,
170
+ "fresh": fresh,
171
+ }
172
+ resp = await client.post(f"{API_PREFIX}/crawl", json=payload)
173
+ if resp.status_code >= 400:
174
+ console.print(
175
+ f"[red]Fetch failed:[/red] {resp.status_code} {resp.text}",
176
+ )
177
+ raise typer.Exit(1)
178
+ doc = resp.json()
179
+
180
+ status_code = doc.get("status_code") or 0
181
+ deduped = (not fresh) and doc["id"] <= prior_max_id
182
+
183
+ if deduped:
184
+ marker, suffix = "[cyan]↻[/cyan]", " [cyan](cached • hash match)[/cyan]"
185
+ elif status_code < 400:
186
+ marker, suffix = "[green]✓[/green]", ""
187
+ else:
188
+ marker, suffix = "[red]✗[/red]", ""
189
+
190
+ console.print(
191
+ f"{marker} "
192
+ f"[cyan]id={doc['id']}[/cyan] "
193
+ f"status={_status(status_code)} "
194
+ f"[yellow]type={doc.get('content_type')}[/yellow] "
195
+ f"[dim]bytes={format_bytes(len(doc.get('content') or ''))}[/dim]"
196
+ f"{suffix}"
197
+ )
198
+
199
+
200
+ # ---------------------------------------------------------------------
201
+ # list
202
+ # ---------------------------------------------------------------------
203
+
204
+
205
+ @app.command(help="Crawl one or more URLs as a background job.")
206
+ async def site(
207
+ urls: list[str] = typer.Argument(..., help="Seed URLs. A bare domain is fine."),
208
+ depth: int = typer.Option(
209
+ 0, "--depth", "-d", min=0, max=MAX_DEPTH, help="How many links deep to follow."
210
+ ),
211
+ max_pages: int = typer.Option(
212
+ 25, "--max-pages", min=1, max=500, help="Ceiling on pages per seed URL."
213
+ ),
214
+ selector: str = typer.Option(
215
+ "", "--selector", help="CSS selector: keep only this part of each page."
216
+ ),
217
+ min_words: int = typer.Option(
218
+ 0, "--min-words", min=0, help="Drop blocks shorter than this many words."
219
+ ),
220
+ external: bool = typer.Option(
221
+ False, "--external", help="Follow links off the seed URL's site."
222
+ ),
223
+ fresh: bool = typer.Option(False, "--fresh", help="Bypass cache and dedup."),
224
+ source_kind: str = typer.Option("crawl4ai", "--source-kind"),
225
+ ) -> None:
226
+ """The same job the dashboard starts, from a terminal."""
227
+ options = CrawlOptions(
228
+ depth=depth,
229
+ max_pages=max_pages,
230
+ selector=selector or None,
231
+ min_words=min_words,
232
+ stay_on_domain=not external,
233
+ fresh=fresh,
234
+ )
235
+
236
+ async with _client() as client:
237
+ resp = await client.post(
238
+ f"{API_PREFIX}/crawl-batch",
239
+ json={
240
+ "urls": urls,
241
+ "options": options.model_dump(),
242
+ "source_kind": source_kind,
243
+ },
244
+ )
245
+ resp.raise_for_status()
246
+ job_id = resp.json()["job_id"]
247
+
248
+ console.print(f"[dim]job {job_id}[/dim]")
249
+ with console.status("Crawling...") as status:
250
+ while True:
251
+ await asyncio.sleep(1.0)
252
+ job = (await client.get(f"/api/v1/jobs/{job_id}")).json()
253
+ status.update(job.get("label") or "Crawling...")
254
+ if job.get("status") != "running":
255
+ break
256
+
257
+ if job.get("status") != "done":
258
+ console.print(f"[red]{job.get('error') or 'The crawl failed.'}[/red]")
259
+ raise typer.Exit(1)
260
+
261
+ summary = CrawlSummary.model_validate(job.get("result") or {})
262
+ table = Table(title=f"Crawled {summary.fetched} page(s)", box=box.SIMPLE)
263
+ table.add_column("status", justify="right")
264
+ table.add_column("size", justify="right")
265
+ table.add_column("url")
266
+ for page in summary.pages:
267
+ table.add_row(_status(page.status), format_bytes(page.size), page.url)
268
+ console.print(table)
269
+ if summary.failed:
270
+ console.print(f"[yellow]{summary.failed} failed[/yellow]")
271
+
272
+
273
+ @app.command("list", help="List recent pages.")
274
+ async def list_pages(
275
+ limit: int = typer.Option(20, "--limit", "-n", help="Max rows to show."),
276
+ failed: bool = typer.Option(
277
+ False, "--failed", help="Only show fetches with status >= 400."
278
+ ),
279
+ ) -> None:
280
+ params: dict[str, Any] = {"limit": limit}
281
+ if failed:
282
+ params["failed_only"] = "true"
283
+
284
+ async with _client() as client:
285
+ resp = await client.get(f"{API_PREFIX}/pages", params=params)
286
+ resp.raise_for_status()
287
+ rows = resp.json()
288
+
289
+ if not rows:
290
+ console.print("[yellow]No pages yet.[/yellow]")
291
+ return
292
+
293
+ table = Table(
294
+ title=f"Crawled pages ({len(rows)})",
295
+ show_header=True,
296
+ header_style="bold magenta",
297
+ title_style="bold magenta",
298
+ )
299
+ table.add_column("id", justify="right", style="cyan", no_wrap=True)
300
+ table.add_column("status", justify="right", style="bold")
301
+ table.add_column("type", style="yellow")
302
+ table.add_column("bytes", justify="right", style="dim")
303
+ table.add_column("fetched_at", style="dim")
304
+ table.add_column("url", style="green", overflow="fold")
305
+ for r in rows:
306
+ table.add_row(
307
+ str(r["id"]),
308
+ _status(r.get("status_code")),
309
+ r.get("content_type") or "",
310
+ format_bytes(len(r.get("content") or "")),
311
+ format_relative_time(r.get("fetched_at")),
312
+ r.get("source_url") or "",
313
+ )
314
+ console.print(table)
315
+ console.print(f"\n[cyan]Total pages shown:[/cyan] {len(rows)}")
316
+
317
+
318
+ # ---------------------------------------------------------------------
319
+ # show
320
+ # ---------------------------------------------------------------------
321
+
322
+
323
+ @app.command(help="Show one document's body + metadata.")
324
+ async def show(
325
+ page_id: int = typer.Argument(..., help="CrawledPage id to render."),
326
+ raw: bool = typer.Option(
327
+ False,
328
+ "--raw",
329
+ help="Print body as plain text instead of rendered Markdown / "
330
+ "syntax-highlighted HTML.",
331
+ ),
332
+ head: int = typer.Option(
333
+ 0,
334
+ "--head",
335
+ "-h",
336
+ help="Print only the first N characters of the body. 0 = full.",
337
+ ),
338
+ ) -> None:
339
+ async with _client() as client:
340
+ resp = await client.get(f"{API_PREFIX}/pages/{page_id}")
341
+ if resp.status_code == 404:
342
+ console.print(f"[red]No document with id={page_id}[/red]")
343
+ raise typer.Exit(1)
344
+ resp.raise_for_status()
345
+ doc = resp.json()
346
+
347
+ status_code = doc.get("status_code") or 0
348
+
349
+ # Header table — id, url, status, content_type, fetched_at, metadata.
350
+ header = Table(
351
+ title=f"CrawledPage #{doc['id']}",
352
+ show_header=False,
353
+ box=None,
354
+ pad_edge=False,
355
+ title_style="bold magenta",
356
+ )
357
+ header.add_column("k", style="cyan", no_wrap=True)
358
+ header.add_column("v", overflow="fold")
359
+ header.add_row("source_url", f"[green]{doc.get('source_url') or ''}[/green]")
360
+ header.add_row("status", _status(status_code))
361
+ header.add_row("content_type", f"[yellow]{doc.get('content_type') or ''}[/yellow]")
362
+ header.add_row("source_kind", f"[dim]{doc.get('source_kind') or ''}[/dim]")
363
+ header.add_row(
364
+ "fetched_at", f"[dim]{format_relative_time(doc.get('fetched_at'))}[/dim]"
365
+ )
366
+ header.add_row("hash", f"[dim]{doc.get('content_hash') or ''}[/dim]")
367
+ if doc.get("doc_metadata"):
368
+ header.add_row("metadata", json.dumps(doc["doc_metadata"]))
369
+ console.print(header)
370
+ console.print()
371
+
372
+ body = doc.get("content") or ""
373
+ if not body:
374
+ console.print("[dim](no body)[/dim]")
375
+ return
376
+ if head and head > 0:
377
+ body = body[:head]
378
+
379
+ if raw:
380
+ console.print(body)
381
+ return
382
+
383
+ content_type = (doc.get("content_type") or "").lower()
384
+ if content_type == "markdown":
385
+ console.print(Panel(Markdown(body), title="content", border_style="dim"))
386
+ elif content_type in ("html", "json"):
387
+ console.print(
388
+ Panel(
389
+ Syntax(body, content_type, theme="ansi_dark", word_wrap=True),
390
+ title=f"content ({content_type})",
391
+ border_style="dim",
392
+ )
393
+ )
394
+ else:
395
+ console.print(Panel(body, title="content", border_style="dim"))
396
+
397
+
398
+ # ---------------------------------------------------------------------
399
+ # retry-failed
400
+ # ---------------------------------------------------------------------
401
+
402
+
403
+ @app.command("retry-failed", help="Re-fetch every document with status >= 400.")
404
+ async def retry_failed(
405
+ limit: int = typer.Option(50, "--limit", "-n", help="Max retries."),
406
+ ) -> None:
407
+ async with _client() as client:
408
+ resp = await client.get(
409
+ f"{API_PREFIX}/pages",
410
+ params={"limit": limit, "failed_only": "true"},
411
+ )
412
+ resp.raise_for_status()
413
+ failed_rows = resp.json()
414
+
415
+ if not failed_rows:
416
+ console.print("[green]✓ Nothing to retry.[/green]")
417
+ return
418
+
419
+ console.print(
420
+ f"[bold yellow]Retrying {len(failed_rows)} failed fetches…[/bold yellow]"
421
+ )
422
+ successes = 0
423
+ for row in failed_rows:
424
+ url = row.get("source_url")
425
+ if not url:
426
+ continue
427
+ payload = {
428
+ "url": url,
429
+ "doc_metadata": row.get("doc_metadata") or {},
430
+ "source_kind": row.get("source_kind") or "crawl4ai",
431
+ }
432
+ try:
433
+ r = await client.post(f"{API_PREFIX}/crawl", json=payload)
434
+ if r.status_code < 400:
435
+ new_doc = r.json()
436
+ new_status = new_doc.get("status_code") or 0
437
+ if new_status < 400:
438
+ successes += 1
439
+ console.print(f" [green]✓[/green] [dim]{url}[/dim]")
440
+ else:
441
+ console.print(
442
+ f" [red]✗[/red] [dim]{url}[/dim] "
443
+ f"[red](status {new_status})[/red]"
444
+ )
445
+ except httpx.HTTPError as exc:
446
+ console.print(
447
+ f" [red]✗[/red] [dim]{url}[/dim]: [red]{exc}[/red]",
448
+ markup=True,
449
+ )
450
+
451
+ fail_count = len(failed_rows) - successes
452
+ console.print(
453
+ f"\n[bold]Retried[/bold] [cyan]{len(failed_rows)}[/cyan] • "
454
+ f"[green]{successes} succeeded[/green] • "
455
+ f"[red]{fail_count} still failing[/red]"
456
+ )
457
+
458
+
459
+ if __name__ == "__main__":
460
+ sys.exit(app())
@@ -0,0 +1,8 @@
1
+ """Crawler service API package.
2
+
3
+ ``API_PREFIX`` is the one place the crawler's mount point is spelled.
4
+ The plugin spec mounts the router there, the dashboard builds its
5
+ endpoints from it and the CLI calls it, so none of them can drift.
6
+ """
7
+
8
+ API_PREFIX = "/api/v1/crawler"