PlaywrightCapture 1.40.0__tar.gz → 1.40.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {playwrightcapture-1.40.0 → playwrightcapture-1.40.2}/PKG-INFO +2 -10
- {playwrightcapture-1.40.0 → playwrightcapture-1.40.2}/README.md +0 -8
- {playwrightcapture-1.40.0 → playwrightcapture-1.40.2}/playwrightcapture/capture.py +32 -27
- {playwrightcapture-1.40.0 → playwrightcapture-1.40.2}/pyproject.toml +3 -3
- {playwrightcapture-1.40.0 → playwrightcapture-1.40.2}/LICENSE +0 -0
- {playwrightcapture-1.40.0 → playwrightcapture-1.40.2}/playwrightcapture/__init__.py +0 -0
- {playwrightcapture-1.40.0 → playwrightcapture-1.40.2}/playwrightcapture/exceptions.py +0 -0
- {playwrightcapture-1.40.0 → playwrightcapture-1.40.2}/playwrightcapture/helpers.py +0 -0
- {playwrightcapture-1.40.0 → playwrightcapture-1.40.2}/playwrightcapture/py.typed +0 -0
- {playwrightcapture-1.40.0 → playwrightcapture-1.40.2}/playwrightcapture/socks5dnslookup.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: PlaywrightCapture
|
|
3
|
-
Version: 1.40.
|
|
3
|
+
Version: 1.40.2
|
|
4
4
|
Summary: A simple library to capture websites using playwright
|
|
5
5
|
License-Expression: BSD-3-Clause
|
|
6
6
|
License-File: LICENSE
|
|
@@ -27,7 +27,7 @@ Requires-Dist: charset-normalizer (>=3.4.6,<4.0.0)
|
|
|
27
27
|
Requires-Dist: dnspython (>=2.7.0,<3.0.0)
|
|
28
28
|
Requires-Dist: lookyloo-models (>=0.2.9)
|
|
29
29
|
Requires-Dist: orjson (>=3.11.4,<4.0.0)
|
|
30
|
-
Requires-Dist: playwright (>=1.
|
|
30
|
+
Requires-Dist: playwright (>=1.61.0)
|
|
31
31
|
Requires-Dist: playwright-stealth (>=2.0.3)
|
|
32
32
|
Requires-Dist: pure-magic-rs (>=0.4.3)
|
|
33
33
|
Requires-Dist: pydub-ng (>=0.2.0) ; extra == "recaptcha"
|
|
@@ -50,14 +50,6 @@ Simple replacement for [splash](https://github.com/scrapinghub/splash) using [pl
|
|
|
50
50
|
pip install playwrightcapture
|
|
51
51
|
```
|
|
52
52
|
|
|
53
|
-
# Note for Ubuntu 26.04 pre-1.61.0
|
|
54
|
-
|
|
55
|
-
It is not supported, and `playwright install` fails. A quick and dirty fix is
|
|
56
|
-
|
|
57
|
-
```bash
|
|
58
|
-
PLAYWRIGHT_HOST_PLATFORM_OVERRIDE=ubuntu24.04-x64 playwright install
|
|
59
|
-
```
|
|
60
|
-
|
|
61
53
|
# Usage
|
|
62
54
|
|
|
63
55
|
A very basic example:
|
|
@@ -8,14 +8,6 @@ Simple replacement for [splash](https://github.com/scrapinghub/splash) using [pl
|
|
|
8
8
|
pip install playwrightcapture
|
|
9
9
|
```
|
|
10
10
|
|
|
11
|
-
# Note for Ubuntu 26.04 pre-1.61.0
|
|
12
|
-
|
|
13
|
-
It is not supported, and `playwright install` fails. A quick and dirty fix is
|
|
14
|
-
|
|
15
|
-
```bash
|
|
16
|
-
PLAYWRIGHT_HOST_PLATFORM_OVERRIDE=ubuntu24.04-x64 playwright install
|
|
17
|
-
```
|
|
18
|
-
|
|
19
11
|
# Usage
|
|
20
12
|
|
|
21
13
|
A very basic example:
|
|
@@ -1799,7 +1799,7 @@ class Capture():
|
|
|
1799
1799
|
self.logger.debug('Frame not loaded yet, cannot get content.')
|
|
1800
1800
|
else:
|
|
1801
1801
|
try:
|
|
1802
|
-
async with timeout(
|
|
1802
|
+
async with timeout(10):
|
|
1803
1803
|
return await page.content()
|
|
1804
1804
|
except (Error, TimeoutError, asyncio.TimeoutError):
|
|
1805
1805
|
self.logger.debug('Unable to get page content, trying again.')
|
|
@@ -1810,33 +1810,37 @@ class Capture():
|
|
|
1810
1810
|
break
|
|
1811
1811
|
tries -= 1
|
|
1812
1812
|
if tries > 0:
|
|
1813
|
-
|
|
1814
|
-
|
|
1815
|
-
|
|
1816
|
-
|
|
1817
|
-
|
|
1818
|
-
|
|
1819
|
-
|
|
1820
|
-
|
|
1821
|
-
|
|
1822
|
-
|
|
1823
|
-
|
|
1824
|
-
|
|
1825
|
-
|
|
1826
|
-
|
|
1827
|
-
|
|
1828
|
-
|
|
1829
|
-
|
|
1830
|
-
|
|
1831
|
-
|
|
1832
|
-
|
|
1833
|
-
|
|
1834
|
-
|
|
1813
|
+
try:
|
|
1814
|
+
await self._wait_for_random_timeout(page, 2)
|
|
1815
|
+
await self._safe_wait(page, 2)
|
|
1816
|
+
except Exception as e:
|
|
1817
|
+
self.logger.warning(f'The Playwright Page is in a broken state: {e}.')
|
|
1818
|
+
break
|
|
1819
|
+
|
|
1820
|
+
# got no content
|
|
1821
|
+
if page.url and page.url.strip() and page.url.strip().startswith('data'):
|
|
1822
|
+
self.logger.debug(f'Data URL in frame: {page.url}')
|
|
1823
|
+
# 2026-02-10: if the URL starts with data, we have a data URI, and possibly some content
|
|
1824
|
+
if parsed := self.__parse_data_uri(page.url):
|
|
1825
|
+
mime, mime_params, content = parsed
|
|
1826
|
+
charset = 'utf-8'
|
|
1827
|
+
if mime_params:
|
|
1828
|
+
# try to get charset
|
|
1829
|
+
if qs := parse_qs(mime_params):
|
|
1830
|
+
if charsets := qs.get('charset'):
|
|
1831
|
+
try:
|
|
1832
|
+
charset = codecs.lookup(charsets[0]).name
|
|
1833
|
+
except LookupError:
|
|
1834
|
+
charset = 'utf-8'
|
|
1835
|
+
if content:
|
|
1836
|
+
return unquote(content, encoding=charset)
|
|
1835
1837
|
else:
|
|
1836
|
-
self.logger.warning('
|
|
1837
|
-
|
|
1838
|
-
|
|
1839
|
-
|
|
1838
|
+
self.logger.warning('No content: {page.url}')
|
|
1839
|
+
else:
|
|
1840
|
+
self.logger.warning('Attempted to get data URL content and failed: {page.url}')
|
|
1841
|
+
elif page.name and page.name.strip():
|
|
1842
|
+
# 2026-02-16: some frame names are full HTML blobs, with extra things. Better than nothing.
|
|
1843
|
+
return unquote(page.name)
|
|
1840
1844
|
return None
|
|
1841
1845
|
|
|
1842
1846
|
def _get_links_from_rendered_page(self, rendered_url: str, rendered_html: str, rendered_hostname_only: bool) -> list[str]:
|
|
@@ -2008,6 +2012,7 @@ class Capture():
|
|
|
2008
2012
|
'Host unreachable through SOCKSv5 server.',
|
|
2009
2013
|
'Operation was cancelled',
|
|
2010
2014
|
'The URL can’t be shown',
|
|
2015
|
+
'Frame was detached',
|
|
2011
2016
|
# JS stuff
|
|
2012
2017
|
'TurnstileError: [Cloudflare Turnstile] Error: 300030.',
|
|
2013
2018
|
# The browser barfed
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "PlaywrightCapture"
|
|
3
|
-
version = "1.40.
|
|
3
|
+
version = "1.40.2"
|
|
4
4
|
description = "A simple library to capture websites using playwright"
|
|
5
5
|
authors = [
|
|
6
6
|
{name="Raphaël Vinot", email= "raphael.vinot@circl.lu"}
|
|
@@ -12,7 +12,7 @@ requires-python = ">=3.10,<3.15"
|
|
|
12
12
|
dynamic = [ "classifiers" ]
|
|
13
13
|
|
|
14
14
|
dependencies = [
|
|
15
|
-
"playwright (>=1.
|
|
15
|
+
"playwright (>=1.61.0)",
|
|
16
16
|
"beautifulsoup4[charset-normalizer,lxml] (>=4.15.0)",
|
|
17
17
|
"w3lib (>=2.4.1)",
|
|
18
18
|
"playwright-stealth (>=2.0.3)",
|
|
@@ -51,7 +51,7 @@ recaptcha = [
|
|
|
51
51
|
|
|
52
52
|
[tool.poetry.group.dev.dependencies]
|
|
53
53
|
types-beautifulsoup4 = "^4.12.0.20250516"
|
|
54
|
-
pytest = "^9.1.
|
|
54
|
+
pytest = "^9.1.1"
|
|
55
55
|
mypy = "^2.1.0"
|
|
56
56
|
types-dateparser = "^1.4.1.20260617"
|
|
57
57
|
types-pytz = "^2026.2.0.20260518"
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|