PlaywrightCapture 1.40.0__tar.gz → 1.40.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: PlaywrightCapture
3
- Version: 1.40.0
3
+ Version: 1.40.2
4
4
  Summary: A simple library to capture websites using playwright
5
5
  License-Expression: BSD-3-Clause
6
6
  License-File: LICENSE
@@ -27,7 +27,7 @@ Requires-Dist: charset-normalizer (>=3.4.6,<4.0.0)
27
27
  Requires-Dist: dnspython (>=2.7.0,<3.0.0)
28
28
  Requires-Dist: lookyloo-models (>=0.2.9)
29
29
  Requires-Dist: orjson (>=3.11.4,<4.0.0)
30
- Requires-Dist: playwright (>=1.60.0)
30
+ Requires-Dist: playwright (>=1.61.0)
31
31
  Requires-Dist: playwright-stealth (>=2.0.3)
32
32
  Requires-Dist: pure-magic-rs (>=0.4.3)
33
33
  Requires-Dist: pydub-ng (>=0.2.0) ; extra == "recaptcha"
@@ -50,14 +50,6 @@ Simple replacement for [splash](https://github.com/scrapinghub/splash) using [pl
50
50
  pip install playwrightcapture
51
51
  ```
52
52
 
53
- # Note for Ubuntu 26.04 pre-1.61.0
54
-
55
- It is not supported, and `playwright install` fails. A quick and dirty fix is
56
-
57
- ```bash
58
- PLAYWRIGHT_HOST_PLATFORM_OVERRIDE=ubuntu24.04-x64 playwright install
59
- ```
60
-
61
53
  # Usage
62
54
 
63
55
  A very basic example:
@@ -8,14 +8,6 @@ Simple replacement for [splash](https://github.com/scrapinghub/splash) using [pl
8
8
  pip install playwrightcapture
9
9
  ```
10
10
 
11
- # Note for Ubuntu 26.04 pre-1.61.0
12
-
13
- It is not supported, and `playwright install` fails. A quick and dirty fix is
14
-
15
- ```bash
16
- PLAYWRIGHT_HOST_PLATFORM_OVERRIDE=ubuntu24.04-x64 playwright install
17
- ```
18
-
19
11
  # Usage
20
12
 
21
13
  A very basic example:
@@ -1799,7 +1799,7 @@ class Capture():
1799
1799
  self.logger.debug('Frame not loaded yet, cannot get content.')
1800
1800
  else:
1801
1801
  try:
1802
- async with timeout(5):
1802
+ async with timeout(10):
1803
1803
  return await page.content()
1804
1804
  except (Error, TimeoutError, asyncio.TimeoutError):
1805
1805
  self.logger.debug('Unable to get page content, trying again.')
@@ -1810,33 +1810,37 @@ class Capture():
1810
1810
  break
1811
1811
  tries -= 1
1812
1812
  if tries > 0:
1813
- await self._wait_for_random_timeout(page, 2)
1814
- await self._safe_wait(page, 2)
1815
- else:
1816
- # got no content
1817
- if page.url and page.url.strip() and page.url.strip().startswith('data'):
1818
- self.logger.debug(f'Data URL in frame: {page.url}')
1819
- # 2026-02-10: if the URL starts with data, we have a data URI, and possibly some content
1820
- if parsed := self.__parse_data_uri(page.url):
1821
- mime, mime_params, content = parsed
1822
- charset = 'utf-8'
1823
- if mime_params:
1824
- # try to get charset
1825
- if qs := parse_qs(mime_params):
1826
- if charsets := qs.get('charset'):
1827
- try:
1828
- charset = codecs.lookup(charsets[0]).name
1829
- except LookupError:
1830
- charset = 'utf-8'
1831
- if content:
1832
- return unquote(content, encoding=charset)
1833
- else:
1834
- self.logger.warning('No content: {page.url}')
1813
+ try:
1814
+ await self._wait_for_random_timeout(page, 2)
1815
+ await self._safe_wait(page, 2)
1816
+ except Exception as e:
1817
+ self.logger.warning(f'The Playwright Page is in a broken state: {e}.')
1818
+ break
1819
+
1820
+ # got no content
1821
+ if page.url and page.url.strip() and page.url.strip().startswith('data'):
1822
+ self.logger.debug(f'Data URL in frame: {page.url}')
1823
+ # 2026-02-10: if the URL starts with data, we have a data URI, and possibly some content
1824
+ if parsed := self.__parse_data_uri(page.url):
1825
+ mime, mime_params, content = parsed
1826
+ charset = 'utf-8'
1827
+ if mime_params:
1828
+ # try to get charset
1829
+ if qs := parse_qs(mime_params):
1830
+ if charsets := qs.get('charset'):
1831
+ try:
1832
+ charset = codecs.lookup(charsets[0]).name
1833
+ except LookupError:
1834
+ charset = 'utf-8'
1835
+ if content:
1836
+ return unquote(content, encoding=charset)
1835
1837
  else:
1836
- self.logger.warning('Attempted to get data URL content and failed: {page.url}')
1837
- elif page.name and page.name.strip():
1838
- # 2026-02-16: some frame names are full HTML blobs, with extra things. Better than nothing.
1839
- return unquote(page.name)
1838
+ self.logger.warning('No content: {page.url}')
1839
+ else:
1840
+ self.logger.warning('Attempted to get data URL content and failed: {page.url}')
1841
+ elif page.name and page.name.strip():
1842
+ # 2026-02-16: some frame names are full HTML blobs, with extra things. Better than nothing.
1843
+ return unquote(page.name)
1840
1844
  return None
1841
1845
 
1842
1846
  def _get_links_from_rendered_page(self, rendered_url: str, rendered_html: str, rendered_hostname_only: bool) -> list[str]:
@@ -2008,6 +2012,7 @@ class Capture():
2008
2012
  'Host unreachable through SOCKSv5 server.',
2009
2013
  'Operation was cancelled',
2010
2014
  'The URL can’t be shown',
2015
+ 'Frame was detached',
2011
2016
  # JS stuff
2012
2017
  'TurnstileError: [Cloudflare Turnstile] Error: 300030.',
2013
2018
  # The browser barfed
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "PlaywrightCapture"
3
- version = "1.40.0"
3
+ version = "1.40.2"
4
4
  description = "A simple library to capture websites using playwright"
5
5
  authors = [
6
6
  {name="Raphaël Vinot", email= "raphael.vinot@circl.lu"}
@@ -12,7 +12,7 @@ requires-python = ">=3.10,<3.15"
12
12
  dynamic = [ "classifiers" ]
13
13
 
14
14
  dependencies = [
15
- "playwright (>=1.60.0)",
15
+ "playwright (>=1.61.0)",
16
16
  "beautifulsoup4[charset-normalizer,lxml] (>=4.15.0)",
17
17
  "w3lib (>=2.4.1)",
18
18
  "playwright-stealth (>=2.0.3)",
@@ -51,7 +51,7 @@ recaptcha = [
51
51
 
52
52
  [tool.poetry.group.dev.dependencies]
53
53
  types-beautifulsoup4 = "^4.12.0.20250516"
54
- pytest = "^9.1.0"
54
+ pytest = "^9.1.1"
55
55
  mypy = "^2.1.0"
56
56
  types-dateparser = "^1.4.1.20260617"
57
57
  types-pytz = "^2026.2.0.20260518"