PlaywrightCapture 1.41.4__tar.gz → 1.41.6__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: PlaywrightCapture
3
- Version: 1.41.4
3
+ Version: 1.41.6
4
4
  Summary: A simple library to capture websites using playwright
5
5
  License-Expression: BSD-3-Clause
6
6
  License-File: LICENSE
@@ -25,7 +25,7 @@ Requires-Dist: async-timeout (>=5.0.1) ; python_version < "3.11"
25
25
  Requires-Dist: beautifulsoup4[charset-normalizer,lxml] (>=4.15.0)
26
26
  Requires-Dist: charset-normalizer (>=3.4.6,<4.0.0)
27
27
  Requires-Dist: dnspython (>=2.7.0,<3.0.0)
28
- Requires-Dist: lookyloo-models (>=0.4.1)
28
+ Requires-Dist: lookyloo-models (>=0.4.3)
29
29
  Requires-Dist: orjson (>=3.12,<4.0.0)
30
30
  Requires-Dist: playwright (>=1.63.0)
31
31
  Requires-Dist: playwright-stealth (>=2.0.3)
@@ -56,10 +56,13 @@ A very basic example:
56
56
 
57
57
  ```python
58
58
  from playwrightcapture import Capture
59
+ from lookyloo_models import CaptureSettings
59
60
 
60
- async with Capture() as capture:
61
+ capture_settings = CaptureSettings(url='google.com')
62
+
63
+ async with Capture(capture_settings=capture_settings) as capture:
61
64
  await capture.initialize_context()
62
- entries = await capture.capture_page(url, max_depth_capture_time=90)
65
+ entries = await capture.capture_page(max_depth_capture_time=90)
63
66
  ```
64
67
 
65
68
  Entries is a dictionaries that contains (if all goes well) the HAR, the screenshot, all the cookies of the session, the URL as it is in the browser at the end of the capture, and the full HTML page as rendered.
@@ -14,10 +14,13 @@ A very basic example:
14
14
 
15
15
  ```python
16
16
  from playwrightcapture import Capture
17
+ from lookyloo_models import CaptureSettings
17
18
 
18
- async with Capture() as capture:
19
+ capture_settings = CaptureSettings(url='google.com')
20
+
21
+ async with Capture(capture_settings=capture_settings) as capture:
19
22
  await capture.initialize_context()
20
- entries = await capture.capture_page(url, max_depth_capture_time=90)
23
+ entries = await capture.capture_page(max_depth_capture_time=90)
21
24
  ```
22
25
 
23
26
  Entries is a dictionaries that contains (if all goes well) the HAR, the screenshot, all the cookies of the session, the URL as it is in the browser at the end of the capture, and the full HTML page as rendered.
@@ -18,7 +18,7 @@ from base64 import b64decode, b64encode
18
18
  from io import BytesIO
19
19
  from logging import LoggerAdapter, Logger
20
20
  from tempfile import NamedTemporaryFile
21
- from typing import Any, Literal, TYPE_CHECKING, overload
21
+ from typing import Any, Literal, TYPE_CHECKING
22
22
  from collections.abc import Awaitable, Callable, MutableMapping
23
23
  from urllib.parse import urlparse, unquote, urljoin, urlsplit, urlunsplit, parse_qs, unquote_plus
24
24
  from zipfile import ZipFile
@@ -201,6 +201,24 @@ class Capture():
201
201
  self._color_scheme: Literal['dark', 'light', 'no-preference', 'null'] | None = capture_settings.color_scheme if capture_settings.color_scheme else None
202
202
  self._java_script_enabled: bool = capture_settings.java_script_enabled
203
203
  self.capture_timeout = capture_settings.general_timeout_in_sec
204
+ self.allow_tracking = capture_settings.allow_tracking
205
+ self.rendered_hostname_only = capture_settings.rendered_hostname_only
206
+ self.with_screenshot = capture_settings.with_screenshot
207
+ self.with_favicon = capture_settings.with_favicon
208
+ self.with_trusted_timestamps = capture_settings.with_trusted_timestamps
209
+ self.capture_depth = capture_settings.depth
210
+ self.final_wait = capture_settings.final_wait
211
+
212
+ self.initial_url: str
213
+ if capture_settings.url:
214
+ # This url could be None in transit when the thing to capture is a file (so the models allows it)
215
+ # But at this stage, the value must have been set to the local path of the file,
216
+ # if it is none, the capture will for sure fail.
217
+ self.initial_url = capture_settings.url
218
+ else:
219
+ raise InvalidPlaywrightParameter('No URL provided, cannot capture.')
220
+
221
+ self.initial_referer = capture_settings.referer
204
222
 
205
223
  self.should_retry: bool = False
206
224
  self.__network_not_idle: int = 2 # makes sure we do not wait for network idle the max amount of time the capture is allowed to take
@@ -285,6 +303,9 @@ class Capture():
285
303
  # Create the temporary file to store the HAR content.
286
304
  self._temp_harfile = NamedTemporaryFile(delete=False, prefix="playwright_capture_har", suffix=".json")
287
305
 
306
+ # all the errors gathered during the capture
307
+ self.errors: list[str] = []
308
+
288
309
  return self
289
310
 
290
311
  async def __aexit__(self, exc_type: Any, exc_value: Any, traceback: Any) -> bool:
@@ -315,7 +336,7 @@ class Capture():
315
336
  return False
316
337
  return True
317
338
 
318
- async def setup_page_capture(self, *, allow_tracking: bool=False) -> Page:
339
+ async def setup_page_capture(self) -> Page:
319
340
  """Prepare a page for a single-page capture without changing capture semantics.
320
341
 
321
342
  This method preserves the existing per-page setup used by capture_page:
@@ -421,7 +442,7 @@ class Capture():
421
442
  except Error as e:
422
443
  self.logger.warning(f'Failed at fetching PDF in headless chromium: {e}')
423
444
 
424
- if allow_tracking:
445
+ if self.allow_tracking:
425
446
  # Add authorization clickthroughs
426
447
  await self.__dialog_didomi_clickthrough(page)
427
448
  await self.__dialog_onetrust_clickthrough(page)
@@ -884,7 +905,7 @@ class Capture():
884
905
  except Exception as e:
885
906
  self.logger.info(f'Error while moving time forward: {e}')
886
907
 
887
- async def __instrumentation(self, page: Page, url: str, allow_tracking: bool, final_wait: int) -> None:
908
+ async def __instrumentation(self, page: Page, url: str) -> None:
888
909
  try:
889
910
  # NOTE: the clock must be installed after the page is loaded, otherwise it sometimes cause the complete capture to hang.
890
911
  await page.clock.install()
@@ -935,7 +956,7 @@ class Capture():
935
956
  await self._wait_for_random_timeout(page, 5)
936
957
  self.logger.debug('Keep going after moving mouse.')
937
958
 
938
- if allow_tracking:
959
+ if self.allow_tracking:
939
960
  await self._wait_for_random_timeout(page, 5)
940
961
  # This event is required trigger the add_locator_handler
941
962
  try:
@@ -1014,12 +1035,12 @@ class Capture():
1014
1035
 
1015
1036
  self.logger.debug('Done with instrumentation.')
1016
1037
  # Wait at least 5 sec after instrumentation
1017
- self.logger.debug(f'Waiting another {max(final_wait, 5)}s.')
1018
- await self._wait_for_random_timeout(page, max(final_wait, 5))
1038
+ self.logger.debug(f'Waiting another {max(self.final_wait, 5)}s.')
1039
+ await self._wait_for_random_timeout(page, max(self.final_wait, 5))
1019
1040
  await self._safe_wait(page)
1020
1041
  self.logger.debug('Done with waiting.')
1021
1042
 
1022
- async def _safe_get_storage_state(self, errors: list[str]) -> dict[str, Any]:
1043
+ async def _safe_get_storage_state(self) -> dict[str, Any]:
1023
1044
  # Collect storage state, including IndexedDB, to capture the full browser state.
1024
1045
  # 2026-09-08: add WebAuth credentials
1025
1046
  # 2026-09-17: Add opfs
@@ -1031,31 +1052,31 @@ class Capture():
1031
1052
  return await self.context.storage_state(**to_store) # type: ignore[return-value,arg-type]
1032
1053
  except (TimeoutError, asyncio.TimeoutError):
1033
1054
  self.logger.warning("Unable to get storage (timeout).")
1034
- errors.append("Unable to get the storage (timeout).")
1055
+ self.errors.append("Unable to get the storage (timeout).")
1035
1056
  self.should_retry = True
1036
1057
  break
1037
1058
  except Error as e:
1038
1059
  if to_store['indexed_db'] and 'IndexedDB' in str(e):
1039
1060
  to_store['indexed_db'] = False
1040
- errors.append('Unable to get the IndexedDB')
1061
+ self.errors.append('Unable to get the IndexedDB')
1041
1062
  self.logger.warning(f"Unable to get the IndexedDB: {e}")
1042
1063
  continue
1043
1064
  if to_store['opfs'] and 'OPFS' in str(e):
1044
1065
  to_store['opfs'] = False
1045
- errors.append('Unable to get the OPFS')
1066
+ self.errors.append('Unable to get the OPFS')
1046
1067
  self.logger.warning(f"Unable to get the OPFS: {e}")
1047
1068
  continue
1048
1069
 
1049
1070
  if not to_store['indexed_db'] and not to_store['opfs']:
1050
1071
  # we disabled both options, quit
1051
1072
  self.should_retry = True
1052
- errors.append(f'Unable to get the storage at all: {e}')
1073
+ self.errors.append(f'Unable to get the storage at all: {e}')
1053
1074
  self.logger.warning(f"Unable to get the storage at all: {e}")
1054
1075
  break
1055
1076
  except Exception as e:
1056
1077
  # When the driver explodes for no clear reason.
1057
1078
  self.logger.warning(f"[Generic Exception] Unable to get the storage: {e}")
1058
- errors.append(f'[Generic Exception] Unable to get the storage: {e}')
1079
+ self.errors.append(f'[Generic Exception] Unable to get the storage: {e}')
1059
1080
  self.should_retry = True
1060
1081
  break
1061
1082
  return {}
@@ -1065,8 +1086,6 @@ class Capture():
1065
1086
  *,
1066
1087
  page: Page,
1067
1088
  to_return: CaptureResponse,
1068
- errors: list[str],
1069
- with_trusted_timestamps: bool,
1070
1089
  ) -> None:
1071
1090
  """Common finalization logic for captures (downloads, cookies, storage, HAR, socks5, timestamps)."""
1072
1091
 
@@ -1097,19 +1116,19 @@ class Capture():
1097
1116
  to_return['cookies'] = [Cookie.model_validate(c).model_dump(exclude_none=True) for c in await self.context.cookies()]
1098
1117
  except (TimeoutError, asyncio.TimeoutError):
1099
1118
  self.logger.warning("Unable to get cookies (timeout).")
1100
- errors.append("Unable to get the cookies (timeout).")
1119
+ self.errors.append("Unable to get the cookies (timeout).")
1101
1120
  self.should_retry = True
1102
1121
  except Error as e:
1103
1122
  self.logger.warning(f"Unable to get cookies: {e}")
1104
- errors.append(f'Unable to get the cookies: {e}')
1123
+ self.errors.append(f'Unable to get the cookies: {e}')
1105
1124
  self.should_retry = True
1106
1125
  except Exception as e:
1107
1126
  # When the driver explodes for no clear reason.
1108
1127
  self.logger.warning(f"[Generic Exception] Unable to get cookies: {e}")
1109
- errors.append(f'[Generic Exception] Unable to get the cookies: {e}')
1128
+ self.errors.append(f'[Generic Exception] Unable to get the cookies: {e}')
1110
1129
  self.should_retry = True
1111
1130
 
1112
- to_return['storage'] = await self._safe_get_storage_state(errors)
1131
+ to_return['storage'] = await self._safe_get_storage_state()
1113
1132
 
1114
1133
  try:
1115
1134
  if page.is_closed():
@@ -1167,14 +1186,14 @@ class Capture():
1167
1186
  await self.socks5_resolver(har)
1168
1187
  except (TimeoutError, asyncio.TimeoutError):
1169
1188
  self.logger.warning("Unable to resolve all the IPs via the socks5 proxy.")
1170
- errors.append("Unable to resolve all the IPs via the socks5 proxy.")
1189
+ self.errors.append("Unable to resolve all the IPs via the socks5 proxy.")
1171
1190
  self.should_retry = True
1172
1191
 
1173
1192
  except (TimeoutError, asyncio.TimeoutError):
1174
1193
  # If closing the context or generating the HAR takes too long, the
1175
1194
  # capture is considered incomplete but we still return what we have.
1176
1195
  self.logger.warning("[Timeout] Unable to close context at the end of the capture.")
1177
- errors.append("[Timeout] Unable to close context at the end of the capture.")
1196
+ self.errors.append("[Timeout] Unable to close context at the end of the capture.")
1178
1197
  self.should_retry = True
1179
1198
  # In case of timeout, let the exception reach the async calls
1180
1199
  await asyncio.sleep(1)
@@ -1182,11 +1201,11 @@ class Capture():
1182
1201
  # Any other unexpected failure while finalizing the capture is logged
1183
1202
  # and surfaced as a generic HAR-generation error.
1184
1203
  self.logger.warning(f"Other exception while finishing up the capture: {e}.")
1185
- errors.append(f'Unable to generate HAR file: {e}')
1204
+ self.errors.append(f'Unable to generate HAR file: {e}')
1186
1205
 
1187
- if errors:
1188
- to_return['error'] = '\n'.join(errors)
1189
- if with_trusted_timestamps:
1206
+ if self.errors:
1207
+ to_return['error'] = '\n'.join(self.errors)
1208
+ if self.with_trusted_timestamps:
1190
1209
  try:
1191
1210
  await self._get_trusted_timestamps(to_return)
1192
1211
  except Exception as e:
@@ -1237,7 +1256,11 @@ class Capture():
1237
1256
  self.logger.warning(f"Opening a weird URL: {url}")
1238
1257
  return False, f"Attempted to open a weird URL '{url}', blocked."
1239
1258
 
1240
- async def open_page(self, page: Page, url: str, errors: list[str], referer: str | None=None) -> None:
1259
+ async def open_page(self, page: Page, url: str | None = None, referer: str | None=None) -> None:
1260
+ """This method opens the page but does nothing with it. Use it only if you need a custom instrumentation.
1261
+ The usecase in lookyloo's context is to have a headfull capture in Xpra.
1262
+ Prefer using `capture_page` instead.
1263
+ """
1241
1264
 
1242
1265
  async def catch_file_route(route: Route, request: Request) -> None:
1243
1266
  if unquote(request.url) == url:
@@ -1280,9 +1303,10 @@ class Capture():
1280
1303
  ]
1281
1304
  for scheme in allowed_schemes:
1282
1305
  await page.route(scheme, lambda route: route.continue_())
1283
-
1306
+ if not url:
1307
+ url = self.initial_url
1284
1308
  try:
1285
- await page.goto(url, wait_until='domcontentloaded', referer=referer if referer else '')
1309
+ await page.goto(url, wait_until='domcontentloaded', referer=referer if referer else self.initial_referer)
1286
1310
  try:
1287
1311
  await page.bring_to_front()
1288
1312
  self.logger.debug('Page moved to front.')
@@ -1298,7 +1322,7 @@ class Capture():
1298
1322
  try:
1299
1323
  async with page.expect_download() as download_info:
1300
1324
  try:
1301
- await page.goto(url, referer=referer if referer else '')
1325
+ await page.goto(url, referer=referer if referer else self.initial_referer)
1302
1326
  except Exception:
1303
1327
  pass
1304
1328
  with NamedTemporaryFile() as tmp_f:
@@ -1316,7 +1340,7 @@ class Capture():
1316
1340
  error_msg = download.failure()
1317
1341
  if not error_msg:
1318
1342
  raise e
1319
- errors.append(f"Error while downloading: {error_msg}")
1343
+ self.errors.append(f"Error while downloading: {error_msg}")
1320
1344
  self.logger.info(f'Error while downloading: {error_msg}')
1321
1345
  self.should_retry = True
1322
1346
  except Exception:
@@ -1326,44 +1350,10 @@ class Capture():
1326
1350
  else:
1327
1351
  await self._wait_for_random_timeout(page, 5) # Wait 5 sec after document loaded
1328
1352
 
1329
- @overload
1330
- async def capture_page(self, url: str, *, max_depth_capture_time: int,
1331
- referer: str | None=None,
1332
- page: Page | None=None, depth: int=0,
1333
- rendered_hostname_only: bool=True,
1334
- with_screenshot: bool=True,
1335
- with_favicon: bool=False,
1336
- allow_tracking: bool=False,
1337
- with_trusted_timestamps: bool=False,
1338
- current_page_only: bool=False,
1339
- final_wait: int=5
1340
- ) -> CaptureResponse:
1341
- ...
1342
-
1343
- @overload
1344
- async def capture_page(self, url: None=None, *, max_depth_capture_time: int,
1345
- referer: str | None=None,
1346
- page: Page, depth: int=0,
1347
- rendered_hostname_only: bool=True,
1348
- with_screenshot: bool=True,
1349
- with_favicon: bool=False,
1350
- allow_tracking: bool=False,
1351
- with_trusted_timestamps: bool=False,
1352
- current_page_only: bool=False,
1353
- final_wait: int=5
1354
- ) -> CaptureResponse:
1355
- ...
1356
-
1357
1353
  async def capture_page(self, url: str | None=None, *, max_depth_capture_time: int,
1358
1354
  referer: str | None=None,
1359
- page: Page | None=None, depth: int=0,
1360
- rendered_hostname_only: bool=True,
1361
- with_screenshot: bool=True,
1362
- with_favicon: bool=False,
1363
- allow_tracking: bool=False,
1364
- with_trusted_timestamps: bool=False,
1355
+ page: Page | None=None, depth: int | None = None,
1365
1356
  current_page_only: bool=False,
1366
- final_wait: int=5,
1367
1357
  ) -> CaptureResponse:
1368
1358
  """Capture a URL and optionally recurse into child links.
1369
1359
 
@@ -1376,10 +1366,11 @@ class Capture():
1376
1366
  (no navigation, no recursion) and then finalizes. This is the path
1377
1367
  used by remote headfull captures after setup_page_capture has already been
1378
1368
  called by the caller.
1369
+
1370
+ The `url` field is only needed when the URL to capture isn't the initial one (depth>0)
1379
1371
  """
1380
1372
 
1381
1373
  to_return: CaptureResponse = {}
1382
- errors: list[str] = []
1383
1374
  capturing_sub = False
1384
1375
 
1385
1376
  if current_page_only:
@@ -1388,7 +1379,7 @@ class Capture():
1388
1379
  raise InvalidPlaywrightParameter('current_page_only requires an initialized page')
1389
1380
  else:
1390
1381
  if page is None:
1391
- page = await self.setup_page_capture(allow_tracking=allow_tracking)
1382
+ page = await self.setup_page_capture()
1392
1383
  else:
1393
1384
  # Automated capture with depth > 0
1394
1385
  capturing_sub = True
@@ -1397,12 +1388,13 @@ class Capture():
1397
1388
  if not current_page_only:
1398
1389
  # Standard navigation + capture path.
1399
1390
  if not url:
1400
- raise InvalidPlaywrightParameter('The URL to capture is missing.')
1401
- await self.open_page(page, url, errors, referer)
1391
+ url = self.initial_url
1392
+ await self.open_page(page, url=url if url else self.initial_url,
1393
+ referer=referer if referer else self.initial_referer)
1402
1394
 
1403
1395
  try:
1404
1396
  if self.headless:
1405
- await self.__instrumentation(page, url, allow_tracking, final_wait)
1397
+ await self.__instrumentation(page, url)
1406
1398
  else:
1407
1399
  self.logger.debug('Headed mode, skipping instrumentation.')
1408
1400
  await self._wait_for_random_timeout(page, self._capture_timeout - 5)
@@ -1436,7 +1428,7 @@ class Capture():
1436
1428
  u = '/!\\ Unknown /!\\'
1437
1429
  to_return['last_redirected_url'] = u
1438
1430
 
1439
- if 'html' in to_return and to_return['html'] is not None and with_favicon:
1431
+ if 'html' in to_return and to_return['html'] is not None and self.with_favicon:
1440
1432
  # We're probably (?) safe only looking for favicons in the main frame.
1441
1433
  # TODO: check that?
1442
1434
  try:
@@ -1447,7 +1439,7 @@ class Capture():
1447
1439
  except Exception as e:
1448
1440
  self.logger.warning(f'Unable to get favicons: {e}')
1449
1441
 
1450
- if with_screenshot:
1442
+ if self.with_screenshot:
1451
1443
  to_return['png'] = await self._failsafe_get_screenshot(page)
1452
1444
 
1453
1445
  # Keep that all the way down there in case the capture failed.
@@ -1456,11 +1448,15 @@ class Capture():
1456
1448
  else:
1457
1449
  self._already_captured.add(page.url)
1458
1450
 
1459
- if depth > 0 and to_return.get('html') and to_return['html']:
1451
+ if depth is None:
1452
+ # fallback for the first call
1453
+ depth = self.capture_depth
1454
+
1455
+ if depth is not None and depth > 0 and to_return.get('html') and to_return['html']:
1460
1456
  # TODO with children frames:
1461
1457
  # 1. if the frame has a URL, use that as base URL/referer for the subsequent captures
1462
1458
  # 2. if it doesn't, the base URL is the url of the parent (which may or may not be the main frame)
1463
- if child_urls := self._get_links_from_rendered_page(page.url, to_return['html'], rendered_hostname_only):
1459
+ if child_urls := self._get_links_from_rendered_page(page.url, to_return['html']):
1464
1460
  to_return['children'] = []
1465
1461
  depth -= 1
1466
1462
  total_urls = len(child_urls)
@@ -1486,11 +1482,8 @@ class Capture():
1486
1482
  child_capture = await self.capture_page(
1487
1483
  url=url, referer=page.url,
1488
1484
  page=page, depth=depth,
1489
- rendered_hostname_only=rendered_hostname_only,
1490
- max_depth_capture_time=max_capture_time,
1491
- with_screenshot=with_screenshot,
1492
- final_wait=final_wait)
1493
- if with_trusted_timestamps:
1485
+ max_depth_capture_time=max_capture_time)
1486
+ if self.with_trusted_timestamps:
1494
1487
  try:
1495
1488
  await self._get_trusted_timestamps(child_capture)
1496
1489
  except Exception as e:
@@ -1510,7 +1503,7 @@ class Capture():
1510
1503
  if consecutive_errors >= 5:
1511
1504
  # if we have more than 5 consecutive errors, the capture is most probably broken, breaking.
1512
1505
  self.logger.warning('Got more than 5 consecutive errors while capturing children, breaking.')
1513
- errors.append("Got more than 5 consecutive errors while capturing children")
1506
+ self.errors.append("Got more than 5 consecutive errors while capturing children")
1514
1507
  self.should_retry = True
1515
1508
  break
1516
1509
 
@@ -1522,19 +1515,19 @@ class Capture():
1522
1515
  self.logger.info(f'Unable to go back: {e}.')
1523
1516
 
1524
1517
  except PlaywrightTimeoutError as e:
1525
- errors.append(f"The capture took too long - {e.message}")
1518
+ self.errors.append(f"The capture took too long - {e.message}")
1526
1519
  self.should_retry = True
1527
1520
  except (asyncio.TimeoutError, TimeoutError):
1528
- errors.append("Something in the capture took too long")
1521
+ self.errors.append("Something in the capture took too long")
1529
1522
  self.should_retry = True
1530
1523
  except TargetClosedError as e:
1531
- errors.append(f"The target was closed - {e}")
1524
+ self.errors.append(f"The target was closed - {e}")
1532
1525
  self.should_retry = True
1533
1526
  except Error as e:
1534
1527
  # NOTE: there are a lot of errors that look like duplicates and they are triggered at different times in the process.
1535
1528
  # it is tricky to figure our which one should (and should not) trigger a retry. Below is our best guess and it will change over time.
1536
1529
  self._update_exceptions(e)
1537
- errors.append(e.message)
1530
+ self.errors.append(e.message)
1538
1531
  to_return['error_name'] = e.name
1539
1532
  # NOTE: e.name is generally (always?) "Error"
1540
1533
  if self._fatal_network_error(e) or self._fatal_auth_error(e) or self.fatal_browser_error(e):
@@ -1542,7 +1535,7 @@ class Capture():
1542
1535
  elif self._retry_network_error(e) or self._retry_browser_error(e):
1543
1536
  # this one sounds like something we can retry...
1544
1537
  self.logger.info(f'Issue with {url} (retrying): {e.message}')
1545
- errors.append(f'Issue with {url}: {e.message}')
1538
+ self.errors.append(f'Issue with {url}: {e.message}')
1546
1539
  self.should_retry = True
1547
1540
  else:
1548
1541
  # Unexpected ones
@@ -1550,15 +1543,15 @@ class Capture():
1550
1543
  except PlaywrightCaptureException as e:
1551
1544
  # unrecoverable exeptions
1552
1545
  self.logger.warning(f'Unable to run capture: {e}')
1553
- errors.append(f'Unable to run capture: {e}')
1546
+ self.errors.append(f'Unable to run capture: {e}')
1554
1547
  raise e
1555
1548
  except Exception as e:
1556
1549
  # we may get a non-playwright exception to.
1557
1550
  # The ones we try to handle here should be treated as if they were.
1558
- errors.append(str(e))
1551
+ self.errors.append(str(e))
1559
1552
  if str(e) in ['Connection closed while reading from the driver']:
1560
1553
  self.logger.info(f'Issue with {url} (retrying): {e}')
1561
- errors.append(f'Issue with {url}: {e}')
1554
+ self.errors.append(f'Issue with {url}: {e}')
1562
1555
  self.should_retry = True
1563
1556
  else:
1564
1557
  raise e
@@ -1569,8 +1562,6 @@ class Capture():
1569
1562
  await self._finalize_capture(
1570
1563
  page=page,
1571
1564
  to_return=to_return,
1572
- errors=errors,
1573
- with_trusted_timestamps=with_trusted_timestamps,
1574
1565
  )
1575
1566
  self.logger.debug('Capture done')
1576
1567
  return to_return
@@ -1756,7 +1747,7 @@ class Capture():
1756
1747
  return unquote(page.name)
1757
1748
  return None
1758
1749
 
1759
- def _get_links_from_rendered_page(self, rendered_url: str, rendered_html: str, rendered_hostname_only: bool) -> list[str]:
1750
+ def _get_links_from_rendered_page(self, rendered_url: str, rendered_html: str) -> list[str]:
1760
1751
  def _sanitize(maybe_url: str) -> str | None:
1761
1752
  href = strip_html5_whitespace(maybe_url)
1762
1753
  href = safe_url_string(href)
@@ -1784,7 +1775,7 @@ class Capture():
1784
1775
  continue
1785
1776
  try:
1786
1777
  if href := _sanitize(href):
1787
- if not rendered_hostname_only:
1778
+ if not self.rendered_hostname_only:
1788
1779
  urls.add(href)
1789
1780
  elif rendered_hostname and urlparse(href).hostname == rendered_hostname:
1790
1781
  urls.add(href)
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "PlaywrightCapture"
3
- version = "1.41.4"
3
+ version = "1.41.6"
4
4
  description = "A simple library to capture websites using playwright"
5
5
  authors = [
6
6
  {name="Raphaël Vinot", email= "raphael.vinot@circl.lu"}
@@ -25,7 +25,7 @@ dependencies = [
25
25
  "rfc3161-client (>=1.0.4,<2.0.0)",
26
26
  "orjson (>=3.12,<4.0.0)",
27
27
  "pure-magic-rs (>=0.5)",
28
- "lookyloo-models (>=0.4.1)",
28
+ "lookyloo-models (>=0.4.3)",
29
29
  "charset-normalizer (>=3.4.6,<4.0.0)",
30
30
  "pyfaup-rs (>=0.4.6,<0.5.0)"
31
31
  ]