PlaywrightCapture 1.41.4__tar.gz → 1.41.6__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {playwrightcapture-1.41.4 → playwrightcapture-1.41.6}/PKG-INFO +7 -4
- {playwrightcapture-1.41.4 → playwrightcapture-1.41.6}/README.md +5 -2
- {playwrightcapture-1.41.4 → playwrightcapture-1.41.6}/playwrightcapture/capture.py +83 -92
- {playwrightcapture-1.41.4 → playwrightcapture-1.41.6}/pyproject.toml +2 -2
- {playwrightcapture-1.41.4 → playwrightcapture-1.41.6}/LICENSE +0 -0
- {playwrightcapture-1.41.4 → playwrightcapture-1.41.6}/playwrightcapture/__init__.py +0 -0
- {playwrightcapture-1.41.4 → playwrightcapture-1.41.6}/playwrightcapture/exceptions.py +0 -0
- {playwrightcapture-1.41.4 → playwrightcapture-1.41.6}/playwrightcapture/helpers.py +0 -0
- {playwrightcapture-1.41.4 → playwrightcapture-1.41.6}/playwrightcapture/py.typed +0 -0
- {playwrightcapture-1.41.4 → playwrightcapture-1.41.6}/playwrightcapture/socks5dnslookup.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: PlaywrightCapture
|
|
3
|
-
Version: 1.41.
|
|
3
|
+
Version: 1.41.6
|
|
4
4
|
Summary: A simple library to capture websites using playwright
|
|
5
5
|
License-Expression: BSD-3-Clause
|
|
6
6
|
License-File: LICENSE
|
|
@@ -25,7 +25,7 @@ Requires-Dist: async-timeout (>=5.0.1) ; python_version < "3.11"
|
|
|
25
25
|
Requires-Dist: beautifulsoup4[charset-normalizer,lxml] (>=4.15.0)
|
|
26
26
|
Requires-Dist: charset-normalizer (>=3.4.6,<4.0.0)
|
|
27
27
|
Requires-Dist: dnspython (>=2.7.0,<3.0.0)
|
|
28
|
-
Requires-Dist: lookyloo-models (>=0.4.
|
|
28
|
+
Requires-Dist: lookyloo-models (>=0.4.3)
|
|
29
29
|
Requires-Dist: orjson (>=3.12,<4.0.0)
|
|
30
30
|
Requires-Dist: playwright (>=1.63.0)
|
|
31
31
|
Requires-Dist: playwright-stealth (>=2.0.3)
|
|
@@ -56,10 +56,13 @@ A very basic example:
|
|
|
56
56
|
|
|
57
57
|
```python
|
|
58
58
|
from playwrightcapture import Capture
|
|
59
|
+
from lookyloo_models import CaptureSettings
|
|
59
60
|
|
|
60
|
-
|
|
61
|
+
capture_settings = CaptureSettings(url='google.com')
|
|
62
|
+
|
|
63
|
+
async with Capture(capture_settings=capture_settings) as capture:
|
|
61
64
|
await capture.initialize_context()
|
|
62
|
-
entries = await capture.capture_page(
|
|
65
|
+
entries = await capture.capture_page(max_depth_capture_time=90)
|
|
63
66
|
```
|
|
64
67
|
|
|
65
68
|
Entries is a dictionaries that contains (if all goes well) the HAR, the screenshot, all the cookies of the session, the URL as it is in the browser at the end of the capture, and the full HTML page as rendered.
|
|
@@ -14,10 +14,13 @@ A very basic example:
|
|
|
14
14
|
|
|
15
15
|
```python
|
|
16
16
|
from playwrightcapture import Capture
|
|
17
|
+
from lookyloo_models import CaptureSettings
|
|
17
18
|
|
|
18
|
-
|
|
19
|
+
capture_settings = CaptureSettings(url='google.com')
|
|
20
|
+
|
|
21
|
+
async with Capture(capture_settings=capture_settings) as capture:
|
|
19
22
|
await capture.initialize_context()
|
|
20
|
-
entries = await capture.capture_page(
|
|
23
|
+
entries = await capture.capture_page(max_depth_capture_time=90)
|
|
21
24
|
```
|
|
22
25
|
|
|
23
26
|
Entries is a dictionaries that contains (if all goes well) the HAR, the screenshot, all the cookies of the session, the URL as it is in the browser at the end of the capture, and the full HTML page as rendered.
|
|
@@ -18,7 +18,7 @@ from base64 import b64decode, b64encode
|
|
|
18
18
|
from io import BytesIO
|
|
19
19
|
from logging import LoggerAdapter, Logger
|
|
20
20
|
from tempfile import NamedTemporaryFile
|
|
21
|
-
from typing import Any, Literal, TYPE_CHECKING
|
|
21
|
+
from typing import Any, Literal, TYPE_CHECKING
|
|
22
22
|
from collections.abc import Awaitable, Callable, MutableMapping
|
|
23
23
|
from urllib.parse import urlparse, unquote, urljoin, urlsplit, urlunsplit, parse_qs, unquote_plus
|
|
24
24
|
from zipfile import ZipFile
|
|
@@ -201,6 +201,24 @@ class Capture():
|
|
|
201
201
|
self._color_scheme: Literal['dark', 'light', 'no-preference', 'null'] | None = capture_settings.color_scheme if capture_settings.color_scheme else None
|
|
202
202
|
self._java_script_enabled: bool = capture_settings.java_script_enabled
|
|
203
203
|
self.capture_timeout = capture_settings.general_timeout_in_sec
|
|
204
|
+
self.allow_tracking = capture_settings.allow_tracking
|
|
205
|
+
self.rendered_hostname_only = capture_settings.rendered_hostname_only
|
|
206
|
+
self.with_screenshot = capture_settings.with_screenshot
|
|
207
|
+
self.with_favicon = capture_settings.with_favicon
|
|
208
|
+
self.with_trusted_timestamps = capture_settings.with_trusted_timestamps
|
|
209
|
+
self.capture_depth = capture_settings.depth
|
|
210
|
+
self.final_wait = capture_settings.final_wait
|
|
211
|
+
|
|
212
|
+
self.initial_url: str
|
|
213
|
+
if capture_settings.url:
|
|
214
|
+
# This url could be None in transit when the thing to capture is a file (so the models allows it)
|
|
215
|
+
# But at this stage, the value must have been set to the local path of the file,
|
|
216
|
+
# if it is none, the capture will for sure fail.
|
|
217
|
+
self.initial_url = capture_settings.url
|
|
218
|
+
else:
|
|
219
|
+
raise InvalidPlaywrightParameter('No URL provided, cannot capture.')
|
|
220
|
+
|
|
221
|
+
self.initial_referer = capture_settings.referer
|
|
204
222
|
|
|
205
223
|
self.should_retry: bool = False
|
|
206
224
|
self.__network_not_idle: int = 2 # makes sure we do not wait for network idle the max amount of time the capture is allowed to take
|
|
@@ -285,6 +303,9 @@ class Capture():
|
|
|
285
303
|
# Create the temporary file to store the HAR content.
|
|
286
304
|
self._temp_harfile = NamedTemporaryFile(delete=False, prefix="playwright_capture_har", suffix=".json")
|
|
287
305
|
|
|
306
|
+
# all the errors gathered during the capture
|
|
307
|
+
self.errors: list[str] = []
|
|
308
|
+
|
|
288
309
|
return self
|
|
289
310
|
|
|
290
311
|
async def __aexit__(self, exc_type: Any, exc_value: Any, traceback: Any) -> bool:
|
|
@@ -315,7 +336,7 @@ class Capture():
|
|
|
315
336
|
return False
|
|
316
337
|
return True
|
|
317
338
|
|
|
318
|
-
async def setup_page_capture(self
|
|
339
|
+
async def setup_page_capture(self) -> Page:
|
|
319
340
|
"""Prepare a page for a single-page capture without changing capture semantics.
|
|
320
341
|
|
|
321
342
|
This method preserves the existing per-page setup used by capture_page:
|
|
@@ -421,7 +442,7 @@ class Capture():
|
|
|
421
442
|
except Error as e:
|
|
422
443
|
self.logger.warning(f'Failed at fetching PDF in headless chromium: {e}')
|
|
423
444
|
|
|
424
|
-
if allow_tracking:
|
|
445
|
+
if self.allow_tracking:
|
|
425
446
|
# Add authorization clickthroughs
|
|
426
447
|
await self.__dialog_didomi_clickthrough(page)
|
|
427
448
|
await self.__dialog_onetrust_clickthrough(page)
|
|
@@ -884,7 +905,7 @@ class Capture():
|
|
|
884
905
|
except Exception as e:
|
|
885
906
|
self.logger.info(f'Error while moving time forward: {e}')
|
|
886
907
|
|
|
887
|
-
async def __instrumentation(self, page: Page, url: str
|
|
908
|
+
async def __instrumentation(self, page: Page, url: str) -> None:
|
|
888
909
|
try:
|
|
889
910
|
# NOTE: the clock must be installed after the page is loaded, otherwise it sometimes cause the complete capture to hang.
|
|
890
911
|
await page.clock.install()
|
|
@@ -935,7 +956,7 @@ class Capture():
|
|
|
935
956
|
await self._wait_for_random_timeout(page, 5)
|
|
936
957
|
self.logger.debug('Keep going after moving mouse.')
|
|
937
958
|
|
|
938
|
-
if allow_tracking:
|
|
959
|
+
if self.allow_tracking:
|
|
939
960
|
await self._wait_for_random_timeout(page, 5)
|
|
940
961
|
# This event is required trigger the add_locator_handler
|
|
941
962
|
try:
|
|
@@ -1014,12 +1035,12 @@ class Capture():
|
|
|
1014
1035
|
|
|
1015
1036
|
self.logger.debug('Done with instrumentation.')
|
|
1016
1037
|
# Wait at least 5 sec after instrumentation
|
|
1017
|
-
self.logger.debug(f'Waiting another {max(final_wait, 5)}s.')
|
|
1018
|
-
await self._wait_for_random_timeout(page, max(final_wait, 5))
|
|
1038
|
+
self.logger.debug(f'Waiting another {max(self.final_wait, 5)}s.')
|
|
1039
|
+
await self._wait_for_random_timeout(page, max(self.final_wait, 5))
|
|
1019
1040
|
await self._safe_wait(page)
|
|
1020
1041
|
self.logger.debug('Done with waiting.')
|
|
1021
1042
|
|
|
1022
|
-
async def _safe_get_storage_state(self
|
|
1043
|
+
async def _safe_get_storage_state(self) -> dict[str, Any]:
|
|
1023
1044
|
# Collect storage state, including IndexedDB, to capture the full browser state.
|
|
1024
1045
|
# 2026-09-08: add WebAuth credentials
|
|
1025
1046
|
# 2026-09-17: Add opfs
|
|
@@ -1031,31 +1052,31 @@ class Capture():
|
|
|
1031
1052
|
return await self.context.storage_state(**to_store) # type: ignore[return-value,arg-type]
|
|
1032
1053
|
except (TimeoutError, asyncio.TimeoutError):
|
|
1033
1054
|
self.logger.warning("Unable to get storage (timeout).")
|
|
1034
|
-
errors.append("Unable to get the storage (timeout).")
|
|
1055
|
+
self.errors.append("Unable to get the storage (timeout).")
|
|
1035
1056
|
self.should_retry = True
|
|
1036
1057
|
break
|
|
1037
1058
|
except Error as e:
|
|
1038
1059
|
if to_store['indexed_db'] and 'IndexedDB' in str(e):
|
|
1039
1060
|
to_store['indexed_db'] = False
|
|
1040
|
-
errors.append('Unable to get the IndexedDB')
|
|
1061
|
+
self.errors.append('Unable to get the IndexedDB')
|
|
1041
1062
|
self.logger.warning(f"Unable to get the IndexedDB: {e}")
|
|
1042
1063
|
continue
|
|
1043
1064
|
if to_store['opfs'] and 'OPFS' in str(e):
|
|
1044
1065
|
to_store['opfs'] = False
|
|
1045
|
-
errors.append('Unable to get the OPFS')
|
|
1066
|
+
self.errors.append('Unable to get the OPFS')
|
|
1046
1067
|
self.logger.warning(f"Unable to get the OPFS: {e}")
|
|
1047
1068
|
continue
|
|
1048
1069
|
|
|
1049
1070
|
if not to_store['indexed_db'] and not to_store['opfs']:
|
|
1050
1071
|
# we disabled both options, quit
|
|
1051
1072
|
self.should_retry = True
|
|
1052
|
-
errors.append(f'Unable to get the storage at all: {e}')
|
|
1073
|
+
self.errors.append(f'Unable to get the storage at all: {e}')
|
|
1053
1074
|
self.logger.warning(f"Unable to get the storage at all: {e}")
|
|
1054
1075
|
break
|
|
1055
1076
|
except Exception as e:
|
|
1056
1077
|
# When the driver explodes for no clear reason.
|
|
1057
1078
|
self.logger.warning(f"[Generic Exception] Unable to get the storage: {e}")
|
|
1058
|
-
errors.append(f'[Generic Exception] Unable to get the storage: {e}')
|
|
1079
|
+
self.errors.append(f'[Generic Exception] Unable to get the storage: {e}')
|
|
1059
1080
|
self.should_retry = True
|
|
1060
1081
|
break
|
|
1061
1082
|
return {}
|
|
@@ -1065,8 +1086,6 @@ class Capture():
|
|
|
1065
1086
|
*,
|
|
1066
1087
|
page: Page,
|
|
1067
1088
|
to_return: CaptureResponse,
|
|
1068
|
-
errors: list[str],
|
|
1069
|
-
with_trusted_timestamps: bool,
|
|
1070
1089
|
) -> None:
|
|
1071
1090
|
"""Common finalization logic for captures (downloads, cookies, storage, HAR, socks5, timestamps)."""
|
|
1072
1091
|
|
|
@@ -1097,19 +1116,19 @@ class Capture():
|
|
|
1097
1116
|
to_return['cookies'] = [Cookie.model_validate(c).model_dump(exclude_none=True) for c in await self.context.cookies()]
|
|
1098
1117
|
except (TimeoutError, asyncio.TimeoutError):
|
|
1099
1118
|
self.logger.warning("Unable to get cookies (timeout).")
|
|
1100
|
-
errors.append("Unable to get the cookies (timeout).")
|
|
1119
|
+
self.errors.append("Unable to get the cookies (timeout).")
|
|
1101
1120
|
self.should_retry = True
|
|
1102
1121
|
except Error as e:
|
|
1103
1122
|
self.logger.warning(f"Unable to get cookies: {e}")
|
|
1104
|
-
errors.append(f'Unable to get the cookies: {e}')
|
|
1123
|
+
self.errors.append(f'Unable to get the cookies: {e}')
|
|
1105
1124
|
self.should_retry = True
|
|
1106
1125
|
except Exception as e:
|
|
1107
1126
|
# When the driver explodes for no clear reason.
|
|
1108
1127
|
self.logger.warning(f"[Generic Exception] Unable to get cookies: {e}")
|
|
1109
|
-
errors.append(f'[Generic Exception] Unable to get the cookies: {e}')
|
|
1128
|
+
self.errors.append(f'[Generic Exception] Unable to get the cookies: {e}')
|
|
1110
1129
|
self.should_retry = True
|
|
1111
1130
|
|
|
1112
|
-
to_return['storage'] = await self._safe_get_storage_state(
|
|
1131
|
+
to_return['storage'] = await self._safe_get_storage_state()
|
|
1113
1132
|
|
|
1114
1133
|
try:
|
|
1115
1134
|
if page.is_closed():
|
|
@@ -1167,14 +1186,14 @@ class Capture():
|
|
|
1167
1186
|
await self.socks5_resolver(har)
|
|
1168
1187
|
except (TimeoutError, asyncio.TimeoutError):
|
|
1169
1188
|
self.logger.warning("Unable to resolve all the IPs via the socks5 proxy.")
|
|
1170
|
-
errors.append("Unable to resolve all the IPs via the socks5 proxy.")
|
|
1189
|
+
self.errors.append("Unable to resolve all the IPs via the socks5 proxy.")
|
|
1171
1190
|
self.should_retry = True
|
|
1172
1191
|
|
|
1173
1192
|
except (TimeoutError, asyncio.TimeoutError):
|
|
1174
1193
|
# If closing the context or generating the HAR takes too long, the
|
|
1175
1194
|
# capture is considered incomplete but we still return what we have.
|
|
1176
1195
|
self.logger.warning("[Timeout] Unable to close context at the end of the capture.")
|
|
1177
|
-
errors.append("[Timeout] Unable to close context at the end of the capture.")
|
|
1196
|
+
self.errors.append("[Timeout] Unable to close context at the end of the capture.")
|
|
1178
1197
|
self.should_retry = True
|
|
1179
1198
|
# In case of timeout, let the exception reach the async calls
|
|
1180
1199
|
await asyncio.sleep(1)
|
|
@@ -1182,11 +1201,11 @@ class Capture():
|
|
|
1182
1201
|
# Any other unexpected failure while finalizing the capture is logged
|
|
1183
1202
|
# and surfaced as a generic HAR-generation error.
|
|
1184
1203
|
self.logger.warning(f"Other exception while finishing up the capture: {e}.")
|
|
1185
|
-
errors.append(f'Unable to generate HAR file: {e}')
|
|
1204
|
+
self.errors.append(f'Unable to generate HAR file: {e}')
|
|
1186
1205
|
|
|
1187
|
-
if errors:
|
|
1188
|
-
to_return['error'] = '\n'.join(errors)
|
|
1189
|
-
if with_trusted_timestamps:
|
|
1206
|
+
if self.errors:
|
|
1207
|
+
to_return['error'] = '\n'.join(self.errors)
|
|
1208
|
+
if self.with_trusted_timestamps:
|
|
1190
1209
|
try:
|
|
1191
1210
|
await self._get_trusted_timestamps(to_return)
|
|
1192
1211
|
except Exception as e:
|
|
@@ -1237,7 +1256,11 @@ class Capture():
|
|
|
1237
1256
|
self.logger.warning(f"Opening a weird URL: {url}")
|
|
1238
1257
|
return False, f"Attempted to open a weird URL '{url}', blocked."
|
|
1239
1258
|
|
|
1240
|
-
async def open_page(self, page: Page, url: str
|
|
1259
|
+
async def open_page(self, page: Page, url: str | None = None, referer: str | None=None) -> None:
|
|
1260
|
+
"""This method opens the page but does nothing with it. Use it only if you need a custom instrumentation.
|
|
1261
|
+
The usecase in lookyloo's context is to have a headfull capture in Xpra.
|
|
1262
|
+
Prefer using `capture_page` instead.
|
|
1263
|
+
"""
|
|
1241
1264
|
|
|
1242
1265
|
async def catch_file_route(route: Route, request: Request) -> None:
|
|
1243
1266
|
if unquote(request.url) == url:
|
|
@@ -1280,9 +1303,10 @@ class Capture():
|
|
|
1280
1303
|
]
|
|
1281
1304
|
for scheme in allowed_schemes:
|
|
1282
1305
|
await page.route(scheme, lambda route: route.continue_())
|
|
1283
|
-
|
|
1306
|
+
if not url:
|
|
1307
|
+
url = self.initial_url
|
|
1284
1308
|
try:
|
|
1285
|
-
await page.goto(url, wait_until='domcontentloaded', referer=referer if referer else
|
|
1309
|
+
await page.goto(url, wait_until='domcontentloaded', referer=referer if referer else self.initial_referer)
|
|
1286
1310
|
try:
|
|
1287
1311
|
await page.bring_to_front()
|
|
1288
1312
|
self.logger.debug('Page moved to front.')
|
|
@@ -1298,7 +1322,7 @@ class Capture():
|
|
|
1298
1322
|
try:
|
|
1299
1323
|
async with page.expect_download() as download_info:
|
|
1300
1324
|
try:
|
|
1301
|
-
await page.goto(url, referer=referer if referer else
|
|
1325
|
+
await page.goto(url, referer=referer if referer else self.initial_referer)
|
|
1302
1326
|
except Exception:
|
|
1303
1327
|
pass
|
|
1304
1328
|
with NamedTemporaryFile() as tmp_f:
|
|
@@ -1316,7 +1340,7 @@ class Capture():
|
|
|
1316
1340
|
error_msg = download.failure()
|
|
1317
1341
|
if not error_msg:
|
|
1318
1342
|
raise e
|
|
1319
|
-
errors.append(f"Error while downloading: {error_msg}")
|
|
1343
|
+
self.errors.append(f"Error while downloading: {error_msg}")
|
|
1320
1344
|
self.logger.info(f'Error while downloading: {error_msg}')
|
|
1321
1345
|
self.should_retry = True
|
|
1322
1346
|
except Exception:
|
|
@@ -1326,44 +1350,10 @@ class Capture():
|
|
|
1326
1350
|
else:
|
|
1327
1351
|
await self._wait_for_random_timeout(page, 5) # Wait 5 sec after document loaded
|
|
1328
1352
|
|
|
1329
|
-
@overload
|
|
1330
|
-
async def capture_page(self, url: str, *, max_depth_capture_time: int,
|
|
1331
|
-
referer: str | None=None,
|
|
1332
|
-
page: Page | None=None, depth: int=0,
|
|
1333
|
-
rendered_hostname_only: bool=True,
|
|
1334
|
-
with_screenshot: bool=True,
|
|
1335
|
-
with_favicon: bool=False,
|
|
1336
|
-
allow_tracking: bool=False,
|
|
1337
|
-
with_trusted_timestamps: bool=False,
|
|
1338
|
-
current_page_only: bool=False,
|
|
1339
|
-
final_wait: int=5
|
|
1340
|
-
) -> CaptureResponse:
|
|
1341
|
-
...
|
|
1342
|
-
|
|
1343
|
-
@overload
|
|
1344
|
-
async def capture_page(self, url: None=None, *, max_depth_capture_time: int,
|
|
1345
|
-
referer: str | None=None,
|
|
1346
|
-
page: Page, depth: int=0,
|
|
1347
|
-
rendered_hostname_only: bool=True,
|
|
1348
|
-
with_screenshot: bool=True,
|
|
1349
|
-
with_favicon: bool=False,
|
|
1350
|
-
allow_tracking: bool=False,
|
|
1351
|
-
with_trusted_timestamps: bool=False,
|
|
1352
|
-
current_page_only: bool=False,
|
|
1353
|
-
final_wait: int=5
|
|
1354
|
-
) -> CaptureResponse:
|
|
1355
|
-
...
|
|
1356
|
-
|
|
1357
1353
|
async def capture_page(self, url: str | None=None, *, max_depth_capture_time: int,
|
|
1358
1354
|
referer: str | None=None,
|
|
1359
|
-
page: Page | None=None, depth: int=
|
|
1360
|
-
rendered_hostname_only: bool=True,
|
|
1361
|
-
with_screenshot: bool=True,
|
|
1362
|
-
with_favicon: bool=False,
|
|
1363
|
-
allow_tracking: bool=False,
|
|
1364
|
-
with_trusted_timestamps: bool=False,
|
|
1355
|
+
page: Page | None=None, depth: int | None = None,
|
|
1365
1356
|
current_page_only: bool=False,
|
|
1366
|
-
final_wait: int=5,
|
|
1367
1357
|
) -> CaptureResponse:
|
|
1368
1358
|
"""Capture a URL and optionally recurse into child links.
|
|
1369
1359
|
|
|
@@ -1376,10 +1366,11 @@ class Capture():
|
|
|
1376
1366
|
(no navigation, no recursion) and then finalizes. This is the path
|
|
1377
1367
|
used by remote headfull captures after setup_page_capture has already been
|
|
1378
1368
|
called by the caller.
|
|
1369
|
+
|
|
1370
|
+
The `url` field is only needed when the URL to capture isn't the initial one (depth>0)
|
|
1379
1371
|
"""
|
|
1380
1372
|
|
|
1381
1373
|
to_return: CaptureResponse = {}
|
|
1382
|
-
errors: list[str] = []
|
|
1383
1374
|
capturing_sub = False
|
|
1384
1375
|
|
|
1385
1376
|
if current_page_only:
|
|
@@ -1388,7 +1379,7 @@ class Capture():
|
|
|
1388
1379
|
raise InvalidPlaywrightParameter('current_page_only requires an initialized page')
|
|
1389
1380
|
else:
|
|
1390
1381
|
if page is None:
|
|
1391
|
-
page = await self.setup_page_capture(
|
|
1382
|
+
page = await self.setup_page_capture()
|
|
1392
1383
|
else:
|
|
1393
1384
|
# Automated capture with depth > 0
|
|
1394
1385
|
capturing_sub = True
|
|
@@ -1397,12 +1388,13 @@ class Capture():
|
|
|
1397
1388
|
if not current_page_only:
|
|
1398
1389
|
# Standard navigation + capture path.
|
|
1399
1390
|
if not url:
|
|
1400
|
-
|
|
1401
|
-
await self.open_page(page, url
|
|
1391
|
+
url = self.initial_url
|
|
1392
|
+
await self.open_page(page, url=url if url else self.initial_url,
|
|
1393
|
+
referer=referer if referer else self.initial_referer)
|
|
1402
1394
|
|
|
1403
1395
|
try:
|
|
1404
1396
|
if self.headless:
|
|
1405
|
-
await self.__instrumentation(page, url
|
|
1397
|
+
await self.__instrumentation(page, url)
|
|
1406
1398
|
else:
|
|
1407
1399
|
self.logger.debug('Headed mode, skipping instrumentation.')
|
|
1408
1400
|
await self._wait_for_random_timeout(page, self._capture_timeout - 5)
|
|
@@ -1436,7 +1428,7 @@ class Capture():
|
|
|
1436
1428
|
u = '/!\\ Unknown /!\\'
|
|
1437
1429
|
to_return['last_redirected_url'] = u
|
|
1438
1430
|
|
|
1439
|
-
if 'html' in to_return and to_return['html'] is not None and with_favicon:
|
|
1431
|
+
if 'html' in to_return and to_return['html'] is not None and self.with_favicon:
|
|
1440
1432
|
# We're probably (?) safe only looking for favicons in the main frame.
|
|
1441
1433
|
# TODO: check that?
|
|
1442
1434
|
try:
|
|
@@ -1447,7 +1439,7 @@ class Capture():
|
|
|
1447
1439
|
except Exception as e:
|
|
1448
1440
|
self.logger.warning(f'Unable to get favicons: {e}')
|
|
1449
1441
|
|
|
1450
|
-
if with_screenshot:
|
|
1442
|
+
if self.with_screenshot:
|
|
1451
1443
|
to_return['png'] = await self._failsafe_get_screenshot(page)
|
|
1452
1444
|
|
|
1453
1445
|
# Keep that all the way down there in case the capture failed.
|
|
@@ -1456,11 +1448,15 @@ class Capture():
|
|
|
1456
1448
|
else:
|
|
1457
1449
|
self._already_captured.add(page.url)
|
|
1458
1450
|
|
|
1459
|
-
if depth
|
|
1451
|
+
if depth is None:
|
|
1452
|
+
# fallback for the first call
|
|
1453
|
+
depth = self.capture_depth
|
|
1454
|
+
|
|
1455
|
+
if depth is not None and depth > 0 and to_return.get('html') and to_return['html']:
|
|
1460
1456
|
# TODO with children frames:
|
|
1461
1457
|
# 1. if the frame has a URL, use that as base URL/referer for the subsequent captures
|
|
1462
1458
|
# 2. if it doesn't, the base URL is the url of the parent (which may or may not be the main frame)
|
|
1463
|
-
if child_urls := self._get_links_from_rendered_page(page.url, to_return['html']
|
|
1459
|
+
if child_urls := self._get_links_from_rendered_page(page.url, to_return['html']):
|
|
1464
1460
|
to_return['children'] = []
|
|
1465
1461
|
depth -= 1
|
|
1466
1462
|
total_urls = len(child_urls)
|
|
@@ -1486,11 +1482,8 @@ class Capture():
|
|
|
1486
1482
|
child_capture = await self.capture_page(
|
|
1487
1483
|
url=url, referer=page.url,
|
|
1488
1484
|
page=page, depth=depth,
|
|
1489
|
-
|
|
1490
|
-
|
|
1491
|
-
with_screenshot=with_screenshot,
|
|
1492
|
-
final_wait=final_wait)
|
|
1493
|
-
if with_trusted_timestamps:
|
|
1485
|
+
max_depth_capture_time=max_capture_time)
|
|
1486
|
+
if self.with_trusted_timestamps:
|
|
1494
1487
|
try:
|
|
1495
1488
|
await self._get_trusted_timestamps(child_capture)
|
|
1496
1489
|
except Exception as e:
|
|
@@ -1510,7 +1503,7 @@ class Capture():
|
|
|
1510
1503
|
if consecutive_errors >= 5:
|
|
1511
1504
|
# if we have more than 5 consecutive errors, the capture is most probably broken, breaking.
|
|
1512
1505
|
self.logger.warning('Got more than 5 consecutive errors while capturing children, breaking.')
|
|
1513
|
-
errors.append("Got more than 5 consecutive errors while capturing children")
|
|
1506
|
+
self.errors.append("Got more than 5 consecutive errors while capturing children")
|
|
1514
1507
|
self.should_retry = True
|
|
1515
1508
|
break
|
|
1516
1509
|
|
|
@@ -1522,19 +1515,19 @@ class Capture():
|
|
|
1522
1515
|
self.logger.info(f'Unable to go back: {e}.')
|
|
1523
1516
|
|
|
1524
1517
|
except PlaywrightTimeoutError as e:
|
|
1525
|
-
errors.append(f"The capture took too long - {e.message}")
|
|
1518
|
+
self.errors.append(f"The capture took too long - {e.message}")
|
|
1526
1519
|
self.should_retry = True
|
|
1527
1520
|
except (asyncio.TimeoutError, TimeoutError):
|
|
1528
|
-
errors.append("Something in the capture took too long")
|
|
1521
|
+
self.errors.append("Something in the capture took too long")
|
|
1529
1522
|
self.should_retry = True
|
|
1530
1523
|
except TargetClosedError as e:
|
|
1531
|
-
errors.append(f"The target was closed - {e}")
|
|
1524
|
+
self.errors.append(f"The target was closed - {e}")
|
|
1532
1525
|
self.should_retry = True
|
|
1533
1526
|
except Error as e:
|
|
1534
1527
|
# NOTE: there are a lot of errors that look like duplicates and they are triggered at different times in the process.
|
|
1535
1528
|
# it is tricky to figure our which one should (and should not) trigger a retry. Below is our best guess and it will change over time.
|
|
1536
1529
|
self._update_exceptions(e)
|
|
1537
|
-
errors.append(e.message)
|
|
1530
|
+
self.errors.append(e.message)
|
|
1538
1531
|
to_return['error_name'] = e.name
|
|
1539
1532
|
# NOTE: e.name is generally (always?) "Error"
|
|
1540
1533
|
if self._fatal_network_error(e) or self._fatal_auth_error(e) or self.fatal_browser_error(e):
|
|
@@ -1542,7 +1535,7 @@ class Capture():
|
|
|
1542
1535
|
elif self._retry_network_error(e) or self._retry_browser_error(e):
|
|
1543
1536
|
# this one sounds like something we can retry...
|
|
1544
1537
|
self.logger.info(f'Issue with {url} (retrying): {e.message}')
|
|
1545
|
-
errors.append(f'Issue with {url}: {e.message}')
|
|
1538
|
+
self.errors.append(f'Issue with {url}: {e.message}')
|
|
1546
1539
|
self.should_retry = True
|
|
1547
1540
|
else:
|
|
1548
1541
|
# Unexpected ones
|
|
@@ -1550,15 +1543,15 @@ class Capture():
|
|
|
1550
1543
|
except PlaywrightCaptureException as e:
|
|
1551
1544
|
# unrecoverable exeptions
|
|
1552
1545
|
self.logger.warning(f'Unable to run capture: {e}')
|
|
1553
|
-
errors.append(f'Unable to run capture: {e}')
|
|
1546
|
+
self.errors.append(f'Unable to run capture: {e}')
|
|
1554
1547
|
raise e
|
|
1555
1548
|
except Exception as e:
|
|
1556
1549
|
# we may get a non-playwright exception to.
|
|
1557
1550
|
# The ones we try to handle here should be treated as if they were.
|
|
1558
|
-
errors.append(str(e))
|
|
1551
|
+
self.errors.append(str(e))
|
|
1559
1552
|
if str(e) in ['Connection closed while reading from the driver']:
|
|
1560
1553
|
self.logger.info(f'Issue with {url} (retrying): {e}')
|
|
1561
|
-
errors.append(f'Issue with {url}: {e}')
|
|
1554
|
+
self.errors.append(f'Issue with {url}: {e}')
|
|
1562
1555
|
self.should_retry = True
|
|
1563
1556
|
else:
|
|
1564
1557
|
raise e
|
|
@@ -1569,8 +1562,6 @@ class Capture():
|
|
|
1569
1562
|
await self._finalize_capture(
|
|
1570
1563
|
page=page,
|
|
1571
1564
|
to_return=to_return,
|
|
1572
|
-
errors=errors,
|
|
1573
|
-
with_trusted_timestamps=with_trusted_timestamps,
|
|
1574
1565
|
)
|
|
1575
1566
|
self.logger.debug('Capture done')
|
|
1576
1567
|
return to_return
|
|
@@ -1756,7 +1747,7 @@ class Capture():
|
|
|
1756
1747
|
return unquote(page.name)
|
|
1757
1748
|
return None
|
|
1758
1749
|
|
|
1759
|
-
def _get_links_from_rendered_page(self, rendered_url: str, rendered_html: str
|
|
1750
|
+
def _get_links_from_rendered_page(self, rendered_url: str, rendered_html: str) -> list[str]:
|
|
1760
1751
|
def _sanitize(maybe_url: str) -> str | None:
|
|
1761
1752
|
href = strip_html5_whitespace(maybe_url)
|
|
1762
1753
|
href = safe_url_string(href)
|
|
@@ -1784,7 +1775,7 @@ class Capture():
|
|
|
1784
1775
|
continue
|
|
1785
1776
|
try:
|
|
1786
1777
|
if href := _sanitize(href):
|
|
1787
|
-
if not rendered_hostname_only:
|
|
1778
|
+
if not self.rendered_hostname_only:
|
|
1788
1779
|
urls.add(href)
|
|
1789
1780
|
elif rendered_hostname and urlparse(href).hostname == rendered_hostname:
|
|
1790
1781
|
urls.add(href)
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "PlaywrightCapture"
|
|
3
|
-
version = "1.41.
|
|
3
|
+
version = "1.41.6"
|
|
4
4
|
description = "A simple library to capture websites using playwright"
|
|
5
5
|
authors = [
|
|
6
6
|
{name="Raphaël Vinot", email= "raphael.vinot@circl.lu"}
|
|
@@ -25,7 +25,7 @@ dependencies = [
|
|
|
25
25
|
"rfc3161-client (>=1.0.4,<2.0.0)",
|
|
26
26
|
"orjson (>=3.12,<4.0.0)",
|
|
27
27
|
"pure-magic-rs (>=0.5)",
|
|
28
|
-
"lookyloo-models (>=0.4.
|
|
28
|
+
"lookyloo-models (>=0.4.3)",
|
|
29
29
|
"charset-normalizer (>=3.4.6,<4.0.0)",
|
|
30
30
|
"pyfaup-rs (>=0.4.6,<0.5.0)"
|
|
31
31
|
]
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|