browserfetch 0.11.0__tar.gz → 0.12.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {browserfetch-0.11.0 → browserfetch-0.12.0}/PKG-INFO +1 -1
- {browserfetch-0.11.0 → browserfetch-0.12.0}/browserfetch/__init__.py +16 -21
- browserfetch-0.12.0/browserfetch/browserfetch.egg-info/PKG-INFO +71 -0
- browserfetch-0.12.0/browserfetch/browserfetch.egg-info/SOURCES.txt +9 -0
- browserfetch-0.12.0/browserfetch/browserfetch.egg-info/dependency_links.txt +1 -0
- browserfetch-0.12.0/browserfetch/browserfetch.egg-info/not-zip-safe +1 -0
- browserfetch-0.12.0/browserfetch/browserfetch.egg-info/requires.txt +2 -0
- browserfetch-0.12.0/browserfetch/browserfetch.egg-info/top_level.txt +1 -0
- {browserfetch-0.11.0 → browserfetch-0.12.0}/browserfetch/browserfetch.js +2 -2
- {browserfetch-0.11.0 → browserfetch-0.12.0}/pyproject.toml +15 -0
- browserfetch-0.11.0/tests/test_init.py +0 -25
- {browserfetch-0.11.0 → browserfetch-0.12.0}/LICENSE +0 -0
- {browserfetch-0.11.0 → browserfetch-0.12.0}/README.rst +0 -0
- {browserfetch-0.11.0 → browserfetch-0.12.0}/browserfetch/__main__.py +0 -0
|
@@ -1,11 +1,11 @@
|
|
|
1
|
-
__version__ = '0.
|
|
1
|
+
__version__ = '0.12.0'
|
|
2
2
|
__all__ = [
|
|
3
3
|
'BrowserError',
|
|
4
|
+
'Response',
|
|
4
5
|
'evaluate',
|
|
5
6
|
'fetch',
|
|
6
7
|
'get',
|
|
7
8
|
'post',
|
|
8
|
-
'Response',
|
|
9
9
|
'start_server',
|
|
10
10
|
]
|
|
11
11
|
import atexit
|
|
@@ -21,8 +21,9 @@ from collections import defaultdict
|
|
|
21
21
|
from dataclasses import dataclass
|
|
22
22
|
from functools import partial as _partial
|
|
23
23
|
from html import escape
|
|
24
|
-
from json import dumps as _dumps, loads
|
|
24
|
+
from json import dumps as _dumps, loads as _loads
|
|
25
25
|
from logging import getLogger
|
|
26
|
+
from typing import Any as _Any
|
|
26
27
|
from urllib.parse import parse_qsl, urlencode, urlparse, urlunparse
|
|
27
28
|
|
|
28
29
|
from aiohttp import ClientSession, ClientWebSocketResponse
|
|
@@ -72,8 +73,8 @@ class Response:
|
|
|
72
73
|
|
|
73
74
|
def json(self, encoding=None, errors='strict'):
|
|
74
75
|
if encoding is None:
|
|
75
|
-
return
|
|
76
|
-
return
|
|
76
|
+
return _loads(self.body)
|
|
77
|
+
return _loads(self.text(encoding=encoding, errors=errors))
|
|
77
78
|
|
|
78
79
|
|
|
79
80
|
def extract_host(url: str) -> str:
|
|
@@ -123,7 +124,7 @@ async def receive_responses(ws: WebSocketResponse | ClientWebSocketResponse):
|
|
|
123
124
|
while True:
|
|
124
125
|
blob = await ws.receive_bytes()
|
|
125
126
|
json_blob, _, body = blob.partition(b'\0')
|
|
126
|
-
j =
|
|
127
|
+
j = _loads(json_blob)
|
|
127
128
|
j['body'] = body
|
|
128
129
|
event_id = j.pop('event_id')
|
|
129
130
|
try:
|
|
@@ -175,7 +176,7 @@ async def _(request: Request) -> WebSocketResponse:
|
|
|
175
176
|
except TypeError: # ws closed
|
|
176
177
|
return ws
|
|
177
178
|
data, null, body = bytes_.partition(b'\0')
|
|
178
|
-
data =
|
|
179
|
+
data = _loads(data)
|
|
179
180
|
relay_event_id = data['event_id']
|
|
180
181
|
|
|
181
182
|
try:
|
|
@@ -231,14 +232,18 @@ async def relay_client(server_host, server_port):
|
|
|
231
232
|
|
|
232
233
|
|
|
233
234
|
async def evaluate(
|
|
234
|
-
|
|
235
|
+
expression: str,
|
|
236
|
+
/,
|
|
237
|
+
*,
|
|
235
238
|
host: str,
|
|
236
239
|
timeout: int | float = 95,
|
|
240
|
+
arg: _Any = None,
|
|
237
241
|
):
|
|
238
242
|
"""Evaluate string in browser context and return JSON.stringify(result)."""
|
|
239
|
-
|
|
240
|
-
|
|
241
|
-
|
|
243
|
+
data = {'action': 'eval', 'string': expression, 'timeout': timeout}
|
|
244
|
+
if arg is not None:
|
|
245
|
+
data['arg'] = arg
|
|
246
|
+
d = await _request(host, data, None)
|
|
242
247
|
return d['result']
|
|
243
248
|
|
|
244
249
|
|
|
@@ -352,18 +357,9 @@ app.add_routes(routes)
|
|
|
352
357
|
app_runner = AppRunner(app)
|
|
353
358
|
|
|
354
359
|
|
|
355
|
-
def _shutdown_server(loop: AbstractEventLoop):
|
|
356
|
-
logger.info('waiting for app_runner.cleanup()')
|
|
357
|
-
loop.run_until_complete(app_runner.cleanup())
|
|
358
|
-
|
|
359
|
-
|
|
360
360
|
def _cancel_relay_task(loop: AbstractEventLoop, task: Task):
|
|
361
361
|
logger.info('cancelling relay task')
|
|
362
362
|
task.cancel()
|
|
363
|
-
try:
|
|
364
|
-
loop.run_until_complete(task)
|
|
365
|
-
except CancelledError:
|
|
366
|
-
pass
|
|
367
363
|
|
|
368
364
|
|
|
369
365
|
_server = False
|
|
@@ -389,5 +385,4 @@ async def start_server(*, host=_host, port=_port):
|
|
|
389
385
|
atexit.register(_cancel_relay_task, loop, relay_task)
|
|
390
386
|
else:
|
|
391
387
|
_server = True
|
|
392
|
-
atexit.register(_shutdown_server, loop)
|
|
393
388
|
logger.info('server started at http://%s:%s', host, port)
|
|
@@ -0,0 +1,71 @@
|
|
|
1
|
+
Metadata-Version: 2.1
|
|
2
|
+
Name: browserfetch
|
|
3
|
+
Version: 0.1.1.dev0
|
|
4
|
+
Summary: fetch in Python using your browser!
|
|
5
|
+
License: GNU General Public License v3 (GPLv3)
|
|
6
|
+
Project-URL: Homepage, https://github.com/5j9/browserfetch
|
|
7
|
+
Keywords: browser,fetch,python,cookies
|
|
8
|
+
Classifier: License :: OSI Approved :: GNU General Public License v3 (GPLv3)
|
|
9
|
+
Requires-Python: >=3.11
|
|
10
|
+
Description-Content-Type: text/x-rst
|
|
11
|
+
License-File: LICENSE
|
|
12
|
+
Requires-Dist: aiohttp
|
|
13
|
+
Requires-Dist: pyperclip
|
|
14
|
+
|
|
15
|
+
Fetch using your browser.
|
|
16
|
+
|
|
17
|
+
Let the browser manage cookies for you.
|
|
18
|
+
|
|
19
|
+
⚠️ This project is a very simple implementation. Not tested thoroughly. Consider it a proof of concept.
|
|
20
|
+
|
|
21
|
+
Usage
|
|
22
|
+
-----
|
|
23
|
+
1. You'll run a Python script containing some code like this:
|
|
24
|
+
|
|
25
|
+
.. code-block:: python
|
|
26
|
+
|
|
27
|
+
from asyncio import gather, get_event_loop
|
|
28
|
+
|
|
29
|
+
from browserfetch import fetch, run_server
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
async def main():
|
|
33
|
+
response1, response2, reponse3 = await gather(
|
|
34
|
+
fetch('https://example.com/path1'),
|
|
35
|
+
fetch('https://example.com/image.png'),
|
|
36
|
+
fetch('https://example.com/path2'),
|
|
37
|
+
)
|
|
38
|
+
# do stuff with retrieved responses
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
loop = get_event_loop()
|
|
42
|
+
loop.create_task(main())
|
|
43
|
+
run_server(loop=loop)
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
2. Open your browser, goto http://example.com (perhaps solve a captcha and log in).
|
|
47
|
+
3. Copy the contents of `browserfetch.js`_ file and paste it in browser's console. (You can use a browser extensions like violentmonkey_/tampermonkey_ to do this step for you.)
|
|
48
|
+
|
|
49
|
+
That's it! Your Python script starts handling requests.
|
|
50
|
+
The browser tab should remain open of-coarse.
|
|
51
|
+
|
|
52
|
+
The server can handle multiple websocket connections from different websites simultaneously.
|
|
53
|
+
|
|
54
|
+
How it works
|
|
55
|
+
------------
|
|
56
|
+
``browserfetch`` communicates with your browser using a websocket. The ``fetch`` function just passes the request to browser and it is the browser that handles the actual request. Response data is sent back to Python using the same WebSocket connection.
|
|
57
|
+
|
|
58
|
+
Motivations
|
|
59
|
+
-----------
|
|
60
|
+
* `browser_cookie3 stopped working on Chrome-based browsers`_. There is a workaround: ShadowCopy, but it requires admin privilege.
|
|
61
|
+
* Another issue with browser_cookie's approach is that it retrieves cookies from cookie files, but these files are not updated instantly. Thus, you might have to wait or retry a few times before you can successfully access newly set cookies.
|
|
62
|
+
* ShadowCopying and File access are slow and inefficient operations.
|
|
63
|
+
|
|
64
|
+
Downsides
|
|
65
|
+
---------
|
|
66
|
+
* Setting up ``browserfetch`` is a more cumbersome since it requires running a Python server and also injecting a small script into the webpage. Using ``browser_cookie3`` might be a better choice if there are many websites that you need to communicate with.
|
|
67
|
+
|
|
68
|
+
.. _`browser_cookie3 stopped working on Chrome-based browsers`: https://github.com/borisbabic/browser_cookie3/issues/180
|
|
69
|
+
.. _tampermonkey: https://github.com/Tampermonkey/tampermonkey
|
|
70
|
+
.. _violentmonkey: https://github.com/violentmonkey/violentmonkey
|
|
71
|
+
.. _browserfetch.js: https://github.com/5j9/browserfetch/blob/master/browserfetch/browserfetch.js
|
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
LICENSE
|
|
2
|
+
README.rst
|
|
3
|
+
pyproject.toml
|
|
4
|
+
browserfetch/browserfetch.egg-info/PKG-INFO
|
|
5
|
+
browserfetch/browserfetch.egg-info/SOURCES.txt
|
|
6
|
+
browserfetch/browserfetch.egg-info/dependency_links.txt
|
|
7
|
+
browserfetch/browserfetch.egg-info/not-zip-safe
|
|
8
|
+
browserfetch/browserfetch.egg-info/requires.txt
|
|
9
|
+
browserfetch/browserfetch.egg-info/top_level.txt
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
|
|
@@ -57,13 +57,13 @@
|
|
|
57
57
|
evalled = eval(req['string']);
|
|
58
58
|
switch (evalled.constructor.name) {
|
|
59
59
|
case 'AsyncFunction':
|
|
60
|
-
evalled = await evalled;
|
|
60
|
+
evalled = await evalled(req['arg']);
|
|
61
61
|
break;
|
|
62
62
|
case 'Promise':
|
|
63
63
|
evalled = await evalled;
|
|
64
64
|
break;
|
|
65
65
|
case 'Function':
|
|
66
|
-
evalled = evalled();
|
|
66
|
+
evalled = evalled(req['arg']);
|
|
67
67
|
break;
|
|
68
68
|
}
|
|
69
69
|
resp = { 'result': evalled, 'event_id': req['event_id'] };
|
|
@@ -40,13 +40,21 @@ lint.extend-select = [
|
|
|
40
40
|
'FA', # flake8-future-annotations
|
|
41
41
|
'I', # isort
|
|
42
42
|
'UP', # pyupgrade
|
|
43
|
+
'RUF', # Ruff-specific rules (RUF)
|
|
43
44
|
]
|
|
44
45
|
lint.ignore = [
|
|
45
46
|
'E721', # Do not compare types, use `isinstance()`
|
|
47
|
+
'RUF001', # ambiguous-unicode-character-string
|
|
48
|
+
'RUF002', # ambiguous-unicode-character-docstring
|
|
49
|
+
'RUF003', # ambiguous-unicode-character-comment
|
|
50
|
+
'RUF012', # mutable-class-default
|
|
46
51
|
]
|
|
47
52
|
|
|
48
53
|
[tool.pytest.ini_options]
|
|
49
54
|
addopts = '--quiet --tb=short'
|
|
55
|
+
asyncio_mode = "auto"
|
|
56
|
+
asyncio_default_test_loop_scope = "session"
|
|
57
|
+
asyncio_default_fixture_loop_scope = "session"
|
|
50
58
|
|
|
51
59
|
[tool.pyright]
|
|
52
60
|
typeCheckingMode = 'standard'
|
|
@@ -60,3 +68,10 @@ reportInvalidStringEscapeSequence = false
|
|
|
60
68
|
reportConstantRedefinition = 'error'
|
|
61
69
|
reportTypeCommentUsage = 'warning'
|
|
62
70
|
reportUnnecessaryComparison = 'warning'
|
|
71
|
+
|
|
72
|
+
[dependency-groups]
|
|
73
|
+
dev = [
|
|
74
|
+
"playwright>=1.52.0",
|
|
75
|
+
"pytest>=8.4.0",
|
|
76
|
+
"pytest-asyncio>=1.0.0",
|
|
77
|
+
]
|
|
@@ -1,25 +0,0 @@
|
|
|
1
|
-
from contextlib import suppress
|
|
2
|
-
from unittest.mock import Mock, patch
|
|
3
|
-
|
|
4
|
-
from browserfetch import fetch
|
|
5
|
-
|
|
6
|
-
request_mock = Mock()
|
|
7
|
-
|
|
8
|
-
|
|
9
|
-
@patch('browserfetch._request', new=request_mock)
|
|
10
|
-
async def test_params():
|
|
11
|
-
with suppress(TypeError):
|
|
12
|
-
await fetch(
|
|
13
|
-
url='http://stackoverflow.com/search?q=question',
|
|
14
|
-
params={'lang': 'en', 'tag': 'python', 'q': True},
|
|
15
|
-
)
|
|
16
|
-
request_mock.assert_called_once_with(
|
|
17
|
-
None, # host
|
|
18
|
-
{
|
|
19
|
-
'action': 'fetch',
|
|
20
|
-
'url': 'http://stackoverflow.com/search?q=question&lang=en&tag=python&q=True',
|
|
21
|
-
'options': None,
|
|
22
|
-
'timeout': 95,
|
|
23
|
-
}, # data
|
|
24
|
-
None, # body
|
|
25
|
-
)
|
|
File without changes
|
|
File without changes
|
|
File without changes
|