browserfetch 0.2.0__tar.gz → 0.4.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: browserfetch
3
- Version: 0.2.0
3
+ Version: 0.4.0
4
4
  Summary: fetch in Python using your browser!
5
5
  License: GNU General Public License v3 (GPLv3)
6
6
  Project-URL: Homepage, https://github.com/5j9/browserfetch
@@ -26,21 +26,21 @@ Usage
26
26
 
27
27
  from asyncio import gather, get_event_loop
28
28
 
29
- from browserfetch import fetch, run_server
29
+ from browserfetch import fetch, get, post, run_server
30
30
 
31
31
 
32
32
  async def main():
33
33
  response1, response2, reponse3 = await gather(
34
- fetch('https://example.com/path1'),
34
+ get('https://example.com/path1', params={'a': 1}),
35
35
  fetch('https://example.com/image.png'),
36
- fetch('https://example.com/path2'),
36
+ post('https://example.com/path2', data={'a': 1}),
37
37
  )
38
38
  # do stuff with retrieved responses
39
39
 
40
40
 
41
41
  loop = get_event_loop()
42
- loop.create_task(main())
43
- run_server(loop=loop)
42
+ loop.create_task(start_server())
43
+ loop.run_until_complete(main())
44
44
 
45
45
 
46
46
  2. Open your browser, goto http://example.com (perhaps solve a captcha and log in).
@@ -63,7 +63,8 @@ Motivations
63
63
 
64
64
  Downsides
65
65
  ---------
66
- * Setting up ``browserfetch`` is a more cumbersome since it requires running a Python server and also injecting a small script into the webpage. Using ``browser_cookie3`` might be a better choice if there are many websites that you need to communicate with.
66
+ * Setting up ``browserfetch`` is more cumbersome since it requires running a Python server and also injecting a small script into the webpage. Using ``browser_cookie3`` might be a better choice if there are many websites that you need to communicate with.
67
+ * To run more than one server at the same time, you have to change the port number in both `start_server(port=)` call and in the corresponding JavaScript for each instance of the program.
67
68
 
68
69
  .. _`browser_cookie3 stopped working on Chrome-based browsers`: https://github.com/borisbabic/browser_cookie3/issues/180
69
70
  .. _tampermonkey: https://github.com/Tampermonkey/tampermonkey
@@ -12,21 +12,21 @@ Usage
12
12
 
13
13
  from asyncio import gather, get_event_loop
14
14
 
15
- from browserfetch import fetch, run_server
15
+ from browserfetch import fetch, get, post, run_server
16
16
 
17
17
 
18
18
  async def main():
19
19
  response1, response2, reponse3 = await gather(
20
- fetch('https://example.com/path1'),
20
+ get('https://example.com/path1', params={'a': 1}),
21
21
  fetch('https://example.com/image.png'),
22
- fetch('https://example.com/path2'),
22
+ post('https://example.com/path2', data={'a': 1}),
23
23
  )
24
24
  # do stuff with retrieved responses
25
25
 
26
26
 
27
27
  loop = get_event_loop()
28
- loop.create_task(main())
29
- run_server(loop=loop)
28
+ loop.create_task(start_server())
29
+ loop.run_until_complete(main())
30
30
 
31
31
 
32
32
  2. Open your browser, goto http://example.com (perhaps solve a captcha and log in).
@@ -49,7 +49,8 @@ Motivations
49
49
 
50
50
  Downsides
51
51
  ---------
52
- * Setting up ``browserfetch`` is a more cumbersome since it requires running a Python server and also injecting a small script into the webpage. Using ``browser_cookie3`` might be a better choice if there are many websites that you need to communicate with.
52
+ * Setting up ``browserfetch`` is more cumbersome since it requires running a Python server and also injecting a small script into the webpage. Using ``browser_cookie3`` might be a better choice if there are many websites that you need to communicate with.
53
+ * To run more than one server at the same time, you have to change the port number in both `start_server(port=)` call and in the corresponding JavaScript for each instance of the program.
53
54
 
54
55
  .. _`browser_cookie3 stopped working on Chrome-based browsers`: https://github.com/borisbabic/browser_cookie3/issues/180
55
56
  .. _tampermonkey: https://github.com/Tampermonkey/tampermonkey
@@ -0,0 +1,217 @@
1
+ __version__ = '0.4.0'
2
+
3
+ import atexit
4
+ from asyncio import Lock, Queue, gather, get_event_loop, wait_for
5
+ from dataclasses import dataclass
6
+ from json import dumps, loads
7
+ from logging import getLogger
8
+ from typing import Any
9
+ from urllib.parse import urlencode
10
+
11
+ from aiohttp.web import Application, RouteTableDef, WebSocketResponse
12
+ from aiohttp.web_runner import AppRunner, TCPSite
13
+
14
+ logger = getLogger(__name__)
15
+ # maps each domain the queue object of that domain
16
+ queues: dict[str, Queue] = {}
17
+ # maps each lock id to either the active lock or the resolved response
18
+ locks: dict[int, Lock | dict] = {}
19
+
20
+
21
+ class BrowserError(Exception):
22
+ pass
23
+
24
+
25
+ @dataclass(slots=True, weakref_slot=True)
26
+ class Response:
27
+ """
28
+ For the meaning of attributes see:
29
+ https://developer.mozilla.org/en-US/docs/Web/API/Response
30
+ """
31
+
32
+ body: bytes
33
+ ok: bool
34
+ redirected: bool
35
+ status: int
36
+ status_text: str
37
+ type: str
38
+ url: str
39
+ headers: dict
40
+
41
+ def text(self, encoding=None, errors='strict') -> str:
42
+ return self.body.decode(encoding or 'utf-8', errors)
43
+
44
+ def json(self, encoding=None, errors='strict'):
45
+ if encoding is None:
46
+ return loads(self.body)
47
+ return loads(self.text(encoding=encoding, errors=errors))
48
+
49
+
50
+ def extract_host(url: str) -> str:
51
+ return url.partition('//')[2].partition('/')[0]
52
+
53
+
54
+ async def send_requests(q, ws):
55
+ while True:
56
+ url, options, lock_id, timeout, body = await q.get()
57
+ request_blob = dumps(
58
+ {
59
+ 'url': url,
60
+ 'options': options,
61
+ 'lock_id': lock_id,
62
+ 'timeout': timeout,
63
+ }
64
+ ).encode()
65
+ if body is not None:
66
+ request_blob += b'\0' + body
67
+ await ws.send_bytes(request_blob)
68
+
69
+
70
+ async def receive_responses(ws: WebSocketResponse):
71
+ while True:
72
+ blob = await ws.receive_bytes()
73
+ json_blob, _, body = blob.partition(b'\0')
74
+ j = loads(json_blob)
75
+ j['body'] = body
76
+ lock_id = j.pop('lock_id')
77
+ try:
78
+ lock = locks[lock_id]
79
+ except KeyError: # lock has reached timeout already
80
+ continue
81
+ locks[lock_id] = j
82
+ lock.release()
83
+
84
+
85
+ routes = RouteTableDef()
86
+
87
+
88
+ @routes.get('/ws')
89
+ async def _(request):
90
+ ws = WebSocketResponse()
91
+ await ws.prepare(request)
92
+
93
+ host = await ws.receive_str()
94
+ logger.info('registering host %s', host)
95
+ q = queues.get(host) or queues.setdefault(host, Queue())
96
+
97
+ await gather(send_requests(q, ws), receive_responses(ws))
98
+
99
+
100
+ async def fetch(
101
+ url: str,
102
+ *,
103
+ params: dict = None,
104
+ body: bytes = None,
105
+ timeout: int | float = None,
106
+ options: dict = None,
107
+ host=None,
108
+ ) -> Response:
109
+ """Fetch using browser fetch API available on host.
110
+
111
+ :param url: the URL of the resource you want to fetch.
112
+ :param params: parameters to be url-encoded and added to url.
113
+ :param body: the body of the request (do not add to options).
114
+ :param timeout: timeout in seconds (do not add to options).
115
+ :param options: See https://developer.mozilla.org/en-US/docs/Web/API/fetch
116
+ :param host: `location.host` of the tab that is supposed to handle this
117
+ request.
118
+ :return: a dict of response values.
119
+ """
120
+ if params is not None:
121
+ url += urlencode(params)
122
+
123
+ if host is None:
124
+ host = extract_host(url)
125
+ q = queues.get(host) or queues.setdefault(host, Queue())
126
+ lock = Lock()
127
+ lock_id = id(lock)
128
+ locks[lock_id] = lock
129
+ await lock.acquire()
130
+
131
+ await q.put((url, options, lock_id, timeout, body))
132
+
133
+ try:
134
+ await wait_for(lock.acquire(), timeout)
135
+ except TimeoutError:
136
+ locks.pop(lock_id, None)
137
+ raise
138
+
139
+ j = locks.pop(lock_id)
140
+ if (err := j.get('error')) is not None:
141
+ raise BrowserError(err)
142
+
143
+ return Response(**j)
144
+
145
+
146
+ async def get(
147
+ url: str,
148
+ *,
149
+ params: dict = None,
150
+ options: dict = None,
151
+ host: str = None,
152
+ timeout: int | float = None,
153
+ ) -> Response:
154
+ if options is None:
155
+ options = {'method': 'GET'}
156
+ else:
157
+ options['method'] = 'GET'
158
+ return await fetch(
159
+ url, options=options, host=host, timeout=timeout, params=params
160
+ )
161
+
162
+
163
+ async def post(
164
+ url: str,
165
+ *,
166
+ params: dict = None,
167
+ body: bytes = None,
168
+ data: dict = None,
169
+ json=None,
170
+ timeout: int | float = None,
171
+ options: dict = None,
172
+ host: str = None,
173
+ ) -> Response:
174
+ if options is None:
175
+ options: dict[str, Any] = {'method': 'POST'}
176
+ else:
177
+ options['method'] = 'POST'
178
+
179
+ if json is not None:
180
+ assert body is None
181
+ body = dumps(json).encode()
182
+ headers = options.setdefault('headers', {})
183
+ headers['Content-Type'] = 'application/json'
184
+
185
+ if data is not None:
186
+ assert body is None
187
+ body = urlencode(data).encode()
188
+ headers = options.setdefault('headers', {})
189
+ headers['Content-Type'] = 'application/x-www-form-urlencoded'
190
+
191
+ return await fetch(
192
+ url,
193
+ options=options,
194
+ host=host,
195
+ timeout=timeout,
196
+ body=body,
197
+ params=params,
198
+ )
199
+
200
+
201
+ app = Application()
202
+ app.add_routes(routes)
203
+ app_runner = AppRunner(app)
204
+
205
+
206
+ @atexit.register
207
+ def shutdown_server():
208
+ loop = get_event_loop()
209
+ logger.info('waiting for app_runner.cleanup()')
210
+ loop.run_until_complete(app_runner.cleanup())
211
+
212
+
213
+ async def start_server(*, host='127.0.0.1', port=9404):
214
+ await app_runner.setup()
215
+ site = TCPSite(app_runner, host, port)
216
+ await site.start()
217
+ logger.info('server started at http://%s:%s', host, port)
@@ -2,10 +2,12 @@
2
2
  // @name browserfetch
3
3
  // @namespace https://github.com/5j9/browserfetch
4
4
  // @match https://example.com/
5
+ // @grant GM_registerMenuCommand
5
6
  // ==/UserScript==
6
7
  (() => {
7
8
  function connect() {
8
9
  var ws = new WebSocket("ws://127.0.0.1:9404/ws");
10
+ ws.binaryType = "arraybuffer";
9
11
 
10
12
  ws.onopen = () => {
11
13
  ws.send(location.host);
@@ -17,19 +19,30 @@
17
19
  };
18
20
 
19
21
  ws.onmessage = async (evt) => {
20
- var returnData, responseBlob;
21
- var j = JSON.parse(evt.data);
22
+ var returnData, responseBlob, body, jArray;
23
+ var requestArray = new Uint8Array(evt.data);
24
+ var nullIndex = requestArray.indexOf(0);
25
+ if (nullIndex === -1) {
26
+ body = null;
27
+ jArray = requestArray;
28
+ } else {
29
+ body = requestArray.slice(nullIndex + 1);
30
+ jArray = requestArray.slice(0, nullIndex)
31
+ }
32
+ var j = JSON.parse(new TextDecoder().decode(jArray));
33
+ var options = j['options'] || {};
34
+
22
35
  if (j['timeout']) {
23
- var signal = AbortSignal.timeout(j['timeout'] * 1000);
24
- if (j['options']) {
25
- j['options'].signal = signal;
26
- } else {
27
- j['options'] = { 'signal': signal };
28
- }
36
+ options.signal = AbortSignal.timeout(j['timeout'] * 1000);
37
+ }
38
+
39
+ if (body !== null) {
40
+ options.body = body;
29
41
  }
42
+
30
43
  try {
31
- var r = await fetch(j['url'], j['options']);
32
- var returnData = {
44
+ var r = await fetch(j['url'], options);
45
+ returnData = {
33
46
  'lock_id': j['lock_id'],
34
47
  'headers': Object.fromEntries([...r.headers]),
35
48
  'ok': r.ok,
@@ -51,5 +64,12 @@
51
64
  }
52
65
  };
53
66
 
54
- connect();
67
+ if (window.GM_registerMenuCommand) {
68
+ GM_registerMenuCommand(
69
+ 'connect to browserfetch',
70
+ connect
71
+ );
72
+ } else {
73
+ connect();
74
+ }
55
75
  })();
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: browserfetch
3
- Version: 0.2.0
3
+ Version: 0.4.0
4
4
  Summary: fetch in Python using your browser!
5
5
  License: GNU General Public License v3 (GPLv3)
6
6
  Project-URL: Homepage, https://github.com/5j9/browserfetch
@@ -26,21 +26,21 @@ Usage
26
26
 
27
27
  from asyncio import gather, get_event_loop
28
28
 
29
- from browserfetch import fetch, run_server
29
+ from browserfetch import fetch, get, post, run_server
30
30
 
31
31
 
32
32
  async def main():
33
33
  response1, response2, reponse3 = await gather(
34
- fetch('https://example.com/path1'),
34
+ get('https://example.com/path1', params={'a': 1}),
35
35
  fetch('https://example.com/image.png'),
36
- fetch('https://example.com/path2'),
36
+ post('https://example.com/path2', data={'a': 1}),
37
37
  )
38
38
  # do stuff with retrieved responses
39
39
 
40
40
 
41
41
  loop = get_event_loop()
42
- loop.create_task(main())
43
- run_server(loop=loop)
42
+ loop.create_task(start_server())
43
+ loop.run_until_complete(main())
44
44
 
45
45
 
46
46
  2. Open your browser, goto http://example.com (perhaps solve a captcha and log in).
@@ -63,7 +63,8 @@ Motivations
63
63
 
64
64
  Downsides
65
65
  ---------
66
- * Setting up ``browserfetch`` is a more cumbersome since it requires running a Python server and also injecting a small script into the webpage. Using ``browser_cookie3`` might be a better choice if there are many websites that you need to communicate with.
66
+ * Setting up ``browserfetch`` is more cumbersome since it requires running a Python server and also injecting a small script into the webpage. Using ``browser_cookie3`` might be a better choice if there are many websites that you need to communicate with.
67
+ * To run more than one server at the same time, you have to change the port number in both `start_server(port=)` call and in the corresponding JavaScript for each instance of the program.
67
68
 
68
69
  .. _`browser_cookie3 stopped working on Chrome-based browsers`: https://github.com/borisbabic/browser_cookie3/issues/180
69
70
  .. _tampermonkey: https://github.com/Tampermonkey/tampermonkey
@@ -1,128 +0,0 @@
1
- __version__ = '0.2.0'
2
-
3
- from asyncio import Lock, Queue, gather, wait_for
4
- from dataclasses import dataclass
5
- from json import loads
6
- from logging import getLogger
7
-
8
- from aiohttp.web import Application, RouteTableDef, WebSocketResponse, run_app
9
-
10
- logger = getLogger(__name__)
11
- # maps each domain the queue object of that domain
12
- queues: dict[str, Queue] = {}
13
- # maps each lock id to either the active lock or the resolved response
14
- locks: dict[int, Lock | dict] = {}
15
-
16
-
17
- class BrowserError(Exception):
18
- pass
19
-
20
-
21
- @dataclass(slots=True, weakref_slot=True)
22
- class FetchResponse:
23
- body: bytes
24
- ok: bool
25
- redirected: bool
26
- status: int
27
- status_text: str
28
- type: str
29
- url: str
30
- headers: dict
31
-
32
- def text(self, encoding=None, errors='strict') -> str:
33
- return self.body.decode(encoding or 'utf-8', errors)
34
-
35
- def json(self, encoding=None, errors='strict'):
36
- if encoding is None:
37
- return loads(self.body, errors=errors)
38
- return loads(self.text(encoding=encoding, errors=errors))
39
-
40
-
41
- def extract_host(url: str) -> str:
42
- return url.partition('//')[2].partition('/')[0]
43
-
44
-
45
- async def send_requests(q, ws):
46
- while True:
47
- url, options, lock, timeout = await q.get()
48
- await ws.send_json(
49
- {
50
- 'url': url,
51
- 'options': options,
52
- 'lock_id': id(lock),
53
- 'timeout': timeout,
54
- }
55
- )
56
-
57
-
58
- async def receive_responses(ws: WebSocketResponse):
59
- while True:
60
- blob = await ws.receive_bytes()
61
- json_blob, _, body = blob.partition(b'\0')
62
- j = loads(json_blob)
63
- j['body'] = body
64
- lock_id = j.pop('lock_id')
65
- try:
66
- lock = locks[lock_id]
67
- except KeyError: # lock has reached timeout already
68
- continue
69
- locks[lock_id] = j
70
- lock.release()
71
-
72
-
73
- routes = RouteTableDef()
74
-
75
-
76
- @routes.get('/ws')
77
- async def _(request):
78
- ws = WebSocketResponse()
79
- await ws.prepare(request)
80
-
81
- host = await ws.receive_str()
82
- logger.info('registering host %s', host)
83
- q = queues.get(host) or queues.setdefault(host, Queue())
84
-
85
- await gather(send_requests(q, ws), receive_responses(ws))
86
-
87
-
88
- async def fetch(
89
- url: str, options: dict = None, *, host=None, timeout: int | float = None
90
- ) -> FetchResponse:
91
- """fetch using browser fetch API available on host.
92
-
93
- :param url: the URL of the resource you want to fetch.
94
- :param options: See https://developer.mozilla.org/en-US/docs/Web/API/fetch
95
- :param host: `location.host` of the tab that is supposed to handle this
96
- request.
97
- :param timeout: timeout in seconds.
98
- :return: a dict of response values.
99
- """
100
- if host is None:
101
- host = extract_host(url)
102
- q = queues.get(host) or queues.setdefault(host, Queue())
103
- lock = Lock()
104
- lock_id = id(lock)
105
- locks[lock_id] = lock
106
- await lock.acquire()
107
-
108
- await q.put((url, options, lock, timeout))
109
-
110
- try:
111
- await wait_for(lock.acquire(), timeout)
112
- except TimeoutError:
113
- locks.pop(lock_id, None)
114
- raise
115
-
116
- j = locks.pop(lock_id)
117
- if (err := j.get('error')) is not None:
118
- raise BrowserError(err)
119
-
120
- return FetchResponse(**j)
121
-
122
-
123
- app = Application()
124
- app.add_routes(routes)
125
-
126
-
127
- def run_server(*, host='127.0.0.1', port=9404, loop=None):
128
- run_app(app, host=host, port=port, loop=loop)
File without changes
File without changes