codetac 0.1.0 → 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +10 -5
- package/readme.md +81 -14
- package/src/ai.mjs +4 -2
- package/src/boundaries.mjs +12 -9
- package/src/browser/bar.js +5 -3
- package/src/cli.mjs +251 -33
- package/src/detect-python.mjs +286 -0
- package/src/detect.mjs +21 -6
- package/src/diagnose.mjs +35 -5
- package/src/digest.mjs +18 -4
- package/src/page.mjs +19 -7
- package/src/panel.mjs +56 -14
- package/src/python/codetac_py/__init__.py +88 -0
- package/src/python/codetac_py/boundaries.py +435 -0
- package/src/python/codetac_py/capture.py +351 -0
- package/src/python/codetac_py/context.py +50 -0
- package/src/python/codetac_py/detail.py +333 -0
- package/src/python/codetac_py/files.py +135 -0
- package/src/python/codetac_py/frameworks.py +72 -0
- package/src/python/codetac_py/hooks.py +72 -0
- package/src/python/codetac_py/jinja_map.py +132 -0
- package/src/python/codetac_py/network.py +704 -0
- package/src/python/codetac_py/page.py +513 -0
- package/src/python/codetac_py/project.py +60 -0
- package/src/python/codetac_py/redact.py +104 -0
- package/src/python/codetac_py/servers.py +514 -0
- package/src/python/codetac_py/sitecustomize.py +62 -0
- package/src/python/codetac_py/writer.py +237 -0
- package/src/python/probe.py +99 -0
- package/src/recording.mjs +22 -4
- package/src/runtime.mjs +13 -1
- package/src/sentences.mjs +4 -0
- package/src/store.mjs +61 -9
|
@@ -0,0 +1,513 @@
|
|
|
1
|
+
"""Browser side, served by the observed app's own Python server, as src/page.mjs
|
|
2
|
+
does for Node: HTML pages get a <script src="/__codetac/bar.js">, and the
|
|
3
|
+
/__codetac/ routes are answered here without ever reaching the app. It works
|
|
4
|
+
at the protocol level (WSGI and ASGI), for any framework on top of them.
|
|
5
|
+
|
|
6
|
+
- the action of a request: the x-codetac-action header set by the page script
|
|
7
|
+
on fetch and XHR, or the short cookie it sets before a full navigation;
|
|
8
|
+
- /__codetac/bar.js serves the same bar.js as the Node side (CODETAC_BAR_JS),
|
|
9
|
+
and /__codetac/events records the page's actions (`browser-action`);
|
|
10
|
+
- the script tag goes into complete HTML documents: known length (or a body
|
|
11
|
+
given whole), not compressed, not a file, 1 MB at most. The length is
|
|
12
|
+
adjusted, and validators (ETag, Last-Modified) are dropped only when the
|
|
13
|
+
body changes. Streams (no length), files and compressed bodies pass
|
|
14
|
+
untouched. Error pages (500) get the bar too.
|
|
15
|
+
|
|
16
|
+
Keep it importable on old Pythons (3.8+): the minimal mode has the bar too.
|
|
17
|
+
"""
|
|
18
|
+
import json
|
|
19
|
+
import math
|
|
20
|
+
import os
|
|
21
|
+
import re
|
|
22
|
+
|
|
23
|
+
PREFIX = '/__codetac/'
|
|
24
|
+
ACTION_HEADER = 'x-codetac-action'
|
|
25
|
+
INTERNAL_HEADER = 'x-codetac-internal'
|
|
26
|
+
_ACTION_ID = re.compile(r'^[A-Za-z0-9_-]{6,40}$')
|
|
27
|
+
_ACTION_NUMBER = re.compile(r'^\d{1,5}$')
|
|
28
|
+
_ACTION_COOKIE = re.compile(r'(?:^|;\s*)codetac_action=([A-Za-z0-9_-]{6,40})(?:;|$)')
|
|
29
|
+
MAX_EVENT_BYTES = 64 * 1024
|
|
30
|
+
MAX_PAGE_BYTES = 1024 * 1024
|
|
31
|
+
TAG = b'<script src="/__codetac/bar.js" data-codetac=""></script>'
|
|
32
|
+
_HEAD = re.compile(br'<head(?:\s[^>]*)?>', re.IGNORECASE)
|
|
33
|
+
_BODY = re.compile(br'<body[\s>]', re.IGNORECASE)
|
|
34
|
+
_LEADING = br'(?:\s|<!--.*?-->)*'
|
|
35
|
+
_OPENING_HTML = re.compile(_LEADING + br'(?:<!doctype html[^>]*>' + _LEADING + br')?<html(?:\s[^>]*)?>', re.IGNORECASE | re.DOTALL)
|
|
36
|
+
_OPENING_DOCTYPE = re.compile(_LEADING + br'<!doctype html[^>]*>', re.IGNORECASE | re.DOTALL)
|
|
37
|
+
_HTML = re.compile(r'text/html', re.IGNORECASE)
|
|
38
|
+
DROPPED = ('content-length', 'etag', 'last-modified')
|
|
39
|
+
BAR_JS = os.path.realpath(os.path.join(os.path.dirname(os.path.abspath(__file__)), '..', '..', 'browser', 'bar.js'))
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def enabled():
|
|
43
|
+
return os.environ.get('CODETAC_PAGE') != '0'
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def panel_url():
|
|
47
|
+
"""src/page.mjs panelUrl."""
|
|
48
|
+
return os.environ.get('CODETAC_PANEL_URL') or 'http://127.0.0.1:%s' % (os.environ.get('CODETAC_PANEL_PORT') or 4000)
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
_script = None
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def bar_script():
|
|
55
|
+
global _script
|
|
56
|
+
if _script is None:
|
|
57
|
+
config = {'panel': panel_url(), 'run': os.environ.get('CODETAC_RUN', '')}
|
|
58
|
+
with open(os.environ.get('CODETAC_BAR_JS') or BAR_JS, 'rb') as file:
|
|
59
|
+
source = file.read()
|
|
60
|
+
_script = ('window.__CODETAC_CONFIG__=%s;\n' % json.dumps(config, separators=(',', ':'))).encode('utf-8') + source
|
|
61
|
+
return _script
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def action_of(header, fetch_mode, cookie):
|
|
65
|
+
"""src/page.mjs actionOf: {'action', 'actionRequest'} or None."""
|
|
66
|
+
if isinstance(header, str):
|
|
67
|
+
parts = header.split('.')
|
|
68
|
+
if _ACTION_ID.match(parts[0]):
|
|
69
|
+
number = parts[1] if len(parts) > 1 else ''
|
|
70
|
+
return {'action': parts[0], 'actionRequest': int(number) if _ACTION_NUMBER.match(number) else None}
|
|
71
|
+
if fetch_mode == 'navigate':
|
|
72
|
+
match = _ACTION_COOKIE.search(cookie or '')
|
|
73
|
+
if match:
|
|
74
|
+
return {'action': match.group(1), 'actionRequest': None}
|
|
75
|
+
return None
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def wants_page(method, destination, accept):
|
|
79
|
+
"""src/page.mjs wantsPage: top-level HTML documents."""
|
|
80
|
+
if method != 'GET':
|
|
81
|
+
return False
|
|
82
|
+
if destination:
|
|
83
|
+
return destination == 'document'
|
|
84
|
+
return bool(_HTML.search(accept or ''))
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
def insert_tag(body, final):
|
|
88
|
+
"""src/page.mjs tagPosition: after <head ...>; in a whole document without a head,
|
|
89
|
+
before <body>, or else after <html ...> or <!doctype html> (werkzeug's error pages have
|
|
90
|
+
neither head nor body). None while more of the body is needed; the body unchanged when
|
|
91
|
+
there is nowhere to put it (fragments)."""
|
|
92
|
+
head = _HEAD.search(body)
|
|
93
|
+
if head:
|
|
94
|
+
at = head.end()
|
|
95
|
+
elif not final:
|
|
96
|
+
return None
|
|
97
|
+
else:
|
|
98
|
+
match = _BODY.search(body)
|
|
99
|
+
opening = match is None and (_OPENING_HTML.match(body) or _OPENING_DOCTYPE.match(body))
|
|
100
|
+
if match:
|
|
101
|
+
at = match.start()
|
|
102
|
+
elif opening:
|
|
103
|
+
at = opening.end()
|
|
104
|
+
else:
|
|
105
|
+
return body
|
|
106
|
+
return body[:at] + TAG + body[at:]
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
def injectable(status, headers):
|
|
110
|
+
"""Whether the response is an HTML document the tag can go into: `headers` as
|
|
111
|
+
(lower-case name, value) pairs. Returns the declared length (or -1), or None."""
|
|
112
|
+
if not 200 <= status < 600 or status in (204, 206, 304):
|
|
113
|
+
return None
|
|
114
|
+
kind = length = None
|
|
115
|
+
for name, value in headers:
|
|
116
|
+
if name == 'content-type':
|
|
117
|
+
kind = value
|
|
118
|
+
elif name in ('content-encoding', 'content-range') and value.strip().lower() not in ('', 'identity'):
|
|
119
|
+
return None
|
|
120
|
+
elif name == 'content-length':
|
|
121
|
+
try:
|
|
122
|
+
length = int(value.strip())
|
|
123
|
+
except ValueError:
|
|
124
|
+
return None
|
|
125
|
+
if not kind or not _HTML.search(kind):
|
|
126
|
+
return None
|
|
127
|
+
if length is None:
|
|
128
|
+
return -1
|
|
129
|
+
return length if length <= MAX_PAGE_BYTES else None
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
def adjusted(headers, length):
|
|
133
|
+
"""Headers (name, value), as given, for a body that changed to `length` bytes."""
|
|
134
|
+
result = [(name, value) for name, value in headers if name.lower() not in DROPPED]
|
|
135
|
+
result.append(('Content-Length', str(length)))
|
|
136
|
+
return result
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
# Browser actions -----------------------------------------------------------------
|
|
140
|
+
|
|
141
|
+
def _text(value, maximum=120):
|
|
142
|
+
if not isinstance(value, str):
|
|
143
|
+
return None
|
|
144
|
+
return ' '.join(value.split())[:maximum]
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
def _number(value):
|
|
148
|
+
if isinstance(value, bool) or not isinstance(value, (int, float)):
|
|
149
|
+
return None
|
|
150
|
+
return value if math.isfinite(value) else None
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
def _frames(items):
|
|
154
|
+
if not isinstance(items, list):
|
|
155
|
+
return None
|
|
156
|
+
result = []
|
|
157
|
+
for frame in items[:16]:
|
|
158
|
+
frame = frame if isinstance(frame, dict) else {}
|
|
159
|
+
item = {'fn': _text(frame.get('fn'), 80), 'url': _text(frame.get('url'), 500), 'line': _number(frame.get('line')),
|
|
160
|
+
'column': _number(frame.get('column')), 'file': _text(frame.get('file'), 500)}
|
|
161
|
+
if item['url'] or item['file']:
|
|
162
|
+
result.append(item)
|
|
163
|
+
return result
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
def _names(items):
|
|
167
|
+
if not isinstance(items, list):
|
|
168
|
+
return None
|
|
169
|
+
result = []
|
|
170
|
+
for item in items[:40]:
|
|
171
|
+
item = item if isinstance(item, dict) else {}
|
|
172
|
+
entry = {'name': _text(item.get('name'), 80), 'count': _number(item.get('count')), 'frames': _frames(item.get('frames'))}
|
|
173
|
+
if entry['name']:
|
|
174
|
+
result.append(entry)
|
|
175
|
+
return result
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
def _path_of(raw):
|
|
179
|
+
if not isinstance(raw, str):
|
|
180
|
+
return {}
|
|
181
|
+
from .servers import split_target
|
|
182
|
+
try:
|
|
183
|
+
path, keys = split_target(raw)
|
|
184
|
+
except Exception:
|
|
185
|
+
return {'path': '[inválido]', 'queryKeys': []}
|
|
186
|
+
return {'path': path, 'queryKeys': keys}
|
|
187
|
+
|
|
188
|
+
|
|
189
|
+
def _clean(value):
|
|
190
|
+
"""Drops the fields that are None, as JSON.stringify drops undefined."""
|
|
191
|
+
if isinstance(value, dict):
|
|
192
|
+
return {key: _clean(item) for key, item in value.items() if item is not None}
|
|
193
|
+
if isinstance(value, list):
|
|
194
|
+
return [_clean(item) for item in value]
|
|
195
|
+
return value
|
|
196
|
+
|
|
197
|
+
|
|
198
|
+
def browser_action(event):
|
|
199
|
+
"""src/page.mjs recordBrowserAction: only known fields; URLs lose their query values.
|
|
200
|
+
The writer redacts the event before it is written."""
|
|
201
|
+
if not isinstance(event, dict) or not _ACTION_ID.match(str(event.get('actionId'))):
|
|
202
|
+
return None
|
|
203
|
+
trigger = event.get('trigger') if isinstance(event.get('trigger'), dict) else None
|
|
204
|
+
shaped = None
|
|
205
|
+
if trigger:
|
|
206
|
+
element = trigger.get('element') if isinstance(trigger.get('element'), dict) else {}
|
|
207
|
+
component = trigger.get('component') if isinstance(trigger.get('component'), dict) else None
|
|
208
|
+
handler = trigger.get('handler') if isinstance(trigger.get('handler'), dict) else None
|
|
209
|
+
owners = None
|
|
210
|
+
if component and isinstance(component.get('owners'), list):
|
|
211
|
+
owners = [{'name': _text(owner.get('name'), 80), 'frames': _frames(owner.get('frames'))}
|
|
212
|
+
for owner in component['owners'][:6] if isinstance(owner, dict)]
|
|
213
|
+
owners = [owner for owner in owners if owner['name']]
|
|
214
|
+
shaped = {
|
|
215
|
+
'event': _text(trigger.get('event'), 20),
|
|
216
|
+
'element': {'tag': _text(element.get('tag'), 20), 'type': _text(element.get('type'), 20), 'role': _text(element.get('role'), 30),
|
|
217
|
+
'text': _text(element.get('text'), 80), 'label': _text(element.get('label'), 80), 'name': _text(element.get('name'), 60),
|
|
218
|
+
'id': _text(element.get('id'), 60), 'href': _path_of(element['href']).get('path') if element.get('href') else None},
|
|
219
|
+
'component': {'name': _text(component.get('name'), 80), 'owners': owners, 'frames': _frames(component.get('frames'))} if component else None,
|
|
220
|
+
'handler': {'name': _text(handler.get('name'), 80), 'prop': _text(handler.get('prop'), 30),
|
|
221
|
+
'source': _text(handler.get('source'), 20)} if handler else None,
|
|
222
|
+
}
|
|
223
|
+
requests = []
|
|
224
|
+
for item in event.get('requests')[:200] if isinstance(event.get('requests'), list) else []:
|
|
225
|
+
item = item if isinstance(item, dict) else {}
|
|
226
|
+
same = bool(item.get('sameOrigin'))
|
|
227
|
+
entry = {'n': _number(item.get('n')), 'kind': _text(item.get('kind'), 20), 'method': _text(item.get('method'), 10)}
|
|
228
|
+
entry.update(_path_of(item.get('url')))
|
|
229
|
+
entry.update({'sameOrigin': same, 'host': None if same else _text(item.get('host'), 200),
|
|
230
|
+
'startMs': _number(item.get('startMs')), 'durationMs': _number(item.get('durationMs')), 'status': _number(item.get('status')),
|
|
231
|
+
'error': True if item.get('error') else None, 'frames': _frames(item.get('frames'))})
|
|
232
|
+
requests.append(entry)
|
|
233
|
+
screen = event.get('screen') if isinstance(event.get('screen'), dict) else None
|
|
234
|
+
navigations = None
|
|
235
|
+
if isinstance(event.get('navigations'), list):
|
|
236
|
+
navigations = []
|
|
237
|
+
for item in event['navigations'][:20]:
|
|
238
|
+
item = item if isinstance(item, dict) else {}
|
|
239
|
+
entry = {'kind': _text(item.get('kind'), 20)}
|
|
240
|
+
entry.update(_path_of(item.get('to')))
|
|
241
|
+
entry['atMs'] = _number(item.get('atMs'))
|
|
242
|
+
navigations.append(entry)
|
|
243
|
+
segment = _number(event.get('segment'))
|
|
244
|
+
return _clean({
|
|
245
|
+
'type': 'browser-action', 'actionId': event['actionId'], 'segment': 1 if segment is None else segment,
|
|
246
|
+
'origin': _text(event.get('origin'), 200), 'page': _path_of(event.get('page')),
|
|
247
|
+
'startedAt': _number(event.get('startedAt')), 'durationMs': _number(event.get('durationMs')), 'closedBy': _text(event.get('closedBy'), 40),
|
|
248
|
+
'trigger': shaped, 'requests': requests,
|
|
249
|
+
'screen': {'added': _number(screen.get('added')), 'removed': _number(screen.get('removed')), 'text': _number(screen.get('text')),
|
|
250
|
+
'attributes': _number(screen.get('attributes')), 'title': True if screen.get('title') else None,
|
|
251
|
+
'stateChanged': _names(screen.get('stateChanged')), 'mounted': _names(screen.get('mounted')),
|
|
252
|
+
'unmounted': _names(screen.get('unmounted'))} if screen else None,
|
|
253
|
+
'scripts': [url for url in (_text(url, 500) for url in event['scripts'][:80]) if url] if isinstance(event.get('scripts'), list) else None,
|
|
254
|
+
'navigations': navigations,
|
|
255
|
+
})
|
|
256
|
+
|
|
257
|
+
|
|
258
|
+
def own_route(method, path, origin, host, read_body):
|
|
259
|
+
"""Answers a /__codetac/ request: (status, headers, body). `read_body(limit)` returns
|
|
260
|
+
the request body, or None when it is longer than the limit."""
|
|
261
|
+
if path == PREFIX + 'bar.js' and method == 'GET':
|
|
262
|
+
return 200, [('Content-Type', 'text/javascript; charset=utf-8'), ('Cache-Control', 'no-store')], bar_script()
|
|
263
|
+
if path == PREFIX + 'events' and method == 'POST':
|
|
264
|
+
# Only the page itself may report actions: another site open in the
|
|
265
|
+
# browser cannot write into the recording (browsers send Origin here).
|
|
266
|
+
same = not origin
|
|
267
|
+
if not same:
|
|
268
|
+
from urllib.parse import urlsplit
|
|
269
|
+
try:
|
|
270
|
+
same = urlsplit(origin).netloc == host
|
|
271
|
+
except ValueError:
|
|
272
|
+
same = False
|
|
273
|
+
if not same:
|
|
274
|
+
return 403, [('Content-Type', 'text/plain; charset=utf-8')], 'CodeTAC: origem recusada.'.encode('utf-8')
|
|
275
|
+
body = read_body(MAX_EVENT_BYTES)
|
|
276
|
+
if body is None:
|
|
277
|
+
return 413, [('Cache-Control', 'no-store')], b''
|
|
278
|
+
try:
|
|
279
|
+
event = browser_action(json.loads(body.decode('utf-8')))
|
|
280
|
+
if event is not None:
|
|
281
|
+
import codetac_py
|
|
282
|
+
codetac_py.writer.emit(event)
|
|
283
|
+
except Exception:
|
|
284
|
+
pass
|
|
285
|
+
return 204, [('Cache-Control', 'no-store')], b''
|
|
286
|
+
return 404, [('Content-Type', 'text/plain; charset=utf-8')], 'CodeTAC: rota desconhecida.'.encode('utf-8')
|
|
287
|
+
|
|
288
|
+
|
|
289
|
+
def emit_page(path):
|
|
290
|
+
import codetac_py
|
|
291
|
+
codetac_py.writer.emit({'type': 'page', 'path': path})
|
|
292
|
+
|
|
293
|
+
|
|
294
|
+
# WSGI ------------------------------------------------------------------------------
|
|
295
|
+
|
|
296
|
+
def wsgi_own_route(environ, start_response):
|
|
297
|
+
def read_body(limit):
|
|
298
|
+
try:
|
|
299
|
+
length = int(environ.get('CONTENT_LENGTH') or 0)
|
|
300
|
+
except ValueError:
|
|
301
|
+
length = 0
|
|
302
|
+
if length > limit:
|
|
303
|
+
return None
|
|
304
|
+
stream = environ.get('wsgi.input')
|
|
305
|
+
return stream.read(length) if stream is not None and length > 0 else b''
|
|
306
|
+
|
|
307
|
+
status, headers, body = own_route(environ.get('REQUEST_METHOD', 'GET'), environ.get('PATH_INFO', ''),
|
|
308
|
+
environ.get('HTTP_ORIGIN'), environ.get('HTTP_HOST'), read_body)
|
|
309
|
+
reasons = {200: 'OK', 204: 'No Content', 403: 'Forbidden', 404: 'Not Found', 413: 'Payload Too Large'}
|
|
310
|
+
start_response('%d %s' % (status, reasons[status]), headers + [('Content-Length', str(len(body)))])
|
|
311
|
+
return [body]
|
|
312
|
+
|
|
313
|
+
|
|
314
|
+
class WsgiPage(object):
|
|
315
|
+
"""The response of a page request: start_response is held until the body shows whether
|
|
316
|
+
the tag goes in. The server sends nothing before the first chunk anyway (PEP 3333)."""
|
|
317
|
+
|
|
318
|
+
def __init__(self, start_response):
|
|
319
|
+
self.start_response = start_response
|
|
320
|
+
self.held = None # (status, headers, exc_info) while undecided
|
|
321
|
+
self.mode = None # None: undecided; 'pass'; 'buffer'
|
|
322
|
+
self.length = -1
|
|
323
|
+
|
|
324
|
+
def start(self, status, headers, exc_info=None):
|
|
325
|
+
if self.mode == 'pass':
|
|
326
|
+
return self.start_response(status, headers, exc_info) if exc_info else self.start_response(status, headers)
|
|
327
|
+
try:
|
|
328
|
+
code = int(str(status).split(' ', 1)[0])
|
|
329
|
+
length = injectable(code, [(name.lower(), value) for name, value in headers])
|
|
330
|
+
except Exception:
|
|
331
|
+
length = None
|
|
332
|
+
if length is None:
|
|
333
|
+
self.mode = 'pass'
|
|
334
|
+
return self.start_response(status, headers, exc_info) if exc_info else self.start_response(status, headers)
|
|
335
|
+
self.mode = 'buffer'
|
|
336
|
+
self.length = length
|
|
337
|
+
self.held = (status, headers, exc_info)
|
|
338
|
+
return self._write
|
|
339
|
+
|
|
340
|
+
def _write(self, data):
|
|
341
|
+
# The legacy write() callable: the headers must go now, unchanged.
|
|
342
|
+
self._release(None)
|
|
343
|
+
return self.write(data)
|
|
344
|
+
|
|
345
|
+
def _release(self, body):
|
|
346
|
+
"""Sends the held headers, for `body` (None: the original response goes on unchanged)."""
|
|
347
|
+
status, headers, exc_info = self.held
|
|
348
|
+
self.held = None
|
|
349
|
+
self.mode = 'pass'
|
|
350
|
+
if body is not None:
|
|
351
|
+
headers = adjusted(headers, len(body))
|
|
352
|
+
self.write = self.start_response(status, headers, exc_info) if exc_info else self.start_response(status, headers)
|
|
353
|
+
|
|
354
|
+
def body(self, result):
|
|
355
|
+
"""The iterable to return to the server."""
|
|
356
|
+
if self.mode == 'pass':
|
|
357
|
+
return result
|
|
358
|
+
# Files (werkzeug's and wsgiref's FileWrapper, the server's file_wrapper): untouched.
|
|
359
|
+
if type(result).__name__ == 'FileWrapper' or hasattr(result, 'filelike'):
|
|
360
|
+
if self.mode == 'buffer':
|
|
361
|
+
self._release(None)
|
|
362
|
+
return result
|
|
363
|
+
return _Iterated(self, result)
|
|
364
|
+
|
|
365
|
+
|
|
366
|
+
class _Iterated(object):
|
|
367
|
+
"""The body of a page request; close() is the original body's."""
|
|
368
|
+
|
|
369
|
+
def __init__(self, page, result):
|
|
370
|
+
self.page = page
|
|
371
|
+
self.result = result
|
|
372
|
+
self.iterator = None
|
|
373
|
+
|
|
374
|
+
def __iter__(self):
|
|
375
|
+
if self.iterator is None:
|
|
376
|
+
self.iterator = self._iterate()
|
|
377
|
+
return self.iterator
|
|
378
|
+
|
|
379
|
+
def close(self):
|
|
380
|
+
close = getattr(self.result, 'close', None)
|
|
381
|
+
if close is not None:
|
|
382
|
+
close()
|
|
383
|
+
|
|
384
|
+
def _iterate(self):
|
|
385
|
+
# start_response may be called while the body is being iterated
|
|
386
|
+
# (generators, the werkzeug debugger).
|
|
387
|
+
page, result = self.page, self.result
|
|
388
|
+
piecewise = not isinstance(result, (list, tuple))
|
|
389
|
+
chunks = []
|
|
390
|
+
size = 0
|
|
391
|
+
for chunk in result:
|
|
392
|
+
if page.mode != 'buffer':
|
|
393
|
+
if chunks:
|
|
394
|
+
yield b''.join(chunks)
|
|
395
|
+
chunks = []
|
|
396
|
+
yield chunk
|
|
397
|
+
continue
|
|
398
|
+
if page.length < 0 and piecewise:
|
|
399
|
+
# A stream (no length, given piece by piece): untouched.
|
|
400
|
+
page._release(None)
|
|
401
|
+
yield chunk
|
|
402
|
+
continue
|
|
403
|
+
chunks.append(chunk)
|
|
404
|
+
size += len(chunk)
|
|
405
|
+
if size > MAX_PAGE_BYTES:
|
|
406
|
+
page._release(None)
|
|
407
|
+
yield b''.join(chunks)
|
|
408
|
+
chunks = []
|
|
409
|
+
if page.mode == 'buffer':
|
|
410
|
+
body = b''.join(chunks)
|
|
411
|
+
changed = insert_tag(body, True)
|
|
412
|
+
if page.length >= 0 and len(body) != page.length or changed == body:
|
|
413
|
+
page._release(None)
|
|
414
|
+
else:
|
|
415
|
+
page._release(changed)
|
|
416
|
+
body = changed
|
|
417
|
+
yield body
|
|
418
|
+
elif chunks:
|
|
419
|
+
yield b''.join(chunks)
|
|
420
|
+
|
|
421
|
+
|
|
422
|
+
# ASGI ------------------------------------------------------------------------------
|
|
423
|
+
|
|
424
|
+
async def asgi_own_route(scope, receive, send):
|
|
425
|
+
headers = dict((key.decode('latin-1').lower(), value.decode('latin-1')) for key, value in scope.get('headers') or [])
|
|
426
|
+
received = []
|
|
427
|
+
|
|
428
|
+
async def read_all():
|
|
429
|
+
size = 0
|
|
430
|
+
while True:
|
|
431
|
+
message = await receive()
|
|
432
|
+
if message.get('type') != 'http.request':
|
|
433
|
+
return None
|
|
434
|
+
chunk = message.get('body') or b''
|
|
435
|
+
size += len(chunk)
|
|
436
|
+
if size <= MAX_EVENT_BYTES:
|
|
437
|
+
received.append(chunk)
|
|
438
|
+
if not message.get('more_body'):
|
|
439
|
+
return size
|
|
440
|
+
|
|
441
|
+
size = await read_all() if scope.get('method') == 'POST' else 0
|
|
442
|
+
status, answer, body = own_route(scope.get('method', 'GET'), scope.get('path', ''), headers.get('origin'), headers.get('host'),
|
|
443
|
+
lambda limit: None if size is None or size > limit else b''.join(received))
|
|
444
|
+
await send({'type': 'http.response.start', 'status': status,
|
|
445
|
+
'headers': [(name.lower().encode('latin-1'), value.encode('latin-1')) for name, value in answer]
|
|
446
|
+
+ [(b'content-length', str(len(body)).encode())]})
|
|
447
|
+
await send({'type': 'http.response.body', 'body': body})
|
|
448
|
+
|
|
449
|
+
|
|
450
|
+
class AsgiPage(object):
|
|
451
|
+
"""The response of a page request, for ASGI: http.response.start is held until the
|
|
452
|
+
body shows whether the tag goes in."""
|
|
453
|
+
|
|
454
|
+
def __init__(self, send):
|
|
455
|
+
self.send = send
|
|
456
|
+
self.mode = None
|
|
457
|
+
self.start = None
|
|
458
|
+
self.length = -1
|
|
459
|
+
self.chunks = []
|
|
460
|
+
self.size = 0
|
|
461
|
+
|
|
462
|
+
async def __call__(self, message):
|
|
463
|
+
kind = message.get('type')
|
|
464
|
+
if self.mode == 'pass' or kind not in ('http.response.start', 'http.response.body'):
|
|
465
|
+
if self.mode == 'buffer':
|
|
466
|
+
# Files (http.response.pathsend, zerocopy) or anything else: untouched.
|
|
467
|
+
await self._release(None)
|
|
468
|
+
return await self.send(message)
|
|
469
|
+
if kind == 'http.response.start':
|
|
470
|
+
try:
|
|
471
|
+
self.length = injectable(int(message.get('status')), [(key.decode('latin-1').lower(), value.decode('latin-1'))
|
|
472
|
+
for key, value in message.get('headers') or []])
|
|
473
|
+
except Exception:
|
|
474
|
+
self.length = None
|
|
475
|
+
if self.length is None or self.length < 0:
|
|
476
|
+
# Without a length it is a stream (StreamingResponse): untouched.
|
|
477
|
+
self.mode = 'pass'
|
|
478
|
+
return await self.send(message)
|
|
479
|
+
self.mode = 'buffer'
|
|
480
|
+
self.start = message
|
|
481
|
+
return None
|
|
482
|
+
body = message.get('body') or b''
|
|
483
|
+
self.chunks.append(body)
|
|
484
|
+
self.size += len(body)
|
|
485
|
+
if message.get('more_body', False):
|
|
486
|
+
if self.size > MAX_PAGE_BYTES:
|
|
487
|
+
await self._release(None)
|
|
488
|
+
return None
|
|
489
|
+
whole = b''.join(self.chunks)
|
|
490
|
+
self.chunks = []
|
|
491
|
+
changed = insert_tag(whole, True)
|
|
492
|
+
if len(whole) != self.length or changed == whole:
|
|
493
|
+
await self._release(None, whole)
|
|
494
|
+
else:
|
|
495
|
+
await self._release(changed)
|
|
496
|
+
|
|
497
|
+
async def _release(self, changed, whole=None):
|
|
498
|
+
start, self.start = self.start, None
|
|
499
|
+
self.mode = 'pass'
|
|
500
|
+
if changed is not None:
|
|
501
|
+
headers = [(key, value) for key, value in start.get('headers') or [] if key.decode('latin-1').lower() not in DROPPED]
|
|
502
|
+
headers.append((b'content-length', str(len(changed)).encode()))
|
|
503
|
+
start = dict(start, headers=headers)
|
|
504
|
+
await self.send(start)
|
|
505
|
+
await self.send({'type': 'http.response.body', 'body': changed})
|
|
506
|
+
return
|
|
507
|
+
await self.send(start)
|
|
508
|
+
if whole is not None:
|
|
509
|
+
await self.send({'type': 'http.response.body', 'body': whole})
|
|
510
|
+
return
|
|
511
|
+
chunks, self.chunks = self.chunks, []
|
|
512
|
+
for chunk in chunks:
|
|
513
|
+
await self.send({'type': 'http.response.body', 'body': chunk, 'more_body': True})
|
|
@@ -0,0 +1,60 @@
|
|
|
1
|
+
"""Which files are the project's own code: used by the capture (functions to
|
|
2
|
+
follow) and by the file boundaries (who is writing).
|
|
3
|
+
|
|
4
|
+
Keep it importable on old Pythons (3.8+): the minimal mode records file
|
|
5
|
+
writes too.
|
|
6
|
+
"""
|
|
7
|
+
import os
|
|
8
|
+
|
|
9
|
+
CAPTOR = os.path.realpath(os.path.dirname(os.path.abspath(__file__)))
|
|
10
|
+
# Folders that never hold the project's own code, wherever they are.
|
|
11
|
+
EXCLUDED = {'site-packages', 'dist-packages', '__pycache__', 'node_modules', 'venv'}
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def project_path(filename, root, python=True):
|
|
15
|
+
"""The real path of a .py file of the project under root, or None. With python=False,
|
|
16
|
+
any file of the project (Jinja templates, jinja_map.py)."""
|
|
17
|
+
# <frozen ...>, <string> and other generated code are not the project's
|
|
18
|
+
# .py files; templates are recognised by their code (jinja_map.py).
|
|
19
|
+
if not filename or filename.startswith('<') or (python and not filename.endswith('.py')):
|
|
20
|
+
return None
|
|
21
|
+
path = os.path.realpath(filename)
|
|
22
|
+
if not os.path.isfile(path) or path.startswith(CAPTOR + os.sep):
|
|
23
|
+
return None
|
|
24
|
+
relative = os.path.relpath(path, root)
|
|
25
|
+
if relative == os.pardir or relative.startswith(os.pardir + os.sep) or os.path.isabs(relative):
|
|
26
|
+
return None
|
|
27
|
+
folder = root
|
|
28
|
+
for part in relative.split(os.sep)[:-1]:
|
|
29
|
+
# Hidden folders (.venv, .git, .tox...), dependencies, and any
|
|
30
|
+
# virtual environment, whatever its name, inside the project.
|
|
31
|
+
if part.startswith('.') or part in EXCLUDED:
|
|
32
|
+
return None
|
|
33
|
+
folder = os.path.join(folder, part)
|
|
34
|
+
if os.path.exists(os.path.join(folder, 'pyvenv.cfg')):
|
|
35
|
+
return None
|
|
36
|
+
return path
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
class ProjectFiles(object):
|
|
40
|
+
"""project_path with a cache per file name. With python=False, the files of the
|
|
41
|
+
project that are not .py (templates)."""
|
|
42
|
+
|
|
43
|
+
def __init__(self, root, python=True):
|
|
44
|
+
self.root = os.path.realpath(root)
|
|
45
|
+
self.python = python
|
|
46
|
+
self.files = {}
|
|
47
|
+
|
|
48
|
+
def __call__(self, filename):
|
|
49
|
+
known = self.files.get(filename, False)
|
|
50
|
+
if known is not False:
|
|
51
|
+
return known
|
|
52
|
+
path = None
|
|
53
|
+
try:
|
|
54
|
+
path = project_path(filename, self.root, self.python)
|
|
55
|
+
if path is not None and not self.python and path.endswith('.py'):
|
|
56
|
+
path = None
|
|
57
|
+
except (OSError, ValueError):
|
|
58
|
+
pass
|
|
59
|
+
self.files[filename] = path
|
|
60
|
+
return path
|
|
@@ -0,0 +1,104 @@
|
|
|
1
|
+
"""Port of src/redact.mjs with identical behaviour.
|
|
2
|
+
|
|
3
|
+
test/vetores-redacao.json holds inputs and expected outputs shared by the Node
|
|
4
|
+
and the Python tests: any difference fails in both.
|
|
5
|
+
|
|
6
|
+
JavaScript regular expressions without the u flag have ASCII \\w, \\d and \\b,
|
|
7
|
+
ASCII-only case folding, and a Unicode \\s. The patterns below use re.ASCII
|
|
8
|
+
and spell \\s out, to match them character for character.
|
|
9
|
+
|
|
10
|
+
Keep it importable on old Pythons (3.8+): the minimal mode uses it too.
|
|
11
|
+
"""
|
|
12
|
+
import os
|
|
13
|
+
import re
|
|
14
|
+
|
|
15
|
+
_S = '[\\t\\n\\v\\f\\r \\u00a0\\u1680\\u2000-\\u200a\\u2028\\u2029\\u202f\\u205f\\u3000\\ufeff]'
|
|
16
|
+
_NOT_S = _S.replace('[', '[^', 1)
|
|
17
|
+
_FLAGS = re.ASCII | re.IGNORECASE
|
|
18
|
+
_NAMES = 'password|secret|token|authorization|api[ _-]?key|credential|private[ _-]?key|session'
|
|
19
|
+
|
|
20
|
+
_sensitive = re.compile(_NAMES, _FLAGS)
|
|
21
|
+
MARKER = '[REDACTED]'
|
|
22
|
+
MIN_SECRET_LENGTH = 8
|
|
23
|
+
|
|
24
|
+
_bearer = re.compile('Bearer' + _S + '+' + _NOT_S[:-1] + '"\',;]+', _FLAGS)
|
|
25
|
+
_private_key = re.compile('-----BEGIN [\\w ]*PRIVATE KEY-----[\\s\\S]*?-----END [\\w ]*PRIVATE KEY-----', re.ASCII)
|
|
26
|
+
_tokens = re.compile('\\b(?:sk-(?:proj-|ant-)?[\\w-]{8,}|gh[pousr]_[\\w]{10,}|github_pat_[\\w]{10,}|AKIA[A-Z0-9]{16}'
|
|
27
|
+
'|AIza[\\w-]{20,}|xox[baprs]-[\\w-]{10,}|eyJ[\\w-]+\\.[\\w-]+\\.[\\w-]+)\\b', re.ASCII)
|
|
28
|
+
_assignment = re.compile('((?:' + _NAMES + ')' + _S + '*[=:]' + _S + '*)(?:"[^"]*"|\'[^\']*\'|' + _NOT_S[:-1] + ',;]+)', _FLAGS)
|
|
29
|
+
_email = re.compile('\\b([\\w.+-])[\\w.+-]*@([\\w-])[\\w.-]*\\.[a-z]{2,}\\b', _FLAGS)
|
|
30
|
+
_phone = re.compile('(?<![\\w])\\+?\\d[\\d ()-]{7,}\\d(?![\\w])', re.ASCII)
|
|
31
|
+
_surrogate = re.compile('[\\ud800-\\udfff]')
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def is_sensitive_name(name):
|
|
35
|
+
"""A name (field, parameter, function) that announces a secret."""
|
|
36
|
+
return bool(_sensitive.search('' if name is None else str(name)))
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def create_redactor(env=None, max_bytes=10 * 1024):
|
|
40
|
+
env = os.environ if env is None else env
|
|
41
|
+
# Very short values ("1", "true") are flags, not secrets; matching them
|
|
42
|
+
# would destroy unrelated text such as paths and numbers.
|
|
43
|
+
secrets = sorted((value for key, value in env.items()
|
|
44
|
+
if _sensitive.search(key) and isinstance(value, str) and len(value) >= MIN_SECRET_LENGTH),
|
|
45
|
+
key=len, reverse=True)
|
|
46
|
+
|
|
47
|
+
def text(value):
|
|
48
|
+
for secret in secrets:
|
|
49
|
+
value = value.replace(secret, MARKER)
|
|
50
|
+
value = _bearer.sub('Bearer ' + MARKER, value)
|
|
51
|
+
value = _private_key.sub(MARKER, value)
|
|
52
|
+
value = _tokens.sub(MARKER, value)
|
|
53
|
+
value = _assignment.sub(lambda match: match.group(1) + MARKER, value)
|
|
54
|
+
value = _email.sub(lambda match: match.group(1) + '***@' + match.group(2) + '***', value)
|
|
55
|
+
value = _phone.sub(lambda match: match.group(0)[:2] + '***' + match.group(0)[-2:], value)
|
|
56
|
+
# Node counts and cuts UTF-8 bytes, with lone surrogates as U+FFFD.
|
|
57
|
+
encoded = value.encode('utf-8', 'surrogatepass')
|
|
58
|
+
if len(encoded) > max_bytes:
|
|
59
|
+
encoded = _surrogate.sub('�', value).encode('utf-8')
|
|
60
|
+
cut = encoded[:max_bytes].decode('utf-8', 'replace')
|
|
61
|
+
if cut.endswith('�'):
|
|
62
|
+
cut = cut[:-1]
|
|
63
|
+
value = cut + '[TRUNCATED]'
|
|
64
|
+
return value
|
|
65
|
+
|
|
66
|
+
# text() depends only on its input (the secrets are fixed here): short texts repeat
|
|
67
|
+
# (event keys, library names, SQL of the same query) and are redacted once.
|
|
68
|
+
# Bounded, so a stream of distinct values cannot grow it.
|
|
69
|
+
cache = {}
|
|
70
|
+
keys = {}
|
|
71
|
+
|
|
72
|
+
def cached(value):
|
|
73
|
+
if len(value) > 512:
|
|
74
|
+
return text(value)
|
|
75
|
+
found = cache.get(value)
|
|
76
|
+
if found is None:
|
|
77
|
+
if len(cache) >= 4096:
|
|
78
|
+
cache.clear()
|
|
79
|
+
found = cache[value] = text(value)
|
|
80
|
+
return found
|
|
81
|
+
|
|
82
|
+
def key_of(key):
|
|
83
|
+
found = keys.get(key)
|
|
84
|
+
if found is None:
|
|
85
|
+
name = str(key)
|
|
86
|
+
if len(keys) >= 4096:
|
|
87
|
+
keys.clear()
|
|
88
|
+
found = keys[key] = (cached(name), bool(_sensitive.search(name)))
|
|
89
|
+
return found
|
|
90
|
+
|
|
91
|
+
def redact(value):
|
|
92
|
+
if isinstance(value, str):
|
|
93
|
+
return cached(value)
|
|
94
|
+
if isinstance(value, (list, tuple)):
|
|
95
|
+
return [redact(item) for item in value]
|
|
96
|
+
if isinstance(value, dict):
|
|
97
|
+
out = {}
|
|
98
|
+
for key, item in value.items():
|
|
99
|
+
name, sensitive = key_of(key)
|
|
100
|
+
out[name] = MARKER if sensitive else redact(item)
|
|
101
|
+
return out
|
|
102
|
+
return value
|
|
103
|
+
|
|
104
|
+
return redact
|