codetac 0.1.0 → 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,513 @@
1
+ """Browser side, served by the observed app's own Python server, as src/page.mjs
2
+ does for Node: HTML pages get a <script src="/__codetac/bar.js">, and the
3
+ /__codetac/ routes are answered here without ever reaching the app. It works
4
+ at the protocol level (WSGI and ASGI), for any framework on top of them.
5
+
6
+ - the action of a request: the x-codetac-action header set by the page script
7
+ on fetch and XHR, or the short cookie it sets before a full navigation;
8
+ - /__codetac/bar.js serves the same bar.js as the Node side (CODETAC_BAR_JS),
9
+ and /__codetac/events records the page's actions (`browser-action`);
10
+ - the script tag goes into complete HTML documents: known length (or a body
11
+ given whole), not compressed, not a file, 1 MB at most. The length is
12
+ adjusted, and validators (ETag, Last-Modified) are dropped only when the
13
+ body changes. Streams (no length), files and compressed bodies pass
14
+ untouched. Error pages (500) get the bar too.
15
+
16
+ Keep it importable on old Pythons (3.8+): the minimal mode has the bar too.
17
+ """
18
+ import json
19
+ import math
20
+ import os
21
+ import re
22
+
23
+ PREFIX = '/__codetac/'
24
+ ACTION_HEADER = 'x-codetac-action'
25
+ INTERNAL_HEADER = 'x-codetac-internal'
26
+ _ACTION_ID = re.compile(r'^[A-Za-z0-9_-]{6,40}$')
27
+ _ACTION_NUMBER = re.compile(r'^\d{1,5}$')
28
+ _ACTION_COOKIE = re.compile(r'(?:^|;\s*)codetac_action=([A-Za-z0-9_-]{6,40})(?:;|$)')
29
+ MAX_EVENT_BYTES = 64 * 1024
30
+ MAX_PAGE_BYTES = 1024 * 1024
31
+ TAG = b'<script src="/__codetac/bar.js" data-codetac=""></script>'
32
+ _HEAD = re.compile(br'<head(?:\s[^>]*)?>', re.IGNORECASE)
33
+ _BODY = re.compile(br'<body[\s>]', re.IGNORECASE)
34
+ _LEADING = br'(?:\s|<!--.*?-->)*'
35
+ _OPENING_HTML = re.compile(_LEADING + br'(?:<!doctype html[^>]*>' + _LEADING + br')?<html(?:\s[^>]*)?>', re.IGNORECASE | re.DOTALL)
36
+ _OPENING_DOCTYPE = re.compile(_LEADING + br'<!doctype html[^>]*>', re.IGNORECASE | re.DOTALL)
37
+ _HTML = re.compile(r'text/html', re.IGNORECASE)
38
+ DROPPED = ('content-length', 'etag', 'last-modified')
39
+ BAR_JS = os.path.realpath(os.path.join(os.path.dirname(os.path.abspath(__file__)), '..', '..', 'browser', 'bar.js'))
40
+
41
+
42
+ def enabled():
43
+ return os.environ.get('CODETAC_PAGE') != '0'
44
+
45
+
46
+ def panel_url():
47
+ """src/page.mjs panelUrl."""
48
+ return os.environ.get('CODETAC_PANEL_URL') or 'http://127.0.0.1:%s' % (os.environ.get('CODETAC_PANEL_PORT') or 4000)
49
+
50
+
51
+ _script = None
52
+
53
+
54
+ def bar_script():
55
+ global _script
56
+ if _script is None:
57
+ config = {'panel': panel_url(), 'run': os.environ.get('CODETAC_RUN', '')}
58
+ with open(os.environ.get('CODETAC_BAR_JS') or BAR_JS, 'rb') as file:
59
+ source = file.read()
60
+ _script = ('window.__CODETAC_CONFIG__=%s;\n' % json.dumps(config, separators=(',', ':'))).encode('utf-8') + source
61
+ return _script
62
+
63
+
64
+ def action_of(header, fetch_mode, cookie):
65
+ """src/page.mjs actionOf: {'action', 'actionRequest'} or None."""
66
+ if isinstance(header, str):
67
+ parts = header.split('.')
68
+ if _ACTION_ID.match(parts[0]):
69
+ number = parts[1] if len(parts) > 1 else ''
70
+ return {'action': parts[0], 'actionRequest': int(number) if _ACTION_NUMBER.match(number) else None}
71
+ if fetch_mode == 'navigate':
72
+ match = _ACTION_COOKIE.search(cookie or '')
73
+ if match:
74
+ return {'action': match.group(1), 'actionRequest': None}
75
+ return None
76
+
77
+
78
+ def wants_page(method, destination, accept):
79
+ """src/page.mjs wantsPage: top-level HTML documents."""
80
+ if method != 'GET':
81
+ return False
82
+ if destination:
83
+ return destination == 'document'
84
+ return bool(_HTML.search(accept or ''))
85
+
86
+
87
+ def insert_tag(body, final):
88
+ """src/page.mjs tagPosition: after <head ...>; in a whole document without a head,
89
+ before <body>, or else after <html ...> or <!doctype html> (werkzeug's error pages have
90
+ neither head nor body). None while more of the body is needed; the body unchanged when
91
+ there is nowhere to put it (fragments)."""
92
+ head = _HEAD.search(body)
93
+ if head:
94
+ at = head.end()
95
+ elif not final:
96
+ return None
97
+ else:
98
+ match = _BODY.search(body)
99
+ opening = match is None and (_OPENING_HTML.match(body) or _OPENING_DOCTYPE.match(body))
100
+ if match:
101
+ at = match.start()
102
+ elif opening:
103
+ at = opening.end()
104
+ else:
105
+ return body
106
+ return body[:at] + TAG + body[at:]
107
+
108
+
109
+ def injectable(status, headers):
110
+ """Whether the response is an HTML document the tag can go into: `headers` as
111
+ (lower-case name, value) pairs. Returns the declared length (or -1), or None."""
112
+ if not 200 <= status < 600 or status in (204, 206, 304):
113
+ return None
114
+ kind = length = None
115
+ for name, value in headers:
116
+ if name == 'content-type':
117
+ kind = value
118
+ elif name in ('content-encoding', 'content-range') and value.strip().lower() not in ('', 'identity'):
119
+ return None
120
+ elif name == 'content-length':
121
+ try:
122
+ length = int(value.strip())
123
+ except ValueError:
124
+ return None
125
+ if not kind or not _HTML.search(kind):
126
+ return None
127
+ if length is None:
128
+ return -1
129
+ return length if length <= MAX_PAGE_BYTES else None
130
+
131
+
132
+ def adjusted(headers, length):
133
+ """Headers (name, value), as given, for a body that changed to `length` bytes."""
134
+ result = [(name, value) for name, value in headers if name.lower() not in DROPPED]
135
+ result.append(('Content-Length', str(length)))
136
+ return result
137
+
138
+
139
+ # Browser actions -----------------------------------------------------------------
140
+
141
+ def _text(value, maximum=120):
142
+ if not isinstance(value, str):
143
+ return None
144
+ return ' '.join(value.split())[:maximum]
145
+
146
+
147
+ def _number(value):
148
+ if isinstance(value, bool) or not isinstance(value, (int, float)):
149
+ return None
150
+ return value if math.isfinite(value) else None
151
+
152
+
153
+ def _frames(items):
154
+ if not isinstance(items, list):
155
+ return None
156
+ result = []
157
+ for frame in items[:16]:
158
+ frame = frame if isinstance(frame, dict) else {}
159
+ item = {'fn': _text(frame.get('fn'), 80), 'url': _text(frame.get('url'), 500), 'line': _number(frame.get('line')),
160
+ 'column': _number(frame.get('column')), 'file': _text(frame.get('file'), 500)}
161
+ if item['url'] or item['file']:
162
+ result.append(item)
163
+ return result
164
+
165
+
166
+ def _names(items):
167
+ if not isinstance(items, list):
168
+ return None
169
+ result = []
170
+ for item in items[:40]:
171
+ item = item if isinstance(item, dict) else {}
172
+ entry = {'name': _text(item.get('name'), 80), 'count': _number(item.get('count')), 'frames': _frames(item.get('frames'))}
173
+ if entry['name']:
174
+ result.append(entry)
175
+ return result
176
+
177
+
178
+ def _path_of(raw):
179
+ if not isinstance(raw, str):
180
+ return {}
181
+ from .servers import split_target
182
+ try:
183
+ path, keys = split_target(raw)
184
+ except Exception:
185
+ return {'path': '[inválido]', 'queryKeys': []}
186
+ return {'path': path, 'queryKeys': keys}
187
+
188
+
189
+ def _clean(value):
190
+ """Drops the fields that are None, as JSON.stringify drops undefined."""
191
+ if isinstance(value, dict):
192
+ return {key: _clean(item) for key, item in value.items() if item is not None}
193
+ if isinstance(value, list):
194
+ return [_clean(item) for item in value]
195
+ return value
196
+
197
+
198
+ def browser_action(event):
199
+ """src/page.mjs recordBrowserAction: only known fields; URLs lose their query values.
200
+ The writer redacts the event before it is written."""
201
+ if not isinstance(event, dict) or not _ACTION_ID.match(str(event.get('actionId'))):
202
+ return None
203
+ trigger = event.get('trigger') if isinstance(event.get('trigger'), dict) else None
204
+ shaped = None
205
+ if trigger:
206
+ element = trigger.get('element') if isinstance(trigger.get('element'), dict) else {}
207
+ component = trigger.get('component') if isinstance(trigger.get('component'), dict) else None
208
+ handler = trigger.get('handler') if isinstance(trigger.get('handler'), dict) else None
209
+ owners = None
210
+ if component and isinstance(component.get('owners'), list):
211
+ owners = [{'name': _text(owner.get('name'), 80), 'frames': _frames(owner.get('frames'))}
212
+ for owner in component['owners'][:6] if isinstance(owner, dict)]
213
+ owners = [owner for owner in owners if owner['name']]
214
+ shaped = {
215
+ 'event': _text(trigger.get('event'), 20),
216
+ 'element': {'tag': _text(element.get('tag'), 20), 'type': _text(element.get('type'), 20), 'role': _text(element.get('role'), 30),
217
+ 'text': _text(element.get('text'), 80), 'label': _text(element.get('label'), 80), 'name': _text(element.get('name'), 60),
218
+ 'id': _text(element.get('id'), 60), 'href': _path_of(element['href']).get('path') if element.get('href') else None},
219
+ 'component': {'name': _text(component.get('name'), 80), 'owners': owners, 'frames': _frames(component.get('frames'))} if component else None,
220
+ 'handler': {'name': _text(handler.get('name'), 80), 'prop': _text(handler.get('prop'), 30),
221
+ 'source': _text(handler.get('source'), 20)} if handler else None,
222
+ }
223
+ requests = []
224
+ for item in event.get('requests')[:200] if isinstance(event.get('requests'), list) else []:
225
+ item = item if isinstance(item, dict) else {}
226
+ same = bool(item.get('sameOrigin'))
227
+ entry = {'n': _number(item.get('n')), 'kind': _text(item.get('kind'), 20), 'method': _text(item.get('method'), 10)}
228
+ entry.update(_path_of(item.get('url')))
229
+ entry.update({'sameOrigin': same, 'host': None if same else _text(item.get('host'), 200),
230
+ 'startMs': _number(item.get('startMs')), 'durationMs': _number(item.get('durationMs')), 'status': _number(item.get('status')),
231
+ 'error': True if item.get('error') else None, 'frames': _frames(item.get('frames'))})
232
+ requests.append(entry)
233
+ screen = event.get('screen') if isinstance(event.get('screen'), dict) else None
234
+ navigations = None
235
+ if isinstance(event.get('navigations'), list):
236
+ navigations = []
237
+ for item in event['navigations'][:20]:
238
+ item = item if isinstance(item, dict) else {}
239
+ entry = {'kind': _text(item.get('kind'), 20)}
240
+ entry.update(_path_of(item.get('to')))
241
+ entry['atMs'] = _number(item.get('atMs'))
242
+ navigations.append(entry)
243
+ segment = _number(event.get('segment'))
244
+ return _clean({
245
+ 'type': 'browser-action', 'actionId': event['actionId'], 'segment': 1 if segment is None else segment,
246
+ 'origin': _text(event.get('origin'), 200), 'page': _path_of(event.get('page')),
247
+ 'startedAt': _number(event.get('startedAt')), 'durationMs': _number(event.get('durationMs')), 'closedBy': _text(event.get('closedBy'), 40),
248
+ 'trigger': shaped, 'requests': requests,
249
+ 'screen': {'added': _number(screen.get('added')), 'removed': _number(screen.get('removed')), 'text': _number(screen.get('text')),
250
+ 'attributes': _number(screen.get('attributes')), 'title': True if screen.get('title') else None,
251
+ 'stateChanged': _names(screen.get('stateChanged')), 'mounted': _names(screen.get('mounted')),
252
+ 'unmounted': _names(screen.get('unmounted'))} if screen else None,
253
+ 'scripts': [url for url in (_text(url, 500) for url in event['scripts'][:80]) if url] if isinstance(event.get('scripts'), list) else None,
254
+ 'navigations': navigations,
255
+ })
256
+
257
+
258
+ def own_route(method, path, origin, host, read_body):
259
+ """Answers a /__codetac/ request: (status, headers, body). `read_body(limit)` returns
260
+ the request body, or None when it is longer than the limit."""
261
+ if path == PREFIX + 'bar.js' and method == 'GET':
262
+ return 200, [('Content-Type', 'text/javascript; charset=utf-8'), ('Cache-Control', 'no-store')], bar_script()
263
+ if path == PREFIX + 'events' and method == 'POST':
264
+ # Only the page itself may report actions: another site open in the
265
+ # browser cannot write into the recording (browsers send Origin here).
266
+ same = not origin
267
+ if not same:
268
+ from urllib.parse import urlsplit
269
+ try:
270
+ same = urlsplit(origin).netloc == host
271
+ except ValueError:
272
+ same = False
273
+ if not same:
274
+ return 403, [('Content-Type', 'text/plain; charset=utf-8')], 'CodeTAC: origem recusada.'.encode('utf-8')
275
+ body = read_body(MAX_EVENT_BYTES)
276
+ if body is None:
277
+ return 413, [('Cache-Control', 'no-store')], b''
278
+ try:
279
+ event = browser_action(json.loads(body.decode('utf-8')))
280
+ if event is not None:
281
+ import codetac_py
282
+ codetac_py.writer.emit(event)
283
+ except Exception:
284
+ pass
285
+ return 204, [('Cache-Control', 'no-store')], b''
286
+ return 404, [('Content-Type', 'text/plain; charset=utf-8')], 'CodeTAC: rota desconhecida.'.encode('utf-8')
287
+
288
+
289
+ def emit_page(path):
290
+ import codetac_py
291
+ codetac_py.writer.emit({'type': 'page', 'path': path})
292
+
293
+
294
+ # WSGI ------------------------------------------------------------------------------
295
+
296
+ def wsgi_own_route(environ, start_response):
297
+ def read_body(limit):
298
+ try:
299
+ length = int(environ.get('CONTENT_LENGTH') or 0)
300
+ except ValueError:
301
+ length = 0
302
+ if length > limit:
303
+ return None
304
+ stream = environ.get('wsgi.input')
305
+ return stream.read(length) if stream is not None and length > 0 else b''
306
+
307
+ status, headers, body = own_route(environ.get('REQUEST_METHOD', 'GET'), environ.get('PATH_INFO', ''),
308
+ environ.get('HTTP_ORIGIN'), environ.get('HTTP_HOST'), read_body)
309
+ reasons = {200: 'OK', 204: 'No Content', 403: 'Forbidden', 404: 'Not Found', 413: 'Payload Too Large'}
310
+ start_response('%d %s' % (status, reasons[status]), headers + [('Content-Length', str(len(body)))])
311
+ return [body]
312
+
313
+
314
+ class WsgiPage(object):
315
+ """The response of a page request: start_response is held until the body shows whether
316
+ the tag goes in. The server sends nothing before the first chunk anyway (PEP 3333)."""
317
+
318
+ def __init__(self, start_response):
319
+ self.start_response = start_response
320
+ self.held = None # (status, headers, exc_info) while undecided
321
+ self.mode = None # None: undecided; 'pass'; 'buffer'
322
+ self.length = -1
323
+
324
+ def start(self, status, headers, exc_info=None):
325
+ if self.mode == 'pass':
326
+ return self.start_response(status, headers, exc_info) if exc_info else self.start_response(status, headers)
327
+ try:
328
+ code = int(str(status).split(' ', 1)[0])
329
+ length = injectable(code, [(name.lower(), value) for name, value in headers])
330
+ except Exception:
331
+ length = None
332
+ if length is None:
333
+ self.mode = 'pass'
334
+ return self.start_response(status, headers, exc_info) if exc_info else self.start_response(status, headers)
335
+ self.mode = 'buffer'
336
+ self.length = length
337
+ self.held = (status, headers, exc_info)
338
+ return self._write
339
+
340
+ def _write(self, data):
341
+ # The legacy write() callable: the headers must go now, unchanged.
342
+ self._release(None)
343
+ return self.write(data)
344
+
345
+ def _release(self, body):
346
+ """Sends the held headers, for `body` (None: the original response goes on unchanged)."""
347
+ status, headers, exc_info = self.held
348
+ self.held = None
349
+ self.mode = 'pass'
350
+ if body is not None:
351
+ headers = adjusted(headers, len(body))
352
+ self.write = self.start_response(status, headers, exc_info) if exc_info else self.start_response(status, headers)
353
+
354
+ def body(self, result):
355
+ """The iterable to return to the server."""
356
+ if self.mode == 'pass':
357
+ return result
358
+ # Files (werkzeug's and wsgiref's FileWrapper, the server's file_wrapper): untouched.
359
+ if type(result).__name__ == 'FileWrapper' or hasattr(result, 'filelike'):
360
+ if self.mode == 'buffer':
361
+ self._release(None)
362
+ return result
363
+ return _Iterated(self, result)
364
+
365
+
366
+ class _Iterated(object):
367
+ """The body of a page request; close() is the original body's."""
368
+
369
+ def __init__(self, page, result):
370
+ self.page = page
371
+ self.result = result
372
+ self.iterator = None
373
+
374
+ def __iter__(self):
375
+ if self.iterator is None:
376
+ self.iterator = self._iterate()
377
+ return self.iterator
378
+
379
+ def close(self):
380
+ close = getattr(self.result, 'close', None)
381
+ if close is not None:
382
+ close()
383
+
384
+ def _iterate(self):
385
+ # start_response may be called while the body is being iterated
386
+ # (generators, the werkzeug debugger).
387
+ page, result = self.page, self.result
388
+ piecewise = not isinstance(result, (list, tuple))
389
+ chunks = []
390
+ size = 0
391
+ for chunk in result:
392
+ if page.mode != 'buffer':
393
+ if chunks:
394
+ yield b''.join(chunks)
395
+ chunks = []
396
+ yield chunk
397
+ continue
398
+ if page.length < 0 and piecewise:
399
+ # A stream (no length, given piece by piece): untouched.
400
+ page._release(None)
401
+ yield chunk
402
+ continue
403
+ chunks.append(chunk)
404
+ size += len(chunk)
405
+ if size > MAX_PAGE_BYTES:
406
+ page._release(None)
407
+ yield b''.join(chunks)
408
+ chunks = []
409
+ if page.mode == 'buffer':
410
+ body = b''.join(chunks)
411
+ changed = insert_tag(body, True)
412
+ if page.length >= 0 and len(body) != page.length or changed == body:
413
+ page._release(None)
414
+ else:
415
+ page._release(changed)
416
+ body = changed
417
+ yield body
418
+ elif chunks:
419
+ yield b''.join(chunks)
420
+
421
+
422
+ # ASGI ------------------------------------------------------------------------------
423
+
424
+ async def asgi_own_route(scope, receive, send):
425
+ headers = dict((key.decode('latin-1').lower(), value.decode('latin-1')) for key, value in scope.get('headers') or [])
426
+ received = []
427
+
428
+ async def read_all():
429
+ size = 0
430
+ while True:
431
+ message = await receive()
432
+ if message.get('type') != 'http.request':
433
+ return None
434
+ chunk = message.get('body') or b''
435
+ size += len(chunk)
436
+ if size <= MAX_EVENT_BYTES:
437
+ received.append(chunk)
438
+ if not message.get('more_body'):
439
+ return size
440
+
441
+ size = await read_all() if scope.get('method') == 'POST' else 0
442
+ status, answer, body = own_route(scope.get('method', 'GET'), scope.get('path', ''), headers.get('origin'), headers.get('host'),
443
+ lambda limit: None if size is None or size > limit else b''.join(received))
444
+ await send({'type': 'http.response.start', 'status': status,
445
+ 'headers': [(name.lower().encode('latin-1'), value.encode('latin-1')) for name, value in answer]
446
+ + [(b'content-length', str(len(body)).encode())]})
447
+ await send({'type': 'http.response.body', 'body': body})
448
+
449
+
450
+ class AsgiPage(object):
451
+ """The response of a page request, for ASGI: http.response.start is held until the
452
+ body shows whether the tag goes in."""
453
+
454
+ def __init__(self, send):
455
+ self.send = send
456
+ self.mode = None
457
+ self.start = None
458
+ self.length = -1
459
+ self.chunks = []
460
+ self.size = 0
461
+
462
+ async def __call__(self, message):
463
+ kind = message.get('type')
464
+ if self.mode == 'pass' or kind not in ('http.response.start', 'http.response.body'):
465
+ if self.mode == 'buffer':
466
+ # Files (http.response.pathsend, zerocopy) or anything else: untouched.
467
+ await self._release(None)
468
+ return await self.send(message)
469
+ if kind == 'http.response.start':
470
+ try:
471
+ self.length = injectable(int(message.get('status')), [(key.decode('latin-1').lower(), value.decode('latin-1'))
472
+ for key, value in message.get('headers') or []])
473
+ except Exception:
474
+ self.length = None
475
+ if self.length is None or self.length < 0:
476
+ # Without a length it is a stream (StreamingResponse): untouched.
477
+ self.mode = 'pass'
478
+ return await self.send(message)
479
+ self.mode = 'buffer'
480
+ self.start = message
481
+ return None
482
+ body = message.get('body') or b''
483
+ self.chunks.append(body)
484
+ self.size += len(body)
485
+ if message.get('more_body', False):
486
+ if self.size > MAX_PAGE_BYTES:
487
+ await self._release(None)
488
+ return None
489
+ whole = b''.join(self.chunks)
490
+ self.chunks = []
491
+ changed = insert_tag(whole, True)
492
+ if len(whole) != self.length or changed == whole:
493
+ await self._release(None, whole)
494
+ else:
495
+ await self._release(changed)
496
+
497
+ async def _release(self, changed, whole=None):
498
+ start, self.start = self.start, None
499
+ self.mode = 'pass'
500
+ if changed is not None:
501
+ headers = [(key, value) for key, value in start.get('headers') or [] if key.decode('latin-1').lower() not in DROPPED]
502
+ headers.append((b'content-length', str(len(changed)).encode()))
503
+ start = dict(start, headers=headers)
504
+ await self.send(start)
505
+ await self.send({'type': 'http.response.body', 'body': changed})
506
+ return
507
+ await self.send(start)
508
+ if whole is not None:
509
+ await self.send({'type': 'http.response.body', 'body': whole})
510
+ return
511
+ chunks, self.chunks = self.chunks, []
512
+ for chunk in chunks:
513
+ await self.send({'type': 'http.response.body', 'body': chunk, 'more_body': True})
@@ -0,0 +1,60 @@
1
+ """Which files are the project's own code: used by the capture (functions to
2
+ follow) and by the file boundaries (who is writing).
3
+
4
+ Keep it importable on old Pythons (3.8+): the minimal mode records file
5
+ writes too.
6
+ """
7
+ import os
8
+
9
+ CAPTOR = os.path.realpath(os.path.dirname(os.path.abspath(__file__)))
10
+ # Folders that never hold the project's own code, wherever they are.
11
+ EXCLUDED = {'site-packages', 'dist-packages', '__pycache__', 'node_modules', 'venv'}
12
+
13
+
14
+ def project_path(filename, root, python=True):
15
+ """The real path of a .py file of the project under root, or None. With python=False,
16
+ any file of the project (Jinja templates, jinja_map.py)."""
17
+ # <frozen ...>, <string> and other generated code are not the project's
18
+ # .py files; templates are recognised by their code (jinja_map.py).
19
+ if not filename or filename.startswith('<') or (python and not filename.endswith('.py')):
20
+ return None
21
+ path = os.path.realpath(filename)
22
+ if not os.path.isfile(path) or path.startswith(CAPTOR + os.sep):
23
+ return None
24
+ relative = os.path.relpath(path, root)
25
+ if relative == os.pardir or relative.startswith(os.pardir + os.sep) or os.path.isabs(relative):
26
+ return None
27
+ folder = root
28
+ for part in relative.split(os.sep)[:-1]:
29
+ # Hidden folders (.venv, .git, .tox...), dependencies, and any
30
+ # virtual environment, whatever its name, inside the project.
31
+ if part.startswith('.') or part in EXCLUDED:
32
+ return None
33
+ folder = os.path.join(folder, part)
34
+ if os.path.exists(os.path.join(folder, 'pyvenv.cfg')):
35
+ return None
36
+ return path
37
+
38
+
39
+ class ProjectFiles(object):
40
+ """project_path with a cache per file name. With python=False, the files of the
41
+ project that are not .py (templates)."""
42
+
43
+ def __init__(self, root, python=True):
44
+ self.root = os.path.realpath(root)
45
+ self.python = python
46
+ self.files = {}
47
+
48
+ def __call__(self, filename):
49
+ known = self.files.get(filename, False)
50
+ if known is not False:
51
+ return known
52
+ path = None
53
+ try:
54
+ path = project_path(filename, self.root, self.python)
55
+ if path is not None and not self.python and path.endswith('.py'):
56
+ path = None
57
+ except (OSError, ValueError):
58
+ pass
59
+ self.files[filename] = path
60
+ return path
@@ -0,0 +1,104 @@
1
+ """Port of src/redact.mjs with identical behaviour.
2
+
3
+ test/vetores-redacao.json holds inputs and expected outputs shared by the Node
4
+ and the Python tests: any difference fails in both.
5
+
6
+ JavaScript regular expressions without the u flag have ASCII \\w, \\d and \\b,
7
+ ASCII-only case folding, and a Unicode \\s. The patterns below use re.ASCII
8
+ and spell \\s out, to match them character for character.
9
+
10
+ Keep it importable on old Pythons (3.8+): the minimal mode uses it too.
11
+ """
12
+ import os
13
+ import re
14
+
15
+ _S = '[\\t\\n\\v\\f\\r \\u00a0\\u1680\\u2000-\\u200a\\u2028\\u2029\\u202f\\u205f\\u3000\\ufeff]'
16
+ _NOT_S = _S.replace('[', '[^', 1)
17
+ _FLAGS = re.ASCII | re.IGNORECASE
18
+ _NAMES = 'password|secret|token|authorization|api[ _-]?key|credential|private[ _-]?key|session'
19
+
20
+ _sensitive = re.compile(_NAMES, _FLAGS)
21
+ MARKER = '[REDACTED]'
22
+ MIN_SECRET_LENGTH = 8
23
+
24
+ _bearer = re.compile('Bearer' + _S + '+' + _NOT_S[:-1] + '"\',;]+', _FLAGS)
25
+ _private_key = re.compile('-----BEGIN [\\w ]*PRIVATE KEY-----[\\s\\S]*?-----END [\\w ]*PRIVATE KEY-----', re.ASCII)
26
+ _tokens = re.compile('\\b(?:sk-(?:proj-|ant-)?[\\w-]{8,}|gh[pousr]_[\\w]{10,}|github_pat_[\\w]{10,}|AKIA[A-Z0-9]{16}'
27
+ '|AIza[\\w-]{20,}|xox[baprs]-[\\w-]{10,}|eyJ[\\w-]+\\.[\\w-]+\\.[\\w-]+)\\b', re.ASCII)
28
+ _assignment = re.compile('((?:' + _NAMES + ')' + _S + '*[=:]' + _S + '*)(?:"[^"]*"|\'[^\']*\'|' + _NOT_S[:-1] + ',;]+)', _FLAGS)
29
+ _email = re.compile('\\b([\\w.+-])[\\w.+-]*@([\\w-])[\\w.-]*\\.[a-z]{2,}\\b', _FLAGS)
30
+ _phone = re.compile('(?<![\\w])\\+?\\d[\\d ()-]{7,}\\d(?![\\w])', re.ASCII)
31
+ _surrogate = re.compile('[\\ud800-\\udfff]')
32
+
33
+
34
+ def is_sensitive_name(name):
35
+ """A name (field, parameter, function) that announces a secret."""
36
+ return bool(_sensitive.search('' if name is None else str(name)))
37
+
38
+
39
+ def create_redactor(env=None, max_bytes=10 * 1024):
40
+ env = os.environ if env is None else env
41
+ # Very short values ("1", "true") are flags, not secrets; matching them
42
+ # would destroy unrelated text such as paths and numbers.
43
+ secrets = sorted((value for key, value in env.items()
44
+ if _sensitive.search(key) and isinstance(value, str) and len(value) >= MIN_SECRET_LENGTH),
45
+ key=len, reverse=True)
46
+
47
+ def text(value):
48
+ for secret in secrets:
49
+ value = value.replace(secret, MARKER)
50
+ value = _bearer.sub('Bearer ' + MARKER, value)
51
+ value = _private_key.sub(MARKER, value)
52
+ value = _tokens.sub(MARKER, value)
53
+ value = _assignment.sub(lambda match: match.group(1) + MARKER, value)
54
+ value = _email.sub(lambda match: match.group(1) + '***@' + match.group(2) + '***', value)
55
+ value = _phone.sub(lambda match: match.group(0)[:2] + '***' + match.group(0)[-2:], value)
56
+ # Node counts and cuts UTF-8 bytes, with lone surrogates as U+FFFD.
57
+ encoded = value.encode('utf-8', 'surrogatepass')
58
+ if len(encoded) > max_bytes:
59
+ encoded = _surrogate.sub('�', value).encode('utf-8')
60
+ cut = encoded[:max_bytes].decode('utf-8', 'replace')
61
+ if cut.endswith('�'):
62
+ cut = cut[:-1]
63
+ value = cut + '[TRUNCATED]'
64
+ return value
65
+
66
+ # text() depends only on its input (the secrets are fixed here): short texts repeat
67
+ # (event keys, library names, SQL of the same query) and are redacted once.
68
+ # Bounded, so a stream of distinct values cannot grow it.
69
+ cache = {}
70
+ keys = {}
71
+
72
+ def cached(value):
73
+ if len(value) > 512:
74
+ return text(value)
75
+ found = cache.get(value)
76
+ if found is None:
77
+ if len(cache) >= 4096:
78
+ cache.clear()
79
+ found = cache[value] = text(value)
80
+ return found
81
+
82
+ def key_of(key):
83
+ found = keys.get(key)
84
+ if found is None:
85
+ name = str(key)
86
+ if len(keys) >= 4096:
87
+ keys.clear()
88
+ found = keys[key] = (cached(name), bool(_sensitive.search(name)))
89
+ return found
90
+
91
+ def redact(value):
92
+ if isinstance(value, str):
93
+ return cached(value)
94
+ if isinstance(value, (list, tuple)):
95
+ return [redact(item) for item in value]
96
+ if isinstance(value, dict):
97
+ out = {}
98
+ for key, item in value.items():
99
+ name, sensitive = key_of(key)
100
+ out[name] = MARKER if sensitive else redact(item)
101
+ return out
102
+ return value
103
+
104
+ return redact