codetac 0.1.0 → 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,704 @@
1
+ """Network boundaries (stage 7): external HTTP, AI, S3 and email, with the
2
+ classification and the fields of src/boundaries.mjs:
3
+
4
+ boundary {kind: 'http' | 'ia' | 'ficheiros' | 'email' | ..., library, method, host, path, queryKeys, ...}
5
+ boundary-end {status, error; for AI: model, promptExcerpt, usage, answerExcerpt}
6
+
7
+ classify_http, ai_request_details and ai_response_details are ports of the
8
+ Node functions; test/vetores-http.json holds cases shared by both tests.
9
+ Headers only classify (never recorded); URLs keep the query keys, not the
10
+ values; AI bodies are read, bounded, for the usage and short excerpts.
11
+
12
+ Where each client is seen:
13
+ - http.client (urllib.request, and requests/urllib3): putrequest/putheader/
14
+ endheaders, and getresponse for the status;
15
+ - httpx and httpx2, sync and async (httpx2 carries the openai and anthropic
16
+ SDKs): the transports, whose response stream is followed to its end;
17
+ - aiohttp (client): ClientSession._request, and ClientResponse.read for AI;
18
+ - boto3/botocore: Endpoint._send, one boundary per attempt;
19
+ - smtplib: SMTP.sendmail (send_message goes through it).
20
+
21
+ Keep it importable on old Pythons (3.8+): the minimal mode records boundaries.
22
+ """
23
+ import functools
24
+ import json
25
+ import re
26
+ import weakref
27
+ import zlib
28
+ from urllib.parse import parse_qsl, urlencode, urlsplit
29
+
30
+ from .boundaries import _inside, _js_trim, start_boundary
31
+ from .hooks import register
32
+ from .servers import safe_path
33
+
34
+ LIMIT = 1048576 # bytes of an AI exchange read for the usage and excerpts
35
+ BODY = 262144 # bodies larger than this are not parsed for the classification
36
+ EXCERPT = 200
37
+
38
+ AI = {'api.openai.com': 'OpenAI', 'api.anthropic.com': 'Anthropic', 'generativelanguage.googleapis.com': 'Google',
39
+ 'api.mistral.ai': 'Mistral', 'api.groq.com': 'Groq', 'openrouter.ai': 'OpenRouter', 'api.deepseek.com': 'DeepSeek',
40
+ 'api.cohere.com': 'Cohere', 'api.together.xyz': 'Together'}
41
+ MAIL = {'api.resend.com': 'Resend', 'api.sendgrid.com': 'SendGrid', 'api.postmarkapp.com': 'Postmark',
42
+ 'api.mailgun.net': 'Mailgun', 'api.eu.mailgun.net': 'Mailgun', 'api.brevo.com': 'Brevo', 'api.twilio.com': 'Twilio'}
43
+ S3_OPERATIONS = {'GET': 'leitura', 'HEAD': 'verificação', 'PUT': 'escrita', 'POST': 'escrita', 'DELETE': 'remoção'}
44
+ SUPABASE = {'GET': 'SELECT', 'HEAD': 'SELECT', 'POST': 'INSERT', 'PATCH': 'UPDATE', 'PUT': 'UPSERT', 'DELETE': 'DELETE'}
45
+
46
+ _local = re.compile(r'^(localhost|127\.0\.0\.1|\[?::1\]?)$')
47
+ _compatible = re.compile(r'^/(v1/(chat/completions|completions|responses|messages)|api/(chat|generate))$')
48
+ _space = re.compile('[\\t\\n\\v\\f\\r \\u00a0\\u1680\\u2000-\\u200a\\u2028\\u2029\\u202f\\u205f\\u3000\\ufeff]+')
49
+
50
+
51
+ # URLs (src/boundaries.mjs splitUrl) ------------------------------------------------
52
+
53
+ def split_url(raw):
54
+ """(host, pathname) or None, the path to record, and the query's keys."""
55
+ try:
56
+ parts = urlsplit(str(raw))
57
+ if not parts.scheme or not parts.netloc:
58
+ parts = urlsplit('http://localhost' + ('' if str(raw).startswith('/') else '/') + str(raw))
59
+ host = parts.hostname or ''
60
+ parts.port # noqa: B018 (an invalid port makes the URL invalid, as in Node)
61
+ except ValueError:
62
+ return None, '[inválido]', []
63
+ if ':' in host:
64
+ host = '[%s]' % host
65
+ pathname = parts.path or '/'
66
+ keys = []
67
+ for key, _ in parse_qsl(parts.query, keep_blank_values=True):
68
+ if key not in keys:
69
+ keys.append(key)
70
+ return (host, pathname), safe_path(pathname), keys
71
+
72
+
73
+ # Bodies ----------------------------------------------------------------------------
74
+
75
+ def body_text(body):
76
+ if isinstance(body, str):
77
+ return body if len(body) <= BODY else None
78
+ if isinstance(body, (bytes, bytearray, memoryview)) and len(body) <= BODY:
79
+ return bytes(body).decode('utf-8', 'replace')
80
+ return None
81
+
82
+
83
+ def json_body(body):
84
+ text = body_text(body)
85
+ if not text:
86
+ return None
87
+ try:
88
+ return json.loads(text)
89
+ except ValueError:
90
+ return None
91
+
92
+
93
+ def _at(value, *path):
94
+ """value?.a?.[0]?.b as in JavaScript: None when any step is missing."""
95
+ for key in path:
96
+ if isinstance(key, int):
97
+ if not isinstance(value, list) or not -len(value) <= key < len(value):
98
+ return None
99
+ value = value[key]
100
+ elif isinstance(value, dict):
101
+ value = value.get(key)
102
+ else:
103
+ return None
104
+ return value
105
+
106
+
107
+ def _first(*values):
108
+ """a ?? b ?? c: the first value that is not None (the others are lambdas)."""
109
+ for value in values:
110
+ value = value() if callable(value) else value
111
+ if value is not None:
112
+ return value
113
+ return None
114
+
115
+
116
+ def _truthy(value):
117
+ return value is not None and value is not False and value != 0 and value != ''
118
+
119
+
120
+ # Classification (src/boundaries.mjs classifyHttp) ----------------------------------
121
+
122
+ def classify_http(method, url, headers, body, client):
123
+ parsed, path, query_keys = split_url(url)
124
+ host = parsed[0] if parsed else ''
125
+ base = {'kind': 'http', 'library': client, 'method': method, 'host': host, 'path': path, 'queryKeys': query_keys}
126
+ if parsed is None:
127
+ return base
128
+ pathname = parsed[1]
129
+ local = bool(_local.match(host))
130
+ if host in AI or _compatible.match(pathname):
131
+ data = json_body(body)
132
+ result = dict(base, kind='ia', provider=AI.get(host) or ('modelo local' if local else 'compatível (%s)' % host), operation=path)
133
+ if local:
134
+ result['local'] = True
135
+ if isinstance(_at(data, 'model'), str):
136
+ result['model'] = data['model']
137
+ return result
138
+ # S3-compatible storage: signed header, or a presigned URL (signature in
139
+ # the query: SigV4, or SigV2, which boto3 still makes by default). Before
140
+ # the local case: a local MinIO or LocalStack is S3.
141
+ authorization = str(headers.get('authorization', ''))
142
+ if (authorization.startswith('AWS4-HMAC-SHA256') or _truthy(headers.get('x-amz-content-sha256'))
143
+ or any(re.match(r'^x-amz-(signature|algorithm)$', key, re.IGNORECASE) for key in query_keys)
144
+ or ('AWSAccessKeyId' in query_keys and 'Signature' in query_keys)):
145
+ virtual = re.match(r'^(.+?)\.s3[.-]', host)
146
+ segments = pathname.split('/')
147
+ bucket = virtual.group(1) if virtual else (segments[1] if len(segments) > 1 else None)
148
+ result = dict(base, kind='ficheiros', provider='S3', operation=S3_OPERATIONS.get(method, method))
149
+ if bucket is not None:
150
+ result['bucket'] = bucket
151
+ size = _number(headers.get('content-length'))
152
+ if size is not None:
153
+ result['bytes'] = size
154
+ return result
155
+ if local:
156
+ return dict(base, local=True)
157
+ if host == 'api.stripe.com':
158
+ form = {}
159
+ for key, value in parse_qsl(body_text(body) or '', keep_blank_values=True):
160
+ form.setdefault(key, value) # URLSearchParams.get: the first value
161
+ mode = ('teste' if re.search('(sk|rk)_test_', authorization) else
162
+ 'produção' if re.search('(sk|rk)_live_', authorization) else 'desconhecido')
163
+ result = dict(base, kind='pagamento', provider='Stripe', operation='%s %s' % (method, path), mode=mode)
164
+ for key in ('amount', 'currency'):
165
+ if key in form:
166
+ result[key] = form[key]
167
+ return result
168
+ if host in MAIL:
169
+ data = json_body(body)
170
+ result = dict(base, kind='mensagem' if MAIL[host] == 'Twilio' else 'email', provider=MAIL[host])
171
+ to = _first(_at(data, 'to'), lambda: _at(data, 'personalizations', 0, 'to'), lambda: _at(data, 'To'))
172
+ if to is not None:
173
+ items = to if isinstance(to, list) else [to]
174
+ names = [item if isinstance(item, str) else _at(item, 'email') for item in items]
175
+ result['to'] = [name for name in names if _truthy(name)]
176
+ subject = _at(data, 'subject') if isinstance(_at(data, 'subject'), str) else _at(data, 'Subject')
177
+ if isinstance(subject, str):
178
+ result['subject'] = subject
179
+ return result
180
+ if host.endswith('.supabase.co'):
181
+ segments = (pathname.split('/') + [None] * 5)[:5]
182
+ area, version, first, second = segments[1:5]
183
+ if area == 'rest' and version == 'v1':
184
+ return dict(base, kind='base-de-dados', provider='Supabase', operation=SUPABASE.get(method, method),
185
+ tables=['rpc:%s' % second] if first == 'rpc' else [first])
186
+ if area == 'auth':
187
+ return dict(base, kind='autenticação', provider='Supabase Auth', operation=first)
188
+ if area == 'storage':
189
+ return dict(base, kind='ficheiros', provider='Supabase Storage', operation=method,
190
+ bucket=first if second is None else second)
191
+ if host == 'api.clerk.com' or host.endswith('.clerk.accounts.dev'):
192
+ return dict(base, kind='autenticação', provider='Clerk')
193
+ return base
194
+
195
+
196
+ def _number(value):
197
+ """Number(value) in JavaScript, for a header: None when it is not a number."""
198
+ if value is None:
199
+ return None
200
+ text = str(value).strip()
201
+ if not text:
202
+ return 0
203
+ try:
204
+ number = float(text)
205
+ except ValueError:
206
+ return None
207
+ if number != number or number in (float('inf'), float('-inf')):
208
+ return None
209
+ return int(number) if number.is_integer() else number
210
+
211
+
212
+ # AI: model, usage and excerpts (src/boundaries.mjs aiRequestDetails, aiResponseDetails)
213
+
214
+ def _excerpt(text):
215
+ if not isinstance(text, str) or not text:
216
+ return None
217
+ return _js_trim.sub('', _space.sub(' ', text))[:EXCERPT]
218
+
219
+
220
+ def _text_of(content):
221
+ if isinstance(content, str):
222
+ return content
223
+ if isinstance(content, list):
224
+ return ' '.join(part if isinstance(part, str) else
225
+ (_first(_at(part, 'text'), lambda: _at(part, 'input_text'), '')) for part in content)
226
+ return None
227
+
228
+
229
+ def _compact(values):
230
+ return {key: value for key, value in values.items() if value is not None}
231
+
232
+
233
+ def ai_request_details(data):
234
+ if not isinstance(data, dict):
235
+ return {}
236
+ messages = _first(data.get('messages'), lambda: data.get('contents'),
237
+ lambda: data['input'] if isinstance(data.get('input'), list) else None)
238
+ prompt = data['input'] if isinstance(data.get('input'), str) else data['prompt'] if isinstance(data.get('prompt'), str) else None
239
+ if not prompt and isinstance(messages, list):
240
+ users = [message for message in messages if _first(_at(message, 'role'), 'user') == 'user']
241
+ last = users[-1] if users else (messages[-1] if messages else None)
242
+ prompt = _first(_text_of(_at(last, 'content')), lambda: _text_of(_at(last, 'parts')))
243
+ return _compact({'model': data['model'] if isinstance(data.get('model'), str) else None,
244
+ 'promptExcerpt': _excerpt(prompt), 'stream': True if data.get('stream') is True else None})
245
+
246
+
247
+ def _usage_of(value):
248
+ usage = _first(_at(value, 'usage'), lambda: _at(value, 'response', 'usage'), lambda: _at(value, 'message', 'usage'))
249
+ if _truthy(usage):
250
+ found = {'input': _first(_at(usage, 'input_tokens'), lambda: _at(usage, 'prompt_tokens')),
251
+ 'output': _first(_at(usage, 'output_tokens'), lambda: _at(usage, 'completion_tokens'))}
252
+ if found['input'] is not None or found['output'] is not None:
253
+ return found
254
+ meta = _at(value, 'usageMetadata')
255
+ if _truthy(meta):
256
+ return {'input': _at(meta, 'promptTokenCount'), 'output': _at(meta, 'candidatesTokenCount')}
257
+ return None
258
+
259
+
260
+ def _answer_of(value):
261
+ def listed(key, kind, *path):
262
+ items = _at(value, key)
263
+ if not isinstance(items, list):
264
+ return None
265
+ found = next((item for item in items if _at(item, 'type') == kind), None)
266
+ return _at(found, *path)
267
+ return _first(_at(value, 'output_text'), lambda: _at(value, 'choices', 0, 'message', 'content'),
268
+ lambda: _at(value, 'choices', 0, 'text'), lambda: listed('output', 'message', 'content', 0, 'text'),
269
+ lambda: listed('content', 'text', 'text'), lambda: _at(value, 'candidates', 0, 'content', 'parts', 0, 'text'))
270
+
271
+
272
+ def ai_response_details(text):
273
+ result = {}
274
+ answer = ['']
275
+
276
+ def take(value):
277
+ # Named "usage" (not "tokens") so the secret redaction keeps the counts.
278
+ usage = _usage_of(value)
279
+ if usage:
280
+ result['usage'] = dict(result.get('usage', {}), **_compact(usage))
281
+ full = _answer_of(value)
282
+ if isinstance(full, str):
283
+ answer[0] = full
284
+ kind = _at(value, 'type')
285
+ delta = _first(_at(value, 'choices', 0, 'delta', 'content'),
286
+ lambda: _at(value, 'delta') if kind == 'response.output_text.delta' else None,
287
+ lambda: _at(value, 'delta', 'text') if kind == 'content_block_delta' else None)
288
+ if isinstance(delta, str) and len(answer[0]) < EXCERPT:
289
+ answer[0] += delta
290
+
291
+ try:
292
+ take(json.loads(text))
293
+ except ValueError:
294
+ for line in text.split('\n'):
295
+ if not line.startswith('data:'):
296
+ continue
297
+ try:
298
+ take(json.loads(line[5:].strip()))
299
+ except ValueError:
300
+ pass
301
+ if answer[0]:
302
+ result['answerExcerpt'] = _excerpt(answer[0])
303
+ return result
304
+
305
+
306
+ def _decoded(data, encoding):
307
+ """The bytes of a gzip or deflate body; other encodings (br, zstd) are not read (M68)."""
308
+ encoding = (encoding or '').strip().lower()
309
+ try:
310
+ if encoding in ('gzip', 'x-gzip'):
311
+ return zlib.decompressobj(16 + zlib.MAX_WBITS).decompress(data)
312
+ if encoding == 'deflate':
313
+ try:
314
+ return zlib.decompressobj().decompress(data)
315
+ except zlib.error:
316
+ return zlib.decompressobj(-zlib.MAX_WBITS).decompress(data)
317
+ except zlib.error:
318
+ return b''
319
+ return data if encoding in ('', 'identity') else b''
320
+
321
+
322
+ # One outgoing HTTP exchange ----------------------------------------------------------
323
+
324
+ class Exchange(object):
325
+ """A started HTTP boundary; for AI, the bounded bodies until the end."""
326
+
327
+ def __init__(self, details, body):
328
+ self.end = start_boundary(details)
329
+ self.ai = details.get('kind') == 'ia'
330
+ self.status = None
331
+ self.encoding = None
332
+ self.request = ai_request_details(json_body(body)) if self.ai else {}
333
+ self.received = []
334
+ self.size = 0
335
+
336
+ def collect(self, chunk):
337
+ if self.ai and self.size <= LIMIT and chunk:
338
+ self.received.append(bytes(chunk))
339
+ self.size += len(chunk)
340
+
341
+ def finish(self, error=False):
342
+ result = {}
343
+ if self.status is not None:
344
+ result['status'] = self.status
345
+ if error:
346
+ result['error'] = True
347
+ if self.ai:
348
+ try:
349
+ result.update(self.request)
350
+ result.update(ai_response_details(_decoded(b''.join(self.received), self.encoding).decode('utf-8', 'replace')))
351
+ except Exception:
352
+ pass
353
+ self.received = []
354
+ self.end(result)
355
+
356
+
357
+ def start_exchange(client, method, url, headers, body):
358
+ """An Exchange, or None inside another boundary or when it cannot be described."""
359
+ if _inside.get():
360
+ return None
361
+ try:
362
+ method = method.decode('ascii', 'replace') if isinstance(method, bytes) else str(method).upper()
363
+ return Exchange(classify_http(method, url, headers, body, client), body)
364
+ except Exception:
365
+ return None
366
+
367
+
368
+ def _header_map(items):
369
+ found = {}
370
+ for key, value in items:
371
+ key = (key.decode('latin-1') if isinstance(key, bytes) else str(key)).lower()
372
+ value = value.decode('latin-1') if isinstance(value, bytes) else str(value)
373
+ found[key] = found[key] + ', ' + value if key in found else value
374
+ return found
375
+
376
+
377
+ # http.client (urllib.request, requests/urllib3) ------------------------------------------
378
+
379
+ def _patch_http_client(module):
380
+ connection = module.HTTPConnection
381
+ if getattr(connection.putrequest, '__codetac__', False):
382
+ return
383
+ putrequest, putheader, endheaders, getresponse = (
384
+ connection.putrequest, connection.putheader, connection.endheaders, connection.getresponse)
385
+
386
+ def url_of(self, target):
387
+ if '://' in target: # a proxy receives the absolute URL
388
+ return target
389
+ host, port = getattr(self, '_tunnel_host', None) or self.host, getattr(self, '_tunnel_port', None) or self.port
390
+ https = isinstance(self, getattr(module, 'HTTPSConnection', ())) or bool(getattr(self, '_tunnel_host', None))
391
+ default = module.HTTPS_PORT if https else module.HTTP_PORT
392
+ netloc = ('[%s]' % host if ':' in host else host) + ('' if port in (None, default) else ':%d' % port)
393
+ return '%s://%s%s' % ('https' if https else 'http', netloc, target)
394
+
395
+ @functools.wraps(putrequest)
396
+ def codetac_putrequest(self, method, url, *args, **kwargs):
397
+ self.__dict__.pop('_codetac_exchange', None)
398
+ self._codetac_request = None if _inside.get() else [method, url, []]
399
+ return putrequest(self, method, url, *args, **kwargs)
400
+
401
+ @functools.wraps(putheader)
402
+ def codetac_putheader(self, header, *values):
403
+ pending = self.__dict__.get('_codetac_request')
404
+ if pending is not None:
405
+ pending[2].append((header, b', '.join(value if isinstance(value, bytes) else str(value).encode('latin-1', 'replace')
406
+ for value in values)))
407
+ return putheader(self, header, *values)
408
+
409
+ @functools.wraps(endheaders)
410
+ def codetac_endheaders(self, message_body=None, *args, **kwargs):
411
+ pending = self.__dict__.pop('_codetac_request', None)
412
+ exchange = None
413
+ if pending is not None:
414
+ library = 'urllib3' if type(self).__module__.startswith(('urllib3', 'botocore')) else 'http.client'
415
+ try:
416
+ url = url_of(self, pending[1] if isinstance(pending[1], str) else pending[1].decode('latin-1'))
417
+ except Exception:
418
+ url = '[inválido]'
419
+ exchange = start_exchange(library, pending[0], url, _header_map(pending[2]), message_body)
420
+ try:
421
+ result = endheaders(self, message_body, *args, **kwargs)
422
+ except BaseException:
423
+ if exchange is not None:
424
+ exchange.finish(error=True)
425
+ raise
426
+ if exchange is not None:
427
+ self._codetac_exchange = exchange
428
+ return result
429
+
430
+ @functools.wraps(getresponse)
431
+ def codetac_getresponse(self, *args, **kwargs):
432
+ exchange = self.__dict__.pop('_codetac_exchange', None)
433
+ try:
434
+ response = getresponse(self, *args, **kwargs)
435
+ except BaseException:
436
+ if exchange is not None:
437
+ exchange.finish(error=True)
438
+ raise
439
+ if exchange is not None:
440
+ # The status line and headers: the body is read later by the
441
+ # caller (usage and excerpts of AI are not read here: M65).
442
+ exchange.status = getattr(response, 'status', None)
443
+ exchange.finish()
444
+ return response
445
+
446
+ for name, wrapper in (('putrequest', codetac_putrequest), ('putheader', codetac_putheader),
447
+ ('endheaders', codetac_endheaders), ('getresponse', codetac_getresponse)):
448
+ wrapper.__codetac__ = True
449
+ setattr(connection, name, wrapper)
450
+
451
+
452
+ # httpx (and the SDKs of OpenAI and Anthropic) --------------------------------------------
453
+
454
+ def _httpx_exchange(library, request):
455
+ try:
456
+ body = request.content
457
+ except Exception: # a streamed body not read yet
458
+ body = None
459
+ return start_exchange(library, request.method, str(request.url), _header_map(request.headers.raw), body)
460
+
461
+
462
+ def _patch_httpx(module, library='httpx'):
463
+ """httpx, and httpx2 (its successor, used by the openai and anthropic SDKs), with the same API."""
464
+ sync, asynchronous = module.HTTPTransport, module.AsyncHTTPTransport
465
+ if getattr(sync.handle_request, '__codetac__', False):
466
+ return
467
+
468
+ class Stream(module.SyncByteStream):
469
+ """The response body, followed to its end (or close)."""
470
+
471
+ def __init__(self, stream, exchange):
472
+ self._stream, self._exchange = stream, exchange
473
+
474
+ def __iter__(self):
475
+ try:
476
+ for chunk in self._stream:
477
+ self._exchange.collect(chunk)
478
+ yield chunk
479
+ except Exception:
480
+ self._exchange.finish(error=True)
481
+ raise
482
+ self._exchange.finish()
483
+
484
+ def close(self):
485
+ try:
486
+ self._stream.close()
487
+ finally:
488
+ self._exchange.finish()
489
+
490
+ class AsyncStream(module.AsyncByteStream):
491
+ def __init__(self, stream, exchange):
492
+ self._stream, self._exchange = stream, exchange
493
+
494
+ async def __aiter__(self):
495
+ try:
496
+ async for chunk in self._stream:
497
+ self._exchange.collect(chunk)
498
+ yield chunk
499
+ except Exception:
500
+ self._exchange.finish(error=True)
501
+ raise
502
+ self._exchange.finish()
503
+
504
+ async def aclose(self):
505
+ try:
506
+ await self._stream.aclose()
507
+ finally:
508
+ self._exchange.finish()
509
+
510
+ handle_request = sync.handle_request
511
+ handle_async_request = asynchronous.handle_async_request
512
+
513
+ @functools.wraps(handle_request)
514
+ def codetac_handle_request(self, request):
515
+ exchange = _httpx_exchange(library, request)
516
+ if exchange is None:
517
+ return handle_request(self, request)
518
+ try:
519
+ response = handle_request(self, request)
520
+ except BaseException:
521
+ exchange.finish(error=True)
522
+ raise
523
+ exchange.status, exchange.encoding = response.status_code, response.headers.get('content-encoding')
524
+ response.stream = Stream(response.stream, exchange)
525
+ return response
526
+
527
+ @functools.wraps(handle_async_request)
528
+ async def codetac_handle_async_request(self, request):
529
+ exchange = _httpx_exchange(library, request)
530
+ if exchange is None:
531
+ return await handle_async_request(self, request)
532
+ try:
533
+ response = await handle_async_request(self, request)
534
+ except BaseException:
535
+ exchange.finish(error=True)
536
+ raise
537
+ exchange.status, exchange.encoding = response.status_code, response.headers.get('content-encoding')
538
+ response.stream = AsyncStream(response.stream, exchange)
539
+ return response
540
+
541
+ codetac_handle_request.__codetac__ = codetac_handle_async_request.__codetac__ = True
542
+ sync.handle_request = codetac_handle_request
543
+ asynchronous.handle_async_request = codetac_handle_async_request
544
+
545
+
546
+ # aiohttp (client) ------------------------------------------------------------------------
547
+
548
+ def _patch_aiohttp(module):
549
+ session, response_class = module.ClientSession, module.ClientResponse
550
+ if getattr(session._request, '__codetac__', False):
551
+ return
552
+ request = session._request
553
+ read = response_class.read
554
+ # AI responses waiting for their body; a response never read is dropped with it.
555
+ pending = weakref.WeakKeyDictionary()
556
+
557
+ def describe(self, method, str_or_url, kwargs):
558
+ url = str(str_or_url)
559
+ base = getattr(self, '_base_url', None)
560
+ if base is not None and '://' not in url:
561
+ url = str(base.join(module.URL(url)))
562
+ params = kwargs.get('params')
563
+ if params:
564
+ query = params if isinstance(params, str) else urlencode(list(params.items()) if hasattr(params, 'items') else list(params))
565
+ url += ('&' if '?' in url else '?') + query
566
+ headers = []
567
+ for source in (getattr(self, '_default_headers', None), kwargs.get('headers')):
568
+ if source:
569
+ headers.extend(source.items() if hasattr(source, 'items') else source)
570
+ body = kwargs.get('data')
571
+ if kwargs.get('json') is not None:
572
+ body = json.dumps(kwargs['json'])
573
+ return start_exchange('aiohttp', method, url, _header_map(headers), body if isinstance(body, (str, bytes)) else None)
574
+
575
+ @functools.wraps(request)
576
+ async def codetac_request(self, method, str_or_url, **kwargs):
577
+ try:
578
+ exchange = describe(self, method, str_or_url, kwargs)
579
+ except Exception:
580
+ exchange = None
581
+ if exchange is None:
582
+ return await request(self, method, str_or_url, **kwargs)
583
+ try:
584
+ response = await request(self, method, str_or_url, **kwargs)
585
+ except BaseException:
586
+ exchange.finish(error=True)
587
+ raise
588
+ exchange.status = response.status
589
+ if exchange.ai:
590
+ try:
591
+ pending[response] = exchange
592
+ return response
593
+ except TypeError: # not weak-referenceable in this version: no usage
594
+ exchange.ai = False
595
+ exchange.finish()
596
+ return response
597
+
598
+ @functools.wraps(read)
599
+ async def codetac_read(self):
600
+ exchange = pending.pop(self, None)
601
+ if exchange is None:
602
+ return await read(self)
603
+ try:
604
+ body = await read(self) # already decompressed by aiohttp
605
+ except BaseException:
606
+ exchange.finish(error=True)
607
+ raise
608
+ exchange.collect(body[:LIMIT])
609
+ exchange.finish()
610
+ return body
611
+
612
+ codetac_request.__codetac__ = codetac_read.__codetac__ = True
613
+ session._request = codetac_request
614
+ response_class.read = codetac_read
615
+
616
+
617
+ # boto3 / botocore -------------------------------------------------------------------------
618
+
619
+ def _aws_details(service, request):
620
+ headers = _header_map(request.headers.items())
621
+ if service == 's3':
622
+ details = classify_http(request.method, request.url, headers, None, 'botocore')
623
+ if details['kind'] == 'ficheiros':
624
+ return details
625
+ parsed, path, keys = split_url(request.url)
626
+ return {'kind': 'http', 'library': 'botocore', 'provider': 'AWS %s' % service, 'method': request.method,
627
+ 'host': parsed[0] if parsed else '', 'path': path, 'queryKeys': keys}
628
+
629
+
630
+ def _patch_botocore(module):
631
+ endpoint = module.Endpoint
632
+ send = endpoint._send
633
+ if getattr(send, '__codetac__', False):
634
+ return
635
+
636
+ @functools.wraps(send)
637
+ def codetac_send(self, request):
638
+ if _inside.get():
639
+ return send(self, request)
640
+ try:
641
+ end = start_boundary(_aws_details(getattr(self, '_endpoint_prefix', None) or '?', request))
642
+ except Exception:
643
+ return send(self, request)
644
+ token = _inside.set(True) # urllib3 underneath is the same exchange
645
+ try:
646
+ response = send(self, request)
647
+ except BaseException:
648
+ end({'error': True})
649
+ raise
650
+ finally:
651
+ _inside.reset(token)
652
+ end({'status': getattr(response, 'status_code', None)})
653
+ return response
654
+
655
+ codetac_send.__codetac__ = True
656
+ endpoint._send = codetac_send
657
+
658
+
659
+ # smtplib --------------------------------------------------------------------------------
660
+ # Only the recipients (redacted when written) and their count: never the
661
+ # sender, the subject or the message.
662
+
663
+ def _patch_smtplib(module):
664
+ smtp = module.SMTP
665
+ sendmail = smtp.sendmail
666
+ if getattr(sendmail, '__codetac__', False):
667
+ return
668
+
669
+ @functools.wraps(sendmail)
670
+ def codetac_sendmail(self, from_addr, to_addrs, *args, **kwargs):
671
+ if _inside.get():
672
+ return sendmail(self, from_addr, to_addrs, *args, **kwargs)
673
+ try:
674
+ if not isinstance(to_addrs, (str, bytes)):
675
+ to_addrs = list(to_addrs) # an iterator is read once, here
676
+ items = [to_addrs] if isinstance(to_addrs, (str, bytes)) else to_addrs
677
+ recipients = [item.decode('utf-8', 'replace') if isinstance(item, bytes) else str(item) for item in items]
678
+ end = start_boundary({'kind': 'email', 'library': 'smtplib', 'provider': 'SMTP', 'host': str(getattr(self, '_host', '') or ''),
679
+ 'to': recipients, 'count': len(recipients)})
680
+ except Exception:
681
+ return sendmail(self, from_addr, to_addrs, *args, **kwargs)
682
+ token = _inside.set(True)
683
+ try:
684
+ refused = sendmail(self, from_addr, to_addrs, *args, **kwargs)
685
+ except BaseException:
686
+ end({'error': True})
687
+ raise
688
+ finally:
689
+ _inside.reset(token)
690
+ rejected = len(refused) if isinstance(refused, dict) else 0
691
+ end({'accepted': len(recipients) - rejected, 'rejected': rejected})
692
+ return refused
693
+
694
+ codetac_sendmail.__codetac__ = True
695
+ smtp.sendmail = codetac_sendmail
696
+
697
+
698
+ def install():
699
+ register('http.client', _patch_http_client)
700
+ register('httpx', _patch_httpx)
701
+ register('httpx2', functools.partial(_patch_httpx, library='httpx2'))
702
+ register('aiohttp.client', _patch_aiohttp)
703
+ register('botocore.endpoint', _patch_botocore)
704
+ register('smtplib', _patch_smtplib)