codetac 0.1.0 → 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +10 -5
- package/readme.md +81 -14
- package/src/ai.mjs +4 -2
- package/src/boundaries.mjs +12 -9
- package/src/browser/bar.js +5 -3
- package/src/cli.mjs +251 -33
- package/src/detect-python.mjs +286 -0
- package/src/detect.mjs +21 -6
- package/src/diagnose.mjs +35 -5
- package/src/digest.mjs +18 -4
- package/src/page.mjs +19 -7
- package/src/panel.mjs +56 -14
- package/src/python/codetac_py/__init__.py +88 -0
- package/src/python/codetac_py/boundaries.py +435 -0
- package/src/python/codetac_py/capture.py +351 -0
- package/src/python/codetac_py/context.py +50 -0
- package/src/python/codetac_py/detail.py +333 -0
- package/src/python/codetac_py/files.py +135 -0
- package/src/python/codetac_py/frameworks.py +72 -0
- package/src/python/codetac_py/hooks.py +72 -0
- package/src/python/codetac_py/jinja_map.py +132 -0
- package/src/python/codetac_py/network.py +704 -0
- package/src/python/codetac_py/page.py +513 -0
- package/src/python/codetac_py/project.py +60 -0
- package/src/python/codetac_py/redact.py +104 -0
- package/src/python/codetac_py/servers.py +514 -0
- package/src/python/codetac_py/sitecustomize.py +62 -0
- package/src/python/codetac_py/writer.py +237 -0
- package/src/python/probe.py +99 -0
- package/src/recording.mjs +22 -4
- package/src/runtime.mjs +13 -1
- package/src/sentences.mjs +4 -0
- package/src/store.mjs +61 -9
|
@@ -0,0 +1,704 @@
|
|
|
1
|
+
"""Network boundaries (stage 7): external HTTP, AI, S3 and email, with the
|
|
2
|
+
classification and the fields of src/boundaries.mjs:
|
|
3
|
+
|
|
4
|
+
boundary {kind: 'http' | 'ia' | 'ficheiros' | 'email' | ..., library, method, host, path, queryKeys, ...}
|
|
5
|
+
boundary-end {status, error; for AI: model, promptExcerpt, usage, answerExcerpt}
|
|
6
|
+
|
|
7
|
+
classify_http, ai_request_details and ai_response_details are ports of the
|
|
8
|
+
Node functions; test/vetores-http.json holds cases shared by both tests.
|
|
9
|
+
Headers only classify (never recorded); URLs keep the query keys, not the
|
|
10
|
+
values; AI bodies are read, bounded, for the usage and short excerpts.
|
|
11
|
+
|
|
12
|
+
Where each client is seen:
|
|
13
|
+
- http.client (urllib.request, and requests/urllib3): putrequest/putheader/
|
|
14
|
+
endheaders, and getresponse for the status;
|
|
15
|
+
- httpx and httpx2, sync and async (httpx2 carries the openai and anthropic
|
|
16
|
+
SDKs): the transports, whose response stream is followed to its end;
|
|
17
|
+
- aiohttp (client): ClientSession._request, and ClientResponse.read for AI;
|
|
18
|
+
- boto3/botocore: Endpoint._send, one boundary per attempt;
|
|
19
|
+
- smtplib: SMTP.sendmail (send_message goes through it).
|
|
20
|
+
|
|
21
|
+
Keep it importable on old Pythons (3.8+): the minimal mode records boundaries.
|
|
22
|
+
"""
|
|
23
|
+
import functools
|
|
24
|
+
import json
|
|
25
|
+
import re
|
|
26
|
+
import weakref
|
|
27
|
+
import zlib
|
|
28
|
+
from urllib.parse import parse_qsl, urlencode, urlsplit
|
|
29
|
+
|
|
30
|
+
from .boundaries import _inside, _js_trim, start_boundary
|
|
31
|
+
from .hooks import register
|
|
32
|
+
from .servers import safe_path
|
|
33
|
+
|
|
34
|
+
LIMIT = 1048576 # bytes of an AI exchange read for the usage and excerpts
|
|
35
|
+
BODY = 262144 # bodies larger than this are not parsed for the classification
|
|
36
|
+
EXCERPT = 200
|
|
37
|
+
|
|
38
|
+
AI = {'api.openai.com': 'OpenAI', 'api.anthropic.com': 'Anthropic', 'generativelanguage.googleapis.com': 'Google',
|
|
39
|
+
'api.mistral.ai': 'Mistral', 'api.groq.com': 'Groq', 'openrouter.ai': 'OpenRouter', 'api.deepseek.com': 'DeepSeek',
|
|
40
|
+
'api.cohere.com': 'Cohere', 'api.together.xyz': 'Together'}
|
|
41
|
+
MAIL = {'api.resend.com': 'Resend', 'api.sendgrid.com': 'SendGrid', 'api.postmarkapp.com': 'Postmark',
|
|
42
|
+
'api.mailgun.net': 'Mailgun', 'api.eu.mailgun.net': 'Mailgun', 'api.brevo.com': 'Brevo', 'api.twilio.com': 'Twilio'}
|
|
43
|
+
S3_OPERATIONS = {'GET': 'leitura', 'HEAD': 'verificação', 'PUT': 'escrita', 'POST': 'escrita', 'DELETE': 'remoção'}
|
|
44
|
+
SUPABASE = {'GET': 'SELECT', 'HEAD': 'SELECT', 'POST': 'INSERT', 'PATCH': 'UPDATE', 'PUT': 'UPSERT', 'DELETE': 'DELETE'}
|
|
45
|
+
|
|
46
|
+
_local = re.compile(r'^(localhost|127\.0\.0\.1|\[?::1\]?)$')
|
|
47
|
+
_compatible = re.compile(r'^/(v1/(chat/completions|completions|responses|messages)|api/(chat|generate))$')
|
|
48
|
+
_space = re.compile('[\\t\\n\\v\\f\\r \\u00a0\\u1680\\u2000-\\u200a\\u2028\\u2029\\u202f\\u205f\\u3000\\ufeff]+')
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
# URLs (src/boundaries.mjs splitUrl) ------------------------------------------------
|
|
52
|
+
|
|
53
|
+
def split_url(raw):
|
|
54
|
+
"""(host, pathname) or None, the path to record, and the query's keys."""
|
|
55
|
+
try:
|
|
56
|
+
parts = urlsplit(str(raw))
|
|
57
|
+
if not parts.scheme or not parts.netloc:
|
|
58
|
+
parts = urlsplit('http://localhost' + ('' if str(raw).startswith('/') else '/') + str(raw))
|
|
59
|
+
host = parts.hostname or ''
|
|
60
|
+
parts.port # noqa: B018 (an invalid port makes the URL invalid, as in Node)
|
|
61
|
+
except ValueError:
|
|
62
|
+
return None, '[inválido]', []
|
|
63
|
+
if ':' in host:
|
|
64
|
+
host = '[%s]' % host
|
|
65
|
+
pathname = parts.path or '/'
|
|
66
|
+
keys = []
|
|
67
|
+
for key, _ in parse_qsl(parts.query, keep_blank_values=True):
|
|
68
|
+
if key not in keys:
|
|
69
|
+
keys.append(key)
|
|
70
|
+
return (host, pathname), safe_path(pathname), keys
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
# Bodies ----------------------------------------------------------------------------
|
|
74
|
+
|
|
75
|
+
def body_text(body):
|
|
76
|
+
if isinstance(body, str):
|
|
77
|
+
return body if len(body) <= BODY else None
|
|
78
|
+
if isinstance(body, (bytes, bytearray, memoryview)) and len(body) <= BODY:
|
|
79
|
+
return bytes(body).decode('utf-8', 'replace')
|
|
80
|
+
return None
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def json_body(body):
|
|
84
|
+
text = body_text(body)
|
|
85
|
+
if not text:
|
|
86
|
+
return None
|
|
87
|
+
try:
|
|
88
|
+
return json.loads(text)
|
|
89
|
+
except ValueError:
|
|
90
|
+
return None
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def _at(value, *path):
|
|
94
|
+
"""value?.a?.[0]?.b as in JavaScript: None when any step is missing."""
|
|
95
|
+
for key in path:
|
|
96
|
+
if isinstance(key, int):
|
|
97
|
+
if not isinstance(value, list) or not -len(value) <= key < len(value):
|
|
98
|
+
return None
|
|
99
|
+
value = value[key]
|
|
100
|
+
elif isinstance(value, dict):
|
|
101
|
+
value = value.get(key)
|
|
102
|
+
else:
|
|
103
|
+
return None
|
|
104
|
+
return value
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
def _first(*values):
|
|
108
|
+
"""a ?? b ?? c: the first value that is not None (the others are lambdas)."""
|
|
109
|
+
for value in values:
|
|
110
|
+
value = value() if callable(value) else value
|
|
111
|
+
if value is not None:
|
|
112
|
+
return value
|
|
113
|
+
return None
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
def _truthy(value):
|
|
117
|
+
return value is not None and value is not False and value != 0 and value != ''
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
# Classification (src/boundaries.mjs classifyHttp) ----------------------------------
|
|
121
|
+
|
|
122
|
+
def classify_http(method, url, headers, body, client):
|
|
123
|
+
parsed, path, query_keys = split_url(url)
|
|
124
|
+
host = parsed[0] if parsed else ''
|
|
125
|
+
base = {'kind': 'http', 'library': client, 'method': method, 'host': host, 'path': path, 'queryKeys': query_keys}
|
|
126
|
+
if parsed is None:
|
|
127
|
+
return base
|
|
128
|
+
pathname = parsed[1]
|
|
129
|
+
local = bool(_local.match(host))
|
|
130
|
+
if host in AI or _compatible.match(pathname):
|
|
131
|
+
data = json_body(body)
|
|
132
|
+
result = dict(base, kind='ia', provider=AI.get(host) or ('modelo local' if local else 'compatível (%s)' % host), operation=path)
|
|
133
|
+
if local:
|
|
134
|
+
result['local'] = True
|
|
135
|
+
if isinstance(_at(data, 'model'), str):
|
|
136
|
+
result['model'] = data['model']
|
|
137
|
+
return result
|
|
138
|
+
# S3-compatible storage: signed header, or a presigned URL (signature in
|
|
139
|
+
# the query: SigV4, or SigV2, which boto3 still makes by default). Before
|
|
140
|
+
# the local case: a local MinIO or LocalStack is S3.
|
|
141
|
+
authorization = str(headers.get('authorization', ''))
|
|
142
|
+
if (authorization.startswith('AWS4-HMAC-SHA256') or _truthy(headers.get('x-amz-content-sha256'))
|
|
143
|
+
or any(re.match(r'^x-amz-(signature|algorithm)$', key, re.IGNORECASE) for key in query_keys)
|
|
144
|
+
or ('AWSAccessKeyId' in query_keys and 'Signature' in query_keys)):
|
|
145
|
+
virtual = re.match(r'^(.+?)\.s3[.-]', host)
|
|
146
|
+
segments = pathname.split('/')
|
|
147
|
+
bucket = virtual.group(1) if virtual else (segments[1] if len(segments) > 1 else None)
|
|
148
|
+
result = dict(base, kind='ficheiros', provider='S3', operation=S3_OPERATIONS.get(method, method))
|
|
149
|
+
if bucket is not None:
|
|
150
|
+
result['bucket'] = bucket
|
|
151
|
+
size = _number(headers.get('content-length'))
|
|
152
|
+
if size is not None:
|
|
153
|
+
result['bytes'] = size
|
|
154
|
+
return result
|
|
155
|
+
if local:
|
|
156
|
+
return dict(base, local=True)
|
|
157
|
+
if host == 'api.stripe.com':
|
|
158
|
+
form = {}
|
|
159
|
+
for key, value in parse_qsl(body_text(body) or '', keep_blank_values=True):
|
|
160
|
+
form.setdefault(key, value) # URLSearchParams.get: the first value
|
|
161
|
+
mode = ('teste' if re.search('(sk|rk)_test_', authorization) else
|
|
162
|
+
'produção' if re.search('(sk|rk)_live_', authorization) else 'desconhecido')
|
|
163
|
+
result = dict(base, kind='pagamento', provider='Stripe', operation='%s %s' % (method, path), mode=mode)
|
|
164
|
+
for key in ('amount', 'currency'):
|
|
165
|
+
if key in form:
|
|
166
|
+
result[key] = form[key]
|
|
167
|
+
return result
|
|
168
|
+
if host in MAIL:
|
|
169
|
+
data = json_body(body)
|
|
170
|
+
result = dict(base, kind='mensagem' if MAIL[host] == 'Twilio' else 'email', provider=MAIL[host])
|
|
171
|
+
to = _first(_at(data, 'to'), lambda: _at(data, 'personalizations', 0, 'to'), lambda: _at(data, 'To'))
|
|
172
|
+
if to is not None:
|
|
173
|
+
items = to if isinstance(to, list) else [to]
|
|
174
|
+
names = [item if isinstance(item, str) else _at(item, 'email') for item in items]
|
|
175
|
+
result['to'] = [name for name in names if _truthy(name)]
|
|
176
|
+
subject = _at(data, 'subject') if isinstance(_at(data, 'subject'), str) else _at(data, 'Subject')
|
|
177
|
+
if isinstance(subject, str):
|
|
178
|
+
result['subject'] = subject
|
|
179
|
+
return result
|
|
180
|
+
if host.endswith('.supabase.co'):
|
|
181
|
+
segments = (pathname.split('/') + [None] * 5)[:5]
|
|
182
|
+
area, version, first, second = segments[1:5]
|
|
183
|
+
if area == 'rest' and version == 'v1':
|
|
184
|
+
return dict(base, kind='base-de-dados', provider='Supabase', operation=SUPABASE.get(method, method),
|
|
185
|
+
tables=['rpc:%s' % second] if first == 'rpc' else [first])
|
|
186
|
+
if area == 'auth':
|
|
187
|
+
return dict(base, kind='autenticação', provider='Supabase Auth', operation=first)
|
|
188
|
+
if area == 'storage':
|
|
189
|
+
return dict(base, kind='ficheiros', provider='Supabase Storage', operation=method,
|
|
190
|
+
bucket=first if second is None else second)
|
|
191
|
+
if host == 'api.clerk.com' or host.endswith('.clerk.accounts.dev'):
|
|
192
|
+
return dict(base, kind='autenticação', provider='Clerk')
|
|
193
|
+
return base
|
|
194
|
+
|
|
195
|
+
|
|
196
|
+
def _number(value):
|
|
197
|
+
"""Number(value) in JavaScript, for a header: None when it is not a number."""
|
|
198
|
+
if value is None:
|
|
199
|
+
return None
|
|
200
|
+
text = str(value).strip()
|
|
201
|
+
if not text:
|
|
202
|
+
return 0
|
|
203
|
+
try:
|
|
204
|
+
number = float(text)
|
|
205
|
+
except ValueError:
|
|
206
|
+
return None
|
|
207
|
+
if number != number or number in (float('inf'), float('-inf')):
|
|
208
|
+
return None
|
|
209
|
+
return int(number) if number.is_integer() else number
|
|
210
|
+
|
|
211
|
+
|
|
212
|
+
# AI: model, usage and excerpts (src/boundaries.mjs aiRequestDetails, aiResponseDetails)
|
|
213
|
+
|
|
214
|
+
def _excerpt(text):
|
|
215
|
+
if not isinstance(text, str) or not text:
|
|
216
|
+
return None
|
|
217
|
+
return _js_trim.sub('', _space.sub(' ', text))[:EXCERPT]
|
|
218
|
+
|
|
219
|
+
|
|
220
|
+
def _text_of(content):
|
|
221
|
+
if isinstance(content, str):
|
|
222
|
+
return content
|
|
223
|
+
if isinstance(content, list):
|
|
224
|
+
return ' '.join(part if isinstance(part, str) else
|
|
225
|
+
(_first(_at(part, 'text'), lambda: _at(part, 'input_text'), '')) for part in content)
|
|
226
|
+
return None
|
|
227
|
+
|
|
228
|
+
|
|
229
|
+
def _compact(values):
|
|
230
|
+
return {key: value for key, value in values.items() if value is not None}
|
|
231
|
+
|
|
232
|
+
|
|
233
|
+
def ai_request_details(data):
|
|
234
|
+
if not isinstance(data, dict):
|
|
235
|
+
return {}
|
|
236
|
+
messages = _first(data.get('messages'), lambda: data.get('contents'),
|
|
237
|
+
lambda: data['input'] if isinstance(data.get('input'), list) else None)
|
|
238
|
+
prompt = data['input'] if isinstance(data.get('input'), str) else data['prompt'] if isinstance(data.get('prompt'), str) else None
|
|
239
|
+
if not prompt and isinstance(messages, list):
|
|
240
|
+
users = [message for message in messages if _first(_at(message, 'role'), 'user') == 'user']
|
|
241
|
+
last = users[-1] if users else (messages[-1] if messages else None)
|
|
242
|
+
prompt = _first(_text_of(_at(last, 'content')), lambda: _text_of(_at(last, 'parts')))
|
|
243
|
+
return _compact({'model': data['model'] if isinstance(data.get('model'), str) else None,
|
|
244
|
+
'promptExcerpt': _excerpt(prompt), 'stream': True if data.get('stream') is True else None})
|
|
245
|
+
|
|
246
|
+
|
|
247
|
+
def _usage_of(value):
|
|
248
|
+
usage = _first(_at(value, 'usage'), lambda: _at(value, 'response', 'usage'), lambda: _at(value, 'message', 'usage'))
|
|
249
|
+
if _truthy(usage):
|
|
250
|
+
found = {'input': _first(_at(usage, 'input_tokens'), lambda: _at(usage, 'prompt_tokens')),
|
|
251
|
+
'output': _first(_at(usage, 'output_tokens'), lambda: _at(usage, 'completion_tokens'))}
|
|
252
|
+
if found['input'] is not None or found['output'] is not None:
|
|
253
|
+
return found
|
|
254
|
+
meta = _at(value, 'usageMetadata')
|
|
255
|
+
if _truthy(meta):
|
|
256
|
+
return {'input': _at(meta, 'promptTokenCount'), 'output': _at(meta, 'candidatesTokenCount')}
|
|
257
|
+
return None
|
|
258
|
+
|
|
259
|
+
|
|
260
|
+
def _answer_of(value):
|
|
261
|
+
def listed(key, kind, *path):
|
|
262
|
+
items = _at(value, key)
|
|
263
|
+
if not isinstance(items, list):
|
|
264
|
+
return None
|
|
265
|
+
found = next((item for item in items if _at(item, 'type') == kind), None)
|
|
266
|
+
return _at(found, *path)
|
|
267
|
+
return _first(_at(value, 'output_text'), lambda: _at(value, 'choices', 0, 'message', 'content'),
|
|
268
|
+
lambda: _at(value, 'choices', 0, 'text'), lambda: listed('output', 'message', 'content', 0, 'text'),
|
|
269
|
+
lambda: listed('content', 'text', 'text'), lambda: _at(value, 'candidates', 0, 'content', 'parts', 0, 'text'))
|
|
270
|
+
|
|
271
|
+
|
|
272
|
+
def ai_response_details(text):
|
|
273
|
+
result = {}
|
|
274
|
+
answer = ['']
|
|
275
|
+
|
|
276
|
+
def take(value):
|
|
277
|
+
# Named "usage" (not "tokens") so the secret redaction keeps the counts.
|
|
278
|
+
usage = _usage_of(value)
|
|
279
|
+
if usage:
|
|
280
|
+
result['usage'] = dict(result.get('usage', {}), **_compact(usage))
|
|
281
|
+
full = _answer_of(value)
|
|
282
|
+
if isinstance(full, str):
|
|
283
|
+
answer[0] = full
|
|
284
|
+
kind = _at(value, 'type')
|
|
285
|
+
delta = _first(_at(value, 'choices', 0, 'delta', 'content'),
|
|
286
|
+
lambda: _at(value, 'delta') if kind == 'response.output_text.delta' else None,
|
|
287
|
+
lambda: _at(value, 'delta', 'text') if kind == 'content_block_delta' else None)
|
|
288
|
+
if isinstance(delta, str) and len(answer[0]) < EXCERPT:
|
|
289
|
+
answer[0] += delta
|
|
290
|
+
|
|
291
|
+
try:
|
|
292
|
+
take(json.loads(text))
|
|
293
|
+
except ValueError:
|
|
294
|
+
for line in text.split('\n'):
|
|
295
|
+
if not line.startswith('data:'):
|
|
296
|
+
continue
|
|
297
|
+
try:
|
|
298
|
+
take(json.loads(line[5:].strip()))
|
|
299
|
+
except ValueError:
|
|
300
|
+
pass
|
|
301
|
+
if answer[0]:
|
|
302
|
+
result['answerExcerpt'] = _excerpt(answer[0])
|
|
303
|
+
return result
|
|
304
|
+
|
|
305
|
+
|
|
306
|
+
def _decoded(data, encoding):
|
|
307
|
+
"""The bytes of a gzip or deflate body; other encodings (br, zstd) are not read (M68)."""
|
|
308
|
+
encoding = (encoding or '').strip().lower()
|
|
309
|
+
try:
|
|
310
|
+
if encoding in ('gzip', 'x-gzip'):
|
|
311
|
+
return zlib.decompressobj(16 + zlib.MAX_WBITS).decompress(data)
|
|
312
|
+
if encoding == 'deflate':
|
|
313
|
+
try:
|
|
314
|
+
return zlib.decompressobj().decompress(data)
|
|
315
|
+
except zlib.error:
|
|
316
|
+
return zlib.decompressobj(-zlib.MAX_WBITS).decompress(data)
|
|
317
|
+
except zlib.error:
|
|
318
|
+
return b''
|
|
319
|
+
return data if encoding in ('', 'identity') else b''
|
|
320
|
+
|
|
321
|
+
|
|
322
|
+
# One outgoing HTTP exchange ----------------------------------------------------------
|
|
323
|
+
|
|
324
|
+
class Exchange(object):
|
|
325
|
+
"""A started HTTP boundary; for AI, the bounded bodies until the end."""
|
|
326
|
+
|
|
327
|
+
def __init__(self, details, body):
|
|
328
|
+
self.end = start_boundary(details)
|
|
329
|
+
self.ai = details.get('kind') == 'ia'
|
|
330
|
+
self.status = None
|
|
331
|
+
self.encoding = None
|
|
332
|
+
self.request = ai_request_details(json_body(body)) if self.ai else {}
|
|
333
|
+
self.received = []
|
|
334
|
+
self.size = 0
|
|
335
|
+
|
|
336
|
+
def collect(self, chunk):
|
|
337
|
+
if self.ai and self.size <= LIMIT and chunk:
|
|
338
|
+
self.received.append(bytes(chunk))
|
|
339
|
+
self.size += len(chunk)
|
|
340
|
+
|
|
341
|
+
def finish(self, error=False):
|
|
342
|
+
result = {}
|
|
343
|
+
if self.status is not None:
|
|
344
|
+
result['status'] = self.status
|
|
345
|
+
if error:
|
|
346
|
+
result['error'] = True
|
|
347
|
+
if self.ai:
|
|
348
|
+
try:
|
|
349
|
+
result.update(self.request)
|
|
350
|
+
result.update(ai_response_details(_decoded(b''.join(self.received), self.encoding).decode('utf-8', 'replace')))
|
|
351
|
+
except Exception:
|
|
352
|
+
pass
|
|
353
|
+
self.received = []
|
|
354
|
+
self.end(result)
|
|
355
|
+
|
|
356
|
+
|
|
357
|
+
def start_exchange(client, method, url, headers, body):
|
|
358
|
+
"""An Exchange, or None inside another boundary or when it cannot be described."""
|
|
359
|
+
if _inside.get():
|
|
360
|
+
return None
|
|
361
|
+
try:
|
|
362
|
+
method = method.decode('ascii', 'replace') if isinstance(method, bytes) else str(method).upper()
|
|
363
|
+
return Exchange(classify_http(method, url, headers, body, client), body)
|
|
364
|
+
except Exception:
|
|
365
|
+
return None
|
|
366
|
+
|
|
367
|
+
|
|
368
|
+
def _header_map(items):
|
|
369
|
+
found = {}
|
|
370
|
+
for key, value in items:
|
|
371
|
+
key = (key.decode('latin-1') if isinstance(key, bytes) else str(key)).lower()
|
|
372
|
+
value = value.decode('latin-1') if isinstance(value, bytes) else str(value)
|
|
373
|
+
found[key] = found[key] + ', ' + value if key in found else value
|
|
374
|
+
return found
|
|
375
|
+
|
|
376
|
+
|
|
377
|
+
# http.client (urllib.request, requests/urllib3) ------------------------------------------
|
|
378
|
+
|
|
379
|
+
def _patch_http_client(module):
|
|
380
|
+
connection = module.HTTPConnection
|
|
381
|
+
if getattr(connection.putrequest, '__codetac__', False):
|
|
382
|
+
return
|
|
383
|
+
putrequest, putheader, endheaders, getresponse = (
|
|
384
|
+
connection.putrequest, connection.putheader, connection.endheaders, connection.getresponse)
|
|
385
|
+
|
|
386
|
+
def url_of(self, target):
|
|
387
|
+
if '://' in target: # a proxy receives the absolute URL
|
|
388
|
+
return target
|
|
389
|
+
host, port = getattr(self, '_tunnel_host', None) or self.host, getattr(self, '_tunnel_port', None) or self.port
|
|
390
|
+
https = isinstance(self, getattr(module, 'HTTPSConnection', ())) or bool(getattr(self, '_tunnel_host', None))
|
|
391
|
+
default = module.HTTPS_PORT if https else module.HTTP_PORT
|
|
392
|
+
netloc = ('[%s]' % host if ':' in host else host) + ('' if port in (None, default) else ':%d' % port)
|
|
393
|
+
return '%s://%s%s' % ('https' if https else 'http', netloc, target)
|
|
394
|
+
|
|
395
|
+
@functools.wraps(putrequest)
|
|
396
|
+
def codetac_putrequest(self, method, url, *args, **kwargs):
|
|
397
|
+
self.__dict__.pop('_codetac_exchange', None)
|
|
398
|
+
self._codetac_request = None if _inside.get() else [method, url, []]
|
|
399
|
+
return putrequest(self, method, url, *args, **kwargs)
|
|
400
|
+
|
|
401
|
+
@functools.wraps(putheader)
|
|
402
|
+
def codetac_putheader(self, header, *values):
|
|
403
|
+
pending = self.__dict__.get('_codetac_request')
|
|
404
|
+
if pending is not None:
|
|
405
|
+
pending[2].append((header, b', '.join(value if isinstance(value, bytes) else str(value).encode('latin-1', 'replace')
|
|
406
|
+
for value in values)))
|
|
407
|
+
return putheader(self, header, *values)
|
|
408
|
+
|
|
409
|
+
@functools.wraps(endheaders)
|
|
410
|
+
def codetac_endheaders(self, message_body=None, *args, **kwargs):
|
|
411
|
+
pending = self.__dict__.pop('_codetac_request', None)
|
|
412
|
+
exchange = None
|
|
413
|
+
if pending is not None:
|
|
414
|
+
library = 'urllib3' if type(self).__module__.startswith(('urllib3', 'botocore')) else 'http.client'
|
|
415
|
+
try:
|
|
416
|
+
url = url_of(self, pending[1] if isinstance(pending[1], str) else pending[1].decode('latin-1'))
|
|
417
|
+
except Exception:
|
|
418
|
+
url = '[inválido]'
|
|
419
|
+
exchange = start_exchange(library, pending[0], url, _header_map(pending[2]), message_body)
|
|
420
|
+
try:
|
|
421
|
+
result = endheaders(self, message_body, *args, **kwargs)
|
|
422
|
+
except BaseException:
|
|
423
|
+
if exchange is not None:
|
|
424
|
+
exchange.finish(error=True)
|
|
425
|
+
raise
|
|
426
|
+
if exchange is not None:
|
|
427
|
+
self._codetac_exchange = exchange
|
|
428
|
+
return result
|
|
429
|
+
|
|
430
|
+
@functools.wraps(getresponse)
|
|
431
|
+
def codetac_getresponse(self, *args, **kwargs):
|
|
432
|
+
exchange = self.__dict__.pop('_codetac_exchange', None)
|
|
433
|
+
try:
|
|
434
|
+
response = getresponse(self, *args, **kwargs)
|
|
435
|
+
except BaseException:
|
|
436
|
+
if exchange is not None:
|
|
437
|
+
exchange.finish(error=True)
|
|
438
|
+
raise
|
|
439
|
+
if exchange is not None:
|
|
440
|
+
# The status line and headers: the body is read later by the
|
|
441
|
+
# caller (usage and excerpts of AI are not read here: M65).
|
|
442
|
+
exchange.status = getattr(response, 'status', None)
|
|
443
|
+
exchange.finish()
|
|
444
|
+
return response
|
|
445
|
+
|
|
446
|
+
for name, wrapper in (('putrequest', codetac_putrequest), ('putheader', codetac_putheader),
|
|
447
|
+
('endheaders', codetac_endheaders), ('getresponse', codetac_getresponse)):
|
|
448
|
+
wrapper.__codetac__ = True
|
|
449
|
+
setattr(connection, name, wrapper)
|
|
450
|
+
|
|
451
|
+
|
|
452
|
+
# httpx (and the SDKs of OpenAI and Anthropic) --------------------------------------------
|
|
453
|
+
|
|
454
|
+
def _httpx_exchange(library, request):
|
|
455
|
+
try:
|
|
456
|
+
body = request.content
|
|
457
|
+
except Exception: # a streamed body not read yet
|
|
458
|
+
body = None
|
|
459
|
+
return start_exchange(library, request.method, str(request.url), _header_map(request.headers.raw), body)
|
|
460
|
+
|
|
461
|
+
|
|
462
|
+
def _patch_httpx(module, library='httpx'):
|
|
463
|
+
"""httpx, and httpx2 (its successor, used by the openai and anthropic SDKs), with the same API."""
|
|
464
|
+
sync, asynchronous = module.HTTPTransport, module.AsyncHTTPTransport
|
|
465
|
+
if getattr(sync.handle_request, '__codetac__', False):
|
|
466
|
+
return
|
|
467
|
+
|
|
468
|
+
class Stream(module.SyncByteStream):
|
|
469
|
+
"""The response body, followed to its end (or close)."""
|
|
470
|
+
|
|
471
|
+
def __init__(self, stream, exchange):
|
|
472
|
+
self._stream, self._exchange = stream, exchange
|
|
473
|
+
|
|
474
|
+
def __iter__(self):
|
|
475
|
+
try:
|
|
476
|
+
for chunk in self._stream:
|
|
477
|
+
self._exchange.collect(chunk)
|
|
478
|
+
yield chunk
|
|
479
|
+
except Exception:
|
|
480
|
+
self._exchange.finish(error=True)
|
|
481
|
+
raise
|
|
482
|
+
self._exchange.finish()
|
|
483
|
+
|
|
484
|
+
def close(self):
|
|
485
|
+
try:
|
|
486
|
+
self._stream.close()
|
|
487
|
+
finally:
|
|
488
|
+
self._exchange.finish()
|
|
489
|
+
|
|
490
|
+
class AsyncStream(module.AsyncByteStream):
|
|
491
|
+
def __init__(self, stream, exchange):
|
|
492
|
+
self._stream, self._exchange = stream, exchange
|
|
493
|
+
|
|
494
|
+
async def __aiter__(self):
|
|
495
|
+
try:
|
|
496
|
+
async for chunk in self._stream:
|
|
497
|
+
self._exchange.collect(chunk)
|
|
498
|
+
yield chunk
|
|
499
|
+
except Exception:
|
|
500
|
+
self._exchange.finish(error=True)
|
|
501
|
+
raise
|
|
502
|
+
self._exchange.finish()
|
|
503
|
+
|
|
504
|
+
async def aclose(self):
|
|
505
|
+
try:
|
|
506
|
+
await self._stream.aclose()
|
|
507
|
+
finally:
|
|
508
|
+
self._exchange.finish()
|
|
509
|
+
|
|
510
|
+
handle_request = sync.handle_request
|
|
511
|
+
handle_async_request = asynchronous.handle_async_request
|
|
512
|
+
|
|
513
|
+
@functools.wraps(handle_request)
|
|
514
|
+
def codetac_handle_request(self, request):
|
|
515
|
+
exchange = _httpx_exchange(library, request)
|
|
516
|
+
if exchange is None:
|
|
517
|
+
return handle_request(self, request)
|
|
518
|
+
try:
|
|
519
|
+
response = handle_request(self, request)
|
|
520
|
+
except BaseException:
|
|
521
|
+
exchange.finish(error=True)
|
|
522
|
+
raise
|
|
523
|
+
exchange.status, exchange.encoding = response.status_code, response.headers.get('content-encoding')
|
|
524
|
+
response.stream = Stream(response.stream, exchange)
|
|
525
|
+
return response
|
|
526
|
+
|
|
527
|
+
@functools.wraps(handle_async_request)
|
|
528
|
+
async def codetac_handle_async_request(self, request):
|
|
529
|
+
exchange = _httpx_exchange(library, request)
|
|
530
|
+
if exchange is None:
|
|
531
|
+
return await handle_async_request(self, request)
|
|
532
|
+
try:
|
|
533
|
+
response = await handle_async_request(self, request)
|
|
534
|
+
except BaseException:
|
|
535
|
+
exchange.finish(error=True)
|
|
536
|
+
raise
|
|
537
|
+
exchange.status, exchange.encoding = response.status_code, response.headers.get('content-encoding')
|
|
538
|
+
response.stream = AsyncStream(response.stream, exchange)
|
|
539
|
+
return response
|
|
540
|
+
|
|
541
|
+
codetac_handle_request.__codetac__ = codetac_handle_async_request.__codetac__ = True
|
|
542
|
+
sync.handle_request = codetac_handle_request
|
|
543
|
+
asynchronous.handle_async_request = codetac_handle_async_request
|
|
544
|
+
|
|
545
|
+
|
|
546
|
+
# aiohttp (client) ------------------------------------------------------------------------
|
|
547
|
+
|
|
548
|
+
def _patch_aiohttp(module):
|
|
549
|
+
session, response_class = module.ClientSession, module.ClientResponse
|
|
550
|
+
if getattr(session._request, '__codetac__', False):
|
|
551
|
+
return
|
|
552
|
+
request = session._request
|
|
553
|
+
read = response_class.read
|
|
554
|
+
# AI responses waiting for their body; a response never read is dropped with it.
|
|
555
|
+
pending = weakref.WeakKeyDictionary()
|
|
556
|
+
|
|
557
|
+
def describe(self, method, str_or_url, kwargs):
|
|
558
|
+
url = str(str_or_url)
|
|
559
|
+
base = getattr(self, '_base_url', None)
|
|
560
|
+
if base is not None and '://' not in url:
|
|
561
|
+
url = str(base.join(module.URL(url)))
|
|
562
|
+
params = kwargs.get('params')
|
|
563
|
+
if params:
|
|
564
|
+
query = params if isinstance(params, str) else urlencode(list(params.items()) if hasattr(params, 'items') else list(params))
|
|
565
|
+
url += ('&' if '?' in url else '?') + query
|
|
566
|
+
headers = []
|
|
567
|
+
for source in (getattr(self, '_default_headers', None), kwargs.get('headers')):
|
|
568
|
+
if source:
|
|
569
|
+
headers.extend(source.items() if hasattr(source, 'items') else source)
|
|
570
|
+
body = kwargs.get('data')
|
|
571
|
+
if kwargs.get('json') is not None:
|
|
572
|
+
body = json.dumps(kwargs['json'])
|
|
573
|
+
return start_exchange('aiohttp', method, url, _header_map(headers), body if isinstance(body, (str, bytes)) else None)
|
|
574
|
+
|
|
575
|
+
@functools.wraps(request)
|
|
576
|
+
async def codetac_request(self, method, str_or_url, **kwargs):
|
|
577
|
+
try:
|
|
578
|
+
exchange = describe(self, method, str_or_url, kwargs)
|
|
579
|
+
except Exception:
|
|
580
|
+
exchange = None
|
|
581
|
+
if exchange is None:
|
|
582
|
+
return await request(self, method, str_or_url, **kwargs)
|
|
583
|
+
try:
|
|
584
|
+
response = await request(self, method, str_or_url, **kwargs)
|
|
585
|
+
except BaseException:
|
|
586
|
+
exchange.finish(error=True)
|
|
587
|
+
raise
|
|
588
|
+
exchange.status = response.status
|
|
589
|
+
if exchange.ai:
|
|
590
|
+
try:
|
|
591
|
+
pending[response] = exchange
|
|
592
|
+
return response
|
|
593
|
+
except TypeError: # not weak-referenceable in this version: no usage
|
|
594
|
+
exchange.ai = False
|
|
595
|
+
exchange.finish()
|
|
596
|
+
return response
|
|
597
|
+
|
|
598
|
+
@functools.wraps(read)
|
|
599
|
+
async def codetac_read(self):
|
|
600
|
+
exchange = pending.pop(self, None)
|
|
601
|
+
if exchange is None:
|
|
602
|
+
return await read(self)
|
|
603
|
+
try:
|
|
604
|
+
body = await read(self) # already decompressed by aiohttp
|
|
605
|
+
except BaseException:
|
|
606
|
+
exchange.finish(error=True)
|
|
607
|
+
raise
|
|
608
|
+
exchange.collect(body[:LIMIT])
|
|
609
|
+
exchange.finish()
|
|
610
|
+
return body
|
|
611
|
+
|
|
612
|
+
codetac_request.__codetac__ = codetac_read.__codetac__ = True
|
|
613
|
+
session._request = codetac_request
|
|
614
|
+
response_class.read = codetac_read
|
|
615
|
+
|
|
616
|
+
|
|
617
|
+
# boto3 / botocore -------------------------------------------------------------------------
|
|
618
|
+
|
|
619
|
+
def _aws_details(service, request):
|
|
620
|
+
headers = _header_map(request.headers.items())
|
|
621
|
+
if service == 's3':
|
|
622
|
+
details = classify_http(request.method, request.url, headers, None, 'botocore')
|
|
623
|
+
if details['kind'] == 'ficheiros':
|
|
624
|
+
return details
|
|
625
|
+
parsed, path, keys = split_url(request.url)
|
|
626
|
+
return {'kind': 'http', 'library': 'botocore', 'provider': 'AWS %s' % service, 'method': request.method,
|
|
627
|
+
'host': parsed[0] if parsed else '', 'path': path, 'queryKeys': keys}
|
|
628
|
+
|
|
629
|
+
|
|
630
|
+
def _patch_botocore(module):
|
|
631
|
+
endpoint = module.Endpoint
|
|
632
|
+
send = endpoint._send
|
|
633
|
+
if getattr(send, '__codetac__', False):
|
|
634
|
+
return
|
|
635
|
+
|
|
636
|
+
@functools.wraps(send)
|
|
637
|
+
def codetac_send(self, request):
|
|
638
|
+
if _inside.get():
|
|
639
|
+
return send(self, request)
|
|
640
|
+
try:
|
|
641
|
+
end = start_boundary(_aws_details(getattr(self, '_endpoint_prefix', None) or '?', request))
|
|
642
|
+
except Exception:
|
|
643
|
+
return send(self, request)
|
|
644
|
+
token = _inside.set(True) # urllib3 underneath is the same exchange
|
|
645
|
+
try:
|
|
646
|
+
response = send(self, request)
|
|
647
|
+
except BaseException:
|
|
648
|
+
end({'error': True})
|
|
649
|
+
raise
|
|
650
|
+
finally:
|
|
651
|
+
_inside.reset(token)
|
|
652
|
+
end({'status': getattr(response, 'status_code', None)})
|
|
653
|
+
return response
|
|
654
|
+
|
|
655
|
+
codetac_send.__codetac__ = True
|
|
656
|
+
endpoint._send = codetac_send
|
|
657
|
+
|
|
658
|
+
|
|
659
|
+
# smtplib --------------------------------------------------------------------------------
|
|
660
|
+
# Only the recipients (redacted when written) and their count: never the
|
|
661
|
+
# sender, the subject or the message.
|
|
662
|
+
|
|
663
|
+
def _patch_smtplib(module):
|
|
664
|
+
smtp = module.SMTP
|
|
665
|
+
sendmail = smtp.sendmail
|
|
666
|
+
if getattr(sendmail, '__codetac__', False):
|
|
667
|
+
return
|
|
668
|
+
|
|
669
|
+
@functools.wraps(sendmail)
|
|
670
|
+
def codetac_sendmail(self, from_addr, to_addrs, *args, **kwargs):
|
|
671
|
+
if _inside.get():
|
|
672
|
+
return sendmail(self, from_addr, to_addrs, *args, **kwargs)
|
|
673
|
+
try:
|
|
674
|
+
if not isinstance(to_addrs, (str, bytes)):
|
|
675
|
+
to_addrs = list(to_addrs) # an iterator is read once, here
|
|
676
|
+
items = [to_addrs] if isinstance(to_addrs, (str, bytes)) else to_addrs
|
|
677
|
+
recipients = [item.decode('utf-8', 'replace') if isinstance(item, bytes) else str(item) for item in items]
|
|
678
|
+
end = start_boundary({'kind': 'email', 'library': 'smtplib', 'provider': 'SMTP', 'host': str(getattr(self, '_host', '') or ''),
|
|
679
|
+
'to': recipients, 'count': len(recipients)})
|
|
680
|
+
except Exception:
|
|
681
|
+
return sendmail(self, from_addr, to_addrs, *args, **kwargs)
|
|
682
|
+
token = _inside.set(True)
|
|
683
|
+
try:
|
|
684
|
+
refused = sendmail(self, from_addr, to_addrs, *args, **kwargs)
|
|
685
|
+
except BaseException:
|
|
686
|
+
end({'error': True})
|
|
687
|
+
raise
|
|
688
|
+
finally:
|
|
689
|
+
_inside.reset(token)
|
|
690
|
+
rejected = len(refused) if isinstance(refused, dict) else 0
|
|
691
|
+
end({'accepted': len(recipients) - rejected, 'rejected': rejected})
|
|
692
|
+
return refused
|
|
693
|
+
|
|
694
|
+
codetac_sendmail.__codetac__ = True
|
|
695
|
+
smtp.sendmail = codetac_sendmail
|
|
696
|
+
|
|
697
|
+
|
|
698
|
+
def install():
|
|
699
|
+
register('http.client', _patch_http_client)
|
|
700
|
+
register('httpx', _patch_httpx)
|
|
701
|
+
register('httpx2', functools.partial(_patch_httpx, library='httpx2'))
|
|
702
|
+
register('aiohttp.client', _patch_aiohttp)
|
|
703
|
+
register('botocore.endpoint', _patch_botocore)
|
|
704
|
+
register('smtplib', _patch_smtplib)
|