borges 1.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (55) hide show
  1. borges/__init__.py +20 -0
  2. borges/analyzers/__init__.py +14 -0
  3. borges/analyzers/llm_analyzer.py +500 -0
  4. borges/analyzers/network_consolidator.py +1989 -0
  5. borges/analyzers/number_validator.py +311 -0
  6. borges/analyzers/redirect_analyzer.py +203 -0
  7. borges/analyzers/whois_analyzer.py +184 -0
  8. borges/cli.py +634 -0
  9. borges/config.py +395 -0
  10. borges/data/__init__.py +46 -0
  11. borges/data/download.py +330 -0
  12. borges/data/exporters.py +319 -0
  13. borges/data/loaders.py +510 -0
  14. borges/data/processors.py +476 -0
  15. borges/defaults/config.yaml +252 -0
  16. borges/defaults/negative_samples/bootstrap_hexagon_b_variant.png +0 -0
  17. borges/defaults/negative_samples/bootstrap_purple_square_b.png +0 -0
  18. borges/defaults/negative_samples/chinese_hosting_text_logo.png +0 -0
  19. borges/defaults/negative_samples/facebook_blue_f_logo.png +0 -0
  20. borges/defaults/negative_samples/generic_corporate_logo.png +0 -0
  21. borges/defaults/negative_samples/generic_globe_web_icon.png +0 -0
  22. borges/defaults/negative_samples/generic_home_house_icon.png +0 -0
  23. borges/defaults/negative_samples/generic_hosting_server_icon.png +0 -0
  24. borges/defaults/negative_samples/generic_isp_branding.png +0 -0
  25. borges/defaults/negative_samples/generic_platform_icon.png +0 -0
  26. borges/defaults/negative_samples/generic_provider_logo.png +0 -0
  27. borges/defaults/negative_samples/generic_warning_triangle.png +0 -0
  28. borges/defaults/negative_samples/google_play_triangular_button.png +0 -0
  29. borges/defaults/negative_samples/mega_group_209asn_generic_e99155e1_urls.txt +28 -0
  30. borges/defaults/negative_samples/ngnix.png +0 -0
  31. borges/defaults/negative_samples/peeringdb_default_favicon.png +0 -0
  32. borges/defaults/negative_samples/unknown_generic_favicon.png +0 -0
  33. borges/defaults/negative_samples/wordpress.png +0 -0
  34. borges/defaults/negative_samples/wordpress2.png +0 -0
  35. borges/defaults/negative_samples/wordpress_default_w_logo.png +0 -0
  36. borges/models/__init__.py +27 -0
  37. borges/models/as_network.py +330 -0
  38. borges/models/schemas.py +212 -0
  39. borges/pipeline/__init__.py +27 -0
  40. borges/pipeline/runner.py +409 -0
  41. borges/pipeline/stages.py +1404 -0
  42. borges/resources.py +34 -0
  43. borges/scrapers/__init__.py +11 -0
  44. borges/scrapers/favicon_scraper.py +269 -0
  45. borges/scrapers/html_scraper.py +239 -0
  46. borges/scrapers/redirect_scraper.py +249 -0
  47. borges/utils/__init__.py +19 -0
  48. borges/utils/http_client.py +231 -0
  49. borges/utils/llm_client.py +231 -0
  50. borges/utils/logging.py +171 -0
  51. borges-1.1.0.dist-info/METADATA +321 -0
  52. borges-1.1.0.dist-info/RECORD +55 -0
  53. borges-1.1.0.dist-info/WHEEL +4 -0
  54. borges-1.1.0.dist-info/entry_points.txt +2 -0
  55. borges-1.1.0.dist-info/licenses/LICENSE +21 -0
borges/__init__.py ADDED
@@ -0,0 +1,20 @@
1
+ """Borges - AS Sibling Relationship Inference System.
2
+
3
+ A tool for inferring sibling relationships between Autonomous Systems using
4
+ data from PeeringDB, WHOIS, and web scraping with AI-powered analysis.
5
+ """
6
+
7
+ __version__ = "1.1.0"
8
+
9
+ from .config import Config, get_config, load_config
10
+ from .models import ASNetwork, ASNetworkReport
11
+ from .pipeline import Pipeline
12
+
13
+ __all__ = [
14
+ "Config",
15
+ "get_config",
16
+ "load_config",
17
+ "ASNetwork",
18
+ "ASNetworkReport",
19
+ "Pipeline",
20
+ ]
@@ -0,0 +1,14 @@
1
+ """Borges analysis modules."""
2
+
3
+ from .llm_analyzer import ASRelationshipAnalyzer, FaviconAnalyzer
4
+ from .network_consolidator import NetworkGroupConsolidator
5
+ from .redirect_analyzer import RedirectAnalyzer
6
+ from .whois_analyzer import WHOISAnalyzer
7
+
8
+ __all__ = [
9
+ "ASRelationshipAnalyzer",
10
+ "FaviconAnalyzer",
11
+ "NetworkGroupConsolidator",
12
+ "RedirectAnalyzer",
13
+ "WHOISAnalyzer",
14
+ ]
@@ -0,0 +1,500 @@
1
+ """LLM-based analysis for sibling relationships and favicon identification."""
2
+
3
+ import base64
4
+ import time
5
+ from typing import Dict, List, Optional
6
+
7
+ import pandas as pd
8
+ from langchain_core.messages import HumanMessage
9
+ from langchain_core.output_parsers import JsonOutputParser
10
+ from langchain_core.prompts import PromptTemplate
11
+ from pydantic import BaseModel, Field
12
+ from langchain_openai import ChatOpenAI
13
+ from PIL import Image
14
+ from io import BytesIO
15
+
16
+ from ..config import get_config
17
+ from ..data.processors import ASNProcessor
18
+ from ..models import APIUsageStats, ASRelationship, FaviconAnalysis, NetworkGroup
19
+ from ..utils import get_logger
20
+ from .number_validator import validate_llm_output
21
+
22
+ logger = get_logger(__name__)
23
+
24
+
25
+ class ASList(BaseModel):
26
+ """Model for LLM AS detection output."""
27
+
28
+ ASs: Optional[List[int]] = Field(None, description="List of AS numbers")
29
+
30
+
31
+ class ASRelationshipAnalyzer:
32
+ """Analyze AS relationships using LLM."""
33
+
34
+ def __init__(self):
35
+ """Initialize AS relationship analyzer."""
36
+ config = get_config()
37
+ self.config = config.api.openai
38
+ self.prompt_template = config.processing.prompts.get("as_detection")
39
+ self.asn_blocklist = set(config.processing.asn_blocklist)
40
+
41
+ # Initialize LLM with rate limiting
42
+ from ..utils.llm_client import create_llm_client
43
+
44
+ self.llm_client = create_llm_client()
45
+ self.llm = self.llm_client.llm
46
+
47
+ # Initialize parser
48
+ self.parser = JsonOutputParser(pydantic_object=ASList)
49
+
50
+ # Create prompt
51
+ self.prompt = PromptTemplate(
52
+ template=self.prompt_template,
53
+ input_variables=["aka", "notes", "asn"],
54
+ partial_variables={
55
+ "format_instructions": self.parser.get_format_instructions()
56
+ },
57
+ )
58
+
59
+ # Create chain
60
+ self.chain = self.prompt | self.llm | self.parser
61
+
62
+ # Initialize cost tracking
63
+ self.api_usage = APIUsageStats()
64
+
65
+ def analyze_as(self, asn: int, notes: str, aka: str) -> ASRelationship:
66
+ """Analyze AS relationships from notes and AKA fields.
67
+
68
+ Args:
69
+ asn: Source AS number
70
+ notes: PeeringDB notes
71
+ aka: AKA field
72
+
73
+ Returns:
74
+ AS relationship
75
+ """
76
+ # Skip analysis for blocklisted ASNs to prevent upstream pollution
77
+ if asn in self.asn_blocklist:
78
+ return ASRelationship(
79
+ source_asn=asn,
80
+ related_asns=[],
81
+ relationship_type="sibling",
82
+ confidence=0.0,
83
+ sources=[],
84
+ detected_by="llm_analysis",
85
+ )
86
+
87
+ try:
88
+ # Run LLM analysis with proper rate limiting
89
+ prompt_input = {"asn": asn, "notes": notes or "", "aka": aka or ""}
90
+
91
+ # Format the prompt
92
+ formatted_prompt = self.prompt.format(**prompt_input)
93
+ messages = [{"role": "user", "content": formatted_prompt}]
94
+
95
+ # Use LLM client with rate limiting
96
+ response = self.llm_client.invoke(messages)
97
+
98
+ # Parse the response
99
+ result = self.parser.parse(response.content)
100
+
101
+ # Track API usage (estimate tokens and cost)
102
+ self._track_api_usage(notes or "", aka or "")
103
+
104
+ # Extract ASNs from LLM
105
+ llm_asns = result.get("ASs", []) if result else []
106
+
107
+ # Validate LLM output - filter out hallucinated ASNs
108
+ input_text = f"{notes or ''} {aka or ''}".strip()
109
+ validated_llm_asns = validate_llm_output(
110
+ input_text, llm_asns, source_asn=asn, blocklist=self.asn_blocklist
111
+ )
112
+
113
+ # Also extract ASNs using regex (keep existing functionality)
114
+ text_asns = []
115
+ if notes:
116
+ text_asns.extend(ASNProcessor.detect_related_asns(notes, asn))
117
+ if aka:
118
+ text_asns.extend(ASNProcessor.detect_related_asns(aka, asn))
119
+
120
+ # Combine validated LLM results with regex results
121
+ all_asns = list(set(validated_llm_asns + text_asns))
122
+ all_asns = [asn_num for asn_num in all_asns if asn_num != asn]
123
+
124
+ # Filter out blocklisted ASNs from related_asns to prevent pollution
125
+ all_asns = [
126
+ asn_num for asn_num in all_asns if asn_num not in self.asn_blocklist
127
+ ]
128
+
129
+ # Calculate dynamic confidence score
130
+ confidence = self._calculate_confidence(
131
+ llm_asns=validated_llm_asns, text_asns=text_asns, notes=notes, aka=aka
132
+ )
133
+
134
+ return ASRelationship(
135
+ source_asn=asn,
136
+ related_asns=all_asns,
137
+ relationship_type="organization_related",
138
+ confidence=confidence,
139
+ evidence=f"Notes: {notes}" if notes else f"AKA: {aka}" if aka else None,
140
+ detected_by="llm_analysis",
141
+ )
142
+
143
+ except Exception as e:
144
+ # Return empty relationship on error
145
+ return ASRelationship(
146
+ source_asn=asn,
147
+ related_asns=[],
148
+ relationship_type="organization_related",
149
+ confidence=0.0,
150
+ error=str(e),
151
+ detected_by="llm_analysis",
152
+ )
153
+
154
+ def analyze_dataframe(self, df: pd.DataFrame) -> List[ASRelationship]:
155
+ """Analyze AS relationships from a DataFrame.
156
+
157
+ Args:
158
+ df: DataFrame with asn, notes, and aka columns
159
+
160
+ Returns:
161
+ List of AS relationships
162
+ """
163
+ relationships = []
164
+
165
+ # Filter to rows that might have AS references
166
+ filtered_df = df[
167
+ (df["notes"].notna() & df["notes"].apply(ASNProcessor.has_asn_reference))
168
+ | (df["aka"].notna() & df["aka"].apply(ASNProcessor.has_asn_reference))
169
+ ]
170
+
171
+ for _, row in filtered_df.iterrows():
172
+ relationship = self.analyze_as(
173
+ asn=row["asn"], notes=row.get("notes", ""), aka=row.get("aka", "")
174
+ )
175
+ if relationship.related_asns:
176
+ relationships.append(relationship)
177
+
178
+ return relationships
179
+
180
+ def _calculate_confidence(
181
+ self, llm_asns: List[int], text_asns: List[int], notes: str, aka: str
182
+ ) -> float:
183
+ """Calculate confidence score for AS relationship detection.
184
+
185
+ Args:
186
+ llm_asns: ASNs found by LLM analysis
187
+ text_asns: ASNs found by text processing
188
+ notes: Notes field content
189
+ aka: AKA field content
190
+
191
+ Returns:
192
+ Confidence score between 0.0 and 1.0
193
+ """
194
+ if not llm_asns and not text_asns:
195
+ return 0.0
196
+
197
+ confidence = 0.0
198
+
199
+ # Base confidence for finding any ASNs
200
+ if llm_asns or text_asns:
201
+ confidence += 0.3
202
+
203
+ # Higher confidence if both LLM and text processing agree
204
+ if llm_asns and text_asns:
205
+ overlap = set(llm_asns) & set(text_asns)
206
+ if overlap:
207
+ confidence += 0.4 # Strong agreement
208
+ else:
209
+ confidence += 0.2 # Both found ASNs, but different ones
210
+
211
+ # Confidence boost based on evidence quality
212
+ evidence_text = (notes or "") + " " + (aka or "")
213
+ evidence_length = len(evidence_text.strip())
214
+
215
+ if evidence_length > 100:
216
+ confidence += 0.2 # Rich evidence
217
+ elif evidence_length > 20:
218
+ confidence += 0.1 # Some evidence
219
+
220
+ # Confidence based on number of ASNs found
221
+ total_asns = len(set(llm_asns + text_asns))
222
+ if total_asns == 1:
223
+ confidence += 0.1 # Single ASN is more reliable
224
+ elif total_asns <= 3:
225
+ confidence += 0.05 # Small group is reasonable
226
+ # No bonus for large groups (might be noisy)
227
+
228
+ # Evidence source preference (notes are more reliable than aka)
229
+ if notes and llm_asns:
230
+ confidence += 0.05
231
+
232
+ return min(confidence, 1.0) # Cap at 1.0
233
+
234
+ def _track_api_usage(self, notes: str, aka: str) -> None:
235
+ """Track API usage for cost estimation.
236
+
237
+ Args:
238
+ notes: Notes text processed
239
+ aka: AKA text processed
240
+ """
241
+ # Rough token estimation (1 token ≈ 4 characters for English text)
242
+ input_text = f"{self.prompt_template} {notes} {aka}"
243
+ estimated_input_tokens = len(input_text) // 4
244
+ estimated_output_tokens = 50 # Estimated JSON response size
245
+
246
+ # Update usage stats
247
+ self.api_usage.total_requests += 1
248
+ self.api_usage.total_input_tokens += estimated_input_tokens
249
+ self.api_usage.total_output_tokens += estimated_output_tokens
250
+
251
+ # Calculate cost based on gpt-4o-mini pricing (as of 2024)
252
+ # Input: $0.15 per 1M tokens, Output: $0.60 per 1M tokens
253
+ input_cost = (estimated_input_tokens / 1_000_000) * 0.15
254
+ output_cost = (estimated_output_tokens / 1_000_000) * 0.60
255
+ self.api_usage.estimated_cost_usd += input_cost + output_cost
256
+
257
+ def get_api_usage(self) -> APIUsageStats:
258
+ """Get current API usage statistics.
259
+
260
+ Returns:
261
+ API usage statistics
262
+ """
263
+ return self.api_usage
264
+
265
+
266
+ class FaviconAnalyzer:
267
+ """Analyze favicons using LLM vision capabilities."""
268
+
269
+ def __init__(self):
270
+ """Initialize favicon analyzer."""
271
+ config = get_config()
272
+ self.config = config.api.openai
273
+ self.prompt_template = config.processing.prompts.get("favicon_analysis")
274
+
275
+ # Initialize vision LLM with rate limiting
276
+ from ..utils.llm_client import create_llm_client
277
+
278
+ self.llm_client = create_llm_client(use_vision=True)
279
+ self.llm = self.llm_client.llm
280
+
281
+ # Load negative example images
282
+ self._load_negative_examples()
283
+
284
+ def _load_negative_examples(self):
285
+ """Load negative example favicon images for comparison."""
286
+ from pathlib import Path
287
+
288
+ from ..resources import negative_samples_dir as default_samples_dir
289
+
290
+ self.negative_examples = []
291
+ # A project's own data/reference/ overrides the samples shipped with Borges
292
+ negative_samples_dir = Path("data/reference/negative_samples")
293
+ if not negative_samples_dir.exists():
294
+ try:
295
+ negative_samples_dir = default_samples_dir()
296
+ except FileNotFoundError as e:
297
+ logger.warning(str(e))
298
+ return
299
+
300
+ # Load all PNG images from negative samples directory
301
+ for image_file in negative_samples_dir.glob("*.png"):
302
+ try:
303
+ with open(image_file, "rb") as f:
304
+ image_bytes = f.read()
305
+
306
+ # Create descriptive name from filename
307
+ name = image_file.stem.replace("_", " ").title()
308
+
309
+ # Store raw bytes for now, will encode when needed
310
+ self.negative_examples.append(
311
+ {"name": name, "filename": image_file.name, "bytes": image_bytes}
312
+ )
313
+
314
+ except Exception as e:
315
+ logger.warning(f"Failed to load negative example {image_file}: {e}")
316
+
317
+ logger.info(f"Loaded {len(self.negative_examples)} negative favicon examples")
318
+
319
+ def _encode_image(self, image_bytes: bytes) -> str:
320
+ """Encode image to base64.
321
+
322
+ Args:
323
+ image_bytes: Image binary data
324
+
325
+ Returns:
326
+ Base64 encoded string
327
+ """
328
+ return base64.b64encode(image_bytes).decode("utf-8")
329
+
330
+ def analyze_favicon(self, favicon_bytes: bytes, urls: List[str]) -> FaviconAnalysis:
331
+ """Analyze a favicon to identify company.
332
+
333
+ Args:
334
+ favicon_bytes: Favicon binary data
335
+ urls: URLs using this favicon
336
+
337
+ Returns:
338
+ Favicon analysis result
339
+ """
340
+ try:
341
+ # Encode image
342
+ image_data = self._encode_image(favicon_bytes)
343
+
344
+ # Create message content starting with text and the favicon to analyze
345
+ content = [
346
+ {
347
+ "type": "text",
348
+ "text": self.prompt_template.format(urls=", ".join(urls[:5])),
349
+ },
350
+ {"type": "text", "text": "\n\n**FAVICON TO ANALYZE:**"},
351
+ {
352
+ "type": "image_url",
353
+ "image_url": {"url": f"data:image/jpeg;base64,{image_data}"},
354
+ },
355
+ ]
356
+
357
+ # Add negative examples if available
358
+ # TEMPORARILY COMMENTED OUT: Visual negative examples consume too many tokens and cause rate limiting
359
+ # Keeping text-based negative examples in the prompt for protection
360
+ # if hasattr(self, 'negative_examples') and self.negative_examples:
361
+ # content.append({
362
+ # "type": "text",
363
+ # "text": "\n\n**NEGATIVE EXAMPLES - DO NOT GROUP if the favicon matches any of these common defaults:**"
364
+ # })
365
+ #
366
+ # # Add up to 8 negative examples to avoid token limits
367
+ # for i, example in enumerate(self.negative_examples[:8]):
368
+ # content.extend([
369
+ # {
370
+ # "type": "text",
371
+ # "text": f"\n{i+1}. {example['name']}:"
372
+ # },
373
+ # {
374
+ # "type": "image_url",
375
+ # "image_url": {"url": f"data:image/jpeg;base64,{self._encode_image(example['bytes'])}"},
376
+ # }
377
+ # ])
378
+
379
+ message = HumanMessage(content=content)
380
+
381
+ # Get LLM response with rate limiting
382
+ response = self.llm_client.invoke([message])
383
+ content = response.content.lower()
384
+
385
+ # Parse response
386
+ company_name = None
387
+ is_telecom = False
388
+ is_hosting = False
389
+ confidence = 0.0
390
+
391
+ if "no sé" not in content and "no se" not in content:
392
+ company_name = response.content.strip()
393
+ confidence = 0.8
394
+
395
+ # Check for telecom keywords
396
+ telecom_keywords = [
397
+ "telecom",
398
+ "telco",
399
+ "communications",
400
+ "móvil",
401
+ "mobile",
402
+ "fiber",
403
+ ]
404
+ if any(keyword in content for keyword in telecom_keywords):
405
+ is_telecom = True
406
+
407
+ # Check for hosting keywords
408
+ hosting_keywords = ["hosting", "cloud", "server", "data center", "cdn"]
409
+ if any(keyword in content for keyword in hosting_keywords):
410
+ is_hosting = True
411
+
412
+ # Generate hash
413
+ from hashlib import sha256
414
+
415
+ favicon_hash = sha256(favicon_bytes).hexdigest()
416
+
417
+ return FaviconAnalysis(
418
+ favicon_hash=favicon_hash,
419
+ urls=urls,
420
+ company_name=company_name,
421
+ is_telecom=is_telecom,
422
+ is_hosting=is_hosting,
423
+ confidence=confidence,
424
+ llm_response=response.content,
425
+ )
426
+
427
+ except Exception as e:
428
+ # Return empty analysis on error
429
+ from hashlib import sha256
430
+
431
+ favicon_hash = sha256(favicon_bytes).hexdigest()
432
+
433
+ return FaviconAnalysis(
434
+ favicon_hash=favicon_hash,
435
+ urls=urls,
436
+ confidence=0.0,
437
+ llm_response=f"Error: {str(e)}",
438
+ )
439
+
440
+ def analyze_favicons(
441
+ self, favicon_data: Dict[str, bytes], url_groups: Dict[str, List[str]]
442
+ ) -> List[FaviconAnalysis]:
443
+ """Analyze multiple favicons.
444
+
445
+ Args:
446
+ favicon_data: Dictionary of favicon_hash -> favicon_bytes
447
+ url_groups: Dictionary of favicon_hash -> list of URLs
448
+
449
+ Returns:
450
+ List of favicon analyses
451
+ """
452
+ analyses = []
453
+
454
+ for favicon_hash, favicon_bytes in favicon_data.items():
455
+ urls = url_groups.get(favicon_hash, [])
456
+ if urls and len(urls) >= 2: # Data-driven: analyze if used by 2+ sites
457
+ analysis = self.analyze_favicon(favicon_bytes, urls)
458
+ analyses.append(analysis)
459
+
460
+ return analyses
461
+
462
+ def create_favicon_network_groups(
463
+ self,
464
+ favicon_hash_to_asns: Dict[str, List[int]],
465
+ favicon_hash_to_urls: Dict[str, List[str]],
466
+ min_asns: int = 2,
467
+ ) -> List[NetworkGroup]:
468
+ """Create NetworkGroups from favicon matches.
469
+
470
+ Args:
471
+ favicon_hash_to_asns: Mapping of favicon hash to ASNs using it
472
+ favicon_hash_to_urls: Mapping of favicon hash to URLs using it
473
+ min_asns: Minimum number of ASNs to form a group
474
+
475
+ Returns:
476
+ List of NetworkGroups based on shared favicons
477
+ """
478
+ groups = []
479
+
480
+ for favicon_hash, asns in favicon_hash_to_asns.items():
481
+ # Only create group if multiple ASNs share the favicon
482
+ if len(asns) >= min_asns:
483
+ # Get URLs for metadata
484
+ urls = favicon_hash_to_urls.get(favicon_hash, [])
485
+
486
+ group = NetworkGroup(
487
+ group_id=f"favicon_{favicon_hash[:16]}",
488
+ group_type="favicon_match",
489
+ asns=sorted(set(asns)),
490
+ common_attribute=favicon_hash,
491
+ metadata={
492
+ "favicon_hash": favicon_hash,
493
+ "url_count": len(urls),
494
+ "sample_urls": urls[:5], # Keep first 5 URLs as examples
495
+ "confidence": 0.6, # Medium confidence for favicon matches
496
+ },
497
+ )
498
+ groups.append(group)
499
+
500
+ return groups