borges 1.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- borges/__init__.py +20 -0
- borges/analyzers/__init__.py +14 -0
- borges/analyzers/llm_analyzer.py +500 -0
- borges/analyzers/network_consolidator.py +1989 -0
- borges/analyzers/number_validator.py +311 -0
- borges/analyzers/redirect_analyzer.py +203 -0
- borges/analyzers/whois_analyzer.py +184 -0
- borges/cli.py +634 -0
- borges/config.py +395 -0
- borges/data/__init__.py +46 -0
- borges/data/download.py +330 -0
- borges/data/exporters.py +319 -0
- borges/data/loaders.py +510 -0
- borges/data/processors.py +476 -0
- borges/defaults/config.yaml +252 -0
- borges/defaults/negative_samples/bootstrap_hexagon_b_variant.png +0 -0
- borges/defaults/negative_samples/bootstrap_purple_square_b.png +0 -0
- borges/defaults/negative_samples/chinese_hosting_text_logo.png +0 -0
- borges/defaults/negative_samples/facebook_blue_f_logo.png +0 -0
- borges/defaults/negative_samples/generic_corporate_logo.png +0 -0
- borges/defaults/negative_samples/generic_globe_web_icon.png +0 -0
- borges/defaults/negative_samples/generic_home_house_icon.png +0 -0
- borges/defaults/negative_samples/generic_hosting_server_icon.png +0 -0
- borges/defaults/negative_samples/generic_isp_branding.png +0 -0
- borges/defaults/negative_samples/generic_platform_icon.png +0 -0
- borges/defaults/negative_samples/generic_provider_logo.png +0 -0
- borges/defaults/negative_samples/generic_warning_triangle.png +0 -0
- borges/defaults/negative_samples/google_play_triangular_button.png +0 -0
- borges/defaults/negative_samples/mega_group_209asn_generic_e99155e1_urls.txt +28 -0
- borges/defaults/negative_samples/ngnix.png +0 -0
- borges/defaults/negative_samples/peeringdb_default_favicon.png +0 -0
- borges/defaults/negative_samples/unknown_generic_favicon.png +0 -0
- borges/defaults/negative_samples/wordpress.png +0 -0
- borges/defaults/negative_samples/wordpress2.png +0 -0
- borges/defaults/negative_samples/wordpress_default_w_logo.png +0 -0
- borges/models/__init__.py +27 -0
- borges/models/as_network.py +330 -0
- borges/models/schemas.py +212 -0
- borges/pipeline/__init__.py +27 -0
- borges/pipeline/runner.py +409 -0
- borges/pipeline/stages.py +1404 -0
- borges/resources.py +34 -0
- borges/scrapers/__init__.py +11 -0
- borges/scrapers/favicon_scraper.py +269 -0
- borges/scrapers/html_scraper.py +239 -0
- borges/scrapers/redirect_scraper.py +249 -0
- borges/utils/__init__.py +19 -0
- borges/utils/http_client.py +231 -0
- borges/utils/llm_client.py +231 -0
- borges/utils/logging.py +171 -0
- borges-1.1.0.dist-info/METADATA +321 -0
- borges-1.1.0.dist-info/RECORD +55 -0
- borges-1.1.0.dist-info/WHEEL +4 -0
- borges-1.1.0.dist-info/entry_points.txt +2 -0
- borges-1.1.0.dist-info/licenses/LICENSE +21 -0
borges/__init__.py
ADDED
|
@@ -0,0 +1,20 @@
|
|
|
1
|
+
"""Borges - AS Sibling Relationship Inference System.
|
|
2
|
+
|
|
3
|
+
A tool for inferring sibling relationships between Autonomous Systems using
|
|
4
|
+
data from PeeringDB, WHOIS, and web scraping with AI-powered analysis.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
__version__ = "1.1.0"
|
|
8
|
+
|
|
9
|
+
from .config import Config, get_config, load_config
|
|
10
|
+
from .models import ASNetwork, ASNetworkReport
|
|
11
|
+
from .pipeline import Pipeline
|
|
12
|
+
|
|
13
|
+
__all__ = [
|
|
14
|
+
"Config",
|
|
15
|
+
"get_config",
|
|
16
|
+
"load_config",
|
|
17
|
+
"ASNetwork",
|
|
18
|
+
"ASNetworkReport",
|
|
19
|
+
"Pipeline",
|
|
20
|
+
]
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
"""Borges analysis modules."""
|
|
2
|
+
|
|
3
|
+
from .llm_analyzer import ASRelationshipAnalyzer, FaviconAnalyzer
|
|
4
|
+
from .network_consolidator import NetworkGroupConsolidator
|
|
5
|
+
from .redirect_analyzer import RedirectAnalyzer
|
|
6
|
+
from .whois_analyzer import WHOISAnalyzer
|
|
7
|
+
|
|
8
|
+
__all__ = [
|
|
9
|
+
"ASRelationshipAnalyzer",
|
|
10
|
+
"FaviconAnalyzer",
|
|
11
|
+
"NetworkGroupConsolidator",
|
|
12
|
+
"RedirectAnalyzer",
|
|
13
|
+
"WHOISAnalyzer",
|
|
14
|
+
]
|
|
@@ -0,0 +1,500 @@
|
|
|
1
|
+
"""LLM-based analysis for sibling relationships and favicon identification."""
|
|
2
|
+
|
|
3
|
+
import base64
|
|
4
|
+
import time
|
|
5
|
+
from typing import Dict, List, Optional
|
|
6
|
+
|
|
7
|
+
import pandas as pd
|
|
8
|
+
from langchain_core.messages import HumanMessage
|
|
9
|
+
from langchain_core.output_parsers import JsonOutputParser
|
|
10
|
+
from langchain_core.prompts import PromptTemplate
|
|
11
|
+
from pydantic import BaseModel, Field
|
|
12
|
+
from langchain_openai import ChatOpenAI
|
|
13
|
+
from PIL import Image
|
|
14
|
+
from io import BytesIO
|
|
15
|
+
|
|
16
|
+
from ..config import get_config
|
|
17
|
+
from ..data.processors import ASNProcessor
|
|
18
|
+
from ..models import APIUsageStats, ASRelationship, FaviconAnalysis, NetworkGroup
|
|
19
|
+
from ..utils import get_logger
|
|
20
|
+
from .number_validator import validate_llm_output
|
|
21
|
+
|
|
22
|
+
logger = get_logger(__name__)
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
class ASList(BaseModel):
|
|
26
|
+
"""Model for LLM AS detection output."""
|
|
27
|
+
|
|
28
|
+
ASs: Optional[List[int]] = Field(None, description="List of AS numbers")
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
class ASRelationshipAnalyzer:
|
|
32
|
+
"""Analyze AS relationships using LLM."""
|
|
33
|
+
|
|
34
|
+
def __init__(self):
|
|
35
|
+
"""Initialize AS relationship analyzer."""
|
|
36
|
+
config = get_config()
|
|
37
|
+
self.config = config.api.openai
|
|
38
|
+
self.prompt_template = config.processing.prompts.get("as_detection")
|
|
39
|
+
self.asn_blocklist = set(config.processing.asn_blocklist)
|
|
40
|
+
|
|
41
|
+
# Initialize LLM with rate limiting
|
|
42
|
+
from ..utils.llm_client import create_llm_client
|
|
43
|
+
|
|
44
|
+
self.llm_client = create_llm_client()
|
|
45
|
+
self.llm = self.llm_client.llm
|
|
46
|
+
|
|
47
|
+
# Initialize parser
|
|
48
|
+
self.parser = JsonOutputParser(pydantic_object=ASList)
|
|
49
|
+
|
|
50
|
+
# Create prompt
|
|
51
|
+
self.prompt = PromptTemplate(
|
|
52
|
+
template=self.prompt_template,
|
|
53
|
+
input_variables=["aka", "notes", "asn"],
|
|
54
|
+
partial_variables={
|
|
55
|
+
"format_instructions": self.parser.get_format_instructions()
|
|
56
|
+
},
|
|
57
|
+
)
|
|
58
|
+
|
|
59
|
+
# Create chain
|
|
60
|
+
self.chain = self.prompt | self.llm | self.parser
|
|
61
|
+
|
|
62
|
+
# Initialize cost tracking
|
|
63
|
+
self.api_usage = APIUsageStats()
|
|
64
|
+
|
|
65
|
+
def analyze_as(self, asn: int, notes: str, aka: str) -> ASRelationship:
|
|
66
|
+
"""Analyze AS relationships from notes and AKA fields.
|
|
67
|
+
|
|
68
|
+
Args:
|
|
69
|
+
asn: Source AS number
|
|
70
|
+
notes: PeeringDB notes
|
|
71
|
+
aka: AKA field
|
|
72
|
+
|
|
73
|
+
Returns:
|
|
74
|
+
AS relationship
|
|
75
|
+
"""
|
|
76
|
+
# Skip analysis for blocklisted ASNs to prevent upstream pollution
|
|
77
|
+
if asn in self.asn_blocklist:
|
|
78
|
+
return ASRelationship(
|
|
79
|
+
source_asn=asn,
|
|
80
|
+
related_asns=[],
|
|
81
|
+
relationship_type="sibling",
|
|
82
|
+
confidence=0.0,
|
|
83
|
+
sources=[],
|
|
84
|
+
detected_by="llm_analysis",
|
|
85
|
+
)
|
|
86
|
+
|
|
87
|
+
try:
|
|
88
|
+
# Run LLM analysis with proper rate limiting
|
|
89
|
+
prompt_input = {"asn": asn, "notes": notes or "", "aka": aka or ""}
|
|
90
|
+
|
|
91
|
+
# Format the prompt
|
|
92
|
+
formatted_prompt = self.prompt.format(**prompt_input)
|
|
93
|
+
messages = [{"role": "user", "content": formatted_prompt}]
|
|
94
|
+
|
|
95
|
+
# Use LLM client with rate limiting
|
|
96
|
+
response = self.llm_client.invoke(messages)
|
|
97
|
+
|
|
98
|
+
# Parse the response
|
|
99
|
+
result = self.parser.parse(response.content)
|
|
100
|
+
|
|
101
|
+
# Track API usage (estimate tokens and cost)
|
|
102
|
+
self._track_api_usage(notes or "", aka or "")
|
|
103
|
+
|
|
104
|
+
# Extract ASNs from LLM
|
|
105
|
+
llm_asns = result.get("ASs", []) if result else []
|
|
106
|
+
|
|
107
|
+
# Validate LLM output - filter out hallucinated ASNs
|
|
108
|
+
input_text = f"{notes or ''} {aka or ''}".strip()
|
|
109
|
+
validated_llm_asns = validate_llm_output(
|
|
110
|
+
input_text, llm_asns, source_asn=asn, blocklist=self.asn_blocklist
|
|
111
|
+
)
|
|
112
|
+
|
|
113
|
+
# Also extract ASNs using regex (keep existing functionality)
|
|
114
|
+
text_asns = []
|
|
115
|
+
if notes:
|
|
116
|
+
text_asns.extend(ASNProcessor.detect_related_asns(notes, asn))
|
|
117
|
+
if aka:
|
|
118
|
+
text_asns.extend(ASNProcessor.detect_related_asns(aka, asn))
|
|
119
|
+
|
|
120
|
+
# Combine validated LLM results with regex results
|
|
121
|
+
all_asns = list(set(validated_llm_asns + text_asns))
|
|
122
|
+
all_asns = [asn_num for asn_num in all_asns if asn_num != asn]
|
|
123
|
+
|
|
124
|
+
# Filter out blocklisted ASNs from related_asns to prevent pollution
|
|
125
|
+
all_asns = [
|
|
126
|
+
asn_num for asn_num in all_asns if asn_num not in self.asn_blocklist
|
|
127
|
+
]
|
|
128
|
+
|
|
129
|
+
# Calculate dynamic confidence score
|
|
130
|
+
confidence = self._calculate_confidence(
|
|
131
|
+
llm_asns=validated_llm_asns, text_asns=text_asns, notes=notes, aka=aka
|
|
132
|
+
)
|
|
133
|
+
|
|
134
|
+
return ASRelationship(
|
|
135
|
+
source_asn=asn,
|
|
136
|
+
related_asns=all_asns,
|
|
137
|
+
relationship_type="organization_related",
|
|
138
|
+
confidence=confidence,
|
|
139
|
+
evidence=f"Notes: {notes}" if notes else f"AKA: {aka}" if aka else None,
|
|
140
|
+
detected_by="llm_analysis",
|
|
141
|
+
)
|
|
142
|
+
|
|
143
|
+
except Exception as e:
|
|
144
|
+
# Return empty relationship on error
|
|
145
|
+
return ASRelationship(
|
|
146
|
+
source_asn=asn,
|
|
147
|
+
related_asns=[],
|
|
148
|
+
relationship_type="organization_related",
|
|
149
|
+
confidence=0.0,
|
|
150
|
+
error=str(e),
|
|
151
|
+
detected_by="llm_analysis",
|
|
152
|
+
)
|
|
153
|
+
|
|
154
|
+
def analyze_dataframe(self, df: pd.DataFrame) -> List[ASRelationship]:
|
|
155
|
+
"""Analyze AS relationships from a DataFrame.
|
|
156
|
+
|
|
157
|
+
Args:
|
|
158
|
+
df: DataFrame with asn, notes, and aka columns
|
|
159
|
+
|
|
160
|
+
Returns:
|
|
161
|
+
List of AS relationships
|
|
162
|
+
"""
|
|
163
|
+
relationships = []
|
|
164
|
+
|
|
165
|
+
# Filter to rows that might have AS references
|
|
166
|
+
filtered_df = df[
|
|
167
|
+
(df["notes"].notna() & df["notes"].apply(ASNProcessor.has_asn_reference))
|
|
168
|
+
| (df["aka"].notna() & df["aka"].apply(ASNProcessor.has_asn_reference))
|
|
169
|
+
]
|
|
170
|
+
|
|
171
|
+
for _, row in filtered_df.iterrows():
|
|
172
|
+
relationship = self.analyze_as(
|
|
173
|
+
asn=row["asn"], notes=row.get("notes", ""), aka=row.get("aka", "")
|
|
174
|
+
)
|
|
175
|
+
if relationship.related_asns:
|
|
176
|
+
relationships.append(relationship)
|
|
177
|
+
|
|
178
|
+
return relationships
|
|
179
|
+
|
|
180
|
+
def _calculate_confidence(
|
|
181
|
+
self, llm_asns: List[int], text_asns: List[int], notes: str, aka: str
|
|
182
|
+
) -> float:
|
|
183
|
+
"""Calculate confidence score for AS relationship detection.
|
|
184
|
+
|
|
185
|
+
Args:
|
|
186
|
+
llm_asns: ASNs found by LLM analysis
|
|
187
|
+
text_asns: ASNs found by text processing
|
|
188
|
+
notes: Notes field content
|
|
189
|
+
aka: AKA field content
|
|
190
|
+
|
|
191
|
+
Returns:
|
|
192
|
+
Confidence score between 0.0 and 1.0
|
|
193
|
+
"""
|
|
194
|
+
if not llm_asns and not text_asns:
|
|
195
|
+
return 0.0
|
|
196
|
+
|
|
197
|
+
confidence = 0.0
|
|
198
|
+
|
|
199
|
+
# Base confidence for finding any ASNs
|
|
200
|
+
if llm_asns or text_asns:
|
|
201
|
+
confidence += 0.3
|
|
202
|
+
|
|
203
|
+
# Higher confidence if both LLM and text processing agree
|
|
204
|
+
if llm_asns and text_asns:
|
|
205
|
+
overlap = set(llm_asns) & set(text_asns)
|
|
206
|
+
if overlap:
|
|
207
|
+
confidence += 0.4 # Strong agreement
|
|
208
|
+
else:
|
|
209
|
+
confidence += 0.2 # Both found ASNs, but different ones
|
|
210
|
+
|
|
211
|
+
# Confidence boost based on evidence quality
|
|
212
|
+
evidence_text = (notes or "") + " " + (aka or "")
|
|
213
|
+
evidence_length = len(evidence_text.strip())
|
|
214
|
+
|
|
215
|
+
if evidence_length > 100:
|
|
216
|
+
confidence += 0.2 # Rich evidence
|
|
217
|
+
elif evidence_length > 20:
|
|
218
|
+
confidence += 0.1 # Some evidence
|
|
219
|
+
|
|
220
|
+
# Confidence based on number of ASNs found
|
|
221
|
+
total_asns = len(set(llm_asns + text_asns))
|
|
222
|
+
if total_asns == 1:
|
|
223
|
+
confidence += 0.1 # Single ASN is more reliable
|
|
224
|
+
elif total_asns <= 3:
|
|
225
|
+
confidence += 0.05 # Small group is reasonable
|
|
226
|
+
# No bonus for large groups (might be noisy)
|
|
227
|
+
|
|
228
|
+
# Evidence source preference (notes are more reliable than aka)
|
|
229
|
+
if notes and llm_asns:
|
|
230
|
+
confidence += 0.05
|
|
231
|
+
|
|
232
|
+
return min(confidence, 1.0) # Cap at 1.0
|
|
233
|
+
|
|
234
|
+
def _track_api_usage(self, notes: str, aka: str) -> None:
|
|
235
|
+
"""Track API usage for cost estimation.
|
|
236
|
+
|
|
237
|
+
Args:
|
|
238
|
+
notes: Notes text processed
|
|
239
|
+
aka: AKA text processed
|
|
240
|
+
"""
|
|
241
|
+
# Rough token estimation (1 token ≈ 4 characters for English text)
|
|
242
|
+
input_text = f"{self.prompt_template} {notes} {aka}"
|
|
243
|
+
estimated_input_tokens = len(input_text) // 4
|
|
244
|
+
estimated_output_tokens = 50 # Estimated JSON response size
|
|
245
|
+
|
|
246
|
+
# Update usage stats
|
|
247
|
+
self.api_usage.total_requests += 1
|
|
248
|
+
self.api_usage.total_input_tokens += estimated_input_tokens
|
|
249
|
+
self.api_usage.total_output_tokens += estimated_output_tokens
|
|
250
|
+
|
|
251
|
+
# Calculate cost based on gpt-4o-mini pricing (as of 2024)
|
|
252
|
+
# Input: $0.15 per 1M tokens, Output: $0.60 per 1M tokens
|
|
253
|
+
input_cost = (estimated_input_tokens / 1_000_000) * 0.15
|
|
254
|
+
output_cost = (estimated_output_tokens / 1_000_000) * 0.60
|
|
255
|
+
self.api_usage.estimated_cost_usd += input_cost + output_cost
|
|
256
|
+
|
|
257
|
+
def get_api_usage(self) -> APIUsageStats:
|
|
258
|
+
"""Get current API usage statistics.
|
|
259
|
+
|
|
260
|
+
Returns:
|
|
261
|
+
API usage statistics
|
|
262
|
+
"""
|
|
263
|
+
return self.api_usage
|
|
264
|
+
|
|
265
|
+
|
|
266
|
+
class FaviconAnalyzer:
|
|
267
|
+
"""Analyze favicons using LLM vision capabilities."""
|
|
268
|
+
|
|
269
|
+
def __init__(self):
|
|
270
|
+
"""Initialize favicon analyzer."""
|
|
271
|
+
config = get_config()
|
|
272
|
+
self.config = config.api.openai
|
|
273
|
+
self.prompt_template = config.processing.prompts.get("favicon_analysis")
|
|
274
|
+
|
|
275
|
+
# Initialize vision LLM with rate limiting
|
|
276
|
+
from ..utils.llm_client import create_llm_client
|
|
277
|
+
|
|
278
|
+
self.llm_client = create_llm_client(use_vision=True)
|
|
279
|
+
self.llm = self.llm_client.llm
|
|
280
|
+
|
|
281
|
+
# Load negative example images
|
|
282
|
+
self._load_negative_examples()
|
|
283
|
+
|
|
284
|
+
def _load_negative_examples(self):
|
|
285
|
+
"""Load negative example favicon images for comparison."""
|
|
286
|
+
from pathlib import Path
|
|
287
|
+
|
|
288
|
+
from ..resources import negative_samples_dir as default_samples_dir
|
|
289
|
+
|
|
290
|
+
self.negative_examples = []
|
|
291
|
+
# A project's own data/reference/ overrides the samples shipped with Borges
|
|
292
|
+
negative_samples_dir = Path("data/reference/negative_samples")
|
|
293
|
+
if not negative_samples_dir.exists():
|
|
294
|
+
try:
|
|
295
|
+
negative_samples_dir = default_samples_dir()
|
|
296
|
+
except FileNotFoundError as e:
|
|
297
|
+
logger.warning(str(e))
|
|
298
|
+
return
|
|
299
|
+
|
|
300
|
+
# Load all PNG images from negative samples directory
|
|
301
|
+
for image_file in negative_samples_dir.glob("*.png"):
|
|
302
|
+
try:
|
|
303
|
+
with open(image_file, "rb") as f:
|
|
304
|
+
image_bytes = f.read()
|
|
305
|
+
|
|
306
|
+
# Create descriptive name from filename
|
|
307
|
+
name = image_file.stem.replace("_", " ").title()
|
|
308
|
+
|
|
309
|
+
# Store raw bytes for now, will encode when needed
|
|
310
|
+
self.negative_examples.append(
|
|
311
|
+
{"name": name, "filename": image_file.name, "bytes": image_bytes}
|
|
312
|
+
)
|
|
313
|
+
|
|
314
|
+
except Exception as e:
|
|
315
|
+
logger.warning(f"Failed to load negative example {image_file}: {e}")
|
|
316
|
+
|
|
317
|
+
logger.info(f"Loaded {len(self.negative_examples)} negative favicon examples")
|
|
318
|
+
|
|
319
|
+
def _encode_image(self, image_bytes: bytes) -> str:
|
|
320
|
+
"""Encode image to base64.
|
|
321
|
+
|
|
322
|
+
Args:
|
|
323
|
+
image_bytes: Image binary data
|
|
324
|
+
|
|
325
|
+
Returns:
|
|
326
|
+
Base64 encoded string
|
|
327
|
+
"""
|
|
328
|
+
return base64.b64encode(image_bytes).decode("utf-8")
|
|
329
|
+
|
|
330
|
+
def analyze_favicon(self, favicon_bytes: bytes, urls: List[str]) -> FaviconAnalysis:
|
|
331
|
+
"""Analyze a favicon to identify company.
|
|
332
|
+
|
|
333
|
+
Args:
|
|
334
|
+
favicon_bytes: Favicon binary data
|
|
335
|
+
urls: URLs using this favicon
|
|
336
|
+
|
|
337
|
+
Returns:
|
|
338
|
+
Favicon analysis result
|
|
339
|
+
"""
|
|
340
|
+
try:
|
|
341
|
+
# Encode image
|
|
342
|
+
image_data = self._encode_image(favicon_bytes)
|
|
343
|
+
|
|
344
|
+
# Create message content starting with text and the favicon to analyze
|
|
345
|
+
content = [
|
|
346
|
+
{
|
|
347
|
+
"type": "text",
|
|
348
|
+
"text": self.prompt_template.format(urls=", ".join(urls[:5])),
|
|
349
|
+
},
|
|
350
|
+
{"type": "text", "text": "\n\n**FAVICON TO ANALYZE:**"},
|
|
351
|
+
{
|
|
352
|
+
"type": "image_url",
|
|
353
|
+
"image_url": {"url": f"data:image/jpeg;base64,{image_data}"},
|
|
354
|
+
},
|
|
355
|
+
]
|
|
356
|
+
|
|
357
|
+
# Add negative examples if available
|
|
358
|
+
# TEMPORARILY COMMENTED OUT: Visual negative examples consume too many tokens and cause rate limiting
|
|
359
|
+
# Keeping text-based negative examples in the prompt for protection
|
|
360
|
+
# if hasattr(self, 'negative_examples') and self.negative_examples:
|
|
361
|
+
# content.append({
|
|
362
|
+
# "type": "text",
|
|
363
|
+
# "text": "\n\n**NEGATIVE EXAMPLES - DO NOT GROUP if the favicon matches any of these common defaults:**"
|
|
364
|
+
# })
|
|
365
|
+
#
|
|
366
|
+
# # Add up to 8 negative examples to avoid token limits
|
|
367
|
+
# for i, example in enumerate(self.negative_examples[:8]):
|
|
368
|
+
# content.extend([
|
|
369
|
+
# {
|
|
370
|
+
# "type": "text",
|
|
371
|
+
# "text": f"\n{i+1}. {example['name']}:"
|
|
372
|
+
# },
|
|
373
|
+
# {
|
|
374
|
+
# "type": "image_url",
|
|
375
|
+
# "image_url": {"url": f"data:image/jpeg;base64,{self._encode_image(example['bytes'])}"},
|
|
376
|
+
# }
|
|
377
|
+
# ])
|
|
378
|
+
|
|
379
|
+
message = HumanMessage(content=content)
|
|
380
|
+
|
|
381
|
+
# Get LLM response with rate limiting
|
|
382
|
+
response = self.llm_client.invoke([message])
|
|
383
|
+
content = response.content.lower()
|
|
384
|
+
|
|
385
|
+
# Parse response
|
|
386
|
+
company_name = None
|
|
387
|
+
is_telecom = False
|
|
388
|
+
is_hosting = False
|
|
389
|
+
confidence = 0.0
|
|
390
|
+
|
|
391
|
+
if "no sé" not in content and "no se" not in content:
|
|
392
|
+
company_name = response.content.strip()
|
|
393
|
+
confidence = 0.8
|
|
394
|
+
|
|
395
|
+
# Check for telecom keywords
|
|
396
|
+
telecom_keywords = [
|
|
397
|
+
"telecom",
|
|
398
|
+
"telco",
|
|
399
|
+
"communications",
|
|
400
|
+
"móvil",
|
|
401
|
+
"mobile",
|
|
402
|
+
"fiber",
|
|
403
|
+
]
|
|
404
|
+
if any(keyword in content for keyword in telecom_keywords):
|
|
405
|
+
is_telecom = True
|
|
406
|
+
|
|
407
|
+
# Check for hosting keywords
|
|
408
|
+
hosting_keywords = ["hosting", "cloud", "server", "data center", "cdn"]
|
|
409
|
+
if any(keyword in content for keyword in hosting_keywords):
|
|
410
|
+
is_hosting = True
|
|
411
|
+
|
|
412
|
+
# Generate hash
|
|
413
|
+
from hashlib import sha256
|
|
414
|
+
|
|
415
|
+
favicon_hash = sha256(favicon_bytes).hexdigest()
|
|
416
|
+
|
|
417
|
+
return FaviconAnalysis(
|
|
418
|
+
favicon_hash=favicon_hash,
|
|
419
|
+
urls=urls,
|
|
420
|
+
company_name=company_name,
|
|
421
|
+
is_telecom=is_telecom,
|
|
422
|
+
is_hosting=is_hosting,
|
|
423
|
+
confidence=confidence,
|
|
424
|
+
llm_response=response.content,
|
|
425
|
+
)
|
|
426
|
+
|
|
427
|
+
except Exception as e:
|
|
428
|
+
# Return empty analysis on error
|
|
429
|
+
from hashlib import sha256
|
|
430
|
+
|
|
431
|
+
favicon_hash = sha256(favicon_bytes).hexdigest()
|
|
432
|
+
|
|
433
|
+
return FaviconAnalysis(
|
|
434
|
+
favicon_hash=favicon_hash,
|
|
435
|
+
urls=urls,
|
|
436
|
+
confidence=0.0,
|
|
437
|
+
llm_response=f"Error: {str(e)}",
|
|
438
|
+
)
|
|
439
|
+
|
|
440
|
+
def analyze_favicons(
|
|
441
|
+
self, favicon_data: Dict[str, bytes], url_groups: Dict[str, List[str]]
|
|
442
|
+
) -> List[FaviconAnalysis]:
|
|
443
|
+
"""Analyze multiple favicons.
|
|
444
|
+
|
|
445
|
+
Args:
|
|
446
|
+
favicon_data: Dictionary of favicon_hash -> favicon_bytes
|
|
447
|
+
url_groups: Dictionary of favicon_hash -> list of URLs
|
|
448
|
+
|
|
449
|
+
Returns:
|
|
450
|
+
List of favicon analyses
|
|
451
|
+
"""
|
|
452
|
+
analyses = []
|
|
453
|
+
|
|
454
|
+
for favicon_hash, favicon_bytes in favicon_data.items():
|
|
455
|
+
urls = url_groups.get(favicon_hash, [])
|
|
456
|
+
if urls and len(urls) >= 2: # Data-driven: analyze if used by 2+ sites
|
|
457
|
+
analysis = self.analyze_favicon(favicon_bytes, urls)
|
|
458
|
+
analyses.append(analysis)
|
|
459
|
+
|
|
460
|
+
return analyses
|
|
461
|
+
|
|
462
|
+
def create_favicon_network_groups(
|
|
463
|
+
self,
|
|
464
|
+
favicon_hash_to_asns: Dict[str, List[int]],
|
|
465
|
+
favicon_hash_to_urls: Dict[str, List[str]],
|
|
466
|
+
min_asns: int = 2,
|
|
467
|
+
) -> List[NetworkGroup]:
|
|
468
|
+
"""Create NetworkGroups from favicon matches.
|
|
469
|
+
|
|
470
|
+
Args:
|
|
471
|
+
favicon_hash_to_asns: Mapping of favicon hash to ASNs using it
|
|
472
|
+
favicon_hash_to_urls: Mapping of favicon hash to URLs using it
|
|
473
|
+
min_asns: Minimum number of ASNs to form a group
|
|
474
|
+
|
|
475
|
+
Returns:
|
|
476
|
+
List of NetworkGroups based on shared favicons
|
|
477
|
+
"""
|
|
478
|
+
groups = []
|
|
479
|
+
|
|
480
|
+
for favicon_hash, asns in favicon_hash_to_asns.items():
|
|
481
|
+
# Only create group if multiple ASNs share the favicon
|
|
482
|
+
if len(asns) >= min_asns:
|
|
483
|
+
# Get URLs for metadata
|
|
484
|
+
urls = favicon_hash_to_urls.get(favicon_hash, [])
|
|
485
|
+
|
|
486
|
+
group = NetworkGroup(
|
|
487
|
+
group_id=f"favicon_{favicon_hash[:16]}",
|
|
488
|
+
group_type="favicon_match",
|
|
489
|
+
asns=sorted(set(asns)),
|
|
490
|
+
common_attribute=favicon_hash,
|
|
491
|
+
metadata={
|
|
492
|
+
"favicon_hash": favicon_hash,
|
|
493
|
+
"url_count": len(urls),
|
|
494
|
+
"sample_urls": urls[:5], # Keep first 5 URLs as examples
|
|
495
|
+
"confidence": 0.6, # Medium confidence for favicon matches
|
|
496
|
+
},
|
|
497
|
+
)
|
|
498
|
+
groups.append(group)
|
|
499
|
+
|
|
500
|
+
return groups
|