wormhole-proxy 3.3.1__tar.gz → 3.3.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.3
2
2
  Name: wormhole-proxy
3
- Version: 3.3.1
3
+ Version: 3.3.2
4
4
  Summary: Asynchronous I/O HTTP and HTTPS Proxy on Python >= 3.11
5
5
  License: MIT
6
6
  Keywords: wormhole,asynchronous,web,proxy
@@ -1,7 +1,7 @@
1
1
  # Main project metadata (PEP 621 standard)
2
2
  [project]
3
3
  name = "wormhole-proxy"
4
- version = "3.3.1"
4
+ version = "3.3.2"
5
5
  description = "Asynchronous I/O HTTP and HTTPS Proxy on Python >= 3.11"
6
6
  readme = "README.md"
7
7
  authors = [
@@ -100,18 +100,27 @@ async def update_database(
100
100
  ) -> None:
101
101
  """
102
102
  Fetches all public blocklists, filters them against an allowlist,
103
- and compiles them into a SQLite database.
103
+ optimizes them, and compiles them into a SQLite database.
104
104
  """
105
105
  db_path = Path(db_path_str)
106
106
  db_path.parent.mkdir(parents=True, exist_ok=True)
107
107
 
108
- # Start with the hardcoded default allowlist
109
- allowlist_domains: set[str] = DEFAULT_ALLOWLIST.copy()
108
+ # Fetch and parse all blocklists
109
+ all_blocked_domains: set[str] = set()
110
+ async with aiohttp.ClientSession() as session:
111
+ tasks = [_fetch_list(session, url) for url in BLOCKLIST_URLS]
112
+ results = await asyncio.gather(*tasks)
113
+
114
+ logger.info("\nParsing all fetched lists...")
115
+ for content in results:
116
+ all_blocked_domains.update(_parse_domains_from_content(content))
117
+
110
118
  logger.info(
111
- f"Loaded {len(allowlist_domains)} domains from the default allowlist."
119
+ f"Found {len(all_blocked_domains)} unique domains before filtering."
112
120
  )
113
121
 
114
- # Add domains from the user-provided file, if any
122
+ # Load allowlist to filter domains before optimization and writing to the DB.
123
+ allowlist_domains: set[str] = DEFAULT_ALLOWLIST.copy()
115
124
  if allowlist_path_str:
116
125
  logger.info(f"Loading custom allowlist from: {allowlist_path_str}")
117
126
  try:
@@ -120,34 +129,22 @@ async def update_database(
120
129
  if line.strip() and not line.startswith("#"):
121
130
  allowlist_domains.add(line.strip().lower())
122
131
  logger.info(
123
- f"Total allowlist size is now {len(allowlist_domains)} domains."
132
+ f"Total allowlist size is {len(allowlist_domains)} domains."
124
133
  )
125
134
  except FileNotFoundError:
126
135
  logger.warning(
127
- "Custom allowlist file not found. Proceeding without it."
136
+ "Custom allowlist file not found. Proceeding with defaults."
128
137
  )
129
138
 
130
- # Fetch and parse all blocklists
131
- all_blocked_domains: set[str] = set()
132
- async with aiohttp.ClientSession() as session:
133
- tasks = [_fetch_list(session, url) for url in BLOCKLIST_URLS]
134
- results = await asyncio.gather(*tasks)
135
-
136
- logger.info("\nParsing all fetched lists...")
137
- for content in results:
138
- all_blocked_domains.update(_parse_domains_from_content(content))
139
-
140
- logger.info(
141
- f"Found {len(all_blocked_domains)} unique domains before filtering."
142
- )
143
-
144
- # Remove allowed domains from the blocklist
139
+ # Remove any domains from the blocklist that are exactly in the allowlist.
140
+ # This allows a user to allow a parent domain (e.g., 'x.com') while still
141
+ # letting specific ad-serving subdomains (e.g., 'ad-api.x.com') be blocked.
145
142
  if allowlist_domains:
146
143
  original_count = len(all_blocked_domains)
147
144
  all_blocked_domains -= allowlist_domains
148
145
  num_removed = original_count - len(all_blocked_domains)
149
146
  logger.info(
150
- f"Removed {num_removed} domains that were in the allowlist."
147
+ f"Removed {num_removed} domains that were present in the allowlist."
151
148
  )
152
149
 
153
150
  # Optimize the final blocklist
@@ -134,11 +134,11 @@ class Resolver:
134
134
  return [ip]
135
135
 
136
136
  # 2. Query DNS using aiodns for IPv4 and IPv6 addresses concurrently
137
- tasks = [
137
+ results = await asyncio.gather(
138
138
  self.resolver.query(hostname, "A"),
139
139
  self.resolver.query(hostname, "AAAA"),
140
- ]
141
- results = await asyncio.gather(*tasks, return_exceptions=True)
140
+ return_exceptions=True,
141
+ )
142
142
 
143
143
  resolved_ips: set[str] = set()
144
144
  for res in results:
@@ -136,33 +136,44 @@ def load_allowlist(path: str, host: str) -> int:
136
136
 
137
137
  def is_ad_domain(hostname: str) -> bool:
138
138
  """
139
- Checks if a hostname is blocked. It first checks against a runtime allowlist,
140
- then checks the main blocklist.
139
+ Checks if a hostname is blocked using a more specific block/allow logic.
140
+ The blocklist is checked before the allowlist to allow for more granular control.
141
+
142
+ The order of checks is:
143
+ 1. Exact match in blocklist -> Block
144
+ 2. Exact match in allowlist -> Allow
145
+ 3. Parent domain in blocklist -> Block
146
+ 4. Parent domain in allowlist -> Allow
141
147
  """
142
148
  hostname_lower = hostname.lower()
143
149
 
144
- # 1. Check the allowlist first.
145
- if ALLOW_LIST_SET:
146
- if hostname_lower in ALLOW_LIST_SET:
147
- return False # It's explicitly allowed
148
- # Check for parent domains in allowlist
149
- parts = hostname_lower.split(".")
150
- for i in range(1, len(parts)):
151
- parent_domain = ".".join(parts[i:])
152
- if parent_domain in ALLOW_LIST_SET:
153
- return False # A parent domain is allowed
154
-
155
- # 2. If not allowed, check the blocklist.
156
- if not AD_BLOCK_SET:
157
- return False
158
-
150
+ # --- Highest Priority: Check for an exact match in the blocklist ---
151
+ # This ensures that if 'ad-api.x.com' is specifically in the blocklist,
152
+ # it is blocked immediately, even if 'x.com' is on the allowlist.
159
153
  if hostname_lower in AD_BLOCK_SET:
160
154
  return True
161
155
 
156
+ # --- Second Priority: Check for an exact match in the allowlist ---
157
+ if hostname_lower in ALLOW_LIST_SET:
158
+ return False
159
+
160
+ # --- Third Priority: Check for parent domains in the blocklist ---
161
+ # This blocks subdomains of a blocked parent (e.g., if 'ad-server.com'
162
+ # is blocked, 'analytics.ad-server.com' will also be blocked).
162
163
  parts = hostname_lower.split(".")
163
164
  for i in range(1, len(parts)):
164
165
  parent_domain = ".".join(parts[i:])
165
166
  if parent_domain in AD_BLOCK_SET:
166
167
  return True
167
168
 
169
+ # --- Fourth Priority: Check for parent domains in the allowlist ---
170
+ # This allows subdomains of an allowed parent (e.g., if 'x.com' is
171
+ # allowed, 'www.x.com' will also be allowed), unless the subdomain
172
+ # itself was caught by the blocklist checks above.
173
+ for i in range(1, len(parts)):
174
+ parent_domain = ".".join(parts[i:])
175
+ if parent_domain in ALLOW_LIST_SET:
176
+ return False
177
+
178
+ # Default to not blocking if no specific rules match
168
179
  return False
File without changes
File without changes