PyPI - crawlo - Versions diffs - 1.2.2__py3-none-any.whl → 1.2.4__py3-none-any.whl - Mend

crawlo 1.2.2py3-none-any.whl → 1.2.4py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.

Potentially problematic release.

This version of crawlo might be problematic. Click here for more details.

Files changed (222) hide show

crawlo/__init__.py +61 -61
crawlo/__version__.py +1 -1
crawlo/cleaners/__init__.py +60 -60
crawlo/cleaners/data_formatter.py +225 -225
crawlo/cleaners/encoding_converter.py +125 -125
crawlo/cleaners/text_cleaner.py +232 -232
crawlo/cli.py +81 -81
crawlo/commands/__init__.py +14 -14
crawlo/commands/check.py +594 -594
crawlo/commands/genspider.py +151 -151
crawlo/commands/help.py +144 -142
crawlo/commands/list.py +155 -155
crawlo/commands/run.py +323 -292
crawlo/commands/startproject.py +420 -418
crawlo/commands/stats.py +188 -188
crawlo/commands/utils.py +186 -186
crawlo/config.py +312 -312
crawlo/config_validator.py +251 -252
crawlo/core/__init__.py +2 -2
crawlo/core/engine.py +354 -354
crawlo/core/processor.py +40 -40
crawlo/core/scheduler.py +143 -143
crawlo/crawler.py +1110 -1027
crawlo/data/__init__.py +6 -0
crawlo/data/user_agents.py +108 -0
crawlo/downloader/__init__.py +266 -266
crawlo/downloader/aiohttp_downloader.py +220 -220
crawlo/downloader/cffi_downloader.py +256 -256
crawlo/downloader/httpx_downloader.py +259 -259
crawlo/downloader/hybrid_downloader.py +212 -213
crawlo/downloader/playwright_downloader.py +402 -402
crawlo/downloader/selenium_downloader.py +472 -472
crawlo/event.py +11 -11
crawlo/exceptions.py +81 -81
crawlo/extension/__init__.py +37 -37
crawlo/extension/health_check.py +141 -141
crawlo/extension/log_interval.py +57 -57
crawlo/extension/log_stats.py +81 -81
crawlo/extension/logging_extension.py +43 -43
crawlo/extension/memory_monitor.py +104 -104
crawlo/extension/performance_profiler.py +133 -133
crawlo/extension/request_recorder.py +107 -107
crawlo/filters/__init__.py +154 -154
crawlo/filters/aioredis_filter.py +280 -280
crawlo/filters/memory_filter.py +269 -269
crawlo/items/__init__.py +23 -23
crawlo/items/base.py +21 -21
crawlo/items/fields.py +52 -53
crawlo/items/items.py +104 -104
crawlo/middleware/__init__.py +21 -21
crawlo/middleware/default_header.py +131 -131
crawlo/middleware/download_delay.py +104 -104
crawlo/middleware/middleware_manager.py +135 -135
crawlo/middleware/offsite.py +114 -115
crawlo/middleware/proxy.py +367 -366
crawlo/middleware/request_ignore.py +86 -87
crawlo/middleware/response_code.py +163 -164
crawlo/middleware/response_filter.py +136 -137
crawlo/middleware/retry.py +124 -124
crawlo/mode_manager.py +211 -211
crawlo/network/__init__.py +21 -21
crawlo/network/request.py +338 -338
crawlo/network/response.py +359 -359
crawlo/pipelines/__init__.py +21 -21
crawlo/pipelines/bloom_dedup_pipeline.py +156 -156
crawlo/pipelines/console_pipeline.py +39 -39
crawlo/pipelines/csv_pipeline.py +316 -316
crawlo/pipelines/database_dedup_pipeline.py +222 -224
crawlo/pipelines/json_pipeline.py +218 -218
crawlo/pipelines/memory_dedup_pipeline.py +115 -115
crawlo/pipelines/mongo_pipeline.py +131 -131
crawlo/pipelines/mysql_pipeline.py +317 -316
crawlo/pipelines/pipeline_manager.py +61 -61
crawlo/pipelines/redis_dedup_pipeline.py +165 -167
crawlo/project.py +279 -187
crawlo/queue/pqueue.py +37 -37
crawlo/queue/queue_manager.py +337 -337
crawlo/queue/redis_priority_queue.py +298 -298
crawlo/settings/__init__.py +7 -7
crawlo/settings/default_settings.py +217 -226
crawlo/settings/setting_manager.py +122 -122
crawlo/spider/__init__.py +639 -639
crawlo/stats_collector.py +59 -59
crawlo/subscriber.py +129 -130
crawlo/task_manager.py +30 -30
crawlo/templates/crawlo.cfg.tmpl +10 -10
crawlo/templates/project/__init__.py.tmpl +3 -3
crawlo/templates/project/items.py.tmpl +17 -17
crawlo/templates/project/middlewares.py.tmpl +118 -118
crawlo/templates/project/pipelines.py.tmpl +96 -96
crawlo/templates/project/run.py.tmpl +47 -45
crawlo/templates/project/settings.py.tmpl +350 -327
crawlo/templates/project/settings_distributed.py.tmpl +160 -119
crawlo/templates/project/settings_gentle.py.tmpl +133 -94
crawlo/templates/project/settings_high_performance.py.tmpl +155 -151
crawlo/templates/project/settings_simple.py.tmpl +108 -68
crawlo/templates/project/spiders/__init__.py.tmpl +5 -5
crawlo/templates/spider/spider.py.tmpl +143 -143
crawlo/tools/__init__.py +182 -182
crawlo/tools/anti_crawler.py +268 -268
crawlo/tools/authenticated_proxy.py +240 -240
crawlo/tools/data_validator.py +180 -180
crawlo/tools/date_tools.py +35 -35
crawlo/tools/distributed_coordinator.py +386 -386
crawlo/tools/retry_mechanism.py +220 -220
crawlo/tools/scenario_adapter.py +262 -262
crawlo/utils/__init__.py +35 -35
crawlo/utils/batch_processor.py +259 -260
crawlo/utils/controlled_spider_mixin.py +439 -439
crawlo/utils/date_tools.py +290 -290
crawlo/utils/db_helper.py +343 -343
crawlo/utils/enhanced_error_handler.py +356 -359
crawlo/utils/env_config.py +105 -105
crawlo/utils/error_handler.py +123 -125
crawlo/utils/func_tools.py +82 -82
crawlo/utils/large_scale_config.py +286 -286
crawlo/utils/large_scale_helper.py +344 -343
crawlo/utils/log.py +128 -128
crawlo/utils/performance_monitor.py +285 -284
crawlo/utils/queue_helper.py +175 -175
crawlo/utils/redis_connection_pool.py +334 -334
crawlo/utils/redis_key_validator.py +198 -199
crawlo/utils/request.py +267 -267
crawlo/utils/request_serializer.py +218 -219
crawlo/utils/spider_loader.py +61 -62
crawlo/utils/system.py +11 -11
crawlo/utils/tools.py +4 -4
crawlo/utils/url.py +39 -39
{crawlo-1.2.2.dist-info → crawlo-1.2.4.dist-info}/METADATA +764 -692
crawlo-1.2.4.dist-info/RECORD +206 -0
examples/__init__.py +7 -7
tests/DOUBLE_CRAWLO_PREFIX_FIX_REPORT.md +81 -81
tests/__init__.py +7 -7
tests/advanced_tools_example.py +275 -275
tests/authenticated_proxy_example.py +236 -236
tests/cleaners_example.py +160 -160
tests/config_validation_demo.py +102 -102
tests/controlled_spider_example.py +205 -205
tests/date_tools_example.py +180 -180
tests/dynamic_loading_example.py +523 -523
tests/dynamic_loading_test.py +104 -104
tests/env_config_example.py +133 -133
tests/error_handling_example.py +171 -171
tests/redis_key_validation_demo.py +130 -130
tests/response_improvements_example.py +144 -144
tests/test_advanced_tools.py +148 -148
tests/test_all_redis_key_configs.py +145 -145
tests/test_authenticated_proxy.py +141 -141
tests/test_cleaners.py +54 -54
tests/test_comprehensive.py +146 -146
tests/test_config_validator.py +193 -193
tests/test_crawlo_proxy_integration.py +172 -172
tests/test_date_tools.py +123 -123
tests/test_default_header_middleware.py +158 -158
tests/test_double_crawlo_fix.py +207 -207
tests/test_double_crawlo_fix_simple.py +124 -124
tests/test_download_delay_middleware.py +221 -221
tests/test_downloader_proxy_compatibility.py +268 -268
tests/test_dynamic_downloaders_proxy.py +124 -124
tests/test_dynamic_proxy.py +92 -92
tests/test_dynamic_proxy_config.py +146 -146
tests/test_dynamic_proxy_real.py +109 -109
tests/test_edge_cases.py +303 -303
tests/test_enhanced_error_handler.py +270 -270
tests/test_env_config.py +121 -121
tests/test_error_handler_compatibility.py +112 -112
tests/test_final_validation.py +153 -153
tests/test_framework_env_usage.py +103 -103
tests/test_integration.py +356 -356
tests/test_item_dedup_redis_key.py +122 -122
tests/test_offsite_middleware.py +221 -221
tests/test_parsel.py +29 -29
tests/test_performance.py +327 -327
tests/test_proxy_api.py +264 -264
tests/test_proxy_health_check.py +32 -32
tests/test_proxy_middleware.py +121 -121
tests/test_proxy_middleware_enhanced.py +216 -216
tests/test_proxy_middleware_integration.py +136 -136
tests/test_proxy_providers.py +56 -56
tests/test_proxy_stats.py +19 -19
tests/test_proxy_strategies.py +59 -59
tests/test_queue_manager_double_crawlo.py +173 -173
tests/test_queue_manager_redis_key.py +176 -176
tests/test_real_scenario_proxy.py +195 -195
tests/test_redis_config.py +28 -28
tests/test_redis_connection_pool.py +294 -294
tests/test_redis_key_naming.py +181 -181
tests/test_redis_key_validator.py +123 -123
tests/test_redis_queue.py +224 -224
tests/test_request_ignore_middleware.py +182 -182
tests/test_request_serialization.py +70 -70
tests/test_response_code_middleware.py +349 -349
tests/test_response_filter_middleware.py +427 -427
tests/test_response_improvements.py +152 -152
tests/test_retry_middleware.py +241 -241
tests/test_scheduler.py +241 -241
tests/test_simple_response.py +61 -61
tests/test_telecom_spider_redis_key.py +205 -205
tests/test_template_content.py +87 -87
tests/test_template_redis_key.py +134 -134
tests/test_tools.py +153 -153
tests/tools_example.py +257 -257
crawlo-1.2.2.dist-info/RECORD +0 -220
examples/aiohttp_settings.py +0 -42
examples/curl_cffi_settings.py +0 -41
examples/default_header_middleware_example.py +0 -107
examples/default_header_spider_example.py +0 -129
examples/download_delay_middleware_example.py +0 -160
examples/httpx_settings.py +0 -42
examples/multi_downloader_proxy_example.py +0 -81
examples/offsite_middleware_example.py +0 -55
examples/offsite_spider_example.py +0 -107
examples/proxy_spider_example.py +0 -166
examples/request_ignore_middleware_example.py +0 -51
examples/request_ignore_spider_example.py +0 -99
examples/response_code_middleware_example.py +0 -52
examples/response_filter_middleware_example.py +0 -67
examples/tong_hua_shun_settings.py +0 -62
examples/tong_hua_shun_spider.py +0 -170
{crawlo-1.2.2.dist-info → crawlo-1.2.4.dist-info}/WHEEL +0 -0
{crawlo-1.2.2.dist-info → crawlo-1.2.4.dist-info}/entry_points.txt +0 -0
{crawlo-1.2.2.dist-info → crawlo-1.2.4.dist-info}/top_level.txt +0 -0

tests/test_proxy_api.py CHANGED Viewed

@@ -1,265 +1,265 @@
-#!/usr/bin/python
-# -*- coding: UTF-8 -*-
-"""
-代理API测试脚本
-================
-测试指定的代理API接口是否能正常工作
-"""
-import asyncio
-import aiohttp
-import sys
-import os
-from urllib.parse import urlparse
-# 添加项目根目录到Python路径
-sys.path.insert(0, os.path.join(os.path.dirname(__file__), '..'))
-from crawlo.middleware.proxy import ProxyMiddleware
-from crawlo.network.request import Request
-from crawlo.settings.setting_manager import SettingManager
-async def test_proxy_api(proxy_api_url):
-    """测试代理API接口"""
-    print(f"=== 测试代理API接口 ===")
-    print(f"API地址: {proxy_api_url}")
-    try:
-        timeout = aiohttp.ClientTimeout(total=10)
-        async with aiohttp.ClientSession(timeout=timeout) as session:
-            async with session.get(proxy_api_url) as response:
-                print(f"状态码: {response.status}")
-                print(f"响应头: {response.headers.get('content-type', 'Unknown')}")
-                # 尝试解析JSON响应
-                try:
-                    data = await response.json()
-                    print(f"响应数据: {data}")
-                    return data
-                except Exception as e:
-                    # 如果不是JSON，尝试获取文本
-                    try:
-                        text = await response.text()
-                        print(f"响应文本: {text[:200]}{'...' if len(text) > 200 else ''}")
-                        return text
-                    except Exception as e2:
-                        print(f"无法解析响应内容: {e2}")
-                        return None
-    except asyncio.TimeoutError:
-        print("请求超时")
-        return None
-    except Exception as e:
-        print(f"请求失败: {e}")
-        return None
-def extract_proxy_url(proxy_data):
-    """从API响应中提取代理URL"""
-    proxy_url = None
-    if isinstance(proxy_data, dict):
-        # 检查是否有status字段且为成功状态
-        if proxy_data.get('status') == 0:
-            # 获取proxy字段
-            proxy_info = proxy_data.get('proxy', {})
-            if isinstance(proxy_info, dict):
-                # 优先使用https代理，否则使用http代理
-                proxy_url = proxy_info.get('https') or proxy_info.get('http')
-            elif isinstance(proxy_info, str):
-                proxy_url = proxy_info
-        else:
-            # 直接尝试常见的字段名
-            for key in ['proxy', 'data', 'url', 'http', 'https']:
-                if key in proxy_data:
-                    value = proxy_data[key]
-                    if isinstance(value, str):
-                        proxy_url = value
-                        break
-                    elif isinstance(value, dict):
-                        proxy_url = value.get('https') or value.get('http')
-                        break
-        # 如果还是没有找到，尝试更深层的嵌套
-        if not proxy_url:
-            for key, value in proxy_data.items():
-                if isinstance(value, str) and (value.startswith('http://') or value.startswith('https://')):
-                    proxy_url = value
-                    break
-                elif isinstance(value, dict):
-                    # 递归查找
-                    for sub_key, sub_value in value.items():
-                        if isinstance(sub_value, str) and (sub_value.startswith('http://') or sub_value.startswith('https://')):
-                            proxy_url = sub_value
-                            break
-                    if proxy_url:
-                        break
-    elif isinstance(proxy_data, str):
-        # 如果响应是字符串，直接使用
-        if proxy_data.startswith('http://') or proxy_data.startswith('https://'):
-            proxy_url = proxy_data
-    return proxy_url
-async def test_target_url_without_proxy(target_url):
-    """不使用代理直接测试访问目标URL"""
-    print(f"\n=== 直接访问目标URL（不使用代理） ===")
-    print(f"目标URL: {target_url}")
-    try:
-        timeout = aiohttp.ClientTimeout(total=15)
-        async with aiohttp.ClientSession(timeout=timeout) as session:
-            # 添加用户代理头，避免被反爬虫机制拦截
-            headers = {
-                'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/124.0.0.0 Safari/537.36'
-            }
-            async with session.get(target_url, headers=headers) as response:
-                print(f"状态码: {response.status}")
-                print(f"响应头: {response.headers.get('content-type', 'Unknown')}")
-                # 只读取响应状态，不尝试解码内容
-                return response.status == 200
-    except asyncio.TimeoutError:
-        print("请求超时")
-        return False
-    except Exception as e:
-        print(f"请求失败: {e}")
-        return False
-async def test_target_url_with_proxy(proxy_url, target_url, max_retries=3):
-    """使用代理测试访问目标URL"""
-    print(f"\n=== 使用代理测试访问目标URL ===")
-    print(f"代理地址: {proxy_url}")
-    print(f"目标URL: {target_url}")
-    # 添加用户代理头，避免被反爬虫机制拦截
-    headers = {
-        'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/124.0.0.0 Safari/537.36'
-    }
-    for attempt in range(max_retries):
-        if attempt > 0:
-            print(f"\n第 {attempt + 1} 次重试...")
-        try:
-            # 创建aiohttp客户端会话
-            timeout = aiohttp.ClientTimeout(total=15)
-            async with aiohttp.ClientSession(timeout=timeout, headers=headers) as session:
-                # 处理代理URL，支持带认证的代理
-                if isinstance(proxy_url, str) and "@" in proxy_url and "://" in proxy_url:
-                    parsed = urlparse(proxy_url)
-                    if parsed.username and parsed.password:
-                        # 提取认证信息
-                        auth = aiohttp.BasicAuth(parsed.username, parsed.password)
-                        # 清理代理URL，移除认证信息
-                        clean_proxy = f"{parsed.scheme}://{parsed.hostname}"
-                        if parsed.port:
-                            clean_proxy += f":{parsed.port}"
-                        print(f"使用带认证的代理: {clean_proxy}")
-                        async with session.get(target_url, proxy=clean_proxy, proxy_auth=auth) as response:
-                            print(f"状态码: {response.status}")
-                            print(f"响应头: {response.headers.get('content-type', 'Unknown')}")
-                            return response.status == 200
-                    else:
-                        # 没有认证信息的代理
-                        print(f"使用普通代理: {proxy_url}")
-                        async with session.get(target_url, proxy=proxy_url) as response:
-                            print(f"状态码: {response.status}")
-                            print(f"响应头: {response.headers.get('content-type', 'Unknown')}")
-                            return response.status == 200
-                else:
-                    # 直接使用代理URL
-                    print(f"使用代理: {proxy_url}")
-                    async with session.get(target_url, proxy=proxy_url) as response:
-                        print(f"状态码: {response.status}")
-                        print(f"响应头: {response.headers.get('content-type', 'Unknown')}")
-                        return response.status == 200
-        except asyncio.TimeoutError:
-            print("请求超时")
-            if attempt < max_retries - 1:
-                await asyncio.sleep(2)  # 等待2秒后重试
-            continue
-        except aiohttp.ClientConnectorError as e:
-            print(f"连接错误: {e}")
-            if attempt < max_retries - 1:
-                await asyncio.sleep(2)  # 等待2秒后重试
-            continue
-        except aiohttp.ClientHttpProxyError as e:
-            print(f"代理HTTP错误: {e}")
-            if attempt < max_retries - 1:
-                await asyncio.sleep(2)  # 等待2秒后重试
-            continue
-        except aiohttp.ServerDisconnectedError as e:
-            print(f"服务器断开连接: {e}")
-            if attempt < max_retries - 1:
-                await asyncio.sleep(2)  # 等待2秒后重试
-            continue
-        except Exception as e:
-            print(f"请求失败: {e}")
-            if attempt < max_retries - 1:
-                await asyncio.sleep(2)  # 等待2秒后重试
-            continue
-    return False
-async def main():
-    """主测试函数"""
-    # 指定的代理API和测试链接
-    proxy_api = 'http://test.proxy.api:8080/proxy/getitem/'
-    target_url = 'https://stock.10jqka.com.cn/20240315/c655957791.shtml'
-    print("开始测试代理接口和目标链接访问...\n")
-    # 1. 测试代理API接口
-    proxy_data = await test_proxy_api(proxy_api)
-    if not proxy_data:
-        print("代理API测试失败，无法获取代理信息")
-        return
-    # 2. 从API响应中提取代理URL
-    proxy_url = extract_proxy_url(proxy_data)
-    if not proxy_url:
-        print("无法从API响应中提取代理URL")
-        print(f"API响应内容: {proxy_data}")
-        return
-    print(f"\n提取到的代理URL: {proxy_url}")
-    # 3. 首先尝试直接访问，确认目标URL是否可访问
-    print("\n=== 测试直接访问目标URL ===")
-    direct_success = await test_target_url_without_proxy(target_url)
-    if direct_success:
-        print("✅ 直接访问目标URL成功")
-    else:
-        print("❌ 直接访问目标URL失败")
-    # 4. 使用代理访问目标URL
-    print("\n=== 测试使用代理访问目标URL ===")
-    proxy_success = await test_target_url_with_proxy(proxy_url, target_url)
-    if proxy_success:
-        print(f"✅ 代理测试成功！代理 {proxy_url} 可以正常访问目标链接")
-    else:
-        print(f"❌ 代理测试失败！代理 {proxy_url} 无法访问目标链接")
-    # 5. 总结
-    print(f"\n=== 测试总结 ===")
-    print(f"代理API访问: {'成功' if proxy_data else '失败'}")
-    print(f"代理提取: {'成功' if proxy_url else '失败'}")
-    print(f"直接访问: {'成功' if direct_success else '失败'}")
-    print(f"代理访问: {'成功' if proxy_success else '失败'}")
-if __name__ == "__main__":
+#!/usr/bin/python
+# -*- coding: UTF-8 -*-
+"""
+代理API测试脚本
+================
+测试指定的代理API接口是否能正常工作
+"""
+import asyncio
+import aiohttp
+import sys
+import os
+from urllib.parse import urlparse
+# 添加项目根目录到Python路径
+sys.path.insert(0, os.path.join(os.path.dirname(__file__), '..'))
+from crawlo.middleware.proxy import ProxyMiddleware
+from crawlo.network.request import Request
+from crawlo.settings.setting_manager import SettingManager
+async def test_proxy_api(proxy_api_url):
+    """测试代理API接口"""
+    print(f"=== 测试代理API接口 ===")
+    print(f"API地址: {proxy_api_url}")
+    try:
+        timeout = aiohttp.ClientTimeout(total=10)
+        async with aiohttp.ClientSession(timeout=timeout) as session:
+            async with session.get(proxy_api_url) as response:
+                print(f"状态码: {response.status}")
+                print(f"响应头: {response.headers.get('content-type', 'Unknown')}")
+                # 尝试解析JSON响应
+                try:
+                    data = await response.json()
+                    print(f"响应数据: {data}")
+                    return data
+                except Exception as e:
+                    # 如果不是JSON，尝试获取文本
+                    try:
+                        text = await response.text()
+                        print(f"响应文本: {text[:200]}{'...' if len(text) > 200 else ''}")
+                        return text
+                    except Exception as e2:
+                        print(f"无法解析响应内容: {e2}")
+                        return None
+    except asyncio.TimeoutError:
+        print("请求超时")
+        return None
+    except Exception as e:
+        print(f"请求失败: {e}")
+        return None
+def extract_proxy_url(proxy_data):
+    """从API响应中提取代理URL"""
+    proxy_url = None
+    if isinstance(proxy_data, dict):
+        # 检查是否有status字段且为成功状态
+        if proxy_data.get('status') == 0:
+            # 获取proxy字段
+            proxy_info = proxy_data.get('proxy', {})
+            if isinstance(proxy_info, dict):
+                # 优先使用https代理，否则使用http代理
+                proxy_url = proxy_info.get('https') or proxy_info.get('http')
+            elif isinstance(proxy_info, str):
+                proxy_url = proxy_info
+        else:
+            # 直接尝试常见的字段名
+            for key in ['proxy', 'data', 'url', 'http', 'https']:
+                if key in proxy_data:
+                    value = proxy_data[key]
+                    if isinstance(value, str):
+                        proxy_url = value
+                        break
+                    elif isinstance(value, dict):
+                        proxy_url = value.get('https') or value.get('http')
+                        break
+        # 如果还是没有找到，尝试更深层的嵌套
+        if not proxy_url:
+            for key, value in proxy_data.items():
+                if isinstance(value, str) and (value.startswith('http://') or value.startswith('https://')):
+                    proxy_url = value
+                    break
+                elif isinstance(value, dict):
+                    # 递归查找
+                    for sub_key, sub_value in value.items():
+                        if isinstance(sub_value, str) and (sub_value.startswith('http://') or sub_value.startswith('https://')):
+                            proxy_url = sub_value
+                            break
+                    if proxy_url:
+                        break
+    elif isinstance(proxy_data, str):
+        # 如果响应是字符串，直接使用
+        if proxy_data.startswith('http://') or proxy_data.startswith('https://'):
+            proxy_url = proxy_data
+    return proxy_url
+async def test_target_url_without_proxy(target_url):
+    """不使用代理直接测试访问目标URL"""
+    print(f"\n=== 直接访问目标URL（不使用代理） ===")
+    print(f"目标URL: {target_url}")
+    try:
+        timeout = aiohttp.ClientTimeout(total=15)
+        async with aiohttp.ClientSession(timeout=timeout) as session:
+            # 添加用户代理头，避免被反爬虫机制拦截
+            headers = {
+                'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/124.0.0.0 Safari/537.36'
+            }
+            async with session.get(target_url, headers=headers) as response:
+                print(f"状态码: {response.status}")
+                print(f"响应头: {response.headers.get('content-type', 'Unknown')}")
+                # 只读取响应状态，不尝试解码内容
+                return response.status == 200
+    except asyncio.TimeoutError:
+        print("请求超时")
+        return False
+    except Exception as e:
+        print(f"请求失败: {e}")
+        return False
+async def test_target_url_with_proxy(proxy_url, target_url, max_retries=3):
+    """使用代理测试访问目标URL"""
+    print(f"\n=== 使用代理测试访问目标URL ===")
+    print(f"代理地址: {proxy_url}")
+    print(f"目标URL: {target_url}")
+    # 添加用户代理头，避免被反爬虫机制拦截
+    headers = {
+        'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/124.0.0.0 Safari/537.36'
+    }
+    for attempt in range(max_retries):
+        if attempt > 0:
+            print(f"\n第 {attempt + 1} 次重试...")
+        try:
+            # 创建aiohttp客户端会话
+            timeout = aiohttp.ClientTimeout(total=15)
+            async with aiohttp.ClientSession(timeout=timeout, headers=headers) as session:
+                # 处理代理URL，支持带认证的代理
+                if isinstance(proxy_url, str) and "@" in proxy_url and "://" in proxy_url:
+                    parsed = urlparse(proxy_url)
+                    if parsed.username and parsed.password:
+                        # 提取认证信息
+                        auth = aiohttp.BasicAuth(parsed.username, parsed.password)
+                        # 清理代理URL，移除认证信息
+                        clean_proxy = f"{parsed.scheme}://{parsed.hostname}"
+                        if parsed.port:
+                            clean_proxy += f":{parsed.port}"
+                        print(f"使用带认证的代理: {clean_proxy}")
+                        async with session.get(target_url, proxy=clean_proxy, proxy_auth=auth) as response:
+                            print(f"状态码: {response.status}")
+                            print(f"响应头: {response.headers.get('content-type', 'Unknown')}")
+                            return response.status == 200
+                    else:
+                        # 没有认证信息的代理
+                        print(f"使用普通代理: {proxy_url}")
+                        async with session.get(target_url, proxy=proxy_url) as response:
+                            print(f"状态码: {response.status}")
+                            print(f"响应头: {response.headers.get('content-type', 'Unknown')}")
+                            return response.status == 200
+                else:
+                    # 直接使用代理URL
+                    print(f"使用代理: {proxy_url}")
+                    async with session.get(target_url, proxy=proxy_url) as response:
+                        print(f"状态码: {response.status}")
+                        print(f"响应头: {response.headers.get('content-type', 'Unknown')}")
+                        return response.status == 200
+        except asyncio.TimeoutError:
+            print("请求超时")
+            if attempt < max_retries - 1:
+                await asyncio.sleep(2)  # 等待2秒后重试
+            continue
+        except aiohttp.ClientConnectorError as e:
+            print(f"连接错误: {e}")
+            if attempt < max_retries - 1:
+                await asyncio.sleep(2)  # 等待2秒后重试
+            continue
+        except aiohttp.ClientHttpProxyError as e:
+            print(f"代理HTTP错误: {e}")
+            if attempt < max_retries - 1:
+                await asyncio.sleep(2)  # 等待2秒后重试
+            continue
+        except aiohttp.ServerDisconnectedError as e:
+            print(f"服务器断开连接: {e}")
+            if attempt < max_retries - 1:
+                await asyncio.sleep(2)  # 等待2秒后重试
+            continue
+        except Exception as e:
+            print(f"请求失败: {e}")
+            if attempt < max_retries - 1:
+                await asyncio.sleep(2)  # 等待2秒后重试
+            continue
+    return False
+async def main():
+    """主测试函数"""
+    # 指定的代理API和测试链接
+    proxy_api = 'http://test.proxy.api:8080/proxy/getitem/'
+    target_url = 'https://stock.10jqka.com.cn/20240315/c655957791.shtml'
+    print("开始测试代理接口和目标链接访问...\n")
+    # 1. 测试代理API接口
+    proxy_data = await test_proxy_api(proxy_api)
+    if not proxy_data:
+        print("代理API测试失败，无法获取代理信息")
+        return
+    # 2. 从API响应中提取代理URL
+    proxy_url = extract_proxy_url(proxy_data)
+    if not proxy_url:
+        print("无法从API响应中提取代理URL")
+        print(f"API响应内容: {proxy_data}")
+        return
+    print(f"\n提取到的代理URL: {proxy_url}")
+    # 3. 首先尝试直接访问，确认目标URL是否可访问
+    print("\n=== 测试直接访问目标URL ===")
+    direct_success = await test_target_url_without_proxy(target_url)
+    if direct_success:
+        print("✅ 直接访问目标URL成功")
+    else:
+        print("❌ 直接访问目标URL失败")
+    # 4. 使用代理访问目标URL
+    print("\n=== 测试使用代理访问目标URL ===")
+    proxy_success = await test_target_url_with_proxy(proxy_url, target_url)
+    if proxy_success:
+        print(f"✅ 代理测试成功！代理 {proxy_url} 可以正常访问目标链接")
+    else:
+        print(f"❌ 代理测试失败！代理 {proxy_url} 无法访问目标链接")
+    # 5. 总结
+    print(f"\n=== 测试总结 ===")
+    print(f"代理API访问: {'成功' if proxy_data else '失败'}")
+    print(f"代理提取: {'成功' if proxy_url else '失败'}")
+    print(f"直接访问: {'成功' if direct_success else '失败'}")
+    print(f"代理访问: {'成功' if proxy_success else '失败'}")
+if __name__ == "__main__":
     asyncio.run(main())

tests/test_proxy_health_check.py CHANGED Viewed

@@ -1,33 +1,33 @@
-# tests/test_proxy_health_check.py
-import pytest
-from unittest.mock import AsyncMock, patch
-from crawlo.proxy.health_check import check_single_proxy
-import httpx
-@pytest.mark.asyncio
-@patch('httpx.AsyncClient')
-async def test_health_check_success(mock_client_class):
-    """测试健康检查：成功"""
-    mock_resp = AsyncMock()
-    mock_resp.status_code = 200
-    mock_client_class.return_value.__aenter__.return_value.get.return_value = mock_resp
-    proxy_info = {'url': 'http://good:8080', 'healthy': False}
-    await check_single_proxy(proxy_info)
-    assert proxy_info['healthy'] is True
-    assert proxy_info['failures'] == 0
-@pytest.mark.asyncio
-@patch('httpx.AsyncClient')
-async def test_health_check_failure(mock_client_class):
-    """测试健康检查：失败"""
-    mock_client_class.return_value.__aenter__.return_value.get.side_effect = httpx.ConnectError("Failed")
-    proxy_info = {'url': 'http://bad:8080', 'healthy': True, 'failures': 0}
-    await check_single_proxy(proxy_info)
-    assert proxy_info['healthy'] is False
+# tests/test_proxy_health_check.py
+import pytest
+from unittest.mock import AsyncMock, patch
+from crawlo.proxy.health_check import check_single_proxy
+import httpx
+@pytest.mark.asyncio
+@patch('httpx.AsyncClient')
+async def test_health_check_success(mock_client_class):
+    """测试健康检查：成功"""
+    mock_resp = AsyncMock()
+    mock_resp.status_code = 200
+    mock_client_class.return_value.__aenter__.return_value.get.return_value = mock_resp
+    proxy_info = {'url': 'http://good:8080', 'healthy': False}
+    await check_single_proxy(proxy_info)
+    assert proxy_info['healthy'] is True
+    assert proxy_info['failures'] == 0
+@pytest.mark.asyncio
+@patch('httpx.AsyncClient')
+async def test_health_check_failure(mock_client_class):
+    """测试健康检查：失败"""
+    mock_client_class.return_value.__aenter__.return_value.get.side_effect = httpx.ConnectError("Failed")
+    proxy_info = {'url': 'http://bad:8080', 'healthy': True, 'failures': 0}
+    await check_single_proxy(proxy_info)
+    assert proxy_info['healthy'] is False
     assert proxy_info['failures'] == 1

crawlo 1.2.2__py3-none-any.whl → 1.2.4__py3-none-any.whl

Potentially problematic release.

crawlo 1.2.2py3-none-any.whl → 1.2.4py3-none-any.whl