crawlo 1.2.6__py3-none-any.whl → 1.2.8__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.

Potentially problematic release.


This version of crawlo might be problematic. Click here for more details.

Files changed (209) hide show
  1. crawlo/__init__.py +61 -61
  2. crawlo/__version__.py +1 -1
  3. crawlo/cleaners/__init__.py +60 -60
  4. crawlo/cleaners/data_formatter.py +225 -225
  5. crawlo/cleaners/encoding_converter.py +125 -125
  6. crawlo/cleaners/text_cleaner.py +232 -232
  7. crawlo/cli.py +75 -88
  8. crawlo/commands/__init__.py +14 -14
  9. crawlo/commands/check.py +594 -594
  10. crawlo/commands/genspider.py +151 -151
  11. crawlo/commands/help.py +138 -144
  12. crawlo/commands/list.py +155 -155
  13. crawlo/commands/run.py +323 -323
  14. crawlo/commands/startproject.py +436 -436
  15. crawlo/commands/stats.py +187 -187
  16. crawlo/commands/utils.py +186 -186
  17. crawlo/config.py +312 -312
  18. crawlo/config_validator.py +251 -251
  19. crawlo/core/__init__.py +2 -2
  20. crawlo/core/engine.py +365 -356
  21. crawlo/core/processor.py +40 -40
  22. crawlo/core/scheduler.py +251 -239
  23. crawlo/crawler.py +1099 -1110
  24. crawlo/data/__init__.py +5 -5
  25. crawlo/data/user_agents.py +107 -107
  26. crawlo/downloader/__init__.py +266 -266
  27. crawlo/downloader/aiohttp_downloader.py +228 -221
  28. crawlo/downloader/cffi_downloader.py +256 -256
  29. crawlo/downloader/httpx_downloader.py +259 -259
  30. crawlo/downloader/hybrid_downloader.py +212 -212
  31. crawlo/downloader/playwright_downloader.py +402 -402
  32. crawlo/downloader/selenium_downloader.py +472 -472
  33. crawlo/event.py +11 -11
  34. crawlo/exceptions.py +81 -81
  35. crawlo/extension/__init__.py +39 -38
  36. crawlo/extension/health_check.py +141 -141
  37. crawlo/extension/log_interval.py +57 -57
  38. crawlo/extension/log_stats.py +81 -81
  39. crawlo/extension/logging_extension.py +43 -43
  40. crawlo/extension/memory_monitor.py +104 -104
  41. crawlo/extension/performance_profiler.py +133 -133
  42. crawlo/extension/request_recorder.py +107 -107
  43. crawlo/filters/__init__.py +154 -154
  44. crawlo/filters/aioredis_filter.py +234 -234
  45. crawlo/filters/memory_filter.py +269 -269
  46. crawlo/items/__init__.py +23 -23
  47. crawlo/items/base.py +21 -21
  48. crawlo/items/fields.py +52 -52
  49. crawlo/items/items.py +104 -104
  50. crawlo/middleware/__init__.py +21 -21
  51. crawlo/middleware/default_header.py +131 -131
  52. crawlo/middleware/download_delay.py +104 -104
  53. crawlo/middleware/middleware_manager.py +136 -135
  54. crawlo/middleware/offsite.py +114 -114
  55. crawlo/middleware/proxy.py +367 -367
  56. crawlo/middleware/request_ignore.py +86 -86
  57. crawlo/middleware/response_code.py +163 -163
  58. crawlo/middleware/response_filter.py +136 -136
  59. crawlo/middleware/retry.py +124 -124
  60. crawlo/mode_manager.py +211 -211
  61. crawlo/network/__init__.py +21 -21
  62. crawlo/network/request.py +338 -338
  63. crawlo/network/response.py +359 -359
  64. crawlo/pipelines/__init__.py +21 -21
  65. crawlo/pipelines/bloom_dedup_pipeline.py +156 -156
  66. crawlo/pipelines/console_pipeline.py +39 -39
  67. crawlo/pipelines/csv_pipeline.py +316 -316
  68. crawlo/pipelines/database_dedup_pipeline.py +222 -222
  69. crawlo/pipelines/json_pipeline.py +218 -218
  70. crawlo/pipelines/memory_dedup_pipeline.py +115 -115
  71. crawlo/pipelines/mongo_pipeline.py +131 -131
  72. crawlo/pipelines/mysql_pipeline.py +317 -317
  73. crawlo/pipelines/pipeline_manager.py +62 -61
  74. crawlo/pipelines/redis_dedup_pipeline.py +166 -165
  75. crawlo/project.py +314 -279
  76. crawlo/queue/pqueue.py +37 -37
  77. crawlo/queue/queue_manager.py +377 -376
  78. crawlo/queue/redis_priority_queue.py +306 -306
  79. crawlo/settings/__init__.py +7 -7
  80. crawlo/settings/default_settings.py +219 -215
  81. crawlo/settings/setting_manager.py +122 -122
  82. crawlo/spider/__init__.py +639 -639
  83. crawlo/stats_collector.py +59 -59
  84. crawlo/subscriber.py +129 -129
  85. crawlo/task_manager.py +30 -30
  86. crawlo/templates/crawlo.cfg.tmpl +10 -10
  87. crawlo/templates/project/__init__.py.tmpl +3 -3
  88. crawlo/templates/project/items.py.tmpl +17 -17
  89. crawlo/templates/project/middlewares.py.tmpl +118 -118
  90. crawlo/templates/project/pipelines.py.tmpl +96 -96
  91. crawlo/templates/project/settings.py.tmpl +288 -288
  92. crawlo/templates/project/settings_distributed.py.tmpl +157 -157
  93. crawlo/templates/project/settings_gentle.py.tmpl +100 -100
  94. crawlo/templates/project/settings_high_performance.py.tmpl +134 -134
  95. crawlo/templates/project/settings_simple.py.tmpl +98 -98
  96. crawlo/templates/project/spiders/__init__.py.tmpl +5 -5
  97. crawlo/templates/run.py.tmpl +47 -45
  98. crawlo/templates/spider/spider.py.tmpl +143 -143
  99. crawlo/tools/__init__.py +182 -182
  100. crawlo/tools/anti_crawler.py +268 -268
  101. crawlo/tools/authenticated_proxy.py +240 -240
  102. crawlo/tools/data_validator.py +180 -180
  103. crawlo/tools/date_tools.py +35 -35
  104. crawlo/tools/distributed_coordinator.py +386 -386
  105. crawlo/tools/retry_mechanism.py +220 -220
  106. crawlo/tools/scenario_adapter.py +262 -262
  107. crawlo/utils/__init__.py +35 -35
  108. crawlo/utils/batch_processor.py +259 -259
  109. crawlo/utils/controlled_spider_mixin.py +439 -439
  110. crawlo/utils/date_tools.py +290 -290
  111. crawlo/utils/db_helper.py +343 -343
  112. crawlo/utils/enhanced_error_handler.py +356 -356
  113. crawlo/utils/env_config.py +143 -106
  114. crawlo/utils/error_handler.py +123 -123
  115. crawlo/utils/func_tools.py +82 -82
  116. crawlo/utils/large_scale_config.py +286 -286
  117. crawlo/utils/large_scale_helper.py +344 -344
  118. crawlo/utils/log.py +128 -128
  119. crawlo/utils/performance_monitor.py +285 -285
  120. crawlo/utils/queue_helper.py +175 -175
  121. crawlo/utils/redis_connection_pool.py +351 -351
  122. crawlo/utils/redis_key_validator.py +198 -198
  123. crawlo/utils/request.py +267 -267
  124. crawlo/utils/request_serializer.py +218 -218
  125. crawlo/utils/spider_loader.py +61 -61
  126. crawlo/utils/system.py +11 -11
  127. crawlo/utils/tools.py +4 -4
  128. crawlo/utils/url.py +39 -39
  129. {crawlo-1.2.6.dist-info → crawlo-1.2.8.dist-info}/METADATA +764 -764
  130. crawlo-1.2.8.dist-info/RECORD +209 -0
  131. examples/__init__.py +7 -7
  132. tests/DOUBLE_CRAWLO_PREFIX_FIX_REPORT.md +81 -81
  133. tests/__init__.py +7 -7
  134. tests/advanced_tools_example.py +275 -275
  135. tests/authenticated_proxy_example.py +236 -236
  136. tests/cleaners_example.py +160 -160
  137. tests/config_validation_demo.py +102 -102
  138. tests/controlled_spider_example.py +205 -205
  139. tests/date_tools_example.py +180 -180
  140. tests/dynamic_loading_example.py +523 -523
  141. tests/dynamic_loading_test.py +104 -104
  142. tests/env_config_example.py +133 -133
  143. tests/error_handling_example.py +171 -171
  144. tests/redis_key_validation_demo.py +130 -130
  145. tests/response_improvements_example.py +144 -144
  146. tests/test_advanced_tools.py +148 -148
  147. tests/test_all_redis_key_configs.py +145 -145
  148. tests/test_authenticated_proxy.py +141 -141
  149. tests/test_cleaners.py +54 -54
  150. tests/test_comprehensive.py +146 -146
  151. tests/test_config_consistency.py +81 -0
  152. tests/test_config_validator.py +193 -193
  153. tests/test_crawlo_proxy_integration.py +172 -172
  154. tests/test_date_tools.py +123 -123
  155. tests/test_default_header_middleware.py +158 -158
  156. tests/test_double_crawlo_fix.py +207 -207
  157. tests/test_double_crawlo_fix_simple.py +124 -124
  158. tests/test_download_delay_middleware.py +221 -221
  159. tests/test_downloader_proxy_compatibility.py +268 -268
  160. tests/test_dynamic_downloaders_proxy.py +124 -124
  161. tests/test_dynamic_proxy.py +92 -92
  162. tests/test_dynamic_proxy_config.py +146 -146
  163. tests/test_dynamic_proxy_real.py +109 -109
  164. tests/test_edge_cases.py +303 -303
  165. tests/test_enhanced_error_handler.py +270 -270
  166. tests/test_env_config.py +121 -121
  167. tests/test_error_handler_compatibility.py +112 -112
  168. tests/test_final_validation.py +153 -153
  169. tests/test_framework_env_usage.py +103 -103
  170. tests/test_integration.py +356 -356
  171. tests/test_item_dedup_redis_key.py +122 -122
  172. tests/test_mode_consistency.py +52 -0
  173. tests/test_offsite_middleware.py +221 -221
  174. tests/test_parsel.py +29 -29
  175. tests/test_performance.py +327 -327
  176. tests/test_proxy_api.py +264 -264
  177. tests/test_proxy_health_check.py +32 -32
  178. tests/test_proxy_middleware.py +121 -121
  179. tests/test_proxy_middleware_enhanced.py +216 -216
  180. tests/test_proxy_middleware_integration.py +136 -136
  181. tests/test_proxy_providers.py +56 -56
  182. tests/test_proxy_stats.py +19 -19
  183. tests/test_proxy_strategies.py +59 -59
  184. tests/test_queue_manager_double_crawlo.py +173 -173
  185. tests/test_queue_manager_redis_key.py +176 -176
  186. tests/test_real_scenario_proxy.py +195 -195
  187. tests/test_redis_config.py +28 -28
  188. tests/test_redis_connection_pool.py +294 -294
  189. tests/test_redis_key_naming.py +181 -181
  190. tests/test_redis_key_validator.py +123 -123
  191. tests/test_redis_queue.py +224 -224
  192. tests/test_request_ignore_middleware.py +182 -182
  193. tests/test_request_serialization.py +70 -70
  194. tests/test_response_code_middleware.py +349 -349
  195. tests/test_response_filter_middleware.py +427 -427
  196. tests/test_response_improvements.py +152 -152
  197. tests/test_retry_middleware.py +241 -241
  198. tests/test_scheduler.py +252 -241
  199. tests/test_scheduler_config_update.py +134 -0
  200. tests/test_simple_response.py +61 -61
  201. tests/test_telecom_spider_redis_key.py +205 -205
  202. tests/test_template_content.py +87 -87
  203. tests/test_template_redis_key.py +134 -134
  204. tests/test_tools.py +153 -153
  205. tests/tools_example.py +257 -257
  206. crawlo-1.2.6.dist-info/RECORD +0 -206
  207. {crawlo-1.2.6.dist-info → crawlo-1.2.8.dist-info}/WHEEL +0 -0
  208. {crawlo-1.2.6.dist-info → crawlo-1.2.8.dist-info}/entry_points.txt +0 -0
  209. {crawlo-1.2.6.dist-info → crawlo-1.2.8.dist-info}/top_level.txt +0 -0
@@ -1,146 +1,146 @@
1
- #!/usr/bin/env python3
2
- # -*- coding: utf-8 -*-
3
- """
4
- 所有Redis Key配置测试脚本
5
- 用于验证所有配置文件是否符合新的Redis key命名规范
6
- """
7
- import sys
8
- import os
9
- import re
10
-
11
- # 添加项目根目录到路径
12
- sys.path.insert(0, os.path.join(os.path.dirname(__file__), '..'))
13
-
14
-
15
- def test_all_redis_key_configs():
16
- """测试所有Redis key配置"""
17
- print("🔍 测试所有Redis key配置...")
18
-
19
- try:
20
- # 检查示例项目配置文件
21
- example_projects = [
22
- "examples/books_distributed/books_distributed/settings.py",
23
- "examples/api_data_collection/api_data_collection/settings.py",
24
- "examples/telecom_licenses_distributed/telecom_licenses_distributed/settings.py"
25
- ]
26
-
27
- for project_config in example_projects:
28
- print(f" 检查 {project_config}...")
29
- if not os.path.exists(project_config):
30
- print(f"❌ 配置文件不存在: {project_config}")
31
- return False
32
-
33
- with open(project_config, 'r', encoding='utf-8') as f:
34
- content = f.read()
35
-
36
- # 检查是否移除了旧的REDIS_KEY配置
37
- if re.search(r'REDIS_KEY\s*=', content) and 'crawlo:{PROJECT_NAME}:filter:fingerprint' not in content:
38
- print(f"❌ {project_config}中仍然存在旧的REDIS_KEY配置")
39
- return False
40
-
41
- # 检查是否添加了新的注释
42
- if 'crawlo:{PROJECT_NAME}:filter:fingerprint' not in content:
43
- print(f"❌ {project_config}中缺少新的Redis key命名规范注释")
44
- return False
45
-
46
- print(f" ✅ {project_config}符合新的Redis key命名规范")
47
-
48
- # 检查模板文件
49
- template_file = "crawlo/templates/project/settings.py.tmpl"
50
- print(f" 检查 {template_file}...")
51
- if not os.path.exists(template_file):
52
- print(f"❌ 模板文件不存在: {template_file}")
53
- return False
54
-
55
- with open(template_file, 'r', encoding='utf-8') as f:
56
- template_content = f.read()
57
-
58
- # 检查是否移除了旧的REDIS_KEY配置
59
- if "REDIS_KEY = f'{{project_name}}:fingerprint'" in template_content:
60
- print("❌ 模板文件中仍然存在旧的REDIS_KEY配置")
61
- return False
62
-
63
- # 检查是否添加了新的注释
64
- if '# crawlo:{project_name}:filter:fingerprint (请求去重)' not in template_content:
65
- print("❌ 模板文件中缺少请求去重的Redis key命名规范注释")
66
- return False
67
-
68
- if '# crawlo:{project_name}:item:fingerprint (数据项去重)' not in template_content:
69
- print("❌ 模板文件中缺少数据项去重的Redis key命名规范注释")
70
- return False
71
-
72
- print(f" ✅ {template_file}符合新的Redis key命名规范")
73
-
74
- # 检查mode_manager.py
75
- mode_manager_file = "crawlo/mode_manager.py"
76
- print(f" 检查 {mode_manager_file}...")
77
- if not os.path.exists(mode_manager_file):
78
- print(f"❌ 文件不存在: {mode_manager_file}")
79
- return False
80
-
81
- with open(mode_manager_file, 'r', encoding='utf-8') as f:
82
- mode_manager_content = f.read()
83
-
84
- # 检查是否移除了旧的REDIS_KEY配置
85
- if "'REDIS_KEY': f'{project_name}:fingerprint'" in mode_manager_content:
86
- print("❌ mode_manager.py中仍然存在旧的REDIS_KEY配置")
87
- return False
88
-
89
- # 检查是否添加了新的注释
90
- if 'crawlo:{project_name}:filter:fingerprint (请求去重)' not in mode_manager_content:
91
- print("❌ mode_manager.py中缺少新的Redis key命名规范注释")
92
- return False
93
-
94
- print(f" ✅ {mode_manager_file}符合新的Redis key命名规范")
95
-
96
- # 检查默认设置文件
97
- default_settings_file = "crawlo/settings/default_settings.py"
98
- print(f" 检查 {default_settings_file}...")
99
- if not os.path.exists(default_settings_file):
100
- print(f"❌ 文件不存在: {default_settings_file}")
101
- return False
102
-
103
- with open(default_settings_file, 'r', encoding='utf-8') as f:
104
- default_settings_content = f.read()
105
-
106
- # 检查是否移除了旧的REDIS_KEY配置
107
- if re.search(r'REDIS_KEY\s*=\s*.*fingerprint', default_settings_content):
108
- print("❌ 默认设置文件中仍然存在旧的REDIS_KEY配置")
109
- return False
110
-
111
- print(f" ✅ {default_settings_file}符合新的Redis key命名规范")
112
-
113
- print("✅ 所有Redis key配置测试通过!")
114
- return True
115
-
116
- except Exception as e:
117
- print(f"❌ 测试过程中发生错误: {e}")
118
- return False
119
-
120
-
121
- def main():
122
- """主测试函数"""
123
- print("🚀 开始所有Redis key配置测试...")
124
- print("=" * 50)
125
-
126
- try:
127
- success = test_all_redis_key_configs()
128
-
129
- print("=" * 50)
130
- if success:
131
- print("🎉 所有测试通过!所有配置文件符合新的Redis key命名规范")
132
- else:
133
- print("❌ 测试失败,请检查配置文件")
134
- return 1
135
-
136
- except Exception as e:
137
- print("=" * 50)
138
- print(f"❌ 测试过程中发生异常: {e}")
139
- return 1
140
-
141
- return 0
142
-
143
-
144
- if __name__ == "__main__":
145
- exit_code = main()
1
+ #!/usr/bin/env python3
2
+ # -*- coding: utf-8 -*-
3
+ """
4
+ 所有Redis Key配置测试脚本
5
+ 用于验证所有配置文件是否符合新的Redis key命名规范
6
+ """
7
+ import sys
8
+ import os
9
+ import re
10
+
11
+ # 添加项目根目录到路径
12
+ sys.path.insert(0, os.path.join(os.path.dirname(__file__), '..'))
13
+
14
+
15
+ def test_all_redis_key_configs():
16
+ """测试所有Redis key配置"""
17
+ print("🔍 测试所有Redis key配置...")
18
+
19
+ try:
20
+ # 检查示例项目配置文件
21
+ example_projects = [
22
+ "examples/books_distributed/books_distributed/settings.py",
23
+ "examples/api_data_collection/api_data_collection/settings.py",
24
+ "examples/telecom_licenses_distributed/telecom_licenses_distributed/settings.py"
25
+ ]
26
+
27
+ for project_config in example_projects:
28
+ print(f" 检查 {project_config}...")
29
+ if not os.path.exists(project_config):
30
+ print(f"❌ 配置文件不存在: {project_config}")
31
+ return False
32
+
33
+ with open(project_config, 'r', encoding='utf-8') as f:
34
+ content = f.read()
35
+
36
+ # 检查是否移除了旧的REDIS_KEY配置
37
+ if re.search(r'REDIS_KEY\s*=', content) and 'crawlo:{PROJECT_NAME}:filter:fingerprint' not in content:
38
+ print(f"❌ {project_config}中仍然存在旧的REDIS_KEY配置")
39
+ return False
40
+
41
+ # 检查是否添加了新的注释
42
+ if 'crawlo:{PROJECT_NAME}:filter:fingerprint' not in content:
43
+ print(f"❌ {project_config}中缺少新的Redis key命名规范注释")
44
+ return False
45
+
46
+ print(f" ✅ {project_config}符合新的Redis key命名规范")
47
+
48
+ # 检查模板文件
49
+ template_file = "crawlo/templates/project/settings.py.tmpl"
50
+ print(f" 检查 {template_file}...")
51
+ if not os.path.exists(template_file):
52
+ print(f"❌ 模板文件不存在: {template_file}")
53
+ return False
54
+
55
+ with open(template_file, 'r', encoding='utf-8') as f:
56
+ template_content = f.read()
57
+
58
+ # 检查是否移除了旧的REDIS_KEY配置
59
+ if "REDIS_KEY = f'{{project_name}}:fingerprint'" in template_content:
60
+ print("❌ 模板文件中仍然存在旧的REDIS_KEY配置")
61
+ return False
62
+
63
+ # 检查是否添加了新的注释
64
+ if '# crawlo:{project_name}:filter:fingerprint (请求去重)' not in template_content:
65
+ print("❌ 模板文件中缺少请求去重的Redis key命名规范注释")
66
+ return False
67
+
68
+ if '# crawlo:{project_name}:item:fingerprint (数据项去重)' not in template_content:
69
+ print("❌ 模板文件中缺少数据项去重的Redis key命名规范注释")
70
+ return False
71
+
72
+ print(f" ✅ {template_file}符合新的Redis key命名规范")
73
+
74
+ # 检查mode_manager.py
75
+ mode_manager_file = "crawlo/mode_manager.py"
76
+ print(f" 检查 {mode_manager_file}...")
77
+ if not os.path.exists(mode_manager_file):
78
+ print(f"❌ 文件不存在: {mode_manager_file}")
79
+ return False
80
+
81
+ with open(mode_manager_file, 'r', encoding='utf-8') as f:
82
+ mode_manager_content = f.read()
83
+
84
+ # 检查是否移除了旧的REDIS_KEY配置
85
+ if "'REDIS_KEY': f'{project_name}:fingerprint'" in mode_manager_content:
86
+ print("❌ mode_manager.py中仍然存在旧的REDIS_KEY配置")
87
+ return False
88
+
89
+ # 检查是否添加了新的注释
90
+ if 'crawlo:{project_name}:filter:fingerprint (请求去重)' not in mode_manager_content:
91
+ print("❌ mode_manager.py中缺少新的Redis key命名规范注释")
92
+ return False
93
+
94
+ print(f" ✅ {mode_manager_file}符合新的Redis key命名规范")
95
+
96
+ # 检查默认设置文件
97
+ default_settings_file = "crawlo/settings/default_settings.py"
98
+ print(f" 检查 {default_settings_file}...")
99
+ if not os.path.exists(default_settings_file):
100
+ print(f"❌ 文件不存在: {default_settings_file}")
101
+ return False
102
+
103
+ with open(default_settings_file, 'r', encoding='utf-8') as f:
104
+ default_settings_content = f.read()
105
+
106
+ # 检查是否移除了旧的REDIS_KEY配置
107
+ if re.search(r'REDIS_KEY\s*=\s*.*fingerprint', default_settings_content):
108
+ print("❌ 默认设置文件中仍然存在旧的REDIS_KEY配置")
109
+ return False
110
+
111
+ print(f" ✅ {default_settings_file}符合新的Redis key命名规范")
112
+
113
+ print("✅ 所有Redis key配置测试通过!")
114
+ return True
115
+
116
+ except Exception as e:
117
+ print(f"❌ 测试过程中发生错误: {e}")
118
+ return False
119
+
120
+
121
+ def main():
122
+ """主测试函数"""
123
+ print("🚀 开始所有Redis key配置测试...")
124
+ print("=" * 50)
125
+
126
+ try:
127
+ success = test_all_redis_key_configs()
128
+
129
+ print("=" * 50)
130
+ if success:
131
+ print("🎉 所有测试通过!所有配置文件符合新的Redis key命名规范")
132
+ else:
133
+ print("❌ 测试失败,请检查配置文件")
134
+ return 1
135
+
136
+ except Exception as e:
137
+ print("=" * 50)
138
+ print(f"❌ 测试过程中发生异常: {e}")
139
+ return 1
140
+
141
+ return 0
142
+
143
+
144
+ if __name__ == "__main__":
145
+ exit_code = main()
146
146
  sys.exit(exit_code)
@@ -1,142 +1,142 @@
1
- #!/usr/bin/python
2
- # -*- coding: UTF-8 -*-
3
- """
4
- 测试带认证代理的功能
5
- """
6
-
7
- import asyncio
8
- import aiohttp
9
- import httpx
10
- from crawlo.network.request import Request
11
- from crawlo.tools import AuthenticatedProxy
12
-
13
-
14
- async def test_proxy_with_aiohttp():
15
- """测试AioHttp与认证代理"""
16
- print("=== 测试AioHttp与认证代理 ===")
17
-
18
- # 代理配置
19
- proxy_config = {
20
- "http": "http://dwe20241014:Dwe0101014@182.201.243.186:58111",
21
- "https": "http://dwe20241014:Dwe0101014@182.201.243.186:58111"
22
- }
23
-
24
- # 创建代理对象
25
- proxy_url = proxy_config["http"]
26
- proxy = AuthenticatedProxy(proxy_url)
27
-
28
- print(f"原始代理URL: {proxy_url}")
29
- print(f"清洁URL: {proxy.clean_url}")
30
- print(f"认证信息: {proxy.get_auth_credentials()}")
31
-
32
- # 使用aiohttp直接测试
33
- try:
34
- auth = proxy.get_auth_credentials()
35
- if auth:
36
- basic_auth = aiohttp.BasicAuth(auth['username'], auth['password'])
37
- else:
38
- basic_auth = None
39
-
40
- async with aiohttp.ClientSession() as session:
41
- async with session.get(
42
- "https://httpbin.org/ip",
43
- proxy=proxy.clean_url,
44
- proxy_auth=basic_auth
45
- ) as response:
46
- print(f"AioHttp测试成功!")
47
- print(f"状态码: {response.status}")
48
- content = await response.text()
49
- print(f"响应内容: {content[:200]}...")
50
-
51
- except Exception as e:
52
- print(f"AioHttp测试失败: {e}")
53
- import traceback
54
- traceback.print_exc()
55
-
56
-
57
- def test_proxy_with_httpx():
58
- """测试HttpX与认证代理"""
59
- print("\n=== 测试HttpX与认证代理 ===")
60
-
61
- # 代理配置
62
- proxy_config = {
63
- "http": "http://dwe20241014:Dwe0101014@182.201.243.186:58111",
64
- "https": "http://dwe20241014:Dwe0101014@182.201.243.186:58111"
65
- }
66
-
67
- # 使用httpx直接测试
68
- try:
69
- # HttpX可以直接使用带认证的URL作为proxy参数
70
- proxy_url = proxy_config["http"]
71
-
72
- with httpx.Client(proxy=proxy_url) as client:
73
- response = client.get("https://httpbin.org/ip")
74
- print(f"HttpX测试成功!")
75
- print(f"状态码: {response.status_code}")
76
- print(f"响应内容: {response.text[:200]}...")
77
-
78
- except Exception as e:
79
- print(f"HttpX测试失败: {e}")
80
- import traceback
81
- traceback.print_exc()
82
-
83
-
84
- async def test_proxy_with_curl_cffi():
85
- """测试CurlCffi与认证代理"""
86
- print("\n=== 测试CurlCffi与认证代理 ===")
87
-
88
- # 代理配置
89
- proxy_config = {
90
- "http": "http://dwe20241014:Dwe0101014@182.201.243.186:58111",
91
- "https": "http://dwe20241014:Dwe0101014@182.201.243.186:58111"
92
- }
93
-
94
- # 创建代理对象
95
- proxy_url = proxy_config["http"]
96
- proxy = AuthenticatedProxy(proxy_url)
97
-
98
- print(f"原始代理URL: {proxy_url}")
99
- print(f"代理字典: {proxy.proxy_dict}")
100
- print(f"认证头: {proxy.get_auth_header()}")
101
-
102
- # 使用curl-cffi直接测试
103
- try:
104
- from curl_cffi import requests as curl_requests
105
-
106
- # 设置代理和认证头
107
- proxies = proxy.proxy_dict
108
- headers = {}
109
- auth_header = proxy.get_auth_header()
110
- if auth_header:
111
- headers["Proxy-Authorization"] = auth_header
112
-
113
- response = curl_requests.get(
114
- "https://httpbin.org/ip",
115
- proxies=proxies,
116
- headers=headers
117
- )
118
-
119
- print(f"CurlCffi测试成功!")
120
- print(f"状态码: {response.status_code}")
121
- print(f"响应内容: {response.text[:200]}...")
122
-
123
- except Exception as e:
124
- print(f"CurlCffi测试失败: {e}")
125
- import traceback
126
- traceback.print_exc()
127
-
128
-
129
- async def main():
130
- """主测试函数"""
131
- print("开始测试带认证代理的功能...\n")
132
-
133
- # 测试各个库
134
- await test_proxy_with_aiohttp()
135
- test_proxy_with_httpx()
136
- await test_proxy_with_curl_cffi()
137
-
138
- print("\n所有测试完成!")
139
-
140
-
141
- if __name__ == "__main__":
1
+ #!/usr/bin/python
2
+ # -*- coding: UTF-8 -*-
3
+ """
4
+ 测试带认证代理的功能
5
+ """
6
+
7
+ import asyncio
8
+ import aiohttp
9
+ import httpx
10
+ from crawlo.network.request import Request
11
+ from crawlo.tools import AuthenticatedProxy
12
+
13
+
14
+ async def test_proxy_with_aiohttp():
15
+ """测试AioHttp与认证代理"""
16
+ print("=== 测试AioHttp与认证代理 ===")
17
+
18
+ # 代理配置
19
+ proxy_config = {
20
+ "http": "http://dwe20241014:Dwe0101014@182.201.243.186:58111",
21
+ "https": "http://dwe20241014:Dwe0101014@182.201.243.186:58111"
22
+ }
23
+
24
+ # 创建代理对象
25
+ proxy_url = proxy_config["http"]
26
+ proxy = AuthenticatedProxy(proxy_url)
27
+
28
+ print(f"原始代理URL: {proxy_url}")
29
+ print(f"清洁URL: {proxy.clean_url}")
30
+ print(f"认证信息: {proxy.get_auth_credentials()}")
31
+
32
+ # 使用aiohttp直接测试
33
+ try:
34
+ auth = proxy.get_auth_credentials()
35
+ if auth:
36
+ basic_auth = aiohttp.BasicAuth(auth['username'], auth['password'])
37
+ else:
38
+ basic_auth = None
39
+
40
+ async with aiohttp.ClientSession() as session:
41
+ async with session.get(
42
+ "https://httpbin.org/ip",
43
+ proxy=proxy.clean_url,
44
+ proxy_auth=basic_auth
45
+ ) as response:
46
+ print(f"AioHttp测试成功!")
47
+ print(f"状态码: {response.status}")
48
+ content = await response.text()
49
+ print(f"响应内容: {content[:200]}...")
50
+
51
+ except Exception as e:
52
+ print(f"AioHttp测试失败: {e}")
53
+ import traceback
54
+ traceback.print_exc()
55
+
56
+
57
+ def test_proxy_with_httpx():
58
+ """测试HttpX与认证代理"""
59
+ print("\n=== 测试HttpX与认证代理 ===")
60
+
61
+ # 代理配置
62
+ proxy_config = {
63
+ "http": "http://dwe20241014:Dwe0101014@182.201.243.186:58111",
64
+ "https": "http://dwe20241014:Dwe0101014@182.201.243.186:58111"
65
+ }
66
+
67
+ # 使用httpx直接测试
68
+ try:
69
+ # HttpX可以直接使用带认证的URL作为proxy参数
70
+ proxy_url = proxy_config["http"]
71
+
72
+ with httpx.Client(proxy=proxy_url) as client:
73
+ response = client.get("https://httpbin.org/ip")
74
+ print(f"HttpX测试成功!")
75
+ print(f"状态码: {response.status_code}")
76
+ print(f"响应内容: {response.text[:200]}...")
77
+
78
+ except Exception as e:
79
+ print(f"HttpX测试失败: {e}")
80
+ import traceback
81
+ traceback.print_exc()
82
+
83
+
84
+ async def test_proxy_with_curl_cffi():
85
+ """测试CurlCffi与认证代理"""
86
+ print("\n=== 测试CurlCffi与认证代理 ===")
87
+
88
+ # 代理配置
89
+ proxy_config = {
90
+ "http": "http://dwe20241014:Dwe0101014@182.201.243.186:58111",
91
+ "https": "http://dwe20241014:Dwe0101014@182.201.243.186:58111"
92
+ }
93
+
94
+ # 创建代理对象
95
+ proxy_url = proxy_config["http"]
96
+ proxy = AuthenticatedProxy(proxy_url)
97
+
98
+ print(f"原始代理URL: {proxy_url}")
99
+ print(f"代理字典: {proxy.proxy_dict}")
100
+ print(f"认证头: {proxy.get_auth_header()}")
101
+
102
+ # 使用curl-cffi直接测试
103
+ try:
104
+ from curl_cffi import requests as curl_requests
105
+
106
+ # 设置代理和认证头
107
+ proxies = proxy.proxy_dict
108
+ headers = {}
109
+ auth_header = proxy.get_auth_header()
110
+ if auth_header:
111
+ headers["Proxy-Authorization"] = auth_header
112
+
113
+ response = curl_requests.get(
114
+ "https://httpbin.org/ip",
115
+ proxies=proxies,
116
+ headers=headers
117
+ )
118
+
119
+ print(f"CurlCffi测试成功!")
120
+ print(f"状态码: {response.status_code}")
121
+ print(f"响应内容: {response.text[:200]}...")
122
+
123
+ except Exception as e:
124
+ print(f"CurlCffi测试失败: {e}")
125
+ import traceback
126
+ traceback.print_exc()
127
+
128
+
129
+ async def main():
130
+ """主测试函数"""
131
+ print("开始测试带认证代理的功能...\n")
132
+
133
+ # 测试各个库
134
+ await test_proxy_with_aiohttp()
135
+ test_proxy_with_httpx()
136
+ await test_proxy_with_curl_cffi()
137
+
138
+ print("\n所有测试完成!")
139
+
140
+
141
+ if __name__ == "__main__":
142
142
  asyncio.run(main())