crawlo 1.4.5__py3-none-any.whl → 1.4.7__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.

Potentially problematic release.


This version of crawlo might be problematic. Click here for more details.

Files changed (375) hide show
  1. crawlo/__init__.py +90 -89
  2. crawlo/__version__.py +1 -1
  3. crawlo/cli.py +75 -75
  4. crawlo/commands/__init__.py +14 -14
  5. crawlo/commands/check.py +594 -594
  6. crawlo/commands/genspider.py +186 -186
  7. crawlo/commands/help.py +140 -138
  8. crawlo/commands/list.py +155 -155
  9. crawlo/commands/run.py +379 -341
  10. crawlo/commands/startproject.py +460 -460
  11. crawlo/commands/stats.py +187 -187
  12. crawlo/commands/utils.py +196 -196
  13. crawlo/config.py +320 -312
  14. crawlo/config_validator.py +277 -277
  15. crawlo/core/__init__.py +52 -52
  16. crawlo/core/engine.py +451 -438
  17. crawlo/core/processor.py +47 -47
  18. crawlo/core/scheduler.py +290 -291
  19. crawlo/crawler.py +698 -657
  20. crawlo/data/__init__.py +5 -5
  21. crawlo/data/user_agents.py +194 -194
  22. crawlo/downloader/__init__.py +280 -276
  23. crawlo/downloader/aiohttp_downloader.py +233 -233
  24. crawlo/downloader/cffi_downloader.py +250 -245
  25. crawlo/downloader/httpx_downloader.py +265 -259
  26. crawlo/downloader/hybrid_downloader.py +212 -212
  27. crawlo/downloader/playwright_downloader.py +425 -402
  28. crawlo/downloader/selenium_downloader.py +486 -472
  29. crawlo/event.py +45 -11
  30. crawlo/exceptions.py +215 -82
  31. crawlo/extension/__init__.py +65 -64
  32. crawlo/extension/health_check.py +141 -141
  33. crawlo/extension/log_interval.py +94 -94
  34. crawlo/extension/log_stats.py +70 -70
  35. crawlo/extension/logging_extension.py +53 -61
  36. crawlo/extension/memory_monitor.py +104 -104
  37. crawlo/extension/performance_profiler.py +133 -133
  38. crawlo/extension/request_recorder.py +107 -107
  39. crawlo/factories/__init__.py +27 -27
  40. crawlo/factories/base.py +68 -68
  41. crawlo/factories/crawler.py +104 -103
  42. crawlo/factories/registry.py +84 -84
  43. crawlo/factories/utils.py +135 -0
  44. crawlo/filters/__init__.py +170 -153
  45. crawlo/filters/aioredis_filter.py +348 -264
  46. crawlo/filters/memory_filter.py +261 -276
  47. crawlo/framework.py +306 -292
  48. crawlo/initialization/__init__.py +44 -44
  49. crawlo/initialization/built_in.py +391 -434
  50. crawlo/initialization/context.py +141 -141
  51. crawlo/initialization/core.py +240 -194
  52. crawlo/initialization/phases.py +230 -149
  53. crawlo/initialization/registry.py +143 -145
  54. crawlo/initialization/utils.py +49 -0
  55. crawlo/interfaces.py +23 -23
  56. crawlo/items/__init__.py +23 -23
  57. crawlo/items/base.py +23 -23
  58. crawlo/items/fields.py +52 -52
  59. crawlo/items/items.py +104 -104
  60. crawlo/logging/__init__.py +42 -46
  61. crawlo/logging/config.py +277 -197
  62. crawlo/logging/factory.py +175 -171
  63. crawlo/logging/manager.py +104 -112
  64. crawlo/middleware/__init__.py +87 -24
  65. crawlo/middleware/default_header.py +132 -132
  66. crawlo/middleware/download_delay.py +104 -104
  67. crawlo/middleware/middleware_manager.py +142 -142
  68. crawlo/middleware/offsite.py +123 -123
  69. crawlo/middleware/proxy.py +209 -386
  70. crawlo/middleware/request_ignore.py +86 -86
  71. crawlo/middleware/response_code.py +150 -150
  72. crawlo/middleware/response_filter.py +136 -136
  73. crawlo/middleware/retry.py +124 -124
  74. crawlo/mode_manager.py +287 -253
  75. crawlo/network/__init__.py +21 -21
  76. crawlo/network/request.py +375 -379
  77. crawlo/network/response.py +569 -664
  78. crawlo/pipelines/__init__.py +53 -22
  79. crawlo/pipelines/base_pipeline.py +452 -0
  80. crawlo/pipelines/bloom_dedup_pipeline.py +146 -146
  81. crawlo/pipelines/console_pipeline.py +39 -39
  82. crawlo/pipelines/csv_pipeline.py +316 -316
  83. crawlo/pipelines/database_dedup_pipeline.py +197 -197
  84. crawlo/pipelines/json_pipeline.py +218 -218
  85. crawlo/pipelines/memory_dedup_pipeline.py +105 -105
  86. crawlo/pipelines/mongo_pipeline.py +140 -132
  87. crawlo/pipelines/mysql_pipeline.py +470 -326
  88. crawlo/pipelines/pipeline_manager.py +100 -100
  89. crawlo/pipelines/redis_dedup_pipeline.py +155 -156
  90. crawlo/project.py +347 -347
  91. crawlo/queue/__init__.py +10 -0
  92. crawlo/queue/pqueue.py +38 -38
  93. crawlo/queue/queue_manager.py +591 -525
  94. crawlo/queue/redis_priority_queue.py +519 -370
  95. crawlo/settings/__init__.py +7 -7
  96. crawlo/settings/default_settings.py +285 -270
  97. crawlo/settings/setting_manager.py +219 -219
  98. crawlo/spider/__init__.py +657 -657
  99. crawlo/stats_collector.py +82 -73
  100. crawlo/subscriber.py +129 -129
  101. crawlo/task_manager.py +138 -138
  102. crawlo/templates/crawlo.cfg.tmpl +10 -10
  103. crawlo/templates/project/__init__.py.tmpl +2 -4
  104. crawlo/templates/project/items.py.tmpl +13 -17
  105. crawlo/templates/project/middlewares.py.tmpl +38 -38
  106. crawlo/templates/project/pipelines.py.tmpl +35 -36
  107. crawlo/templates/project/settings.py.tmpl +110 -157
  108. crawlo/templates/project/settings_distributed.py.tmpl +156 -161
  109. crawlo/templates/project/settings_gentle.py.tmpl +170 -171
  110. crawlo/templates/project/settings_high_performance.py.tmpl +171 -172
  111. crawlo/templates/project/settings_minimal.py.tmpl +99 -77
  112. crawlo/templates/project/settings_simple.py.tmpl +168 -169
  113. crawlo/templates/project/spiders/__init__.py.tmpl +9 -9
  114. crawlo/templates/run.py.tmpl +23 -30
  115. crawlo/templates/spider/spider.py.tmpl +33 -144
  116. crawlo/templates/spiders_init.py.tmpl +5 -10
  117. crawlo/tools/__init__.py +86 -189
  118. crawlo/tools/date_tools.py +289 -289
  119. crawlo/tools/distributed_coordinator.py +384 -384
  120. crawlo/tools/scenario_adapter.py +262 -262
  121. crawlo/tools/text_cleaner.py +232 -232
  122. crawlo/utils/__init__.py +50 -50
  123. crawlo/utils/batch_processor.py +276 -259
  124. crawlo/utils/config_manager.py +442 -0
  125. crawlo/utils/controlled_spider_mixin.py +439 -439
  126. crawlo/utils/db_helper.py +250 -244
  127. crawlo/utils/error_handler.py +410 -410
  128. crawlo/utils/fingerprint.py +121 -121
  129. crawlo/utils/func_tools.py +82 -82
  130. crawlo/utils/large_scale_helper.py +344 -344
  131. crawlo/utils/leak_detector.py +335 -0
  132. crawlo/utils/log.py +79 -79
  133. crawlo/utils/misc.py +81 -81
  134. crawlo/utils/mongo_connection_pool.py +157 -0
  135. crawlo/utils/mysql_connection_pool.py +197 -0
  136. crawlo/utils/performance_monitor.py +285 -285
  137. crawlo/utils/queue_helper.py +175 -175
  138. crawlo/utils/redis_checker.py +91 -0
  139. crawlo/utils/redis_connection_pool.py +578 -388
  140. crawlo/utils/redis_key_validator.py +198 -198
  141. crawlo/utils/request.py +278 -256
  142. crawlo/utils/request_serializer.py +225 -225
  143. crawlo/utils/resource_manager.py +337 -0
  144. crawlo/utils/selector_helper.py +137 -137
  145. crawlo/utils/singleton.py +70 -0
  146. crawlo/utils/spider_loader.py +201 -201
  147. crawlo/utils/text_helper.py +94 -94
  148. crawlo/utils/{url.py → url_utils.py} +39 -39
  149. crawlo-1.4.7.dist-info/METADATA +689 -0
  150. crawlo-1.4.7.dist-info/RECORD +347 -0
  151. examples/__init__.py +7 -7
  152. tests/__init__.py +7 -7
  153. tests/advanced_tools_example.py +217 -275
  154. tests/authenticated_proxy_example.py +110 -106
  155. tests/baidu_performance_test.py +108 -108
  156. tests/baidu_test.py +59 -59
  157. tests/bug_check_test.py +250 -250
  158. tests/cleaners_example.py +160 -160
  159. tests/comprehensive_framework_test.py +212 -212
  160. tests/comprehensive_test.py +81 -81
  161. tests/comprehensive_testing_summary.md +186 -186
  162. tests/config_validation_demo.py +142 -142
  163. tests/controlled_spider_example.py +205 -205
  164. tests/date_tools_example.py +180 -180
  165. tests/debug_configure.py +69 -69
  166. tests/debug_framework_logger.py +84 -84
  167. tests/debug_log_config.py +126 -126
  168. tests/debug_log_levels.py +63 -63
  169. tests/debug_pipelines.py +66 -66
  170. tests/detailed_log_test.py +233 -233
  171. tests/direct_selector_helper_test.py +96 -96
  172. tests/distributed_dedup_test.py +467 -0
  173. tests/distributed_test.py +66 -66
  174. tests/distributed_test_debug.py +76 -76
  175. tests/dynamic_loading_example.py +523 -523
  176. tests/dynamic_loading_test.py +104 -104
  177. tests/error_handling_example.py +171 -171
  178. tests/explain_mysql_update_behavior.py +77 -0
  179. tests/final_comprehensive_test.py +151 -151
  180. tests/final_log_test.py +260 -260
  181. tests/final_validation_test.py +182 -182
  182. tests/fix_log_test.py +142 -142
  183. tests/framework_performance_test.py +202 -202
  184. tests/log_buffering_test.py +111 -111
  185. tests/log_generation_timing_test.py +153 -153
  186. tests/monitor_redis_dedup.sh +72 -0
  187. tests/ofweek_scrapy/ofweek_scrapy/items.py +12 -12
  188. tests/ofweek_scrapy/ofweek_scrapy/middlewares.py +100 -100
  189. tests/ofweek_scrapy/ofweek_scrapy/pipelines.py +13 -13
  190. tests/ofweek_scrapy/ofweek_scrapy/settings.py +84 -84
  191. tests/ofweek_scrapy/scrapy.cfg +11 -11
  192. tests/optimized_performance_test.py +211 -211
  193. tests/performance_comparison.py +244 -244
  194. tests/queue_blocking_test.py +113 -113
  195. tests/queue_test.py +89 -89
  196. tests/redis_key_validation_demo.py +130 -130
  197. tests/request_params_example.py +150 -150
  198. tests/response_improvements_example.py +144 -144
  199. tests/scrapy_comparison/ofweek_scrapy.py +138 -138
  200. tests/scrapy_comparison/scrapy_test.py +133 -133
  201. tests/simple_cli_test.py +55 -0
  202. tests/simple_command_test.py +119 -119
  203. tests/simple_crawlo_test.py +126 -126
  204. tests/simple_follow_test.py +38 -38
  205. tests/simple_log_test2.py +137 -137
  206. tests/simple_optimization_test.py +128 -128
  207. tests/simple_queue_type_test.py +41 -41
  208. tests/simple_response_selector_test.py +94 -94
  209. tests/simple_selector_helper_test.py +154 -154
  210. tests/simple_selector_test.py +207 -207
  211. tests/simple_spider_test.py +49 -49
  212. tests/simple_url_test.py +73 -73
  213. tests/simulate_mysql_update_test.py +140 -0
  214. tests/spider_log_timing_test.py +177 -177
  215. tests/test_advanced_tools.py +148 -148
  216. tests/test_all_commands.py +230 -230
  217. tests/test_all_pipeline_fingerprints.py +133 -133
  218. tests/test_all_redis_key_configs.py +145 -145
  219. tests/test_asyncmy_usage.py +57 -0
  220. tests/test_batch_processor.py +178 -178
  221. tests/test_cleaners.py +54 -54
  222. tests/test_cli_arguments.py +119 -0
  223. tests/test_component_factory.py +174 -174
  224. tests/test_config_consistency.py +80 -80
  225. tests/test_config_merge.py +152 -152
  226. tests/test_config_validator.py +182 -182
  227. tests/test_controlled_spider_mixin.py +79 -79
  228. tests/test_crawler_process_import.py +38 -38
  229. tests/test_crawler_process_spider_modules.py +47 -47
  230. tests/test_crawlo_proxy_integration.py +114 -108
  231. tests/test_date_tools.py +123 -123
  232. tests/test_dedup_fix.py +220 -220
  233. tests/test_dedup_pipeline_consistency.py +124 -124
  234. tests/test_default_header_middleware.py +313 -313
  235. tests/test_distributed.py +65 -65
  236. tests/test_double_crawlo_fix.py +204 -204
  237. tests/test_double_crawlo_fix_simple.py +124 -124
  238. tests/test_download_delay_middleware.py +221 -221
  239. tests/test_downloader_proxy_compatibility.py +272 -268
  240. tests/test_edge_cases.py +305 -305
  241. tests/test_encoding_core.py +56 -56
  242. tests/test_encoding_detection.py +126 -126
  243. tests/test_enhanced_error_handler.py +270 -270
  244. tests/test_enhanced_error_handler_comprehensive.py +245 -245
  245. tests/test_error_handler_compatibility.py +112 -112
  246. tests/test_factories.py +252 -252
  247. tests/test_factory_compatibility.py +196 -196
  248. tests/test_final_validation.py +153 -153
  249. tests/test_fingerprint_consistency.py +135 -135
  250. tests/test_fingerprint_simple.py +51 -51
  251. tests/test_get_component_logger.py +83 -83
  252. tests/test_hash_performance.py +99 -99
  253. tests/test_integration.py +169 -169
  254. tests/test_item_dedup_redis_key.py +122 -122
  255. tests/test_large_scale_helper.py +235 -235
  256. tests/test_logging_enhancements.py +374 -374
  257. tests/test_logging_final.py +184 -184
  258. tests/test_logging_integration.py +312 -312
  259. tests/test_logging_system.py +282 -282
  260. tests/test_middleware_debug.py +141 -141
  261. tests/test_mode_consistency.py +51 -51
  262. tests/test_multi_directory.py +67 -67
  263. tests/test_multiple_spider_modules.py +80 -80
  264. tests/test_mysql_pipeline_config.py +165 -0
  265. tests/test_mysql_pipeline_error.py +99 -0
  266. tests/test_mysql_pipeline_init_log.py +83 -0
  267. tests/test_mysql_pipeline_integration.py +133 -0
  268. tests/test_mysql_pipeline_refactor.py +144 -0
  269. tests/test_mysql_pipeline_refactor_simple.py +86 -0
  270. tests/test_mysql_pipeline_robustness.py +196 -0
  271. tests/test_mysql_pipeline_types.py +89 -0
  272. tests/test_mysql_update_columns.py +94 -0
  273. tests/test_offsite_middleware.py +244 -244
  274. tests/test_offsite_middleware_simple.py +203 -203
  275. tests/test_optimized_selector_naming.py +100 -100
  276. tests/test_parsel.py +29 -29
  277. tests/test_performance.py +327 -327
  278. tests/test_performance_monitor.py +115 -115
  279. tests/test_pipeline_fingerprint_consistency.py +86 -86
  280. tests/test_priority_behavior.py +211 -211
  281. tests/test_priority_consistency.py +151 -151
  282. tests/test_priority_consistency_fixed.py +249 -249
  283. tests/test_proxy_health_check.py +32 -32
  284. tests/test_proxy_middleware.py +217 -121
  285. tests/test_proxy_middleware_enhanced.py +212 -216
  286. tests/test_proxy_middleware_integration.py +142 -137
  287. tests/test_proxy_middleware_refactored.py +207 -184
  288. tests/test_proxy_only.py +84 -0
  289. tests/test_proxy_providers.py +56 -56
  290. tests/test_proxy_stats.py +19 -19
  291. tests/test_proxy_strategies.py +59 -59
  292. tests/test_proxy_with_downloader.py +153 -0
  293. tests/test_queue_empty_check.py +41 -41
  294. tests/test_queue_manager_double_crawlo.py +173 -173
  295. tests/test_queue_manager_redis_key.py +179 -179
  296. tests/test_queue_naming.py +154 -154
  297. tests/test_queue_type.py +106 -106
  298. tests/test_queue_type_redis_config_consistency.py +130 -130
  299. tests/test_random_headers_default.py +322 -322
  300. tests/test_random_headers_necessity.py +308 -308
  301. tests/test_random_user_agent.py +72 -72
  302. tests/test_redis_config.py +28 -28
  303. tests/test_redis_connection_pool.py +294 -294
  304. tests/test_redis_key_naming.py +181 -181
  305. tests/test_redis_key_validator.py +123 -123
  306. tests/test_redis_queue.py +224 -224
  307. tests/test_redis_queue_name_fix.py +175 -175
  308. tests/test_redis_queue_type_fallback.py +129 -129
  309. tests/test_request_ignore_middleware.py +182 -182
  310. tests/test_request_params.py +111 -111
  311. tests/test_request_serialization.py +70 -70
  312. tests/test_response_code_middleware.py +349 -349
  313. tests/test_response_filter_middleware.py +427 -427
  314. tests/test_response_follow.py +104 -104
  315. tests/test_response_improvements.py +152 -152
  316. tests/test_response_selector_methods.py +92 -92
  317. tests/test_response_url_methods.py +70 -70
  318. tests/test_response_urljoin.py +86 -86
  319. tests/test_retry_middleware.py +333 -333
  320. tests/test_retry_middleware_realistic.py +273 -273
  321. tests/test_scheduler.py +252 -252
  322. tests/test_scheduler_config_update.py +133 -133
  323. tests/test_scrapy_style_encoding.py +112 -112
  324. tests/test_selector_helper.py +100 -100
  325. tests/test_selector_optimizations.py +146 -146
  326. tests/test_simple_response.py +61 -61
  327. tests/test_spider_loader.py +49 -49
  328. tests/test_spider_loader_comprehensive.py +69 -69
  329. tests/test_spider_modules.py +84 -84
  330. tests/test_spiders/test_spider.py +9 -9
  331. tests/test_telecom_spider_redis_key.py +205 -205
  332. tests/test_template_content.py +87 -87
  333. tests/test_template_redis_key.py +134 -134
  334. tests/test_tools.py +159 -159
  335. tests/test_user_agent_randomness.py +176 -176
  336. tests/test_user_agents.py +96 -96
  337. tests/untested_features_report.md +138 -138
  338. tests/verify_debug.py +51 -51
  339. tests/verify_distributed.py +117 -117
  340. tests/verify_log_fix.py +111 -111
  341. tests/verify_mysql_warnings.py +110 -0
  342. crawlo/logging/async_handler.py +0 -181
  343. crawlo/logging/monitor.py +0 -153
  344. crawlo/logging/sampler.py +0 -167
  345. crawlo/middleware/simple_proxy.py +0 -65
  346. crawlo/tools/authenticated_proxy.py +0 -241
  347. crawlo/tools/data_formatter.py +0 -226
  348. crawlo/tools/data_validator.py +0 -181
  349. crawlo/tools/encoding_converter.py +0 -127
  350. crawlo/tools/network_diagnostic.py +0 -365
  351. crawlo/tools/request_tools.py +0 -83
  352. crawlo/tools/retry_mechanism.py +0 -224
  353. crawlo/utils/env_config.py +0 -143
  354. crawlo/utils/large_scale_config.py +0 -287
  355. crawlo/utils/system.py +0 -11
  356. crawlo/utils/tools.py +0 -5
  357. crawlo-1.4.5.dist-info/METADATA +0 -329
  358. crawlo-1.4.5.dist-info/RECORD +0 -347
  359. tests/env_config_example.py +0 -134
  360. tests/ofweek_scrapy/ofweek_scrapy/spiders/ofweek_spider.py +0 -162
  361. tests/test_authenticated_proxy.py +0 -142
  362. tests/test_comprehensive.py +0 -147
  363. tests/test_dynamic_downloaders_proxy.py +0 -125
  364. tests/test_dynamic_proxy.py +0 -93
  365. tests/test_dynamic_proxy_config.py +0 -147
  366. tests/test_dynamic_proxy_real.py +0 -110
  367. tests/test_env_config.py +0 -122
  368. tests/test_framework_env_usage.py +0 -104
  369. tests/test_large_scale_config.py +0 -113
  370. tests/test_proxy_api.py +0 -265
  371. tests/test_real_scenario_proxy.py +0 -196
  372. tests/tools_example.py +0 -261
  373. {crawlo-1.4.5.dist-info → crawlo-1.4.7.dist-info}/WHEEL +0 -0
  374. {crawlo-1.4.5.dist-info → crawlo-1.4.7.dist-info}/entry_points.txt +0 -0
  375. {crawlo-1.4.5.dist-info → crawlo-1.4.7.dist-info}/top_level.txt +0 -0
@@ -1,142 +0,0 @@
1
- #!/usr/bin/python
2
- # -*- coding: UTF-8 -*-
3
- """
4
- 测试带认证代理的功能
5
- """
6
-
7
- import asyncio
8
- import aiohttp
9
- import httpx
10
- from crawlo.network.request import Request
11
- from crawlo.tools import AuthenticatedProxy
12
-
13
-
14
- async def test_proxy_with_aiohttp():
15
- """测试AioHttp与认证代理"""
16
- print("=== 测试AioHttp与认证代理 ===")
17
-
18
- # 代理配置
19
- proxy_config = {
20
- "http": "http://dwe20241014:Dwe0101014@182.201.243.186:58111",
21
- "https": "http://dwe20241014:Dwe0101014@182.201.243.186:58111"
22
- }
23
-
24
- # 创建代理对象
25
- proxy_url = proxy_config["http"]
26
- proxy = AuthenticatedProxy(proxy_url)
27
-
28
- print(f"原始代理URL: {proxy_url}")
29
- print(f"清洁URL: {proxy.clean_url}")
30
- print(f"认证信息: {proxy.get_auth_credentials()}")
31
-
32
- # 使用aiohttp直接测试
33
- try:
34
- auth = proxy.get_auth_credentials()
35
- if auth:
36
- basic_auth = aiohttp.BasicAuth(auth['username'], auth['password'])
37
- else:
38
- basic_auth = None
39
-
40
- async with aiohttp.ClientSession() as session:
41
- async with session.get(
42
- "https://httpbin.org/ip",
43
- proxy=proxy.clean_url,
44
- proxy_auth=basic_auth
45
- ) as response:
46
- print(f"AioHttp测试成功!")
47
- print(f"状态码: {response.status}")
48
- content = await response.text()
49
- print(f"响应内容: {content[:200]}...")
50
-
51
- except Exception as e:
52
- print(f"AioHttp测试失败: {e}")
53
- import traceback
54
- traceback.print_exc()
55
-
56
-
57
- def test_proxy_with_httpx():
58
- """测试HttpX与认证代理"""
59
- print("\n=== 测试HttpX与认证代理 ===")
60
-
61
- # 代理配置
62
- proxy_config = {
63
- "http": "http://dwe20241014:Dwe0101014@182.201.243.186:58111",
64
- "https": "http://dwe20241014:Dwe0101014@182.201.243.186:58111"
65
- }
66
-
67
- # 使用httpx直接测试
68
- try:
69
- # HttpX可以直接使用带认证的URL作为proxy参数
70
- proxy_url = proxy_config["http"]
71
-
72
- with httpx.Client(proxy=proxy_url) as client:
73
- response = client.get("https://httpbin.org/ip")
74
- print(f"HttpX测试成功!")
75
- print(f"状态码: {response.status_code}")
76
- print(f"响应内容: {response.text[:200]}...")
77
-
78
- except Exception as e:
79
- print(f"HttpX测试失败: {e}")
80
- import traceback
81
- traceback.print_exc()
82
-
83
-
84
- async def test_proxy_with_curl_cffi():
85
- """测试CurlCffi与认证代理"""
86
- print("\n=== 测试CurlCffi与认证代理 ===")
87
-
88
- # 代理配置
89
- proxy_config = {
90
- "http": "http://dwe20241014:Dwe0101014@182.201.243.186:58111",
91
- "https": "http://dwe20241014:Dwe0101014@182.201.243.186:58111"
92
- }
93
-
94
- # 创建代理对象
95
- proxy_url = proxy_config["http"]
96
- proxy = AuthenticatedProxy(proxy_url)
97
-
98
- print(f"原始代理URL: {proxy_url}")
99
- print(f"代理字典: {proxy.proxy_dict}")
100
- print(f"认证头: {proxy.get_auth_header()}")
101
-
102
- # 使用curl-cffi直接测试
103
- try:
104
- from curl_cffi import requests as curl_requests
105
-
106
- # 设置代理和认证头
107
- proxies = proxy.proxy_dict
108
- headers = {}
109
- auth_header = proxy.get_auth_header()
110
- if auth_header:
111
- headers["Proxy-Authorization"] = auth_header
112
-
113
- response = curl_requests.get(
114
- "https://httpbin.org/ip",
115
- proxies=proxies,
116
- headers=headers
117
- )
118
-
119
- print(f"CurlCffi测试成功!")
120
- print(f"状态码: {response.status_code}")
121
- print(f"响应内容: {response.text[:200]}...")
122
-
123
- except Exception as e:
124
- print(f"CurlCffi测试失败: {e}")
125
- import traceback
126
- traceback.print_exc()
127
-
128
-
129
- async def main():
130
- """主测试函数"""
131
- print("开始测试带认证代理的功能...\n")
132
-
133
- # 测试各个库
134
- await test_proxy_with_aiohttp()
135
- test_proxy_with_httpx()
136
- await test_proxy_with_curl_cffi()
137
-
138
- print("\n所有测试完成!")
139
-
140
-
141
- if __name__ == "__main__":
142
- asyncio.run(main())
@@ -1,147 +0,0 @@
1
- #!/usr/bin/env python3
2
- # -*- coding: utf-8 -*-
3
- """
4
- 综合测试
5
- 验证所有改进的集成效果
6
- """
7
- import sys
8
- import os
9
- import asyncio
10
- import unittest
11
- from unittest.mock import patch, MagicMock
12
-
13
- # 添加项目根目录到Python路径
14
- sys.path.insert(0, os.path.join(os.path.dirname(__file__), '..'))
15
-
16
- from crawlo.utils.env_config import get_env_var, get_redis_config, get_runtime_config
17
- from crawlo.utils.error_handler import ErrorHandler, handle_exception
18
- from crawlo.core.engine import Engine
19
- from crawlo.settings.setting_manager import SettingManager
20
- from crawlo.settings import default_settings
21
- from crawlo.queue.queue_manager import QueueManager, QueueConfig, QueueType
22
-
23
-
24
- class TestComprehensiveIntegration(unittest.TestCase):
25
- """综合集成测试"""
26
-
27
- def setUp(self):
28
- """测试前准备"""
29
- # 设置测试环境变量
30
- self.test_env = {
31
- 'PROJECT_NAME': 'test_project',
32
- 'CONCURRENCY': '4',
33
- 'REDIS_HOST': 'localhost',
34
- 'REDIS_PORT': '6379'
35
- }
36
- self.original_env = {}
37
- for key, value in self.test_env.items():
38
- self.original_env[key] = os.environ.get(key)
39
- os.environ[key] = value
40
-
41
- def tearDown(self):
42
- """测试后清理"""
43
- # 恢复原始环境变量
44
- for key, value in self.original_env.items():
45
- if value is None:
46
- os.environ.pop(key, None)
47
- else:
48
- os.environ[key] = value
49
-
50
- def test_env_config_integration(self):
51
- """测试环境变量配置集成"""
52
- # 验证环境变量工具正常工作
53
- project_name = get_env_var('PROJECT_NAME', 'default', str)
54
- self.assertEqual(project_name, 'test_project')
55
-
56
- concurrency = get_env_var('CONCURRENCY', 1, int)
57
- self.assertEqual(concurrency, 4)
58
-
59
- # 验证Redis配置工具
60
- redis_config = get_redis_config()
61
- self.assertEqual(redis_config['REDIS_HOST'], 'localhost')
62
- self.assertEqual(redis_config['REDIS_PORT'], 6379)
63
-
64
- def test_error_handler_integration(self):
65
- """测试错误处理集成"""
66
- # 验证错误处理模块正常工作
67
- error_handler = ErrorHandler("test")
68
-
69
- # 测试错误处理
70
- try:
71
- error_handler.handle_error(ValueError("Test error"), raise_error=False)
72
- except Exception:
73
- self.fail("Error handler should not raise exception when raise_error=False")
74
-
75
- # 测试安全调用
76
- result = error_handler.safe_call(lambda x: x*2, 5, default_return=0)
77
- self.assertEqual(result, 10)
78
-
79
- # 测试装饰器
80
- @handle_exception(raise_error=False)
81
- def failing_function():
82
- raise RuntimeError("Test")
83
-
84
- try:
85
- failing_function()
86
- except Exception:
87
- self.fail("Decorated function should not raise exception")
88
-
89
- def test_settings_integration(self):
90
- """测试设置管理器集成"""
91
- # 重新加载默认设置以获取环境变量
92
- import importlib
93
- import crawlo.settings.default_settings
94
- importlib.reload(crawlo.settings.default_settings)
95
-
96
- # 创建设置管理器
97
- settings = SettingManager()
98
- settings.set_settings(crawlo.settings.default_settings)
99
-
100
- # 验证设置正确加载
101
- self.assertEqual(settings.get('PROJECT_NAME'), 'test_project')
102
- self.assertEqual(settings.get_int('CONCURRENCY'), 4)
103
- self.assertEqual(settings.get('REDIS_HOST'), 'localhost')
104
-
105
- def test_queue_manager_config(self):
106
- """测试队列管理器配置"""
107
- # 重新加载默认设置
108
- import importlib
109
- import crawlo.settings.default_settings
110
- importlib.reload(crawlo.settings.default_settings)
111
-
112
- # 创建设置管理器
113
- settings = SettingManager()
114
- settings.set_settings(crawlo.settings.default_settings)
115
-
116
- # 从设置创建队列配置
117
- queue_config = QueueConfig.from_settings(settings)
118
-
119
- # 验证配置正确
120
- self.assertEqual(queue_config.queue_type, QueueType.AUTO)
121
- self.assertIn('test_project', queue_config.queue_name)
122
-
123
- async def test_async_components(self):
124
- """测试异步组件"""
125
- # 测试异步错误处理装饰器
126
- @handle_exception(raise_error=False)
127
- async def async_failing_function():
128
- raise RuntimeError("Async test")
129
-
130
- try:
131
- await async_failing_function()
132
- except Exception:
133
- self.fail("Async decorated function should not raise exception")
134
-
135
-
136
- if __name__ == '__main__':
137
- # 运行同步测试
138
- unittest.main(exit=False)
139
-
140
- # 运行异步测试
141
- async def run_async_tests():
142
- test_instance = TestComprehensiveIntegration()
143
- test_instance.setUp()
144
- await test_instance.test_async_components()
145
- test_instance.tearDown()
146
-
147
- asyncio.run(run_async_tests())
@@ -1,125 +0,0 @@
1
- #!/usr/bin/python
2
- # -*- coding: UTF-8 -*-
3
- """
4
- 测试动态下载器(Selenium和Playwright)的代理功能
5
- """
6
-
7
- import asyncio
8
- from crawlo.tools import AuthenticatedProxy
9
-
10
-
11
- def test_selenium_proxy_configuration():
12
- """测试Selenium下载器的代理配置"""
13
- print("=== 测试Selenium下载器的代理配置 ===")
14
-
15
- # 代理配置
16
- proxy_config = {
17
- "http": "http://dwe20241014:Dwe0101014@182.201.243.186:58111",
18
- "https": "http://dwe20241014:Dwe0101014@182.201.243.186:58111"
19
- }
20
-
21
- # 创建代理对象
22
- proxy_url = proxy_config["http"]
23
- proxy = AuthenticatedProxy(proxy_url)
24
-
25
- print(f"原始代理URL: {proxy_url}")
26
- print(f"清洁URL: {proxy.clean_url}")
27
- print(f"认证信息: {proxy.get_auth_credentials()}")
28
-
29
- # Selenium的代理设置方式
30
- print(f"\nSelenium代理设置方式:")
31
- print(f" 1. 在爬虫设置中配置:")
32
- print(f" settings = {{")
33
- print(f" 'SELENIUM_PROXY': '{proxy.clean_url}',")
34
- print(f" }}")
35
-
36
- # 对于带认证的代理,需要特殊处理
37
- if proxy.username and proxy.password:
38
- print(f"\n 2. 带认证代理的处理:")
39
- print(f" - 用户名: {proxy.username}")
40
- print(f" - 密码: {proxy.password}")
41
- print(f" - 认证头: {proxy.get_auth_header()}")
42
- print(f" - 注意: Selenium需要通过扩展或其他方式处理认证")
43
-
44
- print("\nSelenium测试完成!")
45
-
46
-
47
- async def test_playwright_proxy_configuration():
48
- """测试Playwright下载器的代理配置"""
49
- print("\n=== 测试Playwright下载器的代理配置 ===")
50
-
51
- # 代理配置
52
- proxy_config = {
53
- "http": "http://dwe20241014:Dwe0101014@182.201.243.186:58111",
54
- "https": "http://dwe20241014:Dwe0101014@182.201.243.186:58111"
55
- }
56
-
57
- # 创建代理对象
58
- proxy_url = proxy_config["http"]
59
- proxy = AuthenticatedProxy(proxy_url)
60
-
61
- print(f"原始代理URL: {proxy_url}")
62
- print(f"清洁URL: {proxy.clean_url}")
63
- print(f"认证信息: {proxy.get_auth_credentials()}")
64
-
65
- # Playwright的代理设置方式
66
- print(f"\nPlaywright代理设置方式:")
67
- print(f" 1. 简单代理配置:")
68
- print(f" settings = {{")
69
- print(f" 'PLAYWRIGHT_PROXY': '{proxy.clean_url}',")
70
- print(f" }}")
71
-
72
- # 对于带认证的代理,Playwright可以直接在代理配置中包含认证信息
73
- if proxy.username and proxy.password:
74
- print(f"\n 2. 带认证的代理配置:")
75
- print(f" settings = {{")
76
- print(f" 'PLAYWRIGHT_PROXY': {{")
77
- print(f" 'server': '{proxy.clean_url}',")
78
- print(f" 'username': '{proxy.username}',")
79
- print(f" 'password': '{proxy.password}'")
80
- print(f" }}")
81
- print(f" }}")
82
-
83
- print("\nPlaywright测试完成!")
84
-
85
-
86
- def show_proxy_usage_examples():
87
- """显示代理使用示例"""
88
- print("\n=== 代理使用示例 ===")
89
-
90
- # 代理配置示例
91
- proxy_examples = [
92
- "http://username:password@proxy.example.com:8080", # 带认证HTTP代理
93
- "https://user:pass@secure-proxy.example.com:443", # 带认证HTTPS代理
94
- "http://proxy.example.com:8080", # 不带认证代理
95
- "socks5://username:password@socks-proxy.example.com:1080" # SOCKS5代理
96
- ]
97
-
98
- for i, proxy_url in enumerate(proxy_examples, 1):
99
- print(f"\n示例 {i}: {proxy_url}")
100
- try:
101
- proxy = AuthenticatedProxy(proxy_url)
102
- print(f" 清洁URL: {proxy.clean_url}")
103
- print(f" 用户名: {proxy.username or '无'}")
104
- print(f" 密码: {proxy.password or '无'}")
105
- print(f" 是否有效: {proxy.is_valid()}")
106
- if proxy.username and proxy.password:
107
- print(f" 认证头: {proxy.get_auth_header()}")
108
- except Exception as e:
109
- print(f" 错误: {e}")
110
-
111
-
112
- async def main():
113
- """主测试函数"""
114
- print("开始测试动态下载器的代理功能...\n")
115
-
116
- # 测试各个下载器
117
- test_selenium_proxy_configuration()
118
- await test_playwright_proxy_configuration()
119
- show_proxy_usage_examples()
120
-
121
- print("\n所有测试完成!")
122
-
123
-
124
- if __name__ == "__main__":
125
- asyncio.run(main())
@@ -1,93 +0,0 @@
1
- #!/usr/bin/python
2
- # -*- coding: UTF-8 -*-
3
- """
4
- 测试动态下载器(Selenium和Playwright)的代理功能
5
- """
6
-
7
- import asyncio
8
- from crawlo.network.request import Request
9
- from crawlo.tools import AuthenticatedProxy
10
-
11
-
12
- def test_proxy_with_selenium():
13
- """测试Selenium下载器与认证代理"""
14
- print("=== 测试Selenium下载器与认证代理 ===")
15
-
16
- # 代理配置
17
- proxy_config = {
18
- "http": "http://dwe20241014:Dwe0101014@182.201.243.186:58111",
19
- "https": "http://dwe20241014:Dwe0101014@182.201.243.186:58111"
20
- }
21
-
22
- # 创建代理对象
23
- proxy_url = proxy_config["http"]
24
- proxy = AuthenticatedProxy(proxy_url)
25
-
26
- print(f"原始代理URL: {proxy_url}")
27
- print(f"清洁URL: {proxy.clean_url}")
28
- print(f"认证信息: {proxy.get_auth_credentials()}")
29
-
30
- # Selenium的代理设置方式
31
- print(f"Selenium代理设置方式:")
32
- print(f" 1. 在设置中配置: SELENIUM_PROXY = '{proxy.clean_url}'")
33
- print(f" 2. 认证信息需要通过其他方式处理")
34
-
35
- # 对于带认证的代理,Selenium需要特殊处理
36
- if proxy.username and proxy.password:
37
- print(f" 3. 认证信息:")
38
- print(f" 用户名: {proxy.username}")
39
- print(f" 密码: {proxy.password}")
40
- print(f" 认证头: {proxy.get_auth_header()}")
41
-
42
- print("Selenium测试完成!")
43
-
44
-
45
- async def test_proxy_with_playwright():
46
- """测试Playwright下载器与认证代理"""
47
- print("\n=== 测试Playwright下载器与认证代理 ===")
48
-
49
- # 代理配置
50
- proxy_config = {
51
- "http": "http://dwe20241014:Dwe0101014@182.201.243.186:58111",
52
- "https": "http://dwe20241014:Dwe0101014@182.201.243.186:58111"
53
- }
54
-
55
- # 创建代理对象
56
- proxy_url = proxy_config["http"]
57
- proxy = AuthenticatedProxy(proxy_url)
58
-
59
- print(f"原始代理URL: {proxy_url}")
60
- print(f"清洁URL: {proxy.clean_url}")
61
- print(f"认证信息: {proxy.get_auth_credentials()}")
62
-
63
- # Playwright的代理设置方式
64
- print(f"Playwright代理设置方式:")
65
- print(f" 1. 在启动浏览器时配置代理:")
66
- print(f" browser = await playwright.chromium.launch(proxy={{'server': '{proxy.clean_url}'}})")
67
-
68
- # 对于带认证的代理,Playwright需要在代理配置中包含认证信息
69
- if proxy.username and proxy.password:
70
- print(f" 2. 带认证的代理配置:")
71
- print(f" proxy_config = {{")
72
- print(f" 'server': '{proxy.clean_url}',")
73
- print(f" 'username': '{proxy.username}',")
74
- print(f" 'password': '{proxy.password}'")
75
- print(f" }}")
76
- print(f" browser = await playwright.chromium.launch(proxy=proxy_config)")
77
-
78
- print("Playwright测试完成!")
79
-
80
-
81
- async def main():
82
- """主测试函数"""
83
- print("开始测试动态下载器的代理功能...\n")
84
-
85
- # 测试各个下载器
86
- test_proxy_with_selenium()
87
- await test_proxy_with_playwright()
88
-
89
- print("\n所有测试完成!")
90
-
91
-
92
- if __name__ == "__main__":
93
- asyncio.run(main())
@@ -1,147 +0,0 @@
1
- #!/usr/bin/python
2
- # -*- coding: UTF-8 -*-
3
- """
4
- 测试动态下载器(Selenium和Playwright)的代理配置逻辑
5
- """
6
-
7
- from crawlo.tools import AuthenticatedProxy
8
-
9
-
10
- def test_selenium_proxy_logic():
11
- """测试Selenium下载器的代理配置逻辑"""
12
- print("=== 测试Selenium下载器的代理配置逻辑 ===")
13
-
14
- # 代理配置
15
- proxy_config = {
16
- "http": "http://dwe20241014:Dwe0101014@182.201.243.186:58111",
17
- "https": "http://dwe20241014:Dwe0101014@182.201.243.186:58111"
18
- }
19
-
20
- # 创建代理对象
21
- proxy_url = proxy_config["http"]
22
- proxy = AuthenticatedProxy(proxy_url)
23
-
24
- print(f"原始代理URL: {proxy_url}")
25
- print(f"清洁URL: {proxy.clean_url}")
26
- print(f"认证信息: {proxy.get_auth_credentials()}")
27
-
28
- # 模拟Selenium的代理配置逻辑
29
- print(f"\nSelenium代理配置逻辑:")
30
- print(f" 1. 在爬虫设置中配置:")
31
- print(f" settings = {{")
32
- print(f" 'SELENIUM_PROXY': '{proxy.clean_url}',")
33
- print(f" }}")
34
-
35
- # 对于带认证的代理,需要特殊处理
36
- if proxy.username and proxy.password:
37
- print(f"\n 2. 带认证代理的处理逻辑:")
38
- print(f" - 用户名: {proxy.username}")
39
- print(f" - 密码: {proxy.password}")
40
- print(f" - 认证头: {proxy.get_auth_header()}")
41
- print(f" - 处理方式: 通过浏览器扩展或手动输入认证信息")
42
-
43
- print("\nSelenium配置逻辑测试完成!")
44
-
45
-
46
- def test_playwright_proxy_logic():
47
- """测试Playwright下载器的代理配置逻辑"""
48
- print("\n=== 测试Playwright下载器的代理配置逻辑 ===")
49
-
50
- # 代理配置
51
- proxy_config = {
52
- "http": "http://dwe20241014:Dwe0101014@182.201.243.186:58111",
53
- "https": "http://dwe20241014:Dwe0101014@182.201.243.186:58111"
54
- }
55
-
56
- # 创建代理对象
57
- proxy_url = proxy_config["http"]
58
- proxy = AuthenticatedProxy(proxy_url)
59
-
60
- print(f"原始代理URL: {proxy_url}")
61
- print(f"清洁URL: {proxy.clean_url}")
62
- print(f"认证信息: {proxy.get_auth_credentials()}")
63
-
64
- # 模拟Playwright的代理配置逻辑
65
- print(f"\nPlaywright代理配置逻辑:")
66
- print(f" 1. 简单代理配置:")
67
- print(f" settings = {{")
68
- print(f" 'PLAYWRIGHT_PROXY': '{proxy.clean_url}',")
69
- print(f" }}")
70
-
71
- # 对于带认证的代理,Playwright可以直接在代理配置中包含认证信息
72
- if proxy.username and proxy.password:
73
- print(f"\n 2. 带认证的代理配置逻辑:")
74
- print(f" settings = {{")
75
- print(f" 'PLAYWRIGHT_PROXY': {{")
76
- print(f" 'server': '{proxy.clean_url}',")
77
- print(f" 'username': '{proxy.username}',")
78
- print(f" 'password': '{proxy.password}'")
79
- print(f" }}")
80
- print(f" }}")
81
- print(f" 实现方式: 在启动浏览器时传递proxy参数")
82
-
83
- print("\nPlaywright配置逻辑测试完成!")
84
-
85
-
86
- def show_proxy_configuration_examples():
87
- """显示代理配置示例"""
88
- print("\n=== 代理配置示例 ===")
89
-
90
- # 不同类型的代理配置示例
91
- examples = [
92
- {
93
- "name": "带认证HTTP代理",
94
- "url": "http://username:password@proxy.example.com:8080",
95
- "selenium_config": "SELENIUM_PROXY = 'http://proxy.example.com:8080'",
96
- "playwright_config": "PLAYWRIGHT_PROXY = {'server': 'http://proxy.example.com:8080', 'username': 'username', 'password': 'password'}"
97
- },
98
- {
99
- "name": "带认证HTTPS代理",
100
- "url": "https://user:pass@secure-proxy.example.com:443",
101
- "selenium_config": "SELENIUM_PROXY = 'https://secure-proxy.example.com:443'",
102
- "playwright_config": "PLAYWRIGHT_PROXY = {'server': 'https://secure-proxy.example.com:443', 'username': 'user', 'password': 'pass'}"
103
- },
104
- {
105
- "name": "SOCKS5代理",
106
- "url": "socks5://username:password@socks-proxy.example.com:1080",
107
- "selenium_config": "SELENIUM_PROXY = 'socks5://socks-proxy.example.com:1080'",
108
- "playwright_config": "PLAYWRIGHT_PROXY = {'server': 'socks5://socks-proxy.example.com:1080', 'username': 'username', 'password': 'password'}"
109
- },
110
- {
111
- "name": "不带认证代理",
112
- "url": "http://proxy.example.com:8080",
113
- "selenium_config": "SELENIUM_PROXY = 'http://proxy.example.com:8080'",
114
- "playwright_config": "PLAYWRIGHT_PROXY = 'http://proxy.example.com:8080'"
115
- }
116
- ]
117
-
118
- for i, example in enumerate(examples, 1):
119
- print(f"\n示例 {i}: {example['name']}")
120
- print(f" 代理URL: {example['url']}")
121
- proxy = AuthenticatedProxy(example['url'])
122
- print(f" 清洁URL: {proxy.clean_url}")
123
- print(f" 用户名: {proxy.username or '无'}")
124
- print(f" 密码: {proxy.password or '无'}")
125
- print(f" Selenium配置: {example['selenium_config']}")
126
- print(f" Playwright配置: {example['playwright_config']}")
127
-
128
-
129
- def main():
130
- """主函数"""
131
- print("开始测试动态下载器的代理配置逻辑...\n")
132
-
133
- # 测试各个下载器的配置逻辑
134
- test_selenium_proxy_logic()
135
- test_playwright_proxy_logic()
136
- show_proxy_configuration_examples()
137
-
138
- print("\n所有配置逻辑测试完成!")
139
- print("\n总结:")
140
- print("1. Selenium和Playwright都支持代理配置")
141
- print("2. 带认证的代理需要特殊处理")
142
- print("3. Playwright对带认证代理的支持更直接")
143
- print("4. Selenium需要通过扩展或其他方式处理认证")
144
-
145
-
146
- if __name__ == "__main__":
147
- main()