crawlo 1.2.5__py3-none-any.whl → 1.2.7__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.

Potentially problematic release.


This version of crawlo might be problematic. Click here for more details.

Files changed (209) hide show
  1. crawlo/__init__.py +61 -61
  2. crawlo/__version__.py +1 -1
  3. crawlo/cleaners/__init__.py +60 -60
  4. crawlo/cleaners/data_formatter.py +225 -225
  5. crawlo/cleaners/encoding_converter.py +125 -125
  6. crawlo/cleaners/text_cleaner.py +232 -232
  7. crawlo/cli.py +75 -88
  8. crawlo/commands/__init__.py +14 -14
  9. crawlo/commands/check.py +594 -594
  10. crawlo/commands/genspider.py +151 -151
  11. crawlo/commands/help.py +138 -144
  12. crawlo/commands/list.py +155 -155
  13. crawlo/commands/run.py +323 -323
  14. crawlo/commands/startproject.py +436 -436
  15. crawlo/commands/stats.py +187 -187
  16. crawlo/commands/utils.py +186 -186
  17. crawlo/config.py +312 -312
  18. crawlo/config_validator.py +251 -251
  19. crawlo/core/__init__.py +2 -2
  20. crawlo/core/engine.py +365 -354
  21. crawlo/core/processor.py +40 -40
  22. crawlo/core/scheduler.py +251 -143
  23. crawlo/crawler.py +1099 -1110
  24. crawlo/data/__init__.py +5 -5
  25. crawlo/data/user_agents.py +107 -107
  26. crawlo/downloader/__init__.py +266 -266
  27. crawlo/downloader/aiohttp_downloader.py +228 -221
  28. crawlo/downloader/cffi_downloader.py +256 -256
  29. crawlo/downloader/httpx_downloader.py +259 -259
  30. crawlo/downloader/hybrid_downloader.py +212 -212
  31. crawlo/downloader/playwright_downloader.py +402 -402
  32. crawlo/downloader/selenium_downloader.py +472 -472
  33. crawlo/event.py +11 -11
  34. crawlo/exceptions.py +81 -81
  35. crawlo/extension/__init__.py +39 -38
  36. crawlo/extension/health_check.py +141 -141
  37. crawlo/extension/log_interval.py +57 -57
  38. crawlo/extension/log_stats.py +81 -81
  39. crawlo/extension/logging_extension.py +43 -43
  40. crawlo/extension/memory_monitor.py +104 -104
  41. crawlo/extension/performance_profiler.py +133 -133
  42. crawlo/extension/request_recorder.py +107 -107
  43. crawlo/filters/__init__.py +154 -154
  44. crawlo/filters/aioredis_filter.py +234 -281
  45. crawlo/filters/memory_filter.py +269 -269
  46. crawlo/items/__init__.py +23 -23
  47. crawlo/items/base.py +21 -21
  48. crawlo/items/fields.py +52 -52
  49. crawlo/items/items.py +104 -104
  50. crawlo/middleware/__init__.py +21 -21
  51. crawlo/middleware/default_header.py +131 -131
  52. crawlo/middleware/download_delay.py +104 -104
  53. crawlo/middleware/middleware_manager.py +136 -135
  54. crawlo/middleware/offsite.py +114 -114
  55. crawlo/middleware/proxy.py +367 -367
  56. crawlo/middleware/request_ignore.py +86 -86
  57. crawlo/middleware/response_code.py +163 -163
  58. crawlo/middleware/response_filter.py +136 -136
  59. crawlo/middleware/retry.py +124 -124
  60. crawlo/mode_manager.py +211 -211
  61. crawlo/network/__init__.py +21 -21
  62. crawlo/network/request.py +338 -338
  63. crawlo/network/response.py +359 -359
  64. crawlo/pipelines/__init__.py +21 -21
  65. crawlo/pipelines/bloom_dedup_pipeline.py +156 -156
  66. crawlo/pipelines/console_pipeline.py +39 -39
  67. crawlo/pipelines/csv_pipeline.py +316 -316
  68. crawlo/pipelines/database_dedup_pipeline.py +222 -222
  69. crawlo/pipelines/json_pipeline.py +218 -218
  70. crawlo/pipelines/memory_dedup_pipeline.py +115 -115
  71. crawlo/pipelines/mongo_pipeline.py +131 -131
  72. crawlo/pipelines/mysql_pipeline.py +317 -317
  73. crawlo/pipelines/pipeline_manager.py +62 -61
  74. crawlo/pipelines/redis_dedup_pipeline.py +166 -165
  75. crawlo/project.py +314 -279
  76. crawlo/queue/pqueue.py +37 -37
  77. crawlo/queue/queue_manager.py +377 -337
  78. crawlo/queue/redis_priority_queue.py +306 -299
  79. crawlo/settings/__init__.py +7 -7
  80. crawlo/settings/default_settings.py +219 -217
  81. crawlo/settings/setting_manager.py +122 -122
  82. crawlo/spider/__init__.py +639 -639
  83. crawlo/stats_collector.py +59 -59
  84. crawlo/subscriber.py +129 -129
  85. crawlo/task_manager.py +30 -30
  86. crawlo/templates/crawlo.cfg.tmpl +10 -10
  87. crawlo/templates/project/__init__.py.tmpl +3 -3
  88. crawlo/templates/project/items.py.tmpl +17 -17
  89. crawlo/templates/project/middlewares.py.tmpl +118 -118
  90. crawlo/templates/project/pipelines.py.tmpl +96 -96
  91. crawlo/templates/project/settings.py.tmpl +288 -324
  92. crawlo/templates/project/settings_distributed.py.tmpl +157 -154
  93. crawlo/templates/project/settings_gentle.py.tmpl +101 -128
  94. crawlo/templates/project/settings_high_performance.py.tmpl +135 -150
  95. crawlo/templates/project/settings_simple.py.tmpl +99 -103
  96. crawlo/templates/project/spiders/__init__.py.tmpl +5 -5
  97. crawlo/templates/run.py.tmpl +45 -47
  98. crawlo/templates/spider/spider.py.tmpl +143 -143
  99. crawlo/tools/__init__.py +182 -182
  100. crawlo/tools/anti_crawler.py +268 -268
  101. crawlo/tools/authenticated_proxy.py +240 -240
  102. crawlo/tools/data_validator.py +180 -180
  103. crawlo/tools/date_tools.py +35 -35
  104. crawlo/tools/distributed_coordinator.py +386 -386
  105. crawlo/tools/retry_mechanism.py +220 -220
  106. crawlo/tools/scenario_adapter.py +262 -262
  107. crawlo/utils/__init__.py +35 -35
  108. crawlo/utils/batch_processor.py +259 -259
  109. crawlo/utils/controlled_spider_mixin.py +439 -439
  110. crawlo/utils/date_tools.py +290 -290
  111. crawlo/utils/db_helper.py +343 -343
  112. crawlo/utils/enhanced_error_handler.py +356 -356
  113. crawlo/utils/env_config.py +143 -106
  114. crawlo/utils/error_handler.py +123 -123
  115. crawlo/utils/func_tools.py +82 -82
  116. crawlo/utils/large_scale_config.py +286 -286
  117. crawlo/utils/large_scale_helper.py +344 -344
  118. crawlo/utils/log.py +128 -128
  119. crawlo/utils/performance_monitor.py +285 -285
  120. crawlo/utils/queue_helper.py +175 -175
  121. crawlo/utils/redis_connection_pool.py +351 -334
  122. crawlo/utils/redis_key_validator.py +198 -198
  123. crawlo/utils/request.py +267 -267
  124. crawlo/utils/request_serializer.py +218 -218
  125. crawlo/utils/spider_loader.py +61 -61
  126. crawlo/utils/system.py +11 -11
  127. crawlo/utils/tools.py +4 -4
  128. crawlo/utils/url.py +39 -39
  129. {crawlo-1.2.5.dist-info → crawlo-1.2.7.dist-info}/METADATA +764 -764
  130. crawlo-1.2.7.dist-info/RECORD +209 -0
  131. examples/__init__.py +7 -7
  132. tests/DOUBLE_CRAWLO_PREFIX_FIX_REPORT.md +81 -81
  133. tests/__init__.py +7 -7
  134. tests/advanced_tools_example.py +275 -275
  135. tests/authenticated_proxy_example.py +236 -236
  136. tests/cleaners_example.py +160 -160
  137. tests/config_validation_demo.py +102 -102
  138. tests/controlled_spider_example.py +205 -205
  139. tests/date_tools_example.py +180 -180
  140. tests/dynamic_loading_example.py +523 -523
  141. tests/dynamic_loading_test.py +104 -104
  142. tests/env_config_example.py +133 -133
  143. tests/error_handling_example.py +171 -171
  144. tests/redis_key_validation_demo.py +130 -130
  145. tests/response_improvements_example.py +144 -144
  146. tests/test_advanced_tools.py +148 -148
  147. tests/test_all_redis_key_configs.py +145 -145
  148. tests/test_authenticated_proxy.py +141 -141
  149. tests/test_cleaners.py +54 -54
  150. tests/test_comprehensive.py +146 -146
  151. tests/test_config_consistency.py +81 -0
  152. tests/test_config_validator.py +193 -193
  153. tests/test_crawlo_proxy_integration.py +172 -172
  154. tests/test_date_tools.py +123 -123
  155. tests/test_default_header_middleware.py +158 -158
  156. tests/test_double_crawlo_fix.py +207 -207
  157. tests/test_double_crawlo_fix_simple.py +124 -124
  158. tests/test_download_delay_middleware.py +221 -221
  159. tests/test_downloader_proxy_compatibility.py +268 -268
  160. tests/test_dynamic_downloaders_proxy.py +124 -124
  161. tests/test_dynamic_proxy.py +92 -92
  162. tests/test_dynamic_proxy_config.py +146 -146
  163. tests/test_dynamic_proxy_real.py +109 -109
  164. tests/test_edge_cases.py +303 -303
  165. tests/test_enhanced_error_handler.py +270 -270
  166. tests/test_env_config.py +121 -121
  167. tests/test_error_handler_compatibility.py +112 -112
  168. tests/test_final_validation.py +153 -153
  169. tests/test_framework_env_usage.py +103 -103
  170. tests/test_integration.py +356 -356
  171. tests/test_item_dedup_redis_key.py +122 -122
  172. tests/test_mode_consistency.py +52 -0
  173. tests/test_offsite_middleware.py +221 -221
  174. tests/test_parsel.py +29 -29
  175. tests/test_performance.py +327 -327
  176. tests/test_proxy_api.py +264 -264
  177. tests/test_proxy_health_check.py +32 -32
  178. tests/test_proxy_middleware.py +121 -121
  179. tests/test_proxy_middleware_enhanced.py +216 -216
  180. tests/test_proxy_middleware_integration.py +136 -136
  181. tests/test_proxy_providers.py +56 -56
  182. tests/test_proxy_stats.py +19 -19
  183. tests/test_proxy_strategies.py +59 -59
  184. tests/test_queue_manager_double_crawlo.py +173 -173
  185. tests/test_queue_manager_redis_key.py +176 -176
  186. tests/test_real_scenario_proxy.py +195 -195
  187. tests/test_redis_config.py +28 -28
  188. tests/test_redis_connection_pool.py +294 -294
  189. tests/test_redis_key_naming.py +181 -181
  190. tests/test_redis_key_validator.py +123 -123
  191. tests/test_redis_queue.py +224 -224
  192. tests/test_request_ignore_middleware.py +182 -182
  193. tests/test_request_serialization.py +70 -70
  194. tests/test_response_code_middleware.py +349 -349
  195. tests/test_response_filter_middleware.py +427 -427
  196. tests/test_response_improvements.py +152 -152
  197. tests/test_retry_middleware.py +241 -241
  198. tests/test_scheduler.py +252 -241
  199. tests/test_scheduler_config_update.py +134 -0
  200. tests/test_simple_response.py +61 -61
  201. tests/test_telecom_spider_redis_key.py +205 -205
  202. tests/test_template_content.py +87 -87
  203. tests/test_template_redis_key.py +134 -134
  204. tests/test_tools.py +153 -153
  205. tests/tools_example.py +257 -257
  206. crawlo-1.2.5.dist-info/RECORD +0 -206
  207. {crawlo-1.2.5.dist-info → crawlo-1.2.7.dist-info}/WHEEL +0 -0
  208. {crawlo-1.2.5.dist-info → crawlo-1.2.7.dist-info}/entry_points.txt +0 -0
  209. {crawlo-1.2.5.dist-info → crawlo-1.2.7.dist-info}/top_level.txt +0 -0
crawlo/commands/check.py CHANGED
@@ -1,595 +1,595 @@
1
- #!/usr/bin/python
2
- # -*- coding: UTF-8 -*-
3
- """
4
- # @Time : 2025-08-31 22:35
5
- # @Author : crawl-coder
6
- # @Desc : 命令行入口:crawlo check,检查所有爬虫定义是否合规。
7
- """
8
- import sys
9
- import ast
10
- import astor
11
- import re
12
- import time
13
- from pathlib import Path
14
- import configparser
15
- from importlib import import_module
16
-
17
- from rich.console import Console
18
- from rich.panel import Panel
19
- from rich.table import Table
20
- from rich.text import Text
21
- from rich import box
22
-
23
- from watchdog.observers import Observer
24
- from watchdog.events import FileSystemEventHandler
25
-
26
- from crawlo.crawler import CrawlerProcess
27
- from crawlo.utils.log import get_logger
28
-
29
-
30
- logger = get_logger(__name__)
31
- console = Console()
32
-
33
-
34
- def get_project_root():
35
- """
36
- 从当前目录向上查找 crawlo.cfg,确定项目根目录
37
- """
38
- current = Path.cwd()
39
- for _ in range(10):
40
- cfg = current / "crawlo.cfg"
41
- if cfg.exists():
42
- return current
43
- if current == current.parent:
44
- break
45
- current = current.parent
46
- return None
47
-
48
-
49
- def auto_fix_spider_file(spider_cls, file_path: Path):
50
- """自动修复 spider 文件中的常见问题"""
51
- try:
52
- with open(file_path, "r", encoding="utf-8") as f:
53
- source = f.read()
54
-
55
- fixed = False
56
- tree = ast.parse(source)
57
-
58
- # 查找 Spider 类定义
59
- class_node = None
60
- for node in ast.walk(tree):
61
- if isinstance(node, ast.ClassDef) and node.name == spider_cls.__name__:
62
- class_node = node
63
- break
64
-
65
- if not class_node:
66
- return False, "在文件中找不到类定义。"
67
-
68
- # 1. 修复 name 为空或缺失
69
- name_assign = None
70
- for node in class_node.body:
71
- if isinstance(node, ast.Assign):
72
- for target in node.targets:
73
- if isinstance(target, ast.Name) and target.id == "name":
74
- name_assign = node
75
- break
76
-
77
- if not name_assign or (
78
- isinstance(name_assign.value, ast.Constant) and not name_assign.value.value
79
- ):
80
- # 生成默认 name:类名转 snake_case
81
- default_name = re.sub(r'(?<!^)(?=[A-Z])', '_', spider_cls.__name__).lower().replace("_spider", "")
82
- new_assign = ast.Assign(
83
- targets=[ast.Name(id="name", ctx=ast.Store())],
84
- value=ast.Constant(value=default_name)
85
- )
86
- if name_assign:
87
- index = class_node.body.index(name_assign)
88
- class_node.body[index] = new_assign
89
- else:
90
- class_node.body.insert(0, new_assign)
91
- fixed = True
92
-
93
- # 2. 修复 start_urls 是字符串
94
- start_urls_assign = None
95
- for node in class_node.body:
96
- if isinstance(node, ast.Assign):
97
- for target in node.targets:
98
- if isinstance(target, ast.Name) and target.id == "start_urls":
99
- start_urls_assign = node
100
- break
101
-
102
- if start_urls_assign and isinstance(start_urls_assign.value, ast.Constant) and isinstance(start_urls_assign.value.value, str):
103
- new_value = ast.List(elts=[ast.Constant(value=start_urls_assign.value.value)], ctx=ast.Load())
104
- start_urls_assign.value = new_value
105
- fixed = True
106
-
107
- # 3. 修复缺少 parse 方法
108
- has_parse = any(
109
- isinstance(node, ast.FunctionDef) and node.name == "parse"
110
- for node in class_node.body
111
- )
112
- if not has_parse:
113
- parse_method = ast.FunctionDef(
114
- name="parse",
115
- args=ast.arguments(
116
- posonlyargs=[],
117
- args=[ast.arg(arg="self"), ast.arg(arg="response")],
118
- kwonlyargs=[],
119
- kw_defaults=[],
120
- defaults=[],
121
- vararg=None,
122
- kwarg=None
123
- ),
124
- body=[
125
- ast.Expr(value=ast.Constant(value="默认 parse 方法,返回 item 或继续请求")),
126
- ast.Pass()
127
- ],
128
- decorator_list=[],
129
- returns=None
130
- )
131
- class_node.body.append(parse_method)
132
- fixed = True
133
-
134
- # 4. 修复 allowed_domains 是字符串
135
- allowed_domains_assign = None
136
- for node in class_node.body:
137
- if isinstance(node, ast.Assign):
138
- for target in node.targets:
139
- if isinstance(target, ast.Name) and target.id == "allowed_domains":
140
- allowed_domains_assign = node
141
- break
142
-
143
- if allowed_domains_assign and isinstance(allowed_domains_assign.value, ast.Constant) and isinstance(allowed_domains_assign.value.value, str):
144
- new_value = ast.List(elts=[ast.Constant(value=allowed_domains_assign.value.value)], ctx=ast.Load())
145
- allowed_domains_assign.value = new_value
146
- fixed = True
147
-
148
- # 5. 修复缺失 custom_settings
149
- has_custom_settings = any(
150
- isinstance(node, ast.Assign) and
151
- any(isinstance(t, ast.Name) and t.id == "custom_settings" for t in node.targets)
152
- for node in class_node.body
153
- )
154
- if not has_custom_settings:
155
- new_assign = ast.Assign(
156
- targets=[ast.Name(id="custom_settings", ctx=ast.Store())],
157
- value=ast.Dict(keys=[], values=[])
158
- )
159
- # 插入在 name 之后
160
- insert_index = 1
161
- for i, node in enumerate(class_node.body):
162
- if isinstance(node, ast.Assign) and any(
163
- isinstance(t, ast.Name) and t.id == "name" for t in node.targets
164
- ):
165
- insert_index = i + 1
166
- break
167
- class_node.body.insert(insert_index, new_assign)
168
- fixed = True
169
-
170
- # 6. 修复缺失 start_requests 方法
171
- has_start_requests = any(
172
- isinstance(node, ast.FunctionDef) and node.name == "start_requests"
173
- for node in class_node.body
174
- )
175
- if not has_start_requests:
176
- start_requests_method = ast.FunctionDef(
177
- name="start_requests",
178
- args=ast.arguments(
179
- posonlyargs=[],
180
- args=[ast.arg(arg="self")],
181
- kwonlyargs=[],
182
- kw_defaults=[],
183
- defaults=[],
184
- vararg=None,
185
- kwarg=None
186
- ),
187
- body=[
188
- ast.Expr(value=ast.Constant(value="默认 start_requests,从 start_urls 生成请求")),
189
- ast.For(
190
- target=ast.Name(id="url", ctx=ast.Store()),
191
- iter=ast.Attribute(value=ast.Name(id="self", ctx=ast.Load()), attr="start_urls", ctx=ast.Load()),
192
- body=[
193
- ast.Expr(
194
- value=ast.Call(
195
- func=ast.Attribute(value=ast.Name(id="self", ctx=ast.Load()), attr="make_request", ctx=ast.Load()),
196
- args=[ast.Name(id="url", ctx=ast.Load())],
197
- keywords=[]
198
- )
199
- )
200
- ],
201
- orelse=[]
202
- )
203
- ],
204
- decorator_list=[],
205
- returns=None
206
- )
207
- # 插入在 custom_settings 或 name 之后,parse 之前
208
- insert_index = 2
209
- for i, node in enumerate(class_node.body):
210
- if isinstance(node, ast.FunctionDef) and node.name == "parse":
211
- insert_index = i
212
- break
213
- elif isinstance(node, ast.Assign) and any(
214
- isinstance(t, ast.Name) and t.id in ("name", "custom_settings") for t in node.targets
215
- ):
216
- insert_index = i + 1
217
- class_node.body.insert(insert_index, start_requests_method)
218
- fixed = True
219
-
220
- if fixed:
221
- fixed_source = astor.to_source(tree)
222
- with open(file_path, "w", encoding="utf-8") as f:
223
- f.write(fixed_source)
224
- return True, "文件自动修复成功。"
225
- else:
226
- return False, "未找到可修复的问题。"
227
-
228
- except Exception as e:
229
- return False, f"自动修复失败: {e}"
230
-
231
-
232
- class SpiderChangeHandler(FileSystemEventHandler):
233
- def __init__(self, project_root, spider_modules, show_fix=False, console=None):
234
- self.project_root = project_root
235
- self.spider_modules = spider_modules
236
- self.show_fix = show_fix
237
- self.console = console or Console()
238
-
239
- def on_modified(self, event):
240
- if event.is_directory:
241
- return
242
- if event.src_path.endswith(".py") and "spiders" in event.src_path:
243
- file_path = Path(event.src_path)
244
- spider_name = file_path.stem
245
- self.console.print(f"\n:eyes: [bold blue]检测到变更[/bold blue] [cyan]{file_path}[/cyan]")
246
- self.check_and_fix_spider(spider_name)
247
-
248
- def check_and_fix_spider(self, spider_name):
249
- try:
250
- process = CrawlerProcess(spider_modules=self.spider_modules)
251
- if spider_name not in process.get_spider_names():
252
- self.console.print(f"[yellow]⚠️ {spider_name} 不是已注册的爬虫。[/yellow]")
253
- return
254
-
255
- cls = process.get_spider_class(spider_name)
256
- issues = []
257
-
258
- # 简化检查
259
- if not getattr(cls, "name", None):
260
- issues.append("缺少或为空的 'name' 属性")
261
- if not callable(getattr(cls, "start_requests", None)):
262
- issues.append("缺少 'start_requests' 方法")
263
- if hasattr(cls, "start_urls") and isinstance(cls.start_urls, str):
264
- issues.append("'start_urls' 是字符串")
265
- if hasattr(cls, "allowed_domains") and isinstance(cls.allowed_domains, str):
266
- issues.append("'allowed_domains' 是字符串")
267
-
268
- try:
269
- spider = cls.create_instance(None)
270
- if not callable(getattr(spider, "parse", None)):
271
- issues.append("缺少 'parse' 方法")
272
- except Exception:
273
- issues.append("实例化失败")
274
-
275
- if issues:
276
- self.console.print(f"[red]❌ {spider_name} 存在问题:[/red]")
277
- for issue in issues:
278
- self.console.print(f" • {issue}")
279
-
280
- if self.show_fix:
281
- file_path = Path(cls.__file__)
282
- fixed, msg = auto_fix_spider_file(cls, file_path)
283
- if fixed:
284
- self.console.print(f"[green]✅ 自动修复: {msg}[/green]")
285
- else:
286
- self.console.print(f"[yellow]⚠️ 无法修复: {msg}[/yellow]")
287
- else:
288
- self.console.print(f"[green]✅ {spider_name} 合规。[/green]")
289
-
290
- except Exception as e:
291
- self.console.print(f"[red]❌ 检查 {spider_name} 时出错: {e}[/red]")
292
-
293
-
294
- def watch_spiders(project_root: Path, project_package: str, show_fix: bool):
295
- """监听 spiders 目录变化并自动检查"""
296
- spider_path = project_root / project_package / "spiders"
297
- if not spider_path.exists():
298
- console.print(f"[bold red]❌ Spider 目录未找到:[/bold red] {spider_path}")
299
- return
300
-
301
- spider_modules = [f"{project_package}.spiders"]
302
- event_handler = SpiderChangeHandler(project_root, spider_modules, show_fix, console)
303
- observer = Observer()
304
- observer.schedule(event_handler, str(spider_path), recursive=False)
305
-
306
- console.print(Panel(
307
- f":eyes: [bold blue]监听[/bold blue] [cyan]{spider_path}[/cyan] 中的变更\n"
308
- "编辑任何爬虫文件以触发自动检查...",
309
- title="🚀 已启动监听模式",
310
- border_style="blue"
311
- ))
312
-
313
- observer.start()
314
- try:
315
- while True:
316
- time.sleep(1)
317
- except KeyboardInterrupt:
318
- console.print("\n[bold red]🛑 监听模式已停止。[/bold red]")
319
- observer.stop()
320
- observer.join()
321
-
322
-
323
- def main(args):
324
- """
325
- 主函数:检查所有爬虫定义的合规性
326
- 用法:
327
- crawlo check
328
- crawlo check --fix
329
- crawlo check --ci
330
- crawlo check --json
331
- crawlo check --watch
332
- """
333
- show_fix = "--fix" in args or "-f" in args
334
- show_ci = "--ci" in args
335
- show_json = "--json" in args
336
- show_watch = "--watch" in args
337
-
338
- valid_args = {"--fix", "-f", "--ci", "--json", "--watch"}
339
- if any(arg not in valid_args for arg in args):
340
- console.print("[bold red]❌ 错误:[/bold red] 用法: [blue]crawlo check[/blue] [--fix] [--ci] [--json] [--watch]")
341
- return 1
342
-
343
- try:
344
- # 1. 查找项目根目录
345
- project_root = get_project_root()
346
- if not project_root:
347
- msg = ":cross_mark: [bold red]找不到 'crawlo.cfg'[/bold red]\n💡 请在项目目录中运行此命令。"
348
- if show_json:
349
- console.print_json(data={"success": False, "error": "未找到项目根目录"})
350
- return 1
351
- elif show_ci:
352
- console.print("❌ 未找到项目根目录。缺少 crawlo.cfg。")
353
- return 1
354
- else:
355
- console.print(Panel(
356
- Text.from_markup(msg),
357
- title="❌ 非Crawlo项目",
358
- border_style="red",
359
- padding=(1, 2)
360
- ))
361
- return 1
362
-
363
- project_root_str = str(project_root)
364
- if project_root_str not in sys.path:
365
- sys.path.insert(0, project_root_str)
366
-
367
- # 2. 读取 crawlo.cfg
368
- cfg_file = project_root / "crawlo.cfg"
369
- if not cfg_file.exists():
370
- msg = f"配置文件未找到: {cfg_file}"
371
- if show_json:
372
- console.print_json(data={"success": False, "error": msg})
373
- return 1
374
- elif show_ci:
375
- console.print(f"❌ {msg}")
376
- return 1
377
- else:
378
- console.print(Panel(msg, title="❌ 缺少配置文件", border_style="red"))
379
- return 1
380
-
381
- config = configparser.ConfigParser()
382
- config.read(cfg_file, encoding="utf-8")
383
-
384
- if not config.has_section("settings") or not config.has_option("settings", "default"):
385
- msg = "crawlo.cfg 中缺少 [settings] 部分或 'default' 选项"
386
- if show_json:
387
- console.print_json(data={"success": False, "error": msg})
388
- return 1
389
- elif show_ci:
390
- console.print(f"❌ {msg}")
391
- return 1
392
- else:
393
- console.print(Panel(msg, title="❌ 无效配置", border_style="red"))
394
- return 1
395
-
396
- settings_module = config.get("settings", "default")
397
- project_package = settings_module.split(".")[0]
398
-
399
- # 3. 确保项目包可导入
400
- try:
401
- import_module(project_package)
402
- except ImportError as e:
403
- msg = f"导入项目包 '{project_package}' 失败: {e}"
404
- if show_json:
405
- console.print_json(data={"success": False, "error": msg})
406
- return 1
407
- elif show_ci:
408
- console.print(f"❌ {msg}")
409
- return 1
410
- else:
411
- console.print(Panel(msg, title="❌ 导入错误", border_style="red"))
412
- return 1
413
-
414
- # 4. 加载爬虫
415
- spider_modules = [f"{project_package}.spiders"]
416
- process = CrawlerProcess(spider_modules=spider_modules)
417
- spider_names = process.get_spider_names()
418
-
419
- if not spider_names:
420
- msg = "未找到爬虫。"
421
- if show_json:
422
- console.print_json(data={"success": True, "warning": msg})
423
- return 0
424
- elif show_ci:
425
- console.print("📭 未找到爬虫。")
426
- return 0
427
- else:
428
- console.print(Panel(
429
- Text.from_markup(
430
- ":envelope_with_arrow: [bold]未找到爬虫[/bold]\n\n"
431
- "[bold]💡 确保:[/bold]\n"
432
- " • 爬虫定义于 '[cyan]spiders[/cyan]' 模块\n"
433
- " • 具有 [green]`name`[/green] 属性\n"
434
- " • 模块已正确导入"
435
- ),
436
- title="📭 未找到爬虫",
437
- border_style="yellow",
438
- padding=(1, 2)
439
- ))
440
- return 0
441
-
442
- # 5. 如果启用 watch 模式,启动监听
443
- if show_watch:
444
- console.print("[bold blue]:eyes: 启动监听模式...[/bold blue]")
445
- watch_spiders(project_root, project_package, show_fix)
446
- return 0 # watch 是长期运行,不返回
447
-
448
- # 6. 开始检查(非 watch 模式)
449
- if not show_ci and not show_json:
450
- console.print(f":mag: [bold]正在检查 {len(spider_names)} 个爬虫...[/bold]\n")
451
-
452
- issues_found = False
453
- results = []
454
-
455
- for name in sorted(spider_names):
456
- cls = process.get_spider_class(name)
457
- issues = []
458
-
459
- # 检查 name 属性
460
- if not getattr(cls, "name", None):
461
- issues.append("缺少或为空的 'name' 属性")
462
- elif not isinstance(cls.name, str):
463
- issues.append("'name' 不是字符串")
464
-
465
- # 检查 start_requests 是否可调用
466
- if not callable(getattr(cls, "start_requests", None)):
467
- issues.append("缺少或不可调用的 'start_requests' 方法")
468
-
469
- # 检查 start_urls 类型(不应是字符串)
470
- if hasattr(cls, "start_urls") and isinstance(cls.start_urls, str):
471
- issues.append("'start_urls' 是字符串;应为列表或元组")
472
-
473
- # 检查 allowed_domains 类型
474
- if hasattr(cls, "allowed_domains") and isinstance(cls.allowed_domains, str):
475
- issues.append("'allowed_domains' 是字符串;应为列表或元组")
476
-
477
- # 实例化并检查 parse 方法
478
- try:
479
- spider = cls.create_instance(None)
480
- if not callable(getattr(spider, "parse", None)):
481
- issues.append("未定义 'parse' 方法(推荐)")
482
- except Exception as e:
483
- issues.append(f"实例化爬虫失败: {e}")
484
-
485
- # 自动修复(如果启用)
486
- if issues and show_fix:
487
- try:
488
- file_path = Path(cls.__file__)
489
- fixed, msg = auto_fix_spider_file(cls, file_path)
490
- if fixed:
491
- if not show_ci and not show_json:
492
- console.print(f"[green]🔧 已自动修复 {name} → {msg}[/green]")
493
- issues = [] # 认为已修复
494
- else:
495
- if not show_ci and not show_json:
496
- console.print(f"[yellow]⚠️ 无法自动修复 {name}: {msg}[/yellow]")
497
- except Exception as e:
498
- if not show_ci and not show_json:
499
- console.print(f"[yellow]⚠️ 找不到 {name} 的源文件: {e}[/yellow]")
500
-
501
- results.append({
502
- "name": name,
503
- "class": cls.__name__,
504
- "file": getattr(cls, "__file__", "unknown"),
505
- "issues": issues
506
- })
507
-
508
- if issues:
509
- issues_found = True
510
-
511
- # 7. 生成报告数据
512
- report = {
513
- "success": not issues_found,
514
- "total_spiders": len(spider_names),
515
- "issues": [
516
- {"name": r["name"], "class": r["class"], "file": r["file"], "problems": r["issues"]}
517
- for r in results if r["issues"]
518
- ]
519
- }
520
-
521
- # 8. 输出(根据模式)
522
- if show_json:
523
- console.print_json(data=report)
524
- return 1 if issues_found else 0
525
-
526
- if show_ci:
527
- if issues_found:
528
- console.print("❌ 合规性检查失败。")
529
- for r in results:
530
- if r["issues"]:
531
- console.print(f" • {r['name']}: {', '.join(r['issues'])}")
532
- else:
533
- console.print("✅ 所有爬虫合规。")
534
- return 1 if issues_found else 0
535
-
536
- # 9. 默认 rich 输出
537
- table = Table(
538
- title="🔍 爬虫合规性检查结果",
539
- box=box.ROUNDED,
540
- show_header=True,
541
- header_style="bold magenta",
542
- title_style="bold green"
543
- )
544
- table.add_column("状态", style="bold", width=4)
545
- table.add_column("名称", style="cyan")
546
- table.add_column("类名", style="green")
547
- table.add_column("问题", style="yellow", overflow="fold")
548
-
549
- for res in results:
550
- if res["issues"]:
551
- status = "[red]❌[/red]"
552
- issues_text = "\n".join(f"• {issue}" for issue in res["issues"])
553
- else:
554
- status = "[green]✅[/green]"
555
- issues_text = "—"
556
-
557
- table.add_row(status, res["name"], res["class"], issues_text)
558
-
559
- console.print(table)
560
- console.print()
561
-
562
- if issues_found:
563
- console.print(Panel(
564
- ":warning: [bold red]一些爬虫存在问题。[/bold red]\n请在运行前修复这些问题。",
565
- title="⚠️ 合规性检查失败",
566
- border_style="red",
567
- padding=(1, 2)
568
- ))
569
- return 1
570
- else:
571
- console.print(Panel(
572
- ":tada: [bold green]所有爬虫都合规且定义良好![/bold green]\n准备开始爬取! 🕷️🚀",
573
- title="🎉 检查通过",
574
- border_style="green",
575
- padding=(1, 2)
576
- ))
577
- return 0
578
-
579
- except Exception as e:
580
- logger.exception("执行 'crawlo check' 时发生异常")
581
- if show_json:
582
- console.print_json(data={"success": False, "error": str(e)})
583
- elif show_ci:
584
- console.print(f"❌ 意外错误: {e}")
585
- else:
586
- console.print(f"[bold red]❌ 检查过程中发生意外错误:[/bold red] {e}")
587
- return 1
588
-
589
-
590
- if __name__ == "__main__":
591
- """
592
- 支持直接运行:
593
- python -m crawlo.commands.check
594
- """
1
+ #!/usr/bin/python
2
+ # -*- coding: UTF-8 -*-
3
+ """
4
+ # @Time : 2025-08-31 22:35
5
+ # @Author : crawl-coder
6
+ # @Desc : 命令行入口:crawlo check,检查所有爬虫定义是否合规。
7
+ """
8
+ import sys
9
+ import ast
10
+ import astor
11
+ import re
12
+ import time
13
+ from pathlib import Path
14
+ import configparser
15
+ from importlib import import_module
16
+
17
+ from rich.console import Console
18
+ from rich.panel import Panel
19
+ from rich.table import Table
20
+ from rich.text import Text
21
+ from rich import box
22
+
23
+ from watchdog.observers import Observer
24
+ from watchdog.events import FileSystemEventHandler
25
+
26
+ from crawlo.crawler import CrawlerProcess
27
+ from crawlo.utils.log import get_logger
28
+
29
+
30
+ logger = get_logger(__name__)
31
+ console = Console()
32
+
33
+
34
+ def get_project_root():
35
+ """
36
+ 从当前目录向上查找 crawlo.cfg,确定项目根目录
37
+ """
38
+ current = Path.cwd()
39
+ for _ in range(10):
40
+ cfg = current / "crawlo.cfg"
41
+ if cfg.exists():
42
+ return current
43
+ if current == current.parent:
44
+ break
45
+ current = current.parent
46
+ return None
47
+
48
+
49
+ def auto_fix_spider_file(spider_cls, file_path: Path):
50
+ """自动修复 spider 文件中的常见问题"""
51
+ try:
52
+ with open(file_path, "r", encoding="utf-8") as f:
53
+ source = f.read()
54
+
55
+ fixed = False
56
+ tree = ast.parse(source)
57
+
58
+ # 查找 Spider 类定义
59
+ class_node = None
60
+ for node in ast.walk(tree):
61
+ if isinstance(node, ast.ClassDef) and node.name == spider_cls.__name__:
62
+ class_node = node
63
+ break
64
+
65
+ if not class_node:
66
+ return False, "在文件中找不到类定义。"
67
+
68
+ # 1. 修复 name 为空或缺失
69
+ name_assign = None
70
+ for node in class_node.body:
71
+ if isinstance(node, ast.Assign):
72
+ for target in node.targets:
73
+ if isinstance(target, ast.Name) and target.id == "name":
74
+ name_assign = node
75
+ break
76
+
77
+ if not name_assign or (
78
+ isinstance(name_assign.value, ast.Constant) and not name_assign.value.value
79
+ ):
80
+ # 生成默认 name:类名转 snake_case
81
+ default_name = re.sub(r'(?<!^)(?=[A-Z])', '_', spider_cls.__name__).lower().replace("_spider", "")
82
+ new_assign = ast.Assign(
83
+ targets=[ast.Name(id="name", ctx=ast.Store())],
84
+ value=ast.Constant(value=default_name)
85
+ )
86
+ if name_assign:
87
+ index = class_node.body.index(name_assign)
88
+ class_node.body[index] = new_assign
89
+ else:
90
+ class_node.body.insert(0, new_assign)
91
+ fixed = True
92
+
93
+ # 2. 修复 start_urls 是字符串
94
+ start_urls_assign = None
95
+ for node in class_node.body:
96
+ if isinstance(node, ast.Assign):
97
+ for target in node.targets:
98
+ if isinstance(target, ast.Name) and target.id == "start_urls":
99
+ start_urls_assign = node
100
+ break
101
+
102
+ if start_urls_assign and isinstance(start_urls_assign.value, ast.Constant) and isinstance(start_urls_assign.value.value, str):
103
+ new_value = ast.List(elts=[ast.Constant(value=start_urls_assign.value.value)], ctx=ast.Load())
104
+ start_urls_assign.value = new_value
105
+ fixed = True
106
+
107
+ # 3. 修复缺少 parse 方法
108
+ has_parse = any(
109
+ isinstance(node, ast.FunctionDef) and node.name == "parse"
110
+ for node in class_node.body
111
+ )
112
+ if not has_parse:
113
+ parse_method = ast.FunctionDef(
114
+ name="parse",
115
+ args=ast.arguments(
116
+ posonlyargs=[],
117
+ args=[ast.arg(arg="self"), ast.arg(arg="response")],
118
+ kwonlyargs=[],
119
+ kw_defaults=[],
120
+ defaults=[],
121
+ vararg=None,
122
+ kwarg=None
123
+ ),
124
+ body=[
125
+ ast.Expr(value=ast.Constant(value="默认 parse 方法,返回 item 或继续请求")),
126
+ ast.Pass()
127
+ ],
128
+ decorator_list=[],
129
+ returns=None
130
+ )
131
+ class_node.body.append(parse_method)
132
+ fixed = True
133
+
134
+ # 4. 修复 allowed_domains 是字符串
135
+ allowed_domains_assign = None
136
+ for node in class_node.body:
137
+ if isinstance(node, ast.Assign):
138
+ for target in node.targets:
139
+ if isinstance(target, ast.Name) and target.id == "allowed_domains":
140
+ allowed_domains_assign = node
141
+ break
142
+
143
+ if allowed_domains_assign and isinstance(allowed_domains_assign.value, ast.Constant) and isinstance(allowed_domains_assign.value.value, str):
144
+ new_value = ast.List(elts=[ast.Constant(value=allowed_domains_assign.value.value)], ctx=ast.Load())
145
+ allowed_domains_assign.value = new_value
146
+ fixed = True
147
+
148
+ # 5. 修复缺失 custom_settings
149
+ has_custom_settings = any(
150
+ isinstance(node, ast.Assign) and
151
+ any(isinstance(t, ast.Name) and t.id == "custom_settings" for t in node.targets)
152
+ for node in class_node.body
153
+ )
154
+ if not has_custom_settings:
155
+ new_assign = ast.Assign(
156
+ targets=[ast.Name(id="custom_settings", ctx=ast.Store())],
157
+ value=ast.Dict(keys=[], values=[])
158
+ )
159
+ # 插入在 name 之后
160
+ insert_index = 1
161
+ for i, node in enumerate(class_node.body):
162
+ if isinstance(node, ast.Assign) and any(
163
+ isinstance(t, ast.Name) and t.id == "name" for t in node.targets
164
+ ):
165
+ insert_index = i + 1
166
+ break
167
+ class_node.body.insert(insert_index, new_assign)
168
+ fixed = True
169
+
170
+ # 6. 修复缺失 start_requests 方法
171
+ has_start_requests = any(
172
+ isinstance(node, ast.FunctionDef) and node.name == "start_requests"
173
+ for node in class_node.body
174
+ )
175
+ if not has_start_requests:
176
+ start_requests_method = ast.FunctionDef(
177
+ name="start_requests",
178
+ args=ast.arguments(
179
+ posonlyargs=[],
180
+ args=[ast.arg(arg="self")],
181
+ kwonlyargs=[],
182
+ kw_defaults=[],
183
+ defaults=[],
184
+ vararg=None,
185
+ kwarg=None
186
+ ),
187
+ body=[
188
+ ast.Expr(value=ast.Constant(value="默认 start_requests,从 start_urls 生成请求")),
189
+ ast.For(
190
+ target=ast.Name(id="url", ctx=ast.Store()),
191
+ iter=ast.Attribute(value=ast.Name(id="self", ctx=ast.Load()), attr="start_urls", ctx=ast.Load()),
192
+ body=[
193
+ ast.Expr(
194
+ value=ast.Call(
195
+ func=ast.Attribute(value=ast.Name(id="self", ctx=ast.Load()), attr="make_request", ctx=ast.Load()),
196
+ args=[ast.Name(id="url", ctx=ast.Load())],
197
+ keywords=[]
198
+ )
199
+ )
200
+ ],
201
+ orelse=[]
202
+ )
203
+ ],
204
+ decorator_list=[],
205
+ returns=None
206
+ )
207
+ # 插入在 custom_settings 或 name 之后,parse 之前
208
+ insert_index = 2
209
+ for i, node in enumerate(class_node.body):
210
+ if isinstance(node, ast.FunctionDef) and node.name == "parse":
211
+ insert_index = i
212
+ break
213
+ elif isinstance(node, ast.Assign) and any(
214
+ isinstance(t, ast.Name) and t.id in ("name", "custom_settings") for t in node.targets
215
+ ):
216
+ insert_index = i + 1
217
+ class_node.body.insert(insert_index, start_requests_method)
218
+ fixed = True
219
+
220
+ if fixed:
221
+ fixed_source = astor.to_source(tree)
222
+ with open(file_path, "w", encoding="utf-8") as f:
223
+ f.write(fixed_source)
224
+ return True, "文件自动修复成功。"
225
+ else:
226
+ return False, "未找到可修复的问题。"
227
+
228
+ except Exception as e:
229
+ return False, f"自动修复失败: {e}"
230
+
231
+
232
+ class SpiderChangeHandler(FileSystemEventHandler):
233
+ def __init__(self, project_root, spider_modules, show_fix=False, console=None):
234
+ self.project_root = project_root
235
+ self.spider_modules = spider_modules
236
+ self.show_fix = show_fix
237
+ self.console = console or Console()
238
+
239
+ def on_modified(self, event):
240
+ if event.is_directory:
241
+ return
242
+ if event.src_path.endswith(".py") and "spiders" in event.src_path:
243
+ file_path = Path(event.src_path)
244
+ spider_name = file_path.stem
245
+ self.console.print(f"\n:eyes: [bold blue]检测到变更[/bold blue] [cyan]{file_path}[/cyan]")
246
+ self.check_and_fix_spider(spider_name)
247
+
248
+ def check_and_fix_spider(self, spider_name):
249
+ try:
250
+ process = CrawlerProcess(spider_modules=self.spider_modules)
251
+ if spider_name not in process.get_spider_names():
252
+ self.console.print(f"[yellow]⚠️ {spider_name} 不是已注册的爬虫。[/yellow]")
253
+ return
254
+
255
+ cls = process.get_spider_class(spider_name)
256
+ issues = []
257
+
258
+ # 简化检查
259
+ if not getattr(cls, "name", None):
260
+ issues.append("缺少或为空的 'name' 属性")
261
+ if not callable(getattr(cls, "start_requests", None)):
262
+ issues.append("缺少 'start_requests' 方法")
263
+ if hasattr(cls, "start_urls") and isinstance(cls.start_urls, str):
264
+ issues.append("'start_urls' 是字符串")
265
+ if hasattr(cls, "allowed_domains") and isinstance(cls.allowed_domains, str):
266
+ issues.append("'allowed_domains' 是字符串")
267
+
268
+ try:
269
+ spider = cls.create_instance(None)
270
+ if not callable(getattr(spider, "parse", None)):
271
+ issues.append("缺少 'parse' 方法")
272
+ except Exception:
273
+ issues.append("实例化失败")
274
+
275
+ if issues:
276
+ self.console.print(f"[red]❌ {spider_name} 存在问题:[/red]")
277
+ for issue in issues:
278
+ self.console.print(f" • {issue}")
279
+
280
+ if self.show_fix:
281
+ file_path = Path(cls.__file__)
282
+ fixed, msg = auto_fix_spider_file(cls, file_path)
283
+ if fixed:
284
+ self.console.print(f"[green]✅ 自动修复: {msg}[/green]")
285
+ else:
286
+ self.console.print(f"[yellow]⚠️ 无法修复: {msg}[/yellow]")
287
+ else:
288
+ self.console.print(f"[green]✅ {spider_name} 合规。[/green]")
289
+
290
+ except Exception as e:
291
+ self.console.print(f"[red]❌ 检查 {spider_name} 时出错: {e}[/red]")
292
+
293
+
294
+ def watch_spiders(project_root: Path, project_package: str, show_fix: bool):
295
+ """监听 spiders 目录变化并自动检查"""
296
+ spider_path = project_root / project_package / "spiders"
297
+ if not spider_path.exists():
298
+ console.print(f"[bold red]❌ Spider 目录未找到:[/bold red] {spider_path}")
299
+ return
300
+
301
+ spider_modules = [f"{project_package}.spiders"]
302
+ event_handler = SpiderChangeHandler(project_root, spider_modules, show_fix, console)
303
+ observer = Observer()
304
+ observer.schedule(event_handler, str(spider_path), recursive=False)
305
+
306
+ console.print(Panel(
307
+ f":eyes: [bold blue]监听[/bold blue] [cyan]{spider_path}[/cyan] 中的变更\n"
308
+ "编辑任何爬虫文件以触发自动检查...",
309
+ title="🚀 已启动监听模式",
310
+ border_style="blue"
311
+ ))
312
+
313
+ observer.start()
314
+ try:
315
+ while True:
316
+ time.sleep(1)
317
+ except KeyboardInterrupt:
318
+ console.print("\n[bold red]🛑 监听模式已停止。[/bold red]")
319
+ observer.stop()
320
+ observer.join()
321
+
322
+
323
+ def main(args):
324
+ """
325
+ 主函数:检查所有爬虫定义的合规性
326
+ 用法:
327
+ crawlo check
328
+ crawlo check --fix
329
+ crawlo check --ci
330
+ crawlo check --json
331
+ crawlo check --watch
332
+ """
333
+ show_fix = "--fix" in args or "-f" in args
334
+ show_ci = "--ci" in args
335
+ show_json = "--json" in args
336
+ show_watch = "--watch" in args
337
+
338
+ valid_args = {"--fix", "-f", "--ci", "--json", "--watch"}
339
+ if any(arg not in valid_args for arg in args):
340
+ console.print("[bold red]❌ 错误:[/bold red] 用法: [blue]crawlo check[/blue] [--fix] [--ci] [--json] [--watch]")
341
+ return 1
342
+
343
+ try:
344
+ # 1. 查找项目根目录
345
+ project_root = get_project_root()
346
+ if not project_root:
347
+ msg = ":cross_mark: [bold red]找不到 'crawlo.cfg'[/bold red]\n💡 请在项目目录中运行此命令。"
348
+ if show_json:
349
+ console.print_json(data={"success": False, "error": "未找到项目根目录"})
350
+ return 1
351
+ elif show_ci:
352
+ console.print("❌ 未找到项目根目录。缺少 crawlo.cfg。")
353
+ return 1
354
+ else:
355
+ console.print(Panel(
356
+ Text.from_markup(msg),
357
+ title="❌ 非Crawlo项目",
358
+ border_style="red",
359
+ padding=(1, 2)
360
+ ))
361
+ return 1
362
+
363
+ project_root_str = str(project_root)
364
+ if project_root_str not in sys.path:
365
+ sys.path.insert(0, project_root_str)
366
+
367
+ # 2. 读取 crawlo.cfg
368
+ cfg_file = project_root / "crawlo.cfg"
369
+ if not cfg_file.exists():
370
+ msg = f"配置文件未找到: {cfg_file}"
371
+ if show_json:
372
+ console.print_json(data={"success": False, "error": msg})
373
+ return 1
374
+ elif show_ci:
375
+ console.print(f"❌ {msg}")
376
+ return 1
377
+ else:
378
+ console.print(Panel(msg, title="❌ 缺少配置文件", border_style="red"))
379
+ return 1
380
+
381
+ config = configparser.ConfigParser()
382
+ config.read(cfg_file, encoding="utf-8")
383
+
384
+ if not config.has_section("settings") or not config.has_option("settings", "default"):
385
+ msg = "crawlo.cfg 中缺少 [settings] 部分或 'default' 选项"
386
+ if show_json:
387
+ console.print_json(data={"success": False, "error": msg})
388
+ return 1
389
+ elif show_ci:
390
+ console.print(f"❌ {msg}")
391
+ return 1
392
+ else:
393
+ console.print(Panel(msg, title="❌ 无效配置", border_style="red"))
394
+ return 1
395
+
396
+ settings_module = config.get("settings", "default")
397
+ project_package = settings_module.split(".")[0]
398
+
399
+ # 3. 确保项目包可导入
400
+ try:
401
+ import_module(project_package)
402
+ except ImportError as e:
403
+ msg = f"导入项目包 '{project_package}' 失败: {e}"
404
+ if show_json:
405
+ console.print_json(data={"success": False, "error": msg})
406
+ return 1
407
+ elif show_ci:
408
+ console.print(f"❌ {msg}")
409
+ return 1
410
+ else:
411
+ console.print(Panel(msg, title="❌ 导入错误", border_style="red"))
412
+ return 1
413
+
414
+ # 4. 加载爬虫
415
+ spider_modules = [f"{project_package}.spiders"]
416
+ process = CrawlerProcess(spider_modules=spider_modules)
417
+ spider_names = process.get_spider_names()
418
+
419
+ if not spider_names:
420
+ msg = "未找到爬虫。"
421
+ if show_json:
422
+ console.print_json(data={"success": True, "warning": msg})
423
+ return 0
424
+ elif show_ci:
425
+ console.print("📭 未找到爬虫。")
426
+ return 0
427
+ else:
428
+ console.print(Panel(
429
+ Text.from_markup(
430
+ ":envelope_with_arrow: [bold]未找到爬虫[/bold]\n\n"
431
+ "[bold]💡 确保:[/bold]\n"
432
+ " • 爬虫定义于 '[cyan]spiders[/cyan]' 模块\n"
433
+ " • 具有 [green]`name`[/green] 属性\n"
434
+ " • 模块已正确导入"
435
+ ),
436
+ title="📭 未找到爬虫",
437
+ border_style="yellow",
438
+ padding=(1, 2)
439
+ ))
440
+ return 0
441
+
442
+ # 5. 如果启用 watch 模式,启动监听
443
+ if show_watch:
444
+ console.print("[bold blue]:eyes: 启动监听模式...[/bold blue]")
445
+ watch_spiders(project_root, project_package, show_fix)
446
+ return 0 # watch 是长期运行,不返回
447
+
448
+ # 6. 开始检查(非 watch 模式)
449
+ if not show_ci and not show_json:
450
+ console.print(f":mag: [bold]正在检查 {len(spider_names)} 个爬虫...[/bold]\n")
451
+
452
+ issues_found = False
453
+ results = []
454
+
455
+ for name in sorted(spider_names):
456
+ cls = process.get_spider_class(name)
457
+ issues = []
458
+
459
+ # 检查 name 属性
460
+ if not getattr(cls, "name", None):
461
+ issues.append("缺少或为空的 'name' 属性")
462
+ elif not isinstance(cls.name, str):
463
+ issues.append("'name' 不是字符串")
464
+
465
+ # 检查 start_requests 是否可调用
466
+ if not callable(getattr(cls, "start_requests", None)):
467
+ issues.append("缺少或不可调用的 'start_requests' 方法")
468
+
469
+ # 检查 start_urls 类型(不应是字符串)
470
+ if hasattr(cls, "start_urls") and isinstance(cls.start_urls, str):
471
+ issues.append("'start_urls' 是字符串;应为列表或元组")
472
+
473
+ # 检查 allowed_domains 类型
474
+ if hasattr(cls, "allowed_domains") and isinstance(cls.allowed_domains, str):
475
+ issues.append("'allowed_domains' 是字符串;应为列表或元组")
476
+
477
+ # 实例化并检查 parse 方法
478
+ try:
479
+ spider = cls.create_instance(None)
480
+ if not callable(getattr(spider, "parse", None)):
481
+ issues.append("未定义 'parse' 方法(推荐)")
482
+ except Exception as e:
483
+ issues.append(f"实例化爬虫失败: {e}")
484
+
485
+ # 自动修复(如果启用)
486
+ if issues and show_fix:
487
+ try:
488
+ file_path = Path(cls.__file__)
489
+ fixed, msg = auto_fix_spider_file(cls, file_path)
490
+ if fixed:
491
+ if not show_ci and not show_json:
492
+ console.print(f"[green]🔧 已自动修复 {name} → {msg}[/green]")
493
+ issues = [] # 认为已修复
494
+ else:
495
+ if not show_ci and not show_json:
496
+ console.print(f"[yellow]⚠️ 无法自动修复 {name}: {msg}[/yellow]")
497
+ except Exception as e:
498
+ if not show_ci and not show_json:
499
+ console.print(f"[yellow]⚠️ 找不到 {name} 的源文件: {e}[/yellow]")
500
+
501
+ results.append({
502
+ "name": name,
503
+ "class": cls.__name__,
504
+ "file": getattr(cls, "__file__", "unknown"),
505
+ "issues": issues
506
+ })
507
+
508
+ if issues:
509
+ issues_found = True
510
+
511
+ # 7. 生成报告数据
512
+ report = {
513
+ "success": not issues_found,
514
+ "total_spiders": len(spider_names),
515
+ "issues": [
516
+ {"name": r["name"], "class": r["class"], "file": r["file"], "problems": r["issues"]}
517
+ for r in results if r["issues"]
518
+ ]
519
+ }
520
+
521
+ # 8. 输出(根据模式)
522
+ if show_json:
523
+ console.print_json(data=report)
524
+ return 1 if issues_found else 0
525
+
526
+ if show_ci:
527
+ if issues_found:
528
+ console.print("❌ 合规性检查失败。")
529
+ for r in results:
530
+ if r["issues"]:
531
+ console.print(f" • {r['name']}: {', '.join(r['issues'])}")
532
+ else:
533
+ console.print("✅ 所有爬虫合规。")
534
+ return 1 if issues_found else 0
535
+
536
+ # 9. 默认 rich 输出
537
+ table = Table(
538
+ title="🔍 爬虫合规性检查结果",
539
+ box=box.ROUNDED,
540
+ show_header=True,
541
+ header_style="bold magenta",
542
+ title_style="bold green"
543
+ )
544
+ table.add_column("状态", style="bold", width=4)
545
+ table.add_column("名称", style="cyan")
546
+ table.add_column("类名", style="green")
547
+ table.add_column("问题", style="yellow", overflow="fold")
548
+
549
+ for res in results:
550
+ if res["issues"]:
551
+ status = "[red]❌[/red]"
552
+ issues_text = "\n".join(f"• {issue}" for issue in res["issues"])
553
+ else:
554
+ status = "[green]✅[/green]"
555
+ issues_text = "—"
556
+
557
+ table.add_row(status, res["name"], res["class"], issues_text)
558
+
559
+ console.print(table)
560
+ console.print()
561
+
562
+ if issues_found:
563
+ console.print(Panel(
564
+ ":warning: [bold red]一些爬虫存在问题。[/bold red]\n请在运行前修复这些问题。",
565
+ title="⚠️ 合规性检查失败",
566
+ border_style="red",
567
+ padding=(1, 2)
568
+ ))
569
+ return 1
570
+ else:
571
+ console.print(Panel(
572
+ ":tada: [bold green]所有爬虫都合规且定义良好![/bold green]\n准备开始爬取! 🕷️🚀",
573
+ title="🎉 检查通过",
574
+ border_style="green",
575
+ padding=(1, 2)
576
+ ))
577
+ return 0
578
+
579
+ except Exception as e:
580
+ logger.exception("执行 'crawlo check' 时发生异常")
581
+ if show_json:
582
+ console.print_json(data={"success": False, "error": str(e)})
583
+ elif show_ci:
584
+ console.print(f"❌ 意外错误: {e}")
585
+ else:
586
+ console.print(f"[bold red]❌ 检查过程中发生意外错误:[/bold red] {e}")
587
+ return 1
588
+
589
+
590
+ if __name__ == "__main__":
591
+ """
592
+ 支持直接运行:
593
+ python -m crawlo.commands.check
594
+ """
595
595
  sys.exit(main(sys.argv[1:]))