alpha-dc-common 0.1.23__py2.py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (38) hide show
  1. alpha_dc_common-0.1.23.dist-info/METADATA +6 -0
  2. alpha_dc_common-0.1.23.dist-info/RECORD +38 -0
  3. alpha_dc_common-0.1.23.dist-info/WHEEL +5 -0
  4. dc_common/__init__.py +15 -0
  5. dc_common/core/logger.py +48 -0
  6. dc_common/data_sources/__init__.py +16 -0
  7. dc_common/data_sources/base_source.py +176 -0
  8. dc_common/data_sources/daily_quote_source.py +195 -0
  9. dc_common/schemas/__init__.py +46 -0
  10. dc_common/schemas/a_stock.py +24 -0
  11. dc_common/schemas/base.py +45 -0
  12. dc_common/schemas/hk_announcement.py +31 -0
  13. dc_common/schemas/hk_company_profile.py +51 -0
  14. dc_common/schemas/hk_stock.py +25 -0
  15. dc_common/schemas/hs_industry.py +26 -0
  16. dc_common/schemas/hs_industry_company.py +27 -0
  17. dc_common/schemas/index_basic.py +26 -0
  18. dc_common/schemas/index_company.py +19 -0
  19. dc_common/schemas/index_daily.py +30 -0
  20. dc_common/schemas/margin_account.py +29 -0
  21. dc_common/schemas/margin_analysis.py +40 -0
  22. dc_common/schemas/margin_detail.py +29 -0
  23. dc_common/schemas/results.py +50 -0
  24. dc_common/schemas/sw_index_daily.py +34 -0
  25. dc_common/schemas/sw_industry.py +54 -0
  26. dc_common/schemas/sw_industry_company.py +28 -0
  27. dc_common/utils/auth_token.py +54 -0
  28. dc_common/utils/baidu_utils.py +122 -0
  29. dc_common/utils/cache_utils.py +141 -0
  30. dc_common/utils/config_reader.py +79 -0
  31. dc_common/utils/eastmoney_utils.py +118 -0
  32. dc_common/utils/jin10_utils.py +153 -0
  33. dc_common/utils/logging_config.py +104 -0
  34. dc_common/utils/markdown_convert.py +121 -0
  35. dc_common/utils/money_format.py +21 -0
  36. dc_common/utils/proxy_utils.py +253 -0
  37. dc_common/utils/schema_extract.py +179 -0
  38. dc_common/utils/xueqiu_utils.py +253 -0
@@ -0,0 +1,6 @@
1
+ Metadata-Version: 2.4
2
+ Name: alpha-dc-common
3
+ Version: 0.1.23
4
+ Summary: Shared schemas and utils for the datacenter project.
5
+ Author-email: Hang GuangLiang <hanguangliang@alphaaidig.com>
6
+ Requires-Dist: pydantic>=2.0
@@ -0,0 +1,38 @@
1
+ dc_common/__init__.py,sha256=OtW3SuOD7_04PgwCPor5AFdWE049sWd_uchQcyrwkJk,284
2
+ dc_common/core/logger.py,sha256=pNiOjkL0TIidKZTww0X8wDZiCyoBSeWcl2h645yhf9k,1347
3
+ dc_common/data_sources/__init__.py,sha256=j6nxSaoK8YtLJr7FU2qr-oXJzw3bt3wWXf2RXHvW9D4,363
4
+ dc_common/data_sources/base_source.py,sha256=zgCHLF0fAfdO3xOR1WC2R7NK5ih1cd6BhhycUg2a_lM,4477
5
+ dc_common/data_sources/daily_quote_source.py,sha256=Lq6SbvIIq_2mvG1T5EHery4G9nw5WkrDGWfV1__H5FI,6754
6
+ dc_common/schemas/__init__.py,sha256=XEI89qPJnW0VUDjthP-hoMn49Rj30CL8yI4JnL9Eirc,1206
7
+ dc_common/schemas/a_stock.py,sha256=FV3utksDO1FdHaCe9xPjr5CgnAbUKgLCPTPaKnuqMRU,920
8
+ dc_common/schemas/base.py,sha256=34ydqpF1ascnqQQ3hNcOesXYLfTXwihwVOHmZ2cu5dM,1076
9
+ dc_common/schemas/hk_announcement.py,sha256=VG2TInENO7R8aRabf92DIp3qdXQWW0oMOWq5fV5u5z8,1560
10
+ dc_common/schemas/hk_company_profile.py,sha256=lLjsRpScBDIp-TXTAmjEBr4D0sq6BUjfPoHskLTgTQ4,2628
11
+ dc_common/schemas/hk_stock.py,sha256=aroehzwxB28qhOSkjG7PN2ZRYJshj7j7oT_XkOkT2vY,961
12
+ dc_common/schemas/hs_industry.py,sha256=Q3JKsUEG5jifRRttTKcv6Zk0isHXVsLyT8tkVP0eUEw,800
13
+ dc_common/schemas/hs_industry_company.py,sha256=J2nPIw0KzqA0GvHIlk2wDRtiWHW8PLWFqV4MxJv6Ytw,1127
14
+ dc_common/schemas/index_basic.py,sha256=2ggaAwW6o9Xnf3igiC7u-JT5IPS2gwu5ijs6dNjnR8A,1093
15
+ dc_common/schemas/index_company.py,sha256=LkA6Zxr4SyilPw9KQ8dCygspA-qr8xmlswtJ-sO6-uM,516
16
+ dc_common/schemas/index_daily.py,sha256=Uy_-3a2XIE-huepP-tFMgsO4gpEx3Qgsb4mxMqiiChs,1311
17
+ dc_common/schemas/margin_account.py,sha256=9IeDYKT7j7BlRkYUzIbSpj7cemdtiU6afbj1ktDz8rE,1567
18
+ dc_common/schemas/margin_analysis.py,sha256=uh16WPAtUKdIEucedJNEpp5ksxhGeiljFM9gENYYqKA,2029
19
+ dc_common/schemas/margin_detail.py,sha256=yN5LTB3-hdq2Us9YP8KpvQNjmky0Kg1txX8LBSZz8Hg,1315
20
+ dc_common/schemas/results.py,sha256=kyJWlJxm8UB8LCREID1oXww465FCxK1md64bOmeXXX4,1392
21
+ dc_common/schemas/sw_index_daily.py,sha256=erSsYQWDafFVOdhym9d48C6KMlsPhtKobGP53YQElcM,1597
22
+ dc_common/schemas/sw_industry.py,sha256=aM1wwV9xdcAJBztzTGOlzhJcIfAj4YtHLh5Q0U_8AP4,2432
23
+ dc_common/schemas/sw_industry_company.py,sha256=lUkSQDL6TqFuIl5DmbUQC9dzIucn08DlolDyY0VPGJg,791
24
+ dc_common/utils/auth_token.py,sha256=jDW_HLhcM_UXb0VAWaiJEMDfuUzWjDHDTRng89G5c3c,1789
25
+ dc_common/utils/baidu_utils.py,sha256=x20NcxkBY3W0LOShoEG2FvwoJ8Wne9FQKrvkyR-_XzA,4199
26
+ dc_common/utils/cache_utils.py,sha256=CFDvckeZcbe0fq7iYzx0FfLmYJR9XdWwAvWcu8QChNE,5054
27
+ dc_common/utils/config_reader.py,sha256=fPBs5XcAHyNlZvWbldpsOBaVLs1s_ABwhM5ZcsA8CSE,2449
28
+ dc_common/utils/eastmoney_utils.py,sha256=un5uGcsVTZrTVOiVvhEQwZk5usGbT4Y6gtnc6ErVZiM,3527
29
+ dc_common/utils/jin10_utils.py,sha256=x6azWD3oEIDdf1yGKh_FdDWcSjbhSXRPypYGKoTZ2BQ,5477
30
+ dc_common/utils/logging_config.py,sha256=lIlpfWZbQzH6ZkGV7_ujIli7KRDGz-T4e6SadfeMLE8,3805
31
+ dc_common/utils/markdown_convert.py,sha256=jsfoOR8aEefCQmkmd3UqCzVVYntqU2zBbIXfRwbWXGA,4430
32
+ dc_common/utils/money_format.py,sha256=pYL03L1ErLkVEzoeTBhMholaI6CCZ2orEHGONk2xN3s,539
33
+ dc_common/utils/proxy_utils.py,sha256=uEg3bI_6nBJkn2DYn_wJLMYbrYmNGSZkyG7173IP3oM,8183
34
+ dc_common/utils/schema_extract.py,sha256=U8OIVdX31fyI8kL-p-7wpEb9qIgj-gohwr37HccKsbc,6673
35
+ dc_common/utils/xueqiu_utils.py,sha256=4kaHn_uGBtRAkUpbDgV5D8bJQVbSxUvLPSAPgXERRUo,9055
36
+ alpha_dc_common-0.1.23.dist-info/METADATA,sha256=Sj8IyB8PVphSFwFOXXi2ROY1wZ1KdUpGjUGPHIeA52Y,212
37
+ alpha_dc_common-0.1.23.dist-info/WHEEL,sha256=aha0VrrYvgDJ3Xxl3db_g_MDIW-ZexDdrc_m-Hk8YY4,105
38
+ alpha_dc_common-0.1.23.dist-info/RECORD,,
@@ -0,0 +1,5 @@
1
+ Wheel-Version: 1.0
2
+ Generator: hatchling 1.28.0
3
+ Root-Is-Purelib: true
4
+ Tag: py2-none-any
5
+ Tag: py3-none-any
dc_common/__init__.py ADDED
@@ -0,0 +1,15 @@
1
+ """
2
+ DataCenter Common - 公共库
3
+
4
+ 提供:
5
+ - 数据源(Tushare、本地缓存等)
6
+ - 工具函数
7
+ - 核心组件
8
+ """
9
+ from .data_sources import BaseDataSource, DataSourceStatus, DailyQuoteSource
10
+
11
+ __all__ = [
12
+ 'BaseDataSource',
13
+ 'DataSourceStatus',
14
+ 'DailyQuoteSource',
15
+ ]
@@ -0,0 +1,48 @@
1
+ import logging
2
+ import sys
3
+ import os
4
+ from pathlib import Path
5
+
6
+ def setup_logger(name: str = "datacenter") -> logging.Logger:
7
+ """设置并返回配置好的logger实例"""
8
+
9
+ # 区分开发和生产环境
10
+ if os.environ.get('ENVIRONMENT') == 'production':
11
+ # 生产环境使用标准日志目录
12
+ log_dir = Path("/var/log/datacenter")
13
+ log_dir.mkdir(exist_ok=True, mode=0o755)
14
+ else:
15
+ # 开发环境使用相对路径
16
+ log_dir = Path("logs")
17
+ log_dir.mkdir(exist_ok=True)
18
+
19
+ # 创建logger
20
+ logger = logging.getLogger(name)
21
+ logger.setLevel(logging.INFO)
22
+
23
+ # 清除现有的handlers
24
+ logger.handlers.clear()
25
+
26
+ # 创建formatter
27
+ formatter = logging.Formatter(
28
+ '%(asctime)s - %(name)s - %(levelname)s - %(message)s'
29
+ )
30
+
31
+ # 文件handler
32
+ file_handler = logging.FileHandler(log_dir / "app.log", encoding="utf-8")
33
+ file_handler.setLevel(logging.INFO)
34
+ file_handler.setFormatter(formatter)
35
+
36
+ # 控制台handler
37
+ console_handler = logging.StreamHandler(sys.stdout)
38
+ console_handler.setLevel(logging.INFO)
39
+ console_handler.setFormatter(formatter)
40
+
41
+ # 添加handlers
42
+ logger.addHandler(file_handler)
43
+ logger.addHandler(console_handler)
44
+
45
+ return logger
46
+
47
+ # 创建默认logger实例
48
+ logger = setup_logger()
@@ -0,0 +1,16 @@
1
+ """
2
+ 数据源模块 - 提供各类数据源的实现
3
+
4
+ 本模块提供统一的数据源接口,支持:
5
+ - Tushare 数据源
6
+ - 本地缓存管理
7
+ - 数据质量验证
8
+ """
9
+ from .base_source import BaseDataSource, DataSourceStatus
10
+ from .daily_quote_source import DailyQuoteSource
11
+
12
+ __all__ = [
13
+ 'BaseDataSource',
14
+ 'DataSourceStatus',
15
+ 'DailyQuoteSource'
16
+ ]
@@ -0,0 +1,176 @@
1
+ """
2
+ 数据源基类定义
3
+ """
4
+ from abc import ABC, abstractmethod
5
+ from datetime import datetime, time
6
+ from typing import Dict, Any, Optional
7
+ import pandas as pd
8
+ import logging
9
+ from pathlib import Path
10
+ from enum import Enum
11
+
12
+
13
+ class DataSourceStatus(Enum):
14
+ """数据源状态"""
15
+ PENDING = "pending" # 等待中
16
+ READY = "ready" # 数据就绪
17
+ UPDATING = "updating" # 更新中
18
+ FAILED = "failed" # 更新失败
19
+ NOT_NEEDED = "not_needed" # 该日不需要更新
20
+
21
+
22
+ class BaseDataSource(ABC):
23
+ """数据源基类
24
+
25
+ 每个数据源有独立的调度策略和更新时间
26
+ """
27
+
28
+ def __init__(self, data_dir: str, logger: Optional[logging.Logger] = None):
29
+ self.data_dir = Path(data_dir)
30
+ self.data_dir.mkdir(parents=True, exist_ok=True)
31
+ self.logger = logger or logging.getLogger(__name__)
32
+
33
+ @property
34
+ @abstractmethod
35
+ def source_name(self) -> str:
36
+ """数据源唯一标识"""
37
+ pass
38
+
39
+ @property
40
+ @abstractmethod
41
+ def display_name(self) -> str:
42
+ """显示名称"""
43
+ pass
44
+
45
+ @property
46
+ @abstractmethod
47
+ def update_time(self) -> str:
48
+ """
49
+ 期望更新时间
50
+ 例如: "17:00", "09:30"
51
+ """
52
+ pass
53
+
54
+ @property
55
+ def update_delay_days(self) -> int:
56
+ """
57
+ 更新延迟天数
58
+ 0 = 当天更新
59
+ 1 = 次日更新
60
+ """
61
+ return 0
62
+
63
+ @property
64
+ def priority(self) -> int:
65
+ """
66
+ 优先级(数字越小越优先)
67
+ 基础数据优先级高,衍生数据优先级低
68
+ """
69
+ return 100
70
+
71
+ @abstractmethod
72
+ def fetch_data(self, trade_date: str) -> pd.DataFrame:
73
+ """
74
+ 获取数据
75
+
76
+ Args:
77
+ trade_date: 交易日期 (YYYYMMDD)
78
+
79
+ Returns:
80
+ 数据 DataFrame
81
+ """
82
+ pass
83
+
84
+ @abstractmethod
85
+ def validate_data(self, df: pd.DataFrame) -> bool:
86
+ """
87
+ 验证数据质量
88
+
89
+ Returns:
90
+ True=数据有效, False=数据无效
91
+ """
92
+ pass
93
+
94
+ def is_ready(self, trade_date: str) -> DataSourceStatus:
95
+ """
96
+ 检查数据是否就绪
97
+
98
+ Args:
99
+ trade_date: 交易日期
100
+
101
+ Returns:
102
+ 数据源状态
103
+ """
104
+ # 1. 检查本地是否已有数据
105
+ if self._has_local_data(trade_date):
106
+ return DataSourceStatus.READY
107
+
108
+ # 2. 检查是否到了更新时间
109
+ if not self._is_update_time():
110
+ return DataSourceStatus.PENDING
111
+
112
+ return DataSourceStatus.READY
113
+
114
+ def update(self, trade_date: str) -> Dict[str, Any]:
115
+ """
116
+ 更新数据
117
+
118
+ Args:
119
+ trade_date: 交易日期
120
+
121
+ Returns:
122
+ 更新结果字典
123
+ """
124
+ result = {
125
+ 'source': self.source_name,
126
+ 'trade_date': trade_date,
127
+ 'status': 'unknown',
128
+ 'rows': 0,
129
+ 'message': ''
130
+ }
131
+
132
+ try:
133
+ self.logger.info(f"🔄 开始更新 {self.display_name}: {trade_date}")
134
+
135
+ # 1. 获取数据
136
+ df = self.fetch_data(trade_date)
137
+
138
+ # 2. 验证数据
139
+ if not self.validate_data(df):
140
+ result['status'] = 'failed'
141
+ result['message'] = '数据验证失败'
142
+ return result
143
+
144
+ # 3. 保存数据
145
+ self._save_data(df, trade_date)
146
+
147
+ result['status'] = 'success'
148
+ result['rows'] = len(df)
149
+ result['message'] = f'成功更新 {len(df)} 条数据'
150
+
151
+ self.logger.info(f"✅ {self.display_name} 更新完成: {len(df)} 条")
152
+
153
+ except Exception as e:
154
+ result['status'] = 'failed'
155
+ result['message'] = str(e)
156
+ self.logger.error(f"❌ {self.display_name} 更新失败: {e}")
157
+
158
+ return result
159
+
160
+ def _is_update_time(self) -> bool:
161
+ """检查是否到了更新时间"""
162
+ try:
163
+ hour, minute = map(int, self.update_time.split(':'))
164
+ target_time = time(hour, minute)
165
+ now = datetime.now().time()
166
+ return now >= target_time
167
+ except:
168
+ return True
169
+
170
+ def _has_local_data(self, trade_date: str) -> bool:
171
+ """检查本地是否已有数据(子类覆盖)"""
172
+ return False
173
+
174
+ def _save_data(self, df: pd.DataFrame, trade_date: str):
175
+ """保存数据(子类实现)"""
176
+ pass
@@ -0,0 +1,195 @@
1
+ """
2
+ 日线行情数据源
3
+ """
4
+ import os
5
+ import pandas as pd
6
+ from pathlib import Path
7
+ from typing import Optional
8
+ import logging
9
+
10
+ from .base_source import BaseDataSource, DataSourceStatus
11
+
12
+
13
+ class DailyQuoteSource(BaseDataSource):
14
+ """日线行情数据源
15
+
16
+ 从 Tushare 获取日线行情数据
17
+ 更新时间:每天 17:00
18
+ """
19
+
20
+ def __init__(
21
+ self,
22
+ data_dir: str,
23
+ tushare_token: Optional[str] = None,
24
+ logger: Optional[logging.Logger] = None
25
+ ):
26
+ super().__init__(data_dir, logger)
27
+
28
+ self.token = tushare_token or os.getenv('TUSHARE_TOKEN')
29
+ if not self.token:
30
+ raise ValueError("Tushare token 未配置,请设置 TUSHARE_TOKEN 环境变量")
31
+
32
+ # 缓存文件直接存储在 data_dir 下(不再添加 raw 子目录)
33
+ self.cache_file = self.data_dir / 'daily_quotes.parquet'
34
+ self.raw_cache_file = self.data_dir / 'daily_quotes_raw.parquet'
35
+
36
+ # 延迟初始化 Tushare(避免导入时立即连接)
37
+ self._pro = None
38
+
39
+ @property
40
+ def pro(self):
41
+ """延迟初始化 Tushare Pro API"""
42
+ if self._pro is None:
43
+ import tushare as ts
44
+ ts.set_token(self.token)
45
+ self._pro = ts.pro_api()
46
+ return self._pro
47
+
48
+ @property
49
+ def source_name(self) -> str:
50
+ return "daily_quotes"
51
+
52
+ @property
53
+ def display_name(self) -> str:
54
+ return "日线行情"
55
+
56
+ @property
57
+ def update_time(self) -> str:
58
+ return "17:00"
59
+
60
+ @property
61
+ def update_delay_days(self) -> int:
62
+ return 0
63
+
64
+ @property
65
+ def priority(self) -> int:
66
+ return 10
67
+
68
+ def fetch_data(self, trade_date: str) -> pd.DataFrame:
69
+ """从 Tushare 获取日线数据"""
70
+ self.logger.info(f"从 Tushare 获取 {trade_date} 的日线数据...")
71
+
72
+ df = self.pro.daily(
73
+ trade_date=trade_date
74
+ )
75
+
76
+ if df.empty:
77
+ self.logger.warning(f"{trade_date} 无数据(可能是非交易日)")
78
+ else:
79
+ self.logger.info(f"获取到 {len(df)} 条数据")
80
+
81
+ return df
82
+
83
+ def validate_data(self, df: pd.DataFrame) -> bool:
84
+ """验证数据质量"""
85
+ if df.empty:
86
+ return True # 空数据也是有效的(可能是非交易日)
87
+
88
+ required_fields = ['ts_code', 'trade_date', 'close']
89
+ return all(field in df.columns for field in required_fields)
90
+
91
+ def _has_local_data(self, trade_date: str) -> bool:
92
+ """检查本地是否已有数据"""
93
+ if not self.raw_cache_file.exists():
94
+ return False
95
+
96
+ try:
97
+ df = pd.read_parquet(self.raw_cache_file)
98
+ return trade_date in df['trade_date'].values
99
+ except Exception as e:
100
+ self.logger.warning(f"检查本地数据失败: {e}")
101
+ return False
102
+
103
+ def _save_data(self, df: pd.DataFrame, trade_date: str):
104
+ """保存数据(追加模式)"""
105
+ # 确保目录存在
106
+ self.raw_cache_file.parent.mkdir(parents=True, exist_ok=True)
107
+ self.cache_file.parent.mkdir(parents=True, exist_ok=True)
108
+
109
+ # 保存原始长格式数据
110
+ if self.raw_cache_file.exists():
111
+ existing_df = pd.read_parquet(self.raw_cache_file)
112
+
113
+ # 移除旧的同日期数据(如果有)
114
+ existing_df = existing_df[existing_df['trade_date'] != trade_date]
115
+
116
+ # 合并数据
117
+ combined_df = pd.concat([existing_df, df], ignore_index=True)
118
+
119
+ # 去重(基于 trade_date 和 ts_code,保留后出现的)
120
+ combined_df = combined_df.drop_duplicates(
121
+ subset=['trade_date', 'ts_code'],
122
+ keep='last'
123
+ )
124
+ else:
125
+ combined_df = df
126
+
127
+ combined_df.to_parquet(self.raw_cache_file, compression='snappy')
128
+ self.logger.info(f"原始数据已保存: {self.raw_cache_file}")
129
+
130
+ # 转换并保存宽表格式(用于因子计算)
131
+ self._save_wide_format(combined_df)
132
+
133
+ def _save_wide_format(self, df: pd.DataFrame):
134
+ """转换为宽表格式并保存(增量追加)"""
135
+ if df.empty:
136
+ return
137
+
138
+ # 排序
139
+ df = df.sort_values(['trade_date', 'ts_code'])
140
+
141
+ # 如果文件已存在,进行增量追加
142
+ if self.cache_file.exists():
143
+ try:
144
+ # 读取已有数据
145
+ existing_df = pd.read_parquet(self.cache_file)
146
+
147
+ # 移除与新数据重复的日期
148
+ existing_dates = existing_df['trade_date'].unique()
149
+ new_dates = df['trade_date'].unique()
150
+
151
+ # 只保留新数据中没有的旧日期
152
+ dates_to_keep = set(existing_dates) - set(new_dates)
153
+ if dates_to_keep:
154
+ existing_df = existing_df[existing_df['trade_date'].isin(dates_to_keep)]
155
+ else:
156
+ existing_df = pd.DataFrame()
157
+
158
+ # 合并数据
159
+ if not existing_df.empty:
160
+ combined_df = pd.concat([existing_df, df], ignore_index=True)
161
+ else:
162
+ combined_df = df
163
+
164
+ # 去重(基于 trade_date 和 ts_code,保留后出现的)
165
+ combined_df = combined_df.drop_duplicates(
166
+ subset=['trade_date', 'ts_code'],
167
+ keep='last'
168
+ )
169
+
170
+ # 排序
171
+ combined_df = combined_df.sort_values(['trade_date', 'ts_code'])
172
+
173
+ # 保存
174
+ combined_df.to_parquet(self.cache_file, compression='snappy', index=False)
175
+
176
+ old_len = len(existing_df) if dates_to_keep else 0
177
+ new_len = len(combined_df)
178
+ added_rows = new_len - old_len
179
+
180
+ self.logger.info(
181
+ f"宽表数据已增量更新: {self.cache_file} "
182
+ f"(原有: {old_len}, 新增: {added_rows}, 总计: {new_len})"
183
+ )
184
+ except Exception as e:
185
+ self.logger.warning(f"增量更新失败: {e},执行覆盖保存")
186
+ df.to_parquet(self.cache_file, compression='snappy', index=False)
187
+ self.logger.info(f"宽表数据已保存: {self.cache_file}")
188
+ else:
189
+ # 首次保存(先去重)
190
+ df = df.drop_duplicates(subset=['trade_date', 'ts_code'], keep='last')
191
+ df.to_parquet(self.cache_file, compression='snappy', index=False)
192
+ self.logger.info(f"宽表数据已保存: {self.cache_file}")
193
+
194
+ self.logger.info(f"数据范围: {df['trade_date'].min()} ~ {df['trade_date'].max()}")
195
+ self.logger.info(f"股票数量: {df['ts_code'].nunique()}")
@@ -0,0 +1,46 @@
1
+ """
2
+ 数据中心通用schemas模块
3
+ """
4
+ from .base import BaseResponse, PaginatedResponse, PaginationInfo
5
+ from .results import ResultType, BaseResult, TableResult, PageResult, DictResult, SingleResult
6
+ from .a_stock import AStock
7
+ from .hk_stock import HKStock
8
+ from .index_basic import IndexBasic
9
+ from .index_company import IndexCompany
10
+ from .margin_account import MarginAccount
11
+ from .margin_analysis import MarginAnalysis
12
+ from .margin_detail import MarginDetail
13
+
14
+ from .hs_industry import HSIndustry, HSIndustryCategory
15
+ from .hs_industry_company import HSIndustryCompany
16
+ from .sw_industry import SWIndustry
17
+ from .sw_industry_company import SWIndustryCompany
18
+ from .index_daily import IndexDaily
19
+ from .sw_index_daily import SWIndexDaily
20
+
21
+ __all__ = [
22
+ "BaseResponse",
23
+ "PaginatedResponse",
24
+ "PaginationInfo",
25
+ "ResultType",
26
+ "BaseResult",
27
+ "TableResult",
28
+ "PageResult",
29
+ "DictResult",
30
+ "SingleResult",
31
+ "AStock",
32
+ "HKStock",
33
+ "IndexBasic",
34
+ "IndexCompany",
35
+ "MarginAccount",
36
+ "MarginAnalysis",
37
+ "MarginDetail",
38
+
39
+ "HSIndustry",
40
+ "HSIndustryCategory",
41
+ "HSIndustryCompany",
42
+ "SWIndustry",
43
+ "SWIndustryCompany",
44
+ "IndexDaily",
45
+ "SWIndexDaily",
46
+ ]
@@ -0,0 +1,24 @@
1
+ """
2
+ A股相关的Pydantic模型
3
+ """
4
+ from pydantic import BaseModel, Field
5
+ from typing import Optional
6
+ from datetime import datetime
7
+
8
+ class AStockBase(BaseModel):
9
+ stock_code: str = Field(..., description="股票代码")
10
+ stock_name: str = Field(..., description="股票名称")
11
+ exchange: str = Field(..., description="交易所(SH/SZ/BJ)")
12
+ list_status: Optional[str] = Field(None, description="上市状态(L上市/D退市/P暂停上市)")
13
+ list_date: Optional[datetime] = Field(None, description="上市日期")
14
+ curr_type: Optional[str] = Field(None, description="交易货币(CNY)")
15
+ market: Optional[str] = Field(None, description="市场类型(主板/创业板/科创板/北交所)")
16
+ ric_code: Optional[str] = Field(None, description="路透代码")
17
+
18
+ class AStock(AStockBase):
19
+ id: int
20
+ created_at: datetime
21
+ updated_at: datetime
22
+
23
+ class Config:
24
+ from_attributes = True
@@ -0,0 +1,45 @@
1
+ """
2
+ 基础Pydantic模型
3
+ """
4
+ from typing import Any, Generic, TypeVar, List, Optional
5
+ from pydantic import BaseModel, Field
6
+
7
+ T = TypeVar("T")
8
+
9
+ class PaginatedResponse(BaseModel, Generic[T]):
10
+ """分页响应模型"""
11
+ total: int
12
+ page: int
13
+ size: int
14
+ items: List[T]
15
+
16
+ class BaseResponse(BaseModel):
17
+ """基础响应模型"""
18
+ status: str = "success"
19
+ message: Optional[str] = None
20
+ data: Optional[dict] = None
21
+
22
+ class PaginationInfo(BaseModel):
23
+ """分页信息响应模型"""
24
+ total: int
25
+ page: int
26
+ page_size: int
27
+ total_pages: int
28
+
29
+ class StandardListResponse(BaseModel):
30
+ """标准列表响应模型,类似Java中的Result<T>"""
31
+ status: str = "success"
32
+ data: dict
33
+
34
+ class StandardResponse(BaseModel):
35
+ """标准通用响应模型"""
36
+ status: str = "success"
37
+ data: Any
38
+
39
+ class PydanticPaginationResult(BaseModel, Generic[T]):
40
+ """支持Pydantic模型的分页查询结果"""
41
+ items: List[T] = Field(default_factory=list)
42
+ total: int = 0
43
+ page: int = 1
44
+ page_size: int = 10
45
+ total_pages: int = 0
@@ -0,0 +1,31 @@
1
+ """
2
+ 港股公告数据模式
3
+ """
4
+ from datetime import datetime
5
+ from typing import Optional
6
+ from pydantic import Field, BaseModel
7
+
8
+
9
+ class HKAnnouncement(BaseModel):
10
+ """港股公告数据模式"""
11
+
12
+ id: Optional[int] = Field(None, description="主键ID")
13
+ announcement_id: str = Field(..., description="公告唯一标识")
14
+ stock_code: str = Field(..., description="股票代码(逗号分隔)")
15
+ ric_code: str = Field(..., description="RIC代码(逗号分隔)")
16
+ sec_name: Optional[str] = Field(None, description="股票名称(逗号分隔)")
17
+ announcement_title: str = Field(..., description="公告标题")
18
+ announcement_time: datetime = Field(..., description="公告发布时间")
19
+ announcement_type: Optional[str] = Field(None, description="公告类型")
20
+ org_id: Optional[str] = Field(None, description="机构ID")
21
+ adjunct_url: str = Field(..., description="附件URL")
22
+ adjunct_size: Optional[int] = Field(None, description="附件大小(字节)")
23
+ adjunct_type: Optional[str] = Field(None, description="附件类型")
24
+ market_sector: str = Field(..., description="市场板块(HKZB主板/HKCY创业板)")
25
+ is_external_pdf: Optional[bool] = Field(False, description="是否外部PDF")
26
+ critical_announcement: Optional[int] = Field(0, description="是否重要公告")
27
+ created_at: Optional[datetime] = Field(None, description="记录创建时间")
28
+ updated_at: Optional[datetime] = Field(None, description="记录更新时间")
29
+
30
+ class Config:
31
+ from_attributes = True