quantdb-sdk 0.2.5__tar.gz → 0.2.6__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {quantdb_sdk-0.2.5 → quantdb_sdk-0.2.6}/CHANGELOG.md +7 -0
- {quantdb_sdk-0.2.5/quantdb_sdk.egg-info → quantdb_sdk-0.2.6}/PKG-INFO +1 -1
- {quantdb_sdk-0.2.5 → quantdb_sdk-0.2.6}/pyproject.toml +1 -1
- {quantdb_sdk-0.2.5 → quantdb_sdk-0.2.6}/quantdb_sdk/async_client.py +41 -9
- {quantdb_sdk-0.2.5 → quantdb_sdk-0.2.6}/quantdb_sdk/client.py +89 -56
- {quantdb_sdk-0.2.5 → quantdb_sdk-0.2.6/quantdb_sdk.egg-info}/PKG-INFO +1 -1
- {quantdb_sdk-0.2.5 → quantdb_sdk-0.2.6}/LICENSE +0 -0
- {quantdb_sdk-0.2.5 → quantdb_sdk-0.2.6}/MANIFEST.in +0 -0
- {quantdb_sdk-0.2.5 → quantdb_sdk-0.2.6}/README.md +0 -0
- {quantdb_sdk-0.2.5 → quantdb_sdk-0.2.6}/quantdb_sdk/__init__.py +0 -0
- {quantdb_sdk-0.2.5 → quantdb_sdk-0.2.6}/quantdb_sdk/__main__.py +0 -0
- {quantdb_sdk-0.2.5 → quantdb_sdk-0.2.6}/quantdb_sdk/_utils.py +0 -0
- {quantdb_sdk-0.2.5 → quantdb_sdk-0.2.6}/quantdb_sdk/errors.py +0 -0
- {quantdb_sdk-0.2.5 → quantdb_sdk-0.2.6}/quantdb_sdk/py.typed +0 -0
- {quantdb_sdk-0.2.5 → quantdb_sdk-0.2.6}/quantdb_sdk.egg-info/SOURCES.txt +0 -0
- {quantdb_sdk-0.2.5 → quantdb_sdk-0.2.6}/quantdb_sdk.egg-info/dependency_links.txt +0 -0
- {quantdb_sdk-0.2.5 → quantdb_sdk-0.2.6}/quantdb_sdk.egg-info/entry_points.txt +0 -0
- {quantdb_sdk-0.2.5 → quantdb_sdk-0.2.6}/quantdb_sdk.egg-info/requires.txt +0 -0
- {quantdb_sdk-0.2.5 → quantdb_sdk-0.2.6}/quantdb_sdk.egg-info/top_level.txt +0 -0
- {quantdb_sdk-0.2.5 → quantdb_sdk-0.2.6}/setup.cfg +0 -0
- {quantdb_sdk-0.2.5 → quantdb_sdk-0.2.6}/tests/test_async_client.py +0 -0
- {quantdb_sdk-0.2.5 → quantdb_sdk-0.2.6}/tests/test_client.py +0 -0
- {quantdb_sdk-0.2.5 → quantdb_sdk-0.2.6}/tests/test_technical_indicators.py +0 -0
|
@@ -3,6 +3,13 @@
|
|
|
3
3
|
所有 notable 变更都会记录在此文件。格式基于 [Keep a Changelog](https://keepachangelog.com/zh-CN/1.1.0/),
|
|
4
4
|
版本号遵循 [Semantic Versioning](https://semver.org/lang/zh-CN/)。
|
|
5
5
|
|
|
6
|
+
## [0.2.6] - 2026-07-27
|
|
7
|
+
|
|
8
|
+
### Fixed
|
|
9
|
+
- **适配服务端 CDN 直连下载(302 跳转)**:异步客户端(httpx 默认不跟随重定向)此前在服务端开启 302 直连后所有下载/同步接口失效,现已修复。
|
|
10
|
+
- **凭证保护**:同步客户端此前跟随 302 时会把 `X-API-Key` 透传给 CDN 域;现两个客户端统一改为「手动处理 302 + 无鉴权头裸请求直连 CDN」,凭证只发给 QuantDB 网关。
|
|
11
|
+
- 302 响应中的 COS ETag 回填至 CDN 响应,保证进程内 ETag 缓存与 If-None-Match 304 逻辑不受影响。
|
|
12
|
+
|
|
6
13
|
## [0.2.5] - 2026-07-26
|
|
7
14
|
|
|
8
15
|
### Security
|
|
@@ -5,6 +5,7 @@ import io
|
|
|
5
5
|
import os
|
|
6
6
|
import re
|
|
7
7
|
import sqlite3
|
|
8
|
+
from contextlib import asynccontextmanager
|
|
8
9
|
from typing import Any, Dict, List, Optional, Literal
|
|
9
10
|
|
|
10
11
|
import httpx
|
|
@@ -25,6 +26,9 @@ from .errors import (
|
|
|
25
26
|
ValidationError,
|
|
26
27
|
)
|
|
27
28
|
|
|
29
|
+
# 维护提示:每次发版必须与 pyproject.toml 版本号同步
|
|
30
|
+
_USER_AGENT = "QuantDB-Python-SDK/0.2.6"
|
|
31
|
+
|
|
28
32
|
|
|
29
33
|
class AsyncQuantDBClient:
|
|
30
34
|
"""QuantDB 异步客户端。
|
|
@@ -41,7 +45,7 @@ class AsyncQuantDBClient:
|
|
|
41
45
|
):
|
|
42
46
|
self.api_host = validate_api_host(api_host)
|
|
43
47
|
self.timeout = timeout
|
|
44
|
-
headers = {"User-Agent":
|
|
48
|
+
headers = {"User-Agent": _USER_AGENT}
|
|
45
49
|
if api_key:
|
|
46
50
|
headers["X-API-Key"] = api_key
|
|
47
51
|
elif token:
|
|
@@ -53,17 +57,49 @@ class AsyncQuantDBClient:
|
|
|
53
57
|
headers=headers,
|
|
54
58
|
timeout=httpx.Timeout(timeout, connect=5.0),
|
|
55
59
|
)
|
|
60
|
+
# 裸客户端:跟随 302 直连 CDN 用,不携带任何鉴权头,避免凭证泄露给 CDN 域。
|
|
61
|
+
self._bare_client = httpx.AsyncClient(
|
|
62
|
+
headers={"User-Agent": _USER_AGENT},
|
|
63
|
+
timeout=httpx.Timeout(timeout, connect=5.0),
|
|
64
|
+
)
|
|
56
65
|
# 进程内 parquet ETag 缓存:同对象(ETag 未变)不重复下载,避免重复计费。
|
|
57
66
|
self._cache: Dict[str, Dict[str, Any]] = {}
|
|
58
67
|
|
|
59
68
|
async def close(self) -> None:
|
|
60
69
|
"""关闭底层 httpx 客户端。"""
|
|
61
70
|
await self.client.aclose()
|
|
71
|
+
await self._bare_client.aclose()
|
|
62
72
|
|
|
63
73
|
def clear_cache(self) -> None:
|
|
64
74
|
"""清空进程内 Parquet 缓存(强制下次重新下载最新数据)。"""
|
|
65
75
|
self._cache.clear()
|
|
66
76
|
|
|
77
|
+
@asynccontextmanager
|
|
78
|
+
async def _download_stream(self, params: Dict[str, Any], headers: Optional[Dict[str, str]] = None):
|
|
79
|
+
"""下载专用流式请求:手动处理服务端 302 CDN 直连跳转。
|
|
80
|
+
|
|
81
|
+
服务端开启 CDN 直连后,/api/v1/data/download 预扣流量成功时返回
|
|
82
|
+
302 -> CDN 签名 URL。httpx 默认不跟随重定向;这里用不带鉴权头的
|
|
83
|
+
裸客户端直连 Location,鉴权头不会泄露给 CDN 域。
|
|
84
|
+
"""
|
|
85
|
+
async with self.client.stream(
|
|
86
|
+
"GET", f"{self.api_host}/api/v1/data/download", params=params, headers=headers
|
|
87
|
+
) as resp:
|
|
88
|
+
if resp.status_code not in (301, 302, 303, 307, 308):
|
|
89
|
+
yield resp
|
|
90
|
+
return
|
|
91
|
+
location = resp.headers.get("Location", "")
|
|
92
|
+
etag = resp.headers.get("ETag", "")
|
|
93
|
+
if not location:
|
|
94
|
+
raise ServerError("下载重定向缺少 Location 头")
|
|
95
|
+
async with self._bare_client.stream("GET", location) as cdn_resp:
|
|
96
|
+
if cdn_resp.status_code != 200:
|
|
97
|
+
raise ServerError(f"CDN 直连下载失败:HTTP {cdn_resp.status_code}")
|
|
98
|
+
# 网关 302 响应携带 COS ETag;若 CDN 响应缺失则回填,保证缓存逻辑一致。
|
|
99
|
+
if etag and not cdn_resp.headers.get("ETag"):
|
|
100
|
+
cdn_resp.headers["ETag"] = etag
|
|
101
|
+
yield cdn_resp
|
|
102
|
+
|
|
67
103
|
@staticmethod
|
|
68
104
|
def _validate_layout(layout: str) -> Literal["auto", "v1", "v2"]:
|
|
69
105
|
if layout not in {"auto", "v1", "v2"}:
|
|
@@ -398,9 +434,7 @@ class AsyncQuantDBClient:
|
|
|
398
434
|
if object_key:
|
|
399
435
|
params["object_key"] = object_key
|
|
400
436
|
|
|
401
|
-
async with self.
|
|
402
|
-
"GET", f"{self.api_host}/api/v1/data/download", params=params
|
|
403
|
-
) as resp:
|
|
437
|
+
async with self._download_stream(params) as resp:
|
|
404
438
|
if resp.status_code != 200:
|
|
405
439
|
body = await resp.aread()
|
|
406
440
|
# 构造一个完整的 httpx.Response 用于错误解析,然后立即抛出
|
|
@@ -467,9 +501,7 @@ class AsyncQuantDBClient:
|
|
|
467
501
|
req_headers = {}
|
|
468
502
|
if cached and cached.get("etag"):
|
|
469
503
|
req_headers["If-None-Match"] = cached["etag"]
|
|
470
|
-
async with self.
|
|
471
|
-
"GET", f"{self.api_host}/api/v1/data/download", params=params, headers=req_headers
|
|
472
|
-
) as resp:
|
|
504
|
+
async with self._download_stream(params, headers=req_headers) as resp:
|
|
473
505
|
if resp.status_code == 304 and cached:
|
|
474
506
|
return cached["df"]
|
|
475
507
|
if resp.status_code != 200:
|
|
@@ -567,7 +599,7 @@ class AsyncQuantDBClient:
|
|
|
567
599
|
if old and old[0] == obj.get("etag") and old[1] == obj.get("sha256") and os.path.exists(old[3]) and (expected_size is None or os.path.getsize(old[3]) == expected_size): continue
|
|
568
600
|
os.makedirs(os.path.dirname(target), exist_ok=True)
|
|
569
601
|
tmp, digest = target + ".part", hashlib.sha256()
|
|
570
|
-
async with self.
|
|
602
|
+
async with self._download_stream({"category_id": SYNC_DATASET_CATEGORIES[dataset], "sub_category": dataset, "layout": "v2", "object_key": key}) as resp:
|
|
571
603
|
if resp.status_code != 200:
|
|
572
604
|
body = await resp.aread(); self._check_response(httpx.Response(resp.status_code, content=body)); raise QuantDBError("下载失败")
|
|
573
605
|
try:
|
|
@@ -595,7 +627,7 @@ class AsyncQuantDBClient:
|
|
|
595
627
|
os.makedirs(os.path.dirname(target), exist_ok=True)
|
|
596
628
|
tmp, written_size = target + ".part", 0
|
|
597
629
|
try:
|
|
598
|
-
async with self.
|
|
630
|
+
async with self._download_stream({"category_id": SYNC_DATASET_CATEGORIES[dataset], "sub_category": dataset, "layout": "v1", "symbol": obj.get("symbol", "")}) as resp:
|
|
599
631
|
if resp.status_code != 200:
|
|
600
632
|
body = await resp.aread(); self._check_response(httpx.Response(resp.status_code, content=body)); raise QuantDBError("下载失败")
|
|
601
633
|
with open(tmp, "wb") as fh:
|
|
@@ -13,11 +13,11 @@ import requests
|
|
|
13
13
|
from requests.adapters import HTTPAdapter
|
|
14
14
|
from urllib3.util.retry import Retry
|
|
15
15
|
|
|
16
|
-
from ._utils import (
|
|
17
|
-
SYNC_DATASET_CATEGORIES, bytes_to_gb, check_download_size, default_download_dir,
|
|
18
|
-
max_download_bytes, parse_filename_from_content_disposition, safe_filename, safe_join,
|
|
19
|
-
validate_api_host,
|
|
20
|
-
)
|
|
16
|
+
from ._utils import (
|
|
17
|
+
SYNC_DATASET_CATEGORIES, bytes_to_gb, check_download_size, default_download_dir,
|
|
18
|
+
max_download_bytes, parse_filename_from_content_disposition, safe_filename, safe_join,
|
|
19
|
+
validate_api_host,
|
|
20
|
+
)
|
|
21
21
|
from .errors import (
|
|
22
22
|
AuthError,
|
|
23
23
|
InsufficientTrafficError,
|
|
@@ -28,6 +28,9 @@ from .errors import (
|
|
|
28
28
|
ValidationError,
|
|
29
29
|
)
|
|
30
30
|
|
|
31
|
+
# 维护提示:每次发版必须与 pyproject.toml 版本号同步
|
|
32
|
+
_USER_AGENT = "QuantDB-Python-SDK/0.2.6"
|
|
33
|
+
|
|
31
34
|
|
|
32
35
|
class QuantDBClient:
|
|
33
36
|
"""QuantDB 同步客户端。
|
|
@@ -45,14 +48,12 @@ class QuantDBClient:
|
|
|
45
48
|
timeout: tuple = (5, 60),
|
|
46
49
|
max_retries: int = 2,
|
|
47
50
|
):
|
|
48
|
-
self.api_host = validate_api_host(api_host)
|
|
51
|
+
self.api_host = validate_api_host(api_host)
|
|
49
52
|
self.timeout = timeout
|
|
50
53
|
self.token: Optional[str] = None
|
|
51
54
|
self.headers: Dict[str, str] = {}
|
|
52
55
|
self.session = requests.Session()
|
|
53
|
-
|
|
54
|
-
# 维护提示:每次版本号变化必须同步改这里(init 里的 __version__ 走 metadata 自动同步)
|
|
55
|
-
self.session.headers.update({"User-Agent": "QuantDB-Python-SDK/0.2.5"})
|
|
56
|
+
self.session.headers.update({"User-Agent": _USER_AGENT})
|
|
56
57
|
|
|
57
58
|
if api_key:
|
|
58
59
|
self.headers = {"X-API-Key": api_key}
|
|
@@ -98,6 +99,7 @@ class QuantDBClient:
|
|
|
98
99
|
json: Optional[dict] = None,
|
|
99
100
|
stream: bool = False,
|
|
100
101
|
headers: Optional[dict] = None,
|
|
102
|
+
allow_redirects: bool = True,
|
|
101
103
|
) -> requests.Response:
|
|
102
104
|
url = f"{self.api_host}{path}"
|
|
103
105
|
resp = self.session.request(
|
|
@@ -108,6 +110,7 @@ class QuantDBClient:
|
|
|
108
110
|
stream=stream,
|
|
109
111
|
timeout=self.timeout,
|
|
110
112
|
headers=headers,
|
|
113
|
+
allow_redirects=allow_redirects,
|
|
111
114
|
)
|
|
112
115
|
return resp
|
|
113
116
|
|
|
@@ -115,6 +118,38 @@ class QuantDBClient:
|
|
|
115
118
|
resp = self._request("GET", path, params=params)
|
|
116
119
|
return self._check_response(resp)
|
|
117
120
|
|
|
121
|
+
def _download_stream(self, params: dict, headers: Optional[dict] = None) -> requests.Response:
|
|
122
|
+
"""下载专用流式请求:手动处理服务端 302 CDN 直连跳转。
|
|
123
|
+
|
|
124
|
+
服务端开启 CDN 直连后,/api/v1/data/download 预扣流量成功时返回
|
|
125
|
+
302 -> CDN 签名 URL。requests 自动跟随会把自定义 X-API-Key 头
|
|
126
|
+
透传给 CDN 域,故禁用自动跟随,改用不带鉴权头的裸请求直连。
|
|
127
|
+
"""
|
|
128
|
+
resp = self._request(
|
|
129
|
+
"GET", "/api/v1/data/download", params=params, stream=True,
|
|
130
|
+
headers=headers, allow_redirects=False,
|
|
131
|
+
)
|
|
132
|
+
if resp.status_code not in (301, 302, 303, 307, 308):
|
|
133
|
+
return resp
|
|
134
|
+
location = resp.headers.get("Location", "")
|
|
135
|
+
etag = resp.headers.get("ETag", "")
|
|
136
|
+
resp.close()
|
|
137
|
+
if not location:
|
|
138
|
+
raise ServerError("下载重定向缺少 Location 头")
|
|
139
|
+
cdn_resp = requests.get(
|
|
140
|
+
location,
|
|
141
|
+
stream=True,
|
|
142
|
+
timeout=self.timeout,
|
|
143
|
+
headers={"User-Agent": _USER_AGENT},
|
|
144
|
+
)
|
|
145
|
+
if cdn_resp.status_code != 200:
|
|
146
|
+
cdn_resp.close()
|
|
147
|
+
raise ServerError(f"CDN 直连下载失败:HTTP {cdn_resp.status_code}")
|
|
148
|
+
# 网关 302 响应携带 COS ETag;若 CDN 响应缺失则回填,保证客户端缓存逻辑一致。
|
|
149
|
+
if etag and not cdn_resp.headers.get("ETag"):
|
|
150
|
+
cdn_resp.headers["ETag"] = etag
|
|
151
|
+
return cdn_resp
|
|
152
|
+
|
|
118
153
|
def _post(self, path: str, json: Optional[dict] = None) -> dict:
|
|
119
154
|
resp = self._request("POST", path, json=json)
|
|
120
155
|
return self._check_response(resp)
|
|
@@ -486,38 +521,36 @@ class QuantDBClient:
|
|
|
486
521
|
if object_key:
|
|
487
522
|
params["object_key"] = object_key
|
|
488
523
|
|
|
489
|
-
resp = self.
|
|
490
|
-
"GET", "/api/v1/data/download", params=params, stream=True
|
|
491
|
-
)
|
|
524
|
+
resp = self._download_stream(params)
|
|
492
525
|
if resp.status_code != 200:
|
|
493
526
|
self._check_response(resp)
|
|
494
527
|
|
|
495
528
|
fallback = f"{sub_category}.parquet"
|
|
496
529
|
if symbol:
|
|
497
530
|
fallback = f"{sub_category}_{symbol}.parquet"
|
|
498
|
-
filename = safe_filename(
|
|
499
|
-
parse_filename_from_content_disposition(resp.headers.get("Content-Disposition", ""), fallback),
|
|
500
|
-
safe_filename(fallback),
|
|
501
|
-
)
|
|
531
|
+
filename = safe_filename(
|
|
532
|
+
parse_filename_from_content_disposition(resp.headers.get("Content-Disposition", ""), fallback),
|
|
533
|
+
safe_filename(fallback),
|
|
534
|
+
)
|
|
502
535
|
|
|
503
536
|
save_path = os.path.join(save_dir, filename)
|
|
504
537
|
tmp_path = save_path + ".part"
|
|
505
|
-
maximum = max_download_bytes()
|
|
506
|
-
check_download_size(resp.headers.get("Content-Length"), maximum)
|
|
507
|
-
written = 0
|
|
508
|
-
try:
|
|
509
|
-
with open(tmp_path, "wb") as f:
|
|
510
|
-
for chunk in resp.iter_content(chunk_size=8192):
|
|
511
|
-
if chunk:
|
|
512
|
-
written += len(chunk)
|
|
513
|
-
if written > maximum:
|
|
514
|
-
raise ServerError(f"下载文件超过大小限制({maximum} 字节)")
|
|
515
|
-
f.write(chunk)
|
|
516
|
-
os.replace(tmp_path, save_path)
|
|
517
|
-
except Exception:
|
|
518
|
-
if os.path.exists(tmp_path):
|
|
519
|
-
os.remove(tmp_path)
|
|
520
|
-
raise
|
|
538
|
+
maximum = max_download_bytes()
|
|
539
|
+
check_download_size(resp.headers.get("Content-Length"), maximum)
|
|
540
|
+
written = 0
|
|
541
|
+
try:
|
|
542
|
+
with open(tmp_path, "wb") as f:
|
|
543
|
+
for chunk in resp.iter_content(chunk_size=8192):
|
|
544
|
+
if chunk:
|
|
545
|
+
written += len(chunk)
|
|
546
|
+
if written > maximum:
|
|
547
|
+
raise ServerError(f"下载文件超过大小限制({maximum} 字节)")
|
|
548
|
+
f.write(chunk)
|
|
549
|
+
os.replace(tmp_path, save_path)
|
|
550
|
+
except Exception:
|
|
551
|
+
if os.path.exists(tmp_path):
|
|
552
|
+
os.remove(tmp_path)
|
|
553
|
+
raise
|
|
521
554
|
return os.path.abspath(save_path)
|
|
522
555
|
|
|
523
556
|
def load_as_df(
|
|
@@ -554,7 +587,7 @@ class QuantDBClient:
|
|
|
554
587
|
headers = {}
|
|
555
588
|
if cached and cached.get("etag"):
|
|
556
589
|
headers["If-None-Match"] = cached["etag"]
|
|
557
|
-
resp = self.
|
|
590
|
+
resp = self._download_stream(params, headers=headers)
|
|
558
591
|
if resp.status_code == 304 and cached:
|
|
559
592
|
return cached["df"]
|
|
560
593
|
if resp.status_code != 200:
|
|
@@ -563,16 +596,16 @@ class QuantDBClient:
|
|
|
563
596
|
# 命中缓存(ETag 未变):复用已解析的 df,不重复消耗解析
|
|
564
597
|
if cached and etag and cached.get("etag") == etag:
|
|
565
598
|
return cached["df"]
|
|
566
|
-
maximum = max_download_bytes()
|
|
567
|
-
check_download_size(resp.headers.get("Content-Length"), maximum)
|
|
568
|
-
payload, written = io.BytesIO(), 0
|
|
569
|
-
for chunk in resp.iter_content(chunk_size=1024 * 1024):
|
|
570
|
-
if chunk:
|
|
571
|
-
written += len(chunk)
|
|
572
|
-
if written > maximum:
|
|
573
|
-
raise ServerError(f"下载文件超过大小限制({maximum} 字节)")
|
|
574
|
-
payload.write(chunk)
|
|
575
|
-
df = pd.read_parquet(payload)
|
|
599
|
+
maximum = max_download_bytes()
|
|
600
|
+
check_download_size(resp.headers.get("Content-Length"), maximum)
|
|
601
|
+
payload, written = io.BytesIO(), 0
|
|
602
|
+
for chunk in resp.iter_content(chunk_size=1024 * 1024):
|
|
603
|
+
if chunk:
|
|
604
|
+
written += len(chunk)
|
|
605
|
+
if written > maximum:
|
|
606
|
+
raise ServerError(f"下载文件超过大小限制({maximum} 字节)")
|
|
607
|
+
payload.write(chunk)
|
|
608
|
+
df = pd.read_parquet(payload)
|
|
576
609
|
if etag:
|
|
577
610
|
self._cache[cache_key] = {"etag": etag, "df": df}
|
|
578
611
|
return df
|
|
@@ -589,12 +622,12 @@ class QuantDBClient:
|
|
|
589
622
|
|
|
590
623
|
若文件不存在会先自动下载。
|
|
591
624
|
|
|
592
|
-
``sql`` 仅接受 WHERE 条件字符串;SDK 固定查询下载的单个 Parquet 文件。
|
|
625
|
+
``sql`` 仅接受 WHERE 条件字符串;SDK 固定查询下载的单个 Parquet 文件。
|
|
593
626
|
|
|
594
627
|
安全限制:
|
|
595
628
|
- 不允许分号(;),防止多语句注入
|
|
596
629
|
- 不允许注释(-- 或 /* */),防止注释注入
|
|
597
|
-
- 不允许子查询、JOIN、外部表函数或任何数据修改语句
|
|
630
|
+
- 不允许子查询、JOIN、外部表函数或任何数据修改语句
|
|
598
631
|
"""
|
|
599
632
|
file_path = self.download_file(
|
|
600
633
|
category_id=category_id,
|
|
@@ -611,19 +644,19 @@ class QuantDBClient:
|
|
|
611
644
|
|
|
612
645
|
# 安全检查:拒绝危险字符和关键字
|
|
613
646
|
dangerous_keywords = [
|
|
614
|
-
";", "--", "/*", "*/", "select", "from", "join", "union", "drop",
|
|
615
|
-
"delete", "insert", "update", "alter", "create", "exec", "execute",
|
|
616
|
-
"attach", "copy", "pragma", "install", "load", "read_", "http", "glob",
|
|
647
|
+
";", "--", "/*", "*/", "select", "from", "join", "union", "drop",
|
|
648
|
+
"delete", "insert", "update", "alter", "create", "exec", "execute",
|
|
649
|
+
"attach", "copy", "pragma", "install", "load", "read_", "http", "glob",
|
|
617
650
|
]
|
|
618
651
|
sql_upper = sql.upper()
|
|
619
652
|
for kw in dangerous_keywords:
|
|
620
653
|
if kw in sql_upper:
|
|
621
654
|
raise ValidationError(
|
|
622
655
|
f"SQL 包含危险关键字 '{kw}',已被拒绝。"
|
|
623
|
-
"query_local 仅支持针对已下载文件的 WHERE 条件。"
|
|
624
|
-
)
|
|
625
|
-
sql = f"SELECT * FROM '{clean_path}' WHERE {sql}"
|
|
626
|
-
return duckdb.query(sql).df()
|
|
656
|
+
"query_local 仅支持针对已下载文件的 WHERE 条件。"
|
|
657
|
+
)
|
|
658
|
+
sql = f"SELECT * FROM '{clean_path}' WHERE {sql}"
|
|
659
|
+
return duckdb.query(sql).df()
|
|
627
660
|
|
|
628
661
|
def get_local_warehouse(self, save_dir: Optional[str] = None) -> "DuckDBWarehouse":
|
|
629
662
|
"""获取 DuckDB 本地数据仓库实例,用于离线多表 SQL JOIN 查询。"""
|
|
@@ -656,12 +689,12 @@ class QuantDBClient:
|
|
|
656
689
|
for obj in release.get("objects", []):
|
|
657
690
|
key = self._normalise_release_key(obj["key"])
|
|
658
691
|
relative_path = obj.get("relative_path") or key
|
|
659
|
-
target = safe_join(root, relative_path)
|
|
692
|
+
target = safe_join(root, relative_path)
|
|
660
693
|
expected_size = obj.get("size")
|
|
661
694
|
old = state.execute("SELECT etag, sha256, size, path FROM objects WHERE key=?", (key,)).fetchone()
|
|
662
695
|
if old and old[0] == obj.get("etag") and old[1] == obj.get("sha256") and os.path.exists(old[3]) and (expected_size is None or os.path.getsize(old[3]) == expected_size):
|
|
663
696
|
continue
|
|
664
|
-
resp = self.
|
|
697
|
+
resp = self._download_stream({"category_id": SYNC_DATASET_CATEGORIES[dataset], "sub_category": dataset, "layout": "v2", "object_key": key})
|
|
665
698
|
if resp.status_code != 200:
|
|
666
699
|
self._check_response(resp)
|
|
667
700
|
os.makedirs(os.path.dirname(target), exist_ok=True)
|
|
@@ -694,12 +727,12 @@ class QuantDBClient:
|
|
|
694
727
|
manifest = self._get("/api/v1/data/download/manifest", {"category_id": SYNC_DATASET_CATEGORIES[dataset], "sub_category": dataset, "layout": "v1"})
|
|
695
728
|
for obj in manifest.get("files", []):
|
|
696
729
|
key, relative_path = obj["key"], obj.get("relative_path") or obj["key"]
|
|
697
|
-
target = safe_join(root, relative_path)
|
|
730
|
+
target = safe_join(root, relative_path)
|
|
698
731
|
expected_size = obj.get("size")
|
|
699
732
|
old = state.execute("SELECT etag, size, path FROM objects WHERE key=?", (key,)).fetchone()
|
|
700
733
|
if old and old[0] == obj.get("etag") and os.path.exists(old[2]) and (expected_size is None or os.path.getsize(old[2]) == expected_size):
|
|
701
734
|
continue
|
|
702
|
-
resp = self.
|
|
735
|
+
resp = self._download_stream({"category_id": SYNC_DATASET_CATEGORIES[dataset], "sub_category": dataset, "layout": "v1", "symbol": obj.get("symbol", "")})
|
|
703
736
|
if resp.status_code != 200: self._check_response(resp)
|
|
704
737
|
os.makedirs(os.path.dirname(target), exist_ok=True)
|
|
705
738
|
tmp = target + ".part"
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|