datamule-hub 0.2.3__py3-none-win_amd64.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- datamule_hub-0.2.3.dist-info/METADATA +23 -0
- datamule_hub-0.2.3.dist-info/RECORD +35 -0
- datamule_hub-0.2.3.dist-info/WHEEL +5 -0
- datamule_hub-0.2.3.dist-info/licenses/LICENSE +7 -0
- datamule_hub-0.2.3.dist-info/top_level.txt +1 -0
- datamulehub/__init__.py +16 -0
- datamulehub/api_key.py +10 -0
- datamulehub/bin/datamule-archive-downloader.exe +0 -0
- datamulehub/object_transfer/__init__.py +11 -0
- datamulehub/object_transfer/gcs/__init__.py +0 -0
- datamulehub/object_transfer/gcs/bucket_transfer.py +80 -0
- datamulehub/object_transfer/gcs/datasets_transfer.py +87 -0
- datamulehub/object_transfer/gcs/utils.py +18 -0
- datamulehub/object_transfer/s3/__init__.py +0 -0
- datamulehub/object_transfer/s3/bucket_transfer.py +85 -0
- datamulehub/object_transfer/s3/datasets_transfer.py +142 -0
- datamulehub/object_transfer/utils.py +63 -0
- datamulehub/utils/__init__.py +0 -0
- datamulehub/utils/format_accession.py +20 -0
- datamulehub/v3/__init__.py +13 -0
- datamulehub/v3/databases/__init__.py +3 -0
- datamulehub/v3/databases/athena.py +231 -0
- datamulehub/v3/databases/fastest.py +312 -0
- datamulehub/v3/datasets/__init__.py +3 -0
- datamulehub/v3/datasets/datasets.py +161 -0
- datamulehub/v3/sec_filings_archive/__init__.py +4 -0
- datamulehub/v3/sec_filings_archive/_rust.py +110 -0
- datamulehub/v3/sec_filings_archive/sgml.py +288 -0
- datamulehub/v3/sec_filings_archive/tar.py +806 -0
- datamulehub/v3/sec_filings_archive/utils.py +136 -0
- datamulehub/v3/sec_filings_lookup/__init__.py +3 -0
- datamulehub/v3/sec_filings_lookup/sec_filings_lookup.py +356 -0
- datamulehub/v3/sec_filings_notifications/__init__.py +10 -0
- datamulehub/v3/sec_filings_notifications/webhooks.py +102 -0
- datamulehub/v3/sec_filings_notifications/websocket.py +391 -0
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: datamule-hub
|
|
3
|
+
Version: 0.2.3
|
|
4
|
+
Summary: Access Datamule cloud
|
|
5
|
+
Home-page: https://github.com/john-friedman/datamule-hub
|
|
6
|
+
Author: John Friedman
|
|
7
|
+
Requires-Python: >=3.9
|
|
8
|
+
License-File: LICENSE
|
|
9
|
+
Requires-Dist: tqdm
|
|
10
|
+
Requires-Dist: aiohttp
|
|
11
|
+
Requires-Dist: aioboto3
|
|
12
|
+
Requires-Dist: gcloud-aio-storage
|
|
13
|
+
Requires-Dist: google-auth
|
|
14
|
+
Requires-Dist: google-cloud-storage
|
|
15
|
+
Requires-Dist: pyarrow
|
|
16
|
+
Requires-Dist: websocket-client
|
|
17
|
+
Requires-Dist: zstandard
|
|
18
|
+
Dynamic: author
|
|
19
|
+
Dynamic: home-page
|
|
20
|
+
Dynamic: license-file
|
|
21
|
+
Dynamic: requires-dist
|
|
22
|
+
Dynamic: requires-python
|
|
23
|
+
Dynamic: summary
|
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
datamule_hub-0.2.3.dist-info/licenses/LICENSE,sha256=e4vO5__LjkwRec1EOexxayzJEymAkwH9gtWEpkF2oRY,1066
|
|
2
|
+
datamulehub/__init__.py,sha256=1San9XeBEKF6Fsp8tBr2zs0VU-1W-w9mERjH8YmJHwc,368
|
|
3
|
+
datamulehub/api_key.py,sha256=ku5HQhtnORtomdka6uPdij3H5INiQZHj2XdNDfYZeEE,270
|
|
4
|
+
datamulehub/bin/datamule-archive-downloader.exe,sha256=qifj77g4rOtq5AS71Bv8MRV7Y7d44q6FcNUYrROGWDk,4059648
|
|
5
|
+
datamulehub/object_transfer/__init__.py,sha256=wy4EgEwMDvDZWv0T4tlTfNvKxfleDsNLs2n5pWEY8nE,432
|
|
6
|
+
datamulehub/object_transfer/utils.py,sha256=2nRtl9LSfkPnWHNmRWaLddEycz6S-zTFtj55z2Gc9bo,2442
|
|
7
|
+
datamulehub/object_transfer/gcs/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
8
|
+
datamulehub/object_transfer/gcs/bucket_transfer.py,sha256=OxgCL2ccmh-CKvDf5jDefnYyiu3PcLKmv5MItIDIDVc,4003
|
|
9
|
+
datamulehub/object_transfer/gcs/datasets_transfer.py,sha256=8ZpXRMrW4JC49xVCvi7po1zCez75PNAzuNxaSSFuA70,3656
|
|
10
|
+
datamulehub/object_transfer/gcs/utils.py,sha256=yOQZJoOpYphoSy3YKQXtNkW-nkPoJQzM3VUzi66SGo0,664
|
|
11
|
+
datamulehub/object_transfer/s3/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
12
|
+
datamulehub/object_transfer/s3/bucket_transfer.py,sha256=t79qKzew2yHGIzq1LeemOl620LBTFSSDQo2QUy-YUUs,4251
|
|
13
|
+
datamulehub/object_transfer/s3/datasets_transfer.py,sha256=r6a54lyzACuqH69SPKOSJNS8FOpZme2BjkzH-HJCTAU,6737
|
|
14
|
+
datamulehub/utils/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
15
|
+
datamulehub/utils/format_accession.py,sha256=Es_Y48lOq_Z_tByDbWTifcO3-1GA4izNAA2_nFIgAy8,697
|
|
16
|
+
datamulehub/v3/__init__.py,sha256=9kZ5EV1j8xnW1yW8jvh9ho3xU7vu8dt8vCt66d6SMtk,301
|
|
17
|
+
datamulehub/v3/databases/__init__.py,sha256=4rIDESjUEJpUWDMRFpYyKDo2D66zJOQtYKRlq_CE0aQ,76
|
|
18
|
+
datamulehub/v3/databases/athena.py,sha256=fSuzORl4pon8BCfG2IKxy40fEF8tu4rwZ91ZP2VXvWE,7247
|
|
19
|
+
datamulehub/v3/databases/fastest.py,sha256=JsfGwcdTeMQgGck8mfL15-GvC3xvp5Z5Q_XB-4LV4S0,10113
|
|
20
|
+
datamulehub/v3/datasets/__init__.py,sha256=Z2TaPOCpIjWYY6hv7Abi-uplUk27fMBxDcXnjg9bk_U,110
|
|
21
|
+
datamulehub/v3/datasets/datasets.py,sha256=f0GylDgsT9yoKXAoIiNpes2vP4TRccVXCZrSKl2S_io,5564
|
|
22
|
+
datamulehub/v3/sec_filings_archive/__init__.py,sha256=VxEAv15gqJ3LfUbeaALE7We0lU8Vwc-SuSaBSi0HTKM,111
|
|
23
|
+
datamulehub/v3/sec_filings_archive/_rust.py,sha256=Wvk3abMqRnzdQxuVijyOVB4KfSNsMBkPseRxKdzMrWY,3257
|
|
24
|
+
datamulehub/v3/sec_filings_archive/sgml.py,sha256=fVNgoRK6nr8xXcvXovgXx4HAxTaS4wIuLuRVJdjzGbo,9649
|
|
25
|
+
datamulehub/v3/sec_filings_archive/tar.py,sha256=wkENjqweIcbjnbjRYAFv0PHTW9Cy2Om_82GQlCS3oXs,26742
|
|
26
|
+
datamulehub/v3/sec_filings_archive/utils.py,sha256=6SYP-O5Kok-4BAASEjOCZwLgHAuL4m_g5VCE1lDUgMQ,4427
|
|
27
|
+
datamulehub/v3/sec_filings_lookup/__init__.py,sha256=fjAgAQwr8Jjx1VV9zqx6fW0uCLlg-2wJzFYV-1qhY-Y,154
|
|
28
|
+
datamulehub/v3/sec_filings_lookup/sec_filings_lookup.py,sha256=abxWSfquspOMqb0PzLajrpeGsb2CjjphdV-aGbXkrwY,9665
|
|
29
|
+
datamulehub/v3/sec_filings_notifications/__init__.py,sha256=M2a6PEnwGqgr6I3j6levqC44dWKkFpLXQQtcbhspJgs,272
|
|
30
|
+
datamulehub/v3/sec_filings_notifications/webhooks.py,sha256=MLhMuYc88YyQNDhMTsBI_q7njaKVn6D5Bqp0dqU9JTQ,2991
|
|
31
|
+
datamulehub/v3/sec_filings_notifications/websocket.py,sha256=vHrsvWQqodc7ApVVpkzVHWfaNBZXSMdXQhF-V9AX0Ac,12176
|
|
32
|
+
datamule_hub-0.2.3.dist-info/METADATA,sha256=0ppVioqPa7e7ZjUcppCsWv74SngoeIxDsFmsq-3X844,600
|
|
33
|
+
datamule_hub-0.2.3.dist-info/WHEEL,sha256=Zx98gwb_dQKckJK3HEOVKnxxD52EfU_PXDG_usMJ2ng,98
|
|
34
|
+
datamule_hub-0.2.3.dist-info/top_level.txt,sha256=aef5Z5BB_fOwDXmIWr9yJ1zncoNVce6MWgNwtaxrUSk,12
|
|
35
|
+
datamule_hub-0.2.3.dist-info/RECORD,,
|
|
@@ -0,0 +1,7 @@
|
|
|
1
|
+
Copyright 2026 John Friedman
|
|
2
|
+
|
|
3
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy of this software and associated documentation files (the “Software”), to deal in the Software without restriction, including without limitation the rights to use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the Software, and to permit persons to whom the Software is furnished to do so, subject to the following conditions:
|
|
4
|
+
|
|
5
|
+
The above copyright notice and this permission notice shall be included in all copies or substantial portions of the Software.
|
|
6
|
+
|
|
7
|
+
THE SOFTWARE IS PROVIDED “AS IS”, WITHOUT WARRANTY OF ANY KIND, EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
datamulehub
|
datamulehub/__init__.py
ADDED
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
from . import object_transfer
|
|
2
|
+
from .v3 import databases
|
|
3
|
+
from .v3 import datasets
|
|
4
|
+
from .v3 import sec_filings_archive
|
|
5
|
+
from .v3 import sec_filings_lookup
|
|
6
|
+
from .v3 import sec_filings_notifications
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
__all__ = [
|
|
10
|
+
"databases",
|
|
11
|
+
"datasets",
|
|
12
|
+
"object_transfer",
|
|
13
|
+
"sec_filings_archive",
|
|
14
|
+
"sec_filings_lookup",
|
|
15
|
+
"sec_filings_notifications",
|
|
16
|
+
]
|
datamulehub/api_key.py
ADDED
|
Binary file
|
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
from .gcs.bucket_transfer import bucket_transfer as gcs_archive_transfer
|
|
2
|
+
from .gcs.datasets_transfer import datasets_transfer as gcs_dataset_transfer
|
|
3
|
+
from .s3.bucket_transfer import bucket_transfer as s3_archive_transfer
|
|
4
|
+
from .s3.datasets_transfer import datasets_transfer as s3_dataset_transfer
|
|
5
|
+
|
|
6
|
+
__all__ = [
|
|
7
|
+
"gcs_archive_transfer",
|
|
8
|
+
"gcs_dataset_transfer",
|
|
9
|
+
"s3_archive_transfer",
|
|
10
|
+
"s3_dataset_transfer",
|
|
11
|
+
]
|
|
File without changes
|
|
@@ -0,0 +1,80 @@
|
|
|
1
|
+
import asyncio
|
|
2
|
+
import aiohttp
|
|
3
|
+
import ssl
|
|
4
|
+
import json
|
|
5
|
+
from urllib.parse import urlparse
|
|
6
|
+
from tqdm import tqdm
|
|
7
|
+
from datetime import datetime, timezone
|
|
8
|
+
|
|
9
|
+
from ..utils import _generate_dates, _get_urls
|
|
10
|
+
from .utils import _get_storage
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
async def _transfer_file(session, storage, semaphore, url, bucket, prefix=None, retry_errors=3):
|
|
14
|
+
async with semaphore:
|
|
15
|
+
key = urlparse(url).path.split('/')[-1]
|
|
16
|
+
if prefix:
|
|
17
|
+
key = f"{prefix.rstrip('/')}/{key}"
|
|
18
|
+
last_error = None
|
|
19
|
+
for attempt in range(retry_errors + 1):
|
|
20
|
+
try:
|
|
21
|
+
async with session.get(url) as response:
|
|
22
|
+
if response.status != 200:
|
|
23
|
+
raise aiohttp.ClientResponseError(response.request_info, response.history, status=response.status)
|
|
24
|
+
content = await response.read()
|
|
25
|
+
await storage.upload(bucket, key, content,
|
|
26
|
+
content_type=response.headers.get('Content-Type', 'application/octet-stream'),
|
|
27
|
+
metadata={'source-url': url, 'transfer-date': datetime.now(timezone.utc).isoformat()}
|
|
28
|
+
)
|
|
29
|
+
return {'success': True, 'url': url, 'size_bytes': len(content)}
|
|
30
|
+
except Exception as e:
|
|
31
|
+
if attempt < retry_errors:
|
|
32
|
+
await asyncio.sleep(2 ** attempt)
|
|
33
|
+
last_error = e
|
|
34
|
+
return {'success': False, 'url': url, 'error': str(last_error)}
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
async def _transfer_urls(urls, gcs_credentials, max_workers, retry_errors, prefix=None):
|
|
38
|
+
connector = aiohttp.TCPConnector(limit=max_workers, ssl=ssl.create_default_context(), ttl_dns_cache=300)
|
|
39
|
+
async with aiohttp.ClientSession(connector=connector, timeout=aiohttp.ClientTimeout(total=7200)) as session:
|
|
40
|
+
async with _get_storage(gcs_credentials, session) as storage:
|
|
41
|
+
semaphore = asyncio.Semaphore(max_workers)
|
|
42
|
+
tasks = [_transfer_file(session, storage, semaphore, url, gcs_credentials['bucket_name'], prefix, retry_errors) for url in urls]
|
|
43
|
+
|
|
44
|
+
failed, total_bytes = [], 0
|
|
45
|
+
with tqdm(total=len(urls), desc="Transferring files", unit="file") as pbar:
|
|
46
|
+
for coro in asyncio.as_completed(tasks):
|
|
47
|
+
result = await coro
|
|
48
|
+
if result['success']:
|
|
49
|
+
total_bytes += result['size_bytes']
|
|
50
|
+
else:
|
|
51
|
+
failed.append(result)
|
|
52
|
+
pbar.set_postfix({'Total': f"{total_bytes / (1024**3):.2f} GB"})
|
|
53
|
+
pbar.update(1)
|
|
54
|
+
|
|
55
|
+
return failed
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def bucket_transfer(datamule_bucket, gcs_credentials, max_workers=4, errors_json_filename='errors_bucket_transfer.json',
|
|
59
|
+
retry_errors=3, force_daily=True, cik=None, submission_type=None, filing_date=None, accession_number=None, prefix=None):
|
|
60
|
+
|
|
61
|
+
if datamule_bucket not in ['filings_sgml_r2', 'sec_filings_sgml_r2']:
|
|
62
|
+
raise ValueError('Datamule S3 bucket not found.')
|
|
63
|
+
|
|
64
|
+
if accession_number is not None and any(p is not None for p in [cik, submission_type, filing_date]):
|
|
65
|
+
raise ValueError('If accession_number is provided, cik, submission_type, and filing_date must be None.')
|
|
66
|
+
|
|
67
|
+
dates = _generate_dates(filing_date) if force_daily and filing_date else [filing_date]
|
|
68
|
+
|
|
69
|
+
for date in dates:
|
|
70
|
+
if len(dates) > 1:
|
|
71
|
+
print(f"Transferring {date}")
|
|
72
|
+
urls = _get_urls(submission_type=submission_type, cik=cik, filing_date=date, accession_number=accession_number)
|
|
73
|
+
failed = asyncio.run(_transfer_urls(urls, gcs_credentials, max_workers, retry_errors, prefix=prefix))
|
|
74
|
+
|
|
75
|
+
if failed and errors_json_filename:
|
|
76
|
+
with open(errors_json_filename, 'w') as f:
|
|
77
|
+
json.dump(failed, f, indent=2)
|
|
78
|
+
print(f"Saved {len(failed)} errors to {errors_json_filename}")
|
|
79
|
+
|
|
80
|
+
print(f"Transfer complete: {len(urls) - len(failed)}/{len(urls)} files successful")
|
|
@@ -0,0 +1,87 @@
|
|
|
1
|
+
import json
|
|
2
|
+
import urllib
|
|
3
|
+
import urllib.request
|
|
4
|
+
from tqdm import tqdm
|
|
5
|
+
from datetime import datetime, timezone
|
|
6
|
+
from google.cloud import storage as gcs
|
|
7
|
+
|
|
8
|
+
from ...api_key import get_api_key
|
|
9
|
+
from ...v3.datasets import resolve_path
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def _get_dataset_url(dataset, api_key=None):
|
|
13
|
+
object_key = resolve_path(dataset)
|
|
14
|
+
body = json.dumps({"path": object_key}).encode("utf-8")
|
|
15
|
+
request = urllib.request.Request(
|
|
16
|
+
"https://api.datamule.xyz/v3/get-s3-link",
|
|
17
|
+
data=body,
|
|
18
|
+
headers={
|
|
19
|
+
"Authorization": f"Bearer {get_api_key(api_key)}",
|
|
20
|
+
"Content-Type": "application/json",
|
|
21
|
+
"User-Agent": "datamule-hub",
|
|
22
|
+
},
|
|
23
|
+
method="POST",
|
|
24
|
+
)
|
|
25
|
+
with urllib.request.urlopen(request) as response:
|
|
26
|
+
data = json.loads(response.read().decode('utf-8'))
|
|
27
|
+
if not data.get('success'):
|
|
28
|
+
raise Exception(f"API error: {data.get('error', 'Unknown error')}")
|
|
29
|
+
billing = data.get('metadata', {}).get('billing', {})
|
|
30
|
+
return data['data']['download_url'], data['data']['size_gb'], billing, data['data'].get('object_key', object_key)
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def _get_gcs_client(gcs_credentials):
|
|
34
|
+
if 'service_file' in gcs_credentials:
|
|
35
|
+
return gcs.Client.from_service_account_json(gcs_credentials['service_file'])
|
|
36
|
+
return gcs.Client()
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def _transfer_dataset(client, bucket, dataset, prefix=None, retry_errors=3):
|
|
40
|
+
download_url, size_gb, billing, object_key = _get_dataset_url(dataset)
|
|
41
|
+
|
|
42
|
+
key = object_key.strip("/") or f"{dataset}.download"
|
|
43
|
+
if prefix:
|
|
44
|
+
key = f"{prefix.rstrip('/')}/{key}"
|
|
45
|
+
|
|
46
|
+
last_error = None
|
|
47
|
+
for attempt in range(retry_errors + 1):
|
|
48
|
+
try:
|
|
49
|
+
request = urllib.request.Request(download_url, headers={'User-Agent': 'datamule-python'})
|
|
50
|
+
with urllib.request.urlopen(request) as response:
|
|
51
|
+
content_type = response.headers.get('Content-Type', 'application/octet-stream')
|
|
52
|
+
blob = bucket.blob(key)
|
|
53
|
+
blob.metadata = {
|
|
54
|
+
'source-url': download_url,
|
|
55
|
+
'transfer-date': datetime.now(timezone.utc).isoformat(),
|
|
56
|
+
}
|
|
57
|
+
blob.upload_from_file(response, content_type=content_type)
|
|
58
|
+
return {'success': True, 'dataset': dataset, 'size_bytes': blob.size, 'billing': billing}
|
|
59
|
+
except Exception as e:
|
|
60
|
+
if attempt < retry_errors:
|
|
61
|
+
download_url, _, _, _ = _get_dataset_url(dataset)
|
|
62
|
+
last_error = e
|
|
63
|
+
|
|
64
|
+
return {'success': False, 'dataset': dataset, 'error': str(last_error)}
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def datasets_transfer(datasets, gcs_credentials, errors_json_filename='errors_datasets_transfer.json', retry_errors=3, prefix=None):
|
|
68
|
+
client = _get_gcs_client(gcs_credentials)
|
|
69
|
+
bucket = client.bucket(gcs_credentials['bucket_name'])
|
|
70
|
+
|
|
71
|
+
failed, total_bytes = [], 0
|
|
72
|
+
with tqdm(total=len(datasets), desc="Transferring datasets", unit="dataset") as pbar:
|
|
73
|
+
for dataset in datasets:
|
|
74
|
+
result = _transfer_dataset(client, bucket, dataset, prefix, retry_errors)
|
|
75
|
+
if result['success']:
|
|
76
|
+
total_bytes += result.get('size_bytes', 0)
|
|
77
|
+
else:
|
|
78
|
+
failed.append(result)
|
|
79
|
+
pbar.set_postfix({'Total': f"{total_bytes / (1024**3):.2f} GB"})
|
|
80
|
+
pbar.update(1)
|
|
81
|
+
|
|
82
|
+
if failed and errors_json_filename:
|
|
83
|
+
with open(errors_json_filename, 'w') as f:
|
|
84
|
+
json.dump(failed, f, indent=2)
|
|
85
|
+
print(f"Saved {len(failed)} errors to {errors_json_filename}")
|
|
86
|
+
|
|
87
|
+
print(f"Transfer complete: {len(datasets) - len(failed)}/{len(datasets)} datasets successful")
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
import os
|
|
2
|
+
import platform
|
|
3
|
+
|
|
4
|
+
def set_adc_credentials():
|
|
5
|
+
if platform.system() == 'Windows':
|
|
6
|
+
path = os.path.join(os.environ['APPDATA'], 'gcloud', 'application_default_credentials.json')
|
|
7
|
+
else:
|
|
8
|
+
path = os.path.expanduser('~/.config/gcloud/application_default_credentials.json')
|
|
9
|
+
os.environ['GOOGLE_APPLICATION_CREDENTIALS'] = path
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def _get_storage(gcs_credentials, session):
|
|
13
|
+
from gcloud.aio.storage import Storage
|
|
14
|
+
if 'service_file' in gcs_credentials:
|
|
15
|
+
return Storage(service_file=gcs_credentials['service_file'], session=session)
|
|
16
|
+
else:
|
|
17
|
+
set_adc_credentials()
|
|
18
|
+
return Storage(session=session)
|
|
File without changes
|
|
@@ -0,0 +1,85 @@
|
|
|
1
|
+
import asyncio
|
|
2
|
+
import aiohttp
|
|
3
|
+
import aioboto3
|
|
4
|
+
import ssl
|
|
5
|
+
import json
|
|
6
|
+
from urllib.parse import urlparse
|
|
7
|
+
from tqdm import tqdm
|
|
8
|
+
from datetime import datetime, timezone
|
|
9
|
+
|
|
10
|
+
from ..utils import _generate_dates, _get_urls
|
|
11
|
+
|
|
12
|
+
async def _transfer_file(session, s3_client, semaphore, url, bucket, prefix=None, retry_errors=3):
|
|
13
|
+
async with semaphore:
|
|
14
|
+
key = urlparse(url).path.split('/')[-1]
|
|
15
|
+
if prefix:
|
|
16
|
+
key = f"{prefix.rstrip('/')}/{key}"
|
|
17
|
+
last_error = None
|
|
18
|
+
for attempt in range(retry_errors + 1):
|
|
19
|
+
try:
|
|
20
|
+
async with session.get(url) as response:
|
|
21
|
+
if response.status != 200:
|
|
22
|
+
raise aiohttp.ClientResponseError(response.request_info, response.history, status=response.status)
|
|
23
|
+
content = await response.read()
|
|
24
|
+
await s3_client.put_object(
|
|
25
|
+
Bucket=bucket, Key=key, Body=content,
|
|
26
|
+
ContentType=response.headers.get('Content-Type', 'application/octet-stream'),
|
|
27
|
+
Metadata={'source-url': url, 'transfer-date': datetime.now(timezone.utc).isoformat()}
|
|
28
|
+
)
|
|
29
|
+
return {'success': True, 'url': url, 'size_bytes': len(content)}
|
|
30
|
+
except Exception as e:
|
|
31
|
+
if attempt < retry_errors:
|
|
32
|
+
await asyncio.sleep(2 ** attempt)
|
|
33
|
+
last_error = e
|
|
34
|
+
return {'success': False, 'url': url, 'error': str(last_error)}
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
async def _transfer_urls(urls, s3_credentials, max_workers, retry_errors, prefix=None):
|
|
38
|
+
connector = aiohttp.TCPConnector(limit=max_workers, ssl=ssl.create_default_context(), ttl_dns_cache=300)
|
|
39
|
+
async with aiohttp.ClientSession(connector=connector, timeout=aiohttp.ClientTimeout(total=7200)) as session:
|
|
40
|
+
async with aioboto3.Session().client(
|
|
41
|
+
's3',
|
|
42
|
+
aws_access_key_id=s3_credentials['aws_access_key_id'],
|
|
43
|
+
aws_secret_access_key=s3_credentials['aws_secret_access_key'],
|
|
44
|
+
region_name=s3_credentials['region_name']
|
|
45
|
+
) as s3_client:
|
|
46
|
+
semaphore = asyncio.Semaphore(max_workers)
|
|
47
|
+
tasks = [_transfer_file(session, s3_client, semaphore, url, s3_credentials['bucket_name'], prefix, retry_errors) for url in urls]
|
|
48
|
+
|
|
49
|
+
failed, total_bytes = [], 0
|
|
50
|
+
with tqdm(total=len(urls), desc="Transferring files", unit="file") as pbar:
|
|
51
|
+
for coro in asyncio.as_completed(tasks):
|
|
52
|
+
result = await coro
|
|
53
|
+
if result['success']:
|
|
54
|
+
total_bytes += result['size_bytes']
|
|
55
|
+
else:
|
|
56
|
+
failed.append(result)
|
|
57
|
+
pbar.set_postfix({'Total': f"{total_bytes / (1024**3):.2f} GB"})
|
|
58
|
+
pbar.update(1)
|
|
59
|
+
|
|
60
|
+
return failed
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def bucket_transfer(datamule_bucket, s3_credentials, max_workers=4, errors_json_filename='errors_bucket_transfer.json',
|
|
64
|
+
retry_errors=3, force_daily=True, cik=None, submission_type=None, filing_date=None, accession_number=None, prefix=None):
|
|
65
|
+
|
|
66
|
+
if datamule_bucket not in ['filings_sgml_r2', 'sec_filings_sgml_r2']:
|
|
67
|
+
raise ValueError('Datamule S3 bucket not found.')
|
|
68
|
+
|
|
69
|
+
if accession_number is not None and any(p is not None for p in [cik, submission_type, filing_date]):
|
|
70
|
+
raise ValueError('If accession_number is provided, cik, submission_type, and filing_date must be None.')
|
|
71
|
+
|
|
72
|
+
dates = _generate_dates(filing_date) if force_daily and filing_date else [filing_date]
|
|
73
|
+
|
|
74
|
+
for date in dates:
|
|
75
|
+
if len(dates) > 1:
|
|
76
|
+
print(f"Transferring {date}")
|
|
77
|
+
urls = _get_urls(submission_type=submission_type, cik=cik, filing_date=date, accession_number=accession_number)
|
|
78
|
+
failed = asyncio.run(_transfer_urls(urls, s3_credentials, max_workers, retry_errors, prefix=prefix))
|
|
79
|
+
|
|
80
|
+
if failed and errors_json_filename:
|
|
81
|
+
with open(errors_json_filename, 'w') as f:
|
|
82
|
+
json.dump(failed, f, indent=2)
|
|
83
|
+
print(f"Saved {len(failed)} errors to {errors_json_filename}")
|
|
84
|
+
|
|
85
|
+
print(f"Transfer complete: {len(urls) - len(failed)}/{len(urls)} files successful")
|
|
@@ -0,0 +1,142 @@
|
|
|
1
|
+
import asyncio
|
|
2
|
+
import aiohttp
|
|
3
|
+
import aioboto3
|
|
4
|
+
import ssl
|
|
5
|
+
import json
|
|
6
|
+
import urllib
|
|
7
|
+
from tqdm import tqdm
|
|
8
|
+
from datetime import datetime, timezone
|
|
9
|
+
|
|
10
|
+
from ...api_key import get_api_key
|
|
11
|
+
from ...v3.datasets import resolve_path
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
async def _get_dataset_url(session, dataset, api_key=None):
|
|
15
|
+
object_key = resolve_path(dataset)
|
|
16
|
+
async with session.post(
|
|
17
|
+
"https://api.datamule.xyz/v3/get-s3-link",
|
|
18
|
+
json={"path": object_key},
|
|
19
|
+
headers={
|
|
20
|
+
"Authorization": f"Bearer {get_api_key(api_key)}",
|
|
21
|
+
"User-Agent": "datamule-hub",
|
|
22
|
+
},
|
|
23
|
+
) as response:
|
|
24
|
+
data = await response.json()
|
|
25
|
+
if not data.get('success'):
|
|
26
|
+
raise Exception(f"API error: {data.get('error', 'Unknown error')}")
|
|
27
|
+
billing = data.get('metadata', {}).get('billing', {})
|
|
28
|
+
return data['data']['download_url'], data['data']['size_gb'], billing, data['data'].get('object_key', object_key)
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
async def _transfer_dataset(session, s3_client, semaphore, dataset, bucket, prefix=None, retry_errors=3, multipart_threshold_mb=100, chunk_size_mb=8):
|
|
32
|
+
download_url, size_gb, billing, object_key = await _get_dataset_url(session, dataset)
|
|
33
|
+
|
|
34
|
+
key = object_key.strip("/") or f"{dataset}.download"
|
|
35
|
+
if prefix:
|
|
36
|
+
key = f"{prefix.rstrip('/')}/{key}"
|
|
37
|
+
|
|
38
|
+
use_multipart = size_gb * 1024 > multipart_threshold_mb
|
|
39
|
+
chunk_size = chunk_size_mb * 1024 * 1024
|
|
40
|
+
|
|
41
|
+
async with semaphore:
|
|
42
|
+
last_error = None
|
|
43
|
+
for attempt in range(retry_errors + 1):
|
|
44
|
+
mpu_id = None
|
|
45
|
+
try:
|
|
46
|
+
async with session.get(download_url) as response:
|
|
47
|
+
if response.status != 200:
|
|
48
|
+
raise aiohttp.ClientResponseError(response.request_info, response.history, status=response.status)
|
|
49
|
+
|
|
50
|
+
if use_multipart:
|
|
51
|
+
mpu = await s3_client.create_multipart_upload(
|
|
52
|
+
Bucket=bucket, Key=key,
|
|
53
|
+
ContentType=response.headers.get('Content-Type', 'application/octet-stream'),
|
|
54
|
+
Metadata={'source-url': download_url, 'transfer-date': datetime.now(timezone.utc).isoformat()}
|
|
55
|
+
)
|
|
56
|
+
mpu_id = mpu['UploadId']
|
|
57
|
+
parts = []
|
|
58
|
+
part_number = 1
|
|
59
|
+
buffer = b""
|
|
60
|
+
|
|
61
|
+
async for chunk in response.content.iter_chunked(chunk_size):
|
|
62
|
+
buffer += chunk
|
|
63
|
+
if len(buffer) >= chunk_size:
|
|
64
|
+
part = await s3_client.upload_part(
|
|
65
|
+
Bucket=bucket, Key=key,
|
|
66
|
+
PartNumber=part_number, UploadId=mpu_id, Body=buffer
|
|
67
|
+
)
|
|
68
|
+
parts.append({'PartNumber': part_number, 'ETag': part['ETag']})
|
|
69
|
+
part_number += 1
|
|
70
|
+
buffer = b""
|
|
71
|
+
|
|
72
|
+
if buffer:
|
|
73
|
+
part = await s3_client.upload_part(
|
|
74
|
+
Bucket=bucket, Key=key,
|
|
75
|
+
PartNumber=part_number, UploadId=mpu_id, Body=buffer
|
|
76
|
+
)
|
|
77
|
+
parts.append({'PartNumber': part_number, 'ETag': part['ETag']})
|
|
78
|
+
|
|
79
|
+
await s3_client.complete_multipart_upload(
|
|
80
|
+
Bucket=bucket, Key=key,
|
|
81
|
+
UploadId=mpu_id, MultipartUpload={'Parts': parts}
|
|
82
|
+
)
|
|
83
|
+
|
|
84
|
+
else:
|
|
85
|
+
content = await response.read()
|
|
86
|
+
await s3_client.put_object(
|
|
87
|
+
Bucket=bucket, Key=key, Body=content,
|
|
88
|
+
ContentType=response.headers.get('Content-Type', 'application/octet-stream'),
|
|
89
|
+
Metadata={'source-url': download_url, 'transfer-date': datetime.now(timezone.utc).isoformat()}
|
|
90
|
+
)
|
|
91
|
+
|
|
92
|
+
return {'success': True, 'dataset': dataset, 'size_bytes': int(size_gb * 1024**3), 'billing': billing}
|
|
93
|
+
|
|
94
|
+
except Exception as e:
|
|
95
|
+
if mpu_id:
|
|
96
|
+
try:
|
|
97
|
+
await s3_client.abort_multipart_upload(Bucket=bucket, Key=key, UploadId=mpu_id)
|
|
98
|
+
except Exception:
|
|
99
|
+
pass
|
|
100
|
+
if attempt < retry_errors:
|
|
101
|
+
download_url, _, _, _ = await _get_dataset_url(session, dataset)
|
|
102
|
+
await asyncio.sleep(2 ** attempt)
|
|
103
|
+
last_error = e
|
|
104
|
+
|
|
105
|
+
return {'success': False, 'dataset': dataset, 'error': str(last_error)}
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
async def _transfer_datasets(datasets, s3_credentials, max_workers, retry_errors, prefix=None):
|
|
109
|
+
connector = aiohttp.TCPConnector(limit=max_workers, ssl=ssl.create_default_context())
|
|
110
|
+
async with aiohttp.ClientSession(connector=connector, timeout=aiohttp.ClientTimeout(total=7200)) as session:
|
|
111
|
+
async with aioboto3.Session().client(
|
|
112
|
+
's3',
|
|
113
|
+
aws_access_key_id=s3_credentials['aws_access_key_id'],
|
|
114
|
+
aws_secret_access_key=s3_credentials['aws_secret_access_key'],
|
|
115
|
+
region_name=s3_credentials['region_name']
|
|
116
|
+
) as s3_client:
|
|
117
|
+
semaphore = asyncio.Semaphore(max_workers)
|
|
118
|
+
tasks = [_transfer_dataset(session, s3_client, semaphore, d, s3_credentials['bucket_name'], prefix, retry_errors) for d in datasets]
|
|
119
|
+
|
|
120
|
+
failed, total_bytes = [], 0
|
|
121
|
+
with tqdm(total=len(datasets), desc="Transferring datasets", unit="dataset") as pbar:
|
|
122
|
+
for coro in asyncio.as_completed(tasks):
|
|
123
|
+
result = await coro
|
|
124
|
+
if result['success']:
|
|
125
|
+
total_bytes += result['size_bytes']
|
|
126
|
+
else:
|
|
127
|
+
failed.append(result)
|
|
128
|
+
pbar.set_postfix({'Total': f"{total_bytes / (1024**3):.2f} GB"})
|
|
129
|
+
pbar.update(1)
|
|
130
|
+
|
|
131
|
+
return failed
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
def datasets_transfer(datasets, s3_credentials, max_workers=4, errors_json_filename='errors_datasets_transfer.json', retry_errors=3, prefix=None):
|
|
135
|
+
failed = asyncio.run(_transfer_datasets(datasets, s3_credentials, max_workers, retry_errors, prefix=prefix))
|
|
136
|
+
|
|
137
|
+
if failed and errors_json_filename:
|
|
138
|
+
with open(errors_json_filename, 'w') as f:
|
|
139
|
+
json.dump(failed, f, indent=2)
|
|
140
|
+
print(f"Saved {len(failed)} errors to {errors_json_filename}")
|
|
141
|
+
|
|
142
|
+
print(f"Transfer complete: {len(datasets) - len(failed)}/{len(datasets)} datasets successful")
|
|
@@ -0,0 +1,63 @@
|
|
|
1
|
+
from datetime import datetime, timedelta
|
|
2
|
+
from ..v3.databases import read_query
|
|
3
|
+
from ..utils.format_accession import format_accession
|
|
4
|
+
|
|
5
|
+
def _generate_dates(filing_date):
|
|
6
|
+
if isinstance(filing_date, str):
|
|
7
|
+
return [filing_date]
|
|
8
|
+
elif isinstance(filing_date, list):
|
|
9
|
+
return filing_date
|
|
10
|
+
elif isinstance(filing_date, tuple):
|
|
11
|
+
start = datetime.strptime(filing_date[0], '%Y-%m-%d')
|
|
12
|
+
end = datetime.strptime(filing_date[1], '%Y-%m-%d')
|
|
13
|
+
dates = []
|
|
14
|
+
current = start
|
|
15
|
+
while current <= end:
|
|
16
|
+
dates.append(current.strftime('%Y-%m-%d'))
|
|
17
|
+
current += timedelta(days=1)
|
|
18
|
+
return dates
|
|
19
|
+
raise ValueError('filing_date must be a string, list, or (start, end) tuple')
|
|
20
|
+
|
|
21
|
+
def _sql_literal(value):
|
|
22
|
+
return "'" + str(value).replace("'", "''") + "'"
|
|
23
|
+
|
|
24
|
+
def _normalize_accession(value):
|
|
25
|
+
return str(value).replace("-", "")
|
|
26
|
+
|
|
27
|
+
def _sql_value(value, quote=True):
|
|
28
|
+
return _sql_literal(value) if quote else str(value)
|
|
29
|
+
|
|
30
|
+
def _sql_filter(column, value, transform=None, quote=True):
|
|
31
|
+
if value is None:
|
|
32
|
+
return None
|
|
33
|
+
|
|
34
|
+
if isinstance(value, tuple):
|
|
35
|
+
start = transform(value[0]) if transform else value[0]
|
|
36
|
+
end = transform(value[1]) if transform else value[1]
|
|
37
|
+
return f"{column} BETWEEN {_sql_value(start, quote)} AND {_sql_value(end, quote)}"
|
|
38
|
+
|
|
39
|
+
if isinstance(value, (list, set)):
|
|
40
|
+
values = [transform(item) if transform else item for item in value]
|
|
41
|
+
return f"{column} IN ({', '.join(_sql_value(item, quote) for item in values)})"
|
|
42
|
+
|
|
43
|
+
value = transform(value) if transform else value
|
|
44
|
+
return f"{column} = {_sql_value(value, quote)}"
|
|
45
|
+
|
|
46
|
+
def _get_urls(submission_type=None, cik=None, filing_date=None, accession_number=None):
|
|
47
|
+
filters = [
|
|
48
|
+
_sql_filter("cik", cik, quote=False),
|
|
49
|
+
_sql_filter("form", submission_type),
|
|
50
|
+
_sql_filter("filingdate", filing_date),
|
|
51
|
+
_sql_filter("accessionnumber", accession_number, _normalize_accession, quote=False),
|
|
52
|
+
]
|
|
53
|
+
where = " AND ".join(item for item in filters if item)
|
|
54
|
+
if where:
|
|
55
|
+
where = f"WHERE {where}"
|
|
56
|
+
|
|
57
|
+
rows = read_query(f"""
|
|
58
|
+
SELECT accessionnumber
|
|
59
|
+
FROM submissions_metadata
|
|
60
|
+
{where}
|
|
61
|
+
ORDER BY filingdate, accessionnumber
|
|
62
|
+
""")
|
|
63
|
+
return [f"https://sec-library.datamule.xyz/{format_accession(row['accessionnumber'], 'no-dash')}.sgml" for row in rows]
|
|
File without changes
|
|
@@ -0,0 +1,20 @@
|
|
|
1
|
+
def format_accession(accession, format):
|
|
2
|
+
if format == 'int':
|
|
3
|
+
accession = int(str(accession).replace('-',''))
|
|
4
|
+
elif format == 'dash':
|
|
5
|
+
accession = str(int(str(accession).replace('-',''))).zfill(18)
|
|
6
|
+
accession = f"{accession[:10]}-{accession[10:12]}-{accession[12:]}"
|
|
7
|
+
elif format == 'no-dash':
|
|
8
|
+
accession = str(int(str(accession).replace('-',''))).zfill(18)
|
|
9
|
+
else:
|
|
10
|
+
raise ValueError("unrecognized format")
|
|
11
|
+
return accession
|
|
12
|
+
|
|
13
|
+
def detect_accession_type(accession):
|
|
14
|
+
accession = str(accession)
|
|
15
|
+
if '-' in accession:
|
|
16
|
+
return 'dash'
|
|
17
|
+
elif len(accession) == 18:
|
|
18
|
+
return 'no-dash'
|
|
19
|
+
else:
|
|
20
|
+
return 'int'
|
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
from . import databases
|
|
2
|
+
from . import datasets
|
|
3
|
+
from . import sec_filings_archive
|
|
4
|
+
from . import sec_filings_lookup
|
|
5
|
+
from . import sec_filings_notifications
|
|
6
|
+
|
|
7
|
+
__all__ = [
|
|
8
|
+
"databases",
|
|
9
|
+
"datasets",
|
|
10
|
+
"sec_filings_archive",
|
|
11
|
+
"sec_filings_lookup",
|
|
12
|
+
"sec_filings_notifications",
|
|
13
|
+
]
|