datamule-hub 0.2.3__py3-none-win_amd64.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (35) hide show
  1. datamule_hub-0.2.3.dist-info/METADATA +23 -0
  2. datamule_hub-0.2.3.dist-info/RECORD +35 -0
  3. datamule_hub-0.2.3.dist-info/WHEEL +5 -0
  4. datamule_hub-0.2.3.dist-info/licenses/LICENSE +7 -0
  5. datamule_hub-0.2.3.dist-info/top_level.txt +1 -0
  6. datamulehub/__init__.py +16 -0
  7. datamulehub/api_key.py +10 -0
  8. datamulehub/bin/datamule-archive-downloader.exe +0 -0
  9. datamulehub/object_transfer/__init__.py +11 -0
  10. datamulehub/object_transfer/gcs/__init__.py +0 -0
  11. datamulehub/object_transfer/gcs/bucket_transfer.py +80 -0
  12. datamulehub/object_transfer/gcs/datasets_transfer.py +87 -0
  13. datamulehub/object_transfer/gcs/utils.py +18 -0
  14. datamulehub/object_transfer/s3/__init__.py +0 -0
  15. datamulehub/object_transfer/s3/bucket_transfer.py +85 -0
  16. datamulehub/object_transfer/s3/datasets_transfer.py +142 -0
  17. datamulehub/object_transfer/utils.py +63 -0
  18. datamulehub/utils/__init__.py +0 -0
  19. datamulehub/utils/format_accession.py +20 -0
  20. datamulehub/v3/__init__.py +13 -0
  21. datamulehub/v3/databases/__init__.py +3 -0
  22. datamulehub/v3/databases/athena.py +231 -0
  23. datamulehub/v3/databases/fastest.py +312 -0
  24. datamulehub/v3/datasets/__init__.py +3 -0
  25. datamulehub/v3/datasets/datasets.py +161 -0
  26. datamulehub/v3/sec_filings_archive/__init__.py +4 -0
  27. datamulehub/v3/sec_filings_archive/_rust.py +110 -0
  28. datamulehub/v3/sec_filings_archive/sgml.py +288 -0
  29. datamulehub/v3/sec_filings_archive/tar.py +806 -0
  30. datamulehub/v3/sec_filings_archive/utils.py +136 -0
  31. datamulehub/v3/sec_filings_lookup/__init__.py +3 -0
  32. datamulehub/v3/sec_filings_lookup/sec_filings_lookup.py +356 -0
  33. datamulehub/v3/sec_filings_notifications/__init__.py +10 -0
  34. datamulehub/v3/sec_filings_notifications/webhooks.py +102 -0
  35. datamulehub/v3/sec_filings_notifications/websocket.py +391 -0
@@ -0,0 +1,23 @@
1
+ Metadata-Version: 2.4
2
+ Name: datamule-hub
3
+ Version: 0.2.3
4
+ Summary: Access Datamule cloud
5
+ Home-page: https://github.com/john-friedman/datamule-hub
6
+ Author: John Friedman
7
+ Requires-Python: >=3.9
8
+ License-File: LICENSE
9
+ Requires-Dist: tqdm
10
+ Requires-Dist: aiohttp
11
+ Requires-Dist: aioboto3
12
+ Requires-Dist: gcloud-aio-storage
13
+ Requires-Dist: google-auth
14
+ Requires-Dist: google-cloud-storage
15
+ Requires-Dist: pyarrow
16
+ Requires-Dist: websocket-client
17
+ Requires-Dist: zstandard
18
+ Dynamic: author
19
+ Dynamic: home-page
20
+ Dynamic: license-file
21
+ Dynamic: requires-dist
22
+ Dynamic: requires-python
23
+ Dynamic: summary
@@ -0,0 +1,35 @@
1
+ datamule_hub-0.2.3.dist-info/licenses/LICENSE,sha256=e4vO5__LjkwRec1EOexxayzJEymAkwH9gtWEpkF2oRY,1066
2
+ datamulehub/__init__.py,sha256=1San9XeBEKF6Fsp8tBr2zs0VU-1W-w9mERjH8YmJHwc,368
3
+ datamulehub/api_key.py,sha256=ku5HQhtnORtomdka6uPdij3H5INiQZHj2XdNDfYZeEE,270
4
+ datamulehub/bin/datamule-archive-downloader.exe,sha256=qifj77g4rOtq5AS71Bv8MRV7Y7d44q6FcNUYrROGWDk,4059648
5
+ datamulehub/object_transfer/__init__.py,sha256=wy4EgEwMDvDZWv0T4tlTfNvKxfleDsNLs2n5pWEY8nE,432
6
+ datamulehub/object_transfer/utils.py,sha256=2nRtl9LSfkPnWHNmRWaLddEycz6S-zTFtj55z2Gc9bo,2442
7
+ datamulehub/object_transfer/gcs/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
8
+ datamulehub/object_transfer/gcs/bucket_transfer.py,sha256=OxgCL2ccmh-CKvDf5jDefnYyiu3PcLKmv5MItIDIDVc,4003
9
+ datamulehub/object_transfer/gcs/datasets_transfer.py,sha256=8ZpXRMrW4JC49xVCvi7po1zCez75PNAzuNxaSSFuA70,3656
10
+ datamulehub/object_transfer/gcs/utils.py,sha256=yOQZJoOpYphoSy3YKQXtNkW-nkPoJQzM3VUzi66SGo0,664
11
+ datamulehub/object_transfer/s3/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
12
+ datamulehub/object_transfer/s3/bucket_transfer.py,sha256=t79qKzew2yHGIzq1LeemOl620LBTFSSDQo2QUy-YUUs,4251
13
+ datamulehub/object_transfer/s3/datasets_transfer.py,sha256=r6a54lyzACuqH69SPKOSJNS8FOpZme2BjkzH-HJCTAU,6737
14
+ datamulehub/utils/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
15
+ datamulehub/utils/format_accession.py,sha256=Es_Y48lOq_Z_tByDbWTifcO3-1GA4izNAA2_nFIgAy8,697
16
+ datamulehub/v3/__init__.py,sha256=9kZ5EV1j8xnW1yW8jvh9ho3xU7vu8dt8vCt66d6SMtk,301
17
+ datamulehub/v3/databases/__init__.py,sha256=4rIDESjUEJpUWDMRFpYyKDo2D66zJOQtYKRlq_CE0aQ,76
18
+ datamulehub/v3/databases/athena.py,sha256=fSuzORl4pon8BCfG2IKxy40fEF8tu4rwZ91ZP2VXvWE,7247
19
+ datamulehub/v3/databases/fastest.py,sha256=JsfGwcdTeMQgGck8mfL15-GvC3xvp5Z5Q_XB-4LV4S0,10113
20
+ datamulehub/v3/datasets/__init__.py,sha256=Z2TaPOCpIjWYY6hv7Abi-uplUk27fMBxDcXnjg9bk_U,110
21
+ datamulehub/v3/datasets/datasets.py,sha256=f0GylDgsT9yoKXAoIiNpes2vP4TRccVXCZrSKl2S_io,5564
22
+ datamulehub/v3/sec_filings_archive/__init__.py,sha256=VxEAv15gqJ3LfUbeaALE7We0lU8Vwc-SuSaBSi0HTKM,111
23
+ datamulehub/v3/sec_filings_archive/_rust.py,sha256=Wvk3abMqRnzdQxuVijyOVB4KfSNsMBkPseRxKdzMrWY,3257
24
+ datamulehub/v3/sec_filings_archive/sgml.py,sha256=fVNgoRK6nr8xXcvXovgXx4HAxTaS4wIuLuRVJdjzGbo,9649
25
+ datamulehub/v3/sec_filings_archive/tar.py,sha256=wkENjqweIcbjnbjRYAFv0PHTW9Cy2Om_82GQlCS3oXs,26742
26
+ datamulehub/v3/sec_filings_archive/utils.py,sha256=6SYP-O5Kok-4BAASEjOCZwLgHAuL4m_g5VCE1lDUgMQ,4427
27
+ datamulehub/v3/sec_filings_lookup/__init__.py,sha256=fjAgAQwr8Jjx1VV9zqx6fW0uCLlg-2wJzFYV-1qhY-Y,154
28
+ datamulehub/v3/sec_filings_lookup/sec_filings_lookup.py,sha256=abxWSfquspOMqb0PzLajrpeGsb2CjjphdV-aGbXkrwY,9665
29
+ datamulehub/v3/sec_filings_notifications/__init__.py,sha256=M2a6PEnwGqgr6I3j6levqC44dWKkFpLXQQtcbhspJgs,272
30
+ datamulehub/v3/sec_filings_notifications/webhooks.py,sha256=MLhMuYc88YyQNDhMTsBI_q7njaKVn6D5Bqp0dqU9JTQ,2991
31
+ datamulehub/v3/sec_filings_notifications/websocket.py,sha256=vHrsvWQqodc7ApVVpkzVHWfaNBZXSMdXQhF-V9AX0Ac,12176
32
+ datamule_hub-0.2.3.dist-info/METADATA,sha256=0ppVioqPa7e7ZjUcppCsWv74SngoeIxDsFmsq-3X844,600
33
+ datamule_hub-0.2.3.dist-info/WHEEL,sha256=Zx98gwb_dQKckJK3HEOVKnxxD52EfU_PXDG_usMJ2ng,98
34
+ datamule_hub-0.2.3.dist-info/top_level.txt,sha256=aef5Z5BB_fOwDXmIWr9yJ1zncoNVce6MWgNwtaxrUSk,12
35
+ datamule_hub-0.2.3.dist-info/RECORD,,
@@ -0,0 +1,5 @@
1
+ Wheel-Version: 1.0
2
+ Generator: setuptools (84.0.0)
3
+ Root-Is-Purelib: false
4
+ Tag: py3-none-win_amd64
5
+
@@ -0,0 +1,7 @@
1
+ Copyright 2026 John Friedman
2
+
3
+ Permission is hereby granted, free of charge, to any person obtaining a copy of this software and associated documentation files (the “Software”), to deal in the Software without restriction, including without limitation the rights to use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the Software, and to permit persons to whom the Software is furnished to do so, subject to the following conditions:
4
+
5
+ The above copyright notice and this permission notice shall be included in all copies or substantial portions of the Software.
6
+
7
+ THE SOFTWARE IS PROVIDED “AS IS”, WITHOUT WARRANTY OF ANY KIND, EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
@@ -0,0 +1 @@
1
+ datamulehub
@@ -0,0 +1,16 @@
1
+ from . import object_transfer
2
+ from .v3 import databases
3
+ from .v3 import datasets
4
+ from .v3 import sec_filings_archive
5
+ from .v3 import sec_filings_lookup
6
+ from .v3 import sec_filings_notifications
7
+
8
+
9
+ __all__ = [
10
+ "databases",
11
+ "datasets",
12
+ "object_transfer",
13
+ "sec_filings_archive",
14
+ "sec_filings_lookup",
15
+ "sec_filings_notifications",
16
+ ]
datamulehub/api_key.py ADDED
@@ -0,0 +1,10 @@
1
+ import os
2
+
3
+ def get_api_key(api_key=None):
4
+ key = api_key or os.environ.get('DATAMULE_API_KEY')
5
+ if not key:
6
+ raise EnvironmentError("DATAMULE_API_KEY environment variable is not set.")
7
+ return key
8
+
9
+
10
+ api_key = os.environ.get('DATAMULE_API_KEY')
@@ -0,0 +1,11 @@
1
+ from .gcs.bucket_transfer import bucket_transfer as gcs_archive_transfer
2
+ from .gcs.datasets_transfer import datasets_transfer as gcs_dataset_transfer
3
+ from .s3.bucket_transfer import bucket_transfer as s3_archive_transfer
4
+ from .s3.datasets_transfer import datasets_transfer as s3_dataset_transfer
5
+
6
+ __all__ = [
7
+ "gcs_archive_transfer",
8
+ "gcs_dataset_transfer",
9
+ "s3_archive_transfer",
10
+ "s3_dataset_transfer",
11
+ ]
File without changes
@@ -0,0 +1,80 @@
1
+ import asyncio
2
+ import aiohttp
3
+ import ssl
4
+ import json
5
+ from urllib.parse import urlparse
6
+ from tqdm import tqdm
7
+ from datetime import datetime, timezone
8
+
9
+ from ..utils import _generate_dates, _get_urls
10
+ from .utils import _get_storage
11
+
12
+
13
+ async def _transfer_file(session, storage, semaphore, url, bucket, prefix=None, retry_errors=3):
14
+ async with semaphore:
15
+ key = urlparse(url).path.split('/')[-1]
16
+ if prefix:
17
+ key = f"{prefix.rstrip('/')}/{key}"
18
+ last_error = None
19
+ for attempt in range(retry_errors + 1):
20
+ try:
21
+ async with session.get(url) as response:
22
+ if response.status != 200:
23
+ raise aiohttp.ClientResponseError(response.request_info, response.history, status=response.status)
24
+ content = await response.read()
25
+ await storage.upload(bucket, key, content,
26
+ content_type=response.headers.get('Content-Type', 'application/octet-stream'),
27
+ metadata={'source-url': url, 'transfer-date': datetime.now(timezone.utc).isoformat()}
28
+ )
29
+ return {'success': True, 'url': url, 'size_bytes': len(content)}
30
+ except Exception as e:
31
+ if attempt < retry_errors:
32
+ await asyncio.sleep(2 ** attempt)
33
+ last_error = e
34
+ return {'success': False, 'url': url, 'error': str(last_error)}
35
+
36
+
37
+ async def _transfer_urls(urls, gcs_credentials, max_workers, retry_errors, prefix=None):
38
+ connector = aiohttp.TCPConnector(limit=max_workers, ssl=ssl.create_default_context(), ttl_dns_cache=300)
39
+ async with aiohttp.ClientSession(connector=connector, timeout=aiohttp.ClientTimeout(total=7200)) as session:
40
+ async with _get_storage(gcs_credentials, session) as storage:
41
+ semaphore = asyncio.Semaphore(max_workers)
42
+ tasks = [_transfer_file(session, storage, semaphore, url, gcs_credentials['bucket_name'], prefix, retry_errors) for url in urls]
43
+
44
+ failed, total_bytes = [], 0
45
+ with tqdm(total=len(urls), desc="Transferring files", unit="file") as pbar:
46
+ for coro in asyncio.as_completed(tasks):
47
+ result = await coro
48
+ if result['success']:
49
+ total_bytes += result['size_bytes']
50
+ else:
51
+ failed.append(result)
52
+ pbar.set_postfix({'Total': f"{total_bytes / (1024**3):.2f} GB"})
53
+ pbar.update(1)
54
+
55
+ return failed
56
+
57
+
58
+ def bucket_transfer(datamule_bucket, gcs_credentials, max_workers=4, errors_json_filename='errors_bucket_transfer.json',
59
+ retry_errors=3, force_daily=True, cik=None, submission_type=None, filing_date=None, accession_number=None, prefix=None):
60
+
61
+ if datamule_bucket not in ['filings_sgml_r2', 'sec_filings_sgml_r2']:
62
+ raise ValueError('Datamule S3 bucket not found.')
63
+
64
+ if accession_number is not None and any(p is not None for p in [cik, submission_type, filing_date]):
65
+ raise ValueError('If accession_number is provided, cik, submission_type, and filing_date must be None.')
66
+
67
+ dates = _generate_dates(filing_date) if force_daily and filing_date else [filing_date]
68
+
69
+ for date in dates:
70
+ if len(dates) > 1:
71
+ print(f"Transferring {date}")
72
+ urls = _get_urls(submission_type=submission_type, cik=cik, filing_date=date, accession_number=accession_number)
73
+ failed = asyncio.run(_transfer_urls(urls, gcs_credentials, max_workers, retry_errors, prefix=prefix))
74
+
75
+ if failed and errors_json_filename:
76
+ with open(errors_json_filename, 'w') as f:
77
+ json.dump(failed, f, indent=2)
78
+ print(f"Saved {len(failed)} errors to {errors_json_filename}")
79
+
80
+ print(f"Transfer complete: {len(urls) - len(failed)}/{len(urls)} files successful")
@@ -0,0 +1,87 @@
1
+ import json
2
+ import urllib
3
+ import urllib.request
4
+ from tqdm import tqdm
5
+ from datetime import datetime, timezone
6
+ from google.cloud import storage as gcs
7
+
8
+ from ...api_key import get_api_key
9
+ from ...v3.datasets import resolve_path
10
+
11
+
12
+ def _get_dataset_url(dataset, api_key=None):
13
+ object_key = resolve_path(dataset)
14
+ body = json.dumps({"path": object_key}).encode("utf-8")
15
+ request = urllib.request.Request(
16
+ "https://api.datamule.xyz/v3/get-s3-link",
17
+ data=body,
18
+ headers={
19
+ "Authorization": f"Bearer {get_api_key(api_key)}",
20
+ "Content-Type": "application/json",
21
+ "User-Agent": "datamule-hub",
22
+ },
23
+ method="POST",
24
+ )
25
+ with urllib.request.urlopen(request) as response:
26
+ data = json.loads(response.read().decode('utf-8'))
27
+ if not data.get('success'):
28
+ raise Exception(f"API error: {data.get('error', 'Unknown error')}")
29
+ billing = data.get('metadata', {}).get('billing', {})
30
+ return data['data']['download_url'], data['data']['size_gb'], billing, data['data'].get('object_key', object_key)
31
+
32
+
33
+ def _get_gcs_client(gcs_credentials):
34
+ if 'service_file' in gcs_credentials:
35
+ return gcs.Client.from_service_account_json(gcs_credentials['service_file'])
36
+ return gcs.Client()
37
+
38
+
39
+ def _transfer_dataset(client, bucket, dataset, prefix=None, retry_errors=3):
40
+ download_url, size_gb, billing, object_key = _get_dataset_url(dataset)
41
+
42
+ key = object_key.strip("/") or f"{dataset}.download"
43
+ if prefix:
44
+ key = f"{prefix.rstrip('/')}/{key}"
45
+
46
+ last_error = None
47
+ for attempt in range(retry_errors + 1):
48
+ try:
49
+ request = urllib.request.Request(download_url, headers={'User-Agent': 'datamule-python'})
50
+ with urllib.request.urlopen(request) as response:
51
+ content_type = response.headers.get('Content-Type', 'application/octet-stream')
52
+ blob = bucket.blob(key)
53
+ blob.metadata = {
54
+ 'source-url': download_url,
55
+ 'transfer-date': datetime.now(timezone.utc).isoformat(),
56
+ }
57
+ blob.upload_from_file(response, content_type=content_type)
58
+ return {'success': True, 'dataset': dataset, 'size_bytes': blob.size, 'billing': billing}
59
+ except Exception as e:
60
+ if attempt < retry_errors:
61
+ download_url, _, _, _ = _get_dataset_url(dataset)
62
+ last_error = e
63
+
64
+ return {'success': False, 'dataset': dataset, 'error': str(last_error)}
65
+
66
+
67
+ def datasets_transfer(datasets, gcs_credentials, errors_json_filename='errors_datasets_transfer.json', retry_errors=3, prefix=None):
68
+ client = _get_gcs_client(gcs_credentials)
69
+ bucket = client.bucket(gcs_credentials['bucket_name'])
70
+
71
+ failed, total_bytes = [], 0
72
+ with tqdm(total=len(datasets), desc="Transferring datasets", unit="dataset") as pbar:
73
+ for dataset in datasets:
74
+ result = _transfer_dataset(client, bucket, dataset, prefix, retry_errors)
75
+ if result['success']:
76
+ total_bytes += result.get('size_bytes', 0)
77
+ else:
78
+ failed.append(result)
79
+ pbar.set_postfix({'Total': f"{total_bytes / (1024**3):.2f} GB"})
80
+ pbar.update(1)
81
+
82
+ if failed and errors_json_filename:
83
+ with open(errors_json_filename, 'w') as f:
84
+ json.dump(failed, f, indent=2)
85
+ print(f"Saved {len(failed)} errors to {errors_json_filename}")
86
+
87
+ print(f"Transfer complete: {len(datasets) - len(failed)}/{len(datasets)} datasets successful")
@@ -0,0 +1,18 @@
1
+ import os
2
+ import platform
3
+
4
+ def set_adc_credentials():
5
+ if platform.system() == 'Windows':
6
+ path = os.path.join(os.environ['APPDATA'], 'gcloud', 'application_default_credentials.json')
7
+ else:
8
+ path = os.path.expanduser('~/.config/gcloud/application_default_credentials.json')
9
+ os.environ['GOOGLE_APPLICATION_CREDENTIALS'] = path
10
+
11
+
12
+ def _get_storage(gcs_credentials, session):
13
+ from gcloud.aio.storage import Storage
14
+ if 'service_file' in gcs_credentials:
15
+ return Storage(service_file=gcs_credentials['service_file'], session=session)
16
+ else:
17
+ set_adc_credentials()
18
+ return Storage(session=session)
File without changes
@@ -0,0 +1,85 @@
1
+ import asyncio
2
+ import aiohttp
3
+ import aioboto3
4
+ import ssl
5
+ import json
6
+ from urllib.parse import urlparse
7
+ from tqdm import tqdm
8
+ from datetime import datetime, timezone
9
+
10
+ from ..utils import _generate_dates, _get_urls
11
+
12
+ async def _transfer_file(session, s3_client, semaphore, url, bucket, prefix=None, retry_errors=3):
13
+ async with semaphore:
14
+ key = urlparse(url).path.split('/')[-1]
15
+ if prefix:
16
+ key = f"{prefix.rstrip('/')}/{key}"
17
+ last_error = None
18
+ for attempt in range(retry_errors + 1):
19
+ try:
20
+ async with session.get(url) as response:
21
+ if response.status != 200:
22
+ raise aiohttp.ClientResponseError(response.request_info, response.history, status=response.status)
23
+ content = await response.read()
24
+ await s3_client.put_object(
25
+ Bucket=bucket, Key=key, Body=content,
26
+ ContentType=response.headers.get('Content-Type', 'application/octet-stream'),
27
+ Metadata={'source-url': url, 'transfer-date': datetime.now(timezone.utc).isoformat()}
28
+ )
29
+ return {'success': True, 'url': url, 'size_bytes': len(content)}
30
+ except Exception as e:
31
+ if attempt < retry_errors:
32
+ await asyncio.sleep(2 ** attempt)
33
+ last_error = e
34
+ return {'success': False, 'url': url, 'error': str(last_error)}
35
+
36
+
37
+ async def _transfer_urls(urls, s3_credentials, max_workers, retry_errors, prefix=None):
38
+ connector = aiohttp.TCPConnector(limit=max_workers, ssl=ssl.create_default_context(), ttl_dns_cache=300)
39
+ async with aiohttp.ClientSession(connector=connector, timeout=aiohttp.ClientTimeout(total=7200)) as session:
40
+ async with aioboto3.Session().client(
41
+ 's3',
42
+ aws_access_key_id=s3_credentials['aws_access_key_id'],
43
+ aws_secret_access_key=s3_credentials['aws_secret_access_key'],
44
+ region_name=s3_credentials['region_name']
45
+ ) as s3_client:
46
+ semaphore = asyncio.Semaphore(max_workers)
47
+ tasks = [_transfer_file(session, s3_client, semaphore, url, s3_credentials['bucket_name'], prefix, retry_errors) for url in urls]
48
+
49
+ failed, total_bytes = [], 0
50
+ with tqdm(total=len(urls), desc="Transferring files", unit="file") as pbar:
51
+ for coro in asyncio.as_completed(tasks):
52
+ result = await coro
53
+ if result['success']:
54
+ total_bytes += result['size_bytes']
55
+ else:
56
+ failed.append(result)
57
+ pbar.set_postfix({'Total': f"{total_bytes / (1024**3):.2f} GB"})
58
+ pbar.update(1)
59
+
60
+ return failed
61
+
62
+
63
+ def bucket_transfer(datamule_bucket, s3_credentials, max_workers=4, errors_json_filename='errors_bucket_transfer.json',
64
+ retry_errors=3, force_daily=True, cik=None, submission_type=None, filing_date=None, accession_number=None, prefix=None):
65
+
66
+ if datamule_bucket not in ['filings_sgml_r2', 'sec_filings_sgml_r2']:
67
+ raise ValueError('Datamule S3 bucket not found.')
68
+
69
+ if accession_number is not None and any(p is not None for p in [cik, submission_type, filing_date]):
70
+ raise ValueError('If accession_number is provided, cik, submission_type, and filing_date must be None.')
71
+
72
+ dates = _generate_dates(filing_date) if force_daily and filing_date else [filing_date]
73
+
74
+ for date in dates:
75
+ if len(dates) > 1:
76
+ print(f"Transferring {date}")
77
+ urls = _get_urls(submission_type=submission_type, cik=cik, filing_date=date, accession_number=accession_number)
78
+ failed = asyncio.run(_transfer_urls(urls, s3_credentials, max_workers, retry_errors, prefix=prefix))
79
+
80
+ if failed and errors_json_filename:
81
+ with open(errors_json_filename, 'w') as f:
82
+ json.dump(failed, f, indent=2)
83
+ print(f"Saved {len(failed)} errors to {errors_json_filename}")
84
+
85
+ print(f"Transfer complete: {len(urls) - len(failed)}/{len(urls)} files successful")
@@ -0,0 +1,142 @@
1
+ import asyncio
2
+ import aiohttp
3
+ import aioboto3
4
+ import ssl
5
+ import json
6
+ import urllib
7
+ from tqdm import tqdm
8
+ from datetime import datetime, timezone
9
+
10
+ from ...api_key import get_api_key
11
+ from ...v3.datasets import resolve_path
12
+
13
+
14
+ async def _get_dataset_url(session, dataset, api_key=None):
15
+ object_key = resolve_path(dataset)
16
+ async with session.post(
17
+ "https://api.datamule.xyz/v3/get-s3-link",
18
+ json={"path": object_key},
19
+ headers={
20
+ "Authorization": f"Bearer {get_api_key(api_key)}",
21
+ "User-Agent": "datamule-hub",
22
+ },
23
+ ) as response:
24
+ data = await response.json()
25
+ if not data.get('success'):
26
+ raise Exception(f"API error: {data.get('error', 'Unknown error')}")
27
+ billing = data.get('metadata', {}).get('billing', {})
28
+ return data['data']['download_url'], data['data']['size_gb'], billing, data['data'].get('object_key', object_key)
29
+
30
+
31
+ async def _transfer_dataset(session, s3_client, semaphore, dataset, bucket, prefix=None, retry_errors=3, multipart_threshold_mb=100, chunk_size_mb=8):
32
+ download_url, size_gb, billing, object_key = await _get_dataset_url(session, dataset)
33
+
34
+ key = object_key.strip("/") or f"{dataset}.download"
35
+ if prefix:
36
+ key = f"{prefix.rstrip('/')}/{key}"
37
+
38
+ use_multipart = size_gb * 1024 > multipart_threshold_mb
39
+ chunk_size = chunk_size_mb * 1024 * 1024
40
+
41
+ async with semaphore:
42
+ last_error = None
43
+ for attempt in range(retry_errors + 1):
44
+ mpu_id = None
45
+ try:
46
+ async with session.get(download_url) as response:
47
+ if response.status != 200:
48
+ raise aiohttp.ClientResponseError(response.request_info, response.history, status=response.status)
49
+
50
+ if use_multipart:
51
+ mpu = await s3_client.create_multipart_upload(
52
+ Bucket=bucket, Key=key,
53
+ ContentType=response.headers.get('Content-Type', 'application/octet-stream'),
54
+ Metadata={'source-url': download_url, 'transfer-date': datetime.now(timezone.utc).isoformat()}
55
+ )
56
+ mpu_id = mpu['UploadId']
57
+ parts = []
58
+ part_number = 1
59
+ buffer = b""
60
+
61
+ async for chunk in response.content.iter_chunked(chunk_size):
62
+ buffer += chunk
63
+ if len(buffer) >= chunk_size:
64
+ part = await s3_client.upload_part(
65
+ Bucket=bucket, Key=key,
66
+ PartNumber=part_number, UploadId=mpu_id, Body=buffer
67
+ )
68
+ parts.append({'PartNumber': part_number, 'ETag': part['ETag']})
69
+ part_number += 1
70
+ buffer = b""
71
+
72
+ if buffer:
73
+ part = await s3_client.upload_part(
74
+ Bucket=bucket, Key=key,
75
+ PartNumber=part_number, UploadId=mpu_id, Body=buffer
76
+ )
77
+ parts.append({'PartNumber': part_number, 'ETag': part['ETag']})
78
+
79
+ await s3_client.complete_multipart_upload(
80
+ Bucket=bucket, Key=key,
81
+ UploadId=mpu_id, MultipartUpload={'Parts': parts}
82
+ )
83
+
84
+ else:
85
+ content = await response.read()
86
+ await s3_client.put_object(
87
+ Bucket=bucket, Key=key, Body=content,
88
+ ContentType=response.headers.get('Content-Type', 'application/octet-stream'),
89
+ Metadata={'source-url': download_url, 'transfer-date': datetime.now(timezone.utc).isoformat()}
90
+ )
91
+
92
+ return {'success': True, 'dataset': dataset, 'size_bytes': int(size_gb * 1024**3), 'billing': billing}
93
+
94
+ except Exception as e:
95
+ if mpu_id:
96
+ try:
97
+ await s3_client.abort_multipart_upload(Bucket=bucket, Key=key, UploadId=mpu_id)
98
+ except Exception:
99
+ pass
100
+ if attempt < retry_errors:
101
+ download_url, _, _, _ = await _get_dataset_url(session, dataset)
102
+ await asyncio.sleep(2 ** attempt)
103
+ last_error = e
104
+
105
+ return {'success': False, 'dataset': dataset, 'error': str(last_error)}
106
+
107
+
108
+ async def _transfer_datasets(datasets, s3_credentials, max_workers, retry_errors, prefix=None):
109
+ connector = aiohttp.TCPConnector(limit=max_workers, ssl=ssl.create_default_context())
110
+ async with aiohttp.ClientSession(connector=connector, timeout=aiohttp.ClientTimeout(total=7200)) as session:
111
+ async with aioboto3.Session().client(
112
+ 's3',
113
+ aws_access_key_id=s3_credentials['aws_access_key_id'],
114
+ aws_secret_access_key=s3_credentials['aws_secret_access_key'],
115
+ region_name=s3_credentials['region_name']
116
+ ) as s3_client:
117
+ semaphore = asyncio.Semaphore(max_workers)
118
+ tasks = [_transfer_dataset(session, s3_client, semaphore, d, s3_credentials['bucket_name'], prefix, retry_errors) for d in datasets]
119
+
120
+ failed, total_bytes = [], 0
121
+ with tqdm(total=len(datasets), desc="Transferring datasets", unit="dataset") as pbar:
122
+ for coro in asyncio.as_completed(tasks):
123
+ result = await coro
124
+ if result['success']:
125
+ total_bytes += result['size_bytes']
126
+ else:
127
+ failed.append(result)
128
+ pbar.set_postfix({'Total': f"{total_bytes / (1024**3):.2f} GB"})
129
+ pbar.update(1)
130
+
131
+ return failed
132
+
133
+
134
+ def datasets_transfer(datasets, s3_credentials, max_workers=4, errors_json_filename='errors_datasets_transfer.json', retry_errors=3, prefix=None):
135
+ failed = asyncio.run(_transfer_datasets(datasets, s3_credentials, max_workers, retry_errors, prefix=prefix))
136
+
137
+ if failed and errors_json_filename:
138
+ with open(errors_json_filename, 'w') as f:
139
+ json.dump(failed, f, indent=2)
140
+ print(f"Saved {len(failed)} errors to {errors_json_filename}")
141
+
142
+ print(f"Transfer complete: {len(datasets) - len(failed)}/{len(datasets)} datasets successful")
@@ -0,0 +1,63 @@
1
+ from datetime import datetime, timedelta
2
+ from ..v3.databases import read_query
3
+ from ..utils.format_accession import format_accession
4
+
5
+ def _generate_dates(filing_date):
6
+ if isinstance(filing_date, str):
7
+ return [filing_date]
8
+ elif isinstance(filing_date, list):
9
+ return filing_date
10
+ elif isinstance(filing_date, tuple):
11
+ start = datetime.strptime(filing_date[0], '%Y-%m-%d')
12
+ end = datetime.strptime(filing_date[1], '%Y-%m-%d')
13
+ dates = []
14
+ current = start
15
+ while current <= end:
16
+ dates.append(current.strftime('%Y-%m-%d'))
17
+ current += timedelta(days=1)
18
+ return dates
19
+ raise ValueError('filing_date must be a string, list, or (start, end) tuple')
20
+
21
+ def _sql_literal(value):
22
+ return "'" + str(value).replace("'", "''") + "'"
23
+
24
+ def _normalize_accession(value):
25
+ return str(value).replace("-", "")
26
+
27
+ def _sql_value(value, quote=True):
28
+ return _sql_literal(value) if quote else str(value)
29
+
30
+ def _sql_filter(column, value, transform=None, quote=True):
31
+ if value is None:
32
+ return None
33
+
34
+ if isinstance(value, tuple):
35
+ start = transform(value[0]) if transform else value[0]
36
+ end = transform(value[1]) if transform else value[1]
37
+ return f"{column} BETWEEN {_sql_value(start, quote)} AND {_sql_value(end, quote)}"
38
+
39
+ if isinstance(value, (list, set)):
40
+ values = [transform(item) if transform else item for item in value]
41
+ return f"{column} IN ({', '.join(_sql_value(item, quote) for item in values)})"
42
+
43
+ value = transform(value) if transform else value
44
+ return f"{column} = {_sql_value(value, quote)}"
45
+
46
+ def _get_urls(submission_type=None, cik=None, filing_date=None, accession_number=None):
47
+ filters = [
48
+ _sql_filter("cik", cik, quote=False),
49
+ _sql_filter("form", submission_type),
50
+ _sql_filter("filingdate", filing_date),
51
+ _sql_filter("accessionnumber", accession_number, _normalize_accession, quote=False),
52
+ ]
53
+ where = " AND ".join(item for item in filters if item)
54
+ if where:
55
+ where = f"WHERE {where}"
56
+
57
+ rows = read_query(f"""
58
+ SELECT accessionnumber
59
+ FROM submissions_metadata
60
+ {where}
61
+ ORDER BY filingdate, accessionnumber
62
+ """)
63
+ return [f"https://sec-library.datamule.xyz/{format_accession(row['accessionnumber'], 'no-dash')}.sgml" for row in rows]
File without changes
@@ -0,0 +1,20 @@
1
+ def format_accession(accession, format):
2
+ if format == 'int':
3
+ accession = int(str(accession).replace('-',''))
4
+ elif format == 'dash':
5
+ accession = str(int(str(accession).replace('-',''))).zfill(18)
6
+ accession = f"{accession[:10]}-{accession[10:12]}-{accession[12:]}"
7
+ elif format == 'no-dash':
8
+ accession = str(int(str(accession).replace('-',''))).zfill(18)
9
+ else:
10
+ raise ValueError("unrecognized format")
11
+ return accession
12
+
13
+ def detect_accession_type(accession):
14
+ accession = str(accession)
15
+ if '-' in accession:
16
+ return 'dash'
17
+ elif len(accession) == 18:
18
+ return 'no-dash'
19
+ else:
20
+ return 'int'
@@ -0,0 +1,13 @@
1
+ from . import databases
2
+ from . import datasets
3
+ from . import sec_filings_archive
4
+ from . import sec_filings_lookup
5
+ from . import sec_filings_notifications
6
+
7
+ __all__ = [
8
+ "databases",
9
+ "datasets",
10
+ "sec_filings_archive",
11
+ "sec_filings_lookup",
12
+ "sec_filings_notifications",
13
+ ]
@@ -0,0 +1,3 @@
1
+ from .athena import query, read_query
2
+
3
+ __all__ = ["query", "read_query"]