cwms-python 1.1.0__tar.gz → 1.1.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (45) hide show
  1. {cwms_python-1.1.0 → cwms_python-1.1.2}/PKG-INFO +42 -1
  2. {cwms_python-1.1.0 → cwms_python-1.1.2}/README.md +41 -0
  3. {cwms_python-1.1.0 → cwms_python-1.1.2}/cwms/api.py +97 -35
  4. {cwms_python-1.1.0 → cwms_python-1.1.2}/cwms/ratings/ratings.py +4 -4
  5. {cwms_python-1.1.0 → cwms_python-1.1.2}/cwms/ratings/ratings_spec.py +2 -2
  6. {cwms_python-1.1.0 → cwms_python-1.1.2}/cwms/timeseries/timeseries.py +171 -82
  7. {cwms_python-1.1.0 → cwms_python-1.1.2}/cwms/users/users.py +6 -12
  8. {cwms_python-1.1.0 → cwms_python-1.1.2}/pyproject.toml +1 -1
  9. {cwms_python-1.1.0 → cwms_python-1.1.2}/LICENSE +0 -0
  10. {cwms_python-1.1.0 → cwms_python-1.1.2}/cwms/__init__.py +0 -0
  11. {cwms_python-1.1.0 → cwms_python-1.1.2}/cwms/catalog/blobs.py +0 -0
  12. {cwms_python-1.1.0 → cwms_python-1.1.2}/cwms/catalog/catalog.py +0 -0
  13. {cwms_python-1.1.0 → cwms_python-1.1.2}/cwms/catalog/clobs.py +0 -0
  14. {cwms_python-1.1.0 → cwms_python-1.1.2}/cwms/cwms_types.py +0 -0
  15. {cwms_python-1.1.0 → cwms_python-1.1.2}/cwms/forecast/forecast_instance.py +0 -0
  16. {cwms_python-1.1.0 → cwms_python-1.1.2}/cwms/forecast/forecast_spec.py +0 -0
  17. {cwms_python-1.1.0 → cwms_python-1.1.2}/cwms/levels/location_levels.py +0 -0
  18. {cwms_python-1.1.0 → cwms_python-1.1.2}/cwms/levels/specified_levels.py +0 -0
  19. {cwms_python-1.1.0 → cwms_python-1.1.2}/cwms/locations/gate_changes.py +0 -0
  20. {cwms_python-1.1.0 → cwms_python-1.1.2}/cwms/locations/location_groups.py +0 -0
  21. {cwms_python-1.1.0 → cwms_python-1.1.2}/cwms/locations/lookups.py +0 -0
  22. {cwms_python-1.1.0 → cwms_python-1.1.2}/cwms/locations/physical_locations.py +0 -0
  23. {cwms_python-1.1.0 → cwms_python-1.1.2}/cwms/locks/locks.py +0 -0
  24. {cwms_python-1.1.0 → cwms_python-1.1.2}/cwms/measurements/measurements.py +0 -0
  25. {cwms_python-1.1.0 → cwms_python-1.1.2}/cwms/outlets/outlets.py +0 -0
  26. {cwms_python-1.1.0 → cwms_python-1.1.2}/cwms/outlets/virtual_outlets.py +0 -0
  27. {cwms_python-1.1.0 → cwms_python-1.1.2}/cwms/projects/project_lock_rights.py +0 -0
  28. {cwms_python-1.1.0 → cwms_python-1.1.2}/cwms/projects/project_locks.py +0 -0
  29. {cwms_python-1.1.0 → cwms_python-1.1.2}/cwms/projects/projects.py +0 -0
  30. {cwms_python-1.1.0 → cwms_python-1.1.2}/cwms/projects/water_supply/accounting.py +0 -0
  31. {cwms_python-1.1.0 → cwms_python-1.1.2}/cwms/projects/water_supply/water_contracts.py +0 -0
  32. {cwms_python-1.1.0 → cwms_python-1.1.2}/cwms/projects/water_supply/water_users.py +0 -0
  33. {cwms_python-1.1.0 → cwms_python-1.1.2}/cwms/properties/properties.py +0 -0
  34. {cwms_python-1.1.0 → cwms_python-1.1.2}/cwms/ratings/ratings_template.py +0 -0
  35. {cwms_python-1.1.0 → cwms_python-1.1.2}/cwms/standard_text/standard_text.py +0 -0
  36. {cwms_python-1.1.0 → cwms_python-1.1.2}/cwms/timeseries/timeseries_bin.py +0 -0
  37. {cwms_python-1.1.0 → cwms_python-1.1.2}/cwms/timeseries/timeseries_group.py +0 -0
  38. {cwms_python-1.1.0 → cwms_python-1.1.2}/cwms/timeseries/timeseries_identifier.py +0 -0
  39. {cwms_python-1.1.0 → cwms_python-1.1.2}/cwms/timeseries/timeseries_profile.py +0 -0
  40. {cwms_python-1.1.0 → cwms_python-1.1.2}/cwms/timeseries/timeseries_profile_instance.py +0 -0
  41. {cwms_python-1.1.0 → cwms_python-1.1.2}/cwms/timeseries/timeseries_profile_parser.py +0 -0
  42. {cwms_python-1.1.0 → cwms_python-1.1.2}/cwms/timeseries/timeseries_txt.py +0 -0
  43. {cwms_python-1.1.0 → cwms_python-1.1.2}/cwms/turbines/turbines.py +0 -0
  44. {cwms_python-1.1.0 → cwms_python-1.1.2}/cwms/utils/__init__.py +0 -0
  45. {cwms_python-1.1.0 → cwms_python-1.1.2}/cwms/utils/checks.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: cwms-python
3
- Version: 1.1.0
3
+ Version: 1.1.2
4
4
  Summary: Corps water management systems (CWMS) REST API for Data Retrieval of USACE water data
5
5
  License: LICENSE
6
6
  License-File: LICENSE
@@ -69,6 +69,47 @@ cwms.init_session(
69
69
  If both `token` and `api_key` are provided, `cwms-python` will use the token
70
70
  and log a warning.
71
71
 
72
+ ### HTTP connection pools
73
+
74
+ Sessions created with `cwms.init_session(api_root=..., pool_connections=100)`
75
+ configure the same connection pool size and retry policy for HTTP and HTTPS.
76
+ This includes unencrypted internal HTTP CDA roots used by Batch jobs. The default
77
+ pool retains up to 100 connections per host for reuse; it does not cap concurrent
78
+ requests. Set `pool_connections` when creating the session to size that pool for
79
+ your workload. A `Connection pool is full, discarding connection` warning means
80
+ an extra connection is being closed instead of retained, not that a response or
81
+ its data was discarded. Check raised exceptions for actual request failures.
82
+
83
+ ### Errors and debugging
84
+
85
+ Failed HTTP requests raise `cwms.api.ApiError`. Its message includes the HTTP
86
+ status, method, URL, and CDA response body (including incident details when
87
+ provided). The original response is available as `error.response`. Custom
88
+ user-management errors retain these details too. Network exceptions propagate
89
+ with their original type. Invalid JSON responses raise `ApiError` with the
90
+ decoding exception as the cause; empty response bodies return an empty dictionary.
91
+
92
+ Concurrent time-series reads and writes raise `cwms.api.BatchError` if any
93
+ series or chunk fails. It subclasses `RuntimeError`, and `error.failures`
94
+ contains `(series_or_chunk, original_exception)` pairs for every failure.
95
+ Reads do not return incomplete results as success. Successful writes are not
96
+ rolled back. Failures while looking up time-series extents also propagate.
97
+ Chunk retries are limited to connection errors, timeouts, and HTTP
98
+ 429/500/502/503/504; validation and other permanent errors fail immediately
99
+ at the chunk layer. The shared HTTP adapter retains its existing retry policy.
100
+
101
+ Enable request outcome and chunk diagnostics with Python logging:
102
+
103
+ ```python
104
+ import logging
105
+
106
+ logging.basicConfig(level=logging.WARNING)
107
+ logging.getLogger("cwms").setLevel(logging.DEBUG)
108
+ ```
109
+
110
+ Request diagnostics include the method, endpoint, and response status, without
111
+ request bodies or authentication headers.
112
+
72
113
  ## Getting Started
73
114
 
74
115
  ```python
@@ -38,6 +38,47 @@ cwms.init_session(
38
38
  If both `token` and `api_key` are provided, `cwms-python` will use the token
39
39
  and log a warning.
40
40
 
41
+ ### HTTP connection pools
42
+
43
+ Sessions created with `cwms.init_session(api_root=..., pool_connections=100)`
44
+ configure the same connection pool size and retry policy for HTTP and HTTPS.
45
+ This includes unencrypted internal HTTP CDA roots used by Batch jobs. The default
46
+ pool retains up to 100 connections per host for reuse; it does not cap concurrent
47
+ requests. Set `pool_connections` when creating the session to size that pool for
48
+ your workload. A `Connection pool is full, discarding connection` warning means
49
+ an extra connection is being closed instead of retained, not that a response or
50
+ its data was discarded. Check raised exceptions for actual request failures.
51
+
52
+ ### Errors and debugging
53
+
54
+ Failed HTTP requests raise `cwms.api.ApiError`. Its message includes the HTTP
55
+ status, method, URL, and CDA response body (including incident details when
56
+ provided). The original response is available as `error.response`. Custom
57
+ user-management errors retain these details too. Network exceptions propagate
58
+ with their original type. Invalid JSON responses raise `ApiError` with the
59
+ decoding exception as the cause; empty response bodies return an empty dictionary.
60
+
61
+ Concurrent time-series reads and writes raise `cwms.api.BatchError` if any
62
+ series or chunk fails. It subclasses `RuntimeError`, and `error.failures`
63
+ contains `(series_or_chunk, original_exception)` pairs for every failure.
64
+ Reads do not return incomplete results as success. Successful writes are not
65
+ rolled back. Failures while looking up time-series extents also propagate.
66
+ Chunk retries are limited to connection errors, timeouts, and HTTP
67
+ 429/500/502/503/504; validation and other permanent errors fail immediately
68
+ at the chunk layer. The shared HTTP adapter retains its existing retry policy.
69
+
70
+ Enable request outcome and chunk diagnostics with Python logging:
71
+
72
+ ```python
73
+ import logging
74
+
75
+ logging.basicConfig(level=logging.WARNING)
76
+ logging.getLogger("cwms").setLevel(logging.DEBUG)
77
+ ```
78
+
79
+ Request diagnostics include the method, endpoint, and response status, without
80
+ request bodies or authentication headers.
81
+
41
82
  ## Getting Started
42
83
 
43
84
  ```python
@@ -33,13 +33,14 @@ import base64
33
33
  import json
34
34
  import logging
35
35
  from http import HTTPStatus
36
- from json import JSONDecodeError
37
- from typing import Any, Optional, cast
36
+ from typing import Any, Optional, Union, cast
38
37
 
39
38
  from requests import Response, adapters
39
+ from requests.exceptions import JSONDecodeError, RequestException
40
40
  from requests.exceptions import RetryError as RequestsRetryError
41
41
  from requests_toolbelt import sessions # type: ignore
42
42
  from requests_toolbelt.sessions import BaseUrlSession # type: ignore
43
+ from urllib3.exceptions import HTTPError as Urllib3HTTPError
43
44
  from urllib3.util.retry import Retry
44
45
 
45
46
  from cwms.cwms_types import JSON, RequestParams
@@ -47,6 +48,7 @@ from cwms.cwms_types import JSON, RequestParams
47
48
  # Specify the default API root URL and version.
48
49
  API_ROOT = "https://cwms-data.usace.army.mil/cwms-data/"
49
50
  API_VERSION = 2
51
+ logger = logging.getLogger(__name__)
50
52
 
51
53
  # Specify whether LRTS will use new ID format
52
54
  USE_NEW_LRTS_IDS = False
@@ -71,6 +73,7 @@ adapter = adapters.HTTPAdapter(
71
73
  pool_connections=100, pool_maxsize=100, max_retries=retry_strategy
72
74
  )
73
75
  SESSION.mount("https://", adapter)
76
+ SESSION.mount("http://", adapter)
74
77
 
75
78
 
76
79
  class InvalidVersion(Exception):
@@ -90,17 +93,19 @@ class ApiError(Exception):
90
93
  self.message = message
91
94
 
92
95
  def __str__(self) -> str:
93
- if self.message:
94
- return self.message
95
-
96
96
  # Include the request URL in the error message.
97
- message = f"CWMS API Error ({self.response.url})"
97
+ message = f"CWMS API Error ({self.response.url}) {self.response.status_code}"
98
+ request = getattr(self.response, "request", None)
99
+ if request is not None:
100
+ message += f" {request.method}"
98
101
 
99
102
  # If a reason is provided in the response, add it to the message.
100
103
  if reason := self.response.reason:
101
104
  message += f" {reason}"
102
105
 
103
106
  message += "."
107
+ if self.message:
108
+ message += f" {self.message}"
104
109
 
105
110
  # Add additional context to help the user resolve the issue.
106
111
  hint = self.hint()
@@ -111,10 +116,7 @@ class ApiError(Exception):
111
116
  content = getattr(self.response, "content", None)
112
117
  if content:
113
118
  if isinstance(content, bytes):
114
- try:
115
- text = content.decode("utf-8", errors="replace")
116
- except Exception:
117
- text = repr(content)
119
+ text = content.decode("utf-8", errors="replace")
118
120
  else:
119
121
  text = str(content)
120
122
  message += f" {text}"
@@ -130,6 +132,9 @@ class ApiError(Exception):
130
132
  if status == 400:
131
133
  return "Check that your parameters are correct."
132
134
  if status == 404:
135
+ request = getattr(self.response, "request", None)
136
+ if request is not None and request.method != "GET":
137
+ return "Check the CDA response body for the failed operation."
133
138
  return "May be the result of an empty query."
134
139
 
135
140
  # No hint for other codes
@@ -144,8 +149,29 @@ class PermissionError(ApiError):
144
149
  """Raised when the CDA request is not authorized for the current caller."""
145
150
 
146
151
 
147
- def _unwrap_retry_error(error: RequestsRetryError) -> Exception:
148
- """Return the original retry cause when requests wraps it in RetryError."""
152
+ class BatchError(RuntimeError):
153
+ """A concurrent operation failed; ``failures`` retains every original exception.
154
+
155
+ Each failure is a (series or chunk description, exception) pair. Successful
156
+ writes are not rolled back. This remains compatible with RuntimeError handlers.
157
+ """
158
+
159
+ def __init__(self, message: str, failures: list[tuple[str, Exception]]):
160
+ self.failures = failures
161
+ super().__init__(
162
+ message
163
+ + "\n"
164
+ + "\n".join(f"{context}: {error}" for context, error in failures)
165
+ )
166
+
167
+
168
+ def _unwrap_retry_error(
169
+ error: RequestsRetryError,
170
+ ) -> Union[RequestException, Urllib3HTTPError]:
171
+ """Unwrap transport errors; retain RetryError for other causes.
172
+
173
+ Unknown causes remain available through the original wrapper's chain or args.
174
+ """
149
175
 
150
176
  current: Exception = error
151
177
  cause = error.__cause__
@@ -162,7 +188,9 @@ def _unwrap_retry_error(error: RequestsRetryError) -> Exception:
162
188
  current = reason
163
189
  reason = getattr(current, "reason", None)
164
190
 
165
- return current
191
+ if isinstance(current, (RequestException, Urllib3HTTPError)):
192
+ return current
193
+ return error
166
194
 
167
195
 
168
196
  def init_session(
@@ -185,6 +213,9 @@ def init_session(
185
213
  api_key (optional): An authentication key.
186
214
  token (optional): A Keycloak access token. If both token and api_key are
187
215
  provided, token is used.
216
+ pool_connections (optional): Number of host pools and reusable connections
217
+ per host when creating a session with api_root. Defaults to 100 for
218
+ both HTTP and HTTPS; this is not a limit on concurrent requests.
188
219
 
189
220
  Returns:
190
221
  Returns the updated session object.
@@ -202,6 +233,7 @@ def init_session(
202
233
  max_retries=retry_strategy,
203
234
  )
204
235
  SESSION.mount("https://", adapter)
236
+ SESSION.mount("http://", adapter)
205
237
  if token:
206
238
  if api_key:
207
239
  logging.warning(
@@ -336,6 +368,8 @@ def get_xml(
336
368
 
337
369
 
338
370
  def _process_response(response: Response) -> Any:
371
+ if not response.content:
372
+ return {}
339
373
  try:
340
374
  # Avoid case sensitivity issues with the content type header
341
375
  content_type = response.headers.get("Content-Type", "").lower()
@@ -356,10 +390,14 @@ def _process_response(response: Response) -> Any:
356
390
  # Fallback for remaining content types
357
391
  return response.content.decode("utf-8")
358
392
  except JSONDecodeError as error:
359
- logging.error(
360
- f"Error decoding CDA response as JSON: {error} on line {error.lineno}\n\tFalling back to text"
361
- )
362
- return response.text
393
+ raise ApiError(response, "Invalid JSON in CDA response.") from error
394
+
395
+
396
+ def _check_response(response: Response, method: str, endpoint: str) -> None:
397
+ """Record request outcomes without logging credentials or request bodies."""
398
+ logger.debug("CDA %s %s returned HTTP %s", method, endpoint, response.status_code)
399
+ if not response.ok:
400
+ raise ApiError(response)
363
401
 
364
402
 
365
403
  def get(
@@ -391,12 +429,16 @@ def get(
391
429
  }
392
430
  try:
393
431
  with SESSION.get(endpoint, params=params, headers=headers) as response:
394
- if not response.ok:
395
- logging.error(f"CDA Error: response={response}")
396
- raise ApiError(response)
432
+ _check_response(response, "GET", endpoint)
397
433
  return _process_response(response)
398
434
  except RequestsRetryError as error:
399
- raise _unwrap_retry_error(error) from None
435
+ cause = _unwrap_retry_error(error)
436
+ if cause is error:
437
+ raise
438
+ raise cause from error
439
+ except RequestException as error:
440
+ logger.debug("CDA GET %s failed: %s", endpoint, type(error).__name__)
441
+ raise
400
442
 
401
443
 
402
444
  def get_with_paging(
@@ -459,12 +501,16 @@ def _post_function(
459
501
  with SESSION.post(
460
502
  endpoint, params=params, headers=headers, data=data
461
503
  ) as response:
462
- if not response.ok:
463
- logging.error(f"CDA Error: response={response}")
464
- raise ApiError(response)
504
+ _check_response(response, "POST", endpoint)
465
505
  return response
466
506
  except RequestsRetryError as error:
467
- raise _unwrap_retry_error(error) from None
507
+ cause = _unwrap_retry_error(error)
508
+ if cause is error:
509
+ raise
510
+ raise cause from error
511
+ except RequestException as error:
512
+ logger.debug("CDA POST %s failed: %s", endpoint, type(error).__name__)
513
+ raise
468
514
 
469
515
 
470
516
  def post(
@@ -556,17 +602,21 @@ def patch(
556
602
  "X-CWMS-LRTS-Formatting": str(USE_NEW_LRTS_IDS).lower(),
557
603
  }
558
604
 
559
- if data and isinstance(data, dict) or isinstance(data, list):
605
+ if isinstance(data, (dict, list)):
560
606
  data = json.dumps(data)
561
607
  try:
562
608
  with SESSION.patch(
563
609
  endpoint, params=params, headers=headers, data=data
564
610
  ) as response:
565
- if not response.ok:
566
- logging.error(f"CDA Error: response={response}")
567
- raise ApiError(response)
611
+ _check_response(response, "PATCH", endpoint)
568
612
  except RequestsRetryError as error:
569
- raise _unwrap_retry_error(error) from None
613
+ cause = _unwrap_retry_error(error)
614
+ if cause is error:
615
+ raise
616
+ raise cause from error
617
+ except RequestException as error:
618
+ logger.debug("CDA PATCH %s failed: %s", endpoint, type(error).__name__)
619
+ raise
570
620
 
571
621
 
572
622
  def delete(
@@ -574,6 +624,7 @@ def delete(
574
624
  params: Optional[RequestParams] = None,
575
625
  *,
576
626
  api_version: int = API_VERSION,
627
+ data: Optional[Any] = None,
577
628
  ) -> None:
578
629
  """Make a DELETE request to the CWMS Data API.
579
630
 
@@ -584,6 +635,7 @@ def delete(
584
635
  Keyword Args:
585
636
  api_version (optional): The CDA version to use for the request. If not specified,
586
637
  the default API_VERSION will be used.
638
+ data (optional): Request body, JSON-encoded for dictionaries and lists.
587
639
 
588
640
  Raises:
589
641
  ApiError: If an error response is return by the API.
@@ -593,10 +645,20 @@ def delete(
593
645
  "Accept": api_version_text(api_version),
594
646
  "X-CWMS-LRTS-Formatting": str(USE_NEW_LRTS_IDS).lower(),
595
647
  }
648
+ kwargs: dict[str, Any] = {}
649
+ if data is not None:
650
+ headers["Content-Type"] = api_version_text(api_version)
651
+ kwargs["data"] = json.dumps(data) if isinstance(data, (dict, list)) else data
596
652
  try:
597
- with SESSION.delete(endpoint, params=params, headers=headers) as response:
598
- if not response.ok:
599
- logging.error(f"CDA Error: response={response}")
600
- raise ApiError(response)
653
+ with SESSION.delete(
654
+ endpoint, params=params, headers=headers, **kwargs
655
+ ) as response:
656
+ _check_response(response, "DELETE", endpoint)
601
657
  except RequestsRetryError as error:
602
- raise _unwrap_retry_error(error) from None
658
+ cause = _unwrap_retry_error(error)
659
+ if cause is error:
660
+ raise
661
+ raise cause from error
662
+ except RequestException as error:
663
+ logger.debug("CDA DELETE %s failed: %s", endpoint, type(error).__name__)
664
+ raise
@@ -424,8 +424,8 @@ def _validate_rating_params(
424
424
  raise ValueError(f"Invalid rating identifer: {rating_id}")
425
425
  try:
426
426
  ind_params, _ = parts[1].split(";")
427
- except Exception:
428
- raise ValueError(f"Invalid rating template: {parts[1]}")
427
+ except ValueError as error:
428
+ raise ValueError(f"Invalid rating template: {parts[1]}") from error
429
429
  if not office_id:
430
430
  raise ValueError("Cannot rate values without an office identifier")
431
431
  if not units:
@@ -466,8 +466,8 @@ def _perform_value_rating(
466
466
  ind_params = _validate_rating_params(rating_id, office_id, units, values)
467
467
  try:
468
468
  ind_units_str, dep_unit = units.split(";")
469
- except Exception:
470
- raise ValueError("Invalid units string")
469
+ except ValueError as error:
470
+ raise ValueError("Invalid units string") from error
471
471
  value_count = len(values[0])
472
472
  times = _get_times(value_count, times)
473
473
  if not rating_time:
@@ -127,7 +127,7 @@ def rating_spec_df_to_xml(data: pd.DataFrame) -> str:
127
127
  try:
128
128
  spec_xml += f"""
129
129
  <source-agency>{data.loc[0,'source-agency']}</source-agency>"""
130
- except Exception:
130
+ except KeyError:
131
131
  spec_xml += """
132
132
  <source-agency/>"""
133
133
  spec_xml += f"""
@@ -154,7 +154,7 @@ def rating_spec_df_to_xml(data: pd.DataFrame) -> str:
154
154
  try:
155
155
  spec_xml2 += f"""
156
156
  <description>{data.loc[0,'description']}</description>"""
157
- except Exception:
157
+ except KeyError:
158
158
  spec_xml2 += """
159
159
  <description/>"""
160
160
  spec_xml2 += """
@@ -1,15 +1,87 @@
1
1
  import concurrent.futures
2
2
  import logging
3
+ import re
3
4
  from datetime import datetime, timedelta, timezone
4
5
  from typing import Any, Dict, List, Optional, Tuple
5
6
 
6
7
  import pandas as pd
7
8
  from pandas import DataFrame
9
+ from requests.exceptions import ConnectionError, Timeout
8
10
 
9
11
  import cwms.api as api
10
12
  from cwms.catalog.catalog import get_ts_extents
11
13
  from cwms.cwms_types import JSON, Data
12
14
 
15
+ logger = logging.getLogger(__name__)
16
+
17
+ _DEFAULT_CHUNK_DAYS = 365
18
+ _MIN_INTERVAL_MINUTES = 2
19
+ _FINE_INTERVAL_MINUTES = 15
20
+ _FINE_INTERVAL_CHUNK_DAYS = 365
21
+ _HOURLY_CHUNK_DAYS = 365
22
+ _SIX_HOURLY_CHUNK_DAYS = 1460
23
+ _COARSE_INTERVAL_CHUNK_DAYS = 2920
24
+ _INTERVAL_PATTERN = re.compile(
25
+ r"^(?P<count>\d+)(?P<unit>"
26
+ r"Minute|Minutes|Hour|Hours|Day|Days|"
27
+ r"Week|Weeks|Month|Months|Year|Years)$"
28
+ )
29
+ _INTERVAL_MINUTES = {
30
+ "Minute": 1,
31
+ "Minutes": 1,
32
+ "Hour": 60,
33
+ "Hours": 60,
34
+ "Day": 24 * 60,
35
+ "Days": 24 * 60,
36
+ "Week": 7 * 24 * 60,
37
+ "Weeks": 7 * 24 * 60,
38
+ "Month": 30 * 24 * 60,
39
+ "Months": 30 * 24 * 60,
40
+ "Year": 365 * 24 * 60,
41
+ "Years": 365 * 24 * 60,
42
+ }
43
+
44
+
45
+ def get_timeseries_chunk_size(ts_id: str) -> timedelta:
46
+ """Return the default request chunk size for a time series interval.
47
+
48
+ Local regular time series intervals, such as ``~15Minutes``, use the same
49
+ chunk size as their regular interval. Unrecognized intervals retain the
50
+ conservative default used for 15-minute through hourly data.
51
+ """
52
+ ts_id_parts = ts_id.split(".")
53
+ if len(ts_id_parts) < 4:
54
+ return timedelta(days=_DEFAULT_CHUNK_DAYS)
55
+
56
+ interval = ts_id_parts[3].removeprefix("~")
57
+ match = _INTERVAL_PATTERN.fullmatch(interval)
58
+ if match is None:
59
+ return timedelta(days=_DEFAULT_CHUNK_DAYS)
60
+
61
+ interval_minutes = (
62
+ int(match.group("count")) * _INTERVAL_MINUTES[match.group("unit")]
63
+ )
64
+ if interval_minutes < _MIN_INTERVAL_MINUTES:
65
+ return timedelta(days=_DEFAULT_CHUNK_DAYS)
66
+
67
+ # These bands balance request overhead and response size based on production
68
+ # CDA timings. Fine intervals scale toward about 35,000 expected values.
69
+ if interval_minutes < _FINE_INTERVAL_MINUTES:
70
+ chunk_days = max(
71
+ 1,
72
+ round(
73
+ _FINE_INTERVAL_CHUNK_DAYS * interval_minutes / _FINE_INTERVAL_MINUTES
74
+ ),
75
+ )
76
+ elif interval_minutes <= 60:
77
+ chunk_days = _HOURLY_CHUNK_DAYS
78
+ elif interval_minutes <= 6 * 60:
79
+ chunk_days = _SIX_HOURLY_CHUNK_DAYS
80
+ else:
81
+ chunk_days = _COARSE_INTERVAL_CHUNK_DAYS
82
+
83
+ return timedelta(days=chunk_days)
84
+
13
85
 
14
86
  def get_multi_timeseries_df(
15
87
  ts_ids: list[str],
@@ -60,45 +132,53 @@ def get_multi_timeseries_df(
60
132
  """
61
133
 
62
134
  def get_ts_ids(ts_id: str) -> Any:
63
- try:
64
- if ":" in ts_id:
65
- ts_id, version_date = ts_id.split(":", 1)
66
- version_date_dt = pd.to_datetime(version_date)
67
- else:
68
- version_date_dt = None
69
- data = get_timeseries(
70
- ts_id=ts_id,
71
- office_id=office_id,
72
- unit=unit,
73
- begin=begin,
74
- end=end,
75
- version_date=version_date_dt,
76
- multithread=False,
77
- )
78
- result_dict = {
79
- "ts_id": ts_id,
80
- "unit": data.json["units"],
81
- "version_date": version_date_dt,
82
- "values": data.df,
83
- }
84
- return result_dict
85
- except Exception as e:
86
- logging.error(f"Error processing {ts_id}: {e}")
87
- return None
135
+ if ":" in ts_id:
136
+ ts_id, version_date = ts_id.split(":", 1)
137
+ version_date_dt = pd.to_datetime(version_date)
138
+ else:
139
+ version_date_dt = None
140
+ data = get_timeseries(
141
+ ts_id=ts_id,
142
+ office_id=office_id,
143
+ unit=unit,
144
+ begin=begin,
145
+ end=end,
146
+ version_date=version_date_dt,
147
+ multithread=False,
148
+ )
149
+ result_dict = {
150
+ "ts_id": ts_id,
151
+ "unit": data.json["units"],
152
+ "version_date": version_date_dt,
153
+ "values": data.df,
154
+ }
155
+ return result_dict
88
156
 
157
+ logger.debug(
158
+ "Fetching %s time series with up to %s workers", len(ts_ids), max_workers
159
+ )
160
+ failures: list[tuple[str, Exception]] = []
161
+ result_dict = []
89
162
  with concurrent.futures.ThreadPoolExecutor(max_workers=max_workers) as executor:
90
- results = executor.map(get_ts_ids, ts_ids)
163
+ futures = [(ts_id, executor.submit(get_ts_ids, ts_id)) for ts_id in ts_ids]
164
+ for ts_id, future in futures:
165
+ try:
166
+ result_dict.append(future.result())
167
+ except Exception as error:
168
+ failures.append((ts_id, error))
169
+ if failures:
170
+ raise api.BatchError(
171
+ f"{len(failures)} time series failed to fetch:", failures
172
+ ) from failures[0][1]
91
173
 
92
- result_dict = list(results)
93
174
  data = pd.DataFrame()
94
175
  for row in result_dict:
95
- if row:
96
- temp_df = row["values"]
97
- temp_df = temp_df.assign(ts_id=row["ts_id"], units=row["unit"])
98
- if "version_date" in row.keys():
99
- temp_df = temp_df.assign(version_date=row["version_date"])
100
- temp_df.dropna(how="all", axis=1, inplace=True)
101
- data = pd.concat([data, temp_df], ignore_index=True)
176
+ temp_df = row["values"]
177
+ temp_df = temp_df.assign(ts_id=row["ts_id"], units=row["unit"])
178
+ if "version_date" in row.keys():
179
+ temp_df = temp_df.assign(version_date=row["version_date"])
180
+ temp_df.dropna(how="all", axis=1, inplace=True)
181
+ data = pd.concat([data, temp_df], ignore_index=True)
102
182
 
103
183
  if not melted and "date-time" in data.columns:
104
184
  cols = ["ts_id", "units"]
@@ -136,6 +216,9 @@ def chunk_timeseries_time_range(
136
216
  List[Tuple[datetime, datetime]]
137
217
  A list of tuples, where each tuple represents the start and end of a chunk.
138
218
  """
219
+ if chunk_size <= timedelta(0):
220
+ raise ValueError("chunk_size must be greater than zero")
221
+
139
222
  chunks = []
140
223
  current = begin
141
224
  while current < end:
@@ -162,16 +245,24 @@ _CHUNK_ATTEMPTS = 6
162
245
 
163
246
 
164
247
  def _call_with_retry(fn: Any, *args: Any, attempts: int = _CHUNK_ATTEMPTS) -> Any:
248
+ if attempts < 1:
249
+ raise ValueError("attempts must be at least 1")
165
250
  for i in range(attempts):
166
251
  try:
167
252
  return fn(*args)
168
- except Exception as e:
253
+ except (api.ApiError, ConnectionError, Timeout) as e:
169
254
  status_code = getattr(getattr(e, "response", None), "status_code", None)
170
- if status_code == 404:
255
+ if isinstance(e, api.ApiError) and status_code not in {
256
+ 429,
257
+ 500,
258
+ 502,
259
+ 503,
260
+ 504,
261
+ }:
171
262
  raise
172
263
  if i == attempts - 1:
173
264
  raise
174
- logging.warning(f"chunk attempt {i + 1}/{attempts} failed: {e}")
265
+ logger.warning(f"chunk attempt {i + 1}/{attempts} failed: {e}")
175
266
 
176
267
 
177
268
  def fetch_timeseries_chunks(
@@ -182,7 +273,7 @@ def fetch_timeseries_chunks(
182
273
  max_workers: int,
183
274
  ) -> List[Data]:
184
275
  results: List[Data] = []
185
- errors: List[str] = []
276
+ errors: list[tuple[str, Exception]] = []
186
277
 
187
278
  with concurrent.futures.ThreadPoolExecutor(max_workers=max_workers) as executor:
188
279
  future_to_chunk = {
@@ -206,14 +297,15 @@ def fetch_timeseries_chunks(
206
297
  error_msg = (
207
298
  f"Failed to fetch data from {chunk_start} to {chunk_end}: {e}"
208
299
  )
209
- logging.error(error_msg)
210
- errors.append(error_msg)
300
+ logger.debug(error_msg)
301
+ errors.append(
302
+ (f"Failed to fetch data from {chunk_start} to {chunk_end}", e)
303
+ )
211
304
 
212
305
  if errors:
213
- raise RuntimeError(
214
- f"{len(errors)} of {len(chunks)} chunk(s) failed to fetch:\n"
215
- + "\n".join(errors)
216
- )
306
+ raise api.BatchError(
307
+ f"{len(errors)} of {len(chunks)} chunk(s) failed to fetch:", errors
308
+ ) from errors[0][1]
217
309
 
218
310
  return results
219
311
 
@@ -298,7 +390,7 @@ def get_timeseries(
298
390
  trim: Optional[bool] = True,
299
391
  multithread: Optional[bool] = True,
300
392
  max_workers: int = 20,
301
- max_days_per_chunk: int = 14,
393
+ max_days_per_chunk: Optional[int] = None,
302
394
  ) -> Data:
303
395
  """Retrieves time series values from a specified time series and time window. Value date-times
304
396
  obtained are always in UTC.
@@ -338,15 +430,18 @@ def get_timeseries(
338
430
  trim: boolean, optional, default is True
339
431
  Specifies whether to trim missing values from the beginning and end of the retrieved values.
340
432
  multithread: boolean, optional, default is True
341
- Specifies whether to trim missing values from the beginning and end of the retrieved values.
433
+ Specifies whether to retrieve time series chunks concurrently.
342
434
  max_workers: integer, default is 20
343
- The maximum number of worker threads that will be spawned for multithreading, If calling more than 3 years of 15 minute data, consider using 30 max_workers
344
- max_days_per_chunk: integer, default is 14
345
- The maximum number of days that would be included in a thread. If calling more than 1 year of 15 minute data, consider using 30 days
435
+ The maximum number of worker threads used for concurrent requests.
436
+ max_days_per_chunk: integer, optional, default is None
437
+ The maximum number of days included in each request. By default,
438
+ the chunk size is selected from the time series interval.
346
439
  Returns
347
440
  -------
348
441
  cwms data type. data.json will return the JSON output and data.df will return a dataframe. dates are all in UTC
349
442
  """
443
+ if max_days_per_chunk is not None and max_days_per_chunk <= 0:
444
+ raise ValueError("max_days_per_chunk must be greater than zero")
350
445
 
351
446
  selector = "values"
352
447
  endpoint = "timeseries"
@@ -367,27 +462,20 @@ def get_timeseries(
367
462
 
368
463
  # grab extents if begin is before CWMS DB were implemented to prevent empty queries outside of extents
369
464
  if begin < datetime(2014, 1, 1, tzinfo=timezone.utc) and multithread:
370
- try:
371
- begin_extent, _, _ = get_ts_extents(ts_id=ts_id, office_id=office_id)
372
- # replace begin with begin extent if outside extents
373
- if begin < begin_extent:
374
- begin = begin_extent
375
- logging.debug(
376
- f"Requested begin was before any data in this timeseries. Reseting to {begin}"
377
- )
378
- except Exception as e:
379
- # If getting extents fails, fall back to single-threaded mode
380
- logging.debug(
381
- f"Could not retrieve time series extents ({e}). Falling back to single-threaded mode."
382
- )
383
-
384
- response = api.get_with_paging(
385
- selector=selector, endpoint=endpoint, params=params
465
+ begin_extent, _, _ = get_ts_extents(ts_id=ts_id, office_id=office_id)
466
+ # replace begin with begin extent if outside extents
467
+ if begin < begin_extent:
468
+ begin = begin_extent
469
+ logger.debug(
470
+ f"Requested begin was before any data in this timeseries. Reseting to {begin}"
386
471
  )
387
- return Data(response, selector=selector)
388
472
 
389
- # divide the time range into chunks
390
- chunks = chunk_timeseries_time_range(begin, end, timedelta(days=max_days_per_chunk))
473
+ chunk_size = (
474
+ timedelta(days=max_days_per_chunk)
475
+ if max_days_per_chunk is not None
476
+ else get_timeseries_chunk_size(ts_id)
477
+ )
478
+ chunks = chunk_timeseries_time_range(begin, end, chunk_size)
391
479
 
392
480
  # find max worker thread
393
481
  max_workers = max(min(len(chunks), max_workers), 1)
@@ -399,7 +487,7 @@ def get_timeseries(
399
487
  )
400
488
  return Data(response, selector=selector)
401
489
  else:
402
- logging.debug(
490
+ logger.debug(
403
491
  f"Fetching {len(chunks)} chunks of timeseries data with {max_workers} threads"
404
492
  )
405
493
  # fetch the data
@@ -580,7 +668,7 @@ def store_multi_timeseries_df(
580
668
  ts_data_all["ts_id"].astype(str) + ":" + ts_data_all["version_date"].astype(str)
581
669
  ).unique()
582
670
 
583
- errors: List[str] = []
671
+ errors: list[tuple[str, Exception]] = []
584
672
  with concurrent.futures.ThreadPoolExecutor(max_workers=max_workers) as executor:
585
673
  futures = {}
586
674
  for unique_tsid in unique_tsids:
@@ -606,12 +694,12 @@ def store_multi_timeseries_df(
606
694
  try:
607
695
  future.result()
608
696
  except Exception as e:
609
- errors.append(f"{futures[future]}: {e}")
697
+ errors.append((str(futures[future]), e))
610
698
 
611
699
  if errors:
612
- raise RuntimeError(
613
- f"{len(errors)} time series failed to store:\n" + "\n".join(errors)
614
- )
700
+ raise api.BatchError(
701
+ f"{len(errors)} time series failed to store:", errors
702
+ ) from errors[0][1]
615
703
 
616
704
 
617
705
  def chunk_timeseries_data(
@@ -711,13 +799,13 @@ def store_timeseries(
711
799
  _call_with_retry(api.post, endpoint, chunks[0], params)
712
800
  remaining_chunks = chunks[1:]
713
801
  actual_workers = min(max_workers, len(remaining_chunks))
714
- logging.debug(
802
+ logger.debug(
715
803
  f"Storing {len(chunks)} chunks of timeseries data with {actual_workers} threads"
716
804
  )
717
805
 
718
806
  # Store chunks concurrently
719
807
  responses: List[Dict[str, Any]] = []
720
- errors: List[str] = []
808
+ errors: list[tuple[str, Exception]] = []
721
809
 
722
810
  with concurrent.futures.ThreadPoolExecutor(max_workers=actual_workers) as executor:
723
811
  future_to_chunk = {
@@ -733,15 +821,16 @@ def store_timeseries(
733
821
  start_time = chunk["values"][0][0]
734
822
  end_time = chunk["values"][-1][0]
735
823
  error_msg = f"Error storing chunk from {start_time} to {end_time}: {e}"
736
- logging.error(error_msg)
737
- errors.append(error_msg)
824
+ logger.debug(error_msg)
825
+ errors.append(
826
+ (f"Error storing chunk from {start_time} to {end_time}", e)
827
+ )
738
828
  responses.append({"error": error_msg})
739
829
 
740
830
  if errors:
741
- raise RuntimeError(
742
- f"{len(errors)} of {len(chunks)} chunk(s) failed to store:\n"
743
- + "\n".join(errors)
744
- )
831
+ raise api.BatchError(
832
+ f"{len(errors)} of {len(chunks)} chunk(s) failed to store:", errors
833
+ ) from errors[0][1]
745
834
 
746
835
  return
747
836
 
@@ -1,4 +1,3 @@
1
- import json
2
1
  from typing import Any, List, Optional
3
2
 
4
3
  import cwms.api as api
@@ -14,7 +13,7 @@ def _raise_user_management_error(error: api.ApiError, action: str) -> None:
14
13
  "are not authorized for user-management access or are missing the "
15
14
  f"required role assignment. CDA responded with 403 {response_hint}."
16
15
  )
17
- raise api.PermissionError(error.response, message) from None
16
+ raise api.PermissionError(error.response, message) from error
18
17
  raise error
19
18
 
20
19
 
@@ -110,7 +109,7 @@ def get_user(user_name: str) -> dict[str, Any]:
110
109
  if status_code == 404:
111
110
  raise api.NotFoundError(
112
111
  error.response, f"User '{user_name}' was not found."
113
- ) from None
112
+ ) from error
114
113
  if status_code == 403:
115
114
  _raise_user_management_error(error, f"User '{user_name}' retrieval")
116
115
  raise
@@ -153,15 +152,10 @@ def delete_user_roles(user_name: str, office_id: str, roles: List[str]) -> None:
153
152
  raise ValueError("Delete user roles requires a roles list")
154
153
 
155
154
  endpoint = f"user/{user_name}/roles/{office_id}"
156
- headers = {"accept": "*/*", "Content-Type": api.api_version_text(api.API_VERSION)}
157
- # TODO: Delete does not currently support a body in the api module. Use SESSION directly
158
- with api.SESSION.delete(
159
- endpoint, headers=headers, data=json.dumps(roles)
160
- ) as response:
161
- if not response.ok:
162
- _raise_user_management_error(
163
- api.ApiError(response), f"User '{user_name}' role deletion"
164
- )
155
+ try:
156
+ api.delete(endpoint, data=roles)
157
+ except api.ApiError as error:
158
+ _raise_user_management_error(error, f"User '{user_name}' role deletion")
165
159
 
166
160
 
167
161
  def update_user(user_name: str, office_id: str, roles: List[str]) -> None:
@@ -3,7 +3,7 @@ name = "cwms-python"
3
3
  repository = "https://github.com/HydrologicEngineeringCenter/cwms-python"
4
4
 
5
5
  # Managed by Release Please; runtime versions come from package metadata.
6
- version = "1.1.0"
6
+ version = "1.1.2"
7
7
 
8
8
  packages = [
9
9
  { include = "cwms" },
File without changes