dately 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- dately/__init__.py +2 -0
- dately/core.py +507 -0
- dately/mold/__init__.py +0 -0
- dately/mold/pyd/Compiled.cp38-win_amd64.pyd +0 -0
- dately/mold/pyd/__init__.py +0 -0
- dately/mold/pyd/cdatetime/UniversalDateFormatter.cp38-win_amd64.pyd +0 -0
- dately/mold/pyd/cdatetime/__init__.py +0 -0
- dately/mold/pyd/cdatetime/iso8601T.cp38-win_amd64.pyd +0 -0
- dately/mold/pyd/cdatetime/iso8601Z.cp38-win_amd64.pyd +0 -0
- dately/mold/pyd/cdatetime/whichformat.cp38-win_amd64.pyd +0 -0
- dately/mold/pyd/clean_str.cp38-win_amd64.pyd +0 -0
- dately/mold/pyd/time_zones.cp38-win_amd64.pyd +0 -0
- dately/timeutils.py +376 -0
- dately/timezone.py +475 -0
- dately/utils.py +116 -0
- dately-1.0.0.dist-info/METADATA +31 -0
- dately-1.0.0.dist-info/RECORD +19 -0
- dately-1.0.0.dist-info/WHEEL +5 -0
- dately-1.0.0.dist-info/top_level.txt +1 -0
dately/__init__.py
ADDED
dately/core.py
ADDED
|
@@ -0,0 +1,507 @@
|
|
|
1
|
+
import threading
|
|
2
|
+
|
|
3
|
+
def import_specific_parts():
|
|
4
|
+
global tz
|
|
5
|
+
from .timezone import tz
|
|
6
|
+
|
|
7
|
+
thread = threading.Thread(target=import_specific_parts)
|
|
8
|
+
thread.start()
|
|
9
|
+
|
|
10
|
+
import datetime
|
|
11
|
+
import numpy as np
|
|
12
|
+
import pandas as pd
|
|
13
|
+
from copy import deepcopy
|
|
14
|
+
|
|
15
|
+
# Import all functions and classes from custom utility modules using relative imports
|
|
16
|
+
from .mold.pyd.cdatetime.whichformat import *
|
|
17
|
+
from .utils import *
|
|
18
|
+
from .timeutils import *
|
|
19
|
+
from .mold.pyd.cdatetime.UniversalDateFormatter import *
|
|
20
|
+
from .mold.pyd.cdatetime.iso8601T import isISOT as is_iso_date
|
|
21
|
+
from .mold.pyd.cdatetime.iso8601Z import replaceZ
|
|
22
|
+
from .mold.pyd.clean_str import *
|
|
23
|
+
from .mold.pyd.Compiled import (
|
|
24
|
+
datetime_regex as datetime_pattern_search,
|
|
25
|
+
anytime_regex,
|
|
26
|
+
timemeridiem_regex,
|
|
27
|
+
timeboundary_regex,
|
|
28
|
+
time_only_regex,
|
|
29
|
+
iana_timezone_identifier_regex,
|
|
30
|
+
timezone_offset_regex,
|
|
31
|
+
timezone_abbreviation_regex,
|
|
32
|
+
full_timezone_name_regex
|
|
33
|
+
)
|
|
34
|
+
thread.join()
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
class DatelyDate:
|
|
38
|
+
"""
|
|
39
|
+
A versatile date and datetime utility class for extracting, detecting, converting, and replacing components of date strings.
|
|
40
|
+
|
|
41
|
+
This class provides methods to handle various date and datetime formats, allowing for extraction of specific components,
|
|
42
|
+
detection and adjustment of date formats, conversion of dates with optional formatting and temporal adjustments,
|
|
43
|
+
and replacement of specific components within datetime strings. It supports operations on individual strings as well as
|
|
44
|
+
collections of date strings in lists, numpy arrays, and pandas Series.
|
|
45
|
+
|
|
46
|
+
Methods
|
|
47
|
+
-------
|
|
48
|
+
extract_datetime_component(date_strings, component, ret_format=False):
|
|
49
|
+
Extract specific date components from a single date string or a collection of date strings using precompiled regex patterns.
|
|
50
|
+
|
|
51
|
+
detect_date_format(date_strings):
|
|
52
|
+
Detect and adjust the date format based on the components of a single date string or a collection of date strings.
|
|
53
|
+
|
|
54
|
+
convert_date(dates, to_format=None, delta=0, dict_keys=None, dict_inplace=False):
|
|
55
|
+
Convert single or multiple date strings or datetime objects into a specified format or datetime objects, with optional date modification.
|
|
56
|
+
|
|
57
|
+
replace_timestring(datetime_strings, *args, **kwargs):
|
|
58
|
+
Modify various time components within a single datetime string or a collection of datetime strings, supporting both ISO and non-ISO formatted strings.
|
|
59
|
+
|
|
60
|
+
replace_datestring(date_strings, year=None, month=None, day=None):
|
|
61
|
+
Replace specific components in a date string or a collection of date strings with new values.
|
|
62
|
+
|
|
63
|
+
sequence(start_date, end_date, to_format='%b %-d, %Y'):
|
|
64
|
+
Generate a sequence of formatted dates between two dates.
|
|
65
|
+
|
|
66
|
+
"""
|
|
67
|
+
|
|
68
|
+
def __dir__(self):
|
|
69
|
+
# Override the __dir__ method to exclude private methods
|
|
70
|
+
original_dir = super().__dir__()
|
|
71
|
+
return [item for item in original_dir if not item.startswith('_DatelyDate__')]
|
|
72
|
+
|
|
73
|
+
def extract_datetime_component(self, date_strings, component, ret_format=False):
|
|
74
|
+
"""
|
|
75
|
+
Extract specific date components from a single date string or a collection of date strings using precompiled regex patterns.
|
|
76
|
+
|
|
77
|
+
This function parses a date string to extract specified components, ensuring accurate extraction
|
|
78
|
+
of year, month, day, hour, minute, and second components, which are critical for consistent date
|
|
79
|
+
formatting across different platforms.
|
|
80
|
+
|
|
81
|
+
Parameters:
|
|
82
|
+
date_strings (str): The date string to parse.
|
|
83
|
+
component (str): The component to extract. Valid components include:
|
|
84
|
+
'year', 'month', 'day', 'hour24', 'hour12', 'minutes', 'seconds'.
|
|
85
|
+
|
|
86
|
+
Returns:
|
|
87
|
+
str or None: The extracted component as a string, or None if no match is found.
|
|
88
|
+
|
|
89
|
+
Components:
|
|
90
|
+
- 'year': Year component of the date.
|
|
91
|
+
- 'month': Month component of the date.
|
|
92
|
+
- 'day': Day component of the date.
|
|
93
|
+
- 'weekday': Weekday component of the date.
|
|
94
|
+
- 'hour24': Hour component in 24-hour format.
|
|
95
|
+
- 'hour12': Hour component in 12-hour format.
|
|
96
|
+
- 'minute': Minute component of the time.
|
|
97
|
+
- 'second': Second component of the time.
|
|
98
|
+
- 'microsecond': Microsecond component of the time.
|
|
99
|
+
"""
|
|
100
|
+
def process(date_string, component, ret_format=False):
|
|
101
|
+
try:
|
|
102
|
+
detected_format = DateFormatFinder().search(date_string)
|
|
103
|
+
pattern = datetime_pattern_search(detected_format)
|
|
104
|
+
match = pattern.match(date_string)
|
|
105
|
+
if match and component in match.groupdict():
|
|
106
|
+
value = match.group(component)
|
|
107
|
+
if component == 'hour12' and 'am_pm' in match.groupdict():
|
|
108
|
+
am_pm = match.group('am_pm')
|
|
109
|
+
if am_pm == 'PM' and value != '12':
|
|
110
|
+
value = str(int(value) + 12)
|
|
111
|
+
elif am_pm == 'AM' and value == '12':
|
|
112
|
+
value = '00'
|
|
113
|
+
if ret_format:
|
|
114
|
+
return (detected_format, value)
|
|
115
|
+
return value
|
|
116
|
+
except ValueError:
|
|
117
|
+
return None
|
|
118
|
+
return None
|
|
119
|
+
|
|
120
|
+
if isinstance(date_strings, str):
|
|
121
|
+
return process(date_strings, component, ret_format)
|
|
122
|
+
elif isinstance(date_strings, list):
|
|
123
|
+
return [process(date_string, component, ret_format) for date_string in date_strings]
|
|
124
|
+
elif isinstance(date_strings, np.ndarray):
|
|
125
|
+
vectorized_func = np.vectorize(process, otypes=[np.object])
|
|
126
|
+
return vectorized_func(date_strings, component, ret_format)
|
|
127
|
+
elif isinstance(date_strings, pd.Series):
|
|
128
|
+
date_strings = date_strings.astype(str)
|
|
129
|
+
return date_strings.apply(process, args=(component, ret_format))
|
|
130
|
+
else:
|
|
131
|
+
raise ValueError("Unsupported data type. The input must be a scalar (str), list, numpy.ndarray, or pandas.Series.")
|
|
132
|
+
|
|
133
|
+
def detect_date_format(self, date_strings):
|
|
134
|
+
"""
|
|
135
|
+
Detect and adjust the date format based on the components of a single date string or a collection of date strings.
|
|
136
|
+
This function analyzes each date string to identify its format and makes adjustments to handle leading zeros in the date components.
|
|
137
|
+
It ensures consistent date formatting across different platforms by replacing zero-padded specifiers with their non-zero-padded
|
|
138
|
+
counterparts where applicable.
|
|
139
|
+
|
|
140
|
+
Parameters:
|
|
141
|
+
date_strings (str, list, np.ndarray, pd.Series): The date string or collection of date strings to analyze.
|
|
142
|
+
|
|
143
|
+
Returns:
|
|
144
|
+
str, list, np.ndarray, pd.Series: Depending on the input type, returns either a single format or a collection of formats with detected and possibly adjusted date format strings.
|
|
145
|
+
|
|
146
|
+
Raises:
|
|
147
|
+
ValueError: If no matching format is found for any of the given date strings.
|
|
148
|
+
"""
|
|
149
|
+
def process(date_string):
|
|
150
|
+
detected_format = DateFormatFinder().search(date_string)
|
|
151
|
+
try:
|
|
152
|
+
for comp_key, comp_details in zero_handling_date_formats().items():
|
|
153
|
+
component_value = self.extract_datetime_component(date_string, comp_key)
|
|
154
|
+
|
|
155
|
+
if has_leading_zero(component_value) is False:
|
|
156
|
+
detected_format = detected_format.replace(comp_details['zero_padded']['format'], comp_details['no_leading_zero']['format'])
|
|
157
|
+
return detected_format
|
|
158
|
+
except (ValueError, TypeError):
|
|
159
|
+
raise ValueError("No matching format found for the given date string.")
|
|
160
|
+
|
|
161
|
+
if isinstance(date_strings, str):
|
|
162
|
+
return process(date_strings)
|
|
163
|
+
elif isinstance(date_strings, list):
|
|
164
|
+
return [process(date_string) for date_string in date_strings]
|
|
165
|
+
elif isinstance(date_strings, np.ndarray):
|
|
166
|
+
vectorized_detect = np.vectorize(process, otypes=[np.object])
|
|
167
|
+
return vectorized_detect(date_strings)
|
|
168
|
+
elif isinstance(date_strings, pd.Series):
|
|
169
|
+
date_strings = date_strings.astype(str)
|
|
170
|
+
return date_strings.apply(process)
|
|
171
|
+
else:
|
|
172
|
+
raise ValueError("Unsupported data type. The input must be a scalar (str), list, numpy.ndarray, or pandas.Series.")
|
|
173
|
+
|
|
174
|
+
def convert_date(self, dates, to_format=None, delta=0, dict_keys=None, dict_inplace=False):
|
|
175
|
+
"""
|
|
176
|
+
This function serves as a versatile converter for date and datetime inputs. It supports converting single or
|
|
177
|
+
multiple date strings or datetime objects into a specified format or datetime objects, with the option to modify
|
|
178
|
+
the date by a given delta of days. Additionally, it handles dictionaries containing date information by applying
|
|
179
|
+
conversions recursively to specified keys.
|
|
180
|
+
|
|
181
|
+
This function is particularly useful in data preprocessing where dates might come in various formats and need
|
|
182
|
+
standardization or adjustment based on a temporal delta for further analysis or storage.
|
|
183
|
+
|
|
184
|
+
Parameters:
|
|
185
|
+
dates (str, list, np.ndarray, pd.Series, datetime.datetime, dict): The input date(s) which can be a single date string,
|
|
186
|
+
a datetime object, a collection (list, array, series) of date strings or datetime objects, or a dictionary containing
|
|
187
|
+
date strings or datetime objects nested under specified keys.
|
|
188
|
+
to_format (str, optional): The desired output format of the date(s) as a string according to datetime.strftime conventions.
|
|
189
|
+
If None, the function will return datetime objects instead of formatted strings.
|
|
190
|
+
delta (int, default=0): An integer representing the number of days to add or subtract from the input date(s). Positive
|
|
191
|
+
values move the date forward, while negative values move it backwards.
|
|
192
|
+
dict_keys (list, optional): When the 'dates' parameter is a dictionary, this list specifies which keys contain the
|
|
193
|
+
date information to be converted. This parameter is mandatory if 'dates' is a dictionary.
|
|
194
|
+
dict_inplace (bool, default=False): Determines whether the dictionary is modified in place. If True, the dictionary is
|
|
195
|
+
modified directly and the function returns None. If False, the function modifies a copy of the dictionary and returns it.
|
|
196
|
+
|
|
197
|
+
Returns:
|
|
198
|
+
The function returns the converted date(s) either as formatted strings (if 'to_format' is specified) or as datetime
|
|
199
|
+
objects. If the input is a dictionary and 'dict_inplace' is False, it returns a new dictionary with the dates converted.
|
|
200
|
+
If 'dict_inplace' is True, the input dictionary is modified directly and the function returns None.
|
|
201
|
+
|
|
202
|
+
Raises:
|
|
203
|
+
ValueError: If 'dates' is a dictionary and 'dict_keys' is not provided, or if any input date string format is unrecognized
|
|
204
|
+
or incorrect, making it impossible to parse the date.
|
|
205
|
+
"""
|
|
206
|
+
def process(date, to_format, delta):
|
|
207
|
+
if isinstance(date, (datetime.datetime, datetime.date)):
|
|
208
|
+
parsed_date = date + datetime.timedelta(days=int(delta))
|
|
209
|
+
else:
|
|
210
|
+
input_format = self.detect_date_format(date)
|
|
211
|
+
try:
|
|
212
|
+
parsed_date = datetime.datetime.strptime(date, input_format) + datetime.timedelta(days=int(delta))
|
|
213
|
+
except ValueError:
|
|
214
|
+
input_format = replace_non_padded_with_padded(input_format)
|
|
215
|
+
parsed_date = datetime.datetime.strptime(date, input_format) + datetime.timedelta(days=int(delta))
|
|
216
|
+
|
|
217
|
+
if to_format and isinstance(parsed_date, datetime.datetime):
|
|
218
|
+
return date_format_leading_zero(parsed_date, to_format)
|
|
219
|
+
else:
|
|
220
|
+
return parsed_date
|
|
221
|
+
|
|
222
|
+
def recursive_convert(data, keys):
|
|
223
|
+
if isinstance(data, dict):
|
|
224
|
+
for key, value in data.items():
|
|
225
|
+
if key in keys:
|
|
226
|
+
if isinstance(value, list):
|
|
227
|
+
data[key] = [process(item, to_format, delta) if not isinstance(item, (dict, list)) else recursive_convert(item, keys) for item in value]
|
|
228
|
+
else:
|
|
229
|
+
data[key] = process(value, to_format, delta)
|
|
230
|
+
elif isinstance(value, dict):
|
|
231
|
+
recursive_convert(value, keys)
|
|
232
|
+
elif isinstance(value, list):
|
|
233
|
+
for item in value:
|
|
234
|
+
if isinstance(item, dict):
|
|
235
|
+
recursive_convert(item, keys)
|
|
236
|
+
elif isinstance(data, list):
|
|
237
|
+
for i, item in enumerate(data):
|
|
238
|
+
if isinstance(item, dict):
|
|
239
|
+
data[i] = recursive_convert(item, keys)
|
|
240
|
+
return data
|
|
241
|
+
|
|
242
|
+
if isinstance(dates, dict):
|
|
243
|
+
if dict_keys is None:
|
|
244
|
+
raise ValueError("dict_keys must be provided when dates is a dictionary")
|
|
245
|
+
if not dict_inplace:
|
|
246
|
+
dates = deepcopy(dates)
|
|
247
|
+
processed_data = recursive_convert(dates, dict_keys)
|
|
248
|
+
if dict_inplace:
|
|
249
|
+
return
|
|
250
|
+
else:
|
|
251
|
+
return processed_data
|
|
252
|
+
|
|
253
|
+
elif isinstance(dates, (list, np.ndarray, pd.Series)):
|
|
254
|
+
if isinstance(dates, list):
|
|
255
|
+
return [process(date, to_format, delta) for date in dates]
|
|
256
|
+
elif isinstance(dates, np.ndarray):
|
|
257
|
+
vectorized_process = np.vectorize(process, excluded=['to_format', 'delta'], otypes=[np.object])
|
|
258
|
+
return vectorized_process(dates, to_format=to_format, delta=delta)
|
|
259
|
+
elif isinstance(dates, pd.Series):
|
|
260
|
+
return dates.apply(process, to_format=to_format, delta=delta)
|
|
261
|
+
else:
|
|
262
|
+
return process(dates, to_format, delta)
|
|
263
|
+
|
|
264
|
+
def _repl_timestring(self, datetime_strings, hour=None, minute=None, second=None, microsecond=None, tzinfo=None, time_indicator=None):
|
|
265
|
+
def process(datetime_string):
|
|
266
|
+
datetime_string = make_datetime_string(datetime_string)
|
|
267
|
+
if hour is not None:
|
|
268
|
+
datetime_string = replace_time_by_position(datetime_string, 'hour', hour)
|
|
269
|
+
if minute is not None:
|
|
270
|
+
datetime_string = replace_time_by_position(datetime_string, 'minute', minute)
|
|
271
|
+
if second is not None:
|
|
272
|
+
datetime_string = replace_time_by_position(datetime_string, 'second', second)
|
|
273
|
+
if microsecond is not None:
|
|
274
|
+
datetime_string = replace_time_by_position(datetime_string, 'microsecond', microsecond)
|
|
275
|
+
if tzinfo is not None:
|
|
276
|
+
datetime_string = replace_time_by_position(datetime_string, 'tzinfo', tzinfo)
|
|
277
|
+
if time_indicator == '':
|
|
278
|
+
return stripTimeIndicator(datetime_string)
|
|
279
|
+
|
|
280
|
+
if time_indicator is not None and time_indicator.upper() in ["AM", "PM"]:
|
|
281
|
+
timepattern = anytime_regex
|
|
282
|
+
timematch = timepattern.search(datetime_string)
|
|
283
|
+
if timematch:
|
|
284
|
+
time_fragment_str = timematch.group()
|
|
285
|
+
if not exist_meridiem(time_fragment_str):
|
|
286
|
+
datetime_string = datetime_string[:timematch.end()] + f' {time_indicator.upper()}' + datetime_string[timematch.end():]
|
|
287
|
+
result = validate_timezone(datetime_string)
|
|
288
|
+
if result[0] is False:
|
|
289
|
+
raise ValueError
|
|
290
|
+
return datetime_string
|
|
291
|
+
|
|
292
|
+
if isinstance(datetime_strings, str):
|
|
293
|
+
return process(datetime_strings)
|
|
294
|
+
elif isinstance(datetime_strings, list):
|
|
295
|
+
return [process(dt_string) for dt_string in datetime_strings]
|
|
296
|
+
elif isinstance(datetime_strings, np.ndarray):
|
|
297
|
+
vectorized_func = np.vectorize(process, otypes=[np.object])
|
|
298
|
+
return vectorized_func(datetime_strings)
|
|
299
|
+
elif isinstance(datetime_strings, pd.Series):
|
|
300
|
+
datetime_strings = datetime_strings.astype(str)
|
|
301
|
+
return datetime_strings.apply(process)
|
|
302
|
+
else:
|
|
303
|
+
raise ValueError("Unsupported data type. The input must be a scalar (str), list, numpy.ndarray, or pandas.Series.")
|
|
304
|
+
|
|
305
|
+
def __repl_iso_timestring(self, datetime_strings, hour=None, minute=None, second=None, microsecond=None, tzinfo=None):
|
|
306
|
+
def process(datetime_string, hour=None, minute=None, second=None, microsecond=None, tzinfo=None):
|
|
307
|
+
datetime_string = replaceZ(datetime_string)
|
|
308
|
+
dt = datetime.datetime.fromisoformat(datetime_string)
|
|
309
|
+
|
|
310
|
+
if isinstance(tzinfo, (int, float)):
|
|
311
|
+
tzinfo = datetime_offset(tzinfo)
|
|
312
|
+
|
|
313
|
+
new_dt = dt.replace(
|
|
314
|
+
hour=hour if hour is not None else dt.hour,
|
|
315
|
+
minute=minute if minute is not None else dt.minute,
|
|
316
|
+
second=second if second is not None else dt.second,
|
|
317
|
+
microsecond=microsecond if microsecond is not None else dt.microsecond,
|
|
318
|
+
tzinfo=tzinfo if tzinfo is not None else dt.tzinfo
|
|
319
|
+
)
|
|
320
|
+
return new_dt.isoformat()
|
|
321
|
+
|
|
322
|
+
if isinstance(datetime_strings, str):
|
|
323
|
+
return process(datetime_strings, hour, minute, second, microsecond, tzinfo)
|
|
324
|
+
elif isinstance(datetime_strings, list):
|
|
325
|
+
return [process(dt_string, hour, minute, second, microsecond, tzinfo) for dt_string in datetime_strings]
|
|
326
|
+
elif isinstance(datetime_strings, np.ndarray):
|
|
327
|
+
vectorized_func = np.vectorize(process, otypes=[np.object], excluded=['hour', 'minute', 'second', 'microsecond', 'tzinfo'])
|
|
328
|
+
return vectorized_func(datetime_strings, hour=hour, minute=minute, second=second, microsecond=microsecond, tzinfo=tzinfo)
|
|
329
|
+
elif isinstance(datetime_strings, pd.Series):
|
|
330
|
+
datetime_strings = datetime_strings.astype(str)
|
|
331
|
+
return datetime_strings.apply(lambda dt_string: process(dt_string, hour, minute, second, microsecond, tzinfo))
|
|
332
|
+
else:
|
|
333
|
+
raise ValueError("Unsupported data type. The input must be a scalar (str), list, numpy.ndarray, or pandas.Series.")
|
|
334
|
+
|
|
335
|
+
def replace_timestring(self, datetime_strings, *args, **kwargs):
|
|
336
|
+
"""
|
|
337
|
+
Modifies various time components within a single datetime string or a collection of datetime strings,
|
|
338
|
+
supporting both ISO and non-ISO formatted strings. This function is adaptable to handle updates to time
|
|
339
|
+
components including hours, minutes, seconds, microseconds, and time zones. It can also add a time indicator
|
|
340
|
+
(AM/PM) for non-ISO formats.
|
|
341
|
+
|
|
342
|
+
This utility is particularly useful in data processing workflows where datetime strings require uniform
|
|
343
|
+
time components across datasets, or adjustments to individual components are necessary for standardization,
|
|
344
|
+
time zone corrections, or formatting for further analysis or display.
|
|
345
|
+
|
|
346
|
+
Parameters:
|
|
347
|
+
datetime_strings (str, list, np.ndarray, pd.Series): The datetime string or collection of datetime
|
|
348
|
+
strings to be modified. This allows the function to integrate seamlessly into various data handling
|
|
349
|
+
contexts, whether the data is a single datetime string, a list from typical Python data manipulations,
|
|
350
|
+
a numpy array from numerical Python operations, or a pandas Series from dataframe manipulations.
|
|
351
|
+
hour (str or int, optional): New hour value, formatted as a string (0-23) or an integer.
|
|
352
|
+
Optional; if not provided, the hour is not modified.
|
|
353
|
+
minute (str or int, optional): New minute value, formatted as a string (0-59) or an integer.
|
|
354
|
+
Optional; if not provided, the minute is not modified.
|
|
355
|
+
second (str or int, optional): New second value, formatted as a string (0-59) or an integer.
|
|
356
|
+
Optional; if not provided, the second is not modified.
|
|
357
|
+
microsecond (str or int, optional): New microsecond value, formatted as a string or an integer.
|
|
358
|
+
Optional; if not provided, the microsecond is not modified.
|
|
359
|
+
tzinfo (str, timezone, int, or float, optional): New timezone information, formatted as a string
|
|
360
|
+
(e.g., "+0200", "UTC"), a timezone object (e.g., from pytz), or an offset in hours (int or float).
|
|
361
|
+
Optional; if not provided, the timezone is not modified.
|
|
362
|
+
time_indicator (str, optional): Time indicator to add to the datetime string, "AM" or "PM" only.
|
|
363
|
+
Optional; if not provided, no time indicator is added.
|
|
364
|
+
|
|
365
|
+
Returns:
|
|
366
|
+
str, list, np.ndarray, pd.Series: Depending on the input type, returns either a single modified datetime
|
|
367
|
+
string or a collection of modified datetime strings. This allows for easy integration of the function's
|
|
368
|
+
output back into the data processing pipeline, maintaining the original data structure for seamless
|
|
369
|
+
further processing.
|
|
370
|
+
|
|
371
|
+
Raises:
|
|
372
|
+
ValueError: If the input data type is not supported, or if the datetime string is not in the expected
|
|
373
|
+
format, or if the tzinfo is invalid, raises an error to ensure that the function usage is clear and safe
|
|
374
|
+
within expected data types.
|
|
375
|
+
"""
|
|
376
|
+
if is_iso_date(datetime_strings):
|
|
377
|
+
return self.__repl_iso_timestring(datetime_strings, *args, **kwargs)
|
|
378
|
+
else:
|
|
379
|
+
return self._repl_timestring(datetime_strings, *args, **kwargs)
|
|
380
|
+
|
|
381
|
+
def replace_datestring(self, date_strings, year=None, month=None, day=None):
|
|
382
|
+
"""
|
|
383
|
+
Replace specific components in a date string or a collection of date strings with new values.
|
|
384
|
+
|
|
385
|
+
This function parses a date string to identify existing components and replaces them with
|
|
386
|
+
new values provided as arguments. It reconstructs the date string with the new values.
|
|
387
|
+
|
|
388
|
+
Parameters:
|
|
389
|
+
date_strings (str, list, np.ndarray, pd.Series): The date string or collection to modify.
|
|
390
|
+
year (str or int or None): The new year value to replace the existing year.
|
|
391
|
+
month (str or int or None): The new month value to replace the existing month.
|
|
392
|
+
day (str or int or None): The new day value to replace the existing day.
|
|
393
|
+
|
|
394
|
+
Returns:
|
|
395
|
+
str, list, np.ndarray, pd.Series: The modified date string or collection with the new values.
|
|
396
|
+
"""
|
|
397
|
+
def process(date_string):
|
|
398
|
+
time_match = None
|
|
399
|
+
result = strTime(date_string)
|
|
400
|
+
|
|
401
|
+
if result:
|
|
402
|
+
time_match = result['full_time_details']['full_time_string']
|
|
403
|
+
fulltime_start = result['full_time_details']['start']
|
|
404
|
+
|
|
405
|
+
datestr = date_string[:fulltime_start]
|
|
406
|
+
date_string = cleanstr(datestr)
|
|
407
|
+
|
|
408
|
+
components = {
|
|
409
|
+
"year": (year, None),
|
|
410
|
+
"month": (month, None),
|
|
411
|
+
"day": (day, None)
|
|
412
|
+
}
|
|
413
|
+
|
|
414
|
+
detected_format = DateFormatFinder().search(date_string)
|
|
415
|
+
pattern = datetime_pattern_search(detected_format)
|
|
416
|
+
match = pattern.match(date_string)
|
|
417
|
+
if match:
|
|
418
|
+
for key in components.keys():
|
|
419
|
+
if key in match.groupdict():
|
|
420
|
+
components[key] = (components[key][0], match.span(key))
|
|
421
|
+
|
|
422
|
+
for key, (new_value, span) in sorted(components.items(), key=lambda item: item[1][1] if item[1][1] else (0, 0), reverse=True):
|
|
423
|
+
if new_value is not None and span:
|
|
424
|
+
start, end = span
|
|
425
|
+
date_string = date_string[:start] + str(new_value) + date_string[end:]
|
|
426
|
+
if time_match:
|
|
427
|
+
date_string += f' {time_match}'
|
|
428
|
+
|
|
429
|
+
if validate_date(date_string, date_format=detected_format) is False:
|
|
430
|
+
raise ValueError
|
|
431
|
+
|
|
432
|
+
return date_string
|
|
433
|
+
|
|
434
|
+
if isinstance(date_strings, str):
|
|
435
|
+
return process(date_strings)
|
|
436
|
+
elif isinstance(date_strings, list):
|
|
437
|
+
return [process(date_string) for date_string in date_strings]
|
|
438
|
+
elif isinstance(date_strings, np.ndarray):
|
|
439
|
+
vectorized_func = np.vectorize(process, otypes=[np.object])
|
|
440
|
+
return vectorized_func(date_strings)
|
|
441
|
+
elif isinstance(date_strings, pd.Series):
|
|
442
|
+
date_strings = date_strings.astype(str)
|
|
443
|
+
return date_strings.apply(process)
|
|
444
|
+
else:
|
|
445
|
+
raise ValueError("Unsupported data type. The input must be a scalar (str), list, numpy.ndarray, or pandas.Series.")
|
|
446
|
+
|
|
447
|
+
def _is_datetime(self, dt):
|
|
448
|
+
"""
|
|
449
|
+
Checks if the dt is a datetime object or an iterable of datetime objects.
|
|
450
|
+
|
|
451
|
+
Parameters:
|
|
452
|
+
dt (any): The date to check.
|
|
453
|
+
|
|
454
|
+
Returns:
|
|
455
|
+
bool: True if dt is a datetime object or an iterable of datetime objects, False otherwise.
|
|
456
|
+
"""
|
|
457
|
+
if isinstance(dt, (datetime.datetime, datetime.date)):
|
|
458
|
+
return True
|
|
459
|
+
if np.isscalar(dt):
|
|
460
|
+
return False
|
|
461
|
+
if isinstance(dt, (list, tuple)):
|
|
462
|
+
if all(isinstance(x, (datetime.datetime, datetime.date)) for x in dt):
|
|
463
|
+
return True
|
|
464
|
+
if isinstance(dt, pd.Series):
|
|
465
|
+
if dt.ndim == 1 and (pd.api.types._is_datetime64_any_dtype(dt) or pd.api.types.is_object_dtype(dt) and all(isinstance(x, (datetime.datetime, datetime.date)) for x in dt)):
|
|
466
|
+
return True
|
|
467
|
+
if isinstance(dt, np.ndarray):
|
|
468
|
+
if dt.ndim == 1 and (np.issubdtype(dt.dtype, np.datetime64) or np.issubdtype(dt.dtype, np.object_) and all(isinstance(x, (datetime.datetime, datetime.date)) for x in dt)):
|
|
469
|
+
return True
|
|
470
|
+
return False
|
|
471
|
+
|
|
472
|
+
def sequence(self, start_date, end_date, to_format='%b %-d, %Y'):
|
|
473
|
+
"""
|
|
474
|
+
Generate a sequence of formatted dates between two dates.
|
|
475
|
+
|
|
476
|
+
Parameters:
|
|
477
|
+
start_date (any): The start date of the sequence, which can be a datetime object or convertible to one.
|
|
478
|
+
end_date (any): The end date of the sequence, which can be a datetime object or convertible to one.
|
|
479
|
+
to_format (str, optional): The format string to use for formatting the dates.
|
|
480
|
+
Defaults to '%b %-d, %Y'.
|
|
481
|
+
|
|
482
|
+
Returns:
|
|
483
|
+
list of str: List of formatted date strings from start to end date inclusive.
|
|
484
|
+
"""
|
|
485
|
+
if not self._is_datetime(start_date):
|
|
486
|
+
start_date = self.convert_date(start_date)
|
|
487
|
+
if not self._is_datetime(end_date):
|
|
488
|
+
end_date = self.convert_date(end_date)
|
|
489
|
+
|
|
490
|
+
# Calculate the number of days between the start and end dates
|
|
491
|
+
delta = end_date - start_date
|
|
492
|
+
|
|
493
|
+
# Generate the list of dates and format them
|
|
494
|
+
date_list = [(start_date + datetime.timedelta(days=i)).strftime(to_format) for i in range(delta.days + 1)]
|
|
495
|
+
|
|
496
|
+
return date_list
|
|
497
|
+
|
|
498
|
+
|
|
499
|
+
# Instance
|
|
500
|
+
dt = DatelyDate()
|
|
501
|
+
|
|
502
|
+
|
|
503
|
+
# Define public interface
|
|
504
|
+
__all__ = [
|
|
505
|
+
"dt",
|
|
506
|
+
"tz"
|
|
507
|
+
]
|
dately/mold/__init__.py
ADDED
|
File without changes
|
|
Binary file
|
|
File without changes
|
|
Binary file
|
|
File without changes
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|