GJDutils 0.2.2__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- gjdutils/__init__.py +12 -0
- gjdutils/audios.py +39 -0
- gjdutils/cacheing.py +237 -0
- gjdutils/cmd.py +149 -0
- gjdutils/colab.py +39 -0
- gjdutils/collections.py +36 -0
- gjdutils/decorators.py +34 -0
- gjdutils/dicts.py +216 -0
- gjdutils/dsci.py +202 -0
- gjdutils/dt.py +296 -0
- gjdutils/env.py +64 -0
- gjdutils/errors.py +12 -0
- gjdutils/files.py +140 -0
- gjdutils/functions.py +6 -0
- gjdutils/google_translate.py +80 -0
- gjdutils/hashing.py +32 -0
- gjdutils/html.py +87 -0
- gjdutils/indexing.py +97 -0
- gjdutils/iterfunc.py +99 -0
- gjdutils/jsons.py +70 -0
- gjdutils/lists.py +13 -0
- gjdutils/llm_utils.py +167 -0
- gjdutils/llms_claude.py +131 -0
- gjdutils/llms_openai.py +299 -0
- gjdutils/misc.py +30 -0
- gjdutils/num.py +77 -0
- gjdutils/obsolete/google_text_to_speech.py +46 -0
- gjdutils/obsolete/llms_obsolete.py +298 -0
- gjdutils/outloud_text_to_speech.py +230 -0
- gjdutils/prompt_templates.py +20 -0
- gjdutils/pypi_build.py +112 -0
- gjdutils/pytest_utils.py +24 -0
- gjdutils/rand.py +65 -0
- gjdutils/regex.py +78 -0
- gjdutils/requirements_dev.txt +2 -0
- gjdutils/runtime.py +19 -0
- gjdutils/sets.py +5 -0
- gjdutils/shell.py +69 -0
- gjdutils/sorteddict.py +34 -0
- gjdutils/stopwatch.py +79 -0
- gjdutils/strings.py +218 -0
- gjdutils/todo/convert_parquet.py +28 -0
- gjdutils/typ.py +37 -0
- gjdutils/voice_speechrecognition.py +29 -0
- gjdutils/web.py +68 -0
- gjdutils-0.2.2.dist-info/METADATA +101 -0
- gjdutils-0.2.2.dist-info/RECORD +49 -0
- gjdutils-0.2.2.dist-info/WHEEL +4 -0
- gjdutils-0.2.2.dist-info/licenses/LICENSE +21 -0
gjdutils/dicts.py
ADDED
|
@@ -0,0 +1,216 @@
|
|
|
1
|
+
import json
|
|
2
|
+
from typing import Iterable, Optional
|
|
3
|
+
|
|
4
|
+
from gjdutils.strings import is_string, jinja_render
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
def compare_dict(d1, d2, ignore_underscores=True):
|
|
8
|
+
"""
|
|
9
|
+
Returns None if they're the same, else returns the first
|
|
10
|
+
key that differs.
|
|
11
|
+
"""
|
|
12
|
+
for k, v in d1.items():
|
|
13
|
+
if ignore_underscores and k.startswith("_"):
|
|
14
|
+
continue
|
|
15
|
+
if v != d2[k]:
|
|
16
|
+
return k
|
|
17
|
+
return None
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def dict_from_module(mod):
|
|
21
|
+
"""
|
|
22
|
+
e.g.
|
|
23
|
+
import config
|
|
24
|
+
config.BLAH
|
|
25
|
+
cfg = dict_from_module(config)
|
|
26
|
+
cfg['BLAH']
|
|
27
|
+
# OR
|
|
28
|
+
cfg = dict_from_module('config')
|
|
29
|
+
"""
|
|
30
|
+
if is_string(mod):
|
|
31
|
+
mod = __import__(mod)
|
|
32
|
+
else:
|
|
33
|
+
raise ValueError("Expected a string, got %s" % type(mod))
|
|
34
|
+
return {k: v for k, v in mod.__dict__.items() if not k.startswith("__")}
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def pop_copy(d, k):
|
|
38
|
+
"""
|
|
39
|
+
A non-destructive version of POP that returns both the
|
|
40
|
+
updated copy of the dictionary and the popped value,
|
|
41
|
+
e.g.
|
|
42
|
+
|
|
43
|
+
pop_copy({'a': 100, 'b': 200}, 'a') -> {'b': 200}, 100
|
|
44
|
+
"""
|
|
45
|
+
out = d.copy()
|
|
46
|
+
v = out.pop(k)
|
|
47
|
+
return out, v
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def pop_safe(d, k, default=None):
|
|
51
|
+
"""
|
|
52
|
+
Destructive, like the default pop, but returns DEFAULT
|
|
53
|
+
if K is not a key in D.
|
|
54
|
+
"""
|
|
55
|
+
return d.pop(k) if k in d else default
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def whittle_dict(d, keys):
|
|
59
|
+
d2 = {}
|
|
60
|
+
for k in keys:
|
|
61
|
+
d2[k] = d[k]
|
|
62
|
+
return d2
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def these_fields_only(d: dict, fields: Iterable[str], required=True):
|
|
66
|
+
"""
|
|
67
|
+
Returns D2, containing just the KEYS from dictionary D, e.g.
|
|
68
|
+
|
|
69
|
+
these_fields_only({'a': 100, 'b': 200}, ['a']) -> {'a': 100}
|
|
70
|
+
|
|
71
|
+
If REQUIRED, will raise an exception if any of KEYS aren't keys in D.
|
|
72
|
+
"""
|
|
73
|
+
# used to be called WHITTLE_DICT
|
|
74
|
+
if required:
|
|
75
|
+
# will fail if any of the fields in FIELDS are missing from D
|
|
76
|
+
d2 = {k: d[k] for k in fields}
|
|
77
|
+
else:
|
|
78
|
+
d2 = {k: d[k] for k in fields if k in d}
|
|
79
|
+
return d2
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def print_dict(d: dict):
|
|
83
|
+
print("\n".join(["%s: %s" % (k, d[k]) for k in sorted(d.keys())]))
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
def truncate_dict(d: dict, n: Optional[int], reverse=False) -> dict:
|
|
87
|
+
l = list(d.items())
|
|
88
|
+
if reverse:
|
|
89
|
+
l = list(reversed(l))
|
|
90
|
+
l_trunc = l[:n]
|
|
91
|
+
d_trunc = dict(l_trunc)
|
|
92
|
+
return d_trunc
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
# this is no longer needed. if you want to update a dict in an expression:
|
|
96
|
+
# d_both = {**d1, **d2}
|
|
97
|
+
# or
|
|
98
|
+
# d_both = d1 | d2
|
|
99
|
+
# def update_d(d1, d2): ...
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
def combine_dicts(d1: dict, d2: dict, require_unique: bool = True) -> dict:
|
|
103
|
+
"""
|
|
104
|
+
UPDATE: you can now do the simple version of this with `d1 | d2`
|
|
105
|
+
"""
|
|
106
|
+
if require_unique:
|
|
107
|
+
d1_keys = set(d1.keys())
|
|
108
|
+
d2_keys = set(d2.keys())
|
|
109
|
+
overlapping_k = set.intersection(d1_keys, d2_keys)
|
|
110
|
+
assert not overlapping_k, "Overlapping keys: %s" % overlapping_k
|
|
111
|
+
# d = d1.copy()
|
|
112
|
+
# d.update(d2)
|
|
113
|
+
d = d1 | d2
|
|
114
|
+
# check we've maintained ordering. only makes sense if REQUIRE_UNIQUE is False
|
|
115
|
+
# assert list(d1.keys()) + list(d2.keys()) == list(d.keys())
|
|
116
|
+
return d
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
def update_params_from_defaults(defaults: dict, params_in: dict):
|
|
120
|
+
for param_k in params_in.keys():
|
|
121
|
+
assert param_k in defaults.keys(), "Unexpected param '%s'" % param_k
|
|
122
|
+
params_out = combine_dicts(defaults, params_in, require_unique=False)
|
|
123
|
+
return params_out
|
|
124
|
+
|
|
125
|
+
|
|
126
|
+
def reverse_dict(d):
|
|
127
|
+
"""
|
|
128
|
+
Returns a dictionary with values as keys and vice versa.
|
|
129
|
+
|
|
130
|
+
Will fail if the values aren't unique or aren't
|
|
131
|
+
hashable.
|
|
132
|
+
"""
|
|
133
|
+
vals = d.values()
|
|
134
|
+
# confirm that all the values are unique
|
|
135
|
+
assert len(vals) == len(set(vals))
|
|
136
|
+
d2 = {}
|
|
137
|
+
for k, v in d.items():
|
|
138
|
+
d2[v] = k
|
|
139
|
+
return d2
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
class HashableDict(dict):
|
|
143
|
+
# http://code.activestate.com/recipes/414283-frozen-dictionaries/
|
|
144
|
+
def __hash__(self):
|
|
145
|
+
return hash(tuple(sorted(self.items())))
|
|
146
|
+
|
|
147
|
+
|
|
148
|
+
def rename_fields(d: dict, renames: dict):
|
|
149
|
+
"""
|
|
150
|
+
Renames multiple fields in a dictionary at the same time,
|
|
151
|
+
returning a new one, and preserving the order, e.g.
|
|
152
|
+
|
|
153
|
+
D = {'a': 100, 'b': 200}
|
|
154
|
+
RENAMES = {'a': 'c'}
|
|
155
|
+
->
|
|
156
|
+
D2 = {'c': 100, 'b': 200}
|
|
157
|
+
"""
|
|
158
|
+
d2 = {}
|
|
159
|
+
for k, v in d.items():
|
|
160
|
+
if k in renames.keys():
|
|
161
|
+
k2 = renames[k]
|
|
162
|
+
d2[k2] = v
|
|
163
|
+
else:
|
|
164
|
+
d2[k] = v
|
|
165
|
+
return d2
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
def dict_from_list(lst: list[dict], key, skip_duplicates: bool = False):
|
|
169
|
+
"""
|
|
170
|
+
Given a list of dicts LST, return a dict DIC where D[v] = an item IT in LST where lst[key]=v.
|
|
171
|
+
|
|
172
|
+
For example, given a list of EVENTS, return a dict where D[event_id] = event:
|
|
173
|
+
|
|
174
|
+
events_d = dict_from_list(events_l, key="id")
|
|
175
|
+
"""
|
|
176
|
+
dic = {}
|
|
177
|
+
for it in lst:
|
|
178
|
+
assert key in it, f"Key {key} not found in {it}"
|
|
179
|
+
v = it[key]
|
|
180
|
+
if skip_duplicates and (v in dic):
|
|
181
|
+
continue
|
|
182
|
+
assert v not in dic, f"Found duplicate key {v}"
|
|
183
|
+
dic[v] = it
|
|
184
|
+
return dic
|
|
185
|
+
|
|
186
|
+
|
|
187
|
+
def pprint_dict(d: dict):
|
|
188
|
+
return print(json.dumps(d, indent=4))
|
|
189
|
+
|
|
190
|
+
|
|
191
|
+
def dict_as_html(d: dict) -> str:
|
|
192
|
+
return jinja_render(
|
|
193
|
+
"""
|
|
194
|
+
<div class="dict-view">
|
|
195
|
+
<ul>
|
|
196
|
+
{% for key, value in d.items() %}
|
|
197
|
+
<li>
|
|
198
|
+
<strong>{{ key }}</strong>:
|
|
199
|
+
{% if value is mapping %}
|
|
200
|
+
{{ dict_as_html(value) | safe }}
|
|
201
|
+
{% elif value is sequence and value is not string %}
|
|
202
|
+
<ul>
|
|
203
|
+
{% for item in value %}
|
|
204
|
+
<li>{{ item }}</li>
|
|
205
|
+
{% endfor %}
|
|
206
|
+
</ul>
|
|
207
|
+
{% else %}
|
|
208
|
+
{{ value }}
|
|
209
|
+
{% endif %}
|
|
210
|
+
</li>
|
|
211
|
+
{% endfor %}
|
|
212
|
+
</ul>
|
|
213
|
+
</div>
|
|
214
|
+
""",
|
|
215
|
+
context={"d": d, "dict_as_html": dict_as_html},
|
|
216
|
+
)
|
gjdutils/dsci.py
ADDED
|
@@ -0,0 +1,202 @@
|
|
|
1
|
+
from collections import Counter
|
|
2
|
+
import itertools
|
|
3
|
+
import numpy as np
|
|
4
|
+
import pandas as pd
|
|
5
|
+
import random
|
|
6
|
+
from scipy import spatial, sparse
|
|
7
|
+
import sys
|
|
8
|
+
from typing import Any, Iterable, Sequence
|
|
9
|
+
|
|
10
|
+
from .rand import set_seeds
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def init_display_options():
|
|
14
|
+
# so that it's easier to see things in the terminal
|
|
15
|
+
pd.set_option("display.max_rows", 10000)
|
|
16
|
+
pd.set_option("display.max_columns", 1000)
|
|
17
|
+
pd.set_option("display.max_colwidth", 100)
|
|
18
|
+
np.set_printoptions(precision=2)
|
|
19
|
+
np.set_printoptions(suppress=True)
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def jaccard_similarity(list1, list2) -> float:
|
|
23
|
+
"""
|
|
24
|
+
For comparing how much overlap there is between two sets.
|
|
25
|
+
|
|
26
|
+
Returns a normalised 0-1 similarity score, where higher = more similar.
|
|
27
|
+
|
|
28
|
+
USAGE:
|
|
29
|
+
a = ['hello', 'foo', 'foo', 'tux']
|
|
30
|
+
b = ['blah', 'hello', 'foo']
|
|
31
|
+
jaccard_similarity(a, b)
|
|
32
|
+
|
|
33
|
+
from https://stackoverflow.com/a/56774335
|
|
34
|
+
"""
|
|
35
|
+
intersection = len(set(list1).intersection(list2))
|
|
36
|
+
union = len(set(list1)) + len(set(list2)) - intersection
|
|
37
|
+
return intersection / union
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def calc_proportion_identical(lst: Any) -> float:
|
|
41
|
+
"""
|
|
42
|
+
Returns a value between 0 and 1 for the uniformity of the values
|
|
43
|
+
in LST, i.e. higher if they're all the same.
|
|
44
|
+
"""
|
|
45
|
+
|
|
46
|
+
def count_most_common(lst):
|
|
47
|
+
"""
|
|
48
|
+
Find the most common item in LST, and count how many times it occurs.
|
|
49
|
+
"""
|
|
50
|
+
# Counter(['a', 'b', 'a']).most_common(2) -> [
|
|
51
|
+
# ('a', 2),
|
|
52
|
+
# ('b', 1),
|
|
53
|
+
# ]
|
|
54
|
+
# so this gives the count of the most common (in this case 2 occurrences of 'a')
|
|
55
|
+
return Counter(lst).most_common(1)[0][1]
|
|
56
|
+
|
|
57
|
+
most_common = count_most_common(lst)
|
|
58
|
+
if most_common == 1:
|
|
59
|
+
return 0
|
|
60
|
+
else:
|
|
61
|
+
return most_common / len(lst)
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def calc_normalised_std_tightness(vals: Sequence[float]) -> float:
|
|
65
|
+
"""
|
|
66
|
+
The standard deviation STD is in the same units as VALS, i.e.
|
|
67
|
+
it's unnormalised. We normalise by the (absolute) mean,
|
|
68
|
+
subtract from 1, and truncate.
|
|
69
|
+
|
|
70
|
+
This gives us a unbounded 'tightness' score,
|
|
71
|
+
i.e. 1 means no variability, 0 means a lot of variability, e.g.
|
|
72
|
+
|
|
73
|
+
[19, 21, 20, 20] -> 0.96
|
|
74
|
+
[19, 1, 40, 20] -> 0.31
|
|
75
|
+
[ 9, 1, 70, 0] -> 0
|
|
76
|
+
"""
|
|
77
|
+
n = len(vals)
|
|
78
|
+
if n == 0:
|
|
79
|
+
raise Exception("Empty")
|
|
80
|
+
elif n == 1:
|
|
81
|
+
return 1.0
|
|
82
|
+
if n == 2:
|
|
83
|
+
deviation = abs(vals[0] - vals[1])
|
|
84
|
+
else:
|
|
85
|
+
deviation = float(np.std(vals))
|
|
86
|
+
|
|
87
|
+
average = abs(sum(vals) / n)
|
|
88
|
+
if average < 0.01:
|
|
89
|
+
# e.g. mean([-50, 50]) -> 0
|
|
90
|
+
# risking a divide-by-zero, which could produce unstable results.
|
|
91
|
+
# better to default to treating as not part of the cluster?
|
|
92
|
+
return 0
|
|
93
|
+
|
|
94
|
+
normalised_std = deviation / average
|
|
95
|
+
tightness = 1 - min(1, normalised_std)
|
|
96
|
+
assert 0 <= tightness <= 1
|
|
97
|
+
return tightness
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def calc_pair_amounts_closeness(amounts: Sequence[float]) -> float:
|
|
101
|
+
"""
|
|
102
|
+
Returns higher the closer the two numbers.
|
|
103
|
+
|
|
104
|
+
Returns 0 if one number is zero but the other isn't,
|
|
105
|
+
or if they're of different signs.
|
|
106
|
+
"""
|
|
107
|
+
assert len(amounts) == 2
|
|
108
|
+
amount1, amount2 = max(amounts), min(amounts)
|
|
109
|
+
if amount1 == 0.0 and amount2 == 0.0:
|
|
110
|
+
# avoid divide-by-zero
|
|
111
|
+
return 1.0
|
|
112
|
+
if amount1 > 0 and amount2 < 0:
|
|
113
|
+
# because a debit and a credit are never similar, no matter what their values
|
|
114
|
+
return 0.0
|
|
115
|
+
if amount1 < 0:
|
|
116
|
+
# if it's negative, they're both negative, and this only works
|
|
117
|
+
# for positive numbers, so swap sign (and therefore max/min
|
|
118
|
+
# will be swapped too)
|
|
119
|
+
amount1, amount2 = abs(amount2), abs(amount1)
|
|
120
|
+
val = 1 - (amount1 - amount2) / (amount1 + amount2)
|
|
121
|
+
assert 0 <= val <= 1
|
|
122
|
+
return val
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
def convert_sim_dist_reciprocal(val: float) -> float:
|
|
126
|
+
"""
|
|
127
|
+
Convert from similarity to distance with 1/x, dealing with divide-by-zero.
|
|
128
|
+
"""
|
|
129
|
+
assert 0 <= val <= 1
|
|
130
|
+
out = sys.maxsize if val == 0 else (1 / val)
|
|
131
|
+
assert 0 <= out <= 1
|
|
132
|
+
return out
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
def convert_sim_dist_oneminus(val: float) -> float:
|
|
136
|
+
"""
|
|
137
|
+
Convert from similarity to distance with 1 - x.
|
|
138
|
+
"""
|
|
139
|
+
assert 0 <= val <= 1
|
|
140
|
+
out = 1 - val
|
|
141
|
+
assert 0 <= out <= 1
|
|
142
|
+
return out
|
|
143
|
+
|
|
144
|
+
|
|
145
|
+
def square_df_from_square(sq, features):
|
|
146
|
+
df = pd.DataFrame(sq)
|
|
147
|
+
# create a 'Feature' column
|
|
148
|
+
df["Feature"] = features
|
|
149
|
+
rename_dict = dict(zip(range(len(features)), features))
|
|
150
|
+
df.rename(columns=rename_dict, inplace=True)
|
|
151
|
+
# df = df.reindex_axis(['Feature'] + features, axis=1)
|
|
152
|
+
df = df.set_index("Feature")
|
|
153
|
+
return df
|
|
154
|
+
|
|
155
|
+
|
|
156
|
+
def long_df_from_flat(dists_flat, features):
|
|
157
|
+
combos = [(f1, f2) for f1, f2 in itertools.permutations(features, 2)]
|
|
158
|
+
assert len(combos) == len(dists_flat)
|
|
159
|
+
dists_triplet = [
|
|
160
|
+
(combo[0], combo[1], dist) for combo, dist in zip(combos, dists_flat)
|
|
161
|
+
]
|
|
162
|
+
dists_long_df = pd.DataFrame(dists_triplet, columns=["Feature", "Brand", "Score"])
|
|
163
|
+
return dists_long_df
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
def square_df_from_flat(dists_flat, features):
|
|
167
|
+
nFeatures = len(features)
|
|
168
|
+
dists_sq = spatial.distance.squareform(np.array(dists_flat))
|
|
169
|
+
assert dists_sq.shape == (nFeatures, nFeatures)
|
|
170
|
+
return square_df_from_square(dists_sq, features)
|
|
171
|
+
|
|
172
|
+
|
|
173
|
+
def pairwise_local(data, distance_func, format="long"):
|
|
174
|
+
"""
|
|
175
|
+
Returns squareform pairwise distances DataFrame (run symmetrically).
|
|
176
|
+
|
|
177
|
+
DATA should be a iterable of arrays (e.g. a list of bitarrays).
|
|
178
|
+
Pairs of rows from DATA will be passed into DISTANCE_FUNC, which
|
|
179
|
+
should return a float.
|
|
180
|
+
|
|
181
|
+
If format 'square' (default), returns an (nFeatures x nFeatures)
|
|
182
|
+
square distances matrix.
|
|
183
|
+
|
|
184
|
+
If format 'flat', returns a vector of distances (that could be fed
|
|
185
|
+
into scipy squareform to produce the 'square' version).
|
|
186
|
+
|
|
187
|
+
If format 'long', returns recs weighted-sum model format.
|
|
188
|
+
"""
|
|
189
|
+
features = sorted(data.keys())
|
|
190
|
+
dists_flat = [
|
|
191
|
+
distance_func(data[f1], data[f2])
|
|
192
|
+
for f1, f2 in itertools.permutations(features, 2)
|
|
193
|
+
]
|
|
194
|
+
if format == "flat":
|
|
195
|
+
return dists_flat
|
|
196
|
+
if format == "long":
|
|
197
|
+
return long_df_from_flat(dists_flat, features)
|
|
198
|
+
elif format == "square":
|
|
199
|
+
dists_sq_df = square_df_from_flat(dists_flat, features)
|
|
200
|
+
return dists_sq_df
|
|
201
|
+
else:
|
|
202
|
+
raise Exception("Unknown FORMAT %s" % format)
|
gjdutils/dt.py
ADDED
|
@@ -0,0 +1,296 @@
|
|
|
1
|
+
from calendar import monthrange
|
|
2
|
+
from datetime import datetime, date, timedelta
|
|
3
|
+
import pendulum
|
|
4
|
+
from typing import Optional, Union
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
def dt_str(
|
|
8
|
+
dt: Optional[datetime] = None, seconds: bool = True, tz: Optional[str] = None
|
|
9
|
+
) -> str:
|
|
10
|
+
"""
|
|
11
|
+
e.g. 2020-Nov-18 at 7:39:20pm -> '201118_1939_20'
|
|
12
|
+
|
|
13
|
+
If TZ is None, defaults to UTC. Or set e.g. 'Europe/London'.
|
|
14
|
+
"""
|
|
15
|
+
if dt is None:
|
|
16
|
+
dt = pendulum.now(tz=tz)
|
|
17
|
+
else:
|
|
18
|
+
dt = pendulum.instance(dt, tz=tz)
|
|
19
|
+
format = "YYMMDD_HHmm_ss" if seconds else "YYMMDD_HHmm"
|
|
20
|
+
return dt.format(format)
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
# def dt_str(dt=None, hoursmins=True, seconds=True):
|
|
24
|
+
# """
|
|
25
|
+
# Returns the current date/time as a yymmdd_HHMM_S string,
|
|
26
|
+
# e.g. 091016_1916_21 for 16th Oct, 2009, at 7.16pm in the
|
|
27
|
+
# evening.
|
|
28
|
+
|
|
29
|
+
# By default, returns for NOW, unless you feed in DT.
|
|
30
|
+
# """
|
|
31
|
+
# if dt is None:
|
|
32
|
+
# dt = datetime.datetime.now()
|
|
33
|
+
# fmt = "%y%m%d"
|
|
34
|
+
# if hoursmins:
|
|
35
|
+
# fmt += "_%H%M"
|
|
36
|
+
# if seconds:
|
|
37
|
+
# fmt += "_%S"
|
|
38
|
+
# return dt.strftime(fmt)
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def str_dt(s):
|
|
42
|
+
"""
|
|
43
|
+
Returns the current date/time as a DATETIME object, when
|
|
44
|
+
fed in a YYYYmmdd_HHMM_S string. See DT_STR.
|
|
45
|
+
"""
|
|
46
|
+
now = datetime.now()
|
|
47
|
+
try:
|
|
48
|
+
# TODO this was a django function. need to find a replacement
|
|
49
|
+
return now.strptime(s, "%Y%m%d_%H%M_%S")
|
|
50
|
+
except ValueError:
|
|
51
|
+
# without seconds
|
|
52
|
+
try:
|
|
53
|
+
return now.strptime(s, "%Y%m%d_%H%M")
|
|
54
|
+
except ValueError:
|
|
55
|
+
# without time at all
|
|
56
|
+
return now.strptime(s, "%Y%m%d")
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def pendulum_from_date(date: Union[datetime, date]) -> date:
|
|
60
|
+
"""
|
|
61
|
+
It's easier to always work with Pendulum objects, so
|
|
62
|
+
convert to that (works from Datetime, Date, or Pendulum).
|
|
63
|
+
"""
|
|
64
|
+
return pendulum.date(date.year, date.month, date.day)
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def timedelta_float(td, units="days"):
|
|
68
|
+
"""
|
|
69
|
+
Returns a float of timedelta TD in UNITS
|
|
70
|
+
(either 'days' or 'seconds').
|
|
71
|
+
|
|
72
|
+
Can be negative for things in the past.
|
|
73
|
+
|
|
74
|
+
timedelta returns the number of
|
|
75
|
+
days and the number of seconds, but you have to combine
|
|
76
|
+
them to get a float timedelta.
|
|
77
|
+
|
|
78
|
+
e.g. timedelta_float(now() - dt_last_week) == c. 7.0
|
|
79
|
+
"""
|
|
80
|
+
# 86400 = number of seconds in a day
|
|
81
|
+
if units == "days":
|
|
82
|
+
return td.days + td.seconds / 86400.0
|
|
83
|
+
elif units == "seconds":
|
|
84
|
+
return td.days * 86400.0 + td.seconds
|
|
85
|
+
else:
|
|
86
|
+
raise Exception("Unknown units %s" % units)
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def serialize_datetimes(d, level=0):
|
|
90
|
+
"""
|
|
91
|
+
Recursively walks through a dictionary, serializing datetime objects
|
|
92
|
+
into ISO 8601 formatted strings.
|
|
93
|
+
|
|
94
|
+
Set a (somewhat arbitrary) maximum of 20 levels of dictionaries. We
|
|
95
|
+
should never get to more than that, but if we do, it will just stop
|
|
96
|
+
serializing datetimes.
|
|
97
|
+
"""
|
|
98
|
+
if level > 20:
|
|
99
|
+
raise ValueError("Too many levels trying to serialize datetimes")
|
|
100
|
+
|
|
101
|
+
for k in d.keys():
|
|
102
|
+
if type(d[k]) == datetime:
|
|
103
|
+
d[k] = d[k].isoformat()
|
|
104
|
+
elif type(d[k]) == dict:
|
|
105
|
+
d[k] = serialize_datetimes(d[k], level + 1)
|
|
106
|
+
else:
|
|
107
|
+
pass
|
|
108
|
+
return d
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
def datetime_to_date(dt):
|
|
112
|
+
return date(year=dt.year, month=dt.month, day=dt.day)
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
def date_to_datetime(dt):
|
|
116
|
+
if isinstance(dt, datetime):
|
|
117
|
+
return dt
|
|
118
|
+
return datetime(year=dt.year, month=dt.month, day=dt.day)
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
def month_name(dt):
|
|
122
|
+
return datetime.strftime(dt, "%B")
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
def near_in_time(dt1, dt2=None):
|
|
126
|
+
"""
|
|
127
|
+
Compares two datetime and ensures that they're within 1s
|
|
128
|
+
of each other. Doesn't care which came first. Useful for unit tests.
|
|
129
|
+
"""
|
|
130
|
+
if dt2 is None:
|
|
131
|
+
dt2 = datetime.now()
|
|
132
|
+
dt_diff = abs(dt1 - dt2)
|
|
133
|
+
return dt_diff.days == 0 and dt_diff.seconds < 1
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
def pp_date(dt):
|
|
137
|
+
"""
|
|
138
|
+
Human-readable (i.e. pretty-print) dates, e.g. for spreadsheets:
|
|
139
|
+
|
|
140
|
+
See http://docs.python.org/tutorial/stdlib.html
|
|
141
|
+
|
|
142
|
+
e.g. 31-Oct-2011
|
|
143
|
+
"""
|
|
144
|
+
d = date_to_datetime(dt)
|
|
145
|
+
return d.strftime("%d-%b-%Y")
|
|
146
|
+
|
|
147
|
+
|
|
148
|
+
def humanize_minutes(minutes: int):
|
|
149
|
+
"""
|
|
150
|
+
e.g.
|
|
151
|
+
humanize_minutes(5) -> '5 minutes'
|
|
152
|
+
humanize_minutes(61) -> '1 hour'
|
|
153
|
+
humanize_minutes(1500) -> 'a day'
|
|
154
|
+
from https://github.com/python-humanize/humanize
|
|
155
|
+
"""
|
|
156
|
+
from humanize import naturalday, naturaldelta
|
|
157
|
+
|
|
158
|
+
delta = timedelta(minutes=minutes)
|
|
159
|
+
# minimum_unit="minutes" is not supported
|
|
160
|
+
ndelta = naturaldelta(delta)
|
|
161
|
+
return ndelta
|
|
162
|
+
|
|
163
|
+
|
|
164
|
+
def alltime():
|
|
165
|
+
# return YourModel.happened.order_by('dt')[0].dt
|
|
166
|
+
#
|
|
167
|
+
# hardcode to avoid the query
|
|
168
|
+
return datetime(year=2009, month=10, day=31, hour=22, minute=34, second=6)
|
|
169
|
+
|
|
170
|
+
|
|
171
|
+
def first_last_day_of_month(dt):
|
|
172
|
+
"""
|
|
173
|
+
Returns two DATETIMES, one for the first and one for the
|
|
174
|
+
last day of the month of DT.
|
|
175
|
+
"""
|
|
176
|
+
first_day = datetime(year=dt.year, month=dt.month, day=1)
|
|
177
|
+
nDays = monthrange(dt.year, dt.month)[1]
|
|
178
|
+
last_day = datetime(year=dt.year, month=dt.month, day=nDays)
|
|
179
|
+
return first_day, last_day
|
|
180
|
+
|
|
181
|
+
|
|
182
|
+
def recent_hour(nHours=1, dt=None):
|
|
183
|
+
"""
|
|
184
|
+
Returns the DT for 1 hour (i.e. 3600 seconds) ago.
|
|
185
|
+
"""
|
|
186
|
+
seconds = nHours * 3600
|
|
187
|
+
dt = dt or datetime.now()
|
|
188
|
+
return dt - timedelta(seconds=seconds)
|
|
189
|
+
|
|
190
|
+
|
|
191
|
+
def recent_day(nDays=1, dt=None):
|
|
192
|
+
"""
|
|
193
|
+
Returns the DT for 1 day (i.e. 24 hours) ago.
|
|
194
|
+
|
|
195
|
+
If NDAYS == 24 hours * NDAYS.
|
|
196
|
+
"""
|
|
197
|
+
dt = dt or datetime.now()
|
|
198
|
+
return dt - timedelta(days=nDays)
|
|
199
|
+
|
|
200
|
+
|
|
201
|
+
def start_of_day(dt=None):
|
|
202
|
+
"""
|
|
203
|
+
Returns the Datetime for DT at midnight, i.e. the start of the day.
|
|
204
|
+
"""
|
|
205
|
+
dt = dt or datetime.now()
|
|
206
|
+
return datetime(year=dt.year, month=dt.month, day=dt.day)
|
|
207
|
+
|
|
208
|
+
|
|
209
|
+
def end_of_day(dt=None):
|
|
210
|
+
dt = dt or datetime.now()
|
|
211
|
+
return datetime(
|
|
212
|
+
year=dt.year, month=dt.month, day=dt.day, hour=23, minute=59, second=59
|
|
213
|
+
)
|
|
214
|
+
|
|
215
|
+
|
|
216
|
+
def start_of_week(dt=None):
|
|
217
|
+
"""
|
|
218
|
+
Returns the DT for the beginning of the week (i.e. the most recent Monday at 00:01.
|
|
219
|
+
"""
|
|
220
|
+
# weekday(): Monday = 0. http://docs.python.org/library/datetime.html
|
|
221
|
+
dt = dt or datetime.now()
|
|
222
|
+
# subtract however many days since Monday from today to get to Monday
|
|
223
|
+
return start_of_day(dt - timedelta(days=dt.weekday()))
|
|
224
|
+
|
|
225
|
+
|
|
226
|
+
def end_of_week(dt=None):
|
|
227
|
+
dt = dt or datetime.now()
|
|
228
|
+
return end_of_day(dt + timedelta(days=(6 - dt.weekday())))
|
|
229
|
+
|
|
230
|
+
|
|
231
|
+
def start_of_month(dt=None):
|
|
232
|
+
dt = dt or datetime.now()
|
|
233
|
+
# xxx - we could have also used:
|
|
234
|
+
# start_of_day(now - timedelta(days=now.day))
|
|
235
|
+
return first_last_day_of_month(dt)[0]
|
|
236
|
+
|
|
237
|
+
|
|
238
|
+
def end_of_month(dt=None):
|
|
239
|
+
dt = dt or datetime.now()
|
|
240
|
+
return first_last_day_of_month(dt)[1] + timedelta(hours=23, minutes=59, seconds=59)
|
|
241
|
+
|
|
242
|
+
|
|
243
|
+
def day_containing(dt=None):
|
|
244
|
+
"""Return the half-open day interval containing dt.
|
|
245
|
+
|
|
246
|
+
i.e. if dt is Today 12:26, return (Today 00:00, Tomorrow 00:00).
|
|
247
|
+
This can be used for a half-open comparison:
|
|
248
|
+
|
|
249
|
+
p, n = day_containing()
|
|
250
|
+
if x >= p and x < n:
|
|
251
|
+
# Do something because x is today.
|
|
252
|
+
"""
|
|
253
|
+
|
|
254
|
+
p = start_of_day(dt)
|
|
255
|
+
n = p + timedelta(days=1)
|
|
256
|
+
|
|
257
|
+
return p, n
|
|
258
|
+
|
|
259
|
+
|
|
260
|
+
def daily_iter(start, end):
|
|
261
|
+
"""Iterate over half-open day intervals pairwise until the end of the range falls after end."""
|
|
262
|
+
|
|
263
|
+
p = start
|
|
264
|
+
n = start + timedelta(days=1)
|
|
265
|
+
|
|
266
|
+
while n < end:
|
|
267
|
+
yield p, n
|
|
268
|
+
p = n
|
|
269
|
+
n = n + timedelta(days=1)
|
|
270
|
+
|
|
271
|
+
|
|
272
|
+
def week_containing(dt=None):
|
|
273
|
+
"""Returns a half-open interval of the week containing dt, starting on Sunday."""
|
|
274
|
+
|
|
275
|
+
p = start_of_week(dt)
|
|
276
|
+
n = p + timedelta(days=7)
|
|
277
|
+
return p, n
|
|
278
|
+
|
|
279
|
+
|
|
280
|
+
def weekly_iter(start, end):
|
|
281
|
+
"""Iterate over weeks pairwise until the end of the range falls after end."""
|
|
282
|
+
|
|
283
|
+
p = start
|
|
284
|
+
n = start + timedelta(days=7)
|
|
285
|
+
|
|
286
|
+
while n < end:
|
|
287
|
+
yield p, n
|
|
288
|
+
p = n
|
|
289
|
+
n = n + timedelta(days=7)
|
|
290
|
+
|
|
291
|
+
|
|
292
|
+
def date_from_datetime(d: date, as_pendulum: bool = False):
|
|
293
|
+
if as_pendulum:
|
|
294
|
+
return pendulum.date(d.year, d.month, d.day)
|
|
295
|
+
else:
|
|
296
|
+
return datetime(d.year, d.month, d.day, hour=0, minute=0)
|