GJDutils 0.2.2__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- gjdutils/__init__.py +12 -0
- gjdutils/audios.py +39 -0
- gjdutils/cacheing.py +237 -0
- gjdutils/cmd.py +149 -0
- gjdutils/colab.py +39 -0
- gjdutils/collections.py +36 -0
- gjdutils/decorators.py +34 -0
- gjdutils/dicts.py +216 -0
- gjdutils/dsci.py +202 -0
- gjdutils/dt.py +296 -0
- gjdutils/env.py +64 -0
- gjdutils/errors.py +12 -0
- gjdutils/files.py +140 -0
- gjdutils/functions.py +6 -0
- gjdutils/google_translate.py +80 -0
- gjdutils/hashing.py +32 -0
- gjdutils/html.py +87 -0
- gjdutils/indexing.py +97 -0
- gjdutils/iterfunc.py +99 -0
- gjdutils/jsons.py +70 -0
- gjdutils/lists.py +13 -0
- gjdutils/llm_utils.py +167 -0
- gjdutils/llms_claude.py +131 -0
- gjdutils/llms_openai.py +299 -0
- gjdutils/misc.py +30 -0
- gjdutils/num.py +77 -0
- gjdutils/obsolete/google_text_to_speech.py +46 -0
- gjdutils/obsolete/llms_obsolete.py +298 -0
- gjdutils/outloud_text_to_speech.py +230 -0
- gjdutils/prompt_templates.py +20 -0
- gjdutils/pypi_build.py +112 -0
- gjdutils/pytest_utils.py +24 -0
- gjdutils/rand.py +65 -0
- gjdutils/regex.py +78 -0
- gjdutils/requirements_dev.txt +2 -0
- gjdutils/runtime.py +19 -0
- gjdutils/sets.py +5 -0
- gjdutils/shell.py +69 -0
- gjdutils/sorteddict.py +34 -0
- gjdutils/stopwatch.py +79 -0
- gjdutils/strings.py +218 -0
- gjdutils/todo/convert_parquet.py +28 -0
- gjdutils/typ.py +37 -0
- gjdutils/voice_speechrecognition.py +29 -0
- gjdutils/web.py +68 -0
- gjdutils-0.2.2.dist-info/METADATA +101 -0
- gjdutils-0.2.2.dist-info/RECORD +49 -0
- gjdutils-0.2.2.dist-info/WHEEL +4 -0
- gjdutils-0.2.2.dist-info/licenses/LICENSE +21 -0
gjdutils/env.py
ADDED
|
@@ -0,0 +1,64 @@
|
|
|
1
|
+
import os
|
|
2
|
+
from pathlib import Path
|
|
3
|
+
from typing import Any, TypeVar, cast
|
|
4
|
+
from pydantic import StrictStr, TypeAdapter
|
|
5
|
+
|
|
6
|
+
T = TypeVar("T")
|
|
7
|
+
_processed_vars = set()
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
def get_env_var(name: str, typ: Any = StrictStr) -> T:
|
|
11
|
+
"""Get environment variable with type validation, e.g.
|
|
12
|
+
|
|
13
|
+
OPENAI_API_KEY = get_env_var("OPENAI_API_KEY")
|
|
14
|
+
NUM_WORKERS = get_env_var("NUM_WORKERS", typ=int)
|
|
15
|
+
|
|
16
|
+
Args:
|
|
17
|
+
name: Name of environment variable
|
|
18
|
+
type_: Pydantic type to validate against (default: StrictStr for non-empty string)
|
|
19
|
+
|
|
20
|
+
Returns:
|
|
21
|
+
The validated value with the specified type
|
|
22
|
+
|
|
23
|
+
Raises:
|
|
24
|
+
ValueError: If variable is missing or fails validation
|
|
25
|
+
"""
|
|
26
|
+
try:
|
|
27
|
+
value = os.environ[name]
|
|
28
|
+
_processed_vars.add(name)
|
|
29
|
+
|
|
30
|
+
# Use TypeAdapter for validation
|
|
31
|
+
adapter = TypeAdapter(typ)
|
|
32
|
+
validated = adapter.validate_python(value)
|
|
33
|
+
|
|
34
|
+
# Return validated value directly
|
|
35
|
+
return cast(T, validated)
|
|
36
|
+
except KeyError:
|
|
37
|
+
raise ValueError(f"Missing required environment variable: {name}")
|
|
38
|
+
except Exception as e:
|
|
39
|
+
raise ValueError(f"Invalid value for {name}: {e}")
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def list_env_example_vars(env_example_filen: Path) -> set[str]:
|
|
43
|
+
"""Get set of required variables from .env.example.
|
|
44
|
+
|
|
45
|
+
Args:
|
|
46
|
+
env_example_filen: Path to the .env.example file
|
|
47
|
+
|
|
48
|
+
Returns:
|
|
49
|
+
Set of environment variable names found in the file
|
|
50
|
+
"""
|
|
51
|
+
assert env_example_filen.exists(), f"Missing env example file: {env_example_filen}"
|
|
52
|
+
|
|
53
|
+
required_vars = set()
|
|
54
|
+
with env_example_filen.open() as f:
|
|
55
|
+
for line in f:
|
|
56
|
+
line = line.strip()
|
|
57
|
+
# Skip comments and empty lines
|
|
58
|
+
if not line or line.startswith("#"):
|
|
59
|
+
continue
|
|
60
|
+
# Get variable name (everything before =)
|
|
61
|
+
var_name = line.split("=")[0].strip()
|
|
62
|
+
required_vars.add(var_name)
|
|
63
|
+
|
|
64
|
+
return required_vars
|
gjdutils/errors.py
ADDED
gjdutils/files.py
ADDED
|
@@ -0,0 +1,140 @@
|
|
|
1
|
+
import os
|
|
2
|
+
from pathlib import Path
|
|
3
|
+
from typing import Sequence
|
|
4
|
+
|
|
5
|
+
from .cmd import run_cmd
|
|
6
|
+
from .strings import is_string, PathOrStr
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
def split_filen(filen: Path | str):
|
|
10
|
+
"""
|
|
11
|
+
Splits a filename into its path, stem, and extension (without dot), e.g.
|
|
12
|
+
|
|
13
|
+
split_filen('data/blah.mp4') -> ('data', 'blah', 'mp4')
|
|
14
|
+
"""
|
|
15
|
+
filen = Path(filen)
|
|
16
|
+
return filen.parent, filen.stem, filen.suffix[1:] if filen.suffix else ""
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def create_dir_if_not_exists(dirn: str):
|
|
20
|
+
if not os.path.exists(dirn):
|
|
21
|
+
os.makedirs(dirn)
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def validate_ext(ext):
|
|
25
|
+
assert is_string(ext)
|
|
26
|
+
assert ext.lower() == ext
|
|
27
|
+
assert ext
|
|
28
|
+
assert ext[0] != "."
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def validate_dir(dirn):
|
|
32
|
+
dirn_path = Path(dirn)
|
|
33
|
+
assert dirn_path.exists() and dirn_path.is_dir()
|
|
34
|
+
return dirn_path
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def fulltext(
|
|
38
|
+
filens: Sequence[str],
|
|
39
|
+
patterns: list[str],
|
|
40
|
+
dirn: str,
|
|
41
|
+
file_ext: str,
|
|
42
|
+
case_sensitive=False,
|
|
43
|
+
):
|
|
44
|
+
"""
|
|
45
|
+
Returns: FOUND_FILES (list of filename strings)
|
|
46
|
+
|
|
47
|
+
Feed in a list of filenames (complete with extensions),
|
|
48
|
+
which will be fed to agrep for full-text
|
|
49
|
+
searching. Returns a list of files.
|
|
50
|
+
|
|
51
|
+
FILENS is a list of strings. If its non-empty, then
|
|
52
|
+
these will be fed in to agrep. If it's empty, then we'll
|
|
53
|
+
just feed in a '*.[freex_extension]'. Spaces in
|
|
54
|
+
filenames are escaped with backslashes, but this is the
|
|
55
|
+
only thing we're escaping.
|
|
56
|
+
|
|
57
|
+
PATTERNS is a list of strings, which will be ANDed
|
|
58
|
+
together in the agrep regex. Currently, this doesn't
|
|
59
|
+
escape the pattern regex at all, though it does surround it in
|
|
60
|
+
quotes, so the usual agrep rules apply.
|
|
61
|
+
|
|
62
|
+
Unless case_sensitive==True, will append a -i flag.
|
|
63
|
+
"""
|
|
64
|
+
# from freex_sqlalchemy.py
|
|
65
|
+
|
|
66
|
+
# xxx this should check that all the files have extensions
|
|
67
|
+
|
|
68
|
+
if case_sensitive:
|
|
69
|
+
case_flag = ""
|
|
70
|
+
else:
|
|
71
|
+
case_flag = "-i"
|
|
72
|
+
|
|
73
|
+
# xxx should check that all the items in the pattern
|
|
74
|
+
# list are strings...
|
|
75
|
+
#
|
|
76
|
+
# first strip each of the pattern strings of whitespace,
|
|
77
|
+
# and remove the surrounding quotes - we'll add them
|
|
78
|
+
# back to the whole pattern_str when we create the CMD
|
|
79
|
+
#
|
|
80
|
+
# then AND together multiple patterns with agrep,
|
|
81
|
+
# using semicolons
|
|
82
|
+
for pat in patterns:
|
|
83
|
+
if pat[0] == '"':
|
|
84
|
+
pat = pat[1:]
|
|
85
|
+
if pat[-1] == '"':
|
|
86
|
+
pat = pat[0:-1]
|
|
87
|
+
|
|
88
|
+
pattern_str = ";".join([x.strip() for x in patterns])
|
|
89
|
+
|
|
90
|
+
if len(filens) > 0:
|
|
91
|
+
# escape all the spaces with back-slashes
|
|
92
|
+
filens = [x.replace(" ", "\\ ") for x in filens]
|
|
93
|
+
|
|
94
|
+
# convert to a space-delimited string (with spaces
|
|
95
|
+
# escaped by backslashes), and each file prepended by the
|
|
96
|
+
# database_dir, e.g.
|
|
97
|
+
# /blah/test0.freex /blah/hello\ world.freex
|
|
98
|
+
fnames_str = " ".join([os.path.join(dirn, filen) for filen in filens])
|
|
99
|
+
|
|
100
|
+
# the -l says to just return filenames only (no text
|
|
101
|
+
# context)
|
|
102
|
+
#
|
|
103
|
+
# put the pattern in quotes
|
|
104
|
+
#
|
|
105
|
+
# and then just list the files at the end
|
|
106
|
+
cmd = 'agrep -l %s "%s" %s' % (case_flag, pattern_str, fnames_str)
|
|
107
|
+
|
|
108
|
+
else:
|
|
109
|
+
# if we're not restricting the files we're looking
|
|
110
|
+
# through, then there could be too many files to run
|
|
111
|
+
# agrep on directly, so we have to pipe it from a
|
|
112
|
+
# find
|
|
113
|
+
#
|
|
114
|
+
# this is to avoid the '/usr/local/bin/agrep:
|
|
115
|
+
# Argument list too long' error
|
|
116
|
+
cmd = 'find %s -name "*.%s" -print0 | xargs -0 agrep -l %s "%s"' % (
|
|
117
|
+
dirn,
|
|
118
|
+
file_ext,
|
|
119
|
+
case_flag,
|
|
120
|
+
pattern_str,
|
|
121
|
+
)
|
|
122
|
+
|
|
123
|
+
# Run command with minimal output unless there's an error
|
|
124
|
+
retcode, out_str, _ = run_cmd(
|
|
125
|
+
cmd,
|
|
126
|
+
verbose=0,
|
|
127
|
+
check=False, # Don't raise exception if no matches found (agrep returns 1)
|
|
128
|
+
)
|
|
129
|
+
|
|
130
|
+
if len(out_str) > 0:
|
|
131
|
+
# strip away the path to yield just the filename for
|
|
132
|
+
# each of the files in out_str
|
|
133
|
+
found_files = [os.path.basename(x) for x in out_str.strip().split("\n")]
|
|
134
|
+
else:
|
|
135
|
+
# if you run the above on an empty string, you get
|
|
136
|
+
# [''], whereas we really want to return an empty
|
|
137
|
+
# list if we didn't find anything
|
|
138
|
+
found_files = []
|
|
139
|
+
|
|
140
|
+
return found_files
|
gjdutils/functions.py
ADDED
|
@@ -0,0 +1,80 @@
|
|
|
1
|
+
from cachetools import cached, LRUCache, TTLCache
|
|
2
|
+
from google.cloud import translate_v2 as translate
|
|
3
|
+
import html
|
|
4
|
+
from typing import Optional
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
def translate_text(
|
|
8
|
+
text: str,
|
|
9
|
+
lang_src_code: Optional[str],
|
|
10
|
+
lang_tgt_code: str,
|
|
11
|
+
verbose: int = 0,
|
|
12
|
+
):
|
|
13
|
+
"""Translates text into the target language.
|
|
14
|
+
|
|
15
|
+
Target must be an ISO 639-1 language code.
|
|
16
|
+
See https://g.co/cloud/translate/v2/translate-reference#supported_languages
|
|
17
|
+
"""
|
|
18
|
+
translate_client = translate.Client()
|
|
19
|
+
|
|
20
|
+
lang_src_code = (
|
|
21
|
+
lang_src_code[:2].lower() if isinstance(lang_src_code, str) else None
|
|
22
|
+
)
|
|
23
|
+
lang_tgt_code = lang_tgt_code[:2].lower()
|
|
24
|
+
if lang_src_code == lang_tgt_code:
|
|
25
|
+
return text, None
|
|
26
|
+
|
|
27
|
+
# assert lang_src_code != lang_tgt_code, (
|
|
28
|
+
# "Identical src and tgt language codes: %s" % lang_src_code
|
|
29
|
+
# )
|
|
30
|
+
|
|
31
|
+
# Text can also be a sequence of strings, in which case this method
|
|
32
|
+
# will return a sequence of results for each text.
|
|
33
|
+
if lang_src_code is None:
|
|
34
|
+
result = translate_client.translate(text, target_language=lang_tgt_code)
|
|
35
|
+
else:
|
|
36
|
+
result = translate_client.translate(
|
|
37
|
+
text,
|
|
38
|
+
target_language=lang_tgt_code,
|
|
39
|
+
source_language=lang_src_code,
|
|
40
|
+
)
|
|
41
|
+
|
|
42
|
+
translated_text = result["translatedText"]
|
|
43
|
+
|
|
44
|
+
# fix escaping, e.g.
|
|
45
|
+
# I've done it a week with no improvement
|
|
46
|
+
# ->
|
|
47
|
+
# I've done it a week with no improvement
|
|
48
|
+
translated_text = html.unescape(translated_text)
|
|
49
|
+
|
|
50
|
+
if verbose > 0:
|
|
51
|
+
print(f"{lang_src_code} -> {lang_tgt_code}")
|
|
52
|
+
print(f"\t\"{result['input']}\" -> \"{translated_text}\"")
|
|
53
|
+
if lang_src_code is None:
|
|
54
|
+
print(f"\t\tDetected source language: {result['detectedSourceLanguage']}")
|
|
55
|
+
|
|
56
|
+
return translated_text, result
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
# translated_text, result = translate_text(
|
|
60
|
+
# text="Hello, world",
|
|
61
|
+
# lang_src_code="en",
|
|
62
|
+
# lang_tgt_code="el",
|
|
63
|
+
# verbose=0,
|
|
64
|
+
# )
|
|
65
|
+
# translated_text
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
@cached(cache={})
|
|
69
|
+
def detect_language(text: str, verbose: int = 0) -> tuple[str, dict]:
|
|
70
|
+
"""
|
|
71
|
+
Detects the text's language.
|
|
72
|
+
"""
|
|
73
|
+
translate_client = translate.Client()
|
|
74
|
+
|
|
75
|
+
# Text can also be a sequence of strings, in which case this method
|
|
76
|
+
# will return a sequence of results for each text.
|
|
77
|
+
result = translate_client.detect_language(text)
|
|
78
|
+
language, confidence = result["language"], result["confidence"]
|
|
79
|
+
print(f"Ran detect_language for {text} -> {language} at confidence {confidence}")
|
|
80
|
+
return language, confidence
|
gjdutils/hashing.py
ADDED
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
import base64
|
|
2
|
+
import hashlib
|
|
3
|
+
|
|
4
|
+
|
|
5
|
+
def hash_readable(s, n=10):
|
|
6
|
+
"""
|
|
7
|
+
Returns a string hash that contains base32 characters instead of a number,
|
|
8
|
+
to make it more readable (and still low risk of collisions if you truncate it).
|
|
9
|
+
|
|
10
|
+
e.g. hash_readable('hello') => 'vl2mmho4yx'
|
|
11
|
+
|
|
12
|
+
Unlike Python's default hash function, this should be deterministic
|
|
13
|
+
across sessions (because we're using 'hashlib').
|
|
14
|
+
|
|
15
|
+
I'm using this for anonymising email addresses if I don't have a user UUID.
|
|
16
|
+
"""
|
|
17
|
+
if isinstance(s, str):
|
|
18
|
+
s = bytes(s, "utf-8")
|
|
19
|
+
hasher = hashlib.sha1(s)
|
|
20
|
+
b32 = base64.b32encode(hasher.digest())[:n]
|
|
21
|
+
return b32.decode("utf-8").lower()
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def hash_consistent(obj):
|
|
25
|
+
"""
|
|
26
|
+
Supposedly gives the same response every time you call it, even after restarting the kernel.
|
|
27
|
+
|
|
28
|
+
N.B. This is based on output from GitHub Copilot, and I haven't tried it.
|
|
29
|
+
"""
|
|
30
|
+
obj_str = str(obj)
|
|
31
|
+
hash_obj = hashlib.sha256(obj_str.encode()).hexdigest()
|
|
32
|
+
return hash_obj
|
gjdutils/html.py
ADDED
|
@@ -0,0 +1,87 @@
|
|
|
1
|
+
from bs4 import BeautifulSoup
|
|
2
|
+
from lxml import html as lxml_html
|
|
3
|
+
from lxml.html import tostring, fromstring
|
|
4
|
+
from lxml.etree import Element, _Element as ElementType
|
|
5
|
+
from typing import Optional, Iterable, Union
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
def remove_html_tags(html: str):
|
|
9
|
+
soup = BeautifulSoup(html, "html.parser")
|
|
10
|
+
return soup.get_text()
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def contents_of_body(soup):
|
|
14
|
+
"""
|
|
15
|
+
e.g.
|
|
16
|
+
BeautifulSoup('<p>hello</p><p>world</world>', features='lxml')
|
|
17
|
+
=>
|
|
18
|
+
<p>hello</p>
|
|
19
|
+
<p>world</p>
|
|
20
|
+
|
|
21
|
+
N.B. for html.parser, you might just be able to do: str(soup)
|
|
22
|
+
"""
|
|
23
|
+
# it might be better to prettify with body hidden=True???
|
|
24
|
+
return "\n".join([str(t) for t in soup.body.contents])
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def compare_html(h1, h2):
|
|
28
|
+
h1p = BeautifulSoup(h1, features="html.parser").prettify().strip()
|
|
29
|
+
h2p = BeautifulSoup(h2, features="html.parser").prettify().strip()
|
|
30
|
+
assert h1p == h2p
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def remove_attrs_from_html(h):
|
|
34
|
+
"""
|
|
35
|
+
Gets rid of all the attrs in the html.
|
|
36
|
+
"""
|
|
37
|
+
soup = BeautifulSoup(h, features="lxml")
|
|
38
|
+
for t in soup.recursiveChildGenerator():
|
|
39
|
+
t.attrs = {} # type: ignore
|
|
40
|
+
# whitespace_from_linebreaks(
|
|
41
|
+
# contents_of_body(soup)
|
|
42
|
+
# )
|
|
43
|
+
return contents_of_body(soup)
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def adjust_indentation(pretty_html, indent: int):
|
|
47
|
+
# from https://www.perplexity.ai/search/can-you-customise-the-beautifu-225tf.pISaiggsL5tNL.gA
|
|
48
|
+
lines = pretty_html.split("\n")
|
|
49
|
+
adjusted_lines = []
|
|
50
|
+
for line in lines:
|
|
51
|
+
line_lstrip = line.lstrip(" ")
|
|
52
|
+
leading_spaces = len(line) - len(line_lstrip)
|
|
53
|
+
indent_level = leading_spaces // 1 # default indent is 1 space
|
|
54
|
+
adjusted_lines.append(" " * (indent_level * indent) + line_lstrip)
|
|
55
|
+
return "\n".join(adjusted_lines)
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def prettify_html(
|
|
59
|
+
html: Union[str, ElementType, list[ElementType]], # BeautifulSoup
|
|
60
|
+
indent: int = 2,
|
|
61
|
+
n: Optional[int] = None, # number of chars to show
|
|
62
|
+
):
|
|
63
|
+
# if isinstance(html, pq):
|
|
64
|
+
# html = html.outer_html() # type: ignore
|
|
65
|
+
if isinstance(html, list):
|
|
66
|
+
# then we'll handle it as a string in a moment
|
|
67
|
+
html = "".join(
|
|
68
|
+
[tostring(e, method="html").decode() for e in html]
|
|
69
|
+
) # type: ignore
|
|
70
|
+
if isinstance(html, ElementType):
|
|
71
|
+
# this will do some cleaning and fixing. but
|
|
72
|
+
# you need document_fromstring() if you want to make sure
|
|
73
|
+
# that it's a full html doc, e.g. with html, body
|
|
74
|
+
html = tostring(html, method="html").decode()
|
|
75
|
+
|
|
76
|
+
soup = BeautifulSoup(html, "html.parser")
|
|
77
|
+
# html2 = tostring(html, pretty_print=True, method="html").decode() # type: ignore
|
|
78
|
+
# the lxml pretty_print just isn't as good as BS4, e.g. with a list of elements
|
|
79
|
+
# it wraps things in a div fragment, but the pretty-print of that isn't right
|
|
80
|
+
html2 = soup.prettify()
|
|
81
|
+
prettified = adjust_indentation(html2, indent=indent)[:n] # type: ignore
|
|
82
|
+
return prettified
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def pprettify_html(*args, **kwargs) -> None:
|
|
86
|
+
html = prettify_html(*args, **kwargs)
|
|
87
|
+
print(html)
|
gjdutils/indexing.py
ADDED
|
@@ -0,0 +1,97 @@
|
|
|
1
|
+
from decimal import Decimal, getcontext
|
|
2
|
+
from typing import Optional
|
|
3
|
+
|
|
4
|
+
"""
|
|
5
|
+
For manual ordering and reording in a database:
|
|
6
|
+
- every item gets a Decimal location (LOC) between 0 and 1
|
|
7
|
+
- the LOCs of all items are sorted
|
|
8
|
+
- the LOC of a new item is calculated as the average of the LOCs of the items before and after it
|
|
9
|
+
- the LOC won't ever be 0 or 1, so there will always be a gap for you to insert afterwards
|
|
10
|
+
- if you insert at the beginning or end, the LOC will be half of the first or last item's LOC
|
|
11
|
+
|
|
12
|
+
I think Figma used this.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
# Set precision high enough to handle many divisions
|
|
16
|
+
getcontext().prec = 28
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def locs_for(n: int) -> list[Decimal]:
|
|
20
|
+
# Assign initial idx values with buffers at both ends
|
|
21
|
+
locs = []
|
|
22
|
+
for i in range(n):
|
|
23
|
+
loc = Decimal(i + 1) / Decimal(n + 1)
|
|
24
|
+
locs.append(loc)
|
|
25
|
+
assert len(locs) == n
|
|
26
|
+
return locs
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def loc_for_insert_at(locs: list[Decimal], position: int, do_insert: bool = True):
|
|
30
|
+
"""
|
|
31
|
+
TODO: rewrite in terms of LOC_BETWEEN
|
|
32
|
+
"""
|
|
33
|
+
assert locs == sorted(
|
|
34
|
+
locs
|
|
35
|
+
), f"Input LOCS are unsorted, so things are already broken - {locs}"
|
|
36
|
+
list_length = len(locs)
|
|
37
|
+
if position == 0: # Insert at the beginning
|
|
38
|
+
newloc = locs[0] / 2 if list_length > 0 else Decimal("0.5")
|
|
39
|
+
elif position >= list_length: # Insert at the end
|
|
40
|
+
newloc = locs[-1] + (1 - locs[-1]) / 2 if list_length > 0 else Decimal("0.5")
|
|
41
|
+
elif position < 0:
|
|
42
|
+
raise Exception(f"Position must be non-negative, but got {position}")
|
|
43
|
+
else: # Insert between two items
|
|
44
|
+
newloc = (locs[position - 1] + locs[position]) / 2
|
|
45
|
+
if do_insert:
|
|
46
|
+
locs.insert(position, newloc)
|
|
47
|
+
assert locs == sorted(locs), f"Somehow we've broken the LOCS sorting: {locs}"
|
|
48
|
+
return newloc
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def loc_for_insert_at2(locs: list[Decimal], position: int, do_insert: bool = True):
|
|
52
|
+
"""
|
|
53
|
+
Functional version of LOC_FOR_INSERT_AT that returns a new list instead of
|
|
54
|
+
modifying the input list.
|
|
55
|
+
|
|
56
|
+
Uses LOC_BETWEEN.
|
|
57
|
+
"""
|
|
58
|
+
assert locs == sorted(
|
|
59
|
+
locs
|
|
60
|
+
), f"Input LOCS are unsorted, so things are already broken - {locs}"
|
|
61
|
+
list_length = len(locs)
|
|
62
|
+
if position < 0:
|
|
63
|
+
raise Exception(f"Position must be non-negative, but got {position}")
|
|
64
|
+
elif position == 0: # Insert at the beginning
|
|
65
|
+
loc1 = None
|
|
66
|
+
loc2 = locs[0] if list_length > 0 else None
|
|
67
|
+
elif position >= list_length: # Insert at the end
|
|
68
|
+
loc1 = locs[-1] if list_length > 0 else None
|
|
69
|
+
loc2 = None
|
|
70
|
+
else: # Insert between two items
|
|
71
|
+
loc1 = locs[position - 1]
|
|
72
|
+
loc2 = locs[position]
|
|
73
|
+
newloc = loc_between(loc1, loc2)
|
|
74
|
+
if do_insert:
|
|
75
|
+
locs.insert(position, newloc)
|
|
76
|
+
assert locs == sorted(locs), f"Somehow we've broken the LOCS sorting: {locs}"
|
|
77
|
+
return newloc
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def loc_between(loc1: Optional[Decimal], loc2: Optional[Decimal]) -> Decimal:
|
|
81
|
+
if loc1 is not None and loc2 is not None:
|
|
82
|
+
assert (
|
|
83
|
+
loc1 >= 0 and loc2 <= 1
|
|
84
|
+
), f"LOCs must be between 0 and 1, but got {loc1} and {loc2}"
|
|
85
|
+
return (loc1 + loc2) / 2
|
|
86
|
+
elif loc1 is None and loc2 is not None:
|
|
87
|
+
return loc2 / 2
|
|
88
|
+
elif loc1 is not None and loc2 is None:
|
|
89
|
+
return loc1 + (1 - loc1) / 2
|
|
90
|
+
elif loc1 is None and loc2 is None:
|
|
91
|
+
return Decimal("0.5")
|
|
92
|
+
else:
|
|
93
|
+
raise Exception(f"This should never happen: {loc1}, {loc2}")
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def disp(locs: list[Decimal]):
|
|
97
|
+
print(", ".join(["%.3f" % loc for loc in locs]))
|
gjdutils/iterfunc.py
ADDED
|
@@ -0,0 +1,99 @@
|
|
|
1
|
+
import itertools
|
|
2
|
+
from typing import Sequence
|
|
3
|
+
|
|
4
|
+
|
|
5
|
+
def contiguous_pairs(lst: Sequence):
|
|
6
|
+
"""
|
|
7
|
+
Given a list LST, return the contiguous pairs, e.g.
|
|
8
|
+
|
|
9
|
+
[10, 20, 30, 40, 50]
|
|
10
|
+
->
|
|
11
|
+
[(10, 20), (20, 30), (30, 40), (40, 50)]
|
|
12
|
+
|
|
13
|
+
(from GitHub Copilot)
|
|
14
|
+
"""
|
|
15
|
+
pairs = [(lst[i], lst[i + 1]) for i in range(len(lst) - 1)]
|
|
16
|
+
return pairs
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def flatten(lol):
|
|
20
|
+
"""
|
|
21
|
+
See http://stackoverflow.com/questions/406121/flattening-a-shallow-list-in-python
|
|
22
|
+
|
|
23
|
+
e.g. [['image00', 'image01'], ['image10'], []] -> ['image00', 'image01', 'image10']
|
|
24
|
+
"""
|
|
25
|
+
|
|
26
|
+
chain = list(itertools.chain(*lol))
|
|
27
|
+
return chain
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
# def flatten(list_of_lists):
|
|
31
|
+
# """
|
|
32
|
+
# Flatten one level of nesting
|
|
33
|
+
|
|
34
|
+
# from https://docs.python.org/3/library/itertools.html#itertools-recipes
|
|
35
|
+
# """
|
|
36
|
+
# return list(chain.from_iterable(list_of_lists))
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def unique(items):
|
|
40
|
+
"""
|
|
41
|
+
Returns KEEP, a list based on ITEMS, but with duplicates
|
|
42
|
+
removed (preserving order, based on first new example).
|
|
43
|
+
|
|
44
|
+
http://stackoverflow.com/questions/89178/in-python-what-is-the-fastest-algorithm-for-removing-duplicates-from-a-list-so-t
|
|
45
|
+
|
|
46
|
+
unique([1, 1, 2, 'a', 'a', 3]) -> [1, 2, 'a', 3]
|
|
47
|
+
"""
|
|
48
|
+
found = set([])
|
|
49
|
+
keep = []
|
|
50
|
+
for item in items:
|
|
51
|
+
if item not in found:
|
|
52
|
+
found.add(item)
|
|
53
|
+
keep.append(item)
|
|
54
|
+
return keep
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def uniquify_list(lst):
|
|
58
|
+
"""Return a list of the elements in s, but without duplicates, preserving order.
|
|
59
|
+
|
|
60
|
+
from comment in http://aspn.activestate.com/ASPN/Cookbook/Python/Recipe/52560
|
|
61
|
+
|
|
62
|
+
Lightweight and fast ..., Raymond Hettinger, 2002/03/17
|
|
63
|
+
|
|
64
|
+
"""
|
|
65
|
+
|
|
66
|
+
set = {}
|
|
67
|
+
return [set.setdefault(e, e) for e in lst if e not in set]
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def grouper(iterable, n):
|
|
71
|
+
"""
|
|
72
|
+
Collect data into fixed-length chunks or blocks. If
|
|
73
|
+
the last block is too small, returns a truncated block.
|
|
74
|
+
|
|
75
|
+
e.g. grouper('ABCDEFG', 3) --> ABC DEF G
|
|
76
|
+
|
|
77
|
+
From http://stackoverflow.com/a/8991553/230523
|
|
78
|
+
"""
|
|
79
|
+
it = iter(iterable)
|
|
80
|
+
while True:
|
|
81
|
+
chunk = tuple(itertools.islice(it, n))
|
|
82
|
+
if not chunk:
|
|
83
|
+
return
|
|
84
|
+
yield chunk
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
def grouper_ragged(iterable, n):
|
|
88
|
+
"""
|
|
89
|
+
Collect data into non-overlapping chunks - the last one might be shorter than the others
|
|
90
|
+
|
|
91
|
+
>>> print(list(grouper('ABCDEFG', 3))) # [('A', 'B', 'C'), ('D', 'E', 'F'), ('G',)]
|
|
92
|
+
|
|
93
|
+
from https://stackoverflow.com/a/41333827/230523
|
|
94
|
+
"""
|
|
95
|
+
it = iter(iterable)
|
|
96
|
+
group = tuple(itertools.islice(it, n))
|
|
97
|
+
while group:
|
|
98
|
+
yield group
|
|
99
|
+
group = tuple(itertools.islice(it, n))
|
gjdutils/jsons.py
ADDED
|
@@ -0,0 +1,70 @@
|
|
|
1
|
+
import json
|
|
2
|
+
from typing import Optional
|
|
3
|
+
|
|
4
|
+
|
|
5
|
+
def jsonify(x):
|
|
6
|
+
def json_dumper_robust(obj):
|
|
7
|
+
try:
|
|
8
|
+
return obj.toJSON()
|
|
9
|
+
except:
|
|
10
|
+
try:
|
|
11
|
+
return str(obj)
|
|
12
|
+
except:
|
|
13
|
+
return None
|
|
14
|
+
|
|
15
|
+
return json.dumps(x, sort_keys=True, indent=4, default=json_dumper_robust)
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
# from o-1
|
|
19
|
+
# class RobustJSONEncoder(json.JSONEncoder):
|
|
20
|
+
# def __init__(self, *args, **kwargs):
|
|
21
|
+
# self.seen = set()
|
|
22
|
+
# super().__init__(*args, **kwargs)
|
|
23
|
+
|
|
24
|
+
# def default(self, obj):
|
|
25
|
+
# if id(obj) in self.seen:
|
|
26
|
+
# return None # Replace circular references with None or a placeholder
|
|
27
|
+
# self.seen.add(id(obj))
|
|
28
|
+
# try:
|
|
29
|
+
# return obj.toJSON()
|
|
30
|
+
# except:
|
|
31
|
+
# try:
|
|
32
|
+
# return str(obj)
|
|
33
|
+
# except:
|
|
34
|
+
# return None
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
# def jsonify(x):
|
|
38
|
+
# return json.dumps(x, cls=RobustJSONEncoder, sort_keys=True, indent=4)
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def to_json(
|
|
42
|
+
inps: list,
|
|
43
|
+
fields: Optional[list] = None,
|
|
44
|
+
skip_if_missing: bool = False,
|
|
45
|
+
skip_empties: bool = True,
|
|
46
|
+
max_str_len: Optional[int] = 1000,
|
|
47
|
+
) -> str:
|
|
48
|
+
"""
|
|
49
|
+
Convert a list of dicts to a JSON string, with only the fields we want,
|
|
50
|
+
and in the same order as FIELDS.
|
|
51
|
+
"""
|
|
52
|
+
outs = []
|
|
53
|
+
for inp in inps:
|
|
54
|
+
if fields is None:
|
|
55
|
+
fields = inp.keys()
|
|
56
|
+
# we want to make sure to return a dict with only the fields we want,
|
|
57
|
+
# and in the same order as FIELDS
|
|
58
|
+
out = {}
|
|
59
|
+
for k in fields:
|
|
60
|
+
if skip_if_missing and (k not in inp):
|
|
61
|
+
continue
|
|
62
|
+
v = inp[k] # will error if missing and !SKIP_IF_MISSING
|
|
63
|
+
if skip_empties and (v is None or v == ""):
|
|
64
|
+
continue
|
|
65
|
+
if max_str_len and isinstance(v, str) and len(v) > max_str_len:
|
|
66
|
+
v = v[:max_str_len] + "..."
|
|
67
|
+
out[k] = v
|
|
68
|
+
outs.append(out)
|
|
69
|
+
outs_j = json.dumps(outs, indent=2)
|
|
70
|
+
return outs_j
|