GJDutils 0.2.2__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (49) hide show
  1. gjdutils/__init__.py +12 -0
  2. gjdutils/audios.py +39 -0
  3. gjdutils/cacheing.py +237 -0
  4. gjdutils/cmd.py +149 -0
  5. gjdutils/colab.py +39 -0
  6. gjdutils/collections.py +36 -0
  7. gjdutils/decorators.py +34 -0
  8. gjdutils/dicts.py +216 -0
  9. gjdutils/dsci.py +202 -0
  10. gjdutils/dt.py +296 -0
  11. gjdutils/env.py +64 -0
  12. gjdutils/errors.py +12 -0
  13. gjdutils/files.py +140 -0
  14. gjdutils/functions.py +6 -0
  15. gjdutils/google_translate.py +80 -0
  16. gjdutils/hashing.py +32 -0
  17. gjdutils/html.py +87 -0
  18. gjdutils/indexing.py +97 -0
  19. gjdutils/iterfunc.py +99 -0
  20. gjdutils/jsons.py +70 -0
  21. gjdutils/lists.py +13 -0
  22. gjdutils/llm_utils.py +167 -0
  23. gjdutils/llms_claude.py +131 -0
  24. gjdutils/llms_openai.py +299 -0
  25. gjdutils/misc.py +30 -0
  26. gjdutils/num.py +77 -0
  27. gjdutils/obsolete/google_text_to_speech.py +46 -0
  28. gjdutils/obsolete/llms_obsolete.py +298 -0
  29. gjdutils/outloud_text_to_speech.py +230 -0
  30. gjdutils/prompt_templates.py +20 -0
  31. gjdutils/pypi_build.py +112 -0
  32. gjdutils/pytest_utils.py +24 -0
  33. gjdutils/rand.py +65 -0
  34. gjdutils/regex.py +78 -0
  35. gjdutils/requirements_dev.txt +2 -0
  36. gjdutils/runtime.py +19 -0
  37. gjdutils/sets.py +5 -0
  38. gjdutils/shell.py +69 -0
  39. gjdutils/sorteddict.py +34 -0
  40. gjdutils/stopwatch.py +79 -0
  41. gjdutils/strings.py +218 -0
  42. gjdutils/todo/convert_parquet.py +28 -0
  43. gjdutils/typ.py +37 -0
  44. gjdutils/voice_speechrecognition.py +29 -0
  45. gjdutils/web.py +68 -0
  46. gjdutils-0.2.2.dist-info/METADATA +101 -0
  47. gjdutils-0.2.2.dist-info/RECORD +49 -0
  48. gjdutils-0.2.2.dist-info/WHEEL +4 -0
  49. gjdutils-0.2.2.dist-info/licenses/LICENSE +21 -0
gjdutils/env.py ADDED
@@ -0,0 +1,64 @@
1
+ import os
2
+ from pathlib import Path
3
+ from typing import Any, TypeVar, cast
4
+ from pydantic import StrictStr, TypeAdapter
5
+
6
+ T = TypeVar("T")
7
+ _processed_vars = set()
8
+
9
+
10
+ def get_env_var(name: str, typ: Any = StrictStr) -> T:
11
+ """Get environment variable with type validation, e.g.
12
+
13
+ OPENAI_API_KEY = get_env_var("OPENAI_API_KEY")
14
+ NUM_WORKERS = get_env_var("NUM_WORKERS", typ=int)
15
+
16
+ Args:
17
+ name: Name of environment variable
18
+ type_: Pydantic type to validate against (default: StrictStr for non-empty string)
19
+
20
+ Returns:
21
+ The validated value with the specified type
22
+
23
+ Raises:
24
+ ValueError: If variable is missing or fails validation
25
+ """
26
+ try:
27
+ value = os.environ[name]
28
+ _processed_vars.add(name)
29
+
30
+ # Use TypeAdapter for validation
31
+ adapter = TypeAdapter(typ)
32
+ validated = adapter.validate_python(value)
33
+
34
+ # Return validated value directly
35
+ return cast(T, validated)
36
+ except KeyError:
37
+ raise ValueError(f"Missing required environment variable: {name}")
38
+ except Exception as e:
39
+ raise ValueError(f"Invalid value for {name}: {e}")
40
+
41
+
42
+ def list_env_example_vars(env_example_filen: Path) -> set[str]:
43
+ """Get set of required variables from .env.example.
44
+
45
+ Args:
46
+ env_example_filen: Path to the .env.example file
47
+
48
+ Returns:
49
+ Set of environment variable names found in the file
50
+ """
51
+ assert env_example_filen.exists(), f"Missing env example file: {env_example_filen}"
52
+
53
+ required_vars = set()
54
+ with env_example_filen.open() as f:
55
+ for line in f:
56
+ line = line.strip()
57
+ # Skip comments and empty lines
58
+ if not line or line.startswith("#"):
59
+ continue
60
+ # Get variable name (everything before =)
61
+ var_name = line.split("=")[0].strip()
62
+ required_vars.add(var_name)
63
+
64
+ return required_vars
gjdutils/errors.py ADDED
@@ -0,0 +1,12 @@
1
+ import inspect
2
+ import sys
3
+ import traceback
4
+
5
+ # see also functions.func_name()
6
+
7
+
8
+ def str_from_exception(name=None):
9
+ return {
10
+ "name": name,
11
+ "msg": "".join(traceback.format_exception(*sys.exc_info())),
12
+ }
gjdutils/files.py ADDED
@@ -0,0 +1,140 @@
1
+ import os
2
+ from pathlib import Path
3
+ from typing import Sequence
4
+
5
+ from .cmd import run_cmd
6
+ from .strings import is_string, PathOrStr
7
+
8
+
9
+ def split_filen(filen: Path | str):
10
+ """
11
+ Splits a filename into its path, stem, and extension (without dot), e.g.
12
+
13
+ split_filen('data/blah.mp4') -> ('data', 'blah', 'mp4')
14
+ """
15
+ filen = Path(filen)
16
+ return filen.parent, filen.stem, filen.suffix[1:] if filen.suffix else ""
17
+
18
+
19
+ def create_dir_if_not_exists(dirn: str):
20
+ if not os.path.exists(dirn):
21
+ os.makedirs(dirn)
22
+
23
+
24
+ def validate_ext(ext):
25
+ assert is_string(ext)
26
+ assert ext.lower() == ext
27
+ assert ext
28
+ assert ext[0] != "."
29
+
30
+
31
+ def validate_dir(dirn):
32
+ dirn_path = Path(dirn)
33
+ assert dirn_path.exists() and dirn_path.is_dir()
34
+ return dirn_path
35
+
36
+
37
+ def fulltext(
38
+ filens: Sequence[str],
39
+ patterns: list[str],
40
+ dirn: str,
41
+ file_ext: str,
42
+ case_sensitive=False,
43
+ ):
44
+ """
45
+ Returns: FOUND_FILES (list of filename strings)
46
+
47
+ Feed in a list of filenames (complete with extensions),
48
+ which will be fed to agrep for full-text
49
+ searching. Returns a list of files.
50
+
51
+ FILENS is a list of strings. If its non-empty, then
52
+ these will be fed in to agrep. If it's empty, then we'll
53
+ just feed in a '*.[freex_extension]'. Spaces in
54
+ filenames are escaped with backslashes, but this is the
55
+ only thing we're escaping.
56
+
57
+ PATTERNS is a list of strings, which will be ANDed
58
+ together in the agrep regex. Currently, this doesn't
59
+ escape the pattern regex at all, though it does surround it in
60
+ quotes, so the usual agrep rules apply.
61
+
62
+ Unless case_sensitive==True, will append a -i flag.
63
+ """
64
+ # from freex_sqlalchemy.py
65
+
66
+ # xxx this should check that all the files have extensions
67
+
68
+ if case_sensitive:
69
+ case_flag = ""
70
+ else:
71
+ case_flag = "-i"
72
+
73
+ # xxx should check that all the items in the pattern
74
+ # list are strings...
75
+ #
76
+ # first strip each of the pattern strings of whitespace,
77
+ # and remove the surrounding quotes - we'll add them
78
+ # back to the whole pattern_str when we create the CMD
79
+ #
80
+ # then AND together multiple patterns with agrep,
81
+ # using semicolons
82
+ for pat in patterns:
83
+ if pat[0] == '"':
84
+ pat = pat[1:]
85
+ if pat[-1] == '"':
86
+ pat = pat[0:-1]
87
+
88
+ pattern_str = ";".join([x.strip() for x in patterns])
89
+
90
+ if len(filens) > 0:
91
+ # escape all the spaces with back-slashes
92
+ filens = [x.replace(" ", "\\ ") for x in filens]
93
+
94
+ # convert to a space-delimited string (with spaces
95
+ # escaped by backslashes), and each file prepended by the
96
+ # database_dir, e.g.
97
+ # /blah/test0.freex /blah/hello\ world.freex
98
+ fnames_str = " ".join([os.path.join(dirn, filen) for filen in filens])
99
+
100
+ # the -l says to just return filenames only (no text
101
+ # context)
102
+ #
103
+ # put the pattern in quotes
104
+ #
105
+ # and then just list the files at the end
106
+ cmd = 'agrep -l %s "%s" %s' % (case_flag, pattern_str, fnames_str)
107
+
108
+ else:
109
+ # if we're not restricting the files we're looking
110
+ # through, then there could be too many files to run
111
+ # agrep on directly, so we have to pipe it from a
112
+ # find
113
+ #
114
+ # this is to avoid the '/usr/local/bin/agrep:
115
+ # Argument list too long' error
116
+ cmd = 'find %s -name "*.%s" -print0 | xargs -0 agrep -l %s "%s"' % (
117
+ dirn,
118
+ file_ext,
119
+ case_flag,
120
+ pattern_str,
121
+ )
122
+
123
+ # Run command with minimal output unless there's an error
124
+ retcode, out_str, _ = run_cmd(
125
+ cmd,
126
+ verbose=0,
127
+ check=False, # Don't raise exception if no matches found (agrep returns 1)
128
+ )
129
+
130
+ if len(out_str) > 0:
131
+ # strip away the path to yield just the filename for
132
+ # each of the files in out_str
133
+ found_files = [os.path.basename(x) for x in out_str.strip().split("\n")]
134
+ else:
135
+ # if you run the above on an empty string, you get
136
+ # [''], whereas we really want to return an empty
137
+ # list if we didn't find anything
138
+ found_files = []
139
+
140
+ return found_files
gjdutils/functions.py ADDED
@@ -0,0 +1,6 @@
1
+ import inspect
2
+
3
+
4
+ def func_name():
5
+ # https://stackoverflow.com/a/13514318/230523
6
+ return inspect.currentframe().f_back.f_code.co_name
@@ -0,0 +1,80 @@
1
+ from cachetools import cached, LRUCache, TTLCache
2
+ from google.cloud import translate_v2 as translate
3
+ import html
4
+ from typing import Optional
5
+
6
+
7
+ def translate_text(
8
+ text: str,
9
+ lang_src_code: Optional[str],
10
+ lang_tgt_code: str,
11
+ verbose: int = 0,
12
+ ):
13
+ """Translates text into the target language.
14
+
15
+ Target must be an ISO 639-1 language code.
16
+ See https://g.co/cloud/translate/v2/translate-reference#supported_languages
17
+ """
18
+ translate_client = translate.Client()
19
+
20
+ lang_src_code = (
21
+ lang_src_code[:2].lower() if isinstance(lang_src_code, str) else None
22
+ )
23
+ lang_tgt_code = lang_tgt_code[:2].lower()
24
+ if lang_src_code == lang_tgt_code:
25
+ return text, None
26
+
27
+ # assert lang_src_code != lang_tgt_code, (
28
+ # "Identical src and tgt language codes: %s" % lang_src_code
29
+ # )
30
+
31
+ # Text can also be a sequence of strings, in which case this method
32
+ # will return a sequence of results for each text.
33
+ if lang_src_code is None:
34
+ result = translate_client.translate(text, target_language=lang_tgt_code)
35
+ else:
36
+ result = translate_client.translate(
37
+ text,
38
+ target_language=lang_tgt_code,
39
+ source_language=lang_src_code,
40
+ )
41
+
42
+ translated_text = result["translatedText"]
43
+
44
+ # fix escaping, e.g.
45
+ # I've done it a week with no improvement
46
+ # ->
47
+ # I've done it a week with no improvement
48
+ translated_text = html.unescape(translated_text)
49
+
50
+ if verbose > 0:
51
+ print(f"{lang_src_code} -> {lang_tgt_code}")
52
+ print(f"\t\"{result['input']}\" -> \"{translated_text}\"")
53
+ if lang_src_code is None:
54
+ print(f"\t\tDetected source language: {result['detectedSourceLanguage']}")
55
+
56
+ return translated_text, result
57
+
58
+
59
+ # translated_text, result = translate_text(
60
+ # text="Hello, world",
61
+ # lang_src_code="en",
62
+ # lang_tgt_code="el",
63
+ # verbose=0,
64
+ # )
65
+ # translated_text
66
+
67
+
68
+ @cached(cache={})
69
+ def detect_language(text: str, verbose: int = 0) -> tuple[str, dict]:
70
+ """
71
+ Detects the text's language.
72
+ """
73
+ translate_client = translate.Client()
74
+
75
+ # Text can also be a sequence of strings, in which case this method
76
+ # will return a sequence of results for each text.
77
+ result = translate_client.detect_language(text)
78
+ language, confidence = result["language"], result["confidence"]
79
+ print(f"Ran detect_language for {text} -> {language} at confidence {confidence}")
80
+ return language, confidence
gjdutils/hashing.py ADDED
@@ -0,0 +1,32 @@
1
+ import base64
2
+ import hashlib
3
+
4
+
5
+ def hash_readable(s, n=10):
6
+ """
7
+ Returns a string hash that contains base32 characters instead of a number,
8
+ to make it more readable (and still low risk of collisions if you truncate it).
9
+
10
+ e.g. hash_readable('hello') => 'vl2mmho4yx'
11
+
12
+ Unlike Python's default hash function, this should be deterministic
13
+ across sessions (because we're using 'hashlib').
14
+
15
+ I'm using this for anonymising email addresses if I don't have a user UUID.
16
+ """
17
+ if isinstance(s, str):
18
+ s = bytes(s, "utf-8")
19
+ hasher = hashlib.sha1(s)
20
+ b32 = base64.b32encode(hasher.digest())[:n]
21
+ return b32.decode("utf-8").lower()
22
+
23
+
24
+ def hash_consistent(obj):
25
+ """
26
+ Supposedly gives the same response every time you call it, even after restarting the kernel.
27
+
28
+ N.B. This is based on output from GitHub Copilot, and I haven't tried it.
29
+ """
30
+ obj_str = str(obj)
31
+ hash_obj = hashlib.sha256(obj_str.encode()).hexdigest()
32
+ return hash_obj
gjdutils/html.py ADDED
@@ -0,0 +1,87 @@
1
+ from bs4 import BeautifulSoup
2
+ from lxml import html as lxml_html
3
+ from lxml.html import tostring, fromstring
4
+ from lxml.etree import Element, _Element as ElementType
5
+ from typing import Optional, Iterable, Union
6
+
7
+
8
+ def remove_html_tags(html: str):
9
+ soup = BeautifulSoup(html, "html.parser")
10
+ return soup.get_text()
11
+
12
+
13
+ def contents_of_body(soup):
14
+ """
15
+ e.g.
16
+ BeautifulSoup('<p>hello</p><p>world</world>', features='lxml')
17
+ =>
18
+ <p>hello</p>
19
+ <p>world</p>
20
+
21
+ N.B. for html.parser, you might just be able to do: str(soup)
22
+ """
23
+ # it might be better to prettify with body hidden=True???
24
+ return "\n".join([str(t) for t in soup.body.contents])
25
+
26
+
27
+ def compare_html(h1, h2):
28
+ h1p = BeautifulSoup(h1, features="html.parser").prettify().strip()
29
+ h2p = BeautifulSoup(h2, features="html.parser").prettify().strip()
30
+ assert h1p == h2p
31
+
32
+
33
+ def remove_attrs_from_html(h):
34
+ """
35
+ Gets rid of all the attrs in the html.
36
+ """
37
+ soup = BeautifulSoup(h, features="lxml")
38
+ for t in soup.recursiveChildGenerator():
39
+ t.attrs = {} # type: ignore
40
+ # whitespace_from_linebreaks(
41
+ # contents_of_body(soup)
42
+ # )
43
+ return contents_of_body(soup)
44
+
45
+
46
+ def adjust_indentation(pretty_html, indent: int):
47
+ # from https://www.perplexity.ai/search/can-you-customise-the-beautifu-225tf.pISaiggsL5tNL.gA
48
+ lines = pretty_html.split("\n")
49
+ adjusted_lines = []
50
+ for line in lines:
51
+ line_lstrip = line.lstrip(" ")
52
+ leading_spaces = len(line) - len(line_lstrip)
53
+ indent_level = leading_spaces // 1 # default indent is 1 space
54
+ adjusted_lines.append(" " * (indent_level * indent) + line_lstrip)
55
+ return "\n".join(adjusted_lines)
56
+
57
+
58
+ def prettify_html(
59
+ html: Union[str, ElementType, list[ElementType]], # BeautifulSoup
60
+ indent: int = 2,
61
+ n: Optional[int] = None, # number of chars to show
62
+ ):
63
+ # if isinstance(html, pq):
64
+ # html = html.outer_html() # type: ignore
65
+ if isinstance(html, list):
66
+ # then we'll handle it as a string in a moment
67
+ html = "".join(
68
+ [tostring(e, method="html").decode() for e in html]
69
+ ) #  type: ignore
70
+ if isinstance(html, ElementType):
71
+ # this will do some cleaning and fixing. but
72
+ # you need document_fromstring() if you want to make sure
73
+ # that it's a full html doc, e.g. with html, body
74
+ html = tostring(html, method="html").decode()
75
+
76
+ soup = BeautifulSoup(html, "html.parser")
77
+ # html2 = tostring(html, pretty_print=True, method="html").decode() # type: ignore
78
+ # the lxml pretty_print just isn't as good as BS4, e.g. with a list of elements
79
+ # it wraps things in a div fragment, but the pretty-print of that isn't right
80
+ html2 = soup.prettify()
81
+ prettified = adjust_indentation(html2, indent=indent)[:n] # type: ignore
82
+ return prettified
83
+
84
+
85
+ def pprettify_html(*args, **kwargs) -> None:
86
+ html = prettify_html(*args, **kwargs)
87
+ print(html)
gjdutils/indexing.py ADDED
@@ -0,0 +1,97 @@
1
+ from decimal import Decimal, getcontext
2
+ from typing import Optional
3
+
4
+ """
5
+ For manual ordering and reording in a database:
6
+ - every item gets a Decimal location (LOC) between 0 and 1
7
+ - the LOCs of all items are sorted
8
+ - the LOC of a new item is calculated as the average of the LOCs of the items before and after it
9
+ - the LOC won't ever be 0 or 1, so there will always be a gap for you to insert afterwards
10
+ - if you insert at the beginning or end, the LOC will be half of the first or last item's LOC
11
+
12
+ I think Figma used this.
13
+ """
14
+
15
+ # Set precision high enough to handle many divisions
16
+ getcontext().prec = 28
17
+
18
+
19
+ def locs_for(n: int) -> list[Decimal]:
20
+ # Assign initial idx values with buffers at both ends
21
+ locs = []
22
+ for i in range(n):
23
+ loc = Decimal(i + 1) / Decimal(n + 1)
24
+ locs.append(loc)
25
+ assert len(locs) == n
26
+ return locs
27
+
28
+
29
+ def loc_for_insert_at(locs: list[Decimal], position: int, do_insert: bool = True):
30
+ """
31
+ TODO: rewrite in terms of LOC_BETWEEN
32
+ """
33
+ assert locs == sorted(
34
+ locs
35
+ ), f"Input LOCS are unsorted, so things are already broken - {locs}"
36
+ list_length = len(locs)
37
+ if position == 0: # Insert at the beginning
38
+ newloc = locs[0] / 2 if list_length > 0 else Decimal("0.5")
39
+ elif position >= list_length: # Insert at the end
40
+ newloc = locs[-1] + (1 - locs[-1]) / 2 if list_length > 0 else Decimal("0.5")
41
+ elif position < 0:
42
+ raise Exception(f"Position must be non-negative, but got {position}")
43
+ else: # Insert between two items
44
+ newloc = (locs[position - 1] + locs[position]) / 2
45
+ if do_insert:
46
+ locs.insert(position, newloc)
47
+ assert locs == sorted(locs), f"Somehow we've broken the LOCS sorting: {locs}"
48
+ return newloc
49
+
50
+
51
+ def loc_for_insert_at2(locs: list[Decimal], position: int, do_insert: bool = True):
52
+ """
53
+ Functional version of LOC_FOR_INSERT_AT that returns a new list instead of
54
+ modifying the input list.
55
+
56
+ Uses LOC_BETWEEN.
57
+ """
58
+ assert locs == sorted(
59
+ locs
60
+ ), f"Input LOCS are unsorted, so things are already broken - {locs}"
61
+ list_length = len(locs)
62
+ if position < 0:
63
+ raise Exception(f"Position must be non-negative, but got {position}")
64
+ elif position == 0: # Insert at the beginning
65
+ loc1 = None
66
+ loc2 = locs[0] if list_length > 0 else None
67
+ elif position >= list_length: # Insert at the end
68
+ loc1 = locs[-1] if list_length > 0 else None
69
+ loc2 = None
70
+ else: # Insert between two items
71
+ loc1 = locs[position - 1]
72
+ loc2 = locs[position]
73
+ newloc = loc_between(loc1, loc2)
74
+ if do_insert:
75
+ locs.insert(position, newloc)
76
+ assert locs == sorted(locs), f"Somehow we've broken the LOCS sorting: {locs}"
77
+ return newloc
78
+
79
+
80
+ def loc_between(loc1: Optional[Decimal], loc2: Optional[Decimal]) -> Decimal:
81
+ if loc1 is not None and loc2 is not None:
82
+ assert (
83
+ loc1 >= 0 and loc2 <= 1
84
+ ), f"LOCs must be between 0 and 1, but got {loc1} and {loc2}"
85
+ return (loc1 + loc2) / 2
86
+ elif loc1 is None and loc2 is not None:
87
+ return loc2 / 2
88
+ elif loc1 is not None and loc2 is None:
89
+ return loc1 + (1 - loc1) / 2
90
+ elif loc1 is None and loc2 is None:
91
+ return Decimal("0.5")
92
+ else:
93
+ raise Exception(f"This should never happen: {loc1}, {loc2}")
94
+
95
+
96
+ def disp(locs: list[Decimal]):
97
+ print(", ".join(["%.3f" % loc for loc in locs]))
gjdutils/iterfunc.py ADDED
@@ -0,0 +1,99 @@
1
+ import itertools
2
+ from typing import Sequence
3
+
4
+
5
+ def contiguous_pairs(lst: Sequence):
6
+ """
7
+ Given a list LST, return the contiguous pairs, e.g.
8
+
9
+ [10, 20, 30, 40, 50]
10
+ ->
11
+ [(10, 20), (20, 30), (30, 40), (40, 50)]
12
+
13
+ (from GitHub Copilot)
14
+ """
15
+ pairs = [(lst[i], lst[i + 1]) for i in range(len(lst) - 1)]
16
+ return pairs
17
+
18
+
19
+ def flatten(lol):
20
+ """
21
+ See http://stackoverflow.com/questions/406121/flattening-a-shallow-list-in-python
22
+
23
+ e.g. [['image00', 'image01'], ['image10'], []] -> ['image00', 'image01', 'image10']
24
+ """
25
+
26
+ chain = list(itertools.chain(*lol))
27
+ return chain
28
+
29
+
30
+ # def flatten(list_of_lists):
31
+ # """
32
+ # Flatten one level of nesting
33
+
34
+ # from https://docs.python.org/3/library/itertools.html#itertools-recipes
35
+ # """
36
+ # return list(chain.from_iterable(list_of_lists))
37
+
38
+
39
+ def unique(items):
40
+ """
41
+ Returns KEEP, a list based on ITEMS, but with duplicates
42
+ removed (preserving order, based on first new example).
43
+
44
+ http://stackoverflow.com/questions/89178/in-python-what-is-the-fastest-algorithm-for-removing-duplicates-from-a-list-so-t
45
+
46
+ unique([1, 1, 2, 'a', 'a', 3]) -> [1, 2, 'a', 3]
47
+ """
48
+ found = set([])
49
+ keep = []
50
+ for item in items:
51
+ if item not in found:
52
+ found.add(item)
53
+ keep.append(item)
54
+ return keep
55
+
56
+
57
+ def uniquify_list(lst):
58
+ """Return a list of the elements in s, but without duplicates, preserving order.
59
+
60
+ from comment in http://aspn.activestate.com/ASPN/Cookbook/Python/Recipe/52560
61
+
62
+ Lightweight and fast ..., Raymond Hettinger, 2002/03/17
63
+
64
+ """
65
+
66
+ set = {}
67
+ return [set.setdefault(e, e) for e in lst if e not in set]
68
+
69
+
70
+ def grouper(iterable, n):
71
+ """
72
+ Collect data into fixed-length chunks or blocks. If
73
+ the last block is too small, returns a truncated block.
74
+
75
+ e.g. grouper('ABCDEFG', 3) --> ABC DEF G
76
+
77
+ From http://stackoverflow.com/a/8991553/230523
78
+ """
79
+ it = iter(iterable)
80
+ while True:
81
+ chunk = tuple(itertools.islice(it, n))
82
+ if not chunk:
83
+ return
84
+ yield chunk
85
+
86
+
87
+ def grouper_ragged(iterable, n):
88
+ """
89
+ Collect data into non-overlapping chunks - the last one might be shorter than the others
90
+
91
+ >>> print(list(grouper('ABCDEFG', 3))) # [('A', 'B', 'C'), ('D', 'E', 'F'), ('G',)]
92
+
93
+ from https://stackoverflow.com/a/41333827/230523
94
+ """
95
+ it = iter(iterable)
96
+ group = tuple(itertools.islice(it, n))
97
+ while group:
98
+ yield group
99
+ group = tuple(itertools.islice(it, n))
gjdutils/jsons.py ADDED
@@ -0,0 +1,70 @@
1
+ import json
2
+ from typing import Optional
3
+
4
+
5
+ def jsonify(x):
6
+ def json_dumper_robust(obj):
7
+ try:
8
+ return obj.toJSON()
9
+ except:
10
+ try:
11
+ return str(obj)
12
+ except:
13
+ return None
14
+
15
+ return json.dumps(x, sort_keys=True, indent=4, default=json_dumper_robust)
16
+
17
+
18
+ # from o-1
19
+ # class RobustJSONEncoder(json.JSONEncoder):
20
+ # def __init__(self, *args, **kwargs):
21
+ # self.seen = set()
22
+ # super().__init__(*args, **kwargs)
23
+
24
+ # def default(self, obj):
25
+ # if id(obj) in self.seen:
26
+ # return None # Replace circular references with None or a placeholder
27
+ # self.seen.add(id(obj))
28
+ # try:
29
+ # return obj.toJSON()
30
+ # except:
31
+ # try:
32
+ # return str(obj)
33
+ # except:
34
+ # return None
35
+
36
+
37
+ # def jsonify(x):
38
+ # return json.dumps(x, cls=RobustJSONEncoder, sort_keys=True, indent=4)
39
+
40
+
41
+ def to_json(
42
+ inps: list,
43
+ fields: Optional[list] = None,
44
+ skip_if_missing: bool = False,
45
+ skip_empties: bool = True,
46
+ max_str_len: Optional[int] = 1000,
47
+ ) -> str:
48
+ """
49
+ Convert a list of dicts to a JSON string, with only the fields we want,
50
+ and in the same order as FIELDS.
51
+ """
52
+ outs = []
53
+ for inp in inps:
54
+ if fields is None:
55
+ fields = inp.keys()
56
+ # we want to make sure to return a dict with only the fields we want,
57
+ # and in the same order as FIELDS
58
+ out = {}
59
+ for k in fields:
60
+ if skip_if_missing and (k not in inp):
61
+ continue
62
+ v = inp[k] # will error if missing and !SKIP_IF_MISSING
63
+ if skip_empties and (v is None or v == ""):
64
+ continue
65
+ if max_str_len and isinstance(v, str) and len(v) > max_str_len:
66
+ v = v[:max_str_len] + "..."
67
+ out[k] = v
68
+ outs.append(out)
69
+ outs_j = json.dumps(outs, indent=2)
70
+ return outs_j