scripturelookup 0.1.0__tar.gz → 0.1.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (34) hide show
  1. {scripturelookup-0.1.0/src/scripturelookup.egg-info → scripturelookup-0.1.1}/PKG-INFO +1 -1
  2. {scripturelookup-0.1.0 → scripturelookup-0.1.1}/src/scripturelookup/data.py +147 -0
  3. {scripturelookup-0.1.0 → scripturelookup-0.1.1}/src/scripturelookup/lookup.py +351 -35
  4. {scripturelookup-0.1.0 → scripturelookup-0.1.1}/src/scripturelookup/patterns.py +25 -12
  5. {scripturelookup-0.1.0 → scripturelookup-0.1.1/src/scripturelookup.egg-info}/PKG-INFO +1 -1
  6. scripturelookup-0.1.1/src/scripturelookup.egg-info/scm_version.json +8 -0
  7. scripturelookup-0.1.1/tests/test_detect.py +393 -0
  8. scripturelookup-0.1.0/src/scripturelookup.egg-info/scm_version.json +0 -8
  9. scripturelookup-0.1.0/tests/test_detect.py +0 -137
  10. {scripturelookup-0.1.0 → scripturelookup-0.1.1}/.github/workflows/publish.yml +0 -0
  11. {scripturelookup-0.1.0 → scripturelookup-0.1.1}/.gitignore +0 -0
  12. {scripturelookup-0.1.0 → scripturelookup-0.1.1}/LICENSE +0 -0
  13. {scripturelookup-0.1.0 → scripturelookup-0.1.1}/README.md +0 -0
  14. {scripturelookup-0.1.0 → scripturelookup-0.1.1}/pyproject.toml +0 -0
  15. {scripturelookup-0.1.0 → scripturelookup-0.1.1}/requirements.txt +0 -0
  16. {scripturelookup-0.1.0 → scripturelookup-0.1.1}/setup.cfg +0 -0
  17. {scripturelookup-0.1.0 → scripturelookup-0.1.1}/src/geezify-python-main/LICENSE +0 -0
  18. {scripturelookup-0.1.0 → scripturelookup-0.1.1}/src/geezify-python-main/README.md +0 -0
  19. {scripturelookup-0.1.0 → scripturelookup-0.1.1}/src/geezify-python-main/arabify.py +0 -0
  20. {scripturelookup-0.1.0 → scripturelookup-0.1.1}/src/geezify-python-main/arabify_test.py +0 -0
  21. {scripturelookup-0.1.0 → scripturelookup-0.1.1}/src/geezify-python-main/geezify.py +0 -0
  22. {scripturelookup-0.1.0 → scripturelookup-0.1.1}/src/geezify-python-main/geezify_test.py +0 -0
  23. {scripturelookup-0.1.0 → scripturelookup-0.1.1}/src/geezify-python-main/test_data.py +0 -0
  24. {scripturelookup-0.1.0 → scripturelookup-0.1.1}/src/scripturelookup/__init__.py +0 -0
  25. {scripturelookup-0.1.0 → scripturelookup-0.1.1}/src/scripturelookup/command_line.py +0 -0
  26. {scripturelookup-0.1.0 → scripturelookup-0.1.1}/src/scripturelookup/data/metadata-languages.min.json +0 -0
  27. {scripturelookup-0.1.0 → scripturelookup-0.1.1}/src/scripturelookup/data/metadata-scriptures.min.json +0 -0
  28. {scripturelookup-0.1.0 → scripturelookup-0.1.1}/src/scripturelookup/numbers.py +0 -0
  29. {scripturelookup-0.1.0 → scripturelookup-0.1.1}/src/scripturelookup.egg-info/SOURCES.txt +0 -0
  30. {scripturelookup-0.1.0 → scripturelookup-0.1.1}/src/scripturelookup.egg-info/dependency_links.txt +0 -0
  31. {scripturelookup-0.1.0 → scripturelookup-0.1.1}/src/scripturelookup.egg-info/entry_points.txt +0 -0
  32. {scripturelookup-0.1.0 → scripturelookup-0.1.1}/src/scripturelookup.egg-info/requires.txt +0 -0
  33. {scripturelookup-0.1.0 → scripturelookup-0.1.1}/src/scripturelookup.egg-info/scm_file_list.json +0 -0
  34. {scripturelookup-0.1.0 → scripturelookup-0.1.1}/src/scripturelookup.egg-info/top_level.txt +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: scripturelookup
3
- Version: 0.1.0
3
+ Version: 0.1.1
4
4
  Summary: Python and command-line utility for converting scripture references between formats.
5
5
  Author-email: Samuel Bradshaw <samuel.h.bradshaw@gmail.com>
6
6
  License: MIT License
@@ -79,6 +79,90 @@ reference_words = {
79
79
  }
80
80
  reference_word_keys = ('list_conjunctions', 'range_conjunctions', 'verse_words', 'chapter_words',)
81
81
 
82
+ # Ordinal forms 1–13, for citing an article of faith by position ("First Article of Faith"), and – in the
83
+ # first four entries – for a numbered book cited as "First Nephi" or "1st Nephi". A form's position in the
84
+ # list is the number it stands for, so there are no numbers written out separately to keep in sync.
85
+ # Gendered forms appear where the noun takes them: Portuguese cites "Regras de Fé", so an article there is
86
+ # "Primeira regra de fé", while Spanish cites "Artículos de Fe" and so takes "Primer artículo de fe".
87
+ ordinal_words = {
88
+ 'en': [
89
+ ('first', '1st',), ('second', '2nd',), ('third', '3rd',), ('fourth', '4th',),
90
+ ('fifth', '5th',), ('sixth', '6th',), ('seventh', '7th',), ('eighth', '8th',),
91
+ ('ninth', '9th',), ('tenth', '10th',), ('eleventh', '11th',), ('twelfth', '12th',),
92
+ ('thirteenth', '13th',),
93
+ ],
94
+ 'fr': [
95
+ ('premier', 'première', '1er', '1re', '1ère',),
96
+ ('deuxième', 'second', 'seconde', '2e', '2ème', '2d', '2de',),
97
+ ('troisième', '3e', '3ème',), ('quatrième', '4e', '4ème',), ('cinquième', '5e', '5ème',),
98
+ ('sixième', '6e', '6ème',), ('septième', '7e', '7ème',), ('huitième', '8e', '8ème',),
99
+ ('neuvième', '9e', '9ème',), ('dixième', '10e', '10ème',), ('onzième', '11e', '11ème',),
100
+ ('douzième', '12e', '12ème',), ('treizième', '13e', '13ème',),
101
+ ],
102
+ # Spanish has two accepted systems for 11 and 12 – the classical "undécimo" and the modern
103
+ # "decimoprimero" – and each ordinal has a masculine, a feminine, and (for 1, 3 and their compounds) an
104
+ # apocopated form used before a masculine noun. Every spelling is listed, so a reference resolves whichever
105
+ # the writer reached for, and whether or not the gender agrees with the name being cited.
106
+ 'es': [
107
+ ('primero', 'primer', 'primera', '1º', '1er', '1ª',),
108
+ ('segundo', 'segunda', '2º', '2ª',),
109
+ ('tercero', 'tercer', 'tercera', '3º', '3er', '3ª',),
110
+ ('cuarto', 'cuarta', '4º', '4ª',),
111
+ ('quinto', 'quinta', '5º', '5ª',),
112
+ ('sexto', 'sexta', '6º', '6ª',),
113
+ ('séptimo', 'séptima', '7º', '7ª',),
114
+ ('octavo', 'octava', '8º', '8ª',),
115
+ ('noveno', 'novena', '9º', '9ª',),
116
+ ('décimo', 'décima', '10º', '10ª',),
117
+ ('undécimo', 'undécima', 'decimoprimero', 'decimoprimera', 'decimoprimer', 'décimo primero', 'décima primera', 'décimo primer', '11º', '11ª',),
118
+ ('duodécimo', 'duodécima', 'decimosegundo', 'decimosegunda', 'décimo segundo', 'décima segunda', '12º', '12ª',),
119
+ ('decimotercero', 'decimotercera', 'decimotercer', 'décimo tercero', 'décima tercera', 'décimo tercer', '13º', '13ª',),
120
+ ],
121
+ # Portuguese cites "Regras de Fé", which is feminine, so the feminine forms are the ones that occur – but
122
+ # the masculine are listed too, along with the classical "undécimo" and "duodécimo" alongside the ordinary
123
+ # two-word spellings.
124
+ 'pt': [
125
+ ('primeiro', 'primeira', '1º', '1ª',), ('segundo', 'segunda', '2º', '2ª',),
126
+ ('terceiro', 'terceira', '3º', '3ª',), ('quarto', 'quarta', '4º', '4ª',),
127
+ ('quinto', 'quinta', '5º', '5ª',), ('sexto', 'sexta', '6º', '6ª',),
128
+ ('sétimo', 'sétima', '7º', '7ª',), ('oitavo', 'oitava', '8º', '8ª',),
129
+ ('nono', 'nona', '9º', '9ª',), ('décimo', 'décima', '10º', '10ª',),
130
+ ('décimo primeiro', 'décima primeira', 'undécimo', 'undécima', '11º', '11ª',),
131
+ ('décimo segundo', 'décima segunda', 'duodécimo', 'duodécima', '12º', '12ª',),
132
+ ('décimo terceiro', 'décima terceira', '13º', '13ª',),
133
+ ],
134
+ }
135
+
136
+ # Roman numerals are typography rather than language, so they stand in for a book number in any language.
137
+ # Only 1–4 are needed, since no book is numbered higher.
138
+ roman_numeral_words = ['i', 'ii', 'iii', 'iv']
139
+
140
+ # The highest number any book name starts with ("4 Nephi"). patterns.leading_book_number is built from this.
141
+ highest_book_number = len(roman_numeral_words)
142
+
143
+ # Get a map of ordinal form to the number it stands for, for an article of faith cited by position
144
+ ordinal_forms_cache = {}
145
+ def get_ordinal_forms(lang):
146
+ if lang not in ordinal_forms_cache:
147
+ ordinal_forms_cache[lang] = {
148
+ form.lower(): index + 1
149
+ for index, group in enumerate(ordinal_words.get(lang, []))
150
+ for form in group
151
+ }
152
+ return ordinal_forms_cache[lang]
153
+
154
+ # Get a map of number form to the number it stands for, for the prefix on a numbered book name. It's the
155
+ # ordinals above, stopping at the highest number any book carries, plus the roman numerals that work in
156
+ # every language.
157
+ book_number_forms_cache = {}
158
+ def get_book_number_forms(lang):
159
+ if lang not in book_number_forms_cache:
160
+ forms = {form: number for form, number in get_ordinal_forms(lang).items() if number <= highest_book_number}
161
+ for index, form in enumerate(roman_numeral_words):
162
+ forms.setdefault(form, index + 1)
163
+ book_number_forms_cache[lang] = forms
164
+ return book_number_forms_cache[lang]
165
+
82
166
  # List conjunctions from every language, split into the ones written as a word ("and", "y", "et") and the ones written as a symbol ("&")
83
167
  all_list_conjunctions = sorted({c for words in reference_words.values() for c in words['list_conjunctions']})
84
168
  list_conjunction_words = [c for c in all_list_conjunctions if any(char.isalpha() for char in c)]
@@ -113,6 +197,69 @@ def get_map_to_slug_normalized():
113
197
  return map_to_slug_normalized
114
198
 
115
199
 
200
+ # The shortest normalized name that's worth matching loosely. Below this, a single typo is most of the word – "Alma" is one edit away from far too much ordinary text.
201
+ minimum_fuzzy_name_length = 5
202
+
203
+ # Every spelling of a name with one letter taken out. Example: "alma" –> ["lma", "ama", "ala", "alm"]
204
+ # Digits are never dropped, so a number always has to match exactly – it's part of which book is meant, not
205
+ # something that can be mistyped into another book. Without that, "5th Nephi" would resolve to "4th Nephi",
206
+ # which is a real name one substitution away.
207
+ def get_single_character_deletions(key):
208
+ return [key[:i] + key[i + 1:] for i in range(len(key)) if not key[i].isdigit()]
209
+
210
+ # Get a map of every single-character deletion of every known name to its book slug. Comparing deletions on
211
+ # both sides is what makes a one-character difference cheap to find: a missing letter, an extra letter, or a
212
+ # wrong letter all line up on some shared deletion, so a lookup costs one dict get per character instead of
213
+ # a comparison against all ~10,000 names. Names that two different slugs both claim are dropped, since an
214
+ # ambiguous match is worse than none. Built the first time a name fails to match exactly, since input that's
215
+ # spelled correctly never needs it.
216
+ map_to_slug_deletions = None
217
+ def get_map_to_slug_deletions():
218
+ global map_to_slug_deletions
219
+ if map_to_slug_deletions is None:
220
+ map_to_slug_deletions = {}
221
+ for key, value in scriptures['mapToSlug'].items():
222
+ normalized_key = normalize_for_compare(key)
223
+ if len(normalized_key) < minimum_fuzzy_name_length:
224
+ continue
225
+ for deletion in get_single_character_deletions(normalized_key):
226
+ map_to_slug_deletions[deletion] = value if map_to_slug_deletions.get(deletion, value) == value else None
227
+ return map_to_slug_deletions
228
+
229
+ # Look up a book slug for a normalized name that's off by one character. Diacritics are already handled by
230
+ # normalize_for_compare, so only letter-level slips reach this point. Like mapToSlug itself, the index covers
231
+ # every language's names, so a typo resolves no matter which language was asked for.
232
+ def fuzzy_map_to_slug(normalized_name):
233
+ if len(normalized_name) < minimum_fuzzy_name_length:
234
+ return None
235
+ deletions = get_map_to_slug_deletions()
236
+
237
+ # The known name has one character that the input is missing
238
+ slug = deletions.get(normalized_name)
239
+ if slug:
240
+ return slug
241
+
242
+ # The input has one character too many, or one character wrong
243
+ for deletion in get_single_character_deletions(normalized_name):
244
+ slug = deletions.get(deletion)
245
+ if slug:
246
+ return slug
247
+
248
+ return None
249
+
250
+
251
+ # Look up the book slug for a name as it was written. The name is tried as-is, then normalized (which folds
252
+ # case, diacritics and punctuation), and finally – unless allow_loose_match is False – as a name that's off
253
+ # by one character. Every caller that resolves a name should come through here, so the three stages stay in
254
+ # one order.
255
+ def get_book_slug(book_string, allow_loose_match = True):
256
+ slug = scriptures['mapToSlug'].get(book_string)
257
+ if slug:
258
+ return slug
259
+ normalized_name = normalize_for_compare(book_string)
260
+ return get_map_to_slug_normalized().get(normalized_name) or (fuzzy_map_to_slug(normalized_name) if allow_loose_match else None)
261
+
262
+
116
263
  # Get regex patterns for the words that can appear in a scripture reference in a given language. Languages that aren't listed above return empty patterns.
117
264
  reference_words_cache = {}
118
265
  def get_reference_words(lang):