scripturelookup 0.0.7__tar.gz → 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (32) hide show
  1. {scripturelookup-0.0.7/src/scripturelookup.egg-info → scripturelookup-0.1.0}/PKG-INFO +18 -2
  2. {scripturelookup-0.0.7 → scripturelookup-0.1.0}/README.md +17 -1
  3. {scripturelookup-0.0.7 → scripturelookup-0.1.0}/src/scripturelookup/command_line.py +15 -6
  4. {scripturelookup-0.0.7 → scripturelookup-0.1.0}/src/scripturelookup/data.py +80 -15
  5. {scripturelookup-0.0.7 → scripturelookup-0.1.0}/src/scripturelookup/lookup.py +194 -56
  6. scripturelookup-0.1.0/src/scripturelookup/patterns.py +132 -0
  7. {scripturelookup-0.0.7 → scripturelookup-0.1.0/src/scripturelookup.egg-info}/PKG-INFO +18 -2
  8. {scripturelookup-0.0.7 → scripturelookup-0.1.0}/src/scripturelookup.egg-info/SOURCES.txt +5 -1
  9. scripturelookup-0.1.0/src/scripturelookup.egg-info/scm_file_list.json +26 -0
  10. scripturelookup-0.1.0/src/scripturelookup.egg-info/scm_version.json +8 -0
  11. scripturelookup-0.1.0/tests/test_detect.py +137 -0
  12. {scripturelookup-0.0.7 → scripturelookup-0.1.0}/.github/workflows/publish.yml +0 -0
  13. {scripturelookup-0.0.7 → scripturelookup-0.1.0}/.gitignore +0 -0
  14. {scripturelookup-0.0.7 → scripturelookup-0.1.0}/LICENSE +0 -0
  15. {scripturelookup-0.0.7 → scripturelookup-0.1.0}/pyproject.toml +0 -0
  16. {scripturelookup-0.0.7 → scripturelookup-0.1.0}/requirements.txt +0 -0
  17. {scripturelookup-0.0.7 → scripturelookup-0.1.0}/setup.cfg +0 -0
  18. {scripturelookup-0.0.7 → scripturelookup-0.1.0}/src/geezify-python-main/LICENSE +0 -0
  19. {scripturelookup-0.0.7 → scripturelookup-0.1.0}/src/geezify-python-main/README.md +0 -0
  20. {scripturelookup-0.0.7 → scripturelookup-0.1.0}/src/geezify-python-main/arabify.py +0 -0
  21. {scripturelookup-0.0.7 → scripturelookup-0.1.0}/src/geezify-python-main/arabify_test.py +0 -0
  22. {scripturelookup-0.0.7 → scripturelookup-0.1.0}/src/geezify-python-main/geezify.py +0 -0
  23. {scripturelookup-0.0.7 → scripturelookup-0.1.0}/src/geezify-python-main/geezify_test.py +0 -0
  24. {scripturelookup-0.0.7 → scripturelookup-0.1.0}/src/geezify-python-main/test_data.py +0 -0
  25. {scripturelookup-0.0.7 → scripturelookup-0.1.0}/src/scripturelookup/__init__.py +0 -0
  26. {scripturelookup-0.0.7 → scripturelookup-0.1.0}/src/scripturelookup/data/metadata-languages.min.json +0 -0
  27. {scripturelookup-0.0.7 → scripturelookup-0.1.0}/src/scripturelookup/data/metadata-scriptures.min.json +0 -0
  28. {scripturelookup-0.0.7 → scripturelookup-0.1.0}/src/scripturelookup/numbers.py +0 -0
  29. {scripturelookup-0.0.7 → scripturelookup-0.1.0}/src/scripturelookup.egg-info/dependency_links.txt +0 -0
  30. {scripturelookup-0.0.7 → scripturelookup-0.1.0}/src/scripturelookup.egg-info/entry_points.txt +0 -0
  31. {scripturelookup-0.0.7 → scripturelookup-0.1.0}/src/scripturelookup.egg-info/requires.txt +0 -0
  32. {scripturelookup-0.0.7 → scripturelookup-0.1.0}/src/scripturelookup.egg-info/top_level.txt +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: scripturelookup
3
- Version: 0.0.7
3
+ Version: 0.1.0
4
4
  Summary: Python and command-line utility for converting scripture references between formats.
5
5
  Author-email: Samuel Bradshaw <samuel.h.bradshaw@gmail.com>
6
6
  License: MIT License
@@ -40,7 +40,7 @@ Dynamic: license-file
40
40
 
41
41
  # Scripture Lookup
42
42
 
43
- Scripture Lookup is a Python library and command-line tool for looking up scripture verses or chapters. It can also convert scripture references between formats and languages. The tool is built around data from [Python Scripture Scraper](https://github.com/samuelbradshaw/python-scripture-scraper).
43
+ Scripture Lookup is a Python library and command-line tool for looking up scripture verses or chapters. It can also detect scripture references in a string, and convert them between formats and languages. The tool uses scripture and language metadata from [Python Scripture Scraper](https://github.com/samuelbradshaw/python-scripture-scraper).
44
44
 
45
45
 
46
46
  ## Installation
@@ -79,6 +79,12 @@ https://www.churchofjesuschrist.org/study/scriptures/bofm/hel/5?id=p12&lang=eng#
79
79
 
80
80
  % scripturelookup get_church_url "/scriptures/ot" --lang "es"
81
81
  https://www.churchofjesuschrist.org/study/scriptures/ot?lang=spa
82
+
83
+ % scripturelookup detect_references "As Nephi taught in 1 Ne. 3:7, and again in 2 Nephi 2:25."
84
+ [['1 Ne. 3:7', 19, 28], ['2 Nephi 2:25', 43, 55]]
85
+
86
+ % scripturelookup refresh_metadata
87
+ Updated metadata: metadata-languages.min.json, metadata-scriptures.min.json
82
88
  ```
83
89
 
84
90
 
@@ -102,6 +108,12 @@ lookup.get_label('/scriptures/nt/1-jn/3.2-3', lang = 'cmn-Hant')
102
108
 
103
109
  lookup.get_church_url('/scriptures/ot', lang = 'es')
104
110
  # https://www.churchofjesuschrist.org/study/scriptures/ot?lang=spa
111
+
112
+ lookup.detect_references('As Nephi taught in 1 Ne. 3:7, and again in 2 Nephi 2:25.')
113
+ # [['1 Ne. 3:7', 19, 28], ['2 Nephi 2:25', 43, 55]]
114
+
115
+ lookup.refresh_metadata()
116
+ # Updated metadata: metadata-languages.min.json, metadata-scriptures.min.json
105
117
  ```
106
118
 
107
119
 
@@ -117,6 +129,8 @@ lookup.get_church_url('/scriptures/ot', lang = 'es')
117
129
  - **get_reference_objects** – Get a list of references as objects.
118
130
  - **get_reference_attributes** – Get a list of references as dictionaries.
119
131
  - **sort_references** – Sort a list of references by label or in traditional book order.
132
+ - **detect_references** – Find scripture references embedded in a string of text.
133
+ - **refresh_metadata** – Download the latest scripture and language metadata. Takes no input.
120
134
 
121
135
  ### Inputs
122
136
 
@@ -145,6 +159,8 @@ Any of the following input types are supported. You can also provide several inp
145
159
  - http://lds.org/scriptures/bofm/1-ne/3.7?lang=eng
146
160
  - [gospellibrary://content/scriptures/nt/john/3.16?lang=eng#16](https://www.churchofjesuschrist.org/study/scriptures/nt/john/3?id=p16&lang=eng#p16)
147
161
 
162
+ `detect_references` takes an arbitrary text string and returns a list of `[label, start_offset, end_offset]`. The label is the exact detected text (not normalized). Start and end offset indicate the exclusive range of characters in the input string where the reference was found.
163
+
148
164
  ### Options
149
165
 
150
166
  Several options are available. Some are only applicable to certain commands.
@@ -1,6 +1,6 @@
1
1
  # Scripture Lookup
2
2
 
3
- Scripture Lookup is a Python library and command-line tool for looking up scripture verses or chapters. It can also convert scripture references between formats and languages. The tool is built around data from [Python Scripture Scraper](https://github.com/samuelbradshaw/python-scripture-scraper).
3
+ Scripture Lookup is a Python library and command-line tool for looking up scripture verses or chapters. It can also detect scripture references in a string, and convert them between formats and languages. The tool uses scripture and language metadata from [Python Scripture Scraper](https://github.com/samuelbradshaw/python-scripture-scraper).
4
4
 
5
5
 
6
6
  ## Installation
@@ -39,6 +39,12 @@ https://www.churchofjesuschrist.org/study/scriptures/bofm/hel/5?id=p12&lang=eng#
39
39
 
40
40
  % scripturelookup get_church_url "/scriptures/ot" --lang "es"
41
41
  https://www.churchofjesuschrist.org/study/scriptures/ot?lang=spa
42
+
43
+ % scripturelookup detect_references "As Nephi taught in 1 Ne. 3:7, and again in 2 Nephi 2:25."
44
+ [['1 Ne. 3:7', 19, 28], ['2 Nephi 2:25', 43, 55]]
45
+
46
+ % scripturelookup refresh_metadata
47
+ Updated metadata: metadata-languages.min.json, metadata-scriptures.min.json
42
48
  ```
43
49
 
44
50
 
@@ -62,6 +68,12 @@ lookup.get_label('/scriptures/nt/1-jn/3.2-3', lang = 'cmn-Hant')
62
68
 
63
69
  lookup.get_church_url('/scriptures/ot', lang = 'es')
64
70
  # https://www.churchofjesuschrist.org/study/scriptures/ot?lang=spa
71
+
72
+ lookup.detect_references('As Nephi taught in 1 Ne. 3:7, and again in 2 Nephi 2:25.')
73
+ # [['1 Ne. 3:7', 19, 28], ['2 Nephi 2:25', 43, 55]]
74
+
75
+ lookup.refresh_metadata()
76
+ # Updated metadata: metadata-languages.min.json, metadata-scriptures.min.json
65
77
  ```
66
78
 
67
79
 
@@ -77,6 +89,8 @@ lookup.get_church_url('/scriptures/ot', lang = 'es')
77
89
  - **get_reference_objects** – Get a list of references as objects.
78
90
  - **get_reference_attributes** – Get a list of references as dictionaries.
79
91
  - **sort_references** – Sort a list of references by label or in traditional book order.
92
+ - **detect_references** – Find scripture references embedded in a string of text.
93
+ - **refresh_metadata** – Download the latest scripture and language metadata. Takes no input.
80
94
 
81
95
  ### Inputs
82
96
 
@@ -105,6 +119,8 @@ Any of the following input types are supported. You can also provide several inp
105
119
  - http://lds.org/scriptures/bofm/1-ne/3.7?lang=eng
106
120
  - [gospellibrary://content/scriptures/nt/john/3.16?lang=eng#16](https://www.churchofjesuschrist.org/study/scriptures/nt/john/3?id=p16&lang=eng#p16)
107
121
 
122
+ `detect_references` takes an arbitrary text string and returns a list of `[label, start_offset, end_offset]`. The label is the exact detected text (not normalized). Start and end offset indicate the exclusive range of characters in the input string where the reference was found.
123
+
108
124
  ### Options
109
125
 
110
126
  Several options are available. Some are only applicable to certain commands.
@@ -1,13 +1,14 @@
1
1
  # Python standard libraries
2
2
  import argparse
3
+ import inspect
3
4
 
4
5
  # Internal imports
5
- from . import data, numbers, lookup
6
+ from . import lookup
6
7
 
7
8
  def main_cli():
8
9
  parser = argparse.ArgumentParser(description='Scripture lookup')
9
10
  parser.add_argument('command', help='Command to run. Required.')
10
- parser.add_argument('input', help='Input text to parse (one or more references).')
11
+ parser.add_argument('input', nargs='?', default='', help='Input text to parse (one or more references). Not needed for commands that don’t take input, such as refresh_metadata.')
11
12
  parser.add_argument('--lang', help='Output language. Default: "en".')
12
13
  parser.add_argument('--separator', help='Separator when there are multiple results. Default: "\n".')
13
14
  parser.add_argument('--sort-by', help='Sort the returned references ("none", "traditional", or "label"). Default: "none".')
@@ -24,9 +25,16 @@ def main_cli():
24
25
 
25
26
  args = parser.parse_args()
26
27
 
27
- command = getattr(lookup, args.command)
28
- result = command(
29
- args.input,
28
+ command = getattr(lookup, args.command, None)
29
+ if not callable(command) or args.command.startswith('_'):
30
+ parser.error(f'Unknown command: “{args.command}”. See README.md for the list of commands.')
31
+
32
+ # Some commands (refresh_metadata, get_langs, get_punctuation, get_numerals) don't take input text
33
+ command_takes_input = 'input_string' in inspect.signature(command).parameters
34
+ if command_takes_input and not args.input:
35
+ parser.error(f'The “{args.command}” command needs input text.')
36
+
37
+ options = dict(
30
38
  lang = args.lang or 'en',
31
39
  separator = args.separator or '\n',
32
40
  sort_by = args.sort_by,
@@ -41,5 +49,6 @@ def main_cli():
41
49
  skip_cleanup = args.skip_cleanup,
42
50
  range_split_limit = args.range_split_limit or 1,
43
51
  )
44
-
52
+ result = command(args.input, **options) if command_takes_input else command(**options)
53
+
45
54
  print(result)
@@ -6,9 +6,8 @@ import time
6
6
  import re
7
7
  import unicodedata
8
8
 
9
- # Third-party libraries
10
- import requests
11
- from bs4 import BeautifulSoup
9
+ # Third-party libraries are imported where they're used, rather than here, so that commands that
10
+ # never reach the network don't pay for loading them. Together they add about 0.12 seconds to startup.
12
11
 
13
12
 
14
13
  data_directory = os.path.join(os.path.abspath(os.path.dirname(__file__)), 'data')
@@ -16,6 +15,8 @@ os.makedirs(data_directory, exist_ok = True)
16
15
 
17
16
  # Download JSON data
18
17
  def download_data(filename, filepath):
18
+ import requests
19
+
19
20
  request_url = f'https://cdn.jsdelivr.net/gh/samuelbradshaw/python-scripture-scraper@main/sample/{filename}'
20
21
  r = requests.get(request_url)
21
22
  r.encoding = 'utf-8'
@@ -36,10 +37,12 @@ def load_data(filename):
36
37
  download_data(filename, filepath)
37
38
  return load_data(filename)
38
39
 
39
- # Update JSON data
40
+ # Update JSON data. Returns the names of the files that were downloaded.
40
41
  def update_data():
41
- for filename in ('metadata-languages.min.json', 'metadata-scriptures.min.json',):
42
+ filenames = ('metadata-languages.min.json', 'metadata-scriptures.min.json',)
43
+ for filename in filenames:
42
44
  download_data(filename, os.path.join(data_directory, filename))
45
+ return filenames
43
46
 
44
47
  # Normalize text by removing anything that's not a letter or number, and converting to lowercase. This allows for a fuzzy comparison between input text and a known list of values.
45
48
  def normalize_for_compare(text):
@@ -47,20 +50,79 @@ def normalize_for_compare(text):
47
50
  normalized_text = ''.join([c for c in decomposed_text if unicodedata.category(c)[0] in ['L', 'N']]).lower()
48
51
  return normalized_text
49
52
 
53
+ # Words that can appear in a scripture reference, by language. Longer words should come before shorter words that they start with, so that (for example) "chapters" is matched before "chapter".
54
+ reference_words = {
55
+ 'en': {
56
+ 'list_conjunctions': ['and', '&'],
57
+ 'range_conjunctions': ['through', 'thru', 'to'],
58
+ 'verse_words': ['verses', 'verse', 'vv.', 'v.'],
59
+ 'chapter_words': ['chapters', 'chapter', 'chs.', 'ch.'],
60
+ },
61
+ 'es': {
62
+ 'list_conjunctions': ['y', 'e'],
63
+ 'range_conjunctions': ['a', 'al'],
64
+ 'verse_words': ['versículos', 'versículo'],
65
+ 'chapter_words': ['capítulos', 'capítulo'],
66
+ },
67
+ 'fr': {
68
+ 'list_conjunctions': ['et'],
69
+ 'range_conjunctions': ['à'],
70
+ 'verse_words': ['versets', 'verset'],
71
+ 'chapter_words': ['chapitres', 'chapitre'],
72
+ },
73
+ 'pt': {
74
+ 'list_conjunctions': ['e'],
75
+ 'range_conjunctions': ['a'],
76
+ 'verse_words': ['versículos', 'versículo'],
77
+ 'chapter_words': ['capítulos', 'capítulo'],
78
+ },
79
+ }
80
+ reference_word_keys = ('list_conjunctions', 'range_conjunctions', 'verse_words', 'chapter_words',)
81
+
82
+ # List conjunctions from every language, split into the ones written as a word ("and", "y", "et") and the ones written as a symbol ("&")
83
+ all_list_conjunctions = sorted({c for words in reference_words.values() for c in words['list_conjunctions']})
84
+ list_conjunction_words = [c for c in all_list_conjunctions if any(char.isalpha() for char in c)]
85
+ list_conjunction_symbols = [c for c in all_list_conjunctions if not any(char.isalpha() for char in c)]
86
+
87
+ # Get alternate spellings of a name where a spelled-out list conjunction is swapped for a symbol, or the other way around. Example: "Doctrine and Covenants" –> "Doctrine & Covenants"
88
+ def get_conjunction_aliases(name):
89
+ aliases = set()
90
+ for word in list_conjunction_words:
91
+ for symbol in list_conjunction_symbols:
92
+ if f' {word} ' in name:
93
+ aliases.add(name.replace(f' {word} ', f' {symbol} '))
94
+ if f' {symbol} ' in name:
95
+ aliases.add(name.replace(f' {symbol} ', f' {word} '))
96
+ return aliases
97
+
98
+
50
99
  languages = load_data('metadata-languages.min.json')
51
100
  scriptures = load_data('metadata-scriptures.min.json')
52
101
 
53
- scriptures['mapToSlugNormalized'] = {}
54
- for key, value in scriptures['mapToSlug'].items():
55
- normalized_key = normalize_for_compare(key)
56
- scriptures['mapToSlugNormalized'][normalized_key] = value
102
+ # Accept a symbol conjunction ("&") wherever a name spells the conjunction out, and vice versa. The words come from the language data below rather than being hard-coded, so this covers "Doctrine & Covenants" for "Doctrine and Covenants" and "Doctrina & Convenios" for "Doctrina y Convenios".
103
+ for key, value in list(scriptures['mapToSlug'].items()):
104
+ for alias in get_conjunction_aliases(key):
105
+ scriptures['mapToSlug'].setdefault(alias, value)
106
+
107
+ # Get a map of normalized names to book slugs, for comparing input that doesn't match a known name exactly. It's built the first time it's needed, rather than at import, since normalizing all ~10,000 names takes a moment and input that's already well-formed never needs it.
108
+ map_to_slug_normalized = None
109
+ def get_map_to_slug_normalized():
110
+ global map_to_slug_normalized
111
+ if map_to_slug_normalized is None:
112
+ map_to_slug_normalized = {normalize_for_compare(key): value for key, value in scriptures['mapToSlug'].items()}
113
+ return map_to_slug_normalized
57
114
 
58
- reference_separators_pattern = r'|'.join([re.escape(s.strip()) for s in scriptures['summary']['punctuation']['referenceSeparator']] + [re.escape(';'), re.escape('|'), re.escape('•'), re.escape('\n')])
59
- chapter_verse_separators_pattern = r'|'.join([re.escape(s.strip()) for s in scriptures['summary']['punctuation']['chapterVerseSeparator']] + [re.escape(':')])
60
- verse_group_separators_pattern = r'|'.join([re.escape(s.strip()) for s in scriptures['summary']['punctuation']['verseGroupSeparator']] + [re.escape(',')])
61
- verse_range_separators_pattern = r'|'.join([re.escape(s.strip()) for s in scriptures['summary']['punctuation']['verseRangeSeparator']] + [re.escape('-'), re.escape('–'), re.escape('〜')])
62
- opening_parenthesis_pattern = r'|'.join([re.escape(s.strip()) for s in scriptures['summary']['punctuation']['openingParenthesis']] + [re.escape('(')])
63
- closing_parenthesis_pattern = r'|'.join([re.escape(s.strip()) for s in scriptures['summary']['punctuation']['closingParenthesis']] + [re.escape(')')])
115
+
116
+ # Get regex patterns for the words that can appear in a scripture reference in a given language. Languages that aren't listed above return empty patterns.
117
+ reference_words_cache = {}
118
+ def get_reference_words(lang):
119
+ if lang not in reference_words_cache:
120
+ words_for_lang = reference_words.get(lang, {})
121
+ reference_words_cache[lang] = {
122
+ key: r'|'.join([re.escape(w) for w in words_for_lang.get(key, [])])
123
+ for key in reference_word_keys
124
+ }
125
+ return reference_words_cache[lang]
64
126
 
65
127
 
66
128
  # Get the BCP 47 language tag for a given language code
@@ -81,6 +143,9 @@ def get_bcp47(lang):
81
143
 
82
144
  # Get the content for a given chapter verse from python-scripture-scraper or ChurchofJesusChrist.org
83
145
  def request_content(publication_slug, book_slug, chapter, verse_groups, church_url, lang = 'en', source = 'python-scripture-scraper'):
146
+ import requests
147
+ from bs4 import BeautifulSoup
148
+
84
149
  text_content = ''
85
150
 
86
151
  if not publication_slug and book_slug and chapter:
@@ -6,10 +6,118 @@ import re
6
6
  import icu
7
7
 
8
8
  # Internal imports
9
- from . import data, numbers
10
-
9
+ from . import data, numbers, patterns
11
10
 
11
+ # Caching dicts
12
12
  natural_sort_collators = {}
13
+ book_name_cache = {}
14
+ detection_pattern_cache = {}
15
+
16
+
17
+ # Get scripture book names and abbreviations for a language, sorted longest first so that (for example) "1 John" is recognized before "John"
18
+ def get_book_names(lang):
19
+ if lang not in book_name_cache:
20
+ scripture_book_names = set()
21
+ for volume_data in data.scriptures['structure'].values():
22
+ for book_slug in volume_data['books'].keys():
23
+ book_info = data.scriptures['languages'][lang]['translatedNames'].get(book_slug)
24
+ if book_info:
25
+ book_name = patterns.whitespace.sub(' ', book_info.get('name') or '')
26
+ if book_name:
27
+ scripture_book_names.add(book_name)
28
+ book_abbrev = patterns.whitespace.sub(' ', book_info.get('abbrev') or '')
29
+ if book_abbrev:
30
+ scripture_book_names.add(book_abbrev)
31
+ scripture_book_names = sorted(scripture_book_names, key=lambda x: (-len(x), x))
32
+
33
+ # Book names with commas removed, so further normalization doesn't try to split one book name into two references. Example: "JST, Genesis 1" –> "JST Genesis 1"
34
+ scripture_book_names_without_commas = []
35
+ scripture_book_names_with_commas = []
36
+ for scripture_book_name in scripture_book_names:
37
+ scripture_book_name_without_comma = patterns.verse_group_separators.sub('', scripture_book_name)
38
+ scripture_book_names_without_commas.append(scripture_book_name_without_comma)
39
+ if patterns.verse_group_separators.search(scripture_book_name):
40
+ scripture_book_names_with_commas.append((scripture_book_name, scripture_book_name_without_comma,))
41
+
42
+ names_pattern = patterns.build_trie_pattern(scripture_book_names_without_commas)
43
+ book_name_cache[lang] = {
44
+ 'names_with_commas': scripture_book_names_with_commas,
45
+ 'separate_references_pattern': re.compile(rf'(?:^|[^\-])\b({names_pattern})', flags=re.IGNORECASE),
46
+ }
47
+ return book_name_cache[lang]
48
+
49
+
50
+ # Study helps aren't scripture references, and their abbreviations ("TG", "BD", "IT", "GS") are too short to tell apart from ordinary text
51
+ slugs_to_skip_when_detecting = ('topical-guide', 'bible-dictionary', 'index-to-the-triple-combination', 'guide-to-the-scriptures', 'joseph-smith-translation',)
52
+
53
+ # Build a regex alternation that matches anything that can start a scripture reference, allowing flexible whitespace and optional commas. Example: "JST, Genesis" –> "JST(?:,?)\sGenesis"
54
+ def build_book_names_pattern(lang):
55
+ # Every translated name is included, not just the books in "structure". That picks up publication names ("Doctrine and Covenants"), alternate forms ("Psalm" alongside "Psalms"), and things like "Official Declaration" and "Facsimile", which can all start a reference.
56
+ names = set()
57
+ for slug, translated_name in data.scriptures['languages'][lang]['translatedNames'].items():
58
+ if slug in slugs_to_skip_when_detecting:
59
+ continue
60
+ for key in ('name', 'abbrev',):
61
+ # Whitespace inside the name doesn't need to be normalized here – it's all converted to "\s" below
62
+ name = translated_name.get(key) or ''
63
+ if name:
64
+ names.add(name)
65
+ names.update(data.get_conjunction_aliases(name))
66
+
67
+ # A comma in a name is optional, so both spellings are listed. They have to be separate entries
68
+ # rather than an optional character, so that every path through the trie consumes the same
69
+ # characters – otherwise "TJS Génesis" and "TJS, Génesis 1–8" end up on branches that can both
70
+ # match the same text, and the shorter one wins.
71
+ for name in list(names):
72
+ name_without_comma = patterns.verse_group_separators.sub('', name)
73
+ if name_without_comma != name:
74
+ names.add(name_without_comma)
75
+
76
+ # Whitespace inside a name is matched loosely, so a non-breaking space in the data still matches a
77
+ # regular space in the text. That mapping is one character for one character, so the trie is safe.
78
+ normalized_names = [patterns.whitespace.sub(' ', name) for name in names]
79
+ return patterns.build_trie_pattern(normalized_names, character_patterns = {' ': r'\s'})
80
+
81
+
82
+ # Get a compiled regex for detecting scripture references embedded in text
83
+ def get_detection_pattern(lang):
84
+ if lang not in detection_pattern_cache:
85
+ words = data.get_reference_words(lang)
86
+
87
+ # Words that can be spelled out between two numbers, or between a book name and a number. Examples: "1, 2, and 3"; "Alma chapter 32 verse 21"
88
+ inline_words = r'|'.join([w for w in (words['list_conjunctions'], words['range_conjunctions'], words['verse_words'], words['chapter_words'],) if w])
89
+ leading_words = r'|'.join([w for w in (words['verse_words'], words['chapter_words'],) if w])
90
+
91
+ books = build_book_names_pattern(lang)
92
+ anchor = rf'(?P<book>{books})'
93
+ if words['chapter_words']:
94
+ anchor += rf'|(?P<chapter_word>{words["chapter_words"]})'
95
+ anchor_continued = rf'{books}|{words["chapter_words"]}' if words['chapter_words'] else books
96
+
97
+ # Separator between two numbers in a reference. A period is one of the possible chapter/verse separators, and it can't be followed by whitespace – without that restriction, "Alma 32. 5 people" would be read as "Alma 32:5". The other chapter/verse separators are unambiguous, so "Mosiah 28: 13" is fine.
98
+ chapter_verse_separators_allowing_space = r'|'.join([s for s in patterns.chapter_verse_separators_pattern.split(r'|') if s != re.escape('.')])
99
+ number_separator = rf'(?:{chapter_verse_separators_allowing_space})\s*|(?:{patterns.chapter_verse_separators_pattern})|\s*(?:{patterns.verse_group_separators_pattern}|{patterns.verse_range_separators_pattern})\s*'
100
+ if inline_words:
101
+ number_separator = rf'(?:{number_separator})(?:(?:{inline_words})\s+)?|\s+(?:{inline_words})\s+'
102
+
103
+ # Chapter and verses, ending on a digit or a closing parenthesis so that trailing whitespace and punctuation are never included in the match
104
+ leading_word = rf'(?:(?:{leading_words})\s+)?' if leading_words else ''
105
+ # A number that starts a book name belongs to the next reference, not to this one's verse list. Without this, "Alma 5 and 2 Ne. 2:25" would run together as "Alma 5 and 2".
106
+ not_the_next_book = rf'(?!(?:{books})\s*\d)'
107
+
108
+ # A chapter or verse is never more than three digits, so a longer run of digits – a year, a page number – isn't part of a reference. The lookahead rejects the whole number, rather than matching just its first three digits.
109
+ number = r'\d{1,3}(?!\d)'
110
+
111
+ context = rf'(?:\s*(?:{patterns.opening_parenthesis_pattern})\s*{number}(?:(?:{number_separator}){number})*\s*(?:{patterns.closing_parenthesis_pattern}))?'
112
+ tail = rf'{leading_word}{number}(?:(?:{number_separator}){not_the_next_book}{number})*{context}'
113
+
114
+ # Additional references that continue the same run. A conjunction can follow the separator, as in "Genesis 11:29; 22:23; and 24:15".
115
+ conjunction_after_separator = rf'(?:(?:{inline_words})\s+)?' if inline_words else ''
116
+ continued = rf'(?:\s*(?:{patterns.reference_separators_pattern})\s*{conjunction_after_separator}(?:(?:{anchor_continued})\s*)?{tail})*'
117
+
118
+ detection_pattern_cache[lang] = re.compile(rf'(?<![-\w])(?:{anchor})\s*{tail}{continued}', flags=re.IGNORECASE)
119
+ return detection_pattern_cache[lang]
120
+
13
121
 
14
122
  class Reference:
15
123
  def __init__(self, lang = 'en', publication_slug = None, book_slug = None, chapter = None, verse_groups = [], context_verse_groups = []):
@@ -61,13 +169,13 @@ class Reference:
61
169
 
62
170
  # Get localized chapter name
63
171
  def format_chapter_range(chapter_string):
64
- groups = re.split(data.verse_group_separators_pattern, chapter_string)
172
+ groups = patterns.verse_group_separators.split(chapter_string)
65
173
  new_groups = []
66
174
  for group in groups:
67
- range_parts = re.split(data.verse_range_separators_pattern, group)
175
+ range_parts = patterns.verse_range_separators.split(group)
68
176
  new_range_parts = []
69
177
  for range_part in range_parts:
70
- chapter_verse_parts = re.split(data.chapter_verse_separators_pattern, range_part)
178
+ chapter_verse_parts = patterns.chapter_verse_separators.split(range_part)
71
179
  new_chapter_verse_parts = []
72
180
  for num in chapter_verse_parts:
73
181
  new_chapter_verse_parts.append(numbers.get_formatted_number(num, target_lang = self.lang, target_custom_numerals = numerals))
@@ -109,9 +217,9 @@ class Reference:
109
217
 
110
218
  if uri and self.chapter:
111
219
  chapter = str(self.chapter)
112
- if re.match(rf'\d+{data.verse_range_separators_pattern}\d+', chapter):
220
+ if patterns.chapter_range.match(chapter):
113
221
  # Chapter range – only use the first chapter
114
- chapter = re.split(data.verse_range_separators_pattern, chapter)[0]
222
+ chapter = patterns.verse_range_separators.split(chapter)[0]
115
223
  uri += '/' + str(chapter)
116
224
  if self.verse_groups:
117
225
  if use_query_parameters:
@@ -174,8 +282,8 @@ def parse_verses_string(verses_string, lang = 'en', range_split_limit = 1):
174
282
 
175
283
  unique_verses = set()
176
284
  all_verses_are_integers = True
177
- for verse_group_string in re.split(rf'(?:{data.verse_group_separators_pattern})+', verses_string):
178
- verse_strings = re.split(data.verse_range_separators_pattern, verse_group_string)
285
+ for verse_group_string in patterns.verse_group_separators_repeated.split(verses_string):
286
+ verse_strings = patterns.verse_range_separators.split(verse_group_string)
179
287
  lower_int = numbers.convert_number_to_int(verse_strings[0])
180
288
  upper_int = numbers.convert_number_to_int(verse_strings[-1])
181
289
 
@@ -257,66 +365,61 @@ def parse_references_string(input_string, lang = 'en', sort_by = None, skip_clea
257
365
  input_string = input_string.strip().strip(punctuation_to_strip).rstrip(':').strip()
258
366
 
259
367
  # If language is English, replace roman numerals with numbers. Example: 'II Corinthians" –> "2 Corinthians"
260
- if lang == 'en':
261
- input_string = re.sub(r'\bi\s', '1', input_string, flags=re.IGNORECASE)
262
- input_string = re.sub(r'\bii\s', '2', input_string, flags=re.IGNORECASE)
263
- input_string = re.sub(r'\biii\s', '3', input_string, flags=re.IGNORECASE)
264
- input_string = re.sub(r'\biv\s', '4', input_string, flags=re.IGNORECASE)
368
+ # The substitutions have to run in order, since replacing "i " with "1" can join it to the text that follows, so they're guarded by a single search rather than combined into one pass.
369
+ if lang == 'en' and patterns.any_roman_numeral.search(input_string):
370
+ for roman_numeral_pattern, arabic_numeral in patterns.roman_numerals:
371
+ input_string = roman_numeral_pattern.sub(arabic_numeral, input_string)
265
372
 
373
+ # Normalize whitespace so it matches the spaces in book names, but keep line breaks, since they separate one reference from the next
374
+ input_string = patterns.whitespace_except_line_breaks.sub(' ', input_string)
375
+
266
376
  # Remove commas from book names so further normalization doesn't try to split it into two references. Example: "JST, Genesis 1" –> "JST Genesis 1"
267
- input_string = input_string.replace('\xa0', ' ')
268
- scripture_book_names = set()
269
- scripture_book_names_without_commas = []
270
- for volume_data in data.scriptures['structure'].values():
271
- for book_slug in volume_data['books'].keys():
272
- book_info = data.scriptures['languages'][lang]['translatedNames'].get(book_slug)
273
- if book_info:
274
- book_name = (book_info.get('name') or '').replace('\xa0', ' ')
275
- if book_name:
276
- scripture_book_names.add(book_name)
277
- book_abbrev = (book_info.get('abbrev') or '').replace('\xa0', ' ')
278
- if book_abbrev:
279
- scripture_book_names.add(book_abbrev)
280
- scripture_book_names = sorted(scripture_book_names, key=lambda x: (-len(x), x))
281
- for scripture_book_name in scripture_book_names:
282
- scripture_book_name_without_comma = re.sub(data.verse_group_separators_pattern, '', scripture_book_name)
283
- scripture_book_names_without_commas.append(scripture_book_name_without_comma)
284
- if scripture_book_name in input_string and re.search(data.verse_group_separators_pattern, scripture_book_name):
377
+ book_names = get_book_names(lang)
378
+ for scripture_book_name, scripture_book_name_without_comma in book_names['names_with_commas']:
379
+ if scripture_book_name in input_string:
285
380
  input_string = input_string.replace(scripture_book_name, scripture_book_name_without_comma)
286
-
381
+
287
382
  # Normalize whitespace-separated references. Example: "Genesis 1:2 1 Nephi 3:7" –> "; Genesis 1:2 ; 1 Nephi 3:7"
288
- scripture_book_names_pattern = '|'.join([re.escape(sbn) for sbn in scripture_book_names_without_commas])
289
- input_string = re.sub(rf'(?:^|[^\-])\b({scripture_book_names_pattern})', r'; \1', input_string, flags=re.IGNORECASE)
290
-
383
+ input_string = book_names['separate_references_pattern'].sub(r'; \1', input_string)
384
+
385
+ normalization_patterns = patterns.get_normalization_patterns(lang)
386
+
291
387
  # Normalize lists and ranges. Example: "Genesis 12:1, 2, and 3; verses 1 and 4; John 2 through 7" –> "Genesis 12:1, 2,,3; verses 1,4; John 2–7"
292
- input_string = re.sub(r'\s+(?:and|y|e|et|&)\s+(\d+)', r',\1', input_string)
293
- input_string = re.sub(r'\s+(?:through|thru|to|al|a|à)\s+(\d+)', r'–\1', input_string)
294
-
388
+ if normalization_patterns['list_conjunctions']:
389
+ input_string = normalization_patterns['list_conjunctions'].sub(r',\1', input_string)
390
+ if normalization_patterns['range_conjunctions']:
391
+ input_string = normalization_patterns['range_conjunctions'].sub(r'–\1', input_string)
392
+
295
393
  # Normalize verse sets. Example: "chapter 3 verse 7; vv. 3, 6" –> "chapter 3:7; :3, 6"
296
- input_string = re.sub(r'(?:^|\s)(?:verses|verse|vv\.|v\.|versículos|versículo|versets|verset)\s(\d+)', r':\1', input_string).replace('::', ':')
297
-
394
+ if normalization_patterns['verse_words']:
395
+ input_string = normalization_patterns['verse_words'].sub(r':\1', input_string).replace('::', ':')
396
+
397
+ # Normalize chapter words. Example: "Alma chapter 32 verse 21" –> "Alma 32:21"
398
+ if normalization_patterns['chapter_words']:
399
+ input_string = normalization_patterns['chapter_words'].sub(r' \1', input_string)
400
+
298
401
  # Normalize chapter sets. Example: "Genesis 1, 2, 4–5, Exodus 10; Alma 32" –> "Genesis 1; 2; 4–5; Exodus 10; Alma 32"
299
- if re.search(data.verse_group_separators_pattern, input_string) and not re.search(data.chapter_verse_separators_pattern, input_string):
300
- input_string = re.sub(rf'(?:{data.verse_group_separators_pattern})+', ';', input_string)
402
+ if patterns.verse_group_separators.search(input_string) and not patterns.chapter_verse_separators.search(input_string):
403
+ input_string = patterns.verse_group_separators_repeated.sub(';', input_string)
301
404
 
302
405
  # Normalize chapter:verse sets. Example: "Genesis 6:7a, 6:13a, 15; 1 Nephi 3:7 (twice), 8:21" –> "Genesis 6:7a; 6:13a, 15; 1 Nephi 3:7 (twice); 8:21"
303
- if re.search(data.chapter_verse_separators_pattern, input_string):
304
- references_list = re.split(data.reference_separators_pattern, input_string)
406
+ if patterns.chapter_verse_separators.search(input_string):
407
+ references_list = patterns.reference_separators.split(input_string)
305
408
  new_references_list = []
306
409
  for reference in references_list:
307
- reference_parts = re.split(data.verse_group_separators_pattern, reference)
410
+ reference_parts = patterns.verse_group_separators.split(reference)
308
411
  reference_input_string = ''
309
412
  for part in reference_parts:
310
413
  if reference_input_string == '':
311
414
  reference_input_string += part
312
- elif re.search(data.chapter_verse_separators_pattern, part):
415
+ elif patterns.chapter_verse_separators.search(part):
313
416
  reference_input_string += ';' + part
314
417
  else:
315
418
  reference_input_string += ',' + part
316
419
  new_references_list.append(reference_input_string)
317
420
  input_string = ';'.join(new_references_list)
318
421
 
319
- input_list = re.split(data.reference_separators_pattern, input_string)
422
+ input_list = patterns.reference_separators.split(input_string)
320
423
 
321
424
  references = []
322
425
  previous_book_slug = None
@@ -329,7 +432,7 @@ def parse_references_string(input_string, lang = 'en', sort_by = None, skip_clea
329
432
 
330
433
  if not skip_cleanup:
331
434
  # Remove trailing text. Example: "1 John 3:2 2" –> "1 John 3:2"
332
- trailing_text_match = re.match(rf'^.*?\d((?:\:|{data.closing_parenthesis_pattern})?\s+[^{data.opening_parenthesis_pattern}|\s]+)$', input_string)
435
+ trailing_text_match = patterns.trailing_text.match(input_string)
333
436
  if trailing_text_match:
334
437
  trailing_text_string = trailing_text_match.group(1)
335
438
  input_string = input_string.removesuffix(trailing_text_string)
@@ -376,8 +479,14 @@ def parse_references_string(input_string, lang = 'en', sort_by = None, skip_clea
376
479
  # Scripture reference or slug
377
480
  # Examples: Old Testament; 1 Nephi; Matthew 1; Helaman 5:12; words-of-mormon
378
481
 
379
- # Get verses string
380
- parts = re.split(data.chapter_verse_separators_pattern, input_string)
482
+ # Get context verses from a trailing parenthetical, so it doesn't interfere with parsing the rest of the reference. Example: "Gen. 1:3 (3–4)"
483
+ context_match = patterns.trailing_parenthetical.match(input_string)
484
+ if context_match and patterns.numbers_and_separators.match(context_match.group(2)):
485
+ input_string = context_match.group(1)
486
+ context_verses_string = context_match.group(2)
487
+
488
+ # Get verses string. The separator has to sit between two digits, so that the period in an abbreviation like "Gen." isn't mistaken for a chapter/verse separator.
489
+ parts = patterns.chapter_verse_separator_between_digits.split(input_string)
381
490
  if len(parts) == 2:
382
491
  # Regular chapter and verse found
383
492
  unparsed, verses_string = parts
@@ -385,11 +494,13 @@ def parse_references_string(input_string, lang = 'en', sort_by = None, skip_clea
385
494
  # Chapter only, or special case like 'Genesis 7:17–8:9' or 'Genesis 1–5'
386
495
  unparsed = input_string
387
496
  verses_string = ''
388
- verses_string, context_verses_string = (re.split(data.opening_parenthesis_pattern, re.sub(data.closing_parenthesis_pattern, '', verses_string)) + [''])[:2]
389
-
497
+ verses_string, remaining_context_verses_string = (patterns.opening_parenthesis.split(patterns.closing_parenthesis.sub('', verses_string)) + [''])[:2]
498
+ if context_verses_string is None:
499
+ context_verses_string = remaining_context_verses_string
500
+
390
501
  # Get chapter string and book string
391
502
  book_string = unparsed
392
- chapter_match = re.match(rf'^.*?(\d(?:\d|\s|{data.chapter_verse_separators_pattern}|{data.verse_range_separators_pattern}|{data.verse_group_separators_pattern})*)$', book_string)
503
+ chapter_match = patterns.trailing_chapter.match(book_string)
393
504
  if chapter_match:
394
505
  chapter_string = chapter_match.group(1)
395
506
  book_string = book_string.removesuffix(chapter_string).strip()
@@ -400,7 +511,7 @@ def parse_references_string(input_string, lang = 'en', sort_by = None, skip_clea
400
511
  book_slug = None
401
512
  skip_book_name = False
402
513
  if book_string:
403
- book_slug = data.scriptures['mapToSlug'].get(book_string, None) or data.scriptures['mapToSlugNormalized'].get(data.normalize_for_compare(book_string), None)
514
+ book_slug = data.scriptures['mapToSlug'].get(book_string, None) or data.get_map_to_slug_normalized().get(data.normalize_for_compare(book_string), None)
404
515
  # Special handling for Abraham facsimiles
405
516
  if book_slug == 'facsimiles' or (not book_slug and 'fac' in book_string.lower()):
406
517
  if previous_book_slug == 'abraham' and not previous_chapter:
@@ -456,7 +567,7 @@ def get_label(input_string, lang = 'en', separator = '\n', sort_by = None, skip_
456
567
  references = parse_references_string(input_string, lang = lang, sort_by = sort_by, skip_cleanup = skip_cleanup, range_split_limit = range_split_limit)
457
568
  return separator.join([ref.label(skip_book_name = skip_book_name, abbreviated = abbreviated) for ref in references])
458
569
 
459
- def get_church_uri(input_string, separator = '\n', sort_by = None, use_query_parameters = False, skip_cleanup = False, range_split_limit = 1, **kwargs):
570
+ def get_church_uri(input_string, lang = 'en', separator = '\n', sort_by = None, use_query_parameters = False, skip_cleanup = False, range_split_limit = 1, **kwargs):
460
571
  references = parse_references_string(input_string, lang = lang, sort_by = sort_by, skip_cleanup = skip_cleanup, range_split_limit = range_split_limit)
461
572
  return separator.join([ref.church_uri(use_query_parameters = use_query_parameters) for ref in references])
462
573
 
@@ -468,6 +579,28 @@ def get_church_link(input_string, lang = 'en', separator = '\n', sort_by = None,
468
579
  references = parse_references_string(input_string, lang = lang, sort_by = sort_by, skip_cleanup = skip_cleanup, range_split_limit = range_split_limit)
469
580
  return separator.join([ref.church_link(link_class = link_class, link_target = link_target, skip_book_name = skip_book_name, abbreviated = abbreviated, skip_lang = skip_lang, skip_fragment = skip_fragment) for ref in references])
470
581
 
582
+ def detect_references(input_string, lang = 'en', range_split_limit = 1, **kwargs):
583
+ # Every reference includes a chapter number, so text with no digits in it can be skipped without
584
+ # building the language's detection pattern or scanning for book names.
585
+ if not input_string or not patterns.any_digit.search(input_string):
586
+ return []
587
+
588
+ lang = data.get_bcp47(lang)
589
+
590
+ # Replace newlines and other whitespace with regular spaces, one character for one character, so offsets stay valid in the original string
591
+ working_string = patterns.whitespace.sub(' ', input_string)
592
+
593
+ detections = []
594
+ for match in get_detection_pattern(lang).finditer(working_string):
595
+ references = parse_references_string(match.group(0), lang = lang, range_split_limit = range_split_limit)
596
+ if match.group('book') and not any(ref.book_slug or ref.publication_slug for ref in references):
597
+ # Looked like a book name, but it didn't resolve to a known book
598
+ continue
599
+ if not any(ref.chapter for ref in references):
600
+ continue
601
+ detections.append([input_string[match.start():match.end()], match.start(), match.end()])
602
+ return detections
603
+
471
604
  def get_reference_objects(input_string, lang = 'en', sort_by = None, skip_cleanup = False, range_split_limit = 1, **kwargs):
472
605
  return parse_references_string(input_string, lang = lang, sort_by = sort_by, skip_cleanup = skip_cleanup, range_split_limit = range_split_limit)
473
606
 
@@ -475,6 +608,11 @@ def get_reference_attributes(input_string, lang = 'en', sort_by = None, skip_cle
475
608
  references = parse_references_string(input_string, lang = lang, sort_by = sort_by, skip_cleanup = skip_cleanup, range_split_limit = range_split_limit)
476
609
  return [ref.attributes() for ref in references]
477
610
 
611
+ # Download the latest scripture and language metadata. Metadata that's already been loaded stays in memory, so the new data is used from the next run onward.
612
+ def refresh_metadata(**kwargs):
613
+ filenames = data.update_data()
614
+ return 'Updated metadata: ' + ', '.join(filenames)
615
+
478
616
  def get_langs(**kwargs):
479
617
  return data.scriptures['languages'].keys()
480
618
 
@@ -0,0 +1,132 @@
1
+ # Regex patterns used for parsing and detecting scripture references.
2
+ #
3
+ # Patterns ending in "_pattern" are strings, since they get interpolated into larger patterns.
4
+ # Everything else is compiled once here, rather than being rebuilt every time a reference is parsed.
5
+
6
+ # Python standard libraries
7
+ import re
8
+
9
+ # Internal imports
10
+ from . import data
11
+
12
+
13
+ # Separators and parentheses, gathered from the punctuation used across all languages
14
+ reference_separators_pattern = r'|'.join([re.escape(s.strip()) for s in data.scriptures['summary']['punctuation']['referenceSeparator']] + [re.escape(';'), re.escape('|'), re.escape('•'), re.escape('\n')])
15
+ chapter_verse_separators_pattern = r'|'.join([re.escape(s.strip()) for s in data.scriptures['summary']['punctuation']['chapterVerseSeparator']] + [re.escape(':'), re.escape('.')])
16
+ verse_group_separators_pattern = r'|'.join([re.escape(s.strip()) for s in data.scriptures['summary']['punctuation']['verseGroupSeparator']] + [re.escape(',')])
17
+ verse_range_separators_pattern = r'|'.join([re.escape(s.strip()) for s in data.scriptures['summary']['punctuation']['verseRangeSeparator']] + [re.escape('-'), re.escape('–'), re.escape('〜'), re.escape('~')])
18
+ opening_parenthesis_pattern = r'|'.join([re.escape(s.strip()) for s in data.scriptures['summary']['punctuation']['openingParenthesis']] + [re.escape('(')])
19
+ closing_parenthesis_pattern = r'|'.join([re.escape(s.strip()) for s in data.scriptures['summary']['punctuation']['closingParenthesis']] + [re.escape(')')])
20
+
21
+ # Any Unicode whitespace character – newlines, tabs, non-breaking spaces, ideographic spaces, and so on
22
+ whitespace = re.compile(r'\s')
23
+ # The same, except line breaks, which separate one reference from the next
24
+ whitespace_except_line_breaks = re.compile(r'[^\S\n]')
25
+
26
+ # Separators between references, chapters, verses, and verse groups
27
+ reference_separators = re.compile(reference_separators_pattern)
28
+ chapter_verse_separators = re.compile(chapter_verse_separators_pattern)
29
+ verse_group_separators = re.compile(verse_group_separators_pattern)
30
+ verse_range_separators = re.compile(verse_range_separators_pattern)
31
+ verse_group_separators_repeated = re.compile(rf'(?:{verse_group_separators_pattern})+')
32
+ opening_parenthesis = re.compile(opening_parenthesis_pattern)
33
+ closing_parenthesis = re.compile(closing_parenthesis_pattern)
34
+
35
+ # A chapter/verse separator that sits between two digits, so the period in an abbreviation like "Gen." isn't mistaken for one
36
+ chapter_verse_separator_between_digits = re.compile(rf'(?<=\d)(?:{chapter_verse_separators_pattern})(?=\d)')
37
+ # A range of chapters. Example: "1–5"
38
+ chapter_range = re.compile(rf'\d+(?:{verse_range_separators_pattern})\d+')
39
+
40
+ # Anything that can follow the first digit of a chapter or verse: another digit, whitespace, or any separator
41
+ number_continuation_pattern = rf'\d|\s|{chapter_verse_separators_pattern}|{verse_range_separators_pattern}|{verse_group_separators_pattern}'
42
+ # A string made up entirely of numbers and separators. Example: "3, 5–7"
43
+ numbers_and_separators = re.compile(rf'^\d(?:{number_continuation_pattern})*$')
44
+ # The chapter and verses at the end of a reference, so the book name can be split off the front
45
+ trailing_chapter = re.compile(rf'^.*?(\d(?:{number_continuation_pattern})*)$')
46
+ # Text after the end of a reference. Example: "1 John 3:2 2" –> " 2"
47
+ trailing_text = re.compile(rf'^.*?\d((?:\:|{closing_parenthesis_pattern})?\s+[^{opening_parenthesis_pattern}|\s]+)$')
48
+ # A parenthetical at the end of a reference, which holds context verses. Example: "Gen. 1:3 (3–4)"
49
+ trailing_parenthetical = re.compile(rf'^(.*?)\s*(?:{opening_parenthesis_pattern})([^))]*)(?:{closing_parenthesis_pattern})$')
50
+
51
+ # English roman numeral prefixes, paired with the number they stand for. Example: "II Corinthians" –> "2 Corinthians"
52
+ roman_numerals = [
53
+ (re.compile(r'\bi\s', flags=re.IGNORECASE), '1'),
54
+ (re.compile(r'\bii\s', flags=re.IGNORECASE), '2'),
55
+ (re.compile(r'\biii\s', flags=re.IGNORECASE), '3'),
56
+ (re.compile(r'\biv\s', flags=re.IGNORECASE), '4'),
57
+ ]
58
+ # Matches any of the prefixes above, so a string with no roman numeral at all – which is most of them – can skip the substitutions entirely
59
+ any_roman_numeral = re.compile(r'\b(?:i|ii|iii|iv)\s', flags=re.IGNORECASE)
60
+
61
+ # A single digit. Every reference needs a chapter number, so text with no digits anywhere can't contain one.
62
+ any_digit = re.compile(r'\d')
63
+
64
+
65
+ # Build a regex alternation that matches any of the given words, sharing common prefixes so the regex
66
+ # engine doesn't retry every word at every position. Example: ["Genesis", "Genes"] –> "Gene(?:sis|s)".
67
+ # Shared prefixes make this several times faster than a flat "Genesis|Genes" alternation, which the
68
+ # engine has to walk one branch at a time. Longer words still win over shorter ones that they start
69
+ # with, because the trailing group is greedy.
70
+ #
71
+ # Each character can be given its own sub-pattern, for names where a character shouldn't be matched
72
+ # literally – see character_patterns in build_book_names_pattern.
73
+ def build_trie_pattern(words, character_patterns = None, case_insensitive = True):
74
+ character_patterns = character_patterns or {}
75
+
76
+ # Characters that differ only by case have to share a trie node, since the pattern is meant to be
77
+ # compiled with re.IGNORECASE. Without this, "OP" and "Opisyal" would sit in sibling branches, and
78
+ # "OP" would win on the input "Opisyal" – a flat alternation avoids that by sorting longest first.
79
+ def trie_key(character):
80
+ lowercased = character.lower()
81
+ return lowercased if case_insensitive and len(lowercased) == 1 else character
82
+
83
+ trie = {}
84
+ for word in words:
85
+ node = trie
86
+ for character in word:
87
+ node = node.setdefault(trie_key(character), {})
88
+ node[''] = {}
89
+
90
+ def build_branch(node):
91
+ # A node with nothing but an end marker is the end of a word
92
+ if '' in node and len(node) == 1:
93
+ return None
94
+
95
+ branches = []
96
+ word_ends_here = False
97
+ for character, child_node in sorted(node.items()):
98
+ if character == '':
99
+ word_ends_here = True
100
+ continue
101
+ remainder = build_branch(child_node)
102
+ character_pattern = character_patterns.get(character) or re.escape(character)
103
+ branches.append(character_pattern + (f'(?:{remainder})' if remainder and len(remainder) > 1 else (remainder or '')))
104
+
105
+ branch_pattern = r'|'.join(branches)
106
+ return f'(?:{branch_pattern})?' if word_ends_here else branch_pattern
107
+
108
+ return build_branch(trie)
109
+
110
+
111
+ # Get compiled regexes for normalizing the words that can appear in a reference in a given language
112
+ normalization_pattern_cache = {}
113
+ def get_normalization_patterns(lang):
114
+ if lang not in normalization_pattern_cache:
115
+ words = data.get_reference_words(lang)
116
+
117
+ # A conjunction sits between two numbers, so it needs whitespace on both sides. A verse or chapter word can also start the string.
118
+ conjunction_template = r'\s+(?:{words})\s+(\d+)'
119
+ reference_word_template = r'(?:^|\s)(?:{words})\s+(\d+)'
120
+ templates = {
121
+ 'list_conjunctions': conjunction_template,
122
+ 'range_conjunctions': conjunction_template,
123
+ 'verse_words': reference_word_template,
124
+ 'chapter_words': reference_word_template,
125
+ }
126
+
127
+ normalization_patterns = {}
128
+ for key in data.reference_word_keys:
129
+ template = templates[key]
130
+ normalization_patterns[key] = re.compile(template.format(words = words[key]), flags=re.IGNORECASE) if words[key] else None
131
+ normalization_pattern_cache[lang] = normalization_patterns
132
+ return normalization_pattern_cache[lang]
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: scripturelookup
3
- Version: 0.0.7
3
+ Version: 0.1.0
4
4
  Summary: Python and command-line utility for converting scripture references between formats.
5
5
  Author-email: Samuel Bradshaw <samuel.h.bradshaw@gmail.com>
6
6
  License: MIT License
@@ -40,7 +40,7 @@ Dynamic: license-file
40
40
 
41
41
  # Scripture Lookup
42
42
 
43
- Scripture Lookup is a Python library and command-line tool for looking up scripture verses or chapters. It can also convert scripture references between formats and languages. The tool is built around data from [Python Scripture Scraper](https://github.com/samuelbradshaw/python-scripture-scraper).
43
+ Scripture Lookup is a Python library and command-line tool for looking up scripture verses or chapters. It can also detect scripture references in a string, and convert them between formats and languages. The tool uses scripture and language metadata from [Python Scripture Scraper](https://github.com/samuelbradshaw/python-scripture-scraper).
44
44
 
45
45
 
46
46
  ## Installation
@@ -79,6 +79,12 @@ https://www.churchofjesuschrist.org/study/scriptures/bofm/hel/5?id=p12&lang=eng#
79
79
 
80
80
  % scripturelookup get_church_url "/scriptures/ot" --lang "es"
81
81
  https://www.churchofjesuschrist.org/study/scriptures/ot?lang=spa
82
+
83
+ % scripturelookup detect_references "As Nephi taught in 1 Ne. 3:7, and again in 2 Nephi 2:25."
84
+ [['1 Ne. 3:7', 19, 28], ['2 Nephi 2:25', 43, 55]]
85
+
86
+ % scripturelookup refresh_metadata
87
+ Updated metadata: metadata-languages.min.json, metadata-scriptures.min.json
82
88
  ```
83
89
 
84
90
 
@@ -102,6 +108,12 @@ lookup.get_label('/scriptures/nt/1-jn/3.2-3', lang = 'cmn-Hant')
102
108
 
103
109
  lookup.get_church_url('/scriptures/ot', lang = 'es')
104
110
  # https://www.churchofjesuschrist.org/study/scriptures/ot?lang=spa
111
+
112
+ lookup.detect_references('As Nephi taught in 1 Ne. 3:7, and again in 2 Nephi 2:25.')
113
+ # [['1 Ne. 3:7', 19, 28], ['2 Nephi 2:25', 43, 55]]
114
+
115
+ lookup.refresh_metadata()
116
+ # Updated metadata: metadata-languages.min.json, metadata-scriptures.min.json
105
117
  ```
106
118
 
107
119
 
@@ -117,6 +129,8 @@ lookup.get_church_url('/scriptures/ot', lang = 'es')
117
129
  - **get_reference_objects** – Get a list of references as objects.
118
130
  - **get_reference_attributes** – Get a list of references as dictionaries.
119
131
  - **sort_references** – Sort a list of references by label or in traditional book order.
132
+ - **detect_references** – Find scripture references embedded in a string of text.
133
+ - **refresh_metadata** – Download the latest scripture and language metadata. Takes no input.
120
134
 
121
135
  ### Inputs
122
136
 
@@ -145,6 +159,8 @@ Any of the following input types are supported. You can also provide several inp
145
159
  - http://lds.org/scriptures/bofm/1-ne/3.7?lang=eng
146
160
  - [gospellibrary://content/scriptures/nt/john/3.16?lang=eng#16](https://www.churchofjesuschrist.org/study/scriptures/nt/john/3?id=p16&lang=eng#p16)
147
161
 
162
+ `detect_references` takes an arbitrary text string and returns a list of `[label, start_offset, end_offset]`. The label is the exact detected text (not normalized). Start and end offset indicate the exclusive range of characters in the input string where the reference was found.
163
+
148
164
  ### Options
149
165
 
150
166
  Several options are available. Some are only applicable to certain commands.
@@ -16,11 +16,15 @@ src/scripturelookup/command_line.py
16
16
  src/scripturelookup/data.py
17
17
  src/scripturelookup/lookup.py
18
18
  src/scripturelookup/numbers.py
19
+ src/scripturelookup/patterns.py
19
20
  src/scripturelookup.egg-info/PKG-INFO
20
21
  src/scripturelookup.egg-info/SOURCES.txt
21
22
  src/scripturelookup.egg-info/dependency_links.txt
22
23
  src/scripturelookup.egg-info/entry_points.txt
23
24
  src/scripturelookup.egg-info/requires.txt
25
+ src/scripturelookup.egg-info/scm_file_list.json
26
+ src/scripturelookup.egg-info/scm_version.json
24
27
  src/scripturelookup.egg-info/top_level.txt
25
28
  src/scripturelookup/data/metadata-languages.min.json
26
- src/scripturelookup/data/metadata-scriptures.min.json
29
+ src/scripturelookup/data/metadata-scriptures.min.json
30
+ tests/test_detect.py
@@ -0,0 +1,26 @@
1
+ {
2
+ "files": [
3
+ ".github/workflows/publish.yml",
4
+ ".gitignore",
5
+ "LICENSE",
6
+ "README.md",
7
+ "pyproject.toml",
8
+ "requirements.txt",
9
+ "src/geezify-python-main/LICENSE",
10
+ "src/geezify-python-main/README.md",
11
+ "src/geezify-python-main/arabify.py",
12
+ "src/geezify-python-main/arabify_test.py",
13
+ "src/geezify-python-main/geezify.py",
14
+ "src/geezify-python-main/geezify_test.py",
15
+ "src/geezify-python-main/test_data.py",
16
+ "src/scripturelookup/__init__.py",
17
+ "src/scripturelookup/command_line.py",
18
+ "src/scripturelookup/data.py",
19
+ "src/scripturelookup/data/metadata-languages.min.json",
20
+ "src/scripturelookup/data/metadata-scriptures.min.json",
21
+ "src/scripturelookup/lookup.py",
22
+ "src/scripturelookup/numbers.py",
23
+ "src/scripturelookup/patterns.py",
24
+ "tests/test_detect.py"
25
+ ]
26
+ }
@@ -0,0 +1,8 @@
1
+ {
2
+ "tag": "0.1.0",
3
+ "distance": 0,
4
+ "node": "ga8b30f8a91100afe8fd53885f129ff7f22d83e70",
5
+ "dirty": false,
6
+ "branch": "HEAD",
7
+ "node_date": "2026-08-28"
8
+ }
@@ -0,0 +1,137 @@
1
+ # Tests for lookup.detect_references
2
+ #
3
+ # Run with:
4
+ # python3 -m pytest tests
5
+
6
+ import os
7
+ import sys
8
+
9
+ import pytest
10
+
11
+ sys.path.insert(0, os.path.join(os.path.abspath(os.path.dirname(__file__)), '..', 'src'))
12
+ from scripturelookup import lookup
13
+
14
+
15
+ # (input text, language, expected labels in order)
16
+ detection_cases = [
17
+ # Nothing to find
18
+ ('', 'en', []),
19
+ ('Nothing here at all.', 'en', []),
20
+
21
+ # A run of related references stays a single detection
22
+ ('See 1 Nephi 3:7, 5:2; Alma 5 for details.', 'en', ['1 Nephi 3:7, 5:2; Alma 5']),
23
+ ('Doctrine and Covenants 45:56-59 | 1 Nephi 22:24-28', 'en', ['Doctrine and Covenants 45:56-59 | 1 Nephi 22:24-28']),
24
+ ('Doctrine & Covenants 45:56 • 1 Nephi 22:24', 'en', ['Doctrine & Covenants 45:56 • 1 Nephi 22:24']),
25
+
26
+ # "&" stands in for a spelled-out list conjunction in any language, not just English
27
+ ('Vea Doctrina & Convenios 45:56 aquí.', 'es', ['Doctrina & Convenios 45:56']),
28
+ ('Vea Doctrina y Convenios 45:56 aquí.', 'es', ['Doctrina y Convenios 45:56']),
29
+
30
+ # Abbreviations
31
+ ('In 2 Ne. 2:25 we read.', 'en', ['2 Ne. 2:25']),
32
+ ('Compare Matt. 5:3 today.', 'en', ['Matt. 5:3']),
33
+
34
+ # A number is required, so ordinary words that are also book names are skipped
35
+ ('He found a job in Job 5.', 'en', ['Job 5']),
36
+ ('Read the Book of Mormon.', 'en', []),
37
+
38
+ # Trailing whitespace and unbalanced parentheses are never part of a match
39
+ ('Read Genesis 1-3 (see also Moses 2).', 'en', ['Genesis 1-3', 'Moses 2']),
40
+ ('Gen. 1:3 (3-4) is nice.', 'en', ['Gen. 1:3 (3-4)']),
41
+
42
+ # A period is a chapter/verse separator in some languages, but only between digits
43
+ ('Alma 32. 5 people came.', 'en', ['Alma 32']),
44
+
45
+ # A chapter or verse is never more than three digits, so years and page numbers aren't references.
46
+ # The whole number is rejected rather than being truncated to its first three digits.
47
+ ('Psalm 119:176 is the last verse.', 'en', ['Psalm 119:176']),
48
+ ('D&C 124:123-45 was given.', 'en', ['D&C 124:123-45']),
49
+ ('Alma 1978 was a year.', 'en', []),
50
+ ('Alma 12345 xyz', 'en', []),
51
+ ('Alma 32:21 was quoted in 1978.', 'en', ['Alma 32:21']),
52
+
53
+ # Chapter words can stand in for a book name; verse words cannot
54
+ ('See chapter 3 and Alma chapter 32 verse 21.', 'en', ['chapter 3', 'Alma chapter 32 verse 21']),
55
+ ('See verses 3-5 below.', 'en', []),
56
+
57
+ # Bare chapter:verse with no preceding anchor is not a reference
58
+ ('As in Alma 32. See also 3:7.', 'en', ['Alma 32']),
59
+ ('Meet at 3:15 tomorrow.', 'en', []),
60
+
61
+ # Spelled-out lists and ranges
62
+ ('Genesis 12:1, 2, and 3 plus John 2 through 7.', 'en', ['Genesis 12:1, 2, and 3', 'John 2 through 7']),
63
+
64
+ # A number that starts the next book name isn't pulled into this reference's verse list
65
+ ('See 1 Nephi 3:7, 5:2; Alma 5 and 2 Ne. 2:25 in chapter 9.', 'en', ['1 Nephi 3:7, 5:2; Alma 5', '2 Ne. 2:25', 'chapter 9']),
66
+ ('Alma 5, 2 Ne. 2:25 today.', 'en', ['Alma 5', '2 Ne. 2:25']),
67
+ ('Read 1 Nephi 3 and 3 John 1:4.', 'en', ['1 Nephi 3', '3 John 1:4']),
68
+
69
+ # A conjunction can follow a reference separator
70
+ ('mentioned in Genesis 11:29; 22:23; and 24:15 as well as in Abraham 2:2.', 'en', ['Genesis 11:29; 22:23; and 24:15', 'Abraham 2:2']),
71
+
72
+ # A colon may be followed by a space, but a period may not (a period is also a chapter/verse separator)
73
+ ('Mosiah 28: 13-15 is next.', 'en', ['Mosiah 28: 13-15']),
74
+ ('Genesis 1.3 and Abr. 3.22-23 here.', 'en', ['Genesis 1.3', 'Abr. 3.22-23']),
75
+ ('It cost 1.50 today.', 'en', []),
76
+
77
+ # Names that aren't in the "structure" book list
78
+ ('See Psalm 23 and Official Declaration 2 and Facsimile 2.', 'en', ['Psalm 23', 'Official Declaration 2', 'Facsimile 2']),
79
+
80
+ # Line-wrapped references are found, and the label keeps the original whitespace
81
+ ('Alma\n32:21 was quoted.', 'en', ['Alma\n32:21']),
82
+ ('Alma\r\n32:21 was quoted.', 'en', ['Alma\r\n32:21']),
83
+ ('See Alma\n 32:21, 22 here.', 'en', ['Alma\n 32:21, 22']),
84
+
85
+ # Any kind of Unicode whitespace works, not just a regular space
86
+ ('See 1 Samuel 3:1 now.', 'en', ['1 Samuel 3:1']),
87
+ ('See 1\xa0Samuel 3:1 now.', 'en', ['1\xa0Samuel 3:1']),
88
+ ('See Alma 32:21 now.', 'en', ['Alma 32:21']),
89
+
90
+ # Reference words are language-specific. German has no seeded word list, so "capítulo 3" isn't a
91
+ # reference there, but "Alma" is spelled the same in German and Spanish and is still detected.
92
+ ('Consulte el capítulo 3 y Alma 32:21 aquí.', 'es', ['capítulo 3', 'Alma 32:21']),
93
+ ('Consulte el capítulo 3 y Alma 32:21 aquí.', 'de', ['Alma 32:21']),
94
+
95
+ # Book names are language-specific too
96
+ ('Siehe Alma 32:21 hier.', 'de', ['Alma 32:21']),
97
+ ('Siehe Alma 32:21 hier.', 'ko', []),
98
+ ]
99
+
100
+
101
+ @pytest.mark.parametrize('text, lang, expected_labels', detection_cases)
102
+ def test_detect_references(text, lang, expected_labels):
103
+ detections = lookup.detect_references(text, lang = lang)
104
+ assert [label for label, start, end in detections] == expected_labels
105
+
106
+
107
+ @pytest.mark.parametrize('text, lang, expected_labels', detection_cases)
108
+ def test_offsets_match_the_original_string(text, lang, expected_labels):
109
+ for label, start, end in lookup.detect_references(text, lang = lang):
110
+ assert text[start:end] == label
111
+
112
+
113
+ @pytest.mark.parametrize('text, lang, expected_labels', detection_cases)
114
+ def test_detections_do_not_overlap(text, lang, expected_labels):
115
+ previous_end = 0
116
+ for label, start, end in lookup.detect_references(text, lang = lang):
117
+ assert start >= previous_end
118
+ assert start < end
119
+ previous_end = end
120
+
121
+
122
+ def test_detected_labels_can_be_parsed_back():
123
+ text = 'See 1 Nephi 3:7 and Mosiah 2:17 for details.'
124
+ labels = [lookup.get_label(label) for label, start, end in lookup.detect_references(text)]
125
+ assert labels == ['1\xa0Nephi\xa03:7', 'Mosiah\xa02:17']
126
+
127
+
128
+ # Parsing (not detection) treats a line break as a separator between references, so whitespace
129
+ # normalization there has to leave line breaks alone.
130
+ def test_line_breaks_still_separate_references_when_parsing():
131
+ assert lookup.get_label('Alma 5\nMosiah 2:17') == 'Alma\xa05\nMosiah\xa02:17'
132
+ assert lookup.get_label('Alma 5\r\nMosiah 2:17') == 'Alma\xa05\nMosiah\xa02:17'
133
+
134
+
135
+ @pytest.mark.parametrize('space', [' ', '\xa0', ' ', ' '])
136
+ def test_book_names_match_any_kind_of_space(space):
137
+ assert lookup.get_label(f'1{space}Samuel 3:1') == '1\xa0Samuel\xa03:1'
File without changes