scripturelookup 0.0.7__tar.gz → 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {scripturelookup-0.0.7/src/scripturelookup.egg-info → scripturelookup-0.1.0}/PKG-INFO +18 -2
- {scripturelookup-0.0.7 → scripturelookup-0.1.0}/README.md +17 -1
- {scripturelookup-0.0.7 → scripturelookup-0.1.0}/src/scripturelookup/command_line.py +15 -6
- {scripturelookup-0.0.7 → scripturelookup-0.1.0}/src/scripturelookup/data.py +80 -15
- {scripturelookup-0.0.7 → scripturelookup-0.1.0}/src/scripturelookup/lookup.py +194 -56
- scripturelookup-0.1.0/src/scripturelookup/patterns.py +132 -0
- {scripturelookup-0.0.7 → scripturelookup-0.1.0/src/scripturelookup.egg-info}/PKG-INFO +18 -2
- {scripturelookup-0.0.7 → scripturelookup-0.1.0}/src/scripturelookup.egg-info/SOURCES.txt +5 -1
- scripturelookup-0.1.0/src/scripturelookup.egg-info/scm_file_list.json +26 -0
- scripturelookup-0.1.0/src/scripturelookup.egg-info/scm_version.json +8 -0
- scripturelookup-0.1.0/tests/test_detect.py +137 -0
- {scripturelookup-0.0.7 → scripturelookup-0.1.0}/.github/workflows/publish.yml +0 -0
- {scripturelookup-0.0.7 → scripturelookup-0.1.0}/.gitignore +0 -0
- {scripturelookup-0.0.7 → scripturelookup-0.1.0}/LICENSE +0 -0
- {scripturelookup-0.0.7 → scripturelookup-0.1.0}/pyproject.toml +0 -0
- {scripturelookup-0.0.7 → scripturelookup-0.1.0}/requirements.txt +0 -0
- {scripturelookup-0.0.7 → scripturelookup-0.1.0}/setup.cfg +0 -0
- {scripturelookup-0.0.7 → scripturelookup-0.1.0}/src/geezify-python-main/LICENSE +0 -0
- {scripturelookup-0.0.7 → scripturelookup-0.1.0}/src/geezify-python-main/README.md +0 -0
- {scripturelookup-0.0.7 → scripturelookup-0.1.0}/src/geezify-python-main/arabify.py +0 -0
- {scripturelookup-0.0.7 → scripturelookup-0.1.0}/src/geezify-python-main/arabify_test.py +0 -0
- {scripturelookup-0.0.7 → scripturelookup-0.1.0}/src/geezify-python-main/geezify.py +0 -0
- {scripturelookup-0.0.7 → scripturelookup-0.1.0}/src/geezify-python-main/geezify_test.py +0 -0
- {scripturelookup-0.0.7 → scripturelookup-0.1.0}/src/geezify-python-main/test_data.py +0 -0
- {scripturelookup-0.0.7 → scripturelookup-0.1.0}/src/scripturelookup/__init__.py +0 -0
- {scripturelookup-0.0.7 → scripturelookup-0.1.0}/src/scripturelookup/data/metadata-languages.min.json +0 -0
- {scripturelookup-0.0.7 → scripturelookup-0.1.0}/src/scripturelookup/data/metadata-scriptures.min.json +0 -0
- {scripturelookup-0.0.7 → scripturelookup-0.1.0}/src/scripturelookup/numbers.py +0 -0
- {scripturelookup-0.0.7 → scripturelookup-0.1.0}/src/scripturelookup.egg-info/dependency_links.txt +0 -0
- {scripturelookup-0.0.7 → scripturelookup-0.1.0}/src/scripturelookup.egg-info/entry_points.txt +0 -0
- {scripturelookup-0.0.7 → scripturelookup-0.1.0}/src/scripturelookup.egg-info/requires.txt +0 -0
- {scripturelookup-0.0.7 → scripturelookup-0.1.0}/src/scripturelookup.egg-info/top_level.txt +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: scripturelookup
|
|
3
|
-
Version: 0.0
|
|
3
|
+
Version: 0.1.0
|
|
4
4
|
Summary: Python and command-line utility for converting scripture references between formats.
|
|
5
5
|
Author-email: Samuel Bradshaw <samuel.h.bradshaw@gmail.com>
|
|
6
6
|
License: MIT License
|
|
@@ -40,7 +40,7 @@ Dynamic: license-file
|
|
|
40
40
|
|
|
41
41
|
# Scripture Lookup
|
|
42
42
|
|
|
43
|
-
Scripture Lookup is a Python library and command-line tool for looking up scripture verses or chapters. It can also
|
|
43
|
+
Scripture Lookup is a Python library and command-line tool for looking up scripture verses or chapters. It can also detect scripture references in a string, and convert them between formats and languages. The tool uses scripture and language metadata from [Python Scripture Scraper](https://github.com/samuelbradshaw/python-scripture-scraper).
|
|
44
44
|
|
|
45
45
|
|
|
46
46
|
## Installation
|
|
@@ -79,6 +79,12 @@ https://www.churchofjesuschrist.org/study/scriptures/bofm/hel/5?id=p12&lang=eng#
|
|
|
79
79
|
|
|
80
80
|
% scripturelookup get_church_url "/scriptures/ot" --lang "es"
|
|
81
81
|
https://www.churchofjesuschrist.org/study/scriptures/ot?lang=spa
|
|
82
|
+
|
|
83
|
+
% scripturelookup detect_references "As Nephi taught in 1 Ne. 3:7, and again in 2 Nephi 2:25."
|
|
84
|
+
[['1 Ne. 3:7', 19, 28], ['2 Nephi 2:25', 43, 55]]
|
|
85
|
+
|
|
86
|
+
% scripturelookup refresh_metadata
|
|
87
|
+
Updated metadata: metadata-languages.min.json, metadata-scriptures.min.json
|
|
82
88
|
```
|
|
83
89
|
|
|
84
90
|
|
|
@@ -102,6 +108,12 @@ lookup.get_label('/scriptures/nt/1-jn/3.2-3', lang = 'cmn-Hant')
|
|
|
102
108
|
|
|
103
109
|
lookup.get_church_url('/scriptures/ot', lang = 'es')
|
|
104
110
|
# https://www.churchofjesuschrist.org/study/scriptures/ot?lang=spa
|
|
111
|
+
|
|
112
|
+
lookup.detect_references('As Nephi taught in 1 Ne. 3:7, and again in 2 Nephi 2:25.')
|
|
113
|
+
# [['1 Ne. 3:7', 19, 28], ['2 Nephi 2:25', 43, 55]]
|
|
114
|
+
|
|
115
|
+
lookup.refresh_metadata()
|
|
116
|
+
# Updated metadata: metadata-languages.min.json, metadata-scriptures.min.json
|
|
105
117
|
```
|
|
106
118
|
|
|
107
119
|
|
|
@@ -117,6 +129,8 @@ lookup.get_church_url('/scriptures/ot', lang = 'es')
|
|
|
117
129
|
- **get_reference_objects** – Get a list of references as objects.
|
|
118
130
|
- **get_reference_attributes** – Get a list of references as dictionaries.
|
|
119
131
|
- **sort_references** – Sort a list of references by label or in traditional book order.
|
|
132
|
+
- **detect_references** – Find scripture references embedded in a string of text.
|
|
133
|
+
- **refresh_metadata** – Download the latest scripture and language metadata. Takes no input.
|
|
120
134
|
|
|
121
135
|
### Inputs
|
|
122
136
|
|
|
@@ -145,6 +159,8 @@ Any of the following input types are supported. You can also provide several inp
|
|
|
145
159
|
- http://lds.org/scriptures/bofm/1-ne/3.7?lang=eng
|
|
146
160
|
- [gospellibrary://content/scriptures/nt/john/3.16?lang=eng#16](https://www.churchofjesuschrist.org/study/scriptures/nt/john/3?id=p16&lang=eng#p16)
|
|
147
161
|
|
|
162
|
+
`detect_references` takes an arbitrary text string and returns a list of `[label, start_offset, end_offset]`. The label is the exact detected text (not normalized). Start and end offset indicate the exclusive range of characters in the input string where the reference was found.
|
|
163
|
+
|
|
148
164
|
### Options
|
|
149
165
|
|
|
150
166
|
Several options are available. Some are only applicable to certain commands.
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
# Scripture Lookup
|
|
2
2
|
|
|
3
|
-
Scripture Lookup is a Python library and command-line tool for looking up scripture verses or chapters. It can also
|
|
3
|
+
Scripture Lookup is a Python library and command-line tool for looking up scripture verses or chapters. It can also detect scripture references in a string, and convert them between formats and languages. The tool uses scripture and language metadata from [Python Scripture Scraper](https://github.com/samuelbradshaw/python-scripture-scraper).
|
|
4
4
|
|
|
5
5
|
|
|
6
6
|
## Installation
|
|
@@ -39,6 +39,12 @@ https://www.churchofjesuschrist.org/study/scriptures/bofm/hel/5?id=p12&lang=eng#
|
|
|
39
39
|
|
|
40
40
|
% scripturelookup get_church_url "/scriptures/ot" --lang "es"
|
|
41
41
|
https://www.churchofjesuschrist.org/study/scriptures/ot?lang=spa
|
|
42
|
+
|
|
43
|
+
% scripturelookup detect_references "As Nephi taught in 1 Ne. 3:7, and again in 2 Nephi 2:25."
|
|
44
|
+
[['1 Ne. 3:7', 19, 28], ['2 Nephi 2:25', 43, 55]]
|
|
45
|
+
|
|
46
|
+
% scripturelookup refresh_metadata
|
|
47
|
+
Updated metadata: metadata-languages.min.json, metadata-scriptures.min.json
|
|
42
48
|
```
|
|
43
49
|
|
|
44
50
|
|
|
@@ -62,6 +68,12 @@ lookup.get_label('/scriptures/nt/1-jn/3.2-3', lang = 'cmn-Hant')
|
|
|
62
68
|
|
|
63
69
|
lookup.get_church_url('/scriptures/ot', lang = 'es')
|
|
64
70
|
# https://www.churchofjesuschrist.org/study/scriptures/ot?lang=spa
|
|
71
|
+
|
|
72
|
+
lookup.detect_references('As Nephi taught in 1 Ne. 3:7, and again in 2 Nephi 2:25.')
|
|
73
|
+
# [['1 Ne. 3:7', 19, 28], ['2 Nephi 2:25', 43, 55]]
|
|
74
|
+
|
|
75
|
+
lookup.refresh_metadata()
|
|
76
|
+
# Updated metadata: metadata-languages.min.json, metadata-scriptures.min.json
|
|
65
77
|
```
|
|
66
78
|
|
|
67
79
|
|
|
@@ -77,6 +89,8 @@ lookup.get_church_url('/scriptures/ot', lang = 'es')
|
|
|
77
89
|
- **get_reference_objects** – Get a list of references as objects.
|
|
78
90
|
- **get_reference_attributes** – Get a list of references as dictionaries.
|
|
79
91
|
- **sort_references** – Sort a list of references by label or in traditional book order.
|
|
92
|
+
- **detect_references** – Find scripture references embedded in a string of text.
|
|
93
|
+
- **refresh_metadata** – Download the latest scripture and language metadata. Takes no input.
|
|
80
94
|
|
|
81
95
|
### Inputs
|
|
82
96
|
|
|
@@ -105,6 +119,8 @@ Any of the following input types are supported. You can also provide several inp
|
|
|
105
119
|
- http://lds.org/scriptures/bofm/1-ne/3.7?lang=eng
|
|
106
120
|
- [gospellibrary://content/scriptures/nt/john/3.16?lang=eng#16](https://www.churchofjesuschrist.org/study/scriptures/nt/john/3?id=p16&lang=eng#p16)
|
|
107
121
|
|
|
122
|
+
`detect_references` takes an arbitrary text string and returns a list of `[label, start_offset, end_offset]`. The label is the exact detected text (not normalized). Start and end offset indicate the exclusive range of characters in the input string where the reference was found.
|
|
123
|
+
|
|
108
124
|
### Options
|
|
109
125
|
|
|
110
126
|
Several options are available. Some are only applicable to certain commands.
|
|
@@ -1,13 +1,14 @@
|
|
|
1
1
|
# Python standard libraries
|
|
2
2
|
import argparse
|
|
3
|
+
import inspect
|
|
3
4
|
|
|
4
5
|
# Internal imports
|
|
5
|
-
from . import
|
|
6
|
+
from . import lookup
|
|
6
7
|
|
|
7
8
|
def main_cli():
|
|
8
9
|
parser = argparse.ArgumentParser(description='Scripture lookup')
|
|
9
10
|
parser.add_argument('command', help='Command to run. Required.')
|
|
10
|
-
parser.add_argument('input', help='Input text to parse (one or more references).')
|
|
11
|
+
parser.add_argument('input', nargs='?', default='', help='Input text to parse (one or more references). Not needed for commands that don’t take input, such as refresh_metadata.')
|
|
11
12
|
parser.add_argument('--lang', help='Output language. Default: "en".')
|
|
12
13
|
parser.add_argument('--separator', help='Separator when there are multiple results. Default: "\n".')
|
|
13
14
|
parser.add_argument('--sort-by', help='Sort the returned references ("none", "traditional", or "label"). Default: "none".')
|
|
@@ -24,9 +25,16 @@ def main_cli():
|
|
|
24
25
|
|
|
25
26
|
args = parser.parse_args()
|
|
26
27
|
|
|
27
|
-
command = getattr(lookup, args.command)
|
|
28
|
-
|
|
29
|
-
args.
|
|
28
|
+
command = getattr(lookup, args.command, None)
|
|
29
|
+
if not callable(command) or args.command.startswith('_'):
|
|
30
|
+
parser.error(f'Unknown command: “{args.command}”. See README.md for the list of commands.')
|
|
31
|
+
|
|
32
|
+
# Some commands (refresh_metadata, get_langs, get_punctuation, get_numerals) don't take input text
|
|
33
|
+
command_takes_input = 'input_string' in inspect.signature(command).parameters
|
|
34
|
+
if command_takes_input and not args.input:
|
|
35
|
+
parser.error(f'The “{args.command}” command needs input text.')
|
|
36
|
+
|
|
37
|
+
options = dict(
|
|
30
38
|
lang = args.lang or 'en',
|
|
31
39
|
separator = args.separator or '\n',
|
|
32
40
|
sort_by = args.sort_by,
|
|
@@ -41,5 +49,6 @@ def main_cli():
|
|
|
41
49
|
skip_cleanup = args.skip_cleanup,
|
|
42
50
|
range_split_limit = args.range_split_limit or 1,
|
|
43
51
|
)
|
|
44
|
-
|
|
52
|
+
result = command(args.input, **options) if command_takes_input else command(**options)
|
|
53
|
+
|
|
45
54
|
print(result)
|
|
@@ -6,9 +6,8 @@ import time
|
|
|
6
6
|
import re
|
|
7
7
|
import unicodedata
|
|
8
8
|
|
|
9
|
-
# Third-party libraries
|
|
10
|
-
|
|
11
|
-
from bs4 import BeautifulSoup
|
|
9
|
+
# Third-party libraries are imported where they're used, rather than here, so that commands that
|
|
10
|
+
# never reach the network don't pay for loading them. Together they add about 0.12 seconds to startup.
|
|
12
11
|
|
|
13
12
|
|
|
14
13
|
data_directory = os.path.join(os.path.abspath(os.path.dirname(__file__)), 'data')
|
|
@@ -16,6 +15,8 @@ os.makedirs(data_directory, exist_ok = True)
|
|
|
16
15
|
|
|
17
16
|
# Download JSON data
|
|
18
17
|
def download_data(filename, filepath):
|
|
18
|
+
import requests
|
|
19
|
+
|
|
19
20
|
request_url = f'https://cdn.jsdelivr.net/gh/samuelbradshaw/python-scripture-scraper@main/sample/{filename}'
|
|
20
21
|
r = requests.get(request_url)
|
|
21
22
|
r.encoding = 'utf-8'
|
|
@@ -36,10 +37,12 @@ def load_data(filename):
|
|
|
36
37
|
download_data(filename, filepath)
|
|
37
38
|
return load_data(filename)
|
|
38
39
|
|
|
39
|
-
# Update JSON data
|
|
40
|
+
# Update JSON data. Returns the names of the files that were downloaded.
|
|
40
41
|
def update_data():
|
|
41
|
-
|
|
42
|
+
filenames = ('metadata-languages.min.json', 'metadata-scriptures.min.json',)
|
|
43
|
+
for filename in filenames:
|
|
42
44
|
download_data(filename, os.path.join(data_directory, filename))
|
|
45
|
+
return filenames
|
|
43
46
|
|
|
44
47
|
# Normalize text by removing anything that's not a letter or number, and converting to lowercase. This allows for a fuzzy comparison between input text and a known list of values.
|
|
45
48
|
def normalize_for_compare(text):
|
|
@@ -47,20 +50,79 @@ def normalize_for_compare(text):
|
|
|
47
50
|
normalized_text = ''.join([c for c in decomposed_text if unicodedata.category(c)[0] in ['L', 'N']]).lower()
|
|
48
51
|
return normalized_text
|
|
49
52
|
|
|
53
|
+
# Words that can appear in a scripture reference, by language. Longer words should come before shorter words that they start with, so that (for example) "chapters" is matched before "chapter".
|
|
54
|
+
reference_words = {
|
|
55
|
+
'en': {
|
|
56
|
+
'list_conjunctions': ['and', '&'],
|
|
57
|
+
'range_conjunctions': ['through', 'thru', 'to'],
|
|
58
|
+
'verse_words': ['verses', 'verse', 'vv.', 'v.'],
|
|
59
|
+
'chapter_words': ['chapters', 'chapter', 'chs.', 'ch.'],
|
|
60
|
+
},
|
|
61
|
+
'es': {
|
|
62
|
+
'list_conjunctions': ['y', 'e'],
|
|
63
|
+
'range_conjunctions': ['a', 'al'],
|
|
64
|
+
'verse_words': ['versículos', 'versículo'],
|
|
65
|
+
'chapter_words': ['capítulos', 'capítulo'],
|
|
66
|
+
},
|
|
67
|
+
'fr': {
|
|
68
|
+
'list_conjunctions': ['et'],
|
|
69
|
+
'range_conjunctions': ['à'],
|
|
70
|
+
'verse_words': ['versets', 'verset'],
|
|
71
|
+
'chapter_words': ['chapitres', 'chapitre'],
|
|
72
|
+
},
|
|
73
|
+
'pt': {
|
|
74
|
+
'list_conjunctions': ['e'],
|
|
75
|
+
'range_conjunctions': ['a'],
|
|
76
|
+
'verse_words': ['versículos', 'versículo'],
|
|
77
|
+
'chapter_words': ['capítulos', 'capítulo'],
|
|
78
|
+
},
|
|
79
|
+
}
|
|
80
|
+
reference_word_keys = ('list_conjunctions', 'range_conjunctions', 'verse_words', 'chapter_words',)
|
|
81
|
+
|
|
82
|
+
# List conjunctions from every language, split into the ones written as a word ("and", "y", "et") and the ones written as a symbol ("&")
|
|
83
|
+
all_list_conjunctions = sorted({c for words in reference_words.values() for c in words['list_conjunctions']})
|
|
84
|
+
list_conjunction_words = [c for c in all_list_conjunctions if any(char.isalpha() for char in c)]
|
|
85
|
+
list_conjunction_symbols = [c for c in all_list_conjunctions if not any(char.isalpha() for char in c)]
|
|
86
|
+
|
|
87
|
+
# Get alternate spellings of a name where a spelled-out list conjunction is swapped for a symbol, or the other way around. Example: "Doctrine and Covenants" –> "Doctrine & Covenants"
|
|
88
|
+
def get_conjunction_aliases(name):
|
|
89
|
+
aliases = set()
|
|
90
|
+
for word in list_conjunction_words:
|
|
91
|
+
for symbol in list_conjunction_symbols:
|
|
92
|
+
if f' {word} ' in name:
|
|
93
|
+
aliases.add(name.replace(f' {word} ', f' {symbol} '))
|
|
94
|
+
if f' {symbol} ' in name:
|
|
95
|
+
aliases.add(name.replace(f' {symbol} ', f' {word} '))
|
|
96
|
+
return aliases
|
|
97
|
+
|
|
98
|
+
|
|
50
99
|
languages = load_data('metadata-languages.min.json')
|
|
51
100
|
scriptures = load_data('metadata-scriptures.min.json')
|
|
52
101
|
|
|
53
|
-
|
|
54
|
-
for key, value in scriptures['mapToSlug'].items():
|
|
55
|
-
|
|
56
|
-
|
|
102
|
+
# Accept a symbol conjunction ("&") wherever a name spells the conjunction out, and vice versa. The words come from the language data below rather than being hard-coded, so this covers "Doctrine & Covenants" for "Doctrine and Covenants" and "Doctrina & Convenios" for "Doctrina y Convenios".
|
|
103
|
+
for key, value in list(scriptures['mapToSlug'].items()):
|
|
104
|
+
for alias in get_conjunction_aliases(key):
|
|
105
|
+
scriptures['mapToSlug'].setdefault(alias, value)
|
|
106
|
+
|
|
107
|
+
# Get a map of normalized names to book slugs, for comparing input that doesn't match a known name exactly. It's built the first time it's needed, rather than at import, since normalizing all ~10,000 names takes a moment and input that's already well-formed never needs it.
|
|
108
|
+
map_to_slug_normalized = None
|
|
109
|
+
def get_map_to_slug_normalized():
|
|
110
|
+
global map_to_slug_normalized
|
|
111
|
+
if map_to_slug_normalized is None:
|
|
112
|
+
map_to_slug_normalized = {normalize_for_compare(key): value for key, value in scriptures['mapToSlug'].items()}
|
|
113
|
+
return map_to_slug_normalized
|
|
57
114
|
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
|
|
115
|
+
|
|
116
|
+
# Get regex patterns for the words that can appear in a scripture reference in a given language. Languages that aren't listed above return empty patterns.
|
|
117
|
+
reference_words_cache = {}
|
|
118
|
+
def get_reference_words(lang):
|
|
119
|
+
if lang not in reference_words_cache:
|
|
120
|
+
words_for_lang = reference_words.get(lang, {})
|
|
121
|
+
reference_words_cache[lang] = {
|
|
122
|
+
key: r'|'.join([re.escape(w) for w in words_for_lang.get(key, [])])
|
|
123
|
+
for key in reference_word_keys
|
|
124
|
+
}
|
|
125
|
+
return reference_words_cache[lang]
|
|
64
126
|
|
|
65
127
|
|
|
66
128
|
# Get the BCP 47 language tag for a given language code
|
|
@@ -81,6 +143,9 @@ def get_bcp47(lang):
|
|
|
81
143
|
|
|
82
144
|
# Get the content for a given chapter verse from python-scripture-scraper or ChurchofJesusChrist.org
|
|
83
145
|
def request_content(publication_slug, book_slug, chapter, verse_groups, church_url, lang = 'en', source = 'python-scripture-scraper'):
|
|
146
|
+
import requests
|
|
147
|
+
from bs4 import BeautifulSoup
|
|
148
|
+
|
|
84
149
|
text_content = ''
|
|
85
150
|
|
|
86
151
|
if not publication_slug and book_slug and chapter:
|
|
@@ -6,10 +6,118 @@ import re
|
|
|
6
6
|
import icu
|
|
7
7
|
|
|
8
8
|
# Internal imports
|
|
9
|
-
from . import data, numbers
|
|
10
|
-
|
|
9
|
+
from . import data, numbers, patterns
|
|
11
10
|
|
|
11
|
+
# Caching dicts
|
|
12
12
|
natural_sort_collators = {}
|
|
13
|
+
book_name_cache = {}
|
|
14
|
+
detection_pattern_cache = {}
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
# Get scripture book names and abbreviations for a language, sorted longest first so that (for example) "1 John" is recognized before "John"
|
|
18
|
+
def get_book_names(lang):
|
|
19
|
+
if lang not in book_name_cache:
|
|
20
|
+
scripture_book_names = set()
|
|
21
|
+
for volume_data in data.scriptures['structure'].values():
|
|
22
|
+
for book_slug in volume_data['books'].keys():
|
|
23
|
+
book_info = data.scriptures['languages'][lang]['translatedNames'].get(book_slug)
|
|
24
|
+
if book_info:
|
|
25
|
+
book_name = patterns.whitespace.sub(' ', book_info.get('name') or '')
|
|
26
|
+
if book_name:
|
|
27
|
+
scripture_book_names.add(book_name)
|
|
28
|
+
book_abbrev = patterns.whitespace.sub(' ', book_info.get('abbrev') or '')
|
|
29
|
+
if book_abbrev:
|
|
30
|
+
scripture_book_names.add(book_abbrev)
|
|
31
|
+
scripture_book_names = sorted(scripture_book_names, key=lambda x: (-len(x), x))
|
|
32
|
+
|
|
33
|
+
# Book names with commas removed, so further normalization doesn't try to split one book name into two references. Example: "JST, Genesis 1" –> "JST Genesis 1"
|
|
34
|
+
scripture_book_names_without_commas = []
|
|
35
|
+
scripture_book_names_with_commas = []
|
|
36
|
+
for scripture_book_name in scripture_book_names:
|
|
37
|
+
scripture_book_name_without_comma = patterns.verse_group_separators.sub('', scripture_book_name)
|
|
38
|
+
scripture_book_names_without_commas.append(scripture_book_name_without_comma)
|
|
39
|
+
if patterns.verse_group_separators.search(scripture_book_name):
|
|
40
|
+
scripture_book_names_with_commas.append((scripture_book_name, scripture_book_name_without_comma,))
|
|
41
|
+
|
|
42
|
+
names_pattern = patterns.build_trie_pattern(scripture_book_names_without_commas)
|
|
43
|
+
book_name_cache[lang] = {
|
|
44
|
+
'names_with_commas': scripture_book_names_with_commas,
|
|
45
|
+
'separate_references_pattern': re.compile(rf'(?:^|[^\-])\b({names_pattern})', flags=re.IGNORECASE),
|
|
46
|
+
}
|
|
47
|
+
return book_name_cache[lang]
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
# Study helps aren't scripture references, and their abbreviations ("TG", "BD", "IT", "GS") are too short to tell apart from ordinary text
|
|
51
|
+
slugs_to_skip_when_detecting = ('topical-guide', 'bible-dictionary', 'index-to-the-triple-combination', 'guide-to-the-scriptures', 'joseph-smith-translation',)
|
|
52
|
+
|
|
53
|
+
# Build a regex alternation that matches anything that can start a scripture reference, allowing flexible whitespace and optional commas. Example: "JST, Genesis" –> "JST(?:,?)\sGenesis"
|
|
54
|
+
def build_book_names_pattern(lang):
|
|
55
|
+
# Every translated name is included, not just the books in "structure". That picks up publication names ("Doctrine and Covenants"), alternate forms ("Psalm" alongside "Psalms"), and things like "Official Declaration" and "Facsimile", which can all start a reference.
|
|
56
|
+
names = set()
|
|
57
|
+
for slug, translated_name in data.scriptures['languages'][lang]['translatedNames'].items():
|
|
58
|
+
if slug in slugs_to_skip_when_detecting:
|
|
59
|
+
continue
|
|
60
|
+
for key in ('name', 'abbrev',):
|
|
61
|
+
# Whitespace inside the name doesn't need to be normalized here – it's all converted to "\s" below
|
|
62
|
+
name = translated_name.get(key) or ''
|
|
63
|
+
if name:
|
|
64
|
+
names.add(name)
|
|
65
|
+
names.update(data.get_conjunction_aliases(name))
|
|
66
|
+
|
|
67
|
+
# A comma in a name is optional, so both spellings are listed. They have to be separate entries
|
|
68
|
+
# rather than an optional character, so that every path through the trie consumes the same
|
|
69
|
+
# characters – otherwise "TJS Génesis" and "TJS, Génesis 1–8" end up on branches that can both
|
|
70
|
+
# match the same text, and the shorter one wins.
|
|
71
|
+
for name in list(names):
|
|
72
|
+
name_without_comma = patterns.verse_group_separators.sub('', name)
|
|
73
|
+
if name_without_comma != name:
|
|
74
|
+
names.add(name_without_comma)
|
|
75
|
+
|
|
76
|
+
# Whitespace inside a name is matched loosely, so a non-breaking space in the data still matches a
|
|
77
|
+
# regular space in the text. That mapping is one character for one character, so the trie is safe.
|
|
78
|
+
normalized_names = [patterns.whitespace.sub(' ', name) for name in names]
|
|
79
|
+
return patterns.build_trie_pattern(normalized_names, character_patterns = {' ': r'\s'})
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
# Get a compiled regex for detecting scripture references embedded in text
|
|
83
|
+
def get_detection_pattern(lang):
|
|
84
|
+
if lang not in detection_pattern_cache:
|
|
85
|
+
words = data.get_reference_words(lang)
|
|
86
|
+
|
|
87
|
+
# Words that can be spelled out between two numbers, or between a book name and a number. Examples: "1, 2, and 3"; "Alma chapter 32 verse 21"
|
|
88
|
+
inline_words = r'|'.join([w for w in (words['list_conjunctions'], words['range_conjunctions'], words['verse_words'], words['chapter_words'],) if w])
|
|
89
|
+
leading_words = r'|'.join([w for w in (words['verse_words'], words['chapter_words'],) if w])
|
|
90
|
+
|
|
91
|
+
books = build_book_names_pattern(lang)
|
|
92
|
+
anchor = rf'(?P<book>{books})'
|
|
93
|
+
if words['chapter_words']:
|
|
94
|
+
anchor += rf'|(?P<chapter_word>{words["chapter_words"]})'
|
|
95
|
+
anchor_continued = rf'{books}|{words["chapter_words"]}' if words['chapter_words'] else books
|
|
96
|
+
|
|
97
|
+
# Separator between two numbers in a reference. A period is one of the possible chapter/verse separators, and it can't be followed by whitespace – without that restriction, "Alma 32. 5 people" would be read as "Alma 32:5". The other chapter/verse separators are unambiguous, so "Mosiah 28: 13" is fine.
|
|
98
|
+
chapter_verse_separators_allowing_space = r'|'.join([s for s in patterns.chapter_verse_separators_pattern.split(r'|') if s != re.escape('.')])
|
|
99
|
+
number_separator = rf'(?:{chapter_verse_separators_allowing_space})\s*|(?:{patterns.chapter_verse_separators_pattern})|\s*(?:{patterns.verse_group_separators_pattern}|{patterns.verse_range_separators_pattern})\s*'
|
|
100
|
+
if inline_words:
|
|
101
|
+
number_separator = rf'(?:{number_separator})(?:(?:{inline_words})\s+)?|\s+(?:{inline_words})\s+'
|
|
102
|
+
|
|
103
|
+
# Chapter and verses, ending on a digit or a closing parenthesis so that trailing whitespace and punctuation are never included in the match
|
|
104
|
+
leading_word = rf'(?:(?:{leading_words})\s+)?' if leading_words else ''
|
|
105
|
+
# A number that starts a book name belongs to the next reference, not to this one's verse list. Without this, "Alma 5 and 2 Ne. 2:25" would run together as "Alma 5 and 2".
|
|
106
|
+
not_the_next_book = rf'(?!(?:{books})\s*\d)'
|
|
107
|
+
|
|
108
|
+
# A chapter or verse is never more than three digits, so a longer run of digits – a year, a page number – isn't part of a reference. The lookahead rejects the whole number, rather than matching just its first three digits.
|
|
109
|
+
number = r'\d{1,3}(?!\d)'
|
|
110
|
+
|
|
111
|
+
context = rf'(?:\s*(?:{patterns.opening_parenthesis_pattern})\s*{number}(?:(?:{number_separator}){number})*\s*(?:{patterns.closing_parenthesis_pattern}))?'
|
|
112
|
+
tail = rf'{leading_word}{number}(?:(?:{number_separator}){not_the_next_book}{number})*{context}'
|
|
113
|
+
|
|
114
|
+
# Additional references that continue the same run. A conjunction can follow the separator, as in "Genesis 11:29; 22:23; and 24:15".
|
|
115
|
+
conjunction_after_separator = rf'(?:(?:{inline_words})\s+)?' if inline_words else ''
|
|
116
|
+
continued = rf'(?:\s*(?:{patterns.reference_separators_pattern})\s*{conjunction_after_separator}(?:(?:{anchor_continued})\s*)?{tail})*'
|
|
117
|
+
|
|
118
|
+
detection_pattern_cache[lang] = re.compile(rf'(?<![-\w])(?:{anchor})\s*{tail}{continued}', flags=re.IGNORECASE)
|
|
119
|
+
return detection_pattern_cache[lang]
|
|
120
|
+
|
|
13
121
|
|
|
14
122
|
class Reference:
|
|
15
123
|
def __init__(self, lang = 'en', publication_slug = None, book_slug = None, chapter = None, verse_groups = [], context_verse_groups = []):
|
|
@@ -61,13 +169,13 @@ class Reference:
|
|
|
61
169
|
|
|
62
170
|
# Get localized chapter name
|
|
63
171
|
def format_chapter_range(chapter_string):
|
|
64
|
-
groups =
|
|
172
|
+
groups = patterns.verse_group_separators.split(chapter_string)
|
|
65
173
|
new_groups = []
|
|
66
174
|
for group in groups:
|
|
67
|
-
range_parts =
|
|
175
|
+
range_parts = patterns.verse_range_separators.split(group)
|
|
68
176
|
new_range_parts = []
|
|
69
177
|
for range_part in range_parts:
|
|
70
|
-
chapter_verse_parts =
|
|
178
|
+
chapter_verse_parts = patterns.chapter_verse_separators.split(range_part)
|
|
71
179
|
new_chapter_verse_parts = []
|
|
72
180
|
for num in chapter_verse_parts:
|
|
73
181
|
new_chapter_verse_parts.append(numbers.get_formatted_number(num, target_lang = self.lang, target_custom_numerals = numerals))
|
|
@@ -109,9 +217,9 @@ class Reference:
|
|
|
109
217
|
|
|
110
218
|
if uri and self.chapter:
|
|
111
219
|
chapter = str(self.chapter)
|
|
112
|
-
if
|
|
220
|
+
if patterns.chapter_range.match(chapter):
|
|
113
221
|
# Chapter range – only use the first chapter
|
|
114
|
-
chapter =
|
|
222
|
+
chapter = patterns.verse_range_separators.split(chapter)[0]
|
|
115
223
|
uri += '/' + str(chapter)
|
|
116
224
|
if self.verse_groups:
|
|
117
225
|
if use_query_parameters:
|
|
@@ -174,8 +282,8 @@ def parse_verses_string(verses_string, lang = 'en', range_split_limit = 1):
|
|
|
174
282
|
|
|
175
283
|
unique_verses = set()
|
|
176
284
|
all_verses_are_integers = True
|
|
177
|
-
for verse_group_string in
|
|
178
|
-
verse_strings =
|
|
285
|
+
for verse_group_string in patterns.verse_group_separators_repeated.split(verses_string):
|
|
286
|
+
verse_strings = patterns.verse_range_separators.split(verse_group_string)
|
|
179
287
|
lower_int = numbers.convert_number_to_int(verse_strings[0])
|
|
180
288
|
upper_int = numbers.convert_number_to_int(verse_strings[-1])
|
|
181
289
|
|
|
@@ -257,66 +365,61 @@ def parse_references_string(input_string, lang = 'en', sort_by = None, skip_clea
|
|
|
257
365
|
input_string = input_string.strip().strip(punctuation_to_strip).rstrip(':').strip()
|
|
258
366
|
|
|
259
367
|
# If language is English, replace roman numerals with numbers. Example: 'II Corinthians" –> "2 Corinthians"
|
|
260
|
-
|
|
261
|
-
|
|
262
|
-
|
|
263
|
-
|
|
264
|
-
input_string = re.sub(r'\biv\s', '4', input_string, flags=re.IGNORECASE)
|
|
368
|
+
# The substitutions have to run in order, since replacing "i " with "1" can join it to the text that follows, so they're guarded by a single search rather than combined into one pass.
|
|
369
|
+
if lang == 'en' and patterns.any_roman_numeral.search(input_string):
|
|
370
|
+
for roman_numeral_pattern, arabic_numeral in patterns.roman_numerals:
|
|
371
|
+
input_string = roman_numeral_pattern.sub(arabic_numeral, input_string)
|
|
265
372
|
|
|
373
|
+
# Normalize whitespace so it matches the spaces in book names, but keep line breaks, since they separate one reference from the next
|
|
374
|
+
input_string = patterns.whitespace_except_line_breaks.sub(' ', input_string)
|
|
375
|
+
|
|
266
376
|
# Remove commas from book names so further normalization doesn't try to split it into two references. Example: "JST, Genesis 1" –> "JST Genesis 1"
|
|
267
|
-
|
|
268
|
-
|
|
269
|
-
|
|
270
|
-
for volume_data in data.scriptures['structure'].values():
|
|
271
|
-
for book_slug in volume_data['books'].keys():
|
|
272
|
-
book_info = data.scriptures['languages'][lang]['translatedNames'].get(book_slug)
|
|
273
|
-
if book_info:
|
|
274
|
-
book_name = (book_info.get('name') or '').replace('\xa0', ' ')
|
|
275
|
-
if book_name:
|
|
276
|
-
scripture_book_names.add(book_name)
|
|
277
|
-
book_abbrev = (book_info.get('abbrev') or '').replace('\xa0', ' ')
|
|
278
|
-
if book_abbrev:
|
|
279
|
-
scripture_book_names.add(book_abbrev)
|
|
280
|
-
scripture_book_names = sorted(scripture_book_names, key=lambda x: (-len(x), x))
|
|
281
|
-
for scripture_book_name in scripture_book_names:
|
|
282
|
-
scripture_book_name_without_comma = re.sub(data.verse_group_separators_pattern, '', scripture_book_name)
|
|
283
|
-
scripture_book_names_without_commas.append(scripture_book_name_without_comma)
|
|
284
|
-
if scripture_book_name in input_string and re.search(data.verse_group_separators_pattern, scripture_book_name):
|
|
377
|
+
book_names = get_book_names(lang)
|
|
378
|
+
for scripture_book_name, scripture_book_name_without_comma in book_names['names_with_commas']:
|
|
379
|
+
if scripture_book_name in input_string:
|
|
285
380
|
input_string = input_string.replace(scripture_book_name, scripture_book_name_without_comma)
|
|
286
|
-
|
|
381
|
+
|
|
287
382
|
# Normalize whitespace-separated references. Example: "Genesis 1:2 1 Nephi 3:7" –> "; Genesis 1:2 ; 1 Nephi 3:7"
|
|
288
|
-
|
|
289
|
-
|
|
290
|
-
|
|
383
|
+
input_string = book_names['separate_references_pattern'].sub(r'; \1', input_string)
|
|
384
|
+
|
|
385
|
+
normalization_patterns = patterns.get_normalization_patterns(lang)
|
|
386
|
+
|
|
291
387
|
# Normalize lists and ranges. Example: "Genesis 12:1, 2, and 3; verses 1 and 4; John 2 through 7" –> "Genesis 12:1, 2,,3; verses 1,4; John 2–7"
|
|
292
|
-
|
|
293
|
-
|
|
294
|
-
|
|
388
|
+
if normalization_patterns['list_conjunctions']:
|
|
389
|
+
input_string = normalization_patterns['list_conjunctions'].sub(r',\1', input_string)
|
|
390
|
+
if normalization_patterns['range_conjunctions']:
|
|
391
|
+
input_string = normalization_patterns['range_conjunctions'].sub(r'–\1', input_string)
|
|
392
|
+
|
|
295
393
|
# Normalize verse sets. Example: "chapter 3 verse 7; vv. 3, 6" –> "chapter 3:7; :3, 6"
|
|
296
|
-
|
|
297
|
-
|
|
394
|
+
if normalization_patterns['verse_words']:
|
|
395
|
+
input_string = normalization_patterns['verse_words'].sub(r':\1', input_string).replace('::', ':')
|
|
396
|
+
|
|
397
|
+
# Normalize chapter words. Example: "Alma chapter 32 verse 21" –> "Alma 32:21"
|
|
398
|
+
if normalization_patterns['chapter_words']:
|
|
399
|
+
input_string = normalization_patterns['chapter_words'].sub(r' \1', input_string)
|
|
400
|
+
|
|
298
401
|
# Normalize chapter sets. Example: "Genesis 1, 2, 4–5, Exodus 10; Alma 32" –> "Genesis 1; 2; 4–5; Exodus 10; Alma 32"
|
|
299
|
-
if
|
|
300
|
-
input_string =
|
|
402
|
+
if patterns.verse_group_separators.search(input_string) and not patterns.chapter_verse_separators.search(input_string):
|
|
403
|
+
input_string = patterns.verse_group_separators_repeated.sub(';', input_string)
|
|
301
404
|
|
|
302
405
|
# Normalize chapter:verse sets. Example: "Genesis 6:7a, 6:13a, 15; 1 Nephi 3:7 (twice), 8:21" –> "Genesis 6:7a; 6:13a, 15; 1 Nephi 3:7 (twice); 8:21"
|
|
303
|
-
if
|
|
304
|
-
references_list =
|
|
406
|
+
if patterns.chapter_verse_separators.search(input_string):
|
|
407
|
+
references_list = patterns.reference_separators.split(input_string)
|
|
305
408
|
new_references_list = []
|
|
306
409
|
for reference in references_list:
|
|
307
|
-
reference_parts =
|
|
410
|
+
reference_parts = patterns.verse_group_separators.split(reference)
|
|
308
411
|
reference_input_string = ''
|
|
309
412
|
for part in reference_parts:
|
|
310
413
|
if reference_input_string == '':
|
|
311
414
|
reference_input_string += part
|
|
312
|
-
elif
|
|
415
|
+
elif patterns.chapter_verse_separators.search(part):
|
|
313
416
|
reference_input_string += ';' + part
|
|
314
417
|
else:
|
|
315
418
|
reference_input_string += ',' + part
|
|
316
419
|
new_references_list.append(reference_input_string)
|
|
317
420
|
input_string = ';'.join(new_references_list)
|
|
318
421
|
|
|
319
|
-
input_list =
|
|
422
|
+
input_list = patterns.reference_separators.split(input_string)
|
|
320
423
|
|
|
321
424
|
references = []
|
|
322
425
|
previous_book_slug = None
|
|
@@ -329,7 +432,7 @@ def parse_references_string(input_string, lang = 'en', sort_by = None, skip_clea
|
|
|
329
432
|
|
|
330
433
|
if not skip_cleanup:
|
|
331
434
|
# Remove trailing text. Example: "1 John 3:2 2" –> "1 John 3:2"
|
|
332
|
-
trailing_text_match =
|
|
435
|
+
trailing_text_match = patterns.trailing_text.match(input_string)
|
|
333
436
|
if trailing_text_match:
|
|
334
437
|
trailing_text_string = trailing_text_match.group(1)
|
|
335
438
|
input_string = input_string.removesuffix(trailing_text_string)
|
|
@@ -376,8 +479,14 @@ def parse_references_string(input_string, lang = 'en', sort_by = None, skip_clea
|
|
|
376
479
|
# Scripture reference or slug
|
|
377
480
|
# Examples: Old Testament; 1 Nephi; Matthew 1; Helaman 5:12; words-of-mormon
|
|
378
481
|
|
|
379
|
-
# Get verses
|
|
380
|
-
|
|
482
|
+
# Get context verses from a trailing parenthetical, so it doesn't interfere with parsing the rest of the reference. Example: "Gen. 1:3 (3–4)"
|
|
483
|
+
context_match = patterns.trailing_parenthetical.match(input_string)
|
|
484
|
+
if context_match and patterns.numbers_and_separators.match(context_match.group(2)):
|
|
485
|
+
input_string = context_match.group(1)
|
|
486
|
+
context_verses_string = context_match.group(2)
|
|
487
|
+
|
|
488
|
+
# Get verses string. The separator has to sit between two digits, so that the period in an abbreviation like "Gen." isn't mistaken for a chapter/verse separator.
|
|
489
|
+
parts = patterns.chapter_verse_separator_between_digits.split(input_string)
|
|
381
490
|
if len(parts) == 2:
|
|
382
491
|
# Regular chapter and verse found
|
|
383
492
|
unparsed, verses_string = parts
|
|
@@ -385,11 +494,13 @@ def parse_references_string(input_string, lang = 'en', sort_by = None, skip_clea
|
|
|
385
494
|
# Chapter only, or special case like 'Genesis 7:17–8:9' or 'Genesis 1–5'
|
|
386
495
|
unparsed = input_string
|
|
387
496
|
verses_string = ''
|
|
388
|
-
verses_string,
|
|
389
|
-
|
|
497
|
+
verses_string, remaining_context_verses_string = (patterns.opening_parenthesis.split(patterns.closing_parenthesis.sub('', verses_string)) + [''])[:2]
|
|
498
|
+
if context_verses_string is None:
|
|
499
|
+
context_verses_string = remaining_context_verses_string
|
|
500
|
+
|
|
390
501
|
# Get chapter string and book string
|
|
391
502
|
book_string = unparsed
|
|
392
|
-
chapter_match =
|
|
503
|
+
chapter_match = patterns.trailing_chapter.match(book_string)
|
|
393
504
|
if chapter_match:
|
|
394
505
|
chapter_string = chapter_match.group(1)
|
|
395
506
|
book_string = book_string.removesuffix(chapter_string).strip()
|
|
@@ -400,7 +511,7 @@ def parse_references_string(input_string, lang = 'en', sort_by = None, skip_clea
|
|
|
400
511
|
book_slug = None
|
|
401
512
|
skip_book_name = False
|
|
402
513
|
if book_string:
|
|
403
|
-
book_slug = data.scriptures['mapToSlug'].get(book_string, None) or data.
|
|
514
|
+
book_slug = data.scriptures['mapToSlug'].get(book_string, None) or data.get_map_to_slug_normalized().get(data.normalize_for_compare(book_string), None)
|
|
404
515
|
# Special handling for Abraham facsimiles
|
|
405
516
|
if book_slug == 'facsimiles' or (not book_slug and 'fac' in book_string.lower()):
|
|
406
517
|
if previous_book_slug == 'abraham' and not previous_chapter:
|
|
@@ -456,7 +567,7 @@ def get_label(input_string, lang = 'en', separator = '\n', sort_by = None, skip_
|
|
|
456
567
|
references = parse_references_string(input_string, lang = lang, sort_by = sort_by, skip_cleanup = skip_cleanup, range_split_limit = range_split_limit)
|
|
457
568
|
return separator.join([ref.label(skip_book_name = skip_book_name, abbreviated = abbreviated) for ref in references])
|
|
458
569
|
|
|
459
|
-
def get_church_uri(input_string, separator = '\n', sort_by = None, use_query_parameters = False, skip_cleanup = False, range_split_limit = 1, **kwargs):
|
|
570
|
+
def get_church_uri(input_string, lang = 'en', separator = '\n', sort_by = None, use_query_parameters = False, skip_cleanup = False, range_split_limit = 1, **kwargs):
|
|
460
571
|
references = parse_references_string(input_string, lang = lang, sort_by = sort_by, skip_cleanup = skip_cleanup, range_split_limit = range_split_limit)
|
|
461
572
|
return separator.join([ref.church_uri(use_query_parameters = use_query_parameters) for ref in references])
|
|
462
573
|
|
|
@@ -468,6 +579,28 @@ def get_church_link(input_string, lang = 'en', separator = '\n', sort_by = None,
|
|
|
468
579
|
references = parse_references_string(input_string, lang = lang, sort_by = sort_by, skip_cleanup = skip_cleanup, range_split_limit = range_split_limit)
|
|
469
580
|
return separator.join([ref.church_link(link_class = link_class, link_target = link_target, skip_book_name = skip_book_name, abbreviated = abbreviated, skip_lang = skip_lang, skip_fragment = skip_fragment) for ref in references])
|
|
470
581
|
|
|
582
|
+
def detect_references(input_string, lang = 'en', range_split_limit = 1, **kwargs):
|
|
583
|
+
# Every reference includes a chapter number, so text with no digits in it can be skipped without
|
|
584
|
+
# building the language's detection pattern or scanning for book names.
|
|
585
|
+
if not input_string or not patterns.any_digit.search(input_string):
|
|
586
|
+
return []
|
|
587
|
+
|
|
588
|
+
lang = data.get_bcp47(lang)
|
|
589
|
+
|
|
590
|
+
# Replace newlines and other whitespace with regular spaces, one character for one character, so offsets stay valid in the original string
|
|
591
|
+
working_string = patterns.whitespace.sub(' ', input_string)
|
|
592
|
+
|
|
593
|
+
detections = []
|
|
594
|
+
for match in get_detection_pattern(lang).finditer(working_string):
|
|
595
|
+
references = parse_references_string(match.group(0), lang = lang, range_split_limit = range_split_limit)
|
|
596
|
+
if match.group('book') and not any(ref.book_slug or ref.publication_slug for ref in references):
|
|
597
|
+
# Looked like a book name, but it didn't resolve to a known book
|
|
598
|
+
continue
|
|
599
|
+
if not any(ref.chapter for ref in references):
|
|
600
|
+
continue
|
|
601
|
+
detections.append([input_string[match.start():match.end()], match.start(), match.end()])
|
|
602
|
+
return detections
|
|
603
|
+
|
|
471
604
|
def get_reference_objects(input_string, lang = 'en', sort_by = None, skip_cleanup = False, range_split_limit = 1, **kwargs):
|
|
472
605
|
return parse_references_string(input_string, lang = lang, sort_by = sort_by, skip_cleanup = skip_cleanup, range_split_limit = range_split_limit)
|
|
473
606
|
|
|
@@ -475,6 +608,11 @@ def get_reference_attributes(input_string, lang = 'en', sort_by = None, skip_cle
|
|
|
475
608
|
references = parse_references_string(input_string, lang = lang, sort_by = sort_by, skip_cleanup = skip_cleanup, range_split_limit = range_split_limit)
|
|
476
609
|
return [ref.attributes() for ref in references]
|
|
477
610
|
|
|
611
|
+
# Download the latest scripture and language metadata. Metadata that's already been loaded stays in memory, so the new data is used from the next run onward.
|
|
612
|
+
def refresh_metadata(**kwargs):
|
|
613
|
+
filenames = data.update_data()
|
|
614
|
+
return 'Updated metadata: ' + ', '.join(filenames)
|
|
615
|
+
|
|
478
616
|
def get_langs(**kwargs):
|
|
479
617
|
return data.scriptures['languages'].keys()
|
|
480
618
|
|
|
@@ -0,0 +1,132 @@
|
|
|
1
|
+
# Regex patterns used for parsing and detecting scripture references.
|
|
2
|
+
#
|
|
3
|
+
# Patterns ending in "_pattern" are strings, since they get interpolated into larger patterns.
|
|
4
|
+
# Everything else is compiled once here, rather than being rebuilt every time a reference is parsed.
|
|
5
|
+
|
|
6
|
+
# Python standard libraries
|
|
7
|
+
import re
|
|
8
|
+
|
|
9
|
+
# Internal imports
|
|
10
|
+
from . import data
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
# Separators and parentheses, gathered from the punctuation used across all languages
|
|
14
|
+
reference_separators_pattern = r'|'.join([re.escape(s.strip()) for s in data.scriptures['summary']['punctuation']['referenceSeparator']] + [re.escape(';'), re.escape('|'), re.escape('•'), re.escape('\n')])
|
|
15
|
+
chapter_verse_separators_pattern = r'|'.join([re.escape(s.strip()) for s in data.scriptures['summary']['punctuation']['chapterVerseSeparator']] + [re.escape(':'), re.escape('.')])
|
|
16
|
+
verse_group_separators_pattern = r'|'.join([re.escape(s.strip()) for s in data.scriptures['summary']['punctuation']['verseGroupSeparator']] + [re.escape(',')])
|
|
17
|
+
verse_range_separators_pattern = r'|'.join([re.escape(s.strip()) for s in data.scriptures['summary']['punctuation']['verseRangeSeparator']] + [re.escape('-'), re.escape('–'), re.escape('〜'), re.escape('~')])
|
|
18
|
+
opening_parenthesis_pattern = r'|'.join([re.escape(s.strip()) for s in data.scriptures['summary']['punctuation']['openingParenthesis']] + [re.escape('(')])
|
|
19
|
+
closing_parenthesis_pattern = r'|'.join([re.escape(s.strip()) for s in data.scriptures['summary']['punctuation']['closingParenthesis']] + [re.escape(')')])
|
|
20
|
+
|
|
21
|
+
# Any Unicode whitespace character – newlines, tabs, non-breaking spaces, ideographic spaces, and so on
|
|
22
|
+
whitespace = re.compile(r'\s')
|
|
23
|
+
# The same, except line breaks, which separate one reference from the next
|
|
24
|
+
whitespace_except_line_breaks = re.compile(r'[^\S\n]')
|
|
25
|
+
|
|
26
|
+
# Separators between references, chapters, verses, and verse groups
|
|
27
|
+
reference_separators = re.compile(reference_separators_pattern)
|
|
28
|
+
chapter_verse_separators = re.compile(chapter_verse_separators_pattern)
|
|
29
|
+
verse_group_separators = re.compile(verse_group_separators_pattern)
|
|
30
|
+
verse_range_separators = re.compile(verse_range_separators_pattern)
|
|
31
|
+
verse_group_separators_repeated = re.compile(rf'(?:{verse_group_separators_pattern})+')
|
|
32
|
+
opening_parenthesis = re.compile(opening_parenthesis_pattern)
|
|
33
|
+
closing_parenthesis = re.compile(closing_parenthesis_pattern)
|
|
34
|
+
|
|
35
|
+
# A chapter/verse separator that sits between two digits, so the period in an abbreviation like "Gen." isn't mistaken for one
|
|
36
|
+
chapter_verse_separator_between_digits = re.compile(rf'(?<=\d)(?:{chapter_verse_separators_pattern})(?=\d)')
|
|
37
|
+
# A range of chapters. Example: "1–5"
|
|
38
|
+
chapter_range = re.compile(rf'\d+(?:{verse_range_separators_pattern})\d+')
|
|
39
|
+
|
|
40
|
+
# Anything that can follow the first digit of a chapter or verse: another digit, whitespace, or any separator
|
|
41
|
+
number_continuation_pattern = rf'\d|\s|{chapter_verse_separators_pattern}|{verse_range_separators_pattern}|{verse_group_separators_pattern}'
|
|
42
|
+
# A string made up entirely of numbers and separators. Example: "3, 5–7"
|
|
43
|
+
numbers_and_separators = re.compile(rf'^\d(?:{number_continuation_pattern})*$')
|
|
44
|
+
# The chapter and verses at the end of a reference, so the book name can be split off the front
|
|
45
|
+
trailing_chapter = re.compile(rf'^.*?(\d(?:{number_continuation_pattern})*)$')
|
|
46
|
+
# Text after the end of a reference. Example: "1 John 3:2 2" –> " 2"
|
|
47
|
+
trailing_text = re.compile(rf'^.*?\d((?:\:|{closing_parenthesis_pattern})?\s+[^{opening_parenthesis_pattern}|\s]+)$')
|
|
48
|
+
# A parenthetical at the end of a reference, which holds context verses. Example: "Gen. 1:3 (3–4)"
|
|
49
|
+
trailing_parenthetical = re.compile(rf'^(.*?)\s*(?:{opening_parenthesis_pattern})([^))]*)(?:{closing_parenthesis_pattern})$')
|
|
50
|
+
|
|
51
|
+
# English roman numeral prefixes, paired with the number they stand for. Example: "II Corinthians" –> "2 Corinthians"
|
|
52
|
+
roman_numerals = [
|
|
53
|
+
(re.compile(r'\bi\s', flags=re.IGNORECASE), '1'),
|
|
54
|
+
(re.compile(r'\bii\s', flags=re.IGNORECASE), '2'),
|
|
55
|
+
(re.compile(r'\biii\s', flags=re.IGNORECASE), '3'),
|
|
56
|
+
(re.compile(r'\biv\s', flags=re.IGNORECASE), '4'),
|
|
57
|
+
]
|
|
58
|
+
# Matches any of the prefixes above, so a string with no roman numeral at all – which is most of them – can skip the substitutions entirely
|
|
59
|
+
any_roman_numeral = re.compile(r'\b(?:i|ii|iii|iv)\s', flags=re.IGNORECASE)
|
|
60
|
+
|
|
61
|
+
# A single digit. Every reference needs a chapter number, so text with no digits anywhere can't contain one.
|
|
62
|
+
any_digit = re.compile(r'\d')
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
# Build a regex alternation that matches any of the given words, sharing common prefixes so the regex
|
|
66
|
+
# engine doesn't retry every word at every position. Example: ["Genesis", "Genes"] –> "Gene(?:sis|s)".
|
|
67
|
+
# Shared prefixes make this several times faster than a flat "Genesis|Genes" alternation, which the
|
|
68
|
+
# engine has to walk one branch at a time. Longer words still win over shorter ones that they start
|
|
69
|
+
# with, because the trailing group is greedy.
|
|
70
|
+
#
|
|
71
|
+
# Each character can be given its own sub-pattern, for names where a character shouldn't be matched
|
|
72
|
+
# literally – see character_patterns in build_book_names_pattern.
|
|
73
|
+
def build_trie_pattern(words, character_patterns = None, case_insensitive = True):
|
|
74
|
+
character_patterns = character_patterns or {}
|
|
75
|
+
|
|
76
|
+
# Characters that differ only by case have to share a trie node, since the pattern is meant to be
|
|
77
|
+
# compiled with re.IGNORECASE. Without this, "OP" and "Opisyal" would sit in sibling branches, and
|
|
78
|
+
# "OP" would win on the input "Opisyal" – a flat alternation avoids that by sorting longest first.
|
|
79
|
+
def trie_key(character):
|
|
80
|
+
lowercased = character.lower()
|
|
81
|
+
return lowercased if case_insensitive and len(lowercased) == 1 else character
|
|
82
|
+
|
|
83
|
+
trie = {}
|
|
84
|
+
for word in words:
|
|
85
|
+
node = trie
|
|
86
|
+
for character in word:
|
|
87
|
+
node = node.setdefault(trie_key(character), {})
|
|
88
|
+
node[''] = {}
|
|
89
|
+
|
|
90
|
+
def build_branch(node):
|
|
91
|
+
# A node with nothing but an end marker is the end of a word
|
|
92
|
+
if '' in node and len(node) == 1:
|
|
93
|
+
return None
|
|
94
|
+
|
|
95
|
+
branches = []
|
|
96
|
+
word_ends_here = False
|
|
97
|
+
for character, child_node in sorted(node.items()):
|
|
98
|
+
if character == '':
|
|
99
|
+
word_ends_here = True
|
|
100
|
+
continue
|
|
101
|
+
remainder = build_branch(child_node)
|
|
102
|
+
character_pattern = character_patterns.get(character) or re.escape(character)
|
|
103
|
+
branches.append(character_pattern + (f'(?:{remainder})' if remainder and len(remainder) > 1 else (remainder or '')))
|
|
104
|
+
|
|
105
|
+
branch_pattern = r'|'.join(branches)
|
|
106
|
+
return f'(?:{branch_pattern})?' if word_ends_here else branch_pattern
|
|
107
|
+
|
|
108
|
+
return build_branch(trie)
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
# Get compiled regexes for normalizing the words that can appear in a reference in a given language
|
|
112
|
+
normalization_pattern_cache = {}
|
|
113
|
+
def get_normalization_patterns(lang):
|
|
114
|
+
if lang not in normalization_pattern_cache:
|
|
115
|
+
words = data.get_reference_words(lang)
|
|
116
|
+
|
|
117
|
+
# A conjunction sits between two numbers, so it needs whitespace on both sides. A verse or chapter word can also start the string.
|
|
118
|
+
conjunction_template = r'\s+(?:{words})\s+(\d+)'
|
|
119
|
+
reference_word_template = r'(?:^|\s)(?:{words})\s+(\d+)'
|
|
120
|
+
templates = {
|
|
121
|
+
'list_conjunctions': conjunction_template,
|
|
122
|
+
'range_conjunctions': conjunction_template,
|
|
123
|
+
'verse_words': reference_word_template,
|
|
124
|
+
'chapter_words': reference_word_template,
|
|
125
|
+
}
|
|
126
|
+
|
|
127
|
+
normalization_patterns = {}
|
|
128
|
+
for key in data.reference_word_keys:
|
|
129
|
+
template = templates[key]
|
|
130
|
+
normalization_patterns[key] = re.compile(template.format(words = words[key]), flags=re.IGNORECASE) if words[key] else None
|
|
131
|
+
normalization_pattern_cache[lang] = normalization_patterns
|
|
132
|
+
return normalization_pattern_cache[lang]
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: scripturelookup
|
|
3
|
-
Version: 0.0
|
|
3
|
+
Version: 0.1.0
|
|
4
4
|
Summary: Python and command-line utility for converting scripture references between formats.
|
|
5
5
|
Author-email: Samuel Bradshaw <samuel.h.bradshaw@gmail.com>
|
|
6
6
|
License: MIT License
|
|
@@ -40,7 +40,7 @@ Dynamic: license-file
|
|
|
40
40
|
|
|
41
41
|
# Scripture Lookup
|
|
42
42
|
|
|
43
|
-
Scripture Lookup is a Python library and command-line tool for looking up scripture verses or chapters. It can also
|
|
43
|
+
Scripture Lookup is a Python library and command-line tool for looking up scripture verses or chapters. It can also detect scripture references in a string, and convert them between formats and languages. The tool uses scripture and language metadata from [Python Scripture Scraper](https://github.com/samuelbradshaw/python-scripture-scraper).
|
|
44
44
|
|
|
45
45
|
|
|
46
46
|
## Installation
|
|
@@ -79,6 +79,12 @@ https://www.churchofjesuschrist.org/study/scriptures/bofm/hel/5?id=p12&lang=eng#
|
|
|
79
79
|
|
|
80
80
|
% scripturelookup get_church_url "/scriptures/ot" --lang "es"
|
|
81
81
|
https://www.churchofjesuschrist.org/study/scriptures/ot?lang=spa
|
|
82
|
+
|
|
83
|
+
% scripturelookup detect_references "As Nephi taught in 1 Ne. 3:7, and again in 2 Nephi 2:25."
|
|
84
|
+
[['1 Ne. 3:7', 19, 28], ['2 Nephi 2:25', 43, 55]]
|
|
85
|
+
|
|
86
|
+
% scripturelookup refresh_metadata
|
|
87
|
+
Updated metadata: metadata-languages.min.json, metadata-scriptures.min.json
|
|
82
88
|
```
|
|
83
89
|
|
|
84
90
|
|
|
@@ -102,6 +108,12 @@ lookup.get_label('/scriptures/nt/1-jn/3.2-3', lang = 'cmn-Hant')
|
|
|
102
108
|
|
|
103
109
|
lookup.get_church_url('/scriptures/ot', lang = 'es')
|
|
104
110
|
# https://www.churchofjesuschrist.org/study/scriptures/ot?lang=spa
|
|
111
|
+
|
|
112
|
+
lookup.detect_references('As Nephi taught in 1 Ne. 3:7, and again in 2 Nephi 2:25.')
|
|
113
|
+
# [['1 Ne. 3:7', 19, 28], ['2 Nephi 2:25', 43, 55]]
|
|
114
|
+
|
|
115
|
+
lookup.refresh_metadata()
|
|
116
|
+
# Updated metadata: metadata-languages.min.json, metadata-scriptures.min.json
|
|
105
117
|
```
|
|
106
118
|
|
|
107
119
|
|
|
@@ -117,6 +129,8 @@ lookup.get_church_url('/scriptures/ot', lang = 'es')
|
|
|
117
129
|
- **get_reference_objects** – Get a list of references as objects.
|
|
118
130
|
- **get_reference_attributes** – Get a list of references as dictionaries.
|
|
119
131
|
- **sort_references** – Sort a list of references by label or in traditional book order.
|
|
132
|
+
- **detect_references** – Find scripture references embedded in a string of text.
|
|
133
|
+
- **refresh_metadata** – Download the latest scripture and language metadata. Takes no input.
|
|
120
134
|
|
|
121
135
|
### Inputs
|
|
122
136
|
|
|
@@ -145,6 +159,8 @@ Any of the following input types are supported. You can also provide several inp
|
|
|
145
159
|
- http://lds.org/scriptures/bofm/1-ne/3.7?lang=eng
|
|
146
160
|
- [gospellibrary://content/scriptures/nt/john/3.16?lang=eng#16](https://www.churchofjesuschrist.org/study/scriptures/nt/john/3?id=p16&lang=eng#p16)
|
|
147
161
|
|
|
162
|
+
`detect_references` takes an arbitrary text string and returns a list of `[label, start_offset, end_offset]`. The label is the exact detected text (not normalized). Start and end offset indicate the exclusive range of characters in the input string where the reference was found.
|
|
163
|
+
|
|
148
164
|
### Options
|
|
149
165
|
|
|
150
166
|
Several options are available. Some are only applicable to certain commands.
|
|
@@ -16,11 +16,15 @@ src/scripturelookup/command_line.py
|
|
|
16
16
|
src/scripturelookup/data.py
|
|
17
17
|
src/scripturelookup/lookup.py
|
|
18
18
|
src/scripturelookup/numbers.py
|
|
19
|
+
src/scripturelookup/patterns.py
|
|
19
20
|
src/scripturelookup.egg-info/PKG-INFO
|
|
20
21
|
src/scripturelookup.egg-info/SOURCES.txt
|
|
21
22
|
src/scripturelookup.egg-info/dependency_links.txt
|
|
22
23
|
src/scripturelookup.egg-info/entry_points.txt
|
|
23
24
|
src/scripturelookup.egg-info/requires.txt
|
|
25
|
+
src/scripturelookup.egg-info/scm_file_list.json
|
|
26
|
+
src/scripturelookup.egg-info/scm_version.json
|
|
24
27
|
src/scripturelookup.egg-info/top_level.txt
|
|
25
28
|
src/scripturelookup/data/metadata-languages.min.json
|
|
26
|
-
src/scripturelookup/data/metadata-scriptures.min.json
|
|
29
|
+
src/scripturelookup/data/metadata-scriptures.min.json
|
|
30
|
+
tests/test_detect.py
|
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
{
|
|
2
|
+
"files": [
|
|
3
|
+
".github/workflows/publish.yml",
|
|
4
|
+
".gitignore",
|
|
5
|
+
"LICENSE",
|
|
6
|
+
"README.md",
|
|
7
|
+
"pyproject.toml",
|
|
8
|
+
"requirements.txt",
|
|
9
|
+
"src/geezify-python-main/LICENSE",
|
|
10
|
+
"src/geezify-python-main/README.md",
|
|
11
|
+
"src/geezify-python-main/arabify.py",
|
|
12
|
+
"src/geezify-python-main/arabify_test.py",
|
|
13
|
+
"src/geezify-python-main/geezify.py",
|
|
14
|
+
"src/geezify-python-main/geezify_test.py",
|
|
15
|
+
"src/geezify-python-main/test_data.py",
|
|
16
|
+
"src/scripturelookup/__init__.py",
|
|
17
|
+
"src/scripturelookup/command_line.py",
|
|
18
|
+
"src/scripturelookup/data.py",
|
|
19
|
+
"src/scripturelookup/data/metadata-languages.min.json",
|
|
20
|
+
"src/scripturelookup/data/metadata-scriptures.min.json",
|
|
21
|
+
"src/scripturelookup/lookup.py",
|
|
22
|
+
"src/scripturelookup/numbers.py",
|
|
23
|
+
"src/scripturelookup/patterns.py",
|
|
24
|
+
"tests/test_detect.py"
|
|
25
|
+
]
|
|
26
|
+
}
|
|
@@ -0,0 +1,137 @@
|
|
|
1
|
+
# Tests for lookup.detect_references
|
|
2
|
+
#
|
|
3
|
+
# Run with:
|
|
4
|
+
# python3 -m pytest tests
|
|
5
|
+
|
|
6
|
+
import os
|
|
7
|
+
import sys
|
|
8
|
+
|
|
9
|
+
import pytest
|
|
10
|
+
|
|
11
|
+
sys.path.insert(0, os.path.join(os.path.abspath(os.path.dirname(__file__)), '..', 'src'))
|
|
12
|
+
from scripturelookup import lookup
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
# (input text, language, expected labels in order)
|
|
16
|
+
detection_cases = [
|
|
17
|
+
# Nothing to find
|
|
18
|
+
('', 'en', []),
|
|
19
|
+
('Nothing here at all.', 'en', []),
|
|
20
|
+
|
|
21
|
+
# A run of related references stays a single detection
|
|
22
|
+
('See 1 Nephi 3:7, 5:2; Alma 5 for details.', 'en', ['1 Nephi 3:7, 5:2; Alma 5']),
|
|
23
|
+
('Doctrine and Covenants 45:56-59 | 1 Nephi 22:24-28', 'en', ['Doctrine and Covenants 45:56-59 | 1 Nephi 22:24-28']),
|
|
24
|
+
('Doctrine & Covenants 45:56 • 1 Nephi 22:24', 'en', ['Doctrine & Covenants 45:56 • 1 Nephi 22:24']),
|
|
25
|
+
|
|
26
|
+
# "&" stands in for a spelled-out list conjunction in any language, not just English
|
|
27
|
+
('Vea Doctrina & Convenios 45:56 aquí.', 'es', ['Doctrina & Convenios 45:56']),
|
|
28
|
+
('Vea Doctrina y Convenios 45:56 aquí.', 'es', ['Doctrina y Convenios 45:56']),
|
|
29
|
+
|
|
30
|
+
# Abbreviations
|
|
31
|
+
('In 2 Ne. 2:25 we read.', 'en', ['2 Ne. 2:25']),
|
|
32
|
+
('Compare Matt. 5:3 today.', 'en', ['Matt. 5:3']),
|
|
33
|
+
|
|
34
|
+
# A number is required, so ordinary words that are also book names are skipped
|
|
35
|
+
('He found a job in Job 5.', 'en', ['Job 5']),
|
|
36
|
+
('Read the Book of Mormon.', 'en', []),
|
|
37
|
+
|
|
38
|
+
# Trailing whitespace and unbalanced parentheses are never part of a match
|
|
39
|
+
('Read Genesis 1-3 (see also Moses 2).', 'en', ['Genesis 1-3', 'Moses 2']),
|
|
40
|
+
('Gen. 1:3 (3-4) is nice.', 'en', ['Gen. 1:3 (3-4)']),
|
|
41
|
+
|
|
42
|
+
# A period is a chapter/verse separator in some languages, but only between digits
|
|
43
|
+
('Alma 32. 5 people came.', 'en', ['Alma 32']),
|
|
44
|
+
|
|
45
|
+
# A chapter or verse is never more than three digits, so years and page numbers aren't references.
|
|
46
|
+
# The whole number is rejected rather than being truncated to its first three digits.
|
|
47
|
+
('Psalm 119:176 is the last verse.', 'en', ['Psalm 119:176']),
|
|
48
|
+
('D&C 124:123-45 was given.', 'en', ['D&C 124:123-45']),
|
|
49
|
+
('Alma 1978 was a year.', 'en', []),
|
|
50
|
+
('Alma 12345 xyz', 'en', []),
|
|
51
|
+
('Alma 32:21 was quoted in 1978.', 'en', ['Alma 32:21']),
|
|
52
|
+
|
|
53
|
+
# Chapter words can stand in for a book name; verse words cannot
|
|
54
|
+
('See chapter 3 and Alma chapter 32 verse 21.', 'en', ['chapter 3', 'Alma chapter 32 verse 21']),
|
|
55
|
+
('See verses 3-5 below.', 'en', []),
|
|
56
|
+
|
|
57
|
+
# Bare chapter:verse with no preceding anchor is not a reference
|
|
58
|
+
('As in Alma 32. See also 3:7.', 'en', ['Alma 32']),
|
|
59
|
+
('Meet at 3:15 tomorrow.', 'en', []),
|
|
60
|
+
|
|
61
|
+
# Spelled-out lists and ranges
|
|
62
|
+
('Genesis 12:1, 2, and 3 plus John 2 through 7.', 'en', ['Genesis 12:1, 2, and 3', 'John 2 through 7']),
|
|
63
|
+
|
|
64
|
+
# A number that starts the next book name isn't pulled into this reference's verse list
|
|
65
|
+
('See 1 Nephi 3:7, 5:2; Alma 5 and 2 Ne. 2:25 in chapter 9.', 'en', ['1 Nephi 3:7, 5:2; Alma 5', '2 Ne. 2:25', 'chapter 9']),
|
|
66
|
+
('Alma 5, 2 Ne. 2:25 today.', 'en', ['Alma 5', '2 Ne. 2:25']),
|
|
67
|
+
('Read 1 Nephi 3 and 3 John 1:4.', 'en', ['1 Nephi 3', '3 John 1:4']),
|
|
68
|
+
|
|
69
|
+
# A conjunction can follow a reference separator
|
|
70
|
+
('mentioned in Genesis 11:29; 22:23; and 24:15 as well as in Abraham 2:2.', 'en', ['Genesis 11:29; 22:23; and 24:15', 'Abraham 2:2']),
|
|
71
|
+
|
|
72
|
+
# A colon may be followed by a space, but a period may not (a period is also a chapter/verse separator)
|
|
73
|
+
('Mosiah 28: 13-15 is next.', 'en', ['Mosiah 28: 13-15']),
|
|
74
|
+
('Genesis 1.3 and Abr. 3.22-23 here.', 'en', ['Genesis 1.3', 'Abr. 3.22-23']),
|
|
75
|
+
('It cost 1.50 today.', 'en', []),
|
|
76
|
+
|
|
77
|
+
# Names that aren't in the "structure" book list
|
|
78
|
+
('See Psalm 23 and Official Declaration 2 and Facsimile 2.', 'en', ['Psalm 23', 'Official Declaration 2', 'Facsimile 2']),
|
|
79
|
+
|
|
80
|
+
# Line-wrapped references are found, and the label keeps the original whitespace
|
|
81
|
+
('Alma\n32:21 was quoted.', 'en', ['Alma\n32:21']),
|
|
82
|
+
('Alma\r\n32:21 was quoted.', 'en', ['Alma\r\n32:21']),
|
|
83
|
+
('See Alma\n 32:21, 22 here.', 'en', ['Alma\n 32:21, 22']),
|
|
84
|
+
|
|
85
|
+
# Any kind of Unicode whitespace works, not just a regular space
|
|
86
|
+
('See 1 Samuel 3:1 now.', 'en', ['1 Samuel 3:1']),
|
|
87
|
+
('See 1\xa0Samuel 3:1 now.', 'en', ['1\xa0Samuel 3:1']),
|
|
88
|
+
('See Alma 32:21 now.', 'en', ['Alma 32:21']),
|
|
89
|
+
|
|
90
|
+
# Reference words are language-specific. German has no seeded word list, so "capítulo 3" isn't a
|
|
91
|
+
# reference there, but "Alma" is spelled the same in German and Spanish and is still detected.
|
|
92
|
+
('Consulte el capítulo 3 y Alma 32:21 aquí.', 'es', ['capítulo 3', 'Alma 32:21']),
|
|
93
|
+
('Consulte el capítulo 3 y Alma 32:21 aquí.', 'de', ['Alma 32:21']),
|
|
94
|
+
|
|
95
|
+
# Book names are language-specific too
|
|
96
|
+
('Siehe Alma 32:21 hier.', 'de', ['Alma 32:21']),
|
|
97
|
+
('Siehe Alma 32:21 hier.', 'ko', []),
|
|
98
|
+
]
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
@pytest.mark.parametrize('text, lang, expected_labels', detection_cases)
|
|
102
|
+
def test_detect_references(text, lang, expected_labels):
|
|
103
|
+
detections = lookup.detect_references(text, lang = lang)
|
|
104
|
+
assert [label for label, start, end in detections] == expected_labels
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
@pytest.mark.parametrize('text, lang, expected_labels', detection_cases)
|
|
108
|
+
def test_offsets_match_the_original_string(text, lang, expected_labels):
|
|
109
|
+
for label, start, end in lookup.detect_references(text, lang = lang):
|
|
110
|
+
assert text[start:end] == label
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
@pytest.mark.parametrize('text, lang, expected_labels', detection_cases)
|
|
114
|
+
def test_detections_do_not_overlap(text, lang, expected_labels):
|
|
115
|
+
previous_end = 0
|
|
116
|
+
for label, start, end in lookup.detect_references(text, lang = lang):
|
|
117
|
+
assert start >= previous_end
|
|
118
|
+
assert start < end
|
|
119
|
+
previous_end = end
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def test_detected_labels_can_be_parsed_back():
|
|
123
|
+
text = 'See 1 Nephi 3:7 and Mosiah 2:17 for details.'
|
|
124
|
+
labels = [lookup.get_label(label) for label, start, end in lookup.detect_references(text)]
|
|
125
|
+
assert labels == ['1\xa0Nephi\xa03:7', 'Mosiah\xa02:17']
|
|
126
|
+
|
|
127
|
+
|
|
128
|
+
# Parsing (not detection) treats a line break as a separator between references, so whitespace
|
|
129
|
+
# normalization there has to leave line breaks alone.
|
|
130
|
+
def test_line_breaks_still_separate_references_when_parsing():
|
|
131
|
+
assert lookup.get_label('Alma 5\nMosiah 2:17') == 'Alma\xa05\nMosiah\xa02:17'
|
|
132
|
+
assert lookup.get_label('Alma 5\r\nMosiah 2:17') == 'Alma\xa05\nMosiah\xa02:17'
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
@pytest.mark.parametrize('space', [' ', '\xa0', ' ', ' '])
|
|
136
|
+
def test_book_names_match_any_kind_of_space(space):
|
|
137
|
+
assert lookup.get_label(f'1{space}Samuel 3:1') == '1\xa0Samuel\xa03:1'
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{scripturelookup-0.0.7 → scripturelookup-0.1.0}/src/scripturelookup/data/metadata-languages.min.json
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{scripturelookup-0.0.7 → scripturelookup-0.1.0}/src/scripturelookup.egg-info/dependency_links.txt
RENAMED
|
File without changes
|
{scripturelookup-0.0.7 → scripturelookup-0.1.0}/src/scripturelookup.egg-info/entry_points.txt
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|