scripturelookup 0.0.7__tar.gz → 0.1.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {scripturelookup-0.0.7/src/scripturelookup.egg-info → scripturelookup-0.1.1}/PKG-INFO +18 -2
- {scripturelookup-0.0.7 → scripturelookup-0.1.1}/README.md +17 -1
- {scripturelookup-0.0.7 → scripturelookup-0.1.1}/src/scripturelookup/command_line.py +15 -6
- scripturelookup-0.1.1/src/scripturelookup/data.py +349 -0
- scripturelookup-0.1.1/src/scripturelookup/lookup.py +991 -0
- scripturelookup-0.1.1/src/scripturelookup/patterns.py +145 -0
- {scripturelookup-0.0.7 → scripturelookup-0.1.1/src/scripturelookup.egg-info}/PKG-INFO +18 -2
- {scripturelookup-0.0.7 → scripturelookup-0.1.1}/src/scripturelookup.egg-info/SOURCES.txt +5 -1
- scripturelookup-0.1.1/src/scripturelookup.egg-info/scm_file_list.json +26 -0
- scripturelookup-0.1.1/src/scripturelookup.egg-info/scm_version.json +8 -0
- scripturelookup-0.1.1/tests/test_detect.py +393 -0
- scripturelookup-0.0.7/src/scripturelookup/data.py +0 -137
- scripturelookup-0.0.7/src/scripturelookup/lookup.py +0 -537
- {scripturelookup-0.0.7 → scripturelookup-0.1.1}/.github/workflows/publish.yml +0 -0
- {scripturelookup-0.0.7 → scripturelookup-0.1.1}/.gitignore +0 -0
- {scripturelookup-0.0.7 → scripturelookup-0.1.1}/LICENSE +0 -0
- {scripturelookup-0.0.7 → scripturelookup-0.1.1}/pyproject.toml +0 -0
- {scripturelookup-0.0.7 → scripturelookup-0.1.1}/requirements.txt +0 -0
- {scripturelookup-0.0.7 → scripturelookup-0.1.1}/setup.cfg +0 -0
- {scripturelookup-0.0.7 → scripturelookup-0.1.1}/src/geezify-python-main/LICENSE +0 -0
- {scripturelookup-0.0.7 → scripturelookup-0.1.1}/src/geezify-python-main/README.md +0 -0
- {scripturelookup-0.0.7 → scripturelookup-0.1.1}/src/geezify-python-main/arabify.py +0 -0
- {scripturelookup-0.0.7 → scripturelookup-0.1.1}/src/geezify-python-main/arabify_test.py +0 -0
- {scripturelookup-0.0.7 → scripturelookup-0.1.1}/src/geezify-python-main/geezify.py +0 -0
- {scripturelookup-0.0.7 → scripturelookup-0.1.1}/src/geezify-python-main/geezify_test.py +0 -0
- {scripturelookup-0.0.7 → scripturelookup-0.1.1}/src/geezify-python-main/test_data.py +0 -0
- {scripturelookup-0.0.7 → scripturelookup-0.1.1}/src/scripturelookup/__init__.py +0 -0
- {scripturelookup-0.0.7 → scripturelookup-0.1.1}/src/scripturelookup/data/metadata-languages.min.json +0 -0
- {scripturelookup-0.0.7 → scripturelookup-0.1.1}/src/scripturelookup/data/metadata-scriptures.min.json +0 -0
- {scripturelookup-0.0.7 → scripturelookup-0.1.1}/src/scripturelookup/numbers.py +0 -0
- {scripturelookup-0.0.7 → scripturelookup-0.1.1}/src/scripturelookup.egg-info/dependency_links.txt +0 -0
- {scripturelookup-0.0.7 → scripturelookup-0.1.1}/src/scripturelookup.egg-info/entry_points.txt +0 -0
- {scripturelookup-0.0.7 → scripturelookup-0.1.1}/src/scripturelookup.egg-info/requires.txt +0 -0
- {scripturelookup-0.0.7 → scripturelookup-0.1.1}/src/scripturelookup.egg-info/top_level.txt +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: scripturelookup
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.1.1
|
|
4
4
|
Summary: Python and command-line utility for converting scripture references between formats.
|
|
5
5
|
Author-email: Samuel Bradshaw <samuel.h.bradshaw@gmail.com>
|
|
6
6
|
License: MIT License
|
|
@@ -40,7 +40,7 @@ Dynamic: license-file
|
|
|
40
40
|
|
|
41
41
|
# Scripture Lookup
|
|
42
42
|
|
|
43
|
-
Scripture Lookup is a Python library and command-line tool for looking up scripture verses or chapters. It can also
|
|
43
|
+
Scripture Lookup is a Python library and command-line tool for looking up scripture verses or chapters. It can also detect scripture references in a string, and convert them between formats and languages. The tool uses scripture and language metadata from [Python Scripture Scraper](https://github.com/samuelbradshaw/python-scripture-scraper).
|
|
44
44
|
|
|
45
45
|
|
|
46
46
|
## Installation
|
|
@@ -79,6 +79,12 @@ https://www.churchofjesuschrist.org/study/scriptures/bofm/hel/5?id=p12&lang=eng#
|
|
|
79
79
|
|
|
80
80
|
% scripturelookup get_church_url "/scriptures/ot" --lang "es"
|
|
81
81
|
https://www.churchofjesuschrist.org/study/scriptures/ot?lang=spa
|
|
82
|
+
|
|
83
|
+
% scripturelookup detect_references "As Nephi taught in 1 Ne. 3:7, and again in 2 Nephi 2:25."
|
|
84
|
+
[['1 Ne. 3:7', 19, 28], ['2 Nephi 2:25', 43, 55]]
|
|
85
|
+
|
|
86
|
+
% scripturelookup refresh_metadata
|
|
87
|
+
Updated metadata: metadata-languages.min.json, metadata-scriptures.min.json
|
|
82
88
|
```
|
|
83
89
|
|
|
84
90
|
|
|
@@ -102,6 +108,12 @@ lookup.get_label('/scriptures/nt/1-jn/3.2-3', lang = 'cmn-Hant')
|
|
|
102
108
|
|
|
103
109
|
lookup.get_church_url('/scriptures/ot', lang = 'es')
|
|
104
110
|
# https://www.churchofjesuschrist.org/study/scriptures/ot?lang=spa
|
|
111
|
+
|
|
112
|
+
lookup.detect_references('As Nephi taught in 1 Ne. 3:7, and again in 2 Nephi 2:25.')
|
|
113
|
+
# [['1 Ne. 3:7', 19, 28], ['2 Nephi 2:25', 43, 55]]
|
|
114
|
+
|
|
115
|
+
lookup.refresh_metadata()
|
|
116
|
+
# Updated metadata: metadata-languages.min.json, metadata-scriptures.min.json
|
|
105
117
|
```
|
|
106
118
|
|
|
107
119
|
|
|
@@ -117,6 +129,8 @@ lookup.get_church_url('/scriptures/ot', lang = 'es')
|
|
|
117
129
|
- **get_reference_objects** – Get a list of references as objects.
|
|
118
130
|
- **get_reference_attributes** – Get a list of references as dictionaries.
|
|
119
131
|
- **sort_references** – Sort a list of references by label or in traditional book order.
|
|
132
|
+
- **detect_references** – Find scripture references embedded in a string of text.
|
|
133
|
+
- **refresh_metadata** – Download the latest scripture and language metadata. Takes no input.
|
|
120
134
|
|
|
121
135
|
### Inputs
|
|
122
136
|
|
|
@@ -145,6 +159,8 @@ Any of the following input types are supported. You can also provide several inp
|
|
|
145
159
|
- http://lds.org/scriptures/bofm/1-ne/3.7?lang=eng
|
|
146
160
|
- [gospellibrary://content/scriptures/nt/john/3.16?lang=eng#16](https://www.churchofjesuschrist.org/study/scriptures/nt/john/3?id=p16&lang=eng#p16)
|
|
147
161
|
|
|
162
|
+
`detect_references` takes an arbitrary text string and returns a list of `[label, start_offset, end_offset]`. The label is the exact detected text (not normalized). Start and end offset indicate the exclusive range of characters in the input string where the reference was found.
|
|
163
|
+
|
|
148
164
|
### Options
|
|
149
165
|
|
|
150
166
|
Several options are available. Some are only applicable to certain commands.
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
# Scripture Lookup
|
|
2
2
|
|
|
3
|
-
Scripture Lookup is a Python library and command-line tool for looking up scripture verses or chapters. It can also
|
|
3
|
+
Scripture Lookup is a Python library and command-line tool for looking up scripture verses or chapters. It can also detect scripture references in a string, and convert them between formats and languages. The tool uses scripture and language metadata from [Python Scripture Scraper](https://github.com/samuelbradshaw/python-scripture-scraper).
|
|
4
4
|
|
|
5
5
|
|
|
6
6
|
## Installation
|
|
@@ -39,6 +39,12 @@ https://www.churchofjesuschrist.org/study/scriptures/bofm/hel/5?id=p12&lang=eng#
|
|
|
39
39
|
|
|
40
40
|
% scripturelookup get_church_url "/scriptures/ot" --lang "es"
|
|
41
41
|
https://www.churchofjesuschrist.org/study/scriptures/ot?lang=spa
|
|
42
|
+
|
|
43
|
+
% scripturelookup detect_references "As Nephi taught in 1 Ne. 3:7, and again in 2 Nephi 2:25."
|
|
44
|
+
[['1 Ne. 3:7', 19, 28], ['2 Nephi 2:25', 43, 55]]
|
|
45
|
+
|
|
46
|
+
% scripturelookup refresh_metadata
|
|
47
|
+
Updated metadata: metadata-languages.min.json, metadata-scriptures.min.json
|
|
42
48
|
```
|
|
43
49
|
|
|
44
50
|
|
|
@@ -62,6 +68,12 @@ lookup.get_label('/scriptures/nt/1-jn/3.2-3', lang = 'cmn-Hant')
|
|
|
62
68
|
|
|
63
69
|
lookup.get_church_url('/scriptures/ot', lang = 'es')
|
|
64
70
|
# https://www.churchofjesuschrist.org/study/scriptures/ot?lang=spa
|
|
71
|
+
|
|
72
|
+
lookup.detect_references('As Nephi taught in 1 Ne. 3:7, and again in 2 Nephi 2:25.')
|
|
73
|
+
# [['1 Ne. 3:7', 19, 28], ['2 Nephi 2:25', 43, 55]]
|
|
74
|
+
|
|
75
|
+
lookup.refresh_metadata()
|
|
76
|
+
# Updated metadata: metadata-languages.min.json, metadata-scriptures.min.json
|
|
65
77
|
```
|
|
66
78
|
|
|
67
79
|
|
|
@@ -77,6 +89,8 @@ lookup.get_church_url('/scriptures/ot', lang = 'es')
|
|
|
77
89
|
- **get_reference_objects** – Get a list of references as objects.
|
|
78
90
|
- **get_reference_attributes** – Get a list of references as dictionaries.
|
|
79
91
|
- **sort_references** – Sort a list of references by label or in traditional book order.
|
|
92
|
+
- **detect_references** – Find scripture references embedded in a string of text.
|
|
93
|
+
- **refresh_metadata** – Download the latest scripture and language metadata. Takes no input.
|
|
80
94
|
|
|
81
95
|
### Inputs
|
|
82
96
|
|
|
@@ -105,6 +119,8 @@ Any of the following input types are supported. You can also provide several inp
|
|
|
105
119
|
- http://lds.org/scriptures/bofm/1-ne/3.7?lang=eng
|
|
106
120
|
- [gospellibrary://content/scriptures/nt/john/3.16?lang=eng#16](https://www.churchofjesuschrist.org/study/scriptures/nt/john/3?id=p16&lang=eng#p16)
|
|
107
121
|
|
|
122
|
+
`detect_references` takes an arbitrary text string and returns a list of `[label, start_offset, end_offset]`. The label is the exact detected text (not normalized). Start and end offset indicate the exclusive range of characters in the input string where the reference was found.
|
|
123
|
+
|
|
108
124
|
### Options
|
|
109
125
|
|
|
110
126
|
Several options are available. Some are only applicable to certain commands.
|
|
@@ -1,13 +1,14 @@
|
|
|
1
1
|
# Python standard libraries
|
|
2
2
|
import argparse
|
|
3
|
+
import inspect
|
|
3
4
|
|
|
4
5
|
# Internal imports
|
|
5
|
-
from . import
|
|
6
|
+
from . import lookup
|
|
6
7
|
|
|
7
8
|
def main_cli():
|
|
8
9
|
parser = argparse.ArgumentParser(description='Scripture lookup')
|
|
9
10
|
parser.add_argument('command', help='Command to run. Required.')
|
|
10
|
-
parser.add_argument('input', help='Input text to parse (one or more references).')
|
|
11
|
+
parser.add_argument('input', nargs='?', default='', help='Input text to parse (one or more references). Not needed for commands that don’t take input, such as refresh_metadata.')
|
|
11
12
|
parser.add_argument('--lang', help='Output language. Default: "en".')
|
|
12
13
|
parser.add_argument('--separator', help='Separator when there are multiple results. Default: "\n".')
|
|
13
14
|
parser.add_argument('--sort-by', help='Sort the returned references ("none", "traditional", or "label"). Default: "none".')
|
|
@@ -24,9 +25,16 @@ def main_cli():
|
|
|
24
25
|
|
|
25
26
|
args = parser.parse_args()
|
|
26
27
|
|
|
27
|
-
command = getattr(lookup, args.command)
|
|
28
|
-
|
|
29
|
-
args.
|
|
28
|
+
command = getattr(lookup, args.command, None)
|
|
29
|
+
if not callable(command) or args.command.startswith('_'):
|
|
30
|
+
parser.error(f'Unknown command: “{args.command}”. See README.md for the list of commands.')
|
|
31
|
+
|
|
32
|
+
# Some commands (refresh_metadata, get_langs, get_punctuation, get_numerals) don't take input text
|
|
33
|
+
command_takes_input = 'input_string' in inspect.signature(command).parameters
|
|
34
|
+
if command_takes_input and not args.input:
|
|
35
|
+
parser.error(f'The “{args.command}” command needs input text.')
|
|
36
|
+
|
|
37
|
+
options = dict(
|
|
30
38
|
lang = args.lang or 'en',
|
|
31
39
|
separator = args.separator or '\n',
|
|
32
40
|
sort_by = args.sort_by,
|
|
@@ -41,5 +49,6 @@ def main_cli():
|
|
|
41
49
|
skip_cleanup = args.skip_cleanup,
|
|
42
50
|
range_split_limit = args.range_split_limit or 1,
|
|
43
51
|
)
|
|
44
|
-
|
|
52
|
+
result = command(args.input, **options) if command_takes_input else command(**options)
|
|
53
|
+
|
|
45
54
|
print(result)
|
|
@@ -0,0 +1,349 @@
|
|
|
1
|
+
# Python standard libraries
|
|
2
|
+
import os
|
|
3
|
+
import sys
|
|
4
|
+
import json
|
|
5
|
+
import time
|
|
6
|
+
import re
|
|
7
|
+
import unicodedata
|
|
8
|
+
|
|
9
|
+
# Third-party libraries are imported where they're used, rather than here, so that commands that
|
|
10
|
+
# never reach the network don't pay for loading them. Together they add about 0.12 seconds to startup.
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
data_directory = os.path.join(os.path.abspath(os.path.dirname(__file__)), 'data')
|
|
14
|
+
os.makedirs(data_directory, exist_ok = True)
|
|
15
|
+
|
|
16
|
+
# Download JSON data
|
|
17
|
+
def download_data(filename, filepath):
|
|
18
|
+
import requests
|
|
19
|
+
|
|
20
|
+
request_url = f'https://cdn.jsdelivr.net/gh/samuelbradshaw/python-scripture-scraper@main/sample/{filename}'
|
|
21
|
+
r = requests.get(request_url)
|
|
22
|
+
r.encoding = 'utf-8'
|
|
23
|
+
if r and r.status_code == 200:
|
|
24
|
+
data = r.content
|
|
25
|
+
with open(filepath, 'wb') as f:
|
|
26
|
+
f.write(data)
|
|
27
|
+
else:
|
|
28
|
+
sys.exit('\nError: Couldn’t download JSON data:\n{request_url}\n')
|
|
29
|
+
|
|
30
|
+
# Load JSON data
|
|
31
|
+
def load_data(filename):
|
|
32
|
+
filepath = os.path.join(data_directory, filename)
|
|
33
|
+
if os.path.isfile(filepath):
|
|
34
|
+
with open(filepath, 'r', encoding='utf-8') as f:
|
|
35
|
+
return json.load(f)
|
|
36
|
+
else:
|
|
37
|
+
download_data(filename, filepath)
|
|
38
|
+
return load_data(filename)
|
|
39
|
+
|
|
40
|
+
# Update JSON data. Returns the names of the files that were downloaded.
|
|
41
|
+
def update_data():
|
|
42
|
+
filenames = ('metadata-languages.min.json', 'metadata-scriptures.min.json',)
|
|
43
|
+
for filename in filenames:
|
|
44
|
+
download_data(filename, os.path.join(data_directory, filename))
|
|
45
|
+
return filenames
|
|
46
|
+
|
|
47
|
+
# Normalize text by removing anything that's not a letter or number, and converting to lowercase. This allows for a fuzzy comparison between input text and a known list of values.
|
|
48
|
+
def normalize_for_compare(text):
|
|
49
|
+
decomposed_text = unicodedata.normalize('NFKD', text)
|
|
50
|
+
normalized_text = ''.join([c for c in decomposed_text if unicodedata.category(c)[0] in ['L', 'N']]).lower()
|
|
51
|
+
return normalized_text
|
|
52
|
+
|
|
53
|
+
# Words that can appear in a scripture reference, by language. Longer words should come before shorter words that they start with, so that (for example) "chapters" is matched before "chapter".
|
|
54
|
+
reference_words = {
|
|
55
|
+
'en': {
|
|
56
|
+
'list_conjunctions': ['and', '&'],
|
|
57
|
+
'range_conjunctions': ['through', 'thru', 'to'],
|
|
58
|
+
'verse_words': ['verses', 'verse', 'vv.', 'v.'],
|
|
59
|
+
'chapter_words': ['chapters', 'chapter', 'chs.', 'ch.'],
|
|
60
|
+
},
|
|
61
|
+
'es': {
|
|
62
|
+
'list_conjunctions': ['y', 'e'],
|
|
63
|
+
'range_conjunctions': ['a', 'al'],
|
|
64
|
+
'verse_words': ['versículos', 'versículo'],
|
|
65
|
+
'chapter_words': ['capítulos', 'capítulo'],
|
|
66
|
+
},
|
|
67
|
+
'fr': {
|
|
68
|
+
'list_conjunctions': ['et'],
|
|
69
|
+
'range_conjunctions': ['à'],
|
|
70
|
+
'verse_words': ['versets', 'verset'],
|
|
71
|
+
'chapter_words': ['chapitres', 'chapitre'],
|
|
72
|
+
},
|
|
73
|
+
'pt': {
|
|
74
|
+
'list_conjunctions': ['e'],
|
|
75
|
+
'range_conjunctions': ['a'],
|
|
76
|
+
'verse_words': ['versículos', 'versículo'],
|
|
77
|
+
'chapter_words': ['capítulos', 'capítulo'],
|
|
78
|
+
},
|
|
79
|
+
}
|
|
80
|
+
reference_word_keys = ('list_conjunctions', 'range_conjunctions', 'verse_words', 'chapter_words',)
|
|
81
|
+
|
|
82
|
+
# Ordinal forms 1–13, for citing an article of faith by position ("First Article of Faith"), and – in the
|
|
83
|
+
# first four entries – for a numbered book cited as "First Nephi" or "1st Nephi". A form's position in the
|
|
84
|
+
# list is the number it stands for, so there are no numbers written out separately to keep in sync.
|
|
85
|
+
# Gendered forms appear where the noun takes them: Portuguese cites "Regras de Fé", so an article there is
|
|
86
|
+
# "Primeira regra de fé", while Spanish cites "Artículos de Fe" and so takes "Primer artículo de fe".
|
|
87
|
+
ordinal_words = {
|
|
88
|
+
'en': [
|
|
89
|
+
('first', '1st',), ('second', '2nd',), ('third', '3rd',), ('fourth', '4th',),
|
|
90
|
+
('fifth', '5th',), ('sixth', '6th',), ('seventh', '7th',), ('eighth', '8th',),
|
|
91
|
+
('ninth', '9th',), ('tenth', '10th',), ('eleventh', '11th',), ('twelfth', '12th',),
|
|
92
|
+
('thirteenth', '13th',),
|
|
93
|
+
],
|
|
94
|
+
'fr': [
|
|
95
|
+
('premier', 'première', '1er', '1re', '1ère',),
|
|
96
|
+
('deuxième', 'second', 'seconde', '2e', '2ème', '2d', '2de',),
|
|
97
|
+
('troisième', '3e', '3ème',), ('quatrième', '4e', '4ème',), ('cinquième', '5e', '5ème',),
|
|
98
|
+
('sixième', '6e', '6ème',), ('septième', '7e', '7ème',), ('huitième', '8e', '8ème',),
|
|
99
|
+
('neuvième', '9e', '9ème',), ('dixième', '10e', '10ème',), ('onzième', '11e', '11ème',),
|
|
100
|
+
('douzième', '12e', '12ème',), ('treizième', '13e', '13ème',),
|
|
101
|
+
],
|
|
102
|
+
# Spanish has two accepted systems for 11 and 12 – the classical "undécimo" and the modern
|
|
103
|
+
# "decimoprimero" – and each ordinal has a masculine, a feminine, and (for 1, 3 and their compounds) an
|
|
104
|
+
# apocopated form used before a masculine noun. Every spelling is listed, so a reference resolves whichever
|
|
105
|
+
# the writer reached for, and whether or not the gender agrees with the name being cited.
|
|
106
|
+
'es': [
|
|
107
|
+
('primero', 'primer', 'primera', '1º', '1er', '1ª',),
|
|
108
|
+
('segundo', 'segunda', '2º', '2ª',),
|
|
109
|
+
('tercero', 'tercer', 'tercera', '3º', '3er', '3ª',),
|
|
110
|
+
('cuarto', 'cuarta', '4º', '4ª',),
|
|
111
|
+
('quinto', 'quinta', '5º', '5ª',),
|
|
112
|
+
('sexto', 'sexta', '6º', '6ª',),
|
|
113
|
+
('séptimo', 'séptima', '7º', '7ª',),
|
|
114
|
+
('octavo', 'octava', '8º', '8ª',),
|
|
115
|
+
('noveno', 'novena', '9º', '9ª',),
|
|
116
|
+
('décimo', 'décima', '10º', '10ª',),
|
|
117
|
+
('undécimo', 'undécima', 'decimoprimero', 'decimoprimera', 'decimoprimer', 'décimo primero', 'décima primera', 'décimo primer', '11º', '11ª',),
|
|
118
|
+
('duodécimo', 'duodécima', 'decimosegundo', 'decimosegunda', 'décimo segundo', 'décima segunda', '12º', '12ª',),
|
|
119
|
+
('decimotercero', 'decimotercera', 'decimotercer', 'décimo tercero', 'décima tercera', 'décimo tercer', '13º', '13ª',),
|
|
120
|
+
],
|
|
121
|
+
# Portuguese cites "Regras de Fé", which is feminine, so the feminine forms are the ones that occur – but
|
|
122
|
+
# the masculine are listed too, along with the classical "undécimo" and "duodécimo" alongside the ordinary
|
|
123
|
+
# two-word spellings.
|
|
124
|
+
'pt': [
|
|
125
|
+
('primeiro', 'primeira', '1º', '1ª',), ('segundo', 'segunda', '2º', '2ª',),
|
|
126
|
+
('terceiro', 'terceira', '3º', '3ª',), ('quarto', 'quarta', '4º', '4ª',),
|
|
127
|
+
('quinto', 'quinta', '5º', '5ª',), ('sexto', 'sexta', '6º', '6ª',),
|
|
128
|
+
('sétimo', 'sétima', '7º', '7ª',), ('oitavo', 'oitava', '8º', '8ª',),
|
|
129
|
+
('nono', 'nona', '9º', '9ª',), ('décimo', 'décima', '10º', '10ª',),
|
|
130
|
+
('décimo primeiro', 'décima primeira', 'undécimo', 'undécima', '11º', '11ª',),
|
|
131
|
+
('décimo segundo', 'décima segunda', 'duodécimo', 'duodécima', '12º', '12ª',),
|
|
132
|
+
('décimo terceiro', 'décima terceira', '13º', '13ª',),
|
|
133
|
+
],
|
|
134
|
+
}
|
|
135
|
+
|
|
136
|
+
# Roman numerals are typography rather than language, so they stand in for a book number in any language.
|
|
137
|
+
# Only 1–4 are needed, since no book is numbered higher.
|
|
138
|
+
roman_numeral_words = ['i', 'ii', 'iii', 'iv']
|
|
139
|
+
|
|
140
|
+
# The highest number any book name starts with ("4 Nephi"). patterns.leading_book_number is built from this.
|
|
141
|
+
highest_book_number = len(roman_numeral_words)
|
|
142
|
+
|
|
143
|
+
# Get a map of ordinal form to the number it stands for, for an article of faith cited by position
|
|
144
|
+
ordinal_forms_cache = {}
|
|
145
|
+
def get_ordinal_forms(lang):
|
|
146
|
+
if lang not in ordinal_forms_cache:
|
|
147
|
+
ordinal_forms_cache[lang] = {
|
|
148
|
+
form.lower(): index + 1
|
|
149
|
+
for index, group in enumerate(ordinal_words.get(lang, []))
|
|
150
|
+
for form in group
|
|
151
|
+
}
|
|
152
|
+
return ordinal_forms_cache[lang]
|
|
153
|
+
|
|
154
|
+
# Get a map of number form to the number it stands for, for the prefix on a numbered book name. It's the
|
|
155
|
+
# ordinals above, stopping at the highest number any book carries, plus the roman numerals that work in
|
|
156
|
+
# every language.
|
|
157
|
+
book_number_forms_cache = {}
|
|
158
|
+
def get_book_number_forms(lang):
|
|
159
|
+
if lang not in book_number_forms_cache:
|
|
160
|
+
forms = {form: number for form, number in get_ordinal_forms(lang).items() if number <= highest_book_number}
|
|
161
|
+
for index, form in enumerate(roman_numeral_words):
|
|
162
|
+
forms.setdefault(form, index + 1)
|
|
163
|
+
book_number_forms_cache[lang] = forms
|
|
164
|
+
return book_number_forms_cache[lang]
|
|
165
|
+
|
|
166
|
+
# List conjunctions from every language, split into the ones written as a word ("and", "y", "et") and the ones written as a symbol ("&")
|
|
167
|
+
all_list_conjunctions = sorted({c for words in reference_words.values() for c in words['list_conjunctions']})
|
|
168
|
+
list_conjunction_words = [c for c in all_list_conjunctions if any(char.isalpha() for char in c)]
|
|
169
|
+
list_conjunction_symbols = [c for c in all_list_conjunctions if not any(char.isalpha() for char in c)]
|
|
170
|
+
|
|
171
|
+
# Get alternate spellings of a name where a spelled-out list conjunction is swapped for a symbol, or the other way around. Example: "Doctrine and Covenants" –> "Doctrine & Covenants"
|
|
172
|
+
def get_conjunction_aliases(name):
|
|
173
|
+
aliases = set()
|
|
174
|
+
for word in list_conjunction_words:
|
|
175
|
+
for symbol in list_conjunction_symbols:
|
|
176
|
+
if f' {word} ' in name:
|
|
177
|
+
aliases.add(name.replace(f' {word} ', f' {symbol} '))
|
|
178
|
+
if f' {symbol} ' in name:
|
|
179
|
+
aliases.add(name.replace(f' {symbol} ', f' {word} '))
|
|
180
|
+
return aliases
|
|
181
|
+
|
|
182
|
+
|
|
183
|
+
languages = load_data('metadata-languages.min.json')
|
|
184
|
+
scriptures = load_data('metadata-scriptures.min.json')
|
|
185
|
+
|
|
186
|
+
# Accept a symbol conjunction ("&") wherever a name spells the conjunction out, and vice versa. The words come from the language data below rather than being hard-coded, so this covers "Doctrine & Covenants" for "Doctrine and Covenants" and "Doctrina & Convenios" for "Doctrina y Convenios".
|
|
187
|
+
for key, value in list(scriptures['mapToSlug'].items()):
|
|
188
|
+
for alias in get_conjunction_aliases(key):
|
|
189
|
+
scriptures['mapToSlug'].setdefault(alias, value)
|
|
190
|
+
|
|
191
|
+
# Get a map of normalized names to book slugs, for comparing input that doesn't match a known name exactly. It's built the first time it's needed, rather than at import, since normalizing all ~10,000 names takes a moment and input that's already well-formed never needs it.
|
|
192
|
+
map_to_slug_normalized = None
|
|
193
|
+
def get_map_to_slug_normalized():
|
|
194
|
+
global map_to_slug_normalized
|
|
195
|
+
if map_to_slug_normalized is None:
|
|
196
|
+
map_to_slug_normalized = {normalize_for_compare(key): value for key, value in scriptures['mapToSlug'].items()}
|
|
197
|
+
return map_to_slug_normalized
|
|
198
|
+
|
|
199
|
+
|
|
200
|
+
# The shortest normalized name that's worth matching loosely. Below this, a single typo is most of the word – "Alma" is one edit away from far too much ordinary text.
|
|
201
|
+
minimum_fuzzy_name_length = 5
|
|
202
|
+
|
|
203
|
+
# Every spelling of a name with one letter taken out. Example: "alma" –> ["lma", "ama", "ala", "alm"]
|
|
204
|
+
# Digits are never dropped, so a number always has to match exactly – it's part of which book is meant, not
|
|
205
|
+
# something that can be mistyped into another book. Without that, "5th Nephi" would resolve to "4th Nephi",
|
|
206
|
+
# which is a real name one substitution away.
|
|
207
|
+
def get_single_character_deletions(key):
|
|
208
|
+
return [key[:i] + key[i + 1:] for i in range(len(key)) if not key[i].isdigit()]
|
|
209
|
+
|
|
210
|
+
# Get a map of every single-character deletion of every known name to its book slug. Comparing deletions on
|
|
211
|
+
# both sides is what makes a one-character difference cheap to find: a missing letter, an extra letter, or a
|
|
212
|
+
# wrong letter all line up on some shared deletion, so a lookup costs one dict get per character instead of
|
|
213
|
+
# a comparison against all ~10,000 names. Names that two different slugs both claim are dropped, since an
|
|
214
|
+
# ambiguous match is worse than none. Built the first time a name fails to match exactly, since input that's
|
|
215
|
+
# spelled correctly never needs it.
|
|
216
|
+
map_to_slug_deletions = None
|
|
217
|
+
def get_map_to_slug_deletions():
|
|
218
|
+
global map_to_slug_deletions
|
|
219
|
+
if map_to_slug_deletions is None:
|
|
220
|
+
map_to_slug_deletions = {}
|
|
221
|
+
for key, value in scriptures['mapToSlug'].items():
|
|
222
|
+
normalized_key = normalize_for_compare(key)
|
|
223
|
+
if len(normalized_key) < minimum_fuzzy_name_length:
|
|
224
|
+
continue
|
|
225
|
+
for deletion in get_single_character_deletions(normalized_key):
|
|
226
|
+
map_to_slug_deletions[deletion] = value if map_to_slug_deletions.get(deletion, value) == value else None
|
|
227
|
+
return map_to_slug_deletions
|
|
228
|
+
|
|
229
|
+
# Look up a book slug for a normalized name that's off by one character. Diacritics are already handled by
|
|
230
|
+
# normalize_for_compare, so only letter-level slips reach this point. Like mapToSlug itself, the index covers
|
|
231
|
+
# every language's names, so a typo resolves no matter which language was asked for.
|
|
232
|
+
def fuzzy_map_to_slug(normalized_name):
|
|
233
|
+
if len(normalized_name) < minimum_fuzzy_name_length:
|
|
234
|
+
return None
|
|
235
|
+
deletions = get_map_to_slug_deletions()
|
|
236
|
+
|
|
237
|
+
# The known name has one character that the input is missing
|
|
238
|
+
slug = deletions.get(normalized_name)
|
|
239
|
+
if slug:
|
|
240
|
+
return slug
|
|
241
|
+
|
|
242
|
+
# The input has one character too many, or one character wrong
|
|
243
|
+
for deletion in get_single_character_deletions(normalized_name):
|
|
244
|
+
slug = deletions.get(deletion)
|
|
245
|
+
if slug:
|
|
246
|
+
return slug
|
|
247
|
+
|
|
248
|
+
return None
|
|
249
|
+
|
|
250
|
+
|
|
251
|
+
# Look up the book slug for a name as it was written. The name is tried as-is, then normalized (which folds
|
|
252
|
+
# case, diacritics and punctuation), and finally – unless allow_loose_match is False – as a name that's off
|
|
253
|
+
# by one character. Every caller that resolves a name should come through here, so the three stages stay in
|
|
254
|
+
# one order.
|
|
255
|
+
def get_book_slug(book_string, allow_loose_match = True):
|
|
256
|
+
slug = scriptures['mapToSlug'].get(book_string)
|
|
257
|
+
if slug:
|
|
258
|
+
return slug
|
|
259
|
+
normalized_name = normalize_for_compare(book_string)
|
|
260
|
+
return get_map_to_slug_normalized().get(normalized_name) or (fuzzy_map_to_slug(normalized_name) if allow_loose_match else None)
|
|
261
|
+
|
|
262
|
+
|
|
263
|
+
# Get regex patterns for the words that can appear in a scripture reference in a given language. Languages that aren't listed above return empty patterns.
|
|
264
|
+
reference_words_cache = {}
|
|
265
|
+
def get_reference_words(lang):
|
|
266
|
+
if lang not in reference_words_cache:
|
|
267
|
+
words_for_lang = reference_words.get(lang, {})
|
|
268
|
+
reference_words_cache[lang] = {
|
|
269
|
+
key: r'|'.join([re.escape(w) for w in words_for_lang.get(key, [])])
|
|
270
|
+
for key in reference_word_keys
|
|
271
|
+
}
|
|
272
|
+
return reference_words_cache[lang]
|
|
273
|
+
|
|
274
|
+
|
|
275
|
+
# Get the BCP 47 language tag for a given language code
|
|
276
|
+
def get_bcp47(lang):
|
|
277
|
+
if lang and 'Hant' in lang:
|
|
278
|
+
lang = 'cmn-Hant'
|
|
279
|
+
elif lang and 'Hans' in lang:
|
|
280
|
+
lang = 'cmn-Hans'
|
|
281
|
+
|
|
282
|
+
bcp47 = languages.get('mapToBcp47', {}).get(lang)
|
|
283
|
+
if not bcp47:
|
|
284
|
+
bcp47 = 'en'
|
|
285
|
+
if lang:
|
|
286
|
+
sys.stdout.write(f'Warning: Couldn’t find BCP 47 language tag for “{lang}” – falling back to “en” (English).\n')
|
|
287
|
+
|
|
288
|
+
return bcp47
|
|
289
|
+
|
|
290
|
+
|
|
291
|
+
# Get the content for a given chapter verse from python-scripture-scraper or ChurchofJesusChrist.org
|
|
292
|
+
def request_content(publication_slug, book_slug, chapter, verse_groups, church_url, lang = 'en', source = 'python-scripture-scraper'):
|
|
293
|
+
import requests
|
|
294
|
+
from bs4 import BeautifulSoup
|
|
295
|
+
|
|
296
|
+
text_content = ''
|
|
297
|
+
|
|
298
|
+
if not publication_slug and book_slug and chapter:
|
|
299
|
+
return text_content
|
|
300
|
+
|
|
301
|
+
verse_numbers = []
|
|
302
|
+
if verse_groups:
|
|
303
|
+
for verse_group in verse_groups:
|
|
304
|
+
for verse_number in verse_group:
|
|
305
|
+
verse_numbers.append(str(verse_number))
|
|
306
|
+
|
|
307
|
+
if source == 'python-scripture-scraper':
|
|
308
|
+
request_url = f'https://cdn.jsdelivr.net/gh/samuelbradshaw/python-scripture-scraper@main/sample/en-json/{publication_slug}/{book_slug}/{book_slug}-{chapter}.json'
|
|
309
|
+
r = requests.get(request_url)
|
|
310
|
+
r.encoding = 'utf-8'
|
|
311
|
+
if r and r.status_code == 200:
|
|
312
|
+
chapter_data = r.json()
|
|
313
|
+
|
|
314
|
+
if verse_numbers:
|
|
315
|
+
for paragraph in chapter_data['paragraphs']:
|
|
316
|
+
if paragraph['type'] == 'verse' and paragraph['number'] in verse_numbers:
|
|
317
|
+
text_content += paragraph['number'] + ' ' + paragraph['content'] + '\n\n'
|
|
318
|
+
else:
|
|
319
|
+
for paragraph in chapter_data['paragraphs']:
|
|
320
|
+
text_content += (paragraph['number'] + ' ' if paragraph['number'] else '') + paragraph['content'] + '\n\n'
|
|
321
|
+
|
|
322
|
+
text_content += '---------------------\n'
|
|
323
|
+
text_content += 'Source: https://github.com/samuelbradshaw/python-scripture-scraper/tree/main/sample\n'
|
|
324
|
+
text_content += 'Public domain.\n'
|
|
325
|
+
|
|
326
|
+
elif source == 'ChurchofJesusChrist.org' and church_url:
|
|
327
|
+
r = requests.get(church_url)
|
|
328
|
+
r.encoding = 'utf-8'
|
|
329
|
+
if r and r.status_code == 200:
|
|
330
|
+
soup = BeautifulSoup(r.text, 'html.parser')
|
|
331
|
+
paragraphs = soup.select('header [data-aid], .body-block [data-aid]')
|
|
332
|
+
if verse_numbers:
|
|
333
|
+
for paragraph in paragraphs:
|
|
334
|
+
verse_number_span = paragraph.select_one('.verse-number')
|
|
335
|
+
if verse_number_span and verse_number_span.text.strip() in verse_numbers:
|
|
336
|
+
text_content += paragraph.text.strip() + '\n\n'
|
|
337
|
+
else:
|
|
338
|
+
for paragraph in paragraphs:
|
|
339
|
+
text_content += paragraph.text.strip() + '\n\n'
|
|
340
|
+
|
|
341
|
+
text_content += '---------------------\n'
|
|
342
|
+
text_content += f'Source: {church_url}\n'
|
|
343
|
+
text_content += 'Some content from this source may be subject to copyright.\n'
|
|
344
|
+
|
|
345
|
+
# Pause for 1 second between requests to avoid overloading server
|
|
346
|
+
time.sleep(1)
|
|
347
|
+
|
|
348
|
+
return text_content
|
|
349
|
+
|