iso-langcodes 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- iso_langcodes/__init__.py +16 -0
- iso_langcodes/cli/__init__.py +1 -0
- iso_langcodes/cli/iso639.py +68 -0
- iso_langcodes/cli/langglottolog.py +58 -0
- iso_langcodes/cli/langgroup.py +96 -0
- iso_langcodes/cli/langscript.py +135 -0
- iso_langcodes/data/__init__.py +1 -0
- iso_langcodes/data/iso15924_data.py +9446 -0
- iso_langcodes/data/iso639_3_data.py +17380 -0
- iso_langcodes/data/iso639_5_data.py +6343 -0
- iso_langcodes/iso15924.py +521 -0
- iso_langcodes/iso639_3.py +148 -0
- iso_langcodes/iso639_5.py +79 -0
- iso_langcodes-1.0.0.dist-info/METADATA +167 -0
- iso_langcodes-1.0.0.dist-info/RECORD +19 -0
- iso_langcodes-1.0.0.dist-info/WHEEL +5 -0
- iso_langcodes-1.0.0.dist-info/entry_points.txt +5 -0
- iso_langcodes-1.0.0.dist-info/licenses/LICENSE +21 -0
- iso_langcodes-1.0.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
"""
|
|
2
|
+
langcodes - Python modules for working with language codes, language groups and language scripts.
|
|
3
|
+
|
|
4
|
+
This package provides:
|
|
5
|
+
- ISO 639-3: conversion between ISO language codes
|
|
6
|
+
- ISO 639-5: language groups and families according to ISO 639-5
|
|
7
|
+
- ISO 15924: language scripts and their connection to languages
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from . import iso639_3
|
|
11
|
+
from . import iso639_5
|
|
12
|
+
from . import iso15924
|
|
13
|
+
|
|
14
|
+
__version__ = "1.0.0"
|
|
15
|
+
__author__ = "Joerg Tiedemann"
|
|
16
|
+
__all__ = ["iso639_3", "iso639_5", "iso15924"]
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
"""CLI package for langcodes."""
|
|
@@ -0,0 +1,68 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""
|
|
3
|
+
iso639 - a simple script to convert language codes.
|
|
4
|
+
|
|
5
|
+
Usage: iso639 [-2|-3|-m|-n|-k] [langcode]*
|
|
6
|
+
|
|
7
|
+
Convert to 3-letter-code if 2-letter code is given and vice versa.
|
|
8
|
+
-2 ... print 2-letter code (even if the input is a 2-letter code)
|
|
9
|
+
-3 ... print 3-letter code (even if the input is a 3-letter code)
|
|
10
|
+
-m ... print macro language instead of local language variants
|
|
11
|
+
-n ... don't print a final new-line
|
|
12
|
+
-k ... keep original code if no mapping is found
|
|
13
|
+
-p ... convert language pairs
|
|
14
|
+
-P ... convert language pairs and sort alphabetically
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
import argparse
|
|
18
|
+
import sys
|
|
19
|
+
|
|
20
|
+
from .. import iso639_3
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def convert_pairs(type_, langs, sorted_, keep):
|
|
24
|
+
"""Convert language pairs instead of single language codes."""
|
|
25
|
+
converted = [iso639_3.convert_iso639(type_, lang, keep) for lang in langs.split('-')]
|
|
26
|
+
if sorted_:
|
|
27
|
+
converted = sorted(converted)
|
|
28
|
+
return '-'.join(converted)
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def main():
|
|
32
|
+
parser = argparse.ArgumentParser(description='Convert language codes')
|
|
33
|
+
parser.add_argument('-2', '--iso639-1', action='store_true', help='convert to two-letter code (ISO 639-1)')
|
|
34
|
+
parser.add_argument('-3', '--iso639-3', action='store_true', help='convert to three-letter code (ISO 639-3)')
|
|
35
|
+
parser.add_argument('-m', '--macro', action='store_true', help='convert to macro language')
|
|
36
|
+
parser.add_argument('-n', '--no-newline', action='store_true', help="don't print a final new line")
|
|
37
|
+
parser.add_argument('-k', '--keep', action='store_true', help='keep original code if no mapping is found')
|
|
38
|
+
parser.add_argument('-p', '--pairs', action='store_true', help='convert language pairs')
|
|
39
|
+
parser.add_argument('-P', '--pairs-sorted', action='store_true', help='convert language pairs and sort alphabetically')
|
|
40
|
+
parser.add_argument('langcodes', nargs='*', help='language codes to convert')
|
|
41
|
+
|
|
42
|
+
args = parser.parse_args()
|
|
43
|
+
|
|
44
|
+
if args.iso639_1:
|
|
45
|
+
type_ = 'iso639-1'
|
|
46
|
+
elif args.iso639_3:
|
|
47
|
+
type_ = 'iso639-3'
|
|
48
|
+
elif args.macro:
|
|
49
|
+
type_ = 'macro'
|
|
50
|
+
else:
|
|
51
|
+
type_ = 'name'
|
|
52
|
+
|
|
53
|
+
if args.pairs or args.pairs_sorted:
|
|
54
|
+
converted = [convert_pairs(type_, lang, args.pairs_sorted, args.keep) for lang in args.langcodes]
|
|
55
|
+
else:
|
|
56
|
+
converted = [iso639_3.convert_iso639(type_, lang, args.keep) for lang in args.langcodes]
|
|
57
|
+
|
|
58
|
+
if type_ == 'name' and converted:
|
|
59
|
+
print('"' + '" "'.join(converted) + '"', end='')
|
|
60
|
+
else:
|
|
61
|
+
print(' '.join(converted), end='')
|
|
62
|
+
|
|
63
|
+
if not args.no_newline:
|
|
64
|
+
print()
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
if __name__ == '__main__':
|
|
68
|
+
main()
|
|
@@ -0,0 +1,58 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""
|
|
3
|
+
langglottolog - convert between ISO 639-5 codes and Glottolog codes.
|
|
4
|
+
|
|
5
|
+
Usage: langglottolog [OPTIONS] CODE*
|
|
6
|
+
|
|
7
|
+
Options:
|
|
8
|
+
-h: help
|
|
9
|
+
-i: convert from ISO to Glottolog (default)
|
|
10
|
+
-g: convert from Glottolog to ISO
|
|
11
|
+
-n: don't print final newline
|
|
12
|
+
-v: verbose output
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
import argparse
|
|
16
|
+
import sys
|
|
17
|
+
|
|
18
|
+
from .. import iso639_5
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def main():
|
|
22
|
+
parser = argparse.ArgumentParser(description='Convert between ISO 639-5 codes and Glottolog codes')
|
|
23
|
+
parser.add_argument('-i', '--iso2glottolog', action='store_true', help='convert from ISO to Glottolog (default)')
|
|
24
|
+
parser.add_argument('-g', '--glottolog2iso', action='store_true', help='convert from Glottolog to ISO')
|
|
25
|
+
parser.add_argument('-n', '--no-newline', action='store_true', help="don't print final newline")
|
|
26
|
+
parser.add_argument('-v', '--verbose', action='store_true', help='verbose output')
|
|
27
|
+
parser.add_argument('codes', nargs='*', help='codes to convert')
|
|
28
|
+
|
|
29
|
+
args = parser.parse_args()
|
|
30
|
+
|
|
31
|
+
if args.verbose:
|
|
32
|
+
iso639_5.VERBOSE = 1
|
|
33
|
+
|
|
34
|
+
# Read from STDIN or take codes from command line args
|
|
35
|
+
if args.codes:
|
|
36
|
+
codes = args.codes
|
|
37
|
+
else:
|
|
38
|
+
codes = sys.stdin.read().split()
|
|
39
|
+
|
|
40
|
+
# Default to ISO to Glottolog if neither flag is set
|
|
41
|
+
if not args.glottolog2iso:
|
|
42
|
+
args.iso2glottolog = True
|
|
43
|
+
|
|
44
|
+
converted = []
|
|
45
|
+
for code in codes:
|
|
46
|
+
if args.glottolog2iso:
|
|
47
|
+
result = iso639_5.glottolog2iso(code)
|
|
48
|
+
else:
|
|
49
|
+
result = iso639_5.iso2glottolog(code)
|
|
50
|
+
converted.append(result if result else 'unknown')
|
|
51
|
+
|
|
52
|
+
print(' '.join(converted), end='')
|
|
53
|
+
if not args.no_newline:
|
|
54
|
+
print()
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
if __name__ == '__main__':
|
|
58
|
+
main()
|
|
@@ -0,0 +1,96 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""
|
|
3
|
+
langgroup - print language groups according to ISO639-5.
|
|
4
|
+
|
|
5
|
+
Usage: langgroup [OPTIONS] LANGCODE*
|
|
6
|
+
|
|
7
|
+
Options:
|
|
8
|
+
-h: help
|
|
9
|
+
-c: group children
|
|
10
|
+
-g: group languages
|
|
11
|
+
-G: group by grandparent
|
|
12
|
+
-n: don't print final newline
|
|
13
|
+
-p: print parent code of each given language code
|
|
14
|
+
-P: print parents all the way up the language tree
|
|
15
|
+
-v: verbose output
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
import argparse
|
|
19
|
+
import sys
|
|
20
|
+
|
|
21
|
+
from .. import iso639_5
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def main():
|
|
25
|
+
parser = argparse.ArgumentParser(description='Print language groups according to ISO639-5')
|
|
26
|
+
parser.add_argument('-c', '--children', action='store_true', help='group children')
|
|
27
|
+
parser.add_argument('-g', '--group', action='store_true', help='group languages')
|
|
28
|
+
parser.add_argument('-G', '--grandparent', action='store_true', help='group by grandparent')
|
|
29
|
+
parser.add_argument('-n', '--no-newline', action='store_true', help="don't print final newline")
|
|
30
|
+
parser.add_argument('-p', '--parent', action='store_true', help='print parent code of each given language code')
|
|
31
|
+
parser.add_argument('-P', '--parents', action='store_true', help='print parents all the way up the language tree')
|
|
32
|
+
parser.add_argument('-v', '--verbose', action='store_true', help='verbose output')
|
|
33
|
+
parser.add_argument('langcodes', nargs='*', help='language codes')
|
|
34
|
+
|
|
35
|
+
args = parser.parse_args()
|
|
36
|
+
|
|
37
|
+
if args.verbose:
|
|
38
|
+
iso639_5.VERBOSE = 1
|
|
39
|
+
|
|
40
|
+
# Read from STDIN or take codes from command line args
|
|
41
|
+
if args.langcodes:
|
|
42
|
+
codes = args.langcodes
|
|
43
|
+
else:
|
|
44
|
+
codes = sys.stdin.read().split()
|
|
45
|
+
|
|
46
|
+
if args.parent:
|
|
47
|
+
converted = [iso639_5.language_parent(code) for code in codes]
|
|
48
|
+
print(' '.join(str(c) for c in converted), end='')
|
|
49
|
+
if not args.no_newline:
|
|
50
|
+
print()
|
|
51
|
+
elif args.parents:
|
|
52
|
+
trees = []
|
|
53
|
+
for lang in codes:
|
|
54
|
+
parents = []
|
|
55
|
+
while True:
|
|
56
|
+
p = iso639_5.language_parent(lang)
|
|
57
|
+
if not p:
|
|
58
|
+
break
|
|
59
|
+
parents.append(p)
|
|
60
|
+
lang = p
|
|
61
|
+
if p == 'mul':
|
|
62
|
+
break
|
|
63
|
+
if parents:
|
|
64
|
+
trees.append(':'.join(parents))
|
|
65
|
+
else:
|
|
66
|
+
trees.append('none')
|
|
67
|
+
print(' '.join(trees), end='')
|
|
68
|
+
if not args.no_newline:
|
|
69
|
+
print()
|
|
70
|
+
elif args.group or args.grandparent:
|
|
71
|
+
groups = {}
|
|
72
|
+
for code in codes:
|
|
73
|
+
parent = iso639_5.language_parent(code) or code
|
|
74
|
+
if args.grandparent:
|
|
75
|
+
parent = iso639_5.language_parent(parent) or parent
|
|
76
|
+
if parent not in groups:
|
|
77
|
+
groups[parent] = set()
|
|
78
|
+
groups[parent].add(code)
|
|
79
|
+
for key in sorted(groups.keys()):
|
|
80
|
+
if args.no_newline:
|
|
81
|
+
print('+'.join(sorted(groups[key])), end=' ')
|
|
82
|
+
else:
|
|
83
|
+
print(key, end='\t')
|
|
84
|
+
print(' '.join(sorted(groups[key])))
|
|
85
|
+
else:
|
|
86
|
+
for code in codes:
|
|
87
|
+
if args.children:
|
|
88
|
+
print(' '.join(iso639_5.language_group_children(code)), end='')
|
|
89
|
+
else:
|
|
90
|
+
print(' '.join(iso639_5.language_group(code)), end='')
|
|
91
|
+
if not args.no_newline:
|
|
92
|
+
print()
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
if __name__ == '__main__':
|
|
96
|
+
main()
|
|
@@ -0,0 +1,135 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""
|
|
3
|
+
langscript - detect characters from various scripts.
|
|
4
|
+
|
|
5
|
+
Usage: langscript [OPTIONS] < input.txt > script-codes.txt
|
|
6
|
+
|
|
7
|
+
Find the script of a text and print the script codes line by line.
|
|
8
|
+
|
|
9
|
+
Options:
|
|
10
|
+
-a ............ print all scripts found in each line
|
|
11
|
+
-l <langid> ... language hint (start by looking at language-specific scripts first)
|
|
12
|
+
-L ............ two-column input (langid <TAB> text)
|
|
13
|
+
-n ............ print script names instead of script codes
|
|
14
|
+
-h ............ print usage information
|
|
15
|
+
-r ............ also print region/territory
|
|
16
|
+
-R ............ like -r but print default region if no other region found
|
|
17
|
+
-D ............ suppress default script codes
|
|
18
|
+
-1 ............ also print ISO-639-1 code
|
|
19
|
+
-3 ............ also print ISO-639-3 code
|
|
20
|
+
"""
|
|
21
|
+
|
|
22
|
+
import argparse
|
|
23
|
+
import sys
|
|
24
|
+
|
|
25
|
+
from .. import iso15924
|
|
26
|
+
from .. import iso639_3
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def get_region(lang, use_default=False):
|
|
30
|
+
"""Get region and cache values to avoid looking them up again."""
|
|
31
|
+
region = iso15924.language_territory(lang, use_default) or "XX"
|
|
32
|
+
return region
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def get_default_region(lang):
|
|
36
|
+
"""Get default region and cache values to avoid looking them up again."""
|
|
37
|
+
region = iso15924.default_territory(lang) or "XX"
|
|
38
|
+
return region
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def get_default_script(lang):
|
|
42
|
+
"""Get default script and cache values to avoid looking them up again."""
|
|
43
|
+
script = iso15924.default_script(lang) or ""
|
|
44
|
+
return script
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def main():
|
|
48
|
+
parser = argparse.ArgumentParser(description='Detect characters from various scripts')
|
|
49
|
+
parser.add_argument('-a', '--all', action='store_true', help='print all scripts found in each line')
|
|
50
|
+
parser.add_argument('-l', '--lang', help='language hint')
|
|
51
|
+
parser.add_argument('-L', '--lang-text', action='store_true', help='two-column input (langid <TAB> text)')
|
|
52
|
+
parser.add_argument('-n', '--names', action='store_true', help='print script names instead of script codes')
|
|
53
|
+
parser.add_argument('-r', '--region', action='store_true', help='also print region/territory')
|
|
54
|
+
parser.add_argument('-R', '--region-default', action='store_true', help='like -r but print default region if no other region found')
|
|
55
|
+
parser.add_argument('-D', '--suppress-default', action='store_true', help='suppress default script codes')
|
|
56
|
+
parser.add_argument('-1', '--iso639-1', action='store_true', help='also print ISO-639-1 code')
|
|
57
|
+
parser.add_argument('-3', '--iso639-3', action='store_true', help='also print ISO-639-3 code')
|
|
58
|
+
|
|
59
|
+
args = parser.parse_args()
|
|
60
|
+
|
|
61
|
+
default_script = ""
|
|
62
|
+
default_region = "XX"
|
|
63
|
+
region = "XX"
|
|
64
|
+
|
|
65
|
+
if args.lang:
|
|
66
|
+
default_script = get_default_script(args.lang)
|
|
67
|
+
default_region = get_default_region(args.lang)
|
|
68
|
+
region = get_region(args.lang, args.region_default)
|
|
69
|
+
|
|
70
|
+
for line in sys.stdin:
|
|
71
|
+
line = line.rstrip('\n')
|
|
72
|
+
if args.lang_text:
|
|
73
|
+
parts = line.split('\t')
|
|
74
|
+
if len(parts) >= 2:
|
|
75
|
+
lang = parts[0]
|
|
76
|
+
text = parts[1]
|
|
77
|
+
else:
|
|
78
|
+
lang = args.lang
|
|
79
|
+
text = line
|
|
80
|
+
else:
|
|
81
|
+
lang = args.lang
|
|
82
|
+
text = line
|
|
83
|
+
|
|
84
|
+
if args.lang_text:
|
|
85
|
+
default_region = get_default_region(lang)
|
|
86
|
+
default_script = get_default_script(lang)
|
|
87
|
+
|
|
88
|
+
if args.all:
|
|
89
|
+
scripts = iso15924.script_of_string(text, lang, allow_non_standard=True)
|
|
90
|
+
if isinstance(scripts, dict):
|
|
91
|
+
sorted_scripts = sorted(scripts.keys(), key=lambda s: scripts[s], reverse=True)
|
|
92
|
+
for s in sorted_scripts:
|
|
93
|
+
if args.names:
|
|
94
|
+
print(f"{iso15924.script_name(s)} ({scripts[s]})", end=' ')
|
|
95
|
+
else:
|
|
96
|
+
print(f"{s} ({scripts[s]})", end=' ')
|
|
97
|
+
print()
|
|
98
|
+
else:
|
|
99
|
+
output = []
|
|
100
|
+
langcode = lang
|
|
101
|
+
|
|
102
|
+
if lang:
|
|
103
|
+
if lang == 'ku' or lang == 'kur':
|
|
104
|
+
script = iso15924.script_of_string(text, lang)
|
|
105
|
+
if script:
|
|
106
|
+
langid = f'ku_{script}'
|
|
107
|
+
default_script = get_default_script(iso639_3.get_iso639_3(langid))
|
|
108
|
+
if args.iso639_1:
|
|
109
|
+
output.append(iso639_3.get_iso639_1(langcode, True))
|
|
110
|
+
elif args.iso639_3:
|
|
111
|
+
output.append(iso639_3.get_iso639_3(langcode, True))
|
|
112
|
+
|
|
113
|
+
script = iso15924.script_of_string(text, lang) or default_script
|
|
114
|
+
|
|
115
|
+
if not args.suppress_default or (script != default_script and script != 'Zyyy'):
|
|
116
|
+
if args.names:
|
|
117
|
+
if script:
|
|
118
|
+
output.append(iso15924.script_name(script))
|
|
119
|
+
else:
|
|
120
|
+
if script:
|
|
121
|
+
output.append(script)
|
|
122
|
+
|
|
123
|
+
# Update region if we have dynamic language labels
|
|
124
|
+
if lang and (args.region or args.region_default):
|
|
125
|
+
if args.lang_text:
|
|
126
|
+
region = get_region(lang, args.region_default)
|
|
127
|
+
if not args.suppress_default or (region != default_region and region != 'XX'):
|
|
128
|
+
if region:
|
|
129
|
+
output.append(region)
|
|
130
|
+
|
|
131
|
+
print('_'.join(output))
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
if __name__ == '__main__':
|
|
135
|
+
main()
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
"""Data package for langcodes."""
|