iso-langcodes 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,16 @@
1
+ """
2
+ langcodes - Python modules for working with language codes, language groups and language scripts.
3
+
4
+ This package provides:
5
+ - ISO 639-3: conversion between ISO language codes
6
+ - ISO 639-5: language groups and families according to ISO 639-5
7
+ - ISO 15924: language scripts and their connection to languages
8
+ """
9
+
10
+ from . import iso639_3
11
+ from . import iso639_5
12
+ from . import iso15924
13
+
14
+ __version__ = "1.0.0"
15
+ __author__ = "Joerg Tiedemann"
16
+ __all__ = ["iso639_3", "iso639_5", "iso15924"]
@@ -0,0 +1 @@
1
+ """CLI package for langcodes."""
@@ -0,0 +1,68 @@
1
+ #!/usr/bin/env python3
2
+ """
3
+ iso639 - a simple script to convert language codes.
4
+
5
+ Usage: iso639 [-2|-3|-m|-n|-k] [langcode]*
6
+
7
+ Convert to 3-letter-code if 2-letter code is given and vice versa.
8
+ -2 ... print 2-letter code (even if the input is a 2-letter code)
9
+ -3 ... print 3-letter code (even if the input is a 3-letter code)
10
+ -m ... print macro language instead of local language variants
11
+ -n ... don't print a final new-line
12
+ -k ... keep original code if no mapping is found
13
+ -p ... convert language pairs
14
+ -P ... convert language pairs and sort alphabetically
15
+ """
16
+
17
+ import argparse
18
+ import sys
19
+
20
+ from .. import iso639_3
21
+
22
+
23
+ def convert_pairs(type_, langs, sorted_, keep):
24
+ """Convert language pairs instead of single language codes."""
25
+ converted = [iso639_3.convert_iso639(type_, lang, keep) for lang in langs.split('-')]
26
+ if sorted_:
27
+ converted = sorted(converted)
28
+ return '-'.join(converted)
29
+
30
+
31
+ def main():
32
+ parser = argparse.ArgumentParser(description='Convert language codes')
33
+ parser.add_argument('-2', '--iso639-1', action='store_true', help='convert to two-letter code (ISO 639-1)')
34
+ parser.add_argument('-3', '--iso639-3', action='store_true', help='convert to three-letter code (ISO 639-3)')
35
+ parser.add_argument('-m', '--macro', action='store_true', help='convert to macro language')
36
+ parser.add_argument('-n', '--no-newline', action='store_true', help="don't print a final new line")
37
+ parser.add_argument('-k', '--keep', action='store_true', help='keep original code if no mapping is found')
38
+ parser.add_argument('-p', '--pairs', action='store_true', help='convert language pairs')
39
+ parser.add_argument('-P', '--pairs-sorted', action='store_true', help='convert language pairs and sort alphabetically')
40
+ parser.add_argument('langcodes', nargs='*', help='language codes to convert')
41
+
42
+ args = parser.parse_args()
43
+
44
+ if args.iso639_1:
45
+ type_ = 'iso639-1'
46
+ elif args.iso639_3:
47
+ type_ = 'iso639-3'
48
+ elif args.macro:
49
+ type_ = 'macro'
50
+ else:
51
+ type_ = 'name'
52
+
53
+ if args.pairs or args.pairs_sorted:
54
+ converted = [convert_pairs(type_, lang, args.pairs_sorted, args.keep) for lang in args.langcodes]
55
+ else:
56
+ converted = [iso639_3.convert_iso639(type_, lang, args.keep) for lang in args.langcodes]
57
+
58
+ if type_ == 'name' and converted:
59
+ print('"' + '" "'.join(converted) + '"', end='')
60
+ else:
61
+ print(' '.join(converted), end='')
62
+
63
+ if not args.no_newline:
64
+ print()
65
+
66
+
67
+ if __name__ == '__main__':
68
+ main()
@@ -0,0 +1,58 @@
1
+ #!/usr/bin/env python3
2
+ """
3
+ langglottolog - convert between ISO 639-5 codes and Glottolog codes.
4
+
5
+ Usage: langglottolog [OPTIONS] CODE*
6
+
7
+ Options:
8
+ -h: help
9
+ -i: convert from ISO to Glottolog (default)
10
+ -g: convert from Glottolog to ISO
11
+ -n: don't print final newline
12
+ -v: verbose output
13
+ """
14
+
15
+ import argparse
16
+ import sys
17
+
18
+ from .. import iso639_5
19
+
20
+
21
+ def main():
22
+ parser = argparse.ArgumentParser(description='Convert between ISO 639-5 codes and Glottolog codes')
23
+ parser.add_argument('-i', '--iso2glottolog', action='store_true', help='convert from ISO to Glottolog (default)')
24
+ parser.add_argument('-g', '--glottolog2iso', action='store_true', help='convert from Glottolog to ISO')
25
+ parser.add_argument('-n', '--no-newline', action='store_true', help="don't print final newline")
26
+ parser.add_argument('-v', '--verbose', action='store_true', help='verbose output')
27
+ parser.add_argument('codes', nargs='*', help='codes to convert')
28
+
29
+ args = parser.parse_args()
30
+
31
+ if args.verbose:
32
+ iso639_5.VERBOSE = 1
33
+
34
+ # Read from STDIN or take codes from command line args
35
+ if args.codes:
36
+ codes = args.codes
37
+ else:
38
+ codes = sys.stdin.read().split()
39
+
40
+ # Default to ISO to Glottolog if neither flag is set
41
+ if not args.glottolog2iso:
42
+ args.iso2glottolog = True
43
+
44
+ converted = []
45
+ for code in codes:
46
+ if args.glottolog2iso:
47
+ result = iso639_5.glottolog2iso(code)
48
+ else:
49
+ result = iso639_5.iso2glottolog(code)
50
+ converted.append(result if result else 'unknown')
51
+
52
+ print(' '.join(converted), end='')
53
+ if not args.no_newline:
54
+ print()
55
+
56
+
57
+ if __name__ == '__main__':
58
+ main()
@@ -0,0 +1,96 @@
1
+ #!/usr/bin/env python3
2
+ """
3
+ langgroup - print language groups according to ISO639-5.
4
+
5
+ Usage: langgroup [OPTIONS] LANGCODE*
6
+
7
+ Options:
8
+ -h: help
9
+ -c: group children
10
+ -g: group languages
11
+ -G: group by grandparent
12
+ -n: don't print final newline
13
+ -p: print parent code of each given language code
14
+ -P: print parents all the way up the language tree
15
+ -v: verbose output
16
+ """
17
+
18
+ import argparse
19
+ import sys
20
+
21
+ from .. import iso639_5
22
+
23
+
24
+ def main():
25
+ parser = argparse.ArgumentParser(description='Print language groups according to ISO639-5')
26
+ parser.add_argument('-c', '--children', action='store_true', help='group children')
27
+ parser.add_argument('-g', '--group', action='store_true', help='group languages')
28
+ parser.add_argument('-G', '--grandparent', action='store_true', help='group by grandparent')
29
+ parser.add_argument('-n', '--no-newline', action='store_true', help="don't print final newline")
30
+ parser.add_argument('-p', '--parent', action='store_true', help='print parent code of each given language code')
31
+ parser.add_argument('-P', '--parents', action='store_true', help='print parents all the way up the language tree')
32
+ parser.add_argument('-v', '--verbose', action='store_true', help='verbose output')
33
+ parser.add_argument('langcodes', nargs='*', help='language codes')
34
+
35
+ args = parser.parse_args()
36
+
37
+ if args.verbose:
38
+ iso639_5.VERBOSE = 1
39
+
40
+ # Read from STDIN or take codes from command line args
41
+ if args.langcodes:
42
+ codes = args.langcodes
43
+ else:
44
+ codes = sys.stdin.read().split()
45
+
46
+ if args.parent:
47
+ converted = [iso639_5.language_parent(code) for code in codes]
48
+ print(' '.join(str(c) for c in converted), end='')
49
+ if not args.no_newline:
50
+ print()
51
+ elif args.parents:
52
+ trees = []
53
+ for lang in codes:
54
+ parents = []
55
+ while True:
56
+ p = iso639_5.language_parent(lang)
57
+ if not p:
58
+ break
59
+ parents.append(p)
60
+ lang = p
61
+ if p == 'mul':
62
+ break
63
+ if parents:
64
+ trees.append(':'.join(parents))
65
+ else:
66
+ trees.append('none')
67
+ print(' '.join(trees), end='')
68
+ if not args.no_newline:
69
+ print()
70
+ elif args.group or args.grandparent:
71
+ groups = {}
72
+ for code in codes:
73
+ parent = iso639_5.language_parent(code) or code
74
+ if args.grandparent:
75
+ parent = iso639_5.language_parent(parent) or parent
76
+ if parent not in groups:
77
+ groups[parent] = set()
78
+ groups[parent].add(code)
79
+ for key in sorted(groups.keys()):
80
+ if args.no_newline:
81
+ print('+'.join(sorted(groups[key])), end=' ')
82
+ else:
83
+ print(key, end='\t')
84
+ print(' '.join(sorted(groups[key])))
85
+ else:
86
+ for code in codes:
87
+ if args.children:
88
+ print(' '.join(iso639_5.language_group_children(code)), end='')
89
+ else:
90
+ print(' '.join(iso639_5.language_group(code)), end='')
91
+ if not args.no_newline:
92
+ print()
93
+
94
+
95
+ if __name__ == '__main__':
96
+ main()
@@ -0,0 +1,135 @@
1
+ #!/usr/bin/env python3
2
+ """
3
+ langscript - detect characters from various scripts.
4
+
5
+ Usage: langscript [OPTIONS] < input.txt > script-codes.txt
6
+
7
+ Find the script of a text and print the script codes line by line.
8
+
9
+ Options:
10
+ -a ............ print all scripts found in each line
11
+ -l <langid> ... language hint (start by looking at language-specific scripts first)
12
+ -L ............ two-column input (langid <TAB> text)
13
+ -n ............ print script names instead of script codes
14
+ -h ............ print usage information
15
+ -r ............ also print region/territory
16
+ -R ............ like -r but print default region if no other region found
17
+ -D ............ suppress default script codes
18
+ -1 ............ also print ISO-639-1 code
19
+ -3 ............ also print ISO-639-3 code
20
+ """
21
+
22
+ import argparse
23
+ import sys
24
+
25
+ from .. import iso15924
26
+ from .. import iso639_3
27
+
28
+
29
+ def get_region(lang, use_default=False):
30
+ """Get region and cache values to avoid looking them up again."""
31
+ region = iso15924.language_territory(lang, use_default) or "XX"
32
+ return region
33
+
34
+
35
+ def get_default_region(lang):
36
+ """Get default region and cache values to avoid looking them up again."""
37
+ region = iso15924.default_territory(lang) or "XX"
38
+ return region
39
+
40
+
41
+ def get_default_script(lang):
42
+ """Get default script and cache values to avoid looking them up again."""
43
+ script = iso15924.default_script(lang) or ""
44
+ return script
45
+
46
+
47
+ def main():
48
+ parser = argparse.ArgumentParser(description='Detect characters from various scripts')
49
+ parser.add_argument('-a', '--all', action='store_true', help='print all scripts found in each line')
50
+ parser.add_argument('-l', '--lang', help='language hint')
51
+ parser.add_argument('-L', '--lang-text', action='store_true', help='two-column input (langid <TAB> text)')
52
+ parser.add_argument('-n', '--names', action='store_true', help='print script names instead of script codes')
53
+ parser.add_argument('-r', '--region', action='store_true', help='also print region/territory')
54
+ parser.add_argument('-R', '--region-default', action='store_true', help='like -r but print default region if no other region found')
55
+ parser.add_argument('-D', '--suppress-default', action='store_true', help='suppress default script codes')
56
+ parser.add_argument('-1', '--iso639-1', action='store_true', help='also print ISO-639-1 code')
57
+ parser.add_argument('-3', '--iso639-3', action='store_true', help='also print ISO-639-3 code')
58
+
59
+ args = parser.parse_args()
60
+
61
+ default_script = ""
62
+ default_region = "XX"
63
+ region = "XX"
64
+
65
+ if args.lang:
66
+ default_script = get_default_script(args.lang)
67
+ default_region = get_default_region(args.lang)
68
+ region = get_region(args.lang, args.region_default)
69
+
70
+ for line in sys.stdin:
71
+ line = line.rstrip('\n')
72
+ if args.lang_text:
73
+ parts = line.split('\t')
74
+ if len(parts) >= 2:
75
+ lang = parts[0]
76
+ text = parts[1]
77
+ else:
78
+ lang = args.lang
79
+ text = line
80
+ else:
81
+ lang = args.lang
82
+ text = line
83
+
84
+ if args.lang_text:
85
+ default_region = get_default_region(lang)
86
+ default_script = get_default_script(lang)
87
+
88
+ if args.all:
89
+ scripts = iso15924.script_of_string(text, lang, allow_non_standard=True)
90
+ if isinstance(scripts, dict):
91
+ sorted_scripts = sorted(scripts.keys(), key=lambda s: scripts[s], reverse=True)
92
+ for s in sorted_scripts:
93
+ if args.names:
94
+ print(f"{iso15924.script_name(s)} ({scripts[s]})", end=' ')
95
+ else:
96
+ print(f"{s} ({scripts[s]})", end=' ')
97
+ print()
98
+ else:
99
+ output = []
100
+ langcode = lang
101
+
102
+ if lang:
103
+ if lang == 'ku' or lang == 'kur':
104
+ script = iso15924.script_of_string(text, lang)
105
+ if script:
106
+ langid = f'ku_{script}'
107
+ default_script = get_default_script(iso639_3.get_iso639_3(langid))
108
+ if args.iso639_1:
109
+ output.append(iso639_3.get_iso639_1(langcode, True))
110
+ elif args.iso639_3:
111
+ output.append(iso639_3.get_iso639_3(langcode, True))
112
+
113
+ script = iso15924.script_of_string(text, lang) or default_script
114
+
115
+ if not args.suppress_default or (script != default_script and script != 'Zyyy'):
116
+ if args.names:
117
+ if script:
118
+ output.append(iso15924.script_name(script))
119
+ else:
120
+ if script:
121
+ output.append(script)
122
+
123
+ # Update region if we have dynamic language labels
124
+ if lang and (args.region or args.region_default):
125
+ if args.lang_text:
126
+ region = get_region(lang, args.region_default)
127
+ if not args.suppress_default or (region != default_region and region != 'XX'):
128
+ if region:
129
+ output.append(region)
130
+
131
+ print('_'.join(output))
132
+
133
+
134
+ if __name__ == '__main__':
135
+ main()
@@ -0,0 +1 @@
1
+ """Data package for langcodes."""