camertron-eprun 1.1.1 → 2.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +5 -5
- data/CHANGELOG.md +12 -0
- data/Gemfile +3 -7
- data/README.md +89 -62
- data/Rakefile +55 -17
- data/lib/eprun/normalizer.rb +215 -0
- data/lib/eprun/ucd_version.rb +83 -0
- data/lib/eprun/v14_0_0/tables.rb +9207 -0
- data/lib/eprun/{core_ext/string.rb → v14_0_0.rb} +3 -12
- data/lib/eprun/v15_1_0/tables.rb +9284 -0
- data/lib/eprun/v15_1_0.rb +11 -0
- data/lib/eprun/v16_0_0/tables.rb +9420 -0
- data/lib/eprun/v16_0_0.rb +11 -0
- data/lib/eprun/v17_0_0/tables.rb +9461 -0
- data/lib/eprun/v17_0_0.rb +11 -0
- data/lib/eprun/v18_0_0/tables.rb +9573 -0
- data/lib/eprun/v18_0_0.rb +11 -0
- data/lib/eprun/version.rb +9 -2
- data/lib/eprun.rb +43 -3
- data/test/test_normalize.rb +198 -0
- data/test/test_normalizer.rb +29 -0
- data/test/test_ucd_version.rb +57 -0
- data/test/test_versions.rb +65 -0
- metadata +23 -15
- data/History.txt +0 -4
- data/lib/eprun/helpers.rb +0 -27
- data/lib/eprun/normalize.rb +0 -185
- data/lib/eprun/ruby18/normalize.rb +0 -198
- data/lib/eprun/ruby18/tables.rb +0 -621
- data/lib/eprun/tables.rb +0 -621
checksums.yaml
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
---
|
|
2
|
-
|
|
3
|
-
metadata.gz:
|
|
4
|
-
data.tar.gz:
|
|
2
|
+
SHA256:
|
|
3
|
+
metadata.gz: 26933843a3512031c7c7d634c1fe12df78d39ee78639f673d1fb175b8df50c21
|
|
4
|
+
data.tar.gz: 8ef34886023216ee7b39aee4d93b9fe1374c855528af2e5885be4cc4d7dd8ade
|
|
5
5
|
SHA512:
|
|
6
|
-
metadata.gz:
|
|
7
|
-
data.tar.gz:
|
|
6
|
+
metadata.gz: ce0baa5b5f798e53d8ab831f7e1865665a0aeaf673cf32cf1223dadeef983cd2cf99fc3be5415c2253370e6ae4da7a4546ba403ebb49df70a32b8e794d2aff54
|
|
7
|
+
data.tar.gz: 06a218b4865a111ebf9696f9bd08b52bdab15a8b1eda0ea0d2250a36ab14856400dd8e11032dbd5f8cacd896238a3f99f507365e0f058165d9bfb5106907e4ea
|
data/CHANGELOG.md
ADDED
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
== 2.0.0
|
|
2
|
+
* Removed MRI 1.8 compatibility.
|
|
3
|
+
* Big internal refactor, but same API.
|
|
4
|
+
* Support multiple versions of the Unicode Character Database (UCD), including the latest (v18.0.0).
|
|
5
|
+
* Switch CI from Travis to GitHub Actions.
|
|
6
|
+
* Simplify benchmarks
|
|
7
|
+
- All the 3rd-party gems used for comparison are very old and use old versions of the UCD.
|
|
8
|
+
- Now we test against Ruby's normalization implementation, which is actually also written in Ruby and was itself upstreamed from the original eprun by @duerst.
|
|
9
|
+
|
|
10
|
+
== 1.0.0
|
|
11
|
+
* Repo converted into gem.
|
|
12
|
+
* Added MRI 1.8 compatibility.
|
data/Gemfile
CHANGED
|
@@ -4,14 +4,10 @@ gemspec
|
|
|
4
4
|
|
|
5
5
|
group :development do
|
|
6
6
|
gem "rake"
|
|
7
|
-
gem "
|
|
8
|
-
gem
|
|
9
|
-
gem 'unf'
|
|
10
|
-
# gem 'unicode_utils'
|
|
11
|
-
gem 'activesupport'
|
|
7
|
+
gem "debug"
|
|
8
|
+
gem "benchmark-ips"
|
|
12
9
|
end
|
|
13
10
|
|
|
14
11
|
group :test do
|
|
15
|
-
gem "
|
|
16
|
-
gem "rr"
|
|
12
|
+
gem "test-unit"
|
|
17
13
|
end
|
data/README.md
CHANGED
|
@@ -1,63 +1,90 @@
|
|
|
1
1
|
Efficient Pure Ruby Unicode Normalization (eprun)
|
|
2
|
-
|
|
3
|
-
|
|
4
|
-
(pronounced
|
|
5
|
-
|
|
6
|
-
|
|
7
|
-
|
|
8
|
-
|
|
9
|
-
|
|
10
|
-
|
|
11
|
-
|
|
12
|
-
|
|
13
|
-
|
|
14
|
-
|
|
15
|
-
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
|
|
2
|
+
===
|
|
3
|
+
|
|
4
|
+
(pronounced ee-prune)
|
|
5
|
+
|
|
6
|
+
## Introduction
|
|
7
|
+
|
|
8
|
+
This library was originally developed by Martin J. Dürst and Ayumu Nojima, a professor and graduate student at Aoyama Gakuin university in Japan. In 2013, Dürst presented eprun, a fast and efficient pure-ruby Unicode normalization algorithm, at the [International Unicode Conference 37](https://www.unicodeconference.org/iuc37/Conference-Program.pdf) in Santa Clara, California. A textual version of the presentation can be found [here](http://www.sw.it.aoyama.ac.jp/2013/pub/RubyNorm/).
|
|
9
|
+
|
|
10
|
+
Shorty after @camertron met Dürst at the conference, @camertron packaged up eprun into a Rubygem, published it to rubygems.org, and replaced the original normalization algorithm in [TwitterCLDR](https://github.com/ruby-i18n/twitter-cldr-rb) with eprun.
|
|
11
|
+
|
|
12
|
+
Dürst, an active member of the Ruby community back in those days, was eventually able to upstream a version of eprun into Ruby itself. The `String#unicode_normalize` method uses a modified version of eprun to perform normalization work.
|
|
13
|
+
|
|
14
|
+
## Purpose
|
|
15
|
+
|
|
16
|
+
The version of eprun built into Ruby only supports a single version of the Unicode Character Database (UCD), which evolves over time. TwitterCLDR uses a specific version of the UCD, and needs eprun to use the same one for consistency and compatibility. Other libraries dependent on eprun may also have the same need. For this reason, eprun supports the last 5 UCD versions.
|
|
17
|
+
|
|
18
|
+
## Usage
|
|
19
|
+
|
|
20
|
+
First, require the library:
|
|
21
|
+
|
|
22
|
+
```ruby
|
|
23
|
+
require "eprun"
|
|
24
|
+
```
|
|
25
|
+
|
|
26
|
+
### Normalizing
|
|
27
|
+
|
|
28
|
+
Eprun's normalization API consists of two methods. To normalize a string:
|
|
29
|
+
|
|
30
|
+
```ruby
|
|
31
|
+
Eprun.normalize("abc")
|
|
32
|
+
```
|
|
33
|
+
|
|
34
|
+
`Eprun.normalize` accepts the normalization form as the second argument. The valid normalization forms are `:nfc` (the default), `:nfd`, `:nfkc`, and `:nfkd`. For more information on Unicode normalization forms, see the [Unicode Standard Annex #15](https://www.unicode.org/reports/tr15/).
|
|
35
|
+
|
|
36
|
+
### Checking normalization
|
|
37
|
+
|
|
38
|
+
The second API method tests if a string is already normalized:
|
|
39
|
+
|
|
40
|
+
```ruby
|
|
41
|
+
Eprun.normalized?("abc")
|
|
42
|
+
```
|
|
43
|
+
|
|
44
|
+
## Using Different UCD Versions
|
|
45
|
+
|
|
46
|
+
The currently supported UCD versions can be inspected via the `Eprun::UCD_VERSIONS` constant, which is an array of strings of the form "1.2.3."
|
|
47
|
+
|
|
48
|
+
To change the version eprun will use during normalization operations, set `Eprun.current_version`.
|
|
49
|
+
|
|
50
|
+
```ruby
|
|
51
|
+
Eprun.current_version = "17.0.0"
|
|
52
|
+
|
|
53
|
+
# normalizes using UCD v17
|
|
54
|
+
Eprun.normalize("abc")
|
|
55
|
+
```
|
|
56
|
+
|
|
57
|
+
There's also a `.with_version` method that changes the current version for the duration of the given block:
|
|
58
|
+
|
|
59
|
+
```ruby
|
|
60
|
+
Eprun.with_version("17.0.0") do
|
|
61
|
+
# normalizes using UCD v17
|
|
62
|
+
Eprun.normalize("abc")
|
|
63
|
+
end
|
|
64
|
+
```
|
|
65
|
+
|
|
66
|
+
## Running Tests
|
|
67
|
+
|
|
68
|
+
Run tests by executing:
|
|
69
|
+
|
|
70
|
+
```shell
|
|
71
|
+
bundle exec rake
|
|
72
|
+
```
|
|
73
|
+
|
|
74
|
+
## Re-generating Normalization Data
|
|
75
|
+
|
|
76
|
+
To download the necessary Unicode files, first run:
|
|
77
|
+
|
|
78
|
+
```shell
|
|
79
|
+
bundle exec rake update_data
|
|
80
|
+
```
|
|
81
|
+
|
|
82
|
+
Then, to re-generate the normalization tables, regexes, etc, run:
|
|
83
|
+
|
|
84
|
+
```shell
|
|
85
|
+
bundle exec rake generate_tables
|
|
86
|
+
```
|
|
87
|
+
|
|
88
|
+
## License
|
|
89
|
+
|
|
90
|
+
Licensed under the terms of the original eprun library, which is licensed under the same terms as Ruby.
|
data/Rakefile
CHANGED
|
@@ -4,37 +4,75 @@
|
|
|
4
4
|
# available under the same licence as Ruby itself
|
|
5
5
|
# (see http://www.ruby-lang.org/en/LICENSE.txt)
|
|
6
6
|
|
|
7
|
-
ROOT_DIR = Pathname.new(
|
|
7
|
+
ROOT_DIR = Pathname.new(__dir__)
|
|
8
8
|
$:.push(ROOT_DIR.to_s)
|
|
9
9
|
|
|
10
|
-
require
|
|
11
|
-
require
|
|
12
|
-
require 'tasks/tables_generator'
|
|
13
|
-
|
|
14
|
-
require 'rubygems/package_task'
|
|
10
|
+
require "eprun"
|
|
11
|
+
require "rubygems/package_task"
|
|
15
12
|
|
|
16
13
|
task :default => :test
|
|
17
14
|
Bundler::GemHelper.install_tasks
|
|
18
15
|
|
|
19
16
|
task :test do
|
|
20
|
-
require
|
|
17
|
+
require "test/unit"
|
|
18
|
+
|
|
21
19
|
files = Dir.glob("./test/test_*.rb")
|
|
22
20
|
runner = Test::Unit::AutoRunner.new(true)
|
|
23
21
|
runner.process_args(files)
|
|
24
|
-
runner.run
|
|
22
|
+
succeeded = runner.run
|
|
23
|
+
exit(succeeded ? 0 : 1)
|
|
24
|
+
end
|
|
25
|
+
|
|
26
|
+
task :update_data do
|
|
27
|
+
require "open-uri"
|
|
28
|
+
require "fileutils"
|
|
29
|
+
|
|
30
|
+
files = ["UnicodeData.txt", "NormalizationTest.txt", "CompositionExclusions.txt"]
|
|
31
|
+
|
|
32
|
+
Eprun::UCD_VERSIONS.each do |ucd_version_string|
|
|
33
|
+
base_url = "https://www.unicode.org/Public/#{ucd_version_string}/ucd"
|
|
34
|
+
ucd_version = Eprun::UcdVersion.get(ucd_version_string)
|
|
35
|
+
|
|
36
|
+
files.each do |file|
|
|
37
|
+
url = "#{base_url}/#{file}"
|
|
38
|
+
FileUtils.mkdir_p(ucd_version.data_path)
|
|
39
|
+
File.write(File.join(ucd_version.data_path, file), URI.open(url).read)
|
|
40
|
+
end
|
|
41
|
+
end
|
|
25
42
|
end
|
|
26
43
|
|
|
27
44
|
task :generate_tables do
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
|
|
45
|
+
require "tasks/tables_generator"
|
|
46
|
+
|
|
47
|
+
Eprun::UCD_VERSIONS.each do |ucd_version_string|
|
|
48
|
+
ucd_version = Eprun::UcdVersion.get(ucd_version_string)
|
|
49
|
+
generator = EprunTasks::TablesGenerator.new(
|
|
50
|
+
ROOT_DIR.join("data").to_s,
|
|
51
|
+
ROOT_DIR.join("lib", "eprun").to_s,
|
|
52
|
+
ucd_version,
|
|
53
|
+
)
|
|
54
|
+
|
|
55
|
+
generator.generate
|
|
56
|
+
end
|
|
32
57
|
end
|
|
33
58
|
|
|
34
59
|
task :benchmark do
|
|
35
|
-
require
|
|
36
|
-
|
|
60
|
+
require "benchmark/ips"
|
|
61
|
+
|
|
62
|
+
deutsch = File.read(File.join(ROOT_DIR, "benchmark", "Deutsch.txt"))
|
|
63
|
+
japanese = File.read(File.join(ROOT_DIR, "benchmark", "Japanese.txt"))
|
|
37
64
|
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
65
|
+
Benchmark.ips do |x|
|
|
66
|
+
x.report("eprun Deutsch") { Eprun.normalize(deutsch, :nfc) }
|
|
67
|
+
x.report("ruby Deutsch") { deutsch.unicode_normalize(:nfc) }
|
|
68
|
+
x.compare!
|
|
69
|
+
end
|
|
70
|
+
|
|
71
|
+
puts
|
|
72
|
+
|
|
73
|
+
Benchmark.ips do |x|
|
|
74
|
+
x.report("eprun Japanese") { Eprun.normalize(japanese, :nfc) }
|
|
75
|
+
x.report("ruby Japanese") { japanese.unicode_normalize(:nfc) }
|
|
76
|
+
x.compare!
|
|
77
|
+
end
|
|
78
|
+
end
|
|
@@ -0,0 +1,215 @@
|
|
|
1
|
+
# encoding: utf-8
|
|
2
|
+
|
|
3
|
+
# Copyright 2010-2013 Ayumu Nojima (野島 歩) and Martin J. Dürst (duerst@it.aoyama.ac.jp)
|
|
4
|
+
# available under the same licence as Ruby itself
|
|
5
|
+
# (see http://www.ruby-lang.org/en/LICENSE.txt)
|
|
6
|
+
|
|
7
|
+
module Eprun
|
|
8
|
+
class Normalizer
|
|
9
|
+
## Constant for max hash capacity to avoid DoS attack
|
|
10
|
+
MAX_HASH_LENGTH = 18000 # enough for all test cases, otherwise tests get slow
|
|
11
|
+
|
|
12
|
+
## Constants For Hangul
|
|
13
|
+
# for details such as the meaning of the identifiers below, please see
|
|
14
|
+
# http://www.unicode.org/versions/Unicode7.0.0/ch03.pdf, pp. 144/145
|
|
15
|
+
SBASE = 0xAC00
|
|
16
|
+
LBASE = 0x1100
|
|
17
|
+
VBASE = 0x1161
|
|
18
|
+
TBASE = 0x11A7
|
|
19
|
+
LCOUNT = 19
|
|
20
|
+
VCOUNT = 21
|
|
21
|
+
TCOUNT = 28
|
|
22
|
+
NCOUNT = VCOUNT * TCOUNT
|
|
23
|
+
SCOUNT = LCOUNT * NCOUNT
|
|
24
|
+
|
|
25
|
+
# Unicode-based encodings (except UTF-8)
|
|
26
|
+
UNICODE_ENCODINGS = [
|
|
27
|
+
Encoding::UTF_16BE,
|
|
28
|
+
Encoding::UTF_16LE,
|
|
29
|
+
Encoding::UTF_32BE,
|
|
30
|
+
Encoding::UTF_32LE,
|
|
31
|
+
Encoding::GB18030,
|
|
32
|
+
Encoding::UCS_2BE,
|
|
33
|
+
Encoding::UCS_4BE
|
|
34
|
+
]
|
|
35
|
+
|
|
36
|
+
class << self
|
|
37
|
+
def get(version = Eprun.current_version)
|
|
38
|
+
version = Eprun::UcdVersion.get(version)
|
|
39
|
+
cache[version.string] ||= new(version)
|
|
40
|
+
end
|
|
41
|
+
|
|
42
|
+
private
|
|
43
|
+
|
|
44
|
+
def cache
|
|
45
|
+
@cache ||= {}
|
|
46
|
+
end
|
|
47
|
+
end
|
|
48
|
+
|
|
49
|
+
attr_reader :version
|
|
50
|
+
|
|
51
|
+
def initialize(version)
|
|
52
|
+
@version = version
|
|
53
|
+
|
|
54
|
+
tables = version.namespace::Tables
|
|
55
|
+
|
|
56
|
+
@class_table = tables.class_table
|
|
57
|
+
@composition_table = tables.composition_table
|
|
58
|
+
@decomposition_table = tables.decomposition_table
|
|
59
|
+
@kompatible_table = tables.kompatible_table
|
|
60
|
+
|
|
61
|
+
## Regular Expressions and Hash Constants
|
|
62
|
+
@regexp_d = Regexp.compile(tables.regexp_d_string, Regexp::EXTENDED)
|
|
63
|
+
@regexp_c = Regexp.compile(tables.regexp_c_string, Regexp::EXTENDED)
|
|
64
|
+
@regexp_k = Regexp.compile(tables.regexp_k_string, Regexp::EXTENDED)
|
|
65
|
+
@nf_hash_d = Hash.new do |hash, key|
|
|
66
|
+
hash.shift if hash.length > MAX_HASH_LENGTH # prevent DoS attack
|
|
67
|
+
hash[key] = nfd_one(key)
|
|
68
|
+
end
|
|
69
|
+
|
|
70
|
+
@nf_hash_c = Hash.new do |hash, key|
|
|
71
|
+
hash.shift if hash.length > MAX_HASH_LENGTH # prevent DoS attack
|
|
72
|
+
hash[key] = nfc_one(key)
|
|
73
|
+
end
|
|
74
|
+
end
|
|
75
|
+
|
|
76
|
+
## Hangul Algorithm
|
|
77
|
+
def hangul_decomp_one(target)
|
|
78
|
+
syllable_index = target.ord - SBASE
|
|
79
|
+
return target if syllable_index < 0 || syllable_index >= SCOUNT
|
|
80
|
+
|
|
81
|
+
l = LBASE + syllable_index / NCOUNT
|
|
82
|
+
v = VBASE + (syllable_index % NCOUNT) / TCOUNT
|
|
83
|
+
t = TBASE + syllable_index % TCOUNT
|
|
84
|
+
(t == TBASE ? [l, v] : [l, v, t]).pack("U*") + target[1..-1]
|
|
85
|
+
end
|
|
86
|
+
|
|
87
|
+
def hangul_comp_one(string)
|
|
88
|
+
length = string.length
|
|
89
|
+
|
|
90
|
+
if length > 1 && 0 <= (lead = string[0].ord - LBASE) && lead < LCOUNT && 0 <= (vowel = string[1].ord - VBASE) && vowel < VCOUNT
|
|
91
|
+
lead_vowel = SBASE + (lead * VCOUNT + vowel) * TCOUNT
|
|
92
|
+
if length > 2 && 0 < (trail = string[2].ord - TBASE) && trail < TCOUNT
|
|
93
|
+
(lead_vowel + trail).chr(Encoding::UTF_8) + string[3..-1]
|
|
94
|
+
else
|
|
95
|
+
lead_vowel.chr(Encoding::UTF_8) + string[2..-1]
|
|
96
|
+
end
|
|
97
|
+
else
|
|
98
|
+
string
|
|
99
|
+
end
|
|
100
|
+
end
|
|
101
|
+
|
|
102
|
+
## Canonical Ordering
|
|
103
|
+
def canonical_ordering_one(string)
|
|
104
|
+
result = +""
|
|
105
|
+
unordered = []
|
|
106
|
+
chars = string.chars
|
|
107
|
+
n = chars.size
|
|
108
|
+
|
|
109
|
+
chars.each_with_index do |char, i|
|
|
110
|
+
ccc = @class_table[char]
|
|
111
|
+
|
|
112
|
+
if ccc == 0
|
|
113
|
+
unordered.sort!.each { |elem| result << chars[elem % n] }
|
|
114
|
+
unordered.clear
|
|
115
|
+
result << char
|
|
116
|
+
else
|
|
117
|
+
unordered << ccc * n + i
|
|
118
|
+
end
|
|
119
|
+
end
|
|
120
|
+
|
|
121
|
+
unordered.sort!.each { |elem| result << chars[elem % n] }
|
|
122
|
+
result
|
|
123
|
+
end
|
|
124
|
+
|
|
125
|
+
## Normalization Forms for Patterns (not whole Strings)
|
|
126
|
+
def nfd_one(string)
|
|
127
|
+
string = string.chars.map! { |c| @decomposition_table[c] || c }.join("")
|
|
128
|
+
canonical_ordering_one(hangul_decomp_one(string))
|
|
129
|
+
end
|
|
130
|
+
|
|
131
|
+
def nfc_one(string)
|
|
132
|
+
nfd_string = nfd_one(string)
|
|
133
|
+
start = nfd_string[0]
|
|
134
|
+
last_class = @class_table[start]-1
|
|
135
|
+
accents = +""
|
|
136
|
+
result = +""
|
|
137
|
+
|
|
138
|
+
nfd_string[1..-1].each_char do |accent|
|
|
139
|
+
accent_class = @class_table[accent]
|
|
140
|
+
|
|
141
|
+
if last_class < accent_class && composite = @composition_table[start + accent]
|
|
142
|
+
start = composite
|
|
143
|
+
elsif accent_class == 0
|
|
144
|
+
result << start << accents
|
|
145
|
+
start = accent
|
|
146
|
+
accents = ""
|
|
147
|
+
last_class = -1
|
|
148
|
+
else
|
|
149
|
+
accents << accent
|
|
150
|
+
last_class = accent_class
|
|
151
|
+
end
|
|
152
|
+
end
|
|
153
|
+
|
|
154
|
+
hangul_comp_one(result + start + accents)
|
|
155
|
+
end
|
|
156
|
+
|
|
157
|
+
def normalize(string, form = :nfc)
|
|
158
|
+
encoding = string.encoding
|
|
159
|
+
|
|
160
|
+
case encoding
|
|
161
|
+
when Encoding::UTF_8
|
|
162
|
+
case form
|
|
163
|
+
when :nfc
|
|
164
|
+
string.gsub(@regexp_c, @nf_hash_c)
|
|
165
|
+
when :nfd
|
|
166
|
+
string.gsub(@regexp_d, @nf_hash_d)
|
|
167
|
+
when :nfkc
|
|
168
|
+
string.gsub(@regexp_k, @kompatible_table).gsub(@regexp_c, @nf_hash_c)
|
|
169
|
+
when :nfkd
|
|
170
|
+
string.gsub(@regexp_k, @kompatible_table).gsub(@regexp_d, @nf_hash_d)
|
|
171
|
+
else
|
|
172
|
+
raise ArgumentError, "Invalid normalization form #{form}."
|
|
173
|
+
end
|
|
174
|
+
when Encoding::US_ASCII
|
|
175
|
+
string
|
|
176
|
+
when *UNICODE_ENCODINGS
|
|
177
|
+
normalize(string.encode(Encoding::UTF_8), form).encode(encoding)
|
|
178
|
+
else
|
|
179
|
+
raise Encoding::CompatibilityError, "Unicode Normalization not appropriate for #{encoding}"
|
|
180
|
+
end
|
|
181
|
+
end
|
|
182
|
+
|
|
183
|
+
def normalized?(string, form = :nfc)
|
|
184
|
+
encoding = string.encoding
|
|
185
|
+
|
|
186
|
+
case encoding
|
|
187
|
+
when Encoding::UTF_8
|
|
188
|
+
case form
|
|
189
|
+
when :nfc
|
|
190
|
+
string.scan(@regexp_c) do |match|
|
|
191
|
+
return false if @nf_hash_c[match] != match
|
|
192
|
+
end
|
|
193
|
+
true
|
|
194
|
+
when :nfd
|
|
195
|
+
string.scan(@regexp_d) do |match|
|
|
196
|
+
return false if @nf_hash_d[match] != match
|
|
197
|
+
end
|
|
198
|
+
true
|
|
199
|
+
when :nfkc
|
|
200
|
+
normalized?(string, :nfc) && string !~ @regexp_k
|
|
201
|
+
when :nfkd
|
|
202
|
+
normalized?(string, :nfd) && string !~ @regexp_k
|
|
203
|
+
else
|
|
204
|
+
raise ArgumentError, "Invalid normalization form #{form}."
|
|
205
|
+
end
|
|
206
|
+
when Encoding::US_ASCII
|
|
207
|
+
true
|
|
208
|
+
when *UNICODE_ENCODINGS
|
|
209
|
+
normalized?(string.encode(Encoding::UTF_8), form)
|
|
210
|
+
else
|
|
211
|
+
raise Encoding::CompatibilityError, "Unicode Normalization not appropriate for #{encoding}"
|
|
212
|
+
end
|
|
213
|
+
end
|
|
214
|
+
end
|
|
215
|
+
end
|
|
@@ -0,0 +1,83 @@
|
|
|
1
|
+
# encoding: utf-8
|
|
2
|
+
|
|
3
|
+
# Copyright 2010-2013 Ayumu Nojima (野島 歩) and Martin J. Dürst (duerst@it.aoyama.ac.jp)
|
|
4
|
+
# available under the same licence as Ruby itself
|
|
5
|
+
# (see http://www.ruby-lang.org/en/LICENSE.txt)
|
|
6
|
+
|
|
7
|
+
module Eprun
|
|
8
|
+
class UcdVersion
|
|
9
|
+
class << self
|
|
10
|
+
def get(str_or_version)
|
|
11
|
+
return str_or_version if str_or_version.is_a?(self)
|
|
12
|
+
|
|
13
|
+
m = str_or_version.match(/v?(\d+)\.(\d+)\.(\d+)/)
|
|
14
|
+
raise(ArgumentError, "#{str_or_version.inspect} is not a valid UCD version") unless m
|
|
15
|
+
|
|
16
|
+
parsed_version = m.captures.join(".")
|
|
17
|
+
|
|
18
|
+
if !Eprun::UCD_VERSIONS.include?(parsed_version)
|
|
19
|
+
raise ArgumentError, "Version #{parsed_version.inspect} is unsupported; please provide a version from `Eprun::UCD_VERSIONS`"
|
|
20
|
+
end
|
|
21
|
+
|
|
22
|
+
numbers = m.captures.map(&:to_i)
|
|
23
|
+
cache[numbers] ||= new(*numbers)
|
|
24
|
+
end
|
|
25
|
+
|
|
26
|
+
def latest
|
|
27
|
+
@latest ||= get(::Eprun::UCD_VERSIONS.first)
|
|
28
|
+
end
|
|
29
|
+
|
|
30
|
+
private
|
|
31
|
+
|
|
32
|
+
def cache
|
|
33
|
+
@cache ||= {}
|
|
34
|
+
end
|
|
35
|
+
end
|
|
36
|
+
|
|
37
|
+
attr_reader :major, :minor, :patch
|
|
38
|
+
|
|
39
|
+
def initialize(major, minor, patch)
|
|
40
|
+
@major = major
|
|
41
|
+
@minor = minor
|
|
42
|
+
@patch = patch
|
|
43
|
+
end
|
|
44
|
+
|
|
45
|
+
def data_path
|
|
46
|
+
@data_path ||= File.expand_path(File.join("..", "..", "data", path_string), __dir__)
|
|
47
|
+
end
|
|
48
|
+
|
|
49
|
+
def composition_exclusions_path
|
|
50
|
+
@composition_exclusions_path ||= File.join(data_path, "CompositionExclusions.txt")
|
|
51
|
+
end
|
|
52
|
+
|
|
53
|
+
def normalization_test_path
|
|
54
|
+
@normalization_test_path ||= File.join(data_path, "NormalizationTest.txt")
|
|
55
|
+
end
|
|
56
|
+
|
|
57
|
+
def unicode_data_path
|
|
58
|
+
@unicode_data_path ||= File.join(data_path, "UnicodeData.txt")
|
|
59
|
+
end
|
|
60
|
+
|
|
61
|
+
def string
|
|
62
|
+
@string ||= "v#{bare_string}"
|
|
63
|
+
end
|
|
64
|
+
|
|
65
|
+
def bare_string
|
|
66
|
+
@bare_string ||= "#{major}.#{minor}.#{patch}"
|
|
67
|
+
end
|
|
68
|
+
|
|
69
|
+
def path_string
|
|
70
|
+
@path_string ||= "v#{major}_#{minor}_#{patch}"
|
|
71
|
+
end
|
|
72
|
+
|
|
73
|
+
alias underscored_string path_string
|
|
74
|
+
|
|
75
|
+
def module_string
|
|
76
|
+
@module_string ||= "V#{major}_#{minor}_#{patch}"
|
|
77
|
+
end
|
|
78
|
+
|
|
79
|
+
def namespace
|
|
80
|
+
@namespace ||= ::Eprun.const_get(module_string)
|
|
81
|
+
end
|
|
82
|
+
end
|
|
83
|
+
end
|