psychowl 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +7 -0
- data/CHANGELOG.md +21 -0
- data/LICENSE.txt +21 -0
- data/README.md +212 -0
- data/exe/psychowl +106 -0
- data/ext/psychowl_native/core/data/LICENSE-whatlang +26 -0
- data/ext/psychowl_native/core/data/README.md +32 -0
- data/ext/psychowl_native/core/data/alphabets.tsv +16 -0
- data/ext/psychowl_native/core/data/languages.tsv +20 -0
- data/ext/psychowl_native/core/data/scripts.tsv +28 -0
- data/ext/psychowl_native/core/data/trigrams/ara.txt +302 -0
- data/ext/psychowl_native/core/data/trigrams/deu.txt +302 -0
- data/ext/psychowl_native/core/data/trigrams/eng.txt +302 -0
- data/ext/psychowl_native/core/data/trigrams/fra.txt +302 -0
- data/ext/psychowl_native/core/data/trigrams/hin.txt +302 -0
- data/ext/psychowl_native/core/data/trigrams/ind.txt +302 -0
- data/ext/psychowl_native/core/data/trigrams/ita.txt +302 -0
- data/ext/psychowl_native/core/data/trigrams/nld.txt +302 -0
- data/ext/psychowl_native/core/data/trigrams/pol.txt +302 -0
- data/ext/psychowl_native/core/data/trigrams/por.txt +302 -0
- data/ext/psychowl_native/core/data/trigrams/rus.txt +302 -0
- data/ext/psychowl_native/core/data/trigrams/spa.txt +302 -0
- data/ext/psychowl_native/core/data/trigrams/tur.txt +302 -0
- data/ext/psychowl_native/core/data/trigrams/ukr.txt +302 -0
- data/ext/psychowl_native/core/data/trigrams/vie.txt +302 -0
- data/lib/psychowl/active_model.rb +64 -0
- data/lib/psychowl/data_file.rb +14 -0
- data/lib/psychowl/detector.rb +123 -0
- data/lib/psychowl/engine.rb +23 -0
- data/lib/psychowl/info.rb +39 -0
- data/lib/psychowl/lang.rb +55 -0
- data/lib/psychowl/locale/en.yml +6 -0
- data/lib/psychowl/lookup.rb +25 -0
- data/lib/psychowl/native_speedup.rb +46 -0
- data/lib/psychowl/postgres.rb +35 -0
- data/lib/psychowl/railtie.rb +10 -0
- data/lib/psychowl/ruby_engine/segmentation.rb +84 -0
- data/lib/psychowl/ruby_engine/trigrams.rb +74 -0
- data/lib/psychowl/ruby_engine.rb +166 -0
- data/lib/psychowl/script.rb +39 -0
- data/lib/psychowl/segment.rb +24 -0
- data/lib/psychowl/tables.rb +55 -0
- data/lib/psychowl/text.rb +26 -0
- data/lib/psychowl/version.rb +5 -0
- data/lib/psychowl.rb +82 -0
- metadata +91 -0
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Psychowl
|
|
4
|
+
# PostgreSQL full-text search configurations for supported languages, for
|
|
5
|
+
# to_tsvector / to_tsquery. Languages without a built-in stemmer map to
|
|
6
|
+
# "simple".
|
|
7
|
+
#
|
|
8
|
+
# config = Psychowl.detect(body)&.lang&.pg_regconfig || "simple"
|
|
9
|
+
# Post.where("to_tsvector(?::regconfig, body) @@ plainto_tsquery(?::regconfig, ?)", config, config, query)
|
|
10
|
+
module Postgres
|
|
11
|
+
REGCONFIGS = {
|
|
12
|
+
'ara' => 'arabic',
|
|
13
|
+
'deu' => 'german',
|
|
14
|
+
'eng' => 'english',
|
|
15
|
+
'fra' => 'french',
|
|
16
|
+
'ind' => 'indonesian',
|
|
17
|
+
'ita' => 'italian',
|
|
18
|
+
'nld' => 'dutch',
|
|
19
|
+
'por' => 'portuguese',
|
|
20
|
+
'rus' => 'russian',
|
|
21
|
+
'spa' => 'spanish',
|
|
22
|
+
'tur' => 'turkish'
|
|
23
|
+
}.freeze
|
|
24
|
+
|
|
25
|
+
# @param lang [Lang, String, Symbol]
|
|
26
|
+
# @return [String] a built-in PostgreSQL text search configuration
|
|
27
|
+
def self.regconfig(lang) = REGCONFIGS.fetch(Lang.fetch(lang).code, 'simple')
|
|
28
|
+
end
|
|
29
|
+
|
|
30
|
+
# PostgreSQL text search configuration per language.
|
|
31
|
+
class Lang
|
|
32
|
+
# @return [String] PostgreSQL text search configuration, see {Postgres}
|
|
33
|
+
def pg_regconfig = Postgres.regconfig(self)
|
|
34
|
+
end
|
|
35
|
+
end
|
|
@@ -0,0 +1,84 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Psychowl
|
|
4
|
+
# Per-sentence detection, mirrored by segment.rs.
|
|
5
|
+
module RubyEngine
|
|
6
|
+
# Sentence-ending marks that need a following space ("3.14" is no boundary).
|
|
7
|
+
SENTENCE_MARKS = %W[. ! ? \u2026].freeze
|
|
8
|
+
# CJK sentence-ending marks, no space needed.
|
|
9
|
+
CLOSING_MARKS = %W[\u3002 \uFF01 \uFF1F].freeze
|
|
10
|
+
|
|
11
|
+
class << self
|
|
12
|
+
# Runs of same-language sentences. Sentences without a language join the
|
|
13
|
+
# previous run (or the first). Each run is re-detected whole and keeps its
|
|
14
|
+
# first sentence's result if that changes the language.
|
|
15
|
+
#
|
|
16
|
+
# @return [Array<Array(Integer, Integer, Integer, Integer, Float)>]
|
|
17
|
+
# [[start, end, language, script, confidence], ...] with character
|
|
18
|
+
# offsets, end exclusive
|
|
19
|
+
def segments(text, filter_mode, filter_langs)
|
|
20
|
+
runs = [] # [start, end, first sentence's detection]
|
|
21
|
+
leading_start = nil
|
|
22
|
+
|
|
23
|
+
sentences(text).each do |start, finish|
|
|
24
|
+
result = detect(text[start...finish], filter_mode, filter_langs)
|
|
25
|
+
|
|
26
|
+
if result.nil?
|
|
27
|
+
if runs.empty?
|
|
28
|
+
leading_start ||= start
|
|
29
|
+
else
|
|
30
|
+
runs.last[1] = finish
|
|
31
|
+
end
|
|
32
|
+
elsif runs.any? && runs.last[2][0] == result[0]
|
|
33
|
+
runs.last[1] = finish
|
|
34
|
+
else
|
|
35
|
+
runs << [leading_start || start, finish, result]
|
|
36
|
+
leading_start = nil
|
|
37
|
+
end
|
|
38
|
+
end
|
|
39
|
+
|
|
40
|
+
runs.map do |start, finish, first|
|
|
41
|
+
whole = detect(text[start...finish], filter_mode, filter_langs)
|
|
42
|
+
whole && whole[0] == first[0] ? [start, finish, *whole] : [start, finish, *first]
|
|
43
|
+
end
|
|
44
|
+
end
|
|
45
|
+
|
|
46
|
+
# [start, end] character offsets. A sentence ends after a newline, 。!?,
|
|
47
|
+
# or . ! ? … followed by a space; trailing spaces stay with it.
|
|
48
|
+
def sentences(text)
|
|
49
|
+
chars = text.chars
|
|
50
|
+
sentences = []
|
|
51
|
+
start = 0
|
|
52
|
+
index = 0
|
|
53
|
+
|
|
54
|
+
while index < chars.size
|
|
55
|
+
char = chars[index]
|
|
56
|
+
boundary = false
|
|
57
|
+
|
|
58
|
+
if char == "\n" || CLOSING_MARKS.include?(char)
|
|
59
|
+
index += 1
|
|
60
|
+
boundary = true
|
|
61
|
+
elsif SENTENCE_MARKS.include?(char)
|
|
62
|
+
index += 1 while index < chars.size && SENTENCE_MARKS.include?(chars[index])
|
|
63
|
+
boundary = index == chars.size || space?(chars[index])
|
|
64
|
+
else
|
|
65
|
+
index += 1
|
|
66
|
+
end
|
|
67
|
+
next unless boundary
|
|
68
|
+
|
|
69
|
+
index += 1 while index < chars.size && space?(chars[index])
|
|
70
|
+
sentences << [start, index]
|
|
71
|
+
start = index
|
|
72
|
+
end
|
|
73
|
+
|
|
74
|
+
sentences << [start, chars.size] if start < chars.size
|
|
75
|
+
sentences
|
|
76
|
+
end
|
|
77
|
+
|
|
78
|
+
private
|
|
79
|
+
|
|
80
|
+
# ASCII whitespace and controls, no-break space, ideographic space.
|
|
81
|
+
def space?(char) = char <= ' ' || char == "\u00A0" || char == "\u3000"
|
|
82
|
+
end
|
|
83
|
+
end
|
|
84
|
+
end
|
|
@@ -0,0 +1,74 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Psychowl
|
|
4
|
+
# Trigram model, mirrored by trigram.rs.
|
|
5
|
+
module RubyEngine
|
|
6
|
+
# A profile trigram missing from the text costs this much distance.
|
|
7
|
+
MAX_TRIGRAM_DISTANCE = 300
|
|
8
|
+
MAX_TOTAL_DISTANCE = MAX_TRIGRAM_DISTANCE * MAX_TRIGRAM_DISTANCE
|
|
9
|
+
# Only the most frequent trigrams of the text are compared.
|
|
10
|
+
TEXT_TRIGRAMS_SIZE = 600
|
|
11
|
+
|
|
12
|
+
class << self
|
|
13
|
+
private
|
|
14
|
+
|
|
15
|
+
# Similarity between the text's trigram ranking and each profile.
|
|
16
|
+
def trigram_scores(lowercase, languages)
|
|
17
|
+
positions = trigram_positions(lowercase)
|
|
18
|
+
max_distance = positions.size * MAX_TRIGRAM_DISTANCE
|
|
19
|
+
|
|
20
|
+
scores = languages.to_h do |language|
|
|
21
|
+
profile = Tables::TRIGRAMS[language]
|
|
22
|
+
if profile
|
|
23
|
+
distance = trigram_distance(profile, positions)
|
|
24
|
+
[language, (max_distance - distance).to_f / max_distance]
|
|
25
|
+
else
|
|
26
|
+
[language, 0.0]
|
|
27
|
+
end
|
|
28
|
+
end
|
|
29
|
+
[scores, positions.size]
|
|
30
|
+
end
|
|
31
|
+
|
|
32
|
+
# {trigram => rank}: the text's most frequent trigrams, most frequent
|
|
33
|
+
# first (ties broken by the trigram itself, descending).
|
|
34
|
+
def trigram_positions(lowercase)
|
|
35
|
+
ranked = count_trigrams(lowercase).sort do |(trigram_a, count_a), (trigram_b, count_b)|
|
|
36
|
+
[count_b, trigram_b] <=> [count_a, trigram_a]
|
|
37
|
+
end
|
|
38
|
+
ranked.first(TEXT_TRIGRAMS_SIZE).each_with_index.to_h { |(trigram, _count), rank| [trigram, rank] }
|
|
39
|
+
end
|
|
40
|
+
|
|
41
|
+
# {trigram => occurrences}. Punctuation and digits act as spaces;
|
|
42
|
+
# trigrams that are only word boundary are skipped.
|
|
43
|
+
def count_trigrams(lowercase)
|
|
44
|
+
counts = Hash.new(0)
|
|
45
|
+
first = ' '
|
|
46
|
+
second = nil
|
|
47
|
+
|
|
48
|
+
"#{lowercase.tr(Tables::STOP_CHARS, ' ')} ".each_char do |third|
|
|
49
|
+
if second.nil?
|
|
50
|
+
second = third
|
|
51
|
+
next
|
|
52
|
+
end
|
|
53
|
+
|
|
54
|
+
counts["#{first}#{second}#{third}"] += 1 unless second == ' ' && (first == ' ' || third == ' ')
|
|
55
|
+
first = second
|
|
56
|
+
second = third
|
|
57
|
+
end
|
|
58
|
+
counts
|
|
59
|
+
end
|
|
60
|
+
|
|
61
|
+
def trigram_distance(profile, positions)
|
|
62
|
+
total = 0
|
|
63
|
+
profile.each_with_index do |trigram, index|
|
|
64
|
+
position = positions[trigram]
|
|
65
|
+
total += position ? (position - index).abs : MAX_TRIGRAM_DISTANCE
|
|
66
|
+
end
|
|
67
|
+
|
|
68
|
+
unique = positions.size
|
|
69
|
+
total -= (MAX_TRIGRAM_DISTANCE - unique) * MAX_TRIGRAM_DISTANCE if unique < MAX_TRIGRAM_DISTANCE
|
|
70
|
+
total.clamp(0, MAX_TOTAL_DISTANCE)
|
|
71
|
+
end
|
|
72
|
+
end
|
|
73
|
+
end
|
|
74
|
+
end
|
|
@@ -0,0 +1,166 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Psychowl
|
|
4
|
+
# Pure Ruby engine, mirrored by the psychowl crate; parity_test.rb keeps them
|
|
5
|
+
# identical. Primitives in and out: languages and scripts are Tables indices,
|
|
6
|
+
# text is valid UTF-8, a filter is an Engine::FILTER_* mode plus indices.
|
|
7
|
+
# Internal calls never go through Engine.
|
|
8
|
+
module RubyEngine
|
|
9
|
+
class << self
|
|
10
|
+
# @return [Array(Integer, Integer, Float), nil] [language, script, confidence]
|
|
11
|
+
def detect(text, filter_mode, filter_langs)
|
|
12
|
+
case evaluate(text, filter_mode, filter_langs)
|
|
13
|
+
in [:decided, result] then result
|
|
14
|
+
in [:scored, script, [[best, best_score], [_, runner_up_score], *], trigram_count]
|
|
15
|
+
[best, script, confidence(best_score, runner_up_score, trigram_count)]
|
|
16
|
+
in [:nothing] then nil
|
|
17
|
+
end
|
|
18
|
+
end
|
|
19
|
+
|
|
20
|
+
# Every candidate language with its score, best first.
|
|
21
|
+
#
|
|
22
|
+
# @return [Array<Array(Integer, Float)>] [[language, score], ...]
|
|
23
|
+
def candidates(text, filter_mode, filter_langs)
|
|
24
|
+
case evaluate(text, filter_mode, filter_langs)
|
|
25
|
+
in [:decided, [language, _script, confidence]] then [[language, confidence]]
|
|
26
|
+
in [:scored, _script, scores, _trigram_count] then scores
|
|
27
|
+
in [:nothing] then []
|
|
28
|
+
end
|
|
29
|
+
end
|
|
30
|
+
|
|
31
|
+
# @return [Integer, nil] the script with the most characters
|
|
32
|
+
def detect_script(text) = main_script(script_counts(text))
|
|
33
|
+
|
|
34
|
+
# @return [Array<Integer>] number of characters per script, by script index
|
|
35
|
+
def script_counts(text)
|
|
36
|
+
Tables::SCRIPTS.map { text.count(it.charset) }
|
|
37
|
+
end
|
|
38
|
+
|
|
39
|
+
# @return [Array<Array(Integer, Integer, Float), nil>] one #detect result per text
|
|
40
|
+
def detect_many(texts, filter_mode, filter_langs)
|
|
41
|
+
texts.map { detect(it, filter_mode, filter_langs) }
|
|
42
|
+
end
|
|
43
|
+
|
|
44
|
+
private
|
|
45
|
+
|
|
46
|
+
# Mirrors the Rust `Outcome`:
|
|
47
|
+
# [:nothing] no letters / no allowed language
|
|
48
|
+
# [:decided, [language, script, confidence]] one allowed language, or Han
|
|
49
|
+
# [:scored, script, scores, trigram_count] several languages, best first
|
|
50
|
+
def evaluate(text, filter_mode, filter_langs)
|
|
51
|
+
counts = script_counts(text)
|
|
52
|
+
script = main_script(counts)
|
|
53
|
+
return [:nothing] if script.nil?
|
|
54
|
+
|
|
55
|
+
if script == Tables::HAN
|
|
56
|
+
result = detect_han(counts, filter_mode, filter_langs)
|
|
57
|
+
return result ? [:decided, result] : [:nothing]
|
|
58
|
+
end
|
|
59
|
+
|
|
60
|
+
case allowed(Tables::SCRIPTS[script].languages, filter_mode, filter_langs)
|
|
61
|
+
in [] then [:nothing]
|
|
62
|
+
in [only] then [:decided, [only, script, 1.0]]
|
|
63
|
+
in languages then [:scored, script, *language_scores(text, languages)]
|
|
64
|
+
end
|
|
65
|
+
end
|
|
66
|
+
|
|
67
|
+
# Ties go to the script listed first in scripts.tsv.
|
|
68
|
+
def main_script(counts)
|
|
69
|
+
best = nil
|
|
70
|
+
counts.each_with_index do |count, index|
|
|
71
|
+
best = index if count.positive? && (best.nil? || count > counts[best])
|
|
72
|
+
end
|
|
73
|
+
best
|
|
74
|
+
end
|
|
75
|
+
|
|
76
|
+
def allowed(languages, filter_mode, filter_langs)
|
|
77
|
+
case filter_mode
|
|
78
|
+
when Engine::FILTER_ALLOW then languages.select { filter_langs.include?(it) }
|
|
79
|
+
when Engine::FILTER_DENY then languages.reject { filter_langs.include?(it) }
|
|
80
|
+
else languages
|
|
81
|
+
end
|
|
82
|
+
end
|
|
83
|
+
|
|
84
|
+
# Han characters are shared by Chinese and Japanese; kana tips it to
|
|
85
|
+
# Japanese.
|
|
86
|
+
def detect_han(counts, filter_mode, filter_langs)
|
|
87
|
+
languages = allowed(Tables::SCRIPTS[Tables::HAN].languages, filter_mode, filter_langs)
|
|
88
|
+
mandarin = languages.include?(Tables::CMN)
|
|
89
|
+
japanese = languages.include?(Tables::JPN)
|
|
90
|
+
|
|
91
|
+
case [mandarin, japanese]
|
|
92
|
+
in [false, false] then return nil
|
|
93
|
+
in [true, false] then return [Tables::CMN, Tables::HAN, 1.0]
|
|
94
|
+
in [false, true] then return [Tables::JPN, Tables::HAN, 1.0]
|
|
95
|
+
in [true, true] then nil
|
|
96
|
+
end
|
|
97
|
+
|
|
98
|
+
kana = counts[Tables::HIRAGANA] + counts[Tables::KATAKANA]
|
|
99
|
+
kana_ratio = kana.to_f / (counts[Tables::HAN] + kana)
|
|
100
|
+
|
|
101
|
+
if kana_ratio > 0.2 then [Tables::JPN, Tables::HAN, 1.0]
|
|
102
|
+
elsif kana_ratio > 0.05 then [Tables::JPN, Tables::HAN, 0.5]
|
|
103
|
+
elsif kana_ratio > 0.02 then [Tables::CMN, Tables::HAN, 0.5]
|
|
104
|
+
else [Tables::CMN, Tables::HAN, 1.0]
|
|
105
|
+
end
|
|
106
|
+
end
|
|
107
|
+
|
|
108
|
+
# @return [Array(Array<Array(Integer, Float)>, Integer)] scores best first, trigram count
|
|
109
|
+
def language_scores(text, languages)
|
|
110
|
+
lowercase = text.downcase
|
|
111
|
+
alphabet_scores, char_count = alphabet_scores(lowercase, languages)
|
|
112
|
+
trigram_scores, trigram_count = trigram_scores(lowercase, languages)
|
|
113
|
+
|
|
114
|
+
alphabet_weight = alphabet_weight(char_count)
|
|
115
|
+
trigram_weight = 1.0 - alphabet_weight
|
|
116
|
+
|
|
117
|
+
scores = languages.map do |language|
|
|
118
|
+
score = (alphabet_scores[language] * alphabet_weight) + (trigram_scores[language] * trigram_weight)
|
|
119
|
+
[language, score]
|
|
120
|
+
end
|
|
121
|
+
[sort_by_score(scores), trigram_count]
|
|
122
|
+
end
|
|
123
|
+
|
|
124
|
+
# 2/3 for an empty text, falling to 1/3 at 100 letters and beyond.
|
|
125
|
+
def alphabet_weight(char_count) = (-(char_count / 300.0) + (2.0 / 3.0)).clamp(1.0 / 3.0, 2.0 / 3.0)
|
|
126
|
+
|
|
127
|
+
# Highest score first; ties go to the lower language index.
|
|
128
|
+
def sort_by_score(scores)
|
|
129
|
+
scores.sort do |(language_a, score_a), (language_b, score_b)|
|
|
130
|
+
order = score_b <=> score_a
|
|
131
|
+
order.zero? ? language_a <=> language_b : order
|
|
132
|
+
end
|
|
133
|
+
end
|
|
134
|
+
|
|
135
|
+
# Share of letters that belong to each language's alphabet. A letter
|
|
136
|
+
# outside the alphabet counts against the language.
|
|
137
|
+
def alphabet_scores(lowercase, languages)
|
|
138
|
+
char_count = lowercase.length - lowercase.count(Tables::STOP_CHARS)
|
|
139
|
+
|
|
140
|
+
unless languages.any? { Tables::ALPHABETS.key?(it) }
|
|
141
|
+
return [languages.to_h { [it, 1.0] }, 1]
|
|
142
|
+
end
|
|
143
|
+
|
|
144
|
+
scores = languages.to_h do |language|
|
|
145
|
+
letters = Tables::ALPHABETS[language]
|
|
146
|
+
hits = letters ? lowercase.count(letters) : 0
|
|
147
|
+
raw = (2 * hits) - char_count
|
|
148
|
+
raw = 0 if raw.negative?
|
|
149
|
+
[language, raw.to_f / char_count]
|
|
150
|
+
end
|
|
151
|
+
[scores, char_count]
|
|
152
|
+
end
|
|
153
|
+
|
|
154
|
+
# How far the best score is ahead of the runner-up, scaled by how much
|
|
155
|
+
# evidence (trigrams) there was.
|
|
156
|
+
def confidence(best, runner_up, count)
|
|
157
|
+
return 0.0 if best.zero?
|
|
158
|
+
return best if runner_up.zero?
|
|
159
|
+
|
|
160
|
+
confident_rate = (3.0 / count) + 0.015
|
|
161
|
+
rate = (best - runner_up) / runner_up
|
|
162
|
+
rate > confident_rate ? 1.0 : rate / confident_rate
|
|
163
|
+
end
|
|
164
|
+
end
|
|
165
|
+
end
|
|
166
|
+
end
|
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Psychowl
|
|
4
|
+
# A writing system, e.g. Latin or Cyrillic. Instances are canonical.
|
|
5
|
+
#
|
|
6
|
+
# @!attribute [r] name
|
|
7
|
+
# @return [String] e.g. "Cyrillic"
|
|
8
|
+
# @!attribute [r] langs
|
|
9
|
+
# @return [Array<Lang>] supported languages written in this script
|
|
10
|
+
Script = ::Data.define(:name, :langs) do
|
|
11
|
+
# @return [String] the script name
|
|
12
|
+
def to_s = name
|
|
13
|
+
|
|
14
|
+
def inspect = "#<#{self.class} #{name}>"
|
|
15
|
+
end
|
|
16
|
+
|
|
17
|
+
# The canonical instances and lookup by name.
|
|
18
|
+
class Script
|
|
19
|
+
singleton_class.prepend(Lookup) # ahead of Data.define's own `[]` constructor
|
|
20
|
+
|
|
21
|
+
ALL = Tables::SCRIPTS.map do |script|
|
|
22
|
+
new(name: script.name, langs: script.languages.map { Lang.at(it) }.freeze)
|
|
23
|
+
end.freeze
|
|
24
|
+
BY_NAME = ALL.to_h { |script| [script.name.downcase, script] }.freeze
|
|
25
|
+
private_constant :ALL, :BY_NAME
|
|
26
|
+
|
|
27
|
+
class << self
|
|
28
|
+
# @return [Array<Script>] every known script (frozen)
|
|
29
|
+
def all = ALL
|
|
30
|
+
|
|
31
|
+
private :new
|
|
32
|
+
|
|
33
|
+
private
|
|
34
|
+
|
|
35
|
+
def by_key = BY_NAME
|
|
36
|
+
def unknown_key_message = 'unknown script'
|
|
37
|
+
end
|
|
38
|
+
end
|
|
39
|
+
end
|
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Psychowl
|
|
4
|
+
# A stretch of text in one language, from {Psychowl.segments}.
|
|
5
|
+
#
|
|
6
|
+
# @!attribute [r] text
|
|
7
|
+
# @return [String] the segment's text
|
|
8
|
+
# @!attribute [r] range
|
|
9
|
+
# @return [Range] character offsets in the original text
|
|
10
|
+
# @!attribute [r] info
|
|
11
|
+
# @return [Info] the detection for this segment
|
|
12
|
+
Segment = ::Data.define(:text, :range, :info) do
|
|
13
|
+
# @return [Lang]
|
|
14
|
+
def lang = info.lang
|
|
15
|
+
|
|
16
|
+
# @return [Script]
|
|
17
|
+
def script = info.script
|
|
18
|
+
|
|
19
|
+
# @return [Float]
|
|
20
|
+
def confidence = info.confidence
|
|
21
|
+
|
|
22
|
+
def inspect = "#<#{self.class} #{range} #{info.lang.code} #{text[0, 30].inspect}>"
|
|
23
|
+
end
|
|
24
|
+
end
|
|
@@ -0,0 +1,55 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Psychowl
|
|
4
|
+
# The data tables shared with the Rust crate, read once at load time.
|
|
5
|
+
# Languages and scripts are referred to by line index, as in the crate.
|
|
6
|
+
module Tables
|
|
7
|
+
DIR = File.expand_path('../../ext/psychowl_native/core/data', __dir__)
|
|
8
|
+
|
|
9
|
+
# ASCII controls, spaces, digits and punctuation, as a String#count set.
|
|
10
|
+
STOP_CHARS = "\u0000-@[-`{-~"
|
|
11
|
+
|
|
12
|
+
class << self
|
|
13
|
+
def rows(file) = DataFile.lines(File.join(DIR, file)).map { it.split("\t") }
|
|
14
|
+
|
|
15
|
+
# Escapes the characters String#count treats specially.
|
|
16
|
+
def charset(chars) = chars.gsub(/[\\^-]/) { |char| "\\#{char}" }
|
|
17
|
+
|
|
18
|
+
def range_charset(ranges)
|
|
19
|
+
ranges.split(',').map do |range|
|
|
20
|
+
low, high = range.split('-').map { it.to_i(16).chr(Encoding::UTF_8) }
|
|
21
|
+
high ? "#{charset(low)}-#{charset(high)}" : charset(low)
|
|
22
|
+
end.join
|
|
23
|
+
end
|
|
24
|
+
end
|
|
25
|
+
|
|
26
|
+
# [[code, iso639_1, eng_name, name], ...]
|
|
27
|
+
LANGUAGES = rows('languages.tsv').map(&:freeze).freeze
|
|
28
|
+
LANGUAGE_INDEX = LANGUAGES.each_with_index.to_h { |(code, *), index| [code, index] }.freeze
|
|
29
|
+
|
|
30
|
+
Script = ::Data.define(:name, :languages, :charset)
|
|
31
|
+
# [Script(name, [language index, ...], String#count set), ...]
|
|
32
|
+
SCRIPTS = rows('scripts.tsv').map do |name, codes, ranges|
|
|
33
|
+
languages = codes == '-' ? [] : codes.split(',').map { LANGUAGE_INDEX.fetch(it) }
|
|
34
|
+
Script.new(name: name.freeze, languages: languages.freeze, charset: range_charset(ranges).freeze)
|
|
35
|
+
end.freeze
|
|
36
|
+
SCRIPT_INDEX = SCRIPTS.each_with_index.to_h { |script, index| [script.name, index] }.freeze
|
|
37
|
+
|
|
38
|
+
# {language index => String#count set of its letters}
|
|
39
|
+
ALPHABETS = rows('alphabets.tsv').to_h do |code, letters|
|
|
40
|
+
[LANGUAGE_INDEX.fetch(code), charset(letters).freeze]
|
|
41
|
+
end.freeze
|
|
42
|
+
|
|
43
|
+
# {language index => [trigram, ...]} most frequent first, " " for a word boundary.
|
|
44
|
+
TRIGRAMS = Dir[File.join(DIR, 'trigrams', '*.txt')].to_h do |path|
|
|
45
|
+
code = File.basename(path, '.txt')
|
|
46
|
+
[LANGUAGE_INDEX.fetch(code), DataFile.lines(path).map { it.tr('_', ' ').freeze }.freeze]
|
|
47
|
+
end.freeze
|
|
48
|
+
|
|
49
|
+
HAN = SCRIPT_INDEX.fetch('Han')
|
|
50
|
+
HIRAGANA = SCRIPT_INDEX.fetch('Hiragana')
|
|
51
|
+
KATAKANA = SCRIPT_INDEX.fetch('Katakana')
|
|
52
|
+
CMN = LANGUAGE_INDEX['cmn']
|
|
53
|
+
JPN = LANGUAGE_INDEX['jpn']
|
|
54
|
+
end
|
|
55
|
+
end
|
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Psychowl
|
|
4
|
+
# Input validation shared by both engines: they only ever see valid UTF-8.
|
|
5
|
+
module Text
|
|
6
|
+
class << self
|
|
7
|
+
# @param text [String, #to_str]
|
|
8
|
+
# @return [String] valid UTF-8
|
|
9
|
+
# @raise [TypeError] not a String
|
|
10
|
+
# @raise [EncodingError] invalid bytes, or not convertible to UTF-8
|
|
11
|
+
def prepare(text)
|
|
12
|
+
string = String.try_convert(text)
|
|
13
|
+
raise TypeError, "no implicit conversion of #{text.nil? ? 'nil' : text.class} into String" unless string
|
|
14
|
+
|
|
15
|
+
utf8 =
|
|
16
|
+
case string.encoding
|
|
17
|
+
when Encoding::UTF_8, Encoding::US_ASCII, Encoding::BINARY then string.dup.force_encoding(Encoding::UTF_8)
|
|
18
|
+
else string.encode(Encoding::UTF_8)
|
|
19
|
+
end
|
|
20
|
+
raise EncodingError, 'text is not valid UTF-8' unless utf8.valid_encoding?
|
|
21
|
+
|
|
22
|
+
utf8
|
|
23
|
+
end
|
|
24
|
+
end
|
|
25
|
+
end
|
|
26
|
+
end
|
data/lib/psychowl.rb
ADDED
|
@@ -0,0 +1,82 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require_relative 'psychowl/version'
|
|
4
|
+
require_relative 'psychowl/data_file'
|
|
5
|
+
require_relative 'psychowl/tables'
|
|
6
|
+
require_relative 'psychowl/text'
|
|
7
|
+
require_relative 'psychowl/engine'
|
|
8
|
+
require_relative 'psychowl/ruby_engine'
|
|
9
|
+
require_relative 'psychowl/ruby_engine/segmentation'
|
|
10
|
+
require_relative 'psychowl/ruby_engine/trigrams'
|
|
11
|
+
require_relative 'psychowl/lookup'
|
|
12
|
+
require_relative 'psychowl/lang'
|
|
13
|
+
require_relative 'psychowl/script'
|
|
14
|
+
require_relative 'psychowl/info'
|
|
15
|
+
require_relative 'psychowl/segment'
|
|
16
|
+
require_relative 'psychowl/detector'
|
|
17
|
+
|
|
18
|
+
# Natural language and script detection. It knows what language you speak.
|
|
19
|
+
# It knows you skipped your lesson.
|
|
20
|
+
#
|
|
21
|
+
# info = Psychowl.detect("Die Deutsche Bahn ist heute pünktlich. Wir ermitteln.")
|
|
22
|
+
# info.lang.code # => "deu"
|
|
23
|
+
# info.script.name # => "Latin"
|
|
24
|
+
module Psychowl
|
|
25
|
+
DEFAULT_DETECTOR = Detector.new
|
|
26
|
+
private_constant :DEFAULT_DETECTOR
|
|
27
|
+
|
|
28
|
+
class << self
|
|
29
|
+
# @param text [String]
|
|
30
|
+
# @param allowlist [Array<Lang, String, Symbol>, nil] only consider these
|
|
31
|
+
# @param denylist [Array<Lang, String, Symbol>, nil] never consider these
|
|
32
|
+
# @return [Info, nil] nil when no supported language was found
|
|
33
|
+
# @raise [TypeError] text is not a String
|
|
34
|
+
# @raise [EncodingError] text is not valid UTF-8 (or convertible to it)
|
|
35
|
+
# @raise [ArgumentError] invalid allowlist/denylist, see {Detector#initialize}
|
|
36
|
+
def detect(text, allowlist: nil, denylist: nil) = detector(allowlist, denylist).detect(text)
|
|
37
|
+
|
|
38
|
+
# @param (see .detect)
|
|
39
|
+
# @return [Lang, nil]
|
|
40
|
+
def detect_lang(text, allowlist: nil, denylist: nil) = detector(allowlist, denylist).detect_lang(text)
|
|
41
|
+
|
|
42
|
+
# @param text [String]
|
|
43
|
+
# @return [Script, nil]
|
|
44
|
+
def detect_script(text) = DEFAULT_DETECTOR.detect_script(text)
|
|
45
|
+
|
|
46
|
+
# @param (see .detect)
|
|
47
|
+
# @param limit [Integer, nil]
|
|
48
|
+
# @return [Array<Array(Lang, Float)>] see {Detector#candidates}
|
|
49
|
+
def candidates(text, limit: nil, allowlist: nil, denylist: nil)
|
|
50
|
+
detector(allowlist, denylist).candidates(text, limit:)
|
|
51
|
+
end
|
|
52
|
+
|
|
53
|
+
# @param texts [Array<String>]
|
|
54
|
+
# @param allowlist [Array<Lang, String, Symbol>, nil]
|
|
55
|
+
# @param denylist [Array<Lang, String, Symbol>, nil]
|
|
56
|
+
# @return [Array<Info, nil>] see {Detector#detect_many}
|
|
57
|
+
def detect_many(texts, allowlist: nil, denylist: nil) = detector(allowlist, denylist).detect_many(texts)
|
|
58
|
+
|
|
59
|
+
# @param (see .detect)
|
|
60
|
+
# @return [Array<Segment>] see {Detector#segments}
|
|
61
|
+
def segments(text, allowlist: nil, denylist: nil) = detector(allowlist, denylist).segments(text)
|
|
62
|
+
|
|
63
|
+
# @param text [String]
|
|
64
|
+
# @return [Hash{Script => Float}] see {Detector#scripts}
|
|
65
|
+
def scripts(text) = DEFAULT_DETECTOR.scripts(text)
|
|
66
|
+
|
|
67
|
+
# @return [Symbol] :native when the Rust extension is loaded, :ruby otherwise
|
|
68
|
+
def backend = Engine.native? ? :native : :ruby
|
|
69
|
+
|
|
70
|
+
private
|
|
71
|
+
|
|
72
|
+
def detector(allowlist, denylist)
|
|
73
|
+
return DEFAULT_DETECTOR if allowlist.nil? && denylist.nil?
|
|
74
|
+
|
|
75
|
+
Detector.new(allowlist:, denylist:)
|
|
76
|
+
end
|
|
77
|
+
end
|
|
78
|
+
end
|
|
79
|
+
|
|
80
|
+
require_relative 'psychowl/postgres'
|
|
81
|
+
require_relative 'psychowl/native_speedup'
|
|
82
|
+
require_relative 'psychowl/railtie' if defined?(Rails::Railtie)
|