psychowl 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (46) hide show
  1. checksums.yaml +7 -0
  2. data/CHANGELOG.md +21 -0
  3. data/LICENSE.txt +21 -0
  4. data/README.md +212 -0
  5. data/exe/psychowl +106 -0
  6. data/ext/psychowl_native/core/data/LICENSE-whatlang +26 -0
  7. data/ext/psychowl_native/core/data/README.md +32 -0
  8. data/ext/psychowl_native/core/data/alphabets.tsv +16 -0
  9. data/ext/psychowl_native/core/data/languages.tsv +20 -0
  10. data/ext/psychowl_native/core/data/scripts.tsv +28 -0
  11. data/ext/psychowl_native/core/data/trigrams/ara.txt +302 -0
  12. data/ext/psychowl_native/core/data/trigrams/deu.txt +302 -0
  13. data/ext/psychowl_native/core/data/trigrams/eng.txt +302 -0
  14. data/ext/psychowl_native/core/data/trigrams/fra.txt +302 -0
  15. data/ext/psychowl_native/core/data/trigrams/hin.txt +302 -0
  16. data/ext/psychowl_native/core/data/trigrams/ind.txt +302 -0
  17. data/ext/psychowl_native/core/data/trigrams/ita.txt +302 -0
  18. data/ext/psychowl_native/core/data/trigrams/nld.txt +302 -0
  19. data/ext/psychowl_native/core/data/trigrams/pol.txt +302 -0
  20. data/ext/psychowl_native/core/data/trigrams/por.txt +302 -0
  21. data/ext/psychowl_native/core/data/trigrams/rus.txt +302 -0
  22. data/ext/psychowl_native/core/data/trigrams/spa.txt +302 -0
  23. data/ext/psychowl_native/core/data/trigrams/tur.txt +302 -0
  24. data/ext/psychowl_native/core/data/trigrams/ukr.txt +302 -0
  25. data/ext/psychowl_native/core/data/trigrams/vie.txt +302 -0
  26. data/lib/psychowl/active_model.rb +64 -0
  27. data/lib/psychowl/data_file.rb +14 -0
  28. data/lib/psychowl/detector.rb +123 -0
  29. data/lib/psychowl/engine.rb +23 -0
  30. data/lib/psychowl/info.rb +39 -0
  31. data/lib/psychowl/lang.rb +55 -0
  32. data/lib/psychowl/locale/en.yml +6 -0
  33. data/lib/psychowl/lookup.rb +25 -0
  34. data/lib/psychowl/native_speedup.rb +46 -0
  35. data/lib/psychowl/postgres.rb +35 -0
  36. data/lib/psychowl/railtie.rb +10 -0
  37. data/lib/psychowl/ruby_engine/segmentation.rb +84 -0
  38. data/lib/psychowl/ruby_engine/trigrams.rb +74 -0
  39. data/lib/psychowl/ruby_engine.rb +166 -0
  40. data/lib/psychowl/script.rb +39 -0
  41. data/lib/psychowl/segment.rb +24 -0
  42. data/lib/psychowl/tables.rb +55 -0
  43. data/lib/psychowl/text.rb +26 -0
  44. data/lib/psychowl/version.rb +5 -0
  45. data/lib/psychowl.rb +82 -0
  46. metadata +91 -0
@@ -0,0 +1,35 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Psychowl
4
+ # PostgreSQL full-text search configurations for supported languages, for
5
+ # to_tsvector / to_tsquery. Languages without a built-in stemmer map to
6
+ # "simple".
7
+ #
8
+ # config = Psychowl.detect(body)&.lang&.pg_regconfig || "simple"
9
+ # Post.where("to_tsvector(?::regconfig, body) @@ plainto_tsquery(?::regconfig, ?)", config, config, query)
10
+ module Postgres
11
+ REGCONFIGS = {
12
+ 'ara' => 'arabic',
13
+ 'deu' => 'german',
14
+ 'eng' => 'english',
15
+ 'fra' => 'french',
16
+ 'ind' => 'indonesian',
17
+ 'ita' => 'italian',
18
+ 'nld' => 'dutch',
19
+ 'por' => 'portuguese',
20
+ 'rus' => 'russian',
21
+ 'spa' => 'spanish',
22
+ 'tur' => 'turkish'
23
+ }.freeze
24
+
25
+ # @param lang [Lang, String, Symbol]
26
+ # @return [String] a built-in PostgreSQL text search configuration
27
+ def self.regconfig(lang) = REGCONFIGS.fetch(Lang.fetch(lang).code, 'simple')
28
+ end
29
+
30
+ # PostgreSQL text search configuration per language.
31
+ class Lang
32
+ # @return [String] PostgreSQL text search configuration, see {Postgres}
33
+ def pg_regconfig = Postgres.regconfig(self)
34
+ end
35
+ end
@@ -0,0 +1,10 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Psychowl
4
+ # Makes the `language:` validator available in Rails apps.
5
+ class Railtie < Rails::Railtie
6
+ initializer 'psychowl.active_model' do
7
+ require 'psychowl/active_model'
8
+ end
9
+ end
10
+ end
@@ -0,0 +1,84 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Psychowl
4
+ # Per-sentence detection, mirrored by segment.rs.
5
+ module RubyEngine
6
+ # Sentence-ending marks that need a following space ("3.14" is no boundary).
7
+ SENTENCE_MARKS = %W[. ! ? \u2026].freeze
8
+ # CJK sentence-ending marks, no space needed.
9
+ CLOSING_MARKS = %W[\u3002 \uFF01 \uFF1F].freeze
10
+
11
+ class << self
12
+ # Runs of same-language sentences. Sentences without a language join the
13
+ # previous run (or the first). Each run is re-detected whole and keeps its
14
+ # first sentence's result if that changes the language.
15
+ #
16
+ # @return [Array<Array(Integer, Integer, Integer, Integer, Float)>]
17
+ # [[start, end, language, script, confidence], ...] with character
18
+ # offsets, end exclusive
19
+ def segments(text, filter_mode, filter_langs)
20
+ runs = [] # [start, end, first sentence's detection]
21
+ leading_start = nil
22
+
23
+ sentences(text).each do |start, finish|
24
+ result = detect(text[start...finish], filter_mode, filter_langs)
25
+
26
+ if result.nil?
27
+ if runs.empty?
28
+ leading_start ||= start
29
+ else
30
+ runs.last[1] = finish
31
+ end
32
+ elsif runs.any? && runs.last[2][0] == result[0]
33
+ runs.last[1] = finish
34
+ else
35
+ runs << [leading_start || start, finish, result]
36
+ leading_start = nil
37
+ end
38
+ end
39
+
40
+ runs.map do |start, finish, first|
41
+ whole = detect(text[start...finish], filter_mode, filter_langs)
42
+ whole && whole[0] == first[0] ? [start, finish, *whole] : [start, finish, *first]
43
+ end
44
+ end
45
+
46
+ # [start, end] character offsets. A sentence ends after a newline, 。!?,
47
+ # or . ! ? … followed by a space; trailing spaces stay with it.
48
+ def sentences(text)
49
+ chars = text.chars
50
+ sentences = []
51
+ start = 0
52
+ index = 0
53
+
54
+ while index < chars.size
55
+ char = chars[index]
56
+ boundary = false
57
+
58
+ if char == "\n" || CLOSING_MARKS.include?(char)
59
+ index += 1
60
+ boundary = true
61
+ elsif SENTENCE_MARKS.include?(char)
62
+ index += 1 while index < chars.size && SENTENCE_MARKS.include?(chars[index])
63
+ boundary = index == chars.size || space?(chars[index])
64
+ else
65
+ index += 1
66
+ end
67
+ next unless boundary
68
+
69
+ index += 1 while index < chars.size && space?(chars[index])
70
+ sentences << [start, index]
71
+ start = index
72
+ end
73
+
74
+ sentences << [start, chars.size] if start < chars.size
75
+ sentences
76
+ end
77
+
78
+ private
79
+
80
+ # ASCII whitespace and controls, no-break space, ideographic space.
81
+ def space?(char) = char <= ' ' || char == "\u00A0" || char == "\u3000"
82
+ end
83
+ end
84
+ end
@@ -0,0 +1,74 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Psychowl
4
+ # Trigram model, mirrored by trigram.rs.
5
+ module RubyEngine
6
+ # A profile trigram missing from the text costs this much distance.
7
+ MAX_TRIGRAM_DISTANCE = 300
8
+ MAX_TOTAL_DISTANCE = MAX_TRIGRAM_DISTANCE * MAX_TRIGRAM_DISTANCE
9
+ # Only the most frequent trigrams of the text are compared.
10
+ TEXT_TRIGRAMS_SIZE = 600
11
+
12
+ class << self
13
+ private
14
+
15
+ # Similarity between the text's trigram ranking and each profile.
16
+ def trigram_scores(lowercase, languages)
17
+ positions = trigram_positions(lowercase)
18
+ max_distance = positions.size * MAX_TRIGRAM_DISTANCE
19
+
20
+ scores = languages.to_h do |language|
21
+ profile = Tables::TRIGRAMS[language]
22
+ if profile
23
+ distance = trigram_distance(profile, positions)
24
+ [language, (max_distance - distance).to_f / max_distance]
25
+ else
26
+ [language, 0.0]
27
+ end
28
+ end
29
+ [scores, positions.size]
30
+ end
31
+
32
+ # {trigram => rank}: the text's most frequent trigrams, most frequent
33
+ # first (ties broken by the trigram itself, descending).
34
+ def trigram_positions(lowercase)
35
+ ranked = count_trigrams(lowercase).sort do |(trigram_a, count_a), (trigram_b, count_b)|
36
+ [count_b, trigram_b] <=> [count_a, trigram_a]
37
+ end
38
+ ranked.first(TEXT_TRIGRAMS_SIZE).each_with_index.to_h { |(trigram, _count), rank| [trigram, rank] }
39
+ end
40
+
41
+ # {trigram => occurrences}. Punctuation and digits act as spaces;
42
+ # trigrams that are only word boundary are skipped.
43
+ def count_trigrams(lowercase)
44
+ counts = Hash.new(0)
45
+ first = ' '
46
+ second = nil
47
+
48
+ "#{lowercase.tr(Tables::STOP_CHARS, ' ')} ".each_char do |third|
49
+ if second.nil?
50
+ second = third
51
+ next
52
+ end
53
+
54
+ counts["#{first}#{second}#{third}"] += 1 unless second == ' ' && (first == ' ' || third == ' ')
55
+ first = second
56
+ second = third
57
+ end
58
+ counts
59
+ end
60
+
61
+ def trigram_distance(profile, positions)
62
+ total = 0
63
+ profile.each_with_index do |trigram, index|
64
+ position = positions[trigram]
65
+ total += position ? (position - index).abs : MAX_TRIGRAM_DISTANCE
66
+ end
67
+
68
+ unique = positions.size
69
+ total -= (MAX_TRIGRAM_DISTANCE - unique) * MAX_TRIGRAM_DISTANCE if unique < MAX_TRIGRAM_DISTANCE
70
+ total.clamp(0, MAX_TOTAL_DISTANCE)
71
+ end
72
+ end
73
+ end
74
+ end
@@ -0,0 +1,166 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Psychowl
4
+ # Pure Ruby engine, mirrored by the psychowl crate; parity_test.rb keeps them
5
+ # identical. Primitives in and out: languages and scripts are Tables indices,
6
+ # text is valid UTF-8, a filter is an Engine::FILTER_* mode plus indices.
7
+ # Internal calls never go through Engine.
8
+ module RubyEngine
9
+ class << self
10
+ # @return [Array(Integer, Integer, Float), nil] [language, script, confidence]
11
+ def detect(text, filter_mode, filter_langs)
12
+ case evaluate(text, filter_mode, filter_langs)
13
+ in [:decided, result] then result
14
+ in [:scored, script, [[best, best_score], [_, runner_up_score], *], trigram_count]
15
+ [best, script, confidence(best_score, runner_up_score, trigram_count)]
16
+ in [:nothing] then nil
17
+ end
18
+ end
19
+
20
+ # Every candidate language with its score, best first.
21
+ #
22
+ # @return [Array<Array(Integer, Float)>] [[language, score], ...]
23
+ def candidates(text, filter_mode, filter_langs)
24
+ case evaluate(text, filter_mode, filter_langs)
25
+ in [:decided, [language, _script, confidence]] then [[language, confidence]]
26
+ in [:scored, _script, scores, _trigram_count] then scores
27
+ in [:nothing] then []
28
+ end
29
+ end
30
+
31
+ # @return [Integer, nil] the script with the most characters
32
+ def detect_script(text) = main_script(script_counts(text))
33
+
34
+ # @return [Array<Integer>] number of characters per script, by script index
35
+ def script_counts(text)
36
+ Tables::SCRIPTS.map { text.count(it.charset) }
37
+ end
38
+
39
+ # @return [Array<Array(Integer, Integer, Float), nil>] one #detect result per text
40
+ def detect_many(texts, filter_mode, filter_langs)
41
+ texts.map { detect(it, filter_mode, filter_langs) }
42
+ end
43
+
44
+ private
45
+
46
+ # Mirrors the Rust `Outcome`:
47
+ # [:nothing] no letters / no allowed language
48
+ # [:decided, [language, script, confidence]] one allowed language, or Han
49
+ # [:scored, script, scores, trigram_count] several languages, best first
50
+ def evaluate(text, filter_mode, filter_langs)
51
+ counts = script_counts(text)
52
+ script = main_script(counts)
53
+ return [:nothing] if script.nil?
54
+
55
+ if script == Tables::HAN
56
+ result = detect_han(counts, filter_mode, filter_langs)
57
+ return result ? [:decided, result] : [:nothing]
58
+ end
59
+
60
+ case allowed(Tables::SCRIPTS[script].languages, filter_mode, filter_langs)
61
+ in [] then [:nothing]
62
+ in [only] then [:decided, [only, script, 1.0]]
63
+ in languages then [:scored, script, *language_scores(text, languages)]
64
+ end
65
+ end
66
+
67
+ # Ties go to the script listed first in scripts.tsv.
68
+ def main_script(counts)
69
+ best = nil
70
+ counts.each_with_index do |count, index|
71
+ best = index if count.positive? && (best.nil? || count > counts[best])
72
+ end
73
+ best
74
+ end
75
+
76
+ def allowed(languages, filter_mode, filter_langs)
77
+ case filter_mode
78
+ when Engine::FILTER_ALLOW then languages.select { filter_langs.include?(it) }
79
+ when Engine::FILTER_DENY then languages.reject { filter_langs.include?(it) }
80
+ else languages
81
+ end
82
+ end
83
+
84
+ # Han characters are shared by Chinese and Japanese; kana tips it to
85
+ # Japanese.
86
+ def detect_han(counts, filter_mode, filter_langs)
87
+ languages = allowed(Tables::SCRIPTS[Tables::HAN].languages, filter_mode, filter_langs)
88
+ mandarin = languages.include?(Tables::CMN)
89
+ japanese = languages.include?(Tables::JPN)
90
+
91
+ case [mandarin, japanese]
92
+ in [false, false] then return nil
93
+ in [true, false] then return [Tables::CMN, Tables::HAN, 1.0]
94
+ in [false, true] then return [Tables::JPN, Tables::HAN, 1.0]
95
+ in [true, true] then nil
96
+ end
97
+
98
+ kana = counts[Tables::HIRAGANA] + counts[Tables::KATAKANA]
99
+ kana_ratio = kana.to_f / (counts[Tables::HAN] + kana)
100
+
101
+ if kana_ratio > 0.2 then [Tables::JPN, Tables::HAN, 1.0]
102
+ elsif kana_ratio > 0.05 then [Tables::JPN, Tables::HAN, 0.5]
103
+ elsif kana_ratio > 0.02 then [Tables::CMN, Tables::HAN, 0.5]
104
+ else [Tables::CMN, Tables::HAN, 1.0]
105
+ end
106
+ end
107
+
108
+ # @return [Array(Array<Array(Integer, Float)>, Integer)] scores best first, trigram count
109
+ def language_scores(text, languages)
110
+ lowercase = text.downcase
111
+ alphabet_scores, char_count = alphabet_scores(lowercase, languages)
112
+ trigram_scores, trigram_count = trigram_scores(lowercase, languages)
113
+
114
+ alphabet_weight = alphabet_weight(char_count)
115
+ trigram_weight = 1.0 - alphabet_weight
116
+
117
+ scores = languages.map do |language|
118
+ score = (alphabet_scores[language] * alphabet_weight) + (trigram_scores[language] * trigram_weight)
119
+ [language, score]
120
+ end
121
+ [sort_by_score(scores), trigram_count]
122
+ end
123
+
124
+ # 2/3 for an empty text, falling to 1/3 at 100 letters and beyond.
125
+ def alphabet_weight(char_count) = (-(char_count / 300.0) + (2.0 / 3.0)).clamp(1.0 / 3.0, 2.0 / 3.0)
126
+
127
+ # Highest score first; ties go to the lower language index.
128
+ def sort_by_score(scores)
129
+ scores.sort do |(language_a, score_a), (language_b, score_b)|
130
+ order = score_b <=> score_a
131
+ order.zero? ? language_a <=> language_b : order
132
+ end
133
+ end
134
+
135
+ # Share of letters that belong to each language's alphabet. A letter
136
+ # outside the alphabet counts against the language.
137
+ def alphabet_scores(lowercase, languages)
138
+ char_count = lowercase.length - lowercase.count(Tables::STOP_CHARS)
139
+
140
+ unless languages.any? { Tables::ALPHABETS.key?(it) }
141
+ return [languages.to_h { [it, 1.0] }, 1]
142
+ end
143
+
144
+ scores = languages.to_h do |language|
145
+ letters = Tables::ALPHABETS[language]
146
+ hits = letters ? lowercase.count(letters) : 0
147
+ raw = (2 * hits) - char_count
148
+ raw = 0 if raw.negative?
149
+ [language, raw.to_f / char_count]
150
+ end
151
+ [scores, char_count]
152
+ end
153
+
154
+ # How far the best score is ahead of the runner-up, scaled by how much
155
+ # evidence (trigrams) there was.
156
+ def confidence(best, runner_up, count)
157
+ return 0.0 if best.zero?
158
+ return best if runner_up.zero?
159
+
160
+ confident_rate = (3.0 / count) + 0.015
161
+ rate = (best - runner_up) / runner_up
162
+ rate > confident_rate ? 1.0 : rate / confident_rate
163
+ end
164
+ end
165
+ end
166
+ end
@@ -0,0 +1,39 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Psychowl
4
+ # A writing system, e.g. Latin or Cyrillic. Instances are canonical.
5
+ #
6
+ # @!attribute [r] name
7
+ # @return [String] e.g. "Cyrillic"
8
+ # @!attribute [r] langs
9
+ # @return [Array<Lang>] supported languages written in this script
10
+ Script = ::Data.define(:name, :langs) do
11
+ # @return [String] the script name
12
+ def to_s = name
13
+
14
+ def inspect = "#<#{self.class} #{name}>"
15
+ end
16
+
17
+ # The canonical instances and lookup by name.
18
+ class Script
19
+ singleton_class.prepend(Lookup) # ahead of Data.define's own `[]` constructor
20
+
21
+ ALL = Tables::SCRIPTS.map do |script|
22
+ new(name: script.name, langs: script.languages.map { Lang.at(it) }.freeze)
23
+ end.freeze
24
+ BY_NAME = ALL.to_h { |script| [script.name.downcase, script] }.freeze
25
+ private_constant :ALL, :BY_NAME
26
+
27
+ class << self
28
+ # @return [Array<Script>] every known script (frozen)
29
+ def all = ALL
30
+
31
+ private :new
32
+
33
+ private
34
+
35
+ def by_key = BY_NAME
36
+ def unknown_key_message = 'unknown script'
37
+ end
38
+ end
39
+ end
@@ -0,0 +1,24 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Psychowl
4
+ # A stretch of text in one language, from {Psychowl.segments}.
5
+ #
6
+ # @!attribute [r] text
7
+ # @return [String] the segment's text
8
+ # @!attribute [r] range
9
+ # @return [Range] character offsets in the original text
10
+ # @!attribute [r] info
11
+ # @return [Info] the detection for this segment
12
+ Segment = ::Data.define(:text, :range, :info) do
13
+ # @return [Lang]
14
+ def lang = info.lang
15
+
16
+ # @return [Script]
17
+ def script = info.script
18
+
19
+ # @return [Float]
20
+ def confidence = info.confidence
21
+
22
+ def inspect = "#<#{self.class} #{range} #{info.lang.code} #{text[0, 30].inspect}>"
23
+ end
24
+ end
@@ -0,0 +1,55 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Psychowl
4
+ # The data tables shared with the Rust crate, read once at load time.
5
+ # Languages and scripts are referred to by line index, as in the crate.
6
+ module Tables
7
+ DIR = File.expand_path('../../ext/psychowl_native/core/data', __dir__)
8
+
9
+ # ASCII controls, spaces, digits and punctuation, as a String#count set.
10
+ STOP_CHARS = "\u0000-@[-`{-~"
11
+
12
+ class << self
13
+ def rows(file) = DataFile.lines(File.join(DIR, file)).map { it.split("\t") }
14
+
15
+ # Escapes the characters String#count treats specially.
16
+ def charset(chars) = chars.gsub(/[\\^-]/) { |char| "\\#{char}" }
17
+
18
+ def range_charset(ranges)
19
+ ranges.split(',').map do |range|
20
+ low, high = range.split('-').map { it.to_i(16).chr(Encoding::UTF_8) }
21
+ high ? "#{charset(low)}-#{charset(high)}" : charset(low)
22
+ end.join
23
+ end
24
+ end
25
+
26
+ # [[code, iso639_1, eng_name, name], ...]
27
+ LANGUAGES = rows('languages.tsv').map(&:freeze).freeze
28
+ LANGUAGE_INDEX = LANGUAGES.each_with_index.to_h { |(code, *), index| [code, index] }.freeze
29
+
30
+ Script = ::Data.define(:name, :languages, :charset)
31
+ # [Script(name, [language index, ...], String#count set), ...]
32
+ SCRIPTS = rows('scripts.tsv').map do |name, codes, ranges|
33
+ languages = codes == '-' ? [] : codes.split(',').map { LANGUAGE_INDEX.fetch(it) }
34
+ Script.new(name: name.freeze, languages: languages.freeze, charset: range_charset(ranges).freeze)
35
+ end.freeze
36
+ SCRIPT_INDEX = SCRIPTS.each_with_index.to_h { |script, index| [script.name, index] }.freeze
37
+
38
+ # {language index => String#count set of its letters}
39
+ ALPHABETS = rows('alphabets.tsv').to_h do |code, letters|
40
+ [LANGUAGE_INDEX.fetch(code), charset(letters).freeze]
41
+ end.freeze
42
+
43
+ # {language index => [trigram, ...]} most frequent first, " " for a word boundary.
44
+ TRIGRAMS = Dir[File.join(DIR, 'trigrams', '*.txt')].to_h do |path|
45
+ code = File.basename(path, '.txt')
46
+ [LANGUAGE_INDEX.fetch(code), DataFile.lines(path).map { it.tr('_', ' ').freeze }.freeze]
47
+ end.freeze
48
+
49
+ HAN = SCRIPT_INDEX.fetch('Han')
50
+ HIRAGANA = SCRIPT_INDEX.fetch('Hiragana')
51
+ KATAKANA = SCRIPT_INDEX.fetch('Katakana')
52
+ CMN = LANGUAGE_INDEX['cmn']
53
+ JPN = LANGUAGE_INDEX['jpn']
54
+ end
55
+ end
@@ -0,0 +1,26 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Psychowl
4
+ # Input validation shared by both engines: they only ever see valid UTF-8.
5
+ module Text
6
+ class << self
7
+ # @param text [String, #to_str]
8
+ # @return [String] valid UTF-8
9
+ # @raise [TypeError] not a String
10
+ # @raise [EncodingError] invalid bytes, or not convertible to UTF-8
11
+ def prepare(text)
12
+ string = String.try_convert(text)
13
+ raise TypeError, "no implicit conversion of #{text.nil? ? 'nil' : text.class} into String" unless string
14
+
15
+ utf8 =
16
+ case string.encoding
17
+ when Encoding::UTF_8, Encoding::US_ASCII, Encoding::BINARY then string.dup.force_encoding(Encoding::UTF_8)
18
+ else string.encode(Encoding::UTF_8)
19
+ end
20
+ raise EncodingError, 'text is not valid UTF-8' unless utf8.valid_encoding?
21
+
22
+ utf8
23
+ end
24
+ end
25
+ end
26
+ end
@@ -0,0 +1,5 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Psychowl
4
+ VERSION = '0.1.0'
5
+ end
data/lib/psychowl.rb ADDED
@@ -0,0 +1,82 @@
1
+ # frozen_string_literal: true
2
+
3
+ require_relative 'psychowl/version'
4
+ require_relative 'psychowl/data_file'
5
+ require_relative 'psychowl/tables'
6
+ require_relative 'psychowl/text'
7
+ require_relative 'psychowl/engine'
8
+ require_relative 'psychowl/ruby_engine'
9
+ require_relative 'psychowl/ruby_engine/segmentation'
10
+ require_relative 'psychowl/ruby_engine/trigrams'
11
+ require_relative 'psychowl/lookup'
12
+ require_relative 'psychowl/lang'
13
+ require_relative 'psychowl/script'
14
+ require_relative 'psychowl/info'
15
+ require_relative 'psychowl/segment'
16
+ require_relative 'psychowl/detector'
17
+
18
+ # Natural language and script detection. It knows what language you speak.
19
+ # It knows you skipped your lesson.
20
+ #
21
+ # info = Psychowl.detect("Die Deutsche Bahn ist heute pünktlich. Wir ermitteln.")
22
+ # info.lang.code # => "deu"
23
+ # info.script.name # => "Latin"
24
+ module Psychowl
25
+ DEFAULT_DETECTOR = Detector.new
26
+ private_constant :DEFAULT_DETECTOR
27
+
28
+ class << self
29
+ # @param text [String]
30
+ # @param allowlist [Array<Lang, String, Symbol>, nil] only consider these
31
+ # @param denylist [Array<Lang, String, Symbol>, nil] never consider these
32
+ # @return [Info, nil] nil when no supported language was found
33
+ # @raise [TypeError] text is not a String
34
+ # @raise [EncodingError] text is not valid UTF-8 (or convertible to it)
35
+ # @raise [ArgumentError] invalid allowlist/denylist, see {Detector#initialize}
36
+ def detect(text, allowlist: nil, denylist: nil) = detector(allowlist, denylist).detect(text)
37
+
38
+ # @param (see .detect)
39
+ # @return [Lang, nil]
40
+ def detect_lang(text, allowlist: nil, denylist: nil) = detector(allowlist, denylist).detect_lang(text)
41
+
42
+ # @param text [String]
43
+ # @return [Script, nil]
44
+ def detect_script(text) = DEFAULT_DETECTOR.detect_script(text)
45
+
46
+ # @param (see .detect)
47
+ # @param limit [Integer, nil]
48
+ # @return [Array<Array(Lang, Float)>] see {Detector#candidates}
49
+ def candidates(text, limit: nil, allowlist: nil, denylist: nil)
50
+ detector(allowlist, denylist).candidates(text, limit:)
51
+ end
52
+
53
+ # @param texts [Array<String>]
54
+ # @param allowlist [Array<Lang, String, Symbol>, nil]
55
+ # @param denylist [Array<Lang, String, Symbol>, nil]
56
+ # @return [Array<Info, nil>] see {Detector#detect_many}
57
+ def detect_many(texts, allowlist: nil, denylist: nil) = detector(allowlist, denylist).detect_many(texts)
58
+
59
+ # @param (see .detect)
60
+ # @return [Array<Segment>] see {Detector#segments}
61
+ def segments(text, allowlist: nil, denylist: nil) = detector(allowlist, denylist).segments(text)
62
+
63
+ # @param text [String]
64
+ # @return [Hash{Script => Float}] see {Detector#scripts}
65
+ def scripts(text) = DEFAULT_DETECTOR.scripts(text)
66
+
67
+ # @return [Symbol] :native when the Rust extension is loaded, :ruby otherwise
68
+ def backend = Engine.native? ? :native : :ruby
69
+
70
+ private
71
+
72
+ def detector(allowlist, denylist)
73
+ return DEFAULT_DETECTOR if allowlist.nil? && denylist.nil?
74
+
75
+ Detector.new(allowlist:, denylist:)
76
+ end
77
+ end
78
+ end
79
+
80
+ require_relative 'psychowl/postgres'
81
+ require_relative 'psychowl/native_speedup'
82
+ require_relative 'psychowl/railtie' if defined?(Rails::Railtie)