cjk_index 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,236 @@
1
+ # frozen_string_literal: true
2
+
3
+ require "json"
4
+ require "set"
5
+
6
+ module CJKIndex
7
+ # Queries an index in Ruby, with the same matching, ranking, highlighting
8
+ # and excerpts as the browser runtime (runtime.js). The test suite runs
9
+ # both on the same index and checks that they agree.
10
+ #
11
+ # searcher = CJKIndex::Searcher.new(builder) # or a parsed JSON index
12
+ # searcher.search("鳥瞰図") # => [{ ref: "item1.html", score: 1.2, fields: ["title"] }]
13
+ # searcher.highlight("東京帝國大學", "帝国大学") # => "東京<mark>帝國大學</mark>"
14
+ #
15
+ # The normalizer is taken from the tables the index was built with, so a
16
+ # query folds exactly like the indexed text.
17
+ class Searcher
18
+ # BM25 parameters (the usual defaults, as in Lucene and Pagefind).
19
+ K1 = 1.2
20
+ B = 0.75
21
+ # Score multiplier for matches found by prefix or typo expansion.
22
+ EXPANDED = 0.6
23
+
24
+ attr_reader :fields, :refs, :normalizer
25
+
26
+ # index is a Builder, its to_h, or the JSON it wrote (String or parsed).
27
+ # prefix: the last Latin word of a query also matches longer words.
28
+ # typos: Latin words of 5+ letters also match words one edit away.
29
+ def self.load(path, **options)
30
+ new(File.read(path, encoding: "UTF-8"), **options)
31
+ end
32
+
33
+ def initialize(index, prefix: true, typos: true, normalizer: nil)
34
+ data = case index
35
+ when Builder then index.to_h
36
+ when String then JSON.parse(index)
37
+ else index
38
+ end
39
+ unless data.is_a?(Hash) && data["format"] == Builder::FORMAT
40
+ raise ArgumentError, "cjk_index: unsupported index format #{data.is_a?(Hash) ? data['format'].inspect : data.class}"
41
+ end
42
+
43
+ @normalizer = normalizer || Normalizer.new(tables: data["variants"])
44
+ if @normalizer.tables != data["variants"]
45
+ raise ArgumentError, "cjk_index: index built with variant tables #{data['variants'].inspect} " \
46
+ "but this normalizer folds with #{@normalizer.tables.inspect}; rebuild both together"
47
+ end
48
+
49
+ @prefix = prefix
50
+ @typos = typos
51
+ @fields = data["fields"]
52
+ @boosts = data["boosts"]
53
+ @refs = data["refs"]
54
+ @lengths = data["lengths"]
55
+ @tokens = data["tokens"]
56
+ nf = @fields.length
57
+ avg = Array.new(nf, 0.0)
58
+ @lengths.each_with_index { |len, i| avg[i % nf] += len }
59
+ @avg_length = avg.map { |s| (v = s / [1, @refs.length].max).zero? ? 1 : v }
60
+ @latin_tokens = @tokens.keys.select { |t| latin?(t) }.sort
61
+ end
62
+
63
+ # Every phrase of the query must occur in the document (in any field),
64
+ # with its tokens adjacent and in order. Returns [{ ref:, score:, fields: }]
65
+ # with the best match first; fields lists the fields that matched.
66
+ def search(query)
67
+ nf = @fields.length
68
+ n = @refs.length
69
+ qp = Tokenizer.phrases(query, normalizer: @normalizer)
70
+ return [] if qp.empty?
71
+
72
+ scores = Hash.new(0.0)
73
+ matched_fields = Hash.new { |h, k| h[k] = [] }
74
+ docs_so_far = nil
75
+
76
+ qp.each_with_index do |phrase, q|
77
+ expanded = false
78
+ parts = phrase.each_with_index.map do |(token, offset), i|
79
+ tokens, exp = alternatives(token, q == qp.length - 1 && i == phrase.length - 1)
80
+ expanded ||= exp
81
+ [offset, positions(tokens)]
82
+ end
83
+ return [] if parts.any? { |_, pos| pos.empty? }
84
+
85
+ parts = parts.sort_by.with_index { |(_, pos), i| [pos.size, i] }
86
+ rare_offset, rare = parts.first
87
+ hits = {}
88
+ rare.each do |key, set|
89
+ tf = set.count do |pos|
90
+ start = pos - rare_offset
91
+ parts.drop(1).all? { |offset, other| other[key]&.include?(start + offset) }
92
+ end
93
+ hits[key] = tf if tf.positive?
94
+ end
95
+ docs = hits.keys.map { |key| key / nf }.to_set
96
+ docs &= docs_so_far if docs_so_far
97
+ return [] if docs.empty?
98
+
99
+ docs_so_far = docs
100
+ idf = Math.log(1 + ((n - docs.size + 0.5) / (docs.size + 0.5)))
101
+ weight = expanded ? EXPANDED : 1
102
+ hits.each do |key, tf|
103
+ doc = key / nf
104
+ field = key % nf
105
+ norm = 1 - B + (B * @lengths[key] / @avg_length[field])
106
+ scores[doc] += weight * @boosts[field] * idf * (tf * (K1 + 1)) / (tf + (K1 * norm))
107
+ matched_fields[doc] |= [@fields[field]]
108
+ end
109
+ end
110
+
111
+ docs_so_far.sort_by { |doc| [-scores[doc], doc] }
112
+ .map { |doc| { ref: @refs[doc], score: scores[doc], fields: matched_fields[doc] } }
113
+ end
114
+
115
+ def highlight(text, query) = CJKIndex.highlight(text, query, normalizer: @normalizer)
116
+
117
+ def excerpt(text, query, length = 80) = CJKIndex.excerpt(text, query, length, normalizer: @normalizer)
118
+
119
+ private
120
+
121
+ def latin?(token) = !token.match?(Tokenizer::CJK)
122
+
123
+ # "doc * nf + field" => Set of positions, merged over the alternatives.
124
+ def positions(tokens)
125
+ nf = @fields.length
126
+ map = {}
127
+ tokens.each do |token|
128
+ postings = @tokens[token] or next
129
+ i = 0
130
+ while i < postings.length
131
+ count = postings[i + 2]
132
+ (map[(postings[i] * nf) + postings[i + 1]] ||= Set.new).merge(postings[i + 3, count])
133
+ i += 3 + count
134
+ end
135
+ end
136
+ map
137
+ end
138
+
139
+ # Alternatives for one query token: itself, and for Latin words the index
140
+ # words it is a prefix of (last word only) or one edit away from.
141
+ def alternatives(token, last)
142
+ return [[token], false] unless latin?(token)
143
+
144
+ out = @tokens.key?(token) ? [token] : []
145
+ if @prefix && last
146
+ from = @latin_tokens.bsearch_index { |k| k >= token } || @latin_tokens.length
147
+ @latin_tokens[from..].each do |k|
148
+ break unless k.start_with?(token)
149
+
150
+ out << k unless k == token
151
+ end
152
+ end
153
+ out.concat(@latin_tokens.select { |k| one_edit?(token, k) }) if @typos && !@tokens.key?(token) && token.length >= 5
154
+ [out, !out.empty? && out.first != token]
155
+ end
156
+
157
+ # true when a and b differ by exactly one insertion, deletion or substitution.
158
+ def one_edit?(a, b)
159
+ return false if a == b || (a.length - b.length).abs > 1
160
+
161
+ i = 0
162
+ i += 1 while i < a.length && i < b.length && a[i] == b[i]
163
+ if a.length == b.length then a[(i + 1)..] == b[(i + 1)..]
164
+ elsif a.length > b.length then a[(i + 1)..] == b[i..]
165
+ else a[i..] == b[(i + 1)..]
166
+ end
167
+ end
168
+ end
169
+
170
+ HTML_ESCAPES = { "&" => "&amp;", "<" => "&lt;", ">" => "&gt;", '"' => "&quot;", "'" => "&#39;" }.freeze
171
+
172
+ def self.escape_html(text) = text.gsub(/[&<>"']/, HTML_ESCAPES)
173
+
174
+ # [start, end) ranges, in characters of the original text, where a phrase
175
+ # of the query occurs after normalization.
176
+ def self.match_ranges(text, query, normalizer: Normalizer.default)
177
+ chars = text.to_s.chars
178
+ norm = +""
179
+ origin = []
180
+ chars.each_with_index do |ch, i|
181
+ folded = normalizer.normalize(ch)
182
+ # 々 depends on the previous character, so fold it in context.
183
+ folded = norm[-1] if folded == Normalizer::ITERATION_MARK && !norm.empty?
184
+ norm << folded
185
+ folded.length.times { origin << i }
186
+ end
187
+ needles = normalizer.normalize(query).split(Tokenizer::SEPARATOR).reject(&:empty?)
188
+ ranges = needles.flat_map do |needle|
189
+ found = []
190
+ from = 0
191
+ while (at = norm.index(needle, from))
192
+ found << [origin[at], origin[at + needle.length - 1] + 1]
193
+ from = at + needle.length
194
+ end
195
+ found
196
+ end
197
+ merged = ranges.sort_by(&:first).each_with_object([]) do |r, acc|
198
+ if acc.last && r[0] <= acc.last[1] then acc.last[1] = [acc.last[1], r[1]].max
199
+ else acc << r.dup
200
+ end
201
+ end
202
+ [chars, merged]
203
+ end
204
+
205
+ # HTML-escaped text with every match of the query wrapped in <mark>.
206
+ # 「東京帝國大學」 is marked for the query 「帝国大学」.
207
+ def self.highlight(text, query, normalizer: Normalizer.default)
208
+ chars, ranges = match_ranges(text, query, normalizer: normalizer)
209
+ at = 0
210
+ out = +""
211
+ ranges.each do |from, to|
212
+ out << escape_html(chars[at...from].join) << "<mark>" << escape_html(chars[from...to].join) << "</mark>"
213
+ at = to
214
+ end
215
+ out << escape_html(chars[at..].join)
216
+ end
217
+
218
+ # A highlighted window of about `length` characters around the first match.
219
+ def self.excerpt(text, query, length = 80, normalizer: Normalizer.default)
220
+ chars, ranges = match_ranges(text, query, normalizer: normalizer)
221
+ if ranges.empty?
222
+ return chars.length > length ? "#{escape_html(chars[0, length].join)}…" : escape_html(chars.join)
223
+ end
224
+
225
+ start = [0, ranges[0][0] - (length / 3)].max
226
+ stop = [chars.length, start + length].min
227
+ "#{start.positive? ? '…' : ''}#{highlight(chars[start...stop].join, query, normalizer: normalizer)}" \
228
+ "#{stop < chars.length ? '…' : ''}"
229
+ end
230
+
231
+ # Substring test on normalized text, ignoring whitespace.
232
+ def self.includes?(haystack, needle, normalizer: Normalizer.default)
233
+ strip = ->(s) { normalizer.normalize(s).gsub(/\s+/, "") }
234
+ strip.call(haystack).include?(strip.call(needle))
235
+ end
236
+ end
@@ -0,0 +1,114 @@
1
+ # frozen_string_literal: true
2
+
3
+ module CJKIndex
4
+ # Splits normalized text into search tokens.
5
+ #
6
+ # Runs of CJK characters (kanji, kana, hangul) have no spaces between words,
7
+ # so they become overlapping character bigrams: 「鳥瞰図」 -> 鳥瞰, 瞰図.
8
+ # Other runs (Latin letters, digits) stay whole words. Each token carries
9
+ # its offset (in characters of the normalized text), so a query can require
10
+ # its bigrams to be adjacent and in order: an exact substring match, found
11
+ # without a dictionary.
12
+ #
13
+ # The character classes are kept as code point ranges so that Runtime can
14
+ # emit exactly the same classes into the browser script, where the same
15
+ # character-by-character loop runs.
16
+ module Tokenizer
17
+ CJK_RANGES = [
18
+ [0x1100, 0x11FF], # hangul jamo (old hangul is written as jamo sequences)
19
+ [0x3005, 0x3005], # 々 iteration mark
20
+ [0x3007, 0x3007], # 〇
21
+ [0x3040, 0x309F], # hiragana
22
+ [0x30A0, 0x30FF], # katakana
23
+ [0x3130, 0x318F], # hangul compatibility jamo
24
+ [0x3400, 0x4DBF], # CJK extension A
25
+ [0x4E00, 0x9FFF], # CJK unified ideographs
26
+ [0xA960, 0xA97F], # hangul jamo extended-A
27
+ [0xAC00, 0xD7AF], # hangul syllables
28
+ [0xD7B0, 0xD7FF], # hangul jamo extended-B
29
+ [0xF900, 0xFAFF], # CJK compatibility ideographs
30
+ [0x20000, 0x2FA1F] # CJK extensions B- and compatibility supplement
31
+ ].freeze
32
+
33
+ SEPARATOR_RANGES = [
34
+ [0x09, 0x0D], [0x20, 0x20], [0xA0, 0xA0], [0x1680, 0x1680],
35
+ [0x2000, 0x200A], [0x2028, 0x2029], [0x202F, 0x202F], [0x205F, 0x205F],
36
+ [0x3000, 0x3000], [0xFEFF, 0xFEFF]
37
+ ].freeze
38
+ SEPARATOR_CHARS = "、。,.・:;!?「」『』()()[][]【】〈〉《》〔〕{}{}\"'`,.:;!?/\\|-‐―-~〜=+*&%$#@^<>"
39
+
40
+ module_function
41
+
42
+ # "[...]" character class usable in both Ruby and JavaScript (with /u).
43
+ def char_class(ranges, chars = "")
44
+ body = ranges.map do |from, to|
45
+ from == to ? escape(from) : "#{escape(from)}-#{escape(to)}"
46
+ end
47
+ body += chars.each_char.map { |c| escape(c.ord) }
48
+ "[#{body.join}]"
49
+ end
50
+
51
+ def escape(code)
52
+ format("\\u{%x}", code)
53
+ end
54
+
55
+ CJK = Regexp.new(char_class(CJK_RANGES))
56
+ SEPARATOR = Regexp.new(char_class(SEPARATOR_RANGES, SEPARATOR_CHARS))
57
+
58
+ # Token strings only. unigrams: true adds every single CJK character as
59
+ # well; the index needs them so that one-character queries (「竹」) and
60
+ # CJK characters between digits (「第1輯」) can be found.
61
+ def tokenize(text, unigrams: false, normalizer: Normalizer.default)
62
+ tokens_with_offsets(text, unigrams: unigrams, normalizer: normalizer).map(&:first)
63
+ end
64
+
65
+ # [[token, offset], ...] for the whole text. Offsets count characters of
66
+ # the normalized text, so the index and the query agree on them.
67
+ def tokens_with_offsets(text, unigrams: false, normalizer: Normalizer.default)
68
+ phrases(text, unigrams: unigrams, normalizer: normalizer).flatten(1)
69
+ end
70
+
71
+ # Splits at separators (spaces, punctuation) into phrases, each a list of
72
+ # [token, offset]. A query matches when every phrase occurs somewhere in
73
+ # the document, with the tokens of each phrase at the same relative offsets.
74
+ def phrases(text, unigrams: false, normalizer: Normalizer.default)
75
+ result = []
76
+ phrase = []
77
+ run = +""
78
+ run_cjk = nil
79
+ run_start = 0
80
+ flush = lambda do
81
+ next if run.empty?
82
+
83
+ if run_cjk
84
+ chars = run.chars
85
+ if chars.length == 1 || unigrams
86
+ chars.each_with_index { |c, i| phrase << [c, run_start + i] }
87
+ end
88
+ chars.each_cons(2).with_index { |(a, b), i| phrase << [a + b, run_start + i] }
89
+ else
90
+ phrase << [run.dup, run_start]
91
+ end
92
+ run = +""
93
+ end
94
+
95
+ normalizer.normalize(text).each_char.with_index do |ch, offset|
96
+ if ch.match?(SEPARATOR)
97
+ flush.call
98
+ result << phrase unless phrase.empty?
99
+ phrase = []
100
+ run_cjk = nil
101
+ next
102
+ end
103
+ cjk = ch.match?(CJK)
104
+ flush.call if !run_cjk.nil? && cjk != run_cjk
105
+ run_start = offset if run.empty?
106
+ run_cjk = cjk
107
+ run << ch
108
+ end
109
+ flush.call
110
+ result << phrase unless phrase.empty?
111
+ result
112
+ end
113
+ end
114
+ end
@@ -0,0 +1,5 @@
1
+ # frozen_string_literal: true
2
+
3
+ module CJKIndex
4
+ VERSION = "0.1.0"
5
+ end
data/lib/cjk_index.rb ADDED
@@ -0,0 +1,19 @@
1
+ # frozen_string_literal: true
2
+
3
+ require_relative "cjk_index/version"
4
+ require_relative "cjk_index/normalizer"
5
+ require_relative "cjk_index/tokenizer"
6
+ require_relative "cjk_index/builder"
7
+ require_relative "cjk_index/runtime"
8
+ require_relative "cjk_index/searcher"
9
+
10
+ # Dictionary-free full-text search for Chinese, Japanese and Korean text:
11
+ # build the index in Ruby, query it in Ruby (Searcher) or in the browser
12
+ # (Runtime), with the same results.
13
+ module CJKIndex
14
+ def self.normalize(text, normalizer: Normalizer.default) = normalizer.normalize(text)
15
+
16
+ def self.tokenize(text, unigrams: false, normalizer: Normalizer.default)
17
+ Tokenizer.tokenize(text, unigrams: unigrams, normalizer: normalizer)
18
+ end
19
+ end
metadata ADDED
@@ -0,0 +1,59 @@
1
+ --- !ruby/object:Gem::Specification
2
+ name: cjk_index
3
+ version: !ruby/object:Gem::Version
4
+ version: 0.1.0
5
+ platform: ruby
6
+ authors:
7
+ - Satoru Nakamura
8
+ bindir: bin
9
+ cert_chain: []
10
+ date: 1980-01-02 00:00:00.000000000 Z
11
+ dependencies: []
12
+ description: Builds a bigram search index in Ruby (with kana, old/new kanji and NFKC
13
+ normalization), and queries it in Ruby or with a small browser runtime that returns
14
+ the same results. Includes a Jekyll plugin and a CollectionBuilder preset.
15
+ executables: []
16
+ extensions: []
17
+ extra_rdoc_files: []
18
+ files:
19
+ - LICENSE
20
+ - LICENSE-OpenCC
21
+ - NOTICE
22
+ - README.md
23
+ - data/ja-kyujitai.tsv
24
+ - data/zh-hant-hans.tsv
25
+ - lib/cjk_index.rb
26
+ - lib/cjk_index/builder.rb
27
+ - lib/cjk_index/jekyll.rb
28
+ - lib/cjk_index/normalizer.rb
29
+ - lib/cjk_index/runtime.js
30
+ - lib/cjk_index/runtime.rb
31
+ - lib/cjk_index/searcher.rb
32
+ - lib/cjk_index/tokenizer.rb
33
+ - lib/cjk_index/version.rb
34
+ homepage: https://github.com/nakamura196/cjk_index
35
+ licenses:
36
+ - MIT
37
+ - Apache-2.0
38
+ metadata:
39
+ source_code_uri: https://github.com/nakamura196/cjk_index
40
+ rubygems_mfa_required: 'true'
41
+ rdoc_options: []
42
+ require_paths:
43
+ - lib
44
+ required_ruby_version: !ruby/object:Gem::Requirement
45
+ requirements:
46
+ - - ">="
47
+ - !ruby/object:Gem::Version
48
+ version: '3.1'
49
+ required_rubygems_version: !ruby/object:Gem::Requirement
50
+ requirements:
51
+ - - ">="
52
+ - !ruby/object:Gem::Version
53
+ version: '0'
54
+ requirements: []
55
+ rubygems_version: 4.0.3
56
+ specification_version: 4
57
+ summary: Dictionary-free full-text search for Chinese, Japanese and Korean in pure
58
+ Ruby and the browser
59
+ test_files: []