cjk_index 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +7 -0
- data/LICENSE +21 -0
- data/LICENSE-OpenCC +56 -0
- data/NOTICE +7 -0
- data/README.md +171 -0
- data/data/ja-kyujitai.tsv +292 -0
- data/data/zh-hant-hans.tsv +3207 -0
- data/lib/cjk_index/builder.rb +86 -0
- data/lib/cjk_index/jekyll.rb +127 -0
- data/lib/cjk_index/normalizer.rb +89 -0
- data/lib/cjk_index/runtime.js +318 -0
- data/lib/cjk_index/runtime.rb +26 -0
- data/lib/cjk_index/searcher.rb +236 -0
- data/lib/cjk_index/tokenizer.rb +114 -0
- data/lib/cjk_index/version.rb +5 -0
- data/lib/cjk_index.rb +19 -0
- metadata +59 -0
|
@@ -0,0 +1,236 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "json"
|
|
4
|
+
require "set"
|
|
5
|
+
|
|
6
|
+
module CJKIndex
|
|
7
|
+
# Queries an index in Ruby, with the same matching, ranking, highlighting
|
|
8
|
+
# and excerpts as the browser runtime (runtime.js). The test suite runs
|
|
9
|
+
# both on the same index and checks that they agree.
|
|
10
|
+
#
|
|
11
|
+
# searcher = CJKIndex::Searcher.new(builder) # or a parsed JSON index
|
|
12
|
+
# searcher.search("鳥瞰図") # => [{ ref: "item1.html", score: 1.2, fields: ["title"] }]
|
|
13
|
+
# searcher.highlight("東京帝國大學", "帝国大学") # => "東京<mark>帝國大學</mark>"
|
|
14
|
+
#
|
|
15
|
+
# The normalizer is taken from the tables the index was built with, so a
|
|
16
|
+
# query folds exactly like the indexed text.
|
|
17
|
+
class Searcher
|
|
18
|
+
# BM25 parameters (the usual defaults, as in Lucene and Pagefind).
|
|
19
|
+
K1 = 1.2
|
|
20
|
+
B = 0.75
|
|
21
|
+
# Score multiplier for matches found by prefix or typo expansion.
|
|
22
|
+
EXPANDED = 0.6
|
|
23
|
+
|
|
24
|
+
attr_reader :fields, :refs, :normalizer
|
|
25
|
+
|
|
26
|
+
# index is a Builder, its to_h, or the JSON it wrote (String or parsed).
|
|
27
|
+
# prefix: the last Latin word of a query also matches longer words.
|
|
28
|
+
# typos: Latin words of 5+ letters also match words one edit away.
|
|
29
|
+
def self.load(path, **options)
|
|
30
|
+
new(File.read(path, encoding: "UTF-8"), **options)
|
|
31
|
+
end
|
|
32
|
+
|
|
33
|
+
def initialize(index, prefix: true, typos: true, normalizer: nil)
|
|
34
|
+
data = case index
|
|
35
|
+
when Builder then index.to_h
|
|
36
|
+
when String then JSON.parse(index)
|
|
37
|
+
else index
|
|
38
|
+
end
|
|
39
|
+
unless data.is_a?(Hash) && data["format"] == Builder::FORMAT
|
|
40
|
+
raise ArgumentError, "cjk_index: unsupported index format #{data.is_a?(Hash) ? data['format'].inspect : data.class}"
|
|
41
|
+
end
|
|
42
|
+
|
|
43
|
+
@normalizer = normalizer || Normalizer.new(tables: data["variants"])
|
|
44
|
+
if @normalizer.tables != data["variants"]
|
|
45
|
+
raise ArgumentError, "cjk_index: index built with variant tables #{data['variants'].inspect} " \
|
|
46
|
+
"but this normalizer folds with #{@normalizer.tables.inspect}; rebuild both together"
|
|
47
|
+
end
|
|
48
|
+
|
|
49
|
+
@prefix = prefix
|
|
50
|
+
@typos = typos
|
|
51
|
+
@fields = data["fields"]
|
|
52
|
+
@boosts = data["boosts"]
|
|
53
|
+
@refs = data["refs"]
|
|
54
|
+
@lengths = data["lengths"]
|
|
55
|
+
@tokens = data["tokens"]
|
|
56
|
+
nf = @fields.length
|
|
57
|
+
avg = Array.new(nf, 0.0)
|
|
58
|
+
@lengths.each_with_index { |len, i| avg[i % nf] += len }
|
|
59
|
+
@avg_length = avg.map { |s| (v = s / [1, @refs.length].max).zero? ? 1 : v }
|
|
60
|
+
@latin_tokens = @tokens.keys.select { |t| latin?(t) }.sort
|
|
61
|
+
end
|
|
62
|
+
|
|
63
|
+
# Every phrase of the query must occur in the document (in any field),
|
|
64
|
+
# with its tokens adjacent and in order. Returns [{ ref:, score:, fields: }]
|
|
65
|
+
# with the best match first; fields lists the fields that matched.
|
|
66
|
+
def search(query)
|
|
67
|
+
nf = @fields.length
|
|
68
|
+
n = @refs.length
|
|
69
|
+
qp = Tokenizer.phrases(query, normalizer: @normalizer)
|
|
70
|
+
return [] if qp.empty?
|
|
71
|
+
|
|
72
|
+
scores = Hash.new(0.0)
|
|
73
|
+
matched_fields = Hash.new { |h, k| h[k] = [] }
|
|
74
|
+
docs_so_far = nil
|
|
75
|
+
|
|
76
|
+
qp.each_with_index do |phrase, q|
|
|
77
|
+
expanded = false
|
|
78
|
+
parts = phrase.each_with_index.map do |(token, offset), i|
|
|
79
|
+
tokens, exp = alternatives(token, q == qp.length - 1 && i == phrase.length - 1)
|
|
80
|
+
expanded ||= exp
|
|
81
|
+
[offset, positions(tokens)]
|
|
82
|
+
end
|
|
83
|
+
return [] if parts.any? { |_, pos| pos.empty? }
|
|
84
|
+
|
|
85
|
+
parts = parts.sort_by.with_index { |(_, pos), i| [pos.size, i] }
|
|
86
|
+
rare_offset, rare = parts.first
|
|
87
|
+
hits = {}
|
|
88
|
+
rare.each do |key, set|
|
|
89
|
+
tf = set.count do |pos|
|
|
90
|
+
start = pos - rare_offset
|
|
91
|
+
parts.drop(1).all? { |offset, other| other[key]&.include?(start + offset) }
|
|
92
|
+
end
|
|
93
|
+
hits[key] = tf if tf.positive?
|
|
94
|
+
end
|
|
95
|
+
docs = hits.keys.map { |key| key / nf }.to_set
|
|
96
|
+
docs &= docs_so_far if docs_so_far
|
|
97
|
+
return [] if docs.empty?
|
|
98
|
+
|
|
99
|
+
docs_so_far = docs
|
|
100
|
+
idf = Math.log(1 + ((n - docs.size + 0.5) / (docs.size + 0.5)))
|
|
101
|
+
weight = expanded ? EXPANDED : 1
|
|
102
|
+
hits.each do |key, tf|
|
|
103
|
+
doc = key / nf
|
|
104
|
+
field = key % nf
|
|
105
|
+
norm = 1 - B + (B * @lengths[key] / @avg_length[field])
|
|
106
|
+
scores[doc] += weight * @boosts[field] * idf * (tf * (K1 + 1)) / (tf + (K1 * norm))
|
|
107
|
+
matched_fields[doc] |= [@fields[field]]
|
|
108
|
+
end
|
|
109
|
+
end
|
|
110
|
+
|
|
111
|
+
docs_so_far.sort_by { |doc| [-scores[doc], doc] }
|
|
112
|
+
.map { |doc| { ref: @refs[doc], score: scores[doc], fields: matched_fields[doc] } }
|
|
113
|
+
end
|
|
114
|
+
|
|
115
|
+
def highlight(text, query) = CJKIndex.highlight(text, query, normalizer: @normalizer)
|
|
116
|
+
|
|
117
|
+
def excerpt(text, query, length = 80) = CJKIndex.excerpt(text, query, length, normalizer: @normalizer)
|
|
118
|
+
|
|
119
|
+
private
|
|
120
|
+
|
|
121
|
+
def latin?(token) = !token.match?(Tokenizer::CJK)
|
|
122
|
+
|
|
123
|
+
# "doc * nf + field" => Set of positions, merged over the alternatives.
|
|
124
|
+
def positions(tokens)
|
|
125
|
+
nf = @fields.length
|
|
126
|
+
map = {}
|
|
127
|
+
tokens.each do |token|
|
|
128
|
+
postings = @tokens[token] or next
|
|
129
|
+
i = 0
|
|
130
|
+
while i < postings.length
|
|
131
|
+
count = postings[i + 2]
|
|
132
|
+
(map[(postings[i] * nf) + postings[i + 1]] ||= Set.new).merge(postings[i + 3, count])
|
|
133
|
+
i += 3 + count
|
|
134
|
+
end
|
|
135
|
+
end
|
|
136
|
+
map
|
|
137
|
+
end
|
|
138
|
+
|
|
139
|
+
# Alternatives for one query token: itself, and for Latin words the index
|
|
140
|
+
# words it is a prefix of (last word only) or one edit away from.
|
|
141
|
+
def alternatives(token, last)
|
|
142
|
+
return [[token], false] unless latin?(token)
|
|
143
|
+
|
|
144
|
+
out = @tokens.key?(token) ? [token] : []
|
|
145
|
+
if @prefix && last
|
|
146
|
+
from = @latin_tokens.bsearch_index { |k| k >= token } || @latin_tokens.length
|
|
147
|
+
@latin_tokens[from..].each do |k|
|
|
148
|
+
break unless k.start_with?(token)
|
|
149
|
+
|
|
150
|
+
out << k unless k == token
|
|
151
|
+
end
|
|
152
|
+
end
|
|
153
|
+
out.concat(@latin_tokens.select { |k| one_edit?(token, k) }) if @typos && !@tokens.key?(token) && token.length >= 5
|
|
154
|
+
[out, !out.empty? && out.first != token]
|
|
155
|
+
end
|
|
156
|
+
|
|
157
|
+
# true when a and b differ by exactly one insertion, deletion or substitution.
|
|
158
|
+
def one_edit?(a, b)
|
|
159
|
+
return false if a == b || (a.length - b.length).abs > 1
|
|
160
|
+
|
|
161
|
+
i = 0
|
|
162
|
+
i += 1 while i < a.length && i < b.length && a[i] == b[i]
|
|
163
|
+
if a.length == b.length then a[(i + 1)..] == b[(i + 1)..]
|
|
164
|
+
elsif a.length > b.length then a[(i + 1)..] == b[i..]
|
|
165
|
+
else a[i..] == b[(i + 1)..]
|
|
166
|
+
end
|
|
167
|
+
end
|
|
168
|
+
end
|
|
169
|
+
|
|
170
|
+
HTML_ESCAPES = { "&" => "&", "<" => "<", ">" => ">", '"' => """, "'" => "'" }.freeze
|
|
171
|
+
|
|
172
|
+
def self.escape_html(text) = text.gsub(/[&<>"']/, HTML_ESCAPES)
|
|
173
|
+
|
|
174
|
+
# [start, end) ranges, in characters of the original text, where a phrase
|
|
175
|
+
# of the query occurs after normalization.
|
|
176
|
+
def self.match_ranges(text, query, normalizer: Normalizer.default)
|
|
177
|
+
chars = text.to_s.chars
|
|
178
|
+
norm = +""
|
|
179
|
+
origin = []
|
|
180
|
+
chars.each_with_index do |ch, i|
|
|
181
|
+
folded = normalizer.normalize(ch)
|
|
182
|
+
# 々 depends on the previous character, so fold it in context.
|
|
183
|
+
folded = norm[-1] if folded == Normalizer::ITERATION_MARK && !norm.empty?
|
|
184
|
+
norm << folded
|
|
185
|
+
folded.length.times { origin << i }
|
|
186
|
+
end
|
|
187
|
+
needles = normalizer.normalize(query).split(Tokenizer::SEPARATOR).reject(&:empty?)
|
|
188
|
+
ranges = needles.flat_map do |needle|
|
|
189
|
+
found = []
|
|
190
|
+
from = 0
|
|
191
|
+
while (at = norm.index(needle, from))
|
|
192
|
+
found << [origin[at], origin[at + needle.length - 1] + 1]
|
|
193
|
+
from = at + needle.length
|
|
194
|
+
end
|
|
195
|
+
found
|
|
196
|
+
end
|
|
197
|
+
merged = ranges.sort_by(&:first).each_with_object([]) do |r, acc|
|
|
198
|
+
if acc.last && r[0] <= acc.last[1] then acc.last[1] = [acc.last[1], r[1]].max
|
|
199
|
+
else acc << r.dup
|
|
200
|
+
end
|
|
201
|
+
end
|
|
202
|
+
[chars, merged]
|
|
203
|
+
end
|
|
204
|
+
|
|
205
|
+
# HTML-escaped text with every match of the query wrapped in <mark>.
|
|
206
|
+
# 「東京帝國大學」 is marked for the query 「帝国大学」.
|
|
207
|
+
def self.highlight(text, query, normalizer: Normalizer.default)
|
|
208
|
+
chars, ranges = match_ranges(text, query, normalizer: normalizer)
|
|
209
|
+
at = 0
|
|
210
|
+
out = +""
|
|
211
|
+
ranges.each do |from, to|
|
|
212
|
+
out << escape_html(chars[at...from].join) << "<mark>" << escape_html(chars[from...to].join) << "</mark>"
|
|
213
|
+
at = to
|
|
214
|
+
end
|
|
215
|
+
out << escape_html(chars[at..].join)
|
|
216
|
+
end
|
|
217
|
+
|
|
218
|
+
# A highlighted window of about `length` characters around the first match.
|
|
219
|
+
def self.excerpt(text, query, length = 80, normalizer: Normalizer.default)
|
|
220
|
+
chars, ranges = match_ranges(text, query, normalizer: normalizer)
|
|
221
|
+
if ranges.empty?
|
|
222
|
+
return chars.length > length ? "#{escape_html(chars[0, length].join)}…" : escape_html(chars.join)
|
|
223
|
+
end
|
|
224
|
+
|
|
225
|
+
start = [0, ranges[0][0] - (length / 3)].max
|
|
226
|
+
stop = [chars.length, start + length].min
|
|
227
|
+
"#{start.positive? ? '…' : ''}#{highlight(chars[start...stop].join, query, normalizer: normalizer)}" \
|
|
228
|
+
"#{stop < chars.length ? '…' : ''}"
|
|
229
|
+
end
|
|
230
|
+
|
|
231
|
+
# Substring test on normalized text, ignoring whitespace.
|
|
232
|
+
def self.includes?(haystack, needle, normalizer: Normalizer.default)
|
|
233
|
+
strip = ->(s) { normalizer.normalize(s).gsub(/\s+/, "") }
|
|
234
|
+
strip.call(haystack).include?(strip.call(needle))
|
|
235
|
+
end
|
|
236
|
+
end
|
|
@@ -0,0 +1,114 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module CJKIndex
|
|
4
|
+
# Splits normalized text into search tokens.
|
|
5
|
+
#
|
|
6
|
+
# Runs of CJK characters (kanji, kana, hangul) have no spaces between words,
|
|
7
|
+
# so they become overlapping character bigrams: 「鳥瞰図」 -> 鳥瞰, 瞰図.
|
|
8
|
+
# Other runs (Latin letters, digits) stay whole words. Each token carries
|
|
9
|
+
# its offset (in characters of the normalized text), so a query can require
|
|
10
|
+
# its bigrams to be adjacent and in order: an exact substring match, found
|
|
11
|
+
# without a dictionary.
|
|
12
|
+
#
|
|
13
|
+
# The character classes are kept as code point ranges so that Runtime can
|
|
14
|
+
# emit exactly the same classes into the browser script, where the same
|
|
15
|
+
# character-by-character loop runs.
|
|
16
|
+
module Tokenizer
|
|
17
|
+
CJK_RANGES = [
|
|
18
|
+
[0x1100, 0x11FF], # hangul jamo (old hangul is written as jamo sequences)
|
|
19
|
+
[0x3005, 0x3005], # 々 iteration mark
|
|
20
|
+
[0x3007, 0x3007], # 〇
|
|
21
|
+
[0x3040, 0x309F], # hiragana
|
|
22
|
+
[0x30A0, 0x30FF], # katakana
|
|
23
|
+
[0x3130, 0x318F], # hangul compatibility jamo
|
|
24
|
+
[0x3400, 0x4DBF], # CJK extension A
|
|
25
|
+
[0x4E00, 0x9FFF], # CJK unified ideographs
|
|
26
|
+
[0xA960, 0xA97F], # hangul jamo extended-A
|
|
27
|
+
[0xAC00, 0xD7AF], # hangul syllables
|
|
28
|
+
[0xD7B0, 0xD7FF], # hangul jamo extended-B
|
|
29
|
+
[0xF900, 0xFAFF], # CJK compatibility ideographs
|
|
30
|
+
[0x20000, 0x2FA1F] # CJK extensions B- and compatibility supplement
|
|
31
|
+
].freeze
|
|
32
|
+
|
|
33
|
+
SEPARATOR_RANGES = [
|
|
34
|
+
[0x09, 0x0D], [0x20, 0x20], [0xA0, 0xA0], [0x1680, 0x1680],
|
|
35
|
+
[0x2000, 0x200A], [0x2028, 0x2029], [0x202F, 0x202F], [0x205F, 0x205F],
|
|
36
|
+
[0x3000, 0x3000], [0xFEFF, 0xFEFF]
|
|
37
|
+
].freeze
|
|
38
|
+
SEPARATOR_CHARS = "、。,.・:;!?「」『』()()[][]【】〈〉《》〔〕{}{}\"'`,.:;!?/\\|-‐―-~〜=+*&%$#@^<>"
|
|
39
|
+
|
|
40
|
+
module_function
|
|
41
|
+
|
|
42
|
+
# "[...]" character class usable in both Ruby and JavaScript (with /u).
|
|
43
|
+
def char_class(ranges, chars = "")
|
|
44
|
+
body = ranges.map do |from, to|
|
|
45
|
+
from == to ? escape(from) : "#{escape(from)}-#{escape(to)}"
|
|
46
|
+
end
|
|
47
|
+
body += chars.each_char.map { |c| escape(c.ord) }
|
|
48
|
+
"[#{body.join}]"
|
|
49
|
+
end
|
|
50
|
+
|
|
51
|
+
def escape(code)
|
|
52
|
+
format("\\u{%x}", code)
|
|
53
|
+
end
|
|
54
|
+
|
|
55
|
+
CJK = Regexp.new(char_class(CJK_RANGES))
|
|
56
|
+
SEPARATOR = Regexp.new(char_class(SEPARATOR_RANGES, SEPARATOR_CHARS))
|
|
57
|
+
|
|
58
|
+
# Token strings only. unigrams: true adds every single CJK character as
|
|
59
|
+
# well; the index needs them so that one-character queries (「竹」) and
|
|
60
|
+
# CJK characters between digits (「第1輯」) can be found.
|
|
61
|
+
def tokenize(text, unigrams: false, normalizer: Normalizer.default)
|
|
62
|
+
tokens_with_offsets(text, unigrams: unigrams, normalizer: normalizer).map(&:first)
|
|
63
|
+
end
|
|
64
|
+
|
|
65
|
+
# [[token, offset], ...] for the whole text. Offsets count characters of
|
|
66
|
+
# the normalized text, so the index and the query agree on them.
|
|
67
|
+
def tokens_with_offsets(text, unigrams: false, normalizer: Normalizer.default)
|
|
68
|
+
phrases(text, unigrams: unigrams, normalizer: normalizer).flatten(1)
|
|
69
|
+
end
|
|
70
|
+
|
|
71
|
+
# Splits at separators (spaces, punctuation) into phrases, each a list of
|
|
72
|
+
# [token, offset]. A query matches when every phrase occurs somewhere in
|
|
73
|
+
# the document, with the tokens of each phrase at the same relative offsets.
|
|
74
|
+
def phrases(text, unigrams: false, normalizer: Normalizer.default)
|
|
75
|
+
result = []
|
|
76
|
+
phrase = []
|
|
77
|
+
run = +""
|
|
78
|
+
run_cjk = nil
|
|
79
|
+
run_start = 0
|
|
80
|
+
flush = lambda do
|
|
81
|
+
next if run.empty?
|
|
82
|
+
|
|
83
|
+
if run_cjk
|
|
84
|
+
chars = run.chars
|
|
85
|
+
if chars.length == 1 || unigrams
|
|
86
|
+
chars.each_with_index { |c, i| phrase << [c, run_start + i] }
|
|
87
|
+
end
|
|
88
|
+
chars.each_cons(2).with_index { |(a, b), i| phrase << [a + b, run_start + i] }
|
|
89
|
+
else
|
|
90
|
+
phrase << [run.dup, run_start]
|
|
91
|
+
end
|
|
92
|
+
run = +""
|
|
93
|
+
end
|
|
94
|
+
|
|
95
|
+
normalizer.normalize(text).each_char.with_index do |ch, offset|
|
|
96
|
+
if ch.match?(SEPARATOR)
|
|
97
|
+
flush.call
|
|
98
|
+
result << phrase unless phrase.empty?
|
|
99
|
+
phrase = []
|
|
100
|
+
run_cjk = nil
|
|
101
|
+
next
|
|
102
|
+
end
|
|
103
|
+
cjk = ch.match?(CJK)
|
|
104
|
+
flush.call if !run_cjk.nil? && cjk != run_cjk
|
|
105
|
+
run_start = offset if run.empty?
|
|
106
|
+
run_cjk = cjk
|
|
107
|
+
run << ch
|
|
108
|
+
end
|
|
109
|
+
flush.call
|
|
110
|
+
result << phrase unless phrase.empty?
|
|
111
|
+
result
|
|
112
|
+
end
|
|
113
|
+
end
|
|
114
|
+
end
|
data/lib/cjk_index.rb
ADDED
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require_relative "cjk_index/version"
|
|
4
|
+
require_relative "cjk_index/normalizer"
|
|
5
|
+
require_relative "cjk_index/tokenizer"
|
|
6
|
+
require_relative "cjk_index/builder"
|
|
7
|
+
require_relative "cjk_index/runtime"
|
|
8
|
+
require_relative "cjk_index/searcher"
|
|
9
|
+
|
|
10
|
+
# Dictionary-free full-text search for Chinese, Japanese and Korean text:
|
|
11
|
+
# build the index in Ruby, query it in Ruby (Searcher) or in the browser
|
|
12
|
+
# (Runtime), with the same results.
|
|
13
|
+
module CJKIndex
|
|
14
|
+
def self.normalize(text, normalizer: Normalizer.default) = normalizer.normalize(text)
|
|
15
|
+
|
|
16
|
+
def self.tokenize(text, unigrams: false, normalizer: Normalizer.default)
|
|
17
|
+
Tokenizer.tokenize(text, unigrams: unigrams, normalizer: normalizer)
|
|
18
|
+
end
|
|
19
|
+
end
|
metadata
ADDED
|
@@ -0,0 +1,59 @@
|
|
|
1
|
+
--- !ruby/object:Gem::Specification
|
|
2
|
+
name: cjk_index
|
|
3
|
+
version: !ruby/object:Gem::Version
|
|
4
|
+
version: 0.1.0
|
|
5
|
+
platform: ruby
|
|
6
|
+
authors:
|
|
7
|
+
- Satoru Nakamura
|
|
8
|
+
bindir: bin
|
|
9
|
+
cert_chain: []
|
|
10
|
+
date: 1980-01-02 00:00:00.000000000 Z
|
|
11
|
+
dependencies: []
|
|
12
|
+
description: Builds a bigram search index in Ruby (with kana, old/new kanji and NFKC
|
|
13
|
+
normalization), and queries it in Ruby or with a small browser runtime that returns
|
|
14
|
+
the same results. Includes a Jekyll plugin and a CollectionBuilder preset.
|
|
15
|
+
executables: []
|
|
16
|
+
extensions: []
|
|
17
|
+
extra_rdoc_files: []
|
|
18
|
+
files:
|
|
19
|
+
- LICENSE
|
|
20
|
+
- LICENSE-OpenCC
|
|
21
|
+
- NOTICE
|
|
22
|
+
- README.md
|
|
23
|
+
- data/ja-kyujitai.tsv
|
|
24
|
+
- data/zh-hant-hans.tsv
|
|
25
|
+
- lib/cjk_index.rb
|
|
26
|
+
- lib/cjk_index/builder.rb
|
|
27
|
+
- lib/cjk_index/jekyll.rb
|
|
28
|
+
- lib/cjk_index/normalizer.rb
|
|
29
|
+
- lib/cjk_index/runtime.js
|
|
30
|
+
- lib/cjk_index/runtime.rb
|
|
31
|
+
- lib/cjk_index/searcher.rb
|
|
32
|
+
- lib/cjk_index/tokenizer.rb
|
|
33
|
+
- lib/cjk_index/version.rb
|
|
34
|
+
homepage: https://github.com/nakamura196/cjk_index
|
|
35
|
+
licenses:
|
|
36
|
+
- MIT
|
|
37
|
+
- Apache-2.0
|
|
38
|
+
metadata:
|
|
39
|
+
source_code_uri: https://github.com/nakamura196/cjk_index
|
|
40
|
+
rubygems_mfa_required: 'true'
|
|
41
|
+
rdoc_options: []
|
|
42
|
+
require_paths:
|
|
43
|
+
- lib
|
|
44
|
+
required_ruby_version: !ruby/object:Gem::Requirement
|
|
45
|
+
requirements:
|
|
46
|
+
- - ">="
|
|
47
|
+
- !ruby/object:Gem::Version
|
|
48
|
+
version: '3.1'
|
|
49
|
+
required_rubygems_version: !ruby/object:Gem::Requirement
|
|
50
|
+
requirements:
|
|
51
|
+
- - ">="
|
|
52
|
+
- !ruby/object:Gem::Version
|
|
53
|
+
version: '0'
|
|
54
|
+
requirements: []
|
|
55
|
+
rubygems_version: 4.0.3
|
|
56
|
+
specification_version: 4
|
|
57
|
+
summary: Dictionary-free full-text search for Chinese, Japanese and Korean in pure
|
|
58
|
+
Ruby and the browser
|
|
59
|
+
test_files: []
|