cjk_index 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +7 -0
- data/LICENSE +21 -0
- data/LICENSE-OpenCC +56 -0
- data/NOTICE +7 -0
- data/README.md +171 -0
- data/data/ja-kyujitai.tsv +292 -0
- data/data/zh-hant-hans.tsv +3207 -0
- data/lib/cjk_index/builder.rb +86 -0
- data/lib/cjk_index/jekyll.rb +127 -0
- data/lib/cjk_index/normalizer.rb +89 -0
- data/lib/cjk_index/runtime.js +318 -0
- data/lib/cjk_index/runtime.rb +26 -0
- data/lib/cjk_index/searcher.rb +236 -0
- data/lib/cjk_index/tokenizer.rb +114 -0
- data/lib/cjk_index/version.rb +5 -0
- data/lib/cjk_index.rb +19 -0
- metadata +59 -0
|
@@ -0,0 +1,86 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "json"
|
|
4
|
+
|
|
5
|
+
module CJKIndex
|
|
6
|
+
# Builds a prebuilt inverted index that the browser runtime loads as JSON.
|
|
7
|
+
#
|
|
8
|
+
# builder = CJKIndex::Builder.new(fields: %w[title creator], boosts: { "title" => 3 })
|
|
9
|
+
# builder.add("item1.html", "title" => "東京帝國大學", "creator" => "…")
|
|
10
|
+
# File.write("search-index.json", builder.to_json)
|
|
11
|
+
#
|
|
12
|
+
# Format (version 2):
|
|
13
|
+
# {
|
|
14
|
+
# "format": "cjk_index/2",
|
|
15
|
+
# "variants": ["ja"],
|
|
16
|
+
# "fields": ["title", "creator"],
|
|
17
|
+
# "boosts": [3, 1],
|
|
18
|
+
# "refs": ["item1.html", ...],
|
|
19
|
+
# "lengths": [len(doc0.title), len(doc0.creator), len(doc1.title), ...],
|
|
20
|
+
# "tokens": { "東京": [doc, field, n, pos1, ..., posn, doc, field, n, ...], ... }
|
|
21
|
+
# }
|
|
22
|
+
# doc and field are positions in "refs" and "fields"; n is the number of
|
|
23
|
+
# occurrences and pos are character offsets in the normalized field text.
|
|
24
|
+
# Offsets let the runtime require the bigrams of a query to be adjacent
|
|
25
|
+
# (exact substring match); lengths feed BM25 ranking.
|
|
26
|
+
class Builder
|
|
27
|
+
FORMAT = "cjk_index/2"
|
|
28
|
+
|
|
29
|
+
# Gap between the values of a multi-valued field, so that a phrase cannot
|
|
30
|
+
# match across two values.
|
|
31
|
+
VALUE_GAP = 1
|
|
32
|
+
|
|
33
|
+
attr_reader :fields, :refs, :normalizer
|
|
34
|
+
|
|
35
|
+
# normalizer must match the one the runtime script was generated with
|
|
36
|
+
# (Runtime.source(normalizer:)); the index records its tables and the
|
|
37
|
+
# runtime refuses an index built with different ones.
|
|
38
|
+
def initialize(fields:, boosts: {}, normalizer: Normalizer.default)
|
|
39
|
+
raise ArgumentError, "fields must not be empty" if fields.empty?
|
|
40
|
+
|
|
41
|
+
@normalizer = normalizer
|
|
42
|
+
@fields = fields.map(&:to_s)
|
|
43
|
+
@boosts = @fields.map { |f| (boosts[f] || boosts[f.to_sym] || 1).to_f }
|
|
44
|
+
@refs = []
|
|
45
|
+
@lengths = []
|
|
46
|
+
@postings = Hash.new { |h, k| h[k] = [] }
|
|
47
|
+
end
|
|
48
|
+
|
|
49
|
+
# doc is a Hash of field name => value. Array values are indexed as
|
|
50
|
+
# separate strings (e.g. multi-valued subject fields).
|
|
51
|
+
def add(ref, doc)
|
|
52
|
+
doc_index = @refs.length
|
|
53
|
+
@refs << ref.to_s
|
|
54
|
+
@fields.each_with_index do |field, field_index|
|
|
55
|
+
value = doc[field] || doc[field.to_sym]
|
|
56
|
+
positions = Hash.new { |h, k| h[k] = [] }
|
|
57
|
+
base = 0
|
|
58
|
+
Array(value).each do |v|
|
|
59
|
+
text = v.to_s
|
|
60
|
+
Tokenizer.tokens_with_offsets(text, unigrams: true, normalizer: @normalizer)
|
|
61
|
+
.each { |t, offset| positions[t] << (base + offset) }
|
|
62
|
+
base += @normalizer.normalize(text).length + VALUE_GAP
|
|
63
|
+
end
|
|
64
|
+
@lengths << [base - VALUE_GAP, 0].max
|
|
65
|
+
positions.each { |token, offsets| @postings[token].push(doc_index, field_index, offsets.length, *offsets) }
|
|
66
|
+
end
|
|
67
|
+
self
|
|
68
|
+
end
|
|
69
|
+
|
|
70
|
+
def to_h
|
|
71
|
+
{
|
|
72
|
+
"format" => FORMAT,
|
|
73
|
+
"variants" => @normalizer.tables,
|
|
74
|
+
"fields" => @fields,
|
|
75
|
+
"boosts" => @boosts.map { |b| b == b.to_i ? b.to_i : b },
|
|
76
|
+
"refs" => @refs,
|
|
77
|
+
"lengths" => @lengths,
|
|
78
|
+
"tokens" => @postings.sort.to_h
|
|
79
|
+
}
|
|
80
|
+
end
|
|
81
|
+
|
|
82
|
+
def to_json(*_args)
|
|
83
|
+
JSON.generate(to_h)
|
|
84
|
+
end
|
|
85
|
+
end
|
|
86
|
+
end
|
|
@@ -0,0 +1,127 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "jekyll"
|
|
4
|
+
require_relative "../cjk_index"
|
|
5
|
+
|
|
6
|
+
module CJKIndex
|
|
7
|
+
# Jekyll plugin. Add `gem "cjk_index"` to the :jekyll_plugins group (or list
|
|
8
|
+
# "cjk_index/jekyll" under `plugins:`), then configure in _config.yml:
|
|
9
|
+
#
|
|
10
|
+
# cjk_index:
|
|
11
|
+
# output: assets/cjk-index # where files are written (default)
|
|
12
|
+
# variants: [ja] # variant tables: ja (default), zh (Traditional -> Simplified)
|
|
13
|
+
# indexes:
|
|
14
|
+
# - name: posts # -> assets/cjk-index/posts.json
|
|
15
|
+
# collection: posts # a Jekyll collection, or...
|
|
16
|
+
# fields: [title, tags, content]
|
|
17
|
+
# boosts: { title: 3 }
|
|
18
|
+
# - name: items
|
|
19
|
+
# data: metadata # ...rows of _data/metadata.csv
|
|
20
|
+
# ref: objectid # column used as the result's ref
|
|
21
|
+
# fields: [title, creator, subject]
|
|
22
|
+
# - name: search
|
|
23
|
+
# preset: collectionbuilder # reads CollectionBuilder's own config
|
|
24
|
+
#
|
|
25
|
+
# The browser script is written to <output>/cjk-index.js.
|
|
26
|
+
module Jekyll
|
|
27
|
+
DEFAULT_OUTPUT = "assets/cjk-index"
|
|
28
|
+
RUNTIME_NAME = "cjk-index.js"
|
|
29
|
+
|
|
30
|
+
# A file whose content is produced at build time (no source file, no Liquid).
|
|
31
|
+
class GeneratedFile < ::Jekyll::StaticFile
|
|
32
|
+
def initialize(site, dir, name, content)
|
|
33
|
+
super(site, site.source, dir, name)
|
|
34
|
+
@generated_content = content
|
|
35
|
+
end
|
|
36
|
+
|
|
37
|
+
def write(dest)
|
|
38
|
+
path = destination(dest)
|
|
39
|
+
return false if File.exist?(path) && File.read(path, encoding: "UTF-8") == @generated_content
|
|
40
|
+
|
|
41
|
+
FileUtils.mkdir_p(File.dirname(path))
|
|
42
|
+
File.write(path, @generated_content)
|
|
43
|
+
true
|
|
44
|
+
end
|
|
45
|
+
end
|
|
46
|
+
|
|
47
|
+
class Generator < ::Jekyll::Generator
|
|
48
|
+
safe true
|
|
49
|
+
priority :lowest
|
|
50
|
+
|
|
51
|
+
def generate(site)
|
|
52
|
+
config = site.config["cjk_index"]
|
|
53
|
+
return unless config.is_a?(Hash)
|
|
54
|
+
|
|
55
|
+
output = config.fetch("output", DEFAULT_OUTPUT).sub(%r{\A/+}, "").sub(%r{/+\z}, "")
|
|
56
|
+
normalizer = Normalizer.new(tables: Array(config.fetch("variants", Normalizer::DEFAULT_TABLES)))
|
|
57
|
+
site.static_files << GeneratedFile.new(site, output, RUNTIME_NAME, Runtime.source(normalizer: normalizer))
|
|
58
|
+
Array(config["indexes"]).each do |spec|
|
|
59
|
+
builder = build(site, spec, normalizer)
|
|
60
|
+
name = "#{spec.fetch('name')}.json"
|
|
61
|
+
site.static_files << GeneratedFile.new(site, output, name, builder.to_json)
|
|
62
|
+
::Jekyll.logger.info "cjk_index:", "#{output}/#{name} (#{builder.refs.length} documents)"
|
|
63
|
+
end
|
|
64
|
+
end
|
|
65
|
+
|
|
66
|
+
private
|
|
67
|
+
|
|
68
|
+
def build(site, spec, normalizer)
|
|
69
|
+
return Presets::CollectionBuilder.build(site, spec, normalizer) if spec["preset"] == "collectionbuilder"
|
|
70
|
+
raise ::Jekyll::Errors::FatalException, "cjk_index: unknown preset #{spec['preset']}" if spec["preset"]
|
|
71
|
+
|
|
72
|
+
builder = Builder.new(fields: Array(spec.fetch("fields")), boosts: spec.fetch("boosts", {}),
|
|
73
|
+
normalizer: normalizer)
|
|
74
|
+
if spec["collection"]
|
|
75
|
+
docs = site.collections.fetch(spec["collection"]) do
|
|
76
|
+
raise ::Jekyll::Errors::FatalException, "cjk_index: no collection #{spec['collection']}"
|
|
77
|
+
end.docs
|
|
78
|
+
docs.each { |doc| builder.add(doc.url, document_fields(doc, builder.fields)) }
|
|
79
|
+
elsif spec["data"]
|
|
80
|
+
ref = spec.fetch("ref", "id")
|
|
81
|
+
Array(site.data[spec["data"]]).each do |row|
|
|
82
|
+
builder.add(row[ref], row) unless row[ref].to_s.empty?
|
|
83
|
+
end
|
|
84
|
+
else
|
|
85
|
+
raise ::Jekyll::Errors::FatalException, "cjk_index: index #{spec['name']} needs collection or data"
|
|
86
|
+
end
|
|
87
|
+
builder
|
|
88
|
+
end
|
|
89
|
+
|
|
90
|
+
def document_fields(doc, fields)
|
|
91
|
+
fields.to_h { |f| [f, f == "content" ? doc.content : doc.data[f]] }
|
|
92
|
+
end
|
|
93
|
+
end
|
|
94
|
+
|
|
95
|
+
module Presets
|
|
96
|
+
# Mirrors assets/js/lunr-store.js of CollectionBuilder (collectionbuilder-csv):
|
|
97
|
+
# items with an objectid, child objects only when the theme asks for them,
|
|
98
|
+
# fields marked index=true in _data/config-search.csv, and the same "id"
|
|
99
|
+
# values, so results can be looked up in the existing search store.
|
|
100
|
+
module CollectionBuilder
|
|
101
|
+
module_function
|
|
102
|
+
|
|
103
|
+
def build(site, spec, normalizer)
|
|
104
|
+
rows = Array(site.data[site.config["metadata"]])
|
|
105
|
+
children = site.data.dig("theme", "search-child-objects") == true
|
|
106
|
+
fields = Array(site.data["config-search"])
|
|
107
|
+
.select { |f| f["index"].to_s == "true" }
|
|
108
|
+
.map { |f| f["field"] }
|
|
109
|
+
boosts = spec.fetch("boosts") { fields.first ? { fields.first => 3 } : {} }
|
|
110
|
+
builder = Builder.new(fields: fields, boosts: boosts, normalizer: normalizer)
|
|
111
|
+
rows.each do |row|
|
|
112
|
+
next if row["objectid"].to_s.empty?
|
|
113
|
+
next if !row["parentid"].to_s.empty? && !children
|
|
114
|
+
|
|
115
|
+
builder.add(ref_for(row), row.to_h { |k, v| [k, v.is_a?(String) ? v.split.join(" ") : v] })
|
|
116
|
+
end
|
|
117
|
+
builder
|
|
118
|
+
end
|
|
119
|
+
|
|
120
|
+
def ref_for(row)
|
|
121
|
+
parent = row["parentid"].to_s
|
|
122
|
+
parent.empty? ? "#{row['objectid']}.html" : "#{parent}.html##{row['objectid']}"
|
|
123
|
+
end
|
|
124
|
+
end
|
|
125
|
+
end
|
|
126
|
+
end
|
|
127
|
+
end
|
|
@@ -0,0 +1,89 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module CJKIndex
|
|
4
|
+
# Folds text so that spelling variants a reader treats as "the same" compare
|
|
5
|
+
# equal: NFKC (full-width -> half-width etc.), lowercase, katakana ->
|
|
6
|
+
# hiragana, character variants from the selected tables, and the iteration
|
|
7
|
+
# mark 々 expanded to the preceding character.
|
|
8
|
+
#
|
|
9
|
+
# The same rules run in the browser (see Runtime), so an index built here
|
|
10
|
+
# and a query typed there are folded identically.
|
|
11
|
+
#
|
|
12
|
+
# Variant tables (data/*.tsv):
|
|
13
|
+
# ja Japanese old forms -> modern forms (圖 -> 図). Default.
|
|
14
|
+
# zh Traditional -> Simplified Chinese, from OpenCC. Opt in: it also
|
|
15
|
+
# merges characters that are distinct in Japanese (後 and 后).
|
|
16
|
+
# With several tables, characters linked by any table fold to one form,
|
|
17
|
+
# preferring the target of the table listed first: with [ja, zh],
|
|
18
|
+
# 圖, 図 and 图 all become 図.
|
|
19
|
+
class Normalizer
|
|
20
|
+
DATA_DIR = File.expand_path("../../data", __dir__)
|
|
21
|
+
TABLES = {
|
|
22
|
+
"ja" => "ja-kyujitai.tsv",
|
|
23
|
+
"zh" => "zh-hant-hans.tsv"
|
|
24
|
+
}.freeze
|
|
25
|
+
DEFAULT_TABLES = %w[ja].freeze
|
|
26
|
+
|
|
27
|
+
# Katakana ァ (U+30A1) .. ヶ (U+30F6) map to hiragana by subtracting 0x60.
|
|
28
|
+
KATAKANA_FIRST = 0x30A1
|
|
29
|
+
KATAKANA_LAST = 0x30F6
|
|
30
|
+
KANA_OFFSET = 0x60
|
|
31
|
+
ITERATION_MARK = "々"
|
|
32
|
+
|
|
33
|
+
def self.default
|
|
34
|
+
@default ||= new
|
|
35
|
+
end
|
|
36
|
+
|
|
37
|
+
def self.read_table(name)
|
|
38
|
+
file = TABLES.fetch(name) { raise ArgumentError, "unknown variant table #{name.inspect} (#{TABLES.keys.join(', ')})" }
|
|
39
|
+
File.readlines(File.join(DATA_DIR, file), chomp: true, encoding: "UTF-8")
|
|
40
|
+
.reject { |line| line.empty? || line.start_with?("#") }
|
|
41
|
+
.map { |line| line.split("\t", 2) }
|
|
42
|
+
end
|
|
43
|
+
|
|
44
|
+
attr_reader :tables, :variants
|
|
45
|
+
|
|
46
|
+
def initialize(tables: DEFAULT_TABLES)
|
|
47
|
+
@tables = tables.map(&:to_s).uniq.freeze
|
|
48
|
+
@variants = build_variants.freeze
|
|
49
|
+
end
|
|
50
|
+
|
|
51
|
+
def normalize(text)
|
|
52
|
+
return "" if text.nil?
|
|
53
|
+
|
|
54
|
+
prev = nil
|
|
55
|
+
text.to_s.unicode_normalize(:nfkc).downcase.each_char.map do |ch|
|
|
56
|
+
code = ch.ord
|
|
57
|
+
ch = (code - KANA_OFFSET).chr(Encoding::UTF_8) if code.between?(KATAKANA_FIRST, KATAKANA_LAST)
|
|
58
|
+
ch = @variants.fetch(ch, ch)
|
|
59
|
+
ch = prev if ch == ITERATION_MARK && prev
|
|
60
|
+
prev = ch
|
|
61
|
+
end.join
|
|
62
|
+
end
|
|
63
|
+
|
|
64
|
+
private
|
|
65
|
+
|
|
66
|
+
# Union-find over all pairs; each class folds to the target preferred by
|
|
67
|
+
# the earliest table (lowest code point among equals, for determinism).
|
|
68
|
+
def build_variants
|
|
69
|
+
parent = {}
|
|
70
|
+
find = lambda do |x|
|
|
71
|
+
parent[x] ||= x
|
|
72
|
+
parent[x] = find.call(parent[x]) unless parent[x] == x
|
|
73
|
+
parent[x]
|
|
74
|
+
end
|
|
75
|
+
rank = {}
|
|
76
|
+
@tables.each_with_index do |name, priority|
|
|
77
|
+
self.class.read_table(name).each do |from, to|
|
|
78
|
+
parent[find.call(from)] = find.call(to)
|
|
79
|
+
rank[to] = [rank.fetch(to, [priority, to.ord]), [priority, to.ord]].min
|
|
80
|
+
end
|
|
81
|
+
end
|
|
82
|
+
classes = parent.keys.group_by { |x| find.call(x) }
|
|
83
|
+
classes.each_value.with_object({}) do |members, map|
|
|
84
|
+
canonical = members.min_by { |m| rank.fetch(m, [Float::INFINITY, m.ord]) }
|
|
85
|
+
members.each { |m| map[m] = canonical unless m == canonical }
|
|
86
|
+
end
|
|
87
|
+
end
|
|
88
|
+
end
|
|
89
|
+
end
|
|
@@ -0,0 +1,318 @@
|
|
|
1
|
+
/*!
|
|
2
|
+
* cjk_index runtime — queries an index built by the cjk_index Ruby gem.
|
|
3
|
+
* Generated file: the variant table and character classes below are written
|
|
4
|
+
* by CJKIndex::Runtime so that they match the Ruby side exactly.
|
|
5
|
+
* License: MIT
|
|
6
|
+
*/
|
|
7
|
+
(function (global) {
|
|
8
|
+
'use strict';
|
|
9
|
+
|
|
10
|
+
var VARIANTS = /*@VARIANTS@*/{};
|
|
11
|
+
var TABLES = /*@TABLES@*/[];
|
|
12
|
+
var CJK = new RegExp(/*@CJK@*/'', 'u');
|
|
13
|
+
var SEPARATOR = new RegExp(/*@SEPARATOR@*/'', 'u');
|
|
14
|
+
var FORMAT = 'cjk_index/2';
|
|
15
|
+
|
|
16
|
+
// BM25 parameters (the usual defaults, as in Lucene and Pagefind).
|
|
17
|
+
var K1 = 1.2;
|
|
18
|
+
var B = 0.75;
|
|
19
|
+
// Score multiplier for matches found by prefix or typo expansion.
|
|
20
|
+
var EXPANDED = 0.6;
|
|
21
|
+
|
|
22
|
+
// Folds one character; prev is the previous folded character (for 々).
|
|
23
|
+
function foldChar(ch, prev) {
|
|
24
|
+
var c = ch.codePointAt(0);
|
|
25
|
+
if (c >= 0x30a1 && c <= 0x30f6) ch = String.fromCodePoint(c - 0x60);
|
|
26
|
+
if (VARIANTS[ch]) ch = VARIANTS[ch];
|
|
27
|
+
if (ch === '々' && prev) ch = prev;
|
|
28
|
+
return ch;
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
function normalize(text) {
|
|
32
|
+
if (text === null || text === undefined) return '';
|
|
33
|
+
var out = '';
|
|
34
|
+
var prev = null;
|
|
35
|
+
for (var ch of String(text).normalize('NFKC').toLowerCase()) {
|
|
36
|
+
prev = foldChar(ch, prev);
|
|
37
|
+
out += prev;
|
|
38
|
+
}
|
|
39
|
+
return out;
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
// Same loop as CJKIndex::Tokenizer.phrases in Ruby: split at separators
|
|
43
|
+
// into phrases of [token, offset]; CJK runs become bigrams (plus unigrams
|
|
44
|
+
// when asked), other runs stay words. Offsets count code points.
|
|
45
|
+
function phrases(text, unigrams) {
|
|
46
|
+
var result = [];
|
|
47
|
+
var phrase = [];
|
|
48
|
+
var run = [];
|
|
49
|
+
var runCjk = null;
|
|
50
|
+
var runStart = 0;
|
|
51
|
+
var flush = function () {
|
|
52
|
+
if (!run.length) return;
|
|
53
|
+
if (runCjk) {
|
|
54
|
+
if (run.length === 1 || unigrams) {
|
|
55
|
+
for (var i = 0; i < run.length; i++) phrase.push([run[i], runStart + i]);
|
|
56
|
+
}
|
|
57
|
+
for (var j = 0; j < run.length - 1; j++) phrase.push([run[j] + run[j + 1], runStart + j]);
|
|
58
|
+
} else {
|
|
59
|
+
phrase.push([run.join(''), runStart]);
|
|
60
|
+
}
|
|
61
|
+
run = [];
|
|
62
|
+
};
|
|
63
|
+
var offset = 0;
|
|
64
|
+
for (var ch of normalize(text)) {
|
|
65
|
+
if (SEPARATOR.test(ch)) {
|
|
66
|
+
flush();
|
|
67
|
+
if (phrase.length) result.push(phrase);
|
|
68
|
+
phrase = [];
|
|
69
|
+
runCjk = null;
|
|
70
|
+
} else {
|
|
71
|
+
var cjk = CJK.test(ch);
|
|
72
|
+
if (runCjk !== null && cjk !== runCjk) flush();
|
|
73
|
+
if (!run.length) runStart = offset;
|
|
74
|
+
runCjk = cjk;
|
|
75
|
+
run.push(ch);
|
|
76
|
+
}
|
|
77
|
+
offset++;
|
|
78
|
+
}
|
|
79
|
+
flush();
|
|
80
|
+
if (phrase.length) result.push(phrase);
|
|
81
|
+
return result;
|
|
82
|
+
}
|
|
83
|
+
|
|
84
|
+
function tokenize(text, unigrams) {
|
|
85
|
+
var out = [];
|
|
86
|
+
phrases(text, unigrams).forEach(function (p) { p.forEach(function (t) { out.push(t[0]); }); });
|
|
87
|
+
return out;
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
// Substring test on normalized text, ignoring whitespace. Useful for
|
|
91
|
+
// filters that do not need the index (e.g. a browse page).
|
|
92
|
+
function includes(haystack, needle) {
|
|
93
|
+
var strip = function (s) { return normalize(s).replace(/\s+/g, ''); };
|
|
94
|
+
return strip(haystack).indexOf(strip(needle)) !== -1;
|
|
95
|
+
}
|
|
96
|
+
|
|
97
|
+
function isLatin(token) { return !CJK.test(token); }
|
|
98
|
+
|
|
99
|
+
// true when a and b differ by exactly one insertion, deletion or substitution.
|
|
100
|
+
function oneEdit(a, b) {
|
|
101
|
+
if (a === b || Math.abs(a.length - b.length) > 1) return false;
|
|
102
|
+
var i = 0;
|
|
103
|
+
while (i < a.length && i < b.length && a[i] === b[i]) i++;
|
|
104
|
+
if (a.length === b.length) return a.slice(i + 1) === b.slice(i + 1);
|
|
105
|
+
if (a.length > b.length) return a.slice(i + 1) === b.slice(i);
|
|
106
|
+
return a.slice(i) === b.slice(i + 1);
|
|
107
|
+
}
|
|
108
|
+
|
|
109
|
+
function Index(data, options) {
|
|
110
|
+
if (!data || data.format !== FORMAT) {
|
|
111
|
+
throw new Error('cjk_index: unsupported index format ' + (data && data.format));
|
|
112
|
+
}
|
|
113
|
+
if (JSON.stringify(data.variants) !== JSON.stringify(TABLES)) {
|
|
114
|
+
throw new Error('cjk_index: index built with variant tables ' + JSON.stringify(data.variants) +
|
|
115
|
+
' but this script folds with ' + JSON.stringify(TABLES) + '; rebuild both together');
|
|
116
|
+
}
|
|
117
|
+
options = options || {};
|
|
118
|
+
this.prefix = options.prefix !== false; // last Latin word matches as a prefix
|
|
119
|
+
this.typos = options.typos !== false; // Latin words of 5+ letters allow one edit
|
|
120
|
+
this.fields = data.fields;
|
|
121
|
+
this.boosts = data.boosts;
|
|
122
|
+
this.refs = data.refs;
|
|
123
|
+
this.lengths = data.lengths;
|
|
124
|
+
this.tokens = data.tokens;
|
|
125
|
+
var nf = this.fields.length;
|
|
126
|
+
var avg = new Array(nf).fill(0);
|
|
127
|
+
for (var i = 0; i < this.lengths.length; i++) avg[i % nf] += this.lengths[i];
|
|
128
|
+
this.avgLength = avg.map(function (s) { return s / Math.max(1, data.refs.length) || 1; });
|
|
129
|
+
this.latinTokens = Object.keys(this.tokens).filter(isLatin).sort();
|
|
130
|
+
}
|
|
131
|
+
|
|
132
|
+
// Map of "doc,field" key -> Set of positions, merged over the alternatives.
|
|
133
|
+
Index.prototype._positions = function (tokens) {
|
|
134
|
+
var nf = this.fields.length;
|
|
135
|
+
var map = new Map();
|
|
136
|
+
for (var t = 0; t < tokens.length; t++) {
|
|
137
|
+
var p = this.tokens[tokens[t]];
|
|
138
|
+
if (!p) continue;
|
|
139
|
+
for (var i = 0; i < p.length;) {
|
|
140
|
+
var key = p[i] * nf + p[i + 1];
|
|
141
|
+
var n = p[i + 2];
|
|
142
|
+
var set = map.get(key);
|
|
143
|
+
if (!set) { set = new Set(); map.set(key, set); }
|
|
144
|
+
for (var k = 0; k < n; k++) set.add(p[i + 3 + k]);
|
|
145
|
+
i += 3 + n;
|
|
146
|
+
}
|
|
147
|
+
}
|
|
148
|
+
return map;
|
|
149
|
+
};
|
|
150
|
+
|
|
151
|
+
// Alternatives for one query token: itself, and for Latin words the index
|
|
152
|
+
// words it is a prefix of (last word only) or one edit away from.
|
|
153
|
+
Index.prototype._alternatives = function (token, isLast) {
|
|
154
|
+
if (!isLatin(token)) return { tokens: [token], expanded: false };
|
|
155
|
+
var out = this.tokens[token] ? [token] : [];
|
|
156
|
+
var keys = this.latinTokens;
|
|
157
|
+
if (this.prefix && isLast) {
|
|
158
|
+
var lo = 0, hi = keys.length;
|
|
159
|
+
while (lo < hi) { var mid = (lo + hi) >> 1; if (keys[mid] < token) lo = mid + 1; else hi = mid; }
|
|
160
|
+
for (var i = lo; i < keys.length && keys[i].indexOf(token) === 0; i++) if (keys[i] !== token) out.push(keys[i]);
|
|
161
|
+
}
|
|
162
|
+
if (this.typos && !this.tokens[token] && Array.from(token).length >= 5) {
|
|
163
|
+
keys.forEach(function (k) { if (oneEdit(token, k)) out.push(k); });
|
|
164
|
+
}
|
|
165
|
+
return { tokens: out, expanded: out.length > 0 && out[0] !== token };
|
|
166
|
+
};
|
|
167
|
+
|
|
168
|
+
// Every phrase of the query must occur in the document (in any field),
|
|
169
|
+
// with its tokens adjacent and in order. Returns [{ref, score, fields}]
|
|
170
|
+
// with the best match first; fields lists the names of the fields that matched.
|
|
171
|
+
Index.prototype.search = function (query) {
|
|
172
|
+
var self = this;
|
|
173
|
+
var nf = this.fields.length;
|
|
174
|
+
var n = this.refs.length;
|
|
175
|
+
var qp = phrases(query, false);
|
|
176
|
+
if (!qp.length) return [];
|
|
177
|
+
var scores = new Map();
|
|
178
|
+
var matchedFields = new Map();
|
|
179
|
+
var docsSoFar = null;
|
|
180
|
+
|
|
181
|
+
for (var q = 0; q < qp.length; q++) {
|
|
182
|
+
var phrase = qp[q];
|
|
183
|
+
var expanded = false;
|
|
184
|
+
var parts = phrase.map(function (tok, i) {
|
|
185
|
+
var alt = self._alternatives(tok[0], q === qp.length - 1 && i === phrase.length - 1);
|
|
186
|
+
if (alt.expanded) expanded = true;
|
|
187
|
+
return { offset: tok[1], positions: self._positions(alt.tokens) };
|
|
188
|
+
});
|
|
189
|
+
if (parts.some(function (p) { return p.positions.size === 0; })) return [];
|
|
190
|
+
parts.sort(function (a, b) { return a.positions.size - b.positions.size; });
|
|
191
|
+
var rare = parts[0];
|
|
192
|
+
var hits = new Map(); // "doc,field" key -> phrase occurrences
|
|
193
|
+
rare.positions.forEach(function (set, key) {
|
|
194
|
+
var tf = 0;
|
|
195
|
+
set.forEach(function (pos) {
|
|
196
|
+
var start = pos - rare.offset;
|
|
197
|
+
for (var i = 1; i < parts.length; i++) {
|
|
198
|
+
var other = parts[i].positions.get(key);
|
|
199
|
+
if (!other || !other.has(start + parts[i].offset)) return;
|
|
200
|
+
}
|
|
201
|
+
tf++;
|
|
202
|
+
});
|
|
203
|
+
if (tf) hits.set(key, tf);
|
|
204
|
+
});
|
|
205
|
+
var docs = new Set();
|
|
206
|
+
hits.forEach(function (tf, key) { docs.add(Math.floor(key / nf)); });
|
|
207
|
+
if (!docs.size) return [];
|
|
208
|
+
if (docsSoFar) {
|
|
209
|
+
docs.forEach(function (d) { if (!docsSoFar.has(d)) docs.delete(d); });
|
|
210
|
+
if (!docs.size) return [];
|
|
211
|
+
}
|
|
212
|
+
docsSoFar = docs;
|
|
213
|
+
var idf = Math.log(1 + (n - docs.size + 0.5) / (docs.size + 0.5));
|
|
214
|
+
var weight = expanded ? EXPANDED : 1;
|
|
215
|
+
hits.forEach(function (tf, key) {
|
|
216
|
+
var doc = Math.floor(key / nf), field = key % nf;
|
|
217
|
+
var norm = 1 - B + B * self.lengths[key] / self.avgLength[field];
|
|
218
|
+
var s = weight * self.boosts[field] * idf * (tf * (K1 + 1)) / (tf + K1 * norm);
|
|
219
|
+
scores.set(doc, (scores.get(doc) || 0) + s);
|
|
220
|
+
var f = matchedFields.get(doc);
|
|
221
|
+
if (!f) { f = new Set(); matchedFields.set(doc, f); }
|
|
222
|
+
f.add(self.fields[field]);
|
|
223
|
+
});
|
|
224
|
+
}
|
|
225
|
+
|
|
226
|
+
var results = [];
|
|
227
|
+
docsSoFar.forEach(function (doc) {
|
|
228
|
+
results.push({ ref: self.refs[doc], score: scores.get(doc), fields: Array.from(matchedFields.get(doc)), doc: doc });
|
|
229
|
+
});
|
|
230
|
+
results.sort(function (a, b) { return b.score - a.score || a.doc - b.doc; });
|
|
231
|
+
return results.map(function (r) { return { ref: r.ref, score: r.score, fields: r.fields }; });
|
|
232
|
+
};
|
|
233
|
+
|
|
234
|
+
function escapeHtml(s) {
|
|
235
|
+
return s.replace(/[&<>"']/g, function (c) {
|
|
236
|
+
return { '&': '&', '<': '<', '>': '>', '"': '"', "'": ''' }[c];
|
|
237
|
+
});
|
|
238
|
+
}
|
|
239
|
+
|
|
240
|
+
// [start, end) ranges, in code points of the original text, where a phrase
|
|
241
|
+
// of the query occurs after normalization.
|
|
242
|
+
function matchRanges(text, query) {
|
|
243
|
+
var chars = Array.from(String(text === null || text === undefined ? '' : text));
|
|
244
|
+
var norm = [];
|
|
245
|
+
var origin = [];
|
|
246
|
+
var prev = null;
|
|
247
|
+
chars.forEach(function (ch, i) {
|
|
248
|
+
for (var c of ch.normalize('NFKC').toLowerCase()) {
|
|
249
|
+
prev = foldChar(c, prev);
|
|
250
|
+
norm.push(prev);
|
|
251
|
+
origin.push(i);
|
|
252
|
+
}
|
|
253
|
+
});
|
|
254
|
+
var normText = norm.join('');
|
|
255
|
+
var ranges = [];
|
|
256
|
+
var needles = [];
|
|
257
|
+
var current = '';
|
|
258
|
+
for (var qc of normalize(query)) {
|
|
259
|
+
if (SEPARATOR.test(qc)) { if (current) needles.push(current); current = ''; } else current += qc;
|
|
260
|
+
}
|
|
261
|
+
if (current) needles.push(current);
|
|
262
|
+
needles.forEach(function (needle) {
|
|
263
|
+
var len = Array.from(needle).length;
|
|
264
|
+
var from = 0;
|
|
265
|
+
var at;
|
|
266
|
+
while ((at = normText.indexOf(needle, from)) !== -1) {
|
|
267
|
+
var cp = Array.from(normText.slice(0, at)).length;
|
|
268
|
+
ranges.push([origin[cp], origin[cp + len - 1] + 1]);
|
|
269
|
+
from = at + needle.length;
|
|
270
|
+
}
|
|
271
|
+
});
|
|
272
|
+
ranges.sort(function (a, b) { return a[0] - b[0]; });
|
|
273
|
+
var merged = [];
|
|
274
|
+
ranges.forEach(function (r) {
|
|
275
|
+
var last = merged[merged.length - 1];
|
|
276
|
+
if (last && r[0] <= last[1]) last[1] = Math.max(last[1], r[1]); else merged.push(r.slice());
|
|
277
|
+
});
|
|
278
|
+
return { chars: chars, ranges: merged };
|
|
279
|
+
}
|
|
280
|
+
|
|
281
|
+
// HTML-escaped text with every match of the query wrapped in <mark>.
|
|
282
|
+
// 「東京帝國大學」 is marked for the query 「帝国大学」.
|
|
283
|
+
function highlight(text, query) {
|
|
284
|
+
var m = matchRanges(text, query);
|
|
285
|
+
var out = '';
|
|
286
|
+
var at = 0;
|
|
287
|
+
m.ranges.forEach(function (r) {
|
|
288
|
+
out += escapeHtml(m.chars.slice(at, r[0]).join('')) + '<mark>' + escapeHtml(m.chars.slice(r[0], r[1]).join('')) + '</mark>';
|
|
289
|
+
at = r[1];
|
|
290
|
+
});
|
|
291
|
+
return out + escapeHtml(m.chars.slice(at).join(''));
|
|
292
|
+
}
|
|
293
|
+
|
|
294
|
+
// A highlighted window of about `length` characters around the first match.
|
|
295
|
+
function excerpt(text, query, length) {
|
|
296
|
+
length = length || 80;
|
|
297
|
+
var m = matchRanges(text, query);
|
|
298
|
+
if (!m.ranges.length) return m.chars.length > length ? escapeHtml(m.chars.slice(0, length).join('')) + '…' : escapeHtml(m.chars.join(''));
|
|
299
|
+
var start = Math.max(0, m.ranges[0][0] - Math.floor(length / 3));
|
|
300
|
+
var end = Math.min(m.chars.length, start + length);
|
|
301
|
+
var part = m.chars.slice(start, end).join('');
|
|
302
|
+
return (start > 0 ? '…' : '') + highlight(part, query) + (end < m.chars.length ? '…' : '');
|
|
303
|
+
}
|
|
304
|
+
|
|
305
|
+
function load(url, options) {
|
|
306
|
+
return fetch(url).then(function (res) {
|
|
307
|
+
if (!res.ok) throw new Error('cjk_index: ' + url + ' returned ' + res.status);
|
|
308
|
+
return res.json();
|
|
309
|
+
}).then(function (data) { return new Index(data, options); });
|
|
310
|
+
}
|
|
311
|
+
|
|
312
|
+
var api = {
|
|
313
|
+
tables: TABLES, normalize: normalize, tokenize: tokenize, phrases: phrases, includes: includes,
|
|
314
|
+
highlight: highlight, excerpt: excerpt, Index: Index, load: load
|
|
315
|
+
};
|
|
316
|
+
if (typeof module !== 'undefined' && module.exports) module.exports = api;
|
|
317
|
+
else global.CJKIndex = api;
|
|
318
|
+
})(typeof window !== 'undefined' ? window : globalThis);
|
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "json"
|
|
4
|
+
|
|
5
|
+
module CJKIndex
|
|
6
|
+
# Writes the browser script. The normalization table and character classes
|
|
7
|
+
# come from Normalizer and Tokenizer, so Ruby (indexing) and JavaScript
|
|
8
|
+
# (querying) cannot drift apart.
|
|
9
|
+
module Runtime
|
|
10
|
+
TEMPLATE_PATH = File.expand_path("runtime.js", __dir__)
|
|
11
|
+
|
|
12
|
+
module_function
|
|
13
|
+
|
|
14
|
+
def source(normalizer: Normalizer.default)
|
|
15
|
+
File.read(TEMPLATE_PATH, encoding: "UTF-8")
|
|
16
|
+
# Block form: a replacement string would treat its backslashes as
|
|
17
|
+
# back-references and drop half of the regexp escapes.
|
|
18
|
+
.sub("/*@VARIANTS@*/{}") { JSON.generate(normalizer.variants) }
|
|
19
|
+
.sub("/*@TABLES@*/[]") { JSON.generate(normalizer.tables) }
|
|
20
|
+
.sub("/*@CJK@*/''") { JSON.generate(Tokenizer.char_class(Tokenizer::CJK_RANGES)) }
|
|
21
|
+
.sub("/*@SEPARATOR@*/''") do
|
|
22
|
+
JSON.generate(Tokenizer.char_class(Tokenizer::SEPARATOR_RANGES, Tokenizer::SEPARATOR_CHARS))
|
|
23
|
+
end
|
|
24
|
+
end
|
|
25
|
+
end
|
|
26
|
+
end
|