cjk_index 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,86 @@
1
+ # frozen_string_literal: true
2
+
3
+ require "json"
4
+
5
+ module CJKIndex
6
+ # Builds a prebuilt inverted index that the browser runtime loads as JSON.
7
+ #
8
+ # builder = CJKIndex::Builder.new(fields: %w[title creator], boosts: { "title" => 3 })
9
+ # builder.add("item1.html", "title" => "東京帝國大學", "creator" => "…")
10
+ # File.write("search-index.json", builder.to_json)
11
+ #
12
+ # Format (version 2):
13
+ # {
14
+ # "format": "cjk_index/2",
15
+ # "variants": ["ja"],
16
+ # "fields": ["title", "creator"],
17
+ # "boosts": [3, 1],
18
+ # "refs": ["item1.html", ...],
19
+ # "lengths": [len(doc0.title), len(doc0.creator), len(doc1.title), ...],
20
+ # "tokens": { "東京": [doc, field, n, pos1, ..., posn, doc, field, n, ...], ... }
21
+ # }
22
+ # doc and field are positions in "refs" and "fields"; n is the number of
23
+ # occurrences and pos are character offsets in the normalized field text.
24
+ # Offsets let the runtime require the bigrams of a query to be adjacent
25
+ # (exact substring match); lengths feed BM25 ranking.
26
+ class Builder
27
+ FORMAT = "cjk_index/2"
28
+
29
+ # Gap between the values of a multi-valued field, so that a phrase cannot
30
+ # match across two values.
31
+ VALUE_GAP = 1
32
+
33
+ attr_reader :fields, :refs, :normalizer
34
+
35
+ # normalizer must match the one the runtime script was generated with
36
+ # (Runtime.source(normalizer:)); the index records its tables and the
37
+ # runtime refuses an index built with different ones.
38
+ def initialize(fields:, boosts: {}, normalizer: Normalizer.default)
39
+ raise ArgumentError, "fields must not be empty" if fields.empty?
40
+
41
+ @normalizer = normalizer
42
+ @fields = fields.map(&:to_s)
43
+ @boosts = @fields.map { |f| (boosts[f] || boosts[f.to_sym] || 1).to_f }
44
+ @refs = []
45
+ @lengths = []
46
+ @postings = Hash.new { |h, k| h[k] = [] }
47
+ end
48
+
49
+ # doc is a Hash of field name => value. Array values are indexed as
50
+ # separate strings (e.g. multi-valued subject fields).
51
+ def add(ref, doc)
52
+ doc_index = @refs.length
53
+ @refs << ref.to_s
54
+ @fields.each_with_index do |field, field_index|
55
+ value = doc[field] || doc[field.to_sym]
56
+ positions = Hash.new { |h, k| h[k] = [] }
57
+ base = 0
58
+ Array(value).each do |v|
59
+ text = v.to_s
60
+ Tokenizer.tokens_with_offsets(text, unigrams: true, normalizer: @normalizer)
61
+ .each { |t, offset| positions[t] << (base + offset) }
62
+ base += @normalizer.normalize(text).length + VALUE_GAP
63
+ end
64
+ @lengths << [base - VALUE_GAP, 0].max
65
+ positions.each { |token, offsets| @postings[token].push(doc_index, field_index, offsets.length, *offsets) }
66
+ end
67
+ self
68
+ end
69
+
70
+ def to_h
71
+ {
72
+ "format" => FORMAT,
73
+ "variants" => @normalizer.tables,
74
+ "fields" => @fields,
75
+ "boosts" => @boosts.map { |b| b == b.to_i ? b.to_i : b },
76
+ "refs" => @refs,
77
+ "lengths" => @lengths,
78
+ "tokens" => @postings.sort.to_h
79
+ }
80
+ end
81
+
82
+ def to_json(*_args)
83
+ JSON.generate(to_h)
84
+ end
85
+ end
86
+ end
@@ -0,0 +1,127 @@
1
+ # frozen_string_literal: true
2
+
3
+ require "jekyll"
4
+ require_relative "../cjk_index"
5
+
6
+ module CJKIndex
7
+ # Jekyll plugin. Add `gem "cjk_index"` to the :jekyll_plugins group (or list
8
+ # "cjk_index/jekyll" under `plugins:`), then configure in _config.yml:
9
+ #
10
+ # cjk_index:
11
+ # output: assets/cjk-index # where files are written (default)
12
+ # variants: [ja] # variant tables: ja (default), zh (Traditional -> Simplified)
13
+ # indexes:
14
+ # - name: posts # -> assets/cjk-index/posts.json
15
+ # collection: posts # a Jekyll collection, or...
16
+ # fields: [title, tags, content]
17
+ # boosts: { title: 3 }
18
+ # - name: items
19
+ # data: metadata # ...rows of _data/metadata.csv
20
+ # ref: objectid # column used as the result's ref
21
+ # fields: [title, creator, subject]
22
+ # - name: search
23
+ # preset: collectionbuilder # reads CollectionBuilder's own config
24
+ #
25
+ # The browser script is written to <output>/cjk-index.js.
26
+ module Jekyll
27
+ DEFAULT_OUTPUT = "assets/cjk-index"
28
+ RUNTIME_NAME = "cjk-index.js"
29
+
30
+ # A file whose content is produced at build time (no source file, no Liquid).
31
+ class GeneratedFile < ::Jekyll::StaticFile
32
+ def initialize(site, dir, name, content)
33
+ super(site, site.source, dir, name)
34
+ @generated_content = content
35
+ end
36
+
37
+ def write(dest)
38
+ path = destination(dest)
39
+ return false if File.exist?(path) && File.read(path, encoding: "UTF-8") == @generated_content
40
+
41
+ FileUtils.mkdir_p(File.dirname(path))
42
+ File.write(path, @generated_content)
43
+ true
44
+ end
45
+ end
46
+
47
+ class Generator < ::Jekyll::Generator
48
+ safe true
49
+ priority :lowest
50
+
51
+ def generate(site)
52
+ config = site.config["cjk_index"]
53
+ return unless config.is_a?(Hash)
54
+
55
+ output = config.fetch("output", DEFAULT_OUTPUT).sub(%r{\A/+}, "").sub(%r{/+\z}, "")
56
+ normalizer = Normalizer.new(tables: Array(config.fetch("variants", Normalizer::DEFAULT_TABLES)))
57
+ site.static_files << GeneratedFile.new(site, output, RUNTIME_NAME, Runtime.source(normalizer: normalizer))
58
+ Array(config["indexes"]).each do |spec|
59
+ builder = build(site, spec, normalizer)
60
+ name = "#{spec.fetch('name')}.json"
61
+ site.static_files << GeneratedFile.new(site, output, name, builder.to_json)
62
+ ::Jekyll.logger.info "cjk_index:", "#{output}/#{name} (#{builder.refs.length} documents)"
63
+ end
64
+ end
65
+
66
+ private
67
+
68
+ def build(site, spec, normalizer)
69
+ return Presets::CollectionBuilder.build(site, spec, normalizer) if spec["preset"] == "collectionbuilder"
70
+ raise ::Jekyll::Errors::FatalException, "cjk_index: unknown preset #{spec['preset']}" if spec["preset"]
71
+
72
+ builder = Builder.new(fields: Array(spec.fetch("fields")), boosts: spec.fetch("boosts", {}),
73
+ normalizer: normalizer)
74
+ if spec["collection"]
75
+ docs = site.collections.fetch(spec["collection"]) do
76
+ raise ::Jekyll::Errors::FatalException, "cjk_index: no collection #{spec['collection']}"
77
+ end.docs
78
+ docs.each { |doc| builder.add(doc.url, document_fields(doc, builder.fields)) }
79
+ elsif spec["data"]
80
+ ref = spec.fetch("ref", "id")
81
+ Array(site.data[spec["data"]]).each do |row|
82
+ builder.add(row[ref], row) unless row[ref].to_s.empty?
83
+ end
84
+ else
85
+ raise ::Jekyll::Errors::FatalException, "cjk_index: index #{spec['name']} needs collection or data"
86
+ end
87
+ builder
88
+ end
89
+
90
+ def document_fields(doc, fields)
91
+ fields.to_h { |f| [f, f == "content" ? doc.content : doc.data[f]] }
92
+ end
93
+ end
94
+
95
+ module Presets
96
+ # Mirrors assets/js/lunr-store.js of CollectionBuilder (collectionbuilder-csv):
97
+ # items with an objectid, child objects only when the theme asks for them,
98
+ # fields marked index=true in _data/config-search.csv, and the same "id"
99
+ # values, so results can be looked up in the existing search store.
100
+ module CollectionBuilder
101
+ module_function
102
+
103
+ def build(site, spec, normalizer)
104
+ rows = Array(site.data[site.config["metadata"]])
105
+ children = site.data.dig("theme", "search-child-objects") == true
106
+ fields = Array(site.data["config-search"])
107
+ .select { |f| f["index"].to_s == "true" }
108
+ .map { |f| f["field"] }
109
+ boosts = spec.fetch("boosts") { fields.first ? { fields.first => 3 } : {} }
110
+ builder = Builder.new(fields: fields, boosts: boosts, normalizer: normalizer)
111
+ rows.each do |row|
112
+ next if row["objectid"].to_s.empty?
113
+ next if !row["parentid"].to_s.empty? && !children
114
+
115
+ builder.add(ref_for(row), row.to_h { |k, v| [k, v.is_a?(String) ? v.split.join(" ") : v] })
116
+ end
117
+ builder
118
+ end
119
+
120
+ def ref_for(row)
121
+ parent = row["parentid"].to_s
122
+ parent.empty? ? "#{row['objectid']}.html" : "#{parent}.html##{row['objectid']}"
123
+ end
124
+ end
125
+ end
126
+ end
127
+ end
@@ -0,0 +1,89 @@
1
+ # frozen_string_literal: true
2
+
3
+ module CJKIndex
4
+ # Folds text so that spelling variants a reader treats as "the same" compare
5
+ # equal: NFKC (full-width -> half-width etc.), lowercase, katakana ->
6
+ # hiragana, character variants from the selected tables, and the iteration
7
+ # mark 々 expanded to the preceding character.
8
+ #
9
+ # The same rules run in the browser (see Runtime), so an index built here
10
+ # and a query typed there are folded identically.
11
+ #
12
+ # Variant tables (data/*.tsv):
13
+ # ja Japanese old forms -> modern forms (圖 -> 図). Default.
14
+ # zh Traditional -> Simplified Chinese, from OpenCC. Opt in: it also
15
+ # merges characters that are distinct in Japanese (後 and 后).
16
+ # With several tables, characters linked by any table fold to one form,
17
+ # preferring the target of the table listed first: with [ja, zh],
18
+ # 圖, 図 and 图 all become 図.
19
+ class Normalizer
20
+ DATA_DIR = File.expand_path("../../data", __dir__)
21
+ TABLES = {
22
+ "ja" => "ja-kyujitai.tsv",
23
+ "zh" => "zh-hant-hans.tsv"
24
+ }.freeze
25
+ DEFAULT_TABLES = %w[ja].freeze
26
+
27
+ # Katakana ァ (U+30A1) .. ヶ (U+30F6) map to hiragana by subtracting 0x60.
28
+ KATAKANA_FIRST = 0x30A1
29
+ KATAKANA_LAST = 0x30F6
30
+ KANA_OFFSET = 0x60
31
+ ITERATION_MARK = "々"
32
+
33
+ def self.default
34
+ @default ||= new
35
+ end
36
+
37
+ def self.read_table(name)
38
+ file = TABLES.fetch(name) { raise ArgumentError, "unknown variant table #{name.inspect} (#{TABLES.keys.join(', ')})" }
39
+ File.readlines(File.join(DATA_DIR, file), chomp: true, encoding: "UTF-8")
40
+ .reject { |line| line.empty? || line.start_with?("#") }
41
+ .map { |line| line.split("\t", 2) }
42
+ end
43
+
44
+ attr_reader :tables, :variants
45
+
46
+ def initialize(tables: DEFAULT_TABLES)
47
+ @tables = tables.map(&:to_s).uniq.freeze
48
+ @variants = build_variants.freeze
49
+ end
50
+
51
+ def normalize(text)
52
+ return "" if text.nil?
53
+
54
+ prev = nil
55
+ text.to_s.unicode_normalize(:nfkc).downcase.each_char.map do |ch|
56
+ code = ch.ord
57
+ ch = (code - KANA_OFFSET).chr(Encoding::UTF_8) if code.between?(KATAKANA_FIRST, KATAKANA_LAST)
58
+ ch = @variants.fetch(ch, ch)
59
+ ch = prev if ch == ITERATION_MARK && prev
60
+ prev = ch
61
+ end.join
62
+ end
63
+
64
+ private
65
+
66
+ # Union-find over all pairs; each class folds to the target preferred by
67
+ # the earliest table (lowest code point among equals, for determinism).
68
+ def build_variants
69
+ parent = {}
70
+ find = lambda do |x|
71
+ parent[x] ||= x
72
+ parent[x] = find.call(parent[x]) unless parent[x] == x
73
+ parent[x]
74
+ end
75
+ rank = {}
76
+ @tables.each_with_index do |name, priority|
77
+ self.class.read_table(name).each do |from, to|
78
+ parent[find.call(from)] = find.call(to)
79
+ rank[to] = [rank.fetch(to, [priority, to.ord]), [priority, to.ord]].min
80
+ end
81
+ end
82
+ classes = parent.keys.group_by { |x| find.call(x) }
83
+ classes.each_value.with_object({}) do |members, map|
84
+ canonical = members.min_by { |m| rank.fetch(m, [Float::INFINITY, m.ord]) }
85
+ members.each { |m| map[m] = canonical unless m == canonical }
86
+ end
87
+ end
88
+ end
89
+ end
@@ -0,0 +1,318 @@
1
+ /*!
2
+ * cjk_index runtime — queries an index built by the cjk_index Ruby gem.
3
+ * Generated file: the variant table and character classes below are written
4
+ * by CJKIndex::Runtime so that they match the Ruby side exactly.
5
+ * License: MIT
6
+ */
7
+ (function (global) {
8
+ 'use strict';
9
+
10
+ var VARIANTS = /*@VARIANTS@*/{};
11
+ var TABLES = /*@TABLES@*/[];
12
+ var CJK = new RegExp(/*@CJK@*/'', 'u');
13
+ var SEPARATOR = new RegExp(/*@SEPARATOR@*/'', 'u');
14
+ var FORMAT = 'cjk_index/2';
15
+
16
+ // BM25 parameters (the usual defaults, as in Lucene and Pagefind).
17
+ var K1 = 1.2;
18
+ var B = 0.75;
19
+ // Score multiplier for matches found by prefix or typo expansion.
20
+ var EXPANDED = 0.6;
21
+
22
+ // Folds one character; prev is the previous folded character (for 々).
23
+ function foldChar(ch, prev) {
24
+ var c = ch.codePointAt(0);
25
+ if (c >= 0x30a1 && c <= 0x30f6) ch = String.fromCodePoint(c - 0x60);
26
+ if (VARIANTS[ch]) ch = VARIANTS[ch];
27
+ if (ch === '々' && prev) ch = prev;
28
+ return ch;
29
+ }
30
+
31
+ function normalize(text) {
32
+ if (text === null || text === undefined) return '';
33
+ var out = '';
34
+ var prev = null;
35
+ for (var ch of String(text).normalize('NFKC').toLowerCase()) {
36
+ prev = foldChar(ch, prev);
37
+ out += prev;
38
+ }
39
+ return out;
40
+ }
41
+
42
+ // Same loop as CJKIndex::Tokenizer.phrases in Ruby: split at separators
43
+ // into phrases of [token, offset]; CJK runs become bigrams (plus unigrams
44
+ // when asked), other runs stay words. Offsets count code points.
45
+ function phrases(text, unigrams) {
46
+ var result = [];
47
+ var phrase = [];
48
+ var run = [];
49
+ var runCjk = null;
50
+ var runStart = 0;
51
+ var flush = function () {
52
+ if (!run.length) return;
53
+ if (runCjk) {
54
+ if (run.length === 1 || unigrams) {
55
+ for (var i = 0; i < run.length; i++) phrase.push([run[i], runStart + i]);
56
+ }
57
+ for (var j = 0; j < run.length - 1; j++) phrase.push([run[j] + run[j + 1], runStart + j]);
58
+ } else {
59
+ phrase.push([run.join(''), runStart]);
60
+ }
61
+ run = [];
62
+ };
63
+ var offset = 0;
64
+ for (var ch of normalize(text)) {
65
+ if (SEPARATOR.test(ch)) {
66
+ flush();
67
+ if (phrase.length) result.push(phrase);
68
+ phrase = [];
69
+ runCjk = null;
70
+ } else {
71
+ var cjk = CJK.test(ch);
72
+ if (runCjk !== null && cjk !== runCjk) flush();
73
+ if (!run.length) runStart = offset;
74
+ runCjk = cjk;
75
+ run.push(ch);
76
+ }
77
+ offset++;
78
+ }
79
+ flush();
80
+ if (phrase.length) result.push(phrase);
81
+ return result;
82
+ }
83
+
84
+ function tokenize(text, unigrams) {
85
+ var out = [];
86
+ phrases(text, unigrams).forEach(function (p) { p.forEach(function (t) { out.push(t[0]); }); });
87
+ return out;
88
+ }
89
+
90
+ // Substring test on normalized text, ignoring whitespace. Useful for
91
+ // filters that do not need the index (e.g. a browse page).
92
+ function includes(haystack, needle) {
93
+ var strip = function (s) { return normalize(s).replace(/\s+/g, ''); };
94
+ return strip(haystack).indexOf(strip(needle)) !== -1;
95
+ }
96
+
97
+ function isLatin(token) { return !CJK.test(token); }
98
+
99
+ // true when a and b differ by exactly one insertion, deletion or substitution.
100
+ function oneEdit(a, b) {
101
+ if (a === b || Math.abs(a.length - b.length) > 1) return false;
102
+ var i = 0;
103
+ while (i < a.length && i < b.length && a[i] === b[i]) i++;
104
+ if (a.length === b.length) return a.slice(i + 1) === b.slice(i + 1);
105
+ if (a.length > b.length) return a.slice(i + 1) === b.slice(i);
106
+ return a.slice(i) === b.slice(i + 1);
107
+ }
108
+
109
+ function Index(data, options) {
110
+ if (!data || data.format !== FORMAT) {
111
+ throw new Error('cjk_index: unsupported index format ' + (data && data.format));
112
+ }
113
+ if (JSON.stringify(data.variants) !== JSON.stringify(TABLES)) {
114
+ throw new Error('cjk_index: index built with variant tables ' + JSON.stringify(data.variants) +
115
+ ' but this script folds with ' + JSON.stringify(TABLES) + '; rebuild both together');
116
+ }
117
+ options = options || {};
118
+ this.prefix = options.prefix !== false; // last Latin word matches as a prefix
119
+ this.typos = options.typos !== false; // Latin words of 5+ letters allow one edit
120
+ this.fields = data.fields;
121
+ this.boosts = data.boosts;
122
+ this.refs = data.refs;
123
+ this.lengths = data.lengths;
124
+ this.tokens = data.tokens;
125
+ var nf = this.fields.length;
126
+ var avg = new Array(nf).fill(0);
127
+ for (var i = 0; i < this.lengths.length; i++) avg[i % nf] += this.lengths[i];
128
+ this.avgLength = avg.map(function (s) { return s / Math.max(1, data.refs.length) || 1; });
129
+ this.latinTokens = Object.keys(this.tokens).filter(isLatin).sort();
130
+ }
131
+
132
+ // Map of "doc,field" key -> Set of positions, merged over the alternatives.
133
+ Index.prototype._positions = function (tokens) {
134
+ var nf = this.fields.length;
135
+ var map = new Map();
136
+ for (var t = 0; t < tokens.length; t++) {
137
+ var p = this.tokens[tokens[t]];
138
+ if (!p) continue;
139
+ for (var i = 0; i < p.length;) {
140
+ var key = p[i] * nf + p[i + 1];
141
+ var n = p[i + 2];
142
+ var set = map.get(key);
143
+ if (!set) { set = new Set(); map.set(key, set); }
144
+ for (var k = 0; k < n; k++) set.add(p[i + 3 + k]);
145
+ i += 3 + n;
146
+ }
147
+ }
148
+ return map;
149
+ };
150
+
151
+ // Alternatives for one query token: itself, and for Latin words the index
152
+ // words it is a prefix of (last word only) or one edit away from.
153
+ Index.prototype._alternatives = function (token, isLast) {
154
+ if (!isLatin(token)) return { tokens: [token], expanded: false };
155
+ var out = this.tokens[token] ? [token] : [];
156
+ var keys = this.latinTokens;
157
+ if (this.prefix && isLast) {
158
+ var lo = 0, hi = keys.length;
159
+ while (lo < hi) { var mid = (lo + hi) >> 1; if (keys[mid] < token) lo = mid + 1; else hi = mid; }
160
+ for (var i = lo; i < keys.length && keys[i].indexOf(token) === 0; i++) if (keys[i] !== token) out.push(keys[i]);
161
+ }
162
+ if (this.typos && !this.tokens[token] && Array.from(token).length >= 5) {
163
+ keys.forEach(function (k) { if (oneEdit(token, k)) out.push(k); });
164
+ }
165
+ return { tokens: out, expanded: out.length > 0 && out[0] !== token };
166
+ };
167
+
168
+ // Every phrase of the query must occur in the document (in any field),
169
+ // with its tokens adjacent and in order. Returns [{ref, score, fields}]
170
+ // with the best match first; fields lists the names of the fields that matched.
171
+ Index.prototype.search = function (query) {
172
+ var self = this;
173
+ var nf = this.fields.length;
174
+ var n = this.refs.length;
175
+ var qp = phrases(query, false);
176
+ if (!qp.length) return [];
177
+ var scores = new Map();
178
+ var matchedFields = new Map();
179
+ var docsSoFar = null;
180
+
181
+ for (var q = 0; q < qp.length; q++) {
182
+ var phrase = qp[q];
183
+ var expanded = false;
184
+ var parts = phrase.map(function (tok, i) {
185
+ var alt = self._alternatives(tok[0], q === qp.length - 1 && i === phrase.length - 1);
186
+ if (alt.expanded) expanded = true;
187
+ return { offset: tok[1], positions: self._positions(alt.tokens) };
188
+ });
189
+ if (parts.some(function (p) { return p.positions.size === 0; })) return [];
190
+ parts.sort(function (a, b) { return a.positions.size - b.positions.size; });
191
+ var rare = parts[0];
192
+ var hits = new Map(); // "doc,field" key -> phrase occurrences
193
+ rare.positions.forEach(function (set, key) {
194
+ var tf = 0;
195
+ set.forEach(function (pos) {
196
+ var start = pos - rare.offset;
197
+ for (var i = 1; i < parts.length; i++) {
198
+ var other = parts[i].positions.get(key);
199
+ if (!other || !other.has(start + parts[i].offset)) return;
200
+ }
201
+ tf++;
202
+ });
203
+ if (tf) hits.set(key, tf);
204
+ });
205
+ var docs = new Set();
206
+ hits.forEach(function (tf, key) { docs.add(Math.floor(key / nf)); });
207
+ if (!docs.size) return [];
208
+ if (docsSoFar) {
209
+ docs.forEach(function (d) { if (!docsSoFar.has(d)) docs.delete(d); });
210
+ if (!docs.size) return [];
211
+ }
212
+ docsSoFar = docs;
213
+ var idf = Math.log(1 + (n - docs.size + 0.5) / (docs.size + 0.5));
214
+ var weight = expanded ? EXPANDED : 1;
215
+ hits.forEach(function (tf, key) {
216
+ var doc = Math.floor(key / nf), field = key % nf;
217
+ var norm = 1 - B + B * self.lengths[key] / self.avgLength[field];
218
+ var s = weight * self.boosts[field] * idf * (tf * (K1 + 1)) / (tf + K1 * norm);
219
+ scores.set(doc, (scores.get(doc) || 0) + s);
220
+ var f = matchedFields.get(doc);
221
+ if (!f) { f = new Set(); matchedFields.set(doc, f); }
222
+ f.add(self.fields[field]);
223
+ });
224
+ }
225
+
226
+ var results = [];
227
+ docsSoFar.forEach(function (doc) {
228
+ results.push({ ref: self.refs[doc], score: scores.get(doc), fields: Array.from(matchedFields.get(doc)), doc: doc });
229
+ });
230
+ results.sort(function (a, b) { return b.score - a.score || a.doc - b.doc; });
231
+ return results.map(function (r) { return { ref: r.ref, score: r.score, fields: r.fields }; });
232
+ };
233
+
234
+ function escapeHtml(s) {
235
+ return s.replace(/[&<>"']/g, function (c) {
236
+ return { '&': '&amp;', '<': '&lt;', '>': '&gt;', '"': '&quot;', "'": '&#39;' }[c];
237
+ });
238
+ }
239
+
240
+ // [start, end) ranges, in code points of the original text, where a phrase
241
+ // of the query occurs after normalization.
242
+ function matchRanges(text, query) {
243
+ var chars = Array.from(String(text === null || text === undefined ? '' : text));
244
+ var norm = [];
245
+ var origin = [];
246
+ var prev = null;
247
+ chars.forEach(function (ch, i) {
248
+ for (var c of ch.normalize('NFKC').toLowerCase()) {
249
+ prev = foldChar(c, prev);
250
+ norm.push(prev);
251
+ origin.push(i);
252
+ }
253
+ });
254
+ var normText = norm.join('');
255
+ var ranges = [];
256
+ var needles = [];
257
+ var current = '';
258
+ for (var qc of normalize(query)) {
259
+ if (SEPARATOR.test(qc)) { if (current) needles.push(current); current = ''; } else current += qc;
260
+ }
261
+ if (current) needles.push(current);
262
+ needles.forEach(function (needle) {
263
+ var len = Array.from(needle).length;
264
+ var from = 0;
265
+ var at;
266
+ while ((at = normText.indexOf(needle, from)) !== -1) {
267
+ var cp = Array.from(normText.slice(0, at)).length;
268
+ ranges.push([origin[cp], origin[cp + len - 1] + 1]);
269
+ from = at + needle.length;
270
+ }
271
+ });
272
+ ranges.sort(function (a, b) { return a[0] - b[0]; });
273
+ var merged = [];
274
+ ranges.forEach(function (r) {
275
+ var last = merged[merged.length - 1];
276
+ if (last && r[0] <= last[1]) last[1] = Math.max(last[1], r[1]); else merged.push(r.slice());
277
+ });
278
+ return { chars: chars, ranges: merged };
279
+ }
280
+
281
+ // HTML-escaped text with every match of the query wrapped in <mark>.
282
+ // 「東京帝國大學」 is marked for the query 「帝国大学」.
283
+ function highlight(text, query) {
284
+ var m = matchRanges(text, query);
285
+ var out = '';
286
+ var at = 0;
287
+ m.ranges.forEach(function (r) {
288
+ out += escapeHtml(m.chars.slice(at, r[0]).join('')) + '<mark>' + escapeHtml(m.chars.slice(r[0], r[1]).join('')) + '</mark>';
289
+ at = r[1];
290
+ });
291
+ return out + escapeHtml(m.chars.slice(at).join(''));
292
+ }
293
+
294
+ // A highlighted window of about `length` characters around the first match.
295
+ function excerpt(text, query, length) {
296
+ length = length || 80;
297
+ var m = matchRanges(text, query);
298
+ if (!m.ranges.length) return m.chars.length > length ? escapeHtml(m.chars.slice(0, length).join('')) + '…' : escapeHtml(m.chars.join(''));
299
+ var start = Math.max(0, m.ranges[0][0] - Math.floor(length / 3));
300
+ var end = Math.min(m.chars.length, start + length);
301
+ var part = m.chars.slice(start, end).join('');
302
+ return (start > 0 ? '…' : '') + highlight(part, query) + (end < m.chars.length ? '…' : '');
303
+ }
304
+
305
+ function load(url, options) {
306
+ return fetch(url).then(function (res) {
307
+ if (!res.ok) throw new Error('cjk_index: ' + url + ' returned ' + res.status);
308
+ return res.json();
309
+ }).then(function (data) { return new Index(data, options); });
310
+ }
311
+
312
+ var api = {
313
+ tables: TABLES, normalize: normalize, tokenize: tokenize, phrases: phrases, includes: includes,
314
+ highlight: highlight, excerpt: excerpt, Index: Index, load: load
315
+ };
316
+ if (typeof module !== 'undefined' && module.exports) module.exports = api;
317
+ else global.CJKIndex = api;
318
+ })(typeof window !== 'undefined' ? window : globalThis);
@@ -0,0 +1,26 @@
1
+ # frozen_string_literal: true
2
+
3
+ require "json"
4
+
5
+ module CJKIndex
6
+ # Writes the browser script. The normalization table and character classes
7
+ # come from Normalizer and Tokenizer, so Ruby (indexing) and JavaScript
8
+ # (querying) cannot drift apart.
9
+ module Runtime
10
+ TEMPLATE_PATH = File.expand_path("runtime.js", __dir__)
11
+
12
+ module_function
13
+
14
+ def source(normalizer: Normalizer.default)
15
+ File.read(TEMPLATE_PATH, encoding: "UTF-8")
16
+ # Block form: a replacement string would treat its backslashes as
17
+ # back-references and drop half of the regexp escapes.
18
+ .sub("/*@VARIANTS@*/{}") { JSON.generate(normalizer.variants) }
19
+ .sub("/*@TABLES@*/[]") { JSON.generate(normalizer.tables) }
20
+ .sub("/*@CJK@*/''") { JSON.generate(Tokenizer.char_class(Tokenizer::CJK_RANGES)) }
21
+ .sub("/*@SEPARATOR@*/''") do
22
+ JSON.generate(Tokenizer.char_class(Tokenizer::SEPARATOR_RANGES, Tokenizer::SEPARATOR_CHARS))
23
+ end
24
+ end
25
+ end
26
+ end