wp2txt 2.3.3 → 2.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/.dockerignore +0 -4
- data/.gitignore +3 -5
- data/CHANGELOG.md +18 -0
- data/DEVELOPMENT.md +1 -1
- data/DEVELOPMENT_ja.md +1 -1
- data/README.md +44 -2
- data/README_ja.md +35 -2
- data/Rakefile +10 -21
- data/bin/wp2txt +79 -17
- data/bin/wp2txt-mcp +1 -1
- data/docs/INDEXES.md +61 -1
- data/lib/wp2txt/article.rb +1 -1
- data/lib/wp2txt/cli.rb +42 -0
- data/lib/wp2txt/constants.rb +24 -0
- data/lib/wp2txt/corpus.rb +11 -2
- data/lib/wp2txt/data/template_aliases.json +1 -1
- data/lib/wp2txt/extractor.rb +10 -1
- data/lib/wp2txt/formatter.rb +20 -0
- data/lib/wp2txt/index_commands.rb +89 -1
- data/lib/wp2txt/langlinks_importer.rb +19 -35
- data/lib/wp2txt/lead_terms.rb +228 -0
- data/lib/wp2txt/link_counter.rb +171 -0
- data/lib/wp2txt/metadata_index.rb +79 -3
- data/lib/wp2txt/multistream.rb +52 -2
- data/lib/wp2txt/output_writer.rb +8 -0
- data/lib/wp2txt/page_props_importer.rb +170 -0
- data/lib/wp2txt/sql_dump_reader.rb +57 -0
- data/lib/wp2txt/stream_processor.rb +42 -15
- data/lib/wp2txt/template_expander.rb +19 -0
- data/lib/wp2txt/utils.rb +5 -3
- data/lib/wp2txt/version.rb +1 -1
- data/lib/wp2txt/wikitext_regions.rb +66 -0
- data/lib/wp2txt.rb +7 -5
- data/spec/docs_sync_spec.rb +3 -3
- data/spec/langlinks_importer_spec.rb +27 -0
- data/spec/lead_terms_edge_cases_spec.rb +194 -0
- data/spec/lead_terms_links_qids_spec.rb +204 -0
- data/spec/output_integrity_spec.rb +146 -0
- data/spec/p1_correctness_spec.rb +32 -14
- data/spec/page_properties_spec.rb +161 -0
- data/spec/region_semantics_spec.rb +71 -0
- data/spec/template_passthrough_spec.rb +43 -0
- data/spec/titles_output_path_spec.rb +12 -0
- metadata +18 -1
data/lib/wp2txt/corpus.rb
CHANGED
|
@@ -115,7 +115,9 @@ module Wp2txt
|
|
|
115
115
|
fulltext_current: fts.built? && fts.valid_for?(@multistream_path),
|
|
116
116
|
stats: stats,
|
|
117
117
|
fulltext: fts.built? ? fts.stats : nil,
|
|
118
|
-
langlinks: @metadata.built? ? @metadata.langlinks_provenance : nil
|
|
118
|
+
langlinks: @metadata.built? ? @metadata.langlinks_provenance : nil,
|
|
119
|
+
page_properties: @metadata.built? ? @metadata.page_props_provenance : nil,
|
|
120
|
+
inlinks: @metadata.built? ? @metadata.links_provenance : nil
|
|
119
121
|
}
|
|
120
122
|
end
|
|
121
123
|
|
|
@@ -147,7 +149,10 @@ module Wp2txt
|
|
|
147
149
|
else
|
|
148
150
|
render_text(page)
|
|
149
151
|
end
|
|
150
|
-
result = { id: page[:id], title: page[:title]
|
|
152
|
+
result = { id: page[:id], title: page[:title] }
|
|
153
|
+
properties = @metadata.built? ? @metadata.properties_for(page[:id]) : nil
|
|
154
|
+
result.merge!(properties) if properties
|
|
155
|
+
result[:format] = format.to_s
|
|
151
156
|
if max_chars && body.length > max_chars
|
|
152
157
|
result.merge(text: body[0, max_chars], truncated: true, total_chars: body.length)
|
|
153
158
|
else
|
|
@@ -401,12 +406,16 @@ module Wp2txt
|
|
|
401
406
|
titles.each_slice(EXTRACT_BATCH_SIZE) do |batch|
|
|
402
407
|
raise Cancelled if cancel_check&.call
|
|
403
408
|
|
|
409
|
+
# Forked workers inherit this file's buffer and would write it again on exit
|
|
410
|
+
f.flush
|
|
404
411
|
pages = reader.extract_articles_parallel(batch, num_processes: num_processes)
|
|
405
412
|
batch.each do |t|
|
|
406
413
|
page = pages[t]
|
|
407
414
|
next unless page
|
|
408
415
|
|
|
409
416
|
records = build_records(page, content, resolved_sections, chunk_size, chunk_overlap)
|
|
417
|
+
properties = @metadata.built? ? @metadata.properties_for(page[:id]) : nil
|
|
418
|
+
records.each { |record| record.merge!(properties) } if properties
|
|
410
419
|
next if records.empty?
|
|
411
420
|
|
|
412
421
|
articles_extracted += 1
|
data/lib/wp2txt/extractor.rb
CHANGED
|
@@ -111,6 +111,7 @@ module Wp2txt
|
|
|
111
111
|
|
|
112
112
|
if page
|
|
113
113
|
article = Article.new(page[:text], page[:title], !config[:marker])
|
|
114
|
+
set_source_fields(article, page[:id], page[:revision_id], page[:text], config)
|
|
114
115
|
result = format_article(article, config)
|
|
115
116
|
writer.write(result)
|
|
116
117
|
extracted_count += 1
|
|
@@ -253,11 +254,17 @@ module Wp2txt
|
|
|
253
254
|
bz2_gem: opts[:bz2_gem]
|
|
254
255
|
}
|
|
255
256
|
|
|
256
|
-
%i[title list heading table pre ref redirect multiline category category_only
|
|
257
|
+
%i[lead_terms title list heading table pre ref redirect multiline category category_only
|
|
257
258
|
summary_only metadata_only marker extract_citations expand_templates].each do |opt|
|
|
258
259
|
config[opt] = opts[opt]
|
|
259
260
|
end
|
|
260
261
|
|
|
262
|
+
# Page properties, when imported into the index of the full cached dump
|
|
263
|
+
if format == :json && opts[:lang]
|
|
264
|
+
full_dump = Wp2txt::DumpManager.new(opts[:lang], cache_dir: opts[:cache_dir]).cached_multistream_path
|
|
265
|
+
config[:page_properties] = load_page_properties(full_dump, opts[:cache_dir])
|
|
266
|
+
end
|
|
267
|
+
|
|
261
268
|
# Section extraction options
|
|
262
269
|
%i[sections section_output min_section_length skip_empty
|
|
263
270
|
alias_file no_section_aliases show_matched_sections].each do |opt|
|
|
@@ -480,6 +487,7 @@ module Wp2txt
|
|
|
480
487
|
|
|
481
488
|
pages.each do |page|
|
|
482
489
|
article = Article.new(page[:text], page[:title], !config[:marker])
|
|
490
|
+
set_source_fields(article, page[:id], page[:revision_id], page[:text], config)
|
|
483
491
|
result = format_article(article, config)
|
|
484
492
|
writer.write(result)
|
|
485
493
|
extracted_count += 1
|
|
@@ -493,6 +501,7 @@ module Wp2txt
|
|
|
493
501
|
|
|
494
502
|
if page
|
|
495
503
|
article = Article.new(page[:text], page[:title], !config[:marker])
|
|
504
|
+
set_source_fields(article, page[:id], page[:revision_id], page[:text], config)
|
|
496
505
|
result = format_article(article, config)
|
|
497
506
|
writer.write(result)
|
|
498
507
|
extracted_count += 1
|
data/lib/wp2txt/formatter.rb
CHANGED
|
@@ -14,6 +14,26 @@ module Wp2txt
|
|
|
14
14
|
|
|
15
15
|
# Format article based on configuration and output format
|
|
16
16
|
def format_article(article, config)
|
|
17
|
+
with_source_fields(format_article_body(article, config), article)
|
|
18
|
+
end
|
|
19
|
+
|
|
20
|
+
# JSON records carry the dump's page and revision IDs (and page properties,
|
|
21
|
+
# when imported) right after the title, so a record can be traced
|
|
22
|
+
# back to its source; lead terms, when requested, come last.
|
|
23
|
+
def with_source_fields(result, article)
|
|
24
|
+
return result unless result.is_a?(Hash)
|
|
25
|
+
|
|
26
|
+
ids = { "page_id" => article.page_id, "revision_id" => article.revision_id }.compact
|
|
27
|
+
ids.merge!(article.page_properties.transform_keys(&:to_s)) if article.page_properties
|
|
28
|
+
out = result.each_with_object({}) do |(key, value), acc|
|
|
29
|
+
acc[key] = value
|
|
30
|
+
acc.merge!(ids) if key == "title"
|
|
31
|
+
end
|
|
32
|
+
out["lead_terms"] = article.lead_terms if article.lead_terms
|
|
33
|
+
out
|
|
34
|
+
end
|
|
35
|
+
|
|
36
|
+
def format_article_body(article, config)
|
|
17
37
|
# Store original title for magic word expansion in content
|
|
18
38
|
original_title = article.title.dup
|
|
19
39
|
article.title = format_wiki(article.title, config)
|
|
@@ -4,6 +4,8 @@ require "json"
|
|
|
4
4
|
require_relative "metadata_index"
|
|
5
5
|
require_relative "fts_index"
|
|
6
6
|
require_relative "langlinks_importer"
|
|
7
|
+
require_relative "page_props_importer"
|
|
8
|
+
require_relative "link_counter"
|
|
7
9
|
require_relative "corpus"
|
|
8
10
|
require_relative "multistream"
|
|
9
11
|
require_relative "memory_monitor"
|
|
@@ -322,11 +324,97 @@ module Wp2txt
|
|
|
322
324
|
end
|
|
323
325
|
end
|
|
324
326
|
CliUI::EXIT_SUCCESS
|
|
325
|
-
rescue ArgumentError => e
|
|
327
|
+
rescue ArgumentError, Wp2txt::Error => e
|
|
326
328
|
print_error(e.message)
|
|
327
329
|
CliUI::EXIT_ERROR
|
|
328
330
|
end
|
|
329
331
|
|
|
332
|
+
# Import each article's page properties from the page_props dump of the
|
|
333
|
+
# same date as the metadata index
|
|
334
|
+
def run_import_page_props(opts)
|
|
335
|
+
db_path, dump_date, manager = built_index_for(opts)
|
|
336
|
+
return CliUI::EXIT_ERROR unless db_path
|
|
337
|
+
|
|
338
|
+
source = opts[:page_props_file] || begin
|
|
339
|
+
print_header("Downloading page_props for '#{opts[:lang]}' (#{dump_date})")
|
|
340
|
+
manager.download_page_props(date: dump_date)
|
|
341
|
+
end
|
|
342
|
+
print_mode_banner("Import Page Properties", { "Source" => File.basename(source), "Metadata DB" => db_path })
|
|
343
|
+
|
|
344
|
+
time_start = Time.now
|
|
345
|
+
result = PagePropsImporter.new(db_path).import!(source, force: opts[:update_cache])
|
|
346
|
+
if result[:status] == :already_imported
|
|
347
|
+
print_success("Page properties already imported (at #{result[:imported_at]}, #{result[:row_count]} pages).")
|
|
348
|
+
print_info_message("Use -U/--update-cache to re-import.")
|
|
349
|
+
else
|
|
350
|
+
print_success("Page properties imported: #{result[:row_count]} pages in #{format_duration(Time.now - time_start)}")
|
|
351
|
+
provenance = result[:provenance]
|
|
352
|
+
print_info("Properties", "#{provenance[:qid_count]} QIDs, #{provenance[:disambiguation_count]} disambiguation flags, " \
|
|
353
|
+
"#{provenance[:sort_key_count]} sort keys; #{provenance[:skipped_invalid_sort_keys]} invalid sort keys skipped")
|
|
354
|
+
print_info("SHA-256", result[:provenance][:source_sha256].to_s)
|
|
355
|
+
end
|
|
356
|
+
CliUI::EXIT_SUCCESS
|
|
357
|
+
rescue ArgumentError, Wp2txt::Error => e
|
|
358
|
+
print_error(e.message)
|
|
359
|
+
CliUI::EXIT_ERROR
|
|
360
|
+
end
|
|
361
|
+
|
|
362
|
+
# Count incoming links to every article and store them in the metadata index
|
|
363
|
+
def run_count_links(opts)
|
|
364
|
+
db_path, _dump_date, manager = built_index_for(opts)
|
|
365
|
+
return CliUI::EXIT_ERROR unless db_path
|
|
366
|
+
|
|
367
|
+
multistream = manager.cached_multistream_path
|
|
368
|
+
stream_offsets, = load_stream_offsets(manager.cached_index_path, opts)
|
|
369
|
+
num_processes = opts[:num_procs] || MemoryMonitor.optimal_processes
|
|
370
|
+
print_mode_banner("Count Incoming Links", {
|
|
371
|
+
"Dump" => File.basename(multistream), "Streams" => stream_offsets.size.to_s,
|
|
372
|
+
"Processes" => num_processes.to_s
|
|
373
|
+
})
|
|
374
|
+
|
|
375
|
+
time_start = Time.now
|
|
376
|
+
last_report = Time.now
|
|
377
|
+
result = LinkCounter.new(multistream, stream_offsets, db_path: db_path,
|
|
378
|
+
num_processes: num_processes).count! do |done, total|
|
|
379
|
+
now = Time.now
|
|
380
|
+
if !quiet? && (now - last_report >= DEFAULT_PROGRESS_INTERVAL || done == total)
|
|
381
|
+
last_report = now
|
|
382
|
+
puts pastel.dim(format(" [%s] %d/%d batches", now.strftime("%H:%M:%S"), done, total))
|
|
383
|
+
end
|
|
384
|
+
end
|
|
385
|
+
print_success("Incoming links counted for #{result[:articles]} articles " \
|
|
386
|
+
"(#{result[:with_inlinks]} linked at least once) in #{format_duration(Time.now - time_start)}")
|
|
387
|
+
CliUI::EXIT_SUCCESS
|
|
388
|
+
rescue ArgumentError, Wp2txt::Error => e
|
|
389
|
+
print_error(e.message)
|
|
390
|
+
CliUI::EXIT_ERROR
|
|
391
|
+
end
|
|
392
|
+
|
|
393
|
+
# The metadata index of the cached dump for --lang, or nil with an error printed
|
|
394
|
+
# @return [Array(String, String, DumpManager)] [db_path, dump_date, manager]
|
|
395
|
+
def built_index_for(opts)
|
|
396
|
+
manager = DumpManager.new(opts[:lang], cache_dir: opts[:cache_dir],
|
|
397
|
+
dump_expiry_days: CLI.config.dump_expiry_days)
|
|
398
|
+
multistream = manager.cached_multistream_path
|
|
399
|
+
unless File.exist?(multistream)
|
|
400
|
+
print_error("No cached dump found for '#{opts[:lang]}'.")
|
|
401
|
+
print_info_message("Download and index it with: wp2txt --build-index -L #{opts[:lang]}")
|
|
402
|
+
return nil
|
|
403
|
+
end
|
|
404
|
+
|
|
405
|
+
db_path = MetadataIndex.path_for(multistream, cache_dir: opts[:cache_dir])
|
|
406
|
+
meta = MetadataIndex.new(db_path)
|
|
407
|
+
unless meta.built?
|
|
408
|
+
meta.close
|
|
409
|
+
print_error("Metadata index not found for this dump.")
|
|
410
|
+
print_info_message("Build it first with: wp2txt --build-index -L #{opts[:lang]}")
|
|
411
|
+
return nil
|
|
412
|
+
end
|
|
413
|
+
dump_date = meta.stats[:dump_name][/\d{8}\z/]
|
|
414
|
+
meta.close
|
|
415
|
+
[db_path, dump_date, manager]
|
|
416
|
+
end
|
|
417
|
+
|
|
330
418
|
# Query the metadata index and print matching article titles
|
|
331
419
|
def run_find_articles(opts)
|
|
332
420
|
multistream_path, = resolve_dump_paths(opts, download: false)
|
|
@@ -4,6 +4,7 @@ require "sqlite3"
|
|
|
4
4
|
require "set"
|
|
5
5
|
require "time"
|
|
6
6
|
require "zlib"
|
|
7
|
+
require_relative "sql_dump_reader"
|
|
7
8
|
require_relative "metadata_index"
|
|
8
9
|
require_relative "version"
|
|
9
10
|
|
|
@@ -28,14 +29,6 @@ module Wp2txt
|
|
|
28
29
|
SANITY_SAMPLE_SIZE = 1000
|
|
29
30
|
SANITY_WARN_THRESHOLD = 0.9
|
|
30
31
|
|
|
31
|
-
INSERT_PREFIX = /\A\s*INSERT\s+INTO\s+`langlinks`\s+VALUES\s+/i
|
|
32
|
-
|
|
33
|
-
# MySQL backslash escapes inside mysqldump string literals
|
|
34
|
-
UNESCAPES = {
|
|
35
|
-
"0" => "\0", "'" => "'", '"' => '"', "b" => "\b", "n" => "\n",
|
|
36
|
-
"r" => "\r", "t" => "\t", "Z" => "\x1A", "\\" => "\\"
|
|
37
|
-
}.freeze
|
|
38
|
-
|
|
39
32
|
def initialize(db_path, cache_dir: nil)
|
|
40
33
|
@db_path = db_path
|
|
41
34
|
@cache_dir = cache_dir
|
|
@@ -99,7 +92,9 @@ module Wp2txt
|
|
|
99
92
|
batch.clear
|
|
100
93
|
end
|
|
101
94
|
|
|
95
|
+
rows_seen = 0
|
|
102
96
|
skipped_invalid = each_source_row(source_path) do |ll_from, ll_lang, ll_title|
|
|
97
|
+
rows_seen += 1
|
|
103
98
|
next if lang_filter && !lang_filter.include?(ll_lang)
|
|
104
99
|
|
|
105
100
|
batch << [ll_from, ll_lang, MetadataIndex.normalize_title(ll_title)]
|
|
@@ -107,6 +102,13 @@ module Wp2txt
|
|
|
107
102
|
end
|
|
108
103
|
flush.call unless batch.empty?
|
|
109
104
|
|
|
105
|
+
# Reporting success after reading nothing would leave every language
|
|
106
|
+
# query silently empty; an unreadable dump must fail loudly instead.
|
|
107
|
+
if rows_seen.zero? && skipped_invalid.zero?
|
|
108
|
+
raise Wp2txt::Error, "no langlinks rows found in #{File.basename(source_path)}; " \
|
|
109
|
+
"the file may be empty or in an unrecognized format"
|
|
110
|
+
end
|
|
111
|
+
|
|
110
112
|
# Indexes are created after the load, not before (insert speed)
|
|
111
113
|
db.execute("CREATE INDEX idx_langlinks_from ON langlinks(ll_from, ll_lang)")
|
|
112
114
|
db.execute("CREATE INDEX idx_langlinks_lang_title ON langlinks(ll_lang, ll_title)")
|
|
@@ -224,32 +226,18 @@ module Wp2txt
|
|
|
224
226
|
# tagged UTF-8 and validated — a garbled title could never join
|
|
225
227
|
# pages.title anyway, so such rows are skipped (and counted), not scrubbed.
|
|
226
228
|
def each_source_row(source_path)
|
|
227
|
-
io = if source_path.end_with?(".gz")
|
|
228
|
-
# GzipReader ignores set_encoding; the encoding must be given
|
|
229
|
-
# at open time (lines must come out as BINARY — see below)
|
|
230
|
-
Zlib::GzipReader.open(source_path, encoding: Encoding::BINARY.to_s)
|
|
231
|
-
else
|
|
232
|
-
File.open(source_path, "rb")
|
|
233
|
-
end
|
|
234
|
-
|
|
235
229
|
skipped = 0
|
|
236
|
-
|
|
237
|
-
|
|
238
|
-
|
|
239
|
-
|
|
240
|
-
|
|
241
|
-
|
|
242
|
-
|
|
243
|
-
unless lang.valid_encoding? && title.valid_encoding?
|
|
244
|
-
skipped += 1
|
|
245
|
-
next
|
|
246
|
-
end
|
|
247
|
-
|
|
248
|
-
yield ll_from.to_i, lang, title
|
|
230
|
+
SqlDumpReader.each_insert_line(source_path, "langlinks") do |line|
|
|
231
|
+
line.scan(TUPLE_REGEX) do |ll_from, ll_lang, ll_title|
|
|
232
|
+
lang = SqlDumpReader.unescape(ll_lang).force_encoding(Encoding::UTF_8)
|
|
233
|
+
title = SqlDumpReader.unescape(ll_title).force_encoding(Encoding::UTF_8)
|
|
234
|
+
unless lang.valid_encoding? && title.valid_encoding?
|
|
235
|
+
skipped += 1
|
|
236
|
+
next
|
|
249
237
|
end
|
|
238
|
+
|
|
239
|
+
yield ll_from.to_i, lang, title
|
|
250
240
|
end
|
|
251
|
-
ensure
|
|
252
|
-
io.close
|
|
253
241
|
end
|
|
254
242
|
skipped
|
|
255
243
|
end
|
|
@@ -262,12 +250,8 @@ module Wp2txt
|
|
|
262
250
|
# (equivalent to the old parser skipping malformed tuples)
|
|
263
251
|
TUPLE_REGEX = /\((\d+),'((?:[^'\\]|\\.)*)','((?:[^'\\]|\\.)*)'\)/
|
|
264
252
|
|
|
265
|
-
UNESCAPE_REGEX = /\\(.)/m
|
|
266
253
|
|
|
267
254
|
# Resolve MySQL backslash escapes in a captured string literal
|
|
268
255
|
# (same mapping as the old hand-rolled parser)
|
|
269
|
-
def unescape_mysql(str)
|
|
270
|
-
str.gsub(UNESCAPE_REGEX) { UNESCAPES[::Regexp.last_match(1)] || ::Regexp.last_match(1) }
|
|
271
|
-
end
|
|
272
256
|
end
|
|
273
257
|
end
|
|
@@ -0,0 +1,228 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require_relative "wikitext_regions"
|
|
4
|
+
|
|
5
|
+
module Wp2txt
|
|
6
|
+
# Finds the terms an article introduces in its lead: bold spans in the first
|
|
7
|
+
# paragraph that has one, the parenthesized notes written right after each,
|
|
8
|
+
# and reading (ruby) templates. Nothing is judged — which note is a reading,
|
|
9
|
+
# a native spelling, or a date is left to the caller.
|
|
10
|
+
#
|
|
11
|
+
# Positions are character offsets [start, end) into the article's wikitext
|
|
12
|
+
# as stored in the dump, XML entities decoded, before any other processing
|
|
13
|
+
# (comments included), so a caller holding the same text can cut out and
|
|
14
|
+
# hash exactly the span a term came from.
|
|
15
|
+
module LeadTerms
|
|
16
|
+
MAX_TERMS = 5
|
|
17
|
+
HEADING = /^={2,6}[^=\n].*?={2,6}[ \t]*$/
|
|
18
|
+
OPENERS = { "(" => ")", "(" => ")" }.freeze
|
|
19
|
+
SEPARATORS = ["、", ",", ",", ";", ";"].freeze
|
|
20
|
+
PARAGRAPH_BREAK = /\n[ \t\r]*\n/
|
|
21
|
+
# Regions whose bold text is not the article's own lead prose
|
|
22
|
+
SKIP_OPEN = { "{{" => "}}", "{|" => "|}", "<!--" => "-->" }.freeze
|
|
23
|
+
REF_OPEN = /\A<ref(?:\s[^>]*)?>/i
|
|
24
|
+
REF_SELF_CLOSING = /\A<ref(?:\s[^>]*)?\/>/i
|
|
25
|
+
FILE_LINK = /\A\[\[\s*(?:file|image|ファイル|画像|media)\s*:/i
|
|
26
|
+
|
|
27
|
+
module_function
|
|
28
|
+
|
|
29
|
+
# @param wikitext [String] decoded article wikitext
|
|
30
|
+
# @param render [#call] turns a wikitext fragment into clean text
|
|
31
|
+
# @return [Array<Hash>] terms in order of appearance, at most MAX_TERMS
|
|
32
|
+
def extract(wikitext, render:)
|
|
33
|
+
return [] if wikitext.nil? || wikitext.empty?
|
|
34
|
+
|
|
35
|
+
visible = WikitextRegions.mask_lead(wikitext)
|
|
36
|
+
bolds, rubies, lead_end = scan(visible, visible.length, source: wikitext)
|
|
37
|
+
original_render = render
|
|
38
|
+
render = ->(fragment) { original_render.call(fragment.gsub(WikitextRegions::COMMENT, "")) }
|
|
39
|
+
terms = []
|
|
40
|
+
|
|
41
|
+
if (first = bolds.first)
|
|
42
|
+
paragraph = paragraph_bounds(visible, first[0], lead_end)
|
|
43
|
+
bolds.select { |s, e| s >= paragraph[0] && e <= paragraph[1] }.first(MAX_TERMS).each do |s, e|
|
|
44
|
+
terms << bold_term(wikitext, s, e, paragraph[1], render)
|
|
45
|
+
end
|
|
46
|
+
end
|
|
47
|
+
rubies.each do |s, e, parts|
|
|
48
|
+
terms << { "text" => render.call(parts[1].to_s).strip, "reading" => render.call(parts[2].to_s).strip,
|
|
49
|
+
"source" => parts[0].strip, "span" => { "template" => [s, e] } }
|
|
50
|
+
end
|
|
51
|
+
|
|
52
|
+
terms.sort_by { |t| t["span"].values.first.first }.first(MAX_TERMS)
|
|
53
|
+
.each_with_index.map { |t, i| { "index" => i }.merge(t) }
|
|
54
|
+
end
|
|
55
|
+
|
|
56
|
+
# Top-level bold spans [start, end) including the quote marks, and ruby
|
|
57
|
+
# templates [start, end, [name, text, reading]] found in the lead
|
|
58
|
+
def scan(text, limit, source: text)
|
|
59
|
+
bolds = []
|
|
60
|
+
rubies = []
|
|
61
|
+
i = 0
|
|
62
|
+
open_bold = nil
|
|
63
|
+
while i < limit
|
|
64
|
+
if (i.zero? || text[i - 1] == "\n") && text[i] == "=" &&
|
|
65
|
+
HEADING.match?(text[i...(text.index("\n", i) || limit)])
|
|
66
|
+
limit = i
|
|
67
|
+
break
|
|
68
|
+
elsif text[i] == "\n"
|
|
69
|
+
open_bold = nil # bold does not continue across lines
|
|
70
|
+
i += 1
|
|
71
|
+
elsif (close = SKIP_OPEN[text[i, 4] == "<!--" ? "<!--" : text[i, 2]])
|
|
72
|
+
opener = text[i, 4] == "<!--" ? "<!--" : text[i, 2]
|
|
73
|
+
stop = matching_end(text, i, opener, close)
|
|
74
|
+
if opener == "{{" && (ruby = ruby_template(source[(i + 2)...(stop - 2)]))
|
|
75
|
+
rubies << [i, stop, ruby]
|
|
76
|
+
end
|
|
77
|
+
i = stop
|
|
78
|
+
elsif text[i, 2] == "[[" && FILE_LINK.match?(text[i, 40])
|
|
79
|
+
i = matching_end(text, i, "[[", "]]")
|
|
80
|
+
elsif text[i] == "<" && (m = REF_SELF_CLOSING.match(text[i, 200]))
|
|
81
|
+
i += m[0].length
|
|
82
|
+
elsif text[i] == "<" && (m = REF_OPEN.match(text[i, 200]))
|
|
83
|
+
close_at = text.index(%r{</ref\s*>}i, i + m[0].length)
|
|
84
|
+
i = close_at ? text.index(">", close_at) + 1 : limit
|
|
85
|
+
elsif text[i, 3] == "'''"
|
|
86
|
+
run = text[i..][/\A'+/].length
|
|
87
|
+
if open_bold
|
|
88
|
+
bolds << [open_bold, i + run]
|
|
89
|
+
open_bold = nil
|
|
90
|
+
else
|
|
91
|
+
open_bold = i
|
|
92
|
+
end
|
|
93
|
+
i += run
|
|
94
|
+
else
|
|
95
|
+
i += 1
|
|
96
|
+
end
|
|
97
|
+
end
|
|
98
|
+
[bolds, rubies, limit]
|
|
99
|
+
end
|
|
100
|
+
|
|
101
|
+
# End index (exclusive) of the construct opened at start, honouring nesting
|
|
102
|
+
def matching_end(text, start, opener, closer)
|
|
103
|
+
depth = 0
|
|
104
|
+
i = start
|
|
105
|
+
while i < text.length
|
|
106
|
+
if text[i, opener.length] == opener
|
|
107
|
+
depth += 1
|
|
108
|
+
i += opener.length
|
|
109
|
+
elsif text[i, closer.length] == closer
|
|
110
|
+
depth -= 1
|
|
111
|
+
i += closer.length
|
|
112
|
+
return i if depth.zero?
|
|
113
|
+
else
|
|
114
|
+
i += 1
|
|
115
|
+
end
|
|
116
|
+
end
|
|
117
|
+
text.length
|
|
118
|
+
end
|
|
119
|
+
|
|
120
|
+
def ruby_template(content)
|
|
121
|
+
parts = split_top_level(content, ["|"])
|
|
122
|
+
name = parts.first.to_s
|
|
123
|
+
return nil unless ruby_name?(name)
|
|
124
|
+
|
|
125
|
+
[name, parts[1], parts[2]]
|
|
126
|
+
end
|
|
127
|
+
|
|
128
|
+
# Same name rule as the cleaner: "_" and " " alike, case-insensitive
|
|
129
|
+
def ruby_name?(name)
|
|
130
|
+
@ruby_names ||= Wp2txt::RUBY_TEXT_TEMPLATES.to_set { |t| t.tr("_", " ").strip.downcase }
|
|
131
|
+
@ruby_names.include?(name.to_s.tr("_", " ").strip.downcase)
|
|
132
|
+
end
|
|
133
|
+
|
|
134
|
+
def paragraph_bounds(text, pos, limit)
|
|
135
|
+
start = 0
|
|
136
|
+
text[0...pos].to_enum(:scan, PARAGRAPH_BREAK).each { start = Regexp.last_match.end(0) }
|
|
137
|
+
stop = text.index(PARAGRAPH_BREAK, pos) || limit
|
|
138
|
+
[start, [stop, limit].min]
|
|
139
|
+
end
|
|
140
|
+
|
|
141
|
+
def bold_term(text, s, e, limit, render)
|
|
142
|
+
inner = text[s...e].sub(/\A'+/, "").sub(/'+\z/, "")
|
|
143
|
+
term = { "text" => render.call(plain_ruby(inner)).strip, "notes" => [], "notes_text" => nil,
|
|
144
|
+
"source" => "bold", "span" => { "bold" => [s, e] } }
|
|
145
|
+
j = e
|
|
146
|
+
j += 1 while j < limit && [" ", "\t", " ", "\r", "\n"].include?(text[j])
|
|
147
|
+
return term if j >= limit
|
|
148
|
+
closer = OPENERS[text[j]]
|
|
149
|
+
return term unless closer
|
|
150
|
+
|
|
151
|
+
pe = paren_end(text, j, text[j], closer, limit)
|
|
152
|
+
return term unless pe
|
|
153
|
+
|
|
154
|
+
raw = text[(j + 1)...(pe - 1)]
|
|
155
|
+
term["notes_text"] = render.call(raw).strip
|
|
156
|
+
term["notes"] = split_top_level(raw, SEPARATORS).map { |part| render.call(part).strip }.reject(&:empty?)
|
|
157
|
+
term["span"]["paren"] = [j, pe]
|
|
158
|
+
term
|
|
159
|
+
end
|
|
160
|
+
|
|
161
|
+
# Index after the bracket closing the one at start, or nil if unclosed
|
|
162
|
+
def paren_end(text, start, opener, closer, limit)
|
|
163
|
+
depth = 0
|
|
164
|
+
i = start
|
|
165
|
+
while i < limit
|
|
166
|
+
two = text[i, 2]
|
|
167
|
+
if text[i] == "<" && (stop = WikitextRegions.end_at(text, i, lead: true))
|
|
168
|
+
i = stop
|
|
169
|
+
next
|
|
170
|
+
elsif ["{{", "[["].include?(two)
|
|
171
|
+
i = matching_end(text, i, two, two == "{{" ? "}}" : "]]")
|
|
172
|
+
next
|
|
173
|
+
end
|
|
174
|
+
if [opener, "(", "("].include?(text[i])
|
|
175
|
+
depth += 1
|
|
176
|
+
elsif [closer, ")", ")"].include?(text[i])
|
|
177
|
+
depth -= 1
|
|
178
|
+
return i + 1 if depth.zero?
|
|
179
|
+
elsif text[i] == "\n" && /\A\n[ \t\r]*\n/.match?(text[i..])
|
|
180
|
+
return nil
|
|
181
|
+
end
|
|
182
|
+
i += 1
|
|
183
|
+
end
|
|
184
|
+
nil
|
|
185
|
+
end
|
|
186
|
+
|
|
187
|
+
# Split at separators that sit outside brackets, templates, and links
|
|
188
|
+
def split_top_level(text, separators)
|
|
189
|
+
parts = [+""]
|
|
190
|
+
depth = 0
|
|
191
|
+
i = 0
|
|
192
|
+
while i < text.length
|
|
193
|
+
two = text[i, 2]
|
|
194
|
+
if text[i] == "<" && (stop = WikitextRegions.end_at(text, i, lead: true))
|
|
195
|
+
parts.last << text[i...stop]
|
|
196
|
+
i = stop
|
|
197
|
+
elsif ["{{", "[["].include?(two)
|
|
198
|
+
depth += 1
|
|
199
|
+
parts.last << two
|
|
200
|
+
i += 2
|
|
201
|
+
elsif ["}}", "]]"].include?(two)
|
|
202
|
+
depth -= 1
|
|
203
|
+
parts.last << two
|
|
204
|
+
i += 2
|
|
205
|
+
else
|
|
206
|
+
ch = text[i]
|
|
207
|
+
depth += 1 if ["(", "("].include?(ch)
|
|
208
|
+
depth -= 1 if [")", ")"].include?(ch)
|
|
209
|
+
if depth <= 0 && separators.include?(ch)
|
|
210
|
+
parts << +""
|
|
211
|
+
else
|
|
212
|
+
parts.last << ch
|
|
213
|
+
end
|
|
214
|
+
i += 1
|
|
215
|
+
end
|
|
216
|
+
end
|
|
217
|
+
parts
|
|
218
|
+
end
|
|
219
|
+
|
|
220
|
+
# In a bold headword, a ruby template stands for its base text; the
|
|
221
|
+
# reading is reported separately as its own term
|
|
222
|
+
def plain_ruby(fragment)
|
|
223
|
+
fragment.gsub(/\{\{([^{}]*)\}\}/) do |whole|
|
|
224
|
+
(ruby = ruby_template(Regexp.last_match(1))) ? ruby[1].to_s : whole
|
|
225
|
+
end
|
|
226
|
+
end
|
|
227
|
+
end
|
|
228
|
+
end
|