wp2txt 2.3.4 → 2.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/.dockerignore +0 -4
- data/.gitignore +3 -5
- data/CHANGELOG.md +9 -0
- data/DEVELOPMENT.md +1 -1
- data/DEVELOPMENT_ja.md +1 -1
- data/README.md +39 -0
- data/README_ja.md +30 -0
- data/Rakefile +10 -21
- data/bin/wp2txt +61 -9
- data/bin/wp2txt-mcp +1 -1
- data/docs/INDEXES.md +61 -1
- data/lib/wp2txt/article.rb +1 -1
- data/lib/wp2txt/cli.rb +42 -0
- data/lib/wp2txt/constants.rb +13 -0
- data/lib/wp2txt/corpus.rb +9 -2
- data/lib/wp2txt/data/template_aliases.json +1 -1
- data/lib/wp2txt/extractor.rb +10 -7
- data/lib/wp2txt/formatter.rb +14 -12
- data/lib/wp2txt/index_commands.rb +89 -1
- data/lib/wp2txt/langlinks_importer.rb +19 -35
- data/lib/wp2txt/lead_terms.rb +228 -0
- data/lib/wp2txt/link_counter.rb +171 -0
- data/lib/wp2txt/metadata_index.rb +79 -3
- data/lib/wp2txt/multistream.rb +21 -0
- data/lib/wp2txt/page_props_importer.rb +170 -0
- data/lib/wp2txt/sql_dump_reader.rb +57 -0
- data/lib/wp2txt/stream_processor.rb +9 -2
- data/lib/wp2txt/template_expander.rb +19 -0
- data/lib/wp2txt/utils.rb +5 -3
- data/lib/wp2txt/version.rb +1 -1
- data/lib/wp2txt/wikitext_regions.rb +66 -0
- data/spec/docs_sync_spec.rb +3 -3
- data/spec/langlinks_importer_spec.rb +27 -0
- data/spec/lead_terms_edge_cases_spec.rb +194 -0
- data/spec/lead_terms_links_qids_spec.rb +204 -0
- data/spec/p1_correctness_spec.rb +12 -9
- data/spec/page_properties_spec.rb +161 -0
- data/spec/region_semantics_spec.rb +71 -0
- data/spec/template_passthrough_spec.rb +43 -0
- metadata +16 -1
data/lib/wp2txt/formatter.rb
CHANGED
|
@@ -14,21 +14,23 @@ module Wp2txt
|
|
|
14
14
|
|
|
15
15
|
# Format article based on configuration and output format
|
|
16
16
|
def format_article(article, config)
|
|
17
|
-
|
|
17
|
+
with_source_fields(format_article_body(article, config), article)
|
|
18
18
|
end
|
|
19
19
|
|
|
20
|
-
# JSON records carry the dump's page and revision IDs
|
|
21
|
-
# when
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
|
|
20
|
+
# JSON records carry the dump's page and revision IDs (and page properties,
|
|
21
|
+
# when imported) right after the title, so a record can be traced
|
|
22
|
+
# back to its source; lead terms, when requested, come last.
|
|
23
|
+
def with_source_fields(result, article)
|
|
24
|
+
return result unless result.is_a?(Hash)
|
|
25
|
+
|
|
26
|
+
ids = { "page_id" => article.page_id, "revision_id" => article.revision_id }.compact
|
|
27
|
+
ids.merge!(article.page_properties.transform_keys(&:to_s)) if article.page_properties
|
|
28
|
+
out = result.each_with_object({}) do |(key, value), acc|
|
|
29
|
+
acc[key] = value
|
|
30
|
+
acc.merge!(ids) if key == "title"
|
|
31
31
|
end
|
|
32
|
+
out["lead_terms"] = article.lead_terms if article.lead_terms
|
|
33
|
+
out
|
|
32
34
|
end
|
|
33
35
|
|
|
34
36
|
def format_article_body(article, config)
|
|
@@ -4,6 +4,8 @@ require "json"
|
|
|
4
4
|
require_relative "metadata_index"
|
|
5
5
|
require_relative "fts_index"
|
|
6
6
|
require_relative "langlinks_importer"
|
|
7
|
+
require_relative "page_props_importer"
|
|
8
|
+
require_relative "link_counter"
|
|
7
9
|
require_relative "corpus"
|
|
8
10
|
require_relative "multistream"
|
|
9
11
|
require_relative "memory_monitor"
|
|
@@ -322,11 +324,97 @@ module Wp2txt
|
|
|
322
324
|
end
|
|
323
325
|
end
|
|
324
326
|
CliUI::EXIT_SUCCESS
|
|
325
|
-
rescue ArgumentError => e
|
|
327
|
+
rescue ArgumentError, Wp2txt::Error => e
|
|
326
328
|
print_error(e.message)
|
|
327
329
|
CliUI::EXIT_ERROR
|
|
328
330
|
end
|
|
329
331
|
|
|
332
|
+
# Import each article's page properties from the page_props dump of the
|
|
333
|
+
# same date as the metadata index
|
|
334
|
+
def run_import_page_props(opts)
|
|
335
|
+
db_path, dump_date, manager = built_index_for(opts)
|
|
336
|
+
return CliUI::EXIT_ERROR unless db_path
|
|
337
|
+
|
|
338
|
+
source = opts[:page_props_file] || begin
|
|
339
|
+
print_header("Downloading page_props for '#{opts[:lang]}' (#{dump_date})")
|
|
340
|
+
manager.download_page_props(date: dump_date)
|
|
341
|
+
end
|
|
342
|
+
print_mode_banner("Import Page Properties", { "Source" => File.basename(source), "Metadata DB" => db_path })
|
|
343
|
+
|
|
344
|
+
time_start = Time.now
|
|
345
|
+
result = PagePropsImporter.new(db_path).import!(source, force: opts[:update_cache])
|
|
346
|
+
if result[:status] == :already_imported
|
|
347
|
+
print_success("Page properties already imported (at #{result[:imported_at]}, #{result[:row_count]} pages).")
|
|
348
|
+
print_info_message("Use -U/--update-cache to re-import.")
|
|
349
|
+
else
|
|
350
|
+
print_success("Page properties imported: #{result[:row_count]} pages in #{format_duration(Time.now - time_start)}")
|
|
351
|
+
provenance = result[:provenance]
|
|
352
|
+
print_info("Properties", "#{provenance[:qid_count]} QIDs, #{provenance[:disambiguation_count]} disambiguation flags, " \
|
|
353
|
+
"#{provenance[:sort_key_count]} sort keys; #{provenance[:skipped_invalid_sort_keys]} invalid sort keys skipped")
|
|
354
|
+
print_info("SHA-256", result[:provenance][:source_sha256].to_s)
|
|
355
|
+
end
|
|
356
|
+
CliUI::EXIT_SUCCESS
|
|
357
|
+
rescue ArgumentError, Wp2txt::Error => e
|
|
358
|
+
print_error(e.message)
|
|
359
|
+
CliUI::EXIT_ERROR
|
|
360
|
+
end
|
|
361
|
+
|
|
362
|
+
# Count incoming links to every article and store them in the metadata index
|
|
363
|
+
def run_count_links(opts)
|
|
364
|
+
db_path, _dump_date, manager = built_index_for(opts)
|
|
365
|
+
return CliUI::EXIT_ERROR unless db_path
|
|
366
|
+
|
|
367
|
+
multistream = manager.cached_multistream_path
|
|
368
|
+
stream_offsets, = load_stream_offsets(manager.cached_index_path, opts)
|
|
369
|
+
num_processes = opts[:num_procs] || MemoryMonitor.optimal_processes
|
|
370
|
+
print_mode_banner("Count Incoming Links", {
|
|
371
|
+
"Dump" => File.basename(multistream), "Streams" => stream_offsets.size.to_s,
|
|
372
|
+
"Processes" => num_processes.to_s
|
|
373
|
+
})
|
|
374
|
+
|
|
375
|
+
time_start = Time.now
|
|
376
|
+
last_report = Time.now
|
|
377
|
+
result = LinkCounter.new(multistream, stream_offsets, db_path: db_path,
|
|
378
|
+
num_processes: num_processes).count! do |done, total|
|
|
379
|
+
now = Time.now
|
|
380
|
+
if !quiet? && (now - last_report >= DEFAULT_PROGRESS_INTERVAL || done == total)
|
|
381
|
+
last_report = now
|
|
382
|
+
puts pastel.dim(format(" [%s] %d/%d batches", now.strftime("%H:%M:%S"), done, total))
|
|
383
|
+
end
|
|
384
|
+
end
|
|
385
|
+
print_success("Incoming links counted for #{result[:articles]} articles " \
|
|
386
|
+
"(#{result[:with_inlinks]} linked at least once) in #{format_duration(Time.now - time_start)}")
|
|
387
|
+
CliUI::EXIT_SUCCESS
|
|
388
|
+
rescue ArgumentError, Wp2txt::Error => e
|
|
389
|
+
print_error(e.message)
|
|
390
|
+
CliUI::EXIT_ERROR
|
|
391
|
+
end
|
|
392
|
+
|
|
393
|
+
# The metadata index of the cached dump for --lang, or nil with an error printed
|
|
394
|
+
# @return [Array(String, String, DumpManager)] [db_path, dump_date, manager]
|
|
395
|
+
def built_index_for(opts)
|
|
396
|
+
manager = DumpManager.new(opts[:lang], cache_dir: opts[:cache_dir],
|
|
397
|
+
dump_expiry_days: CLI.config.dump_expiry_days)
|
|
398
|
+
multistream = manager.cached_multistream_path
|
|
399
|
+
unless File.exist?(multistream)
|
|
400
|
+
print_error("No cached dump found for '#{opts[:lang]}'.")
|
|
401
|
+
print_info_message("Download and index it with: wp2txt --build-index -L #{opts[:lang]}")
|
|
402
|
+
return nil
|
|
403
|
+
end
|
|
404
|
+
|
|
405
|
+
db_path = MetadataIndex.path_for(multistream, cache_dir: opts[:cache_dir])
|
|
406
|
+
meta = MetadataIndex.new(db_path)
|
|
407
|
+
unless meta.built?
|
|
408
|
+
meta.close
|
|
409
|
+
print_error("Metadata index not found for this dump.")
|
|
410
|
+
print_info_message("Build it first with: wp2txt --build-index -L #{opts[:lang]}")
|
|
411
|
+
return nil
|
|
412
|
+
end
|
|
413
|
+
dump_date = meta.stats[:dump_name][/\d{8}\z/]
|
|
414
|
+
meta.close
|
|
415
|
+
[db_path, dump_date, manager]
|
|
416
|
+
end
|
|
417
|
+
|
|
330
418
|
# Query the metadata index and print matching article titles
|
|
331
419
|
def run_find_articles(opts)
|
|
332
420
|
multistream_path, = resolve_dump_paths(opts, download: false)
|
|
@@ -4,6 +4,7 @@ require "sqlite3"
|
|
|
4
4
|
require "set"
|
|
5
5
|
require "time"
|
|
6
6
|
require "zlib"
|
|
7
|
+
require_relative "sql_dump_reader"
|
|
7
8
|
require_relative "metadata_index"
|
|
8
9
|
require_relative "version"
|
|
9
10
|
|
|
@@ -28,14 +29,6 @@ module Wp2txt
|
|
|
28
29
|
SANITY_SAMPLE_SIZE = 1000
|
|
29
30
|
SANITY_WARN_THRESHOLD = 0.9
|
|
30
31
|
|
|
31
|
-
INSERT_PREFIX = /\A\s*INSERT\s+INTO\s+`langlinks`\s+VALUES\s+/i
|
|
32
|
-
|
|
33
|
-
# MySQL backslash escapes inside mysqldump string literals
|
|
34
|
-
UNESCAPES = {
|
|
35
|
-
"0" => "\0", "'" => "'", '"' => '"', "b" => "\b", "n" => "\n",
|
|
36
|
-
"r" => "\r", "t" => "\t", "Z" => "\x1A", "\\" => "\\"
|
|
37
|
-
}.freeze
|
|
38
|
-
|
|
39
32
|
def initialize(db_path, cache_dir: nil)
|
|
40
33
|
@db_path = db_path
|
|
41
34
|
@cache_dir = cache_dir
|
|
@@ -99,7 +92,9 @@ module Wp2txt
|
|
|
99
92
|
batch.clear
|
|
100
93
|
end
|
|
101
94
|
|
|
95
|
+
rows_seen = 0
|
|
102
96
|
skipped_invalid = each_source_row(source_path) do |ll_from, ll_lang, ll_title|
|
|
97
|
+
rows_seen += 1
|
|
103
98
|
next if lang_filter && !lang_filter.include?(ll_lang)
|
|
104
99
|
|
|
105
100
|
batch << [ll_from, ll_lang, MetadataIndex.normalize_title(ll_title)]
|
|
@@ -107,6 +102,13 @@ module Wp2txt
|
|
|
107
102
|
end
|
|
108
103
|
flush.call unless batch.empty?
|
|
109
104
|
|
|
105
|
+
# Reporting success after reading nothing would leave every language
|
|
106
|
+
# query silently empty; an unreadable dump must fail loudly instead.
|
|
107
|
+
if rows_seen.zero? && skipped_invalid.zero?
|
|
108
|
+
raise Wp2txt::Error, "no langlinks rows found in #{File.basename(source_path)}; " \
|
|
109
|
+
"the file may be empty or in an unrecognized format"
|
|
110
|
+
end
|
|
111
|
+
|
|
110
112
|
# Indexes are created after the load, not before (insert speed)
|
|
111
113
|
db.execute("CREATE INDEX idx_langlinks_from ON langlinks(ll_from, ll_lang)")
|
|
112
114
|
db.execute("CREATE INDEX idx_langlinks_lang_title ON langlinks(ll_lang, ll_title)")
|
|
@@ -224,32 +226,18 @@ module Wp2txt
|
|
|
224
226
|
# tagged UTF-8 and validated — a garbled title could never join
|
|
225
227
|
# pages.title anyway, so such rows are skipped (and counted), not scrubbed.
|
|
226
228
|
def each_source_row(source_path)
|
|
227
|
-
io = if source_path.end_with?(".gz")
|
|
228
|
-
# GzipReader ignores set_encoding; the encoding must be given
|
|
229
|
-
# at open time (lines must come out as BINARY — see below)
|
|
230
|
-
Zlib::GzipReader.open(source_path, encoding: Encoding::BINARY.to_s)
|
|
231
|
-
else
|
|
232
|
-
File.open(source_path, "rb")
|
|
233
|
-
end
|
|
234
|
-
|
|
235
229
|
skipped = 0
|
|
236
|
-
|
|
237
|
-
|
|
238
|
-
|
|
239
|
-
|
|
240
|
-
|
|
241
|
-
|
|
242
|
-
|
|
243
|
-
unless lang.valid_encoding? && title.valid_encoding?
|
|
244
|
-
skipped += 1
|
|
245
|
-
next
|
|
246
|
-
end
|
|
247
|
-
|
|
248
|
-
yield ll_from.to_i, lang, title
|
|
230
|
+
SqlDumpReader.each_insert_line(source_path, "langlinks") do |line|
|
|
231
|
+
line.scan(TUPLE_REGEX) do |ll_from, ll_lang, ll_title|
|
|
232
|
+
lang = SqlDumpReader.unescape(ll_lang).force_encoding(Encoding::UTF_8)
|
|
233
|
+
title = SqlDumpReader.unescape(ll_title).force_encoding(Encoding::UTF_8)
|
|
234
|
+
unless lang.valid_encoding? && title.valid_encoding?
|
|
235
|
+
skipped += 1
|
|
236
|
+
next
|
|
249
237
|
end
|
|
238
|
+
|
|
239
|
+
yield ll_from.to_i, lang, title
|
|
250
240
|
end
|
|
251
|
-
ensure
|
|
252
|
-
io.close
|
|
253
241
|
end
|
|
254
242
|
skipped
|
|
255
243
|
end
|
|
@@ -262,12 +250,8 @@ module Wp2txt
|
|
|
262
250
|
# (equivalent to the old parser skipping malformed tuples)
|
|
263
251
|
TUPLE_REGEX = /\((\d+),'((?:[^'\\]|\\.)*)','((?:[^'\\]|\\.)*)'\)/
|
|
264
252
|
|
|
265
|
-
UNESCAPE_REGEX = /\\(.)/m
|
|
266
253
|
|
|
267
254
|
# Resolve MySQL backslash escapes in a captured string literal
|
|
268
255
|
# (same mapping as the old hand-rolled parser)
|
|
269
|
-
def unescape_mysql(str)
|
|
270
|
-
str.gsub(UNESCAPE_REGEX) { UNESCAPES[::Regexp.last_match(1)] || ::Regexp.last_match(1) }
|
|
271
|
-
end
|
|
272
256
|
end
|
|
273
257
|
end
|
|
@@ -0,0 +1,228 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require_relative "wikitext_regions"
|
|
4
|
+
|
|
5
|
+
module Wp2txt
|
|
6
|
+
# Finds the terms an article introduces in its lead: bold spans in the first
|
|
7
|
+
# paragraph that has one, the parenthesized notes written right after each,
|
|
8
|
+
# and reading (ruby) templates. Nothing is judged — which note is a reading,
|
|
9
|
+
# a native spelling, or a date is left to the caller.
|
|
10
|
+
#
|
|
11
|
+
# Positions are character offsets [start, end) into the article's wikitext
|
|
12
|
+
# as stored in the dump, XML entities decoded, before any other processing
|
|
13
|
+
# (comments included), so a caller holding the same text can cut out and
|
|
14
|
+
# hash exactly the span a term came from.
|
|
15
|
+
module LeadTerms
|
|
16
|
+
MAX_TERMS = 5
|
|
17
|
+
HEADING = /^={2,6}[^=\n].*?={2,6}[ \t]*$/
|
|
18
|
+
OPENERS = { "(" => ")", "(" => ")" }.freeze
|
|
19
|
+
SEPARATORS = ["、", ",", ",", ";", ";"].freeze
|
|
20
|
+
PARAGRAPH_BREAK = /\n[ \t\r]*\n/
|
|
21
|
+
# Regions whose bold text is not the article's own lead prose
|
|
22
|
+
SKIP_OPEN = { "{{" => "}}", "{|" => "|}", "<!--" => "-->" }.freeze
|
|
23
|
+
REF_OPEN = /\A<ref(?:\s[^>]*)?>/i
|
|
24
|
+
REF_SELF_CLOSING = /\A<ref(?:\s[^>]*)?\/>/i
|
|
25
|
+
FILE_LINK = /\A\[\[\s*(?:file|image|ファイル|画像|media)\s*:/i
|
|
26
|
+
|
|
27
|
+
module_function
|
|
28
|
+
|
|
29
|
+
# @param wikitext [String] decoded article wikitext
|
|
30
|
+
# @param render [#call] turns a wikitext fragment into clean text
|
|
31
|
+
# @return [Array<Hash>] terms in order of appearance, at most MAX_TERMS
|
|
32
|
+
def extract(wikitext, render:)
|
|
33
|
+
return [] if wikitext.nil? || wikitext.empty?
|
|
34
|
+
|
|
35
|
+
visible = WikitextRegions.mask_lead(wikitext)
|
|
36
|
+
bolds, rubies, lead_end = scan(visible, visible.length, source: wikitext)
|
|
37
|
+
original_render = render
|
|
38
|
+
render = ->(fragment) { original_render.call(fragment.gsub(WikitextRegions::COMMENT, "")) }
|
|
39
|
+
terms = []
|
|
40
|
+
|
|
41
|
+
if (first = bolds.first)
|
|
42
|
+
paragraph = paragraph_bounds(visible, first[0], lead_end)
|
|
43
|
+
bolds.select { |s, e| s >= paragraph[0] && e <= paragraph[1] }.first(MAX_TERMS).each do |s, e|
|
|
44
|
+
terms << bold_term(wikitext, s, e, paragraph[1], render)
|
|
45
|
+
end
|
|
46
|
+
end
|
|
47
|
+
rubies.each do |s, e, parts|
|
|
48
|
+
terms << { "text" => render.call(parts[1].to_s).strip, "reading" => render.call(parts[2].to_s).strip,
|
|
49
|
+
"source" => parts[0].strip, "span" => { "template" => [s, e] } }
|
|
50
|
+
end
|
|
51
|
+
|
|
52
|
+
terms.sort_by { |t| t["span"].values.first.first }.first(MAX_TERMS)
|
|
53
|
+
.each_with_index.map { |t, i| { "index" => i }.merge(t) }
|
|
54
|
+
end
|
|
55
|
+
|
|
56
|
+
# Top-level bold spans [start, end) including the quote marks, and ruby
|
|
57
|
+
# templates [start, end, [name, text, reading]] found in the lead
|
|
58
|
+
def scan(text, limit, source: text)
|
|
59
|
+
bolds = []
|
|
60
|
+
rubies = []
|
|
61
|
+
i = 0
|
|
62
|
+
open_bold = nil
|
|
63
|
+
while i < limit
|
|
64
|
+
if (i.zero? || text[i - 1] == "\n") && text[i] == "=" &&
|
|
65
|
+
HEADING.match?(text[i...(text.index("\n", i) || limit)])
|
|
66
|
+
limit = i
|
|
67
|
+
break
|
|
68
|
+
elsif text[i] == "\n"
|
|
69
|
+
open_bold = nil # bold does not continue across lines
|
|
70
|
+
i += 1
|
|
71
|
+
elsif (close = SKIP_OPEN[text[i, 4] == "<!--" ? "<!--" : text[i, 2]])
|
|
72
|
+
opener = text[i, 4] == "<!--" ? "<!--" : text[i, 2]
|
|
73
|
+
stop = matching_end(text, i, opener, close)
|
|
74
|
+
if opener == "{{" && (ruby = ruby_template(source[(i + 2)...(stop - 2)]))
|
|
75
|
+
rubies << [i, stop, ruby]
|
|
76
|
+
end
|
|
77
|
+
i = stop
|
|
78
|
+
elsif text[i, 2] == "[[" && FILE_LINK.match?(text[i, 40])
|
|
79
|
+
i = matching_end(text, i, "[[", "]]")
|
|
80
|
+
elsif text[i] == "<" && (m = REF_SELF_CLOSING.match(text[i, 200]))
|
|
81
|
+
i += m[0].length
|
|
82
|
+
elsif text[i] == "<" && (m = REF_OPEN.match(text[i, 200]))
|
|
83
|
+
close_at = text.index(%r{</ref\s*>}i, i + m[0].length)
|
|
84
|
+
i = close_at ? text.index(">", close_at) + 1 : limit
|
|
85
|
+
elsif text[i, 3] == "'''"
|
|
86
|
+
run = text[i..][/\A'+/].length
|
|
87
|
+
if open_bold
|
|
88
|
+
bolds << [open_bold, i + run]
|
|
89
|
+
open_bold = nil
|
|
90
|
+
else
|
|
91
|
+
open_bold = i
|
|
92
|
+
end
|
|
93
|
+
i += run
|
|
94
|
+
else
|
|
95
|
+
i += 1
|
|
96
|
+
end
|
|
97
|
+
end
|
|
98
|
+
[bolds, rubies, limit]
|
|
99
|
+
end
|
|
100
|
+
|
|
101
|
+
# End index (exclusive) of the construct opened at start, honouring nesting
|
|
102
|
+
def matching_end(text, start, opener, closer)
|
|
103
|
+
depth = 0
|
|
104
|
+
i = start
|
|
105
|
+
while i < text.length
|
|
106
|
+
if text[i, opener.length] == opener
|
|
107
|
+
depth += 1
|
|
108
|
+
i += opener.length
|
|
109
|
+
elsif text[i, closer.length] == closer
|
|
110
|
+
depth -= 1
|
|
111
|
+
i += closer.length
|
|
112
|
+
return i if depth.zero?
|
|
113
|
+
else
|
|
114
|
+
i += 1
|
|
115
|
+
end
|
|
116
|
+
end
|
|
117
|
+
text.length
|
|
118
|
+
end
|
|
119
|
+
|
|
120
|
+
def ruby_template(content)
|
|
121
|
+
parts = split_top_level(content, ["|"])
|
|
122
|
+
name = parts.first.to_s
|
|
123
|
+
return nil unless ruby_name?(name)
|
|
124
|
+
|
|
125
|
+
[name, parts[1], parts[2]]
|
|
126
|
+
end
|
|
127
|
+
|
|
128
|
+
# Same name rule as the cleaner: "_" and " " alike, case-insensitive
|
|
129
|
+
def ruby_name?(name)
|
|
130
|
+
@ruby_names ||= Wp2txt::RUBY_TEXT_TEMPLATES.to_set { |t| t.tr("_", " ").strip.downcase }
|
|
131
|
+
@ruby_names.include?(name.to_s.tr("_", " ").strip.downcase)
|
|
132
|
+
end
|
|
133
|
+
|
|
134
|
+
def paragraph_bounds(text, pos, limit)
|
|
135
|
+
start = 0
|
|
136
|
+
text[0...pos].to_enum(:scan, PARAGRAPH_BREAK).each { start = Regexp.last_match.end(0) }
|
|
137
|
+
stop = text.index(PARAGRAPH_BREAK, pos) || limit
|
|
138
|
+
[start, [stop, limit].min]
|
|
139
|
+
end
|
|
140
|
+
|
|
141
|
+
def bold_term(text, s, e, limit, render)
|
|
142
|
+
inner = text[s...e].sub(/\A'+/, "").sub(/'+\z/, "")
|
|
143
|
+
term = { "text" => render.call(plain_ruby(inner)).strip, "notes" => [], "notes_text" => nil,
|
|
144
|
+
"source" => "bold", "span" => { "bold" => [s, e] } }
|
|
145
|
+
j = e
|
|
146
|
+
j += 1 while j < limit && [" ", "\t", " ", "\r", "\n"].include?(text[j])
|
|
147
|
+
return term if j >= limit
|
|
148
|
+
closer = OPENERS[text[j]]
|
|
149
|
+
return term unless closer
|
|
150
|
+
|
|
151
|
+
pe = paren_end(text, j, text[j], closer, limit)
|
|
152
|
+
return term unless pe
|
|
153
|
+
|
|
154
|
+
raw = text[(j + 1)...(pe - 1)]
|
|
155
|
+
term["notes_text"] = render.call(raw).strip
|
|
156
|
+
term["notes"] = split_top_level(raw, SEPARATORS).map { |part| render.call(part).strip }.reject(&:empty?)
|
|
157
|
+
term["span"]["paren"] = [j, pe]
|
|
158
|
+
term
|
|
159
|
+
end
|
|
160
|
+
|
|
161
|
+
# Index after the bracket closing the one at start, or nil if unclosed
|
|
162
|
+
def paren_end(text, start, opener, closer, limit)
|
|
163
|
+
depth = 0
|
|
164
|
+
i = start
|
|
165
|
+
while i < limit
|
|
166
|
+
two = text[i, 2]
|
|
167
|
+
if text[i] == "<" && (stop = WikitextRegions.end_at(text, i, lead: true))
|
|
168
|
+
i = stop
|
|
169
|
+
next
|
|
170
|
+
elsif ["{{", "[["].include?(two)
|
|
171
|
+
i = matching_end(text, i, two, two == "{{" ? "}}" : "]]")
|
|
172
|
+
next
|
|
173
|
+
end
|
|
174
|
+
if [opener, "(", "("].include?(text[i])
|
|
175
|
+
depth += 1
|
|
176
|
+
elsif [closer, ")", ")"].include?(text[i])
|
|
177
|
+
depth -= 1
|
|
178
|
+
return i + 1 if depth.zero?
|
|
179
|
+
elsif text[i] == "\n" && /\A\n[ \t\r]*\n/.match?(text[i..])
|
|
180
|
+
return nil
|
|
181
|
+
end
|
|
182
|
+
i += 1
|
|
183
|
+
end
|
|
184
|
+
nil
|
|
185
|
+
end
|
|
186
|
+
|
|
187
|
+
# Split at separators that sit outside brackets, templates, and links
|
|
188
|
+
def split_top_level(text, separators)
|
|
189
|
+
parts = [+""]
|
|
190
|
+
depth = 0
|
|
191
|
+
i = 0
|
|
192
|
+
while i < text.length
|
|
193
|
+
two = text[i, 2]
|
|
194
|
+
if text[i] == "<" && (stop = WikitextRegions.end_at(text, i, lead: true))
|
|
195
|
+
parts.last << text[i...stop]
|
|
196
|
+
i = stop
|
|
197
|
+
elsif ["{{", "[["].include?(two)
|
|
198
|
+
depth += 1
|
|
199
|
+
parts.last << two
|
|
200
|
+
i += 2
|
|
201
|
+
elsif ["}}", "]]"].include?(two)
|
|
202
|
+
depth -= 1
|
|
203
|
+
parts.last << two
|
|
204
|
+
i += 2
|
|
205
|
+
else
|
|
206
|
+
ch = text[i]
|
|
207
|
+
depth += 1 if ["(", "("].include?(ch)
|
|
208
|
+
depth -= 1 if [")", ")"].include?(ch)
|
|
209
|
+
if depth <= 0 && separators.include?(ch)
|
|
210
|
+
parts << +""
|
|
211
|
+
else
|
|
212
|
+
parts.last << ch
|
|
213
|
+
end
|
|
214
|
+
i += 1
|
|
215
|
+
end
|
|
216
|
+
end
|
|
217
|
+
parts
|
|
218
|
+
end
|
|
219
|
+
|
|
220
|
+
# In a bold headword, a ruby template stands for its base text; the
|
|
221
|
+
# reading is reported separately as its own term
|
|
222
|
+
def plain_ruby(fragment)
|
|
223
|
+
fragment.gsub(/\{\{([^{}]*)\}\}/) do |whole|
|
|
224
|
+
(ruby = ruby_template(Regexp.last_match(1))) ? ruby[1].to_s : whole
|
|
225
|
+
end
|
|
226
|
+
end
|
|
227
|
+
end
|
|
228
|
+
end
|
|
@@ -0,0 +1,171 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "parallel"
|
|
4
|
+
require "sqlite3"
|
|
5
|
+
require "time"
|
|
6
|
+
require_relative "metadata_index"
|
|
7
|
+
require_relative "version"
|
|
8
|
+
require_relative "text_processing"
|
|
9
|
+
require_relative "wikitext_regions"
|
|
10
|
+
|
|
11
|
+
module Wp2txt
|
|
12
|
+
# Counts, for every article, how many other articles link to it, and stores
|
|
13
|
+
# the result in the metadata index as page_inlinks(page_id, inlinks,
|
|
14
|
+
# via_redirects).
|
|
15
|
+
#
|
|
16
|
+
# Rules (recorded as RULE_VERSION so counts from different rules are never
|
|
17
|
+
# compared unknowingly):
|
|
18
|
+
# - sources are articles (namespace 0) that are not redirects
|
|
19
|
+
# - each source counts once per target, however many times it links there
|
|
20
|
+
# - a link to a redirect counts for the redirect's target (one hop);
|
|
21
|
+
# via_redirects is how many sources reached the target only that way
|
|
22
|
+
# - only links written in the article's own wikitext count; links that
|
|
23
|
+
# templates add when rendered (navigation boxes) are not in the dump text
|
|
24
|
+
# - commented-out links do not count
|
|
25
|
+
class LinkCounter
|
|
26
|
+
include Wp2txt
|
|
27
|
+
|
|
28
|
+
RULE_VERSION = "2"
|
|
29
|
+
STREAMS_PER_BATCH = 50
|
|
30
|
+
LINK_REGEX = /\[\[([^\[\]|<>{}\n]+)(?=\||\]\])/
|
|
31
|
+
COMMENT_REGEX = /<!--.*?-->/m
|
|
32
|
+
|
|
33
|
+
def initialize(multistream_path, stream_offsets, db_path:, num_processes: 4)
|
|
34
|
+
@multistream_path = multistream_path
|
|
35
|
+
@stream_offsets = stream_offsets
|
|
36
|
+
@db_path = db_path
|
|
37
|
+
@num_processes = num_processes
|
|
38
|
+
end
|
|
39
|
+
|
|
40
|
+
# @return [Hash] { articles:, with_inlinks:, rule_version: }
|
|
41
|
+
def count!(&progress)
|
|
42
|
+
titles, redirects = load_titles
|
|
43
|
+
# Inherited by the forked workers (copy-on-write); they only read it
|
|
44
|
+
@redirects = redirects
|
|
45
|
+
|
|
46
|
+
direct = Hash.new(0)
|
|
47
|
+
via = Hash.new(0)
|
|
48
|
+
pairs = @stream_offsets.zip(@stream_offsets[1..].to_a + [nil])
|
|
49
|
+
batches = pairs.each_slice(STREAMS_PER_BATCH).to_a
|
|
50
|
+
done = 0
|
|
51
|
+
|
|
52
|
+
Parallel.each(
|
|
53
|
+
batches,
|
|
54
|
+
in_processes: @num_processes,
|
|
55
|
+
finish: lambda { |_item, _idx, result|
|
|
56
|
+
result[:direct].each { |t, n| direct[t] += n }
|
|
57
|
+
result[:via].each { |t, n| via[t] += n }
|
|
58
|
+
done += 1
|
|
59
|
+
progress&.call(done, batches.size)
|
|
60
|
+
}
|
|
61
|
+
) do |batch|
|
|
62
|
+
scan_batch(batch)
|
|
63
|
+
end
|
|
64
|
+
|
|
65
|
+
write(titles, direct, via)
|
|
66
|
+
end
|
|
67
|
+
|
|
68
|
+
private
|
|
69
|
+
|
|
70
|
+
# Articles (title => page_id) and redirects (title => target title)
|
|
71
|
+
def load_titles
|
|
72
|
+
db = SQLite3::Database.new(@db_path, readonly: true)
|
|
73
|
+
@case_rule = db.get_first_value("SELECT value FROM metadata WHERE key = 'case_rule'") || "first-letter"
|
|
74
|
+
titles = {}
|
|
75
|
+
redirects = {}
|
|
76
|
+
db.execute("SELECT page_id, title, redirect_to FROM pages WHERE namespace = 0") do |id, title, target|
|
|
77
|
+
if target
|
|
78
|
+
redirects[title] = normalize_target(target)
|
|
79
|
+
else
|
|
80
|
+
titles[title] = id
|
|
81
|
+
end
|
|
82
|
+
end
|
|
83
|
+
[titles, redirects]
|
|
84
|
+
ensure
|
|
85
|
+
db&.close
|
|
86
|
+
end
|
|
87
|
+
|
|
88
|
+
# Runs in a worker: returns per-target counts for this batch of streams
|
|
89
|
+
def scan_batch(offset_pairs)
|
|
90
|
+
direct = Hash.new(0)
|
|
91
|
+
via = Hash.new(0)
|
|
92
|
+
File.open(@multistream_path, "rb") do |f|
|
|
93
|
+
offset_pairs.each do |offset, next_offset|
|
|
94
|
+
f.seek(offset)
|
|
95
|
+
data = next_offset ? f.read(next_offset - offset) : f.read
|
|
96
|
+
xml = MetadataIndexBuilder.decompress_bz2(data)
|
|
97
|
+
xml.scan(MetadataIndexBuilder::PAGE_BLOCK_REGEX) do
|
|
98
|
+
count_page(::Regexp.last_match(1), direct, via)
|
|
99
|
+
end
|
|
100
|
+
end
|
|
101
|
+
end
|
|
102
|
+
{ direct: direct, via: via }
|
|
103
|
+
end
|
|
104
|
+
|
|
105
|
+
def count_page(block, direct, via)
|
|
106
|
+
return unless Wp2txt.namespace_id(block[MetadataIndexBuilder::NS_REGEX, 1]).zero?
|
|
107
|
+
|
|
108
|
+
text = MetadataIndexBuilder.unescape_xml(block[MetadataIndexBuilder::TEXT_REGEX, 1] || "")
|
|
109
|
+
return if Wp2txt::REDIRECT_REGEX.match?(text)
|
|
110
|
+
|
|
111
|
+
reached = {} # target => true if reached directly at least once
|
|
112
|
+
WikitextRegions.remove_literal(text).scan(LINK_REGEX) do |(raw)|
|
|
113
|
+
name = normalize_target(raw)
|
|
114
|
+
next if name.empty?
|
|
115
|
+
|
|
116
|
+
if (target = @redirects[name])
|
|
117
|
+
reached[target] ||= false
|
|
118
|
+
else
|
|
119
|
+
reached[name] = true
|
|
120
|
+
end
|
|
121
|
+
end
|
|
122
|
+
reached.each do |target, directly|
|
|
123
|
+
direct[target] += 1
|
|
124
|
+
via[target] += 1 unless directly
|
|
125
|
+
end
|
|
126
|
+
end
|
|
127
|
+
|
|
128
|
+
def normalize_target(raw)
|
|
129
|
+
# Decode each reference in the original input, never references created
|
|
130
|
+
# by decoding another one (special_chr has two decoding stages).
|
|
131
|
+
decoded = raw.gsub(/&(?:#[xX][0-9a-fA-F]+|#\d+|[a-zA-Z][a-zA-Z0-9]*);/) { |entity| special_chr(entity) }
|
|
132
|
+
title = decoded.split("#", 2).first.to_s.strip.sub(/\A:/, "")
|
|
133
|
+
MetadataIndex.normalize_title(title, case_rule: @case_rule || "first-letter")
|
|
134
|
+
end
|
|
135
|
+
|
|
136
|
+
def write(titles, direct, via)
|
|
137
|
+
db = SQLite3::Database.new(@db_path)
|
|
138
|
+
db.busy_timeout = 5000
|
|
139
|
+
db.execute("DROP TABLE IF EXISTS page_inlinks")
|
|
140
|
+
db.execute("DELETE FROM metadata WHERE key LIKE 'links\\_%' ESCAPE '\\'")
|
|
141
|
+
db.execute(<<~SQL)
|
|
142
|
+
CREATE TABLE page_inlinks (
|
|
143
|
+
page_id INTEGER PRIMARY KEY,
|
|
144
|
+
inlinks INTEGER NOT NULL,
|
|
145
|
+
via_redirects INTEGER NOT NULL
|
|
146
|
+
)
|
|
147
|
+
SQL
|
|
148
|
+
with_inlinks = 0
|
|
149
|
+
db.transaction do
|
|
150
|
+
stmt = db.prepare("INSERT INTO page_inlinks (page_id, inlinks, via_redirects) VALUES (?, ?, ?)")
|
|
151
|
+
titles.each do |title, page_id|
|
|
152
|
+
n = direct[title]
|
|
153
|
+
with_inlinks += 1 if n.positive?
|
|
154
|
+
stmt.execute([page_id, n, via[title]])
|
|
155
|
+
end
|
|
156
|
+
stmt.close
|
|
157
|
+
end
|
|
158
|
+
db.execute("CREATE INDEX idx_page_inlinks_count ON page_inlinks(inlinks)")
|
|
159
|
+
{
|
|
160
|
+
links_counted_at: Time.now.utc.iso8601,
|
|
161
|
+
links_rule_version: RULE_VERSION,
|
|
162
|
+
links_wp2txt_version: Wp2txt::VERSION,
|
|
163
|
+
links_article_count: titles.size,
|
|
164
|
+
links_with_inlinks: with_inlinks
|
|
165
|
+
}.each { |k, v| db.execute("INSERT OR REPLACE INTO metadata (key, value) VALUES (?, ?)", [k.to_s, v.to_s]) }
|
|
166
|
+
{ articles: titles.size, with_inlinks: with_inlinks, rule_version: RULE_VERSION }
|
|
167
|
+
ensure
|
|
168
|
+
db&.close
|
|
169
|
+
end
|
|
170
|
+
end
|
|
171
|
+
end
|