wp2txt 2.3.3 → 2.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/.dockerignore +0 -4
- data/.gitignore +3 -5
- data/CHANGELOG.md +18 -0
- data/DEVELOPMENT.md +1 -1
- data/DEVELOPMENT_ja.md +1 -1
- data/README.md +44 -2
- data/README_ja.md +35 -2
- data/Rakefile +10 -21
- data/bin/wp2txt +79 -17
- data/bin/wp2txt-mcp +1 -1
- data/docs/INDEXES.md +61 -1
- data/lib/wp2txt/article.rb +1 -1
- data/lib/wp2txt/cli.rb +42 -0
- data/lib/wp2txt/constants.rb +24 -0
- data/lib/wp2txt/corpus.rb +11 -2
- data/lib/wp2txt/data/template_aliases.json +1 -1
- data/lib/wp2txt/extractor.rb +10 -1
- data/lib/wp2txt/formatter.rb +20 -0
- data/lib/wp2txt/index_commands.rb +89 -1
- data/lib/wp2txt/langlinks_importer.rb +19 -35
- data/lib/wp2txt/lead_terms.rb +228 -0
- data/lib/wp2txt/link_counter.rb +171 -0
- data/lib/wp2txt/metadata_index.rb +79 -3
- data/lib/wp2txt/multistream.rb +52 -2
- data/lib/wp2txt/output_writer.rb +8 -0
- data/lib/wp2txt/page_props_importer.rb +170 -0
- data/lib/wp2txt/sql_dump_reader.rb +57 -0
- data/lib/wp2txt/stream_processor.rb +42 -15
- data/lib/wp2txt/template_expander.rb +19 -0
- data/lib/wp2txt/utils.rb +5 -3
- data/lib/wp2txt/version.rb +1 -1
- data/lib/wp2txt/wikitext_regions.rb +66 -0
- data/lib/wp2txt.rb +7 -5
- data/spec/docs_sync_spec.rb +3 -3
- data/spec/langlinks_importer_spec.rb +27 -0
- data/spec/lead_terms_edge_cases_spec.rb +194 -0
- data/spec/lead_terms_links_qids_spec.rb +204 -0
- data/spec/output_integrity_spec.rb +146 -0
- data/spec/p1_correctness_spec.rb +32 -14
- data/spec/page_properties_spec.rb +161 -0
- data/spec/region_semantics_spec.rb +71 -0
- data/spec/template_passthrough_spec.rb +43 -0
- data/spec/titles_output_path_spec.rb +12 -0
- metadata +18 -1
|
@@ -0,0 +1,171 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "parallel"
|
|
4
|
+
require "sqlite3"
|
|
5
|
+
require "time"
|
|
6
|
+
require_relative "metadata_index"
|
|
7
|
+
require_relative "version"
|
|
8
|
+
require_relative "text_processing"
|
|
9
|
+
require_relative "wikitext_regions"
|
|
10
|
+
|
|
11
|
+
module Wp2txt
|
|
12
|
+
# Counts, for every article, how many other articles link to it, and stores
|
|
13
|
+
# the result in the metadata index as page_inlinks(page_id, inlinks,
|
|
14
|
+
# via_redirects).
|
|
15
|
+
#
|
|
16
|
+
# Rules (recorded as RULE_VERSION so counts from different rules are never
|
|
17
|
+
# compared unknowingly):
|
|
18
|
+
# - sources are articles (namespace 0) that are not redirects
|
|
19
|
+
# - each source counts once per target, however many times it links there
|
|
20
|
+
# - a link to a redirect counts for the redirect's target (one hop);
|
|
21
|
+
# via_redirects is how many sources reached the target only that way
|
|
22
|
+
# - only links written in the article's own wikitext count; links that
|
|
23
|
+
# templates add when rendered (navigation boxes) are not in the dump text
|
|
24
|
+
# - commented-out links do not count
|
|
25
|
+
class LinkCounter
|
|
26
|
+
include Wp2txt
|
|
27
|
+
|
|
28
|
+
RULE_VERSION = "2"
|
|
29
|
+
STREAMS_PER_BATCH = 50
|
|
30
|
+
LINK_REGEX = /\[\[([^\[\]|<>{}\n]+)(?=\||\]\])/
|
|
31
|
+
COMMENT_REGEX = /<!--.*?-->/m
|
|
32
|
+
|
|
33
|
+
def initialize(multistream_path, stream_offsets, db_path:, num_processes: 4)
|
|
34
|
+
@multistream_path = multistream_path
|
|
35
|
+
@stream_offsets = stream_offsets
|
|
36
|
+
@db_path = db_path
|
|
37
|
+
@num_processes = num_processes
|
|
38
|
+
end
|
|
39
|
+
|
|
40
|
+
# @return [Hash] { articles:, with_inlinks:, rule_version: }
|
|
41
|
+
def count!(&progress)
|
|
42
|
+
titles, redirects = load_titles
|
|
43
|
+
# Inherited by the forked workers (copy-on-write); they only read it
|
|
44
|
+
@redirects = redirects
|
|
45
|
+
|
|
46
|
+
direct = Hash.new(0)
|
|
47
|
+
via = Hash.new(0)
|
|
48
|
+
pairs = @stream_offsets.zip(@stream_offsets[1..].to_a + [nil])
|
|
49
|
+
batches = pairs.each_slice(STREAMS_PER_BATCH).to_a
|
|
50
|
+
done = 0
|
|
51
|
+
|
|
52
|
+
Parallel.each(
|
|
53
|
+
batches,
|
|
54
|
+
in_processes: @num_processes,
|
|
55
|
+
finish: lambda { |_item, _idx, result|
|
|
56
|
+
result[:direct].each { |t, n| direct[t] += n }
|
|
57
|
+
result[:via].each { |t, n| via[t] += n }
|
|
58
|
+
done += 1
|
|
59
|
+
progress&.call(done, batches.size)
|
|
60
|
+
}
|
|
61
|
+
) do |batch|
|
|
62
|
+
scan_batch(batch)
|
|
63
|
+
end
|
|
64
|
+
|
|
65
|
+
write(titles, direct, via)
|
|
66
|
+
end
|
|
67
|
+
|
|
68
|
+
private
|
|
69
|
+
|
|
70
|
+
# Articles (title => page_id) and redirects (title => target title)
|
|
71
|
+
def load_titles
|
|
72
|
+
db = SQLite3::Database.new(@db_path, readonly: true)
|
|
73
|
+
@case_rule = db.get_first_value("SELECT value FROM metadata WHERE key = 'case_rule'") || "first-letter"
|
|
74
|
+
titles = {}
|
|
75
|
+
redirects = {}
|
|
76
|
+
db.execute("SELECT page_id, title, redirect_to FROM pages WHERE namespace = 0") do |id, title, target|
|
|
77
|
+
if target
|
|
78
|
+
redirects[title] = normalize_target(target)
|
|
79
|
+
else
|
|
80
|
+
titles[title] = id
|
|
81
|
+
end
|
|
82
|
+
end
|
|
83
|
+
[titles, redirects]
|
|
84
|
+
ensure
|
|
85
|
+
db&.close
|
|
86
|
+
end
|
|
87
|
+
|
|
88
|
+
# Runs in a worker: returns per-target counts for this batch of streams
|
|
89
|
+
def scan_batch(offset_pairs)
|
|
90
|
+
direct = Hash.new(0)
|
|
91
|
+
via = Hash.new(0)
|
|
92
|
+
File.open(@multistream_path, "rb") do |f|
|
|
93
|
+
offset_pairs.each do |offset, next_offset|
|
|
94
|
+
f.seek(offset)
|
|
95
|
+
data = next_offset ? f.read(next_offset - offset) : f.read
|
|
96
|
+
xml = MetadataIndexBuilder.decompress_bz2(data)
|
|
97
|
+
xml.scan(MetadataIndexBuilder::PAGE_BLOCK_REGEX) do
|
|
98
|
+
count_page(::Regexp.last_match(1), direct, via)
|
|
99
|
+
end
|
|
100
|
+
end
|
|
101
|
+
end
|
|
102
|
+
{ direct: direct, via: via }
|
|
103
|
+
end
|
|
104
|
+
|
|
105
|
+
def count_page(block, direct, via)
|
|
106
|
+
return unless Wp2txt.namespace_id(block[MetadataIndexBuilder::NS_REGEX, 1]).zero?
|
|
107
|
+
|
|
108
|
+
text = MetadataIndexBuilder.unescape_xml(block[MetadataIndexBuilder::TEXT_REGEX, 1] || "")
|
|
109
|
+
return if Wp2txt::REDIRECT_REGEX.match?(text)
|
|
110
|
+
|
|
111
|
+
reached = {} # target => true if reached directly at least once
|
|
112
|
+
WikitextRegions.remove_literal(text).scan(LINK_REGEX) do |(raw)|
|
|
113
|
+
name = normalize_target(raw)
|
|
114
|
+
next if name.empty?
|
|
115
|
+
|
|
116
|
+
if (target = @redirects[name])
|
|
117
|
+
reached[target] ||= false
|
|
118
|
+
else
|
|
119
|
+
reached[name] = true
|
|
120
|
+
end
|
|
121
|
+
end
|
|
122
|
+
reached.each do |target, directly|
|
|
123
|
+
direct[target] += 1
|
|
124
|
+
via[target] += 1 unless directly
|
|
125
|
+
end
|
|
126
|
+
end
|
|
127
|
+
|
|
128
|
+
def normalize_target(raw)
|
|
129
|
+
# Decode each reference in the original input, never references created
|
|
130
|
+
# by decoding another one (special_chr has two decoding stages).
|
|
131
|
+
decoded = raw.gsub(/&(?:#[xX][0-9a-fA-F]+|#\d+|[a-zA-Z][a-zA-Z0-9]*);/) { |entity| special_chr(entity) }
|
|
132
|
+
title = decoded.split("#", 2).first.to_s.strip.sub(/\A:/, "")
|
|
133
|
+
MetadataIndex.normalize_title(title, case_rule: @case_rule || "first-letter")
|
|
134
|
+
end
|
|
135
|
+
|
|
136
|
+
def write(titles, direct, via)
|
|
137
|
+
db = SQLite3::Database.new(@db_path)
|
|
138
|
+
db.busy_timeout = 5000
|
|
139
|
+
db.execute("DROP TABLE IF EXISTS page_inlinks")
|
|
140
|
+
db.execute("DELETE FROM metadata WHERE key LIKE 'links\\_%' ESCAPE '\\'")
|
|
141
|
+
db.execute(<<~SQL)
|
|
142
|
+
CREATE TABLE page_inlinks (
|
|
143
|
+
page_id INTEGER PRIMARY KEY,
|
|
144
|
+
inlinks INTEGER NOT NULL,
|
|
145
|
+
via_redirects INTEGER NOT NULL
|
|
146
|
+
)
|
|
147
|
+
SQL
|
|
148
|
+
with_inlinks = 0
|
|
149
|
+
db.transaction do
|
|
150
|
+
stmt = db.prepare("INSERT INTO page_inlinks (page_id, inlinks, via_redirects) VALUES (?, ?, ?)")
|
|
151
|
+
titles.each do |title, page_id|
|
|
152
|
+
n = direct[title]
|
|
153
|
+
with_inlinks += 1 if n.positive?
|
|
154
|
+
stmt.execute([page_id, n, via[title]])
|
|
155
|
+
end
|
|
156
|
+
stmt.close
|
|
157
|
+
end
|
|
158
|
+
db.execute("CREATE INDEX idx_page_inlinks_count ON page_inlinks(inlinks)")
|
|
159
|
+
{
|
|
160
|
+
links_counted_at: Time.now.utc.iso8601,
|
|
161
|
+
links_rule_version: RULE_VERSION,
|
|
162
|
+
links_wp2txt_version: Wp2txt::VERSION,
|
|
163
|
+
links_article_count: titles.size,
|
|
164
|
+
links_with_inlinks: with_inlinks
|
|
165
|
+
}.each { |k, v| db.execute("INSERT OR REPLACE INTO metadata (key, value) VALUES (?, ?)", [k.to_s, v.to_s]) }
|
|
166
|
+
{ articles: titles.size, with_inlinks: with_inlinks, rule_version: RULE_VERSION }
|
|
167
|
+
ensure
|
|
168
|
+
db&.close
|
|
169
|
+
end
|
|
170
|
+
end
|
|
171
|
+
end
|
|
@@ -39,9 +39,10 @@ module Wp2txt
|
|
|
39
39
|
|
|
40
40
|
# Normalize a page title the way MediaWiki treats titles:
|
|
41
41
|
# underscores to spaces, trimmed, first letter capitalized
|
|
42
|
-
def self.normalize_title(name)
|
|
42
|
+
def self.normalize_title(name, case_rule: "first-letter")
|
|
43
43
|
n = name.to_s.tr("_", " ").strip.squeeze(" ")
|
|
44
44
|
return n if n.empty?
|
|
45
|
+
return n if case_rule == "case-sensitive"
|
|
45
46
|
|
|
46
47
|
n[0].upcase + n[1..].to_s
|
|
47
48
|
end
|
|
@@ -161,6 +162,69 @@ module Wp2txt
|
|
|
161
162
|
skipped_invalid: meta[:langlinks_skipped_invalid].to_i }
|
|
162
163
|
end
|
|
163
164
|
|
|
165
|
+
# Provenance and counts of imported page properties, or nil
|
|
166
|
+
def page_props_provenance
|
|
167
|
+
return nil unless File.exist?(@db_path)
|
|
168
|
+
|
|
169
|
+
meta = read_metadata
|
|
170
|
+
return nil unless meta && meta[:page_props_imported_at] && page_properties_imported?
|
|
171
|
+
|
|
172
|
+
{ source: meta[:page_props_source],
|
|
173
|
+
source_size: meta[:page_props_source_size].to_i,
|
|
174
|
+
source_sha256: meta[:page_props_source_sha256],
|
|
175
|
+
imported_at: meta[:page_props_imported_at],
|
|
176
|
+
imported_with: meta[:page_props_wp2txt_version],
|
|
177
|
+
page_count: meta[:page_props_page_count].to_i,
|
|
178
|
+
qid_count: meta[:page_props_qid_count].to_i,
|
|
179
|
+
disambiguation_count: meta[:page_props_disambiguation_count].to_i,
|
|
180
|
+
sort_key_count: meta[:page_props_sort_key_count].to_i,
|
|
181
|
+
skipped_invalid_sort_keys: meta[:page_props_skipped_invalid_sort_keys].to_i }
|
|
182
|
+
end
|
|
183
|
+
|
|
184
|
+
# How incoming links were counted (wp2txt --count-links), or nil
|
|
185
|
+
def links_provenance
|
|
186
|
+
return nil unless File.exist?(@db_path)
|
|
187
|
+
|
|
188
|
+
meta = read_metadata
|
|
189
|
+
return nil unless meta && meta[:links_counted_at]
|
|
190
|
+
|
|
191
|
+
{ counted_at: meta[:links_counted_at],
|
|
192
|
+
rule_version: meta[:links_rule_version],
|
|
193
|
+
counted_with: meta[:links_wp2txt_version],
|
|
194
|
+
article_count: meta[:links_article_count].to_i,
|
|
195
|
+
with_inlinks: meta[:links_with_inlinks].to_i }
|
|
196
|
+
end
|
|
197
|
+
|
|
198
|
+
# An incomplete import must not be reported as an absent property.
|
|
199
|
+
def self.page_properties_imported?(db)
|
|
200
|
+
!!db.get_first_value(<<~SQL)
|
|
201
|
+
SELECT 1 FROM sqlite_master WHERE type = 'table' AND name = 'page_properties'
|
|
202
|
+
AND EXISTS (SELECT 1 FROM metadata WHERE key = 'page_props_imported_at')
|
|
203
|
+
SQL
|
|
204
|
+
end
|
|
205
|
+
|
|
206
|
+
def page_properties_imported?
|
|
207
|
+
File.exist?(@db_path) && self.class.page_properties_imported?(open_db)
|
|
208
|
+
rescue SQLite3::Exception
|
|
209
|
+
false
|
|
210
|
+
end
|
|
211
|
+
|
|
212
|
+
def self.property_values(row = nil)
|
|
213
|
+
qid, sort_key, disambiguation = row
|
|
214
|
+
{ qid: qid, sort_key: sort_key, disambiguation: disambiguation == 1 }
|
|
215
|
+
end
|
|
216
|
+
|
|
217
|
+
# nil means not imported; null/false values mean imported but absent.
|
|
218
|
+
def properties_for(page_id)
|
|
219
|
+
return nil unless page_properties_imported?
|
|
220
|
+
|
|
221
|
+
self.class.property_values(open_db.get_first_row(
|
|
222
|
+
"SELECT qid, sort_key, disambiguation FROM page_properties WHERE page_id = ?", [page_id]
|
|
223
|
+
))
|
|
224
|
+
rescue SQLite3::Exception
|
|
225
|
+
nil
|
|
226
|
+
end
|
|
227
|
+
|
|
164
228
|
def close
|
|
165
229
|
@db&.close
|
|
166
230
|
@db = nil
|
|
@@ -230,7 +294,7 @@ module Wp2txt
|
|
|
230
294
|
end
|
|
231
295
|
end
|
|
232
296
|
|
|
233
|
-
def finalize_build!(source_path)
|
|
297
|
+
def finalize_build!(source_path, case_rule: "first-letter")
|
|
234
298
|
db = open_db
|
|
235
299
|
db.execute("CREATE INDEX IF NOT EXISTS idx_pages_title ON pages(title)")
|
|
236
300
|
db.execute("CREATE INDEX IF NOT EXISTS idx_pc_category ON page_categories(category)")
|
|
@@ -248,6 +312,7 @@ module Wp2txt
|
|
|
248
312
|
source_size: stat.size,
|
|
249
313
|
source_mtime: stat.mtime.to_i,
|
|
250
314
|
dump_name: dump_name,
|
|
315
|
+
case_rule: case_rule,
|
|
251
316
|
built_at: Time.now.utc.iso8601
|
|
252
317
|
)
|
|
253
318
|
db.execute("ANALYZE")
|
|
@@ -629,6 +694,7 @@ module Wp2txt
|
|
|
629
694
|
# Build the index. Yields (batches_done, batches_total) after each batch.
|
|
630
695
|
# @return [MetadataIndex] the built index
|
|
631
696
|
def build(&progress)
|
|
697
|
+
case_rule = case_rule_from_header
|
|
632
698
|
index = MetadataIndex.new(@db_path)
|
|
633
699
|
index.prepare_build!
|
|
634
700
|
# Close before Parallel forks workers so children do not inherit a
|
|
@@ -652,10 +718,20 @@ module Wp2txt
|
|
|
652
718
|
self.class.scan_batch(@multistream_path, batch)
|
|
653
719
|
end
|
|
654
720
|
|
|
655
|
-
index.finalize_build!(@multistream_path)
|
|
721
|
+
index.finalize_build!(@multistream_path, case_rule: case_rule)
|
|
656
722
|
index
|
|
657
723
|
end
|
|
658
724
|
|
|
725
|
+
# The siteinfo may occupy its own stream before the first page offset.
|
|
726
|
+
def case_rule_from_header
|
|
727
|
+
return "first-letter" if @stream_offsets.empty?
|
|
728
|
+
|
|
729
|
+
header_end = @stream_offsets.first.positive? ? @stream_offsets.first : @stream_offsets[1]
|
|
730
|
+
xml = self.class.decompress_bz2(File.binread(@multistream_path, header_end))
|
|
731
|
+
siteinfo = xml[%r{<siteinfo\b[^>]*>(.*?)</siteinfo>}m, 1].to_s
|
|
732
|
+
siteinfo[%r{<case>\s*(first-letter|case-sensitive)\s*</case>}, 1] || "first-letter"
|
|
733
|
+
end
|
|
734
|
+
|
|
659
735
|
# Scan a batch of [offset, next_offset] stream pairs.
|
|
660
736
|
# Runs inside worker processes: must not touch SQLite.
|
|
661
737
|
def self.scan_batch(multistream_path, offset_pairs)
|
data/lib/wp2txt/multistream.rb
CHANGED
|
@@ -95,6 +95,11 @@ module Wp2txt
|
|
|
95
95
|
@early_terminated == true
|
|
96
96
|
end
|
|
97
97
|
|
|
98
|
+
# Where the stream holding the last found article ends. Scanning stops as
|
|
99
|
+
# soon as every target is found, so without this the reader cannot tell how
|
|
100
|
+
# far that stream extends and would read the dump to its end.
|
|
101
|
+
attr_reader :stream_end_offset
|
|
102
|
+
|
|
98
103
|
def find_by_title(title)
|
|
99
104
|
@entries_by_title[title]
|
|
100
105
|
end
|
|
@@ -179,6 +184,14 @@ module Wp2txt
|
|
|
179
184
|
end
|
|
180
185
|
end
|
|
181
186
|
|
|
187
|
+
def next_stream_offset_after(io, offset)
|
|
188
|
+
io.each_line do |line|
|
|
189
|
+
next_offset = line.split(":", 2).first.to_i
|
|
190
|
+
return next_offset if next_offset > offset
|
|
191
|
+
end
|
|
192
|
+
nil
|
|
193
|
+
end
|
|
194
|
+
|
|
182
195
|
def parse_index_stream(io)
|
|
183
196
|
count = 0
|
|
184
197
|
io.each_line do |line|
|
|
@@ -205,6 +218,7 @@ module Wp2txt
|
|
|
205
218
|
@found_targets << title if @target_titles.include?(title)
|
|
206
219
|
if @found_targets.size == @target_titles.size
|
|
207
220
|
@early_terminated = true
|
|
221
|
+
@stream_end_offset = next_stream_offset_after(io, offset)
|
|
208
222
|
print "\r Found all #{@target_titles.size} target articles" if @show_progress
|
|
209
223
|
puts if @show_progress
|
|
210
224
|
break
|
|
@@ -223,6 +237,10 @@ module Wp2txt
|
|
|
223
237
|
|
|
224
238
|
# Reads articles from multistream bz2 files
|
|
225
239
|
class MultistreamReader
|
|
240
|
+
# A single multistream bz2 stream holds about 100 pages; anything this large
|
|
241
|
+
# past the last known offset is several streams, not one.
|
|
242
|
+
MAX_TAIL_STREAM_BYTES = 64 * 1024 * 1024
|
|
243
|
+
|
|
226
244
|
attr_reader :multistream_path, :index
|
|
227
245
|
|
|
228
246
|
# Initialize reader with multistream file and index
|
|
@@ -357,7 +375,14 @@ module Wp2txt
|
|
|
357
375
|
if next_offset
|
|
358
376
|
compressed_data = f.read(next_offset - offset)
|
|
359
377
|
else
|
|
360
|
-
# Last stream
|
|
378
|
+
# Last stream of the dump: the rest of the file is that one stream.
|
|
379
|
+
# A large remainder means the end was never recorded, and reading it
|
|
380
|
+
# in one call fails outright on some platforms (EINVAL on macOS).
|
|
381
|
+
remaining = File.size(@multistream_path) - offset
|
|
382
|
+
if remaining > MAX_TAIL_STREAM_BYTES
|
|
383
|
+
raise Wp2txt::Error, "cannot locate the end of the stream at offset #{offset} " \
|
|
384
|
+
"(#{remaining} bytes to end of file); the index may be incomplete"
|
|
385
|
+
end
|
|
361
386
|
compressed_data = f.read
|
|
362
387
|
end
|
|
363
388
|
|
|
@@ -372,7 +397,9 @@ module Wp2txt
|
|
|
372
397
|
# here would silently read gigabytes to EOF, so fail fast instead
|
|
373
398
|
raise "Stream offset #{current_offset} not found in index (#{offsets.size} streams known)" unless idx
|
|
374
399
|
|
|
375
|
-
offsets[idx + 1]
|
|
400
|
+
return offsets[idx + 1] if idx + 1 < offsets.size
|
|
401
|
+
|
|
402
|
+
@index.respond_to?(:stream_end_offset) ? @index.stream_end_offset : nil
|
|
376
403
|
end
|
|
377
404
|
|
|
378
405
|
def decompress_bz2(data)
|
|
@@ -394,6 +421,7 @@ module Wp2txt
|
|
|
394
421
|
return {
|
|
395
422
|
title: page_title,
|
|
396
423
|
id: page_node.at_xpath("id")&.text&.to_i,
|
|
424
|
+
revision_id: page_node.at_xpath("revision/id")&.text&.to_i,
|
|
397
425
|
text: page_node.at_xpath(".//text")&.text || ""
|
|
398
426
|
}
|
|
399
427
|
end
|
|
@@ -409,6 +437,7 @@ module Wp2txt
|
|
|
409
437
|
page = {
|
|
410
438
|
title: page_node.at_xpath("title")&.text,
|
|
411
439
|
id: page_node.at_xpath("id")&.text&.to_i,
|
|
440
|
+
revision_id: page_node.at_xpath("revision/id")&.text&.to_i,
|
|
412
441
|
text: page_node.at_xpath(".//text")&.text || ""
|
|
413
442
|
}
|
|
414
443
|
yield page if page[:title]
|
|
@@ -837,6 +866,27 @@ module Wp2txt
|
|
|
837
866
|
File.join(@cache_dir, "#{@lang}wiki-#{date}-langlinks.sql.gz")
|
|
838
867
|
end
|
|
839
868
|
|
|
869
|
+
def page_props_url(date)
|
|
870
|
+
wiki = "#{@lang}wiki"
|
|
871
|
+
"#{DUMP_BASE_URL}/#{wiki}/#{date}/#{wiki}-#{date}-page_props.sql.gz"
|
|
872
|
+
end
|
|
873
|
+
|
|
874
|
+
def cached_page_props_path(date)
|
|
875
|
+
File.join(@cache_dir, "#{@lang}wiki-#{date}-page_props.sql.gz")
|
|
876
|
+
end
|
|
877
|
+
|
|
878
|
+
# Download the page_props dump for an explicit dump date (must match the index)
|
|
879
|
+
def download_page_props(date:, force: false)
|
|
880
|
+
path = cached_page_props_path(date)
|
|
881
|
+
return path if File.exist?(path) && !force
|
|
882
|
+
|
|
883
|
+
url = page_props_url(date)
|
|
884
|
+
puts "Downloading page_props: #{url}"
|
|
885
|
+
$stdout.flush
|
|
886
|
+
download_file(url, path)
|
|
887
|
+
path
|
|
888
|
+
end
|
|
889
|
+
|
|
840
890
|
# Download the langlinks dump for an explicit dump date. The date MUST
|
|
841
891
|
# come from the built metadata index (not latest_dump_date) so the
|
|
842
892
|
# imported links stay pinned to the indexed dump version.
|
data/lib/wp2txt/output_writer.rb
CHANGED
|
@@ -100,6 +100,14 @@ module Wp2txt
|
|
|
100
100
|
raise Wp2txt::FileIOError, "Write failed: #{e.message}"
|
|
101
101
|
end
|
|
102
102
|
|
|
103
|
+
# Push buffered output to the file. Call before forking: a child process
|
|
104
|
+
# inherits the buffer and writes it out again when it exits.
|
|
105
|
+
def flush
|
|
106
|
+
@mutex.synchronize do
|
|
107
|
+
@current_file.flush if @current_file && !@current_file.closed?
|
|
108
|
+
end
|
|
109
|
+
end
|
|
110
|
+
|
|
103
111
|
# Close current file and finalize
|
|
104
112
|
def close
|
|
105
113
|
@mutex.synchronize do
|
|
@@ -0,0 +1,170 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "sqlite3"
|
|
4
|
+
require "time"
|
|
5
|
+
require "digest"
|
|
6
|
+
require_relative "sql_dump_reader"
|
|
7
|
+
require_relative "version"
|
|
8
|
+
|
|
9
|
+
module Wp2txt
|
|
10
|
+
# Imports Wikidata IDs, disambiguation flags, and default sort keys
|
|
11
|
+
# from the official page_props dump into the metadata index, as
|
|
12
|
+
# page_properties. Like langlinks, the dump must carry the same
|
|
13
|
+
# date as the index: IDs are only meaningful for the pages they came with.
|
|
14
|
+
class PagePropsImporter
|
|
15
|
+
BATCH_SIZE = 50_000
|
|
16
|
+
|
|
17
|
+
# (pp_page,'pp_propname','pp_value',pp_sortkey) — string classes accept
|
|
18
|
+
# escaped quotes and backslashes; sortkey is NULL or a number
|
|
19
|
+
TUPLE_REGEX = /\((\d+),'((?:[^'\\]|\\.)*)','((?:[^'\\]|\\.)*)',(?:NULL|[-+.\deE]+)\)/
|
|
20
|
+
QID_REGEX = /\AQ\d+\z/
|
|
21
|
+
|
|
22
|
+
def initialize(db_path)
|
|
23
|
+
@db_path = db_path
|
|
24
|
+
end
|
|
25
|
+
|
|
26
|
+
# "jawiki-20260901-page_props.sql.gz" => "jawiki-20260901"
|
|
27
|
+
def self.dump_name_of(path)
|
|
28
|
+
File.basename(path)[/\A[a-z0-9_\-]+?-\d{8}/]
|
|
29
|
+
end
|
|
30
|
+
|
|
31
|
+
# @return [Hash] { status: :imported | :already_imported, row_count:, provenance: }
|
|
32
|
+
def import!(source_path, force: false)
|
|
33
|
+
raise ArgumentError, "page_props file not found: #{source_path}" unless File.exist?(source_path)
|
|
34
|
+
|
|
35
|
+
db = SQLite3::Database.new(@db_path)
|
|
36
|
+
db.busy_timeout = 5000
|
|
37
|
+
dump_name = metadata_value(db, "dump_name")
|
|
38
|
+
raise ArgumentError, "metadata index is not built: #{@db_path}" unless dump_name
|
|
39
|
+
|
|
40
|
+
source_dump = self.class.dump_name_of(source_path)
|
|
41
|
+
unless source_dump && source_dump == dump_name
|
|
42
|
+
raise ArgumentError,
|
|
43
|
+
"dump version mismatch: the metadata index is #{dump_name} but the page_props file is " \
|
|
44
|
+
"#{source_dump || File.basename(source_path)} (versions must match; there is no override)"
|
|
45
|
+
end
|
|
46
|
+
|
|
47
|
+
if !force && (existing = imported_at(db))
|
|
48
|
+
return { status: :already_imported, imported_at: existing,
|
|
49
|
+
row_count: db.get_first_value("SELECT COUNT(*) FROM page_properties").to_i, provenance: read_provenance(db) }
|
|
50
|
+
end
|
|
51
|
+
|
|
52
|
+
db.execute("DROP TABLE IF EXISTS page_qids") # replace the unreleased QID-only schema
|
|
53
|
+
db.execute("DROP TABLE IF EXISTS page_properties")
|
|
54
|
+
# A failed load must leave the index looking "not imported"
|
|
55
|
+
db.execute("DELETE FROM metadata WHERE key LIKE 'page\\_props\\_%' ESCAPE '\\'")
|
|
56
|
+
db.execute(<<~SQL)
|
|
57
|
+
CREATE TABLE page_properties (
|
|
58
|
+
page_id INTEGER PRIMARY KEY,
|
|
59
|
+
qid TEXT,
|
|
60
|
+
disambiguation INTEGER NOT NULL DEFAULT 0,
|
|
61
|
+
sort_key TEXT
|
|
62
|
+
)
|
|
63
|
+
SQL
|
|
64
|
+
|
|
65
|
+
rows_seen = 0
|
|
66
|
+
skipped_invalid = 0
|
|
67
|
+
batch = []
|
|
68
|
+
flush = lambda do
|
|
69
|
+
db.transaction do
|
|
70
|
+
stmt = db.prepare(<<~SQL)
|
|
71
|
+
INSERT INTO page_properties (page_id, qid, disambiguation, sort_key) VALUES (?, ?, ?, ?)
|
|
72
|
+
ON CONFLICT(page_id) DO UPDATE SET
|
|
73
|
+
qid = COALESCE(excluded.qid, page_properties.qid),
|
|
74
|
+
disambiguation = MAX(excluded.disambiguation, page_properties.disambiguation),
|
|
75
|
+
sort_key = COALESCE(excluded.sort_key, page_properties.sort_key)
|
|
76
|
+
SQL
|
|
77
|
+
batch.each { |row| stmt.execute(row) }
|
|
78
|
+
stmt.close
|
|
79
|
+
end
|
|
80
|
+
batch.clear
|
|
81
|
+
end
|
|
82
|
+
|
|
83
|
+
SqlDumpReader.each_insert_line(source_path, "page_props") do |line|
|
|
84
|
+
line.scan(TUPLE_REGEX) do |page, name, value|
|
|
85
|
+
rows_seen += 1
|
|
86
|
+
row = [page.to_i, nil, 0, nil]
|
|
87
|
+
case name
|
|
88
|
+
when "wikibase_item"
|
|
89
|
+
qid = SqlDumpReader.unescape(value)
|
|
90
|
+
next unless QID_REGEX.match?(qid)
|
|
91
|
+
|
|
92
|
+
row[1] = qid.force_encoding(Encoding::UTF_8)
|
|
93
|
+
when "disambiguation"
|
|
94
|
+
row[2] = 1
|
|
95
|
+
when "defaultsort"
|
|
96
|
+
key = SqlDumpReader.unescape(value).force_encoding(Encoding::UTF_8)
|
|
97
|
+
unless key.valid_encoding?
|
|
98
|
+
skipped_invalid += 1
|
|
99
|
+
next
|
|
100
|
+
end
|
|
101
|
+
row[3] = key
|
|
102
|
+
else
|
|
103
|
+
next
|
|
104
|
+
end
|
|
105
|
+
batch << row
|
|
106
|
+
flush.call if batch.size >= BATCH_SIZE
|
|
107
|
+
end
|
|
108
|
+
end
|
|
109
|
+
flush.call unless batch.empty?
|
|
110
|
+
|
|
111
|
+
if rows_seen.zero?
|
|
112
|
+
raise Wp2txt::Error, "no page_props rows found in #{File.basename(source_path)}; " \
|
|
113
|
+
"the file may be empty or in an unrecognized format"
|
|
114
|
+
end
|
|
115
|
+
|
|
116
|
+
counts = db.get_first_row("SELECT COUNT(*), COUNT(qid), SUM(disambiguation), COUNT(sort_key) FROM page_properties")
|
|
117
|
+
row_count = counts[0]
|
|
118
|
+
stamp_provenance(db, source_path, counts, skipped_invalid)
|
|
119
|
+
{ status: :imported, row_count: row_count, provenance: read_provenance(db) }
|
|
120
|
+
ensure
|
|
121
|
+
db&.close
|
|
122
|
+
end
|
|
123
|
+
|
|
124
|
+
private
|
|
125
|
+
|
|
126
|
+
def metadata_value(db, key)
|
|
127
|
+
db.get_first_value("SELECT value FROM metadata WHERE key = ?", [key])
|
|
128
|
+
rescue SQLite3::Exception
|
|
129
|
+
nil
|
|
130
|
+
end
|
|
131
|
+
|
|
132
|
+
def imported_at(db)
|
|
133
|
+
table = db.get_first_value("SELECT name FROM sqlite_master WHERE type = 'table' AND name = 'page_properties'")
|
|
134
|
+
table && metadata_value(db, "page_props_imported_at")
|
|
135
|
+
end
|
|
136
|
+
|
|
137
|
+
def stamp_provenance(db, source_path, counts, skipped_invalid)
|
|
138
|
+
values = {
|
|
139
|
+
page_props_source: File.basename(source_path),
|
|
140
|
+
page_props_source_size: File.size(source_path),
|
|
141
|
+
page_props_source_sha256: Digest::SHA256.file(source_path).hexdigest,
|
|
142
|
+
page_props_imported_at: Time.now.utc.iso8601,
|
|
143
|
+
page_props_wp2txt_version: Wp2txt::VERSION,
|
|
144
|
+
page_props_page_count: counts[0],
|
|
145
|
+
page_props_qid_count: counts[1],
|
|
146
|
+
page_props_disambiguation_count: counts[2].to_i,
|
|
147
|
+
page_props_sort_key_count: counts[3],
|
|
148
|
+
page_props_skipped_invalid_sort_keys: skipped_invalid
|
|
149
|
+
}
|
|
150
|
+
stmt = db.prepare("INSERT OR REPLACE INTO metadata (key, value) VALUES (?, ?)")
|
|
151
|
+
values.each { |k, v| stmt.execute([k.to_s, v.to_s]) }
|
|
152
|
+
stmt.close
|
|
153
|
+
end
|
|
154
|
+
|
|
155
|
+
def read_provenance(db)
|
|
156
|
+
{
|
|
157
|
+
source: metadata_value(db, "page_props_source"),
|
|
158
|
+
source_size: metadata_value(db, "page_props_source_size").to_i,
|
|
159
|
+
source_sha256: metadata_value(db, "page_props_source_sha256"),
|
|
160
|
+
imported_at: metadata_value(db, "page_props_imported_at"),
|
|
161
|
+
imported_with: metadata_value(db, "page_props_wp2txt_version"),
|
|
162
|
+
page_count: metadata_value(db, "page_props_page_count").to_i,
|
|
163
|
+
qid_count: metadata_value(db, "page_props_qid_count").to_i,
|
|
164
|
+
disambiguation_count: metadata_value(db, "page_props_disambiguation_count").to_i,
|
|
165
|
+
sort_key_count: metadata_value(db, "page_props_sort_key_count").to_i,
|
|
166
|
+
skipped_invalid_sort_keys: metadata_value(db, "page_props_skipped_invalid_sort_keys").to_i
|
|
167
|
+
}
|
|
168
|
+
end
|
|
169
|
+
end
|
|
170
|
+
end
|
|
@@ -0,0 +1,57 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "zlib"
|
|
4
|
+
|
|
5
|
+
module Wp2txt
|
|
6
|
+
# Streams the rows of one table's INSERT statements out of an official
|
|
7
|
+
# MySQL table dump (.sql or .sql.gz), without loading the file whole.
|
|
8
|
+
#
|
|
9
|
+
# Dumps write each INSERT either on one line or with the header on its own
|
|
10
|
+
# line and one tuple per following line (current dumps do the latter), so
|
|
11
|
+
# every line of a statement is read up to its terminating semicolon.
|
|
12
|
+
# Statements for other tables are skipped.
|
|
13
|
+
#
|
|
14
|
+
# Lines are read as BINARY: string columns are VARBINARY and real dumps
|
|
15
|
+
# contain historically corrupted bytes. Callers match tuples with ASCII-only
|
|
16
|
+
# patterns and validate the captures themselves.
|
|
17
|
+
module SqlDumpReader
|
|
18
|
+
STATEMENT_START = /\A\s*INSERT\s+INTO\s/i
|
|
19
|
+
|
|
20
|
+
module_function
|
|
21
|
+
|
|
22
|
+
# @yield [String] each line (BINARY) that belongs to an INSERT statement for table
|
|
23
|
+
def each_insert_line(source_path, table)
|
|
24
|
+
header = /\A\s*INSERT\s+INTO\s+`#{Regexp.escape(table)}`\s+VALUES\b/i
|
|
25
|
+
io = if source_path.end_with?(".gz")
|
|
26
|
+
# GzipReader ignores set_encoding; the encoding must be given at open time
|
|
27
|
+
Zlib::GzipReader.open(source_path, encoding: Encoding::BINARY.to_s)
|
|
28
|
+
else
|
|
29
|
+
File.open(source_path, "rb")
|
|
30
|
+
end
|
|
31
|
+
inside = false
|
|
32
|
+
begin
|
|
33
|
+
io.each_line do |line|
|
|
34
|
+
inside = header.match?(line) if STATEMENT_START.match?(line)
|
|
35
|
+
next unless inside
|
|
36
|
+
|
|
37
|
+
inside = false if line.rstrip.end_with?(";")
|
|
38
|
+
yield line
|
|
39
|
+
end
|
|
40
|
+
ensure
|
|
41
|
+
io.close
|
|
42
|
+
end
|
|
43
|
+
end
|
|
44
|
+
|
|
45
|
+
UNESCAPE_REGEX = /\\(.)/m
|
|
46
|
+
# MySQL backslash escapes inside mysqldump string literals
|
|
47
|
+
UNESCAPES = {
|
|
48
|
+
"0" => "\0", "'" => "'", '"' => '"', "b" => "\b", "n" => "\n",
|
|
49
|
+
"r" => "\r", "t" => "\t", "Z" => "\x1A", "\\" => "\\"
|
|
50
|
+
}.freeze
|
|
51
|
+
|
|
52
|
+
# Undo MySQL string escaping (\' \\ \n ...) in a captured value
|
|
53
|
+
def unescape(value)
|
|
54
|
+
value.gsub(UNESCAPE_REGEX) { UNESCAPES.fetch(Regexp.last_match(1), Regexp.last_match(1)) }
|
|
55
|
+
end
|
|
56
|
+
end
|
|
57
|
+
end
|