wp2txt 2.3.4 → 2.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/.dockerignore +0 -4
- data/.gitignore +3 -5
- data/CHANGELOG.md +9 -0
- data/DEVELOPMENT.md +1 -1
- data/DEVELOPMENT_ja.md +1 -1
- data/README.md +39 -0
- data/README_ja.md +30 -0
- data/Rakefile +10 -21
- data/bin/wp2txt +61 -9
- data/bin/wp2txt-mcp +1 -1
- data/docs/INDEXES.md +61 -1
- data/lib/wp2txt/article.rb +1 -1
- data/lib/wp2txt/cli.rb +42 -0
- data/lib/wp2txt/constants.rb +13 -0
- data/lib/wp2txt/corpus.rb +9 -2
- data/lib/wp2txt/data/template_aliases.json +1 -1
- data/lib/wp2txt/extractor.rb +10 -7
- data/lib/wp2txt/formatter.rb +14 -12
- data/lib/wp2txt/index_commands.rb +89 -1
- data/lib/wp2txt/langlinks_importer.rb +19 -35
- data/lib/wp2txt/lead_terms.rb +228 -0
- data/lib/wp2txt/link_counter.rb +171 -0
- data/lib/wp2txt/metadata_index.rb +79 -3
- data/lib/wp2txt/multistream.rb +21 -0
- data/lib/wp2txt/page_props_importer.rb +170 -0
- data/lib/wp2txt/sql_dump_reader.rb +57 -0
- data/lib/wp2txt/stream_processor.rb +9 -2
- data/lib/wp2txt/template_expander.rb +19 -0
- data/lib/wp2txt/utils.rb +5 -3
- data/lib/wp2txt/version.rb +1 -1
- data/lib/wp2txt/wikitext_regions.rb +66 -0
- data/spec/docs_sync_spec.rb +3 -3
- data/spec/langlinks_importer_spec.rb +27 -0
- data/spec/lead_terms_edge_cases_spec.rb +194 -0
- data/spec/lead_terms_links_qids_spec.rb +204 -0
- data/spec/p1_correctness_spec.rb +12 -9
- data/spec/page_properties_spec.rb +161 -0
- data/spec/region_semantics_spec.rb +71 -0
- data/spec/template_passthrough_spec.rb +43 -0
- metadata +16 -1
|
@@ -39,9 +39,10 @@ module Wp2txt
|
|
|
39
39
|
|
|
40
40
|
# Normalize a page title the way MediaWiki treats titles:
|
|
41
41
|
# underscores to spaces, trimmed, first letter capitalized
|
|
42
|
-
def self.normalize_title(name)
|
|
42
|
+
def self.normalize_title(name, case_rule: "first-letter")
|
|
43
43
|
n = name.to_s.tr("_", " ").strip.squeeze(" ")
|
|
44
44
|
return n if n.empty?
|
|
45
|
+
return n if case_rule == "case-sensitive"
|
|
45
46
|
|
|
46
47
|
n[0].upcase + n[1..].to_s
|
|
47
48
|
end
|
|
@@ -161,6 +162,69 @@ module Wp2txt
|
|
|
161
162
|
skipped_invalid: meta[:langlinks_skipped_invalid].to_i }
|
|
162
163
|
end
|
|
163
164
|
|
|
165
|
+
# Provenance and counts of imported page properties, or nil
|
|
166
|
+
def page_props_provenance
|
|
167
|
+
return nil unless File.exist?(@db_path)
|
|
168
|
+
|
|
169
|
+
meta = read_metadata
|
|
170
|
+
return nil unless meta && meta[:page_props_imported_at] && page_properties_imported?
|
|
171
|
+
|
|
172
|
+
{ source: meta[:page_props_source],
|
|
173
|
+
source_size: meta[:page_props_source_size].to_i,
|
|
174
|
+
source_sha256: meta[:page_props_source_sha256],
|
|
175
|
+
imported_at: meta[:page_props_imported_at],
|
|
176
|
+
imported_with: meta[:page_props_wp2txt_version],
|
|
177
|
+
page_count: meta[:page_props_page_count].to_i,
|
|
178
|
+
qid_count: meta[:page_props_qid_count].to_i,
|
|
179
|
+
disambiguation_count: meta[:page_props_disambiguation_count].to_i,
|
|
180
|
+
sort_key_count: meta[:page_props_sort_key_count].to_i,
|
|
181
|
+
skipped_invalid_sort_keys: meta[:page_props_skipped_invalid_sort_keys].to_i }
|
|
182
|
+
end
|
|
183
|
+
|
|
184
|
+
# How incoming links were counted (wp2txt --count-links), or nil
|
|
185
|
+
def links_provenance
|
|
186
|
+
return nil unless File.exist?(@db_path)
|
|
187
|
+
|
|
188
|
+
meta = read_metadata
|
|
189
|
+
return nil unless meta && meta[:links_counted_at]
|
|
190
|
+
|
|
191
|
+
{ counted_at: meta[:links_counted_at],
|
|
192
|
+
rule_version: meta[:links_rule_version],
|
|
193
|
+
counted_with: meta[:links_wp2txt_version],
|
|
194
|
+
article_count: meta[:links_article_count].to_i,
|
|
195
|
+
with_inlinks: meta[:links_with_inlinks].to_i }
|
|
196
|
+
end
|
|
197
|
+
|
|
198
|
+
# An incomplete import must not be reported as an absent property.
|
|
199
|
+
def self.page_properties_imported?(db)
|
|
200
|
+
!!db.get_first_value(<<~SQL)
|
|
201
|
+
SELECT 1 FROM sqlite_master WHERE type = 'table' AND name = 'page_properties'
|
|
202
|
+
AND EXISTS (SELECT 1 FROM metadata WHERE key = 'page_props_imported_at')
|
|
203
|
+
SQL
|
|
204
|
+
end
|
|
205
|
+
|
|
206
|
+
def page_properties_imported?
|
|
207
|
+
File.exist?(@db_path) && self.class.page_properties_imported?(open_db)
|
|
208
|
+
rescue SQLite3::Exception
|
|
209
|
+
false
|
|
210
|
+
end
|
|
211
|
+
|
|
212
|
+
def self.property_values(row = nil)
|
|
213
|
+
qid, sort_key, disambiguation = row
|
|
214
|
+
{ qid: qid, sort_key: sort_key, disambiguation: disambiguation == 1 }
|
|
215
|
+
end
|
|
216
|
+
|
|
217
|
+
# nil means not imported; null/false values mean imported but absent.
|
|
218
|
+
def properties_for(page_id)
|
|
219
|
+
return nil unless page_properties_imported?
|
|
220
|
+
|
|
221
|
+
self.class.property_values(open_db.get_first_row(
|
|
222
|
+
"SELECT qid, sort_key, disambiguation FROM page_properties WHERE page_id = ?", [page_id]
|
|
223
|
+
))
|
|
224
|
+
rescue SQLite3::Exception
|
|
225
|
+
nil
|
|
226
|
+
end
|
|
227
|
+
|
|
164
228
|
def close
|
|
165
229
|
@db&.close
|
|
166
230
|
@db = nil
|
|
@@ -230,7 +294,7 @@ module Wp2txt
|
|
|
230
294
|
end
|
|
231
295
|
end
|
|
232
296
|
|
|
233
|
-
def finalize_build!(source_path)
|
|
297
|
+
def finalize_build!(source_path, case_rule: "first-letter")
|
|
234
298
|
db = open_db
|
|
235
299
|
db.execute("CREATE INDEX IF NOT EXISTS idx_pages_title ON pages(title)")
|
|
236
300
|
db.execute("CREATE INDEX IF NOT EXISTS idx_pc_category ON page_categories(category)")
|
|
@@ -248,6 +312,7 @@ module Wp2txt
|
|
|
248
312
|
source_size: stat.size,
|
|
249
313
|
source_mtime: stat.mtime.to_i,
|
|
250
314
|
dump_name: dump_name,
|
|
315
|
+
case_rule: case_rule,
|
|
251
316
|
built_at: Time.now.utc.iso8601
|
|
252
317
|
)
|
|
253
318
|
db.execute("ANALYZE")
|
|
@@ -629,6 +694,7 @@ module Wp2txt
|
|
|
629
694
|
# Build the index. Yields (batches_done, batches_total) after each batch.
|
|
630
695
|
# @return [MetadataIndex] the built index
|
|
631
696
|
def build(&progress)
|
|
697
|
+
case_rule = case_rule_from_header
|
|
632
698
|
index = MetadataIndex.new(@db_path)
|
|
633
699
|
index.prepare_build!
|
|
634
700
|
# Close before Parallel forks workers so children do not inherit a
|
|
@@ -652,10 +718,20 @@ module Wp2txt
|
|
|
652
718
|
self.class.scan_batch(@multistream_path, batch)
|
|
653
719
|
end
|
|
654
720
|
|
|
655
|
-
index.finalize_build!(@multistream_path)
|
|
721
|
+
index.finalize_build!(@multistream_path, case_rule: case_rule)
|
|
656
722
|
index
|
|
657
723
|
end
|
|
658
724
|
|
|
725
|
+
# The siteinfo may occupy its own stream before the first page offset.
|
|
726
|
+
def case_rule_from_header
|
|
727
|
+
return "first-letter" if @stream_offsets.empty?
|
|
728
|
+
|
|
729
|
+
header_end = @stream_offsets.first.positive? ? @stream_offsets.first : @stream_offsets[1]
|
|
730
|
+
xml = self.class.decompress_bz2(File.binread(@multistream_path, header_end))
|
|
731
|
+
siteinfo = xml[%r{<siteinfo\b[^>]*>(.*?)</siteinfo>}m, 1].to_s
|
|
732
|
+
siteinfo[%r{<case>\s*(first-letter|case-sensitive)\s*</case>}, 1] || "first-letter"
|
|
733
|
+
end
|
|
734
|
+
|
|
659
735
|
# Scan a batch of [offset, next_offset] stream pairs.
|
|
660
736
|
# Runs inside worker processes: must not touch SQLite.
|
|
661
737
|
def self.scan_batch(multistream_path, offset_pairs)
|
data/lib/wp2txt/multistream.rb
CHANGED
|
@@ -866,6 +866,27 @@ module Wp2txt
|
|
|
866
866
|
File.join(@cache_dir, "#{@lang}wiki-#{date}-langlinks.sql.gz")
|
|
867
867
|
end
|
|
868
868
|
|
|
869
|
+
def page_props_url(date)
|
|
870
|
+
wiki = "#{@lang}wiki"
|
|
871
|
+
"#{DUMP_BASE_URL}/#{wiki}/#{date}/#{wiki}-#{date}-page_props.sql.gz"
|
|
872
|
+
end
|
|
873
|
+
|
|
874
|
+
def cached_page_props_path(date)
|
|
875
|
+
File.join(@cache_dir, "#{@lang}wiki-#{date}-page_props.sql.gz")
|
|
876
|
+
end
|
|
877
|
+
|
|
878
|
+
# Download the page_props dump for an explicit dump date (must match the index)
|
|
879
|
+
def download_page_props(date:, force: false)
|
|
880
|
+
path = cached_page_props_path(date)
|
|
881
|
+
return path if File.exist?(path) && !force
|
|
882
|
+
|
|
883
|
+
url = page_props_url(date)
|
|
884
|
+
puts "Downloading page_props: #{url}"
|
|
885
|
+
$stdout.flush
|
|
886
|
+
download_file(url, path)
|
|
887
|
+
path
|
|
888
|
+
end
|
|
889
|
+
|
|
869
890
|
# Download the langlinks dump for an explicit dump date. The date MUST
|
|
870
891
|
# come from the built metadata index (not latest_dump_date) so the
|
|
871
892
|
# imported links stay pinned to the indexed dump version.
|
|
@@ -0,0 +1,170 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "sqlite3"
|
|
4
|
+
require "time"
|
|
5
|
+
require "digest"
|
|
6
|
+
require_relative "sql_dump_reader"
|
|
7
|
+
require_relative "version"
|
|
8
|
+
|
|
9
|
+
module Wp2txt
|
|
10
|
+
# Imports Wikidata IDs, disambiguation flags, and default sort keys
|
|
11
|
+
# from the official page_props dump into the metadata index, as
|
|
12
|
+
# page_properties. Like langlinks, the dump must carry the same
|
|
13
|
+
# date as the index: IDs are only meaningful for the pages they came with.
|
|
14
|
+
class PagePropsImporter
|
|
15
|
+
BATCH_SIZE = 50_000
|
|
16
|
+
|
|
17
|
+
# (pp_page,'pp_propname','pp_value',pp_sortkey) — string classes accept
|
|
18
|
+
# escaped quotes and backslashes; sortkey is NULL or a number
|
|
19
|
+
TUPLE_REGEX = /\((\d+),'((?:[^'\\]|\\.)*)','((?:[^'\\]|\\.)*)',(?:NULL|[-+.\deE]+)\)/
|
|
20
|
+
QID_REGEX = /\AQ\d+\z/
|
|
21
|
+
|
|
22
|
+
def initialize(db_path)
|
|
23
|
+
@db_path = db_path
|
|
24
|
+
end
|
|
25
|
+
|
|
26
|
+
# "jawiki-20260901-page_props.sql.gz" => "jawiki-20260901"
|
|
27
|
+
def self.dump_name_of(path)
|
|
28
|
+
File.basename(path)[/\A[a-z0-9_\-]+?-\d{8}/]
|
|
29
|
+
end
|
|
30
|
+
|
|
31
|
+
# @return [Hash] { status: :imported | :already_imported, row_count:, provenance: }
|
|
32
|
+
def import!(source_path, force: false)
|
|
33
|
+
raise ArgumentError, "page_props file not found: #{source_path}" unless File.exist?(source_path)
|
|
34
|
+
|
|
35
|
+
db = SQLite3::Database.new(@db_path)
|
|
36
|
+
db.busy_timeout = 5000
|
|
37
|
+
dump_name = metadata_value(db, "dump_name")
|
|
38
|
+
raise ArgumentError, "metadata index is not built: #{@db_path}" unless dump_name
|
|
39
|
+
|
|
40
|
+
source_dump = self.class.dump_name_of(source_path)
|
|
41
|
+
unless source_dump && source_dump == dump_name
|
|
42
|
+
raise ArgumentError,
|
|
43
|
+
"dump version mismatch: the metadata index is #{dump_name} but the page_props file is " \
|
|
44
|
+
"#{source_dump || File.basename(source_path)} (versions must match; there is no override)"
|
|
45
|
+
end
|
|
46
|
+
|
|
47
|
+
if !force && (existing = imported_at(db))
|
|
48
|
+
return { status: :already_imported, imported_at: existing,
|
|
49
|
+
row_count: db.get_first_value("SELECT COUNT(*) FROM page_properties").to_i, provenance: read_provenance(db) }
|
|
50
|
+
end
|
|
51
|
+
|
|
52
|
+
db.execute("DROP TABLE IF EXISTS page_qids") # replace the unreleased QID-only schema
|
|
53
|
+
db.execute("DROP TABLE IF EXISTS page_properties")
|
|
54
|
+
# A failed load must leave the index looking "not imported"
|
|
55
|
+
db.execute("DELETE FROM metadata WHERE key LIKE 'page\\_props\\_%' ESCAPE '\\'")
|
|
56
|
+
db.execute(<<~SQL)
|
|
57
|
+
CREATE TABLE page_properties (
|
|
58
|
+
page_id INTEGER PRIMARY KEY,
|
|
59
|
+
qid TEXT,
|
|
60
|
+
disambiguation INTEGER NOT NULL DEFAULT 0,
|
|
61
|
+
sort_key TEXT
|
|
62
|
+
)
|
|
63
|
+
SQL
|
|
64
|
+
|
|
65
|
+
rows_seen = 0
|
|
66
|
+
skipped_invalid = 0
|
|
67
|
+
batch = []
|
|
68
|
+
flush = lambda do
|
|
69
|
+
db.transaction do
|
|
70
|
+
stmt = db.prepare(<<~SQL)
|
|
71
|
+
INSERT INTO page_properties (page_id, qid, disambiguation, sort_key) VALUES (?, ?, ?, ?)
|
|
72
|
+
ON CONFLICT(page_id) DO UPDATE SET
|
|
73
|
+
qid = COALESCE(excluded.qid, page_properties.qid),
|
|
74
|
+
disambiguation = MAX(excluded.disambiguation, page_properties.disambiguation),
|
|
75
|
+
sort_key = COALESCE(excluded.sort_key, page_properties.sort_key)
|
|
76
|
+
SQL
|
|
77
|
+
batch.each { |row| stmt.execute(row) }
|
|
78
|
+
stmt.close
|
|
79
|
+
end
|
|
80
|
+
batch.clear
|
|
81
|
+
end
|
|
82
|
+
|
|
83
|
+
SqlDumpReader.each_insert_line(source_path, "page_props") do |line|
|
|
84
|
+
line.scan(TUPLE_REGEX) do |page, name, value|
|
|
85
|
+
rows_seen += 1
|
|
86
|
+
row = [page.to_i, nil, 0, nil]
|
|
87
|
+
case name
|
|
88
|
+
when "wikibase_item"
|
|
89
|
+
qid = SqlDumpReader.unescape(value)
|
|
90
|
+
next unless QID_REGEX.match?(qid)
|
|
91
|
+
|
|
92
|
+
row[1] = qid.force_encoding(Encoding::UTF_8)
|
|
93
|
+
when "disambiguation"
|
|
94
|
+
row[2] = 1
|
|
95
|
+
when "defaultsort"
|
|
96
|
+
key = SqlDumpReader.unescape(value).force_encoding(Encoding::UTF_8)
|
|
97
|
+
unless key.valid_encoding?
|
|
98
|
+
skipped_invalid += 1
|
|
99
|
+
next
|
|
100
|
+
end
|
|
101
|
+
row[3] = key
|
|
102
|
+
else
|
|
103
|
+
next
|
|
104
|
+
end
|
|
105
|
+
batch << row
|
|
106
|
+
flush.call if batch.size >= BATCH_SIZE
|
|
107
|
+
end
|
|
108
|
+
end
|
|
109
|
+
flush.call unless batch.empty?
|
|
110
|
+
|
|
111
|
+
if rows_seen.zero?
|
|
112
|
+
raise Wp2txt::Error, "no page_props rows found in #{File.basename(source_path)}; " \
|
|
113
|
+
"the file may be empty or in an unrecognized format"
|
|
114
|
+
end
|
|
115
|
+
|
|
116
|
+
counts = db.get_first_row("SELECT COUNT(*), COUNT(qid), SUM(disambiguation), COUNT(sort_key) FROM page_properties")
|
|
117
|
+
row_count = counts[0]
|
|
118
|
+
stamp_provenance(db, source_path, counts, skipped_invalid)
|
|
119
|
+
{ status: :imported, row_count: row_count, provenance: read_provenance(db) }
|
|
120
|
+
ensure
|
|
121
|
+
db&.close
|
|
122
|
+
end
|
|
123
|
+
|
|
124
|
+
private
|
|
125
|
+
|
|
126
|
+
def metadata_value(db, key)
|
|
127
|
+
db.get_first_value("SELECT value FROM metadata WHERE key = ?", [key])
|
|
128
|
+
rescue SQLite3::Exception
|
|
129
|
+
nil
|
|
130
|
+
end
|
|
131
|
+
|
|
132
|
+
def imported_at(db)
|
|
133
|
+
table = db.get_first_value("SELECT name FROM sqlite_master WHERE type = 'table' AND name = 'page_properties'")
|
|
134
|
+
table && metadata_value(db, "page_props_imported_at")
|
|
135
|
+
end
|
|
136
|
+
|
|
137
|
+
def stamp_provenance(db, source_path, counts, skipped_invalid)
|
|
138
|
+
values = {
|
|
139
|
+
page_props_source: File.basename(source_path),
|
|
140
|
+
page_props_source_size: File.size(source_path),
|
|
141
|
+
page_props_source_sha256: Digest::SHA256.file(source_path).hexdigest,
|
|
142
|
+
page_props_imported_at: Time.now.utc.iso8601,
|
|
143
|
+
page_props_wp2txt_version: Wp2txt::VERSION,
|
|
144
|
+
page_props_page_count: counts[0],
|
|
145
|
+
page_props_qid_count: counts[1],
|
|
146
|
+
page_props_disambiguation_count: counts[2].to_i,
|
|
147
|
+
page_props_sort_key_count: counts[3],
|
|
148
|
+
page_props_skipped_invalid_sort_keys: skipped_invalid
|
|
149
|
+
}
|
|
150
|
+
stmt = db.prepare("INSERT OR REPLACE INTO metadata (key, value) VALUES (?, ?)")
|
|
151
|
+
values.each { |k, v| stmt.execute([k.to_s, v.to_s]) }
|
|
152
|
+
stmt.close
|
|
153
|
+
end
|
|
154
|
+
|
|
155
|
+
def read_provenance(db)
|
|
156
|
+
{
|
|
157
|
+
source: metadata_value(db, "page_props_source"),
|
|
158
|
+
source_size: metadata_value(db, "page_props_source_size").to_i,
|
|
159
|
+
source_sha256: metadata_value(db, "page_props_source_sha256"),
|
|
160
|
+
imported_at: metadata_value(db, "page_props_imported_at"),
|
|
161
|
+
imported_with: metadata_value(db, "page_props_wp2txt_version"),
|
|
162
|
+
page_count: metadata_value(db, "page_props_page_count").to_i,
|
|
163
|
+
qid_count: metadata_value(db, "page_props_qid_count").to_i,
|
|
164
|
+
disambiguation_count: metadata_value(db, "page_props_disambiguation_count").to_i,
|
|
165
|
+
sort_key_count: metadata_value(db, "page_props_sort_key_count").to_i,
|
|
166
|
+
skipped_invalid_sort_keys: metadata_value(db, "page_props_skipped_invalid_sort_keys").to_i
|
|
167
|
+
}
|
|
168
|
+
end
|
|
169
|
+
end
|
|
170
|
+
end
|
|
@@ -0,0 +1,57 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "zlib"
|
|
4
|
+
|
|
5
|
+
module Wp2txt
|
|
6
|
+
# Streams the rows of one table's INSERT statements out of an official
|
|
7
|
+
# MySQL table dump (.sql or .sql.gz), without loading the file whole.
|
|
8
|
+
#
|
|
9
|
+
# Dumps write each INSERT either on one line or with the header on its own
|
|
10
|
+
# line and one tuple per following line (current dumps do the latter), so
|
|
11
|
+
# every line of a statement is read up to its terminating semicolon.
|
|
12
|
+
# Statements for other tables are skipped.
|
|
13
|
+
#
|
|
14
|
+
# Lines are read as BINARY: string columns are VARBINARY and real dumps
|
|
15
|
+
# contain historically corrupted bytes. Callers match tuples with ASCII-only
|
|
16
|
+
# patterns and validate the captures themselves.
|
|
17
|
+
module SqlDumpReader
|
|
18
|
+
STATEMENT_START = /\A\s*INSERT\s+INTO\s/i
|
|
19
|
+
|
|
20
|
+
module_function
|
|
21
|
+
|
|
22
|
+
# @yield [String] each line (BINARY) that belongs to an INSERT statement for table
|
|
23
|
+
def each_insert_line(source_path, table)
|
|
24
|
+
header = /\A\s*INSERT\s+INTO\s+`#{Regexp.escape(table)}`\s+VALUES\b/i
|
|
25
|
+
io = if source_path.end_with?(".gz")
|
|
26
|
+
# GzipReader ignores set_encoding; the encoding must be given at open time
|
|
27
|
+
Zlib::GzipReader.open(source_path, encoding: Encoding::BINARY.to_s)
|
|
28
|
+
else
|
|
29
|
+
File.open(source_path, "rb")
|
|
30
|
+
end
|
|
31
|
+
inside = false
|
|
32
|
+
begin
|
|
33
|
+
io.each_line do |line|
|
|
34
|
+
inside = header.match?(line) if STATEMENT_START.match?(line)
|
|
35
|
+
next unless inside
|
|
36
|
+
|
|
37
|
+
inside = false if line.rstrip.end_with?(";")
|
|
38
|
+
yield line
|
|
39
|
+
end
|
|
40
|
+
ensure
|
|
41
|
+
io.close
|
|
42
|
+
end
|
|
43
|
+
end
|
|
44
|
+
|
|
45
|
+
UNESCAPE_REGEX = /\\(.)/m
|
|
46
|
+
# MySQL backslash escapes inside mysqldump string literals
|
|
47
|
+
UNESCAPES = {
|
|
48
|
+
"0" => "\0", "'" => "'", '"' => '"', "b" => "\b", "n" => "\n",
|
|
49
|
+
"r" => "\r", "t" => "\t", "Z" => "\x1A", "\\" => "\\"
|
|
50
|
+
}.freeze
|
|
51
|
+
|
|
52
|
+
# Undo MySQL string escaping (\' \\ \n ...) in a captured value
|
|
53
|
+
def unescape(value)
|
|
54
|
+
value.gsub(UNESCAPE_REGEX) { UNESCAPES.fetch(Regexp.last_match(1), Regexp.last_match(1)) }
|
|
55
|
+
end
|
|
56
|
+
end
|
|
57
|
+
end
|
|
@@ -19,8 +19,12 @@ module Wp2txt
|
|
|
19
19
|
|
|
20
20
|
attr_reader :buffer_size, :pages_processed, :bytes_read, :redirects_skipped
|
|
21
21
|
|
|
22
|
-
|
|
22
|
+
# keep_raw_text: also hand back each article's wikitext as stored in the
|
|
23
|
+
# dump (before comment removal), as :raw_text among the with_ids values
|
|
24
|
+
def initialize(input_path, bz2_gem: false, adaptive_buffer: true, validate_bz2: true, skip_redirects: true,
|
|
25
|
+
keep_raw_text: false)
|
|
23
26
|
@input_path = input_path
|
|
27
|
+
@keep_raw_text = keep_raw_text
|
|
24
28
|
@bz2_gem = bz2_gem
|
|
25
29
|
@buffer = +""
|
|
26
30
|
@pending_bytes = +"".b
|
|
@@ -257,6 +261,7 @@ module Wp2txt
|
|
|
257
261
|
return nil unless Wp2txt.namespace_id(namespace).zero?
|
|
258
262
|
|
|
259
263
|
text = text_node.content
|
|
264
|
+
raw_text = text if @keep_raw_text
|
|
260
265
|
|
|
261
266
|
# Early redirect detection and skip (before expensive processing)
|
|
262
267
|
# Redirects start with # or # followed by redirect keyword and [[target]]
|
|
@@ -272,7 +277,9 @@ module Wp2txt
|
|
|
272
277
|
end
|
|
273
278
|
|
|
274
279
|
@pages_processed += 1
|
|
275
|
-
|
|
280
|
+
meta = Wp2txt.page_ids(page_xml)
|
|
281
|
+
meta[:raw_text] = raw_text if @keep_raw_text
|
|
282
|
+
[title, text, meta]
|
|
276
283
|
rescue Nokogiri::XML::SyntaxError
|
|
277
284
|
# Skip malformed XML
|
|
278
285
|
nil
|
|
@@ -171,6 +171,19 @@ module Wp2txt
|
|
|
171
171
|
nil
|
|
172
172
|
end
|
|
173
173
|
|
|
174
|
+
# Templates whose rendering the text-cleaning stage defines (see
|
|
175
|
+
# Wp2txt#correct_inline_template), keyed by normalized name. Limited to
|
|
176
|
+
# kinds whose output does not depend on marker settings.
|
|
177
|
+
def rendered_by_cleaner?(template_name)
|
|
178
|
+
@rendered_by_cleaner ||= [Wp2txt::RUBY_TEXT_TEMPLATES, Wp2txt::INTERWIKI_LINK_TEMPLATES]
|
|
179
|
+
.flatten.to_set { |name| name.to_s.tr("_", " ").strip.downcase }
|
|
180
|
+
@rendered_by_cleaner.include?(template_name.tr("_", " "))
|
|
181
|
+
end
|
|
182
|
+
|
|
183
|
+
def text_renderer
|
|
184
|
+
@text_renderer ||= Object.new.extend(Wp2txt)
|
|
185
|
+
end
|
|
186
|
+
|
|
174
187
|
def expand_single_template(content)
|
|
175
188
|
parts = split_template_parts(content)
|
|
176
189
|
return "" if parts.empty?
|
|
@@ -263,6 +276,12 @@ module Wp2txt
|
|
|
263
276
|
# Handle lang-xx templates (e.g., lang-fr, lang-de, lang-ja)
|
|
264
277
|
if template_name.start_with?("lang-")
|
|
265
278
|
expand_lang_xx(template_name, params)
|
|
279
|
+
elsif rendered_by_cleaner?(template_name)
|
|
280
|
+
# These carry words the article needs (a reading, a link's display
|
|
281
|
+
# text). Deleting them dropped those words; leaving the raw template
|
|
282
|
+
# for later confused links that contain it (an image caption with a
|
|
283
|
+
# "|" inside). Render them now, with the cleaner's own rules.
|
|
284
|
+
text_renderer.correct_inline_template("{{#{expand(content)}}}")
|
|
266
285
|
else
|
|
267
286
|
@preserve_unknown ? "{{#{content}}}" : ""
|
|
268
287
|
end
|
data/lib/wp2txt/utils.rb
CHANGED
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
# frozen_string_literal: true
|
|
2
2
|
|
|
3
3
|
require "strscan"
|
|
4
|
+
require_relative "wikitext_regions"
|
|
4
5
|
require_relative "constants"
|
|
5
6
|
require_relative "regex"
|
|
6
7
|
require_relative "text_processing"
|
|
@@ -574,8 +575,9 @@ module Wp2txt
|
|
|
574
575
|
# Helper to check if template name matches any in a list (case-insensitive)
|
|
575
576
|
def template_matches?(name, template_list)
|
|
576
577
|
return false if template_list.nil? || template_list.empty?
|
|
577
|
-
|
|
578
|
-
|
|
578
|
+
# MediaWiki treats "_" and " " in template names as the same character
|
|
579
|
+
normalized_name = name.to_s.tr("_", " ").strip.downcase
|
|
580
|
+
template_list.any? { |t| t.tr("_", " ").downcase == normalized_name }
|
|
579
581
|
end
|
|
580
582
|
|
|
581
583
|
def correct_inline_template(str, enabled_markers = [], extract_citations = false)
|
|
@@ -583,7 +585,7 @@ module Wp2txt
|
|
|
583
585
|
return str unless str.include?("{{")
|
|
584
586
|
|
|
585
587
|
process_nested_single_pass(str, "{{", "}}") do |contents|
|
|
586
|
-
parts =
|
|
588
|
+
parts = WikitextRegions.split_pipes(contents)
|
|
587
589
|
template_name = (parts[0] || "").strip
|
|
588
590
|
template_name_lower = template_name.downcase
|
|
589
591
|
|
data/lib/wp2txt/version.rb
CHANGED
|
@@ -0,0 +1,66 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Wp2txt
|
|
4
|
+
# Literal regions suppress wikitext parsing. Galleries and timelines can
|
|
5
|
+
# contain links, but their contents are not the article's lead prose.
|
|
6
|
+
module WikitextRegions
|
|
7
|
+
LITERAL_TAGS = %w[nowiki pre math chem ce score syntaxhighlight source graph mapframe templatedata].freeze
|
|
8
|
+
NON_PROSE_TAGS = %w[gallery timeline].freeze
|
|
9
|
+
COMMENT = /<!--.*?(?:-->|\z)/m
|
|
10
|
+
|
|
11
|
+
def self.region_pattern(tags)
|
|
12
|
+
/<!--.*?(?:-->|\z)|<(?<tag>#{tags.join('|')})(?=[\s\/>])(?:"[^"]*"|'[^']*'|[^'">])*?(?:\/>|>.*?(?:<\/\k<tag>\s*>|\z))/mi
|
|
13
|
+
end
|
|
14
|
+
|
|
15
|
+
LITERAL_REGION = region_pattern(LITERAL_TAGS)
|
|
16
|
+
LEAD_REGION = region_pattern(LITERAL_TAGS + NON_PROSE_TAGS)
|
|
17
|
+
LITERAL_REGION_AT = /\G(?:#{LITERAL_REGION})/
|
|
18
|
+
LEAD_REGION_AT = /\G(?:#{LEAD_REGION})/
|
|
19
|
+
|
|
20
|
+
module_function
|
|
21
|
+
|
|
22
|
+
def end_at(text, offset, lead: false)
|
|
23
|
+
match = (lead ? LEAD_REGION_AT : LITERAL_REGION_AT).match(text, offset)
|
|
24
|
+
match.end(0) if match
|
|
25
|
+
end
|
|
26
|
+
|
|
27
|
+
def remove_literal(text)
|
|
28
|
+
text.gsub(LITERAL_REGION, "")
|
|
29
|
+
end
|
|
30
|
+
|
|
31
|
+
# One space per codepoint, including internal newlines: an excluded
|
|
32
|
+
# region must not introduce a paragraph or heading boundary.
|
|
33
|
+
def mask_lead(text)
|
|
34
|
+
text.gsub(LEAD_REGION) { |region| " " * region.length }
|
|
35
|
+
end
|
|
36
|
+
|
|
37
|
+
def split_pipes(text)
|
|
38
|
+
parts = [+""]
|
|
39
|
+
stack = []
|
|
40
|
+
i = 0
|
|
41
|
+
while i < text.length
|
|
42
|
+
two = text[i, 2]
|
|
43
|
+
if text[i] == "<" && (stop = end_at(text, i))
|
|
44
|
+
parts.last << text[i...stop]
|
|
45
|
+
i = stop
|
|
46
|
+
elsif ["{{", "[["].include?(two)
|
|
47
|
+
stack << (two == "{{" ? "}}" : "]]")
|
|
48
|
+
parts.last << two
|
|
49
|
+
i += 2
|
|
50
|
+
elsif !stack.empty? && two == stack.last
|
|
51
|
+
stack.pop
|
|
52
|
+
parts.last << two
|
|
53
|
+
i += 2
|
|
54
|
+
else
|
|
55
|
+
if text[i] == "|" && stack.empty?
|
|
56
|
+
parts << +""
|
|
57
|
+
else
|
|
58
|
+
parts.last << text[i]
|
|
59
|
+
end
|
|
60
|
+
i += 1
|
|
61
|
+
end
|
|
62
|
+
end
|
|
63
|
+
parts
|
|
64
|
+
end
|
|
65
|
+
end
|
|
66
|
+
end
|
data/spec/docs_sync_spec.rb
CHANGED
|
@@ -37,14 +37,14 @@ RSpec.describe "documentation surface sync" do
|
|
|
37
37
|
expect(phantom).to be_empty, "documented tools that do not exist: #{phantom.join(', ')}"
|
|
38
38
|
end
|
|
39
39
|
|
|
40
|
-
# Tripwire: tracked files must not contain tokens listed in .
|
|
40
|
+
# Tripwire: tracked files must not contain tokens listed in .local/doc-tokens,
|
|
41
41
|
# an untracked, machine-local file (one substring per line; # starts a comment).
|
|
42
42
|
# The file exists only on machines that maintain such a list; everywhere else
|
|
43
43
|
# (CI, other contributors) this example skips — loudly, so a silently dead
|
|
44
44
|
# check cannot be mistaken for a passing one.
|
|
45
45
|
it "keeps machine-local private tokens out of tracked files" do
|
|
46
|
-
token_file = File.join(repo_root, ".
|
|
47
|
-
skip "SKIPPED: no .
|
|
46
|
+
token_file = File.join(repo_root, ".local", "doc-tokens")
|
|
47
|
+
skip "SKIPPED: no .local/doc-tokens on this machine — tripwire not checked" unless File.exist?(token_file)
|
|
48
48
|
|
|
49
49
|
tokens = File.readlines(token_file, encoding: "UTF-8")
|
|
50
50
|
.map(&:strip).reject { |t| t.empty? || t.start_with?("#") }
|
|
@@ -65,6 +65,33 @@ RSpec.describe Wp2txt::LanglinksImporter do
|
|
|
65
65
|
rows
|
|
66
66
|
end
|
|
67
67
|
|
|
68
|
+
# Current dumps put the INSERT header on its own line and one tuple per line
|
|
69
|
+
MULTILINE_SQL = <<~MSQL
|
|
70
|
+
INSERT INTO `langlinks` VALUES
|
|
71
|
+
(1,'en','Film A'),
|
|
72
|
+
(1,'ja','映画A'),
|
|
73
|
+
(2,'en','It\\'s a Film; Really');
|
|
74
|
+
INSERT INTO `pagelinks` VALUES
|
|
75
|
+
(9,'Ignored',0);
|
|
76
|
+
INSERT INTO `langlinks` VALUES
|
|
77
|
+
(3,'fr','Film B');
|
|
78
|
+
MSQL
|
|
79
|
+
|
|
80
|
+
describe "dumps with one tuple per line" do
|
|
81
|
+
it "imports every tuple of every langlinks statement, and nothing else" do
|
|
82
|
+
path = write_langlinks("testwiki-20260101-langlinks.sql.gz", MULTILINE_SQL, gzip: true)
|
|
83
|
+
expect(import(path)[:row_count]).to eq(4)
|
|
84
|
+
expect(langlinks_rows).to contain_exactly(
|
|
85
|
+
[1, "en", "Film A"], [1, "ja", "映画A"], [2, "en", "It's a Film; Really"], [3, "fr", "Film B"]
|
|
86
|
+
)
|
|
87
|
+
end
|
|
88
|
+
|
|
89
|
+
it "refuses to report success when no rows could be read" do
|
|
90
|
+
path = write_langlinks("testwiki-20260101-langlinks.sql", "-- nothing here\nUNLOCK TABLES;\n")
|
|
91
|
+
expect { import(path) }.to raise_error(Wp2txt::Error, /no langlinks rows found/)
|
|
92
|
+
end
|
|
93
|
+
end
|
|
94
|
+
|
|
68
95
|
describe "parsing and normalization" do
|
|
69
96
|
it "imports tuples with escapes, commas, parens, and multiple INSERT statements" do
|
|
70
97
|
path = write_langlinks("testwiki-20260101-langlinks.sql")
|