wp2txt 2.3.3 → 2.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/.dockerignore +0 -4
- data/.gitignore +3 -5
- data/CHANGELOG.md +18 -0
- data/DEVELOPMENT.md +1 -1
- data/DEVELOPMENT_ja.md +1 -1
- data/README.md +44 -2
- data/README_ja.md +35 -2
- data/Rakefile +10 -21
- data/bin/wp2txt +79 -17
- data/bin/wp2txt-mcp +1 -1
- data/docs/INDEXES.md +61 -1
- data/lib/wp2txt/article.rb +1 -1
- data/lib/wp2txt/cli.rb +42 -0
- data/lib/wp2txt/constants.rb +24 -0
- data/lib/wp2txt/corpus.rb +11 -2
- data/lib/wp2txt/data/template_aliases.json +1 -1
- data/lib/wp2txt/extractor.rb +10 -1
- data/lib/wp2txt/formatter.rb +20 -0
- data/lib/wp2txt/index_commands.rb +89 -1
- data/lib/wp2txt/langlinks_importer.rb +19 -35
- data/lib/wp2txt/lead_terms.rb +228 -0
- data/lib/wp2txt/link_counter.rb +171 -0
- data/lib/wp2txt/metadata_index.rb +79 -3
- data/lib/wp2txt/multistream.rb +52 -2
- data/lib/wp2txt/output_writer.rb +8 -0
- data/lib/wp2txt/page_props_importer.rb +170 -0
- data/lib/wp2txt/sql_dump_reader.rb +57 -0
- data/lib/wp2txt/stream_processor.rb +42 -15
- data/lib/wp2txt/template_expander.rb +19 -0
- data/lib/wp2txt/utils.rb +5 -3
- data/lib/wp2txt/version.rb +1 -1
- data/lib/wp2txt/wikitext_regions.rb +66 -0
- data/lib/wp2txt.rb +7 -5
- data/spec/docs_sync_spec.rb +3 -3
- data/spec/langlinks_importer_spec.rb +27 -0
- data/spec/lead_terms_edge_cases_spec.rb +194 -0
- data/spec/lead_terms_links_qids_spec.rb +204 -0
- data/spec/output_integrity_spec.rb +146 -0
- data/spec/p1_correctness_spec.rb +32 -14
- data/spec/page_properties_spec.rb +161 -0
- data/spec/region_semantics_spec.rb +71 -0
- data/spec/template_passthrough_spec.rb +43 -0
- data/spec/titles_output_path_spec.rb +12 -0
- metadata +18 -1
|
@@ -19,8 +19,12 @@ module Wp2txt
|
|
|
19
19
|
|
|
20
20
|
attr_reader :buffer_size, :pages_processed, :bytes_read, :redirects_skipped
|
|
21
21
|
|
|
22
|
-
|
|
22
|
+
# keep_raw_text: also hand back each article's wikitext as stored in the
|
|
23
|
+
# dump (before comment removal), as :raw_text among the with_ids values
|
|
24
|
+
def initialize(input_path, bz2_gem: false, adaptive_buffer: true, validate_bz2: true, skip_redirects: true,
|
|
25
|
+
keep_raw_text: false)
|
|
23
26
|
@input_path = input_path
|
|
27
|
+
@keep_raw_text = keep_raw_text
|
|
24
28
|
@bz2_gem = bz2_gem
|
|
25
29
|
@buffer = +""
|
|
26
30
|
@pending_bytes = +"".b
|
|
@@ -64,20 +68,28 @@ module Wp2txt
|
|
|
64
68
|
|
|
65
69
|
# Iterate over each page in the input
|
|
66
70
|
# Yields [title, text] for each page
|
|
67
|
-
|
|
68
|
-
|
|
71
|
+
# Yields title and text of each article. With with_ids: true, also yields
|
|
72
|
+
# { page_id:, revision_id: } taken from the dump.
|
|
73
|
+
def each_page(with_ids: false, &block)
|
|
74
|
+
return enum_for(:each_page, with_ids: with_ids) unless block
|
|
75
|
+
|
|
76
|
+
emit = if with_ids
|
|
77
|
+
->(title, text, ids) { block.call(title, text, ids) }
|
|
78
|
+
else
|
|
79
|
+
->(title, text, _ids) { block.call(title, text) }
|
|
80
|
+
end
|
|
69
81
|
|
|
70
82
|
if File.directory?(@input_path)
|
|
71
83
|
# Process XML files in directory
|
|
72
84
|
Dir.glob(File.join(@input_path, "*.xml")).sort.each do |xml_file|
|
|
73
|
-
process_xml_file(xml_file) { |title, text|
|
|
85
|
+
process_xml_file(xml_file) { |title, text, ids| emit.call(title, text, ids) }
|
|
74
86
|
end
|
|
75
87
|
elsif @input_path.end_with?(".bz2")
|
|
76
88
|
# Process bz2 compressed file with streaming
|
|
77
|
-
process_bz2_stream { |title, text|
|
|
89
|
+
process_bz2_stream { |title, text, ids| emit.call(title, text, ids) }
|
|
78
90
|
elsif @input_path.end_with?(".xml")
|
|
79
91
|
# Process single XML file
|
|
80
|
-
process_xml_file(@input_path) { |title, text|
|
|
92
|
+
process_xml_file(@input_path) { |title, text, ids| emit.call(title, text, ids) }
|
|
81
93
|
else
|
|
82
94
|
raise ArgumentError, "Unsupported input format: #{@input_path}"
|
|
83
95
|
end
|
|
@@ -162,20 +174,32 @@ module Wp2txt
|
|
|
162
174
|
def fill_buffer
|
|
163
175
|
chunk = @file_pointer.read(@buffer_size)
|
|
164
176
|
unless chunk
|
|
165
|
-
|
|
166
|
-
|
|
177
|
+
unless @pending_bytes.to_s.empty?
|
|
178
|
+
raise Wp2txt::EncodingError,
|
|
179
|
+
"input ends in the middle of a UTF-8 character (byte #{@bytes_read}): #{@input_path}"
|
|
180
|
+
end
|
|
167
181
|
return false
|
|
168
182
|
end
|
|
169
183
|
|
|
170
|
-
|
|
171
|
-
|
|
172
|
-
|
|
173
|
-
#
|
|
184
|
+
carried = @pending_bytes.to_s.b
|
|
185
|
+
start = @bytes_read - carried.bytesize # stream position of the first carried byte
|
|
186
|
+
bytes = carried + chunk.b
|
|
187
|
+
# A read can end partway through a multi-byte character; carry those
|
|
188
|
+
# bytes into the next read rather than judging half a character.
|
|
174
189
|
tail = bytes[/[\xC2-\xF4][\x80-\xBF]{0,2}\z/n]
|
|
175
190
|
width = tail && (tail.getbyte(0) < 0xE0 ? 2 : tail.getbyte(0) < 0xF0 ? 3 : 4)
|
|
176
191
|
@pending_bytes = tail && tail.bytesize < width ? tail : +"".b
|
|
177
|
-
bytes = bytes.byteslice(0, bytes.bytesize - @pending_bytes.bytesize)
|
|
178
|
-
|
|
192
|
+
bytes = bytes.byteslice(0, bytes.bytesize - @pending_bytes.bytesize).force_encoding(Encoding::UTF_8)
|
|
193
|
+
# Anything still invalid is corrupt input. Dropping it would change the
|
|
194
|
+
# text without a trace, so stop and say where it is.
|
|
195
|
+
unless bytes.valid_encoding?
|
|
196
|
+
bad = bytes.each_char.find_index { |c| !c.valid_encoding? }.to_i
|
|
197
|
+
offset = start + bytes[0, bad].bytesize
|
|
198
|
+
raise Wp2txt::EncodingError,
|
|
199
|
+
"invalid UTF-8 near decompressed byte #{offset}: #{@input_path}"
|
|
200
|
+
end
|
|
201
|
+
@bytes_read += chunk.bytesize
|
|
202
|
+
@buffer << bytes
|
|
179
203
|
|
|
180
204
|
# Adaptive buffer adjustment: if memory is low, reduce buffer size
|
|
181
205
|
if @adaptive_buffer && MemoryMonitor.memory_low?
|
|
@@ -237,6 +261,7 @@ module Wp2txt
|
|
|
237
261
|
return nil unless Wp2txt.namespace_id(namespace).zero?
|
|
238
262
|
|
|
239
263
|
text = text_node.content
|
|
264
|
+
raw_text = text if @keep_raw_text
|
|
240
265
|
|
|
241
266
|
# Early redirect detection and skip (before expensive processing)
|
|
242
267
|
# Redirects start with # or # followed by redirect keyword and [[target]]
|
|
@@ -252,7 +277,9 @@ module Wp2txt
|
|
|
252
277
|
end
|
|
253
278
|
|
|
254
279
|
@pages_processed += 1
|
|
255
|
-
|
|
280
|
+
meta = Wp2txt.page_ids(page_xml)
|
|
281
|
+
meta[:raw_text] = raw_text if @keep_raw_text
|
|
282
|
+
[title, text, meta]
|
|
256
283
|
rescue Nokogiri::XML::SyntaxError
|
|
257
284
|
# Skip malformed XML
|
|
258
285
|
nil
|
|
@@ -171,6 +171,19 @@ module Wp2txt
|
|
|
171
171
|
nil
|
|
172
172
|
end
|
|
173
173
|
|
|
174
|
+
# Templates whose rendering the text-cleaning stage defines (see
|
|
175
|
+
# Wp2txt#correct_inline_template), keyed by normalized name. Limited to
|
|
176
|
+
# kinds whose output does not depend on marker settings.
|
|
177
|
+
def rendered_by_cleaner?(template_name)
|
|
178
|
+
@rendered_by_cleaner ||= [Wp2txt::RUBY_TEXT_TEMPLATES, Wp2txt::INTERWIKI_LINK_TEMPLATES]
|
|
179
|
+
.flatten.to_set { |name| name.to_s.tr("_", " ").strip.downcase }
|
|
180
|
+
@rendered_by_cleaner.include?(template_name.tr("_", " "))
|
|
181
|
+
end
|
|
182
|
+
|
|
183
|
+
def text_renderer
|
|
184
|
+
@text_renderer ||= Object.new.extend(Wp2txt)
|
|
185
|
+
end
|
|
186
|
+
|
|
174
187
|
def expand_single_template(content)
|
|
175
188
|
parts = split_template_parts(content)
|
|
176
189
|
return "" if parts.empty?
|
|
@@ -263,6 +276,12 @@ module Wp2txt
|
|
|
263
276
|
# Handle lang-xx templates (e.g., lang-fr, lang-de, lang-ja)
|
|
264
277
|
if template_name.start_with?("lang-")
|
|
265
278
|
expand_lang_xx(template_name, params)
|
|
279
|
+
elsif rendered_by_cleaner?(template_name)
|
|
280
|
+
# These carry words the article needs (a reading, a link's display
|
|
281
|
+
# text). Deleting them dropped those words; leaving the raw template
|
|
282
|
+
# for later confused links that contain it (an image caption with a
|
|
283
|
+
# "|" inside). Render them now, with the cleaner's own rules.
|
|
284
|
+
text_renderer.correct_inline_template("{{#{expand(content)}}}")
|
|
266
285
|
else
|
|
267
286
|
@preserve_unknown ? "{{#{content}}}" : ""
|
|
268
287
|
end
|
data/lib/wp2txt/utils.rb
CHANGED
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
# frozen_string_literal: true
|
|
2
2
|
|
|
3
3
|
require "strscan"
|
|
4
|
+
require_relative "wikitext_regions"
|
|
4
5
|
require_relative "constants"
|
|
5
6
|
require_relative "regex"
|
|
6
7
|
require_relative "text_processing"
|
|
@@ -574,8 +575,9 @@ module Wp2txt
|
|
|
574
575
|
# Helper to check if template name matches any in a list (case-insensitive)
|
|
575
576
|
def template_matches?(name, template_list)
|
|
576
577
|
return false if template_list.nil? || template_list.empty?
|
|
577
|
-
|
|
578
|
-
|
|
578
|
+
# MediaWiki treats "_" and " " in template names as the same character
|
|
579
|
+
normalized_name = name.to_s.tr("_", " ").strip.downcase
|
|
580
|
+
template_list.any? { |t| t.tr("_", " ").downcase == normalized_name }
|
|
579
581
|
end
|
|
580
582
|
|
|
581
583
|
def correct_inline_template(str, enabled_markers = [], extract_citations = false)
|
|
@@ -583,7 +585,7 @@ module Wp2txt
|
|
|
583
585
|
return str unless str.include?("{{")
|
|
584
586
|
|
|
585
587
|
process_nested_single_pass(str, "{{", "}}") do |contents|
|
|
586
|
-
parts =
|
|
588
|
+
parts = WikitextRegions.split_pipes(contents)
|
|
587
589
|
template_name = (parts[0] || "").strip
|
|
588
590
|
template_name_lower = template_name.downcase
|
|
589
591
|
|
data/lib/wp2txt/version.rb
CHANGED
|
@@ -0,0 +1,66 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Wp2txt
|
|
4
|
+
# Literal regions suppress wikitext parsing. Galleries and timelines can
|
|
5
|
+
# contain links, but their contents are not the article's lead prose.
|
|
6
|
+
module WikitextRegions
|
|
7
|
+
LITERAL_TAGS = %w[nowiki pre math chem ce score syntaxhighlight source graph mapframe templatedata].freeze
|
|
8
|
+
NON_PROSE_TAGS = %w[gallery timeline].freeze
|
|
9
|
+
COMMENT = /<!--.*?(?:-->|\z)/m
|
|
10
|
+
|
|
11
|
+
def self.region_pattern(tags)
|
|
12
|
+
/<!--.*?(?:-->|\z)|<(?<tag>#{tags.join('|')})(?=[\s\/>])(?:"[^"]*"|'[^']*'|[^'">])*?(?:\/>|>.*?(?:<\/\k<tag>\s*>|\z))/mi
|
|
13
|
+
end
|
|
14
|
+
|
|
15
|
+
LITERAL_REGION = region_pattern(LITERAL_TAGS)
|
|
16
|
+
LEAD_REGION = region_pattern(LITERAL_TAGS + NON_PROSE_TAGS)
|
|
17
|
+
LITERAL_REGION_AT = /\G(?:#{LITERAL_REGION})/
|
|
18
|
+
LEAD_REGION_AT = /\G(?:#{LEAD_REGION})/
|
|
19
|
+
|
|
20
|
+
module_function
|
|
21
|
+
|
|
22
|
+
def end_at(text, offset, lead: false)
|
|
23
|
+
match = (lead ? LEAD_REGION_AT : LITERAL_REGION_AT).match(text, offset)
|
|
24
|
+
match.end(0) if match
|
|
25
|
+
end
|
|
26
|
+
|
|
27
|
+
def remove_literal(text)
|
|
28
|
+
text.gsub(LITERAL_REGION, "")
|
|
29
|
+
end
|
|
30
|
+
|
|
31
|
+
# One space per codepoint, including internal newlines: an excluded
|
|
32
|
+
# region must not introduce a paragraph or heading boundary.
|
|
33
|
+
def mask_lead(text)
|
|
34
|
+
text.gsub(LEAD_REGION) { |region| " " * region.length }
|
|
35
|
+
end
|
|
36
|
+
|
|
37
|
+
def split_pipes(text)
|
|
38
|
+
parts = [+""]
|
|
39
|
+
stack = []
|
|
40
|
+
i = 0
|
|
41
|
+
while i < text.length
|
|
42
|
+
two = text[i, 2]
|
|
43
|
+
if text[i] == "<" && (stop = end_at(text, i))
|
|
44
|
+
parts.last << text[i...stop]
|
|
45
|
+
i = stop
|
|
46
|
+
elsif ["{{", "[["].include?(two)
|
|
47
|
+
stack << (two == "{{" ? "}}" : "]]")
|
|
48
|
+
parts.last << two
|
|
49
|
+
i += 2
|
|
50
|
+
elsif !stack.empty? && two == stack.last
|
|
51
|
+
stack.pop
|
|
52
|
+
parts.last << two
|
|
53
|
+
i += 2
|
|
54
|
+
else
|
|
55
|
+
if text[i] == "|" && stack.empty?
|
|
56
|
+
parts << +""
|
|
57
|
+
else
|
|
58
|
+
parts.last << text[i]
|
|
59
|
+
end
|
|
60
|
+
i += 1
|
|
61
|
+
end
|
|
62
|
+
end
|
|
63
|
+
parts
|
|
64
|
+
end
|
|
65
|
+
end
|
|
66
|
+
end
|
data/lib/wp2txt.rb
CHANGED
|
@@ -269,12 +269,14 @@ module Wp2txt
|
|
|
269
269
|
end
|
|
270
270
|
page << line if inside_page
|
|
271
271
|
end
|
|
272
|
-
if page.empty?
|
|
273
|
-
|
|
274
|
-
|
|
275
|
-
|
|
272
|
+
return false if page.empty?
|
|
273
|
+
|
|
274
|
+
page.force_encoding("utf-8")
|
|
275
|
+
# Corrupt input must not be cleaned into different text without a trace
|
|
276
|
+
unless page.valid_encoding?
|
|
277
|
+
title = page[%r{<title>([^<]*)</title>}, 1]&.scrub("?")
|
|
278
|
+
raise Wp2txt::EncodingError, "invalid UTF-8 in page #{title.inspect} of #{@input_file}"
|
|
276
279
|
end
|
|
277
|
-
rescue ::Encoding::InvalidByteSequenceError, ::Encoding::UndefinedConversionError
|
|
278
280
|
page
|
|
279
281
|
end
|
|
280
282
|
|
data/spec/docs_sync_spec.rb
CHANGED
|
@@ -37,14 +37,14 @@ RSpec.describe "documentation surface sync" do
|
|
|
37
37
|
expect(phantom).to be_empty, "documented tools that do not exist: #{phantom.join(', ')}"
|
|
38
38
|
end
|
|
39
39
|
|
|
40
|
-
# Tripwire: tracked files must not contain tokens listed in .
|
|
40
|
+
# Tripwire: tracked files must not contain tokens listed in .local/doc-tokens,
|
|
41
41
|
# an untracked, machine-local file (one substring per line; # starts a comment).
|
|
42
42
|
# The file exists only on machines that maintain such a list; everywhere else
|
|
43
43
|
# (CI, other contributors) this example skips — loudly, so a silently dead
|
|
44
44
|
# check cannot be mistaken for a passing one.
|
|
45
45
|
it "keeps machine-local private tokens out of tracked files" do
|
|
46
|
-
token_file = File.join(repo_root, ".
|
|
47
|
-
skip "SKIPPED: no .
|
|
46
|
+
token_file = File.join(repo_root, ".local", "doc-tokens")
|
|
47
|
+
skip "SKIPPED: no .local/doc-tokens on this machine — tripwire not checked" unless File.exist?(token_file)
|
|
48
48
|
|
|
49
49
|
tokens = File.readlines(token_file, encoding: "UTF-8")
|
|
50
50
|
.map(&:strip).reject { |t| t.empty? || t.start_with?("#") }
|
|
@@ -65,6 +65,33 @@ RSpec.describe Wp2txt::LanglinksImporter do
|
|
|
65
65
|
rows
|
|
66
66
|
end
|
|
67
67
|
|
|
68
|
+
# Current dumps put the INSERT header on its own line and one tuple per line
|
|
69
|
+
MULTILINE_SQL = <<~MSQL
|
|
70
|
+
INSERT INTO `langlinks` VALUES
|
|
71
|
+
(1,'en','Film A'),
|
|
72
|
+
(1,'ja','映画A'),
|
|
73
|
+
(2,'en','It\\'s a Film; Really');
|
|
74
|
+
INSERT INTO `pagelinks` VALUES
|
|
75
|
+
(9,'Ignored',0);
|
|
76
|
+
INSERT INTO `langlinks` VALUES
|
|
77
|
+
(3,'fr','Film B');
|
|
78
|
+
MSQL
|
|
79
|
+
|
|
80
|
+
describe "dumps with one tuple per line" do
|
|
81
|
+
it "imports every tuple of every langlinks statement, and nothing else" do
|
|
82
|
+
path = write_langlinks("testwiki-20260101-langlinks.sql.gz", MULTILINE_SQL, gzip: true)
|
|
83
|
+
expect(import(path)[:row_count]).to eq(4)
|
|
84
|
+
expect(langlinks_rows).to contain_exactly(
|
|
85
|
+
[1, "en", "Film A"], [1, "ja", "映画A"], [2, "en", "It's a Film; Really"], [3, "fr", "Film B"]
|
|
86
|
+
)
|
|
87
|
+
end
|
|
88
|
+
|
|
89
|
+
it "refuses to report success when no rows could be read" do
|
|
90
|
+
path = write_langlinks("testwiki-20260101-langlinks.sql", "-- nothing here\nUNLOCK TABLES;\n")
|
|
91
|
+
expect { import(path) }.to raise_error(Wp2txt::Error, /no langlinks rows found/)
|
|
92
|
+
end
|
|
93
|
+
end
|
|
94
|
+
|
|
68
95
|
describe "parsing and normalization" do
|
|
69
96
|
it "imports tuples with escapes, commas, parens, and multiple INSERT statements" do
|
|
70
97
|
path = write_langlinks("testwiki-20260101-langlinks.sql")
|
|
@@ -0,0 +1,194 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require 'spec_helper'
|
|
4
|
+
require 'wp2txt'
|
|
5
|
+
require 'wp2txt/lead_terms'
|
|
6
|
+
require 'wp2txt/link_counter'
|
|
7
|
+
require 'wp2txt/metadata_index'
|
|
8
|
+
require 'open3'
|
|
9
|
+
require 'tmpdir'
|
|
10
|
+
require_relative 'support/multistream_fixture'
|
|
11
|
+
|
|
12
|
+
RSpec.describe 'lead terms and link counting on unusual wikitext' do
|
|
13
|
+
include MultistreamFixture
|
|
14
|
+
|
|
15
|
+
let(:cleaner) { Object.new.extend(Wp2txt) }
|
|
16
|
+
let(:render) { ->(s) { cleaner.format_wiki(s, expand_templates: true, markers: [:all]) } }
|
|
17
|
+
def terms(text)
|
|
18
|
+
Wp2txt::LeadTerms.extract(text, render: render)
|
|
19
|
+
end
|
|
20
|
+
|
|
21
|
+
it '#1 preserves link display and reading with expansion enabled or disabled' do
|
|
22
|
+
{
|
|
23
|
+
'{{読み仮名|[[東京|東京都]]|とうきょう}}' => '東京都(とうきょう)',
|
|
24
|
+
'{{仮リンク|[[東京|東京都]]|en|Tokyo}}' => '東京都',
|
|
25
|
+
'{{読み仮名|{{lang|ja|[[東京|東京都]]}}|とうきょう}}' => '東京都(とうきょう)',
|
|
26
|
+
'{{読み仮名|東京|}}' => '東京'
|
|
27
|
+
}.each do |input, expected|
|
|
28
|
+
[true, false].each do |expand|
|
|
29
|
+
expect(cleaner.format_wiki(input, expand_templates: expand)).to eq(expected)
|
|
30
|
+
end
|
|
31
|
+
end
|
|
32
|
+
end
|
|
33
|
+
|
|
34
|
+
def build_case_dump(dir, rule)
|
|
35
|
+
alias_title = rule == 'first-letter' ? 'Alias' : 'alias'
|
|
36
|
+
pages = [[1, 'apple', ''], [2, 'Apple', ''], [3, alias_title, '#REDIRECT [[apple]]'],
|
|
37
|
+
[4, 'Direct', '[[apple]]'], [5, 'Via', '[[alias]]'], [6, '東京', '']]
|
|
38
|
+
xml = "<mediawiki><siteinfo><case>#{rule}</case></siteinfo>" + pages.map do |id, title, text|
|
|
39
|
+
page_xml(id: id, ns: 0, title: title, text: text)
|
|
40
|
+
end.join + '</mediawiki>'
|
|
41
|
+
dump = File.join(dir, 'testwiki-20260101-multistream.xml.bz2')
|
|
42
|
+
File.binwrite(dump, bzip2(xml))
|
|
43
|
+
db = File.join(dir, 'meta.sqlite3')
|
|
44
|
+
Wp2txt::MetadataIndexBuilder.new(dump, [0], db_path: db, num_processes: 0).build.close
|
|
45
|
+
[dump, db]
|
|
46
|
+
end
|
|
47
|
+
|
|
48
|
+
it '#2 stores siteinfo case and preserves direct and redirected case-sensitive targets' do
|
|
49
|
+
Dir.mktmpdir do |dir|
|
|
50
|
+
dump, path = build_case_dump(dir, 'case-sensitive')
|
|
51
|
+
db = SQLite3::Database.new(path)
|
|
52
|
+
expect(db.get_first_value("SELECT value FROM metadata WHERE key = 'case_rule'")).to eq('case-sensitive')
|
|
53
|
+
db.close
|
|
54
|
+
Wp2txt::LinkCounter.new(dump, [0], db_path: path, num_processes: 0).count!
|
|
55
|
+
db = SQLite3::Database.new(path)
|
|
56
|
+
expect(db.execute('SELECT page_id,inlinks,via_redirects FROM page_inlinks WHERE page_id IN (1,2) ORDER BY page_id'))
|
|
57
|
+
.to eq([[1, 2, 1], [2, 0, 0]])
|
|
58
|
+
db.close
|
|
59
|
+
expect(Wp2txt::MetadataIndex.normalize_title('apple')).to eq('Apple')
|
|
60
|
+
end
|
|
61
|
+
end
|
|
62
|
+
|
|
63
|
+
it '#2 uses first-letter for explicit siteinfo and legacy metadata without the rule' do
|
|
64
|
+
Dir.mktmpdir do |dir|
|
|
65
|
+
dump, path = build_case_dump(dir, 'first-letter')
|
|
66
|
+
[false, true].each do |legacy|
|
|
67
|
+
db = SQLite3::Database.new(path)
|
|
68
|
+
if legacy
|
|
69
|
+
db.execute("DELETE FROM metadata WHERE key = 'case_rule'")
|
|
70
|
+
else
|
|
71
|
+
expect(db.get_first_value("SELECT value FROM metadata WHERE key = 'case_rule'")).to eq('first-letter')
|
|
72
|
+
end
|
|
73
|
+
db.close
|
|
74
|
+
Wp2txt::LinkCounter.new(dump, [0], db_path: path, num_processes: 0).count!
|
|
75
|
+
db = SQLite3::Database.new(path)
|
|
76
|
+
expect(db.get_first_value('SELECT inlinks FROM page_inlinks WHERE page_id=1')).to eq(0)
|
|
77
|
+
expect(db.execute('SELECT inlinks,via_redirects FROM page_inlinks WHERE page_id=2')).to eq([[2, 1]])
|
|
78
|
+
db.close
|
|
79
|
+
end
|
|
80
|
+
end
|
|
81
|
+
end
|
|
82
|
+
|
|
83
|
+
it '#2 reads a separate siteinfo stream before the first indexed page' do
|
|
84
|
+
Dir.mktmpdir do |dir|
|
|
85
|
+
header = bzip2('<mediawiki><siteinfo><case>case-sensitive</case></siteinfo>')
|
|
86
|
+
dump = File.join(dir, 'dump.xml.bz2')
|
|
87
|
+
File.binwrite(dump, header + bzip2(page_xml(id: 1, ns: 0, title: 'apple', text: '') + '</mediawiki>'))
|
|
88
|
+
path = File.join(dir, 'meta.sqlite3')
|
|
89
|
+
Wp2txt::MetadataIndexBuilder.new(dump, [header.bytesize], db_path: path, num_processes: 0).build.close
|
|
90
|
+
db = SQLite3::Database.new(path)
|
|
91
|
+
expect(db.get_first_value("SELECT value FROM metadata WHERE key = 'case_rule'")).to eq('case-sensitive')
|
|
92
|
+
db.close
|
|
93
|
+
end
|
|
94
|
+
end
|
|
95
|
+
|
|
96
|
+
%w[nowiki pre math syntaxhighlight source gallery score timeline chem ce graph mapframe templatedata].each do |tag|
|
|
97
|
+
it "#3 skips #{tag} contents and self-closing forms while preserving source offsets" do
|
|
98
|
+
text = "<#{tag.upcase} data-x='>'>'''偽'''(にせ){{ruby|偽|にせ}}</#{tag.upcase}>\n\n" \
|
|
99
|
+
"<#{tag} />'''東京'''(とうきょう)。"
|
|
100
|
+
result = terms(text)
|
|
101
|
+
expect(result.map { |t| t['text'] }).to eq(['東京'])
|
|
102
|
+
s, e = result.first['span']['bold']
|
|
103
|
+
expect(text[s...e]).to eq("'''東京'''")
|
|
104
|
+
expect(terms("<#{tag}>'''偽'''" )).to eq([])
|
|
105
|
+
end
|
|
106
|
+
end
|
|
107
|
+
|
|
108
|
+
it '#4 ignores delimiters in comments and excluded tags inside parentheses' do
|
|
109
|
+
text = "'''東京'''(とうきょう<!-- ) , -->、<nowiki>),</nowiki>Tokyo)は都市。"
|
|
110
|
+
result = terms(text).first
|
|
111
|
+
s, e = result['span']['paren']
|
|
112
|
+
expect(text[s...e]).to eq('(とうきょう<!-- ) , -->、<nowiki>),</nowiki>Tokyo)')
|
|
113
|
+
expect(result['notes'].size).to eq(2)
|
|
114
|
+
expect(result['notes'].first).to eq('とうきょう')
|
|
115
|
+
expect(result['notes_text']).not_to include('<!--')
|
|
116
|
+
end
|
|
117
|
+
|
|
118
|
+
it 'retains the original text of ruby arguments containing excluded regions' do
|
|
119
|
+
expect(terms('{{ruby|<nowiki>東京</nowiki>|とうきょう}}').first.slice('text', 'reading'))
|
|
120
|
+
.to eq('text' => '東京', 'reading' => 'とうきょう')
|
|
121
|
+
end
|
|
122
|
+
|
|
123
|
+
it '#5 detects only headings outside skipped regions including trailing comments' do
|
|
124
|
+
["<!--\n== 偽 ==\n-->", "<nowiki>\n== 偽 ==\n</nowiki>", "{{box|\n== 偽 ==\n}}", "{|\n== 偽 ==\n|}"].each do |prefix|
|
|
125
|
+
expect(terms("#{prefix}\n'''東京'''(とうきょう)").map { |t| t['text'] }).to eq(['東京'])
|
|
126
|
+
end
|
|
127
|
+
expect(terms("導入文。\n\n== 歴史 == <!-- comment -->\n'''後の語'''(あと)。")).to eq([])
|
|
128
|
+
end
|
|
129
|
+
|
|
130
|
+
it '#6 allows single newlines before and inside notes but stops at a blank line' do
|
|
131
|
+
expect(terms("'''東京'''\n(とうきょう)は都市。").first['notes']).to eq(['とうきょう'])
|
|
132
|
+
text = "'''東京'''(とうきょう、\nTokyo)は都市。"
|
|
133
|
+
term = terms(text).first
|
|
134
|
+
expect(term['notes']).to eq(['とうきょう', 'Tokyo'])
|
|
135
|
+
s, e = term['span']['paren']
|
|
136
|
+
expect(text[s...e]).to eq("(とうきょう、\nTokyo)")
|
|
137
|
+
["'''東京'''\n\n(とうきょう)", "'''東京'''(とうきょう、\n \nTokyo)"].each do |input|
|
|
138
|
+
expect(terms(input).first['notes']).to eq([])
|
|
139
|
+
end
|
|
140
|
+
end
|
|
141
|
+
|
|
142
|
+
it '#7 stops at empty lines containing whitespace' do
|
|
143
|
+
["\n \n", "\n\t\n", "\r\n \r\n"].each do |gap|
|
|
144
|
+
expect(terms("'''東京'''(とうきょう)。#{gap}'''別段落'''(べつだんらく)。").map { |t| t['text'] }).to eq(['東京'])
|
|
145
|
+
end
|
|
146
|
+
end
|
|
147
|
+
|
|
148
|
+
def counts(text)
|
|
149
|
+
counter = Wp2txt::LinkCounter.new('', [], db_path: '')
|
|
150
|
+
counter.instance_variable_set(:@redirects, {})
|
|
151
|
+
direct, via = Hash.new(0), Hash.new(0)
|
|
152
|
+
counter.send(:count_page, page_xml(id: 1, ns: 0, title: 'Source', text: text), direct, via)
|
|
153
|
+
direct
|
|
154
|
+
end
|
|
155
|
+
|
|
156
|
+
it '#8 decodes title entities before sections and normalizes whitespace before colon' do
|
|
157
|
+
expect(counts('[[M&A]] [[Café]] [[ :東京#節|表示]] [[Café#section]] [[#節]] [[]]'))
|
|
158
|
+
.to eq('M&A' => 1, 'Café' => 1, '東京' => 1)
|
|
159
|
+
expect(counts('[[東京#節]] [[東京|表示]]')).to eq('東京' => 1)
|
|
160
|
+
end
|
|
161
|
+
|
|
162
|
+
it '#9 excludes every non-wikitext tag and comments from incoming links' do
|
|
163
|
+
%w[nowiki pre math syntaxhighlight source score chem ce graph mapframe templatedata].each do |tag|
|
|
164
|
+
expect(counts("<#{tag}>[[東京]]</#{tag}>[[大阪]]<#{tag}/><!-- [[京都]] -->")).to eq('大阪' => 1)
|
|
165
|
+
end
|
|
166
|
+
end
|
|
167
|
+
|
|
168
|
+
it 'handles empty text and binary invalid bytes in the shared region helper' do
|
|
169
|
+
expect(terms('')).to eq([])
|
|
170
|
+
expect(counts('')).to eq({})
|
|
171
|
+
expect(Wp2txt::WikitextRegions.remove_literal("<nowiki>\xFF</nowiki>\xFE".b)).to eq("\xFE".b)
|
|
172
|
+
end
|
|
173
|
+
|
|
174
|
+
def cli(*options)
|
|
175
|
+
Dir.mktmpdir do |dir|
|
|
176
|
+
input = File.join(dir, 'x.xml.bz2')
|
|
177
|
+
File.binwrite(input, bzip2("<mediawiki>#{page_xml(id: 1, ns: 0, title: '東京', text: "'''東京'''。")}</mediawiki>"))
|
|
178
|
+
Open3.capture3(RbConfig.ruby, '-I', File.expand_path('../lib', __dir__),
|
|
179
|
+
File.expand_path('../bin/wp2txt', __dir__), '-i', input, '-o', dir, *options)
|
|
180
|
+
end
|
|
181
|
+
end
|
|
182
|
+
|
|
183
|
+
it '#10 rejects lead terms with ractor' do
|
|
184
|
+
_, err, status = cli('--format', 'json', '--lead-terms', '--ractor', '--no-turbo')
|
|
185
|
+
expect(status.success?).to be(false)
|
|
186
|
+
expect(err).to include('--lead-terms cannot be combined with --ractor')
|
|
187
|
+
end
|
|
188
|
+
|
|
189
|
+
it '#10 warns about missing source identifiers in ractor JSON without rejecting it' do
|
|
190
|
+
_, err, status = cli('--format', 'json', '--ractor', '--no-turbo', '-n', '1')
|
|
191
|
+
expect(status.success?).to be(true), err
|
|
192
|
+
expect(err).to include('does not include page IDs or revision IDs')
|
|
193
|
+
end
|
|
194
|
+
end
|