wp2txt 2.3.4 → 2.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/.dockerignore +0 -4
- data/.gitignore +3 -5
- data/CHANGELOG.md +9 -0
- data/DEVELOPMENT.md +1 -1
- data/DEVELOPMENT_ja.md +1 -1
- data/README.md +39 -0
- data/README_ja.md +30 -0
- data/Rakefile +10 -21
- data/bin/wp2txt +61 -9
- data/bin/wp2txt-mcp +1 -1
- data/docs/INDEXES.md +61 -1
- data/lib/wp2txt/article.rb +1 -1
- data/lib/wp2txt/cli.rb +42 -0
- data/lib/wp2txt/constants.rb +13 -0
- data/lib/wp2txt/corpus.rb +9 -2
- data/lib/wp2txt/data/template_aliases.json +1 -1
- data/lib/wp2txt/extractor.rb +10 -7
- data/lib/wp2txt/formatter.rb +14 -12
- data/lib/wp2txt/index_commands.rb +89 -1
- data/lib/wp2txt/langlinks_importer.rb +19 -35
- data/lib/wp2txt/lead_terms.rb +228 -0
- data/lib/wp2txt/link_counter.rb +171 -0
- data/lib/wp2txt/metadata_index.rb +79 -3
- data/lib/wp2txt/multistream.rb +21 -0
- data/lib/wp2txt/page_props_importer.rb +170 -0
- data/lib/wp2txt/sql_dump_reader.rb +57 -0
- data/lib/wp2txt/stream_processor.rb +9 -2
- data/lib/wp2txt/template_expander.rb +19 -0
- data/lib/wp2txt/utils.rb +5 -3
- data/lib/wp2txt/version.rb +1 -1
- data/lib/wp2txt/wikitext_regions.rb +66 -0
- data/spec/docs_sync_spec.rb +3 -3
- data/spec/langlinks_importer_spec.rb +27 -0
- data/spec/lead_terms_edge_cases_spec.rb +194 -0
- data/spec/lead_terms_links_qids_spec.rb +204 -0
- data/spec/p1_correctness_spec.rb +12 -9
- data/spec/page_properties_spec.rb +161 -0
- data/spec/region_semantics_spec.rb +71 -0
- data/spec/template_passthrough_spec.rb +43 -0
- metadata +16 -1
|
@@ -0,0 +1,194 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require 'spec_helper'
|
|
4
|
+
require 'wp2txt'
|
|
5
|
+
require 'wp2txt/lead_terms'
|
|
6
|
+
require 'wp2txt/link_counter'
|
|
7
|
+
require 'wp2txt/metadata_index'
|
|
8
|
+
require 'open3'
|
|
9
|
+
require 'tmpdir'
|
|
10
|
+
require_relative 'support/multistream_fixture'
|
|
11
|
+
|
|
12
|
+
RSpec.describe 'lead terms and link counting on unusual wikitext' do
|
|
13
|
+
include MultistreamFixture
|
|
14
|
+
|
|
15
|
+
let(:cleaner) { Object.new.extend(Wp2txt) }
|
|
16
|
+
let(:render) { ->(s) { cleaner.format_wiki(s, expand_templates: true, markers: [:all]) } }
|
|
17
|
+
def terms(text)
|
|
18
|
+
Wp2txt::LeadTerms.extract(text, render: render)
|
|
19
|
+
end
|
|
20
|
+
|
|
21
|
+
it '#1 preserves link display and reading with expansion enabled or disabled' do
|
|
22
|
+
{
|
|
23
|
+
'{{読み仮名|[[東京|東京都]]|とうきょう}}' => '東京都(とうきょう)',
|
|
24
|
+
'{{仮リンク|[[東京|東京都]]|en|Tokyo}}' => '東京都',
|
|
25
|
+
'{{読み仮名|{{lang|ja|[[東京|東京都]]}}|とうきょう}}' => '東京都(とうきょう)',
|
|
26
|
+
'{{読み仮名|東京|}}' => '東京'
|
|
27
|
+
}.each do |input, expected|
|
|
28
|
+
[true, false].each do |expand|
|
|
29
|
+
expect(cleaner.format_wiki(input, expand_templates: expand)).to eq(expected)
|
|
30
|
+
end
|
|
31
|
+
end
|
|
32
|
+
end
|
|
33
|
+
|
|
34
|
+
def build_case_dump(dir, rule)
|
|
35
|
+
alias_title = rule == 'first-letter' ? 'Alias' : 'alias'
|
|
36
|
+
pages = [[1, 'apple', ''], [2, 'Apple', ''], [3, alias_title, '#REDIRECT [[apple]]'],
|
|
37
|
+
[4, 'Direct', '[[apple]]'], [5, 'Via', '[[alias]]'], [6, '東京', '']]
|
|
38
|
+
xml = "<mediawiki><siteinfo><case>#{rule}</case></siteinfo>" + pages.map do |id, title, text|
|
|
39
|
+
page_xml(id: id, ns: 0, title: title, text: text)
|
|
40
|
+
end.join + '</mediawiki>'
|
|
41
|
+
dump = File.join(dir, 'testwiki-20260101-multistream.xml.bz2')
|
|
42
|
+
File.binwrite(dump, bzip2(xml))
|
|
43
|
+
db = File.join(dir, 'meta.sqlite3')
|
|
44
|
+
Wp2txt::MetadataIndexBuilder.new(dump, [0], db_path: db, num_processes: 0).build.close
|
|
45
|
+
[dump, db]
|
|
46
|
+
end
|
|
47
|
+
|
|
48
|
+
it '#2 stores siteinfo case and preserves direct and redirected case-sensitive targets' do
|
|
49
|
+
Dir.mktmpdir do |dir|
|
|
50
|
+
dump, path = build_case_dump(dir, 'case-sensitive')
|
|
51
|
+
db = SQLite3::Database.new(path)
|
|
52
|
+
expect(db.get_first_value("SELECT value FROM metadata WHERE key = 'case_rule'")).to eq('case-sensitive')
|
|
53
|
+
db.close
|
|
54
|
+
Wp2txt::LinkCounter.new(dump, [0], db_path: path, num_processes: 0).count!
|
|
55
|
+
db = SQLite3::Database.new(path)
|
|
56
|
+
expect(db.execute('SELECT page_id,inlinks,via_redirects FROM page_inlinks WHERE page_id IN (1,2) ORDER BY page_id'))
|
|
57
|
+
.to eq([[1, 2, 1], [2, 0, 0]])
|
|
58
|
+
db.close
|
|
59
|
+
expect(Wp2txt::MetadataIndex.normalize_title('apple')).to eq('Apple')
|
|
60
|
+
end
|
|
61
|
+
end
|
|
62
|
+
|
|
63
|
+
it '#2 uses first-letter for explicit siteinfo and legacy metadata without the rule' do
|
|
64
|
+
Dir.mktmpdir do |dir|
|
|
65
|
+
dump, path = build_case_dump(dir, 'first-letter')
|
|
66
|
+
[false, true].each do |legacy|
|
|
67
|
+
db = SQLite3::Database.new(path)
|
|
68
|
+
if legacy
|
|
69
|
+
db.execute("DELETE FROM metadata WHERE key = 'case_rule'")
|
|
70
|
+
else
|
|
71
|
+
expect(db.get_first_value("SELECT value FROM metadata WHERE key = 'case_rule'")).to eq('first-letter')
|
|
72
|
+
end
|
|
73
|
+
db.close
|
|
74
|
+
Wp2txt::LinkCounter.new(dump, [0], db_path: path, num_processes: 0).count!
|
|
75
|
+
db = SQLite3::Database.new(path)
|
|
76
|
+
expect(db.get_first_value('SELECT inlinks FROM page_inlinks WHERE page_id=1')).to eq(0)
|
|
77
|
+
expect(db.execute('SELECT inlinks,via_redirects FROM page_inlinks WHERE page_id=2')).to eq([[2, 1]])
|
|
78
|
+
db.close
|
|
79
|
+
end
|
|
80
|
+
end
|
|
81
|
+
end
|
|
82
|
+
|
|
83
|
+
it '#2 reads a separate siteinfo stream before the first indexed page' do
|
|
84
|
+
Dir.mktmpdir do |dir|
|
|
85
|
+
header = bzip2('<mediawiki><siteinfo><case>case-sensitive</case></siteinfo>')
|
|
86
|
+
dump = File.join(dir, 'dump.xml.bz2')
|
|
87
|
+
File.binwrite(dump, header + bzip2(page_xml(id: 1, ns: 0, title: 'apple', text: '') + '</mediawiki>'))
|
|
88
|
+
path = File.join(dir, 'meta.sqlite3')
|
|
89
|
+
Wp2txt::MetadataIndexBuilder.new(dump, [header.bytesize], db_path: path, num_processes: 0).build.close
|
|
90
|
+
db = SQLite3::Database.new(path)
|
|
91
|
+
expect(db.get_first_value("SELECT value FROM metadata WHERE key = 'case_rule'")).to eq('case-sensitive')
|
|
92
|
+
db.close
|
|
93
|
+
end
|
|
94
|
+
end
|
|
95
|
+
|
|
96
|
+
%w[nowiki pre math syntaxhighlight source gallery score timeline chem ce graph mapframe templatedata].each do |tag|
|
|
97
|
+
it "#3 skips #{tag} contents and self-closing forms while preserving source offsets" do
|
|
98
|
+
text = "<#{tag.upcase} data-x='>'>'''偽'''(にせ){{ruby|偽|にせ}}</#{tag.upcase}>\n\n" \
|
|
99
|
+
"<#{tag} />'''東京'''(とうきょう)。"
|
|
100
|
+
result = terms(text)
|
|
101
|
+
expect(result.map { |t| t['text'] }).to eq(['東京'])
|
|
102
|
+
s, e = result.first['span']['bold']
|
|
103
|
+
expect(text[s...e]).to eq("'''東京'''")
|
|
104
|
+
expect(terms("<#{tag}>'''偽'''" )).to eq([])
|
|
105
|
+
end
|
|
106
|
+
end
|
|
107
|
+
|
|
108
|
+
it '#4 ignores delimiters in comments and excluded tags inside parentheses' do
|
|
109
|
+
text = "'''東京'''(とうきょう<!-- ) , -->、<nowiki>),</nowiki>Tokyo)は都市。"
|
|
110
|
+
result = terms(text).first
|
|
111
|
+
s, e = result['span']['paren']
|
|
112
|
+
expect(text[s...e]).to eq('(とうきょう<!-- ) , -->、<nowiki>),</nowiki>Tokyo)')
|
|
113
|
+
expect(result['notes'].size).to eq(2)
|
|
114
|
+
expect(result['notes'].first).to eq('とうきょう')
|
|
115
|
+
expect(result['notes_text']).not_to include('<!--')
|
|
116
|
+
end
|
|
117
|
+
|
|
118
|
+
it 'retains the original text of ruby arguments containing excluded regions' do
|
|
119
|
+
expect(terms('{{ruby|<nowiki>東京</nowiki>|とうきょう}}').first.slice('text', 'reading'))
|
|
120
|
+
.to eq('text' => '東京', 'reading' => 'とうきょう')
|
|
121
|
+
end
|
|
122
|
+
|
|
123
|
+
it '#5 detects only headings outside skipped regions including trailing comments' do
|
|
124
|
+
["<!--\n== 偽 ==\n-->", "<nowiki>\n== 偽 ==\n</nowiki>", "{{box|\n== 偽 ==\n}}", "{|\n== 偽 ==\n|}"].each do |prefix|
|
|
125
|
+
expect(terms("#{prefix}\n'''東京'''(とうきょう)").map { |t| t['text'] }).to eq(['東京'])
|
|
126
|
+
end
|
|
127
|
+
expect(terms("導入文。\n\n== 歴史 == <!-- comment -->\n'''後の語'''(あと)。")).to eq([])
|
|
128
|
+
end
|
|
129
|
+
|
|
130
|
+
it '#6 allows single newlines before and inside notes but stops at a blank line' do
|
|
131
|
+
expect(terms("'''東京'''\n(とうきょう)は都市。").first['notes']).to eq(['とうきょう'])
|
|
132
|
+
text = "'''東京'''(とうきょう、\nTokyo)は都市。"
|
|
133
|
+
term = terms(text).first
|
|
134
|
+
expect(term['notes']).to eq(['とうきょう', 'Tokyo'])
|
|
135
|
+
s, e = term['span']['paren']
|
|
136
|
+
expect(text[s...e]).to eq("(とうきょう、\nTokyo)")
|
|
137
|
+
["'''東京'''\n\n(とうきょう)", "'''東京'''(とうきょう、\n \nTokyo)"].each do |input|
|
|
138
|
+
expect(terms(input).first['notes']).to eq([])
|
|
139
|
+
end
|
|
140
|
+
end
|
|
141
|
+
|
|
142
|
+
it '#7 stops at empty lines containing whitespace' do
|
|
143
|
+
["\n \n", "\n\t\n", "\r\n \r\n"].each do |gap|
|
|
144
|
+
expect(terms("'''東京'''(とうきょう)。#{gap}'''別段落'''(べつだんらく)。").map { |t| t['text'] }).to eq(['東京'])
|
|
145
|
+
end
|
|
146
|
+
end
|
|
147
|
+
|
|
148
|
+
def counts(text)
|
|
149
|
+
counter = Wp2txt::LinkCounter.new('', [], db_path: '')
|
|
150
|
+
counter.instance_variable_set(:@redirects, {})
|
|
151
|
+
direct, via = Hash.new(0), Hash.new(0)
|
|
152
|
+
counter.send(:count_page, page_xml(id: 1, ns: 0, title: 'Source', text: text), direct, via)
|
|
153
|
+
direct
|
|
154
|
+
end
|
|
155
|
+
|
|
156
|
+
it '#8 decodes title entities before sections and normalizes whitespace before colon' do
|
|
157
|
+
expect(counts('[[M&A]] [[Café]] [[ :東京#節|表示]] [[Café#section]] [[#節]] [[]]'))
|
|
158
|
+
.to eq('M&A' => 1, 'Café' => 1, '東京' => 1)
|
|
159
|
+
expect(counts('[[東京#節]] [[東京|表示]]')).to eq('東京' => 1)
|
|
160
|
+
end
|
|
161
|
+
|
|
162
|
+
it '#9 excludes every non-wikitext tag and comments from incoming links' do
|
|
163
|
+
%w[nowiki pre math syntaxhighlight source score chem ce graph mapframe templatedata].each do |tag|
|
|
164
|
+
expect(counts("<#{tag}>[[東京]]</#{tag}>[[大阪]]<#{tag}/><!-- [[京都]] -->")).to eq('大阪' => 1)
|
|
165
|
+
end
|
|
166
|
+
end
|
|
167
|
+
|
|
168
|
+
it 'handles empty text and binary invalid bytes in the shared region helper' do
|
|
169
|
+
expect(terms('')).to eq([])
|
|
170
|
+
expect(counts('')).to eq({})
|
|
171
|
+
expect(Wp2txt::WikitextRegions.remove_literal("<nowiki>\xFF</nowiki>\xFE".b)).to eq("\xFE".b)
|
|
172
|
+
end
|
|
173
|
+
|
|
174
|
+
def cli(*options)
|
|
175
|
+
Dir.mktmpdir do |dir|
|
|
176
|
+
input = File.join(dir, 'x.xml.bz2')
|
|
177
|
+
File.binwrite(input, bzip2("<mediawiki>#{page_xml(id: 1, ns: 0, title: '東京', text: "'''東京'''。")}</mediawiki>"))
|
|
178
|
+
Open3.capture3(RbConfig.ruby, '-I', File.expand_path('../lib', __dir__),
|
|
179
|
+
File.expand_path('../bin/wp2txt', __dir__), '-i', input, '-o', dir, *options)
|
|
180
|
+
end
|
|
181
|
+
end
|
|
182
|
+
|
|
183
|
+
it '#10 rejects lead terms with ractor' do
|
|
184
|
+
_, err, status = cli('--format', 'json', '--lead-terms', '--ractor', '--no-turbo')
|
|
185
|
+
expect(status.success?).to be(false)
|
|
186
|
+
expect(err).to include('--lead-terms cannot be combined with --ractor')
|
|
187
|
+
end
|
|
188
|
+
|
|
189
|
+
it '#10 warns about missing source identifiers in ractor JSON without rejecting it' do
|
|
190
|
+
_, err, status = cli('--format', 'json', '--ractor', '--no-turbo', '-n', '1')
|
|
191
|
+
expect(status.success?).to be(true), err
|
|
192
|
+
expect(err).to include('does not include page IDs or revision IDs')
|
|
193
|
+
end
|
|
194
|
+
end
|
|
@@ -0,0 +1,204 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "spec_helper"
|
|
4
|
+
require "json"
|
|
5
|
+
require "open3"
|
|
6
|
+
require "tmpdir"
|
|
7
|
+
require "zlib"
|
|
8
|
+
require_relative "support/multistream_fixture"
|
|
9
|
+
require "wp2txt"
|
|
10
|
+
require "wp2txt/metadata_index"
|
|
11
|
+
require "wp2txt/multistream"
|
|
12
|
+
require "wp2txt/sql_dump_reader"
|
|
13
|
+
require "wp2txt/page_props_importer"
|
|
14
|
+
require "wp2txt/link_counter"
|
|
15
|
+
require "wp2txt/lead_terms"
|
|
16
|
+
|
|
17
|
+
RSpec.describe "lead terms, incoming links, and Wikidata IDs" do
|
|
18
|
+
include MultistreamFixture
|
|
19
|
+
|
|
20
|
+
around do |example|
|
|
21
|
+
Dir.mktmpdir("wp2txt-ltq-") do |dir|
|
|
22
|
+
@dir = dir
|
|
23
|
+
example.run
|
|
24
|
+
end
|
|
25
|
+
end
|
|
26
|
+
|
|
27
|
+
def xml_page(id:, title:, text:, ns: 0)
|
|
28
|
+
esc = ->(s) { s.gsub("&", "&").gsub("<", "<").gsub(">", ">") }
|
|
29
|
+
"<page>\n<title>#{esc.(title)}</title>\n<ns>#{ns}</ns>\n<id>#{id}</id>\n<revision>\n<id>#{id * 100}</id>\n" \
|
|
30
|
+
"<text bytes=\"#{text.bytesize}\">#{esc.(text)}</text>\n</revision>\n</page>\n"
|
|
31
|
+
end
|
|
32
|
+
|
|
33
|
+
PAGES = [
|
|
34
|
+
[1, "東京", "'''東京'''(とうきょう、Tokyo)は首都。\n\n== 歴史 ==\n'''後'''(あと)", 0],
|
|
35
|
+
[2, "東京都区部", "#REDIRECT [[東京]]", 0],
|
|
36
|
+
[3, "記事A", "[[東京]]と[[東京]]、[[東京都区部]]。[[M&A|買収]]。<!-- [[隠れ]] -->[[#節]][[]]", 0],
|
|
37
|
+
[4, "記事B", "[[東京#歴史|東京の歴史]]", 0],
|
|
38
|
+
[5, "記事C", "[[東京都区部]]のみ。", 0],
|
|
39
|
+
[6, "M&A", "'''{{読み仮名|合併|がっぺい}}'''と買収。", 0],
|
|
40
|
+
[7, "隠れ", "", 0],
|
|
41
|
+
[8, "Category:X", "[[東京]]", 14]
|
|
42
|
+
].freeze
|
|
43
|
+
|
|
44
|
+
# One-stream dump with its index and a built metadata index
|
|
45
|
+
def build_dump
|
|
46
|
+
dump = File.join(@dir, "testwiki-20260101-pages-articles-multistream.xml.bz2")
|
|
47
|
+
File.binwrite(dump, bzip2(PAGES.map { |id, t, x, ns| xml_page(id: id, title: t, text: x, ns: ns) }.join))
|
|
48
|
+
File.write(File.join(@dir, "testwiki-20260101-pages-articles-multistream-index.txt"),
|
|
49
|
+
PAGES.map { |id, t, _, _| "0:#{id}:#{t}" }.join("\n") + "\n")
|
|
50
|
+
db = Wp2txt::MetadataIndex.path_for(dump, cache_dir: @dir)
|
|
51
|
+
Wp2txt::MetadataIndexBuilder.new(dump, [0], db_path: db, num_processes: 0).build.close
|
|
52
|
+
[dump, db]
|
|
53
|
+
end
|
|
54
|
+
|
|
55
|
+
def page_props_file(content, name: "testwiki-20260101-page_props.sql.gz")
|
|
56
|
+
path = File.join(@dir, name)
|
|
57
|
+
Zlib::GzipWriter.open(path) { |gz| gz.write(content) }
|
|
58
|
+
path
|
|
59
|
+
end
|
|
60
|
+
|
|
61
|
+
PAGE_PROPS_SQL = <<~SQL
|
|
62
|
+
INSERT INTO `page_props` VALUES
|
|
63
|
+
(1,'wikibase_item','Q1490',NULL),
|
|
64
|
+
(1,'page_image_free','T\\'kyo.jpg',NULL),
|
|
65
|
+
(3,'wikibase_item','Q9',1.5),
|
|
66
|
+
(6,'wikibase_item','Q1\\'bad',NULL),
|
|
67
|
+
(6,'defaultsort','\xFF\xFE',NULL);
|
|
68
|
+
INSERT INTO `page_restrictions` VALUES
|
|
69
|
+
(1,'wikibase_item','Q777',NULL);
|
|
70
|
+
SQL
|
|
71
|
+
|
|
72
|
+
describe Wp2txt::SqlDumpReader do
|
|
73
|
+
it "reads statements written on one line and one tuple per line, skipping other tables" do
|
|
74
|
+
lines = []
|
|
75
|
+
path = page_props_file("INSERT INTO `t` VALUES (1),(2);\nINSERT INTO `u` VALUES\n(9);\nINSERT INTO `t` VALUES\n(3),\n(4);\n")
|
|
76
|
+
described_class.each_insert_line(path, "t") { |l| lines << l.strip }
|
|
77
|
+
expect(lines).to eq(["INSERT INTO `t` VALUES (1),(2);", "INSERT INTO `t` VALUES", "(3),", "(4);"])
|
|
78
|
+
end
|
|
79
|
+
end
|
|
80
|
+
|
|
81
|
+
describe Wp2txt::PagePropsImporter do
|
|
82
|
+
it "imports only well-formed wikibase_item values, and records where they came from" do
|
|
83
|
+
_dump, db = build_dump
|
|
84
|
+
path = page_props_file(PAGE_PROPS_SQL.b)
|
|
85
|
+
result = described_class.new(db).import!(path)
|
|
86
|
+
expect(result[:row_count]).to eq(2)
|
|
87
|
+
rows = SQLite3::Database.new(db, readonly: true).execute("SELECT page_id, qid FROM page_properties ORDER BY page_id")
|
|
88
|
+
expect(rows).to eq([[1, "Q1490"], [3, "Q9"]])
|
|
89
|
+
expect(result[:provenance][:source_sha256]).to eq(Digest::SHA256.file(path).hexdigest)
|
|
90
|
+
expect(described_class.new(db).import!(path)[:status]).to eq(:already_imported)
|
|
91
|
+
end
|
|
92
|
+
|
|
93
|
+
it "refuses a dump of another date" do
|
|
94
|
+
_dump, db = build_dump
|
|
95
|
+
path = page_props_file(PAGE_PROPS_SQL.b, name: "testwiki-20250101-page_props.sql.gz")
|
|
96
|
+
expect { described_class.new(db).import!(path) }.to raise_error(ArgumentError, /version mismatch/)
|
|
97
|
+
end
|
|
98
|
+
|
|
99
|
+
it "fails when nothing could be read" do
|
|
100
|
+
_dump, db = build_dump
|
|
101
|
+
path = page_props_file("-- empty\n")
|
|
102
|
+
expect { described_class.new(db).import!(path) }.to raise_error(Wp2txt::Error, /no page_props rows/)
|
|
103
|
+
end
|
|
104
|
+
end
|
|
105
|
+
|
|
106
|
+
describe Wp2txt::LinkCounter do
|
|
107
|
+
it "counts each linking article once, through redirects, ignoring comments and non-articles" do
|
|
108
|
+
dump, db = build_dump
|
|
109
|
+
described_class.new(dump, [0], db_path: db, num_processes: 0).count!
|
|
110
|
+
counts = SQLite3::Database.new(db, readonly: true)
|
|
111
|
+
.execute("SELECT p.title, i.inlinks, i.via_redirects FROM page_inlinks i " \
|
|
112
|
+
"JOIN pages p USING (page_id)").to_h { |t, n, v| [t, [n, v]] }
|
|
113
|
+
expect(counts["東京"]).to eq([3, 1]) # 記事A and 記事B directly, 記事C only via the redirect
|
|
114
|
+
expect(counts["M&A"]).to eq([1, 0])
|
|
115
|
+
expect(counts["隠れ"]).to eq([0, 0]) # linked only from a comment
|
|
116
|
+
expect(counts["記事A"]).to eq([0, 0])
|
|
117
|
+
expect(counts).not_to have_key("東京都区部") # redirects get no row of their own
|
|
118
|
+
end
|
|
119
|
+
end
|
|
120
|
+
|
|
121
|
+
describe Wp2txt::LeadTerms do
|
|
122
|
+
let(:render) { ->(fragment) { Object.new.extend(Wp2txt).format_wiki(fragment, { expand_templates: true, markers: [:all] }) } }
|
|
123
|
+
|
|
124
|
+
def terms(text)
|
|
125
|
+
Wp2txt::LeadTerms.extract(text, render: render)
|
|
126
|
+
end
|
|
127
|
+
|
|
128
|
+
it "takes bold terms of the first paragraph that has one, skipping templates, files, refs, and comments" do
|
|
129
|
+
text = "{{Otheruses|'''x'''}}\n[[ファイル:A.jpg|thumb|'''偽''']]<!-- '''隠''' --><ref>'''注'''</ref>\n" \
|
|
130
|
+
"'''日本語'''(にほんご、にっぽんご{{Refnest|注}})は言語。'''和語'''とも。\n\n次の段落の'''別'''。"
|
|
131
|
+
expect(terms(text).map { |t| [t["text"], t["notes"]] }).to eq([["日本語", %w[にほんご にっぽんご]], ["和語", []]])
|
|
132
|
+
end
|
|
133
|
+
|
|
134
|
+
it "reports spans that cut the bold and the parentheses out of the original text" do
|
|
135
|
+
text = "前置き。'''東京''' (とうきょう)は首都。"
|
|
136
|
+
term = terms(text).first
|
|
137
|
+
expect(text[Range.new(*term["span"]["bold"], true)]).to eq("'''東京'''")
|
|
138
|
+
expect(text[Range.new(*term["span"]["paren"], true)]).to eq("(とうきょう)")
|
|
139
|
+
expect(term["notes_text"]).to eq("とうきょう")
|
|
140
|
+
end
|
|
141
|
+
|
|
142
|
+
it "does not split inside nested brackets, templates, or links" do
|
|
143
|
+
text = "'''山野太郎'''(やまの たろう、[[2001年]](平成13年、辛巳)[[4月1日]] - 、{{lang|en|a, b}})は架空の人物。"
|
|
144
|
+
expect(terms(text).first["notes"]).to eq(["やまの たろう", "2001年(平成13年、辛巳)4月1日 -", "a, b"])
|
|
145
|
+
end
|
|
146
|
+
|
|
147
|
+
it "reports reading templates as pairs and keeps their reading out of the bold text" do
|
|
148
|
+
result = terms("'''{{読み仮名|言語|げんご}}'''は記号体系。")
|
|
149
|
+
expect(result.map { |t| t.slice("text", "reading", "source") })
|
|
150
|
+
.to eq([{ "text" => "言語", "source" => "bold" },
|
|
151
|
+
{ "text" => "言語", "reading" => "げんご", "source" => "読み仮名" }])
|
|
152
|
+
end
|
|
153
|
+
|
|
154
|
+
it "stops at the first heading, ignores an unclosed bracket, and caps the count" do
|
|
155
|
+
expect(terms("'''甲'''(こう\n\n== 節 ==\n'''乙'''").map { |t| [t["text"], t["notes"]] }).to eq([["甲", []]])
|
|
156
|
+
many = (1..8).map { |i| "'''語#{i}'''" }.join("、")
|
|
157
|
+
expect(terms(many).map { |t| t["index"] }).to eq([0, 1, 2, 3, 4])
|
|
158
|
+
end
|
|
159
|
+
|
|
160
|
+
it "returns nothing for empty text" do
|
|
161
|
+
expect(terms("")).to eq([])
|
|
162
|
+
end
|
|
163
|
+
end
|
|
164
|
+
|
|
165
|
+
describe "command line" do
|
|
166
|
+
let(:cli) { File.expand_path("../bin/wp2txt", __dir__) }
|
|
167
|
+
let(:lib) { File.expand_path("../lib", __dir__) }
|
|
168
|
+
|
|
169
|
+
def records(*args)
|
|
170
|
+
out = File.join(@dir, "out#{args.hash.abs}")
|
|
171
|
+
Dir.mkdir(out)
|
|
172
|
+
_stdout, stderr, status = Open3.capture3(RbConfig.ruby, "-I", lib, cli, *args, "-o", out)
|
|
173
|
+
expect(status.success?).to be(true), stderr
|
|
174
|
+
Dir[File.join(out, "*")].flat_map { |f| File.readlines(f) }.map { |l| JSON.parse(l) }.to_h { |r| [r["title"], r] }
|
|
175
|
+
end
|
|
176
|
+
|
|
177
|
+
it "adds the Wikidata ID and lead terms to JSON, identically on both extraction paths" do
|
|
178
|
+
dump, db = build_dump
|
|
179
|
+
Wp2txt::PagePropsImporter.new(db).import!(page_props_file(PAGE_PROPS_SQL.b))
|
|
180
|
+
common = ["-i", dump, "--cache-dir", @dir, "--format", "json", "--summary-only", "--lead-terms"]
|
|
181
|
+
turbo = records(*common)
|
|
182
|
+
streamed = records(*common, "--no-turbo")
|
|
183
|
+
|
|
184
|
+
expect(turbo["東京"].keys.first(4)).to eq(%w[title page_id revision_id qid])
|
|
185
|
+
expect(turbo["東京"]["qid"]).to eq("Q1490")
|
|
186
|
+
expect(turbo["記事B"].slice("qid", "sort_key", "disambiguation")).to eq("qid" => nil, "sort_key" => nil, "disambiguation" => false)
|
|
187
|
+
expect(turbo["東京"]["lead_terms"].first.slice("text", "notes"))
|
|
188
|
+
.to eq("text" => "東京", "notes" => %w[とうきょう Tokyo])
|
|
189
|
+
# (the default path skips articles with empty text; compare what both emit)
|
|
190
|
+
common_titles = turbo.keys & streamed.keys
|
|
191
|
+
expect(common_titles.size).to eq(turbo.size)
|
|
192
|
+
expect(streamed.slice(*common_titles).transform_values { |r| r["lead_terms"] })
|
|
193
|
+
.to eq(turbo.transform_values { |r| r["lead_terms"] })
|
|
194
|
+
end
|
|
195
|
+
|
|
196
|
+
it "requires JSON output for --lead-terms" do
|
|
197
|
+
input = File.join(@dir, "x.xml.bz2")
|
|
198
|
+
File.write(input, "")
|
|
199
|
+
_stdout, stderr, status = Open3.capture3(RbConfig.ruby, "-I", lib, cli, "-i", input, "--lead-terms")
|
|
200
|
+
expect(status.success?).to be(false)
|
|
201
|
+
expect(stderr).to include("--lead-terms requires --format json")
|
|
202
|
+
end
|
|
203
|
+
end
|
|
204
|
+
end
|
data/spec/p1_correctness_spec.rb
CHANGED
|
@@ -356,17 +356,20 @@ RSpec.describe "P1 correctness contracts" do
|
|
|
356
356
|
Rake.application = previous
|
|
357
357
|
end
|
|
358
358
|
|
|
359
|
-
|
|
360
|
-
|
|
361
|
-
|
|
362
|
-
|
|
363
|
-
|
|
359
|
+
it "builds from a clean clone and hands the image to the CI gate with that clone as context" do
|
|
360
|
+
commands = []
|
|
361
|
+
allow(TOPLEVEL_BINDING.receiver).to receive(:sh) { |*args| commands << args }
|
|
362
|
+
Rake::Task[:check_image].invoke
|
|
363
|
+
clone, build, gate = commands
|
|
364
|
+
expect(clone.first(4)).to eq(%w[git clone --quiet --no-local])
|
|
365
|
+
dir = clone.last
|
|
366
|
+
expect(build).to eq(["docker", "build", "-t", "wp2txt-verify:local", dir])
|
|
367
|
+
expect(gate).to eq(["ruby", "scripts/verify_image.rb", "wp2txt-verify:local", "--context", dir])
|
|
364
368
|
end
|
|
365
369
|
|
|
366
|
-
it "
|
|
367
|
-
expect(
|
|
368
|
-
|
|
369
|
-
expect { Rake::Task[:verify_image].invoke("test-image") }.to output(/OK:/).to_stdout
|
|
370
|
+
it "no longer carries a hand-written list of forbidden paths" do
|
|
371
|
+
expect(Rake::Task.task_defined?(:verify_image)).to be(false)
|
|
372
|
+
expect(defined?(IMAGE_FORBIDDEN_PATHS)).to be_nil
|
|
370
373
|
end
|
|
371
374
|
end
|
|
372
375
|
end
|
|
@@ -0,0 +1,161 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "spec_helper"
|
|
4
|
+
require "wp2txt/page_props_importer"
|
|
5
|
+
require "wp2txt/metadata_index"
|
|
6
|
+
require "wp2txt/corpus"
|
|
7
|
+
require "wp2txt/link_counter"
|
|
8
|
+
require "tmpdir"
|
|
9
|
+
require "open3"
|
|
10
|
+
require "json"
|
|
11
|
+
require "zlib"
|
|
12
|
+
require_relative "support/multistream_fixture"
|
|
13
|
+
|
|
14
|
+
RSpec.describe "page properties" do
|
|
15
|
+
include MultistreamFixture
|
|
16
|
+
TITLES = %w[東京 QID パイプ Sort Invalid Other Empty Absent].freeze
|
|
17
|
+
FIELDS = %w[qid sort_key disambiguation].freeze
|
|
18
|
+
|
|
19
|
+
around do |example|
|
|
20
|
+
Dir.mktmpdir do |dir|
|
|
21
|
+
@dir = dir
|
|
22
|
+
@dump = File.join(dir, "testwiki-20260101-pages-articles-multistream.xml.bz2")
|
|
23
|
+
xml = TITLES.each_with_index.map { |t, i| page_xml(id: i + 1, ns: 0, title: t, text: "'''#{t}'''です。") }.join
|
|
24
|
+
File.binwrite(@dump, bzip2("<mediawiki>\n#{xml}</mediawiki>"))
|
|
25
|
+
File.write(@dump.sub(".xml.bz2", "-index.txt"), TITLES.each_with_index.map { |t, i| "0:#{i + 1}:#{t}\n" }.join)
|
|
26
|
+
@db_path = Wp2txt::MetadataIndex.path_for(@dump, cache_dir: dir)
|
|
27
|
+
Wp2txt::MetadataIndexBuilder.new(@dump, [0], db_path: @db_path, num_processes: 0).build.close
|
|
28
|
+
@source = File.join(dir, "testwiki-20260101-page_props.sql.gz")
|
|
29
|
+
@sql = <<~SQL
|
|
30
|
+
INSERT INTO `page_props` VALUES
|
|
31
|
+
(1,'defaultsort','とうきよう',NULL),
|
|
32
|
+
(1,'disambiguation','ignored',NULL),
|
|
33
|
+
(1,'wikibase_item','Q1490',NULL),
|
|
34
|
+
(2,'wikibase_item','Q2',NULL),
|
|
35
|
+
(3,'disambiguation','',NULL),
|
|
36
|
+
(4,'defaultsort','O\\'Brien あ',NULL),
|
|
37
|
+
(5,'defaultsort','\xFF\xFE',NULL),
|
|
38
|
+
(6,'other','ignored',NULL),
|
|
39
|
+
(7,'defaultsort','',NULL),
|
|
40
|
+
(8,'wikibase_item','bad',NULL);
|
|
41
|
+
INSERT INTO `other_table` VALUES (8,'wikibase_item','Q999',NULL);
|
|
42
|
+
SQL
|
|
43
|
+
Zlib::GzipWriter.open(@source) { |gz| gz.write(@sql.b) }
|
|
44
|
+
example.run
|
|
45
|
+
end
|
|
46
|
+
end
|
|
47
|
+
|
|
48
|
+
def import(**options)
|
|
49
|
+
Wp2txt::PagePropsImporter.new(@db_path).import!(@source, **options)
|
|
50
|
+
end
|
|
51
|
+
|
|
52
|
+
def with_db
|
|
53
|
+
db = SQLite3::Database.new(@db_path)
|
|
54
|
+
yield db
|
|
55
|
+
ensure
|
|
56
|
+
db&.close
|
|
57
|
+
end
|
|
58
|
+
|
|
59
|
+
it "merges all three properties across batches and records counts and invalid sort keys" do
|
|
60
|
+
stub_const("Wp2txt::PagePropsImporter::BATCH_SIZE", 1)
|
|
61
|
+
result = import
|
|
62
|
+
expect(result[:row_count]).to eq(5)
|
|
63
|
+
with_db do |db|
|
|
64
|
+
expect(db.execute("SELECT page_id,qid,disambiguation,sort_key FROM page_properties ORDER BY page_id"))
|
|
65
|
+
.to eq([[1, "Q1490", 1, "とうきよう"], [2, "Q2", 0, nil], [3, nil, 1, nil],
|
|
66
|
+
[4, nil, 0, "O'Brien あ"], [7, nil, 0, ""]])
|
|
67
|
+
end
|
|
68
|
+
expect(result[:provenance]).to include(page_count: 5, qid_count: 2, disambiguation_count: 2,
|
|
69
|
+
sort_key_count: 3, skipped_invalid_sort_keys: 1,
|
|
70
|
+
source_sha256: Digest::SHA256.file(@source).hexdigest)
|
|
71
|
+
expect(import).to include(status: :already_imported, row_count: 5, provenance: result[:provenance])
|
|
72
|
+
end
|
|
73
|
+
|
|
74
|
+
it "replaces the unreleased QID table without requiring force" do
|
|
75
|
+
with_db do |db|
|
|
76
|
+
db.execute("CREATE TABLE page_qids (page_id INTEGER PRIMARY KEY, qid TEXT)")
|
|
77
|
+
db.execute("INSERT INTO metadata VALUES ('page_props_imported_at','old')")
|
|
78
|
+
end
|
|
79
|
+
expect(import[:status]).to eq(:imported)
|
|
80
|
+
with_db { |db| expect(db.get_first_value("SELECT 1 FROM sqlite_master WHERE name='page_qids'")).to be_nil }
|
|
81
|
+
end
|
|
82
|
+
|
|
83
|
+
def records(*flags)
|
|
84
|
+
out = Dir.mktmpdir("out-", @dir)
|
|
85
|
+
_, stderr, status = Open3.capture3(RbConfig.ruby, "-I", File.expand_path("../lib", __dir__),
|
|
86
|
+
File.expand_path("../bin/wp2txt", __dir__), "-i", @dump,
|
|
87
|
+
"--cache-dir", @dir, "--format", "json", "-n", "2", "-o", out, *flags)
|
|
88
|
+
expect(status.success?).to be(true), stderr
|
|
89
|
+
Dir[File.join(out, "*.jsonl")].flat_map { |f| File.readlines(f).map { |s| JSON.parse(s) } }.to_h { |r| [r["title"], r] }
|
|
90
|
+
end
|
|
91
|
+
|
|
92
|
+
it "always emits the three fields after import on both CLI paths, including null and false" do
|
|
93
|
+
import
|
|
94
|
+
turbo, stream = records, records("--no-turbo")
|
|
95
|
+
expect(turbo.size).to eq(8)
|
|
96
|
+
expect(turbo).to eq(stream)
|
|
97
|
+
turbo.each_value { |r| expect(r.keys.first(6)).to eq(%w[title page_id revision_id qid sort_key disambiguation]) }
|
|
98
|
+
expect(turbo["東京"].slice(*FIELDS)).to eq("qid" => "Q1490", "sort_key" => "とうきよう", "disambiguation" => true)
|
|
99
|
+
expect(turbo["パイプ"].slice(*FIELDS)).to eq("qid" => nil, "sort_key" => nil, "disambiguation" => true)
|
|
100
|
+
expect(turbo["Absent"].slice(*FIELDS)).to eq("qid" => nil, "sort_key" => nil, "disambiguation" => false)
|
|
101
|
+
expect(turbo["Empty"]["sort_key"]).to eq("")
|
|
102
|
+
end
|
|
103
|
+
|
|
104
|
+
it "omits all three fields before import and after a failed force import" do
|
|
105
|
+
[false, true].each do |fail_import|
|
|
106
|
+
if fail_import
|
|
107
|
+
import
|
|
108
|
+
Zlib::GzipWriter.open(@source) { |gz| gz.write("-- empty\n") }
|
|
109
|
+
expect { import(force: true) }.to raise_error(Wp2txt::Error, /no page_props rows/)
|
|
110
|
+
end
|
|
111
|
+
[[], ["--no-turbo"]].each do |flags|
|
|
112
|
+
records(*flags).each_value { |r| expect(r.keys & FIELDS).to eq([]) }
|
|
113
|
+
end
|
|
114
|
+
end
|
|
115
|
+
end
|
|
116
|
+
|
|
117
|
+
it "attaches properties to Ractor JSON in the parent without omitting nulls" do
|
|
118
|
+
import
|
|
119
|
+
result = records("--no-turbo", "--ractor")
|
|
120
|
+
expect(result["東京"].slice(*FIELDS)).to eq("qid" => "Q1490", "sort_key" => "とうきよう", "disambiguation" => true)
|
|
121
|
+
expect(result["Absent"].slice(*FIELDS)).to eq("qid" => nil, "sort_key" => nil, "disambiguation" => false)
|
|
122
|
+
end
|
|
123
|
+
|
|
124
|
+
it "distinguishes an imported dump with no relevant properties from an unimported dump" do
|
|
125
|
+
Zlib::GzipWriter.open(@source) { |gz| gz.write("INSERT INTO `page_props` VALUES (1,'other','',NULL);\n") }
|
|
126
|
+
expect(import[:row_count]).to eq(0)
|
|
127
|
+
expect(records["東京"].slice(*FIELDS)).to eq("qid" => nil, "sort_key" => nil, "disambiguation" => false)
|
|
128
|
+
end
|
|
129
|
+
|
|
130
|
+
it "applies the same contract to MCP's Corpus methods and provenance" do
|
|
131
|
+
corpus = Wp2txt::Corpus.for_input(@dump, cache_dir: @dir)
|
|
132
|
+
expect(corpus.get_article("東京").keys & FIELDS.map(&:to_sym)).to eq([])
|
|
133
|
+
unimported = File.join(@dir, "unimported.jsonl")
|
|
134
|
+
corpus.extract_corpus(output_path: unimported, titles: ["東京", "Absent"], content: "full", num_processes: 0)
|
|
135
|
+
File.readlines(unimported).each { |line| expect(JSON.parse(line).keys & FIELDS).to eq([]) }
|
|
136
|
+
corpus.close
|
|
137
|
+
import
|
|
138
|
+
corpus = Wp2txt::Corpus.for_input(@dump, cache_dir: @dir)
|
|
139
|
+
expect(corpus.get_article("東京")).to include(qid: "Q1490", sort_key: "とうきよう", disambiguation: true)
|
|
140
|
+
expect(corpus.get_article("Absent")).to include(qid: nil, sort_key: nil, disambiguation: false)
|
|
141
|
+
output = File.join(@dir, "corpus.jsonl")
|
|
142
|
+
corpus.extract_corpus(output_path: output, titles: ["東京", "パイプ", "Absent"], content: "full", num_processes: 0)
|
|
143
|
+
rows = File.readlines(output).map { |s| JSON.parse(s) }
|
|
144
|
+
expect(rows.size).to eq(3)
|
|
145
|
+
rows.each { |r| expect(r.keys & FIELDS).to match_array(FIELDS) }
|
|
146
|
+
expect(rows.last.slice(*FIELDS)).to eq("qid" => nil, "sort_key" => nil, "disambiguation" => false)
|
|
147
|
+
expect(corpus.dump_info[:page_properties]).to include(qid_count: 2, disambiguation_count: 2, sort_key_count: 3,
|
|
148
|
+
skipped_invalid_sort_keys: 1)
|
|
149
|
+
ensure
|
|
150
|
+
corpus&.close
|
|
151
|
+
end
|
|
152
|
+
|
|
153
|
+
it "decodes title references once, including escaped ampersands and numeric references" do
|
|
154
|
+
counter = Wp2txt::LinkCounter.new("", [], db_path: "")
|
|
155
|
+
counter.instance_variable_set(:@redirects, {})
|
|
156
|
+
direct, via = Hash.new(0), Hash.new(0)
|
|
157
|
+
text = "[[A&amp;B]] [[Café]] [[東京#節]] [[X&lt;Y]]"
|
|
158
|
+
counter.send(:count_page, page_xml(id: 1, ns: 0, title: "Source", text: text), direct, via)
|
|
159
|
+
expect(direct).to eq("A&B" => 1, "Café" => 1, "東京" => 1, "X<Y" => 1)
|
|
160
|
+
end
|
|
161
|
+
end
|
|
@@ -0,0 +1,71 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "spec_helper"
|
|
4
|
+
require "wp2txt/lead_terms"
|
|
5
|
+
require "wp2txt/link_counter"
|
|
6
|
+
require_relative "support/multistream_fixture"
|
|
7
|
+
|
|
8
|
+
RSpec.describe "literal regions and lead prose" do
|
|
9
|
+
include MultistreamFixture
|
|
10
|
+
|
|
11
|
+
def terms(text)
|
|
12
|
+
render = ->(s) { Object.new.extend(Wp2txt).format_wiki(s, expand_templates: true, markers: [:all]) }
|
|
13
|
+
Wp2txt::LeadTerms.extract(text, render: render)
|
|
14
|
+
end
|
|
15
|
+
|
|
16
|
+
def counts(text)
|
|
17
|
+
counter = Wp2txt::LinkCounter.new("", [], db_path: "")
|
|
18
|
+
counter.instance_variable_set(:@redirects, {})
|
|
19
|
+
direct, via = Hash.new(0), Hash.new(0)
|
|
20
|
+
counter.send(:count_page, page_xml(id: 1, ns: 0, title: "Source", text: text), direct, via)
|
|
21
|
+
direct
|
|
22
|
+
end
|
|
23
|
+
|
|
24
|
+
%w[gallery timeline].each do |tag|
|
|
25
|
+
it "counts links inside #{tag}, but excludes its bold text from the lead" do
|
|
26
|
+
text = "<#{tag}>[[東京]] '''偽'''(にせ)</#{tag}>\n\n'''大阪'''(おおさか)。"
|
|
27
|
+
expect(counts(text)).to eq("東京" => 1)
|
|
28
|
+
expect(terms(text).map { |term| term["text"] }).to eq(["大阪"])
|
|
29
|
+
expect(counts("<#{tag}>[[東京]]<nowiki>[[京都]]</nowiki>[[大阪]]</#{tag}>"))
|
|
30
|
+
.to eq("東京" => 1, "大阪" => 1)
|
|
31
|
+
end
|
|
32
|
+
end
|
|
33
|
+
|
|
34
|
+
it "counts links and extracts bold terms inside code as ordinary wikitext" do
|
|
35
|
+
text = "<code>'''東京'''(とうきょう) [[大阪]]</code>"
|
|
36
|
+
expect(counts(text)).to eq("大阪" => 1)
|
|
37
|
+
term = terms(text).first
|
|
38
|
+
expect(term).not_to be_nil
|
|
39
|
+
expect(term.slice("text", "notes")).to eq("text" => "東京", "notes" => ["とうきょう"])
|
|
40
|
+
s, e = term["span"]["bold"]
|
|
41
|
+
expect(text[s...e]).to eq("'''東京'''")
|
|
42
|
+
s, e = term["span"]["paren"]
|
|
43
|
+
expect(text[s...e]).to eq("(とうきょう)")
|
|
44
|
+
end
|
|
45
|
+
|
|
46
|
+
%w[nowiki pre math chem ce score syntaxhighlight source graph mapframe templatedata].each do |tag|
|
|
47
|
+
it "excludes literal #{tag} contents from both consumers and protects its pipes" do
|
|
48
|
+
region = "<#{tag}>[[東京]] '''偽'''(にせ)A|B</#{tag}>"
|
|
49
|
+
expect(counts(region + "[[大阪]]")).to eq("大阪" => 1)
|
|
50
|
+
expect(terms(region + "\n\n'''大阪'''(おおさか)。").map { |term| term["text"] }).to eq(["大阪"])
|
|
51
|
+
expect(Wp2txt::WikitextRegions.split_pipes("ruby|#{region}|よみ")).to eq(["ruby", region, "よみ"])
|
|
52
|
+
end
|
|
53
|
+
end
|
|
54
|
+
|
|
55
|
+
%w[gallery timeline code].each do |tag|
|
|
56
|
+
it "splits template pipes inside #{tag} while still protecting nested literal regions" do
|
|
57
|
+
expect(Wp2txt::WikitextRegions.split_pipes("ruby|<#{tag}>A|B</#{tag}>|よみ"))
|
|
58
|
+
.to eq(["ruby", "<#{tag}>A", "B</#{tag}>", "よみ"])
|
|
59
|
+
expect(Wp2txt::WikitextRegions.split_pipes("ruby|<#{tag}>A<nowiki>|</nowiki>B|C</#{tag}>|よみ"))
|
|
60
|
+
.to eq(["ruby", "<#{tag}>A<nowiki>|</nowiki>B", "C</#{tag}>", "よみ"])
|
|
61
|
+
end
|
|
62
|
+
end
|
|
63
|
+
|
|
64
|
+
it "continues to exclude comments and keeps rule version 2" do
|
|
65
|
+
expect(counts("<!-- [[東京]] -->[[大阪]]")).to eq("大阪" => 1)
|
|
66
|
+
expect(terms("<!-- '''偽''' -->'''大阪'''(おおさか)").first["text"]).to eq("大阪")
|
|
67
|
+
expect(Wp2txt::WikitextRegions.split_pipes("ruby|<!-- a|b -->東京|よみ"))
|
|
68
|
+
.to eq(["ruby", "<!-- a|b -->東京", "よみ"])
|
|
69
|
+
expect(Wp2txt::LinkCounter::RULE_VERSION).to eq("2")
|
|
70
|
+
end
|
|
71
|
+
end
|