wp2txt 2.3.4 → 2.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,194 @@
1
+ # frozen_string_literal: true
2
+
3
+ require 'spec_helper'
4
+ require 'wp2txt'
5
+ require 'wp2txt/lead_terms'
6
+ require 'wp2txt/link_counter'
7
+ require 'wp2txt/metadata_index'
8
+ require 'open3'
9
+ require 'tmpdir'
10
+ require_relative 'support/multistream_fixture'
11
+
12
+ RSpec.describe 'lead terms and link counting on unusual wikitext' do
13
+ include MultistreamFixture
14
+
15
+ let(:cleaner) { Object.new.extend(Wp2txt) }
16
+ let(:render) { ->(s) { cleaner.format_wiki(s, expand_templates: true, markers: [:all]) } }
17
+ def terms(text)
18
+ Wp2txt::LeadTerms.extract(text, render: render)
19
+ end
20
+
21
+ it '#1 preserves link display and reading with expansion enabled or disabled' do
22
+ {
23
+ '{{読み仮名|[[東京|東京都]]|とうきょう}}' => '東京都(とうきょう)',
24
+ '{{仮リンク|[[東京|東京都]]|en|Tokyo}}' => '東京都',
25
+ '{{読み仮名|{{lang|ja|[[東京|東京都]]}}|とうきょう}}' => '東京都(とうきょう)',
26
+ '{{読み仮名|東京|}}' => '東京'
27
+ }.each do |input, expected|
28
+ [true, false].each do |expand|
29
+ expect(cleaner.format_wiki(input, expand_templates: expand)).to eq(expected)
30
+ end
31
+ end
32
+ end
33
+
34
+ def build_case_dump(dir, rule)
35
+ alias_title = rule == 'first-letter' ? 'Alias' : 'alias'
36
+ pages = [[1, 'apple', ''], [2, 'Apple', ''], [3, alias_title, '#REDIRECT [[apple]]'],
37
+ [4, 'Direct', '[[apple]]'], [5, 'Via', '[[alias]]'], [6, '東京', '']]
38
+ xml = "<mediawiki><siteinfo><case>#{rule}</case></siteinfo>" + pages.map do |id, title, text|
39
+ page_xml(id: id, ns: 0, title: title, text: text)
40
+ end.join + '</mediawiki>'
41
+ dump = File.join(dir, 'testwiki-20260101-multistream.xml.bz2')
42
+ File.binwrite(dump, bzip2(xml))
43
+ db = File.join(dir, 'meta.sqlite3')
44
+ Wp2txt::MetadataIndexBuilder.new(dump, [0], db_path: db, num_processes: 0).build.close
45
+ [dump, db]
46
+ end
47
+
48
+ it '#2 stores siteinfo case and preserves direct and redirected case-sensitive targets' do
49
+ Dir.mktmpdir do |dir|
50
+ dump, path = build_case_dump(dir, 'case-sensitive')
51
+ db = SQLite3::Database.new(path)
52
+ expect(db.get_first_value("SELECT value FROM metadata WHERE key = 'case_rule'")).to eq('case-sensitive')
53
+ db.close
54
+ Wp2txt::LinkCounter.new(dump, [0], db_path: path, num_processes: 0).count!
55
+ db = SQLite3::Database.new(path)
56
+ expect(db.execute('SELECT page_id,inlinks,via_redirects FROM page_inlinks WHERE page_id IN (1,2) ORDER BY page_id'))
57
+ .to eq([[1, 2, 1], [2, 0, 0]])
58
+ db.close
59
+ expect(Wp2txt::MetadataIndex.normalize_title('apple')).to eq('Apple')
60
+ end
61
+ end
62
+
63
+ it '#2 uses first-letter for explicit siteinfo and legacy metadata without the rule' do
64
+ Dir.mktmpdir do |dir|
65
+ dump, path = build_case_dump(dir, 'first-letter')
66
+ [false, true].each do |legacy|
67
+ db = SQLite3::Database.new(path)
68
+ if legacy
69
+ db.execute("DELETE FROM metadata WHERE key = 'case_rule'")
70
+ else
71
+ expect(db.get_first_value("SELECT value FROM metadata WHERE key = 'case_rule'")).to eq('first-letter')
72
+ end
73
+ db.close
74
+ Wp2txt::LinkCounter.new(dump, [0], db_path: path, num_processes: 0).count!
75
+ db = SQLite3::Database.new(path)
76
+ expect(db.get_first_value('SELECT inlinks FROM page_inlinks WHERE page_id=1')).to eq(0)
77
+ expect(db.execute('SELECT inlinks,via_redirects FROM page_inlinks WHERE page_id=2')).to eq([[2, 1]])
78
+ db.close
79
+ end
80
+ end
81
+ end
82
+
83
+ it '#2 reads a separate siteinfo stream before the first indexed page' do
84
+ Dir.mktmpdir do |dir|
85
+ header = bzip2('<mediawiki><siteinfo><case>case-sensitive</case></siteinfo>')
86
+ dump = File.join(dir, 'dump.xml.bz2')
87
+ File.binwrite(dump, header + bzip2(page_xml(id: 1, ns: 0, title: 'apple', text: '') + '</mediawiki>'))
88
+ path = File.join(dir, 'meta.sqlite3')
89
+ Wp2txt::MetadataIndexBuilder.new(dump, [header.bytesize], db_path: path, num_processes: 0).build.close
90
+ db = SQLite3::Database.new(path)
91
+ expect(db.get_first_value("SELECT value FROM metadata WHERE key = 'case_rule'")).to eq('case-sensitive')
92
+ db.close
93
+ end
94
+ end
95
+
96
+ %w[nowiki pre math syntaxhighlight source gallery score timeline chem ce graph mapframe templatedata].each do |tag|
97
+ it "#3 skips #{tag} contents and self-closing forms while preserving source offsets" do
98
+ text = "<#{tag.upcase} data-x='>'>'''偽'''(にせ){{ruby|偽|にせ}}</#{tag.upcase}>\n\n" \
99
+ "<#{tag} />'''東京'''(とうきょう)。"
100
+ result = terms(text)
101
+ expect(result.map { |t| t['text'] }).to eq(['東京'])
102
+ s, e = result.first['span']['bold']
103
+ expect(text[s...e]).to eq("'''東京'''")
104
+ expect(terms("<#{tag}>'''偽'''" )).to eq([])
105
+ end
106
+ end
107
+
108
+ it '#4 ignores delimiters in comments and excluded tags inside parentheses' do
109
+ text = "'''東京'''(とうきょう<!-- ) , -->、<nowiki>),</nowiki>Tokyo)は都市。"
110
+ result = terms(text).first
111
+ s, e = result['span']['paren']
112
+ expect(text[s...e]).to eq('(とうきょう<!-- ) , -->、<nowiki>),</nowiki>Tokyo)')
113
+ expect(result['notes'].size).to eq(2)
114
+ expect(result['notes'].first).to eq('とうきょう')
115
+ expect(result['notes_text']).not_to include('<!--')
116
+ end
117
+
118
+ it 'retains the original text of ruby arguments containing excluded regions' do
119
+ expect(terms('{{ruby|<nowiki>東京</nowiki>|とうきょう}}').first.slice('text', 'reading'))
120
+ .to eq('text' => '東京', 'reading' => 'とうきょう')
121
+ end
122
+
123
+ it '#5 detects only headings outside skipped regions including trailing comments' do
124
+ ["<!--\n== 偽 ==\n-->", "<nowiki>\n== 偽 ==\n</nowiki>", "{{box|\n== 偽 ==\n}}", "{|\n== 偽 ==\n|}"].each do |prefix|
125
+ expect(terms("#{prefix}\n'''東京'''(とうきょう)").map { |t| t['text'] }).to eq(['東京'])
126
+ end
127
+ expect(terms("導入文。\n\n== 歴史 == <!-- comment -->\n'''後の語'''(あと)。")).to eq([])
128
+ end
129
+
130
+ it '#6 allows single newlines before and inside notes but stops at a blank line' do
131
+ expect(terms("'''東京'''\n(とうきょう)は都市。").first['notes']).to eq(['とうきょう'])
132
+ text = "'''東京'''(とうきょう、\nTokyo)は都市。"
133
+ term = terms(text).first
134
+ expect(term['notes']).to eq(['とうきょう', 'Tokyo'])
135
+ s, e = term['span']['paren']
136
+ expect(text[s...e]).to eq("(とうきょう、\nTokyo)")
137
+ ["'''東京'''\n\n(とうきょう)", "'''東京'''(とうきょう、\n \nTokyo)"].each do |input|
138
+ expect(terms(input).first['notes']).to eq([])
139
+ end
140
+ end
141
+
142
+ it '#7 stops at empty lines containing whitespace' do
143
+ ["\n \n", "\n\t\n", "\r\n \r\n"].each do |gap|
144
+ expect(terms("'''東京'''(とうきょう)。#{gap}'''別段落'''(べつだんらく)。").map { |t| t['text'] }).to eq(['東京'])
145
+ end
146
+ end
147
+
148
+ def counts(text)
149
+ counter = Wp2txt::LinkCounter.new('', [], db_path: '')
150
+ counter.instance_variable_set(:@redirects, {})
151
+ direct, via = Hash.new(0), Hash.new(0)
152
+ counter.send(:count_page, page_xml(id: 1, ns: 0, title: 'Source', text: text), direct, via)
153
+ direct
154
+ end
155
+
156
+ it '#8 decodes title entities before sections and normalizes whitespace before colon' do
157
+ expect(counts('[[M&amp;A]] [[Caf&#233;]] [[ :東京#節|表示]] [[Caf&#xE9;#section]] [[#節]] [[]]'))
158
+ .to eq('M&A' => 1, 'Café' => 1, '東京' => 1)
159
+ expect(counts('[[東京&#35;節]] [[東京|表示]]')).to eq('東京' => 1)
160
+ end
161
+
162
+ it '#9 excludes every non-wikitext tag and comments from incoming links' do
163
+ %w[nowiki pre math syntaxhighlight source score chem ce graph mapframe templatedata].each do |tag|
164
+ expect(counts("<#{tag}>[[東京]]</#{tag}>[[大阪]]<#{tag}/><!-- [[京都]] -->")).to eq('大阪' => 1)
165
+ end
166
+ end
167
+
168
+ it 'handles empty text and binary invalid bytes in the shared region helper' do
169
+ expect(terms('')).to eq([])
170
+ expect(counts('')).to eq({})
171
+ expect(Wp2txt::WikitextRegions.remove_literal("<nowiki>\xFF</nowiki>\xFE".b)).to eq("\xFE".b)
172
+ end
173
+
174
+ def cli(*options)
175
+ Dir.mktmpdir do |dir|
176
+ input = File.join(dir, 'x.xml.bz2')
177
+ File.binwrite(input, bzip2("<mediawiki>#{page_xml(id: 1, ns: 0, title: '東京', text: "'''東京'''。")}</mediawiki>"))
178
+ Open3.capture3(RbConfig.ruby, '-I', File.expand_path('../lib', __dir__),
179
+ File.expand_path('../bin/wp2txt', __dir__), '-i', input, '-o', dir, *options)
180
+ end
181
+ end
182
+
183
+ it '#10 rejects lead terms with ractor' do
184
+ _, err, status = cli('--format', 'json', '--lead-terms', '--ractor', '--no-turbo')
185
+ expect(status.success?).to be(false)
186
+ expect(err).to include('--lead-terms cannot be combined with --ractor')
187
+ end
188
+
189
+ it '#10 warns about missing source identifiers in ractor JSON without rejecting it' do
190
+ _, err, status = cli('--format', 'json', '--ractor', '--no-turbo', '-n', '1')
191
+ expect(status.success?).to be(true), err
192
+ expect(err).to include('does not include page IDs or revision IDs')
193
+ end
194
+ end
@@ -0,0 +1,204 @@
1
+ # frozen_string_literal: true
2
+
3
+ require "spec_helper"
4
+ require "json"
5
+ require "open3"
6
+ require "tmpdir"
7
+ require "zlib"
8
+ require_relative "support/multistream_fixture"
9
+ require "wp2txt"
10
+ require "wp2txt/metadata_index"
11
+ require "wp2txt/multistream"
12
+ require "wp2txt/sql_dump_reader"
13
+ require "wp2txt/page_props_importer"
14
+ require "wp2txt/link_counter"
15
+ require "wp2txt/lead_terms"
16
+
17
+ RSpec.describe "lead terms, incoming links, and Wikidata IDs" do
18
+ include MultistreamFixture
19
+
20
+ around do |example|
21
+ Dir.mktmpdir("wp2txt-ltq-") do |dir|
22
+ @dir = dir
23
+ example.run
24
+ end
25
+ end
26
+
27
+ def xml_page(id:, title:, text:, ns: 0)
28
+ esc = ->(s) { s.gsub("&", "&amp;").gsub("<", "&lt;").gsub(">", "&gt;") }
29
+ "<page>\n<title>#{esc.(title)}</title>\n<ns>#{ns}</ns>\n<id>#{id}</id>\n<revision>\n<id>#{id * 100}</id>\n" \
30
+ "<text bytes=\"#{text.bytesize}\">#{esc.(text)}</text>\n</revision>\n</page>\n"
31
+ end
32
+
33
+ PAGES = [
34
+ [1, "東京", "'''東京'''(とうきょう、Tokyo)は首都。\n\n== 歴史 ==\n'''後'''(あと)", 0],
35
+ [2, "東京都区部", "#REDIRECT [[東京]]", 0],
36
+ [3, "記事A", "[[東京]]と[[東京]]、[[東京都区部]]。[[M&A|買収]]。<!-- [[隠れ]] -->[[#節]][[]]", 0],
37
+ [4, "記事B", "[[東京#歴史|東京の歴史]]", 0],
38
+ [5, "記事C", "[[東京都区部]]のみ。", 0],
39
+ [6, "M&A", "'''{{読み仮名|合併|がっぺい}}'''と買収。", 0],
40
+ [7, "隠れ", "", 0],
41
+ [8, "Category:X", "[[東京]]", 14]
42
+ ].freeze
43
+
44
+ # One-stream dump with its index and a built metadata index
45
+ def build_dump
46
+ dump = File.join(@dir, "testwiki-20260101-pages-articles-multistream.xml.bz2")
47
+ File.binwrite(dump, bzip2(PAGES.map { |id, t, x, ns| xml_page(id: id, title: t, text: x, ns: ns) }.join))
48
+ File.write(File.join(@dir, "testwiki-20260101-pages-articles-multistream-index.txt"),
49
+ PAGES.map { |id, t, _, _| "0:#{id}:#{t}" }.join("\n") + "\n")
50
+ db = Wp2txt::MetadataIndex.path_for(dump, cache_dir: @dir)
51
+ Wp2txt::MetadataIndexBuilder.new(dump, [0], db_path: db, num_processes: 0).build.close
52
+ [dump, db]
53
+ end
54
+
55
+ def page_props_file(content, name: "testwiki-20260101-page_props.sql.gz")
56
+ path = File.join(@dir, name)
57
+ Zlib::GzipWriter.open(path) { |gz| gz.write(content) }
58
+ path
59
+ end
60
+
61
+ PAGE_PROPS_SQL = <<~SQL
62
+ INSERT INTO `page_props` VALUES
63
+ (1,'wikibase_item','Q1490',NULL),
64
+ (1,'page_image_free','T\\'kyo.jpg',NULL),
65
+ (3,'wikibase_item','Q9',1.5),
66
+ (6,'wikibase_item','Q1\\'bad',NULL),
67
+ (6,'defaultsort','\xFF\xFE',NULL);
68
+ INSERT INTO `page_restrictions` VALUES
69
+ (1,'wikibase_item','Q777',NULL);
70
+ SQL
71
+
72
+ describe Wp2txt::SqlDumpReader do
73
+ it "reads statements written on one line and one tuple per line, skipping other tables" do
74
+ lines = []
75
+ path = page_props_file("INSERT INTO `t` VALUES (1),(2);\nINSERT INTO `u` VALUES\n(9);\nINSERT INTO `t` VALUES\n(3),\n(4);\n")
76
+ described_class.each_insert_line(path, "t") { |l| lines << l.strip }
77
+ expect(lines).to eq(["INSERT INTO `t` VALUES (1),(2);", "INSERT INTO `t` VALUES", "(3),", "(4);"])
78
+ end
79
+ end
80
+
81
+ describe Wp2txt::PagePropsImporter do
82
+ it "imports only well-formed wikibase_item values, and records where they came from" do
83
+ _dump, db = build_dump
84
+ path = page_props_file(PAGE_PROPS_SQL.b)
85
+ result = described_class.new(db).import!(path)
86
+ expect(result[:row_count]).to eq(2)
87
+ rows = SQLite3::Database.new(db, readonly: true).execute("SELECT page_id, qid FROM page_properties ORDER BY page_id")
88
+ expect(rows).to eq([[1, "Q1490"], [3, "Q9"]])
89
+ expect(result[:provenance][:source_sha256]).to eq(Digest::SHA256.file(path).hexdigest)
90
+ expect(described_class.new(db).import!(path)[:status]).to eq(:already_imported)
91
+ end
92
+
93
+ it "refuses a dump of another date" do
94
+ _dump, db = build_dump
95
+ path = page_props_file(PAGE_PROPS_SQL.b, name: "testwiki-20250101-page_props.sql.gz")
96
+ expect { described_class.new(db).import!(path) }.to raise_error(ArgumentError, /version mismatch/)
97
+ end
98
+
99
+ it "fails when nothing could be read" do
100
+ _dump, db = build_dump
101
+ path = page_props_file("-- empty\n")
102
+ expect { described_class.new(db).import!(path) }.to raise_error(Wp2txt::Error, /no page_props rows/)
103
+ end
104
+ end
105
+
106
+ describe Wp2txt::LinkCounter do
107
+ it "counts each linking article once, through redirects, ignoring comments and non-articles" do
108
+ dump, db = build_dump
109
+ described_class.new(dump, [0], db_path: db, num_processes: 0).count!
110
+ counts = SQLite3::Database.new(db, readonly: true)
111
+ .execute("SELECT p.title, i.inlinks, i.via_redirects FROM page_inlinks i " \
112
+ "JOIN pages p USING (page_id)").to_h { |t, n, v| [t, [n, v]] }
113
+ expect(counts["東京"]).to eq([3, 1]) # 記事A and 記事B directly, 記事C only via the redirect
114
+ expect(counts["M&A"]).to eq([1, 0])
115
+ expect(counts["隠れ"]).to eq([0, 0]) # linked only from a comment
116
+ expect(counts["記事A"]).to eq([0, 0])
117
+ expect(counts).not_to have_key("東京都区部") # redirects get no row of their own
118
+ end
119
+ end
120
+
121
+ describe Wp2txt::LeadTerms do
122
+ let(:render) { ->(fragment) { Object.new.extend(Wp2txt).format_wiki(fragment, { expand_templates: true, markers: [:all] }) } }
123
+
124
+ def terms(text)
125
+ Wp2txt::LeadTerms.extract(text, render: render)
126
+ end
127
+
128
+ it "takes bold terms of the first paragraph that has one, skipping templates, files, refs, and comments" do
129
+ text = "{{Otheruses|'''x'''}}\n[[ファイル:A.jpg|thumb|'''偽''']]<!-- '''隠''' --><ref>'''注'''</ref>\n" \
130
+ "'''日本語'''(にほんご、にっぽんご{{Refnest|注}})は言語。'''和語'''とも。\n\n次の段落の'''別'''。"
131
+ expect(terms(text).map { |t| [t["text"], t["notes"]] }).to eq([["日本語", %w[にほんご にっぽんご]], ["和語", []]])
132
+ end
133
+
134
+ it "reports spans that cut the bold and the parentheses out of the original text" do
135
+ text = "前置き。'''東京''' (とうきょう)は首都。"
136
+ term = terms(text).first
137
+ expect(text[Range.new(*term["span"]["bold"], true)]).to eq("'''東京'''")
138
+ expect(text[Range.new(*term["span"]["paren"], true)]).to eq("(とうきょう)")
139
+ expect(term["notes_text"]).to eq("とうきょう")
140
+ end
141
+
142
+ it "does not split inside nested brackets, templates, or links" do
143
+ text = "'''山野太郎'''(やまの たろう、[[2001年]](平成13年、辛巳)[[4月1日]] - 、{{lang|en|a, b}})は架空の人物。"
144
+ expect(terms(text).first["notes"]).to eq(["やまの たろう", "2001年(平成13年、辛巳)4月1日 -", "a, b"])
145
+ end
146
+
147
+ it "reports reading templates as pairs and keeps their reading out of the bold text" do
148
+ result = terms("'''{{読み仮名|言語|げんご}}'''は記号体系。")
149
+ expect(result.map { |t| t.slice("text", "reading", "source") })
150
+ .to eq([{ "text" => "言語", "source" => "bold" },
151
+ { "text" => "言語", "reading" => "げんご", "source" => "読み仮名" }])
152
+ end
153
+
154
+ it "stops at the first heading, ignores an unclosed bracket, and caps the count" do
155
+ expect(terms("'''甲'''(こう\n\n== 節 ==\n'''乙'''").map { |t| [t["text"], t["notes"]] }).to eq([["甲", []]])
156
+ many = (1..8).map { |i| "'''語#{i}'''" }.join("、")
157
+ expect(terms(many).map { |t| t["index"] }).to eq([0, 1, 2, 3, 4])
158
+ end
159
+
160
+ it "returns nothing for empty text" do
161
+ expect(terms("")).to eq([])
162
+ end
163
+ end
164
+
165
+ describe "command line" do
166
+ let(:cli) { File.expand_path("../bin/wp2txt", __dir__) }
167
+ let(:lib) { File.expand_path("../lib", __dir__) }
168
+
169
+ def records(*args)
170
+ out = File.join(@dir, "out#{args.hash.abs}")
171
+ Dir.mkdir(out)
172
+ _stdout, stderr, status = Open3.capture3(RbConfig.ruby, "-I", lib, cli, *args, "-o", out)
173
+ expect(status.success?).to be(true), stderr
174
+ Dir[File.join(out, "*")].flat_map { |f| File.readlines(f) }.map { |l| JSON.parse(l) }.to_h { |r| [r["title"], r] }
175
+ end
176
+
177
+ it "adds the Wikidata ID and lead terms to JSON, identically on both extraction paths" do
178
+ dump, db = build_dump
179
+ Wp2txt::PagePropsImporter.new(db).import!(page_props_file(PAGE_PROPS_SQL.b))
180
+ common = ["-i", dump, "--cache-dir", @dir, "--format", "json", "--summary-only", "--lead-terms"]
181
+ turbo = records(*common)
182
+ streamed = records(*common, "--no-turbo")
183
+
184
+ expect(turbo["東京"].keys.first(4)).to eq(%w[title page_id revision_id qid])
185
+ expect(turbo["東京"]["qid"]).to eq("Q1490")
186
+ expect(turbo["記事B"].slice("qid", "sort_key", "disambiguation")).to eq("qid" => nil, "sort_key" => nil, "disambiguation" => false)
187
+ expect(turbo["東京"]["lead_terms"].first.slice("text", "notes"))
188
+ .to eq("text" => "東京", "notes" => %w[とうきょう Tokyo])
189
+ # (the default path skips articles with empty text; compare what both emit)
190
+ common_titles = turbo.keys & streamed.keys
191
+ expect(common_titles.size).to eq(turbo.size)
192
+ expect(streamed.slice(*common_titles).transform_values { |r| r["lead_terms"] })
193
+ .to eq(turbo.transform_values { |r| r["lead_terms"] })
194
+ end
195
+
196
+ it "requires JSON output for --lead-terms" do
197
+ input = File.join(@dir, "x.xml.bz2")
198
+ File.write(input, "")
199
+ _stdout, stderr, status = Open3.capture3(RbConfig.ruby, "-I", lib, cli, "-i", input, "--lead-terms")
200
+ expect(status.success?).to be(false)
201
+ expect(stderr).to include("--lead-terms requires --format json")
202
+ end
203
+ end
204
+ end
@@ -356,17 +356,20 @@ RSpec.describe "P1 correctness contracts" do
356
356
  Rake.application = previous
357
357
  end
358
358
 
359
- [[false, "Cannot connect to daemon"], [false, "image missing"], [true, "LEAK:/wp2txt/tmp\n"]].each do |success, output|
360
- it "fails verification for #{output.strip}" do
361
- allow(Open3).to receive(:capture2e).and_return([output, double(success?: success)])
362
- expect { suppress_stderr { Rake::Task[:verify_image].invoke("test-image") } }.to raise_error(SystemExit)
363
- end
359
+ it "builds from a clean clone and hands the image to the CI gate with that clone as context" do
360
+ commands = []
361
+ allow(TOPLEVEL_BINDING.receiver).to receive(:sh) { |*args| commands << args }
362
+ Rake::Task[:check_image].invoke
363
+ clone, build, gate = commands
364
+ expect(clone.first(4)).to eq(%w[git clone --quiet --no-local])
365
+ dir = clone.last
366
+ expect(build).to eq(["docker", "build", "-t", "wp2txt-verify:local", dir])
367
+ expect(gate).to eq(["ruby", "scripts/verify_image.rb", "wp2txt-verify:local", "--context", dir])
364
368
  end
365
369
 
366
- it "passes only after a successful clean container inspection" do
367
- expect(Open3).to receive(:capture2e).with("docker", "run", "--rm", "test-image", "sh", "-c", kind_of(String))
368
- .and_return(["", double(success?: true)])
369
- expect { Rake::Task[:verify_image].invoke("test-image") }.to output(/OK:/).to_stdout
370
+ it "no longer carries a hand-written list of forbidden paths" do
371
+ expect(Rake::Task.task_defined?(:verify_image)).to be(false)
372
+ expect(defined?(IMAGE_FORBIDDEN_PATHS)).to be_nil
370
373
  end
371
374
  end
372
375
  end
@@ -0,0 +1,161 @@
1
+ # frozen_string_literal: true
2
+
3
+ require "spec_helper"
4
+ require "wp2txt/page_props_importer"
5
+ require "wp2txt/metadata_index"
6
+ require "wp2txt/corpus"
7
+ require "wp2txt/link_counter"
8
+ require "tmpdir"
9
+ require "open3"
10
+ require "json"
11
+ require "zlib"
12
+ require_relative "support/multistream_fixture"
13
+
14
+ RSpec.describe "page properties" do
15
+ include MultistreamFixture
16
+ TITLES = %w[東京 QID パイプ Sort Invalid Other Empty Absent].freeze
17
+ FIELDS = %w[qid sort_key disambiguation].freeze
18
+
19
+ around do |example|
20
+ Dir.mktmpdir do |dir|
21
+ @dir = dir
22
+ @dump = File.join(dir, "testwiki-20260101-pages-articles-multistream.xml.bz2")
23
+ xml = TITLES.each_with_index.map { |t, i| page_xml(id: i + 1, ns: 0, title: t, text: "'''#{t}'''です。") }.join
24
+ File.binwrite(@dump, bzip2("<mediawiki>\n#{xml}</mediawiki>"))
25
+ File.write(@dump.sub(".xml.bz2", "-index.txt"), TITLES.each_with_index.map { |t, i| "0:#{i + 1}:#{t}\n" }.join)
26
+ @db_path = Wp2txt::MetadataIndex.path_for(@dump, cache_dir: dir)
27
+ Wp2txt::MetadataIndexBuilder.new(@dump, [0], db_path: @db_path, num_processes: 0).build.close
28
+ @source = File.join(dir, "testwiki-20260101-page_props.sql.gz")
29
+ @sql = <<~SQL
30
+ INSERT INTO `page_props` VALUES
31
+ (1,'defaultsort','とうきよう',NULL),
32
+ (1,'disambiguation','ignored',NULL),
33
+ (1,'wikibase_item','Q1490',NULL),
34
+ (2,'wikibase_item','Q2',NULL),
35
+ (3,'disambiguation','',NULL),
36
+ (4,'defaultsort','O\\'Brien あ',NULL),
37
+ (5,'defaultsort','\xFF\xFE',NULL),
38
+ (6,'other','ignored',NULL),
39
+ (7,'defaultsort','',NULL),
40
+ (8,'wikibase_item','bad',NULL);
41
+ INSERT INTO `other_table` VALUES (8,'wikibase_item','Q999',NULL);
42
+ SQL
43
+ Zlib::GzipWriter.open(@source) { |gz| gz.write(@sql.b) }
44
+ example.run
45
+ end
46
+ end
47
+
48
+ def import(**options)
49
+ Wp2txt::PagePropsImporter.new(@db_path).import!(@source, **options)
50
+ end
51
+
52
+ def with_db
53
+ db = SQLite3::Database.new(@db_path)
54
+ yield db
55
+ ensure
56
+ db&.close
57
+ end
58
+
59
+ it "merges all three properties across batches and records counts and invalid sort keys" do
60
+ stub_const("Wp2txt::PagePropsImporter::BATCH_SIZE", 1)
61
+ result = import
62
+ expect(result[:row_count]).to eq(5)
63
+ with_db do |db|
64
+ expect(db.execute("SELECT page_id,qid,disambiguation,sort_key FROM page_properties ORDER BY page_id"))
65
+ .to eq([[1, "Q1490", 1, "とうきよう"], [2, "Q2", 0, nil], [3, nil, 1, nil],
66
+ [4, nil, 0, "O'Brien あ"], [7, nil, 0, ""]])
67
+ end
68
+ expect(result[:provenance]).to include(page_count: 5, qid_count: 2, disambiguation_count: 2,
69
+ sort_key_count: 3, skipped_invalid_sort_keys: 1,
70
+ source_sha256: Digest::SHA256.file(@source).hexdigest)
71
+ expect(import).to include(status: :already_imported, row_count: 5, provenance: result[:provenance])
72
+ end
73
+
74
+ it "replaces the unreleased QID table without requiring force" do
75
+ with_db do |db|
76
+ db.execute("CREATE TABLE page_qids (page_id INTEGER PRIMARY KEY, qid TEXT)")
77
+ db.execute("INSERT INTO metadata VALUES ('page_props_imported_at','old')")
78
+ end
79
+ expect(import[:status]).to eq(:imported)
80
+ with_db { |db| expect(db.get_first_value("SELECT 1 FROM sqlite_master WHERE name='page_qids'")).to be_nil }
81
+ end
82
+
83
+ def records(*flags)
84
+ out = Dir.mktmpdir("out-", @dir)
85
+ _, stderr, status = Open3.capture3(RbConfig.ruby, "-I", File.expand_path("../lib", __dir__),
86
+ File.expand_path("../bin/wp2txt", __dir__), "-i", @dump,
87
+ "--cache-dir", @dir, "--format", "json", "-n", "2", "-o", out, *flags)
88
+ expect(status.success?).to be(true), stderr
89
+ Dir[File.join(out, "*.jsonl")].flat_map { |f| File.readlines(f).map { |s| JSON.parse(s) } }.to_h { |r| [r["title"], r] }
90
+ end
91
+
92
+ it "always emits the three fields after import on both CLI paths, including null and false" do
93
+ import
94
+ turbo, stream = records, records("--no-turbo")
95
+ expect(turbo.size).to eq(8)
96
+ expect(turbo).to eq(stream)
97
+ turbo.each_value { |r| expect(r.keys.first(6)).to eq(%w[title page_id revision_id qid sort_key disambiguation]) }
98
+ expect(turbo["東京"].slice(*FIELDS)).to eq("qid" => "Q1490", "sort_key" => "とうきよう", "disambiguation" => true)
99
+ expect(turbo["パイプ"].slice(*FIELDS)).to eq("qid" => nil, "sort_key" => nil, "disambiguation" => true)
100
+ expect(turbo["Absent"].slice(*FIELDS)).to eq("qid" => nil, "sort_key" => nil, "disambiguation" => false)
101
+ expect(turbo["Empty"]["sort_key"]).to eq("")
102
+ end
103
+
104
+ it "omits all three fields before import and after a failed force import" do
105
+ [false, true].each do |fail_import|
106
+ if fail_import
107
+ import
108
+ Zlib::GzipWriter.open(@source) { |gz| gz.write("-- empty\n") }
109
+ expect { import(force: true) }.to raise_error(Wp2txt::Error, /no page_props rows/)
110
+ end
111
+ [[], ["--no-turbo"]].each do |flags|
112
+ records(*flags).each_value { |r| expect(r.keys & FIELDS).to eq([]) }
113
+ end
114
+ end
115
+ end
116
+
117
+ it "attaches properties to Ractor JSON in the parent without omitting nulls" do
118
+ import
119
+ result = records("--no-turbo", "--ractor")
120
+ expect(result["東京"].slice(*FIELDS)).to eq("qid" => "Q1490", "sort_key" => "とうきよう", "disambiguation" => true)
121
+ expect(result["Absent"].slice(*FIELDS)).to eq("qid" => nil, "sort_key" => nil, "disambiguation" => false)
122
+ end
123
+
124
+ it "distinguishes an imported dump with no relevant properties from an unimported dump" do
125
+ Zlib::GzipWriter.open(@source) { |gz| gz.write("INSERT INTO `page_props` VALUES (1,'other','',NULL);\n") }
126
+ expect(import[:row_count]).to eq(0)
127
+ expect(records["東京"].slice(*FIELDS)).to eq("qid" => nil, "sort_key" => nil, "disambiguation" => false)
128
+ end
129
+
130
+ it "applies the same contract to MCP's Corpus methods and provenance" do
131
+ corpus = Wp2txt::Corpus.for_input(@dump, cache_dir: @dir)
132
+ expect(corpus.get_article("東京").keys & FIELDS.map(&:to_sym)).to eq([])
133
+ unimported = File.join(@dir, "unimported.jsonl")
134
+ corpus.extract_corpus(output_path: unimported, titles: ["東京", "Absent"], content: "full", num_processes: 0)
135
+ File.readlines(unimported).each { |line| expect(JSON.parse(line).keys & FIELDS).to eq([]) }
136
+ corpus.close
137
+ import
138
+ corpus = Wp2txt::Corpus.for_input(@dump, cache_dir: @dir)
139
+ expect(corpus.get_article("東京")).to include(qid: "Q1490", sort_key: "とうきよう", disambiguation: true)
140
+ expect(corpus.get_article("Absent")).to include(qid: nil, sort_key: nil, disambiguation: false)
141
+ output = File.join(@dir, "corpus.jsonl")
142
+ corpus.extract_corpus(output_path: output, titles: ["東京", "パイプ", "Absent"], content: "full", num_processes: 0)
143
+ rows = File.readlines(output).map { |s| JSON.parse(s) }
144
+ expect(rows.size).to eq(3)
145
+ rows.each { |r| expect(r.keys & FIELDS).to match_array(FIELDS) }
146
+ expect(rows.last.slice(*FIELDS)).to eq("qid" => nil, "sort_key" => nil, "disambiguation" => false)
147
+ expect(corpus.dump_info[:page_properties]).to include(qid_count: 2, disambiguation_count: 2, sort_key_count: 3,
148
+ skipped_invalid_sort_keys: 1)
149
+ ensure
150
+ corpus&.close
151
+ end
152
+
153
+ it "decodes title references once, including escaped ampersands and numeric references" do
154
+ counter = Wp2txt::LinkCounter.new("", [], db_path: "")
155
+ counter.instance_variable_set(:@redirects, {})
156
+ direct, via = Hash.new(0), Hash.new(0)
157
+ text = "[[A&amp;amp;B]] [[Caf&#233;]] [[東京&#35;節]] [[X&amp;lt;Y]]"
158
+ counter.send(:count_page, page_xml(id: 1, ns: 0, title: "Source", text: text), direct, via)
159
+ expect(direct).to eq("A&amp;B" => 1, "Café" => 1, "東京" => 1, "X&lt;Y" => 1)
160
+ end
161
+ end
@@ -0,0 +1,71 @@
1
+ # frozen_string_literal: true
2
+
3
+ require "spec_helper"
4
+ require "wp2txt/lead_terms"
5
+ require "wp2txt/link_counter"
6
+ require_relative "support/multistream_fixture"
7
+
8
+ RSpec.describe "literal regions and lead prose" do
9
+ include MultistreamFixture
10
+
11
+ def terms(text)
12
+ render = ->(s) { Object.new.extend(Wp2txt).format_wiki(s, expand_templates: true, markers: [:all]) }
13
+ Wp2txt::LeadTerms.extract(text, render: render)
14
+ end
15
+
16
+ def counts(text)
17
+ counter = Wp2txt::LinkCounter.new("", [], db_path: "")
18
+ counter.instance_variable_set(:@redirects, {})
19
+ direct, via = Hash.new(0), Hash.new(0)
20
+ counter.send(:count_page, page_xml(id: 1, ns: 0, title: "Source", text: text), direct, via)
21
+ direct
22
+ end
23
+
24
+ %w[gallery timeline].each do |tag|
25
+ it "counts links inside #{tag}, but excludes its bold text from the lead" do
26
+ text = "<#{tag}>[[東京]] '''偽'''(にせ)</#{tag}>\n\n'''大阪'''(おおさか)。"
27
+ expect(counts(text)).to eq("東京" => 1)
28
+ expect(terms(text).map { |term| term["text"] }).to eq(["大阪"])
29
+ expect(counts("<#{tag}>[[東京]]<nowiki>[[京都]]</nowiki>[[大阪]]</#{tag}>"))
30
+ .to eq("東京" => 1, "大阪" => 1)
31
+ end
32
+ end
33
+
34
+ it "counts links and extracts bold terms inside code as ordinary wikitext" do
35
+ text = "<code>'''東京'''(とうきょう) [[大阪]]</code>"
36
+ expect(counts(text)).to eq("大阪" => 1)
37
+ term = terms(text).first
38
+ expect(term).not_to be_nil
39
+ expect(term.slice("text", "notes")).to eq("text" => "東京", "notes" => ["とうきょう"])
40
+ s, e = term["span"]["bold"]
41
+ expect(text[s...e]).to eq("'''東京'''")
42
+ s, e = term["span"]["paren"]
43
+ expect(text[s...e]).to eq("(とうきょう)")
44
+ end
45
+
46
+ %w[nowiki pre math chem ce score syntaxhighlight source graph mapframe templatedata].each do |tag|
47
+ it "excludes literal #{tag} contents from both consumers and protects its pipes" do
48
+ region = "<#{tag}>[[東京]] '''偽'''(にせ)A|B</#{tag}>"
49
+ expect(counts(region + "[[大阪]]")).to eq("大阪" => 1)
50
+ expect(terms(region + "\n\n'''大阪'''(おおさか)。").map { |term| term["text"] }).to eq(["大阪"])
51
+ expect(Wp2txt::WikitextRegions.split_pipes("ruby|#{region}|よみ")).to eq(["ruby", region, "よみ"])
52
+ end
53
+ end
54
+
55
+ %w[gallery timeline code].each do |tag|
56
+ it "splits template pipes inside #{tag} while still protecting nested literal regions" do
57
+ expect(Wp2txt::WikitextRegions.split_pipes("ruby|<#{tag}>A|B</#{tag}>|よみ"))
58
+ .to eq(["ruby", "<#{tag}>A", "B</#{tag}>", "よみ"])
59
+ expect(Wp2txt::WikitextRegions.split_pipes("ruby|<#{tag}>A<nowiki>|</nowiki>B|C</#{tag}>|よみ"))
60
+ .to eq(["ruby", "<#{tag}>A<nowiki>|</nowiki>B", "C</#{tag}>", "よみ"])
61
+ end
62
+ end
63
+
64
+ it "continues to exclude comments and keeps rule version 2" do
65
+ expect(counts("<!-- [[東京]] -->[[大阪]]")).to eq("大阪" => 1)
66
+ expect(terms("<!-- '''偽''' -->'''大阪'''(おおさか)").first["text"]).to eq("大阪")
67
+ expect(Wp2txt::WikitextRegions.split_pipes("ruby|<!-- a|b -->東京|よみ"))
68
+ .to eq(["ruby", "<!-- a|b -->東京", "よみ"])
69
+ expect(Wp2txt::LinkCounter::RULE_VERSION).to eq("2")
70
+ end
71
+ end