wp2txt 2.3.3 → 2.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/.dockerignore +0 -4
- data/.gitignore +3 -5
- data/CHANGELOG.md +18 -0
- data/DEVELOPMENT.md +1 -1
- data/DEVELOPMENT_ja.md +1 -1
- data/README.md +44 -2
- data/README_ja.md +35 -2
- data/Rakefile +10 -21
- data/bin/wp2txt +79 -17
- data/bin/wp2txt-mcp +1 -1
- data/docs/INDEXES.md +61 -1
- data/lib/wp2txt/article.rb +1 -1
- data/lib/wp2txt/cli.rb +42 -0
- data/lib/wp2txt/constants.rb +24 -0
- data/lib/wp2txt/corpus.rb +11 -2
- data/lib/wp2txt/data/template_aliases.json +1 -1
- data/lib/wp2txt/extractor.rb +10 -1
- data/lib/wp2txt/formatter.rb +20 -0
- data/lib/wp2txt/index_commands.rb +89 -1
- data/lib/wp2txt/langlinks_importer.rb +19 -35
- data/lib/wp2txt/lead_terms.rb +228 -0
- data/lib/wp2txt/link_counter.rb +171 -0
- data/lib/wp2txt/metadata_index.rb +79 -3
- data/lib/wp2txt/multistream.rb +52 -2
- data/lib/wp2txt/output_writer.rb +8 -0
- data/lib/wp2txt/page_props_importer.rb +170 -0
- data/lib/wp2txt/sql_dump_reader.rb +57 -0
- data/lib/wp2txt/stream_processor.rb +42 -15
- data/lib/wp2txt/template_expander.rb +19 -0
- data/lib/wp2txt/utils.rb +5 -3
- data/lib/wp2txt/version.rb +1 -1
- data/lib/wp2txt/wikitext_regions.rb +66 -0
- data/lib/wp2txt.rb +7 -5
- data/spec/docs_sync_spec.rb +3 -3
- data/spec/langlinks_importer_spec.rb +27 -0
- data/spec/lead_terms_edge_cases_spec.rb +194 -0
- data/spec/lead_terms_links_qids_spec.rb +204 -0
- data/spec/output_integrity_spec.rb +146 -0
- data/spec/p1_correctness_spec.rb +32 -14
- data/spec/page_properties_spec.rb +161 -0
- data/spec/region_semantics_spec.rb +71 -0
- data/spec/template_passthrough_spec.rb +43 -0
- data/spec/titles_output_path_spec.rb +12 -0
- metadata +18 -1
|
@@ -0,0 +1,204 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "spec_helper"
|
|
4
|
+
require "json"
|
|
5
|
+
require "open3"
|
|
6
|
+
require "tmpdir"
|
|
7
|
+
require "zlib"
|
|
8
|
+
require_relative "support/multistream_fixture"
|
|
9
|
+
require "wp2txt"
|
|
10
|
+
require "wp2txt/metadata_index"
|
|
11
|
+
require "wp2txt/multistream"
|
|
12
|
+
require "wp2txt/sql_dump_reader"
|
|
13
|
+
require "wp2txt/page_props_importer"
|
|
14
|
+
require "wp2txt/link_counter"
|
|
15
|
+
require "wp2txt/lead_terms"
|
|
16
|
+
|
|
17
|
+
RSpec.describe "lead terms, incoming links, and Wikidata IDs" do
|
|
18
|
+
include MultistreamFixture
|
|
19
|
+
|
|
20
|
+
around do |example|
|
|
21
|
+
Dir.mktmpdir("wp2txt-ltq-") do |dir|
|
|
22
|
+
@dir = dir
|
|
23
|
+
example.run
|
|
24
|
+
end
|
|
25
|
+
end
|
|
26
|
+
|
|
27
|
+
def xml_page(id:, title:, text:, ns: 0)
|
|
28
|
+
esc = ->(s) { s.gsub("&", "&").gsub("<", "<").gsub(">", ">") }
|
|
29
|
+
"<page>\n<title>#{esc.(title)}</title>\n<ns>#{ns}</ns>\n<id>#{id}</id>\n<revision>\n<id>#{id * 100}</id>\n" \
|
|
30
|
+
"<text bytes=\"#{text.bytesize}\">#{esc.(text)}</text>\n</revision>\n</page>\n"
|
|
31
|
+
end
|
|
32
|
+
|
|
33
|
+
PAGES = [
|
|
34
|
+
[1, "東京", "'''東京'''(とうきょう、Tokyo)は首都。\n\n== 歴史 ==\n'''後'''(あと)", 0],
|
|
35
|
+
[2, "東京都区部", "#REDIRECT [[東京]]", 0],
|
|
36
|
+
[3, "記事A", "[[東京]]と[[東京]]、[[東京都区部]]。[[M&A|買収]]。<!-- [[隠れ]] -->[[#節]][[]]", 0],
|
|
37
|
+
[4, "記事B", "[[東京#歴史|東京の歴史]]", 0],
|
|
38
|
+
[5, "記事C", "[[東京都区部]]のみ。", 0],
|
|
39
|
+
[6, "M&A", "'''{{読み仮名|合併|がっぺい}}'''と買収。", 0],
|
|
40
|
+
[7, "隠れ", "", 0],
|
|
41
|
+
[8, "Category:X", "[[東京]]", 14]
|
|
42
|
+
].freeze
|
|
43
|
+
|
|
44
|
+
# One-stream dump with its index and a built metadata index
|
|
45
|
+
def build_dump
|
|
46
|
+
dump = File.join(@dir, "testwiki-20260101-pages-articles-multistream.xml.bz2")
|
|
47
|
+
File.binwrite(dump, bzip2(PAGES.map { |id, t, x, ns| xml_page(id: id, title: t, text: x, ns: ns) }.join))
|
|
48
|
+
File.write(File.join(@dir, "testwiki-20260101-pages-articles-multistream-index.txt"),
|
|
49
|
+
PAGES.map { |id, t, _, _| "0:#{id}:#{t}" }.join("\n") + "\n")
|
|
50
|
+
db = Wp2txt::MetadataIndex.path_for(dump, cache_dir: @dir)
|
|
51
|
+
Wp2txt::MetadataIndexBuilder.new(dump, [0], db_path: db, num_processes: 0).build.close
|
|
52
|
+
[dump, db]
|
|
53
|
+
end
|
|
54
|
+
|
|
55
|
+
def page_props_file(content, name: "testwiki-20260101-page_props.sql.gz")
|
|
56
|
+
path = File.join(@dir, name)
|
|
57
|
+
Zlib::GzipWriter.open(path) { |gz| gz.write(content) }
|
|
58
|
+
path
|
|
59
|
+
end
|
|
60
|
+
|
|
61
|
+
PAGE_PROPS_SQL = <<~SQL
|
|
62
|
+
INSERT INTO `page_props` VALUES
|
|
63
|
+
(1,'wikibase_item','Q1490',NULL),
|
|
64
|
+
(1,'page_image_free','T\\'kyo.jpg',NULL),
|
|
65
|
+
(3,'wikibase_item','Q9',1.5),
|
|
66
|
+
(6,'wikibase_item','Q1\\'bad',NULL),
|
|
67
|
+
(6,'defaultsort','\xFF\xFE',NULL);
|
|
68
|
+
INSERT INTO `page_restrictions` VALUES
|
|
69
|
+
(1,'wikibase_item','Q777',NULL);
|
|
70
|
+
SQL
|
|
71
|
+
|
|
72
|
+
describe Wp2txt::SqlDumpReader do
|
|
73
|
+
it "reads statements written on one line and one tuple per line, skipping other tables" do
|
|
74
|
+
lines = []
|
|
75
|
+
path = page_props_file("INSERT INTO `t` VALUES (1),(2);\nINSERT INTO `u` VALUES\n(9);\nINSERT INTO `t` VALUES\n(3),\n(4);\n")
|
|
76
|
+
described_class.each_insert_line(path, "t") { |l| lines << l.strip }
|
|
77
|
+
expect(lines).to eq(["INSERT INTO `t` VALUES (1),(2);", "INSERT INTO `t` VALUES", "(3),", "(4);"])
|
|
78
|
+
end
|
|
79
|
+
end
|
|
80
|
+
|
|
81
|
+
describe Wp2txt::PagePropsImporter do
|
|
82
|
+
it "imports only well-formed wikibase_item values, and records where they came from" do
|
|
83
|
+
_dump, db = build_dump
|
|
84
|
+
path = page_props_file(PAGE_PROPS_SQL.b)
|
|
85
|
+
result = described_class.new(db).import!(path)
|
|
86
|
+
expect(result[:row_count]).to eq(2)
|
|
87
|
+
rows = SQLite3::Database.new(db, readonly: true).execute("SELECT page_id, qid FROM page_properties ORDER BY page_id")
|
|
88
|
+
expect(rows).to eq([[1, "Q1490"], [3, "Q9"]])
|
|
89
|
+
expect(result[:provenance][:source_sha256]).to eq(Digest::SHA256.file(path).hexdigest)
|
|
90
|
+
expect(described_class.new(db).import!(path)[:status]).to eq(:already_imported)
|
|
91
|
+
end
|
|
92
|
+
|
|
93
|
+
it "refuses a dump of another date" do
|
|
94
|
+
_dump, db = build_dump
|
|
95
|
+
path = page_props_file(PAGE_PROPS_SQL.b, name: "testwiki-20250101-page_props.sql.gz")
|
|
96
|
+
expect { described_class.new(db).import!(path) }.to raise_error(ArgumentError, /version mismatch/)
|
|
97
|
+
end
|
|
98
|
+
|
|
99
|
+
it "fails when nothing could be read" do
|
|
100
|
+
_dump, db = build_dump
|
|
101
|
+
path = page_props_file("-- empty\n")
|
|
102
|
+
expect { described_class.new(db).import!(path) }.to raise_error(Wp2txt::Error, /no page_props rows/)
|
|
103
|
+
end
|
|
104
|
+
end
|
|
105
|
+
|
|
106
|
+
describe Wp2txt::LinkCounter do
|
|
107
|
+
it "counts each linking article once, through redirects, ignoring comments and non-articles" do
|
|
108
|
+
dump, db = build_dump
|
|
109
|
+
described_class.new(dump, [0], db_path: db, num_processes: 0).count!
|
|
110
|
+
counts = SQLite3::Database.new(db, readonly: true)
|
|
111
|
+
.execute("SELECT p.title, i.inlinks, i.via_redirects FROM page_inlinks i " \
|
|
112
|
+
"JOIN pages p USING (page_id)").to_h { |t, n, v| [t, [n, v]] }
|
|
113
|
+
expect(counts["東京"]).to eq([3, 1]) # 記事A and 記事B directly, 記事C only via the redirect
|
|
114
|
+
expect(counts["M&A"]).to eq([1, 0])
|
|
115
|
+
expect(counts["隠れ"]).to eq([0, 0]) # linked only from a comment
|
|
116
|
+
expect(counts["記事A"]).to eq([0, 0])
|
|
117
|
+
expect(counts).not_to have_key("東京都区部") # redirects get no row of their own
|
|
118
|
+
end
|
|
119
|
+
end
|
|
120
|
+
|
|
121
|
+
describe Wp2txt::LeadTerms do
|
|
122
|
+
let(:render) { ->(fragment) { Object.new.extend(Wp2txt).format_wiki(fragment, { expand_templates: true, markers: [:all] }) } }
|
|
123
|
+
|
|
124
|
+
def terms(text)
|
|
125
|
+
Wp2txt::LeadTerms.extract(text, render: render)
|
|
126
|
+
end
|
|
127
|
+
|
|
128
|
+
it "takes bold terms of the first paragraph that has one, skipping templates, files, refs, and comments" do
|
|
129
|
+
text = "{{Otheruses|'''x'''}}\n[[ファイル:A.jpg|thumb|'''偽''']]<!-- '''隠''' --><ref>'''注'''</ref>\n" \
|
|
130
|
+
"'''日本語'''(にほんご、にっぽんご{{Refnest|注}})は言語。'''和語'''とも。\n\n次の段落の'''別'''。"
|
|
131
|
+
expect(terms(text).map { |t| [t["text"], t["notes"]] }).to eq([["日本語", %w[にほんご にっぽんご]], ["和語", []]])
|
|
132
|
+
end
|
|
133
|
+
|
|
134
|
+
it "reports spans that cut the bold and the parentheses out of the original text" do
|
|
135
|
+
text = "前置き。'''東京''' (とうきょう)は首都。"
|
|
136
|
+
term = terms(text).first
|
|
137
|
+
expect(text[Range.new(*term["span"]["bold"], true)]).to eq("'''東京'''")
|
|
138
|
+
expect(text[Range.new(*term["span"]["paren"], true)]).to eq("(とうきょう)")
|
|
139
|
+
expect(term["notes_text"]).to eq("とうきょう")
|
|
140
|
+
end
|
|
141
|
+
|
|
142
|
+
it "does not split inside nested brackets, templates, or links" do
|
|
143
|
+
text = "'''山野太郎'''(やまの たろう、[[2001年]](平成13年、辛巳)[[4月1日]] - 、{{lang|en|a, b}})は架空の人物。"
|
|
144
|
+
expect(terms(text).first["notes"]).to eq(["やまの たろう", "2001年(平成13年、辛巳)4月1日 -", "a, b"])
|
|
145
|
+
end
|
|
146
|
+
|
|
147
|
+
it "reports reading templates as pairs and keeps their reading out of the bold text" do
|
|
148
|
+
result = terms("'''{{読み仮名|言語|げんご}}'''は記号体系。")
|
|
149
|
+
expect(result.map { |t| t.slice("text", "reading", "source") })
|
|
150
|
+
.to eq([{ "text" => "言語", "source" => "bold" },
|
|
151
|
+
{ "text" => "言語", "reading" => "げんご", "source" => "読み仮名" }])
|
|
152
|
+
end
|
|
153
|
+
|
|
154
|
+
it "stops at the first heading, ignores an unclosed bracket, and caps the count" do
|
|
155
|
+
expect(terms("'''甲'''(こう\n\n== 節 ==\n'''乙'''").map { |t| [t["text"], t["notes"]] }).to eq([["甲", []]])
|
|
156
|
+
many = (1..8).map { |i| "'''語#{i}'''" }.join("、")
|
|
157
|
+
expect(terms(many).map { |t| t["index"] }).to eq([0, 1, 2, 3, 4])
|
|
158
|
+
end
|
|
159
|
+
|
|
160
|
+
it "returns nothing for empty text" do
|
|
161
|
+
expect(terms("")).to eq([])
|
|
162
|
+
end
|
|
163
|
+
end
|
|
164
|
+
|
|
165
|
+
describe "command line" do
|
|
166
|
+
let(:cli) { File.expand_path("../bin/wp2txt", __dir__) }
|
|
167
|
+
let(:lib) { File.expand_path("../lib", __dir__) }
|
|
168
|
+
|
|
169
|
+
def records(*args)
|
|
170
|
+
out = File.join(@dir, "out#{args.hash.abs}")
|
|
171
|
+
Dir.mkdir(out)
|
|
172
|
+
_stdout, stderr, status = Open3.capture3(RbConfig.ruby, "-I", lib, cli, *args, "-o", out)
|
|
173
|
+
expect(status.success?).to be(true), stderr
|
|
174
|
+
Dir[File.join(out, "*")].flat_map { |f| File.readlines(f) }.map { |l| JSON.parse(l) }.to_h { |r| [r["title"], r] }
|
|
175
|
+
end
|
|
176
|
+
|
|
177
|
+
it "adds the Wikidata ID and lead terms to JSON, identically on both extraction paths" do
|
|
178
|
+
dump, db = build_dump
|
|
179
|
+
Wp2txt::PagePropsImporter.new(db).import!(page_props_file(PAGE_PROPS_SQL.b))
|
|
180
|
+
common = ["-i", dump, "--cache-dir", @dir, "--format", "json", "--summary-only", "--lead-terms"]
|
|
181
|
+
turbo = records(*common)
|
|
182
|
+
streamed = records(*common, "--no-turbo")
|
|
183
|
+
|
|
184
|
+
expect(turbo["東京"].keys.first(4)).to eq(%w[title page_id revision_id qid])
|
|
185
|
+
expect(turbo["東京"]["qid"]).to eq("Q1490")
|
|
186
|
+
expect(turbo["記事B"].slice("qid", "sort_key", "disambiguation")).to eq("qid" => nil, "sort_key" => nil, "disambiguation" => false)
|
|
187
|
+
expect(turbo["東京"]["lead_terms"].first.slice("text", "notes"))
|
|
188
|
+
.to eq("text" => "東京", "notes" => %w[とうきょう Tokyo])
|
|
189
|
+
# (the default path skips articles with empty text; compare what both emit)
|
|
190
|
+
common_titles = turbo.keys & streamed.keys
|
|
191
|
+
expect(common_titles.size).to eq(turbo.size)
|
|
192
|
+
expect(streamed.slice(*common_titles).transform_values { |r| r["lead_terms"] })
|
|
193
|
+
.to eq(turbo.transform_values { |r| r["lead_terms"] })
|
|
194
|
+
end
|
|
195
|
+
|
|
196
|
+
it "requires JSON output for --lead-terms" do
|
|
197
|
+
input = File.join(@dir, "x.xml.bz2")
|
|
198
|
+
File.write(input, "")
|
|
199
|
+
_stdout, stderr, status = Open3.capture3(RbConfig.ruby, "-I", lib, cli, "-i", input, "--lead-terms")
|
|
200
|
+
expect(status.success?).to be(false)
|
|
201
|
+
expect(stderr).to include("--lead-terms requires --format json")
|
|
202
|
+
end
|
|
203
|
+
end
|
|
204
|
+
end
|
|
@@ -0,0 +1,146 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "spec_helper"
|
|
4
|
+
require "open3"
|
|
5
|
+
require "json"
|
|
6
|
+
require "tmpdir"
|
|
7
|
+
require "parallel"
|
|
8
|
+
require_relative "support/multistream_fixture"
|
|
9
|
+
require "wp2txt/multistream"
|
|
10
|
+
require "wp2txt/output_writer"
|
|
11
|
+
require "wp2txt/stream_processor"
|
|
12
|
+
|
|
13
|
+
RSpec.describe "output integrity" do
|
|
14
|
+
include MultistreamFixture
|
|
15
|
+
|
|
16
|
+
around do |example|
|
|
17
|
+
Dir.mktmpdir("wp2txt-integrity-") do |dir|
|
|
18
|
+
@dir = dir
|
|
19
|
+
example.run
|
|
20
|
+
end
|
|
21
|
+
end
|
|
22
|
+
|
|
23
|
+
let(:cli) { File.expand_path("../bin/wp2txt", __dir__) }
|
|
24
|
+
let(:lib) { File.expand_path("../lib", __dir__) }
|
|
25
|
+
|
|
26
|
+
def run_cli(*args)
|
|
27
|
+
Open3.capture3(RbConfig.ruby, "-I", lib, cli, *args)
|
|
28
|
+
end
|
|
29
|
+
|
|
30
|
+
def json_records(dir)
|
|
31
|
+
Dir[File.join(dir, "*")].flat_map { |f| File.readlines(f) }.map { |line| JSON.parse(line) }
|
|
32
|
+
end
|
|
33
|
+
|
|
34
|
+
def write_pages(path, count)
|
|
35
|
+
body = (1..count).map { |i| page_xml(id: i, ns: 0, title: "記事#{i}", text: "本文#{i}。日本語の文。\n") }.join
|
|
36
|
+
File.write(path, "<mediawiki>\n#{body}</mediawiki>\n")
|
|
37
|
+
end
|
|
38
|
+
|
|
39
|
+
describe "forked workers" do
|
|
40
|
+
it "do not write the parent's buffered output again when they exit" do
|
|
41
|
+
writer = Wp2txt::OutputWriter.new(output_dir: @dir, base_name: "out", format: :text, file_size_mb: 10)
|
|
42
|
+
writer.write("one line still sitting in the buffer")
|
|
43
|
+
writer.flush
|
|
44
|
+
Parallel.map([1, 2, 3], in_processes: 3) { |x| x }
|
|
45
|
+
files = writer.close
|
|
46
|
+
expect(File.readlines(files.first).size).to eq(1)
|
|
47
|
+
end
|
|
48
|
+
|
|
49
|
+
it "leave every article exactly once in streaming output across batch boundaries" do
|
|
50
|
+
input = File.join(@dir, "pages.xml")
|
|
51
|
+
out = File.join(@dir, "out")
|
|
52
|
+
Dir.mkdir(out)
|
|
53
|
+
write_pages(input, 450) # batches of 200 with -n 2
|
|
54
|
+
_stdout, stderr, status = run_cli("-i", input, "-o", out, "--format", "json", "-n", "2")
|
|
55
|
+
expect(status.success?).to be(true), stderr
|
|
56
|
+
|
|
57
|
+
titles = json_records(out).map { |r| r["title"] }
|
|
58
|
+
expect(titles.size).to eq(450)
|
|
59
|
+
expect(titles.tally.select { |_, n| n > 1 }).to be_empty
|
|
60
|
+
end
|
|
61
|
+
end
|
|
62
|
+
|
|
63
|
+
describe "--num-procs" do
|
|
64
|
+
it "uses the requested process count" do
|
|
65
|
+
input = File.join(@dir, "pages.xml")
|
|
66
|
+
out = File.join(@dir, "out")
|
|
67
|
+
Dir.mkdir(out)
|
|
68
|
+
write_pages(input, 3)
|
|
69
|
+
stdout, stderr, status = run_cli("-i", input, "-o", out, "--format", "json", "-n", "1")
|
|
70
|
+
expect(status.success?).to be(true), stderr
|
|
71
|
+
expect(stdout.gsub(/\e\[[\d;]*m/, "")).to match(/CPU cores:\s*1\b/)
|
|
72
|
+
end
|
|
73
|
+
end
|
|
74
|
+
|
|
75
|
+
describe "page and revision IDs" do
|
|
76
|
+
it "appear in JSON records from the streaming path, including summaries" do
|
|
77
|
+
input = File.join(@dir, "pages.xml")
|
|
78
|
+
write_pages(input, 3)
|
|
79
|
+
[[], ["--summary-only"]].each do |extra|
|
|
80
|
+
out = File.join(@dir, "out#{extra.size}")
|
|
81
|
+
Dir.mkdir(out)
|
|
82
|
+
_stdout, stderr, status = run_cli("-i", input, "-o", out, "--format", "json", *extra)
|
|
83
|
+
expect(status.success?).to be(true), stderr
|
|
84
|
+
record = json_records(out).find { |r| r["title"] == "記事2" }
|
|
85
|
+
expect(record).to include("page_id" => 2, "revision_id" => 200)
|
|
86
|
+
expect(record.keys.first(3)).to eq(%w[title page_id revision_id])
|
|
87
|
+
end
|
|
88
|
+
end
|
|
89
|
+
|
|
90
|
+
it "appear in JSON records from the default path for bz2 dumps" do
|
|
91
|
+
dump, = create_fixture(@dir)
|
|
92
|
+
out = File.join(@dir, "out")
|
|
93
|
+
Dir.mkdir(out)
|
|
94
|
+
_stdout, stderr, status = run_cli("-i", dump, "-o", out, "--format", "json")
|
|
95
|
+
expect(status.success?).to be(true), stderr
|
|
96
|
+
record = json_records(out).find { |r| r["title"] == "Film B" }
|
|
97
|
+
expect(record).to include("page_id" => 2, "revision_id" => 200)
|
|
98
|
+
end
|
|
99
|
+
|
|
100
|
+
it "are read from the page header, not from the contributor" do
|
|
101
|
+
xml = "<page><title>X</title><ns>0</ns><id>12</id><revision><id>34</id>" \
|
|
102
|
+
"<contributor><id>9</id></contributor><text>t</text></revision></page>"
|
|
103
|
+
expect(Wp2txt.page_ids(xml)).to eq(page_id: 12, revision_id: 34)
|
|
104
|
+
end
|
|
105
|
+
|
|
106
|
+
it "are offered by StreamProcessor only on request, keeping each_page's shape" do
|
|
107
|
+
input = File.join(@dir, "pages.xml")
|
|
108
|
+
write_pages(input, 1)
|
|
109
|
+
processor = Wp2txt::StreamProcessor.new(input, adaptive_buffer: false)
|
|
110
|
+
expect(processor.each_page.to_a).to eq([["記事1", "本文1。日本語の文。\n"]])
|
|
111
|
+
with_ids = Wp2txt::StreamProcessor.new(input, adaptive_buffer: false).each_page(with_ids: true).to_a
|
|
112
|
+
expect(with_ids.first.last).to eq(page_id: 1, revision_id: 100)
|
|
113
|
+
end
|
|
114
|
+
end
|
|
115
|
+
|
|
116
|
+
describe "targeted extraction after an early index stop" do
|
|
117
|
+
it "knows where the last found article's stream ends" do
|
|
118
|
+
_dump, index_path = create_fixture(@dir)
|
|
119
|
+
index = Wp2txt::MultistreamIndex.new(index_path, use_cache: false, target_titles: ["Film A"],
|
|
120
|
+
show_progress: false)
|
|
121
|
+
expect(index.early_terminated?).to be(true)
|
|
122
|
+
expect(index.stream_offsets).to eq([0])
|
|
123
|
+
second_stream = File.readlines(index_path).map { |line| line.split(":").first.to_i }.uniq[1]
|
|
124
|
+
expect(index.stream_end_offset).to eq(second_stream)
|
|
125
|
+
end
|
|
126
|
+
|
|
127
|
+
it "reads only that stream" do
|
|
128
|
+
dump, index_path = create_fixture(@dir)
|
|
129
|
+
index = Wp2txt::MultistreamIndex.new(index_path, use_cache: false, target_titles: ["Film A"],
|
|
130
|
+
show_progress: false)
|
|
131
|
+
reader = Wp2txt::MultistreamReader.new(dump, index)
|
|
132
|
+
stub_const("Wp2txt::MultistreamReader::MAX_TAIL_STREAM_BYTES", 0)
|
|
133
|
+
expect(reader.extract_article("Film A")).to include(title: "Film A", id: 1, revision_id: 100)
|
|
134
|
+
end
|
|
135
|
+
|
|
136
|
+
it "refuses to read a large remainder in one call when the end is unknown" do
|
|
137
|
+
dump, index_path = create_fixture(@dir)
|
|
138
|
+
index = Wp2txt::MultistreamIndex.new(index_path, use_cache: false, target_titles: ["Film A"],
|
|
139
|
+
show_progress: false)
|
|
140
|
+
index.instance_variable_set(:@stream_end_offset, nil)
|
|
141
|
+
reader = Wp2txt::MultistreamReader.new(dump, index)
|
|
142
|
+
stub_const("Wp2txt::MultistreamReader::MAX_TAIL_STREAM_BYTES", 0)
|
|
143
|
+
expect { reader.extract_article("Film A") }.to raise_error(Wp2txt::Error, /cannot locate the end/)
|
|
144
|
+
end
|
|
145
|
+
end
|
|
146
|
+
end
|
data/spec/p1_correctness_spec.rb
CHANGED
|
@@ -40,10 +40,25 @@ RSpec.describe "P1 correctness contracts" do
|
|
|
40
40
|
end
|
|
41
41
|
end
|
|
42
42
|
|
|
43
|
-
it "
|
|
44
|
-
processor = processor_for("
|
|
45
|
-
nil while processor.send(:fill_buffer)
|
|
46
|
-
|
|
43
|
+
it "stops on an invalid byte and reports where it is" do
|
|
44
|
+
processor = processor_for("éあ\xFF😀".b, 1)
|
|
45
|
+
expect { nil while processor.send(:fill_buffer) }
|
|
46
|
+
.to raise_error(Wp2txt::EncodingError, /byte 5\b/)
|
|
47
|
+
end
|
|
48
|
+
|
|
49
|
+
it "stops when the input ends partway through a character" do
|
|
50
|
+
processor = processor_for("éあ\xE3\x81".b, 1)
|
|
51
|
+
expect { nil while processor.send(:fill_buffer) }
|
|
52
|
+
.to raise_error(Wp2txt::EncodingError, /middle of a UTF-8 character/)
|
|
53
|
+
end
|
|
54
|
+
|
|
55
|
+
it "passes valid text through unchanged at every buffer size" do
|
|
56
|
+
text = "é\nあ😀終\n" * 7
|
|
57
|
+
[1, 2, 3, 5, 64].each do |size|
|
|
58
|
+
processor = processor_for(text.b, size)
|
|
59
|
+
nil while processor.send(:fill_buffer)
|
|
60
|
+
expect(processor.instance_variable_get(:@buffer)).to eq(text)
|
|
61
|
+
end
|
|
47
62
|
end
|
|
48
63
|
|
|
49
64
|
it "handles empty input" do
|
|
@@ -105,7 +120,7 @@ RSpec.describe "P1 correctness contracts" do
|
|
|
105
120
|
it "retains the historical ns=0 default for missing ns in both parsers" do
|
|
106
121
|
xml = page_xml(id: 1, ns: 0, title: "作品: 東京", text: "本文").sub(/<ns>.*?<\/ns>/, "")
|
|
107
122
|
processor = Wp2txt::StreamProcessor.new("unused.xml", adaptive_buffer: false)
|
|
108
|
-
expect(processor.send(:parse_page_xml, xml)).to eq(["作品: 東京", "本文"])
|
|
123
|
+
expect(processor.send(:parse_page_xml, xml)).to eq(["作品: 東京", "本文", { page_id: 1, revision_id: 100 }])
|
|
109
124
|
rows = { pages: [], categories: [], sections: [], hierarchy: [] }
|
|
110
125
|
Wp2txt::MetadataIndexBuilder.scan_page(xml, rows)
|
|
111
126
|
expect(rows[:pages].first[2]).to eq(0)
|
|
@@ -341,17 +356,20 @@ RSpec.describe "P1 correctness contracts" do
|
|
|
341
356
|
Rake.application = previous
|
|
342
357
|
end
|
|
343
358
|
|
|
344
|
-
|
|
345
|
-
|
|
346
|
-
|
|
347
|
-
|
|
348
|
-
|
|
359
|
+
it "builds from a clean clone and hands the image to the CI gate with that clone as context" do
|
|
360
|
+
commands = []
|
|
361
|
+
allow(TOPLEVEL_BINDING.receiver).to receive(:sh) { |*args| commands << args }
|
|
362
|
+
Rake::Task[:check_image].invoke
|
|
363
|
+
clone, build, gate = commands
|
|
364
|
+
expect(clone.first(4)).to eq(%w[git clone --quiet --no-local])
|
|
365
|
+
dir = clone.last
|
|
366
|
+
expect(build).to eq(["docker", "build", "-t", "wp2txt-verify:local", dir])
|
|
367
|
+
expect(gate).to eq(["ruby", "scripts/verify_image.rb", "wp2txt-verify:local", "--context", dir])
|
|
349
368
|
end
|
|
350
369
|
|
|
351
|
-
it "
|
|
352
|
-
expect(
|
|
353
|
-
|
|
354
|
-
expect { Rake::Task[:verify_image].invoke("test-image") }.to output(/OK:/).to_stdout
|
|
370
|
+
it "no longer carries a hand-written list of forbidden paths" do
|
|
371
|
+
expect(Rake::Task.task_defined?(:verify_image)).to be(false)
|
|
372
|
+
expect(defined?(IMAGE_FORBIDDEN_PATHS)).to be_nil
|
|
355
373
|
end
|
|
356
374
|
end
|
|
357
375
|
end
|
|
@@ -0,0 +1,161 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "spec_helper"
|
|
4
|
+
require "wp2txt/page_props_importer"
|
|
5
|
+
require "wp2txt/metadata_index"
|
|
6
|
+
require "wp2txt/corpus"
|
|
7
|
+
require "wp2txt/link_counter"
|
|
8
|
+
require "tmpdir"
|
|
9
|
+
require "open3"
|
|
10
|
+
require "json"
|
|
11
|
+
require "zlib"
|
|
12
|
+
require_relative "support/multistream_fixture"
|
|
13
|
+
|
|
14
|
+
RSpec.describe "page properties" do
|
|
15
|
+
include MultistreamFixture
|
|
16
|
+
TITLES = %w[東京 QID パイプ Sort Invalid Other Empty Absent].freeze
|
|
17
|
+
FIELDS = %w[qid sort_key disambiguation].freeze
|
|
18
|
+
|
|
19
|
+
around do |example|
|
|
20
|
+
Dir.mktmpdir do |dir|
|
|
21
|
+
@dir = dir
|
|
22
|
+
@dump = File.join(dir, "testwiki-20260101-pages-articles-multistream.xml.bz2")
|
|
23
|
+
xml = TITLES.each_with_index.map { |t, i| page_xml(id: i + 1, ns: 0, title: t, text: "'''#{t}'''です。") }.join
|
|
24
|
+
File.binwrite(@dump, bzip2("<mediawiki>\n#{xml}</mediawiki>"))
|
|
25
|
+
File.write(@dump.sub(".xml.bz2", "-index.txt"), TITLES.each_with_index.map { |t, i| "0:#{i + 1}:#{t}\n" }.join)
|
|
26
|
+
@db_path = Wp2txt::MetadataIndex.path_for(@dump, cache_dir: dir)
|
|
27
|
+
Wp2txt::MetadataIndexBuilder.new(@dump, [0], db_path: @db_path, num_processes: 0).build.close
|
|
28
|
+
@source = File.join(dir, "testwiki-20260101-page_props.sql.gz")
|
|
29
|
+
@sql = <<~SQL
|
|
30
|
+
INSERT INTO `page_props` VALUES
|
|
31
|
+
(1,'defaultsort','とうきよう',NULL),
|
|
32
|
+
(1,'disambiguation','ignored',NULL),
|
|
33
|
+
(1,'wikibase_item','Q1490',NULL),
|
|
34
|
+
(2,'wikibase_item','Q2',NULL),
|
|
35
|
+
(3,'disambiguation','',NULL),
|
|
36
|
+
(4,'defaultsort','O\\'Brien あ',NULL),
|
|
37
|
+
(5,'defaultsort','\xFF\xFE',NULL),
|
|
38
|
+
(6,'other','ignored',NULL),
|
|
39
|
+
(7,'defaultsort','',NULL),
|
|
40
|
+
(8,'wikibase_item','bad',NULL);
|
|
41
|
+
INSERT INTO `other_table` VALUES (8,'wikibase_item','Q999',NULL);
|
|
42
|
+
SQL
|
|
43
|
+
Zlib::GzipWriter.open(@source) { |gz| gz.write(@sql.b) }
|
|
44
|
+
example.run
|
|
45
|
+
end
|
|
46
|
+
end
|
|
47
|
+
|
|
48
|
+
def import(**options)
|
|
49
|
+
Wp2txt::PagePropsImporter.new(@db_path).import!(@source, **options)
|
|
50
|
+
end
|
|
51
|
+
|
|
52
|
+
def with_db
|
|
53
|
+
db = SQLite3::Database.new(@db_path)
|
|
54
|
+
yield db
|
|
55
|
+
ensure
|
|
56
|
+
db&.close
|
|
57
|
+
end
|
|
58
|
+
|
|
59
|
+
it "merges all three properties across batches and records counts and invalid sort keys" do
|
|
60
|
+
stub_const("Wp2txt::PagePropsImporter::BATCH_SIZE", 1)
|
|
61
|
+
result = import
|
|
62
|
+
expect(result[:row_count]).to eq(5)
|
|
63
|
+
with_db do |db|
|
|
64
|
+
expect(db.execute("SELECT page_id,qid,disambiguation,sort_key FROM page_properties ORDER BY page_id"))
|
|
65
|
+
.to eq([[1, "Q1490", 1, "とうきよう"], [2, "Q2", 0, nil], [3, nil, 1, nil],
|
|
66
|
+
[4, nil, 0, "O'Brien あ"], [7, nil, 0, ""]])
|
|
67
|
+
end
|
|
68
|
+
expect(result[:provenance]).to include(page_count: 5, qid_count: 2, disambiguation_count: 2,
|
|
69
|
+
sort_key_count: 3, skipped_invalid_sort_keys: 1,
|
|
70
|
+
source_sha256: Digest::SHA256.file(@source).hexdigest)
|
|
71
|
+
expect(import).to include(status: :already_imported, row_count: 5, provenance: result[:provenance])
|
|
72
|
+
end
|
|
73
|
+
|
|
74
|
+
it "replaces the unreleased QID table without requiring force" do
|
|
75
|
+
with_db do |db|
|
|
76
|
+
db.execute("CREATE TABLE page_qids (page_id INTEGER PRIMARY KEY, qid TEXT)")
|
|
77
|
+
db.execute("INSERT INTO metadata VALUES ('page_props_imported_at','old')")
|
|
78
|
+
end
|
|
79
|
+
expect(import[:status]).to eq(:imported)
|
|
80
|
+
with_db { |db| expect(db.get_first_value("SELECT 1 FROM sqlite_master WHERE name='page_qids'")).to be_nil }
|
|
81
|
+
end
|
|
82
|
+
|
|
83
|
+
def records(*flags)
|
|
84
|
+
out = Dir.mktmpdir("out-", @dir)
|
|
85
|
+
_, stderr, status = Open3.capture3(RbConfig.ruby, "-I", File.expand_path("../lib", __dir__),
|
|
86
|
+
File.expand_path("../bin/wp2txt", __dir__), "-i", @dump,
|
|
87
|
+
"--cache-dir", @dir, "--format", "json", "-n", "2", "-o", out, *flags)
|
|
88
|
+
expect(status.success?).to be(true), stderr
|
|
89
|
+
Dir[File.join(out, "*.jsonl")].flat_map { |f| File.readlines(f).map { |s| JSON.parse(s) } }.to_h { |r| [r["title"], r] }
|
|
90
|
+
end
|
|
91
|
+
|
|
92
|
+
it "always emits the three fields after import on both CLI paths, including null and false" do
|
|
93
|
+
import
|
|
94
|
+
turbo, stream = records, records("--no-turbo")
|
|
95
|
+
expect(turbo.size).to eq(8)
|
|
96
|
+
expect(turbo).to eq(stream)
|
|
97
|
+
turbo.each_value { |r| expect(r.keys.first(6)).to eq(%w[title page_id revision_id qid sort_key disambiguation]) }
|
|
98
|
+
expect(turbo["東京"].slice(*FIELDS)).to eq("qid" => "Q1490", "sort_key" => "とうきよう", "disambiguation" => true)
|
|
99
|
+
expect(turbo["パイプ"].slice(*FIELDS)).to eq("qid" => nil, "sort_key" => nil, "disambiguation" => true)
|
|
100
|
+
expect(turbo["Absent"].slice(*FIELDS)).to eq("qid" => nil, "sort_key" => nil, "disambiguation" => false)
|
|
101
|
+
expect(turbo["Empty"]["sort_key"]).to eq("")
|
|
102
|
+
end
|
|
103
|
+
|
|
104
|
+
it "omits all three fields before import and after a failed force import" do
|
|
105
|
+
[false, true].each do |fail_import|
|
|
106
|
+
if fail_import
|
|
107
|
+
import
|
|
108
|
+
Zlib::GzipWriter.open(@source) { |gz| gz.write("-- empty\n") }
|
|
109
|
+
expect { import(force: true) }.to raise_error(Wp2txt::Error, /no page_props rows/)
|
|
110
|
+
end
|
|
111
|
+
[[], ["--no-turbo"]].each do |flags|
|
|
112
|
+
records(*flags).each_value { |r| expect(r.keys & FIELDS).to eq([]) }
|
|
113
|
+
end
|
|
114
|
+
end
|
|
115
|
+
end
|
|
116
|
+
|
|
117
|
+
it "attaches properties to Ractor JSON in the parent without omitting nulls" do
|
|
118
|
+
import
|
|
119
|
+
result = records("--no-turbo", "--ractor")
|
|
120
|
+
expect(result["東京"].slice(*FIELDS)).to eq("qid" => "Q1490", "sort_key" => "とうきよう", "disambiguation" => true)
|
|
121
|
+
expect(result["Absent"].slice(*FIELDS)).to eq("qid" => nil, "sort_key" => nil, "disambiguation" => false)
|
|
122
|
+
end
|
|
123
|
+
|
|
124
|
+
it "distinguishes an imported dump with no relevant properties from an unimported dump" do
|
|
125
|
+
Zlib::GzipWriter.open(@source) { |gz| gz.write("INSERT INTO `page_props` VALUES (1,'other','',NULL);\n") }
|
|
126
|
+
expect(import[:row_count]).to eq(0)
|
|
127
|
+
expect(records["東京"].slice(*FIELDS)).to eq("qid" => nil, "sort_key" => nil, "disambiguation" => false)
|
|
128
|
+
end
|
|
129
|
+
|
|
130
|
+
it "applies the same contract to MCP's Corpus methods and provenance" do
|
|
131
|
+
corpus = Wp2txt::Corpus.for_input(@dump, cache_dir: @dir)
|
|
132
|
+
expect(corpus.get_article("東京").keys & FIELDS.map(&:to_sym)).to eq([])
|
|
133
|
+
unimported = File.join(@dir, "unimported.jsonl")
|
|
134
|
+
corpus.extract_corpus(output_path: unimported, titles: ["東京", "Absent"], content: "full", num_processes: 0)
|
|
135
|
+
File.readlines(unimported).each { |line| expect(JSON.parse(line).keys & FIELDS).to eq([]) }
|
|
136
|
+
corpus.close
|
|
137
|
+
import
|
|
138
|
+
corpus = Wp2txt::Corpus.for_input(@dump, cache_dir: @dir)
|
|
139
|
+
expect(corpus.get_article("東京")).to include(qid: "Q1490", sort_key: "とうきよう", disambiguation: true)
|
|
140
|
+
expect(corpus.get_article("Absent")).to include(qid: nil, sort_key: nil, disambiguation: false)
|
|
141
|
+
output = File.join(@dir, "corpus.jsonl")
|
|
142
|
+
corpus.extract_corpus(output_path: output, titles: ["東京", "パイプ", "Absent"], content: "full", num_processes: 0)
|
|
143
|
+
rows = File.readlines(output).map { |s| JSON.parse(s) }
|
|
144
|
+
expect(rows.size).to eq(3)
|
|
145
|
+
rows.each { |r| expect(r.keys & FIELDS).to match_array(FIELDS) }
|
|
146
|
+
expect(rows.last.slice(*FIELDS)).to eq("qid" => nil, "sort_key" => nil, "disambiguation" => false)
|
|
147
|
+
expect(corpus.dump_info[:page_properties]).to include(qid_count: 2, disambiguation_count: 2, sort_key_count: 3,
|
|
148
|
+
skipped_invalid_sort_keys: 1)
|
|
149
|
+
ensure
|
|
150
|
+
corpus&.close
|
|
151
|
+
end
|
|
152
|
+
|
|
153
|
+
it "decodes title references once, including escaped ampersands and numeric references" do
|
|
154
|
+
counter = Wp2txt::LinkCounter.new("", [], db_path: "")
|
|
155
|
+
counter.instance_variable_set(:@redirects, {})
|
|
156
|
+
direct, via = Hash.new(0), Hash.new(0)
|
|
157
|
+
text = "[[A&amp;B]] [[Café]] [[東京#節]] [[X&lt;Y]]"
|
|
158
|
+
counter.send(:count_page, page_xml(id: 1, ns: 0, title: "Source", text: text), direct, via)
|
|
159
|
+
expect(direct).to eq("A&B" => 1, "Café" => 1, "東京" => 1, "X<Y" => 1)
|
|
160
|
+
end
|
|
161
|
+
end
|
|
@@ -0,0 +1,71 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "spec_helper"
|
|
4
|
+
require "wp2txt/lead_terms"
|
|
5
|
+
require "wp2txt/link_counter"
|
|
6
|
+
require_relative "support/multistream_fixture"
|
|
7
|
+
|
|
8
|
+
RSpec.describe "literal regions and lead prose" do
|
|
9
|
+
include MultistreamFixture
|
|
10
|
+
|
|
11
|
+
def terms(text)
|
|
12
|
+
render = ->(s) { Object.new.extend(Wp2txt).format_wiki(s, expand_templates: true, markers: [:all]) }
|
|
13
|
+
Wp2txt::LeadTerms.extract(text, render: render)
|
|
14
|
+
end
|
|
15
|
+
|
|
16
|
+
def counts(text)
|
|
17
|
+
counter = Wp2txt::LinkCounter.new("", [], db_path: "")
|
|
18
|
+
counter.instance_variable_set(:@redirects, {})
|
|
19
|
+
direct, via = Hash.new(0), Hash.new(0)
|
|
20
|
+
counter.send(:count_page, page_xml(id: 1, ns: 0, title: "Source", text: text), direct, via)
|
|
21
|
+
direct
|
|
22
|
+
end
|
|
23
|
+
|
|
24
|
+
%w[gallery timeline].each do |tag|
|
|
25
|
+
it "counts links inside #{tag}, but excludes its bold text from the lead" do
|
|
26
|
+
text = "<#{tag}>[[東京]] '''偽'''(にせ)</#{tag}>\n\n'''大阪'''(おおさか)。"
|
|
27
|
+
expect(counts(text)).to eq("東京" => 1)
|
|
28
|
+
expect(terms(text).map { |term| term["text"] }).to eq(["大阪"])
|
|
29
|
+
expect(counts("<#{tag}>[[東京]]<nowiki>[[京都]]</nowiki>[[大阪]]</#{tag}>"))
|
|
30
|
+
.to eq("東京" => 1, "大阪" => 1)
|
|
31
|
+
end
|
|
32
|
+
end
|
|
33
|
+
|
|
34
|
+
it "counts links and extracts bold terms inside code as ordinary wikitext" do
|
|
35
|
+
text = "<code>'''東京'''(とうきょう) [[大阪]]</code>"
|
|
36
|
+
expect(counts(text)).to eq("大阪" => 1)
|
|
37
|
+
term = terms(text).first
|
|
38
|
+
expect(term).not_to be_nil
|
|
39
|
+
expect(term.slice("text", "notes")).to eq("text" => "東京", "notes" => ["とうきょう"])
|
|
40
|
+
s, e = term["span"]["bold"]
|
|
41
|
+
expect(text[s...e]).to eq("'''東京'''")
|
|
42
|
+
s, e = term["span"]["paren"]
|
|
43
|
+
expect(text[s...e]).to eq("(とうきょう)")
|
|
44
|
+
end
|
|
45
|
+
|
|
46
|
+
%w[nowiki pre math chem ce score syntaxhighlight source graph mapframe templatedata].each do |tag|
|
|
47
|
+
it "excludes literal #{tag} contents from both consumers and protects its pipes" do
|
|
48
|
+
region = "<#{tag}>[[東京]] '''偽'''(にせ)A|B</#{tag}>"
|
|
49
|
+
expect(counts(region + "[[大阪]]")).to eq("大阪" => 1)
|
|
50
|
+
expect(terms(region + "\n\n'''大阪'''(おおさか)。").map { |term| term["text"] }).to eq(["大阪"])
|
|
51
|
+
expect(Wp2txt::WikitextRegions.split_pipes("ruby|#{region}|よみ")).to eq(["ruby", region, "よみ"])
|
|
52
|
+
end
|
|
53
|
+
end
|
|
54
|
+
|
|
55
|
+
%w[gallery timeline code].each do |tag|
|
|
56
|
+
it "splits template pipes inside #{tag} while still protecting nested literal regions" do
|
|
57
|
+
expect(Wp2txt::WikitextRegions.split_pipes("ruby|<#{tag}>A|B</#{tag}>|よみ"))
|
|
58
|
+
.to eq(["ruby", "<#{tag}>A", "B</#{tag}>", "よみ"])
|
|
59
|
+
expect(Wp2txt::WikitextRegions.split_pipes("ruby|<#{tag}>A<nowiki>|</nowiki>B|C</#{tag}>|よみ"))
|
|
60
|
+
.to eq(["ruby", "<#{tag}>A<nowiki>|</nowiki>B", "C</#{tag}>", "よみ"])
|
|
61
|
+
end
|
|
62
|
+
end
|
|
63
|
+
|
|
64
|
+
it "continues to exclude comments and keeps rule version 2" do
|
|
65
|
+
expect(counts("<!-- [[東京]] -->[[大阪]]")).to eq("大阪" => 1)
|
|
66
|
+
expect(terms("<!-- '''偽''' -->'''大阪'''(おおさか)").first["text"]).to eq("大阪")
|
|
67
|
+
expect(Wp2txt::WikitextRegions.split_pipes("ruby|<!-- a|b -->東京|よみ"))
|
|
68
|
+
.to eq(["ruby", "<!-- a|b -->東京", "よみ"])
|
|
69
|
+
expect(Wp2txt::LinkCounter::RULE_VERSION).to eq("2")
|
|
70
|
+
end
|
|
71
|
+
end
|