wp2txt 2.3.3 → 2.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (45) hide show
  1. checksums.yaml +4 -4
  2. data/.dockerignore +0 -4
  3. data/.gitignore +3 -5
  4. data/CHANGELOG.md +18 -0
  5. data/DEVELOPMENT.md +1 -1
  6. data/DEVELOPMENT_ja.md +1 -1
  7. data/README.md +44 -2
  8. data/README_ja.md +35 -2
  9. data/Rakefile +10 -21
  10. data/bin/wp2txt +79 -17
  11. data/bin/wp2txt-mcp +1 -1
  12. data/docs/INDEXES.md +61 -1
  13. data/lib/wp2txt/article.rb +1 -1
  14. data/lib/wp2txt/cli.rb +42 -0
  15. data/lib/wp2txt/constants.rb +24 -0
  16. data/lib/wp2txt/corpus.rb +11 -2
  17. data/lib/wp2txt/data/template_aliases.json +1 -1
  18. data/lib/wp2txt/extractor.rb +10 -1
  19. data/lib/wp2txt/formatter.rb +20 -0
  20. data/lib/wp2txt/index_commands.rb +89 -1
  21. data/lib/wp2txt/langlinks_importer.rb +19 -35
  22. data/lib/wp2txt/lead_terms.rb +228 -0
  23. data/lib/wp2txt/link_counter.rb +171 -0
  24. data/lib/wp2txt/metadata_index.rb +79 -3
  25. data/lib/wp2txt/multistream.rb +52 -2
  26. data/lib/wp2txt/output_writer.rb +8 -0
  27. data/lib/wp2txt/page_props_importer.rb +170 -0
  28. data/lib/wp2txt/sql_dump_reader.rb +57 -0
  29. data/lib/wp2txt/stream_processor.rb +42 -15
  30. data/lib/wp2txt/template_expander.rb +19 -0
  31. data/lib/wp2txt/utils.rb +5 -3
  32. data/lib/wp2txt/version.rb +1 -1
  33. data/lib/wp2txt/wikitext_regions.rb +66 -0
  34. data/lib/wp2txt.rb +7 -5
  35. data/spec/docs_sync_spec.rb +3 -3
  36. data/spec/langlinks_importer_spec.rb +27 -0
  37. data/spec/lead_terms_edge_cases_spec.rb +194 -0
  38. data/spec/lead_terms_links_qids_spec.rb +204 -0
  39. data/spec/output_integrity_spec.rb +146 -0
  40. data/spec/p1_correctness_spec.rb +32 -14
  41. data/spec/page_properties_spec.rb +161 -0
  42. data/spec/region_semantics_spec.rb +71 -0
  43. data/spec/template_passthrough_spec.rb +43 -0
  44. data/spec/titles_output_path_spec.rb +12 -0
  45. metadata +18 -1
@@ -0,0 +1,204 @@
1
+ # frozen_string_literal: true
2
+
3
+ require "spec_helper"
4
+ require "json"
5
+ require "open3"
6
+ require "tmpdir"
7
+ require "zlib"
8
+ require_relative "support/multistream_fixture"
9
+ require "wp2txt"
10
+ require "wp2txt/metadata_index"
11
+ require "wp2txt/multistream"
12
+ require "wp2txt/sql_dump_reader"
13
+ require "wp2txt/page_props_importer"
14
+ require "wp2txt/link_counter"
15
+ require "wp2txt/lead_terms"
16
+
17
+ RSpec.describe "lead terms, incoming links, and Wikidata IDs" do
18
+ include MultistreamFixture
19
+
20
+ around do |example|
21
+ Dir.mktmpdir("wp2txt-ltq-") do |dir|
22
+ @dir = dir
23
+ example.run
24
+ end
25
+ end
26
+
27
+ def xml_page(id:, title:, text:, ns: 0)
28
+ esc = ->(s) { s.gsub("&", "&amp;").gsub("<", "&lt;").gsub(">", "&gt;") }
29
+ "<page>\n<title>#{esc.(title)}</title>\n<ns>#{ns}</ns>\n<id>#{id}</id>\n<revision>\n<id>#{id * 100}</id>\n" \
30
+ "<text bytes=\"#{text.bytesize}\">#{esc.(text)}</text>\n</revision>\n</page>\n"
31
+ end
32
+
33
+ PAGES = [
34
+ [1, "東京", "'''東京'''(とうきょう、Tokyo)は首都。\n\n== 歴史 ==\n'''後'''(あと)", 0],
35
+ [2, "東京都区部", "#REDIRECT [[東京]]", 0],
36
+ [3, "記事A", "[[東京]]と[[東京]]、[[東京都区部]]。[[M&A|買収]]。<!-- [[隠れ]] -->[[#節]][[]]", 0],
37
+ [4, "記事B", "[[東京#歴史|東京の歴史]]", 0],
38
+ [5, "記事C", "[[東京都区部]]のみ。", 0],
39
+ [6, "M&A", "'''{{読み仮名|合併|がっぺい}}'''と買収。", 0],
40
+ [7, "隠れ", "", 0],
41
+ [8, "Category:X", "[[東京]]", 14]
42
+ ].freeze
43
+
44
+ # One-stream dump with its index and a built metadata index
45
+ def build_dump
46
+ dump = File.join(@dir, "testwiki-20260101-pages-articles-multistream.xml.bz2")
47
+ File.binwrite(dump, bzip2(PAGES.map { |id, t, x, ns| xml_page(id: id, title: t, text: x, ns: ns) }.join))
48
+ File.write(File.join(@dir, "testwiki-20260101-pages-articles-multistream-index.txt"),
49
+ PAGES.map { |id, t, _, _| "0:#{id}:#{t}" }.join("\n") + "\n")
50
+ db = Wp2txt::MetadataIndex.path_for(dump, cache_dir: @dir)
51
+ Wp2txt::MetadataIndexBuilder.new(dump, [0], db_path: db, num_processes: 0).build.close
52
+ [dump, db]
53
+ end
54
+
55
+ def page_props_file(content, name: "testwiki-20260101-page_props.sql.gz")
56
+ path = File.join(@dir, name)
57
+ Zlib::GzipWriter.open(path) { |gz| gz.write(content) }
58
+ path
59
+ end
60
+
61
+ PAGE_PROPS_SQL = <<~SQL
62
+ INSERT INTO `page_props` VALUES
63
+ (1,'wikibase_item','Q1490',NULL),
64
+ (1,'page_image_free','T\\'kyo.jpg',NULL),
65
+ (3,'wikibase_item','Q9',1.5),
66
+ (6,'wikibase_item','Q1\\'bad',NULL),
67
+ (6,'defaultsort','\xFF\xFE',NULL);
68
+ INSERT INTO `page_restrictions` VALUES
69
+ (1,'wikibase_item','Q777',NULL);
70
+ SQL
71
+
72
+ describe Wp2txt::SqlDumpReader do
73
+ it "reads statements written on one line and one tuple per line, skipping other tables" do
74
+ lines = []
75
+ path = page_props_file("INSERT INTO `t` VALUES (1),(2);\nINSERT INTO `u` VALUES\n(9);\nINSERT INTO `t` VALUES\n(3),\n(4);\n")
76
+ described_class.each_insert_line(path, "t") { |l| lines << l.strip }
77
+ expect(lines).to eq(["INSERT INTO `t` VALUES (1),(2);", "INSERT INTO `t` VALUES", "(3),", "(4);"])
78
+ end
79
+ end
80
+
81
+ describe Wp2txt::PagePropsImporter do
82
+ it "imports only well-formed wikibase_item values, and records where they came from" do
83
+ _dump, db = build_dump
84
+ path = page_props_file(PAGE_PROPS_SQL.b)
85
+ result = described_class.new(db).import!(path)
86
+ expect(result[:row_count]).to eq(2)
87
+ rows = SQLite3::Database.new(db, readonly: true).execute("SELECT page_id, qid FROM page_properties ORDER BY page_id")
88
+ expect(rows).to eq([[1, "Q1490"], [3, "Q9"]])
89
+ expect(result[:provenance][:source_sha256]).to eq(Digest::SHA256.file(path).hexdigest)
90
+ expect(described_class.new(db).import!(path)[:status]).to eq(:already_imported)
91
+ end
92
+
93
+ it "refuses a dump of another date" do
94
+ _dump, db = build_dump
95
+ path = page_props_file(PAGE_PROPS_SQL.b, name: "testwiki-20250101-page_props.sql.gz")
96
+ expect { described_class.new(db).import!(path) }.to raise_error(ArgumentError, /version mismatch/)
97
+ end
98
+
99
+ it "fails when nothing could be read" do
100
+ _dump, db = build_dump
101
+ path = page_props_file("-- empty\n")
102
+ expect { described_class.new(db).import!(path) }.to raise_error(Wp2txt::Error, /no page_props rows/)
103
+ end
104
+ end
105
+
106
+ describe Wp2txt::LinkCounter do
107
+ it "counts each linking article once, through redirects, ignoring comments and non-articles" do
108
+ dump, db = build_dump
109
+ described_class.new(dump, [0], db_path: db, num_processes: 0).count!
110
+ counts = SQLite3::Database.new(db, readonly: true)
111
+ .execute("SELECT p.title, i.inlinks, i.via_redirects FROM page_inlinks i " \
112
+ "JOIN pages p USING (page_id)").to_h { |t, n, v| [t, [n, v]] }
113
+ expect(counts["東京"]).to eq([3, 1]) # 記事A and 記事B directly, 記事C only via the redirect
114
+ expect(counts["M&A"]).to eq([1, 0])
115
+ expect(counts["隠れ"]).to eq([0, 0]) # linked only from a comment
116
+ expect(counts["記事A"]).to eq([0, 0])
117
+ expect(counts).not_to have_key("東京都区部") # redirects get no row of their own
118
+ end
119
+ end
120
+
121
+ describe Wp2txt::LeadTerms do
122
+ let(:render) { ->(fragment) { Object.new.extend(Wp2txt).format_wiki(fragment, { expand_templates: true, markers: [:all] }) } }
123
+
124
+ def terms(text)
125
+ Wp2txt::LeadTerms.extract(text, render: render)
126
+ end
127
+
128
+ it "takes bold terms of the first paragraph that has one, skipping templates, files, refs, and comments" do
129
+ text = "{{Otheruses|'''x'''}}\n[[ファイル:A.jpg|thumb|'''偽''']]<!-- '''隠''' --><ref>'''注'''</ref>\n" \
130
+ "'''日本語'''(にほんご、にっぽんご{{Refnest|注}})は言語。'''和語'''とも。\n\n次の段落の'''別'''。"
131
+ expect(terms(text).map { |t| [t["text"], t["notes"]] }).to eq([["日本語", %w[にほんご にっぽんご]], ["和語", []]])
132
+ end
133
+
134
+ it "reports spans that cut the bold and the parentheses out of the original text" do
135
+ text = "前置き。'''東京''' (とうきょう)は首都。"
136
+ term = terms(text).first
137
+ expect(text[Range.new(*term["span"]["bold"], true)]).to eq("'''東京'''")
138
+ expect(text[Range.new(*term["span"]["paren"], true)]).to eq("(とうきょう)")
139
+ expect(term["notes_text"]).to eq("とうきょう")
140
+ end
141
+
142
+ it "does not split inside nested brackets, templates, or links" do
143
+ text = "'''山野太郎'''(やまの たろう、[[2001年]](平成13年、辛巳)[[4月1日]] - 、{{lang|en|a, b}})は架空の人物。"
144
+ expect(terms(text).first["notes"]).to eq(["やまの たろう", "2001年(平成13年、辛巳)4月1日 -", "a, b"])
145
+ end
146
+
147
+ it "reports reading templates as pairs and keeps their reading out of the bold text" do
148
+ result = terms("'''{{読み仮名|言語|げんご}}'''は記号体系。")
149
+ expect(result.map { |t| t.slice("text", "reading", "source") })
150
+ .to eq([{ "text" => "言語", "source" => "bold" },
151
+ { "text" => "言語", "reading" => "げんご", "source" => "読み仮名" }])
152
+ end
153
+
154
+ it "stops at the first heading, ignores an unclosed bracket, and caps the count" do
155
+ expect(terms("'''甲'''(こう\n\n== 節 ==\n'''乙'''").map { |t| [t["text"], t["notes"]] }).to eq([["甲", []]])
156
+ many = (1..8).map { |i| "'''語#{i}'''" }.join("、")
157
+ expect(terms(many).map { |t| t["index"] }).to eq([0, 1, 2, 3, 4])
158
+ end
159
+
160
+ it "returns nothing for empty text" do
161
+ expect(terms("")).to eq([])
162
+ end
163
+ end
164
+
165
+ describe "command line" do
166
+ let(:cli) { File.expand_path("../bin/wp2txt", __dir__) }
167
+ let(:lib) { File.expand_path("../lib", __dir__) }
168
+
169
+ def records(*args)
170
+ out = File.join(@dir, "out#{args.hash.abs}")
171
+ Dir.mkdir(out)
172
+ _stdout, stderr, status = Open3.capture3(RbConfig.ruby, "-I", lib, cli, *args, "-o", out)
173
+ expect(status.success?).to be(true), stderr
174
+ Dir[File.join(out, "*")].flat_map { |f| File.readlines(f) }.map { |l| JSON.parse(l) }.to_h { |r| [r["title"], r] }
175
+ end
176
+
177
+ it "adds the Wikidata ID and lead terms to JSON, identically on both extraction paths" do
178
+ dump, db = build_dump
179
+ Wp2txt::PagePropsImporter.new(db).import!(page_props_file(PAGE_PROPS_SQL.b))
180
+ common = ["-i", dump, "--cache-dir", @dir, "--format", "json", "--summary-only", "--lead-terms"]
181
+ turbo = records(*common)
182
+ streamed = records(*common, "--no-turbo")
183
+
184
+ expect(turbo["東京"].keys.first(4)).to eq(%w[title page_id revision_id qid])
185
+ expect(turbo["東京"]["qid"]).to eq("Q1490")
186
+ expect(turbo["記事B"].slice("qid", "sort_key", "disambiguation")).to eq("qid" => nil, "sort_key" => nil, "disambiguation" => false)
187
+ expect(turbo["東京"]["lead_terms"].first.slice("text", "notes"))
188
+ .to eq("text" => "東京", "notes" => %w[とうきょう Tokyo])
189
+ # (the default path skips articles with empty text; compare what both emit)
190
+ common_titles = turbo.keys & streamed.keys
191
+ expect(common_titles.size).to eq(turbo.size)
192
+ expect(streamed.slice(*common_titles).transform_values { |r| r["lead_terms"] })
193
+ .to eq(turbo.transform_values { |r| r["lead_terms"] })
194
+ end
195
+
196
+ it "requires JSON output for --lead-terms" do
197
+ input = File.join(@dir, "x.xml.bz2")
198
+ File.write(input, "")
199
+ _stdout, stderr, status = Open3.capture3(RbConfig.ruby, "-I", lib, cli, "-i", input, "--lead-terms")
200
+ expect(status.success?).to be(false)
201
+ expect(stderr).to include("--lead-terms requires --format json")
202
+ end
203
+ end
204
+ end
@@ -0,0 +1,146 @@
1
+ # frozen_string_literal: true
2
+
3
+ require "spec_helper"
4
+ require "open3"
5
+ require "json"
6
+ require "tmpdir"
7
+ require "parallel"
8
+ require_relative "support/multistream_fixture"
9
+ require "wp2txt/multistream"
10
+ require "wp2txt/output_writer"
11
+ require "wp2txt/stream_processor"
12
+
13
+ RSpec.describe "output integrity" do
14
+ include MultistreamFixture
15
+
16
+ around do |example|
17
+ Dir.mktmpdir("wp2txt-integrity-") do |dir|
18
+ @dir = dir
19
+ example.run
20
+ end
21
+ end
22
+
23
+ let(:cli) { File.expand_path("../bin/wp2txt", __dir__) }
24
+ let(:lib) { File.expand_path("../lib", __dir__) }
25
+
26
+ def run_cli(*args)
27
+ Open3.capture3(RbConfig.ruby, "-I", lib, cli, *args)
28
+ end
29
+
30
+ def json_records(dir)
31
+ Dir[File.join(dir, "*")].flat_map { |f| File.readlines(f) }.map { |line| JSON.parse(line) }
32
+ end
33
+
34
+ def write_pages(path, count)
35
+ body = (1..count).map { |i| page_xml(id: i, ns: 0, title: "記事#{i}", text: "本文#{i}。日本語の文。\n") }.join
36
+ File.write(path, "<mediawiki>\n#{body}</mediawiki>\n")
37
+ end
38
+
39
+ describe "forked workers" do
40
+ it "do not write the parent's buffered output again when they exit" do
41
+ writer = Wp2txt::OutputWriter.new(output_dir: @dir, base_name: "out", format: :text, file_size_mb: 10)
42
+ writer.write("one line still sitting in the buffer")
43
+ writer.flush
44
+ Parallel.map([1, 2, 3], in_processes: 3) { |x| x }
45
+ files = writer.close
46
+ expect(File.readlines(files.first).size).to eq(1)
47
+ end
48
+
49
+ it "leave every article exactly once in streaming output across batch boundaries" do
50
+ input = File.join(@dir, "pages.xml")
51
+ out = File.join(@dir, "out")
52
+ Dir.mkdir(out)
53
+ write_pages(input, 450) # batches of 200 with -n 2
54
+ _stdout, stderr, status = run_cli("-i", input, "-o", out, "--format", "json", "-n", "2")
55
+ expect(status.success?).to be(true), stderr
56
+
57
+ titles = json_records(out).map { |r| r["title"] }
58
+ expect(titles.size).to eq(450)
59
+ expect(titles.tally.select { |_, n| n > 1 }).to be_empty
60
+ end
61
+ end
62
+
63
+ describe "--num-procs" do
64
+ it "uses the requested process count" do
65
+ input = File.join(@dir, "pages.xml")
66
+ out = File.join(@dir, "out")
67
+ Dir.mkdir(out)
68
+ write_pages(input, 3)
69
+ stdout, stderr, status = run_cli("-i", input, "-o", out, "--format", "json", "-n", "1")
70
+ expect(status.success?).to be(true), stderr
71
+ expect(stdout.gsub(/\e\[[\d;]*m/, "")).to match(/CPU cores:\s*1\b/)
72
+ end
73
+ end
74
+
75
+ describe "page and revision IDs" do
76
+ it "appear in JSON records from the streaming path, including summaries" do
77
+ input = File.join(@dir, "pages.xml")
78
+ write_pages(input, 3)
79
+ [[], ["--summary-only"]].each do |extra|
80
+ out = File.join(@dir, "out#{extra.size}")
81
+ Dir.mkdir(out)
82
+ _stdout, stderr, status = run_cli("-i", input, "-o", out, "--format", "json", *extra)
83
+ expect(status.success?).to be(true), stderr
84
+ record = json_records(out).find { |r| r["title"] == "記事2" }
85
+ expect(record).to include("page_id" => 2, "revision_id" => 200)
86
+ expect(record.keys.first(3)).to eq(%w[title page_id revision_id])
87
+ end
88
+ end
89
+
90
+ it "appear in JSON records from the default path for bz2 dumps" do
91
+ dump, = create_fixture(@dir)
92
+ out = File.join(@dir, "out")
93
+ Dir.mkdir(out)
94
+ _stdout, stderr, status = run_cli("-i", dump, "-o", out, "--format", "json")
95
+ expect(status.success?).to be(true), stderr
96
+ record = json_records(out).find { |r| r["title"] == "Film B" }
97
+ expect(record).to include("page_id" => 2, "revision_id" => 200)
98
+ end
99
+
100
+ it "are read from the page header, not from the contributor" do
101
+ xml = "<page><title>X</title><ns>0</ns><id>12</id><revision><id>34</id>" \
102
+ "<contributor><id>9</id></contributor><text>t</text></revision></page>"
103
+ expect(Wp2txt.page_ids(xml)).to eq(page_id: 12, revision_id: 34)
104
+ end
105
+
106
+ it "are offered by StreamProcessor only on request, keeping each_page's shape" do
107
+ input = File.join(@dir, "pages.xml")
108
+ write_pages(input, 1)
109
+ processor = Wp2txt::StreamProcessor.new(input, adaptive_buffer: false)
110
+ expect(processor.each_page.to_a).to eq([["記事1", "本文1。日本語の文。\n"]])
111
+ with_ids = Wp2txt::StreamProcessor.new(input, adaptive_buffer: false).each_page(with_ids: true).to_a
112
+ expect(with_ids.first.last).to eq(page_id: 1, revision_id: 100)
113
+ end
114
+ end
115
+
116
+ describe "targeted extraction after an early index stop" do
117
+ it "knows where the last found article's stream ends" do
118
+ _dump, index_path = create_fixture(@dir)
119
+ index = Wp2txt::MultistreamIndex.new(index_path, use_cache: false, target_titles: ["Film A"],
120
+ show_progress: false)
121
+ expect(index.early_terminated?).to be(true)
122
+ expect(index.stream_offsets).to eq([0])
123
+ second_stream = File.readlines(index_path).map { |line| line.split(":").first.to_i }.uniq[1]
124
+ expect(index.stream_end_offset).to eq(second_stream)
125
+ end
126
+
127
+ it "reads only that stream" do
128
+ dump, index_path = create_fixture(@dir)
129
+ index = Wp2txt::MultistreamIndex.new(index_path, use_cache: false, target_titles: ["Film A"],
130
+ show_progress: false)
131
+ reader = Wp2txt::MultistreamReader.new(dump, index)
132
+ stub_const("Wp2txt::MultistreamReader::MAX_TAIL_STREAM_BYTES", 0)
133
+ expect(reader.extract_article("Film A")).to include(title: "Film A", id: 1, revision_id: 100)
134
+ end
135
+
136
+ it "refuses to read a large remainder in one call when the end is unknown" do
137
+ dump, index_path = create_fixture(@dir)
138
+ index = Wp2txt::MultistreamIndex.new(index_path, use_cache: false, target_titles: ["Film A"],
139
+ show_progress: false)
140
+ index.instance_variable_set(:@stream_end_offset, nil)
141
+ reader = Wp2txt::MultistreamReader.new(dump, index)
142
+ stub_const("Wp2txt::MultistreamReader::MAX_TAIL_STREAM_BYTES", 0)
143
+ expect { reader.extract_article("Film A") }.to raise_error(Wp2txt::Error, /cannot locate the end/)
144
+ end
145
+ end
146
+ end
@@ -40,10 +40,25 @@ RSpec.describe "P1 correctness contracts" do
40
40
  end
41
41
  end
42
42
 
43
- it "scrubs invalid bytes and incomplete EOF without losing adjacent valid text" do
44
- processor = processor_for("é\xFFあ😀\xE3\x81".b, 1)
45
- nil while processor.send(:fill_buffer)
46
- expect(processor.instance_variable_get(:@buffer)).to eq("éあ😀")
43
+ it "stops on an invalid byte and reports where it is" do
44
+ processor = processor_for("éあ\xFF😀".b, 1)
45
+ expect { nil while processor.send(:fill_buffer) }
46
+ .to raise_error(Wp2txt::EncodingError, /byte 5\b/)
47
+ end
48
+
49
+ it "stops when the input ends partway through a character" do
50
+ processor = processor_for("éあ\xE3\x81".b, 1)
51
+ expect { nil while processor.send(:fill_buffer) }
52
+ .to raise_error(Wp2txt::EncodingError, /middle of a UTF-8 character/)
53
+ end
54
+
55
+ it "passes valid text through unchanged at every buffer size" do
56
+ text = "é\nあ😀終\n" * 7
57
+ [1, 2, 3, 5, 64].each do |size|
58
+ processor = processor_for(text.b, size)
59
+ nil while processor.send(:fill_buffer)
60
+ expect(processor.instance_variable_get(:@buffer)).to eq(text)
61
+ end
47
62
  end
48
63
 
49
64
  it "handles empty input" do
@@ -105,7 +120,7 @@ RSpec.describe "P1 correctness contracts" do
105
120
  it "retains the historical ns=0 default for missing ns in both parsers" do
106
121
  xml = page_xml(id: 1, ns: 0, title: "作品: 東京", text: "本文").sub(/<ns>.*?<\/ns>/, "")
107
122
  processor = Wp2txt::StreamProcessor.new("unused.xml", adaptive_buffer: false)
108
- expect(processor.send(:parse_page_xml, xml)).to eq(["作品: 東京", "本文"])
123
+ expect(processor.send(:parse_page_xml, xml)).to eq(["作品: 東京", "本文", { page_id: 1, revision_id: 100 }])
109
124
  rows = { pages: [], categories: [], sections: [], hierarchy: [] }
110
125
  Wp2txt::MetadataIndexBuilder.scan_page(xml, rows)
111
126
  expect(rows[:pages].first[2]).to eq(0)
@@ -341,17 +356,20 @@ RSpec.describe "P1 correctness contracts" do
341
356
  Rake.application = previous
342
357
  end
343
358
 
344
- [[false, "Cannot connect to daemon"], [false, "image missing"], [true, "LEAK:/wp2txt/tmp\n"]].each do |success, output|
345
- it "fails verification for #{output.strip}" do
346
- allow(Open3).to receive(:capture2e).and_return([output, double(success?: success)])
347
- expect { suppress_stderr { Rake::Task[:verify_image].invoke("test-image") } }.to raise_error(SystemExit)
348
- end
359
+ it "builds from a clean clone and hands the image to the CI gate with that clone as context" do
360
+ commands = []
361
+ allow(TOPLEVEL_BINDING.receiver).to receive(:sh) { |*args| commands << args }
362
+ Rake::Task[:check_image].invoke
363
+ clone, build, gate = commands
364
+ expect(clone.first(4)).to eq(%w[git clone --quiet --no-local])
365
+ dir = clone.last
366
+ expect(build).to eq(["docker", "build", "-t", "wp2txt-verify:local", dir])
367
+ expect(gate).to eq(["ruby", "scripts/verify_image.rb", "wp2txt-verify:local", "--context", dir])
349
368
  end
350
369
 
351
- it "passes only after a successful clean container inspection" do
352
- expect(Open3).to receive(:capture2e).with("docker", "run", "--rm", "test-image", "sh", "-c", kind_of(String))
353
- .and_return(["", double(success?: true)])
354
- expect { Rake::Task[:verify_image].invoke("test-image") }.to output(/OK:/).to_stdout
370
+ it "no longer carries a hand-written list of forbidden paths" do
371
+ expect(Rake::Task.task_defined?(:verify_image)).to be(false)
372
+ expect(defined?(IMAGE_FORBIDDEN_PATHS)).to be_nil
355
373
  end
356
374
  end
357
375
  end
@@ -0,0 +1,161 @@
1
+ # frozen_string_literal: true
2
+
3
+ require "spec_helper"
4
+ require "wp2txt/page_props_importer"
5
+ require "wp2txt/metadata_index"
6
+ require "wp2txt/corpus"
7
+ require "wp2txt/link_counter"
8
+ require "tmpdir"
9
+ require "open3"
10
+ require "json"
11
+ require "zlib"
12
+ require_relative "support/multistream_fixture"
13
+
14
+ RSpec.describe "page properties" do
15
+ include MultistreamFixture
16
+ TITLES = %w[東京 QID パイプ Sort Invalid Other Empty Absent].freeze
17
+ FIELDS = %w[qid sort_key disambiguation].freeze
18
+
19
+ around do |example|
20
+ Dir.mktmpdir do |dir|
21
+ @dir = dir
22
+ @dump = File.join(dir, "testwiki-20260101-pages-articles-multistream.xml.bz2")
23
+ xml = TITLES.each_with_index.map { |t, i| page_xml(id: i + 1, ns: 0, title: t, text: "'''#{t}'''です。") }.join
24
+ File.binwrite(@dump, bzip2("<mediawiki>\n#{xml}</mediawiki>"))
25
+ File.write(@dump.sub(".xml.bz2", "-index.txt"), TITLES.each_with_index.map { |t, i| "0:#{i + 1}:#{t}\n" }.join)
26
+ @db_path = Wp2txt::MetadataIndex.path_for(@dump, cache_dir: dir)
27
+ Wp2txt::MetadataIndexBuilder.new(@dump, [0], db_path: @db_path, num_processes: 0).build.close
28
+ @source = File.join(dir, "testwiki-20260101-page_props.sql.gz")
29
+ @sql = <<~SQL
30
+ INSERT INTO `page_props` VALUES
31
+ (1,'defaultsort','とうきよう',NULL),
32
+ (1,'disambiguation','ignored',NULL),
33
+ (1,'wikibase_item','Q1490',NULL),
34
+ (2,'wikibase_item','Q2',NULL),
35
+ (3,'disambiguation','',NULL),
36
+ (4,'defaultsort','O\\'Brien あ',NULL),
37
+ (5,'defaultsort','\xFF\xFE',NULL),
38
+ (6,'other','ignored',NULL),
39
+ (7,'defaultsort','',NULL),
40
+ (8,'wikibase_item','bad',NULL);
41
+ INSERT INTO `other_table` VALUES (8,'wikibase_item','Q999',NULL);
42
+ SQL
43
+ Zlib::GzipWriter.open(@source) { |gz| gz.write(@sql.b) }
44
+ example.run
45
+ end
46
+ end
47
+
48
+ def import(**options)
49
+ Wp2txt::PagePropsImporter.new(@db_path).import!(@source, **options)
50
+ end
51
+
52
+ def with_db
53
+ db = SQLite3::Database.new(@db_path)
54
+ yield db
55
+ ensure
56
+ db&.close
57
+ end
58
+
59
+ it "merges all three properties across batches and records counts and invalid sort keys" do
60
+ stub_const("Wp2txt::PagePropsImporter::BATCH_SIZE", 1)
61
+ result = import
62
+ expect(result[:row_count]).to eq(5)
63
+ with_db do |db|
64
+ expect(db.execute("SELECT page_id,qid,disambiguation,sort_key FROM page_properties ORDER BY page_id"))
65
+ .to eq([[1, "Q1490", 1, "とうきよう"], [2, "Q2", 0, nil], [3, nil, 1, nil],
66
+ [4, nil, 0, "O'Brien あ"], [7, nil, 0, ""]])
67
+ end
68
+ expect(result[:provenance]).to include(page_count: 5, qid_count: 2, disambiguation_count: 2,
69
+ sort_key_count: 3, skipped_invalid_sort_keys: 1,
70
+ source_sha256: Digest::SHA256.file(@source).hexdigest)
71
+ expect(import).to include(status: :already_imported, row_count: 5, provenance: result[:provenance])
72
+ end
73
+
74
+ it "replaces the unreleased QID table without requiring force" do
75
+ with_db do |db|
76
+ db.execute("CREATE TABLE page_qids (page_id INTEGER PRIMARY KEY, qid TEXT)")
77
+ db.execute("INSERT INTO metadata VALUES ('page_props_imported_at','old')")
78
+ end
79
+ expect(import[:status]).to eq(:imported)
80
+ with_db { |db| expect(db.get_first_value("SELECT 1 FROM sqlite_master WHERE name='page_qids'")).to be_nil }
81
+ end
82
+
83
+ def records(*flags)
84
+ out = Dir.mktmpdir("out-", @dir)
85
+ _, stderr, status = Open3.capture3(RbConfig.ruby, "-I", File.expand_path("../lib", __dir__),
86
+ File.expand_path("../bin/wp2txt", __dir__), "-i", @dump,
87
+ "--cache-dir", @dir, "--format", "json", "-n", "2", "-o", out, *flags)
88
+ expect(status.success?).to be(true), stderr
89
+ Dir[File.join(out, "*.jsonl")].flat_map { |f| File.readlines(f).map { |s| JSON.parse(s) } }.to_h { |r| [r["title"], r] }
90
+ end
91
+
92
+ it "always emits the three fields after import on both CLI paths, including null and false" do
93
+ import
94
+ turbo, stream = records, records("--no-turbo")
95
+ expect(turbo.size).to eq(8)
96
+ expect(turbo).to eq(stream)
97
+ turbo.each_value { |r| expect(r.keys.first(6)).to eq(%w[title page_id revision_id qid sort_key disambiguation]) }
98
+ expect(turbo["東京"].slice(*FIELDS)).to eq("qid" => "Q1490", "sort_key" => "とうきよう", "disambiguation" => true)
99
+ expect(turbo["パイプ"].slice(*FIELDS)).to eq("qid" => nil, "sort_key" => nil, "disambiguation" => true)
100
+ expect(turbo["Absent"].slice(*FIELDS)).to eq("qid" => nil, "sort_key" => nil, "disambiguation" => false)
101
+ expect(turbo["Empty"]["sort_key"]).to eq("")
102
+ end
103
+
104
+ it "omits all three fields before import and after a failed force import" do
105
+ [false, true].each do |fail_import|
106
+ if fail_import
107
+ import
108
+ Zlib::GzipWriter.open(@source) { |gz| gz.write("-- empty\n") }
109
+ expect { import(force: true) }.to raise_error(Wp2txt::Error, /no page_props rows/)
110
+ end
111
+ [[], ["--no-turbo"]].each do |flags|
112
+ records(*flags).each_value { |r| expect(r.keys & FIELDS).to eq([]) }
113
+ end
114
+ end
115
+ end
116
+
117
+ it "attaches properties to Ractor JSON in the parent without omitting nulls" do
118
+ import
119
+ result = records("--no-turbo", "--ractor")
120
+ expect(result["東京"].slice(*FIELDS)).to eq("qid" => "Q1490", "sort_key" => "とうきよう", "disambiguation" => true)
121
+ expect(result["Absent"].slice(*FIELDS)).to eq("qid" => nil, "sort_key" => nil, "disambiguation" => false)
122
+ end
123
+
124
+ it "distinguishes an imported dump with no relevant properties from an unimported dump" do
125
+ Zlib::GzipWriter.open(@source) { |gz| gz.write("INSERT INTO `page_props` VALUES (1,'other','',NULL);\n") }
126
+ expect(import[:row_count]).to eq(0)
127
+ expect(records["東京"].slice(*FIELDS)).to eq("qid" => nil, "sort_key" => nil, "disambiguation" => false)
128
+ end
129
+
130
+ it "applies the same contract to MCP's Corpus methods and provenance" do
131
+ corpus = Wp2txt::Corpus.for_input(@dump, cache_dir: @dir)
132
+ expect(corpus.get_article("東京").keys & FIELDS.map(&:to_sym)).to eq([])
133
+ unimported = File.join(@dir, "unimported.jsonl")
134
+ corpus.extract_corpus(output_path: unimported, titles: ["東京", "Absent"], content: "full", num_processes: 0)
135
+ File.readlines(unimported).each { |line| expect(JSON.parse(line).keys & FIELDS).to eq([]) }
136
+ corpus.close
137
+ import
138
+ corpus = Wp2txt::Corpus.for_input(@dump, cache_dir: @dir)
139
+ expect(corpus.get_article("東京")).to include(qid: "Q1490", sort_key: "とうきよう", disambiguation: true)
140
+ expect(corpus.get_article("Absent")).to include(qid: nil, sort_key: nil, disambiguation: false)
141
+ output = File.join(@dir, "corpus.jsonl")
142
+ corpus.extract_corpus(output_path: output, titles: ["東京", "パイプ", "Absent"], content: "full", num_processes: 0)
143
+ rows = File.readlines(output).map { |s| JSON.parse(s) }
144
+ expect(rows.size).to eq(3)
145
+ rows.each { |r| expect(r.keys & FIELDS).to match_array(FIELDS) }
146
+ expect(rows.last.slice(*FIELDS)).to eq("qid" => nil, "sort_key" => nil, "disambiguation" => false)
147
+ expect(corpus.dump_info[:page_properties]).to include(qid_count: 2, disambiguation_count: 2, sort_key_count: 3,
148
+ skipped_invalid_sort_keys: 1)
149
+ ensure
150
+ corpus&.close
151
+ end
152
+
153
+ it "decodes title references once, including escaped ampersands and numeric references" do
154
+ counter = Wp2txt::LinkCounter.new("", [], db_path: "")
155
+ counter.instance_variable_set(:@redirects, {})
156
+ direct, via = Hash.new(0), Hash.new(0)
157
+ text = "[[A&amp;amp;B]] [[Caf&#233;]] [[東京&#35;節]] [[X&amp;lt;Y]]"
158
+ counter.send(:count_page, page_xml(id: 1, ns: 0, title: "Source", text: text), direct, via)
159
+ expect(direct).to eq("A&amp;B" => 1, "Café" => 1, "東京" => 1, "X&lt;Y" => 1)
160
+ end
161
+ end
@@ -0,0 +1,71 @@
1
+ # frozen_string_literal: true
2
+
3
+ require "spec_helper"
4
+ require "wp2txt/lead_terms"
5
+ require "wp2txt/link_counter"
6
+ require_relative "support/multistream_fixture"
7
+
8
+ RSpec.describe "literal regions and lead prose" do
9
+ include MultistreamFixture
10
+
11
+ def terms(text)
12
+ render = ->(s) { Object.new.extend(Wp2txt).format_wiki(s, expand_templates: true, markers: [:all]) }
13
+ Wp2txt::LeadTerms.extract(text, render: render)
14
+ end
15
+
16
+ def counts(text)
17
+ counter = Wp2txt::LinkCounter.new("", [], db_path: "")
18
+ counter.instance_variable_set(:@redirects, {})
19
+ direct, via = Hash.new(0), Hash.new(0)
20
+ counter.send(:count_page, page_xml(id: 1, ns: 0, title: "Source", text: text), direct, via)
21
+ direct
22
+ end
23
+
24
+ %w[gallery timeline].each do |tag|
25
+ it "counts links inside #{tag}, but excludes its bold text from the lead" do
26
+ text = "<#{tag}>[[東京]] '''偽'''(にせ)</#{tag}>\n\n'''大阪'''(おおさか)。"
27
+ expect(counts(text)).to eq("東京" => 1)
28
+ expect(terms(text).map { |term| term["text"] }).to eq(["大阪"])
29
+ expect(counts("<#{tag}>[[東京]]<nowiki>[[京都]]</nowiki>[[大阪]]</#{tag}>"))
30
+ .to eq("東京" => 1, "大阪" => 1)
31
+ end
32
+ end
33
+
34
+ it "counts links and extracts bold terms inside code as ordinary wikitext" do
35
+ text = "<code>'''東京'''(とうきょう) [[大阪]]</code>"
36
+ expect(counts(text)).to eq("大阪" => 1)
37
+ term = terms(text).first
38
+ expect(term).not_to be_nil
39
+ expect(term.slice("text", "notes")).to eq("text" => "東京", "notes" => ["とうきょう"])
40
+ s, e = term["span"]["bold"]
41
+ expect(text[s...e]).to eq("'''東京'''")
42
+ s, e = term["span"]["paren"]
43
+ expect(text[s...e]).to eq("(とうきょう)")
44
+ end
45
+
46
+ %w[nowiki pre math chem ce score syntaxhighlight source graph mapframe templatedata].each do |tag|
47
+ it "excludes literal #{tag} contents from both consumers and protects its pipes" do
48
+ region = "<#{tag}>[[東京]] '''偽'''(にせ)A|B</#{tag}>"
49
+ expect(counts(region + "[[大阪]]")).to eq("大阪" => 1)
50
+ expect(terms(region + "\n\n'''大阪'''(おおさか)。").map { |term| term["text"] }).to eq(["大阪"])
51
+ expect(Wp2txt::WikitextRegions.split_pipes("ruby|#{region}|よみ")).to eq(["ruby", region, "よみ"])
52
+ end
53
+ end
54
+
55
+ %w[gallery timeline code].each do |tag|
56
+ it "splits template pipes inside #{tag} while still protecting nested literal regions" do
57
+ expect(Wp2txt::WikitextRegions.split_pipes("ruby|<#{tag}>A|B</#{tag}>|よみ"))
58
+ .to eq(["ruby", "<#{tag}>A", "B</#{tag}>", "よみ"])
59
+ expect(Wp2txt::WikitextRegions.split_pipes("ruby|<#{tag}>A<nowiki>|</nowiki>B|C</#{tag}>|よみ"))
60
+ .to eq(["ruby", "<#{tag}>A<nowiki>|</nowiki>B", "C</#{tag}>", "よみ"])
61
+ end
62
+ end
63
+
64
+ it "continues to exclude comments and keeps rule version 2" do
65
+ expect(counts("<!-- [[東京]] -->[[大阪]]")).to eq("大阪" => 1)
66
+ expect(terms("<!-- '''偽''' -->'''大阪'''(おおさか)").first["text"]).to eq("大阪")
67
+ expect(Wp2txt::WikitextRegions.split_pipes("ruby|<!-- a|b -->東京|よみ"))
68
+ .to eq(["ruby", "<!-- a|b -->東京", "よみ"])
69
+ expect(Wp2txt::LinkCounter::RULE_VERSION).to eq("2")
70
+ end
71
+ end