wp2txt 2.3.2 → 2.3.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/.dockerignore +3 -0
- data/.github/workflows/publish-image.yml +140 -0
- data/CHANGELOG.md +23 -0
- data/DEVELOPMENT.md +12 -4
- data/DEVELOPMENT_ja.md +2 -3
- data/README.md +6 -3
- data/README_ja.md +6 -3
- data/Rakefile +22 -16
- data/bin/wp2txt +26 -14
- data/bin/wp2txt-mcp +4 -3
- data/docs/INDEXES.md +4 -0
- data/lib/wp2txt/article.rb +1 -1
- data/lib/wp2txt/constants.rb +16 -0
- data/lib/wp2txt/corpus.rb +98 -79
- data/lib/wp2txt/corpus_jobs.rb +15 -13
- data/lib/wp2txt/extractor.rb +6 -0
- data/lib/wp2txt/formatter.rb +18 -0
- data/lib/wp2txt/fts_index.rb +12 -3
- data/lib/wp2txt/metadata_index.rb +3 -3
- data/lib/wp2txt/multistream.rb +31 -2
- data/lib/wp2txt/output_path.rb +72 -12
- data/lib/wp2txt/output_writer.rb +8 -0
- data/lib/wp2txt/stream_processor.rb +45 -15
- data/lib/wp2txt/version.rb +1 -1
- data/lib/wp2txt.rb +7 -5
- data/spec/fts_index_spec.rb +40 -0
- data/spec/output_integrity_spec.rb +146 -0
- data/spec/p1_correctness_spec.rb +372 -0
- data/spec/stream_processor_spec.rb +3 -1
- data/spec/titles_output_path_spec.rb +61 -1
- data/wp2txt.gemspec +1 -1
- metadata +6 -9
- data/image/wp2txt-logo.svg +0 -16
- data/image/wp2txt.svg +0 -31
- data/scripts/benchmark_regex.rb +0 -161
- data/scripts/fetch_html_entities.rb +0 -94
- data/scripts/fetch_language_metadata.rb +0 -180
- data/scripts/fetch_mediawiki_data.rb +0 -334
- data/scripts/fetch_template_data.rb +0 -186
- data/scripts/profile_memory.rb +0 -139
|
@@ -23,6 +23,7 @@ module Wp2txt
|
|
|
23
23
|
@input_path = input_path
|
|
24
24
|
@bz2_gem = bz2_gem
|
|
25
25
|
@buffer = +""
|
|
26
|
+
@pending_bytes = +"".b
|
|
26
27
|
@file_pointer = nil
|
|
27
28
|
@adaptive_buffer = adaptive_buffer
|
|
28
29
|
@buffer_size = adaptive_buffer ? calculate_optimal_buffer_size : DEFAULT_BUFFER_SIZE
|
|
@@ -63,20 +64,28 @@ module Wp2txt
|
|
|
63
64
|
|
|
64
65
|
# Iterate over each page in the input
|
|
65
66
|
# Yields [title, text] for each page
|
|
66
|
-
|
|
67
|
-
|
|
67
|
+
# Yields title and text of each article. With with_ids: true, also yields
|
|
68
|
+
# { page_id:, revision_id: } taken from the dump.
|
|
69
|
+
def each_page(with_ids: false, &block)
|
|
70
|
+
return enum_for(:each_page, with_ids: with_ids) unless block
|
|
71
|
+
|
|
72
|
+
emit = if with_ids
|
|
73
|
+
->(title, text, ids) { block.call(title, text, ids) }
|
|
74
|
+
else
|
|
75
|
+
->(title, text, _ids) { block.call(title, text) }
|
|
76
|
+
end
|
|
68
77
|
|
|
69
78
|
if File.directory?(@input_path)
|
|
70
79
|
# Process XML files in directory
|
|
71
80
|
Dir.glob(File.join(@input_path, "*.xml")).sort.each do |xml_file|
|
|
72
|
-
process_xml_file(xml_file) { |title, text|
|
|
81
|
+
process_xml_file(xml_file) { |title, text, ids| emit.call(title, text, ids) }
|
|
73
82
|
end
|
|
74
83
|
elsif @input_path.end_with?(".bz2")
|
|
75
84
|
# Process bz2 compressed file with streaming
|
|
76
|
-
process_bz2_stream { |title, text|
|
|
85
|
+
process_bz2_stream { |title, text, ids| emit.call(title, text, ids) }
|
|
77
86
|
elsif @input_path.end_with?(".xml")
|
|
78
87
|
# Process single XML file
|
|
79
|
-
process_xml_file(@input_path) { |title, text|
|
|
88
|
+
process_xml_file(@input_path) { |title, text, ids| emit.call(title, text, ids) }
|
|
80
89
|
else
|
|
81
90
|
raise ArgumentError, "Unsupported input format: #{@input_path}"
|
|
82
91
|
end
|
|
@@ -99,7 +108,8 @@ module Wp2txt
|
|
|
99
108
|
# Process a single XML file
|
|
100
109
|
def process_xml_file(xml_file)
|
|
101
110
|
@buffer = +""
|
|
102
|
-
@
|
|
111
|
+
@pending_bytes = +"".b
|
|
112
|
+
@file_pointer = File.open(xml_file, "rb")
|
|
103
113
|
|
|
104
114
|
while (page = extract_next_page)
|
|
105
115
|
result = parse_page_xml(page)
|
|
@@ -120,6 +130,7 @@ module Wp2txt
|
|
|
120
130
|
end
|
|
121
131
|
|
|
122
132
|
@buffer = +""
|
|
133
|
+
@pending_bytes = +"".b
|
|
123
134
|
@file_pointer = open_bz2_stream
|
|
124
135
|
|
|
125
136
|
while (page = extract_next_page)
|
|
@@ -158,14 +169,33 @@ module Wp2txt
|
|
|
158
169
|
# Fill buffer from file pointer
|
|
159
170
|
def fill_buffer
|
|
160
171
|
chunk = @file_pointer.read(@buffer_size)
|
|
161
|
-
|
|
172
|
+
unless chunk
|
|
173
|
+
unless @pending_bytes.to_s.empty?
|
|
174
|
+
raise Wp2txt::EncodingError,
|
|
175
|
+
"input ends in the middle of a UTF-8 character (byte #{@bytes_read}): #{@input_path}"
|
|
176
|
+
end
|
|
177
|
+
return false
|
|
178
|
+
end
|
|
162
179
|
|
|
180
|
+
carried = @pending_bytes.to_s.b
|
|
181
|
+
start = @bytes_read - carried.bytesize # stream position of the first carried byte
|
|
182
|
+
bytes = carried + chunk.b
|
|
183
|
+
# A read can end partway through a multi-byte character; carry those
|
|
184
|
+
# bytes into the next read rather than judging half a character.
|
|
185
|
+
tail = bytes[/[\xC2-\xF4][\x80-\xBF]{0,2}\z/n]
|
|
186
|
+
width = tail && (tail.getbyte(0) < 0xE0 ? 2 : tail.getbyte(0) < 0xF0 ? 3 : 4)
|
|
187
|
+
@pending_bytes = tail && tail.bytesize < width ? tail : +"".b
|
|
188
|
+
bytes = bytes.byteslice(0, bytes.bytesize - @pending_bytes.bytesize).force_encoding(Encoding::UTF_8)
|
|
189
|
+
# Anything still invalid is corrupt input. Dropping it would change the
|
|
190
|
+
# text without a trace, so stop and say where it is.
|
|
191
|
+
unless bytes.valid_encoding?
|
|
192
|
+
bad = bytes.each_char.find_index { |c| !c.valid_encoding? }.to_i
|
|
193
|
+
offset = start + bytes[0, bad].bytesize
|
|
194
|
+
raise Wp2txt::EncodingError,
|
|
195
|
+
"invalid UTF-8 near decompressed byte #{offset}: #{@input_path}"
|
|
196
|
+
end
|
|
163
197
|
@bytes_read += chunk.bytesize
|
|
164
|
-
|
|
165
|
-
# Handle encoding for bz2 streams
|
|
166
|
-
chunk = chunk.force_encoding("UTF-8")
|
|
167
|
-
chunk = chunk.scrub("")
|
|
168
|
-
@buffer << chunk
|
|
198
|
+
@buffer << bytes
|
|
169
199
|
|
|
170
200
|
# Adaptive buffer adjustment: if memory is low, reduce buffer size
|
|
171
201
|
if @adaptive_buffer && MemoryMonitor.memory_low?
|
|
@@ -223,8 +253,8 @@ module Wp2txt
|
|
|
223
253
|
return nil unless title_node
|
|
224
254
|
|
|
225
255
|
title = title_node.content
|
|
226
|
-
|
|
227
|
-
return nil
|
|
256
|
+
namespace = title_node.parent.at_css("ns")&.text
|
|
257
|
+
return nil unless Wp2txt.namespace_id(namespace).zero?
|
|
228
258
|
|
|
229
259
|
text = text_node.content
|
|
230
260
|
|
|
@@ -242,7 +272,7 @@ module Wp2txt
|
|
|
242
272
|
end
|
|
243
273
|
|
|
244
274
|
@pages_processed += 1
|
|
245
|
-
[title, text]
|
|
275
|
+
[title, text, Wp2txt.page_ids(page_xml)]
|
|
246
276
|
rescue Nokogiri::XML::SyntaxError
|
|
247
277
|
# Skip malformed XML
|
|
248
278
|
nil
|
data/lib/wp2txt/version.rb
CHANGED
data/lib/wp2txt.rb
CHANGED
|
@@ -269,12 +269,14 @@ module Wp2txt
|
|
|
269
269
|
end
|
|
270
270
|
page << line if inside_page
|
|
271
271
|
end
|
|
272
|
-
if page.empty?
|
|
273
|
-
|
|
274
|
-
|
|
275
|
-
|
|
272
|
+
return false if page.empty?
|
|
273
|
+
|
|
274
|
+
page.force_encoding("utf-8")
|
|
275
|
+
# Corrupt input must not be cleaned into different text without a trace
|
|
276
|
+
unless page.valid_encoding?
|
|
277
|
+
title = page[%r{<title>([^<]*)</title>}, 1]&.scrub("?")
|
|
278
|
+
raise Wp2txt::EncodingError, "invalid UTF-8 in page #{title.inspect} of #{@input_file}"
|
|
276
279
|
end
|
|
277
|
-
rescue ::Encoding::InvalidByteSequenceError, ::Encoding::UndefinedConversionError
|
|
278
280
|
page
|
|
279
281
|
end
|
|
280
282
|
|
data/spec/fts_index_spec.rb
CHANGED
|
@@ -171,6 +171,46 @@ RSpec.describe "Wp2txt Full-Text Search" do
|
|
|
171
171
|
end
|
|
172
172
|
end
|
|
173
173
|
|
|
174
|
+
["", "東", "東京", "é", "😀"].each do |query|
|
|
175
|
+
it "rejects the short phrase #{query.inspect} with a stable code" do
|
|
176
|
+
expect { @fts.search(query, count: "exact") }.to raise_error(Wp2txt::FtsIndex::ShortQueryError) { |error|
|
|
177
|
+
expect(error.code).to eq("query_too_short")
|
|
178
|
+
}
|
|
179
|
+
end
|
|
180
|
+
end
|
|
181
|
+
|
|
182
|
+
it "leaves raw FTS syntax to SQLite even for short expressions" do
|
|
183
|
+
expect(@fts.search('"東"', mode: "query", count: "exact")[:total]).to eq(0)
|
|
184
|
+
expect(@fts.search("\xFF".b, mode: "query", count: "exact")[:total]).to eq(0)
|
|
185
|
+
end
|
|
186
|
+
|
|
187
|
+
it "accepts three Unicode characters" do
|
|
188
|
+
expect(@fts.search("東京都", count: "exact")[:total]).to eq(0)
|
|
189
|
+
end
|
|
190
|
+
|
|
191
|
+
it "returns a coded MCP tool error instead of a zero-result success" do
|
|
192
|
+
@fts.close
|
|
193
|
+
requests = [
|
|
194
|
+
{ jsonrpc: "2.0", id: 1, method: "initialize", params: {
|
|
195
|
+
protocolVersion: "2025-03-26", capabilities: {}, clientInfo: { name: "regression", version: "1" }
|
|
196
|
+
} },
|
|
197
|
+
{ jsonrpc: "2.0", method: "notifications/initialized" },
|
|
198
|
+
{ jsonrpc: "2.0", id: 2, method: "tools/call", params: {
|
|
199
|
+
name: "search_text", arguments: { query: "東京", count: "exact" }
|
|
200
|
+
} }
|
|
201
|
+
]
|
|
202
|
+
stdout, stderr, status = Open3.capture3(RbConfig.ruby,
|
|
203
|
+
File.expand_path("../bin/wp2txt-mcp", __dir__), "--input", @multistream_path,
|
|
204
|
+
"--cache-dir", File.dirname(@multistream_path),
|
|
205
|
+
stdin_data: requests.map { |request| JSON.generate(request) }.join("\n") + "\n")
|
|
206
|
+
expect(status.success?).to be(true), stderr
|
|
207
|
+
response = stdout.lines.map { |line| JSON.parse(line) }.find { |message| message["id"] == 2 }
|
|
208
|
+
expect(response.dig("result", "isError")).to be true
|
|
209
|
+
payload = JSON.parse(response.dig("result", "content", 0, "text"))
|
|
210
|
+
expect(payload["code"]).to eq("query_too_short")
|
|
211
|
+
expect(payload).not_to have_key("total")
|
|
212
|
+
end
|
|
213
|
+
|
|
174
214
|
it "matches substrings of three or more characters" do
|
|
175
215
|
result = @fts.search("tory", count: "exact")
|
|
176
216
|
expect(result[:total]).to eq(2)
|
|
@@ -0,0 +1,146 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "spec_helper"
|
|
4
|
+
require "open3"
|
|
5
|
+
require "json"
|
|
6
|
+
require "tmpdir"
|
|
7
|
+
require "parallel"
|
|
8
|
+
require_relative "support/multistream_fixture"
|
|
9
|
+
require "wp2txt/multistream"
|
|
10
|
+
require "wp2txt/output_writer"
|
|
11
|
+
require "wp2txt/stream_processor"
|
|
12
|
+
|
|
13
|
+
RSpec.describe "output integrity" do
|
|
14
|
+
include MultistreamFixture
|
|
15
|
+
|
|
16
|
+
around do |example|
|
|
17
|
+
Dir.mktmpdir("wp2txt-integrity-") do |dir|
|
|
18
|
+
@dir = dir
|
|
19
|
+
example.run
|
|
20
|
+
end
|
|
21
|
+
end
|
|
22
|
+
|
|
23
|
+
let(:cli) { File.expand_path("../bin/wp2txt", __dir__) }
|
|
24
|
+
let(:lib) { File.expand_path("../lib", __dir__) }
|
|
25
|
+
|
|
26
|
+
def run_cli(*args)
|
|
27
|
+
Open3.capture3(RbConfig.ruby, "-I", lib, cli, *args)
|
|
28
|
+
end
|
|
29
|
+
|
|
30
|
+
def json_records(dir)
|
|
31
|
+
Dir[File.join(dir, "*")].flat_map { |f| File.readlines(f) }.map { |line| JSON.parse(line) }
|
|
32
|
+
end
|
|
33
|
+
|
|
34
|
+
def write_pages(path, count)
|
|
35
|
+
body = (1..count).map { |i| page_xml(id: i, ns: 0, title: "記事#{i}", text: "本文#{i}。日本語の文。\n") }.join
|
|
36
|
+
File.write(path, "<mediawiki>\n#{body}</mediawiki>\n")
|
|
37
|
+
end
|
|
38
|
+
|
|
39
|
+
describe "forked workers" do
|
|
40
|
+
it "do not write the parent's buffered output again when they exit" do
|
|
41
|
+
writer = Wp2txt::OutputWriter.new(output_dir: @dir, base_name: "out", format: :text, file_size_mb: 10)
|
|
42
|
+
writer.write("one line still sitting in the buffer")
|
|
43
|
+
writer.flush
|
|
44
|
+
Parallel.map([1, 2, 3], in_processes: 3) { |x| x }
|
|
45
|
+
files = writer.close
|
|
46
|
+
expect(File.readlines(files.first).size).to eq(1)
|
|
47
|
+
end
|
|
48
|
+
|
|
49
|
+
it "leave every article exactly once in streaming output across batch boundaries" do
|
|
50
|
+
input = File.join(@dir, "pages.xml")
|
|
51
|
+
out = File.join(@dir, "out")
|
|
52
|
+
Dir.mkdir(out)
|
|
53
|
+
write_pages(input, 450) # batches of 200 with -n 2
|
|
54
|
+
_stdout, stderr, status = run_cli("-i", input, "-o", out, "--format", "json", "-n", "2")
|
|
55
|
+
expect(status.success?).to be(true), stderr
|
|
56
|
+
|
|
57
|
+
titles = json_records(out).map { |r| r["title"] }
|
|
58
|
+
expect(titles.size).to eq(450)
|
|
59
|
+
expect(titles.tally.select { |_, n| n > 1 }).to be_empty
|
|
60
|
+
end
|
|
61
|
+
end
|
|
62
|
+
|
|
63
|
+
describe "--num-procs" do
|
|
64
|
+
it "uses the requested process count" do
|
|
65
|
+
input = File.join(@dir, "pages.xml")
|
|
66
|
+
out = File.join(@dir, "out")
|
|
67
|
+
Dir.mkdir(out)
|
|
68
|
+
write_pages(input, 3)
|
|
69
|
+
stdout, stderr, status = run_cli("-i", input, "-o", out, "--format", "json", "-n", "1")
|
|
70
|
+
expect(status.success?).to be(true), stderr
|
|
71
|
+
expect(stdout.gsub(/\e\[[\d;]*m/, "")).to match(/CPU cores:\s*1\b/)
|
|
72
|
+
end
|
|
73
|
+
end
|
|
74
|
+
|
|
75
|
+
describe "page and revision IDs" do
|
|
76
|
+
it "appear in JSON records from the streaming path, including summaries" do
|
|
77
|
+
input = File.join(@dir, "pages.xml")
|
|
78
|
+
write_pages(input, 3)
|
|
79
|
+
[[], ["--summary-only"]].each do |extra|
|
|
80
|
+
out = File.join(@dir, "out#{extra.size}")
|
|
81
|
+
Dir.mkdir(out)
|
|
82
|
+
_stdout, stderr, status = run_cli("-i", input, "-o", out, "--format", "json", *extra)
|
|
83
|
+
expect(status.success?).to be(true), stderr
|
|
84
|
+
record = json_records(out).find { |r| r["title"] == "記事2" }
|
|
85
|
+
expect(record).to include("page_id" => 2, "revision_id" => 200)
|
|
86
|
+
expect(record.keys.first(3)).to eq(%w[title page_id revision_id])
|
|
87
|
+
end
|
|
88
|
+
end
|
|
89
|
+
|
|
90
|
+
it "appear in JSON records from the default path for bz2 dumps" do
|
|
91
|
+
dump, = create_fixture(@dir)
|
|
92
|
+
out = File.join(@dir, "out")
|
|
93
|
+
Dir.mkdir(out)
|
|
94
|
+
_stdout, stderr, status = run_cli("-i", dump, "-o", out, "--format", "json")
|
|
95
|
+
expect(status.success?).to be(true), stderr
|
|
96
|
+
record = json_records(out).find { |r| r["title"] == "Film B" }
|
|
97
|
+
expect(record).to include("page_id" => 2, "revision_id" => 200)
|
|
98
|
+
end
|
|
99
|
+
|
|
100
|
+
it "are read from the page header, not from the contributor" do
|
|
101
|
+
xml = "<page><title>X</title><ns>0</ns><id>12</id><revision><id>34</id>" \
|
|
102
|
+
"<contributor><id>9</id></contributor><text>t</text></revision></page>"
|
|
103
|
+
expect(Wp2txt.page_ids(xml)).to eq(page_id: 12, revision_id: 34)
|
|
104
|
+
end
|
|
105
|
+
|
|
106
|
+
it "are offered by StreamProcessor only on request, keeping each_page's shape" do
|
|
107
|
+
input = File.join(@dir, "pages.xml")
|
|
108
|
+
write_pages(input, 1)
|
|
109
|
+
processor = Wp2txt::StreamProcessor.new(input, adaptive_buffer: false)
|
|
110
|
+
expect(processor.each_page.to_a).to eq([["記事1", "本文1。日本語の文。\n"]])
|
|
111
|
+
with_ids = Wp2txt::StreamProcessor.new(input, adaptive_buffer: false).each_page(with_ids: true).to_a
|
|
112
|
+
expect(with_ids.first.last).to eq(page_id: 1, revision_id: 100)
|
|
113
|
+
end
|
|
114
|
+
end
|
|
115
|
+
|
|
116
|
+
describe "targeted extraction after an early index stop" do
|
|
117
|
+
it "knows where the last found article's stream ends" do
|
|
118
|
+
_dump, index_path = create_fixture(@dir)
|
|
119
|
+
index = Wp2txt::MultistreamIndex.new(index_path, use_cache: false, target_titles: ["Film A"],
|
|
120
|
+
show_progress: false)
|
|
121
|
+
expect(index.early_terminated?).to be(true)
|
|
122
|
+
expect(index.stream_offsets).to eq([0])
|
|
123
|
+
second_stream = File.readlines(index_path).map { |line| line.split(":").first.to_i }.uniq[1]
|
|
124
|
+
expect(index.stream_end_offset).to eq(second_stream)
|
|
125
|
+
end
|
|
126
|
+
|
|
127
|
+
it "reads only that stream" do
|
|
128
|
+
dump, index_path = create_fixture(@dir)
|
|
129
|
+
index = Wp2txt::MultistreamIndex.new(index_path, use_cache: false, target_titles: ["Film A"],
|
|
130
|
+
show_progress: false)
|
|
131
|
+
reader = Wp2txt::MultistreamReader.new(dump, index)
|
|
132
|
+
stub_const("Wp2txt::MultistreamReader::MAX_TAIL_STREAM_BYTES", 0)
|
|
133
|
+
expect(reader.extract_article("Film A")).to include(title: "Film A", id: 1, revision_id: 100)
|
|
134
|
+
end
|
|
135
|
+
|
|
136
|
+
it "refuses to read a large remainder in one call when the end is unknown" do
|
|
137
|
+
dump, index_path = create_fixture(@dir)
|
|
138
|
+
index = Wp2txt::MultistreamIndex.new(index_path, use_cache: false, target_titles: ["Film A"],
|
|
139
|
+
show_progress: false)
|
|
140
|
+
index.instance_variable_set(:@stream_end_offset, nil)
|
|
141
|
+
reader = Wp2txt::MultistreamReader.new(dump, index)
|
|
142
|
+
stub_const("Wp2txt::MultistreamReader::MAX_TAIL_STREAM_BYTES", 0)
|
|
143
|
+
expect { reader.extract_article("Film A") }.to raise_error(Wp2txt::Error, /cannot locate the end/)
|
|
144
|
+
end
|
|
145
|
+
end
|
|
146
|
+
end
|