wp2txt 2.3.2 → 2.3.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -23,6 +23,7 @@ module Wp2txt
23
23
  @input_path = input_path
24
24
  @bz2_gem = bz2_gem
25
25
  @buffer = +""
26
+ @pending_bytes = +"".b
26
27
  @file_pointer = nil
27
28
  @adaptive_buffer = adaptive_buffer
28
29
  @buffer_size = adaptive_buffer ? calculate_optimal_buffer_size : DEFAULT_BUFFER_SIZE
@@ -63,20 +64,28 @@ module Wp2txt
63
64
 
64
65
  # Iterate over each page in the input
65
66
  # Yields [title, text] for each page
66
- def each_page
67
- return enum_for(:each_page) unless block_given?
67
+ # Yields title and text of each article. With with_ids: true, also yields
68
+ # { page_id:, revision_id: } taken from the dump.
69
+ def each_page(with_ids: false, &block)
70
+ return enum_for(:each_page, with_ids: with_ids) unless block
71
+
72
+ emit = if with_ids
73
+ ->(title, text, ids) { block.call(title, text, ids) }
74
+ else
75
+ ->(title, text, _ids) { block.call(title, text) }
76
+ end
68
77
 
69
78
  if File.directory?(@input_path)
70
79
  # Process XML files in directory
71
80
  Dir.glob(File.join(@input_path, "*.xml")).sort.each do |xml_file|
72
- process_xml_file(xml_file) { |title, text| yield title, text }
81
+ process_xml_file(xml_file) { |title, text, ids| emit.call(title, text, ids) }
73
82
  end
74
83
  elsif @input_path.end_with?(".bz2")
75
84
  # Process bz2 compressed file with streaming
76
- process_bz2_stream { |title, text| yield title, text }
85
+ process_bz2_stream { |title, text, ids| emit.call(title, text, ids) }
77
86
  elsif @input_path.end_with?(".xml")
78
87
  # Process single XML file
79
- process_xml_file(@input_path) { |title, text| yield title, text }
88
+ process_xml_file(@input_path) { |title, text, ids| emit.call(title, text, ids) }
80
89
  else
81
90
  raise ArgumentError, "Unsupported input format: #{@input_path}"
82
91
  end
@@ -99,7 +108,8 @@ module Wp2txt
99
108
  # Process a single XML file
100
109
  def process_xml_file(xml_file)
101
110
  @buffer = +""
102
- @file_pointer = File.open(xml_file, "r:UTF-8")
111
+ @pending_bytes = +"".b
112
+ @file_pointer = File.open(xml_file, "rb")
103
113
 
104
114
  while (page = extract_next_page)
105
115
  result = parse_page_xml(page)
@@ -120,6 +130,7 @@ module Wp2txt
120
130
  end
121
131
 
122
132
  @buffer = +""
133
+ @pending_bytes = +"".b
123
134
  @file_pointer = open_bz2_stream
124
135
 
125
136
  while (page = extract_next_page)
@@ -158,14 +169,33 @@ module Wp2txt
158
169
  # Fill buffer from file pointer
159
170
  def fill_buffer
160
171
  chunk = @file_pointer.read(@buffer_size)
161
- return false unless chunk
172
+ unless chunk
173
+ unless @pending_bytes.to_s.empty?
174
+ raise Wp2txt::EncodingError,
175
+ "input ends in the middle of a UTF-8 character (byte #{@bytes_read}): #{@input_path}"
176
+ end
177
+ return false
178
+ end
162
179
 
180
+ carried = @pending_bytes.to_s.b
181
+ start = @bytes_read - carried.bytesize # stream position of the first carried byte
182
+ bytes = carried + chunk.b
183
+ # A read can end partway through a multi-byte character; carry those
184
+ # bytes into the next read rather than judging half a character.
185
+ tail = bytes[/[\xC2-\xF4][\x80-\xBF]{0,2}\z/n]
186
+ width = tail && (tail.getbyte(0) < 0xE0 ? 2 : tail.getbyte(0) < 0xF0 ? 3 : 4)
187
+ @pending_bytes = tail && tail.bytesize < width ? tail : +"".b
188
+ bytes = bytes.byteslice(0, bytes.bytesize - @pending_bytes.bytesize).force_encoding(Encoding::UTF_8)
189
+ # Anything still invalid is corrupt input. Dropping it would change the
190
+ # text without a trace, so stop and say where it is.
191
+ unless bytes.valid_encoding?
192
+ bad = bytes.each_char.find_index { |c| !c.valid_encoding? }.to_i
193
+ offset = start + bytes[0, bad].bytesize
194
+ raise Wp2txt::EncodingError,
195
+ "invalid UTF-8 near decompressed byte #{offset}: #{@input_path}"
196
+ end
163
197
  @bytes_read += chunk.bytesize
164
-
165
- # Handle encoding for bz2 streams
166
- chunk = chunk.force_encoding("UTF-8")
167
- chunk = chunk.scrub("")
168
- @buffer << chunk
198
+ @buffer << bytes
169
199
 
170
200
  # Adaptive buffer adjustment: if memory is low, reduce buffer size
171
201
  if @adaptive_buffer && MemoryMonitor.memory_low?
@@ -223,8 +253,8 @@ module Wp2txt
223
253
  return nil unless title_node
224
254
 
225
255
  title = title_node.content
226
- # Skip special pages (containing colon in title like "Wikipedia:", "File:", etc.)
227
- return nil if title.include?(":")
256
+ namespace = title_node.parent.at_css("ns")&.text
257
+ return nil unless Wp2txt.namespace_id(namespace).zero?
228
258
 
229
259
  text = text_node.content
230
260
 
@@ -242,7 +272,7 @@ module Wp2txt
242
272
  end
243
273
 
244
274
  @pages_processed += 1
245
- [title, text]
275
+ [title, text, Wp2txt.page_ids(page_xml)]
246
276
  rescue Nokogiri::XML::SyntaxError
247
277
  # Skip malformed XML
248
278
  nil
@@ -1,5 +1,5 @@
1
1
  # frozen_string_literal: true
2
2
 
3
3
  module Wp2txt
4
- VERSION = "2.3.2"
4
+ VERSION = "2.3.4"
5
5
  end
data/lib/wp2txt.rb CHANGED
@@ -269,12 +269,14 @@ module Wp2txt
269
269
  end
270
270
  page << line if inside_page
271
271
  end
272
- if page.empty?
273
- false
274
- else
275
- page.force_encoding("utf-8")
272
+ return false if page.empty?
273
+
274
+ page.force_encoding("utf-8")
275
+ # Corrupt input must not be cleaned into different text without a trace
276
+ unless page.valid_encoding?
277
+ title = page[%r{<title>([^<]*)</title>}, 1]&.scrub("?")
278
+ raise Wp2txt::EncodingError, "invalid UTF-8 in page #{title.inspect} of #{@input_file}"
276
279
  end
277
- rescue ::Encoding::InvalidByteSequenceError, ::Encoding::UndefinedConversionError
278
280
  page
279
281
  end
280
282
 
@@ -171,6 +171,46 @@ RSpec.describe "Wp2txt Full-Text Search" do
171
171
  end
172
172
  end
173
173
 
174
+ ["", "東", "東京", "é", "😀"].each do |query|
175
+ it "rejects the short phrase #{query.inspect} with a stable code" do
176
+ expect { @fts.search(query, count: "exact") }.to raise_error(Wp2txt::FtsIndex::ShortQueryError) { |error|
177
+ expect(error.code).to eq("query_too_short")
178
+ }
179
+ end
180
+ end
181
+
182
+ it "leaves raw FTS syntax to SQLite even for short expressions" do
183
+ expect(@fts.search('"東"', mode: "query", count: "exact")[:total]).to eq(0)
184
+ expect(@fts.search("\xFF".b, mode: "query", count: "exact")[:total]).to eq(0)
185
+ end
186
+
187
+ it "accepts three Unicode characters" do
188
+ expect(@fts.search("東京都", count: "exact")[:total]).to eq(0)
189
+ end
190
+
191
+ it "returns a coded MCP tool error instead of a zero-result success" do
192
+ @fts.close
193
+ requests = [
194
+ { jsonrpc: "2.0", id: 1, method: "initialize", params: {
195
+ protocolVersion: "2025-03-26", capabilities: {}, clientInfo: { name: "regression", version: "1" }
196
+ } },
197
+ { jsonrpc: "2.0", method: "notifications/initialized" },
198
+ { jsonrpc: "2.0", id: 2, method: "tools/call", params: {
199
+ name: "search_text", arguments: { query: "東京", count: "exact" }
200
+ } }
201
+ ]
202
+ stdout, stderr, status = Open3.capture3(RbConfig.ruby,
203
+ File.expand_path("../bin/wp2txt-mcp", __dir__), "--input", @multistream_path,
204
+ "--cache-dir", File.dirname(@multistream_path),
205
+ stdin_data: requests.map { |request| JSON.generate(request) }.join("\n") + "\n")
206
+ expect(status.success?).to be(true), stderr
207
+ response = stdout.lines.map { |line| JSON.parse(line) }.find { |message| message["id"] == 2 }
208
+ expect(response.dig("result", "isError")).to be true
209
+ payload = JSON.parse(response.dig("result", "content", 0, "text"))
210
+ expect(payload["code"]).to eq("query_too_short")
211
+ expect(payload).not_to have_key("total")
212
+ end
213
+
174
214
  it "matches substrings of three or more characters" do
175
215
  result = @fts.search("tory", count: "exact")
176
216
  expect(result[:total]).to eq(2)
@@ -0,0 +1,146 @@
1
+ # frozen_string_literal: true
2
+
3
+ require "spec_helper"
4
+ require "open3"
5
+ require "json"
6
+ require "tmpdir"
7
+ require "parallel"
8
+ require_relative "support/multistream_fixture"
9
+ require "wp2txt/multistream"
10
+ require "wp2txt/output_writer"
11
+ require "wp2txt/stream_processor"
12
+
13
+ RSpec.describe "output integrity" do
14
+ include MultistreamFixture
15
+
16
+ around do |example|
17
+ Dir.mktmpdir("wp2txt-integrity-") do |dir|
18
+ @dir = dir
19
+ example.run
20
+ end
21
+ end
22
+
23
+ let(:cli) { File.expand_path("../bin/wp2txt", __dir__) }
24
+ let(:lib) { File.expand_path("../lib", __dir__) }
25
+
26
+ def run_cli(*args)
27
+ Open3.capture3(RbConfig.ruby, "-I", lib, cli, *args)
28
+ end
29
+
30
+ def json_records(dir)
31
+ Dir[File.join(dir, "*")].flat_map { |f| File.readlines(f) }.map { |line| JSON.parse(line) }
32
+ end
33
+
34
+ def write_pages(path, count)
35
+ body = (1..count).map { |i| page_xml(id: i, ns: 0, title: "記事#{i}", text: "本文#{i}。日本語の文。\n") }.join
36
+ File.write(path, "<mediawiki>\n#{body}</mediawiki>\n")
37
+ end
38
+
39
+ describe "forked workers" do
40
+ it "do not write the parent's buffered output again when they exit" do
41
+ writer = Wp2txt::OutputWriter.new(output_dir: @dir, base_name: "out", format: :text, file_size_mb: 10)
42
+ writer.write("one line still sitting in the buffer")
43
+ writer.flush
44
+ Parallel.map([1, 2, 3], in_processes: 3) { |x| x }
45
+ files = writer.close
46
+ expect(File.readlines(files.first).size).to eq(1)
47
+ end
48
+
49
+ it "leave every article exactly once in streaming output across batch boundaries" do
50
+ input = File.join(@dir, "pages.xml")
51
+ out = File.join(@dir, "out")
52
+ Dir.mkdir(out)
53
+ write_pages(input, 450) # batches of 200 with -n 2
54
+ _stdout, stderr, status = run_cli("-i", input, "-o", out, "--format", "json", "-n", "2")
55
+ expect(status.success?).to be(true), stderr
56
+
57
+ titles = json_records(out).map { |r| r["title"] }
58
+ expect(titles.size).to eq(450)
59
+ expect(titles.tally.select { |_, n| n > 1 }).to be_empty
60
+ end
61
+ end
62
+
63
+ describe "--num-procs" do
64
+ it "uses the requested process count" do
65
+ input = File.join(@dir, "pages.xml")
66
+ out = File.join(@dir, "out")
67
+ Dir.mkdir(out)
68
+ write_pages(input, 3)
69
+ stdout, stderr, status = run_cli("-i", input, "-o", out, "--format", "json", "-n", "1")
70
+ expect(status.success?).to be(true), stderr
71
+ expect(stdout.gsub(/\e\[[\d;]*m/, "")).to match(/CPU cores:\s*1\b/)
72
+ end
73
+ end
74
+
75
+ describe "page and revision IDs" do
76
+ it "appear in JSON records from the streaming path, including summaries" do
77
+ input = File.join(@dir, "pages.xml")
78
+ write_pages(input, 3)
79
+ [[], ["--summary-only"]].each do |extra|
80
+ out = File.join(@dir, "out#{extra.size}")
81
+ Dir.mkdir(out)
82
+ _stdout, stderr, status = run_cli("-i", input, "-o", out, "--format", "json", *extra)
83
+ expect(status.success?).to be(true), stderr
84
+ record = json_records(out).find { |r| r["title"] == "記事2" }
85
+ expect(record).to include("page_id" => 2, "revision_id" => 200)
86
+ expect(record.keys.first(3)).to eq(%w[title page_id revision_id])
87
+ end
88
+ end
89
+
90
+ it "appear in JSON records from the default path for bz2 dumps" do
91
+ dump, = create_fixture(@dir)
92
+ out = File.join(@dir, "out")
93
+ Dir.mkdir(out)
94
+ _stdout, stderr, status = run_cli("-i", dump, "-o", out, "--format", "json")
95
+ expect(status.success?).to be(true), stderr
96
+ record = json_records(out).find { |r| r["title"] == "Film B" }
97
+ expect(record).to include("page_id" => 2, "revision_id" => 200)
98
+ end
99
+
100
+ it "are read from the page header, not from the contributor" do
101
+ xml = "<page><title>X</title><ns>0</ns><id>12</id><revision><id>34</id>" \
102
+ "<contributor><id>9</id></contributor><text>t</text></revision></page>"
103
+ expect(Wp2txt.page_ids(xml)).to eq(page_id: 12, revision_id: 34)
104
+ end
105
+
106
+ it "are offered by StreamProcessor only on request, keeping each_page's shape" do
107
+ input = File.join(@dir, "pages.xml")
108
+ write_pages(input, 1)
109
+ processor = Wp2txt::StreamProcessor.new(input, adaptive_buffer: false)
110
+ expect(processor.each_page.to_a).to eq([["記事1", "本文1。日本語の文。\n"]])
111
+ with_ids = Wp2txt::StreamProcessor.new(input, adaptive_buffer: false).each_page(with_ids: true).to_a
112
+ expect(with_ids.first.last).to eq(page_id: 1, revision_id: 100)
113
+ end
114
+ end
115
+
116
+ describe "targeted extraction after an early index stop" do
117
+ it "knows where the last found article's stream ends" do
118
+ _dump, index_path = create_fixture(@dir)
119
+ index = Wp2txt::MultistreamIndex.new(index_path, use_cache: false, target_titles: ["Film A"],
120
+ show_progress: false)
121
+ expect(index.early_terminated?).to be(true)
122
+ expect(index.stream_offsets).to eq([0])
123
+ second_stream = File.readlines(index_path).map { |line| line.split(":").first.to_i }.uniq[1]
124
+ expect(index.stream_end_offset).to eq(second_stream)
125
+ end
126
+
127
+ it "reads only that stream" do
128
+ dump, index_path = create_fixture(@dir)
129
+ index = Wp2txt::MultistreamIndex.new(index_path, use_cache: false, target_titles: ["Film A"],
130
+ show_progress: false)
131
+ reader = Wp2txt::MultistreamReader.new(dump, index)
132
+ stub_const("Wp2txt::MultistreamReader::MAX_TAIL_STREAM_BYTES", 0)
133
+ expect(reader.extract_article("Film A")).to include(title: "Film A", id: 1, revision_id: 100)
134
+ end
135
+
136
+ it "refuses to read a large remainder in one call when the end is unknown" do
137
+ dump, index_path = create_fixture(@dir)
138
+ index = Wp2txt::MultistreamIndex.new(index_path, use_cache: false, target_titles: ["Film A"],
139
+ show_progress: false)
140
+ index.instance_variable_set(:@stream_end_offset, nil)
141
+ reader = Wp2txt::MultistreamReader.new(dump, index)
142
+ stub_const("Wp2txt::MultistreamReader::MAX_TAIL_STREAM_BYTES", 0)
143
+ expect { reader.extract_article("Film A") }.to raise_error(Wp2txt::Error, /cannot locate the end/)
144
+ end
145
+ end
146
+ end