wp2txt 2.3.3 → 2.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (45) hide show
  1. checksums.yaml +4 -4
  2. data/.dockerignore +0 -4
  3. data/.gitignore +3 -5
  4. data/CHANGELOG.md +18 -0
  5. data/DEVELOPMENT.md +1 -1
  6. data/DEVELOPMENT_ja.md +1 -1
  7. data/README.md +44 -2
  8. data/README_ja.md +35 -2
  9. data/Rakefile +10 -21
  10. data/bin/wp2txt +79 -17
  11. data/bin/wp2txt-mcp +1 -1
  12. data/docs/INDEXES.md +61 -1
  13. data/lib/wp2txt/article.rb +1 -1
  14. data/lib/wp2txt/cli.rb +42 -0
  15. data/lib/wp2txt/constants.rb +24 -0
  16. data/lib/wp2txt/corpus.rb +11 -2
  17. data/lib/wp2txt/data/template_aliases.json +1 -1
  18. data/lib/wp2txt/extractor.rb +10 -1
  19. data/lib/wp2txt/formatter.rb +20 -0
  20. data/lib/wp2txt/index_commands.rb +89 -1
  21. data/lib/wp2txt/langlinks_importer.rb +19 -35
  22. data/lib/wp2txt/lead_terms.rb +228 -0
  23. data/lib/wp2txt/link_counter.rb +171 -0
  24. data/lib/wp2txt/metadata_index.rb +79 -3
  25. data/lib/wp2txt/multistream.rb +52 -2
  26. data/lib/wp2txt/output_writer.rb +8 -0
  27. data/lib/wp2txt/page_props_importer.rb +170 -0
  28. data/lib/wp2txt/sql_dump_reader.rb +57 -0
  29. data/lib/wp2txt/stream_processor.rb +42 -15
  30. data/lib/wp2txt/template_expander.rb +19 -0
  31. data/lib/wp2txt/utils.rb +5 -3
  32. data/lib/wp2txt/version.rb +1 -1
  33. data/lib/wp2txt/wikitext_regions.rb +66 -0
  34. data/lib/wp2txt.rb +7 -5
  35. data/spec/docs_sync_spec.rb +3 -3
  36. data/spec/langlinks_importer_spec.rb +27 -0
  37. data/spec/lead_terms_edge_cases_spec.rb +194 -0
  38. data/spec/lead_terms_links_qids_spec.rb +204 -0
  39. data/spec/output_integrity_spec.rb +146 -0
  40. data/spec/p1_correctness_spec.rb +32 -14
  41. data/spec/page_properties_spec.rb +161 -0
  42. data/spec/region_semantics_spec.rb +71 -0
  43. data/spec/template_passthrough_spec.rb +43 -0
  44. data/spec/titles_output_path_spec.rb +12 -0
  45. metadata +18 -1
@@ -19,8 +19,12 @@ module Wp2txt
19
19
 
20
20
  attr_reader :buffer_size, :pages_processed, :bytes_read, :redirects_skipped
21
21
 
22
- def initialize(input_path, bz2_gem: false, adaptive_buffer: true, validate_bz2: true, skip_redirects: true)
22
+ # keep_raw_text: also hand back each article's wikitext as stored in the
23
+ # dump (before comment removal), as :raw_text among the with_ids values
24
+ def initialize(input_path, bz2_gem: false, adaptive_buffer: true, validate_bz2: true, skip_redirects: true,
25
+ keep_raw_text: false)
23
26
  @input_path = input_path
27
+ @keep_raw_text = keep_raw_text
24
28
  @bz2_gem = bz2_gem
25
29
  @buffer = +""
26
30
  @pending_bytes = +"".b
@@ -64,20 +68,28 @@ module Wp2txt
64
68
 
65
69
  # Iterate over each page in the input
66
70
  # Yields [title, text] for each page
67
- def each_page
68
- return enum_for(:each_page) unless block_given?
71
+ # Yields title and text of each article. With with_ids: true, also yields
72
+ # { page_id:, revision_id: } taken from the dump.
73
+ def each_page(with_ids: false, &block)
74
+ return enum_for(:each_page, with_ids: with_ids) unless block
75
+
76
+ emit = if with_ids
77
+ ->(title, text, ids) { block.call(title, text, ids) }
78
+ else
79
+ ->(title, text, _ids) { block.call(title, text) }
80
+ end
69
81
 
70
82
  if File.directory?(@input_path)
71
83
  # Process XML files in directory
72
84
  Dir.glob(File.join(@input_path, "*.xml")).sort.each do |xml_file|
73
- process_xml_file(xml_file) { |title, text| yield title, text }
85
+ process_xml_file(xml_file) { |title, text, ids| emit.call(title, text, ids) }
74
86
  end
75
87
  elsif @input_path.end_with?(".bz2")
76
88
  # Process bz2 compressed file with streaming
77
- process_bz2_stream { |title, text| yield title, text }
89
+ process_bz2_stream { |title, text, ids| emit.call(title, text, ids) }
78
90
  elsif @input_path.end_with?(".xml")
79
91
  # Process single XML file
80
- process_xml_file(@input_path) { |title, text| yield title, text }
92
+ process_xml_file(@input_path) { |title, text, ids| emit.call(title, text, ids) }
81
93
  else
82
94
  raise ArgumentError, "Unsupported input format: #{@input_path}"
83
95
  end
@@ -162,20 +174,32 @@ module Wp2txt
162
174
  def fill_buffer
163
175
  chunk = @file_pointer.read(@buffer_size)
164
176
  unless chunk
165
- @buffer << @pending_bytes.to_s.dup.force_encoding(Encoding::UTF_8).scrub("")
166
- @pending_bytes = +"".b
177
+ unless @pending_bytes.to_s.empty?
178
+ raise Wp2txt::EncodingError,
179
+ "input ends in the middle of a UTF-8 character (byte #{@bytes_read}): #{@input_path}"
180
+ end
167
181
  return false
168
182
  end
169
183
 
170
- @bytes_read += chunk.bytesize
171
- bytes = @pending_bytes.to_s.b + chunk.b
172
- # Retain a trailing UTF-8 sequence until the next read. Only complete
173
- # chunks are scrubbed, so valid characters split by read are preserved.
184
+ carried = @pending_bytes.to_s.b
185
+ start = @bytes_read - carried.bytesize # stream position of the first carried byte
186
+ bytes = carried + chunk.b
187
+ # A read can end partway through a multi-byte character; carry those
188
+ # bytes into the next read rather than judging half a character.
174
189
  tail = bytes[/[\xC2-\xF4][\x80-\xBF]{0,2}\z/n]
175
190
  width = tail && (tail.getbyte(0) < 0xE0 ? 2 : tail.getbyte(0) < 0xF0 ? 3 : 4)
176
191
  @pending_bytes = tail && tail.bytesize < width ? tail : +"".b
177
- bytes = bytes.byteslice(0, bytes.bytesize - @pending_bytes.bytesize)
178
- @buffer << bytes.force_encoding(Encoding::UTF_8).scrub("")
192
+ bytes = bytes.byteslice(0, bytes.bytesize - @pending_bytes.bytesize).force_encoding(Encoding::UTF_8)
193
+ # Anything still invalid is corrupt input. Dropping it would change the
194
+ # text without a trace, so stop and say where it is.
195
+ unless bytes.valid_encoding?
196
+ bad = bytes.each_char.find_index { |c| !c.valid_encoding? }.to_i
197
+ offset = start + bytes[0, bad].bytesize
198
+ raise Wp2txt::EncodingError,
199
+ "invalid UTF-8 near decompressed byte #{offset}: #{@input_path}"
200
+ end
201
+ @bytes_read += chunk.bytesize
202
+ @buffer << bytes
179
203
 
180
204
  # Adaptive buffer adjustment: if memory is low, reduce buffer size
181
205
  if @adaptive_buffer && MemoryMonitor.memory_low?
@@ -237,6 +261,7 @@ module Wp2txt
237
261
  return nil unless Wp2txt.namespace_id(namespace).zero?
238
262
 
239
263
  text = text_node.content
264
+ raw_text = text if @keep_raw_text
240
265
 
241
266
  # Early redirect detection and skip (before expensive processing)
242
267
  # Redirects start with # or # followed by redirect keyword and [[target]]
@@ -252,7 +277,9 @@ module Wp2txt
252
277
  end
253
278
 
254
279
  @pages_processed += 1
255
- [title, text]
280
+ meta = Wp2txt.page_ids(page_xml)
281
+ meta[:raw_text] = raw_text if @keep_raw_text
282
+ [title, text, meta]
256
283
  rescue Nokogiri::XML::SyntaxError
257
284
  # Skip malformed XML
258
285
  nil
@@ -171,6 +171,19 @@ module Wp2txt
171
171
  nil
172
172
  end
173
173
 
174
+ # Templates whose rendering the text-cleaning stage defines (see
175
+ # Wp2txt#correct_inline_template), keyed by normalized name. Limited to
176
+ # kinds whose output does not depend on marker settings.
177
+ def rendered_by_cleaner?(template_name)
178
+ @rendered_by_cleaner ||= [Wp2txt::RUBY_TEXT_TEMPLATES, Wp2txt::INTERWIKI_LINK_TEMPLATES]
179
+ .flatten.to_set { |name| name.to_s.tr("_", " ").strip.downcase }
180
+ @rendered_by_cleaner.include?(template_name.tr("_", " "))
181
+ end
182
+
183
+ def text_renderer
184
+ @text_renderer ||= Object.new.extend(Wp2txt)
185
+ end
186
+
174
187
  def expand_single_template(content)
175
188
  parts = split_template_parts(content)
176
189
  return "" if parts.empty?
@@ -263,6 +276,12 @@ module Wp2txt
263
276
  # Handle lang-xx templates (e.g., lang-fr, lang-de, lang-ja)
264
277
  if template_name.start_with?("lang-")
265
278
  expand_lang_xx(template_name, params)
279
+ elsif rendered_by_cleaner?(template_name)
280
+ # These carry words the article needs (a reading, a link's display
281
+ # text). Deleting them dropped those words; leaving the raw template
282
+ # for later confused links that contain it (an image caption with a
283
+ # "|" inside). Render them now, with the cleaner's own rules.
284
+ text_renderer.correct_inline_template("{{#{expand(content)}}}")
266
285
  else
267
286
  @preserve_unknown ? "{{#{content}}}" : ""
268
287
  end
data/lib/wp2txt/utils.rb CHANGED
@@ -1,6 +1,7 @@
1
1
  # frozen_string_literal: true
2
2
 
3
3
  require "strscan"
4
+ require_relative "wikitext_regions"
4
5
  require_relative "constants"
5
6
  require_relative "regex"
6
7
  require_relative "text_processing"
@@ -574,8 +575,9 @@ module Wp2txt
574
575
  # Helper to check if template name matches any in a list (case-insensitive)
575
576
  def template_matches?(name, template_list)
576
577
  return false if template_list.nil? || template_list.empty?
577
- normalized_name = name.to_s.strip.downcase
578
- template_list.any? { |t| t.downcase == normalized_name }
578
+ # MediaWiki treats "_" and " " in template names as the same character
579
+ normalized_name = name.to_s.tr("_", " ").strip.downcase
580
+ template_list.any? { |t| t.tr("_", " ").downcase == normalized_name }
579
581
  end
580
582
 
581
583
  def correct_inline_template(str, enabled_markers = [], extract_citations = false)
@@ -583,7 +585,7 @@ module Wp2txt
583
585
  return str unless str.include?("{{")
584
586
 
585
587
  process_nested_single_pass(str, "{{", "}}") do |contents|
586
- parts = contents.split("|")
588
+ parts = WikitextRegions.split_pipes(contents)
587
589
  template_name = (parts[0] || "").strip
588
590
  template_name_lower = template_name.downcase
589
591
 
@@ -1,5 +1,5 @@
1
1
  # frozen_string_literal: true
2
2
 
3
3
  module Wp2txt
4
- VERSION = "2.3.3"
4
+ VERSION = "2.4.0"
5
5
  end
@@ -0,0 +1,66 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Wp2txt
4
+ # Literal regions suppress wikitext parsing. Galleries and timelines can
5
+ # contain links, but their contents are not the article's lead prose.
6
+ module WikitextRegions
7
+ LITERAL_TAGS = %w[nowiki pre math chem ce score syntaxhighlight source graph mapframe templatedata].freeze
8
+ NON_PROSE_TAGS = %w[gallery timeline].freeze
9
+ COMMENT = /<!--.*?(?:-->|\z)/m
10
+
11
+ def self.region_pattern(tags)
12
+ /<!--.*?(?:-->|\z)|<(?<tag>#{tags.join('|')})(?=[\s\/>])(?:"[^"]*"|'[^']*'|[^'">])*?(?:\/>|>.*?(?:<\/\k<tag>\s*>|\z))/mi
13
+ end
14
+
15
+ LITERAL_REGION = region_pattern(LITERAL_TAGS)
16
+ LEAD_REGION = region_pattern(LITERAL_TAGS + NON_PROSE_TAGS)
17
+ LITERAL_REGION_AT = /\G(?:#{LITERAL_REGION})/
18
+ LEAD_REGION_AT = /\G(?:#{LEAD_REGION})/
19
+
20
+ module_function
21
+
22
+ def end_at(text, offset, lead: false)
23
+ match = (lead ? LEAD_REGION_AT : LITERAL_REGION_AT).match(text, offset)
24
+ match.end(0) if match
25
+ end
26
+
27
+ def remove_literal(text)
28
+ text.gsub(LITERAL_REGION, "")
29
+ end
30
+
31
+ # One space per codepoint, including internal newlines: an excluded
32
+ # region must not introduce a paragraph or heading boundary.
33
+ def mask_lead(text)
34
+ text.gsub(LEAD_REGION) { |region| " " * region.length }
35
+ end
36
+
37
+ def split_pipes(text)
38
+ parts = [+""]
39
+ stack = []
40
+ i = 0
41
+ while i < text.length
42
+ two = text[i, 2]
43
+ if text[i] == "<" && (stop = end_at(text, i))
44
+ parts.last << text[i...stop]
45
+ i = stop
46
+ elsif ["{{", "[["].include?(two)
47
+ stack << (two == "{{" ? "}}" : "]]")
48
+ parts.last << two
49
+ i += 2
50
+ elsif !stack.empty? && two == stack.last
51
+ stack.pop
52
+ parts.last << two
53
+ i += 2
54
+ else
55
+ if text[i] == "|" && stack.empty?
56
+ parts << +""
57
+ else
58
+ parts.last << text[i]
59
+ end
60
+ i += 1
61
+ end
62
+ end
63
+ parts
64
+ end
65
+ end
66
+ end
data/lib/wp2txt.rb CHANGED
@@ -269,12 +269,14 @@ module Wp2txt
269
269
  end
270
270
  page << line if inside_page
271
271
  end
272
- if page.empty?
273
- false
274
- else
275
- page.force_encoding("utf-8")
272
+ return false if page.empty?
273
+
274
+ page.force_encoding("utf-8")
275
+ # Corrupt input must not be cleaned into different text without a trace
276
+ unless page.valid_encoding?
277
+ title = page[%r{<title>([^<]*)</title>}, 1]&.scrub("?")
278
+ raise Wp2txt::EncodingError, "invalid UTF-8 in page #{title.inspect} of #{@input_file}"
276
279
  end
277
- rescue ::Encoding::InvalidByteSequenceError, ::Encoding::UndefinedConversionError
278
280
  page
279
281
  end
280
282
 
@@ -37,14 +37,14 @@ RSpec.describe "documentation surface sync" do
37
37
  expect(phantom).to be_empty, "documented tools that do not exist: #{phantom.join(', ')}"
38
38
  end
39
39
 
40
- # Tripwire: tracked files must not contain tokens listed in .private-doc-tokens,
40
+ # Tripwire: tracked files must not contain tokens listed in .local/doc-tokens,
41
41
  # an untracked, machine-local file (one substring per line; # starts a comment).
42
42
  # The file exists only on machines that maintain such a list; everywhere else
43
43
  # (CI, other contributors) this example skips — loudly, so a silently dead
44
44
  # check cannot be mistaken for a passing one.
45
45
  it "keeps machine-local private tokens out of tracked files" do
46
- token_file = File.join(repo_root, ".private-doc-tokens")
47
- skip "SKIPPED: no .private-doc-tokens on this machine — tripwire not checked" unless File.exist?(token_file)
46
+ token_file = File.join(repo_root, ".local", "doc-tokens")
47
+ skip "SKIPPED: no .local/doc-tokens on this machine — tripwire not checked" unless File.exist?(token_file)
48
48
 
49
49
  tokens = File.readlines(token_file, encoding: "UTF-8")
50
50
  .map(&:strip).reject { |t| t.empty? || t.start_with?("#") }
@@ -65,6 +65,33 @@ RSpec.describe Wp2txt::LanglinksImporter do
65
65
  rows
66
66
  end
67
67
 
68
+ # Current dumps put the INSERT header on its own line and one tuple per line
69
+ MULTILINE_SQL = <<~MSQL
70
+ INSERT INTO `langlinks` VALUES
71
+ (1,'en','Film A'),
72
+ (1,'ja','映画A'),
73
+ (2,'en','It\\'s a Film; Really');
74
+ INSERT INTO `pagelinks` VALUES
75
+ (9,'Ignored',0);
76
+ INSERT INTO `langlinks` VALUES
77
+ (3,'fr','Film B');
78
+ MSQL
79
+
80
+ describe "dumps with one tuple per line" do
81
+ it "imports every tuple of every langlinks statement, and nothing else" do
82
+ path = write_langlinks("testwiki-20260101-langlinks.sql.gz", MULTILINE_SQL, gzip: true)
83
+ expect(import(path)[:row_count]).to eq(4)
84
+ expect(langlinks_rows).to contain_exactly(
85
+ [1, "en", "Film A"], [1, "ja", "映画A"], [2, "en", "It's a Film; Really"], [3, "fr", "Film B"]
86
+ )
87
+ end
88
+
89
+ it "refuses to report success when no rows could be read" do
90
+ path = write_langlinks("testwiki-20260101-langlinks.sql", "-- nothing here\nUNLOCK TABLES;\n")
91
+ expect { import(path) }.to raise_error(Wp2txt::Error, /no langlinks rows found/)
92
+ end
93
+ end
94
+
68
95
  describe "parsing and normalization" do
69
96
  it "imports tuples with escapes, commas, parens, and multiple INSERT statements" do
70
97
  path = write_langlinks("testwiki-20260101-langlinks.sql")
@@ -0,0 +1,194 @@
1
+ # frozen_string_literal: true
2
+
3
+ require 'spec_helper'
4
+ require 'wp2txt'
5
+ require 'wp2txt/lead_terms'
6
+ require 'wp2txt/link_counter'
7
+ require 'wp2txt/metadata_index'
8
+ require 'open3'
9
+ require 'tmpdir'
10
+ require_relative 'support/multistream_fixture'
11
+
12
+ RSpec.describe 'lead terms and link counting on unusual wikitext' do
13
+ include MultistreamFixture
14
+
15
+ let(:cleaner) { Object.new.extend(Wp2txt) }
16
+ let(:render) { ->(s) { cleaner.format_wiki(s, expand_templates: true, markers: [:all]) } }
17
+ def terms(text)
18
+ Wp2txt::LeadTerms.extract(text, render: render)
19
+ end
20
+
21
+ it '#1 preserves link display and reading with expansion enabled or disabled' do
22
+ {
23
+ '{{読み仮名|[[東京|東京都]]|とうきょう}}' => '東京都(とうきょう)',
24
+ '{{仮リンク|[[東京|東京都]]|en|Tokyo}}' => '東京都',
25
+ '{{読み仮名|{{lang|ja|[[東京|東京都]]}}|とうきょう}}' => '東京都(とうきょう)',
26
+ '{{読み仮名|東京|}}' => '東京'
27
+ }.each do |input, expected|
28
+ [true, false].each do |expand|
29
+ expect(cleaner.format_wiki(input, expand_templates: expand)).to eq(expected)
30
+ end
31
+ end
32
+ end
33
+
34
+ def build_case_dump(dir, rule)
35
+ alias_title = rule == 'first-letter' ? 'Alias' : 'alias'
36
+ pages = [[1, 'apple', ''], [2, 'Apple', ''], [3, alias_title, '#REDIRECT [[apple]]'],
37
+ [4, 'Direct', '[[apple]]'], [5, 'Via', '[[alias]]'], [6, '東京', '']]
38
+ xml = "<mediawiki><siteinfo><case>#{rule}</case></siteinfo>" + pages.map do |id, title, text|
39
+ page_xml(id: id, ns: 0, title: title, text: text)
40
+ end.join + '</mediawiki>'
41
+ dump = File.join(dir, 'testwiki-20260101-multistream.xml.bz2')
42
+ File.binwrite(dump, bzip2(xml))
43
+ db = File.join(dir, 'meta.sqlite3')
44
+ Wp2txt::MetadataIndexBuilder.new(dump, [0], db_path: db, num_processes: 0).build.close
45
+ [dump, db]
46
+ end
47
+
48
+ it '#2 stores siteinfo case and preserves direct and redirected case-sensitive targets' do
49
+ Dir.mktmpdir do |dir|
50
+ dump, path = build_case_dump(dir, 'case-sensitive')
51
+ db = SQLite3::Database.new(path)
52
+ expect(db.get_first_value("SELECT value FROM metadata WHERE key = 'case_rule'")).to eq('case-sensitive')
53
+ db.close
54
+ Wp2txt::LinkCounter.new(dump, [0], db_path: path, num_processes: 0).count!
55
+ db = SQLite3::Database.new(path)
56
+ expect(db.execute('SELECT page_id,inlinks,via_redirects FROM page_inlinks WHERE page_id IN (1,2) ORDER BY page_id'))
57
+ .to eq([[1, 2, 1], [2, 0, 0]])
58
+ db.close
59
+ expect(Wp2txt::MetadataIndex.normalize_title('apple')).to eq('Apple')
60
+ end
61
+ end
62
+
63
+ it '#2 uses first-letter for explicit siteinfo and legacy metadata without the rule' do
64
+ Dir.mktmpdir do |dir|
65
+ dump, path = build_case_dump(dir, 'first-letter')
66
+ [false, true].each do |legacy|
67
+ db = SQLite3::Database.new(path)
68
+ if legacy
69
+ db.execute("DELETE FROM metadata WHERE key = 'case_rule'")
70
+ else
71
+ expect(db.get_first_value("SELECT value FROM metadata WHERE key = 'case_rule'")).to eq('first-letter')
72
+ end
73
+ db.close
74
+ Wp2txt::LinkCounter.new(dump, [0], db_path: path, num_processes: 0).count!
75
+ db = SQLite3::Database.new(path)
76
+ expect(db.get_first_value('SELECT inlinks FROM page_inlinks WHERE page_id=1')).to eq(0)
77
+ expect(db.execute('SELECT inlinks,via_redirects FROM page_inlinks WHERE page_id=2')).to eq([[2, 1]])
78
+ db.close
79
+ end
80
+ end
81
+ end
82
+
83
+ it '#2 reads a separate siteinfo stream before the first indexed page' do
84
+ Dir.mktmpdir do |dir|
85
+ header = bzip2('<mediawiki><siteinfo><case>case-sensitive</case></siteinfo>')
86
+ dump = File.join(dir, 'dump.xml.bz2')
87
+ File.binwrite(dump, header + bzip2(page_xml(id: 1, ns: 0, title: 'apple', text: '') + '</mediawiki>'))
88
+ path = File.join(dir, 'meta.sqlite3')
89
+ Wp2txt::MetadataIndexBuilder.new(dump, [header.bytesize], db_path: path, num_processes: 0).build.close
90
+ db = SQLite3::Database.new(path)
91
+ expect(db.get_first_value("SELECT value FROM metadata WHERE key = 'case_rule'")).to eq('case-sensitive')
92
+ db.close
93
+ end
94
+ end
95
+
96
+ %w[nowiki pre math syntaxhighlight source gallery score timeline chem ce graph mapframe templatedata].each do |tag|
97
+ it "#3 skips #{tag} contents and self-closing forms while preserving source offsets" do
98
+ text = "<#{tag.upcase} data-x='>'>'''偽'''(にせ){{ruby|偽|にせ}}</#{tag.upcase}>\n\n" \
99
+ "<#{tag} />'''東京'''(とうきょう)。"
100
+ result = terms(text)
101
+ expect(result.map { |t| t['text'] }).to eq(['東京'])
102
+ s, e = result.first['span']['bold']
103
+ expect(text[s...e]).to eq("'''東京'''")
104
+ expect(terms("<#{tag}>'''偽'''" )).to eq([])
105
+ end
106
+ end
107
+
108
+ it '#4 ignores delimiters in comments and excluded tags inside parentheses' do
109
+ text = "'''東京'''(とうきょう<!-- ) , -->、<nowiki>),</nowiki>Tokyo)は都市。"
110
+ result = terms(text).first
111
+ s, e = result['span']['paren']
112
+ expect(text[s...e]).to eq('(とうきょう<!-- ) , -->、<nowiki>),</nowiki>Tokyo)')
113
+ expect(result['notes'].size).to eq(2)
114
+ expect(result['notes'].first).to eq('とうきょう')
115
+ expect(result['notes_text']).not_to include('<!--')
116
+ end
117
+
118
+ it 'retains the original text of ruby arguments containing excluded regions' do
119
+ expect(terms('{{ruby|<nowiki>東京</nowiki>|とうきょう}}').first.slice('text', 'reading'))
120
+ .to eq('text' => '東京', 'reading' => 'とうきょう')
121
+ end
122
+
123
+ it '#5 detects only headings outside skipped regions including trailing comments' do
124
+ ["<!--\n== 偽 ==\n-->", "<nowiki>\n== 偽 ==\n</nowiki>", "{{box|\n== 偽 ==\n}}", "{|\n== 偽 ==\n|}"].each do |prefix|
125
+ expect(terms("#{prefix}\n'''東京'''(とうきょう)").map { |t| t['text'] }).to eq(['東京'])
126
+ end
127
+ expect(terms("導入文。\n\n== 歴史 == <!-- comment -->\n'''後の語'''(あと)。")).to eq([])
128
+ end
129
+
130
+ it '#6 allows single newlines before and inside notes but stops at a blank line' do
131
+ expect(terms("'''東京'''\n(とうきょう)は都市。").first['notes']).to eq(['とうきょう'])
132
+ text = "'''東京'''(とうきょう、\nTokyo)は都市。"
133
+ term = terms(text).first
134
+ expect(term['notes']).to eq(['とうきょう', 'Tokyo'])
135
+ s, e = term['span']['paren']
136
+ expect(text[s...e]).to eq("(とうきょう、\nTokyo)")
137
+ ["'''東京'''\n\n(とうきょう)", "'''東京'''(とうきょう、\n \nTokyo)"].each do |input|
138
+ expect(terms(input).first['notes']).to eq([])
139
+ end
140
+ end
141
+
142
+ it '#7 stops at empty lines containing whitespace' do
143
+ ["\n \n", "\n\t\n", "\r\n \r\n"].each do |gap|
144
+ expect(terms("'''東京'''(とうきょう)。#{gap}'''別段落'''(べつだんらく)。").map { |t| t['text'] }).to eq(['東京'])
145
+ end
146
+ end
147
+
148
+ def counts(text)
149
+ counter = Wp2txt::LinkCounter.new('', [], db_path: '')
150
+ counter.instance_variable_set(:@redirects, {})
151
+ direct, via = Hash.new(0), Hash.new(0)
152
+ counter.send(:count_page, page_xml(id: 1, ns: 0, title: 'Source', text: text), direct, via)
153
+ direct
154
+ end
155
+
156
+ it '#8 decodes title entities before sections and normalizes whitespace before colon' do
157
+ expect(counts('[[M&amp;A]] [[Caf&#233;]] [[ :東京#節|表示]] [[Caf&#xE9;#section]] [[#節]] [[]]'))
158
+ .to eq('M&A' => 1, 'Café' => 1, '東京' => 1)
159
+ expect(counts('[[東京&#35;節]] [[東京|表示]]')).to eq('東京' => 1)
160
+ end
161
+
162
+ it '#9 excludes every non-wikitext tag and comments from incoming links' do
163
+ %w[nowiki pre math syntaxhighlight source score chem ce graph mapframe templatedata].each do |tag|
164
+ expect(counts("<#{tag}>[[東京]]</#{tag}>[[大阪]]<#{tag}/><!-- [[京都]] -->")).to eq('大阪' => 1)
165
+ end
166
+ end
167
+
168
+ it 'handles empty text and binary invalid bytes in the shared region helper' do
169
+ expect(terms('')).to eq([])
170
+ expect(counts('')).to eq({})
171
+ expect(Wp2txt::WikitextRegions.remove_literal("<nowiki>\xFF</nowiki>\xFE".b)).to eq("\xFE".b)
172
+ end
173
+
174
+ def cli(*options)
175
+ Dir.mktmpdir do |dir|
176
+ input = File.join(dir, 'x.xml.bz2')
177
+ File.binwrite(input, bzip2("<mediawiki>#{page_xml(id: 1, ns: 0, title: '東京', text: "'''東京'''。")}</mediawiki>"))
178
+ Open3.capture3(RbConfig.ruby, '-I', File.expand_path('../lib', __dir__),
179
+ File.expand_path('../bin/wp2txt', __dir__), '-i', input, '-o', dir, *options)
180
+ end
181
+ end
182
+
183
+ it '#10 rejects lead terms with ractor' do
184
+ _, err, status = cli('--format', 'json', '--lead-terms', '--ractor', '--no-turbo')
185
+ expect(status.success?).to be(false)
186
+ expect(err).to include('--lead-terms cannot be combined with --ractor')
187
+ end
188
+
189
+ it '#10 warns about missing source identifiers in ractor JSON without rejecting it' do
190
+ _, err, status = cli('--format', 'json', '--ractor', '--no-turbo', '-n', '1')
191
+ expect(status.success?).to be(true), err
192
+ expect(err).to include('does not include page IDs or revision IDs')
193
+ end
194
+ end