wp2txt 2.3.4 → 2.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -39,9 +39,10 @@ module Wp2txt
39
39
 
40
40
  # Normalize a page title the way MediaWiki treats titles:
41
41
  # underscores to spaces, trimmed, first letter capitalized
42
- def self.normalize_title(name)
42
+ def self.normalize_title(name, case_rule: "first-letter")
43
43
  n = name.to_s.tr("_", " ").strip.squeeze(" ")
44
44
  return n if n.empty?
45
+ return n if case_rule == "case-sensitive"
45
46
 
46
47
  n[0].upcase + n[1..].to_s
47
48
  end
@@ -161,6 +162,69 @@ module Wp2txt
161
162
  skipped_invalid: meta[:langlinks_skipped_invalid].to_i }
162
163
  end
163
164
 
165
+ # Provenance and counts of imported page properties, or nil
166
+ def page_props_provenance
167
+ return nil unless File.exist?(@db_path)
168
+
169
+ meta = read_metadata
170
+ return nil unless meta && meta[:page_props_imported_at] && page_properties_imported?
171
+
172
+ { source: meta[:page_props_source],
173
+ source_size: meta[:page_props_source_size].to_i,
174
+ source_sha256: meta[:page_props_source_sha256],
175
+ imported_at: meta[:page_props_imported_at],
176
+ imported_with: meta[:page_props_wp2txt_version],
177
+ page_count: meta[:page_props_page_count].to_i,
178
+ qid_count: meta[:page_props_qid_count].to_i,
179
+ disambiguation_count: meta[:page_props_disambiguation_count].to_i,
180
+ sort_key_count: meta[:page_props_sort_key_count].to_i,
181
+ skipped_invalid_sort_keys: meta[:page_props_skipped_invalid_sort_keys].to_i }
182
+ end
183
+
184
+ # How incoming links were counted (wp2txt --count-links), or nil
185
+ def links_provenance
186
+ return nil unless File.exist?(@db_path)
187
+
188
+ meta = read_metadata
189
+ return nil unless meta && meta[:links_counted_at]
190
+
191
+ { counted_at: meta[:links_counted_at],
192
+ rule_version: meta[:links_rule_version],
193
+ counted_with: meta[:links_wp2txt_version],
194
+ article_count: meta[:links_article_count].to_i,
195
+ with_inlinks: meta[:links_with_inlinks].to_i }
196
+ end
197
+
198
+ # An incomplete import must not be reported as an absent property.
199
+ def self.page_properties_imported?(db)
200
+ !!db.get_first_value(<<~SQL)
201
+ SELECT 1 FROM sqlite_master WHERE type = 'table' AND name = 'page_properties'
202
+ AND EXISTS (SELECT 1 FROM metadata WHERE key = 'page_props_imported_at')
203
+ SQL
204
+ end
205
+
206
+ def page_properties_imported?
207
+ File.exist?(@db_path) && self.class.page_properties_imported?(open_db)
208
+ rescue SQLite3::Exception
209
+ false
210
+ end
211
+
212
+ def self.property_values(row = nil)
213
+ qid, sort_key, disambiguation = row
214
+ { qid: qid, sort_key: sort_key, disambiguation: disambiguation == 1 }
215
+ end
216
+
217
+ # nil means not imported; null/false values mean imported but absent.
218
+ def properties_for(page_id)
219
+ return nil unless page_properties_imported?
220
+
221
+ self.class.property_values(open_db.get_first_row(
222
+ "SELECT qid, sort_key, disambiguation FROM page_properties WHERE page_id = ?", [page_id]
223
+ ))
224
+ rescue SQLite3::Exception
225
+ nil
226
+ end
227
+
164
228
  def close
165
229
  @db&.close
166
230
  @db = nil
@@ -230,7 +294,7 @@ module Wp2txt
230
294
  end
231
295
  end
232
296
 
233
- def finalize_build!(source_path)
297
+ def finalize_build!(source_path, case_rule: "first-letter")
234
298
  db = open_db
235
299
  db.execute("CREATE INDEX IF NOT EXISTS idx_pages_title ON pages(title)")
236
300
  db.execute("CREATE INDEX IF NOT EXISTS idx_pc_category ON page_categories(category)")
@@ -248,6 +312,7 @@ module Wp2txt
248
312
  source_size: stat.size,
249
313
  source_mtime: stat.mtime.to_i,
250
314
  dump_name: dump_name,
315
+ case_rule: case_rule,
251
316
  built_at: Time.now.utc.iso8601
252
317
  )
253
318
  db.execute("ANALYZE")
@@ -629,6 +694,7 @@ module Wp2txt
629
694
  # Build the index. Yields (batches_done, batches_total) after each batch.
630
695
  # @return [MetadataIndex] the built index
631
696
  def build(&progress)
697
+ case_rule = case_rule_from_header
632
698
  index = MetadataIndex.new(@db_path)
633
699
  index.prepare_build!
634
700
  # Close before Parallel forks workers so children do not inherit a
@@ -652,10 +718,20 @@ module Wp2txt
652
718
  self.class.scan_batch(@multistream_path, batch)
653
719
  end
654
720
 
655
- index.finalize_build!(@multistream_path)
721
+ index.finalize_build!(@multistream_path, case_rule: case_rule)
656
722
  index
657
723
  end
658
724
 
725
+ # The siteinfo may occupy its own stream before the first page offset.
726
+ def case_rule_from_header
727
+ return "first-letter" if @stream_offsets.empty?
728
+
729
+ header_end = @stream_offsets.first.positive? ? @stream_offsets.first : @stream_offsets[1]
730
+ xml = self.class.decompress_bz2(File.binread(@multistream_path, header_end))
731
+ siteinfo = xml[%r{<siteinfo\b[^>]*>(.*?)</siteinfo>}m, 1].to_s
732
+ siteinfo[%r{<case>\s*(first-letter|case-sensitive)\s*</case>}, 1] || "first-letter"
733
+ end
734
+
659
735
  # Scan a batch of [offset, next_offset] stream pairs.
660
736
  # Runs inside worker processes: must not touch SQLite.
661
737
  def self.scan_batch(multistream_path, offset_pairs)
@@ -866,6 +866,27 @@ module Wp2txt
866
866
  File.join(@cache_dir, "#{@lang}wiki-#{date}-langlinks.sql.gz")
867
867
  end
868
868
 
869
+ def page_props_url(date)
870
+ wiki = "#{@lang}wiki"
871
+ "#{DUMP_BASE_URL}/#{wiki}/#{date}/#{wiki}-#{date}-page_props.sql.gz"
872
+ end
873
+
874
+ def cached_page_props_path(date)
875
+ File.join(@cache_dir, "#{@lang}wiki-#{date}-page_props.sql.gz")
876
+ end
877
+
878
+ # Download the page_props dump for an explicit dump date (must match the index)
879
+ def download_page_props(date:, force: false)
880
+ path = cached_page_props_path(date)
881
+ return path if File.exist?(path) && !force
882
+
883
+ url = page_props_url(date)
884
+ puts "Downloading page_props: #{url}"
885
+ $stdout.flush
886
+ download_file(url, path)
887
+ path
888
+ end
889
+
869
890
  # Download the langlinks dump for an explicit dump date. The date MUST
870
891
  # come from the built metadata index (not latest_dump_date) so the
871
892
  # imported links stay pinned to the indexed dump version.
@@ -0,0 +1,170 @@
1
+ # frozen_string_literal: true
2
+
3
+ require "sqlite3"
4
+ require "time"
5
+ require "digest"
6
+ require_relative "sql_dump_reader"
7
+ require_relative "version"
8
+
9
+ module Wp2txt
10
+ # Imports Wikidata IDs, disambiguation flags, and default sort keys
11
+ # from the official page_props dump into the metadata index, as
12
+ # page_properties. Like langlinks, the dump must carry the same
13
+ # date as the index: IDs are only meaningful for the pages they came with.
14
+ class PagePropsImporter
15
+ BATCH_SIZE = 50_000
16
+
17
+ # (pp_page,'pp_propname','pp_value',pp_sortkey) — string classes accept
18
+ # escaped quotes and backslashes; sortkey is NULL or a number
19
+ TUPLE_REGEX = /\((\d+),'((?:[^'\\]|\\.)*)','((?:[^'\\]|\\.)*)',(?:NULL|[-+.\deE]+)\)/
20
+ QID_REGEX = /\AQ\d+\z/
21
+
22
+ def initialize(db_path)
23
+ @db_path = db_path
24
+ end
25
+
26
+ # "jawiki-20260901-page_props.sql.gz" => "jawiki-20260901"
27
+ def self.dump_name_of(path)
28
+ File.basename(path)[/\A[a-z0-9_\-]+?-\d{8}/]
29
+ end
30
+
31
+ # @return [Hash] { status: :imported | :already_imported, row_count:, provenance: }
32
+ def import!(source_path, force: false)
33
+ raise ArgumentError, "page_props file not found: #{source_path}" unless File.exist?(source_path)
34
+
35
+ db = SQLite3::Database.new(@db_path)
36
+ db.busy_timeout = 5000
37
+ dump_name = metadata_value(db, "dump_name")
38
+ raise ArgumentError, "metadata index is not built: #{@db_path}" unless dump_name
39
+
40
+ source_dump = self.class.dump_name_of(source_path)
41
+ unless source_dump && source_dump == dump_name
42
+ raise ArgumentError,
43
+ "dump version mismatch: the metadata index is #{dump_name} but the page_props file is " \
44
+ "#{source_dump || File.basename(source_path)} (versions must match; there is no override)"
45
+ end
46
+
47
+ if !force && (existing = imported_at(db))
48
+ return { status: :already_imported, imported_at: existing,
49
+ row_count: db.get_first_value("SELECT COUNT(*) FROM page_properties").to_i, provenance: read_provenance(db) }
50
+ end
51
+
52
+ db.execute("DROP TABLE IF EXISTS page_qids") # replace the unreleased QID-only schema
53
+ db.execute("DROP TABLE IF EXISTS page_properties")
54
+ # A failed load must leave the index looking "not imported"
55
+ db.execute("DELETE FROM metadata WHERE key LIKE 'page\\_props\\_%' ESCAPE '\\'")
56
+ db.execute(<<~SQL)
57
+ CREATE TABLE page_properties (
58
+ page_id INTEGER PRIMARY KEY,
59
+ qid TEXT,
60
+ disambiguation INTEGER NOT NULL DEFAULT 0,
61
+ sort_key TEXT
62
+ )
63
+ SQL
64
+
65
+ rows_seen = 0
66
+ skipped_invalid = 0
67
+ batch = []
68
+ flush = lambda do
69
+ db.transaction do
70
+ stmt = db.prepare(<<~SQL)
71
+ INSERT INTO page_properties (page_id, qid, disambiguation, sort_key) VALUES (?, ?, ?, ?)
72
+ ON CONFLICT(page_id) DO UPDATE SET
73
+ qid = COALESCE(excluded.qid, page_properties.qid),
74
+ disambiguation = MAX(excluded.disambiguation, page_properties.disambiguation),
75
+ sort_key = COALESCE(excluded.sort_key, page_properties.sort_key)
76
+ SQL
77
+ batch.each { |row| stmt.execute(row) }
78
+ stmt.close
79
+ end
80
+ batch.clear
81
+ end
82
+
83
+ SqlDumpReader.each_insert_line(source_path, "page_props") do |line|
84
+ line.scan(TUPLE_REGEX) do |page, name, value|
85
+ rows_seen += 1
86
+ row = [page.to_i, nil, 0, nil]
87
+ case name
88
+ when "wikibase_item"
89
+ qid = SqlDumpReader.unescape(value)
90
+ next unless QID_REGEX.match?(qid)
91
+
92
+ row[1] = qid.force_encoding(Encoding::UTF_8)
93
+ when "disambiguation"
94
+ row[2] = 1
95
+ when "defaultsort"
96
+ key = SqlDumpReader.unescape(value).force_encoding(Encoding::UTF_8)
97
+ unless key.valid_encoding?
98
+ skipped_invalid += 1
99
+ next
100
+ end
101
+ row[3] = key
102
+ else
103
+ next
104
+ end
105
+ batch << row
106
+ flush.call if batch.size >= BATCH_SIZE
107
+ end
108
+ end
109
+ flush.call unless batch.empty?
110
+
111
+ if rows_seen.zero?
112
+ raise Wp2txt::Error, "no page_props rows found in #{File.basename(source_path)}; " \
113
+ "the file may be empty or in an unrecognized format"
114
+ end
115
+
116
+ counts = db.get_first_row("SELECT COUNT(*), COUNT(qid), SUM(disambiguation), COUNT(sort_key) FROM page_properties")
117
+ row_count = counts[0]
118
+ stamp_provenance(db, source_path, counts, skipped_invalid)
119
+ { status: :imported, row_count: row_count, provenance: read_provenance(db) }
120
+ ensure
121
+ db&.close
122
+ end
123
+
124
+ private
125
+
126
+ def metadata_value(db, key)
127
+ db.get_first_value("SELECT value FROM metadata WHERE key = ?", [key])
128
+ rescue SQLite3::Exception
129
+ nil
130
+ end
131
+
132
+ def imported_at(db)
133
+ table = db.get_first_value("SELECT name FROM sqlite_master WHERE type = 'table' AND name = 'page_properties'")
134
+ table && metadata_value(db, "page_props_imported_at")
135
+ end
136
+
137
+ def stamp_provenance(db, source_path, counts, skipped_invalid)
138
+ values = {
139
+ page_props_source: File.basename(source_path),
140
+ page_props_source_size: File.size(source_path),
141
+ page_props_source_sha256: Digest::SHA256.file(source_path).hexdigest,
142
+ page_props_imported_at: Time.now.utc.iso8601,
143
+ page_props_wp2txt_version: Wp2txt::VERSION,
144
+ page_props_page_count: counts[0],
145
+ page_props_qid_count: counts[1],
146
+ page_props_disambiguation_count: counts[2].to_i,
147
+ page_props_sort_key_count: counts[3],
148
+ page_props_skipped_invalid_sort_keys: skipped_invalid
149
+ }
150
+ stmt = db.prepare("INSERT OR REPLACE INTO metadata (key, value) VALUES (?, ?)")
151
+ values.each { |k, v| stmt.execute([k.to_s, v.to_s]) }
152
+ stmt.close
153
+ end
154
+
155
+ def read_provenance(db)
156
+ {
157
+ source: metadata_value(db, "page_props_source"),
158
+ source_size: metadata_value(db, "page_props_source_size").to_i,
159
+ source_sha256: metadata_value(db, "page_props_source_sha256"),
160
+ imported_at: metadata_value(db, "page_props_imported_at"),
161
+ imported_with: metadata_value(db, "page_props_wp2txt_version"),
162
+ page_count: metadata_value(db, "page_props_page_count").to_i,
163
+ qid_count: metadata_value(db, "page_props_qid_count").to_i,
164
+ disambiguation_count: metadata_value(db, "page_props_disambiguation_count").to_i,
165
+ sort_key_count: metadata_value(db, "page_props_sort_key_count").to_i,
166
+ skipped_invalid_sort_keys: metadata_value(db, "page_props_skipped_invalid_sort_keys").to_i
167
+ }
168
+ end
169
+ end
170
+ end
@@ -0,0 +1,57 @@
1
+ # frozen_string_literal: true
2
+
3
+ require "zlib"
4
+
5
+ module Wp2txt
6
+ # Streams the rows of one table's INSERT statements out of an official
7
+ # MySQL table dump (.sql or .sql.gz), without loading the file whole.
8
+ #
9
+ # Dumps write each INSERT either on one line or with the header on its own
10
+ # line and one tuple per following line (current dumps do the latter), so
11
+ # every line of a statement is read up to its terminating semicolon.
12
+ # Statements for other tables are skipped.
13
+ #
14
+ # Lines are read as BINARY: string columns are VARBINARY and real dumps
15
+ # contain historically corrupted bytes. Callers match tuples with ASCII-only
16
+ # patterns and validate the captures themselves.
17
+ module SqlDumpReader
18
+ STATEMENT_START = /\A\s*INSERT\s+INTO\s/i
19
+
20
+ module_function
21
+
22
+ # @yield [String] each line (BINARY) that belongs to an INSERT statement for table
23
+ def each_insert_line(source_path, table)
24
+ header = /\A\s*INSERT\s+INTO\s+`#{Regexp.escape(table)}`\s+VALUES\b/i
25
+ io = if source_path.end_with?(".gz")
26
+ # GzipReader ignores set_encoding; the encoding must be given at open time
27
+ Zlib::GzipReader.open(source_path, encoding: Encoding::BINARY.to_s)
28
+ else
29
+ File.open(source_path, "rb")
30
+ end
31
+ inside = false
32
+ begin
33
+ io.each_line do |line|
34
+ inside = header.match?(line) if STATEMENT_START.match?(line)
35
+ next unless inside
36
+
37
+ inside = false if line.rstrip.end_with?(";")
38
+ yield line
39
+ end
40
+ ensure
41
+ io.close
42
+ end
43
+ end
44
+
45
+ UNESCAPE_REGEX = /\\(.)/m
46
+ # MySQL backslash escapes inside mysqldump string literals
47
+ UNESCAPES = {
48
+ "0" => "\0", "'" => "'", '"' => '"', "b" => "\b", "n" => "\n",
49
+ "r" => "\r", "t" => "\t", "Z" => "\x1A", "\\" => "\\"
50
+ }.freeze
51
+
52
+ # Undo MySQL string escaping (\' \\ \n ...) in a captured value
53
+ def unescape(value)
54
+ value.gsub(UNESCAPE_REGEX) { UNESCAPES.fetch(Regexp.last_match(1), Regexp.last_match(1)) }
55
+ end
56
+ end
57
+ end
@@ -19,8 +19,12 @@ module Wp2txt
19
19
 
20
20
  attr_reader :buffer_size, :pages_processed, :bytes_read, :redirects_skipped
21
21
 
22
- def initialize(input_path, bz2_gem: false, adaptive_buffer: true, validate_bz2: true, skip_redirects: true)
22
+ # keep_raw_text: also hand back each article's wikitext as stored in the
23
+ # dump (before comment removal), as :raw_text among the with_ids values
24
+ def initialize(input_path, bz2_gem: false, adaptive_buffer: true, validate_bz2: true, skip_redirects: true,
25
+ keep_raw_text: false)
23
26
  @input_path = input_path
27
+ @keep_raw_text = keep_raw_text
24
28
  @bz2_gem = bz2_gem
25
29
  @buffer = +""
26
30
  @pending_bytes = +"".b
@@ -257,6 +261,7 @@ module Wp2txt
257
261
  return nil unless Wp2txt.namespace_id(namespace).zero?
258
262
 
259
263
  text = text_node.content
264
+ raw_text = text if @keep_raw_text
260
265
 
261
266
  # Early redirect detection and skip (before expensive processing)
262
267
  # Redirects start with # or # followed by redirect keyword and [[target]]
@@ -272,7 +277,9 @@ module Wp2txt
272
277
  end
273
278
 
274
279
  @pages_processed += 1
275
- [title, text, Wp2txt.page_ids(page_xml)]
280
+ meta = Wp2txt.page_ids(page_xml)
281
+ meta[:raw_text] = raw_text if @keep_raw_text
282
+ [title, text, meta]
276
283
  rescue Nokogiri::XML::SyntaxError
277
284
  # Skip malformed XML
278
285
  nil
@@ -171,6 +171,19 @@ module Wp2txt
171
171
  nil
172
172
  end
173
173
 
174
+ # Templates whose rendering the text-cleaning stage defines (see
175
+ # Wp2txt#correct_inline_template), keyed by normalized name. Limited to
176
+ # kinds whose output does not depend on marker settings.
177
+ def rendered_by_cleaner?(template_name)
178
+ @rendered_by_cleaner ||= [Wp2txt::RUBY_TEXT_TEMPLATES, Wp2txt::INTERWIKI_LINK_TEMPLATES]
179
+ .flatten.to_set { |name| name.to_s.tr("_", " ").strip.downcase }
180
+ @rendered_by_cleaner.include?(template_name.tr("_", " "))
181
+ end
182
+
183
+ def text_renderer
184
+ @text_renderer ||= Object.new.extend(Wp2txt)
185
+ end
186
+
174
187
  def expand_single_template(content)
175
188
  parts = split_template_parts(content)
176
189
  return "" if parts.empty?
@@ -263,6 +276,12 @@ module Wp2txt
263
276
  # Handle lang-xx templates (e.g., lang-fr, lang-de, lang-ja)
264
277
  if template_name.start_with?("lang-")
265
278
  expand_lang_xx(template_name, params)
279
+ elsif rendered_by_cleaner?(template_name)
280
+ # These carry words the article needs (a reading, a link's display
281
+ # text). Deleting them dropped those words; leaving the raw template
282
+ # for later confused links that contain it (an image caption with a
283
+ # "|" inside). Render them now, with the cleaner's own rules.
284
+ text_renderer.correct_inline_template("{{#{expand(content)}}}")
266
285
  else
267
286
  @preserve_unknown ? "{{#{content}}}" : ""
268
287
  end
data/lib/wp2txt/utils.rb CHANGED
@@ -1,6 +1,7 @@
1
1
  # frozen_string_literal: true
2
2
 
3
3
  require "strscan"
4
+ require_relative "wikitext_regions"
4
5
  require_relative "constants"
5
6
  require_relative "regex"
6
7
  require_relative "text_processing"
@@ -574,8 +575,9 @@ module Wp2txt
574
575
  # Helper to check if template name matches any in a list (case-insensitive)
575
576
  def template_matches?(name, template_list)
576
577
  return false if template_list.nil? || template_list.empty?
577
- normalized_name = name.to_s.strip.downcase
578
- template_list.any? { |t| t.downcase == normalized_name }
578
+ # MediaWiki treats "_" and " " in template names as the same character
579
+ normalized_name = name.to_s.tr("_", " ").strip.downcase
580
+ template_list.any? { |t| t.tr("_", " ").downcase == normalized_name }
579
581
  end
580
582
 
581
583
  def correct_inline_template(str, enabled_markers = [], extract_citations = false)
@@ -583,7 +585,7 @@ module Wp2txt
583
585
  return str unless str.include?("{{")
584
586
 
585
587
  process_nested_single_pass(str, "{{", "}}") do |contents|
586
- parts = contents.split("|")
588
+ parts = WikitextRegions.split_pipes(contents)
587
589
  template_name = (parts[0] || "").strip
588
590
  template_name_lower = template_name.downcase
589
591
 
@@ -1,5 +1,5 @@
1
1
  # frozen_string_literal: true
2
2
 
3
3
  module Wp2txt
4
- VERSION = "2.3.4"
4
+ VERSION = "2.4.0"
5
5
  end
@@ -0,0 +1,66 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Wp2txt
4
+ # Literal regions suppress wikitext parsing. Galleries and timelines can
5
+ # contain links, but their contents are not the article's lead prose.
6
+ module WikitextRegions
7
+ LITERAL_TAGS = %w[nowiki pre math chem ce score syntaxhighlight source graph mapframe templatedata].freeze
8
+ NON_PROSE_TAGS = %w[gallery timeline].freeze
9
+ COMMENT = /<!--.*?(?:-->|\z)/m
10
+
11
+ def self.region_pattern(tags)
12
+ /<!--.*?(?:-->|\z)|<(?<tag>#{tags.join('|')})(?=[\s\/>])(?:"[^"]*"|'[^']*'|[^'">])*?(?:\/>|>.*?(?:<\/\k<tag>\s*>|\z))/mi
13
+ end
14
+
15
+ LITERAL_REGION = region_pattern(LITERAL_TAGS)
16
+ LEAD_REGION = region_pattern(LITERAL_TAGS + NON_PROSE_TAGS)
17
+ LITERAL_REGION_AT = /\G(?:#{LITERAL_REGION})/
18
+ LEAD_REGION_AT = /\G(?:#{LEAD_REGION})/
19
+
20
+ module_function
21
+
22
+ def end_at(text, offset, lead: false)
23
+ match = (lead ? LEAD_REGION_AT : LITERAL_REGION_AT).match(text, offset)
24
+ match.end(0) if match
25
+ end
26
+
27
+ def remove_literal(text)
28
+ text.gsub(LITERAL_REGION, "")
29
+ end
30
+
31
+ # One space per codepoint, including internal newlines: an excluded
32
+ # region must not introduce a paragraph or heading boundary.
33
+ def mask_lead(text)
34
+ text.gsub(LEAD_REGION) { |region| " " * region.length }
35
+ end
36
+
37
+ def split_pipes(text)
38
+ parts = [+""]
39
+ stack = []
40
+ i = 0
41
+ while i < text.length
42
+ two = text[i, 2]
43
+ if text[i] == "<" && (stop = end_at(text, i))
44
+ parts.last << text[i...stop]
45
+ i = stop
46
+ elsif ["{{", "[["].include?(two)
47
+ stack << (two == "{{" ? "}}" : "]]")
48
+ parts.last << two
49
+ i += 2
50
+ elsif !stack.empty? && two == stack.last
51
+ stack.pop
52
+ parts.last << two
53
+ i += 2
54
+ else
55
+ if text[i] == "|" && stack.empty?
56
+ parts << +""
57
+ else
58
+ parts.last << text[i]
59
+ end
60
+ i += 1
61
+ end
62
+ end
63
+ parts
64
+ end
65
+ end
66
+ end
@@ -37,14 +37,14 @@ RSpec.describe "documentation surface sync" do
37
37
  expect(phantom).to be_empty, "documented tools that do not exist: #{phantom.join(', ')}"
38
38
  end
39
39
 
40
- # Tripwire: tracked files must not contain tokens listed in .private-doc-tokens,
40
+ # Tripwire: tracked files must not contain tokens listed in .local/doc-tokens,
41
41
  # an untracked, machine-local file (one substring per line; # starts a comment).
42
42
  # The file exists only on machines that maintain such a list; everywhere else
43
43
  # (CI, other contributors) this example skips — loudly, so a silently dead
44
44
  # check cannot be mistaken for a passing one.
45
45
  it "keeps machine-local private tokens out of tracked files" do
46
- token_file = File.join(repo_root, ".private-doc-tokens")
47
- skip "SKIPPED: no .private-doc-tokens on this machine — tripwire not checked" unless File.exist?(token_file)
46
+ token_file = File.join(repo_root, ".local", "doc-tokens")
47
+ skip "SKIPPED: no .local/doc-tokens on this machine — tripwire not checked" unless File.exist?(token_file)
48
48
 
49
49
  tokens = File.readlines(token_file, encoding: "UTF-8")
50
50
  .map(&:strip).reject { |t| t.empty? || t.start_with?("#") }
@@ -65,6 +65,33 @@ RSpec.describe Wp2txt::LanglinksImporter do
65
65
  rows
66
66
  end
67
67
 
68
+ # Current dumps put the INSERT header on its own line and one tuple per line
69
+ MULTILINE_SQL = <<~MSQL
70
+ INSERT INTO `langlinks` VALUES
71
+ (1,'en','Film A'),
72
+ (1,'ja','映画A'),
73
+ (2,'en','It\\'s a Film; Really');
74
+ INSERT INTO `pagelinks` VALUES
75
+ (9,'Ignored',0);
76
+ INSERT INTO `langlinks` VALUES
77
+ (3,'fr','Film B');
78
+ MSQL
79
+
80
+ describe "dumps with one tuple per line" do
81
+ it "imports every tuple of every langlinks statement, and nothing else" do
82
+ path = write_langlinks("testwiki-20260101-langlinks.sql.gz", MULTILINE_SQL, gzip: true)
83
+ expect(import(path)[:row_count]).to eq(4)
84
+ expect(langlinks_rows).to contain_exactly(
85
+ [1, "en", "Film A"], [1, "ja", "映画A"], [2, "en", "It's a Film; Really"], [3, "fr", "Film B"]
86
+ )
87
+ end
88
+
89
+ it "refuses to report success when no rows could be read" do
90
+ path = write_langlinks("testwiki-20260101-langlinks.sql", "-- nothing here\nUNLOCK TABLES;\n")
91
+ expect { import(path) }.to raise_error(Wp2txt::Error, /no langlinks rows found/)
92
+ end
93
+ end
94
+
68
95
  describe "parsing and normalization" do
69
96
  it "imports tuples with escapes, commas, parens, and multiple INSERT statements" do
70
97
  path = write_langlinks("testwiki-20260101-langlinks.sql")