wp2txt 2.3.3 → 2.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (45) hide show
  1. checksums.yaml +4 -4
  2. data/.dockerignore +0 -4
  3. data/.gitignore +3 -5
  4. data/CHANGELOG.md +18 -0
  5. data/DEVELOPMENT.md +1 -1
  6. data/DEVELOPMENT_ja.md +1 -1
  7. data/README.md +44 -2
  8. data/README_ja.md +35 -2
  9. data/Rakefile +10 -21
  10. data/bin/wp2txt +79 -17
  11. data/bin/wp2txt-mcp +1 -1
  12. data/docs/INDEXES.md +61 -1
  13. data/lib/wp2txt/article.rb +1 -1
  14. data/lib/wp2txt/cli.rb +42 -0
  15. data/lib/wp2txt/constants.rb +24 -0
  16. data/lib/wp2txt/corpus.rb +11 -2
  17. data/lib/wp2txt/data/template_aliases.json +1 -1
  18. data/lib/wp2txt/extractor.rb +10 -1
  19. data/lib/wp2txt/formatter.rb +20 -0
  20. data/lib/wp2txt/index_commands.rb +89 -1
  21. data/lib/wp2txt/langlinks_importer.rb +19 -35
  22. data/lib/wp2txt/lead_terms.rb +228 -0
  23. data/lib/wp2txt/link_counter.rb +171 -0
  24. data/lib/wp2txt/metadata_index.rb +79 -3
  25. data/lib/wp2txt/multistream.rb +52 -2
  26. data/lib/wp2txt/output_writer.rb +8 -0
  27. data/lib/wp2txt/page_props_importer.rb +170 -0
  28. data/lib/wp2txt/sql_dump_reader.rb +57 -0
  29. data/lib/wp2txt/stream_processor.rb +42 -15
  30. data/lib/wp2txt/template_expander.rb +19 -0
  31. data/lib/wp2txt/utils.rb +5 -3
  32. data/lib/wp2txt/version.rb +1 -1
  33. data/lib/wp2txt/wikitext_regions.rb +66 -0
  34. data/lib/wp2txt.rb +7 -5
  35. data/spec/docs_sync_spec.rb +3 -3
  36. data/spec/langlinks_importer_spec.rb +27 -0
  37. data/spec/lead_terms_edge_cases_spec.rb +194 -0
  38. data/spec/lead_terms_links_qids_spec.rb +204 -0
  39. data/spec/output_integrity_spec.rb +146 -0
  40. data/spec/p1_correctness_spec.rb +32 -14
  41. data/spec/page_properties_spec.rb +161 -0
  42. data/spec/region_semantics_spec.rb +71 -0
  43. data/spec/template_passthrough_spec.rb +43 -0
  44. data/spec/titles_output_path_spec.rb +12 -0
  45. metadata +18 -1
@@ -0,0 +1,171 @@
1
+ # frozen_string_literal: true
2
+
3
+ require "parallel"
4
+ require "sqlite3"
5
+ require "time"
6
+ require_relative "metadata_index"
7
+ require_relative "version"
8
+ require_relative "text_processing"
9
+ require_relative "wikitext_regions"
10
+
11
+ module Wp2txt
12
+ # Counts, for every article, how many other articles link to it, and stores
13
+ # the result in the metadata index as page_inlinks(page_id, inlinks,
14
+ # via_redirects).
15
+ #
16
+ # Rules (recorded as RULE_VERSION so counts from different rules are never
17
+ # compared unknowingly):
18
+ # - sources are articles (namespace 0) that are not redirects
19
+ # - each source counts once per target, however many times it links there
20
+ # - a link to a redirect counts for the redirect's target (one hop);
21
+ # via_redirects is how many sources reached the target only that way
22
+ # - only links written in the article's own wikitext count; links that
23
+ # templates add when rendered (navigation boxes) are not in the dump text
24
+ # - commented-out links do not count
25
+ class LinkCounter
26
+ include Wp2txt
27
+
28
+ RULE_VERSION = "2"
29
+ STREAMS_PER_BATCH = 50
30
+ LINK_REGEX = /\[\[([^\[\]|<>{}\n]+)(?=\||\]\])/
31
+ COMMENT_REGEX = /<!--.*?-->/m
32
+
33
+ def initialize(multistream_path, stream_offsets, db_path:, num_processes: 4)
34
+ @multistream_path = multistream_path
35
+ @stream_offsets = stream_offsets
36
+ @db_path = db_path
37
+ @num_processes = num_processes
38
+ end
39
+
40
+ # @return [Hash] { articles:, with_inlinks:, rule_version: }
41
+ def count!(&progress)
42
+ titles, redirects = load_titles
43
+ # Inherited by the forked workers (copy-on-write); they only read it
44
+ @redirects = redirects
45
+
46
+ direct = Hash.new(0)
47
+ via = Hash.new(0)
48
+ pairs = @stream_offsets.zip(@stream_offsets[1..].to_a + [nil])
49
+ batches = pairs.each_slice(STREAMS_PER_BATCH).to_a
50
+ done = 0
51
+
52
+ Parallel.each(
53
+ batches,
54
+ in_processes: @num_processes,
55
+ finish: lambda { |_item, _idx, result|
56
+ result[:direct].each { |t, n| direct[t] += n }
57
+ result[:via].each { |t, n| via[t] += n }
58
+ done += 1
59
+ progress&.call(done, batches.size)
60
+ }
61
+ ) do |batch|
62
+ scan_batch(batch)
63
+ end
64
+
65
+ write(titles, direct, via)
66
+ end
67
+
68
+ private
69
+
70
+ # Articles (title => page_id) and redirects (title => target title)
71
+ def load_titles
72
+ db = SQLite3::Database.new(@db_path, readonly: true)
73
+ @case_rule = db.get_first_value("SELECT value FROM metadata WHERE key = 'case_rule'") || "first-letter"
74
+ titles = {}
75
+ redirects = {}
76
+ db.execute("SELECT page_id, title, redirect_to FROM pages WHERE namespace = 0") do |id, title, target|
77
+ if target
78
+ redirects[title] = normalize_target(target)
79
+ else
80
+ titles[title] = id
81
+ end
82
+ end
83
+ [titles, redirects]
84
+ ensure
85
+ db&.close
86
+ end
87
+
88
+ # Runs in a worker: returns per-target counts for this batch of streams
89
+ def scan_batch(offset_pairs)
90
+ direct = Hash.new(0)
91
+ via = Hash.new(0)
92
+ File.open(@multistream_path, "rb") do |f|
93
+ offset_pairs.each do |offset, next_offset|
94
+ f.seek(offset)
95
+ data = next_offset ? f.read(next_offset - offset) : f.read
96
+ xml = MetadataIndexBuilder.decompress_bz2(data)
97
+ xml.scan(MetadataIndexBuilder::PAGE_BLOCK_REGEX) do
98
+ count_page(::Regexp.last_match(1), direct, via)
99
+ end
100
+ end
101
+ end
102
+ { direct: direct, via: via }
103
+ end
104
+
105
+ def count_page(block, direct, via)
106
+ return unless Wp2txt.namespace_id(block[MetadataIndexBuilder::NS_REGEX, 1]).zero?
107
+
108
+ text = MetadataIndexBuilder.unescape_xml(block[MetadataIndexBuilder::TEXT_REGEX, 1] || "")
109
+ return if Wp2txt::REDIRECT_REGEX.match?(text)
110
+
111
+ reached = {} # target => true if reached directly at least once
112
+ WikitextRegions.remove_literal(text).scan(LINK_REGEX) do |(raw)|
113
+ name = normalize_target(raw)
114
+ next if name.empty?
115
+
116
+ if (target = @redirects[name])
117
+ reached[target] ||= false
118
+ else
119
+ reached[name] = true
120
+ end
121
+ end
122
+ reached.each do |target, directly|
123
+ direct[target] += 1
124
+ via[target] += 1 unless directly
125
+ end
126
+ end
127
+
128
+ def normalize_target(raw)
129
+ # Decode each reference in the original input, never references created
130
+ # by decoding another one (special_chr has two decoding stages).
131
+ decoded = raw.gsub(/&(?:#[xX][0-9a-fA-F]+|#\d+|[a-zA-Z][a-zA-Z0-9]*);/) { |entity| special_chr(entity) }
132
+ title = decoded.split("#", 2).first.to_s.strip.sub(/\A:/, "")
133
+ MetadataIndex.normalize_title(title, case_rule: @case_rule || "first-letter")
134
+ end
135
+
136
+ def write(titles, direct, via)
137
+ db = SQLite3::Database.new(@db_path)
138
+ db.busy_timeout = 5000
139
+ db.execute("DROP TABLE IF EXISTS page_inlinks")
140
+ db.execute("DELETE FROM metadata WHERE key LIKE 'links\\_%' ESCAPE '\\'")
141
+ db.execute(<<~SQL)
142
+ CREATE TABLE page_inlinks (
143
+ page_id INTEGER PRIMARY KEY,
144
+ inlinks INTEGER NOT NULL,
145
+ via_redirects INTEGER NOT NULL
146
+ )
147
+ SQL
148
+ with_inlinks = 0
149
+ db.transaction do
150
+ stmt = db.prepare("INSERT INTO page_inlinks (page_id, inlinks, via_redirects) VALUES (?, ?, ?)")
151
+ titles.each do |title, page_id|
152
+ n = direct[title]
153
+ with_inlinks += 1 if n.positive?
154
+ stmt.execute([page_id, n, via[title]])
155
+ end
156
+ stmt.close
157
+ end
158
+ db.execute("CREATE INDEX idx_page_inlinks_count ON page_inlinks(inlinks)")
159
+ {
160
+ links_counted_at: Time.now.utc.iso8601,
161
+ links_rule_version: RULE_VERSION,
162
+ links_wp2txt_version: Wp2txt::VERSION,
163
+ links_article_count: titles.size,
164
+ links_with_inlinks: with_inlinks
165
+ }.each { |k, v| db.execute("INSERT OR REPLACE INTO metadata (key, value) VALUES (?, ?)", [k.to_s, v.to_s]) }
166
+ { articles: titles.size, with_inlinks: with_inlinks, rule_version: RULE_VERSION }
167
+ ensure
168
+ db&.close
169
+ end
170
+ end
171
+ end
@@ -39,9 +39,10 @@ module Wp2txt
39
39
 
40
40
  # Normalize a page title the way MediaWiki treats titles:
41
41
  # underscores to spaces, trimmed, first letter capitalized
42
- def self.normalize_title(name)
42
+ def self.normalize_title(name, case_rule: "first-letter")
43
43
  n = name.to_s.tr("_", " ").strip.squeeze(" ")
44
44
  return n if n.empty?
45
+ return n if case_rule == "case-sensitive"
45
46
 
46
47
  n[0].upcase + n[1..].to_s
47
48
  end
@@ -161,6 +162,69 @@ module Wp2txt
161
162
  skipped_invalid: meta[:langlinks_skipped_invalid].to_i }
162
163
  end
163
164
 
165
+ # Provenance and counts of imported page properties, or nil
166
+ def page_props_provenance
167
+ return nil unless File.exist?(@db_path)
168
+
169
+ meta = read_metadata
170
+ return nil unless meta && meta[:page_props_imported_at] && page_properties_imported?
171
+
172
+ { source: meta[:page_props_source],
173
+ source_size: meta[:page_props_source_size].to_i,
174
+ source_sha256: meta[:page_props_source_sha256],
175
+ imported_at: meta[:page_props_imported_at],
176
+ imported_with: meta[:page_props_wp2txt_version],
177
+ page_count: meta[:page_props_page_count].to_i,
178
+ qid_count: meta[:page_props_qid_count].to_i,
179
+ disambiguation_count: meta[:page_props_disambiguation_count].to_i,
180
+ sort_key_count: meta[:page_props_sort_key_count].to_i,
181
+ skipped_invalid_sort_keys: meta[:page_props_skipped_invalid_sort_keys].to_i }
182
+ end
183
+
184
+ # How incoming links were counted (wp2txt --count-links), or nil
185
+ def links_provenance
186
+ return nil unless File.exist?(@db_path)
187
+
188
+ meta = read_metadata
189
+ return nil unless meta && meta[:links_counted_at]
190
+
191
+ { counted_at: meta[:links_counted_at],
192
+ rule_version: meta[:links_rule_version],
193
+ counted_with: meta[:links_wp2txt_version],
194
+ article_count: meta[:links_article_count].to_i,
195
+ with_inlinks: meta[:links_with_inlinks].to_i }
196
+ end
197
+
198
+ # An incomplete import must not be reported as an absent property.
199
+ def self.page_properties_imported?(db)
200
+ !!db.get_first_value(<<~SQL)
201
+ SELECT 1 FROM sqlite_master WHERE type = 'table' AND name = 'page_properties'
202
+ AND EXISTS (SELECT 1 FROM metadata WHERE key = 'page_props_imported_at')
203
+ SQL
204
+ end
205
+
206
+ def page_properties_imported?
207
+ File.exist?(@db_path) && self.class.page_properties_imported?(open_db)
208
+ rescue SQLite3::Exception
209
+ false
210
+ end
211
+
212
+ def self.property_values(row = nil)
213
+ qid, sort_key, disambiguation = row
214
+ { qid: qid, sort_key: sort_key, disambiguation: disambiguation == 1 }
215
+ end
216
+
217
+ # nil means not imported; null/false values mean imported but absent.
218
+ def properties_for(page_id)
219
+ return nil unless page_properties_imported?
220
+
221
+ self.class.property_values(open_db.get_first_row(
222
+ "SELECT qid, sort_key, disambiguation FROM page_properties WHERE page_id = ?", [page_id]
223
+ ))
224
+ rescue SQLite3::Exception
225
+ nil
226
+ end
227
+
164
228
  def close
165
229
  @db&.close
166
230
  @db = nil
@@ -230,7 +294,7 @@ module Wp2txt
230
294
  end
231
295
  end
232
296
 
233
- def finalize_build!(source_path)
297
+ def finalize_build!(source_path, case_rule: "first-letter")
234
298
  db = open_db
235
299
  db.execute("CREATE INDEX IF NOT EXISTS idx_pages_title ON pages(title)")
236
300
  db.execute("CREATE INDEX IF NOT EXISTS idx_pc_category ON page_categories(category)")
@@ -248,6 +312,7 @@ module Wp2txt
248
312
  source_size: stat.size,
249
313
  source_mtime: stat.mtime.to_i,
250
314
  dump_name: dump_name,
315
+ case_rule: case_rule,
251
316
  built_at: Time.now.utc.iso8601
252
317
  )
253
318
  db.execute("ANALYZE")
@@ -629,6 +694,7 @@ module Wp2txt
629
694
  # Build the index. Yields (batches_done, batches_total) after each batch.
630
695
  # @return [MetadataIndex] the built index
631
696
  def build(&progress)
697
+ case_rule = case_rule_from_header
632
698
  index = MetadataIndex.new(@db_path)
633
699
  index.prepare_build!
634
700
  # Close before Parallel forks workers so children do not inherit a
@@ -652,10 +718,20 @@ module Wp2txt
652
718
  self.class.scan_batch(@multistream_path, batch)
653
719
  end
654
720
 
655
- index.finalize_build!(@multistream_path)
721
+ index.finalize_build!(@multistream_path, case_rule: case_rule)
656
722
  index
657
723
  end
658
724
 
725
+ # The siteinfo may occupy its own stream before the first page offset.
726
+ def case_rule_from_header
727
+ return "first-letter" if @stream_offsets.empty?
728
+
729
+ header_end = @stream_offsets.first.positive? ? @stream_offsets.first : @stream_offsets[1]
730
+ xml = self.class.decompress_bz2(File.binread(@multistream_path, header_end))
731
+ siteinfo = xml[%r{<siteinfo\b[^>]*>(.*?)</siteinfo>}m, 1].to_s
732
+ siteinfo[%r{<case>\s*(first-letter|case-sensitive)\s*</case>}, 1] || "first-letter"
733
+ end
734
+
659
735
  # Scan a batch of [offset, next_offset] stream pairs.
660
736
  # Runs inside worker processes: must not touch SQLite.
661
737
  def self.scan_batch(multistream_path, offset_pairs)
@@ -95,6 +95,11 @@ module Wp2txt
95
95
  @early_terminated == true
96
96
  end
97
97
 
98
+ # Where the stream holding the last found article ends. Scanning stops as
99
+ # soon as every target is found, so without this the reader cannot tell how
100
+ # far that stream extends and would read the dump to its end.
101
+ attr_reader :stream_end_offset
102
+
98
103
  def find_by_title(title)
99
104
  @entries_by_title[title]
100
105
  end
@@ -179,6 +184,14 @@ module Wp2txt
179
184
  end
180
185
  end
181
186
 
187
+ def next_stream_offset_after(io, offset)
188
+ io.each_line do |line|
189
+ next_offset = line.split(":", 2).first.to_i
190
+ return next_offset if next_offset > offset
191
+ end
192
+ nil
193
+ end
194
+
182
195
  def parse_index_stream(io)
183
196
  count = 0
184
197
  io.each_line do |line|
@@ -205,6 +218,7 @@ module Wp2txt
205
218
  @found_targets << title if @target_titles.include?(title)
206
219
  if @found_targets.size == @target_titles.size
207
220
  @early_terminated = true
221
+ @stream_end_offset = next_stream_offset_after(io, offset)
208
222
  print "\r Found all #{@target_titles.size} target articles" if @show_progress
209
223
  puts if @show_progress
210
224
  break
@@ -223,6 +237,10 @@ module Wp2txt
223
237
 
224
238
  # Reads articles from multistream bz2 files
225
239
  class MultistreamReader
240
+ # A single multistream bz2 stream holds about 100 pages; anything this large
241
+ # past the last known offset is several streams, not one.
242
+ MAX_TAIL_STREAM_BYTES = 64 * 1024 * 1024
243
+
226
244
  attr_reader :multistream_path, :index
227
245
 
228
246
  # Initialize reader with multistream file and index
@@ -357,7 +375,14 @@ module Wp2txt
357
375
  if next_offset
358
376
  compressed_data = f.read(next_offset - offset)
359
377
  else
360
- # Last stream - read to end
378
+ # Last stream of the dump: the rest of the file is that one stream.
379
+ # A large remainder means the end was never recorded, and reading it
380
+ # in one call fails outright on some platforms (EINVAL on macOS).
381
+ remaining = File.size(@multistream_path) - offset
382
+ if remaining > MAX_TAIL_STREAM_BYTES
383
+ raise Wp2txt::Error, "cannot locate the end of the stream at offset #{offset} " \
384
+ "(#{remaining} bytes to end of file); the index may be incomplete"
385
+ end
361
386
  compressed_data = f.read
362
387
  end
363
388
 
@@ -372,7 +397,9 @@ module Wp2txt
372
397
  # here would silently read gigabytes to EOF, so fail fast instead
373
398
  raise "Stream offset #{current_offset} not found in index (#{offsets.size} streams known)" unless idx
374
399
 
375
- offsets[idx + 1]
400
+ return offsets[idx + 1] if idx + 1 < offsets.size
401
+
402
+ @index.respond_to?(:stream_end_offset) ? @index.stream_end_offset : nil
376
403
  end
377
404
 
378
405
  def decompress_bz2(data)
@@ -394,6 +421,7 @@ module Wp2txt
394
421
  return {
395
422
  title: page_title,
396
423
  id: page_node.at_xpath("id")&.text&.to_i,
424
+ revision_id: page_node.at_xpath("revision/id")&.text&.to_i,
397
425
  text: page_node.at_xpath(".//text")&.text || ""
398
426
  }
399
427
  end
@@ -409,6 +437,7 @@ module Wp2txt
409
437
  page = {
410
438
  title: page_node.at_xpath("title")&.text,
411
439
  id: page_node.at_xpath("id")&.text&.to_i,
440
+ revision_id: page_node.at_xpath("revision/id")&.text&.to_i,
412
441
  text: page_node.at_xpath(".//text")&.text || ""
413
442
  }
414
443
  yield page if page[:title]
@@ -837,6 +866,27 @@ module Wp2txt
837
866
  File.join(@cache_dir, "#{@lang}wiki-#{date}-langlinks.sql.gz")
838
867
  end
839
868
 
869
+ def page_props_url(date)
870
+ wiki = "#{@lang}wiki"
871
+ "#{DUMP_BASE_URL}/#{wiki}/#{date}/#{wiki}-#{date}-page_props.sql.gz"
872
+ end
873
+
874
+ def cached_page_props_path(date)
875
+ File.join(@cache_dir, "#{@lang}wiki-#{date}-page_props.sql.gz")
876
+ end
877
+
878
+ # Download the page_props dump for an explicit dump date (must match the index)
879
+ def download_page_props(date:, force: false)
880
+ path = cached_page_props_path(date)
881
+ return path if File.exist?(path) && !force
882
+
883
+ url = page_props_url(date)
884
+ puts "Downloading page_props: #{url}"
885
+ $stdout.flush
886
+ download_file(url, path)
887
+ path
888
+ end
889
+
840
890
  # Download the langlinks dump for an explicit dump date. The date MUST
841
891
  # come from the built metadata index (not latest_dump_date) so the
842
892
  # imported links stay pinned to the indexed dump version.
@@ -100,6 +100,14 @@ module Wp2txt
100
100
  raise Wp2txt::FileIOError, "Write failed: #{e.message}"
101
101
  end
102
102
 
103
+ # Push buffered output to the file. Call before forking: a child process
104
+ # inherits the buffer and writes it out again when it exits.
105
+ def flush
106
+ @mutex.synchronize do
107
+ @current_file.flush if @current_file && !@current_file.closed?
108
+ end
109
+ end
110
+
103
111
  # Close current file and finalize
104
112
  def close
105
113
  @mutex.synchronize do
@@ -0,0 +1,170 @@
1
+ # frozen_string_literal: true
2
+
3
+ require "sqlite3"
4
+ require "time"
5
+ require "digest"
6
+ require_relative "sql_dump_reader"
7
+ require_relative "version"
8
+
9
+ module Wp2txt
10
+ # Imports Wikidata IDs, disambiguation flags, and default sort keys
11
+ # from the official page_props dump into the metadata index, as
12
+ # page_properties. Like langlinks, the dump must carry the same
13
+ # date as the index: IDs are only meaningful for the pages they came with.
14
+ class PagePropsImporter
15
+ BATCH_SIZE = 50_000
16
+
17
+ # (pp_page,'pp_propname','pp_value',pp_sortkey) — string classes accept
18
+ # escaped quotes and backslashes; sortkey is NULL or a number
19
+ TUPLE_REGEX = /\((\d+),'((?:[^'\\]|\\.)*)','((?:[^'\\]|\\.)*)',(?:NULL|[-+.\deE]+)\)/
20
+ QID_REGEX = /\AQ\d+\z/
21
+
22
+ def initialize(db_path)
23
+ @db_path = db_path
24
+ end
25
+
26
+ # "jawiki-20260901-page_props.sql.gz" => "jawiki-20260901"
27
+ def self.dump_name_of(path)
28
+ File.basename(path)[/\A[a-z0-9_\-]+?-\d{8}/]
29
+ end
30
+
31
+ # @return [Hash] { status: :imported | :already_imported, row_count:, provenance: }
32
+ def import!(source_path, force: false)
33
+ raise ArgumentError, "page_props file not found: #{source_path}" unless File.exist?(source_path)
34
+
35
+ db = SQLite3::Database.new(@db_path)
36
+ db.busy_timeout = 5000
37
+ dump_name = metadata_value(db, "dump_name")
38
+ raise ArgumentError, "metadata index is not built: #{@db_path}" unless dump_name
39
+
40
+ source_dump = self.class.dump_name_of(source_path)
41
+ unless source_dump && source_dump == dump_name
42
+ raise ArgumentError,
43
+ "dump version mismatch: the metadata index is #{dump_name} but the page_props file is " \
44
+ "#{source_dump || File.basename(source_path)} (versions must match; there is no override)"
45
+ end
46
+
47
+ if !force && (existing = imported_at(db))
48
+ return { status: :already_imported, imported_at: existing,
49
+ row_count: db.get_first_value("SELECT COUNT(*) FROM page_properties").to_i, provenance: read_provenance(db) }
50
+ end
51
+
52
+ db.execute("DROP TABLE IF EXISTS page_qids") # replace the unreleased QID-only schema
53
+ db.execute("DROP TABLE IF EXISTS page_properties")
54
+ # A failed load must leave the index looking "not imported"
55
+ db.execute("DELETE FROM metadata WHERE key LIKE 'page\\_props\\_%' ESCAPE '\\'")
56
+ db.execute(<<~SQL)
57
+ CREATE TABLE page_properties (
58
+ page_id INTEGER PRIMARY KEY,
59
+ qid TEXT,
60
+ disambiguation INTEGER NOT NULL DEFAULT 0,
61
+ sort_key TEXT
62
+ )
63
+ SQL
64
+
65
+ rows_seen = 0
66
+ skipped_invalid = 0
67
+ batch = []
68
+ flush = lambda do
69
+ db.transaction do
70
+ stmt = db.prepare(<<~SQL)
71
+ INSERT INTO page_properties (page_id, qid, disambiguation, sort_key) VALUES (?, ?, ?, ?)
72
+ ON CONFLICT(page_id) DO UPDATE SET
73
+ qid = COALESCE(excluded.qid, page_properties.qid),
74
+ disambiguation = MAX(excluded.disambiguation, page_properties.disambiguation),
75
+ sort_key = COALESCE(excluded.sort_key, page_properties.sort_key)
76
+ SQL
77
+ batch.each { |row| stmt.execute(row) }
78
+ stmt.close
79
+ end
80
+ batch.clear
81
+ end
82
+
83
+ SqlDumpReader.each_insert_line(source_path, "page_props") do |line|
84
+ line.scan(TUPLE_REGEX) do |page, name, value|
85
+ rows_seen += 1
86
+ row = [page.to_i, nil, 0, nil]
87
+ case name
88
+ when "wikibase_item"
89
+ qid = SqlDumpReader.unescape(value)
90
+ next unless QID_REGEX.match?(qid)
91
+
92
+ row[1] = qid.force_encoding(Encoding::UTF_8)
93
+ when "disambiguation"
94
+ row[2] = 1
95
+ when "defaultsort"
96
+ key = SqlDumpReader.unescape(value).force_encoding(Encoding::UTF_8)
97
+ unless key.valid_encoding?
98
+ skipped_invalid += 1
99
+ next
100
+ end
101
+ row[3] = key
102
+ else
103
+ next
104
+ end
105
+ batch << row
106
+ flush.call if batch.size >= BATCH_SIZE
107
+ end
108
+ end
109
+ flush.call unless batch.empty?
110
+
111
+ if rows_seen.zero?
112
+ raise Wp2txt::Error, "no page_props rows found in #{File.basename(source_path)}; " \
113
+ "the file may be empty or in an unrecognized format"
114
+ end
115
+
116
+ counts = db.get_first_row("SELECT COUNT(*), COUNT(qid), SUM(disambiguation), COUNT(sort_key) FROM page_properties")
117
+ row_count = counts[0]
118
+ stamp_provenance(db, source_path, counts, skipped_invalid)
119
+ { status: :imported, row_count: row_count, provenance: read_provenance(db) }
120
+ ensure
121
+ db&.close
122
+ end
123
+
124
+ private
125
+
126
+ def metadata_value(db, key)
127
+ db.get_first_value("SELECT value FROM metadata WHERE key = ?", [key])
128
+ rescue SQLite3::Exception
129
+ nil
130
+ end
131
+
132
+ def imported_at(db)
133
+ table = db.get_first_value("SELECT name FROM sqlite_master WHERE type = 'table' AND name = 'page_properties'")
134
+ table && metadata_value(db, "page_props_imported_at")
135
+ end
136
+
137
+ def stamp_provenance(db, source_path, counts, skipped_invalid)
138
+ values = {
139
+ page_props_source: File.basename(source_path),
140
+ page_props_source_size: File.size(source_path),
141
+ page_props_source_sha256: Digest::SHA256.file(source_path).hexdigest,
142
+ page_props_imported_at: Time.now.utc.iso8601,
143
+ page_props_wp2txt_version: Wp2txt::VERSION,
144
+ page_props_page_count: counts[0],
145
+ page_props_qid_count: counts[1],
146
+ page_props_disambiguation_count: counts[2].to_i,
147
+ page_props_sort_key_count: counts[3],
148
+ page_props_skipped_invalid_sort_keys: skipped_invalid
149
+ }
150
+ stmt = db.prepare("INSERT OR REPLACE INTO metadata (key, value) VALUES (?, ?)")
151
+ values.each { |k, v| stmt.execute([k.to_s, v.to_s]) }
152
+ stmt.close
153
+ end
154
+
155
+ def read_provenance(db)
156
+ {
157
+ source: metadata_value(db, "page_props_source"),
158
+ source_size: metadata_value(db, "page_props_source_size").to_i,
159
+ source_sha256: metadata_value(db, "page_props_source_sha256"),
160
+ imported_at: metadata_value(db, "page_props_imported_at"),
161
+ imported_with: metadata_value(db, "page_props_wp2txt_version"),
162
+ page_count: metadata_value(db, "page_props_page_count").to_i,
163
+ qid_count: metadata_value(db, "page_props_qid_count").to_i,
164
+ disambiguation_count: metadata_value(db, "page_props_disambiguation_count").to_i,
165
+ sort_key_count: metadata_value(db, "page_props_sort_key_count").to_i,
166
+ skipped_invalid_sort_keys: metadata_value(db, "page_props_skipped_invalid_sort_keys").to_i
167
+ }
168
+ end
169
+ end
170
+ end
@@ -0,0 +1,57 @@
1
+ # frozen_string_literal: true
2
+
3
+ require "zlib"
4
+
5
+ module Wp2txt
6
+ # Streams the rows of one table's INSERT statements out of an official
7
+ # MySQL table dump (.sql or .sql.gz), without loading the file whole.
8
+ #
9
+ # Dumps write each INSERT either on one line or with the header on its own
10
+ # line and one tuple per following line (current dumps do the latter), so
11
+ # every line of a statement is read up to its terminating semicolon.
12
+ # Statements for other tables are skipped.
13
+ #
14
+ # Lines are read as BINARY: string columns are VARBINARY and real dumps
15
+ # contain historically corrupted bytes. Callers match tuples with ASCII-only
16
+ # patterns and validate the captures themselves.
17
+ module SqlDumpReader
18
+ STATEMENT_START = /\A\s*INSERT\s+INTO\s/i
19
+
20
+ module_function
21
+
22
+ # @yield [String] each line (BINARY) that belongs to an INSERT statement for table
23
+ def each_insert_line(source_path, table)
24
+ header = /\A\s*INSERT\s+INTO\s+`#{Regexp.escape(table)}`\s+VALUES\b/i
25
+ io = if source_path.end_with?(".gz")
26
+ # GzipReader ignores set_encoding; the encoding must be given at open time
27
+ Zlib::GzipReader.open(source_path, encoding: Encoding::BINARY.to_s)
28
+ else
29
+ File.open(source_path, "rb")
30
+ end
31
+ inside = false
32
+ begin
33
+ io.each_line do |line|
34
+ inside = header.match?(line) if STATEMENT_START.match?(line)
35
+ next unless inside
36
+
37
+ inside = false if line.rstrip.end_with?(";")
38
+ yield line
39
+ end
40
+ ensure
41
+ io.close
42
+ end
43
+ end
44
+
45
+ UNESCAPE_REGEX = /\\(.)/m
46
+ # MySQL backslash escapes inside mysqldump string literals
47
+ UNESCAPES = {
48
+ "0" => "\0", "'" => "'", '"' => '"', "b" => "\b", "n" => "\n",
49
+ "r" => "\r", "t" => "\t", "Z" => "\x1A", "\\" => "\\"
50
+ }.freeze
51
+
52
+ # Undo MySQL string escaping (\' \\ \n ...) in a captured value
53
+ def unescape(value)
54
+ value.gsub(UNESCAPE_REGEX) { UNESCAPES.fetch(Regexp.last_match(1), Regexp.last_match(1)) }
55
+ end
56
+ end
57
+ end