wp2txt 2.3.4 → 2.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -14,21 +14,23 @@ module Wp2txt
14
14
 
15
15
  # Format article based on configuration and output format
16
16
  def format_article(article, config)
17
- with_page_ids(format_article_body(article, config), article)
17
+ with_source_fields(format_article_body(article, config), article)
18
18
  end
19
19
 
20
- # JSON records carry the dump's page and revision IDs right after the title,
21
- # when the reader supplied them, so a record can be traced back to its source.
22
- def with_page_ids(result, article)
23
- return result unless result.is_a?(Hash) && (article.page_id || article.revision_id)
24
-
25
- result.each_with_object({}) do |(key, value), out|
26
- out[key] = value
27
- next unless key == "title"
28
-
29
- out["page_id"] = article.page_id
30
- out["revision_id"] = article.revision_id
20
+ # JSON records carry the dump's page and revision IDs (and page properties,
21
+ # when imported) right after the title, so a record can be traced
22
+ # back to its source; lead terms, when requested, come last.
23
+ def with_source_fields(result, article)
24
+ return result unless result.is_a?(Hash)
25
+
26
+ ids = { "page_id" => article.page_id, "revision_id" => article.revision_id }.compact
27
+ ids.merge!(article.page_properties.transform_keys(&:to_s)) if article.page_properties
28
+ out = result.each_with_object({}) do |(key, value), acc|
29
+ acc[key] = value
30
+ acc.merge!(ids) if key == "title"
31
31
  end
32
+ out["lead_terms"] = article.lead_terms if article.lead_terms
33
+ out
32
34
  end
33
35
 
34
36
  def format_article_body(article, config)
@@ -4,6 +4,8 @@ require "json"
4
4
  require_relative "metadata_index"
5
5
  require_relative "fts_index"
6
6
  require_relative "langlinks_importer"
7
+ require_relative "page_props_importer"
8
+ require_relative "link_counter"
7
9
  require_relative "corpus"
8
10
  require_relative "multistream"
9
11
  require_relative "memory_monitor"
@@ -322,11 +324,97 @@ module Wp2txt
322
324
  end
323
325
  end
324
326
  CliUI::EXIT_SUCCESS
325
- rescue ArgumentError => e
327
+ rescue ArgumentError, Wp2txt::Error => e
326
328
  print_error(e.message)
327
329
  CliUI::EXIT_ERROR
328
330
  end
329
331
 
332
+ # Import each article's page properties from the page_props dump of the
333
+ # same date as the metadata index
334
+ def run_import_page_props(opts)
335
+ db_path, dump_date, manager = built_index_for(opts)
336
+ return CliUI::EXIT_ERROR unless db_path
337
+
338
+ source = opts[:page_props_file] || begin
339
+ print_header("Downloading page_props for '#{opts[:lang]}' (#{dump_date})")
340
+ manager.download_page_props(date: dump_date)
341
+ end
342
+ print_mode_banner("Import Page Properties", { "Source" => File.basename(source), "Metadata DB" => db_path })
343
+
344
+ time_start = Time.now
345
+ result = PagePropsImporter.new(db_path).import!(source, force: opts[:update_cache])
346
+ if result[:status] == :already_imported
347
+ print_success("Page properties already imported (at #{result[:imported_at]}, #{result[:row_count]} pages).")
348
+ print_info_message("Use -U/--update-cache to re-import.")
349
+ else
350
+ print_success("Page properties imported: #{result[:row_count]} pages in #{format_duration(Time.now - time_start)}")
351
+ provenance = result[:provenance]
352
+ print_info("Properties", "#{provenance[:qid_count]} QIDs, #{provenance[:disambiguation_count]} disambiguation flags, " \
353
+ "#{provenance[:sort_key_count]} sort keys; #{provenance[:skipped_invalid_sort_keys]} invalid sort keys skipped")
354
+ print_info("SHA-256", result[:provenance][:source_sha256].to_s)
355
+ end
356
+ CliUI::EXIT_SUCCESS
357
+ rescue ArgumentError, Wp2txt::Error => e
358
+ print_error(e.message)
359
+ CliUI::EXIT_ERROR
360
+ end
361
+
362
+ # Count incoming links to every article and store them in the metadata index
363
+ def run_count_links(opts)
364
+ db_path, _dump_date, manager = built_index_for(opts)
365
+ return CliUI::EXIT_ERROR unless db_path
366
+
367
+ multistream = manager.cached_multistream_path
368
+ stream_offsets, = load_stream_offsets(manager.cached_index_path, opts)
369
+ num_processes = opts[:num_procs] || MemoryMonitor.optimal_processes
370
+ print_mode_banner("Count Incoming Links", {
371
+ "Dump" => File.basename(multistream), "Streams" => stream_offsets.size.to_s,
372
+ "Processes" => num_processes.to_s
373
+ })
374
+
375
+ time_start = Time.now
376
+ last_report = Time.now
377
+ result = LinkCounter.new(multistream, stream_offsets, db_path: db_path,
378
+ num_processes: num_processes).count! do |done, total|
379
+ now = Time.now
380
+ if !quiet? && (now - last_report >= DEFAULT_PROGRESS_INTERVAL || done == total)
381
+ last_report = now
382
+ puts pastel.dim(format(" [%s] %d/%d batches", now.strftime("%H:%M:%S"), done, total))
383
+ end
384
+ end
385
+ print_success("Incoming links counted for #{result[:articles]} articles " \
386
+ "(#{result[:with_inlinks]} linked at least once) in #{format_duration(Time.now - time_start)}")
387
+ CliUI::EXIT_SUCCESS
388
+ rescue ArgumentError, Wp2txt::Error => e
389
+ print_error(e.message)
390
+ CliUI::EXIT_ERROR
391
+ end
392
+
393
+ # The metadata index of the cached dump for --lang, or nil with an error printed
394
+ # @return [Array(String, String, DumpManager)] [db_path, dump_date, manager]
395
+ def built_index_for(opts)
396
+ manager = DumpManager.new(opts[:lang], cache_dir: opts[:cache_dir],
397
+ dump_expiry_days: CLI.config.dump_expiry_days)
398
+ multistream = manager.cached_multistream_path
399
+ unless File.exist?(multistream)
400
+ print_error("No cached dump found for '#{opts[:lang]}'.")
401
+ print_info_message("Download and index it with: wp2txt --build-index -L #{opts[:lang]}")
402
+ return nil
403
+ end
404
+
405
+ db_path = MetadataIndex.path_for(multistream, cache_dir: opts[:cache_dir])
406
+ meta = MetadataIndex.new(db_path)
407
+ unless meta.built?
408
+ meta.close
409
+ print_error("Metadata index not found for this dump.")
410
+ print_info_message("Build it first with: wp2txt --build-index -L #{opts[:lang]}")
411
+ return nil
412
+ end
413
+ dump_date = meta.stats[:dump_name][/\d{8}\z/]
414
+ meta.close
415
+ [db_path, dump_date, manager]
416
+ end
417
+
330
418
  # Query the metadata index and print matching article titles
331
419
  def run_find_articles(opts)
332
420
  multistream_path, = resolve_dump_paths(opts, download: false)
@@ -4,6 +4,7 @@ require "sqlite3"
4
4
  require "set"
5
5
  require "time"
6
6
  require "zlib"
7
+ require_relative "sql_dump_reader"
7
8
  require_relative "metadata_index"
8
9
  require_relative "version"
9
10
 
@@ -28,14 +29,6 @@ module Wp2txt
28
29
  SANITY_SAMPLE_SIZE = 1000
29
30
  SANITY_WARN_THRESHOLD = 0.9
30
31
 
31
- INSERT_PREFIX = /\A\s*INSERT\s+INTO\s+`langlinks`\s+VALUES\s+/i
32
-
33
- # MySQL backslash escapes inside mysqldump string literals
34
- UNESCAPES = {
35
- "0" => "\0", "'" => "'", '"' => '"', "b" => "\b", "n" => "\n",
36
- "r" => "\r", "t" => "\t", "Z" => "\x1A", "\\" => "\\"
37
- }.freeze
38
-
39
32
  def initialize(db_path, cache_dir: nil)
40
33
  @db_path = db_path
41
34
  @cache_dir = cache_dir
@@ -99,7 +92,9 @@ module Wp2txt
99
92
  batch.clear
100
93
  end
101
94
 
95
+ rows_seen = 0
102
96
  skipped_invalid = each_source_row(source_path) do |ll_from, ll_lang, ll_title|
97
+ rows_seen += 1
103
98
  next if lang_filter && !lang_filter.include?(ll_lang)
104
99
 
105
100
  batch << [ll_from, ll_lang, MetadataIndex.normalize_title(ll_title)]
@@ -107,6 +102,13 @@ module Wp2txt
107
102
  end
108
103
  flush.call unless batch.empty?
109
104
 
105
+ # Reporting success after reading nothing would leave every language
106
+ # query silently empty; an unreadable dump must fail loudly instead.
107
+ if rows_seen.zero? && skipped_invalid.zero?
108
+ raise Wp2txt::Error, "no langlinks rows found in #{File.basename(source_path)}; " \
109
+ "the file may be empty or in an unrecognized format"
110
+ end
111
+
110
112
  # Indexes are created after the load, not before (insert speed)
111
113
  db.execute("CREATE INDEX idx_langlinks_from ON langlinks(ll_from, ll_lang)")
112
114
  db.execute("CREATE INDEX idx_langlinks_lang_title ON langlinks(ll_lang, ll_title)")
@@ -224,32 +226,18 @@ module Wp2txt
224
226
  # tagged UTF-8 and validated — a garbled title could never join
225
227
  # pages.title anyway, so such rows are skipped (and counted), not scrubbed.
226
228
  def each_source_row(source_path)
227
- io = if source_path.end_with?(".gz")
228
- # GzipReader ignores set_encoding; the encoding must be given
229
- # at open time (lines must come out as BINARY — see below)
230
- Zlib::GzipReader.open(source_path, encoding: Encoding::BINARY.to_s)
231
- else
232
- File.open(source_path, "rb")
233
- end
234
-
235
229
  skipped = 0
236
- begin
237
- io.each_line do |line|
238
- next unless INSERT_PREFIX.match?(line)
239
-
240
- line.scan(TUPLE_REGEX) do |ll_from, ll_lang, ll_title|
241
- lang = unescape_mysql(ll_lang).force_encoding(Encoding::UTF_8)
242
- title = unescape_mysql(ll_title).force_encoding(Encoding::UTF_8)
243
- unless lang.valid_encoding? && title.valid_encoding?
244
- skipped += 1
245
- next
246
- end
247
-
248
- yield ll_from.to_i, lang, title
230
+ SqlDumpReader.each_insert_line(source_path, "langlinks") do |line|
231
+ line.scan(TUPLE_REGEX) do |ll_from, ll_lang, ll_title|
232
+ lang = SqlDumpReader.unescape(ll_lang).force_encoding(Encoding::UTF_8)
233
+ title = SqlDumpReader.unescape(ll_title).force_encoding(Encoding::UTF_8)
234
+ unless lang.valid_encoding? && title.valid_encoding?
235
+ skipped += 1
236
+ next
249
237
  end
238
+
239
+ yield ll_from.to_i, lang, title
250
240
  end
251
- ensure
252
- io.close
253
241
  end
254
242
  skipped
255
243
  end
@@ -262,12 +250,8 @@ module Wp2txt
262
250
  # (equivalent to the old parser skipping malformed tuples)
263
251
  TUPLE_REGEX = /\((\d+),'((?:[^'\\]|\\.)*)','((?:[^'\\]|\\.)*)'\)/
264
252
 
265
- UNESCAPE_REGEX = /\\(.)/m
266
253
 
267
254
  # Resolve MySQL backslash escapes in a captured string literal
268
255
  # (same mapping as the old hand-rolled parser)
269
- def unescape_mysql(str)
270
- str.gsub(UNESCAPE_REGEX) { UNESCAPES[::Regexp.last_match(1)] || ::Regexp.last_match(1) }
271
- end
272
256
  end
273
257
  end
@@ -0,0 +1,228 @@
1
+ # frozen_string_literal: true
2
+
3
+ require_relative "wikitext_regions"
4
+
5
+ module Wp2txt
6
+ # Finds the terms an article introduces in its lead: bold spans in the first
7
+ # paragraph that has one, the parenthesized notes written right after each,
8
+ # and reading (ruby) templates. Nothing is judged — which note is a reading,
9
+ # a native spelling, or a date is left to the caller.
10
+ #
11
+ # Positions are character offsets [start, end) into the article's wikitext
12
+ # as stored in the dump, XML entities decoded, before any other processing
13
+ # (comments included), so a caller holding the same text can cut out and
14
+ # hash exactly the span a term came from.
15
+ module LeadTerms
16
+ MAX_TERMS = 5
17
+ HEADING = /^={2,6}[^=\n].*?={2,6}[ \t]*$/
18
+ OPENERS = { "(" => ")", "(" => ")" }.freeze
19
+ SEPARATORS = ["、", ",", ",", ";", ";"].freeze
20
+ PARAGRAPH_BREAK = /\n[ \t\r]*\n/
21
+ # Regions whose bold text is not the article's own lead prose
22
+ SKIP_OPEN = { "{{" => "}}", "{|" => "|}", "<!--" => "-->" }.freeze
23
+ REF_OPEN = /\A<ref(?:\s[^>]*)?>/i
24
+ REF_SELF_CLOSING = /\A<ref(?:\s[^>]*)?\/>/i
25
+ FILE_LINK = /\A\[\[\s*(?:file|image|ファイル|画像|media)\s*:/i
26
+
27
+ module_function
28
+
29
+ # @param wikitext [String] decoded article wikitext
30
+ # @param render [#call] turns a wikitext fragment into clean text
31
+ # @return [Array<Hash>] terms in order of appearance, at most MAX_TERMS
32
+ def extract(wikitext, render:)
33
+ return [] if wikitext.nil? || wikitext.empty?
34
+
35
+ visible = WikitextRegions.mask_lead(wikitext)
36
+ bolds, rubies, lead_end = scan(visible, visible.length, source: wikitext)
37
+ original_render = render
38
+ render = ->(fragment) { original_render.call(fragment.gsub(WikitextRegions::COMMENT, "")) }
39
+ terms = []
40
+
41
+ if (first = bolds.first)
42
+ paragraph = paragraph_bounds(visible, first[0], lead_end)
43
+ bolds.select { |s, e| s >= paragraph[0] && e <= paragraph[1] }.first(MAX_TERMS).each do |s, e|
44
+ terms << bold_term(wikitext, s, e, paragraph[1], render)
45
+ end
46
+ end
47
+ rubies.each do |s, e, parts|
48
+ terms << { "text" => render.call(parts[1].to_s).strip, "reading" => render.call(parts[2].to_s).strip,
49
+ "source" => parts[0].strip, "span" => { "template" => [s, e] } }
50
+ end
51
+
52
+ terms.sort_by { |t| t["span"].values.first.first }.first(MAX_TERMS)
53
+ .each_with_index.map { |t, i| { "index" => i }.merge(t) }
54
+ end
55
+
56
+ # Top-level bold spans [start, end) including the quote marks, and ruby
57
+ # templates [start, end, [name, text, reading]] found in the lead
58
+ def scan(text, limit, source: text)
59
+ bolds = []
60
+ rubies = []
61
+ i = 0
62
+ open_bold = nil
63
+ while i < limit
64
+ if (i.zero? || text[i - 1] == "\n") && text[i] == "=" &&
65
+ HEADING.match?(text[i...(text.index("\n", i) || limit)])
66
+ limit = i
67
+ break
68
+ elsif text[i] == "\n"
69
+ open_bold = nil # bold does not continue across lines
70
+ i += 1
71
+ elsif (close = SKIP_OPEN[text[i, 4] == "<!--" ? "<!--" : text[i, 2]])
72
+ opener = text[i, 4] == "<!--" ? "<!--" : text[i, 2]
73
+ stop = matching_end(text, i, opener, close)
74
+ if opener == "{{" && (ruby = ruby_template(source[(i + 2)...(stop - 2)]))
75
+ rubies << [i, stop, ruby]
76
+ end
77
+ i = stop
78
+ elsif text[i, 2] == "[[" && FILE_LINK.match?(text[i, 40])
79
+ i = matching_end(text, i, "[[", "]]")
80
+ elsif text[i] == "<" && (m = REF_SELF_CLOSING.match(text[i, 200]))
81
+ i += m[0].length
82
+ elsif text[i] == "<" && (m = REF_OPEN.match(text[i, 200]))
83
+ close_at = text.index(%r{</ref\s*>}i, i + m[0].length)
84
+ i = close_at ? text.index(">", close_at) + 1 : limit
85
+ elsif text[i, 3] == "'''"
86
+ run = text[i..][/\A'+/].length
87
+ if open_bold
88
+ bolds << [open_bold, i + run]
89
+ open_bold = nil
90
+ else
91
+ open_bold = i
92
+ end
93
+ i += run
94
+ else
95
+ i += 1
96
+ end
97
+ end
98
+ [bolds, rubies, limit]
99
+ end
100
+
101
+ # End index (exclusive) of the construct opened at start, honouring nesting
102
+ def matching_end(text, start, opener, closer)
103
+ depth = 0
104
+ i = start
105
+ while i < text.length
106
+ if text[i, opener.length] == opener
107
+ depth += 1
108
+ i += opener.length
109
+ elsif text[i, closer.length] == closer
110
+ depth -= 1
111
+ i += closer.length
112
+ return i if depth.zero?
113
+ else
114
+ i += 1
115
+ end
116
+ end
117
+ text.length
118
+ end
119
+
120
+ def ruby_template(content)
121
+ parts = split_top_level(content, ["|"])
122
+ name = parts.first.to_s
123
+ return nil unless ruby_name?(name)
124
+
125
+ [name, parts[1], parts[2]]
126
+ end
127
+
128
+ # Same name rule as the cleaner: "_" and " " alike, case-insensitive
129
+ def ruby_name?(name)
130
+ @ruby_names ||= Wp2txt::RUBY_TEXT_TEMPLATES.to_set { |t| t.tr("_", " ").strip.downcase }
131
+ @ruby_names.include?(name.to_s.tr("_", " ").strip.downcase)
132
+ end
133
+
134
+ def paragraph_bounds(text, pos, limit)
135
+ start = 0
136
+ text[0...pos].to_enum(:scan, PARAGRAPH_BREAK).each { start = Regexp.last_match.end(0) }
137
+ stop = text.index(PARAGRAPH_BREAK, pos) || limit
138
+ [start, [stop, limit].min]
139
+ end
140
+
141
+ def bold_term(text, s, e, limit, render)
142
+ inner = text[s...e].sub(/\A'+/, "").sub(/'+\z/, "")
143
+ term = { "text" => render.call(plain_ruby(inner)).strip, "notes" => [], "notes_text" => nil,
144
+ "source" => "bold", "span" => { "bold" => [s, e] } }
145
+ j = e
146
+ j += 1 while j < limit && [" ", "\t", " ", "\r", "\n"].include?(text[j])
147
+ return term if j >= limit
148
+ closer = OPENERS[text[j]]
149
+ return term unless closer
150
+
151
+ pe = paren_end(text, j, text[j], closer, limit)
152
+ return term unless pe
153
+
154
+ raw = text[(j + 1)...(pe - 1)]
155
+ term["notes_text"] = render.call(raw).strip
156
+ term["notes"] = split_top_level(raw, SEPARATORS).map { |part| render.call(part).strip }.reject(&:empty?)
157
+ term["span"]["paren"] = [j, pe]
158
+ term
159
+ end
160
+
161
+ # Index after the bracket closing the one at start, or nil if unclosed
162
+ def paren_end(text, start, opener, closer, limit)
163
+ depth = 0
164
+ i = start
165
+ while i < limit
166
+ two = text[i, 2]
167
+ if text[i] == "<" && (stop = WikitextRegions.end_at(text, i, lead: true))
168
+ i = stop
169
+ next
170
+ elsif ["{{", "[["].include?(two)
171
+ i = matching_end(text, i, two, two == "{{" ? "}}" : "]]")
172
+ next
173
+ end
174
+ if [opener, "(", "("].include?(text[i])
175
+ depth += 1
176
+ elsif [closer, ")", ")"].include?(text[i])
177
+ depth -= 1
178
+ return i + 1 if depth.zero?
179
+ elsif text[i] == "\n" && /\A\n[ \t\r]*\n/.match?(text[i..])
180
+ return nil
181
+ end
182
+ i += 1
183
+ end
184
+ nil
185
+ end
186
+
187
+ # Split at separators that sit outside brackets, templates, and links
188
+ def split_top_level(text, separators)
189
+ parts = [+""]
190
+ depth = 0
191
+ i = 0
192
+ while i < text.length
193
+ two = text[i, 2]
194
+ if text[i] == "<" && (stop = WikitextRegions.end_at(text, i, lead: true))
195
+ parts.last << text[i...stop]
196
+ i = stop
197
+ elsif ["{{", "[["].include?(two)
198
+ depth += 1
199
+ parts.last << two
200
+ i += 2
201
+ elsif ["}}", "]]"].include?(two)
202
+ depth -= 1
203
+ parts.last << two
204
+ i += 2
205
+ else
206
+ ch = text[i]
207
+ depth += 1 if ["(", "("].include?(ch)
208
+ depth -= 1 if [")", ")"].include?(ch)
209
+ if depth <= 0 && separators.include?(ch)
210
+ parts << +""
211
+ else
212
+ parts.last << ch
213
+ end
214
+ i += 1
215
+ end
216
+ end
217
+ parts
218
+ end
219
+
220
+ # In a bold headword, a ruby template stands for its base text; the
221
+ # reading is reported separately as its own term
222
+ def plain_ruby(fragment)
223
+ fragment.gsub(/\{\{([^{}]*)\}\}/) do |whole|
224
+ (ruby = ruby_template(Regexp.last_match(1))) ? ruby[1].to_s : whole
225
+ end
226
+ end
227
+ end
228
+ end
@@ -0,0 +1,171 @@
1
+ # frozen_string_literal: true
2
+
3
+ require "parallel"
4
+ require "sqlite3"
5
+ require "time"
6
+ require_relative "metadata_index"
7
+ require_relative "version"
8
+ require_relative "text_processing"
9
+ require_relative "wikitext_regions"
10
+
11
+ module Wp2txt
12
+ # Counts, for every article, how many other articles link to it, and stores
13
+ # the result in the metadata index as page_inlinks(page_id, inlinks,
14
+ # via_redirects).
15
+ #
16
+ # Rules (recorded as RULE_VERSION so counts from different rules are never
17
+ # compared unknowingly):
18
+ # - sources are articles (namespace 0) that are not redirects
19
+ # - each source counts once per target, however many times it links there
20
+ # - a link to a redirect counts for the redirect's target (one hop);
21
+ # via_redirects is how many sources reached the target only that way
22
+ # - only links written in the article's own wikitext count; links that
23
+ # templates add when rendered (navigation boxes) are not in the dump text
24
+ # - commented-out links do not count
25
+ class LinkCounter
26
+ include Wp2txt
27
+
28
+ RULE_VERSION = "2"
29
+ STREAMS_PER_BATCH = 50
30
+ LINK_REGEX = /\[\[([^\[\]|<>{}\n]+)(?=\||\]\])/
31
+ COMMENT_REGEX = /<!--.*?-->/m
32
+
33
+ def initialize(multistream_path, stream_offsets, db_path:, num_processes: 4)
34
+ @multistream_path = multistream_path
35
+ @stream_offsets = stream_offsets
36
+ @db_path = db_path
37
+ @num_processes = num_processes
38
+ end
39
+
40
+ # @return [Hash] { articles:, with_inlinks:, rule_version: }
41
+ def count!(&progress)
42
+ titles, redirects = load_titles
43
+ # Inherited by the forked workers (copy-on-write); they only read it
44
+ @redirects = redirects
45
+
46
+ direct = Hash.new(0)
47
+ via = Hash.new(0)
48
+ pairs = @stream_offsets.zip(@stream_offsets[1..].to_a + [nil])
49
+ batches = pairs.each_slice(STREAMS_PER_BATCH).to_a
50
+ done = 0
51
+
52
+ Parallel.each(
53
+ batches,
54
+ in_processes: @num_processes,
55
+ finish: lambda { |_item, _idx, result|
56
+ result[:direct].each { |t, n| direct[t] += n }
57
+ result[:via].each { |t, n| via[t] += n }
58
+ done += 1
59
+ progress&.call(done, batches.size)
60
+ }
61
+ ) do |batch|
62
+ scan_batch(batch)
63
+ end
64
+
65
+ write(titles, direct, via)
66
+ end
67
+
68
+ private
69
+
70
+ # Articles (title => page_id) and redirects (title => target title)
71
+ def load_titles
72
+ db = SQLite3::Database.new(@db_path, readonly: true)
73
+ @case_rule = db.get_first_value("SELECT value FROM metadata WHERE key = 'case_rule'") || "first-letter"
74
+ titles = {}
75
+ redirects = {}
76
+ db.execute("SELECT page_id, title, redirect_to FROM pages WHERE namespace = 0") do |id, title, target|
77
+ if target
78
+ redirects[title] = normalize_target(target)
79
+ else
80
+ titles[title] = id
81
+ end
82
+ end
83
+ [titles, redirects]
84
+ ensure
85
+ db&.close
86
+ end
87
+
88
+ # Runs in a worker: returns per-target counts for this batch of streams
89
+ def scan_batch(offset_pairs)
90
+ direct = Hash.new(0)
91
+ via = Hash.new(0)
92
+ File.open(@multistream_path, "rb") do |f|
93
+ offset_pairs.each do |offset, next_offset|
94
+ f.seek(offset)
95
+ data = next_offset ? f.read(next_offset - offset) : f.read
96
+ xml = MetadataIndexBuilder.decompress_bz2(data)
97
+ xml.scan(MetadataIndexBuilder::PAGE_BLOCK_REGEX) do
98
+ count_page(::Regexp.last_match(1), direct, via)
99
+ end
100
+ end
101
+ end
102
+ { direct: direct, via: via }
103
+ end
104
+
105
+ def count_page(block, direct, via)
106
+ return unless Wp2txt.namespace_id(block[MetadataIndexBuilder::NS_REGEX, 1]).zero?
107
+
108
+ text = MetadataIndexBuilder.unescape_xml(block[MetadataIndexBuilder::TEXT_REGEX, 1] || "")
109
+ return if Wp2txt::REDIRECT_REGEX.match?(text)
110
+
111
+ reached = {} # target => true if reached directly at least once
112
+ WikitextRegions.remove_literal(text).scan(LINK_REGEX) do |(raw)|
113
+ name = normalize_target(raw)
114
+ next if name.empty?
115
+
116
+ if (target = @redirects[name])
117
+ reached[target] ||= false
118
+ else
119
+ reached[name] = true
120
+ end
121
+ end
122
+ reached.each do |target, directly|
123
+ direct[target] += 1
124
+ via[target] += 1 unless directly
125
+ end
126
+ end
127
+
128
+ def normalize_target(raw)
129
+ # Decode each reference in the original input, never references created
130
+ # by decoding another one (special_chr has two decoding stages).
131
+ decoded = raw.gsub(/&(?:#[xX][0-9a-fA-F]+|#\d+|[a-zA-Z][a-zA-Z0-9]*);/) { |entity| special_chr(entity) }
132
+ title = decoded.split("#", 2).first.to_s.strip.sub(/\A:/, "")
133
+ MetadataIndex.normalize_title(title, case_rule: @case_rule || "first-letter")
134
+ end
135
+
136
+ def write(titles, direct, via)
137
+ db = SQLite3::Database.new(@db_path)
138
+ db.busy_timeout = 5000
139
+ db.execute("DROP TABLE IF EXISTS page_inlinks")
140
+ db.execute("DELETE FROM metadata WHERE key LIKE 'links\\_%' ESCAPE '\\'")
141
+ db.execute(<<~SQL)
142
+ CREATE TABLE page_inlinks (
143
+ page_id INTEGER PRIMARY KEY,
144
+ inlinks INTEGER NOT NULL,
145
+ via_redirects INTEGER NOT NULL
146
+ )
147
+ SQL
148
+ with_inlinks = 0
149
+ db.transaction do
150
+ stmt = db.prepare("INSERT INTO page_inlinks (page_id, inlinks, via_redirects) VALUES (?, ?, ?)")
151
+ titles.each do |title, page_id|
152
+ n = direct[title]
153
+ with_inlinks += 1 if n.positive?
154
+ stmt.execute([page_id, n, via[title]])
155
+ end
156
+ stmt.close
157
+ end
158
+ db.execute("CREATE INDEX idx_page_inlinks_count ON page_inlinks(inlinks)")
159
+ {
160
+ links_counted_at: Time.now.utc.iso8601,
161
+ links_rule_version: RULE_VERSION,
162
+ links_wp2txt_version: Wp2txt::VERSION,
163
+ links_article_count: titles.size,
164
+ links_with_inlinks: with_inlinks
165
+ }.each { |k, v| db.execute("INSERT OR REPLACE INTO metadata (key, value) VALUES (?, ?)", [k.to_s, v.to_s]) }
166
+ { articles: titles.size, with_inlinks: with_inlinks, rule_version: RULE_VERSION }
167
+ ensure
168
+ db&.close
169
+ end
170
+ end
171
+ end