wp2txt 2.3.3 → 2.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (45) hide show
  1. checksums.yaml +4 -4
  2. data/.dockerignore +0 -4
  3. data/.gitignore +3 -5
  4. data/CHANGELOG.md +18 -0
  5. data/DEVELOPMENT.md +1 -1
  6. data/DEVELOPMENT_ja.md +1 -1
  7. data/README.md +44 -2
  8. data/README_ja.md +35 -2
  9. data/Rakefile +10 -21
  10. data/bin/wp2txt +79 -17
  11. data/bin/wp2txt-mcp +1 -1
  12. data/docs/INDEXES.md +61 -1
  13. data/lib/wp2txt/article.rb +1 -1
  14. data/lib/wp2txt/cli.rb +42 -0
  15. data/lib/wp2txt/constants.rb +24 -0
  16. data/lib/wp2txt/corpus.rb +11 -2
  17. data/lib/wp2txt/data/template_aliases.json +1 -1
  18. data/lib/wp2txt/extractor.rb +10 -1
  19. data/lib/wp2txt/formatter.rb +20 -0
  20. data/lib/wp2txt/index_commands.rb +89 -1
  21. data/lib/wp2txt/langlinks_importer.rb +19 -35
  22. data/lib/wp2txt/lead_terms.rb +228 -0
  23. data/lib/wp2txt/link_counter.rb +171 -0
  24. data/lib/wp2txt/metadata_index.rb +79 -3
  25. data/lib/wp2txt/multistream.rb +52 -2
  26. data/lib/wp2txt/output_writer.rb +8 -0
  27. data/lib/wp2txt/page_props_importer.rb +170 -0
  28. data/lib/wp2txt/sql_dump_reader.rb +57 -0
  29. data/lib/wp2txt/stream_processor.rb +42 -15
  30. data/lib/wp2txt/template_expander.rb +19 -0
  31. data/lib/wp2txt/utils.rb +5 -3
  32. data/lib/wp2txt/version.rb +1 -1
  33. data/lib/wp2txt/wikitext_regions.rb +66 -0
  34. data/lib/wp2txt.rb +7 -5
  35. data/spec/docs_sync_spec.rb +3 -3
  36. data/spec/langlinks_importer_spec.rb +27 -0
  37. data/spec/lead_terms_edge_cases_spec.rb +194 -0
  38. data/spec/lead_terms_links_qids_spec.rb +204 -0
  39. data/spec/output_integrity_spec.rb +146 -0
  40. data/spec/p1_correctness_spec.rb +32 -14
  41. data/spec/page_properties_spec.rb +161 -0
  42. data/spec/region_semantics_spec.rb +71 -0
  43. data/spec/template_passthrough_spec.rb +43 -0
  44. data/spec/titles_output_path_spec.rb +12 -0
  45. metadata +18 -1
data/lib/wp2txt/corpus.rb CHANGED
@@ -115,7 +115,9 @@ module Wp2txt
115
115
  fulltext_current: fts.built? && fts.valid_for?(@multistream_path),
116
116
  stats: stats,
117
117
  fulltext: fts.built? ? fts.stats : nil,
118
- langlinks: @metadata.built? ? @metadata.langlinks_provenance : nil
118
+ langlinks: @metadata.built? ? @metadata.langlinks_provenance : nil,
119
+ page_properties: @metadata.built? ? @metadata.page_props_provenance : nil,
120
+ inlinks: @metadata.built? ? @metadata.links_provenance : nil
119
121
  }
120
122
  end
121
123
 
@@ -147,7 +149,10 @@ module Wp2txt
147
149
  else
148
150
  render_text(page)
149
151
  end
150
- result = { id: page[:id], title: page[:title], format: format.to_s }
152
+ result = { id: page[:id], title: page[:title] }
153
+ properties = @metadata.built? ? @metadata.properties_for(page[:id]) : nil
154
+ result.merge!(properties) if properties
155
+ result[:format] = format.to_s
151
156
  if max_chars && body.length > max_chars
152
157
  result.merge(text: body[0, max_chars], truncated: true, total_chars: body.length)
153
158
  else
@@ -401,12 +406,16 @@ module Wp2txt
401
406
  titles.each_slice(EXTRACT_BATCH_SIZE) do |batch|
402
407
  raise Cancelled if cancel_check&.call
403
408
 
409
+ # Forked workers inherit this file's buffer and would write it again on exit
410
+ f.flush
404
411
  pages = reader.extract_articles_parallel(batch, num_processes: num_processes)
405
412
  batch.each do |t|
406
413
  page = pages[t]
407
414
  next unless page
408
415
 
409
416
  records = build_records(page, content, resolved_sections, chunk_size, chunk_overlap)
417
+ properties = @metadata.built? ? @metadata.properties_for(page[:id]) : nil
418
+ records.each { |record| record.merge!(properties) } if properties
410
419
  next if records.empty?
411
420
 
412
421
  articles_extracted += 1
@@ -82,7 +82,7 @@
82
82
  "Cite web", "Citeweb", "Cit"
83
83
  ],
84
84
  "ruby_text_templates": [
85
- "読み仮名", "よみがな", "ふりがな",
85
+ "読み仮名", "読み仮名 ruby不使用", "よみがな", "ふりがな",
86
86
  "ruby", "Ruby", "ruby-ja", "Ruby-ja",
87
87
  "ruby text", "ruby annotation",
88
88
  "rubi", "rubytext",
@@ -111,6 +111,7 @@ module Wp2txt
111
111
 
112
112
  if page
113
113
  article = Article.new(page[:text], page[:title], !config[:marker])
114
+ set_source_fields(article, page[:id], page[:revision_id], page[:text], config)
114
115
  result = format_article(article, config)
115
116
  writer.write(result)
116
117
  extracted_count += 1
@@ -253,11 +254,17 @@ module Wp2txt
253
254
  bz2_gem: opts[:bz2_gem]
254
255
  }
255
256
 
256
- %i[title list heading table pre ref redirect multiline category category_only
257
+ %i[lead_terms title list heading table pre ref redirect multiline category category_only
257
258
  summary_only metadata_only marker extract_citations expand_templates].each do |opt|
258
259
  config[opt] = opts[opt]
259
260
  end
260
261
 
262
+ # Page properties, when imported into the index of the full cached dump
263
+ if format == :json && opts[:lang]
264
+ full_dump = Wp2txt::DumpManager.new(opts[:lang], cache_dir: opts[:cache_dir]).cached_multistream_path
265
+ config[:page_properties] = load_page_properties(full_dump, opts[:cache_dir])
266
+ end
267
+
261
268
  # Section extraction options
262
269
  %i[sections section_output min_section_length skip_empty
263
270
  alias_file no_section_aliases show_matched_sections].each do |opt|
@@ -480,6 +487,7 @@ module Wp2txt
480
487
 
481
488
  pages.each do |page|
482
489
  article = Article.new(page[:text], page[:title], !config[:marker])
490
+ set_source_fields(article, page[:id], page[:revision_id], page[:text], config)
483
491
  result = format_article(article, config)
484
492
  writer.write(result)
485
493
  extracted_count += 1
@@ -493,6 +501,7 @@ module Wp2txt
493
501
 
494
502
  if page
495
503
  article = Article.new(page[:text], page[:title], !config[:marker])
504
+ set_source_fields(article, page[:id], page[:revision_id], page[:text], config)
496
505
  result = format_article(article, config)
497
506
  writer.write(result)
498
507
  extracted_count += 1
@@ -14,6 +14,26 @@ module Wp2txt
14
14
 
15
15
  # Format article based on configuration and output format
16
16
  def format_article(article, config)
17
+ with_source_fields(format_article_body(article, config), article)
18
+ end
19
+
20
+ # JSON records carry the dump's page and revision IDs (and page properties,
21
+ # when imported) right after the title, so a record can be traced
22
+ # back to its source; lead terms, when requested, come last.
23
+ def with_source_fields(result, article)
24
+ return result unless result.is_a?(Hash)
25
+
26
+ ids = { "page_id" => article.page_id, "revision_id" => article.revision_id }.compact
27
+ ids.merge!(article.page_properties.transform_keys(&:to_s)) if article.page_properties
28
+ out = result.each_with_object({}) do |(key, value), acc|
29
+ acc[key] = value
30
+ acc.merge!(ids) if key == "title"
31
+ end
32
+ out["lead_terms"] = article.lead_terms if article.lead_terms
33
+ out
34
+ end
35
+
36
+ def format_article_body(article, config)
17
37
  # Store original title for magic word expansion in content
18
38
  original_title = article.title.dup
19
39
  article.title = format_wiki(article.title, config)
@@ -4,6 +4,8 @@ require "json"
4
4
  require_relative "metadata_index"
5
5
  require_relative "fts_index"
6
6
  require_relative "langlinks_importer"
7
+ require_relative "page_props_importer"
8
+ require_relative "link_counter"
7
9
  require_relative "corpus"
8
10
  require_relative "multistream"
9
11
  require_relative "memory_monitor"
@@ -322,11 +324,97 @@ module Wp2txt
322
324
  end
323
325
  end
324
326
  CliUI::EXIT_SUCCESS
325
- rescue ArgumentError => e
327
+ rescue ArgumentError, Wp2txt::Error => e
326
328
  print_error(e.message)
327
329
  CliUI::EXIT_ERROR
328
330
  end
329
331
 
332
+ # Import each article's page properties from the page_props dump of the
333
+ # same date as the metadata index
334
+ def run_import_page_props(opts)
335
+ db_path, dump_date, manager = built_index_for(opts)
336
+ return CliUI::EXIT_ERROR unless db_path
337
+
338
+ source = opts[:page_props_file] || begin
339
+ print_header("Downloading page_props for '#{opts[:lang]}' (#{dump_date})")
340
+ manager.download_page_props(date: dump_date)
341
+ end
342
+ print_mode_banner("Import Page Properties", { "Source" => File.basename(source), "Metadata DB" => db_path })
343
+
344
+ time_start = Time.now
345
+ result = PagePropsImporter.new(db_path).import!(source, force: opts[:update_cache])
346
+ if result[:status] == :already_imported
347
+ print_success("Page properties already imported (at #{result[:imported_at]}, #{result[:row_count]} pages).")
348
+ print_info_message("Use -U/--update-cache to re-import.")
349
+ else
350
+ print_success("Page properties imported: #{result[:row_count]} pages in #{format_duration(Time.now - time_start)}")
351
+ provenance = result[:provenance]
352
+ print_info("Properties", "#{provenance[:qid_count]} QIDs, #{provenance[:disambiguation_count]} disambiguation flags, " \
353
+ "#{provenance[:sort_key_count]} sort keys; #{provenance[:skipped_invalid_sort_keys]} invalid sort keys skipped")
354
+ print_info("SHA-256", result[:provenance][:source_sha256].to_s)
355
+ end
356
+ CliUI::EXIT_SUCCESS
357
+ rescue ArgumentError, Wp2txt::Error => e
358
+ print_error(e.message)
359
+ CliUI::EXIT_ERROR
360
+ end
361
+
362
+ # Count incoming links to every article and store them in the metadata index
363
+ def run_count_links(opts)
364
+ db_path, _dump_date, manager = built_index_for(opts)
365
+ return CliUI::EXIT_ERROR unless db_path
366
+
367
+ multistream = manager.cached_multistream_path
368
+ stream_offsets, = load_stream_offsets(manager.cached_index_path, opts)
369
+ num_processes = opts[:num_procs] || MemoryMonitor.optimal_processes
370
+ print_mode_banner("Count Incoming Links", {
371
+ "Dump" => File.basename(multistream), "Streams" => stream_offsets.size.to_s,
372
+ "Processes" => num_processes.to_s
373
+ })
374
+
375
+ time_start = Time.now
376
+ last_report = Time.now
377
+ result = LinkCounter.new(multistream, stream_offsets, db_path: db_path,
378
+ num_processes: num_processes).count! do |done, total|
379
+ now = Time.now
380
+ if !quiet? && (now - last_report >= DEFAULT_PROGRESS_INTERVAL || done == total)
381
+ last_report = now
382
+ puts pastel.dim(format(" [%s] %d/%d batches", now.strftime("%H:%M:%S"), done, total))
383
+ end
384
+ end
385
+ print_success("Incoming links counted for #{result[:articles]} articles " \
386
+ "(#{result[:with_inlinks]} linked at least once) in #{format_duration(Time.now - time_start)}")
387
+ CliUI::EXIT_SUCCESS
388
+ rescue ArgumentError, Wp2txt::Error => e
389
+ print_error(e.message)
390
+ CliUI::EXIT_ERROR
391
+ end
392
+
393
+ # The metadata index of the cached dump for --lang, or nil with an error printed
394
+ # @return [Array(String, String, DumpManager)] [db_path, dump_date, manager]
395
+ def built_index_for(opts)
396
+ manager = DumpManager.new(opts[:lang], cache_dir: opts[:cache_dir],
397
+ dump_expiry_days: CLI.config.dump_expiry_days)
398
+ multistream = manager.cached_multistream_path
399
+ unless File.exist?(multistream)
400
+ print_error("No cached dump found for '#{opts[:lang]}'.")
401
+ print_info_message("Download and index it with: wp2txt --build-index -L #{opts[:lang]}")
402
+ return nil
403
+ end
404
+
405
+ db_path = MetadataIndex.path_for(multistream, cache_dir: opts[:cache_dir])
406
+ meta = MetadataIndex.new(db_path)
407
+ unless meta.built?
408
+ meta.close
409
+ print_error("Metadata index not found for this dump.")
410
+ print_info_message("Build it first with: wp2txt --build-index -L #{opts[:lang]}")
411
+ return nil
412
+ end
413
+ dump_date = meta.stats[:dump_name][/\d{8}\z/]
414
+ meta.close
415
+ [db_path, dump_date, manager]
416
+ end
417
+
330
418
  # Query the metadata index and print matching article titles
331
419
  def run_find_articles(opts)
332
420
  multistream_path, = resolve_dump_paths(opts, download: false)
@@ -4,6 +4,7 @@ require "sqlite3"
4
4
  require "set"
5
5
  require "time"
6
6
  require "zlib"
7
+ require_relative "sql_dump_reader"
7
8
  require_relative "metadata_index"
8
9
  require_relative "version"
9
10
 
@@ -28,14 +29,6 @@ module Wp2txt
28
29
  SANITY_SAMPLE_SIZE = 1000
29
30
  SANITY_WARN_THRESHOLD = 0.9
30
31
 
31
- INSERT_PREFIX = /\A\s*INSERT\s+INTO\s+`langlinks`\s+VALUES\s+/i
32
-
33
- # MySQL backslash escapes inside mysqldump string literals
34
- UNESCAPES = {
35
- "0" => "\0", "'" => "'", '"' => '"', "b" => "\b", "n" => "\n",
36
- "r" => "\r", "t" => "\t", "Z" => "\x1A", "\\" => "\\"
37
- }.freeze
38
-
39
32
  def initialize(db_path, cache_dir: nil)
40
33
  @db_path = db_path
41
34
  @cache_dir = cache_dir
@@ -99,7 +92,9 @@ module Wp2txt
99
92
  batch.clear
100
93
  end
101
94
 
95
+ rows_seen = 0
102
96
  skipped_invalid = each_source_row(source_path) do |ll_from, ll_lang, ll_title|
97
+ rows_seen += 1
103
98
  next if lang_filter && !lang_filter.include?(ll_lang)
104
99
 
105
100
  batch << [ll_from, ll_lang, MetadataIndex.normalize_title(ll_title)]
@@ -107,6 +102,13 @@ module Wp2txt
107
102
  end
108
103
  flush.call unless batch.empty?
109
104
 
105
+ # Reporting success after reading nothing would leave every language
106
+ # query silently empty; an unreadable dump must fail loudly instead.
107
+ if rows_seen.zero? && skipped_invalid.zero?
108
+ raise Wp2txt::Error, "no langlinks rows found in #{File.basename(source_path)}; " \
109
+ "the file may be empty or in an unrecognized format"
110
+ end
111
+
110
112
  # Indexes are created after the load, not before (insert speed)
111
113
  db.execute("CREATE INDEX idx_langlinks_from ON langlinks(ll_from, ll_lang)")
112
114
  db.execute("CREATE INDEX idx_langlinks_lang_title ON langlinks(ll_lang, ll_title)")
@@ -224,32 +226,18 @@ module Wp2txt
224
226
  # tagged UTF-8 and validated — a garbled title could never join
225
227
  # pages.title anyway, so such rows are skipped (and counted), not scrubbed.
226
228
  def each_source_row(source_path)
227
- io = if source_path.end_with?(".gz")
228
- # GzipReader ignores set_encoding; the encoding must be given
229
- # at open time (lines must come out as BINARY — see below)
230
- Zlib::GzipReader.open(source_path, encoding: Encoding::BINARY.to_s)
231
- else
232
- File.open(source_path, "rb")
233
- end
234
-
235
229
  skipped = 0
236
- begin
237
- io.each_line do |line|
238
- next unless INSERT_PREFIX.match?(line)
239
-
240
- line.scan(TUPLE_REGEX) do |ll_from, ll_lang, ll_title|
241
- lang = unescape_mysql(ll_lang).force_encoding(Encoding::UTF_8)
242
- title = unescape_mysql(ll_title).force_encoding(Encoding::UTF_8)
243
- unless lang.valid_encoding? && title.valid_encoding?
244
- skipped += 1
245
- next
246
- end
247
-
248
- yield ll_from.to_i, lang, title
230
+ SqlDumpReader.each_insert_line(source_path, "langlinks") do |line|
231
+ line.scan(TUPLE_REGEX) do |ll_from, ll_lang, ll_title|
232
+ lang = SqlDumpReader.unescape(ll_lang).force_encoding(Encoding::UTF_8)
233
+ title = SqlDumpReader.unescape(ll_title).force_encoding(Encoding::UTF_8)
234
+ unless lang.valid_encoding? && title.valid_encoding?
235
+ skipped += 1
236
+ next
249
237
  end
238
+
239
+ yield ll_from.to_i, lang, title
250
240
  end
251
- ensure
252
- io.close
253
241
  end
254
242
  skipped
255
243
  end
@@ -262,12 +250,8 @@ module Wp2txt
262
250
  # (equivalent to the old parser skipping malformed tuples)
263
251
  TUPLE_REGEX = /\((\d+),'((?:[^'\\]|\\.)*)','((?:[^'\\]|\\.)*)'\)/
264
252
 
265
- UNESCAPE_REGEX = /\\(.)/m
266
253
 
267
254
  # Resolve MySQL backslash escapes in a captured string literal
268
255
  # (same mapping as the old hand-rolled parser)
269
- def unescape_mysql(str)
270
- str.gsub(UNESCAPE_REGEX) { UNESCAPES[::Regexp.last_match(1)] || ::Regexp.last_match(1) }
271
- end
272
256
  end
273
257
  end
@@ -0,0 +1,228 @@
1
+ # frozen_string_literal: true
2
+
3
+ require_relative "wikitext_regions"
4
+
5
+ module Wp2txt
6
+ # Finds the terms an article introduces in its lead: bold spans in the first
7
+ # paragraph that has one, the parenthesized notes written right after each,
8
+ # and reading (ruby) templates. Nothing is judged — which note is a reading,
9
+ # a native spelling, or a date is left to the caller.
10
+ #
11
+ # Positions are character offsets [start, end) into the article's wikitext
12
+ # as stored in the dump, XML entities decoded, before any other processing
13
+ # (comments included), so a caller holding the same text can cut out and
14
+ # hash exactly the span a term came from.
15
+ module LeadTerms
16
+ MAX_TERMS = 5
17
+ HEADING = /^={2,6}[^=\n].*?={2,6}[ \t]*$/
18
+ OPENERS = { "(" => ")", "(" => ")" }.freeze
19
+ SEPARATORS = ["、", ",", ",", ";", ";"].freeze
20
+ PARAGRAPH_BREAK = /\n[ \t\r]*\n/
21
+ # Regions whose bold text is not the article's own lead prose
22
+ SKIP_OPEN = { "{{" => "}}", "{|" => "|}", "<!--" => "-->" }.freeze
23
+ REF_OPEN = /\A<ref(?:\s[^>]*)?>/i
24
+ REF_SELF_CLOSING = /\A<ref(?:\s[^>]*)?\/>/i
25
+ FILE_LINK = /\A\[\[\s*(?:file|image|ファイル|画像|media)\s*:/i
26
+
27
+ module_function
28
+
29
+ # @param wikitext [String] decoded article wikitext
30
+ # @param render [#call] turns a wikitext fragment into clean text
31
+ # @return [Array<Hash>] terms in order of appearance, at most MAX_TERMS
32
+ def extract(wikitext, render:)
33
+ return [] if wikitext.nil? || wikitext.empty?
34
+
35
+ visible = WikitextRegions.mask_lead(wikitext)
36
+ bolds, rubies, lead_end = scan(visible, visible.length, source: wikitext)
37
+ original_render = render
38
+ render = ->(fragment) { original_render.call(fragment.gsub(WikitextRegions::COMMENT, "")) }
39
+ terms = []
40
+
41
+ if (first = bolds.first)
42
+ paragraph = paragraph_bounds(visible, first[0], lead_end)
43
+ bolds.select { |s, e| s >= paragraph[0] && e <= paragraph[1] }.first(MAX_TERMS).each do |s, e|
44
+ terms << bold_term(wikitext, s, e, paragraph[1], render)
45
+ end
46
+ end
47
+ rubies.each do |s, e, parts|
48
+ terms << { "text" => render.call(parts[1].to_s).strip, "reading" => render.call(parts[2].to_s).strip,
49
+ "source" => parts[0].strip, "span" => { "template" => [s, e] } }
50
+ end
51
+
52
+ terms.sort_by { |t| t["span"].values.first.first }.first(MAX_TERMS)
53
+ .each_with_index.map { |t, i| { "index" => i }.merge(t) }
54
+ end
55
+
56
+ # Top-level bold spans [start, end) including the quote marks, and ruby
57
+ # templates [start, end, [name, text, reading]] found in the lead
58
+ def scan(text, limit, source: text)
59
+ bolds = []
60
+ rubies = []
61
+ i = 0
62
+ open_bold = nil
63
+ while i < limit
64
+ if (i.zero? || text[i - 1] == "\n") && text[i] == "=" &&
65
+ HEADING.match?(text[i...(text.index("\n", i) || limit)])
66
+ limit = i
67
+ break
68
+ elsif text[i] == "\n"
69
+ open_bold = nil # bold does not continue across lines
70
+ i += 1
71
+ elsif (close = SKIP_OPEN[text[i, 4] == "<!--" ? "<!--" : text[i, 2]])
72
+ opener = text[i, 4] == "<!--" ? "<!--" : text[i, 2]
73
+ stop = matching_end(text, i, opener, close)
74
+ if opener == "{{" && (ruby = ruby_template(source[(i + 2)...(stop - 2)]))
75
+ rubies << [i, stop, ruby]
76
+ end
77
+ i = stop
78
+ elsif text[i, 2] == "[[" && FILE_LINK.match?(text[i, 40])
79
+ i = matching_end(text, i, "[[", "]]")
80
+ elsif text[i] == "<" && (m = REF_SELF_CLOSING.match(text[i, 200]))
81
+ i += m[0].length
82
+ elsif text[i] == "<" && (m = REF_OPEN.match(text[i, 200]))
83
+ close_at = text.index(%r{</ref\s*>}i, i + m[0].length)
84
+ i = close_at ? text.index(">", close_at) + 1 : limit
85
+ elsif text[i, 3] == "'''"
86
+ run = text[i..][/\A'+/].length
87
+ if open_bold
88
+ bolds << [open_bold, i + run]
89
+ open_bold = nil
90
+ else
91
+ open_bold = i
92
+ end
93
+ i += run
94
+ else
95
+ i += 1
96
+ end
97
+ end
98
+ [bolds, rubies, limit]
99
+ end
100
+
101
+ # End index (exclusive) of the construct opened at start, honouring nesting
102
+ def matching_end(text, start, opener, closer)
103
+ depth = 0
104
+ i = start
105
+ while i < text.length
106
+ if text[i, opener.length] == opener
107
+ depth += 1
108
+ i += opener.length
109
+ elsif text[i, closer.length] == closer
110
+ depth -= 1
111
+ i += closer.length
112
+ return i if depth.zero?
113
+ else
114
+ i += 1
115
+ end
116
+ end
117
+ text.length
118
+ end
119
+
120
+ def ruby_template(content)
121
+ parts = split_top_level(content, ["|"])
122
+ name = parts.first.to_s
123
+ return nil unless ruby_name?(name)
124
+
125
+ [name, parts[1], parts[2]]
126
+ end
127
+
128
+ # Same name rule as the cleaner: "_" and " " alike, case-insensitive
129
+ def ruby_name?(name)
130
+ @ruby_names ||= Wp2txt::RUBY_TEXT_TEMPLATES.to_set { |t| t.tr("_", " ").strip.downcase }
131
+ @ruby_names.include?(name.to_s.tr("_", " ").strip.downcase)
132
+ end
133
+
134
+ def paragraph_bounds(text, pos, limit)
135
+ start = 0
136
+ text[0...pos].to_enum(:scan, PARAGRAPH_BREAK).each { start = Regexp.last_match.end(0) }
137
+ stop = text.index(PARAGRAPH_BREAK, pos) || limit
138
+ [start, [stop, limit].min]
139
+ end
140
+
141
+ def bold_term(text, s, e, limit, render)
142
+ inner = text[s...e].sub(/\A'+/, "").sub(/'+\z/, "")
143
+ term = { "text" => render.call(plain_ruby(inner)).strip, "notes" => [], "notes_text" => nil,
144
+ "source" => "bold", "span" => { "bold" => [s, e] } }
145
+ j = e
146
+ j += 1 while j < limit && [" ", "\t", " ", "\r", "\n"].include?(text[j])
147
+ return term if j >= limit
148
+ closer = OPENERS[text[j]]
149
+ return term unless closer
150
+
151
+ pe = paren_end(text, j, text[j], closer, limit)
152
+ return term unless pe
153
+
154
+ raw = text[(j + 1)...(pe - 1)]
155
+ term["notes_text"] = render.call(raw).strip
156
+ term["notes"] = split_top_level(raw, SEPARATORS).map { |part| render.call(part).strip }.reject(&:empty?)
157
+ term["span"]["paren"] = [j, pe]
158
+ term
159
+ end
160
+
161
+ # Index after the bracket closing the one at start, or nil if unclosed
162
+ def paren_end(text, start, opener, closer, limit)
163
+ depth = 0
164
+ i = start
165
+ while i < limit
166
+ two = text[i, 2]
167
+ if text[i] == "<" && (stop = WikitextRegions.end_at(text, i, lead: true))
168
+ i = stop
169
+ next
170
+ elsif ["{{", "[["].include?(two)
171
+ i = matching_end(text, i, two, two == "{{" ? "}}" : "]]")
172
+ next
173
+ end
174
+ if [opener, "(", "("].include?(text[i])
175
+ depth += 1
176
+ elsif [closer, ")", ")"].include?(text[i])
177
+ depth -= 1
178
+ return i + 1 if depth.zero?
179
+ elsif text[i] == "\n" && /\A\n[ \t\r]*\n/.match?(text[i..])
180
+ return nil
181
+ end
182
+ i += 1
183
+ end
184
+ nil
185
+ end
186
+
187
+ # Split at separators that sit outside brackets, templates, and links
188
+ def split_top_level(text, separators)
189
+ parts = [+""]
190
+ depth = 0
191
+ i = 0
192
+ while i < text.length
193
+ two = text[i, 2]
194
+ if text[i] == "<" && (stop = WikitextRegions.end_at(text, i, lead: true))
195
+ parts.last << text[i...stop]
196
+ i = stop
197
+ elsif ["{{", "[["].include?(two)
198
+ depth += 1
199
+ parts.last << two
200
+ i += 2
201
+ elsif ["}}", "]]"].include?(two)
202
+ depth -= 1
203
+ parts.last << two
204
+ i += 2
205
+ else
206
+ ch = text[i]
207
+ depth += 1 if ["(", "("].include?(ch)
208
+ depth -= 1 if [")", ")"].include?(ch)
209
+ if depth <= 0 && separators.include?(ch)
210
+ parts << +""
211
+ else
212
+ parts.last << ch
213
+ end
214
+ i += 1
215
+ end
216
+ end
217
+ parts
218
+ end
219
+
220
+ # In a bold headword, a ruby template stands for its base text; the
221
+ # reading is reported separately as its own term
222
+ def plain_ruby(fragment)
223
+ fragment.gsub(/\{\{([^{}]*)\}\}/) do |whole|
224
+ (ruby = ruby_template(Regexp.last_match(1))) ? ruby[1].to_s : whole
225
+ end
226
+ end
227
+ end
228
+ end