wp2txt 2.3.2 → 2.3.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/.dockerignore +3 -0
- data/.github/workflows/publish-image.yml +140 -0
- data/CHANGELOG.md +23 -0
- data/DEVELOPMENT.md +12 -4
- data/DEVELOPMENT_ja.md +2 -3
- data/README.md +6 -3
- data/README_ja.md +6 -3
- data/Rakefile +22 -16
- data/bin/wp2txt +26 -14
- data/bin/wp2txt-mcp +4 -3
- data/docs/INDEXES.md +4 -0
- data/lib/wp2txt/article.rb +1 -1
- data/lib/wp2txt/constants.rb +16 -0
- data/lib/wp2txt/corpus.rb +98 -79
- data/lib/wp2txt/corpus_jobs.rb +15 -13
- data/lib/wp2txt/extractor.rb +6 -0
- data/lib/wp2txt/formatter.rb +18 -0
- data/lib/wp2txt/fts_index.rb +12 -3
- data/lib/wp2txt/metadata_index.rb +3 -3
- data/lib/wp2txt/multistream.rb +31 -2
- data/lib/wp2txt/output_path.rb +72 -12
- data/lib/wp2txt/output_writer.rb +8 -0
- data/lib/wp2txt/stream_processor.rb +45 -15
- data/lib/wp2txt/version.rb +1 -1
- data/lib/wp2txt.rb +7 -5
- data/spec/fts_index_spec.rb +40 -0
- data/spec/output_integrity_spec.rb +146 -0
- data/spec/p1_correctness_spec.rb +372 -0
- data/spec/stream_processor_spec.rb +3 -1
- data/spec/titles_output_path_spec.rb +61 -1
- data/wp2txt.gemspec +1 -1
- metadata +6 -9
- data/image/wp2txt-logo.svg +0 -16
- data/image/wp2txt.svg +0 -31
- data/scripts/benchmark_regex.rb +0 -161
- data/scripts/fetch_html_entities.rb +0 -94
- data/scripts/fetch_language_metadata.rb +0 -180
- data/scripts/fetch_mediawiki_data.rb +0 -334
- data/scripts/fetch_template_data.rb +0 -186
- data/scripts/profile_memory.rb +0 -139
data/lib/wp2txt/corpus.rb
CHANGED
|
@@ -14,6 +14,7 @@ require_relative "metadata_index"
|
|
|
14
14
|
require_relative "fts_index"
|
|
15
15
|
require_relative "section_extractor"
|
|
16
16
|
require_relative "version"
|
|
17
|
+
require_relative "output_path"
|
|
17
18
|
|
|
18
19
|
module Wp2txt
|
|
19
20
|
# Facade over a local dump: single-article access (Tier 0, multistream),
|
|
@@ -324,7 +325,7 @@ module Wp2txt
|
|
|
324
325
|
title_match: nil, limit: 0, titles: nil,
|
|
325
326
|
chunk_size: nil, chunk_overlap: 0,
|
|
326
327
|
max_articles: DEFAULT_MAX_SYNC_ARTICLES, num_processes: 4,
|
|
327
|
-
progress: nil, cancel_check: nil)
|
|
328
|
+
progress: nil, cancel_check: nil, overwrite: false)
|
|
328
329
|
if content == "sections" && Array(sections).empty? && alias_set.nil?
|
|
329
330
|
raise ArgumentError, "content: \"sections\" requires sections or alias_set"
|
|
330
331
|
end
|
|
@@ -394,46 +395,50 @@ module Wp2txt
|
|
|
394
395
|
titles_done = 0
|
|
395
396
|
sample = []
|
|
396
397
|
|
|
397
|
-
|
|
398
|
-
|
|
399
|
-
|
|
400
|
-
|
|
401
|
-
|
|
402
|
-
|
|
403
|
-
|
|
404
|
-
|
|
405
|
-
|
|
406
|
-
|
|
407
|
-
|
|
408
|
-
|
|
409
|
-
|
|
410
|
-
|
|
411
|
-
|
|
412
|
-
|
|
413
|
-
|
|
398
|
+
meta_path = "#{output_path}.meta.json"
|
|
399
|
+
OutputPath.write_pair(output_path, overwrite: overwrite) do |staged_output, staged_meta|
|
|
400
|
+
File.open(staged_output, "w") do |f|
|
|
401
|
+
titles.each_slice(EXTRACT_BATCH_SIZE) do |batch|
|
|
402
|
+
raise Cancelled if cancel_check&.call
|
|
403
|
+
|
|
404
|
+
# Forked workers inherit this file's buffer and would write it again on exit
|
|
405
|
+
f.flush
|
|
406
|
+
pages = reader.extract_articles_parallel(batch, num_processes: num_processes)
|
|
407
|
+
batch.each do |t|
|
|
408
|
+
page = pages[t]
|
|
409
|
+
next unless page
|
|
410
|
+
|
|
411
|
+
records = build_records(page, content, resolved_sections, chunk_size, chunk_overlap)
|
|
412
|
+
next if records.empty?
|
|
413
|
+
|
|
414
|
+
articles_extracted += 1
|
|
415
|
+
records.each do |record|
|
|
416
|
+
f.puts(JSON.generate(record))
|
|
417
|
+
records_written += 1
|
|
418
|
+
sample << record if sample.size < 3
|
|
419
|
+
end
|
|
414
420
|
end
|
|
421
|
+
titles_done += batch.size
|
|
422
|
+
progress&.call(titles_done, titles.size)
|
|
415
423
|
end
|
|
416
|
-
titles_done += batch.size
|
|
417
|
-
progress&.call(titles_done, titles.size)
|
|
418
424
|
end
|
|
419
|
-
end
|
|
420
425
|
|
|
421
|
-
|
|
422
|
-
|
|
423
|
-
|
|
424
|
-
|
|
425
|
-
|
|
426
|
-
|
|
427
|
-
|
|
428
|
-
|
|
429
|
-
|
|
430
|
-
|
|
431
|
-
|
|
432
|
-
|
|
433
|
-
|
|
434
|
-
|
|
435
|
-
|
|
436
|
-
|
|
426
|
+
File.write(staged_meta, JSON.pretty_generate(
|
|
427
|
+
tool: "wp2txt #{Wp2txt::VERSION}",
|
|
428
|
+
dump: dump_name,
|
|
429
|
+
generated_at: Time.now.utc.iso8601,
|
|
430
|
+
query: (titles_record || filters.compact).merge(
|
|
431
|
+
content: content, resolved_sections: resolved_sections,
|
|
432
|
+
chunk_size: chunk_size, chunk_overlap: chunk_size ? chunk_overlap : nil
|
|
433
|
+
).compact,
|
|
434
|
+
alias_set_contents: alias_contents,
|
|
435
|
+
total_matching: total,
|
|
436
|
+
articles_extracted: articles_extracted,
|
|
437
|
+
records_written: records_written,
|
|
438
|
+
truncated: truncated,
|
|
439
|
+
not_found: not_found
|
|
440
|
+
))
|
|
441
|
+
end
|
|
437
442
|
|
|
438
443
|
{ output_path: output_path, meta_path: meta_path, dump: dump_name,
|
|
439
444
|
total_matching: total, articles_extracted: articles_extracted,
|
|
@@ -661,15 +666,19 @@ module Wp2txt
|
|
|
661
666
|
def run_sql_on(db, sql, limit)
|
|
662
667
|
columns = nil
|
|
663
668
|
rows = []
|
|
669
|
+
truncated = false
|
|
664
670
|
db.query(sql) do |result|
|
|
665
671
|
columns = result.columns
|
|
666
672
|
result.each do |row|
|
|
667
|
-
|
|
673
|
+
if rows.size >= limit
|
|
674
|
+
truncated = true
|
|
675
|
+
break
|
|
676
|
+
end
|
|
668
677
|
|
|
669
678
|
rows << row.map { |v| v.is_a?(String) && v.length > SQL_CELL_LIMIT ? "#{v[0, SQL_CELL_LIMIT]}…" : v }
|
|
670
679
|
end
|
|
671
680
|
end
|
|
672
|
-
{ columns: columns, rows: rows, row_count: rows.size, truncated:
|
|
681
|
+
{ columns: columns, rows: rows, row_count: rows.size, truncated: truncated }
|
|
673
682
|
end
|
|
674
683
|
|
|
675
684
|
# Execute the query in a forked child with a hard deadline: the child opens
|
|
@@ -727,38 +736,24 @@ module Wp2txt
|
|
|
727
736
|
end
|
|
728
737
|
end
|
|
729
738
|
|
|
730
|
-
#
|
|
731
|
-
#
|
|
732
|
-
# renames it into place only on success and removes it on every failure
|
|
733
|
-
# path (child crash, timeout kill, error over the pipe) — a partially
|
|
734
|
-
# written file is never presented as a result. The .meta.json sidecar is
|
|
735
|
-
# written by the parent after the rename succeeds.
|
|
739
|
+
# Stage JSONL and provenance in unique files, then publish on success.
|
|
740
|
+
# OutputPath owns exclusive destination reservations and failure cleanup.
|
|
736
741
|
def query_sql_to_file(sql, timeout, attachments, output_path, overwrite)
|
|
737
|
-
|
|
738
|
-
|
|
739
|
-
|
|
740
|
-
|
|
741
|
-
|
|
742
|
-
|
|
743
|
-
|
|
744
|
-
|
|
745
|
-
|
|
746
|
-
|
|
747
|
-
|
|
748
|
-
|
|
749
|
-
|
|
750
|
-
ensure
|
|
751
|
-
db.close
|
|
752
|
-
end
|
|
753
|
-
end
|
|
754
|
-
rescue StandardError
|
|
755
|
-
FileUtils.rm_f(partial)
|
|
756
|
-
raise
|
|
742
|
+
outcome = OutputPath.write_pair(output_path, overwrite: overwrite) do |partial, staged_meta|
|
|
743
|
+
result = if Process.respond_to?(:fork)
|
|
744
|
+
run_sql_file_in_subprocess(sql, timeout, attachments, partial)
|
|
745
|
+
else
|
|
746
|
+
db = build_readonly_connection(attach_fts: fts.built?, attachments: attachments)
|
|
747
|
+
begin
|
|
748
|
+
run_sql_file_on(db, sql, partial)
|
|
749
|
+
ensure
|
|
750
|
+
db.close
|
|
751
|
+
end
|
|
752
|
+
end
|
|
753
|
+
write_sql_sidecar(staged_meta, sql, attachments, result)
|
|
754
|
+
result
|
|
757
755
|
end
|
|
758
756
|
|
|
759
|
-
File.rename(partial, output_path)
|
|
760
|
-
write_sql_sidecar(output_path, sql, attachments, outcome)
|
|
761
|
-
|
|
762
757
|
result = { output_path: output_path, meta_path: "#{output_path}.meta.json",
|
|
763
758
|
columns: outcome[:columns], row_count: outcome[:row_count],
|
|
764
759
|
truncated: outcome[:truncated], cells_clipped: outcome[:cells_clipped],
|
|
@@ -772,6 +767,7 @@ module Wp2txt
|
|
|
772
767
|
# subprocess path: the 30s SIGKILL deadline covers the writing too.
|
|
773
768
|
def run_sql_file_on(db, sql, partial_path)
|
|
774
769
|
columns = nil
|
|
770
|
+
column_mapping = nil
|
|
775
771
|
row_count = 0
|
|
776
772
|
cells_clipped = 0
|
|
777
773
|
truncated = false
|
|
@@ -779,7 +775,11 @@ module Wp2txt
|
|
|
779
775
|
|
|
780
776
|
File.open(partial_path, "w") do |f|
|
|
781
777
|
db.query(sql) do |result|
|
|
782
|
-
|
|
778
|
+
original_columns = result.columns
|
|
779
|
+
columns = unique_columns(original_columns)
|
|
780
|
+
column_mapping = original_columns.each_with_index.map do |name, ordinal|
|
|
781
|
+
{ ordinal: ordinal, original_name: name, output_name: columns[ordinal] }
|
|
782
|
+
end
|
|
783
783
|
result.each do |row|
|
|
784
784
|
if row_count >= SQL_FILE_ROW_LIMIT
|
|
785
785
|
truncated = true
|
|
@@ -788,8 +788,8 @@ module Wp2txt
|
|
|
788
788
|
|
|
789
789
|
record = {}
|
|
790
790
|
row.each_with_index do |value, i|
|
|
791
|
-
if value.is_a?(String) && value.
|
|
792
|
-
value =
|
|
791
|
+
if value.is_a?(String) && value.bytesize > SQL_FILE_CELL_LIMIT
|
|
792
|
+
value = clip_file_cell(value)
|
|
793
793
|
cells_clipped += 1
|
|
794
794
|
end
|
|
795
795
|
record[columns[i]] = value
|
|
@@ -801,20 +801,37 @@ module Wp2txt
|
|
|
801
801
|
end
|
|
802
802
|
end
|
|
803
803
|
|
|
804
|
-
{ columns: columns, row_count: row_count, truncated: truncated,
|
|
804
|
+
{ columns: columns, column_mapping: column_mapping, row_count: row_count, truncated: truncated,
|
|
805
805
|
cells_clipped: cells_clipped, sample: sample }
|
|
806
806
|
end
|
|
807
807
|
|
|
808
808
|
# Duplicate result column names (SELECT 1 AS x, 2 AS x) are suffixed
|
|
809
809
|
# (_2, _3, ...) so every JSONL record key is unique
|
|
810
810
|
def unique_columns(columns)
|
|
811
|
-
|
|
812
|
-
|
|
813
|
-
|
|
814
|
-
|
|
811
|
+
reserved = columns.to_h { |name| [name, true] }
|
|
812
|
+
used = {}
|
|
813
|
+
columns.map do |name|
|
|
814
|
+
candidate = name
|
|
815
|
+
suffix = 2
|
|
816
|
+
while used[candidate] || (candidate != name && reserved[candidate])
|
|
817
|
+
candidate = "#{name}_#{suffix}"
|
|
818
|
+
suffix += 1
|
|
819
|
+
end
|
|
820
|
+
used[candidate] = true
|
|
821
|
+
candidate
|
|
815
822
|
end
|
|
816
823
|
end
|
|
817
824
|
|
|
825
|
+
# SQL_FILE_CELL_LIMIT is a byte ceiling INCLUDING the UTF-8 ellipsis.
|
|
826
|
+
# Remove only the incomplete UTF-8 suffix after a byte-based cut.
|
|
827
|
+
def clip_file_cell(value)
|
|
828
|
+
utf8 = value.dup.force_encoding(Encoding::UTF_8)
|
|
829
|
+
raise JSON::GeneratorError, "SQL cell contains invalid UTF-8" unless utf8.valid_encoding?
|
|
830
|
+
|
|
831
|
+
prefix = utf8.byteslice(0, SQL_FILE_CELL_LIMIT - "…".bytesize).force_encoding(Encoding::UTF_8)
|
|
832
|
+
"#{prefix.scrub("")}…"
|
|
833
|
+
end
|
|
834
|
+
|
|
818
835
|
# Subprocess driver for file-output mode; same fork/pipe/SIGKILL
|
|
819
836
|
# structure as run_sql_in_subprocess, but the child writes the rows to
|
|
820
837
|
# partial_path and ships back only the summary
|
|
@@ -856,9 +873,9 @@ module Wp2txt
|
|
|
856
873
|
reader_io&.close
|
|
857
874
|
end
|
|
858
875
|
|
|
859
|
-
# Reproducibility sidecar,
|
|
860
|
-
def write_sql_sidecar(
|
|
861
|
-
File.write(
|
|
876
|
+
# Reproducibility sidecar, staged by the parent before publication.
|
|
877
|
+
def write_sql_sidecar(meta_path, sql, attachments, outcome)
|
|
878
|
+
File.write(meta_path, JSON.pretty_generate(
|
|
862
879
|
tool: "query_sql",
|
|
863
880
|
dump: dump_name,
|
|
864
881
|
built_with: @metadata.stats&.dig(:built_with),
|
|
@@ -867,6 +884,8 @@ module Wp2txt
|
|
|
867
884
|
row_count: outcome[:row_count],
|
|
868
885
|
truncated: outcome[:truncated],
|
|
869
886
|
cells_clipped: outcome[:cells_clipped],
|
|
887
|
+
column_mapping: outcome[:column_mapping],
|
|
888
|
+
cell_byte_limit: SQL_FILE_CELL_LIMIT,
|
|
870
889
|
generated_at: Time.now.utc.iso8601,
|
|
871
890
|
wp2txt_version: Wp2txt::VERSION
|
|
872
891
|
))
|
data/lib/wp2txt/corpus_jobs.rb
CHANGED
|
@@ -22,22 +22,24 @@ module Wp2txt
|
|
|
22
22
|
# unbounded concurrent jobs would multiply workers against the same dump.
|
|
23
23
|
# @return [Hash] { job_id:, status: "running" } or { error: ... }
|
|
24
24
|
def start_extract(params)
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
|
|
25
|
+
job_id = @mutex.synchronize do
|
|
26
|
+
running = @jobs.values.find { |state| state[:status] == "running" }
|
|
27
|
+
if running
|
|
28
|
+
return { error: "another job is already running (#{running[:job_id]}); " \
|
|
29
|
+
"poll job_status or cancel_job before starting a new one" }
|
|
30
|
+
end
|
|
31
|
+
id = format("job-%04d", @seq += 1)
|
|
32
|
+
@jobs[id] = {
|
|
33
|
+
job_id: id, status: "running", started_at: Time.now.utc.iso8601,
|
|
34
|
+
params: params, titles_done: 0, titles_total: nil, cancel: false
|
|
35
|
+
}
|
|
36
|
+
id
|
|
29
37
|
end
|
|
30
38
|
|
|
31
|
-
job_id = @mutex.synchronize { format("job-%04d", @seq += 1) }
|
|
32
|
-
state = {
|
|
33
|
-
job_id: job_id, status: "running", started_at: Time.now.utc.iso8601,
|
|
34
|
-
params: params, titles_done: 0, titles_total: nil, cancel: false
|
|
35
|
-
}
|
|
36
|
-
@mutex.synchronize { @jobs[job_id] = state }
|
|
37
|
-
|
|
38
39
|
thread = Thread.new do
|
|
39
|
-
corpus =
|
|
40
|
+
corpus = nil
|
|
40
41
|
begin
|
|
42
|
+
corpus = @factory.call
|
|
41
43
|
result = corpus.extract_corpus(
|
|
42
44
|
**params,
|
|
43
45
|
max_articles: nil,
|
|
@@ -52,7 +54,7 @@ module Wp2txt
|
|
|
52
54
|
rescue StandardError => e
|
|
53
55
|
update(job_id) { |s| s[:status] = "error"; s[:error] = "#{e.class}: #{e.message}"; s[:finished_at] = Time.now.utc.iso8601 }
|
|
54
56
|
ensure
|
|
55
|
-
corpus
|
|
57
|
+
corpus&.close
|
|
56
58
|
end
|
|
57
59
|
end
|
|
58
60
|
thread.report_on_exception = false
|
data/lib/wp2txt/extractor.rb
CHANGED
|
@@ -111,6 +111,8 @@ module Wp2txt
|
|
|
111
111
|
|
|
112
112
|
if page
|
|
113
113
|
article = Article.new(page[:text], page[:title], !config[:marker])
|
|
114
|
+
article.page_id = page[:id]
|
|
115
|
+
article.revision_id = page[:revision_id]
|
|
114
116
|
result = format_article(article, config)
|
|
115
117
|
writer.write(result)
|
|
116
118
|
extracted_count += 1
|
|
@@ -480,6 +482,8 @@ module Wp2txt
|
|
|
480
482
|
|
|
481
483
|
pages.each do |page|
|
|
482
484
|
article = Article.new(page[:text], page[:title], !config[:marker])
|
|
485
|
+
article.page_id = page[:id]
|
|
486
|
+
article.revision_id = page[:revision_id]
|
|
483
487
|
result = format_article(article, config)
|
|
484
488
|
writer.write(result)
|
|
485
489
|
extracted_count += 1
|
|
@@ -493,6 +497,8 @@ module Wp2txt
|
|
|
493
497
|
|
|
494
498
|
if page
|
|
495
499
|
article = Article.new(page[:text], page[:title], !config[:marker])
|
|
500
|
+
article.page_id = page[:id]
|
|
501
|
+
article.revision_id = page[:revision_id]
|
|
496
502
|
result = format_article(article, config)
|
|
497
503
|
writer.write(result)
|
|
498
504
|
extracted_count += 1
|
data/lib/wp2txt/formatter.rb
CHANGED
|
@@ -14,6 +14,24 @@ module Wp2txt
|
|
|
14
14
|
|
|
15
15
|
# Format article based on configuration and output format
|
|
16
16
|
def format_article(article, config)
|
|
17
|
+
with_page_ids(format_article_body(article, config), article)
|
|
18
|
+
end
|
|
19
|
+
|
|
20
|
+
# JSON records carry the dump's page and revision IDs right after the title,
|
|
21
|
+
# when the reader supplied them, so a record can be traced back to its source.
|
|
22
|
+
def with_page_ids(result, article)
|
|
23
|
+
return result unless result.is_a?(Hash) && (article.page_id || article.revision_id)
|
|
24
|
+
|
|
25
|
+
result.each_with_object({}) do |(key, value), out|
|
|
26
|
+
out[key] = value
|
|
27
|
+
next unless key == "title"
|
|
28
|
+
|
|
29
|
+
out["page_id"] = article.page_id
|
|
30
|
+
out["revision_id"] = article.revision_id
|
|
31
|
+
end
|
|
32
|
+
end
|
|
33
|
+
|
|
34
|
+
def format_article_body(article, config)
|
|
17
35
|
# Store original title for magic word expansion in content
|
|
18
36
|
original_title = article.title.dup
|
|
19
37
|
article.title = format_wiki(article.title, config)
|
data/lib/wp2txt/fts_index.rb
CHANGED
|
@@ -19,6 +19,12 @@ module Wp2txt
|
|
|
19
19
|
# Queries ATTACH the Tier 1 metadata DB so category/section/redirect
|
|
20
20
|
# filters compose with MATCH in plain SQL.
|
|
21
21
|
class FtsIndex
|
|
22
|
+
class ShortQueryError < ArgumentError
|
|
23
|
+
def code
|
|
24
|
+
"query_too_short"
|
|
25
|
+
end
|
|
26
|
+
end
|
|
27
|
+
|
|
22
28
|
SCHEMA_VERSION = 2
|
|
23
29
|
CACHE_SUFFIX = "_fts.sqlite3"
|
|
24
30
|
|
|
@@ -206,6 +212,9 @@ module Wp2txt
|
|
|
206
212
|
# @return [Hash] { total:, total_is_capped:, hits: [{page_id:, title:, heading:, ord:}] }
|
|
207
213
|
def search(query, mode: "phrase", sections: nil, category: nil, depth: 0,
|
|
208
214
|
limit: 20, offset: 0, count: "capped", count_cap: 1000)
|
|
215
|
+
if mode == "phrase" && tokenizer == "trigram" && query.length < 3
|
|
216
|
+
raise ShortQueryError, "trigram phrase searches require at least 3 Unicode characters"
|
|
217
|
+
end
|
|
209
218
|
match_expr = mode == "query" ? query : phrase_query(query)
|
|
210
219
|
|
|
211
220
|
conds = ["fts_sections MATCH ?", "p.namespace = 0", "p.redirect_to IS NULL"]
|
|
@@ -386,10 +395,10 @@ module Wp2txt
|
|
|
386
395
|
batches = pairs.each_slice(STREAMS_PER_BATCH).to_a
|
|
387
396
|
done = 0
|
|
388
397
|
|
|
389
|
-
Parallel.
|
|
398
|
+
Parallel.each(
|
|
390
399
|
batches,
|
|
391
400
|
in_processes: @num_processes,
|
|
392
|
-
|
|
401
|
+
finish: lambda { |_item, _idx, rows|
|
|
393
402
|
index.insert_batch(rows)
|
|
394
403
|
done += 1
|
|
395
404
|
progress&.call(done, batches.size)
|
|
@@ -423,7 +432,7 @@ module Wp2txt
|
|
|
423
432
|
title = block[MetadataIndexBuilder::TITLE_REGEX, 1]
|
|
424
433
|
return unless title && !title.empty?
|
|
425
434
|
|
|
426
|
-
ns = (block[MetadataIndexBuilder::NS_REGEX, 1]
|
|
435
|
+
ns = Wp2txt.namespace_id(block[MetadataIndexBuilder::NS_REGEX, 1])
|
|
427
436
|
return unless ns.zero?
|
|
428
437
|
|
|
429
438
|
page_id = block[MetadataIndexBuilder::ID_REGEX, 1]&.to_i
|
|
@@ -640,10 +640,10 @@ module Wp2txt
|
|
|
640
640
|
batches = pairs.each_slice(STREAMS_PER_BATCH).to_a
|
|
641
641
|
done = 0
|
|
642
642
|
|
|
643
|
-
Parallel.
|
|
643
|
+
Parallel.each(
|
|
644
644
|
batches,
|
|
645
645
|
in_processes: @num_processes,
|
|
646
|
-
|
|
646
|
+
finish: lambda { |_item, _idx, result|
|
|
647
647
|
index.insert_batch(result)
|
|
648
648
|
done += 1
|
|
649
649
|
progress&.call(done, batches.size)
|
|
@@ -686,7 +686,7 @@ module Wp2txt
|
|
|
686
686
|
return unless page_id
|
|
687
687
|
|
|
688
688
|
title = unescape_xml(title)
|
|
689
|
-
ns = (block[NS_REGEX, 1]
|
|
689
|
+
ns = Wp2txt.namespace_id(block[NS_REGEX, 1])
|
|
690
690
|
text = block[TEXT_REGEX, 1] || ""
|
|
691
691
|
text = unescape_xml(text)
|
|
692
692
|
# Strip HTML comments before scanning, matching what the Article parser
|
data/lib/wp2txt/multistream.rb
CHANGED
|
@@ -95,6 +95,11 @@ module Wp2txt
|
|
|
95
95
|
@early_terminated == true
|
|
96
96
|
end
|
|
97
97
|
|
|
98
|
+
# Where the stream holding the last found article ends. Scanning stops as
|
|
99
|
+
# soon as every target is found, so without this the reader cannot tell how
|
|
100
|
+
# far that stream extends and would read the dump to its end.
|
|
101
|
+
attr_reader :stream_end_offset
|
|
102
|
+
|
|
98
103
|
def find_by_title(title)
|
|
99
104
|
@entries_by_title[title]
|
|
100
105
|
end
|
|
@@ -179,6 +184,14 @@ module Wp2txt
|
|
|
179
184
|
end
|
|
180
185
|
end
|
|
181
186
|
|
|
187
|
+
def next_stream_offset_after(io, offset)
|
|
188
|
+
io.each_line do |line|
|
|
189
|
+
next_offset = line.split(":", 2).first.to_i
|
|
190
|
+
return next_offset if next_offset > offset
|
|
191
|
+
end
|
|
192
|
+
nil
|
|
193
|
+
end
|
|
194
|
+
|
|
182
195
|
def parse_index_stream(io)
|
|
183
196
|
count = 0
|
|
184
197
|
io.each_line do |line|
|
|
@@ -205,6 +218,7 @@ module Wp2txt
|
|
|
205
218
|
@found_targets << title if @target_titles.include?(title)
|
|
206
219
|
if @found_targets.size == @target_titles.size
|
|
207
220
|
@early_terminated = true
|
|
221
|
+
@stream_end_offset = next_stream_offset_after(io, offset)
|
|
208
222
|
print "\r Found all #{@target_titles.size} target articles" if @show_progress
|
|
209
223
|
puts if @show_progress
|
|
210
224
|
break
|
|
@@ -223,6 +237,10 @@ module Wp2txt
|
|
|
223
237
|
|
|
224
238
|
# Reads articles from multistream bz2 files
|
|
225
239
|
class MultistreamReader
|
|
240
|
+
# A single multistream bz2 stream holds about 100 pages; anything this large
|
|
241
|
+
# past the last known offset is several streams, not one.
|
|
242
|
+
MAX_TAIL_STREAM_BYTES = 64 * 1024 * 1024
|
|
243
|
+
|
|
226
244
|
attr_reader :multistream_path, :index
|
|
227
245
|
|
|
228
246
|
# Initialize reader with multistream file and index
|
|
@@ -357,7 +375,14 @@ module Wp2txt
|
|
|
357
375
|
if next_offset
|
|
358
376
|
compressed_data = f.read(next_offset - offset)
|
|
359
377
|
else
|
|
360
|
-
# Last stream
|
|
378
|
+
# Last stream of the dump: the rest of the file is that one stream.
|
|
379
|
+
# A large remainder means the end was never recorded, and reading it
|
|
380
|
+
# in one call fails outright on some platforms (EINVAL on macOS).
|
|
381
|
+
remaining = File.size(@multistream_path) - offset
|
|
382
|
+
if remaining > MAX_TAIL_STREAM_BYTES
|
|
383
|
+
raise Wp2txt::Error, "cannot locate the end of the stream at offset #{offset} " \
|
|
384
|
+
"(#{remaining} bytes to end of file); the index may be incomplete"
|
|
385
|
+
end
|
|
361
386
|
compressed_data = f.read
|
|
362
387
|
end
|
|
363
388
|
|
|
@@ -372,7 +397,9 @@ module Wp2txt
|
|
|
372
397
|
# here would silently read gigabytes to EOF, so fail fast instead
|
|
373
398
|
raise "Stream offset #{current_offset} not found in index (#{offsets.size} streams known)" unless idx
|
|
374
399
|
|
|
375
|
-
offsets[idx + 1]
|
|
400
|
+
return offsets[idx + 1] if idx + 1 < offsets.size
|
|
401
|
+
|
|
402
|
+
@index.respond_to?(:stream_end_offset) ? @index.stream_end_offset : nil
|
|
376
403
|
end
|
|
377
404
|
|
|
378
405
|
def decompress_bz2(data)
|
|
@@ -394,6 +421,7 @@ module Wp2txt
|
|
|
394
421
|
return {
|
|
395
422
|
title: page_title,
|
|
396
423
|
id: page_node.at_xpath("id")&.text&.to_i,
|
|
424
|
+
revision_id: page_node.at_xpath("revision/id")&.text&.to_i,
|
|
397
425
|
text: page_node.at_xpath(".//text")&.text || ""
|
|
398
426
|
}
|
|
399
427
|
end
|
|
@@ -409,6 +437,7 @@ module Wp2txt
|
|
|
409
437
|
page = {
|
|
410
438
|
title: page_node.at_xpath("title")&.text,
|
|
411
439
|
id: page_node.at_xpath("id")&.text&.to_i,
|
|
440
|
+
revision_id: page_node.at_xpath("revision/id")&.text&.to_i,
|
|
412
441
|
text: page_node.at_xpath(".//text")&.text || ""
|
|
413
442
|
}
|
|
414
443
|
yield page if page[:title]
|
data/lib/wp2txt/output_path.rb
CHANGED
|
@@ -1,27 +1,87 @@
|
|
|
1
1
|
# frozen_string_literal: true
|
|
2
2
|
|
|
3
|
+
require "tempfile"
|
|
4
|
+
require "tmpdir"
|
|
5
|
+
|
|
3
6
|
module Wp2txt
|
|
4
|
-
#
|
|
5
|
-
# (
|
|
6
|
-
#
|
|
7
|
-
# up paths must not be able to clobber arbitrary user files — and existing
|
|
8
|
-
# files are not replaced unless overwrite is set.
|
|
7
|
+
# Output directories are dedicated, trusted directories. Reject links below
|
|
8
|
+
# the trusted platform temp root (whose ancestors may be OS aliases on macOS).
|
|
9
|
+
# This does not defend against an attacker replacing parent directories.
|
|
9
10
|
module OutputPath
|
|
10
11
|
module_function
|
|
11
12
|
|
|
12
|
-
|
|
13
|
-
|
|
13
|
+
def reject_symlinks!(path)
|
|
14
|
+
current = File.expand_path(path)
|
|
15
|
+
temp_roots = [File.expand_path(Dir.tmpdir), File.realpath(Dir.tmpdir)]
|
|
16
|
+
loop do
|
|
17
|
+
raise ArgumentError, "symbolic links are not allowed in output paths: #{current}" if File.symlink?(current)
|
|
18
|
+
parent = File.dirname(current)
|
|
19
|
+
break if parent == current || temp_roots.include?(current)
|
|
20
|
+
|
|
21
|
+
current = parent
|
|
22
|
+
end
|
|
23
|
+
end
|
|
24
|
+
|
|
25
|
+
def validate_pair!(path, overwrite: false)
|
|
26
|
+
[path, "#{path}.meta.json"].each do |destination|
|
|
27
|
+
reject_symlinks!(destination)
|
|
28
|
+
if File.exist?(destination) && !overwrite
|
|
29
|
+
raise ArgumentError, "output file already exists: #{destination} (pass overwrite: true to replace it)"
|
|
30
|
+
end
|
|
31
|
+
if File.exist?(destination) && !File.file?(destination)
|
|
32
|
+
raise ArgumentError, "output destination must be a file: #{destination}"
|
|
33
|
+
end
|
|
34
|
+
end
|
|
35
|
+
end
|
|
36
|
+
|
|
37
|
+
# Return the confined absolute path; the writer repeats validation and
|
|
38
|
+
# reserves both destinations with EXCL to close the check/create race.
|
|
14
39
|
def confine(output_path, base_dir, overwrite: false)
|
|
15
40
|
path = File.expand_path(output_path, base_dir)
|
|
16
41
|
base = File.expand_path(base_dir)
|
|
17
|
-
unless path
|
|
42
|
+
unless path.start_with?(base + File::SEPARATOR)
|
|
18
43
|
raise ArgumentError, "output_path must stay within the server output directory (#{base})"
|
|
19
44
|
end
|
|
20
|
-
|
|
21
|
-
raise ArgumentError, "output file already exists: #{path} (pass overwrite: true to replace it)"
|
|
22
|
-
end
|
|
23
|
-
|
|
45
|
+
validate_pair!(path, overwrite: overwrite)
|
|
24
46
|
path
|
|
25
47
|
end
|
|
48
|
+
|
|
49
|
+
# Reserve destinations, stage both files, then publish with rename. EXCL
|
|
50
|
+
# reservations are empty until publication; the pair is not a transaction.
|
|
51
|
+
# Failure removes only reservations owned by this call, never prior output.
|
|
52
|
+
def write_pair(path, overwrite: false)
|
|
53
|
+
validate_pair!(path, overwrite: overwrite)
|
|
54
|
+
destinations = [path, "#{path}.meta.json"]
|
|
55
|
+
reservations = {}
|
|
56
|
+
temporary = []
|
|
57
|
+
begin
|
|
58
|
+
unless overwrite
|
|
59
|
+
destinations.each do |destination|
|
|
60
|
+
File.open(destination, File::WRONLY | File::CREAT | File::EXCL, 0o600) do |file|
|
|
61
|
+
reservations[destination] = file.stat
|
|
62
|
+
end
|
|
63
|
+
end
|
|
64
|
+
end
|
|
65
|
+
destinations.each do |destination|
|
|
66
|
+
temporary << Tempfile.create([".wp2txt-", ".partial"], File.dirname(destination))
|
|
67
|
+
temporary.last.close
|
|
68
|
+
end
|
|
69
|
+
result = yield(*temporary.map(&:path))
|
|
70
|
+
destinations.each_with_index do |destination, index|
|
|
71
|
+
reject_symlinks!(destination)
|
|
72
|
+
File.rename(temporary[index].path, destination)
|
|
73
|
+
reservations.delete(destination)
|
|
74
|
+
end
|
|
75
|
+
result
|
|
76
|
+
rescue Errno::EEXIST => e
|
|
77
|
+
raise ArgumentError, "output file already exists: #{e.message} (pass overwrite: true to replace it)"
|
|
78
|
+
ensure
|
|
79
|
+
temporary.each { |file| File.unlink(file.path) if File.exist?(file.path) }
|
|
80
|
+
reservations.each do |destination, stat|
|
|
81
|
+
current = File.lstat(destination) rescue nil
|
|
82
|
+
File.unlink(destination) if current && current.dev == stat.dev && current.ino == stat.ino
|
|
83
|
+
end
|
|
84
|
+
end
|
|
85
|
+
end
|
|
26
86
|
end
|
|
27
87
|
end
|
data/lib/wp2txt/output_writer.rb
CHANGED
|
@@ -100,6 +100,14 @@ module Wp2txt
|
|
|
100
100
|
raise Wp2txt::FileIOError, "Write failed: #{e.message}"
|
|
101
101
|
end
|
|
102
102
|
|
|
103
|
+
# Push buffered output to the file. Call before forking: a child process
|
|
104
|
+
# inherits the buffer and writes it out again when it exits.
|
|
105
|
+
def flush
|
|
106
|
+
@mutex.synchronize do
|
|
107
|
+
@current_file.flush if @current_file && !@current_file.closed?
|
|
108
|
+
end
|
|
109
|
+
end
|
|
110
|
+
|
|
103
111
|
# Close current file and finalize
|
|
104
112
|
def close
|
|
105
113
|
@mutex.synchronize do
|