wp2txt 2.3.2 → 2.3.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
data/lib/wp2txt/corpus.rb CHANGED
@@ -14,6 +14,7 @@ require_relative "metadata_index"
14
14
  require_relative "fts_index"
15
15
  require_relative "section_extractor"
16
16
  require_relative "version"
17
+ require_relative "output_path"
17
18
 
18
19
  module Wp2txt
19
20
  # Facade over a local dump: single-article access (Tier 0, multistream),
@@ -324,7 +325,7 @@ module Wp2txt
324
325
  title_match: nil, limit: 0, titles: nil,
325
326
  chunk_size: nil, chunk_overlap: 0,
326
327
  max_articles: DEFAULT_MAX_SYNC_ARTICLES, num_processes: 4,
327
- progress: nil, cancel_check: nil)
328
+ progress: nil, cancel_check: nil, overwrite: false)
328
329
  if content == "sections" && Array(sections).empty? && alias_set.nil?
329
330
  raise ArgumentError, "content: \"sections\" requires sections or alias_set"
330
331
  end
@@ -394,46 +395,50 @@ module Wp2txt
394
395
  titles_done = 0
395
396
  sample = []
396
397
 
397
- File.open(output_path, "w") do |f|
398
- titles.each_slice(EXTRACT_BATCH_SIZE) do |batch|
399
- raise Cancelled if cancel_check&.call
400
-
401
- pages = reader.extract_articles_parallel(batch, num_processes: num_processes)
402
- batch.each do |t|
403
- page = pages[t]
404
- next unless page
405
-
406
- records = build_records(page, content, resolved_sections, chunk_size, chunk_overlap)
407
- next if records.empty?
408
-
409
- articles_extracted += 1
410
- records.each do |record|
411
- f.puts(JSON.generate(record))
412
- records_written += 1
413
- sample << record if sample.size < 3
398
+ meta_path = "#{output_path}.meta.json"
399
+ OutputPath.write_pair(output_path, overwrite: overwrite) do |staged_output, staged_meta|
400
+ File.open(staged_output, "w") do |f|
401
+ titles.each_slice(EXTRACT_BATCH_SIZE) do |batch|
402
+ raise Cancelled if cancel_check&.call
403
+
404
+ # Forked workers inherit this file's buffer and would write it again on exit
405
+ f.flush
406
+ pages = reader.extract_articles_parallel(batch, num_processes: num_processes)
407
+ batch.each do |t|
408
+ page = pages[t]
409
+ next unless page
410
+
411
+ records = build_records(page, content, resolved_sections, chunk_size, chunk_overlap)
412
+ next if records.empty?
413
+
414
+ articles_extracted += 1
415
+ records.each do |record|
416
+ f.puts(JSON.generate(record))
417
+ records_written += 1
418
+ sample << record if sample.size < 3
419
+ end
414
420
  end
421
+ titles_done += batch.size
422
+ progress&.call(titles_done, titles.size)
415
423
  end
416
- titles_done += batch.size
417
- progress&.call(titles_done, titles.size)
418
424
  end
419
- end
420
425
 
421
- meta_path = "#{output_path}.meta.json"
422
- File.write(meta_path, JSON.pretty_generate(
423
- tool: "wp2txt #{Wp2txt::VERSION}",
424
- dump: dump_name,
425
- generated_at: Time.now.utc.iso8601,
426
- query: (titles_record || filters.compact).merge(
427
- content: content, resolved_sections: resolved_sections,
428
- chunk_size: chunk_size, chunk_overlap: chunk_size ? chunk_overlap : nil
429
- ).compact,
430
- alias_set_contents: alias_contents,
431
- total_matching: total,
432
- articles_extracted: articles_extracted,
433
- records_written: records_written,
434
- truncated: truncated,
435
- not_found: not_found
436
- ))
426
+ File.write(staged_meta, JSON.pretty_generate(
427
+ tool: "wp2txt #{Wp2txt::VERSION}",
428
+ dump: dump_name,
429
+ generated_at: Time.now.utc.iso8601,
430
+ query: (titles_record || filters.compact).merge(
431
+ content: content, resolved_sections: resolved_sections,
432
+ chunk_size: chunk_size, chunk_overlap: chunk_size ? chunk_overlap : nil
433
+ ).compact,
434
+ alias_set_contents: alias_contents,
435
+ total_matching: total,
436
+ articles_extracted: articles_extracted,
437
+ records_written: records_written,
438
+ truncated: truncated,
439
+ not_found: not_found
440
+ ))
441
+ end
437
442
 
438
443
  { output_path: output_path, meta_path: meta_path, dump: dump_name,
439
444
  total_matching: total, articles_extracted: articles_extracted,
@@ -661,15 +666,19 @@ module Wp2txt
661
666
  def run_sql_on(db, sql, limit)
662
667
  columns = nil
663
668
  rows = []
669
+ truncated = false
664
670
  db.query(sql) do |result|
665
671
  columns = result.columns
666
672
  result.each do |row|
667
- break if rows.size >= limit
673
+ if rows.size >= limit
674
+ truncated = true
675
+ break
676
+ end
668
677
 
669
678
  rows << row.map { |v| v.is_a?(String) && v.length > SQL_CELL_LIMIT ? "#{v[0, SQL_CELL_LIMIT]}…" : v }
670
679
  end
671
680
  end
672
- { columns: columns, rows: rows, row_count: rows.size, truncated: rows.size >= limit }
681
+ { columns: columns, rows: rows, row_count: rows.size, truncated: truncated }
673
682
  end
674
683
 
675
684
  # Execute the query in a forked child with a hard deadline: the child opens
@@ -727,38 +736,24 @@ module Wp2txt
727
736
  end
728
737
  end
729
738
 
730
- # Write the full query result to output_path as JSONL. Atomicity: the
731
- # child (or inline fallback) writes "#{output_path}.partial"; the parent
732
- # renames it into place only on success and removes it on every failure
733
- # path (child crash, timeout kill, error over the pipe) — a partially
734
- # written file is never presented as a result. The .meta.json sidecar is
735
- # written by the parent after the rename succeeds.
739
+ # Stage JSONL and provenance in unique files, then publish on success.
740
+ # OutputPath owns exclusive destination reservations and failure cleanup.
736
741
  def query_sql_to_file(sql, timeout, attachments, output_path, overwrite)
737
- if File.exist?(output_path) && !overwrite
738
- raise ArgumentError, "output file already exists: #{output_path} (pass overwrite: true to replace it)"
739
- end
740
-
741
- partial = "#{output_path}.partial"
742
- FileUtils.rm_f(partial)
743
- outcome = begin
744
- if Process.respond_to?(:fork)
745
- run_sql_file_in_subprocess(sql, timeout, attachments, partial)
746
- else
747
- db = build_readonly_connection(attach_fts: fts.built?, attachments: attachments)
748
- begin
749
- run_sql_file_on(db, sql, partial)
750
- ensure
751
- db.close
752
- end
753
- end
754
- rescue StandardError
755
- FileUtils.rm_f(partial)
756
- raise
742
+ outcome = OutputPath.write_pair(output_path, overwrite: overwrite) do |partial, staged_meta|
743
+ result = if Process.respond_to?(:fork)
744
+ run_sql_file_in_subprocess(sql, timeout, attachments, partial)
745
+ else
746
+ db = build_readonly_connection(attach_fts: fts.built?, attachments: attachments)
747
+ begin
748
+ run_sql_file_on(db, sql, partial)
749
+ ensure
750
+ db.close
751
+ end
752
+ end
753
+ write_sql_sidecar(staged_meta, sql, attachments, result)
754
+ result
757
755
  end
758
756
 
759
- File.rename(partial, output_path)
760
- write_sql_sidecar(output_path, sql, attachments, outcome)
761
-
762
757
  result = { output_path: output_path, meta_path: "#{output_path}.meta.json",
763
758
  columns: outcome[:columns], row_count: outcome[:row_count],
764
759
  truncated: outcome[:truncated], cells_clipped: outcome[:cells_clipped],
@@ -772,6 +767,7 @@ module Wp2txt
772
767
  # subprocess path: the 30s SIGKILL deadline covers the writing too.
773
768
  def run_sql_file_on(db, sql, partial_path)
774
769
  columns = nil
770
+ column_mapping = nil
775
771
  row_count = 0
776
772
  cells_clipped = 0
777
773
  truncated = false
@@ -779,7 +775,11 @@ module Wp2txt
779
775
 
780
776
  File.open(partial_path, "w") do |f|
781
777
  db.query(sql) do |result|
782
- columns = unique_columns(result.columns)
778
+ original_columns = result.columns
779
+ columns = unique_columns(original_columns)
780
+ column_mapping = original_columns.each_with_index.map do |name, ordinal|
781
+ { ordinal: ordinal, original_name: name, output_name: columns[ordinal] }
782
+ end
783
783
  result.each do |row|
784
784
  if row_count >= SQL_FILE_ROW_LIMIT
785
785
  truncated = true
@@ -788,8 +788,8 @@ module Wp2txt
788
788
 
789
789
  record = {}
790
790
  row.each_with_index do |value, i|
791
- if value.is_a?(String) && value.length > SQL_FILE_CELL_LIMIT
792
- value = "#{value[0, SQL_FILE_CELL_LIMIT]}…"
791
+ if value.is_a?(String) && value.bytesize > SQL_FILE_CELL_LIMIT
792
+ value = clip_file_cell(value)
793
793
  cells_clipped += 1
794
794
  end
795
795
  record[columns[i]] = value
@@ -801,20 +801,37 @@ module Wp2txt
801
801
  end
802
802
  end
803
803
 
804
- { columns: columns, row_count: row_count, truncated: truncated,
804
+ { columns: columns, column_mapping: column_mapping, row_count: row_count, truncated: truncated,
805
805
  cells_clipped: cells_clipped, sample: sample }
806
806
  end
807
807
 
808
808
  # Duplicate result column names (SELECT 1 AS x, 2 AS x) are suffixed
809
809
  # (_2, _3, ...) so every JSONL record key is unique
810
810
  def unique_columns(columns)
811
- seen = Hash.new(0)
812
- columns.map do |c|
813
- seen[c] += 1
814
- seen[c] == 1 ? c : "#{c}_#{seen[c]}"
811
+ reserved = columns.to_h { |name| [name, true] }
812
+ used = {}
813
+ columns.map do |name|
814
+ candidate = name
815
+ suffix = 2
816
+ while used[candidate] || (candidate != name && reserved[candidate])
817
+ candidate = "#{name}_#{suffix}"
818
+ suffix += 1
819
+ end
820
+ used[candidate] = true
821
+ candidate
815
822
  end
816
823
  end
817
824
 
825
+ # SQL_FILE_CELL_LIMIT is a byte ceiling INCLUDING the UTF-8 ellipsis.
826
+ # Remove only the incomplete UTF-8 suffix after a byte-based cut.
827
+ def clip_file_cell(value)
828
+ utf8 = value.dup.force_encoding(Encoding::UTF_8)
829
+ raise JSON::GeneratorError, "SQL cell contains invalid UTF-8" unless utf8.valid_encoding?
830
+
831
+ prefix = utf8.byteslice(0, SQL_FILE_CELL_LIMIT - "…".bytesize).force_encoding(Encoding::UTF_8)
832
+ "#{prefix.scrub("")}…"
833
+ end
834
+
818
835
  # Subprocess driver for file-output mode; same fork/pipe/SIGKILL
819
836
  # structure as run_sql_in_subprocess, but the child writes the rows to
820
837
  # partial_path and ships back only the summary
@@ -856,9 +873,9 @@ module Wp2txt
856
873
  reader_io&.close
857
874
  end
858
875
 
859
- # Reproducibility sidecar, written by the parent after the atomic rename
860
- def write_sql_sidecar(output_path, sql, attachments, outcome)
861
- File.write("#{output_path}.meta.json", JSON.pretty_generate(
876
+ # Reproducibility sidecar, staged by the parent before publication.
877
+ def write_sql_sidecar(meta_path, sql, attachments, outcome)
878
+ File.write(meta_path, JSON.pretty_generate(
862
879
  tool: "query_sql",
863
880
  dump: dump_name,
864
881
  built_with: @metadata.stats&.dig(:built_with),
@@ -867,6 +884,8 @@ module Wp2txt
867
884
  row_count: outcome[:row_count],
868
885
  truncated: outcome[:truncated],
869
886
  cells_clipped: outcome[:cells_clipped],
887
+ column_mapping: outcome[:column_mapping],
888
+ cell_byte_limit: SQL_FILE_CELL_LIMIT,
870
889
  generated_at: Time.now.utc.iso8601,
871
890
  wp2txt_version: Wp2txt::VERSION
872
891
  ))
@@ -22,22 +22,24 @@ module Wp2txt
22
22
  # unbounded concurrent jobs would multiply workers against the same dump.
23
23
  # @return [Hash] { job_id:, status: "running" } or { error: ... }
24
24
  def start_extract(params)
25
- running = @mutex.synchronize { @jobs.values.find { |s| s[:status] == "running" } }
26
- if running
27
- return { error: "another job is already running (#{running[:job_id]}); " \
28
- "poll job_status or cancel_job before starting a new one" }
25
+ job_id = @mutex.synchronize do
26
+ running = @jobs.values.find { |state| state[:status] == "running" }
27
+ if running
28
+ return { error: "another job is already running (#{running[:job_id]}); " \
29
+ "poll job_status or cancel_job before starting a new one" }
30
+ end
31
+ id = format("job-%04d", @seq += 1)
32
+ @jobs[id] = {
33
+ job_id: id, status: "running", started_at: Time.now.utc.iso8601,
34
+ params: params, titles_done: 0, titles_total: nil, cancel: false
35
+ }
36
+ id
29
37
  end
30
38
 
31
- job_id = @mutex.synchronize { format("job-%04d", @seq += 1) }
32
- state = {
33
- job_id: job_id, status: "running", started_at: Time.now.utc.iso8601,
34
- params: params, titles_done: 0, titles_total: nil, cancel: false
35
- }
36
- @mutex.synchronize { @jobs[job_id] = state }
37
-
38
39
  thread = Thread.new do
39
- corpus = @factory.call
40
+ corpus = nil
40
41
  begin
42
+ corpus = @factory.call
41
43
  result = corpus.extract_corpus(
42
44
  **params,
43
45
  max_articles: nil,
@@ -52,7 +54,7 @@ module Wp2txt
52
54
  rescue StandardError => e
53
55
  update(job_id) { |s| s[:status] = "error"; s[:error] = "#{e.class}: #{e.message}"; s[:finished_at] = Time.now.utc.iso8601 }
54
56
  ensure
55
- corpus.close
57
+ corpus&.close
56
58
  end
57
59
  end
58
60
  thread.report_on_exception = false
@@ -111,6 +111,8 @@ module Wp2txt
111
111
 
112
112
  if page
113
113
  article = Article.new(page[:text], page[:title], !config[:marker])
114
+ article.page_id = page[:id]
115
+ article.revision_id = page[:revision_id]
114
116
  result = format_article(article, config)
115
117
  writer.write(result)
116
118
  extracted_count += 1
@@ -480,6 +482,8 @@ module Wp2txt
480
482
 
481
483
  pages.each do |page|
482
484
  article = Article.new(page[:text], page[:title], !config[:marker])
485
+ article.page_id = page[:id]
486
+ article.revision_id = page[:revision_id]
483
487
  result = format_article(article, config)
484
488
  writer.write(result)
485
489
  extracted_count += 1
@@ -493,6 +497,8 @@ module Wp2txt
493
497
 
494
498
  if page
495
499
  article = Article.new(page[:text], page[:title], !config[:marker])
500
+ article.page_id = page[:id]
501
+ article.revision_id = page[:revision_id]
496
502
  result = format_article(article, config)
497
503
  writer.write(result)
498
504
  extracted_count += 1
@@ -14,6 +14,24 @@ module Wp2txt
14
14
 
15
15
  # Format article based on configuration and output format
16
16
  def format_article(article, config)
17
+ with_page_ids(format_article_body(article, config), article)
18
+ end
19
+
20
+ # JSON records carry the dump's page and revision IDs right after the title,
21
+ # when the reader supplied them, so a record can be traced back to its source.
22
+ def with_page_ids(result, article)
23
+ return result unless result.is_a?(Hash) && (article.page_id || article.revision_id)
24
+
25
+ result.each_with_object({}) do |(key, value), out|
26
+ out[key] = value
27
+ next unless key == "title"
28
+
29
+ out["page_id"] = article.page_id
30
+ out["revision_id"] = article.revision_id
31
+ end
32
+ end
33
+
34
+ def format_article_body(article, config)
17
35
  # Store original title for magic word expansion in content
18
36
  original_title = article.title.dup
19
37
  article.title = format_wiki(article.title, config)
@@ -19,6 +19,12 @@ module Wp2txt
19
19
  # Queries ATTACH the Tier 1 metadata DB so category/section/redirect
20
20
  # filters compose with MATCH in plain SQL.
21
21
  class FtsIndex
22
+ class ShortQueryError < ArgumentError
23
+ def code
24
+ "query_too_short"
25
+ end
26
+ end
27
+
22
28
  SCHEMA_VERSION = 2
23
29
  CACHE_SUFFIX = "_fts.sqlite3"
24
30
 
@@ -206,6 +212,9 @@ module Wp2txt
206
212
  # @return [Hash] { total:, total_is_capped:, hits: [{page_id:, title:, heading:, ord:}] }
207
213
  def search(query, mode: "phrase", sections: nil, category: nil, depth: 0,
208
214
  limit: 20, offset: 0, count: "capped", count_cap: 1000)
215
+ if mode == "phrase" && tokenizer == "trigram" && query.length < 3
216
+ raise ShortQueryError, "trigram phrase searches require at least 3 Unicode characters"
217
+ end
209
218
  match_expr = mode == "query" ? query : phrase_query(query)
210
219
 
211
220
  conds = ["fts_sections MATCH ?", "p.namespace = 0", "p.redirect_to IS NULL"]
@@ -386,10 +395,10 @@ module Wp2txt
386
395
  batches = pairs.each_slice(STREAMS_PER_BATCH).to_a
387
396
  done = 0
388
397
 
389
- Parallel.map(
398
+ Parallel.each(
390
399
  batches,
391
400
  in_processes: @num_processes,
392
- finish: lambda { |_item, _idx, rows|
401
+ finish: lambda { |_item, _idx, rows|
393
402
  index.insert_batch(rows)
394
403
  done += 1
395
404
  progress&.call(done, batches.size)
@@ -423,7 +432,7 @@ module Wp2txt
423
432
  title = block[MetadataIndexBuilder::TITLE_REGEX, 1]
424
433
  return unless title && !title.empty?
425
434
 
426
- ns = (block[MetadataIndexBuilder::NS_REGEX, 1] || "0").to_i
435
+ ns = Wp2txt.namespace_id(block[MetadataIndexBuilder::NS_REGEX, 1])
427
436
  return unless ns.zero?
428
437
 
429
438
  page_id = block[MetadataIndexBuilder::ID_REGEX, 1]&.to_i
@@ -640,10 +640,10 @@ module Wp2txt
640
640
  batches = pairs.each_slice(STREAMS_PER_BATCH).to_a
641
641
  done = 0
642
642
 
643
- Parallel.map(
643
+ Parallel.each(
644
644
  batches,
645
645
  in_processes: @num_processes,
646
- finish: lambda { |_item, _idx, result|
646
+ finish: lambda { |_item, _idx, result|
647
647
  index.insert_batch(result)
648
648
  done += 1
649
649
  progress&.call(done, batches.size)
@@ -686,7 +686,7 @@ module Wp2txt
686
686
  return unless page_id
687
687
 
688
688
  title = unescape_xml(title)
689
- ns = (block[NS_REGEX, 1] || "0").to_i
689
+ ns = Wp2txt.namespace_id(block[NS_REGEX, 1])
690
690
  text = block[TEXT_REGEX, 1] || ""
691
691
  text = unescape_xml(text)
692
692
  # Strip HTML comments before scanning, matching what the Article parser
@@ -95,6 +95,11 @@ module Wp2txt
95
95
  @early_terminated == true
96
96
  end
97
97
 
98
+ # Where the stream holding the last found article ends. Scanning stops as
99
+ # soon as every target is found, so without this the reader cannot tell how
100
+ # far that stream extends and would read the dump to its end.
101
+ attr_reader :stream_end_offset
102
+
98
103
  def find_by_title(title)
99
104
  @entries_by_title[title]
100
105
  end
@@ -179,6 +184,14 @@ module Wp2txt
179
184
  end
180
185
  end
181
186
 
187
+ def next_stream_offset_after(io, offset)
188
+ io.each_line do |line|
189
+ next_offset = line.split(":", 2).first.to_i
190
+ return next_offset if next_offset > offset
191
+ end
192
+ nil
193
+ end
194
+
182
195
  def parse_index_stream(io)
183
196
  count = 0
184
197
  io.each_line do |line|
@@ -205,6 +218,7 @@ module Wp2txt
205
218
  @found_targets << title if @target_titles.include?(title)
206
219
  if @found_targets.size == @target_titles.size
207
220
  @early_terminated = true
221
+ @stream_end_offset = next_stream_offset_after(io, offset)
208
222
  print "\r Found all #{@target_titles.size} target articles" if @show_progress
209
223
  puts if @show_progress
210
224
  break
@@ -223,6 +237,10 @@ module Wp2txt
223
237
 
224
238
  # Reads articles from multistream bz2 files
225
239
  class MultistreamReader
240
+ # A single multistream bz2 stream holds about 100 pages; anything this large
241
+ # past the last known offset is several streams, not one.
242
+ MAX_TAIL_STREAM_BYTES = 64 * 1024 * 1024
243
+
226
244
  attr_reader :multistream_path, :index
227
245
 
228
246
  # Initialize reader with multistream file and index
@@ -357,7 +375,14 @@ module Wp2txt
357
375
  if next_offset
358
376
  compressed_data = f.read(next_offset - offset)
359
377
  else
360
- # Last stream - read to end
378
+ # Last stream of the dump: the rest of the file is that one stream.
379
+ # A large remainder means the end was never recorded, and reading it
380
+ # in one call fails outright on some platforms (EINVAL on macOS).
381
+ remaining = File.size(@multistream_path) - offset
382
+ if remaining > MAX_TAIL_STREAM_BYTES
383
+ raise Wp2txt::Error, "cannot locate the end of the stream at offset #{offset} " \
384
+ "(#{remaining} bytes to end of file); the index may be incomplete"
385
+ end
361
386
  compressed_data = f.read
362
387
  end
363
388
 
@@ -372,7 +397,9 @@ module Wp2txt
372
397
  # here would silently read gigabytes to EOF, so fail fast instead
373
398
  raise "Stream offset #{current_offset} not found in index (#{offsets.size} streams known)" unless idx
374
399
 
375
- offsets[idx + 1]
400
+ return offsets[idx + 1] if idx + 1 < offsets.size
401
+
402
+ @index.respond_to?(:stream_end_offset) ? @index.stream_end_offset : nil
376
403
  end
377
404
 
378
405
  def decompress_bz2(data)
@@ -394,6 +421,7 @@ module Wp2txt
394
421
  return {
395
422
  title: page_title,
396
423
  id: page_node.at_xpath("id")&.text&.to_i,
424
+ revision_id: page_node.at_xpath("revision/id")&.text&.to_i,
397
425
  text: page_node.at_xpath(".//text")&.text || ""
398
426
  }
399
427
  end
@@ -409,6 +437,7 @@ module Wp2txt
409
437
  page = {
410
438
  title: page_node.at_xpath("title")&.text,
411
439
  id: page_node.at_xpath("id")&.text&.to_i,
440
+ revision_id: page_node.at_xpath("revision/id")&.text&.to_i,
412
441
  text: page_node.at_xpath(".//text")&.text || ""
413
442
  }
414
443
  yield page if page[:title]
@@ -1,27 +1,87 @@
1
1
  # frozen_string_literal: true
2
2
 
3
+ require "tempfile"
4
+ require "tmpdir"
5
+
3
6
  module Wp2txt
4
- # Server-side output path confinement shared by the file-writing tools
5
- # (extract_corpus / start_extract_job / query_sql with output_path).
6
- # Paths must resolve under the server's output directory — an agent mixing
7
- # up paths must not be able to clobber arbitrary user files — and existing
8
- # files are not replaced unless overwrite is set.
7
+ # Output directories are dedicated, trusted directories. Reject links below
8
+ # the trusted platform temp root (whose ancestors may be OS aliases on macOS).
9
+ # This does not defend against an attacker replacing parent directories.
9
10
  module OutputPath
10
11
  module_function
11
12
 
12
- # @return [String] the confined, absolute output path
13
- # @raise [ArgumentError] when the path escapes base_dir or the file exists
13
+ def reject_symlinks!(path)
14
+ current = File.expand_path(path)
15
+ temp_roots = [File.expand_path(Dir.tmpdir), File.realpath(Dir.tmpdir)]
16
+ loop do
17
+ raise ArgumentError, "symbolic links are not allowed in output paths: #{current}" if File.symlink?(current)
18
+ parent = File.dirname(current)
19
+ break if parent == current || temp_roots.include?(current)
20
+
21
+ current = parent
22
+ end
23
+ end
24
+
25
+ def validate_pair!(path, overwrite: false)
26
+ [path, "#{path}.meta.json"].each do |destination|
27
+ reject_symlinks!(destination)
28
+ if File.exist?(destination) && !overwrite
29
+ raise ArgumentError, "output file already exists: #{destination} (pass overwrite: true to replace it)"
30
+ end
31
+ if File.exist?(destination) && !File.file?(destination)
32
+ raise ArgumentError, "output destination must be a file: #{destination}"
33
+ end
34
+ end
35
+ end
36
+
37
+ # Return the confined absolute path; the writer repeats validation and
38
+ # reserves both destinations with EXCL to close the check/create race.
14
39
  def confine(output_path, base_dir, overwrite: false)
15
40
  path = File.expand_path(output_path, base_dir)
16
41
  base = File.expand_path(base_dir)
17
- unless path == base || path.start_with?(base + File::SEPARATOR)
42
+ unless path.start_with?(base + File::SEPARATOR)
18
43
  raise ArgumentError, "output_path must stay within the server output directory (#{base})"
19
44
  end
20
- if File.exist?(path) && !overwrite
21
- raise ArgumentError, "output file already exists: #{path} (pass overwrite: true to replace it)"
22
- end
23
-
45
+ validate_pair!(path, overwrite: overwrite)
24
46
  path
25
47
  end
48
+
49
+ # Reserve destinations, stage both files, then publish with rename. EXCL
50
+ # reservations are empty until publication; the pair is not a transaction.
51
+ # Failure removes only reservations owned by this call, never prior output.
52
+ def write_pair(path, overwrite: false)
53
+ validate_pair!(path, overwrite: overwrite)
54
+ destinations = [path, "#{path}.meta.json"]
55
+ reservations = {}
56
+ temporary = []
57
+ begin
58
+ unless overwrite
59
+ destinations.each do |destination|
60
+ File.open(destination, File::WRONLY | File::CREAT | File::EXCL, 0o600) do |file|
61
+ reservations[destination] = file.stat
62
+ end
63
+ end
64
+ end
65
+ destinations.each do |destination|
66
+ temporary << Tempfile.create([".wp2txt-", ".partial"], File.dirname(destination))
67
+ temporary.last.close
68
+ end
69
+ result = yield(*temporary.map(&:path))
70
+ destinations.each_with_index do |destination, index|
71
+ reject_symlinks!(destination)
72
+ File.rename(temporary[index].path, destination)
73
+ reservations.delete(destination)
74
+ end
75
+ result
76
+ rescue Errno::EEXIST => e
77
+ raise ArgumentError, "output file already exists: #{e.message} (pass overwrite: true to replace it)"
78
+ ensure
79
+ temporary.each { |file| File.unlink(file.path) if File.exist?(file.path) }
80
+ reservations.each do |destination, stat|
81
+ current = File.lstat(destination) rescue nil
82
+ File.unlink(destination) if current && current.dev == stat.dev && current.ino == stat.ino
83
+ end
84
+ end
85
+ end
26
86
  end
27
87
  end
@@ -100,6 +100,14 @@ module Wp2txt
100
100
  raise Wp2txt::FileIOError, "Write failed: #{e.message}"
101
101
  end
102
102
 
103
+ # Push buffered output to the file. Call before forking: a child process
104
+ # inherits the buffer and writes it out again when it exits.
105
+ def flush
106
+ @mutex.synchronize do
107
+ @current_file.flush if @current_file && !@current_file.closed?
108
+ end
109
+ end
110
+
103
111
  # Close current file and finalize
104
112
  def close
105
113
  @mutex.synchronize do