wp2txt 2.1.2 → 2.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,338 @@
1
+ # frozen_string_literal: true
2
+
3
+ require "spec_helper"
4
+ require "tmpdir"
5
+ require "json"
6
+ require_relative "support/multistream_fixture"
7
+ require_relative "support/meta_db_fixture"
8
+ require_relative "../lib/wp2txt/corpus"
9
+ require_relative "../lib/wp2txt/corpus_jobs"
10
+ require_relative "../lib/wp2txt/output_path"
11
+
12
+ # extract_corpus titles: and query_sql output_path: (design doc 04)
13
+ RSpec.describe "titles extraction and SQL file output" do
14
+ include MultistreamFixture
15
+ include MetaDbFixture
16
+
17
+ around do |example|
18
+ Dir.mktmpdir do |dir|
19
+ @dir = dir
20
+ @multistream_path, @index_path = create_fixture(dir)
21
+
22
+ ms_index = Wp2txt::MultistreamIndex.new(@index_path, use_cache: false, show_progress: false)
23
+ @db_path = Wp2txt::MetadataIndex.path_for(@multistream_path, cache_dir: dir)
24
+ Wp2txt::MetadataIndexBuilder.new(
25
+ @multistream_path, ms_index.stream_offsets,
26
+ db_path: @db_path, num_processes: 0
27
+ ).build
28
+
29
+ @corpus = Wp2txt::Corpus.for_input(@multistream_path, cache_dir: dir)
30
+ example.run
31
+ @corpus.close
32
+ end
33
+ end
34
+
35
+ def read_jsonl(path)
36
+ File.readlines(path).map { |l| JSON.parse(l) }
37
+ end
38
+
39
+ describe "extract_corpus titles:" do
40
+ it "extracts an explicit set with normalization, dedup, and input order" do
41
+ out = File.join(@dir, "t.jsonl")
42
+ result = @corpus.extract_corpus(
43
+ output_path: out, content: "summary",
44
+ titles: ["Film B", "Film_A", "film A", "Film B"], num_processes: 0
45
+ )
46
+
47
+ expect(result[:total_matching]).to eq(2) # normalized + deduplicated
48
+ expect(result[:articles_extracted]).to eq(2)
49
+ expect(read_jsonl(out).map { |r| r["title"] }).to eq(["Film B", "Film A"])
50
+ expect(result[:not_found][:count]).to eq(0)
51
+ end
52
+
53
+ it "resolves one redirect hop and extracts under the resolved title" do
54
+ out = File.join(@dir, "t.jsonl")
55
+ result = @corpus.extract_corpus(
56
+ output_path: out, content: "summary", titles: ["Old Film"], num_processes: 0
57
+ )
58
+
59
+ expect(result[:articles_extracted]).to eq(1)
60
+ expect(read_jsonl(out).first["title"]).to eq("Film A")
61
+ end
62
+
63
+ it "reports missing titles in not_found (count + sample)" do
64
+ out = File.join(@dir, "t.jsonl")
65
+ result = @corpus.extract_corpus(
66
+ output_path: out, content: "summary",
67
+ titles: ["Film A", "Ghost One", "Ghost Two"], num_processes: 0
68
+ )
69
+
70
+ expect(result[:articles_extracted]).to eq(1)
71
+ expect(result[:not_found]).to eq({ count: 2, sample: ["Ghost One", "Ghost Two"] })
72
+ meta = JSON.parse(File.read(result[:meta_path]))
73
+ expect(meta["not_found"]["count"]).to eq(2)
74
+ end
75
+
76
+ it "counts a redirect whose target does not exist as not found" do
77
+ db = SQLite3::Database.new(@db_path)
78
+ db.execute("INSERT INTO pages (page_id, title, namespace, redirect_to, text_length) VALUES (99, 'Broken Redirect', 0, 'Nowhere', 0)")
79
+ db.close
80
+
81
+ out = File.join(@dir, "t.jsonl")
82
+ result = @corpus.extract_corpus(
83
+ output_path: out, content: "summary", titles: ["Broken Redirect"], num_processes: 0
84
+ )
85
+ expect(result[:not_found]).to eq({ count: 1, sample: ["Broken Redirect"] })
86
+ expect(result[:articles_extracted]).to eq(0)
87
+ end
88
+
89
+ it "does not flag truncation for missing titles when the cap is not reached" do
90
+ out = File.join(@dir, "t.jsonl")
91
+ result = @corpus.extract_corpus(
92
+ output_path: out, content: "summary",
93
+ titles: ["Film A", "Ghost One", "Ghost Two"], num_processes: 0
94
+ )
95
+
96
+ expect(result[:not_found][:count]).to eq(2)
97
+ expect(result[:articles_extracted]).to eq(1)
98
+ expect(result[:truncated]).to be false
99
+ end
100
+
101
+ it "deduplicates titles that resolve to the same article" do
102
+ db = SQLite3::Database.new(@db_path)
103
+ db.execute("INSERT INTO pages (page_id, title, namespace, redirect_to, text_length) VALUES (98, 'Ancient Film', 0, 'Film A', 0)")
104
+ db.close
105
+
106
+ # (a) two aliases redirecting to the same target
107
+ out1 = File.join(@dir, "t1.jsonl")
108
+ result1 = @corpus.extract_corpus(
109
+ output_path: out1, content: "summary",
110
+ titles: ["Old Film", "Ancient Film"], num_processes: 0
111
+ )
112
+ expect(result1[:articles_extracted]).to eq(1)
113
+ expect(read_jsonl(out1).map { |r| r["title"] }).to eq(["Film A"])
114
+
115
+ # (b) a direct title plus its redirect alias
116
+ out2 = File.join(@dir, "t2.jsonl")
117
+ result2 = @corpus.extract_corpus(
118
+ output_path: out2, content: "summary",
119
+ titles: ["Film A", "Old Film"], num_processes: 0
120
+ )
121
+ expect(result2[:articles_extracted]).to eq(1)
122
+ expect(read_jsonl(out2).map { |r| r["title"] }).to eq(["Film A"])
123
+ end
124
+
125
+ it "rejects combination with filter arguments (set is defined twice)" do
126
+ out = File.join(@dir, "t.jsonl")
127
+ [
128
+ { category: "Japanese films" },
129
+ { categories: ["Japanese films"] },
130
+ { category_match: "films" },
131
+ { title_match: "Film" }
132
+ ].each do |filter|
133
+ expect do
134
+ @corpus.extract_corpus(output_path: out, content: "summary", titles: ["Film A"], **filter)
135
+ end.to raise_error(ArgumentError, /query_sql/), "expected #{filter.keys.first} to conflict"
136
+ end
137
+ end
138
+
139
+ it "rejects more than 10,000 titles" do
140
+ expect do
141
+ @corpus.extract_corpus(output_path: File.join(@dir, "t.jsonl"), content: "summary",
142
+ titles: Array.new(10_001) { |i| "T#{i}" })
143
+ end.to raise_error(ArgumentError, /10_000|10000/)
144
+ end
145
+
146
+ it "interacts with limit and max_articles like the filter path" do
147
+ titles = ["Film A", "Film B", "Person X"]
148
+ limited = @corpus.extract_corpus(
149
+ output_path: File.join(@dir, "t1.jsonl"), content: "summary",
150
+ titles: titles, limit: 2, num_processes: 0
151
+ )
152
+ expect(limited[:articles_extracted]).to eq(2)
153
+ expect(limited[:truncated]).to be true
154
+ expect(limited[:total_matching]).to eq(3)
155
+
156
+ capped = @corpus.extract_corpus(
157
+ output_path: File.join(@dir, "t2.jsonl"), content: "summary",
158
+ titles: titles, max_articles: 1, num_processes: 0
159
+ )
160
+ expect(capped[:articles_extracted]).to eq(1)
161
+ expect(capped[:truncated]).to be true
162
+ end
163
+
164
+ it "enumerates titles in .meta.json when 100 or fewer, and the sha is order-independent" do
165
+ result1 = @corpus.extract_corpus(
166
+ output_path: File.join(@dir, "t1.jsonl"), content: "summary",
167
+ titles: ["Film A", "Film B"], num_processes: 0
168
+ )
169
+ result2 = @corpus.extract_corpus(
170
+ output_path: File.join(@dir, "t2.jsonl"), content: "summary",
171
+ titles: ["Film B", "Film_A"], num_processes: 0
172
+ )
173
+ meta1 = JSON.parse(File.read(result1[:meta_path]))
174
+ meta2 = JSON.parse(File.read(result2[:meta_path]))
175
+
176
+ expect(meta1["query"]["titles"]).to eq(["Film A", "Film B"])
177
+ expect(meta1["query"]["titles_count"]).to eq(2)
178
+ expect(meta1["query"]["titles_sha256"]).to eq(meta2["query"]["titles_sha256"])
179
+ end
180
+
181
+ it "records only count + sha256 in .meta.json when more than 100 titles" do
182
+ titles = ["Film A"] + Array.new(100) { |i| "Ghost #{i}" }
183
+ result = @corpus.extract_corpus(
184
+ output_path: File.join(@dir, "t.jsonl"), content: "summary",
185
+ titles: titles, num_processes: 0
186
+ )
187
+ meta = JSON.parse(File.read(result[:meta_path]))
188
+
189
+ expect(meta["query"]["titles_count"]).to eq(101)
190
+ expect(meta["query"]["titles_sha256"]).to match(/\A[0-9a-f]{64}\z/)
191
+ expect(meta["query"]).not_to have_key("titles")
192
+ end
193
+
194
+ it "passes titles through start_extract_job" do
195
+ out = File.join(@dir, "job.jsonl")
196
+ manager = Wp2txt::CorpusJobManager.new(
197
+ -> { Wp2txt::Corpus.for_input(@multistream_path, cache_dir: @dir) }
198
+ )
199
+ start = manager.start_extract(
200
+ output_path: out, content: "summary",
201
+ titles: ["Film A", "Ghost"], num_processes: 0
202
+ )
203
+
204
+ deadline = Time.now + 30
205
+ status = nil
206
+ loop do
207
+ status = manager.status(start[:job_id])
208
+ break unless status[:status] == "running"
209
+ raise "job did not finish in time" if Time.now > deadline
210
+
211
+ sleep 0.1
212
+ end
213
+
214
+ expect(status[:status]).to eq("completed")
215
+ expect(status[:result][:articles_extracted]).to eq(1)
216
+ expect(status[:result][:not_found][:count]).to eq(1)
217
+ end
218
+ end
219
+
220
+ describe "query_sql output_path:" do
221
+ it "writes all rows as JSONL with NULLs and deduplicated column names" do
222
+ out = File.join(@dir, "q.jsonl")
223
+ result = @corpus.query_sql(
224
+ "SELECT title, NULL AS note, page_id AS x, page_id AS x FROM pages " \
225
+ "WHERE namespace = 0 AND redirect_to IS NULL ORDER BY page_id",
226
+ output_path: out
227
+ )
228
+
229
+ expect(result[:columns]).to eq(["title", "note", "x", "x_2"])
230
+ expect(result[:row_count]).to eq(3)
231
+ expect(result[:sample].size).to eq(3)
232
+ expect(result).not_to have_key(:rows)
233
+
234
+ records = read_jsonl(out)
235
+ expect(records.size).to eq(3)
236
+ expect(records.first).to eq({ "title" => "Film A", "note" => nil, "x" => 1, "x_2" => 1 })
237
+ expect(File.exist?("#{out}.partial")).to be false
238
+ end
239
+
240
+ it "ignores limit in file-output mode" do
241
+ out = File.join(@dir, "q.jsonl")
242
+ result = @corpus.query_sql("SELECT title FROM pages", output_path: out, limit: 1)
243
+ expect(result[:row_count]).to eq(8)
244
+ expect(read_jsonl(out).size).to eq(8)
245
+ end
246
+
247
+ it "refuses an existing output file unless overwrite is set" do
248
+ out = File.join(@dir, "q.jsonl")
249
+ File.write(out, "old")
250
+ expect { @corpus.query_sql("SELECT 1", output_path: out) }
251
+ .to raise_error(ArgumentError, /already exists/)
252
+ expect(File.read(out)).to eq("old")
253
+
254
+ @corpus.query_sql("SELECT 1 AS one", output_path: out, overwrite: true)
255
+ expect(read_jsonl(out)).to eq([{ "one" => 1 }])
256
+ end
257
+
258
+ it "removes .partial and leaves no output on SQL error" do
259
+ out = File.join(@dir, "q.jsonl")
260
+ expect { @corpus.query_sql("SELECT * FROM no_such_table", output_path: out) }
261
+ .to raise_error(ArgumentError, /SQL error/)
262
+ expect(File.exist?(out)).to be false
263
+ expect(File.exist?("#{out}.partial")).to be false
264
+ end
265
+
266
+ it "removes .partial and leaves no output on timeout kill" do
267
+ out = File.join(@dir, "q.jsonl")
268
+ runaway = "WITH RECURSIVE c(x) AS (SELECT 1 UNION ALL SELECT x + 1 FROM c) SELECT x FROM c"
269
+ expect { @corpus.query_sql(runaway, output_path: out, timeout: 1) }
270
+ .to raise_error(ArgumentError, /time limit/)
271
+ expect(File.exist?(out)).to be false
272
+ expect(File.exist?("#{out}.partial")).to be false
273
+ end
274
+
275
+ it "truncates at SQL_FILE_ROW_LIMIT" do
276
+ stub_const("Wp2txt::Corpus::SQL_FILE_ROW_LIMIT", 3)
277
+ out = File.join(@dir, "q.jsonl")
278
+ result = @corpus.query_sql("SELECT title FROM pages", output_path: out)
279
+
280
+ expect(result[:row_count]).to eq(3)
281
+ expect(result[:truncated]).to be true
282
+ expect(read_jsonl(out).size).to eq(3)
283
+ end
284
+
285
+ it "clips cells over 64KB and counts them" do
286
+ out = File.join(@dir, "q.jsonl")
287
+ result = @corpus.query_sql(
288
+ "SELECT printf('%.70000d', 0) AS big",
289
+ output_path: out
290
+ )
291
+
292
+ expect(result[:cells_clipped]).to eq(1)
293
+ value = read_jsonl(out).first["big"]
294
+ expect(value.length).to eq(Wp2txt::Corpus::SQL_FILE_CELL_LIMIT + 1)
295
+ end
296
+
297
+ it "writes a .meta.json sidecar with SQL, attach provenance, and counts" do
298
+ create_meta_db(@dir, lang: "en", date: "20260101",
299
+ pages: [[101, "Film A", 0, nil, 10]])
300
+ out = File.join(@dir, "q.jsonl")
301
+ result = @corpus.query_sql(
302
+ "SELECT COUNT(*) AS n FROM en_meta.pages", attach: ["en"], output_path: out
303
+ )
304
+
305
+ expect(result[:attached].first[:lang]).to eq("en")
306
+ meta = JSON.parse(File.read("#{out}.meta.json"))
307
+ expect(meta["tool"]).to eq("query_sql")
308
+ expect(meta["dump"]).to eq("testwiki-20260101")
309
+ expect(meta["sql"]).to include("en_meta.pages")
310
+ expect(meta["attached"]).to eq([
311
+ { "lang" => "en", "dump_name" => "enwiki-20260101", "built_with" => Wp2txt::VERSION }
312
+ ])
313
+ expect(meta["row_count"]).to eq(1)
314
+ expect(meta["truncated"]).to be false
315
+ expect(meta["cells_clipped"]).to eq(0)
316
+ expect(meta["generated_at"]).to match(/\A\d{4}-\d{2}-\d{2}T/)
317
+ expect(meta["wp2txt_version"]).to eq(Wp2txt::VERSION)
318
+ end
319
+ end
320
+
321
+ describe Wp2txt::OutputPath do
322
+ it "confines paths under the server output directory" do
323
+ expect(described_class.confine("sub/out.jsonl", @dir)).to eq(File.join(@dir, "sub/out.jsonl"))
324
+ expect { described_class.confine("../escape.jsonl", @dir) }
325
+ .to raise_error(ArgumentError, /output directory/)
326
+ expect { described_class.confine("/etc/passwd", @dir) }
327
+ .to raise_error(ArgumentError, /output directory/)
328
+ end
329
+
330
+ it "refuses existing files unless overwrite is set" do
331
+ existing = File.join(@dir, "exists.jsonl")
332
+ File.write(existing, "x")
333
+ expect { described_class.confine("exists.jsonl", @dir) }
334
+ .to raise_error(ArgumentError, /already exists/)
335
+ expect(described_class.confine("exists.jsonl", @dir, overwrite: true)).to eq(existing)
336
+ end
337
+ end
338
+ end
metadata CHANGED
@@ -1,7 +1,7 @@
1
1
  --- !ruby/object:Gem::Specification
2
2
  name: wp2txt
3
3
  version: !ruby/object:Gem::Version
4
- version: 2.1.2
4
+ version: 2.3.0
5
5
  platform: ruby
6
6
  authors:
7
7
  - Yoichiro Hasebe
@@ -197,6 +197,7 @@ email:
197
197
  - yohasebe@gmail.com
198
198
  executables:
199
199
  - wp2txt
200
+ - wp2txt-mcp
200
201
  extensions: []
201
202
  extra_rdoc_files: []
202
203
  files:
@@ -215,6 +216,8 @@ files:
215
216
  - README_ja.md
216
217
  - Rakefile
217
218
  - bin/wp2txt
219
+ - bin/wp2txt-mcp
220
+ - docs/RESEARCH.md
218
221
  - image/wp2txt-logo.svg
219
222
  - image/wp2txt.svg
220
223
  - lib/wp2txt.rb
@@ -225,6 +228,8 @@ files:
225
228
  - lib/wp2txt/cli_ui.rb
226
229
  - lib/wp2txt/config.rb
227
230
  - lib/wp2txt/constants.rb
231
+ - lib/wp2txt/corpus.rb
232
+ - lib/wp2txt/corpus_jobs.rb
228
233
  - lib/wp2txt/data/html_entities.json
229
234
  - lib/wp2txt/data/language_metadata.json
230
235
  - lib/wp2txt/data/language_tiers.json
@@ -234,11 +239,16 @@ files:
234
239
  - lib/wp2txt/extractor.rb
235
240
  - lib/wp2txt/file_utils.rb
236
241
  - lib/wp2txt/formatter.rb
242
+ - lib/wp2txt/fts_index.rb
237
243
  - lib/wp2txt/global_data_cache.rb
238
244
  - lib/wp2txt/index_cache.rb
245
+ - lib/wp2txt/index_commands.rb
246
+ - lib/wp2txt/langlinks_importer.rb
239
247
  - lib/wp2txt/magic_words.rb
240
248
  - lib/wp2txt/memory_monitor.rb
249
+ - lib/wp2txt/metadata_index.rb
241
250
  - lib/wp2txt/multistream.rb
251
+ - lib/wp2txt/output_path.rb
242
252
  - lib/wp2txt/output_writer.rb
243
253
  - lib/wp2txt/parser_functions.rb
244
254
  - lib/wp2txt/ractor_worker.rb
@@ -265,15 +275,20 @@ files:
265
275
  - spec/cli_spec.rb
266
276
  - spec/config_spec.rb
267
277
  - spec/constants_spec.rb
278
+ - spec/corpus_spec.rb
268
279
  - spec/file_utils_spec.rb
269
280
  - spec/fixtures/samples.rb
270
281
  - spec/formatter_sections_spec.rb
282
+ - spec/fts_index_spec.rb
271
283
  - spec/global_data_cache_spec.rb
272
284
  - spec/index_cache_spec.rb
273
285
  - spec/integration_spec.rb
286
+ - spec/langlinks_importer_spec.rb
274
287
  - spec/magic_words_spec.rb
275
288
  - spec/markers_spec.rb
276
289
  - spec/memory_monitor_spec.rb
290
+ - spec/metadata_index_spec.rb
291
+ - spec/multi_dump_attach_spec.rb
277
292
  - spec/multistream_spec.rb
278
293
  - spec/output_writer_spec.rb
279
294
  - spec/parser_functions_spec.rb
@@ -282,10 +297,13 @@ files:
282
297
  - spec/section_extractor_spec.rb
283
298
  - spec/spec_helper.rb
284
299
  - spec/stream_processor_spec.rb
300
+ - spec/support/meta_db_fixture.rb
301
+ - spec/support/multistream_fixture.rb
285
302
  - spec/template_data_spec.rb
286
303
  - spec/template_expander_spec.rb
287
304
  - spec/template_processing_spec.rb
288
305
  - spec/text_processing_spec.rb
306
+ - spec/titles_output_path_spec.rb
289
307
  - spec/utils_spec.rb
290
308
  - spec/wp2txt_spec.rb
291
309
  - wp2txt.gemspec
@@ -322,15 +340,20 @@ test_files:
322
340
  - spec/cli_spec.rb
323
341
  - spec/config_spec.rb
324
342
  - spec/constants_spec.rb
343
+ - spec/corpus_spec.rb
325
344
  - spec/file_utils_spec.rb
326
345
  - spec/fixtures/samples.rb
327
346
  - spec/formatter_sections_spec.rb
347
+ - spec/fts_index_spec.rb
328
348
  - spec/global_data_cache_spec.rb
329
349
  - spec/index_cache_spec.rb
330
350
  - spec/integration_spec.rb
351
+ - spec/langlinks_importer_spec.rb
331
352
  - spec/magic_words_spec.rb
332
353
  - spec/markers_spec.rb
333
354
  - spec/memory_monitor_spec.rb
355
+ - spec/metadata_index_spec.rb
356
+ - spec/multi_dump_attach_spec.rb
334
357
  - spec/multistream_spec.rb
335
358
  - spec/output_writer_spec.rb
336
359
  - spec/parser_functions_spec.rb
@@ -339,9 +362,12 @@ test_files:
339
362
  - spec/section_extractor_spec.rb
340
363
  - spec/spec_helper.rb
341
364
  - spec/stream_processor_spec.rb
365
+ - spec/support/meta_db_fixture.rb
366
+ - spec/support/multistream_fixture.rb
342
367
  - spec/template_data_spec.rb
343
368
  - spec/template_expander_spec.rb
344
369
  - spec/template_processing_spec.rb
345
370
  - spec/text_processing_spec.rb
371
+ - spec/titles_output_path_spec.rb
346
372
  - spec/utils_spec.rb
347
373
  - spec/wp2txt_spec.rb