wp2txt 2.3.3 → 2.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/.dockerignore +0 -4
- data/.gitignore +3 -5
- data/CHANGELOG.md +18 -0
- data/DEVELOPMENT.md +1 -1
- data/DEVELOPMENT_ja.md +1 -1
- data/README.md +44 -2
- data/README_ja.md +35 -2
- data/Rakefile +10 -21
- data/bin/wp2txt +79 -17
- data/bin/wp2txt-mcp +1 -1
- data/docs/INDEXES.md +61 -1
- data/lib/wp2txt/article.rb +1 -1
- data/lib/wp2txt/cli.rb +42 -0
- data/lib/wp2txt/constants.rb +24 -0
- data/lib/wp2txt/corpus.rb +11 -2
- data/lib/wp2txt/data/template_aliases.json +1 -1
- data/lib/wp2txt/extractor.rb +10 -1
- data/lib/wp2txt/formatter.rb +20 -0
- data/lib/wp2txt/index_commands.rb +89 -1
- data/lib/wp2txt/langlinks_importer.rb +19 -35
- data/lib/wp2txt/lead_terms.rb +228 -0
- data/lib/wp2txt/link_counter.rb +171 -0
- data/lib/wp2txt/metadata_index.rb +79 -3
- data/lib/wp2txt/multistream.rb +52 -2
- data/lib/wp2txt/output_writer.rb +8 -0
- data/lib/wp2txt/page_props_importer.rb +170 -0
- data/lib/wp2txt/sql_dump_reader.rb +57 -0
- data/lib/wp2txt/stream_processor.rb +42 -15
- data/lib/wp2txt/template_expander.rb +19 -0
- data/lib/wp2txt/utils.rb +5 -3
- data/lib/wp2txt/version.rb +1 -1
- data/lib/wp2txt/wikitext_regions.rb +66 -0
- data/lib/wp2txt.rb +7 -5
- data/spec/docs_sync_spec.rb +3 -3
- data/spec/langlinks_importer_spec.rb +27 -0
- data/spec/lead_terms_edge_cases_spec.rb +194 -0
- data/spec/lead_terms_links_qids_spec.rb +204 -0
- data/spec/output_integrity_spec.rb +146 -0
- data/spec/p1_correctness_spec.rb +32 -14
- data/spec/page_properties_spec.rb +161 -0
- data/spec/region_semantics_spec.rb +71 -0
- data/spec/template_passthrough_spec.rb +43 -0
- data/spec/titles_output_path_spec.rb +12 -0
- metadata +18 -1
|
@@ -0,0 +1,43 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "spec_helper"
|
|
4
|
+
require "wp2txt"
|
|
5
|
+
require "wp2txt/formatter"
|
|
6
|
+
|
|
7
|
+
# Templates the cleaning stage renders itself must survive template expansion;
|
|
8
|
+
# deleting them earlier silently dropped readings and link text from articles.
|
|
9
|
+
RSpec.describe "templates rendered by the cleaning stage" do
|
|
10
|
+
include Wp2txt
|
|
11
|
+
include Wp2txt::Formatter
|
|
12
|
+
|
|
13
|
+
def summary_of(wikitext)
|
|
14
|
+
article = Wp2txt::Article.new("#{wikitext}\n\n== 節 ==\n本文\n", "T", true)
|
|
15
|
+
format_article(article, { format: :json, summary_only: true, expand_templates: true,
|
|
16
|
+
markers: [:all] })["text"].strip
|
|
17
|
+
end
|
|
18
|
+
|
|
19
|
+
{
|
|
20
|
+
"'''{{読み仮名|言語|げんご}}'''は記号体系。" => "言語(げんご)は記号体系。",
|
|
21
|
+
"{{読み仮名_ruby不使用|東京|とうきょう}}都" => "東京(とうきょう)都",
|
|
22
|
+
"作曲は{{仮リンク|ジョン・ドウ|en|John Doe}}が担当した。" => "作曲はジョン・ドウが担当した。",
|
|
23
|
+
"{{ruby|漢字|かんじ}}を読む。" => "漢字(かんじ)を読む。",
|
|
24
|
+
"作曲は{{仮リンク|{{lang|en|John Doe}}|en|John Doe}}。" => "作曲はJohn Doe。"
|
|
25
|
+
}.each do |source, expected|
|
|
26
|
+
it "renders #{source.inspect}" do
|
|
27
|
+
expect(summary_of(source)).to eq(expected)
|
|
28
|
+
end
|
|
29
|
+
end
|
|
30
|
+
|
|
31
|
+
it "keeps an image caption that contains a link template" do
|
|
32
|
+
source = "[[ファイル:Stele.jpg|thumb|200px|[[ナラム・シン]]。画像は{{仮リンク|戦勝記念碑|en|Victory Stele}}。]]\n'''紀元前23世紀'''は世紀。"
|
|
33
|
+
expect(summary_of(source)).to include("ナラム・シン。画像は戦勝記念碑。")
|
|
34
|
+
end
|
|
35
|
+
|
|
36
|
+
it "still removes templates nothing downstream knows" do
|
|
37
|
+
expect(summary_of("前{{Otheruses|x|y}}後")).to eq("前後")
|
|
38
|
+
end
|
|
39
|
+
|
|
40
|
+
it "adds no empty brackets for an empty reading" do
|
|
41
|
+
expect(summary_of("{{読み仮名|語|}}は")).to eq("語は")
|
|
42
|
+
end
|
|
43
|
+
end
|
|
@@ -36,6 +36,18 @@ RSpec.describe "titles extraction and SQL file output" do
|
|
|
36
36
|
File.readlines(path).map { |l| JSON.parse(l) }
|
|
37
37
|
end
|
|
38
38
|
|
|
39
|
+
describe "extract_corpus across several batches with worker processes" do
|
|
40
|
+
it "writes each record once (workers must not re-flush the output buffer)" do
|
|
41
|
+
stub_const("Wp2txt::Corpus::EXTRACT_BATCH_SIZE", 1)
|
|
42
|
+
out = File.join(@dir, "batched.jsonl")
|
|
43
|
+
titles = ["Film A", "Film B", "Person X"]
|
|
44
|
+
result = @corpus.extract_corpus(output_path: out, content: "full", titles: titles, num_processes: 2)
|
|
45
|
+
records = read_jsonl(out)
|
|
46
|
+
expect(records.map { |r| r["title"] }.tally).to eq(titles.to_h { |t| [t, 1] })
|
|
47
|
+
expect(records.size).to eq(result[:records_written])
|
|
48
|
+
end
|
|
49
|
+
end
|
|
50
|
+
|
|
39
51
|
describe "extract_corpus titles:" do
|
|
40
52
|
it "extracts an explicit set with normalization, dedup, and input order" do
|
|
41
53
|
out = File.join(@dir, "t.jsonl")
|
metadata
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
--- !ruby/object:Gem::Specification
|
|
2
2
|
name: wp2txt
|
|
3
3
|
version: !ruby/object:Gem::Version
|
|
4
|
-
version: 2.
|
|
4
|
+
version: 2.4.0
|
|
5
5
|
platform: ruby
|
|
6
6
|
authors:
|
|
7
7
|
- Yoichiro Hasebe
|
|
@@ -243,21 +243,26 @@ files:
|
|
|
243
243
|
- lib/wp2txt/index_cache.rb
|
|
244
244
|
- lib/wp2txt/index_commands.rb
|
|
245
245
|
- lib/wp2txt/langlinks_importer.rb
|
|
246
|
+
- lib/wp2txt/lead_terms.rb
|
|
247
|
+
- lib/wp2txt/link_counter.rb
|
|
246
248
|
- lib/wp2txt/magic_words.rb
|
|
247
249
|
- lib/wp2txt/memory_monitor.rb
|
|
248
250
|
- lib/wp2txt/metadata_index.rb
|
|
249
251
|
- lib/wp2txt/multistream.rb
|
|
250
252
|
- lib/wp2txt/output_path.rb
|
|
251
253
|
- lib/wp2txt/output_writer.rb
|
|
254
|
+
- lib/wp2txt/page_props_importer.rb
|
|
252
255
|
- lib/wp2txt/parser_functions.rb
|
|
253
256
|
- lib/wp2txt/ractor_worker.rb
|
|
254
257
|
- lib/wp2txt/regex.rb
|
|
255
258
|
- lib/wp2txt/section_extractor.rb
|
|
259
|
+
- lib/wp2txt/sql_dump_reader.rb
|
|
256
260
|
- lib/wp2txt/stream_processor.rb
|
|
257
261
|
- lib/wp2txt/template_expander.rb
|
|
258
262
|
- lib/wp2txt/text_processing.rb
|
|
259
263
|
- lib/wp2txt/utils.rb
|
|
260
264
|
- lib/wp2txt/version.rb
|
|
265
|
+
- lib/wp2txt/wikitext_regions.rb
|
|
261
266
|
- spec/article_spec.rb
|
|
262
267
|
- spec/auto_download_spec.rb
|
|
263
268
|
- spec/bz2_validator_spec.rb
|
|
@@ -278,17 +283,22 @@ files:
|
|
|
278
283
|
- spec/index_cache_spec.rb
|
|
279
284
|
- spec/integration_spec.rb
|
|
280
285
|
- spec/langlinks_importer_spec.rb
|
|
286
|
+
- spec/lead_terms_edge_cases_spec.rb
|
|
287
|
+
- spec/lead_terms_links_qids_spec.rb
|
|
281
288
|
- spec/magic_words_spec.rb
|
|
282
289
|
- spec/markers_spec.rb
|
|
283
290
|
- spec/memory_monitor_spec.rb
|
|
284
291
|
- spec/metadata_index_spec.rb
|
|
285
292
|
- spec/multi_dump_attach_spec.rb
|
|
286
293
|
- spec/multistream_spec.rb
|
|
294
|
+
- spec/output_integrity_spec.rb
|
|
287
295
|
- spec/output_writer_spec.rb
|
|
288
296
|
- spec/p1_correctness_spec.rb
|
|
297
|
+
- spec/page_properties_spec.rb
|
|
289
298
|
- spec/parser_functions_spec.rb
|
|
290
299
|
- spec/ractor_worker_spec.rb
|
|
291
300
|
- spec/regex_spec.rb
|
|
301
|
+
- spec/region_semantics_spec.rb
|
|
292
302
|
- spec/section_extractor_spec.rb
|
|
293
303
|
- spec/spec_helper.rb
|
|
294
304
|
- spec/stream_processor_spec.rb
|
|
@@ -296,6 +306,7 @@ files:
|
|
|
296
306
|
- spec/support/multistream_fixture.rb
|
|
297
307
|
- spec/template_data_spec.rb
|
|
298
308
|
- spec/template_expander_spec.rb
|
|
309
|
+
- spec/template_passthrough_spec.rb
|
|
299
310
|
- spec/template_processing_spec.rb
|
|
300
311
|
- spec/text_processing_spec.rb
|
|
301
312
|
- spec/titles_output_path_spec.rb
|
|
@@ -345,17 +356,22 @@ test_files:
|
|
|
345
356
|
- spec/index_cache_spec.rb
|
|
346
357
|
- spec/integration_spec.rb
|
|
347
358
|
- spec/langlinks_importer_spec.rb
|
|
359
|
+
- spec/lead_terms_edge_cases_spec.rb
|
|
360
|
+
- spec/lead_terms_links_qids_spec.rb
|
|
348
361
|
- spec/magic_words_spec.rb
|
|
349
362
|
- spec/markers_spec.rb
|
|
350
363
|
- spec/memory_monitor_spec.rb
|
|
351
364
|
- spec/metadata_index_spec.rb
|
|
352
365
|
- spec/multi_dump_attach_spec.rb
|
|
353
366
|
- spec/multistream_spec.rb
|
|
367
|
+
- spec/output_integrity_spec.rb
|
|
354
368
|
- spec/output_writer_spec.rb
|
|
355
369
|
- spec/p1_correctness_spec.rb
|
|
370
|
+
- spec/page_properties_spec.rb
|
|
356
371
|
- spec/parser_functions_spec.rb
|
|
357
372
|
- spec/ractor_worker_spec.rb
|
|
358
373
|
- spec/regex_spec.rb
|
|
374
|
+
- spec/region_semantics_spec.rb
|
|
359
375
|
- spec/section_extractor_spec.rb
|
|
360
376
|
- spec/spec_helper.rb
|
|
361
377
|
- spec/stream_processor_spec.rb
|
|
@@ -363,6 +379,7 @@ test_files:
|
|
|
363
379
|
- spec/support/multistream_fixture.rb
|
|
364
380
|
- spec/template_data_spec.rb
|
|
365
381
|
- spec/template_expander_spec.rb
|
|
382
|
+
- spec/template_passthrough_spec.rb
|
|
366
383
|
- spec/template_processing_spec.rb
|
|
367
384
|
- spec/text_processing_spec.rb
|
|
368
385
|
- spec/titles_output_path_spec.rb
|