wp2txt 2.3.3 → 2.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/.dockerignore +0 -4
- data/.gitignore +3 -5
- data/CHANGELOG.md +18 -0
- data/DEVELOPMENT.md +1 -1
- data/DEVELOPMENT_ja.md +1 -1
- data/README.md +44 -2
- data/README_ja.md +35 -2
- data/Rakefile +10 -21
- data/bin/wp2txt +79 -17
- data/bin/wp2txt-mcp +1 -1
- data/docs/INDEXES.md +61 -1
- data/lib/wp2txt/article.rb +1 -1
- data/lib/wp2txt/cli.rb +42 -0
- data/lib/wp2txt/constants.rb +24 -0
- data/lib/wp2txt/corpus.rb +11 -2
- data/lib/wp2txt/data/template_aliases.json +1 -1
- data/lib/wp2txt/extractor.rb +10 -1
- data/lib/wp2txt/formatter.rb +20 -0
- data/lib/wp2txt/index_commands.rb +89 -1
- data/lib/wp2txt/langlinks_importer.rb +19 -35
- data/lib/wp2txt/lead_terms.rb +228 -0
- data/lib/wp2txt/link_counter.rb +171 -0
- data/lib/wp2txt/metadata_index.rb +79 -3
- data/lib/wp2txt/multistream.rb +52 -2
- data/lib/wp2txt/output_writer.rb +8 -0
- data/lib/wp2txt/page_props_importer.rb +170 -0
- data/lib/wp2txt/sql_dump_reader.rb +57 -0
- data/lib/wp2txt/stream_processor.rb +42 -15
- data/lib/wp2txt/template_expander.rb +19 -0
- data/lib/wp2txt/utils.rb +5 -3
- data/lib/wp2txt/version.rb +1 -1
- data/lib/wp2txt/wikitext_regions.rb +66 -0
- data/lib/wp2txt.rb +7 -5
- data/spec/docs_sync_spec.rb +3 -3
- data/spec/langlinks_importer_spec.rb +27 -0
- data/spec/lead_terms_edge_cases_spec.rb +194 -0
- data/spec/lead_terms_links_qids_spec.rb +204 -0
- data/spec/output_integrity_spec.rb +146 -0
- data/spec/p1_correctness_spec.rb +32 -14
- data/spec/page_properties_spec.rb +161 -0
- data/spec/region_semantics_spec.rb +71 -0
- data/spec/template_passthrough_spec.rb +43 -0
- data/spec/titles_output_path_spec.rb +12 -0
- metadata +18 -1
checksums.yaml
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
---
|
|
2
2
|
SHA256:
|
|
3
|
-
metadata.gz:
|
|
4
|
-
data.tar.gz:
|
|
3
|
+
metadata.gz: 54e6c29a38bad95092494bf88db73029769d942940953480e1db4e6433982cad
|
|
4
|
+
data.tar.gz: 811faae9a5b59bd3e58ca4210314d52daadfe356de0da4629732e9fe3af3fe60
|
|
5
5
|
SHA512:
|
|
6
|
-
metadata.gz:
|
|
7
|
-
data.tar.gz:
|
|
6
|
+
metadata.gz: 387f9374165f327a5f398306232ed32e5fa096e01b84519f8e1ab41f0e8b773d9f2cfccbc2e9504080550b9dcdb31ac269102845cae4a7a0e547ad6db864b5f1
|
|
7
|
+
data.tar.gz: 1a10d09d4e12bd94147fd3fa09126bb6450d141041c07013c71c4d51abc8537f80e95e822186f86591da94b4f7a6a0066b469e9379880f315938a82c7e03cbfb
|
data/.dockerignore
CHANGED
|
@@ -1,13 +1,11 @@
|
|
|
1
1
|
.git
|
|
2
2
|
.github
|
|
3
|
-
.claude
|
|
4
3
|
image
|
|
5
4
|
pkg
|
|
6
5
|
spec
|
|
7
6
|
coverage
|
|
8
7
|
tmp
|
|
9
8
|
.bundle
|
|
10
|
-
research-notes
|
|
11
9
|
benchmark_results
|
|
12
10
|
data/output_samples
|
|
13
11
|
scripts
|
|
@@ -15,10 +13,8 @@ scripts
|
|
|
15
13
|
.solargraph.yml
|
|
16
14
|
.rubocop.yml
|
|
17
15
|
Gemfile.lock
|
|
18
|
-
.private-doc-tokens
|
|
19
16
|
**/.DS_Store
|
|
20
17
|
.ruby-version
|
|
21
|
-
CLAUDE.md
|
|
22
18
|
DEVELOPMENT.md
|
|
23
19
|
DEVELOPMENT_ja.md
|
|
24
20
|
*.gem
|
data/.gitignore
CHANGED
|
@@ -20,11 +20,7 @@ tmp
|
|
|
20
20
|
*.~
|
|
21
21
|
tags
|
|
22
22
|
|
|
23
|
-
# Claude Code development files
|
|
24
|
-
CLAUDE.md
|
|
25
23
|
|
|
26
|
-
# Private design/research notes (not for publication)
|
|
27
|
-
research-notes/
|
|
28
24
|
|
|
29
25
|
# Error logs
|
|
30
26
|
error_log.txt
|
|
@@ -36,4 +32,6 @@ benchmark_results/
|
|
|
36
32
|
|
|
37
33
|
# Developer-specific Ruby version
|
|
38
34
|
.ruby-version
|
|
39
|
-
|
|
35
|
+
|
|
36
|
+
# Machine-local files (never committed)
|
|
37
|
+
.local/
|
data/CHANGELOG.md
CHANGED
|
@@ -8,6 +8,24 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
|
|
|
8
8
|
Entries were rewritten in August 2026 to describe what changed for people using
|
|
9
9
|
wp2txt, rather than how it was implemented. The changes themselves are unaltered.
|
|
10
10
|
|
|
11
|
+
## [2.4.0] - 2026-10-02
|
|
12
|
+
|
|
13
|
+
- **Readings and link text are no longer dropped from Japanese articles**: template expansion, on by default, removed `{{読み仮名}}`, ruby templates, and `{{仮リンク}}` before they could be rendered, so a lead like "'''{{読み仮名|言語|げんご}}'''は…" came out as "は…" and names linked through 仮リンク vanished from running text (one in six Japanese articles uses 仮リンク). Re-extract if you rely on these articles
|
|
14
|
+
- **Interlanguage links import from current dumps again**: dumps now write one row per line, which the importer read as nothing while still reporting success. If you imported langlinks from a dump dated September 2026 or later, re-import with `-U`. An import that reads no rows now fails instead of succeeding quietly
|
|
15
|
+
- **`--lead-terms` (JSON output)**: lists the terms an article introduces in its lead — the bold terms of the first lead paragraph that has one (up to five), the parenthesized text written right after each (split at top-level commas and semicolons, and unsplit), and reading templates as text/reading pairs. Each term carries character offsets into the article's wikitext as stored in the dump, so its source can be cut out and checked. Nothing is interpreted: whether a note is a reading, a native spelling, or a date is left to you. Not available with `--ractor`
|
|
16
|
+
- **`--count-links`**: adds, for every article in an existing metadata index, how many articles link to it — each linking article counted once, through one redirect hop, using the wiki's own capitalization rule, and ignoring links that the wiki does not render (comments, `nowiki`, `pre`, `math`, and similar). About 12 minutes for Japanese Wikipedia
|
|
17
|
+
- **`--import-page-props`**: adds each article's Wikidata ID, MediaWiki's disambiguation mark, and its sort key from the `page_props` dump of the same date as the index. Once imported, JSON output, `get_article`, and `extract_corpus` carry `qid`, `sort_key`, and `disambiguation` (null or false when a page has none). The disambiguation mark finds many more disambiguation pages than their titles do. On Japanese Wikipedia the sort key is a reading normalized for sorting (voicing marks dropped), useful for checking reading candidates
|
|
18
|
+
- **`dump_info` reports where the new data came from**, including the `page_props` file's SHA-256 and the rule used to count links
|
|
19
|
+
|
|
20
|
+
## [2.3.4] - 2026-09-30
|
|
21
|
+
|
|
22
|
+
- **Records are no longer duplicated in `--no-turbo` output or in large `extract_corpus` runs**: the last record written before each batch could appear once more for every worker process — up to eight identical lines in a row. The default mode for `.bz2` dumps was not affected. If you extracted more than 200 articles with `extract_corpus`, or used `--no-turbo`, check earlier output for repeated lines; the repeats are byte-for-byte identical, so removing exact duplicate lines restores it
|
|
23
|
+
- **`--num-procs` is now honoured**: any value smaller than the number of CPU cores was silently replaced with the automatic choice. The value you give is used, and one outside 1 to the core count is adjusted with a warning
|
|
24
|
+
- **`--articles` no longer fails on large dumps**: extracting specific articles could stop with `Invalid argument` (seen on macOS) when the last requested article sat early in a multi-gigabyte dump
|
|
25
|
+
- **JSON records include `page_id` and `revision_id`**: the dump's own IDs for the page and its revision, right after `title`, in every mode — so each record can be traced to the exact version it came from. From Ruby, `StreamProcessor#each_page(with_ids: true)` yields them as a third value; plain `each_page` is unchanged
|
|
26
|
+
- **Corrupt input now stops processing instead of being cleaned silently**: invalid UTF-8 raises `Wp2txt::EncodingError` naming where it was found. Official dumps are valid UTF-8, so this only triggers on damaged files
|
|
27
|
+
- **Articles that fail to render are reported**: the default mode skipped them without a word; each one is now named on stderr with the reason
|
|
28
|
+
|
|
11
29
|
## [2.3.3] - 2026-09-08
|
|
12
30
|
|
|
13
31
|
- **Short searches no longer report a false zero**: in Japanese, Chinese, and Korean indexes a phrase of one or two characters could match nothing and come back as `0 matches`, which reads exactly like a term that is genuinely absent from the dump. Such a search now fails with an explicit error instead. If you recorded a zero result for a short term with an earlier version, re-check it
|
data/DEVELOPMENT.md
CHANGED
|
@@ -406,7 +406,7 @@ local build sends the working tree as its context, so untracked files ride
|
|
|
406
406
|
along, while a runner starts from a clean checkout.
|
|
407
407
|
|
|
408
408
|
```bash
|
|
409
|
-
rake check_image # Build
|
|
409
|
+
rake check_image # Build from a clean copy of the last commit and run the workflow's gate
|
|
410
410
|
```
|
|
411
411
|
|
|
412
412
|
The gate (`scripts/verify_image.rb`) lists what the image holds under `/wp2txt`
|
data/DEVELOPMENT_ja.md
CHANGED
data/README.md
CHANGED
|
@@ -172,13 +172,55 @@ CATEGORIES: Category1, Category2, Category3
|
|
|
172
172
|
Each line contains one JSON object:
|
|
173
173
|
|
|
174
174
|
```json
|
|
175
|
-
{"title": "Article Title", "categories": ["Cat1", "Cat2"], "text": "...", "redirect": null}
|
|
175
|
+
{"title": "Article Title", "page_id": 12345, "revision_id": 67890, "categories": ["Cat1", "Cat2"], "text": "...", "redirect": null}
|
|
176
176
|
```
|
|
177
177
|
|
|
178
|
+
`page_id` and `revision_id` are the dump's own `<id>` values for the page and its
|
|
179
|
+
revision, so a record can be traced back to the exact version of the article it came from.
|
|
180
|
+
If page properties have been imported into the dump's metadata index
|
|
181
|
+
(`--import-page-props`, see [docs/INDEXES.md](docs/INDEXES.md)), three fields always
|
|
182
|
+
follow the source IDs: `qid`, `sort_key`, and `disambiguation`. Missing QIDs and sort
|
|
183
|
+
keys are `null`; an absent disambiguation flag is `false`. Before import, all three
|
|
184
|
+
fields are omitted.
|
|
185
|
+
|
|
186
|
+
`disambiguation` is MediaWiki's own flag, which identifies pages that a title-based
|
|
187
|
+
check would miss. `sort_key` is the stored sorting key, whose meaning varies by wiki:
|
|
188
|
+
Japanese Wikipedia commonly uses normalized readings without voicing marks and with
|
|
189
|
+
small kana enlarged; English Wikipedia may use "surname, given name". It is not a
|
|
190
|
+
reading itself, but can be used to compare reading candidates. WP2TXT returns the
|
|
191
|
+
stored value without interpreting it.
|
|
192
|
+
|
|
193
|
+
With `--lead-terms`, each record also lists the terms the article introduces in its lead:
|
|
194
|
+
|
|
195
|
+
```json
|
|
196
|
+
"lead_terms": [
|
|
197
|
+
{"index": 0, "text": "東京", "notes": ["とうきょう", "Tokyo"], "notes_text": "とうきょう、Tokyo",
|
|
198
|
+
"source": "bold", "span": {"bold": [0, 8], "paren": [8, 21]}}
|
|
199
|
+
]
|
|
200
|
+
```
|
|
201
|
+
|
|
202
|
+
- `source: "bold"` — a bold term in the first lead paragraph that has one (up to five),
|
|
203
|
+
with the parenthesized text written right after it, split at top-level commas and
|
|
204
|
+
semicolons into `notes` (`notes_text` keeps it unsplit). Bold text inside templates,
|
|
205
|
+
image captions, references, literal regions such as `nowiki` and `pre`, and
|
|
206
|
+
`gallery`/`timeline` content is not counted. `code` is an ordinary formatting tag:
|
|
207
|
+
bold text inside it can be a lead term.
|
|
208
|
+
- `source: <template name>` — a reading template such as `{{読み仮名}}`, reported as
|
|
209
|
+
`text` and `reading`.
|
|
210
|
+
- `span` gives character offsets `[start, end)` into the article's wikitext as stored in
|
|
211
|
+
the dump (XML entities decoded, nothing else changed), so the source of each term can be
|
|
212
|
+
cut out and checked later.
|
|
213
|
+
- Nothing is interpreted: whether a note is a reading, a native spelling, or a date is left
|
|
214
|
+
to you.
|
|
215
|
+
|
|
216
|
+
`--lead-terms` cannot be combined with `--ractor`. The experimental Ractor JSON path
|
|
217
|
+
also omits page IDs and revision IDs; the CLI warns when it is selected. Imported
|
|
218
|
+
page properties are attached in the parent process on this path too.
|
|
219
|
+
|
|
178
220
|
For redirect articles:
|
|
179
221
|
|
|
180
222
|
```json
|
|
181
|
-
{"title": "NYC", "categories": [], "text": "", "redirect": "New York City"}
|
|
223
|
+
{"title": "NYC", "page_id": 23456, "revision_id": 78901, "categories": [], "text": "", "redirect": "New York City"}
|
|
182
224
|
```
|
|
183
225
|
|
|
184
226
|
## Cache Management
|
data/README_ja.md
CHANGED
|
@@ -156,13 +156,46 @@ CATEGORIES: カテゴリ1, カテゴリ2, カテゴリ3
|
|
|
156
156
|
各行に1つのJSONオブジェクト:
|
|
157
157
|
|
|
158
158
|
```json
|
|
159
|
-
{"title": "記事タイトル", "categories": ["カテゴリ1", "カテゴリ2"], "text": "...", "redirect": null}
|
|
159
|
+
{"title": "記事タイトル", "page_id": 12345, "revision_id": 67890, "categories": ["カテゴリ1", "カテゴリ2"], "text": "...", "redirect": null}
|
|
160
160
|
```
|
|
161
161
|
|
|
162
|
+
`page_id` と `revision_id` はダンプに記録されたページと版の `<id>` です。各レコードが
|
|
163
|
+
どの記事のどの版から取り出されたかを後から辿れます。
|
|
164
|
+
そのダンプのメタデータ索引にページ属性を取り込んである場合(`--import-page-props`、
|
|
165
|
+
[docs/INDEXES.md](docs/INDEXES.md) 参照)、その後に `qid`・`sort_key`・`disambiguation` の3項目が常に続きます。
|
|
166
|
+
QID・並べ替えの鍵がなければ `null`、曖昧さ回避の印がなければ `false`。未取り込みなら3項目とも出しません。
|
|
167
|
+
|
|
168
|
+
`disambiguation` は MediaWiki 自身の印で、題名の形だけでは拾えない曖昧さ回避ページも識別できます。
|
|
169
|
+
`sort_key` の意味は言語版ごとに異なります。日本語版では通常、濁点・半濁点を落とし、小書きの仮名を
|
|
170
|
+
大書きにした並べ替え用の読み(例: 言語→「けんこ」)、英語版では「姓, 名」の形などです。
|
|
171
|
+
読みそのものではありませんが、読みの候補の照合に使えます。WP2TXT は値を解釈せず、そのまま渡します。
|
|
172
|
+
|
|
173
|
+
`--lead-terms` を付けると、記事の導入部で示される語の一覧も出力します:
|
|
174
|
+
|
|
175
|
+
```json
|
|
176
|
+
"lead_terms": [
|
|
177
|
+
{"index": 0, "text": "東京", "notes": ["とうきょう", "Tokyo"], "notes_text": "とうきょう、Tokyo",
|
|
178
|
+
"source": "bold", "span": {"bold": [0, 8], "paren": [8, 21]}}
|
|
179
|
+
]
|
|
180
|
+
```
|
|
181
|
+
|
|
182
|
+
- `source: "bold"` — 導入部で太字を含む最初の段落の太字(最大5個)と、その直後の括弧の中身。
|
|
183
|
+
括弧の中身は最上位の読点・カンマ・セミコロンで区切って `notes` に入れる(`notes_text` は区切る前)。
|
|
184
|
+
テンプレート・画像の説明文・脚注、`nowiki`・`pre` などの字義どおりの領域、および
|
|
185
|
+
`gallery`・`timeline` の太字は対象外。`code` は通常の書式タグなので、中の太字も対象
|
|
186
|
+
- `source: <テンプレート名>` — `{{読み仮名}}` などの読みのテンプレート。`text` と `reading` の組
|
|
187
|
+
- `span` はダンプに記録された記事の wikitext(XML の実体参照を戻しただけのもの)での文字位置
|
|
188
|
+
`[開始, 終了)`。各語の根拠をあとから切り出して確かめられる
|
|
189
|
+
- 解釈はしない。括弧の中のどれが読みで、どれが原語や日付かの判断は利用側に任せる
|
|
190
|
+
|
|
191
|
+
`--lead-terms` と `--ractor` は併用できません。実験的な Ractor 経路の JSON には
|
|
192
|
+
ページ ID・リビジョン ID が含まれないため、選択時に警告します。取り込み済みの属性3項目は、
|
|
193
|
+
この経路でも親プロセス側で付与します。
|
|
194
|
+
|
|
162
195
|
リダイレクト記事の場合:
|
|
163
196
|
|
|
164
197
|
```json
|
|
165
|
-
{"title": "NYC", "categories": [], "text": "", "redirect": "New York City"}
|
|
198
|
+
{"title": "NYC", "page_id": 23456, "revision_id": 78901, "categories": [], "text": "", "redirect": "New York City"}
|
|
166
199
|
```
|
|
167
200
|
|
|
168
201
|
## キャッシュ管理
|
data/Rakefile
CHANGED
|
@@ -30,34 +30,23 @@ end
|
|
|
30
30
|
Rake::Task["build"].enhance([:normalize_permissions])
|
|
31
31
|
|
|
32
32
|
# Pre-release gate: verify the built gem's payload against spec.files and scan
|
|
33
|
-
# it for names, content, and modes that must never ship
|
|
33
|
+
# it for names, content, and modes that must never ship.
|
|
34
34
|
Rake::Task["build"].enhance { sh "ruby", "scripts/verify_gem.rb" }
|
|
35
35
|
|
|
36
36
|
# =============================================================================
|
|
37
37
|
# Docker
|
|
38
38
|
# =============================================================================
|
|
39
39
|
|
|
40
|
-
|
|
41
|
-
# working tree, so anything ignored locally (private notes, scratch files)
|
|
42
|
-
# would otherwise ride along.
|
|
43
|
-
IMAGE_FORBIDDEN_PATHS = %w[/wp2txt/research-notes /wp2txt/tmp /wp2txt/.git /wp2txt/CLAUDE.md /wp2txt/.claude /wp2txt/.private-doc-tokens].freeze
|
|
44
|
-
|
|
45
|
-
desc "Verify a built image contains no private material (run before pushing)"
|
|
46
|
-
task :verify_image, [:tag] do |_t, args|
|
|
47
|
-
tag = args[:tag] || "wp2txt-verify:local"
|
|
48
|
-
checks = IMAGE_FORBIDDEN_PATHS.map { |p| "test -e #{p} && echo LEAK:#{p}" }.join("; ")
|
|
49
|
-
out, status = Open3.capture2e("docker", "run", "--rm", tag, "sh", "-c", "#{checks}; true")
|
|
50
|
-
abort "Image verification failed for #{tag}: #{out}" unless status.success?
|
|
51
|
-
leaks = out.lines.grep(/^LEAK:/).map(&:strip)
|
|
52
|
-
abort "Image #{tag} contains private paths:\n #{leaks.join("\n ")}" unless leaks.empty?
|
|
53
|
-
|
|
54
|
-
puts "OK: #{tag} contains none of #{IMAGE_FORBIDDEN_PATHS.join(', ')}"
|
|
55
|
-
end
|
|
56
|
-
|
|
57
|
-
desc "Build the image locally and verify it, without pushing"
|
|
40
|
+
desc "Build the image from a clean copy of the last commit and run the gate CI runs"
|
|
58
41
|
task :check_image do
|
|
59
|
-
|
|
60
|
-
|
|
42
|
+
# A clean clone is what CI builds from: files that exist only in this working
|
|
43
|
+
# tree never reach the context, and the gate compares the image against it.
|
|
44
|
+
require "tmpdir"
|
|
45
|
+
Dir.mktmpdir("wp2txt-image-") do |dir|
|
|
46
|
+
sh "git", "clone", "--quiet", "--no-local", ".", dir
|
|
47
|
+
sh "docker", "build", "-t", "wp2txt-verify:local", dir
|
|
48
|
+
sh "ruby", "scripts/verify_image.rb", "wp2txt-verify:local", "--context", dir
|
|
49
|
+
end
|
|
61
50
|
end
|
|
62
51
|
|
|
63
52
|
desc "Explain how images are published (they are built and pushed by CI)"
|
data/bin/wp2txt
CHANGED
|
@@ -15,6 +15,7 @@ require_relative "../lib/wp2txt/formatter"
|
|
|
15
15
|
require_relative "../lib/wp2txt/extractor"
|
|
16
16
|
require_relative "../lib/wp2txt/ractor_worker"
|
|
17
17
|
require_relative "../lib/wp2txt/index_commands"
|
|
18
|
+
require_relative "../lib/wp2txt/lead_terms"
|
|
18
19
|
|
|
19
20
|
require "etc"
|
|
20
21
|
require "json"
|
|
@@ -45,14 +46,48 @@ class WpApp
|
|
|
45
46
|
def calculate_num_processes(opts)
|
|
46
47
|
optimal = Wp2txt::MemoryMonitor.optimal_processes
|
|
47
48
|
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
|
|
49
|
+
return optimal unless opts[:num_procs]
|
|
50
|
+
|
|
51
|
+
requested = opts[:num_procs].to_i
|
|
52
|
+
used = requested.clamp(1, Etc.nprocessors)
|
|
53
|
+
if used != requested
|
|
54
|
+
print_warning("--num-procs #{requested} is outside 1..#{Etc.nprocessors}; using #{used}.")
|
|
55
|
+
end
|
|
56
|
+
used
|
|
57
|
+
end
|
|
58
|
+
|
|
59
|
+
# Attach source identifiers and, when asked, lead terms to an article.
|
|
60
|
+
# raw_wikitext is the article as stored in the dump (XML-decoded, before
|
|
61
|
+
# comment removal): lead-term positions are offsets into exactly that text.
|
|
62
|
+
def set_source_fields(article, page_id, revision_id, raw_wikitext, config)
|
|
63
|
+
article.page_id = page_id
|
|
64
|
+
article.revision_id = revision_id
|
|
65
|
+
if config[:page_properties]
|
|
66
|
+
article.page_properties = Wp2txt::MetadataIndex.property_values(config[:page_properties][page_id])
|
|
67
|
+
end
|
|
68
|
+
return unless config[:lead_terms] && raw_wikitext
|
|
69
|
+
|
|
70
|
+
article.lead_terms = Wp2txt::LeadTerms.extract(raw_wikitext, render: ->(fragment) { format_wiki(fragment, config) })
|
|
71
|
+
end
|
|
72
|
+
|
|
73
|
+
# Page properties from the metadata index of this dump, when imported
|
|
74
|
+
# (wp2txt --import-page-props); nil otherwise
|
|
75
|
+
def load_page_properties(input_path, cache_dir)
|
|
76
|
+
db_path = Wp2txt::MetadataIndex.path_for(File.expand_path(input_path), cache_dir: cache_dir)
|
|
77
|
+
return nil unless File.exist?(db_path)
|
|
78
|
+
|
|
79
|
+
db = SQLite3::Database.new(db_path, readonly: true)
|
|
80
|
+
return nil unless Wp2txt::MetadataIndex.page_properties_imported?(db)
|
|
81
|
+
|
|
82
|
+
rows = {}
|
|
83
|
+
db.execute("SELECT page_id, qid, sort_key, disambiguation FROM page_properties") do |id, qid, key, flag|
|
|
84
|
+
rows[id] = [qid, key, flag]
|
|
85
|
+
end
|
|
86
|
+
rows
|
|
87
|
+
rescue SQLite3::Exception
|
|
88
|
+
nil
|
|
89
|
+
ensure
|
|
90
|
+
db&.close
|
|
56
91
|
end
|
|
57
92
|
|
|
58
93
|
# Process articles using turbo mode (split-first architecture from v1.x)
|
|
@@ -232,6 +267,8 @@ class WpApp
|
|
|
232
267
|
text = page_xml[text_start...text_end_match.begin(0)]
|
|
233
268
|
next if text.nil? || text.empty?
|
|
234
269
|
|
|
270
|
+
raw_wikitext = Wp2txt.xml_decode(text) if config[:lead_terms]
|
|
271
|
+
|
|
235
272
|
# Decode XML entities
|
|
236
273
|
text = text.gsub("<", "<").gsub(">", ">").gsub("&", "&").gsub(""", '"')
|
|
237
274
|
|
|
@@ -244,6 +281,8 @@ class WpApp
|
|
|
244
281
|
next if redirect_page?(text)
|
|
245
282
|
|
|
246
283
|
article = Article.new(text, title, strip_tmarker)
|
|
284
|
+
ids = Wp2txt.page_ids(page_xml)
|
|
285
|
+
set_source_fields(article, ids[:page_id], ids[:revision_id], raw_wikitext, config)
|
|
247
286
|
result = format_article(article, config)
|
|
248
287
|
next unless result
|
|
249
288
|
|
|
@@ -254,7 +293,9 @@ class WpApp
|
|
|
254
293
|
out.puts(result)
|
|
255
294
|
end
|
|
256
295
|
article_count += 1
|
|
257
|
-
rescue StandardError
|
|
296
|
+
rescue StandardError => e
|
|
297
|
+
# Keep going, but never lose an article silently
|
|
298
|
+
warn "wp2txt: skipped #{title.inspect}: #{e.class}: #{e.message}"
|
|
258
299
|
next
|
|
259
300
|
end
|
|
260
301
|
end
|
|
@@ -285,7 +326,7 @@ class WpApp
|
|
|
285
326
|
base_name = base_name.sub(/\.xml$/, "") # Handle .xml.bz2
|
|
286
327
|
|
|
287
328
|
# Create stream processor
|
|
288
|
-
stream = StreamProcessor.new(input_path, bz2_gem: bz2_gem)
|
|
329
|
+
stream = StreamProcessor.new(input_path, bz2_gem: bz2_gem, keep_raw_text: config[:lead_terms])
|
|
289
330
|
|
|
290
331
|
# Create output writer
|
|
291
332
|
writer = OutputWriter.new(
|
|
@@ -355,8 +396,8 @@ class WpApp
|
|
|
355
396
|
end
|
|
356
397
|
puts
|
|
357
398
|
|
|
358
|
-
stream.each_page do |title, text|
|
|
359
|
-
pages << [title, text]
|
|
399
|
+
stream.each_page(with_ids: true) do |title, text, ids|
|
|
400
|
+
pages << [title, text, ids]
|
|
360
401
|
page_count += 1
|
|
361
402
|
|
|
362
403
|
# Process batch when full
|
|
@@ -439,23 +480,38 @@ class WpApp
|
|
|
439
480
|
# Process a batch of pages in parallel
|
|
440
481
|
# Uses Ractor for true parallelism when enabled, otherwise falls back to Parallel gem
|
|
441
482
|
def process_batch(pages, writer, config, strip_tmarker, num_processes)
|
|
442
|
-
|
|
483
|
+
use_ractor = config[:use_ractor] && Wp2txt::RactorWorker.available?
|
|
484
|
+
results = if use_ractor
|
|
443
485
|
# Use Ractor-based parallel processing (true parallelism)
|
|
444
486
|
Wp2txt::RactorWorker.process_articles(
|
|
445
487
|
pages,
|
|
446
|
-
config: config,
|
|
488
|
+
config: config.reject { |key, _| key == :page_properties },
|
|
447
489
|
strip_tmarker: strip_tmarker,
|
|
448
490
|
num_workers: num_processes
|
|
449
491
|
)
|
|
450
492
|
else
|
|
451
493
|
# Fall back to Parallel gem (process-based parallelism)
|
|
452
|
-
|
|
494
|
+
# Anything the writer still buffers would be inherited by every
|
|
495
|
+
# forked worker and written again as each one exits.
|
|
496
|
+
writer.flush
|
|
497
|
+
Parallel.map(pages, in_processes: num_processes) do |title, text, ids|
|
|
453
498
|
article = Article.new(text, title, strip_tmarker)
|
|
499
|
+
set_source_fields(article, ids&.dig(:page_id), ids&.dig(:revision_id), ids&.dig(:raw_text), config)
|
|
454
500
|
format_article(article, config)
|
|
455
501
|
end
|
|
456
502
|
end
|
|
457
503
|
|
|
458
|
-
results.
|
|
504
|
+
results.each_with_index do |result, index|
|
|
505
|
+
# Keep the bulk properties table in the parent, outside Ractor's
|
|
506
|
+
# shareable config. Results retain the order of the input pages.
|
|
507
|
+
if use_ractor && result.is_a?(Hash) && config[:page_properties]
|
|
508
|
+
id = pages[index][2]&.dig(:page_id)
|
|
509
|
+
properties = Wp2txt::MetadataIndex.property_values(config[:page_properties][id]).transform_keys(&:to_s)
|
|
510
|
+
result = result.each_with_object({}) do |(key, value), row|
|
|
511
|
+
row[key] = value
|
|
512
|
+
row.merge!(properties) if key == "title"
|
|
513
|
+
end
|
|
514
|
+
end
|
|
459
515
|
writer.write(result) if result
|
|
460
516
|
end
|
|
461
517
|
end
|
|
@@ -714,6 +770,8 @@ class WpApp
|
|
|
714
770
|
return run_search(opts) if opts[:search]
|
|
715
771
|
return run_fts_optimize(opts) if opts[:fts_optimize]
|
|
716
772
|
return run_import_langlinks(opts) if opts[:import_langlinks]
|
|
773
|
+
return run_import_page_props(opts) if opts[:import_page_props]
|
|
774
|
+
return run_count_links(opts) if opts[:count_links]
|
|
717
775
|
|
|
718
776
|
# Determine input source
|
|
719
777
|
if opts[:from_category] && opts[:lang]
|
|
@@ -753,7 +811,7 @@ class WpApp
|
|
|
753
811
|
no_turbo: opts[:no_turbo]
|
|
754
812
|
}
|
|
755
813
|
|
|
756
|
-
%i[title list heading table pre ref redirect multiline category category_only
|
|
814
|
+
%i[lead_terms title list heading table pre ref redirect multiline category category_only
|
|
757
815
|
summary_only metadata_only marker extract_citations expand_templates
|
|
758
816
|
section_output min_section_length skip_empty
|
|
759
817
|
alias_file no_section_aliases section_stats show_matched_sections].each do |opt|
|
|
@@ -768,6 +826,10 @@ class WpApp
|
|
|
768
826
|
# Parse markers option
|
|
769
827
|
config[:markers] = parse_markers_option(opts[:markers])
|
|
770
828
|
|
|
829
|
+
# Imported page properties ride along in JSON output;
|
|
830
|
+
# loaded here, before any worker is forked, so workers share the table
|
|
831
|
+
config[:page_properties] = load_page_properties(input_path, opts[:cache_dir]) if format == :json
|
|
832
|
+
|
|
771
833
|
# Handle section-stats mode (standalone, outputs to stdout)
|
|
772
834
|
if opts[:section_stats]
|
|
773
835
|
return process_section_stats(input_path, config)
|
data/bin/wp2txt-mcp
CHANGED
|
@@ -80,7 +80,7 @@ end
|
|
|
80
80
|
|
|
81
81
|
server.define_tool(
|
|
82
82
|
name: "dump_info",
|
|
83
|
-
description: "Dump identity (language/date), available index tiers, and article/category/section counts. Call first to learn what this corpus contains.",
|
|
83
|
+
description: "Dump identity (language/date), available index tiers, and article/category/section counts, plus provenance and counts for imported page properties (Wikidata IDs, disambiguation flags, sort keys), interlanguage links, and incoming-link counts. Call first to learn what this corpus contains.",
|
|
84
84
|
input_schema: { properties: {}, required: [] }
|
|
85
85
|
) do |server_context:|
|
|
86
86
|
respond { corpus.dump_info }
|
data/docs/INDEXES.md
CHANGED
|
@@ -71,6 +71,66 @@ $ wp2txt --import-langlinks -L ja --langlinks-langs en,de,fr,zh,ko
|
|
|
71
71
|
This adds a `langlinks` table (`ll_from` = source page_id, `ll_lang`, `ll_title`) that can
|
|
72
72
|
be joined in SQL. Tip: filter `ll_title != ''` — real dumps contain a few empty-title rows.
|
|
73
73
|
|
|
74
|
+
### Page properties and incoming links
|
|
75
|
+
|
|
76
|
+
Two more signals can be added to an existing metadata index, each with one command:
|
|
77
|
+
|
|
78
|
+
```console
|
|
79
|
+
$ wp2txt --import-page-props -L ja # each article's Wikidata item ID
|
|
80
|
+
$ wp2txt --count-links -L ja # how many articles link to each article
|
|
81
|
+
```
|
|
82
|
+
|
|
83
|
+
- `--import-page-props` reads the official `page_props` dump of the **same date** as the
|
|
84
|
+
index (a mismatch is refused) and creates `page_properties` with columns `page_id`
|
|
85
|
+
(integer primary key), `qid` (nullable text), `disambiguation` (integer, default 0,
|
|
86
|
+
not null), and `sort_key` (nullable text). A row exists only for a page with at least
|
|
87
|
+
one of these properties.
|
|
88
|
+
The source file's name, size, SHA-256, import time, page count, each property's count,
|
|
89
|
+
and the count of invalid UTF-8 sort keys skipped are reported under
|
|
90
|
+
`dump_info.page_properties`. Once imported, JSON records always carry `qid`,
|
|
91
|
+
`sort_key`, and `disambiguation`, including null/false values for absent properties.
|
|
92
|
+
Before import, all three are omitted. `get_article` and `extract_corpus` use the
|
|
93
|
+
same rule. The three imported properties are `wikibase_item`, `defaultsort`, and
|
|
94
|
+
`disambiguation`; unrelated properties are ignored.
|
|
95
|
+
MediaWiki's disambiguation flag is more reliable than matching the page title.
|
|
96
|
+
Sort keys are language-specific: Japanese Wikipedia commonly removes voicing marks
|
|
97
|
+
and enlarges small kana (言語 → けんこ), while English may use "surname, given name".
|
|
98
|
+
A sort key is not a reading itself, but can help compare reading candidates.
|
|
99
|
+
- `--count-links` scans the dump and adds `page_inlinks` (`page_id`, `inlinks`,
|
|
100
|
+
`via_redirects`) for every article. Each linking article counts **once** per target,
|
|
101
|
+
however many times it links; a link to a redirect counts for the redirect's target
|
|
102
|
+
(`via_redirects` is how many articles reached it only that way). Only links written in
|
|
103
|
+
articles' own text count — links that navigation templates add when a page is rendered
|
|
104
|
+
are not in the dump, so they are not counted. Links inside comments or literal
|
|
105
|
+
regions (`nowiki`, `pre`, `math`, `chem`, `ce`, `score`, `syntaxhighlight`, `source`,
|
|
106
|
+
`graph`, `mapframe`, `templatedata`) and links from redirects or other namespaces
|
|
107
|
+
do not count. Links inside `gallery`, `timeline`, and ordinary formatting tags
|
|
108
|
+
such as `code` do count. Lead-term extraction separately excludes `gallery` and
|
|
109
|
+
`timeline` because their contents are not lead prose.
|
|
110
|
+
Japanese Wikipedia takes about 12 minutes on
|
|
111
|
+
an Apple Silicon laptop.
|
|
112
|
+
Counting rule version 2 decodes title entities and respects the dump's siteinfo case
|
|
113
|
+
rule, recorded as `case_rule` in metadata when the index is built. Existing indexes
|
|
114
|
+
without this value use `first-letter`; rebuild the metadata index before counting
|
|
115
|
+
links for a case-sensitive wiki. Re-run `--count-links` to replace counts from rule 1.
|
|
116
|
+
|
|
117
|
+
Together with langlinks these let a query select articles by how referenced they are
|
|
118
|
+
within an edition, how many editions cover them, and what Wikidata says they are:
|
|
119
|
+
|
|
120
|
+
```sql
|
|
121
|
+
SELECT p.title, i.inlinks, q.qid, q.sort_key, q.disambiguation,
|
|
122
|
+
(SELECT COUNT(DISTINCT ll_lang) FROM langlinks l WHERE l.ll_from = p.page_id) AS editions
|
|
123
|
+
FROM pages p
|
|
124
|
+
JOIN page_inlinks i USING (page_id)
|
|
125
|
+
LEFT JOIN page_properties q USING (page_id)
|
|
126
|
+
WHERE p.namespace = 0 AND p.redirect_to IS NULL AND i.inlinks >= 5
|
|
127
|
+
ORDER BY i.inlinks DESC
|
|
128
|
+
```
|
|
129
|
+
|
|
130
|
+
Link counts measure how much an edition refers to a topic, not whether it is a proper
|
|
131
|
+
noun: common nouns ("singer", "high school") rank high too. Use the Wikidata ID to tell
|
|
132
|
+
kinds of entities apart.
|
|
133
|
+
|
|
74
134
|
## 4. The MCP server
|
|
75
135
|
|
|
76
136
|
`wp2txt-mcp` exposes a local dump to any MCP-capable LLM client (Claude, ChatGPT, Gemini,
|
|
@@ -105,7 +165,7 @@ $ claude mcp add wp2txt -- docker run -i --rm -v wp2txt:/root/.wp2txt ghcr.io/yo
|
|
|
105
165
|
|
|
106
166
|
| Tool | Purpose |
|
|
107
167
|
|------|---------|
|
|
108
|
-
| `dump_info` | Dump identity, installed indexes, corpus statistics, langlinks
|
|
168
|
+
| `dump_info` | Dump identity, installed indexes, corpus statistics, provenance of langlinks, Wikidata IDs, and link counts |
|
|
109
169
|
| `get_article` / `get_sections` / `list_headings` / `get_categories` | Single-article access (redirect-aware) |
|
|
110
170
|
| `find_articles` | Exhaustive filtered listing (category recursion, category AND / pattern match, section headings, title match) |
|
|
111
171
|
| `category_tree` / `section_stats` | Scope exploration and heading-frequency discovery |
|
data/lib/wp2txt/article.rb
CHANGED
|
@@ -26,7 +26,7 @@ module Wp2txt
|
|
|
26
26
|
# an article contains elements, each of which is [TYPE, string]
|
|
27
27
|
class Article
|
|
28
28
|
include Wp2txt
|
|
29
|
-
attr_accessor :elements, :title, :categories
|
|
29
|
+
attr_accessor :elements, :title, :categories, :page_id, :revision_id, :page_properties, :lead_terms
|
|
30
30
|
|
|
31
31
|
def initialize(text, title = "", strip_tmarker = false)
|
|
32
32
|
@title = title.strip
|
data/lib/wp2txt/cli.rb
CHANGED
|
@@ -150,6 +150,16 @@ module Wp2txt
|
|
|
150
150
|
opt :langlinks_langs, "Comma-separated target languages to import with --import-langlinks (default: all)",
|
|
151
151
|
type: String, short: :none
|
|
152
152
|
|
|
153
|
+
# Wikidata item IDs and incoming-link counts, added to an existing metadata index
|
|
154
|
+
opt :import_page_props, "Import Wikidata IDs, disambiguation flags, and sort keys (page_props dump) into the metadata index (requires --lang)",
|
|
155
|
+
default: false, short: :none
|
|
156
|
+
opt :page_props_file, "Use a local page_props .sql(.gz) file instead of downloading (with --import-page-props)",
|
|
157
|
+
type: String, short: :none
|
|
158
|
+
opt :count_links, "Count incoming links to each article and store them in the metadata index (requires --lang)",
|
|
159
|
+
default: false, short: :none
|
|
160
|
+
opt :lead_terms, "Add the lead's bold terms, their parenthesized notes, and reading templates to JSON output",
|
|
161
|
+
default: false, short: :none
|
|
162
|
+
|
|
153
163
|
opt :file_size, "Approximate size (in MB) of each output file (0 for single file)",
|
|
154
164
|
default: 10, short: "-f"
|
|
155
165
|
opt :num_procs, "Number of parallel processes (auto-detected based on CPU/memory)",
|
|
@@ -377,6 +387,38 @@ module Wp2txt
|
|
|
377
387
|
end
|
|
378
388
|
end
|
|
379
389
|
|
|
390
|
+
# Page-props import and link counting are standalone modes on an existing index
|
|
391
|
+
{ import_page_props: "--import-page-props", count_links: "--count-links" }.each do |key, flag|
|
|
392
|
+
next unless opts[key]
|
|
393
|
+
|
|
394
|
+
Optimist.die "#{flag} requires --lang" if opts[:lang].nil?
|
|
395
|
+
others = { build_index: "--build-index", find_articles: "--find-articles", search: "--search",
|
|
396
|
+
fts_optimize: "--fts-optimize", import_langlinks: "--import-langlinks",
|
|
397
|
+
import_page_props: "--import-page-props", count_links: "--count-links",
|
|
398
|
+
articles: "--articles", from_category: "--from-category", section_stats: "--section-stats" }
|
|
399
|
+
conflicts = others.reject { |k, _| k == key }.select { |k, _| opts[k] }.values
|
|
400
|
+
Optimist.die "#{flag} cannot be combined with #{conflicts.join(', ')}" unless conflicts.empty?
|
|
401
|
+
end
|
|
402
|
+
|
|
403
|
+
if opts[:page_props_file] && !opts[:import_page_props]
|
|
404
|
+
Optimist.die "--page-props-file requires --import-page-props"
|
|
405
|
+
end
|
|
406
|
+
|
|
407
|
+
if opts[:page_props_file] && !File.exist?(opts[:page_props_file])
|
|
408
|
+
Optimist.die :page_props_file, "file does not exist"
|
|
409
|
+
end
|
|
410
|
+
|
|
411
|
+
if opts[:lead_terms] && opts[:format].to_s != "json"
|
|
412
|
+
Optimist.die "--lead-terms requires --format json"
|
|
413
|
+
end
|
|
414
|
+
|
|
415
|
+
if opts[:lead_terms] && opts[:ractor]
|
|
416
|
+
Optimist.die "--lead-terms cannot be combined with --ractor"
|
|
417
|
+
end
|
|
418
|
+
if opts[:ractor] && opts[:format].to_s == "json"
|
|
419
|
+
warn "Warning: --ractor JSON output does not include page IDs or revision IDs"
|
|
420
|
+
end
|
|
421
|
+
|
|
380
422
|
if opts[:langlinks_file] && !opts[:import_langlinks]
|
|
381
423
|
Optimist.die "--langlinks-file requires --import-langlinks"
|
|
382
424
|
end
|
data/lib/wp2txt/constants.rb
CHANGED
|
@@ -6,6 +6,30 @@ module Wp2txt
|
|
|
6
6
|
(value || "0").to_i
|
|
7
7
|
end
|
|
8
8
|
|
|
9
|
+
XML_ENTITIES = { "lt" => "<", "gt" => ">", "amp" => "&", "quot" => '"', "apos" => "'" }.freeze
|
|
10
|
+
|
|
11
|
+
# Decode XML character references in one pass, as an XML parser does
|
|
12
|
+
# (a single pass keeps "&lt;" as the literal text "<")
|
|
13
|
+
def self.xml_decode(str)
|
|
14
|
+
str.gsub(/&(?:(lt|gt|amp|quot|apos)|#(\d+)|#x(\h+));/) do
|
|
15
|
+
if Regexp.last_match(1) then XML_ENTITIES[Regexp.last_match(1)]
|
|
16
|
+
elsif Regexp.last_match(2) then Regexp.last_match(2).to_i.chr(Encoding::UTF_8)
|
|
17
|
+
else Regexp.last_match(3).to_i(16).chr(Encoding::UTF_8)
|
|
18
|
+
end
|
|
19
|
+
end
|
|
20
|
+
end
|
|
21
|
+
|
|
22
|
+
# Page and revision IDs from one <page> element of a dump. The page's own
|
|
23
|
+
# <id> precedes <revision>; the revision's <id> is its first child (the
|
|
24
|
+
# contributor's <id> comes later). Only the header before <text> is read.
|
|
25
|
+
def self.page_ids(page_xml)
|
|
26
|
+
head = page_xml[0, page_xml.index("<text") || page_xml.size]
|
|
27
|
+
{
|
|
28
|
+
page_id: head[%r{<page>.*?<id>(\d+)</id>}m, 1]&.to_i,
|
|
29
|
+
revision_id: head[%r{<revision>\s*<id>(\d+)</id>}, 1]&.to_i
|
|
30
|
+
}
|
|
31
|
+
end
|
|
32
|
+
|
|
9
33
|
# =========================================================================
|
|
10
34
|
# Custom Exception Classes
|
|
11
35
|
# =========================================================================
|