wp2txt 2.1.1 → 2.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,245 @@
1
+ # frozen_string_literal: true
2
+
3
+ require "spec_helper"
4
+ require "tmpdir"
5
+ require "fileutils"
6
+ require "json"
7
+ require_relative "support/multistream_fixture"
8
+ require_relative "../lib/wp2txt/fts_index"
9
+ require_relative "../lib/wp2txt/corpus"
10
+
11
+ RSpec.describe "Wp2txt Full-Text Search" do
12
+ include MultistreamFixture
13
+
14
+ def build_indexes(dir, tokenizer: "unicode61", optimize: true)
15
+ multistream_path, index_path = create_fixture(dir)
16
+ ms_index = Wp2txt::MultistreamIndex.new(index_path, use_cache: false, show_progress: false)
17
+
18
+ meta_db = Wp2txt::MetadataIndex.path_for(multistream_path, cache_dir: dir)
19
+ Wp2txt::MetadataIndexBuilder.new(
20
+ multistream_path, ms_index.stream_offsets, db_path: meta_db, num_processes: 0
21
+ ).build
22
+
23
+ fts_db = Wp2txt::FtsIndex.path_for(multistream_path, cache_dir: dir)
24
+ Wp2txt::FtsIndexBuilder.new(
25
+ multistream_path, ms_index.stream_offsets,
26
+ db_path: fts_db, meta_db_path: meta_db, tokenizer: tokenizer,
27
+ num_processes: 0, optimize: optimize
28
+ ).build
29
+
30
+ [multistream_path, fts_db, meta_db]
31
+ end
32
+
33
+ describe Wp2txt::FtsIndex do
34
+ describe ".default_tokenizer" do
35
+ it "picks trigram for CJK language dumps" do
36
+ expect(described_class.default_tokenizer("/x/jawiki-20260701-multistream.xml.bz2")).to eq("trigram")
37
+ expect(described_class.default_tokenizer("/x/zhwiki-20260701-multistream.xml.bz2")).to eq("trigram")
38
+ end
39
+
40
+ it "picks unicode61 for space-delimited language dumps" do
41
+ expect(described_class.default_tokenizer("/x/enwiki-20260701-multistream.xml.bz2")).to eq("unicode61")
42
+ expect(described_class.default_tokenizer("/x/dewiki-20260701-multistream.xml.bz2")).to eq("unicode61")
43
+ end
44
+ end
45
+
46
+ context "with a built unicode61 index" do
47
+ around do |example|
48
+ Dir.mktmpdir do |dir|
49
+ @multistream_path, fts_db, meta_db = build_indexes(dir)
50
+ @fts = described_class.new(fts_db, meta_db)
51
+ example.run
52
+ @fts.close
53
+ end
54
+ end
55
+
56
+ it "is built, valid, and records the tokenizer" do
57
+ expect(@fts.built?).to be true
58
+ expect(@fts.valid_for?(@multistream_path)).to be true
59
+ expect(@fts.tokenizer).to eq("unicode61")
60
+ expect(@fts.stats[:section_count]).to be > 0
61
+ end
62
+
63
+ it "finds matches across articles with exact counts" do
64
+ result = @fts.search("Story", count: "exact")
65
+ expect(result[:total]).to eq(2)
66
+ expect(result[:total_is_capped]).to be false
67
+ expect(result[:hits].map { |h| h[:title] }).to contain_exactly("Film A", "Film B")
68
+ end
69
+
70
+ it "searches lead sections (empty heading)" do
71
+ result = @fts.search("Intro", count: "exact")
72
+ expect(result[:total]).to eq(2)
73
+ expect(result[:hits].map { |h| h[:heading] }.uniq).to eq([""])
74
+ end
75
+
76
+ it "excludes redirect pages from the index" do
77
+ result = @fts.search("Old Film", count: "exact")
78
+ expect(result[:total]).to eq(0)
79
+ end
80
+
81
+ it "returns zero for absent strings (absence claim)" do
82
+ expect(@fts.search("zebra unicorn", count: "exact")[:total]).to eq(0)
83
+ end
84
+
85
+ it "composes with an exact category filter" do
86
+ result = @fts.search("Story", category: "Japanese films", count: "exact")
87
+ expect(result[:hits].map { |h| h[:title] }).to eq(["Film A"])
88
+ end
89
+
90
+ it "composes with a recursive category filter" do
91
+ result = @fts.search("Story", category: "Films", depth: 1, count: "exact")
92
+ expect(result[:total]).to eq(2)
93
+ end
94
+
95
+ it "composes with a section filter" do
96
+ result = @fts.search("Story", sections: ["Plot"], count: "exact")
97
+ expect(result[:hits].map { |h| h[:heading] }).to eq(["Plot"])
98
+ end
99
+
100
+ it "normalizes decorated headings the same way as the metadata index" do
101
+ result = @fts.search("Distinct prose", count: "exact")
102
+ expect(result[:hits].map { |h| h[:heading] }).to eq(["Style"])
103
+ expect(@fts.search("Distinct prose", sections: ["Style"], count: "exact")[:total]).to eq(1)
104
+ end
105
+
106
+ it "keeps ord aligned between page_sections and fts_map (shared semantics)" do
107
+ meta_db = SQLite3::Database.new(@fts.meta_db_path, readonly: true)
108
+ meta_ord = meta_db.get_first_value(
109
+ "SELECT ps.ord FROM page_sections ps JOIN pages p ON p.page_id = ps.page_id " \
110
+ "WHERE p.title = 'Person X' AND ps.heading = 'Career'"
111
+ )
112
+ meta_db.close
113
+ fts_db = SQLite3::Database.new(@fts.db_path, readonly: true)
114
+ fts_ord = fts_db.get_first_value(
115
+ "SELECT fm.ord FROM fts_map fm WHERE fm.heading = 'Career'"
116
+ )
117
+ fts_db.close
118
+ expect(meta_ord).to eq(1) # lead = 0 (not stored), first heading = 1
119
+ expect(fts_ord).to eq(meta_ord)
120
+ end
121
+
122
+ it "caps counting when requested" do
123
+ result = @fts.search("Story", count: "capped", count_cap: 1)
124
+ expect(result[:total]).to eq(1)
125
+ expect(result[:total_is_capped]).to be true
126
+ end
127
+
128
+ it "supports raw FTS5 query mode" do
129
+ result = @fts.search("Story OR Acting", mode: "query", count: "exact")
130
+ expect(result[:total]).to eq(3)
131
+ end
132
+
133
+ it "escapes quotes in phrase mode" do
134
+ expect { @fts.search('say "hi" now', count: "exact") }.not_to raise_error
135
+ end
136
+ end
137
+
138
+ context "built with optimize: false" do
139
+ around do |example|
140
+ Dir.mktmpdir do |dir|
141
+ @multistream_path, fts_db, meta_db = build_indexes(dir, optimize: false)
142
+ @fts = described_class.new(fts_db, meta_db)
143
+ example.run
144
+ @fts.close
145
+ end
146
+ end
147
+
148
+ it "is valid, flagged unoptimized, and fully searchable" do
149
+ expect(@fts.built?).to be true
150
+ expect(@fts.valid_for?(@multistream_path)).to be true
151
+ expect(@fts.optimized?).to be false
152
+ expect(@fts.stats[:optimized]).to be false
153
+ expect(@fts.search("Story", count: "exact")[:total]).to eq(2)
154
+ end
155
+
156
+ it "can be optimized afterwards (idempotent)" do
157
+ expect(@fts.optimize!).to be true
158
+ expect(@fts.optimized?).to be true
159
+ expect(@fts.search("Story", count: "exact")[:total]).to eq(2)
160
+ expect(@fts.optimize!).to be true
161
+ end
162
+ end
163
+
164
+ context "with a trigram index" do
165
+ around do |example|
166
+ Dir.mktmpdir do |dir|
167
+ @multistream_path, fts_db, meta_db = build_indexes(dir, tokenizer: "trigram")
168
+ @fts = described_class.new(fts_db, meta_db)
169
+ example.run
170
+ @fts.close
171
+ end
172
+ end
173
+
174
+ it "matches substrings of three or more characters" do
175
+ result = @fts.search("tory", count: "exact")
176
+ expect(result[:total]).to eq(2)
177
+ end
178
+ end
179
+ end
180
+
181
+ describe "Corpus#search_text" do
182
+ around do |example|
183
+ Dir.mktmpdir do |dir|
184
+ @multistream_path, = build_indexes(dir)
185
+ @corpus = Wp2txt::Corpus.for_input(@multistream_path, cache_dir: dir)
186
+ example.run
187
+ @corpus.close
188
+ end
189
+ end
190
+
191
+ it "reports the fulltext tier in dump_info" do
192
+ info = @corpus.dump_info
193
+ expect(info[:tiers][:fulltext]).to be true
194
+ expect(info[:fulltext_current]).to be true
195
+ expect(info[:fulltext][:tokenizer]).to eq("unicode61")
196
+ end
197
+
198
+ it "exposes the fts tables to query_sql" do
199
+ result = @corpus.query_sql("SELECT COUNT(*) FROM fts.fts_map")
200
+ expect(result[:rows].first.first).to be > 0
201
+ expect(@corpus.describe_schema[:fts].join).to include("fts_map")
202
+ end
203
+
204
+ it "returns hits with section paths and dump identity" do
205
+ result = @corpus.search_text("Story", count: "exact")
206
+ expect(result[:dump]).to eq("testwiki-20260101")
207
+ expect(result[:total]).to eq(2)
208
+ paths = result[:hits].map { |h| h[:section_path] }
209
+ expect(paths).to contain_exactly("Film A > Plot", "Film B > Synopsis")
210
+ end
211
+
212
+ it "renders snippets containing the search term" do
213
+ result = @corpus.search_text("Story")
214
+ expect(result[:hits].first[:snippet]).to include("Story")
215
+ end
216
+
217
+ it "uses the article title as section_path for lead hits" do
218
+ result = @corpus.search_text("Intro", count: "exact")
219
+ expect(result[:hits].map { |h| h[:section_path] }).to contain_exactly("Film A", "Film B")
220
+ expect(result[:hits].first).not_to have_key(:section)
221
+ end
222
+
223
+ it "expands section filters through alias sets" do
224
+ @corpus.save_alias_set("plot", [%w[Plot Synopsis]], min_articles: 1)
225
+ result = @corpus.search_text("Story", sections: ["Plot"], alias_set: "plot", count: "exact")
226
+ expect(result[:total]).to eq(2)
227
+ end
228
+
229
+ it "raises a helpful error when the index is missing" do
230
+ Dir.mktmpdir do |dir2|
231
+ ms2, = create_fixture(dir2)
232
+ ms_index = Wp2txt::MultistreamIndex.new(
233
+ ms2.sub(/multistream\.xml\.bz2\z/, "multistream-index.txt"), use_cache: false, show_progress: false
234
+ )
235
+ Wp2txt::MetadataIndexBuilder.new(
236
+ ms2, ms_index.stream_offsets,
237
+ db_path: Wp2txt::MetadataIndex.path_for(ms2, cache_dir: dir2), num_processes: 0
238
+ ).build
239
+ corpus2 = Wp2txt::Corpus.for_input(ms2, cache_dir: dir2)
240
+ expect { corpus2.search_text("x") }.to raise_error(ArgumentError, /Full-text index not built/)
241
+ corpus2.close
242
+ end
243
+ end
244
+ end
245
+ end
@@ -0,0 +1,208 @@
1
+ # frozen_string_literal: true
2
+
3
+ require "spec_helper"
4
+ require "tmpdir"
5
+ require "fileutils"
6
+ require "open3"
7
+ require_relative "support/multistream_fixture"
8
+ require_relative "../lib/wp2txt/metadata_index"
9
+ require_relative "../lib/wp2txt/multistream"
10
+ require_relative "../lib/wp2txt/cli"
11
+
12
+ RSpec.describe "Wp2txt Metadata Index" do
13
+ include MultistreamFixture
14
+
15
+ describe Wp2txt::MetadataIndex do
16
+ describe ".normalize_category" do
17
+ it "replaces underscores, trims, and capitalizes the first letter" do
18
+ expect(described_class.normalize_category(" japanese_films ")).to eq("Japanese films")
19
+ end
20
+
21
+ it "leaves non-ASCII names unchanged" do
22
+ expect(described_class.normalize_category("日本の映画作品")).to eq("日本の映画作品")
23
+ end
24
+ end
25
+
26
+ describe ".clean_heading" do
27
+ it "strips bold markup and resolves links" do
28
+ expect(described_class.clean_heading("'''Bold''' [[link|Label]]")).to eq("Bold Label")
29
+ end
30
+
31
+ it "strips HTML tags" do
32
+ expect(described_class.clean_heading("<small>Notes</small>")).to eq("Notes")
33
+ end
34
+ end
35
+
36
+ describe ".expand_section_names" do
37
+ it "expands a canonical name to its alias group" do
38
+ expect(described_class.expand_section_names("Plot")).to include("Plot", "Synopsis")
39
+ end
40
+
41
+ it "expands an alias back to the full group (bidirectional)" do
42
+ expect(described_class.expand_section_names("Synopsis")).to include("Plot", "Synopsis")
43
+ end
44
+
45
+ it "returns the name itself when no alias group matches" do
46
+ expect(described_class.expand_section_names("Nonexistent Section")).to eq(["Nonexistent Section"])
47
+ end
48
+ end
49
+
50
+ describe ".path_for" do
51
+ it "builds a cache path keyed to the dump file" do
52
+ path = described_class.path_for("/dumps/jawiki-20260101-pages-articles-multistream.xml.bz2", cache_dir: "/cache")
53
+ expect(path).to start_with("/cache/jawiki-20260101-pages-articles-multistream")
54
+ expect(path).to end_with("_meta.sqlite3")
55
+ end
56
+ end
57
+ end
58
+
59
+ describe "build and query" do
60
+ around do |example|
61
+ Dir.mktmpdir do |dir|
62
+ @dir = dir
63
+ @multistream_path, @index_path = create_fixture(dir)
64
+ @db_path = File.join(dir, "meta.sqlite3")
65
+ ms_index = Wp2txt::MultistreamIndex.new(@index_path, use_cache: false, show_progress: false)
66
+ builder = Wp2txt::MetadataIndexBuilder.new(
67
+ @multistream_path, ms_index.stream_offsets,
68
+ db_path: @db_path, num_processes: 0
69
+ )
70
+ @index = builder.build
71
+ example.run
72
+ @index.close
73
+ end
74
+ end
75
+
76
+ it "records all pages and counts articles excluding redirects and category pages" do
77
+ stats = @index.stats
78
+ expect(stats[:page_count]).to eq(8)
79
+ expect(stats[:article_count]).to eq(3)
80
+ expect(stats[:dump_name]).to eq("testwiki-20260101")
81
+ end
82
+
83
+ it "finds articles by exact category" do
84
+ expect(@index.find_articles(category: "Japanese films")).to eq(["Film A"])
85
+ end
86
+
87
+ it "finds articles through subcategories with depth" do
88
+ expect(@index.find_articles(category: "Films", depth: 1)).to contain_exactly("Film A", "Film B")
89
+ end
90
+
91
+ it "does not include subcategory members at depth 0" do
92
+ expect(@index.find_articles(category: "Films", depth: 0)).to be_empty
93
+ end
94
+
95
+ it "matches sections through aliases" do
96
+ titles = @index.find_articles(category: "Films", depth: 1, has_section: "Plot")
97
+ expect(titles).to contain_exactly("Film A", "Film B")
98
+ end
99
+
100
+ it "matches sections exactly when aliases are disabled" do
101
+ titles = @index.find_articles(category: "Films", depth: 1, has_section: "Plot", use_aliases: false)
102
+ expect(titles).to eq(["Film A"])
103
+ end
104
+
105
+ it "excludes redirects from results" do
106
+ expect(@index.find_articles(title_match: "Old Film")).to be_empty
107
+ end
108
+
109
+ it "filters by title substring" do
110
+ expect(@index.find_articles(title_match: "Film")).to contain_exactly("Film A", "Film B")
111
+ end
112
+
113
+ it "applies limit and reports full count separately" do
114
+ titles = @index.find_articles(title_match: "Film", limit: 1)
115
+ expect(titles.size).to eq(1)
116
+ expect(@index.count_articles(title_match: "Film")).to eq(2)
117
+ end
118
+
119
+ it "returns the category tree with depths" do
120
+ tree = @index.category_tree("Films", depth: 1)
121
+ expect(tree).to include({ name: "Films", depth: 0 },
122
+ { name: "Japanese films", depth: 1 },
123
+ { name: "French films", depth: 1 })
124
+ end
125
+
126
+ it "collects section statistics" do
127
+ stats = @index.section_stats
128
+ expect(stats).to include(["Plot", 1], ["Synopsis", 1], ["Career", 1])
129
+ end
130
+
131
+ it "scopes section statistics to a category" do
132
+ stats = @index.section_stats(category: "Films", depth: 1)
133
+ headings = stats.map(&:first)
134
+ expect(headings).to include("Plot", "Synopsis")
135
+ expect(headings).not_to include("Career")
136
+ end
137
+
138
+ it "records redirect targets" do
139
+ db = SQLite3::Database.new(@db_path)
140
+ target = db.get_first_value("SELECT redirect_to FROM pages WHERE title = 'Old Film'")
141
+ db.close
142
+ expect(target).to eq("Film A")
143
+ end
144
+
145
+ it "validates against the source dump file" do
146
+ expect(@index.built?).to be true
147
+ expect(@index.valid_for?(@multistream_path)).to be true
148
+ end
149
+
150
+ it "records the wp2txt version it was built with" do
151
+ expect(@index.stats[:built_with]).to eq(Wp2txt::VERSION)
152
+ end
153
+
154
+ it "sees headings even when a trailing HTML comment follows the closing markers" do
155
+ expect(@index.find_articles(has_section: "Career", use_aliases: false)).to eq(["Person X"])
156
+ end
157
+
158
+ it "ignores commented-out category links" do
159
+ expect(@index.categories_of("Person X")).to eq(["Japanese actors"])
160
+ end
161
+
162
+ it "builds atomically: the existing index survives an unfinished rebuild" do
163
+ expect(@index.built?).to be true
164
+ rebuilding = Wp2txt::MetadataIndex.new(@db_path)
165
+ rebuilding.prepare_build!
166
+ rebuilding.insert_batch(pages: [[99, "Partial", 0, nil, 10]], categories: [], sections: [], hierarchy: [])
167
+ rebuilding.close # abandon before finalize
168
+
169
+ survivor = Wp2txt::MetadataIndex.new(@db_path)
170
+ expect(survivor.built?).to be true
171
+ expect(survivor.find_articles(title_match: "Film A")).to eq(["Film A"])
172
+ expect(survivor.find_articles(title_match: "Partial")).to be_empty
173
+ survivor.close
174
+ FileUtils.rm_f(Dir.glob("#{@db_path}.building*"))
175
+ end
176
+
177
+ it "detects a changed source dump file" do
178
+ File.binwrite(@multistream_path, File.binread(@multistream_path) + "x")
179
+ expect(@index.valid_for?(@multistream_path)).to be false
180
+ end
181
+
182
+ it "reports not built for a missing index file" do
183
+ missing = Wp2txt::MetadataIndex.new(File.join(@dir, "nope.sqlite3"))
184
+ expect(missing.built?).to be false
185
+ end
186
+ end
187
+
188
+ describe "CLI option validation" do
189
+ it "rejects --in-category without --find-articles" do
190
+ expect do
191
+ Wp2txt::CLI.parse_options(["-L", "ja", "--in-category", "Films"])
192
+ end.to raise_error(SystemExit)
193
+ end
194
+
195
+ it "rejects combining --build-index with --find-articles" do
196
+ expect do
197
+ Wp2txt::CLI.parse_options(["-L", "ja", "--build-index", "--find-articles"])
198
+ end.to raise_error(SystemExit)
199
+ end
200
+
201
+ it "accepts --find-articles with filters" do
202
+ opts = Wp2txt::CLI.parse_options(["-L", "ja", "--find-articles", "--in-category", "Films", "--has-section", "Plot"])
203
+ expect(opts[:find_articles]).to be true
204
+ expect(opts[:in_category]).to eq("Films")
205
+ expect(opts[:has_section]).to eq("Plot")
206
+ end
207
+ end
208
+ end
@@ -0,0 +1,66 @@
1
+ # frozen_string_literal: true
2
+
3
+ require "open3"
4
+
5
+ # Synthetic two-stream multistream dump used by metadata index / corpus specs.
6
+ # Stream 1 holds articles (ns 0), stream 2 holds category pages (ns 14).
7
+ module MultistreamFixture
8
+ def page_xml(id:, ns:, title:, text:)
9
+ escaped = text.gsub("&", "&amp;").gsub("<", "&lt;").gsub(">", "&gt;")
10
+ <<~XML
11
+ <page>
12
+ <title>#{title}</title>
13
+ <ns>#{ns}</ns>
14
+ <id>#{id}</id>
15
+ <revision>
16
+ <id>#{id * 100}</id>
17
+ <text bytes="#{text.bytesize}">#{escaped}</text>
18
+ </revision>
19
+ </page>
20
+ XML
21
+ end
22
+
23
+ def bzip2(data)
24
+ out, status = Open3.capture2("bzip2", "-c", stdin_data: data)
25
+ raise "bzip2 failed" unless status.success?
26
+
27
+ out
28
+ end
29
+
30
+ # @return [Array(String, String)] [multistream_path, index_path]
31
+ def create_fixture(dir)
32
+ stream1_pages = [
33
+ page_xml(id: 1, ns: 0, title: "Film A",
34
+ text: "Intro.\n== Plot ==\nStory here.\n== Reception ==\nGood.\n[[Category:Japanese films]]\n"),
35
+ page_xml(id: 2, ns: 0, title: "Film B",
36
+ text: "Intro.\n== Synopsis ==\nStory here.\n[[Category:French films|B]]\n"),
37
+ page_xml(id: 3, ns: 0, title: "Person X",
38
+ text: "Bio.\n== Career == <!-- legacy anchor -->\nActing.\n== '''Style''' ==\nDistinct prose.\n<!-- [[Category:Hidden]] -->\n[[Category:Japanese actors]]\n"),
39
+ page_xml(id: 4, ns: 0, title: "Old Film",
40
+ text: "#REDIRECT [[Film A]]\n[[Category:Japanese films]]\n")
41
+ ]
42
+ stream2_pages = [
43
+ page_xml(id: 5, ns: 14, title: "Category:Japanese films", text: "[[Category:Films]]\n"),
44
+ page_xml(id: 6, ns: 14, title: "Category:French films", text: "[[Category:Films]]\n"),
45
+ page_xml(id: 7, ns: 14, title: "Category:Films", text: "Top category.\n"),
46
+ page_xml(id: 8, ns: 14, title: "Category:Japanese actors", text: "[[Category:People]]\n")
47
+ ]
48
+
49
+ stream1 = bzip2(stream1_pages.join)
50
+ stream2 = bzip2(stream2_pages.join)
51
+
52
+ multistream_path = File.join(dir, "testwiki-20260101-pages-articles-multistream.xml.bz2")
53
+ File.binwrite(multistream_path, stream1 + stream2)
54
+
55
+ offset2 = stream1.bytesize
56
+ index_lines = [
57
+ "0:1:Film A", "0:2:Film B", "0:3:Person X", "0:4:Old Film",
58
+ "#{offset2}:5:Category:Japanese films", "#{offset2}:6:Category:French films",
59
+ "#{offset2}:7:Category:Films", "#{offset2}:8:Category:Japanese actors"
60
+ ]
61
+ index_path = File.join(dir, "testwiki-20260101-pages-articles-multistream-index.txt")
62
+ File.write(index_path, index_lines.join("\n") + "\n")
63
+
64
+ [multistream_path, index_path]
65
+ end
66
+ end
metadata CHANGED
@@ -1,7 +1,7 @@
1
1
  --- !ruby/object:Gem::Specification
2
2
  name: wp2txt
3
3
  version: !ruby/object:Gem::Version
4
- version: 2.1.1
4
+ version: 2.2.0
5
5
  platform: ruby
6
6
  authors:
7
7
  - Yoichiro Hasebe
@@ -197,10 +197,12 @@ email:
197
197
  - yohasebe@gmail.com
198
198
  executables:
199
199
  - wp2txt
200
+ - wp2txt-mcp
200
201
  extensions: []
201
202
  extra_rdoc_files: []
202
203
  files:
203
204
  - ".dockerignore"
205
+ - ".github/FUNDING.yml"
204
206
  - ".github/workflows/ci.yml"
205
207
  - ".gitignore"
206
208
  - ".solargraph.yml"
@@ -214,6 +216,7 @@ files:
214
216
  - README_ja.md
215
217
  - Rakefile
216
218
  - bin/wp2txt
219
+ - bin/wp2txt-mcp
217
220
  - image/wp2txt-logo.svg
218
221
  - image/wp2txt.svg
219
222
  - lib/wp2txt.rb
@@ -224,6 +227,8 @@ files:
224
227
  - lib/wp2txt/cli_ui.rb
225
228
  - lib/wp2txt/config.rb
226
229
  - lib/wp2txt/constants.rb
230
+ - lib/wp2txt/corpus.rb
231
+ - lib/wp2txt/corpus_jobs.rb
227
232
  - lib/wp2txt/data/html_entities.json
228
233
  - lib/wp2txt/data/language_metadata.json
229
234
  - lib/wp2txt/data/language_tiers.json
@@ -233,10 +238,13 @@ files:
233
238
  - lib/wp2txt/extractor.rb
234
239
  - lib/wp2txt/file_utils.rb
235
240
  - lib/wp2txt/formatter.rb
241
+ - lib/wp2txt/fts_index.rb
236
242
  - lib/wp2txt/global_data_cache.rb
237
243
  - lib/wp2txt/index_cache.rb
244
+ - lib/wp2txt/index_commands.rb
238
245
  - lib/wp2txt/magic_words.rb
239
246
  - lib/wp2txt/memory_monitor.rb
247
+ - lib/wp2txt/metadata_index.rb
240
248
  - lib/wp2txt/multistream.rb
241
249
  - lib/wp2txt/output_writer.rb
242
250
  - lib/wp2txt/parser_functions.rb
@@ -264,15 +272,18 @@ files:
264
272
  - spec/cli_spec.rb
265
273
  - spec/config_spec.rb
266
274
  - spec/constants_spec.rb
275
+ - spec/corpus_spec.rb
267
276
  - spec/file_utils_spec.rb
268
277
  - spec/fixtures/samples.rb
269
278
  - spec/formatter_sections_spec.rb
279
+ - spec/fts_index_spec.rb
270
280
  - spec/global_data_cache_spec.rb
271
281
  - spec/index_cache_spec.rb
272
282
  - spec/integration_spec.rb
273
283
  - spec/magic_words_spec.rb
274
284
  - spec/markers_spec.rb
275
285
  - spec/memory_monitor_spec.rb
286
+ - spec/metadata_index_spec.rb
276
287
  - spec/multistream_spec.rb
277
288
  - spec/output_writer_spec.rb
278
289
  - spec/parser_functions_spec.rb
@@ -281,6 +292,7 @@ files:
281
292
  - spec/section_extractor_spec.rb
282
293
  - spec/spec_helper.rb
283
294
  - spec/stream_processor_spec.rb
295
+ - spec/support/multistream_fixture.rb
284
296
  - spec/template_data_spec.rb
285
297
  - spec/template_expander_spec.rb
286
298
  - spec/template_processing_spec.rb
@@ -321,15 +333,18 @@ test_files:
321
333
  - spec/cli_spec.rb
322
334
  - spec/config_spec.rb
323
335
  - spec/constants_spec.rb
336
+ - spec/corpus_spec.rb
324
337
  - spec/file_utils_spec.rb
325
338
  - spec/fixtures/samples.rb
326
339
  - spec/formatter_sections_spec.rb
340
+ - spec/fts_index_spec.rb
327
341
  - spec/global_data_cache_spec.rb
328
342
  - spec/index_cache_spec.rb
329
343
  - spec/integration_spec.rb
330
344
  - spec/magic_words_spec.rb
331
345
  - spec/markers_spec.rb
332
346
  - spec/memory_monitor_spec.rb
347
+ - spec/metadata_index_spec.rb
333
348
  - spec/multistream_spec.rb
334
349
  - spec/output_writer_spec.rb
335
350
  - spec/parser_functions_spec.rb
@@ -338,6 +353,7 @@ test_files:
338
353
  - spec/section_extractor_spec.rb
339
354
  - spec/spec_helper.rb
340
355
  - spec/stream_processor_spec.rb
356
+ - spec/support/multistream_fixture.rb
341
357
  - spec/template_data_spec.rb
342
358
  - spec/template_expander_spec.rb
343
359
  - spec/template_processing_spec.rb