twfilter 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +7 -0
- data/CHANGELOG.md +37 -0
- data/LICENSE +19 -0
- data/NOTICES.md +93 -0
- data/README.md +364 -0
- data/lib/twfilter/block.rb +36 -0
- data/lib/twfilter/checks/erhua.rb +40 -0
- data/lib/twfilter/checks/lexicon.rb +83 -0
- data/lib/twfilter/checks/script.rb +71 -0
- data/lib/twfilter/checks/shape.rb +40 -0
- data/lib/twfilter/checks/wenyan.rb +32 -0
- data/lib/twfilter/data/MANIFEST.json +85 -0
- data/lib/twfilter/data/cantonese.txt +17 -0
- data/lib/twfilter/data/converted_orthography.txt +14 -0
- data/lib/twfilter/data/erhua_headed.txt +14 -0
- data/lib/twfilter/data/erhua_tailed.txt +36 -0
- data/lib/twfilter/data/foreign_topics.txt +80 -0
- data/lib/twfilter/data/mainland_exceptions.tsv +6 -0
- data/lib/twfilter/data/mainland_hard.tsv +149 -0
- data/lib/twfilter/data/mainland_soft.tsv +38 -0
- data/lib/twfilter/data/moe_common.txt +4808 -0
- data/lib/twfilter/data/moe_exception.txt +12 -0
- data/lib/twfilter/data/moe_rare.txt +18356 -0
- data/lib/twfilter/data/moe_secondary.txt +6343 -0
- data/lib/twfilter/data/regional_hard.tsv +19 -0
- data/lib/twfilter/data/simplified_only.txt +3785 -0
- data/lib/twfilter/data/taiwan_grammar.txt +20 -0
- data/lib/twfilter/data/taiwan_lexicon.txt +51 -0
- data/lib/twfilter/data/taiwan_markers.txt +66 -0
- data/lib/twfilter/data/taiwan_particles.txt +7 -0
- data/lib/twfilter/data/variants_used_in_taiwan.txt +16 -0
- data/lib/twfilter/data/wenyan.txt +4 -0
- data/lib/twfilter/errors.rb +12 -0
- data/lib/twfilter/evidence.rb +44 -0
- data/lib/twfilter/finding.rb +42 -0
- data/lib/twfilter/han.rb +30 -0
- data/lib/twfilter/policy.rb +68 -0
- data/lib/twfilter/punctuation.rb +106 -0
- data/lib/twfilter/sentences.rb +32 -0
- data/lib/twfilter/subject.rb +28 -0
- data/lib/twfilter/tables.rb +140 -0
- data/lib/twfilter/version.rb +5 -0
- data/lib/twfilter.rb +57 -0
- data/licenses/APACHE-2.0.txt +202 -0
- data/sig/twfilter.rbs +215 -0
- metadata +137 -0
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
伺服器
|
|
2
|
+
便當
|
|
3
|
+
光碟
|
|
4
|
+
公車
|
|
5
|
+
冷氣
|
|
6
|
+
列印
|
|
7
|
+
叫車
|
|
8
|
+
品質
|
|
9
|
+
宵夜
|
|
10
|
+
寬頻
|
|
11
|
+
寮國
|
|
12
|
+
專案
|
|
13
|
+
幼稚園
|
|
14
|
+
影印
|
|
15
|
+
影片
|
|
16
|
+
捷運
|
|
17
|
+
早安
|
|
18
|
+
智慧型手機
|
|
19
|
+
桌上型電腦
|
|
20
|
+
機車
|
|
21
|
+
水準
|
|
22
|
+
泡麵
|
|
23
|
+
滑鼠
|
|
24
|
+
硬碟
|
|
25
|
+
硬體
|
|
26
|
+
程式
|
|
27
|
+
筆記型電腦
|
|
28
|
+
管道
|
|
29
|
+
簡訊
|
|
30
|
+
紐西蘭
|
|
31
|
+
網路
|
|
32
|
+
網際網路
|
|
33
|
+
義大利
|
|
34
|
+
腳踏車
|
|
35
|
+
螢幕
|
|
36
|
+
行動電話
|
|
37
|
+
視訊
|
|
38
|
+
計程車
|
|
39
|
+
記憶體
|
|
40
|
+
貓熊
|
|
41
|
+
資料庫
|
|
42
|
+
資訊
|
|
43
|
+
超商
|
|
44
|
+
身分證
|
|
45
|
+
軟體
|
|
46
|
+
部落格
|
|
47
|
+
錄影
|
|
48
|
+
隨身碟
|
|
49
|
+
雪梨
|
|
50
|
+
雷射
|
|
51
|
+
鳳梨
|
|
@@ -0,0 +1,66 @@
|
|
|
1
|
+
中華民國
|
|
2
|
+
交通部
|
|
3
|
+
便利商店
|
|
4
|
+
健保
|
|
5
|
+
僑委會
|
|
6
|
+
內政部
|
|
7
|
+
勞保
|
|
8
|
+
南投
|
|
9
|
+
台中
|
|
10
|
+
台北
|
|
11
|
+
台南
|
|
12
|
+
台東
|
|
13
|
+
台灣
|
|
14
|
+
台鐵
|
|
15
|
+
司法院
|
|
16
|
+
嘉義
|
|
17
|
+
國中
|
|
18
|
+
國小
|
|
19
|
+
國防部
|
|
20
|
+
基隆
|
|
21
|
+
學測
|
|
22
|
+
宜蘭
|
|
23
|
+
屏東
|
|
24
|
+
彰化
|
|
25
|
+
悠遊卡
|
|
26
|
+
戶政
|
|
27
|
+
技專校院
|
|
28
|
+
指考
|
|
29
|
+
教育部
|
|
30
|
+
文化部
|
|
31
|
+
新北
|
|
32
|
+
新台幣
|
|
33
|
+
新竹
|
|
34
|
+
新臺幣
|
|
35
|
+
會考
|
|
36
|
+
桃園
|
|
37
|
+
民國
|
|
38
|
+
澎湖
|
|
39
|
+
環保署
|
|
40
|
+
監察院
|
|
41
|
+
立法院
|
|
42
|
+
統一發票
|
|
43
|
+
統測
|
|
44
|
+
經濟部
|
|
45
|
+
縣市政府
|
|
46
|
+
總統府
|
|
47
|
+
考試院
|
|
48
|
+
臺中
|
|
49
|
+
臺北
|
|
50
|
+
臺南
|
|
51
|
+
臺東
|
|
52
|
+
臺灣
|
|
53
|
+
花蓮
|
|
54
|
+
苗栗
|
|
55
|
+
行政院
|
|
56
|
+
衛福部
|
|
57
|
+
農委會
|
|
58
|
+
鄉鎮市
|
|
59
|
+
里長
|
|
60
|
+
金管會
|
|
61
|
+
金門
|
|
62
|
+
陸委會
|
|
63
|
+
雲林
|
|
64
|
+
馬祖
|
|
65
|
+
高職
|
|
66
|
+
高雄
|
|
@@ -0,0 +1,44 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module TWFilter
|
|
4
|
+
module Evidence
|
|
5
|
+
# 沒有 V and 有沒有 V are pan-Mandarin negation and A-not-A, not the Southern Min substrate
|
|
6
|
+
# perfective. One lookbehind covers both: in 有沒有看 the 有看 match is also preceded by 沒.
|
|
7
|
+
NEGATED = "沒"
|
|
8
|
+
|
|
9
|
+
class << self
|
|
10
|
+
def count(text) = scanners.sum { |scanner| text.scan(scanner).length }
|
|
11
|
+
|
|
12
|
+
def per_100(text)
|
|
13
|
+
han = Han.count(text)
|
|
14
|
+
han.zero? ? 0.0 : count(text) * 100.0 / han
|
|
15
|
+
end
|
|
16
|
+
|
|
17
|
+
def hits(text) = terms.zip(scanners).filter_map { |term, scanner| term if found?(text, scanner) }
|
|
18
|
+
|
|
19
|
+
def markers(text) = Tables.rows("taiwan_markers.txt").select { |term| text.include?(term) }
|
|
20
|
+
|
|
21
|
+
def reset! = @terms = @scanners = nil
|
|
22
|
+
|
|
23
|
+
def terms
|
|
24
|
+
@terms ||= (Tables.rows("taiwan_lexicon.txt") +
|
|
25
|
+
Tables.rows("taiwan_particles.txt") +
|
|
26
|
+
Tables.rows("taiwan_markers.txt") +
|
|
27
|
+
Tables.rows("taiwan_grammar.txt"))
|
|
28
|
+
.uniq
|
|
29
|
+
.freeze
|
|
30
|
+
end
|
|
31
|
+
|
|
32
|
+
# A plain string scans faster than an equivalent regexp; only 有-initial terms need one.
|
|
33
|
+
def scanners
|
|
34
|
+
@scanners ||= terms.map { |term|
|
|
35
|
+
term.start_with?("有") ? /(?<!#{NEGATED})#{Regexp.escape(term)}/ : term
|
|
36
|
+
}.freeze
|
|
37
|
+
end
|
|
38
|
+
|
|
39
|
+
private
|
|
40
|
+
|
|
41
|
+
def found?(text, scanner) = scanner.is_a?(Regexp) ? scanner.match?(text) : text.include?(scanner)
|
|
42
|
+
end
|
|
43
|
+
end
|
|
44
|
+
end
|
|
@@ -0,0 +1,42 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module TWFilter
|
|
4
|
+
# Offending items quoted in a finding detail. The check fires on all of them; this only
|
|
5
|
+
# bounds log width.
|
|
6
|
+
DETAIL_SAMPLE = 4
|
|
7
|
+
|
|
8
|
+
Finding = Data.define(:check, :code, :severity, :detail) do
|
|
9
|
+
def initialize(check:, code:, severity: :reject, detail: nil)
|
|
10
|
+
super
|
|
11
|
+
end
|
|
12
|
+
|
|
13
|
+
def reject? = severity == :reject
|
|
14
|
+
|
|
15
|
+
def mark? = severity == :mark
|
|
16
|
+
|
|
17
|
+
def to_h = {check: check.to_s, code: code.to_s, severity: severity.to_s, detail: detail}.compact
|
|
18
|
+
end
|
|
19
|
+
|
|
20
|
+
Report = Data.define(:text, :findings, :tier, :han, :evidence) do
|
|
21
|
+
def ok? = findings.none?(&:reject?)
|
|
22
|
+
|
|
23
|
+
def rejects = findings.select(&:reject?)
|
|
24
|
+
|
|
25
|
+
def marks = findings.select(&:mark?)
|
|
26
|
+
|
|
27
|
+
def codes = findings.map(&:code)
|
|
28
|
+
|
|
29
|
+
def reasons = rejects.map { |finding| [finding.code, finding.detail].compact.join(": ") }
|
|
30
|
+
|
|
31
|
+
def to_h
|
|
32
|
+
{
|
|
33
|
+
text: text,
|
|
34
|
+
ok: ok?,
|
|
35
|
+
tier: tier,
|
|
36
|
+
han: han,
|
|
37
|
+
evidence: evidence,
|
|
38
|
+
findings: findings.map(&:to_h)
|
|
39
|
+
}
|
|
40
|
+
end
|
|
41
|
+
end
|
|
42
|
+
end
|
data/lib/twfilter/han.rb
ADDED
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module TWFilter
|
|
4
|
+
module Han
|
|
5
|
+
CHAR = /\p{Han}/
|
|
6
|
+
RUN = /\p{Han}+/
|
|
7
|
+
# CJK Unified Ideographs and Extension A — the planes a Taiwanese font is expected to
|
|
8
|
+
# render. \p{Han} is wider; .basic? distinguishes the two.
|
|
9
|
+
BMP = (0x4E00..0x9FFF).freeze
|
|
10
|
+
EXTENSION_A = (0x3400..0x4DBF).freeze
|
|
11
|
+
|
|
12
|
+
module_function
|
|
13
|
+
|
|
14
|
+
def char?(char) = char.match?(CHAR)
|
|
15
|
+
|
|
16
|
+
def count(text) = text.each_char.count { |char| char.match?(CHAR) }
|
|
17
|
+
|
|
18
|
+
def runs(text) = text.scan(RUN)
|
|
19
|
+
|
|
20
|
+
def ratio(text)
|
|
21
|
+
length = text.length
|
|
22
|
+
length.zero? ? 0.0 : count(text).fdiv(length)
|
|
23
|
+
end
|
|
24
|
+
|
|
25
|
+
def basic?(char)
|
|
26
|
+
code = char.ord
|
|
27
|
+
BMP.cover?(code) || EXTENSION_A.cover?(code)
|
|
28
|
+
end
|
|
29
|
+
end
|
|
30
|
+
end
|
|
@@ -0,0 +1,68 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module TWFilter
|
|
4
|
+
# 教育部 chart indices, ordered by decreasing frequency of use.
|
|
5
|
+
TIERS = {common: 0, secondary: 1, rare: 2}.freeze
|
|
6
|
+
|
|
7
|
+
Policy = Data.define(
|
|
8
|
+
:han_range,
|
|
9
|
+
:min_han_ratio,
|
|
10
|
+
:max_tier,
|
|
11
|
+
:allow_latin,
|
|
12
|
+
:orthography_rejects,
|
|
13
|
+
:foreign_topics_reject,
|
|
14
|
+
:punctuation,
|
|
15
|
+
:soft_lexicon_rejects,
|
|
16
|
+
:wenyan_min_han,
|
|
17
|
+
:wenyan_min_hits,
|
|
18
|
+
:wenyan_max_density,
|
|
19
|
+
:block_tolerance,
|
|
20
|
+
:evidence_per_100
|
|
21
|
+
) do
|
|
22
|
+
# Defaults impose no shape constraint. Literary-Chinese density fires only above 8 Han
|
|
23
|
+
# characters and 2 particle hits, since the ratio is unstable on shorter spans; 0.05 is
|
|
24
|
+
# the observed ceiling for modern prose quoting classical text. Block tolerance 0.0
|
|
25
|
+
# rejects a window on one failing member; 1.0 evidence per 100 sentences is the floor
|
|
26
|
+
# below which a window carries no Taiwan-specific signal.
|
|
27
|
+
def initialize(
|
|
28
|
+
han_range: (1..Float::INFINITY),
|
|
29
|
+
min_han_ratio: 0.0,
|
|
30
|
+
max_tier: :rare,
|
|
31
|
+
allow_latin: true,
|
|
32
|
+
orthography_rejects: true,
|
|
33
|
+
foreign_topics_reject: false,
|
|
34
|
+
punctuation: :wide,
|
|
35
|
+
soft_lexicon_rejects: false,
|
|
36
|
+
wenyan_min_han: 8,
|
|
37
|
+
wenyan_min_hits: 2,
|
|
38
|
+
wenyan_max_density: 0.05,
|
|
39
|
+
block_tolerance: 0.0,
|
|
40
|
+
evidence_per_100: 1.0
|
|
41
|
+
)
|
|
42
|
+
super
|
|
43
|
+
end
|
|
44
|
+
|
|
45
|
+
def tier_limit = TIERS.fetch(max_tier)
|
|
46
|
+
|
|
47
|
+
class << self
|
|
48
|
+
# 6..60 Han characters spans a usable sentence; 0.65 Han excludes list and table
|
|
49
|
+
# fragments while admitting embedded Latin and numerals.
|
|
50
|
+
def corpus = new(han_range: (6..60), min_han_ratio: 0.65)
|
|
51
|
+
|
|
52
|
+
# Tighter bounds for material shown to learners: 40 Han is a readable card, 0.70
|
|
53
|
+
# suppresses code-mixed text, and the 常用 chart is the pedagogical inventory.
|
|
54
|
+
def publishable
|
|
55
|
+
new(
|
|
56
|
+
han_range: (6..40),
|
|
57
|
+
min_han_ratio: 0.70,
|
|
58
|
+
max_tier: :common,
|
|
59
|
+
soft_lexicon_rejects: true,
|
|
60
|
+
punctuation: :strict,
|
|
61
|
+
foreign_topics_reject: true
|
|
62
|
+
)
|
|
63
|
+
end
|
|
64
|
+
|
|
65
|
+
def permissive = new
|
|
66
|
+
end
|
|
67
|
+
end
|
|
68
|
+
end
|
|
@@ -0,0 +1,106 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module TWFilter
|
|
4
|
+
module Punctuation
|
|
5
|
+
TAIWAN = ",。、;:?!「」『』()〔〕【】《》〈〉──……‧—".chars.to_set.freeze
|
|
6
|
+
LATIN = ".,;:?!()[]{}\"'-/%&+=*\#@_<>|$^`\\".chars.to_set.freeze
|
|
7
|
+
SPACE = " \t\u{3000}\u{00A0}".chars.to_set.freeze
|
|
8
|
+
# STRICT is the 教育部 handbook inventory; WIDE adds marks in common Taiwanese use but outside it.
|
|
9
|
+
STRICT = (TAIWAN | LATIN | SPACE | "…⋯.·─".chars.to_set).freeze
|
|
10
|
+
WIDE = (STRICT | "~~℃°※○●–→§±×÷№‼⁉".chars.to_set).freeze
|
|
11
|
+
SETS = {strict: STRICT, wide: WIDE}.freeze
|
|
12
|
+
ALLOWED = WIDE
|
|
13
|
+
|
|
14
|
+
# Bopomofo and its extended block: the Taiwanese phonetic script, admitted as text.
|
|
15
|
+
BOPOMOFO = [0x3100..0x312F, 0x31A0..0x31BF].freeze
|
|
16
|
+
TONE_MARKS = "\u{02CA}\u{02C7}\u{02CB}\u{02D9}\u{02C9}".chars.to_set.freeze
|
|
17
|
+
INVISIBLE = /[\u{200B}-\u{200F}\u{2028}\u{2029}\u{FEFF}\u{FE00}-\u{FE0F}\u{00AD}]/
|
|
18
|
+
|
|
19
|
+
WIDTH = {
|
|
20
|
+
""" => "\"",
|
|
21
|
+
"#" => "#",
|
|
22
|
+
"$" => "$",
|
|
23
|
+
"%" => "%",
|
|
24
|
+
"&" => "&",
|
|
25
|
+
"'" => "'",
|
|
26
|
+
"*" => "*",
|
|
27
|
+
"+" => "+",
|
|
28
|
+
"-" => "-",
|
|
29
|
+
"." => ".",
|
|
30
|
+
"/" => "/",
|
|
31
|
+
"<" => "<",
|
|
32
|
+
"=" => "=",
|
|
33
|
+
">" => ">",
|
|
34
|
+
"@" => "@",
|
|
35
|
+
"[" => "[",
|
|
36
|
+
"\" => "\\",
|
|
37
|
+
"]" => "]",
|
|
38
|
+
"^" => "^",
|
|
39
|
+
"_" => "_",
|
|
40
|
+
"`" => "`",
|
|
41
|
+
"{" => "{",
|
|
42
|
+
"|" => "|",
|
|
43
|
+
"}" => "}"
|
|
44
|
+
}.freeze
|
|
45
|
+
|
|
46
|
+
RULES = [
|
|
47
|
+
[INVISIBLE, ""],
|
|
48
|
+
[/[#{Regexp.escape(WIDTH.keys.join)}]/, WIDTH],
|
|
49
|
+
# Fullwidth digits and Latin letters fold to ASCII; 0xFEE0 is the fixed offset between the blocks.
|
|
50
|
+
[/[\u{FF10}-\u{FF19}\u{FF21}-\u{FF3A}\u{FF41}-\u{FF5A}]/, -> (m) { (m.ord - 0xFEE0).chr(Encoding::UTF_8) }],
|
|
51
|
+
[/[\u{201C}\u{301D}]/, "「"],
|
|
52
|
+
[/[\u{201D}\u{301E}\u{301F}]/, "」"],
|
|
53
|
+
[/\u{2018}(?=[^\u{2019}]*\p{Han})/, "『"],
|
|
54
|
+
[/(?<=\p{Han})\u{2019}/, "』"],
|
|
55
|
+
[/\u{FE50}/, ","],
|
|
56
|
+
[/\u{FE55}|\u{FE30}|\u{FE13}/, ":"],
|
|
57
|
+
[/[\u{30FB}\u{FF65}]/, "\u{2027}"],
|
|
58
|
+
[/(?<=\p{Han})[\u{00B7}\u{2022}\u{2219}]|[\u{00B7}\u{2022}\u{2219}](?=\p{Han})/, "\u{2027}"],
|
|
59
|
+
[/(?:\u{2026}|\u{22EF}|。。|\.{3,}|・{3,}){1,}/, "\u{2026}\u{2026}"],
|
|
60
|
+
[/[\u{2014}\u{2015}\u{2500}]{1,}|(?<=\p{Han})--(?=\p{Han})/, "\u{2500}\u{2500}"],
|
|
61
|
+
["\u{301C}", "~"],
|
|
62
|
+
[/(?<=\p{Han}),(?=\s|\p{Han}|\z)/, ","],
|
|
63
|
+
[/(?<=\p{Han})\.(?=\s|\p{Han}|\z)/, "。"],
|
|
64
|
+
[/(?<=\p{Han})\?(?=\s|\p{Han}|\z)/, "?"],
|
|
65
|
+
[/(?<=\p{Han})!(?=\s|\p{Han}|\z)/, "!"],
|
|
66
|
+
[/(?<=\p{Han});(?=\s|\p{Han}|\z)/, ";"],
|
|
67
|
+
[/(?<=\p{Han}):(?=\s|\p{Han}|\z)/, ":"],
|
|
68
|
+
[/(?<=\p{Han})\s+(?=\p{Han})/, ""],
|
|
69
|
+
[/\s+(?=[,。、;:?!」』)〕】》〉])/, ""],
|
|
70
|
+
[/(?<=[,。、;:?!「『(〔【《〈])\s+/, ""],
|
|
71
|
+
[/[ \t\u{00A0}]{2,}/, " "]
|
|
72
|
+
].freeze
|
|
73
|
+
|
|
74
|
+
COLLAPSE = /([,。!?;:])\1+/
|
|
75
|
+
|
|
76
|
+
module_function
|
|
77
|
+
|
|
78
|
+
def normalize(text, collapse_repeats: true)
|
|
79
|
+
value = RULES.reduce(text.to_s.unicode_normalize(:nfc)) { |memo, (pattern, replacement)|
|
|
80
|
+
case replacement
|
|
81
|
+
in Proc
|
|
82
|
+
memo.gsub(pattern) { replacement.call(Regexp.last_match(0)) }
|
|
83
|
+
in Hash
|
|
84
|
+
memo.gsub(pattern, replacement)
|
|
85
|
+
in String
|
|
86
|
+
memo.gsub(pattern, replacement)
|
|
87
|
+
end
|
|
88
|
+
}
|
|
89
|
+
value = value.gsub(COLLAPSE, "\\1") if collapse_repeats
|
|
90
|
+
value.strip
|
|
91
|
+
end
|
|
92
|
+
|
|
93
|
+
def acceptable?(char, set: WIDE)
|
|
94
|
+
return true if Han.char?(char) || set.include?(char) || TONE_MARKS.include?(char)
|
|
95
|
+
return true if char.match?(/[A-Za-z0-9]/)
|
|
96
|
+
|
|
97
|
+
code = char.ord
|
|
98
|
+
BOPOMOFO.any? { |range| range.cover?(code) }
|
|
99
|
+
end
|
|
100
|
+
|
|
101
|
+
def offenders(text, punctuation: :wide)
|
|
102
|
+
set = SETS.fetch(punctuation)
|
|
103
|
+
text.each_char.reject { |char| acceptable?(char, set: set) }.uniq
|
|
104
|
+
end
|
|
105
|
+
end
|
|
106
|
+
end
|
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module TWFilter
|
|
4
|
+
module Sentences
|
|
5
|
+
# A boundary is a run of terminators followed by any closing brackets; 、 is not a boundary.
|
|
6
|
+
TERMINATORS = "。!?…‼⁉"
|
|
7
|
+
CLAUSE = ";"
|
|
8
|
+
CLOSERS = "」』)〕】》〉\u{201D}\u{2019}"
|
|
9
|
+
|
|
10
|
+
module_function
|
|
11
|
+
|
|
12
|
+
PIECES = {
|
|
13
|
+
true => /[^#{TERMINATORS}#{CLAUSE}]*[#{TERMINATORS}#{CLAUSE}]+[#{CLOSERS}]*|[^#{TERMINATORS}#{CLAUSE}]+/,
|
|
14
|
+
false => /[^#{TERMINATORS}]*[#{TERMINATORS}]+[#{CLOSERS}]*|[^#{TERMINATORS}]+/
|
|
15
|
+
}.freeze
|
|
16
|
+
|
|
17
|
+
def split(text, clause: true)
|
|
18
|
+
text
|
|
19
|
+
.to_s
|
|
20
|
+
.scan(PIECES.fetch(clause))
|
|
21
|
+
.map { |piece| piece.gsub(/\s+/, " ").strip }
|
|
22
|
+
.reject(&:empty?)
|
|
23
|
+
end
|
|
24
|
+
|
|
25
|
+
def terminal?(text) = TERMINATORS.chars.any? { |mark| text.end_with?(mark) }
|
|
26
|
+
|
|
27
|
+
def shaped?(text, policy: Policy.corpus)
|
|
28
|
+
han = Han.count(text)
|
|
29
|
+
policy.han_range.cover?(han) && Han.ratio(text) >= policy.min_han_ratio
|
|
30
|
+
end
|
|
31
|
+
end
|
|
32
|
+
end
|
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module TWFilter
|
|
4
|
+
class Subject
|
|
5
|
+
attr_reader :text, :policy
|
|
6
|
+
|
|
7
|
+
def initialize(text, policy: Policy.corpus)
|
|
8
|
+
@text = text.to_s
|
|
9
|
+
@policy = policy
|
|
10
|
+
end
|
|
11
|
+
|
|
12
|
+
def chars = @chars ||= text.chars
|
|
13
|
+
|
|
14
|
+
def han_chars = @han_chars ||= chars.select { |char| Han.char?(char) }
|
|
15
|
+
|
|
16
|
+
def han = @han ||= han_chars.length
|
|
17
|
+
|
|
18
|
+
def ratio = @ratio ||= text.empty? ? 0.0 : han.fdiv(text.length)
|
|
19
|
+
|
|
20
|
+
def empty? = text.strip.empty?
|
|
21
|
+
|
|
22
|
+
def include?(term) = text.include?(term)
|
|
23
|
+
|
|
24
|
+
def masked(exceptions)
|
|
25
|
+
exceptions.reduce(text) { |memo, allowed| memo.gsub(allowed, "") }
|
|
26
|
+
end
|
|
27
|
+
end
|
|
28
|
+
end
|
|
@@ -0,0 +1,140 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "digest"
|
|
4
|
+
require "json"
|
|
5
|
+
require "pathname"
|
|
6
|
+
require "set"
|
|
7
|
+
|
|
8
|
+
module TWFilter
|
|
9
|
+
# Reference tables. The bundled copy is the source of truth for behavior and is versioned
|
|
10
|
+
# with the gem; .dir replaces it wholesale and .overlay supplements it. Any deviation from
|
|
11
|
+
# the bundled copy changes what "Taiwan Mandarin" means, so both are reported by .provenance.
|
|
12
|
+
module Tables
|
|
13
|
+
BUNDLED = Pathname(__dir__).join("data").freeze
|
|
14
|
+
MANIFEST = "MANIFEST.json"
|
|
15
|
+
ADD = ".add"
|
|
16
|
+
REMOVE = ".remove"
|
|
17
|
+
|
|
18
|
+
# Row shape per table; unknown names default to :terms so user tables are accepted.
|
|
19
|
+
SCHEMA = {
|
|
20
|
+
"moe_common.txt" => :chars,
|
|
21
|
+
"moe_secondary.txt" => :chars,
|
|
22
|
+
"moe_rare.txt" => :chars,
|
|
23
|
+
"moe_exception.txt" => :chars,
|
|
24
|
+
"simplified_only.txt" => :chars,
|
|
25
|
+
"variants_used_in_taiwan.txt" => :chars,
|
|
26
|
+
"converted_orthography.txt" => :chars,
|
|
27
|
+
"cantonese.txt" => :chars,
|
|
28
|
+
"wenyan.txt" => :chars,
|
|
29
|
+
"mainland_hard.tsv" => :pairs,
|
|
30
|
+
"mainland_soft.tsv" => :pairs,
|
|
31
|
+
"mainland_exceptions.tsv" => :pairs,
|
|
32
|
+
"regional_hard.tsv" => :pairs
|
|
33
|
+
}.freeze
|
|
34
|
+
|
|
35
|
+
class << self
|
|
36
|
+
attr_reader :overlay
|
|
37
|
+
|
|
38
|
+
def dir = @dir ||= Pathname(ENV.fetch("TWFILTER_DATA_DIR", BUNDLED.to_s))
|
|
39
|
+
|
|
40
|
+
def dir=(path)
|
|
41
|
+
@dir = path.nil? ? nil : Pathname(path)
|
|
42
|
+
TWFilter.reset!
|
|
43
|
+
end
|
|
44
|
+
|
|
45
|
+
def overlay=(path)
|
|
46
|
+
@overlay = path.nil? ? nil : Pathname(path)
|
|
47
|
+
TWFilter.reset!
|
|
48
|
+
end
|
|
49
|
+
|
|
50
|
+
def rows(name) = cache[name] ||= resolve(name)
|
|
51
|
+
|
|
52
|
+
def set(name) = cache[:"#{name}/set"] ||= rows(name).to_set
|
|
53
|
+
|
|
54
|
+
def pairs(name) = cache[:"#{name}/pairs"] ||= rows(name).to_h { |row| row.split("\t", 2) }.freeze
|
|
55
|
+
|
|
56
|
+
def columns(name) = cache[:"#{name}/columns"] ||= rows(name).map { |row| row.split("\t") }.freeze
|
|
57
|
+
|
|
58
|
+
def groups(name) = cache[:"#{name}/groups"] ||= columns(name)
|
|
59
|
+
.to_h { |row| [row.first, row.drop(1).freeze] }
|
|
60
|
+
.freeze
|
|
61
|
+
|
|
62
|
+
def names
|
|
63
|
+
base = dir.children.map { |path| path.basename.to_s }
|
|
64
|
+
extra = overlay ? overlay.children.map { |path|
|
|
65
|
+
path.basename.to_s.sub(/#{Regexp.escape(ADD)}|#{Regexp.escape(REMOVE)}(?=\.)/, "")
|
|
66
|
+
} : []
|
|
67
|
+
(base + extra).uniq.reject { |name| name == MANIFEST }.sort
|
|
68
|
+
end
|
|
69
|
+
|
|
70
|
+
def manifest
|
|
71
|
+
path = dir.join(MANIFEST)
|
|
72
|
+
path.exist? ? JSON.parse(path.read).freeze : {}
|
|
73
|
+
end
|
|
74
|
+
|
|
75
|
+
# Fingerprint of the effective tables, for recording alongside derived measurements.
|
|
76
|
+
def provenance
|
|
77
|
+
digest = Digest::SHA256.new
|
|
78
|
+
names.each { |name| digest << name << rows(name).join("\n") }
|
|
79
|
+
|
|
80
|
+
{
|
|
81
|
+
version: VERSION,
|
|
82
|
+
dir: dir.to_s,
|
|
83
|
+
overlay: overlay&.to_s,
|
|
84
|
+
tables: names.length,
|
|
85
|
+
# 64 bits, sufficient to detect an accidental table change
|
|
86
|
+
digest: digest.hexdigest[0, 16]
|
|
87
|
+
}.freeze
|
|
88
|
+
end
|
|
89
|
+
|
|
90
|
+
def reset!
|
|
91
|
+
@cache = nil
|
|
92
|
+
ENV["TWFILTER_OVERLAY_DIR"]&.then { |path| @overlay ||= Pathname(path) }
|
|
93
|
+
end
|
|
94
|
+
|
|
95
|
+
private
|
|
96
|
+
|
|
97
|
+
def cache = @cache ||= {}
|
|
98
|
+
|
|
99
|
+
def resolve(name)
|
|
100
|
+
base = overlay&.join(name)&.exist? ? read(overlay.join(name), name) : read(dir.join(name), name)
|
|
101
|
+
return base.freeze if overlay.nil?
|
|
102
|
+
|
|
103
|
+
added = optional(overlay.join(with_suffix(name, ADD)), name)
|
|
104
|
+
removed = optional(overlay.join(with_suffix(name, REMOVE)), nil).to_set
|
|
105
|
+
|
|
106
|
+
(base + added).reject { |row| removed.include?(row) || removed.include?(row.split("\t").first) }.uniq.freeze
|
|
107
|
+
end
|
|
108
|
+
|
|
109
|
+
def with_suffix(name, suffix)
|
|
110
|
+
extension = File.extname(name)
|
|
111
|
+
"#{File.basename(name, extension)}#{suffix}#{extension}"
|
|
112
|
+
end
|
|
113
|
+
|
|
114
|
+
def optional(path, name) = path.exist? ? read(path, name) : []
|
|
115
|
+
|
|
116
|
+
def read(path, name)
|
|
117
|
+
raise MissingTableError, "missing table: #{path}" unless path.exist?
|
|
118
|
+
|
|
119
|
+
rows = path.each_line.map(&:chomp).reject(&:empty?)
|
|
120
|
+
validate(rows, SCHEMA.fetch(name, :terms), path) unless name.nil?
|
|
121
|
+
rows
|
|
122
|
+
end
|
|
123
|
+
|
|
124
|
+
def validate(rows, shape, path)
|
|
125
|
+
offender = case shape
|
|
126
|
+
in :chars
|
|
127
|
+
rows.find { |row| row.length != 1 || !Han.char?(row) }
|
|
128
|
+
in :pairs
|
|
129
|
+
rows.find { |row| row.split("\t").length < 2 || row.start_with?("\t") }
|
|
130
|
+
in :terms
|
|
131
|
+
rows.find { |row| row.include?("\t") }
|
|
132
|
+
end
|
|
133
|
+
|
|
134
|
+
return if offender.nil?
|
|
135
|
+
|
|
136
|
+
raise InvalidTableError, "#{path}: expected #{shape}, got #{offender.inspect}"
|
|
137
|
+
end
|
|
138
|
+
end
|
|
139
|
+
end
|
|
140
|
+
end
|