iev 0.4.7 → 0.4.8
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/.claude/scheduled_tasks.lock +1 -0
- data/.rubocop_todo.yml +5 -5
- data/Gemfile +1 -0
- data/TODO.reconcile/01-design.md +57 -0
- data/TODO.reconcile/02-change-models.md +44 -0
- data/TODO.reconcile/02-termbase-loader.md +48 -0
- data/TODO.reconcile/03-live-loader.md +28 -0
- data/TODO.reconcile/03-status-mapper.md +24 -0
- data/TODO.reconcile/04-concept-merger.md +45 -0
- data/TODO.reconcile/04-termbase-loader.md +28 -0
- data/TODO.reconcile/05-content-diff.md +45 -0
- data/TODO.reconcile/05-live-loader.md +20 -0
- data/TODO.reconcile/06-content-differ.md +32 -0
- data/TODO.reconcile/06-v3-serializer.md +42 -0
- data/TODO.reconcile/07-cli-and-run.md +27 -0
- data/TODO.reconcile/07-concept-merger.md +32 -0
- data/TODO.reconcile/08-report.md +53 -0
- data/TODO.reconcile/09-pipeline.md +23 -0
- data/TODO.reconcile/10-script-and-run.md +15 -0
- data/data/locales/glossarist_enums.yml +705 -0
- data/iev.gemspec +3 -1
- data/lib/iev/cli/command.rb +29 -0
- data/lib/iev/cli/command_helper.rb +1 -6
- data/lib/iev/config.rb +10 -1
- data/lib/iev/exporter.rb +4 -4
- data/lib/iev/multi_doc_yaml.rb +60 -0
- data/lib/iev/reconciler/change.rb +29 -0
- data/lib/iev/reconciler/change_set.rb +48 -0
- data/lib/iev/reconciler/concept_merger.rb +130 -0
- data/lib/iev/reconciler/content_differ.rb +157 -0
- data/lib/iev/reconciler/live_loader.rb +163 -0
- data/lib/iev/reconciler/pipeline.rb +104 -0
- data/lib/iev/reconciler/reconciled_concept.rb +14 -0
- data/lib/iev/reconciler/report.rb +118 -0
- data/lib/iev/reconciler/status_mapper.rb +29 -0
- data/lib/iev/reconciler/term_marker_parser.rb +196 -0
- data/lib/iev/reconciler/termbase_loader.rb +141 -0
- data/lib/iev/reconciler.rb +17 -0
- data/lib/iev/scraper/browser.rb +155 -50
- data/lib/iev/scraper/page_parser.rb +116 -24
- data/lib/iev/source_parser.rb +12 -1
- data/lib/iev/subject_area_concepts.rb +4 -2
- data/lib/iev/utilities.rb +7 -5
- data/lib/iev/version.rb +1 -1
- data/lib/iev.rb +2 -0
- data/scripts/find_unparseable.rb +22 -0
- metadata +50 -4
|
@@ -0,0 +1,104 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "date"
|
|
4
|
+
require "fileutils"
|
|
5
|
+
require "yaml"
|
|
6
|
+
|
|
7
|
+
module Iev
|
|
8
|
+
module Reconciler
|
|
9
|
+
# Orchestrates the full reconciliation by streaming through codes
|
|
10
|
+
# one at a time. Each code is loaded → merged → serialized → discarded,
|
|
11
|
+
# so memory usage is bounded regardless of dataset size.
|
|
12
|
+
class Pipeline
|
|
13
|
+
attr_reader :stats
|
|
14
|
+
|
|
15
|
+
# @param termbase_path [String] path to termbase.yaml
|
|
16
|
+
# @param pages_dir [String] directory with mirrored HTML pages
|
|
17
|
+
# @param output_dir [String] where to write concepts + reports
|
|
18
|
+
# @param detected_at [String] ISO 8601 date for detected changes
|
|
19
|
+
def initialize(termbase_path:, pages_dir:, output_dir:,
|
|
20
|
+
detected_at: Date.today.iso8601)
|
|
21
|
+
@termbase_path = termbase_path
|
|
22
|
+
@pages_dir = pages_dir
|
|
23
|
+
@output_dir = output_dir
|
|
24
|
+
@detected_at = detected_at
|
|
25
|
+
@stats = {}
|
|
26
|
+
end
|
|
27
|
+
|
|
28
|
+
# Run the full pipeline.
|
|
29
|
+
# @return [void]
|
|
30
|
+
def run
|
|
31
|
+
FileUtils.mkdir_p(File.join(@output_dir, "concepts"))
|
|
32
|
+
FileUtils.mkdir_p(File.join(@output_dir, "report"))
|
|
33
|
+
|
|
34
|
+
warn "Indexing sources..."
|
|
35
|
+
termbase = TermbaseLoader.new(@termbase_path)
|
|
36
|
+
live = LiveLoader.new(@pages_dir)
|
|
37
|
+
all_codes = (termbase.codes + live.codes).uniq.sort
|
|
38
|
+
warn " #{termbase.codes.size} termbase + #{live.codes.size} live = #{all_codes.size} total"
|
|
39
|
+
|
|
40
|
+
merger = ConceptMerger.new
|
|
41
|
+
reconciled = []
|
|
42
|
+
errors = []
|
|
43
|
+
|
|
44
|
+
all_codes.each_with_index do |code, idx|
|
|
45
|
+
begin
|
|
46
|
+
rc = merger.merge(
|
|
47
|
+
code: code,
|
|
48
|
+
termbase_concept: termbase.get(code),
|
|
49
|
+
live_concept: live.get(code),
|
|
50
|
+
detected_at: @detected_at,
|
|
51
|
+
)
|
|
52
|
+
|
|
53
|
+
if rc
|
|
54
|
+
reconciled << rc
|
|
55
|
+
save_concept(rc.managed_concept)
|
|
56
|
+
end
|
|
57
|
+
rescue StandardError => e
|
|
58
|
+
errors << { code: code, error: "#{e.class}: #{e.message}" }
|
|
59
|
+
warn " ERROR on #{code}: #{e.message[0, 100]}"
|
|
60
|
+
end
|
|
61
|
+
|
|
62
|
+
if (idx + 1) % 1000 == 0
|
|
63
|
+
warn " #{idx + 1}/#{all_codes.size}... (#{errors.size} errors)"
|
|
64
|
+
end
|
|
65
|
+
end
|
|
66
|
+
|
|
67
|
+
compute_stats(reconciled)
|
|
68
|
+
compute_error_stats(errors) if errors.any?
|
|
69
|
+
Report.new(reconciled).write_to(File.join(@output_dir, "report"))
|
|
70
|
+
warn "Done."
|
|
71
|
+
end
|
|
72
|
+
|
|
73
|
+
private
|
|
74
|
+
|
|
75
|
+
def save_concept(concept)
|
|
76
|
+
path = File.join(@output_dir, "concepts", "#{concept.id}.yaml")
|
|
77
|
+
Iev::MultiDocYaml.write(path, Iev::MultiDocYaml.parts_for(concept))
|
|
78
|
+
end
|
|
79
|
+
|
|
80
|
+
def compute_stats(reconciled)
|
|
81
|
+
@stats = {
|
|
82
|
+
total: reconciled.size,
|
|
83
|
+
in_both: reconciled.count { |r| r.source == :both },
|
|
84
|
+
termbase_only: reconciled.count { |r| r.source == :termbase_only },
|
|
85
|
+
live_only: reconciled.count { |r| r.source == :live_only },
|
|
86
|
+
changed: reconciled.count { |r| !r.change_set.empty? },
|
|
87
|
+
}
|
|
88
|
+
warn ""
|
|
89
|
+
warn "Results:"
|
|
90
|
+
warn " In both (merged): #{@stats[:in_both]}"
|
|
91
|
+
warn " Termbase only (retired): #{@stats[:termbase_only]}"
|
|
92
|
+
warn " Live only (new): #{@stats[:live_only]}"
|
|
93
|
+
warn " Concepts with changes: #{@stats[:changed]}"
|
|
94
|
+
warn " Total: #{@stats[:total]}"
|
|
95
|
+
end
|
|
96
|
+
|
|
97
|
+
def compute_error_stats(errors)
|
|
98
|
+
warn ""
|
|
99
|
+
warn "Errors: #{errors.size}"
|
|
100
|
+
errors.first(10).each { |e| warn " #{e[:code]}: #{e[:error]}" }
|
|
101
|
+
end
|
|
102
|
+
end
|
|
103
|
+
end
|
|
104
|
+
end
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Iev
|
|
4
|
+
module Reconciler
|
|
5
|
+
# Wraps a reconciled ManagedConcept with its ChangeSet and origin.
|
|
6
|
+
# Value object produced by ConceptMerger, consumed by Pipeline + Report.
|
|
7
|
+
#
|
|
8
|
+
# @attr managed_concept [Glossarist::ManagedConcept]
|
|
9
|
+
# @attr change_set [ChangeSet] field-level diffs detected
|
|
10
|
+
# @attr source [Symbol] :both | :termbase_only | :live_only
|
|
11
|
+
ReconciledConcept = Struct.new(:managed_concept, :change_set, :source,
|
|
12
|
+
keyword_init: true)
|
|
13
|
+
end
|
|
14
|
+
end
|
|
@@ -0,0 +1,118 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "csv"
|
|
4
|
+
require "yaml"
|
|
5
|
+
|
|
6
|
+
module Iev
|
|
7
|
+
module Reconciler
|
|
8
|
+
# Generates dataset-level change reports from reconciled concepts.
|
|
9
|
+
# Answers "what changed across the entire IEV?" with machine-readable
|
|
10
|
+
# summary files.
|
|
11
|
+
class Report
|
|
12
|
+
# @param reconciled [Array<ReconciledConcept>] all reconciled concepts
|
|
13
|
+
def initialize(reconciled)
|
|
14
|
+
@reconciled = reconciled
|
|
15
|
+
end
|
|
16
|
+
|
|
17
|
+
# Write all report files to the given directory.
|
|
18
|
+
# @param dir [String, Pathname]
|
|
19
|
+
def write_to(dir)
|
|
20
|
+
require "fileutils"
|
|
21
|
+
FileUtils.mkdir_p(dir)
|
|
22
|
+
|
|
23
|
+
write_summary(File.join(dir, "summary.yaml"))
|
|
24
|
+
write_changes_csv(File.join(dir, "changes.csv"))
|
|
25
|
+
write_retired(File.join(dir, "retired.yaml"))
|
|
26
|
+
write_new_concepts(File.join(dir, "new_concepts.yaml"))
|
|
27
|
+
end
|
|
28
|
+
|
|
29
|
+
# @return [Hash] aggregate statistics
|
|
30
|
+
def summary
|
|
31
|
+
{
|
|
32
|
+
total_concepts: @reconciled.size,
|
|
33
|
+
sources: source_counts,
|
|
34
|
+
changes: change_stats,
|
|
35
|
+
}
|
|
36
|
+
end
|
|
37
|
+
|
|
38
|
+
private
|
|
39
|
+
|
|
40
|
+
def write_summary(path)
|
|
41
|
+
File.write(path, YAML.dump(summary))
|
|
42
|
+
end
|
|
43
|
+
|
|
44
|
+
def write_changes_csv(path)
|
|
45
|
+
CSV.open(path, "w") do |csv|
|
|
46
|
+
csv << %w[code section field language detected_at old_value new_value]
|
|
47
|
+
@reconciled.each do |rc|
|
|
48
|
+
rc.change_set.each do |change|
|
|
49
|
+
csv << [
|
|
50
|
+
change.code,
|
|
51
|
+
section_of(change.code),
|
|
52
|
+
change.field,
|
|
53
|
+
change.language,
|
|
54
|
+
change.detected_at,
|
|
55
|
+
truncate(change.old_value),
|
|
56
|
+
truncate(change.new_value),
|
|
57
|
+
]
|
|
58
|
+
end
|
|
59
|
+
end
|
|
60
|
+
end
|
|
61
|
+
end
|
|
62
|
+
|
|
63
|
+
def write_retired(path)
|
|
64
|
+
retired = @reconciled
|
|
65
|
+
.select { |rc| rc.source == :termbase_only && rc.managed_concept }
|
|
66
|
+
.map { |rc| rc.managed_concept.id }
|
|
67
|
+
File.write(path, YAML.dump(retired))
|
|
68
|
+
end
|
|
69
|
+
|
|
70
|
+
def write_new_concepts(path)
|
|
71
|
+
new_concepts = @reconciled
|
|
72
|
+
.select { |rc| rc.source == :live_only && rc.managed_concept }
|
|
73
|
+
.map { |rc| rc.managed_concept.id }
|
|
74
|
+
File.write(path, YAML.dump(new_concepts))
|
|
75
|
+
end
|
|
76
|
+
|
|
77
|
+
def source_counts
|
|
78
|
+
@reconciled.group_by(&:source).transform_values(&:size)
|
|
79
|
+
end
|
|
80
|
+
|
|
81
|
+
def change_stats
|
|
82
|
+
all_changes = @reconciled.flat_map { |rc| rc.change_set.to_a }
|
|
83
|
+
{
|
|
84
|
+
total: all_changes.size,
|
|
85
|
+
by_field: tally_by(all_changes, :field),
|
|
86
|
+
by_language: tally_by(all_changes, :language),
|
|
87
|
+
by_section: tally_by_section(all_changes),
|
|
88
|
+
}
|
|
89
|
+
end
|
|
90
|
+
|
|
91
|
+
def tally_by(changes, attr)
|
|
92
|
+
changes
|
|
93
|
+
.group_by { |c| c.send(attr) }
|
|
94
|
+
.transform_values(&:size)
|
|
95
|
+
.sort_by { |_, v| -v }
|
|
96
|
+
.to_h
|
|
97
|
+
end
|
|
98
|
+
|
|
99
|
+
def tally_by_section(changes)
|
|
100
|
+
changes
|
|
101
|
+
.group_by { |c| section_of(c.code) }
|
|
102
|
+
.transform_values(&:size)
|
|
103
|
+
.sort_by { |_, v| -v }
|
|
104
|
+
.to_h
|
|
105
|
+
end
|
|
106
|
+
|
|
107
|
+
def section_of(code)
|
|
108
|
+
code.to_s.rpartition("-").first
|
|
109
|
+
end
|
|
110
|
+
|
|
111
|
+
def truncate(value, max = 200)
|
|
112
|
+
return "" if value.nil?
|
|
113
|
+
value = value.to_s
|
|
114
|
+
value.size > max ? value[0, max] + "..." : value
|
|
115
|
+
end
|
|
116
|
+
end
|
|
117
|
+
end
|
|
118
|
+
end
|
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Iev
|
|
4
|
+
module Reconciler
|
|
5
|
+
# Maps termbase entry_status strings to Glossarist V3 ConceptStatus
|
|
6
|
+
# enum values. Stateless and pure.
|
|
7
|
+
module StatusMapper
|
|
8
|
+
MAP = {
|
|
9
|
+
"Standard" => "valid",
|
|
10
|
+
"Published" => "valid",
|
|
11
|
+
"Effective" => "valid",
|
|
12
|
+
"Draft" => "draft",
|
|
13
|
+
"Not Valid" => "notValid",
|
|
14
|
+
"Superseded" => "superseded",
|
|
15
|
+
"Retired" => "retired",
|
|
16
|
+
}.freeze
|
|
17
|
+
|
|
18
|
+
DEFAULT = "valid"
|
|
19
|
+
|
|
20
|
+
module_function
|
|
21
|
+
|
|
22
|
+
# @param entry_status [String, nil]
|
|
23
|
+
# @return [String] a valid ConceptStatus value
|
|
24
|
+
def call(entry_status)
|
|
25
|
+
MAP[entry_status.to_s] || DEFAULT
|
|
26
|
+
end
|
|
27
|
+
end
|
|
28
|
+
end
|
|
29
|
+
end
|
|
@@ -0,0 +1,196 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Iev
|
|
4
|
+
module Reconciler
|
|
5
|
+
# Extracts grammatical markers, context qualifiers, and part-of-speech
|
|
6
|
+
# tags that are inline in Electropedia's live HTML term text, matching
|
|
7
|
+
# how the termbase stores them in separate fields.
|
|
8
|
+
#
|
|
9
|
+
# Two-pass approach:
|
|
10
|
+
# Pass 1: extract context qualifiers and IEV cross-references from
|
|
11
|
+
# angle brackets (<text>, <相关条目:IEV xxx>)
|
|
12
|
+
# Pass 2: extract gender/number/POS/prefix markers from remaining text
|
|
13
|
+
#
|
|
14
|
+
# Marker formats handled:
|
|
15
|
+
# Gender: , f , m , n (also multiple: , f, n)
|
|
16
|
+
# , ж , м , с (Serbian Cyrillic)
|
|
17
|
+
# , m/f (slash = both apply)
|
|
18
|
+
# Number: , pl , sg (also Serbian: јд, мн)
|
|
19
|
+
# , f pl (space-separated gender+number)
|
|
20
|
+
# Usage info: <text> → usage_info field
|
|
21
|
+
# Part of speech: , 名詞 , 명사 , noun , verb , adj. , agg , etc.
|
|
22
|
+
# Prefix: , Präfix , 접두사 , 接頭語 , etc. → prefix flag
|
|
23
|
+
class TermMarkerParser
|
|
24
|
+
GENDER_MAP = {
|
|
25
|
+
"f" => "feminine", "m" => "masculine", "n" => "neuter",
|
|
26
|
+
"ж" => "feminine", "м" => "masculine", "с" => "neuter",
|
|
27
|
+
}.freeze
|
|
28
|
+
|
|
29
|
+
NUMBER_MAP = {
|
|
30
|
+
"јд" => "singular", "мн" => "plural",
|
|
31
|
+
"sg" => "singular", "pl" => "plural",
|
|
32
|
+
}.freeze
|
|
33
|
+
|
|
34
|
+
PART_OF_SPEECH_MAP = {
|
|
35
|
+
"noun" => "noun",
|
|
36
|
+
"명사" => "noun", "名詞" => "noun", "اسم" => "noun",
|
|
37
|
+
"именица" => "noun",
|
|
38
|
+
"verb" => "verb", "verbo" => "verb",
|
|
39
|
+
"동사" => "verb", "動詞" => "verb", "فعل" => "verb",
|
|
40
|
+
"глагол" => "verb",
|
|
41
|
+
"adj" => "adj", "adj." => "adj", "adjective" => "adj",
|
|
42
|
+
"agg" => "adj", "agg." => "adj",
|
|
43
|
+
"형용사" => "adj", "形容詞" => "adj", "صفة" => "adj",
|
|
44
|
+
"adjektiv" => "adj", "придев" => "adj",
|
|
45
|
+
"adv" => "adv", "adv." => "adv", "adverb" => "adv",
|
|
46
|
+
"부사" => "adv", "副詞" => "adv",
|
|
47
|
+
"прилог" => "adv",
|
|
48
|
+
}.freeze
|
|
49
|
+
|
|
50
|
+
PREFIX_KEYWORDS = %w[
|
|
51
|
+
Präfix prefix préfixe 접두사 接頭語 接尾語
|
|
52
|
+
].freeze
|
|
53
|
+
|
|
54
|
+
DOMAIN_RE = /<([^>]+)>/
|
|
55
|
+
USAGE_INFO_RE = /\(([^)]+)\)\s*(?=[,]?\s*(?:[fmnжмс]\b|[fmn]\/[fmn]|sg|pl|јд|мн|\z))/i
|
|
56
|
+
IEV_XREF_RE = /<[^>]*IEV\s*(\d{3}-\d{2}-\d{2,3})[^>]*>/i
|
|
57
|
+
|
|
58
|
+
SERBIAN_RE = /,\s*([жмс])\s+(јд|мн)\s*\z/
|
|
59
|
+
WESTERN_GENDER_NUMBER_RE = /,\s*([fmn])\s+(sg|pl)\.?\s*\z/i
|
|
60
|
+
WESTERN_NUMBER_RE = /,\s*(sg|pl)\.?\s*\z/i
|
|
61
|
+
SLASH_GENDER_RE = /,\s*([fmn])\/([fmn])\s*\z/i
|
|
62
|
+
WESTERN_RE = /[,]?\s+([fmn])\s*\z/
|
|
63
|
+
SPACE_GENDER_RE = /\s+([fmn])\s*\z/
|
|
64
|
+
|
|
65
|
+
POS_RE = /,\s*(#{PART_OF_SPEECH_MAP.keys.map { |k| Regexp.escape(k) }.join("|")})\s*\z/i
|
|
66
|
+
PREFIX_RE = /,\s*(#{PREFIX_KEYWORDS.map { |k| Regexp.escape(k) }.join("|")})\s*\z/
|
|
67
|
+
|
|
68
|
+
TRAILING_PUNCT_RE = /[,;\s]+$/
|
|
69
|
+
|
|
70
|
+
Result = Struct.new(:designation, :genders, :numbers, :domain,
|
|
71
|
+
:usage_info, :part_of_speech, :related_refs,
|
|
72
|
+
:is_prefix, keyword_init: true)
|
|
73
|
+
|
|
74
|
+
class << self
|
|
75
|
+
def parse(term)
|
|
76
|
+
return Result.new(designation: nil) unless term
|
|
77
|
+
|
|
78
|
+
text = decode_entities(term).strip.gsub(/\s+/, " ").gsub(/ /, " ")
|
|
79
|
+
related_refs, text = extract_iev_xrefs(text)
|
|
80
|
+
domain, text = extract_domain(text)
|
|
81
|
+
usage_info, text = extract_usage_info(text)
|
|
82
|
+
designation, genders, numbers, pos, is_prefix = extract_markers(text)
|
|
83
|
+
|
|
84
|
+
Result.new(
|
|
85
|
+
designation: designation,
|
|
86
|
+
genders: genders,
|
|
87
|
+
numbers: numbers,
|
|
88
|
+
domain: domain,
|
|
89
|
+
usage_info: usage_info,
|
|
90
|
+
part_of_speech: pos,
|
|
91
|
+
related_refs: related_refs,
|
|
92
|
+
is_prefix: is_prefix,
|
|
93
|
+
)
|
|
94
|
+
end
|
|
95
|
+
|
|
96
|
+
def parse_multiple(term_text)
|
|
97
|
+
return [] unless term_text
|
|
98
|
+
|
|
99
|
+
term_text
|
|
100
|
+
.split(/\n+/)
|
|
101
|
+
.map(&:strip)
|
|
102
|
+
.reject(&:empty?)
|
|
103
|
+
.map { |part| parse(part) }
|
|
104
|
+
end
|
|
105
|
+
|
|
106
|
+
private
|
|
107
|
+
|
|
108
|
+
def decode_entities(str)
|
|
109
|
+
str
|
|
110
|
+
.gsub(/</, "<")
|
|
111
|
+
.gsub(/>/, ">")
|
|
112
|
+
.gsub(/&/, "&")
|
|
113
|
+
.gsub(/"/, '"')
|
|
114
|
+
.gsub(/'/, "'")
|
|
115
|
+
end
|
|
116
|
+
|
|
117
|
+
# Extract domain qualifier from angle brackets: <text> -> domain
|
|
118
|
+
def extract_domain(text)
|
|
119
|
+
match = text.match(DOMAIN_RE)
|
|
120
|
+
return [nil, text] unless match
|
|
121
|
+
|
|
122
|
+
domain = match[1].strip
|
|
123
|
+
remaining = text.sub(match[0], "").gsub(TRAILING_PUNCT_RE, "").strip
|
|
124
|
+
remaining = remaining.sub(/^[,;\s]+/, "").strip
|
|
125
|
+
[domain, remaining]
|
|
126
|
+
end
|
|
127
|
+
|
|
128
|
+
def extract_iev_xrefs(text)
|
|
129
|
+
refs = []
|
|
130
|
+
remaining = text
|
|
131
|
+
while (match = remaining.match(IEV_XREF_RE))
|
|
132
|
+
refs << match[1]
|
|
133
|
+
remaining = remaining.sub(match[0], "").gsub(TRAILING_PUNCT_RE, "").strip
|
|
134
|
+
remaining = remaining.sub(/^[,;\s]+/, "").strip
|
|
135
|
+
end
|
|
136
|
+
[refs.empty? ? nil : refs, remaining]
|
|
137
|
+
end
|
|
138
|
+
|
|
139
|
+
# Extract usage_info from parentheses that precede trailing markers.
|
|
140
|
+
# E.g. "Orientierung (einer Kurve) f" -> usage_info="einer Kurve"
|
|
141
|
+
def extract_usage_info(text)
|
|
142
|
+
match = text.match(USAGE_INFO_RE)
|
|
143
|
+
return [nil, text] unless match
|
|
144
|
+
|
|
145
|
+
usage = match[1].strip
|
|
146
|
+
remaining = text.sub(match[0], "").gsub(/\s+/, " ").strip
|
|
147
|
+
[usage, remaining]
|
|
148
|
+
end
|
|
149
|
+
|
|
150
|
+
def extract_markers(text)
|
|
151
|
+
designation = text
|
|
152
|
+
genders = []
|
|
153
|
+
numbers = []
|
|
154
|
+
pos = nil
|
|
155
|
+
is_prefix = false
|
|
156
|
+
|
|
157
|
+
loop do
|
|
158
|
+
break if designation.nil? || designation.empty?
|
|
159
|
+
|
|
160
|
+
if (m = designation.match(SERBIAN_RE))
|
|
161
|
+
designation = designation[0, m.begin(0)].strip
|
|
162
|
+
genders << GENDER_MAP[m[1]] if GENDER_MAP[m[1]]
|
|
163
|
+
numbers << NUMBER_MAP[m[2]] if NUMBER_MAP[m[2]]
|
|
164
|
+
elsif (m = designation.match(SLASH_GENDER_RE))
|
|
165
|
+
designation = designation[0, m.begin(0)].strip
|
|
166
|
+
genders << GENDER_MAP[m[1]] if GENDER_MAP[m[1]]
|
|
167
|
+
genders << GENDER_MAP[m[2]] if GENDER_MAP[m[2]]
|
|
168
|
+
elsif (m = designation.match(WESTERN_GENDER_NUMBER_RE))
|
|
169
|
+
designation = designation[0, m.begin(0)].strip
|
|
170
|
+
genders << GENDER_MAP[m[1]] if GENDER_MAP[m[1]]
|
|
171
|
+
num_key = m[2].downcase.sub(/\.$/, "")
|
|
172
|
+
numbers << NUMBER_MAP[num_key] if NUMBER_MAP[num_key]
|
|
173
|
+
elsif (m = designation.match(WESTERN_NUMBER_RE))
|
|
174
|
+
designation = designation[0, m.begin(0)].strip
|
|
175
|
+
num_key = m[1].downcase.sub(/\.$/, "")
|
|
176
|
+
numbers << NUMBER_MAP[num_key] if NUMBER_MAP[num_key]
|
|
177
|
+
elsif (m = designation.match(POS_RE))
|
|
178
|
+
designation = designation[0, m.begin(0)].strip
|
|
179
|
+
pos = PART_OF_SPEECH_MAP[m[1].downcase]
|
|
180
|
+
elsif (m = designation.match(PREFIX_RE))
|
|
181
|
+
designation = designation[0, m.begin(0)].strip
|
|
182
|
+
is_prefix = true
|
|
183
|
+
elsif (m = designation.match(WESTERN_RE))
|
|
184
|
+
designation = designation[0, m.begin(0)].strip
|
|
185
|
+
genders << GENDER_MAP[m[1]] if GENDER_MAP[m[1]]
|
|
186
|
+
else
|
|
187
|
+
break
|
|
188
|
+
end
|
|
189
|
+
end
|
|
190
|
+
|
|
191
|
+
[designation, genders.uniq, numbers.uniq, pos, is_prefix]
|
|
192
|
+
end
|
|
193
|
+
end
|
|
194
|
+
end
|
|
195
|
+
end
|
|
196
|
+
end
|
|
@@ -0,0 +1,141 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "yaml"
|
|
4
|
+
|
|
5
|
+
module Iev
|
|
6
|
+
module Reconciler
|
|
7
|
+
# Loads termbase.yaml into a raw hash, then builds
|
|
8
|
+
# Glossarist::ManagedConcept objects on demand per code.
|
|
9
|
+
# This avoids the cost of constructing 22k model objects upfront.
|
|
10
|
+
class TermbaseLoader
|
|
11
|
+
# @param path [String, Pathname] path to termbase.yaml
|
|
12
|
+
def initialize(path)
|
|
13
|
+
@path = path.to_s
|
|
14
|
+
@data = nil
|
|
15
|
+
end
|
|
16
|
+
|
|
17
|
+
# @return [Hash<String, Hash>] raw termbase data keyed by code
|
|
18
|
+
def raw
|
|
19
|
+
@data ||= YAML.load_file(@path)
|
|
20
|
+
end
|
|
21
|
+
|
|
22
|
+
# @return [Array<String>] all codes in the termbase
|
|
23
|
+
def codes
|
|
24
|
+
raw.keys.map(&:to_s)
|
|
25
|
+
end
|
|
26
|
+
|
|
27
|
+
# Build a single ManagedConcept for the given code.
|
|
28
|
+
# @param code [String]
|
|
29
|
+
# @return [Glossarist::ManagedConcept, nil]
|
|
30
|
+
def get(code)
|
|
31
|
+
entry = raw[code] || raw[code.to_sym]
|
|
32
|
+
return nil unless entry.is_a?(Hash)
|
|
33
|
+
|
|
34
|
+
build_concept(code.to_s, entry)
|
|
35
|
+
end
|
|
36
|
+
|
|
37
|
+
private
|
|
38
|
+
|
|
39
|
+
def build_concept(code, entry)
|
|
40
|
+
concept = Glossarist::ManagedConcept.of_yaml(
|
|
41
|
+
"id" => code, "data" => { "id" => code },
|
|
42
|
+
)
|
|
43
|
+
concept.status = concept_status(entry)
|
|
44
|
+
concept.dates = concept_dates(entry)
|
|
45
|
+
concept.schema_version = "3"
|
|
46
|
+
|
|
47
|
+
build_localized(code, entry).each do |lc|
|
|
48
|
+
concept.add_l10n(lc)
|
|
49
|
+
end
|
|
50
|
+
|
|
51
|
+
concept
|
|
52
|
+
end
|
|
53
|
+
|
|
54
|
+
def concept_status(entry)
|
|
55
|
+
lang_data = first_language(entry)
|
|
56
|
+
StatusMapper.call(lang_data&.dig("entry_status"))
|
|
57
|
+
end
|
|
58
|
+
|
|
59
|
+
def concept_dates(entry)
|
|
60
|
+
lang_data = first_language(entry)
|
|
61
|
+
return [] unless lang_data
|
|
62
|
+
|
|
63
|
+
dates = []
|
|
64
|
+
if lang_data["date_accepted"]
|
|
65
|
+
dates << Glossarist::ConceptDate.new(
|
|
66
|
+
type: "accepted",
|
|
67
|
+
date: lang_data["date_accepted"],
|
|
68
|
+
)
|
|
69
|
+
end
|
|
70
|
+
if lang_data["date_amended"] && lang_data["date_amended"] != lang_data["date_accepted"]
|
|
71
|
+
dates << Glossarist::ConceptDate.new(
|
|
72
|
+
type: "amended",
|
|
73
|
+
date: lang_data["date_amended"],
|
|
74
|
+
)
|
|
75
|
+
end
|
|
76
|
+
dates
|
|
77
|
+
end
|
|
78
|
+
|
|
79
|
+
def first_language(entry)
|
|
80
|
+
entry.values.find { |v| v.is_a?(Hash) && v["terms"] }
|
|
81
|
+
end
|
|
82
|
+
|
|
83
|
+
def build_localized(code, entry)
|
|
84
|
+
result = []
|
|
85
|
+
entry.each do |lang, data|
|
|
86
|
+
next unless data.is_a?(Hash) && data["terms"]
|
|
87
|
+
lc = build_localized_concept(code, lang, data)
|
|
88
|
+
result << lc if lc
|
|
89
|
+
end
|
|
90
|
+
result
|
|
91
|
+
end
|
|
92
|
+
|
|
93
|
+
def build_localized_concept(code, lang, data)
|
|
94
|
+
cdata = Glossarist::ConceptData.new
|
|
95
|
+
cdata.id = code
|
|
96
|
+
cdata.language_code = lang
|
|
97
|
+
cdata.entry_status = StatusMapper.call(data["entry_status"])
|
|
98
|
+
|
|
99
|
+
cdata.terms = (data["terms"] || []).map do |t|
|
|
100
|
+
build_term_expression(t)
|
|
101
|
+
end
|
|
102
|
+
|
|
103
|
+
if data["definition"] && !data["definition"].empty?
|
|
104
|
+
cdata.definition = [Glossarist::DetailedDefinition.new(content: data["definition"])]
|
|
105
|
+
end
|
|
106
|
+
|
|
107
|
+
cdata.notes = Array(data["notes"]).map { |n| Glossarist::DetailedDefinition.new(content: n) }
|
|
108
|
+
cdata.examples = Array(data["examples"]).map { |e| Glossarist::DetailedDefinition.new(content: e) }
|
|
109
|
+
|
|
110
|
+
lc = Glossarist::LocalizedConcept.new
|
|
111
|
+
lc.id = code
|
|
112
|
+
lc.data = cdata
|
|
113
|
+
lc
|
|
114
|
+
end
|
|
115
|
+
|
|
116
|
+
def build_term_expression(term_data)
|
|
117
|
+
designation = term_data["designation"].to_s
|
|
118
|
+
parsed = TermMarkerParser.parse(designation)
|
|
119
|
+
|
|
120
|
+
expr = Glossarist::Designation::Expression.new(
|
|
121
|
+
designation: parsed.designation,
|
|
122
|
+
normative_status: term_data["normative_status"] || "preferred",
|
|
123
|
+
type: term_data["type"] || "expression",
|
|
124
|
+
)
|
|
125
|
+
expr.geographical_area = term_data["geographical_area"] if term_data["geographical_area"]
|
|
126
|
+
expr.field_of_application = term_data["usage_info"] || parsed.domain
|
|
127
|
+
expr.usage_info = parsed.usage_info if parsed.usage_info
|
|
128
|
+
expr.prefix = parsed.is_prefix if parsed.is_prefix
|
|
129
|
+
|
|
130
|
+
grammar = Glossarist::Designation::GrammarInfo.new
|
|
131
|
+
merged_genders = (Array(term_data["gender"]) + parsed.genders).flatten.compact.uniq
|
|
132
|
+
merged_numbers = (Array(term_data["plurality"]) + parsed.numbers).flatten.compact.uniq
|
|
133
|
+
grammar.gender = merged_genders if merged_genders.any?
|
|
134
|
+
grammar.number = merged_numbers if merged_numbers.any?
|
|
135
|
+
grammar.part_of_speech = parsed.part_of_speech if parsed.part_of_speech
|
|
136
|
+
expr.grammar_info = [grammar] if grammar.gender&.any? || grammar.number&.any? || grammar.part_of_speech
|
|
137
|
+
expr
|
|
138
|
+
end
|
|
139
|
+
end
|
|
140
|
+
end
|
|
141
|
+
end
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Iev
|
|
4
|
+
module Reconciler
|
|
5
|
+
autoload :Change, "iev/reconciler/change"
|
|
6
|
+
autoload :ChangeSet, "iev/reconciler/change_set"
|
|
7
|
+
autoload :ConceptMerger, "iev/reconciler/concept_merger"
|
|
8
|
+
autoload :ContentDiffer, "iev/reconciler/content_differ"
|
|
9
|
+
autoload :LiveLoader, "iev/reconciler/live_loader"
|
|
10
|
+
autoload :Pipeline, "iev/reconciler/pipeline"
|
|
11
|
+
autoload :ReconciledConcept, "iev/reconciler/reconciled_concept"
|
|
12
|
+
autoload :Report, "iev/reconciler/report"
|
|
13
|
+
autoload :StatusMapper, "iev/reconciler/status_mapper"
|
|
14
|
+
autoload :TermMarkerParser, "iev/reconciler/term_marker_parser"
|
|
15
|
+
autoload :TermbaseLoader, "iev/reconciler/termbase_loader"
|
|
16
|
+
end
|
|
17
|
+
end
|