iev 0.4.7 → 0.4.8

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (48) hide show
  1. checksums.yaml +4 -4
  2. data/.claude/scheduled_tasks.lock +1 -0
  3. data/.rubocop_todo.yml +5 -5
  4. data/Gemfile +1 -0
  5. data/TODO.reconcile/01-design.md +57 -0
  6. data/TODO.reconcile/02-change-models.md +44 -0
  7. data/TODO.reconcile/02-termbase-loader.md +48 -0
  8. data/TODO.reconcile/03-live-loader.md +28 -0
  9. data/TODO.reconcile/03-status-mapper.md +24 -0
  10. data/TODO.reconcile/04-concept-merger.md +45 -0
  11. data/TODO.reconcile/04-termbase-loader.md +28 -0
  12. data/TODO.reconcile/05-content-diff.md +45 -0
  13. data/TODO.reconcile/05-live-loader.md +20 -0
  14. data/TODO.reconcile/06-content-differ.md +32 -0
  15. data/TODO.reconcile/06-v3-serializer.md +42 -0
  16. data/TODO.reconcile/07-cli-and-run.md +27 -0
  17. data/TODO.reconcile/07-concept-merger.md +32 -0
  18. data/TODO.reconcile/08-report.md +53 -0
  19. data/TODO.reconcile/09-pipeline.md +23 -0
  20. data/TODO.reconcile/10-script-and-run.md +15 -0
  21. data/data/locales/glossarist_enums.yml +705 -0
  22. data/iev.gemspec +3 -1
  23. data/lib/iev/cli/command.rb +29 -0
  24. data/lib/iev/cli/command_helper.rb +1 -6
  25. data/lib/iev/config.rb +10 -1
  26. data/lib/iev/exporter.rb +4 -4
  27. data/lib/iev/multi_doc_yaml.rb +60 -0
  28. data/lib/iev/reconciler/change.rb +29 -0
  29. data/lib/iev/reconciler/change_set.rb +48 -0
  30. data/lib/iev/reconciler/concept_merger.rb +130 -0
  31. data/lib/iev/reconciler/content_differ.rb +157 -0
  32. data/lib/iev/reconciler/live_loader.rb +163 -0
  33. data/lib/iev/reconciler/pipeline.rb +104 -0
  34. data/lib/iev/reconciler/reconciled_concept.rb +14 -0
  35. data/lib/iev/reconciler/report.rb +118 -0
  36. data/lib/iev/reconciler/status_mapper.rb +29 -0
  37. data/lib/iev/reconciler/term_marker_parser.rb +196 -0
  38. data/lib/iev/reconciler/termbase_loader.rb +141 -0
  39. data/lib/iev/reconciler.rb +17 -0
  40. data/lib/iev/scraper/browser.rb +155 -50
  41. data/lib/iev/scraper/page_parser.rb +116 -24
  42. data/lib/iev/source_parser.rb +12 -1
  43. data/lib/iev/subject_area_concepts.rb +4 -2
  44. data/lib/iev/utilities.rb +7 -5
  45. data/lib/iev/version.rb +1 -1
  46. data/lib/iev.rb +2 -0
  47. data/scripts/find_unparseable.rb +22 -0
  48. metadata +50 -4
@@ -0,0 +1,104 @@
1
+ # frozen_string_literal: true
2
+
3
+ require "date"
4
+ require "fileutils"
5
+ require "yaml"
6
+
7
+ module Iev
8
+ module Reconciler
9
+ # Orchestrates the full reconciliation by streaming through codes
10
+ # one at a time. Each code is loaded → merged → serialized → discarded,
11
+ # so memory usage is bounded regardless of dataset size.
12
+ class Pipeline
13
+ attr_reader :stats
14
+
15
+ # @param termbase_path [String] path to termbase.yaml
16
+ # @param pages_dir [String] directory with mirrored HTML pages
17
+ # @param output_dir [String] where to write concepts + reports
18
+ # @param detected_at [String] ISO 8601 date for detected changes
19
+ def initialize(termbase_path:, pages_dir:, output_dir:,
20
+ detected_at: Date.today.iso8601)
21
+ @termbase_path = termbase_path
22
+ @pages_dir = pages_dir
23
+ @output_dir = output_dir
24
+ @detected_at = detected_at
25
+ @stats = {}
26
+ end
27
+
28
+ # Run the full pipeline.
29
+ # @return [void]
30
+ def run
31
+ FileUtils.mkdir_p(File.join(@output_dir, "concepts"))
32
+ FileUtils.mkdir_p(File.join(@output_dir, "report"))
33
+
34
+ warn "Indexing sources..."
35
+ termbase = TermbaseLoader.new(@termbase_path)
36
+ live = LiveLoader.new(@pages_dir)
37
+ all_codes = (termbase.codes + live.codes).uniq.sort
38
+ warn " #{termbase.codes.size} termbase + #{live.codes.size} live = #{all_codes.size} total"
39
+
40
+ merger = ConceptMerger.new
41
+ reconciled = []
42
+ errors = []
43
+
44
+ all_codes.each_with_index do |code, idx|
45
+ begin
46
+ rc = merger.merge(
47
+ code: code,
48
+ termbase_concept: termbase.get(code),
49
+ live_concept: live.get(code),
50
+ detected_at: @detected_at,
51
+ )
52
+
53
+ if rc
54
+ reconciled << rc
55
+ save_concept(rc.managed_concept)
56
+ end
57
+ rescue StandardError => e
58
+ errors << { code: code, error: "#{e.class}: #{e.message}" }
59
+ warn " ERROR on #{code}: #{e.message[0, 100]}"
60
+ end
61
+
62
+ if (idx + 1) % 1000 == 0
63
+ warn " #{idx + 1}/#{all_codes.size}... (#{errors.size} errors)"
64
+ end
65
+ end
66
+
67
+ compute_stats(reconciled)
68
+ compute_error_stats(errors) if errors.any?
69
+ Report.new(reconciled).write_to(File.join(@output_dir, "report"))
70
+ warn "Done."
71
+ end
72
+
73
+ private
74
+
75
+ def save_concept(concept)
76
+ path = File.join(@output_dir, "concepts", "#{concept.id}.yaml")
77
+ Iev::MultiDocYaml.write(path, Iev::MultiDocYaml.parts_for(concept))
78
+ end
79
+
80
+ def compute_stats(reconciled)
81
+ @stats = {
82
+ total: reconciled.size,
83
+ in_both: reconciled.count { |r| r.source == :both },
84
+ termbase_only: reconciled.count { |r| r.source == :termbase_only },
85
+ live_only: reconciled.count { |r| r.source == :live_only },
86
+ changed: reconciled.count { |r| !r.change_set.empty? },
87
+ }
88
+ warn ""
89
+ warn "Results:"
90
+ warn " In both (merged): #{@stats[:in_both]}"
91
+ warn " Termbase only (retired): #{@stats[:termbase_only]}"
92
+ warn " Live only (new): #{@stats[:live_only]}"
93
+ warn " Concepts with changes: #{@stats[:changed]}"
94
+ warn " Total: #{@stats[:total]}"
95
+ end
96
+
97
+ def compute_error_stats(errors)
98
+ warn ""
99
+ warn "Errors: #{errors.size}"
100
+ errors.first(10).each { |e| warn " #{e[:code]}: #{e[:error]}" }
101
+ end
102
+ end
103
+ end
104
+ end
@@ -0,0 +1,14 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Iev
4
+ module Reconciler
5
+ # Wraps a reconciled ManagedConcept with its ChangeSet and origin.
6
+ # Value object produced by ConceptMerger, consumed by Pipeline + Report.
7
+ #
8
+ # @attr managed_concept [Glossarist::ManagedConcept]
9
+ # @attr change_set [ChangeSet] field-level diffs detected
10
+ # @attr source [Symbol] :both | :termbase_only | :live_only
11
+ ReconciledConcept = Struct.new(:managed_concept, :change_set, :source,
12
+ keyword_init: true)
13
+ end
14
+ end
@@ -0,0 +1,118 @@
1
+ # frozen_string_literal: true
2
+
3
+ require "csv"
4
+ require "yaml"
5
+
6
+ module Iev
7
+ module Reconciler
8
+ # Generates dataset-level change reports from reconciled concepts.
9
+ # Answers "what changed across the entire IEV?" with machine-readable
10
+ # summary files.
11
+ class Report
12
+ # @param reconciled [Array<ReconciledConcept>] all reconciled concepts
13
+ def initialize(reconciled)
14
+ @reconciled = reconciled
15
+ end
16
+
17
+ # Write all report files to the given directory.
18
+ # @param dir [String, Pathname]
19
+ def write_to(dir)
20
+ require "fileutils"
21
+ FileUtils.mkdir_p(dir)
22
+
23
+ write_summary(File.join(dir, "summary.yaml"))
24
+ write_changes_csv(File.join(dir, "changes.csv"))
25
+ write_retired(File.join(dir, "retired.yaml"))
26
+ write_new_concepts(File.join(dir, "new_concepts.yaml"))
27
+ end
28
+
29
+ # @return [Hash] aggregate statistics
30
+ def summary
31
+ {
32
+ total_concepts: @reconciled.size,
33
+ sources: source_counts,
34
+ changes: change_stats,
35
+ }
36
+ end
37
+
38
+ private
39
+
40
+ def write_summary(path)
41
+ File.write(path, YAML.dump(summary))
42
+ end
43
+
44
+ def write_changes_csv(path)
45
+ CSV.open(path, "w") do |csv|
46
+ csv << %w[code section field language detected_at old_value new_value]
47
+ @reconciled.each do |rc|
48
+ rc.change_set.each do |change|
49
+ csv << [
50
+ change.code,
51
+ section_of(change.code),
52
+ change.field,
53
+ change.language,
54
+ change.detected_at,
55
+ truncate(change.old_value),
56
+ truncate(change.new_value),
57
+ ]
58
+ end
59
+ end
60
+ end
61
+ end
62
+
63
+ def write_retired(path)
64
+ retired = @reconciled
65
+ .select { |rc| rc.source == :termbase_only && rc.managed_concept }
66
+ .map { |rc| rc.managed_concept.id }
67
+ File.write(path, YAML.dump(retired))
68
+ end
69
+
70
+ def write_new_concepts(path)
71
+ new_concepts = @reconciled
72
+ .select { |rc| rc.source == :live_only && rc.managed_concept }
73
+ .map { |rc| rc.managed_concept.id }
74
+ File.write(path, YAML.dump(new_concepts))
75
+ end
76
+
77
+ def source_counts
78
+ @reconciled.group_by(&:source).transform_values(&:size)
79
+ end
80
+
81
+ def change_stats
82
+ all_changes = @reconciled.flat_map { |rc| rc.change_set.to_a }
83
+ {
84
+ total: all_changes.size,
85
+ by_field: tally_by(all_changes, :field),
86
+ by_language: tally_by(all_changes, :language),
87
+ by_section: tally_by_section(all_changes),
88
+ }
89
+ end
90
+
91
+ def tally_by(changes, attr)
92
+ changes
93
+ .group_by { |c| c.send(attr) }
94
+ .transform_values(&:size)
95
+ .sort_by { |_, v| -v }
96
+ .to_h
97
+ end
98
+
99
+ def tally_by_section(changes)
100
+ changes
101
+ .group_by { |c| section_of(c.code) }
102
+ .transform_values(&:size)
103
+ .sort_by { |_, v| -v }
104
+ .to_h
105
+ end
106
+
107
+ def section_of(code)
108
+ code.to_s.rpartition("-").first
109
+ end
110
+
111
+ def truncate(value, max = 200)
112
+ return "" if value.nil?
113
+ value = value.to_s
114
+ value.size > max ? value[0, max] + "..." : value
115
+ end
116
+ end
117
+ end
118
+ end
@@ -0,0 +1,29 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Iev
4
+ module Reconciler
5
+ # Maps termbase entry_status strings to Glossarist V3 ConceptStatus
6
+ # enum values. Stateless and pure.
7
+ module StatusMapper
8
+ MAP = {
9
+ "Standard" => "valid",
10
+ "Published" => "valid",
11
+ "Effective" => "valid",
12
+ "Draft" => "draft",
13
+ "Not Valid" => "notValid",
14
+ "Superseded" => "superseded",
15
+ "Retired" => "retired",
16
+ }.freeze
17
+
18
+ DEFAULT = "valid"
19
+
20
+ module_function
21
+
22
+ # @param entry_status [String, nil]
23
+ # @return [String] a valid ConceptStatus value
24
+ def call(entry_status)
25
+ MAP[entry_status.to_s] || DEFAULT
26
+ end
27
+ end
28
+ end
29
+ end
@@ -0,0 +1,196 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Iev
4
+ module Reconciler
5
+ # Extracts grammatical markers, context qualifiers, and part-of-speech
6
+ # tags that are inline in Electropedia's live HTML term text, matching
7
+ # how the termbase stores them in separate fields.
8
+ #
9
+ # Two-pass approach:
10
+ # Pass 1: extract context qualifiers and IEV cross-references from
11
+ # angle brackets (<text>, <相关条目:IEV xxx>)
12
+ # Pass 2: extract gender/number/POS/prefix markers from remaining text
13
+ #
14
+ # Marker formats handled:
15
+ # Gender: , f , m , n (also multiple: , f, n)
16
+ # , ж , м , с (Serbian Cyrillic)
17
+ # , m/f (slash = both apply)
18
+ # Number: , pl , sg (also Serbian: јд, мн)
19
+ # , f pl (space-separated gender+number)
20
+ # Usage info: <text> → usage_info field
21
+ # Part of speech: , 名詞 , 명사 , noun , verb , adj. , agg , etc.
22
+ # Prefix: , Präfix , 접두사 , 接頭語 , etc. → prefix flag
23
+ class TermMarkerParser
24
+ GENDER_MAP = {
25
+ "f" => "feminine", "m" => "masculine", "n" => "neuter",
26
+ "ж" => "feminine", "м" => "masculine", "с" => "neuter",
27
+ }.freeze
28
+
29
+ NUMBER_MAP = {
30
+ "јд" => "singular", "мн" => "plural",
31
+ "sg" => "singular", "pl" => "plural",
32
+ }.freeze
33
+
34
+ PART_OF_SPEECH_MAP = {
35
+ "noun" => "noun",
36
+ "명사" => "noun", "名詞" => "noun", "اسم" => "noun",
37
+ "именица" => "noun",
38
+ "verb" => "verb", "verbo" => "verb",
39
+ "동사" => "verb", "動詞" => "verb", "فعل" => "verb",
40
+ "глагол" => "verb",
41
+ "adj" => "adj", "adj." => "adj", "adjective" => "adj",
42
+ "agg" => "adj", "agg." => "adj",
43
+ "형용사" => "adj", "形容詞" => "adj", "صفة" => "adj",
44
+ "adjektiv" => "adj", "придев" => "adj",
45
+ "adv" => "adv", "adv." => "adv", "adverb" => "adv",
46
+ "부사" => "adv", "副詞" => "adv",
47
+ "прилог" => "adv",
48
+ }.freeze
49
+
50
+ PREFIX_KEYWORDS = %w[
51
+ Präfix prefix préfixe 접두사 接頭語 接尾語
52
+ ].freeze
53
+
54
+ DOMAIN_RE = /<([^>]+)>/
55
+ USAGE_INFO_RE = /\(([^)]+)\)\s*(?=[,]?\s*(?:[fmnжмс]\b|[fmn]\/[fmn]|sg|pl|јд|мн|\z))/i
56
+ IEV_XREF_RE = /<[^>]*IEV\s*(\d{3}-\d{2}-\d{2,3})[^>]*>/i
57
+
58
+ SERBIAN_RE = /,\s*([жмс])\s+(јд|мн)\s*\z/
59
+ WESTERN_GENDER_NUMBER_RE = /,\s*([fmn])\s+(sg|pl)\.?\s*\z/i
60
+ WESTERN_NUMBER_RE = /,\s*(sg|pl)\.?\s*\z/i
61
+ SLASH_GENDER_RE = /,\s*([fmn])\/([fmn])\s*\z/i
62
+ WESTERN_RE = /[,]?\s+([fmn])\s*\z/
63
+ SPACE_GENDER_RE = /\s+([fmn])\s*\z/
64
+
65
+ POS_RE = /,\s*(#{PART_OF_SPEECH_MAP.keys.map { |k| Regexp.escape(k) }.join("|")})\s*\z/i
66
+ PREFIX_RE = /,\s*(#{PREFIX_KEYWORDS.map { |k| Regexp.escape(k) }.join("|")})\s*\z/
67
+
68
+ TRAILING_PUNCT_RE = /[,;\s]+$/
69
+
70
+ Result = Struct.new(:designation, :genders, :numbers, :domain,
71
+ :usage_info, :part_of_speech, :related_refs,
72
+ :is_prefix, keyword_init: true)
73
+
74
+ class << self
75
+ def parse(term)
76
+ return Result.new(designation: nil) unless term
77
+
78
+ text = decode_entities(term).strip.gsub(/\s+/, " ").gsub(/ /, " ")
79
+ related_refs, text = extract_iev_xrefs(text)
80
+ domain, text = extract_domain(text)
81
+ usage_info, text = extract_usage_info(text)
82
+ designation, genders, numbers, pos, is_prefix = extract_markers(text)
83
+
84
+ Result.new(
85
+ designation: designation,
86
+ genders: genders,
87
+ numbers: numbers,
88
+ domain: domain,
89
+ usage_info: usage_info,
90
+ part_of_speech: pos,
91
+ related_refs: related_refs,
92
+ is_prefix: is_prefix,
93
+ )
94
+ end
95
+
96
+ def parse_multiple(term_text)
97
+ return [] unless term_text
98
+
99
+ term_text
100
+ .split(/\n+/)
101
+ .map(&:strip)
102
+ .reject(&:empty?)
103
+ .map { |part| parse(part) }
104
+ end
105
+
106
+ private
107
+
108
+ def decode_entities(str)
109
+ str
110
+ .gsub(/&lt;/, "<")
111
+ .gsub(/&gt;/, ">")
112
+ .gsub(/&amp;/, "&")
113
+ .gsub(/&quot;/, '"')
114
+ .gsub(/&#39;/, "'")
115
+ end
116
+
117
+ # Extract domain qualifier from angle brackets: <text> -> domain
118
+ def extract_domain(text)
119
+ match = text.match(DOMAIN_RE)
120
+ return [nil, text] unless match
121
+
122
+ domain = match[1].strip
123
+ remaining = text.sub(match[0], "").gsub(TRAILING_PUNCT_RE, "").strip
124
+ remaining = remaining.sub(/^[,;\s]+/, "").strip
125
+ [domain, remaining]
126
+ end
127
+
128
+ def extract_iev_xrefs(text)
129
+ refs = []
130
+ remaining = text
131
+ while (match = remaining.match(IEV_XREF_RE))
132
+ refs << match[1]
133
+ remaining = remaining.sub(match[0], "").gsub(TRAILING_PUNCT_RE, "").strip
134
+ remaining = remaining.sub(/^[,;\s]+/, "").strip
135
+ end
136
+ [refs.empty? ? nil : refs, remaining]
137
+ end
138
+
139
+ # Extract usage_info from parentheses that precede trailing markers.
140
+ # E.g. "Orientierung (einer Kurve) f" -> usage_info="einer Kurve"
141
+ def extract_usage_info(text)
142
+ match = text.match(USAGE_INFO_RE)
143
+ return [nil, text] unless match
144
+
145
+ usage = match[1].strip
146
+ remaining = text.sub(match[0], "").gsub(/\s+/, " ").strip
147
+ [usage, remaining]
148
+ end
149
+
150
+ def extract_markers(text)
151
+ designation = text
152
+ genders = []
153
+ numbers = []
154
+ pos = nil
155
+ is_prefix = false
156
+
157
+ loop do
158
+ break if designation.nil? || designation.empty?
159
+
160
+ if (m = designation.match(SERBIAN_RE))
161
+ designation = designation[0, m.begin(0)].strip
162
+ genders << GENDER_MAP[m[1]] if GENDER_MAP[m[1]]
163
+ numbers << NUMBER_MAP[m[2]] if NUMBER_MAP[m[2]]
164
+ elsif (m = designation.match(SLASH_GENDER_RE))
165
+ designation = designation[0, m.begin(0)].strip
166
+ genders << GENDER_MAP[m[1]] if GENDER_MAP[m[1]]
167
+ genders << GENDER_MAP[m[2]] if GENDER_MAP[m[2]]
168
+ elsif (m = designation.match(WESTERN_GENDER_NUMBER_RE))
169
+ designation = designation[0, m.begin(0)].strip
170
+ genders << GENDER_MAP[m[1]] if GENDER_MAP[m[1]]
171
+ num_key = m[2].downcase.sub(/\.$/, "")
172
+ numbers << NUMBER_MAP[num_key] if NUMBER_MAP[num_key]
173
+ elsif (m = designation.match(WESTERN_NUMBER_RE))
174
+ designation = designation[0, m.begin(0)].strip
175
+ num_key = m[1].downcase.sub(/\.$/, "")
176
+ numbers << NUMBER_MAP[num_key] if NUMBER_MAP[num_key]
177
+ elsif (m = designation.match(POS_RE))
178
+ designation = designation[0, m.begin(0)].strip
179
+ pos = PART_OF_SPEECH_MAP[m[1].downcase]
180
+ elsif (m = designation.match(PREFIX_RE))
181
+ designation = designation[0, m.begin(0)].strip
182
+ is_prefix = true
183
+ elsif (m = designation.match(WESTERN_RE))
184
+ designation = designation[0, m.begin(0)].strip
185
+ genders << GENDER_MAP[m[1]] if GENDER_MAP[m[1]]
186
+ else
187
+ break
188
+ end
189
+ end
190
+
191
+ [designation, genders.uniq, numbers.uniq, pos, is_prefix]
192
+ end
193
+ end
194
+ end
195
+ end
196
+ end
@@ -0,0 +1,141 @@
1
+ # frozen_string_literal: true
2
+
3
+ require "yaml"
4
+
5
+ module Iev
6
+ module Reconciler
7
+ # Loads termbase.yaml into a raw hash, then builds
8
+ # Glossarist::ManagedConcept objects on demand per code.
9
+ # This avoids the cost of constructing 22k model objects upfront.
10
+ class TermbaseLoader
11
+ # @param path [String, Pathname] path to termbase.yaml
12
+ def initialize(path)
13
+ @path = path.to_s
14
+ @data = nil
15
+ end
16
+
17
+ # @return [Hash<String, Hash>] raw termbase data keyed by code
18
+ def raw
19
+ @data ||= YAML.load_file(@path)
20
+ end
21
+
22
+ # @return [Array<String>] all codes in the termbase
23
+ def codes
24
+ raw.keys.map(&:to_s)
25
+ end
26
+
27
+ # Build a single ManagedConcept for the given code.
28
+ # @param code [String]
29
+ # @return [Glossarist::ManagedConcept, nil]
30
+ def get(code)
31
+ entry = raw[code] || raw[code.to_sym]
32
+ return nil unless entry.is_a?(Hash)
33
+
34
+ build_concept(code.to_s, entry)
35
+ end
36
+
37
+ private
38
+
39
+ def build_concept(code, entry)
40
+ concept = Glossarist::ManagedConcept.of_yaml(
41
+ "id" => code, "data" => { "id" => code },
42
+ )
43
+ concept.status = concept_status(entry)
44
+ concept.dates = concept_dates(entry)
45
+ concept.schema_version = "3"
46
+
47
+ build_localized(code, entry).each do |lc|
48
+ concept.add_l10n(lc)
49
+ end
50
+
51
+ concept
52
+ end
53
+
54
+ def concept_status(entry)
55
+ lang_data = first_language(entry)
56
+ StatusMapper.call(lang_data&.dig("entry_status"))
57
+ end
58
+
59
+ def concept_dates(entry)
60
+ lang_data = first_language(entry)
61
+ return [] unless lang_data
62
+
63
+ dates = []
64
+ if lang_data["date_accepted"]
65
+ dates << Glossarist::ConceptDate.new(
66
+ type: "accepted",
67
+ date: lang_data["date_accepted"],
68
+ )
69
+ end
70
+ if lang_data["date_amended"] && lang_data["date_amended"] != lang_data["date_accepted"]
71
+ dates << Glossarist::ConceptDate.new(
72
+ type: "amended",
73
+ date: lang_data["date_amended"],
74
+ )
75
+ end
76
+ dates
77
+ end
78
+
79
+ def first_language(entry)
80
+ entry.values.find { |v| v.is_a?(Hash) && v["terms"] }
81
+ end
82
+
83
+ def build_localized(code, entry)
84
+ result = []
85
+ entry.each do |lang, data|
86
+ next unless data.is_a?(Hash) && data["terms"]
87
+ lc = build_localized_concept(code, lang, data)
88
+ result << lc if lc
89
+ end
90
+ result
91
+ end
92
+
93
+ def build_localized_concept(code, lang, data)
94
+ cdata = Glossarist::ConceptData.new
95
+ cdata.id = code
96
+ cdata.language_code = lang
97
+ cdata.entry_status = StatusMapper.call(data["entry_status"])
98
+
99
+ cdata.terms = (data["terms"] || []).map do |t|
100
+ build_term_expression(t)
101
+ end
102
+
103
+ if data["definition"] && !data["definition"].empty?
104
+ cdata.definition = [Glossarist::DetailedDefinition.new(content: data["definition"])]
105
+ end
106
+
107
+ cdata.notes = Array(data["notes"]).map { |n| Glossarist::DetailedDefinition.new(content: n) }
108
+ cdata.examples = Array(data["examples"]).map { |e| Glossarist::DetailedDefinition.new(content: e) }
109
+
110
+ lc = Glossarist::LocalizedConcept.new
111
+ lc.id = code
112
+ lc.data = cdata
113
+ lc
114
+ end
115
+
116
+ def build_term_expression(term_data)
117
+ designation = term_data["designation"].to_s
118
+ parsed = TermMarkerParser.parse(designation)
119
+
120
+ expr = Glossarist::Designation::Expression.new(
121
+ designation: parsed.designation,
122
+ normative_status: term_data["normative_status"] || "preferred",
123
+ type: term_data["type"] || "expression",
124
+ )
125
+ expr.geographical_area = term_data["geographical_area"] if term_data["geographical_area"]
126
+ expr.field_of_application = term_data["usage_info"] || parsed.domain
127
+ expr.usage_info = parsed.usage_info if parsed.usage_info
128
+ expr.prefix = parsed.is_prefix if parsed.is_prefix
129
+
130
+ grammar = Glossarist::Designation::GrammarInfo.new
131
+ merged_genders = (Array(term_data["gender"]) + parsed.genders).flatten.compact.uniq
132
+ merged_numbers = (Array(term_data["plurality"]) + parsed.numbers).flatten.compact.uniq
133
+ grammar.gender = merged_genders if merged_genders.any?
134
+ grammar.number = merged_numbers if merged_numbers.any?
135
+ grammar.part_of_speech = parsed.part_of_speech if parsed.part_of_speech
136
+ expr.grammar_info = [grammar] if grammar.gender&.any? || grammar.number&.any? || grammar.part_of_speech
137
+ expr
138
+ end
139
+ end
140
+ end
141
+ end
@@ -0,0 +1,17 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Iev
4
+ module Reconciler
5
+ autoload :Change, "iev/reconciler/change"
6
+ autoload :ChangeSet, "iev/reconciler/change_set"
7
+ autoload :ConceptMerger, "iev/reconciler/concept_merger"
8
+ autoload :ContentDiffer, "iev/reconciler/content_differ"
9
+ autoload :LiveLoader, "iev/reconciler/live_loader"
10
+ autoload :Pipeline, "iev/reconciler/pipeline"
11
+ autoload :ReconciledConcept, "iev/reconciler/reconciled_concept"
12
+ autoload :Report, "iev/reconciler/report"
13
+ autoload :StatusMapper, "iev/reconciler/status_mapper"
14
+ autoload :TermMarkerParser, "iev/reconciler/term_marker_parser"
15
+ autoload :TermbaseLoader, "iev/reconciler/termbase_loader"
16
+ end
17
+ end