iev 0.4.6 → 0.4.8
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/.claude/scheduled_tasks.lock +1 -0
- data/.rubocop_todo.yml +5 -47
- data/CLAUDE.md +11 -1
- data/Gemfile +4 -2
- data/TODO.reconcile/01-design.md +57 -0
- data/TODO.reconcile/02-change-models.md +44 -0
- data/TODO.reconcile/02-termbase-loader.md +48 -0
- data/TODO.reconcile/03-live-loader.md +28 -0
- data/TODO.reconcile/03-status-mapper.md +24 -0
- data/TODO.reconcile/04-concept-merger.md +45 -0
- data/TODO.reconcile/04-termbase-loader.md +28 -0
- data/TODO.reconcile/05-content-diff.md +45 -0
- data/TODO.reconcile/05-live-loader.md +20 -0
- data/TODO.reconcile/06-content-differ.md +32 -0
- data/TODO.reconcile/06-v3-serializer.md +42 -0
- data/TODO.reconcile/07-cli-and-run.md +27 -0
- data/TODO.reconcile/07-concept-merger.md +32 -0
- data/TODO.reconcile/08-report.md +53 -0
- data/TODO.reconcile/09-pipeline.md +23 -0
- data/TODO.reconcile/10-script-and-run.md +15 -0
- data/data/locales/glossarist_enums.yml +705 -0
- data/iev.gemspec +4 -2
- data/lib/iev/bibliography_builder.rb +87 -0
- data/lib/iev/cli/command.rb +29 -0
- data/lib/iev/cli/command_helper.rb +1 -6
- data/lib/iev/config.rb +10 -1
- data/lib/iev/exporter.rb +37 -4
- data/lib/iev/figure_builder.rb +186 -0
- data/lib/iev/multi_doc_yaml.rb +60 -0
- data/lib/iev/reconciler/change.rb +29 -0
- data/lib/iev/reconciler/change_set.rb +48 -0
- data/lib/iev/reconciler/concept_merger.rb +130 -0
- data/lib/iev/reconciler/content_differ.rb +157 -0
- data/lib/iev/reconciler/live_loader.rb +163 -0
- data/lib/iev/reconciler/pipeline.rb +104 -0
- data/lib/iev/reconciler/reconciled_concept.rb +14 -0
- data/lib/iev/reconciler/report.rb +118 -0
- data/lib/iev/reconciler/status_mapper.rb +29 -0
- data/lib/iev/reconciler/term_marker_parser.rb +196 -0
- data/lib/iev/reconciler/termbase_loader.rb +141 -0
- data/lib/iev/reconciler.rb +17 -0
- data/lib/iev/scraper/browser.rb +155 -50
- data/lib/iev/scraper/page_parser.rb +116 -24
- data/lib/iev/source_parser.rb +12 -1
- data/lib/iev/subject_area_concepts.rb +4 -2
- data/lib/iev/utilities.rb +7 -5
- data/lib/iev/version.rb +1 -1
- data/lib/iev.rb +4 -0
- data/scripts/find_unparseable.rb +22 -0
- metadata +54 -6
|
@@ -0,0 +1,157 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Iev
|
|
4
|
+
module Reconciler
|
|
5
|
+
# Compares two Glossarist::ManagedConcept objects and produces a
|
|
6
|
+
# ChangeSet with field-level differences. This powers the change
|
|
7
|
+
# reporting that answers "what changed, from what to what?"
|
|
8
|
+
class ContentDiffer
|
|
9
|
+
# @param old_concept [Glossarist::ManagedConcept] the termbase snapshot
|
|
10
|
+
# @param new_concept [Glossarist::ManagedConcept] the live snapshot
|
|
11
|
+
# @param detected_at [String] ISO 8601 date the change was observed
|
|
12
|
+
# @return [ChangeSet]
|
|
13
|
+
def diff(old_concept, new_concept, detected_at:)
|
|
14
|
+
code = new_concept&.id || old_concept&.id
|
|
15
|
+
change_set = ChangeSet.new(code)
|
|
16
|
+
|
|
17
|
+
return change_set unless old_concept && new_concept
|
|
18
|
+
|
|
19
|
+
diff_status(change_set, old_concept, new_concept, detected_at)
|
|
20
|
+
diff_languages(change_set, old_concept, new_concept, detected_at)
|
|
21
|
+
diff_localized(change_set, old_concept, new_concept, detected_at)
|
|
22
|
+
|
|
23
|
+
change_set
|
|
24
|
+
end
|
|
25
|
+
|
|
26
|
+
private
|
|
27
|
+
|
|
28
|
+
def diff_status(cs, old_c, new_c, date)
|
|
29
|
+
old_status = old_c.status
|
|
30
|
+
new_status = new_c.status
|
|
31
|
+
return if old_status == new_status
|
|
32
|
+
|
|
33
|
+
cs.add(Change.new(
|
|
34
|
+
code: cs.code,
|
|
35
|
+
field: :status,
|
|
36
|
+
language: nil,
|
|
37
|
+
old_value: old_status,
|
|
38
|
+
new_value: new_status,
|
|
39
|
+
detected_at: date,
|
|
40
|
+
))
|
|
41
|
+
end
|
|
42
|
+
|
|
43
|
+
def diff_languages(cs, old_c, new_c, date)
|
|
44
|
+
old_langs = language_keys(old_c)
|
|
45
|
+
new_langs = language_keys(new_c)
|
|
46
|
+
|
|
47
|
+
(new_langs - old_langs).each do |lang|
|
|
48
|
+
cs.add(Change.new(
|
|
49
|
+
code: cs.code, field: :language_added, language: lang,
|
|
50
|
+
old_value: nil, new_value: "present", detected_at: date,
|
|
51
|
+
))
|
|
52
|
+
end
|
|
53
|
+
end
|
|
54
|
+
|
|
55
|
+
def diff_localized(cs, old_c, new_c, date)
|
|
56
|
+
common_langs = language_keys(old_c) & language_keys(new_c)
|
|
57
|
+
common_langs.each do |lang|
|
|
58
|
+
old_lc = localized_data(old_c, lang)
|
|
59
|
+
new_lc = localized_data(new_c, lang)
|
|
60
|
+
next unless old_lc && new_lc
|
|
61
|
+
|
|
62
|
+
diff_designation(cs, lang, old_lc, new_lc, date)
|
|
63
|
+
diff_definition(cs, lang, old_lc, new_lc, date)
|
|
64
|
+
diff_notes(cs, lang, old_lc, new_lc, date)
|
|
65
|
+
diff_examples(cs, lang, old_lc, new_lc, date)
|
|
66
|
+
end
|
|
67
|
+
end
|
|
68
|
+
|
|
69
|
+
def diff_designation(cs, lang, old_lc, new_lc, date)
|
|
70
|
+
old_term = extract_designation(old_lc)
|
|
71
|
+
new_term = extract_designation(new_lc)
|
|
72
|
+
return if normalize(old_term) == normalize(new_term)
|
|
73
|
+
|
|
74
|
+
cs.add(Change.new(
|
|
75
|
+
code: cs.code, field: :designation, language: lang,
|
|
76
|
+
old_value: old_term, new_value: new_term, detected_at: date,
|
|
77
|
+
))
|
|
78
|
+
end
|
|
79
|
+
|
|
80
|
+
def diff_definition(cs, lang, old_lc, new_lc, date)
|
|
81
|
+
old_def = extract_definition(old_lc)
|
|
82
|
+
new_def = extract_definition(new_lc)
|
|
83
|
+
return if normalize(old_def) == normalize(new_def)
|
|
84
|
+
|
|
85
|
+
cs.add(Change.new(
|
|
86
|
+
code: cs.code, field: :definition, language: lang,
|
|
87
|
+
old_value: old_def, new_value: new_def, detected_at: date,
|
|
88
|
+
))
|
|
89
|
+
end
|
|
90
|
+
|
|
91
|
+
def diff_notes(cs, lang, old_lc, new_lc, date)
|
|
92
|
+
old_notes = extract_notes(old_lc)
|
|
93
|
+
new_notes = extract_notes(new_lc)
|
|
94
|
+
return if old_notes == new_notes
|
|
95
|
+
|
|
96
|
+
cs.add(Change.new(
|
|
97
|
+
code: cs.code, field: :notes, language: lang,
|
|
98
|
+
old_value: old_notes.join(" | "), new_value: new_notes.join(" | "),
|
|
99
|
+
detected_at: date,
|
|
100
|
+
))
|
|
101
|
+
end
|
|
102
|
+
|
|
103
|
+
def diff_examples(cs, lang, old_lc, new_lc, date)
|
|
104
|
+
old_ex = extract_examples(old_lc)
|
|
105
|
+
new_ex = extract_examples(new_lc)
|
|
106
|
+
return if old_ex == new_ex
|
|
107
|
+
|
|
108
|
+
cs.add(Change.new(
|
|
109
|
+
code: cs.code, field: :examples, language: lang,
|
|
110
|
+
old_value: old_ex.join(" | "), new_value: new_ex.join(" | "),
|
|
111
|
+
detected_at: date,
|
|
112
|
+
))
|
|
113
|
+
end
|
|
114
|
+
|
|
115
|
+
# --- extraction helpers ---
|
|
116
|
+
|
|
117
|
+
def language_keys(concept)
|
|
118
|
+
concept.localized_concepts&.keys || []
|
|
119
|
+
end
|
|
120
|
+
|
|
121
|
+
def localized_data(concept, lang)
|
|
122
|
+
lc = concept.localization(lang)
|
|
123
|
+
lc&.data
|
|
124
|
+
end
|
|
125
|
+
|
|
126
|
+
def extract_designation(cdata)
|
|
127
|
+
cdata&.terms&.first&.designation.to_s
|
|
128
|
+
end
|
|
129
|
+
|
|
130
|
+
def extract_definition(cdata)
|
|
131
|
+
cdata&.definition&.first&.content.to_s
|
|
132
|
+
end
|
|
133
|
+
|
|
134
|
+
def extract_notes(cdata)
|
|
135
|
+
Array(cdata&.notes).map(&:content)
|
|
136
|
+
end
|
|
137
|
+
|
|
138
|
+
def extract_examples(cdata)
|
|
139
|
+
Array(cdata&.examples).map(&:content)
|
|
140
|
+
end
|
|
141
|
+
|
|
142
|
+
def normalize(str)
|
|
143
|
+
return "" unless str
|
|
144
|
+
str
|
|
145
|
+
.to_s
|
|
146
|
+
.unicode_normalize(:nfc)
|
|
147
|
+
.gsub(/<[^>]+>/, "")
|
|
148
|
+
.gsub(/[ ]/, "")
|
|
149
|
+
.gsub(/[""″‴„‟‚‛]/, '"')
|
|
150
|
+
.gsub(/[''‚‛]/, "'")
|
|
151
|
+
.gsub(/\s+/, " ")
|
|
152
|
+
.strip
|
|
153
|
+
.downcase
|
|
154
|
+
end
|
|
155
|
+
end
|
|
156
|
+
end
|
|
157
|
+
end
|
|
@@ -0,0 +1,163 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "nokogiri"
|
|
4
|
+
|
|
5
|
+
module Iev
|
|
6
|
+
module Reconciler
|
|
7
|
+
# Indexes live HTML pages and builds Glossarist::ManagedConcept
|
|
8
|
+
# objects on demand per code. Uses Iev::Scraper::PageParser which
|
|
9
|
+
# runs the full semantic enrichment pipeline (HTML → AsciiDoc with
|
|
10
|
+
# stem:[] for math, {{urn:...}} for cross-refs, etc.), so parsed
|
|
11
|
+
# output matches the format used in termbase.yaml.
|
|
12
|
+
class LiveLoader
|
|
13
|
+
# @param pages_dir [String, Pathname] directory containing *.html pages
|
|
14
|
+
def initialize(pages_dir)
|
|
15
|
+
@pages_dir = pages_dir.to_s
|
|
16
|
+
@index = nil
|
|
17
|
+
end
|
|
18
|
+
|
|
19
|
+
def codes
|
|
20
|
+
index.keys
|
|
21
|
+
end
|
|
22
|
+
|
|
23
|
+
def get(code)
|
|
24
|
+
path = index[code]
|
|
25
|
+
return nil unless path
|
|
26
|
+
|
|
27
|
+
html = File.read(path, encoding: "utf-8")
|
|
28
|
+
parse_concept(code, html)
|
|
29
|
+
end
|
|
30
|
+
|
|
31
|
+
private
|
|
32
|
+
|
|
33
|
+
def index
|
|
34
|
+
@index ||= Dir.glob(File.join(@pages_dir, "*.html")).each_with_object({}) do |path, h|
|
|
35
|
+
h[File.basename(path, ".html")] = path
|
|
36
|
+
end
|
|
37
|
+
end
|
|
38
|
+
|
|
39
|
+
def parse_concept(code, html)
|
|
40
|
+
doc = Nokogiri::HTML(html)
|
|
41
|
+
parsed = Iev::Scraper::PageParser.new(doc, code).parse
|
|
42
|
+
return nil unless parsed && parsed.dig("data", "localized_concepts")&.any?
|
|
43
|
+
|
|
44
|
+
build_concept(code, parsed["data"]["localized_concepts"])
|
|
45
|
+
end
|
|
46
|
+
|
|
47
|
+
def build_concept(code, localized_data)
|
|
48
|
+
concept = Glossarist::ManagedConcept.of_yaml(
|
|
49
|
+
"id" => code, "data" => { "id" => code },
|
|
50
|
+
)
|
|
51
|
+
concept.status = "valid"
|
|
52
|
+
concept.schema_version = "3"
|
|
53
|
+
|
|
54
|
+
localized_data.each do |lang, lc_data|
|
|
55
|
+
lc = build_localized_concept(code, lang, lc_data)
|
|
56
|
+
concept.add_l10n(lc)
|
|
57
|
+
end
|
|
58
|
+
|
|
59
|
+
concept
|
|
60
|
+
end
|
|
61
|
+
|
|
62
|
+
def build_localized_concept(code, lang, lc_data)
|
|
63
|
+
cdata = Glossarist::ConceptData.new
|
|
64
|
+
cdata.id = code
|
|
65
|
+
cdata.language_code = lang
|
|
66
|
+
cdata.entry_status = "valid"
|
|
67
|
+
|
|
68
|
+
term = lc_data["term"].to_s
|
|
69
|
+
areas = lc_data["term_areas"] || {}
|
|
70
|
+
cdata.terms = if term.empty?
|
|
71
|
+
[]
|
|
72
|
+
else
|
|
73
|
+
TermMarkerParser.parse_multiple(term).map do |parsed|
|
|
74
|
+
build_expression(parsed, areas[parsed.designation])
|
|
75
|
+
end
|
|
76
|
+
end
|
|
77
|
+
|
|
78
|
+
definition = lc_data["definition"]
|
|
79
|
+
if definition && !definition.empty?
|
|
80
|
+
parts = split_notes_examples(definition)
|
|
81
|
+
if parts[:definition] && !parts[:definition].empty?
|
|
82
|
+
cdata.definition = [Glossarist::DetailedDefinition.new(content: parts[:definition])]
|
|
83
|
+
end
|
|
84
|
+
cdata.notes = parts[:notes].map { |n| Glossarist::DetailedDefinition.new(content: n) }
|
|
85
|
+
cdata.examples = parts[:examples].map { |e| Glossarist::DetailedDefinition.new(content: e) }
|
|
86
|
+
end
|
|
87
|
+
|
|
88
|
+
lc = Glossarist::LocalizedConcept.new
|
|
89
|
+
lc.id = code
|
|
90
|
+
lc.data = cdata
|
|
91
|
+
lc
|
|
92
|
+
end
|
|
93
|
+
|
|
94
|
+
def build_expression(parsed, area = nil)
|
|
95
|
+
expr = Glossarist::Designation::Expression.new(
|
|
96
|
+
designation: parsed.designation,
|
|
97
|
+
normative_status: "preferred",
|
|
98
|
+
)
|
|
99
|
+
expr.geographical_area = area if area
|
|
100
|
+
expr.field_of_application = parsed.domain if parsed.domain
|
|
101
|
+
expr.usage_info = parsed.usage_info if parsed.usage_info
|
|
102
|
+
expr.prefix = parsed.is_prefix if parsed.is_prefix
|
|
103
|
+
|
|
104
|
+
grammar = build_grammar_info(parsed)
|
|
105
|
+
expr.grammar_info = [grammar] if grammar
|
|
106
|
+
expr
|
|
107
|
+
end
|
|
108
|
+
|
|
109
|
+
def build_related_concepts(parsed)
|
|
110
|
+
return [] unless parsed.related_refs&.any?
|
|
111
|
+
|
|
112
|
+
parsed.related_refs.map do |code|
|
|
113
|
+
Glossarist::RelatedConcept.new(
|
|
114
|
+
type: "see",
|
|
115
|
+
ref: { "source" => "IEV", "id" => code },
|
|
116
|
+
)
|
|
117
|
+
end
|
|
118
|
+
end
|
|
119
|
+
|
|
120
|
+
def build_grammar_info(parsed)
|
|
121
|
+
has_gender = parsed.genders&.any?
|
|
122
|
+
has_number = parsed.numbers&.any?
|
|
123
|
+
has_pos = !parsed.part_of_speech.nil?
|
|
124
|
+
return nil unless has_gender || has_number || has_pos
|
|
125
|
+
|
|
126
|
+
grammar = Glossarist::Designation::GrammarInfo.new
|
|
127
|
+
grammar.gender = parsed.genders if has_gender
|
|
128
|
+
grammar.number = parsed.numbers if has_number
|
|
129
|
+
grammar.part_of_speech = parsed.part_of_speech if has_pos
|
|
130
|
+
grammar
|
|
131
|
+
end
|
|
132
|
+
|
|
133
|
+
NOTE_RE = /^(Note\s+\d+\s+to\s+entry:.*)$/i
|
|
134
|
+
EXAMPLE_RE = /^(Example\s*:.*)$/i
|
|
135
|
+
NOTE_PREFIX_RE = /^Note\s+\d+\s+to\s+entry:\s*/i
|
|
136
|
+
EXAMPLE_PREFIX_RE = /^Example\s*:\s*/i
|
|
137
|
+
|
|
138
|
+
def split_notes_examples(text)
|
|
139
|
+
notes = []
|
|
140
|
+
examples = []
|
|
141
|
+
|
|
142
|
+
lines = text.split("\n")
|
|
143
|
+
definition_lines = []
|
|
144
|
+
|
|
145
|
+
lines.each do |line|
|
|
146
|
+
stripped = line.strip
|
|
147
|
+
if stripped.match?(NOTE_RE)
|
|
148
|
+
notes << stripped.sub(NOTE_PREFIX_RE, "")
|
|
149
|
+
elsif stripped.match?(EXAMPLE_RE)
|
|
150
|
+
examples << stripped.sub(EXAMPLE_PREFIX_RE, "")
|
|
151
|
+
else
|
|
152
|
+
definition_lines << line
|
|
153
|
+
end
|
|
154
|
+
end
|
|
155
|
+
|
|
156
|
+
definition = definition_lines.join("\n").strip
|
|
157
|
+
definition = nil if definition.empty?
|
|
158
|
+
|
|
159
|
+
{ definition: definition, notes: notes, examples: examples }
|
|
160
|
+
end
|
|
161
|
+
end
|
|
162
|
+
end
|
|
163
|
+
end
|
|
@@ -0,0 +1,104 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "date"
|
|
4
|
+
require "fileutils"
|
|
5
|
+
require "yaml"
|
|
6
|
+
|
|
7
|
+
module Iev
|
|
8
|
+
module Reconciler
|
|
9
|
+
# Orchestrates the full reconciliation by streaming through codes
|
|
10
|
+
# one at a time. Each code is loaded → merged → serialized → discarded,
|
|
11
|
+
# so memory usage is bounded regardless of dataset size.
|
|
12
|
+
class Pipeline
|
|
13
|
+
attr_reader :stats
|
|
14
|
+
|
|
15
|
+
# @param termbase_path [String] path to termbase.yaml
|
|
16
|
+
# @param pages_dir [String] directory with mirrored HTML pages
|
|
17
|
+
# @param output_dir [String] where to write concepts + reports
|
|
18
|
+
# @param detected_at [String] ISO 8601 date for detected changes
|
|
19
|
+
def initialize(termbase_path:, pages_dir:, output_dir:,
|
|
20
|
+
detected_at: Date.today.iso8601)
|
|
21
|
+
@termbase_path = termbase_path
|
|
22
|
+
@pages_dir = pages_dir
|
|
23
|
+
@output_dir = output_dir
|
|
24
|
+
@detected_at = detected_at
|
|
25
|
+
@stats = {}
|
|
26
|
+
end
|
|
27
|
+
|
|
28
|
+
# Run the full pipeline.
|
|
29
|
+
# @return [void]
|
|
30
|
+
def run
|
|
31
|
+
FileUtils.mkdir_p(File.join(@output_dir, "concepts"))
|
|
32
|
+
FileUtils.mkdir_p(File.join(@output_dir, "report"))
|
|
33
|
+
|
|
34
|
+
warn "Indexing sources..."
|
|
35
|
+
termbase = TermbaseLoader.new(@termbase_path)
|
|
36
|
+
live = LiveLoader.new(@pages_dir)
|
|
37
|
+
all_codes = (termbase.codes + live.codes).uniq.sort
|
|
38
|
+
warn " #{termbase.codes.size} termbase + #{live.codes.size} live = #{all_codes.size} total"
|
|
39
|
+
|
|
40
|
+
merger = ConceptMerger.new
|
|
41
|
+
reconciled = []
|
|
42
|
+
errors = []
|
|
43
|
+
|
|
44
|
+
all_codes.each_with_index do |code, idx|
|
|
45
|
+
begin
|
|
46
|
+
rc = merger.merge(
|
|
47
|
+
code: code,
|
|
48
|
+
termbase_concept: termbase.get(code),
|
|
49
|
+
live_concept: live.get(code),
|
|
50
|
+
detected_at: @detected_at,
|
|
51
|
+
)
|
|
52
|
+
|
|
53
|
+
if rc
|
|
54
|
+
reconciled << rc
|
|
55
|
+
save_concept(rc.managed_concept)
|
|
56
|
+
end
|
|
57
|
+
rescue StandardError => e
|
|
58
|
+
errors << { code: code, error: "#{e.class}: #{e.message}" }
|
|
59
|
+
warn " ERROR on #{code}: #{e.message[0, 100]}"
|
|
60
|
+
end
|
|
61
|
+
|
|
62
|
+
if (idx + 1) % 1000 == 0
|
|
63
|
+
warn " #{idx + 1}/#{all_codes.size}... (#{errors.size} errors)"
|
|
64
|
+
end
|
|
65
|
+
end
|
|
66
|
+
|
|
67
|
+
compute_stats(reconciled)
|
|
68
|
+
compute_error_stats(errors) if errors.any?
|
|
69
|
+
Report.new(reconciled).write_to(File.join(@output_dir, "report"))
|
|
70
|
+
warn "Done."
|
|
71
|
+
end
|
|
72
|
+
|
|
73
|
+
private
|
|
74
|
+
|
|
75
|
+
def save_concept(concept)
|
|
76
|
+
path = File.join(@output_dir, "concepts", "#{concept.id}.yaml")
|
|
77
|
+
Iev::MultiDocYaml.write(path, Iev::MultiDocYaml.parts_for(concept))
|
|
78
|
+
end
|
|
79
|
+
|
|
80
|
+
def compute_stats(reconciled)
|
|
81
|
+
@stats = {
|
|
82
|
+
total: reconciled.size,
|
|
83
|
+
in_both: reconciled.count { |r| r.source == :both },
|
|
84
|
+
termbase_only: reconciled.count { |r| r.source == :termbase_only },
|
|
85
|
+
live_only: reconciled.count { |r| r.source == :live_only },
|
|
86
|
+
changed: reconciled.count { |r| !r.change_set.empty? },
|
|
87
|
+
}
|
|
88
|
+
warn ""
|
|
89
|
+
warn "Results:"
|
|
90
|
+
warn " In both (merged): #{@stats[:in_both]}"
|
|
91
|
+
warn " Termbase only (retired): #{@stats[:termbase_only]}"
|
|
92
|
+
warn " Live only (new): #{@stats[:live_only]}"
|
|
93
|
+
warn " Concepts with changes: #{@stats[:changed]}"
|
|
94
|
+
warn " Total: #{@stats[:total]}"
|
|
95
|
+
end
|
|
96
|
+
|
|
97
|
+
def compute_error_stats(errors)
|
|
98
|
+
warn ""
|
|
99
|
+
warn "Errors: #{errors.size}"
|
|
100
|
+
errors.first(10).each { |e| warn " #{e[:code]}: #{e[:error]}" }
|
|
101
|
+
end
|
|
102
|
+
end
|
|
103
|
+
end
|
|
104
|
+
end
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Iev
|
|
4
|
+
module Reconciler
|
|
5
|
+
# Wraps a reconciled ManagedConcept with its ChangeSet and origin.
|
|
6
|
+
# Value object produced by ConceptMerger, consumed by Pipeline + Report.
|
|
7
|
+
#
|
|
8
|
+
# @attr managed_concept [Glossarist::ManagedConcept]
|
|
9
|
+
# @attr change_set [ChangeSet] field-level diffs detected
|
|
10
|
+
# @attr source [Symbol] :both | :termbase_only | :live_only
|
|
11
|
+
ReconciledConcept = Struct.new(:managed_concept, :change_set, :source,
|
|
12
|
+
keyword_init: true)
|
|
13
|
+
end
|
|
14
|
+
end
|
|
@@ -0,0 +1,118 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "csv"
|
|
4
|
+
require "yaml"
|
|
5
|
+
|
|
6
|
+
module Iev
|
|
7
|
+
module Reconciler
|
|
8
|
+
# Generates dataset-level change reports from reconciled concepts.
|
|
9
|
+
# Answers "what changed across the entire IEV?" with machine-readable
|
|
10
|
+
# summary files.
|
|
11
|
+
class Report
|
|
12
|
+
# @param reconciled [Array<ReconciledConcept>] all reconciled concepts
|
|
13
|
+
def initialize(reconciled)
|
|
14
|
+
@reconciled = reconciled
|
|
15
|
+
end
|
|
16
|
+
|
|
17
|
+
# Write all report files to the given directory.
|
|
18
|
+
# @param dir [String, Pathname]
|
|
19
|
+
def write_to(dir)
|
|
20
|
+
require "fileutils"
|
|
21
|
+
FileUtils.mkdir_p(dir)
|
|
22
|
+
|
|
23
|
+
write_summary(File.join(dir, "summary.yaml"))
|
|
24
|
+
write_changes_csv(File.join(dir, "changes.csv"))
|
|
25
|
+
write_retired(File.join(dir, "retired.yaml"))
|
|
26
|
+
write_new_concepts(File.join(dir, "new_concepts.yaml"))
|
|
27
|
+
end
|
|
28
|
+
|
|
29
|
+
# @return [Hash] aggregate statistics
|
|
30
|
+
def summary
|
|
31
|
+
{
|
|
32
|
+
total_concepts: @reconciled.size,
|
|
33
|
+
sources: source_counts,
|
|
34
|
+
changes: change_stats,
|
|
35
|
+
}
|
|
36
|
+
end
|
|
37
|
+
|
|
38
|
+
private
|
|
39
|
+
|
|
40
|
+
def write_summary(path)
|
|
41
|
+
File.write(path, YAML.dump(summary))
|
|
42
|
+
end
|
|
43
|
+
|
|
44
|
+
def write_changes_csv(path)
|
|
45
|
+
CSV.open(path, "w") do |csv|
|
|
46
|
+
csv << %w[code section field language detected_at old_value new_value]
|
|
47
|
+
@reconciled.each do |rc|
|
|
48
|
+
rc.change_set.each do |change|
|
|
49
|
+
csv << [
|
|
50
|
+
change.code,
|
|
51
|
+
section_of(change.code),
|
|
52
|
+
change.field,
|
|
53
|
+
change.language,
|
|
54
|
+
change.detected_at,
|
|
55
|
+
truncate(change.old_value),
|
|
56
|
+
truncate(change.new_value),
|
|
57
|
+
]
|
|
58
|
+
end
|
|
59
|
+
end
|
|
60
|
+
end
|
|
61
|
+
end
|
|
62
|
+
|
|
63
|
+
def write_retired(path)
|
|
64
|
+
retired = @reconciled
|
|
65
|
+
.select { |rc| rc.source == :termbase_only && rc.managed_concept }
|
|
66
|
+
.map { |rc| rc.managed_concept.id }
|
|
67
|
+
File.write(path, YAML.dump(retired))
|
|
68
|
+
end
|
|
69
|
+
|
|
70
|
+
def write_new_concepts(path)
|
|
71
|
+
new_concepts = @reconciled
|
|
72
|
+
.select { |rc| rc.source == :live_only && rc.managed_concept }
|
|
73
|
+
.map { |rc| rc.managed_concept.id }
|
|
74
|
+
File.write(path, YAML.dump(new_concepts))
|
|
75
|
+
end
|
|
76
|
+
|
|
77
|
+
def source_counts
|
|
78
|
+
@reconciled.group_by(&:source).transform_values(&:size)
|
|
79
|
+
end
|
|
80
|
+
|
|
81
|
+
def change_stats
|
|
82
|
+
all_changes = @reconciled.flat_map { |rc| rc.change_set.to_a }
|
|
83
|
+
{
|
|
84
|
+
total: all_changes.size,
|
|
85
|
+
by_field: tally_by(all_changes, :field),
|
|
86
|
+
by_language: tally_by(all_changes, :language),
|
|
87
|
+
by_section: tally_by_section(all_changes),
|
|
88
|
+
}
|
|
89
|
+
end
|
|
90
|
+
|
|
91
|
+
def tally_by(changes, attr)
|
|
92
|
+
changes
|
|
93
|
+
.group_by { |c| c.send(attr) }
|
|
94
|
+
.transform_values(&:size)
|
|
95
|
+
.sort_by { |_, v| -v }
|
|
96
|
+
.to_h
|
|
97
|
+
end
|
|
98
|
+
|
|
99
|
+
def tally_by_section(changes)
|
|
100
|
+
changes
|
|
101
|
+
.group_by { |c| section_of(c.code) }
|
|
102
|
+
.transform_values(&:size)
|
|
103
|
+
.sort_by { |_, v| -v }
|
|
104
|
+
.to_h
|
|
105
|
+
end
|
|
106
|
+
|
|
107
|
+
def section_of(code)
|
|
108
|
+
code.to_s.rpartition("-").first
|
|
109
|
+
end
|
|
110
|
+
|
|
111
|
+
def truncate(value, max = 200)
|
|
112
|
+
return "" if value.nil?
|
|
113
|
+
value = value.to_s
|
|
114
|
+
value.size > max ? value[0, max] + "..." : value
|
|
115
|
+
end
|
|
116
|
+
end
|
|
117
|
+
end
|
|
118
|
+
end
|
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Iev
|
|
4
|
+
module Reconciler
|
|
5
|
+
# Maps termbase entry_status strings to Glossarist V3 ConceptStatus
|
|
6
|
+
# enum values. Stateless and pure.
|
|
7
|
+
module StatusMapper
|
|
8
|
+
MAP = {
|
|
9
|
+
"Standard" => "valid",
|
|
10
|
+
"Published" => "valid",
|
|
11
|
+
"Effective" => "valid",
|
|
12
|
+
"Draft" => "draft",
|
|
13
|
+
"Not Valid" => "notValid",
|
|
14
|
+
"Superseded" => "superseded",
|
|
15
|
+
"Retired" => "retired",
|
|
16
|
+
}.freeze
|
|
17
|
+
|
|
18
|
+
DEFAULT = "valid"
|
|
19
|
+
|
|
20
|
+
module_function
|
|
21
|
+
|
|
22
|
+
# @param entry_status [String, nil]
|
|
23
|
+
# @return [String] a valid ConceptStatus value
|
|
24
|
+
def call(entry_status)
|
|
25
|
+
MAP[entry_status.to_s] || DEFAULT
|
|
26
|
+
end
|
|
27
|
+
end
|
|
28
|
+
end
|
|
29
|
+
end
|