uniword 1.5.0 → 1.5.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +27 -0
- data/lib/uniword/cli/main.rb +134 -0
- data/lib/uniword/cli/review_cli.rb +34 -0
- data/lib/uniword/configuration.rb +17 -0
- data/lib/uniword/docx/deterministic_output.rb +47 -0
- data/lib/uniword/docx.rb +1 -0
- data/lib/uniword/find_replace/body_scope.rb +23 -0
- data/lib/uniword/find_replace/comment_scope.rb +23 -0
- data/lib/uniword/find_replace/endnote_scope.rb +23 -0
- data/lib/uniword/find_replace/engine.rb +84 -0
- data/lib/uniword/find_replace/footer_scope.rb +24 -0
- data/lib/uniword/find_replace/footnote_scope.rb +23 -0
- data/lib/uniword/find_replace/header_scope.rb +24 -0
- data/lib/uniword/find_replace/matcher.rb +23 -0
- data/lib/uniword/find_replace/paragraph_walker.rb +61 -0
- data/lib/uniword/find_replace/regex_matcher.rb +38 -0
- data/lib/uniword/find_replace/result.rb +54 -0
- data/lib/uniword/find_replace/scope.rb +98 -0
- data/lib/uniword/find_replace/string_matcher.rb +72 -0
- data/lib/uniword/find_replace/styles_scope.rb +38 -0
- data/lib/uniword/find_replace.rb +29 -0
- data/lib/uniword/infrastructure/zip_packager.rb +49 -3
- data/lib/uniword/lint/builtin_rules/banned_words.rb +35 -0
- data/lib/uniword/lint/builtin_rules/max_paragraph_length.rb +32 -0
- data/lib/uniword/lint/builtin_rules/require_body.rb +23 -0
- data/lib/uniword/lint/builtin_rules/required_style.rb +31 -0
- data/lib/uniword/lint/builtin_rules.rb +23 -0
- data/lib/uniword/lint/engine.rb +32 -0
- data/lib/uniword/lint/result.rb +59 -0
- data/lib/uniword/lint/rule.rb +77 -0
- data/lib/uniword/lint/ruleset.rb +65 -0
- data/lib/uniword/lint.rb +17 -0
- data/lib/uniword/redact/engine.rb +52 -0
- data/lib/uniword/redact/pattern.rb +23 -0
- data/lib/uniword/redact/pattern_library.rb +60 -0
- data/lib/uniword/redact/result.rb +42 -0
- data/lib/uniword/redact.rb +15 -0
- data/lib/uniword/version.rb +1 -1
- data/lib/uniword/wordprocessingml/document_styling.rb +91 -0
- data/lib/uniword/wordprocessingml/settings.rb +2 -0
- data/lib/uniword/wordprocessingml/track_changes.rb +19 -0
- data/lib/uniword/wordprocessingml.rb +1 -0
- data/lib/uniword.rb +10 -0
- metadata +34 -2
|
@@ -0,0 +1,54 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Uniword
|
|
4
|
+
module FindReplace
|
|
5
|
+
# Aggregated result of a find-replace pass.
|
|
6
|
+
#
|
|
7
|
+
# Counts total substitutions and breaks the count down by scope
|
|
8
|
+
# so callers can see where matches happened. Returned by
|
|
9
|
+
# `Engine#run` and `DocumentRoot#find_replace`.
|
|
10
|
+
#
|
|
11
|
+
# @example
|
|
12
|
+
# result = engine.run
|
|
13
|
+
# result.count # => 7
|
|
14
|
+
# result.by_scope # => { body: 5, headers: 2 }
|
|
15
|
+
# result.scopes_touched # => [:body, :headers]
|
|
16
|
+
class Result
|
|
17
|
+
attr_reader :by_scope
|
|
18
|
+
|
|
19
|
+
def initialize
|
|
20
|
+
@by_scope = Hash.new { |hash, key| hash[key] = 0 }
|
|
21
|
+
end
|
|
22
|
+
|
|
23
|
+
# Increment the count for one scope.
|
|
24
|
+
#
|
|
25
|
+
# @param scope [Symbol] scope name (e.g. :body, :headers)
|
|
26
|
+
# @param substitutions [Integer] substitutions added
|
|
27
|
+
# @return [void]
|
|
28
|
+
def add(scope, substitutions)
|
|
29
|
+
@by_scope[scope] += substitutions
|
|
30
|
+
end
|
|
31
|
+
|
|
32
|
+
# Total substitutions across all scopes.
|
|
33
|
+
#
|
|
34
|
+
# @return [Integer]
|
|
35
|
+
def count
|
|
36
|
+
@by_scope.values.sum
|
|
37
|
+
end
|
|
38
|
+
|
|
39
|
+
# Scopes that produced at least one substitution.
|
|
40
|
+
#
|
|
41
|
+
# @return [Array<Symbol>]
|
|
42
|
+
def scopes_touched
|
|
43
|
+
@by_scope.select { |_, c| c.positive? }.keys
|
|
44
|
+
end
|
|
45
|
+
|
|
46
|
+
# True when zero substitutions were made.
|
|
47
|
+
#
|
|
48
|
+
# @return [Boolean]
|
|
49
|
+
def empty?
|
|
50
|
+
count.zero?
|
|
51
|
+
end
|
|
52
|
+
end
|
|
53
|
+
end
|
|
54
|
+
end
|
|
@@ -0,0 +1,98 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Uniword
|
|
4
|
+
module FindReplace
|
|
5
|
+
# Abstract scope: knows how to enumerate text-bearing nodes in
|
|
6
|
+
# one part of the package and run a matcher over them.
|
|
7
|
+
#
|
|
8
|
+
# Each subclass implements `each_text_node`, yielding
|
|
9
|
+
# `[holder, accessor]` pairs where `holder` is the object whose
|
|
10
|
+
# text we read/write and `accessor` is a `TextAccessor` value
|
|
11
|
+
# object with `value` and `value=` methods.
|
|
12
|
+
#
|
|
13
|
+
# Open/closed: a new scope = a new subclass + registration in
|
|
14
|
+
# `Engine::SCOPE_REGISTRY`. Engine and other scopes are
|
|
15
|
+
# unchanged.
|
|
16
|
+
class Scope
|
|
17
|
+
# @param document [Wordprocessingml::DocumentRoot]
|
|
18
|
+
def initialize(document)
|
|
19
|
+
@document = document
|
|
20
|
+
end
|
|
21
|
+
|
|
22
|
+
# @return [Symbol] scope name (e.g. :body, :headers)
|
|
23
|
+
def name
|
|
24
|
+
raise NotImplementedError
|
|
25
|
+
end
|
|
26
|
+
|
|
27
|
+
# Yield `[holder, accessor]` for every text node this scope
|
|
28
|
+
# covers. Engine applies the matcher to each accessor.
|
|
29
|
+
#
|
|
30
|
+
# @yieldparam holder [Wordprocessingml::Text] the Text element
|
|
31
|
+
# being read/written (its `content` attribute holds the string)
|
|
32
|
+
# @yieldparam accessor [Scope::TextAccessor]
|
|
33
|
+
# @return [void]
|
|
34
|
+
def each_text_node
|
|
35
|
+
raise NotImplementedError
|
|
36
|
+
end
|
|
37
|
+
|
|
38
|
+
protected
|
|
39
|
+
|
|
40
|
+
# Yield every Text element inside a run. A run carries its
|
|
41
|
+
# `<w:t>` Text element on the `text` accessor (lutaml-model
|
|
42
|
+
# returns the instance directly even when the attribute is
|
|
43
|
+
# declared as a collection).
|
|
44
|
+
#
|
|
45
|
+
# @param run [Wordprocessingml::Run, nil]
|
|
46
|
+
# @yieldparam text_element [Wordprocessingml::Text]
|
|
47
|
+
# @yieldparam accessor [TextAccessor]
|
|
48
|
+
# @return [void]
|
|
49
|
+
def each_text_in_run(run)
|
|
50
|
+
return unless run
|
|
51
|
+
|
|
52
|
+
text_element = run.text
|
|
53
|
+
return unless text_element
|
|
54
|
+
|
|
55
|
+
accessor = TextAccessor.new(
|
|
56
|
+
-> { text_element.content },
|
|
57
|
+
->(value) { text_element.content = value },
|
|
58
|
+
)
|
|
59
|
+
yield text_element, accessor
|
|
60
|
+
end
|
|
61
|
+
|
|
62
|
+
# Walk every paragraph in `containers` and yield each run's
|
|
63
|
+
# text elements. Shared by body / headers / footers / footnotes
|
|
64
|
+
# / endnotes / comments scopes.
|
|
65
|
+
#
|
|
66
|
+
# @param containers [Enumerable<#paragraphs, #tables,
|
|
67
|
+
# #structured_document_tags>]
|
|
68
|
+
# @yieldparam text_element [Wordprocessingml::Text]
|
|
69
|
+
# @yieldparam accessor [TextAccessor]
|
|
70
|
+
# @return [void]
|
|
71
|
+
def each_text_in_containers(containers)
|
|
72
|
+
ParagraphWalker.each_paragraph(containers) do |paragraph|
|
|
73
|
+
paragraph.runs&.each do |run|
|
|
74
|
+
each_text_in_run(run) { |*a| yield(*a) }
|
|
75
|
+
end
|
|
76
|
+
end
|
|
77
|
+
end
|
|
78
|
+
|
|
79
|
+
# Small value object with a `value` reader and `value=`
|
|
80
|
+
# writer, backed by lambdas. Avoids `instance_variable_set` /
|
|
81
|
+
# `send`; lets callers wire read/write however they like.
|
|
82
|
+
class TextAccessor
|
|
83
|
+
def initialize(reader, writer)
|
|
84
|
+
@reader = reader
|
|
85
|
+
@writer = writer
|
|
86
|
+
end
|
|
87
|
+
|
|
88
|
+
def value
|
|
89
|
+
@reader.call
|
|
90
|
+
end
|
|
91
|
+
|
|
92
|
+
def value=(new_value)
|
|
93
|
+
@writer.call(new_value)
|
|
94
|
+
end
|
|
95
|
+
end
|
|
96
|
+
end
|
|
97
|
+
end
|
|
98
|
+
end
|
|
@@ -0,0 +1,72 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Uniword
|
|
4
|
+
module FindReplace
|
|
5
|
+
# Literal substring matcher. Replaces every non-overlapping
|
|
6
|
+
# occurrence of `pattern` with `replacement`.
|
|
7
|
+
class StringMatcher < Matcher
|
|
8
|
+
# @param pattern [String] literal substring to find
|
|
9
|
+
# @param replacement [String] replacement text
|
|
10
|
+
# @param ignore_case [Boolean] match case-insensitively
|
|
11
|
+
def initialize(pattern:, replacement:, ignore_case: false)
|
|
12
|
+
raise ArgumentError, "pattern cannot be empty" if pattern.empty?
|
|
13
|
+
|
|
14
|
+
@pattern = pattern
|
|
15
|
+
@replacement = replacement
|
|
16
|
+
@ignore_case = ignore_case
|
|
17
|
+
end
|
|
18
|
+
|
|
19
|
+
# @param text [String]
|
|
20
|
+
# @return [Array(String, Integer)]
|
|
21
|
+
def apply(text)
|
|
22
|
+
unless text.include?(@pattern) || matches_ignore_case?(text)
|
|
23
|
+
return [text,
|
|
24
|
+
0]
|
|
25
|
+
end
|
|
26
|
+
|
|
27
|
+
text.size
|
|
28
|
+
result = replace_all(text)
|
|
29
|
+
substitutions = count_substitutions(text, result)
|
|
30
|
+
[result, substitutions]
|
|
31
|
+
end
|
|
32
|
+
|
|
33
|
+
private
|
|
34
|
+
|
|
35
|
+
def matches_ignore_case?(text)
|
|
36
|
+
@ignore_case && text.downcase.include?(@pattern.downcase)
|
|
37
|
+
end
|
|
38
|
+
|
|
39
|
+
def replace_all(text)
|
|
40
|
+
if @ignore_case
|
|
41
|
+
pattern = /#{Regexp.escape(@pattern)}/i
|
|
42
|
+
text.gsub(pattern, @replacement)
|
|
43
|
+
else
|
|
44
|
+
text.gsub(@pattern, @replacement)
|
|
45
|
+
end
|
|
46
|
+
end
|
|
47
|
+
|
|
48
|
+
# Net change in match count from the substitution. Computed by
|
|
49
|
+
# walking the original text and counting non-overlapping
|
|
50
|
+
# matches; the gsub above cannot be relied on directly when
|
|
51
|
+
# ignore_case is on (the pattern escapes are regex-based).
|
|
52
|
+
def count_substitutions(original, _replacement)
|
|
53
|
+
pattern = @ignore_case ? /#{Regexp.escape(@pattern)}/i : nil
|
|
54
|
+
if @ignore_case
|
|
55
|
+
original.scan(pattern).size
|
|
56
|
+
else
|
|
57
|
+
count_literal(original)
|
|
58
|
+
end
|
|
59
|
+
end
|
|
60
|
+
|
|
61
|
+
def count_literal(text)
|
|
62
|
+
count = 0
|
|
63
|
+
idx = 0
|
|
64
|
+
while (idx = text.index(@pattern, idx))
|
|
65
|
+
count += 1
|
|
66
|
+
idx += @pattern.length
|
|
67
|
+
end
|
|
68
|
+
count
|
|
69
|
+
end
|
|
70
|
+
end
|
|
71
|
+
end
|
|
72
|
+
end
|
|
@@ -0,0 +1,38 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Uniword
|
|
4
|
+
module FindReplace
|
|
5
|
+
# Styles scope: text in style display names (`<w:name w:val=...>`
|
|
6
|
+
# inside `word/styles.xml`).
|
|
7
|
+
#
|
|
8
|
+
# Style identifiers (`w:styleId`) are deliberately not touched —
|
|
9
|
+
# renaming a styleId breaks every reference. Use
|
|
10
|
+
# `DocumentRoot#rename_style` for that, which keeps references
|
|
11
|
+
# intact.
|
|
12
|
+
class StylesScope < Scope
|
|
13
|
+
# @return [Symbol]
|
|
14
|
+
def name
|
|
15
|
+
:styles
|
|
16
|
+
end
|
|
17
|
+
|
|
18
|
+
# @yieldparam name_element [Wordprocessingml::StyleName]
|
|
19
|
+
# @yieldparam accessor [Scope::TextAccessor]
|
|
20
|
+
def each_text_node
|
|
21
|
+
styles = @document.styles_configuration&.styles
|
|
22
|
+
return unless styles
|
|
23
|
+
|
|
24
|
+
styles.each do |style|
|
|
25
|
+
name_element = style.name
|
|
26
|
+
next unless name_element
|
|
27
|
+
next unless name_element.val
|
|
28
|
+
|
|
29
|
+
accessor = TextAccessor.new(
|
|
30
|
+
-> { name_element.val },
|
|
31
|
+
->(value) { name_element.val = value },
|
|
32
|
+
)
|
|
33
|
+
yield name_element, accessor
|
|
34
|
+
end
|
|
35
|
+
end
|
|
36
|
+
end
|
|
37
|
+
end
|
|
38
|
+
end
|
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Uniword
|
|
4
|
+
# Find & replace over the text-bearing parts of a DOCX package.
|
|
5
|
+
#
|
|
6
|
+
# Public surface: `Engine` (orchestrator), `Scope` (per-part
|
|
7
|
+
# strategy), `Matcher` (literal or regex), `Result` (counts +
|
|
8
|
+
# per-scope breakdown).
|
|
9
|
+
#
|
|
10
|
+
# Open/closed: adding a new scope = new subclass of `Scope` plus
|
|
11
|
+
# registration. Adding a new matcher type = new subclass of
|
|
12
|
+
# `Matcher`. Engine unchanged in both cases.
|
|
13
|
+
module FindReplace
|
|
14
|
+
autoload :Engine, "#{__dir__}/find_replace/engine"
|
|
15
|
+
autoload :Matcher, "#{__dir__}/find_replace/matcher"
|
|
16
|
+
autoload :StringMatcher, "#{__dir__}/find_replace/string_matcher"
|
|
17
|
+
autoload :RegexMatcher, "#{__dir__}/find_replace/regex_matcher"
|
|
18
|
+
autoload :Scope, "#{__dir__}/find_replace/scope"
|
|
19
|
+
autoload :BodyScope, "#{__dir__}/find_replace/body_scope"
|
|
20
|
+
autoload :HeaderScope, "#{__dir__}/find_replace/header_scope"
|
|
21
|
+
autoload :FooterScope, "#{__dir__}/find_replace/footer_scope"
|
|
22
|
+
autoload :FootnoteScope, "#{__dir__}/find_replace/footnote_scope"
|
|
23
|
+
autoload :EndnoteScope, "#{__dir__}/find_replace/endnote_scope"
|
|
24
|
+
autoload :CommentScope, "#{__dir__}/find_replace/comment_scope"
|
|
25
|
+
autoload :StylesScope, "#{__dir__}/find_replace/styles_scope"
|
|
26
|
+
autoload :ParagraphWalker, "#{__dir__}/find_replace/paragraph_walker"
|
|
27
|
+
autoload :Result, "#{__dir__}/find_replace/result"
|
|
28
|
+
end
|
|
29
|
+
end
|
|
@@ -44,15 +44,19 @@ module Uniword
|
|
|
44
44
|
# causing put_next_entry to discard our Entry and create a fresh one.
|
|
45
45
|
temp_path = "#{output_path}.#{Process.pid}.tmp"
|
|
46
46
|
|
|
47
|
+
ordered_content = order_content(content)
|
|
48
|
+
fixed_time = deterministic_timestamp
|
|
49
|
+
|
|
47
50
|
was_zip64 = Zip.write_zip64_support
|
|
48
51
|
Zip.write_zip64_support = false
|
|
49
52
|
begin
|
|
50
53
|
Zip::OutputStream.open(temp_path) do |zos|
|
|
51
|
-
|
|
54
|
+
ordered_content.each do |entry_path, entry_content|
|
|
52
55
|
entry = Zip::Entry.new(temp_path, entry_path.to_s)
|
|
53
56
|
entry.internal_file_attributes = 0
|
|
54
57
|
entry.external_file_attributes = 0
|
|
55
58
|
entry.fstype = Zip::FSTYPE_FAT
|
|
59
|
+
entry.time = fixed_time if fixed_time
|
|
56
60
|
|
|
57
61
|
zos.put_next_entry(entry)
|
|
58
62
|
|
|
@@ -73,7 +77,7 @@ module Uniword
|
|
|
73
77
|
|
|
74
78
|
move_temp_to_output(temp_path, output_path)
|
|
75
79
|
ensure
|
|
76
|
-
|
|
80
|
+
remove_temp_file(temp_path)
|
|
77
81
|
end
|
|
78
82
|
|
|
79
83
|
# Add a file to an existing ZIP archive.
|
|
@@ -145,6 +149,29 @@ module Uniword
|
|
|
145
149
|
|
|
146
150
|
private
|
|
147
151
|
|
|
152
|
+
# When `Uniword.configuration.deterministic_output` is true,
|
|
153
|
+
# reorder entries (priority for OPC-required first, alphabetical
|
|
154
|
+
# for the rest). Otherwise return content unchanged (insertion
|
|
155
|
+
# order, which matches Word's behavior).
|
|
156
|
+
#
|
|
157
|
+
# @param content [Hash<String, String>]
|
|
158
|
+
# @return [Hash<String, String>] ordered hash
|
|
159
|
+
def order_content(content)
|
|
160
|
+
return content unless Uniword.configuration.deterministic_output
|
|
161
|
+
|
|
162
|
+
ordered_keys = Docx::DeterministicOutput.reorder_entries(content.keys)
|
|
163
|
+
ordered_keys.to_h { |k| [k, content[k]] }
|
|
164
|
+
end
|
|
165
|
+
|
|
166
|
+
# Fixed timestamp for deterministic mode; nil otherwise.
|
|
167
|
+
#
|
|
168
|
+
# @return [Time, nil]
|
|
169
|
+
def deterministic_timestamp
|
|
170
|
+
return nil unless Uniword.configuration.deterministic_output
|
|
171
|
+
|
|
172
|
+
Docx::DeterministicOutput::FIXED_TIMESTAMP
|
|
173
|
+
end
|
|
174
|
+
|
|
148
175
|
# Write content to a ZIP file using a temp file and atomic move.
|
|
149
176
|
# This avoids Windows file locking issues by ensuring we never
|
|
150
177
|
# write directly to the target file while it might be open.
|
|
@@ -184,7 +211,7 @@ module Uniword
|
|
|
184
211
|
|
|
185
212
|
move_temp_to_output(temp_path, output_path)
|
|
186
213
|
ensure
|
|
187
|
-
|
|
214
|
+
remove_temp_file(temp_path)
|
|
188
215
|
end
|
|
189
216
|
|
|
190
217
|
def move_temp_to_output(temp_path, output_path)
|
|
@@ -203,6 +230,25 @@ module Uniword
|
|
|
203
230
|
end
|
|
204
231
|
end
|
|
205
232
|
|
|
233
|
+
# Best-effort temp file removal with Windows-safe retries. AV
|
|
234
|
+
# scanners and the indexer briefly hold newly-written files; a
|
|
235
|
+
# single `FileUtils.rm_f` can return EACCES and leak the temp.
|
|
236
|
+
def remove_temp_file(temp_path)
|
|
237
|
+
return unless defined?(temp_path) && temp_path
|
|
238
|
+
return unless File.exist?(temp_path)
|
|
239
|
+
|
|
240
|
+
retries = 5
|
|
241
|
+
begin
|
|
242
|
+
FileUtils.rm_f(temp_path)
|
|
243
|
+
rescue Errno::EACCES
|
|
244
|
+
retries -= 1
|
|
245
|
+
if retries.positive?
|
|
246
|
+
sleep(0.3)
|
|
247
|
+
retry
|
|
248
|
+
end
|
|
249
|
+
end
|
|
250
|
+
end
|
|
251
|
+
|
|
206
252
|
# Validate the content hash.
|
|
207
253
|
#
|
|
208
254
|
# @param content [Object] The content to validate
|
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "set"
|
|
4
|
+
|
|
5
|
+
module Uniword
|
|
6
|
+
module Lint
|
|
7
|
+
module BuiltinRules
|
|
8
|
+
# Rule: paragraphs containing banned words trigger a finding.
|
|
9
|
+
class BannedWords < Rule
|
|
10
|
+
register :banned_words, self
|
|
11
|
+
|
|
12
|
+
# @param words [Array<String>] words to flag
|
|
13
|
+
def initialize(words:, **rest)
|
|
14
|
+
super(**rest)
|
|
15
|
+
@banned = Set.new(Array(words).map(&:downcase))
|
|
16
|
+
end
|
|
17
|
+
|
|
18
|
+
def check(document)
|
|
19
|
+
document.paragraphs.each_with_index do |paragraph, idx|
|
|
20
|
+
text = paragraph.text.to_s.downcase
|
|
21
|
+
@banned.each do |word|
|
|
22
|
+
next unless text.match?(/\b#{Regexp.escape(word)}\b/)
|
|
23
|
+
|
|
24
|
+
yield finding(
|
|
25
|
+
message: "Paragraph #{idx + 1} contains banned word " \
|
|
26
|
+
"'#{word}'",
|
|
27
|
+
path: "paragraph[#{idx}]",
|
|
28
|
+
)
|
|
29
|
+
end
|
|
30
|
+
end
|
|
31
|
+
end
|
|
32
|
+
end
|
|
33
|
+
end
|
|
34
|
+
end
|
|
35
|
+
end
|
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Uniword
|
|
4
|
+
module Lint
|
|
5
|
+
module BuiltinRules
|
|
6
|
+
# Rule: paragraphs longer than `max` words trigger a finding.
|
|
7
|
+
class MaxParagraphLength < Rule
|
|
8
|
+
register :max_paragraph_length, self
|
|
9
|
+
|
|
10
|
+
# @param max [Integer] word count threshold (default 200)
|
|
11
|
+
def initialize(max: 200, **rest)
|
|
12
|
+
super(**rest)
|
|
13
|
+
@max = max
|
|
14
|
+
end
|
|
15
|
+
|
|
16
|
+
def check(document)
|
|
17
|
+
document.paragraphs.each_with_index do |paragraph, idx|
|
|
18
|
+
text = paragraph.text.to_s
|
|
19
|
+
word_count = text.split.length
|
|
20
|
+
next if word_count <= @max
|
|
21
|
+
|
|
22
|
+
yield finding(
|
|
23
|
+
message: "Paragraph #{idx + 1} has #{word_count} words " \
|
|
24
|
+
"(max #{@max})",
|
|
25
|
+
path: "paragraph[#{idx}]",
|
|
26
|
+
)
|
|
27
|
+
end
|
|
28
|
+
end
|
|
29
|
+
end
|
|
30
|
+
end
|
|
31
|
+
end
|
|
32
|
+
end
|
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Uniword
|
|
4
|
+
module Lint
|
|
5
|
+
module BuiltinRules
|
|
6
|
+
# Rule: documents must have a non-empty body.
|
|
7
|
+
class RequireBody < Rule
|
|
8
|
+
register :require_body, self
|
|
9
|
+
|
|
10
|
+
def check(document)
|
|
11
|
+
body = document.body
|
|
12
|
+
paragraphs = body&.paragraphs || []
|
|
13
|
+
return if paragraphs.any?
|
|
14
|
+
|
|
15
|
+
yield finding(
|
|
16
|
+
message: "Document body is empty",
|
|
17
|
+
path: "word/document.xml",
|
|
18
|
+
)
|
|
19
|
+
end
|
|
20
|
+
end
|
|
21
|
+
end
|
|
22
|
+
end
|
|
23
|
+
end
|
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Uniword
|
|
4
|
+
module Lint
|
|
5
|
+
module BuiltinRules
|
|
6
|
+
# Rule: required style presence. Triggers when the named style
|
|
7
|
+
# is missing from the document.
|
|
8
|
+
class RequiredStyle < Rule
|
|
9
|
+
register :required_style, self
|
|
10
|
+
|
|
11
|
+
# @param style_id [String, Array<String>] required styleId(s)
|
|
12
|
+
def initialize(style_id:, **rest)
|
|
13
|
+
super(**rest)
|
|
14
|
+
@required = Array(style_id)
|
|
15
|
+
end
|
|
16
|
+
|
|
17
|
+
def check(document)
|
|
18
|
+
present = document.styles_configuration&.styles&.map(&:styleId) || []
|
|
19
|
+
@required.each do |id|
|
|
20
|
+
next if present.include?(id)
|
|
21
|
+
|
|
22
|
+
yield finding(
|
|
23
|
+
message: "Required style '#{id}' is missing",
|
|
24
|
+
path: "styles.xml",
|
|
25
|
+
)
|
|
26
|
+
end
|
|
27
|
+
end
|
|
28
|
+
end
|
|
29
|
+
end
|
|
30
|
+
end
|
|
31
|
+
end
|
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Uniword
|
|
4
|
+
module Lint
|
|
5
|
+
# Container module for builtin rule classes. Loaded eagerly by
|
|
6
|
+
# `Lint` so each class's `register` call fires.
|
|
7
|
+
module BuiltinRules
|
|
8
|
+
autoload :MaxParagraphLength,
|
|
9
|
+
"#{__dir__}/builtin_rules/max_paragraph_length"
|
|
10
|
+
autoload :BannedWords, "#{__dir__}/builtin_rules/banned_words"
|
|
11
|
+
autoload :RequiredStyle, "#{__dir__}/builtin_rules/required_style"
|
|
12
|
+
autoload :RequireBody, "#{__dir__}/builtin_rules/require_body"
|
|
13
|
+
|
|
14
|
+
# Touch every autoload to force-load + register each rule.
|
|
15
|
+
ALL = [
|
|
16
|
+
MaxParagraphLength,
|
|
17
|
+
BannedWords,
|
|
18
|
+
RequiredStyle,
|
|
19
|
+
RequireBody,
|
|
20
|
+
].freeze
|
|
21
|
+
end
|
|
22
|
+
end
|
|
23
|
+
end
|
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Uniword
|
|
4
|
+
module Lint
|
|
5
|
+
# Walks the document with each rule and aggregates findings.
|
|
6
|
+
class Engine
|
|
7
|
+
# @param document [Wordprocessingml::DocumentRoot]
|
|
8
|
+
# @param ruleset [Ruleset, Array<Rule>] rules to apply
|
|
9
|
+
def initialize(document:, ruleset:)
|
|
10
|
+
@document = document
|
|
11
|
+
@ruleset = wrap_ruleset(ruleset)
|
|
12
|
+
end
|
|
13
|
+
|
|
14
|
+
# @return [Result]
|
|
15
|
+
def run
|
|
16
|
+
result = Result.new
|
|
17
|
+
@ruleset.rules.each do |rule|
|
|
18
|
+
rule.check(@document) { |finding| result.add(finding) }
|
|
19
|
+
end
|
|
20
|
+
result
|
|
21
|
+
end
|
|
22
|
+
|
|
23
|
+
private
|
|
24
|
+
|
|
25
|
+
def wrap_ruleset(ruleset)
|
|
26
|
+
return ruleset if ruleset.is_a?(Ruleset)
|
|
27
|
+
|
|
28
|
+
Ruleset.new(Array(ruleset))
|
|
29
|
+
end
|
|
30
|
+
end
|
|
31
|
+
end
|
|
32
|
+
end
|
|
@@ -0,0 +1,59 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Uniword
|
|
4
|
+
module Lint
|
|
5
|
+
# Aggregated lint result with severity counts.
|
|
6
|
+
class Result
|
|
7
|
+
SEVERITY_ORDER = %i[error warning info].freeze
|
|
8
|
+
|
|
9
|
+
attr_reader :findings
|
|
10
|
+
|
|
11
|
+
def initialize
|
|
12
|
+
@findings = []
|
|
13
|
+
end
|
|
14
|
+
|
|
15
|
+
# @param finding [Hash]
|
|
16
|
+
# @return [void]
|
|
17
|
+
def add(finding)
|
|
18
|
+
@findings << finding
|
|
19
|
+
end
|
|
20
|
+
|
|
21
|
+
# True when at least one :error severity finding exists.
|
|
22
|
+
#
|
|
23
|
+
# @return [Boolean]
|
|
24
|
+
def errors?
|
|
25
|
+
@findings.any? { |f| f[:severity] == :error }
|
|
26
|
+
end
|
|
27
|
+
|
|
28
|
+
# Count by severity.
|
|
29
|
+
#
|
|
30
|
+
# @return [Hash{Symbol => Integer}]
|
|
31
|
+
def by_severity
|
|
32
|
+
@findings.group_by { |f| f[:severity] }
|
|
33
|
+
.transform_values(&:count)
|
|
34
|
+
end
|
|
35
|
+
|
|
36
|
+
# Count by rule name.
|
|
37
|
+
#
|
|
38
|
+
# @return [Hash{Symbol => Integer}]
|
|
39
|
+
def by_rule
|
|
40
|
+
@findings.group_by { |f| f[:rule] }
|
|
41
|
+
.transform_values(&:count)
|
|
42
|
+
end
|
|
43
|
+
|
|
44
|
+
# Total finding count.
|
|
45
|
+
#
|
|
46
|
+
# @return [Integer]
|
|
47
|
+
def count
|
|
48
|
+
@findings.length
|
|
49
|
+
end
|
|
50
|
+
|
|
51
|
+
# True when zero findings.
|
|
52
|
+
#
|
|
53
|
+
# @return [Boolean]
|
|
54
|
+
def empty?
|
|
55
|
+
@findings.empty?
|
|
56
|
+
end
|
|
57
|
+
end
|
|
58
|
+
end
|
|
59
|
+
end
|