uniword 1.2.5 → 1.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +313 -0
- data/CONTRIBUTING.md +1 -1
- data/config/ooxml/schemas/shared_types.yml +5 -4
- data/data/schemas/ecma/dc.xsd +118 -0
- data/data/schemas/ecma/dcmitype.xsd +50 -0
- data/data/schemas/ecma/dcterms.xsd +331 -0
- data/data/schemas/ecma/opc-coreProperties.xsd +2 -2
- data/data/schemas/ecma/xml.xsd +117 -0
- data/lib/uniword/builder/bibliography_builder.rb +1 -1
- data/lib/uniword/builder/chart_builder.rb +22 -14
- data/lib/uniword/builder/comment_anchorer.rb +221 -0
- data/lib/uniword/builder/document_builder.rb +97 -6
- data/lib/uniword/builder/image_builder.rb +11 -10
- data/lib/uniword/builder/run_utils.rb +2 -0
- data/lib/uniword/builder.rb +1 -0
- data/lib/uniword/cli/fonts_cli.rb +52 -0
- data/lib/uniword/cli/main.rb +110 -11
- data/lib/uniword/cli/page_cli.rb +85 -0
- data/lib/uniword/cli/styles_cli.rb +204 -0
- data/lib/uniword/cli/theme_cli.rb +88 -0
- data/lib/uniword/cli/toc_cli.rb +22 -2
- data/lib/uniword/comment.rb +20 -15
- data/lib/uniword/comment_range.rb +3 -3
- data/lib/uniword/comments_part.rb +51 -19
- data/lib/uniword/configuration/configuration_loader.rb +5 -3
- data/lib/uniword/configuration.rb +93 -1
- data/lib/uniword/content_types.rb +19 -39
- data/lib/uniword/document_factory.rb +12 -18
- data/lib/uniword/document_writer.rb +16 -5
- data/lib/uniword/docx/chart_part.rb +34 -0
- data/lib/uniword/docx/custom_xml_item.rb +91 -0
- data/lib/uniword/docx/header_footer_part.rb +140 -0
- data/lib/uniword/docx/header_footer_part_collection.rb +135 -0
- data/lib/uniword/docx/header_footer_view.rb +225 -0
- data/lib/uniword/docx/id_allocator.rb +130 -36
- data/lib/uniword/docx/image_part.rb +83 -0
- data/lib/uniword/docx/package.rb +141 -333
- data/lib/uniword/docx/package_defaults.rb +83 -99
- data/lib/uniword/docx/package_integrity_checker.rb +312 -0
- data/lib/uniword/docx/package_serialization.rb +144 -233
- data/lib/uniword/docx/part.rb +125 -0
- data/lib/uniword/docx/part_collection.rb +131 -0
- data/lib/uniword/docx/part_loader/chart_loader.rb +50 -0
- data/lib/uniword/docx/part_loader/custom_xml_loader.rb +53 -0
- data/lib/uniword/docx/part_loader/embedding_loader.rb +26 -0
- data/lib/uniword/docx/part_loader/header_footer_loader.rb +101 -0
- data/lib/uniword/docx/part_loader/image_loader.rb +105 -0
- data/lib/uniword/docx/part_loader/load_context.rb +85 -0
- data/lib/uniword/docx/part_loader/raw_part_loader.rb +183 -0
- data/lib/uniword/docx/part_loader/theme_media_loader.rb +39 -0
- data/lib/uniword/docx/part_loader/xml_model_loader.rb +56 -0
- data/lib/uniword/docx/part_loader.rb +92 -0
- data/lib/uniword/docx/raw_part.rb +75 -0
- data/lib/uniword/docx/reconciler/body.rb +59 -68
- data/lib/uniword/docx/reconciler/fix.rb +46 -0
- data/lib/uniword/docx/reconciler/fix_codes.rb +37 -30
- data/lib/uniword/docx/reconciler/helpers.rb +18 -31
- data/lib/uniword/docx/reconciler/notes.rb +17 -8
- data/lib/uniword/docx/reconciler/package_structure.rb +168 -221
- data/lib/uniword/docx/reconciler/parts.rb +36 -13
- data/lib/uniword/docx/reconciler/referential_integrity.rb +222 -123
- data/lib/uniword/docx/reconciler/tables.rb +78 -16
- data/lib/uniword/docx/reconciler/theme.rb +7 -4
- data/lib/uniword/docx/reconciler.rb +36 -14
- data/lib/uniword/docx.rb +13 -0
- data/lib/uniword/drawingml/blip.rb +0 -1
- data/lib/uniword/drawingml/color_scheme.rb +5 -6
- data/lib/uniword/drawingml/font_scheme.rb +4 -8
- data/lib/uniword/drawingml/theme.rb +12 -6
- data/lib/uniword/errors.rb +8 -1
- data/lib/uniword/footer.rb +4 -0
- data/lib/uniword/header.rb +4 -0
- data/lib/uniword/hyperlink.rb +1 -1
- data/lib/uniword/image.rb +6 -6
- data/lib/uniword/images/image_manager.rb +13 -13
- data/lib/uniword/lazy_loader.rb +1 -2
- data/lib/uniword/model_attribute_access.rb +20 -4
- data/lib/uniword/ooxml/dotx_package.rb +42 -2
- data/lib/uniword/ooxml/element_order.rb +94 -0
- data/lib/uniword/ooxml/part_definition.rb +329 -0
- data/lib/uniword/ooxml/part_registry.rb +504 -0
- data/lib/uniword/ooxml/relationships/image_relationship.rb +6 -3
- data/lib/uniword/ooxml/relationships/package_relationships.rb +20 -22
- data/lib/uniword/ooxml/schema/element_serializer.rb +8 -7
- data/lib/uniword/ooxml/types/hex_color_value.rb +38 -0
- data/lib/uniword/ooxml/types/ooxml_boolean.rb +27 -14
- data/lib/uniword/ooxml/types/ooxml_boolean_optional.rb +21 -16
- data/lib/uniword/ooxml/types/theme_color_value.rb +22 -0
- data/lib/uniword/ooxml/types/unsigned_decimal_number.rb +30 -0
- data/lib/uniword/ooxml/types.rb +10 -0
- data/lib/uniword/ooxml.rb +3 -0
- data/lib/uniword/properties/alignment.rb +8 -3
- data/lib/uniword/properties/border.rb +52 -6
- data/lib/uniword/properties/cell_vertical_align.rb +4 -1
- data/lib/uniword/properties/color_value.rb +8 -3
- data/lib/uniword/properties/highlight.rb +9 -5
- data/lib/uniword/properties/shading.rb +4 -3
- data/lib/uniword/properties/tab_stop.rb +6 -1
- data/lib/uniword/properties/table_justification.rb +4 -1
- data/lib/uniword/properties/underline.rb +4 -2
- data/lib/uniword/properties/vertical_align.rb +6 -3
- data/lib/uniword/review/review_manager.rb +23 -9
- data/lib/uniword/revision.rb +9 -9
- data/lib/uniword/schema/model_generator.rb +31 -15
- data/lib/uniword/shared_types/hex_color.rb +4 -1
- data/lib/uniword/shared_types/pixel_measure.rb +4 -1
- data/lib/uniword/shared_types/point_measure.rb +4 -1
- data/lib/uniword/shared_types/text_alignment.rb +6 -1
- data/lib/uniword/shared_types/twips_measure.rb +4 -1
- data/lib/uniword/spreadsheetml/phonetic_pr.rb +1 -1
- data/lib/uniword/spreadsheetml/shared_string_table.rb +2 -2
- data/lib/uniword/spreadsheetml/table_formula.rb +1 -1
- data/lib/uniword/template/variable_resolver.rb +40 -7
- data/lib/uniword/themes/theme_transformation.rb +55 -41
- data/lib/uniword/toc/toc_generator.rb +6 -5
- data/lib/uniword/transformation/mhtml_element_renderer.rb +1 -5
- data/lib/uniword/transformation/mhtml_metadata_builder.rb +6 -6
- data/lib/uniword/validation/engine.rb +35 -0
- data/lib/uniword/validation/link_checker.rb +1 -1
- data/lib/uniword/validation/opc_validator.rb +16 -2
- data/lib/uniword/validation/report/layer_result.rb +0 -1
- data/lib/uniword/validation/report/terminal_formatter.rb +0 -1
- data/lib/uniword/validation/report/verification_report.rb +0 -2
- data/lib/uniword/validation/report.rb +20 -0
- data/lib/uniword/validation/rules/base.rb +10 -2
- data/lib/uniword/validation/rules/bookmark_pairing_rule.rb +38 -0
- data/lib/uniword/validation/rules/bookmark_uniqueness_rule.rb +48 -0
- data/lib/uniword/validation/rules/bookmarks_rule.rb +0 -2
- data/lib/uniword/validation/rules/content_types_coverage_rule.rb +0 -2
- data/lib/uniword/validation/rules/core_properties_namespace_rule.rb +0 -2
- data/lib/uniword/validation/rules/document_body_rule.rb +25 -0
- data/lib/uniword/validation/rules/document_context.rb +7 -0
- data/lib/uniword/validation/rules/empty_paragraphs_rule.rb +27 -0
- data/lib/uniword/validation/rules/font_table_signature_rule.rb +0 -2
- data/lib/uniword/validation/rules/fonts_rule.rb +0 -2
- data/lib/uniword/validation/rules/footnotes_rule.rb +0 -2
- data/lib/uniword/validation/rules/headers_footers_rule.rb +0 -2
- data/lib/uniword/validation/rules/images_rule.rb +0 -2
- data/lib/uniword/validation/rules/mc_ignorable_namespace_rule.rb +0 -2
- data/lib/uniword/validation/rules/model_context.rb +32 -0
- data/lib/uniword/validation/rules/model_rule.rb +34 -0
- data/lib/uniword/validation/rules/numbering_preservation_rule.rb +0 -2
- data/lib/uniword/validation/rules/numbering_rule.rb +0 -2
- data/lib/uniword/validation/rules/relationship_integrity_rule.rb +0 -1
- data/lib/uniword/validation/rules/rsid_rule.rb +0 -2
- data/lib/uniword/validation/rules/section_properties_rule.rb +0 -2
- data/lib/uniword/validation/rules/settings_rule.rb +0 -2
- data/lib/uniword/validation/rules/settings_values_rule.rb +0 -2
- data/lib/uniword/validation/rules/style_references_rule.rb +0 -2
- data/lib/uniword/validation/rules/table_grid_rule.rb +27 -0
- data/lib/uniword/validation/rules/table_properties_rule.rb +27 -0
- data/lib/uniword/validation/rules/tables_rule.rb +0 -2
- data/lib/uniword/validation/rules/theme_completeness_rule.rb +0 -2
- data/lib/uniword/validation/rules/theme_rule.rb +0 -2
- data/lib/uniword/validation/rules.rb +61 -23
- data/lib/uniword/validation/schema_registry.rb +11 -0
- data/lib/uniword/validation/validators/document_semantics_validator.rb +1 -9
- data/lib/uniword/validation/validators/xml_schema_validator.rb +46 -2
- data/lib/uniword/validation/validators.rb +0 -9
- data/lib/uniword/validation/verify_orchestrator.rb +0 -7
- data/lib/uniword/validation.rb +7 -2
- data/lib/uniword/version.rb +1 -1
- data/lib/uniword/vml/imagedata.rb +0 -1
- data/lib/uniword/wordprocessingml/attached_template.rb +0 -1
- data/lib/uniword/wordprocessingml/deleted_text.rb +1 -1
- data/lib/uniword/wordprocessingml/document_root.rb +121 -122
- data/lib/uniword/wordprocessingml/document_styling.rb +246 -0
- data/lib/uniword/wordprocessingml/endnotes.rb +4 -1
- data/lib/uniword/wordprocessingml/font_replacer.rb +147 -0
- data/lib/uniword/wordprocessingml/footnotes.rb +4 -1
- data/lib/uniword/wordprocessingml/hdr_shape_defaults.rb +0 -1
- data/lib/uniword/wordprocessingml/hyperlink.rb +0 -1
- data/lib/uniword/wordprocessingml/level.rb +6 -20
- data/lib/uniword/wordprocessingml/math_pr.rb +0 -11
- data/lib/uniword/wordprocessingml/numbering_definition.rb +16 -8
- data/lib/uniword/wordprocessingml/numbering_elements.rb +0 -15
- data/lib/uniword/wordprocessingml/page_setup.rb +181 -0
- data/lib/uniword/wordprocessingml/paragraph_properties.rb +33 -28
- data/lib/uniword/wordprocessingml/rsids.rb +0 -2
- data/lib/uniword/wordprocessingml/run.rb +3 -0
- data/lib/uniword/wordprocessingml/section_properties.rb +2 -2
- data/lib/uniword/wordprocessingml/settings.rb +8 -34
- data/lib/uniword/wordprocessingml/shape_defaults.rb +0 -1
- data/lib/uniword/wordprocessingml/style.rb +2 -1
- data/lib/uniword/wordprocessingml/style_cleanup.rb +198 -0
- data/lib/uniword/wordprocessingml/table_cell_properties.rb +6 -6
- data/lib/uniword/wordprocessingml/update_fields.rb +27 -0
- data/lib/uniword/wordprocessingml/w14_attributes.rb +30 -2
- data/lib/uniword/wordprocessingml.rb +5 -2
- data/lib/uniword.rb +27 -2
- metadata +52 -19
- data/config/validation_rules.yml +0 -60
- data/config/warning_rules.yml +0 -57
- data/lib/uniword/validation/document_validator.rb +0 -272
- data/lib/uniword/validation/structural_validator.rb +0 -116
- data/lib/uniword/validation/validators/content_type_validator.rb +0 -165
- data/lib/uniword/validation/validators/file_structure_validator.rb +0 -104
- data/lib/uniword/validation/validators/ooxml_part_validator.rb +0 -128
- data/lib/uniword/validation/validators/relationship_validator.rb +0 -147
- data/lib/uniword/validation/validators/zip_integrity_validator.rb +0 -110
- data/lib/uniword/validators/element_validator.rb +0 -93
- data/lib/uniword/validators/paragraph_validator.rb +0 -116
- data/lib/uniword/validators/table_validator.rb +0 -134
- data/lib/uniword/validators.rb +0 -9
- data/lib/uniword/warnings/warning.rb +0 -130
- data/lib/uniword/warnings/warning_collector.rb +0 -234
- data/lib/uniword/warnings/warning_report.rb +0 -159
- data/lib/uniword/warnings.rb +0 -9
data/lib/uniword/docx/package.rb
CHANGED
|
@@ -44,7 +44,15 @@ module Uniword
|
|
|
44
44
|
attribute :custom_properties, Ooxml::CustomProperties
|
|
45
45
|
|
|
46
46
|
# Custom XML data items (customXml/item*.xml)
|
|
47
|
-
|
|
47
|
+
#
|
|
48
|
+
# @return [Array<CustomXmlItem>, nil]
|
|
49
|
+
attr_reader :custom_xml_items
|
|
50
|
+
|
|
51
|
+
# Assign custom XML items; accepts CustomXmlItem objects and
|
|
52
|
+
# legacy hashes ({ index:, xml_content:, props_xml:, rels_xml: }).
|
|
53
|
+
def custom_xml_items=(items)
|
|
54
|
+
@custom_xml_items = items && items.map { |i| CustomXmlItem.wrap(i) }
|
|
55
|
+
end
|
|
48
56
|
|
|
49
57
|
# === Document Parts (word/) ===
|
|
50
58
|
# Main document content (word/document.xml)
|
|
@@ -82,9 +90,40 @@ module Uniword
|
|
|
82
90
|
# Endnotes (word/endnotes.xml)
|
|
83
91
|
attribute :endnotes, Uniword::Wordprocessingml::Endnotes
|
|
84
92
|
|
|
93
|
+
# Comments (word/comments.xml)
|
|
94
|
+
attribute :comments, Uniword::CommentsPart
|
|
95
|
+
|
|
85
96
|
# Non-serialized attributes (DOCX packaging helpers)
|
|
86
|
-
attr_accessor :
|
|
87
|
-
attr_accessor :settings_rels, :
|
|
97
|
+
attr_accessor :profile
|
|
98
|
+
attr_accessor :settings_rels, :footnotes_rels, :endnotes_rels
|
|
99
|
+
|
|
100
|
+
# OLE/embedded object binaries (word/embeddings/*), keyed by target.
|
|
101
|
+
#
|
|
102
|
+
# @return [PartCollection] target => Part
|
|
103
|
+
def embeddings
|
|
104
|
+
@embeddings ||= PartCollection.new(:target, Part)
|
|
105
|
+
end
|
|
106
|
+
|
|
107
|
+
# Bulk-assign embeddings (Hash of target => Part/binary; nil clears).
|
|
108
|
+
def embeddings=(value)
|
|
109
|
+
embeddings.replace_all(value)
|
|
110
|
+
end
|
|
111
|
+
|
|
112
|
+
# Raw passthrough parts no registry definition models, keyed by
|
|
113
|
+
# package path ("docProps/meta.xml", "word/glossary/document.xml",
|
|
114
|
+
# ...). Carried byte-for-byte from load to save — see
|
|
115
|
+
# PartLoader::RawPartLoader for claiming and
|
|
116
|
+
# PackageSerialization#serialize_raw_parts for emission.
|
|
117
|
+
#
|
|
118
|
+
# @return [PartCollection] package path => RawPart
|
|
119
|
+
def raw_parts
|
|
120
|
+
@raw_parts ||= PartCollection.new(:path, RawPart)
|
|
121
|
+
end
|
|
122
|
+
|
|
123
|
+
# Bulk-assign raw parts (Hash of path => RawPart/hash; nil clears).
|
|
124
|
+
def raw_parts=(value)
|
|
125
|
+
raw_parts.replace_all(value)
|
|
126
|
+
end
|
|
88
127
|
|
|
89
128
|
# Central ID allocator — owns all rId, footnote, bookmark, etc. assignment.
|
|
90
129
|
# Seeded from template rels on load; used by builders during construction.
|
|
@@ -98,6 +137,15 @@ module Uniword
|
|
|
98
137
|
# Paths that have been modified and should not use raw XML passthrough.
|
|
99
138
|
attr_accessor :modified_part_paths
|
|
100
139
|
|
|
140
|
+
# Audit trail of repairs applied by the Reconciler during the most
|
|
141
|
+
# recent save (to_zip_content). Empty before the first save and when
|
|
142
|
+
# the package needed no repairs.
|
|
143
|
+
#
|
|
144
|
+
# @return [Array<Reconciler::Fix>] Fixes from the most recent save
|
|
145
|
+
def applied_fixes
|
|
146
|
+
@applied_fixes ||= []
|
|
147
|
+
end
|
|
148
|
+
|
|
101
149
|
# Load DOCX package from file
|
|
102
150
|
#
|
|
103
151
|
# @param path [String] Path to .docx file
|
|
@@ -112,305 +160,28 @@ module Uniword
|
|
|
112
160
|
|
|
113
161
|
# Create package from extracted ZIP content
|
|
114
162
|
#
|
|
163
|
+
# Parts load by iterating Ooxml::PartRegistry.loadable and
|
|
164
|
+
# dispatching each definition to its Docx::PartLoader strategy —
|
|
165
|
+
# see PartLoader for the load order and strategy registry.
|
|
166
|
+
#
|
|
115
167
|
# @param zip_content [Hash] Extracted ZIP files
|
|
116
168
|
# @param zip_path [String, nil] Original ZIP path for binary re-extraction
|
|
117
169
|
# @return [Package] Package object
|
|
118
170
|
def self.from_zip_content(zip_content, zip_path = nil)
|
|
119
171
|
package = new
|
|
120
|
-
|
|
121
|
-
# Parse Content Types
|
|
122
|
-
if zip_content["[Content_Types].xml"]
|
|
123
|
-
package.content_types = Uniword::ContentTypes::Types.from_xml(
|
|
124
|
-
zip_content["[Content_Types].xml"]
|
|
125
|
-
)
|
|
126
|
-
end
|
|
127
|
-
|
|
128
|
-
# Parse Package Relationships
|
|
129
|
-
if zip_content["_rels/.rels"]
|
|
130
|
-
package.package_rels = Ooxml::Relationships::PackageRelationships.from_xml(
|
|
131
|
-
zip_content["_rels/.rels"]
|
|
132
|
-
)
|
|
133
|
-
end
|
|
134
|
-
|
|
135
|
-
# Find the main document path from officeDocument relationship
|
|
136
|
-
main_doc_path = find_main_document_path(package.package_rels)
|
|
137
|
-
main_doc_rels_path = find_document_rels_path(main_doc_path)
|
|
138
|
-
|
|
139
|
-
# Parse Document Properties
|
|
140
|
-
if zip_content["docProps/core.xml"]
|
|
141
|
-
package.core_properties = Ooxml::CoreProperties.from_xml(
|
|
142
|
-
zip_content["docProps/core.xml"]
|
|
143
|
-
)
|
|
144
|
-
end
|
|
145
|
-
|
|
146
|
-
if zip_content["docProps/app.xml"]
|
|
147
|
-
package.app_properties = Ooxml::AppProperties.from_xml(
|
|
148
|
-
zip_content["docProps/app.xml"]
|
|
149
|
-
)
|
|
150
|
-
end
|
|
151
|
-
|
|
152
|
-
# Parse Custom Properties
|
|
153
|
-
if zip_content["docProps/custom.xml"]
|
|
154
|
-
package.custom_properties = Ooxml::CustomProperties.from_xml(
|
|
155
|
-
zip_content["docProps/custom.xml"]
|
|
156
|
-
)
|
|
157
|
-
end
|
|
158
|
-
|
|
159
|
-
# Parse Custom XML Data items (customXml/item*.xml)
|
|
160
|
-
custom_xml_files = zip_content.keys.grep(%r{^customXml/item(\d+)\.xml$})
|
|
161
|
-
if custom_xml_files.any?
|
|
162
|
-
package.custom_xml_items = []
|
|
163
|
-
custom_xml_files.sort_by { |f| f[/item(\d+)/, 1].to_i }.each do |item_path|
|
|
164
|
-
index = item_path[/item(\d+)/, 1].to_i
|
|
165
|
-
item = {
|
|
166
|
-
index: index,
|
|
167
|
-
xml_content: zip_content[item_path]
|
|
168
|
-
}
|
|
169
|
-
|
|
170
|
-
props_path = "customXml/itemProps#{index}.xml"
|
|
171
|
-
item[:props_xml] = zip_content[props_path] if zip_content[props_path]
|
|
172
|
-
|
|
173
|
-
rels_path = "customXml/_rels/item#{index}.xml.rels"
|
|
174
|
-
item[:rels_xml] = zip_content[rels_path] if zip_content[rels_path]
|
|
175
|
-
|
|
176
|
-
package.custom_xml_items << item
|
|
177
|
-
end
|
|
178
|
-
end
|
|
179
|
-
|
|
180
|
-
# Parse Document Parts - use dynamic path from package relationships
|
|
181
|
-
if main_doc_path && zip_content[main_doc_path]
|
|
182
|
-
package.document = Uniword::Wordprocessingml::DocumentRoot.from_xml(
|
|
183
|
-
zip_content[main_doc_path]
|
|
184
|
-
)
|
|
185
|
-
elsif zip_content["word/document.xml"]
|
|
186
|
-
package.document = Uniword::Wordprocessingml::DocumentRoot.from_xml(
|
|
187
|
-
zip_content["word/document.xml"]
|
|
188
|
-
)
|
|
189
|
-
end
|
|
190
|
-
|
|
191
|
-
if zip_content["word/styles.xml"]
|
|
192
|
-
package.styles = Uniword::Wordprocessingml::StylesConfiguration.from_xml(
|
|
193
|
-
zip_content["word/styles.xml"]
|
|
194
|
-
)
|
|
195
|
-
end
|
|
196
|
-
|
|
197
|
-
if zip_content["word/numbering.xml"]
|
|
198
|
-
package.numbering = Uniword::Wordprocessingml::NumberingConfiguration.from_xml(
|
|
199
|
-
zip_content["word/numbering.xml"]
|
|
200
|
-
)
|
|
201
|
-
end
|
|
202
|
-
|
|
203
|
-
if zip_content["word/settings.xml"]
|
|
204
|
-
package.settings = Uniword::Wordprocessingml::Settings.from_xml(
|
|
205
|
-
zip_content["word/settings.xml"]
|
|
206
|
-
)
|
|
207
|
-
end
|
|
208
|
-
|
|
209
|
-
if zip_content["word/_rels/settings.xml.rels"]
|
|
210
|
-
package.settings_rels =
|
|
211
|
-
Ooxml::Relationships::PackageRelationships.from_xml(
|
|
212
|
-
zip_content["word/_rels/settings.xml.rels"]
|
|
213
|
-
)
|
|
214
|
-
end
|
|
215
|
-
|
|
216
|
-
if zip_content["word/fontTable.xml"]
|
|
217
|
-
package.font_table = Uniword::Wordprocessingml::FontTable.from_xml(
|
|
218
|
-
zip_content["word/fontTable.xml"]
|
|
219
|
-
)
|
|
220
|
-
end
|
|
221
|
-
|
|
222
|
-
if zip_content["word/webSettings.xml"]
|
|
223
|
-
package.web_settings = Uniword::Wordprocessingml::WebSettings.from_xml(
|
|
224
|
-
zip_content["word/webSettings.xml"]
|
|
225
|
-
)
|
|
226
|
-
end
|
|
227
|
-
|
|
228
|
-
# Parse document relationships - use dynamic path based on main document
|
|
229
|
-
if main_doc_rels_path && zip_content[main_doc_rels_path]
|
|
230
|
-
package.document_rels = Ooxml::Relationships::PackageRelationships.from_xml(
|
|
231
|
-
zip_content[main_doc_rels_path]
|
|
232
|
-
)
|
|
233
|
-
elsif zip_content["word/_rels/document.xml.rels"]
|
|
234
|
-
package.document_rels = Ooxml::Relationships::PackageRelationships.from_xml(
|
|
235
|
-
zip_content["word/_rels/document.xml.rels"]
|
|
236
|
-
)
|
|
237
|
-
end
|
|
238
|
-
|
|
239
|
-
# Parse Theme
|
|
240
|
-
if zip_content["word/theme/theme1.xml"]
|
|
241
|
-
package.theme = Drawingml::Theme.from_xml(
|
|
242
|
-
zip_content["word/theme/theme1.xml"]
|
|
243
|
-
)
|
|
244
|
-
|
|
245
|
-
theme_media = extract_theme_media(zip_content)
|
|
246
|
-
package.theme.media_files = theme_media if theme_media.any?
|
|
247
|
-
end
|
|
248
|
-
|
|
249
|
-
if zip_content["word/theme/_rels/theme1.xml.rels"]
|
|
250
|
-
package.theme_rels = Ooxml::Relationships::PackageRelationships.from_xml(
|
|
251
|
-
zip_content["word/theme/_rels/theme1.xml.rels"]
|
|
252
|
-
)
|
|
253
|
-
end
|
|
254
|
-
|
|
255
|
-
# Parse Footnotes
|
|
256
|
-
if zip_content["word/footnotes.xml"]
|
|
257
|
-
package.footnotes = Uniword::Wordprocessingml::Footnotes.from_xml(
|
|
258
|
-
zip_content["word/footnotes.xml"]
|
|
259
|
-
)
|
|
260
|
-
end
|
|
261
|
-
|
|
262
|
-
# Parse Endnotes
|
|
263
|
-
if zip_content["word/endnotes.xml"]
|
|
264
|
-
package.endnotes = Uniword::Wordprocessingml::Endnotes.from_xml(
|
|
265
|
-
zip_content["word/endnotes.xml"]
|
|
266
|
-
)
|
|
267
|
-
end
|
|
268
|
-
|
|
269
|
-
# Parse Header and Footer parts
|
|
270
|
-
extract_header_footer_parts(zip_content, package)
|
|
271
|
-
|
|
272
|
-
# Parse Chart parts
|
|
273
|
-
chart_files = zip_content.keys.grep(%r{^word/charts/chart\d+\.xml$})
|
|
274
|
-
if chart_files.any? && package.document_rels
|
|
275
|
-
package.document.chart_parts ||= {}
|
|
276
|
-
chart_files.each do |chart_path|
|
|
277
|
-
chart_target = chart_path.sub("word/", "")
|
|
278
|
-
rel = package.document_rels.relationships.find do |r|
|
|
279
|
-
r.target == chart_target &&
|
|
280
|
-
r.type.to_s.include?("officeDocument/2006/relationships/chart")
|
|
281
|
-
end
|
|
282
|
-
next unless rel
|
|
283
|
-
|
|
284
|
-
package.document.chart_parts[rel.id] = {
|
|
285
|
-
xml: zip_content[chart_path],
|
|
286
|
-
target: chart_target
|
|
287
|
-
}
|
|
288
|
-
end
|
|
289
|
-
end
|
|
290
|
-
|
|
291
|
-
# Extract image parts from word/media/ directory
|
|
292
|
-
extract_image_parts(zip_content, package, zip_path)
|
|
293
|
-
|
|
294
|
-
# Extract OLE/embedded object binaries from word/embeddings/
|
|
295
|
-
embedding_files = zip_content.keys.grep(%r{^word/embeddings/.+$})
|
|
296
|
-
if embedding_files.any?
|
|
297
|
-
package.embeddings = {}
|
|
298
|
-
embedding_files.each do |emb_path|
|
|
299
|
-
target = emb_path.sub("word/", "")
|
|
300
|
-
package.embeddings[target] = zip_content[emb_path]
|
|
301
|
-
end
|
|
302
|
-
end
|
|
303
|
-
|
|
172
|
+
PartLoader.load(zip_content, package, zip_path: zip_path)
|
|
304
173
|
package
|
|
305
174
|
end
|
|
306
175
|
|
|
307
|
-
def self.extract_header_footer_parts(zip_content, package)
|
|
308
|
-
return unless package.document && package.document_rels
|
|
309
|
-
|
|
310
|
-
header_files = zip_content.keys.grep(%r{^word/header\d+\.xml$})
|
|
311
|
-
footer_files = zip_content.keys.grep(%r{^word/footer\d+\.xml$})
|
|
312
|
-
|
|
313
|
-
return if header_files.empty? && footer_files.empty?
|
|
314
|
-
|
|
315
|
-
package.document.header_footer_parts ||= []
|
|
316
|
-
|
|
317
|
-
header_files.sort.each do |path|
|
|
318
|
-
target = path.sub("word/", "")
|
|
319
|
-
rel = package.document_rels.relationships.find do |r|
|
|
320
|
-
r.target == target &&
|
|
321
|
-
r.type.to_s.include?("officeDocument/2006/relationships/header")
|
|
322
|
-
end
|
|
323
|
-
next unless rel
|
|
324
|
-
|
|
325
|
-
package.document.header_footer_parts << {
|
|
326
|
-
r_id: rel.id,
|
|
327
|
-
target: target,
|
|
328
|
-
rel_type: rel.type,
|
|
329
|
-
content_type: "application/vnd.openxmlformats-officedocument.wordprocessingml.header+xml",
|
|
330
|
-
content: Uniword::Wordprocessingml::Header.from_xml(zip_content[path]),
|
|
331
|
-
}
|
|
332
|
-
end
|
|
333
|
-
|
|
334
|
-
footer_files.sort.each do |path|
|
|
335
|
-
target = path.sub("word/", "")
|
|
336
|
-
rel = package.document_rels.relationships.find do |r|
|
|
337
|
-
r.target == target &&
|
|
338
|
-
r.type.to_s.include?("officeDocument/2006/relationships/footer")
|
|
339
|
-
end
|
|
340
|
-
next unless rel
|
|
341
|
-
|
|
342
|
-
package.document.header_footer_parts << {
|
|
343
|
-
r_id: rel.id,
|
|
344
|
-
target: target,
|
|
345
|
-
rel_type: rel.type,
|
|
346
|
-
content_type: "application/vnd.openxmlformats-officedocument.wordprocessingml.footer+xml",
|
|
347
|
-
content: Uniword::Wordprocessingml::Footer.from_xml(zip_content[path]),
|
|
348
|
-
}
|
|
349
|
-
end
|
|
350
|
-
end
|
|
351
|
-
|
|
352
|
-
# Extract image files from word/media/ directory in DOCX
|
|
353
|
-
#
|
|
354
|
-
# @param zip_content [Hash] Extracted ZIP content (may have corrupted binary)
|
|
355
|
-
# @param package [Package] Package to populate
|
|
356
|
-
# @param zip_path [String, nil] Original ZIP path for binary re-extraction
|
|
357
|
-
def self.extract_image_parts(zip_content, package, zip_path = nil)
|
|
358
|
-
return unless package.document
|
|
359
|
-
|
|
360
|
-
media_files = zip_content.keys.grep(%r{^word/media/.+$})
|
|
361
|
-
return if media_files.empty?
|
|
362
|
-
|
|
363
|
-
package.document.image_parts ||= {}
|
|
364
|
-
|
|
365
|
-
media_files.each do |media_path|
|
|
366
|
-
filename = File.basename(media_path)
|
|
367
|
-
ext = File.extname(filename).delete(".").downcase
|
|
368
|
-
content_type = case ext
|
|
369
|
-
when "jpg", "jpeg" then "image/jpeg"
|
|
370
|
-
when "png" then "image/png"
|
|
371
|
-
when "gif" then "image/gif"
|
|
372
|
-
when "bmp" then "image/bmp"
|
|
373
|
-
when "tiff", "tif" then "image/tiff"
|
|
374
|
-
when "svg" then "image/svg+xml"
|
|
375
|
-
else "image/#{ext}"
|
|
376
|
-
end
|
|
377
|
-
|
|
378
|
-
r_id = if package.allocator
|
|
379
|
-
package.allocator.alloc_rid(
|
|
380
|
-
target: "media/#{filename}",
|
|
381
|
-
type: IdAllocator::IMAGE_REL_TYPE,
|
|
382
|
-
)
|
|
383
|
-
else
|
|
384
|
-
"rId#{package.document.image_parts.size + 1}"
|
|
385
|
-
end
|
|
386
|
-
|
|
387
|
-
binary_data = if zip_path
|
|
388
|
-
read_binary_from_zip(zip_path, media_path)
|
|
389
|
-
else
|
|
390
|
-
zip_content[media_path]
|
|
391
|
-
end
|
|
392
|
-
|
|
393
|
-
package.document.image_parts[r_id] = {
|
|
394
|
-
data: binary_data,
|
|
395
|
-
target: "media/#{filename}",
|
|
396
|
-
content_type: content_type
|
|
397
|
-
}
|
|
398
|
-
end
|
|
399
|
-
end
|
|
400
|
-
|
|
401
|
-
# Read binary data directly from ZIP file without UTF-8 encoding
|
|
402
|
-
def self.read_binary_from_zip(zip_path, entry_path)
|
|
403
|
-
require "zip"
|
|
404
|
-
Zip::File.open(zip_path) do |zip_file|
|
|
405
|
-
entry = zip_file.find_entry(entry_path)
|
|
406
|
-
return nil unless entry
|
|
407
|
-
|
|
408
|
-
entry.get_input_stream.read
|
|
409
|
-
end
|
|
410
|
-
end
|
|
411
|
-
|
|
412
176
|
# Save document to file (class method for DocumentWriter compatibility)
|
|
413
|
-
|
|
177
|
+
#
|
|
178
|
+
# @param document [Wordprocessingml::DocumentRoot] Document to save
|
|
179
|
+
# @param path [String] Output file path
|
|
180
|
+
# @param profile [Profile, nil] Reconciliation profile
|
|
181
|
+
# @param validate [Boolean, nil] Run the package integrity gate before
|
|
182
|
+
# writing; nil falls back to Uniword.configuration.validate_on_save
|
|
183
|
+
# @return [void]
|
|
184
|
+
def self.to_file(document, path, profile: nil, validate: nil)
|
|
414
185
|
package = new
|
|
415
186
|
package.document = document
|
|
416
187
|
package.profile = profile || Profile.defaults
|
|
@@ -421,7 +192,7 @@ module Uniword
|
|
|
421
192
|
package.settings ||= Uniword::Wordprocessingml::Settings.new
|
|
422
193
|
package.font_table ||= Uniword::Wordprocessingml::FontTable.new
|
|
423
194
|
package.web_settings ||= Uniword::Wordprocessingml::WebSettings.new
|
|
424
|
-
package.to_file(path)
|
|
195
|
+
package.to_file(path, validate: validate)
|
|
425
196
|
end
|
|
426
197
|
|
|
427
198
|
# Populate the allocator from all existing template data.
|
|
@@ -430,33 +201,46 @@ module Uniword
|
|
|
430
201
|
@allocator = IdAllocator.populate_from_package(self)
|
|
431
202
|
end
|
|
432
203
|
|
|
433
|
-
#
|
|
434
|
-
|
|
435
|
-
|
|
436
|
-
|
|
437
|
-
|
|
438
|
-
|
|
439
|
-
|
|
440
|
-
|
|
441
|
-
|
|
442
|
-
|
|
443
|
-
|
|
444
|
-
|
|
445
|
-
)
|
|
446
|
-
end
|
|
447
|
-
|
|
448
|
-
media
|
|
204
|
+
# Guarantee the single rId authority before reconciliation: reuse
|
|
205
|
+
# the package's allocator (loaded/builder-seeded) or start one,
|
|
206
|
+
# then seed it from the package's current relationships. Seeding
|
|
207
|
+
# preserves loaded rIds verbatim; earlier builder allocations keep
|
|
208
|
+
# their ids — a seeded rel whose id collides is reallocated
|
|
209
|
+
# deterministically (IdAllocator#seed_from_rels).
|
|
210
|
+
def prepare_allocator
|
|
211
|
+
self.allocator ||= IdAllocator.new
|
|
212
|
+
allocator.seed_from_rels(package_rels&.relationships,
|
|
213
|
+
scope: :package)
|
|
214
|
+
allocator.seed_from_rels(document_rels&.relationships)
|
|
215
|
+
allocator
|
|
449
216
|
end
|
|
450
217
|
|
|
451
218
|
# Save package to file
|
|
452
|
-
|
|
453
|
-
|
|
219
|
+
#
|
|
220
|
+
# @param path [String] Output file path
|
|
221
|
+
# @param validate [Boolean, nil] Run the package integrity gate before
|
|
222
|
+
# writing; nil falls back to Uniword.configuration.validate_on_save
|
|
223
|
+
# @return [void]
|
|
224
|
+
# @raise [Uniword::ValidationError] when the gate is enabled and the
|
|
225
|
+
# generated package content is invalid
|
|
226
|
+
def to_file(path, validate: nil)
|
|
227
|
+
zip_content = to_zip_content(validate: validate)
|
|
454
228
|
packager = Infrastructure::ZipPackager.new
|
|
455
229
|
packager.package(zip_content, path)
|
|
456
230
|
end
|
|
457
231
|
|
|
458
232
|
# Generate ZIP content hash
|
|
459
|
-
|
|
233
|
+
#
|
|
234
|
+
# Runs the Reconciler (the only mutating pass), records its repair
|
|
235
|
+
# report on #applied_fixes, and — unless validation is disabled —
|
|
236
|
+
# refuses invalid output via the PackageIntegrityChecker gate.
|
|
237
|
+
#
|
|
238
|
+
# @param validate [Boolean, nil] Run the package integrity gate;
|
|
239
|
+
# nil falls back to Uniword.configuration.validate_on_save
|
|
240
|
+
# @return [Hash] File paths => content
|
|
241
|
+
# @raise [Uniword::ValidationError] when the gate is enabled and the
|
|
242
|
+
# generated package content is invalid
|
|
243
|
+
def to_zip_content(validate: nil)
|
|
460
244
|
content = {}
|
|
461
245
|
|
|
462
246
|
self.content_types ||= self.class.minimal_content_types
|
|
@@ -467,15 +251,28 @@ module Uniword
|
|
|
467
251
|
self.font_table ||= Uniword::Wordprocessingml::FontTable.new
|
|
468
252
|
self.web_settings ||= Uniword::Wordprocessingml::WebSettings.new
|
|
469
253
|
|
|
470
|
-
|
|
471
|
-
|
|
472
|
-
|
|
254
|
+
# An allocator carried into the save (builder-managed document
|
|
255
|
+
# or loaded package) selects light-touch repairs; documents
|
|
256
|
+
# without one get the full normalization repertoire. rIds flow
|
|
257
|
+
# through the allocator either way.
|
|
258
|
+
builder_managed = !allocator.nil?
|
|
259
|
+
prepare_allocator
|
|
260
|
+
|
|
261
|
+
reconciler = Reconciler.new(self,
|
|
262
|
+
profile: profile || Profile.defaults,
|
|
263
|
+
allocator: allocator,
|
|
264
|
+
builder_managed: builder_managed)
|
|
265
|
+
reconciler.reconcile
|
|
266
|
+
@applied_fixes = reconciler.applied_fixes
|
|
267
|
+
log_applied_fixes
|
|
473
268
|
|
|
474
269
|
inject_part_relationships(content, content_types, package_rels, document_rels)
|
|
475
270
|
serialize_package_parts(content, content_types, package_rels, document_rels)
|
|
476
271
|
|
|
477
272
|
# OOXML requires [Content_Types].xml as the first ZIP entry.
|
|
478
273
|
reorder_content_hash(content)
|
|
274
|
+
|
|
275
|
+
enforce_package_integrity(content, validate)
|
|
479
276
|
content
|
|
480
277
|
end
|
|
481
278
|
|
|
@@ -519,27 +316,38 @@ module Uniword
|
|
|
519
316
|
document&.styles_configuration
|
|
520
317
|
end
|
|
521
318
|
|
|
522
|
-
|
|
523
|
-
def self.find_main_document_path(package_rels)
|
|
524
|
-
return nil unless package_rels&.relationships
|
|
319
|
+
private
|
|
525
320
|
|
|
526
|
-
|
|
527
|
-
|
|
528
|
-
|
|
529
|
-
|
|
321
|
+
# Log each applied fix via Uniword.logger when policy allows.
|
|
322
|
+
#
|
|
323
|
+
# @return [void]
|
|
324
|
+
def log_applied_fixes
|
|
325
|
+
return if @applied_fixes.empty?
|
|
326
|
+
return unless Uniword.configuration.log_save_fixes
|
|
530
327
|
|
|
531
|
-
|
|
532
|
-
|
|
533
|
-
|
|
328
|
+
@applied_fixes.each do |fix|
|
|
329
|
+
Uniword.logger&.info { "Reconciler fix #{fix}" }
|
|
330
|
+
end
|
|
534
331
|
end
|
|
535
332
|
|
|
536
|
-
#
|
|
537
|
-
|
|
538
|
-
|
|
539
|
-
|
|
540
|
-
|
|
541
|
-
|
|
542
|
-
|
|
333
|
+
# Write-time integrity gate: refuse invalid package content.
|
|
334
|
+
#
|
|
335
|
+
# @param content [Hash] File paths => content
|
|
336
|
+
# @param validate [Boolean, nil] explicit override; nil reads policy
|
|
337
|
+
# @return [void]
|
|
338
|
+
# @raise [Uniword::ValidationError] listing all integrity issues
|
|
339
|
+
def enforce_package_integrity(content, validate)
|
|
340
|
+
validate = Uniword.configuration.validate_on_save if validate.nil?
|
|
341
|
+
return unless validate
|
|
342
|
+
|
|
343
|
+
issues = PackageIntegrityChecker.new.check(content)
|
|
344
|
+
return if issues.empty?
|
|
345
|
+
|
|
346
|
+
raise Uniword::ValidationError.new(
|
|
347
|
+
self,
|
|
348
|
+
issues.map { |issue| "#{issue.code} (#{issue.part}): #{issue.message}" },
|
|
349
|
+
issues: issues,
|
|
350
|
+
)
|
|
543
351
|
end
|
|
544
352
|
end
|
|
545
353
|
end
|