neu-mods 0.14.0 → 0.14.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/.version +1 -1
- data/CHANGELOG.md +32 -0
- data/README.md +62 -223
- data/lib/neu/mods/builders.rb +63 -0
- data/lib/neu/mods/canonicalize.rb +6 -151
- data/lib/neu/mods/document.rb +5 -0
- data/lib/neu/mods/namespaces.rb +13 -0
- data/lib/neu/mods/projection/access.rb +74 -0
- data/lib/neu/mods/projection/dates.rb +170 -0
- data/lib/neu/mods/projection/identifiers.rb +80 -0
- data/lib/neu/mods/projection/languages.rb +59 -0
- data/lib/neu/mods/projection/name_display.rb +110 -0
- data/lib/neu/mods/projection/names.rb +71 -0
- data/lib/neu/mods/projection/origin_info.rb +86 -0
- data/lib/neu/mods/projection/physical_description.rb +53 -0
- data/lib/neu/mods/projection/related_items.rb +100 -0
- data/lib/neu/mods/projection/subjects.rb +148 -0
- data/lib/neu/mods/projection/support.rb +139 -0
- data/lib/neu/mods/projection/titles.rb +109 -0
- data/lib/neu/mods/projection.rb +41 -1516
- data/lib/neu/mods/selectors.rb +19 -94
- data/lib/neu/mods/text_normalizer.rb +118 -0
- data/lib/neu-mods.rb +6 -12
- metadata +57 -14
checksums.yaml
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
---
|
|
2
2
|
SHA256:
|
|
3
|
-
metadata.gz:
|
|
4
|
-
data.tar.gz:
|
|
3
|
+
metadata.gz: abcf4a1fa8c50f807e790a3de6999df9e93356275530ebd7aaac32a7b5c93dd0
|
|
4
|
+
data.tar.gz: b4c3c5b43549e420bfb0ce17b8db3983522f32e5695bf21c7e4a2b2e32fcb2e3
|
|
5
5
|
SHA512:
|
|
6
|
-
metadata.gz:
|
|
7
|
-
data.tar.gz:
|
|
6
|
+
metadata.gz: 1dca7512c1fd31592e5d9bcac19092574924d26da0345eb634e90481938b7e873f3c42c74064b73a276387ebf2ede68f1efdff789451587be0bd8682d508b145
|
|
7
|
+
data.tar.gz: 1471ba6e147cf9a4ab0f2cc5d19605a3bcfd82e5661ec63c81a9e385c271ebdb5bc6d982f726468dd6a667ca1080dec8dc097e12d41a49896b920ab0b686eb55
|
data/.version
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
0.14.
|
|
1
|
+
0.14.1
|
data/CHANGELOG.md
CHANGED
|
@@ -8,6 +8,38 @@ unannounced is one they discover as a nil value or a missing display row.
|
|
|
8
8
|
Releases before 0.14.0 are not recorded. Their diffs are in git; this file
|
|
9
9
|
starts where the convention does.
|
|
10
10
|
|
|
11
|
+
## 0.14.1
|
|
12
|
+
|
|
13
|
+
No projected shape changes. Two fixes change what the write path does, and the
|
|
14
|
+
code is reorganised with the public method list unchanged.
|
|
15
|
+
|
|
16
|
+
### Fixed
|
|
17
|
+
|
|
18
|
+
- **The builders find the MODS namespace by URI.** `build_node` looked for the
|
|
19
|
+
literal prefix `mods`. Under any other prefix, such as `<m:mods xmlns:m=...>`,
|
|
20
|
+
it built elements in no namespace. The projection could not see them, and
|
|
21
|
+
`editable_creator_nodes` could not replace them, so each save added another
|
|
22
|
+
copy. A document whose root declares no MODS namespace now raises
|
|
23
|
+
`ArgumentError`.
|
|
24
|
+
- **A blank authority attribute counts as absent on the edit path.** A name
|
|
25
|
+
with `authority=" "` was reported uncontrolled by `authority_of` but kept out
|
|
26
|
+
of the editable creators. It is now editable, matching the projection.
|
|
27
|
+
|
|
28
|
+
### Changed
|
|
29
|
+
|
|
30
|
+
- `Projection` is one mixin per MODS area under `lib/neu/mods/projection/`.
|
|
31
|
+
`Document`'s public and private methods and `to_h` are unchanged.
|
|
32
|
+
`Projection.compose_title`, `Projection.fold_type` and
|
|
33
|
+
`Projection::HEADING_SEPARATOR` still resolve.
|
|
34
|
+
- The node builders move from `Selectors` to a `Builders` mixin. `Document`
|
|
35
|
+
includes both, so a call through a `Document` is unchanged.
|
|
36
|
+
- Each file requires what it uses, so `require "neu/mods/document"` works
|
|
37
|
+
without the top-level entry file.
|
|
38
|
+
- The date accessors and their `FIELDS` rows are generated from
|
|
39
|
+
`Projection::Dates::DATE_FIELDS` and `DATE_KEYS`. The field names are
|
|
40
|
+
unchanged.
|
|
41
|
+
- The per-area explanation moves from source comments to `docs/`.
|
|
42
|
+
|
|
11
43
|
## 0.14.0
|
|
12
44
|
|
|
13
45
|
The projection gains the facts a consumer needs to offer a metadata value as a
|
data/README.md
CHANGED
|
@@ -1,23 +1,27 @@
|
|
|
1
1
|
# neu-mods
|
|
2
2
|
|
|
3
|
-
|
|
3
|
+
The MODS v3 reading contract for Northeastern's DRS, shared by
|
|
4
4
|
[Cerberus](https://github.com/NEU-Libraries/cerberus) (front end) and
|
|
5
5
|
[Atlas](https://github.com/NEU-Libraries/atlas) (API backend).
|
|
6
6
|
|
|
7
|
-
It is
|
|
8
|
-
|
|
9
|
-
|
|
7
|
+
It is pure functions over a parsed MODS document, with no Rails, no persistence
|
|
8
|
+
and no HTTP. It depends on Nokogiri alone, not on the `sul-dlss/mods` and
|
|
9
|
+
`nom-xml` stack. It answers two questions:
|
|
10
10
|
|
|
11
|
-
- **"Where is X?"**
|
|
12
|
-
the
|
|
13
|
-
|
|
14
|
-
|
|
15
|
-
|
|
16
|
-
(hashes/strings/arrays — never opaque typed objects) for indexing/display.
|
|
11
|
+
- **"Where is X?"** `Selectors` return live Nokogiri nodes. They serve the read
|
|
12
|
+
path and the write path, so the node an editor changes is the node the
|
|
13
|
+
projection reads.
|
|
14
|
+
- **"What does this project to?"** `Projection` returns plain data: hashes,
|
|
15
|
+
strings and arrays, for indexing and display.
|
|
17
16
|
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
|
|
17
|
+
## Installation
|
|
18
|
+
|
|
19
|
+
```ruby
|
|
20
|
+
# Gemfile
|
|
21
|
+
gem "neu-mods"
|
|
22
|
+
```
|
|
23
|
+
|
|
24
|
+
The gem needs Ruby 3.0 or later.
|
|
21
25
|
|
|
22
26
|
## Usage
|
|
23
27
|
|
|
@@ -26,231 +30,66 @@ require "neu-mods"
|
|
|
26
30
|
|
|
27
31
|
doc = NEU::MODS::Document.parse(xml_string)
|
|
28
32
|
|
|
29
|
-
# Projection
|
|
33
|
+
# Projection: plain data
|
|
34
|
+
doc.to_h # => every field in NEU::MODS::FIELDS
|
|
30
35
|
doc.plain_title # => "What's New. How We Respond to Disaster. Episode 1"
|
|
31
36
|
doc.title_parts # => { non_sort:, subtitle:, title:, part_name:, part_number: }
|
|
32
|
-
|
|
33
|
-
doc.abstract # => normalized, paragraph-joined String
|
|
34
|
-
doc.languages # => [{ term: "English", object_part: nil, script: nil }, ...]
|
|
35
|
-
# a code-only <languageTerm>eng</> is read through the
|
|
36
|
-
# ISO 639 registry. @objectPart rides along because
|
|
37
|
-
# objectPart="subtitles" says the SUBTITLES are Spanish,
|
|
38
|
-
# not the resource
|
|
39
|
-
doc.topical_subjects # => ["Civil society", ...] (every <topic>, for the access copy)
|
|
40
|
-
doc.keywords # => [...] (only the editable attribute-free keyword subjects)
|
|
41
|
-
doc.date_created_parts
|
|
42
|
-
# => { value:, precision:, end_value:, end_precision:,
|
|
43
|
-
# qualifier:, key_date:, text: } everything the record
|
|
44
|
-
# declared about one date. w3cdtf YYYY, YYYY-MM and
|
|
45
|
-
# YYYY-MM-DD all parse, and the precision says which
|
|
46
|
-
# shape it gave, so display cannot invent a month or a
|
|
47
|
-
# day. The points are read by @point, not by document
|
|
48
|
-
# order, and the end carries its OWN precision.
|
|
49
|
-
# A keyDate="yes" node chooses the value, ahead of
|
|
50
|
-
# @point and document order; one date per type is the
|
|
51
|
-
# rule, so an unflagged repeat is discarded.
|
|
52
|
-
# A value that is not a w3cdtf date projects NO date
|
|
53
|
-
# and keeps its literal in :text -- "19uu", "ca. 1920"
|
|
54
|
-
# and "undated" are statements a cataloguer made, and
|
|
55
|
-
# guessing a date for them is worse than either losing
|
|
56
|
-
# them or showing them as written.
|
|
57
|
-
# Same for the other six originInfo dates --
|
|
58
|
-
# date_issued, copyright_date, date_captured,
|
|
59
|
-
# date_valid, date_other and date_modified. Each part is
|
|
60
|
-
# also a reader of its own, e.g.
|
|
61
|
-
# doc.date_created_qualifier.
|
|
62
|
-
doc.place_of_publication
|
|
63
|
-
# => [{ value: "Boston", display_label:, href: }, ...]
|
|
64
|
-
# the type="text" placeTerm wins, and a bare marccountry
|
|
65
|
-
# code drops rather than reaching a places facet as a
|
|
66
|
-
# place name. A bare code under any other authority
|
|
67
|
-
# still projects
|
|
68
|
-
|
|
69
|
-
# A name entry also carries @usage (fixed="primary" in the schema, so a record
|
|
70
|
-
# that sets it has said which name leads) and :alternative_names, the MODS 3.7
|
|
71
|
-
# alternativeName composed with the ENCLOSING name's @type.
|
|
72
|
-
|
|
73
|
-
doc.origin_agents # => [{ name:, roles:, affiliation:, display_label:, href:,
|
|
74
|
-
# event_type: }, ...]
|
|
75
|
-
# originInfo/agent, new in MODS 3.8: who performed the
|
|
76
|
-
# event the block records
|
|
77
|
-
|
|
78
|
-
# An originInfo child also carries its block's @eventType, and a place carries
|
|
79
|
-
# the NAMES of the date elements beside it (:date_elements) -- "Creation place"
|
|
80
|
-
# and "Publication place" are the same element under a different date, and the
|
|
81
|
-
# place says nothing about the event itself. Each of the seven dates gains
|
|
82
|
-
# <date>_display_label and <date>_event_type from the same block.
|
|
83
|
-
|
|
84
|
-
# Every DISPLAYED projection carries the @displayLabel and xlink:href of the
|
|
85
|
-
# element its header comes from, as { value:, display_label:, href: } -- or as
|
|
86
|
-
# two extra keys where the entry already had a shape of its own. MODS puts the
|
|
87
|
-
# pair on originInfo and physicalDescription rather than on the publisher,
|
|
88
|
-
# place, extent or digitalOrigin inside them, so those children read it off
|
|
89
|
-
# their parent. The two attribute sets overlap rather than match (26 elements
|
|
90
|
-
# take @displayLabel, 14 take xlink:href); an element the schema gives neither
|
|
91
|
-
# projects nil. The four fields that JOIN several elements into one string --
|
|
92
|
-
# abstract and the three accessCondition fields -- take companion scalars
|
|
93
|
-
# instead (doc.abstract_display_label, doc.abstract_href).
|
|
94
|
-
|
|
95
|
-
# A BROWSABLE projection also carries the vocabulary its value was taken from,
|
|
96
|
-
# as { authority:, authority_uri:, value_uri: } -- names (and originInfo
|
|
97
|
-
# agents), languages, genres and subject headings. A consumer asking "may this
|
|
98
|
-
# value be offered as a browse?" asks for any of the three: MODS lets a record
|
|
99
|
-
# declare its vocabulary by URI alone, so requiring @authority would call an
|
|
100
|
-
# authorityURI-bearing corporate name uncontrolled.
|
|
101
|
-
#
|
|
102
|
-
# This resolves DIFFERENTLY from the pair above, which is why it is a separate
|
|
103
|
-
# port rather than three more keys on qualifiers_of. A header often comes from
|
|
104
|
-
# a PARENT; an authority never does. It is read off the element holding the
|
|
105
|
-
# value, then off an enclosing <subject> -- a pre-coordinated heading declares
|
|
106
|
-
# its vocabulary once, on the heading. A <role>/<roleTerm> authority is never
|
|
107
|
-
# consulted: it is the vocabulary of the RELATOR, and the deposit form writes a
|
|
108
|
-
# marcrelator roleTerm on every creator it collects.
|
|
109
|
-
doc.host_collections
|
|
110
|
-
# => [{ title:, volume:, issue:, start_page:, end_page:,
|
|
111
|
-
# date:, text:, details: [...], extents: [...] }, ...]
|
|
112
|
-
# this work's position in its host. The entry survives on
|
|
113
|
-
# its part alone, so a host with no titleInfo is kept
|
|
114
|
-
doc.identifiers # => [{ type: "isbn", value: "...", invalid: false,
|
|
115
|
-
# display_label:, href: }, ...]
|
|
116
|
-
# @invalid means cancelled or superseded, so it travels
|
|
117
|
-
doc.table_of_contents
|
|
118
|
-
# => [{ value: "Ch 1\nCh 2", ... }]
|
|
119
|
-
# line breaks kept: in a contents list
|
|
120
|
-
# the break is the structure, not stray formatting
|
|
121
|
-
doc.notes # => [{ type: "funding", value: "...", display_label:, href: }, ...]
|
|
122
|
-
doc.related_items # => [{ type: "otherFormat", title: "...",
|
|
123
|
-
# display_label:, href: }, ...]
|
|
124
|
-
# every relatedItem that is not a series or a host
|
|
125
|
-
doc.location # => [{ physical_location:, shelf_location:, url:,
|
|
126
|
-
# display_label:, href: }, ...]
|
|
127
|
-
doc.map_data # => [{ scale:, projection:, coordinates: }, ...]
|
|
128
|
-
doc.title_subjects # => ["The Great Gatsby"] composed like the main title
|
|
37
|
+
doc.names # => [{ name: "Cohen, Daniel J.(Daniel Jared), 1968-", roles: ["Creator"], ... }]
|
|
129
38
|
doc.subject_headings
|
|
130
|
-
# => [{ parts: ["Salt marshes
|
|
131
|
-
#
|
|
132
|
-
|
|
133
|
-
# value_uri:, display_label:, href: }, ...]
|
|
134
|
-
# one top-level <subject> as ONE heading. :heading is the
|
|
135
|
-
# parts joined, here rather than with the caller because
|
|
136
|
-
# it is both the string a display renders and the string
|
|
137
|
-
# a browse index holds -- two joins is how those drift.
|
|
138
|
-
# :axis names the element of the heading's MAIN term
|
|
139
|
-
# (its first child carrying heading text), which is what
|
|
140
|
-
# says which browse the heading belongs to: a place
|
|
141
|
-
# subdivision does not make a topic heading a place.
|
|
142
|
-
# A <name> axis splits by @type: "personal_name" or
|
|
143
|
-
# "corporate_name"
|
|
144
|
-
doc.hierarchical_geographic_subjects
|
|
145
|
-
# => [{ country:, state:, city:, ... }, ...] eleven levels,
|
|
146
|
-
# structured for the reason map_data is
|
|
147
|
-
doc.record_info # => { content_source:, origin:, description_standard:,
|
|
148
|
-
# creation_date:, change_date:, language_of_cataloging: }
|
|
149
|
-
# describes the CATALOGUING, not the resource
|
|
150
|
-
doc.to_h # => full projection, keyed to Atlas's Metadata::MODS attributes
|
|
39
|
+
# => [{ parts: [...], heading: "Salt marshes -- Massachusetts",
|
|
40
|
+
# axis: "topic", authority: "lcsh", ... }]
|
|
41
|
+
doc.date_issued # => a DateTime or nil, beside date_issued_precision, _end, ...
|
|
151
42
|
|
|
152
|
-
# The field registry
|
|
153
|
-
# name => :one or :many. to_h is derived from it, and a consumer builds its own
|
|
154
|
-
# schema from it rather than re-listing the field set by hand. Cardinality
|
|
155
|
-
# follows what MODS marks repeatable, so a field can never silently truncate.
|
|
43
|
+
# The field registry: field name => :one or :many
|
|
156
44
|
NEU::MODS::FIELDS # => { main_title: :one, names: :many, ... }
|
|
157
45
|
|
|
158
|
-
#
|
|
159
|
-
|
|
160
|
-
|
|
161
|
-
|
|
162
|
-
# => "What's New. How We Respond to Disaster. Episode 1" (== doc.plain_title)
|
|
163
|
-
# The part NUMBER precedes the part NAME: "Part 2. The Marshes" is the
|
|
164
|
-
# cataloguing convention, and titleInfo is an unordered choice in the schema.
|
|
46
|
+
# Title composition over parts a caller already holds, with no XML
|
|
47
|
+
NEU::MODS.compose_title(title: "What's New", part_name: "How We Respond to Disaster",
|
|
48
|
+
part_number: "Episode 1")
|
|
49
|
+
# => "What's New. How We Respond to Disaster. Episode 1"
|
|
165
50
|
|
|
166
|
-
# Selectors
|
|
51
|
+
# Selectors: live nodes, for editing
|
|
167
52
|
node = doc.primary_title_info.at_xpath("mods:title", NEU::MODS::NAMESPACE)
|
|
168
53
|
node.content = "New Title" unless NEU::MODS.whitespace_equivalent?(node.text, "New Title")
|
|
169
|
-
doc.to_xml
|
|
170
54
|
|
|
171
|
-
#
|
|
172
|
-
|
|
173
|
-
doc.
|
|
174
|
-
doc.editable_corporate_creators # => [{ name: }]
|
|
175
|
-
doc.preserved_names # => [{ name:, roles: }] (authority-bearing / non-Creator — read-only)
|
|
176
|
-
doc.editable_creator_nodes("personal") # => live <name> nodes to replace
|
|
177
|
-
doc.build_personal_name(given: "Jenny", family: "Smith") # => a plain personal <name> node
|
|
178
|
-
doc.build_corporate_name(name: "Northeastern University") # => a plain corporate <name> node
|
|
55
|
+
# Builders: new nodes in the document's MODS namespace
|
|
56
|
+
doc.doc.root.add_child(doc.build_corporate_name(name: "Northeastern University"))
|
|
57
|
+
doc.to_xml
|
|
179
58
|
```
|
|
180
59
|
|
|
181
|
-
|
|
182
|
-
|
|
183
|
-
|
|
184
|
-
|
|
185
|
-
|
|
186
|
-
|
|
187
|
-
|
|
188
|
-
|
|
189
|
-
|
|
190
|
-
|
|
191
|
-
|
|
192
|
-
|
|
193
|
-
|
|
194
|
-
|
|
195
|
-
|
|
196
|
-
|
|
197
|
-
|
|
198
|
-
|
|
199
|
-
|
|
200
|
-
|
|
201
|
-
|
|
202
|
-
|
|
203
|
-
|
|
204
|
-
|
|
205
|
-
|
|
206
|
-
|
|
207
|
-
projection an edit form reads.
|
|
208
|
-
|
|
209
|
-
## Behavior fidelity & known caveats
|
|
210
|
-
|
|
211
|
-
The projection is **behavior-preserving** with Atlas's prior `mods`-gem-based
|
|
212
|
-
extraction, pinned by `spec/conformance_spec.rb` against `work-mods.xml`. Two
|
|
213
|
-
intentional notes:
|
|
214
|
-
|
|
215
|
-
- **Name display** reproduces the `mods` gem's `display_value_w_date` *including
|
|
216
|
-
its quirks* (e.g. multiple `given` nameParts concatenate with no separator),
|
|
217
|
-
to preserve existing Solr/display output. Cleanups are a deliberate future
|
|
218
|
-
contract change, not a silent one.
|
|
219
|
-
- **Languages are translated; roles are not.** Both read the `type="text"` term
|
|
220
|
-
first. A code-only `languageTerm` is then translated through the vendored ISO
|
|
221
|
-
639 registry (`lib/neu/mods/data/iso639-2.txt`, from the Library of Congress),
|
|
222
|
-
so `eng` projects `English`. That happens here rather than in a consumer's
|
|
223
|
-
display layer because otherwise Solr indexes `eng` while the page shows
|
|
224
|
-
`English`, and the language facet reads in codes.
|
|
225
|
-
A code-only `roleTerm` stays raw. A MARC relator is a display *label*, and the
|
|
226
|
-
label vocabulary belongs to the consumer — Cerberus's edit form and Atlas's
|
|
227
|
-
display word the same role differently. An unrecognised language code also
|
|
228
|
-
stays raw, since the record still said something.
|
|
229
|
-
- **`description` is not projected.** MODS does define `name/description`, but
|
|
230
|
-
that annotates a *name*, not the resource, so it is not the field Atlas once
|
|
231
|
-
called `description`. The two candidates for that one — an `abstract` variant
|
|
232
|
-
and `physicalDescription/note` — describe different things. Projecting a guess
|
|
233
|
-
would put wrong data in the field rather than leave an empty one, so it waits
|
|
234
|
-
on a decision.
|
|
235
|
-
- **A date carries more than a value.** Each of `dateCreated`, `dateIssued` and
|
|
236
|
-
`copyrightDate` projects a value, its precision, an end value with its own
|
|
237
|
-
precision, the `@qualifier` and the `@keyDate` flag. The gem does not *pick*
|
|
238
|
-
the key date, because "which date to sort on" and "which date to display" are
|
|
239
|
-
not necessarily the same answer, and choosing is the consumer's job.
|
|
240
|
-
|
|
241
|
-
## Source convention
|
|
242
|
-
|
|
243
|
-
Every character-class regex in `TextNormalizer` is built **programmatically from
|
|
244
|
-
codepoints**, so the source stays pure ASCII (no literal smart-quotes/dashes, no
|
|
245
|
-
raw control bytes). A spec enforces this. Keep it that way.
|
|
60
|
+
What each field holds, and why, is documented per MODS area in
|
|
61
|
+
[`docs/`](docs/README.md). Start with [`docs/fields.md`](docs/fields.md).
|
|
62
|
+
|
|
63
|
+
## Behavior fidelity and known caveats
|
|
64
|
+
|
|
65
|
+
The projection preserves the output of Atlas's earlier `mods`-gem-based
|
|
66
|
+
extraction. `spec/conformance_spec.rb` pins it against `work-mods.xml`, so a
|
|
67
|
+
change to what the gem projects is a deliberate contract change.
|
|
68
|
+
|
|
69
|
+
- **Name display reproduces the `mods` gem's `display_value_w_date`, quirks
|
|
70
|
+
included.** For example, two `given` name parts join with no separator. This
|
|
71
|
+
preserves Atlas's Solr and display output. A cleanup would be a contract
|
|
72
|
+
change. See [`docs/names.md`](docs/names.md).
|
|
73
|
+
- **Languages are translated; roles are not.** A code-only `languageTerm` is
|
|
74
|
+
read through the vendored ISO 639 registry, so `eng` projects "English", and
|
|
75
|
+
Solr and the page agree. A code-only `roleTerm` stays raw, because a MARC
|
|
76
|
+
relator is a display label and each consumer words it differently. An
|
|
77
|
+
unrecognised language code also stays raw.
|
|
78
|
+
- **`description` is not projected.** MODS `name/description` annotates a name,
|
|
79
|
+
not the resource. The two candidates for a resource description, an
|
|
80
|
+
`abstract` variant and `physicalDescription/note`, describe different things.
|
|
81
|
+
A guess would put wrong data in the field, so it waits on a decision.
|
|
82
|
+
- **A date carries more than a value.** Each of the seven `originInfo` date
|
|
83
|
+
elements projects its value, precision, range end, qualifier, key-date flag,
|
|
84
|
+
literal text and block header. The gem does not choose which date to sort on
|
|
85
|
+
or display; that is the consumer's call. See [`docs/dates.md`](docs/dates.md).
|
|
246
86
|
|
|
247
87
|
## Development
|
|
248
88
|
|
|
249
|
-
```
|
|
89
|
+
```sh
|
|
250
90
|
bundle install
|
|
251
|
-
bundle exec
|
|
252
|
-
bundle exec rubocop
|
|
91
|
+
bundle exec rake # specs, then rubocop
|
|
253
92
|
```
|
|
254
93
|
|
|
255
|
-
|
|
256
|
-
with `bundler/gem_tasks` (`rake release`)
|
|
94
|
+
The version lives in `.version`, which `lib/neu/mods/version.rb` reads. Release
|
|
95
|
+
with `bundler/gem_tasks` (`rake release`).
|
|
@@ -0,0 +1,63 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "nokogiri"
|
|
4
|
+
|
|
5
|
+
require_relative "namespaces"
|
|
6
|
+
|
|
7
|
+
module NEU
|
|
8
|
+
module MODS
|
|
9
|
+
# Node CREATION for the write path. Cerberus's MODSMerge builds every
|
|
10
|
+
# element it adds through these, so their names and signatures are part of
|
|
11
|
+
# the contract. See docs/editing.md.
|
|
12
|
+
module Builders
|
|
13
|
+
# A MODS element reusing the root's namespace declaration, matched by URI:
|
|
14
|
+
# built outside MODS, an element is invisible to every XPath here.
|
|
15
|
+
def build_node(name, text = nil)
|
|
16
|
+
node = Nokogiri::XML::Node.new(name, doc)
|
|
17
|
+
node.namespace = mods_namespace_definition
|
|
18
|
+
node.content = text unless text.nil?
|
|
19
|
+
node
|
|
20
|
+
end
|
|
21
|
+
|
|
22
|
+
# A plain personal creator: given and family parts and a text roleTerm.
|
|
23
|
+
def build_personal_name(given:, family:, role: "Creator")
|
|
24
|
+
name = build_node("name")
|
|
25
|
+
name["type"] = "personal"
|
|
26
|
+
name.add_child(name_part(given, "given")) unless given.to_s.strip.empty?
|
|
27
|
+
name.add_child(name_part(family, "family")) unless family.to_s.strip.empty?
|
|
28
|
+
name.add_child(role_node(role))
|
|
29
|
+
name
|
|
30
|
+
end
|
|
31
|
+
|
|
32
|
+
# A plain corporate creator: one name part and a text roleTerm.
|
|
33
|
+
def build_corporate_name(name:, role: "Creator")
|
|
34
|
+
node = build_node("name")
|
|
35
|
+
node["type"] = "corporate"
|
|
36
|
+
node.add_child(name_part(name)) unless name.to_s.strip.empty?
|
|
37
|
+
node.add_child(role_node(role))
|
|
38
|
+
node
|
|
39
|
+
end
|
|
40
|
+
|
|
41
|
+
private
|
|
42
|
+
|
|
43
|
+
def mods_namespace_definition
|
|
44
|
+
doc.root.namespace_definitions.find { |d| d.href == NAMESPACE["mods"] } ||
|
|
45
|
+
raise(ArgumentError, "document declares no MODS namespace on its root element")
|
|
46
|
+
end
|
|
47
|
+
|
|
48
|
+
def name_part(text, type = nil)
|
|
49
|
+
np = build_node("namePart", text.to_s.strip)
|
|
50
|
+
np["type"] = type if type
|
|
51
|
+
np
|
|
52
|
+
end
|
|
53
|
+
|
|
54
|
+
def role_node(role)
|
|
55
|
+
role_el = build_node("role")
|
|
56
|
+
term = build_node("roleTerm", role)
|
|
57
|
+
term["type"] = "text"
|
|
58
|
+
role_el.add_child(term)
|
|
59
|
+
role_el
|
|
60
|
+
end
|
|
61
|
+
end
|
|
62
|
+
end
|
|
63
|
+
end
|
|
@@ -2,29 +2,22 @@
|
|
|
2
2
|
|
|
3
3
|
module NEU
|
|
4
4
|
module MODS
|
|
5
|
-
#
|
|
6
|
-
#
|
|
7
|
-
#
|
|
8
|
-
# deliberately distinct from TextNormalizer below: this one only folds
|
|
9
|
-
# whitespace; TextNormalizer cleans curator freetext for the access copy.
|
|
5
|
+
# Whitespace canonicalization for the no-op guard: did an edit change
|
|
6
|
+
# anything, or only insignificant whitespace? Distinct from TextNormalizer,
|
|
7
|
+
# which cleans text for the access copy. See docs/text-normalization.md.
|
|
10
8
|
module Canonicalize
|
|
11
9
|
module_function
|
|
12
10
|
|
|
13
11
|
NBSP = [0xA0].pack("U") # U+00A0 non-breaking space, built from codepoint
|
|
14
12
|
|
|
15
|
-
#
|
|
16
|
-
# plain space first, then collapse any whitespace run to one space + strip.
|
|
13
|
+
# Fold NBSP first: Ruby's \s does not match U+00A0.
|
|
17
14
|
def canonical_ws(str)
|
|
18
15
|
str.to_s.tr(NBSP, " ").gsub(/\s+/, " ").strip
|
|
19
16
|
end
|
|
20
17
|
|
|
21
|
-
# canonical_ws per line, keeping the
|
|
22
|
-
# newline is structure rather than formatting -- tableOfContents, where
|
|
23
|
-
# the break separates one entry from the next. Blank lines drop, so a
|
|
24
|
-
# double-spaced list does not project empty entries.
|
|
18
|
+
# canonical_ws per line, keeping the breaks and dropping blank lines.
|
|
25
19
|
def canonical_lines(str)
|
|
26
|
-
str.to_s.
|
|
27
|
-
.reject(&:empty?).join("\n")
|
|
20
|
+
str.to_s.split("\n").map { |line| canonical_ws(line) }.reject(&:empty?).join("\n")
|
|
28
21
|
end
|
|
29
22
|
|
|
30
23
|
# Treat values differing only by insignificant whitespace (NBSP vs space,
|
|
@@ -33,143 +26,5 @@ module NEU
|
|
|
33
26
|
canonical_ws(current) == canonical_ws(incoming)
|
|
34
27
|
end
|
|
35
28
|
end
|
|
36
|
-
|
|
37
|
-
# Normalises curator-authored freetext on the way into the JSON access copy
|
|
38
|
-
# (and Solr); the XML preservation copy stays untouched. Ported from Atlas's
|
|
39
|
-
# TextNormalizer (which carries DRS v1 prior art) so the gem reproduces Atlas's
|
|
40
|
-
# projection byte-for-byte.
|
|
41
|
-
#
|
|
42
|
-
# IMPORTANT: every character-class regex is built *programmatically* from
|
|
43
|
-
# codepoint lists via `format('\\u%04X', cp)`, so this source file stays pure
|
|
44
|
-
# ASCII -- no literal smart-quotes, dashes, or (critically) raw control bytes
|
|
45
|
-
# land on disk. Keep it that way.
|
|
46
|
-
#
|
|
47
|
-
# Pipeline: force UTF-8 + scrub invalid bytes; NFC; map Unicode dashes to '-'
|
|
48
|
-
# (swung-dash to '~'); transliterate the General Punctuation block (smart
|
|
49
|
-
# quotes, ellipsis, etc.) to ASCII; map the separator controls to a newline
|
|
50
|
-
# and drop what is invisible (the soft hyphen, the rest of C0/C1, keeping
|
|
51
|
-
# tab/newline); collapse
|
|
52
|
-
# horizontal-whitespace runs to one space; for paragraph fields, collapse
|
|
53
|
-
# 2+ newlines to exactly two; strip.
|
|
54
|
-
#
|
|
55
|
-
# .normalize(str) -- single-line fields (newlines -> spaces)
|
|
56
|
-
# .normalize_paragraphs(str) -- fields that may carry paragraph breaks
|
|
57
|
-
# (abstract, accessCondition)
|
|
58
|
-
module TextNormalizer
|
|
59
|
-
module_function
|
|
60
|
-
|
|
61
|
-
# Build a character-class Regexp from an array of integer codepoints, as
|
|
62
|
-
# \uXXXX escapes (keeps this source ASCII).
|
|
63
|
-
def self.char_class(codepoints, prefix: "")
|
|
64
|
-
Regexp.new("[#{prefix}#{codepoints.map { |cp| format('\\u%04X', cp) }.join}]")
|
|
65
|
-
end
|
|
66
|
-
|
|
67
|
-
# NOTE: U+2053 (swung dash) is intentionally excluded from dashes -- it is
|
|
68
|
-
# named "dash" but conventionally maps to ASCII '~', not '-' (V1 prior art).
|
|
69
|
-
DASH_CODEPOINTS = [
|
|
70
|
-
0x002D, 0x058A, 0x05BE, 0x1400, 0x1806,
|
|
71
|
-
0x2010, 0x2011, 0x2012, 0x2013, 0x2014, 0x2015,
|
|
72
|
-
0x2043, 0x207B, 0x208B, 0x2212,
|
|
73
|
-
0x2E17, 0x2E1A, 0x2E3A, 0x2E3B, 0x2E40,
|
|
74
|
-
0x301C, 0x3030, 0x30A0, 0xFE31, 0xFE32, 0xFE58,
|
|
75
|
-
0xFE63, 0xFF0D
|
|
76
|
-
].freeze
|
|
77
|
-
DASH_RE = char_class(DASH_CODEPOINTS).freeze
|
|
78
|
-
|
|
79
|
-
SWUNG_DASH_RE = Regexp.new(format('\\u%04X', 0x2053)).freeze
|
|
80
|
-
|
|
81
|
-
# U+00AD is a hint about where a word may break, not a dash: it renders as
|
|
82
|
-
# nothing, and Solr discards it, so "co<00AD>operation" already matches a
|
|
83
|
-
# search for "cooperation". Mapping it to an ASCII hyphen instead would
|
|
84
|
-
# index "co" and "operation" as two tokens and lose the word, so it is
|
|
85
|
-
# dropped and kept out of DASH_CODEPOINTS.
|
|
86
|
-
SOFT_HYPHEN_RE = Regexp.new(format('\\u%04X', 0x00AD)).freeze
|
|
87
|
-
|
|
88
|
-
# U+000B (vertical tab) and U+000C (form feed) separate words rather than
|
|
89
|
-
# meaning nothing: Word writes a manual line break as U+000B and a page
|
|
90
|
-
# break as U+000C. They map to a newline, because deleting one runs the
|
|
91
|
-
# words either side of it together -- normalize_paragraphs then reads that
|
|
92
|
-
# newline as the soft wrap the line break was, and normalize turns it into
|
|
93
|
-
# a space.
|
|
94
|
-
SEPARATOR_CONTROL_CODEPOINTS = [0x000B, 0x000C].freeze
|
|
95
|
-
SEPARATOR_CONTROL_RE = char_class(SEPARATOR_CONTROL_CODEPOINTS).freeze
|
|
96
|
-
|
|
97
|
-
# C0 (U+0000..U+0008, U+000D..U+001F) and C1 (U+007F..U+009F) -- what is
|
|
98
|
-
# left once the separators above are accounted for, and none of it carries
|
|
99
|
-
# meaning in curator text. U+0009 (tab) and U+000A (newline) are preserved.
|
|
100
|
-
# U+000D is not: dropping it reduces a CRLF line ending to the single
|
|
101
|
-
# newline it stands for.
|
|
102
|
-
CONTROL_CODEPOINTS = ((0x0000..0x0008).to_a + (0x000D..0x001F).to_a + (0x007F..0x009F).to_a).freeze
|
|
103
|
-
CONTROL_RE = char_class(CONTROL_CODEPOINTS).freeze
|
|
104
|
-
|
|
105
|
-
HORIZONTAL_WS_CODEPOINTS = [
|
|
106
|
-
0x0009, 0x00A0, 0x1680,
|
|
107
|
-
0x2000, 0x2001, 0x2002, 0x2003, 0x2004, 0x2005, 0x2006,
|
|
108
|
-
0x2007, 0x2008, 0x2009, 0x200A, 0x202F, 0x205F, 0x3000
|
|
109
|
-
].freeze
|
|
110
|
-
# Leading literal space included in the class (the " " prefix); `+` so a run
|
|
111
|
-
# of horizontal whitespace collapses to a single space.
|
|
112
|
-
HORIZONTAL_WS_RE = Regexp.new("#{char_class(HORIZONTAL_WS_CODEPOINTS, prefix: " ").source}+").freeze
|
|
113
|
-
|
|
114
|
-
PARAGRAPH_RUN_RE = /\n{2,}/
|
|
115
|
-
|
|
116
|
-
# General Punctuation block (U+2000..U+206F). Codepoints not listed pass
|
|
117
|
-
# through unchanged. Empty-string values deliberately drop invisible/bidi/
|
|
118
|
-
# format marks so they cannot leak into the access copy.
|
|
119
|
-
GENERAL_PUNCTUATION = {
|
|
120
|
-
0x2000 => " ", 0x2001 => " ", 0x2002 => " ", 0x2003 => " ",
|
|
121
|
-
0x2004 => " ", 0x2005 => " ", 0x2006 => " ", 0x2007 => " ",
|
|
122
|
-
0x2008 => " ", 0x2009 => " ", 0x200A => " ",
|
|
123
|
-
0x200B => "", 0x200C => "", 0x200D => "",
|
|
124
|
-
0x200E => "", 0x200F => "",
|
|
125
|
-
0x2018 => "'", 0x2019 => "'", 0x201A => ",", 0x201B => "'",
|
|
126
|
-
0x201C => '"', 0x201D => '"', 0x201E => '"', 0x201F => '"',
|
|
127
|
-
0x2020 => "+", 0x2021 => "+",
|
|
128
|
-
0x2022 => "*", 0x2023 => "*", 0x2024 => ".", 0x2025 => "..",
|
|
129
|
-
0x2026 => "...",
|
|
130
|
-
0x2028 => "\n", 0x2029 => "\n\n",
|
|
131
|
-
0x202A => "", 0x202B => "", 0x202C => "", 0x202D => "",
|
|
132
|
-
0x202E => "", 0x202F => " ",
|
|
133
|
-
0x2030 => "%", 0x2032 => "'", 0x2033 => '"', 0x2035 => "'",
|
|
134
|
-
0x2036 => '"',
|
|
135
|
-
0x2039 => "<", 0x203A => ">", 0x203C => "!!", 0x203D => "?",
|
|
136
|
-
0x2044 => "/", 0x2052 => "%",
|
|
137
|
-
0x205F => " ", 0x2060 => "", 0x2061 => "", 0x2062 => "",
|
|
138
|
-
0x2063 => "", 0x2064 => "",
|
|
139
|
-
0x206A => "", 0x206B => "", 0x206C => "", 0x206D => "",
|
|
140
|
-
0x206E => "", 0x206F => ""
|
|
141
|
-
}.transform_keys { |cp| [cp].pack("U") }.freeze
|
|
142
|
-
GENERAL_PUNCTUATION_RE = Regexp.new("[#{format('\\u%04X-\\u%04X', 0x2000, 0x206F)}]").freeze
|
|
143
|
-
|
|
144
|
-
def normalize(str)
|
|
145
|
-
return "" if str.nil?
|
|
146
|
-
|
|
147
|
-
s = base_normalize(str.to_s)
|
|
148
|
-
s = s.tr("\n", " ")
|
|
149
|
-
s.gsub(HORIZONTAL_WS_RE, " ").strip
|
|
150
|
-
end
|
|
151
|
-
|
|
152
|
-
def normalize_paragraphs(str)
|
|
153
|
-
return "" if str.nil?
|
|
154
|
-
|
|
155
|
-
s = base_normalize(str.to_s)
|
|
156
|
-
s = s.gsub(HORIZONTAL_WS_RE, " ")
|
|
157
|
-
s = s.gsub(/ *\n */, "\n")
|
|
158
|
-
s.split(PARAGRAPH_RUN_RE).map { |p| p.tr("\n", " ").strip }
|
|
159
|
-
.reject(&:empty?).join("\n\n")
|
|
160
|
-
end
|
|
161
|
-
|
|
162
|
-
def base_normalize(str)
|
|
163
|
-
s = str.dup.force_encoding("UTF-8")
|
|
164
|
-
s = s.scrub("")
|
|
165
|
-
s = s.unicode_normalize(:nfc)
|
|
166
|
-
s = s.gsub(DASH_RE, "-")
|
|
167
|
-
s = s.gsub(SWUNG_DASH_RE, "~")
|
|
168
|
-
s = s.gsub(GENERAL_PUNCTUATION_RE) { |c| GENERAL_PUNCTUATION.fetch(c, c) }
|
|
169
|
-
s = s.gsub(SEPARATOR_CONTROL_RE, "\n")
|
|
170
|
-
s = s.gsub(SOFT_HYPHEN_RE, "")
|
|
171
|
-
s.gsub(CONTROL_RE, "")
|
|
172
|
-
end
|
|
173
|
-
end
|
|
174
29
|
end
|
|
175
30
|
end
|
data/lib/neu/mods/document.rb
CHANGED
|
@@ -2,6 +2,10 @@
|
|
|
2
2
|
|
|
3
3
|
require "nokogiri"
|
|
4
4
|
|
|
5
|
+
require_relative "selectors"
|
|
6
|
+
require_relative "builders"
|
|
7
|
+
require_relative "projection"
|
|
8
|
+
|
|
5
9
|
module NEU
|
|
6
10
|
module MODS
|
|
7
11
|
# The gem's main entry point: a thin facade over a parsed MODS document.
|
|
@@ -16,6 +20,7 @@ module NEU
|
|
|
16
20
|
# avoid spurious whitespace-only text nodes.
|
|
17
21
|
class Document
|
|
18
22
|
include Selectors
|
|
23
|
+
include Builders
|
|
19
24
|
include Projection
|
|
20
25
|
|
|
21
26
|
attr_reader :doc
|