sslabdata 3.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- sslabdata/__init__.py +40 -0
- sslabdata/assembler.py +452 -0
- sslabdata/cli.py +274 -0
- sslabdata/config.py +443 -0
- sslabdata/diagnostics.py +159 -0
- sslabdata/exporters.py +86 -0
- sslabdata/loaders.py +298 -0
- sslabdata/models.py +342 -0
- sslabdata/parsers/__init__.py +0 -0
- sslabdata/parsers/bibtex.py +1132 -0
- sslabdata/parsers/latex.py +215 -0
- sslabdata/resolver.py +517 -0
- sslabdata/schema/__init__.py +0 -0
- sslabdata/schema/v5/output.schema.json +458 -0
- sslabdata-3.0.0.dist-info/METADATA +409 -0
- sslabdata-3.0.0.dist-info/RECORD +20 -0
- sslabdata-3.0.0.dist-info/WHEEL +5 -0
- sslabdata-3.0.0.dist-info/entry_points.txt +2 -0
- sslabdata-3.0.0.dist-info/licenses/LICENSE +21 -0
- sslabdata-3.0.0.dist-info/top_level.txt +1 -0
sslabdata/resolver.py
ADDED
|
@@ -0,0 +1,517 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Entity resolution: link works to people and projects, and compute
|
|
3
|
+
back-links. Names are matched as SPEC.md "How a name is matched" says.
|
|
4
|
+
|
|
5
|
+
Copyright (c) 2024 Personal Robotics Laboratory, University of Washington
|
|
6
|
+
Author: Siddhartha Srinivasa
|
|
7
|
+
MIT License - see LICENSE file for details.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
import re
|
|
11
|
+
import unicodedata
|
|
12
|
+
from difflib import SequenceMatcher
|
|
13
|
+
from typing import Dict, List, NamedTuple, Optional, Sequence, Set, Tuple
|
|
14
|
+
|
|
15
|
+
from .diagnostics import Diagnostic, diagnostic
|
|
16
|
+
from .models import Contributor, Work, Person, Project, LabData
|
|
17
|
+
from .parsers.bibtex import given_words
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
# Default fuzzy match threshold (0.0 to 1.0). A fuzzy match is only ever a
|
|
21
|
+
# suggestion reported to a human; it never links anything.
|
|
22
|
+
FUZZY_THRESHOLD = 0.85
|
|
23
|
+
|
|
24
|
+
# Pattern for abbreviated names: single initial + surname (e.g., "A. Kim")
|
|
25
|
+
# After normalization (no periods): "a kim", "h zhang", etc.
|
|
26
|
+
_ABBREVIATED_NAME_RE = re.compile(r'^[a-z] [a-z]+$')
|
|
27
|
+
|
|
28
|
+
# `resolution.status` and `resolution.method` values, both open strings
|
|
29
|
+
# (SPEC.md §5).
|
|
30
|
+
RESOLVED, UNRESOLVED, AMBIGUOUS = "resolved", "unresolved", "ambiguous"
|
|
31
|
+
BY_NAME = "exact"
|
|
32
|
+
|
|
33
|
+
# The codes this module reports (SPEC.md "Diagnostic codes").
|
|
34
|
+
AMBIGUOUS_NAME = "RESOLVE-AMBIGUOUS-NAME"
|
|
35
|
+
SUGGESTION = "RESOLVE-SUGGESTION"
|
|
36
|
+
ALIAS_AMBIGUOUS = "PEOPLE-ALIAS-AMBIGUOUS"
|
|
37
|
+
PROJECT_UNKNOWN = "RESOLVE-PROJECT-UNKNOWN"
|
|
38
|
+
|
|
39
|
+
# One part of a given name that is an initial rather than a name: a letter,
|
|
40
|
+
# its period optional, and a hyphenated run of them -- `A.`, `A`, `G.-A.`,
|
|
41
|
+
# `J-P`. A Unicode letter, so `Ç.` is read as an initial and a name outside
|
|
42
|
+
# ASCII is not silently exempt. Tested once combining marks are off, so a
|
|
43
|
+
# letter with a mark that has no precomposed form, `Q̇.`, is one too. A part
|
|
44
|
+
# that is anything else is read as a name, which is the safe direction for a
|
|
45
|
+
# warning: it reports one grouping key too few rather than one too many.
|
|
46
|
+
_INITIAL = re.compile(r"^[^\W\d_]\.?(?:-[^\W\d_]\.?)*$", re.UNICODE)
|
|
47
|
+
|
|
48
|
+
# Initials written without a space between them, `S.S.` or `T.A.K.`, read as
|
|
49
|
+
# one initial per letter in a given name only. Tested once combining marks
|
|
50
|
+
# are off. SPEC.md "How a name is matched" has the rule.
|
|
51
|
+
_RUN_TOGETHER = re.compile(r"^[^\W\d_](?:\.[^\W\d_])+\.?$", re.UNICODE)
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def _is_mark(c: str) -> bool:
|
|
55
|
+
return unicodedata.category(c) == 'Mn'
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def _without_marks(text: str) -> str:
|
|
59
|
+
return ''.join(c for c in unicodedata.normalize('NFD', text) if not _is_mark(c))
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def _is_initial(part: str) -> bool:
|
|
63
|
+
return bool(_INITIAL.match(_without_marks(part)))
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def _split_initials(word: str) -> List[str]:
|
|
67
|
+
"""``S.S.`` → ``["S.", "S."]``; any other word is returned whole.
|
|
68
|
+
|
|
69
|
+
The initials come back without their combining marks: every reader of
|
|
70
|
+
them removes marks too (`normalize_name()`, `_is_initial()`).
|
|
71
|
+
"""
|
|
72
|
+
bare = _without_marks(word)
|
|
73
|
+
if not _RUN_TOGETHER.match(bare):
|
|
74
|
+
return [word]
|
|
75
|
+
return [c + '.' for c in bare if c != '.']
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def _spaced_initials(text: str) -> str:
|
|
79
|
+
"""``S.S. Adams`` → ``S. S. Adams``: run-together initials, one per part."""
|
|
80
|
+
return ' '.join(part for word in text.split() for part in _split_initials(word))
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def normalize_name(name: str) -> str:
|
|
84
|
+
"""Normalize a name for matching.
|
|
85
|
+
|
|
86
|
+
Exactly the steps SPEC.md "How a name is matched" lists, so a change
|
|
87
|
+
here is a change to the contract.
|
|
88
|
+
"""
|
|
89
|
+
name = name.lower().strip()
|
|
90
|
+
name = unicodedata.normalize('NFC', ''.join(
|
|
91
|
+
c for c in unicodedata.normalize('NFD', name)
|
|
92
|
+
if unicodedata.category(c) != 'Mn'
|
|
93
|
+
))
|
|
94
|
+
name = name.replace('.', '')
|
|
95
|
+
name = re.sub(r'<sup>.*?</sup>', '', name)
|
|
96
|
+
name = re.sub(r'\s+', ' ', name).strip()
|
|
97
|
+
return name
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def declared_form(name: str) -> str:
|
|
101
|
+
"""How a declared name or alias compares with another declaration
|
|
102
|
+
(SPEC.md "How a name is matched").
|
|
103
|
+
|
|
104
|
+
A declaration is read as `full_form()` joins a name, `Given von Family,
|
|
105
|
+
Suffix`. Where the parse does not line up with the words, it is
|
|
106
|
+
`normalize_name()` exactly.
|
|
107
|
+
"""
|
|
108
|
+
before = name.split(",", 1)[0]
|
|
109
|
+
words = before.split()
|
|
110
|
+
given = given_words(before)
|
|
111
|
+
if words[:len(given)] != given or all(
|
|
112
|
+
_split_initials(word) == [word] for word in given):
|
|
113
|
+
return normalize_name(name)
|
|
114
|
+
spaced = [_spaced_initials(word) for word in given] + words[len(given):]
|
|
115
|
+
return normalize_name(" ".join(spaced) + name[len(before):])
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
def written_form(contributor: Contributor) -> str:
|
|
119
|
+
"""How an unresolved name is grouped: `normalize_name()` of the readable
|
|
120
|
+
name, with run-together initials spaced in its structured given name, so
|
|
121
|
+
`S.S. Quinn` and `S. S. Quinn` are one grouping. A brace-protected name,
|
|
122
|
+
or one with nothing to space, is `normalize_name()` of the name exactly.
|
|
123
|
+
"""
|
|
124
|
+
given = contributor.given or ""
|
|
125
|
+
if (contributor.literal or not contributor.name.startswith(given)
|
|
126
|
+
or _spaced_initials(given) == " ".join(given.split())):
|
|
127
|
+
return normalize_name(contributor.name)
|
|
128
|
+
return normalize_name(_spaced_initials(given) + contributor.name[len(given):])
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
def _given_parts(given: Optional[str]) -> List[str]:
|
|
132
|
+
return [part for part in _spaced_initials(given or "").split() if part]
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
def initials_only(given: Optional[str]) -> bool:
|
|
136
|
+
"""True when every part of a given name is an initial rather than a name."""
|
|
137
|
+
parts = _given_parts(given)
|
|
138
|
+
return bool(parts) and all(_is_initial(part) for part in parts)
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
def has_initial(given: Optional[str]) -> bool:
|
|
142
|
+
"""True when any part of a given name is an initial: ``Dave M.``, ``A.``"""
|
|
143
|
+
return any(_is_initial(part) for part in _given_parts(given))
|
|
144
|
+
|
|
145
|
+
|
|
146
|
+
def given_initials(given: Optional[str]) -> Tuple[str, ...]:
|
|
147
|
+
"""The initial of each part of a given name, normalised.
|
|
148
|
+
|
|
149
|
+
``Alice Jane`` and ``A. J.`` both give ``("a", "j")``; ``Grace-Ann`` and
|
|
150
|
+
``G.-A.`` both give ``("g-a",)``, because both halves of a hyphenated
|
|
151
|
+
given name carry an initial.
|
|
152
|
+
"""
|
|
153
|
+
return tuple("-".join(normalize_name(piece)[:1]
|
|
154
|
+
for piece in part.split("-") if piece)
|
|
155
|
+
for part in _given_parts(given))
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
def _initials(given: str) -> str:
|
|
159
|
+
"""Abbreviate one given name: ``Alice`` → ``A.``, ``Grace-Ann`` → ``G.-A.``"""
|
|
160
|
+
parts = [part for part in given.split("-") if part]
|
|
161
|
+
return "-".join(f"{part[0]}." for part in parts)
|
|
162
|
+
|
|
163
|
+
|
|
164
|
+
def full_form(contributor: Contributor) -> str:
|
|
165
|
+
"""The form a full name is matched on: ``Alice Jane van Last, Jr.``
|
|
166
|
+
|
|
167
|
+
The structured parts in reading order, with the suffix after a comma as
|
|
168
|
+
`people.yaml` writes it. Nothing is abbreviated. A name written as one
|
|
169
|
+
brace-protected unit is matched as written.
|
|
170
|
+
"""
|
|
171
|
+
if contributor.literal:
|
|
172
|
+
return contributor.literal
|
|
173
|
+
name = " ".join(part for part in (contributor.given, contributor.von,
|
|
174
|
+
contributor.family) if part)
|
|
175
|
+
suffix = contributor.suffix
|
|
176
|
+
return f"{name}, {suffix}" if suffix and name else (name or suffix or "")
|
|
177
|
+
|
|
178
|
+
|
|
179
|
+
def is_abbreviated(name: str) -> bool:
|
|
180
|
+
"""Check if a normalized name is a single-initial abbreviation.
|
|
181
|
+
|
|
182
|
+
Returns True for names like "a kim" or "h zhang" — these have
|
|
183
|
+
too little information for reliable fuzzy matching.
|
|
184
|
+
"""
|
|
185
|
+
return bool(_ABBREVIATED_NAME_RE.match(name))
|
|
186
|
+
|
|
187
|
+
|
|
188
|
+
def shared_declarations(people: List[Person], source: str) -> List[Diagnostic]:
|
|
189
|
+
"""One `ALIAS_AMBIGUOUS` warning per spelling more than one person declares.
|
|
190
|
+
|
|
191
|
+
Compared through `declared_form()`, so `S.S. Ivers` and `S. S. Ivers`
|
|
192
|
+
are one spelling.
|
|
193
|
+
"""
|
|
194
|
+
declared: Dict[str, List[Tuple[str, str, str]]] = {}
|
|
195
|
+
for person in people:
|
|
196
|
+
for field_name, written in [("name", person.name)] + [
|
|
197
|
+
("aliases", alias) for alias in person.aliases]:
|
|
198
|
+
key = declared_form(written)
|
|
199
|
+
owners = declared.setdefault(key, [])
|
|
200
|
+
if key and person.id not in [owner for owner, _, _ in owners]:
|
|
201
|
+
owners.append((person.id, field_name, written))
|
|
202
|
+
reported = []
|
|
203
|
+
for owners in declared.values():
|
|
204
|
+
if len(owners) < 2:
|
|
205
|
+
continue
|
|
206
|
+
_, field_name, _ = owners[1]
|
|
207
|
+
spellings = sorted({repr(written) for _, _, written in owners})
|
|
208
|
+
reported.append(diagnostic(
|
|
209
|
+
ALIAS_AMBIGUOUS, source, owners[1][0], field_name,
|
|
210
|
+
f"{' / '.join(spellings)} is declared by "
|
|
211
|
+
f"{', '.join(owner for owner, _, _ in owners)}; a name written "
|
|
212
|
+
"this way fits all of them and resolves to none"))
|
|
213
|
+
return reported
|
|
214
|
+
|
|
215
|
+
|
|
216
|
+
def fuzzy_matches(name: str, candidates: "Candidates",
|
|
217
|
+
threshold: float = FUZZY_THRESHOLD) -> List[str]:
|
|
218
|
+
"""Every id tied at the best similarity at or above the threshold,
|
|
219
|
+
sorted, so a tie does not depend on the order of `people.yaml`.
|
|
220
|
+
|
|
221
|
+
Compared against each normalised form exactly one entity declares: a form
|
|
222
|
+
more than one declares suggests nobody, and `shared_declarations()`
|
|
223
|
+
reports it. A single-initial name such as ``S. Zhang`` is never compared,
|
|
224
|
+
because it carries too little to suggest anyone.
|
|
225
|
+
"""
|
|
226
|
+
normalized = normalize_name(name)
|
|
227
|
+
if is_abbreviated(normalized):
|
|
228
|
+
return []
|
|
229
|
+
|
|
230
|
+
best_ratio = 0.0
|
|
231
|
+
best_ids: Set[str] = set()
|
|
232
|
+
|
|
233
|
+
for key, ids in candidates.exact.items():
|
|
234
|
+
if len(ids) != 1:
|
|
235
|
+
continue
|
|
236
|
+
ratio = SequenceMatcher(None, normalized, key).ratio()
|
|
237
|
+
if ratio > best_ratio:
|
|
238
|
+
best_ratio, best_ids = ratio, set(ids)
|
|
239
|
+
elif ratio == best_ratio:
|
|
240
|
+
best_ids |= ids
|
|
241
|
+
|
|
242
|
+
return sorted(best_ids) if best_ratio >= threshold else []
|
|
243
|
+
|
|
244
|
+
|
|
245
|
+
def _is_initial_token(token: str) -> bool:
|
|
246
|
+
pieces = [piece for piece in token.split("-") if piece]
|
|
247
|
+
return bool(pieces) and all(len(piece) == 1 for piece in pieces)
|
|
248
|
+
|
|
249
|
+
|
|
250
|
+
def _tokens_agree(written: str, declared: str) -> bool:
|
|
251
|
+
"""One given-name part against another: equal, or the same initials
|
|
252
|
+
where either side is only an initial."""
|
|
253
|
+
if _is_initial_token(written) or _is_initial_token(declared):
|
|
254
|
+
return (tuple(p[0] for p in written.split("-") if p)
|
|
255
|
+
== tuple(p[0] for p in declared.split("-") if p))
|
|
256
|
+
return written == declared
|
|
257
|
+
|
|
258
|
+
|
|
259
|
+
class NameKey(NamedTuple):
|
|
260
|
+
"""A name as matching compares it, through `normalize_name()`: its given
|
|
261
|
+
name piece by piece, run-together initials one piece per letter, and the
|
|
262
|
+
rest as one string. ``S.S. van Kim, Jr.`` is ``(("s", "s"), "van kim, jr")``.
|
|
263
|
+
"""
|
|
264
|
+
given: Tuple[str, ...]
|
|
265
|
+
rest: str
|
|
266
|
+
|
|
267
|
+
|
|
268
|
+
def _pieces(given: str) -> Tuple[str, ...]:
|
|
269
|
+
"""The normalised pieces of a given name, run-together initials spaced."""
|
|
270
|
+
return tuple(piece for piece in (normalize_name(part) for part in
|
|
271
|
+
_spaced_initials(given).split()) if piece)
|
|
272
|
+
|
|
273
|
+
|
|
274
|
+
def _written_key(contributor: Contributor) -> NameKey:
|
|
275
|
+
"""The key of a name from a work. A brace-protected name is all rest."""
|
|
276
|
+
if contributor.literal:
|
|
277
|
+
return NameKey((), normalize_name(contributor.literal))
|
|
278
|
+
rest = normalize_name(" ".join(part for part in (contributor.von,
|
|
279
|
+
contributor.family) if part))
|
|
280
|
+
if contributor.suffix:
|
|
281
|
+
rest = f"{rest}, {normalize_name(contributor.suffix)}"
|
|
282
|
+
return NameKey(_pieces(contributor.given or ""), rest)
|
|
283
|
+
|
|
284
|
+
|
|
285
|
+
def _declared_keys(name: str, key: str) -> List[NameKey]:
|
|
286
|
+
"""A declared name or alias divided at each of its words in turn.
|
|
287
|
+
|
|
288
|
+
A declared string is not parsed into name parts: the division that counts
|
|
289
|
+
is the one whose rest is the compared name's own. Its words are
|
|
290
|
+
normalised one at a time and lined up with ``key``, the whole name
|
|
291
|
+
normalised; where they do not line up, each word of ``key`` is one piece.
|
|
292
|
+
"""
|
|
293
|
+
words = [(normalize_name(word), _pieces(word)) for word in name.split()]
|
|
294
|
+
words = [(plain, pieces) for plain, pieces in words if plain]
|
|
295
|
+
if " ".join(plain for plain, _ in words) != key:
|
|
296
|
+
words = [(word, (word,)) for word in key.split()]
|
|
297
|
+
return [NameKey(tuple(piece for _, pieces in words[:at] for piece in pieces),
|
|
298
|
+
" ".join(plain for plain, _ in words[at:]))
|
|
299
|
+
for at in range(len(words) + 1)]
|
|
300
|
+
|
|
301
|
+
|
|
302
|
+
class Candidates:
|
|
303
|
+
"""The names a matcher compares against: ``(id, [name, *aliases])`` each.
|
|
304
|
+
|
|
305
|
+
Every form is read through `normalize_name()`, so case, accents and
|
|
306
|
+
periods do not matter.
|
|
307
|
+
"""
|
|
308
|
+
|
|
309
|
+
def __init__(self, entries: Sequence[Tuple[str, Sequence[str]]]):
|
|
310
|
+
self.exact: Dict[str, Set[str]] = {}
|
|
311
|
+
# (id, declared given name) for each declared key, by its rest.
|
|
312
|
+
self._by_rest: Dict[str, List[Tuple[str, Tuple[str, ...]]]] = {}
|
|
313
|
+
for entity_id, names in entries:
|
|
314
|
+
for name in names:
|
|
315
|
+
key = normalize_name(name)
|
|
316
|
+
if not key:
|
|
317
|
+
continue
|
|
318
|
+
self.exact.setdefault(key, set()).add(entity_id)
|
|
319
|
+
for declared in _declared_keys(name, key):
|
|
320
|
+
self._by_rest.setdefault(declared.rest, []).append(
|
|
321
|
+
(entity_id, declared.given))
|
|
322
|
+
|
|
323
|
+
def named(self, key: NameKey) -> Set[str]:
|
|
324
|
+
"""Every entity with a form whose key is ``key``."""
|
|
325
|
+
return {entity_id for entity_id, given in self._by_rest.get(key.rest, ())
|
|
326
|
+
if given == key.given}
|
|
327
|
+
|
|
328
|
+
def compatible(self, key: NameKey) -> Set[str]:
|
|
329
|
+
"""Every entity one of whose forms this name could be.
|
|
330
|
+
|
|
331
|
+
Same family, particles and suffix, and the given names agreeing part
|
|
332
|
+
by part over the shorter of the two -- an initial agreeing with any
|
|
333
|
+
name it abbreviates. `A. Kim` could be `Alex Kim` or `Alan Kim`;
|
|
334
|
+
`Alan Kim` could not be `Alex Kim`. Used to find everyone a name
|
|
335
|
+
could be, never on its own to link one.
|
|
336
|
+
"""
|
|
337
|
+
return {entity_id for entity_id, given in self._by_rest.get(key.rest, ())
|
|
338
|
+
if key.given and given and all(
|
|
339
|
+
_tokens_agree(w, d) for w, d in zip(key.given, given))}
|
|
340
|
+
|
|
341
|
+
|
|
342
|
+
class Match:
|
|
343
|
+
"""What matching one name found.
|
|
344
|
+
|
|
345
|
+
``status`` is `RESOLVED` with the one id in ``ids``, `AMBIGUOUS` with
|
|
346
|
+
every id the name fits, or `UNRESOLVED` with ``ids`` holding the
|
|
347
|
+
suggestions a human might check, which may be empty.
|
|
348
|
+
"""
|
|
349
|
+
|
|
350
|
+
def __init__(self, status: str, ids: Set[str]):
|
|
351
|
+
self.status = status
|
|
352
|
+
self.ids = sorted(ids)
|
|
353
|
+
|
|
354
|
+
@property
|
|
355
|
+
def id(self) -> Optional[str]:
|
|
356
|
+
return self.ids[0] if self.status == RESOLVED else None
|
|
357
|
+
|
|
358
|
+
|
|
359
|
+
def match(contributor: Contributor, candidates: Candidates) -> Match:
|
|
360
|
+
"""Match one name, on its full form first.
|
|
361
|
+
|
|
362
|
+
The order of the rules is SPEC.md "How a name is matched".
|
|
363
|
+
"""
|
|
364
|
+
written = _written_key(contributor)
|
|
365
|
+
abbreviated = has_initial(contributor.given)
|
|
366
|
+
exact = candidates.named(written)
|
|
367
|
+
if len(exact) == 1 and not abbreviated:
|
|
368
|
+
return Match(RESOLVED, exact)
|
|
369
|
+
if len(exact) > 1:
|
|
370
|
+
return Match(AMBIGUOUS, exact)
|
|
371
|
+
|
|
372
|
+
compatible = (set() if contributor.literal or not contributor.family
|
|
373
|
+
else candidates.compatible(written))
|
|
374
|
+
if contributor.literal or not abbreviated:
|
|
375
|
+
return Match(UNRESOLVED, compatible)
|
|
376
|
+
|
|
377
|
+
declared = exact | candidates.named(written._replace(given=tuple(
|
|
378
|
+
normalize_name(_initials(part)) for part in _given_parts(contributor.given))))
|
|
379
|
+
fits = compatible | declared
|
|
380
|
+
if len(fits) > 1:
|
|
381
|
+
return Match(AMBIGUOUS, fits)
|
|
382
|
+
if declared:
|
|
383
|
+
return Match(RESOLVED, declared)
|
|
384
|
+
return Match(UNRESOLVED, fits)
|
|
385
|
+
|
|
386
|
+
|
|
387
|
+
def person_candidates(people: Sequence[Person]) -> Candidates:
|
|
388
|
+
return Candidates([(p.id, [p.name] + list(p.aliases)) for p in people])
|
|
389
|
+
|
|
390
|
+
|
|
391
|
+
def _resolve(contributors: Sequence[Contributor], candidates: Candidates,
|
|
392
|
+
fuzzy_threshold: float,
|
|
393
|
+
where: Tuple[str, str, str],
|
|
394
|
+
report: Optional[List[Diagnostic]]) -> List[str]:
|
|
395
|
+
"""Resolve one list of contributors in place; return the names left over.
|
|
396
|
+
|
|
397
|
+
``where`` is the ``(file, key, field)`` a diagnostic is located at.
|
|
398
|
+
"""
|
|
399
|
+
unresolved: List[str] = []
|
|
400
|
+
for contributor in contributors:
|
|
401
|
+
found = match(contributor, candidates)
|
|
402
|
+
if found.status == RESOLVED:
|
|
403
|
+
contributor.person_id = found.id
|
|
404
|
+
contributor.resolution_status = RESOLVED
|
|
405
|
+
contributor.resolution_method = BY_NAME
|
|
406
|
+
continue
|
|
407
|
+
|
|
408
|
+
contributor.resolution_status = found.status
|
|
409
|
+
unresolved.append(contributor.name)
|
|
410
|
+
if report is None:
|
|
411
|
+
continue
|
|
412
|
+
if found.status == AMBIGUOUS:
|
|
413
|
+
report.append(diagnostic(
|
|
414
|
+
AMBIGUOUS_NAME, *where,
|
|
415
|
+
f"position {contributor.position}, '{contributor.name}', fits "
|
|
416
|
+
f"more than one person and is left unresolved: "
|
|
417
|
+
f"{', '.join(found.ids)}"))
|
|
418
|
+
continue
|
|
419
|
+
suggested = set(found.ids) | set(
|
|
420
|
+
fuzzy_matches(full_form(contributor), candidates, fuzzy_threshold))
|
|
421
|
+
if suggested:
|
|
422
|
+
report.append(diagnostic(
|
|
423
|
+
SUGGESTION, *where,
|
|
424
|
+
f"position {contributor.position}, '{contributor.name}', "
|
|
425
|
+
f"matched no person but may be {', '.join(sorted(suggested))}; "
|
|
426
|
+
"not linked, declare an alias if it is"))
|
|
427
|
+
return unresolved
|
|
428
|
+
|
|
429
|
+
|
|
430
|
+
def resolve_authors(
|
|
431
|
+
works: List[Work],
|
|
432
|
+
people: List[Person],
|
|
433
|
+
fuzzy_threshold: float = FUZZY_THRESHOLD,
|
|
434
|
+
diagnostics: Optional[List[Diagnostic]] = None,
|
|
435
|
+
bib_dir: str = ".",
|
|
436
|
+
) -> List[str]:
|
|
437
|
+
"""Resolve contributor names in works to person IDs, by `match()`.
|
|
438
|
+
|
|
439
|
+
Ambiguous names and suggestions are reported into ``diagnostics`` when a
|
|
440
|
+
list is given. Editors are resolved too, but are not authorships, so an
|
|
441
|
+
editor that matches nobody is not in the returned list.
|
|
442
|
+
|
|
443
|
+
Mutates ``person_id`` and ``resolution`` in place.
|
|
444
|
+
|
|
445
|
+
Returns:
|
|
446
|
+
The readable names of authorships that matched no person, sorted.
|
|
447
|
+
"""
|
|
448
|
+
if not people:
|
|
449
|
+
return []
|
|
450
|
+
|
|
451
|
+
candidates = person_candidates(people)
|
|
452
|
+
unresolved: Set[str] = set()
|
|
453
|
+
|
|
454
|
+
for work in works:
|
|
455
|
+
file = f"{bib_dir}/{work.source_file}"
|
|
456
|
+
unresolved |= set(_resolve(work.authors, candidates, fuzzy_threshold,
|
|
457
|
+
(file, work.bib_id, "author"), diagnostics))
|
|
458
|
+
_resolve(work.editors, candidates, fuzzy_threshold,
|
|
459
|
+
(file, work.bib_id, "editor"), diagnostics)
|
|
460
|
+
|
|
461
|
+
return sorted(unresolved)
|
|
462
|
+
|
|
463
|
+
|
|
464
|
+
def resolve_projects(
|
|
465
|
+
works: List[Work],
|
|
466
|
+
projects: List[Project],
|
|
467
|
+
diagnostics: Optional[List[Diagnostic]] = None,
|
|
468
|
+
bib_dir: str = ".",
|
|
469
|
+
) -> List[str]:
|
|
470
|
+
"""Return the unknown project IDs in works, sorted, and report each into
|
|
471
|
+
``diagnostics`` when a list is given. They stay on the work (SPEC.md §5).
|
|
472
|
+
"""
|
|
473
|
+
known_ids = {p.id for p in projects}
|
|
474
|
+
unknown: Set[str] = set()
|
|
475
|
+
|
|
476
|
+
for work in works:
|
|
477
|
+
for pid in work.project_ids:
|
|
478
|
+
if pid not in known_ids:
|
|
479
|
+
unknown.add(pid)
|
|
480
|
+
if diagnostics is not None:
|
|
481
|
+
diagnostics.append(diagnostic(
|
|
482
|
+
PROJECT_UNKNOWN, f"{bib_dir}/{work.source_file}",
|
|
483
|
+
work.bib_id, "project",
|
|
484
|
+
f"'{pid}' is not a project id in the projects file"))
|
|
485
|
+
|
|
486
|
+
return sorted(unknown)
|
|
487
|
+
|
|
488
|
+
|
|
489
|
+
def compute_backlinks(data: LabData) -> None:
|
|
490
|
+
"""Populate `Person.work_ids`, `Project.work_ids` and
|
|
491
|
+
`Project.people_ids` in place (SPEC.md §3). Editors are not authorships
|
|
492
|
+
and contribute to none of them."""
|
|
493
|
+
people_by_id = {p.id: p for p in data.people}
|
|
494
|
+
projects_by_id = {p.id: p for p in data.projects}
|
|
495
|
+
|
|
496
|
+
for work in data.works:
|
|
497
|
+
for author in work.authors:
|
|
498
|
+
if author.person_id and author.person_id in people_by_id:
|
|
499
|
+
person = people_by_id[author.person_id]
|
|
500
|
+
if work.bib_id not in person.work_ids:
|
|
501
|
+
person.work_ids.append(work.bib_id)
|
|
502
|
+
|
|
503
|
+
for pid in work.project_ids:
|
|
504
|
+
if pid in projects_by_id:
|
|
505
|
+
project = projects_by_id[pid]
|
|
506
|
+
if work.bib_id not in project.work_ids:
|
|
507
|
+
project.work_ids.append(work.bib_id)
|
|
508
|
+
|
|
509
|
+
for project in data.projects:
|
|
510
|
+
people_set: Set[str] = set()
|
|
511
|
+
for work_id in project.work_ids:
|
|
512
|
+
linked = next((w for w in data.works if w.bib_id == work_id), None)
|
|
513
|
+
if linked:
|
|
514
|
+
for author in linked.authors:
|
|
515
|
+
if author.person_id:
|
|
516
|
+
people_set.add(author.person_id)
|
|
517
|
+
project.people_ids = sorted(people_set)
|
|
File without changes
|