sslabdata 3.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- sslabdata/__init__.py +40 -0
- sslabdata/assembler.py +452 -0
- sslabdata/cli.py +274 -0
- sslabdata/config.py +443 -0
- sslabdata/diagnostics.py +159 -0
- sslabdata/exporters.py +86 -0
- sslabdata/loaders.py +298 -0
- sslabdata/models.py +342 -0
- sslabdata/parsers/__init__.py +0 -0
- sslabdata/parsers/bibtex.py +1132 -0
- sslabdata/parsers/latex.py +215 -0
- sslabdata/resolver.py +517 -0
- sslabdata/schema/__init__.py +0 -0
- sslabdata/schema/v5/output.schema.json +458 -0
- sslabdata-3.0.0.dist-info/METADATA +409 -0
- sslabdata-3.0.0.dist-info/RECORD +20 -0
- sslabdata-3.0.0.dist-info/WHEEL +5 -0
- sslabdata-3.0.0.dist-info/entry_points.txt +2 -0
- sslabdata-3.0.0.dist-info/licenses/LICENSE +21 -0
- sslabdata-3.0.0.dist-info/top_level.txt +1 -0
sslabdata/loaders.py
ADDED
|
@@ -0,0 +1,298 @@
|
|
|
1
|
+
"""
|
|
2
|
+
YAML data loaders for people and projects.
|
|
3
|
+
|
|
4
|
+
Copyright (c) 2024 Personal Robotics Laboratory, University of Washington
|
|
5
|
+
Author: Siddhartha Srinivasa
|
|
6
|
+
MIT License - see LICENSE file for details.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
import yaml
|
|
10
|
+
from dataclasses import dataclass, field
|
|
11
|
+
from typing import Dict, List
|
|
12
|
+
from pathlib import Path
|
|
13
|
+
|
|
14
|
+
from .config import (
|
|
15
|
+
CONTROL_CHARACTER, RepeatedKey, _kind, control_message, dotted,
|
|
16
|
+
read_yaml, repeated_message,
|
|
17
|
+
)
|
|
18
|
+
from .diagnostics import Diagnostic, diagnostic
|
|
19
|
+
from .models import Person, Project
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
# One code per condition and file; the RECORD-* codes cover all three files
|
|
23
|
+
# (SPEC.md "Diagnostic codes").
|
|
24
|
+
PEOPLE_YAML_INVALID = "PEOPLE-YAML-INVALID"
|
|
25
|
+
PEOPLE_NOT_A_LIST = "PEOPLE-NOT-A-LIST"
|
|
26
|
+
PEOPLE_FIELD_MISSING = "PEOPLE-FIELD-MISSING"
|
|
27
|
+
PROJECTS_YAML_INVALID = "PROJECTS-YAML-INVALID"
|
|
28
|
+
PROJECTS_NOT_A_LIST = "PROJECTS-NOT-A-LIST"
|
|
29
|
+
PROJECTS_FIELD_MISSING = "PROJECTS-FIELD-MISSING"
|
|
30
|
+
COLLABORATORS_YAML_INVALID = "COLLABORATORS-YAML-INVALID"
|
|
31
|
+
COLLABORATORS_NOT_A_LIST = "COLLABORATORS-NOT-A-LIST"
|
|
32
|
+
COLLABORATORS_FIELD_MISSING = "COLLABORATORS-FIELD-MISSING"
|
|
33
|
+
PEOPLE_ID_DUPLICATE = "PEOPLE-ID-DUPLICATE"
|
|
34
|
+
PEOPLE_ROLE_INVALID = "PEOPLE-ROLE-INVALID"
|
|
35
|
+
PEOPLE_STATUS_INVALID = "PEOPLE-STATUS-INVALID"
|
|
36
|
+
PROJECTS_ID_DUPLICATE = "PROJECTS-ID-DUPLICATE"
|
|
37
|
+
PROJECTS_STATUS_INVALID = "PROJECTS-STATUS-INVALID"
|
|
38
|
+
RECORD_KEY_UNKNOWN = "RECORD-KEY-UNKNOWN"
|
|
39
|
+
RECORD_KEY_REPEATED = "RECORD-KEY-REPEATED"
|
|
40
|
+
RECORD_TYPE_INVALID = "RECORD-TYPE-INVALID"
|
|
41
|
+
|
|
42
|
+
# The keys each file's records are read for. Any other key is reported and
|
|
43
|
+
# ignored, so a misspelt `webiste` is not silently dropped from the document.
|
|
44
|
+
PERSON_KEYS = ("id", "name", "aliases", "role", "status", "photo", "website",
|
|
45
|
+
"email", "co_advisor", "start_year", "end_year", "degree",
|
|
46
|
+
"thesis_title", "current_position")
|
|
47
|
+
PROJECT_KEYS = ("id", "title", "description", "website", "image", "status")
|
|
48
|
+
COLLABORATOR_KEYS = ("name", "aliases")
|
|
49
|
+
|
|
50
|
+
# The type each optional field accepts; `role` and `status` have codes of
|
|
51
|
+
# their own. A value of another type is read as empty, so it never reaches the
|
|
52
|
+
# document as the wrong type.
|
|
53
|
+
STRING, INTEGER, ALIASES = ("a string", "an integer",
|
|
54
|
+
"a list of non-empty strings")
|
|
55
|
+
PERSON_TYPES = {**dict.fromkeys(("photo", "website", "email", "co_advisor",
|
|
56
|
+
"degree", "thesis_title",
|
|
57
|
+
"current_position"), STRING),
|
|
58
|
+
"start_year": INTEGER, "end_year": INTEGER,
|
|
59
|
+
"aliases": ALIASES}
|
|
60
|
+
PROJECT_TYPES = dict.fromkeys(("description", "website", "image"), STRING)
|
|
61
|
+
COLLABORATOR_TYPES = {"aliases": ALIASES}
|
|
62
|
+
|
|
63
|
+
# There is no list of roles: any non-empty string is one, so that any lab's
|
|
64
|
+
# roles fit (SPEC.md §5).
|
|
65
|
+
PERSON_STATUSES = ("current", "alumni")
|
|
66
|
+
PROJECT_STATUSES = ("active", "completed")
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def _has_type(value, expected: str) -> bool:
|
|
70
|
+
if expected == STRING:
|
|
71
|
+
return isinstance(value, str)
|
|
72
|
+
if expected == INTEGER:
|
|
73
|
+
return isinstance(value, int) and not isinstance(value, bool)
|
|
74
|
+
return isinstance(value, list) and all(
|
|
75
|
+
isinstance(v, str) and v.strip() for v in value)
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def _records(path: str, codes, required, known, optional,
|
|
79
|
+
diagnostics) -> List[dict]:
|
|
80
|
+
"""The records of one people, projects or collaborators file that can be
|
|
81
|
+
emitted, every file checked the same way.
|
|
82
|
+
|
|
83
|
+
``codes`` are the file's YAML-invalid, not-a-list and field-missing
|
|
84
|
+
codes, ``required`` the fields a record cannot be emitted without --
|
|
85
|
+
each a non-empty string, the first naming the record -- ``known`` the
|
|
86
|
+
keys the file's records are read for and ``optional`` the type each
|
|
87
|
+
optional field accepts. A missing file reads as no records: the assembler
|
|
88
|
+
reports it, naming the configuration key.
|
|
89
|
+
"""
|
|
90
|
+
yaml_invalid, not_a_list, field_missing = codes
|
|
91
|
+
fail = diagnostics.append
|
|
92
|
+
if not Path(path).exists():
|
|
93
|
+
return []
|
|
94
|
+
|
|
95
|
+
try:
|
|
96
|
+
with open(path, 'r', encoding='utf-8') as f:
|
|
97
|
+
data, repeated, controls = read_yaml(f)
|
|
98
|
+
except (yaml.YAMLError, UnicodeDecodeError) as error:
|
|
99
|
+
fail(diagnostic(yaml_invalid, path, None, None,
|
|
100
|
+
" ".join(str(error).split())))
|
|
101
|
+
return []
|
|
102
|
+
|
|
103
|
+
# Each located as a repeat is: at the record, named by its first required
|
|
104
|
+
# field, and the path inside it; or, outside any record, at the path.
|
|
105
|
+
for found in controls:
|
|
106
|
+
index = found.path[0] if found.path else None
|
|
107
|
+
record = (data[index] if isinstance(data, list)
|
|
108
|
+
and isinstance(index, int) else None)
|
|
109
|
+
key = record.get(required[0]) if isinstance(record, dict) else None
|
|
110
|
+
fail(diagnostic(CONTROL_CHARACTER, path,
|
|
111
|
+
key if isinstance(key, str) else None,
|
|
112
|
+
dotted(found.path[1:] if isinstance(record, dict)
|
|
113
|
+
else found.path) or None,
|
|
114
|
+
control_message(found.found)))
|
|
115
|
+
|
|
116
|
+
# Each repeat inside a record is reported with that record, below; one
|
|
117
|
+
# anywhere else is reported here, at the file and the path to the key.
|
|
118
|
+
in_record: Dict[int, List[RepeatedKey]] = {}
|
|
119
|
+
for repeat in repeated:
|
|
120
|
+
index = repeat.path[0] if repeat.path else None
|
|
121
|
+
if (isinstance(data, list) and isinstance(index, int)
|
|
122
|
+
and isinstance(data[index], dict)):
|
|
123
|
+
in_record.setdefault(index, []).append(repeat)
|
|
124
|
+
else:
|
|
125
|
+
fail(diagnostic(RECORD_KEY_REPEATED, path, None,
|
|
126
|
+
dotted((*repeat.path, repeat.key)),
|
|
127
|
+
repeated_message(repeat)))
|
|
128
|
+
|
|
129
|
+
if data is None:
|
|
130
|
+
return []
|
|
131
|
+
if not isinstance(data, list):
|
|
132
|
+
fail(diagnostic(not_a_list, path, None, None,
|
|
133
|
+
"the file must be a list of records, one per entry; "
|
|
134
|
+
f"it is a {type(data).__name__}"))
|
|
135
|
+
return []
|
|
136
|
+
|
|
137
|
+
records = []
|
|
138
|
+
for number, entry in enumerate(data, start=1):
|
|
139
|
+
if not isinstance(entry, dict):
|
|
140
|
+
fail(diagnostic(not_a_list, path, None, None,
|
|
141
|
+
f"entry {number} is a {type(entry).__name__}, "
|
|
142
|
+
"not a record"))
|
|
143
|
+
continue
|
|
144
|
+
if number - 1 in in_record:
|
|
145
|
+
# The record is named by its first required field, unless that
|
|
146
|
+
# is the key given twice.
|
|
147
|
+
repeats = in_record[number - 1]
|
|
148
|
+
key = entry.get(required[0])
|
|
149
|
+
if (not isinstance(key, str) or
|
|
150
|
+
any(r.path == (number - 1,) and r.key == required[0]
|
|
151
|
+
for r in repeats)):
|
|
152
|
+
key = None
|
|
153
|
+
for repeat in repeats:
|
|
154
|
+
fail(diagnostic(RECORD_KEY_REPEATED, path, key,
|
|
155
|
+
dotted((*repeat.path[1:], repeat.key)),
|
|
156
|
+
repeated_message(repeat)))
|
|
157
|
+
continue
|
|
158
|
+
missing = [name for name in required
|
|
159
|
+
if not isinstance(entry.get(name), str)
|
|
160
|
+
or not entry[name].strip()]
|
|
161
|
+
if missing:
|
|
162
|
+
name, value = missing[0], entry.get(missing[0])
|
|
163
|
+
key = entry.get('id')
|
|
164
|
+
fail(diagnostic(
|
|
165
|
+
field_missing, path, key if isinstance(key, str) else None,
|
|
166
|
+
name, f"entry {number} has no {name}"
|
|
167
|
+
if value is None or isinstance(value, str) else
|
|
168
|
+
f"entry {number}'s {name} is {_kind(value)}; it must be a "
|
|
169
|
+
"string"))
|
|
170
|
+
continue
|
|
171
|
+
# A YAML key need not be a string (`0:`, `true:`); it is named as
|
|
172
|
+
# text so that every unknown key has a location, even a falsy one.
|
|
173
|
+
for key in entry:
|
|
174
|
+
if key not in known:
|
|
175
|
+
diagnostics.append(diagnostic(
|
|
176
|
+
RECORD_KEY_UNKNOWN, path, entry[required[0]], str(key),
|
|
177
|
+
f"'{key}' is not a key sslabdata reads, and is ignored"))
|
|
178
|
+
for name, expected in optional.items():
|
|
179
|
+
if entry.get(name) is not None and not _has_type(entry[name],
|
|
180
|
+
expected):
|
|
181
|
+
fail(diagnostic(
|
|
182
|
+
RECORD_TYPE_INVALID, path, entry[required[0]], name,
|
|
183
|
+
f"{name} is {_kind(entry[name])}; it must be {expected}, "
|
|
184
|
+
"and is read as empty"))
|
|
185
|
+
entry[name] = None
|
|
186
|
+
records.append(entry)
|
|
187
|
+
return records
|
|
188
|
+
|
|
189
|
+
|
|
190
|
+
def _repeated_ids(records: List[dict], path: str, code: str, report) -> None:
|
|
191
|
+
"""Report each id declared again, at the record that repeats it; both
|
|
192
|
+
are kept, as the parser library keeps a repeated citation key."""
|
|
193
|
+
seen = set()
|
|
194
|
+
for entry in records:
|
|
195
|
+
key = entry['id']
|
|
196
|
+
if key in seen:
|
|
197
|
+
report(diagnostic(code, path, key, 'id',
|
|
198
|
+
f"the id '{key}' is declared more than once"))
|
|
199
|
+
seen.add(key)
|
|
200
|
+
|
|
201
|
+
|
|
202
|
+
def load_people(path: str, diagnostics: List[Diagnostic]) -> List[Person]:
|
|
203
|
+
"""Load people from a YAML file (format: README.md); ``diagnostics``
|
|
204
|
+
receives what is wrong with it."""
|
|
205
|
+
warn = diagnostics.append
|
|
206
|
+
people = []
|
|
207
|
+
records = _records(path, (PEOPLE_YAML_INVALID, PEOPLE_NOT_A_LIST,
|
|
208
|
+
PEOPLE_FIELD_MISSING), ('id', 'name'), PERSON_KEYS,
|
|
209
|
+
PERSON_TYPES, diagnostics)
|
|
210
|
+
_repeated_ids(records, path, PEOPLE_ID_DUPLICATE, diagnostics.append)
|
|
211
|
+
for entry in records:
|
|
212
|
+
role = entry.get('role')
|
|
213
|
+
if not isinstance(role, str) or not role.strip():
|
|
214
|
+
warn(diagnostic(PEOPLE_ROLE_INVALID, path, entry['id'], 'role',
|
|
215
|
+
"role is missing, empty or not a string; any "
|
|
216
|
+
"non-empty string is accepted"))
|
|
217
|
+
status = entry.get('status', 'current')
|
|
218
|
+
if status not in PERSON_STATUSES:
|
|
219
|
+
warn(diagnostic(PEOPLE_STATUS_INVALID, path, entry['id'], 'status',
|
|
220
|
+
f"'{status}' is not one of "
|
|
221
|
+
f"{', '.join(PERSON_STATUSES)}"))
|
|
222
|
+
# A status that is not a string reads as the absent default: the
|
|
223
|
+
# schema requires a string, and the warning has said why.
|
|
224
|
+
if not isinstance(status, str):
|
|
225
|
+
status = 'current'
|
|
226
|
+
person = Person(
|
|
227
|
+
id=entry['id'],
|
|
228
|
+
name=entry['name'],
|
|
229
|
+
aliases=entry.get('aliases') or [],
|
|
230
|
+
role=role if isinstance(role, str) else None,
|
|
231
|
+
status=status,
|
|
232
|
+
photo=entry.get('photo'),
|
|
233
|
+
website=entry.get('website'),
|
|
234
|
+
email=entry.get('email'),
|
|
235
|
+
co_advisor=entry.get('co_advisor'),
|
|
236
|
+
start_year=entry.get('start_year'),
|
|
237
|
+
end_year=entry.get('end_year'),
|
|
238
|
+
degree=entry.get('degree'),
|
|
239
|
+
thesis_title=entry.get('thesis_title'),
|
|
240
|
+
current_position=entry.get('current_position'),
|
|
241
|
+
)
|
|
242
|
+
people.append(person)
|
|
243
|
+
|
|
244
|
+
return people
|
|
245
|
+
|
|
246
|
+
|
|
247
|
+
def load_projects(path: str, diagnostics: List[Diagnostic]) -> List[Project]:
|
|
248
|
+
"""Load projects from a YAML file, checked as `load_people()` checks
|
|
249
|
+
its own."""
|
|
250
|
+
warn = diagnostics.append
|
|
251
|
+
records = _records(path, (PROJECTS_YAML_INVALID, PROJECTS_NOT_A_LIST,
|
|
252
|
+
PROJECTS_FIELD_MISSING), ('id', 'title'), PROJECT_KEYS,
|
|
253
|
+
PROJECT_TYPES, diagnostics)
|
|
254
|
+
_repeated_ids(records, path, PROJECTS_ID_DUPLICATE, diagnostics.append)
|
|
255
|
+
projects = []
|
|
256
|
+
for entry in records:
|
|
257
|
+
status = entry.get('status', 'active')
|
|
258
|
+
if status not in PROJECT_STATUSES:
|
|
259
|
+
warn(diagnostic(PROJECTS_STATUS_INVALID, path, entry['id'],
|
|
260
|
+
'status', f"'{status}' is not one of "
|
|
261
|
+
f"{', '.join(PROJECT_STATUSES)}"))
|
|
262
|
+
if not isinstance(status, str):
|
|
263
|
+
status = 'active'
|
|
264
|
+
project = Project(
|
|
265
|
+
id=entry['id'],
|
|
266
|
+
title=entry['title'],
|
|
267
|
+
description=entry.get('description'),
|
|
268
|
+
website=entry.get('website'),
|
|
269
|
+
image=entry.get('image'),
|
|
270
|
+
status=status,
|
|
271
|
+
)
|
|
272
|
+
projects.append(project)
|
|
273
|
+
|
|
274
|
+
return projects
|
|
275
|
+
|
|
276
|
+
|
|
277
|
+
@dataclass
|
|
278
|
+
class DeclaredCollaborator:
|
|
279
|
+
"""One external co-author declared in `collaborators_file`.
|
|
280
|
+
|
|
281
|
+
It only groups authorships into `collaborators`; it is never a person
|
|
282
|
+
and never produces a `person_id`.
|
|
283
|
+
"""
|
|
284
|
+
name: str
|
|
285
|
+
aliases: List[str] = field(default_factory=list)
|
|
286
|
+
|
|
287
|
+
|
|
288
|
+
def load_collaborators(path: str, diagnostics: List[Diagnostic]
|
|
289
|
+
) -> List[DeclaredCollaborator]:
|
|
290
|
+
"""Load declared external co-authors from a YAML file, checked as
|
|
291
|
+
`load_people()` checks its own."""
|
|
292
|
+
records = _records(path, (COLLABORATORS_YAML_INVALID,
|
|
293
|
+
COLLABORATORS_NOT_A_LIST,
|
|
294
|
+
COLLABORATORS_FIELD_MISSING), ('name',),
|
|
295
|
+
COLLABORATOR_KEYS, COLLABORATOR_TYPES, diagnostics)
|
|
296
|
+
return [DeclaredCollaborator(name=entry['name'],
|
|
297
|
+
aliases=entry.get('aliases') or [])
|
|
298
|
+
for entry in records]
|
sslabdata/models.py
ADDED
|
@@ -0,0 +1,342 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Data models for sslabdata.
|
|
3
|
+
|
|
4
|
+
Defines the core entity types: Work, Author, Person, Project, Collaborator,
|
|
5
|
+
and the assembled LabData output.
|
|
6
|
+
|
|
7
|
+
Copyright (c) 2024 Personal Robotics Laboratory, University of Washington
|
|
8
|
+
Author: Siddhartha Srinivasa
|
|
9
|
+
MIT License - see LICENSE file for details.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from dataclasses import dataclass, field
|
|
13
|
+
from typing import Dict, List, Optional
|
|
14
|
+
|
|
15
|
+
# `config` owns these checks and their codes. `to_dict()` repeats them
|
|
16
|
+
# because every document passes through it, including one built in Python
|
|
17
|
+
# (SPEC.md §1).
|
|
18
|
+
from .config import json_lab, reject_absolute_name
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
# The document's schema version (schema/v5/output.schema.json). When it
|
|
22
|
+
# changes is SPEC.md §6.
|
|
23
|
+
SCHEMA_VERSION = 5
|
|
24
|
+
|
|
25
|
+
GENERATOR_NAME = "sslabdata"
|
|
26
|
+
|
|
27
|
+
# Every to_dict() emits every declared key, `null` when it does not apply;
|
|
28
|
+
# the open maps carry only keys with values (SPEC.md §4).
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
@dataclass
|
|
32
|
+
class Venue:
|
|
33
|
+
"""Where a work appeared (SPEC.md §5)."""
|
|
34
|
+
kind: str
|
|
35
|
+
name: str
|
|
36
|
+
|
|
37
|
+
def to_dict(self) -> dict:
|
|
38
|
+
return {'kind': self.kind, 'name': self.name}
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
@dataclass
|
|
42
|
+
class Link:
|
|
43
|
+
"""One URL a work can be reached at, with its origin and verification
|
|
44
|
+
status (SPEC.md §5)."""
|
|
45
|
+
url: str
|
|
46
|
+
label: Optional[str] = None
|
|
47
|
+
origin: str = "input"
|
|
48
|
+
status: str = "unchecked"
|
|
49
|
+
|
|
50
|
+
def to_dict(self) -> dict:
|
|
51
|
+
return {
|
|
52
|
+
'url': self.url,
|
|
53
|
+
'label': self.label,
|
|
54
|
+
'origin': self.origin,
|
|
55
|
+
'verification': {'status': self.status},
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
@dataclass
|
|
60
|
+
class Contributor:
|
|
61
|
+
"""One person named on a work: the parts of the name, and who it resolved
|
|
62
|
+
to (SPEC.md §5). The record ``work.editors`` carries.
|
|
63
|
+
|
|
64
|
+
``literal`` holds a name written as one brace-protected unit, where the
|
|
65
|
+
other parts do not apply.
|
|
66
|
+
"""
|
|
67
|
+
name: str
|
|
68
|
+
position: int = 0
|
|
69
|
+
person_id: Optional[str] = None
|
|
70
|
+
given: Optional[str] = None
|
|
71
|
+
von: Optional[str] = None
|
|
72
|
+
family: Optional[str] = None
|
|
73
|
+
suffix: Optional[str] = None
|
|
74
|
+
literal: Optional[str] = None
|
|
75
|
+
resolution_status: str = "unresolved"
|
|
76
|
+
resolution_method: Optional[str] = None
|
|
77
|
+
derived: Dict[str, object] = field(default_factory=dict)
|
|
78
|
+
|
|
79
|
+
def to_dict(self) -> dict:
|
|
80
|
+
"""Convert to dictionary for serialization."""
|
|
81
|
+
return {
|
|
82
|
+
'name': self.name,
|
|
83
|
+
'position': self.position,
|
|
84
|
+
'person_id': self.person_id,
|
|
85
|
+
'given': self.given,
|
|
86
|
+
'von': self.von,
|
|
87
|
+
'family': self.family,
|
|
88
|
+
'suffix': self.suffix,
|
|
89
|
+
'literal': self.literal,
|
|
90
|
+
'resolution': {'status': self.resolution_status,
|
|
91
|
+
'method': self.resolution_method},
|
|
92
|
+
'derived': dict(self.derived),
|
|
93
|
+
}
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
@dataclass
|
|
97
|
+
class Author(Contributor):
|
|
98
|
+
"""One authorship of a work, addressed by ``(work.bib_id, position)``
|
|
99
|
+
(SPEC.md §5).
|
|
100
|
+
|
|
101
|
+
Exactly one of ``person_id`` and ``collaborator_key`` is non-null.
|
|
102
|
+
"""
|
|
103
|
+
collaborator_key: Optional[str] = None
|
|
104
|
+
equal_contribution: bool = False
|
|
105
|
+
|
|
106
|
+
def to_dict(self) -> dict:
|
|
107
|
+
"""Convert to dictionary for serialization."""
|
|
108
|
+
d = Contributor.to_dict(self)
|
|
109
|
+
d['collaborator_key'] = self.collaborator_key
|
|
110
|
+
d['equal_contribution'] = self.equal_contribution
|
|
111
|
+
return d
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
@dataclass
|
|
115
|
+
class Work:
|
|
116
|
+
"""A single work with structured, renderer-agnostic data."""
|
|
117
|
+
bib_id: str
|
|
118
|
+
title: str
|
|
119
|
+
authors: List[Author]
|
|
120
|
+
year: Optional[int]
|
|
121
|
+
category: str
|
|
122
|
+
entry_type: str
|
|
123
|
+
|
|
124
|
+
# The configured `bib_files[].name`, relative to `bib_dir`: see `to_dict()`.
|
|
125
|
+
source_file: str = ""
|
|
126
|
+
|
|
127
|
+
editors: List[Contributor] = field(default_factory=list)
|
|
128
|
+
venue: Optional[Venue] = None
|
|
129
|
+
|
|
130
|
+
# The bibliographic parts, flat on the work rather than nested in the
|
|
131
|
+
# venue: they describe the work's placement, not the container.
|
|
132
|
+
volume: Optional[str] = None
|
|
133
|
+
number: Optional[str] = None
|
|
134
|
+
pages: Optional[str] = None
|
|
135
|
+
series: Optional[str] = None
|
|
136
|
+
edition: Optional[str] = None
|
|
137
|
+
publisher: Optional[str] = None
|
|
138
|
+
address: Optional[str] = None
|
|
139
|
+
organization: Optional[str] = None
|
|
140
|
+
chapter: Optional[str] = None
|
|
141
|
+
month: Optional[str] = None
|
|
142
|
+
howpublished: Optional[str] = None
|
|
143
|
+
type: Optional[str] = None
|
|
144
|
+
|
|
145
|
+
abstract: Optional[str] = None
|
|
146
|
+
note: Optional[str] = None
|
|
147
|
+
|
|
148
|
+
identifiers: Dict[str, List[str]] = field(default_factory=dict)
|
|
149
|
+
links: Dict[str, List[Link]] = field(default_factory=dict)
|
|
150
|
+
|
|
151
|
+
project_ids: List[str] = field(default_factory=list)
|
|
152
|
+
bibtex: Optional[str] = None
|
|
153
|
+
derived: Dict[str, object] = field(default_factory=dict)
|
|
154
|
+
|
|
155
|
+
def to_dict(self) -> dict:
|
|
156
|
+
"""Convert to dictionary for serialization.
|
|
157
|
+
|
|
158
|
+
``source.file`` is checked here rather than only where it was set:
|
|
159
|
+
every serializer passes through this method, so a `Work` built by hand
|
|
160
|
+
cannot carry an absolute path into the document.
|
|
161
|
+
"""
|
|
162
|
+
reject_absolute_name(self.source_file)
|
|
163
|
+
return {
|
|
164
|
+
'bib_id': self.bib_id,
|
|
165
|
+
'source': {'file': self.source_file, 'key': self.bib_id},
|
|
166
|
+
'title': self.title,
|
|
167
|
+
'authors': [a.to_dict() for a in self.authors],
|
|
168
|
+
'editors': [e.to_dict() for e in self.editors],
|
|
169
|
+
'year': self.year,
|
|
170
|
+
'venue': self.venue.to_dict() if self.venue else None,
|
|
171
|
+
'volume': self.volume,
|
|
172
|
+
'number': self.number,
|
|
173
|
+
'pages': self.pages,
|
|
174
|
+
'series': self.series,
|
|
175
|
+
'edition': self.edition,
|
|
176
|
+
'publisher': self.publisher,
|
|
177
|
+
'address': self.address,
|
|
178
|
+
'organization': self.organization,
|
|
179
|
+
'chapter': self.chapter,
|
|
180
|
+
'month': self.month,
|
|
181
|
+
'howpublished': self.howpublished,
|
|
182
|
+
'type': self.type,
|
|
183
|
+
'category': self.category,
|
|
184
|
+
'entry_type': self.entry_type,
|
|
185
|
+
'abstract': self.abstract,
|
|
186
|
+
'note': self.note,
|
|
187
|
+
'identifiers': {scheme: list(values)
|
|
188
|
+
for scheme, values in self.identifiers.items()},
|
|
189
|
+
'links': {kind: [link.to_dict() for link in links]
|
|
190
|
+
for kind, links in self.links.items()},
|
|
191
|
+
'project_ids': list(self.project_ids),
|
|
192
|
+
'bibtex': self.bibtex,
|
|
193
|
+
'derived': dict(self.derived),
|
|
194
|
+
}
|
|
195
|
+
|
|
196
|
+
|
|
197
|
+
@dataclass
|
|
198
|
+
class Person:
|
|
199
|
+
"""A lab member (current or alumni)."""
|
|
200
|
+
id: str
|
|
201
|
+
name: str
|
|
202
|
+
aliases: List[str] = field(default_factory=list)
|
|
203
|
+
role: Optional[str] = None
|
|
204
|
+
status: str = "current"
|
|
205
|
+
photo: Optional[str] = None
|
|
206
|
+
website: Optional[str] = None
|
|
207
|
+
email: Optional[str] = None
|
|
208
|
+
co_advisor: Optional[str] = None
|
|
209
|
+
start_year: Optional[int] = None
|
|
210
|
+
|
|
211
|
+
end_year: Optional[int] = None
|
|
212
|
+
degree: Optional[str] = None
|
|
213
|
+
thesis_title: Optional[str] = None
|
|
214
|
+
current_position: Optional[str] = None
|
|
215
|
+
|
|
216
|
+
# Back-linked (computed, not from YAML input)
|
|
217
|
+
work_ids: List[str] = field(default_factory=list)
|
|
218
|
+
derived: Dict[str, object] = field(default_factory=dict)
|
|
219
|
+
|
|
220
|
+
def to_dict(self) -> dict:
|
|
221
|
+
"""Convert to dictionary for serialization; ``aliases`` are read for
|
|
222
|
+
matching and are not emitted."""
|
|
223
|
+
return {
|
|
224
|
+
'id': self.id,
|
|
225
|
+
'name': self.name,
|
|
226
|
+
'role': self.role,
|
|
227
|
+
'status': self.status,
|
|
228
|
+
'photo': self.photo,
|
|
229
|
+
'email': self.email,
|
|
230
|
+
'website': self.website,
|
|
231
|
+
'co_advisor': self.co_advisor,
|
|
232
|
+
'start_year': self.start_year,
|
|
233
|
+
'end_year': self.end_year,
|
|
234
|
+
'degree': self.degree,
|
|
235
|
+
'thesis_title': self.thesis_title,
|
|
236
|
+
'current_position': self.current_position,
|
|
237
|
+
'work_ids': list(self.work_ids),
|
|
238
|
+
'derived': dict(self.derived),
|
|
239
|
+
}
|
|
240
|
+
|
|
241
|
+
|
|
242
|
+
@dataclass
|
|
243
|
+
class Collaborator:
|
|
244
|
+
"""A grouping over unresolved authorships, not an identity (SPEC.md §5)."""
|
|
245
|
+
key: str
|
|
246
|
+
name: str
|
|
247
|
+
grouped_by: str = "normalized_name"
|
|
248
|
+
name_kind: str = "personal"
|
|
249
|
+
given: Optional[str] = None
|
|
250
|
+
von: Optional[str] = None
|
|
251
|
+
family: Optional[str] = None
|
|
252
|
+
suffix: Optional[str] = None
|
|
253
|
+
literal: Optional[str] = None
|
|
254
|
+
name_variants: List[str] = field(default_factory=list)
|
|
255
|
+
authorships: List[Dict[str, object]] = field(default_factory=list)
|
|
256
|
+
work_ids: List[str] = field(default_factory=list)
|
|
257
|
+
last_year: Optional[int] = None
|
|
258
|
+
derived: Dict[str, object] = field(default_factory=dict)
|
|
259
|
+
|
|
260
|
+
def to_dict(self) -> dict:
|
|
261
|
+
return {
|
|
262
|
+
'key': self.key,
|
|
263
|
+
'grouped_by': self.grouped_by,
|
|
264
|
+
'name_kind': self.name_kind,
|
|
265
|
+
'name': self.name,
|
|
266
|
+
'given': self.given,
|
|
267
|
+
'von': self.von,
|
|
268
|
+
'family': self.family,
|
|
269
|
+
'suffix': self.suffix,
|
|
270
|
+
'literal': self.literal,
|
|
271
|
+
'name_variants': list(self.name_variants),
|
|
272
|
+
'authorships': [dict(a) for a in self.authorships],
|
|
273
|
+
'work_ids': list(self.work_ids),
|
|
274
|
+
'last_year': self.last_year,
|
|
275
|
+
'derived': dict(self.derived),
|
|
276
|
+
}
|
|
277
|
+
|
|
278
|
+
|
|
279
|
+
@dataclass
|
|
280
|
+
class Project:
|
|
281
|
+
"""A research project."""
|
|
282
|
+
id: str
|
|
283
|
+
title: str
|
|
284
|
+
description: Optional[str] = None
|
|
285
|
+
website: Optional[str] = None
|
|
286
|
+
status: str = "active"
|
|
287
|
+
|
|
288
|
+
# Back-linked (computed)
|
|
289
|
+
work_ids: List[str] = field(default_factory=list)
|
|
290
|
+
people_ids: List[str] = field(default_factory=list)
|
|
291
|
+
derived: Dict[str, object] = field(default_factory=dict)
|
|
292
|
+
|
|
293
|
+
# Declared last so it takes no other field's position in a positional
|
|
294
|
+
# call; to_dict() emits it beside `website`.
|
|
295
|
+
image: Optional[str] = None
|
|
296
|
+
|
|
297
|
+
def to_dict(self) -> dict:
|
|
298
|
+
"""Convert to dictionary for serialization."""
|
|
299
|
+
return {
|
|
300
|
+
'id': self.id,
|
|
301
|
+
'title': self.title,
|
|
302
|
+
'description': self.description,
|
|
303
|
+
'website': self.website,
|
|
304
|
+
'image': self.image,
|
|
305
|
+
'status': self.status,
|
|
306
|
+
'work_ids': list(self.work_ids),
|
|
307
|
+
'people_ids': list(self.people_ids),
|
|
308
|
+
'derived': dict(self.derived),
|
|
309
|
+
}
|
|
310
|
+
|
|
311
|
+
|
|
312
|
+
@dataclass
|
|
313
|
+
class LabData:
|
|
314
|
+
"""The fully resolved output: all entities with cross-references."""
|
|
315
|
+
works: List[Work] = field(default_factory=list)
|
|
316
|
+
people: List[Person] = field(default_factory=list)
|
|
317
|
+
projects: List[Project] = field(default_factory=list)
|
|
318
|
+
collaborators: List[Collaborator] = field(default_factory=list)
|
|
319
|
+
lab: Optional[Dict[str, object]] = None
|
|
320
|
+
|
|
321
|
+
def to_dict(self) -> dict:
|
|
322
|
+
"""Convert to dictionary for serialization.
|
|
323
|
+
|
|
324
|
+
``generator`` carries no timestamp, so the document is deterministic
|
|
325
|
+
(SPEC.md §3). The version is imported here because the package imports
|
|
326
|
+
this module while defining it.
|
|
327
|
+
"""
|
|
328
|
+
from . import __version__
|
|
329
|
+
|
|
330
|
+
return {
|
|
331
|
+
'schema_version': SCHEMA_VERSION,
|
|
332
|
+
'generator': {
|
|
333
|
+
'name': GENERATOR_NAME,
|
|
334
|
+
'version': __version__,
|
|
335
|
+
'schema_version': SCHEMA_VERSION,
|
|
336
|
+
},
|
|
337
|
+
'lab': json_lab(dict(self.lab or {})),
|
|
338
|
+
'works': [w.to_dict() for w in self.works],
|
|
339
|
+
'people': [p.to_dict() for p in self.people],
|
|
340
|
+
'projects': [p.to_dict() for p in self.projects],
|
|
341
|
+
'collaborators': [c.to_dict() for c in self.collaborators],
|
|
342
|
+
}
|
|
File without changes
|