sslabdata 3.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
sslabdata/loaders.py ADDED
@@ -0,0 +1,298 @@
1
+ """
2
+ YAML data loaders for people and projects.
3
+
4
+ Copyright (c) 2024 Personal Robotics Laboratory, University of Washington
5
+ Author: Siddhartha Srinivasa
6
+ MIT License - see LICENSE file for details.
7
+ """
8
+
9
+ import yaml
10
+ from dataclasses import dataclass, field
11
+ from typing import Dict, List
12
+ from pathlib import Path
13
+
14
+ from .config import (
15
+ CONTROL_CHARACTER, RepeatedKey, _kind, control_message, dotted,
16
+ read_yaml, repeated_message,
17
+ )
18
+ from .diagnostics import Diagnostic, diagnostic
19
+ from .models import Person, Project
20
+
21
+
22
+ # One code per condition and file; the RECORD-* codes cover all three files
23
+ # (SPEC.md "Diagnostic codes").
24
+ PEOPLE_YAML_INVALID = "PEOPLE-YAML-INVALID"
25
+ PEOPLE_NOT_A_LIST = "PEOPLE-NOT-A-LIST"
26
+ PEOPLE_FIELD_MISSING = "PEOPLE-FIELD-MISSING"
27
+ PROJECTS_YAML_INVALID = "PROJECTS-YAML-INVALID"
28
+ PROJECTS_NOT_A_LIST = "PROJECTS-NOT-A-LIST"
29
+ PROJECTS_FIELD_MISSING = "PROJECTS-FIELD-MISSING"
30
+ COLLABORATORS_YAML_INVALID = "COLLABORATORS-YAML-INVALID"
31
+ COLLABORATORS_NOT_A_LIST = "COLLABORATORS-NOT-A-LIST"
32
+ COLLABORATORS_FIELD_MISSING = "COLLABORATORS-FIELD-MISSING"
33
+ PEOPLE_ID_DUPLICATE = "PEOPLE-ID-DUPLICATE"
34
+ PEOPLE_ROLE_INVALID = "PEOPLE-ROLE-INVALID"
35
+ PEOPLE_STATUS_INVALID = "PEOPLE-STATUS-INVALID"
36
+ PROJECTS_ID_DUPLICATE = "PROJECTS-ID-DUPLICATE"
37
+ PROJECTS_STATUS_INVALID = "PROJECTS-STATUS-INVALID"
38
+ RECORD_KEY_UNKNOWN = "RECORD-KEY-UNKNOWN"
39
+ RECORD_KEY_REPEATED = "RECORD-KEY-REPEATED"
40
+ RECORD_TYPE_INVALID = "RECORD-TYPE-INVALID"
41
+
42
+ # The keys each file's records are read for. Any other key is reported and
43
+ # ignored, so a misspelt `webiste` is not silently dropped from the document.
44
+ PERSON_KEYS = ("id", "name", "aliases", "role", "status", "photo", "website",
45
+ "email", "co_advisor", "start_year", "end_year", "degree",
46
+ "thesis_title", "current_position")
47
+ PROJECT_KEYS = ("id", "title", "description", "website", "image", "status")
48
+ COLLABORATOR_KEYS = ("name", "aliases")
49
+
50
+ # The type each optional field accepts; `role` and `status` have codes of
51
+ # their own. A value of another type is read as empty, so it never reaches the
52
+ # document as the wrong type.
53
+ STRING, INTEGER, ALIASES = ("a string", "an integer",
54
+ "a list of non-empty strings")
55
+ PERSON_TYPES = {**dict.fromkeys(("photo", "website", "email", "co_advisor",
56
+ "degree", "thesis_title",
57
+ "current_position"), STRING),
58
+ "start_year": INTEGER, "end_year": INTEGER,
59
+ "aliases": ALIASES}
60
+ PROJECT_TYPES = dict.fromkeys(("description", "website", "image"), STRING)
61
+ COLLABORATOR_TYPES = {"aliases": ALIASES}
62
+
63
+ # There is no list of roles: any non-empty string is one, so that any lab's
64
+ # roles fit (SPEC.md §5).
65
+ PERSON_STATUSES = ("current", "alumni")
66
+ PROJECT_STATUSES = ("active", "completed")
67
+
68
+
69
+ def _has_type(value, expected: str) -> bool:
70
+ if expected == STRING:
71
+ return isinstance(value, str)
72
+ if expected == INTEGER:
73
+ return isinstance(value, int) and not isinstance(value, bool)
74
+ return isinstance(value, list) and all(
75
+ isinstance(v, str) and v.strip() for v in value)
76
+
77
+
78
+ def _records(path: str, codes, required, known, optional,
79
+ diagnostics) -> List[dict]:
80
+ """The records of one people, projects or collaborators file that can be
81
+ emitted, every file checked the same way.
82
+
83
+ ``codes`` are the file's YAML-invalid, not-a-list and field-missing
84
+ codes, ``required`` the fields a record cannot be emitted without --
85
+ each a non-empty string, the first naming the record -- ``known`` the
86
+ keys the file's records are read for and ``optional`` the type each
87
+ optional field accepts. A missing file reads as no records: the assembler
88
+ reports it, naming the configuration key.
89
+ """
90
+ yaml_invalid, not_a_list, field_missing = codes
91
+ fail = diagnostics.append
92
+ if not Path(path).exists():
93
+ return []
94
+
95
+ try:
96
+ with open(path, 'r', encoding='utf-8') as f:
97
+ data, repeated, controls = read_yaml(f)
98
+ except (yaml.YAMLError, UnicodeDecodeError) as error:
99
+ fail(diagnostic(yaml_invalid, path, None, None,
100
+ " ".join(str(error).split())))
101
+ return []
102
+
103
+ # Each located as a repeat is: at the record, named by its first required
104
+ # field, and the path inside it; or, outside any record, at the path.
105
+ for found in controls:
106
+ index = found.path[0] if found.path else None
107
+ record = (data[index] if isinstance(data, list)
108
+ and isinstance(index, int) else None)
109
+ key = record.get(required[0]) if isinstance(record, dict) else None
110
+ fail(diagnostic(CONTROL_CHARACTER, path,
111
+ key if isinstance(key, str) else None,
112
+ dotted(found.path[1:] if isinstance(record, dict)
113
+ else found.path) or None,
114
+ control_message(found.found)))
115
+
116
+ # Each repeat inside a record is reported with that record, below; one
117
+ # anywhere else is reported here, at the file and the path to the key.
118
+ in_record: Dict[int, List[RepeatedKey]] = {}
119
+ for repeat in repeated:
120
+ index = repeat.path[0] if repeat.path else None
121
+ if (isinstance(data, list) and isinstance(index, int)
122
+ and isinstance(data[index], dict)):
123
+ in_record.setdefault(index, []).append(repeat)
124
+ else:
125
+ fail(diagnostic(RECORD_KEY_REPEATED, path, None,
126
+ dotted((*repeat.path, repeat.key)),
127
+ repeated_message(repeat)))
128
+
129
+ if data is None:
130
+ return []
131
+ if not isinstance(data, list):
132
+ fail(diagnostic(not_a_list, path, None, None,
133
+ "the file must be a list of records, one per entry; "
134
+ f"it is a {type(data).__name__}"))
135
+ return []
136
+
137
+ records = []
138
+ for number, entry in enumerate(data, start=1):
139
+ if not isinstance(entry, dict):
140
+ fail(diagnostic(not_a_list, path, None, None,
141
+ f"entry {number} is a {type(entry).__name__}, "
142
+ "not a record"))
143
+ continue
144
+ if number - 1 in in_record:
145
+ # The record is named by its first required field, unless that
146
+ # is the key given twice.
147
+ repeats = in_record[number - 1]
148
+ key = entry.get(required[0])
149
+ if (not isinstance(key, str) or
150
+ any(r.path == (number - 1,) and r.key == required[0]
151
+ for r in repeats)):
152
+ key = None
153
+ for repeat in repeats:
154
+ fail(diagnostic(RECORD_KEY_REPEATED, path, key,
155
+ dotted((*repeat.path[1:], repeat.key)),
156
+ repeated_message(repeat)))
157
+ continue
158
+ missing = [name for name in required
159
+ if not isinstance(entry.get(name), str)
160
+ or not entry[name].strip()]
161
+ if missing:
162
+ name, value = missing[0], entry.get(missing[0])
163
+ key = entry.get('id')
164
+ fail(diagnostic(
165
+ field_missing, path, key if isinstance(key, str) else None,
166
+ name, f"entry {number} has no {name}"
167
+ if value is None or isinstance(value, str) else
168
+ f"entry {number}'s {name} is {_kind(value)}; it must be a "
169
+ "string"))
170
+ continue
171
+ # A YAML key need not be a string (`0:`, `true:`); it is named as
172
+ # text so that every unknown key has a location, even a falsy one.
173
+ for key in entry:
174
+ if key not in known:
175
+ diagnostics.append(diagnostic(
176
+ RECORD_KEY_UNKNOWN, path, entry[required[0]], str(key),
177
+ f"'{key}' is not a key sslabdata reads, and is ignored"))
178
+ for name, expected in optional.items():
179
+ if entry.get(name) is not None and not _has_type(entry[name],
180
+ expected):
181
+ fail(diagnostic(
182
+ RECORD_TYPE_INVALID, path, entry[required[0]], name,
183
+ f"{name} is {_kind(entry[name])}; it must be {expected}, "
184
+ "and is read as empty"))
185
+ entry[name] = None
186
+ records.append(entry)
187
+ return records
188
+
189
+
190
+ def _repeated_ids(records: List[dict], path: str, code: str, report) -> None:
191
+ """Report each id declared again, at the record that repeats it; both
192
+ are kept, as the parser library keeps a repeated citation key."""
193
+ seen = set()
194
+ for entry in records:
195
+ key = entry['id']
196
+ if key in seen:
197
+ report(diagnostic(code, path, key, 'id',
198
+ f"the id '{key}' is declared more than once"))
199
+ seen.add(key)
200
+
201
+
202
+ def load_people(path: str, diagnostics: List[Diagnostic]) -> List[Person]:
203
+ """Load people from a YAML file (format: README.md); ``diagnostics``
204
+ receives what is wrong with it."""
205
+ warn = diagnostics.append
206
+ people = []
207
+ records = _records(path, (PEOPLE_YAML_INVALID, PEOPLE_NOT_A_LIST,
208
+ PEOPLE_FIELD_MISSING), ('id', 'name'), PERSON_KEYS,
209
+ PERSON_TYPES, diagnostics)
210
+ _repeated_ids(records, path, PEOPLE_ID_DUPLICATE, diagnostics.append)
211
+ for entry in records:
212
+ role = entry.get('role')
213
+ if not isinstance(role, str) or not role.strip():
214
+ warn(diagnostic(PEOPLE_ROLE_INVALID, path, entry['id'], 'role',
215
+ "role is missing, empty or not a string; any "
216
+ "non-empty string is accepted"))
217
+ status = entry.get('status', 'current')
218
+ if status not in PERSON_STATUSES:
219
+ warn(diagnostic(PEOPLE_STATUS_INVALID, path, entry['id'], 'status',
220
+ f"'{status}' is not one of "
221
+ f"{', '.join(PERSON_STATUSES)}"))
222
+ # A status that is not a string reads as the absent default: the
223
+ # schema requires a string, and the warning has said why.
224
+ if not isinstance(status, str):
225
+ status = 'current'
226
+ person = Person(
227
+ id=entry['id'],
228
+ name=entry['name'],
229
+ aliases=entry.get('aliases') or [],
230
+ role=role if isinstance(role, str) else None,
231
+ status=status,
232
+ photo=entry.get('photo'),
233
+ website=entry.get('website'),
234
+ email=entry.get('email'),
235
+ co_advisor=entry.get('co_advisor'),
236
+ start_year=entry.get('start_year'),
237
+ end_year=entry.get('end_year'),
238
+ degree=entry.get('degree'),
239
+ thesis_title=entry.get('thesis_title'),
240
+ current_position=entry.get('current_position'),
241
+ )
242
+ people.append(person)
243
+
244
+ return people
245
+
246
+
247
+ def load_projects(path: str, diagnostics: List[Diagnostic]) -> List[Project]:
248
+ """Load projects from a YAML file, checked as `load_people()` checks
249
+ its own."""
250
+ warn = diagnostics.append
251
+ records = _records(path, (PROJECTS_YAML_INVALID, PROJECTS_NOT_A_LIST,
252
+ PROJECTS_FIELD_MISSING), ('id', 'title'), PROJECT_KEYS,
253
+ PROJECT_TYPES, diagnostics)
254
+ _repeated_ids(records, path, PROJECTS_ID_DUPLICATE, diagnostics.append)
255
+ projects = []
256
+ for entry in records:
257
+ status = entry.get('status', 'active')
258
+ if status not in PROJECT_STATUSES:
259
+ warn(diagnostic(PROJECTS_STATUS_INVALID, path, entry['id'],
260
+ 'status', f"'{status}' is not one of "
261
+ f"{', '.join(PROJECT_STATUSES)}"))
262
+ if not isinstance(status, str):
263
+ status = 'active'
264
+ project = Project(
265
+ id=entry['id'],
266
+ title=entry['title'],
267
+ description=entry.get('description'),
268
+ website=entry.get('website'),
269
+ image=entry.get('image'),
270
+ status=status,
271
+ )
272
+ projects.append(project)
273
+
274
+ return projects
275
+
276
+
277
+ @dataclass
278
+ class DeclaredCollaborator:
279
+ """One external co-author declared in `collaborators_file`.
280
+
281
+ It only groups authorships into `collaborators`; it is never a person
282
+ and never produces a `person_id`.
283
+ """
284
+ name: str
285
+ aliases: List[str] = field(default_factory=list)
286
+
287
+
288
+ def load_collaborators(path: str, diagnostics: List[Diagnostic]
289
+ ) -> List[DeclaredCollaborator]:
290
+ """Load declared external co-authors from a YAML file, checked as
291
+ `load_people()` checks its own."""
292
+ records = _records(path, (COLLABORATORS_YAML_INVALID,
293
+ COLLABORATORS_NOT_A_LIST,
294
+ COLLABORATORS_FIELD_MISSING), ('name',),
295
+ COLLABORATOR_KEYS, COLLABORATOR_TYPES, diagnostics)
296
+ return [DeclaredCollaborator(name=entry['name'],
297
+ aliases=entry.get('aliases') or [])
298
+ for entry in records]
sslabdata/models.py ADDED
@@ -0,0 +1,342 @@
1
+ """
2
+ Data models for sslabdata.
3
+
4
+ Defines the core entity types: Work, Author, Person, Project, Collaborator,
5
+ and the assembled LabData output.
6
+
7
+ Copyright (c) 2024 Personal Robotics Laboratory, University of Washington
8
+ Author: Siddhartha Srinivasa
9
+ MIT License - see LICENSE file for details.
10
+ """
11
+
12
+ from dataclasses import dataclass, field
13
+ from typing import Dict, List, Optional
14
+
15
+ # `config` owns these checks and their codes. `to_dict()` repeats them
16
+ # because every document passes through it, including one built in Python
17
+ # (SPEC.md §1).
18
+ from .config import json_lab, reject_absolute_name
19
+
20
+
21
+ # The document's schema version (schema/v5/output.schema.json). When it
22
+ # changes is SPEC.md §6.
23
+ SCHEMA_VERSION = 5
24
+
25
+ GENERATOR_NAME = "sslabdata"
26
+
27
+ # Every to_dict() emits every declared key, `null` when it does not apply;
28
+ # the open maps carry only keys with values (SPEC.md §4).
29
+
30
+
31
+ @dataclass
32
+ class Venue:
33
+ """Where a work appeared (SPEC.md §5)."""
34
+ kind: str
35
+ name: str
36
+
37
+ def to_dict(self) -> dict:
38
+ return {'kind': self.kind, 'name': self.name}
39
+
40
+
41
+ @dataclass
42
+ class Link:
43
+ """One URL a work can be reached at, with its origin and verification
44
+ status (SPEC.md §5)."""
45
+ url: str
46
+ label: Optional[str] = None
47
+ origin: str = "input"
48
+ status: str = "unchecked"
49
+
50
+ def to_dict(self) -> dict:
51
+ return {
52
+ 'url': self.url,
53
+ 'label': self.label,
54
+ 'origin': self.origin,
55
+ 'verification': {'status': self.status},
56
+ }
57
+
58
+
59
+ @dataclass
60
+ class Contributor:
61
+ """One person named on a work: the parts of the name, and who it resolved
62
+ to (SPEC.md §5). The record ``work.editors`` carries.
63
+
64
+ ``literal`` holds a name written as one brace-protected unit, where the
65
+ other parts do not apply.
66
+ """
67
+ name: str
68
+ position: int = 0
69
+ person_id: Optional[str] = None
70
+ given: Optional[str] = None
71
+ von: Optional[str] = None
72
+ family: Optional[str] = None
73
+ suffix: Optional[str] = None
74
+ literal: Optional[str] = None
75
+ resolution_status: str = "unresolved"
76
+ resolution_method: Optional[str] = None
77
+ derived: Dict[str, object] = field(default_factory=dict)
78
+
79
+ def to_dict(self) -> dict:
80
+ """Convert to dictionary for serialization."""
81
+ return {
82
+ 'name': self.name,
83
+ 'position': self.position,
84
+ 'person_id': self.person_id,
85
+ 'given': self.given,
86
+ 'von': self.von,
87
+ 'family': self.family,
88
+ 'suffix': self.suffix,
89
+ 'literal': self.literal,
90
+ 'resolution': {'status': self.resolution_status,
91
+ 'method': self.resolution_method},
92
+ 'derived': dict(self.derived),
93
+ }
94
+
95
+
96
+ @dataclass
97
+ class Author(Contributor):
98
+ """One authorship of a work, addressed by ``(work.bib_id, position)``
99
+ (SPEC.md §5).
100
+
101
+ Exactly one of ``person_id`` and ``collaborator_key`` is non-null.
102
+ """
103
+ collaborator_key: Optional[str] = None
104
+ equal_contribution: bool = False
105
+
106
+ def to_dict(self) -> dict:
107
+ """Convert to dictionary for serialization."""
108
+ d = Contributor.to_dict(self)
109
+ d['collaborator_key'] = self.collaborator_key
110
+ d['equal_contribution'] = self.equal_contribution
111
+ return d
112
+
113
+
114
+ @dataclass
115
+ class Work:
116
+ """A single work with structured, renderer-agnostic data."""
117
+ bib_id: str
118
+ title: str
119
+ authors: List[Author]
120
+ year: Optional[int]
121
+ category: str
122
+ entry_type: str
123
+
124
+ # The configured `bib_files[].name`, relative to `bib_dir`: see `to_dict()`.
125
+ source_file: str = ""
126
+
127
+ editors: List[Contributor] = field(default_factory=list)
128
+ venue: Optional[Venue] = None
129
+
130
+ # The bibliographic parts, flat on the work rather than nested in the
131
+ # venue: they describe the work's placement, not the container.
132
+ volume: Optional[str] = None
133
+ number: Optional[str] = None
134
+ pages: Optional[str] = None
135
+ series: Optional[str] = None
136
+ edition: Optional[str] = None
137
+ publisher: Optional[str] = None
138
+ address: Optional[str] = None
139
+ organization: Optional[str] = None
140
+ chapter: Optional[str] = None
141
+ month: Optional[str] = None
142
+ howpublished: Optional[str] = None
143
+ type: Optional[str] = None
144
+
145
+ abstract: Optional[str] = None
146
+ note: Optional[str] = None
147
+
148
+ identifiers: Dict[str, List[str]] = field(default_factory=dict)
149
+ links: Dict[str, List[Link]] = field(default_factory=dict)
150
+
151
+ project_ids: List[str] = field(default_factory=list)
152
+ bibtex: Optional[str] = None
153
+ derived: Dict[str, object] = field(default_factory=dict)
154
+
155
+ def to_dict(self) -> dict:
156
+ """Convert to dictionary for serialization.
157
+
158
+ ``source.file`` is checked here rather than only where it was set:
159
+ every serializer passes through this method, so a `Work` built by hand
160
+ cannot carry an absolute path into the document.
161
+ """
162
+ reject_absolute_name(self.source_file)
163
+ return {
164
+ 'bib_id': self.bib_id,
165
+ 'source': {'file': self.source_file, 'key': self.bib_id},
166
+ 'title': self.title,
167
+ 'authors': [a.to_dict() for a in self.authors],
168
+ 'editors': [e.to_dict() for e in self.editors],
169
+ 'year': self.year,
170
+ 'venue': self.venue.to_dict() if self.venue else None,
171
+ 'volume': self.volume,
172
+ 'number': self.number,
173
+ 'pages': self.pages,
174
+ 'series': self.series,
175
+ 'edition': self.edition,
176
+ 'publisher': self.publisher,
177
+ 'address': self.address,
178
+ 'organization': self.organization,
179
+ 'chapter': self.chapter,
180
+ 'month': self.month,
181
+ 'howpublished': self.howpublished,
182
+ 'type': self.type,
183
+ 'category': self.category,
184
+ 'entry_type': self.entry_type,
185
+ 'abstract': self.abstract,
186
+ 'note': self.note,
187
+ 'identifiers': {scheme: list(values)
188
+ for scheme, values in self.identifiers.items()},
189
+ 'links': {kind: [link.to_dict() for link in links]
190
+ for kind, links in self.links.items()},
191
+ 'project_ids': list(self.project_ids),
192
+ 'bibtex': self.bibtex,
193
+ 'derived': dict(self.derived),
194
+ }
195
+
196
+
197
+ @dataclass
198
+ class Person:
199
+ """A lab member (current or alumni)."""
200
+ id: str
201
+ name: str
202
+ aliases: List[str] = field(default_factory=list)
203
+ role: Optional[str] = None
204
+ status: str = "current"
205
+ photo: Optional[str] = None
206
+ website: Optional[str] = None
207
+ email: Optional[str] = None
208
+ co_advisor: Optional[str] = None
209
+ start_year: Optional[int] = None
210
+
211
+ end_year: Optional[int] = None
212
+ degree: Optional[str] = None
213
+ thesis_title: Optional[str] = None
214
+ current_position: Optional[str] = None
215
+
216
+ # Back-linked (computed, not from YAML input)
217
+ work_ids: List[str] = field(default_factory=list)
218
+ derived: Dict[str, object] = field(default_factory=dict)
219
+
220
+ def to_dict(self) -> dict:
221
+ """Convert to dictionary for serialization; ``aliases`` are read for
222
+ matching and are not emitted."""
223
+ return {
224
+ 'id': self.id,
225
+ 'name': self.name,
226
+ 'role': self.role,
227
+ 'status': self.status,
228
+ 'photo': self.photo,
229
+ 'email': self.email,
230
+ 'website': self.website,
231
+ 'co_advisor': self.co_advisor,
232
+ 'start_year': self.start_year,
233
+ 'end_year': self.end_year,
234
+ 'degree': self.degree,
235
+ 'thesis_title': self.thesis_title,
236
+ 'current_position': self.current_position,
237
+ 'work_ids': list(self.work_ids),
238
+ 'derived': dict(self.derived),
239
+ }
240
+
241
+
242
+ @dataclass
243
+ class Collaborator:
244
+ """A grouping over unresolved authorships, not an identity (SPEC.md §5)."""
245
+ key: str
246
+ name: str
247
+ grouped_by: str = "normalized_name"
248
+ name_kind: str = "personal"
249
+ given: Optional[str] = None
250
+ von: Optional[str] = None
251
+ family: Optional[str] = None
252
+ suffix: Optional[str] = None
253
+ literal: Optional[str] = None
254
+ name_variants: List[str] = field(default_factory=list)
255
+ authorships: List[Dict[str, object]] = field(default_factory=list)
256
+ work_ids: List[str] = field(default_factory=list)
257
+ last_year: Optional[int] = None
258
+ derived: Dict[str, object] = field(default_factory=dict)
259
+
260
+ def to_dict(self) -> dict:
261
+ return {
262
+ 'key': self.key,
263
+ 'grouped_by': self.grouped_by,
264
+ 'name_kind': self.name_kind,
265
+ 'name': self.name,
266
+ 'given': self.given,
267
+ 'von': self.von,
268
+ 'family': self.family,
269
+ 'suffix': self.suffix,
270
+ 'literal': self.literal,
271
+ 'name_variants': list(self.name_variants),
272
+ 'authorships': [dict(a) for a in self.authorships],
273
+ 'work_ids': list(self.work_ids),
274
+ 'last_year': self.last_year,
275
+ 'derived': dict(self.derived),
276
+ }
277
+
278
+
279
+ @dataclass
280
+ class Project:
281
+ """A research project."""
282
+ id: str
283
+ title: str
284
+ description: Optional[str] = None
285
+ website: Optional[str] = None
286
+ status: str = "active"
287
+
288
+ # Back-linked (computed)
289
+ work_ids: List[str] = field(default_factory=list)
290
+ people_ids: List[str] = field(default_factory=list)
291
+ derived: Dict[str, object] = field(default_factory=dict)
292
+
293
+ # Declared last so it takes no other field's position in a positional
294
+ # call; to_dict() emits it beside `website`.
295
+ image: Optional[str] = None
296
+
297
+ def to_dict(self) -> dict:
298
+ """Convert to dictionary for serialization."""
299
+ return {
300
+ 'id': self.id,
301
+ 'title': self.title,
302
+ 'description': self.description,
303
+ 'website': self.website,
304
+ 'image': self.image,
305
+ 'status': self.status,
306
+ 'work_ids': list(self.work_ids),
307
+ 'people_ids': list(self.people_ids),
308
+ 'derived': dict(self.derived),
309
+ }
310
+
311
+
312
+ @dataclass
313
+ class LabData:
314
+ """The fully resolved output: all entities with cross-references."""
315
+ works: List[Work] = field(default_factory=list)
316
+ people: List[Person] = field(default_factory=list)
317
+ projects: List[Project] = field(default_factory=list)
318
+ collaborators: List[Collaborator] = field(default_factory=list)
319
+ lab: Optional[Dict[str, object]] = None
320
+
321
+ def to_dict(self) -> dict:
322
+ """Convert to dictionary for serialization.
323
+
324
+ ``generator`` carries no timestamp, so the document is deterministic
325
+ (SPEC.md §3). The version is imported here because the package imports
326
+ this module while defining it.
327
+ """
328
+ from . import __version__
329
+
330
+ return {
331
+ 'schema_version': SCHEMA_VERSION,
332
+ 'generator': {
333
+ 'name': GENERATOR_NAME,
334
+ 'version': __version__,
335
+ 'schema_version': SCHEMA_VERSION,
336
+ },
337
+ 'lab': json_lab(dict(self.lab or {})),
338
+ 'works': [w.to_dict() for w in self.works],
339
+ 'people': [p.to_dict() for p in self.people],
340
+ 'projects': [p.to_dict() for p in self.projects],
341
+ 'collaborators': [c.to_dict() for c in self.collaborators],
342
+ }
File without changes