sslabdata 3.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
sslabdata/__init__.py ADDED
@@ -0,0 +1,40 @@
1
+ """
2
+ sslabdata - Renderer-agnostic academic lab data assembler.
3
+
4
+ Transforms BibTeX files and YAML configuration into structured data
5
+ (YAML/JSON) for academic lab websites. Framework-agnostic: works with
6
+ any static site generator, web framework, or other consumer.
7
+
8
+ Copyright (c) 2024 Personal Robotics Laboratory, University of Washington
9
+ Author: Siddhartha Srinivasa
10
+ MIT License - see LICENSE file for details.
11
+ """
12
+
13
+ from .config import ConfigurationError, LabDataConfig, BibFile
14
+ from .models import (
15
+ LabData, Work, Author, Contributor, Venue, Link, Person, Project,
16
+ Collaborator,
17
+ )
18
+ from .assembler import assemble, AssemblyError, AssemblyResult
19
+ from .exporters import export_to_yaml, export_to_json
20
+
21
+ __all__ = [
22
+ "LabDataConfig",
23
+ "BibFile",
24
+ "ConfigurationError",
25
+ "LabData",
26
+ "Work",
27
+ "Author",
28
+ "Contributor",
29
+ "Venue",
30
+ "Link",
31
+ "Person",
32
+ "Project",
33
+ "Collaborator",
34
+ "assemble",
35
+ "AssemblyResult",
36
+ "AssemblyError",
37
+ "export_to_yaml",
38
+ "export_to_json",
39
+ ]
40
+ __version__ = "3.0.0"
sslabdata/assembler.py ADDED
@@ -0,0 +1,452 @@
1
+ """
2
+ Main pipeline orchestrator.
3
+
4
+ Assembles the complete LabData output from configuration:
5
+ config → parse BibTeX → load people/projects → resolve links → back-link.
6
+
7
+ Copyright (c) 2024 Personal Robotics Laboratory, University of Washington
8
+ Author: Siddhartha Srinivasa
9
+ MIT License - see LICENSE file for details.
10
+ """
11
+
12
+ import hashlib
13
+ import re
14
+ import sys
15
+ from dataclasses import dataclass, field
16
+ from pathlib import Path
17
+ from typing import Dict, List, Optional, Tuple
18
+
19
+ from .config import (
20
+ LabDataConfig, reject_absolute_name, reject_name_outside_bib_dir,
21
+ )
22
+ from .diagnostics import (
23
+ ERROR, Diagnostic, diagnostic, in_report_order, severity,
24
+ )
25
+ from .models import Author, Collaborator, LabData, Person, Work
26
+ from .parsers.bibtex import parse_all_works
27
+ from .loaders import (
28
+ DeclaredCollaborator, load_collaborators, load_people, load_projects,
29
+ )
30
+ from .resolver import (
31
+ AMBIGUOUS, RESOLVED, Candidates, compute_backlinks, declared_form,
32
+ given_initials, initials_only, match, normalize_name,
33
+ written_form,
34
+ resolve_authors, resolve_projects, shared_declarations,
35
+ )
36
+
37
+
38
+ # `grouped_by` and `name_kind` values, both open strings (SPEC.md §5).
39
+ GROUPED_BY_NORMALIZED_NAME = "normalized_name"
40
+ GROUPED_BY_DECLARED = "declared"
41
+
42
+ PERSONAL, LITERAL = "personal", "literal"
43
+
44
+ # The digest is always present, never conditional on a collision (SPEC.md §5).
45
+ _DIGEST_LENGTH = 8
46
+
47
+ # A slug long enough to stay readable and short enough to stay a key.
48
+ _SLUG_LENGTH = 60
49
+
50
+ _NOT_SLUG = re.compile(r"-+")
51
+
52
+ # The codes this module reports (SPEC.md "Diagnostic codes").
53
+ GROUPING_SPANS_SPELLINGS = "ID-GROUPING-SPANS-SPELLINGS"
54
+ GROUPING_INITIALS_AMBIGUOUS = "ID-GROUPING-INITIALS-AMBIGUOUS"
55
+ GROUPING_AMBIGUOUS_DECLARED = "ID-GROUPING-AMBIGUOUS-DECLARED"
56
+ UNRESOLVED_NAME = "RESOLVE-UNRESOLVED-NAME"
57
+ COLLABORATOR_ALIAS_IS_MEMBER = "RESOLVE-COLLABORATOR-ALIAS-IS-MEMBER"
58
+ LAB_NAME_MISSING = "CONFIG-LAB-NAME-MISSING"
59
+ FILE_NOT_FOUND = "CONFIG-FILE-NOT-FOUND"
60
+ PATH_WRONG_KIND = "CONFIG-PATH-WRONG-KIND"
61
+
62
+
63
+ def path_problem(path: str, directory: bool = False
64
+ ) -> Optional[Tuple[str, str]]:
65
+ """What is wrong with a configured path, as ``(code, message)``, or
66
+ ``None`` when it is a regular file (a directory, with ``directory``).
67
+
68
+ Symlinks are followed, so a dangling one is missing.
69
+ """
70
+ target = Path(path)
71
+ if not target.exists():
72
+ return FILE_NOT_FOUND, f"'{path}' does not exist"
73
+ if directory:
74
+ if target.is_dir():
75
+ return None
76
+ return PATH_WRONG_KIND, f"'{path}' is not a directory"
77
+ if target.is_file():
78
+ return None
79
+ return PATH_WRONG_KIND, (f"'{path}' is a directory, not a file"
80
+ if target.is_dir() else
81
+ f"'{path}' is not a regular file")
82
+
83
+ KEY_UNKNOWN = "CONFIG-KEY-UNKNOWN"
84
+ BIB_FILES_MISSING = "CONFIG-BIB-FILES-MISSING"
85
+
86
+
87
+ @dataclass
88
+ class AssemblyResult:
89
+ """Result of assembling lab data: the document and every coded
90
+ diagnostic, in report order, with no severity stored (SPEC.md §1)."""
91
+ data: LabData
92
+ unresolved_authors: List[str] = field(default_factory=list)
93
+ unknown_projects: List[str] = field(default_factory=list)
94
+ diagnostics: List[Diagnostic] = field(default_factory=list)
95
+
96
+
97
+ class AssemblyError(ValueError):
98
+ """Raised by `assemble()` when a diagnostic is fatal (SPEC.md §1).
99
+
100
+ The message is the fatal diagnostics, one per line; ``diagnostics``
101
+ holds every diagnostic the run found, in report order.
102
+ """
103
+
104
+ def __init__(self, fatal: List[Diagnostic], diagnostics: List[Diagnostic]):
105
+ super().__init__("\n".join(str(line) for line in fatal))
106
+ self.diagnostics = diagnostics
107
+
108
+
109
+ def collaborator_key(name_kind: str, normalized: str) -> str:
110
+ """A lookup key for one grouping of unresolved authorships, not an
111
+ identity (SPEC.md §5)."""
112
+ digest = hashlib.sha256(
113
+ (name_kind + "\x00" + normalized).encode("utf-8")).hexdigest()[:_DIGEST_LENGTH]
114
+ slug = "".join(c if c.isalnum() else "-" for c in normalized)
115
+ slug = _NOT_SLUG.sub("-", slug).strip("-")[:_SLUG_LENGTH].strip("-")
116
+ return f"{slug}-{digest}" if slug else digest
117
+
118
+
119
+ class _Grouping:
120
+ """One collaborator under construction, in the order the works are read."""
121
+
122
+ def __init__(self, key: str, author: Author, normalized: str,
123
+ grouped_by: str = GROUPED_BY_NORMALIZED_NAME):
124
+ self.key = key
125
+ self.grouped_by = grouped_by
126
+ self.normalized = normalized
127
+ self.name_kind = LITERAL if author.literal else PERSONAL
128
+ self.author = author # the first spelling, in document order
129
+ # Read from the structured parts rather than from the key, and used
130
+ # only by the diagnostics below, which decide nothing about grouping.
131
+ self.family = normalize_name(author.family or "")
132
+ self.von = normalize_name(author.von or "")
133
+ self.suffix = normalize_name(author.suffix or "")
134
+ self.initials = given_initials(author.given)
135
+ self.initials_only = initials_only(author.given)
136
+ self.variants: List[str] = []
137
+ self.authorships: List[Dict[str, object]] = []
138
+ self.work_ids: List[str] = []
139
+ self.last_year: Optional[int] = None
140
+ self.where: Tuple[str, str, str] = ("", "", "")
141
+
142
+ def add(self, work: Work, author: Author,
143
+ where: Tuple[str, str, str]) -> None:
144
+ if not self.authorships:
145
+ self.where = where
146
+ if author.name not in self.variants:
147
+ self.variants.append(author.name)
148
+ self.authorships.append({"work_id": work.bib_id,
149
+ "position": author.position})
150
+ if work.bib_id not in self.work_ids:
151
+ self.work_ids.append(work.bib_id)
152
+ if work.year is not None:
153
+ self.last_year = max(self.last_year or work.year, work.year)
154
+
155
+ def could_be(self, other: "_Grouping") -> bool:
156
+ """True when this initials-only key could be that fuller one
157
+ (`ID-GROUPING-INITIALS-AMBIGUOUS` in SPEC.md "Diagnostic codes").
158
+
159
+ One suffix against none still pairs: an entry that omits its suffix
160
+ has said nothing about it.
161
+ """
162
+ if other.key == self.key or not self.initials_only:
163
+ return False
164
+ if self.name_kind != PERSONAL or other.name_kind != PERSONAL:
165
+ return False
166
+ if other.initials_only or not other.initials:
167
+ return False
168
+ if not self.family or self.family != other.family:
169
+ return False
170
+ if self.von != other.von:
171
+ return False
172
+ if self.suffix and other.suffix and self.suffix != other.suffix:
173
+ return False
174
+ shorter, longer = sorted((self.initials, other.initials), key=len)
175
+ return longer[:len(shorter)] == shorter
176
+
177
+ def build(self) -> Collaborator:
178
+ return Collaborator(
179
+ key=self.key,
180
+ name=self.author.name,
181
+ grouped_by=self.grouped_by,
182
+ name_kind=self.name_kind,
183
+ given=self.author.given,
184
+ von=self.author.von,
185
+ family=self.author.family,
186
+ suffix=self.author.suffix,
187
+ literal=self.author.literal,
188
+ name_variants=sorted(self.variants),
189
+ authorships=self.authorships,
190
+ work_ids=self.work_ids,
191
+ last_year=self.last_year,
192
+ )
193
+
194
+
195
+ def declared_collaborators(declared: List[DeclaredCollaborator],
196
+ people: List[Person], source: str,
197
+ diagnostics: List[Diagnostic]
198
+ ) -> List[Tuple[str, List[str]]]:
199
+ """The `collaborators_file` entries as ``(normalised name, spellings)``,
200
+ minus any spelling a lab member already declares.
201
+
202
+ The normalised name is what the entry's collaborator key is built from.
203
+ A spelling a member declares is reported and left out, so the member is
204
+ never shadowed.
205
+ """
206
+ members: Dict[str, set] = {}
207
+ for person in people:
208
+ for name in [person.name] + list(person.aliases):
209
+ members.setdefault(declared_form(name), set()).add(person.id)
210
+ entries = []
211
+ for collaborator in declared:
212
+ kept = []
213
+ for field_name, name in [("name", collaborator.name)] + [
214
+ ("aliases", alias) for alias in collaborator.aliases]:
215
+ owners = members.get(declared_form(name), set())
216
+ if owners:
217
+ diagnostics.append(diagnostic(
218
+ COLLABORATOR_ALIAS_IS_MEMBER, source, collaborator.name,
219
+ field_name,
220
+ f"'{name}' is also declared by {', '.join(sorted(owners))}; "
221
+ "the member keeps it and the collaborator entry is not "
222
+ "used for it"))
223
+ else:
224
+ kept.append(name)
225
+ entries.append((declared_form(collaborator.name), kept))
226
+ return entries
227
+
228
+
229
+ def group_collaborators(works: List[Work], bib_dir: str,
230
+ diagnostics: List[Diagnostic],
231
+ declared: Optional[List[Tuple[str, List[str]]]] = None,
232
+ people: Optional[List[Person]] = None) -> List[Collaborator]:
233
+ """Group every unresolved authorship (SPEC.md §5), and say where the
234
+ grouping is risky.
235
+
236
+ Mutates ``author.collaborator_key`` in place.
237
+ """
238
+ rivals = None
239
+ if declared:
240
+ rivals = Candidates(
241
+ [(f"collaborator:{name}", spellings) for name, spellings in declared]
242
+ + [(f"person:{p.id}", [p.name] + list(p.aliases))
243
+ for p in (people or [])])
244
+ groups: Dict[str, _Grouping] = {}
245
+ for work in works:
246
+ where = (f"{bib_dir}/{work.source_file}", work.bib_id, "author")
247
+ for author in work.authors:
248
+ if author.person_id:
249
+ continue
250
+ normalized = written_form(author)
251
+ kind = LITERAL if author.literal else PERSONAL
252
+ grouped_by = GROUPED_BY_NORMALIZED_NAME
253
+ if rivals is not None:
254
+ found = match(author, rivals)
255
+ ids = found.ids
256
+ if found.status == RESOLVED and ids[0].startswith("collaborator:"):
257
+ normalized = ids[0].split(":", 1)[1]
258
+ kind, grouped_by = PERSONAL, GROUPED_BY_DECLARED
259
+ elif found.status == AMBIGUOUS and any(
260
+ i.startswith("collaborator:") for i in ids):
261
+ diagnostics.append(diagnostic(
262
+ GROUPING_AMBIGUOUS_DECLARED, *where,
263
+ f"position {author.position}, '{author.name}', fits "
264
+ "more than one collaborators_file entry, or an entry "
265
+ "and a lab member, and is grouped by its own name: "
266
+ f"{', '.join(ids)}"))
267
+ key = collaborator_key(kind, normalized)
268
+ author.collaborator_key = key
269
+ group = groups.get(key)
270
+ if group is None:
271
+ group = groups[key] = _Grouping(key, author, normalized, grouped_by)
272
+ group.add(work, author, where)
273
+
274
+ diagnostics.extend(_grouping_warnings(groups))
275
+
276
+ # The order is SPEC.md §3's; `key` last makes it total.
277
+ ordered = sorted(groups.values(),
278
+ key=lambda g: (g.last_year is None, -(g.last_year or 0),
279
+ -len(g.work_ids), g.author.name, g.key))
280
+ return [group.build() for group in ordered]
281
+
282
+
283
+ def _grouping_warnings(groups: Dict[str, "_Grouping"]) -> List[Diagnostic]:
284
+ """The two ways a key over- or under-groups, located at the first
285
+ authorship the key grouped, which is where a human goes to fix the
286
+ spelling. A declared grouping spans its spellings on purpose, so neither
287
+ is reported against it."""
288
+ reported = []
289
+ for group in sorted(groups.values(), key=lambda g: g.key):
290
+ if group.grouped_by == GROUPED_BY_DECLARED:
291
+ continue
292
+ if len(group.variants) > 1:
293
+ spellings = ", ".join(repr(v) for v in sorted(group.variants))
294
+ reported.append(diagnostic(
295
+ GROUPING_SPANS_SPELLINGS, *group.where,
296
+ f"collaborator key '{group.key}' groups {len(group.variants)} "
297
+ f"spellings of one name: {spellings}"))
298
+ fuller = sorted(other.key for other in groups.values()
299
+ if group.could_be(other))
300
+ if fuller:
301
+ reported.append(diagnostic(
302
+ GROUPING_INITIALS_AMBIGUOUS, *group.where,
303
+ f"collaborator key '{group.key}' is initials only and could "
304
+ "be any of: " + ", ".join(repr(k) for k in fuller)))
305
+ return reported
306
+
307
+
308
+ def unresolved_name_diagnostics(works: List[Work], names: List[str],
309
+ bib_dir: str) -> List[Diagnostic]:
310
+ """One `UNRESOLVED_NAME` diagnostic per name, in the order given.
311
+
312
+ Each is located at the first authorship, in document order, that is
313
+ written that way and linked to no person; its message is the name.
314
+ """
315
+ first: Dict[str, Tuple[Optional[str], Optional[str]]] = {}
316
+ for work in works:
317
+ for author in work.authors:
318
+ if author.person_id is None and author.name not in first:
319
+ first[author.name] = (f"{bib_dir}/{work.source_file}", work.bib_id)
320
+ return [diagnostic(UNRESOLVED_NAME, *first.get(name, (None, None)),
321
+ "author", name)
322
+ for name in names]
323
+
324
+
325
+ def assemble(config: LabDataConfig, diagnostics: bool = False):
326
+ """Main entry point: config → fully resolved LabData.
327
+
328
+ Args:
329
+ config: Lab data configuration
330
+ diagnostics: If True, return AssemblyResult with diagnostics.
331
+ If False (default), return LabData directly, and print
332
+ each diagnostic to standard error first, fatal or not.
333
+
334
+ Raises:
335
+ AssemblyError: when any diagnostic is fatal, whatever
336
+ ``diagnostics`` is (SPEC.md §1).
337
+ """
338
+ result = assemble_result(config)
339
+ if not diagnostics:
340
+ for message in result.diagnostics:
341
+ print(f"Warning: {message}", file=sys.stderr)
342
+ fatal = [line for line in result.diagnostics
343
+ if severity(line, validating=False, strict=False) == ERROR]
344
+ if fatal:
345
+ raise AssemblyError(fatal, result.diagnostics)
346
+ return result if diagnostics else result.data
347
+
348
+
349
+ def assemble_result(config: LabDataConfig) -> AssemblyResult:
350
+ """The document and every diagnostic, whether or not one is fatal.
351
+
352
+ The CLI reads this rather than `assemble()` because it reports a run
353
+ with fatal diagnostics too; it never writes that run's document.
354
+ """
355
+ # For a configuration built in Python; `from_yaml()` has already checked
356
+ # one read from a file. Before anything is parsed, so it fails early.
357
+ for bib_file in config.bib_files:
358
+ reject_absolute_name(getattr(bib_file, 'name', None))
359
+ for bib_file in config.bib_files:
360
+ reject_name_outside_bib_dir(getattr(bib_file, 'name', None),
361
+ config.bib_dir)
362
+
363
+ found: List[Diagnostic] = []
364
+ source = config.path or 'lab.yaml'
365
+
366
+ for key in config.unknown_keys:
367
+ found.append(diagnostic(
368
+ KEY_UNKNOWN, source, key, None,
369
+ f"'{key}' is not a key sslabdata reads, and is ignored"))
370
+ found.extend(config.control_characters)
371
+ if not config.bib_files:
372
+ found.append(diagnostic(
373
+ BIB_FILES_MISSING, source, 'bib_files', None,
374
+ "no bib_files are configured, so the document has no works"))
375
+
376
+ # Every path is checked before any is read, so a problem is reported
377
+ # against the key that names it. An empty path is a mistake, not "none".
378
+ def present(path: Optional[str], key: str, field_name=None,
379
+ directory: bool = False) -> bool:
380
+ if path is None:
381
+ return True
382
+ problem = path_problem(path, directory) if path else (
383
+ FILE_NOT_FOUND,
384
+ f"the path is empty; name a file, or leave {key} out for none")
385
+ if problem is None:
386
+ return True
387
+ found.append(diagnostic(problem[0], source, key, field_name,
388
+ problem[1]))
389
+ return False
390
+
391
+ # `bib_dir` is checked only when a file is read from it. The files are
392
+ # opened as `<bib_dir>/<name>`, so an empty one is the root.
393
+ bib_dir_found = not config.bib_files or present(
394
+ config.bib_dir or "/", 'bib_dir', directory=True)
395
+ bib_files = [{'name': bf.name, 'category': bf.category}
396
+ for bf in config.bib_files if bib_dir_found and
397
+ present(f"{config.bib_dir}/{bf.name}", 'bib_files', 'name')]
398
+ people_found = present(config.people_file, 'people_file')
399
+ projects_found = present(config.projects_file, 'projects_file')
400
+ collaborators_found = present(config.collaborators_file,
401
+ 'collaborators_file')
402
+ works = parse_all_works(
403
+ bib_dir=config.bib_dir,
404
+ bib_files=bib_files,
405
+ diagnostics=found,
406
+ pdf_base_url=config.pdf_base_url,
407
+ )
408
+
409
+ people = (load_people(config.people_file, found)
410
+ if config.people_file and people_found else [])
411
+ projects = (load_projects(config.projects_file, found)
412
+ if config.projects_file and projects_found else [])
413
+ if people and config.people_file:
414
+ found.extend(shared_declarations(people, config.people_file))
415
+
416
+ unresolved_authors = resolve_authors(works, people, diagnostics=found,
417
+ bib_dir=config.bib_dir)
418
+ unknown_projects = resolve_projects(works, projects, found,
419
+ bib_dir=config.bib_dir)
420
+
421
+ declared = None
422
+ if config.collaborators_file and collaborators_found:
423
+ declared = declared_collaborators(
424
+ load_collaborators(config.collaborators_file, found), people,
425
+ config.collaborators_file, found)
426
+ collaborators = group_collaborators(works, config.bib_dir, found,
427
+ declared, people)
428
+
429
+ # A `lab` that is not a mapping is malformed rather than unnamed, and
430
+ # `from_yaml()` rejects it under its own code.
431
+ if config.lab is None or isinstance(config.lab, dict):
432
+ if not (config.lab or {}).get("name"):
433
+ found.append(diagnostic(
434
+ LAB_NAME_MISSING, config.path or 'lab.yaml', "lab", "name",
435
+ "the lab header declares no name"))
436
+
437
+ data = LabData(
438
+ works=works,
439
+ people=people,
440
+ projects=projects,
441
+ collaborators=collaborators,
442
+ lab=config.lab,
443
+ )
444
+
445
+ compute_backlinks(data)
446
+
447
+ return AssemblyResult(
448
+ data=data,
449
+ unresolved_authors=unresolved_authors,
450
+ unknown_projects=unknown_projects,
451
+ diagnostics=in_report_order(found),
452
+ )