sslabdata 4.0.0__tar.gz → 5.0.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (44) hide show
  1. {sslabdata-4.0.0/sslabdata.egg-info → sslabdata-5.0.0}/PKG-INFO +20 -5
  2. {sslabdata-4.0.0 → sslabdata-5.0.0}/README.md +19 -4
  3. {sslabdata-4.0.0 → sslabdata-5.0.0}/SPEC.md +55 -23
  4. {sslabdata-4.0.0 → sslabdata-5.0.0}/pyproject.toml +4 -4
  5. sslabdata-5.0.0/schema/input/v2/collaborators.schema.json +24 -0
  6. sslabdata-5.0.0/schema/input/v2/lab.schema.json +71 -0
  7. sslabdata-5.0.0/schema/input/v2/people.schema.json +51 -0
  8. sslabdata-5.0.0/schema/input/v2/projects.schema.json +34 -0
  9. sslabdata-5.0.0/schema/v7/output.schema.json +485 -0
  10. {sslabdata-4.0.0 → sslabdata-5.0.0}/sslabdata/__init__.py +1 -1
  11. {sslabdata-4.0.0 → sslabdata-5.0.0}/sslabdata/loaders.py +3 -2
  12. {sslabdata-4.0.0 → sslabdata-5.0.0}/sslabdata/models.py +8 -2
  13. {sslabdata-4.0.0 → sslabdata-5.0.0}/sslabdata/templates/init/collaborators.yaml +1 -1
  14. {sslabdata-4.0.0 → sslabdata-5.0.0}/sslabdata/templates/init/lab.yaml +1 -1
  15. {sslabdata-4.0.0 → sslabdata-5.0.0}/sslabdata/templates/init/people.yaml +2 -1
  16. {sslabdata-4.0.0 → sslabdata-5.0.0}/sslabdata/templates/init/projects.yaml +1 -1
  17. {sslabdata-4.0.0 → sslabdata-5.0.0/sslabdata.egg-info}/PKG-INFO +20 -5
  18. {sslabdata-4.0.0 → sslabdata-5.0.0}/sslabdata.egg-info/SOURCES.txt +5 -0
  19. {sslabdata-4.0.0 → sslabdata-5.0.0}/LICENSE +0 -0
  20. {sslabdata-4.0.0 → sslabdata-5.0.0}/MANIFEST.in +0 -0
  21. {sslabdata-4.0.0 → sslabdata-5.0.0}/schema/__init__.py +0 -0
  22. {sslabdata-4.0.0 → sslabdata-5.0.0}/schema/input/v1/collaborators.schema.json +0 -0
  23. {sslabdata-4.0.0 → sslabdata-5.0.0}/schema/input/v1/lab.schema.json +0 -0
  24. {sslabdata-4.0.0 → sslabdata-5.0.0}/schema/input/v1/people.schema.json +0 -0
  25. {sslabdata-4.0.0 → sslabdata-5.0.0}/schema/input/v1/projects.schema.json +0 -0
  26. {sslabdata-4.0.0 → sslabdata-5.0.0}/schema/v3/output.schema.json +0 -0
  27. {sslabdata-4.0.0 → sslabdata-5.0.0}/schema/v4/output.schema.json +0 -0
  28. {sslabdata-4.0.0 → sslabdata-5.0.0}/schema/v5/output.schema.json +0 -0
  29. {sslabdata-4.0.0 → sslabdata-5.0.0}/schema/v6/output.schema.json +0 -0
  30. {sslabdata-4.0.0 → sslabdata-5.0.0}/setup.cfg +0 -0
  31. {sslabdata-4.0.0 → sslabdata-5.0.0}/sslabdata/assembler.py +0 -0
  32. {sslabdata-4.0.0 → sslabdata-5.0.0}/sslabdata/cli.py +0 -0
  33. {sslabdata-4.0.0 → sslabdata-5.0.0}/sslabdata/config.py +0 -0
  34. {sslabdata-4.0.0 → sslabdata-5.0.0}/sslabdata/diagnostics.py +0 -0
  35. {sslabdata-4.0.0 → sslabdata-5.0.0}/sslabdata/exporters.py +0 -0
  36. {sslabdata-4.0.0 → sslabdata-5.0.0}/sslabdata/parsers/__init__.py +0 -0
  37. {sslabdata-4.0.0 → sslabdata-5.0.0}/sslabdata/parsers/bibtex.py +0 -0
  38. {sslabdata-4.0.0 → sslabdata-5.0.0}/sslabdata/parsers/latex.py +0 -0
  39. {sslabdata-4.0.0 → sslabdata-5.0.0}/sslabdata/resolver.py +0 -0
  40. {sslabdata-4.0.0 → sslabdata-5.0.0}/sslabdata/templates/init/bib/publications.bib +0 -0
  41. {sslabdata-4.0.0 → sslabdata-5.0.0}/sslabdata.egg-info/dependency_links.txt +0 -0
  42. {sslabdata-4.0.0 → sslabdata-5.0.0}/sslabdata.egg-info/entry_points.txt +0 -0
  43. {sslabdata-4.0.0 → sslabdata-5.0.0}/sslabdata.egg-info/requires.txt +0 -0
  44. {sslabdata-4.0.0 → sslabdata-5.0.0}/sslabdata.egg-info/top_level.txt +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: sslabdata
3
- Version: 4.0.0
3
+ Version: 5.0.0
4
4
  Summary: Renderer-agnostic academic lab data assembler: BibTeX + YAML → structured data
5
5
  Author: Siddhartha Srinivasa
6
6
  License-Expression: MIT
@@ -255,6 +255,9 @@ author names to people:
255
255
  website: "https://example.org/people/bbrown"
256
256
  co_advisor: "Peggy Park"
257
257
  start_year: 2021
258
+ bio: |
259
+ Bob Brown is a PhD student advised by Alice Adams and Peggy Park.
260
+ He works on shared control for assistive robot arms.
258
261
 
259
262
  - id: "iingram"
260
263
  name: "Ivan Ingram"
@@ -271,6 +274,16 @@ author names to people:
271
274
  `id` and `name` are required. `role` is any non-empty string, so any lab's
272
275
  roles fit; `status` is `current` (the default) or `alumni`.
273
276
 
277
+ `bio` is a short biography in plain text, not Markdown or HTML: it is
278
+ emitted as written and a renderer escapes it. Line breaks are kept as YAML
279
+ reads them, so a block scalar (`|`) keeps each line, and a renderer may treat
280
+ a blank line as a paragraph break. A person without one gets `bio: null`; an
281
+ empty string is emitted as written, as any other person field is. Keep the
282
+ structured facts in their own fields: nothing reads `degree`, years or
283
+ `current_position` out of a bio, so an alumni line such as "PhD 2022, now
284
+ Research Scientist at Example Robotics" is for the renderer to build from
285
+ those fields.
286
+
274
287
  ### External co-authors (optional, `data/collaborators.yaml`)
275
288
 
276
289
  A list of co-authors outside the lab whose spellings you want grouped
@@ -306,24 +319,26 @@ deciding which URLs are safe to render is the renderer's job.
306
319
  ### Checking inputs in an editor
307
320
 
308
321
  Each input file has a JSON Schema in
309
- [`schema/input/v1/`](https://github.com/siddhss5/sslabdata/tree/main/schema/input/v1):
322
+ [`schema/input/v2/`](https://github.com/siddhss5/sslabdata/tree/main/schema/input/v2):
310
323
  `lab.schema.json`, `people.schema.json`, `projects.schema.json` and
311
324
  `collaborators.schema.json`. The wheel installs them under
312
- `sslabdata/schema/input/v1/`. An editor can use them to check and complete
325
+ `sslabdata/schema/input/v2/`. An editor can use them to check and complete
313
326
  the files as you write; `--validate` stays the check, and also reports what
314
327
  a schema cannot see, such as a repeated id or a missing file. With the YAML
315
328
  language server (the VS Code YAML extension, among others), name the schema
316
329
  in a comment at the top of the file:
317
330
 
318
331
  ```yaml
319
- # yaml-language-server: $schema=https://raw.githubusercontent.com/siddhss5/sslabdata/input-schema-v1/schema/input/v1/people.schema.json
332
+ # yaml-language-server: $schema=https://raw.githubusercontent.com/siddhss5/sslabdata/input-schema-v2/schema/input/v2/people.schema.json
320
333
  - id: "aadams"
321
334
  name: "Alice Adams"
322
335
  role: "professor"
323
336
  ```
324
337
 
325
- The URL is the schema's `$id`, served from the `input-schema-v1` tag, which is
338
+ The URL is the schema's `$id`, served from the `input-schema-v2` tag, which is
326
339
  never moved ([`SPEC.md` §6](https://github.com/siddhss5/sslabdata/blob/main/SPEC.md#6-version-policy)).
340
+ A file that names v1 still validates, because v1 allows keys it does not
341
+ list, but only v2 checks and completes `bio`.
327
342
  Use `lab.schema.json`, `projects.schema.json` or `collaborators.schema.json`
328
343
  in the same way for the other files.
329
344
 
@@ -218,6 +218,9 @@ author names to people:
218
218
  website: "https://example.org/people/bbrown"
219
219
  co_advisor: "Peggy Park"
220
220
  start_year: 2021
221
+ bio: |
222
+ Bob Brown is a PhD student advised by Alice Adams and Peggy Park.
223
+ He works on shared control for assistive robot arms.
221
224
 
222
225
  - id: "iingram"
223
226
  name: "Ivan Ingram"
@@ -234,6 +237,16 @@ author names to people:
234
237
  `id` and `name` are required. `role` is any non-empty string, so any lab's
235
238
  roles fit; `status` is `current` (the default) or `alumni`.
236
239
 
240
+ `bio` is a short biography in plain text, not Markdown or HTML: it is
241
+ emitted as written and a renderer escapes it. Line breaks are kept as YAML
242
+ reads them, so a block scalar (`|`) keeps each line, and a renderer may treat
243
+ a blank line as a paragraph break. A person without one gets `bio: null`; an
244
+ empty string is emitted as written, as any other person field is. Keep the
245
+ structured facts in their own fields: nothing reads `degree`, years or
246
+ `current_position` out of a bio, so an alumni line such as "PhD 2022, now
247
+ Research Scientist at Example Robotics" is for the renderer to build from
248
+ those fields.
249
+
237
250
  ### External co-authors (optional, `data/collaborators.yaml`)
238
251
 
239
252
  A list of co-authors outside the lab whose spellings you want grouped
@@ -269,24 +282,26 @@ deciding which URLs are safe to render is the renderer's job.
269
282
  ### Checking inputs in an editor
270
283
 
271
284
  Each input file has a JSON Schema in
272
- [`schema/input/v1/`](https://github.com/siddhss5/sslabdata/tree/main/schema/input/v1):
285
+ [`schema/input/v2/`](https://github.com/siddhss5/sslabdata/tree/main/schema/input/v2):
273
286
  `lab.schema.json`, `people.schema.json`, `projects.schema.json` and
274
287
  `collaborators.schema.json`. The wheel installs them under
275
- `sslabdata/schema/input/v1/`. An editor can use them to check and complete
288
+ `sslabdata/schema/input/v2/`. An editor can use them to check and complete
276
289
  the files as you write; `--validate` stays the check, and also reports what
277
290
  a schema cannot see, such as a repeated id or a missing file. With the YAML
278
291
  language server (the VS Code YAML extension, among others), name the schema
279
292
  in a comment at the top of the file:
280
293
 
281
294
  ```yaml
282
- # yaml-language-server: $schema=https://raw.githubusercontent.com/siddhss5/sslabdata/input-schema-v1/schema/input/v1/people.schema.json
295
+ # yaml-language-server: $schema=https://raw.githubusercontent.com/siddhss5/sslabdata/input-schema-v2/schema/input/v2/people.schema.json
283
296
  - id: "aadams"
284
297
  name: "Alice Adams"
285
298
  role: "professor"
286
299
  ```
287
300
 
288
- The URL is the schema's `$id`, served from the `input-schema-v1` tag, which is
301
+ The URL is the schema's `$id`, served from the `input-schema-v2` tag, which is
289
302
  never moved ([`SPEC.md` §6](https://github.com/siddhss5/sslabdata/blob/main/SPEC.md#6-version-policy)).
303
+ A file that names v1 still validates, because v1 allows keys it does not
304
+ list, but only v2 checks and completes `bio`.
290
305
  Use `lab.schema.json`, `projects.schema.json` or `collaborators.schema.json`
291
306
  in the same way for the other files.
292
307
 
@@ -7,7 +7,7 @@ them.
7
7
  This file states the parts of that contract a JSON Schema cannot express:
8
8
  what the strings in the document are, what order the lists are in, what an
9
9
  absent key means, which fields are computed, when the version changes, and
10
- how a repeated `@string` macro resolves. `schema/v6/output.schema.json`
10
+ how a repeated `@string` macro resolves. `schema/v7/output.schema.json`
11
11
  states the rest.
12
12
 
13
13
  Everything here is normative unless it carries a `Target` note. A `Target`
@@ -16,8 +16,8 @@ names the issue that will make it true. Until that issue lands, the rule is
16
16
  the intent and the note is the fact. What changed at each release, and what
17
17
  it replaced, is in [`CHANGELOG.md`](CHANGELOG.md), not here.
18
18
 
19
- - Applies to: `schema_version` 6 (`sslabdata.models.SCHEMA_VERSION`), package
20
- version 4.0.0 (`sslabdata.__version__`).
19
+ - Applies to: `schema_version` 7 (`sslabdata.models.SCHEMA_VERSION`), package
20
+ version 5.0.0 (`sslabdata.__version__`).
21
21
 
22
22
  ### How this file cites the code
23
23
 
@@ -25,7 +25,7 @@ Every rule below is grounded in a named part of the code rather than a line
25
25
  number, because line numbers rot silently: a function such as
26
26
  `sslabdata.parsers.bibtex.parse_all_works()`, a method such as
27
27
  `Person.to_dict()`, a module-level constant such as `TEXT_FIELDS`, a JSON
28
- Pointer into `schema/v6/output.schema.json` such as `/$defs/person/required`, or
28
+ Pointer into `schema/v7/output.schema.json` such as `/$defs/person/required`, or
29
29
  a `tests/COVERAGE.md` row key such as `config.people_file.missing`. A bare
30
30
  statement is cited by its enclosing function.
31
31
 
@@ -275,7 +275,7 @@ Codes in use:
275
275
  | `PROJECTS-ID-DUPLICATE` | Two projects declare one `id`. Located at the second. Both are kept. A validation error. |
276
276
  | `PROJECTS-STATUS-INVALID` | A project's `status` is present and is not `active` or `completed`. A missing status reads as `active`, and so does one that is not a string. A warning. |
277
277
  | `RECORD-KEY-UNKNOWN` | A person, project or collaborator record holds a key sslabdata does not read (`sslabdata.loaders.PERSON_KEYS`, `PROJECT_KEYS`, `COLLABORATOR_KEYS`), such as `hobby` or a misspelt `webiste`. One code for all three files. Located at `<people_file>:<id>:<key>`, `<projects_file>:<id>:<key>` or `<collaborators_file>:<collaborator name>:<key>`, once per key. The key is ignored and never emitted; the record is kept. Only a record that is loaded is checked, so a record missing a required field reports that alone. A warning. |
278
- | `RECORD-TYPE-INVALID` | An optional field of a person, project or collaborator record has a value of the wrong type, other than the `role` and `status` that have codes of their own. A person's `photo`, `website`, `email`, `co_advisor`, `degree`, `thesis_title` and `current_position`, and a project's `description`, `website` and `image`, are strings; a person's `start_year` and `end_year` are integers, not booleans; a person's or collaborator's `aliases` is a list of non-empty strings. One code for all three files, located at `<file>:<id>:<field>` (a collaborator's name for its id). The value is read as empty, so it is emitted as `null`, and a wrong `aliases` declares none; the record is kept. Only a record that is loaded is checked. A warning. |
278
+ | `RECORD-TYPE-INVALID` | An optional field of a person, project or collaborator record has a value of the wrong type, other than the `role` and `status` that have codes of their own. A person's `photo`, `website`, `email`, `co_advisor`, `degree`, `thesis_title`, `current_position` and `bio`, and a project's `description`, `website` and `image`, are strings; a person's `start_year` and `end_year` are integers, not booleans; a person's or collaborator's `aliases` is a list of non-empty strings. One code for all three files, located at `<file>:<id>:<field>` (a collaborator's name for its id). The value is read as empty, so it is emitted as `null`, and a wrong `aliases` declares none; the record is kept. Only a record that is loaded is checked. A warning. |
279
279
  | `RECORD-KEY-REPEATED` | A key is given twice in one mapping of a people, projects or collaborators file, which YAML alone would read as its last value, silently dropping the first. Every YAML file sslabdata reads is read with one loader that finds these (`sslabdata.config.YAMLLoader`). Keys are compared as YAML reads them, so `1` and `0x1` are one key. A merge key (`<<`) is not a repeat: the keys it merges in are overridden by the mapping's own, as YAML merge keys are defined. One code for all three files. Located at `<file>:<id>:<key>` (a collaborator's name for its id), with the path below the record's top level joined by `.` and a list member's index in brackets; the record is named by nothing when the repeated key is its `id` (a collaborator's `name`), and a repeat outside any record is located at `<file>::<path>`. The prose names the key and both lines. Every repeat is reported; the record is not loaded, and the run is fatal, as for a missing field: which value was meant is not sslabdata's to guess. |
280
280
  | `CONFIG-NOT-A-MAPPING` | `lab.yaml` is not a mapping of keys, or is empty. Fatal at load. |
281
281
  | `CONFIG-KEY-MISSING` | A required key is absent: `bib_dir`, or the `name` or `category` of a `bib_files` entry (`lab.yaml:bib_files:name`). Fatal at load. |
@@ -541,10 +541,20 @@ converts nor checks them:
541
541
  - **The person and project strings supplied in YAML.**
542
542
  `sslabdata.loaders.load_people()` and `load_projects()` perform no conversion
543
543
  of any kind — they check the types of a record's fields, a person's `role` and that a
544
- `status` is one they know, but emit every string as written — so a person's `name`, `role`, `current_position` or
545
- `thesis_title`, and a project's `title` or `description`, are copied
544
+ `status` is one they know, but emit every string as written — so a person's `name`, `role`, `current_position`,
545
+ `thesis_title` or `bio`, and a project's `title` or `description`, are copied
546
546
  straight from `people.yaml` and `projects.yaml`. Not every YAML string is
547
547
  emitted — `aliases` and the configuration paths are not; see heading 3.
548
+ "As written" is the string YAML reads, after control characters are
549
+ removed and NFC is applied as for every input string (above), and nothing
550
+ else: an empty or whitespace-only string is emitted as such, not as
551
+ `null`, and line breaks are kept. That matters most for a person's `bio`,
552
+ the one of these meant to run over several lines: a literal block scalar
553
+ (`|`) keeps each line break and the final one, a folded one (`>`) joins
554
+ lines with a space and keeps a blank line as a line break, and `\n` in a
555
+ double-quoted string is a line break. It is plain text like the rest — not
556
+ Markdown and not HTML — and a renderer escapes it, and may read a blank
557
+ line as a paragraph break.
548
558
  - **`work.category`**, which comes from the `category` of the `bib_files`
549
559
  entry in `lab.yaml`, not from the `.bib` file
550
560
  (`sslabdata.config.LabDataConfig.from_yaml()`, then
@@ -805,7 +815,7 @@ sslabdata's own output as input, and a wrong derivation becomes permanent.
805
815
  | `author.resolution` | **Derived** — `status` over `resolved`, `unresolved` and `ambiguous`, and `method` `exact`, or `null` when nothing matched (`resolve_authors()`). Both are open strings; `fuzzy` is never emitted. |
806
816
  | `author.equal_contribution` | **Derived** — whether the entry wrote a `*` marker on any part of the name (`sslabdata.parsers.bibtex.marks_equal_contribution()`). |
807
817
  | `work.editors[*]` | The same, minus `collaborator_key` and `equal_contribution`. An editor that matched nobody is simply `person_id: null` (`parse_editor_list()`). |
808
- | `person.*` except the two below | Input — the fields of `people_file` (`sslabdata.loaders.load_people()`). `aliases` is read for matching and is **not** emitted. `status` is `current` or `alumni`, and `current` when absent; `role` is open, any non-empty string (`PEOPLE-STATUS-INVALID`, `PEOPLE-ROLE-INVALID`). |
818
+ | `person.*` except the two below | Input — the fields of `people_file` (`sslabdata.loaders.load_people()`). `aliases` is read for matching and is **not** emitted. `status` is `current` or `alumni`, and `current` when absent; `role` is open, any non-empty string (`PEOPLE-STATUS-INVALID`, `PEOPLE-ROLE-INVALID`). `bio` is a short biography in plain text with its line breaks as written (§2), and `null` when absent; it is carried and never read: nothing in the document is derived from it, and no `degree`, year or `current_position` is read out of one. |
809
819
  | `person.work_ids` | **Derived** — back-links over authorships (`sslabdata.resolver.compute_backlinks()`). Editors are not authorships and are not listed. |
810
820
  | `project.id`, `title`, `description`, `website`, `image`, `status` | Input — the fields of `projects_file` (`sslabdata.loaders.load_projects()`). `status` is one of `active` and `completed`, and `active` when absent (`PROJECTS-STATUS-INVALID`). `image` is a URL or a site path, the same kind of value as a person's `photo`, carried as plain text: deciding which URLs are safe to render is the renderer's job. |
811
821
  | `project.work_ids`, `people_ids` | **Derived** — back-links, and the people reached through them (`compute_backlinks()`). |
@@ -1120,14 +1130,14 @@ meaning and guarantees, and each of them is a bump.
1120
1130
  has been published is never edited. Version `N`'s schema stays reachable, byte
1121
1131
  for byte, at its own path after version `N+1` ships, so a consumer pinned to
1122
1132
  `N` keeps a stable target. `schema/v3/output.schema.json`,
1123
- `schema/v4/output.schema.json`, `schema/v5/output.schema.json` and
1124
- `schema/v6/output.schema.json` are those paths, and `tests/COVERAGE.md` row
1125
- `output.versioned_schema` asserts that the older three are unchanged byte for
1126
- byte and still say `3`, `4` and `5`.
1127
-
1128
- **The `$id` is a pinned tag URL.** v6's `$id` is
1129
- `https://raw.githubusercontent.com/siddhss5/sslabdata/schema-v6/schema/v6/output.schema.json`.
1130
- The rule that makes it a contract rather than a guess: **the `schema-v6` tag
1133
+ `schema/v4/output.schema.json`, `schema/v5/output.schema.json`,
1134
+ `schema/v6/output.schema.json` and `schema/v7/output.schema.json` are those
1135
+ paths, and `tests/COVERAGE.md` row `output.versioned_schema` asserts that the
1136
+ older four are unchanged byte for byte and still say `3`, `4`, `5` and `6`.
1137
+
1138
+ **The `$id` is a pinned tag URL.** v7's `$id` is
1139
+ `https://raw.githubusercontent.com/siddhss5/sslabdata/schema-v7/schema/v7/output.schema.json`.
1140
+ The rule that makes it a contract rather than a guess: **the `schema-v7` tag
1131
1141
  is created when this version ships and is never moved.** A branch URL such as
1132
1142
  `blob/main` is not usable — it serves an HTML page rather than the schema, so
1133
1143
  no consumer can ever have resolved v3's `$id` — and this repository publishes
@@ -1142,11 +1152,11 @@ rule.
1142
1152
  descriptions inside those files, name the repository `labdata`, and the files
1143
1153
  are left byte for byte as published rather than rewritten. v4's raw `$id`
1144
1154
  resolves, because GitHub redirects `labdata` to `sslabdata`. v3's `blob/main`
1145
- `$id` does not resolve, as above. The `$id`s, titles and descriptions of v5
1146
- and v6 say `sslabdata`.
1155
+ `$id` does not resolve, as above. The `$id`s, titles and descriptions of v5,
1156
+ v6 and v7 say `sslabdata`.
1147
1157
 
1148
1158
  **The input schemas are versioned apart from the document.**
1149
- `schema/input/v1/` holds a JSON Schema for each input file: `lab.schema.json`,
1159
+ `schema/input/v2/` holds a JSON Schema for each input file: `lab.schema.json`,
1150
1160
  `people.schema.json`, `projects.schema.json` and `collaborators.schema.json`.
1151
1161
  They describe what the loaders accept; the loaders stay the authority, and
1152
1162
  the diagnostic codes above report what a schema cannot see. The input format
@@ -1154,9 +1164,13 @@ and the document change for different reasons, so the input schemas carry
1154
1164
  their own version rather than `schema_version`. They follow the same rules:
1155
1165
  a published input schema is never edited, any change to one is a new
1156
1166
  version at a new path, and its `$id` is a pinned tag URL,
1157
- `https://raw.githubusercontent.com/siddhss5/sslabdata/input-schema-v1/schema/input/v1/<file>.schema.json`,
1158
- whose `input-schema-v1` tag is created when this version ships and is never
1159
- moved.
1167
+ `https://raw.githubusercontent.com/siddhss5/sslabdata/input-schema-v2/schema/input/v2/<file>.schema.json`,
1168
+ whose `input-schema-v2` tag is created when this version ships and is never
1169
+ moved. A version holds all four files, so a change to one of them versions
1170
+ the set: v2 adds a person's `bio` to `people.schema.json`, and its other three
1171
+ files differ from v1's only in the tag their `$id`s and descriptions name. `schema/input/v1/`, served
1172
+ from the `input-schema-v1` tag, stays byte for byte as published; the wheel
1173
+ installs only the current version.
1160
1174
 
1161
1175
  **Version history.** What each `schema_version` changed is in
1162
1176
  [`CHANGELOG.md`](CHANGELOG.md).
@@ -1256,6 +1270,24 @@ So each rejected type is rejected for a stated reason, and each has a home:
1256
1270
  | Alumni | Not a collection — a `status` on a person (`sslabdata.models.Person.status`). |
1257
1271
  | Robots, platforms, facilities | Your site repository; one of 27 surveyed sites had such a page. |
1258
1272
 
1273
+ **An attribute of an entity the document already has is not an entity.**
1274
+ The evidence above, and the question #60 asks of a candidate — would
1275
+ sslabdata do anything with it beyond carrying it — decide which *entities*
1276
+ the document has: collections with records of their own, ids, references
1277
+ and an order, each of which the compiler has to check and a renderer has to
1278
+ understand. A field
1279
+ on a person, work or project raises none of that. Its owner is already
1280
+ admitted, the field has one value of one type, and the only work it asks
1281
+ for is what every input string gets: control characters removed, NFC and
1282
+ the text rule (§2). A person already carries attributes sslabdata does
1283
+ nothing else with — `photo`, `website`, `thesis_title`, `current_position` —
1284
+ and a person's `bio` is one more. Such a field is judged on two questions
1285
+ instead: does it belong to that entity rather than to a page of the site,
1286
+ and is it the only place the fact can live, so that it does not copy a
1287
+ structured field into text that will drift from it. A bio passes both; an
1288
+ alumni line such as "PhD 2012, now at Example Robotics" fails the second,
1289
+ because `degree`, `end_year` and `current_position` already hold it.
1290
+
1259
1291
  **There is no generic extension mechanism and no `collections` escape hatch.**
1260
1292
  What one would carry is mostly prose, and its one real service — catching
1261
1293
  references that point at nothing — is delivered by #58 without the document
@@ -1267,7 +1299,7 @@ vocabularies, not over arbitrary content.
1267
1299
 
1268
1300
  ## 9. What this file is not
1269
1301
 
1270
- It does not list the document's fields; `schema/v6/output.schema.json` does.
1302
+ It does not list the document's fields; `schema/v7/output.schema.json` does.
1271
1303
  It does not describe renderers such as
1272
1304
  [sslabdata-site](https://github.com/siddhss5/sslabdata-site), which are
1273
1305
  downstream consumers in their own repositories. It does not describe the input formats `lab.yaml`,
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "sslabdata"
3
- version = "4.0.0"
3
+ version = "5.0.0"
4
4
  description = "Renderer-agnostic academic lab data assembler: BibTeX + YAML → structured data"
5
5
  readme = "README.md"
6
6
  requires-python = ">=3.10"
@@ -58,8 +58,8 @@ requires = ["setuptools>=77"]
58
58
  build-backend = "setuptools.build_meta"
59
59
 
60
60
  # The current output schema and input schemas are public wheel data, installed
61
- # at sslabdata/schema/v6/output.schema.json and
62
- # sslabdata/schema/input/v1/*.schema.json. The package directory is mapped onto
61
+ # at sslabdata/schema/v7/output.schema.json and
62
+ # sslabdata/schema/input/v2/*.schema.json. The package directory is mapped onto
63
63
  # the repository's schema/ directory so each frozen file has one copy and one
64
64
  # path; earlier versions stay in the sdist only. schema/__init__.py makes the
65
65
  # mapped directory a regular package, which editable installs need to resolve.
@@ -71,7 +71,7 @@ packages = ["sslabdata", "sslabdata.parsers", "sslabdata.schema"]
71
71
  "sslabdata.schema" = "schema"
72
72
 
73
73
  [tool.setuptools.package-data]
74
- "sslabdata.schema" = ["v6/output.schema.json", "input/v1/*.schema.json"]
74
+ "sslabdata.schema" = ["v7/output.schema.json", "input/v2/*.schema.json"]
75
75
  # The starting point `sslabdata init` copies.
76
76
  "sslabdata" = ["templates/init/*.yaml", "templates/init/bib/*.bib"]
77
77
 
@@ -0,0 +1,24 @@
1
+ {
2
+ "$schema": "https://json-schema.org/draft/2020-12/schema",
3
+ "$id": "https://raw.githubusercontent.com/siddhss5/sslabdata/input-schema-v2/schema/input/v2/collaborators.schema.json",
4
+ "title": "sslabdata collaborators file",
5
+ "description": "The file lab.yaml names as collaborators_file: co-authors outside the lab whose spellings are grouped together, one record each. A record never makes anyone a lab member. It describes what sslabdata's loader accepts; the loader stays the authority, and reports each problem under a code in SPEC.md, Diagnostic codes. A record key not listed here is reported as RECORD-KEY-UNKNOWN and ignored, so it is allowed. Published input schemas are immutable: this one is served from the input-schema-v2 tag, which is created when this version ships and is never moved (SPEC.md §6).",
6
+ "type": ["array", "null"],
7
+ "items": { "$ref": "#/$defs/collaborator" },
8
+ "$defs": {
9
+ "nonEmptyString": { "type": "string", "pattern": "\\S" },
10
+ "collaborator": {
11
+ "type": "object",
12
+ "required": ["name"],
13
+ "additionalProperties": true,
14
+ "properties": {
15
+ "name": { "$ref": "#/$defs/nonEmptyString" },
16
+ "aliases": {
17
+ "description": "Other spellings of the name, grouped with it.",
18
+ "type": ["array", "null"],
19
+ "items": { "$ref": "#/$defs/nonEmptyString" }
20
+ }
21
+ }
22
+ }
23
+ }
24
+ }
@@ -0,0 +1,71 @@
1
+ {
2
+ "$schema": "https://json-schema.org/draft/2020-12/schema",
3
+ "$id": "https://raw.githubusercontent.com/siddhss5/sslabdata/input-schema-v2/schema/input/v2/lab.schema.json",
4
+ "title": "sslabdata lab.yaml",
5
+ "description": "The configuration file sslabdata is run with (--config). It describes what sslabdata's loader accepts; the loader stays the authority, and reports each problem under a code in SPEC.md, Diagnostic codes. A key not listed here is reported as CONFIG-KEY-UNKNOWN and ignored, so it is allowed. Published input schemas are immutable: this one is served from the input-schema-v2 tag, which is created when this version ships and is never moved (SPEC.md §6).",
6
+ "type": "object",
7
+ "required": ["bib_dir"],
8
+ "additionalProperties": true,
9
+ "properties": {
10
+ "lab": {
11
+ "description": "The lab header, copied into the document's lab unchanged (SPEC.md §5). The keys below are typed; any other key is allowed and copied through. sslabdata warns when it declares no name (CONFIG-LAB-NAME-MISSING).",
12
+ "type": ["object", "null"],
13
+ "additionalProperties": true,
14
+ "properties": {
15
+ "name": { "type": "string" },
16
+ "description": { "type": "string" },
17
+ "institution": { "type": "string" },
18
+ "department": { "type": "string" },
19
+ "website": { "type": "string" },
20
+ "email": { "type": "string" },
21
+ "address": { "type": "string" },
22
+ "logo": { "type": "string" },
23
+ "links": { "type": "object" }
24
+ }
25
+ },
26
+ "site": {
27
+ "description": "Read by renderers, not by sslabdata, and accepted without being checked."
28
+ },
29
+ "bib_dir": {
30
+ "description": "The directory the bib_files names are under, relative to the directory sslabdata is run from.",
31
+ "type": "string"
32
+ },
33
+ "bib_files": {
34
+ "description": "The .bib files to read. sslabdata warns when there are none (CONFIG-BIB-FILES-MISSING).",
35
+ "type": ["array", "null"],
36
+ "items": {
37
+ "type": "object",
38
+ "required": ["name", "category"],
39
+ "additionalProperties": false,
40
+ "properties": {
41
+ "name": {
42
+ "description": "A name under bib_dir, emitted as work.source.file: never absolute, and never leaving bib_dir (SPEC.md §5).",
43
+ "type": "string"
44
+ },
45
+ "category": {
46
+ "description": "Emitted as the category of every work in this file.",
47
+ "type": "string"
48
+ }
49
+ }
50
+ }
51
+ },
52
+ "pdf_base_url": {
53
+ "description": "A work with no pdf field gets this URL plus its citation key as its pdf link. Leave it out to guess no pdf links; an empty or blank value is an error.",
54
+ "type": ["string", "null"],
55
+ "minLength": 1,
56
+ "pattern": "\\S"
57
+ },
58
+ "people_file": {
59
+ "description": "The people file (people.schema.json), relative to the directory sslabdata is run from.",
60
+ "type": ["string", "null"]
61
+ },
62
+ "projects_file": {
63
+ "description": "The projects file (projects.schema.json), relative to the directory sslabdata is run from.",
64
+ "type": ["string", "null"]
65
+ },
66
+ "collaborators_file": {
67
+ "description": "The external co-authors file (collaborators.schema.json), relative to the directory sslabdata is run from.",
68
+ "type": ["string", "null"]
69
+ }
70
+ }
71
+ }
@@ -0,0 +1,51 @@
1
+ {
2
+ "$schema": "https://json-schema.org/draft/2020-12/schema",
3
+ "$id": "https://raw.githubusercontent.com/siddhss5/sslabdata/input-schema-v2/schema/input/v2/people.schema.json",
4
+ "title": "sslabdata people file",
5
+ "description": "The file lab.yaml names as people_file: a list of lab members and alumni, one record each. It describes what sslabdata's loader accepts; the loader stays the authority, and reports each problem under a code in SPEC.md, Diagnostic codes. A record key not listed here is reported as RECORD-KEY-UNKNOWN and ignored, so it is allowed. What each field becomes in the document is in SPEC.md §5. Published input schemas are immutable: this one is served from the input-schema-v2 tag, which is created when this version ships and is never moved (SPEC.md §6).",
6
+ "type": ["array", "null"],
7
+ "items": { "$ref": "#/$defs/person" },
8
+ "$defs": {
9
+ "nonEmptyString": { "type": "string", "pattern": "\\S" },
10
+ "optionalString": { "type": ["string", "null"] },
11
+ "optionalYear": { "type": ["integer", "null"] },
12
+ "person": {
13
+ "type": "object",
14
+ "required": ["id", "name", "role"],
15
+ "additionalProperties": true,
16
+ "properties": {
17
+ "id": {
18
+ "description": "The person's id, unique in the file (PEOPLE-ID-DUPLICATE).",
19
+ "$ref": "#/$defs/nonEmptyString"
20
+ },
21
+ "name": { "$ref": "#/$defs/nonEmptyString" },
22
+ "aliases": {
23
+ "description": "Other spellings of the name, used to match BibTeX authors and never emitted.",
24
+ "type": ["array", "null"],
25
+ "items": { "$ref": "#/$defs/nonEmptyString" }
26
+ },
27
+ "role": {
28
+ "description": "Any non-empty string; there is no list of roles (PEOPLE-ROLE-INVALID).",
29
+ "$ref": "#/$defs/nonEmptyString"
30
+ },
31
+ "status": {
32
+ "description": "current when absent (PEOPLE-STATUS-INVALID).",
33
+ "enum": ["current", "alumni"]
34
+ },
35
+ "photo": { "$ref": "#/$defs/optionalString" },
36
+ "website": { "$ref": "#/$defs/optionalString" },
37
+ "email": { "$ref": "#/$defs/optionalString" },
38
+ "co_advisor": { "$ref": "#/$defs/optionalString" },
39
+ "start_year": { "$ref": "#/$defs/optionalYear" },
40
+ "end_year": { "$ref": "#/$defs/optionalYear" },
41
+ "degree": { "$ref": "#/$defs/optionalString" },
42
+ "thesis_title": { "$ref": "#/$defs/optionalString" },
43
+ "current_position": { "$ref": "#/$defs/optionalString" },
44
+ "bio": {
45
+ "description": "A short biography, as plain text: not Markdown or HTML. Line breaks are kept as written (SPEC.md §2).",
46
+ "$ref": "#/$defs/optionalString"
47
+ }
48
+ }
49
+ }
50
+ }
51
+ }
@@ -0,0 +1,34 @@
1
+ {
2
+ "$schema": "https://json-schema.org/draft/2020-12/schema",
3
+ "$id": "https://raw.githubusercontent.com/siddhss5/sslabdata/input-schema-v2/schema/input/v2/projects.schema.json",
4
+ "title": "sslabdata projects file",
5
+ "description": "The file lab.yaml names as projects_file: a list of research projects, one record each. It describes what sslabdata's loader accepts; the loader stays the authority, and reports each problem under a code in SPEC.md, Diagnostic codes. A record key not listed here is reported as RECORD-KEY-UNKNOWN and ignored, so it is allowed. What each field becomes in the document is in SPEC.md §5. Published input schemas are immutable: this one is served from the input-schema-v2 tag, which is created when this version ships and is never moved (SPEC.md §6).",
6
+ "type": ["array", "null"],
7
+ "items": { "$ref": "#/$defs/project" },
8
+ "$defs": {
9
+ "nonEmptyString": { "type": "string", "pattern": "\\S" },
10
+ "optionalString": { "type": ["string", "null"] },
11
+ "project": {
12
+ "type": "object",
13
+ "required": ["id", "title"],
14
+ "additionalProperties": true,
15
+ "properties": {
16
+ "id": {
17
+ "description": "The project's id, unique in the file (PROJECTS-ID-DUPLICATE). A work's project field names it.",
18
+ "$ref": "#/$defs/nonEmptyString"
19
+ },
20
+ "title": { "$ref": "#/$defs/nonEmptyString" },
21
+ "description": { "$ref": "#/$defs/optionalString" },
22
+ "website": { "$ref": "#/$defs/optionalString" },
23
+ "image": {
24
+ "description": "A URL or a site path, carried as plain text.",
25
+ "$ref": "#/$defs/optionalString"
26
+ },
27
+ "status": {
28
+ "description": "active when absent (PROJECTS-STATUS-INVALID).",
29
+ "enum": ["active", "completed"]
30
+ }
31
+ }
32
+ }
33
+ }
34
+ }