sslabdata 4.0.0__tar.gz → 5.0.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {sslabdata-4.0.0/sslabdata.egg-info → sslabdata-5.0.0}/PKG-INFO +20 -5
- {sslabdata-4.0.0 → sslabdata-5.0.0}/README.md +19 -4
- {sslabdata-4.0.0 → sslabdata-5.0.0}/SPEC.md +55 -23
- {sslabdata-4.0.0 → sslabdata-5.0.0}/pyproject.toml +4 -4
- sslabdata-5.0.0/schema/input/v2/collaborators.schema.json +24 -0
- sslabdata-5.0.0/schema/input/v2/lab.schema.json +71 -0
- sslabdata-5.0.0/schema/input/v2/people.schema.json +51 -0
- sslabdata-5.0.0/schema/input/v2/projects.schema.json +34 -0
- sslabdata-5.0.0/schema/v7/output.schema.json +485 -0
- {sslabdata-4.0.0 → sslabdata-5.0.0}/sslabdata/__init__.py +1 -1
- {sslabdata-4.0.0 → sslabdata-5.0.0}/sslabdata/loaders.py +3 -2
- {sslabdata-4.0.0 → sslabdata-5.0.0}/sslabdata/models.py +8 -2
- {sslabdata-4.0.0 → sslabdata-5.0.0}/sslabdata/templates/init/collaborators.yaml +1 -1
- {sslabdata-4.0.0 → sslabdata-5.0.0}/sslabdata/templates/init/lab.yaml +1 -1
- {sslabdata-4.0.0 → sslabdata-5.0.0}/sslabdata/templates/init/people.yaml +2 -1
- {sslabdata-4.0.0 → sslabdata-5.0.0}/sslabdata/templates/init/projects.yaml +1 -1
- {sslabdata-4.0.0 → sslabdata-5.0.0/sslabdata.egg-info}/PKG-INFO +20 -5
- {sslabdata-4.0.0 → sslabdata-5.0.0}/sslabdata.egg-info/SOURCES.txt +5 -0
- {sslabdata-4.0.0 → sslabdata-5.0.0}/LICENSE +0 -0
- {sslabdata-4.0.0 → sslabdata-5.0.0}/MANIFEST.in +0 -0
- {sslabdata-4.0.0 → sslabdata-5.0.0}/schema/__init__.py +0 -0
- {sslabdata-4.0.0 → sslabdata-5.0.0}/schema/input/v1/collaborators.schema.json +0 -0
- {sslabdata-4.0.0 → sslabdata-5.0.0}/schema/input/v1/lab.schema.json +0 -0
- {sslabdata-4.0.0 → sslabdata-5.0.0}/schema/input/v1/people.schema.json +0 -0
- {sslabdata-4.0.0 → sslabdata-5.0.0}/schema/input/v1/projects.schema.json +0 -0
- {sslabdata-4.0.0 → sslabdata-5.0.0}/schema/v3/output.schema.json +0 -0
- {sslabdata-4.0.0 → sslabdata-5.0.0}/schema/v4/output.schema.json +0 -0
- {sslabdata-4.0.0 → sslabdata-5.0.0}/schema/v5/output.schema.json +0 -0
- {sslabdata-4.0.0 → sslabdata-5.0.0}/schema/v6/output.schema.json +0 -0
- {sslabdata-4.0.0 → sslabdata-5.0.0}/setup.cfg +0 -0
- {sslabdata-4.0.0 → sslabdata-5.0.0}/sslabdata/assembler.py +0 -0
- {sslabdata-4.0.0 → sslabdata-5.0.0}/sslabdata/cli.py +0 -0
- {sslabdata-4.0.0 → sslabdata-5.0.0}/sslabdata/config.py +0 -0
- {sslabdata-4.0.0 → sslabdata-5.0.0}/sslabdata/diagnostics.py +0 -0
- {sslabdata-4.0.0 → sslabdata-5.0.0}/sslabdata/exporters.py +0 -0
- {sslabdata-4.0.0 → sslabdata-5.0.0}/sslabdata/parsers/__init__.py +0 -0
- {sslabdata-4.0.0 → sslabdata-5.0.0}/sslabdata/parsers/bibtex.py +0 -0
- {sslabdata-4.0.0 → sslabdata-5.0.0}/sslabdata/parsers/latex.py +0 -0
- {sslabdata-4.0.0 → sslabdata-5.0.0}/sslabdata/resolver.py +0 -0
- {sslabdata-4.0.0 → sslabdata-5.0.0}/sslabdata/templates/init/bib/publications.bib +0 -0
- {sslabdata-4.0.0 → sslabdata-5.0.0}/sslabdata.egg-info/dependency_links.txt +0 -0
- {sslabdata-4.0.0 → sslabdata-5.0.0}/sslabdata.egg-info/entry_points.txt +0 -0
- {sslabdata-4.0.0 → sslabdata-5.0.0}/sslabdata.egg-info/requires.txt +0 -0
- {sslabdata-4.0.0 → sslabdata-5.0.0}/sslabdata.egg-info/top_level.txt +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: sslabdata
|
|
3
|
-
Version:
|
|
3
|
+
Version: 5.0.0
|
|
4
4
|
Summary: Renderer-agnostic academic lab data assembler: BibTeX + YAML → structured data
|
|
5
5
|
Author: Siddhartha Srinivasa
|
|
6
6
|
License-Expression: MIT
|
|
@@ -255,6 +255,9 @@ author names to people:
|
|
|
255
255
|
website: "https://example.org/people/bbrown"
|
|
256
256
|
co_advisor: "Peggy Park"
|
|
257
257
|
start_year: 2021
|
|
258
|
+
bio: |
|
|
259
|
+
Bob Brown is a PhD student advised by Alice Adams and Peggy Park.
|
|
260
|
+
He works on shared control for assistive robot arms.
|
|
258
261
|
|
|
259
262
|
- id: "iingram"
|
|
260
263
|
name: "Ivan Ingram"
|
|
@@ -271,6 +274,16 @@ author names to people:
|
|
|
271
274
|
`id` and `name` are required. `role` is any non-empty string, so any lab's
|
|
272
275
|
roles fit; `status` is `current` (the default) or `alumni`.
|
|
273
276
|
|
|
277
|
+
`bio` is a short biography in plain text, not Markdown or HTML: it is
|
|
278
|
+
emitted as written and a renderer escapes it. Line breaks are kept as YAML
|
|
279
|
+
reads them, so a block scalar (`|`) keeps each line, and a renderer may treat
|
|
280
|
+
a blank line as a paragraph break. A person without one gets `bio: null`; an
|
|
281
|
+
empty string is emitted as written, as any other person field is. Keep the
|
|
282
|
+
structured facts in their own fields: nothing reads `degree`, years or
|
|
283
|
+
`current_position` out of a bio, so an alumni line such as "PhD 2022, now
|
|
284
|
+
Research Scientist at Example Robotics" is for the renderer to build from
|
|
285
|
+
those fields.
|
|
286
|
+
|
|
274
287
|
### External co-authors (optional, `data/collaborators.yaml`)
|
|
275
288
|
|
|
276
289
|
A list of co-authors outside the lab whose spellings you want grouped
|
|
@@ -306,24 +319,26 @@ deciding which URLs are safe to render is the renderer's job.
|
|
|
306
319
|
### Checking inputs in an editor
|
|
307
320
|
|
|
308
321
|
Each input file has a JSON Schema in
|
|
309
|
-
[`schema/input/
|
|
322
|
+
[`schema/input/v2/`](https://github.com/siddhss5/sslabdata/tree/main/schema/input/v2):
|
|
310
323
|
`lab.schema.json`, `people.schema.json`, `projects.schema.json` and
|
|
311
324
|
`collaborators.schema.json`. The wheel installs them under
|
|
312
|
-
`sslabdata/schema/input/
|
|
325
|
+
`sslabdata/schema/input/v2/`. An editor can use them to check and complete
|
|
313
326
|
the files as you write; `--validate` stays the check, and also reports what
|
|
314
327
|
a schema cannot see, such as a repeated id or a missing file. With the YAML
|
|
315
328
|
language server (the VS Code YAML extension, among others), name the schema
|
|
316
329
|
in a comment at the top of the file:
|
|
317
330
|
|
|
318
331
|
```yaml
|
|
319
|
-
# yaml-language-server: $schema=https://raw.githubusercontent.com/siddhss5/sslabdata/input-schema-
|
|
332
|
+
# yaml-language-server: $schema=https://raw.githubusercontent.com/siddhss5/sslabdata/input-schema-v2/schema/input/v2/people.schema.json
|
|
320
333
|
- id: "aadams"
|
|
321
334
|
name: "Alice Adams"
|
|
322
335
|
role: "professor"
|
|
323
336
|
```
|
|
324
337
|
|
|
325
|
-
The URL is the schema's `$id`, served from the `input-schema-
|
|
338
|
+
The URL is the schema's `$id`, served from the `input-schema-v2` tag, which is
|
|
326
339
|
never moved ([`SPEC.md` §6](https://github.com/siddhss5/sslabdata/blob/main/SPEC.md#6-version-policy)).
|
|
340
|
+
A file that names v1 still validates, because v1 allows keys it does not
|
|
341
|
+
list, but only v2 checks and completes `bio`.
|
|
327
342
|
Use `lab.schema.json`, `projects.schema.json` or `collaborators.schema.json`
|
|
328
343
|
in the same way for the other files.
|
|
329
344
|
|
|
@@ -218,6 +218,9 @@ author names to people:
|
|
|
218
218
|
website: "https://example.org/people/bbrown"
|
|
219
219
|
co_advisor: "Peggy Park"
|
|
220
220
|
start_year: 2021
|
|
221
|
+
bio: |
|
|
222
|
+
Bob Brown is a PhD student advised by Alice Adams and Peggy Park.
|
|
223
|
+
He works on shared control for assistive robot arms.
|
|
221
224
|
|
|
222
225
|
- id: "iingram"
|
|
223
226
|
name: "Ivan Ingram"
|
|
@@ -234,6 +237,16 @@ author names to people:
|
|
|
234
237
|
`id` and `name` are required. `role` is any non-empty string, so any lab's
|
|
235
238
|
roles fit; `status` is `current` (the default) or `alumni`.
|
|
236
239
|
|
|
240
|
+
`bio` is a short biography in plain text, not Markdown or HTML: it is
|
|
241
|
+
emitted as written and a renderer escapes it. Line breaks are kept as YAML
|
|
242
|
+
reads them, so a block scalar (`|`) keeps each line, and a renderer may treat
|
|
243
|
+
a blank line as a paragraph break. A person without one gets `bio: null`; an
|
|
244
|
+
empty string is emitted as written, as any other person field is. Keep the
|
|
245
|
+
structured facts in their own fields: nothing reads `degree`, years or
|
|
246
|
+
`current_position` out of a bio, so an alumni line such as "PhD 2022, now
|
|
247
|
+
Research Scientist at Example Robotics" is for the renderer to build from
|
|
248
|
+
those fields.
|
|
249
|
+
|
|
237
250
|
### External co-authors (optional, `data/collaborators.yaml`)
|
|
238
251
|
|
|
239
252
|
A list of co-authors outside the lab whose spellings you want grouped
|
|
@@ -269,24 +282,26 @@ deciding which URLs are safe to render is the renderer's job.
|
|
|
269
282
|
### Checking inputs in an editor
|
|
270
283
|
|
|
271
284
|
Each input file has a JSON Schema in
|
|
272
|
-
[`schema/input/
|
|
285
|
+
[`schema/input/v2/`](https://github.com/siddhss5/sslabdata/tree/main/schema/input/v2):
|
|
273
286
|
`lab.schema.json`, `people.schema.json`, `projects.schema.json` and
|
|
274
287
|
`collaborators.schema.json`. The wheel installs them under
|
|
275
|
-
`sslabdata/schema/input/
|
|
288
|
+
`sslabdata/schema/input/v2/`. An editor can use them to check and complete
|
|
276
289
|
the files as you write; `--validate` stays the check, and also reports what
|
|
277
290
|
a schema cannot see, such as a repeated id or a missing file. With the YAML
|
|
278
291
|
language server (the VS Code YAML extension, among others), name the schema
|
|
279
292
|
in a comment at the top of the file:
|
|
280
293
|
|
|
281
294
|
```yaml
|
|
282
|
-
# yaml-language-server: $schema=https://raw.githubusercontent.com/siddhss5/sslabdata/input-schema-
|
|
295
|
+
# yaml-language-server: $schema=https://raw.githubusercontent.com/siddhss5/sslabdata/input-schema-v2/schema/input/v2/people.schema.json
|
|
283
296
|
- id: "aadams"
|
|
284
297
|
name: "Alice Adams"
|
|
285
298
|
role: "professor"
|
|
286
299
|
```
|
|
287
300
|
|
|
288
|
-
The URL is the schema's `$id`, served from the `input-schema-
|
|
301
|
+
The URL is the schema's `$id`, served from the `input-schema-v2` tag, which is
|
|
289
302
|
never moved ([`SPEC.md` §6](https://github.com/siddhss5/sslabdata/blob/main/SPEC.md#6-version-policy)).
|
|
303
|
+
A file that names v1 still validates, because v1 allows keys it does not
|
|
304
|
+
list, but only v2 checks and completes `bio`.
|
|
290
305
|
Use `lab.schema.json`, `projects.schema.json` or `collaborators.schema.json`
|
|
291
306
|
in the same way for the other files.
|
|
292
307
|
|
|
@@ -7,7 +7,7 @@ them.
|
|
|
7
7
|
This file states the parts of that contract a JSON Schema cannot express:
|
|
8
8
|
what the strings in the document are, what order the lists are in, what an
|
|
9
9
|
absent key means, which fields are computed, when the version changes, and
|
|
10
|
-
how a repeated `@string` macro resolves. `schema/
|
|
10
|
+
how a repeated `@string` macro resolves. `schema/v7/output.schema.json`
|
|
11
11
|
states the rest.
|
|
12
12
|
|
|
13
13
|
Everything here is normative unless it carries a `Target` note. A `Target`
|
|
@@ -16,8 +16,8 @@ names the issue that will make it true. Until that issue lands, the rule is
|
|
|
16
16
|
the intent and the note is the fact. What changed at each release, and what
|
|
17
17
|
it replaced, is in [`CHANGELOG.md`](CHANGELOG.md), not here.
|
|
18
18
|
|
|
19
|
-
- Applies to: `schema_version`
|
|
20
|
-
version
|
|
19
|
+
- Applies to: `schema_version` 7 (`sslabdata.models.SCHEMA_VERSION`), package
|
|
20
|
+
version 5.0.0 (`sslabdata.__version__`).
|
|
21
21
|
|
|
22
22
|
### How this file cites the code
|
|
23
23
|
|
|
@@ -25,7 +25,7 @@ Every rule below is grounded in a named part of the code rather than a line
|
|
|
25
25
|
number, because line numbers rot silently: a function such as
|
|
26
26
|
`sslabdata.parsers.bibtex.parse_all_works()`, a method such as
|
|
27
27
|
`Person.to_dict()`, a module-level constant such as `TEXT_FIELDS`, a JSON
|
|
28
|
-
Pointer into `schema/
|
|
28
|
+
Pointer into `schema/v7/output.schema.json` such as `/$defs/person/required`, or
|
|
29
29
|
a `tests/COVERAGE.md` row key such as `config.people_file.missing`. A bare
|
|
30
30
|
statement is cited by its enclosing function.
|
|
31
31
|
|
|
@@ -275,7 +275,7 @@ Codes in use:
|
|
|
275
275
|
| `PROJECTS-ID-DUPLICATE` | Two projects declare one `id`. Located at the second. Both are kept. A validation error. |
|
|
276
276
|
| `PROJECTS-STATUS-INVALID` | A project's `status` is present and is not `active` or `completed`. A missing status reads as `active`, and so does one that is not a string. A warning. |
|
|
277
277
|
| `RECORD-KEY-UNKNOWN` | A person, project or collaborator record holds a key sslabdata does not read (`sslabdata.loaders.PERSON_KEYS`, `PROJECT_KEYS`, `COLLABORATOR_KEYS`), such as `hobby` or a misspelt `webiste`. One code for all three files. Located at `<people_file>:<id>:<key>`, `<projects_file>:<id>:<key>` or `<collaborators_file>:<collaborator name>:<key>`, once per key. The key is ignored and never emitted; the record is kept. Only a record that is loaded is checked, so a record missing a required field reports that alone. A warning. |
|
|
278
|
-
| `RECORD-TYPE-INVALID` | An optional field of a person, project or collaborator record has a value of the wrong type, other than the `role` and `status` that have codes of their own. A person's `photo`, `website`, `email`, `co_advisor`, `degree`, `thesis_title` and `
|
|
278
|
+
| `RECORD-TYPE-INVALID` | An optional field of a person, project or collaborator record has a value of the wrong type, other than the `role` and `status` that have codes of their own. A person's `photo`, `website`, `email`, `co_advisor`, `degree`, `thesis_title`, `current_position` and `bio`, and a project's `description`, `website` and `image`, are strings; a person's `start_year` and `end_year` are integers, not booleans; a person's or collaborator's `aliases` is a list of non-empty strings. One code for all three files, located at `<file>:<id>:<field>` (a collaborator's name for its id). The value is read as empty, so it is emitted as `null`, and a wrong `aliases` declares none; the record is kept. Only a record that is loaded is checked. A warning. |
|
|
279
279
|
| `RECORD-KEY-REPEATED` | A key is given twice in one mapping of a people, projects or collaborators file, which YAML alone would read as its last value, silently dropping the first. Every YAML file sslabdata reads is read with one loader that finds these (`sslabdata.config.YAMLLoader`). Keys are compared as YAML reads them, so `1` and `0x1` are one key. A merge key (`<<`) is not a repeat: the keys it merges in are overridden by the mapping's own, as YAML merge keys are defined. One code for all three files. Located at `<file>:<id>:<key>` (a collaborator's name for its id), with the path below the record's top level joined by `.` and a list member's index in brackets; the record is named by nothing when the repeated key is its `id` (a collaborator's `name`), and a repeat outside any record is located at `<file>::<path>`. The prose names the key and both lines. Every repeat is reported; the record is not loaded, and the run is fatal, as for a missing field: which value was meant is not sslabdata's to guess. |
|
|
280
280
|
| `CONFIG-NOT-A-MAPPING` | `lab.yaml` is not a mapping of keys, or is empty. Fatal at load. |
|
|
281
281
|
| `CONFIG-KEY-MISSING` | A required key is absent: `bib_dir`, or the `name` or `category` of a `bib_files` entry (`lab.yaml:bib_files:name`). Fatal at load. |
|
|
@@ -541,10 +541,20 @@ converts nor checks them:
|
|
|
541
541
|
- **The person and project strings supplied in YAML.**
|
|
542
542
|
`sslabdata.loaders.load_people()` and `load_projects()` perform no conversion
|
|
543
543
|
of any kind — they check the types of a record's fields, a person's `role` and that a
|
|
544
|
-
`status` is one they know, but emit every string as written — so a person's `name`, `role`, `current_position
|
|
545
|
-
`thesis_title`, and a project's `title` or `description`, are copied
|
|
544
|
+
`status` is one they know, but emit every string as written — so a person's `name`, `role`, `current_position`,
|
|
545
|
+
`thesis_title` or `bio`, and a project's `title` or `description`, are copied
|
|
546
546
|
straight from `people.yaml` and `projects.yaml`. Not every YAML string is
|
|
547
547
|
emitted — `aliases` and the configuration paths are not; see heading 3.
|
|
548
|
+
"As written" is the string YAML reads, after control characters are
|
|
549
|
+
removed and NFC is applied as for every input string (above), and nothing
|
|
550
|
+
else: an empty or whitespace-only string is emitted as such, not as
|
|
551
|
+
`null`, and line breaks are kept. That matters most for a person's `bio`,
|
|
552
|
+
the one of these meant to run over several lines: a literal block scalar
|
|
553
|
+
(`|`) keeps each line break and the final one, a folded one (`>`) joins
|
|
554
|
+
lines with a space and keeps a blank line as a line break, and `\n` in a
|
|
555
|
+
double-quoted string is a line break. It is plain text like the rest — not
|
|
556
|
+
Markdown and not HTML — and a renderer escapes it, and may read a blank
|
|
557
|
+
line as a paragraph break.
|
|
548
558
|
- **`work.category`**, which comes from the `category` of the `bib_files`
|
|
549
559
|
entry in `lab.yaml`, not from the `.bib` file
|
|
550
560
|
(`sslabdata.config.LabDataConfig.from_yaml()`, then
|
|
@@ -805,7 +815,7 @@ sslabdata's own output as input, and a wrong derivation becomes permanent.
|
|
|
805
815
|
| `author.resolution` | **Derived** — `status` over `resolved`, `unresolved` and `ambiguous`, and `method` `exact`, or `null` when nothing matched (`resolve_authors()`). Both are open strings; `fuzzy` is never emitted. |
|
|
806
816
|
| `author.equal_contribution` | **Derived** — whether the entry wrote a `*` marker on any part of the name (`sslabdata.parsers.bibtex.marks_equal_contribution()`). |
|
|
807
817
|
| `work.editors[*]` | The same, minus `collaborator_key` and `equal_contribution`. An editor that matched nobody is simply `person_id: null` (`parse_editor_list()`). |
|
|
808
|
-
| `person.*` except the two below | Input — the fields of `people_file` (`sslabdata.loaders.load_people()`). `aliases` is read for matching and is **not** emitted. `status` is `current` or `alumni`, and `current` when absent; `role` is open, any non-empty string (`PEOPLE-STATUS-INVALID`, `PEOPLE-ROLE-INVALID`). |
|
|
818
|
+
| `person.*` except the two below | Input — the fields of `people_file` (`sslabdata.loaders.load_people()`). `aliases` is read for matching and is **not** emitted. `status` is `current` or `alumni`, and `current` when absent; `role` is open, any non-empty string (`PEOPLE-STATUS-INVALID`, `PEOPLE-ROLE-INVALID`). `bio` is a short biography in plain text with its line breaks as written (§2), and `null` when absent; it is carried and never read: nothing in the document is derived from it, and no `degree`, year or `current_position` is read out of one. |
|
|
809
819
|
| `person.work_ids` | **Derived** — back-links over authorships (`sslabdata.resolver.compute_backlinks()`). Editors are not authorships and are not listed. |
|
|
810
820
|
| `project.id`, `title`, `description`, `website`, `image`, `status` | Input — the fields of `projects_file` (`sslabdata.loaders.load_projects()`). `status` is one of `active` and `completed`, and `active` when absent (`PROJECTS-STATUS-INVALID`). `image` is a URL or a site path, the same kind of value as a person's `photo`, carried as plain text: deciding which URLs are safe to render is the renderer's job. |
|
|
811
821
|
| `project.work_ids`, `people_ids` | **Derived** — back-links, and the people reached through them (`compute_backlinks()`). |
|
|
@@ -1120,14 +1130,14 @@ meaning and guarantees, and each of them is a bump.
|
|
|
1120
1130
|
has been published is never edited. Version `N`'s schema stays reachable, byte
|
|
1121
1131
|
for byte, at its own path after version `N+1` ships, so a consumer pinned to
|
|
1122
1132
|
`N` keeps a stable target. `schema/v3/output.schema.json`,
|
|
1123
|
-
`schema/v4/output.schema.json`, `schema/v5/output.schema.json
|
|
1124
|
-
`schema/v6/output.schema.json`
|
|
1125
|
-
`output.versioned_schema` asserts that the
|
|
1126
|
-
byte and still say `3`, `4` and `
|
|
1127
|
-
|
|
1128
|
-
**The `$id` is a pinned tag URL.**
|
|
1129
|
-
`https://raw.githubusercontent.com/siddhss5/sslabdata/schema-
|
|
1130
|
-
The rule that makes it a contract rather than a guess: **the `schema-
|
|
1133
|
+
`schema/v4/output.schema.json`, `schema/v5/output.schema.json`,
|
|
1134
|
+
`schema/v6/output.schema.json` and `schema/v7/output.schema.json` are those
|
|
1135
|
+
paths, and `tests/COVERAGE.md` row `output.versioned_schema` asserts that the
|
|
1136
|
+
older four are unchanged byte for byte and still say `3`, `4`, `5` and `6`.
|
|
1137
|
+
|
|
1138
|
+
**The `$id` is a pinned tag URL.** v7's `$id` is
|
|
1139
|
+
`https://raw.githubusercontent.com/siddhss5/sslabdata/schema-v7/schema/v7/output.schema.json`.
|
|
1140
|
+
The rule that makes it a contract rather than a guess: **the `schema-v7` tag
|
|
1131
1141
|
is created when this version ships and is never moved.** A branch URL such as
|
|
1132
1142
|
`blob/main` is not usable — it serves an HTML page rather than the schema, so
|
|
1133
1143
|
no consumer can ever have resolved v3's `$id` — and this repository publishes
|
|
@@ -1142,11 +1152,11 @@ rule.
|
|
|
1142
1152
|
descriptions inside those files, name the repository `labdata`, and the files
|
|
1143
1153
|
are left byte for byte as published rather than rewritten. v4's raw `$id`
|
|
1144
1154
|
resolves, because GitHub redirects `labdata` to `sslabdata`. v3's `blob/main`
|
|
1145
|
-
`$id` does not resolve, as above. The `$id`s, titles and descriptions of v5
|
|
1146
|
-
and
|
|
1155
|
+
`$id` does not resolve, as above. The `$id`s, titles and descriptions of v5,
|
|
1156
|
+
v6 and v7 say `sslabdata`.
|
|
1147
1157
|
|
|
1148
1158
|
**The input schemas are versioned apart from the document.**
|
|
1149
|
-
`schema/input/
|
|
1159
|
+
`schema/input/v2/` holds a JSON Schema for each input file: `lab.schema.json`,
|
|
1150
1160
|
`people.schema.json`, `projects.schema.json` and `collaborators.schema.json`.
|
|
1151
1161
|
They describe what the loaders accept; the loaders stay the authority, and
|
|
1152
1162
|
the diagnostic codes above report what a schema cannot see. The input format
|
|
@@ -1154,9 +1164,13 @@ and the document change for different reasons, so the input schemas carry
|
|
|
1154
1164
|
their own version rather than `schema_version`. They follow the same rules:
|
|
1155
1165
|
a published input schema is never edited, any change to one is a new
|
|
1156
1166
|
version at a new path, and its `$id` is a pinned tag URL,
|
|
1157
|
-
`https://raw.githubusercontent.com/siddhss5/sslabdata/input-schema-
|
|
1158
|
-
whose `input-schema-
|
|
1159
|
-
moved.
|
|
1167
|
+
`https://raw.githubusercontent.com/siddhss5/sslabdata/input-schema-v2/schema/input/v2/<file>.schema.json`,
|
|
1168
|
+
whose `input-schema-v2` tag is created when this version ships and is never
|
|
1169
|
+
moved. A version holds all four files, so a change to one of them versions
|
|
1170
|
+
the set: v2 adds a person's `bio` to `people.schema.json`, and its other three
|
|
1171
|
+
files differ from v1's only in the tag their `$id`s and descriptions name. `schema/input/v1/`, served
|
|
1172
|
+
from the `input-schema-v1` tag, stays byte for byte as published; the wheel
|
|
1173
|
+
installs only the current version.
|
|
1160
1174
|
|
|
1161
1175
|
**Version history.** What each `schema_version` changed is in
|
|
1162
1176
|
[`CHANGELOG.md`](CHANGELOG.md).
|
|
@@ -1256,6 +1270,24 @@ So each rejected type is rejected for a stated reason, and each has a home:
|
|
|
1256
1270
|
| Alumni | Not a collection — a `status` on a person (`sslabdata.models.Person.status`). |
|
|
1257
1271
|
| Robots, platforms, facilities | Your site repository; one of 27 surveyed sites had such a page. |
|
|
1258
1272
|
|
|
1273
|
+
**An attribute of an entity the document already has is not an entity.**
|
|
1274
|
+
The evidence above, and the question #60 asks of a candidate — would
|
|
1275
|
+
sslabdata do anything with it beyond carrying it — decide which *entities*
|
|
1276
|
+
the document has: collections with records of their own, ids, references
|
|
1277
|
+
and an order, each of which the compiler has to check and a renderer has to
|
|
1278
|
+
understand. A field
|
|
1279
|
+
on a person, work or project raises none of that. Its owner is already
|
|
1280
|
+
admitted, the field has one value of one type, and the only work it asks
|
|
1281
|
+
for is what every input string gets: control characters removed, NFC and
|
|
1282
|
+
the text rule (§2). A person already carries attributes sslabdata does
|
|
1283
|
+
nothing else with — `photo`, `website`, `thesis_title`, `current_position` —
|
|
1284
|
+
and a person's `bio` is one more. Such a field is judged on two questions
|
|
1285
|
+
instead: does it belong to that entity rather than to a page of the site,
|
|
1286
|
+
and is it the only place the fact can live, so that it does not copy a
|
|
1287
|
+
structured field into text that will drift from it. A bio passes both; an
|
|
1288
|
+
alumni line such as "PhD 2012, now at Example Robotics" fails the second,
|
|
1289
|
+
because `degree`, `end_year` and `current_position` already hold it.
|
|
1290
|
+
|
|
1259
1291
|
**There is no generic extension mechanism and no `collections` escape hatch.**
|
|
1260
1292
|
What one would carry is mostly prose, and its one real service — catching
|
|
1261
1293
|
references that point at nothing — is delivered by #58 without the document
|
|
@@ -1267,7 +1299,7 @@ vocabularies, not over arbitrary content.
|
|
|
1267
1299
|
|
|
1268
1300
|
## 9. What this file is not
|
|
1269
1301
|
|
|
1270
|
-
It does not list the document's fields; `schema/
|
|
1302
|
+
It does not list the document's fields; `schema/v7/output.schema.json` does.
|
|
1271
1303
|
It does not describe renderers such as
|
|
1272
1304
|
[sslabdata-site](https://github.com/siddhss5/sslabdata-site), which are
|
|
1273
1305
|
downstream consumers in their own repositories. It does not describe the input formats `lab.yaml`,
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "sslabdata"
|
|
3
|
-
version = "
|
|
3
|
+
version = "5.0.0"
|
|
4
4
|
description = "Renderer-agnostic academic lab data assembler: BibTeX + YAML → structured data"
|
|
5
5
|
readme = "README.md"
|
|
6
6
|
requires-python = ">=3.10"
|
|
@@ -58,8 +58,8 @@ requires = ["setuptools>=77"]
|
|
|
58
58
|
build-backend = "setuptools.build_meta"
|
|
59
59
|
|
|
60
60
|
# The current output schema and input schemas are public wheel data, installed
|
|
61
|
-
# at sslabdata/schema/
|
|
62
|
-
# sslabdata/schema/input/
|
|
61
|
+
# at sslabdata/schema/v7/output.schema.json and
|
|
62
|
+
# sslabdata/schema/input/v2/*.schema.json. The package directory is mapped onto
|
|
63
63
|
# the repository's schema/ directory so each frozen file has one copy and one
|
|
64
64
|
# path; earlier versions stay in the sdist only. schema/__init__.py makes the
|
|
65
65
|
# mapped directory a regular package, which editable installs need to resolve.
|
|
@@ -71,7 +71,7 @@ packages = ["sslabdata", "sslabdata.parsers", "sslabdata.schema"]
|
|
|
71
71
|
"sslabdata.schema" = "schema"
|
|
72
72
|
|
|
73
73
|
[tool.setuptools.package-data]
|
|
74
|
-
"sslabdata.schema" = ["
|
|
74
|
+
"sslabdata.schema" = ["v7/output.schema.json", "input/v2/*.schema.json"]
|
|
75
75
|
# The starting point `sslabdata init` copies.
|
|
76
76
|
"sslabdata" = ["templates/init/*.yaml", "templates/init/bib/*.bib"]
|
|
77
77
|
|
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
{
|
|
2
|
+
"$schema": "https://json-schema.org/draft/2020-12/schema",
|
|
3
|
+
"$id": "https://raw.githubusercontent.com/siddhss5/sslabdata/input-schema-v2/schema/input/v2/collaborators.schema.json",
|
|
4
|
+
"title": "sslabdata collaborators file",
|
|
5
|
+
"description": "The file lab.yaml names as collaborators_file: co-authors outside the lab whose spellings are grouped together, one record each. A record never makes anyone a lab member. It describes what sslabdata's loader accepts; the loader stays the authority, and reports each problem under a code in SPEC.md, Diagnostic codes. A record key not listed here is reported as RECORD-KEY-UNKNOWN and ignored, so it is allowed. Published input schemas are immutable: this one is served from the input-schema-v2 tag, which is created when this version ships and is never moved (SPEC.md §6).",
|
|
6
|
+
"type": ["array", "null"],
|
|
7
|
+
"items": { "$ref": "#/$defs/collaborator" },
|
|
8
|
+
"$defs": {
|
|
9
|
+
"nonEmptyString": { "type": "string", "pattern": "\\S" },
|
|
10
|
+
"collaborator": {
|
|
11
|
+
"type": "object",
|
|
12
|
+
"required": ["name"],
|
|
13
|
+
"additionalProperties": true,
|
|
14
|
+
"properties": {
|
|
15
|
+
"name": { "$ref": "#/$defs/nonEmptyString" },
|
|
16
|
+
"aliases": {
|
|
17
|
+
"description": "Other spellings of the name, grouped with it.",
|
|
18
|
+
"type": ["array", "null"],
|
|
19
|
+
"items": { "$ref": "#/$defs/nonEmptyString" }
|
|
20
|
+
}
|
|
21
|
+
}
|
|
22
|
+
}
|
|
23
|
+
}
|
|
24
|
+
}
|
|
@@ -0,0 +1,71 @@
|
|
|
1
|
+
{
|
|
2
|
+
"$schema": "https://json-schema.org/draft/2020-12/schema",
|
|
3
|
+
"$id": "https://raw.githubusercontent.com/siddhss5/sslabdata/input-schema-v2/schema/input/v2/lab.schema.json",
|
|
4
|
+
"title": "sslabdata lab.yaml",
|
|
5
|
+
"description": "The configuration file sslabdata is run with (--config). It describes what sslabdata's loader accepts; the loader stays the authority, and reports each problem under a code in SPEC.md, Diagnostic codes. A key not listed here is reported as CONFIG-KEY-UNKNOWN and ignored, so it is allowed. Published input schemas are immutable: this one is served from the input-schema-v2 tag, which is created when this version ships and is never moved (SPEC.md §6).",
|
|
6
|
+
"type": "object",
|
|
7
|
+
"required": ["bib_dir"],
|
|
8
|
+
"additionalProperties": true,
|
|
9
|
+
"properties": {
|
|
10
|
+
"lab": {
|
|
11
|
+
"description": "The lab header, copied into the document's lab unchanged (SPEC.md §5). The keys below are typed; any other key is allowed and copied through. sslabdata warns when it declares no name (CONFIG-LAB-NAME-MISSING).",
|
|
12
|
+
"type": ["object", "null"],
|
|
13
|
+
"additionalProperties": true,
|
|
14
|
+
"properties": {
|
|
15
|
+
"name": { "type": "string" },
|
|
16
|
+
"description": { "type": "string" },
|
|
17
|
+
"institution": { "type": "string" },
|
|
18
|
+
"department": { "type": "string" },
|
|
19
|
+
"website": { "type": "string" },
|
|
20
|
+
"email": { "type": "string" },
|
|
21
|
+
"address": { "type": "string" },
|
|
22
|
+
"logo": { "type": "string" },
|
|
23
|
+
"links": { "type": "object" }
|
|
24
|
+
}
|
|
25
|
+
},
|
|
26
|
+
"site": {
|
|
27
|
+
"description": "Read by renderers, not by sslabdata, and accepted without being checked."
|
|
28
|
+
},
|
|
29
|
+
"bib_dir": {
|
|
30
|
+
"description": "The directory the bib_files names are under, relative to the directory sslabdata is run from.",
|
|
31
|
+
"type": "string"
|
|
32
|
+
},
|
|
33
|
+
"bib_files": {
|
|
34
|
+
"description": "The .bib files to read. sslabdata warns when there are none (CONFIG-BIB-FILES-MISSING).",
|
|
35
|
+
"type": ["array", "null"],
|
|
36
|
+
"items": {
|
|
37
|
+
"type": "object",
|
|
38
|
+
"required": ["name", "category"],
|
|
39
|
+
"additionalProperties": false,
|
|
40
|
+
"properties": {
|
|
41
|
+
"name": {
|
|
42
|
+
"description": "A name under bib_dir, emitted as work.source.file: never absolute, and never leaving bib_dir (SPEC.md §5).",
|
|
43
|
+
"type": "string"
|
|
44
|
+
},
|
|
45
|
+
"category": {
|
|
46
|
+
"description": "Emitted as the category of every work in this file.",
|
|
47
|
+
"type": "string"
|
|
48
|
+
}
|
|
49
|
+
}
|
|
50
|
+
}
|
|
51
|
+
},
|
|
52
|
+
"pdf_base_url": {
|
|
53
|
+
"description": "A work with no pdf field gets this URL plus its citation key as its pdf link. Leave it out to guess no pdf links; an empty or blank value is an error.",
|
|
54
|
+
"type": ["string", "null"],
|
|
55
|
+
"minLength": 1,
|
|
56
|
+
"pattern": "\\S"
|
|
57
|
+
},
|
|
58
|
+
"people_file": {
|
|
59
|
+
"description": "The people file (people.schema.json), relative to the directory sslabdata is run from.",
|
|
60
|
+
"type": ["string", "null"]
|
|
61
|
+
},
|
|
62
|
+
"projects_file": {
|
|
63
|
+
"description": "The projects file (projects.schema.json), relative to the directory sslabdata is run from.",
|
|
64
|
+
"type": ["string", "null"]
|
|
65
|
+
},
|
|
66
|
+
"collaborators_file": {
|
|
67
|
+
"description": "The external co-authors file (collaborators.schema.json), relative to the directory sslabdata is run from.",
|
|
68
|
+
"type": ["string", "null"]
|
|
69
|
+
}
|
|
70
|
+
}
|
|
71
|
+
}
|
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
{
|
|
2
|
+
"$schema": "https://json-schema.org/draft/2020-12/schema",
|
|
3
|
+
"$id": "https://raw.githubusercontent.com/siddhss5/sslabdata/input-schema-v2/schema/input/v2/people.schema.json",
|
|
4
|
+
"title": "sslabdata people file",
|
|
5
|
+
"description": "The file lab.yaml names as people_file: a list of lab members and alumni, one record each. It describes what sslabdata's loader accepts; the loader stays the authority, and reports each problem under a code in SPEC.md, Diagnostic codes. A record key not listed here is reported as RECORD-KEY-UNKNOWN and ignored, so it is allowed. What each field becomes in the document is in SPEC.md §5. Published input schemas are immutable: this one is served from the input-schema-v2 tag, which is created when this version ships and is never moved (SPEC.md §6).",
|
|
6
|
+
"type": ["array", "null"],
|
|
7
|
+
"items": { "$ref": "#/$defs/person" },
|
|
8
|
+
"$defs": {
|
|
9
|
+
"nonEmptyString": { "type": "string", "pattern": "\\S" },
|
|
10
|
+
"optionalString": { "type": ["string", "null"] },
|
|
11
|
+
"optionalYear": { "type": ["integer", "null"] },
|
|
12
|
+
"person": {
|
|
13
|
+
"type": "object",
|
|
14
|
+
"required": ["id", "name", "role"],
|
|
15
|
+
"additionalProperties": true,
|
|
16
|
+
"properties": {
|
|
17
|
+
"id": {
|
|
18
|
+
"description": "The person's id, unique in the file (PEOPLE-ID-DUPLICATE).",
|
|
19
|
+
"$ref": "#/$defs/nonEmptyString"
|
|
20
|
+
},
|
|
21
|
+
"name": { "$ref": "#/$defs/nonEmptyString" },
|
|
22
|
+
"aliases": {
|
|
23
|
+
"description": "Other spellings of the name, used to match BibTeX authors and never emitted.",
|
|
24
|
+
"type": ["array", "null"],
|
|
25
|
+
"items": { "$ref": "#/$defs/nonEmptyString" }
|
|
26
|
+
},
|
|
27
|
+
"role": {
|
|
28
|
+
"description": "Any non-empty string; there is no list of roles (PEOPLE-ROLE-INVALID).",
|
|
29
|
+
"$ref": "#/$defs/nonEmptyString"
|
|
30
|
+
},
|
|
31
|
+
"status": {
|
|
32
|
+
"description": "current when absent (PEOPLE-STATUS-INVALID).",
|
|
33
|
+
"enum": ["current", "alumni"]
|
|
34
|
+
},
|
|
35
|
+
"photo": { "$ref": "#/$defs/optionalString" },
|
|
36
|
+
"website": { "$ref": "#/$defs/optionalString" },
|
|
37
|
+
"email": { "$ref": "#/$defs/optionalString" },
|
|
38
|
+
"co_advisor": { "$ref": "#/$defs/optionalString" },
|
|
39
|
+
"start_year": { "$ref": "#/$defs/optionalYear" },
|
|
40
|
+
"end_year": { "$ref": "#/$defs/optionalYear" },
|
|
41
|
+
"degree": { "$ref": "#/$defs/optionalString" },
|
|
42
|
+
"thesis_title": { "$ref": "#/$defs/optionalString" },
|
|
43
|
+
"current_position": { "$ref": "#/$defs/optionalString" },
|
|
44
|
+
"bio": {
|
|
45
|
+
"description": "A short biography, as plain text: not Markdown or HTML. Line breaks are kept as written (SPEC.md §2).",
|
|
46
|
+
"$ref": "#/$defs/optionalString"
|
|
47
|
+
}
|
|
48
|
+
}
|
|
49
|
+
}
|
|
50
|
+
}
|
|
51
|
+
}
|
|
@@ -0,0 +1,34 @@
|
|
|
1
|
+
{
|
|
2
|
+
"$schema": "https://json-schema.org/draft/2020-12/schema",
|
|
3
|
+
"$id": "https://raw.githubusercontent.com/siddhss5/sslabdata/input-schema-v2/schema/input/v2/projects.schema.json",
|
|
4
|
+
"title": "sslabdata projects file",
|
|
5
|
+
"description": "The file lab.yaml names as projects_file: a list of research projects, one record each. It describes what sslabdata's loader accepts; the loader stays the authority, and reports each problem under a code in SPEC.md, Diagnostic codes. A record key not listed here is reported as RECORD-KEY-UNKNOWN and ignored, so it is allowed. What each field becomes in the document is in SPEC.md §5. Published input schemas are immutable: this one is served from the input-schema-v2 tag, which is created when this version ships and is never moved (SPEC.md §6).",
|
|
6
|
+
"type": ["array", "null"],
|
|
7
|
+
"items": { "$ref": "#/$defs/project" },
|
|
8
|
+
"$defs": {
|
|
9
|
+
"nonEmptyString": { "type": "string", "pattern": "\\S" },
|
|
10
|
+
"optionalString": { "type": ["string", "null"] },
|
|
11
|
+
"project": {
|
|
12
|
+
"type": "object",
|
|
13
|
+
"required": ["id", "title"],
|
|
14
|
+
"additionalProperties": true,
|
|
15
|
+
"properties": {
|
|
16
|
+
"id": {
|
|
17
|
+
"description": "The project's id, unique in the file (PROJECTS-ID-DUPLICATE). A work's project field names it.",
|
|
18
|
+
"$ref": "#/$defs/nonEmptyString"
|
|
19
|
+
},
|
|
20
|
+
"title": { "$ref": "#/$defs/nonEmptyString" },
|
|
21
|
+
"description": { "$ref": "#/$defs/optionalString" },
|
|
22
|
+
"website": { "$ref": "#/$defs/optionalString" },
|
|
23
|
+
"image": {
|
|
24
|
+
"description": "A URL or a site path, carried as plain text.",
|
|
25
|
+
"$ref": "#/$defs/optionalString"
|
|
26
|
+
},
|
|
27
|
+
"status": {
|
|
28
|
+
"description": "active when absent (PROJECTS-STATUS-INVALID).",
|
|
29
|
+
"enum": ["active", "completed"]
|
|
30
|
+
}
|
|
31
|
+
}
|
|
32
|
+
}
|
|
33
|
+
}
|
|
34
|
+
}
|