colregs 0.2.4 → 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/PROVENANCE.md +3 -3
- package/README.md +11 -4
- package/data/corpora.json +41 -0
- package/data/editions.json +27 -0
- package/data/rules.json +171 -599
- package/data/text/intl/2016/en-US.uscg.json +877 -0
- package/data/text/intl/2016/es.boe.json +23 -0
- package/data/text/intl/2016/fi.finlex.json +22 -0
- package/data/text/us/inland/2014/en-US.ecfr.json +22 -0
- package/data/version.json +1 -1
- package/docs/adr/0001-name-and-jurisdiction-model.md +42 -2
- package/docs/adr/0003-language-as-a-dimension.md +8 -7
- package/docs/adr/0012-trace-and-rule2-departure-api.md +3 -4
- package/docs/adr/0013-corpus-files-with-editions.md +70 -0
- package/docs/budgets.json +13 -3
- package/docs/gates.json +3 -3
- package/docs/requirements.md +41 -29
- package/docs/timeline.md +102 -0
- package/docs/verification/2026-09-09-text-slug-straw-man.md +3 -2
- package/package.json +1 -1
- package/schema/corpora.schema.json +68 -0
- package/schema/corpus.schema.json +329 -0
- package/schema/editions.schema.json +47 -0
- package/schema/rules.schema.json +20 -90
|
@@ -0,0 +1,68 @@
|
|
|
1
|
+
{
|
|
2
|
+
"$schema": "https://json-schema.org/draft/2020-12/schema",
|
|
3
|
+
"$id": "https://github.com/mark-brannan/colregs/schema/corpora.schema.json",
|
|
4
|
+
"title": "data/corpora.json",
|
|
5
|
+
"description": "Index of the text corpora under data/text/: which exist, where each lives, and how many paragraphs it covers. Derived from the corpus files and checked against them by test/data.test.mjs; a consumer that cannot list a directory reads this instead. See docs/adr/0013-corpus-files-with-editions.md.",
|
|
6
|
+
"type": "object",
|
|
7
|
+
"additionalProperties": false,
|
|
8
|
+
"required": [
|
|
9
|
+
"corpora"
|
|
10
|
+
],
|
|
11
|
+
"properties": {
|
|
12
|
+
"note": {
|
|
13
|
+
"type": "string"
|
|
14
|
+
},
|
|
15
|
+
"corpora": {
|
|
16
|
+
"type": "object",
|
|
17
|
+
"additionalProperties": false,
|
|
18
|
+
"patternProperties": {
|
|
19
|
+
"^[a-z]+(/[a-z]+)*@[a-z0-9-]+\\.[A-Za-z0-9-]+\\.[a-z0-9]+$": {
|
|
20
|
+
"type": "object",
|
|
21
|
+
"additionalProperties": false,
|
|
22
|
+
"required": [
|
|
23
|
+
"file",
|
|
24
|
+
"edition",
|
|
25
|
+
"edition_status",
|
|
26
|
+
"language",
|
|
27
|
+
"source_id",
|
|
28
|
+
"tier",
|
|
29
|
+
"paragraphs"
|
|
30
|
+
],
|
|
31
|
+
"properties": {
|
|
32
|
+
"file": {
|
|
33
|
+
"type": "string",
|
|
34
|
+
"pattern": "^text/.+\\.json$"
|
|
35
|
+
},
|
|
36
|
+
"edition": {
|
|
37
|
+
"type": "string",
|
|
38
|
+
"pattern": "^[a-z]+(/[a-z]+)*@[a-z0-9-]+$"
|
|
39
|
+
},
|
|
40
|
+
"edition_status": {
|
|
41
|
+
"enum": ["verified", "claimed", "unknown"]
|
|
42
|
+
},
|
|
43
|
+
"language": {
|
|
44
|
+
"type": "string",
|
|
45
|
+
"minLength": 2
|
|
46
|
+
},
|
|
47
|
+
"source_id": {
|
|
48
|
+
"type": "string",
|
|
49
|
+
"pattern": "^[a-z0-9]+$"
|
|
50
|
+
},
|
|
51
|
+
"tier": {
|
|
52
|
+
"enum": [
|
|
53
|
+
"authentic",
|
|
54
|
+
"official",
|
|
55
|
+
"national",
|
|
56
|
+
"community"
|
|
57
|
+
]
|
|
58
|
+
},
|
|
59
|
+
"paragraphs": {
|
|
60
|
+
"type": "integer",
|
|
61
|
+
"minimum": 0
|
|
62
|
+
}
|
|
63
|
+
}
|
|
64
|
+
}
|
|
65
|
+
}
|
|
66
|
+
}
|
|
67
|
+
}
|
|
68
|
+
}
|
|
@@ -0,0 +1,329 @@
|
|
|
1
|
+
{
|
|
2
|
+
"$schema": "https://json-schema.org/draft/2020-12/schema",
|
|
3
|
+
"$id": "https://github.com/mark-brannan/colregs/schema/corpus.schema.json",
|
|
4
|
+
"title": "data/text/<jurisdiction>/<edition>/<language>.<source_id>.json",
|
|
5
|
+
"description": "One rule-text corpus: the words of one edition of one jurisdiction's rules, in one language, from one source, keyed by paragraph path into data/rules.json. Verbatim unless withheld under a licence bar. See docs/adr/0003-language-as-a-dimension.md, docs/adr/0010-text-withheld-jurisdictions.md and docs/adr/0013-corpus-files-with-editions.md.",
|
|
6
|
+
"type": "object",
|
|
7
|
+
"additionalProperties": false,
|
|
8
|
+
"required": [
|
|
9
|
+
"id",
|
|
10
|
+
"edition",
|
|
11
|
+
"edition_status",
|
|
12
|
+
"language",
|
|
13
|
+
"source_id",
|
|
14
|
+
"tier",
|
|
15
|
+
"normalization",
|
|
16
|
+
"source",
|
|
17
|
+
"rights",
|
|
18
|
+
"paragraphs"
|
|
19
|
+
],
|
|
20
|
+
"allOf": [
|
|
21
|
+
{
|
|
22
|
+
"$comment": "colregs#85 / ADR 0010: a withheld paragraph is a citation, not a reproduction -- 'retrieved' is when that citation was last checked against the primary source.",
|
|
23
|
+
"if": {
|
|
24
|
+
"properties": {
|
|
25
|
+
"paragraphs": {
|
|
26
|
+
"not": {
|
|
27
|
+
"type": "object",
|
|
28
|
+
"patternProperties": {
|
|
29
|
+
"^.*$": {
|
|
30
|
+
"not": {
|
|
31
|
+
"type": "object",
|
|
32
|
+
"properties": {
|
|
33
|
+
"text_status": {
|
|
34
|
+
"const": "withheld"
|
|
35
|
+
}
|
|
36
|
+
},
|
|
37
|
+
"required": [
|
|
38
|
+
"text_status"
|
|
39
|
+
]
|
|
40
|
+
}
|
|
41
|
+
}
|
|
42
|
+
}
|
|
43
|
+
}
|
|
44
|
+
}
|
|
45
|
+
}
|
|
46
|
+
},
|
|
47
|
+
"then": {
|
|
48
|
+
"properties": {
|
|
49
|
+
"source": {
|
|
50
|
+
"type": "object",
|
|
51
|
+
"properties": {
|
|
52
|
+
"retrieved": {
|
|
53
|
+
"type": "string"
|
|
54
|
+
}
|
|
55
|
+
},
|
|
56
|
+
"required": [
|
|
57
|
+
"retrieved"
|
|
58
|
+
]
|
|
59
|
+
}
|
|
60
|
+
}
|
|
61
|
+
}
|
|
62
|
+
}
|
|
63
|
+
],
|
|
64
|
+
"$defs": {
|
|
65
|
+
"path": {
|
|
66
|
+
"type": "string",
|
|
67
|
+
"pattern": "^[0-9]+(\\([a-z]\\))?(\\([ivx]+\\))?$"
|
|
68
|
+
},
|
|
69
|
+
"date": {
|
|
70
|
+
"type": "string",
|
|
71
|
+
"pattern": "^\\d{4}-\\d{2}-\\d{2}$"
|
|
72
|
+
}
|
|
73
|
+
},
|
|
74
|
+
"properties": {
|
|
75
|
+
"id": {
|
|
76
|
+
"$comment": "<edition>.<language>.<source_id>; CI checks it agrees with the file's path, with data/corpora.json, and that the edition is registered in data/editions.json.",
|
|
77
|
+
"type": "string",
|
|
78
|
+
"pattern": "^[a-z]+(/[a-z]+)*@[a-z0-9-]+\\.[A-Za-z0-9-]+\\.[a-z0-9]+$"
|
|
79
|
+
},
|
|
80
|
+
"edition": {
|
|
81
|
+
"$comment": "Edition id from data/editions.json (<jurisdiction>@<tag>): the consolidated state this text reflects -- REQ-LANG-10. The jurisdiction is the edition's, never restated here.",
|
|
82
|
+
"type": "string",
|
|
83
|
+
"pattern": "^[a-z]+(/[a-z]+)*@[a-z0-9-]+$"
|
|
84
|
+
},
|
|
85
|
+
"edition_status": {
|
|
86
|
+
"$comment": "Confidence in the `edition` claim above, not in the text itself: 'verified' means this corpus's edition was checked against the primary source; 'claimed' means asserted but not yet checked (every stub starts here); 'unknown' means even the claim is inherited/guessed. See ADR 0013.",
|
|
87
|
+
"enum": ["verified", "claimed", "unknown"]
|
|
88
|
+
},
|
|
89
|
+
"language": {
|
|
90
|
+
"$comment": "BCP 47. Carries the language of the text and nothing else -- REQ-LANG-1.",
|
|
91
|
+
"type": "string",
|
|
92
|
+
"pattern": "^[a-z]{2,3}(-[A-Z][a-z]{3})?(-[A-Z]{2}|-[0-9]{3})?$"
|
|
93
|
+
},
|
|
94
|
+
"source_id": {
|
|
95
|
+
"$comment": "Short handle for the publishing route (uscg, finlex, boe, ecfr); the identity of the source is the structured `source` object.",
|
|
96
|
+
"type": "string",
|
|
97
|
+
"pattern": "^[a-z0-9]+$"
|
|
98
|
+
},
|
|
99
|
+
"tier": {
|
|
100
|
+
"$comment": "Legal authority of the source, never of the language -- REQ-LANG-3.",
|
|
101
|
+
"enum": [
|
|
102
|
+
"authentic",
|
|
103
|
+
"official",
|
|
104
|
+
"national",
|
|
105
|
+
"community"
|
|
106
|
+
]
|
|
107
|
+
},
|
|
108
|
+
"normalization": {
|
|
109
|
+
"$comment": "Unicode normalization form the text is stored in -- REQ-LANG-9.",
|
|
110
|
+
"enum": [
|
|
111
|
+
"NFC",
|
|
112
|
+
"NFD",
|
|
113
|
+
"NFKC",
|
|
114
|
+
"NFKD"
|
|
115
|
+
]
|
|
116
|
+
},
|
|
117
|
+
"translation_of": {
|
|
118
|
+
"$comment": "Corpus id this text is a translation of, where it is one and that is known. Absent for independently promulgated national texts.",
|
|
119
|
+
"type": "string"
|
|
120
|
+
},
|
|
121
|
+
"source": {
|
|
122
|
+
"$comment": "REQ-PROV-6: structured source identity, not a prose string.",
|
|
123
|
+
"type": "object",
|
|
124
|
+
"additionalProperties": false,
|
|
125
|
+
"required": [
|
|
126
|
+
"publisher",
|
|
127
|
+
"title",
|
|
128
|
+
"url",
|
|
129
|
+
"retrieved"
|
|
130
|
+
],
|
|
131
|
+
"properties": {
|
|
132
|
+
"publisher": {
|
|
133
|
+
"type": "string",
|
|
134
|
+
"minLength": 1
|
|
135
|
+
},
|
|
136
|
+
"title": {
|
|
137
|
+
"type": "string",
|
|
138
|
+
"minLength": 1
|
|
139
|
+
},
|
|
140
|
+
"edition": {
|
|
141
|
+
"type": "string",
|
|
142
|
+
"minLength": 1
|
|
143
|
+
},
|
|
144
|
+
"published": {
|
|
145
|
+
"$ref": "#/$defs/date"
|
|
146
|
+
},
|
|
147
|
+
"effective": {
|
|
148
|
+
"$ref": "#/$defs/date"
|
|
149
|
+
},
|
|
150
|
+
"url": {
|
|
151
|
+
"type": "string",
|
|
152
|
+
"minLength": 1
|
|
153
|
+
},
|
|
154
|
+
"retrieved": {
|
|
155
|
+
"oneOf": [
|
|
156
|
+
{
|
|
157
|
+
"type": "null"
|
|
158
|
+
},
|
|
159
|
+
{
|
|
160
|
+
"$ref": "#/$defs/date"
|
|
161
|
+
}
|
|
162
|
+
]
|
|
163
|
+
}
|
|
164
|
+
}
|
|
165
|
+
},
|
|
166
|
+
"rights": {
|
|
167
|
+
"$comment": "REQ-PROV-6: three separate statements, never one flat licence field.",
|
|
168
|
+
"type": "object",
|
|
169
|
+
"additionalProperties": false,
|
|
170
|
+
"required": [
|
|
171
|
+
"source_text",
|
|
172
|
+
"redistribution_basis",
|
|
173
|
+
"package_licence"
|
|
174
|
+
],
|
|
175
|
+
"properties": {
|
|
176
|
+
"source_text": {
|
|
177
|
+
"type": "string",
|
|
178
|
+
"minLength": 1
|
|
179
|
+
},
|
|
180
|
+
"redistribution_basis": {
|
|
181
|
+
"type": "string",
|
|
182
|
+
"minLength": 1
|
|
183
|
+
},
|
|
184
|
+
"attribution": {
|
|
185
|
+
"type": "string",
|
|
186
|
+
"minLength": 1
|
|
187
|
+
},
|
|
188
|
+
"package_licence": {
|
|
189
|
+
"type": "string",
|
|
190
|
+
"minLength": 1
|
|
191
|
+
},
|
|
192
|
+
"contributors": {
|
|
193
|
+
"type": "array",
|
|
194
|
+
"items": {
|
|
195
|
+
"type": "string",
|
|
196
|
+
"minLength": 1
|
|
197
|
+
},
|
|
198
|
+
"minItems": 1
|
|
199
|
+
},
|
|
200
|
+
"reviewers": {
|
|
201
|
+
"type": "array",
|
|
202
|
+
"items": {
|
|
203
|
+
"type": "string",
|
|
204
|
+
"minLength": 1
|
|
205
|
+
},
|
|
206
|
+
"minItems": 1
|
|
207
|
+
}
|
|
208
|
+
}
|
|
209
|
+
},
|
|
210
|
+
"note": {
|
|
211
|
+
"type": "string"
|
|
212
|
+
},
|
|
213
|
+
"gaps": {
|
|
214
|
+
"$comment": "Skeleton paths this source could not supply. Distinct from withheld: a gap is text we could not obtain, withheld is text we may not publish.",
|
|
215
|
+
"type": "array",
|
|
216
|
+
"items": {
|
|
217
|
+
"type": "object",
|
|
218
|
+
"additionalProperties": false,
|
|
219
|
+
"required": [
|
|
220
|
+
"path",
|
|
221
|
+
"reason"
|
|
222
|
+
],
|
|
223
|
+
"properties": {
|
|
224
|
+
"path": {
|
|
225
|
+
"$ref": "#/$defs/path"
|
|
226
|
+
},
|
|
227
|
+
"reason": {
|
|
228
|
+
"type": "string",
|
|
229
|
+
"minLength": 1
|
|
230
|
+
}
|
|
231
|
+
}
|
|
232
|
+
}
|
|
233
|
+
},
|
|
234
|
+
"paragraphs": {
|
|
235
|
+
"type": "object",
|
|
236
|
+
"additionalProperties": false,
|
|
237
|
+
"patternProperties": {
|
|
238
|
+
"^[0-9]+(\\([a-z]\\))?(\\([ivx]+\\))?$": {
|
|
239
|
+
"type": "object",
|
|
240
|
+
"additionalProperties": false,
|
|
241
|
+
"required": [
|
|
242
|
+
"rule_title"
|
|
243
|
+
],
|
|
244
|
+
"allOf": [
|
|
245
|
+
{
|
|
246
|
+
"$comment": "ADR 0010: a withheld paragraph is modelled but ships no text.",
|
|
247
|
+
"if": {
|
|
248
|
+
"properties": {
|
|
249
|
+
"text_status": {
|
|
250
|
+
"const": "withheld"
|
|
251
|
+
}
|
|
252
|
+
},
|
|
253
|
+
"required": [
|
|
254
|
+
"text_status"
|
|
255
|
+
]
|
|
256
|
+
},
|
|
257
|
+
"then": {
|
|
258
|
+
"properties": {
|
|
259
|
+
"text": false,
|
|
260
|
+
"withheld_reason": {
|
|
261
|
+
"type": "string",
|
|
262
|
+
"minLength": 1
|
|
263
|
+
}
|
|
264
|
+
},
|
|
265
|
+
"required": [
|
|
266
|
+
"withheld_reason"
|
|
267
|
+
]
|
|
268
|
+
},
|
|
269
|
+
"else": {
|
|
270
|
+
"$comment": "mirrors, text_digest and text_slug are open to verbatim paragraphs too; only withheld_reason is incoherent without a bar.",
|
|
271
|
+
"properties": {
|
|
272
|
+
"text": {
|
|
273
|
+
"type": "string",
|
|
274
|
+
"minLength": 1
|
|
275
|
+
},
|
|
276
|
+
"withheld_reason": false
|
|
277
|
+
},
|
|
278
|
+
"required": [
|
|
279
|
+
"text"
|
|
280
|
+
]
|
|
281
|
+
}
|
|
282
|
+
}
|
|
283
|
+
],
|
|
284
|
+
"properties": {
|
|
285
|
+
"rule_title": {
|
|
286
|
+
"$comment": "Verbatim source material like `text`; on a withheld paragraph, a neutral label written for this package -- ADR 0010.",
|
|
287
|
+
"type": "string",
|
|
288
|
+
"minLength": 1
|
|
289
|
+
},
|
|
290
|
+
"text": {
|
|
291
|
+
"type": "string",
|
|
292
|
+
"minLength": 1
|
|
293
|
+
},
|
|
294
|
+
"text_status": {
|
|
295
|
+
"$comment": "Absent means verbatim. 'withheld' means the text is not ours to publish -- ADR 0010.",
|
|
296
|
+
"enum": [
|
|
297
|
+
"verbatim",
|
|
298
|
+
"withheld"
|
|
299
|
+
]
|
|
300
|
+
},
|
|
301
|
+
"withheld_reason": {
|
|
302
|
+
"type": "string",
|
|
303
|
+
"minLength": 1
|
|
304
|
+
},
|
|
305
|
+
"text_slug": {
|
|
306
|
+
"$comment": "Pencil, ADR 0010: sorted set of lemmatised key nouns and verbs, English counterparts where the source is not English.",
|
|
307
|
+
"type": "array",
|
|
308
|
+
"items": {
|
|
309
|
+
"type": "string",
|
|
310
|
+
"pattern": "^[a-z0-9]+(-[a-z0-9]+)*$"
|
|
311
|
+
},
|
|
312
|
+
"minItems": 1,
|
|
313
|
+
"uniqueItems": true
|
|
314
|
+
},
|
|
315
|
+
"text_digest": {
|
|
316
|
+
"$comment": "Pencil, ADR 0010: algorithm-qualified digest of the normalised text, e.g. 'sha256:<hex>'.",
|
|
317
|
+
"type": "string",
|
|
318
|
+
"pattern": "^[a-z0-9-]+:[A-Za-z0-9+/=]+$"
|
|
319
|
+
},
|
|
320
|
+
"mirrors": {
|
|
321
|
+
"$comment": "Paragraph path in intl this provision restates; display labelled as the international equivalent, never as this jurisdiction's text.",
|
|
322
|
+
"$ref": "#/$defs/path"
|
|
323
|
+
}
|
|
324
|
+
}
|
|
325
|
+
}
|
|
326
|
+
}
|
|
327
|
+
}
|
|
328
|
+
}
|
|
329
|
+
}
|
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
{
|
|
2
|
+
"$schema": "https://json-schema.org/draft/2020-12/schema",
|
|
3
|
+
"$id": "https://github.com/mark-brannan/colregs/schema/editions.schema.json",
|
|
4
|
+
"title": "data/editions.json",
|
|
5
|
+
"description": "Registry of instruments and their editions, per jurisdiction: the layer GATE-2 asked for. The skeleton names the edition it consolidates; every corpus names the edition its text reflects; two editions of one jurisdiction may be registered at once. See docs/adr/0013-corpus-files-with-editions.md.",
|
|
6
|
+
"type": "object",
|
|
7
|
+
"additionalProperties": false,
|
|
8
|
+
"required": ["jurisdictions"],
|
|
9
|
+
"properties": {
|
|
10
|
+
"note": { "type": "string" },
|
|
11
|
+
"jurisdictions": {
|
|
12
|
+
"type": "object",
|
|
13
|
+
"additionalProperties": false,
|
|
14
|
+
"patternProperties": {
|
|
15
|
+
"^[a-z]+(/[a-z]+)*$": {
|
|
16
|
+
"type": "object",
|
|
17
|
+
"additionalProperties": false,
|
|
18
|
+
"required": ["instrument", "skeleton", "editions"],
|
|
19
|
+
"properties": {
|
|
20
|
+
"instrument": { "$comment": "The body of rules this jurisdiction's editions are editions of.", "type": "string", "minLength": 1 },
|
|
21
|
+
"skeleton": {
|
|
22
|
+
"$comment": "Which edition data/rules.json consolidates for this jurisdiction; null while the skeleton has no paragraphs of it.",
|
|
23
|
+
"oneOf": [{ "type": "null" }, { "type": "string", "pattern": "^[a-z]+(/[a-z]+)*@[a-z0-9-]+$" }]
|
|
24
|
+
},
|
|
25
|
+
"editions": {
|
|
26
|
+
"type": "object",
|
|
27
|
+
"additionalProperties": false,
|
|
28
|
+
"patternProperties": {
|
|
29
|
+
"^[a-z]+(/[a-z]+)*@[a-z0-9-]+$": {
|
|
30
|
+
"type": "object",
|
|
31
|
+
"additionalProperties": false,
|
|
32
|
+
"required": ["amended_through", "in_force"],
|
|
33
|
+
"properties": {
|
|
34
|
+
"amended_through": { "$comment": "Free text: national amalgamations cite their own amending acts, not IMO resolutions.", "type": "string", "minLength": 1 },
|
|
35
|
+
"in_force": { "type": "string", "pattern": "^\\d{4}-\\d{2}-\\d{2}$" },
|
|
36
|
+
"superseded_by": { "$comment": "The later edition, where two are registered and one has lapsed.", "type": "string", "pattern": "^[a-z]+(/[a-z]+)*@[a-z0-9-]+$" },
|
|
37
|
+
"note": { "type": "string" }
|
|
38
|
+
}
|
|
39
|
+
}
|
|
40
|
+
}
|
|
41
|
+
}
|
|
42
|
+
}
|
|
43
|
+
}
|
|
44
|
+
}
|
|
45
|
+
}
|
|
46
|
+
}
|
|
47
|
+
}
|
package/schema/rules.schema.json
CHANGED
|
@@ -2,53 +2,15 @@
|
|
|
2
2
|
"$schema": "https://json-schema.org/draft/2020-12/schema",
|
|
3
3
|
"$id": "https://github.com/mark-brannan/colregs/schema/rules.schema.json",
|
|
4
4
|
"title": "data/rules.json",
|
|
5
|
-
"description": "
|
|
5
|
+
"description": "The language-neutral skeleton: paragraph paths, rule numbers, jurisdictions and figures. No text -- the words live in data/text/ corpora (schema/corpus.schema.json); the edition each jurisdiction consolidates is declared in data/editions.json. Structure only -- see docs/adr/0006-json-schema-and-identifier-diff.md, docs/adr/0003-language-as-a-dimension.md and docs/adr/0013-corpus-files-with-editions.md.",
|
|
6
6
|
"type": "object",
|
|
7
7
|
"additionalProperties": false,
|
|
8
|
-
"required": [
|
|
9
|
-
|
|
10
|
-
{
|
|
11
|
-
"$comment": "colregs#85: a withheld paragraph is a citation, not a reproduction (ADR 0010) -- 'retrieved' is when that citation was last checked against the primary source, so it can't be blank on a document that withholds text.",
|
|
12
|
-
"if": {
|
|
13
|
-
"properties": {
|
|
14
|
-
"paragraphs": {
|
|
15
|
-
"not": {
|
|
16
|
-
"type": "object",
|
|
17
|
-
"patternProperties": {
|
|
18
|
-
"^.*$": {
|
|
19
|
-
"not": {
|
|
20
|
-
"type": "object",
|
|
21
|
-
"properties": { "text_status": { "const": "withheld" } },
|
|
22
|
-
"required": ["text_status"]
|
|
23
|
-
}
|
|
24
|
-
}
|
|
25
|
-
}
|
|
26
|
-
}
|
|
27
|
-
}
|
|
28
|
-
}
|
|
29
|
-
},
|
|
30
|
-
"then": {
|
|
31
|
-
"properties": { "retrieved": { "type": "string" } },
|
|
32
|
-
"required": ["retrieved"]
|
|
33
|
-
}
|
|
34
|
-
}
|
|
8
|
+
"required": [
|
|
9
|
+
"paragraphs"
|
|
35
10
|
],
|
|
36
11
|
"properties": {
|
|
37
|
-
"
|
|
38
|
-
|
|
39
|
-
"retrieved": { "type": "string", "pattern": "^\\d{4}-\\d{2}-\\d{2}$" },
|
|
40
|
-
"note": { "type": "string" },
|
|
41
|
-
"gaps": {
|
|
42
|
-
"type": "array",
|
|
43
|
-
"items": {
|
|
44
|
-
"type": "object",
|
|
45
|
-
"additionalProperties": false,
|
|
46
|
-
"required": ["path", "reason"],
|
|
47
|
-
"properties": {
|
|
48
|
-
"path": { "type": "string", "minLength": 1 },
|
|
49
|
-
"reason": { "type": "string", "minLength": 1 }
|
|
50
|
-
}
|
|
51
|
-
}
|
|
12
|
+
"note": {
|
|
13
|
+
"type": "string"
|
|
52
14
|
},
|
|
53
15
|
"paragraphs": {
|
|
54
16
|
"type": "object",
|
|
@@ -57,62 +19,30 @@
|
|
|
57
19
|
"^[0-9]+(\\([a-z]\\))?(\\([ivx]+\\))?$": {
|
|
58
20
|
"type": "object",
|
|
59
21
|
"additionalProperties": false,
|
|
60
|
-
"required": [
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
"if": {
|
|
65
|
-
"properties": { "text_status": { "const": "withheld" } },
|
|
66
|
-
"required": ["text_status"]
|
|
67
|
-
},
|
|
68
|
-
"then": {
|
|
69
|
-
"properties": {
|
|
70
|
-
"text": false,
|
|
71
|
-
"withheld_reason": { "type": "string", "minLength": 1 }
|
|
72
|
-
},
|
|
73
|
-
"required": ["withheld_reason"]
|
|
74
|
-
},
|
|
75
|
-
"else": {
|
|
76
|
-
"$comment": "mirrors, text_digest and text_slug are open to verbatim paragraphs too; only withheld_reason is incoherent without a bar.",
|
|
77
|
-
"properties": {
|
|
78
|
-
"text": { "type": "string", "minLength": 1 },
|
|
79
|
-
"withheld_reason": false
|
|
80
|
-
},
|
|
81
|
-
"required": ["text"]
|
|
82
|
-
}
|
|
83
|
-
}
|
|
22
|
+
"required": [
|
|
23
|
+
"path",
|
|
24
|
+
"rule",
|
|
25
|
+
"jurisdiction"
|
|
84
26
|
],
|
|
85
27
|
"properties": {
|
|
86
|
-
"path": {
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
"jurisdiction": { "type": "string", "pattern": "^[a-z]+(/[a-z]+)*$" },
|
|
90
|
-
"text": { "type": "string", "minLength": 1 },
|
|
91
|
-
"text_status": {
|
|
92
|
-
"$comment": "Absent means verbatim. 'withheld' means the text is not ours to publish -- ADR 0010.",
|
|
93
|
-
"enum": ["verbatim", "withheld"]
|
|
94
|
-
},
|
|
95
|
-
"withheld_reason": { "type": "string", "minLength": 1 },
|
|
96
|
-
"text_slug": {
|
|
97
|
-
"$comment": "Pencil, ADR 0010: sorted set of lemmatised key nouns and verbs, English counterparts where the source is not English. Defined so the option is not walled off; nothing writes it yet.",
|
|
98
|
-
"type": "array",
|
|
99
|
-
"items": { "type": "string", "pattern": "^[a-z0-9]+(-[a-z0-9]+)*$" },
|
|
100
|
-
"minItems": 1,
|
|
101
|
-
"uniqueItems": true
|
|
28
|
+
"path": {
|
|
29
|
+
"type": "string",
|
|
30
|
+
"pattern": "^[0-9]+(\\([a-z]\\))?(\\([ivx]+\\))?$"
|
|
102
31
|
},
|
|
103
|
-
"
|
|
104
|
-
"$comment": "Pencil, ADR 0010: algorithm-qualified digest of the normalised text, e.g. 'sha256:<hex>'. Defined so the option is not walled off by additionalProperties; nothing writes it yet.",
|
|
32
|
+
"rule": {
|
|
105
33
|
"type": "string",
|
|
106
|
-
"pattern": "^[
|
|
34
|
+
"pattern": "^[0-9]+$"
|
|
107
35
|
},
|
|
108
|
-
"
|
|
109
|
-
"$comment": "Paragraph path in intl this provision restates; display labelled as the international equivalent, never as this jurisdiction's text.",
|
|
36
|
+
"jurisdiction": {
|
|
110
37
|
"type": "string",
|
|
111
|
-
"pattern": "^[
|
|
38
|
+
"pattern": "^[a-z]+(/[a-z]+)*$"
|
|
112
39
|
},
|
|
113
40
|
"images": {
|
|
114
41
|
"type": "array",
|
|
115
|
-
"items": {
|
|
42
|
+
"items": {
|
|
43
|
+
"type": "string",
|
|
44
|
+
"minLength": 1
|
|
45
|
+
},
|
|
116
46
|
"minItems": 1
|
|
117
47
|
}
|
|
118
48
|
}
|