@graphty/graph-io 0.3.16 → 0.3.18
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +1 -1
- package/README.md +29 -3
- package/dist/chunks/{escape-D1f9-cwf.js → escape-D-gZWO26.js} +3 -2
- package/dist/chunks/{escape-D1f9-cwf.js.map → escape-D-gZWO26.js.map} +1 -1
- package/dist/chunks/{importer-D7ZcGCeb.js → importer-Br_QeAeE.js} +4 -3
- package/dist/chunks/{importer-D7ZcGCeb.js.map → importer-Br_QeAeE.js.map} +1 -1
- package/dist/chunks/{importer-CXEiicAN.js → importer-DHagxvDD.js} +4 -3
- package/dist/chunks/{importer-CXEiicAN.js.map → importer-DHagxvDD.js.map} +1 -1
- package/dist/chunks/{importer-C7mnGdr_.js → importer-Du5crN9l.js} +510 -26
- package/dist/chunks/importer-Du5crN9l.js.map +1 -0
- package/dist/chunks/importer-aNJfe0qu.js +1614 -0
- package/dist/chunks/importer-aNJfe0qu.js.map +1 -0
- package/dist/chunks/{importer-B8lsjFWx.js → importer-d0uQxFp6.js} +4 -3
- package/dist/chunks/{importer-B8lsjFWx.js.map → importer-d0uQxFp6.js.map} +1 -1
- package/dist/chunks/ontology-BnrJ4I98.js +113 -0
- package/dist/chunks/ontology-BnrJ4I98.js.map +1 -0
- package/dist/chunks/{records-IHsCfv7s.js → records-Bk9jgodz.js} +2 -2
- package/dist/chunks/{records-IHsCfv7s.js.map → records-Bk9jgodz.js.map} +1 -1
- package/dist/chunks/{writer-BtWpUaiH.js → report-BOk0p5y8.js} +181 -912
- package/dist/chunks/report-BOk0p5y8.js.map +1 -0
- package/dist/chunks/writer-GAdltGmC.js +827 -0
- package/dist/chunks/writer-GAdltGmC.js.map +1 -0
- package/dist/csv.js +4 -3
- package/dist/csv.js.map +1 -1
- package/dist/dot.js +1 -1
- package/dist/gexf.js +3 -2
- package/dist/gexf.js.map +1 -1
- package/dist/gml.js +3 -2
- package/dist/gml.js.map +1 -1
- package/dist/graph-io.js +191 -134
- package/dist/graph-io.js.map +1 -1
- package/dist/graphml.js +1 -1
- package/dist/json.js +1 -1
- package/dist/neo4j.js +4 -3
- package/dist/neo4j.js.map +1 -1
- package/dist/obo.d.ts +1 -0
- package/dist/obo.js +6 -0
- package/dist/obo.js.map +1 -0
- package/dist/pajek.js +1 -1
- package/dist/src/common/codes.d.ts +43 -0
- package/dist/src/common/codes.d.ts.map +1 -1
- package/dist/src/common/codes.js +43 -0
- package/dist/src/common/codes.js.map +1 -1
- package/dist/src/common/input.d.ts +9 -0
- package/dist/src/common/input.d.ts.map +1 -1
- package/dist/src/common/input.js +16 -0
- package/dist/src/common/input.js.map +1 -1
- package/dist/src/common/ontology.d.ts +59 -0
- package/dist/src/common/ontology.d.ts.map +1 -0
- package/dist/src/common/ontology.js +147 -0
- package/dist/src/common/ontology.js.map +1 -0
- package/dist/src/common/options.d.ts +12 -1
- package/dist/src/common/options.d.ts.map +1 -1
- package/dist/src/common/options.js +51 -1
- package/dist/src/common/options.js.map +1 -1
- package/dist/src/formats/json/dialect.d.ts +8 -6
- package/dist/src/formats/json/dialect.d.ts.map +1 -1
- package/dist/src/formats/json/dialect.js +26 -3
- package/dist/src/formats/json/dialect.js.map +1 -1
- package/dist/src/formats/json/importer.d.ts +324 -7
- package/dist/src/formats/json/importer.d.ts.map +1 -1
- package/dist/src/formats/json/importer.js +154 -23
- package/dist/src/formats/json/importer.js.map +1 -1
- package/dist/src/formats/json/obographs.d.ts +21 -0
- package/dist/src/formats/json/obographs.d.ts.map +1 -0
- package/dist/src/formats/json/obographs.js +476 -0
- package/dist/src/formats/json/obographs.js.map +1 -0
- package/dist/src/formats/obo/importer.d.ts +90 -0
- package/dist/src/formats/obo/importer.d.ts.map +1 -0
- package/dist/src/formats/obo/importer.js +1248 -0
- package/dist/src/formats/obo/importer.js.map +1 -0
- package/dist/src/formats/obo/index.d.ts +7 -0
- package/dist/src/formats/obo/index.d.ts.map +1 -0
- package/dist/src/formats/obo/index.js +7 -0
- package/dist/src/formats/obo/index.js.map +1 -0
- package/dist/src/formats/obo/syntax.d.ts +121 -0
- package/dist/src/formats/obo/syntax.d.ts.map +1 -0
- package/dist/src/formats/obo/syntax.js +424 -0
- package/dist/src/formats/obo/syntax.js.map +1 -0
- package/dist/src/index.d.ts +5 -4
- package/dist/src/index.d.ts.map +1 -1
- package/dist/src/index.js +4 -3
- package/dist/src/index.js.map +1 -1
- package/dist/src/registry.d.ts +21 -3
- package/dist/src/registry.d.ts.map +1 -1
- package/dist/src/registry.js +32 -1
- package/dist/src/registry.js.map +1 -1
- package/dist/src/sniff.d.ts +1 -1
- package/dist/src/sniff.d.ts.map +1 -1
- package/dist/src/sniff.js +23 -3
- package/dist/src/sniff.js.map +1 -1
- package/dist/src/types.d.ts +31 -0
- package/dist/src/types.d.ts.map +1 -1
- package/dist/src/types.js.map +1 -1
- package/package.json +6 -1
- package/src/common/codes.ts +58 -0
- package/src/common/input.ts +16 -0
- package/src/common/ontology.ts +169 -0
- package/src/common/options.ts +78 -2
- package/src/formats/json/dialect.ts +37 -7
- package/src/formats/json/importer.ts +206 -28
- package/src/formats/json/obographs.ts +563 -0
- package/src/formats/obo/importer.ts +1695 -0
- package/src/formats/obo/index.ts +7 -0
- package/src/formats/obo/syntax.ts +466 -0
- package/src/index.ts +6 -0
- package/src/registry.ts +38 -3
- package/src/sniff.ts +35 -5
- package/src/types.ts +40 -1
- package/dist/chunks/importer-C7mnGdr_.js.map +0 -1
- package/dist/chunks/writer-BtWpUaiH.js.map +0 -1
|
@@ -0,0 +1,1695 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The OBO importer (design section 4; research-obo.md): the line-oriented OBO flat file format of
|
|
3
|
+
* the Gene Ontology and the OBO Foundry, versions 1.0, 1.2 and 1.4 read as their union (real files
|
|
4
|
+
* do not say which they follow: the GO writes `format-version: 1.2` and uses 1.4 tags), streamed
|
|
5
|
+
* through the common LineReader.
|
|
6
|
+
*
|
|
7
|
+
* `[Term]` and `[Instance]` frames are nodes; `is_a`, `relationship` and `instance_of` clauses are
|
|
8
|
+
* directed edges from the frame to the target, with the relation in a `relation` dict column (role
|
|
9
|
+
* `kind`) and the trailing qualifier block in a `qualifiers` json column. `[Typedef]` frames are
|
|
10
|
+
* metadata (`meta.extra.obo.typedefs`) unless `typedefs: "nodes"`. Every other tag fills the
|
|
11
|
+
* column of its name (src/common/ontology.ts); an unknown tag is kept in `obo.unrecognized`.
|
|
12
|
+
* Frames that share an id are merged (spec 4.1.1): list clauses take the union, a single-valued
|
|
13
|
+
* clause keeps the first. A target no frame declares becomes a placeholder node (the guide's
|
|
14
|
+
* recommended reading; ontologies point into their imports this way) unless `addMissingNodes` is
|
|
15
|
+
* false.
|
|
16
|
+
*
|
|
17
|
+
* ponytail: the frames are held until the end of the file (merging needs every frame of an id,
|
|
18
|
+
* dangling detection the whole id set), so memory is about twice the snapshot's; a two-pass reader
|
|
19
|
+
* would lift that for files beyond a few hundred megabytes.
|
|
20
|
+
*/
|
|
21
|
+
|
|
22
|
+
import {
|
|
23
|
+
type ColumnHandle,
|
|
24
|
+
GraphFormatError,
|
|
25
|
+
type GraphMetaPatch,
|
|
26
|
+
type GraphSink,
|
|
27
|
+
INVALID_INDEX,
|
|
28
|
+
type NodeId,
|
|
29
|
+
} from "@graphty/graph-format";
|
|
30
|
+
|
|
31
|
+
import { declareResolved, RENAMED_CODE, ROLE_TAKEN_CODE } from "../../common/attributes.js";
|
|
32
|
+
import {
|
|
33
|
+
BAD_VALUE_CODE,
|
|
34
|
+
DANGLING_REFERENCE_CODE,
|
|
35
|
+
DUPLICATE_ATTRIBUTE_CODE,
|
|
36
|
+
DUPLICATE_NODE_CODE,
|
|
37
|
+
EMPTY_INPUT_CODE,
|
|
38
|
+
ENCODING_FALLBACK_CODE,
|
|
39
|
+
INVALID_ENCODING_CODE,
|
|
40
|
+
INVALID_UTF8_CODE,
|
|
41
|
+
MISSING_ID_CODE,
|
|
42
|
+
OPTION_IGNORED_CODE,
|
|
43
|
+
UNKNOWN_ELEMENT_CODE,
|
|
44
|
+
UNKNOWN_ENCODING_CODE,
|
|
45
|
+
} from "../../common/codes.js";
|
|
46
|
+
import { DirectionResolver } from "../../common/direction.js";
|
|
47
|
+
import { ID_MERGED_CODE, IdCoercer } from "../../common/ids.js";
|
|
48
|
+
import { LineReader, throwIfAborted } from "../../common/input.js";
|
|
49
|
+
import { OBO_NODE_COLUMNS, oboColumnDecl, PLACEHOLDER_COLUMN, SYNONYM_SCOPES } from "../../common/ontology.js";
|
|
50
|
+
import {
|
|
51
|
+
type ImportFormatDefaults,
|
|
52
|
+
reportSinkOptions,
|
|
53
|
+
reportUnusedOptions,
|
|
54
|
+
resolveImportOptions,
|
|
55
|
+
SINK_OPTION_CODE,
|
|
56
|
+
} from "../../common/options.js";
|
|
57
|
+
import { ImportReportBuilder } from "../../common/report.js";
|
|
58
|
+
import { type CommonImportOptions, type GraphImporter, type ImportInput, type ImportReport } from "../../types.js";
|
|
59
|
+
import {
|
|
60
|
+
endsWithContinuation,
|
|
61
|
+
hasStrayBrace,
|
|
62
|
+
parseXref,
|
|
63
|
+
parseXrefList,
|
|
64
|
+
type Qualifiers,
|
|
65
|
+
splitQualifiers,
|
|
66
|
+
splitTagValue,
|
|
67
|
+
stripComment,
|
|
68
|
+
type Token,
|
|
69
|
+
tokenize,
|
|
70
|
+
unescapeObo,
|
|
71
|
+
type Xref,
|
|
72
|
+
} from "./syntax.js";
|
|
73
|
+
|
|
74
|
+
/** The format-specific options of the OBO importer. */
|
|
75
|
+
export interface OboImportOptions {
|
|
76
|
+
/** "keep" (default): obsolete terms are nodes with `is_obsolete` true; "drop": they and their edges are left out. */
|
|
77
|
+
obsolete?: "keep" | "drop" | undefined;
|
|
78
|
+
/** "metadata" (default): `[Typedef]` frames go to `meta.extra.obo.typedefs`; "nodes": they are nodes too, with their `is_a` edges. */
|
|
79
|
+
typedefs?: "metadata" | "nodes" | undefined;
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
/** Issue codes of the OBO importer (design section 4.3). */
|
|
83
|
+
export const OBO_ISSUE = Object.freeze({
|
|
84
|
+
/** The input is empty or whitespace (fatal). */
|
|
85
|
+
EMPTY_INPUT: EMPTY_INPUT_CODE,
|
|
86
|
+
/** The input holds invalid UTF-8 (fatal). */
|
|
87
|
+
INVALID_UTF8: INVALID_UTF8_CODE,
|
|
88
|
+
/** Invalid bytes in the encoding a BOM or the encoding option chose (fatal). */
|
|
89
|
+
INVALID_ENCODING: INVALID_ENCODING_CODE,
|
|
90
|
+
/** Bytes that are not UTF-8 were read as windows-1252. */
|
|
91
|
+
ENCODING_FALLBACK: ENCODING_FALLBACK_CODE,
|
|
92
|
+
/** A declared encoding the platform cannot decode was ignored. */
|
|
93
|
+
UNKNOWN_ENCODING: UNKNOWN_ENCODING_CODE,
|
|
94
|
+
/** A frame without an `id` clause; the frame is skipped. */
|
|
95
|
+
MISSING_ID: MISSING_ID_CODE,
|
|
96
|
+
/** A clause whose value cannot be read (a boolean other than true / false, a relationship with one or three values); the clause is skipped. */
|
|
97
|
+
BAD_VALUE: BAD_VALUE_CODE,
|
|
98
|
+
/** Two frames with one id were merged (spec 4.1.1). */
|
|
99
|
+
DUPLICATE_NODE: DUPLICATE_NODE_CODE,
|
|
100
|
+
/** A single-valued tag given twice for one id (a cardinality violation); the first is kept. */
|
|
101
|
+
DUPLICATE_ATTRIBUTE: DUPLICATE_ATTRIBUTE_CODE,
|
|
102
|
+
/** An unknown tag (kept in `obo.unrecognized`) or frame type (skipped), once per name. */
|
|
103
|
+
UNKNOWN_ELEMENT: UNKNOWN_ELEMENT_CODE,
|
|
104
|
+
/** A target no frame declares: a placeholder node was made, or the edge dropped under addMissingNodes false. */
|
|
105
|
+
DANGLING_REFERENCE: DANGLING_REFERENCE_CODE,
|
|
106
|
+
/** A line without a colon, an unterminated quote, a def without its xref list, a qualifier block that does not parse, an unescaped brace. */
|
|
107
|
+
SYNTAX: "W_OBO_SYNTAX",
|
|
108
|
+
/** The frame's `id` is not its first clause; it is used anyway. */
|
|
109
|
+
ID_NOT_FIRST: "W_OBO_ID_NOT_FIRST",
|
|
110
|
+
/** A synonym without a scope in a file that does not say 1.2, or with a scope that is not one of the four. */
|
|
111
|
+
SYNONYM_SCOPE: "W_OBO_SYNONYM_SCOPE",
|
|
112
|
+
/** A relation, subset or synonym type that nothing declares. */
|
|
113
|
+
UNDECLARED: "W_OBO_UNDECLARED",
|
|
114
|
+
/** One id for a Term and a Typedef (or an Instance); the Term (the first node frame) is the node. */
|
|
115
|
+
ID_KIND_CLASH: "W_OBO_ID_KIND_CLASH",
|
|
116
|
+
/** Fewer than two `intersection_of` or `union_of` clauses on a frame. */
|
|
117
|
+
CARDINALITY: "W_OBO_CARDINALITY",
|
|
118
|
+
/** An obsolete term with is_a / relationship, or replaced_by / consider on a term that is not obsolete. */
|
|
119
|
+
OBSOLETION: "W_OBO_OBSOLETION",
|
|
120
|
+
/** An OBO 1.0 / 1.2 tag read as its 1.4 meaning (exact_synonym, xref_analog, use_term, typeref, version). */
|
|
121
|
+
DEPRECATED_TAG: "W_OBO_DEPRECATED_TAG",
|
|
122
|
+
/** A backslash line continuation (deprecated in 1.4). */
|
|
123
|
+
DEPRECATED_SYNTAX: "W_OBO_DEPRECATED_SYNTAX",
|
|
124
|
+
/** A header clause kept in `meta.extra.obo.header` whose meaning is not applied (import, id-mapping, the treat-xrefs macros, owl-axioms). */
|
|
125
|
+
HEADER_NOT_APPLIED: "W_OBO_HEADER_NOT_APPLIED",
|
|
126
|
+
/** Obsolete terms and their edges left out under obsolete: "drop". */
|
|
127
|
+
OBSOLETE_DROPPED: "W_OBO_OBSOLETE_DROPPED",
|
|
128
|
+
/** Two distinct id texts became one number under ids "number". */
|
|
129
|
+
ID_MERGED: ID_MERGED_CODE,
|
|
130
|
+
/** A vocabulary column renamed `<name>#obo` because the sink already holds the name. */
|
|
131
|
+
COLUMN_RENAMED: RENAMED_CODE,
|
|
132
|
+
/** A vocabulary column declared without its role because the sink already holds it. */
|
|
133
|
+
ROLE_TAKEN: ROLE_TAKEN_CODE,
|
|
134
|
+
/** A common option the format has no use for (defaultDirected, weightFrom, nodeIdFrom, ...). */
|
|
135
|
+
OPTION_IGNORED: OPTION_IGNORED_CODE,
|
|
136
|
+
/** A builder-policy option the caller passed that the caller's sink does not use. */
|
|
137
|
+
SINK_OPTION: SINK_OPTION_CODE,
|
|
138
|
+
});
|
|
139
|
+
|
|
140
|
+
/** The common options the OBO importer reads. */
|
|
141
|
+
const USED_OPTIONS: ReadonlySet<keyof CommonImportOptions> = new Set<keyof CommonImportOptions>([
|
|
142
|
+
"ids",
|
|
143
|
+
"addMissingNodes",
|
|
144
|
+
"duplicateEdges",
|
|
145
|
+
"selfLoops",
|
|
146
|
+
"onMixedDirection",
|
|
147
|
+
"weightDtype",
|
|
148
|
+
"errorLimit",
|
|
149
|
+
"signal",
|
|
150
|
+
"onProgress",
|
|
151
|
+
"encoding",
|
|
152
|
+
]);
|
|
153
|
+
|
|
154
|
+
const DEFAULTS: ImportFormatDefaults = { ids: "keep", defaultDirected: true, weightFrom: null };
|
|
155
|
+
|
|
156
|
+
/** Frames between two checks of the cancellation signal. */
|
|
157
|
+
const ABORT_CHECK_INTERVAL = 64;
|
|
158
|
+
|
|
159
|
+
/** Bytes of the head sniff() reads. */
|
|
160
|
+
const SNIFF_BYTES = 4096;
|
|
161
|
+
|
|
162
|
+
const FRAME_KINDS: ReadonlySet<string> = new Set(["Term", "Typedef", "Instance"]);
|
|
163
|
+
|
|
164
|
+
/** Header tags kept but not applied (design 4.3, W_OBO_HEADER_NOT_APPLIED). */
|
|
165
|
+
const NOT_APPLIED_HEADER = /^(import|typeref|id-mapping|default-relationship-id-prefix|owl-axioms|treat-xrefs-as-.*)$/;
|
|
166
|
+
|
|
167
|
+
/** Deprecated header tags and what they are now. */
|
|
168
|
+
const DEPRECATED_HEADER: ReadonlyMap<string, string> = new Map([
|
|
169
|
+
["typeref", "import"],
|
|
170
|
+
["version", "data-version"],
|
|
171
|
+
]);
|
|
172
|
+
|
|
173
|
+
/** The 1.0 / 1.2 frame tags read as their 1.4 meaning: tag to [1.4 tag, synonym scope]. */
|
|
174
|
+
const DEPRECATED_TAGS: ReadonlyMap<string, readonly [string, string | null]> = new Map([
|
|
175
|
+
["exact_synonym", ["synonym", "EXACT"]],
|
|
176
|
+
["narrow_synonym", ["synonym", "NARROW"]],
|
|
177
|
+
["broad_synonym", ["synonym", "BROAD"]],
|
|
178
|
+
["related_synonym", ["synonym", "RELATED"]],
|
|
179
|
+
["xref_analog", ["xref", null]],
|
|
180
|
+
["xref_unk", ["xref", null]],
|
|
181
|
+
["xref_unknown", ["xref", null]],
|
|
182
|
+
["use_term", ["consider", null]],
|
|
183
|
+
]);
|
|
184
|
+
|
|
185
|
+
/** Relations every OBO file may use without a Typedef (the 1.2 built-ins). */
|
|
186
|
+
const BUILTIN_RELATIONS: ReadonlySet<string> = new Set([
|
|
187
|
+
"is_a",
|
|
188
|
+
"instance_of",
|
|
189
|
+
"disjoint_from",
|
|
190
|
+
"inverse_of",
|
|
191
|
+
"union_of",
|
|
192
|
+
"intersection_of",
|
|
193
|
+
]);
|
|
194
|
+
|
|
195
|
+
/** Single-valued text tags (first value kept). */
|
|
196
|
+
const TEXT_TAGS: ReadonlySet<string> = new Set([
|
|
197
|
+
"name",
|
|
198
|
+
"namespace",
|
|
199
|
+
"comment",
|
|
200
|
+
"created_by",
|
|
201
|
+
"creation_date",
|
|
202
|
+
"domain",
|
|
203
|
+
"range",
|
|
204
|
+
"inverse_of",
|
|
205
|
+
]);
|
|
206
|
+
|
|
207
|
+
/** Single-valued boolean tags. */
|
|
208
|
+
const BOOL_TAGS: ReadonlySet<string> = new Set(
|
|
209
|
+
Object.keys(OBO_NODE_COLUMNS).filter((name) => OBO_NODE_COLUMNS[name].dtype === "bool"),
|
|
210
|
+
);
|
|
211
|
+
|
|
212
|
+
/** Tags whose value is one id, collected in a list column. */
|
|
213
|
+
const ID_LIST_TAGS: ReadonlySet<string> = new Set([
|
|
214
|
+
"alt_id",
|
|
215
|
+
"subset",
|
|
216
|
+
"replaced_by",
|
|
217
|
+
"consider",
|
|
218
|
+
"union_of",
|
|
219
|
+
"equivalent_to",
|
|
220
|
+
"disjoint_from",
|
|
221
|
+
"transitive_over",
|
|
222
|
+
"disjoint_over",
|
|
223
|
+
]);
|
|
224
|
+
|
|
225
|
+
/** Tags that only a Typedef frame may carry (elsewhere they are unrecognized). */
|
|
226
|
+
const TYPEDEF_TAGS: ReadonlySet<string> = new Set([
|
|
227
|
+
"domain",
|
|
228
|
+
"range",
|
|
229
|
+
"inverse_of",
|
|
230
|
+
"transitive_over",
|
|
231
|
+
"disjoint_over",
|
|
232
|
+
"holds_over_chain",
|
|
233
|
+
"equivalent_to_chain",
|
|
234
|
+
"expand_assertion_to",
|
|
235
|
+
"expand_expression_to",
|
|
236
|
+
"is_cyclic",
|
|
237
|
+
"is_reflexive",
|
|
238
|
+
"is_symmetric",
|
|
239
|
+
"is_anti_symmetric",
|
|
240
|
+
"is_asymmetric",
|
|
241
|
+
"is_transitive",
|
|
242
|
+
"is_functional",
|
|
243
|
+
"is_inverse_functional",
|
|
244
|
+
"is_metadata_tag",
|
|
245
|
+
"is_class_level",
|
|
246
|
+
]);
|
|
247
|
+
|
|
248
|
+
type FrameKind = "Term" | "Typedef" | "Instance";
|
|
249
|
+
|
|
250
|
+
/** One clause as read: the tag and the raw value without its comment. */
|
|
251
|
+
interface Clause {
|
|
252
|
+
readonly tag: string;
|
|
253
|
+
readonly value: string;
|
|
254
|
+
readonly line: number;
|
|
255
|
+
}
|
|
256
|
+
|
|
257
|
+
/** An edge held until the end of the file. */
|
|
258
|
+
interface PendingEdge {
|
|
259
|
+
readonly relation: string;
|
|
260
|
+
readonly target: string;
|
|
261
|
+
readonly qualifiers: Qualifiers | null;
|
|
262
|
+
readonly line: number;
|
|
263
|
+
}
|
|
264
|
+
|
|
265
|
+
/** Everything read for one id, over all of its frames. */
|
|
266
|
+
interface NodeRecord {
|
|
267
|
+
readonly id: string;
|
|
268
|
+
readonly kind: FrameKind;
|
|
269
|
+
readonly line: number;
|
|
270
|
+
/** Single values and whole-cell values by column. */
|
|
271
|
+
readonly values: Map<string, unknown>;
|
|
272
|
+
/** List and json-list cells by column. */
|
|
273
|
+
readonly lists: Map<string, unknown[]>;
|
|
274
|
+
readonly edges: PendingEdge[];
|
|
275
|
+
/** The clauses already applied (spec 4.1.1: an identical clause counts once). */
|
|
276
|
+
readonly seen: Set<string>;
|
|
277
|
+
/** Typedef frames: every clause by tag, raw, for `meta.extra.obo.typedefs`. */
|
|
278
|
+
readonly raw: Record<string, string[]>;
|
|
279
|
+
}
|
|
280
|
+
|
|
281
|
+
/** Where a clause is: its line and the id of its frame. */
|
|
282
|
+
interface Where {
|
|
283
|
+
readonly line: number;
|
|
284
|
+
readonly element: string;
|
|
285
|
+
}
|
|
286
|
+
|
|
287
|
+
/** A warning counted while reading and recorded once at the end. */
|
|
288
|
+
interface Tally {
|
|
289
|
+
readonly category: "parse-error" | "validation-error" | "coercion" | "unsupported";
|
|
290
|
+
readonly code: string;
|
|
291
|
+
readonly element: string;
|
|
292
|
+
readonly what: string;
|
|
293
|
+
readonly line: number;
|
|
294
|
+
count: number;
|
|
295
|
+
}
|
|
296
|
+
|
|
297
|
+
/** The OBO-specific options, resolved. */
|
|
298
|
+
interface ResolvedOboOptions {
|
|
299
|
+
readonly obsolete: "keep" | "drop";
|
|
300
|
+
readonly typedefs: "metadata" | "nodes";
|
|
301
|
+
}
|
|
302
|
+
|
|
303
|
+
/**
|
|
304
|
+
* Resolve the OBO-specific options.
|
|
305
|
+
* @param options - the caller's options
|
|
306
|
+
* @returns the resolved options; E_UNSUPPORTED for a value outside its set
|
|
307
|
+
*/
|
|
308
|
+
function resolveOboOptions(options: OboImportOptions | undefined): ResolvedOboOptions {
|
|
309
|
+
const obsolete = options?.obsolete ?? "keep";
|
|
310
|
+
const typedefs = options?.typedefs ?? "metadata";
|
|
311
|
+
if (obsolete !== "keep" && obsolete !== "drop") {
|
|
312
|
+
throw unsupported("obsolete", obsolete, ["keep", "drop"]);
|
|
313
|
+
}
|
|
314
|
+
if (typedefs !== "metadata" && typedefs !== "nodes") {
|
|
315
|
+
throw unsupported("typedefs", typedefs, ["metadata", "nodes"]);
|
|
316
|
+
}
|
|
317
|
+
return { obsolete, typedefs };
|
|
318
|
+
}
|
|
319
|
+
|
|
320
|
+
/**
|
|
321
|
+
* The E_UNSUPPORTED error of an option outside its set.
|
|
322
|
+
* @param option - the option name
|
|
323
|
+
* @param found - the value
|
|
324
|
+
* @param supported - the allowed values
|
|
325
|
+
* @returns the error
|
|
326
|
+
*/
|
|
327
|
+
function unsupported(option: string, found: unknown, supported: readonly string[]): Error {
|
|
328
|
+
return new GraphFormatError(
|
|
329
|
+
"E_UNSUPPORTED",
|
|
330
|
+
`option ${option}: ${JSON.stringify(found)} is not one of ${supported.join(", ")}`,
|
|
331
|
+
{
|
|
332
|
+
option,
|
|
333
|
+
found,
|
|
334
|
+
supported,
|
|
335
|
+
},
|
|
336
|
+
);
|
|
337
|
+
}
|
|
338
|
+
|
|
339
|
+
/** One import's state. */
|
|
340
|
+
class OboReader {
|
|
341
|
+
private readonly report: ImportReportBuilder;
|
|
342
|
+
|
|
343
|
+
private readonly signal: AbortSignal | null;
|
|
344
|
+
|
|
345
|
+
/** Header clauses by tag, raw, in order. */
|
|
346
|
+
private readonly header: Record<string, string[]> = {};
|
|
347
|
+
|
|
348
|
+
private version: string | null = null;
|
|
349
|
+
|
|
350
|
+
private defaultNamespace: string | null = null;
|
|
351
|
+
|
|
352
|
+
private readonly subsets = new Set<string>();
|
|
353
|
+
|
|
354
|
+
/** Declared synonym types and their default scope (null: none). */
|
|
355
|
+
private readonly synonymTypes = new Map<string, string | null>();
|
|
356
|
+
|
|
357
|
+
/** Node frames (Term, Instance) by id, in first-appearance order. */
|
|
358
|
+
readonly nodes = new Map<string, NodeRecord>();
|
|
359
|
+
|
|
360
|
+
/** Typedef frames by id. */
|
|
361
|
+
readonly typedefs = new Map<string, NodeRecord>();
|
|
362
|
+
|
|
363
|
+
private readonly tallies = new Map<string, Tally>();
|
|
364
|
+
|
|
365
|
+
/** The frame being read: its kind (null for an unknown frame type), name and clauses. */
|
|
366
|
+
private frame: { kind: FrameKind | null; name: string; line: number; clauses: Clause[] } | null = null;
|
|
367
|
+
|
|
368
|
+
/** The frames of an unknown type, raw, for `meta.extra.obo.unknownFrames`. */
|
|
369
|
+
readonly unknownFrames: { type: string; clauses: Record<string, string[]> }[] = [];
|
|
370
|
+
|
|
371
|
+
private framesSinceCheck = 0;
|
|
372
|
+
|
|
373
|
+
/** Whether any significant line was read. */
|
|
374
|
+
significant = false;
|
|
375
|
+
|
|
376
|
+
/**
|
|
377
|
+
* Create a reader.
|
|
378
|
+
* @param report - the report
|
|
379
|
+
* @param signal - the cancellation signal
|
|
380
|
+
*/
|
|
381
|
+
constructor(report: ImportReportBuilder, signal: AbortSignal | null) {
|
|
382
|
+
this.report = report;
|
|
383
|
+
this.signal = signal;
|
|
384
|
+
}
|
|
385
|
+
|
|
386
|
+
/**
|
|
387
|
+
* Read one logical line.
|
|
388
|
+
* @param text - the line, continuation joined
|
|
389
|
+
* @param line - its 1-based number
|
|
390
|
+
*/
|
|
391
|
+
line(text: string, line: number): void {
|
|
392
|
+
const t = text.trim();
|
|
393
|
+
if (t.length === 0) {
|
|
394
|
+
return;
|
|
395
|
+
}
|
|
396
|
+
this.significant = true;
|
|
397
|
+
if (t.startsWith("!")) {
|
|
398
|
+
return;
|
|
399
|
+
}
|
|
400
|
+
const header = /^\[([^\]]*)\]\s*(?:!.*)?$/.exec(t);
|
|
401
|
+
if (header !== null) {
|
|
402
|
+
this.startFrame(header[1].trim(), line);
|
|
403
|
+
return;
|
|
404
|
+
}
|
|
405
|
+
// the comment goes first, so a colon inside it never makes `foo ! a:b` a tag-value line
|
|
406
|
+
const tv = splitTagValue(stripComment(t));
|
|
407
|
+
if (tv === null) {
|
|
408
|
+
this.tally("parse-error", OBO_ISSUE.SYNTAX, "a line without a colon was skipped", "line", line);
|
|
409
|
+
return;
|
|
410
|
+
}
|
|
411
|
+
const clause: Clause = { tag: tv.tag, value: stripComment(tv.rest), line };
|
|
412
|
+
if (this.frame === null) {
|
|
413
|
+
this.headerClause(clause);
|
|
414
|
+
} else {
|
|
415
|
+
this.frame.clauses.push(clause);
|
|
416
|
+
}
|
|
417
|
+
}
|
|
418
|
+
|
|
419
|
+
/**
|
|
420
|
+
* Start a frame, finishing the one before.
|
|
421
|
+
* @param name - the frame type
|
|
422
|
+
* @param line - the header's line
|
|
423
|
+
*/
|
|
424
|
+
private startFrame(name: string, line: number): void {
|
|
425
|
+
this.finishFrame();
|
|
426
|
+
const kind = FRAME_KINDS.has(name) ? (name as FrameKind) : null;
|
|
427
|
+
if (kind === null) {
|
|
428
|
+
this.tally(
|
|
429
|
+
"validation-error",
|
|
430
|
+
OBO_ISSUE.UNKNOWN_ELEMENT,
|
|
431
|
+
"frames of an unknown type are not nodes; kept in meta.extra.obo.unknownFrames",
|
|
432
|
+
`[${name}]`,
|
|
433
|
+
line,
|
|
434
|
+
);
|
|
435
|
+
}
|
|
436
|
+
this.frame = { kind, name, line, clauses: [] };
|
|
437
|
+
}
|
|
438
|
+
|
|
439
|
+
/**
|
|
440
|
+
* Record one header clause.
|
|
441
|
+
* @param clause - the clause
|
|
442
|
+
*/
|
|
443
|
+
private headerClause(clause: Clause): void {
|
|
444
|
+
(this.header[clause.tag] ??= []).push(clause.value);
|
|
445
|
+
const { tag } = clause;
|
|
446
|
+
const tokens = tokenize(splitQualifiers(clause.value).value);
|
|
447
|
+
const first = tokens.length > 0 && tokens[0].kind === "word" ? tokens[0].text : null;
|
|
448
|
+
if (tag === "format-version") {
|
|
449
|
+
this.version ??= unescapeObo(clause.value).trim();
|
|
450
|
+
} else if (tag === "default-namespace") {
|
|
451
|
+
this.defaultNamespace ??= first;
|
|
452
|
+
} else if (tag === "subsetdef" && first !== null) {
|
|
453
|
+
this.subsets.add(first);
|
|
454
|
+
} else if (tag === "synonymtypedef" && first !== null) {
|
|
455
|
+
const scope = tokens.slice(1).find((t) => t.kind === "word" && SYNONYM_SCOPES.has(t.text));
|
|
456
|
+
this.synonymTypes.set(first, scope?.text ?? null);
|
|
457
|
+
}
|
|
458
|
+
const now = DEPRECATED_HEADER.get(tag);
|
|
459
|
+
if (now !== undefined) {
|
|
460
|
+
this.tally("coercion", OBO_ISSUE.DEPRECATED_TAG, `the 1.0 header tag is read as ${now}`, tag, clause.line);
|
|
461
|
+
}
|
|
462
|
+
if (NOT_APPLIED_HEADER.test(tag)) {
|
|
463
|
+
this.tally(
|
|
464
|
+
"unsupported",
|
|
465
|
+
OBO_ISSUE.HEADER_NOT_APPLIED,
|
|
466
|
+
"kept in meta.extra.obo.header, not applied",
|
|
467
|
+
tag,
|
|
468
|
+
clause.line,
|
|
469
|
+
);
|
|
470
|
+
}
|
|
471
|
+
}
|
|
472
|
+
|
|
473
|
+
/** Finish the frame being read: find its id and merge its clauses into the record of that id. */
|
|
474
|
+
finishFrame(): void {
|
|
475
|
+
const { frame } = this;
|
|
476
|
+
this.frame = null;
|
|
477
|
+
if (frame === null) {
|
|
478
|
+
return;
|
|
479
|
+
}
|
|
480
|
+
if (frame.kind === null) {
|
|
481
|
+
// the guides: an unrecognized frame must survive, so its clauses are kept, raw
|
|
482
|
+
const clauses: Record<string, string[]> = {};
|
|
483
|
+
for (const clause of frame.clauses) {
|
|
484
|
+
(clauses[clause.tag] ??= []).push(clause.value);
|
|
485
|
+
}
|
|
486
|
+
this.unknownFrames.push({ type: frame.name, clauses });
|
|
487
|
+
return;
|
|
488
|
+
}
|
|
489
|
+
if (++this.framesSinceCheck >= ABORT_CHECK_INTERVAL) {
|
|
490
|
+
this.framesSinceCheck = 0;
|
|
491
|
+
throwIfAborted(this.signal);
|
|
492
|
+
}
|
|
493
|
+
const idAt = frame.clauses.findIndex((c) => c.tag === "id");
|
|
494
|
+
const id = idAt < 0 ? "" : unescapeObo(splitQualifiers(frame.clauses[idAt].value).value).trim();
|
|
495
|
+
if (id.length === 0) {
|
|
496
|
+
this.report.error(
|
|
497
|
+
"missing-value",
|
|
498
|
+
OBO_ISSUE.MISSING_ID,
|
|
499
|
+
`a [${frame.kind}] frame has no id; it is skipped`,
|
|
500
|
+
{
|
|
501
|
+
line: frame.line,
|
|
502
|
+
},
|
|
503
|
+
);
|
|
504
|
+
this.report.counts.skippedNodes++;
|
|
505
|
+
return;
|
|
506
|
+
}
|
|
507
|
+
if (idAt > 0) {
|
|
508
|
+
this.tally(
|
|
509
|
+
"validation-error",
|
|
510
|
+
OBO_ISSUE.ID_NOT_FIRST,
|
|
511
|
+
"the id clause is not the frame's first; it is used",
|
|
512
|
+
"id",
|
|
513
|
+
frame.clauses[idAt].line,
|
|
514
|
+
);
|
|
515
|
+
}
|
|
516
|
+
const record = this.recordFor(frame.kind, id, frame.line);
|
|
517
|
+
for (let i = 0; i < frame.clauses.length; i++) {
|
|
518
|
+
const clause = frame.clauses[i];
|
|
519
|
+
if (clause.tag === "id") {
|
|
520
|
+
if (i !== idAt) {
|
|
521
|
+
this.tally(
|
|
522
|
+
"validation-error",
|
|
523
|
+
OBO_ISSUE.DUPLICATE_ATTRIBUTE,
|
|
524
|
+
"a frame names a second id; the first is kept",
|
|
525
|
+
"id",
|
|
526
|
+
clause.line,
|
|
527
|
+
);
|
|
528
|
+
}
|
|
529
|
+
continue;
|
|
530
|
+
}
|
|
531
|
+
this.applyClause(record, clause);
|
|
532
|
+
}
|
|
533
|
+
if (record.kind === "Typedef") {
|
|
534
|
+
record.raw.id ??= [id];
|
|
535
|
+
}
|
|
536
|
+
}
|
|
537
|
+
|
|
538
|
+
/**
|
|
539
|
+
* The record a frame merges into: the existing one of its id, or a new one.
|
|
540
|
+
* @param kind - the frame type
|
|
541
|
+
* @param id - the id
|
|
542
|
+
* @param line - the frame's line
|
|
543
|
+
* @returns the record
|
|
544
|
+
*/
|
|
545
|
+
private recordFor(kind: FrameKind, id: string, line: number): NodeRecord {
|
|
546
|
+
const map = kind === "Typedef" ? this.typedefs : this.nodes;
|
|
547
|
+
const existing = map.get(id);
|
|
548
|
+
if (existing !== undefined) {
|
|
549
|
+
if (existing.kind !== kind) {
|
|
550
|
+
this.tally(
|
|
551
|
+
"validation-error",
|
|
552
|
+
OBO_ISSUE.ID_KIND_CLASH,
|
|
553
|
+
`an id names both a ${existing.kind} and an ${kind}; the ${existing.kind} is the node`,
|
|
554
|
+
id,
|
|
555
|
+
line,
|
|
556
|
+
);
|
|
557
|
+
} else {
|
|
558
|
+
this.report.warning(
|
|
559
|
+
"merged",
|
|
560
|
+
OBO_ISSUE.DUPLICATE_NODE,
|
|
561
|
+
`two [${kind}] frames have the id ${id}; they are merged (spec 4.1.1)`,
|
|
562
|
+
{
|
|
563
|
+
line,
|
|
564
|
+
element: id,
|
|
565
|
+
},
|
|
566
|
+
);
|
|
567
|
+
}
|
|
568
|
+
return existing;
|
|
569
|
+
}
|
|
570
|
+
const record: NodeRecord = {
|
|
571
|
+
id,
|
|
572
|
+
kind,
|
|
573
|
+
line,
|
|
574
|
+
values: new Map(),
|
|
575
|
+
lists: new Map(),
|
|
576
|
+
edges: [],
|
|
577
|
+
seen: new Set(),
|
|
578
|
+
raw: {},
|
|
579
|
+
};
|
|
580
|
+
map.set(id, record);
|
|
581
|
+
return record;
|
|
582
|
+
}
|
|
583
|
+
|
|
584
|
+
/**
|
|
585
|
+
* A clause's tag (a 1.0 tag read as its current one), the synonym scope that tag implies, its
|
|
586
|
+
* qualifier block and its value, with the syntax warnings about stray braces recorded.
|
|
587
|
+
* @param clause - the clause
|
|
588
|
+
* @returns the parts
|
|
589
|
+
*/
|
|
590
|
+
private clauseParts(clause: Clause): {
|
|
591
|
+
tag: string;
|
|
592
|
+
scope: string | null;
|
|
593
|
+
qualifiers: ReturnType<typeof splitQualifiers>["qualifiers"];
|
|
594
|
+
value: string;
|
|
595
|
+
} {
|
|
596
|
+
let { tag } = clause;
|
|
597
|
+
let scope: string | null = null;
|
|
598
|
+
const deprecated = DEPRECATED_TAGS.get(tag);
|
|
599
|
+
if (deprecated !== undefined) {
|
|
600
|
+
this.tally(
|
|
601
|
+
"coercion",
|
|
602
|
+
OBO_ISSUE.DEPRECATED_TAG,
|
|
603
|
+
`the 1.0 tag is read as ${deprecated[0]}`,
|
|
604
|
+
tag,
|
|
605
|
+
clause.line,
|
|
606
|
+
);
|
|
607
|
+
[tag, scope] = deprecated;
|
|
608
|
+
}
|
|
609
|
+
const split = splitQualifiers(clause.value);
|
|
610
|
+
const { qualifiers } = split;
|
|
611
|
+
if (split.badBlock !== null) {
|
|
612
|
+
this.tally(
|
|
613
|
+
"parse-error",
|
|
614
|
+
OBO_ISSUE.SYNTAX,
|
|
615
|
+
"a closing brace that is not a qualifier block was kept as text",
|
|
616
|
+
tag,
|
|
617
|
+
clause.line,
|
|
618
|
+
);
|
|
619
|
+
} else if (hasStrayBrace(split.value)) {
|
|
620
|
+
this.tally(
|
|
621
|
+
"parse-error",
|
|
622
|
+
OBO_ISSUE.SYNTAX,
|
|
623
|
+
"an unescaped brace inside a value was kept as text",
|
|
624
|
+
tag,
|
|
625
|
+
clause.line,
|
|
626
|
+
);
|
|
627
|
+
}
|
|
628
|
+
return { tag, scope, qualifiers, value: split.value };
|
|
629
|
+
}
|
|
630
|
+
|
|
631
|
+
/**
|
|
632
|
+
* A def, expand_assertion_to or expand_expression_to clause: quoted text and its xref list.
|
|
633
|
+
* @param record - the record
|
|
634
|
+
* @param tag - the tag
|
|
635
|
+
* @param value - the clause value
|
|
636
|
+
* @param tokens - its tokens
|
|
637
|
+
* @param qualifiers - its qualifier block
|
|
638
|
+
* @param line - its line
|
|
639
|
+
*/
|
|
640
|
+
private definition(
|
|
641
|
+
record: NodeRecord,
|
|
642
|
+
tag: string,
|
|
643
|
+
value: string,
|
|
644
|
+
tokens: ReturnType<typeof tokenize>,
|
|
645
|
+
qualifiers: ReturnType<typeof splitQualifiers>["qualifiers"],
|
|
646
|
+
line: number,
|
|
647
|
+
): void {
|
|
648
|
+
if (tag === "def" && record.values.has("def")) {
|
|
649
|
+
// the second def of an id: reported, and none of its xrefs leak into the first's
|
|
650
|
+
this.single(record, "def", null, line);
|
|
651
|
+
return;
|
|
652
|
+
}
|
|
653
|
+
const text = this.quotedWithXrefs(record, tag, value, tokens, line);
|
|
654
|
+
if (tag === "def") {
|
|
655
|
+
this.single(record, "def", text.text, line, () => {
|
|
656
|
+
record.values.set("def.xrefs", text.xrefs);
|
|
657
|
+
});
|
|
658
|
+
this.extraQualifiers(record, tag, text.text, qualifiers);
|
|
659
|
+
} else {
|
|
660
|
+
this.push(record, tag, withQualifiers({ template: text.text, xrefs: text.xrefs }, qualifiers));
|
|
661
|
+
}
|
|
662
|
+
}
|
|
663
|
+
|
|
664
|
+
/**
|
|
665
|
+
* Apply one clause to a record.
|
|
666
|
+
* @param record - the record
|
|
667
|
+
* @param clause - the clause
|
|
668
|
+
*/
|
|
669
|
+
private applyClause(record: NodeRecord, clause: Clause): void {
|
|
670
|
+
const key = `${clause.tag}\u0000${clause.value}`;
|
|
671
|
+
if (record.seen.has(key)) {
|
|
672
|
+
return;
|
|
673
|
+
}
|
|
674
|
+
record.seen.add(key);
|
|
675
|
+
if (record.kind === "Typedef") {
|
|
676
|
+
(record.raw[clause.tag] ??= []).push(clause.value);
|
|
677
|
+
}
|
|
678
|
+
const { tag, scope, qualifiers, value } = this.clauseParts(clause);
|
|
679
|
+
const where = { line: clause.line, element: record.id };
|
|
680
|
+
if (TYPEDEF_TAGS.has(tag) && record.kind !== "Typedef") {
|
|
681
|
+
this.unrecognized(record, clause);
|
|
682
|
+
return;
|
|
683
|
+
}
|
|
684
|
+
if (TEXT_TAGS.has(tag)) {
|
|
685
|
+
this.single(record, tag, unescapeObo(value).trim(), clause.line);
|
|
686
|
+
this.extraQualifiers(record, tag, unescapeObo(value).trim(), qualifiers);
|
|
687
|
+
return;
|
|
688
|
+
}
|
|
689
|
+
if (BOOL_TAGS.has(tag)) {
|
|
690
|
+
const text = unescapeObo(value).trim();
|
|
691
|
+
if (text !== "true" && text !== "false") {
|
|
692
|
+
this.report.error(
|
|
693
|
+
"validation-error",
|
|
694
|
+
OBO_ISSUE.BAD_VALUE,
|
|
695
|
+
`${tag}: ${JSON.stringify(text)} is not true or false; the clause is skipped`,
|
|
696
|
+
where,
|
|
697
|
+
);
|
|
698
|
+
return;
|
|
699
|
+
}
|
|
700
|
+
this.single(record, tag, text === "true", clause.line);
|
|
701
|
+
this.extraQualifiers(record, tag, text, qualifiers);
|
|
702
|
+
return;
|
|
703
|
+
}
|
|
704
|
+
const tokens = tokenize(value);
|
|
705
|
+
if (ID_LIST_TAGS.has(tag)) {
|
|
706
|
+
const id = this.oneWord(tokens, tag, where);
|
|
707
|
+
if (id !== null) {
|
|
708
|
+
this.push(record, tag, id);
|
|
709
|
+
this.extraQualifiers(record, tag, id, qualifiers);
|
|
710
|
+
if (tag === "subset" && !this.subsets.has(id)) {
|
|
711
|
+
this.tally(
|
|
712
|
+
"validation-error",
|
|
713
|
+
OBO_ISSUE.UNDECLARED,
|
|
714
|
+
"a subset no subsetdef declares",
|
|
715
|
+
id,
|
|
716
|
+
clause.line,
|
|
717
|
+
);
|
|
718
|
+
}
|
|
719
|
+
}
|
|
720
|
+
return;
|
|
721
|
+
}
|
|
722
|
+
switch (tag) {
|
|
723
|
+
case "is_a":
|
|
724
|
+
case "instance_of": {
|
|
725
|
+
if (tag === "instance_of" && record.kind !== "Instance") {
|
|
726
|
+
this.unrecognized(record, clause);
|
|
727
|
+
return;
|
|
728
|
+
}
|
|
729
|
+
const target = this.oneWord(tokens, tag, where);
|
|
730
|
+
if (target !== null) {
|
|
731
|
+
record.edges.push({ relation: tag, target, qualifiers, line: clause.line });
|
|
732
|
+
}
|
|
733
|
+
return;
|
|
734
|
+
}
|
|
735
|
+
case "relationship": {
|
|
736
|
+
if (tokens.length !== 2 || tokens.some((t) => t.kind !== "word")) {
|
|
737
|
+
this.report.error(
|
|
738
|
+
"validation-error",
|
|
739
|
+
OBO_ISSUE.BAD_VALUE,
|
|
740
|
+
`relationship needs a relation and a target, found ${JSON.stringify(unescapeObo(value))}; the clause is skipped`,
|
|
741
|
+
where,
|
|
742
|
+
);
|
|
743
|
+
return;
|
|
744
|
+
}
|
|
745
|
+
record.edges.push({ relation: tokens[0].text, target: tokens[1].text, qualifiers, line: clause.line });
|
|
746
|
+
return;
|
|
747
|
+
}
|
|
748
|
+
case "intersection_of": {
|
|
749
|
+
if (tokens.length < 1 || tokens.length > 2 || tokens.some((t) => t.kind !== "word")) {
|
|
750
|
+
this.report.error(
|
|
751
|
+
"validation-error",
|
|
752
|
+
OBO_ISSUE.BAD_VALUE,
|
|
753
|
+
`intersection_of needs a class, or a relation and a class, found ${JSON.stringify(unescapeObo(value))}; the clause is skipped`,
|
|
754
|
+
where,
|
|
755
|
+
);
|
|
756
|
+
return;
|
|
757
|
+
}
|
|
758
|
+
const item: Record<string, unknown> =
|
|
759
|
+
tokens.length === 1
|
|
760
|
+
? { relation: null, target: tokens[0].text }
|
|
761
|
+
: { relation: tokens[0].text, target: tokens[1].text };
|
|
762
|
+
this.push(record, tag, withQualifiers(item, qualifiers));
|
|
763
|
+
return;
|
|
764
|
+
}
|
|
765
|
+
case "holds_over_chain":
|
|
766
|
+
case "equivalent_to_chain": {
|
|
767
|
+
if (tokens.length !== 2 || tokens.some((t) => t.kind !== "word")) {
|
|
768
|
+
this.report.error(
|
|
769
|
+
"validation-error",
|
|
770
|
+
OBO_ISSUE.BAD_VALUE,
|
|
771
|
+
`${tag} needs two relations; the clause is skipped`,
|
|
772
|
+
where,
|
|
773
|
+
);
|
|
774
|
+
return;
|
|
775
|
+
}
|
|
776
|
+
this.push(record, tag, [tokens[0].text, tokens[1].text]);
|
|
777
|
+
this.extraQualifiers(record, tag, `${tokens[0].text} ${tokens[1].text}`, qualifiers);
|
|
778
|
+
return;
|
|
779
|
+
}
|
|
780
|
+
case "def":
|
|
781
|
+
case "expand_assertion_to":
|
|
782
|
+
case "expand_expression_to":
|
|
783
|
+
this.definition(record, tag, value, tokens, qualifiers, clause.line);
|
|
784
|
+
return;
|
|
785
|
+
case "synonym":
|
|
786
|
+
this.synonym(record, tokens, scope, qualifiers, clause.line);
|
|
787
|
+
return;
|
|
788
|
+
case "xref": {
|
|
789
|
+
const xref = parseXref(value);
|
|
790
|
+
if (xref === null) {
|
|
791
|
+
this.report.error(
|
|
792
|
+
"validation-error",
|
|
793
|
+
OBO_ISSUE.BAD_VALUE,
|
|
794
|
+
"xref has no id; the clause is skipped",
|
|
795
|
+
where,
|
|
796
|
+
);
|
|
797
|
+
return;
|
|
798
|
+
}
|
|
799
|
+
this.push(record, "xref", xref.id);
|
|
800
|
+
this.describe(record, [xref]);
|
|
801
|
+
this.extraQualifiers(record, tag, xref.id, qualifiers ?? xref.qualifiers);
|
|
802
|
+
return;
|
|
803
|
+
}
|
|
804
|
+
case "property_value":
|
|
805
|
+
this.propertyValue(record, tokens, qualifiers, where);
|
|
806
|
+
return;
|
|
807
|
+
default:
|
|
808
|
+
this.unrecognized(record, clause);
|
|
809
|
+
}
|
|
810
|
+
}
|
|
811
|
+
|
|
812
|
+
/**
|
|
813
|
+
* Read a value that must be one word.
|
|
814
|
+
* @param tokens - the value's tokens
|
|
815
|
+
* @param tag - the tag, for the message
|
|
816
|
+
* @param where - the location
|
|
817
|
+
* @returns the word, or null after reporting E_BAD_VALUE
|
|
818
|
+
*/
|
|
819
|
+
private oneWord(tokens: readonly Token[], tag: string, where: Where): string | null {
|
|
820
|
+
if (tokens.length === 1 && tokens[0].kind === "word") {
|
|
821
|
+
return tokens[0].text;
|
|
822
|
+
}
|
|
823
|
+
const found = tokens.map((t) => t.text).join(" ");
|
|
824
|
+
this.report.error(
|
|
825
|
+
"validation-error",
|
|
826
|
+
OBO_ISSUE.BAD_VALUE,
|
|
827
|
+
`${tag} needs one id, found ${JSON.stringify(found)}; the clause is skipped`,
|
|
828
|
+
where,
|
|
829
|
+
);
|
|
830
|
+
return null;
|
|
831
|
+
}
|
|
832
|
+
|
|
833
|
+
/**
|
|
834
|
+
* Read `"text" [xrefs]` (def, expand_*), reporting the forms the grammar does not allow.
|
|
835
|
+
* @param record - the record (xref descriptions go to its `xref.descriptions`)
|
|
836
|
+
* @param tag - the tag
|
|
837
|
+
* @param value - the raw value
|
|
838
|
+
* @param tokens - its tokens
|
|
839
|
+
* @param line - the line
|
|
840
|
+
* @returns the text and the xref ids
|
|
841
|
+
*/
|
|
842
|
+
private quotedWithXrefs(
|
|
843
|
+
record: NodeRecord,
|
|
844
|
+
tag: string,
|
|
845
|
+
value: string,
|
|
846
|
+
tokens: readonly Token[],
|
|
847
|
+
line: number,
|
|
848
|
+
): { text: string; xrefs: string[] } {
|
|
849
|
+
if (tokens.length === 0 || tokens[0].kind !== "quoted") {
|
|
850
|
+
this.tally("parse-error", OBO_ISSUE.SYNTAX, "the text is not quoted; the whole value is kept", tag, line);
|
|
851
|
+
return { text: unescapeObo(value).trim(), xrefs: [] };
|
|
852
|
+
}
|
|
853
|
+
if (tokens[0].unterminated) {
|
|
854
|
+
this.tally(
|
|
855
|
+
"parse-error",
|
|
856
|
+
OBO_ISSUE.SYNTAX,
|
|
857
|
+
"an unterminated quote; the text runs to the end of the line",
|
|
858
|
+
tag,
|
|
859
|
+
line,
|
|
860
|
+
);
|
|
861
|
+
return { text: tokens[0].text, xrefs: [] };
|
|
862
|
+
}
|
|
863
|
+
const list = tokens[1];
|
|
864
|
+
if (list?.kind !== "list") {
|
|
865
|
+
this.tally("parse-error", OBO_ISSUE.SYNTAX, "the xref list is missing", tag, line);
|
|
866
|
+
return { text: tokens[0].text, xrefs: [] };
|
|
867
|
+
}
|
|
868
|
+
const xrefs = parseXrefList(list.text);
|
|
869
|
+
this.describe(record, xrefs);
|
|
870
|
+
this.xrefQualifiers(record, tag, xrefs);
|
|
871
|
+
return { text: tokens[0].text, xrefs: xrefs.map((x) => x.id) };
|
|
872
|
+
}
|
|
873
|
+
|
|
874
|
+
/**
|
|
875
|
+
* Read a synonym: `"text" SCOPE? TYPE? [xrefs]` (the scope optional in 1.2).
|
|
876
|
+
* @param record - the record
|
|
877
|
+
* @param tokens - the value's tokens
|
|
878
|
+
* @param fixedScope - the scope of a 1.0 tag (exact_synonym), or null
|
|
879
|
+
* @param qualifiers - the clause's qualifiers
|
|
880
|
+
* @param line - the line
|
|
881
|
+
*/
|
|
882
|
+
private synonym(
|
|
883
|
+
record: NodeRecord,
|
|
884
|
+
tokens: readonly Token[],
|
|
885
|
+
fixedScope: string | null,
|
|
886
|
+
qualifiers: Qualifiers | null,
|
|
887
|
+
line: number,
|
|
888
|
+
): void {
|
|
889
|
+
if (tokens.length === 0 || tokens[0].kind !== "quoted") {
|
|
890
|
+
this.report.error(
|
|
891
|
+
"validation-error",
|
|
892
|
+
OBO_ISSUE.BAD_VALUE,
|
|
893
|
+
"a synonym needs its quoted text; the clause is skipped",
|
|
894
|
+
{
|
|
895
|
+
line,
|
|
896
|
+
element: record.id,
|
|
897
|
+
},
|
|
898
|
+
);
|
|
899
|
+
return;
|
|
900
|
+
}
|
|
901
|
+
if (tokens[0].unterminated) {
|
|
902
|
+
this.tally(
|
|
903
|
+
"parse-error",
|
|
904
|
+
OBO_ISSUE.SYNTAX,
|
|
905
|
+
"an unterminated quote; the text runs to the end of the line",
|
|
906
|
+
"synonym",
|
|
907
|
+
line,
|
|
908
|
+
);
|
|
909
|
+
}
|
|
910
|
+
const words = tokens
|
|
911
|
+
.slice(1)
|
|
912
|
+
.filter((t) => t.kind === "word")
|
|
913
|
+
.map((t) => t.text);
|
|
914
|
+
const list = tokens.slice(1).find((t) => t.kind === "list");
|
|
915
|
+
let scope: string | null = fixedScope;
|
|
916
|
+
let type: string | null = null;
|
|
917
|
+
if (fixedScope === null) {
|
|
918
|
+
if (words.length > 0 && SYNONYM_SCOPES.has(words[0])) {
|
|
919
|
+
[scope] = words;
|
|
920
|
+
type = words[1] ?? null;
|
|
921
|
+
} else if (words.length > 0 && this.synonymTypes.has(words[0])) {
|
|
922
|
+
[type] = words;
|
|
923
|
+
scope = "RELATED";
|
|
924
|
+
} else if (words.length > 0) {
|
|
925
|
+
[type] = words;
|
|
926
|
+
this.tally(
|
|
927
|
+
"validation-error",
|
|
928
|
+
OBO_ISSUE.SYNONYM_SCOPE,
|
|
929
|
+
`a scope that is not EXACT, BROAD, NARROW or RELATED was kept as the type, scope null`,
|
|
930
|
+
words[0],
|
|
931
|
+
line,
|
|
932
|
+
);
|
|
933
|
+
} else {
|
|
934
|
+
scope = "RELATED";
|
|
935
|
+
}
|
|
936
|
+
if (scope === "RELATED" && (words.length === 0 || words[0] === type) && !this.allowsMissingScope()) {
|
|
937
|
+
this.tally(
|
|
938
|
+
"validation-error",
|
|
939
|
+
OBO_ISSUE.SYNONYM_SCOPE,
|
|
940
|
+
"a synonym without a scope (1.2 only) is read as RELATED",
|
|
941
|
+
"synonym",
|
|
942
|
+
line,
|
|
943
|
+
);
|
|
944
|
+
}
|
|
945
|
+
}
|
|
946
|
+
if (type !== null) {
|
|
947
|
+
const declared = this.synonymTypes.get(type);
|
|
948
|
+
if (declared === undefined) {
|
|
949
|
+
if (scope !== null) {
|
|
950
|
+
this.tally(
|
|
951
|
+
"validation-error",
|
|
952
|
+
OBO_ISSUE.UNDECLARED,
|
|
953
|
+
"a synonym type no synonymtypedef declares",
|
|
954
|
+
type,
|
|
955
|
+
line,
|
|
956
|
+
);
|
|
957
|
+
}
|
|
958
|
+
} else if (declared !== null) {
|
|
959
|
+
// the 1.2 guide: a type's default scope is used regardless of the synonym's own
|
|
960
|
+
scope = declared;
|
|
961
|
+
}
|
|
962
|
+
}
|
|
963
|
+
if (words.length > 2 || list === undefined) {
|
|
964
|
+
this.tally(
|
|
965
|
+
"parse-error",
|
|
966
|
+
OBO_ISSUE.SYNTAX,
|
|
967
|
+
list === undefined ? "a synonym without its xref list" : "extra words after the synonym type",
|
|
968
|
+
"synonym",
|
|
969
|
+
line,
|
|
970
|
+
);
|
|
971
|
+
}
|
|
972
|
+
const xrefs = list === undefined ? [] : parseXrefList(list.text);
|
|
973
|
+
this.describe(record, xrefs);
|
|
974
|
+
this.xrefQualifiers(record, "synonym", xrefs);
|
|
975
|
+
this.push(
|
|
976
|
+
record,
|
|
977
|
+
"synonym",
|
|
978
|
+
withQualifiers({ text: tokens[0].text, scope, type, xrefs: xrefs.map((x) => x.id) }, qualifiers),
|
|
979
|
+
);
|
|
980
|
+
}
|
|
981
|
+
|
|
982
|
+
/**
|
|
983
|
+
* Whether the file's version lets a synonym omit its scope (1.0 and 1.2).
|
|
984
|
+
* @returns true for a 1.2 or 1.0 file
|
|
985
|
+
*/
|
|
986
|
+
private allowsMissingScope(): boolean {
|
|
987
|
+
return this.version !== null && /^(1\.[0-2]|GO_1\.[0-2])\b/.test(this.version);
|
|
988
|
+
}
|
|
989
|
+
|
|
990
|
+
/**
|
|
991
|
+
* Read a property_value: `relation "literal" datatype` or `relation id`.
|
|
992
|
+
* @param record - the record
|
|
993
|
+
* @param tokens - the value's tokens
|
|
994
|
+
* @param qualifiers - the clause's qualifiers
|
|
995
|
+
* @param where - the location
|
|
996
|
+
*/
|
|
997
|
+
private propertyValue(
|
|
998
|
+
record: NodeRecord,
|
|
999
|
+
tokens: readonly Token[],
|
|
1000
|
+
qualifiers: Qualifiers | null,
|
|
1001
|
+
where: Where,
|
|
1002
|
+
): void {
|
|
1003
|
+
if (tokens.length < 2 || tokens.length > 3 || tokens[0].kind !== "word" || tokens[1].kind === "list") {
|
|
1004
|
+
this.report.error(
|
|
1005
|
+
"validation-error",
|
|
1006
|
+
OBO_ISSUE.BAD_VALUE,
|
|
1007
|
+
"property_value needs a relation and a value; the clause is skipped",
|
|
1008
|
+
where,
|
|
1009
|
+
);
|
|
1010
|
+
return;
|
|
1011
|
+
}
|
|
1012
|
+
const datatype = tokens.length === 3 ? tokens[2].text : null;
|
|
1013
|
+
this.push(
|
|
1014
|
+
record,
|
|
1015
|
+
"property_value",
|
|
1016
|
+
withQualifiers({ relation: tokens[0].text, value: tokens[1].text, datatype }, qualifiers),
|
|
1017
|
+
);
|
|
1018
|
+
}
|
|
1019
|
+
|
|
1020
|
+
/**
|
|
1021
|
+
* Set a single-valued column, keeping the first value of an id.
|
|
1022
|
+
* @param record - the record
|
|
1023
|
+
* @param column - the column
|
|
1024
|
+
* @param value - the value
|
|
1025
|
+
* @param line - the line
|
|
1026
|
+
* @param also - more to set together with the value
|
|
1027
|
+
*/
|
|
1028
|
+
private single(record: NodeRecord, column: string, value: unknown, line: number, also?: () => void): void {
|
|
1029
|
+
if (record.values.has(column)) {
|
|
1030
|
+
this.tally(
|
|
1031
|
+
"validation-error",
|
|
1032
|
+
OBO_ISSUE.DUPLICATE_ATTRIBUTE,
|
|
1033
|
+
"a single-valued tag given twice for one id; the first is kept",
|
|
1034
|
+
column,
|
|
1035
|
+
line,
|
|
1036
|
+
);
|
|
1037
|
+
return;
|
|
1038
|
+
}
|
|
1039
|
+
record.values.set(column, value);
|
|
1040
|
+
also?.();
|
|
1041
|
+
}
|
|
1042
|
+
|
|
1043
|
+
/**
|
|
1044
|
+
* Append to a list column, without repeats (merged frames take the union).
|
|
1045
|
+
* @param record - the record
|
|
1046
|
+
* @param column - the column
|
|
1047
|
+
* @param value - the item
|
|
1048
|
+
*/
|
|
1049
|
+
private push(record: NodeRecord, column: string, value: unknown): void {
|
|
1050
|
+
let list = record.lists.get(column);
|
|
1051
|
+
if (list === undefined) {
|
|
1052
|
+
list = [];
|
|
1053
|
+
record.lists.set(column, list);
|
|
1054
|
+
}
|
|
1055
|
+
// objects are never repeated: identical clauses were already skipped
|
|
1056
|
+
if (typeof value === "object" || !list.includes(value)) {
|
|
1057
|
+
list.push(value);
|
|
1058
|
+
}
|
|
1059
|
+
}
|
|
1060
|
+
|
|
1061
|
+
/**
|
|
1062
|
+
* Record xref descriptions in `xref.descriptions`.
|
|
1063
|
+
* @param record - the record
|
|
1064
|
+
* @param xrefs - the xrefs
|
|
1065
|
+
*/
|
|
1066
|
+
private describe(record: NodeRecord, xrefs: readonly Xref[]): void {
|
|
1067
|
+
for (const xref of xrefs) {
|
|
1068
|
+
if (xref.description !== null) {
|
|
1069
|
+
const map = (record.values.get("xref.descriptions") ?? {}) as Record<string, string>;
|
|
1070
|
+
map[xref.id] ??= xref.description;
|
|
1071
|
+
record.values.set("xref.descriptions", map);
|
|
1072
|
+
}
|
|
1073
|
+
}
|
|
1074
|
+
}
|
|
1075
|
+
|
|
1076
|
+
/**
|
|
1077
|
+
* Keep the per-xref qualifiers of an xref list (OBO 1.2: `[A:1 {source="x"}]`) in
|
|
1078
|
+
* `obo.qualifiers` under `<tag>.xrefs`.
|
|
1079
|
+
* @param record - the record
|
|
1080
|
+
* @param tag - the clause's tag
|
|
1081
|
+
* @param xrefs - the list's xrefs
|
|
1082
|
+
*/
|
|
1083
|
+
private xrefQualifiers(record: NodeRecord, tag: string, xrefs: readonly Xref[]): void {
|
|
1084
|
+
for (const xref of xrefs) {
|
|
1085
|
+
this.extraQualifiers(record, `${tag}.xrefs`, xref.id, xref.qualifiers);
|
|
1086
|
+
}
|
|
1087
|
+
}
|
|
1088
|
+
|
|
1089
|
+
/**
|
|
1090
|
+
* Keep the qualifiers of a clause whose column has no place for them in `obo.qualifiers`.
|
|
1091
|
+
* @param record - the record
|
|
1092
|
+
* @param tag - the tag
|
|
1093
|
+
* @param value - the clause's value as stored
|
|
1094
|
+
* @param qualifiers - the qualifiers, or null
|
|
1095
|
+
*/
|
|
1096
|
+
private extraQualifiers(record: NodeRecord, tag: string, value: string, qualifiers: Qualifiers | null): void {
|
|
1097
|
+
if (qualifiers === null) {
|
|
1098
|
+
return;
|
|
1099
|
+
}
|
|
1100
|
+
const map = (record.values.get("obo.qualifiers") ?? {}) as Record<string, unknown[]>;
|
|
1101
|
+
(map[tag] ??= []).push({ value, qualifiers });
|
|
1102
|
+
record.values.set("obo.qualifiers", map);
|
|
1103
|
+
}
|
|
1104
|
+
|
|
1105
|
+
/**
|
|
1106
|
+
* Keep an unknown clause in `obo.unrecognized`.
|
|
1107
|
+
* @param record - the record
|
|
1108
|
+
* @param clause - the clause
|
|
1109
|
+
*/
|
|
1110
|
+
private unrecognized(record: NodeRecord, clause: Clause): void {
|
|
1111
|
+
const map = (record.values.get("obo.unrecognized") ?? {}) as Record<string, string[]>;
|
|
1112
|
+
(map[clause.tag] ??= []).push(unescapeObo(clause.value));
|
|
1113
|
+
record.values.set("obo.unrecognized", map);
|
|
1114
|
+
this.tally(
|
|
1115
|
+
"validation-error",
|
|
1116
|
+
OBO_ISSUE.UNKNOWN_ELEMENT,
|
|
1117
|
+
"an unknown tag was kept in obo.unrecognized",
|
|
1118
|
+
clause.tag,
|
|
1119
|
+
clause.line,
|
|
1120
|
+
);
|
|
1121
|
+
}
|
|
1122
|
+
|
|
1123
|
+
/**
|
|
1124
|
+
* Count a warning, recorded once per code and element at the end.
|
|
1125
|
+
* @param category - the category
|
|
1126
|
+
* @param code - the code
|
|
1127
|
+
* @param what - what happened
|
|
1128
|
+
* @param element - the tag, id or name it is about
|
|
1129
|
+
* @param line - the first line
|
|
1130
|
+
*/
|
|
1131
|
+
tally(category: Tally["category"], code: string, what: string, element: string, line: number): void {
|
|
1132
|
+
const key = `${code}\u0000${element}\u0000${what}`;
|
|
1133
|
+
const found = this.tallies.get(key);
|
|
1134
|
+
if (found === undefined) {
|
|
1135
|
+
this.tallies.set(key, { category, code, element, what, line, count: 1 });
|
|
1136
|
+
} else {
|
|
1137
|
+
found.count++;
|
|
1138
|
+
}
|
|
1139
|
+
}
|
|
1140
|
+
|
|
1141
|
+
/** Record the counted warnings, in the order they were first seen. */
|
|
1142
|
+
flushTallies(): void {
|
|
1143
|
+
for (const t of this.tallies.values()) {
|
|
1144
|
+
this.report.warning(
|
|
1145
|
+
t.category,
|
|
1146
|
+
t.code,
|
|
1147
|
+
`${t.element}: ${t.what} (${t.count} time(s), first at line ${t.line})`,
|
|
1148
|
+
{
|
|
1149
|
+
line: t.line,
|
|
1150
|
+
element: t.element,
|
|
1151
|
+
},
|
|
1152
|
+
);
|
|
1153
|
+
}
|
|
1154
|
+
this.tallies.clear();
|
|
1155
|
+
}
|
|
1156
|
+
|
|
1157
|
+
/**
|
|
1158
|
+
* The header as GraphMeta fields (design 4.2): name from `ontology`, the version, `created` from
|
|
1159
|
+
* the `dd:MM:yyyy HH:mm` date, `creator` from `saved-by`.
|
|
1160
|
+
* @returns the metadata patch, without `extra`
|
|
1161
|
+
*/
|
|
1162
|
+
meta(): GraphMetaPatch {
|
|
1163
|
+
const first = (tag: string): string | null => {
|
|
1164
|
+
const values = this.header[tag];
|
|
1165
|
+
return values === undefined ? null : unescapeObo(values[0]).trim();
|
|
1166
|
+
};
|
|
1167
|
+
return {
|
|
1168
|
+
name: first("ontology"),
|
|
1169
|
+
sourceFormat: "obo",
|
|
1170
|
+
sourceVersion: this.version,
|
|
1171
|
+
created: oboDate(first("date")),
|
|
1172
|
+
creator: first("saved-by"),
|
|
1173
|
+
};
|
|
1174
|
+
}
|
|
1175
|
+
|
|
1176
|
+
/**
|
|
1177
|
+
* The raw header clauses.
|
|
1178
|
+
* @returns the header by tag
|
|
1179
|
+
*/
|
|
1180
|
+
headerRecord(): Record<string, string[]> {
|
|
1181
|
+
return this.header;
|
|
1182
|
+
}
|
|
1183
|
+
|
|
1184
|
+
/**
|
|
1185
|
+
* The namespace a frame without one gets.
|
|
1186
|
+
* @returns the default namespace, or null
|
|
1187
|
+
*/
|
|
1188
|
+
namespaceDefault(): string | null {
|
|
1189
|
+
return this.defaultNamespace;
|
|
1190
|
+
}
|
|
1191
|
+
}
|
|
1192
|
+
|
|
1193
|
+
/**
|
|
1194
|
+
* Add the qualifiers to a json item when there are any.
|
|
1195
|
+
* @param item - the item
|
|
1196
|
+
* @param qualifiers - the qualifiers, or null
|
|
1197
|
+
* @returns the item
|
|
1198
|
+
*/
|
|
1199
|
+
function withQualifiers(item: Record<string, unknown>, qualifiers: Qualifiers | null): Record<string, unknown> {
|
|
1200
|
+
return qualifiers === null ? item : { ...item, qualifiers };
|
|
1201
|
+
}
|
|
1202
|
+
|
|
1203
|
+
/**
|
|
1204
|
+
* The OBO header date `dd:MM:yyyy HH:mm` as ISO 8601 (`yyyy-MM-ddTHH:mm`).
|
|
1205
|
+
* @param text - the header value, or null
|
|
1206
|
+
* @returns the ISO text, or null when the date is not of that form
|
|
1207
|
+
*/
|
|
1208
|
+
function oboDate(text: string | null): string | null {
|
|
1209
|
+
const m = text === null ? null : /^(\d{2}):(\d{2}):(\d{4})\s+(\d{2}):(\d{2})$/.exec(text);
|
|
1210
|
+
if (m === null) {
|
|
1211
|
+
return null;
|
|
1212
|
+
}
|
|
1213
|
+
const [, dd, mm, yyyy, hh, min] = m;
|
|
1214
|
+
if (Number(mm) < 1 || Number(mm) > 12 || Number(dd) < 1 || Number(dd) > 31 || Number(hh) > 23 || Number(min) > 59) {
|
|
1215
|
+
return null;
|
|
1216
|
+
}
|
|
1217
|
+
return `${yyyy}-${mm}-${dd}T${hh}:${min}`;
|
|
1218
|
+
}
|
|
1219
|
+
|
|
1220
|
+
/** What writeGraph() pushes, decided before anything reaches the sink. */
|
|
1221
|
+
interface GraphPlan {
|
|
1222
|
+
/** The node records, in first-appearance order. */
|
|
1223
|
+
readonly nodes: readonly NodeRecord[];
|
|
1224
|
+
/** The undeclared targets that become placeholder nodes. */
|
|
1225
|
+
readonly placeholders: readonly string[];
|
|
1226
|
+
/** The edges to push. */
|
|
1227
|
+
readonly edges: readonly { readonly source: string; readonly edge: PendingEdge }[];
|
|
1228
|
+
}
|
|
1229
|
+
|
|
1230
|
+
/**
|
|
1231
|
+
* Decide what is written (design 4.2): the node records (Typedefs too under typedefs "nodes", the
|
|
1232
|
+
* obsolete ones left out under obsolete "drop"), the placeholders of undeclared targets and the
|
|
1233
|
+
* edges; record the warnings that need the whole file (obsoletion rules, undeclared relations,
|
|
1234
|
+
* dangling targets, dropped obsolete terms).
|
|
1235
|
+
* @param reader - the finished reader
|
|
1236
|
+
* @param report - the report
|
|
1237
|
+
* @param addMissingNodes - whether an undeclared target becomes a placeholder node
|
|
1238
|
+
* @param obo - the OBO options
|
|
1239
|
+
* @returns the plan
|
|
1240
|
+
*/
|
|
1241
|
+
function planGraph(
|
|
1242
|
+
reader: OboReader,
|
|
1243
|
+
report: ImportReportBuilder,
|
|
1244
|
+
addMissingNodes: boolean,
|
|
1245
|
+
obo: ResolvedOboOptions,
|
|
1246
|
+
): GraphPlan {
|
|
1247
|
+
const records: NodeRecord[] = [...reader.nodes.values()];
|
|
1248
|
+
for (const [id, typedef] of reader.typedefs) {
|
|
1249
|
+
const node = reader.nodes.get(id);
|
|
1250
|
+
if (node !== undefined) {
|
|
1251
|
+
reader.tally(
|
|
1252
|
+
"validation-error",
|
|
1253
|
+
OBO_ISSUE.ID_KIND_CLASH,
|
|
1254
|
+
`an id names both a ${node.kind} and a Typedef; the ${node.kind} is the node`,
|
|
1255
|
+
id,
|
|
1256
|
+
typedef.line,
|
|
1257
|
+
);
|
|
1258
|
+
} else if (obo.typedefs === "nodes") {
|
|
1259
|
+
records.push(typedef);
|
|
1260
|
+
}
|
|
1261
|
+
}
|
|
1262
|
+
// a class expression needs two or more operands, counted after merging (spec 4.1.1)
|
|
1263
|
+
for (const record of [...reader.nodes.values(), ...reader.typedefs.values()]) {
|
|
1264
|
+
for (const tag of ["intersection_of", "union_of"]) {
|
|
1265
|
+
if (record.lists.get(tag)?.length === 1) {
|
|
1266
|
+
reader.tally(
|
|
1267
|
+
"validation-error",
|
|
1268
|
+
OBO_ISSUE.CARDINALITY,
|
|
1269
|
+
"a single clause (a class expression needs two or more)",
|
|
1270
|
+
tag,
|
|
1271
|
+
record.line,
|
|
1272
|
+
);
|
|
1273
|
+
}
|
|
1274
|
+
}
|
|
1275
|
+
}
|
|
1276
|
+
const dropped = new Set<string>();
|
|
1277
|
+
const kept = new Set<string>();
|
|
1278
|
+
for (const record of records) {
|
|
1279
|
+
const obsolete = record.values.get("is_obsolete") === true;
|
|
1280
|
+
(obo.obsolete === "drop" && obsolete ? dropped : kept).add(record.id);
|
|
1281
|
+
// obsoletion rules (the 1.2 guide; fastobo-validator --obsoletion)
|
|
1282
|
+
if (obsolete && record.edges.length > 0) {
|
|
1283
|
+
reader.tally(
|
|
1284
|
+
"validation-error",
|
|
1285
|
+
OBO_ISSUE.OBSOLETION,
|
|
1286
|
+
"an obsolete term has is_a or relationship clauses; they are kept",
|
|
1287
|
+
"is_obsolete",
|
|
1288
|
+
record.edges[0].line,
|
|
1289
|
+
);
|
|
1290
|
+
}
|
|
1291
|
+
if (!obsolete && (record.lists.has("replaced_by") || record.lists.has("consider"))) {
|
|
1292
|
+
reader.tally(
|
|
1293
|
+
"validation-error",
|
|
1294
|
+
OBO_ISSUE.OBSOLETION,
|
|
1295
|
+
"replaced_by or consider on a term that is not obsolete; kept",
|
|
1296
|
+
"replaced_by",
|
|
1297
|
+
record.line,
|
|
1298
|
+
);
|
|
1299
|
+
}
|
|
1300
|
+
}
|
|
1301
|
+
const declared = new Set<string>(reader.typedefs.keys());
|
|
1302
|
+
for (const typedef of reader.typedefs.values()) {
|
|
1303
|
+
for (const xref of (typedef.lists.get("xref") ?? []) as string[]) {
|
|
1304
|
+
declared.add(xref);
|
|
1305
|
+
}
|
|
1306
|
+
}
|
|
1307
|
+
const dangling: string[] = [];
|
|
1308
|
+
const danglingSet = new Set<string>();
|
|
1309
|
+
let danglingEdges = 0;
|
|
1310
|
+
let droppedEdges = 0;
|
|
1311
|
+
const edges: { source: string; edge: PendingEdge }[] = [];
|
|
1312
|
+
for (const record of records) {
|
|
1313
|
+
for (const edge of record.edges) {
|
|
1314
|
+
if (dropped.has(record.id) || dropped.has(edge.target)) {
|
|
1315
|
+
droppedEdges++;
|
|
1316
|
+
continue;
|
|
1317
|
+
}
|
|
1318
|
+
if (!BUILTIN_RELATIONS.has(edge.relation) && !declared.has(edge.relation)) {
|
|
1319
|
+
reader.tally(
|
|
1320
|
+
"validation-error",
|
|
1321
|
+
OBO_ISSUE.UNDECLARED,
|
|
1322
|
+
"a relation no Typedef declares",
|
|
1323
|
+
edge.relation,
|
|
1324
|
+
edge.line,
|
|
1325
|
+
);
|
|
1326
|
+
}
|
|
1327
|
+
if (!kept.has(edge.target)) {
|
|
1328
|
+
if (!danglingSet.has(edge.target)) {
|
|
1329
|
+
danglingSet.add(edge.target);
|
|
1330
|
+
dangling.push(edge.target);
|
|
1331
|
+
}
|
|
1332
|
+
if (!addMissingNodes) {
|
|
1333
|
+
danglingEdges++;
|
|
1334
|
+
report.counts.skippedEdges++;
|
|
1335
|
+
continue;
|
|
1336
|
+
}
|
|
1337
|
+
}
|
|
1338
|
+
edges.push({ source: record.id, edge });
|
|
1339
|
+
}
|
|
1340
|
+
}
|
|
1341
|
+
reader.flushTallies();
|
|
1342
|
+
reportDangling(records, dangling, addMissingNodes, danglingEdges, report);
|
|
1343
|
+
if (dropped.size > 0) {
|
|
1344
|
+
report.warning(
|
|
1345
|
+
"merged",
|
|
1346
|
+
OBO_ISSUE.OBSOLETE_DROPPED,
|
|
1347
|
+
`${dropped.size} obsolete term(s) and ${droppedEdges} edge(s) to or from them were left out (obsolete: "drop")`,
|
|
1348
|
+
{ element: [...dropped][0] },
|
|
1349
|
+
);
|
|
1350
|
+
}
|
|
1351
|
+
return {
|
|
1352
|
+
nodes: records.filter((record) => kept.has(record.id)),
|
|
1353
|
+
placeholders: addMissingNodes ? dangling : [],
|
|
1354
|
+
edges,
|
|
1355
|
+
};
|
|
1356
|
+
}
|
|
1357
|
+
|
|
1358
|
+
/**
|
|
1359
|
+
* Record the one W_DANGLING_REFERENCE of an import: the count, the first targets, and for a target
|
|
1360
|
+
* that is another term's alt_id, the term it is an alt_id of.
|
|
1361
|
+
* @param records - every node record
|
|
1362
|
+
* @param dangling - the undeclared targets, in first-reference order
|
|
1363
|
+
* @param addMissingNodes - whether they became placeholder nodes
|
|
1364
|
+
* @param droppedEdges - the edges dropped under addMissingNodes false
|
|
1365
|
+
* @param report - the report
|
|
1366
|
+
*/
|
|
1367
|
+
function reportDangling(
|
|
1368
|
+
records: readonly NodeRecord[],
|
|
1369
|
+
dangling: readonly string[],
|
|
1370
|
+
addMissingNodes: boolean,
|
|
1371
|
+
droppedEdges: number,
|
|
1372
|
+
report: ImportReportBuilder,
|
|
1373
|
+
): void {
|
|
1374
|
+
if (dangling.length === 0) {
|
|
1375
|
+
return;
|
|
1376
|
+
}
|
|
1377
|
+
const altOf = new Map<string, string>();
|
|
1378
|
+
for (const record of records) {
|
|
1379
|
+
for (const alt of (record.lists.get("alt_id") ?? []) as string[]) {
|
|
1380
|
+
altOf.set(alt, record.id);
|
|
1381
|
+
}
|
|
1382
|
+
}
|
|
1383
|
+
const shown = dangling.slice(0, 5).map((id) => {
|
|
1384
|
+
const primary = altOf.get(id);
|
|
1385
|
+
return primary === undefined ? id : `${id} (an alt_id of ${primary})`;
|
|
1386
|
+
});
|
|
1387
|
+
const more = dangling.length > 5 ? `, ... (${dangling.length - 5} more)` : "";
|
|
1388
|
+
const action = addMissingNodes
|
|
1389
|
+
? "each became a placeholder node (graphty.placeholder)"
|
|
1390
|
+
: `${droppedEdges} edge(s) to them were dropped (addMissingNodes false)`;
|
|
1391
|
+
report.warning(
|
|
1392
|
+
"validation-error",
|
|
1393
|
+
OBO_ISSUE.DANGLING_REFERENCE,
|
|
1394
|
+
`${dangling.length} target(s) no frame declares: ${shown.join(", ")}${more}; ${action}`,
|
|
1395
|
+
{ element: dangling[0] },
|
|
1396
|
+
);
|
|
1397
|
+
}
|
|
1398
|
+
|
|
1399
|
+
/** Pushes ids into a sink: the id coercion rule and the cancellation check. */
|
|
1400
|
+
class Pusher {
|
|
1401
|
+
readonly sink: GraphSink;
|
|
1402
|
+
|
|
1403
|
+
readonly report: ImportReportBuilder;
|
|
1404
|
+
|
|
1405
|
+
private readonly coercer: IdCoercer;
|
|
1406
|
+
|
|
1407
|
+
private readonly signal: AbortSignal | null;
|
|
1408
|
+
|
|
1409
|
+
private since = 0;
|
|
1410
|
+
|
|
1411
|
+
/**
|
|
1412
|
+
* Create a pusher.
|
|
1413
|
+
* @param sink - the sink
|
|
1414
|
+
* @param report - the report
|
|
1415
|
+
* @param options - the resolved common options
|
|
1416
|
+
*/
|
|
1417
|
+
constructor(sink: GraphSink, report: ImportReportBuilder, options: ReturnType<typeof resolveImportOptions>) {
|
|
1418
|
+
this.sink = sink;
|
|
1419
|
+
this.report = report;
|
|
1420
|
+
this.coercer = new IdCoercer(options.ids);
|
|
1421
|
+
this.signal = options.signal;
|
|
1422
|
+
}
|
|
1423
|
+
|
|
1424
|
+
/**
|
|
1425
|
+
* An id text under the `ids` rule, reporting a merge.
|
|
1426
|
+
* @param text - the id as written
|
|
1427
|
+
* @returns the node id
|
|
1428
|
+
*/
|
|
1429
|
+
id(text: string): NodeId {
|
|
1430
|
+
const id = this.coercer.text(text);
|
|
1431
|
+
const merge = this.coercer.lastMerge;
|
|
1432
|
+
if (merge !== null) {
|
|
1433
|
+
this.report.warning(
|
|
1434
|
+
"coercion",
|
|
1435
|
+
OBO_ISSUE.ID_MERGED,
|
|
1436
|
+
`id text ${JSON.stringify(merge.text)} merged with ${JSON.stringify(merge.previousText)} as ${merge.id}`,
|
|
1437
|
+
{ element: text },
|
|
1438
|
+
);
|
|
1439
|
+
}
|
|
1440
|
+
return id;
|
|
1441
|
+
}
|
|
1442
|
+
|
|
1443
|
+
/**
|
|
1444
|
+
* Add a node, counting it, or record why it was refused.
|
|
1445
|
+
* @param text - the id as written
|
|
1446
|
+
* @param line - the line, when known
|
|
1447
|
+
* @returns the node index, or -1 when refused
|
|
1448
|
+
*/
|
|
1449
|
+
node(text: string, line?: number): number {
|
|
1450
|
+
this.tick();
|
|
1451
|
+
try {
|
|
1452
|
+
const id = this.id(text);
|
|
1453
|
+
const known = this.sink.indexOf(id) !== INVALID_INDEX;
|
|
1454
|
+
const index = this.sink.addNode(id);
|
|
1455
|
+
if (!known) {
|
|
1456
|
+
this.report.counts.nodes++;
|
|
1457
|
+
}
|
|
1458
|
+
return index;
|
|
1459
|
+
} catch (err) {
|
|
1460
|
+
this.report.recordError(err, { line, element: text });
|
|
1461
|
+
this.report.counts.skippedNodes++;
|
|
1462
|
+
return -1;
|
|
1463
|
+
}
|
|
1464
|
+
}
|
|
1465
|
+
|
|
1466
|
+
/** Check the cancellation signal every ABORT_CHECK_INTERVAL pushes. */
|
|
1467
|
+
tick(): void {
|
|
1468
|
+
if (++this.since >= ABORT_CHECK_INTERVAL) {
|
|
1469
|
+
this.since = 0;
|
|
1470
|
+
throwIfAborted(this.signal);
|
|
1471
|
+
}
|
|
1472
|
+
}
|
|
1473
|
+
}
|
|
1474
|
+
|
|
1475
|
+
/**
|
|
1476
|
+
* Write the records into the sink (design 4.2): the nodes with the vocabulary columns they use,
|
|
1477
|
+
* the placeholders of undeclared targets, then the edges.
|
|
1478
|
+
* @param reader - the finished reader
|
|
1479
|
+
* @param sink - the sink
|
|
1480
|
+
* @param report - the report
|
|
1481
|
+
* @param options - the resolved options
|
|
1482
|
+
* @param obo - the OBO options
|
|
1483
|
+
*/
|
|
1484
|
+
function writeGraph(
|
|
1485
|
+
reader: OboReader,
|
|
1486
|
+
sink: GraphSink,
|
|
1487
|
+
report: ImportReportBuilder,
|
|
1488
|
+
options: ReturnType<typeof resolveImportOptions>,
|
|
1489
|
+
obo: ResolvedOboOptions,
|
|
1490
|
+
): void {
|
|
1491
|
+
const plan = planGraph(reader, report, options.addMissingNodes, obo);
|
|
1492
|
+
const push = new Pusher(sink, report, options);
|
|
1493
|
+
const namespace = reader.namespaceDefault();
|
|
1494
|
+
// columns, declared only when used, in the vocabulary's order
|
|
1495
|
+
const used = new Set<string>(["type"]);
|
|
1496
|
+
for (const record of plan.nodes) {
|
|
1497
|
+
for (const name of [...record.values.keys(), ...record.lists.keys()]) {
|
|
1498
|
+
used.add(name);
|
|
1499
|
+
}
|
|
1500
|
+
}
|
|
1501
|
+
if (namespace !== null) {
|
|
1502
|
+
used.add("namespace");
|
|
1503
|
+
}
|
|
1504
|
+
const columns = new Map<string, ColumnHandle>();
|
|
1505
|
+
for (const name of plan.nodes.length > 0 ? Object.keys(OBO_NODE_COLUMNS) : []) {
|
|
1506
|
+
if (used.has(name)) {
|
|
1507
|
+
columns.set(name, declareResolved(sink, "node", oboColumnDecl("node", name), report).handle);
|
|
1508
|
+
}
|
|
1509
|
+
}
|
|
1510
|
+
sink.reserve(plan.nodes.length + plan.placeholders.length, plan.edges.length);
|
|
1511
|
+
for (const record of plan.nodes) {
|
|
1512
|
+
const index = push.node(record.id, record.line);
|
|
1513
|
+
if (index < 0) {
|
|
1514
|
+
continue;
|
|
1515
|
+
}
|
|
1516
|
+
const cells: [string, unknown][] = [["type", record.kind], ...record.values, ...record.lists];
|
|
1517
|
+
if (namespace !== null && !record.values.has("namespace")) {
|
|
1518
|
+
cells.push(["namespace", namespace]);
|
|
1519
|
+
}
|
|
1520
|
+
for (const [name, value] of cells) {
|
|
1521
|
+
const handle = columns.get(name);
|
|
1522
|
+
if (handle !== undefined) {
|
|
1523
|
+
sink.setNodeValue(handle, index, value);
|
|
1524
|
+
}
|
|
1525
|
+
}
|
|
1526
|
+
}
|
|
1527
|
+
if (plan.placeholders.length > 0) {
|
|
1528
|
+
const placeholder = declareResolved(sink, "node", oboColumnDecl("node", PLACEHOLDER_COLUMN), report).handle;
|
|
1529
|
+
for (const text of plan.placeholders) {
|
|
1530
|
+
const index = push.node(text);
|
|
1531
|
+
if (index >= 0) {
|
|
1532
|
+
sink.setNodeValue(placeholder, index, true);
|
|
1533
|
+
}
|
|
1534
|
+
}
|
|
1535
|
+
}
|
|
1536
|
+
writeEdges(plan, push, options);
|
|
1537
|
+
}
|
|
1538
|
+
|
|
1539
|
+
/**
|
|
1540
|
+
* Push the planned edges through the direction resolver, with their relation and qualifiers.
|
|
1541
|
+
* @param plan - the plan
|
|
1542
|
+
* @param push - the pusher
|
|
1543
|
+
* @param options - the resolved options
|
|
1544
|
+
*/
|
|
1545
|
+
function writeEdges(plan: GraphPlan, push: Pusher, options: ReturnType<typeof resolveImportOptions>): void {
|
|
1546
|
+
const { sink, report } = push;
|
|
1547
|
+
const direction = new DirectionResolver(sink, report, options.onMixedDirection);
|
|
1548
|
+
direction.setHeader(true);
|
|
1549
|
+
if (plan.edges.length === 0) {
|
|
1550
|
+
return;
|
|
1551
|
+
}
|
|
1552
|
+
const relation = declareResolved(sink, "edge", oboColumnDecl("edge", "relation"), report).handle;
|
|
1553
|
+
const qualifiers = plan.edges.some((e) => e.edge.qualifiers !== null)
|
|
1554
|
+
? declareResolved(sink, "edge", oboColumnDecl("edge", "qualifiers"), report).handle
|
|
1555
|
+
: null;
|
|
1556
|
+
for (const { source, edge } of plan.edges) {
|
|
1557
|
+
push.tick();
|
|
1558
|
+
const before = sink.edgeCount;
|
|
1559
|
+
try {
|
|
1560
|
+
const e = direction.addEdge(push.id(source), push.id(edge.target), "directed", undefined, {
|
|
1561
|
+
line: edge.line,
|
|
1562
|
+
element: source,
|
|
1563
|
+
});
|
|
1564
|
+
report.counts.edges += sink.edgeCount - before;
|
|
1565
|
+
if (e < 0) {
|
|
1566
|
+
continue;
|
|
1567
|
+
}
|
|
1568
|
+
sink.setEdgeValue(relation, e, edge.relation);
|
|
1569
|
+
if (qualifiers !== null && edge.qualifiers !== null) {
|
|
1570
|
+
sink.setEdgeValue(qualifiers, e, edge.qualifiers);
|
|
1571
|
+
}
|
|
1572
|
+
} catch (err) {
|
|
1573
|
+
report.recordError(err, { line: edge.line, element: source });
|
|
1574
|
+
report.counts.skippedEdges++;
|
|
1575
|
+
}
|
|
1576
|
+
}
|
|
1577
|
+
}
|
|
1578
|
+
|
|
1579
|
+
/**
|
|
1580
|
+
* Confidence that a head is OBO (design 1.5): after a BOM, blank lines and `!` comment lines, 0.9
|
|
1581
|
+
* when the first significant line is `format-version:` or a `[Term]` / `[Typedef]` / `[Instance]`
|
|
1582
|
+
* header, or such a header follows `tag: value` lines; 0.4 for a head of only `tag: value` lines.
|
|
1583
|
+
* @param head - the first bytes
|
|
1584
|
+
* @returns the confidence
|
|
1585
|
+
*/
|
|
1586
|
+
function sniffObo(head: Uint8Array): number {
|
|
1587
|
+
let text = new TextDecoder("utf-8").decode(head.subarray(0, SNIFF_BYTES));
|
|
1588
|
+
if (text.charCodeAt(0) === 0xfeff) {
|
|
1589
|
+
text = text.slice(1);
|
|
1590
|
+
}
|
|
1591
|
+
const lines = text.split(/\r\n|\r|\n|\f/);
|
|
1592
|
+
// the last line may be cut mid-way
|
|
1593
|
+
if (lines.length > 1 && head.byteLength >= SNIFF_BYTES) {
|
|
1594
|
+
lines.pop();
|
|
1595
|
+
}
|
|
1596
|
+
let tagLines = 0;
|
|
1597
|
+
for (const raw of lines) {
|
|
1598
|
+
const line = raw.trim();
|
|
1599
|
+
if (line.length === 0 || line.startsWith("!")) {
|
|
1600
|
+
continue;
|
|
1601
|
+
}
|
|
1602
|
+
if (/^\[(Term|Typedef|Instance)\]\s*(!.*)?$/.test(line)) {
|
|
1603
|
+
return 0.9;
|
|
1604
|
+
}
|
|
1605
|
+
if (tagLines === 0 && /^format-version\s*:/.test(line)) {
|
|
1606
|
+
return 0.9;
|
|
1607
|
+
}
|
|
1608
|
+
if (/^[A-Za-z][A-Za-z0-9_-]*\s*:(\s|$)/.test(line)) {
|
|
1609
|
+
tagLines++;
|
|
1610
|
+
continue;
|
|
1611
|
+
}
|
|
1612
|
+
return 0;
|
|
1613
|
+
}
|
|
1614
|
+
return tagLines > 0 ? 0.4 : 0;
|
|
1615
|
+
}
|
|
1616
|
+
|
|
1617
|
+
/**
|
|
1618
|
+
* The OBO importer plugin (design section 4).
|
|
1619
|
+
*/
|
|
1620
|
+
export const oboImporter: GraphImporter<OboImportOptions> = Object.freeze({
|
|
1621
|
+
format: "obo",
|
|
1622
|
+
extensions: Object.freeze([".obo"]),
|
|
1623
|
+
mimeTypes: Object.freeze(["text/obo", "application/obo"]),
|
|
1624
|
+
|
|
1625
|
+
/**
|
|
1626
|
+
* Confidence that the head is an OBO file.
|
|
1627
|
+
* @param head - the first bytes
|
|
1628
|
+
* @returns the confidence
|
|
1629
|
+
*/
|
|
1630
|
+
sniff(head: Uint8Array): number {
|
|
1631
|
+
return sniffObo(head);
|
|
1632
|
+
},
|
|
1633
|
+
|
|
1634
|
+
/**
|
|
1635
|
+
* Read an OBO file into the sink.
|
|
1636
|
+
* @param input - the text, bytes or stream
|
|
1637
|
+
* @param sink - the sink
|
|
1638
|
+
* @param options - format-specific and common options
|
|
1639
|
+
* @returns the import report; ImportError on a fatal error or beyond the error limit
|
|
1640
|
+
*/
|
|
1641
|
+
async import(
|
|
1642
|
+
input: ImportInput,
|
|
1643
|
+
sink: GraphSink,
|
|
1644
|
+
options?: OboImportOptions & CommonImportOptions,
|
|
1645
|
+
): Promise<ImportReport> {
|
|
1646
|
+
const resolved = resolveImportOptions(options, DEFAULTS);
|
|
1647
|
+
const obo = resolveOboOptions(options);
|
|
1648
|
+
const report = new ImportReportBuilder("obo", resolved.errorLimit);
|
|
1649
|
+
reportSinkOptions(sink, options, report, true);
|
|
1650
|
+
reportUnusedOptions(options, report, USED_OPTIONS);
|
|
1651
|
+
const reader = new OboReader(report, resolved.signal);
|
|
1652
|
+
const lines = new LineReader(input, report, resolved);
|
|
1653
|
+
let pending: string | null = null;
|
|
1654
|
+
for await (const physical of lines) {
|
|
1655
|
+
const { line } = lines;
|
|
1656
|
+
const text: string = pending === null ? physical : pending + physical;
|
|
1657
|
+
pending = null;
|
|
1658
|
+
if (endsWithContinuation(text)) {
|
|
1659
|
+
reader.tally(
|
|
1660
|
+
"coercion",
|
|
1661
|
+
OBO_ISSUE.DEPRECATED_SYNTAX,
|
|
1662
|
+
"a backslash line continuation was joined (deprecated in 1.4)",
|
|
1663
|
+
"\\",
|
|
1664
|
+
line,
|
|
1665
|
+
);
|
|
1666
|
+
pending = text.slice(0, -1);
|
|
1667
|
+
continue;
|
|
1668
|
+
}
|
|
1669
|
+
// the 1.4 grammar counts form feed as a line end; LineReader does not
|
|
1670
|
+
for (const piece of text.includes("\f") ? text.split("\f") : [text]) {
|
|
1671
|
+
reader.line(piece, line);
|
|
1672
|
+
}
|
|
1673
|
+
}
|
|
1674
|
+
if (pending !== null) {
|
|
1675
|
+
reader.line(pending, lines.line);
|
|
1676
|
+
}
|
|
1677
|
+
reader.finishFrame();
|
|
1678
|
+
if (!reader.significant) {
|
|
1679
|
+
report.fail(OBO_ISSUE.EMPTY_INPUT, "the input is empty");
|
|
1680
|
+
}
|
|
1681
|
+
throwIfAborted(resolved.signal);
|
|
1682
|
+
writeGraph(reader, sink, report, resolved, obo);
|
|
1683
|
+
const typedefs: Record<string, Record<string, string[]>> = {};
|
|
1684
|
+
for (const [id, record] of reader.typedefs) {
|
|
1685
|
+
typedefs[id] = record.raw;
|
|
1686
|
+
}
|
|
1687
|
+
const extra: Record<string, unknown> = { header: reader.headerRecord(), typedefs };
|
|
1688
|
+
if (reader.unknownFrames.length > 0) {
|
|
1689
|
+
extra.unknownFrames = reader.unknownFrames;
|
|
1690
|
+
}
|
|
1691
|
+
sink.setMeta({ ...reader.meta(), extra: { obo: extra } });
|
|
1692
|
+
throwIfAborted(resolved.signal);
|
|
1693
|
+
return report.finish();
|
|
1694
|
+
},
|
|
1695
|
+
});
|