@graphty/graph-io 0.3.16 → 0.3.18

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (111) hide show
  1. package/LICENSE +1 -1
  2. package/README.md +29 -3
  3. package/dist/chunks/{escape-D1f9-cwf.js → escape-D-gZWO26.js} +3 -2
  4. package/dist/chunks/{escape-D1f9-cwf.js.map → escape-D-gZWO26.js.map} +1 -1
  5. package/dist/chunks/{importer-D7ZcGCeb.js → importer-Br_QeAeE.js} +4 -3
  6. package/dist/chunks/{importer-D7ZcGCeb.js.map → importer-Br_QeAeE.js.map} +1 -1
  7. package/dist/chunks/{importer-CXEiicAN.js → importer-DHagxvDD.js} +4 -3
  8. package/dist/chunks/{importer-CXEiicAN.js.map → importer-DHagxvDD.js.map} +1 -1
  9. package/dist/chunks/{importer-C7mnGdr_.js → importer-Du5crN9l.js} +510 -26
  10. package/dist/chunks/importer-Du5crN9l.js.map +1 -0
  11. package/dist/chunks/importer-aNJfe0qu.js +1614 -0
  12. package/dist/chunks/importer-aNJfe0qu.js.map +1 -0
  13. package/dist/chunks/{importer-B8lsjFWx.js → importer-d0uQxFp6.js} +4 -3
  14. package/dist/chunks/{importer-B8lsjFWx.js.map → importer-d0uQxFp6.js.map} +1 -1
  15. package/dist/chunks/ontology-BnrJ4I98.js +113 -0
  16. package/dist/chunks/ontology-BnrJ4I98.js.map +1 -0
  17. package/dist/chunks/{records-IHsCfv7s.js → records-Bk9jgodz.js} +2 -2
  18. package/dist/chunks/{records-IHsCfv7s.js.map → records-Bk9jgodz.js.map} +1 -1
  19. package/dist/chunks/{writer-BtWpUaiH.js → report-BOk0p5y8.js} +181 -912
  20. package/dist/chunks/report-BOk0p5y8.js.map +1 -0
  21. package/dist/chunks/writer-GAdltGmC.js +827 -0
  22. package/dist/chunks/writer-GAdltGmC.js.map +1 -0
  23. package/dist/csv.js +4 -3
  24. package/dist/csv.js.map +1 -1
  25. package/dist/dot.js +1 -1
  26. package/dist/gexf.js +3 -2
  27. package/dist/gexf.js.map +1 -1
  28. package/dist/gml.js +3 -2
  29. package/dist/gml.js.map +1 -1
  30. package/dist/graph-io.js +191 -134
  31. package/dist/graph-io.js.map +1 -1
  32. package/dist/graphml.js +1 -1
  33. package/dist/json.js +1 -1
  34. package/dist/neo4j.js +4 -3
  35. package/dist/neo4j.js.map +1 -1
  36. package/dist/obo.d.ts +1 -0
  37. package/dist/obo.js +6 -0
  38. package/dist/obo.js.map +1 -0
  39. package/dist/pajek.js +1 -1
  40. package/dist/src/common/codes.d.ts +43 -0
  41. package/dist/src/common/codes.d.ts.map +1 -1
  42. package/dist/src/common/codes.js +43 -0
  43. package/dist/src/common/codes.js.map +1 -1
  44. package/dist/src/common/input.d.ts +9 -0
  45. package/dist/src/common/input.d.ts.map +1 -1
  46. package/dist/src/common/input.js +16 -0
  47. package/dist/src/common/input.js.map +1 -1
  48. package/dist/src/common/ontology.d.ts +59 -0
  49. package/dist/src/common/ontology.d.ts.map +1 -0
  50. package/dist/src/common/ontology.js +147 -0
  51. package/dist/src/common/ontology.js.map +1 -0
  52. package/dist/src/common/options.d.ts +12 -1
  53. package/dist/src/common/options.d.ts.map +1 -1
  54. package/dist/src/common/options.js +51 -1
  55. package/dist/src/common/options.js.map +1 -1
  56. package/dist/src/formats/json/dialect.d.ts +8 -6
  57. package/dist/src/formats/json/dialect.d.ts.map +1 -1
  58. package/dist/src/formats/json/dialect.js +26 -3
  59. package/dist/src/formats/json/dialect.js.map +1 -1
  60. package/dist/src/formats/json/importer.d.ts +324 -7
  61. package/dist/src/formats/json/importer.d.ts.map +1 -1
  62. package/dist/src/formats/json/importer.js +154 -23
  63. package/dist/src/formats/json/importer.js.map +1 -1
  64. package/dist/src/formats/json/obographs.d.ts +21 -0
  65. package/dist/src/formats/json/obographs.d.ts.map +1 -0
  66. package/dist/src/formats/json/obographs.js +476 -0
  67. package/dist/src/formats/json/obographs.js.map +1 -0
  68. package/dist/src/formats/obo/importer.d.ts +90 -0
  69. package/dist/src/formats/obo/importer.d.ts.map +1 -0
  70. package/dist/src/formats/obo/importer.js +1248 -0
  71. package/dist/src/formats/obo/importer.js.map +1 -0
  72. package/dist/src/formats/obo/index.d.ts +7 -0
  73. package/dist/src/formats/obo/index.d.ts.map +1 -0
  74. package/dist/src/formats/obo/index.js +7 -0
  75. package/dist/src/formats/obo/index.js.map +1 -0
  76. package/dist/src/formats/obo/syntax.d.ts +121 -0
  77. package/dist/src/formats/obo/syntax.d.ts.map +1 -0
  78. package/dist/src/formats/obo/syntax.js +424 -0
  79. package/dist/src/formats/obo/syntax.js.map +1 -0
  80. package/dist/src/index.d.ts +5 -4
  81. package/dist/src/index.d.ts.map +1 -1
  82. package/dist/src/index.js +4 -3
  83. package/dist/src/index.js.map +1 -1
  84. package/dist/src/registry.d.ts +21 -3
  85. package/dist/src/registry.d.ts.map +1 -1
  86. package/dist/src/registry.js +32 -1
  87. package/dist/src/registry.js.map +1 -1
  88. package/dist/src/sniff.d.ts +1 -1
  89. package/dist/src/sniff.d.ts.map +1 -1
  90. package/dist/src/sniff.js +23 -3
  91. package/dist/src/sniff.js.map +1 -1
  92. package/dist/src/types.d.ts +31 -0
  93. package/dist/src/types.d.ts.map +1 -1
  94. package/dist/src/types.js.map +1 -1
  95. package/package.json +6 -1
  96. package/src/common/codes.ts +58 -0
  97. package/src/common/input.ts +16 -0
  98. package/src/common/ontology.ts +169 -0
  99. package/src/common/options.ts +78 -2
  100. package/src/formats/json/dialect.ts +37 -7
  101. package/src/formats/json/importer.ts +206 -28
  102. package/src/formats/json/obographs.ts +563 -0
  103. package/src/formats/obo/importer.ts +1695 -0
  104. package/src/formats/obo/index.ts +7 -0
  105. package/src/formats/obo/syntax.ts +466 -0
  106. package/src/index.ts +6 -0
  107. package/src/registry.ts +38 -3
  108. package/src/sniff.ts +35 -5
  109. package/src/types.ts +40 -1
  110. package/dist/chunks/importer-C7mnGdr_.js.map +0 -1
  111. package/dist/chunks/writer-BtWpUaiH.js.map +0 -1
@@ -0,0 +1,7 @@
1
+ /**
2
+ * The OBO subpath entry (`@graphty/graph-io/obo`, design section 1.5): the importer of the OBO
3
+ * flat file format (the Gene Ontology and the OBO Foundry ontologies), its options and its issue
4
+ * codes. OBO is read-only: there is no exporter.
5
+ */
6
+
7
+ export { OBO_ISSUE, oboImporter, type OboImportOptions } from "./importer.js";
@@ -0,0 +1,466 @@
1
+ /**
2
+ * The lexical layer of the OBO importer (research-obo.md section 4.1): the tag-value line, the
3
+ * hidden `!` comment, the trailing `{name="value", ...}` qualifier block, the escapes of the OBO
4
+ * guides, quoted strings and bracketed xref lists. Pure functions over one logical line; the
5
+ * importer joins backslash continuations and splits form feeds before calling them.
6
+ *
7
+ * Every function works on the RAW text (escapes still in place) and unescapes only the pieces it
8
+ * hands back, so an escaped quote, colon, comma, brace or `!` never ends a construct.
9
+ */
10
+
11
+ /** The OBO escapes with a meaning of their own; any other `\x` is `x` (the guides). */
12
+ const ESCAPES: Readonly<Record<string, string>> = Object.freeze({
13
+ n: "\n",
14
+ t: "\t",
15
+ // the guides read `\W` as a space; the 1.4 BNF (and fastobo) as the letter W (design 4.1)
16
+ W: " ",
17
+ });
18
+
19
+ /**
20
+ * Remove the OBO escapes: `\n` newline, `\t` tab, `\W` space, any other `\x` the character x; a
21
+ * backslash at the very end is dropped.
22
+ * @param text - raw text
23
+ * @returns the unescaped text
24
+ */
25
+ export function unescapeObo(text: string): string {
26
+ if (!text.includes("\\")) {
27
+ return text;
28
+ }
29
+ let out = "";
30
+ for (let i = 0; i < text.length; i++) {
31
+ const ch = text[i];
32
+ if (ch !== "\\") {
33
+ out += ch;
34
+ continue;
35
+ }
36
+ i++;
37
+ if (i < text.length) {
38
+ const next = text[i];
39
+ out += ESCAPES[next] ?? next;
40
+ }
41
+ }
42
+ return out;
43
+ }
44
+
45
+ /**
46
+ * Whether the character at `index` is escaped (preceded by an odd run of backslashes).
47
+ * @param text - the text
48
+ * @param index - the character's index
49
+ * @returns true when escaped
50
+ */
51
+ function isEscaped(text: string, index: number): boolean {
52
+ let run = 0;
53
+ for (let i = index - 1; i >= 0 && text[i] === "\\"; i--) {
54
+ run++;
55
+ }
56
+ return run % 2 === 1;
57
+ }
58
+
59
+ /**
60
+ * Whether a raw line ends with an unescaped backslash (the 1.0 / 1.2 line continuation).
61
+ * @param line - the raw line, trailing whitespace included
62
+ * @returns true when the line continues on the next one
63
+ */
64
+ export function endsWithContinuation(line: string): boolean {
65
+ return line.endsWith("\\") && !isEscaped(line, line.length - 1);
66
+ }
67
+
68
+ /** A tag-value line split at its first unescaped colon. */
69
+ interface TagValue {
70
+ /** The tag, unescaped and trimmed. */
71
+ readonly tag: string;
72
+ /** The raw text after the colon, leading whitespace removed. */
73
+ readonly rest: string;
74
+ }
75
+
76
+ /**
77
+ * Split a line at its first unescaped colon (values contain colons freely: `is_a: GO:0000001`).
78
+ * @param line - the raw line, trimmed
79
+ * @returns the tag and the rest, or null when the line has no colon
80
+ */
81
+ export function splitTagValue(line: string): TagValue | null {
82
+ for (let i = 0; i < line.length; i++) {
83
+ if (line[i] === "\\") {
84
+ i++;
85
+ continue;
86
+ }
87
+ if (line[i] === ":") {
88
+ return { tag: unescapeObo(line.slice(0, i).trim()), rest: line.slice(i + 1).trimStart() };
89
+ }
90
+ }
91
+ return null;
92
+ }
93
+
94
+ /**
95
+ * Strip the hidden comment: an unescaped `!` outside quotes that starts the value or follows
96
+ * whitespace begins a comment running to the end of the line. A `!` glued to the text before it
97
+ * (`#!/`, `Hello!`) is text, so URLs and names are never cut. A quote opens a string only where a
98
+ * token starts, as tokenize() reads it, so the `"` of an unquoted `5" pipe` is text and does not
99
+ * hide the comment after it.
100
+ * @param rest - the raw value
101
+ * @returns the value before the comment, trailing whitespace removed
102
+ */
103
+ export function stripComment(rest: string): string {
104
+ let quoted = false;
105
+ for (let i = 0; i < rest.length; i++) {
106
+ const ch = rest[i];
107
+ if (ch === "\\") {
108
+ i++;
109
+ continue;
110
+ }
111
+ const atStart = i === 0 || rest[i - 1] === " " || rest[i - 1] === "\t";
112
+ if (ch === '"' && (quoted || atStart)) {
113
+ quoted = !quoted;
114
+ } else if (ch === "!" && !quoted && atStart) {
115
+ return rest.slice(0, i).trimEnd();
116
+ }
117
+ }
118
+ return rest.trimEnd();
119
+ }
120
+
121
+ /** The qualifiers of a clause: name to value (a list when a name repeats). */
122
+ export type Qualifiers = Record<string, string | string[]>;
123
+
124
+ /** A value with its trailing qualifier blocks split off. */
125
+ interface QualifiedValue {
126
+ /** The raw value without the blocks, trimmed. */
127
+ readonly value: string;
128
+ /** The qualifiers, or null when the value has none. */
129
+ readonly qualifiers: Qualifiers | null;
130
+ /** A block that closes the value but does not parse, kept in `value` as text. */
131
+ readonly badBlock: string | null;
132
+ }
133
+
134
+ /**
135
+ * Split the trailing qualifier blocks off a value: a `{...}` is a block only when it closes the
136
+ * value (after the comment was stripped); several blocks (`{a="1"}{b="2"}`, 1.2 examples) are
137
+ * merged. A closing brace whose block does not parse is left in the value and reported.
138
+ * @param value - the raw value without its comment
139
+ * @returns the value and its qualifiers
140
+ */
141
+ export function splitQualifiers(value: string): QualifiedValue {
142
+ let rest = value;
143
+ let qualifiers: Qualifiers | null = null;
144
+ for (;;) {
145
+ if (!rest.endsWith("}") || isEscaped(rest, rest.length - 1)) {
146
+ return { value: rest, qualifiers, badBlock: null };
147
+ }
148
+ const open = openingBrace(rest);
149
+ const parsed = open < 0 ? null : parseQualifiers(rest.slice(open + 1, -1));
150
+ if (parsed === null) {
151
+ return { value: rest, qualifiers, badBlock: open < 0 ? rest : rest.slice(open) };
152
+ }
153
+ // a block read right to left: an earlier block's names come first
154
+ qualifiers = qualifiers === null ? parsed : mergeQualifiers(parsed, qualifiers);
155
+ rest = rest.slice(0, open).trimEnd();
156
+ }
157
+ }
158
+
159
+ /**
160
+ * The index of the unescaped `{` outside quotes that opens the block closing the value.
161
+ * @param value - a raw value ending with `}`
162
+ * @returns the index, or -1 when there is none
163
+ */
164
+ function openingBrace(value: string): number {
165
+ let quoted = false;
166
+ let open = -1;
167
+ for (let i = 0; i < value.length - 1; i++) {
168
+ const ch = value[i];
169
+ if (ch === "\\") {
170
+ i++;
171
+ continue;
172
+ }
173
+ if (ch === '"') {
174
+ quoted = !quoted;
175
+ } else if (!quoted && ch === "{") {
176
+ open = i;
177
+ } else if (!quoted && ch === "}") {
178
+ open = -1;
179
+ }
180
+ }
181
+ return quoted ? -1 : open;
182
+ }
183
+
184
+ /**
185
+ * Merge two qualifier records, a repeated name becoming a list.
186
+ * @param first - the earlier record
187
+ * @param second - the later record
188
+ * @returns the merged record
189
+ */
190
+ function mergeQualifiers(first: Qualifiers, second: Qualifiers): Qualifiers {
191
+ const out: Qualifiers = { ...first };
192
+ for (const [name, value] of Object.entries(second)) {
193
+ addQualifier(out, name, value);
194
+ }
195
+ return out;
196
+ }
197
+
198
+ /**
199
+ * Add one qualifier, turning a repeated name into a list.
200
+ * @param record - the record
201
+ * @param name - the name
202
+ * @param value - the value (or values)
203
+ */
204
+ function addQualifier(record: Qualifiers, name: string, value: string | string[]): void {
205
+ if (!Object.prototype.hasOwnProperty.call(record, name)) {
206
+ record[name] = value;
207
+ return;
208
+ }
209
+ const before = record[name];
210
+ record[name] = [...(Array.isArray(before) ? before : [before]), ...(Array.isArray(value) ? value : [value])];
211
+ }
212
+
213
+ /**
214
+ * Parse the inside of a qualifier block: `name="value"` pairs separated by commas, the 1.2
215
+ * unquoted form (`source=PMID:1`) and missing spaces accepted. A name is everything up to `=`
216
+ * (qualifier names may be IRIs with colons).
217
+ * @param inner - the text between the braces
218
+ * @returns the qualifiers, or null when the block does not parse (a pair without `=`, an unclosed quote)
219
+ */
220
+ export function parseQualifiers(inner: string): Qualifiers | null {
221
+ const out: Qualifiers = Object.create(null) as Qualifiers;
222
+ let i = 0;
223
+ const n = inner.length;
224
+ let any = false;
225
+ while (i < n) {
226
+ while (i < n && (inner[i] === " " || inner[i] === "\t" || inner[i] === ",")) {
227
+ i++;
228
+ }
229
+ if (i >= n) {
230
+ break;
231
+ }
232
+ const eq = inner.indexOf("=", i);
233
+ if (eq < 0) {
234
+ return null;
235
+ }
236
+ const name = unescapeObo(inner.slice(i, eq).trim());
237
+ if (name.length === 0 || name.includes('"')) {
238
+ return null;
239
+ }
240
+ i = eq + 1;
241
+ while (i < n && (inner[i] === " " || inner[i] === "\t")) {
242
+ i++;
243
+ }
244
+ let value: string;
245
+ if (inner[i] === '"') {
246
+ const end = closingQuote(inner, i + 1);
247
+ if (end < 0) {
248
+ return null;
249
+ }
250
+ value = unescapeObo(inner.slice(i + 1, end));
251
+ i = end + 1;
252
+ } else {
253
+ let end = i;
254
+ while (end < n && inner[end] !== ",") {
255
+ end += inner[end] === "\\" ? 2 : 1;
256
+ }
257
+ value = unescapeObo(inner.slice(i, end).trim());
258
+ i = end;
259
+ }
260
+ addQualifier(out, name, value);
261
+ any = true;
262
+ }
263
+ return any ? { ...out } : null;
264
+ }
265
+
266
+ /**
267
+ * The index of the unescaped closing quote.
268
+ * @param text - the text
269
+ * @param from - the index after the opening quote
270
+ * @returns the index, or -1 when the text ends first
271
+ */
272
+ function closingQuote(text: string, from: number): number {
273
+ for (let i = from; i < text.length; i++) {
274
+ if (text[i] === "\\") {
275
+ i++;
276
+ } else if (text[i] === '"') {
277
+ return i;
278
+ }
279
+ }
280
+ return -1;
281
+ }
282
+
283
+ /** One token of a value. */
284
+ export interface Token {
285
+ /** A bare word, a quoted string or a bracketed list. */
286
+ readonly kind: "word" | "quoted" | "list";
287
+ /** The token's text: unescaped for a word or a quoted string, raw inside the brackets for a list. */
288
+ readonly text: string;
289
+ /** Whether the quote or bracket was never closed (the token runs to the end of the value). */
290
+ readonly unterminated: boolean;
291
+ }
292
+
293
+ /**
294
+ * Split a value (comment and qualifiers removed) into words, quoted strings and bracketed lists.
295
+ * Whitespace separates tokens; an escaped space belongs to its word.
296
+ * @param value - the raw value
297
+ * @returns the tokens
298
+ */
299
+ export function tokenize(value: string): Token[] {
300
+ const tokens: Token[] = [];
301
+ const n = value.length;
302
+ let i = 0;
303
+ while (i < n) {
304
+ const ch = value[i];
305
+ if (ch === " " || ch === "\t") {
306
+ i++;
307
+ continue;
308
+ }
309
+ if (ch === '"') {
310
+ const end = closingQuote(value, i + 1);
311
+ const stop = end < 0 ? n : end;
312
+ // a 1.4 language tag glued to the closing quote ("chat"@fr) is kept with the text, as an
313
+ // unquoted chat@fr is (research-obo.md 4.1: no language is split out)
314
+ let after = stop + 1;
315
+ if (end >= 0 && value[after] === "@") {
316
+ while (after < n && value[after] !== " " && value[after] !== "\t") {
317
+ after++;
318
+ }
319
+ }
320
+ const suffix = end < 0 ? "" : value.slice(stop + 1, after);
321
+ tokens.push({
322
+ kind: "quoted",
323
+ text: unescapeObo(value.slice(i + 1, stop)) + suffix,
324
+ unterminated: end < 0,
325
+ });
326
+ i = after;
327
+ continue;
328
+ }
329
+ if (ch === "[") {
330
+ const end = closingBracket(value, i + 1);
331
+ const stop = end < 0 ? n : end;
332
+ tokens.push({ kind: "list", text: value.slice(i + 1, stop), unterminated: end < 0 });
333
+ i = stop + 1;
334
+ continue;
335
+ }
336
+ let end = i;
337
+ while (end < n && value[end] !== " " && value[end] !== "\t") {
338
+ end += value[end] === "\\" ? 2 : 1;
339
+ }
340
+ tokens.push({ kind: "word", text: unescapeObo(value.slice(i, Math.min(end, n))), unterminated: false });
341
+ i = end;
342
+ }
343
+ return tokens;
344
+ }
345
+
346
+ /**
347
+ * The index of the unescaped `]` outside quotes closing a list.
348
+ * @param text - the text
349
+ * @param from - the index after `[`
350
+ * @returns the index, or -1 when the text ends first
351
+ */
352
+ function closingBracket(text: string, from: number): number {
353
+ let quoted = false;
354
+ for (let i = from; i < text.length; i++) {
355
+ const ch = text[i];
356
+ if (ch === "\\") {
357
+ i++;
358
+ } else if (ch === '"') {
359
+ quoted = !quoted;
360
+ } else if (ch === "]" && !quoted) {
361
+ return i;
362
+ }
363
+ }
364
+ return -1;
365
+ }
366
+
367
+ /** One cross-reference: `ID "description" {qualifiers}`. */
368
+ export interface Xref {
369
+ /** The id (may hold spaces: `NIST Chemistry WebBook:110-63-4`, which owlapi reads). */
370
+ readonly id: string;
371
+ /** The description, or null. */
372
+ readonly description: string | null;
373
+ /** Per-xref qualifiers (OBO 1.2 inside a list), or null. */
374
+ readonly qualifiers: Qualifiers | null;
375
+ }
376
+
377
+ /**
378
+ * Read one cross-reference: the id is the text before the first unescaped quote, the quoted
379
+ * string its description, a closing block its qualifiers.
380
+ * @param raw - the raw text of one xref
381
+ * @returns the xref, or null when it has no id
382
+ */
383
+ export function parseXref(raw: string): Xref | null {
384
+ const { value, qualifiers } = splitQualifiers(raw.trim());
385
+ let quote = -1;
386
+ for (let i = 0; i < value.length; i++) {
387
+ if (value[i] === "\\") {
388
+ i++;
389
+ } else if (value[i] === '"') {
390
+ quote = i;
391
+ break;
392
+ }
393
+ }
394
+ const id = unescapeObo((quote < 0 ? value : value.slice(0, quote)).trim());
395
+ if (id.length === 0) {
396
+ return null;
397
+ }
398
+ let description: string | null = null;
399
+ if (quote >= 0) {
400
+ const end = closingQuote(value, quote + 1);
401
+ description = unescapeObo(value.slice(quote + 1, end < 0 ? value.length : end));
402
+ }
403
+ return { id, description, qualifiers };
404
+ }
405
+
406
+ /**
407
+ * Split the inside of a bracketed list into xrefs at the unescaped commas outside quotes and
408
+ * braces (a solitary xref needs no escaping of its commas only outside a list).
409
+ * @param inner - the raw text between the brackets
410
+ * @returns the xrefs, in order
411
+ */
412
+ export function parseXrefList(inner: string): Xref[] {
413
+ const out: Xref[] = [];
414
+ let quoted = false;
415
+ let depth = 0;
416
+ let start = 0;
417
+ const flush = (end: number): void => {
418
+ const xref = parseXref(inner.slice(start, end));
419
+ if (xref !== null) {
420
+ out.push(xref);
421
+ }
422
+ };
423
+ for (let i = 0; i < inner.length; i++) {
424
+ const ch = inner[i];
425
+ if (ch === "\\") {
426
+ i++;
427
+ } else if (ch === '"') {
428
+ quoted = !quoted;
429
+ } else if (!quoted && ch === "{") {
430
+ depth++;
431
+ } else if (!quoted && ch === "}") {
432
+ depth = Math.max(0, depth - 1);
433
+ } else if (!quoted && depth === 0 && ch === ",") {
434
+ flush(i);
435
+ start = i + 1;
436
+ }
437
+ }
438
+ flush(inner.length);
439
+ return out;
440
+ }
441
+
442
+ /**
443
+ * Whether the raw value holds an unescaped `{` or `}` outside quotes and outside a bracketed xref
444
+ * list (a brace mid-value that is not a closing qualifier block is literal text the writer should
445
+ * have escaped).
446
+ * @param value - the raw value, qualifiers already split off
447
+ * @returns true when a stray brace is present
448
+ */
449
+ export function hasStrayBrace(value: string): boolean {
450
+ let quoted = false;
451
+ let listed = false;
452
+ for (let i = 0; i < value.length; i++) {
453
+ const ch = value[i];
454
+ if (ch === "\\") {
455
+ i++;
456
+ } else if (ch === '"') {
457
+ quoted = !quoted;
458
+ } else if (!quoted && (ch === "[" || ch === "]")) {
459
+ // inside an xref list a brace opens that xref's qualifiers (OBO 1.2)
460
+ listed = ch === "[";
461
+ } else if (!quoted && !listed && (ch === "{" || ch === "}")) {
462
+ return true;
463
+ }
464
+ }
465
+ return false;
466
+ }
package/src/index.ts CHANGED
@@ -13,8 +13,10 @@ export {
13
13
  type CommonExportOptions,
14
14
  type CommonImportOptions,
15
15
  type ExportCapabilities,
16
+ type GraphChoiceOptions,
16
17
  type GraphExporter,
17
18
  type GraphImporter,
19
+ type GraphListing,
18
20
  ImportError,
19
21
  type ImportInput,
20
22
  type ImportIssue,
@@ -44,6 +46,7 @@ export {
44
46
  importGraph,
45
47
  type ImportGraphOptions,
46
48
  type ImportGraphResult,
49
+ listGraphs,
47
50
  registry,
48
51
  sniff,
49
52
  UNKNOWN_FORMAT_CODE,
@@ -134,6 +137,7 @@ export {
134
137
  ORIGINAL_ID_COLUMN,
135
138
  TYPE_COLUMN,
136
139
  } from "./formats/neo4j/index.js";
140
+ export { OBO_ISSUE, oboImporter, type OboImportOptions } from "./formats/obo/index.js";
137
141
  export {
138
142
  PAJEK_ISSUE,
139
143
  PAJEK_LOSS,
@@ -192,6 +196,7 @@ export {
192
196
  isCanonicalIntegerText,
193
197
  } from "./common/ids.js";
194
198
  export {
199
+ decodeEntryName,
195
200
  inputLength,
196
201
  INVALID_UTF8_CODE,
197
202
  isImportInput,
@@ -202,6 +207,7 @@ export {
202
207
  throwIfAborted,
203
208
  } from "./common/input.js";
204
209
  export {
210
+ chooseGraph,
205
211
  DEFAULT_ERROR_LIMIT,
206
212
  type ImportFormatDefaults,
207
213
  reportSinkOptions,
package/src/registry.ts CHANGED
@@ -25,13 +25,16 @@ import { gmlExporter, gmlImporter } from "./formats/gml/index.js";
25
25
  import { graphmlExporter, graphmlImporter } from "./formats/graphml/index.js";
26
26
  import { jsonExporter, jsonImporter } from "./formats/json/index.js";
27
27
  import { neo4jExporter, neo4jImporter } from "./formats/neo4j/index.js";
28
+ import { oboImporter } from "./formats/obo/index.js";
28
29
  import { pajekExporter, pajekImporter } from "./formats/pajek/index.js";
29
30
  import { rankFormats, SNIFF_HEAD_BYTES, type SniffHints, type SniffResult } from "./sniff.js";
30
31
  import {
31
32
  type CommonExportOptions,
32
33
  type CommonImportOptions,
34
+ type GraphChoiceOptions,
33
35
  type GraphExporter,
34
36
  type GraphImporter,
37
+ type GraphListing,
35
38
  type ImportInput,
36
39
  type ImportReport,
37
40
  type LossNote,
@@ -50,9 +53,10 @@ export type BuilderSeed = Omit<
50
53
  * The options of importGraph(): the common import options (which also seed the registry's
51
54
  * builder, design section 8.4), the format choice and the hints sniffing uses, the builder and
52
55
  * freeze options, and any format-specific option (`delimiter`, `dialect`, ...) passed through to
53
- * the importer unchanged.
56
+ * the importer unchanged. `graphIndex` / `graphName` choose one graph of an input that holds
57
+ * several, for the formats that list their graphs.
54
58
  */
55
- export interface ImportGraphOptions extends CommonImportOptions {
59
+ export interface ImportGraphOptions extends CommonImportOptions, GraphChoiceOptions {
56
60
  /** The format name, or "auto" (default) to sniff it from the filename, MIME type and content. */
57
61
  readonly format?: string | undefined;
58
62
  /** The file name or path the input came from, a hint for sniffing. */
@@ -269,6 +273,26 @@ export class FormatRegistry {
269
273
  return reports.map((report, i) => result(chosen, builders[i], report, options));
270
274
  }
271
275
 
276
+ /**
277
+ * The graphs of an input that can hold several, without importing them (a Cytoscape session's
278
+ * networks, a JGF `graphs` array), so a caller can offer a choice and then pass `graphIndex`
279
+ * or `graphName` to importGraph(). The format is named or sniffed as by importGraph().
280
+ * @param input - the text, bytes, stream or chunks to read
281
+ * @param options - the format, hints, common and format-specific import options
282
+ * @returns one listing per graph, in document order; null when the format's importer does not
283
+ * list its graphs (importGraph() then reads the first)
284
+ */
285
+ async listGraphs(input: ImportInput, options: ImportGraphOptions = {}): Promise<readonly GraphListing[] | null> {
286
+ const chosen = await this.choose(input, options);
287
+ try {
288
+ return chosen.importer.listGraphs === undefined
289
+ ? null
290
+ : await chosen.importer.listGraphs(chosen.source, importerOptions(options));
291
+ } finally {
292
+ await chosen.peeked?.close();
293
+ }
294
+ }
295
+
272
296
  /**
273
297
  * The importer for an input: the named format, or the sniffed one.
274
298
  * @param input - the input
@@ -349,7 +373,8 @@ export function createRegistry(): FormatRegistry {
349
373
  .registerImporter(pajekImporter)
350
374
  .registerExporter(pajekExporter)
351
375
  .registerImporter(neo4jImporter)
352
- .registerExporter(neo4jExporter);
376
+ .registerExporter(neo4jExporter)
377
+ .registerImporter(oboImporter);
353
378
  }
354
379
 
355
380
  /** The default registry: every built-in format. */
@@ -376,6 +401,16 @@ export function importAllGraphs(input: ImportInput, options?: ImportGraphOptions
376
401
  return registry.importAllGraphs(input, options);
377
402
  }
378
403
 
404
+ /**
405
+ * The graphs of an input that can hold several, through the default registry.
406
+ * @param input - the text, bytes, stream or chunks to read
407
+ * @param options - the format, hints, common and format-specific import options
408
+ * @returns one listing per graph, in document order; null when the format does not list its graphs
409
+ */
410
+ export function listGraphs(input: ImportInput, options?: ImportGraphOptions): Promise<readonly GraphListing[] | null> {
411
+ return registry.listGraphs(input, options);
412
+ }
413
+
379
414
  /**
380
415
  * Write a snapshot in a format through the default registry, as UTF-8 chunks.
381
416
  * @param snapshot - the snapshot
package/src/sniff.ts CHANGED
@@ -25,7 +25,7 @@ import { type JsonImportDialect, sniffJsonDialect } from "./formats/json/dialect
25
25
  import { type GraphImporter } from "./types.js";
26
26
 
27
27
  /** The format names of the eight built-in importers and exporters. */
28
- export type GraphFormatName = "gexf" | "graphml" | "gml" | "dot" | "pajek" | "csv" | "json" | "neo4j";
28
+ export type GraphFormatName = "gexf" | "graphml" | "gml" | "dot" | "pajek" | "csv" | "json" | "neo4j" | "obo";
29
29
 
30
30
  /**
31
31
  * The built-in format names in the default registry's order, which is also the tie-break order of
@@ -41,6 +41,7 @@ export const GRAPH_FORMATS: readonly GraphFormatName[] = Object.freeze([
41
41
  "dot",
42
42
  "pajek",
43
43
  "neo4j",
44
+ "obo",
44
45
  ]);
45
46
 
46
47
  /** How many bytes of the input the sniffers look at; the registry reads no more than this before deciding. */
@@ -195,7 +196,15 @@ export function sniffJsonDialectHead(head: Uint8Array | string): JsonImportDiale
195
196
  }
196
197
 
197
198
  /** Where a key was seen while scanning a truncated head. */
198
- type KeyPath = "" | "graph" | "options" | "nodes[0]" | "edges[0]" | "links[0]";
199
+ type KeyPath =
200
+ | ""
201
+ | "graph"
202
+ | "options"
203
+ | "nodes[0]"
204
+ | "edges[0]"
205
+ | "links[0]"
206
+ | "graphs[0].nodes[]"
207
+ | "graphs[0].edges[]";
199
208
 
200
209
  /**
201
210
  * A partial document rebuilt from the keys a truncated head reveals: every key gets a placeholder
@@ -223,9 +232,20 @@ function skeletonOf(text: string): unknown {
223
232
  case "options":
224
233
  root[key] = objectOf(keys.get(key));
225
234
  break;
226
- case "graphs":
227
- root[key] = [];
235
+ case "graphs": {
236
+ // the keys of any node or edge of the first graph: OBO Graphs or JGF (design 1.6)
237
+ const nodes = keys.get("graphs[0].nodes[]");
238
+ const edges = keys.get("graphs[0].edges[]");
239
+ const graph: Record<string, unknown> = {};
240
+ if (nodes !== undefined) {
241
+ graph.nodes = [objectOf(nodes)];
242
+ }
243
+ if (edges !== undefined) {
244
+ graph.edges = [objectOf(edges)];
245
+ }
246
+ root[key] = nodes === undefined && edges === undefined ? [] : [graph];
228
247
  break;
248
+ }
229
249
  case "nodes":
230
250
  case "edges":
231
251
  case "links": {
@@ -262,7 +282,8 @@ function objectOf(names: ReadonlySet<string> | undefined): Record<string, unknow
262
282
  * scanner tracks a container stack (the key each object sits under, the index of each array
263
283
  * element) and records a key when it is at one of the watched paths; a head cut inside a string
264
284
  * or a number simply ends the scan. For a top-level array the first element's keys are recorded
265
- * under `nodes[0]`.
285
+ * under `nodes[0]`; the keys of every node and edge of `graphs[0]` under `graphs[0].nodes[]` and
286
+ * `graphs[0].edges[]`.
266
287
  * @param text - the head, starting with `{` or `[`
267
288
  * @returns key sets by path
268
289
  */
@@ -285,6 +306,15 @@ function scanKeys(text: string): Map<KeyPath, Set<string>> {
285
306
  if (depth === 2 && kinds[0] === "object" && kinds[1] === "object") {
286
307
  return labels[1] === "graph" || labels[1] === "options" ? labels[1] : null;
287
308
  }
309
+ if (
310
+ depth === 5 &&
311
+ kinds.join() === "object,array,object,array,object" &&
312
+ labels[1] === "graphs" &&
313
+ labels[2] === "0" &&
314
+ (labels[3] === "nodes" || labels[3] === "edges")
315
+ ) {
316
+ return labels[3] === "nodes" ? "graphs[0].nodes[]" : "graphs[0].edges[]";
317
+ }
288
318
  if (depth === 3 && kinds[0] === "object" && kinds[1] === "array" && kinds[2] === "object") {
289
319
  const section = labels[1];
290
320
  if ((section === "nodes" || section === "edges" || section === "links") && labels[2] === "0") {