@graphty/graph-io 0.2.4 → 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (180) hide show
  1. package/README.md +29 -7
  2. package/dist/chunks/{escape-DyI8JofU.js → escape-CKied3Ri.js} +16 -10
  3. package/dist/chunks/escape-CKied3Ri.js.map +1 -0
  4. package/dist/chunks/{importer-DbnGYr3_.js → importer-C6QcIRIb.js} +129 -39
  5. package/dist/chunks/importer-C6QcIRIb.js.map +1 -0
  6. package/dist/chunks/{importer-CpCpfbxr.js → importer-D5pAsweZ.js} +136 -31
  7. package/dist/chunks/importer-D5pAsweZ.js.map +1 -0
  8. package/dist/chunks/{importer-GozH8DkN.js → importer-DbMrAR_w.js} +10 -4
  9. package/dist/chunks/importer-DbMrAR_w.js.map +1 -0
  10. package/dist/chunks/{importer-CQnJuWJw.js → importer-DkzjTvHc.js} +82 -40
  11. package/dist/chunks/importer-DkzjTvHc.js.map +1 -0
  12. package/dist/chunks/{records-CGpxszm1.js → records-BKSMowhR.js} +3 -3
  13. package/dist/chunks/{records-CGpxszm1.js.map → records-BKSMowhR.js.map} +1 -1
  14. package/dist/chunks/{text-CajMdVFy.js → text-Dr0Ifpag.js} +2 -2
  15. package/dist/chunks/{text-CajMdVFy.js.map → text-Dr0Ifpag.js.map} +1 -1
  16. package/dist/chunks/{writer-DxSKC7TL.js → writer-BdMak4_J.js} +341 -112
  17. package/dist/chunks/writer-BdMak4_J.js.map +1 -0
  18. package/dist/csv.js +24 -6
  19. package/dist/csv.js.map +1 -1
  20. package/dist/dot.js +1 -1
  21. package/dist/gexf.js +42 -10
  22. package/dist/gexf.js.map +1 -1
  23. package/dist/gml.js +69 -25
  24. package/dist/gml.js.map +1 -1
  25. package/dist/graph-io.js +186 -129
  26. package/dist/graph-io.js.map +1 -1
  27. package/dist/graphml.js +1 -1
  28. package/dist/json.js +1 -1
  29. package/dist/neo4j.js +10 -4
  30. package/dist/neo4j.js.map +1 -1
  31. package/dist/pajek.js +1 -1
  32. package/dist/src/common/codes.d.ts +6 -0
  33. package/dist/src/common/codes.d.ts.map +1 -1
  34. package/dist/src/common/codes.js +6 -0
  35. package/dist/src/common/codes.js.map +1 -1
  36. package/dist/src/common/escape.d.ts +5 -4
  37. package/dist/src/common/escape.d.ts.map +1 -1
  38. package/dist/src/common/escape.js +6 -5
  39. package/dist/src/common/escape.js.map +1 -1
  40. package/dist/src/common/format.d.ts +3 -3
  41. package/dist/src/common/format.js +5 -5
  42. package/dist/src/common/format.js.map +1 -1
  43. package/dist/src/common/input.d.ts +24 -6
  44. package/dist/src/common/input.d.ts.map +1 -1
  45. package/dist/src/common/input.js +259 -23
  46. package/dist/src/common/input.js.map +1 -1
  47. package/dist/src/common/options.d.ts +2 -0
  48. package/dist/src/common/options.d.ts.map +1 -1
  49. package/dist/src/common/options.js +17 -0
  50. package/dist/src/common/options.js.map +1 -1
  51. package/dist/src/common/text.js +4 -4
  52. package/dist/src/common/text.js.map +1 -1
  53. package/dist/src/common/weights.d.ts.map +1 -1
  54. package/dist/src/common/weights.js +8 -0
  55. package/dist/src/common/weights.js.map +1 -1
  56. package/dist/src/common/xml.d.ts +7 -0
  57. package/dist/src/common/xml.d.ts.map +1 -1
  58. package/dist/src/common/xml.js +12 -0
  59. package/dist/src/common/xml.js.map +1 -1
  60. package/dist/src/formats/csv/exporter.d.ts +5 -1
  61. package/dist/src/formats/csv/exporter.d.ts.map +1 -1
  62. package/dist/src/formats/csv/exporter.js +8 -1
  63. package/dist/src/formats/csv/exporter.js.map +1 -1
  64. package/dist/src/formats/csv/importer.d.ts.map +1 -1
  65. package/dist/src/formats/csv/importer.js +1 -0
  66. package/dist/src/formats/csv/importer.js.map +1 -1
  67. package/dist/src/formats/csv/index.d.ts +6 -0
  68. package/dist/src/formats/csv/index.d.ts.map +1 -1
  69. package/dist/src/formats/csv/index.js +7 -1
  70. package/dist/src/formats/csv/index.js.map +1 -1
  71. package/dist/src/formats/csv/records.js +1 -1
  72. package/dist/src/formats/csv/records.js.map +1 -1
  73. package/dist/src/formats/dot/exporter.d.ts +1 -1
  74. package/dist/src/formats/dot/exporter.d.ts.map +1 -1
  75. package/dist/src/formats/dot/exporter.js +3 -3
  76. package/dist/src/formats/dot/exporter.js.map +1 -1
  77. package/dist/src/formats/dot/importer.d.ts +6 -0
  78. package/dist/src/formats/dot/importer.d.ts.map +1 -1
  79. package/dist/src/formats/dot/importer.js +151 -24
  80. package/dist/src/formats/dot/importer.js.map +1 -1
  81. package/dist/src/formats/gexf/exporter.d.ts.map +1 -1
  82. package/dist/src/formats/gexf/exporter.js +32 -7
  83. package/dist/src/formats/gexf/exporter.js.map +1 -1
  84. package/dist/src/formats/gexf/importer.d.ts.map +1 -1
  85. package/dist/src/formats/gexf/importer.js +2 -2
  86. package/dist/src/formats/gexf/importer.js.map +1 -1
  87. package/dist/src/formats/gexf/index.d.ts +6 -0
  88. package/dist/src/formats/gexf/index.d.ts.map +1 -1
  89. package/dist/src/formats/gexf/index.js +7 -1
  90. package/dist/src/formats/gexf/index.js.map +1 -1
  91. package/dist/src/formats/gml/importer.d.ts +2 -2
  92. package/dist/src/formats/gml/importer.d.ts.map +1 -1
  93. package/dist/src/formats/gml/importer.js +66 -25
  94. package/dist/src/formats/gml/importer.js.map +1 -1
  95. package/dist/src/formats/gml/index.d.ts +7 -1
  96. package/dist/src/formats/gml/index.d.ts.map +1 -1
  97. package/dist/src/formats/gml/index.js +8 -2
  98. package/dist/src/formats/gml/index.js.map +1 -1
  99. package/dist/src/formats/graphml/constants.d.ts +6 -0
  100. package/dist/src/formats/graphml/constants.d.ts.map +1 -1
  101. package/dist/src/formats/graphml/constants.js +7 -1
  102. package/dist/src/formats/graphml/constants.js.map +1 -1
  103. package/dist/src/formats/graphml/importer.d.ts.map +1 -1
  104. package/dist/src/formats/graphml/importer.js +2 -2
  105. package/dist/src/formats/graphml/importer.js.map +1 -1
  106. package/dist/src/formats/json/exporter.d.ts.map +1 -1
  107. package/dist/src/formats/json/exporter.js +14 -5
  108. package/dist/src/formats/json/exporter.js.map +1 -1
  109. package/dist/src/formats/json/importer.d.ts +8 -0
  110. package/dist/src/formats/json/importer.d.ts.map +1 -1
  111. package/dist/src/formats/json/importer.js +84 -37
  112. package/dist/src/formats/json/importer.js.map +1 -1
  113. package/dist/src/formats/neo4j/importer.js +1 -1
  114. package/dist/src/formats/neo4j/importer.js.map +1 -1
  115. package/dist/src/formats/neo4j/index.d.ts +6 -0
  116. package/dist/src/formats/neo4j/index.d.ts.map +1 -1
  117. package/dist/src/formats/neo4j/index.js +7 -1
  118. package/dist/src/formats/neo4j/index.js.map +1 -1
  119. package/dist/src/formats/pajek/exporter.d.ts +2 -0
  120. package/dist/src/formats/pajek/exporter.d.ts.map +1 -1
  121. package/dist/src/formats/pajek/exporter.js +13 -2
  122. package/dist/src/formats/pajek/exporter.js.map +1 -1
  123. package/dist/src/formats/pajek/importer.d.ts +8 -2
  124. package/dist/src/formats/pajek/importer.d.ts.map +1 -1
  125. package/dist/src/formats/pajek/importer.js +122 -26
  126. package/dist/src/formats/pajek/importer.js.map +1 -1
  127. package/dist/src/index.d.ts +1 -1
  128. package/dist/src/index.d.ts.map +1 -1
  129. package/dist/src/index.js +1 -1
  130. package/dist/src/index.js.map +1 -1
  131. package/dist/src/registry.d.ts +26 -0
  132. package/dist/src/registry.d.ts.map +1 -1
  133. package/dist/src/registry.js +101 -35
  134. package/dist/src/registry.js.map +1 -1
  135. package/dist/src/sniff.d.ts +1 -1
  136. package/dist/src/sniff.d.ts.map +1 -1
  137. package/dist/src/sniff.js +6 -1
  138. package/dist/src/sniff.js.map +1 -1
  139. package/dist/src/types.d.ts +22 -3
  140. package/dist/src/types.d.ts.map +1 -1
  141. package/dist/src/types.js.map +1 -1
  142. package/package.json +4 -3
  143. package/src/common/codes.ts +9 -0
  144. package/src/common/escape.ts +6 -5
  145. package/src/common/format.ts +5 -5
  146. package/src/common/input.ts +293 -22
  147. package/src/common/options.ts +24 -0
  148. package/src/common/text.ts +4 -4
  149. package/src/common/weights.ts +9 -0
  150. package/src/common/xml.ts +14 -0
  151. package/src/formats/csv/exporter.ts +12 -1
  152. package/src/formats/csv/importer.ts +1 -0
  153. package/src/formats/csv/index.ts +9 -0
  154. package/src/formats/csv/records.ts +1 -1
  155. package/src/formats/dot/exporter.ts +3 -3
  156. package/src/formats/dot/importer.ts +172 -32
  157. package/src/formats/gexf/exporter.ts +38 -7
  158. package/src/formats/gexf/importer.ts +12 -2
  159. package/src/formats/gexf/index.ts +9 -0
  160. package/src/formats/gml/importer.ts +75 -22
  161. package/src/formats/gml/index.ts +10 -1
  162. package/src/formats/graphml/constants.ts +9 -0
  163. package/src/formats/graphml/importer.ts +9 -2
  164. package/src/formats/json/exporter.ts +14 -5
  165. package/src/formats/json/importer.ts +104 -36
  166. package/src/formats/neo4j/importer.ts +1 -1
  167. package/src/formats/neo4j/index.ts +9 -0
  168. package/src/formats/pajek/exporter.ts +19 -2
  169. package/src/formats/pajek/importer.ts +145 -28
  170. package/src/index.ts +1 -0
  171. package/src/registry.ts +131 -40
  172. package/src/sniff.ts +6 -1
  173. package/src/types.ts +26 -3
  174. package/dist/chunks/escape-DyI8JofU.js.map +0 -1
  175. package/dist/chunks/importer-CQnJuWJw.js.map +0 -1
  176. package/dist/chunks/importer-CpCpfbxr.js.map +0 -1
  177. package/dist/chunks/importer-DbnGYr3_.js.map +0 -1
  178. package/dist/chunks/importer-GozH8DkN.js.map +0 -1
  179. package/dist/chunks/writer-DxSKC7TL.js.map +0 -1
  180. package/dist/tsconfig.build.tsbuildinfo +0 -1
package/src/common/xml.ts CHANGED
@@ -226,6 +226,20 @@ function countIllegalRows(column: Column): number {
226
226
  return bad;
227
227
  }
228
228
 
229
+ /** The encoding pseudo-attribute of an XML declaration at the very start of a document. */
230
+ const XML_DECLARED_ENCODING = /^<\?xml\s[^>]*?\bencoding\s*=\s*["']([A-Za-z][A-Za-z0-9._-]*)["']/;
231
+
232
+ /**
233
+ * The encoding an XML document declares in its prolog (`<?xml version="1.0" encoding="..."?>`),
234
+ * for the shared byte decoder (common/input.ts).
235
+ * @param head - the start of the document, decoded as windows-1252
236
+ * @returns the declared label, or null when there is no declaration or it names no encoding
237
+ */
238
+ export function xmlDeclaredEncoding(head: string): string | null {
239
+ const match = XML_DECLARED_ENCODING.exec(head);
240
+ return match === null ? null : match[1];
241
+ }
242
+
229
243
  /**
230
244
  * Whether a text is an XML Name.
231
245
  * @param text - the text
@@ -55,7 +55,11 @@ export const CSV_LOSS = Object.freeze({
55
55
  ID_TEXT_COLLISION: LOSS.ID_TEXT_COLLISION,
56
56
  /** Ids whose text reads back as the other type under the canonical rule. */
57
57
  ID_TEXT_TYPE: LOSS.ID_TEXT_TYPE,
58
- /** The generic dialect has no direction column; an undirected or mixed graph reads back as directed. */
58
+ /**
59
+ * The generic dialect has no direction column; an undirected or mixed graph reads back as
60
+ * directed. The Gephi dialect loses the direction of an undirected graph without edges (no
61
+ * row carries a Type cell).
62
+ */
59
63
  DIRECTION_DROPPED: "W_CSV_DIRECTION_DROPPED",
60
64
  /** Mutual pairs are written as two directed rows. */
61
65
  MUTUAL_EXPANDED: LOSS.MUTUAL_EXPANDED,
@@ -390,6 +394,13 @@ function planExport(
390
394
  mixed,
391
395
  );
392
396
  }
397
+ } else if (csv.table === "edges" && !snapshot.directed && edgeRows.length === 0) {
398
+ note(
399
+ CSV_LOSS.DIRECTION_DROPPED,
400
+ "the direction lives in the Type cell of each edge row; an undirected graph without edges reads back as directed",
401
+ null,
402
+ 0,
403
+ );
393
404
  }
394
405
  if (folding.mutualCount > 0) {
395
406
  note(
@@ -453,6 +453,7 @@ class TableReader {
453
453
  comments: COMMENT_CHARS,
454
454
  signal: state.common.signal,
455
455
  onProgress: progress ? state.common.onProgress : null,
456
+ encoding: state.common.encoding,
456
457
  };
457
458
  this.reader = new CsvRecordReader(input, state.report, readerOptions);
458
459
  }
@@ -6,10 +6,13 @@ import {
6
6
  COLUMN_RENAMED_CODE,
7
7
  DIRECTION_FORCED_CODE,
8
8
  DIRECTION_REFUSED_CODE,
9
+ ENCODING_FALLBACK_CODE,
10
+ INVALID_ENCODING_CODE,
9
11
  INVALID_UTF8_CODE,
10
12
  MIXED_DIRECTION_CODE,
11
13
  OPTION_IGNORED_CODE,
12
14
  SINK_OPTION_CODE,
15
+ UNKNOWN_ENCODING_CODE,
13
16
  } from "../../common/codes.js";
14
17
  import { WIDENING_UNSUPPORTED_CODE } from "../../common/text.js";
15
18
  import {
@@ -43,6 +46,12 @@ export const CSV_ISSUE = Object.freeze({
43
46
  EMPTY_INPUT: EMPTY_INPUT_CODE,
44
47
  /** The input holds invalid UTF-8 (fatal). */
45
48
  INVALID_UTF8: INVALID_UTF8_CODE,
49
+ /** Invalid bytes in the encoding a BOM, a declaration or the encoding option chose (fatal). */
50
+ INVALID_ENCODING: INVALID_ENCODING_CODE,
51
+ /** Bytes that are not UTF-8 and declare no encoding were read as windows-1252. */
52
+ ENCODING_FALLBACK: ENCODING_FALLBACK_CODE,
53
+ /** A declared encoding the platform cannot decode was ignored. */
54
+ UNKNOWN_ENCODING: UNKNOWN_ENCODING_CODE,
46
55
  /** The header names neither endpoint columns nor an id column (fatal). */
47
56
  NO_ENDPOINT_COLUMNS: NO_ENDPOINT_COLUMNS_CODE,
48
57
  /** A node table without an id column (fatal). */
@@ -760,7 +760,7 @@ export class CsvRecordReader implements AsyncIterable<string[]> {
760
760
  input,
761
761
  report,
762
762
  { delimiter: options.delimiter ?? null, quote: '"', comments: options.comments },
763
- { signal: options.signal, onProgress: options.onProgress },
763
+ { signal: options.signal, onProgress: options.onProgress, encoding: options.encoding },
764
764
  );
765
765
  }
766
766
 
@@ -50,7 +50,7 @@ export interface DotExportOptions {
50
50
 
51
51
  /** The LossNote codes of the DOT exporter; the shared ones are LOSS's. */
52
52
  export const DOT_LOSS = Object.freeze({
53
- /** An id, name or text with a backslash before a quote or at its end cannot be written as a DOT quoted string; export() throws. */
53
+ /** An id, name or text with a backslash before a quote or a line break, or at its end, cannot be written as a DOT quoted string; export() throws. */
54
54
  TRAILING_BACKSLASH: "E_DOT_TRAILING_BACKSLASH",
55
55
  /** A non-finite f32 / f64 cell has no numeric DOT spelling and reads back as text. */
56
56
  NON_FINITE: "W_DOT_NON_FINITE",
@@ -448,7 +448,7 @@ class ExportPlan {
448
448
  if (unwritable > 0) {
449
449
  note(
450
450
  DOT_LOSS.TRAILING_BACKSLASH,
451
- `${unwritable} id(s), name(s) or text value(s) hold a backslash before a quote or at the end, which a DOT quoted string cannot carry; export() will throw`,
451
+ `${unwritable} id(s), name(s) or text value(s) hold a backslash before a quote or a line break, or at the end, which a DOT quoted string cannot carry; export() will throw`,
452
452
  null,
453
453
  unwritable,
454
454
  );
@@ -576,7 +576,7 @@ class ExportPlan {
576
576
  if (first !== null) {
577
577
  throw new GraphFormatError(
578
578
  first.kind === "id" ? "E_INVALID_ID" : "E_COLUMN_TYPE",
579
- `${first.kind} ${JSON.stringify(first.text)} ends in a backslash, which a DOT quoted string cannot carry`,
579
+ `${first.kind} ${JSON.stringify(first.text)} holds a backslash before a quote or a line break, or at its end, which a DOT quoted string cannot carry`,
580
580
  { reason: "trailing backslash", kind: first.kind, value: first.text },
581
581
  );
582
582
  }
@@ -47,7 +47,9 @@ import {
47
47
  DIRECTION_FORCED_CODE,
48
48
  DIRECTION_REFUSED_CODE,
49
49
  EMPTY_INPUT_CODE,
50
+ ENCODING_FALLBACK_CODE,
50
51
  ID_MERGED_CODE,
52
+ INVALID_ENCODING_CODE,
51
53
  INVALID_UTF8_CODE,
52
54
  MIXED_DIRECTION_CODE,
53
55
  MULTIPLE_GRAPHS_CODE,
@@ -55,6 +57,7 @@ import {
55
57
  ROLE_TAKEN_CODE,
56
58
  SINK_OPTION_CODE,
57
59
  SYNTAX_CODE,
60
+ UNKNOWN_ENCODING_CODE,
58
61
  } from "../../common/codes.js";
59
62
  import { DirectionResolver, type EdgeKind } from "../../common/direction.js";
60
63
  import { IdCoercer } from "../../common/ids.js";
@@ -106,6 +109,12 @@ export const DOT_ISSUE = Object.freeze({
106
109
  EMPTY_INPUT: EMPTY_INPUT_CODE,
107
110
  /** The input holds invalid UTF-8 (fatal). */
108
111
  INVALID_UTF8: INVALID_UTF8_CODE,
112
+ /** Invalid bytes in the encoding a BOM, a declaration or the encoding option chose (fatal). */
113
+ INVALID_ENCODING: INVALID_ENCODING_CODE,
114
+ /** Bytes that are not UTF-8 and declare no encoding were read as windows-1252. */
115
+ ENCODING_FALLBACK: ENCODING_FALLBACK_CODE,
116
+ /** A declared encoding the platform cannot decode was ignored. */
117
+ UNKNOWN_ENCODING: UNKNOWN_ENCODING_CODE,
109
118
  /** Subgraphs or braces nested deeper than the parser's limit; fatal. */
110
119
  NESTING: "E_DOT_NESTING",
111
120
  /** An edge operator contradicting the graph keyword (warning under "operator" / "header"). */
@@ -264,22 +273,166 @@ export const dotImporter: GraphImporter<DotImportOptions> = Object.freeze({
264
273
  const report = new ImportReportBuilder(DOT_FORMAT, resolved.errorLimit);
265
274
  reportUnusedOptions(options, report, USED_OPTIONS);
266
275
  reportSinkOptions(sink, options, report);
267
- const text = await readText(input, report, resolved);
268
- const parser = new DotParser(text, sink, report, resolved, mismatch);
269
- try {
270
- parser.parse();
271
- } catch (err) {
272
- if (err instanceof DotSyntaxError) {
273
- report.fail(SYNTAX_CODE, err.message, { line: err.line }, { line: err.line });
274
- }
275
- throw err;
276
+ const text = await readText(input, report, { ...resolved, declaredEncoding: dotCharset });
277
+ const reports = { current: report };
278
+ const lexer = dotLexer(text, reports);
279
+ const first = guard(report, () => lexer.next());
280
+ const trailing = parseGraph(new DotParser(lexer, sink, report, resolved, mismatch), report, first);
281
+ if (trailing.kind !== "eof") {
282
+ const skipped = guard(report, () => countGraphs(lexer, trailing));
283
+ report.warning(
284
+ "unsupported",
285
+ MULTIPLE_GRAPHS_CODE,
286
+ `the input holds ${skipped} more graph(s) after the first; import() reads the first, importAll() reads every one`,
287
+ { line: trailing.line },
288
+ );
276
289
  }
277
290
  // an abort raised during the last few statements (after the last periodic check) still rejects
278
291
  throwIfAborted(resolved.signal);
279
292
  return report.finish();
280
293
  },
294
+
295
+ /**
296
+ * Read every graph of a DOT file (Graphviz renders each one), each into its own sink.
297
+ * @param input - the text, bytes or stream
298
+ * @param sinkFor - the sink of the graph with this index, called before its first push
299
+ * @param options - format-specific and common options
300
+ * @returns one report per graph; ImportError (E_IMPORT) on a syntax error anywhere
301
+ */
302
+ async importAll(
303
+ input: ImportInput,
304
+ sinkFor: (index: number) => GraphSink,
305
+ options?: DotImportOptions & CommonImportOptions,
306
+ ): Promise<ImportReport[]> {
307
+ const resolved = resolveImportOptions(options, {
308
+ ids: "canonical",
309
+ defaultDirected: true,
310
+ weightFrom: "weight",
311
+ });
312
+ const mismatch = mismatchOption(options?.mismatchedEdgeOperator);
313
+ const first = new ImportReportBuilder(DOT_FORMAT, resolved.errorLimit);
314
+ reportUnusedOptions(options, first, USED_OPTIONS);
315
+ const text = await readText(input, first, { ...resolved, declaredEncoding: dotCharset });
316
+ const reports = { current: first };
317
+ const lexer = dotLexer(text, reports);
318
+ const done: ImportReport[] = [];
319
+ let token = guard(first, () => lexer.next());
320
+ do {
321
+ const report = done.length === 0 ? first : new ImportReportBuilder(DOT_FORMAT, resolved.errorLimit);
322
+ reports.current = report;
323
+ const sink = sinkFor(done.length);
324
+ reportSinkOptions(sink, options, report);
325
+ token = parseGraph(new DotParser(lexer, sink, report, resolved, mismatch), report, token);
326
+ throwIfAborted(resolved.signal);
327
+ done.push(report.finish());
328
+ } while (token.kind !== "eof");
329
+ return done;
330
+ },
281
331
  });
282
332
 
333
+ /**
334
+ * The tokenizer of one input; its badly-delimited-numeral warnings go to the report of the graph
335
+ * being read.
336
+ * @param text - the DOT text
337
+ * @param reports - holds the current graph's report
338
+ * @param reports.current - the report warnings are recorded in
339
+ * @returns the tokenizer
340
+ */
341
+ function dotLexer(text: string, reports: { current: ImportReportBuilder }): DotTokenizer {
342
+ return new DotTokenizer(text, (numeral, line) => {
343
+ reports.current.warning(
344
+ "validation-error",
345
+ DOT_ISSUE.NUMERAL_AMBIGUITY,
346
+ `badly delimited number ${JSON.stringify(numeral)} splits into two tokens (Graphviz warns the same)`,
347
+ { line, element: numeral },
348
+ );
349
+ });
350
+ }
351
+
352
+ /**
353
+ * Run a step that may throw DotSyntaxError, turning it into the fatal E_SYNTAX issue.
354
+ * @param report - the report of the graph being read
355
+ * @param step - the step
356
+ * @returns the step's result
357
+ */
358
+ function guard<T>(report: ImportReportBuilder, step: () => T): T {
359
+ try {
360
+ return step();
361
+ } catch (err) {
362
+ if (err instanceof DotSyntaxError) {
363
+ report.fail(SYNTAX_CODE, err.message, { line: err.line }, { line: err.line });
364
+ }
365
+ throw err;
366
+ }
367
+ }
368
+
369
+ /**
370
+ * Parse one graph.
371
+ * @param parser - the parser of this graph
372
+ * @param report - its report
373
+ * @param first - the graph's first token (already read)
374
+ * @returns the token after the graph's closing brace
375
+ */
376
+ function parseGraph(parser: DotParser, report: ImportReportBuilder, first: DotToken): DotToken {
377
+ return guard(report, () => parser.parse(first));
378
+ }
379
+
380
+ /**
381
+ * Count the graphs that follow the first one without reading their statements: each must be
382
+ * `[strict] (graph | digraph) [ID] { ... }` with balanced braces, as Graphviz requires.
383
+ * @param lexer - the tokenizer, positioned after `first`
384
+ * @param first - the first token after the first graph
385
+ * @returns how many graphs follow; DotSyntaxError on anything else
386
+ */
387
+ function countGraphs(lexer: DotTokenizer, first: DotToken): number {
388
+ let count = 0;
389
+ let token = first;
390
+ while (token.kind !== "eof") {
391
+ if (isKeyword(token, "strict")) {
392
+ token = lexer.next();
393
+ }
394
+ if (!isKeyword(token, "graph") && !isKeyword(token, "digraph")) {
395
+ throw new DotSyntaxError(`expected "graph" or "digraph", found ${describeToken(token)}`, token.line);
396
+ }
397
+ token = lexer.next();
398
+ if (token.kind === "id") {
399
+ token = lexer.next();
400
+ }
401
+ if (!isPunct(token, "{")) {
402
+ throw new DotSyntaxError(`expected "{" after the graph header, found ${describeToken(token)}`, token.line);
403
+ }
404
+ for (let depth = 1; depth > 0; ) {
405
+ token = lexer.next();
406
+ if (token.kind === "eof") {
407
+ throw new DotSyntaxError('missing "}" at the end of a graph', token.line);
408
+ }
409
+ if (isPunct(token, "{")) {
410
+ depth++;
411
+ } else if (isPunct(token, "}")) {
412
+ depth--;
413
+ }
414
+ }
415
+ count++;
416
+ token = lexer.next();
417
+ }
418
+ return count;
419
+ }
420
+
421
+ /** A `charset` attribute assignment (Graphviz's declaration of the input encoding). */
422
+ const DOT_CHARSET = /\bcharset\s*=\s*"?([A-Za-z][A-Za-z0-9._-]*)/i;
423
+
424
+ /**
425
+ * The encoding a DOT file declares with the graph attribute `charset` (Graphviz reads UTF-8,
426
+ * Latin1 and Big-5), for the shared byte decoder; Graphviz's spellings `latin-1` and `big-5` are
427
+ * mapped to the WHATWG labels.
428
+ * @param head - the start of the file, decoded as windows-1252
429
+ * @returns the declared label, or null
430
+ */
431
+ function dotCharset(head: string): string | null {
432
+ const match = DOT_CHARSET.exec(head);
433
+ return match === null ? null : match[1].replace(/^(latin|big)-/i, "$1");
434
+ }
435
+
283
436
  /**
284
437
  * Resolve the mismatchedEdgeOperator option.
285
438
  * @param value - the caller's value
@@ -447,28 +600,21 @@ class DotParser {
447
600
  private readonly edgeWriters = new Map<string, TextCellWriter>();
448
601
 
449
602
  /**
450
- * Create a parser over one document.
451
- * @param text - the DOT text
603
+ * Create a parser for one graph of a document.
604
+ * @param lexer - the document's tokenizer, positioned at the graph
452
605
  * @param sink - the sink
453
606
  * @param report - the report
454
607
  * @param options - the resolved common options
455
608
  * @param mismatch - the resolved mismatchedEdgeOperator option
456
609
  */
457
610
  constructor(
458
- text: string,
611
+ lexer: DotTokenizer,
459
612
  sink: GraphSink,
460
613
  report: ImportReportBuilder,
461
614
  options: ResolvedImportOptions,
462
615
  mismatch: "operator" | "header" | "error",
463
616
  ) {
464
- this.lexer = new DotTokenizer(text, (numeral, line) => {
465
- report.warning(
466
- "validation-error",
467
- DOT_ISSUE.NUMERAL_AMBIGUITY,
468
- `badly delimited number ${JSON.stringify(numeral)} splits into two tokens (Graphviz warns the same)`,
469
- { line, element: numeral },
470
- );
471
- });
617
+ this.lexer = lexer;
472
618
  this.sink = sink;
473
619
  this.report = report;
474
620
  this.options = options;
@@ -478,11 +624,13 @@ class DotParser {
478
624
  }
479
625
 
480
626
  /**
481
- * Parse the whole document: `[strict] (graph | digraph) [ID] { stmt_list }`.
627
+ * Parse one graph: `[strict] (graph | digraph) [ID] { stmt_list }`.
628
+ * @param first - the graph's first token (already read from the tokenizer)
629
+ * @returns the token after the closing brace (eof when the document ends there)
482
630
  */
483
- parse(): void {
631
+ parse(first: DotToken): DotToken {
484
632
  const { lexer } = this;
485
- let token = lexer.next();
633
+ let token = first;
486
634
  if (token.kind === "eof") {
487
635
  this.report.fail(EMPTY_INPUT_CODE, "the input holds no graph (empty or only comments)", {
488
636
  line: token.line,
@@ -517,15 +665,7 @@ class DotParser {
517
665
  });
518
666
  const root = this.newScope(null, headerLine, true, null);
519
667
  this.statementList(root);
520
- const trailing = lexer.next();
521
- if (trailing.kind !== "eof") {
522
- this.report.warning(
523
- "unsupported",
524
- MULTIPLE_GRAPHS_CODE,
525
- `content after the closing brace of the graph (${describeToken(trailing)}) was not read; one graph per input`,
526
- { line: trailing.line },
527
- );
528
- }
668
+ return lexer.next();
529
669
  }
530
670
 
531
671
  /**
@@ -290,6 +290,14 @@ function planExport(
290
290
  );
291
291
  }
292
292
  }
293
+ if (graphExtraText(snapshot, "timestamp") !== null) {
294
+ note(
295
+ GEXF_LOSS.TIMESTAMP_AS_INTERVAL,
296
+ "the graph timestamp is written as a closed interval (start = end); GEXF 1.2 has no timestamp representation",
297
+ null,
298
+ null,
299
+ );
300
+ }
293
301
  if (edgeRoles.kind !== null) {
294
302
  note(
295
303
  GEXF_LOSS.KIND_DROPPED,
@@ -1647,18 +1655,41 @@ function graphStart(snapshot: GraphSnapshot, plan: ExportPlan): string {
1647
1655
  if (plan.version === "1.3" && plan.timeRepresentation !== null) {
1648
1656
  attrs += ` timerepresentation="${plan.timeRepresentation}"`;
1649
1657
  }
1650
- const gexfExtra = meta.extra.gexf;
1651
- if (typeof gexfExtra === "object" && gexfExtra !== null) {
1652
- for (const key of ["start", "end", "timestamp"]) {
1653
- const value = (gexfExtra as Record<string, unknown>)[key];
1654
- if (typeof value === "string") {
1655
- attrs += ` ${key}="${escapeXmlAttribute(value)}"`;
1656
- }
1658
+ const timestamp = graphExtraText(snapshot, "timestamp");
1659
+ const header: Record<string, string | null> = {
1660
+ start: graphExtraText(snapshot, "start"),
1661
+ end: graphExtraText(snapshot, "end"),
1662
+ timestamp,
1663
+ };
1664
+ if (plan.version === "1.2" && timestamp !== null) {
1665
+ // GEXF 1.2 has no timestamp: the graph's becomes a closed interval, as an element's does
1666
+ header.start ??= timestamp;
1667
+ header.end ??= timestamp;
1668
+ header.timestamp = null;
1669
+ }
1670
+ for (const [key, value] of Object.entries(header)) {
1671
+ if (value !== null) {
1672
+ attrs += ` ${key}="${escapeXmlAttribute(value)}"`;
1657
1673
  }
1658
1674
  }
1659
1675
  return ` <graph${attrs}>\n`;
1660
1676
  }
1661
1677
 
1678
+ /**
1679
+ * A graph header value the GEXF importer recorded in `meta.extra.gexf` (start, end, timestamp).
1680
+ * @param snapshot - the snapshot
1681
+ * @param key - the attribute name
1682
+ * @returns the text, or null when absent
1683
+ */
1684
+ function graphExtraText(snapshot: GraphSnapshot, key: string): string | null {
1685
+ const gexfExtra = snapshot.meta.extra.gexf;
1686
+ if (typeof gexfExtra !== "object" || gexfExtra === null) {
1687
+ return null;
1688
+ }
1689
+ const value = (gexfExtra as Record<string, unknown>)[key];
1690
+ return typeof value === "string" ? value : null;
1691
+ }
1692
+
1662
1693
  /** The GEXF exporter (design section 8.5); `capabilities` describes the default 1.3 output. */
1663
1694
  export const gexfExporter: GraphExporter<GexfExportOptions> = Object.freeze({
1664
1695
  format: GEXF_FORMAT,
@@ -74,7 +74,14 @@ import {
74
74
  import { ImportReportBuilder, type IssueLocation } from "../../common/report.js";
75
75
  import { parseTimeText, type TemporalValue, type TimeFormat, timeTextCompanion } from "../../common/temporal.js";
76
76
  import { isWeightField, parseWeightText } from "../../common/weights.js";
77
- import { isWhitespace, localName, tokenizeXml, type XmlHandler, XmlSyntaxError } from "../../common/xml.js";
77
+ import {
78
+ isWhitespace,
79
+ localName,
80
+ tokenizeXml,
81
+ xmlDeclaredEncoding,
82
+ type XmlHandler,
83
+ XmlSyntaxError,
84
+ } from "../../common/xml.js";
78
85
  import { type CommonImportOptions, type GraphImporter, type ImportInput, type ImportReport } from "../../types.js";
79
86
  import {
80
87
  EDGE_DECLS,
@@ -2318,7 +2325,10 @@ export const gexfImporter: GraphImporter<GexfImportOptions> = Object.freeze({
2318
2325
  reportUnusedOptions(options, report, USED_OPTIONS);
2319
2326
  const reader = new GexfReader(sink, report, resolved, viz);
2320
2327
  try {
2321
- await tokenizeXml(textChunks(input, report, resolved), reader);
2328
+ await tokenizeXml(
2329
+ textChunks(input, report, { ...resolved, declaredEncoding: xmlDeclaredEncoding }),
2330
+ reader,
2331
+ );
2322
2332
  } catch (err) {
2323
2333
  if (err instanceof XmlSyntaxError) {
2324
2334
  report.fail(XML_SYNTAX_CODE, err.message, { line: err.line });
@@ -13,7 +13,9 @@ import {
13
13
  DIRECTION_REFUSED_CODE,
14
14
  DUPLICATE_EDGE_ID_CODE,
15
15
  DUPLICATE_NODE_CODE,
16
+ ENCODING_FALLBACK_CODE,
16
17
  ID_MERGED_CODE,
18
+ INVALID_ENCODING_CODE,
17
19
  INVALID_UTF8_CODE,
18
20
  MISSING_ENDPOINT_CODE,
19
21
  MISSING_ID_CODE,
@@ -26,6 +28,7 @@ import {
26
28
  STRAY_TEXT_CODE,
27
29
  UNKNOWN_ATTR_TYPE_CODE,
28
30
  UNKNOWN_ELEMENT_CODE,
31
+ UNKNOWN_ENCODING_CODE,
29
32
  UNKNOWN_PARENT_CODE,
30
33
  XML_SYNTAX_CODE,
31
34
  } from "../../common/codes.js";
@@ -63,6 +66,12 @@ export const GEXF_ISSUE = Object.freeze({
63
66
  XML_SYNTAX: XML_SYNTAX_CODE,
64
67
  /** The input holds invalid UTF-8 (fatal). */
65
68
  INVALID_UTF8: INVALID_UTF8_CODE,
69
+ /** Invalid bytes in the encoding a BOM, a declaration or the encoding option chose (fatal). */
70
+ INVALID_ENCODING: INVALID_ENCODING_CODE,
71
+ /** Bytes that are not UTF-8 and declare no encoding were read as windows-1252. */
72
+ ENCODING_FALLBACK: ENCODING_FALLBACK_CODE,
73
+ /** A declared encoding the platform cannot decode was ignored. */
74
+ UNKNOWN_ENCODING: UNKNOWN_ENCODING_CODE,
66
75
  /** The root element is not `<gexf>` (fatal). */
67
76
  NOT_GEXF: NOT_GEXF_CODE,
68
77
  /** The document has no `<graph>` (fatal). */
@@ -39,6 +39,7 @@ import {
39
39
  DUPLICATE_NODE_CODE as SHARED_DUPLICATE_NODE_CODE,
40
40
  MISSING_ENDPOINT_CODE as SHARED_MISSING_ENDPOINT_CODE,
41
41
  MISSING_ID_CODE as SHARED_MISSING_ID_CODE,
42
+ MULTIPLE_GRAPHS_CODE,
42
43
  NO_GRAPH_CODE as SHARED_NO_GRAPH_CODE,
43
44
  PRECISION_CODE as SHARED_PRECISION_CODE,
44
45
  ROLE_TAKEN_CODE as SHARED_ROLE_TAKEN_CODE,
@@ -87,8 +88,8 @@ export interface GmlImportOptions {
87
88
 
88
89
  /** Issue code: the text holds no `graph [ ... ]` block. */
89
90
  export const NO_GRAPH_CODE = SHARED_NO_GRAPH_CODE;
90
- /** Issue code: the text holds more than one `graph [ ... ]` block (fatal; one network per file). */
91
- export const SECOND_GRAPH_CODE = "E_GML_SECOND_GRAPH";
91
+ /** Issue code: the text holds more than one `graph [ ... ]` block; import() reads the first, importAll() every one. */
92
+ export const SECOND_GRAPH_CODE = MULTIPLE_GRAPHS_CODE;
92
93
  /** Issue code: a node block has no `id` key. */
93
94
  export const MISSING_ID_CODE = SHARED_MISSING_ID_CODE;
94
95
  /** Issue code: a node block has no `label` key under `nodeIdFrom: "label"`. */
@@ -248,6 +249,9 @@ class GmlImport {
248
249
 
249
250
  private tokens: GmlTokens | null = null;
250
251
 
252
+ /** How many `graph [ ... ]` blocks the text holds (known after the scan). */
253
+ graphCount = 0;
254
+
251
255
  private resolver: DirectionResolver | null = null;
252
256
 
253
257
  private readonly nodeSchema = new Map<string, KeySchema>();
@@ -315,12 +319,14 @@ class GmlImport {
315
319
  * @param report - the report
316
320
  * @param options - the resolved common options
317
321
  * @param format - the format-specific options
322
+ * @param which - which `graph [ ... ]` block to read (0: the first)
318
323
  */
319
324
  constructor(
320
325
  sink: GraphSink,
321
326
  report: ImportReportBuilder,
322
327
  options: ResolvedImportOptions,
323
328
  format: GmlImportOptions | undefined,
329
+ private readonly which = 0,
324
330
  ) {
325
331
  this.sink = sink;
326
332
  this.report = report;
@@ -331,19 +337,11 @@ class GmlImport {
331
337
  }
332
338
 
333
339
  /**
334
- * Run the import over a text.
335
- * @param text - the whole GML text
340
+ * Run the import over the tokens of a text.
341
+ * @param tokens - the whole text, tokenized (tokenize())
336
342
  */
337
- run(text: string): void {
338
- const { report } = this;
339
- try {
340
- this.tokens = tokenizeGml(text);
341
- } catch (err) {
342
- if (err instanceof GmlSyntaxError) {
343
- report.fail(err.code, err.message, { line: err.line });
344
- }
345
- throw err;
346
- }
343
+ run(tokens: GmlTokens): void {
344
+ this.tokens = tokens;
347
345
  this.scan();
348
346
  this.push();
349
347
  }
@@ -361,13 +359,10 @@ class GmlImport {
361
359
  const v = p + 1;
362
360
  const record = t.kind[v] === TOKEN_OPEN;
363
361
  if (key === "graph" && record) {
364
- if (this.graphOpen >= 0) {
365
- this.report.fail(SECOND_GRAPH_CODE, "the input contains more than one graph", {
366
- line: t.line[p],
367
- });
362
+ if (this.graphCount++ === this.which) {
363
+ this.graphOpen = v;
364
+ this.scanGraph(v);
368
365
  }
369
- this.graphOpen = v;
370
- this.scanGraph(v);
371
366
  } else if (key === "Creator" && !record && this.creatorToken < 0) {
372
367
  this.creatorToken = v;
373
368
  } else if (key === "Version" && !record && this.versionToken < 0) {
@@ -1581,11 +1576,69 @@ export const gmlImporter: GraphImporter<GmlImportOptions> = Object.freeze({
1581
1576
  const report = new ImportReportBuilder("gml", resolved.errorLimit);
1582
1577
  reportSinkOptions(sink, options, report);
1583
1578
  reportUnusedOptions(options, report, USED_OPTIONS);
1584
- const text = await readText(input, report, resolved);
1579
+ const tokens = tokenize(await readText(input, report, resolved), report);
1585
1580
  throwIfAborted(resolved.signal);
1586
- new GmlImport(sink, report, resolved, options).run(text);
1581
+ const gml = new GmlImport(sink, report, resolved, options);
1582
+ gml.run(tokens);
1583
+ if (gml.graphCount > 1) {
1584
+ report.warning(
1585
+ "unsupported",
1586
+ SECOND_GRAPH_CODE,
1587
+ `the input holds ${gml.graphCount - 1} more graph block(s) after the first; import() reads the first, importAll() reads every one`,
1588
+ );
1589
+ }
1587
1590
  // an abort raised during the last few elements (after the last periodic check) still rejects
1588
1591
  throwIfAborted(resolved.signal);
1589
1592
  return report.finish();
1590
1593
  },
1594
+
1595
+ /**
1596
+ * Read every `graph [ ... ]` block of a GML text, each into its own sink; the top-level keys
1597
+ * (Creator, Version, ...) apply to each.
1598
+ * @param input - the text, bytes or stream
1599
+ * @param sinkFor - the sink of the graph with this index, called before its first push
1600
+ * @param options - format-specific and common options
1601
+ * @returns one report per graph block
1602
+ */
1603
+ async importAll(
1604
+ input: ImportInput,
1605
+ sinkFor: (index: number) => GraphSink,
1606
+ options?: GmlImportOptions & CommonImportOptions,
1607
+ ): Promise<ImportReport[]> {
1608
+ const resolved = resolveImportOptions(options, FORMAT_DEFAULTS);
1609
+ const first = new ImportReportBuilder("gml", resolved.errorLimit);
1610
+ reportUnusedOptions(options, first, USED_OPTIONS);
1611
+ const tokens = tokenize(await readText(input, first, resolved), first);
1612
+ const reports: ImportReport[] = [];
1613
+ let count = 1;
1614
+ for (let i = 0; i < count; i++) {
1615
+ throwIfAborted(resolved.signal);
1616
+ const report = i === 0 ? first : new ImportReportBuilder("gml", resolved.errorLimit);
1617
+ const sink = sinkFor(i);
1618
+ reportSinkOptions(sink, options, report);
1619
+ const gml = new GmlImport(sink, report, resolved, options, i);
1620
+ gml.run(tokens);
1621
+ count = gml.graphCount;
1622
+ reports.push(report.finish());
1623
+ }
1624
+ throwIfAborted(resolved.signal);
1625
+ return reports;
1626
+ },
1591
1627
  });
1628
+
1629
+ /**
1630
+ * Tokenize a GML text; a syntax error is fatal.
1631
+ * @param text - the whole text
1632
+ * @param report - where the syntax error is recorded
1633
+ * @returns the tokens
1634
+ */
1635
+ function tokenize(text: string, report: ImportReportBuilder): GmlTokens {
1636
+ try {
1637
+ return tokenizeGml(text);
1638
+ } catch (err) {
1639
+ if (err instanceof GmlSyntaxError) {
1640
+ report.fail(err.code, err.message, { line: err.line });
1641
+ }
1642
+ throw err;
1643
+ }
1644
+ }