@graphty/graph-io 0.3.18 → 0.3.20

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (235) hide show
  1. package/README.md +80 -2
  2. package/dist/chunks/{escape-D-gZWO26.js → escape-B-tkk_Cl.js} +74 -16
  3. package/dist/chunks/escape-B-tkk_Cl.js.map +1 -0
  4. package/dist/chunks/{writer-GAdltGmC.js → export-Bh60Dv-n.js} +7 -252
  5. package/dist/chunks/export-Bh60Dv-n.js.map +1 -0
  6. package/dist/chunks/exporter-BGrsWamJ.js +736 -0
  7. package/dist/chunks/exporter-BGrsWamJ.js.map +1 -0
  8. package/dist/chunks/exporter-DIeGJXAZ.js +834 -0
  9. package/dist/chunks/exporter-DIeGJXAZ.js.map +1 -0
  10. package/dist/chunks/{importer-Br_QeAeE.js → importer-BWY2FFc5.js} +28 -6
  11. package/dist/chunks/importer-BWY2FFc5.js.map +1 -0
  12. package/dist/chunks/importer-BaQpCEcJ.js +2194 -0
  13. package/dist/chunks/importer-BaQpCEcJ.js.map +1 -0
  14. package/dist/chunks/importer-DD-xv9_Y.js +1471 -0
  15. package/dist/chunks/importer-DD-xv9_Y.js.map +1 -0
  16. package/dist/chunks/{importer-aNJfe0qu.js → importer-DEcKhsmQ.js} +2 -2
  17. package/dist/chunks/{importer-aNJfe0qu.js.map → importer-DEcKhsmQ.js.map} +1 -1
  18. package/dist/chunks/{importer-DHagxvDD.js → importer-DIrbbnAf.js} +6 -4
  19. package/dist/chunks/{importer-DHagxvDD.js.map → importer-DIrbbnAf.js.map} +1 -1
  20. package/dist/chunks/importer-D_e7e7LX.js +3055 -0
  21. package/dist/chunks/importer-D_e7e7LX.js.map +1 -0
  22. package/dist/chunks/{importer-Du5crN9l.js → importer-Wv4C4gaF.js} +278 -82
  23. package/dist/chunks/importer-Wv4C4gaF.js.map +1 -0
  24. package/dist/chunks/importer-vi3bSdtw.js +1713 -0
  25. package/dist/chunks/importer-vi3bSdtw.js.map +1 -0
  26. package/dist/chunks/{importer-d0uQxFp6.js → importer-zGLo8Dg_.js} +6 -4
  27. package/dist/chunks/{importer-d0uQxFp6.js.map → importer-zGLo8Dg_.js.map} +1 -1
  28. package/dist/chunks/json-elements-DDS8N2Dd.js +779 -0
  29. package/dist/chunks/json-elements-DDS8N2Dd.js.map +1 -0
  30. package/dist/chunks/{records-Bk9jgodz.js → records-dbRkxwaq.js} +2 -2
  31. package/dist/chunks/{records-Bk9jgodz.js.map → records-dbRkxwaq.js.map} +1 -1
  32. package/dist/chunks/{report-BOk0p5y8.js → report-B1z4WT9e.js} +143 -111
  33. package/dist/chunks/{report-BOk0p5y8.js.map → report-B1z4WT9e.js.map} +1 -1
  34. package/dist/chunks/weights-Dzba96G3.js +176 -0
  35. package/dist/chunks/weights-Dzba96G3.js.map +1 -0
  36. package/dist/chunks/writer-C7flM-Ih.js +77 -0
  37. package/dist/chunks/writer-C7flM-Ih.js.map +1 -0
  38. package/dist/csv.js +6 -4
  39. package/dist/csv.js.map +1 -1
  40. package/dist/cx.d.ts +1 -0
  41. package/dist/cx.js +6 -0
  42. package/dist/cx.js.map +1 -0
  43. package/dist/cx2.d.ts +1 -0
  44. package/dist/cx2.js +10 -0
  45. package/dist/cx2.js.map +1 -0
  46. package/dist/cys.d.ts +1 -0
  47. package/dist/cys.js +6 -0
  48. package/dist/cys.js.map +1 -0
  49. package/dist/dot.js +1 -1
  50. package/dist/gexf.js +16 -4
  51. package/dist/gexf.js.map +1 -1
  52. package/dist/gml.js +13 -14
  53. package/dist/gml.js.map +1 -1
  54. package/dist/graph-io.js +302 -399
  55. package/dist/graph-io.js.map +1 -1
  56. package/dist/graphml.js +1 -1
  57. package/dist/json.js +1 -1
  58. package/dist/neo4j.js +6 -4
  59. package/dist/neo4j.js.map +1 -1
  60. package/dist/obo.js +1 -1
  61. package/dist/pajek.js +1 -1
  62. package/dist/src/common/cell-budget.d.ts +84 -0
  63. package/dist/src/common/cell-budget.d.ts.map +1 -0
  64. package/dist/src/common/cell-budget.js +138 -0
  65. package/dist/src/common/cell-budget.js.map +1 -0
  66. package/dist/src/common/input.d.ts +10 -0
  67. package/dist/src/common/input.d.ts.map +1 -1
  68. package/dist/src/common/input.js +39 -0
  69. package/dist/src/common/input.js.map +1 -1
  70. package/dist/src/common/json-elements.d.ts +286 -0
  71. package/dist/src/common/json-elements.d.ts.map +1 -0
  72. package/dist/src/common/json-elements.js +926 -0
  73. package/dist/src/common/json-elements.js.map +1 -0
  74. package/dist/src/common/options.d.ts +6 -0
  75. package/dist/src/common/options.d.ts.map +1 -1
  76. package/dist/src/common/options.js +1 -1
  77. package/dist/src/common/options.js.map +1 -1
  78. package/dist/src/common/xml.d.ts +37 -3
  79. package/dist/src/common/xml.d.ts.map +1 -1
  80. package/dist/src/common/xml.js +86 -6
  81. package/dist/src/common/xml.js.map +1 -1
  82. package/dist/src/common/zip.d.ts +92 -0
  83. package/dist/src/common/zip.d.ts.map +1 -0
  84. package/dist/src/common/zip.js +399 -0
  85. package/dist/src/common/zip.js.map +1 -0
  86. package/dist/src/formats/cx/importer.d.ts +133 -0
  87. package/dist/src/formats/cx/importer.d.ts.map +1 -0
  88. package/dist/src/formats/cx/importer.js +2220 -0
  89. package/dist/src/formats/cx/importer.js.map +1 -0
  90. package/dist/src/formats/cx/index.d.ts +7 -0
  91. package/dist/src/formats/cx/index.d.ts.map +1 -0
  92. package/dist/src/formats/cx/index.js +7 -0
  93. package/dist/src/formats/cx/index.js.map +1 -0
  94. package/dist/src/formats/cx2/exporter.d.ts +59 -0
  95. package/dist/src/formats/cx2/exporter.d.ts.map +1 -0
  96. package/dist/src/formats/cx2/exporter.js +864 -0
  97. package/dist/src/formats/cx2/exporter.js.map +1 -0
  98. package/dist/src/formats/cx2/importer.d.ts +169 -0
  99. package/dist/src/formats/cx2/importer.d.ts.map +1 -0
  100. package/dist/src/formats/cx2/importer.js +1652 -0
  101. package/dist/src/formats/cx2/importer.js.map +1 -0
  102. package/dist/src/formats/cx2/index.d.ts +7 -0
  103. package/dist/src/formats/cx2/index.d.ts.map +1 -0
  104. package/dist/src/formats/cx2/index.js +7 -0
  105. package/dist/src/formats/cx2/index.js.map +1 -0
  106. package/dist/src/formats/cys/constants.d.ts +68 -0
  107. package/dist/src/formats/cys/constants.d.ts.map +1 -0
  108. package/dist/src/formats/cys/constants.js +76 -0
  109. package/dist/src/formats/cys/constants.js.map +1 -0
  110. package/dist/src/formats/cys/importer.d.ts +38 -0
  111. package/dist/src/formats/cys/importer.d.ts.map +1 -0
  112. package/dist/src/formats/cys/importer.js +830 -0
  113. package/dist/src/formats/cys/importer.js.map +1 -0
  114. package/dist/src/formats/cys/index.d.ts +7 -0
  115. package/dist/src/formats/cys/index.d.ts.map +1 -0
  116. package/dist/src/formats/cys/index.js +7 -0
  117. package/dist/src/formats/cys/index.js.map +1 -0
  118. package/dist/src/formats/cys/session.d.ts +132 -0
  119. package/dist/src/formats/cys/session.d.ts.map +1 -0
  120. package/dist/src/formats/cys/session.js +315 -0
  121. package/dist/src/formats/cys/session.js.map +1 -0
  122. package/dist/src/formats/cys/tables.d.ts +90 -0
  123. package/dist/src/formats/cys/tables.d.ts.map +1 -0
  124. package/dist/src/formats/cys/tables.js +293 -0
  125. package/dist/src/formats/cys/tables.js.map +1 -0
  126. package/dist/src/formats/dot/exporter.d.ts +2 -0
  127. package/dist/src/formats/dot/exporter.d.ts.map +1 -1
  128. package/dist/src/formats/dot/exporter.js +12 -0
  129. package/dist/src/formats/dot/exporter.js.map +1 -1
  130. package/dist/src/formats/dot/importer.js +6 -2
  131. package/dist/src/formats/dot/importer.js.map +1 -1
  132. package/dist/src/formats/gexf/exporter.d.ts +2 -0
  133. package/dist/src/formats/gexf/exporter.d.ts.map +1 -1
  134. package/dist/src/formats/gexf/exporter.js +6 -0
  135. package/dist/src/formats/gexf/exporter.js.map +1 -1
  136. package/dist/src/formats/gexf/importer.d.ts +0 -2
  137. package/dist/src/formats/gexf/importer.d.ts.map +1 -1
  138. package/dist/src/formats/gexf/importer.js +1 -3
  139. package/dist/src/formats/gexf/importer.js.map +1 -1
  140. package/dist/src/formats/gexf/index.d.ts +1 -1
  141. package/dist/src/formats/gexf/index.js +2 -2
  142. package/dist/src/formats/gexf/index.js.map +1 -1
  143. package/dist/src/formats/gml/exporter.d.ts +0 -8
  144. package/dist/src/formats/gml/exporter.d.ts.map +1 -1
  145. package/dist/src/formats/gml/exporter.js +8 -18
  146. package/dist/src/formats/gml/exporter.js.map +1 -1
  147. package/dist/src/formats/json/importer.d.ts.map +1 -1
  148. package/dist/src/formats/json/importer.js +6 -104
  149. package/dist/src/formats/json/importer.js.map +1 -1
  150. package/dist/src/formats/xgmml/columns.d.ts +152 -0
  151. package/dist/src/formats/xgmml/columns.d.ts.map +1 -0
  152. package/dist/src/formats/xgmml/columns.js +593 -0
  153. package/dist/src/formats/xgmml/columns.js.map +1 -0
  154. package/dist/src/formats/xgmml/constants.d.ts +196 -0
  155. package/dist/src/formats/xgmml/constants.d.ts.map +1 -0
  156. package/dist/src/formats/xgmml/constants.js +197 -0
  157. package/dist/src/formats/xgmml/constants.js.map +1 -0
  158. package/dist/src/formats/xgmml/document.d.ts +361 -0
  159. package/dist/src/formats/xgmml/document.d.ts.map +1 -0
  160. package/dist/src/formats/xgmml/document.js +695 -0
  161. package/dist/src/formats/xgmml/document.js.map +1 -0
  162. package/dist/src/formats/xgmml/emit.d.ts +341 -0
  163. package/dist/src/formats/xgmml/emit.d.ts.map +1 -0
  164. package/dist/src/formats/xgmml/emit.js +1275 -0
  165. package/dist/src/formats/xgmml/emit.js.map +1 -0
  166. package/dist/src/formats/xgmml/exporter.d.ts +24 -0
  167. package/dist/src/formats/xgmml/exporter.d.ts.map +1 -0
  168. package/dist/src/formats/xgmml/exporter.js +999 -0
  169. package/dist/src/formats/xgmml/exporter.js.map +1 -0
  170. package/dist/src/formats/xgmml/importer.d.ts +58 -0
  171. package/dist/src/formats/xgmml/importer.d.ts.map +1 -0
  172. package/dist/src/formats/xgmml/importer.js +360 -0
  173. package/dist/src/formats/xgmml/importer.js.map +1 -0
  174. package/dist/src/formats/xgmml/index.d.ts +8 -0
  175. package/dist/src/formats/xgmml/index.d.ts.map +1 -0
  176. package/dist/src/formats/xgmml/index.js +8 -0
  177. package/dist/src/formats/xgmml/index.js.map +1 -0
  178. package/dist/src/formats/xgmml/values.d.ts +60 -0
  179. package/dist/src/formats/xgmml/values.d.ts.map +1 -0
  180. package/dist/src/formats/xgmml/values.js +148 -0
  181. package/dist/src/formats/xgmml/values.js.map +1 -0
  182. package/dist/src/index.d.ts +4 -0
  183. package/dist/src/index.d.ts.map +1 -1
  184. package/dist/src/index.js +4 -0
  185. package/dist/src/index.js.map +1 -1
  186. package/dist/src/registry.d.ts +7 -0
  187. package/dist/src/registry.d.ts.map +1 -1
  188. package/dist/src/registry.js +30 -10
  189. package/dist/src/registry.js.map +1 -1
  190. package/dist/src/sniff.d.ts +1 -1
  191. package/dist/src/sniff.d.ts.map +1 -1
  192. package/dist/src/sniff.js +4 -0
  193. package/dist/src/sniff.js.map +1 -1
  194. package/dist/xgmml.d.ts +1 -0
  195. package/dist/xgmml.js +9 -0
  196. package/dist/xgmml.js.map +1 -0
  197. package/package.json +25 -2
  198. package/src/common/cell-budget.ts +170 -0
  199. package/src/common/input.ts +40 -0
  200. package/src/common/json-elements.ts +1147 -0
  201. package/src/common/options.ts +1 -1
  202. package/src/common/xml.ts +127 -6
  203. package/src/common/zip.ts +472 -0
  204. package/src/formats/cx/importer.ts +2733 -0
  205. package/src/formats/cx/index.ts +7 -0
  206. package/src/formats/cx2/exporter.ts +1036 -0
  207. package/src/formats/cx2/importer.ts +2187 -0
  208. package/src/formats/cx2/index.ts +7 -0
  209. package/src/formats/cys/constants.ts +99 -0
  210. package/src/formats/cys/importer.ts +1100 -0
  211. package/src/formats/cys/index.ts +7 -0
  212. package/src/formats/cys/session.ts +428 -0
  213. package/src/formats/cys/tables.ts +400 -0
  214. package/src/formats/dot/exporter.ts +17 -0
  215. package/src/formats/dot/importer.ts +6 -2
  216. package/src/formats/gexf/exporter.ts +11 -0
  217. package/src/formats/gexf/importer.ts +1 -2
  218. package/src/formats/gexf/index.ts +1 -1
  219. package/src/formats/gml/exporter.ts +8 -19
  220. package/src/formats/json/importer.ts +6 -105
  221. package/src/formats/xgmml/columns.ts +747 -0
  222. package/src/formats/xgmml/constants.ts +265 -0
  223. package/src/formats/xgmml/document.ts +975 -0
  224. package/src/formats/xgmml/emit.ts +1563 -0
  225. package/src/formats/xgmml/exporter.ts +1215 -0
  226. package/src/formats/xgmml/importer.ts +491 -0
  227. package/src/formats/xgmml/index.ts +8 -0
  228. package/src/formats/xgmml/values.ts +183 -0
  229. package/src/index.ts +19 -0
  230. package/src/registry.ts +46 -15
  231. package/src/sniff.ts +18 -1
  232. package/dist/chunks/escape-D-gZWO26.js.map +0 -1
  233. package/dist/chunks/importer-Br_QeAeE.js.map +0 -1
  234. package/dist/chunks/importer-Du5crN9l.js.map +0 -1
  235. package/dist/chunks/writer-GAdltGmC.js.map +0 -1
@@ -455,7 +455,7 @@ function encodingOption(value: unknown): string | null {
455
455
  * @param value - the value
456
456
  * @returns the JSON text of a primitive, or the type name otherwise
457
457
  */
458
- function describe(value: unknown): string {
458
+ export function describe(value: unknown): string {
459
459
  switch (typeof value) {
460
460
  case "string":
461
461
  return JSON.stringify(value);
package/src/common/xml.ts CHANGED
@@ -72,6 +72,38 @@ export class XmlSyntaxError extends Error {
72
72
  }
73
73
  }
74
74
 
75
+ /**
76
+ * Opt-in repairs of two defects the Cytoscape XGMML writer is known to produce (research note
77
+ * `research-xgmml.md` 3.8 and 5). Each is off unless its callback is given; GEXF and GraphML never
78
+ * pass them, so their documents stay strictly well-formed. The callback is told the line of each
79
+ * repair, so the importer can warn per occurrence.
80
+ */
81
+ export interface XmlRepairs {
82
+ /**
83
+ * Read an `&` that is not followed by a `;` within the next 7 characters as `&` (Cytoscape's
84
+ * `cytoscape.xgmml.repair.bare.ampersands` lookahead); an `&name;` that does end in time is
85
+ * still decoded, and still fatal when the entity is unknown. A complete numeric character
86
+ * reference is decoded whatever its length (`😀`), where Cytoscape's byte lookahead
87
+ * would turn it into text.
88
+ */
89
+ readonly bareAmpersand?: ((line: number) => void) | undefined;
90
+ /**
91
+ * Join a high-surrogate character reference immediately followed by a low-surrogate one
92
+ * (`��`, which XML 1.0 forbids but Cytoscape writes for astral characters) into
93
+ * the one character they encode; a lone surrogate stays fatal.
94
+ */
95
+ readonly surrogatePair?: ((line: number) => void) | undefined;
96
+ }
97
+
98
+ /** How far Cytoscape's bare-ampersand repair looks for the `;` that ends an entity reference. */
99
+ const BARE_AMPERSAND_LOOKAHEAD = 7;
100
+
101
+ /** A numeric character reference at the start of a text: `{` or ``. */
102
+ const CHAR_REFERENCE = /^&#(?:[xX]([0-9a-fA-F]{1,6})|([0-9]{1,7}));/;
103
+
104
+ /** A numeric character reference at the end of a text. */
105
+ const TRAILING_CHAR_REFERENCE = /&#(?:[xX]([0-9a-fA-F]{1,6})|([0-9]{1,7}));$/;
106
+
75
107
  const NAMED_ENTITIES: Readonly<Record<string, string>> = {
76
108
  lt: "<",
77
109
  gt: ">",
@@ -280,9 +312,10 @@ function isXmlChar(cp: number): boolean {
280
312
  * Decode the predefined entities and character references of a text.
281
313
  * @param raw - the text as written
282
314
  * @param line - the line, for errors
315
+ * @param repairs - the opt-in repairs (XGMML only); none by default
283
316
  * @returns the decoded text
284
317
  */
285
- export function decodeEntities(raw: string, line: number): string {
318
+ export function decodeEntities(raw: string, line: number, repairs?: XmlRepairs): string {
286
319
  let amp = raw.indexOf("&");
287
320
  if (amp < 0) {
288
321
  return raw;
@@ -292,6 +325,18 @@ export function decodeEntities(raw: string, line: number): string {
292
325
  while (amp >= 0) {
293
326
  out += raw.slice(start, amp);
294
327
  const semi = raw.indexOf(";", amp + 1);
328
+ if (
329
+ repairs?.bareAmpersand !== undefined &&
330
+ (semi < 0 || semi - amp > BARE_AMPERSAND_LOOKAHEAD) &&
331
+ // a well-formed character reference longer than the lookahead (&#128512;) is not bare
332
+ !CHAR_REFERENCE.test(raw.slice(amp, amp + 12))
333
+ ) {
334
+ repairs.bareAmpersand(lineAt(raw, amp, line));
335
+ out += "&";
336
+ start = amp + 1;
337
+ amp = raw.indexOf("&", start);
338
+ continue;
339
+ }
295
340
  if (semi < 0) {
296
341
  throw new XmlSyntaxError("unterminated entity reference", line);
297
342
  }
@@ -301,6 +346,14 @@ export function decodeEntities(raw: string, line: number): string {
301
346
  const digits = name.slice(hex ? 2 : 1);
302
347
  const ok = hex ? /^[0-9a-fA-F]{1,6}$/.test(digits) : /^[0-9]{1,7}$/.test(digits);
303
348
  const cp = ok ? Number.parseInt(digits, hex ? 16 : 10) : -1;
349
+ const low = cp >= 0xd800 && cp <= 0xdbff ? lowSurrogateAfter(raw, semi + 1, repairs) : null;
350
+ if (low !== null) {
351
+ repairs?.surrogatePair?.(lineAt(raw, amp, line));
352
+ out += String.fromCharCode(cp, low.unit);
353
+ start = low.end;
354
+ amp = raw.indexOf("&", start);
355
+ continue;
356
+ }
304
357
  if (cp < 0 || !isXmlChar(cp)) {
305
358
  throw new XmlSyntaxError(`invalid character reference &${name};`, line);
306
359
  }
@@ -318,6 +371,45 @@ export function decodeEntities(raw: string, line: number): string {
318
371
  return out + raw.slice(start);
319
372
  }
320
373
 
374
+ /**
375
+ * The line of a position inside a text that starts on a known line.
376
+ * @param raw - the text
377
+ * @param at - the position
378
+ * @param line - the line the text starts on
379
+ * @returns the line of the position
380
+ */
381
+ function lineAt(raw: string, at: number, line: number): number {
382
+ let n = line;
383
+ for (let i = raw.indexOf("\n"); i >= 0 && i < at; i = raw.indexOf("\n", i + 1)) {
384
+ n++;
385
+ }
386
+ return n;
387
+ }
388
+
389
+ /**
390
+ * The low-surrogate character reference that may follow a high-surrogate one, when the
391
+ * surrogate-pair repair is on.
392
+ * @param raw - the text
393
+ * @param at - where the next reference would start
394
+ * @param repairs - the repairs in force
395
+ * @returns the low surrogate code unit and the index after its reference, or null
396
+ */
397
+ function lowSurrogateAfter(
398
+ raw: string,
399
+ at: number,
400
+ repairs: XmlRepairs | undefined,
401
+ ): { readonly unit: number; readonly end: number } | null {
402
+ if (repairs?.surrogatePair === undefined || raw.charCodeAt(at) !== 38) {
403
+ return null;
404
+ }
405
+ const match = CHAR_REFERENCE.exec(raw.slice(at, at + 12));
406
+ if (match === null) {
407
+ return null;
408
+ }
409
+ const unit = match[1] === undefined ? Number.parseInt(match[2], 10) : Number.parseInt(match[1], 16);
410
+ return unit >= 0xdc00 && unit <= 0xdfff ? { unit, end: at + match[0].length } : null;
411
+ }
412
+
321
413
  /**
322
414
  * Whether a text is whitespace only.
323
415
  * @param text - the text
@@ -347,9 +439,14 @@ export function localName(name: string): string {
347
439
  * malformed input; any error the handler throws propagates unchanged.
348
440
  * @param chunks - the text (already UTF-8 decoded, BOM removed)
349
441
  * @param handler - the event sink
442
+ * @param repairs - the opt-in repairs (XGMML only); none by default
350
443
  */
351
- export async function tokenizeXml(chunks: AsyncIterable<string>, handler: XmlHandler): Promise<void> {
352
- const tokenizer = new XmlTokenizer(handler);
444
+ export async function tokenizeXml(
445
+ chunks: AsyncIterable<string>,
446
+ handler: XmlHandler,
447
+ repairs?: XmlRepairs,
448
+ ): Promise<void> {
449
+ const tokenizer = new XmlTokenizer(handler, repairs);
353
450
  for await (const chunk of chunks) {
354
451
  tokenizer.push(chunk);
355
452
  }
@@ -435,12 +532,16 @@ export class XmlTokenizer {
435
532
 
436
533
  private rootClosed = false;
437
534
 
535
+ private readonly repairs: XmlRepairs | undefined;
536
+
438
537
  /**
439
538
  * Create a tokenizer.
440
539
  * @param handler - the event sink
540
+ * @param repairs - the opt-in repairs (XGMML only); none by default
441
541
  */
442
- constructor(handler: XmlHandler) {
542
+ constructor(handler: XmlHandler, repairs?: XmlRepairs) {
443
543
  this.handler = handler;
544
+ this.repairs = repairs;
444
545
  }
445
546
 
446
547
  /**
@@ -677,6 +778,9 @@ export class XmlTokenizer {
677
778
  if (amp >= pos && amp >= length - MAX_ENTITY_LENGTH && buffer.indexOf(";", amp) < 0) {
678
779
  end = amp;
679
780
  }
781
+ if (this.repairs?.surrogatePair !== undefined) {
782
+ end = this.holdHighSurrogate(pos, end);
783
+ }
680
784
  }
681
785
  this.takeText(pos, end);
682
786
  pos = end;
@@ -928,7 +1032,7 @@ export class XmlTokenizer {
928
1032
  this.line,
929
1033
  );
930
1034
  }
931
- attrs.set(attrName, decodeEntities(normalizeAttributeValue(raw), this.line));
1035
+ attrs.set(attrName, decodeEntities(normalizeAttributeValue(raw), this.line, this.repairs));
932
1036
  i = close + 1;
933
1037
  }
934
1038
  }
@@ -962,6 +1066,23 @@ export class XmlTokenizer {
962
1066
  return j >= length && !this.final ? -1 : j;
963
1067
  }
964
1068
 
1069
+ /**
1070
+ * Under the surrogate-pair repair, hold back a high-surrogate character reference that ends
1071
+ * the text taken so far, so it is decoded together with the low one the next chunk may start with.
1072
+ * @param pos - the start of the text
1073
+ * @param end - the end of the text that would be taken
1074
+ * @returns the end to take up to
1075
+ */
1076
+ private holdHighSurrogate(pos: number, end: number): number {
1077
+ const tail = this.buffer.slice(Math.max(pos, end - 12), end);
1078
+ const match = TRAILING_CHAR_REFERENCE.exec(tail);
1079
+ if (match === null) {
1080
+ return end;
1081
+ }
1082
+ const unit = match[1] === undefined ? Number.parseInt(match[2], 10) : Number.parseInt(match[1], 16);
1083
+ return unit >= 0xd800 && unit <= 0xdbff ? end - match[0].length : end;
1084
+ }
1085
+
965
1086
  /**
966
1087
  * Move text from the buffer into the current run.
967
1088
  * @param start - the start index
@@ -978,7 +1099,7 @@ export class XmlTokenizer {
978
1099
  if (hasIllegalXmlChar(raw)) {
979
1100
  throw new XmlSyntaxError("a character XML 1.0 forbids appears in character data", this.line);
980
1101
  }
981
- this.text += decodeEntities(raw, this.line);
1102
+ this.text += decodeEntities(raw, this.line, this.repairs);
982
1103
  this.advanceLine(start, end);
983
1104
  }
984
1105
 
@@ -0,0 +1,472 @@
1
+ /**
2
+ * A zip reader with no dependency (design `design/graph-io/cytoscape-and-obo/design.md` section
3
+ * 3.3; PKWARE APPNOTE 6.3.10, https://pkware.cachefly.net/webdocs/casestudies/APPNOTE.TXT), for
4
+ * the Cytoscape session importer.
5
+ *
6
+ * The whole archive is in memory: every Cytoscape-written entry has a data descriptor and zero
7
+ * sizes in its local header, so the sizes come from the central directory at the end. The end
8
+ * record is searched in the last 65,557 bytes (a comment can be 65,535 bytes long); the zip64
9
+ * locator and the `0x0001` extra field are followed; bytes prepended to the archive (a
10
+ * self-extracting stub) are tolerated by the shift between the directory's stated and actual
11
+ * offsets. Deflate (method 8) is inflated by wrapping the raw deflate bytes in a gzip member
12
+ * (header, data, the directory's CRC-32 and size) and passing it to `DecompressionStream("gzip")`,
13
+ * which every supported runtime has (Node 18, unlike `"deflate-raw"`) and which checks the CRC and
14
+ * the length itself; stored entries (method 0) are checked against a CRC-32 table here. Every
15
+ * other method, and encryption, is refused by name. Entry names are decoded by `decodeEntryName()`
16
+ * (the one place bytes become text) and are matching keys only, never paths.
17
+ */
18
+
19
+ import { decodeEntryName, throwIfAborted } from "./input.js";
20
+
21
+ /** Why a zip could not be read; the importer maps each kind to its own issue code. */
22
+ type ZipErrorKind = "not-zip" | "corrupt" | "unsupported" | "too-large";
23
+
24
+ /** A zip that cannot be read, or an entry that cannot be inflated. */
25
+ export class ZipError extends Error {
26
+ /** What went wrong. */
27
+ readonly kind: ZipErrorKind;
28
+
29
+ /**
30
+ * Create the error.
31
+ * @param kind - what went wrong
32
+ * @param message - a plain-ASCII message
33
+ */
34
+ constructor(kind: ZipErrorKind, message: string) {
35
+ super(message);
36
+ this.name = "ZipError";
37
+ this.kind = kind;
38
+ }
39
+ }
40
+
41
+ /** One entry of the central directory. */
42
+ export interface ZipEntry {
43
+ /** The name, decoded, as the central directory spells it. */
44
+ readonly name: string;
45
+ /** The compression method (0 stored, 8 deflate, ...). */
46
+ readonly method: number;
47
+ /** The general purpose bit flags. */
48
+ readonly flags: number;
49
+ /** The CRC-32 of the uncompressed data. */
50
+ readonly crc32: number;
51
+ /** The compressed size in bytes. */
52
+ readonly compressedSize: number;
53
+ /** The uncompressed size in bytes. */
54
+ readonly size: number;
55
+ /** The offset of the local header in the archive bytes (the prepended-data shift applied). */
56
+ readonly localOffset: number;
57
+ /** The name ends in `/`: a directory entry, which holds no data. */
58
+ readonly directory: boolean;
59
+ }
60
+
61
+ /** How one entry is read. */
62
+ export interface ReadEntryOptions {
63
+ /** The cancellation signal. */
64
+ readonly signal?: AbortSignal | null | undefined;
65
+ /** Called with the bytes inflated so far by this call. */
66
+ readonly onBytes?: ((inflated: number) => void) | undefined;
67
+ /** The most uncompressed bytes this entry may have (the import's remaining budget). */
68
+ readonly maxBytes: number;
69
+ /** The largest uncompressed-to-compressed ratio allowed (above `RATIO_FLOOR` bytes). */
70
+ readonly maxRatio: number;
71
+ }
72
+
73
+ /** A per-entry size under which the ratio limit does not apply (tiny entries compress well). */
74
+ export const RATIO_FLOOR = 1024 * 1024;
75
+
76
+ const EOCD_SIGNATURE = 0x06054b50;
77
+ const ZIP64_LOCATOR_SIGNATURE = 0x07064b50;
78
+ const ZIP64_EOCD_SIGNATURE = 0x06064b50;
79
+ const CENTRAL_SIGNATURE = 0x02014b50;
80
+ const LOCAL_SIGNATURE = 0x04034b50;
81
+ const EOCD_SIZE = 22;
82
+ const MAX_COMMENT = 65535;
83
+ const CENTRAL_SIZE = 46;
84
+ const LOCAL_SIZE = 30;
85
+ const ZIP64_EOCD_SIZE = 56;
86
+ const ZIP64_LOCATOR_SIZE = 20;
87
+ const U16_MAX = 0xffff;
88
+ const U32_MAX = 0xffffffff;
89
+ /** Bytes between two cancellation checks while a stored entry's CRC is computed. */
90
+ const CRC_SLICE = 1024 * 1024;
91
+
92
+ /** The compression methods other tools write, by number, for the refusal message. */
93
+ const METHOD_NAMES: Readonly<Record<number, string>> = {
94
+ 1: "shrink",
95
+ 6: "implode",
96
+ 9: "deflate64",
97
+ 12: "bzip2",
98
+ 14: "LZMA",
99
+ 93: "zstd",
100
+ 95: "xz",
101
+ 96: "JPEG",
102
+ 97: "WavPack",
103
+ 98: "PPMd",
104
+ 99: "AES encryption",
105
+ };
106
+
107
+ /**
108
+ * Whether bytes start with a local file header signature (`PK\x03\x04`).
109
+ * @param bytes - the bytes
110
+ * @returns true for the start of a zip
111
+ */
112
+ export function startsLikeZip(bytes: Uint8Array): boolean {
113
+ return bytes.byteLength >= 4 && bytes[0] === 0x50 && bytes[1] === 0x4b && bytes[2] === 3 && bytes[3] === 4;
114
+ }
115
+
116
+ /**
117
+ * Read the central directory.
118
+ * @param bytes - the whole archive
119
+ * @returns the entries in directory order; ZipError "not-zip" when there is no end record and the
120
+ * bytes do not start like a zip, "corrupt" for an end record or directory that does not fit,
121
+ * "unsupported" for a split archive
122
+ */
123
+ export function readZipDirectory(bytes: Uint8Array): ZipEntry[] {
124
+ const view = new DataView(bytes.buffer, bytes.byteOffset, bytes.byteLength);
125
+ const eocd = findEndRecord(bytes, view);
126
+ if (eocd < 0) {
127
+ throw startsLikeZip(bytes)
128
+ ? new ZipError("corrupt", "the archive has no end of central directory record (truncated?)")
129
+ : new ZipError("not-zip", "the input is not a zip archive");
130
+ }
131
+ let disk = view.getUint16(eocd + 4, true);
132
+ let directoryDisk = view.getUint16(eocd + 6, true);
133
+ let count = view.getUint16(eocd + 10, true);
134
+ let directorySize = view.getUint32(eocd + 12, true);
135
+ let directoryOffset = view.getUint32(eocd + 16, true);
136
+ let directoryEnd = eocd;
137
+ const locator = eocd - ZIP64_LOCATOR_SIZE;
138
+ if (locator >= 0 && view.getUint32(locator, true) === ZIP64_LOCATOR_SIGNATURE) {
139
+ const record = zip64EndRecord(view, locator);
140
+ disk = view.getUint32(record + 16, true);
141
+ directoryDisk = view.getUint32(record + 20, true);
142
+ count = readU64(view, record + 32);
143
+ directorySize = readU64(view, record + 40);
144
+ directoryOffset = readU64(view, record + 48);
145
+ directoryEnd = record;
146
+ } else if (count === U16_MAX || directorySize === U32_MAX || directoryOffset === U32_MAX) {
147
+ throw new ZipError("corrupt", "the end record needs a zip64 record the archive does not have");
148
+ }
149
+ if (disk !== 0 || directoryDisk !== 0) {
150
+ throw new ZipError("unsupported", "a split (multi-disk) archive cannot be read");
151
+ }
152
+ const start = directoryEnd - directorySize;
153
+ if (start < 0 || directoryOffset > bytes.byteLength) {
154
+ throw new ZipError("corrupt", "the central directory lies outside the archive");
155
+ }
156
+ // data prepended to the archive moves every offset by the same amount
157
+ const shift = start - directoryOffset;
158
+ if (shift < 0) {
159
+ throw new ZipError("corrupt", "the central directory offset does not match its position");
160
+ }
161
+ const entries: ZipEntry[] = [];
162
+ let at = start;
163
+ for (let i = 0; i < count; i++) {
164
+ if (at + CENTRAL_SIZE > directoryEnd || view.getUint32(at, true) !== CENTRAL_SIGNATURE) {
165
+ throw new ZipError("corrupt", `central directory entry ${i + 1} of ${count} is damaged`);
166
+ }
167
+ const nameLength = view.getUint16(at + 28, true);
168
+ const extraLength = view.getUint16(at + 30, true);
169
+ const commentLength = view.getUint16(at + 32, true);
170
+ const next = at + CENTRAL_SIZE + nameLength + extraLength + commentLength;
171
+ if (next > directoryEnd) {
172
+ throw new ZipError("corrupt", `central directory entry ${i + 1} runs past the directory`);
173
+ }
174
+ const name = decodeEntryName(bytes.subarray(at + CENTRAL_SIZE, at + CENTRAL_SIZE + nameLength));
175
+ let compressedSize = view.getUint32(at + 20, true);
176
+ let size = view.getUint32(at + 24, true);
177
+ let localOffset = view.getUint32(at + 42, true);
178
+ if (compressedSize === U32_MAX || size === U32_MAX || localOffset === U32_MAX) {
179
+ ({ compressedSize, size, localOffset } = zip64Extra(view, at + CENTRAL_SIZE + nameLength, extraLength, {
180
+ compressedSize,
181
+ size,
182
+ localOffset,
183
+ }));
184
+ }
185
+ entries.push(
186
+ Object.freeze({
187
+ name,
188
+ method: view.getUint16(at + 10, true),
189
+ flags: view.getUint16(at + 8, true),
190
+ crc32: view.getUint32(at + 16, true),
191
+ compressedSize,
192
+ size,
193
+ localOffset: localOffset + shift,
194
+ directory: name.endsWith("/"),
195
+ }),
196
+ );
197
+ at = next;
198
+ }
199
+ return entries;
200
+ }
201
+
202
+ /**
203
+ * The position of the end of central directory record: the last signature whose comment length
204
+ * fits the archive, else the last signature at all (an archive with bytes after its comment).
205
+ * @param bytes - the archive
206
+ * @param view - a view of it
207
+ * @returns the offset, or -1
208
+ */
209
+ function findEndRecord(bytes: Uint8Array, view: DataView): number {
210
+ const lowest = Math.max(0, bytes.byteLength - EOCD_SIZE - MAX_COMMENT);
211
+ let fallback = -1;
212
+ for (let at = bytes.byteLength - EOCD_SIZE; at >= lowest; at--) {
213
+ if (view.getUint32(at, true) !== EOCD_SIGNATURE) {
214
+ continue;
215
+ }
216
+ if (at + EOCD_SIZE + view.getUint16(at + 20, true) === bytes.byteLength) {
217
+ return at;
218
+ }
219
+ if (fallback < 0) {
220
+ fallback = at;
221
+ }
222
+ }
223
+ return fallback;
224
+ }
225
+
226
+ /**
227
+ * The zip64 end of central directory record a locator points at (or, when the archive has bytes
228
+ * prepended, the one right before the locator).
229
+ * @param view - the archive
230
+ * @param locator - the locator's offset
231
+ * @returns the record's offset
232
+ */
233
+ function zip64EndRecord(view: DataView, locator: number): number {
234
+ const stated = readU64(view, locator + 8);
235
+ for (const at of [stated, locator - ZIP64_EOCD_SIZE]) {
236
+ if (at >= 0 && at + ZIP64_EOCD_SIZE <= locator && view.getUint32(at, true) === ZIP64_EOCD_SIGNATURE) {
237
+ return at;
238
+ }
239
+ }
240
+ throw new ZipError("corrupt", "the zip64 end record the locator names is missing");
241
+ }
242
+
243
+ /**
244
+ * The zip64 extended information extra field (header id 0x0001) of a central entry: the 64-bit
245
+ * values of the fields whose 32-bit slot is 0xFFFFFFFF, in the order APPNOTE 4.5.3 gives.
246
+ * @param view - the archive
247
+ * @param at - the start of the extra field
248
+ * @param length - its length
249
+ * @param fields - the 32-bit values
250
+ * @param fields.compressedSize - the compressed size slot
251
+ * @param fields.size - the uncompressed size slot
252
+ * @param fields.localOffset - the local header offset slot
253
+ * @returns the resolved values
254
+ */
255
+ function zip64Extra(
256
+ view: DataView,
257
+ at: number,
258
+ length: number,
259
+ fields: { compressedSize: number; size: number; localOffset: number },
260
+ ): { compressedSize: number; size: number; localOffset: number } {
261
+ const out = { ...fields };
262
+ for (let p = at; p + 4 <= at + length; ) {
263
+ const id = view.getUint16(p, true);
264
+ const dataLength = view.getUint16(p + 2, true);
265
+ if (id === 1) {
266
+ let q = p + 4;
267
+ const end = q + dataLength;
268
+ for (const key of ["size", "compressedSize", "localOffset"] as const) {
269
+ if (fields[key] === U32_MAX) {
270
+ if (q + 8 > end) {
271
+ throw new ZipError("corrupt", "a zip64 extra field is too short");
272
+ }
273
+ out[key] = readU64(view, q);
274
+ q += 8;
275
+ }
276
+ }
277
+ return out;
278
+ }
279
+ p += 4 + dataLength;
280
+ }
281
+ throw new ZipError("corrupt", "an entry needs a zip64 extra field it does not have");
282
+ }
283
+
284
+ /**
285
+ * An unsigned 64-bit little-endian integer as a number (exact up to 2^53).
286
+ * @param view - the bytes
287
+ * @param at - the offset
288
+ * @returns the value; ZipError "unsupported" beyond 2^53
289
+ */
290
+ function readU64(view: DataView, at: number): number {
291
+ const high = view.getUint32(at + 4, true);
292
+ if (high >= 0x200000) {
293
+ throw new ZipError("unsupported", "a zip64 size or offset beyond 2^53 cannot be read");
294
+ }
295
+ return high * 0x100000000 + view.getUint32(at, true);
296
+ }
297
+
298
+ /**
299
+ * Read one entry's data: stored entries are sliced and CRC-checked, deflated ones inflated.
300
+ * @param bytes - the whole archive
301
+ * @param entry - the entry
302
+ * @param options - cancellation, progress and the size limits
303
+ * @returns the uncompressed bytes; ZipError for a damaged, encrypted, unsupported or oversized entry
304
+ */
305
+ export async function readZipEntry(bytes: Uint8Array, entry: ZipEntry, options: ReadEntryOptions): Promise<Uint8Array> {
306
+ if ((entry.flags & 1) !== 0 || entry.method === 99) {
307
+ throw new ZipError("unsupported", `${entry.name}: the entry is encrypted`);
308
+ }
309
+ if (entry.method !== 0 && entry.method !== 8) {
310
+ const name = METHOD_NAMES[entry.method] ?? `method ${entry.method}`;
311
+ throw new ZipError(
312
+ "unsupported",
313
+ `${entry.name}: compression ${name} is not supported (only stored and deflate)`,
314
+ );
315
+ }
316
+ if (entry.size > options.maxBytes) {
317
+ throw new ZipError(
318
+ "too-large",
319
+ `${entry.name}: ${entry.size} uncompressed bytes exceed the limit of ${options.maxBytes}`,
320
+ );
321
+ }
322
+ if (entry.size > RATIO_FLOOR && entry.size > options.maxRatio * Math.max(entry.compressedSize, 1)) {
323
+ throw new ZipError(
324
+ "too-large",
325
+ `${entry.name}: ${entry.size} bytes from ${entry.compressedSize} is a compression ratio above ${options.maxRatio}:1`,
326
+ );
327
+ }
328
+ const data = entryData(bytes, entry);
329
+ if (entry.method === 0) {
330
+ if (data.byteLength !== entry.size) {
331
+ throw new ZipError("corrupt", `${entry.name}: a stored entry's sizes disagree`);
332
+ }
333
+ let crc = 0xffffffff;
334
+ for (let at = 0; at < data.byteLength; at += CRC_SLICE) {
335
+ throwIfAborted(options.signal);
336
+ crc = crcUpdate(crc, data.subarray(at, Math.min(at + CRC_SLICE, data.byteLength)));
337
+ options.onBytes?.(Math.min(at + CRC_SLICE, data.byteLength));
338
+ }
339
+ if ((crc ^ 0xffffffff) >>> 0 !== entry.crc32) {
340
+ throw new ZipError("corrupt", `${entry.name}: the data does not match its CRC-32`);
341
+ }
342
+ return data;
343
+ }
344
+ return inflate(data, entry, options);
345
+ }
346
+
347
+ /**
348
+ * The compressed bytes of an entry, located through its local header (whose name and extra
349
+ * lengths can differ from the central directory's).
350
+ * @param bytes - the archive
351
+ * @param entry - the entry
352
+ * @returns exactly compressedSize bytes
353
+ */
354
+ function entryData(bytes: Uint8Array, entry: ZipEntry): Uint8Array {
355
+ const view = new DataView(bytes.buffer, bytes.byteOffset, bytes.byteLength);
356
+ const at = entry.localOffset;
357
+ if (at + LOCAL_SIZE > bytes.byteLength || view.getUint32(at, true) !== LOCAL_SIGNATURE) {
358
+ throw new ZipError("corrupt", `${entry.name}: the local header is missing`);
359
+ }
360
+ const start = at + LOCAL_SIZE + view.getUint16(at + 26, true) + view.getUint16(at + 28, true);
361
+ const end = start + entry.compressedSize;
362
+ if (end > bytes.byteLength) {
363
+ throw new ZipError("corrupt", `${entry.name}: the entry's data runs past the end of the archive (truncated?)`);
364
+ }
365
+ return bytes.subarray(start, end);
366
+ }
367
+
368
+ /**
369
+ * Inflate a deflate stream through `DecompressionStream("gzip")`, wrapped in a gzip member that
370
+ * carries the directory's CRC-32 and size, so the platform checks both.
371
+ * @param data - the raw deflate bytes
372
+ * @param entry - the entry
373
+ * @param options - cancellation, progress and limits
374
+ * @returns the bytes
375
+ */
376
+ async function inflate(data: Uint8Array, entry: ZipEntry, options: ReadEntryOptions): Promise<Uint8Array> {
377
+ if (typeof DecompressionStream !== "function") {
378
+ throw new ZipError(
379
+ "unsupported",
380
+ "this runtime has no DecompressionStream, so deflated entries cannot be read",
381
+ );
382
+ }
383
+ const member = new Uint8Array(10 + data.byteLength + 8);
384
+ member.set([0x1f, 0x8b, 8, 0, 0, 0, 0, 0, 0, 0xff]);
385
+ member.set(data, 10);
386
+ const trailer = new DataView(member.buffer, 10 + data.byteLength, 8);
387
+ trailer.setUint32(0, entry.crc32, true);
388
+ trailer.setUint32(4, entry.size % 0x100000000, true);
389
+ const source = new ReadableStream<BufferSource>({
390
+ start(controller): void {
391
+ controller.enqueue(member);
392
+ controller.close();
393
+ },
394
+ });
395
+ const reader = source.pipeThrough(new DecompressionStream("gzip")).getReader();
396
+ const parts: Uint8Array[] = [];
397
+ let total = 0;
398
+ try {
399
+ for (;;) {
400
+ throwIfAborted(options.signal);
401
+ let chunk: ReadableStreamReadResult<Uint8Array>;
402
+ try {
403
+ chunk = await reader.read();
404
+ } catch {
405
+ throw new ZipError(
406
+ "corrupt",
407
+ `${entry.name}: the deflate data is damaged or does not match its CRC-32 and size`,
408
+ );
409
+ }
410
+ if (chunk.done) {
411
+ break;
412
+ }
413
+ total += chunk.value.byteLength;
414
+ if (total > entry.size) {
415
+ throw new ZipError("corrupt", `${entry.name}: the data is longer than its directory size`);
416
+ }
417
+ parts.push(chunk.value);
418
+ options.onBytes?.(total);
419
+ }
420
+ } finally {
421
+ await reader.cancel().catch(() => undefined);
422
+ }
423
+ if (total !== entry.size) {
424
+ throw new ZipError("corrupt", `${entry.name}: the data is shorter than its directory size`);
425
+ }
426
+ if (parts.length === 1) {
427
+ return parts[0];
428
+ }
429
+ const out = new Uint8Array(total);
430
+ let at = 0;
431
+ for (const part of parts) {
432
+ out.set(part, at);
433
+ at += part.byteLength;
434
+ }
435
+ return out;
436
+ }
437
+
438
+ /** The CRC-32 table (polynomial 0xEDB88320). */
439
+ const CRC_TABLE = ((): Uint32Array => {
440
+ const table = new Uint32Array(256);
441
+ for (let n = 0; n < 256; n++) {
442
+ let c = n;
443
+ for (let k = 0; k < 8; k++) {
444
+ c = (c & 1) === 1 ? 0xedb88320 ^ (c >>> 1) : c >>> 1;
445
+ }
446
+ table[n] = c >>> 0;
447
+ }
448
+ return table;
449
+ })();
450
+
451
+ /**
452
+ * Continue a CRC-32 over more bytes.
453
+ * @param crc - the running value (start with 0xFFFFFFFF)
454
+ * @param bytes - the bytes
455
+ * @returns the running value (XOR with 0xFFFFFFFF to finish)
456
+ */
457
+ function crcUpdate(crc: number, bytes: Uint8Array): number {
458
+ let c = crc;
459
+ for (const byte of bytes) {
460
+ c = CRC_TABLE[(c ^ byte) & 0xff] ^ (c >>> 8);
461
+ }
462
+ return c >>> 0;
463
+ }
464
+
465
+ /**
466
+ * The CRC-32 of bytes.
467
+ * @param bytes - the bytes
468
+ * @returns the checksum
469
+ */
470
+ export function crc32(bytes: Uint8Array): number {
471
+ return (crcUpdate(0xffffffff, bytes) ^ 0xffffffff) >>> 0;
472
+ }