@graphty/graph-io 0.0.0 → 0.2.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +250 -28
- package/dist/chunks/children-CL3Cy0ez.js +238 -0
- package/dist/chunks/children-CL3Cy0ez.js.map +1 -0
- package/dist/chunks/escape-DyI8JofU.js +938 -0
- package/dist/chunks/escape-DyI8JofU.js.map +1 -0
- package/dist/chunks/importer-CQnJuWJw.js +2987 -0
- package/dist/chunks/importer-CQnJuWJw.js.map +1 -0
- package/dist/chunks/importer-CpCpfbxr.js +2015 -0
- package/dist/chunks/importer-CpCpfbxr.js.map +1 -0
- package/dist/chunks/importer-DbnGYr3_.js +2342 -0
- package/dist/chunks/importer-DbnGYr3_.js.map +1 -0
- package/dist/chunks/importer-GozH8DkN.js +3050 -0
- package/dist/chunks/importer-GozH8DkN.js.map +1 -0
- package/dist/chunks/records-CGpxszm1.js +605 -0
- package/dist/chunks/records-CGpxszm1.js.map +1 -0
- package/dist/chunks/text-CajMdVFy.js +189 -0
- package/dist/chunks/text-CajMdVFy.js.map +1 -0
- package/dist/chunks/writer-DxSKC7TL.js +2842 -0
- package/dist/chunks/writer-DxSKC7TL.js.map +1 -0
- package/dist/csv.d.ts +1 -0
- package/dist/csv.js +1702 -0
- package/dist/csv.js.map +1 -0
- package/dist/dot.d.ts +1 -0
- package/dist/dot.js +8 -0
- package/dist/dot.js.map +1 -0
- package/dist/gexf.d.ts +1 -0
- package/dist/gexf.js +3466 -0
- package/dist/gexf.js.map +1 -0
- package/dist/gml.d.ts +1 -0
- package/dist/gml.js +2647 -0
- package/dist/gml.js.map +1 -0
- package/dist/graph-io.d.ts +1 -0
- package/dist/graph-io.js +790 -0
- package/dist/graph-io.js.map +1 -0
- package/dist/graphml.d.ts +1 -0
- package/dist/graphml.js +8 -0
- package/dist/graphml.js.map +1 -0
- package/dist/json.d.ts +1 -0
- package/dist/json.js +11 -0
- package/dist/json.js.map +1 -0
- package/dist/neo4j.d.ts +1 -0
- package/dist/neo4j.js +2046 -0
- package/dist/neo4j.js.map +1 -0
- package/dist/pajek.d.ts +1 -0
- package/dist/pajek.js +8 -0
- package/dist/pajek.js.map +1 -0
- package/dist/src/children.d.ts +134 -0
- package/dist/src/children.d.ts.map +1 -0
- package/dist/src/children.js +274 -0
- package/dist/src/children.js.map +1 -0
- package/dist/src/common/attributes.d.ts +229 -0
- package/dist/src/common/attributes.d.ts.map +1 -0
- package/dist/src/common/attributes.js +368 -0
- package/dist/src/common/attributes.js.map +1 -0
- package/dist/src/common/codes.d.ts +105 -0
- package/dist/src/common/codes.d.ts.map +1 -0
- package/dist/src/common/codes.js +107 -0
- package/dist/src/common/codes.js.map +1 -0
- package/dist/src/common/declared-types.d.ts +84 -0
- package/dist/src/common/declared-types.d.ts.map +1 -0
- package/dist/src/common/declared-types.js +326 -0
- package/dist/src/common/declared-types.js.map +1 -0
- package/dist/src/common/direction.d.ts +206 -0
- package/dist/src/common/direction.d.ts.map +1 -0
- package/dist/src/common/direction.js +370 -0
- package/dist/src/common/direction.js.map +1 -0
- package/dist/src/common/escape.d.ts +92 -0
- package/dist/src/common/escape.d.ts.map +1 -0
- package/dist/src/common/escape.js +212 -0
- package/dist/src/common/escape.js.map +1 -0
- package/dist/src/common/export.d.ts +249 -0
- package/dist/src/common/export.d.ts.map +1 -0
- package/dist/src/common/export.js +594 -0
- package/dist/src/common/export.js.map +1 -0
- package/dist/src/common/format.d.ts +59 -0
- package/dist/src/common/format.d.ts.map +1 -0
- package/dist/src/common/format.js +106 -0
- package/dist/src/common/format.js.map +1 -0
- package/dist/src/common/ids.d.ts +83 -0
- package/dist/src/common/ids.d.ts.map +1 -0
- package/dist/src/common/ids.js +158 -0
- package/dist/src/common/ids.js.map +1 -0
- package/dist/src/common/input.d.ts +100 -0
- package/dist/src/common/input.d.ts.map +1 -0
- package/dist/src/common/input.js +335 -0
- package/dist/src/common/input.js.map +1 -0
- package/dist/src/common/lists.d.ts +34 -0
- package/dist/src/common/lists.d.ts.map +1 -0
- package/dist/src/common/lists.js +185 -0
- package/dist/src/common/lists.js.map +1 -0
- package/dist/src/common/options.d.ts +108 -0
- package/dist/src/common/options.d.ts.map +1 -0
- package/dist/src/common/options.js +265 -0
- package/dist/src/common/options.js.map +1 -0
- package/dist/src/common/report.d.ts +187 -0
- package/dist/src/common/report.d.ts.map +1 -0
- package/dist/src/common/report.js +274 -0
- package/dist/src/common/report.js.map +1 -0
- package/dist/src/common/temporal.d.ts +71 -0
- package/dist/src/common/temporal.d.ts.map +1 -0
- package/dist/src/common/temporal.js +266 -0
- package/dist/src/common/temporal.js.map +1 -0
- package/dist/src/common/text.d.ts +104 -0
- package/dist/src/common/text.d.ts.map +1 -0
- package/dist/src/common/text.js +255 -0
- package/dist/src/common/text.js.map +1 -0
- package/dist/src/common/weights.d.ts +77 -0
- package/dist/src/common/weights.d.ts.map +1 -0
- package/dist/src/common/weights.js +156 -0
- package/dist/src/common/weights.js.map +1 -0
- package/dist/src/common/writer.d.ts +51 -0
- package/dist/src/common/writer.d.ts.map +1 -0
- package/dist/src/common/writer.js +108 -0
- package/dist/src/common/writer.js.map +1 -0
- package/dist/src/common/xml.d.ts +245 -0
- package/dist/src/common/xml.d.ts.map +1 -0
- package/dist/src/common/xml.js +942 -0
- package/dist/src/common/xml.js.map +1 -0
- package/dist/src/formats/csv/exporter.d.ts +70 -0
- package/dist/src/formats/csv/exporter.d.ts.map +1 -0
- package/dist/src/formats/csv/exporter.js +682 -0
- package/dist/src/formats/csv/exporter.js.map +1 -0
- package/dist/src/formats/csv/header.d.ts +66 -0
- package/dist/src/formats/csv/header.d.ts.map +1 -0
- package/dist/src/formats/csv/header.js +152 -0
- package/dist/src/formats/csv/header.js.map +1 -0
- package/dist/src/formats/csv/importer.d.ts +82 -0
- package/dist/src/formats/csv/importer.d.ts.map +1 -0
- package/dist/src/formats/csv/importer.js +849 -0
- package/dist/src/formats/csv/importer.js.map +1 -0
- package/dist/src/formats/csv/index.d.ts +60 -0
- package/dist/src/formats/csv/index.d.ts.map +1 -0
- package/dist/src/formats/csv/index.js +63 -0
- package/dist/src/formats/csv/index.js.map +1 -0
- package/dist/src/formats/csv/records.d.ts +188 -0
- package/dist/src/formats/csv/records.d.ts.map +1 -0
- package/dist/src/formats/csv/records.js +702 -0
- package/dist/src/formats/csv/records.js.map +1 -0
- package/dist/src/formats/csv/values.d.ts +105 -0
- package/dist/src/formats/csv/values.d.ts.map +1 -0
- package/dist/src/formats/csv/values.js +192 -0
- package/dist/src/formats/csv/values.js.map +1 -0
- package/dist/src/formats/dot/exporter.d.ts +52 -0
- package/dist/src/formats/dot/exporter.d.ts.map +1 -0
- package/dist/src/formats/dot/exporter.js +836 -0
- package/dist/src/formats/dot/exporter.js.map +1 -0
- package/dist/src/formats/dot/importer.d.ts +102 -0
- package/dist/src/formats/dot/importer.d.ts.map +1 -0
- package/dist/src/formats/dot/importer.js +1291 -0
- package/dist/src/formats/dot/importer.js.map +1 -0
- package/dist/src/formats/dot/index.d.ts +7 -0
- package/dist/src/formats/dot/index.d.ts.map +1 -0
- package/dist/src/formats/dot/index.js +7 -0
- package/dist/src/formats/dot/index.js.map +1 -0
- package/dist/src/formats/dot/names.d.ts +29 -0
- package/dist/src/formats/dot/names.d.ts.map +1 -0
- package/dist/src/formats/dot/names.js +28 -0
- package/dist/src/formats/dot/names.js.map +1 -0
- package/dist/src/formats/dot/tokenizer.d.ts +114 -0
- package/dist/src/formats/dot/tokenizer.d.ts.map +1 -0
- package/dist/src/formats/dot/tokenizer.js +341 -0
- package/dist/src/formats/dot/tokenizer.js.map +1 -0
- package/dist/src/formats/gexf/exporter.d.ts +56 -0
- package/dist/src/formats/gexf/exporter.d.ts.map +1 -0
- package/dist/src/formats/gexf/exporter.js +1395 -0
- package/dist/src/formats/gexf/exporter.js.map +1 -0
- package/dist/src/formats/gexf/importer.d.ts +73 -0
- package/dist/src/formats/gexf/importer.d.ts.map +1 -0
- package/dist/src/formats/gexf/importer.js +1880 -0
- package/dist/src/formats/gexf/importer.js.map +1 -0
- package/dist/src/formats/gexf/index.d.ts +96 -0
- package/dist/src/formats/gexf/index.d.ts.map +1 -0
- package/dist/src/formats/gexf/index.js +97 -0
- package/dist/src/formats/gexf/index.js.map +1 -0
- package/dist/src/formats/gexf/schema.d.ts +135 -0
- package/dist/src/formats/gexf/schema.d.ts.map +1 -0
- package/dist/src/formats/gexf/schema.js +323 -0
- package/dist/src/formats/gexf/schema.js.map +1 -0
- package/dist/src/formats/gml/exporter.d.ts +69 -0
- package/dist/src/formats/gml/exporter.d.ts.map +1 -0
- package/dist/src/formats/gml/exporter.js +1093 -0
- package/dist/src/formats/gml/exporter.js.map +1 -0
- package/dist/src/formats/gml/importer.d.ts +66 -0
- package/dist/src/formats/gml/importer.d.ts.map +1 -0
- package/dist/src/formats/gml/importer.js +1331 -0
- package/dist/src/formats/gml/importer.js.map +1 -0
- package/dist/src/formats/gml/index.d.ts +85 -0
- package/dist/src/formats/gml/index.d.ts.map +1 -0
- package/dist/src/formats/gml/index.js +88 -0
- package/dist/src/formats/gml/index.js.map +1 -0
- package/dist/src/formats/gml/syntax.d.ts +186 -0
- package/dist/src/formats/gml/syntax.d.ts.map +1 -0
- package/dist/src/formats/gml/syntax.js +467 -0
- package/dist/src/formats/gml/syntax.js.map +1 -0
- package/dist/src/formats/graphml/constants.d.ts +169 -0
- package/dist/src/formats/graphml/constants.d.ts.map +1 -0
- package/dist/src/formats/graphml/constants.js +165 -0
- package/dist/src/formats/graphml/constants.js.map +1 -0
- package/dist/src/formats/graphml/exporter.d.ts +34 -0
- package/dist/src/formats/graphml/exporter.d.ts.map +1 -0
- package/dist/src/formats/graphml/exporter.js +1176 -0
- package/dist/src/formats/graphml/exporter.js.map +1 -0
- package/dist/src/formats/graphml/importer.d.ts +31 -0
- package/dist/src/formats/graphml/importer.d.ts.map +1 -0
- package/dist/src/formats/graphml/importer.js +1607 -0
- package/dist/src/formats/graphml/importer.js.map +1 -0
- package/dist/src/formats/graphml/index.d.ts +8 -0
- package/dist/src/formats/graphml/index.d.ts.map +1 -0
- package/dist/src/formats/graphml/index.js +8 -0
- package/dist/src/formats/graphml/index.js.map +1 -0
- package/dist/src/formats/graphml/tree.d.ts +72 -0
- package/dist/src/formats/graphml/tree.d.ts.map +1 -0
- package/dist/src/formats/graphml/tree.js +290 -0
- package/dist/src/formats/graphml/tree.js.map +1 -0
- package/dist/src/formats/json/dialect.d.ts +125 -0
- package/dist/src/formats/json/dialect.d.ts.map +1 -0
- package/dist/src/formats/json/dialect.js +262 -0
- package/dist/src/formats/json/dialect.js.map +1 -0
- package/dist/src/formats/json/exporter.d.ts +89 -0
- package/dist/src/formats/json/exporter.d.ts.map +1 -0
- package/dist/src/formats/json/exporter.js +1358 -0
- package/dist/src/formats/json/exporter.js.map +1 -0
- package/dist/src/formats/json/importer.d.ts +108 -0
- package/dist/src/formats/json/importer.d.ts.map +1 -0
- package/dist/src/formats/json/importer.js +1838 -0
- package/dist/src/formats/json/importer.js.map +1 -0
- package/dist/src/formats/json/index.d.ts +8 -0
- package/dist/src/formats/json/index.d.ts.map +1 -0
- package/dist/src/formats/json/index.js +8 -0
- package/dist/src/formats/json/index.js.map +1 -0
- package/dist/src/formats/neo4j/exporter.d.ts +68 -0
- package/dist/src/formats/neo4j/exporter.d.ts.map +1 -0
- package/dist/src/formats/neo4j/exporter.js +1055 -0
- package/dist/src/formats/neo4j/exporter.js.map +1 -0
- package/dist/src/formats/neo4j/header.d.ts +52 -0
- package/dist/src/formats/neo4j/header.d.ts.map +1 -0
- package/dist/src/formats/neo4j/header.js +131 -0
- package/dist/src/formats/neo4j/header.js.map +1 -0
- package/dist/src/formats/neo4j/importer.d.ts +73 -0
- package/dist/src/formats/neo4j/importer.d.ts.map +1 -0
- package/dist/src/formats/neo4j/importer.js +932 -0
- package/dist/src/formats/neo4j/importer.js.map +1 -0
- package/dist/src/formats/neo4j/index.d.ts +79 -0
- package/dist/src/formats/neo4j/index.d.ts.map +1 -0
- package/dist/src/formats/neo4j/index.js +83 -0
- package/dist/src/formats/neo4j/index.js.map +1 -0
- package/dist/src/formats/pajek/exporter.d.ts +58 -0
- package/dist/src/formats/pajek/exporter.d.ts.map +1 -0
- package/dist/src/formats/pajek/exporter.js +825 -0
- package/dist/src/formats/pajek/exporter.js.map +1 -0
- package/dist/src/formats/pajek/importer.d.ts +88 -0
- package/dist/src/formats/pajek/importer.d.ts.map +1 -0
- package/dist/src/formats/pajek/importer.js +1047 -0
- package/dist/src/formats/pajek/importer.js.map +1 -0
- package/dist/src/formats/pajek/index.d.ts +7 -0
- package/dist/src/formats/pajek/index.d.ts.map +1 -0
- package/dist/src/formats/pajek/index.js +7 -0
- package/dist/src/formats/pajek/index.js.map +1 -0
- package/dist/src/formats/pajek/syntax.d.ts +112 -0
- package/dist/src/formats/pajek/syntax.d.ts.map +1 -0
- package/dist/src/formats/pajek/syntax.js +269 -0
- package/dist/src/formats/pajek/syntax.js.map +1 -0
- package/dist/src/index.d.ts +35 -0
- package/dist/src/index.d.ts.map +1 -0
- package/dist/src/index.js +39 -0
- package/dist/src/index.js.map +1 -0
- package/dist/src/registry.d.ts +207 -0
- package/dist/src/registry.d.ts.map +1 -0
- package/dist/src/registry.js +481 -0
- package/dist/src/registry.js.map +1 -0
- package/dist/src/sniff.d.ts +104 -0
- package/dist/src/sniff.d.ts.map +1 -0
- package/dist/src/sniff.js +357 -0
- package/dist/src/sniff.js.map +1 -0
- package/dist/src/types.d.ts +238 -0
- package/dist/src/types.d.ts.map +1 -0
- package/dist/src/types.js +29 -0
- package/dist/src/types.js.map +1 -0
- package/dist/tsconfig.build.tsbuildinfo +1 -0
- package/package.json +122 -7
- package/src/children.ts +335 -0
- package/src/common/attributes.ts +520 -0
- package/src/common/codes.ts +153 -0
- package/src/common/declared-types.ts +374 -0
- package/src/common/direction.ts +518 -0
- package/src/common/escape.ts +231 -0
- package/src/common/export.ts +817 -0
- package/src/common/format.ts +111 -0
- package/src/common/ids.ts +176 -0
- package/src/common/input.ts +378 -0
- package/src/common/lists.ts +196 -0
- package/src/common/options.ts +377 -0
- package/src/common/report.ts +352 -0
- package/src/common/temporal.ts +302 -0
- package/src/common/text.ts +294 -0
- package/src/common/weights.ts +202 -0
- package/src/common/writer.ts +123 -0
- package/src/common/xml.ts +1053 -0
- package/src/formats/csv/exporter.ts +894 -0
- package/src/formats/csv/header.ts +172 -0
- package/src/formats/csv/importer.ts +1104 -0
- package/src/formats/csv/index.ts +88 -0
- package/src/formats/csv/records.ts +813 -0
- package/src/formats/csv/values.ts +224 -0
- package/src/formats/dot/exporter.ts +1014 -0
- package/src/formats/dot/importer.ts +1549 -0
- package/src/formats/dot/index.ts +7 -0
- package/src/formats/dot/names.ts +40 -0
- package/src/formats/dot/tokenizer.ts +384 -0
- package/src/formats/gexf/exporter.ts +1696 -0
- package/src/formats/gexf/importer.ts +2333 -0
- package/src/formats/gexf/index.ts +142 -0
- package/src/formats/gexf/schema.ts +361 -0
- package/src/formats/gml/exporter.ts +1404 -0
- package/src/formats/gml/importer.ts +1591 -0
- package/src/formats/gml/index.ts +128 -0
- package/src/formats/gml/syntax.ts +545 -0
- package/src/formats/graphml/constants.ts +225 -0
- package/src/formats/graphml/exporter.ts +1458 -0
- package/src/formats/graphml/importer.ts +2027 -0
- package/src/formats/graphml/index.ts +8 -0
- package/src/formats/graphml/tree.ts +318 -0
- package/src/formats/json/dialect.ts +317 -0
- package/src/formats/json/exporter.ts +1616 -0
- package/src/formats/json/importer.ts +2271 -0
- package/src/formats/json/index.ts +8 -0
- package/src/formats/neo4j/exporter.ts +1287 -0
- package/src/formats/neo4j/header.ts +156 -0
- package/src/formats/neo4j/importer.ts +1220 -0
- package/src/formats/neo4j/index.ts +116 -0
- package/src/formats/pajek/exporter.ts +1000 -0
- package/src/formats/pajek/importer.ts +1311 -0
- package/src/formats/pajek/index.ts +7 -0
- package/src/formats/pajek/syntax.ts +307 -0
- package/src/index.ts +244 -0
- package/src/registry.ts +617 -0
- package/src/sniff.ts +397 -0
- package/src/types.ts +262 -0
|
@@ -0,0 +1,938 @@
|
|
|
1
|
+
import { GraphFormatError } from "@graphty/graph-format";
|
|
2
|
+
import { aF as isNameChar, a6 as XML_ILLEGAL_CHAR_CODE } from "./writer-DxSKC7TL.js";
|
|
3
|
+
class XmlSyntaxError extends Error {
|
|
4
|
+
/**
|
|
5
|
+
* Create the error.
|
|
6
|
+
* @param message - a plain-ASCII message
|
|
7
|
+
* @param line - the 1-based line
|
|
8
|
+
*/
|
|
9
|
+
constructor(message, line) {
|
|
10
|
+
super(message);
|
|
11
|
+
this.name = "XmlSyntaxError";
|
|
12
|
+
this.line = line;
|
|
13
|
+
}
|
|
14
|
+
}
|
|
15
|
+
const NAMED_ENTITIES = {
|
|
16
|
+
lt: "<",
|
|
17
|
+
gt: ">",
|
|
18
|
+
amp: "&",
|
|
19
|
+
quot: '"',
|
|
20
|
+
apos: "'"
|
|
21
|
+
};
|
|
22
|
+
const LT = 60;
|
|
23
|
+
const GT = 62;
|
|
24
|
+
const SLASH = 47;
|
|
25
|
+
const EQUALS = 61;
|
|
26
|
+
const QUOTE = 34;
|
|
27
|
+
const APOS = 39;
|
|
28
|
+
const OPEN_BRACKET = 91;
|
|
29
|
+
const CLOSE_BRACKET = 93;
|
|
30
|
+
const BANG = 33;
|
|
31
|
+
const DASH = 45;
|
|
32
|
+
const MAX_ENTITY_LENGTH = 16;
|
|
33
|
+
const MIN_DECIDABLE_MARKUP = 9;
|
|
34
|
+
function isSpace(c) {
|
|
35
|
+
return c === 32 || c === 9 || c === 10 || c === 13;
|
|
36
|
+
}
|
|
37
|
+
function isNameStart(cp) {
|
|
38
|
+
if (!isNameChar(cp)) {
|
|
39
|
+
return false;
|
|
40
|
+
}
|
|
41
|
+
if (cp >= 48 && cp <= 57 || cp === 45 || cp === 46 || cp === 183) {
|
|
42
|
+
return false;
|
|
43
|
+
}
|
|
44
|
+
return !(cp >= 768 && cp <= 879) && cp !== 8255 && cp !== 8256;
|
|
45
|
+
}
|
|
46
|
+
const ASCII_NAME_START = new Uint8Array(128);
|
|
47
|
+
const ASCII_NAME_CHAR = new Uint8Array(128);
|
|
48
|
+
for (let c = 0; c < 128; c++) {
|
|
49
|
+
ASCII_NAME_START[c] = isNameStart(c) ? 1 : 0;
|
|
50
|
+
ASCII_NAME_CHAR[c] = isNameChar(c) ? 1 : 0;
|
|
51
|
+
}
|
|
52
|
+
const ILLEGAL_CHAR = new RegExp(
|
|
53
|
+
`[${String.fromCharCode(0)}-${String.fromCharCode(8)}${String.fromCharCode(11)}${String.fromCharCode(12)}${String.fromCharCode(14)}-${String.fromCharCode(31)}\\uFFFE\\uFFFF\\uD800-\\uDFFF]`,
|
|
54
|
+
"u"
|
|
55
|
+
);
|
|
56
|
+
function hasIllegalXmlChar(text) {
|
|
57
|
+
return ILLEGAL_CHAR.test(text);
|
|
58
|
+
}
|
|
59
|
+
function xmlIllegalTextNotes(snapshot) {
|
|
60
|
+
const notes = [];
|
|
61
|
+
let ids = 0;
|
|
62
|
+
for (let i = 0; i < snapshot.nodeCount; i++) {
|
|
63
|
+
const id = snapshot.ids.idOf(i);
|
|
64
|
+
if (typeof id === "string" && hasIllegalXmlChar(id)) {
|
|
65
|
+
ids++;
|
|
66
|
+
}
|
|
67
|
+
}
|
|
68
|
+
if (ids > 0) {
|
|
69
|
+
notes.push(
|
|
70
|
+
Object.freeze({
|
|
71
|
+
code: XML_ILLEGAL_CHAR_CODE,
|
|
72
|
+
message: `${ids} node id(s) hold a character XML 1.0 cannot carry; export() will throw`,
|
|
73
|
+
column: null,
|
|
74
|
+
count: ids
|
|
75
|
+
})
|
|
76
|
+
);
|
|
77
|
+
}
|
|
78
|
+
for (const [domain, table] of [
|
|
79
|
+
["node", snapshot.nodes],
|
|
80
|
+
["edge", snapshot.edges],
|
|
81
|
+
["graph", snapshot.graph]
|
|
82
|
+
]) {
|
|
83
|
+
for (const column of table) {
|
|
84
|
+
const bad = countIllegalRows(column);
|
|
85
|
+
if (bad > 0) {
|
|
86
|
+
notes.push(
|
|
87
|
+
Object.freeze({
|
|
88
|
+
code: XML_ILLEGAL_CHAR_CODE,
|
|
89
|
+
message: `${domain} column "${column.meta.name}": ${bad} value(s) hold a character XML 1.0 cannot carry; export() will throw`,
|
|
90
|
+
column: column.meta.name,
|
|
91
|
+
count: bad
|
|
92
|
+
})
|
|
93
|
+
);
|
|
94
|
+
}
|
|
95
|
+
}
|
|
96
|
+
}
|
|
97
|
+
return notes;
|
|
98
|
+
}
|
|
99
|
+
function countIllegalRows(column) {
|
|
100
|
+
let bad = 0;
|
|
101
|
+
if (column.dtype === "string" || column.dtype === "dict") {
|
|
102
|
+
for (let r = 0; r < column.length; r++) {
|
|
103
|
+
if (column.isSet(r) && hasIllegalXmlChar(column.value(r))) {
|
|
104
|
+
bad++;
|
|
105
|
+
}
|
|
106
|
+
}
|
|
107
|
+
} else if (column.dtype === "list" && (column.meta.itemDtype === "string" || column.meta.itemDtype === "dict")) {
|
|
108
|
+
for (let r = 0; r < column.length; r++) {
|
|
109
|
+
if (column.isSet(r) && column.sliceOf(r).some((item) => typeof item === "string" && hasIllegalXmlChar(item))) {
|
|
110
|
+
bad++;
|
|
111
|
+
}
|
|
112
|
+
}
|
|
113
|
+
}
|
|
114
|
+
return bad;
|
|
115
|
+
}
|
|
116
|
+
function isXmlName(text) {
|
|
117
|
+
if (text.length === 0) {
|
|
118
|
+
return false;
|
|
119
|
+
}
|
|
120
|
+
let first = true;
|
|
121
|
+
for (const ch of text) {
|
|
122
|
+
const cp = ch.codePointAt(0);
|
|
123
|
+
if (cp === void 0 || (first ? !isNameStart(cp) : !isNameChar(cp))) {
|
|
124
|
+
return false;
|
|
125
|
+
}
|
|
126
|
+
first = false;
|
|
127
|
+
}
|
|
128
|
+
return true;
|
|
129
|
+
}
|
|
130
|
+
function isXmlChar(cp) {
|
|
131
|
+
return cp === 9 || cp === 10 || cp === 13 || cp >= 32 && cp <= 55295 || cp >= 57344 && cp <= 65533 || cp >= 65536 && cp <= 1114111;
|
|
132
|
+
}
|
|
133
|
+
function decodeEntities(raw, line) {
|
|
134
|
+
let amp = raw.indexOf("&");
|
|
135
|
+
if (amp < 0) {
|
|
136
|
+
return raw;
|
|
137
|
+
}
|
|
138
|
+
let out = "";
|
|
139
|
+
let start = 0;
|
|
140
|
+
while (amp >= 0) {
|
|
141
|
+
out += raw.slice(start, amp);
|
|
142
|
+
const semi = raw.indexOf(";", amp + 1);
|
|
143
|
+
if (semi < 0) {
|
|
144
|
+
throw new XmlSyntaxError("unterminated entity reference", line);
|
|
145
|
+
}
|
|
146
|
+
const name = raw.slice(amp + 1, semi);
|
|
147
|
+
if (name.startsWith("#")) {
|
|
148
|
+
const hex = name.startsWith("#x") || name.startsWith("#X");
|
|
149
|
+
const digits = name.slice(hex ? 2 : 1);
|
|
150
|
+
const ok = hex ? /^[0-9a-fA-F]{1,6}$/.test(digits) : /^[0-9]{1,7}$/.test(digits);
|
|
151
|
+
const cp = ok ? Number.parseInt(digits, hex ? 16 : 10) : -1;
|
|
152
|
+
if (cp < 0 || !isXmlChar(cp)) {
|
|
153
|
+
throw new XmlSyntaxError(`invalid character reference &${name};`, line);
|
|
154
|
+
}
|
|
155
|
+
out += String.fromCodePoint(cp);
|
|
156
|
+
} else {
|
|
157
|
+
const value = NAMED_ENTITIES[name];
|
|
158
|
+
if (value === void 0) {
|
|
159
|
+
throw new XmlSyntaxError(`unknown entity &${name};`, line);
|
|
160
|
+
}
|
|
161
|
+
out += value;
|
|
162
|
+
}
|
|
163
|
+
start = semi + 1;
|
|
164
|
+
amp = raw.indexOf("&", start);
|
|
165
|
+
}
|
|
166
|
+
return out + raw.slice(start);
|
|
167
|
+
}
|
|
168
|
+
function isWhitespace(text) {
|
|
169
|
+
for (let i = 0; i < text.length; i++) {
|
|
170
|
+
if (!isSpace(text.charCodeAt(i))) {
|
|
171
|
+
return false;
|
|
172
|
+
}
|
|
173
|
+
}
|
|
174
|
+
return true;
|
|
175
|
+
}
|
|
176
|
+
function localName(name) {
|
|
177
|
+
const colon = name.lastIndexOf(":");
|
|
178
|
+
return colon < 0 ? name : name.slice(colon + 1);
|
|
179
|
+
}
|
|
180
|
+
async function tokenizeXml(chunks, handler) {
|
|
181
|
+
const tokenizer = new XmlTokenizer(handler);
|
|
182
|
+
for await (const chunk of chunks) {
|
|
183
|
+
tokenizer.push(chunk);
|
|
184
|
+
}
|
|
185
|
+
tokenizer.finish();
|
|
186
|
+
}
|
|
187
|
+
const TERMINATORS = {
|
|
188
|
+
comment: "-->",
|
|
189
|
+
cdata: "]]>",
|
|
190
|
+
pi: "?>",
|
|
191
|
+
endtag: ">"
|
|
192
|
+
};
|
|
193
|
+
class XmlTokenizer {
|
|
194
|
+
/**
|
|
195
|
+
* Create a tokenizer.
|
|
196
|
+
* @param handler - the event sink
|
|
197
|
+
*/
|
|
198
|
+
constructor(handler) {
|
|
199
|
+
this.buffer = "";
|
|
200
|
+
this.pending = null;
|
|
201
|
+
this.line = 1;
|
|
202
|
+
this.nextBreak = -2;
|
|
203
|
+
this.pendingCr = false;
|
|
204
|
+
this.final = false;
|
|
205
|
+
this.text = "";
|
|
206
|
+
this.textLine = 1;
|
|
207
|
+
this.stack = [];
|
|
208
|
+
this.rootSeen = false;
|
|
209
|
+
this.rootClosed = false;
|
|
210
|
+
this.handler = handler;
|
|
211
|
+
}
|
|
212
|
+
/**
|
|
213
|
+
* Feed one chunk and emit every complete token in it.
|
|
214
|
+
* @param chunk - the text
|
|
215
|
+
*/
|
|
216
|
+
push(chunk) {
|
|
217
|
+
let text = chunk;
|
|
218
|
+
if (this.pendingCr) {
|
|
219
|
+
text = `\r${text}`;
|
|
220
|
+
this.pendingCr = false;
|
|
221
|
+
}
|
|
222
|
+
if (text.endsWith("\r")) {
|
|
223
|
+
this.pendingCr = true;
|
|
224
|
+
text = text.slice(0, -1);
|
|
225
|
+
}
|
|
226
|
+
if (text.includes("\r")) {
|
|
227
|
+
text = text.replace(/\r\n?/g, "\n");
|
|
228
|
+
}
|
|
229
|
+
this.feed(text);
|
|
230
|
+
}
|
|
231
|
+
/** Signal the end of the input: flush the last text run and check well-formedness. */
|
|
232
|
+
finish() {
|
|
233
|
+
this.final = true;
|
|
234
|
+
if (this.pendingCr) {
|
|
235
|
+
this.pendingCr = false;
|
|
236
|
+
this.feed("\n");
|
|
237
|
+
} else {
|
|
238
|
+
this.feed("");
|
|
239
|
+
}
|
|
240
|
+
if (this.pending !== null || this.buffer.length > 0) {
|
|
241
|
+
throw new XmlSyntaxError("unexpected end of input inside markup", this.line);
|
|
242
|
+
}
|
|
243
|
+
this.flushText();
|
|
244
|
+
if (this.stack.length > 0) {
|
|
245
|
+
throw new XmlSyntaxError(`unclosed element <${this.stack[this.stack.length - 1]}>`, this.line);
|
|
246
|
+
}
|
|
247
|
+
if (!this.rootSeen) {
|
|
248
|
+
throw new XmlSyntaxError("no root element", this.line);
|
|
249
|
+
}
|
|
250
|
+
}
|
|
251
|
+
/**
|
|
252
|
+
* Append normalised text: continue a pending token's terminator scan over the new text alone,
|
|
253
|
+
* and once the token is complete (or when none is pending) scan the buffer for tokens.
|
|
254
|
+
* @param text - the text, line breaks normalised
|
|
255
|
+
*/
|
|
256
|
+
feed(text) {
|
|
257
|
+
const { pending } = this;
|
|
258
|
+
if (pending === null) {
|
|
259
|
+
this.buffer = this.buffer.length === 0 ? text : this.buffer + text;
|
|
260
|
+
this.nextBreak = -2;
|
|
261
|
+
this.scan();
|
|
262
|
+
return;
|
|
263
|
+
}
|
|
264
|
+
if (pending.kind === "decl") {
|
|
265
|
+
this.pending = null;
|
|
266
|
+
this.buffer = pending.pieces.join("") + text;
|
|
267
|
+
this.nextBreak = -2;
|
|
268
|
+
this.scan();
|
|
269
|
+
return;
|
|
270
|
+
}
|
|
271
|
+
const end = this.scanPending(pending, text);
|
|
272
|
+
if (end < 0) {
|
|
273
|
+
if (text.length > 0) {
|
|
274
|
+
pending.pieces.push(text);
|
|
275
|
+
}
|
|
276
|
+
if (this.final) {
|
|
277
|
+
throw new XmlSyntaxError("unexpected end of input inside markup", this.line);
|
|
278
|
+
}
|
|
279
|
+
return;
|
|
280
|
+
}
|
|
281
|
+
this.pending = null;
|
|
282
|
+
pending.pieces.push(text.slice(0, end));
|
|
283
|
+
this.buffer = pending.pieces.join("") + text.slice(end);
|
|
284
|
+
this.nextBreak = -2;
|
|
285
|
+
this.scan();
|
|
286
|
+
}
|
|
287
|
+
/**
|
|
288
|
+
* Continue the terminator scan of a pending token over new text.
|
|
289
|
+
* @param pending - the pending token
|
|
290
|
+
* @param text - the new text
|
|
291
|
+
* @returns the index in `text` one past the token's end, or -1 when it does not end there
|
|
292
|
+
*/
|
|
293
|
+
scanPending(pending, text) {
|
|
294
|
+
switch (pending.kind) {
|
|
295
|
+
case "starttag":
|
|
296
|
+
return this.scanStartTag(pending, text, 0);
|
|
297
|
+
case "doctype":
|
|
298
|
+
return this.scanDoctype(pending, text, 0);
|
|
299
|
+
case "decl":
|
|
300
|
+
return -1;
|
|
301
|
+
default: {
|
|
302
|
+
const terminator = TERMINATORS[pending.kind] ?? ">";
|
|
303
|
+
const probe = pending.tail + text;
|
|
304
|
+
const at = probe.indexOf(terminator);
|
|
305
|
+
if (at < 0) {
|
|
306
|
+
pending.tail = probe.slice(Math.max(0, probe.length - (terminator.length - 1)));
|
|
307
|
+
return -1;
|
|
308
|
+
}
|
|
309
|
+
return at + terminator.length - pending.tail.length;
|
|
310
|
+
}
|
|
311
|
+
}
|
|
312
|
+
}
|
|
313
|
+
/**
|
|
314
|
+
* Scan text for the `>` that ends a start tag, skipping quoted attribute values (the quote
|
|
315
|
+
* state survives between calls).
|
|
316
|
+
* @param pending - the pending token
|
|
317
|
+
* @param text - the text
|
|
318
|
+
* @param from - where to start
|
|
319
|
+
* @returns the index one past the `>`, or -1
|
|
320
|
+
*/
|
|
321
|
+
scanStartTag(pending, text, from) {
|
|
322
|
+
let { quote } = pending;
|
|
323
|
+
const n = text.length;
|
|
324
|
+
let i = from;
|
|
325
|
+
while (i < n) {
|
|
326
|
+
if (quote !== 0) {
|
|
327
|
+
const close = text.indexOf(String.fromCharCode(quote), i);
|
|
328
|
+
if (close < 0) {
|
|
329
|
+
break;
|
|
330
|
+
}
|
|
331
|
+
quote = 0;
|
|
332
|
+
i = close + 1;
|
|
333
|
+
continue;
|
|
334
|
+
}
|
|
335
|
+
const c = text.charCodeAt(i);
|
|
336
|
+
if (c === QUOTE || c === APOS) {
|
|
337
|
+
quote = c;
|
|
338
|
+
} else if (c === GT) {
|
|
339
|
+
pending.quote = 0;
|
|
340
|
+
return i + 1;
|
|
341
|
+
}
|
|
342
|
+
i++;
|
|
343
|
+
}
|
|
344
|
+
pending.quote = quote;
|
|
345
|
+
return -1;
|
|
346
|
+
}
|
|
347
|
+
/**
|
|
348
|
+
* Scan text for the `>` that ends a DOCTYPE declaration, skipping a bracketed internal subset,
|
|
349
|
+
* quoted literals and comments inside the subset. A character state machine, so the state
|
|
350
|
+
* (`depth`, `quote`, `comment`, and `run`, the progress through a `<!--` or the dashes before
|
|
351
|
+
* a `-->`) survives between calls and nothing is re-read.
|
|
352
|
+
* @param pending - the pending token
|
|
353
|
+
* @param text - the text
|
|
354
|
+
* @param from - where to start
|
|
355
|
+
* @returns the index one past the `>`, or -1
|
|
356
|
+
*/
|
|
357
|
+
scanDoctype(pending, text, from) {
|
|
358
|
+
let { quote, depth, comment, run } = pending;
|
|
359
|
+
for (let i = from; i < text.length; i++) {
|
|
360
|
+
const c = text.charCodeAt(i);
|
|
361
|
+
if (comment) {
|
|
362
|
+
if (c === DASH) {
|
|
363
|
+
run++;
|
|
364
|
+
} else if (c === GT && run >= 2) {
|
|
365
|
+
comment = false;
|
|
366
|
+
run = 0;
|
|
367
|
+
} else {
|
|
368
|
+
run = 0;
|
|
369
|
+
}
|
|
370
|
+
continue;
|
|
371
|
+
}
|
|
372
|
+
if (quote !== 0) {
|
|
373
|
+
if (c === quote) {
|
|
374
|
+
quote = 0;
|
|
375
|
+
}
|
|
376
|
+
continue;
|
|
377
|
+
}
|
|
378
|
+
if (depth > 0) {
|
|
379
|
+
if (run === 1 && c === BANG) {
|
|
380
|
+
run = 2;
|
|
381
|
+
continue;
|
|
382
|
+
}
|
|
383
|
+
if (run === 2 && c === DASH) {
|
|
384
|
+
run = 3;
|
|
385
|
+
continue;
|
|
386
|
+
}
|
|
387
|
+
if (run === 3 && c === DASH) {
|
|
388
|
+
comment = true;
|
|
389
|
+
run = 0;
|
|
390
|
+
continue;
|
|
391
|
+
}
|
|
392
|
+
run = 0;
|
|
393
|
+
if (c === LT) {
|
|
394
|
+
run = 1;
|
|
395
|
+
continue;
|
|
396
|
+
}
|
|
397
|
+
}
|
|
398
|
+
if (c === QUOTE || c === APOS) {
|
|
399
|
+
quote = c;
|
|
400
|
+
} else if (c === OPEN_BRACKET) {
|
|
401
|
+
depth++;
|
|
402
|
+
} else if (c === CLOSE_BRACKET) {
|
|
403
|
+
depth--;
|
|
404
|
+
} else if (c === GT && depth <= 0) {
|
|
405
|
+
pending.quote = 0;
|
|
406
|
+
pending.depth = 0;
|
|
407
|
+
pending.comment = false;
|
|
408
|
+
pending.run = 0;
|
|
409
|
+
return i + 1;
|
|
410
|
+
}
|
|
411
|
+
}
|
|
412
|
+
pending.quote = quote;
|
|
413
|
+
pending.depth = depth;
|
|
414
|
+
pending.comment = comment;
|
|
415
|
+
pending.run = run;
|
|
416
|
+
return -1;
|
|
417
|
+
}
|
|
418
|
+
/**
|
|
419
|
+
* Consume every complete token at the front of the buffer; an incomplete token at its end
|
|
420
|
+
* becomes the pending token (its text moved out of the buffer), a possibly split entity
|
|
421
|
+
* reference is held back in the buffer.
|
|
422
|
+
*/
|
|
423
|
+
scan() {
|
|
424
|
+
const { buffer } = this;
|
|
425
|
+
const { length } = buffer;
|
|
426
|
+
let pos = 0;
|
|
427
|
+
let incomplete = false;
|
|
428
|
+
for (; ; ) {
|
|
429
|
+
const lt = buffer.indexOf("<", pos);
|
|
430
|
+
if (lt < 0) {
|
|
431
|
+
let end = length;
|
|
432
|
+
if (!this.final) {
|
|
433
|
+
const amp = buffer.lastIndexOf("&");
|
|
434
|
+
if (amp >= pos && amp >= length - MAX_ENTITY_LENGTH && buffer.indexOf(";", amp) < 0) {
|
|
435
|
+
end = amp;
|
|
436
|
+
}
|
|
437
|
+
}
|
|
438
|
+
this.takeText(pos, end);
|
|
439
|
+
pos = end;
|
|
440
|
+
break;
|
|
441
|
+
}
|
|
442
|
+
if (lt > pos) {
|
|
443
|
+
this.takeText(pos, lt);
|
|
444
|
+
pos = lt;
|
|
445
|
+
}
|
|
446
|
+
const next = this.consumeMarkup(pos);
|
|
447
|
+
if (next < 0) {
|
|
448
|
+
incomplete = true;
|
|
449
|
+
break;
|
|
450
|
+
}
|
|
451
|
+
pos = next;
|
|
452
|
+
}
|
|
453
|
+
if (incomplete) {
|
|
454
|
+
this.pending = this.startPending(buffer, pos);
|
|
455
|
+
this.buffer = "";
|
|
456
|
+
} else {
|
|
457
|
+
this.buffer = pos === 0 ? buffer : buffer.slice(pos);
|
|
458
|
+
}
|
|
459
|
+
this.nextBreak = -2;
|
|
460
|
+
}
|
|
461
|
+
/**
|
|
462
|
+
* Turn the incomplete markup at `pos` into a pending token, running the terminator scan over
|
|
463
|
+
* the part already in the buffer so later chunks are scanned alone.
|
|
464
|
+
* @param buffer - the buffer
|
|
465
|
+
* @param pos - the index of the `<`
|
|
466
|
+
* @returns the pending token
|
|
467
|
+
*/
|
|
468
|
+
startPending(buffer, pos) {
|
|
469
|
+
const piece = pos === 0 ? buffer : buffer.slice(pos);
|
|
470
|
+
const pending = {
|
|
471
|
+
kind: "starttag",
|
|
472
|
+
pieces: [piece],
|
|
473
|
+
tail: "",
|
|
474
|
+
quote: 0,
|
|
475
|
+
depth: 0,
|
|
476
|
+
comment: false,
|
|
477
|
+
run: 0
|
|
478
|
+
};
|
|
479
|
+
if (piece.length < MIN_DECIDABLE_MARKUP) {
|
|
480
|
+
pending.kind = "decl";
|
|
481
|
+
return pending;
|
|
482
|
+
}
|
|
483
|
+
if (piece.startsWith("<!--")) {
|
|
484
|
+
pending.kind = "comment";
|
|
485
|
+
} else if (piece.startsWith("<![CDATA[")) {
|
|
486
|
+
pending.kind = "cdata";
|
|
487
|
+
} else if (piece.startsWith("<!DOCTYPE")) {
|
|
488
|
+
pending.kind = "doctype";
|
|
489
|
+
} else if (piece.startsWith("<!")) {
|
|
490
|
+
pending.kind = "decl";
|
|
491
|
+
return pending;
|
|
492
|
+
} else if (piece.startsWith("<?")) {
|
|
493
|
+
pending.kind = "pi";
|
|
494
|
+
} else if (piece.startsWith("</")) {
|
|
495
|
+
pending.kind = "endtag";
|
|
496
|
+
}
|
|
497
|
+
switch (pending.kind) {
|
|
498
|
+
case "starttag":
|
|
499
|
+
this.scanStartTag(pending, piece, 1);
|
|
500
|
+
break;
|
|
501
|
+
case "doctype":
|
|
502
|
+
this.scanDoctype(pending, piece, 9);
|
|
503
|
+
break;
|
|
504
|
+
default: {
|
|
505
|
+
const terminator = TERMINATORS[pending.kind] ?? ">";
|
|
506
|
+
pending.tail = piece.slice(Math.max(0, piece.length - (terminator.length - 1)));
|
|
507
|
+
break;
|
|
508
|
+
}
|
|
509
|
+
}
|
|
510
|
+
return pending;
|
|
511
|
+
}
|
|
512
|
+
/**
|
|
513
|
+
* Consume the markup starting at `pos` (a `<`).
|
|
514
|
+
* @param pos - the index of the `<`
|
|
515
|
+
* @returns the index after the markup, or -1 when the buffer ends before the markup does
|
|
516
|
+
*/
|
|
517
|
+
consumeMarkup(pos) {
|
|
518
|
+
const { buffer } = this;
|
|
519
|
+
if (buffer.startsWith("<!", pos)) {
|
|
520
|
+
if (buffer.length - pos < MIN_DECIDABLE_MARKUP && !this.final) {
|
|
521
|
+
return -1;
|
|
522
|
+
}
|
|
523
|
+
if (buffer.startsWith("<!--", pos)) {
|
|
524
|
+
const end2 = buffer.indexOf("-->", pos + 4);
|
|
525
|
+
if (end2 < 0) {
|
|
526
|
+
return -1;
|
|
527
|
+
}
|
|
528
|
+
this.advanceLine(pos, end2 + 3);
|
|
529
|
+
return end2 + 3;
|
|
530
|
+
}
|
|
531
|
+
if (buffer.startsWith("<![CDATA[", pos)) {
|
|
532
|
+
const end2 = buffer.indexOf("]]>", pos + 9);
|
|
533
|
+
if (end2 < 0) {
|
|
534
|
+
return -1;
|
|
535
|
+
}
|
|
536
|
+
if (this.text.length === 0) {
|
|
537
|
+
this.textLine = this.line;
|
|
538
|
+
}
|
|
539
|
+
const cdata = buffer.slice(pos + 9, end2);
|
|
540
|
+
if (hasIllegalXmlChar(cdata)) {
|
|
541
|
+
throw new XmlSyntaxError("a character XML 1.0 forbids appears in a CDATA section", this.line);
|
|
542
|
+
}
|
|
543
|
+
this.text += cdata;
|
|
544
|
+
this.advanceLine(pos, end2 + 3);
|
|
545
|
+
return end2 + 3;
|
|
546
|
+
}
|
|
547
|
+
if (!buffer.startsWith("<!DOCTYPE", pos)) {
|
|
548
|
+
throw new XmlSyntaxError("unexpected markup declaration", this.line);
|
|
549
|
+
}
|
|
550
|
+
const state = {
|
|
551
|
+
kind: "doctype",
|
|
552
|
+
pieces: [],
|
|
553
|
+
tail: "",
|
|
554
|
+
quote: 0,
|
|
555
|
+
depth: 0,
|
|
556
|
+
comment: false,
|
|
557
|
+
run: 0
|
|
558
|
+
};
|
|
559
|
+
const end = this.scanDoctype(state, buffer, pos + 9);
|
|
560
|
+
if (end < 0) {
|
|
561
|
+
return -1;
|
|
562
|
+
}
|
|
563
|
+
this.advanceLine(pos, end);
|
|
564
|
+
return end;
|
|
565
|
+
}
|
|
566
|
+
if (buffer.startsWith("<?", pos)) {
|
|
567
|
+
const end = buffer.indexOf("?>", pos + 2);
|
|
568
|
+
if (end < 0) {
|
|
569
|
+
return -1;
|
|
570
|
+
}
|
|
571
|
+
this.advanceLine(pos, end + 2);
|
|
572
|
+
return end + 2;
|
|
573
|
+
}
|
|
574
|
+
if (buffer.startsWith("</", pos)) {
|
|
575
|
+
const end = buffer.indexOf(">", pos + 2);
|
|
576
|
+
if (end < 0) {
|
|
577
|
+
return -1;
|
|
578
|
+
}
|
|
579
|
+
const name = buffer.slice(pos + 2, end).trim();
|
|
580
|
+
if (!isXmlName(name)) {
|
|
581
|
+
throw new XmlSyntaxError(`malformed end tag </${name}>`, this.line);
|
|
582
|
+
}
|
|
583
|
+
this.flushText();
|
|
584
|
+
this.endElement(name);
|
|
585
|
+
this.advanceLine(pos, end + 1);
|
|
586
|
+
return end + 1;
|
|
587
|
+
}
|
|
588
|
+
const tag = this.parseStartTag(pos);
|
|
589
|
+
if (tag === null) {
|
|
590
|
+
return -1;
|
|
591
|
+
}
|
|
592
|
+
this.flushText();
|
|
593
|
+
this.startElement(tag.name, tag.attrs);
|
|
594
|
+
if (tag.selfClosing) {
|
|
595
|
+
this.endElement(tag.name);
|
|
596
|
+
}
|
|
597
|
+
this.advanceLine(pos, tag.end);
|
|
598
|
+
return tag.end;
|
|
599
|
+
}
|
|
600
|
+
/**
|
|
601
|
+
* Parse a start tag at `pos`.
|
|
602
|
+
* @param pos - the index of the `<`
|
|
603
|
+
* @returns the tag, or null when the buffer ends inside it
|
|
604
|
+
*/
|
|
605
|
+
parseStartTag(pos) {
|
|
606
|
+
const { buffer } = this;
|
|
607
|
+
const { length } = buffer;
|
|
608
|
+
let i = pos + 1;
|
|
609
|
+
const nameEnd = this.readName(i);
|
|
610
|
+
if (nameEnd < 0) {
|
|
611
|
+
return null;
|
|
612
|
+
}
|
|
613
|
+
if (nameEnd === i) {
|
|
614
|
+
throw new XmlSyntaxError("expected an element name after <", this.line);
|
|
615
|
+
}
|
|
616
|
+
const name = buffer.slice(i, nameEnd);
|
|
617
|
+
i = nameEnd;
|
|
618
|
+
const attrs = /* @__PURE__ */ new Map();
|
|
619
|
+
for (; ; ) {
|
|
620
|
+
while (i < length && isSpace(buffer.charCodeAt(i))) {
|
|
621
|
+
i++;
|
|
622
|
+
}
|
|
623
|
+
if (i >= length) {
|
|
624
|
+
return null;
|
|
625
|
+
}
|
|
626
|
+
const c = buffer.charCodeAt(i);
|
|
627
|
+
if (c === GT) {
|
|
628
|
+
return { name, attrs, selfClosing: false, end: i + 1 };
|
|
629
|
+
}
|
|
630
|
+
if (c === SLASH) {
|
|
631
|
+
if (i + 1 >= length) {
|
|
632
|
+
return null;
|
|
633
|
+
}
|
|
634
|
+
if (buffer.charCodeAt(i + 1) !== GT) {
|
|
635
|
+
throw new XmlSyntaxError(`unexpected "/" in <${name}>`, this.line);
|
|
636
|
+
}
|
|
637
|
+
return { name, attrs, selfClosing: true, end: i + 2 };
|
|
638
|
+
}
|
|
639
|
+
const attrEnd = this.readName(i);
|
|
640
|
+
if (attrEnd < 0) {
|
|
641
|
+
return null;
|
|
642
|
+
}
|
|
643
|
+
if (attrEnd === i) {
|
|
644
|
+
throw new XmlSyntaxError(`malformed attribute in <${name}>`, this.line);
|
|
645
|
+
}
|
|
646
|
+
const attrName = buffer.slice(i, attrEnd);
|
|
647
|
+
i = attrEnd;
|
|
648
|
+
while (i < length && isSpace(buffer.charCodeAt(i))) {
|
|
649
|
+
i++;
|
|
650
|
+
}
|
|
651
|
+
if (i >= length) {
|
|
652
|
+
return null;
|
|
653
|
+
}
|
|
654
|
+
if (buffer.charCodeAt(i) !== EQUALS) {
|
|
655
|
+
throw new XmlSyntaxError(`attribute ${attrName} of <${name}> has no value`, this.line);
|
|
656
|
+
}
|
|
657
|
+
i++;
|
|
658
|
+
while (i < length && isSpace(buffer.charCodeAt(i))) {
|
|
659
|
+
i++;
|
|
660
|
+
}
|
|
661
|
+
if (i >= length) {
|
|
662
|
+
return null;
|
|
663
|
+
}
|
|
664
|
+
const quote = buffer.charCodeAt(i);
|
|
665
|
+
if (quote !== QUOTE && quote !== APOS) {
|
|
666
|
+
throw new XmlSyntaxError(`attribute ${attrName} of <${name}> is not quoted`, this.line);
|
|
667
|
+
}
|
|
668
|
+
const close = buffer.indexOf(quote === QUOTE ? '"' : "'", i + 1);
|
|
669
|
+
if (close < 0) {
|
|
670
|
+
return null;
|
|
671
|
+
}
|
|
672
|
+
if (attrs.has(attrName)) {
|
|
673
|
+
throw new XmlSyntaxError(`duplicate attribute ${attrName} in <${name}>`, this.line);
|
|
674
|
+
}
|
|
675
|
+
const raw = buffer.slice(i + 1, close);
|
|
676
|
+
if (hasIllegalXmlChar(raw)) {
|
|
677
|
+
throw new XmlSyntaxError(
|
|
678
|
+
`a character XML 1.0 forbids appears in attribute ${attrName} of <${name}>`,
|
|
679
|
+
this.line
|
|
680
|
+
);
|
|
681
|
+
}
|
|
682
|
+
attrs.set(attrName, decodeEntities(normalizeAttributeValue(raw), this.line));
|
|
683
|
+
i = close + 1;
|
|
684
|
+
}
|
|
685
|
+
}
|
|
686
|
+
/**
|
|
687
|
+
* The end of the XML Name starting at `i` (an ASCII table for the common case, the code point
|
|
688
|
+
* classes beyond it).
|
|
689
|
+
* @param i - the start index
|
|
690
|
+
* @returns the index after the name; `i` when no name starts there; -1 when the name may
|
|
691
|
+
* continue past the end of the buffer
|
|
692
|
+
*/
|
|
693
|
+
readName(i) {
|
|
694
|
+
const { buffer } = this;
|
|
695
|
+
const { length } = buffer;
|
|
696
|
+
let j = i;
|
|
697
|
+
while (j < length) {
|
|
698
|
+
const c = buffer.charCodeAt(j);
|
|
699
|
+
if (c < 128) {
|
|
700
|
+
if ((j === i ? ASCII_NAME_START[c] : ASCII_NAME_CHAR[c]) === 0) {
|
|
701
|
+
break;
|
|
702
|
+
}
|
|
703
|
+
j++;
|
|
704
|
+
continue;
|
|
705
|
+
}
|
|
706
|
+
const cp = buffer.codePointAt(j);
|
|
707
|
+
if (cp === void 0 || (j === i ? !isNameStart(cp) : !isNameChar(cp))) {
|
|
708
|
+
break;
|
|
709
|
+
}
|
|
710
|
+
j += cp > 65535 ? 2 : 1;
|
|
711
|
+
}
|
|
712
|
+
return j >= length && !this.final ? -1 : j;
|
|
713
|
+
}
|
|
714
|
+
/**
|
|
715
|
+
* Move text from the buffer into the current run.
|
|
716
|
+
* @param start - the start index
|
|
717
|
+
* @param end - the end index (exclusive)
|
|
718
|
+
*/
|
|
719
|
+
takeText(start, end) {
|
|
720
|
+
if (end <= start) {
|
|
721
|
+
return;
|
|
722
|
+
}
|
|
723
|
+
if (this.text.length === 0) {
|
|
724
|
+
this.textLine = this.line;
|
|
725
|
+
}
|
|
726
|
+
const raw = this.buffer.slice(start, end);
|
|
727
|
+
if (hasIllegalXmlChar(raw)) {
|
|
728
|
+
throw new XmlSyntaxError("a character XML 1.0 forbids appears in character data", this.line);
|
|
729
|
+
}
|
|
730
|
+
this.text += decodeEntities(raw, this.line);
|
|
731
|
+
this.advanceLine(start, end);
|
|
732
|
+
}
|
|
733
|
+
/**
|
|
734
|
+
* Count the newlines of a consumed range. Ranges are consumed in order, so the next line break
|
|
735
|
+
* is searched for once per buffer and re-searched only after it was passed; a buffer without
|
|
736
|
+
* a further break is searched once and remembered as such.
|
|
737
|
+
* @param start - the start index
|
|
738
|
+
* @param end - the end index (exclusive)
|
|
739
|
+
*/
|
|
740
|
+
advanceLine(start, end) {
|
|
741
|
+
let i = this.nextBreak;
|
|
742
|
+
if (i === -1) {
|
|
743
|
+
return;
|
|
744
|
+
}
|
|
745
|
+
if (i === -2 || i < start) {
|
|
746
|
+
i = this.buffer.indexOf("\n", start);
|
|
747
|
+
}
|
|
748
|
+
while (i >= 0 && i < end) {
|
|
749
|
+
this.line++;
|
|
750
|
+
i = this.buffer.indexOf("\n", i + 1);
|
|
751
|
+
}
|
|
752
|
+
this.nextBreak = i;
|
|
753
|
+
}
|
|
754
|
+
/** Deliver the accumulated text run, if any. */
|
|
755
|
+
flushText() {
|
|
756
|
+
if (this.text.length === 0) {
|
|
757
|
+
return;
|
|
758
|
+
}
|
|
759
|
+
const { text } = this;
|
|
760
|
+
this.text = "";
|
|
761
|
+
if (this.stack.length === 0) {
|
|
762
|
+
if (!isWhitespace(text)) {
|
|
763
|
+
throw new XmlSyntaxError("text outside the root element", this.textLine);
|
|
764
|
+
}
|
|
765
|
+
return;
|
|
766
|
+
}
|
|
767
|
+
this.handler.text(text, this.textLine);
|
|
768
|
+
}
|
|
769
|
+
/**
|
|
770
|
+
* Open an element.
|
|
771
|
+
* @param name - the element name
|
|
772
|
+
* @param attrs - its attributes
|
|
773
|
+
*/
|
|
774
|
+
startElement(name, attrs) {
|
|
775
|
+
if (this.stack.length === 0) {
|
|
776
|
+
if (this.rootClosed) {
|
|
777
|
+
throw new XmlSyntaxError(`a second root element <${name}> follows the document element`, this.line);
|
|
778
|
+
}
|
|
779
|
+
this.rootSeen = true;
|
|
780
|
+
}
|
|
781
|
+
this.stack.push(name);
|
|
782
|
+
this.handler.start(name, attrs, this.line);
|
|
783
|
+
}
|
|
784
|
+
/**
|
|
785
|
+
* Close an element, checking that it matches the innermost open one.
|
|
786
|
+
* @param name - the element name
|
|
787
|
+
*/
|
|
788
|
+
endElement(name) {
|
|
789
|
+
const open = this.stack.length > 0 ? this.stack[this.stack.length - 1] : null;
|
|
790
|
+
if (open === null) {
|
|
791
|
+
throw new XmlSyntaxError(`unexpected end tag </${name}>`, this.line);
|
|
792
|
+
}
|
|
793
|
+
if (open !== name) {
|
|
794
|
+
throw new XmlSyntaxError(`end tag </${name}> does not match <${open}>`, this.line);
|
|
795
|
+
}
|
|
796
|
+
this.stack.pop();
|
|
797
|
+
if (this.stack.length === 0) {
|
|
798
|
+
this.rootClosed = true;
|
|
799
|
+
}
|
|
800
|
+
this.handler.end(name, this.line);
|
|
801
|
+
}
|
|
802
|
+
}
|
|
803
|
+
function normalizeAttributeValue(raw) {
|
|
804
|
+
return raw.includes("\n") || raw.includes(" ") ? raw.replace(/[\n\t]/g, " ") : raw;
|
|
805
|
+
}
|
|
806
|
+
const XML_TEXT_SPECIAL = /[&<>\r]/g;
|
|
807
|
+
const XML_ATTR_SPECIAL = /[&<>"\t\n\r]/g;
|
|
808
|
+
const XML_REPLACEMENTS = {
|
|
809
|
+
"&": "&",
|
|
810
|
+
"<": "<",
|
|
811
|
+
">": ">",
|
|
812
|
+
'"': """,
|
|
813
|
+
" ": "	",
|
|
814
|
+
"\n": " ",
|
|
815
|
+
"\r": " "
|
|
816
|
+
};
|
|
817
|
+
function escapeXmlText(text) {
|
|
818
|
+
if (hasIllegalXmlChar(text)) {
|
|
819
|
+
throw illegalXml(text);
|
|
820
|
+
}
|
|
821
|
+
return text.replace(XML_TEXT_SPECIAL, (c) => XML_REPLACEMENTS[c]);
|
|
822
|
+
}
|
|
823
|
+
function escapeXmlAttribute(text) {
|
|
824
|
+
if (hasIllegalXmlChar(text)) {
|
|
825
|
+
throw illegalXml(text);
|
|
826
|
+
}
|
|
827
|
+
return text.replace(XML_ATTR_SPECIAL, (c) => XML_REPLACEMENTS[c]);
|
|
828
|
+
}
|
|
829
|
+
function illegalXml(text) {
|
|
830
|
+
return new GraphFormatError("E_COLUMN_TYPE", "the text holds a character XML 1.0 cannot carry", {
|
|
831
|
+
reason: "xml illegal char",
|
|
832
|
+
length: text.length
|
|
833
|
+
});
|
|
834
|
+
}
|
|
835
|
+
const GML_SPECIAL = /["&]|[^ -~]/gu;
|
|
836
|
+
function quoteGmlString(text) {
|
|
837
|
+
const escaped = text.replace(GML_SPECIAL, (c) => {
|
|
838
|
+
const code = c.codePointAt(0);
|
|
839
|
+
if (code === void 0) {
|
|
840
|
+
return c;
|
|
841
|
+
}
|
|
842
|
+
if (code >= 55296 && code <= 57343) {
|
|
843
|
+
throw new GraphFormatError("E_COLUMN_TYPE", "a lone surrogate cannot be written as a GML string", {
|
|
844
|
+
reason: "lone surrogate"
|
|
845
|
+
});
|
|
846
|
+
}
|
|
847
|
+
return `&#${code};`;
|
|
848
|
+
});
|
|
849
|
+
return `"${escaped}"`;
|
|
850
|
+
}
|
|
851
|
+
const GML_ENTITY = /&(#[0-9]+|#x[0-9a-fA-F]+|amp|quot|lt|gt|apos);/g;
|
|
852
|
+
const GML_NAMED = { amp: "&", quot: '"', lt: "<", gt: ">", apos: "'" };
|
|
853
|
+
function decodeGmlString(body) {
|
|
854
|
+
return body.replace(GML_ENTITY, (whole, entity) => {
|
|
855
|
+
if (entity.startsWith("#x")) {
|
|
856
|
+
return String.fromCodePoint(Number.parseInt(entity.slice(2), 16));
|
|
857
|
+
}
|
|
858
|
+
if (entity.startsWith("#")) {
|
|
859
|
+
return String.fromCodePoint(Number.parseInt(entity.slice(1), 10));
|
|
860
|
+
}
|
|
861
|
+
return GML_NAMED[entity] ?? whole;
|
|
862
|
+
});
|
|
863
|
+
}
|
|
864
|
+
const DOT_NUMERAL = /^-?(\.[0-9]+|[0-9]+(\.[0-9]*)?)$/;
|
|
865
|
+
const DOT_KEYWORDS = /* @__PURE__ */ new Set(["node", "edge", "graph", "digraph", "subgraph", "strict"]);
|
|
866
|
+
function isBareDotId(text) {
|
|
867
|
+
if (DOT_KEYWORDS.has(text.toLowerCase())) {
|
|
868
|
+
return false;
|
|
869
|
+
}
|
|
870
|
+
return isDotIdentifier(text) || DOT_NUMERAL.test(text);
|
|
871
|
+
}
|
|
872
|
+
function isDotIdentifier(text) {
|
|
873
|
+
if (text.length === 0) {
|
|
874
|
+
return false;
|
|
875
|
+
}
|
|
876
|
+
for (let i = 0; i < text.length; i++) {
|
|
877
|
+
const c = text.charCodeAt(i);
|
|
878
|
+
const letter = c >= 65 && c <= 90 || c >= 97 && c <= 122 || c === 95 || c >= 128;
|
|
879
|
+
const digit = c >= 48 && c <= 57;
|
|
880
|
+
if (!letter && (i === 0 || !digit)) {
|
|
881
|
+
return false;
|
|
882
|
+
}
|
|
883
|
+
}
|
|
884
|
+
return true;
|
|
885
|
+
}
|
|
886
|
+
function isWritableDotText(text) {
|
|
887
|
+
return !text.endsWith("\\") && !text.includes('\\"');
|
|
888
|
+
}
|
|
889
|
+
function quoteDotId(text) {
|
|
890
|
+
if (isBareDotId(text)) {
|
|
891
|
+
return text;
|
|
892
|
+
}
|
|
893
|
+
return `"${text.replace(/"/g, '\\"')}"`;
|
|
894
|
+
}
|
|
895
|
+
function quoteCsvCell(text, delimiter = ",") {
|
|
896
|
+
if (text.length === 0) {
|
|
897
|
+
return '""';
|
|
898
|
+
}
|
|
899
|
+
if (!text.includes(delimiter) && !text.includes('"') && !text.includes("\n") && !text.includes("\r") && text === text.trim()) {
|
|
900
|
+
return text;
|
|
901
|
+
}
|
|
902
|
+
return `"${text.replace(/"/g, '""')}"`;
|
|
903
|
+
}
|
|
904
|
+
function isPajekLabel(text) {
|
|
905
|
+
return !text.includes('"') && !text.includes("\n") && !text.includes("\r");
|
|
906
|
+
}
|
|
907
|
+
function quotePajekLabel(text) {
|
|
908
|
+
if (!isPajekLabel(text)) {
|
|
909
|
+
throw new GraphFormatError("E_UNSUPPORTED", "a Pajek label cannot contain a double quote or a line break", {
|
|
910
|
+
reason: "pajek label",
|
|
911
|
+
value: text
|
|
912
|
+
});
|
|
913
|
+
}
|
|
914
|
+
if (text.length > 0 && !/[\s"]/.test(text)) {
|
|
915
|
+
return text;
|
|
916
|
+
}
|
|
917
|
+
return `"${text}"`;
|
|
918
|
+
}
|
|
919
|
+
export {
|
|
920
|
+
XmlSyntaxError as X,
|
|
921
|
+
XmlTokenizer as a,
|
|
922
|
+
quoteDotId as b,
|
|
923
|
+
isWritableDotText as c,
|
|
924
|
+
isWhitespace as d,
|
|
925
|
+
escapeXmlText as e,
|
|
926
|
+
escapeXmlAttribute as f,
|
|
927
|
+
decodeGmlString as g,
|
|
928
|
+
hasIllegalXmlChar as h,
|
|
929
|
+
isPajekLabel as i,
|
|
930
|
+
quoteGmlString as j,
|
|
931
|
+
isXmlName as k,
|
|
932
|
+
localName as l,
|
|
933
|
+
quoteCsvCell as m,
|
|
934
|
+
quotePajekLabel as q,
|
|
935
|
+
tokenizeXml as t,
|
|
936
|
+
xmlIllegalTextNotes as x
|
|
937
|
+
};
|
|
938
|
+
//# sourceMappingURL=escape-DyI8JofU.js.map
|