@graphty/graph-io 0.3.3 → 0.3.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +17 -4
- package/dist/chunks/{importer-BELrICWM.js → importer-BOxwmef_.js} +352 -19
- package/dist/chunks/importer-BOxwmef_.js.map +1 -0
- package/dist/graph-io.js +2 -2
- package/dist/graph-io.js.map +1 -1
- package/dist/json.js +1 -1
- package/dist/src/formats/json/dialect.d.ts +21 -4
- package/dist/src/formats/json/dialect.d.ts.map +1 -1
- package/dist/src/formats/json/dialect.js +29 -4
- package/dist/src/formats/json/dialect.js.map +1 -1
- package/dist/src/formats/json/exporter.d.ts.map +1 -1
- package/dist/src/formats/json/exporter.js +27 -7
- package/dist/src/formats/json/exporter.js.map +1 -1
- package/dist/src/formats/json/importer.d.ts +17 -4
- package/dist/src/formats/json/importer.d.ts.map +1 -1
- package/dist/src/formats/json/importer.js +356 -8
- package/dist/src/formats/json/importer.js.map +1 -1
- package/dist/src/formats/json/index.d.ts +1 -1
- package/dist/src/formats/json/index.d.ts.map +1 -1
- package/dist/src/formats/json/index.js +1 -1
- package/dist/src/formats/json/index.js.map +1 -1
- package/dist/src/index.d.ts +1 -1
- package/dist/src/index.d.ts.map +1 -1
- package/dist/src/index.js.map +1 -1
- package/dist/src/sniff.d.ts +3 -3
- package/dist/src/sniff.d.ts.map +1 -1
- package/dist/src/sniff.js.map +1 -1
- package/package.json +1 -1
- package/src/common/escape.ts +5 -4
- package/src/formats/json/dialect.ts +40 -6
- package/src/formats/json/exporter.ts +28 -7
- package/src/formats/json/importer.ts +411 -21
- package/src/formats/json/index.ts +7 -1
- package/src/formats/neo4j/exporter.ts +43 -6
- package/src/formats/neo4j/importer.ts +56 -11
- package/src/formats/neo4j/index.ts +10 -3
- package/src/formats/pajek/exporter.ts +3 -1
- package/src/formats/pajek/importer.ts +284 -60
- package/src/formats/pajek/syntax.ts +98 -12
- package/src/index.ts +2 -0
- package/src/sniff.ts +3 -3
- package/dist/chunks/importer-BELrICWM.js.map +0 -1
|
@@ -5,7 +5,13 @@
|
|
|
5
5
|
* index endpoints), JSON Graph Format v2 (nodes keyed by id, per-edge `directed`, hyperedges),
|
|
6
6
|
* Cytoscape.js elements (`data.id` / `data.source` / `data.target`, `position`, `classes`,
|
|
7
7
|
* `data.parent`), graphology serialisation (`key` / `attributes`, `undirected` edges, `options`)
|
|
8
|
-
*
|
|
8
|
+
* vis.js (`from` / `to`), and NetworkX adjacency_data (`nodes` + `adjacency`) and tree_data (nested
|
|
9
|
+
* `id` / `children`) -- and pushes it scalar by scalar into the sink.
|
|
10
|
+
*
|
|
11
|
+
* Python's json module writes the bare tokens NaN, Infinity and -Infinity, which strict JSON
|
|
12
|
+
* refuses; they are read as the JavaScript numbers with one warning. An integer literal beyond
|
|
13
|
+
* 2^53 is read as its exact digits (a string, the canonical id rule of the text formats) rather
|
|
14
|
+
* than a rounded float, again with one warning.
|
|
9
15
|
*
|
|
10
16
|
* JSON awaits the whole text (design section 8.4: `JSON.parse` on 100 MB is fine; a streaming
|
|
11
17
|
* tokeniser is a later improvement). The parsed records are iterated in place; the importer never
|
|
@@ -72,10 +78,10 @@ import {
|
|
|
72
78
|
CYTOSCAPE_STRUCTURAL_KEYS,
|
|
73
79
|
DIALECT_DEFAULT_DIRECTED,
|
|
74
80
|
hasKey,
|
|
75
|
-
|
|
81
|
+
isJsonImportDialect,
|
|
76
82
|
isJsonObject,
|
|
77
|
-
|
|
78
|
-
type
|
|
83
|
+
JSON_IMPORT_DIALECTS,
|
|
84
|
+
type JsonImportDialect,
|
|
79
85
|
type JsonShapeMeta,
|
|
80
86
|
META_KEY,
|
|
81
87
|
NODE_LINK_SOURCE_KEYS,
|
|
@@ -89,8 +95,11 @@ import {
|
|
|
89
95
|
/** The format-specific options of the JSON importer. */
|
|
90
96
|
export interface JsonImportOptions {
|
|
91
97
|
/** The dialect to read; "auto" (default) sniffs the parsed document. */
|
|
92
|
-
dialect?:
|
|
93
|
-
/**
|
|
98
|
+
dialect?: JsonImportDialect | "auto" | undefined;
|
|
99
|
+
/**
|
|
100
|
+
* node-link / d3 / vis / adjacency / tree: the node key holding the id; auto: "id" (node-link /
|
|
101
|
+
* d3: "id" when any node has it, else "name").
|
|
102
|
+
*/
|
|
94
103
|
nodeIdKey?: string | undefined;
|
|
95
104
|
/** node-link / d3: the top-level key holding the edges; auto: "edges" when present, else "links". */
|
|
96
105
|
edgesKey?: string | undefined;
|
|
@@ -169,6 +178,10 @@ export const JSON_ISSUE = Object.freeze({
|
|
|
169
178
|
INVALID_ENCODING: INVALID_ENCODING_CODE,
|
|
170
179
|
/** Bytes that are not UTF-8 and declare no encoding were read as windows-1252. */
|
|
171
180
|
ENCODING_FALLBACK: ENCODING_FALLBACK_CODE,
|
|
181
|
+
/** The document uses the non-standard tokens NaN / Infinity / -Infinity (Python's json writes them); read as numbers. */
|
|
182
|
+
NONSTANDARD_NUMBER: "W_JSON_NONSTANDARD_NUMBER",
|
|
183
|
+
/** Integer literals beyond 2^53 were read as their exact digits (strings), not as rounded numbers. */
|
|
184
|
+
BIG_INTEGER: "W_JSON_BIG_INTEGER",
|
|
172
185
|
/** A declared encoding the platform cannot decode was ignored. */
|
|
173
186
|
UNKNOWN_ENCODING: UNKNOWN_ENCODING_CODE,
|
|
174
187
|
});
|
|
@@ -243,7 +256,7 @@ interface EdgeIdColumn {
|
|
|
243
256
|
|
|
244
257
|
/** The resolved format-specific options. */
|
|
245
258
|
interface ResolvedJsonOptions {
|
|
246
|
-
readonly dialect:
|
|
259
|
+
readonly dialect: JsonImportDialect | "auto";
|
|
247
260
|
readonly nodeIdKey: string | null;
|
|
248
261
|
readonly edgesKey: string | null;
|
|
249
262
|
readonly sourceKey: string | null;
|
|
@@ -262,8 +275,8 @@ interface ResolvedJsonOptions {
|
|
|
262
275
|
function resolveJsonOptions(options: (JsonImportOptions & CommonImportOptions) | undefined): ResolvedJsonOptions {
|
|
263
276
|
const o = options ?? {};
|
|
264
277
|
const dialect = o.dialect ?? "auto";
|
|
265
|
-
if (dialect !== "auto" && !
|
|
266
|
-
throw unsupportedOption("dialect", dialect, [...
|
|
278
|
+
if (dialect !== "auto" && !isJsonImportDialect(dialect)) {
|
|
279
|
+
throw unsupportedOption("dialect", dialect, [...JSON_IMPORT_DIALECTS, "auto"]);
|
|
267
280
|
}
|
|
268
281
|
const indexLinks = o.indexLinks ?? "auto";
|
|
269
282
|
if (indexLinks !== "auto" && typeof indexLinks !== "boolean") {
|
|
@@ -506,7 +519,7 @@ class ImportContext {
|
|
|
506
519
|
* @param dialect - the dialect
|
|
507
520
|
* @param idField - where the dialect's ids come from, for the message
|
|
508
521
|
*/
|
|
509
|
-
reportNodeIdFrom(dialect:
|
|
522
|
+
reportNodeIdFrom(dialect: JsonImportDialect, idField: string): void {
|
|
510
523
|
if (this.options.nodeIdFrom !== "id") {
|
|
511
524
|
this.report.warning(
|
|
512
525
|
"unsupported",
|
|
@@ -523,7 +536,7 @@ class ImportContext {
|
|
|
523
536
|
* @param dialect - the dialect
|
|
524
537
|
* @returns the direction
|
|
525
538
|
*/
|
|
526
|
-
defaultDirected(dialect:
|
|
539
|
+
defaultDirected(dialect: JsonImportDialect): boolean {
|
|
527
540
|
return this.explicitDefaultDirected ? this.options.defaultDirected : DIALECT_DEFAULT_DIRECTED[dialect];
|
|
528
541
|
}
|
|
529
542
|
|
|
@@ -962,12 +975,153 @@ function parseDocument(text: string, report: ImportReportBuilder): unknown {
|
|
|
962
975
|
if (text.trim().length === 0) {
|
|
963
976
|
report.fail(JSON_ISSUE.EMPTY_INPUT, "the input is empty");
|
|
964
977
|
}
|
|
978
|
+
let root: unknown;
|
|
979
|
+
let syntaxError: string | null = null;
|
|
965
980
|
try {
|
|
966
|
-
|
|
981
|
+
root = JSON.parse(text) as unknown;
|
|
967
982
|
} catch (err) {
|
|
968
|
-
|
|
969
|
-
|
|
983
|
+
syntaxError = err instanceof Error ? err.message : String(err);
|
|
984
|
+
}
|
|
985
|
+
// the fast path: strict JSON without a digit run long enough to be an unsafe integer
|
|
986
|
+
if (syntaxError === null && !MAYBE_UNSAFE_INTEGER.test(text)) {
|
|
987
|
+
return root;
|
|
988
|
+
}
|
|
989
|
+
const scan = rewriteNumbers(text);
|
|
990
|
+
if (scan.tokens.size > 0 || scan.bigIntegers.length > 0) {
|
|
991
|
+
try {
|
|
992
|
+
root = JSON.parse(scan.text, scan.tokens.size > 0 ? reviveNonstandard : undefined) as unknown;
|
|
993
|
+
syntaxError = null;
|
|
994
|
+
} catch {
|
|
995
|
+
// the rewrite did not make it valid JSON: report the parser's message on the original text
|
|
996
|
+
}
|
|
997
|
+
}
|
|
998
|
+
if (syntaxError !== null) {
|
|
999
|
+
return report.fail(JSON_ISSUE.SYNTAX, `invalid JSON: ${syntaxError}`);
|
|
1000
|
+
}
|
|
1001
|
+
if (scan.tokens.size > 0) {
|
|
1002
|
+
report.warning(
|
|
1003
|
+
"coercion",
|
|
1004
|
+
JSON_ISSUE.NONSTANDARD_NUMBER,
|
|
1005
|
+
`the document uses the non-standard token(s) ${[...scan.tokens].join(", ")}, which strict JSON does not allow; read as numbers`,
|
|
1006
|
+
);
|
|
1007
|
+
}
|
|
1008
|
+
if (scan.bigIntegers.length > 0) {
|
|
1009
|
+
const shown = scan.bigIntegers.slice(0, BIG_INTEGERS_SHOWN).join(", ");
|
|
1010
|
+
const more = scan.bigIntegers.length - BIG_INTEGERS_SHOWN;
|
|
1011
|
+
report.warning(
|
|
1012
|
+
"precision",
|
|
1013
|
+
JSON_ISSUE.BIG_INTEGER,
|
|
1014
|
+
`${scan.bigIntegers.length} integer(s) beyond 2^53 kept as text so no digit is lost: ${shown}${more > 0 ? ` and ${more} more` : ""}`,
|
|
1015
|
+
);
|
|
1016
|
+
}
|
|
1017
|
+
return root;
|
|
1018
|
+
}
|
|
1019
|
+
|
|
1020
|
+
/** A run of 16 digits not inside a fraction: the shortest integer literal that can exceed 2^53 (9007199254740992). */
|
|
1021
|
+
const MAYBE_UNSAFE_INTEGER = /(?<![0-9.])[0-9]{16}/;
|
|
1022
|
+
|
|
1023
|
+
/** How many of the integers kept as text the warning lists. */
|
|
1024
|
+
const BIG_INTEGERS_SHOWN = 10;
|
|
1025
|
+
|
|
1026
|
+
/**
|
|
1027
|
+
* The prefix of the string a non-standard token is rewritten to (a NUL character first, which no
|
|
1028
|
+
* sensible attribute value starts with); reviveNonstandard() turns it back into the number.
|
|
1029
|
+
*/
|
|
1030
|
+
const NONSTANDARD_SENTINEL = `${String.fromCharCode(0)}graph-io:`;
|
|
1031
|
+
|
|
1032
|
+
/** The non-standard tokens Python's json module writes, longest first so -Infinity wins over a bare minus. */
|
|
1033
|
+
const NONSTANDARD_TOKENS: readonly (readonly [string, number])[] = [
|
|
1034
|
+
["-Infinity", -Infinity],
|
|
1035
|
+
["Infinity", Infinity],
|
|
1036
|
+
["NaN", NaN],
|
|
1037
|
+
];
|
|
1038
|
+
|
|
1039
|
+
/** A JSON integer literal (no fraction, no exponent, no leading zero), as CANONICAL_INTEGER in common/ids.ts. */
|
|
1040
|
+
const INTEGER_LITERAL = /^-?(0|[1-9][0-9]*)$/;
|
|
1041
|
+
|
|
1042
|
+
/**
|
|
1043
|
+
* Rewrite the numbers JSON.parse cannot read exactly, outside strings and in value positions only:
|
|
1044
|
+
* NaN / Infinity / -Infinity become sentinel strings, an integer literal that is not a safe
|
|
1045
|
+
* integer becomes a string of its digits. A container stack tells a value position (after `:`,
|
|
1046
|
+
* `[`, or `,` inside an array) from a key position, so `{NaN: 1}` stays invalid.
|
|
1047
|
+
* @param text - the document text
|
|
1048
|
+
* @returns the rewritten text, the non-standard tokens seen and the integer literals quoted
|
|
1049
|
+
*/
|
|
1050
|
+
function rewriteNumbers(text: string): { text: string; tokens: Set<string>; bigIntegers: string[] } {
|
|
1051
|
+
const parts: string[] = [];
|
|
1052
|
+
const tokens = new Set<string>();
|
|
1053
|
+
const bigIntegers: string[] = [];
|
|
1054
|
+
const arrays: boolean[] = [];
|
|
1055
|
+
let expectValue = true;
|
|
1056
|
+
let copied = 0;
|
|
1057
|
+
let i = 0;
|
|
1058
|
+
const n = text.length;
|
|
1059
|
+
while (i < n) {
|
|
1060
|
+
const ch = text[i];
|
|
1061
|
+
if (ch === '"') {
|
|
1062
|
+
i++;
|
|
1063
|
+
while (i < n && text[i] !== '"') {
|
|
1064
|
+
i += text[i] === "\\" ? 2 : 1;
|
|
1065
|
+
}
|
|
1066
|
+
i++;
|
|
1067
|
+
expectValue = false;
|
|
1068
|
+
continue;
|
|
1069
|
+
}
|
|
1070
|
+
if (ch === "{" || ch === "[") {
|
|
1071
|
+
arrays.push(ch === "[");
|
|
1072
|
+
expectValue = ch === "[";
|
|
1073
|
+
} else if (ch === "}" || ch === "]") {
|
|
1074
|
+
arrays.pop();
|
|
1075
|
+
expectValue = false;
|
|
1076
|
+
} else if (ch === ":") {
|
|
1077
|
+
expectValue = true;
|
|
1078
|
+
} else if (ch === ",") {
|
|
1079
|
+
expectValue = arrays.length > 0 && arrays[arrays.length - 1];
|
|
1080
|
+
} else if (expectValue && ch !== " " && ch !== "\t" && ch !== "\n" && ch !== "\r") {
|
|
1081
|
+
expectValue = false;
|
|
1082
|
+
const token = NONSTANDARD_TOKENS.find(([word]) => text.startsWith(word, i));
|
|
1083
|
+
let end = i;
|
|
1084
|
+
let replacement: string | null = null;
|
|
1085
|
+
if (token !== undefined) {
|
|
1086
|
+
end = i + token[0].length;
|
|
1087
|
+
tokens.add(token[0]);
|
|
1088
|
+
replacement = JSON.stringify(`${NONSTANDARD_SENTINEL}${token[0]}`);
|
|
1089
|
+
} else if (ch === "-" || (ch >= "0" && ch <= "9")) {
|
|
1090
|
+
end = i + 1;
|
|
1091
|
+
while (end < n && "0123456789+-.eE".includes(text[end])) {
|
|
1092
|
+
end++;
|
|
1093
|
+
}
|
|
1094
|
+
const literal = text.slice(i, end);
|
|
1095
|
+
if (INTEGER_LITERAL.test(literal) && !Number.isSafeInteger(Number(literal))) {
|
|
1096
|
+
bigIntegers.push(literal);
|
|
1097
|
+
replacement = `"${literal}"`;
|
|
1098
|
+
}
|
|
1099
|
+
}
|
|
1100
|
+
if (replacement !== null) {
|
|
1101
|
+
parts.push(text.slice(copied, i), replacement);
|
|
1102
|
+
copied = end;
|
|
1103
|
+
}
|
|
1104
|
+
i = Math.max(end, i + 1);
|
|
1105
|
+
continue;
|
|
1106
|
+
}
|
|
1107
|
+
i++;
|
|
1108
|
+
}
|
|
1109
|
+
parts.push(text.slice(copied));
|
|
1110
|
+
return { text: parts.join(""), tokens, bigIntegers };
|
|
1111
|
+
}
|
|
1112
|
+
|
|
1113
|
+
/**
|
|
1114
|
+
* The JSON.parse reviver that turns the sentinel strings of rewriteNumbers() back into numbers.
|
|
1115
|
+
* @param _key - the member key (unused)
|
|
1116
|
+
* @param value - the parsed value
|
|
1117
|
+
* @returns the number for a sentinel string, the value otherwise
|
|
1118
|
+
*/
|
|
1119
|
+
function reviveNonstandard(_key: string, value: unknown): unknown {
|
|
1120
|
+
if (typeof value === "string" && value.startsWith(NONSTANDARD_SENTINEL)) {
|
|
1121
|
+
const found = NONSTANDARD_TOKENS.find(([word]) => word === value.slice(NONSTANDARD_SENTINEL.length));
|
|
1122
|
+
return found === undefined ? value : found[1];
|
|
970
1123
|
}
|
|
1124
|
+
return value;
|
|
971
1125
|
}
|
|
972
1126
|
|
|
973
1127
|
/**
|
|
@@ -978,7 +1132,11 @@ function parseDocument(text: string, report: ImportReportBuilder): unknown {
|
|
|
978
1132
|
* @param report - the report the failure is recorded in
|
|
979
1133
|
* @returns the dialect
|
|
980
1134
|
*/
|
|
981
|
-
function detectDialect(
|
|
1135
|
+
function detectDialect(
|
|
1136
|
+
root: unknown,
|
|
1137
|
+
forced: JsonImportDialect | "auto",
|
|
1138
|
+
report: ImportReportBuilder,
|
|
1139
|
+
): JsonImportDialect {
|
|
982
1140
|
if (forced !== "auto") {
|
|
983
1141
|
return forced;
|
|
984
1142
|
}
|
|
@@ -997,7 +1155,7 @@ function detectDialect(root: unknown, forced: JsonDialect | "auto", report: Impo
|
|
|
997
1155
|
}
|
|
998
1156
|
return report.fail(
|
|
999
1157
|
JSON_ISSUE.DIALECT,
|
|
1000
|
-
"no known dialect: expected nodes / links / edges (node-link), elements (Cytoscape) or graph (JGF)",
|
|
1158
|
+
"no known dialect: expected nodes / links / edges (node-link), nodes / adjacency (adjacency), children (tree), elements (Cytoscape) or graph (JGF)",
|
|
1001
1159
|
);
|
|
1002
1160
|
}
|
|
1003
1161
|
|
|
@@ -1094,7 +1252,7 @@ function isIndexBelow(value: unknown, bound: number): boolean {
|
|
|
1094
1252
|
* @param root - the document
|
|
1095
1253
|
* @param dialect - "node-link" or "d3"
|
|
1096
1254
|
*/
|
|
1097
|
-
function importNodeLink(ctx: ImportContext, root: JsonRecord, dialect:
|
|
1255
|
+
function importNodeLink(ctx: ImportContext, root: JsonRecord, dialect: "node-link" | "d3"): void {
|
|
1098
1256
|
const { report, json } = ctx;
|
|
1099
1257
|
let { edgesKey } = json;
|
|
1100
1258
|
if (edgesKey === null) {
|
|
@@ -1128,9 +1286,14 @@ function importNodeLink(ctx: ImportContext, root: JsonRecord, dialect: JsonDiale
|
|
|
1128
1286
|
const what = Array.isArray(value)
|
|
1129
1287
|
? `${String(value.length)} ${value.length === 1 ? "entry" : "entries"}`
|
|
1130
1288
|
: describe(value);
|
|
1131
|
-
report.warning(
|
|
1132
|
-
|
|
1133
|
-
|
|
1289
|
+
report.warning(
|
|
1290
|
+
"unsupported",
|
|
1291
|
+
JSON_ISSUE.UNREAD_KEY,
|
|
1292
|
+
`top-level key ${key} (${what}) is not read; dropped`,
|
|
1293
|
+
{
|
|
1294
|
+
element: key,
|
|
1295
|
+
},
|
|
1296
|
+
);
|
|
1134
1297
|
}
|
|
1135
1298
|
}
|
|
1136
1299
|
|
|
@@ -1460,6 +1623,227 @@ function importVis(ctx: ImportContext, root: JsonRecord): void {
|
|
|
1460
1623
|
);
|
|
1461
1624
|
}
|
|
1462
1625
|
|
|
1626
|
+
// ============================================================ NetworkX adjacency_data / tree_data
|
|
1627
|
+
|
|
1628
|
+
/**
|
|
1629
|
+
* Read a NetworkX adjacency_data document: `nodes` as in node-link, and `adjacency[i]` the
|
|
1630
|
+
* neighbour list of `nodes[i]`, one `{ id, key?, ...attributes }` entry per edge. An undirected
|
|
1631
|
+
* file lists every edge from both ends, so an entry whose mirror (the same pair and `key`) was
|
|
1632
|
+
* already read is that edge again and is not pushed twice; a self-loop is listed once.
|
|
1633
|
+
* @param ctx - the context
|
|
1634
|
+
* @param root - the document
|
|
1635
|
+
*/
|
|
1636
|
+
function importAdjacency(ctx: ImportContext, root: JsonRecord): void {
|
|
1637
|
+
const { report } = ctx;
|
|
1638
|
+
const nodes = arraySection(root.nodes, "nodes", report) ?? [];
|
|
1639
|
+
const adjacency = arraySection(root.adjacency, "adjacency", report);
|
|
1640
|
+
if (adjacency === null) {
|
|
1641
|
+
report.error(
|
|
1642
|
+
"missing-value",
|
|
1643
|
+
JSON_ISSUE.MISSING_SECTION,
|
|
1644
|
+
"the document has no adjacency array; the graph has no edges",
|
|
1645
|
+
{
|
|
1646
|
+
element: "adjacency",
|
|
1647
|
+
},
|
|
1648
|
+
);
|
|
1649
|
+
}
|
|
1650
|
+
const directed = flagOf(root.directed, "directed", ctx.defaultDirected("adjacency"), report);
|
|
1651
|
+
const multigraph = hasKey(root, "multigraph") ? flagOf(root.multigraph, "multigraph", false, report) : null;
|
|
1652
|
+
ctx.setHeader(directed);
|
|
1653
|
+
// adjacency_data writes the graph dict as a list of [key, value] pairs
|
|
1654
|
+
ctx.writeGraphDict(isPairList(root.graph) ? Object.fromEntries(root.graph) : root.graph, "graph");
|
|
1655
|
+
for (const key of Object.keys(root)) {
|
|
1656
|
+
if (key !== "nodes" && key !== "adjacency" && key !== "directed" && key !== "multigraph" && key !== "graph") {
|
|
1657
|
+
report.warning(
|
|
1658
|
+
"unsupported",
|
|
1659
|
+
JSON_ISSUE.UNREAD_KEY,
|
|
1660
|
+
`top-level key ${key} (${describe(root[key])}) is not read; dropped`,
|
|
1661
|
+
{
|
|
1662
|
+
element: key,
|
|
1663
|
+
},
|
|
1664
|
+
);
|
|
1665
|
+
}
|
|
1666
|
+
}
|
|
1667
|
+
const lists = adjacency ?? [];
|
|
1668
|
+
ctx.sink.reserve(
|
|
1669
|
+
nodes.length,
|
|
1670
|
+
lists.reduce<number>((sum, list) => sum + (Array.isArray(list) ? list.length : 0), 0),
|
|
1671
|
+
);
|
|
1672
|
+
const idKey = ctx.json.nodeIdKey ?? "id";
|
|
1673
|
+
ctx.reportNodeIdFrom("adjacency", `the ${JSON.stringify(idKey)} key`);
|
|
1674
|
+
const owners: (NodeId | null)[] = [];
|
|
1675
|
+
for (let i = 0; i < nodes.length; i++) {
|
|
1676
|
+
const element = `nodes[${i}]`;
|
|
1677
|
+
const record = nodes[i];
|
|
1678
|
+
let pushed: NodeId | null = null;
|
|
1679
|
+
if (!isJsonObject(record)) {
|
|
1680
|
+
ctx.badElement("node", element);
|
|
1681
|
+
} else {
|
|
1682
|
+
const id = ctx.coerceId(hasKey(record, idKey) ? record[idKey] : undefined, element);
|
|
1683
|
+
if (id === null) {
|
|
1684
|
+
ctx.countSkipped("node");
|
|
1685
|
+
} else {
|
|
1686
|
+
const index = ctx.pushNode(id, element);
|
|
1687
|
+
if (index >= 0) {
|
|
1688
|
+
pushed = id;
|
|
1689
|
+
writeFlat(ctx, ctx.nodes, index, record, id, (key) => key !== idKey);
|
|
1690
|
+
}
|
|
1691
|
+
}
|
|
1692
|
+
}
|
|
1693
|
+
owners.push(pushed);
|
|
1694
|
+
}
|
|
1695
|
+
throwIfAborted(ctx.options.signal);
|
|
1696
|
+
const kind = ctx.uniformKind();
|
|
1697
|
+
const { weightFrom } = ctx.options;
|
|
1698
|
+
// undirected: entries read once whose mirror is still to come, by pair and key
|
|
1699
|
+
const pending = new Map<string, number>();
|
|
1700
|
+
for (let i = 0; i < lists.length; i++) {
|
|
1701
|
+
const element = `adjacency[${i}]`;
|
|
1702
|
+
const list = lists[i];
|
|
1703
|
+
if (!Array.isArray(list)) {
|
|
1704
|
+
report.error("validation-error", JSON_ISSUE.BAD_ELEMENT, `${element} is not an array`, { element });
|
|
1705
|
+
continue;
|
|
1706
|
+
}
|
|
1707
|
+
const source = i < owners.length ? owners[i] : null;
|
|
1708
|
+
if (source === null) {
|
|
1709
|
+
const why = i < nodes.length ? `nodes[${i}] was skipped` : `there is no nodes[${i}]`;
|
|
1710
|
+
report.error(
|
|
1711
|
+
"missing-value",
|
|
1712
|
+
JSON_ISSUE.BAD_INDEX,
|
|
1713
|
+
`${element}: ${why}; its ${list.length} edge(s) are skipped`,
|
|
1714
|
+
{
|
|
1715
|
+
element,
|
|
1716
|
+
},
|
|
1717
|
+
);
|
|
1718
|
+
report.counts.skippedEdges += list.length;
|
|
1719
|
+
continue;
|
|
1720
|
+
}
|
|
1721
|
+
for (let j = 0; j < list.length; j++) {
|
|
1722
|
+
const entry = `${element}[${j}]`;
|
|
1723
|
+
const record = list[j];
|
|
1724
|
+
if (!isJsonObject(record)) {
|
|
1725
|
+
ctx.badElement("edge", entry);
|
|
1726
|
+
continue;
|
|
1727
|
+
}
|
|
1728
|
+
if (!hasKey(record, idKey)) {
|
|
1729
|
+
ctx.missingEndpoint(entry, idKey);
|
|
1730
|
+
continue;
|
|
1731
|
+
}
|
|
1732
|
+
try {
|
|
1733
|
+
const target = ctx.coerceId(record[idKey], `${entry}.${idKey}`);
|
|
1734
|
+
if (target === null) {
|
|
1735
|
+
ctx.countSkipped("edge");
|
|
1736
|
+
continue;
|
|
1737
|
+
}
|
|
1738
|
+
if (!directed && isMirroredEntry(pending, source, target, record.key)) {
|
|
1739
|
+
continue;
|
|
1740
|
+
}
|
|
1741
|
+
const edge = ctx.pushEdge(source, target, kind, ctx.weightOf(record), entry);
|
|
1742
|
+
for (const key of Object.keys(record)) {
|
|
1743
|
+
if (key !== idKey && key !== weightFrom) {
|
|
1744
|
+
ctx.edges.write(edge, key, record[key], SUFFIX.data);
|
|
1745
|
+
}
|
|
1746
|
+
}
|
|
1747
|
+
} catch (err) {
|
|
1748
|
+
ctx.skip(err, "edge", entry);
|
|
1749
|
+
}
|
|
1750
|
+
}
|
|
1751
|
+
}
|
|
1752
|
+
ctx.setMeta({}, { declaredMultigraph: multigraph, ...ctx.weightOriginPatch() });
|
|
1753
|
+
}
|
|
1754
|
+
|
|
1755
|
+
/**
|
|
1756
|
+
* Whether an undirected adjacency entry is the mirror of one already read (the same unordered
|
|
1757
|
+
* pair and key), consuming it; otherwise the entry is remembered as awaiting its mirror. A
|
|
1758
|
+
* self-loop is listed once and never awaits one.
|
|
1759
|
+
* @param pending - the entries awaiting their mirror, by owner, other end and key
|
|
1760
|
+
* @param source - the owner of the list
|
|
1761
|
+
* @param target - the entry's node
|
|
1762
|
+
* @param key - the entry's multigraph key, or undefined
|
|
1763
|
+
* @returns true when the entry is a mirror and must not be pushed
|
|
1764
|
+
*/
|
|
1765
|
+
function isMirroredEntry(pending: Map<string, number>, source: NodeId, target: NodeId, key: unknown): boolean {
|
|
1766
|
+
const a = JSON.stringify(source);
|
|
1767
|
+
const b = JSON.stringify(target);
|
|
1768
|
+
if (a === b) {
|
|
1769
|
+
return false;
|
|
1770
|
+
}
|
|
1771
|
+
const k = JSON.stringify(key ?? null);
|
|
1772
|
+
// an entry is the mirror of one listed by the other end, never of a parallel entry of its own list
|
|
1773
|
+
const mirror = `${b} ${a} ${k}`;
|
|
1774
|
+
const waiting = pending.get(mirror) ?? 0;
|
|
1775
|
+
if (waiting > 0) {
|
|
1776
|
+
pending.set(mirror, waiting - 1);
|
|
1777
|
+
return true;
|
|
1778
|
+
}
|
|
1779
|
+
const own = `${a} ${b} ${k}`;
|
|
1780
|
+
pending.set(own, (pending.get(own) ?? 0) + 1);
|
|
1781
|
+
return false;
|
|
1782
|
+
}
|
|
1783
|
+
|
|
1784
|
+
/**
|
|
1785
|
+
* Whether a value is a list of [string, value] pairs, the shape adjacency_data gives the graph dict.
|
|
1786
|
+
* @param value - the value
|
|
1787
|
+
* @returns true for an array whose items are all two-element arrays with a string first
|
|
1788
|
+
*/
|
|
1789
|
+
function isPairList(value: unknown): value is [string, unknown][] {
|
|
1790
|
+
return Array.isArray(value) && value.every((p) => Array.isArray(p) && p.length === 2 && typeof p[0] === "string");
|
|
1791
|
+
}
|
|
1792
|
+
|
|
1793
|
+
/**
|
|
1794
|
+
* Read a NetworkX tree_data document: a nested record with the node id, its attributes and a
|
|
1795
|
+
* `children` array of records of the same shape; every child gets a directed edge from its parent.
|
|
1796
|
+
* Walked depth first in document order with an explicit stack. A record without a usable id is
|
|
1797
|
+
* reported and skipped, and its children are still read (as roots, without an edge).
|
|
1798
|
+
* @param ctx - the context
|
|
1799
|
+
* @param root - the root record
|
|
1800
|
+
*/
|
|
1801
|
+
function importTree(ctx: ImportContext, root: JsonRecord): void {
|
|
1802
|
+
const idKey = ctx.json.nodeIdKey ?? "id";
|
|
1803
|
+
ctx.setHeader(ctx.defaultDirected("tree"));
|
|
1804
|
+
ctx.reportNodeIdFrom("tree", `the ${JSON.stringify(idKey)} key`);
|
|
1805
|
+
const kind = ctx.uniformKind();
|
|
1806
|
+
const stack: { record: unknown; element: string; parent: NodeId | null }[] = [
|
|
1807
|
+
{ record: root, element: "root", parent: null },
|
|
1808
|
+
];
|
|
1809
|
+
for (let item = stack.pop(); item !== undefined; item = stack.pop()) {
|
|
1810
|
+
const { record, element, parent } = item;
|
|
1811
|
+
if (!isJsonObject(record)) {
|
|
1812
|
+
ctx.badElement("node", element);
|
|
1813
|
+
continue;
|
|
1814
|
+
}
|
|
1815
|
+
let id = ctx.coerceId(hasKey(record, idKey) ? record[idKey] : undefined, element);
|
|
1816
|
+
if (id === null) {
|
|
1817
|
+
ctx.countSkipped("node");
|
|
1818
|
+
} else if (ctx.pushNode(id, element) < 0) {
|
|
1819
|
+
id = null;
|
|
1820
|
+
} else {
|
|
1821
|
+
writeFlat(ctx, ctx.nodes, ctx.sink.indexOf(id), record, id, (key) => key !== idKey && key !== "children");
|
|
1822
|
+
if (parent !== null) {
|
|
1823
|
+
try {
|
|
1824
|
+
ctx.pushEdge(parent, id, kind, undefined, element);
|
|
1825
|
+
} catch (err) {
|
|
1826
|
+
ctx.skip(err, "edge", element);
|
|
1827
|
+
}
|
|
1828
|
+
}
|
|
1829
|
+
}
|
|
1830
|
+
const { children } = record;
|
|
1831
|
+
if (Array.isArray(children)) {
|
|
1832
|
+
for (let k = children.length - 1; k >= 0; k--) {
|
|
1833
|
+
stack.push({ record: children[k], element: `${element}.children[${k}]`, parent: id });
|
|
1834
|
+
}
|
|
1835
|
+
} else if (children !== undefined && children !== null) {
|
|
1836
|
+
ctx.report.error(
|
|
1837
|
+
"validation-error",
|
|
1838
|
+
JSON_ISSUE.BAD_VALUE,
|
|
1839
|
+
`${element}: children must be an array, found ${describe(children)}`,
|
|
1840
|
+
{ element },
|
|
1841
|
+
);
|
|
1842
|
+
}
|
|
1843
|
+
}
|
|
1844
|
+
ctx.setMeta({});
|
|
1845
|
+
}
|
|
1846
|
+
|
|
1463
1847
|
// ============================================================ graphology
|
|
1464
1848
|
|
|
1465
1849
|
/**
|
|
@@ -2308,7 +2692,7 @@ export const jsonImporter: GraphImporter<JsonImportOptions> = Object.freeze({
|
|
|
2308
2692
|
*/
|
|
2309
2693
|
function readGraph(
|
|
2310
2694
|
root: unknown,
|
|
2311
|
-
dialect:
|
|
2695
|
+
dialect: JsonImportDialect,
|
|
2312
2696
|
sink: GraphSink,
|
|
2313
2697
|
report: ImportReportBuilder,
|
|
2314
2698
|
resolved: ResolvedImportOptions,
|
|
@@ -2340,6 +2724,12 @@ function readGraph(
|
|
|
2340
2724
|
case "vis":
|
|
2341
2725
|
importVis(ctx, doc);
|
|
2342
2726
|
break;
|
|
2727
|
+
case "adjacency":
|
|
2728
|
+
importAdjacency(ctx, doc);
|
|
2729
|
+
break;
|
|
2730
|
+
case "tree":
|
|
2731
|
+
importTree(ctx, doc);
|
|
2732
|
+
break;
|
|
2343
2733
|
default: {
|
|
2344
2734
|
const name: string = dialect;
|
|
2345
2735
|
throw new GraphFormatError("E_UNSUPPORTED", `unknown dialect ${name}`, {
|
|
@@ -3,6 +3,12 @@
|
|
|
3
3
|
* objects, their option types, the dialect names and the issue / loss codes they use.
|
|
4
4
|
*/
|
|
5
5
|
|
|
6
|
-
export {
|
|
6
|
+
export {
|
|
7
|
+
dialectCapabilities,
|
|
8
|
+
JSON_DIALECTS,
|
|
9
|
+
type JsonDialect,
|
|
10
|
+
type JsonImportDialect,
|
|
11
|
+
type JsonShapeMeta,
|
|
12
|
+
} from "./dialect.js";
|
|
7
13
|
export { JSON_LOSS, jsonCapabilities, jsonExporter, type JsonExportOptions } from "./exporter.js";
|
|
8
14
|
export { JSON_ISSUE, jsonImporter, type JsonImportOptions } from "./importer.js";
|
|
@@ -16,7 +16,9 @@
|
|
|
16
16
|
* `type[]`), and derive it from the dtype otherwise. Temporal columns write their `.text`
|
|
17
17
|
* companion when set and the canonical ISO form otherwise. Nodes are grouped into sections by
|
|
18
18
|
* (id space, stored-id column) in index order, so a re-import restores the same node order;
|
|
19
|
-
* relationships likewise by (start space, end space).
|
|
19
|
+
* relationships likewise by (start space, end space). A node of an id space is written under its
|
|
20
|
+
* `originalId` value (the id text the importer stored it under as `Space:id`), so a round trip
|
|
21
|
+
* reproduces the source file.
|
|
20
22
|
*/
|
|
21
23
|
|
|
22
24
|
import {
|
|
@@ -291,6 +293,9 @@ class ExportPlan {
|
|
|
291
293
|
|
|
292
294
|
readonly idSpace: Column | null;
|
|
293
295
|
|
|
296
|
+
/** The id text of the nodes of an id space (the importer's `originalId` column), or null. */
|
|
297
|
+
readonly originalId: Column | null;
|
|
298
|
+
|
|
294
299
|
readonly labels: Column | null;
|
|
295
300
|
|
|
296
301
|
readonly kind: Column | null;
|
|
@@ -311,6 +316,7 @@ class ExportPlan {
|
|
|
311
316
|
this.options = options;
|
|
312
317
|
this.common = common;
|
|
313
318
|
this.idSpace = snapshot.nodes.byRole("idSpace");
|
|
319
|
+
this.originalId = [...snapshot.nodes].find((column) => isIdText(column.meta)) ?? null;
|
|
314
320
|
this.labels = snapshot.nodes.byRole("labels");
|
|
315
321
|
this.kind = snapshot.edges.byRole("kind");
|
|
316
322
|
for (const column of snapshot.nodes) {
|
|
@@ -438,7 +444,10 @@ class ExportPlan {
|
|
|
438
444
|
if (companions.has(column) || (role !== null && SKIPPED_ROLES.has(role))) {
|
|
439
445
|
continue;
|
|
440
446
|
}
|
|
441
|
-
if (
|
|
447
|
+
if (
|
|
448
|
+
domain === "node" &&
|
|
449
|
+
(isStoredId(meta) || column === this.idSpace || column === this.labels || column === this.originalId)
|
|
450
|
+
) {
|
|
442
451
|
continue;
|
|
443
452
|
}
|
|
444
453
|
if (domain === "edge" && (role === "id" || column === this.kind)) {
|
|
@@ -833,6 +842,20 @@ class ExportPlan {
|
|
|
833
842
|
return typeof value === "string" ? value : String(value);
|
|
834
843
|
}
|
|
835
844
|
|
|
845
|
+
/**
|
|
846
|
+
* The `:ID` / `:START_ID` / `:END_ID` text of a node: its originalId value when it has an id
|
|
847
|
+
* space (the importer stored it as `Space:id`), its id otherwise.
|
|
848
|
+
* @param index - the node index
|
|
849
|
+
* @returns the id text
|
|
850
|
+
*/
|
|
851
|
+
idTextOf(index: number): string {
|
|
852
|
+
const { originalId } = this;
|
|
853
|
+
if (originalId !== null && originalId.isSet(index) && this.spaceOf(index) !== null) {
|
|
854
|
+
return textOf(originalId, index);
|
|
855
|
+
}
|
|
856
|
+
return idText(this.snapshot.ids.idOf(index));
|
|
857
|
+
}
|
|
858
|
+
|
|
836
859
|
/**
|
|
837
860
|
* The stored-id column of a node: the first one set in declaration order.
|
|
838
861
|
* @param index - the node index
|
|
@@ -872,7 +895,6 @@ class ExportPlan {
|
|
|
872
895
|
*/
|
|
873
896
|
private *nodeLines(): Generator<string, void, undefined> {
|
|
874
897
|
const { snapshot, labels, nodeColumns } = this;
|
|
875
|
-
const { ids } = snapshot;
|
|
876
898
|
const { delimiter } = this.options.syntax;
|
|
877
899
|
const propertyHeaders = nodeColumns.map((plan) => plan.header).join(delimiter);
|
|
878
900
|
let sections = 0;
|
|
@@ -897,7 +919,7 @@ class ExportPlan {
|
|
|
897
919
|
yield `${header.join(delimiter)}\n`;
|
|
898
920
|
}
|
|
899
921
|
cells.length = 0;
|
|
900
|
-
cells.push(this.cell(
|
|
922
|
+
cells.push(this.cell(this.idTextOf(i)));
|
|
901
923
|
if (labels !== null) {
|
|
902
924
|
cells.push(this.cell(this.labelsText(i)));
|
|
903
925
|
}
|
|
@@ -926,7 +948,6 @@ class ExportPlan {
|
|
|
926
948
|
*/
|
|
927
949
|
private *relationshipLines(): Generator<string, void, undefined> {
|
|
928
950
|
const { snapshot, kind, weights, edgeColumns, folding } = this;
|
|
929
|
-
const { ids } = snapshot;
|
|
930
951
|
const { delimiter } = this.options.syntax;
|
|
931
952
|
const list = snapshot.edgeList();
|
|
932
953
|
const propertyHeaders = edgeColumns.map((plan) => plan.header).join(delimiter);
|
|
@@ -964,7 +985,7 @@ class ExportPlan {
|
|
|
964
985
|
yield headerOf(startSpace, endSpace);
|
|
965
986
|
}
|
|
966
987
|
cells.length = 0;
|
|
967
|
-
cells.push(this.cell(
|
|
988
|
+
cells.push(this.cell(this.idTextOf(u)), this.cell(this.idTextOf(v)));
|
|
968
989
|
if (kind !== null) {
|
|
969
990
|
cells.push(this.cell(kind.isSet(e) ? textOf(kind, e) : null));
|
|
970
991
|
}
|
|
@@ -1048,6 +1069,22 @@ class ExportPlan {
|
|
|
1048
1069
|
}
|
|
1049
1070
|
}
|
|
1050
1071
|
|
|
1072
|
+
/**
|
|
1073
|
+
* Whether a node column is the importer's id text of the nodes of an id space (`originalId`).
|
|
1074
|
+
* @param meta - the column metadata
|
|
1075
|
+
* @returns true for the role-less string column the Neo4j importer declares with origin id `:ID` and no type
|
|
1076
|
+
*/
|
|
1077
|
+
function isIdText(meta: ColumnMeta): boolean {
|
|
1078
|
+
return (
|
|
1079
|
+
meta.role === null &&
|
|
1080
|
+
meta.dtype === "string" &&
|
|
1081
|
+
meta.origin !== null &&
|
|
1082
|
+
meta.origin.format === NEO4J &&
|
|
1083
|
+
meta.origin.id === ":ID" &&
|
|
1084
|
+
meta.origin.type === null
|
|
1085
|
+
);
|
|
1086
|
+
}
|
|
1087
|
+
|
|
1051
1088
|
/**
|
|
1052
1089
|
* Whether a node column is a stored id (`name:ID` on import).
|
|
1053
1090
|
* @param meta - the column metadata
|