@graphty/graph-io 0.3.3 → 0.3.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +17 -4
- package/dist/chunks/{importer-BELrICWM.js → importer-BOxwmef_.js} +352 -19
- package/dist/chunks/importer-BOxwmef_.js.map +1 -0
- package/dist/graph-io.js +2 -2
- package/dist/graph-io.js.map +1 -1
- package/dist/json.js +1 -1
- package/dist/src/formats/json/dialect.d.ts +21 -4
- package/dist/src/formats/json/dialect.d.ts.map +1 -1
- package/dist/src/formats/json/dialect.js +29 -4
- package/dist/src/formats/json/dialect.js.map +1 -1
- package/dist/src/formats/json/exporter.d.ts.map +1 -1
- package/dist/src/formats/json/exporter.js +27 -7
- package/dist/src/formats/json/exporter.js.map +1 -1
- package/dist/src/formats/json/importer.d.ts +17 -4
- package/dist/src/formats/json/importer.d.ts.map +1 -1
- package/dist/src/formats/json/importer.js +356 -8
- package/dist/src/formats/json/importer.js.map +1 -1
- package/dist/src/formats/json/index.d.ts +1 -1
- package/dist/src/formats/json/index.d.ts.map +1 -1
- package/dist/src/formats/json/index.js +1 -1
- package/dist/src/formats/json/index.js.map +1 -1
- package/dist/src/index.d.ts +1 -1
- package/dist/src/index.d.ts.map +1 -1
- package/dist/src/index.js.map +1 -1
- package/dist/src/sniff.d.ts +3 -3
- package/dist/src/sniff.d.ts.map +1 -1
- package/dist/src/sniff.js.map +1 -1
- package/package.json +1 -1
- package/src/common/escape.ts +5 -4
- package/src/formats/json/dialect.ts +40 -6
- package/src/formats/json/exporter.ts +28 -7
- package/src/formats/json/importer.ts +411 -21
- package/src/formats/json/index.ts +7 -1
- package/src/formats/neo4j/exporter.ts +43 -6
- package/src/formats/neo4j/importer.ts +56 -11
- package/src/formats/neo4j/index.ts +10 -3
- package/src/formats/pajek/exporter.ts +3 -1
- package/src/formats/pajek/importer.ts +284 -60
- package/src/formats/pajek/syntax.ts +98 -12
- package/src/index.ts +2 -0
- package/src/sniff.ts +3 -3
- package/dist/chunks/importer-BELrICWM.js.map +0 -1
|
@@ -2,9 +2,10 @@
|
|
|
2
2
|
* The lexical layer of the Pajek NET format shared by the importer and the exporter (research
|
|
3
3
|
* note 07 section 2.5): the line tokenizer (whitespace-separated tokens, double quotes group
|
|
4
4
|
* spaces and are removed, no escape mechanism), the section headers (`*Vertices N [N1]`,
|
|
5
|
-
* `*Arcs [:k ["name"]]`, `*Edges`, `*Arcslist`, `*Edgeslist`, `*Matrix`, `*Network name
|
|
6
|
-
* project-file sections the importer does not read), the
|
|
7
|
-
* tokens `[1-5,7-*]` that map to the spells role,
|
|
5
|
+
* `*Arcs [:k ["name"]]`, `*Edges`, `*Arcslist`, `*Edgeslist`, `*Matrix`, `*Network name`,
|
|
6
|
+
* `*Partition name`, `*Vector name` and the project-file sections the importer does not read), the
|
|
7
|
+
* vertex shape keywords, the time interval tokens `[1-5,7-*]` that map to the spells role, the
|
|
8
|
+
* `&#dddd;` character references of labels, and the column names both sides agree on.
|
|
8
9
|
*/
|
|
9
10
|
|
|
10
11
|
/** The node column holding the vertex label (role label). */
|
|
@@ -28,11 +29,49 @@ export const VALUE_COLUMN = "value";
|
|
|
28
29
|
/** The `weightFrom` default: Pajek's third column is the line value (design section 8.4). */
|
|
29
30
|
export const VALUE_FIELD = "value";
|
|
30
31
|
|
|
31
|
-
/** The
|
|
32
|
-
export const
|
|
32
|
+
/** The node column holding the values of a `*Partition` object (i32). */
|
|
33
|
+
export const PARTITION_COLUMN = "partition";
|
|
34
|
+
|
|
35
|
+
/** The node column holding the values of a `*Vector` object (f64). */
|
|
36
|
+
export const VECTOR_COLUMN = "vector";
|
|
37
|
+
|
|
38
|
+
/**
|
|
39
|
+
* The vertex shape keywords of the Pajek manual, lower-cased; a file may write them in any case
|
|
40
|
+
* (use isShapeKeyword()).
|
|
41
|
+
*/
|
|
42
|
+
export const SHAPES: ReadonlySet<string> = new Set([
|
|
43
|
+
"ellipse",
|
|
44
|
+
"box",
|
|
45
|
+
"diamond",
|
|
46
|
+
"triangle",
|
|
47
|
+
"cross",
|
|
48
|
+
"empty",
|
|
49
|
+
"house",
|
|
50
|
+
"man",
|
|
51
|
+
"woman",
|
|
52
|
+
]);
|
|
53
|
+
|
|
54
|
+
/**
|
|
55
|
+
* Whether a token is a vertex shape keyword, in any case (`Ellipse`, `BOX`).
|
|
56
|
+
* @param token - the token
|
|
57
|
+
* @returns true for a shape keyword
|
|
58
|
+
*/
|
|
59
|
+
export function isShapeKeyword(token: string): boolean {
|
|
60
|
+
return SHAPES.has(token.toLowerCase());
|
|
61
|
+
}
|
|
33
62
|
|
|
34
63
|
/** The section keywords the importer reads, lower-cased. */
|
|
35
|
-
type SectionKind =
|
|
64
|
+
type SectionKind =
|
|
65
|
+
| "network"
|
|
66
|
+
| "vertices"
|
|
67
|
+
| "arcs"
|
|
68
|
+
| "edges"
|
|
69
|
+
| "arcslist"
|
|
70
|
+
| "edgeslist"
|
|
71
|
+
| "matrix"
|
|
72
|
+
| "partition"
|
|
73
|
+
| "vector"
|
|
74
|
+
| "unsupported";
|
|
36
75
|
|
|
37
76
|
/**
|
|
38
77
|
* The parameter key the exporter writes a node's original id under when `sanitizeIds: "mangle"`
|
|
@@ -67,6 +106,8 @@ const SECTION_KINDS: ReadonlyMap<string, SectionKind> = new Map([
|
|
|
67
106
|
["arcslist", "arcslist"],
|
|
68
107
|
["edgeslist", "edgeslist"],
|
|
69
108
|
["matrix", "matrix"],
|
|
109
|
+
["partition", "partition"],
|
|
110
|
+
["vector", "vector"],
|
|
70
111
|
]);
|
|
71
112
|
|
|
72
113
|
const INTEGER_TEXT = /^[+-]?[0-9]+$/;
|
|
@@ -76,7 +117,9 @@ const TIME_POINT =
|
|
|
76
117
|
/**
|
|
77
118
|
* Split one line into tokens: runs of non-whitespace, with double quotes grouping whitespace
|
|
78
119
|
* into one token and removed from it (the shlex rule NetworkX applies; Pajek has no escapes, so a
|
|
79
|
-
* quote never appears inside a token). An empty quoted string `""` is one empty token.
|
|
120
|
+
* quote never appears inside a token). An empty quoted string `""` is one empty token. A token
|
|
121
|
+
* that starts with `[` runs to the next `]` whatever whitespace it holds (when no other `[` comes
|
|
122
|
+
* first), so a time set written `[ 1, 3 ]` is one token.
|
|
80
123
|
* @param line - the line without its terminator
|
|
81
124
|
* @returns the tokens, or null when a quote is not closed before the end of the line
|
|
82
125
|
*/
|
|
@@ -92,6 +135,15 @@ export function tokenize(line: string): string[] | null {
|
|
|
92
135
|
started = true;
|
|
93
136
|
continue;
|
|
94
137
|
}
|
|
138
|
+
if (!quoted && !started && c === 91) {
|
|
139
|
+
const close = line.indexOf("]", i);
|
|
140
|
+
if (close > i && line.lastIndexOf("[", close) === i) {
|
|
141
|
+
current += line.slice(i, close + 1);
|
|
142
|
+
started = true;
|
|
143
|
+
i = close;
|
|
144
|
+
continue;
|
|
145
|
+
}
|
|
146
|
+
}
|
|
95
147
|
if (!quoted && (c === 32 || c === 9 || c === 13 || c === 12 || c === 11)) {
|
|
96
148
|
if (started) {
|
|
97
149
|
tokens.push(current);
|
|
@@ -190,6 +242,8 @@ export function parseSectionHeader(line: string): SectionHeader | null {
|
|
|
190
242
|
}
|
|
191
243
|
break;
|
|
192
244
|
case "network":
|
|
245
|
+
case "partition":
|
|
246
|
+
case "vector":
|
|
193
247
|
if (i < tokens.length) {
|
|
194
248
|
name = tokens.slice(i).join(" ");
|
|
195
249
|
i = tokens.length;
|
|
@@ -238,18 +292,19 @@ export function isIntervalToken(token: string): boolean {
|
|
|
238
292
|
|
|
239
293
|
/**
|
|
240
294
|
* Parse a Pajek time interval token into spells: `[1-5,7-*]` is `[[1, 5], [7, Infinity]]`, a
|
|
241
|
-
* single time point `[3]` is `[[3, 3]]`,
|
|
295
|
+
* single time point `[3]` is `[[3, 3]]`, `*` at either end is the corresponding infinity, blanks
|
|
296
|
+
* around the parts are ignored and an empty `[]` is no spell at all.
|
|
242
297
|
* @param token - a token isIntervalToken() accepted
|
|
243
|
-
* @returns the spells as [start, end] pairs
|
|
298
|
+
* @returns the spells as [start, end] pairs (empty for `[]`)
|
|
244
299
|
*/
|
|
245
300
|
export function parseIntervals(token: string): [number, number][] {
|
|
246
301
|
const body = token.slice(1, -1);
|
|
247
302
|
if (body.trim().length === 0) {
|
|
248
|
-
|
|
303
|
+
return [];
|
|
249
304
|
}
|
|
250
305
|
const spells: [number, number][] = [];
|
|
251
306
|
for (const part of body.split(",")) {
|
|
252
|
-
const match = TIME_POINT.exec(part.
|
|
307
|
+
const match = TIME_POINT.exec(part.replace(/\s+/g, ""));
|
|
253
308
|
if (match === null) {
|
|
254
309
|
throw new Error(`malformed time interval ${token}: "${part}" is not a-b, a-* or a`);
|
|
255
310
|
}
|
|
@@ -271,6 +326,37 @@ export function parseIntervals(token: string): [number, number][] {
|
|
|
271
326
|
return spells;
|
|
272
327
|
}
|
|
273
328
|
|
|
329
|
+
const CHARACTER_REFERENCE = /&#(?:[xX]([0-9a-fA-F]{1,6})|([0-9]{1,7}));/g;
|
|
330
|
+
|
|
331
|
+
/**
|
|
332
|
+
* Decode the `&#dddd;` and `&#xhhhh;` character references of a label, as Pajek does; a reference
|
|
333
|
+
* outside the Unicode range is left as written.
|
|
334
|
+
* @param text - the label as written
|
|
335
|
+
* @returns the decoded label
|
|
336
|
+
*/
|
|
337
|
+
export function decodeCharacterReferences(text: string): string {
|
|
338
|
+
if (!text.includes("&#")) {
|
|
339
|
+
return text;
|
|
340
|
+
}
|
|
341
|
+
return text.replace(CHARACTER_REFERENCE, (whole, hex: string | undefined, dec: string | undefined) => {
|
|
342
|
+
const code = hex === undefined ? Number(dec) : Number.parseInt(hex, 16);
|
|
343
|
+
return code <= 0x10ffff ? String.fromCodePoint(code) : whole;
|
|
344
|
+
});
|
|
345
|
+
}
|
|
346
|
+
|
|
347
|
+
/**
|
|
348
|
+
* Protect a label whose text would read back as a character reference: the `&` of every
|
|
349
|
+
* `&#...;` run is written as `&`, the inverse of decodeCharacterReferences().
|
|
350
|
+
* @param text - the label text
|
|
351
|
+
* @returns the text to write
|
|
352
|
+
*/
|
|
353
|
+
export function encodeCharacterReferences(text: string): string {
|
|
354
|
+
if (!text.includes("&#")) {
|
|
355
|
+
return text;
|
|
356
|
+
}
|
|
357
|
+
return text.replace(CHARACTER_REFERENCE, (whole) => `&${whole.slice(1)}`);
|
|
358
|
+
}
|
|
359
|
+
|
|
274
360
|
/**
|
|
275
361
|
* Write spells as a Pajek time interval token, the inverse of parseIntervals().
|
|
276
362
|
* @param spells - [start, end] pairs
|
|
@@ -300,7 +386,7 @@ export function isParameterKey(text: string): boolean {
|
|
|
300
386
|
if (text.length === 0 || /[\s"]/.test(text) || text.startsWith("[") || text.startsWith("*")) {
|
|
301
387
|
return false;
|
|
302
388
|
}
|
|
303
|
-
if (
|
|
389
|
+
if (isShapeKeyword(text)) {
|
|
304
390
|
return false;
|
|
305
391
|
}
|
|
306
392
|
return !/^[+-]?(\.[0-9]|[0-9])/.test(text);
|
package/src/index.ts
CHANGED
|
@@ -116,6 +116,7 @@ export {
|
|
|
116
116
|
type JsonDialect,
|
|
117
117
|
jsonExporter,
|
|
118
118
|
type JsonExportOptions,
|
|
119
|
+
type JsonImportDialect,
|
|
119
120
|
jsonImporter,
|
|
120
121
|
type JsonImportOptions,
|
|
121
122
|
type JsonShapeMeta,
|
|
@@ -130,6 +131,7 @@ export {
|
|
|
130
131
|
type Neo4jExportOptions,
|
|
131
132
|
neo4jImporter,
|
|
132
133
|
type Neo4jImportOptions,
|
|
134
|
+
ORIGINAL_ID_COLUMN,
|
|
133
135
|
TYPE_COLUMN,
|
|
134
136
|
} from "./formats/neo4j/index.js";
|
|
135
137
|
export {
|
package/src/sniff.ts
CHANGED
|
@@ -21,7 +21,7 @@
|
|
|
21
21
|
* (the importer's detection on the parsed document is authoritative, design section 8.2).
|
|
22
22
|
*/
|
|
23
23
|
|
|
24
|
-
import { type
|
|
24
|
+
import { type JsonImportDialect, sniffJsonDialect } from "./formats/json/dialect.js";
|
|
25
25
|
import { type GraphImporter } from "./types.js";
|
|
26
26
|
|
|
27
27
|
/** The format names of the eight built-in importers and exporters. */
|
|
@@ -69,7 +69,7 @@ export interface SniffResult {
|
|
|
69
69
|
/** Whether the MIME type is one the importer claims. */
|
|
70
70
|
readonly mimeType: boolean;
|
|
71
71
|
/** For the JSON format: the dialect the head suggests, or null when unknown; always null for other formats. */
|
|
72
|
-
readonly dialect:
|
|
72
|
+
readonly dialect: JsonImportDialect | null;
|
|
73
73
|
}
|
|
74
74
|
|
|
75
75
|
/**
|
|
@@ -178,7 +178,7 @@ export function sniffFormat(hints: SniffHints, importers: Iterable<GraphImporter
|
|
|
178
178
|
* @param head - the first bytes or characters of the document
|
|
179
179
|
* @returns the dialect, or null when the head is not a JSON graph document
|
|
180
180
|
*/
|
|
181
|
-
export function sniffJsonDialectHead(head: Uint8Array | string):
|
|
181
|
+
export function sniffJsonDialectHead(head: Uint8Array | string): JsonImportDialect | null {
|
|
182
182
|
const text = typeof head === "string" ? head : new TextDecoder("utf-8", { fatal: false }).decode(head);
|
|
183
183
|
const body = text.charCodeAt(0) === 0xfeff ? text.slice(1) : text;
|
|
184
184
|
const trimmed = body.trimStart();
|