@graphty/graph-io 0.3.3 → 0.3.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (42) hide show
  1. package/README.md +17 -4
  2. package/dist/chunks/{importer-BELrICWM.js → importer-BOxwmef_.js} +352 -19
  3. package/dist/chunks/importer-BOxwmef_.js.map +1 -0
  4. package/dist/graph-io.js +2 -2
  5. package/dist/graph-io.js.map +1 -1
  6. package/dist/json.js +1 -1
  7. package/dist/src/formats/json/dialect.d.ts +21 -4
  8. package/dist/src/formats/json/dialect.d.ts.map +1 -1
  9. package/dist/src/formats/json/dialect.js +29 -4
  10. package/dist/src/formats/json/dialect.js.map +1 -1
  11. package/dist/src/formats/json/exporter.d.ts.map +1 -1
  12. package/dist/src/formats/json/exporter.js +27 -7
  13. package/dist/src/formats/json/exporter.js.map +1 -1
  14. package/dist/src/formats/json/importer.d.ts +17 -4
  15. package/dist/src/formats/json/importer.d.ts.map +1 -1
  16. package/dist/src/formats/json/importer.js +356 -8
  17. package/dist/src/formats/json/importer.js.map +1 -1
  18. package/dist/src/formats/json/index.d.ts +1 -1
  19. package/dist/src/formats/json/index.d.ts.map +1 -1
  20. package/dist/src/formats/json/index.js +1 -1
  21. package/dist/src/formats/json/index.js.map +1 -1
  22. package/dist/src/index.d.ts +1 -1
  23. package/dist/src/index.d.ts.map +1 -1
  24. package/dist/src/index.js.map +1 -1
  25. package/dist/src/sniff.d.ts +3 -3
  26. package/dist/src/sniff.d.ts.map +1 -1
  27. package/dist/src/sniff.js.map +1 -1
  28. package/package.json +1 -1
  29. package/src/common/escape.ts +5 -4
  30. package/src/formats/json/dialect.ts +40 -6
  31. package/src/formats/json/exporter.ts +28 -7
  32. package/src/formats/json/importer.ts +411 -21
  33. package/src/formats/json/index.ts +7 -1
  34. package/src/formats/neo4j/exporter.ts +43 -6
  35. package/src/formats/neo4j/importer.ts +56 -11
  36. package/src/formats/neo4j/index.ts +10 -3
  37. package/src/formats/pajek/exporter.ts +3 -1
  38. package/src/formats/pajek/importer.ts +284 -60
  39. package/src/formats/pajek/syntax.ts +98 -12
  40. package/src/index.ts +2 -0
  41. package/src/sniff.ts +3 -3
  42. package/dist/chunks/importer-BELrICWM.js.map +0 -1
@@ -2,9 +2,10 @@
2
2
  * The lexical layer of the Pajek NET format shared by the importer and the exporter (research
3
3
  * note 07 section 2.5): the line tokenizer (whitespace-separated tokens, double quotes group
4
4
  * spaces and are removed, no escape mechanism), the section headers (`*Vertices N [N1]`,
5
- * `*Arcs [:k ["name"]]`, `*Edges`, `*Arcslist`, `*Edgeslist`, `*Matrix`, `*Network name` and the
6
- * project-file sections the importer does not read), the vertex shape keywords, the time interval
7
- * tokens `[1-5,7-*]` that map to the spells role, and the column names both sides agree on.
5
+ * `*Arcs [:k ["name"]]`, `*Edges`, `*Arcslist`, `*Edgeslist`, `*Matrix`, `*Network name`,
6
+ * `*Partition name`, `*Vector name` and the project-file sections the importer does not read), the
7
+ * vertex shape keywords, the time interval tokens `[1-5,7-*]` that map to the spells role, the
8
+ * `&#dddd;` character references of labels, and the column names both sides agree on.
8
9
  */
9
10
 
10
11
  /** The node column holding the vertex label (role label). */
@@ -28,11 +29,49 @@ export const VALUE_COLUMN = "value";
28
29
  /** The `weightFrom` default: Pajek's third column is the line value (design section 8.4). */
29
30
  export const VALUE_FIELD = "value";
30
31
 
31
- /** The vertex shape keywords of the Pajek manual. */
32
- export const SHAPES: ReadonlySet<string> = new Set(["ellipse", "box", "diamond", "triangle", "cross", "empty"]);
32
+ /** The node column holding the values of a `*Partition` object (i32). */
33
+ export const PARTITION_COLUMN = "partition";
34
+
35
+ /** The node column holding the values of a `*Vector` object (f64). */
36
+ export const VECTOR_COLUMN = "vector";
37
+
38
+ /**
39
+ * The vertex shape keywords of the Pajek manual, lower-cased; a file may write them in any case
40
+ * (use isShapeKeyword()).
41
+ */
42
+ export const SHAPES: ReadonlySet<string> = new Set([
43
+ "ellipse",
44
+ "box",
45
+ "diamond",
46
+ "triangle",
47
+ "cross",
48
+ "empty",
49
+ "house",
50
+ "man",
51
+ "woman",
52
+ ]);
53
+
54
+ /**
55
+ * Whether a token is a vertex shape keyword, in any case (`Ellipse`, `BOX`).
56
+ * @param token - the token
57
+ * @returns true for a shape keyword
58
+ */
59
+ export function isShapeKeyword(token: string): boolean {
60
+ return SHAPES.has(token.toLowerCase());
61
+ }
33
62
 
34
63
  /** The section keywords the importer reads, lower-cased. */
35
- type SectionKind = "network" | "vertices" | "arcs" | "edges" | "arcslist" | "edgeslist" | "matrix" | "unsupported";
64
+ type SectionKind =
65
+ | "network"
66
+ | "vertices"
67
+ | "arcs"
68
+ | "edges"
69
+ | "arcslist"
70
+ | "edgeslist"
71
+ | "matrix"
72
+ | "partition"
73
+ | "vector"
74
+ | "unsupported";
36
75
 
37
76
  /**
38
77
  * The parameter key the exporter writes a node's original id under when `sanitizeIds: "mangle"`
@@ -67,6 +106,8 @@ const SECTION_KINDS: ReadonlyMap<string, SectionKind> = new Map([
67
106
  ["arcslist", "arcslist"],
68
107
  ["edgeslist", "edgeslist"],
69
108
  ["matrix", "matrix"],
109
+ ["partition", "partition"],
110
+ ["vector", "vector"],
70
111
  ]);
71
112
 
72
113
  const INTEGER_TEXT = /^[+-]?[0-9]+$/;
@@ -76,7 +117,9 @@ const TIME_POINT =
76
117
  /**
77
118
  * Split one line into tokens: runs of non-whitespace, with double quotes grouping whitespace
78
119
  * into one token and removed from it (the shlex rule NetworkX applies; Pajek has no escapes, so a
79
- * quote never appears inside a token). An empty quoted string `""` is one empty token.
120
+ * quote never appears inside a token). An empty quoted string `""` is one empty token. A token
121
+ * that starts with `[` runs to the next `]` whatever whitespace it holds (when no other `[` comes
122
+ * first), so a time set written `[ 1, 3 ]` is one token.
80
123
  * @param line - the line without its terminator
81
124
  * @returns the tokens, or null when a quote is not closed before the end of the line
82
125
  */
@@ -92,6 +135,15 @@ export function tokenize(line: string): string[] | null {
92
135
  started = true;
93
136
  continue;
94
137
  }
138
+ if (!quoted && !started && c === 91) {
139
+ const close = line.indexOf("]", i);
140
+ if (close > i && line.lastIndexOf("[", close) === i) {
141
+ current += line.slice(i, close + 1);
142
+ started = true;
143
+ i = close;
144
+ continue;
145
+ }
146
+ }
95
147
  if (!quoted && (c === 32 || c === 9 || c === 13 || c === 12 || c === 11)) {
96
148
  if (started) {
97
149
  tokens.push(current);
@@ -190,6 +242,8 @@ export function parseSectionHeader(line: string): SectionHeader | null {
190
242
  }
191
243
  break;
192
244
  case "network":
245
+ case "partition":
246
+ case "vector":
193
247
  if (i < tokens.length) {
194
248
  name = tokens.slice(i).join(" ");
195
249
  i = tokens.length;
@@ -238,18 +292,19 @@ export function isIntervalToken(token: string): boolean {
238
292
 
239
293
  /**
240
294
  * Parse a Pajek time interval token into spells: `[1-5,7-*]` is `[[1, 5], [7, Infinity]]`, a
241
- * single time point `[3]` is `[[3, 3]]`, and `*` at either end is the corresponding infinity.
295
+ * single time point `[3]` is `[[3, 3]]`, `*` at either end is the corresponding infinity, blanks
296
+ * around the parts are ignored and an empty `[]` is no spell at all.
242
297
  * @param token - a token isIntervalToken() accepted
243
- * @returns the spells as [start, end] pairs
298
+ * @returns the spells as [start, end] pairs (empty for `[]`)
244
299
  */
245
300
  export function parseIntervals(token: string): [number, number][] {
246
301
  const body = token.slice(1, -1);
247
302
  if (body.trim().length === 0) {
248
- throw new Error(`empty time interval ${token}`);
303
+ return [];
249
304
  }
250
305
  const spells: [number, number][] = [];
251
306
  for (const part of body.split(",")) {
252
- const match = TIME_POINT.exec(part.trim());
307
+ const match = TIME_POINT.exec(part.replace(/\s+/g, ""));
253
308
  if (match === null) {
254
309
  throw new Error(`malformed time interval ${token}: "${part}" is not a-b, a-* or a`);
255
310
  }
@@ -271,6 +326,37 @@ export function parseIntervals(token: string): [number, number][] {
271
326
  return spells;
272
327
  }
273
328
 
329
+ const CHARACTER_REFERENCE = /&#(?:[xX]([0-9a-fA-F]{1,6})|([0-9]{1,7}));/g;
330
+
331
+ /**
332
+ * Decode the `&#dddd;` and `&#xhhhh;` character references of a label, as Pajek does; a reference
333
+ * outside the Unicode range is left as written.
334
+ * @param text - the label as written
335
+ * @returns the decoded label
336
+ */
337
+ export function decodeCharacterReferences(text: string): string {
338
+ if (!text.includes("&#")) {
339
+ return text;
340
+ }
341
+ return text.replace(CHARACTER_REFERENCE, (whole, hex: string | undefined, dec: string | undefined) => {
342
+ const code = hex === undefined ? Number(dec) : Number.parseInt(hex, 16);
343
+ return code <= 0x10ffff ? String.fromCodePoint(code) : whole;
344
+ });
345
+ }
346
+
347
+ /**
348
+ * Protect a label whose text would read back as a character reference: the `&` of every
349
+ * `&#...;` run is written as `&#38;`, the inverse of decodeCharacterReferences().
350
+ * @param text - the label text
351
+ * @returns the text to write
352
+ */
353
+ export function encodeCharacterReferences(text: string): string {
354
+ if (!text.includes("&#")) {
355
+ return text;
356
+ }
357
+ return text.replace(CHARACTER_REFERENCE, (whole) => `&#38;${whole.slice(1)}`);
358
+ }
359
+
274
360
  /**
275
361
  * Write spells as a Pajek time interval token, the inverse of parseIntervals().
276
362
  * @param spells - [start, end] pairs
@@ -300,7 +386,7 @@ export function isParameterKey(text: string): boolean {
300
386
  if (text.length === 0 || /[\s"]/.test(text) || text.startsWith("[") || text.startsWith("*")) {
301
387
  return false;
302
388
  }
303
- if (SHAPES.has(text)) {
389
+ if (isShapeKeyword(text)) {
304
390
  return false;
305
391
  }
306
392
  return !/^[+-]?(\.[0-9]|[0-9])/.test(text);
package/src/index.ts CHANGED
@@ -116,6 +116,7 @@ export {
116
116
  type JsonDialect,
117
117
  jsonExporter,
118
118
  type JsonExportOptions,
119
+ type JsonImportDialect,
119
120
  jsonImporter,
120
121
  type JsonImportOptions,
121
122
  type JsonShapeMeta,
@@ -130,6 +131,7 @@ export {
130
131
  type Neo4jExportOptions,
131
132
  neo4jImporter,
132
133
  type Neo4jImportOptions,
134
+ ORIGINAL_ID_COLUMN,
133
135
  TYPE_COLUMN,
134
136
  } from "./formats/neo4j/index.js";
135
137
  export {
package/src/sniff.ts CHANGED
@@ -21,7 +21,7 @@
21
21
  * (the importer's detection on the parsed document is authoritative, design section 8.2).
22
22
  */
23
23
 
24
- import { type JsonDialect, sniffJsonDialect } from "./formats/json/dialect.js";
24
+ import { type JsonImportDialect, sniffJsonDialect } from "./formats/json/dialect.js";
25
25
  import { type GraphImporter } from "./types.js";
26
26
 
27
27
  /** The format names of the eight built-in importers and exporters. */
@@ -69,7 +69,7 @@ export interface SniffResult {
69
69
  /** Whether the MIME type is one the importer claims. */
70
70
  readonly mimeType: boolean;
71
71
  /** For the JSON format: the dialect the head suggests, or null when unknown; always null for other formats. */
72
- readonly dialect: JsonDialect | null;
72
+ readonly dialect: JsonImportDialect | null;
73
73
  }
74
74
 
75
75
  /**
@@ -178,7 +178,7 @@ export function sniffFormat(hints: SniffHints, importers: Iterable<GraphImporter
178
178
  * @param head - the first bytes or characters of the document
179
179
  * @returns the dialect, or null when the head is not a JSON graph document
180
180
  */
181
- export function sniffJsonDialectHead(head: Uint8Array | string): JsonDialect | null {
181
+ export function sniffJsonDialectHead(head: Uint8Array | string): JsonImportDialect | null {
182
182
  const text = typeof head === "string" ? head : new TextDecoder("utf-8", { fatal: false }).decode(head);
183
183
  const body = text.charCodeAt(0) === 0xfeff ? text.slice(1) : text;
184
184
  const trimmed = body.trimStart();