@graphty/graph-io 0.0.0 → 0.2.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +250 -28
- package/dist/chunks/children-CL3Cy0ez.js +238 -0
- package/dist/chunks/children-CL3Cy0ez.js.map +1 -0
- package/dist/chunks/escape-DyI8JofU.js +938 -0
- package/dist/chunks/escape-DyI8JofU.js.map +1 -0
- package/dist/chunks/importer-CQnJuWJw.js +2987 -0
- package/dist/chunks/importer-CQnJuWJw.js.map +1 -0
- package/dist/chunks/importer-CpCpfbxr.js +2015 -0
- package/dist/chunks/importer-CpCpfbxr.js.map +1 -0
- package/dist/chunks/importer-DbnGYr3_.js +2342 -0
- package/dist/chunks/importer-DbnGYr3_.js.map +1 -0
- package/dist/chunks/importer-GozH8DkN.js +3050 -0
- package/dist/chunks/importer-GozH8DkN.js.map +1 -0
- package/dist/chunks/records-CGpxszm1.js +605 -0
- package/dist/chunks/records-CGpxszm1.js.map +1 -0
- package/dist/chunks/text-CajMdVFy.js +189 -0
- package/dist/chunks/text-CajMdVFy.js.map +1 -0
- package/dist/chunks/writer-DxSKC7TL.js +2842 -0
- package/dist/chunks/writer-DxSKC7TL.js.map +1 -0
- package/dist/csv.d.ts +1 -0
- package/dist/csv.js +1702 -0
- package/dist/csv.js.map +1 -0
- package/dist/dot.d.ts +1 -0
- package/dist/dot.js +8 -0
- package/dist/dot.js.map +1 -0
- package/dist/gexf.d.ts +1 -0
- package/dist/gexf.js +3466 -0
- package/dist/gexf.js.map +1 -0
- package/dist/gml.d.ts +1 -0
- package/dist/gml.js +2647 -0
- package/dist/gml.js.map +1 -0
- package/dist/graph-io.d.ts +1 -0
- package/dist/graph-io.js +790 -0
- package/dist/graph-io.js.map +1 -0
- package/dist/graphml.d.ts +1 -0
- package/dist/graphml.js +8 -0
- package/dist/graphml.js.map +1 -0
- package/dist/json.d.ts +1 -0
- package/dist/json.js +11 -0
- package/dist/json.js.map +1 -0
- package/dist/neo4j.d.ts +1 -0
- package/dist/neo4j.js +2046 -0
- package/dist/neo4j.js.map +1 -0
- package/dist/pajek.d.ts +1 -0
- package/dist/pajek.js +8 -0
- package/dist/pajek.js.map +1 -0
- package/dist/src/children.d.ts +134 -0
- package/dist/src/children.d.ts.map +1 -0
- package/dist/src/children.js +274 -0
- package/dist/src/children.js.map +1 -0
- package/dist/src/common/attributes.d.ts +229 -0
- package/dist/src/common/attributes.d.ts.map +1 -0
- package/dist/src/common/attributes.js +368 -0
- package/dist/src/common/attributes.js.map +1 -0
- package/dist/src/common/codes.d.ts +105 -0
- package/dist/src/common/codes.d.ts.map +1 -0
- package/dist/src/common/codes.js +107 -0
- package/dist/src/common/codes.js.map +1 -0
- package/dist/src/common/declared-types.d.ts +84 -0
- package/dist/src/common/declared-types.d.ts.map +1 -0
- package/dist/src/common/declared-types.js +326 -0
- package/dist/src/common/declared-types.js.map +1 -0
- package/dist/src/common/direction.d.ts +206 -0
- package/dist/src/common/direction.d.ts.map +1 -0
- package/dist/src/common/direction.js +370 -0
- package/dist/src/common/direction.js.map +1 -0
- package/dist/src/common/escape.d.ts +92 -0
- package/dist/src/common/escape.d.ts.map +1 -0
- package/dist/src/common/escape.js +212 -0
- package/dist/src/common/escape.js.map +1 -0
- package/dist/src/common/export.d.ts +249 -0
- package/dist/src/common/export.d.ts.map +1 -0
- package/dist/src/common/export.js +594 -0
- package/dist/src/common/export.js.map +1 -0
- package/dist/src/common/format.d.ts +59 -0
- package/dist/src/common/format.d.ts.map +1 -0
- package/dist/src/common/format.js +106 -0
- package/dist/src/common/format.js.map +1 -0
- package/dist/src/common/ids.d.ts +83 -0
- package/dist/src/common/ids.d.ts.map +1 -0
- package/dist/src/common/ids.js +158 -0
- package/dist/src/common/ids.js.map +1 -0
- package/dist/src/common/input.d.ts +100 -0
- package/dist/src/common/input.d.ts.map +1 -0
- package/dist/src/common/input.js +335 -0
- package/dist/src/common/input.js.map +1 -0
- package/dist/src/common/lists.d.ts +34 -0
- package/dist/src/common/lists.d.ts.map +1 -0
- package/dist/src/common/lists.js +185 -0
- package/dist/src/common/lists.js.map +1 -0
- package/dist/src/common/options.d.ts +108 -0
- package/dist/src/common/options.d.ts.map +1 -0
- package/dist/src/common/options.js +265 -0
- package/dist/src/common/options.js.map +1 -0
- package/dist/src/common/report.d.ts +187 -0
- package/dist/src/common/report.d.ts.map +1 -0
- package/dist/src/common/report.js +274 -0
- package/dist/src/common/report.js.map +1 -0
- package/dist/src/common/temporal.d.ts +71 -0
- package/dist/src/common/temporal.d.ts.map +1 -0
- package/dist/src/common/temporal.js +266 -0
- package/dist/src/common/temporal.js.map +1 -0
- package/dist/src/common/text.d.ts +104 -0
- package/dist/src/common/text.d.ts.map +1 -0
- package/dist/src/common/text.js +255 -0
- package/dist/src/common/text.js.map +1 -0
- package/dist/src/common/weights.d.ts +77 -0
- package/dist/src/common/weights.d.ts.map +1 -0
- package/dist/src/common/weights.js +156 -0
- package/dist/src/common/weights.js.map +1 -0
- package/dist/src/common/writer.d.ts +51 -0
- package/dist/src/common/writer.d.ts.map +1 -0
- package/dist/src/common/writer.js +108 -0
- package/dist/src/common/writer.js.map +1 -0
- package/dist/src/common/xml.d.ts +245 -0
- package/dist/src/common/xml.d.ts.map +1 -0
- package/dist/src/common/xml.js +942 -0
- package/dist/src/common/xml.js.map +1 -0
- package/dist/src/formats/csv/exporter.d.ts +70 -0
- package/dist/src/formats/csv/exporter.d.ts.map +1 -0
- package/dist/src/formats/csv/exporter.js +682 -0
- package/dist/src/formats/csv/exporter.js.map +1 -0
- package/dist/src/formats/csv/header.d.ts +66 -0
- package/dist/src/formats/csv/header.d.ts.map +1 -0
- package/dist/src/formats/csv/header.js +152 -0
- package/dist/src/formats/csv/header.js.map +1 -0
- package/dist/src/formats/csv/importer.d.ts +82 -0
- package/dist/src/formats/csv/importer.d.ts.map +1 -0
- package/dist/src/formats/csv/importer.js +849 -0
- package/dist/src/formats/csv/importer.js.map +1 -0
- package/dist/src/formats/csv/index.d.ts +60 -0
- package/dist/src/formats/csv/index.d.ts.map +1 -0
- package/dist/src/formats/csv/index.js +63 -0
- package/dist/src/formats/csv/index.js.map +1 -0
- package/dist/src/formats/csv/records.d.ts +188 -0
- package/dist/src/formats/csv/records.d.ts.map +1 -0
- package/dist/src/formats/csv/records.js +702 -0
- package/dist/src/formats/csv/records.js.map +1 -0
- package/dist/src/formats/csv/values.d.ts +105 -0
- package/dist/src/formats/csv/values.d.ts.map +1 -0
- package/dist/src/formats/csv/values.js +192 -0
- package/dist/src/formats/csv/values.js.map +1 -0
- package/dist/src/formats/dot/exporter.d.ts +52 -0
- package/dist/src/formats/dot/exporter.d.ts.map +1 -0
- package/dist/src/formats/dot/exporter.js +836 -0
- package/dist/src/formats/dot/exporter.js.map +1 -0
- package/dist/src/formats/dot/importer.d.ts +102 -0
- package/dist/src/formats/dot/importer.d.ts.map +1 -0
- package/dist/src/formats/dot/importer.js +1291 -0
- package/dist/src/formats/dot/importer.js.map +1 -0
- package/dist/src/formats/dot/index.d.ts +7 -0
- package/dist/src/formats/dot/index.d.ts.map +1 -0
- package/dist/src/formats/dot/index.js +7 -0
- package/dist/src/formats/dot/index.js.map +1 -0
- package/dist/src/formats/dot/names.d.ts +29 -0
- package/dist/src/formats/dot/names.d.ts.map +1 -0
- package/dist/src/formats/dot/names.js +28 -0
- package/dist/src/formats/dot/names.js.map +1 -0
- package/dist/src/formats/dot/tokenizer.d.ts +114 -0
- package/dist/src/formats/dot/tokenizer.d.ts.map +1 -0
- package/dist/src/formats/dot/tokenizer.js +341 -0
- package/dist/src/formats/dot/tokenizer.js.map +1 -0
- package/dist/src/formats/gexf/exporter.d.ts +56 -0
- package/dist/src/formats/gexf/exporter.d.ts.map +1 -0
- package/dist/src/formats/gexf/exporter.js +1395 -0
- package/dist/src/formats/gexf/exporter.js.map +1 -0
- package/dist/src/formats/gexf/importer.d.ts +73 -0
- package/dist/src/formats/gexf/importer.d.ts.map +1 -0
- package/dist/src/formats/gexf/importer.js +1880 -0
- package/dist/src/formats/gexf/importer.js.map +1 -0
- package/dist/src/formats/gexf/index.d.ts +96 -0
- package/dist/src/formats/gexf/index.d.ts.map +1 -0
- package/dist/src/formats/gexf/index.js +97 -0
- package/dist/src/formats/gexf/index.js.map +1 -0
- package/dist/src/formats/gexf/schema.d.ts +135 -0
- package/dist/src/formats/gexf/schema.d.ts.map +1 -0
- package/dist/src/formats/gexf/schema.js +323 -0
- package/dist/src/formats/gexf/schema.js.map +1 -0
- package/dist/src/formats/gml/exporter.d.ts +69 -0
- package/dist/src/formats/gml/exporter.d.ts.map +1 -0
- package/dist/src/formats/gml/exporter.js +1093 -0
- package/dist/src/formats/gml/exporter.js.map +1 -0
- package/dist/src/formats/gml/importer.d.ts +66 -0
- package/dist/src/formats/gml/importer.d.ts.map +1 -0
- package/dist/src/formats/gml/importer.js +1331 -0
- package/dist/src/formats/gml/importer.js.map +1 -0
- package/dist/src/formats/gml/index.d.ts +85 -0
- package/dist/src/formats/gml/index.d.ts.map +1 -0
- package/dist/src/formats/gml/index.js +88 -0
- package/dist/src/formats/gml/index.js.map +1 -0
- package/dist/src/formats/gml/syntax.d.ts +186 -0
- package/dist/src/formats/gml/syntax.d.ts.map +1 -0
- package/dist/src/formats/gml/syntax.js +467 -0
- package/dist/src/formats/gml/syntax.js.map +1 -0
- package/dist/src/formats/graphml/constants.d.ts +169 -0
- package/dist/src/formats/graphml/constants.d.ts.map +1 -0
- package/dist/src/formats/graphml/constants.js +165 -0
- package/dist/src/formats/graphml/constants.js.map +1 -0
- package/dist/src/formats/graphml/exporter.d.ts +34 -0
- package/dist/src/formats/graphml/exporter.d.ts.map +1 -0
- package/dist/src/formats/graphml/exporter.js +1176 -0
- package/dist/src/formats/graphml/exporter.js.map +1 -0
- package/dist/src/formats/graphml/importer.d.ts +31 -0
- package/dist/src/formats/graphml/importer.d.ts.map +1 -0
- package/dist/src/formats/graphml/importer.js +1607 -0
- package/dist/src/formats/graphml/importer.js.map +1 -0
- package/dist/src/formats/graphml/index.d.ts +8 -0
- package/dist/src/formats/graphml/index.d.ts.map +1 -0
- package/dist/src/formats/graphml/index.js +8 -0
- package/dist/src/formats/graphml/index.js.map +1 -0
- package/dist/src/formats/graphml/tree.d.ts +72 -0
- package/dist/src/formats/graphml/tree.d.ts.map +1 -0
- package/dist/src/formats/graphml/tree.js +290 -0
- package/dist/src/formats/graphml/tree.js.map +1 -0
- package/dist/src/formats/json/dialect.d.ts +125 -0
- package/dist/src/formats/json/dialect.d.ts.map +1 -0
- package/dist/src/formats/json/dialect.js +262 -0
- package/dist/src/formats/json/dialect.js.map +1 -0
- package/dist/src/formats/json/exporter.d.ts +89 -0
- package/dist/src/formats/json/exporter.d.ts.map +1 -0
- package/dist/src/formats/json/exporter.js +1358 -0
- package/dist/src/formats/json/exporter.js.map +1 -0
- package/dist/src/formats/json/importer.d.ts +108 -0
- package/dist/src/formats/json/importer.d.ts.map +1 -0
- package/dist/src/formats/json/importer.js +1838 -0
- package/dist/src/formats/json/importer.js.map +1 -0
- package/dist/src/formats/json/index.d.ts +8 -0
- package/dist/src/formats/json/index.d.ts.map +1 -0
- package/dist/src/formats/json/index.js +8 -0
- package/dist/src/formats/json/index.js.map +1 -0
- package/dist/src/formats/neo4j/exporter.d.ts +68 -0
- package/dist/src/formats/neo4j/exporter.d.ts.map +1 -0
- package/dist/src/formats/neo4j/exporter.js +1055 -0
- package/dist/src/formats/neo4j/exporter.js.map +1 -0
- package/dist/src/formats/neo4j/header.d.ts +52 -0
- package/dist/src/formats/neo4j/header.d.ts.map +1 -0
- package/dist/src/formats/neo4j/header.js +131 -0
- package/dist/src/formats/neo4j/header.js.map +1 -0
- package/dist/src/formats/neo4j/importer.d.ts +73 -0
- package/dist/src/formats/neo4j/importer.d.ts.map +1 -0
- package/dist/src/formats/neo4j/importer.js +932 -0
- package/dist/src/formats/neo4j/importer.js.map +1 -0
- package/dist/src/formats/neo4j/index.d.ts +79 -0
- package/dist/src/formats/neo4j/index.d.ts.map +1 -0
- package/dist/src/formats/neo4j/index.js +83 -0
- package/dist/src/formats/neo4j/index.js.map +1 -0
- package/dist/src/formats/pajek/exporter.d.ts +58 -0
- package/dist/src/formats/pajek/exporter.d.ts.map +1 -0
- package/dist/src/formats/pajek/exporter.js +825 -0
- package/dist/src/formats/pajek/exporter.js.map +1 -0
- package/dist/src/formats/pajek/importer.d.ts +88 -0
- package/dist/src/formats/pajek/importer.d.ts.map +1 -0
- package/dist/src/formats/pajek/importer.js +1047 -0
- package/dist/src/formats/pajek/importer.js.map +1 -0
- package/dist/src/formats/pajek/index.d.ts +7 -0
- package/dist/src/formats/pajek/index.d.ts.map +1 -0
- package/dist/src/formats/pajek/index.js +7 -0
- package/dist/src/formats/pajek/index.js.map +1 -0
- package/dist/src/formats/pajek/syntax.d.ts +112 -0
- package/dist/src/formats/pajek/syntax.d.ts.map +1 -0
- package/dist/src/formats/pajek/syntax.js +269 -0
- package/dist/src/formats/pajek/syntax.js.map +1 -0
- package/dist/src/index.d.ts +35 -0
- package/dist/src/index.d.ts.map +1 -0
- package/dist/src/index.js +39 -0
- package/dist/src/index.js.map +1 -0
- package/dist/src/registry.d.ts +207 -0
- package/dist/src/registry.d.ts.map +1 -0
- package/dist/src/registry.js +481 -0
- package/dist/src/registry.js.map +1 -0
- package/dist/src/sniff.d.ts +104 -0
- package/dist/src/sniff.d.ts.map +1 -0
- package/dist/src/sniff.js +357 -0
- package/dist/src/sniff.js.map +1 -0
- package/dist/src/types.d.ts +238 -0
- package/dist/src/types.d.ts.map +1 -0
- package/dist/src/types.js +29 -0
- package/dist/src/types.js.map +1 -0
- package/dist/tsconfig.build.tsbuildinfo +1 -0
- package/package.json +122 -7
- package/src/children.ts +335 -0
- package/src/common/attributes.ts +520 -0
- package/src/common/codes.ts +153 -0
- package/src/common/declared-types.ts +374 -0
- package/src/common/direction.ts +518 -0
- package/src/common/escape.ts +231 -0
- package/src/common/export.ts +817 -0
- package/src/common/format.ts +111 -0
- package/src/common/ids.ts +176 -0
- package/src/common/input.ts +378 -0
- package/src/common/lists.ts +196 -0
- package/src/common/options.ts +377 -0
- package/src/common/report.ts +352 -0
- package/src/common/temporal.ts +302 -0
- package/src/common/text.ts +294 -0
- package/src/common/weights.ts +202 -0
- package/src/common/writer.ts +123 -0
- package/src/common/xml.ts +1053 -0
- package/src/formats/csv/exporter.ts +894 -0
- package/src/formats/csv/header.ts +172 -0
- package/src/formats/csv/importer.ts +1104 -0
- package/src/formats/csv/index.ts +88 -0
- package/src/formats/csv/records.ts +813 -0
- package/src/formats/csv/values.ts +224 -0
- package/src/formats/dot/exporter.ts +1014 -0
- package/src/formats/dot/importer.ts +1549 -0
- package/src/formats/dot/index.ts +7 -0
- package/src/formats/dot/names.ts +40 -0
- package/src/formats/dot/tokenizer.ts +384 -0
- package/src/formats/gexf/exporter.ts +1696 -0
- package/src/formats/gexf/importer.ts +2333 -0
- package/src/formats/gexf/index.ts +142 -0
- package/src/formats/gexf/schema.ts +361 -0
- package/src/formats/gml/exporter.ts +1404 -0
- package/src/formats/gml/importer.ts +1591 -0
- package/src/formats/gml/index.ts +128 -0
- package/src/formats/gml/syntax.ts +545 -0
- package/src/formats/graphml/constants.ts +225 -0
- package/src/formats/graphml/exporter.ts +1458 -0
- package/src/formats/graphml/importer.ts +2027 -0
- package/src/formats/graphml/index.ts +8 -0
- package/src/formats/graphml/tree.ts +318 -0
- package/src/formats/json/dialect.ts +317 -0
- package/src/formats/json/exporter.ts +1616 -0
- package/src/formats/json/importer.ts +2271 -0
- package/src/formats/json/index.ts +8 -0
- package/src/formats/neo4j/exporter.ts +1287 -0
- package/src/formats/neo4j/header.ts +156 -0
- package/src/formats/neo4j/importer.ts +1220 -0
- package/src/formats/neo4j/index.ts +116 -0
- package/src/formats/pajek/exporter.ts +1000 -0
- package/src/formats/pajek/importer.ts +1311 -0
- package/src/formats/pajek/index.ts +7 -0
- package/src/formats/pajek/syntax.ts +307 -0
- package/src/index.ts +244 -0
- package/src/registry.ts +617 -0
- package/src/sniff.ts +397 -0
- package/src/types.ts +262 -0
|
@@ -0,0 +1,894 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The CSV / TSV exporter (design section 8.5; research note 07 section 9): writes the edge table
|
|
3
|
+
* (`Source,Target,Type,Id,Label,Weight,<attributes>` in the Gephi dialect, `source,target,weight,
|
|
4
|
+
* <attributes>` in the generic one) or, with `table: "nodes"`, the node table (`Id,Label,
|
|
5
|
+
* <attributes>`), RFC 4180 quoted, one row per logical edge with expanded pairs folded back
|
|
6
|
+
* through the `pair` role column.
|
|
7
|
+
*
|
|
8
|
+
* What survives a re-import exactly: ids (as text under the canonical rule), topology and
|
|
9
|
+
* orientation, explicit weights (blank cells for defaulted ones), the per-row direction of the
|
|
10
|
+
* Gephi dialect, edge ids and labels (string), bool / i32 / f64 / string columns (an f64 value is
|
|
11
|
+
* written with a decimal point so it reads back as f64), low-cardinality dicts, and set empty
|
|
12
|
+
* strings (the quoted empty cell; an unset cell is written as nothing). check() reports everything
|
|
13
|
+
* else: the generic dialect drops direction, mutual pairs become two directed rows, lists and json
|
|
14
|
+
* values are written as text, non-finite numbers do not read back, attribute names that collide
|
|
15
|
+
* with the reserved headers are not written, node attributes are written by the node table only,
|
|
16
|
+
* and the edge table carries neither isolated nodes nor the node order.
|
|
17
|
+
*/
|
|
18
|
+
|
|
19
|
+
import { type Column, GraphFormatError, type GraphSnapshot, type NodeId } from "@graphty/graph-format";
|
|
20
|
+
|
|
21
|
+
import { DICT_SAMPLE_ROWS, DictHeuristic } from "../../common/attributes.js";
|
|
22
|
+
import { type PairFolding, pairFolding } from "../../common/direction.js";
|
|
23
|
+
import { quoteCsvCell } from "../../common/escape.js";
|
|
24
|
+
import { capabilities, checkCapabilities, countMixedEdges, LOSS } from "../../common/export.js";
|
|
25
|
+
import { formatDecimal, formatInteger } from "../../common/format.js";
|
|
26
|
+
import { canonicalId } from "../../common/ids.js";
|
|
27
|
+
import { joinListText } from "../../common/lists.js";
|
|
28
|
+
import { type ResolvedExportOptions, resolveExportOptions } from "../../common/options.js";
|
|
29
|
+
import { inferTextDtype, type TextDtype } from "../../common/text.js";
|
|
30
|
+
import { type ExplicitWeights, explicitWeights } from "../../common/weights.js";
|
|
31
|
+
import { encodeChunks, joinText } from "../../common/writer.js";
|
|
32
|
+
import { type CommonExportOptions, type ExportCapabilities, type GraphExporter, type LossNote } from "../../types.js";
|
|
33
|
+
import { EDGE_ID_NAMES, findColumn, LABEL_NAMES } from "./header.js";
|
|
34
|
+
|
|
35
|
+
/** The format-specific options of the CSV exporter. */
|
|
36
|
+
export interface CsvExportOptions {
|
|
37
|
+
/**
|
|
38
|
+
* The header spelling: "gephi" (default) writes `Source,Target,Type,...,Weight` with the per-row
|
|
39
|
+
* direction; "generic" writes `source,target,...,weight` and no direction column.
|
|
40
|
+
*/
|
|
41
|
+
dialect?: "gephi" | "generic" | undefined;
|
|
42
|
+
/** Which table to write: the edge table (default) or the node table. */
|
|
43
|
+
table?: "edges" | "nodes" | undefined;
|
|
44
|
+
/** The field delimiter; "," by default. */
|
|
45
|
+
delimiter?: string | undefined;
|
|
46
|
+
/** The line terminator; "\n" by default. */
|
|
47
|
+
newline?: "\n" | "\r\n" | undefined;
|
|
48
|
+
/** Whether to write the header row; true by default. */
|
|
49
|
+
header?: boolean | undefined;
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
/** The CSV loss-note codes of check(); the shared ones are LOSS's. */
|
|
53
|
+
export const CSV_LOSS = Object.freeze({
|
|
54
|
+
/** Two node ids share one text (a number and a string); export() throws E_INVALID_ID. */
|
|
55
|
+
ID_TEXT_COLLISION: LOSS.ID_TEXT_COLLISION,
|
|
56
|
+
/** Ids whose text reads back as the other type under the canonical rule. */
|
|
57
|
+
ID_TEXT_TYPE: LOSS.ID_TEXT_TYPE,
|
|
58
|
+
/** The generic dialect has no direction column; an undirected or mixed graph reads back as directed. */
|
|
59
|
+
DIRECTION_DROPPED: "W_CSV_DIRECTION_DROPPED",
|
|
60
|
+
/** Mutual pairs are written as two directed rows. */
|
|
61
|
+
MUTUAL_EXPANDED: LOSS.MUTUAL_EXPANDED,
|
|
62
|
+
/** An attribute column named like a reserved header is not written. */
|
|
63
|
+
RESERVED_NAME: "W_CSV_RESERVED_NAME",
|
|
64
|
+
/** A column without a role that the importer gives one back by its name. */
|
|
65
|
+
ROLE_ASSUMED: LOSS.ROLE_ASSUMED,
|
|
66
|
+
/** A role column (id, label) whose name the importer does not recognise; the role is lost. */
|
|
67
|
+
ROLE_NAME: "W_CSV_ROLE_NAME",
|
|
68
|
+
/** A role column (id, label) that is not string / dict reads back as string. */
|
|
69
|
+
TEXT_ROLE: "W_CSV_TEXT_ROLE",
|
|
70
|
+
/** NaN / Infinity in a numeric column read back as text. */
|
|
71
|
+
NONFINITE: "W_CSV_NONFINITE",
|
|
72
|
+
/** A text column whose every value reads back as a number or boolean. */
|
|
73
|
+
TEXT_INFERRED: LOSS.TEXT_INFERRED,
|
|
74
|
+
/** A dict column whose cardinality makes the importer read it back as string, or the reverse. */
|
|
75
|
+
STORAGE_CLASS_CHANGED: LOSS.STORAGE_CLASS,
|
|
76
|
+
/** Node attributes are written by a `table: "nodes"` export only. */
|
|
77
|
+
NODE_TABLE: "W_CSV_NODE_TABLE",
|
|
78
|
+
/** The edge table carries no node without an edge: isolated nodes vanish on re-import. */
|
|
79
|
+
ISOLATED_NODES: "W_CSV_ISOLATED_NODES",
|
|
80
|
+
/** The edge table lists nodes by first appearance; the node order (and indices) change on re-import. */
|
|
81
|
+
NODE_ORDER: "W_CSV_NODE_ORDER",
|
|
82
|
+
});
|
|
83
|
+
|
|
84
|
+
/** What the CSV format keeps as declared. */
|
|
85
|
+
export const CSV_CAPABILITIES: ExportCapabilities = capabilities({
|
|
86
|
+
mixedDirection: true,
|
|
87
|
+
multiEdges: true,
|
|
88
|
+
selfLoops: true,
|
|
89
|
+
edgeIds: "optional",
|
|
90
|
+
idCharset: "any",
|
|
91
|
+
dtypes: ["bool", "i32", "f64", "string", "dict"],
|
|
92
|
+
});
|
|
93
|
+
|
|
94
|
+
/** The dialect header spellings. */
|
|
95
|
+
interface Dialect {
|
|
96
|
+
readonly source: string;
|
|
97
|
+
readonly target: string;
|
|
98
|
+
readonly type: string | null;
|
|
99
|
+
readonly id: string;
|
|
100
|
+
readonly label: string;
|
|
101
|
+
readonly weight: string;
|
|
102
|
+
}
|
|
103
|
+
|
|
104
|
+
const DIALECTS: Readonly<Record<"gephi" | "generic", Dialect>> = {
|
|
105
|
+
gephi: { source: "Source", target: "Target", type: "Type", id: "Id", label: "Label", weight: "Weight" },
|
|
106
|
+
generic: { source: "source", target: "target", type: null, id: "id", label: "label", weight: "weight" },
|
|
107
|
+
};
|
|
108
|
+
|
|
109
|
+
/** The roles the tables have a slot for (the id and label headers); every other role is reported. */
|
|
110
|
+
const KEPT_ROLES: ReadonlySet<string> = new Set(["id", "label"]);
|
|
111
|
+
|
|
112
|
+
/** Roles the edge and node tables never write as attributes. */
|
|
113
|
+
const SKIPPED_ROLES: ReadonlySet<string> = new Set([
|
|
114
|
+
"directed",
|
|
115
|
+
"pair",
|
|
116
|
+
"mutual",
|
|
117
|
+
"weight",
|
|
118
|
+
"timeText",
|
|
119
|
+
"position",
|
|
120
|
+
"color",
|
|
121
|
+
"size",
|
|
122
|
+
"shape",
|
|
123
|
+
"thickness",
|
|
124
|
+
"parent",
|
|
125
|
+
"parents",
|
|
126
|
+
"start",
|
|
127
|
+
"end",
|
|
128
|
+
"timestamp",
|
|
129
|
+
"timestamps",
|
|
130
|
+
"spells",
|
|
131
|
+
"open",
|
|
132
|
+
]);
|
|
133
|
+
|
|
134
|
+
/** The CSV options with defaults applied. */
|
|
135
|
+
interface ResolvedCsvExportOptions {
|
|
136
|
+
readonly dialect: Dialect;
|
|
137
|
+
readonly table: "edges" | "nodes";
|
|
138
|
+
readonly delimiter: string;
|
|
139
|
+
readonly newline: string;
|
|
140
|
+
readonly header: boolean;
|
|
141
|
+
}
|
|
142
|
+
|
|
143
|
+
/** A column written under a header. */
|
|
144
|
+
interface ColumnOut {
|
|
145
|
+
readonly column: Column;
|
|
146
|
+
readonly header: string;
|
|
147
|
+
}
|
|
148
|
+
|
|
149
|
+
/** The plan of one export call: what is written, and the notes. */
|
|
150
|
+
interface Plan {
|
|
151
|
+
readonly csv: ResolvedCsvExportOptions;
|
|
152
|
+
readonly common: ResolvedExportOptions;
|
|
153
|
+
readonly notes: LossNote[];
|
|
154
|
+
/** The rows of the edge table (logical edge indices; mirrors of undirected pairs left out). */
|
|
155
|
+
readonly edgeRows: number[];
|
|
156
|
+
readonly folding: PairFolding;
|
|
157
|
+
readonly edgeId: Column | null;
|
|
158
|
+
readonly edgeLabel: Column | null;
|
|
159
|
+
readonly weights: ExplicitWeights;
|
|
160
|
+
readonly edgeColumns: ColumnOut[];
|
|
161
|
+
readonly nodeLabel: Column | null;
|
|
162
|
+
readonly nodeColumns: ColumnOut[];
|
|
163
|
+
/** The id text of every node. */
|
|
164
|
+
readonly idText: (index: number) => string;
|
|
165
|
+
}
|
|
166
|
+
|
|
167
|
+
/**
|
|
168
|
+
* Apply the defaults of the CSV export options and check every value.
|
|
169
|
+
* @param options - the caller's options
|
|
170
|
+
* @returns the resolved options; E_UNSUPPORTED for a value outside its set
|
|
171
|
+
*/
|
|
172
|
+
function resolveCsvExportOptions(
|
|
173
|
+
options: (CsvExportOptions & CommonExportOptions) | undefined,
|
|
174
|
+
): ResolvedCsvExportOptions {
|
|
175
|
+
const o: CsvExportOptions = options ?? {};
|
|
176
|
+
if (o.dialect !== undefined && o.dialect !== "gephi" && o.dialect !== "generic") {
|
|
177
|
+
throw new GraphFormatError("E_UNSUPPORTED", 'option dialect: expected "gephi" or "generic"', {
|
|
178
|
+
option: "dialect",
|
|
179
|
+
found: o.dialect,
|
|
180
|
+
});
|
|
181
|
+
}
|
|
182
|
+
if (o.table !== undefined && o.table !== "edges" && o.table !== "nodes") {
|
|
183
|
+
throw new GraphFormatError("E_UNSUPPORTED", 'option table: expected "edges" or "nodes"', {
|
|
184
|
+
option: "table",
|
|
185
|
+
found: o.table,
|
|
186
|
+
});
|
|
187
|
+
}
|
|
188
|
+
if (
|
|
189
|
+
o.delimiter !== undefined &&
|
|
190
|
+
(typeof o.delimiter !== "string" ||
|
|
191
|
+
o.delimiter.length !== 1 ||
|
|
192
|
+
o.delimiter === '"' ||
|
|
193
|
+
o.delimiter === "\n" ||
|
|
194
|
+
o.delimiter === "\r")
|
|
195
|
+
) {
|
|
196
|
+
throw new GraphFormatError(
|
|
197
|
+
"E_UNSUPPORTED",
|
|
198
|
+
"option delimiter: expected one character other than a quote or a line break",
|
|
199
|
+
{ option: "delimiter", found: o.delimiter },
|
|
200
|
+
);
|
|
201
|
+
}
|
|
202
|
+
if (o.newline !== undefined && o.newline !== "\n" && o.newline !== "\r\n") {
|
|
203
|
+
throw new GraphFormatError("E_UNSUPPORTED", "option newline: expected LF or CRLF", {
|
|
204
|
+
option: "newline",
|
|
205
|
+
found: o.newline,
|
|
206
|
+
});
|
|
207
|
+
}
|
|
208
|
+
if (o.header !== undefined && typeof o.header !== "boolean") {
|
|
209
|
+
throw new GraphFormatError("E_UNSUPPORTED", "option header: expected a boolean", {
|
|
210
|
+
option: "header",
|
|
211
|
+
found: o.header,
|
|
212
|
+
});
|
|
213
|
+
}
|
|
214
|
+
return {
|
|
215
|
+
dialect: DIALECTS[o.dialect ?? "gephi"],
|
|
216
|
+
table: o.table ?? "edges",
|
|
217
|
+
delimiter: o.delimiter ?? ",",
|
|
218
|
+
newline: o.newline ?? "\n",
|
|
219
|
+
header: o.header ?? true,
|
|
220
|
+
};
|
|
221
|
+
}
|
|
222
|
+
|
|
223
|
+
/**
|
|
224
|
+
* One scalar (or one list item) as text by its dtype.
|
|
225
|
+
* @param dtype - the scalar dtype
|
|
226
|
+
* @param value - the value
|
|
227
|
+
* @returns the text
|
|
228
|
+
*/
|
|
229
|
+
function scalarText(dtype: string, value: unknown): string {
|
|
230
|
+
switch (dtype) {
|
|
231
|
+
case "bool":
|
|
232
|
+
return value === true ? "true" : "false";
|
|
233
|
+
case "f32":
|
|
234
|
+
case "f64":
|
|
235
|
+
return typeof value === "number" ? formatDecimal(value, dtype) : String(value);
|
|
236
|
+
case "i32":
|
|
237
|
+
case "u32":
|
|
238
|
+
case "u8":
|
|
239
|
+
return typeof value === "number" ? formatInteger(value) : String(value);
|
|
240
|
+
case "string":
|
|
241
|
+
case "dict":
|
|
242
|
+
return typeof value === "string" ? value : String(value);
|
|
243
|
+
case "json":
|
|
244
|
+
return JSON.stringify(value) ?? "";
|
|
245
|
+
default:
|
|
246
|
+
return String(value);
|
|
247
|
+
}
|
|
248
|
+
}
|
|
249
|
+
|
|
250
|
+
/**
|
|
251
|
+
* The text of one cell: null for an unset row (written as nothing between the delimiters);
|
|
252
|
+
* components and list items joined by `;`.
|
|
253
|
+
* @param column - the column
|
|
254
|
+
* @param row - the row
|
|
255
|
+
* @returns the cell text, or null
|
|
256
|
+
*/
|
|
257
|
+
function cellText(column: Column, row: number): string | null {
|
|
258
|
+
if (!column.isSet(row)) {
|
|
259
|
+
return null;
|
|
260
|
+
}
|
|
261
|
+
switch (column.dtype) {
|
|
262
|
+
case "list": {
|
|
263
|
+
const items = column.sliceOf(row);
|
|
264
|
+
const { dtype } = column.child;
|
|
265
|
+
return joinListText(
|
|
266
|
+
items.map((item) => scalarText(dtype, item)),
|
|
267
|
+
"semicolon",
|
|
268
|
+
);
|
|
269
|
+
}
|
|
270
|
+
case "json":
|
|
271
|
+
return JSON.stringify(column.values[row]) ?? "";
|
|
272
|
+
case "bool":
|
|
273
|
+
case "string":
|
|
274
|
+
case "dict":
|
|
275
|
+
return scalarText(column.dtype, column.value(row));
|
|
276
|
+
default: {
|
|
277
|
+
const { components } = column.meta;
|
|
278
|
+
const { data } = column;
|
|
279
|
+
if (components === 1) {
|
|
280
|
+
return scalarText(column.dtype, data[row]);
|
|
281
|
+
}
|
|
282
|
+
const parts: string[] = [];
|
|
283
|
+
for (let k = 0; k < components; k++) {
|
|
284
|
+
parts.push(scalarText(column.dtype, data[row * components + k]));
|
|
285
|
+
}
|
|
286
|
+
return parts.join(";");
|
|
287
|
+
}
|
|
288
|
+
}
|
|
289
|
+
}
|
|
290
|
+
|
|
291
|
+
/**
|
|
292
|
+
* Build the plan of an export: resolve options, choose the rows and columns, and collect the
|
|
293
|
+
* CSV-specific loss notes (check() adds the generic ones). The value-level notes need a pass over
|
|
294
|
+
* every written cell and are collected for check() only; export() needs the plan and the fatal
|
|
295
|
+
* id-collision note alone.
|
|
296
|
+
* @param snapshot - the snapshot
|
|
297
|
+
* @param options - the caller's options
|
|
298
|
+
* @param values - whether to scan the values for notes
|
|
299
|
+
* @returns the plan
|
|
300
|
+
*/
|
|
301
|
+
/**
|
|
302
|
+
* The id notes: text collisions are fatal (export() throws E_INVALID_ID); type changes under the
|
|
303
|
+
* canonical re-read are reported.
|
|
304
|
+
* @param snapshot - the snapshot
|
|
305
|
+
* @param values - whether the values are inspected (false for a plan that writes nothing)
|
|
306
|
+
* @param note - the note recorder
|
|
307
|
+
*/
|
|
308
|
+
function idNotes(
|
|
309
|
+
snapshot: GraphSnapshot,
|
|
310
|
+
values: boolean,
|
|
311
|
+
note: (code: string, message: string, column?: string | null, count?: number | null) => void,
|
|
312
|
+
): void {
|
|
313
|
+
const { ids } = snapshot;
|
|
314
|
+
if (ids.kind === "mixed") {
|
|
315
|
+
const seen = new Set<string>();
|
|
316
|
+
let collisions = 0;
|
|
317
|
+
for (let i = 0; i < ids.size; i++) {
|
|
318
|
+
const text = String(ids.idOf(i));
|
|
319
|
+
if (seen.has(text)) {
|
|
320
|
+
collisions++;
|
|
321
|
+
} else {
|
|
322
|
+
seen.add(text);
|
|
323
|
+
}
|
|
324
|
+
}
|
|
325
|
+
if (collisions > 0) {
|
|
326
|
+
note(
|
|
327
|
+
CSV_LOSS.ID_TEXT_COLLISION,
|
|
328
|
+
`${collisions} node id(s) share their text with another id (a number and a string); export() will throw E_INVALID_ID`,
|
|
329
|
+
null,
|
|
330
|
+
collisions,
|
|
331
|
+
);
|
|
332
|
+
}
|
|
333
|
+
}
|
|
334
|
+
if (values && ids.kind !== "identity" && ids.kind !== "dense") {
|
|
335
|
+
let changed = 0;
|
|
336
|
+
for (let i = 0; i < ids.size; i++) {
|
|
337
|
+
const id: NodeId = ids.idOf(i);
|
|
338
|
+
if (typeof canonicalId(String(id)) !== typeof id) {
|
|
339
|
+
changed++;
|
|
340
|
+
}
|
|
341
|
+
}
|
|
342
|
+
if (changed > 0) {
|
|
343
|
+
note(
|
|
344
|
+
CSV_LOSS.ID_TEXT_TYPE,
|
|
345
|
+
`${changed} node id(s) read back as the other type under ids: "canonical" (a string "1" becomes 1, a number 1.5 becomes "1.5")`,
|
|
346
|
+
null,
|
|
347
|
+
changed,
|
|
348
|
+
);
|
|
349
|
+
}
|
|
350
|
+
}
|
|
351
|
+
}
|
|
352
|
+
|
|
353
|
+
function planExport(
|
|
354
|
+
snapshot: GraphSnapshot,
|
|
355
|
+
options: (CsvExportOptions & CommonExportOptions) | undefined,
|
|
356
|
+
values: boolean,
|
|
357
|
+
): Plan {
|
|
358
|
+
const common = resolveExportOptions(options);
|
|
359
|
+
const csv = resolveCsvExportOptions(options);
|
|
360
|
+
const notes: LossNote[] = [];
|
|
361
|
+
const note = (code: string, message: string, column: string | null = null, count: number | null = null): void => {
|
|
362
|
+
notes.push(Object.freeze({ code, message, column, count }));
|
|
363
|
+
};
|
|
364
|
+
const { ids } = snapshot;
|
|
365
|
+
const idText = (index: number): string => String(ids.idOf(index));
|
|
366
|
+
idNotes(snapshot, values, note);
|
|
367
|
+
|
|
368
|
+
// edge rows: fold the mirrors of undirected pairs; mutual pairs stay two directed rows
|
|
369
|
+
const folding = pairFolding(snapshot);
|
|
370
|
+
const edgeRows: number[] = [];
|
|
371
|
+
for (let e = 0; e < snapshot.edgeCount; e++) {
|
|
372
|
+
if (!folding.folded(e)) {
|
|
373
|
+
edgeRows.push(e);
|
|
374
|
+
}
|
|
375
|
+
}
|
|
376
|
+
if (csv.dialect.type === null) {
|
|
377
|
+
const mixed = countMixedEdges(snapshot);
|
|
378
|
+
if (!snapshot.directed) {
|
|
379
|
+
note(
|
|
380
|
+
CSV_LOSS.DIRECTION_DROPPED,
|
|
381
|
+
`the generic dialect has no direction column; ${snapshot.edgeCount} undirected edge(s) read back as directed unless the importer is told otherwise`,
|
|
382
|
+
null,
|
|
383
|
+
snapshot.edgeCount,
|
|
384
|
+
);
|
|
385
|
+
} else if (mixed > 0) {
|
|
386
|
+
note(
|
|
387
|
+
CSV_LOSS.DIRECTION_DROPPED,
|
|
388
|
+
`the generic dialect has no direction column; ${mixed} undirected edge(s) of a mixed graph read back as one directed edge each`,
|
|
389
|
+
null,
|
|
390
|
+
mixed,
|
|
391
|
+
);
|
|
392
|
+
}
|
|
393
|
+
}
|
|
394
|
+
if (folding.mutualCount > 0) {
|
|
395
|
+
note(
|
|
396
|
+
CSV_LOSS.MUTUAL_EXPANDED,
|
|
397
|
+
`${folding.mutualCount} mutual pair(s) are written as two directed rows; the mutual mark is lost`,
|
|
398
|
+
null,
|
|
399
|
+
folding.mutualCount,
|
|
400
|
+
);
|
|
401
|
+
}
|
|
402
|
+
if (csv.table === "edges") {
|
|
403
|
+
noteNodeCoverage(snapshot, edgeRows, note);
|
|
404
|
+
}
|
|
405
|
+
|
|
406
|
+
// weights
|
|
407
|
+
const weights = explicitWeights(snapshot);
|
|
408
|
+
|
|
409
|
+
// the edge table's columns
|
|
410
|
+
const edgeId = snapshot.edges.byRole("id");
|
|
411
|
+
const edgeLabel = snapshot.edges.byRole("label");
|
|
412
|
+
const edgeReserved = new Set<string>([csv.dialect.source, csv.dialect.target, csv.dialect.weight].map(lower));
|
|
413
|
+
if (csv.dialect.type !== null) {
|
|
414
|
+
edgeReserved.add(lower(csv.dialect.type));
|
|
415
|
+
}
|
|
416
|
+
const edgeColumns = attributeColumns(
|
|
417
|
+
snapshot.edges,
|
|
418
|
+
"edge",
|
|
419
|
+
edgeReserved,
|
|
420
|
+
edgeId,
|
|
421
|
+
edgeLabel,
|
|
422
|
+
csv.table === "edges" ? note : null,
|
|
423
|
+
);
|
|
424
|
+
if (csv.table === "edges" && values) {
|
|
425
|
+
checkRoleColumn(edgeId, "id", EDGE_ID_NAMES, "edge", edgeRows, note);
|
|
426
|
+
checkRoleColumn(edgeLabel, "label", LABEL_NAMES, "edge", edgeRows, note);
|
|
427
|
+
for (const { column } of edgeColumns) {
|
|
428
|
+
checkValues(column, "edge", edgeRows, note);
|
|
429
|
+
}
|
|
430
|
+
}
|
|
431
|
+
|
|
432
|
+
// the node table's columns
|
|
433
|
+
const nodeLabel = snapshot.nodes.byRole("label");
|
|
434
|
+
const nodeReserved = new Set<string>([lower(csv.dialect.id)]);
|
|
435
|
+
const nodeRows: number[] = [];
|
|
436
|
+
if (csv.table === "nodes" && values) {
|
|
437
|
+
for (let i = 0; i < snapshot.nodeCount; i++) {
|
|
438
|
+
nodeRows.push(i);
|
|
439
|
+
}
|
|
440
|
+
}
|
|
441
|
+
const nodeColumns = attributeColumns(
|
|
442
|
+
snapshot.nodes,
|
|
443
|
+
"node",
|
|
444
|
+
nodeReserved,
|
|
445
|
+
null,
|
|
446
|
+
nodeLabel,
|
|
447
|
+
csv.table === "nodes" ? note : null,
|
|
448
|
+
);
|
|
449
|
+
if (csv.table === "nodes") {
|
|
450
|
+
if (values) {
|
|
451
|
+
checkRoleColumn(nodeLabel, "label", LABEL_NAMES, "node", nodeRows, note);
|
|
452
|
+
for (const { column } of nodeColumns) {
|
|
453
|
+
checkValues(column, "node", nodeRows, note);
|
|
454
|
+
}
|
|
455
|
+
}
|
|
456
|
+
} else {
|
|
457
|
+
const written = nodeColumns.length + (nodeLabel === null ? 0 : 1);
|
|
458
|
+
if (written > 0) {
|
|
459
|
+
note(
|
|
460
|
+
CSV_LOSS.NODE_TABLE,
|
|
461
|
+
`${written} node column(s) are written by a table: "nodes" export only`,
|
|
462
|
+
null,
|
|
463
|
+
written,
|
|
464
|
+
);
|
|
465
|
+
}
|
|
466
|
+
}
|
|
467
|
+
|
|
468
|
+
return {
|
|
469
|
+
csv,
|
|
470
|
+
common,
|
|
471
|
+
notes,
|
|
472
|
+
edgeRows,
|
|
473
|
+
folding,
|
|
474
|
+
edgeId,
|
|
475
|
+
edgeLabel,
|
|
476
|
+
weights,
|
|
477
|
+
edgeColumns,
|
|
478
|
+
nodeLabel,
|
|
479
|
+
nodeColumns,
|
|
480
|
+
idText,
|
|
481
|
+
};
|
|
482
|
+
}
|
|
483
|
+
|
|
484
|
+
/**
|
|
485
|
+
* The edge table is not a node carrier (research note 07 section 2.6: node attributes and node
|
|
486
|
+
* presence come from a separate node table): a node without an edge is not written at all, and
|
|
487
|
+
* the importer creates nodes in the order the edge rows first mention them. Both are reported so
|
|
488
|
+
* check() predicts the ids and the order a re-import gives back (design section 8.5).
|
|
489
|
+
* @param snapshot - the snapshot
|
|
490
|
+
* @param edgeRows - the edge rows written
|
|
491
|
+
* @param note - the recorder
|
|
492
|
+
*/
|
|
493
|
+
function noteNodeCoverage(
|
|
494
|
+
snapshot: GraphSnapshot,
|
|
495
|
+
edgeRows: readonly number[],
|
|
496
|
+
note: (code: string, message: string, column?: string | null, count?: number | null) => void,
|
|
497
|
+
): void {
|
|
498
|
+
const { nodeCount } = snapshot;
|
|
499
|
+
if (nodeCount === 0) {
|
|
500
|
+
return;
|
|
501
|
+
}
|
|
502
|
+
const list = snapshot.edgeList();
|
|
503
|
+
const seen = new Uint8Array(nodeCount);
|
|
504
|
+
let next = 0;
|
|
505
|
+
let reordered = 0;
|
|
506
|
+
const mention = (index: number): void => {
|
|
507
|
+
if (seen[index] === 1) {
|
|
508
|
+
return;
|
|
509
|
+
}
|
|
510
|
+
seen[index] = 1;
|
|
511
|
+
if (index !== next) {
|
|
512
|
+
reordered++;
|
|
513
|
+
}
|
|
514
|
+
next++;
|
|
515
|
+
};
|
|
516
|
+
for (const e of edgeRows) {
|
|
517
|
+
mention(list.src[e]);
|
|
518
|
+
mention(list.dst[e]);
|
|
519
|
+
}
|
|
520
|
+
const isolated = nodeCount - next;
|
|
521
|
+
if (isolated > 0) {
|
|
522
|
+
note(
|
|
523
|
+
CSV_LOSS.ISOLATED_NODES,
|
|
524
|
+
`${isolated} node(s) have no edge and cannot be written by the edge table; write the node table (table: "nodes") to keep them`,
|
|
525
|
+
null,
|
|
526
|
+
isolated,
|
|
527
|
+
);
|
|
528
|
+
}
|
|
529
|
+
if (reordered > 0) {
|
|
530
|
+
note(
|
|
531
|
+
CSV_LOSS.NODE_ORDER,
|
|
532
|
+
`${reordered} node(s) are first mentioned by an edge row out of index order; a re-import numbers nodes by first appearance`,
|
|
533
|
+
null,
|
|
534
|
+
reordered,
|
|
535
|
+
);
|
|
536
|
+
}
|
|
537
|
+
}
|
|
538
|
+
|
|
539
|
+
/**
|
|
540
|
+
* Lower-case a header name for reserved-name comparisons.
|
|
541
|
+
* @param name - the name
|
|
542
|
+
* @returns the lower-case name
|
|
543
|
+
*/
|
|
544
|
+
function lower(name: string): string {
|
|
545
|
+
return name.toLowerCase();
|
|
546
|
+
}
|
|
547
|
+
|
|
548
|
+
/**
|
|
549
|
+
* The attribute columns of a table that are written, in declaration order, skipping the role
|
|
550
|
+
* columns handled elsewhere, the roles the format cannot hold, and names that collide with the
|
|
551
|
+
* reserved headers (reported).
|
|
552
|
+
* @param table - the node or edge table
|
|
553
|
+
* @param domain - node or edge, for messages
|
|
554
|
+
* @param reserved - lower-case header names the dialect writes for structure
|
|
555
|
+
* @param idColumn - the table's id role column, or null
|
|
556
|
+
* @param labelColumn - the table's label role column, or null
|
|
557
|
+
* @param note - the recorder, or null when this table's notes are not wanted
|
|
558
|
+
* @returns the columns and their headers
|
|
559
|
+
*/
|
|
560
|
+
function attributeColumns(
|
|
561
|
+
table: Iterable<Column>,
|
|
562
|
+
domain: "node" | "edge",
|
|
563
|
+
reserved: ReadonlySet<string>,
|
|
564
|
+
idColumn: Column | null,
|
|
565
|
+
labelColumn: Column | null,
|
|
566
|
+
note: ((code: string, message: string, column?: string | null, count?: number | null) => void) | null,
|
|
567
|
+
): ColumnOut[] {
|
|
568
|
+
const out: ColumnOut[] = [];
|
|
569
|
+
for (const column of table) {
|
|
570
|
+
const { name, role } = column.meta;
|
|
571
|
+
if (column === idColumn || column === labelColumn) {
|
|
572
|
+
continue;
|
|
573
|
+
}
|
|
574
|
+
if (role !== null && SKIPPED_ROLES.has(role)) {
|
|
575
|
+
continue;
|
|
576
|
+
}
|
|
577
|
+
const low = lower(name);
|
|
578
|
+
const set = column.length - column.nullCount;
|
|
579
|
+
const isIdName = domain === "edge" ? findColumn([name], EDGE_ID_NAMES) >= 0 : false;
|
|
580
|
+
if (reserved.has(low) || isIdName || (idColumn !== null && lower(idColumn.meta.name) === low)) {
|
|
581
|
+
note?.(
|
|
582
|
+
CSV_LOSS.RESERVED_NAME,
|
|
583
|
+
`${domain} column "${name}" is not written: the name is reserved for the ${isIdName ? "edge id" : low} column`,
|
|
584
|
+
name,
|
|
585
|
+
set,
|
|
586
|
+
);
|
|
587
|
+
continue;
|
|
588
|
+
}
|
|
589
|
+
if (findColumn([name], LABEL_NAMES) >= 0) {
|
|
590
|
+
if (labelColumn !== null) {
|
|
591
|
+
note?.(
|
|
592
|
+
CSV_LOSS.RESERVED_NAME,
|
|
593
|
+
`${domain} column "${name}" is not written: the name is reserved for the label column`,
|
|
594
|
+
name,
|
|
595
|
+
set,
|
|
596
|
+
);
|
|
597
|
+
continue;
|
|
598
|
+
}
|
|
599
|
+
note?.(
|
|
600
|
+
CSV_LOSS.ROLE_ASSUMED,
|
|
601
|
+
`${domain} column "${name}" reads back with the label role (string)`,
|
|
602
|
+
name,
|
|
603
|
+
set,
|
|
604
|
+
);
|
|
605
|
+
}
|
|
606
|
+
out.push({ column, header: name });
|
|
607
|
+
}
|
|
608
|
+
return out;
|
|
609
|
+
}
|
|
610
|
+
|
|
611
|
+
/**
|
|
612
|
+
* Notes about a role column (id or label): a name the importer does not map to the role, and a
|
|
613
|
+
* dtype other than string / dict (read back as string).
|
|
614
|
+
* @param column - the role column, or null
|
|
615
|
+
* @param role - the role
|
|
616
|
+
* @param names - the header names the importer maps to the role
|
|
617
|
+
* @param domain - node or edge
|
|
618
|
+
* @param rows - the rows written
|
|
619
|
+
* @param note - the recorder
|
|
620
|
+
*/
|
|
621
|
+
function checkRoleColumn(
|
|
622
|
+
column: Column | null,
|
|
623
|
+
role: "id" | "label",
|
|
624
|
+
names: readonly string[],
|
|
625
|
+
domain: "node" | "edge",
|
|
626
|
+
rows: readonly number[],
|
|
627
|
+
note: (code: string, message: string, column?: string | null, count?: number | null) => void,
|
|
628
|
+
): void {
|
|
629
|
+
if (column === null) {
|
|
630
|
+
return;
|
|
631
|
+
}
|
|
632
|
+
const { name } = column.meta;
|
|
633
|
+
if (findColumn([name], names) < 0) {
|
|
634
|
+
note(
|
|
635
|
+
CSV_LOSS.ROLE_NAME,
|
|
636
|
+
`${domain} column "${name}" (${role}) is written under its name, which the importer does not map to the ${role} role`,
|
|
637
|
+
name,
|
|
638
|
+
countSet(column, rows),
|
|
639
|
+
);
|
|
640
|
+
}
|
|
641
|
+
if (column.dtype !== "string" && column.dtype !== "dict") {
|
|
642
|
+
note(
|
|
643
|
+
CSV_LOSS.TEXT_ROLE,
|
|
644
|
+
`${domain} column "${name}" (${role}) is ${column.dtype}; it reads back as string`,
|
|
645
|
+
name,
|
|
646
|
+
countSet(column, rows),
|
|
647
|
+
);
|
|
648
|
+
}
|
|
649
|
+
checkValues(column, domain, rows, note, true);
|
|
650
|
+
}
|
|
651
|
+
|
|
652
|
+
/**
|
|
653
|
+
* Set rows among the rows written.
|
|
654
|
+
* @param column - the column
|
|
655
|
+
* @param rows - the rows written
|
|
656
|
+
* @returns the count
|
|
657
|
+
*/
|
|
658
|
+
function countSet(column: Column, rows: readonly number[]): number {
|
|
659
|
+
let count = 0;
|
|
660
|
+
for (const row of rows) {
|
|
661
|
+
if (column.isSet(row)) {
|
|
662
|
+
count++;
|
|
663
|
+
}
|
|
664
|
+
}
|
|
665
|
+
return count;
|
|
666
|
+
}
|
|
667
|
+
|
|
668
|
+
const WIDENING: Readonly<Record<TextDtype, number>> = { bool: 0, i32: 1, f64: 2, string: 3 };
|
|
669
|
+
const TEXT_DTYPES: readonly TextDtype[] = ["bool", "i32", "f64", "string"];
|
|
670
|
+
|
|
671
|
+
/**
|
|
672
|
+
* Value-level notes of one written column over the rows written: non-finite numbers (text on
|
|
673
|
+
* re-import), a text column whose values all read back as numbers or booleans, and the dict
|
|
674
|
+
* heuristic's verdict when it differs from the dtype. A set empty string is written as the quoted
|
|
675
|
+
* empty cell and reads back exactly.
|
|
676
|
+
* @param column - the column
|
|
677
|
+
* @param domain - node or edge
|
|
678
|
+
* @param rows - the rows written
|
|
679
|
+
* @param note - the recorder
|
|
680
|
+
* @param roleColumn - whether the column is declared string on re-import (no inference)
|
|
681
|
+
*/
|
|
682
|
+
function checkValues(
|
|
683
|
+
column: Column,
|
|
684
|
+
domain: "node" | "edge",
|
|
685
|
+
rows: readonly number[],
|
|
686
|
+
note: (code: string, message: string, column?: string | null, count?: number | null) => void,
|
|
687
|
+
roleColumn = false,
|
|
688
|
+
): void {
|
|
689
|
+
const { name, dtype } = column.meta;
|
|
690
|
+
const label = `${domain} column "${name}"`;
|
|
691
|
+
if (column.dtype === "f32" || column.dtype === "f64") {
|
|
692
|
+
let nonFinite = 0;
|
|
693
|
+
const { components } = column.meta;
|
|
694
|
+
const { data } = column;
|
|
695
|
+
for (const row of rows) {
|
|
696
|
+
if (!column.isSet(row)) {
|
|
697
|
+
continue;
|
|
698
|
+
}
|
|
699
|
+
for (let k = 0; k < components; k++) {
|
|
700
|
+
if (!Number.isFinite(data[row * components + k])) {
|
|
701
|
+
nonFinite++;
|
|
702
|
+
break;
|
|
703
|
+
}
|
|
704
|
+
}
|
|
705
|
+
}
|
|
706
|
+
if (nonFinite > 0) {
|
|
707
|
+
note(CSV_LOSS.NONFINITE, `${label}: ${nonFinite} non-finite value(s) read back as text`, name, nonFinite);
|
|
708
|
+
}
|
|
709
|
+
return;
|
|
710
|
+
}
|
|
711
|
+
if (dtype !== "string" && dtype !== "dict") {
|
|
712
|
+
return;
|
|
713
|
+
}
|
|
714
|
+
let widest = -1;
|
|
715
|
+
let candidate = true;
|
|
716
|
+
let set = 0;
|
|
717
|
+
const heuristic = new DictHeuristic(DICT_SAMPLE_ROWS);
|
|
718
|
+
for (const row of rows) {
|
|
719
|
+
if (!column.isSet(row)) {
|
|
720
|
+
continue;
|
|
721
|
+
}
|
|
722
|
+
const text = column.value(row);
|
|
723
|
+
if (typeof text !== "string") {
|
|
724
|
+
continue;
|
|
725
|
+
}
|
|
726
|
+
set++;
|
|
727
|
+
const kind = inferTextDtype(text);
|
|
728
|
+
widest = Math.max(widest, WIDENING[kind]);
|
|
729
|
+
if (!heuristic.decided) {
|
|
730
|
+
if (kind !== "string") {
|
|
731
|
+
candidate = false;
|
|
732
|
+
}
|
|
733
|
+
heuristic.observe(text);
|
|
734
|
+
}
|
|
735
|
+
}
|
|
736
|
+
if (roleColumn || widest < 0) {
|
|
737
|
+
return;
|
|
738
|
+
}
|
|
739
|
+
const readsAsDict = candidate && heuristic.decide() === "dict";
|
|
740
|
+
if (widest < WIDENING.string && !readsAsDict) {
|
|
741
|
+
note(CSV_LOSS.TEXT_INFERRED, `${label}: every value reads back as ${TEXT_DTYPES[widest]}`, name, set);
|
|
742
|
+
return;
|
|
743
|
+
}
|
|
744
|
+
if (dtype === "dict" && !readsAsDict) {
|
|
745
|
+
note(
|
|
746
|
+
CSV_LOSS.STORAGE_CLASS_CHANGED,
|
|
747
|
+
`${label}: reads back as string (cardinality too high for a dict)`,
|
|
748
|
+
name,
|
|
749
|
+
null,
|
|
750
|
+
);
|
|
751
|
+
} else if (dtype === "string" && readsAsDict) {
|
|
752
|
+
note(CSV_LOSS.STORAGE_CLASS_CHANGED, `${label}: reads back as dict (low cardinality)`, name, null);
|
|
753
|
+
}
|
|
754
|
+
}
|
|
755
|
+
|
|
756
|
+
/**
|
|
757
|
+
* The Type cell of an edge row.
|
|
758
|
+
* @param snapshot - the snapshot
|
|
759
|
+
* @param plan - the plan
|
|
760
|
+
* @param e - the edge
|
|
761
|
+
* @returns "Directed" or "Undirected"
|
|
762
|
+
*/
|
|
763
|
+
function typeText(snapshot: GraphSnapshot, plan: Plan, e: number): string {
|
|
764
|
+
if (!snapshot.directed || !plan.folding.sourceDirected(e)) {
|
|
765
|
+
return "Undirected";
|
|
766
|
+
}
|
|
767
|
+
return "Directed";
|
|
768
|
+
}
|
|
769
|
+
|
|
770
|
+
/**
|
|
771
|
+
* The lines of the export, one string per row (terminator included).
|
|
772
|
+
* @param snapshot - the snapshot
|
|
773
|
+
* @param plan - the plan
|
|
774
|
+
* @yields one line at a time
|
|
775
|
+
* @returns nothing
|
|
776
|
+
*/
|
|
777
|
+
function* lines(snapshot: GraphSnapshot, plan: Plan): Generator<string, void, undefined> {
|
|
778
|
+
const { csv } = plan;
|
|
779
|
+
const { delimiter, newline } = csv;
|
|
780
|
+
const quote = (text: string): string => quoteCsvCell(text, delimiter);
|
|
781
|
+
const cell = (column: Column, row: number): string => {
|
|
782
|
+
const text = cellText(column, row);
|
|
783
|
+
return text === null ? "" : quote(text);
|
|
784
|
+
};
|
|
785
|
+
if (csv.table === "nodes") {
|
|
786
|
+
const headers = [csv.dialect.id];
|
|
787
|
+
if (plan.nodeLabel !== null) {
|
|
788
|
+
headers.push(plan.nodeLabel.meta.name);
|
|
789
|
+
}
|
|
790
|
+
for (const { header } of plan.nodeColumns) {
|
|
791
|
+
headers.push(header);
|
|
792
|
+
}
|
|
793
|
+
if (csv.header) {
|
|
794
|
+
yield headers.map(quote).join(delimiter) + newline;
|
|
795
|
+
}
|
|
796
|
+
for (let i = 0; i < snapshot.nodeCount; i++) {
|
|
797
|
+
const cells = [quote(plan.idText(i))];
|
|
798
|
+
if (plan.nodeLabel !== null) {
|
|
799
|
+
cells.push(cell(plan.nodeLabel, i));
|
|
800
|
+
}
|
|
801
|
+
for (const { column } of plan.nodeColumns) {
|
|
802
|
+
cells.push(cell(column, i));
|
|
803
|
+
}
|
|
804
|
+
yield cells.join(delimiter) + newline;
|
|
805
|
+
}
|
|
806
|
+
return;
|
|
807
|
+
}
|
|
808
|
+
const headers = [csv.dialect.source, csv.dialect.target];
|
|
809
|
+
if (csv.dialect.type !== null) {
|
|
810
|
+
headers.push(csv.dialect.type);
|
|
811
|
+
}
|
|
812
|
+
if (plan.edgeId !== null) {
|
|
813
|
+
headers.push(plan.edgeId.meta.name);
|
|
814
|
+
}
|
|
815
|
+
if (plan.edgeLabel !== null) {
|
|
816
|
+
headers.push(plan.edgeLabel.meta.name);
|
|
817
|
+
}
|
|
818
|
+
if (plan.weights.weighted) {
|
|
819
|
+
headers.push(csv.dialect.weight);
|
|
820
|
+
}
|
|
821
|
+
for (const { header } of plan.edgeColumns) {
|
|
822
|
+
headers.push(header);
|
|
823
|
+
}
|
|
824
|
+
if (csv.header) {
|
|
825
|
+
yield headers.map(quote).join(delimiter) + newline;
|
|
826
|
+
}
|
|
827
|
+
const list = snapshot.edgeList();
|
|
828
|
+
for (const e of plan.edgeRows) {
|
|
829
|
+
const cells = [quote(plan.idText(list.src[e])), quote(plan.idText(list.dst[e]))];
|
|
830
|
+
if (csv.dialect.type !== null) {
|
|
831
|
+
cells.push(typeText(snapshot, plan, e));
|
|
832
|
+
}
|
|
833
|
+
if (plan.edgeId !== null) {
|
|
834
|
+
cells.push(cell(plan.edgeId, e));
|
|
835
|
+
}
|
|
836
|
+
if (plan.edgeLabel !== null) {
|
|
837
|
+
cells.push(cell(plan.edgeLabel, e));
|
|
838
|
+
}
|
|
839
|
+
if (plan.weights.weighted) {
|
|
840
|
+
cells.push(plan.weights.text(e) ?? "");
|
|
841
|
+
}
|
|
842
|
+
for (const { column } of plan.edgeColumns) {
|
|
843
|
+
cells.push(cell(column, e));
|
|
844
|
+
}
|
|
845
|
+
yield cells.join(delimiter) + newline;
|
|
846
|
+
}
|
|
847
|
+
}
|
|
848
|
+
|
|
849
|
+
/**
|
|
850
|
+
* Plan an export and refuse it when a fatal note is present (E_INVALID_ID for id text collisions),
|
|
851
|
+
* before anything is written.
|
|
852
|
+
* @param snapshot - the snapshot
|
|
853
|
+
* @param options - the caller's options
|
|
854
|
+
* @returns the plan
|
|
855
|
+
*/
|
|
856
|
+
function prepare(snapshot: GraphSnapshot, options: (CsvExportOptions & CommonExportOptions) | undefined): Plan {
|
|
857
|
+
const plan = planExport(snapshot, options, false);
|
|
858
|
+
const collision = plan.notes.find((n) => n.code === CSV_LOSS.ID_TEXT_COLLISION);
|
|
859
|
+
if (collision !== undefined) {
|
|
860
|
+
throw new GraphFormatError("E_INVALID_ID", collision.message, {
|
|
861
|
+
reason: "collision",
|
|
862
|
+
count: collision.count,
|
|
863
|
+
});
|
|
864
|
+
}
|
|
865
|
+
return plan;
|
|
866
|
+
}
|
|
867
|
+
|
|
868
|
+
/**
|
|
869
|
+
* Pre-flight: what a CSV export would lose. The generic notes describe the snapshot as a whole
|
|
870
|
+
* against the format; the CSV notes concern the table selected by `table`.
|
|
871
|
+
* @param snapshot - the snapshot
|
|
872
|
+
* @param options - CSV and common options
|
|
873
|
+
* @returns the notes, empty when the export is exact
|
|
874
|
+
*/
|
|
875
|
+
function check(snapshot: GraphSnapshot, options?: CsvExportOptions & CommonExportOptions): readonly LossNote[] {
|
|
876
|
+
const plan = planExport(snapshot, options, true);
|
|
877
|
+
return Object.freeze([
|
|
878
|
+
...checkCapabilities(snapshot, CSV_CAPABILITIES, plan.common, { roles: KEPT_ROLES }),
|
|
879
|
+
...plan.notes,
|
|
880
|
+
]);
|
|
881
|
+
}
|
|
882
|
+
|
|
883
|
+
/** The CSV / TSV exporter plugin (subpath `@graphty/graph-io/csv`). */
|
|
884
|
+
export const csvExporter: GraphExporter<CsvExportOptions> = Object.freeze({
|
|
885
|
+
format: "csv",
|
|
886
|
+
capabilities: CSV_CAPABILITIES,
|
|
887
|
+
check,
|
|
888
|
+
export(snapshot: GraphSnapshot, options?: CsvExportOptions & CommonExportOptions): AsyncIterable<Uint8Array> {
|
|
889
|
+
return encodeChunks(lines(snapshot, prepare(snapshot, options)));
|
|
890
|
+
},
|
|
891
|
+
exportToString(snapshot: GraphSnapshot, options?: CsvExportOptions & CommonExportOptions): Promise<string> {
|
|
892
|
+
return joinText(lines(snapshot, prepare(snapshot, options)));
|
|
893
|
+
},
|
|
894
|
+
});
|