@graphty/graph-io 0.0.0 → 0.2.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +250 -28
- package/dist/chunks/children-CL3Cy0ez.js +238 -0
- package/dist/chunks/children-CL3Cy0ez.js.map +1 -0
- package/dist/chunks/escape-DyI8JofU.js +938 -0
- package/dist/chunks/escape-DyI8JofU.js.map +1 -0
- package/dist/chunks/importer-CQnJuWJw.js +2987 -0
- package/dist/chunks/importer-CQnJuWJw.js.map +1 -0
- package/dist/chunks/importer-CpCpfbxr.js +2015 -0
- package/dist/chunks/importer-CpCpfbxr.js.map +1 -0
- package/dist/chunks/importer-DbnGYr3_.js +2342 -0
- package/dist/chunks/importer-DbnGYr3_.js.map +1 -0
- package/dist/chunks/importer-GozH8DkN.js +3050 -0
- package/dist/chunks/importer-GozH8DkN.js.map +1 -0
- package/dist/chunks/records-CGpxszm1.js +605 -0
- package/dist/chunks/records-CGpxszm1.js.map +1 -0
- package/dist/chunks/text-CajMdVFy.js +189 -0
- package/dist/chunks/text-CajMdVFy.js.map +1 -0
- package/dist/chunks/writer-DxSKC7TL.js +2842 -0
- package/dist/chunks/writer-DxSKC7TL.js.map +1 -0
- package/dist/csv.d.ts +1 -0
- package/dist/csv.js +1702 -0
- package/dist/csv.js.map +1 -0
- package/dist/dot.d.ts +1 -0
- package/dist/dot.js +8 -0
- package/dist/dot.js.map +1 -0
- package/dist/gexf.d.ts +1 -0
- package/dist/gexf.js +3466 -0
- package/dist/gexf.js.map +1 -0
- package/dist/gml.d.ts +1 -0
- package/dist/gml.js +2647 -0
- package/dist/gml.js.map +1 -0
- package/dist/graph-io.d.ts +1 -0
- package/dist/graph-io.js +790 -0
- package/dist/graph-io.js.map +1 -0
- package/dist/graphml.d.ts +1 -0
- package/dist/graphml.js +8 -0
- package/dist/graphml.js.map +1 -0
- package/dist/json.d.ts +1 -0
- package/dist/json.js +11 -0
- package/dist/json.js.map +1 -0
- package/dist/neo4j.d.ts +1 -0
- package/dist/neo4j.js +2046 -0
- package/dist/neo4j.js.map +1 -0
- package/dist/pajek.d.ts +1 -0
- package/dist/pajek.js +8 -0
- package/dist/pajek.js.map +1 -0
- package/dist/src/children.d.ts +134 -0
- package/dist/src/children.d.ts.map +1 -0
- package/dist/src/children.js +274 -0
- package/dist/src/children.js.map +1 -0
- package/dist/src/common/attributes.d.ts +229 -0
- package/dist/src/common/attributes.d.ts.map +1 -0
- package/dist/src/common/attributes.js +368 -0
- package/dist/src/common/attributes.js.map +1 -0
- package/dist/src/common/codes.d.ts +105 -0
- package/dist/src/common/codes.d.ts.map +1 -0
- package/dist/src/common/codes.js +107 -0
- package/dist/src/common/codes.js.map +1 -0
- package/dist/src/common/declared-types.d.ts +84 -0
- package/dist/src/common/declared-types.d.ts.map +1 -0
- package/dist/src/common/declared-types.js +326 -0
- package/dist/src/common/declared-types.js.map +1 -0
- package/dist/src/common/direction.d.ts +206 -0
- package/dist/src/common/direction.d.ts.map +1 -0
- package/dist/src/common/direction.js +370 -0
- package/dist/src/common/direction.js.map +1 -0
- package/dist/src/common/escape.d.ts +92 -0
- package/dist/src/common/escape.d.ts.map +1 -0
- package/dist/src/common/escape.js +212 -0
- package/dist/src/common/escape.js.map +1 -0
- package/dist/src/common/export.d.ts +249 -0
- package/dist/src/common/export.d.ts.map +1 -0
- package/dist/src/common/export.js +594 -0
- package/dist/src/common/export.js.map +1 -0
- package/dist/src/common/format.d.ts +59 -0
- package/dist/src/common/format.d.ts.map +1 -0
- package/dist/src/common/format.js +106 -0
- package/dist/src/common/format.js.map +1 -0
- package/dist/src/common/ids.d.ts +83 -0
- package/dist/src/common/ids.d.ts.map +1 -0
- package/dist/src/common/ids.js +158 -0
- package/dist/src/common/ids.js.map +1 -0
- package/dist/src/common/input.d.ts +100 -0
- package/dist/src/common/input.d.ts.map +1 -0
- package/dist/src/common/input.js +335 -0
- package/dist/src/common/input.js.map +1 -0
- package/dist/src/common/lists.d.ts +34 -0
- package/dist/src/common/lists.d.ts.map +1 -0
- package/dist/src/common/lists.js +185 -0
- package/dist/src/common/lists.js.map +1 -0
- package/dist/src/common/options.d.ts +108 -0
- package/dist/src/common/options.d.ts.map +1 -0
- package/dist/src/common/options.js +265 -0
- package/dist/src/common/options.js.map +1 -0
- package/dist/src/common/report.d.ts +187 -0
- package/dist/src/common/report.d.ts.map +1 -0
- package/dist/src/common/report.js +274 -0
- package/dist/src/common/report.js.map +1 -0
- package/dist/src/common/temporal.d.ts +71 -0
- package/dist/src/common/temporal.d.ts.map +1 -0
- package/dist/src/common/temporal.js +266 -0
- package/dist/src/common/temporal.js.map +1 -0
- package/dist/src/common/text.d.ts +104 -0
- package/dist/src/common/text.d.ts.map +1 -0
- package/dist/src/common/text.js +255 -0
- package/dist/src/common/text.js.map +1 -0
- package/dist/src/common/weights.d.ts +77 -0
- package/dist/src/common/weights.d.ts.map +1 -0
- package/dist/src/common/weights.js +156 -0
- package/dist/src/common/weights.js.map +1 -0
- package/dist/src/common/writer.d.ts +51 -0
- package/dist/src/common/writer.d.ts.map +1 -0
- package/dist/src/common/writer.js +108 -0
- package/dist/src/common/writer.js.map +1 -0
- package/dist/src/common/xml.d.ts +245 -0
- package/dist/src/common/xml.d.ts.map +1 -0
- package/dist/src/common/xml.js +942 -0
- package/dist/src/common/xml.js.map +1 -0
- package/dist/src/formats/csv/exporter.d.ts +70 -0
- package/dist/src/formats/csv/exporter.d.ts.map +1 -0
- package/dist/src/formats/csv/exporter.js +682 -0
- package/dist/src/formats/csv/exporter.js.map +1 -0
- package/dist/src/formats/csv/header.d.ts +66 -0
- package/dist/src/formats/csv/header.d.ts.map +1 -0
- package/dist/src/formats/csv/header.js +152 -0
- package/dist/src/formats/csv/header.js.map +1 -0
- package/dist/src/formats/csv/importer.d.ts +82 -0
- package/dist/src/formats/csv/importer.d.ts.map +1 -0
- package/dist/src/formats/csv/importer.js +849 -0
- package/dist/src/formats/csv/importer.js.map +1 -0
- package/dist/src/formats/csv/index.d.ts +60 -0
- package/dist/src/formats/csv/index.d.ts.map +1 -0
- package/dist/src/formats/csv/index.js +63 -0
- package/dist/src/formats/csv/index.js.map +1 -0
- package/dist/src/formats/csv/records.d.ts +188 -0
- package/dist/src/formats/csv/records.d.ts.map +1 -0
- package/dist/src/formats/csv/records.js +702 -0
- package/dist/src/formats/csv/records.js.map +1 -0
- package/dist/src/formats/csv/values.d.ts +105 -0
- package/dist/src/formats/csv/values.d.ts.map +1 -0
- package/dist/src/formats/csv/values.js +192 -0
- package/dist/src/formats/csv/values.js.map +1 -0
- package/dist/src/formats/dot/exporter.d.ts +52 -0
- package/dist/src/formats/dot/exporter.d.ts.map +1 -0
- package/dist/src/formats/dot/exporter.js +836 -0
- package/dist/src/formats/dot/exporter.js.map +1 -0
- package/dist/src/formats/dot/importer.d.ts +102 -0
- package/dist/src/formats/dot/importer.d.ts.map +1 -0
- package/dist/src/formats/dot/importer.js +1291 -0
- package/dist/src/formats/dot/importer.js.map +1 -0
- package/dist/src/formats/dot/index.d.ts +7 -0
- package/dist/src/formats/dot/index.d.ts.map +1 -0
- package/dist/src/formats/dot/index.js +7 -0
- package/dist/src/formats/dot/index.js.map +1 -0
- package/dist/src/formats/dot/names.d.ts +29 -0
- package/dist/src/formats/dot/names.d.ts.map +1 -0
- package/dist/src/formats/dot/names.js +28 -0
- package/dist/src/formats/dot/names.js.map +1 -0
- package/dist/src/formats/dot/tokenizer.d.ts +114 -0
- package/dist/src/formats/dot/tokenizer.d.ts.map +1 -0
- package/dist/src/formats/dot/tokenizer.js +341 -0
- package/dist/src/formats/dot/tokenizer.js.map +1 -0
- package/dist/src/formats/gexf/exporter.d.ts +56 -0
- package/dist/src/formats/gexf/exporter.d.ts.map +1 -0
- package/dist/src/formats/gexf/exporter.js +1395 -0
- package/dist/src/formats/gexf/exporter.js.map +1 -0
- package/dist/src/formats/gexf/importer.d.ts +73 -0
- package/dist/src/formats/gexf/importer.d.ts.map +1 -0
- package/dist/src/formats/gexf/importer.js +1880 -0
- package/dist/src/formats/gexf/importer.js.map +1 -0
- package/dist/src/formats/gexf/index.d.ts +96 -0
- package/dist/src/formats/gexf/index.d.ts.map +1 -0
- package/dist/src/formats/gexf/index.js +97 -0
- package/dist/src/formats/gexf/index.js.map +1 -0
- package/dist/src/formats/gexf/schema.d.ts +135 -0
- package/dist/src/formats/gexf/schema.d.ts.map +1 -0
- package/dist/src/formats/gexf/schema.js +323 -0
- package/dist/src/formats/gexf/schema.js.map +1 -0
- package/dist/src/formats/gml/exporter.d.ts +69 -0
- package/dist/src/formats/gml/exporter.d.ts.map +1 -0
- package/dist/src/formats/gml/exporter.js +1093 -0
- package/dist/src/formats/gml/exporter.js.map +1 -0
- package/dist/src/formats/gml/importer.d.ts +66 -0
- package/dist/src/formats/gml/importer.d.ts.map +1 -0
- package/dist/src/formats/gml/importer.js +1331 -0
- package/dist/src/formats/gml/importer.js.map +1 -0
- package/dist/src/formats/gml/index.d.ts +85 -0
- package/dist/src/formats/gml/index.d.ts.map +1 -0
- package/dist/src/formats/gml/index.js +88 -0
- package/dist/src/formats/gml/index.js.map +1 -0
- package/dist/src/formats/gml/syntax.d.ts +186 -0
- package/dist/src/formats/gml/syntax.d.ts.map +1 -0
- package/dist/src/formats/gml/syntax.js +467 -0
- package/dist/src/formats/gml/syntax.js.map +1 -0
- package/dist/src/formats/graphml/constants.d.ts +169 -0
- package/dist/src/formats/graphml/constants.d.ts.map +1 -0
- package/dist/src/formats/graphml/constants.js +165 -0
- package/dist/src/formats/graphml/constants.js.map +1 -0
- package/dist/src/formats/graphml/exporter.d.ts +34 -0
- package/dist/src/formats/graphml/exporter.d.ts.map +1 -0
- package/dist/src/formats/graphml/exporter.js +1176 -0
- package/dist/src/formats/graphml/exporter.js.map +1 -0
- package/dist/src/formats/graphml/importer.d.ts +31 -0
- package/dist/src/formats/graphml/importer.d.ts.map +1 -0
- package/dist/src/formats/graphml/importer.js +1607 -0
- package/dist/src/formats/graphml/importer.js.map +1 -0
- package/dist/src/formats/graphml/index.d.ts +8 -0
- package/dist/src/formats/graphml/index.d.ts.map +1 -0
- package/dist/src/formats/graphml/index.js +8 -0
- package/dist/src/formats/graphml/index.js.map +1 -0
- package/dist/src/formats/graphml/tree.d.ts +72 -0
- package/dist/src/formats/graphml/tree.d.ts.map +1 -0
- package/dist/src/formats/graphml/tree.js +290 -0
- package/dist/src/formats/graphml/tree.js.map +1 -0
- package/dist/src/formats/json/dialect.d.ts +125 -0
- package/dist/src/formats/json/dialect.d.ts.map +1 -0
- package/dist/src/formats/json/dialect.js +262 -0
- package/dist/src/formats/json/dialect.js.map +1 -0
- package/dist/src/formats/json/exporter.d.ts +89 -0
- package/dist/src/formats/json/exporter.d.ts.map +1 -0
- package/dist/src/formats/json/exporter.js +1358 -0
- package/dist/src/formats/json/exporter.js.map +1 -0
- package/dist/src/formats/json/importer.d.ts +108 -0
- package/dist/src/formats/json/importer.d.ts.map +1 -0
- package/dist/src/formats/json/importer.js +1838 -0
- package/dist/src/formats/json/importer.js.map +1 -0
- package/dist/src/formats/json/index.d.ts +8 -0
- package/dist/src/formats/json/index.d.ts.map +1 -0
- package/dist/src/formats/json/index.js +8 -0
- package/dist/src/formats/json/index.js.map +1 -0
- package/dist/src/formats/neo4j/exporter.d.ts +68 -0
- package/dist/src/formats/neo4j/exporter.d.ts.map +1 -0
- package/dist/src/formats/neo4j/exporter.js +1055 -0
- package/dist/src/formats/neo4j/exporter.js.map +1 -0
- package/dist/src/formats/neo4j/header.d.ts +52 -0
- package/dist/src/formats/neo4j/header.d.ts.map +1 -0
- package/dist/src/formats/neo4j/header.js +131 -0
- package/dist/src/formats/neo4j/header.js.map +1 -0
- package/dist/src/formats/neo4j/importer.d.ts +73 -0
- package/dist/src/formats/neo4j/importer.d.ts.map +1 -0
- package/dist/src/formats/neo4j/importer.js +932 -0
- package/dist/src/formats/neo4j/importer.js.map +1 -0
- package/dist/src/formats/neo4j/index.d.ts +79 -0
- package/dist/src/formats/neo4j/index.d.ts.map +1 -0
- package/dist/src/formats/neo4j/index.js +83 -0
- package/dist/src/formats/neo4j/index.js.map +1 -0
- package/dist/src/formats/pajek/exporter.d.ts +58 -0
- package/dist/src/formats/pajek/exporter.d.ts.map +1 -0
- package/dist/src/formats/pajek/exporter.js +825 -0
- package/dist/src/formats/pajek/exporter.js.map +1 -0
- package/dist/src/formats/pajek/importer.d.ts +88 -0
- package/dist/src/formats/pajek/importer.d.ts.map +1 -0
- package/dist/src/formats/pajek/importer.js +1047 -0
- package/dist/src/formats/pajek/importer.js.map +1 -0
- package/dist/src/formats/pajek/index.d.ts +7 -0
- package/dist/src/formats/pajek/index.d.ts.map +1 -0
- package/dist/src/formats/pajek/index.js +7 -0
- package/dist/src/formats/pajek/index.js.map +1 -0
- package/dist/src/formats/pajek/syntax.d.ts +112 -0
- package/dist/src/formats/pajek/syntax.d.ts.map +1 -0
- package/dist/src/formats/pajek/syntax.js +269 -0
- package/dist/src/formats/pajek/syntax.js.map +1 -0
- package/dist/src/index.d.ts +35 -0
- package/dist/src/index.d.ts.map +1 -0
- package/dist/src/index.js +39 -0
- package/dist/src/index.js.map +1 -0
- package/dist/src/registry.d.ts +207 -0
- package/dist/src/registry.d.ts.map +1 -0
- package/dist/src/registry.js +481 -0
- package/dist/src/registry.js.map +1 -0
- package/dist/src/sniff.d.ts +104 -0
- package/dist/src/sniff.d.ts.map +1 -0
- package/dist/src/sniff.js +357 -0
- package/dist/src/sniff.js.map +1 -0
- package/dist/src/types.d.ts +238 -0
- package/dist/src/types.d.ts.map +1 -0
- package/dist/src/types.js +29 -0
- package/dist/src/types.js.map +1 -0
- package/dist/tsconfig.build.tsbuildinfo +1 -0
- package/package.json +122 -7
- package/src/children.ts +335 -0
- package/src/common/attributes.ts +520 -0
- package/src/common/codes.ts +153 -0
- package/src/common/declared-types.ts +374 -0
- package/src/common/direction.ts +518 -0
- package/src/common/escape.ts +231 -0
- package/src/common/export.ts +817 -0
- package/src/common/format.ts +111 -0
- package/src/common/ids.ts +176 -0
- package/src/common/input.ts +378 -0
- package/src/common/lists.ts +196 -0
- package/src/common/options.ts +377 -0
- package/src/common/report.ts +352 -0
- package/src/common/temporal.ts +302 -0
- package/src/common/text.ts +294 -0
- package/src/common/weights.ts +202 -0
- package/src/common/writer.ts +123 -0
- package/src/common/xml.ts +1053 -0
- package/src/formats/csv/exporter.ts +894 -0
- package/src/formats/csv/header.ts +172 -0
- package/src/formats/csv/importer.ts +1104 -0
- package/src/formats/csv/index.ts +88 -0
- package/src/formats/csv/records.ts +813 -0
- package/src/formats/csv/values.ts +224 -0
- package/src/formats/dot/exporter.ts +1014 -0
- package/src/formats/dot/importer.ts +1549 -0
- package/src/formats/dot/index.ts +7 -0
- package/src/formats/dot/names.ts +40 -0
- package/src/formats/dot/tokenizer.ts +384 -0
- package/src/formats/gexf/exporter.ts +1696 -0
- package/src/formats/gexf/importer.ts +2333 -0
- package/src/formats/gexf/index.ts +142 -0
- package/src/formats/gexf/schema.ts +361 -0
- package/src/formats/gml/exporter.ts +1404 -0
- package/src/formats/gml/importer.ts +1591 -0
- package/src/formats/gml/index.ts +128 -0
- package/src/formats/gml/syntax.ts +545 -0
- package/src/formats/graphml/constants.ts +225 -0
- package/src/formats/graphml/exporter.ts +1458 -0
- package/src/formats/graphml/importer.ts +2027 -0
- package/src/formats/graphml/index.ts +8 -0
- package/src/formats/graphml/tree.ts +318 -0
- package/src/formats/json/dialect.ts +317 -0
- package/src/formats/json/exporter.ts +1616 -0
- package/src/formats/json/importer.ts +2271 -0
- package/src/formats/json/index.ts +8 -0
- package/src/formats/neo4j/exporter.ts +1287 -0
- package/src/formats/neo4j/header.ts +156 -0
- package/src/formats/neo4j/importer.ts +1220 -0
- package/src/formats/neo4j/index.ts +116 -0
- package/src/formats/pajek/exporter.ts +1000 -0
- package/src/formats/pajek/importer.ts +1311 -0
- package/src/formats/pajek/index.ts +7 -0
- package/src/formats/pajek/syntax.ts +307 -0
- package/src/index.ts +244 -0
- package/src/registry.ts +617 -0
- package/src/sniff.ts +397 -0
- package/src/types.ts +262 -0
package/dist/csv.js
ADDED
|
@@ -0,0 +1,1702 @@
|
|
|
1
|
+
import { aY as DICT_SAMPLE_ROWS, aX as DictHeuristic, V as ROLE_TAKEN_CODE$1, n as ID_MERGED_CODE$1, i as DUPLICATE_EDGE_ID_CODE$1, k as DUPLICATE_NODE_CODE$1, w as MISSING_ID_CODE$1, M as MISSING_ENDPOINT_CODE$1, m as EMPTY_INPUT_CODE$1, aC as resolveImportOptions, I as ImportReportBuilder, az as reportSinkOptions, aA as reportUnusedOptions, l as DirectionResolver, s as IdCoercer, a as throwIfAborted, ax as parseWeightText, aI as uniqueColumnName, S as RENAMED_CODE, af as declareResolved, a9 as capabilities, L as LOSS, au as joinText, ah as encodeChunks, aa as checkCapabilities, aB as resolveExportOptions, aw as pairFolding, aG as countMixedEdges, ai as explicitWeights, aV as joinListText, an as formatInteger, aj as formatDecimal, a8 as canonicalId, x as MIXED_DIRECTION_CODE, g as DIRECTION_FORCED_CODE, h as DIRECTION_REFUSED_CODE, W as SINK_OPTION_CODE, J as OPTION_IGNORED_CODE, C as COLUMN_RENAMED_CODE, r as INVALID_UTF8_CODE } from "./chunks/writer-DxSKC7TL.js";
|
|
2
|
+
import { i as inferTextDtype, T as TextCellWriter, p as parseTextCell, W as WIDENING_UNSUPPORTED_CODE } from "./chunks/text-CajMdVFy.js";
|
|
3
|
+
import { GraphFormatError, INVALID_INDEX } from "@graphty/graph-format";
|
|
4
|
+
import { s as sniffNewline, a as sniffDelimiter, C as CsvRecordReader, B as BAD_QUOTE_CODE, U as UNCLOSED_QUOTE_CODE } from "./chunks/records-CGpxszm1.js";
|
|
5
|
+
import { m as quoteCsvCell } from "./chunks/escape-DyI8JofU.js";
|
|
6
|
+
const SOURCE_NAMES = Object.freeze([
|
|
7
|
+
"source",
|
|
8
|
+
"src",
|
|
9
|
+
"from",
|
|
10
|
+
"start",
|
|
11
|
+
"source_id",
|
|
12
|
+
"sourceid",
|
|
13
|
+
"fromnodeid",
|
|
14
|
+
"start_id",
|
|
15
|
+
":start_id"
|
|
16
|
+
]);
|
|
17
|
+
const TARGET_NAMES = Object.freeze([
|
|
18
|
+
"target",
|
|
19
|
+
"dst",
|
|
20
|
+
"dest",
|
|
21
|
+
"to",
|
|
22
|
+
"end",
|
|
23
|
+
"target_id",
|
|
24
|
+
"targetid",
|
|
25
|
+
"tonodeid",
|
|
26
|
+
"end_id",
|
|
27
|
+
":end_id"
|
|
28
|
+
]);
|
|
29
|
+
const ID_NAMES = Object.freeze(["id", "node", "name", "key"]);
|
|
30
|
+
const LABEL_NAMES = Object.freeze(["label"]);
|
|
31
|
+
const EDGE_ID_NAMES = Object.freeze(["id"]);
|
|
32
|
+
const TYPE_NAME = "Type";
|
|
33
|
+
const HEADER_MARKERS = /* @__PURE__ */ new Set([
|
|
34
|
+
...SOURCE_NAMES,
|
|
35
|
+
...TARGET_NAMES,
|
|
36
|
+
...ID_NAMES,
|
|
37
|
+
...LABEL_NAMES,
|
|
38
|
+
"weight",
|
|
39
|
+
"type",
|
|
40
|
+
"value"
|
|
41
|
+
]);
|
|
42
|
+
function findColumn(names, candidates) {
|
|
43
|
+
for (const candidate of candidates) {
|
|
44
|
+
const exact = names.indexOf(candidate);
|
|
45
|
+
if (exact >= 0) {
|
|
46
|
+
return exact;
|
|
47
|
+
}
|
|
48
|
+
const lower2 = candidate.toLowerCase();
|
|
49
|
+
const loose = names.findIndex((name) => name.toLowerCase() === lower2);
|
|
50
|
+
if (loose >= 0) {
|
|
51
|
+
return loose;
|
|
52
|
+
}
|
|
53
|
+
}
|
|
54
|
+
return -1;
|
|
55
|
+
}
|
|
56
|
+
function resolveColumnRef(names, ref, option) {
|
|
57
|
+
if (typeof ref === "number") {
|
|
58
|
+
if (!Number.isInteger(ref) || ref < 0 || ref >= names.length) {
|
|
59
|
+
throw new GraphFormatError(
|
|
60
|
+
"E_UNSUPPORTED",
|
|
61
|
+
`option ${option}: column ${ref} does not exist (the file has ${names.length} column(s))`,
|
|
62
|
+
{ option, found: ref, columns: names.length }
|
|
63
|
+
);
|
|
64
|
+
}
|
|
65
|
+
return ref;
|
|
66
|
+
}
|
|
67
|
+
const index = findColumn(names, [ref]);
|
|
68
|
+
if (index < 0) {
|
|
69
|
+
throw new GraphFormatError("E_UNSUPPORTED", `option ${option}: no column named ${JSON.stringify(ref)}`, {
|
|
70
|
+
option,
|
|
71
|
+
found: ref,
|
|
72
|
+
columns: [...names]
|
|
73
|
+
});
|
|
74
|
+
}
|
|
75
|
+
return index;
|
|
76
|
+
}
|
|
77
|
+
function looksLikeHeader(first, second) {
|
|
78
|
+
let allText = true;
|
|
79
|
+
for (const cell of first) {
|
|
80
|
+
const text = cell.trim();
|
|
81
|
+
if (HEADER_MARKERS.has(text.toLowerCase())) {
|
|
82
|
+
return true;
|
|
83
|
+
}
|
|
84
|
+
if (text.length === 0 || inferTextDtype(text) !== "string") {
|
|
85
|
+
allText = false;
|
|
86
|
+
}
|
|
87
|
+
}
|
|
88
|
+
if (!allText || second === null) {
|
|
89
|
+
return false;
|
|
90
|
+
}
|
|
91
|
+
return second.some((cell) => cell.trim().length > 0 && inferTextDtype(cell.trim()) !== "string");
|
|
92
|
+
}
|
|
93
|
+
function positionalNames(width) {
|
|
94
|
+
const names = [];
|
|
95
|
+
for (let i = 1; i <= width; i++) {
|
|
96
|
+
names.push(`column${i}`);
|
|
97
|
+
}
|
|
98
|
+
return names;
|
|
99
|
+
}
|
|
100
|
+
function headerNames(cells) {
|
|
101
|
+
return cells.map((cell, i) => {
|
|
102
|
+
const name = cell.trim();
|
|
103
|
+
return name.length === 0 ? `column${i + 1}` : name;
|
|
104
|
+
});
|
|
105
|
+
}
|
|
106
|
+
class InferredColumn {
|
|
107
|
+
/**
|
|
108
|
+
* Create a column writer; nothing is declared until the first value or finish().
|
|
109
|
+
* @param name - the column name
|
|
110
|
+
* @param domain - node or edge
|
|
111
|
+
* @param sink - the sink
|
|
112
|
+
* @param report - the report a refused sampled value is recorded in (the flush has no line numbers)
|
|
113
|
+
* @param sampleRows - rows sampled before the dict decision; DICT_SAMPLE_ROWS by default
|
|
114
|
+
*/
|
|
115
|
+
constructor(name, domain, sink, report, sampleRows = DICT_SAMPLE_ROWS) {
|
|
116
|
+
this.handle = INVALID_INDEX;
|
|
117
|
+
this.text = false;
|
|
118
|
+
this.dict = false;
|
|
119
|
+
this.decided = false;
|
|
120
|
+
this.candidate = true;
|
|
121
|
+
this.writer = null;
|
|
122
|
+
this.pendingRows = [];
|
|
123
|
+
this.pendingTexts = [];
|
|
124
|
+
this.name = name;
|
|
125
|
+
this.domain = domain;
|
|
126
|
+
this.sink = sink;
|
|
127
|
+
this.report = report;
|
|
128
|
+
this.heuristic = new DictHeuristic(sampleRows);
|
|
129
|
+
}
|
|
130
|
+
/**
|
|
131
|
+
* Whether the column was declared as a dict.
|
|
132
|
+
* @returns true after a dict decision
|
|
133
|
+
*/
|
|
134
|
+
get isDict() {
|
|
135
|
+
return this.dict;
|
|
136
|
+
}
|
|
137
|
+
/**
|
|
138
|
+
* Write one set cell.
|
|
139
|
+
* @param row - the node or edge index
|
|
140
|
+
* @param text - the cell text (the empty string for a quoted empty cell)
|
|
141
|
+
*/
|
|
142
|
+
write(row, text) {
|
|
143
|
+
if (this.decided) {
|
|
144
|
+
this.push(row, text);
|
|
145
|
+
return;
|
|
146
|
+
}
|
|
147
|
+
this.pendingRows.push(row);
|
|
148
|
+
this.pendingTexts.push(text);
|
|
149
|
+
if (this.candidate && inferTextDtype(text) !== "string") {
|
|
150
|
+
this.candidate = false;
|
|
151
|
+
}
|
|
152
|
+
const full = this.heuristic.observe(text);
|
|
153
|
+
if (full || !this.candidate) {
|
|
154
|
+
this.decide();
|
|
155
|
+
}
|
|
156
|
+
}
|
|
157
|
+
/**
|
|
158
|
+
* Decide an undecided column from what was sampled (or declare an all-empty column as string)
|
|
159
|
+
* and flush the sample. Call once at the end of the input.
|
|
160
|
+
*/
|
|
161
|
+
finish() {
|
|
162
|
+
if (!this.decided) {
|
|
163
|
+
this.decide();
|
|
164
|
+
}
|
|
165
|
+
if (this.handle === INVALID_INDEX) {
|
|
166
|
+
this.handle = this.lookup();
|
|
167
|
+
if (this.handle === INVALID_INDEX) {
|
|
168
|
+
this.handle = this.declare("string");
|
|
169
|
+
}
|
|
170
|
+
}
|
|
171
|
+
}
|
|
172
|
+
/**
|
|
173
|
+
* Make the dtype decision: a dict when every sampled value is text and the cardinality is low
|
|
174
|
+
* and no column of the name exists yet; inference otherwise. A column the sink already holds
|
|
175
|
+
* (a caller's) receives the cell text when it is a string or dict column and the parsed value
|
|
176
|
+
* otherwise. The sample is flushed either way; a sampled value the sink refuses (a caller's
|
|
177
|
+
* typed column of the name) is recorded as an issue of this column and the flush continues.
|
|
178
|
+
*/
|
|
179
|
+
decide() {
|
|
180
|
+
this.decided = true;
|
|
181
|
+
const existing = this.lookup();
|
|
182
|
+
if (existing !== INVALID_INDEX) {
|
|
183
|
+
this.handle = existing;
|
|
184
|
+
this.text = this.holdsText();
|
|
185
|
+
} else if (this.candidate && this.heuristic.decide() === "dict") {
|
|
186
|
+
this.dict = true;
|
|
187
|
+
this.text = true;
|
|
188
|
+
this.handle = this.declare("dict");
|
|
189
|
+
} else {
|
|
190
|
+
this.writer = new TextCellWriter(this.name, this.domain, this.sink, this.report);
|
|
191
|
+
}
|
|
192
|
+
const rows = this.pendingRows;
|
|
193
|
+
const texts = this.pendingTexts;
|
|
194
|
+
for (let i = 0; i < rows.length; i++) {
|
|
195
|
+
try {
|
|
196
|
+
this.push(rows[i], texts[i]);
|
|
197
|
+
} catch (err) {
|
|
198
|
+
this.report.recordError(err, { element: this.name });
|
|
199
|
+
}
|
|
200
|
+
}
|
|
201
|
+
rows.length = 0;
|
|
202
|
+
texts.length = 0;
|
|
203
|
+
}
|
|
204
|
+
/**
|
|
205
|
+
* Push one value: text into a dict or a caller's text column, the parsed scalar into a caller's
|
|
206
|
+
* typed column, and everything else through the inferred-column writer.
|
|
207
|
+
* @param row - the node or edge index
|
|
208
|
+
* @param text - the cell text
|
|
209
|
+
*/
|
|
210
|
+
push(row, text) {
|
|
211
|
+
if (this.writer !== null) {
|
|
212
|
+
this.writer.write(row, text);
|
|
213
|
+
this.handle = this.writer.column;
|
|
214
|
+
return;
|
|
215
|
+
}
|
|
216
|
+
const value = this.text ? text : parseTextCell(text);
|
|
217
|
+
this.set(this.handle, row, value);
|
|
218
|
+
}
|
|
219
|
+
/**
|
|
220
|
+
* Set a cell in the sink's table for this domain.
|
|
221
|
+
* @param column - the handle or name
|
|
222
|
+
* @param row - the row
|
|
223
|
+
* @param value - the value
|
|
224
|
+
*/
|
|
225
|
+
set(column, row, value) {
|
|
226
|
+
if (this.domain === "node") {
|
|
227
|
+
this.sink.setNodeValue(column, row, value);
|
|
228
|
+
} else {
|
|
229
|
+
this.sink.setEdgeValue(column, row, value);
|
|
230
|
+
}
|
|
231
|
+
}
|
|
232
|
+
/**
|
|
233
|
+
* The sink's handle for this column's name.
|
|
234
|
+
* @returns the handle, or INVALID_INDEX when the sink has no such column yet
|
|
235
|
+
*/
|
|
236
|
+
lookup() {
|
|
237
|
+
return this.domain === "node" ? this.sink.nodeColumn(this.name) : this.sink.edgeColumn(this.name);
|
|
238
|
+
}
|
|
239
|
+
/**
|
|
240
|
+
* Whether the sink's existing column of this name is a string or dict column, probed through
|
|
241
|
+
* the sink's declare (the same shape returns the existing handle, another is E_COLUMN_EXISTS).
|
|
242
|
+
* @returns true when cell text is what the column holds
|
|
243
|
+
*/
|
|
244
|
+
holdsText() {
|
|
245
|
+
for (const dtype of ["string", "dict"]) {
|
|
246
|
+
try {
|
|
247
|
+
this.declare(dtype);
|
|
248
|
+
return true;
|
|
249
|
+
} catch (err) {
|
|
250
|
+
if (!(err instanceof GraphFormatError) || err.code !== "E_COLUMN_EXISTS") {
|
|
251
|
+
throw err;
|
|
252
|
+
}
|
|
253
|
+
}
|
|
254
|
+
}
|
|
255
|
+
return false;
|
|
256
|
+
}
|
|
257
|
+
/**
|
|
258
|
+
* Declare the column with a fixed dtype.
|
|
259
|
+
* @param dtype - dict or string
|
|
260
|
+
* @returns the handle
|
|
261
|
+
*/
|
|
262
|
+
declare(dtype) {
|
|
263
|
+
const decl = { name: this.name, dtype, nullable: true, origin: { format: "csv" } };
|
|
264
|
+
return this.domain === "node" ? this.sink.declareNodeColumn(decl) : this.sink.declareEdgeColumn(decl);
|
|
265
|
+
}
|
|
266
|
+
}
|
|
267
|
+
const EMPTY_INPUT_CODE = EMPTY_INPUT_CODE$1;
|
|
268
|
+
const NO_ENDPOINT_COLUMNS_CODE = "E_CSV_NO_ENDPOINT_COLUMNS";
|
|
269
|
+
const NO_ID_COLUMN_CODE = "E_CSV_NO_ID_COLUMN";
|
|
270
|
+
const FIELD_COUNT_CODE = "E_CSV_FIELD_COUNT";
|
|
271
|
+
const MISSING_ENDPOINT_CODE = MISSING_ENDPOINT_CODE$1;
|
|
272
|
+
const MISSING_ID_CODE = MISSING_ID_CODE$1;
|
|
273
|
+
const BAD_TYPE_CODE = "E_CSV_BAD_TYPE";
|
|
274
|
+
const NO_DATA_ROWS_CODE = "W_CSV_NO_DATA_ROWS";
|
|
275
|
+
const DUPLICATE_NODE_CODE = DUPLICATE_NODE_CODE$1;
|
|
276
|
+
const ID_MERGED_CODE = ID_MERGED_CODE$1;
|
|
277
|
+
const COLUMN_MISSING_CODE = "W_CSV_COLUMN_MISSING";
|
|
278
|
+
const ROLE_TAKEN_CODE = ROLE_TAKEN_CODE$1;
|
|
279
|
+
const DUPLICATE_EDGE_ID_CODE = DUPLICATE_EDGE_ID_CODE$1;
|
|
280
|
+
const TABLE_MODES = /* @__PURE__ */ new Set(["edges", "nodes", "auto"]);
|
|
281
|
+
const USED_OPTIONS = /* @__PURE__ */ new Set([
|
|
282
|
+
"ids",
|
|
283
|
+
"addMissingNodes",
|
|
284
|
+
"duplicateEdges",
|
|
285
|
+
"selfLoops",
|
|
286
|
+
"onMixedDirection",
|
|
287
|
+
"defaultDirected",
|
|
288
|
+
"weightFrom",
|
|
289
|
+
"weightDtype",
|
|
290
|
+
"errorLimit",
|
|
291
|
+
"signal",
|
|
292
|
+
"onProgress"
|
|
293
|
+
]);
|
|
294
|
+
const USED_OPTIONS_WITH_NODES = /* @__PURE__ */ new Set([
|
|
295
|
+
...USED_OPTIONS,
|
|
296
|
+
"nodeIdFrom"
|
|
297
|
+
]);
|
|
298
|
+
const BAD_DELIMITERS = /* @__PURE__ */ new Set(['"', "\n", "\r"]);
|
|
299
|
+
const COMMENT_CHARS = Object.freeze(["#", "%"]);
|
|
300
|
+
function commentDirection(comments) {
|
|
301
|
+
for (const comment of comments) {
|
|
302
|
+
const text = comment.slice(1).trim().toLowerCase();
|
|
303
|
+
if (text.startsWith("directed graph") || text.startsWith("asym")) {
|
|
304
|
+
return true;
|
|
305
|
+
}
|
|
306
|
+
if (text.startsWith("undirected graph") || text.startsWith("sym") || text.startsWith("bip")) {
|
|
307
|
+
return false;
|
|
308
|
+
}
|
|
309
|
+
}
|
|
310
|
+
return null;
|
|
311
|
+
}
|
|
312
|
+
function resolveCsvOptions(options) {
|
|
313
|
+
const o = options ?? {};
|
|
314
|
+
if (o.delimiter !== void 0 && (typeof o.delimiter !== "string" || o.delimiter.length === 0)) {
|
|
315
|
+
throw new GraphFormatError("E_UNSUPPORTED", "option delimiter: expected a non-empty string", {
|
|
316
|
+
option: "delimiter",
|
|
317
|
+
found: o.delimiter
|
|
318
|
+
});
|
|
319
|
+
}
|
|
320
|
+
if (o.delimiter !== void 0 && BAD_DELIMITERS.has(o.delimiter)) {
|
|
321
|
+
throw new GraphFormatError("E_UNSUPPORTED", "option delimiter: a quote or a line break cannot delimit", {
|
|
322
|
+
option: "delimiter",
|
|
323
|
+
found: o.delimiter
|
|
324
|
+
});
|
|
325
|
+
}
|
|
326
|
+
if (o.header !== void 0 && o.header !== "auto" && typeof o.header !== "boolean") {
|
|
327
|
+
throw new GraphFormatError("E_UNSUPPORTED", 'option header: expected true, false or "auto"', {
|
|
328
|
+
option: "header",
|
|
329
|
+
found: o.header
|
|
330
|
+
});
|
|
331
|
+
}
|
|
332
|
+
if (o.table !== void 0 && !TABLE_MODES.has(o.table)) {
|
|
333
|
+
throw new GraphFormatError("E_UNSUPPORTED", 'option table: expected "edges", "nodes" or "auto"', {
|
|
334
|
+
option: "table",
|
|
335
|
+
found: o.table
|
|
336
|
+
});
|
|
337
|
+
}
|
|
338
|
+
for (const name of ["sourceColumn", "targetColumn", "idColumn"]) {
|
|
339
|
+
checkColumnRef(name, o[name]);
|
|
340
|
+
}
|
|
341
|
+
if (o.typeColumn !== null) {
|
|
342
|
+
checkColumnRef("typeColumn", o.typeColumn);
|
|
343
|
+
}
|
|
344
|
+
return {
|
|
345
|
+
delimiter: o.delimiter ?? null,
|
|
346
|
+
header: o.header ?? "auto",
|
|
347
|
+
table: o.table ?? "auto",
|
|
348
|
+
sourceColumn: o.sourceColumn ?? null,
|
|
349
|
+
targetColumn: o.targetColumn ?? null,
|
|
350
|
+
typeColumn: o.typeColumn,
|
|
351
|
+
idColumn: o.idColumn ?? null,
|
|
352
|
+
nodes: o.nodes ?? null
|
|
353
|
+
};
|
|
354
|
+
}
|
|
355
|
+
function checkColumnRef(name, value) {
|
|
356
|
+
if (value === void 0) {
|
|
357
|
+
return;
|
|
358
|
+
}
|
|
359
|
+
if (typeof value === "string" && value.length > 0) {
|
|
360
|
+
return;
|
|
361
|
+
}
|
|
362
|
+
if (typeof value === "number" && Number.isInteger(value) && value >= 0) {
|
|
363
|
+
return;
|
|
364
|
+
}
|
|
365
|
+
throw new GraphFormatError("E_UNSUPPORTED", `option ${name}: expected a column name or a 0-based position`, {
|
|
366
|
+
option: name,
|
|
367
|
+
found: value
|
|
368
|
+
});
|
|
369
|
+
}
|
|
370
|
+
function isBlank(text) {
|
|
371
|
+
return text.length === 0 || text.trim().length === 0;
|
|
372
|
+
}
|
|
373
|
+
function isUnset(text, quoted) {
|
|
374
|
+
return quoted !== true && isBlank(text);
|
|
375
|
+
}
|
|
376
|
+
const ABORT_CHECK_INTERVAL = 64;
|
|
377
|
+
function uniqueNames(names, report, line) {
|
|
378
|
+
const seen = /* @__PURE__ */ new Set();
|
|
379
|
+
return names.map((name, i) => {
|
|
380
|
+
const unique = uniqueColumnName(name, String(i + 1), (n) => seen.has(n));
|
|
381
|
+
seen.add(unique);
|
|
382
|
+
if (unique !== name) {
|
|
383
|
+
report.warning(
|
|
384
|
+
"coercion",
|
|
385
|
+
RENAMED_CODE,
|
|
386
|
+
`column ${i + 1} "${name}" renamed to "${unique}": the header repeats the name`,
|
|
387
|
+
{ line, element: name }
|
|
388
|
+
);
|
|
389
|
+
}
|
|
390
|
+
return unique;
|
|
391
|
+
});
|
|
392
|
+
}
|
|
393
|
+
function findFree(names, candidates, claimed) {
|
|
394
|
+
const masked = names.map((name, i) => claimed.has(i) ? "" : name);
|
|
395
|
+
return findColumn(masked, candidates);
|
|
396
|
+
}
|
|
397
|
+
function declareRoleColumn(sink, domain, decl, position, report, line) {
|
|
398
|
+
const withOrigin = { ...decl, origin: { ...decl.origin, id: String(position) } };
|
|
399
|
+
const resolved = declareResolved(sink, domain, withOrigin, report, { line, element: decl.name });
|
|
400
|
+
if (resolved.roleDropped) {
|
|
401
|
+
return resolved.handle;
|
|
402
|
+
}
|
|
403
|
+
return resolved.handle;
|
|
404
|
+
}
|
|
405
|
+
function parseKind(text) {
|
|
406
|
+
if (isBlank(text)) {
|
|
407
|
+
return void 0;
|
|
408
|
+
}
|
|
409
|
+
switch (text.trim().toLowerCase()) {
|
|
410
|
+
case "directed":
|
|
411
|
+
return "directed";
|
|
412
|
+
case "undirected":
|
|
413
|
+
return "undirected";
|
|
414
|
+
case "mutual":
|
|
415
|
+
return "mutual";
|
|
416
|
+
default:
|
|
417
|
+
return null;
|
|
418
|
+
}
|
|
419
|
+
}
|
|
420
|
+
class TableReader {
|
|
421
|
+
/**
|
|
422
|
+
* Create a table reader.
|
|
423
|
+
* @param state - the import state
|
|
424
|
+
* @param input - the table's input
|
|
425
|
+
* @param kind - what the table is, or "auto"
|
|
426
|
+
* @param progress - whether this table reports byte progress
|
|
427
|
+
*/
|
|
428
|
+
constructor(state, input, kind, progress) {
|
|
429
|
+
this.plan = null;
|
|
430
|
+
this.writers = [];
|
|
431
|
+
this.idHandle = INVALID_INDEX;
|
|
432
|
+
this.labelHandle = INVALID_INDEX;
|
|
433
|
+
this.dataRows = 0;
|
|
434
|
+
this.nodeOrdinal = 0;
|
|
435
|
+
this.edgeIds = /* @__PURE__ */ new Set();
|
|
436
|
+
this.where = { line: null, element: null };
|
|
437
|
+
this.state = state;
|
|
438
|
+
this.kind = kind;
|
|
439
|
+
const readerOptions = {
|
|
440
|
+
delimiter: state.csv.delimiter,
|
|
441
|
+
comments: COMMENT_CHARS,
|
|
442
|
+
signal: state.common.signal,
|
|
443
|
+
onProgress: progress ? state.common.onProgress : null
|
|
444
|
+
};
|
|
445
|
+
this.reader = new CsvRecordReader(input, state.report, readerOptions);
|
|
446
|
+
}
|
|
447
|
+
/**
|
|
448
|
+
* Read every row; the reader is closed (and a stream cancelled) when the import aborts midway.
|
|
449
|
+
*/
|
|
450
|
+
async read() {
|
|
451
|
+
const iterator = this.reader[Symbol.asyncIterator]();
|
|
452
|
+
try {
|
|
453
|
+
await this.readRows(iterator);
|
|
454
|
+
} finally {
|
|
455
|
+
await iterator.return(void 0);
|
|
456
|
+
}
|
|
457
|
+
}
|
|
458
|
+
/**
|
|
459
|
+
* Read the header (or decide there is none), resolve the plan, then push every row.
|
|
460
|
+
* @param iterator - the record iterator
|
|
461
|
+
*/
|
|
462
|
+
async readRows(iterator) {
|
|
463
|
+
const { report } = this.state;
|
|
464
|
+
const first = await iterator.next();
|
|
465
|
+
const firstRow = first.done ? report.fail(EMPTY_INPUT_CODE, "the input is empty: no header row and no records") : first.value;
|
|
466
|
+
const firstLine = this.reader.line;
|
|
467
|
+
const firstQuoted = this.reader.quoted.slice(0, firstRow.length);
|
|
468
|
+
const pending = [];
|
|
469
|
+
let header;
|
|
470
|
+
const { header: mode } = this.state.csv;
|
|
471
|
+
if (mode === "auto") {
|
|
472
|
+
const second = await iterator.next();
|
|
473
|
+
const secondRow = second.done ? null : second.value;
|
|
474
|
+
const secondLine = this.reader.line;
|
|
475
|
+
const secondQuoted = this.reader.quoted.slice(0, secondRow?.length ?? 0);
|
|
476
|
+
header = looksLikeHeader(firstRow, secondRow);
|
|
477
|
+
if (!header) {
|
|
478
|
+
pending.push({ row: firstRow, quoted: firstQuoted, line: firstLine });
|
|
479
|
+
}
|
|
480
|
+
if (secondRow !== null) {
|
|
481
|
+
pending.push({ row: secondRow, quoted: secondQuoted, line: secondLine });
|
|
482
|
+
}
|
|
483
|
+
} else {
|
|
484
|
+
header = mode;
|
|
485
|
+
if (!header) {
|
|
486
|
+
pending.push({ row: firstRow, quoted: firstQuoted, line: firstLine });
|
|
487
|
+
}
|
|
488
|
+
}
|
|
489
|
+
const names = header ? uniqueNames(headerNames(firstRow), report, firstLine) : positionalNames(firstRow.length);
|
|
490
|
+
if (this.kind !== "nodes" && this.state.commentDirected === null) {
|
|
491
|
+
this.state.commentDirected = commentDirection(this.reader.leadingComments);
|
|
492
|
+
}
|
|
493
|
+
this.plan = this.resolvePlan(names, header, firstLine);
|
|
494
|
+
this.prepareColumns(firstLine);
|
|
495
|
+
for (const { row, quoted, line } of pending) {
|
|
496
|
+
this.processRow(row, quoted, line);
|
|
497
|
+
}
|
|
498
|
+
const { signal } = this.state.common;
|
|
499
|
+
let sinceCheck = 0;
|
|
500
|
+
for (; ; ) {
|
|
501
|
+
const next = await iterator.next();
|
|
502
|
+
if (next.done) {
|
|
503
|
+
break;
|
|
504
|
+
}
|
|
505
|
+
this.processRow(next.value, this.reader.quoted, this.reader.line);
|
|
506
|
+
if (++sinceCheck >= ABORT_CHECK_INTERVAL) {
|
|
507
|
+
sinceCheck = 0;
|
|
508
|
+
throwIfAborted(signal);
|
|
509
|
+
}
|
|
510
|
+
}
|
|
511
|
+
for (const writer of this.writers) {
|
|
512
|
+
writer?.finish();
|
|
513
|
+
}
|
|
514
|
+
if (this.dataRows === 0 && header) {
|
|
515
|
+
report.warning("missing-value", NO_DATA_ROWS_CODE, "the table has a header and no data rows", {
|
|
516
|
+
line: firstLine
|
|
517
|
+
});
|
|
518
|
+
}
|
|
519
|
+
}
|
|
520
|
+
/**
|
|
521
|
+
* Decide the table's columns from its header.
|
|
522
|
+
* @param names - the column names
|
|
523
|
+
* @param header - whether the file has a header row
|
|
524
|
+
* @param line - the header line
|
|
525
|
+
* @returns the plan; the import aborts when no endpoints (or id) resolve
|
|
526
|
+
*/
|
|
527
|
+
resolvePlan(names, header, line) {
|
|
528
|
+
const { csv, report } = this.state;
|
|
529
|
+
const width = names.length;
|
|
530
|
+
let source = -1;
|
|
531
|
+
let target = -1;
|
|
532
|
+
if (this.kind !== "nodes") {
|
|
533
|
+
if (csv.sourceColumn !== null) {
|
|
534
|
+
source = resolveColumnRef(names, csv.sourceColumn, "sourceColumn");
|
|
535
|
+
} else if (header) {
|
|
536
|
+
source = findColumn(names, SOURCE_NAMES);
|
|
537
|
+
} else {
|
|
538
|
+
source = width >= 2 ? 0 : -1;
|
|
539
|
+
}
|
|
540
|
+
if (csv.targetColumn !== null) {
|
|
541
|
+
target = resolveColumnRef(names, csv.targetColumn, "targetColumn");
|
|
542
|
+
} else if (header) {
|
|
543
|
+
target = findColumn(names, TARGET_NAMES);
|
|
544
|
+
} else {
|
|
545
|
+
target = width >= 2 ? 1 : -1;
|
|
546
|
+
}
|
|
547
|
+
if (source >= 0 && target >= 0 && source === target) {
|
|
548
|
+
throw new GraphFormatError("E_UNSUPPORTED", "sourceColumn and targetColumn name the same column", {
|
|
549
|
+
option: "targetColumn",
|
|
550
|
+
found: names[target]
|
|
551
|
+
});
|
|
552
|
+
}
|
|
553
|
+
}
|
|
554
|
+
if (source >= 0 && target >= 0) {
|
|
555
|
+
return this.edgePlan(names, header, source, target);
|
|
556
|
+
}
|
|
557
|
+
const shown = names.map((n) => JSON.stringify(n)).join(", ");
|
|
558
|
+
if (this.kind === "edges" || this.kind === "auto" && (csv.sourceColumn !== null || csv.targetColumn !== null)) {
|
|
559
|
+
report.fail(
|
|
560
|
+
NO_ENDPOINT_COLUMNS_CODE,
|
|
561
|
+
`no source / target columns in the header (${shown}); a node table goes in the nodes option`,
|
|
562
|
+
{ line },
|
|
563
|
+
{ columns: [...names] }
|
|
564
|
+
);
|
|
565
|
+
}
|
|
566
|
+
const idResolves = header ? findColumn(names, ID_NAMES) >= 0 : width >= 1;
|
|
567
|
+
if (this.kind === "auto" && csv.idColumn === null && !idResolves) {
|
|
568
|
+
report.fail(
|
|
569
|
+
NO_ENDPOINT_COLUMNS_CODE,
|
|
570
|
+
`no source / target columns and no id column in the header (${shown}); the input is neither an edge table nor a node table`,
|
|
571
|
+
{ line },
|
|
572
|
+
{ columns: [...names] }
|
|
573
|
+
);
|
|
574
|
+
}
|
|
575
|
+
return this.nodePlan(names, header, line);
|
|
576
|
+
}
|
|
577
|
+
/**
|
|
578
|
+
* The columns of an edge table.
|
|
579
|
+
* @param names - the column names
|
|
580
|
+
* @param header - whether the file has a header row
|
|
581
|
+
* @param source - the source column
|
|
582
|
+
* @param target - the target column
|
|
583
|
+
* @returns the plan
|
|
584
|
+
*/
|
|
585
|
+
edgePlan(names, header, source, target) {
|
|
586
|
+
const { csv, common, report } = this.state;
|
|
587
|
+
const claimed = /* @__PURE__ */ new Set([source, target]);
|
|
588
|
+
let weight = -1;
|
|
589
|
+
if (common.weightFrom !== null) {
|
|
590
|
+
if (header) {
|
|
591
|
+
weight = findFree(names, [common.weightFrom], claimed);
|
|
592
|
+
if (weight < 0 && this.state.weightFromExplicit) {
|
|
593
|
+
report.warning(
|
|
594
|
+
"missing-value",
|
|
595
|
+
COLUMN_MISSING_CODE,
|
|
596
|
+
`weight column ${JSON.stringify(common.weightFrom)} not found; edges are unweighted`,
|
|
597
|
+
{ line: this.reader.line, element: common.weightFrom }
|
|
598
|
+
);
|
|
599
|
+
}
|
|
600
|
+
} else if (names.length >= 3 && !claimed.has(2)) {
|
|
601
|
+
weight = 2;
|
|
602
|
+
}
|
|
603
|
+
}
|
|
604
|
+
if (weight >= 0) {
|
|
605
|
+
claimed.add(weight);
|
|
606
|
+
}
|
|
607
|
+
let type = -1;
|
|
608
|
+
if (csv.typeColumn === void 0) {
|
|
609
|
+
if (header && names[source] === "Source" && names[target] === "Target") {
|
|
610
|
+
type = findFree(names, [TYPE_NAME], claimed);
|
|
611
|
+
if (type >= 0 && names[type] !== TYPE_NAME) {
|
|
612
|
+
type = -1;
|
|
613
|
+
}
|
|
614
|
+
}
|
|
615
|
+
} else if (csv.typeColumn !== null) {
|
|
616
|
+
type = resolveColumnRef(names, csv.typeColumn, "typeColumn");
|
|
617
|
+
if (claimed.has(type)) {
|
|
618
|
+
throw new GraphFormatError("E_UNSUPPORTED", "typeColumn names an endpoint or weight column", {
|
|
619
|
+
option: "typeColumn",
|
|
620
|
+
found: names[type]
|
|
621
|
+
});
|
|
622
|
+
}
|
|
623
|
+
}
|
|
624
|
+
if (type >= 0) {
|
|
625
|
+
claimed.add(type);
|
|
626
|
+
}
|
|
627
|
+
const id = header ? findFree(names, EDGE_ID_NAMES, claimed) : -1;
|
|
628
|
+
if (id >= 0) {
|
|
629
|
+
claimed.add(id);
|
|
630
|
+
}
|
|
631
|
+
const label = header ? findFree(names, LABEL_NAMES, claimed) : -1;
|
|
632
|
+
if (label >= 0) {
|
|
633
|
+
claimed.add(label);
|
|
634
|
+
}
|
|
635
|
+
const attributes = [];
|
|
636
|
+
for (let i = 0; i < names.length; i++) {
|
|
637
|
+
if (!claimed.has(i)) {
|
|
638
|
+
attributes.push(i);
|
|
639
|
+
}
|
|
640
|
+
}
|
|
641
|
+
return { kind: "edges", names, width: names.length, source, target, weight, type, id, label, attributes };
|
|
642
|
+
}
|
|
643
|
+
/**
|
|
644
|
+
* The columns of a node table.
|
|
645
|
+
* @param names - the column names
|
|
646
|
+
* @param header - whether the file has a header row
|
|
647
|
+
* @param line - the header line
|
|
648
|
+
* @returns the plan; the import aborts when no id column resolves
|
|
649
|
+
*/
|
|
650
|
+
nodePlan(names, header, line) {
|
|
651
|
+
const { csv, common, report } = this.state;
|
|
652
|
+
const claimed = /* @__PURE__ */ new Set();
|
|
653
|
+
let idColumn = -1;
|
|
654
|
+
if (csv.idColumn !== null) {
|
|
655
|
+
idColumn = resolveColumnRef(names, csv.idColumn, "idColumn");
|
|
656
|
+
} else if (header) {
|
|
657
|
+
idColumn = findColumn(names, ID_NAMES);
|
|
658
|
+
} else if (names.length >= 1) {
|
|
659
|
+
idColumn = 0;
|
|
660
|
+
}
|
|
661
|
+
const label = header ? findFree(names, LABEL_NAMES, new Set(idColumn >= 0 ? [idColumn] : [])) : -1;
|
|
662
|
+
let id;
|
|
663
|
+
switch (common.nodeIdFrom) {
|
|
664
|
+
case "label":
|
|
665
|
+
if (label < 0) {
|
|
666
|
+
report.fail(NO_ID_COLUMN_CODE, 'nodeIdFrom is "label" but the node table has no label column', {
|
|
667
|
+
line
|
|
668
|
+
});
|
|
669
|
+
}
|
|
670
|
+
id = label;
|
|
671
|
+
break;
|
|
672
|
+
case "index":
|
|
673
|
+
id = -1;
|
|
674
|
+
break;
|
|
675
|
+
default:
|
|
676
|
+
if (idColumn < 0) {
|
|
677
|
+
report.fail(
|
|
678
|
+
NO_ID_COLUMN_CODE,
|
|
679
|
+
`no id column in the node table header (${names.map((n) => JSON.stringify(n)).join(", ")})`,
|
|
680
|
+
{ line },
|
|
681
|
+
{ columns: [...names] }
|
|
682
|
+
);
|
|
683
|
+
}
|
|
684
|
+
id = idColumn;
|
|
685
|
+
claimed.add(idColumn);
|
|
686
|
+
break;
|
|
687
|
+
}
|
|
688
|
+
if (label >= 0) {
|
|
689
|
+
claimed.add(label);
|
|
690
|
+
}
|
|
691
|
+
const attributes = [];
|
|
692
|
+
for (let i = 0; i < names.length; i++) {
|
|
693
|
+
if (!claimed.has(i)) {
|
|
694
|
+
attributes.push(i);
|
|
695
|
+
}
|
|
696
|
+
}
|
|
697
|
+
return { kind: "nodes", names, width: names.length, id, label, attributes };
|
|
698
|
+
}
|
|
699
|
+
/**
|
|
700
|
+
* Declare the role columns and create a writer per attribute column.
|
|
701
|
+
* @param line - the header line
|
|
702
|
+
*/
|
|
703
|
+
prepareColumns(line) {
|
|
704
|
+
const plan = this.requirePlan();
|
|
705
|
+
const { sink, report } = this.state;
|
|
706
|
+
const domain = plan.kind === "edges" ? "edge" : "node";
|
|
707
|
+
const origin = { format: "csv" };
|
|
708
|
+
if (plan.kind === "edges" && plan.id >= 0) {
|
|
709
|
+
this.idHandle = declareRoleColumn(
|
|
710
|
+
sink,
|
|
711
|
+
"edge",
|
|
712
|
+
{ name: plan.names[plan.id], dtype: "string", nullable: true, role: "id", unique: true, origin },
|
|
713
|
+
plan.id + 1,
|
|
714
|
+
report,
|
|
715
|
+
line
|
|
716
|
+
);
|
|
717
|
+
}
|
|
718
|
+
if (plan.label >= 0) {
|
|
719
|
+
this.labelHandle = declareRoleColumn(
|
|
720
|
+
sink,
|
|
721
|
+
domain,
|
|
722
|
+
{ name: plan.names[plan.label], dtype: "string", nullable: true, role: "label", origin },
|
|
723
|
+
plan.label + 1,
|
|
724
|
+
report,
|
|
725
|
+
line
|
|
726
|
+
);
|
|
727
|
+
}
|
|
728
|
+
this.writers = plan.names.map(() => null);
|
|
729
|
+
for (const index of plan.attributes) {
|
|
730
|
+
this.writers[index] = new InferredColumn(plan.names[index], domain, sink, report);
|
|
731
|
+
}
|
|
732
|
+
}
|
|
733
|
+
/**
|
|
734
|
+
* The plan, which exists once the header was read.
|
|
735
|
+
* @returns the plan
|
|
736
|
+
*/
|
|
737
|
+
requirePlan() {
|
|
738
|
+
if (this.plan === null) {
|
|
739
|
+
throw new GraphFormatError("E_UNSUPPORTED", "the header has not been read", { reason: "no plan" });
|
|
740
|
+
}
|
|
741
|
+
return this.plan;
|
|
742
|
+
}
|
|
743
|
+
/**
|
|
744
|
+
* Push one data row.
|
|
745
|
+
* @param row - the cells
|
|
746
|
+
* @param quoted - whether each cell was quoted (a quoted empty cell is the empty string)
|
|
747
|
+
* @param line - the row's line
|
|
748
|
+
*/
|
|
749
|
+
processRow(row, quoted, line) {
|
|
750
|
+
const plan = this.requirePlan();
|
|
751
|
+
this.dataRows++;
|
|
752
|
+
if (plan.kind === "edges") {
|
|
753
|
+
this.processEdgeRow(plan, row, quoted, line);
|
|
754
|
+
} else {
|
|
755
|
+
this.processNodeRow(plan, row, quoted, line);
|
|
756
|
+
}
|
|
757
|
+
}
|
|
758
|
+
/**
|
|
759
|
+
* Push one edge row: endpoints, weight, direction, then the attribute cells.
|
|
760
|
+
* @param plan - the edge plan
|
|
761
|
+
* @param row - the cells
|
|
762
|
+
* @param quoted - whether each cell was quoted
|
|
763
|
+
* @param line - the row's line
|
|
764
|
+
*/
|
|
765
|
+
processEdgeRow(plan, row, quoted, line) {
|
|
766
|
+
const { report, sink, resolver, common } = this.state;
|
|
767
|
+
const { counts } = report;
|
|
768
|
+
if (row.length !== plan.width) {
|
|
769
|
+
report.error(
|
|
770
|
+
"validation-error",
|
|
771
|
+
FIELD_COUNT_CODE,
|
|
772
|
+
`line ${line}: ${row.length} field(s), the header has ${plan.width}`,
|
|
773
|
+
{ line }
|
|
774
|
+
);
|
|
775
|
+
counts.skippedEdges++;
|
|
776
|
+
return;
|
|
777
|
+
}
|
|
778
|
+
const sourceText = row[plan.source];
|
|
779
|
+
const targetText = row[plan.target];
|
|
780
|
+
const sourceMissing = isUnset(sourceText, quoted[plan.source]);
|
|
781
|
+
if (sourceMissing || isUnset(targetText, quoted[plan.target])) {
|
|
782
|
+
report.error(
|
|
783
|
+
"missing-value",
|
|
784
|
+
MISSING_ENDPOINT_CODE,
|
|
785
|
+
`line ${line}: blank ${sourceMissing ? "source" : "target"} cell`,
|
|
786
|
+
{ line }
|
|
787
|
+
);
|
|
788
|
+
counts.skippedEdges++;
|
|
789
|
+
return;
|
|
790
|
+
}
|
|
791
|
+
let kind = this.state.commentDirected ?? common.defaultDirected ? "directed" : "undirected";
|
|
792
|
+
if (plan.type >= 0) {
|
|
793
|
+
const parsed = parseKind(row[plan.type]);
|
|
794
|
+
if (parsed === null) {
|
|
795
|
+
report.error(
|
|
796
|
+
"validation-error",
|
|
797
|
+
BAD_TYPE_CODE,
|
|
798
|
+
`line ${line}: Type ${JSON.stringify(row[plan.type])} is not Directed, Undirected or Mutual`,
|
|
799
|
+
{ line }
|
|
800
|
+
);
|
|
801
|
+
counts.skippedEdges++;
|
|
802
|
+
return;
|
|
803
|
+
}
|
|
804
|
+
if (parsed !== void 0) {
|
|
805
|
+
kind = parsed;
|
|
806
|
+
}
|
|
807
|
+
}
|
|
808
|
+
const { where } = this;
|
|
809
|
+
where.line = line;
|
|
810
|
+
where.element = null;
|
|
811
|
+
const idText = plan.id >= 0 && !isUnset(row[plan.id], quoted[plan.id]) ? row[plan.id] : null;
|
|
812
|
+
if (idText !== null) {
|
|
813
|
+
if (this.edgeIds.has(idText)) {
|
|
814
|
+
report.error(
|
|
815
|
+
"validation-error",
|
|
816
|
+
DUPLICATE_EDGE_ID_CODE,
|
|
817
|
+
`line ${line}: edge id ${JSON.stringify(idText)} repeats an earlier row's; the row is skipped`,
|
|
818
|
+
{ line, element: idText }
|
|
819
|
+
);
|
|
820
|
+
counts.skippedEdges++;
|
|
821
|
+
return;
|
|
822
|
+
}
|
|
823
|
+
this.edgeIds.add(idText);
|
|
824
|
+
}
|
|
825
|
+
let edge;
|
|
826
|
+
try {
|
|
827
|
+
const source = this.coerce(sourceText);
|
|
828
|
+
const target = this.coerce(targetText);
|
|
829
|
+
const weight = plan.weight >= 0 ? parseWeightText(row[plan.weight]) : void 0;
|
|
830
|
+
if (!this.state.headerSet) {
|
|
831
|
+
this.state.headerSet = true;
|
|
832
|
+
resolver.setHeader(kind !== "undirected", where);
|
|
833
|
+
}
|
|
834
|
+
const sourceNew = sink.indexOf(source) === INVALID_INDEX;
|
|
835
|
+
const targetNew = source !== target && sink.indexOf(target) === INVALID_INDEX;
|
|
836
|
+
const before = sink.edgeCount;
|
|
837
|
+
edge = resolver.addEdge(source, target, kind, weight, where);
|
|
838
|
+
counts.edges += sink.edgeCount - before;
|
|
839
|
+
counts.nodes += (sourceNew ? 1 : 0) + (targetNew ? 1 : 0);
|
|
840
|
+
} catch (err) {
|
|
841
|
+
where.element = idText ?? `${sourceText}->${targetText}`;
|
|
842
|
+
report.recordError(err, where);
|
|
843
|
+
counts.skippedEdges++;
|
|
844
|
+
return;
|
|
845
|
+
}
|
|
846
|
+
if (idText !== null) {
|
|
847
|
+
this.writeRole(this.idHandle, "edge", edge, idText, plan.names[plan.id], line);
|
|
848
|
+
}
|
|
849
|
+
if (plan.label >= 0 && !isUnset(row[plan.label], quoted[plan.label])) {
|
|
850
|
+
this.writeRole(this.labelHandle, "edge", edge, row[plan.label], plan.names[plan.label], line);
|
|
851
|
+
}
|
|
852
|
+
this.writeAttributes(plan, row, quoted, edge, line);
|
|
853
|
+
}
|
|
854
|
+
/**
|
|
855
|
+
* Push one node row: the id, then the label and attribute cells.
|
|
856
|
+
* @param plan - the node plan
|
|
857
|
+
* @param row - the cells
|
|
858
|
+
* @param quoted - whether each cell was quoted
|
|
859
|
+
* @param line - the row's line
|
|
860
|
+
*/
|
|
861
|
+
processNodeRow(plan, row, quoted, line) {
|
|
862
|
+
const { report, sink } = this.state;
|
|
863
|
+
const { counts } = report;
|
|
864
|
+
const ordinal = this.nodeOrdinal++;
|
|
865
|
+
if (row.length !== plan.width) {
|
|
866
|
+
report.error(
|
|
867
|
+
"validation-error",
|
|
868
|
+
FIELD_COUNT_CODE,
|
|
869
|
+
`line ${line}: ${row.length} field(s), the header has ${plan.width}`,
|
|
870
|
+
{ line }
|
|
871
|
+
);
|
|
872
|
+
counts.skippedNodes++;
|
|
873
|
+
return;
|
|
874
|
+
}
|
|
875
|
+
const idText = plan.id >= 0 ? row[plan.id] : String(ordinal);
|
|
876
|
+
if (plan.id >= 0 && isUnset(idText, quoted[plan.id])) {
|
|
877
|
+
report.error("missing-value", MISSING_ID_CODE, `line ${line}: blank id cell`, { line });
|
|
878
|
+
counts.skippedNodes++;
|
|
879
|
+
return;
|
|
880
|
+
}
|
|
881
|
+
const { where } = this;
|
|
882
|
+
where.line = line;
|
|
883
|
+
where.element = idText;
|
|
884
|
+
let index;
|
|
885
|
+
try {
|
|
886
|
+
const id = plan.id >= 0 ? this.coerce(idText) : ordinal;
|
|
887
|
+
if (sink.indexOf(id) !== INVALID_INDEX) {
|
|
888
|
+
report.warning(
|
|
889
|
+
"merged",
|
|
890
|
+
DUPLICATE_NODE_CODE,
|
|
891
|
+
`line ${line}: node ${JSON.stringify(id)} already exists; its attributes are overwritten`,
|
|
892
|
+
where
|
|
893
|
+
);
|
|
894
|
+
} else {
|
|
895
|
+
counts.nodes++;
|
|
896
|
+
}
|
|
897
|
+
index = sink.addNode(id);
|
|
898
|
+
} catch (err) {
|
|
899
|
+
report.recordError(err, where);
|
|
900
|
+
counts.skippedNodes++;
|
|
901
|
+
return;
|
|
902
|
+
}
|
|
903
|
+
if (plan.label >= 0 && !isUnset(row[plan.label], quoted[plan.label])) {
|
|
904
|
+
this.writeRole(this.labelHandle, "node", index, row[plan.label], plan.names[plan.label], line);
|
|
905
|
+
}
|
|
906
|
+
this.writeAttributes(plan, row, quoted, index, line);
|
|
907
|
+
}
|
|
908
|
+
/**
|
|
909
|
+
* Coerce an id cell, reporting a merge under `ids: "number"`.
|
|
910
|
+
* @param text - the cell text
|
|
911
|
+
* @returns the id
|
|
912
|
+
*/
|
|
913
|
+
coerce(text) {
|
|
914
|
+
const id = this.state.coercer.text(text);
|
|
915
|
+
const merge = this.state.coercer.lastMerge;
|
|
916
|
+
if (merge !== null) {
|
|
917
|
+
this.state.report.warnOnce(
|
|
918
|
+
"coercion",
|
|
919
|
+
ID_MERGED_CODE,
|
|
920
|
+
`id ${JSON.stringify(merge.text)} merged with ${JSON.stringify(merge.previousText)} as ${merge.id} under ids: "number"`,
|
|
921
|
+
this.where
|
|
922
|
+
);
|
|
923
|
+
}
|
|
924
|
+
return id;
|
|
925
|
+
}
|
|
926
|
+
/**
|
|
927
|
+
* Write a role column cell (edge id, label).
|
|
928
|
+
* @param handle - the column handle
|
|
929
|
+
* @param domain - node or edge
|
|
930
|
+
* @param row - the node or edge index
|
|
931
|
+
* @param text - the cell text
|
|
932
|
+
* @param name - the column name, for issues
|
|
933
|
+
* @param line - the row's line
|
|
934
|
+
*/
|
|
935
|
+
writeRole(handle, domain, row, text, name, line) {
|
|
936
|
+
try {
|
|
937
|
+
if (domain === "node") {
|
|
938
|
+
this.state.sink.setNodeValue(handle, row, text);
|
|
939
|
+
} else {
|
|
940
|
+
this.state.sink.setEdgeValue(handle, row, text);
|
|
941
|
+
}
|
|
942
|
+
} catch (err) {
|
|
943
|
+
this.state.report.recordError(err, { line, element: name });
|
|
944
|
+
}
|
|
945
|
+
}
|
|
946
|
+
/**
|
|
947
|
+
* Write the attribute cells of a row: an unquoted blank cell is unset, a quoted one (`""`,
|
|
948
|
+
* `" "`) is the text it holds.
|
|
949
|
+
* @param plan - the plan
|
|
950
|
+
* @param row - the cells
|
|
951
|
+
* @param quoted - whether each cell was quoted
|
|
952
|
+
* @param index - the node or edge index
|
|
953
|
+
* @param line - the row's line
|
|
954
|
+
*/
|
|
955
|
+
writeAttributes(plan, row, quoted, index, line) {
|
|
956
|
+
for (const k of plan.attributes) {
|
|
957
|
+
const text = row[k];
|
|
958
|
+
if (isUnset(text, quoted[k])) {
|
|
959
|
+
continue;
|
|
960
|
+
}
|
|
961
|
+
const writer = this.writers[k];
|
|
962
|
+
if (writer === null) {
|
|
963
|
+
continue;
|
|
964
|
+
}
|
|
965
|
+
try {
|
|
966
|
+
writer.write(index, text);
|
|
967
|
+
} catch (err) {
|
|
968
|
+
this.state.report.recordError(err, { line, element: writer.name });
|
|
969
|
+
}
|
|
970
|
+
}
|
|
971
|
+
}
|
|
972
|
+
}
|
|
973
|
+
const HEAD_BYTES = 4096;
|
|
974
|
+
const OTHER_FORMAT = /^\s*(<|[[{]|(strict\s+)?(di)?graph(\s+\S+)?\s*\{|\*vertices|creator\b|graph\s*\[)/i;
|
|
975
|
+
function sniff(head) {
|
|
976
|
+
const text = new TextDecoder("utf-8").decode(head.subarray(0, HEAD_BYTES));
|
|
977
|
+
const body = text.startsWith(String.fromCharCode(65279)) ? text.slice(1) : text;
|
|
978
|
+
if (body.trim().length === 0 || OTHER_FORMAT.test(body)) {
|
|
979
|
+
return 0;
|
|
980
|
+
}
|
|
981
|
+
const newline = sniffNewline(body);
|
|
982
|
+
const delimiter = sniffDelimiter(body, newline);
|
|
983
|
+
if (delimiter === null) {
|
|
984
|
+
return 0;
|
|
985
|
+
}
|
|
986
|
+
const end = body.indexOf(newline);
|
|
987
|
+
const firstLine = end < 0 ? body : body.slice(0, end);
|
|
988
|
+
const names = headerNames(firstLine.replace(/\r$/, "").split(delimiter));
|
|
989
|
+
if (findColumn(names, SOURCE_NAMES) >= 0 && findColumn(names, TARGET_NAMES) >= 0) {
|
|
990
|
+
return 0.9;
|
|
991
|
+
}
|
|
992
|
+
if (findColumn(names, ID_NAMES) >= 0) {
|
|
993
|
+
return 0.6;
|
|
994
|
+
}
|
|
995
|
+
return 0.3;
|
|
996
|
+
}
|
|
997
|
+
async function importCsv(input, sink, options) {
|
|
998
|
+
const common = resolveImportOptions(options, { ids: "canonical", defaultDirected: true, weightFrom: "weight" });
|
|
999
|
+
const csv = resolveCsvOptions(options);
|
|
1000
|
+
const report = new ImportReportBuilder("csv", common.errorLimit);
|
|
1001
|
+
reportSinkOptions(sink, options, report);
|
|
1002
|
+
reportUnusedOptions(
|
|
1003
|
+
options,
|
|
1004
|
+
report,
|
|
1005
|
+
csv.table === "nodes" || csv.nodes !== null ? USED_OPTIONS_WITH_NODES : USED_OPTIONS
|
|
1006
|
+
);
|
|
1007
|
+
const state = {
|
|
1008
|
+
sink,
|
|
1009
|
+
report,
|
|
1010
|
+
common,
|
|
1011
|
+
csv,
|
|
1012
|
+
weightFromExplicit: typeof options?.weightFrom === "string",
|
|
1013
|
+
coercer: new IdCoercer(common.ids),
|
|
1014
|
+
resolver: new DirectionResolver(sink, report, common.onMixedDirection),
|
|
1015
|
+
headerSet: false,
|
|
1016
|
+
commentDirected: null
|
|
1017
|
+
};
|
|
1018
|
+
if (csv.nodes !== null) {
|
|
1019
|
+
await new TableReader(state, csv.nodes, "nodes", false).read();
|
|
1020
|
+
}
|
|
1021
|
+
await new TableReader(state, input, csv.table, true).read();
|
|
1022
|
+
if (state.coercer.mergeCount > 1) {
|
|
1023
|
+
report.warning(
|
|
1024
|
+
"coercion",
|
|
1025
|
+
ID_MERGED_CODE,
|
|
1026
|
+
`${state.coercer.mergeCount} id cell(s) merged into ids other cells already produced under ids: "number"`
|
|
1027
|
+
);
|
|
1028
|
+
}
|
|
1029
|
+
throwIfAborted(common.signal);
|
|
1030
|
+
return report.finish();
|
|
1031
|
+
}
|
|
1032
|
+
const csvImporter = Object.freeze({
|
|
1033
|
+
format: "csv",
|
|
1034
|
+
extensions: Object.freeze([".csv", ".tsv", ".edges", ".edgelist"]),
|
|
1035
|
+
mimeTypes: Object.freeze(["text/csv", "text/tab-separated-values", "text/plain"]),
|
|
1036
|
+
sniff,
|
|
1037
|
+
import: importCsv
|
|
1038
|
+
});
|
|
1039
|
+
const CSV_LOSS = Object.freeze({
|
|
1040
|
+
/** Two node ids share one text (a number and a string); export() throws E_INVALID_ID. */
|
|
1041
|
+
ID_TEXT_COLLISION: LOSS.ID_TEXT_COLLISION,
|
|
1042
|
+
/** Ids whose text reads back as the other type under the canonical rule. */
|
|
1043
|
+
ID_TEXT_TYPE: LOSS.ID_TEXT_TYPE,
|
|
1044
|
+
/** The generic dialect has no direction column; an undirected or mixed graph reads back as directed. */
|
|
1045
|
+
DIRECTION_DROPPED: "W_CSV_DIRECTION_DROPPED",
|
|
1046
|
+
/** Mutual pairs are written as two directed rows. */
|
|
1047
|
+
MUTUAL_EXPANDED: LOSS.MUTUAL_EXPANDED,
|
|
1048
|
+
/** An attribute column named like a reserved header is not written. */
|
|
1049
|
+
RESERVED_NAME: "W_CSV_RESERVED_NAME",
|
|
1050
|
+
/** A column without a role that the importer gives one back by its name. */
|
|
1051
|
+
ROLE_ASSUMED: LOSS.ROLE_ASSUMED,
|
|
1052
|
+
/** A role column (id, label) whose name the importer does not recognise; the role is lost. */
|
|
1053
|
+
ROLE_NAME: "W_CSV_ROLE_NAME",
|
|
1054
|
+
/** A role column (id, label) that is not string / dict reads back as string. */
|
|
1055
|
+
TEXT_ROLE: "W_CSV_TEXT_ROLE",
|
|
1056
|
+
/** NaN / Infinity in a numeric column read back as text. */
|
|
1057
|
+
NONFINITE: "W_CSV_NONFINITE",
|
|
1058
|
+
/** A text column whose every value reads back as a number or boolean. */
|
|
1059
|
+
TEXT_INFERRED: LOSS.TEXT_INFERRED,
|
|
1060
|
+
/** A dict column whose cardinality makes the importer read it back as string, or the reverse. */
|
|
1061
|
+
STORAGE_CLASS_CHANGED: LOSS.STORAGE_CLASS,
|
|
1062
|
+
/** Node attributes are written by a `table: "nodes"` export only. */
|
|
1063
|
+
NODE_TABLE: "W_CSV_NODE_TABLE",
|
|
1064
|
+
/** The edge table carries no node without an edge: isolated nodes vanish on re-import. */
|
|
1065
|
+
ISOLATED_NODES: "W_CSV_ISOLATED_NODES",
|
|
1066
|
+
/** The edge table lists nodes by first appearance; the node order (and indices) change on re-import. */
|
|
1067
|
+
NODE_ORDER: "W_CSV_NODE_ORDER"
|
|
1068
|
+
});
|
|
1069
|
+
const CSV_CAPABILITIES = capabilities({
|
|
1070
|
+
mixedDirection: true,
|
|
1071
|
+
multiEdges: true,
|
|
1072
|
+
selfLoops: true,
|
|
1073
|
+
edgeIds: "optional",
|
|
1074
|
+
idCharset: "any",
|
|
1075
|
+
dtypes: ["bool", "i32", "f64", "string", "dict"]
|
|
1076
|
+
});
|
|
1077
|
+
const DIALECTS = {
|
|
1078
|
+
gephi: { source: "Source", target: "Target", type: "Type", id: "Id", label: "Label", weight: "Weight" },
|
|
1079
|
+
generic: { source: "source", target: "target", type: null, id: "id", label: "label", weight: "weight" }
|
|
1080
|
+
};
|
|
1081
|
+
const KEPT_ROLES = /* @__PURE__ */ new Set(["id", "label"]);
|
|
1082
|
+
const SKIPPED_ROLES = /* @__PURE__ */ new Set([
|
|
1083
|
+
"directed",
|
|
1084
|
+
"pair",
|
|
1085
|
+
"mutual",
|
|
1086
|
+
"weight",
|
|
1087
|
+
"timeText",
|
|
1088
|
+
"position",
|
|
1089
|
+
"color",
|
|
1090
|
+
"size",
|
|
1091
|
+
"shape",
|
|
1092
|
+
"thickness",
|
|
1093
|
+
"parent",
|
|
1094
|
+
"parents",
|
|
1095
|
+
"start",
|
|
1096
|
+
"end",
|
|
1097
|
+
"timestamp",
|
|
1098
|
+
"timestamps",
|
|
1099
|
+
"spells",
|
|
1100
|
+
"open"
|
|
1101
|
+
]);
|
|
1102
|
+
function resolveCsvExportOptions(options) {
|
|
1103
|
+
const o = options ?? {};
|
|
1104
|
+
if (o.dialect !== void 0 && o.dialect !== "gephi" && o.dialect !== "generic") {
|
|
1105
|
+
throw new GraphFormatError("E_UNSUPPORTED", 'option dialect: expected "gephi" or "generic"', {
|
|
1106
|
+
option: "dialect",
|
|
1107
|
+
found: o.dialect
|
|
1108
|
+
});
|
|
1109
|
+
}
|
|
1110
|
+
if (o.table !== void 0 && o.table !== "edges" && o.table !== "nodes") {
|
|
1111
|
+
throw new GraphFormatError("E_UNSUPPORTED", 'option table: expected "edges" or "nodes"', {
|
|
1112
|
+
option: "table",
|
|
1113
|
+
found: o.table
|
|
1114
|
+
});
|
|
1115
|
+
}
|
|
1116
|
+
if (o.delimiter !== void 0 && (typeof o.delimiter !== "string" || o.delimiter.length !== 1 || o.delimiter === '"' || o.delimiter === "\n" || o.delimiter === "\r")) {
|
|
1117
|
+
throw new GraphFormatError(
|
|
1118
|
+
"E_UNSUPPORTED",
|
|
1119
|
+
"option delimiter: expected one character other than a quote or a line break",
|
|
1120
|
+
{ option: "delimiter", found: o.delimiter }
|
|
1121
|
+
);
|
|
1122
|
+
}
|
|
1123
|
+
if (o.newline !== void 0 && o.newline !== "\n" && o.newline !== "\r\n") {
|
|
1124
|
+
throw new GraphFormatError("E_UNSUPPORTED", "option newline: expected LF or CRLF", {
|
|
1125
|
+
option: "newline",
|
|
1126
|
+
found: o.newline
|
|
1127
|
+
});
|
|
1128
|
+
}
|
|
1129
|
+
if (o.header !== void 0 && typeof o.header !== "boolean") {
|
|
1130
|
+
throw new GraphFormatError("E_UNSUPPORTED", "option header: expected a boolean", {
|
|
1131
|
+
option: "header",
|
|
1132
|
+
found: o.header
|
|
1133
|
+
});
|
|
1134
|
+
}
|
|
1135
|
+
return {
|
|
1136
|
+
dialect: DIALECTS[o.dialect ?? "gephi"],
|
|
1137
|
+
table: o.table ?? "edges",
|
|
1138
|
+
delimiter: o.delimiter ?? ",",
|
|
1139
|
+
newline: o.newline ?? "\n",
|
|
1140
|
+
header: o.header ?? true
|
|
1141
|
+
};
|
|
1142
|
+
}
|
|
1143
|
+
function scalarText(dtype, value) {
|
|
1144
|
+
switch (dtype) {
|
|
1145
|
+
case "bool":
|
|
1146
|
+
return value === true ? "true" : "false";
|
|
1147
|
+
case "f32":
|
|
1148
|
+
case "f64":
|
|
1149
|
+
return typeof value === "number" ? formatDecimal(value, dtype) : String(value);
|
|
1150
|
+
case "i32":
|
|
1151
|
+
case "u32":
|
|
1152
|
+
case "u8":
|
|
1153
|
+
return typeof value === "number" ? formatInteger(value) : String(value);
|
|
1154
|
+
case "string":
|
|
1155
|
+
case "dict":
|
|
1156
|
+
return typeof value === "string" ? value : String(value);
|
|
1157
|
+
case "json":
|
|
1158
|
+
return JSON.stringify(value) ?? "";
|
|
1159
|
+
default:
|
|
1160
|
+
return String(value);
|
|
1161
|
+
}
|
|
1162
|
+
}
|
|
1163
|
+
function cellText(column, row) {
|
|
1164
|
+
if (!column.isSet(row)) {
|
|
1165
|
+
return null;
|
|
1166
|
+
}
|
|
1167
|
+
switch (column.dtype) {
|
|
1168
|
+
case "list": {
|
|
1169
|
+
const items = column.sliceOf(row);
|
|
1170
|
+
const { dtype } = column.child;
|
|
1171
|
+
return joinListText(
|
|
1172
|
+
items.map((item) => scalarText(dtype, item)),
|
|
1173
|
+
"semicolon"
|
|
1174
|
+
);
|
|
1175
|
+
}
|
|
1176
|
+
case "json":
|
|
1177
|
+
return JSON.stringify(column.values[row]) ?? "";
|
|
1178
|
+
case "bool":
|
|
1179
|
+
case "string":
|
|
1180
|
+
case "dict":
|
|
1181
|
+
return scalarText(column.dtype, column.value(row));
|
|
1182
|
+
default: {
|
|
1183
|
+
const { components } = column.meta;
|
|
1184
|
+
const { data } = column;
|
|
1185
|
+
if (components === 1) {
|
|
1186
|
+
return scalarText(column.dtype, data[row]);
|
|
1187
|
+
}
|
|
1188
|
+
const parts = [];
|
|
1189
|
+
for (let k = 0; k < components; k++) {
|
|
1190
|
+
parts.push(scalarText(column.dtype, data[row * components + k]));
|
|
1191
|
+
}
|
|
1192
|
+
return parts.join(";");
|
|
1193
|
+
}
|
|
1194
|
+
}
|
|
1195
|
+
}
|
|
1196
|
+
function idNotes(snapshot, values, note) {
|
|
1197
|
+
const { ids } = snapshot;
|
|
1198
|
+
if (ids.kind === "mixed") {
|
|
1199
|
+
const seen = /* @__PURE__ */ new Set();
|
|
1200
|
+
let collisions = 0;
|
|
1201
|
+
for (let i = 0; i < ids.size; i++) {
|
|
1202
|
+
const text = String(ids.idOf(i));
|
|
1203
|
+
if (seen.has(text)) {
|
|
1204
|
+
collisions++;
|
|
1205
|
+
} else {
|
|
1206
|
+
seen.add(text);
|
|
1207
|
+
}
|
|
1208
|
+
}
|
|
1209
|
+
if (collisions > 0) {
|
|
1210
|
+
note(
|
|
1211
|
+
CSV_LOSS.ID_TEXT_COLLISION,
|
|
1212
|
+
`${collisions} node id(s) share their text with another id (a number and a string); export() will throw E_INVALID_ID`,
|
|
1213
|
+
null,
|
|
1214
|
+
collisions
|
|
1215
|
+
);
|
|
1216
|
+
}
|
|
1217
|
+
}
|
|
1218
|
+
if (values && ids.kind !== "identity" && ids.kind !== "dense") {
|
|
1219
|
+
let changed = 0;
|
|
1220
|
+
for (let i = 0; i < ids.size; i++) {
|
|
1221
|
+
const id = ids.idOf(i);
|
|
1222
|
+
if (typeof canonicalId(String(id)) !== typeof id) {
|
|
1223
|
+
changed++;
|
|
1224
|
+
}
|
|
1225
|
+
}
|
|
1226
|
+
if (changed > 0) {
|
|
1227
|
+
note(
|
|
1228
|
+
CSV_LOSS.ID_TEXT_TYPE,
|
|
1229
|
+
`${changed} node id(s) read back as the other type under ids: "canonical" (a string "1" becomes 1, a number 1.5 becomes "1.5")`,
|
|
1230
|
+
null,
|
|
1231
|
+
changed
|
|
1232
|
+
);
|
|
1233
|
+
}
|
|
1234
|
+
}
|
|
1235
|
+
}
|
|
1236
|
+
function planExport(snapshot, options, values) {
|
|
1237
|
+
const common = resolveExportOptions(options);
|
|
1238
|
+
const csv = resolveCsvExportOptions(options);
|
|
1239
|
+
const notes = [];
|
|
1240
|
+
const note = (code, message, column = null, count = null) => {
|
|
1241
|
+
notes.push(Object.freeze({ code, message, column, count }));
|
|
1242
|
+
};
|
|
1243
|
+
const { ids } = snapshot;
|
|
1244
|
+
const idText = (index) => String(ids.idOf(index));
|
|
1245
|
+
idNotes(snapshot, values, note);
|
|
1246
|
+
const folding = pairFolding(snapshot);
|
|
1247
|
+
const edgeRows = [];
|
|
1248
|
+
for (let e = 0; e < snapshot.edgeCount; e++) {
|
|
1249
|
+
if (!folding.folded(e)) {
|
|
1250
|
+
edgeRows.push(e);
|
|
1251
|
+
}
|
|
1252
|
+
}
|
|
1253
|
+
if (csv.dialect.type === null) {
|
|
1254
|
+
const mixed = countMixedEdges(snapshot);
|
|
1255
|
+
if (!snapshot.directed) {
|
|
1256
|
+
note(
|
|
1257
|
+
CSV_LOSS.DIRECTION_DROPPED,
|
|
1258
|
+
`the generic dialect has no direction column; ${snapshot.edgeCount} undirected edge(s) read back as directed unless the importer is told otherwise`,
|
|
1259
|
+
null,
|
|
1260
|
+
snapshot.edgeCount
|
|
1261
|
+
);
|
|
1262
|
+
} else if (mixed > 0) {
|
|
1263
|
+
note(
|
|
1264
|
+
CSV_LOSS.DIRECTION_DROPPED,
|
|
1265
|
+
`the generic dialect has no direction column; ${mixed} undirected edge(s) of a mixed graph read back as one directed edge each`,
|
|
1266
|
+
null,
|
|
1267
|
+
mixed
|
|
1268
|
+
);
|
|
1269
|
+
}
|
|
1270
|
+
}
|
|
1271
|
+
if (folding.mutualCount > 0) {
|
|
1272
|
+
note(
|
|
1273
|
+
CSV_LOSS.MUTUAL_EXPANDED,
|
|
1274
|
+
`${folding.mutualCount} mutual pair(s) are written as two directed rows; the mutual mark is lost`,
|
|
1275
|
+
null,
|
|
1276
|
+
folding.mutualCount
|
|
1277
|
+
);
|
|
1278
|
+
}
|
|
1279
|
+
if (csv.table === "edges") {
|
|
1280
|
+
noteNodeCoverage(snapshot, edgeRows, note);
|
|
1281
|
+
}
|
|
1282
|
+
const weights = explicitWeights(snapshot);
|
|
1283
|
+
const edgeId = snapshot.edges.byRole("id");
|
|
1284
|
+
const edgeLabel = snapshot.edges.byRole("label");
|
|
1285
|
+
const edgeReserved = new Set([csv.dialect.source, csv.dialect.target, csv.dialect.weight].map(lower));
|
|
1286
|
+
if (csv.dialect.type !== null) {
|
|
1287
|
+
edgeReserved.add(lower(csv.dialect.type));
|
|
1288
|
+
}
|
|
1289
|
+
const edgeColumns = attributeColumns(
|
|
1290
|
+
snapshot.edges,
|
|
1291
|
+
"edge",
|
|
1292
|
+
edgeReserved,
|
|
1293
|
+
edgeId,
|
|
1294
|
+
edgeLabel,
|
|
1295
|
+
csv.table === "edges" ? note : null
|
|
1296
|
+
);
|
|
1297
|
+
if (csv.table === "edges" && values) {
|
|
1298
|
+
checkRoleColumn(edgeId, "id", EDGE_ID_NAMES, "edge", edgeRows, note);
|
|
1299
|
+
checkRoleColumn(edgeLabel, "label", LABEL_NAMES, "edge", edgeRows, note);
|
|
1300
|
+
for (const { column } of edgeColumns) {
|
|
1301
|
+
checkValues(column, "edge", edgeRows, note);
|
|
1302
|
+
}
|
|
1303
|
+
}
|
|
1304
|
+
const nodeLabel = snapshot.nodes.byRole("label");
|
|
1305
|
+
const nodeReserved = /* @__PURE__ */ new Set([lower(csv.dialect.id)]);
|
|
1306
|
+
const nodeRows = [];
|
|
1307
|
+
if (csv.table === "nodes" && values) {
|
|
1308
|
+
for (let i = 0; i < snapshot.nodeCount; i++) {
|
|
1309
|
+
nodeRows.push(i);
|
|
1310
|
+
}
|
|
1311
|
+
}
|
|
1312
|
+
const nodeColumns = attributeColumns(
|
|
1313
|
+
snapshot.nodes,
|
|
1314
|
+
"node",
|
|
1315
|
+
nodeReserved,
|
|
1316
|
+
null,
|
|
1317
|
+
nodeLabel,
|
|
1318
|
+
csv.table === "nodes" ? note : null
|
|
1319
|
+
);
|
|
1320
|
+
if (csv.table === "nodes") {
|
|
1321
|
+
if (values) {
|
|
1322
|
+
checkRoleColumn(nodeLabel, "label", LABEL_NAMES, "node", nodeRows, note);
|
|
1323
|
+
for (const { column } of nodeColumns) {
|
|
1324
|
+
checkValues(column, "node", nodeRows, note);
|
|
1325
|
+
}
|
|
1326
|
+
}
|
|
1327
|
+
} else {
|
|
1328
|
+
const written = nodeColumns.length + (nodeLabel === null ? 0 : 1);
|
|
1329
|
+
if (written > 0) {
|
|
1330
|
+
note(
|
|
1331
|
+
CSV_LOSS.NODE_TABLE,
|
|
1332
|
+
`${written} node column(s) are written by a table: "nodes" export only`,
|
|
1333
|
+
null,
|
|
1334
|
+
written
|
|
1335
|
+
);
|
|
1336
|
+
}
|
|
1337
|
+
}
|
|
1338
|
+
return {
|
|
1339
|
+
csv,
|
|
1340
|
+
common,
|
|
1341
|
+
notes,
|
|
1342
|
+
edgeRows,
|
|
1343
|
+
folding,
|
|
1344
|
+
edgeId,
|
|
1345
|
+
edgeLabel,
|
|
1346
|
+
weights,
|
|
1347
|
+
edgeColumns,
|
|
1348
|
+
nodeLabel,
|
|
1349
|
+
nodeColumns,
|
|
1350
|
+
idText
|
|
1351
|
+
};
|
|
1352
|
+
}
|
|
1353
|
+
function noteNodeCoverage(snapshot, edgeRows, note) {
|
|
1354
|
+
const { nodeCount } = snapshot;
|
|
1355
|
+
if (nodeCount === 0) {
|
|
1356
|
+
return;
|
|
1357
|
+
}
|
|
1358
|
+
const list = snapshot.edgeList();
|
|
1359
|
+
const seen = new Uint8Array(nodeCount);
|
|
1360
|
+
let next = 0;
|
|
1361
|
+
let reordered = 0;
|
|
1362
|
+
const mention = (index) => {
|
|
1363
|
+
if (seen[index] === 1) {
|
|
1364
|
+
return;
|
|
1365
|
+
}
|
|
1366
|
+
seen[index] = 1;
|
|
1367
|
+
if (index !== next) {
|
|
1368
|
+
reordered++;
|
|
1369
|
+
}
|
|
1370
|
+
next++;
|
|
1371
|
+
};
|
|
1372
|
+
for (const e of edgeRows) {
|
|
1373
|
+
mention(list.src[e]);
|
|
1374
|
+
mention(list.dst[e]);
|
|
1375
|
+
}
|
|
1376
|
+
const isolated = nodeCount - next;
|
|
1377
|
+
if (isolated > 0) {
|
|
1378
|
+
note(
|
|
1379
|
+
CSV_LOSS.ISOLATED_NODES,
|
|
1380
|
+
`${isolated} node(s) have no edge and cannot be written by the edge table; write the node table (table: "nodes") to keep them`,
|
|
1381
|
+
null,
|
|
1382
|
+
isolated
|
|
1383
|
+
);
|
|
1384
|
+
}
|
|
1385
|
+
if (reordered > 0) {
|
|
1386
|
+
note(
|
|
1387
|
+
CSV_LOSS.NODE_ORDER,
|
|
1388
|
+
`${reordered} node(s) are first mentioned by an edge row out of index order; a re-import numbers nodes by first appearance`,
|
|
1389
|
+
null,
|
|
1390
|
+
reordered
|
|
1391
|
+
);
|
|
1392
|
+
}
|
|
1393
|
+
}
|
|
1394
|
+
function lower(name) {
|
|
1395
|
+
return name.toLowerCase();
|
|
1396
|
+
}
|
|
1397
|
+
function attributeColumns(table, domain, reserved, idColumn, labelColumn, note) {
|
|
1398
|
+
const out = [];
|
|
1399
|
+
for (const column of table) {
|
|
1400
|
+
const { name, role } = column.meta;
|
|
1401
|
+
if (column === idColumn || column === labelColumn) {
|
|
1402
|
+
continue;
|
|
1403
|
+
}
|
|
1404
|
+
if (role !== null && SKIPPED_ROLES.has(role)) {
|
|
1405
|
+
continue;
|
|
1406
|
+
}
|
|
1407
|
+
const low = lower(name);
|
|
1408
|
+
const set = column.length - column.nullCount;
|
|
1409
|
+
const isIdName = domain === "edge" ? findColumn([name], EDGE_ID_NAMES) >= 0 : false;
|
|
1410
|
+
if (reserved.has(low) || isIdName || idColumn !== null && lower(idColumn.meta.name) === low) {
|
|
1411
|
+
note?.(
|
|
1412
|
+
CSV_LOSS.RESERVED_NAME,
|
|
1413
|
+
`${domain} column "${name}" is not written: the name is reserved for the ${isIdName ? "edge id" : low} column`,
|
|
1414
|
+
name,
|
|
1415
|
+
set
|
|
1416
|
+
);
|
|
1417
|
+
continue;
|
|
1418
|
+
}
|
|
1419
|
+
if (findColumn([name], LABEL_NAMES) >= 0) {
|
|
1420
|
+
if (labelColumn !== null) {
|
|
1421
|
+
note?.(
|
|
1422
|
+
CSV_LOSS.RESERVED_NAME,
|
|
1423
|
+
`${domain} column "${name}" is not written: the name is reserved for the label column`,
|
|
1424
|
+
name,
|
|
1425
|
+
set
|
|
1426
|
+
);
|
|
1427
|
+
continue;
|
|
1428
|
+
}
|
|
1429
|
+
note?.(
|
|
1430
|
+
CSV_LOSS.ROLE_ASSUMED,
|
|
1431
|
+
`${domain} column "${name}" reads back with the label role (string)`,
|
|
1432
|
+
name,
|
|
1433
|
+
set
|
|
1434
|
+
);
|
|
1435
|
+
}
|
|
1436
|
+
out.push({ column, header: name });
|
|
1437
|
+
}
|
|
1438
|
+
return out;
|
|
1439
|
+
}
|
|
1440
|
+
function checkRoleColumn(column, role, names, domain, rows, note) {
|
|
1441
|
+
if (column === null) {
|
|
1442
|
+
return;
|
|
1443
|
+
}
|
|
1444
|
+
const { name } = column.meta;
|
|
1445
|
+
if (findColumn([name], names) < 0) {
|
|
1446
|
+
note(
|
|
1447
|
+
CSV_LOSS.ROLE_NAME,
|
|
1448
|
+
`${domain} column "${name}" (${role}) is written under its name, which the importer does not map to the ${role} role`,
|
|
1449
|
+
name,
|
|
1450
|
+
countSet(column, rows)
|
|
1451
|
+
);
|
|
1452
|
+
}
|
|
1453
|
+
if (column.dtype !== "string" && column.dtype !== "dict") {
|
|
1454
|
+
note(
|
|
1455
|
+
CSV_LOSS.TEXT_ROLE,
|
|
1456
|
+
`${domain} column "${name}" (${role}) is ${column.dtype}; it reads back as string`,
|
|
1457
|
+
name,
|
|
1458
|
+
countSet(column, rows)
|
|
1459
|
+
);
|
|
1460
|
+
}
|
|
1461
|
+
checkValues(column, domain, rows, note, true);
|
|
1462
|
+
}
|
|
1463
|
+
function countSet(column, rows) {
|
|
1464
|
+
let count = 0;
|
|
1465
|
+
for (const row of rows) {
|
|
1466
|
+
if (column.isSet(row)) {
|
|
1467
|
+
count++;
|
|
1468
|
+
}
|
|
1469
|
+
}
|
|
1470
|
+
return count;
|
|
1471
|
+
}
|
|
1472
|
+
const WIDENING = { bool: 0, i32: 1, f64: 2, string: 3 };
|
|
1473
|
+
const TEXT_DTYPES = ["bool", "i32", "f64", "string"];
|
|
1474
|
+
function checkValues(column, domain, rows, note, roleColumn = false) {
|
|
1475
|
+
const { name, dtype } = column.meta;
|
|
1476
|
+
const label = `${domain} column "${name}"`;
|
|
1477
|
+
if (column.dtype === "f32" || column.dtype === "f64") {
|
|
1478
|
+
let nonFinite = 0;
|
|
1479
|
+
const { components } = column.meta;
|
|
1480
|
+
const { data } = column;
|
|
1481
|
+
for (const row of rows) {
|
|
1482
|
+
if (!column.isSet(row)) {
|
|
1483
|
+
continue;
|
|
1484
|
+
}
|
|
1485
|
+
for (let k = 0; k < components; k++) {
|
|
1486
|
+
if (!Number.isFinite(data[row * components + k])) {
|
|
1487
|
+
nonFinite++;
|
|
1488
|
+
break;
|
|
1489
|
+
}
|
|
1490
|
+
}
|
|
1491
|
+
}
|
|
1492
|
+
if (nonFinite > 0) {
|
|
1493
|
+
note(CSV_LOSS.NONFINITE, `${label}: ${nonFinite} non-finite value(s) read back as text`, name, nonFinite);
|
|
1494
|
+
}
|
|
1495
|
+
return;
|
|
1496
|
+
}
|
|
1497
|
+
if (dtype !== "string" && dtype !== "dict") {
|
|
1498
|
+
return;
|
|
1499
|
+
}
|
|
1500
|
+
let widest = -1;
|
|
1501
|
+
let candidate = true;
|
|
1502
|
+
let set = 0;
|
|
1503
|
+
const heuristic = new DictHeuristic(DICT_SAMPLE_ROWS);
|
|
1504
|
+
for (const row of rows) {
|
|
1505
|
+
if (!column.isSet(row)) {
|
|
1506
|
+
continue;
|
|
1507
|
+
}
|
|
1508
|
+
const text = column.value(row);
|
|
1509
|
+
if (typeof text !== "string") {
|
|
1510
|
+
continue;
|
|
1511
|
+
}
|
|
1512
|
+
set++;
|
|
1513
|
+
const kind = inferTextDtype(text);
|
|
1514
|
+
widest = Math.max(widest, WIDENING[kind]);
|
|
1515
|
+
if (!heuristic.decided) {
|
|
1516
|
+
if (kind !== "string") {
|
|
1517
|
+
candidate = false;
|
|
1518
|
+
}
|
|
1519
|
+
heuristic.observe(text);
|
|
1520
|
+
}
|
|
1521
|
+
}
|
|
1522
|
+
if (roleColumn || widest < 0) {
|
|
1523
|
+
return;
|
|
1524
|
+
}
|
|
1525
|
+
const readsAsDict = candidate && heuristic.decide() === "dict";
|
|
1526
|
+
if (widest < WIDENING.string && !readsAsDict) {
|
|
1527
|
+
note(CSV_LOSS.TEXT_INFERRED, `${label}: every value reads back as ${TEXT_DTYPES[widest]}`, name, set);
|
|
1528
|
+
return;
|
|
1529
|
+
}
|
|
1530
|
+
if (dtype === "dict" && !readsAsDict) {
|
|
1531
|
+
note(
|
|
1532
|
+
CSV_LOSS.STORAGE_CLASS_CHANGED,
|
|
1533
|
+
`${label}: reads back as string (cardinality too high for a dict)`,
|
|
1534
|
+
name,
|
|
1535
|
+
null
|
|
1536
|
+
);
|
|
1537
|
+
} else if (dtype === "string" && readsAsDict) {
|
|
1538
|
+
note(CSV_LOSS.STORAGE_CLASS_CHANGED, `${label}: reads back as dict (low cardinality)`, name, null);
|
|
1539
|
+
}
|
|
1540
|
+
}
|
|
1541
|
+
function typeText(snapshot, plan, e) {
|
|
1542
|
+
if (!snapshot.directed || !plan.folding.sourceDirected(e)) {
|
|
1543
|
+
return "Undirected";
|
|
1544
|
+
}
|
|
1545
|
+
return "Directed";
|
|
1546
|
+
}
|
|
1547
|
+
function* lines(snapshot, plan) {
|
|
1548
|
+
const { csv } = plan;
|
|
1549
|
+
const { delimiter, newline } = csv;
|
|
1550
|
+
const quote = (text) => quoteCsvCell(text, delimiter);
|
|
1551
|
+
const cell = (column, row) => {
|
|
1552
|
+
const text = cellText(column, row);
|
|
1553
|
+
return text === null ? "" : quote(text);
|
|
1554
|
+
};
|
|
1555
|
+
if (csv.table === "nodes") {
|
|
1556
|
+
const headers2 = [csv.dialect.id];
|
|
1557
|
+
if (plan.nodeLabel !== null) {
|
|
1558
|
+
headers2.push(plan.nodeLabel.meta.name);
|
|
1559
|
+
}
|
|
1560
|
+
for (const { header } of plan.nodeColumns) {
|
|
1561
|
+
headers2.push(header);
|
|
1562
|
+
}
|
|
1563
|
+
if (csv.header) {
|
|
1564
|
+
yield headers2.map(quote).join(delimiter) + newline;
|
|
1565
|
+
}
|
|
1566
|
+
for (let i = 0; i < snapshot.nodeCount; i++) {
|
|
1567
|
+
const cells = [quote(plan.idText(i))];
|
|
1568
|
+
if (plan.nodeLabel !== null) {
|
|
1569
|
+
cells.push(cell(plan.nodeLabel, i));
|
|
1570
|
+
}
|
|
1571
|
+
for (const { column } of plan.nodeColumns) {
|
|
1572
|
+
cells.push(cell(column, i));
|
|
1573
|
+
}
|
|
1574
|
+
yield cells.join(delimiter) + newline;
|
|
1575
|
+
}
|
|
1576
|
+
return;
|
|
1577
|
+
}
|
|
1578
|
+
const headers = [csv.dialect.source, csv.dialect.target];
|
|
1579
|
+
if (csv.dialect.type !== null) {
|
|
1580
|
+
headers.push(csv.dialect.type);
|
|
1581
|
+
}
|
|
1582
|
+
if (plan.edgeId !== null) {
|
|
1583
|
+
headers.push(plan.edgeId.meta.name);
|
|
1584
|
+
}
|
|
1585
|
+
if (plan.edgeLabel !== null) {
|
|
1586
|
+
headers.push(plan.edgeLabel.meta.name);
|
|
1587
|
+
}
|
|
1588
|
+
if (plan.weights.weighted) {
|
|
1589
|
+
headers.push(csv.dialect.weight);
|
|
1590
|
+
}
|
|
1591
|
+
for (const { header } of plan.edgeColumns) {
|
|
1592
|
+
headers.push(header);
|
|
1593
|
+
}
|
|
1594
|
+
if (csv.header) {
|
|
1595
|
+
yield headers.map(quote).join(delimiter) + newline;
|
|
1596
|
+
}
|
|
1597
|
+
const list = snapshot.edgeList();
|
|
1598
|
+
for (const e of plan.edgeRows) {
|
|
1599
|
+
const cells = [quote(plan.idText(list.src[e])), quote(plan.idText(list.dst[e]))];
|
|
1600
|
+
if (csv.dialect.type !== null) {
|
|
1601
|
+
cells.push(typeText(snapshot, plan, e));
|
|
1602
|
+
}
|
|
1603
|
+
if (plan.edgeId !== null) {
|
|
1604
|
+
cells.push(cell(plan.edgeId, e));
|
|
1605
|
+
}
|
|
1606
|
+
if (plan.edgeLabel !== null) {
|
|
1607
|
+
cells.push(cell(plan.edgeLabel, e));
|
|
1608
|
+
}
|
|
1609
|
+
if (plan.weights.weighted) {
|
|
1610
|
+
cells.push(plan.weights.text(e) ?? "");
|
|
1611
|
+
}
|
|
1612
|
+
for (const { column } of plan.edgeColumns) {
|
|
1613
|
+
cells.push(cell(column, e));
|
|
1614
|
+
}
|
|
1615
|
+
yield cells.join(delimiter) + newline;
|
|
1616
|
+
}
|
|
1617
|
+
}
|
|
1618
|
+
function prepare(snapshot, options) {
|
|
1619
|
+
const plan = planExport(snapshot, options, false);
|
|
1620
|
+
const collision = plan.notes.find((n) => n.code === CSV_LOSS.ID_TEXT_COLLISION);
|
|
1621
|
+
if (collision !== void 0) {
|
|
1622
|
+
throw new GraphFormatError("E_INVALID_ID", collision.message, {
|
|
1623
|
+
reason: "collision",
|
|
1624
|
+
count: collision.count
|
|
1625
|
+
});
|
|
1626
|
+
}
|
|
1627
|
+
return plan;
|
|
1628
|
+
}
|
|
1629
|
+
function check(snapshot, options) {
|
|
1630
|
+
const plan = planExport(snapshot, options, true);
|
|
1631
|
+
return Object.freeze([
|
|
1632
|
+
...checkCapabilities(snapshot, CSV_CAPABILITIES, plan.common, { roles: KEPT_ROLES }),
|
|
1633
|
+
...plan.notes
|
|
1634
|
+
]);
|
|
1635
|
+
}
|
|
1636
|
+
const csvExporter = Object.freeze({
|
|
1637
|
+
format: "csv",
|
|
1638
|
+
capabilities: CSV_CAPABILITIES,
|
|
1639
|
+
check,
|
|
1640
|
+
export(snapshot, options) {
|
|
1641
|
+
return encodeChunks(lines(snapshot, prepare(snapshot, options)));
|
|
1642
|
+
},
|
|
1643
|
+
exportToString(snapshot, options) {
|
|
1644
|
+
return joinText(lines(snapshot, prepare(snapshot, options)));
|
|
1645
|
+
}
|
|
1646
|
+
});
|
|
1647
|
+
const CSV_ISSUE = Object.freeze({
|
|
1648
|
+
/** The input is empty (fatal). */
|
|
1649
|
+
EMPTY_INPUT: EMPTY_INPUT_CODE,
|
|
1650
|
+
/** The input holds invalid UTF-8 (fatal). */
|
|
1651
|
+
INVALID_UTF8: INVALID_UTF8_CODE,
|
|
1652
|
+
/** The header names neither endpoint columns nor an id column (fatal). */
|
|
1653
|
+
NO_ENDPOINT_COLUMNS: NO_ENDPOINT_COLUMNS_CODE,
|
|
1654
|
+
/** A node table without an id column (fatal). */
|
|
1655
|
+
NO_ID_COLUMN: NO_ID_COLUMN_CODE,
|
|
1656
|
+
/** A row with a different field count than the header. */
|
|
1657
|
+
FIELD_COUNT: FIELD_COUNT_CODE,
|
|
1658
|
+
/** An edge row with a blank source or target. */
|
|
1659
|
+
MISSING_ENDPOINT: MISSING_ENDPOINT_CODE,
|
|
1660
|
+
/** A node row with a blank id. */
|
|
1661
|
+
MISSING_ID: MISSING_ID_CODE,
|
|
1662
|
+
/** A Type cell outside Directed / Undirected / Mutual. */
|
|
1663
|
+
BAD_TYPE: BAD_TYPE_CODE,
|
|
1664
|
+
/** An unterminated quoted field (fatal). */
|
|
1665
|
+
UNCLOSED_QUOTE: UNCLOSED_QUOTE_CODE,
|
|
1666
|
+
/** Text after a closing quote (fatal). */
|
|
1667
|
+
QUOTE: BAD_QUOTE_CODE,
|
|
1668
|
+
/** A header and no data rows. */
|
|
1669
|
+
NO_DATA_ROWS: NO_DATA_ROWS_CODE,
|
|
1670
|
+
/** A node table row repeating an id. */
|
|
1671
|
+
DUPLICATE_NODE: DUPLICATE_NODE_CODE,
|
|
1672
|
+
/** An edge row repeating an edge id (skipped). */
|
|
1673
|
+
DUPLICATE_EDGE_ID: DUPLICATE_EDGE_ID_CODE,
|
|
1674
|
+
/** Two id cells merged into one number under ids "number". */
|
|
1675
|
+
ID_MERGED: ID_MERGED_CODE,
|
|
1676
|
+
/** An explicitly named weight column the file does not have. */
|
|
1677
|
+
COLUMN_MISSING: COLUMN_MISSING_CODE,
|
|
1678
|
+
/** A column whose role another column of the sink already holds. */
|
|
1679
|
+
ROLE_TAKEN: ROLE_TAKEN_CODE,
|
|
1680
|
+
/** A column renamed `<name>#<position>` (a repeated header, or a name the sink holds with another shape). */
|
|
1681
|
+
COLUMN_RENAMED: COLUMN_RENAMED_CODE,
|
|
1682
|
+
/** A common option the importer has no use for was given. */
|
|
1683
|
+
OPTION_IGNORED: OPTION_IGNORED_CODE,
|
|
1684
|
+
/** A builder-policy option the sink does not honour. */
|
|
1685
|
+
SINK_OPTION: SINK_OPTION_CODE,
|
|
1686
|
+
/** The sink refused the file's direction. */
|
|
1687
|
+
DIRECTION_REFUSED: DIRECTION_REFUSED_CODE,
|
|
1688
|
+
/** Edges forced to the policy's direction. */
|
|
1689
|
+
DIRECTION_FORCED: DIRECTION_FORCED_CODE,
|
|
1690
|
+
/** A mixed file under onMixedDirection "error" (fatal). */
|
|
1691
|
+
MIXED_DIRECTION: MIXED_DIRECTION_CODE,
|
|
1692
|
+
/** A text column the sink could not widen to the dtype its cells imply. */
|
|
1693
|
+
WIDENING_UNSUPPORTED: WIDENING_UNSUPPORTED_CODE
|
|
1694
|
+
});
|
|
1695
|
+
export {
|
|
1696
|
+
CSV_CAPABILITIES,
|
|
1697
|
+
CSV_ISSUE,
|
|
1698
|
+
CSV_LOSS,
|
|
1699
|
+
csvExporter,
|
|
1700
|
+
csvImporter
|
|
1701
|
+
};
|
|
1702
|
+
//# sourceMappingURL=csv.js.map
|