@graphty/graph-io 0.0.0 → 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +250 -28
- package/dist/chunks/children-CL3Cy0ez.js +238 -0
- package/dist/chunks/children-CL3Cy0ez.js.map +1 -0
- package/dist/chunks/escape-DyI8JofU.js +938 -0
- package/dist/chunks/escape-DyI8JofU.js.map +1 -0
- package/dist/chunks/importer-CQnJuWJw.js +2987 -0
- package/dist/chunks/importer-CQnJuWJw.js.map +1 -0
- package/dist/chunks/importer-CpCpfbxr.js +2015 -0
- package/dist/chunks/importer-CpCpfbxr.js.map +1 -0
- package/dist/chunks/importer-DbnGYr3_.js +2342 -0
- package/dist/chunks/importer-DbnGYr3_.js.map +1 -0
- package/dist/chunks/importer-GozH8DkN.js +3050 -0
- package/dist/chunks/importer-GozH8DkN.js.map +1 -0
- package/dist/chunks/records-CGpxszm1.js +605 -0
- package/dist/chunks/records-CGpxszm1.js.map +1 -0
- package/dist/chunks/text-CajMdVFy.js +189 -0
- package/dist/chunks/text-CajMdVFy.js.map +1 -0
- package/dist/chunks/writer-DxSKC7TL.js +2842 -0
- package/dist/chunks/writer-DxSKC7TL.js.map +1 -0
- package/dist/csv.d.ts +1 -0
- package/dist/csv.js +1702 -0
- package/dist/csv.js.map +1 -0
- package/dist/dot.d.ts +1 -0
- package/dist/dot.js +8 -0
- package/dist/dot.js.map +1 -0
- package/dist/gexf.d.ts +1 -0
- package/dist/gexf.js +3466 -0
- package/dist/gexf.js.map +1 -0
- package/dist/gml.d.ts +1 -0
- package/dist/gml.js +2647 -0
- package/dist/gml.js.map +1 -0
- package/dist/graph-io.d.ts +1 -0
- package/dist/graph-io.js +790 -0
- package/dist/graph-io.js.map +1 -0
- package/dist/graphml.d.ts +1 -0
- package/dist/graphml.js +8 -0
- package/dist/graphml.js.map +1 -0
- package/dist/json.d.ts +1 -0
- package/dist/json.js +11 -0
- package/dist/json.js.map +1 -0
- package/dist/neo4j.d.ts +1 -0
- package/dist/neo4j.js +2046 -0
- package/dist/neo4j.js.map +1 -0
- package/dist/pajek.d.ts +1 -0
- package/dist/pajek.js +8 -0
- package/dist/pajek.js.map +1 -0
- package/dist/src/children.d.ts +134 -0
- package/dist/src/children.d.ts.map +1 -0
- package/dist/src/children.js +274 -0
- package/dist/src/children.js.map +1 -0
- package/dist/src/common/attributes.d.ts +229 -0
- package/dist/src/common/attributes.d.ts.map +1 -0
- package/dist/src/common/attributes.js +368 -0
- package/dist/src/common/attributes.js.map +1 -0
- package/dist/src/common/codes.d.ts +105 -0
- package/dist/src/common/codes.d.ts.map +1 -0
- package/dist/src/common/codes.js +107 -0
- package/dist/src/common/codes.js.map +1 -0
- package/dist/src/common/declared-types.d.ts +84 -0
- package/dist/src/common/declared-types.d.ts.map +1 -0
- package/dist/src/common/declared-types.js +326 -0
- package/dist/src/common/declared-types.js.map +1 -0
- package/dist/src/common/direction.d.ts +206 -0
- package/dist/src/common/direction.d.ts.map +1 -0
- package/dist/src/common/direction.js +370 -0
- package/dist/src/common/direction.js.map +1 -0
- package/dist/src/common/escape.d.ts +92 -0
- package/dist/src/common/escape.d.ts.map +1 -0
- package/dist/src/common/escape.js +212 -0
- package/dist/src/common/escape.js.map +1 -0
- package/dist/src/common/export.d.ts +249 -0
- package/dist/src/common/export.d.ts.map +1 -0
- package/dist/src/common/export.js +594 -0
- package/dist/src/common/export.js.map +1 -0
- package/dist/src/common/format.d.ts +59 -0
- package/dist/src/common/format.d.ts.map +1 -0
- package/dist/src/common/format.js +106 -0
- package/dist/src/common/format.js.map +1 -0
- package/dist/src/common/ids.d.ts +83 -0
- package/dist/src/common/ids.d.ts.map +1 -0
- package/dist/src/common/ids.js +158 -0
- package/dist/src/common/ids.js.map +1 -0
- package/dist/src/common/input.d.ts +100 -0
- package/dist/src/common/input.d.ts.map +1 -0
- package/dist/src/common/input.js +335 -0
- package/dist/src/common/input.js.map +1 -0
- package/dist/src/common/lists.d.ts +34 -0
- package/dist/src/common/lists.d.ts.map +1 -0
- package/dist/src/common/lists.js +185 -0
- package/dist/src/common/lists.js.map +1 -0
- package/dist/src/common/options.d.ts +108 -0
- package/dist/src/common/options.d.ts.map +1 -0
- package/dist/src/common/options.js +265 -0
- package/dist/src/common/options.js.map +1 -0
- package/dist/src/common/report.d.ts +187 -0
- package/dist/src/common/report.d.ts.map +1 -0
- package/dist/src/common/report.js +274 -0
- package/dist/src/common/report.js.map +1 -0
- package/dist/src/common/temporal.d.ts +71 -0
- package/dist/src/common/temporal.d.ts.map +1 -0
- package/dist/src/common/temporal.js +266 -0
- package/dist/src/common/temporal.js.map +1 -0
- package/dist/src/common/text.d.ts +104 -0
- package/dist/src/common/text.d.ts.map +1 -0
- package/dist/src/common/text.js +255 -0
- package/dist/src/common/text.js.map +1 -0
- package/dist/src/common/weights.d.ts +77 -0
- package/dist/src/common/weights.d.ts.map +1 -0
- package/dist/src/common/weights.js +156 -0
- package/dist/src/common/weights.js.map +1 -0
- package/dist/src/common/writer.d.ts +51 -0
- package/dist/src/common/writer.d.ts.map +1 -0
- package/dist/src/common/writer.js +108 -0
- package/dist/src/common/writer.js.map +1 -0
- package/dist/src/common/xml.d.ts +245 -0
- package/dist/src/common/xml.d.ts.map +1 -0
- package/dist/src/common/xml.js +942 -0
- package/dist/src/common/xml.js.map +1 -0
- package/dist/src/formats/csv/exporter.d.ts +70 -0
- package/dist/src/formats/csv/exporter.d.ts.map +1 -0
- package/dist/src/formats/csv/exporter.js +682 -0
- package/dist/src/formats/csv/exporter.js.map +1 -0
- package/dist/src/formats/csv/header.d.ts +66 -0
- package/dist/src/formats/csv/header.d.ts.map +1 -0
- package/dist/src/formats/csv/header.js +152 -0
- package/dist/src/formats/csv/header.js.map +1 -0
- package/dist/src/formats/csv/importer.d.ts +82 -0
- package/dist/src/formats/csv/importer.d.ts.map +1 -0
- package/dist/src/formats/csv/importer.js +849 -0
- package/dist/src/formats/csv/importer.js.map +1 -0
- package/dist/src/formats/csv/index.d.ts +60 -0
- package/dist/src/formats/csv/index.d.ts.map +1 -0
- package/dist/src/formats/csv/index.js +63 -0
- package/dist/src/formats/csv/index.js.map +1 -0
- package/dist/src/formats/csv/records.d.ts +188 -0
- package/dist/src/formats/csv/records.d.ts.map +1 -0
- package/dist/src/formats/csv/records.js +702 -0
- package/dist/src/formats/csv/records.js.map +1 -0
- package/dist/src/formats/csv/values.d.ts +105 -0
- package/dist/src/formats/csv/values.d.ts.map +1 -0
- package/dist/src/formats/csv/values.js +192 -0
- package/dist/src/formats/csv/values.js.map +1 -0
- package/dist/src/formats/dot/exporter.d.ts +52 -0
- package/dist/src/formats/dot/exporter.d.ts.map +1 -0
- package/dist/src/formats/dot/exporter.js +836 -0
- package/dist/src/formats/dot/exporter.js.map +1 -0
- package/dist/src/formats/dot/importer.d.ts +102 -0
- package/dist/src/formats/dot/importer.d.ts.map +1 -0
- package/dist/src/formats/dot/importer.js +1291 -0
- package/dist/src/formats/dot/importer.js.map +1 -0
- package/dist/src/formats/dot/index.d.ts +7 -0
- package/dist/src/formats/dot/index.d.ts.map +1 -0
- package/dist/src/formats/dot/index.js +7 -0
- package/dist/src/formats/dot/index.js.map +1 -0
- package/dist/src/formats/dot/names.d.ts +29 -0
- package/dist/src/formats/dot/names.d.ts.map +1 -0
- package/dist/src/formats/dot/names.js +28 -0
- package/dist/src/formats/dot/names.js.map +1 -0
- package/dist/src/formats/dot/tokenizer.d.ts +114 -0
- package/dist/src/formats/dot/tokenizer.d.ts.map +1 -0
- package/dist/src/formats/dot/tokenizer.js +341 -0
- package/dist/src/formats/dot/tokenizer.js.map +1 -0
- package/dist/src/formats/gexf/exporter.d.ts +56 -0
- package/dist/src/formats/gexf/exporter.d.ts.map +1 -0
- package/dist/src/formats/gexf/exporter.js +1395 -0
- package/dist/src/formats/gexf/exporter.js.map +1 -0
- package/dist/src/formats/gexf/importer.d.ts +73 -0
- package/dist/src/formats/gexf/importer.d.ts.map +1 -0
- package/dist/src/formats/gexf/importer.js +1880 -0
- package/dist/src/formats/gexf/importer.js.map +1 -0
- package/dist/src/formats/gexf/index.d.ts +96 -0
- package/dist/src/formats/gexf/index.d.ts.map +1 -0
- package/dist/src/formats/gexf/index.js +97 -0
- package/dist/src/formats/gexf/index.js.map +1 -0
- package/dist/src/formats/gexf/schema.d.ts +135 -0
- package/dist/src/formats/gexf/schema.d.ts.map +1 -0
- package/dist/src/formats/gexf/schema.js +323 -0
- package/dist/src/formats/gexf/schema.js.map +1 -0
- package/dist/src/formats/gml/exporter.d.ts +69 -0
- package/dist/src/formats/gml/exporter.d.ts.map +1 -0
- package/dist/src/formats/gml/exporter.js +1093 -0
- package/dist/src/formats/gml/exporter.js.map +1 -0
- package/dist/src/formats/gml/importer.d.ts +66 -0
- package/dist/src/formats/gml/importer.d.ts.map +1 -0
- package/dist/src/formats/gml/importer.js +1331 -0
- package/dist/src/formats/gml/importer.js.map +1 -0
- package/dist/src/formats/gml/index.d.ts +85 -0
- package/dist/src/formats/gml/index.d.ts.map +1 -0
- package/dist/src/formats/gml/index.js +88 -0
- package/dist/src/formats/gml/index.js.map +1 -0
- package/dist/src/formats/gml/syntax.d.ts +186 -0
- package/dist/src/formats/gml/syntax.d.ts.map +1 -0
- package/dist/src/formats/gml/syntax.js +467 -0
- package/dist/src/formats/gml/syntax.js.map +1 -0
- package/dist/src/formats/graphml/constants.d.ts +169 -0
- package/dist/src/formats/graphml/constants.d.ts.map +1 -0
- package/dist/src/formats/graphml/constants.js +165 -0
- package/dist/src/formats/graphml/constants.js.map +1 -0
- package/dist/src/formats/graphml/exporter.d.ts +34 -0
- package/dist/src/formats/graphml/exporter.d.ts.map +1 -0
- package/dist/src/formats/graphml/exporter.js +1176 -0
- package/dist/src/formats/graphml/exporter.js.map +1 -0
- package/dist/src/formats/graphml/importer.d.ts +31 -0
- package/dist/src/formats/graphml/importer.d.ts.map +1 -0
- package/dist/src/formats/graphml/importer.js +1607 -0
- package/dist/src/formats/graphml/importer.js.map +1 -0
- package/dist/src/formats/graphml/index.d.ts +8 -0
- package/dist/src/formats/graphml/index.d.ts.map +1 -0
- package/dist/src/formats/graphml/index.js +8 -0
- package/dist/src/formats/graphml/index.js.map +1 -0
- package/dist/src/formats/graphml/tree.d.ts +72 -0
- package/dist/src/formats/graphml/tree.d.ts.map +1 -0
- package/dist/src/formats/graphml/tree.js +290 -0
- package/dist/src/formats/graphml/tree.js.map +1 -0
- package/dist/src/formats/json/dialect.d.ts +125 -0
- package/dist/src/formats/json/dialect.d.ts.map +1 -0
- package/dist/src/formats/json/dialect.js +262 -0
- package/dist/src/formats/json/dialect.js.map +1 -0
- package/dist/src/formats/json/exporter.d.ts +89 -0
- package/dist/src/formats/json/exporter.d.ts.map +1 -0
- package/dist/src/formats/json/exporter.js +1358 -0
- package/dist/src/formats/json/exporter.js.map +1 -0
- package/dist/src/formats/json/importer.d.ts +108 -0
- package/dist/src/formats/json/importer.d.ts.map +1 -0
- package/dist/src/formats/json/importer.js +1838 -0
- package/dist/src/formats/json/importer.js.map +1 -0
- package/dist/src/formats/json/index.d.ts +8 -0
- package/dist/src/formats/json/index.d.ts.map +1 -0
- package/dist/src/formats/json/index.js +8 -0
- package/dist/src/formats/json/index.js.map +1 -0
- package/dist/src/formats/neo4j/exporter.d.ts +68 -0
- package/dist/src/formats/neo4j/exporter.d.ts.map +1 -0
- package/dist/src/formats/neo4j/exporter.js +1055 -0
- package/dist/src/formats/neo4j/exporter.js.map +1 -0
- package/dist/src/formats/neo4j/header.d.ts +52 -0
- package/dist/src/formats/neo4j/header.d.ts.map +1 -0
- package/dist/src/formats/neo4j/header.js +131 -0
- package/dist/src/formats/neo4j/header.js.map +1 -0
- package/dist/src/formats/neo4j/importer.d.ts +73 -0
- package/dist/src/formats/neo4j/importer.d.ts.map +1 -0
- package/dist/src/formats/neo4j/importer.js +932 -0
- package/dist/src/formats/neo4j/importer.js.map +1 -0
- package/dist/src/formats/neo4j/index.d.ts +79 -0
- package/dist/src/formats/neo4j/index.d.ts.map +1 -0
- package/dist/src/formats/neo4j/index.js +83 -0
- package/dist/src/formats/neo4j/index.js.map +1 -0
- package/dist/src/formats/pajek/exporter.d.ts +58 -0
- package/dist/src/formats/pajek/exporter.d.ts.map +1 -0
- package/dist/src/formats/pajek/exporter.js +825 -0
- package/dist/src/formats/pajek/exporter.js.map +1 -0
- package/dist/src/formats/pajek/importer.d.ts +88 -0
- package/dist/src/formats/pajek/importer.d.ts.map +1 -0
- package/dist/src/formats/pajek/importer.js +1047 -0
- package/dist/src/formats/pajek/importer.js.map +1 -0
- package/dist/src/formats/pajek/index.d.ts +7 -0
- package/dist/src/formats/pajek/index.d.ts.map +1 -0
- package/dist/src/formats/pajek/index.js +7 -0
- package/dist/src/formats/pajek/index.js.map +1 -0
- package/dist/src/formats/pajek/syntax.d.ts +112 -0
- package/dist/src/formats/pajek/syntax.d.ts.map +1 -0
- package/dist/src/formats/pajek/syntax.js +269 -0
- package/dist/src/formats/pajek/syntax.js.map +1 -0
- package/dist/src/index.d.ts +35 -0
- package/dist/src/index.d.ts.map +1 -0
- package/dist/src/index.js +39 -0
- package/dist/src/index.js.map +1 -0
- package/dist/src/registry.d.ts +207 -0
- package/dist/src/registry.d.ts.map +1 -0
- package/dist/src/registry.js +481 -0
- package/dist/src/registry.js.map +1 -0
- package/dist/src/sniff.d.ts +104 -0
- package/dist/src/sniff.d.ts.map +1 -0
- package/dist/src/sniff.js +357 -0
- package/dist/src/sniff.js.map +1 -0
- package/dist/src/types.d.ts +238 -0
- package/dist/src/types.d.ts.map +1 -0
- package/dist/src/types.js +29 -0
- package/dist/src/types.js.map +1 -0
- package/dist/tsconfig.build.tsbuildinfo +1 -0
- package/package.json +122 -7
- package/src/children.ts +335 -0
- package/src/common/attributes.ts +520 -0
- package/src/common/codes.ts +153 -0
- package/src/common/declared-types.ts +374 -0
- package/src/common/direction.ts +518 -0
- package/src/common/escape.ts +231 -0
- package/src/common/export.ts +817 -0
- package/src/common/format.ts +111 -0
- package/src/common/ids.ts +176 -0
- package/src/common/input.ts +378 -0
- package/src/common/lists.ts +196 -0
- package/src/common/options.ts +377 -0
- package/src/common/report.ts +352 -0
- package/src/common/temporal.ts +302 -0
- package/src/common/text.ts +294 -0
- package/src/common/weights.ts +202 -0
- package/src/common/writer.ts +123 -0
- package/src/common/xml.ts +1053 -0
- package/src/formats/csv/exporter.ts +894 -0
- package/src/formats/csv/header.ts +172 -0
- package/src/formats/csv/importer.ts +1104 -0
- package/src/formats/csv/index.ts +88 -0
- package/src/formats/csv/records.ts +813 -0
- package/src/formats/csv/values.ts +224 -0
- package/src/formats/dot/exporter.ts +1014 -0
- package/src/formats/dot/importer.ts +1549 -0
- package/src/formats/dot/index.ts +7 -0
- package/src/formats/dot/names.ts +40 -0
- package/src/formats/dot/tokenizer.ts +384 -0
- package/src/formats/gexf/exporter.ts +1696 -0
- package/src/formats/gexf/importer.ts +2333 -0
- package/src/formats/gexf/index.ts +142 -0
- package/src/formats/gexf/schema.ts +361 -0
- package/src/formats/gml/exporter.ts +1404 -0
- package/src/formats/gml/importer.ts +1591 -0
- package/src/formats/gml/index.ts +128 -0
- package/src/formats/gml/syntax.ts +545 -0
- package/src/formats/graphml/constants.ts +225 -0
- package/src/formats/graphml/exporter.ts +1458 -0
- package/src/formats/graphml/importer.ts +2027 -0
- package/src/formats/graphml/index.ts +8 -0
- package/src/formats/graphml/tree.ts +318 -0
- package/src/formats/json/dialect.ts +317 -0
- package/src/formats/json/exporter.ts +1616 -0
- package/src/formats/json/importer.ts +2271 -0
- package/src/formats/json/index.ts +8 -0
- package/src/formats/neo4j/exporter.ts +1287 -0
- package/src/formats/neo4j/header.ts +156 -0
- package/src/formats/neo4j/importer.ts +1220 -0
- package/src/formats/neo4j/index.ts +116 -0
- package/src/formats/pajek/exporter.ts +1000 -0
- package/src/formats/pajek/importer.ts +1311 -0
- package/src/formats/pajek/index.ts +7 -0
- package/src/formats/pajek/syntax.ts +307 -0
- package/src/index.ts +244 -0
- package/src/registry.ts +617 -0
- package/src/sniff.ts +397 -0
- package/src/types.ts +262 -0
|
@@ -0,0 +1,1104 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The CSV / TSV importer (design sections 8.4 and 8.6; research note 07 section 2.6): a streaming
|
|
3
|
+
* edge-list reader for the generic (`source,target[,weight,...]`), Gephi (`Source,Target,Type,Id,
|
|
4
|
+
* Label,Weight,...`) and headerless (`u v [w]`) dialects, with an optional node table merged by id.
|
|
5
|
+
*
|
|
6
|
+
* - The delimiter is sniffed from a preview unless given; LF, CRLF and lone-CR files all read.
|
|
7
|
+
* - The first row is a header when it holds a known column name or when it is all text over a
|
|
8
|
+
* numeric second row (`header: "auto"`); a headerless file is positional: source, target,
|
|
9
|
+
* weight (when `weightFrom` is not null), then `column4`... as attributes.
|
|
10
|
+
* - Endpoint and node ids are text cells coerced by ONE rule per import call (`ids`, default
|
|
11
|
+
* "canonical": `"1"` becomes the number 1, `"01"` stays a string) over the node table and the
|
|
12
|
+
* edge table alike, so they agree across the two files.
|
|
13
|
+
* - Direction is per row in the Gephi dialect (`Type` = Directed / Undirected / Mutual, blank =
|
|
14
|
+
* `defaultDirected`) and `defaultDirected` (true) otherwise; the first edge row sets the sink's
|
|
15
|
+
* direction and later rows that differ go through `onMixedDirection` (expansion by default).
|
|
16
|
+
* - The weight column (`weightFrom`, default "weight", matched case-insensitively) is parsed per
|
|
17
|
+
* row; a blank cell means "no weight given" and the edge is pushed without one.
|
|
18
|
+
* - Every other column is an attribute: cells are parsed by the fixed lexical grammar of design
|
|
19
|
+
* section 5.1 and the sink infers the column dtype (widening per column, never per cell); an
|
|
20
|
+
* all-text column of low cardinality becomes a dict (design section 5.4); an `id` column of the
|
|
21
|
+
* edge table is the edge id (role id, unique); a `label` column is the label (role label).
|
|
22
|
+
* - Per-row problems (wrong field count, blank endpoint, invalid weight, bad Type, refused id)
|
|
23
|
+
* are recorded and the row skipped; the import aborts with ImportError once `errorLimit` is
|
|
24
|
+
* exceeded, on a malformed or unterminated quoted field, on an empty input and on a header
|
|
25
|
+
* without endpoint (or id) columns.
|
|
26
|
+
*/
|
|
27
|
+
|
|
28
|
+
import {
|
|
29
|
+
type ColumnDecl,
|
|
30
|
+
type ColumnHandle,
|
|
31
|
+
GraphFormatError,
|
|
32
|
+
type GraphSink,
|
|
33
|
+
INVALID_INDEX,
|
|
34
|
+
type NodeId,
|
|
35
|
+
} from "@graphty/graph-format";
|
|
36
|
+
|
|
37
|
+
import { declareResolved, RENAMED_CODE, uniqueColumnName } from "../../common/attributes.js";
|
|
38
|
+
import {
|
|
39
|
+
DUPLICATE_EDGE_ID_CODE as SHARED_DUPLICATE_EDGE_ID_CODE,
|
|
40
|
+
DUPLICATE_NODE_CODE as SHARED_DUPLICATE_NODE_CODE,
|
|
41
|
+
EMPTY_INPUT_CODE as SHARED_EMPTY_INPUT_CODE,
|
|
42
|
+
ID_MERGED_CODE as SHARED_ID_MERGED_CODE,
|
|
43
|
+
MISSING_ENDPOINT_CODE as SHARED_MISSING_ENDPOINT_CODE,
|
|
44
|
+
MISSING_ID_CODE as SHARED_MISSING_ID_CODE,
|
|
45
|
+
ROLE_TAKEN_CODE as SHARED_ROLE_TAKEN_CODE,
|
|
46
|
+
} from "../../common/codes.js";
|
|
47
|
+
import { DirectionResolver, type EdgeKind } from "../../common/direction.js";
|
|
48
|
+
import { IdCoercer } from "../../common/ids.js";
|
|
49
|
+
import { throwIfAborted } from "../../common/input.js";
|
|
50
|
+
import {
|
|
51
|
+
reportSinkOptions,
|
|
52
|
+
reportUnusedOptions,
|
|
53
|
+
type ResolvedImportOptions,
|
|
54
|
+
resolveImportOptions,
|
|
55
|
+
} from "../../common/options.js";
|
|
56
|
+
import { ImportReportBuilder } from "../../common/report.js";
|
|
57
|
+
import { parseWeightText } from "../../common/weights.js";
|
|
58
|
+
import { type CommonImportOptions, type GraphImporter, type ImportInput, type ImportReport } from "../../types.js";
|
|
59
|
+
import {
|
|
60
|
+
type CsvColumnRef,
|
|
61
|
+
EDGE_ID_NAMES,
|
|
62
|
+
findColumn,
|
|
63
|
+
headerNames,
|
|
64
|
+
ID_NAMES,
|
|
65
|
+
LABEL_NAMES,
|
|
66
|
+
looksLikeHeader,
|
|
67
|
+
positionalNames,
|
|
68
|
+
resolveColumnRef,
|
|
69
|
+
SOURCE_NAMES,
|
|
70
|
+
TARGET_NAMES,
|
|
71
|
+
TYPE_NAME,
|
|
72
|
+
} from "./header.js";
|
|
73
|
+
import { type CsvReaderOptions, CsvRecordReader, sniffDelimiter, sniffNewline } from "./records.js";
|
|
74
|
+
import { InferredColumn } from "./values.js";
|
|
75
|
+
|
|
76
|
+
/** The format-specific options of the CSV importer. */
|
|
77
|
+
export interface CsvImportOptions {
|
|
78
|
+
/** The field delimiter; sniffed from the first rows when omitted (`,`, tab, `;`, `|`, space). */
|
|
79
|
+
delimiter?: string | undefined;
|
|
80
|
+
/** Whether the first row is a header; "auto" (default) decides from its content. */
|
|
81
|
+
header?: boolean | "auto" | undefined;
|
|
82
|
+
/**
|
|
83
|
+
* What the input is: an edge table, a node table, or "auto" (default): an edge table when
|
|
84
|
+
* source and target columns resolve, a node table when only an id column does.
|
|
85
|
+
*/
|
|
86
|
+
table?: "edges" | "nodes" | "auto" | undefined;
|
|
87
|
+
/** The source column, by name or 0-based position; resolved from the header by default. */
|
|
88
|
+
sourceColumn?: CsvColumnRef | undefined;
|
|
89
|
+
/** The target column, by name or 0-based position; resolved from the header by default. */
|
|
90
|
+
targetColumn?: CsvColumnRef | undefined;
|
|
91
|
+
/**
|
|
92
|
+
* The per-row direction column (Directed / Undirected / Mutual); by default the exact `Type`
|
|
93
|
+
* column of a Gephi table (exact `Source` and `Target` headers); null reads no such column.
|
|
94
|
+
*/
|
|
95
|
+
typeColumn?: CsvColumnRef | null | undefined;
|
|
96
|
+
/** The id column of a node table, by name or position; resolved from the header by default. */
|
|
97
|
+
idColumn?: CsvColumnRef | undefined;
|
|
98
|
+
/** A node table read before the edges: its ids become nodes and its other columns node attributes. */
|
|
99
|
+
nodes?: ImportInput | undefined;
|
|
100
|
+
}
|
|
101
|
+
|
|
102
|
+
/** Issue code: the input holds no header row at all. */
|
|
103
|
+
export const EMPTY_INPUT_CODE = SHARED_EMPTY_INPUT_CODE;
|
|
104
|
+
/** Issue code: the header names no source / target (or, for a node table, no id) column. */
|
|
105
|
+
export const NO_ENDPOINT_COLUMNS_CODE = "E_CSV_NO_ENDPOINT_COLUMNS";
|
|
106
|
+
/** Issue code: a node table without an id column. */
|
|
107
|
+
export const NO_ID_COLUMN_CODE = "E_CSV_NO_ID_COLUMN";
|
|
108
|
+
/** Issue code: a row with a different number of fields than the header. */
|
|
109
|
+
export const FIELD_COUNT_CODE = "E_CSV_FIELD_COUNT";
|
|
110
|
+
/** Issue code: an edge row with a blank source or target cell. */
|
|
111
|
+
export const MISSING_ENDPOINT_CODE = SHARED_MISSING_ENDPOINT_CODE;
|
|
112
|
+
/** Issue code: a node row with a blank id cell. */
|
|
113
|
+
export const MISSING_ID_CODE = SHARED_MISSING_ID_CODE;
|
|
114
|
+
/** Issue code: a Type cell that is not Directed, Undirected or Mutual. */
|
|
115
|
+
export const BAD_TYPE_CODE = "E_CSV_BAD_TYPE";
|
|
116
|
+
/** Issue code: the table has a header and no data rows. */
|
|
117
|
+
export const NO_DATA_ROWS_CODE = "W_CSV_NO_DATA_ROWS";
|
|
118
|
+
/** Issue code: a node table row repeats an id; its attributes overwrite the earlier row's. */
|
|
119
|
+
export const DUPLICATE_NODE_CODE = SHARED_DUPLICATE_NODE_CODE;
|
|
120
|
+
/** Issue code: two distinct id cells became one id under `ids: "number"` (design section 4.1). */
|
|
121
|
+
export const ID_MERGED_CODE = SHARED_ID_MERGED_CODE;
|
|
122
|
+
/** Issue code: an explicitly named weight column the file does not have. */
|
|
123
|
+
export const COLUMN_MISSING_CODE = "W_CSV_COLUMN_MISSING";
|
|
124
|
+
/** Issue code: a column whose role (id, label) is already held by another column of the sink. */
|
|
125
|
+
export const ROLE_TAKEN_CODE = SHARED_ROLE_TAKEN_CODE;
|
|
126
|
+
/** Issue code: a repeated edge id (the column is unique); the edge is skipped. */
|
|
127
|
+
export const DUPLICATE_EDGE_ID_CODE = SHARED_DUPLICATE_EDGE_ID_CODE;
|
|
128
|
+
|
|
129
|
+
const TABLE_MODES: ReadonlySet<string> = new Set(["edges", "nodes", "auto"]);
|
|
130
|
+
|
|
131
|
+
/** The common options an edge-table import reads (the rest is reported by reportUnusedOptions). */
|
|
132
|
+
const USED_OPTIONS: ReadonlySet<keyof CommonImportOptions> = new Set<keyof CommonImportOptions>([
|
|
133
|
+
"ids",
|
|
134
|
+
"addMissingNodes",
|
|
135
|
+
"duplicateEdges",
|
|
136
|
+
"selfLoops",
|
|
137
|
+
"onMixedDirection",
|
|
138
|
+
"defaultDirected",
|
|
139
|
+
"weightFrom",
|
|
140
|
+
"weightDtype",
|
|
141
|
+
"errorLimit",
|
|
142
|
+
"signal",
|
|
143
|
+
"onProgress",
|
|
144
|
+
]);
|
|
145
|
+
|
|
146
|
+
/** The common options an import with a node table reads: nodeIdFrom applies to the node table. */
|
|
147
|
+
const USED_OPTIONS_WITH_NODES: ReadonlySet<keyof CommonImportOptions> = new Set<keyof CommonImportOptions>([
|
|
148
|
+
...USED_OPTIONS,
|
|
149
|
+
"nodeIdFrom",
|
|
150
|
+
]);
|
|
151
|
+
const BAD_DELIMITERS: ReadonlySet<string> = new Set(['"', "\n", "\r"]);
|
|
152
|
+
|
|
153
|
+
/** The CSV options with defaults applied. */
|
|
154
|
+
interface ResolvedCsvOptions {
|
|
155
|
+
readonly delimiter: string | null;
|
|
156
|
+
readonly header: boolean | "auto";
|
|
157
|
+
readonly table: "edges" | "nodes" | "auto";
|
|
158
|
+
readonly sourceColumn: CsvColumnRef | null;
|
|
159
|
+
readonly targetColumn: CsvColumnRef | null;
|
|
160
|
+
/** The direction column reference; null for none; undefined for the Gephi rule. */
|
|
161
|
+
readonly typeColumn: CsvColumnRef | null | undefined;
|
|
162
|
+
readonly idColumn: CsvColumnRef | null;
|
|
163
|
+
readonly nodes: ImportInput | null;
|
|
164
|
+
}
|
|
165
|
+
|
|
166
|
+
/** The columns of an edge table, by index. */
|
|
167
|
+
interface EdgePlan {
|
|
168
|
+
readonly kind: "edges";
|
|
169
|
+
readonly names: readonly string[];
|
|
170
|
+
readonly width: number;
|
|
171
|
+
readonly source: number;
|
|
172
|
+
readonly target: number;
|
|
173
|
+
readonly weight: number;
|
|
174
|
+
readonly type: number;
|
|
175
|
+
readonly id: number;
|
|
176
|
+
readonly label: number;
|
|
177
|
+
readonly attributes: readonly number[];
|
|
178
|
+
}
|
|
179
|
+
|
|
180
|
+
/** The columns of a node table, by index. */
|
|
181
|
+
interface NodePlan {
|
|
182
|
+
readonly kind: "nodes";
|
|
183
|
+
readonly names: readonly string[];
|
|
184
|
+
readonly width: number;
|
|
185
|
+
/** The column ids are read from, or -1 under nodeIdFrom "index". */
|
|
186
|
+
readonly id: number;
|
|
187
|
+
readonly label: number;
|
|
188
|
+
readonly attributes: readonly number[];
|
|
189
|
+
}
|
|
190
|
+
|
|
191
|
+
/** Everything one import call shares between its tables. */
|
|
192
|
+
interface ImportState {
|
|
193
|
+
readonly sink: GraphSink;
|
|
194
|
+
readonly report: ImportReportBuilder;
|
|
195
|
+
readonly common: ResolvedImportOptions;
|
|
196
|
+
readonly csv: ResolvedCsvOptions;
|
|
197
|
+
readonly weightFromExplicit: boolean;
|
|
198
|
+
readonly coercer: IdCoercer;
|
|
199
|
+
readonly resolver: DirectionResolver;
|
|
200
|
+
headerSet: boolean;
|
|
201
|
+
/**
|
|
202
|
+
* The file-level direction a SNAP (`# Directed graph` / `# Undirected graph`) or KONECT
|
|
203
|
+
* (`% asym` / `% sym` / `% bip`) comment header declares; null when the file declares none
|
|
204
|
+
* (the `defaultDirected` option applies).
|
|
205
|
+
*/
|
|
206
|
+
commentDirected: boolean | null;
|
|
207
|
+
}
|
|
208
|
+
|
|
209
|
+
/** The comment characters of the SNAP (`#`) and KONECT (`%`) headers (research note 07 section 2.6). */
|
|
210
|
+
const COMMENT_CHARS: readonly string[] = Object.freeze(["#", "%"]);
|
|
211
|
+
|
|
212
|
+
/**
|
|
213
|
+
* The direction a SNAP or KONECT comment header declares (research note 07 section 2.6): SNAP
|
|
214
|
+
* pages write `# Directed graph` / `# Undirected graph`, KONECT's first line is `% sym` (undirected),
|
|
215
|
+
* `% asym` (directed) or `% bip` (bipartite, undirected).
|
|
216
|
+
* @param comments - the leading comment lines
|
|
217
|
+
* @returns true / false when a line declares the direction, null otherwise
|
|
218
|
+
*/
|
|
219
|
+
function commentDirection(comments: readonly string[]): boolean | null {
|
|
220
|
+
for (const comment of comments) {
|
|
221
|
+
const text = comment.slice(1).trim().toLowerCase();
|
|
222
|
+
if (text.startsWith("directed graph") || text.startsWith("asym")) {
|
|
223
|
+
return true;
|
|
224
|
+
}
|
|
225
|
+
if (text.startsWith("undirected graph") || text.startsWith("sym") || text.startsWith("bip")) {
|
|
226
|
+
return false;
|
|
227
|
+
}
|
|
228
|
+
}
|
|
229
|
+
return null;
|
|
230
|
+
}
|
|
231
|
+
|
|
232
|
+
/**
|
|
233
|
+
* Apply the defaults of the CSV options and check every value.
|
|
234
|
+
* @param options - the caller's options
|
|
235
|
+
* @returns the resolved options; E_UNSUPPORTED for a value outside its set
|
|
236
|
+
*/
|
|
237
|
+
function resolveCsvOptions(options: (CsvImportOptions & CommonImportOptions) | undefined): ResolvedCsvOptions {
|
|
238
|
+
const o: CsvImportOptions = options ?? {};
|
|
239
|
+
if (o.delimiter !== undefined && (typeof o.delimiter !== "string" || o.delimiter.length === 0)) {
|
|
240
|
+
throw new GraphFormatError("E_UNSUPPORTED", "option delimiter: expected a non-empty string", {
|
|
241
|
+
option: "delimiter",
|
|
242
|
+
found: o.delimiter,
|
|
243
|
+
});
|
|
244
|
+
}
|
|
245
|
+
if (o.delimiter !== undefined && BAD_DELIMITERS.has(o.delimiter)) {
|
|
246
|
+
throw new GraphFormatError("E_UNSUPPORTED", "option delimiter: a quote or a line break cannot delimit", {
|
|
247
|
+
option: "delimiter",
|
|
248
|
+
found: o.delimiter,
|
|
249
|
+
});
|
|
250
|
+
}
|
|
251
|
+
if (o.header !== undefined && o.header !== "auto" && typeof o.header !== "boolean") {
|
|
252
|
+
throw new GraphFormatError("E_UNSUPPORTED", 'option header: expected true, false or "auto"', {
|
|
253
|
+
option: "header",
|
|
254
|
+
found: o.header,
|
|
255
|
+
});
|
|
256
|
+
}
|
|
257
|
+
if (o.table !== undefined && !TABLE_MODES.has(o.table)) {
|
|
258
|
+
throw new GraphFormatError("E_UNSUPPORTED", 'option table: expected "edges", "nodes" or "auto"', {
|
|
259
|
+
option: "table",
|
|
260
|
+
found: o.table,
|
|
261
|
+
});
|
|
262
|
+
}
|
|
263
|
+
for (const name of ["sourceColumn", "targetColumn", "idColumn"] as const) {
|
|
264
|
+
checkColumnRef(name, o[name]);
|
|
265
|
+
}
|
|
266
|
+
if (o.typeColumn !== null) {
|
|
267
|
+
checkColumnRef("typeColumn", o.typeColumn);
|
|
268
|
+
}
|
|
269
|
+
return {
|
|
270
|
+
delimiter: o.delimiter ?? null,
|
|
271
|
+
header: o.header ?? "auto",
|
|
272
|
+
table: o.table ?? "auto",
|
|
273
|
+
sourceColumn: o.sourceColumn ?? null,
|
|
274
|
+
targetColumn: o.targetColumn ?? null,
|
|
275
|
+
typeColumn: o.typeColumn,
|
|
276
|
+
idColumn: o.idColumn ?? null,
|
|
277
|
+
nodes: o.nodes ?? null,
|
|
278
|
+
};
|
|
279
|
+
}
|
|
280
|
+
|
|
281
|
+
/**
|
|
282
|
+
* Check a column reference option: a string name or a non-negative integer position.
|
|
283
|
+
* @param name - the option name
|
|
284
|
+
* @param value - the value
|
|
285
|
+
*/
|
|
286
|
+
function checkColumnRef(name: string, value: unknown): void {
|
|
287
|
+
if (value === undefined) {
|
|
288
|
+
return;
|
|
289
|
+
}
|
|
290
|
+
if (typeof value === "string" && value.length > 0) {
|
|
291
|
+
return;
|
|
292
|
+
}
|
|
293
|
+
if (typeof value === "number" && Number.isInteger(value) && value >= 0) {
|
|
294
|
+
return;
|
|
295
|
+
}
|
|
296
|
+
throw new GraphFormatError("E_UNSUPPORTED", `option ${name}: expected a column name or a 0-based position`, {
|
|
297
|
+
option: name,
|
|
298
|
+
found: value,
|
|
299
|
+
});
|
|
300
|
+
}
|
|
301
|
+
|
|
302
|
+
/**
|
|
303
|
+
* Whether a cell is empty (nothing or whitespace only): an unset value, a missing endpoint.
|
|
304
|
+
* @param text - the cell text
|
|
305
|
+
* @returns true when blank
|
|
306
|
+
*/
|
|
307
|
+
function isBlank(text: string): boolean {
|
|
308
|
+
return text.length === 0 || text.trim().length === 0;
|
|
309
|
+
}
|
|
310
|
+
|
|
311
|
+
/**
|
|
312
|
+
* Whether a cell is unset: blank and not quoted. A quoted blank cell (`""`, `" "`) is a set value
|
|
313
|
+
* (the empty string is a legal id and a legal value, design section 4.1); an unquoted one is
|
|
314
|
+
* nothing at all.
|
|
315
|
+
* @param text - the cell text
|
|
316
|
+
* @param quoted - whether the cell was quoted
|
|
317
|
+
* @returns true when the cell carries no value
|
|
318
|
+
*/
|
|
319
|
+
function isUnset(text: string, quoted: boolean | undefined): boolean {
|
|
320
|
+
return quoted !== true && isBlank(text);
|
|
321
|
+
}
|
|
322
|
+
|
|
323
|
+
/** Rows between two checks of the cancellation signal on an in-memory input. */
|
|
324
|
+
const ABORT_CHECK_INTERVAL = 64;
|
|
325
|
+
|
|
326
|
+
/**
|
|
327
|
+
* Header names made unique within the table: a repeated name becomes `<name>#<position>` (1-based,
|
|
328
|
+
* the CSV analogue of the `#<origin.id>` rule of design section 5.6) and is reported.
|
|
329
|
+
* @param names - the header names
|
|
330
|
+
* @param report - the report
|
|
331
|
+
* @param line - the header line
|
|
332
|
+
* @returns the unique names
|
|
333
|
+
*/
|
|
334
|
+
function uniqueNames(names: readonly string[], report: ImportReportBuilder, line: number): string[] {
|
|
335
|
+
const seen = new Set<string>();
|
|
336
|
+
return names.map((name, i) => {
|
|
337
|
+
const unique = uniqueColumnName(name, String(i + 1), (n) => seen.has(n));
|
|
338
|
+
seen.add(unique);
|
|
339
|
+
if (unique !== name) {
|
|
340
|
+
report.warning(
|
|
341
|
+
"coercion",
|
|
342
|
+
RENAMED_CODE,
|
|
343
|
+
`column ${i + 1} "${name}" renamed to "${unique}": the header repeats the name`,
|
|
344
|
+
{ line, element: name },
|
|
345
|
+
);
|
|
346
|
+
}
|
|
347
|
+
return unique;
|
|
348
|
+
});
|
|
349
|
+
}
|
|
350
|
+
|
|
351
|
+
/**
|
|
352
|
+
* Find a column by candidates, skipping indices already claimed by another role.
|
|
353
|
+
* @param names - the header names
|
|
354
|
+
* @param candidates - the names to look for
|
|
355
|
+
* @param claimed - indices taken
|
|
356
|
+
* @returns the index, or -1
|
|
357
|
+
*/
|
|
358
|
+
function findFree(names: readonly string[], candidates: readonly string[], claimed: ReadonlySet<number>): number {
|
|
359
|
+
const masked = names.map((name, i) => (claimed.has(i) ? "" : name));
|
|
360
|
+
return findColumn(masked, candidates);
|
|
361
|
+
}
|
|
362
|
+
|
|
363
|
+
/**
|
|
364
|
+
* Declare a role column (edge id, label) on the sink through the shared design section 5.6 rule:
|
|
365
|
+
* a name already declared differently is renamed `<name>#<position>` (reported); a role already
|
|
366
|
+
* held by another column is dropped from this declaration (reported), the values are kept under
|
|
367
|
+
* the name.
|
|
368
|
+
* @param sink - the sink
|
|
369
|
+
* @param domain - node or edge
|
|
370
|
+
* @param decl - the declaration
|
|
371
|
+
* @param position - the 1-based column position, for the rename
|
|
372
|
+
* @param report - the report
|
|
373
|
+
* @param line - the header line
|
|
374
|
+
* @returns the handle
|
|
375
|
+
*/
|
|
376
|
+
function declareRoleColumn(
|
|
377
|
+
sink: GraphSink,
|
|
378
|
+
domain: "node" | "edge",
|
|
379
|
+
decl: ColumnDecl,
|
|
380
|
+
position: number,
|
|
381
|
+
report: ImportReportBuilder,
|
|
382
|
+
line: number,
|
|
383
|
+
): ColumnHandle {
|
|
384
|
+
const withOrigin: ColumnDecl = { ...decl, origin: { ...decl.origin, id: String(position) } };
|
|
385
|
+
const resolved = declareResolved(sink, domain, withOrigin, report, { line, element: decl.name });
|
|
386
|
+
if (resolved.roleDropped) {
|
|
387
|
+
// a unique constraint belongs to the id role; without the role the column is plain text
|
|
388
|
+
return resolved.handle;
|
|
389
|
+
}
|
|
390
|
+
return resolved.handle;
|
|
391
|
+
}
|
|
392
|
+
|
|
393
|
+
/**
|
|
394
|
+
* Parse a Gephi Type cell.
|
|
395
|
+
* @param text - the cell text
|
|
396
|
+
* @returns the kind; undefined for a blank cell (the default applies); null for an unknown word
|
|
397
|
+
*/
|
|
398
|
+
function parseKind(text: string): EdgeKind | null | undefined {
|
|
399
|
+
if (isBlank(text)) {
|
|
400
|
+
return undefined;
|
|
401
|
+
}
|
|
402
|
+
switch (text.trim().toLowerCase()) {
|
|
403
|
+
case "directed":
|
|
404
|
+
return "directed";
|
|
405
|
+
case "undirected":
|
|
406
|
+
return "undirected";
|
|
407
|
+
case "mutual":
|
|
408
|
+
return "mutual";
|
|
409
|
+
default:
|
|
410
|
+
return null;
|
|
411
|
+
}
|
|
412
|
+
}
|
|
413
|
+
|
|
414
|
+
/**
|
|
415
|
+
* Read one CSV table into the sink.
|
|
416
|
+
*/
|
|
417
|
+
class TableReader {
|
|
418
|
+
private readonly state: ImportState;
|
|
419
|
+
|
|
420
|
+
private readonly reader: CsvRecordReader;
|
|
421
|
+
|
|
422
|
+
private readonly kind: "edges" | "nodes" | "auto";
|
|
423
|
+
|
|
424
|
+
private plan: EdgePlan | NodePlan | null = null;
|
|
425
|
+
|
|
426
|
+
private writers: (InferredColumn | null)[] = [];
|
|
427
|
+
|
|
428
|
+
private idHandle: ColumnHandle = INVALID_INDEX as ColumnHandle;
|
|
429
|
+
|
|
430
|
+
private labelHandle: ColumnHandle = INVALID_INDEX as ColumnHandle;
|
|
431
|
+
|
|
432
|
+
private dataRows = 0;
|
|
433
|
+
|
|
434
|
+
private nodeOrdinal = 0;
|
|
435
|
+
|
|
436
|
+
/** The edge ids seen so far (the id column is unique; a repeat is skipped with an issue). */
|
|
437
|
+
private readonly edgeIds = new Set<string>();
|
|
438
|
+
|
|
439
|
+
private readonly where: { line: number | null; element: string | null } = { line: null, element: null };
|
|
440
|
+
|
|
441
|
+
/**
|
|
442
|
+
* Create a table reader.
|
|
443
|
+
* @param state - the import state
|
|
444
|
+
* @param input - the table's input
|
|
445
|
+
* @param kind - what the table is, or "auto"
|
|
446
|
+
* @param progress - whether this table reports byte progress
|
|
447
|
+
*/
|
|
448
|
+
constructor(state: ImportState, input: ImportInput, kind: "edges" | "nodes" | "auto", progress: boolean) {
|
|
449
|
+
this.state = state;
|
|
450
|
+
this.kind = kind;
|
|
451
|
+
const readerOptions: CsvReaderOptions = {
|
|
452
|
+
delimiter: state.csv.delimiter,
|
|
453
|
+
comments: COMMENT_CHARS,
|
|
454
|
+
signal: state.common.signal,
|
|
455
|
+
onProgress: progress ? state.common.onProgress : null,
|
|
456
|
+
};
|
|
457
|
+
this.reader = new CsvRecordReader(input, state.report, readerOptions);
|
|
458
|
+
}
|
|
459
|
+
|
|
460
|
+
/**
|
|
461
|
+
* Read every row; the reader is closed (and a stream cancelled) when the import aborts midway.
|
|
462
|
+
*/
|
|
463
|
+
async read(): Promise<void> {
|
|
464
|
+
const iterator = this.reader[Symbol.asyncIterator]();
|
|
465
|
+
try {
|
|
466
|
+
await this.readRows(iterator);
|
|
467
|
+
} finally {
|
|
468
|
+
await iterator.return(undefined);
|
|
469
|
+
}
|
|
470
|
+
}
|
|
471
|
+
|
|
472
|
+
/**
|
|
473
|
+
* Read the header (or decide there is none), resolve the plan, then push every row.
|
|
474
|
+
* @param iterator - the record iterator
|
|
475
|
+
*/
|
|
476
|
+
private async readRows(iterator: AsyncGenerator<string[], void, undefined>): Promise<void> {
|
|
477
|
+
const { report } = this.state;
|
|
478
|
+
const first = await iterator.next();
|
|
479
|
+
const firstRow: string[] = first.done
|
|
480
|
+
? report.fail(EMPTY_INPUT_CODE, "the input is empty: no header row and no records")
|
|
481
|
+
: first.value;
|
|
482
|
+
const firstLine = this.reader.line;
|
|
483
|
+
const firstQuoted = this.reader.quoted.slice(0, firstRow.length);
|
|
484
|
+
const pending: { row: string[]; quoted: readonly boolean[]; line: number }[] = [];
|
|
485
|
+
let header: boolean;
|
|
486
|
+
const { header: mode } = this.state.csv;
|
|
487
|
+
if (mode === "auto") {
|
|
488
|
+
const second = await iterator.next();
|
|
489
|
+
const secondRow: string[] | null = second.done ? null : second.value;
|
|
490
|
+
const secondLine = this.reader.line;
|
|
491
|
+
const secondQuoted = this.reader.quoted.slice(0, secondRow?.length ?? 0);
|
|
492
|
+
header = looksLikeHeader(firstRow, secondRow);
|
|
493
|
+
if (!header) {
|
|
494
|
+
pending.push({ row: firstRow, quoted: firstQuoted, line: firstLine });
|
|
495
|
+
}
|
|
496
|
+
if (secondRow !== null) {
|
|
497
|
+
pending.push({ row: secondRow, quoted: secondQuoted, line: secondLine });
|
|
498
|
+
}
|
|
499
|
+
} else {
|
|
500
|
+
header = mode;
|
|
501
|
+
if (!header) {
|
|
502
|
+
pending.push({ row: firstRow, quoted: firstQuoted, line: firstLine });
|
|
503
|
+
}
|
|
504
|
+
}
|
|
505
|
+
const names = header ? uniqueNames(headerNames(firstRow), report, firstLine) : positionalNames(firstRow.length);
|
|
506
|
+
if (this.kind !== "nodes" && this.state.commentDirected === null) {
|
|
507
|
+
this.state.commentDirected = commentDirection(this.reader.leadingComments);
|
|
508
|
+
}
|
|
509
|
+
this.plan = this.resolvePlan(names, header, firstLine);
|
|
510
|
+
this.prepareColumns(firstLine);
|
|
511
|
+
for (const { row, quoted, line } of pending) {
|
|
512
|
+
this.processRow(row, quoted, line);
|
|
513
|
+
}
|
|
514
|
+
const { signal } = this.state.common;
|
|
515
|
+
let sinceCheck = 0;
|
|
516
|
+
for (;;) {
|
|
517
|
+
const next = await iterator.next();
|
|
518
|
+
if (next.done) {
|
|
519
|
+
break;
|
|
520
|
+
}
|
|
521
|
+
this.processRow(next.value, this.reader.quoted, this.reader.line);
|
|
522
|
+
if (++sinceCheck >= ABORT_CHECK_INTERVAL) {
|
|
523
|
+
sinceCheck = 0;
|
|
524
|
+
throwIfAborted(signal);
|
|
525
|
+
}
|
|
526
|
+
}
|
|
527
|
+
for (const writer of this.writers) {
|
|
528
|
+
writer?.finish();
|
|
529
|
+
}
|
|
530
|
+
if (this.dataRows === 0 && header) {
|
|
531
|
+
report.warning("missing-value", NO_DATA_ROWS_CODE, "the table has a header and no data rows", {
|
|
532
|
+
line: firstLine,
|
|
533
|
+
});
|
|
534
|
+
}
|
|
535
|
+
}
|
|
536
|
+
|
|
537
|
+
/**
|
|
538
|
+
* Decide the table's columns from its header.
|
|
539
|
+
* @param names - the column names
|
|
540
|
+
* @param header - whether the file has a header row
|
|
541
|
+
* @param line - the header line
|
|
542
|
+
* @returns the plan; the import aborts when no endpoints (or id) resolve
|
|
543
|
+
*/
|
|
544
|
+
private resolvePlan(names: readonly string[], header: boolean, line: number): EdgePlan | NodePlan {
|
|
545
|
+
const { csv, report } = this.state;
|
|
546
|
+
const width = names.length;
|
|
547
|
+
let source = -1;
|
|
548
|
+
let target = -1;
|
|
549
|
+
if (this.kind !== "nodes") {
|
|
550
|
+
if (csv.sourceColumn !== null) {
|
|
551
|
+
source = resolveColumnRef(names, csv.sourceColumn, "sourceColumn");
|
|
552
|
+
} else if (header) {
|
|
553
|
+
source = findColumn(names, SOURCE_NAMES);
|
|
554
|
+
} else {
|
|
555
|
+
source = width >= 2 ? 0 : -1;
|
|
556
|
+
}
|
|
557
|
+
if (csv.targetColumn !== null) {
|
|
558
|
+
target = resolveColumnRef(names, csv.targetColumn, "targetColumn");
|
|
559
|
+
} else if (header) {
|
|
560
|
+
target = findColumn(names, TARGET_NAMES);
|
|
561
|
+
} else {
|
|
562
|
+
target = width >= 2 ? 1 : -1;
|
|
563
|
+
}
|
|
564
|
+
if (source >= 0 && target >= 0 && source === target) {
|
|
565
|
+
throw new GraphFormatError("E_UNSUPPORTED", "sourceColumn and targetColumn name the same column", {
|
|
566
|
+
option: "targetColumn",
|
|
567
|
+
found: names[target],
|
|
568
|
+
});
|
|
569
|
+
}
|
|
570
|
+
}
|
|
571
|
+
if (source >= 0 && target >= 0) {
|
|
572
|
+
return this.edgePlan(names, header, source, target);
|
|
573
|
+
}
|
|
574
|
+
const shown = names.map((n) => JSON.stringify(n)).join(", ");
|
|
575
|
+
if (
|
|
576
|
+
this.kind === "edges" ||
|
|
577
|
+
(this.kind === "auto" && (csv.sourceColumn !== null || csv.targetColumn !== null))
|
|
578
|
+
) {
|
|
579
|
+
report.fail(
|
|
580
|
+
NO_ENDPOINT_COLUMNS_CODE,
|
|
581
|
+
`no source / target columns in the header (${shown}); a node table goes in the nodes option`,
|
|
582
|
+
{ line },
|
|
583
|
+
{ columns: [...names] },
|
|
584
|
+
);
|
|
585
|
+
}
|
|
586
|
+
const idResolves = header ? findColumn(names, ID_NAMES) >= 0 : width >= 1;
|
|
587
|
+
if (this.kind === "auto" && csv.idColumn === null && !idResolves) {
|
|
588
|
+
report.fail(
|
|
589
|
+
NO_ENDPOINT_COLUMNS_CODE,
|
|
590
|
+
`no source / target columns and no id column in the header (${shown}); the input is neither an edge table nor a node table`,
|
|
591
|
+
{ line },
|
|
592
|
+
{ columns: [...names] },
|
|
593
|
+
);
|
|
594
|
+
}
|
|
595
|
+
return this.nodePlan(names, header, line);
|
|
596
|
+
}
|
|
597
|
+
|
|
598
|
+
/**
|
|
599
|
+
* The columns of an edge table.
|
|
600
|
+
* @param names - the column names
|
|
601
|
+
* @param header - whether the file has a header row
|
|
602
|
+
* @param source - the source column
|
|
603
|
+
* @param target - the target column
|
|
604
|
+
* @returns the plan
|
|
605
|
+
*/
|
|
606
|
+
private edgePlan(names: readonly string[], header: boolean, source: number, target: number): EdgePlan {
|
|
607
|
+
const { csv, common, report } = this.state;
|
|
608
|
+
const claimed = new Set<number>([source, target]);
|
|
609
|
+
let weight = -1;
|
|
610
|
+
if (common.weightFrom !== null) {
|
|
611
|
+
if (header) {
|
|
612
|
+
weight = findFree(names, [common.weightFrom], claimed);
|
|
613
|
+
if (weight < 0 && this.state.weightFromExplicit) {
|
|
614
|
+
report.warning(
|
|
615
|
+
"missing-value",
|
|
616
|
+
COLUMN_MISSING_CODE,
|
|
617
|
+
`weight column ${JSON.stringify(common.weightFrom)} not found; edges are unweighted`,
|
|
618
|
+
{ line: this.reader.line, element: common.weightFrom },
|
|
619
|
+
);
|
|
620
|
+
}
|
|
621
|
+
} else if (names.length >= 3 && !claimed.has(2)) {
|
|
622
|
+
weight = 2;
|
|
623
|
+
}
|
|
624
|
+
}
|
|
625
|
+
if (weight >= 0) {
|
|
626
|
+
claimed.add(weight);
|
|
627
|
+
}
|
|
628
|
+
let type = -1;
|
|
629
|
+
if (csv.typeColumn === undefined) {
|
|
630
|
+
if (header && names[source] === "Source" && names[target] === "Target") {
|
|
631
|
+
type = findFree(names, [TYPE_NAME], claimed);
|
|
632
|
+
if (type >= 0 && names[type] !== TYPE_NAME) {
|
|
633
|
+
type = -1;
|
|
634
|
+
}
|
|
635
|
+
}
|
|
636
|
+
} else if (csv.typeColumn !== null) {
|
|
637
|
+
type = resolveColumnRef(names, csv.typeColumn, "typeColumn");
|
|
638
|
+
if (claimed.has(type)) {
|
|
639
|
+
throw new GraphFormatError("E_UNSUPPORTED", "typeColumn names an endpoint or weight column", {
|
|
640
|
+
option: "typeColumn",
|
|
641
|
+
found: names[type],
|
|
642
|
+
});
|
|
643
|
+
}
|
|
644
|
+
}
|
|
645
|
+
if (type >= 0) {
|
|
646
|
+
claimed.add(type);
|
|
647
|
+
}
|
|
648
|
+
const id = header ? findFree(names, EDGE_ID_NAMES, claimed) : -1;
|
|
649
|
+
if (id >= 0) {
|
|
650
|
+
claimed.add(id);
|
|
651
|
+
}
|
|
652
|
+
const label = header ? findFree(names, LABEL_NAMES, claimed) : -1;
|
|
653
|
+
if (label >= 0) {
|
|
654
|
+
claimed.add(label);
|
|
655
|
+
}
|
|
656
|
+
const attributes: number[] = [];
|
|
657
|
+
for (let i = 0; i < names.length; i++) {
|
|
658
|
+
if (!claimed.has(i)) {
|
|
659
|
+
attributes.push(i);
|
|
660
|
+
}
|
|
661
|
+
}
|
|
662
|
+
return { kind: "edges", names, width: names.length, source, target, weight, type, id, label, attributes };
|
|
663
|
+
}
|
|
664
|
+
|
|
665
|
+
/**
|
|
666
|
+
* The columns of a node table.
|
|
667
|
+
* @param names - the column names
|
|
668
|
+
* @param header - whether the file has a header row
|
|
669
|
+
* @param line - the header line
|
|
670
|
+
* @returns the plan; the import aborts when no id column resolves
|
|
671
|
+
*/
|
|
672
|
+
private nodePlan(names: readonly string[], header: boolean, line: number): NodePlan {
|
|
673
|
+
const { csv, common, report } = this.state;
|
|
674
|
+
const claimed = new Set<number>();
|
|
675
|
+
let idColumn = -1;
|
|
676
|
+
if (csv.idColumn !== null) {
|
|
677
|
+
idColumn = resolveColumnRef(names, csv.idColumn, "idColumn");
|
|
678
|
+
} else if (header) {
|
|
679
|
+
idColumn = findColumn(names, ID_NAMES);
|
|
680
|
+
} else if (names.length >= 1) {
|
|
681
|
+
idColumn = 0;
|
|
682
|
+
}
|
|
683
|
+
const label = header ? findFree(names, LABEL_NAMES, new Set(idColumn >= 0 ? [idColumn] : [])) : -1;
|
|
684
|
+
let id: number;
|
|
685
|
+
switch (common.nodeIdFrom) {
|
|
686
|
+
case "label":
|
|
687
|
+
if (label < 0) {
|
|
688
|
+
report.fail(NO_ID_COLUMN_CODE, 'nodeIdFrom is "label" but the node table has no label column', {
|
|
689
|
+
line,
|
|
690
|
+
});
|
|
691
|
+
}
|
|
692
|
+
id = label;
|
|
693
|
+
break;
|
|
694
|
+
case "index":
|
|
695
|
+
id = -1;
|
|
696
|
+
break;
|
|
697
|
+
default:
|
|
698
|
+
if (idColumn < 0) {
|
|
699
|
+
report.fail(
|
|
700
|
+
NO_ID_COLUMN_CODE,
|
|
701
|
+
`no id column in the node table header (${names.map((n) => JSON.stringify(n)).join(", ")})`,
|
|
702
|
+
{ line },
|
|
703
|
+
{ columns: [...names] },
|
|
704
|
+
);
|
|
705
|
+
}
|
|
706
|
+
id = idColumn;
|
|
707
|
+
claimed.add(idColumn);
|
|
708
|
+
break;
|
|
709
|
+
}
|
|
710
|
+
if (label >= 0) {
|
|
711
|
+
claimed.add(label);
|
|
712
|
+
}
|
|
713
|
+
const attributes: number[] = [];
|
|
714
|
+
for (let i = 0; i < names.length; i++) {
|
|
715
|
+
if (!claimed.has(i)) {
|
|
716
|
+
attributes.push(i);
|
|
717
|
+
}
|
|
718
|
+
}
|
|
719
|
+
return { kind: "nodes", names, width: names.length, id, label, attributes };
|
|
720
|
+
}
|
|
721
|
+
|
|
722
|
+
/**
|
|
723
|
+
* Declare the role columns and create a writer per attribute column.
|
|
724
|
+
* @param line - the header line
|
|
725
|
+
*/
|
|
726
|
+
private prepareColumns(line: number): void {
|
|
727
|
+
const plan = this.requirePlan();
|
|
728
|
+
const { sink, report } = this.state;
|
|
729
|
+
const domain = plan.kind === "edges" ? "edge" : "node";
|
|
730
|
+
const origin = { format: "csv" };
|
|
731
|
+
if (plan.kind === "edges" && plan.id >= 0) {
|
|
732
|
+
this.idHandle = declareRoleColumn(
|
|
733
|
+
sink,
|
|
734
|
+
"edge",
|
|
735
|
+
{ name: plan.names[plan.id], dtype: "string", nullable: true, role: "id", unique: true, origin },
|
|
736
|
+
plan.id + 1,
|
|
737
|
+
report,
|
|
738
|
+
line,
|
|
739
|
+
);
|
|
740
|
+
}
|
|
741
|
+
if (plan.label >= 0) {
|
|
742
|
+
this.labelHandle = declareRoleColumn(
|
|
743
|
+
sink,
|
|
744
|
+
domain,
|
|
745
|
+
{ name: plan.names[plan.label], dtype: "string", nullable: true, role: "label", origin },
|
|
746
|
+
plan.label + 1,
|
|
747
|
+
report,
|
|
748
|
+
line,
|
|
749
|
+
);
|
|
750
|
+
}
|
|
751
|
+
this.writers = plan.names.map(() => null);
|
|
752
|
+
for (const index of plan.attributes) {
|
|
753
|
+
this.writers[index] = new InferredColumn(plan.names[index], domain, sink, report);
|
|
754
|
+
}
|
|
755
|
+
}
|
|
756
|
+
|
|
757
|
+
/**
|
|
758
|
+
* The plan, which exists once the header was read.
|
|
759
|
+
* @returns the plan
|
|
760
|
+
*/
|
|
761
|
+
private requirePlan(): EdgePlan | NodePlan {
|
|
762
|
+
if (this.plan === null) {
|
|
763
|
+
throw new GraphFormatError("E_UNSUPPORTED", "the header has not been read", { reason: "no plan" });
|
|
764
|
+
}
|
|
765
|
+
return this.plan;
|
|
766
|
+
}
|
|
767
|
+
|
|
768
|
+
/**
|
|
769
|
+
* Push one data row.
|
|
770
|
+
* @param row - the cells
|
|
771
|
+
* @param quoted - whether each cell was quoted (a quoted empty cell is the empty string)
|
|
772
|
+
* @param line - the row's line
|
|
773
|
+
*/
|
|
774
|
+
private processRow(row: string[], quoted: readonly boolean[], line: number): void {
|
|
775
|
+
const plan = this.requirePlan();
|
|
776
|
+
this.dataRows++;
|
|
777
|
+
if (plan.kind === "edges") {
|
|
778
|
+
this.processEdgeRow(plan, row, quoted, line);
|
|
779
|
+
} else {
|
|
780
|
+
this.processNodeRow(plan, row, quoted, line);
|
|
781
|
+
}
|
|
782
|
+
}
|
|
783
|
+
|
|
784
|
+
/**
|
|
785
|
+
* Push one edge row: endpoints, weight, direction, then the attribute cells.
|
|
786
|
+
* @param plan - the edge plan
|
|
787
|
+
* @param row - the cells
|
|
788
|
+
* @param quoted - whether each cell was quoted
|
|
789
|
+
* @param line - the row's line
|
|
790
|
+
*/
|
|
791
|
+
private processEdgeRow(plan: EdgePlan, row: string[], quoted: readonly boolean[], line: number): void {
|
|
792
|
+
const { report, sink, resolver, common } = this.state;
|
|
793
|
+
const { counts } = report;
|
|
794
|
+
if (row.length !== plan.width) {
|
|
795
|
+
report.error(
|
|
796
|
+
"validation-error",
|
|
797
|
+
FIELD_COUNT_CODE,
|
|
798
|
+
`line ${line}: ${row.length} field(s), the header has ${plan.width}`,
|
|
799
|
+
{ line },
|
|
800
|
+
);
|
|
801
|
+
counts.skippedEdges++;
|
|
802
|
+
return;
|
|
803
|
+
}
|
|
804
|
+
const sourceText = row[plan.source];
|
|
805
|
+
const targetText = row[plan.target];
|
|
806
|
+
const sourceMissing = isUnset(sourceText, quoted[plan.source]);
|
|
807
|
+
if (sourceMissing || isUnset(targetText, quoted[plan.target])) {
|
|
808
|
+
report.error(
|
|
809
|
+
"missing-value",
|
|
810
|
+
MISSING_ENDPOINT_CODE,
|
|
811
|
+
`line ${line}: blank ${sourceMissing ? "source" : "target"} cell`,
|
|
812
|
+
{ line },
|
|
813
|
+
);
|
|
814
|
+
counts.skippedEdges++;
|
|
815
|
+
return;
|
|
816
|
+
}
|
|
817
|
+
let kind: EdgeKind = (this.state.commentDirected ?? common.defaultDirected) ? "directed" : "undirected";
|
|
818
|
+
if (plan.type >= 0) {
|
|
819
|
+
const parsed = parseKind(row[plan.type]);
|
|
820
|
+
if (parsed === null) {
|
|
821
|
+
report.error(
|
|
822
|
+
"validation-error",
|
|
823
|
+
BAD_TYPE_CODE,
|
|
824
|
+
`line ${line}: Type ${JSON.stringify(row[plan.type])} is not Directed, Undirected or Mutual`,
|
|
825
|
+
{ line },
|
|
826
|
+
);
|
|
827
|
+
counts.skippedEdges++;
|
|
828
|
+
return;
|
|
829
|
+
}
|
|
830
|
+
if (parsed !== undefined) {
|
|
831
|
+
kind = parsed;
|
|
832
|
+
}
|
|
833
|
+
}
|
|
834
|
+
const { where } = this;
|
|
835
|
+
where.line = line;
|
|
836
|
+
where.element = null;
|
|
837
|
+
const idText = plan.id >= 0 && !isUnset(row[plan.id], quoted[plan.id]) ? row[plan.id] : null;
|
|
838
|
+
if (idText !== null) {
|
|
839
|
+
if (this.edgeIds.has(idText)) {
|
|
840
|
+
report.error(
|
|
841
|
+
"validation-error",
|
|
842
|
+
DUPLICATE_EDGE_ID_CODE,
|
|
843
|
+
`line ${line}: edge id ${JSON.stringify(idText)} repeats an earlier row's; the row is skipped`,
|
|
844
|
+
{ line, element: idText },
|
|
845
|
+
);
|
|
846
|
+
counts.skippedEdges++;
|
|
847
|
+
return;
|
|
848
|
+
}
|
|
849
|
+
this.edgeIds.add(idText);
|
|
850
|
+
}
|
|
851
|
+
let edge: number;
|
|
852
|
+
try {
|
|
853
|
+
const source = this.coerce(sourceText);
|
|
854
|
+
const target = this.coerce(targetText);
|
|
855
|
+
const weight = plan.weight >= 0 ? parseWeightText(row[plan.weight]) : undefined;
|
|
856
|
+
if (!this.state.headerSet) {
|
|
857
|
+
this.state.headerSet = true;
|
|
858
|
+
resolver.setHeader(kind !== "undirected", where);
|
|
859
|
+
}
|
|
860
|
+
const sourceNew = sink.indexOf(source) === INVALID_INDEX;
|
|
861
|
+
const targetNew = source !== target && sink.indexOf(target) === INVALID_INDEX;
|
|
862
|
+
const before = sink.edgeCount;
|
|
863
|
+
edge = resolver.addEdge(source, target, kind, weight, where);
|
|
864
|
+
counts.edges += sink.edgeCount - before;
|
|
865
|
+
counts.nodes += (sourceNew ? 1 : 0) + (targetNew ? 1 : 0);
|
|
866
|
+
} catch (err) {
|
|
867
|
+
where.element = idText ?? `${sourceText}->${targetText}`;
|
|
868
|
+
report.recordError(err, where);
|
|
869
|
+
counts.skippedEdges++;
|
|
870
|
+
return;
|
|
871
|
+
}
|
|
872
|
+
if (idText !== null) {
|
|
873
|
+
this.writeRole(this.idHandle, "edge", edge, idText, plan.names[plan.id], line);
|
|
874
|
+
}
|
|
875
|
+
if (plan.label >= 0 && !isUnset(row[plan.label], quoted[plan.label])) {
|
|
876
|
+
this.writeRole(this.labelHandle, "edge", edge, row[plan.label], plan.names[plan.label], line);
|
|
877
|
+
}
|
|
878
|
+
this.writeAttributes(plan, row, quoted, edge, line);
|
|
879
|
+
}
|
|
880
|
+
|
|
881
|
+
/**
|
|
882
|
+
* Push one node row: the id, then the label and attribute cells.
|
|
883
|
+
* @param plan - the node plan
|
|
884
|
+
* @param row - the cells
|
|
885
|
+
* @param quoted - whether each cell was quoted
|
|
886
|
+
* @param line - the row's line
|
|
887
|
+
*/
|
|
888
|
+
private processNodeRow(plan: NodePlan, row: string[], quoted: readonly boolean[], line: number): void {
|
|
889
|
+
const { report, sink } = this.state;
|
|
890
|
+
const { counts } = report;
|
|
891
|
+
const ordinal = this.nodeOrdinal++;
|
|
892
|
+
if (row.length !== plan.width) {
|
|
893
|
+
report.error(
|
|
894
|
+
"validation-error",
|
|
895
|
+
FIELD_COUNT_CODE,
|
|
896
|
+
`line ${line}: ${row.length} field(s), the header has ${plan.width}`,
|
|
897
|
+
{ line },
|
|
898
|
+
);
|
|
899
|
+
counts.skippedNodes++;
|
|
900
|
+
return;
|
|
901
|
+
}
|
|
902
|
+
const idText = plan.id >= 0 ? row[plan.id] : String(ordinal);
|
|
903
|
+
if (plan.id >= 0 && isUnset(idText, quoted[plan.id])) {
|
|
904
|
+
report.error("missing-value", MISSING_ID_CODE, `line ${line}: blank id cell`, { line });
|
|
905
|
+
counts.skippedNodes++;
|
|
906
|
+
return;
|
|
907
|
+
}
|
|
908
|
+
const { where } = this;
|
|
909
|
+
where.line = line;
|
|
910
|
+
where.element = idText;
|
|
911
|
+
let index: number;
|
|
912
|
+
try {
|
|
913
|
+
const id = plan.id >= 0 ? this.coerce(idText) : ordinal;
|
|
914
|
+
if (sink.indexOf(id) !== INVALID_INDEX) {
|
|
915
|
+
report.warning(
|
|
916
|
+
"merged",
|
|
917
|
+
DUPLICATE_NODE_CODE,
|
|
918
|
+
`line ${line}: node ${JSON.stringify(id)} already exists; its attributes are overwritten`,
|
|
919
|
+
where,
|
|
920
|
+
);
|
|
921
|
+
} else {
|
|
922
|
+
counts.nodes++;
|
|
923
|
+
}
|
|
924
|
+
index = sink.addNode(id);
|
|
925
|
+
} catch (err) {
|
|
926
|
+
report.recordError(err, where);
|
|
927
|
+
counts.skippedNodes++;
|
|
928
|
+
return;
|
|
929
|
+
}
|
|
930
|
+
if (plan.label >= 0 && !isUnset(row[plan.label], quoted[plan.label])) {
|
|
931
|
+
this.writeRole(this.labelHandle, "node", index, row[plan.label], plan.names[plan.label], line);
|
|
932
|
+
}
|
|
933
|
+
this.writeAttributes(plan, row, quoted, index, line);
|
|
934
|
+
}
|
|
935
|
+
|
|
936
|
+
/**
|
|
937
|
+
* Coerce an id cell, reporting a merge under `ids: "number"`.
|
|
938
|
+
* @param text - the cell text
|
|
939
|
+
* @returns the id
|
|
940
|
+
*/
|
|
941
|
+
private coerce(text: string): NodeId {
|
|
942
|
+
const id = this.state.coercer.text(text);
|
|
943
|
+
const merge = this.state.coercer.lastMerge;
|
|
944
|
+
if (merge !== null) {
|
|
945
|
+
this.state.report.warnOnce(
|
|
946
|
+
"coercion",
|
|
947
|
+
ID_MERGED_CODE,
|
|
948
|
+
`id ${JSON.stringify(merge.text)} merged with ${JSON.stringify(merge.previousText)} as ${merge.id} under ids: "number"`,
|
|
949
|
+
this.where,
|
|
950
|
+
);
|
|
951
|
+
}
|
|
952
|
+
return id;
|
|
953
|
+
}
|
|
954
|
+
|
|
955
|
+
/**
|
|
956
|
+
* Write a role column cell (edge id, label).
|
|
957
|
+
* @param handle - the column handle
|
|
958
|
+
* @param domain - node or edge
|
|
959
|
+
* @param row - the node or edge index
|
|
960
|
+
* @param text - the cell text
|
|
961
|
+
* @param name - the column name, for issues
|
|
962
|
+
* @param line - the row's line
|
|
963
|
+
*/
|
|
964
|
+
private writeRole(
|
|
965
|
+
handle: ColumnHandle,
|
|
966
|
+
domain: "node" | "edge",
|
|
967
|
+
row: number,
|
|
968
|
+
text: string,
|
|
969
|
+
name: string,
|
|
970
|
+
line: number,
|
|
971
|
+
): void {
|
|
972
|
+
try {
|
|
973
|
+
if (domain === "node") {
|
|
974
|
+
this.state.sink.setNodeValue(handle, row, text);
|
|
975
|
+
} else {
|
|
976
|
+
this.state.sink.setEdgeValue(handle, row, text);
|
|
977
|
+
}
|
|
978
|
+
} catch (err) {
|
|
979
|
+
this.state.report.recordError(err, { line, element: name });
|
|
980
|
+
}
|
|
981
|
+
}
|
|
982
|
+
|
|
983
|
+
/**
|
|
984
|
+
* Write the attribute cells of a row: an unquoted blank cell is unset, a quoted one (`""`,
|
|
985
|
+
* `" "`) is the text it holds.
|
|
986
|
+
* @param plan - the plan
|
|
987
|
+
* @param row - the cells
|
|
988
|
+
* @param quoted - whether each cell was quoted
|
|
989
|
+
* @param index - the node or edge index
|
|
990
|
+
* @param line - the row's line
|
|
991
|
+
*/
|
|
992
|
+
private writeAttributes(
|
|
993
|
+
plan: EdgePlan | NodePlan,
|
|
994
|
+
row: readonly string[],
|
|
995
|
+
quoted: readonly boolean[],
|
|
996
|
+
index: number,
|
|
997
|
+
line: number,
|
|
998
|
+
): void {
|
|
999
|
+
for (const k of plan.attributes) {
|
|
1000
|
+
const text = row[k];
|
|
1001
|
+
if (isUnset(text, quoted[k])) {
|
|
1002
|
+
continue;
|
|
1003
|
+
}
|
|
1004
|
+
const writer = this.writers[k];
|
|
1005
|
+
if (writer === null) {
|
|
1006
|
+
continue;
|
|
1007
|
+
}
|
|
1008
|
+
try {
|
|
1009
|
+
writer.write(index, text);
|
|
1010
|
+
} catch (err) {
|
|
1011
|
+
this.state.report.recordError(err, { line, element: writer.name });
|
|
1012
|
+
}
|
|
1013
|
+
}
|
|
1014
|
+
}
|
|
1015
|
+
}
|
|
1016
|
+
|
|
1017
|
+
const HEAD_BYTES = 4096;
|
|
1018
|
+
const OTHER_FORMAT = /^\s*(<|[[{]|(strict\s+)?(di)?graph(\s+\S+)?\s*\{|\*vertices|creator\b|graph\s*\[)/i;
|
|
1019
|
+
|
|
1020
|
+
/**
|
|
1021
|
+
* Sniff confidence for the registry: 0 for XML, JSON, GML, DOT and Pajek openings; otherwise a
|
|
1022
|
+
* delimited first row with endpoint headers is 0.9, with an id header 0.6, any consistently
|
|
1023
|
+
* delimited rows 0.3, a single column 0.
|
|
1024
|
+
* @param head - the first bytes of the input
|
|
1025
|
+
* @returns a confidence in 0..1
|
|
1026
|
+
*/
|
|
1027
|
+
function sniff(head: Uint8Array): number {
|
|
1028
|
+
const text = new TextDecoder("utf-8").decode(head.subarray(0, HEAD_BYTES));
|
|
1029
|
+
const body = text.startsWith(String.fromCharCode(0xfeff)) ? text.slice(1) : text;
|
|
1030
|
+
if (body.trim().length === 0 || OTHER_FORMAT.test(body)) {
|
|
1031
|
+
return 0;
|
|
1032
|
+
}
|
|
1033
|
+
const newline = sniffNewline(body);
|
|
1034
|
+
const delimiter = sniffDelimiter(body, newline);
|
|
1035
|
+
if (delimiter === null) {
|
|
1036
|
+
return 0;
|
|
1037
|
+
}
|
|
1038
|
+
const end = body.indexOf(newline);
|
|
1039
|
+
const firstLine = end < 0 ? body : body.slice(0, end);
|
|
1040
|
+
const names = headerNames(firstLine.replace(/\r$/, "").split(delimiter));
|
|
1041
|
+
if (findColumn(names, SOURCE_NAMES) >= 0 && findColumn(names, TARGET_NAMES) >= 0) {
|
|
1042
|
+
return 0.9;
|
|
1043
|
+
}
|
|
1044
|
+
if (findColumn(names, ID_NAMES) >= 0) {
|
|
1045
|
+
return 0.6;
|
|
1046
|
+
}
|
|
1047
|
+
return 0.3;
|
|
1048
|
+
}
|
|
1049
|
+
|
|
1050
|
+
/**
|
|
1051
|
+
* Import a CSV edge table (and an optional node table) into a sink.
|
|
1052
|
+
* @param input - the edge table (or, with `table: "nodes"` or a header without endpoints, a node table)
|
|
1053
|
+
* @param sink - the sink
|
|
1054
|
+
* @param options - CSV and common options
|
|
1055
|
+
* @returns the report; ImportError with the partial report when the import aborts
|
|
1056
|
+
*/
|
|
1057
|
+
async function importCsv(
|
|
1058
|
+
input: ImportInput,
|
|
1059
|
+
sink: GraphSink,
|
|
1060
|
+
options?: CsvImportOptions & CommonImportOptions,
|
|
1061
|
+
): Promise<ImportReport> {
|
|
1062
|
+
const common = resolveImportOptions(options, { ids: "canonical", defaultDirected: true, weightFrom: "weight" });
|
|
1063
|
+
const csv = resolveCsvOptions(options);
|
|
1064
|
+
const report = new ImportReportBuilder("csv", common.errorLimit);
|
|
1065
|
+
reportSinkOptions(sink, options, report);
|
|
1066
|
+
reportUnusedOptions(
|
|
1067
|
+
options,
|
|
1068
|
+
report,
|
|
1069
|
+
csv.table === "nodes" || csv.nodes !== null ? USED_OPTIONS_WITH_NODES : USED_OPTIONS,
|
|
1070
|
+
);
|
|
1071
|
+
const state: ImportState = {
|
|
1072
|
+
sink,
|
|
1073
|
+
report,
|
|
1074
|
+
common,
|
|
1075
|
+
csv,
|
|
1076
|
+
weightFromExplicit: typeof options?.weightFrom === "string",
|
|
1077
|
+
coercer: new IdCoercer(common.ids),
|
|
1078
|
+
resolver: new DirectionResolver(sink, report, common.onMixedDirection),
|
|
1079
|
+
headerSet: false,
|
|
1080
|
+
commentDirected: null,
|
|
1081
|
+
};
|
|
1082
|
+
if (csv.nodes !== null) {
|
|
1083
|
+
await new TableReader(state, csv.nodes, "nodes", false).read();
|
|
1084
|
+
}
|
|
1085
|
+
await new TableReader(state, input, csv.table, true).read();
|
|
1086
|
+
if (state.coercer.mergeCount > 1) {
|
|
1087
|
+
report.warning(
|
|
1088
|
+
"coercion",
|
|
1089
|
+
ID_MERGED_CODE,
|
|
1090
|
+
`${state.coercer.mergeCount} id cell(s) merged into ids other cells already produced under ids: "number"`,
|
|
1091
|
+
);
|
|
1092
|
+
}
|
|
1093
|
+
throwIfAborted(common.signal);
|
|
1094
|
+
return report.finish();
|
|
1095
|
+
}
|
|
1096
|
+
|
|
1097
|
+
/** The CSV / TSV importer plugin (subpath `@graphty/graph-io/csv`). */
|
|
1098
|
+
export const csvImporter: GraphImporter<CsvImportOptions> = Object.freeze({
|
|
1099
|
+
format: "csv",
|
|
1100
|
+
extensions: Object.freeze([".csv", ".tsv", ".edges", ".edgelist"]),
|
|
1101
|
+
mimeTypes: Object.freeze(["text/csv", "text/tab-separated-values", "text/plain"]),
|
|
1102
|
+
sniff,
|
|
1103
|
+
import: importCsv,
|
|
1104
|
+
});
|