@graphty/graph-io 0.0.0 → 0.2.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (339) hide show
  1. package/LICENSE +21 -0
  2. package/README.md +250 -28
  3. package/dist/chunks/children-CL3Cy0ez.js +238 -0
  4. package/dist/chunks/children-CL3Cy0ez.js.map +1 -0
  5. package/dist/chunks/escape-DyI8JofU.js +938 -0
  6. package/dist/chunks/escape-DyI8JofU.js.map +1 -0
  7. package/dist/chunks/importer-CQnJuWJw.js +2987 -0
  8. package/dist/chunks/importer-CQnJuWJw.js.map +1 -0
  9. package/dist/chunks/importer-CpCpfbxr.js +2015 -0
  10. package/dist/chunks/importer-CpCpfbxr.js.map +1 -0
  11. package/dist/chunks/importer-DbnGYr3_.js +2342 -0
  12. package/dist/chunks/importer-DbnGYr3_.js.map +1 -0
  13. package/dist/chunks/importer-GozH8DkN.js +3050 -0
  14. package/dist/chunks/importer-GozH8DkN.js.map +1 -0
  15. package/dist/chunks/records-CGpxszm1.js +605 -0
  16. package/dist/chunks/records-CGpxszm1.js.map +1 -0
  17. package/dist/chunks/text-CajMdVFy.js +189 -0
  18. package/dist/chunks/text-CajMdVFy.js.map +1 -0
  19. package/dist/chunks/writer-DxSKC7TL.js +2842 -0
  20. package/dist/chunks/writer-DxSKC7TL.js.map +1 -0
  21. package/dist/csv.d.ts +1 -0
  22. package/dist/csv.js +1702 -0
  23. package/dist/csv.js.map +1 -0
  24. package/dist/dot.d.ts +1 -0
  25. package/dist/dot.js +8 -0
  26. package/dist/dot.js.map +1 -0
  27. package/dist/gexf.d.ts +1 -0
  28. package/dist/gexf.js +3466 -0
  29. package/dist/gexf.js.map +1 -0
  30. package/dist/gml.d.ts +1 -0
  31. package/dist/gml.js +2647 -0
  32. package/dist/gml.js.map +1 -0
  33. package/dist/graph-io.d.ts +1 -0
  34. package/dist/graph-io.js +790 -0
  35. package/dist/graph-io.js.map +1 -0
  36. package/dist/graphml.d.ts +1 -0
  37. package/dist/graphml.js +8 -0
  38. package/dist/graphml.js.map +1 -0
  39. package/dist/json.d.ts +1 -0
  40. package/dist/json.js +11 -0
  41. package/dist/json.js.map +1 -0
  42. package/dist/neo4j.d.ts +1 -0
  43. package/dist/neo4j.js +2046 -0
  44. package/dist/neo4j.js.map +1 -0
  45. package/dist/pajek.d.ts +1 -0
  46. package/dist/pajek.js +8 -0
  47. package/dist/pajek.js.map +1 -0
  48. package/dist/src/children.d.ts +134 -0
  49. package/dist/src/children.d.ts.map +1 -0
  50. package/dist/src/children.js +274 -0
  51. package/dist/src/children.js.map +1 -0
  52. package/dist/src/common/attributes.d.ts +229 -0
  53. package/dist/src/common/attributes.d.ts.map +1 -0
  54. package/dist/src/common/attributes.js +368 -0
  55. package/dist/src/common/attributes.js.map +1 -0
  56. package/dist/src/common/codes.d.ts +105 -0
  57. package/dist/src/common/codes.d.ts.map +1 -0
  58. package/dist/src/common/codes.js +107 -0
  59. package/dist/src/common/codes.js.map +1 -0
  60. package/dist/src/common/declared-types.d.ts +84 -0
  61. package/dist/src/common/declared-types.d.ts.map +1 -0
  62. package/dist/src/common/declared-types.js +326 -0
  63. package/dist/src/common/declared-types.js.map +1 -0
  64. package/dist/src/common/direction.d.ts +206 -0
  65. package/dist/src/common/direction.d.ts.map +1 -0
  66. package/dist/src/common/direction.js +370 -0
  67. package/dist/src/common/direction.js.map +1 -0
  68. package/dist/src/common/escape.d.ts +92 -0
  69. package/dist/src/common/escape.d.ts.map +1 -0
  70. package/dist/src/common/escape.js +212 -0
  71. package/dist/src/common/escape.js.map +1 -0
  72. package/dist/src/common/export.d.ts +249 -0
  73. package/dist/src/common/export.d.ts.map +1 -0
  74. package/dist/src/common/export.js +594 -0
  75. package/dist/src/common/export.js.map +1 -0
  76. package/dist/src/common/format.d.ts +59 -0
  77. package/dist/src/common/format.d.ts.map +1 -0
  78. package/dist/src/common/format.js +106 -0
  79. package/dist/src/common/format.js.map +1 -0
  80. package/dist/src/common/ids.d.ts +83 -0
  81. package/dist/src/common/ids.d.ts.map +1 -0
  82. package/dist/src/common/ids.js +158 -0
  83. package/dist/src/common/ids.js.map +1 -0
  84. package/dist/src/common/input.d.ts +100 -0
  85. package/dist/src/common/input.d.ts.map +1 -0
  86. package/dist/src/common/input.js +335 -0
  87. package/dist/src/common/input.js.map +1 -0
  88. package/dist/src/common/lists.d.ts +34 -0
  89. package/dist/src/common/lists.d.ts.map +1 -0
  90. package/dist/src/common/lists.js +185 -0
  91. package/dist/src/common/lists.js.map +1 -0
  92. package/dist/src/common/options.d.ts +108 -0
  93. package/dist/src/common/options.d.ts.map +1 -0
  94. package/dist/src/common/options.js +265 -0
  95. package/dist/src/common/options.js.map +1 -0
  96. package/dist/src/common/report.d.ts +187 -0
  97. package/dist/src/common/report.d.ts.map +1 -0
  98. package/dist/src/common/report.js +274 -0
  99. package/dist/src/common/report.js.map +1 -0
  100. package/dist/src/common/temporal.d.ts +71 -0
  101. package/dist/src/common/temporal.d.ts.map +1 -0
  102. package/dist/src/common/temporal.js +266 -0
  103. package/dist/src/common/temporal.js.map +1 -0
  104. package/dist/src/common/text.d.ts +104 -0
  105. package/dist/src/common/text.d.ts.map +1 -0
  106. package/dist/src/common/text.js +255 -0
  107. package/dist/src/common/text.js.map +1 -0
  108. package/dist/src/common/weights.d.ts +77 -0
  109. package/dist/src/common/weights.d.ts.map +1 -0
  110. package/dist/src/common/weights.js +156 -0
  111. package/dist/src/common/weights.js.map +1 -0
  112. package/dist/src/common/writer.d.ts +51 -0
  113. package/dist/src/common/writer.d.ts.map +1 -0
  114. package/dist/src/common/writer.js +108 -0
  115. package/dist/src/common/writer.js.map +1 -0
  116. package/dist/src/common/xml.d.ts +245 -0
  117. package/dist/src/common/xml.d.ts.map +1 -0
  118. package/dist/src/common/xml.js +942 -0
  119. package/dist/src/common/xml.js.map +1 -0
  120. package/dist/src/formats/csv/exporter.d.ts +70 -0
  121. package/dist/src/formats/csv/exporter.d.ts.map +1 -0
  122. package/dist/src/formats/csv/exporter.js +682 -0
  123. package/dist/src/formats/csv/exporter.js.map +1 -0
  124. package/dist/src/formats/csv/header.d.ts +66 -0
  125. package/dist/src/formats/csv/header.d.ts.map +1 -0
  126. package/dist/src/formats/csv/header.js +152 -0
  127. package/dist/src/formats/csv/header.js.map +1 -0
  128. package/dist/src/formats/csv/importer.d.ts +82 -0
  129. package/dist/src/formats/csv/importer.d.ts.map +1 -0
  130. package/dist/src/formats/csv/importer.js +849 -0
  131. package/dist/src/formats/csv/importer.js.map +1 -0
  132. package/dist/src/formats/csv/index.d.ts +60 -0
  133. package/dist/src/formats/csv/index.d.ts.map +1 -0
  134. package/dist/src/formats/csv/index.js +63 -0
  135. package/dist/src/formats/csv/index.js.map +1 -0
  136. package/dist/src/formats/csv/records.d.ts +188 -0
  137. package/dist/src/formats/csv/records.d.ts.map +1 -0
  138. package/dist/src/formats/csv/records.js +702 -0
  139. package/dist/src/formats/csv/records.js.map +1 -0
  140. package/dist/src/formats/csv/values.d.ts +105 -0
  141. package/dist/src/formats/csv/values.d.ts.map +1 -0
  142. package/dist/src/formats/csv/values.js +192 -0
  143. package/dist/src/formats/csv/values.js.map +1 -0
  144. package/dist/src/formats/dot/exporter.d.ts +52 -0
  145. package/dist/src/formats/dot/exporter.d.ts.map +1 -0
  146. package/dist/src/formats/dot/exporter.js +836 -0
  147. package/dist/src/formats/dot/exporter.js.map +1 -0
  148. package/dist/src/formats/dot/importer.d.ts +102 -0
  149. package/dist/src/formats/dot/importer.d.ts.map +1 -0
  150. package/dist/src/formats/dot/importer.js +1291 -0
  151. package/dist/src/formats/dot/importer.js.map +1 -0
  152. package/dist/src/formats/dot/index.d.ts +7 -0
  153. package/dist/src/formats/dot/index.d.ts.map +1 -0
  154. package/dist/src/formats/dot/index.js +7 -0
  155. package/dist/src/formats/dot/index.js.map +1 -0
  156. package/dist/src/formats/dot/names.d.ts +29 -0
  157. package/dist/src/formats/dot/names.d.ts.map +1 -0
  158. package/dist/src/formats/dot/names.js +28 -0
  159. package/dist/src/formats/dot/names.js.map +1 -0
  160. package/dist/src/formats/dot/tokenizer.d.ts +114 -0
  161. package/dist/src/formats/dot/tokenizer.d.ts.map +1 -0
  162. package/dist/src/formats/dot/tokenizer.js +341 -0
  163. package/dist/src/formats/dot/tokenizer.js.map +1 -0
  164. package/dist/src/formats/gexf/exporter.d.ts +56 -0
  165. package/dist/src/formats/gexf/exporter.d.ts.map +1 -0
  166. package/dist/src/formats/gexf/exporter.js +1395 -0
  167. package/dist/src/formats/gexf/exporter.js.map +1 -0
  168. package/dist/src/formats/gexf/importer.d.ts +73 -0
  169. package/dist/src/formats/gexf/importer.d.ts.map +1 -0
  170. package/dist/src/formats/gexf/importer.js +1880 -0
  171. package/dist/src/formats/gexf/importer.js.map +1 -0
  172. package/dist/src/formats/gexf/index.d.ts +96 -0
  173. package/dist/src/formats/gexf/index.d.ts.map +1 -0
  174. package/dist/src/formats/gexf/index.js +97 -0
  175. package/dist/src/formats/gexf/index.js.map +1 -0
  176. package/dist/src/formats/gexf/schema.d.ts +135 -0
  177. package/dist/src/formats/gexf/schema.d.ts.map +1 -0
  178. package/dist/src/formats/gexf/schema.js +323 -0
  179. package/dist/src/formats/gexf/schema.js.map +1 -0
  180. package/dist/src/formats/gml/exporter.d.ts +69 -0
  181. package/dist/src/formats/gml/exporter.d.ts.map +1 -0
  182. package/dist/src/formats/gml/exporter.js +1093 -0
  183. package/dist/src/formats/gml/exporter.js.map +1 -0
  184. package/dist/src/formats/gml/importer.d.ts +66 -0
  185. package/dist/src/formats/gml/importer.d.ts.map +1 -0
  186. package/dist/src/formats/gml/importer.js +1331 -0
  187. package/dist/src/formats/gml/importer.js.map +1 -0
  188. package/dist/src/formats/gml/index.d.ts +85 -0
  189. package/dist/src/formats/gml/index.d.ts.map +1 -0
  190. package/dist/src/formats/gml/index.js +88 -0
  191. package/dist/src/formats/gml/index.js.map +1 -0
  192. package/dist/src/formats/gml/syntax.d.ts +186 -0
  193. package/dist/src/formats/gml/syntax.d.ts.map +1 -0
  194. package/dist/src/formats/gml/syntax.js +467 -0
  195. package/dist/src/formats/gml/syntax.js.map +1 -0
  196. package/dist/src/formats/graphml/constants.d.ts +169 -0
  197. package/dist/src/formats/graphml/constants.d.ts.map +1 -0
  198. package/dist/src/formats/graphml/constants.js +165 -0
  199. package/dist/src/formats/graphml/constants.js.map +1 -0
  200. package/dist/src/formats/graphml/exporter.d.ts +34 -0
  201. package/dist/src/formats/graphml/exporter.d.ts.map +1 -0
  202. package/dist/src/formats/graphml/exporter.js +1176 -0
  203. package/dist/src/formats/graphml/exporter.js.map +1 -0
  204. package/dist/src/formats/graphml/importer.d.ts +31 -0
  205. package/dist/src/formats/graphml/importer.d.ts.map +1 -0
  206. package/dist/src/formats/graphml/importer.js +1607 -0
  207. package/dist/src/formats/graphml/importer.js.map +1 -0
  208. package/dist/src/formats/graphml/index.d.ts +8 -0
  209. package/dist/src/formats/graphml/index.d.ts.map +1 -0
  210. package/dist/src/formats/graphml/index.js +8 -0
  211. package/dist/src/formats/graphml/index.js.map +1 -0
  212. package/dist/src/formats/graphml/tree.d.ts +72 -0
  213. package/dist/src/formats/graphml/tree.d.ts.map +1 -0
  214. package/dist/src/formats/graphml/tree.js +290 -0
  215. package/dist/src/formats/graphml/tree.js.map +1 -0
  216. package/dist/src/formats/json/dialect.d.ts +125 -0
  217. package/dist/src/formats/json/dialect.d.ts.map +1 -0
  218. package/dist/src/formats/json/dialect.js +262 -0
  219. package/dist/src/formats/json/dialect.js.map +1 -0
  220. package/dist/src/formats/json/exporter.d.ts +89 -0
  221. package/dist/src/formats/json/exporter.d.ts.map +1 -0
  222. package/dist/src/formats/json/exporter.js +1358 -0
  223. package/dist/src/formats/json/exporter.js.map +1 -0
  224. package/dist/src/formats/json/importer.d.ts +108 -0
  225. package/dist/src/formats/json/importer.d.ts.map +1 -0
  226. package/dist/src/formats/json/importer.js +1838 -0
  227. package/dist/src/formats/json/importer.js.map +1 -0
  228. package/dist/src/formats/json/index.d.ts +8 -0
  229. package/dist/src/formats/json/index.d.ts.map +1 -0
  230. package/dist/src/formats/json/index.js +8 -0
  231. package/dist/src/formats/json/index.js.map +1 -0
  232. package/dist/src/formats/neo4j/exporter.d.ts +68 -0
  233. package/dist/src/formats/neo4j/exporter.d.ts.map +1 -0
  234. package/dist/src/formats/neo4j/exporter.js +1055 -0
  235. package/dist/src/formats/neo4j/exporter.js.map +1 -0
  236. package/dist/src/formats/neo4j/header.d.ts +52 -0
  237. package/dist/src/formats/neo4j/header.d.ts.map +1 -0
  238. package/dist/src/formats/neo4j/header.js +131 -0
  239. package/dist/src/formats/neo4j/header.js.map +1 -0
  240. package/dist/src/formats/neo4j/importer.d.ts +73 -0
  241. package/dist/src/formats/neo4j/importer.d.ts.map +1 -0
  242. package/dist/src/formats/neo4j/importer.js +932 -0
  243. package/dist/src/formats/neo4j/importer.js.map +1 -0
  244. package/dist/src/formats/neo4j/index.d.ts +79 -0
  245. package/dist/src/formats/neo4j/index.d.ts.map +1 -0
  246. package/dist/src/formats/neo4j/index.js +83 -0
  247. package/dist/src/formats/neo4j/index.js.map +1 -0
  248. package/dist/src/formats/pajek/exporter.d.ts +58 -0
  249. package/dist/src/formats/pajek/exporter.d.ts.map +1 -0
  250. package/dist/src/formats/pajek/exporter.js +825 -0
  251. package/dist/src/formats/pajek/exporter.js.map +1 -0
  252. package/dist/src/formats/pajek/importer.d.ts +88 -0
  253. package/dist/src/formats/pajek/importer.d.ts.map +1 -0
  254. package/dist/src/formats/pajek/importer.js +1047 -0
  255. package/dist/src/formats/pajek/importer.js.map +1 -0
  256. package/dist/src/formats/pajek/index.d.ts +7 -0
  257. package/dist/src/formats/pajek/index.d.ts.map +1 -0
  258. package/dist/src/formats/pajek/index.js +7 -0
  259. package/dist/src/formats/pajek/index.js.map +1 -0
  260. package/dist/src/formats/pajek/syntax.d.ts +112 -0
  261. package/dist/src/formats/pajek/syntax.d.ts.map +1 -0
  262. package/dist/src/formats/pajek/syntax.js +269 -0
  263. package/dist/src/formats/pajek/syntax.js.map +1 -0
  264. package/dist/src/index.d.ts +35 -0
  265. package/dist/src/index.d.ts.map +1 -0
  266. package/dist/src/index.js +39 -0
  267. package/dist/src/index.js.map +1 -0
  268. package/dist/src/registry.d.ts +207 -0
  269. package/dist/src/registry.d.ts.map +1 -0
  270. package/dist/src/registry.js +481 -0
  271. package/dist/src/registry.js.map +1 -0
  272. package/dist/src/sniff.d.ts +104 -0
  273. package/dist/src/sniff.d.ts.map +1 -0
  274. package/dist/src/sniff.js +357 -0
  275. package/dist/src/sniff.js.map +1 -0
  276. package/dist/src/types.d.ts +238 -0
  277. package/dist/src/types.d.ts.map +1 -0
  278. package/dist/src/types.js +29 -0
  279. package/dist/src/types.js.map +1 -0
  280. package/dist/tsconfig.build.tsbuildinfo +1 -0
  281. package/package.json +122 -7
  282. package/src/children.ts +335 -0
  283. package/src/common/attributes.ts +520 -0
  284. package/src/common/codes.ts +153 -0
  285. package/src/common/declared-types.ts +374 -0
  286. package/src/common/direction.ts +518 -0
  287. package/src/common/escape.ts +231 -0
  288. package/src/common/export.ts +817 -0
  289. package/src/common/format.ts +111 -0
  290. package/src/common/ids.ts +176 -0
  291. package/src/common/input.ts +378 -0
  292. package/src/common/lists.ts +196 -0
  293. package/src/common/options.ts +377 -0
  294. package/src/common/report.ts +352 -0
  295. package/src/common/temporal.ts +302 -0
  296. package/src/common/text.ts +294 -0
  297. package/src/common/weights.ts +202 -0
  298. package/src/common/writer.ts +123 -0
  299. package/src/common/xml.ts +1053 -0
  300. package/src/formats/csv/exporter.ts +894 -0
  301. package/src/formats/csv/header.ts +172 -0
  302. package/src/formats/csv/importer.ts +1104 -0
  303. package/src/formats/csv/index.ts +88 -0
  304. package/src/formats/csv/records.ts +813 -0
  305. package/src/formats/csv/values.ts +224 -0
  306. package/src/formats/dot/exporter.ts +1014 -0
  307. package/src/formats/dot/importer.ts +1549 -0
  308. package/src/formats/dot/index.ts +7 -0
  309. package/src/formats/dot/names.ts +40 -0
  310. package/src/formats/dot/tokenizer.ts +384 -0
  311. package/src/formats/gexf/exporter.ts +1696 -0
  312. package/src/formats/gexf/importer.ts +2333 -0
  313. package/src/formats/gexf/index.ts +142 -0
  314. package/src/formats/gexf/schema.ts +361 -0
  315. package/src/formats/gml/exporter.ts +1404 -0
  316. package/src/formats/gml/importer.ts +1591 -0
  317. package/src/formats/gml/index.ts +128 -0
  318. package/src/formats/gml/syntax.ts +545 -0
  319. package/src/formats/graphml/constants.ts +225 -0
  320. package/src/formats/graphml/exporter.ts +1458 -0
  321. package/src/formats/graphml/importer.ts +2027 -0
  322. package/src/formats/graphml/index.ts +8 -0
  323. package/src/formats/graphml/tree.ts +318 -0
  324. package/src/formats/json/dialect.ts +317 -0
  325. package/src/formats/json/exporter.ts +1616 -0
  326. package/src/formats/json/importer.ts +2271 -0
  327. package/src/formats/json/index.ts +8 -0
  328. package/src/formats/neo4j/exporter.ts +1287 -0
  329. package/src/formats/neo4j/header.ts +156 -0
  330. package/src/formats/neo4j/importer.ts +1220 -0
  331. package/src/formats/neo4j/index.ts +116 -0
  332. package/src/formats/pajek/exporter.ts +1000 -0
  333. package/src/formats/pajek/importer.ts +1311 -0
  334. package/src/formats/pajek/index.ts +7 -0
  335. package/src/formats/pajek/syntax.ts +307 -0
  336. package/src/index.ts +244 -0
  337. package/src/registry.ts +617 -0
  338. package/src/sniff.ts +397 -0
  339. package/src/types.ts +262 -0
@@ -0,0 +1,849 @@
1
+ /**
2
+ * The CSV / TSV importer (design sections 8.4 and 8.6; research note 07 section 2.6): a streaming
3
+ * edge-list reader for the generic (`source,target[,weight,...]`), Gephi (`Source,Target,Type,Id,
4
+ * Label,Weight,...`) and headerless (`u v [w]`) dialects, with an optional node table merged by id.
5
+ *
6
+ * - The delimiter is sniffed from a preview unless given; LF, CRLF and lone-CR files all read.
7
+ * - The first row is a header when it holds a known column name or when it is all text over a
8
+ * numeric second row (`header: "auto"`); a headerless file is positional: source, target,
9
+ * weight (when `weightFrom` is not null), then `column4`... as attributes.
10
+ * - Endpoint and node ids are text cells coerced by ONE rule per import call (`ids`, default
11
+ * "canonical": `"1"` becomes the number 1, `"01"` stays a string) over the node table and the
12
+ * edge table alike, so they agree across the two files.
13
+ * - Direction is per row in the Gephi dialect (`Type` = Directed / Undirected / Mutual, blank =
14
+ * `defaultDirected`) and `defaultDirected` (true) otherwise; the first edge row sets the sink's
15
+ * direction and later rows that differ go through `onMixedDirection` (expansion by default).
16
+ * - The weight column (`weightFrom`, default "weight", matched case-insensitively) is parsed per
17
+ * row; a blank cell means "no weight given" and the edge is pushed without one.
18
+ * - Every other column is an attribute: cells are parsed by the fixed lexical grammar of design
19
+ * section 5.1 and the sink infers the column dtype (widening per column, never per cell); an
20
+ * all-text column of low cardinality becomes a dict (design section 5.4); an `id` column of the
21
+ * edge table is the edge id (role id, unique); a `label` column is the label (role label).
22
+ * - Per-row problems (wrong field count, blank endpoint, invalid weight, bad Type, refused id)
23
+ * are recorded and the row skipped; the import aborts with ImportError once `errorLimit` is
24
+ * exceeded, on a malformed or unterminated quoted field, on an empty input and on a header
25
+ * without endpoint (or id) columns.
26
+ */
27
+ import { GraphFormatError, INVALID_INDEX, } from "@graphty/graph-format";
28
+ import { declareResolved, RENAMED_CODE, uniqueColumnName } from "../../common/attributes.js";
29
+ import { DUPLICATE_EDGE_ID_CODE as SHARED_DUPLICATE_EDGE_ID_CODE, DUPLICATE_NODE_CODE as SHARED_DUPLICATE_NODE_CODE, EMPTY_INPUT_CODE as SHARED_EMPTY_INPUT_CODE, ID_MERGED_CODE as SHARED_ID_MERGED_CODE, MISSING_ENDPOINT_CODE as SHARED_MISSING_ENDPOINT_CODE, MISSING_ID_CODE as SHARED_MISSING_ID_CODE, ROLE_TAKEN_CODE as SHARED_ROLE_TAKEN_CODE, } from "../../common/codes.js";
30
+ import { DirectionResolver } from "../../common/direction.js";
31
+ import { IdCoercer } from "../../common/ids.js";
32
+ import { throwIfAborted } from "../../common/input.js";
33
+ import { reportSinkOptions, reportUnusedOptions, resolveImportOptions, } from "../../common/options.js";
34
+ import { ImportReportBuilder } from "../../common/report.js";
35
+ import { parseWeightText } from "../../common/weights.js";
36
+ import { EDGE_ID_NAMES, findColumn, headerNames, ID_NAMES, LABEL_NAMES, looksLikeHeader, positionalNames, resolveColumnRef, SOURCE_NAMES, TARGET_NAMES, TYPE_NAME, } from "./header.js";
37
+ import { CsvRecordReader, sniffDelimiter, sniffNewline } from "./records.js";
38
+ import { InferredColumn } from "./values.js";
39
+ /** Issue code: the input holds no header row at all. */
40
+ export const EMPTY_INPUT_CODE = SHARED_EMPTY_INPUT_CODE;
41
+ /** Issue code: the header names no source / target (or, for a node table, no id) column. */
42
+ export const NO_ENDPOINT_COLUMNS_CODE = "E_CSV_NO_ENDPOINT_COLUMNS";
43
+ /** Issue code: a node table without an id column. */
44
+ export const NO_ID_COLUMN_CODE = "E_CSV_NO_ID_COLUMN";
45
+ /** Issue code: a row with a different number of fields than the header. */
46
+ export const FIELD_COUNT_CODE = "E_CSV_FIELD_COUNT";
47
+ /** Issue code: an edge row with a blank source or target cell. */
48
+ export const MISSING_ENDPOINT_CODE = SHARED_MISSING_ENDPOINT_CODE;
49
+ /** Issue code: a node row with a blank id cell. */
50
+ export const MISSING_ID_CODE = SHARED_MISSING_ID_CODE;
51
+ /** Issue code: a Type cell that is not Directed, Undirected or Mutual. */
52
+ export const BAD_TYPE_CODE = "E_CSV_BAD_TYPE";
53
+ /** Issue code: the table has a header and no data rows. */
54
+ export const NO_DATA_ROWS_CODE = "W_CSV_NO_DATA_ROWS";
55
+ /** Issue code: a node table row repeats an id; its attributes overwrite the earlier row's. */
56
+ export const DUPLICATE_NODE_CODE = SHARED_DUPLICATE_NODE_CODE;
57
+ /** Issue code: two distinct id cells became one id under `ids: "number"` (design section 4.1). */
58
+ export const ID_MERGED_CODE = SHARED_ID_MERGED_CODE;
59
+ /** Issue code: an explicitly named weight column the file does not have. */
60
+ export const COLUMN_MISSING_CODE = "W_CSV_COLUMN_MISSING";
61
+ /** Issue code: a column whose role (id, label) is already held by another column of the sink. */
62
+ export const ROLE_TAKEN_CODE = SHARED_ROLE_TAKEN_CODE;
63
+ /** Issue code: a repeated edge id (the column is unique); the edge is skipped. */
64
+ export const DUPLICATE_EDGE_ID_CODE = SHARED_DUPLICATE_EDGE_ID_CODE;
65
+ const TABLE_MODES = new Set(["edges", "nodes", "auto"]);
66
+ /** The common options an edge-table import reads (the rest is reported by reportUnusedOptions). */
67
+ const USED_OPTIONS = new Set([
68
+ "ids",
69
+ "addMissingNodes",
70
+ "duplicateEdges",
71
+ "selfLoops",
72
+ "onMixedDirection",
73
+ "defaultDirected",
74
+ "weightFrom",
75
+ "weightDtype",
76
+ "errorLimit",
77
+ "signal",
78
+ "onProgress",
79
+ ]);
80
+ /** The common options an import with a node table reads: nodeIdFrom applies to the node table. */
81
+ const USED_OPTIONS_WITH_NODES = new Set([
82
+ ...USED_OPTIONS,
83
+ "nodeIdFrom",
84
+ ]);
85
+ const BAD_DELIMITERS = new Set(['"', "\n", "\r"]);
86
+ /** The comment characters of the SNAP (`#`) and KONECT (`%`) headers (research note 07 section 2.6). */
87
+ const COMMENT_CHARS = Object.freeze(["#", "%"]);
88
+ /**
89
+ * The direction a SNAP or KONECT comment header declares (research note 07 section 2.6): SNAP
90
+ * pages write `# Directed graph` / `# Undirected graph`, KONECT's first line is `% sym` (undirected),
91
+ * `% asym` (directed) or `% bip` (bipartite, undirected).
92
+ * @param comments - the leading comment lines
93
+ * @returns true / false when a line declares the direction, null otherwise
94
+ */
95
+ function commentDirection(comments) {
96
+ for (const comment of comments) {
97
+ const text = comment.slice(1).trim().toLowerCase();
98
+ if (text.startsWith("directed graph") || text.startsWith("asym")) {
99
+ return true;
100
+ }
101
+ if (text.startsWith("undirected graph") || text.startsWith("sym") || text.startsWith("bip")) {
102
+ return false;
103
+ }
104
+ }
105
+ return null;
106
+ }
107
+ /**
108
+ * Apply the defaults of the CSV options and check every value.
109
+ * @param options - the caller's options
110
+ * @returns the resolved options; E_UNSUPPORTED for a value outside its set
111
+ */
112
+ function resolveCsvOptions(options) {
113
+ const o = options ?? {};
114
+ if (o.delimiter !== undefined && (typeof o.delimiter !== "string" || o.delimiter.length === 0)) {
115
+ throw new GraphFormatError("E_UNSUPPORTED", "option delimiter: expected a non-empty string", {
116
+ option: "delimiter",
117
+ found: o.delimiter,
118
+ });
119
+ }
120
+ if (o.delimiter !== undefined && BAD_DELIMITERS.has(o.delimiter)) {
121
+ throw new GraphFormatError("E_UNSUPPORTED", "option delimiter: a quote or a line break cannot delimit", {
122
+ option: "delimiter",
123
+ found: o.delimiter,
124
+ });
125
+ }
126
+ if (o.header !== undefined && o.header !== "auto" && typeof o.header !== "boolean") {
127
+ throw new GraphFormatError("E_UNSUPPORTED", 'option header: expected true, false or "auto"', {
128
+ option: "header",
129
+ found: o.header,
130
+ });
131
+ }
132
+ if (o.table !== undefined && !TABLE_MODES.has(o.table)) {
133
+ throw new GraphFormatError("E_UNSUPPORTED", 'option table: expected "edges", "nodes" or "auto"', {
134
+ option: "table",
135
+ found: o.table,
136
+ });
137
+ }
138
+ for (const name of ["sourceColumn", "targetColumn", "idColumn"]) {
139
+ checkColumnRef(name, o[name]);
140
+ }
141
+ if (o.typeColumn !== null) {
142
+ checkColumnRef("typeColumn", o.typeColumn);
143
+ }
144
+ return {
145
+ delimiter: o.delimiter ?? null,
146
+ header: o.header ?? "auto",
147
+ table: o.table ?? "auto",
148
+ sourceColumn: o.sourceColumn ?? null,
149
+ targetColumn: o.targetColumn ?? null,
150
+ typeColumn: o.typeColumn,
151
+ idColumn: o.idColumn ?? null,
152
+ nodes: o.nodes ?? null,
153
+ };
154
+ }
155
+ /**
156
+ * Check a column reference option: a string name or a non-negative integer position.
157
+ * @param name - the option name
158
+ * @param value - the value
159
+ */
160
+ function checkColumnRef(name, value) {
161
+ if (value === undefined) {
162
+ return;
163
+ }
164
+ if (typeof value === "string" && value.length > 0) {
165
+ return;
166
+ }
167
+ if (typeof value === "number" && Number.isInteger(value) && value >= 0) {
168
+ return;
169
+ }
170
+ throw new GraphFormatError("E_UNSUPPORTED", `option ${name}: expected a column name or a 0-based position`, {
171
+ option: name,
172
+ found: value,
173
+ });
174
+ }
175
+ /**
176
+ * Whether a cell is empty (nothing or whitespace only): an unset value, a missing endpoint.
177
+ * @param text - the cell text
178
+ * @returns true when blank
179
+ */
180
+ function isBlank(text) {
181
+ return text.length === 0 || text.trim().length === 0;
182
+ }
183
+ /**
184
+ * Whether a cell is unset: blank and not quoted. A quoted blank cell (`""`, `" "`) is a set value
185
+ * (the empty string is a legal id and a legal value, design section 4.1); an unquoted one is
186
+ * nothing at all.
187
+ * @param text - the cell text
188
+ * @param quoted - whether the cell was quoted
189
+ * @returns true when the cell carries no value
190
+ */
191
+ function isUnset(text, quoted) {
192
+ return quoted !== true && isBlank(text);
193
+ }
194
+ /** Rows between two checks of the cancellation signal on an in-memory input. */
195
+ const ABORT_CHECK_INTERVAL = 64;
196
+ /**
197
+ * Header names made unique within the table: a repeated name becomes `<name>#<position>` (1-based,
198
+ * the CSV analogue of the `#<origin.id>` rule of design section 5.6) and is reported.
199
+ * @param names - the header names
200
+ * @param report - the report
201
+ * @param line - the header line
202
+ * @returns the unique names
203
+ */
204
+ function uniqueNames(names, report, line) {
205
+ const seen = new Set();
206
+ return names.map((name, i) => {
207
+ const unique = uniqueColumnName(name, String(i + 1), (n) => seen.has(n));
208
+ seen.add(unique);
209
+ if (unique !== name) {
210
+ report.warning("coercion", RENAMED_CODE, `column ${i + 1} "${name}" renamed to "${unique}": the header repeats the name`, { line, element: name });
211
+ }
212
+ return unique;
213
+ });
214
+ }
215
+ /**
216
+ * Find a column by candidates, skipping indices already claimed by another role.
217
+ * @param names - the header names
218
+ * @param candidates - the names to look for
219
+ * @param claimed - indices taken
220
+ * @returns the index, or -1
221
+ */
222
+ function findFree(names, candidates, claimed) {
223
+ const masked = names.map((name, i) => (claimed.has(i) ? "" : name));
224
+ return findColumn(masked, candidates);
225
+ }
226
+ /**
227
+ * Declare a role column (edge id, label) on the sink through the shared design section 5.6 rule:
228
+ * a name already declared differently is renamed `<name>#<position>` (reported); a role already
229
+ * held by another column is dropped from this declaration (reported), the values are kept under
230
+ * the name.
231
+ * @param sink - the sink
232
+ * @param domain - node or edge
233
+ * @param decl - the declaration
234
+ * @param position - the 1-based column position, for the rename
235
+ * @param report - the report
236
+ * @param line - the header line
237
+ * @returns the handle
238
+ */
239
+ function declareRoleColumn(sink, domain, decl, position, report, line) {
240
+ const withOrigin = { ...decl, origin: { ...decl.origin, id: String(position) } };
241
+ const resolved = declareResolved(sink, domain, withOrigin, report, { line, element: decl.name });
242
+ if (resolved.roleDropped) {
243
+ // a unique constraint belongs to the id role; without the role the column is plain text
244
+ return resolved.handle;
245
+ }
246
+ return resolved.handle;
247
+ }
248
+ /**
249
+ * Parse a Gephi Type cell.
250
+ * @param text - the cell text
251
+ * @returns the kind; undefined for a blank cell (the default applies); null for an unknown word
252
+ */
253
+ function parseKind(text) {
254
+ if (isBlank(text)) {
255
+ return undefined;
256
+ }
257
+ switch (text.trim().toLowerCase()) {
258
+ case "directed":
259
+ return "directed";
260
+ case "undirected":
261
+ return "undirected";
262
+ case "mutual":
263
+ return "mutual";
264
+ default:
265
+ return null;
266
+ }
267
+ }
268
+ /**
269
+ * Read one CSV table into the sink.
270
+ */
271
+ class TableReader {
272
+ /**
273
+ * Create a table reader.
274
+ * @param state - the import state
275
+ * @param input - the table's input
276
+ * @param kind - what the table is, or "auto"
277
+ * @param progress - whether this table reports byte progress
278
+ */
279
+ constructor(state, input, kind, progress) {
280
+ this.plan = null;
281
+ this.writers = [];
282
+ this.idHandle = INVALID_INDEX;
283
+ this.labelHandle = INVALID_INDEX;
284
+ this.dataRows = 0;
285
+ this.nodeOrdinal = 0;
286
+ /** The edge ids seen so far (the id column is unique; a repeat is skipped with an issue). */
287
+ this.edgeIds = new Set();
288
+ this.where = { line: null, element: null };
289
+ this.state = state;
290
+ this.kind = kind;
291
+ const readerOptions = {
292
+ delimiter: state.csv.delimiter,
293
+ comments: COMMENT_CHARS,
294
+ signal: state.common.signal,
295
+ onProgress: progress ? state.common.onProgress : null,
296
+ };
297
+ this.reader = new CsvRecordReader(input, state.report, readerOptions);
298
+ }
299
+ /**
300
+ * Read every row; the reader is closed (and a stream cancelled) when the import aborts midway.
301
+ */
302
+ async read() {
303
+ const iterator = this.reader[Symbol.asyncIterator]();
304
+ try {
305
+ await this.readRows(iterator);
306
+ }
307
+ finally {
308
+ await iterator.return(undefined);
309
+ }
310
+ }
311
+ /**
312
+ * Read the header (or decide there is none), resolve the plan, then push every row.
313
+ * @param iterator - the record iterator
314
+ */
315
+ async readRows(iterator) {
316
+ const { report } = this.state;
317
+ const first = await iterator.next();
318
+ const firstRow = first.done
319
+ ? report.fail(EMPTY_INPUT_CODE, "the input is empty: no header row and no records")
320
+ : first.value;
321
+ const firstLine = this.reader.line;
322
+ const firstQuoted = this.reader.quoted.slice(0, firstRow.length);
323
+ const pending = [];
324
+ let header;
325
+ const { header: mode } = this.state.csv;
326
+ if (mode === "auto") {
327
+ const second = await iterator.next();
328
+ const secondRow = second.done ? null : second.value;
329
+ const secondLine = this.reader.line;
330
+ const secondQuoted = this.reader.quoted.slice(0, secondRow?.length ?? 0);
331
+ header = looksLikeHeader(firstRow, secondRow);
332
+ if (!header) {
333
+ pending.push({ row: firstRow, quoted: firstQuoted, line: firstLine });
334
+ }
335
+ if (secondRow !== null) {
336
+ pending.push({ row: secondRow, quoted: secondQuoted, line: secondLine });
337
+ }
338
+ }
339
+ else {
340
+ header = mode;
341
+ if (!header) {
342
+ pending.push({ row: firstRow, quoted: firstQuoted, line: firstLine });
343
+ }
344
+ }
345
+ const names = header ? uniqueNames(headerNames(firstRow), report, firstLine) : positionalNames(firstRow.length);
346
+ if (this.kind !== "nodes" && this.state.commentDirected === null) {
347
+ this.state.commentDirected = commentDirection(this.reader.leadingComments);
348
+ }
349
+ this.plan = this.resolvePlan(names, header, firstLine);
350
+ this.prepareColumns(firstLine);
351
+ for (const { row, quoted, line } of pending) {
352
+ this.processRow(row, quoted, line);
353
+ }
354
+ const { signal } = this.state.common;
355
+ let sinceCheck = 0;
356
+ for (;;) {
357
+ const next = await iterator.next();
358
+ if (next.done) {
359
+ break;
360
+ }
361
+ this.processRow(next.value, this.reader.quoted, this.reader.line);
362
+ if (++sinceCheck >= ABORT_CHECK_INTERVAL) {
363
+ sinceCheck = 0;
364
+ throwIfAborted(signal);
365
+ }
366
+ }
367
+ for (const writer of this.writers) {
368
+ writer?.finish();
369
+ }
370
+ if (this.dataRows === 0 && header) {
371
+ report.warning("missing-value", NO_DATA_ROWS_CODE, "the table has a header and no data rows", {
372
+ line: firstLine,
373
+ });
374
+ }
375
+ }
376
+ /**
377
+ * Decide the table's columns from its header.
378
+ * @param names - the column names
379
+ * @param header - whether the file has a header row
380
+ * @param line - the header line
381
+ * @returns the plan; the import aborts when no endpoints (or id) resolve
382
+ */
383
+ resolvePlan(names, header, line) {
384
+ const { csv, report } = this.state;
385
+ const width = names.length;
386
+ let source = -1;
387
+ let target = -1;
388
+ if (this.kind !== "nodes") {
389
+ if (csv.sourceColumn !== null) {
390
+ source = resolveColumnRef(names, csv.sourceColumn, "sourceColumn");
391
+ }
392
+ else if (header) {
393
+ source = findColumn(names, SOURCE_NAMES);
394
+ }
395
+ else {
396
+ source = width >= 2 ? 0 : -1;
397
+ }
398
+ if (csv.targetColumn !== null) {
399
+ target = resolveColumnRef(names, csv.targetColumn, "targetColumn");
400
+ }
401
+ else if (header) {
402
+ target = findColumn(names, TARGET_NAMES);
403
+ }
404
+ else {
405
+ target = width >= 2 ? 1 : -1;
406
+ }
407
+ if (source >= 0 && target >= 0 && source === target) {
408
+ throw new GraphFormatError("E_UNSUPPORTED", "sourceColumn and targetColumn name the same column", {
409
+ option: "targetColumn",
410
+ found: names[target],
411
+ });
412
+ }
413
+ }
414
+ if (source >= 0 && target >= 0) {
415
+ return this.edgePlan(names, header, source, target);
416
+ }
417
+ const shown = names.map((n) => JSON.stringify(n)).join(", ");
418
+ if (this.kind === "edges" ||
419
+ (this.kind === "auto" && (csv.sourceColumn !== null || csv.targetColumn !== null))) {
420
+ report.fail(NO_ENDPOINT_COLUMNS_CODE, `no source / target columns in the header (${shown}); a node table goes in the nodes option`, { line }, { columns: [...names] });
421
+ }
422
+ const idResolves = header ? findColumn(names, ID_NAMES) >= 0 : width >= 1;
423
+ if (this.kind === "auto" && csv.idColumn === null && !idResolves) {
424
+ report.fail(NO_ENDPOINT_COLUMNS_CODE, `no source / target columns and no id column in the header (${shown}); the input is neither an edge table nor a node table`, { line }, { columns: [...names] });
425
+ }
426
+ return this.nodePlan(names, header, line);
427
+ }
428
+ /**
429
+ * The columns of an edge table.
430
+ * @param names - the column names
431
+ * @param header - whether the file has a header row
432
+ * @param source - the source column
433
+ * @param target - the target column
434
+ * @returns the plan
435
+ */
436
+ edgePlan(names, header, source, target) {
437
+ const { csv, common, report } = this.state;
438
+ const claimed = new Set([source, target]);
439
+ let weight = -1;
440
+ if (common.weightFrom !== null) {
441
+ if (header) {
442
+ weight = findFree(names, [common.weightFrom], claimed);
443
+ if (weight < 0 && this.state.weightFromExplicit) {
444
+ report.warning("missing-value", COLUMN_MISSING_CODE, `weight column ${JSON.stringify(common.weightFrom)} not found; edges are unweighted`, { line: this.reader.line, element: common.weightFrom });
445
+ }
446
+ }
447
+ else if (names.length >= 3 && !claimed.has(2)) {
448
+ weight = 2;
449
+ }
450
+ }
451
+ if (weight >= 0) {
452
+ claimed.add(weight);
453
+ }
454
+ let type = -1;
455
+ if (csv.typeColumn === undefined) {
456
+ if (header && names[source] === "Source" && names[target] === "Target") {
457
+ type = findFree(names, [TYPE_NAME], claimed);
458
+ if (type >= 0 && names[type] !== TYPE_NAME) {
459
+ type = -1;
460
+ }
461
+ }
462
+ }
463
+ else if (csv.typeColumn !== null) {
464
+ type = resolveColumnRef(names, csv.typeColumn, "typeColumn");
465
+ if (claimed.has(type)) {
466
+ throw new GraphFormatError("E_UNSUPPORTED", "typeColumn names an endpoint or weight column", {
467
+ option: "typeColumn",
468
+ found: names[type],
469
+ });
470
+ }
471
+ }
472
+ if (type >= 0) {
473
+ claimed.add(type);
474
+ }
475
+ const id = header ? findFree(names, EDGE_ID_NAMES, claimed) : -1;
476
+ if (id >= 0) {
477
+ claimed.add(id);
478
+ }
479
+ const label = header ? findFree(names, LABEL_NAMES, claimed) : -1;
480
+ if (label >= 0) {
481
+ claimed.add(label);
482
+ }
483
+ const attributes = [];
484
+ for (let i = 0; i < names.length; i++) {
485
+ if (!claimed.has(i)) {
486
+ attributes.push(i);
487
+ }
488
+ }
489
+ return { kind: "edges", names, width: names.length, source, target, weight, type, id, label, attributes };
490
+ }
491
+ /**
492
+ * The columns of a node table.
493
+ * @param names - the column names
494
+ * @param header - whether the file has a header row
495
+ * @param line - the header line
496
+ * @returns the plan; the import aborts when no id column resolves
497
+ */
498
+ nodePlan(names, header, line) {
499
+ const { csv, common, report } = this.state;
500
+ const claimed = new Set();
501
+ let idColumn = -1;
502
+ if (csv.idColumn !== null) {
503
+ idColumn = resolveColumnRef(names, csv.idColumn, "idColumn");
504
+ }
505
+ else if (header) {
506
+ idColumn = findColumn(names, ID_NAMES);
507
+ }
508
+ else if (names.length >= 1) {
509
+ idColumn = 0;
510
+ }
511
+ const label = header ? findFree(names, LABEL_NAMES, new Set(idColumn >= 0 ? [idColumn] : [])) : -1;
512
+ let id;
513
+ switch (common.nodeIdFrom) {
514
+ case "label":
515
+ if (label < 0) {
516
+ report.fail(NO_ID_COLUMN_CODE, 'nodeIdFrom is "label" but the node table has no label column', {
517
+ line,
518
+ });
519
+ }
520
+ id = label;
521
+ break;
522
+ case "index":
523
+ id = -1;
524
+ break;
525
+ default:
526
+ if (idColumn < 0) {
527
+ report.fail(NO_ID_COLUMN_CODE, `no id column in the node table header (${names.map((n) => JSON.stringify(n)).join(", ")})`, { line }, { columns: [...names] });
528
+ }
529
+ id = idColumn;
530
+ claimed.add(idColumn);
531
+ break;
532
+ }
533
+ if (label >= 0) {
534
+ claimed.add(label);
535
+ }
536
+ const attributes = [];
537
+ for (let i = 0; i < names.length; i++) {
538
+ if (!claimed.has(i)) {
539
+ attributes.push(i);
540
+ }
541
+ }
542
+ return { kind: "nodes", names, width: names.length, id, label, attributes };
543
+ }
544
+ /**
545
+ * Declare the role columns and create a writer per attribute column.
546
+ * @param line - the header line
547
+ */
548
+ prepareColumns(line) {
549
+ const plan = this.requirePlan();
550
+ const { sink, report } = this.state;
551
+ const domain = plan.kind === "edges" ? "edge" : "node";
552
+ const origin = { format: "csv" };
553
+ if (plan.kind === "edges" && plan.id >= 0) {
554
+ this.idHandle = declareRoleColumn(sink, "edge", { name: plan.names[plan.id], dtype: "string", nullable: true, role: "id", unique: true, origin }, plan.id + 1, report, line);
555
+ }
556
+ if (plan.label >= 0) {
557
+ this.labelHandle = declareRoleColumn(sink, domain, { name: plan.names[plan.label], dtype: "string", nullable: true, role: "label", origin }, plan.label + 1, report, line);
558
+ }
559
+ this.writers = plan.names.map(() => null);
560
+ for (const index of plan.attributes) {
561
+ this.writers[index] = new InferredColumn(plan.names[index], domain, sink, report);
562
+ }
563
+ }
564
+ /**
565
+ * The plan, which exists once the header was read.
566
+ * @returns the plan
567
+ */
568
+ requirePlan() {
569
+ if (this.plan === null) {
570
+ throw new GraphFormatError("E_UNSUPPORTED", "the header has not been read", { reason: "no plan" });
571
+ }
572
+ return this.plan;
573
+ }
574
+ /**
575
+ * Push one data row.
576
+ * @param row - the cells
577
+ * @param quoted - whether each cell was quoted (a quoted empty cell is the empty string)
578
+ * @param line - the row's line
579
+ */
580
+ processRow(row, quoted, line) {
581
+ const plan = this.requirePlan();
582
+ this.dataRows++;
583
+ if (plan.kind === "edges") {
584
+ this.processEdgeRow(plan, row, quoted, line);
585
+ }
586
+ else {
587
+ this.processNodeRow(plan, row, quoted, line);
588
+ }
589
+ }
590
+ /**
591
+ * Push one edge row: endpoints, weight, direction, then the attribute cells.
592
+ * @param plan - the edge plan
593
+ * @param row - the cells
594
+ * @param quoted - whether each cell was quoted
595
+ * @param line - the row's line
596
+ */
597
+ processEdgeRow(plan, row, quoted, line) {
598
+ const { report, sink, resolver, common } = this.state;
599
+ const { counts } = report;
600
+ if (row.length !== plan.width) {
601
+ report.error("validation-error", FIELD_COUNT_CODE, `line ${line}: ${row.length} field(s), the header has ${plan.width}`, { line });
602
+ counts.skippedEdges++;
603
+ return;
604
+ }
605
+ const sourceText = row[plan.source];
606
+ const targetText = row[plan.target];
607
+ const sourceMissing = isUnset(sourceText, quoted[plan.source]);
608
+ if (sourceMissing || isUnset(targetText, quoted[plan.target])) {
609
+ report.error("missing-value", MISSING_ENDPOINT_CODE, `line ${line}: blank ${sourceMissing ? "source" : "target"} cell`, { line });
610
+ counts.skippedEdges++;
611
+ return;
612
+ }
613
+ let kind = (this.state.commentDirected ?? common.defaultDirected) ? "directed" : "undirected";
614
+ if (plan.type >= 0) {
615
+ const parsed = parseKind(row[plan.type]);
616
+ if (parsed === null) {
617
+ report.error("validation-error", BAD_TYPE_CODE, `line ${line}: Type ${JSON.stringify(row[plan.type])} is not Directed, Undirected or Mutual`, { line });
618
+ counts.skippedEdges++;
619
+ return;
620
+ }
621
+ if (parsed !== undefined) {
622
+ kind = parsed;
623
+ }
624
+ }
625
+ const { where } = this;
626
+ where.line = line;
627
+ where.element = null;
628
+ const idText = plan.id >= 0 && !isUnset(row[plan.id], quoted[plan.id]) ? row[plan.id] : null;
629
+ if (idText !== null) {
630
+ if (this.edgeIds.has(idText)) {
631
+ report.error("validation-error", DUPLICATE_EDGE_ID_CODE, `line ${line}: edge id ${JSON.stringify(idText)} repeats an earlier row's; the row is skipped`, { line, element: idText });
632
+ counts.skippedEdges++;
633
+ return;
634
+ }
635
+ this.edgeIds.add(idText);
636
+ }
637
+ let edge;
638
+ try {
639
+ const source = this.coerce(sourceText);
640
+ const target = this.coerce(targetText);
641
+ const weight = plan.weight >= 0 ? parseWeightText(row[plan.weight]) : undefined;
642
+ if (!this.state.headerSet) {
643
+ this.state.headerSet = true;
644
+ resolver.setHeader(kind !== "undirected", where);
645
+ }
646
+ const sourceNew = sink.indexOf(source) === INVALID_INDEX;
647
+ const targetNew = source !== target && sink.indexOf(target) === INVALID_INDEX;
648
+ const before = sink.edgeCount;
649
+ edge = resolver.addEdge(source, target, kind, weight, where);
650
+ counts.edges += sink.edgeCount - before;
651
+ counts.nodes += (sourceNew ? 1 : 0) + (targetNew ? 1 : 0);
652
+ }
653
+ catch (err) {
654
+ where.element = idText ?? `${sourceText}->${targetText}`;
655
+ report.recordError(err, where);
656
+ counts.skippedEdges++;
657
+ return;
658
+ }
659
+ if (idText !== null) {
660
+ this.writeRole(this.idHandle, "edge", edge, idText, plan.names[plan.id], line);
661
+ }
662
+ if (plan.label >= 0 && !isUnset(row[plan.label], quoted[plan.label])) {
663
+ this.writeRole(this.labelHandle, "edge", edge, row[plan.label], plan.names[plan.label], line);
664
+ }
665
+ this.writeAttributes(plan, row, quoted, edge, line);
666
+ }
667
+ /**
668
+ * Push one node row: the id, then the label and attribute cells.
669
+ * @param plan - the node plan
670
+ * @param row - the cells
671
+ * @param quoted - whether each cell was quoted
672
+ * @param line - the row's line
673
+ */
674
+ processNodeRow(plan, row, quoted, line) {
675
+ const { report, sink } = this.state;
676
+ const { counts } = report;
677
+ const ordinal = this.nodeOrdinal++;
678
+ if (row.length !== plan.width) {
679
+ report.error("validation-error", FIELD_COUNT_CODE, `line ${line}: ${row.length} field(s), the header has ${plan.width}`, { line });
680
+ counts.skippedNodes++;
681
+ return;
682
+ }
683
+ const idText = plan.id >= 0 ? row[plan.id] : String(ordinal);
684
+ if (plan.id >= 0 && isUnset(idText, quoted[plan.id])) {
685
+ report.error("missing-value", MISSING_ID_CODE, `line ${line}: blank id cell`, { line });
686
+ counts.skippedNodes++;
687
+ return;
688
+ }
689
+ const { where } = this;
690
+ where.line = line;
691
+ where.element = idText;
692
+ let index;
693
+ try {
694
+ const id = plan.id >= 0 ? this.coerce(idText) : ordinal;
695
+ if (sink.indexOf(id) !== INVALID_INDEX) {
696
+ report.warning("merged", DUPLICATE_NODE_CODE, `line ${line}: node ${JSON.stringify(id)} already exists; its attributes are overwritten`, where);
697
+ }
698
+ else {
699
+ counts.nodes++;
700
+ }
701
+ index = sink.addNode(id);
702
+ }
703
+ catch (err) {
704
+ report.recordError(err, where);
705
+ counts.skippedNodes++;
706
+ return;
707
+ }
708
+ if (plan.label >= 0 && !isUnset(row[plan.label], quoted[plan.label])) {
709
+ this.writeRole(this.labelHandle, "node", index, row[plan.label], plan.names[plan.label], line);
710
+ }
711
+ this.writeAttributes(plan, row, quoted, index, line);
712
+ }
713
+ /**
714
+ * Coerce an id cell, reporting a merge under `ids: "number"`.
715
+ * @param text - the cell text
716
+ * @returns the id
717
+ */
718
+ coerce(text) {
719
+ const id = this.state.coercer.text(text);
720
+ const merge = this.state.coercer.lastMerge;
721
+ if (merge !== null) {
722
+ this.state.report.warnOnce("coercion", ID_MERGED_CODE, `id ${JSON.stringify(merge.text)} merged with ${JSON.stringify(merge.previousText)} as ${merge.id} under ids: "number"`, this.where);
723
+ }
724
+ return id;
725
+ }
726
+ /**
727
+ * Write a role column cell (edge id, label).
728
+ * @param handle - the column handle
729
+ * @param domain - node or edge
730
+ * @param row - the node or edge index
731
+ * @param text - the cell text
732
+ * @param name - the column name, for issues
733
+ * @param line - the row's line
734
+ */
735
+ writeRole(handle, domain, row, text, name, line) {
736
+ try {
737
+ if (domain === "node") {
738
+ this.state.sink.setNodeValue(handle, row, text);
739
+ }
740
+ else {
741
+ this.state.sink.setEdgeValue(handle, row, text);
742
+ }
743
+ }
744
+ catch (err) {
745
+ this.state.report.recordError(err, { line, element: name });
746
+ }
747
+ }
748
+ /**
749
+ * Write the attribute cells of a row: an unquoted blank cell is unset, a quoted one (`""`,
750
+ * `" "`) is the text it holds.
751
+ * @param plan - the plan
752
+ * @param row - the cells
753
+ * @param quoted - whether each cell was quoted
754
+ * @param index - the node or edge index
755
+ * @param line - the row's line
756
+ */
757
+ writeAttributes(plan, row, quoted, index, line) {
758
+ for (const k of plan.attributes) {
759
+ const text = row[k];
760
+ if (isUnset(text, quoted[k])) {
761
+ continue;
762
+ }
763
+ const writer = this.writers[k];
764
+ if (writer === null) {
765
+ continue;
766
+ }
767
+ try {
768
+ writer.write(index, text);
769
+ }
770
+ catch (err) {
771
+ this.state.report.recordError(err, { line, element: writer.name });
772
+ }
773
+ }
774
+ }
775
+ }
776
+ const HEAD_BYTES = 4096;
777
+ const OTHER_FORMAT = /^\s*(<|[[{]|(strict\s+)?(di)?graph(\s+\S+)?\s*\{|\*vertices|creator\b|graph\s*\[)/i;
778
+ /**
779
+ * Sniff confidence for the registry: 0 for XML, JSON, GML, DOT and Pajek openings; otherwise a
780
+ * delimited first row with endpoint headers is 0.9, with an id header 0.6, any consistently
781
+ * delimited rows 0.3, a single column 0.
782
+ * @param head - the first bytes of the input
783
+ * @returns a confidence in 0..1
784
+ */
785
+ function sniff(head) {
786
+ const text = new TextDecoder("utf-8").decode(head.subarray(0, HEAD_BYTES));
787
+ const body = text.startsWith(String.fromCharCode(0xfeff)) ? text.slice(1) : text;
788
+ if (body.trim().length === 0 || OTHER_FORMAT.test(body)) {
789
+ return 0;
790
+ }
791
+ const newline = sniffNewline(body);
792
+ const delimiter = sniffDelimiter(body, newline);
793
+ if (delimiter === null) {
794
+ return 0;
795
+ }
796
+ const end = body.indexOf(newline);
797
+ const firstLine = end < 0 ? body : body.slice(0, end);
798
+ const names = headerNames(firstLine.replace(/\r$/, "").split(delimiter));
799
+ if (findColumn(names, SOURCE_NAMES) >= 0 && findColumn(names, TARGET_NAMES) >= 0) {
800
+ return 0.9;
801
+ }
802
+ if (findColumn(names, ID_NAMES) >= 0) {
803
+ return 0.6;
804
+ }
805
+ return 0.3;
806
+ }
807
+ /**
808
+ * Import a CSV edge table (and an optional node table) into a sink.
809
+ * @param input - the edge table (or, with `table: "nodes"` or a header without endpoints, a node table)
810
+ * @param sink - the sink
811
+ * @param options - CSV and common options
812
+ * @returns the report; ImportError with the partial report when the import aborts
813
+ */
814
+ async function importCsv(input, sink, options) {
815
+ const common = resolveImportOptions(options, { ids: "canonical", defaultDirected: true, weightFrom: "weight" });
816
+ const csv = resolveCsvOptions(options);
817
+ const report = new ImportReportBuilder("csv", common.errorLimit);
818
+ reportSinkOptions(sink, options, report);
819
+ reportUnusedOptions(options, report, csv.table === "nodes" || csv.nodes !== null ? USED_OPTIONS_WITH_NODES : USED_OPTIONS);
820
+ const state = {
821
+ sink,
822
+ report,
823
+ common,
824
+ csv,
825
+ weightFromExplicit: typeof options?.weightFrom === "string",
826
+ coercer: new IdCoercer(common.ids),
827
+ resolver: new DirectionResolver(sink, report, common.onMixedDirection),
828
+ headerSet: false,
829
+ commentDirected: null,
830
+ };
831
+ if (csv.nodes !== null) {
832
+ await new TableReader(state, csv.nodes, "nodes", false).read();
833
+ }
834
+ await new TableReader(state, input, csv.table, true).read();
835
+ if (state.coercer.mergeCount > 1) {
836
+ report.warning("coercion", ID_MERGED_CODE, `${state.coercer.mergeCount} id cell(s) merged into ids other cells already produced under ids: "number"`);
837
+ }
838
+ throwIfAborted(common.signal);
839
+ return report.finish();
840
+ }
841
+ /** The CSV / TSV importer plugin (subpath `@graphty/graph-io/csv`). */
842
+ export const csvImporter = Object.freeze({
843
+ format: "csv",
844
+ extensions: Object.freeze([".csv", ".tsv", ".edges", ".edgelist"]),
845
+ mimeTypes: Object.freeze(["text/csv", "text/tab-separated-values", "text/plain"]),
846
+ sniff,
847
+ import: importCsv,
848
+ });
849
+ //# sourceMappingURL=importer.js.map