@graphty/graph-io 0.0.0 → 0.2.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (339) hide show
  1. package/LICENSE +21 -0
  2. package/README.md +250 -28
  3. package/dist/chunks/children-CL3Cy0ez.js +238 -0
  4. package/dist/chunks/children-CL3Cy0ez.js.map +1 -0
  5. package/dist/chunks/escape-DyI8JofU.js +938 -0
  6. package/dist/chunks/escape-DyI8JofU.js.map +1 -0
  7. package/dist/chunks/importer-CQnJuWJw.js +2987 -0
  8. package/dist/chunks/importer-CQnJuWJw.js.map +1 -0
  9. package/dist/chunks/importer-CpCpfbxr.js +2015 -0
  10. package/dist/chunks/importer-CpCpfbxr.js.map +1 -0
  11. package/dist/chunks/importer-DbnGYr3_.js +2342 -0
  12. package/dist/chunks/importer-DbnGYr3_.js.map +1 -0
  13. package/dist/chunks/importer-GozH8DkN.js +3050 -0
  14. package/dist/chunks/importer-GozH8DkN.js.map +1 -0
  15. package/dist/chunks/records-CGpxszm1.js +605 -0
  16. package/dist/chunks/records-CGpxszm1.js.map +1 -0
  17. package/dist/chunks/text-CajMdVFy.js +189 -0
  18. package/dist/chunks/text-CajMdVFy.js.map +1 -0
  19. package/dist/chunks/writer-DxSKC7TL.js +2842 -0
  20. package/dist/chunks/writer-DxSKC7TL.js.map +1 -0
  21. package/dist/csv.d.ts +1 -0
  22. package/dist/csv.js +1702 -0
  23. package/dist/csv.js.map +1 -0
  24. package/dist/dot.d.ts +1 -0
  25. package/dist/dot.js +8 -0
  26. package/dist/dot.js.map +1 -0
  27. package/dist/gexf.d.ts +1 -0
  28. package/dist/gexf.js +3466 -0
  29. package/dist/gexf.js.map +1 -0
  30. package/dist/gml.d.ts +1 -0
  31. package/dist/gml.js +2647 -0
  32. package/dist/gml.js.map +1 -0
  33. package/dist/graph-io.d.ts +1 -0
  34. package/dist/graph-io.js +790 -0
  35. package/dist/graph-io.js.map +1 -0
  36. package/dist/graphml.d.ts +1 -0
  37. package/dist/graphml.js +8 -0
  38. package/dist/graphml.js.map +1 -0
  39. package/dist/json.d.ts +1 -0
  40. package/dist/json.js +11 -0
  41. package/dist/json.js.map +1 -0
  42. package/dist/neo4j.d.ts +1 -0
  43. package/dist/neo4j.js +2046 -0
  44. package/dist/neo4j.js.map +1 -0
  45. package/dist/pajek.d.ts +1 -0
  46. package/dist/pajek.js +8 -0
  47. package/dist/pajek.js.map +1 -0
  48. package/dist/src/children.d.ts +134 -0
  49. package/dist/src/children.d.ts.map +1 -0
  50. package/dist/src/children.js +274 -0
  51. package/dist/src/children.js.map +1 -0
  52. package/dist/src/common/attributes.d.ts +229 -0
  53. package/dist/src/common/attributes.d.ts.map +1 -0
  54. package/dist/src/common/attributes.js +368 -0
  55. package/dist/src/common/attributes.js.map +1 -0
  56. package/dist/src/common/codes.d.ts +105 -0
  57. package/dist/src/common/codes.d.ts.map +1 -0
  58. package/dist/src/common/codes.js +107 -0
  59. package/dist/src/common/codes.js.map +1 -0
  60. package/dist/src/common/declared-types.d.ts +84 -0
  61. package/dist/src/common/declared-types.d.ts.map +1 -0
  62. package/dist/src/common/declared-types.js +326 -0
  63. package/dist/src/common/declared-types.js.map +1 -0
  64. package/dist/src/common/direction.d.ts +206 -0
  65. package/dist/src/common/direction.d.ts.map +1 -0
  66. package/dist/src/common/direction.js +370 -0
  67. package/dist/src/common/direction.js.map +1 -0
  68. package/dist/src/common/escape.d.ts +92 -0
  69. package/dist/src/common/escape.d.ts.map +1 -0
  70. package/dist/src/common/escape.js +212 -0
  71. package/dist/src/common/escape.js.map +1 -0
  72. package/dist/src/common/export.d.ts +249 -0
  73. package/dist/src/common/export.d.ts.map +1 -0
  74. package/dist/src/common/export.js +594 -0
  75. package/dist/src/common/export.js.map +1 -0
  76. package/dist/src/common/format.d.ts +59 -0
  77. package/dist/src/common/format.d.ts.map +1 -0
  78. package/dist/src/common/format.js +106 -0
  79. package/dist/src/common/format.js.map +1 -0
  80. package/dist/src/common/ids.d.ts +83 -0
  81. package/dist/src/common/ids.d.ts.map +1 -0
  82. package/dist/src/common/ids.js +158 -0
  83. package/dist/src/common/ids.js.map +1 -0
  84. package/dist/src/common/input.d.ts +100 -0
  85. package/dist/src/common/input.d.ts.map +1 -0
  86. package/dist/src/common/input.js +335 -0
  87. package/dist/src/common/input.js.map +1 -0
  88. package/dist/src/common/lists.d.ts +34 -0
  89. package/dist/src/common/lists.d.ts.map +1 -0
  90. package/dist/src/common/lists.js +185 -0
  91. package/dist/src/common/lists.js.map +1 -0
  92. package/dist/src/common/options.d.ts +108 -0
  93. package/dist/src/common/options.d.ts.map +1 -0
  94. package/dist/src/common/options.js +265 -0
  95. package/dist/src/common/options.js.map +1 -0
  96. package/dist/src/common/report.d.ts +187 -0
  97. package/dist/src/common/report.d.ts.map +1 -0
  98. package/dist/src/common/report.js +274 -0
  99. package/dist/src/common/report.js.map +1 -0
  100. package/dist/src/common/temporal.d.ts +71 -0
  101. package/dist/src/common/temporal.d.ts.map +1 -0
  102. package/dist/src/common/temporal.js +266 -0
  103. package/dist/src/common/temporal.js.map +1 -0
  104. package/dist/src/common/text.d.ts +104 -0
  105. package/dist/src/common/text.d.ts.map +1 -0
  106. package/dist/src/common/text.js +255 -0
  107. package/dist/src/common/text.js.map +1 -0
  108. package/dist/src/common/weights.d.ts +77 -0
  109. package/dist/src/common/weights.d.ts.map +1 -0
  110. package/dist/src/common/weights.js +156 -0
  111. package/dist/src/common/weights.js.map +1 -0
  112. package/dist/src/common/writer.d.ts +51 -0
  113. package/dist/src/common/writer.d.ts.map +1 -0
  114. package/dist/src/common/writer.js +108 -0
  115. package/dist/src/common/writer.js.map +1 -0
  116. package/dist/src/common/xml.d.ts +245 -0
  117. package/dist/src/common/xml.d.ts.map +1 -0
  118. package/dist/src/common/xml.js +942 -0
  119. package/dist/src/common/xml.js.map +1 -0
  120. package/dist/src/formats/csv/exporter.d.ts +70 -0
  121. package/dist/src/formats/csv/exporter.d.ts.map +1 -0
  122. package/dist/src/formats/csv/exporter.js +682 -0
  123. package/dist/src/formats/csv/exporter.js.map +1 -0
  124. package/dist/src/formats/csv/header.d.ts +66 -0
  125. package/dist/src/formats/csv/header.d.ts.map +1 -0
  126. package/dist/src/formats/csv/header.js +152 -0
  127. package/dist/src/formats/csv/header.js.map +1 -0
  128. package/dist/src/formats/csv/importer.d.ts +82 -0
  129. package/dist/src/formats/csv/importer.d.ts.map +1 -0
  130. package/dist/src/formats/csv/importer.js +849 -0
  131. package/dist/src/formats/csv/importer.js.map +1 -0
  132. package/dist/src/formats/csv/index.d.ts +60 -0
  133. package/dist/src/formats/csv/index.d.ts.map +1 -0
  134. package/dist/src/formats/csv/index.js +63 -0
  135. package/dist/src/formats/csv/index.js.map +1 -0
  136. package/dist/src/formats/csv/records.d.ts +188 -0
  137. package/dist/src/formats/csv/records.d.ts.map +1 -0
  138. package/dist/src/formats/csv/records.js +702 -0
  139. package/dist/src/formats/csv/records.js.map +1 -0
  140. package/dist/src/formats/csv/values.d.ts +105 -0
  141. package/dist/src/formats/csv/values.d.ts.map +1 -0
  142. package/dist/src/formats/csv/values.js +192 -0
  143. package/dist/src/formats/csv/values.js.map +1 -0
  144. package/dist/src/formats/dot/exporter.d.ts +52 -0
  145. package/dist/src/formats/dot/exporter.d.ts.map +1 -0
  146. package/dist/src/formats/dot/exporter.js +836 -0
  147. package/dist/src/formats/dot/exporter.js.map +1 -0
  148. package/dist/src/formats/dot/importer.d.ts +102 -0
  149. package/dist/src/formats/dot/importer.d.ts.map +1 -0
  150. package/dist/src/formats/dot/importer.js +1291 -0
  151. package/dist/src/formats/dot/importer.js.map +1 -0
  152. package/dist/src/formats/dot/index.d.ts +7 -0
  153. package/dist/src/formats/dot/index.d.ts.map +1 -0
  154. package/dist/src/formats/dot/index.js +7 -0
  155. package/dist/src/formats/dot/index.js.map +1 -0
  156. package/dist/src/formats/dot/names.d.ts +29 -0
  157. package/dist/src/formats/dot/names.d.ts.map +1 -0
  158. package/dist/src/formats/dot/names.js +28 -0
  159. package/dist/src/formats/dot/names.js.map +1 -0
  160. package/dist/src/formats/dot/tokenizer.d.ts +114 -0
  161. package/dist/src/formats/dot/tokenizer.d.ts.map +1 -0
  162. package/dist/src/formats/dot/tokenizer.js +341 -0
  163. package/dist/src/formats/dot/tokenizer.js.map +1 -0
  164. package/dist/src/formats/gexf/exporter.d.ts +56 -0
  165. package/dist/src/formats/gexf/exporter.d.ts.map +1 -0
  166. package/dist/src/formats/gexf/exporter.js +1395 -0
  167. package/dist/src/formats/gexf/exporter.js.map +1 -0
  168. package/dist/src/formats/gexf/importer.d.ts +73 -0
  169. package/dist/src/formats/gexf/importer.d.ts.map +1 -0
  170. package/dist/src/formats/gexf/importer.js +1880 -0
  171. package/dist/src/formats/gexf/importer.js.map +1 -0
  172. package/dist/src/formats/gexf/index.d.ts +96 -0
  173. package/dist/src/formats/gexf/index.d.ts.map +1 -0
  174. package/dist/src/formats/gexf/index.js +97 -0
  175. package/dist/src/formats/gexf/index.js.map +1 -0
  176. package/dist/src/formats/gexf/schema.d.ts +135 -0
  177. package/dist/src/formats/gexf/schema.d.ts.map +1 -0
  178. package/dist/src/formats/gexf/schema.js +323 -0
  179. package/dist/src/formats/gexf/schema.js.map +1 -0
  180. package/dist/src/formats/gml/exporter.d.ts +69 -0
  181. package/dist/src/formats/gml/exporter.d.ts.map +1 -0
  182. package/dist/src/formats/gml/exporter.js +1093 -0
  183. package/dist/src/formats/gml/exporter.js.map +1 -0
  184. package/dist/src/formats/gml/importer.d.ts +66 -0
  185. package/dist/src/formats/gml/importer.d.ts.map +1 -0
  186. package/dist/src/formats/gml/importer.js +1331 -0
  187. package/dist/src/formats/gml/importer.js.map +1 -0
  188. package/dist/src/formats/gml/index.d.ts +85 -0
  189. package/dist/src/formats/gml/index.d.ts.map +1 -0
  190. package/dist/src/formats/gml/index.js +88 -0
  191. package/dist/src/formats/gml/index.js.map +1 -0
  192. package/dist/src/formats/gml/syntax.d.ts +186 -0
  193. package/dist/src/formats/gml/syntax.d.ts.map +1 -0
  194. package/dist/src/formats/gml/syntax.js +467 -0
  195. package/dist/src/formats/gml/syntax.js.map +1 -0
  196. package/dist/src/formats/graphml/constants.d.ts +169 -0
  197. package/dist/src/formats/graphml/constants.d.ts.map +1 -0
  198. package/dist/src/formats/graphml/constants.js +165 -0
  199. package/dist/src/formats/graphml/constants.js.map +1 -0
  200. package/dist/src/formats/graphml/exporter.d.ts +34 -0
  201. package/dist/src/formats/graphml/exporter.d.ts.map +1 -0
  202. package/dist/src/formats/graphml/exporter.js +1176 -0
  203. package/dist/src/formats/graphml/exporter.js.map +1 -0
  204. package/dist/src/formats/graphml/importer.d.ts +31 -0
  205. package/dist/src/formats/graphml/importer.d.ts.map +1 -0
  206. package/dist/src/formats/graphml/importer.js +1607 -0
  207. package/dist/src/formats/graphml/importer.js.map +1 -0
  208. package/dist/src/formats/graphml/index.d.ts +8 -0
  209. package/dist/src/formats/graphml/index.d.ts.map +1 -0
  210. package/dist/src/formats/graphml/index.js +8 -0
  211. package/dist/src/formats/graphml/index.js.map +1 -0
  212. package/dist/src/formats/graphml/tree.d.ts +72 -0
  213. package/dist/src/formats/graphml/tree.d.ts.map +1 -0
  214. package/dist/src/formats/graphml/tree.js +290 -0
  215. package/dist/src/formats/graphml/tree.js.map +1 -0
  216. package/dist/src/formats/json/dialect.d.ts +125 -0
  217. package/dist/src/formats/json/dialect.d.ts.map +1 -0
  218. package/dist/src/formats/json/dialect.js +262 -0
  219. package/dist/src/formats/json/dialect.js.map +1 -0
  220. package/dist/src/formats/json/exporter.d.ts +89 -0
  221. package/dist/src/formats/json/exporter.d.ts.map +1 -0
  222. package/dist/src/formats/json/exporter.js +1358 -0
  223. package/dist/src/formats/json/exporter.js.map +1 -0
  224. package/dist/src/formats/json/importer.d.ts +108 -0
  225. package/dist/src/formats/json/importer.d.ts.map +1 -0
  226. package/dist/src/formats/json/importer.js +1838 -0
  227. package/dist/src/formats/json/importer.js.map +1 -0
  228. package/dist/src/formats/json/index.d.ts +8 -0
  229. package/dist/src/formats/json/index.d.ts.map +1 -0
  230. package/dist/src/formats/json/index.js +8 -0
  231. package/dist/src/formats/json/index.js.map +1 -0
  232. package/dist/src/formats/neo4j/exporter.d.ts +68 -0
  233. package/dist/src/formats/neo4j/exporter.d.ts.map +1 -0
  234. package/dist/src/formats/neo4j/exporter.js +1055 -0
  235. package/dist/src/formats/neo4j/exporter.js.map +1 -0
  236. package/dist/src/formats/neo4j/header.d.ts +52 -0
  237. package/dist/src/formats/neo4j/header.d.ts.map +1 -0
  238. package/dist/src/formats/neo4j/header.js +131 -0
  239. package/dist/src/formats/neo4j/header.js.map +1 -0
  240. package/dist/src/formats/neo4j/importer.d.ts +73 -0
  241. package/dist/src/formats/neo4j/importer.d.ts.map +1 -0
  242. package/dist/src/formats/neo4j/importer.js +932 -0
  243. package/dist/src/formats/neo4j/importer.js.map +1 -0
  244. package/dist/src/formats/neo4j/index.d.ts +79 -0
  245. package/dist/src/formats/neo4j/index.d.ts.map +1 -0
  246. package/dist/src/formats/neo4j/index.js +83 -0
  247. package/dist/src/formats/neo4j/index.js.map +1 -0
  248. package/dist/src/formats/pajek/exporter.d.ts +58 -0
  249. package/dist/src/formats/pajek/exporter.d.ts.map +1 -0
  250. package/dist/src/formats/pajek/exporter.js +825 -0
  251. package/dist/src/formats/pajek/exporter.js.map +1 -0
  252. package/dist/src/formats/pajek/importer.d.ts +88 -0
  253. package/dist/src/formats/pajek/importer.d.ts.map +1 -0
  254. package/dist/src/formats/pajek/importer.js +1047 -0
  255. package/dist/src/formats/pajek/importer.js.map +1 -0
  256. package/dist/src/formats/pajek/index.d.ts +7 -0
  257. package/dist/src/formats/pajek/index.d.ts.map +1 -0
  258. package/dist/src/formats/pajek/index.js +7 -0
  259. package/dist/src/formats/pajek/index.js.map +1 -0
  260. package/dist/src/formats/pajek/syntax.d.ts +112 -0
  261. package/dist/src/formats/pajek/syntax.d.ts.map +1 -0
  262. package/dist/src/formats/pajek/syntax.js +269 -0
  263. package/dist/src/formats/pajek/syntax.js.map +1 -0
  264. package/dist/src/index.d.ts +35 -0
  265. package/dist/src/index.d.ts.map +1 -0
  266. package/dist/src/index.js +39 -0
  267. package/dist/src/index.js.map +1 -0
  268. package/dist/src/registry.d.ts +207 -0
  269. package/dist/src/registry.d.ts.map +1 -0
  270. package/dist/src/registry.js +481 -0
  271. package/dist/src/registry.js.map +1 -0
  272. package/dist/src/sniff.d.ts +104 -0
  273. package/dist/src/sniff.d.ts.map +1 -0
  274. package/dist/src/sniff.js +357 -0
  275. package/dist/src/sniff.js.map +1 -0
  276. package/dist/src/types.d.ts +238 -0
  277. package/dist/src/types.d.ts.map +1 -0
  278. package/dist/src/types.js +29 -0
  279. package/dist/src/types.js.map +1 -0
  280. package/dist/tsconfig.build.tsbuildinfo +1 -0
  281. package/package.json +122 -7
  282. package/src/children.ts +335 -0
  283. package/src/common/attributes.ts +520 -0
  284. package/src/common/codes.ts +153 -0
  285. package/src/common/declared-types.ts +374 -0
  286. package/src/common/direction.ts +518 -0
  287. package/src/common/escape.ts +231 -0
  288. package/src/common/export.ts +817 -0
  289. package/src/common/format.ts +111 -0
  290. package/src/common/ids.ts +176 -0
  291. package/src/common/input.ts +378 -0
  292. package/src/common/lists.ts +196 -0
  293. package/src/common/options.ts +377 -0
  294. package/src/common/report.ts +352 -0
  295. package/src/common/temporal.ts +302 -0
  296. package/src/common/text.ts +294 -0
  297. package/src/common/weights.ts +202 -0
  298. package/src/common/writer.ts +123 -0
  299. package/src/common/xml.ts +1053 -0
  300. package/src/formats/csv/exporter.ts +894 -0
  301. package/src/formats/csv/header.ts +172 -0
  302. package/src/formats/csv/importer.ts +1104 -0
  303. package/src/formats/csv/index.ts +88 -0
  304. package/src/formats/csv/records.ts +813 -0
  305. package/src/formats/csv/values.ts +224 -0
  306. package/src/formats/dot/exporter.ts +1014 -0
  307. package/src/formats/dot/importer.ts +1549 -0
  308. package/src/formats/dot/index.ts +7 -0
  309. package/src/formats/dot/names.ts +40 -0
  310. package/src/formats/dot/tokenizer.ts +384 -0
  311. package/src/formats/gexf/exporter.ts +1696 -0
  312. package/src/formats/gexf/importer.ts +2333 -0
  313. package/src/formats/gexf/index.ts +142 -0
  314. package/src/formats/gexf/schema.ts +361 -0
  315. package/src/formats/gml/exporter.ts +1404 -0
  316. package/src/formats/gml/importer.ts +1591 -0
  317. package/src/formats/gml/index.ts +128 -0
  318. package/src/formats/gml/syntax.ts +545 -0
  319. package/src/formats/graphml/constants.ts +225 -0
  320. package/src/formats/graphml/exporter.ts +1458 -0
  321. package/src/formats/graphml/importer.ts +2027 -0
  322. package/src/formats/graphml/index.ts +8 -0
  323. package/src/formats/graphml/tree.ts +318 -0
  324. package/src/formats/json/dialect.ts +317 -0
  325. package/src/formats/json/exporter.ts +1616 -0
  326. package/src/formats/json/importer.ts +2271 -0
  327. package/src/formats/json/index.ts +8 -0
  328. package/src/formats/neo4j/exporter.ts +1287 -0
  329. package/src/formats/neo4j/header.ts +156 -0
  330. package/src/formats/neo4j/importer.ts +1220 -0
  331. package/src/formats/neo4j/index.ts +116 -0
  332. package/src/formats/pajek/exporter.ts +1000 -0
  333. package/src/formats/pajek/importer.ts +1311 -0
  334. package/src/formats/pajek/index.ts +7 -0
  335. package/src/formats/pajek/syntax.ts +307 -0
  336. package/src/index.ts +244 -0
  337. package/src/registry.ts +617 -0
  338. package/src/sniff.ts +397 -0
  339. package/src/types.ts +262 -0
@@ -0,0 +1,1104 @@
1
+ /**
2
+ * The CSV / TSV importer (design sections 8.4 and 8.6; research note 07 section 2.6): a streaming
3
+ * edge-list reader for the generic (`source,target[,weight,...]`), Gephi (`Source,Target,Type,Id,
4
+ * Label,Weight,...`) and headerless (`u v [w]`) dialects, with an optional node table merged by id.
5
+ *
6
+ * - The delimiter is sniffed from a preview unless given; LF, CRLF and lone-CR files all read.
7
+ * - The first row is a header when it holds a known column name or when it is all text over a
8
+ * numeric second row (`header: "auto"`); a headerless file is positional: source, target,
9
+ * weight (when `weightFrom` is not null), then `column4`... as attributes.
10
+ * - Endpoint and node ids are text cells coerced by ONE rule per import call (`ids`, default
11
+ * "canonical": `"1"` becomes the number 1, `"01"` stays a string) over the node table and the
12
+ * edge table alike, so they agree across the two files.
13
+ * - Direction is per row in the Gephi dialect (`Type` = Directed / Undirected / Mutual, blank =
14
+ * `defaultDirected`) and `defaultDirected` (true) otherwise; the first edge row sets the sink's
15
+ * direction and later rows that differ go through `onMixedDirection` (expansion by default).
16
+ * - The weight column (`weightFrom`, default "weight", matched case-insensitively) is parsed per
17
+ * row; a blank cell means "no weight given" and the edge is pushed without one.
18
+ * - Every other column is an attribute: cells are parsed by the fixed lexical grammar of design
19
+ * section 5.1 and the sink infers the column dtype (widening per column, never per cell); an
20
+ * all-text column of low cardinality becomes a dict (design section 5.4); an `id` column of the
21
+ * edge table is the edge id (role id, unique); a `label` column is the label (role label).
22
+ * - Per-row problems (wrong field count, blank endpoint, invalid weight, bad Type, refused id)
23
+ * are recorded and the row skipped; the import aborts with ImportError once `errorLimit` is
24
+ * exceeded, on a malformed or unterminated quoted field, on an empty input and on a header
25
+ * without endpoint (or id) columns.
26
+ */
27
+
28
+ import {
29
+ type ColumnDecl,
30
+ type ColumnHandle,
31
+ GraphFormatError,
32
+ type GraphSink,
33
+ INVALID_INDEX,
34
+ type NodeId,
35
+ } from "@graphty/graph-format";
36
+
37
+ import { declareResolved, RENAMED_CODE, uniqueColumnName } from "../../common/attributes.js";
38
+ import {
39
+ DUPLICATE_EDGE_ID_CODE as SHARED_DUPLICATE_EDGE_ID_CODE,
40
+ DUPLICATE_NODE_CODE as SHARED_DUPLICATE_NODE_CODE,
41
+ EMPTY_INPUT_CODE as SHARED_EMPTY_INPUT_CODE,
42
+ ID_MERGED_CODE as SHARED_ID_MERGED_CODE,
43
+ MISSING_ENDPOINT_CODE as SHARED_MISSING_ENDPOINT_CODE,
44
+ MISSING_ID_CODE as SHARED_MISSING_ID_CODE,
45
+ ROLE_TAKEN_CODE as SHARED_ROLE_TAKEN_CODE,
46
+ } from "../../common/codes.js";
47
+ import { DirectionResolver, type EdgeKind } from "../../common/direction.js";
48
+ import { IdCoercer } from "../../common/ids.js";
49
+ import { throwIfAborted } from "../../common/input.js";
50
+ import {
51
+ reportSinkOptions,
52
+ reportUnusedOptions,
53
+ type ResolvedImportOptions,
54
+ resolveImportOptions,
55
+ } from "../../common/options.js";
56
+ import { ImportReportBuilder } from "../../common/report.js";
57
+ import { parseWeightText } from "../../common/weights.js";
58
+ import { type CommonImportOptions, type GraphImporter, type ImportInput, type ImportReport } from "../../types.js";
59
+ import {
60
+ type CsvColumnRef,
61
+ EDGE_ID_NAMES,
62
+ findColumn,
63
+ headerNames,
64
+ ID_NAMES,
65
+ LABEL_NAMES,
66
+ looksLikeHeader,
67
+ positionalNames,
68
+ resolveColumnRef,
69
+ SOURCE_NAMES,
70
+ TARGET_NAMES,
71
+ TYPE_NAME,
72
+ } from "./header.js";
73
+ import { type CsvReaderOptions, CsvRecordReader, sniffDelimiter, sniffNewline } from "./records.js";
74
+ import { InferredColumn } from "./values.js";
75
+
76
+ /** The format-specific options of the CSV importer. */
77
+ export interface CsvImportOptions {
78
+ /** The field delimiter; sniffed from the first rows when omitted (`,`, tab, `;`, `|`, space). */
79
+ delimiter?: string | undefined;
80
+ /** Whether the first row is a header; "auto" (default) decides from its content. */
81
+ header?: boolean | "auto" | undefined;
82
+ /**
83
+ * What the input is: an edge table, a node table, or "auto" (default): an edge table when
84
+ * source and target columns resolve, a node table when only an id column does.
85
+ */
86
+ table?: "edges" | "nodes" | "auto" | undefined;
87
+ /** The source column, by name or 0-based position; resolved from the header by default. */
88
+ sourceColumn?: CsvColumnRef | undefined;
89
+ /** The target column, by name or 0-based position; resolved from the header by default. */
90
+ targetColumn?: CsvColumnRef | undefined;
91
+ /**
92
+ * The per-row direction column (Directed / Undirected / Mutual); by default the exact `Type`
93
+ * column of a Gephi table (exact `Source` and `Target` headers); null reads no such column.
94
+ */
95
+ typeColumn?: CsvColumnRef | null | undefined;
96
+ /** The id column of a node table, by name or position; resolved from the header by default. */
97
+ idColumn?: CsvColumnRef | undefined;
98
+ /** A node table read before the edges: its ids become nodes and its other columns node attributes. */
99
+ nodes?: ImportInput | undefined;
100
+ }
101
+
102
+ /** Issue code: the input holds no header row at all. */
103
+ export const EMPTY_INPUT_CODE = SHARED_EMPTY_INPUT_CODE;
104
+ /** Issue code: the header names no source / target (or, for a node table, no id) column. */
105
+ export const NO_ENDPOINT_COLUMNS_CODE = "E_CSV_NO_ENDPOINT_COLUMNS";
106
+ /** Issue code: a node table without an id column. */
107
+ export const NO_ID_COLUMN_CODE = "E_CSV_NO_ID_COLUMN";
108
+ /** Issue code: a row with a different number of fields than the header. */
109
+ export const FIELD_COUNT_CODE = "E_CSV_FIELD_COUNT";
110
+ /** Issue code: an edge row with a blank source or target cell. */
111
+ export const MISSING_ENDPOINT_CODE = SHARED_MISSING_ENDPOINT_CODE;
112
+ /** Issue code: a node row with a blank id cell. */
113
+ export const MISSING_ID_CODE = SHARED_MISSING_ID_CODE;
114
+ /** Issue code: a Type cell that is not Directed, Undirected or Mutual. */
115
+ export const BAD_TYPE_CODE = "E_CSV_BAD_TYPE";
116
+ /** Issue code: the table has a header and no data rows. */
117
+ export const NO_DATA_ROWS_CODE = "W_CSV_NO_DATA_ROWS";
118
+ /** Issue code: a node table row repeats an id; its attributes overwrite the earlier row's. */
119
+ export const DUPLICATE_NODE_CODE = SHARED_DUPLICATE_NODE_CODE;
120
+ /** Issue code: two distinct id cells became one id under `ids: "number"` (design section 4.1). */
121
+ export const ID_MERGED_CODE = SHARED_ID_MERGED_CODE;
122
+ /** Issue code: an explicitly named weight column the file does not have. */
123
+ export const COLUMN_MISSING_CODE = "W_CSV_COLUMN_MISSING";
124
+ /** Issue code: a column whose role (id, label) is already held by another column of the sink. */
125
+ export const ROLE_TAKEN_CODE = SHARED_ROLE_TAKEN_CODE;
126
+ /** Issue code: a repeated edge id (the column is unique); the edge is skipped. */
127
+ export const DUPLICATE_EDGE_ID_CODE = SHARED_DUPLICATE_EDGE_ID_CODE;
128
+
129
+ const TABLE_MODES: ReadonlySet<string> = new Set(["edges", "nodes", "auto"]);
130
+
131
+ /** The common options an edge-table import reads (the rest is reported by reportUnusedOptions). */
132
+ const USED_OPTIONS: ReadonlySet<keyof CommonImportOptions> = new Set<keyof CommonImportOptions>([
133
+ "ids",
134
+ "addMissingNodes",
135
+ "duplicateEdges",
136
+ "selfLoops",
137
+ "onMixedDirection",
138
+ "defaultDirected",
139
+ "weightFrom",
140
+ "weightDtype",
141
+ "errorLimit",
142
+ "signal",
143
+ "onProgress",
144
+ ]);
145
+
146
+ /** The common options an import with a node table reads: nodeIdFrom applies to the node table. */
147
+ const USED_OPTIONS_WITH_NODES: ReadonlySet<keyof CommonImportOptions> = new Set<keyof CommonImportOptions>([
148
+ ...USED_OPTIONS,
149
+ "nodeIdFrom",
150
+ ]);
151
+ const BAD_DELIMITERS: ReadonlySet<string> = new Set(['"', "\n", "\r"]);
152
+
153
+ /** The CSV options with defaults applied. */
154
+ interface ResolvedCsvOptions {
155
+ readonly delimiter: string | null;
156
+ readonly header: boolean | "auto";
157
+ readonly table: "edges" | "nodes" | "auto";
158
+ readonly sourceColumn: CsvColumnRef | null;
159
+ readonly targetColumn: CsvColumnRef | null;
160
+ /** The direction column reference; null for none; undefined for the Gephi rule. */
161
+ readonly typeColumn: CsvColumnRef | null | undefined;
162
+ readonly idColumn: CsvColumnRef | null;
163
+ readonly nodes: ImportInput | null;
164
+ }
165
+
166
+ /** The columns of an edge table, by index. */
167
+ interface EdgePlan {
168
+ readonly kind: "edges";
169
+ readonly names: readonly string[];
170
+ readonly width: number;
171
+ readonly source: number;
172
+ readonly target: number;
173
+ readonly weight: number;
174
+ readonly type: number;
175
+ readonly id: number;
176
+ readonly label: number;
177
+ readonly attributes: readonly number[];
178
+ }
179
+
180
+ /** The columns of a node table, by index. */
181
+ interface NodePlan {
182
+ readonly kind: "nodes";
183
+ readonly names: readonly string[];
184
+ readonly width: number;
185
+ /** The column ids are read from, or -1 under nodeIdFrom "index". */
186
+ readonly id: number;
187
+ readonly label: number;
188
+ readonly attributes: readonly number[];
189
+ }
190
+
191
+ /** Everything one import call shares between its tables. */
192
+ interface ImportState {
193
+ readonly sink: GraphSink;
194
+ readonly report: ImportReportBuilder;
195
+ readonly common: ResolvedImportOptions;
196
+ readonly csv: ResolvedCsvOptions;
197
+ readonly weightFromExplicit: boolean;
198
+ readonly coercer: IdCoercer;
199
+ readonly resolver: DirectionResolver;
200
+ headerSet: boolean;
201
+ /**
202
+ * The file-level direction a SNAP (`# Directed graph` / `# Undirected graph`) or KONECT
203
+ * (`% asym` / `% sym` / `% bip`) comment header declares; null when the file declares none
204
+ * (the `defaultDirected` option applies).
205
+ */
206
+ commentDirected: boolean | null;
207
+ }
208
+
209
+ /** The comment characters of the SNAP (`#`) and KONECT (`%`) headers (research note 07 section 2.6). */
210
+ const COMMENT_CHARS: readonly string[] = Object.freeze(["#", "%"]);
211
+
212
+ /**
213
+ * The direction a SNAP or KONECT comment header declares (research note 07 section 2.6): SNAP
214
+ * pages write `# Directed graph` / `# Undirected graph`, KONECT's first line is `% sym` (undirected),
215
+ * `% asym` (directed) or `% bip` (bipartite, undirected).
216
+ * @param comments - the leading comment lines
217
+ * @returns true / false when a line declares the direction, null otherwise
218
+ */
219
+ function commentDirection(comments: readonly string[]): boolean | null {
220
+ for (const comment of comments) {
221
+ const text = comment.slice(1).trim().toLowerCase();
222
+ if (text.startsWith("directed graph") || text.startsWith("asym")) {
223
+ return true;
224
+ }
225
+ if (text.startsWith("undirected graph") || text.startsWith("sym") || text.startsWith("bip")) {
226
+ return false;
227
+ }
228
+ }
229
+ return null;
230
+ }
231
+
232
+ /**
233
+ * Apply the defaults of the CSV options and check every value.
234
+ * @param options - the caller's options
235
+ * @returns the resolved options; E_UNSUPPORTED for a value outside its set
236
+ */
237
+ function resolveCsvOptions(options: (CsvImportOptions & CommonImportOptions) | undefined): ResolvedCsvOptions {
238
+ const o: CsvImportOptions = options ?? {};
239
+ if (o.delimiter !== undefined && (typeof o.delimiter !== "string" || o.delimiter.length === 0)) {
240
+ throw new GraphFormatError("E_UNSUPPORTED", "option delimiter: expected a non-empty string", {
241
+ option: "delimiter",
242
+ found: o.delimiter,
243
+ });
244
+ }
245
+ if (o.delimiter !== undefined && BAD_DELIMITERS.has(o.delimiter)) {
246
+ throw new GraphFormatError("E_UNSUPPORTED", "option delimiter: a quote or a line break cannot delimit", {
247
+ option: "delimiter",
248
+ found: o.delimiter,
249
+ });
250
+ }
251
+ if (o.header !== undefined && o.header !== "auto" && typeof o.header !== "boolean") {
252
+ throw new GraphFormatError("E_UNSUPPORTED", 'option header: expected true, false or "auto"', {
253
+ option: "header",
254
+ found: o.header,
255
+ });
256
+ }
257
+ if (o.table !== undefined && !TABLE_MODES.has(o.table)) {
258
+ throw new GraphFormatError("E_UNSUPPORTED", 'option table: expected "edges", "nodes" or "auto"', {
259
+ option: "table",
260
+ found: o.table,
261
+ });
262
+ }
263
+ for (const name of ["sourceColumn", "targetColumn", "idColumn"] as const) {
264
+ checkColumnRef(name, o[name]);
265
+ }
266
+ if (o.typeColumn !== null) {
267
+ checkColumnRef("typeColumn", o.typeColumn);
268
+ }
269
+ return {
270
+ delimiter: o.delimiter ?? null,
271
+ header: o.header ?? "auto",
272
+ table: o.table ?? "auto",
273
+ sourceColumn: o.sourceColumn ?? null,
274
+ targetColumn: o.targetColumn ?? null,
275
+ typeColumn: o.typeColumn,
276
+ idColumn: o.idColumn ?? null,
277
+ nodes: o.nodes ?? null,
278
+ };
279
+ }
280
+
281
+ /**
282
+ * Check a column reference option: a string name or a non-negative integer position.
283
+ * @param name - the option name
284
+ * @param value - the value
285
+ */
286
+ function checkColumnRef(name: string, value: unknown): void {
287
+ if (value === undefined) {
288
+ return;
289
+ }
290
+ if (typeof value === "string" && value.length > 0) {
291
+ return;
292
+ }
293
+ if (typeof value === "number" && Number.isInteger(value) && value >= 0) {
294
+ return;
295
+ }
296
+ throw new GraphFormatError("E_UNSUPPORTED", `option ${name}: expected a column name or a 0-based position`, {
297
+ option: name,
298
+ found: value,
299
+ });
300
+ }
301
+
302
+ /**
303
+ * Whether a cell is empty (nothing or whitespace only): an unset value, a missing endpoint.
304
+ * @param text - the cell text
305
+ * @returns true when blank
306
+ */
307
+ function isBlank(text: string): boolean {
308
+ return text.length === 0 || text.trim().length === 0;
309
+ }
310
+
311
+ /**
312
+ * Whether a cell is unset: blank and not quoted. A quoted blank cell (`""`, `" "`) is a set value
313
+ * (the empty string is a legal id and a legal value, design section 4.1); an unquoted one is
314
+ * nothing at all.
315
+ * @param text - the cell text
316
+ * @param quoted - whether the cell was quoted
317
+ * @returns true when the cell carries no value
318
+ */
319
+ function isUnset(text: string, quoted: boolean | undefined): boolean {
320
+ return quoted !== true && isBlank(text);
321
+ }
322
+
323
+ /** Rows between two checks of the cancellation signal on an in-memory input. */
324
+ const ABORT_CHECK_INTERVAL = 64;
325
+
326
+ /**
327
+ * Header names made unique within the table: a repeated name becomes `<name>#<position>` (1-based,
328
+ * the CSV analogue of the `#<origin.id>` rule of design section 5.6) and is reported.
329
+ * @param names - the header names
330
+ * @param report - the report
331
+ * @param line - the header line
332
+ * @returns the unique names
333
+ */
334
+ function uniqueNames(names: readonly string[], report: ImportReportBuilder, line: number): string[] {
335
+ const seen = new Set<string>();
336
+ return names.map((name, i) => {
337
+ const unique = uniqueColumnName(name, String(i + 1), (n) => seen.has(n));
338
+ seen.add(unique);
339
+ if (unique !== name) {
340
+ report.warning(
341
+ "coercion",
342
+ RENAMED_CODE,
343
+ `column ${i + 1} "${name}" renamed to "${unique}": the header repeats the name`,
344
+ { line, element: name },
345
+ );
346
+ }
347
+ return unique;
348
+ });
349
+ }
350
+
351
+ /**
352
+ * Find a column by candidates, skipping indices already claimed by another role.
353
+ * @param names - the header names
354
+ * @param candidates - the names to look for
355
+ * @param claimed - indices taken
356
+ * @returns the index, or -1
357
+ */
358
+ function findFree(names: readonly string[], candidates: readonly string[], claimed: ReadonlySet<number>): number {
359
+ const masked = names.map((name, i) => (claimed.has(i) ? "" : name));
360
+ return findColumn(masked, candidates);
361
+ }
362
+
363
+ /**
364
+ * Declare a role column (edge id, label) on the sink through the shared design section 5.6 rule:
365
+ * a name already declared differently is renamed `<name>#<position>` (reported); a role already
366
+ * held by another column is dropped from this declaration (reported), the values are kept under
367
+ * the name.
368
+ * @param sink - the sink
369
+ * @param domain - node or edge
370
+ * @param decl - the declaration
371
+ * @param position - the 1-based column position, for the rename
372
+ * @param report - the report
373
+ * @param line - the header line
374
+ * @returns the handle
375
+ */
376
+ function declareRoleColumn(
377
+ sink: GraphSink,
378
+ domain: "node" | "edge",
379
+ decl: ColumnDecl,
380
+ position: number,
381
+ report: ImportReportBuilder,
382
+ line: number,
383
+ ): ColumnHandle {
384
+ const withOrigin: ColumnDecl = { ...decl, origin: { ...decl.origin, id: String(position) } };
385
+ const resolved = declareResolved(sink, domain, withOrigin, report, { line, element: decl.name });
386
+ if (resolved.roleDropped) {
387
+ // a unique constraint belongs to the id role; without the role the column is plain text
388
+ return resolved.handle;
389
+ }
390
+ return resolved.handle;
391
+ }
392
+
393
+ /**
394
+ * Parse a Gephi Type cell.
395
+ * @param text - the cell text
396
+ * @returns the kind; undefined for a blank cell (the default applies); null for an unknown word
397
+ */
398
+ function parseKind(text: string): EdgeKind | null | undefined {
399
+ if (isBlank(text)) {
400
+ return undefined;
401
+ }
402
+ switch (text.trim().toLowerCase()) {
403
+ case "directed":
404
+ return "directed";
405
+ case "undirected":
406
+ return "undirected";
407
+ case "mutual":
408
+ return "mutual";
409
+ default:
410
+ return null;
411
+ }
412
+ }
413
+
414
+ /**
415
+ * Read one CSV table into the sink.
416
+ */
417
+ class TableReader {
418
+ private readonly state: ImportState;
419
+
420
+ private readonly reader: CsvRecordReader;
421
+
422
+ private readonly kind: "edges" | "nodes" | "auto";
423
+
424
+ private plan: EdgePlan | NodePlan | null = null;
425
+
426
+ private writers: (InferredColumn | null)[] = [];
427
+
428
+ private idHandle: ColumnHandle = INVALID_INDEX as ColumnHandle;
429
+
430
+ private labelHandle: ColumnHandle = INVALID_INDEX as ColumnHandle;
431
+
432
+ private dataRows = 0;
433
+
434
+ private nodeOrdinal = 0;
435
+
436
+ /** The edge ids seen so far (the id column is unique; a repeat is skipped with an issue). */
437
+ private readonly edgeIds = new Set<string>();
438
+
439
+ private readonly where: { line: number | null; element: string | null } = { line: null, element: null };
440
+
441
+ /**
442
+ * Create a table reader.
443
+ * @param state - the import state
444
+ * @param input - the table's input
445
+ * @param kind - what the table is, or "auto"
446
+ * @param progress - whether this table reports byte progress
447
+ */
448
+ constructor(state: ImportState, input: ImportInput, kind: "edges" | "nodes" | "auto", progress: boolean) {
449
+ this.state = state;
450
+ this.kind = kind;
451
+ const readerOptions: CsvReaderOptions = {
452
+ delimiter: state.csv.delimiter,
453
+ comments: COMMENT_CHARS,
454
+ signal: state.common.signal,
455
+ onProgress: progress ? state.common.onProgress : null,
456
+ };
457
+ this.reader = new CsvRecordReader(input, state.report, readerOptions);
458
+ }
459
+
460
+ /**
461
+ * Read every row; the reader is closed (and a stream cancelled) when the import aborts midway.
462
+ */
463
+ async read(): Promise<void> {
464
+ const iterator = this.reader[Symbol.asyncIterator]();
465
+ try {
466
+ await this.readRows(iterator);
467
+ } finally {
468
+ await iterator.return(undefined);
469
+ }
470
+ }
471
+
472
+ /**
473
+ * Read the header (or decide there is none), resolve the plan, then push every row.
474
+ * @param iterator - the record iterator
475
+ */
476
+ private async readRows(iterator: AsyncGenerator<string[], void, undefined>): Promise<void> {
477
+ const { report } = this.state;
478
+ const first = await iterator.next();
479
+ const firstRow: string[] = first.done
480
+ ? report.fail(EMPTY_INPUT_CODE, "the input is empty: no header row and no records")
481
+ : first.value;
482
+ const firstLine = this.reader.line;
483
+ const firstQuoted = this.reader.quoted.slice(0, firstRow.length);
484
+ const pending: { row: string[]; quoted: readonly boolean[]; line: number }[] = [];
485
+ let header: boolean;
486
+ const { header: mode } = this.state.csv;
487
+ if (mode === "auto") {
488
+ const second = await iterator.next();
489
+ const secondRow: string[] | null = second.done ? null : second.value;
490
+ const secondLine = this.reader.line;
491
+ const secondQuoted = this.reader.quoted.slice(0, secondRow?.length ?? 0);
492
+ header = looksLikeHeader(firstRow, secondRow);
493
+ if (!header) {
494
+ pending.push({ row: firstRow, quoted: firstQuoted, line: firstLine });
495
+ }
496
+ if (secondRow !== null) {
497
+ pending.push({ row: secondRow, quoted: secondQuoted, line: secondLine });
498
+ }
499
+ } else {
500
+ header = mode;
501
+ if (!header) {
502
+ pending.push({ row: firstRow, quoted: firstQuoted, line: firstLine });
503
+ }
504
+ }
505
+ const names = header ? uniqueNames(headerNames(firstRow), report, firstLine) : positionalNames(firstRow.length);
506
+ if (this.kind !== "nodes" && this.state.commentDirected === null) {
507
+ this.state.commentDirected = commentDirection(this.reader.leadingComments);
508
+ }
509
+ this.plan = this.resolvePlan(names, header, firstLine);
510
+ this.prepareColumns(firstLine);
511
+ for (const { row, quoted, line } of pending) {
512
+ this.processRow(row, quoted, line);
513
+ }
514
+ const { signal } = this.state.common;
515
+ let sinceCheck = 0;
516
+ for (;;) {
517
+ const next = await iterator.next();
518
+ if (next.done) {
519
+ break;
520
+ }
521
+ this.processRow(next.value, this.reader.quoted, this.reader.line);
522
+ if (++sinceCheck >= ABORT_CHECK_INTERVAL) {
523
+ sinceCheck = 0;
524
+ throwIfAborted(signal);
525
+ }
526
+ }
527
+ for (const writer of this.writers) {
528
+ writer?.finish();
529
+ }
530
+ if (this.dataRows === 0 && header) {
531
+ report.warning("missing-value", NO_DATA_ROWS_CODE, "the table has a header and no data rows", {
532
+ line: firstLine,
533
+ });
534
+ }
535
+ }
536
+
537
+ /**
538
+ * Decide the table's columns from its header.
539
+ * @param names - the column names
540
+ * @param header - whether the file has a header row
541
+ * @param line - the header line
542
+ * @returns the plan; the import aborts when no endpoints (or id) resolve
543
+ */
544
+ private resolvePlan(names: readonly string[], header: boolean, line: number): EdgePlan | NodePlan {
545
+ const { csv, report } = this.state;
546
+ const width = names.length;
547
+ let source = -1;
548
+ let target = -1;
549
+ if (this.kind !== "nodes") {
550
+ if (csv.sourceColumn !== null) {
551
+ source = resolveColumnRef(names, csv.sourceColumn, "sourceColumn");
552
+ } else if (header) {
553
+ source = findColumn(names, SOURCE_NAMES);
554
+ } else {
555
+ source = width >= 2 ? 0 : -1;
556
+ }
557
+ if (csv.targetColumn !== null) {
558
+ target = resolveColumnRef(names, csv.targetColumn, "targetColumn");
559
+ } else if (header) {
560
+ target = findColumn(names, TARGET_NAMES);
561
+ } else {
562
+ target = width >= 2 ? 1 : -1;
563
+ }
564
+ if (source >= 0 && target >= 0 && source === target) {
565
+ throw new GraphFormatError("E_UNSUPPORTED", "sourceColumn and targetColumn name the same column", {
566
+ option: "targetColumn",
567
+ found: names[target],
568
+ });
569
+ }
570
+ }
571
+ if (source >= 0 && target >= 0) {
572
+ return this.edgePlan(names, header, source, target);
573
+ }
574
+ const shown = names.map((n) => JSON.stringify(n)).join(", ");
575
+ if (
576
+ this.kind === "edges" ||
577
+ (this.kind === "auto" && (csv.sourceColumn !== null || csv.targetColumn !== null))
578
+ ) {
579
+ report.fail(
580
+ NO_ENDPOINT_COLUMNS_CODE,
581
+ `no source / target columns in the header (${shown}); a node table goes in the nodes option`,
582
+ { line },
583
+ { columns: [...names] },
584
+ );
585
+ }
586
+ const idResolves = header ? findColumn(names, ID_NAMES) >= 0 : width >= 1;
587
+ if (this.kind === "auto" && csv.idColumn === null && !idResolves) {
588
+ report.fail(
589
+ NO_ENDPOINT_COLUMNS_CODE,
590
+ `no source / target columns and no id column in the header (${shown}); the input is neither an edge table nor a node table`,
591
+ { line },
592
+ { columns: [...names] },
593
+ );
594
+ }
595
+ return this.nodePlan(names, header, line);
596
+ }
597
+
598
+ /**
599
+ * The columns of an edge table.
600
+ * @param names - the column names
601
+ * @param header - whether the file has a header row
602
+ * @param source - the source column
603
+ * @param target - the target column
604
+ * @returns the plan
605
+ */
606
+ private edgePlan(names: readonly string[], header: boolean, source: number, target: number): EdgePlan {
607
+ const { csv, common, report } = this.state;
608
+ const claimed = new Set<number>([source, target]);
609
+ let weight = -1;
610
+ if (common.weightFrom !== null) {
611
+ if (header) {
612
+ weight = findFree(names, [common.weightFrom], claimed);
613
+ if (weight < 0 && this.state.weightFromExplicit) {
614
+ report.warning(
615
+ "missing-value",
616
+ COLUMN_MISSING_CODE,
617
+ `weight column ${JSON.stringify(common.weightFrom)} not found; edges are unweighted`,
618
+ { line: this.reader.line, element: common.weightFrom },
619
+ );
620
+ }
621
+ } else if (names.length >= 3 && !claimed.has(2)) {
622
+ weight = 2;
623
+ }
624
+ }
625
+ if (weight >= 0) {
626
+ claimed.add(weight);
627
+ }
628
+ let type = -1;
629
+ if (csv.typeColumn === undefined) {
630
+ if (header && names[source] === "Source" && names[target] === "Target") {
631
+ type = findFree(names, [TYPE_NAME], claimed);
632
+ if (type >= 0 && names[type] !== TYPE_NAME) {
633
+ type = -1;
634
+ }
635
+ }
636
+ } else if (csv.typeColumn !== null) {
637
+ type = resolveColumnRef(names, csv.typeColumn, "typeColumn");
638
+ if (claimed.has(type)) {
639
+ throw new GraphFormatError("E_UNSUPPORTED", "typeColumn names an endpoint or weight column", {
640
+ option: "typeColumn",
641
+ found: names[type],
642
+ });
643
+ }
644
+ }
645
+ if (type >= 0) {
646
+ claimed.add(type);
647
+ }
648
+ const id = header ? findFree(names, EDGE_ID_NAMES, claimed) : -1;
649
+ if (id >= 0) {
650
+ claimed.add(id);
651
+ }
652
+ const label = header ? findFree(names, LABEL_NAMES, claimed) : -1;
653
+ if (label >= 0) {
654
+ claimed.add(label);
655
+ }
656
+ const attributes: number[] = [];
657
+ for (let i = 0; i < names.length; i++) {
658
+ if (!claimed.has(i)) {
659
+ attributes.push(i);
660
+ }
661
+ }
662
+ return { kind: "edges", names, width: names.length, source, target, weight, type, id, label, attributes };
663
+ }
664
+
665
+ /**
666
+ * The columns of a node table.
667
+ * @param names - the column names
668
+ * @param header - whether the file has a header row
669
+ * @param line - the header line
670
+ * @returns the plan; the import aborts when no id column resolves
671
+ */
672
+ private nodePlan(names: readonly string[], header: boolean, line: number): NodePlan {
673
+ const { csv, common, report } = this.state;
674
+ const claimed = new Set<number>();
675
+ let idColumn = -1;
676
+ if (csv.idColumn !== null) {
677
+ idColumn = resolveColumnRef(names, csv.idColumn, "idColumn");
678
+ } else if (header) {
679
+ idColumn = findColumn(names, ID_NAMES);
680
+ } else if (names.length >= 1) {
681
+ idColumn = 0;
682
+ }
683
+ const label = header ? findFree(names, LABEL_NAMES, new Set(idColumn >= 0 ? [idColumn] : [])) : -1;
684
+ let id: number;
685
+ switch (common.nodeIdFrom) {
686
+ case "label":
687
+ if (label < 0) {
688
+ report.fail(NO_ID_COLUMN_CODE, 'nodeIdFrom is "label" but the node table has no label column', {
689
+ line,
690
+ });
691
+ }
692
+ id = label;
693
+ break;
694
+ case "index":
695
+ id = -1;
696
+ break;
697
+ default:
698
+ if (idColumn < 0) {
699
+ report.fail(
700
+ NO_ID_COLUMN_CODE,
701
+ `no id column in the node table header (${names.map((n) => JSON.stringify(n)).join(", ")})`,
702
+ { line },
703
+ { columns: [...names] },
704
+ );
705
+ }
706
+ id = idColumn;
707
+ claimed.add(idColumn);
708
+ break;
709
+ }
710
+ if (label >= 0) {
711
+ claimed.add(label);
712
+ }
713
+ const attributes: number[] = [];
714
+ for (let i = 0; i < names.length; i++) {
715
+ if (!claimed.has(i)) {
716
+ attributes.push(i);
717
+ }
718
+ }
719
+ return { kind: "nodes", names, width: names.length, id, label, attributes };
720
+ }
721
+
722
+ /**
723
+ * Declare the role columns and create a writer per attribute column.
724
+ * @param line - the header line
725
+ */
726
+ private prepareColumns(line: number): void {
727
+ const plan = this.requirePlan();
728
+ const { sink, report } = this.state;
729
+ const domain = plan.kind === "edges" ? "edge" : "node";
730
+ const origin = { format: "csv" };
731
+ if (plan.kind === "edges" && plan.id >= 0) {
732
+ this.idHandle = declareRoleColumn(
733
+ sink,
734
+ "edge",
735
+ { name: plan.names[plan.id], dtype: "string", nullable: true, role: "id", unique: true, origin },
736
+ plan.id + 1,
737
+ report,
738
+ line,
739
+ );
740
+ }
741
+ if (plan.label >= 0) {
742
+ this.labelHandle = declareRoleColumn(
743
+ sink,
744
+ domain,
745
+ { name: plan.names[plan.label], dtype: "string", nullable: true, role: "label", origin },
746
+ plan.label + 1,
747
+ report,
748
+ line,
749
+ );
750
+ }
751
+ this.writers = plan.names.map(() => null);
752
+ for (const index of plan.attributes) {
753
+ this.writers[index] = new InferredColumn(plan.names[index], domain, sink, report);
754
+ }
755
+ }
756
+
757
+ /**
758
+ * The plan, which exists once the header was read.
759
+ * @returns the plan
760
+ */
761
+ private requirePlan(): EdgePlan | NodePlan {
762
+ if (this.plan === null) {
763
+ throw new GraphFormatError("E_UNSUPPORTED", "the header has not been read", { reason: "no plan" });
764
+ }
765
+ return this.plan;
766
+ }
767
+
768
+ /**
769
+ * Push one data row.
770
+ * @param row - the cells
771
+ * @param quoted - whether each cell was quoted (a quoted empty cell is the empty string)
772
+ * @param line - the row's line
773
+ */
774
+ private processRow(row: string[], quoted: readonly boolean[], line: number): void {
775
+ const plan = this.requirePlan();
776
+ this.dataRows++;
777
+ if (plan.kind === "edges") {
778
+ this.processEdgeRow(plan, row, quoted, line);
779
+ } else {
780
+ this.processNodeRow(plan, row, quoted, line);
781
+ }
782
+ }
783
+
784
+ /**
785
+ * Push one edge row: endpoints, weight, direction, then the attribute cells.
786
+ * @param plan - the edge plan
787
+ * @param row - the cells
788
+ * @param quoted - whether each cell was quoted
789
+ * @param line - the row's line
790
+ */
791
+ private processEdgeRow(plan: EdgePlan, row: string[], quoted: readonly boolean[], line: number): void {
792
+ const { report, sink, resolver, common } = this.state;
793
+ const { counts } = report;
794
+ if (row.length !== plan.width) {
795
+ report.error(
796
+ "validation-error",
797
+ FIELD_COUNT_CODE,
798
+ `line ${line}: ${row.length} field(s), the header has ${plan.width}`,
799
+ { line },
800
+ );
801
+ counts.skippedEdges++;
802
+ return;
803
+ }
804
+ const sourceText = row[plan.source];
805
+ const targetText = row[plan.target];
806
+ const sourceMissing = isUnset(sourceText, quoted[plan.source]);
807
+ if (sourceMissing || isUnset(targetText, quoted[plan.target])) {
808
+ report.error(
809
+ "missing-value",
810
+ MISSING_ENDPOINT_CODE,
811
+ `line ${line}: blank ${sourceMissing ? "source" : "target"} cell`,
812
+ { line },
813
+ );
814
+ counts.skippedEdges++;
815
+ return;
816
+ }
817
+ let kind: EdgeKind = (this.state.commentDirected ?? common.defaultDirected) ? "directed" : "undirected";
818
+ if (plan.type >= 0) {
819
+ const parsed = parseKind(row[plan.type]);
820
+ if (parsed === null) {
821
+ report.error(
822
+ "validation-error",
823
+ BAD_TYPE_CODE,
824
+ `line ${line}: Type ${JSON.stringify(row[plan.type])} is not Directed, Undirected or Mutual`,
825
+ { line },
826
+ );
827
+ counts.skippedEdges++;
828
+ return;
829
+ }
830
+ if (parsed !== undefined) {
831
+ kind = parsed;
832
+ }
833
+ }
834
+ const { where } = this;
835
+ where.line = line;
836
+ where.element = null;
837
+ const idText = plan.id >= 0 && !isUnset(row[plan.id], quoted[plan.id]) ? row[plan.id] : null;
838
+ if (idText !== null) {
839
+ if (this.edgeIds.has(idText)) {
840
+ report.error(
841
+ "validation-error",
842
+ DUPLICATE_EDGE_ID_CODE,
843
+ `line ${line}: edge id ${JSON.stringify(idText)} repeats an earlier row's; the row is skipped`,
844
+ { line, element: idText },
845
+ );
846
+ counts.skippedEdges++;
847
+ return;
848
+ }
849
+ this.edgeIds.add(idText);
850
+ }
851
+ let edge: number;
852
+ try {
853
+ const source = this.coerce(sourceText);
854
+ const target = this.coerce(targetText);
855
+ const weight = plan.weight >= 0 ? parseWeightText(row[plan.weight]) : undefined;
856
+ if (!this.state.headerSet) {
857
+ this.state.headerSet = true;
858
+ resolver.setHeader(kind !== "undirected", where);
859
+ }
860
+ const sourceNew = sink.indexOf(source) === INVALID_INDEX;
861
+ const targetNew = source !== target && sink.indexOf(target) === INVALID_INDEX;
862
+ const before = sink.edgeCount;
863
+ edge = resolver.addEdge(source, target, kind, weight, where);
864
+ counts.edges += sink.edgeCount - before;
865
+ counts.nodes += (sourceNew ? 1 : 0) + (targetNew ? 1 : 0);
866
+ } catch (err) {
867
+ where.element = idText ?? `${sourceText}->${targetText}`;
868
+ report.recordError(err, where);
869
+ counts.skippedEdges++;
870
+ return;
871
+ }
872
+ if (idText !== null) {
873
+ this.writeRole(this.idHandle, "edge", edge, idText, plan.names[plan.id], line);
874
+ }
875
+ if (plan.label >= 0 && !isUnset(row[plan.label], quoted[plan.label])) {
876
+ this.writeRole(this.labelHandle, "edge", edge, row[plan.label], plan.names[plan.label], line);
877
+ }
878
+ this.writeAttributes(plan, row, quoted, edge, line);
879
+ }
880
+
881
+ /**
882
+ * Push one node row: the id, then the label and attribute cells.
883
+ * @param plan - the node plan
884
+ * @param row - the cells
885
+ * @param quoted - whether each cell was quoted
886
+ * @param line - the row's line
887
+ */
888
+ private processNodeRow(plan: NodePlan, row: string[], quoted: readonly boolean[], line: number): void {
889
+ const { report, sink } = this.state;
890
+ const { counts } = report;
891
+ const ordinal = this.nodeOrdinal++;
892
+ if (row.length !== plan.width) {
893
+ report.error(
894
+ "validation-error",
895
+ FIELD_COUNT_CODE,
896
+ `line ${line}: ${row.length} field(s), the header has ${plan.width}`,
897
+ { line },
898
+ );
899
+ counts.skippedNodes++;
900
+ return;
901
+ }
902
+ const idText = plan.id >= 0 ? row[plan.id] : String(ordinal);
903
+ if (plan.id >= 0 && isUnset(idText, quoted[plan.id])) {
904
+ report.error("missing-value", MISSING_ID_CODE, `line ${line}: blank id cell`, { line });
905
+ counts.skippedNodes++;
906
+ return;
907
+ }
908
+ const { where } = this;
909
+ where.line = line;
910
+ where.element = idText;
911
+ let index: number;
912
+ try {
913
+ const id = plan.id >= 0 ? this.coerce(idText) : ordinal;
914
+ if (sink.indexOf(id) !== INVALID_INDEX) {
915
+ report.warning(
916
+ "merged",
917
+ DUPLICATE_NODE_CODE,
918
+ `line ${line}: node ${JSON.stringify(id)} already exists; its attributes are overwritten`,
919
+ where,
920
+ );
921
+ } else {
922
+ counts.nodes++;
923
+ }
924
+ index = sink.addNode(id);
925
+ } catch (err) {
926
+ report.recordError(err, where);
927
+ counts.skippedNodes++;
928
+ return;
929
+ }
930
+ if (plan.label >= 0 && !isUnset(row[plan.label], quoted[plan.label])) {
931
+ this.writeRole(this.labelHandle, "node", index, row[plan.label], plan.names[plan.label], line);
932
+ }
933
+ this.writeAttributes(plan, row, quoted, index, line);
934
+ }
935
+
936
+ /**
937
+ * Coerce an id cell, reporting a merge under `ids: "number"`.
938
+ * @param text - the cell text
939
+ * @returns the id
940
+ */
941
+ private coerce(text: string): NodeId {
942
+ const id = this.state.coercer.text(text);
943
+ const merge = this.state.coercer.lastMerge;
944
+ if (merge !== null) {
945
+ this.state.report.warnOnce(
946
+ "coercion",
947
+ ID_MERGED_CODE,
948
+ `id ${JSON.stringify(merge.text)} merged with ${JSON.stringify(merge.previousText)} as ${merge.id} under ids: "number"`,
949
+ this.where,
950
+ );
951
+ }
952
+ return id;
953
+ }
954
+
955
+ /**
956
+ * Write a role column cell (edge id, label).
957
+ * @param handle - the column handle
958
+ * @param domain - node or edge
959
+ * @param row - the node or edge index
960
+ * @param text - the cell text
961
+ * @param name - the column name, for issues
962
+ * @param line - the row's line
963
+ */
964
+ private writeRole(
965
+ handle: ColumnHandle,
966
+ domain: "node" | "edge",
967
+ row: number,
968
+ text: string,
969
+ name: string,
970
+ line: number,
971
+ ): void {
972
+ try {
973
+ if (domain === "node") {
974
+ this.state.sink.setNodeValue(handle, row, text);
975
+ } else {
976
+ this.state.sink.setEdgeValue(handle, row, text);
977
+ }
978
+ } catch (err) {
979
+ this.state.report.recordError(err, { line, element: name });
980
+ }
981
+ }
982
+
983
+ /**
984
+ * Write the attribute cells of a row: an unquoted blank cell is unset, a quoted one (`""`,
985
+ * `" "`) is the text it holds.
986
+ * @param plan - the plan
987
+ * @param row - the cells
988
+ * @param quoted - whether each cell was quoted
989
+ * @param index - the node or edge index
990
+ * @param line - the row's line
991
+ */
992
+ private writeAttributes(
993
+ plan: EdgePlan | NodePlan,
994
+ row: readonly string[],
995
+ quoted: readonly boolean[],
996
+ index: number,
997
+ line: number,
998
+ ): void {
999
+ for (const k of plan.attributes) {
1000
+ const text = row[k];
1001
+ if (isUnset(text, quoted[k])) {
1002
+ continue;
1003
+ }
1004
+ const writer = this.writers[k];
1005
+ if (writer === null) {
1006
+ continue;
1007
+ }
1008
+ try {
1009
+ writer.write(index, text);
1010
+ } catch (err) {
1011
+ this.state.report.recordError(err, { line, element: writer.name });
1012
+ }
1013
+ }
1014
+ }
1015
+ }
1016
+
1017
+ const HEAD_BYTES = 4096;
1018
+ const OTHER_FORMAT = /^\s*(<|[[{]|(strict\s+)?(di)?graph(\s+\S+)?\s*\{|\*vertices|creator\b|graph\s*\[)/i;
1019
+
1020
+ /**
1021
+ * Sniff confidence for the registry: 0 for XML, JSON, GML, DOT and Pajek openings; otherwise a
1022
+ * delimited first row with endpoint headers is 0.9, with an id header 0.6, any consistently
1023
+ * delimited rows 0.3, a single column 0.
1024
+ * @param head - the first bytes of the input
1025
+ * @returns a confidence in 0..1
1026
+ */
1027
+ function sniff(head: Uint8Array): number {
1028
+ const text = new TextDecoder("utf-8").decode(head.subarray(0, HEAD_BYTES));
1029
+ const body = text.startsWith(String.fromCharCode(0xfeff)) ? text.slice(1) : text;
1030
+ if (body.trim().length === 0 || OTHER_FORMAT.test(body)) {
1031
+ return 0;
1032
+ }
1033
+ const newline = sniffNewline(body);
1034
+ const delimiter = sniffDelimiter(body, newline);
1035
+ if (delimiter === null) {
1036
+ return 0;
1037
+ }
1038
+ const end = body.indexOf(newline);
1039
+ const firstLine = end < 0 ? body : body.slice(0, end);
1040
+ const names = headerNames(firstLine.replace(/\r$/, "").split(delimiter));
1041
+ if (findColumn(names, SOURCE_NAMES) >= 0 && findColumn(names, TARGET_NAMES) >= 0) {
1042
+ return 0.9;
1043
+ }
1044
+ if (findColumn(names, ID_NAMES) >= 0) {
1045
+ return 0.6;
1046
+ }
1047
+ return 0.3;
1048
+ }
1049
+
1050
+ /**
1051
+ * Import a CSV edge table (and an optional node table) into a sink.
1052
+ * @param input - the edge table (or, with `table: "nodes"` or a header without endpoints, a node table)
1053
+ * @param sink - the sink
1054
+ * @param options - CSV and common options
1055
+ * @returns the report; ImportError with the partial report when the import aborts
1056
+ */
1057
+ async function importCsv(
1058
+ input: ImportInput,
1059
+ sink: GraphSink,
1060
+ options?: CsvImportOptions & CommonImportOptions,
1061
+ ): Promise<ImportReport> {
1062
+ const common = resolveImportOptions(options, { ids: "canonical", defaultDirected: true, weightFrom: "weight" });
1063
+ const csv = resolveCsvOptions(options);
1064
+ const report = new ImportReportBuilder("csv", common.errorLimit);
1065
+ reportSinkOptions(sink, options, report);
1066
+ reportUnusedOptions(
1067
+ options,
1068
+ report,
1069
+ csv.table === "nodes" || csv.nodes !== null ? USED_OPTIONS_WITH_NODES : USED_OPTIONS,
1070
+ );
1071
+ const state: ImportState = {
1072
+ sink,
1073
+ report,
1074
+ common,
1075
+ csv,
1076
+ weightFromExplicit: typeof options?.weightFrom === "string",
1077
+ coercer: new IdCoercer(common.ids),
1078
+ resolver: new DirectionResolver(sink, report, common.onMixedDirection),
1079
+ headerSet: false,
1080
+ commentDirected: null,
1081
+ };
1082
+ if (csv.nodes !== null) {
1083
+ await new TableReader(state, csv.nodes, "nodes", false).read();
1084
+ }
1085
+ await new TableReader(state, input, csv.table, true).read();
1086
+ if (state.coercer.mergeCount > 1) {
1087
+ report.warning(
1088
+ "coercion",
1089
+ ID_MERGED_CODE,
1090
+ `${state.coercer.mergeCount} id cell(s) merged into ids other cells already produced under ids: "number"`,
1091
+ );
1092
+ }
1093
+ throwIfAborted(common.signal);
1094
+ return report.finish();
1095
+ }
1096
+
1097
+ /** The CSV / TSV importer plugin (subpath `@graphty/graph-io/csv`). */
1098
+ export const csvImporter: GraphImporter<CsvImportOptions> = Object.freeze({
1099
+ format: "csv",
1100
+ extensions: Object.freeze([".csv", ".tsv", ".edges", ".edgelist"]),
1101
+ mimeTypes: Object.freeze(["text/csv", "text/tab-separated-values", "text/plain"]),
1102
+ sniff,
1103
+ import: importCsv,
1104
+ });