@graphty/graph-io 0.0.0 → 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (339) hide show
  1. package/LICENSE +21 -0
  2. package/README.md +250 -28
  3. package/dist/chunks/children-CL3Cy0ez.js +238 -0
  4. package/dist/chunks/children-CL3Cy0ez.js.map +1 -0
  5. package/dist/chunks/escape-DyI8JofU.js +938 -0
  6. package/dist/chunks/escape-DyI8JofU.js.map +1 -0
  7. package/dist/chunks/importer-CQnJuWJw.js +2987 -0
  8. package/dist/chunks/importer-CQnJuWJw.js.map +1 -0
  9. package/dist/chunks/importer-CpCpfbxr.js +2015 -0
  10. package/dist/chunks/importer-CpCpfbxr.js.map +1 -0
  11. package/dist/chunks/importer-DbnGYr3_.js +2342 -0
  12. package/dist/chunks/importer-DbnGYr3_.js.map +1 -0
  13. package/dist/chunks/importer-GozH8DkN.js +3050 -0
  14. package/dist/chunks/importer-GozH8DkN.js.map +1 -0
  15. package/dist/chunks/records-CGpxszm1.js +605 -0
  16. package/dist/chunks/records-CGpxszm1.js.map +1 -0
  17. package/dist/chunks/text-CajMdVFy.js +189 -0
  18. package/dist/chunks/text-CajMdVFy.js.map +1 -0
  19. package/dist/chunks/writer-DxSKC7TL.js +2842 -0
  20. package/dist/chunks/writer-DxSKC7TL.js.map +1 -0
  21. package/dist/csv.d.ts +1 -0
  22. package/dist/csv.js +1702 -0
  23. package/dist/csv.js.map +1 -0
  24. package/dist/dot.d.ts +1 -0
  25. package/dist/dot.js +8 -0
  26. package/dist/dot.js.map +1 -0
  27. package/dist/gexf.d.ts +1 -0
  28. package/dist/gexf.js +3466 -0
  29. package/dist/gexf.js.map +1 -0
  30. package/dist/gml.d.ts +1 -0
  31. package/dist/gml.js +2647 -0
  32. package/dist/gml.js.map +1 -0
  33. package/dist/graph-io.d.ts +1 -0
  34. package/dist/graph-io.js +790 -0
  35. package/dist/graph-io.js.map +1 -0
  36. package/dist/graphml.d.ts +1 -0
  37. package/dist/graphml.js +8 -0
  38. package/dist/graphml.js.map +1 -0
  39. package/dist/json.d.ts +1 -0
  40. package/dist/json.js +11 -0
  41. package/dist/json.js.map +1 -0
  42. package/dist/neo4j.d.ts +1 -0
  43. package/dist/neo4j.js +2046 -0
  44. package/dist/neo4j.js.map +1 -0
  45. package/dist/pajek.d.ts +1 -0
  46. package/dist/pajek.js +8 -0
  47. package/dist/pajek.js.map +1 -0
  48. package/dist/src/children.d.ts +134 -0
  49. package/dist/src/children.d.ts.map +1 -0
  50. package/dist/src/children.js +274 -0
  51. package/dist/src/children.js.map +1 -0
  52. package/dist/src/common/attributes.d.ts +229 -0
  53. package/dist/src/common/attributes.d.ts.map +1 -0
  54. package/dist/src/common/attributes.js +368 -0
  55. package/dist/src/common/attributes.js.map +1 -0
  56. package/dist/src/common/codes.d.ts +105 -0
  57. package/dist/src/common/codes.d.ts.map +1 -0
  58. package/dist/src/common/codes.js +107 -0
  59. package/dist/src/common/codes.js.map +1 -0
  60. package/dist/src/common/declared-types.d.ts +84 -0
  61. package/dist/src/common/declared-types.d.ts.map +1 -0
  62. package/dist/src/common/declared-types.js +326 -0
  63. package/dist/src/common/declared-types.js.map +1 -0
  64. package/dist/src/common/direction.d.ts +206 -0
  65. package/dist/src/common/direction.d.ts.map +1 -0
  66. package/dist/src/common/direction.js +370 -0
  67. package/dist/src/common/direction.js.map +1 -0
  68. package/dist/src/common/escape.d.ts +92 -0
  69. package/dist/src/common/escape.d.ts.map +1 -0
  70. package/dist/src/common/escape.js +212 -0
  71. package/dist/src/common/escape.js.map +1 -0
  72. package/dist/src/common/export.d.ts +249 -0
  73. package/dist/src/common/export.d.ts.map +1 -0
  74. package/dist/src/common/export.js +594 -0
  75. package/dist/src/common/export.js.map +1 -0
  76. package/dist/src/common/format.d.ts +59 -0
  77. package/dist/src/common/format.d.ts.map +1 -0
  78. package/dist/src/common/format.js +106 -0
  79. package/dist/src/common/format.js.map +1 -0
  80. package/dist/src/common/ids.d.ts +83 -0
  81. package/dist/src/common/ids.d.ts.map +1 -0
  82. package/dist/src/common/ids.js +158 -0
  83. package/dist/src/common/ids.js.map +1 -0
  84. package/dist/src/common/input.d.ts +100 -0
  85. package/dist/src/common/input.d.ts.map +1 -0
  86. package/dist/src/common/input.js +335 -0
  87. package/dist/src/common/input.js.map +1 -0
  88. package/dist/src/common/lists.d.ts +34 -0
  89. package/dist/src/common/lists.d.ts.map +1 -0
  90. package/dist/src/common/lists.js +185 -0
  91. package/dist/src/common/lists.js.map +1 -0
  92. package/dist/src/common/options.d.ts +108 -0
  93. package/dist/src/common/options.d.ts.map +1 -0
  94. package/dist/src/common/options.js +265 -0
  95. package/dist/src/common/options.js.map +1 -0
  96. package/dist/src/common/report.d.ts +187 -0
  97. package/dist/src/common/report.d.ts.map +1 -0
  98. package/dist/src/common/report.js +274 -0
  99. package/dist/src/common/report.js.map +1 -0
  100. package/dist/src/common/temporal.d.ts +71 -0
  101. package/dist/src/common/temporal.d.ts.map +1 -0
  102. package/dist/src/common/temporal.js +266 -0
  103. package/dist/src/common/temporal.js.map +1 -0
  104. package/dist/src/common/text.d.ts +104 -0
  105. package/dist/src/common/text.d.ts.map +1 -0
  106. package/dist/src/common/text.js +255 -0
  107. package/dist/src/common/text.js.map +1 -0
  108. package/dist/src/common/weights.d.ts +77 -0
  109. package/dist/src/common/weights.d.ts.map +1 -0
  110. package/dist/src/common/weights.js +156 -0
  111. package/dist/src/common/weights.js.map +1 -0
  112. package/dist/src/common/writer.d.ts +51 -0
  113. package/dist/src/common/writer.d.ts.map +1 -0
  114. package/dist/src/common/writer.js +108 -0
  115. package/dist/src/common/writer.js.map +1 -0
  116. package/dist/src/common/xml.d.ts +245 -0
  117. package/dist/src/common/xml.d.ts.map +1 -0
  118. package/dist/src/common/xml.js +942 -0
  119. package/dist/src/common/xml.js.map +1 -0
  120. package/dist/src/formats/csv/exporter.d.ts +70 -0
  121. package/dist/src/formats/csv/exporter.d.ts.map +1 -0
  122. package/dist/src/formats/csv/exporter.js +682 -0
  123. package/dist/src/formats/csv/exporter.js.map +1 -0
  124. package/dist/src/formats/csv/header.d.ts +66 -0
  125. package/dist/src/formats/csv/header.d.ts.map +1 -0
  126. package/dist/src/formats/csv/header.js +152 -0
  127. package/dist/src/formats/csv/header.js.map +1 -0
  128. package/dist/src/formats/csv/importer.d.ts +82 -0
  129. package/dist/src/formats/csv/importer.d.ts.map +1 -0
  130. package/dist/src/formats/csv/importer.js +849 -0
  131. package/dist/src/formats/csv/importer.js.map +1 -0
  132. package/dist/src/formats/csv/index.d.ts +60 -0
  133. package/dist/src/formats/csv/index.d.ts.map +1 -0
  134. package/dist/src/formats/csv/index.js +63 -0
  135. package/dist/src/formats/csv/index.js.map +1 -0
  136. package/dist/src/formats/csv/records.d.ts +188 -0
  137. package/dist/src/formats/csv/records.d.ts.map +1 -0
  138. package/dist/src/formats/csv/records.js +702 -0
  139. package/dist/src/formats/csv/records.js.map +1 -0
  140. package/dist/src/formats/csv/values.d.ts +105 -0
  141. package/dist/src/formats/csv/values.d.ts.map +1 -0
  142. package/dist/src/formats/csv/values.js +192 -0
  143. package/dist/src/formats/csv/values.js.map +1 -0
  144. package/dist/src/formats/dot/exporter.d.ts +52 -0
  145. package/dist/src/formats/dot/exporter.d.ts.map +1 -0
  146. package/dist/src/formats/dot/exporter.js +836 -0
  147. package/dist/src/formats/dot/exporter.js.map +1 -0
  148. package/dist/src/formats/dot/importer.d.ts +102 -0
  149. package/dist/src/formats/dot/importer.d.ts.map +1 -0
  150. package/dist/src/formats/dot/importer.js +1291 -0
  151. package/dist/src/formats/dot/importer.js.map +1 -0
  152. package/dist/src/formats/dot/index.d.ts +7 -0
  153. package/dist/src/formats/dot/index.d.ts.map +1 -0
  154. package/dist/src/formats/dot/index.js +7 -0
  155. package/dist/src/formats/dot/index.js.map +1 -0
  156. package/dist/src/formats/dot/names.d.ts +29 -0
  157. package/dist/src/formats/dot/names.d.ts.map +1 -0
  158. package/dist/src/formats/dot/names.js +28 -0
  159. package/dist/src/formats/dot/names.js.map +1 -0
  160. package/dist/src/formats/dot/tokenizer.d.ts +114 -0
  161. package/dist/src/formats/dot/tokenizer.d.ts.map +1 -0
  162. package/dist/src/formats/dot/tokenizer.js +341 -0
  163. package/dist/src/formats/dot/tokenizer.js.map +1 -0
  164. package/dist/src/formats/gexf/exporter.d.ts +56 -0
  165. package/dist/src/formats/gexf/exporter.d.ts.map +1 -0
  166. package/dist/src/formats/gexf/exporter.js +1395 -0
  167. package/dist/src/formats/gexf/exporter.js.map +1 -0
  168. package/dist/src/formats/gexf/importer.d.ts +73 -0
  169. package/dist/src/formats/gexf/importer.d.ts.map +1 -0
  170. package/dist/src/formats/gexf/importer.js +1880 -0
  171. package/dist/src/formats/gexf/importer.js.map +1 -0
  172. package/dist/src/formats/gexf/index.d.ts +96 -0
  173. package/dist/src/formats/gexf/index.d.ts.map +1 -0
  174. package/dist/src/formats/gexf/index.js +97 -0
  175. package/dist/src/formats/gexf/index.js.map +1 -0
  176. package/dist/src/formats/gexf/schema.d.ts +135 -0
  177. package/dist/src/formats/gexf/schema.d.ts.map +1 -0
  178. package/dist/src/formats/gexf/schema.js +323 -0
  179. package/dist/src/formats/gexf/schema.js.map +1 -0
  180. package/dist/src/formats/gml/exporter.d.ts +69 -0
  181. package/dist/src/formats/gml/exporter.d.ts.map +1 -0
  182. package/dist/src/formats/gml/exporter.js +1093 -0
  183. package/dist/src/formats/gml/exporter.js.map +1 -0
  184. package/dist/src/formats/gml/importer.d.ts +66 -0
  185. package/dist/src/formats/gml/importer.d.ts.map +1 -0
  186. package/dist/src/formats/gml/importer.js +1331 -0
  187. package/dist/src/formats/gml/importer.js.map +1 -0
  188. package/dist/src/formats/gml/index.d.ts +85 -0
  189. package/dist/src/formats/gml/index.d.ts.map +1 -0
  190. package/dist/src/formats/gml/index.js +88 -0
  191. package/dist/src/formats/gml/index.js.map +1 -0
  192. package/dist/src/formats/gml/syntax.d.ts +186 -0
  193. package/dist/src/formats/gml/syntax.d.ts.map +1 -0
  194. package/dist/src/formats/gml/syntax.js +467 -0
  195. package/dist/src/formats/gml/syntax.js.map +1 -0
  196. package/dist/src/formats/graphml/constants.d.ts +169 -0
  197. package/dist/src/formats/graphml/constants.d.ts.map +1 -0
  198. package/dist/src/formats/graphml/constants.js +165 -0
  199. package/dist/src/formats/graphml/constants.js.map +1 -0
  200. package/dist/src/formats/graphml/exporter.d.ts +34 -0
  201. package/dist/src/formats/graphml/exporter.d.ts.map +1 -0
  202. package/dist/src/formats/graphml/exporter.js +1176 -0
  203. package/dist/src/formats/graphml/exporter.js.map +1 -0
  204. package/dist/src/formats/graphml/importer.d.ts +31 -0
  205. package/dist/src/formats/graphml/importer.d.ts.map +1 -0
  206. package/dist/src/formats/graphml/importer.js +1607 -0
  207. package/dist/src/formats/graphml/importer.js.map +1 -0
  208. package/dist/src/formats/graphml/index.d.ts +8 -0
  209. package/dist/src/formats/graphml/index.d.ts.map +1 -0
  210. package/dist/src/formats/graphml/index.js +8 -0
  211. package/dist/src/formats/graphml/index.js.map +1 -0
  212. package/dist/src/formats/graphml/tree.d.ts +72 -0
  213. package/dist/src/formats/graphml/tree.d.ts.map +1 -0
  214. package/dist/src/formats/graphml/tree.js +290 -0
  215. package/dist/src/formats/graphml/tree.js.map +1 -0
  216. package/dist/src/formats/json/dialect.d.ts +125 -0
  217. package/dist/src/formats/json/dialect.d.ts.map +1 -0
  218. package/dist/src/formats/json/dialect.js +262 -0
  219. package/dist/src/formats/json/dialect.js.map +1 -0
  220. package/dist/src/formats/json/exporter.d.ts +89 -0
  221. package/dist/src/formats/json/exporter.d.ts.map +1 -0
  222. package/dist/src/formats/json/exporter.js +1358 -0
  223. package/dist/src/formats/json/exporter.js.map +1 -0
  224. package/dist/src/formats/json/importer.d.ts +108 -0
  225. package/dist/src/formats/json/importer.d.ts.map +1 -0
  226. package/dist/src/formats/json/importer.js +1838 -0
  227. package/dist/src/formats/json/importer.js.map +1 -0
  228. package/dist/src/formats/json/index.d.ts +8 -0
  229. package/dist/src/formats/json/index.d.ts.map +1 -0
  230. package/dist/src/formats/json/index.js +8 -0
  231. package/dist/src/formats/json/index.js.map +1 -0
  232. package/dist/src/formats/neo4j/exporter.d.ts +68 -0
  233. package/dist/src/formats/neo4j/exporter.d.ts.map +1 -0
  234. package/dist/src/formats/neo4j/exporter.js +1055 -0
  235. package/dist/src/formats/neo4j/exporter.js.map +1 -0
  236. package/dist/src/formats/neo4j/header.d.ts +52 -0
  237. package/dist/src/formats/neo4j/header.d.ts.map +1 -0
  238. package/dist/src/formats/neo4j/header.js +131 -0
  239. package/dist/src/formats/neo4j/header.js.map +1 -0
  240. package/dist/src/formats/neo4j/importer.d.ts +73 -0
  241. package/dist/src/formats/neo4j/importer.d.ts.map +1 -0
  242. package/dist/src/formats/neo4j/importer.js +932 -0
  243. package/dist/src/formats/neo4j/importer.js.map +1 -0
  244. package/dist/src/formats/neo4j/index.d.ts +79 -0
  245. package/dist/src/formats/neo4j/index.d.ts.map +1 -0
  246. package/dist/src/formats/neo4j/index.js +83 -0
  247. package/dist/src/formats/neo4j/index.js.map +1 -0
  248. package/dist/src/formats/pajek/exporter.d.ts +58 -0
  249. package/dist/src/formats/pajek/exporter.d.ts.map +1 -0
  250. package/dist/src/formats/pajek/exporter.js +825 -0
  251. package/dist/src/formats/pajek/exporter.js.map +1 -0
  252. package/dist/src/formats/pajek/importer.d.ts +88 -0
  253. package/dist/src/formats/pajek/importer.d.ts.map +1 -0
  254. package/dist/src/formats/pajek/importer.js +1047 -0
  255. package/dist/src/formats/pajek/importer.js.map +1 -0
  256. package/dist/src/formats/pajek/index.d.ts +7 -0
  257. package/dist/src/formats/pajek/index.d.ts.map +1 -0
  258. package/dist/src/formats/pajek/index.js +7 -0
  259. package/dist/src/formats/pajek/index.js.map +1 -0
  260. package/dist/src/formats/pajek/syntax.d.ts +112 -0
  261. package/dist/src/formats/pajek/syntax.d.ts.map +1 -0
  262. package/dist/src/formats/pajek/syntax.js +269 -0
  263. package/dist/src/formats/pajek/syntax.js.map +1 -0
  264. package/dist/src/index.d.ts +35 -0
  265. package/dist/src/index.d.ts.map +1 -0
  266. package/dist/src/index.js +39 -0
  267. package/dist/src/index.js.map +1 -0
  268. package/dist/src/registry.d.ts +207 -0
  269. package/dist/src/registry.d.ts.map +1 -0
  270. package/dist/src/registry.js +481 -0
  271. package/dist/src/registry.js.map +1 -0
  272. package/dist/src/sniff.d.ts +104 -0
  273. package/dist/src/sniff.d.ts.map +1 -0
  274. package/dist/src/sniff.js +357 -0
  275. package/dist/src/sniff.js.map +1 -0
  276. package/dist/src/types.d.ts +238 -0
  277. package/dist/src/types.d.ts.map +1 -0
  278. package/dist/src/types.js +29 -0
  279. package/dist/src/types.js.map +1 -0
  280. package/dist/tsconfig.build.tsbuildinfo +1 -0
  281. package/package.json +122 -7
  282. package/src/children.ts +335 -0
  283. package/src/common/attributes.ts +520 -0
  284. package/src/common/codes.ts +153 -0
  285. package/src/common/declared-types.ts +374 -0
  286. package/src/common/direction.ts +518 -0
  287. package/src/common/escape.ts +231 -0
  288. package/src/common/export.ts +817 -0
  289. package/src/common/format.ts +111 -0
  290. package/src/common/ids.ts +176 -0
  291. package/src/common/input.ts +378 -0
  292. package/src/common/lists.ts +196 -0
  293. package/src/common/options.ts +377 -0
  294. package/src/common/report.ts +352 -0
  295. package/src/common/temporal.ts +302 -0
  296. package/src/common/text.ts +294 -0
  297. package/src/common/weights.ts +202 -0
  298. package/src/common/writer.ts +123 -0
  299. package/src/common/xml.ts +1053 -0
  300. package/src/formats/csv/exporter.ts +894 -0
  301. package/src/formats/csv/header.ts +172 -0
  302. package/src/formats/csv/importer.ts +1104 -0
  303. package/src/formats/csv/index.ts +88 -0
  304. package/src/formats/csv/records.ts +813 -0
  305. package/src/formats/csv/values.ts +224 -0
  306. package/src/formats/dot/exporter.ts +1014 -0
  307. package/src/formats/dot/importer.ts +1549 -0
  308. package/src/formats/dot/index.ts +7 -0
  309. package/src/formats/dot/names.ts +40 -0
  310. package/src/formats/dot/tokenizer.ts +384 -0
  311. package/src/formats/gexf/exporter.ts +1696 -0
  312. package/src/formats/gexf/importer.ts +2333 -0
  313. package/src/formats/gexf/index.ts +142 -0
  314. package/src/formats/gexf/schema.ts +361 -0
  315. package/src/formats/gml/exporter.ts +1404 -0
  316. package/src/formats/gml/importer.ts +1591 -0
  317. package/src/formats/gml/index.ts +128 -0
  318. package/src/formats/gml/syntax.ts +545 -0
  319. package/src/formats/graphml/constants.ts +225 -0
  320. package/src/formats/graphml/exporter.ts +1458 -0
  321. package/src/formats/graphml/importer.ts +2027 -0
  322. package/src/formats/graphml/index.ts +8 -0
  323. package/src/formats/graphml/tree.ts +318 -0
  324. package/src/formats/json/dialect.ts +317 -0
  325. package/src/formats/json/exporter.ts +1616 -0
  326. package/src/formats/json/importer.ts +2271 -0
  327. package/src/formats/json/index.ts +8 -0
  328. package/src/formats/neo4j/exporter.ts +1287 -0
  329. package/src/formats/neo4j/header.ts +156 -0
  330. package/src/formats/neo4j/importer.ts +1220 -0
  331. package/src/formats/neo4j/index.ts +116 -0
  332. package/src/formats/pajek/exporter.ts +1000 -0
  333. package/src/formats/pajek/importer.ts +1311 -0
  334. package/src/formats/pajek/index.ts +7 -0
  335. package/src/formats/pajek/syntax.ts +307 -0
  336. package/src/index.ts +244 -0
  337. package/src/registry.ts +617 -0
  338. package/src/sniff.ts +397 -0
  339. package/src/types.ts +262 -0
@@ -0,0 +1,1053 @@
1
+ /**
2
+ * The streaming XML tokenizer shared by the GEXF and GraphML importers (design sections 8.2 and
3
+ * 8.4): SAX-style start / end / text events over the text chunks of the common reader, so both
4
+ * importers run in one pass with bounded memory and every issue carries a line number.
5
+ *
6
+ * Cost model: every character of the input is scanned once. A token that spans many chunks (a
7
+ * long attribute value, comment, CDATA section or name) resumes where the previous chunk left off
8
+ * instead of re-reading the token from its `<`, and line numbers are counted from a cached next
9
+ * line-break position, so a document without line breaks costs the same as one with them.
10
+ *
11
+ * What it handles: the XML declaration and processing instructions (skipped), comments (skipped),
12
+ * CDATA sections (text), a DOCTYPE with an internal subset (skipped; entity declarations are not
13
+ * expanded, so an unknown entity reference is a syntax error), the five predefined entities and
14
+ * numeric character references (decimal and hexadecimal) in text and attribute values, attribute
15
+ * value whitespace normalisation, end-of-line normalisation (CR LF and lone CR become LF), and
16
+ * well-formedness: matching end tags, one root element, no text outside it, no duplicate
17
+ * attributes, no unterminated markup at the end of the input. Namespaces are not resolved; element
18
+ * and attribute names are reported as written (with their prefix).
19
+ *
20
+ * Why not fast-xml-parser: its XMLParser accepts mismatched and unclosed tags without error, does
21
+ * not decode numeric character references unless the deprecated `htmlEntities` option is set, and
22
+ * its validator (`XMLValidator`, the `parse(xml, true)` overload) and `XMLBuilder` are deprecated in
23
+ * 5.x, which the root lint's `no-deprecated` rule forbids using; so the malformed corpus could not
24
+ * be rejected and attribute values written by the common `escapeXmlAttribute` (`&#10;`) could not
25
+ * be read back.
26
+ */
27
+
28
+ import { type Column, type GraphSnapshot } from "@graphty/graph-format";
29
+
30
+ import { type LossNote } from "../types.js";
31
+ import { XML_ILLEGAL_CHAR_CODE } from "./codes.js";
32
+ import { isNameChar } from "./export.js";
33
+
34
+ /** The events of the tokenizer; every callback is synchronous. */
35
+ export interface XmlHandler {
36
+ /**
37
+ * An element starts (a self-closing element produces start then end).
38
+ * @param name - the element name as written, prefix included
39
+ * @param attrs - the attributes, entities decoded, in document order
40
+ * @param line - the 1-based line of the `<`
41
+ */
42
+ start(name: string, attrs: ReadonlyMap<string, string>, line: number): void;
43
+ /**
44
+ * An element ends.
45
+ * @param name - the element name
46
+ * @param line - the 1-based line of the end tag (or of the self-closing start tag)
47
+ */
48
+ end(name: string, line: number): void;
49
+ /**
50
+ * Character data between two tags (entities decoded, CDATA included), one call per run;
51
+ * whitespace-only runs are delivered too.
52
+ * @param text - the text
53
+ * @param line - the 1-based line where the run starts
54
+ */
55
+ text(text: string, line: number): void;
56
+ }
57
+
58
+ /** A well-formedness or syntax error, with the line it was found on. */
59
+ export class XmlSyntaxError extends Error {
60
+ /** The 1-based line. */
61
+ readonly line: number;
62
+
63
+ /**
64
+ * Create the error.
65
+ * @param message - a plain-ASCII message
66
+ * @param line - the 1-based line
67
+ */
68
+ constructor(message: string, line: number) {
69
+ super(message);
70
+ this.name = "XmlSyntaxError";
71
+ this.line = line;
72
+ }
73
+ }
74
+
75
+ const NAMED_ENTITIES: Readonly<Record<string, string>> = {
76
+ lt: "<",
77
+ gt: ">",
78
+ amp: "&",
79
+ quot: '"',
80
+ apos: "'",
81
+ };
82
+
83
+ const LT = 60;
84
+ const GT = 62;
85
+ const SLASH = 47;
86
+ const EQUALS = 61;
87
+ const QUOTE = 34;
88
+ const APOS = 39;
89
+ const OPEN_BRACKET = 91;
90
+ const CLOSE_BRACKET = 93;
91
+ const BANG = 33;
92
+ const DASH = 45;
93
+
94
+ /** The longest entity reference held back at a chunk boundary (`&#x10FFFF;` is 10 characters). */
95
+ const MAX_ENTITY_LENGTH = 16;
96
+
97
+ /** Markup shorter than this (`<![CDATA[` is 9 characters) cannot be classified yet. */
98
+ const MIN_DECIDABLE_MARKUP = 9;
99
+
100
+ /**
101
+ * Whether a char code is XML whitespace (space, tab, LF, CR).
102
+ * @param c - the char code
103
+ * @returns true for whitespace
104
+ */
105
+ function isSpace(c: number): boolean {
106
+ return c === 32 || c === 9 || c === 10 || c === 13;
107
+ }
108
+
109
+ /**
110
+ * Whether a code point may start an XML Name: a NameChar that is not a digit, `-`, `.`, U+00B7,
111
+ * a combining character (U+0300..U+036F) or U+203F / U+2040.
112
+ * @param cp - the code point
113
+ * @returns true for a NameStartChar
114
+ */
115
+ function isNameStart(cp: number): boolean {
116
+ if (!isNameChar(cp)) {
117
+ return false;
118
+ }
119
+ if ((cp >= 0x30 && cp <= 0x39) || cp === 0x2d || cp === 0x2e || cp === 0xb7) {
120
+ return false;
121
+ }
122
+ return !(cp >= 0x300 && cp <= 0x36f) && cp !== 0x203f && cp !== 0x2040;
123
+ }
124
+
125
+ /** ASCII code -> whether it is a NameStartChar / NameChar, for the fast path of readName(). */
126
+ const ASCII_NAME_START = new Uint8Array(128);
127
+ const ASCII_NAME_CHAR = new Uint8Array(128);
128
+ for (let c = 0; c < 128; c++) {
129
+ ASCII_NAME_START[c] = isNameStart(c) ? 1 : 0;
130
+ ASCII_NAME_CHAR[c] = isNameChar(c) ? 1 : 0;
131
+ }
132
+
133
+ /**
134
+ * Characters XML 1.0 forbids in a document (the Char production): C0 controls other than tab, LF
135
+ * and CR, U+FFFE / U+FFFF, and lone surrogates (the `u` flag makes the surrogate class match only
136
+ * an unpaired one). Built from char codes so no control character appears in the source.
137
+ */
138
+ const ILLEGAL_CHAR = new RegExp(
139
+ `[${String.fromCharCode(0)}-${String.fromCharCode(8)}${String.fromCharCode(11)}${String.fromCharCode(12)}` +
140
+ `${String.fromCharCode(14)}-${String.fromCharCode(31)}\\uFFFE\\uFFFF\\uD800-\\uDFFF]`,
141
+ "u",
142
+ );
143
+
144
+ /**
145
+ * Whether a text holds a character XML 1.0 forbids (see ILLEGAL_CHAR); such a character cannot be
146
+ * written even as a character reference, so a conforming parser rejects the whole document.
147
+ * @param text - the text
148
+ * @returns true when the text cannot appear in an XML 1.0 document
149
+ */
150
+ export function hasIllegalXmlChar(text: string): boolean {
151
+ return ILLEGAL_CHAR.test(text);
152
+ }
153
+
154
+ /**
155
+ * The loss notes of the texts a snapshot holds that no XML 1.0 document can carry (see
156
+ * hasIllegalXmlChar): one E_XML_ILLEGAL_CHAR note per column (node, edge and graph string / dict /
157
+ * list-of-text columns) and one for the ids, each counting the affected rows; export() throws on
158
+ * the first such text. Shared by the GEXF and GraphML exporters' check().
159
+ * @param snapshot - the snapshot
160
+ * @returns the notes, empty when every text is writable
161
+ */
162
+ export function xmlIllegalTextNotes(snapshot: GraphSnapshot): LossNote[] {
163
+ const notes: LossNote[] = [];
164
+ let ids = 0;
165
+ for (let i = 0; i < snapshot.nodeCount; i++) {
166
+ const id = snapshot.ids.idOf(i);
167
+ if (typeof id === "string" && hasIllegalXmlChar(id)) {
168
+ ids++;
169
+ }
170
+ }
171
+ if (ids > 0) {
172
+ notes.push(
173
+ Object.freeze({
174
+ code: XML_ILLEGAL_CHAR_CODE,
175
+ message: `${ids} node id(s) hold a character XML 1.0 cannot carry; export() will throw`,
176
+ column: null,
177
+ count: ids,
178
+ }),
179
+ );
180
+ }
181
+ for (const [domain, table] of [
182
+ ["node", snapshot.nodes],
183
+ ["edge", snapshot.edges],
184
+ ["graph", snapshot.graph],
185
+ ] as const) {
186
+ for (const column of table) {
187
+ const bad = countIllegalRows(column);
188
+ if (bad > 0) {
189
+ notes.push(
190
+ Object.freeze({
191
+ code: XML_ILLEGAL_CHAR_CODE,
192
+ message: `${domain} column "${column.meta.name}": ${bad} value(s) hold a character XML 1.0 cannot carry; export() will throw`,
193
+ column: column.meta.name,
194
+ count: bad,
195
+ }),
196
+ );
197
+ }
198
+ }
199
+ }
200
+ return notes;
201
+ }
202
+
203
+ /**
204
+ * How many set rows of a text column hold an XML-illegal character.
205
+ * @param column - the column
206
+ * @returns the count; 0 for a non-text column
207
+ */
208
+ function countIllegalRows(column: Column): number {
209
+ let bad = 0;
210
+ if (column.dtype === "string" || column.dtype === "dict") {
211
+ for (let r = 0; r < column.length; r++) {
212
+ if (column.isSet(r) && hasIllegalXmlChar(column.value(r) as string)) {
213
+ bad++;
214
+ }
215
+ }
216
+ } else if (column.dtype === "list" && (column.meta.itemDtype === "string" || column.meta.itemDtype === "dict")) {
217
+ for (let r = 0; r < column.length; r++) {
218
+ if (
219
+ column.isSet(r) &&
220
+ column.sliceOf(r).some((item) => typeof item === "string" && hasIllegalXmlChar(item))
221
+ ) {
222
+ bad++;
223
+ }
224
+ }
225
+ }
226
+ return bad;
227
+ }
228
+
229
+ /**
230
+ * Whether a text is an XML Name.
231
+ * @param text - the text
232
+ * @returns true for a well-formed name
233
+ */
234
+ export function isXmlName(text: string): boolean {
235
+ if (text.length === 0) {
236
+ return false;
237
+ }
238
+ let first = true;
239
+ for (const ch of text) {
240
+ const cp = ch.codePointAt(0);
241
+ if (cp === undefined || (first ? !isNameStart(cp) : !isNameChar(cp))) {
242
+ return false;
243
+ }
244
+ first = false;
245
+ }
246
+ return true;
247
+ }
248
+
249
+ /**
250
+ * Whether a code point is a legal XML 1.0 character.
251
+ * @param cp - the code point
252
+ * @returns true when a character reference may produce it
253
+ */
254
+ function isXmlChar(cp: number): boolean {
255
+ return (
256
+ cp === 0x9 ||
257
+ cp === 0xa ||
258
+ cp === 0xd ||
259
+ (cp >= 0x20 && cp <= 0xd7ff) ||
260
+ (cp >= 0xe000 && cp <= 0xfffd) ||
261
+ (cp >= 0x10000 && cp <= 0x10ffff)
262
+ );
263
+ }
264
+
265
+ /**
266
+ * Decode the predefined entities and character references of a text.
267
+ * @param raw - the text as written
268
+ * @param line - the line, for errors
269
+ * @returns the decoded text
270
+ */
271
+ export function decodeEntities(raw: string, line: number): string {
272
+ let amp = raw.indexOf("&");
273
+ if (amp < 0) {
274
+ return raw;
275
+ }
276
+ let out = "";
277
+ let start = 0;
278
+ while (amp >= 0) {
279
+ out += raw.slice(start, amp);
280
+ const semi = raw.indexOf(";", amp + 1);
281
+ if (semi < 0) {
282
+ throw new XmlSyntaxError("unterminated entity reference", line);
283
+ }
284
+ const name = raw.slice(amp + 1, semi);
285
+ if (name.startsWith("#")) {
286
+ const hex = name.startsWith("#x") || name.startsWith("#X");
287
+ const digits = name.slice(hex ? 2 : 1);
288
+ const ok = hex ? /^[0-9a-fA-F]{1,6}$/.test(digits) : /^[0-9]{1,7}$/.test(digits);
289
+ const cp = ok ? Number.parseInt(digits, hex ? 16 : 10) : -1;
290
+ if (cp < 0 || !isXmlChar(cp)) {
291
+ throw new XmlSyntaxError(`invalid character reference &${name};`, line);
292
+ }
293
+ out += String.fromCodePoint(cp);
294
+ } else {
295
+ const value = NAMED_ENTITIES[name];
296
+ if (value === undefined) {
297
+ throw new XmlSyntaxError(`unknown entity &${name};`, line);
298
+ }
299
+ out += value;
300
+ }
301
+ start = semi + 1;
302
+ amp = raw.indexOf("&", start);
303
+ }
304
+ return out + raw.slice(start);
305
+ }
306
+
307
+ /**
308
+ * Whether a text is whitespace only.
309
+ * @param text - the text
310
+ * @returns true when every character is XML whitespace (also for an empty text)
311
+ */
312
+ export function isWhitespace(text: string): boolean {
313
+ for (let i = 0; i < text.length; i++) {
314
+ if (!isSpace(text.charCodeAt(i))) {
315
+ return false;
316
+ }
317
+ }
318
+ return true;
319
+ }
320
+
321
+ /**
322
+ * The local part of a possibly prefixed XML name.
323
+ * @param name - the name as written
324
+ * @returns the text after the last colon, or the name itself
325
+ */
326
+ export function localName(name: string): string {
327
+ const colon = name.lastIndexOf(":");
328
+ return colon < 0 ? name : name.slice(colon + 1);
329
+ }
330
+
331
+ /**
332
+ * Tokenize XML text arriving in chunks and deliver events to a handler. Throws XmlSyntaxError on
333
+ * malformed input; any error the handler throws propagates unchanged.
334
+ * @param chunks - the text (already UTF-8 decoded, BOM removed)
335
+ * @param handler - the event sink
336
+ */
337
+ export async function tokenizeXml(chunks: AsyncIterable<string>, handler: XmlHandler): Promise<void> {
338
+ const tokenizer = new XmlTokenizer(handler);
339
+ for await (const chunk of chunks) {
340
+ tokenizer.push(chunk);
341
+ }
342
+ tokenizer.finish();
343
+ }
344
+
345
+ /** The result of parsing one start tag out of the buffer. */
346
+ interface StartTag {
347
+ /** The element name. */
348
+ readonly name: string;
349
+ /** The attributes. */
350
+ readonly attrs: Map<string, string>;
351
+ /** Whether the tag ends with `/>`. */
352
+ readonly selfClosing: boolean;
353
+ /** The buffer index one past the `>`. */
354
+ readonly end: number;
355
+ }
356
+
357
+ /** The kinds of markup whose end may lie in a later chunk. */
358
+ type PendingKind = "decl" | "comment" | "cdata" | "pi" | "endtag" | "doctype" | "starttag";
359
+
360
+ /** The terminator of each fixed-terminator markup kind. */
361
+ const TERMINATORS: Readonly<Partial<Record<PendingKind, string>>> = {
362
+ comment: "-->",
363
+ cdata: "]]>",
364
+ pi: "?>",
365
+ endtag: ">",
366
+ };
367
+
368
+ /**
369
+ * An incomplete piece of markup at the end of the input so far: its text pieces (never joined
370
+ * until the end is found, so a token spanning many chunks costs its length once) and the state
371
+ * the terminator scan is in. `tail` holds the last characters of the pieces a fixed terminator
372
+ * could straddle; `quote` is the open quote character (0 outside a value) of a start tag or a
373
+ * DOCTYPE; `depth`, `comment` and `run` track a DOCTYPE's internal subset.
374
+ */
375
+ interface Pending {
376
+ kind: PendingKind;
377
+ readonly pieces: string[];
378
+ tail: string;
379
+ quote: number;
380
+ depth: number;
381
+ comment: boolean;
382
+ run: number;
383
+ }
384
+
385
+ /**
386
+ * The tokenizer state: the unconsumed tail of the input, the current line, the open element
387
+ * stack and the text run being accumulated. `push()` chunks, then `finish()`.
388
+ */
389
+ export class XmlTokenizer {
390
+ private readonly handler: XmlHandler;
391
+
392
+ /** Unconsumed input that is not part of a pending token: at most a held-back entity or the last chunk. */
393
+ private buffer = "";
394
+
395
+ /** The markup whose end has not arrived yet, or null. */
396
+ private pending: Pending | null = null;
397
+
398
+ /** The line number at the start of `buffer` (or of the pending markup). */
399
+ private line = 1;
400
+
401
+ /**
402
+ * The index of the next line break in `buffer` at or after the last counted position: -2
403
+ * when not yet searched for the current buffer, -1 when the buffer holds no further break.
404
+ */
405
+ private nextBreak = -2;
406
+
407
+ /** A CR held back from the end of the previous chunk (it may be the first half of CR LF). */
408
+ private pendingCr = false;
409
+
410
+ /** Whether the input has ended (an incomplete token is then an error, not a wait). */
411
+ private final = false;
412
+
413
+ /** The text run being accumulated (entities decoded), and the line it started on. */
414
+ private text = "";
415
+
416
+ private textLine = 1;
417
+
418
+ private readonly stack: string[] = [];
419
+
420
+ private rootSeen = false;
421
+
422
+ private rootClosed = false;
423
+
424
+ /**
425
+ * Create a tokenizer.
426
+ * @param handler - the event sink
427
+ */
428
+ constructor(handler: XmlHandler) {
429
+ this.handler = handler;
430
+ }
431
+
432
+ /**
433
+ * Feed one chunk and emit every complete token in it.
434
+ * @param chunk - the text
435
+ */
436
+ push(chunk: string): void {
437
+ let text = chunk;
438
+ if (this.pendingCr) {
439
+ text = `\r${text}`;
440
+ this.pendingCr = false;
441
+ }
442
+ if (text.endsWith("\r")) {
443
+ this.pendingCr = true;
444
+ text = text.slice(0, -1);
445
+ }
446
+ if (text.includes("\r")) {
447
+ text = text.replace(/\r\n?/g, "\n");
448
+ }
449
+ this.feed(text);
450
+ }
451
+
452
+ /** Signal the end of the input: flush the last text run and check well-formedness. */
453
+ finish(): void {
454
+ this.final = true;
455
+ if (this.pendingCr) {
456
+ this.pendingCr = false;
457
+ this.feed("\n");
458
+ } else {
459
+ this.feed("");
460
+ }
461
+ if (this.pending !== null || this.buffer.length > 0) {
462
+ throw new XmlSyntaxError("unexpected end of input inside markup", this.line);
463
+ }
464
+ this.flushText();
465
+ if (this.stack.length > 0) {
466
+ throw new XmlSyntaxError(`unclosed element <${this.stack[this.stack.length - 1]}>`, this.line);
467
+ }
468
+ if (!this.rootSeen) {
469
+ throw new XmlSyntaxError("no root element", this.line);
470
+ }
471
+ }
472
+
473
+ /**
474
+ * Append normalised text: continue a pending token's terminator scan over the new text alone,
475
+ * and once the token is complete (or when none is pending) scan the buffer for tokens.
476
+ * @param text - the text, line breaks normalised
477
+ */
478
+ private feed(text: string): void {
479
+ const { pending } = this;
480
+ if (pending === null) {
481
+ this.buffer = this.buffer.length === 0 ? text : this.buffer + text;
482
+ this.nextBreak = -2;
483
+ this.scan();
484
+ return;
485
+ }
486
+ if (pending.kind === "decl") {
487
+ // fewer than MIN_DECIDABLE_MARKUP characters of markup: the kind is decided once enough arrived
488
+ this.pending = null;
489
+ this.buffer = pending.pieces.join("") + text;
490
+ this.nextBreak = -2;
491
+ this.scan();
492
+ return;
493
+ }
494
+ const end = this.scanPending(pending, text);
495
+ if (end < 0) {
496
+ if (text.length > 0) {
497
+ pending.pieces.push(text);
498
+ }
499
+ if (this.final) {
500
+ throw new XmlSyntaxError("unexpected end of input inside markup", this.line);
501
+ }
502
+ return;
503
+ }
504
+ this.pending = null;
505
+ pending.pieces.push(text.slice(0, end));
506
+ this.buffer = pending.pieces.join("") + text.slice(end);
507
+ this.nextBreak = -2;
508
+ this.scan();
509
+ }
510
+
511
+ /**
512
+ * Continue the terminator scan of a pending token over new text.
513
+ * @param pending - the pending token
514
+ * @param text - the new text
515
+ * @returns the index in `text` one past the token's end, or -1 when it does not end there
516
+ */
517
+ private scanPending(pending: Pending, text: string): number {
518
+ switch (pending.kind) {
519
+ case "starttag":
520
+ return this.scanStartTag(pending, text, 0);
521
+ case "doctype":
522
+ return this.scanDoctype(pending, text, 0);
523
+ case "decl":
524
+ return -1;
525
+ default: {
526
+ const terminator = TERMINATORS[pending.kind] ?? ">";
527
+ const probe = pending.tail + text;
528
+ const at = probe.indexOf(terminator);
529
+ if (at < 0) {
530
+ pending.tail = probe.slice(Math.max(0, probe.length - (terminator.length - 1)));
531
+ return -1;
532
+ }
533
+ return at + terminator.length - pending.tail.length;
534
+ }
535
+ }
536
+ }
537
+
538
+ /**
539
+ * Scan text for the `>` that ends a start tag, skipping quoted attribute values (the quote
540
+ * state survives between calls).
541
+ * @param pending - the pending token
542
+ * @param text - the text
543
+ * @param from - where to start
544
+ * @returns the index one past the `>`, or -1
545
+ */
546
+ private scanStartTag(pending: Pending, text: string, from: number): number {
547
+ let { quote } = pending;
548
+ const n = text.length;
549
+ let i = from;
550
+ while (i < n) {
551
+ if (quote !== 0) {
552
+ // inside an attribute value: jump to its closing quote
553
+ const close = text.indexOf(String.fromCharCode(quote), i);
554
+ if (close < 0) {
555
+ break;
556
+ }
557
+ quote = 0;
558
+ i = close + 1;
559
+ continue;
560
+ }
561
+ const c = text.charCodeAt(i);
562
+ if (c === QUOTE || c === APOS) {
563
+ quote = c;
564
+ } else if (c === GT) {
565
+ pending.quote = 0;
566
+ return i + 1;
567
+ }
568
+ i++;
569
+ }
570
+ pending.quote = quote;
571
+ return -1;
572
+ }
573
+
574
+ /**
575
+ * Scan text for the `>` that ends a DOCTYPE declaration, skipping a bracketed internal subset,
576
+ * quoted literals and comments inside the subset. A character state machine, so the state
577
+ * (`depth`, `quote`, `comment`, and `run`, the progress through a `<!--` or the dashes before
578
+ * a `-->`) survives between calls and nothing is re-read.
579
+ * @param pending - the pending token
580
+ * @param text - the text
581
+ * @param from - where to start
582
+ * @returns the index one past the `>`, or -1
583
+ */
584
+ private scanDoctype(pending: Pending, text: string, from: number): number {
585
+ let { quote, depth, comment, run } = pending;
586
+ for (let i = from; i < text.length; i++) {
587
+ const c = text.charCodeAt(i);
588
+ if (comment) {
589
+ if (c === DASH) {
590
+ run++;
591
+ } else if (c === GT && run >= 2) {
592
+ comment = false;
593
+ run = 0;
594
+ } else {
595
+ run = 0;
596
+ }
597
+ continue;
598
+ }
599
+ if (quote !== 0) {
600
+ if (c === quote) {
601
+ quote = 0;
602
+ }
603
+ continue;
604
+ }
605
+ if (depth > 0) {
606
+ if (run === 1 && c === BANG) {
607
+ run = 2;
608
+ continue;
609
+ }
610
+ if (run === 2 && c === DASH) {
611
+ run = 3;
612
+ continue;
613
+ }
614
+ if (run === 3 && c === DASH) {
615
+ comment = true;
616
+ run = 0;
617
+ continue;
618
+ }
619
+ run = 0;
620
+ if (c === LT) {
621
+ run = 1;
622
+ continue;
623
+ }
624
+ }
625
+ if (c === QUOTE || c === APOS) {
626
+ quote = c;
627
+ } else if (c === OPEN_BRACKET) {
628
+ depth++;
629
+ } else if (c === CLOSE_BRACKET) {
630
+ depth--;
631
+ } else if (c === GT && depth <= 0) {
632
+ pending.quote = 0;
633
+ pending.depth = 0;
634
+ pending.comment = false;
635
+ pending.run = 0;
636
+ return i + 1;
637
+ }
638
+ }
639
+ pending.quote = quote;
640
+ pending.depth = depth;
641
+ pending.comment = comment;
642
+ pending.run = run;
643
+ return -1;
644
+ }
645
+
646
+ /**
647
+ * Consume every complete token at the front of the buffer; an incomplete token at its end
648
+ * becomes the pending token (its text moved out of the buffer), a possibly split entity
649
+ * reference is held back in the buffer.
650
+ */
651
+ private scan(): void {
652
+ const { buffer } = this;
653
+ const { length } = buffer;
654
+ let pos = 0;
655
+ let incomplete = false;
656
+ for (;;) {
657
+ const lt = buffer.indexOf("<", pos);
658
+ if (lt < 0) {
659
+ let end = length;
660
+ if (!this.final) {
661
+ // an entity reference may be split across chunks: hold back from its "&"
662
+ const amp = buffer.lastIndexOf("&");
663
+ if (amp >= pos && amp >= length - MAX_ENTITY_LENGTH && buffer.indexOf(";", amp) < 0) {
664
+ end = amp;
665
+ }
666
+ }
667
+ this.takeText(pos, end);
668
+ pos = end;
669
+ break;
670
+ }
671
+ if (lt > pos) {
672
+ this.takeText(pos, lt);
673
+ pos = lt;
674
+ }
675
+ const next = this.consumeMarkup(pos);
676
+ if (next < 0) {
677
+ incomplete = true;
678
+ break;
679
+ }
680
+ pos = next;
681
+ }
682
+ if (incomplete) {
683
+ this.pending = this.startPending(buffer, pos);
684
+ this.buffer = "";
685
+ } else {
686
+ this.buffer = pos === 0 ? buffer : buffer.slice(pos);
687
+ }
688
+ this.nextBreak = -2;
689
+ }
690
+
691
+ /**
692
+ * Turn the incomplete markup at `pos` into a pending token, running the terminator scan over
693
+ * the part already in the buffer so later chunks are scanned alone.
694
+ * @param buffer - the buffer
695
+ * @param pos - the index of the `<`
696
+ * @returns the pending token
697
+ */
698
+ private startPending(buffer: string, pos: number): Pending {
699
+ const piece = pos === 0 ? buffer : buffer.slice(pos);
700
+ const pending: Pending = {
701
+ kind: "starttag",
702
+ pieces: [piece],
703
+ tail: "",
704
+ quote: 0,
705
+ depth: 0,
706
+ comment: false,
707
+ run: 0,
708
+ };
709
+ if (piece.length < MIN_DECIDABLE_MARKUP) {
710
+ // too short to tell a comment from a start tag: wait for more before choosing a scan
711
+ pending.kind = "decl";
712
+ return pending;
713
+ }
714
+ if (piece.startsWith("<!--")) {
715
+ pending.kind = "comment";
716
+ } else if (piece.startsWith("<![CDATA[")) {
717
+ pending.kind = "cdata";
718
+ } else if (piece.startsWith("<!DOCTYPE")) {
719
+ pending.kind = "doctype";
720
+ } else if (piece.startsWith("<!")) {
721
+ pending.kind = "decl";
722
+ return pending;
723
+ } else if (piece.startsWith("<?")) {
724
+ pending.kind = "pi";
725
+ } else if (piece.startsWith("</")) {
726
+ pending.kind = "endtag";
727
+ }
728
+ // the part in the buffer holds no terminator (consumeMarkup said so); record the scan state
729
+ switch (pending.kind) {
730
+ case "starttag":
731
+ this.scanStartTag(pending, piece, 1);
732
+ break;
733
+ case "doctype":
734
+ this.scanDoctype(pending, piece, 9);
735
+ break;
736
+ default: {
737
+ const terminator = TERMINATORS[pending.kind] ?? ">";
738
+ pending.tail = piece.slice(Math.max(0, piece.length - (terminator.length - 1)));
739
+ break;
740
+ }
741
+ }
742
+ return pending;
743
+ }
744
+
745
+ /**
746
+ * Consume the markup starting at `pos` (a `<`).
747
+ * @param pos - the index of the `<`
748
+ * @returns the index after the markup, or -1 when the buffer ends before the markup does
749
+ */
750
+ private consumeMarkup(pos: number): number {
751
+ const { buffer } = this;
752
+ if (buffer.startsWith("<!", pos)) {
753
+ if (buffer.length - pos < MIN_DECIDABLE_MARKUP && !this.final) {
754
+ // "<![CDATA[" is 9 characters; wait until the kind of declaration is decidable
755
+ return -1;
756
+ }
757
+ if (buffer.startsWith("<!--", pos)) {
758
+ const end = buffer.indexOf("-->", pos + 4);
759
+ if (end < 0) {
760
+ return -1;
761
+ }
762
+ this.advanceLine(pos, end + 3);
763
+ return end + 3;
764
+ }
765
+ if (buffer.startsWith("<![CDATA[", pos)) {
766
+ const end = buffer.indexOf("]]>", pos + 9);
767
+ if (end < 0) {
768
+ return -1;
769
+ }
770
+ if (this.text.length === 0) {
771
+ this.textLine = this.line;
772
+ }
773
+ const cdata = buffer.slice(pos + 9, end);
774
+ if (hasIllegalXmlChar(cdata)) {
775
+ throw new XmlSyntaxError("a character XML 1.0 forbids appears in a CDATA section", this.line);
776
+ }
777
+ this.text += cdata;
778
+ this.advanceLine(pos, end + 3);
779
+ return end + 3;
780
+ }
781
+ if (!buffer.startsWith("<!DOCTYPE", pos)) {
782
+ throw new XmlSyntaxError("unexpected markup declaration", this.line);
783
+ }
784
+ const state: Pending = {
785
+ kind: "doctype",
786
+ pieces: [],
787
+ tail: "",
788
+ quote: 0,
789
+ depth: 0,
790
+ comment: false,
791
+ run: 0,
792
+ };
793
+ const end = this.scanDoctype(state, buffer, pos + 9);
794
+ if (end < 0) {
795
+ return -1;
796
+ }
797
+ this.advanceLine(pos, end);
798
+ return end;
799
+ }
800
+ if (buffer.startsWith("<?", pos)) {
801
+ const end = buffer.indexOf("?>", pos + 2);
802
+ if (end < 0) {
803
+ return -1;
804
+ }
805
+ this.advanceLine(pos, end + 2);
806
+ return end + 2;
807
+ }
808
+ if (buffer.startsWith("</", pos)) {
809
+ const end = buffer.indexOf(">", pos + 2);
810
+ if (end < 0) {
811
+ return -1;
812
+ }
813
+ const name = buffer.slice(pos + 2, end).trim();
814
+ if (!isXmlName(name)) {
815
+ throw new XmlSyntaxError(`malformed end tag </${name}>`, this.line);
816
+ }
817
+ this.flushText();
818
+ this.endElement(name);
819
+ this.advanceLine(pos, end + 1);
820
+ return end + 1;
821
+ }
822
+ const tag = this.parseStartTag(pos);
823
+ if (tag === null) {
824
+ return -1;
825
+ }
826
+ this.flushText();
827
+ this.startElement(tag.name, tag.attrs);
828
+ if (tag.selfClosing) {
829
+ this.endElement(tag.name);
830
+ }
831
+ this.advanceLine(pos, tag.end);
832
+ return tag.end;
833
+ }
834
+
835
+ /**
836
+ * Parse a start tag at `pos`.
837
+ * @param pos - the index of the `<`
838
+ * @returns the tag, or null when the buffer ends inside it
839
+ */
840
+ private parseStartTag(pos: number): StartTag | null {
841
+ const { buffer } = this;
842
+ const { length } = buffer;
843
+ let i = pos + 1;
844
+ const nameEnd = this.readName(i);
845
+ if (nameEnd < 0) {
846
+ return null;
847
+ }
848
+ if (nameEnd === i) {
849
+ throw new XmlSyntaxError("expected an element name after <", this.line);
850
+ }
851
+ const name = buffer.slice(i, nameEnd);
852
+ i = nameEnd;
853
+ const attrs = new Map<string, string>();
854
+ for (;;) {
855
+ while (i < length && isSpace(buffer.charCodeAt(i))) {
856
+ i++;
857
+ }
858
+ if (i >= length) {
859
+ return null;
860
+ }
861
+ const c = buffer.charCodeAt(i);
862
+ if (c === GT) {
863
+ return { name, attrs, selfClosing: false, end: i + 1 };
864
+ }
865
+ if (c === SLASH) {
866
+ if (i + 1 >= length) {
867
+ return null;
868
+ }
869
+ if (buffer.charCodeAt(i + 1) !== GT) {
870
+ throw new XmlSyntaxError(`unexpected "/" in <${name}>`, this.line);
871
+ }
872
+ return { name, attrs, selfClosing: true, end: i + 2 };
873
+ }
874
+ const attrEnd = this.readName(i);
875
+ if (attrEnd < 0) {
876
+ return null;
877
+ }
878
+ if (attrEnd === i) {
879
+ throw new XmlSyntaxError(`malformed attribute in <${name}>`, this.line);
880
+ }
881
+ const attrName = buffer.slice(i, attrEnd);
882
+ i = attrEnd;
883
+ while (i < length && isSpace(buffer.charCodeAt(i))) {
884
+ i++;
885
+ }
886
+ if (i >= length) {
887
+ return null;
888
+ }
889
+ if (buffer.charCodeAt(i) !== EQUALS) {
890
+ throw new XmlSyntaxError(`attribute ${attrName} of <${name}> has no value`, this.line);
891
+ }
892
+ i++;
893
+ while (i < length && isSpace(buffer.charCodeAt(i))) {
894
+ i++;
895
+ }
896
+ if (i >= length) {
897
+ return null;
898
+ }
899
+ const quote = buffer.charCodeAt(i);
900
+ if (quote !== QUOTE && quote !== APOS) {
901
+ throw new XmlSyntaxError(`attribute ${attrName} of <${name}> is not quoted`, this.line);
902
+ }
903
+ const close = buffer.indexOf(quote === QUOTE ? '"' : "'", i + 1);
904
+ if (close < 0) {
905
+ return null;
906
+ }
907
+ if (attrs.has(attrName)) {
908
+ throw new XmlSyntaxError(`duplicate attribute ${attrName} in <${name}>`, this.line);
909
+ }
910
+ const raw = buffer.slice(i + 1, close);
911
+ if (hasIllegalXmlChar(raw)) {
912
+ throw new XmlSyntaxError(
913
+ `a character XML 1.0 forbids appears in attribute ${attrName} of <${name}>`,
914
+ this.line,
915
+ );
916
+ }
917
+ attrs.set(attrName, decodeEntities(normalizeAttributeValue(raw), this.line));
918
+ i = close + 1;
919
+ }
920
+ }
921
+
922
+ /**
923
+ * The end of the XML Name starting at `i` (an ASCII table for the common case, the code point
924
+ * classes beyond it).
925
+ * @param i - the start index
926
+ * @returns the index after the name; `i` when no name starts there; -1 when the name may
927
+ * continue past the end of the buffer
928
+ */
929
+ private readName(i: number): number {
930
+ const { buffer } = this;
931
+ const { length } = buffer;
932
+ let j = i;
933
+ while (j < length) {
934
+ const c = buffer.charCodeAt(j);
935
+ if (c < 0x80) {
936
+ if ((j === i ? ASCII_NAME_START[c] : ASCII_NAME_CHAR[c]) === 0) {
937
+ break;
938
+ }
939
+ j++;
940
+ continue;
941
+ }
942
+ const cp = buffer.codePointAt(j);
943
+ if (cp === undefined || (j === i ? !isNameStart(cp) : !isNameChar(cp))) {
944
+ break;
945
+ }
946
+ j += cp > 0xffff ? 2 : 1;
947
+ }
948
+ return j >= length && !this.final ? -1 : j;
949
+ }
950
+
951
+ /**
952
+ * Move text from the buffer into the current run.
953
+ * @param start - the start index
954
+ * @param end - the end index (exclusive)
955
+ */
956
+ private takeText(start: number, end: number): void {
957
+ if (end <= start) {
958
+ return;
959
+ }
960
+ if (this.text.length === 0) {
961
+ this.textLine = this.line;
962
+ }
963
+ const raw = this.buffer.slice(start, end);
964
+ if (hasIllegalXmlChar(raw)) {
965
+ throw new XmlSyntaxError("a character XML 1.0 forbids appears in character data", this.line);
966
+ }
967
+ this.text += decodeEntities(raw, this.line);
968
+ this.advanceLine(start, end);
969
+ }
970
+
971
+ /**
972
+ * Count the newlines of a consumed range. Ranges are consumed in order, so the next line break
973
+ * is searched for once per buffer and re-searched only after it was passed; a buffer without
974
+ * a further break is searched once and remembered as such.
975
+ * @param start - the start index
976
+ * @param end - the end index (exclusive)
977
+ */
978
+ private advanceLine(start: number, end: number): void {
979
+ let i = this.nextBreak;
980
+ if (i === -1) {
981
+ return;
982
+ }
983
+ if (i === -2 || i < start) {
984
+ i = this.buffer.indexOf("\n", start);
985
+ }
986
+ while (i >= 0 && i < end) {
987
+ this.line++;
988
+ i = this.buffer.indexOf("\n", i + 1);
989
+ }
990
+ this.nextBreak = i;
991
+ }
992
+
993
+ /** Deliver the accumulated text run, if any. */
994
+ private flushText(): void {
995
+ if (this.text.length === 0) {
996
+ return;
997
+ }
998
+ const { text } = this;
999
+ this.text = "";
1000
+ if (this.stack.length === 0) {
1001
+ if (!isWhitespace(text)) {
1002
+ throw new XmlSyntaxError("text outside the root element", this.textLine);
1003
+ }
1004
+ return;
1005
+ }
1006
+ this.handler.text(text, this.textLine);
1007
+ }
1008
+
1009
+ /**
1010
+ * Open an element.
1011
+ * @param name - the element name
1012
+ * @param attrs - its attributes
1013
+ */
1014
+ private startElement(name: string, attrs: Map<string, string>): void {
1015
+ if (this.stack.length === 0) {
1016
+ if (this.rootClosed) {
1017
+ throw new XmlSyntaxError(`a second root element <${name}> follows the document element`, this.line);
1018
+ }
1019
+ this.rootSeen = true;
1020
+ }
1021
+ this.stack.push(name);
1022
+ this.handler.start(name, attrs, this.line);
1023
+ }
1024
+
1025
+ /**
1026
+ * Close an element, checking that it matches the innermost open one.
1027
+ * @param name - the element name
1028
+ */
1029
+ private endElement(name: string): void {
1030
+ const open = this.stack.length > 0 ? this.stack[this.stack.length - 1] : null;
1031
+ if (open === null) {
1032
+ throw new XmlSyntaxError(`unexpected end tag </${name}>`, this.line);
1033
+ }
1034
+ if (open !== name) {
1035
+ throw new XmlSyntaxError(`end tag </${name}> does not match <${open}>`, this.line);
1036
+ }
1037
+ this.stack.pop();
1038
+ if (this.stack.length === 0) {
1039
+ this.rootClosed = true;
1040
+ }
1041
+ this.handler.end(name, this.line);
1042
+ }
1043
+ }
1044
+
1045
+ /**
1046
+ * Attribute value normalisation (XML 1.0 section 3.3.3): a literal tab or line break becomes a
1047
+ * space; a character reference to one is kept, which is why this runs before entity decoding.
1048
+ * @param raw - the value between the quotes
1049
+ * @returns the normalised value
1050
+ */
1051
+ function normalizeAttributeValue(raw: string): string {
1052
+ return raw.includes("\n") || raw.includes("\t") ? raw.replace(/[\n\t]/g, " ") : raw;
1053
+ }