@endevops/effect-codec-xml 0.0.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/LICENSE-is-entities +21 -0
- package/LICENSE-is-xml-naming +21 -0
- package/README.md +415 -0
- package/dist/codec.d.ts +48 -0
- package/dist/codec.d.ts.map +1 -0
- package/dist/codec.js +63 -0
- package/dist/codec.js.map +1 -0
- package/dist/conventions.d.ts +88 -0
- package/dist/conventions.d.ts.map +1 -0
- package/dist/conventions.js +113 -0
- package/dist/conventions.js.map +1 -0
- package/dist/entities/entity-decoder.d.ts +333 -0
- package/dist/entities/entity-decoder.d.ts.map +1 -0
- package/dist/entities/entity-decoder.js +841 -0
- package/dist/entities/entity-decoder.js.map +1 -0
- package/dist/entities/entity-tables.js +16 -0
- package/dist/entities/entity-tables.js.map +1 -0
- package/dist/errors.d.ts +49 -0
- package/dist/errors.d.ts.map +1 -0
- package/dist/errors.js +48 -0
- package/dist/errors.js.map +1 -0
- package/dist/index.d.ts +11 -0
- package/dist/index.js +11 -0
- package/dist/namespaces.d.ts +101 -0
- package/dist/namespaces.d.ts.map +1 -0
- package/dist/namespaces.js +663 -0
- package/dist/namespaces.js.map +1 -0
- package/dist/naming.d.ts +149 -0
- package/dist/naming.d.ts.map +1 -0
- package/dist/naming.js +296 -0
- package/dist/naming.js.map +1 -0
- package/dist/parse.d.ts +75 -0
- package/dist/parse.d.ts.map +1 -0
- package/dist/parse.js +437 -0
- package/dist/parse.js.map +1 -0
- package/dist/render.d.ts +99 -0
- package/dist/render.d.ts.map +1 -0
- package/dist/render.js +509 -0
- package/dist/render.js.map +1 -0
- package/dist/xml-error.d.ts +172 -0
- package/dist/xml-error.d.ts.map +1 -0
- package/dist/xml-error.js +157 -0
- package/dist/xml-error.js.map +1 -0
- package/dist/xml-value.d.ts +42 -0
- package/dist/xml-value.d.ts.map +1 -0
- package/dist/xml-value.js +79 -0
- package/dist/xml-value.js.map +1 -0
- package/package.json +69 -0
- package/src/codec.ts +136 -0
- package/src/conventions.ts +145 -0
- package/src/entities/entity-decoder.ts +1248 -0
- package/src/entities/entity-tables.ts +18 -0
- package/src/errors.ts +55 -0
- package/src/index.ts +79 -0
- package/src/namespaces.ts +968 -0
- package/src/naming.ts +519 -0
- package/src/parse.ts +597 -0
- package/src/render.ts +708 -0
- package/src/xml-error.ts +168 -0
- package/src/xml-value.ts +108 -0
package/src/naming.ts
ADDED
|
@@ -0,0 +1,519 @@
|
|
|
1
|
+
// xml-naming
|
|
2
|
+
// Validates XML Name productions as defined in the XML 1.0 and 1.1 specifications.
|
|
3
|
+
// Covers: Name, NCName, QName, NMToken, NMTokens
|
|
4
|
+
//
|
|
5
|
+
// XML 1.0 spec: https://www.w3.org/TR/xml/#NT-Name
|
|
6
|
+
// XML 1.1 spec: https://www.w3.org/TR/xml11/#NT-NameStartChar
|
|
7
|
+
// XML NS spec: https://www.w3.org/TR/xml-names/#NT-NCName
|
|
8
|
+
//
|
|
9
|
+
// The five predicates and `sanitize` are plain synchronous functions: a regex
|
|
10
|
+
// test cannot fail and a character substitution has nothing to fail about, so
|
|
11
|
+
// there is no effect to model. `validate` does have one failure to report — an
|
|
12
|
+
// unknown production, unreachable from TypeScript where `Production` is a
|
|
13
|
+
// closed union but reachable for an untyped JavaScript caller, or a value that
|
|
14
|
+
// crossed a boundary as `unknown` — so it answers with an `Effect` whose error
|
|
15
|
+
// channel is that {@link XmlError}. The value it produces is still a plain
|
|
16
|
+
// result.
|
|
17
|
+
|
|
18
|
+
import { Effect } from 'effect';
|
|
19
|
+
|
|
20
|
+
import { XmlError } from '#/xml-error.ts';
|
|
21
|
+
|
|
22
|
+
/**
|
|
23
|
+
* @description The XML specification version a production is validated against. The two differ only in their non-ASCII character ranges — see {@link getRegexes}.
|
|
24
|
+
*/
|
|
25
|
+
export type XmlVersion = '1.0' | '1.1';
|
|
26
|
+
|
|
27
|
+
/**
|
|
28
|
+
* @description One of the five XML name productions this package validates. The name is the production's grammar rule: `name` is the full Name production,
|
|
29
|
+
* `ncName` its non-colonized form, `qName` the prefixed form, and `nmToken`/`nmTokens` the attribute-value productions that drop the first-character
|
|
30
|
+
* restriction.
|
|
31
|
+
*/
|
|
32
|
+
export type Production = 'name' | 'ncName' | 'qName' | 'nmToken' | 'nmTokens';
|
|
33
|
+
|
|
34
|
+
/**
|
|
35
|
+
* @description Options shared by every validator.
|
|
36
|
+
*/
|
|
37
|
+
export interface ValidationOptions {
|
|
38
|
+
/**
|
|
39
|
+
* @description XML specification version to validate against. Defaults to '1.0'.
|
|
40
|
+
*/
|
|
41
|
+
xmlVersion?: XmlVersion;
|
|
42
|
+
/**
|
|
43
|
+
* @description Restrict matching to the ASCII subset of the NameStartChar/NameChar productions and skip unicode-aware regex matching entirely. Faster,
|
|
44
|
+
* especially for XML 1.1 (which otherwise requires the `/u` regex flag), but rejects legitimate non-ASCII XML names. Off by default for backward
|
|
45
|
+
* compatibility — opt in only when inputs are known to be ASCII. Defaults to false.
|
|
46
|
+
*/
|
|
47
|
+
asciiOnly?: boolean;
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
/**
|
|
51
|
+
* @description Options for {@link sanitize}.
|
|
52
|
+
*/
|
|
53
|
+
export interface SanitizeOptions {
|
|
54
|
+
/**
|
|
55
|
+
* @description Character used to replace invalid characters. Defaults to '_'.
|
|
56
|
+
*/
|
|
57
|
+
replacement?: string;
|
|
58
|
+
/**
|
|
59
|
+
* @description Also replace any non-ASCII character, not just XML-illegal ones. Defaults to false.
|
|
60
|
+
*/
|
|
61
|
+
asciiOnly?: boolean;
|
|
62
|
+
/**
|
|
63
|
+
* @description Accepted and ignored. Sanitizing is not version-dependent — the character set it considers illegal is the union of both versions — but the option
|
|
64
|
+
* is part of the published signature, so dropping it would break callers that pass it through a shared options object.
|
|
65
|
+
*/
|
|
66
|
+
xmlVersion?: XmlVersion;
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
/**
|
|
70
|
+
* @description The outcome of {@link validate}, discriminated on `valid` so a caller can narrow to the reason and position without a cast.
|
|
71
|
+
*/
|
|
72
|
+
export type ValidationResult =
|
|
73
|
+
| { valid: true; production: Production; input: string }
|
|
74
|
+
| {
|
|
75
|
+
valid: false;
|
|
76
|
+
production: Production;
|
|
77
|
+
input: string;
|
|
78
|
+
reason: string;
|
|
79
|
+
/**
|
|
80
|
+
* @description Index of the first offending character, or `undefined` when the failure is structural (an empty input, or a colon count) rather than a
|
|
81
|
+
* specific character.
|
|
82
|
+
*/
|
|
83
|
+
position: number | undefined;
|
|
84
|
+
};
|
|
85
|
+
|
|
86
|
+
/**
|
|
87
|
+
* @description The five productions, in the order the runtime error message lists them.
|
|
88
|
+
*/
|
|
89
|
+
const PRODUCTIONS = ['name', 'ncName', 'qName', 'nmToken', 'nmTokens'] as const satisfies ReadonlyArray<Production>;
|
|
90
|
+
|
|
91
|
+
/**
|
|
92
|
+
* @description One compiled regex per production.
|
|
93
|
+
*/
|
|
94
|
+
type ProductionRegexes = Record<Production, RegExp>;
|
|
95
|
+
|
|
96
|
+
// ---------------------------------------------------------------------------
|
|
97
|
+
// Character class strings — XML 1.0
|
|
98
|
+
//
|
|
99
|
+
// NameStartChar ::= ":" | [A-Z] | "_" | [a-z]
|
|
100
|
+
// | [#xC0-#xD6] | [#xD8-#xF6] | [#xF8-#x2FF]
|
|
101
|
+
// | [#x370-#x37D] | [#x37F-#x1FFF] <- split to exclude #x0487
|
|
102
|
+
// | [#x200C-#x200D]
|
|
103
|
+
// | [#x2070-#x218F] | [#x2C00-#x2FEF]
|
|
104
|
+
// | [#x3001-#xD7FF] | [#xF900-#xFDCF] | [#xFDF0-#xFFFD]
|
|
105
|
+
//
|
|
106
|
+
// NameChar ::= NameStartChar | "-" | "." | [0-9]
|
|
107
|
+
// | #xB7 | [#x0300-#x036F] | [#x203F-#x2040]
|
|
108
|
+
//
|
|
109
|
+
// Note: \u0487 (Combining Cyrillic Millions Sign) was added in Unicode 4.0,
|
|
110
|
+
// after XML 1.0 was defined against Unicode 2.0. It falls inside the range
|
|
111
|
+
// \u037F-\u1FFF but must be excluded. We split that range into
|
|
112
|
+
// \u037F-\u0486 and \u0488-\u1FFF to exclude it explicitly.
|
|
113
|
+
// ---------------------------------------------------------------------------
|
|
114
|
+
|
|
115
|
+
const nameStartChar10 =
|
|
116
|
+
':A-Za-z_' +
|
|
117
|
+
'\u00C0-\u00D6\u00D8-\u00F6\u00F8-\u02FF' +
|
|
118
|
+
'\u0370-\u037D' +
|
|
119
|
+
'\u037F-\u0486\u0488-\u1FFF' + // split to exclude \u0487
|
|
120
|
+
'\u200C-\u200D' +
|
|
121
|
+
'\u2070-\u218F' +
|
|
122
|
+
'\u2C00-\u2FEF' +
|
|
123
|
+
'\u3001-\uD7FF' +
|
|
124
|
+
'\uF900-\uFDCF' +
|
|
125
|
+
'\uFDF0-\uFFFD';
|
|
126
|
+
|
|
127
|
+
const nameChar10 = nameStartChar10 + '\\-\\.\\d' + '\u00B7' + '\u0300-\u036F' + '\u203F-\u2040';
|
|
128
|
+
|
|
129
|
+
// ---------------------------------------------------------------------------
|
|
130
|
+
// Character class strings — XML 1.1
|
|
131
|
+
//
|
|
132
|
+
// Differences from XML 1.0:
|
|
133
|
+
//
|
|
134
|
+
// NameStartChar:
|
|
135
|
+
// 1.0 has split ranges: \u00C0-\u00D6, \u00D8-\u00F6, \u00F8-\u02FF
|
|
136
|
+
// 1.1 merges them into: \u00C0-\u02FF
|
|
137
|
+
// (\u00D7 x and \u00F7 / are division symbols, excluded in both versions)
|
|
138
|
+
//
|
|
139
|
+
// 1.0 tops out at \uFFFD (BMP only)
|
|
140
|
+
// 1.1 adds \u{10000}-\u{EFFFF} (supplementary planes)
|
|
141
|
+
// These require the /u flag on the RegExp — see buildRegexes below.
|
|
142
|
+
//
|
|
143
|
+
// NameChar:
|
|
144
|
+
// 1.1 adds \u0487 (Combining Cyrillic Millions Sign, added in Unicode 4.0)
|
|
145
|
+
// ---------------------------------------------------------------------------
|
|
146
|
+
|
|
147
|
+
const nameStartChar11 =
|
|
148
|
+
':A-Za-z_' +
|
|
149
|
+
'\u00C0-\u02FF' + // merged — 1.0 had three split ranges here
|
|
150
|
+
'\u0370-\u037D' +
|
|
151
|
+
'\u037F-\u0486\u0488-\u1FFF' + // split to exclude \u0487 (combining mark, never a NameStartChar)
|
|
152
|
+
'\u200C-\u200D' +
|
|
153
|
+
'\u2070-\u218F' +
|
|
154
|
+
'\u2C00-\u2FEF' +
|
|
155
|
+
'\u3001-\uD7FF' +
|
|
156
|
+
'\uF900-\uFDCF' +
|
|
157
|
+
'\uFDF0-\uFFFD' +
|
|
158
|
+
'\u{10000}-\u{EFFFF}'; // supplementary planes — REQUIRES /u flag on RegExp
|
|
159
|
+
|
|
160
|
+
const nameChar11 =
|
|
161
|
+
nameStartChar11 +
|
|
162
|
+
'\\-\\.\\d' +
|
|
163
|
+
'\u00B7' +
|
|
164
|
+
'\u0300-\u036F' +
|
|
165
|
+
'\u0487' + // Combining Cyrillic Millions Sign — valid in 1.1, not 1.0
|
|
166
|
+
'\u203F-\u2040';
|
|
167
|
+
|
|
168
|
+
// ---------------------------------------------------------------------------
|
|
169
|
+
// Regex builders
|
|
170
|
+
//
|
|
171
|
+
// XML 1.0 regexes: no flags — BMP only, standard JS regex behaviour.
|
|
172
|
+
// XML 1.1 regexes: /u flag — required for \u{10000}-\u{EFFFF} to match actual
|
|
173
|
+
// supplementary code points rather than lone surrogates (which are illegal XML).
|
|
174
|
+
// ---------------------------------------------------------------------------
|
|
175
|
+
|
|
176
|
+
/**
|
|
177
|
+
* @description Compiles the five production regexes from a NameStartChar/NameChar pair.
|
|
178
|
+
*
|
|
179
|
+
* @param startChar - Character class body for the first character.
|
|
180
|
+
* @param char - Character class body for every subsequent character.
|
|
181
|
+
* @param flags - RegExp flags. `'u'` for the XML 1.1 set, so its supplementary-plane range matches code points rather than lone surrogates.
|
|
182
|
+
*
|
|
183
|
+
* @returns One regex per {@link Production}.
|
|
184
|
+
*/
|
|
185
|
+
const buildRegexes = (startChar: string, char: string, flags = ''): ProductionRegexes => {
|
|
186
|
+
const ncStart = startChar.replace(':', '');
|
|
187
|
+
const ncChar = char.replace(':', '');
|
|
188
|
+
const ncNamePat = `[${ncStart}][${ncChar}]*`;
|
|
189
|
+
|
|
190
|
+
return {
|
|
191
|
+
name: new RegExp(`^[${startChar}][${char}]*$`, flags),
|
|
192
|
+
ncName: new RegExp(`^${ncNamePat}$`, flags),
|
|
193
|
+
qName: new RegExp(`^${ncNamePat}(?::${ncNamePat})?$`, flags),
|
|
194
|
+
nmToken: new RegExp(`^[${char}]+$`, flags),
|
|
195
|
+
nmTokens: new RegExp(`^[${char}]+(?:\\s+[${char}]+)*$`, flags),
|
|
196
|
+
};
|
|
197
|
+
};
|
|
198
|
+
|
|
199
|
+
const regexes10 = buildRegexes(nameStartChar10, nameChar10); // no /u — BMP only
|
|
200
|
+
const regexes11 = buildRegexes(nameStartChar11, nameChar11, 'u'); // /u — enables \u{10000}-\u{EFFFF}
|
|
201
|
+
|
|
202
|
+
// ---------------------------------------------------------------------------
|
|
203
|
+
// ASCII-only fast path (opt-in, off by default)
|
|
204
|
+
//
|
|
205
|
+
// The XML 1.0 vs 1.1 NameStartChar/NameChar productions differ *only* in
|
|
206
|
+
// their non-ASCII ranges (merged vs split Latin-1 ranges, \u0487, and
|
|
207
|
+
// supplementary planes). Restricted to ASCII, both versions collapse to the
|
|
208
|
+
// same character classes, so a single regex pair covers both xmlVersion
|
|
209
|
+
// values — no /u flag needed.
|
|
210
|
+
//
|
|
211
|
+
// Rationale: unicode-aware regexes (the /u flag, required for XML 1.1's
|
|
212
|
+
// supplementary-plane range) are measurably slower in V8 than plain
|
|
213
|
+
// non-unicode regexes on the same input, even when the input is pure ASCII.
|
|
214
|
+
// For the common case — HTML/SVG ids, XML tags — names are ASCII, so callers
|
|
215
|
+
// who know this can opt in to skip the unicode-aware matching path entirely.
|
|
216
|
+
// This is a real but *conditional* win: mainly for XML 1.1 input (avoids /u),
|
|
217
|
+
// or at scale where the larger unicode character classes add engine
|
|
218
|
+
// overhead. It also changes behaviour (rejects legitimate non-ASCII XML
|
|
219
|
+
// 1.0/1.1 names), so it must never be silently enabled — hence off by
|
|
220
|
+
// default.
|
|
221
|
+
// ---------------------------------------------------------------------------
|
|
222
|
+
|
|
223
|
+
const nameStartCharAscii = ':A-Za-z_';
|
|
224
|
+
const nameCharAscii = nameStartCharAscii + '\\-\\.\\d';
|
|
225
|
+
|
|
226
|
+
const regexesAscii = buildRegexes(nameStartCharAscii, nameCharAscii); // no /u — ASCII only
|
|
227
|
+
|
|
228
|
+
/**
|
|
229
|
+
* @description The compiled regex set for a version/ASCII combination. Only three sets are ever built, at module load; this is a lookup, not a compile.
|
|
230
|
+
*
|
|
231
|
+
* @param xmlVersion - Which XML version's character classes to use.
|
|
232
|
+
* @param asciiOnly - Return the ASCII-only set regardless of `xmlVersion`.
|
|
233
|
+
*
|
|
234
|
+
* @returns The regex set to validate against.
|
|
235
|
+
*/
|
|
236
|
+
const getRegexes = (xmlVersion: XmlVersion = '1.0', asciiOnly = false): ProductionRegexes => {
|
|
237
|
+
if (asciiOnly) return regexesAscii;
|
|
238
|
+
return xmlVersion === '1.1' ? regexes11 : regexes10;
|
|
239
|
+
};
|
|
240
|
+
|
|
241
|
+
// ---------------------------------------------------------------------------
|
|
242
|
+
// Boolean validators
|
|
243
|
+
//
|
|
244
|
+
// One plain predicate per production. A regex test cannot fail, and every one of these is called per name
|
|
245
|
+
// inside a parser's or codec's hot loop, so a boolean is the honest answer and there is no effect to
|
|
246
|
+
// allocate or run.
|
|
247
|
+
// ---------------------------------------------------------------------------
|
|
248
|
+
|
|
249
|
+
/**
|
|
250
|
+
* @description Whether the string is a valid XML Name. Colons are allowed anywhere (Name production). Used for: DOCTYPE entity names, notation names, DTD element
|
|
251
|
+
* declarations.
|
|
252
|
+
*
|
|
253
|
+
* @param str - The candidate name.
|
|
254
|
+
* @param opts - `asciiOnly` skips unicode-aware matching, ASCII names only (default false).
|
|
255
|
+
*
|
|
256
|
+
* @returns Whether `str` satisfies the production.
|
|
257
|
+
*/
|
|
258
|
+
export const isName = (str: string, { xmlVersion = '1.0', asciiOnly = false }: ValidationOptions = {}): boolean =>
|
|
259
|
+
getRegexes(xmlVersion, asciiOnly).name.test(str);
|
|
260
|
+
|
|
261
|
+
/**
|
|
262
|
+
* @description Whether the string is a valid NCName (Non-Colonized Name).\
|
|
263
|
+
* Colons are not permitted.\
|
|
264
|
+
* Used for: namespace prefixes, local names, SVG id attributes.
|
|
265
|
+
*
|
|
266
|
+
* @param str - The candidate name.
|
|
267
|
+
* @param opts - `asciiOnly` skips unicode-aware matching, ASCII names only (default false).
|
|
268
|
+
*
|
|
269
|
+
* @returns Whether `str` satisfies the production.
|
|
270
|
+
*/
|
|
271
|
+
export const isNcName = (str: string, { xmlVersion = '1.0', asciiOnly = false }: ValidationOptions = {}): boolean =>
|
|
272
|
+
getRegexes(xmlVersion, asciiOnly).ncName.test(str);
|
|
273
|
+
|
|
274
|
+
/**
|
|
275
|
+
* @description Whether the string is a valid QName (Qualified Name).\
|
|
276
|
+
* Allows exactly one colon as a prefix separator: `prefix:localName`.\
|
|
277
|
+
* Used for: element and attribute names in namespace-aware XML/SVG.
|
|
278
|
+
*
|
|
279
|
+
* @param str - The candidate name.
|
|
280
|
+
* @param opts - `asciiOnly` skips unicode-aware matching, ASCII names only (default false).
|
|
281
|
+
*
|
|
282
|
+
* @returns Whether `str` satisfies the production.
|
|
283
|
+
*/
|
|
284
|
+
export const isQName = (str: string, { xmlVersion = '1.0', asciiOnly = false }: ValidationOptions = {}): boolean =>
|
|
285
|
+
getRegexes(xmlVersion, asciiOnly).qName.test(str);
|
|
286
|
+
|
|
287
|
+
/**
|
|
288
|
+
* @description Whether the string is a valid NMToken. Like Name but no restriction on the first character. Used for: DTD NMTOKEN attribute values.
|
|
289
|
+
*
|
|
290
|
+
* @param str - The candidate token.
|
|
291
|
+
* @param opts - `asciiOnly` skips unicode-aware matching, ASCII names only (default false).
|
|
292
|
+
*
|
|
293
|
+
* @returns Whether `str` satisfies the production.
|
|
294
|
+
*/
|
|
295
|
+
export const isNmToken = (str: string, { xmlVersion = '1.0', asciiOnly = false }: ValidationOptions = {}): boolean =>
|
|
296
|
+
getRegexes(xmlVersion, asciiOnly).nmToken.test(str);
|
|
297
|
+
|
|
298
|
+
/**
|
|
299
|
+
* @description Whether the string is a valid NMTokens value — a whitespace-separated list of NMToken values. Used for: DTD NMTOKENS attribute values.
|
|
300
|
+
*
|
|
301
|
+
* @param str - The candidate list.
|
|
302
|
+
* @param opts - `asciiOnly` skips unicode-aware matching, ASCII names only (default false).
|
|
303
|
+
*
|
|
304
|
+
* @returns Whether `str` satisfies the production.
|
|
305
|
+
*/
|
|
306
|
+
export const isNmTokens = (str: string, { xmlVersion = '1.0', asciiOnly = false }: ValidationOptions = {}): boolean =>
|
|
307
|
+
getRegexes(xmlVersion, asciiOnly).nmTokens.test(str);
|
|
308
|
+
|
|
309
|
+
/**
|
|
310
|
+
* @description The failure to report when a production is not one of the five this module knows, so `validate` reports it the same way. The single place the
|
|
311
|
+
* unknown-production guard lives. It is unreachable from TypeScript, where `Production` is a closed union; this is the guard for untyped JavaScript
|
|
312
|
+
* callers.
|
|
313
|
+
*
|
|
314
|
+
* @param production - The production to check.
|
|
315
|
+
*
|
|
316
|
+
* @returns The `InvalidProduction` failure, or `null` when the production is known.
|
|
317
|
+
*/
|
|
318
|
+
const productionError = (production: Production): XmlError | null => {
|
|
319
|
+
if (PRODUCTIONS.includes(production)) return null;
|
|
320
|
+
return new XmlError({
|
|
321
|
+
reason: { _tag: 'InvalidProduction', production, expected: PRODUCTIONS.join(', ') },
|
|
322
|
+
message: `Unknown production "${production}". Must be one of: ${PRODUCTIONS.join(', ')}`,
|
|
323
|
+
});
|
|
324
|
+
};
|
|
325
|
+
|
|
326
|
+
const validators: Record<Production, (str: string, opts?: ValidationOptions) => boolean> = {
|
|
327
|
+
name: isName,
|
|
328
|
+
ncName: isNcName,
|
|
329
|
+
qName: isQName,
|
|
330
|
+
nmToken: isNmToken,
|
|
331
|
+
nmTokens: isNmTokens,
|
|
332
|
+
};
|
|
333
|
+
|
|
334
|
+
/**
|
|
335
|
+
* @description The diagnostic body {@link validate} reports, with the production already known to be valid. Kept separate so the reason-finding logic carries no
|
|
336
|
+
* unknown-production check and the batch path can map over it without re-entering the guard per element.
|
|
337
|
+
*
|
|
338
|
+
* @param str - The candidate name.
|
|
339
|
+
* @param production - The production to validate against, already checked.
|
|
340
|
+
* @param xmlVersion - Which version's character classes to use.
|
|
341
|
+
* @param asciiOnly - Whether the ASCII-only fast path applied.
|
|
342
|
+
*
|
|
343
|
+
* @returns The discriminated result.
|
|
344
|
+
*/
|
|
345
|
+
const diagnose = (str: string, production: Production, xmlVersion: XmlVersion, asciiOnly: boolean): ValidationResult =>
|
|
346
|
+
diagnoseWith(str, production, validators[production](str, { xmlVersion, asciiOnly }), asciiOnly);
|
|
347
|
+
|
|
348
|
+
/**
|
|
349
|
+
* @description The colon-specific reason a `qName` or `ncName` failed, or `undefined` when the failure is not about a colon. The three QName forms are checked in
|
|
350
|
+
* the order they can co-occur: a string that both starts and ends with a colon is reported as the leading one, and a two-colon string is reported as
|
|
351
|
+
* the count.
|
|
352
|
+
*
|
|
353
|
+
* @param str - The candidate name.
|
|
354
|
+
* @param production - The production checked.
|
|
355
|
+
*
|
|
356
|
+
* @returns The reason and position, or `undefined`.
|
|
357
|
+
*/
|
|
358
|
+
const diagnoseColon = (str: string, production: Production): { reason: string; position: number } | undefined => {
|
|
359
|
+
if (production === 'ncName' && str.includes(':')) {
|
|
360
|
+
return { reason: 'Colon is not allowed in NCName', position: str.indexOf(':') };
|
|
361
|
+
}
|
|
362
|
+
|
|
363
|
+
if (production !== 'qName') {
|
|
364
|
+
return undefined;
|
|
365
|
+
}
|
|
366
|
+
if (str.startsWith(':')) {
|
|
367
|
+
return { reason: 'QName cannot start with a colon', position: 0 };
|
|
368
|
+
}
|
|
369
|
+
if (str.endsWith(':')) {
|
|
370
|
+
return { reason: 'QName cannot end with a colon', position: str.length - 1 };
|
|
371
|
+
}
|
|
372
|
+
if ((str.match(/:/g) ?? []).length > 1) {
|
|
373
|
+
return { reason: 'QName can have at most one colon', position: str.lastIndexOf(':') };
|
|
374
|
+
}
|
|
375
|
+
return undefined;
|
|
376
|
+
};
|
|
377
|
+
|
|
378
|
+
/**
|
|
379
|
+
* @description The first character that is not a legal `NameChar`, and where it sits.
|
|
380
|
+
*
|
|
381
|
+
* @param str - The candidate name.
|
|
382
|
+
* @param namePattern - The `NameChar` test for the character set the validator used.
|
|
383
|
+
*
|
|
384
|
+
* @returns The reason and position, or `undefined` when every character is legal.
|
|
385
|
+
*/
|
|
386
|
+
const diagnoseNameChar = (str: string, namePattern: RegExp): { reason: string; position: number } | undefined => {
|
|
387
|
+
for (let i = 0; i < str.length; i++) {
|
|
388
|
+
const char = str[i];
|
|
389
|
+
if (char !== undefined && !namePattern.test(char)) {
|
|
390
|
+
return { reason: `Character "${char}" at position ${i} is not a valid NameChar`, position: i };
|
|
391
|
+
}
|
|
392
|
+
}
|
|
393
|
+
return undefined;
|
|
394
|
+
};
|
|
395
|
+
|
|
396
|
+
/**
|
|
397
|
+
* @description Why a name failed, once whether it failed is already known. The first character outranks the rest: a name that cannot start is reported as a bad
|
|
398
|
+
* `NameStartChar` even when a later character is also illegal.
|
|
399
|
+
*
|
|
400
|
+
* @param str - The candidate name.
|
|
401
|
+
* @param production - The production checked.
|
|
402
|
+
* @param startCharPattern - The `NameStartChar` test for the character set the validator used.
|
|
403
|
+
* @param namePattern - The `NameChar` test for the same character set.
|
|
404
|
+
*
|
|
405
|
+
* @returns The reason, and the offending position, which is `undefined` for a structural failure.
|
|
406
|
+
*/
|
|
407
|
+
const findFailure = (
|
|
408
|
+
str: string,
|
|
409
|
+
production: Production,
|
|
410
|
+
startCharPattern: RegExp,
|
|
411
|
+
namePattern: RegExp
|
|
412
|
+
): { reason: string; position: number | undefined } => {
|
|
413
|
+
if (str.length === 0) {
|
|
414
|
+
return { reason: 'Input is empty', position: undefined };
|
|
415
|
+
}
|
|
416
|
+
|
|
417
|
+
const colon = diagnoseColon(str, production);
|
|
418
|
+
if (colon !== undefined) return colon;
|
|
419
|
+
|
|
420
|
+
const firstChar = str[0];
|
|
421
|
+
if (['name', 'ncName', 'qName'].includes(production) && !startCharPattern.test(firstChar ?? '')) {
|
|
422
|
+
return { reason: `First character "${firstChar}" is not a valid NameStartChar`, position: 0 };
|
|
423
|
+
}
|
|
424
|
+
|
|
425
|
+
return diagnoseNameChar(str, namePattern) ?? { reason: 'Does not match the production rules', position: undefined };
|
|
426
|
+
};
|
|
427
|
+
|
|
428
|
+
/**
|
|
429
|
+
* @description Why a name failed, once whether it failed is already known. Split from {@link diagnose} so the effect that asks the question stays a one-liner and
|
|
430
|
+
* the reason-finding is plain.
|
|
431
|
+
*
|
|
432
|
+
* @param str - The candidate name.
|
|
433
|
+
* @param production - The production checked.
|
|
434
|
+
* @param isValid - Whether it passed.
|
|
435
|
+
* @param asciiOnly - Whether the ASCII-only fast path applied.
|
|
436
|
+
*
|
|
437
|
+
* @returns The discriminated result.
|
|
438
|
+
*/
|
|
439
|
+
const diagnoseWith = (str: string, production: Production, isValid: boolean, asciiOnly: boolean): ValidationResult => {
|
|
440
|
+
if (isValid) return { valid: true, production, input: str };
|
|
441
|
+
|
|
442
|
+
// Diagnostic fallback char checks must mirror the same character set the
|
|
443
|
+
// boolean validator above used, or the reported reason/position could
|
|
444
|
+
// contradict the `valid: false` result (e.g. flagging a char as illegal
|
|
445
|
+
// that the unicode-aware check would have accepted).
|
|
446
|
+
const startCharPattern = asciiOnly ? /^[:A-Za-z_]/ : /^[:A-Za-z_\u00C0-\uFFFD]/;
|
|
447
|
+
const namePattern = asciiOnly ? /[\w\-\\.:]/ : /[\w\-\\.:\u00B7\u00C0-\uFFFD]/;
|
|
448
|
+
|
|
449
|
+
return { valid: false, production, input: str, ...findFailure(str, production, startCharPattern, namePattern) };
|
|
450
|
+
};
|
|
451
|
+
|
|
452
|
+
/**
|
|
453
|
+
* @description Validates a string against a named production and, on failure, reports why and where.
|
|
454
|
+
*
|
|
455
|
+
* @example
|
|
456
|
+
* ```typescript
|
|
457
|
+
* import { Effect } from 'effect';
|
|
458
|
+
* import { validate } from '@endevops/effect-xml-codec';
|
|
459
|
+
*
|
|
460
|
+
* Effect.runSync(validate('not a name', 'ncName'));
|
|
461
|
+
* // { valid: false, production: 'ncName', input: 'not a name', reason: 'First character " " is not a valid NameStartChar', position: 0 }
|
|
462
|
+
* ```;
|
|
463
|
+
*
|
|
464
|
+
* @param str - The candidate name.
|
|
465
|
+
* @param production - The production to validate against.
|
|
466
|
+
* @param opts - Version and ASCII-only selection, as for the boolean validators.
|
|
467
|
+
*
|
|
468
|
+
* @returns An effect producing a discriminated result: the plain triple when valid, or the offending `reason` and `position` when not. A name that
|
|
469
|
+
* fails to validate is a `valid: false` result, not a failure — an invalid name is the question being answered. The effect fails only with an
|
|
470
|
+
* {@link XmlError} and the `InvalidProduction` reason for an unknown production, which is unreachable from TypeScript and is the guard for untyped
|
|
471
|
+
* JavaScript callers.
|
|
472
|
+
*/
|
|
473
|
+
export const validate = Effect.fnUntraced(function* (
|
|
474
|
+
str: string,
|
|
475
|
+
production: Production,
|
|
476
|
+
{ xmlVersion = '1.0', asciiOnly = false }: ValidationOptions = {}
|
|
477
|
+
): Effect.fn.Return<ValidationResult, XmlError> {
|
|
478
|
+
const invalid = productionError(production);
|
|
479
|
+
if (invalid) return yield* invalid;
|
|
480
|
+
return diagnose(str, production, xmlVersion, asciiOnly);
|
|
481
|
+
});
|
|
482
|
+
|
|
483
|
+
// ---------------------------------------------------------------------------
|
|
484
|
+
// Sanitizer
|
|
485
|
+
// ---------------------------------------------------------------------------
|
|
486
|
+
|
|
487
|
+
/**
|
|
488
|
+
* @description Transforms an invalid string into the nearest valid XML name for the given production: strips or replaces illegal characters, fixes an invalid
|
|
489
|
+
* start character by prepending the replacement, and removes colons for NCName.
|
|
490
|
+
*
|
|
491
|
+
* @param str - The candidate name.
|
|
492
|
+
* @param production - The production to sanitize for. Defaults to `'name'`.
|
|
493
|
+
* @param opts - `replacement` is the substitute character (default `'_'`); `asciiOnly` also replaces non-ASCII characters.
|
|
494
|
+
*
|
|
495
|
+
* @returns A string that satisfies `production` for the ASCII range, or the nearest approximation of it.
|
|
496
|
+
*/
|
|
497
|
+
export const sanitize = (str: string, production: Production = 'name', { replacement = '_', asciiOnly = false }: SanitizeOptions = {}): string => {
|
|
498
|
+
if (!str) return replacement;
|
|
499
|
+
|
|
500
|
+
let result = str;
|
|
501
|
+
|
|
502
|
+
// Strip colons for NCName
|
|
503
|
+
if (production === 'ncName') {
|
|
504
|
+
result = result.replace(/:/g, '');
|
|
505
|
+
}
|
|
506
|
+
|
|
507
|
+
// Replace illegal characters
|
|
508
|
+
const allowedCharPattern = asciiOnly ? /[^\w\-.:]/g : /[^\w\-.:\u00B7\u00C0-\uFFFD]/g;
|
|
509
|
+
result = result.replace(allowedCharPattern, replacement);
|
|
510
|
+
|
|
511
|
+
// Fix invalid start character for Name / NCName / QName
|
|
512
|
+
if (production !== 'nmToken' && production !== 'nmTokens') {
|
|
513
|
+
if (/^[-.\d]/.test(result)) {
|
|
514
|
+
result = replacement + result;
|
|
515
|
+
}
|
|
516
|
+
}
|
|
517
|
+
|
|
518
|
+
return result || replacement;
|
|
519
|
+
};
|