@endevops/effect-codec-xml 0.0.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (61) hide show
  1. package/LICENSE +21 -0
  2. package/LICENSE-is-entities +21 -0
  3. package/LICENSE-is-xml-naming +21 -0
  4. package/README.md +415 -0
  5. package/dist/codec.d.ts +48 -0
  6. package/dist/codec.d.ts.map +1 -0
  7. package/dist/codec.js +63 -0
  8. package/dist/codec.js.map +1 -0
  9. package/dist/conventions.d.ts +88 -0
  10. package/dist/conventions.d.ts.map +1 -0
  11. package/dist/conventions.js +113 -0
  12. package/dist/conventions.js.map +1 -0
  13. package/dist/entities/entity-decoder.d.ts +333 -0
  14. package/dist/entities/entity-decoder.d.ts.map +1 -0
  15. package/dist/entities/entity-decoder.js +841 -0
  16. package/dist/entities/entity-decoder.js.map +1 -0
  17. package/dist/entities/entity-tables.js +16 -0
  18. package/dist/entities/entity-tables.js.map +1 -0
  19. package/dist/errors.d.ts +49 -0
  20. package/dist/errors.d.ts.map +1 -0
  21. package/dist/errors.js +48 -0
  22. package/dist/errors.js.map +1 -0
  23. package/dist/index.d.ts +11 -0
  24. package/dist/index.js +11 -0
  25. package/dist/namespaces.d.ts +101 -0
  26. package/dist/namespaces.d.ts.map +1 -0
  27. package/dist/namespaces.js +663 -0
  28. package/dist/namespaces.js.map +1 -0
  29. package/dist/naming.d.ts +149 -0
  30. package/dist/naming.d.ts.map +1 -0
  31. package/dist/naming.js +296 -0
  32. package/dist/naming.js.map +1 -0
  33. package/dist/parse.d.ts +75 -0
  34. package/dist/parse.d.ts.map +1 -0
  35. package/dist/parse.js +437 -0
  36. package/dist/parse.js.map +1 -0
  37. package/dist/render.d.ts +99 -0
  38. package/dist/render.d.ts.map +1 -0
  39. package/dist/render.js +509 -0
  40. package/dist/render.js.map +1 -0
  41. package/dist/xml-error.d.ts +172 -0
  42. package/dist/xml-error.d.ts.map +1 -0
  43. package/dist/xml-error.js +157 -0
  44. package/dist/xml-error.js.map +1 -0
  45. package/dist/xml-value.d.ts +42 -0
  46. package/dist/xml-value.d.ts.map +1 -0
  47. package/dist/xml-value.js +79 -0
  48. package/dist/xml-value.js.map +1 -0
  49. package/package.json +69 -0
  50. package/src/codec.ts +136 -0
  51. package/src/conventions.ts +145 -0
  52. package/src/entities/entity-decoder.ts +1248 -0
  53. package/src/entities/entity-tables.ts +18 -0
  54. package/src/errors.ts +55 -0
  55. package/src/index.ts +79 -0
  56. package/src/namespaces.ts +968 -0
  57. package/src/naming.ts +519 -0
  58. package/src/parse.ts +597 -0
  59. package/src/render.ts +708 -0
  60. package/src/xml-error.ts +168 -0
  61. package/src/xml-value.ts +108 -0
package/src/naming.ts ADDED
@@ -0,0 +1,519 @@
1
+ // xml-naming
2
+ // Validates XML Name productions as defined in the XML 1.0 and 1.1 specifications.
3
+ // Covers: Name, NCName, QName, NMToken, NMTokens
4
+ //
5
+ // XML 1.0 spec: https://www.w3.org/TR/xml/#NT-Name
6
+ // XML 1.1 spec: https://www.w3.org/TR/xml11/#NT-NameStartChar
7
+ // XML NS spec: https://www.w3.org/TR/xml-names/#NT-NCName
8
+ //
9
+ // The five predicates and `sanitize` are plain synchronous functions: a regex
10
+ // test cannot fail and a character substitution has nothing to fail about, so
11
+ // there is no effect to model. `validate` does have one failure to report — an
12
+ // unknown production, unreachable from TypeScript where `Production` is a
13
+ // closed union but reachable for an untyped JavaScript caller, or a value that
14
+ // crossed a boundary as `unknown` — so it answers with an `Effect` whose error
15
+ // channel is that {@link XmlError}. The value it produces is still a plain
16
+ // result.
17
+
18
+ import { Effect } from 'effect';
19
+
20
+ import { XmlError } from '#/xml-error.ts';
21
+
22
+ /**
23
+ * @description The XML specification version a production is validated against. The two differ only in their non-ASCII character ranges — see {@link getRegexes}.
24
+ */
25
+ export type XmlVersion = '1.0' | '1.1';
26
+
27
+ /**
28
+ * @description One of the five XML name productions this package validates. The name is the production's grammar rule: `name` is the full Name production,
29
+ * `ncName` its non-colonized form, `qName` the prefixed form, and `nmToken`/`nmTokens` the attribute-value productions that drop the first-character
30
+ * restriction.
31
+ */
32
+ export type Production = 'name' | 'ncName' | 'qName' | 'nmToken' | 'nmTokens';
33
+
34
+ /**
35
+ * @description Options shared by every validator.
36
+ */
37
+ export interface ValidationOptions {
38
+ /**
39
+ * @description XML specification version to validate against. Defaults to '1.0'.
40
+ */
41
+ xmlVersion?: XmlVersion;
42
+ /**
43
+ * @description Restrict matching to the ASCII subset of the NameStartChar/NameChar productions and skip unicode-aware regex matching entirely. Faster,
44
+ * especially for XML 1.1 (which otherwise requires the `/u` regex flag), but rejects legitimate non-ASCII XML names. Off by default for backward
45
+ * compatibility — opt in only when inputs are known to be ASCII. Defaults to false.
46
+ */
47
+ asciiOnly?: boolean;
48
+ }
49
+
50
+ /**
51
+ * @description Options for {@link sanitize}.
52
+ */
53
+ export interface SanitizeOptions {
54
+ /**
55
+ * @description Character used to replace invalid characters. Defaults to '_'.
56
+ */
57
+ replacement?: string;
58
+ /**
59
+ * @description Also replace any non-ASCII character, not just XML-illegal ones. Defaults to false.
60
+ */
61
+ asciiOnly?: boolean;
62
+ /**
63
+ * @description Accepted and ignored. Sanitizing is not version-dependent — the character set it considers illegal is the union of both versions — but the option
64
+ * is part of the published signature, so dropping it would break callers that pass it through a shared options object.
65
+ */
66
+ xmlVersion?: XmlVersion;
67
+ }
68
+
69
+ /**
70
+ * @description The outcome of {@link validate}, discriminated on `valid` so a caller can narrow to the reason and position without a cast.
71
+ */
72
+ export type ValidationResult =
73
+ | { valid: true; production: Production; input: string }
74
+ | {
75
+ valid: false;
76
+ production: Production;
77
+ input: string;
78
+ reason: string;
79
+ /**
80
+ * @description Index of the first offending character, or `undefined` when the failure is structural (an empty input, or a colon count) rather than a
81
+ * specific character.
82
+ */
83
+ position: number | undefined;
84
+ };
85
+
86
+ /**
87
+ * @description The five productions, in the order the runtime error message lists them.
88
+ */
89
+ const PRODUCTIONS = ['name', 'ncName', 'qName', 'nmToken', 'nmTokens'] as const satisfies ReadonlyArray<Production>;
90
+
91
+ /**
92
+ * @description One compiled regex per production.
93
+ */
94
+ type ProductionRegexes = Record<Production, RegExp>;
95
+
96
+ // ---------------------------------------------------------------------------
97
+ // Character class strings — XML 1.0
98
+ //
99
+ // NameStartChar ::= ":" | [A-Z] | "_" | [a-z]
100
+ // | [#xC0-#xD6] | [#xD8-#xF6] | [#xF8-#x2FF]
101
+ // | [#x370-#x37D] | [#x37F-#x1FFF] <- split to exclude #x0487
102
+ // | [#x200C-#x200D]
103
+ // | [#x2070-#x218F] | [#x2C00-#x2FEF]
104
+ // | [#x3001-#xD7FF] | [#xF900-#xFDCF] | [#xFDF0-#xFFFD]
105
+ //
106
+ // NameChar ::= NameStartChar | "-" | "." | [0-9]
107
+ // | #xB7 | [#x0300-#x036F] | [#x203F-#x2040]
108
+ //
109
+ // Note: \u0487 (Combining Cyrillic Millions Sign) was added in Unicode 4.0,
110
+ // after XML 1.0 was defined against Unicode 2.0. It falls inside the range
111
+ // \u037F-\u1FFF but must be excluded. We split that range into
112
+ // \u037F-\u0486 and \u0488-\u1FFF to exclude it explicitly.
113
+ // ---------------------------------------------------------------------------
114
+
115
+ const nameStartChar10 =
116
+ ':A-Za-z_' +
117
+ '\u00C0-\u00D6\u00D8-\u00F6\u00F8-\u02FF' +
118
+ '\u0370-\u037D' +
119
+ '\u037F-\u0486\u0488-\u1FFF' + // split to exclude \u0487
120
+ '\u200C-\u200D' +
121
+ '\u2070-\u218F' +
122
+ '\u2C00-\u2FEF' +
123
+ '\u3001-\uD7FF' +
124
+ '\uF900-\uFDCF' +
125
+ '\uFDF0-\uFFFD';
126
+
127
+ const nameChar10 = nameStartChar10 + '\\-\\.\\d' + '\u00B7' + '\u0300-\u036F' + '\u203F-\u2040';
128
+
129
+ // ---------------------------------------------------------------------------
130
+ // Character class strings — XML 1.1
131
+ //
132
+ // Differences from XML 1.0:
133
+ //
134
+ // NameStartChar:
135
+ // 1.0 has split ranges: \u00C0-\u00D6, \u00D8-\u00F6, \u00F8-\u02FF
136
+ // 1.1 merges them into: \u00C0-\u02FF
137
+ // (\u00D7 x and \u00F7 / are division symbols, excluded in both versions)
138
+ //
139
+ // 1.0 tops out at \uFFFD (BMP only)
140
+ // 1.1 adds \u{10000}-\u{EFFFF} (supplementary planes)
141
+ // These require the /u flag on the RegExp — see buildRegexes below.
142
+ //
143
+ // NameChar:
144
+ // 1.1 adds \u0487 (Combining Cyrillic Millions Sign, added in Unicode 4.0)
145
+ // ---------------------------------------------------------------------------
146
+
147
+ const nameStartChar11 =
148
+ ':A-Za-z_' +
149
+ '\u00C0-\u02FF' + // merged — 1.0 had three split ranges here
150
+ '\u0370-\u037D' +
151
+ '\u037F-\u0486\u0488-\u1FFF' + // split to exclude \u0487 (combining mark, never a NameStartChar)
152
+ '\u200C-\u200D' +
153
+ '\u2070-\u218F' +
154
+ '\u2C00-\u2FEF' +
155
+ '\u3001-\uD7FF' +
156
+ '\uF900-\uFDCF' +
157
+ '\uFDF0-\uFFFD' +
158
+ '\u{10000}-\u{EFFFF}'; // supplementary planes — REQUIRES /u flag on RegExp
159
+
160
+ const nameChar11 =
161
+ nameStartChar11 +
162
+ '\\-\\.\\d' +
163
+ '\u00B7' +
164
+ '\u0300-\u036F' +
165
+ '\u0487' + // Combining Cyrillic Millions Sign — valid in 1.1, not 1.0
166
+ '\u203F-\u2040';
167
+
168
+ // ---------------------------------------------------------------------------
169
+ // Regex builders
170
+ //
171
+ // XML 1.0 regexes: no flags — BMP only, standard JS regex behaviour.
172
+ // XML 1.1 regexes: /u flag — required for \u{10000}-\u{EFFFF} to match actual
173
+ // supplementary code points rather than lone surrogates (which are illegal XML).
174
+ // ---------------------------------------------------------------------------
175
+
176
+ /**
177
+ * @description Compiles the five production regexes from a NameStartChar/NameChar pair.
178
+ *
179
+ * @param startChar - Character class body for the first character.
180
+ * @param char - Character class body for every subsequent character.
181
+ * @param flags - RegExp flags. `'u'` for the XML 1.1 set, so its supplementary-plane range matches code points rather than lone surrogates.
182
+ *
183
+ * @returns One regex per {@link Production}.
184
+ */
185
+ const buildRegexes = (startChar: string, char: string, flags = ''): ProductionRegexes => {
186
+ const ncStart = startChar.replace(':', '');
187
+ const ncChar = char.replace(':', '');
188
+ const ncNamePat = `[${ncStart}][${ncChar}]*`;
189
+
190
+ return {
191
+ name: new RegExp(`^[${startChar}][${char}]*$`, flags),
192
+ ncName: new RegExp(`^${ncNamePat}$`, flags),
193
+ qName: new RegExp(`^${ncNamePat}(?::${ncNamePat})?$`, flags),
194
+ nmToken: new RegExp(`^[${char}]+$`, flags),
195
+ nmTokens: new RegExp(`^[${char}]+(?:\\s+[${char}]+)*$`, flags),
196
+ };
197
+ };
198
+
199
+ const regexes10 = buildRegexes(nameStartChar10, nameChar10); // no /u — BMP only
200
+ const regexes11 = buildRegexes(nameStartChar11, nameChar11, 'u'); // /u — enables \u{10000}-\u{EFFFF}
201
+
202
+ // ---------------------------------------------------------------------------
203
+ // ASCII-only fast path (opt-in, off by default)
204
+ //
205
+ // The XML 1.0 vs 1.1 NameStartChar/NameChar productions differ *only* in
206
+ // their non-ASCII ranges (merged vs split Latin-1 ranges, \u0487, and
207
+ // supplementary planes). Restricted to ASCII, both versions collapse to the
208
+ // same character classes, so a single regex pair covers both xmlVersion
209
+ // values — no /u flag needed.
210
+ //
211
+ // Rationale: unicode-aware regexes (the /u flag, required for XML 1.1's
212
+ // supplementary-plane range) are measurably slower in V8 than plain
213
+ // non-unicode regexes on the same input, even when the input is pure ASCII.
214
+ // For the common case — HTML/SVG ids, XML tags — names are ASCII, so callers
215
+ // who know this can opt in to skip the unicode-aware matching path entirely.
216
+ // This is a real but *conditional* win: mainly for XML 1.1 input (avoids /u),
217
+ // or at scale where the larger unicode character classes add engine
218
+ // overhead. It also changes behaviour (rejects legitimate non-ASCII XML
219
+ // 1.0/1.1 names), so it must never be silently enabled — hence off by
220
+ // default.
221
+ // ---------------------------------------------------------------------------
222
+
223
+ const nameStartCharAscii = ':A-Za-z_';
224
+ const nameCharAscii = nameStartCharAscii + '\\-\\.\\d';
225
+
226
+ const regexesAscii = buildRegexes(nameStartCharAscii, nameCharAscii); // no /u — ASCII only
227
+
228
+ /**
229
+ * @description The compiled regex set for a version/ASCII combination. Only three sets are ever built, at module load; this is a lookup, not a compile.
230
+ *
231
+ * @param xmlVersion - Which XML version's character classes to use.
232
+ * @param asciiOnly - Return the ASCII-only set regardless of `xmlVersion`.
233
+ *
234
+ * @returns The regex set to validate against.
235
+ */
236
+ const getRegexes = (xmlVersion: XmlVersion = '1.0', asciiOnly = false): ProductionRegexes => {
237
+ if (asciiOnly) return regexesAscii;
238
+ return xmlVersion === '1.1' ? regexes11 : regexes10;
239
+ };
240
+
241
+ // ---------------------------------------------------------------------------
242
+ // Boolean validators
243
+ //
244
+ // One plain predicate per production. A regex test cannot fail, and every one of these is called per name
245
+ // inside a parser's or codec's hot loop, so a boolean is the honest answer and there is no effect to
246
+ // allocate or run.
247
+ // ---------------------------------------------------------------------------
248
+
249
+ /**
250
+ * @description Whether the string is a valid XML Name. Colons are allowed anywhere (Name production). Used for: DOCTYPE entity names, notation names, DTD element
251
+ * declarations.
252
+ *
253
+ * @param str - The candidate name.
254
+ * @param opts - `asciiOnly` skips unicode-aware matching, ASCII names only (default false).
255
+ *
256
+ * @returns Whether `str` satisfies the production.
257
+ */
258
+ export const isName = (str: string, { xmlVersion = '1.0', asciiOnly = false }: ValidationOptions = {}): boolean =>
259
+ getRegexes(xmlVersion, asciiOnly).name.test(str);
260
+
261
+ /**
262
+ * @description Whether the string is a valid NCName (Non-Colonized Name).\
263
+ * Colons are not permitted.\
264
+ * Used for: namespace prefixes, local names, SVG id attributes.
265
+ *
266
+ * @param str - The candidate name.
267
+ * @param opts - `asciiOnly` skips unicode-aware matching, ASCII names only (default false).
268
+ *
269
+ * @returns Whether `str` satisfies the production.
270
+ */
271
+ export const isNcName = (str: string, { xmlVersion = '1.0', asciiOnly = false }: ValidationOptions = {}): boolean =>
272
+ getRegexes(xmlVersion, asciiOnly).ncName.test(str);
273
+
274
+ /**
275
+ * @description Whether the string is a valid QName (Qualified Name).\
276
+ * Allows exactly one colon as a prefix separator: `prefix:localName`.\
277
+ * Used for: element and attribute names in namespace-aware XML/SVG.
278
+ *
279
+ * @param str - The candidate name.
280
+ * @param opts - `asciiOnly` skips unicode-aware matching, ASCII names only (default false).
281
+ *
282
+ * @returns Whether `str` satisfies the production.
283
+ */
284
+ export const isQName = (str: string, { xmlVersion = '1.0', asciiOnly = false }: ValidationOptions = {}): boolean =>
285
+ getRegexes(xmlVersion, asciiOnly).qName.test(str);
286
+
287
+ /**
288
+ * @description Whether the string is a valid NMToken. Like Name but no restriction on the first character. Used for: DTD NMTOKEN attribute values.
289
+ *
290
+ * @param str - The candidate token.
291
+ * @param opts - `asciiOnly` skips unicode-aware matching, ASCII names only (default false).
292
+ *
293
+ * @returns Whether `str` satisfies the production.
294
+ */
295
+ export const isNmToken = (str: string, { xmlVersion = '1.0', asciiOnly = false }: ValidationOptions = {}): boolean =>
296
+ getRegexes(xmlVersion, asciiOnly).nmToken.test(str);
297
+
298
+ /**
299
+ * @description Whether the string is a valid NMTokens value — a whitespace-separated list of NMToken values. Used for: DTD NMTOKENS attribute values.
300
+ *
301
+ * @param str - The candidate list.
302
+ * @param opts - `asciiOnly` skips unicode-aware matching, ASCII names only (default false).
303
+ *
304
+ * @returns Whether `str` satisfies the production.
305
+ */
306
+ export const isNmTokens = (str: string, { xmlVersion = '1.0', asciiOnly = false }: ValidationOptions = {}): boolean =>
307
+ getRegexes(xmlVersion, asciiOnly).nmTokens.test(str);
308
+
309
+ /**
310
+ * @description The failure to report when a production is not one of the five this module knows, so `validate` reports it the same way. The single place the
311
+ * unknown-production guard lives. It is unreachable from TypeScript, where `Production` is a closed union; this is the guard for untyped JavaScript
312
+ * callers.
313
+ *
314
+ * @param production - The production to check.
315
+ *
316
+ * @returns The `InvalidProduction` failure, or `null` when the production is known.
317
+ */
318
+ const productionError = (production: Production): XmlError | null => {
319
+ if (PRODUCTIONS.includes(production)) return null;
320
+ return new XmlError({
321
+ reason: { _tag: 'InvalidProduction', production, expected: PRODUCTIONS.join(', ') },
322
+ message: `Unknown production "${production}". Must be one of: ${PRODUCTIONS.join(', ')}`,
323
+ });
324
+ };
325
+
326
+ const validators: Record<Production, (str: string, opts?: ValidationOptions) => boolean> = {
327
+ name: isName,
328
+ ncName: isNcName,
329
+ qName: isQName,
330
+ nmToken: isNmToken,
331
+ nmTokens: isNmTokens,
332
+ };
333
+
334
+ /**
335
+ * @description The diagnostic body {@link validate} reports, with the production already known to be valid. Kept separate so the reason-finding logic carries no
336
+ * unknown-production check and the batch path can map over it without re-entering the guard per element.
337
+ *
338
+ * @param str - The candidate name.
339
+ * @param production - The production to validate against, already checked.
340
+ * @param xmlVersion - Which version's character classes to use.
341
+ * @param asciiOnly - Whether the ASCII-only fast path applied.
342
+ *
343
+ * @returns The discriminated result.
344
+ */
345
+ const diagnose = (str: string, production: Production, xmlVersion: XmlVersion, asciiOnly: boolean): ValidationResult =>
346
+ diagnoseWith(str, production, validators[production](str, { xmlVersion, asciiOnly }), asciiOnly);
347
+
348
+ /**
349
+ * @description The colon-specific reason a `qName` or `ncName` failed, or `undefined` when the failure is not about a colon. The three QName forms are checked in
350
+ * the order they can co-occur: a string that both starts and ends with a colon is reported as the leading one, and a two-colon string is reported as
351
+ * the count.
352
+ *
353
+ * @param str - The candidate name.
354
+ * @param production - The production checked.
355
+ *
356
+ * @returns The reason and position, or `undefined`.
357
+ */
358
+ const diagnoseColon = (str: string, production: Production): { reason: string; position: number } | undefined => {
359
+ if (production === 'ncName' && str.includes(':')) {
360
+ return { reason: 'Colon is not allowed in NCName', position: str.indexOf(':') };
361
+ }
362
+
363
+ if (production !== 'qName') {
364
+ return undefined;
365
+ }
366
+ if (str.startsWith(':')) {
367
+ return { reason: 'QName cannot start with a colon', position: 0 };
368
+ }
369
+ if (str.endsWith(':')) {
370
+ return { reason: 'QName cannot end with a colon', position: str.length - 1 };
371
+ }
372
+ if ((str.match(/:/g) ?? []).length > 1) {
373
+ return { reason: 'QName can have at most one colon', position: str.lastIndexOf(':') };
374
+ }
375
+ return undefined;
376
+ };
377
+
378
+ /**
379
+ * @description The first character that is not a legal `NameChar`, and where it sits.
380
+ *
381
+ * @param str - The candidate name.
382
+ * @param namePattern - The `NameChar` test for the character set the validator used.
383
+ *
384
+ * @returns The reason and position, or `undefined` when every character is legal.
385
+ */
386
+ const diagnoseNameChar = (str: string, namePattern: RegExp): { reason: string; position: number } | undefined => {
387
+ for (let i = 0; i < str.length; i++) {
388
+ const char = str[i];
389
+ if (char !== undefined && !namePattern.test(char)) {
390
+ return { reason: `Character "${char}" at position ${i} is not a valid NameChar`, position: i };
391
+ }
392
+ }
393
+ return undefined;
394
+ };
395
+
396
+ /**
397
+ * @description Why a name failed, once whether it failed is already known. The first character outranks the rest: a name that cannot start is reported as a bad
398
+ * `NameStartChar` even when a later character is also illegal.
399
+ *
400
+ * @param str - The candidate name.
401
+ * @param production - The production checked.
402
+ * @param startCharPattern - The `NameStartChar` test for the character set the validator used.
403
+ * @param namePattern - The `NameChar` test for the same character set.
404
+ *
405
+ * @returns The reason, and the offending position, which is `undefined` for a structural failure.
406
+ */
407
+ const findFailure = (
408
+ str: string,
409
+ production: Production,
410
+ startCharPattern: RegExp,
411
+ namePattern: RegExp
412
+ ): { reason: string; position: number | undefined } => {
413
+ if (str.length === 0) {
414
+ return { reason: 'Input is empty', position: undefined };
415
+ }
416
+
417
+ const colon = diagnoseColon(str, production);
418
+ if (colon !== undefined) return colon;
419
+
420
+ const firstChar = str[0];
421
+ if (['name', 'ncName', 'qName'].includes(production) && !startCharPattern.test(firstChar ?? '')) {
422
+ return { reason: `First character "${firstChar}" is not a valid NameStartChar`, position: 0 };
423
+ }
424
+
425
+ return diagnoseNameChar(str, namePattern) ?? { reason: 'Does not match the production rules', position: undefined };
426
+ };
427
+
428
+ /**
429
+ * @description Why a name failed, once whether it failed is already known. Split from {@link diagnose} so the effect that asks the question stays a one-liner and
430
+ * the reason-finding is plain.
431
+ *
432
+ * @param str - The candidate name.
433
+ * @param production - The production checked.
434
+ * @param isValid - Whether it passed.
435
+ * @param asciiOnly - Whether the ASCII-only fast path applied.
436
+ *
437
+ * @returns The discriminated result.
438
+ */
439
+ const diagnoseWith = (str: string, production: Production, isValid: boolean, asciiOnly: boolean): ValidationResult => {
440
+ if (isValid) return { valid: true, production, input: str };
441
+
442
+ // Diagnostic fallback char checks must mirror the same character set the
443
+ // boolean validator above used, or the reported reason/position could
444
+ // contradict the `valid: false` result (e.g. flagging a char as illegal
445
+ // that the unicode-aware check would have accepted).
446
+ const startCharPattern = asciiOnly ? /^[:A-Za-z_]/ : /^[:A-Za-z_\u00C0-\uFFFD]/;
447
+ const namePattern = asciiOnly ? /[\w\-\\.:]/ : /[\w\-\\.:\u00B7\u00C0-\uFFFD]/;
448
+
449
+ return { valid: false, production, input: str, ...findFailure(str, production, startCharPattern, namePattern) };
450
+ };
451
+
452
+ /**
453
+ * @description Validates a string against a named production and, on failure, reports why and where.
454
+ *
455
+ * @example
456
+ * ```typescript
457
+ * import { Effect } from 'effect';
458
+ * import { validate } from '@endevops/effect-xml-codec';
459
+ *
460
+ * Effect.runSync(validate('not a name', 'ncName'));
461
+ * // { valid: false, production: 'ncName', input: 'not a name', reason: 'First character " " is not a valid NameStartChar', position: 0 }
462
+ * ```;
463
+ *
464
+ * @param str - The candidate name.
465
+ * @param production - The production to validate against.
466
+ * @param opts - Version and ASCII-only selection, as for the boolean validators.
467
+ *
468
+ * @returns An effect producing a discriminated result: the plain triple when valid, or the offending `reason` and `position` when not. A name that
469
+ * fails to validate is a `valid: false` result, not a failure — an invalid name is the question being answered. The effect fails only with an
470
+ * {@link XmlError} and the `InvalidProduction` reason for an unknown production, which is unreachable from TypeScript and is the guard for untyped
471
+ * JavaScript callers.
472
+ */
473
+ export const validate = Effect.fnUntraced(function* (
474
+ str: string,
475
+ production: Production,
476
+ { xmlVersion = '1.0', asciiOnly = false }: ValidationOptions = {}
477
+ ): Effect.fn.Return<ValidationResult, XmlError> {
478
+ const invalid = productionError(production);
479
+ if (invalid) return yield* invalid;
480
+ return diagnose(str, production, xmlVersion, asciiOnly);
481
+ });
482
+
483
+ // ---------------------------------------------------------------------------
484
+ // Sanitizer
485
+ // ---------------------------------------------------------------------------
486
+
487
+ /**
488
+ * @description Transforms an invalid string into the nearest valid XML name for the given production: strips or replaces illegal characters, fixes an invalid
489
+ * start character by prepending the replacement, and removes colons for NCName.
490
+ *
491
+ * @param str - The candidate name.
492
+ * @param production - The production to sanitize for. Defaults to `'name'`.
493
+ * @param opts - `replacement` is the substitute character (default `'_'`); `asciiOnly` also replaces non-ASCII characters.
494
+ *
495
+ * @returns A string that satisfies `production` for the ASCII range, or the nearest approximation of it.
496
+ */
497
+ export const sanitize = (str: string, production: Production = 'name', { replacement = '_', asciiOnly = false }: SanitizeOptions = {}): string => {
498
+ if (!str) return replacement;
499
+
500
+ let result = str;
501
+
502
+ // Strip colons for NCName
503
+ if (production === 'ncName') {
504
+ result = result.replace(/:/g, '');
505
+ }
506
+
507
+ // Replace illegal characters
508
+ const allowedCharPattern = asciiOnly ? /[^\w\-.:]/g : /[^\w\-.:\u00B7\u00C0-\uFFFD]/g;
509
+ result = result.replace(allowedCharPattern, replacement);
510
+
511
+ // Fix invalid start character for Name / NCName / QName
512
+ if (production !== 'nmToken' && production !== 'nmTokens') {
513
+ if (/^[-.\d]/.test(result)) {
514
+ result = replacement + result;
515
+ }
516
+ }
517
+
518
+ return result || replacement;
519
+ };