@endevops/effect-codec-xml 0.0.1 → 0.1.0-beta.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +81 -62
- package/dist/codec.d.ts +17 -9
- package/dist/codec.d.ts.map +1 -1
- package/dist/codec.js +29 -18
- package/dist/codec.js.map +1 -1
- package/dist/conventions.d.ts +4 -4
- package/dist/conventions.js +7 -7
- package/dist/conventions.js.map +1 -1
- package/dist/entities/entity-decoder.d.ts +33 -33
- package/dist/entities/entity-decoder.d.ts.map +1 -1
- package/dist/entities/entity-decoder.js +63 -64
- package/dist/entities/entity-decoder.js.map +1 -1
- package/dist/errors.d.ts +2 -2
- package/dist/errors.js +2 -2
- package/dist/errors.js.map +1 -1
- package/dist/namespaces.js +40 -13
- package/dist/namespaces.js.map +1 -1
- package/dist/naming.d.ts +6 -6
- package/dist/naming.d.ts.map +1 -1
- package/dist/naming.js +3 -3
- package/dist/naming.js.map +1 -1
- package/dist/parse.d.ts +5 -5
- package/dist/parse.js +14 -14
- package/dist/parse.js.map +1 -1
- package/dist/plain-value.js +242 -0
- package/dist/plain-value.js.map +1 -0
- package/dist/render.d.ts +1 -1
- package/dist/render.d.ts.map +1 -1
- package/dist/render.js +28 -28
- package/dist/render.js.map +1 -1
- package/dist/xml-error.d.ts +9 -9
- package/dist/xml-error.js +18 -18
- package/dist/xml-error.js.map +1 -1
- package/dist/xml-value.d.ts +7 -7
- package/dist/xml-value.d.ts.map +1 -1
- package/dist/xml-value.js +6 -7
- package/dist/xml-value.js.map +1 -1
- package/package.json +1 -1
- package/src/codec.ts +98 -71
- package/src/conventions.ts +7 -7
- package/src/entities/entity-decoder.ts +98 -99
- package/src/errors.ts +3 -3
- package/src/index.ts +3 -3
- package/src/namespaces.ts +62 -36
- package/src/naming.ts +34 -35
- package/src/parse.ts +26 -26
- package/src/plain-value.ts +312 -0
- package/src/render.ts +44 -44
- package/src/xml-error.ts +18 -18
- package/src/xml-value.ts +10 -11
|
@@ -0,0 +1,312 @@
|
|
|
1
|
+
// A plain value read from an element that carries more than character data.
|
|
2
|
+
//
|
|
3
|
+
// The parser reduces an element with no attributes and no children to its
|
|
4
|
+
// character data, so a field like `Schema.String` is usually handed a bare
|
|
5
|
+
// string. An element that also carries attributes does not reduce: it arrives
|
|
6
|
+
// as a record holding the attributes, the `#text` character data, and any child
|
|
7
|
+
// elements, because the value model has nowhere else to put them:
|
|
8
|
+
//
|
|
9
|
+
// { "@lang": "en", "#text": "Dune" }
|
|
10
|
+
//
|
|
11
|
+
// Effect's `Schema.toCodecStringTree` derivation knows nothing about the `#text`
|
|
12
|
+
// convention, so where a field wants a scalar it sees a record and fails with an
|
|
13
|
+
// `InvalidType`. This module bridges the two. Walking the derived StringTree AST
|
|
14
|
+
// beside the parsed value:
|
|
15
|
+
//
|
|
16
|
+
// - a node that wants character data takes the `#text` key and discards the
|
|
17
|
+
// attributes, because an attribute is not part of the value the schema
|
|
18
|
+
// describes;
|
|
19
|
+
// - a node that wants character data but also finds a child element fails,
|
|
20
|
+
// because a plain value cannot hold one and silently dropping it would lose
|
|
21
|
+
// a field the document actually carried;
|
|
22
|
+
// - a node that describes a struct, array or record keeps its record and
|
|
23
|
+
// recurses, because `@`-prefixed fields and `#text` are ordinary field names
|
|
24
|
+
// to such a schema;
|
|
25
|
+
// - a union rewrites its record against each object member in turn, so a plain
|
|
26
|
+
// value nested under any branch is still read; a repeated field derives as a
|
|
27
|
+
// union of an array and `undefined`, so a record is also read through the
|
|
28
|
+
// first array node reachable behind a union;
|
|
29
|
+
// - an empty element under an array field reads as one empty object when the
|
|
30
|
+
// member is structural, because an empty array renders as a single empty
|
|
31
|
+
// element (see `renderRepeated`) and a member the schema requires must still
|
|
32
|
+
// read; a plain-value member reads as an empty array.
|
|
33
|
+
//
|
|
34
|
+
// The AST is the one `Schema.toCodecStringTree` produces for the source schema,
|
|
35
|
+
// so it is the same shape the decoder is about to read; this pass only rewrites
|
|
36
|
+
// the value, it does not derive a schema.
|
|
37
|
+
|
|
38
|
+
import { Predicate, Result, SchemaAST } from 'effect';
|
|
39
|
+
|
|
40
|
+
import type { XmlRecord, XmlValue } from './xml-value.ts';
|
|
41
|
+
|
|
42
|
+
import { isAttributeKey, TEXT_KEY } from './conventions.ts';
|
|
43
|
+
|
|
44
|
+
/**
|
|
45
|
+
* @description Whether an AST describes a structure - a struct, an array, or a union reachable to one - rather than a plain value. A node that describes a
|
|
46
|
+
* structure keeps a record as a record; a node that does not is where `#text` is read.
|
|
47
|
+
*
|
|
48
|
+
* @param ast - The derived StringTree AST to classify.
|
|
49
|
+
*
|
|
50
|
+
* @returns Whether the node is structural.
|
|
51
|
+
*/
|
|
52
|
+
const isStructural = (ast: SchemaAST.AST): boolean => {
|
|
53
|
+
if (SchemaAST.isSuspend(ast)) return isStructural(ast.thunk());
|
|
54
|
+
if (SchemaAST.isObjects(ast) || SchemaAST.isArrays(ast)) return true;
|
|
55
|
+
return SchemaAST.isUnion(ast) && ast.types.some(isStructural);
|
|
56
|
+
};
|
|
57
|
+
|
|
58
|
+
/**
|
|
59
|
+
* @description The AST a value is read against, with `Suspend` wrappers unwrapped so a recursive schema reaches its node.
|
|
60
|
+
*
|
|
61
|
+
* @param ast - The AST to unwrap.
|
|
62
|
+
*
|
|
63
|
+
* @returns The first node that is not a `Suspend`.
|
|
64
|
+
*/
|
|
65
|
+
const resolveNode = (ast: SchemaAST.AST): SchemaAST.AST => {
|
|
66
|
+
let node = ast;
|
|
67
|
+
while (SchemaAST.isSuspend(node)) node = node.thunk();
|
|
68
|
+
return node;
|
|
69
|
+
};
|
|
70
|
+
|
|
71
|
+
/**
|
|
72
|
+
* @description The AST a record value under one key is read against: the matching property signature, or the first index signature when the object is a record.
|
|
73
|
+
* `undefined` leaves the value alone, which is what an object with neither describes.
|
|
74
|
+
*
|
|
75
|
+
* @param node - The object node.
|
|
76
|
+
* @param key - The record key.
|
|
77
|
+
*
|
|
78
|
+
* @returns The AST for the value, or `undefined`.
|
|
79
|
+
*/
|
|
80
|
+
const fieldAst = (node: SchemaAST.Objects, key: string): SchemaAST.AST | undefined => {
|
|
81
|
+
const property = node.propertySignatures.find(candidate => candidate.name === key);
|
|
82
|
+
return property?.type ?? node.indexSignatures[0]?.type;
|
|
83
|
+
};
|
|
84
|
+
|
|
85
|
+
/**
|
|
86
|
+
* @description The AST one member of an array is read against: the tuple element at that index, or the array's rest element.
|
|
87
|
+
*
|
|
88
|
+
* @param node - The array node.
|
|
89
|
+
* @param index - The member's index.
|
|
90
|
+
*
|
|
91
|
+
* @returns The AST for the member, or `undefined`.
|
|
92
|
+
*/
|
|
93
|
+
const memberAst = (node: SchemaAST.Arrays, index: number): SchemaAST.AST | undefined => node.elements[index] ?? node.rest[0];
|
|
94
|
+
|
|
95
|
+
/**
|
|
96
|
+
* @description The array node a value is read against. An optional repeated field derives as a union of an array and `undefined`, so an array node can sit behind
|
|
97
|
+
* a union rather than at the top; the first one reachable through the union's members is the one a repeated value belongs to.
|
|
98
|
+
*
|
|
99
|
+
* @param node - The resolved node to search.
|
|
100
|
+
*
|
|
101
|
+
* @returns The array node, or `undefined` when none is reachable.
|
|
102
|
+
*/
|
|
103
|
+
const arrayNode = (node: SchemaAST.AST): SchemaAST.Arrays | undefined => {
|
|
104
|
+
if (SchemaAST.isArrays(node)) return node;
|
|
105
|
+
if (SchemaAST.isUnion(node)) {
|
|
106
|
+
for (const member of node.types) {
|
|
107
|
+
const found = arrayNode(resolveNode(member));
|
|
108
|
+
if (found !== undefined) return found;
|
|
109
|
+
}
|
|
110
|
+
}
|
|
111
|
+
return undefined;
|
|
112
|
+
};
|
|
113
|
+
|
|
114
|
+
/**
|
|
115
|
+
* @description Whether an array's member is a structure - a struct, an array, or a union reachable to one - rather than a plain value. An empty element under such
|
|
116
|
+
* an array reads as one empty object, because a structural member cannot be empty character data; a plain-value member reads as an empty array
|
|
117
|
+
* instead.
|
|
118
|
+
*
|
|
119
|
+
* @param node - The array node.
|
|
120
|
+
*
|
|
121
|
+
* @returns Whether the member is structural.
|
|
122
|
+
*/
|
|
123
|
+
const arrayMemberIsStructural = (node: SchemaAST.Arrays): boolean => {
|
|
124
|
+
const member = memberAst(node, 0);
|
|
125
|
+
return member !== undefined && isStructural(member);
|
|
126
|
+
};
|
|
127
|
+
|
|
128
|
+
/**
|
|
129
|
+
* @description The value an empty element under an array field reads as: one empty object when the member is structural, so a member the schema requires still
|
|
130
|
+
* reads, and an empty array otherwise.
|
|
131
|
+
*
|
|
132
|
+
* @param node - The array node.
|
|
133
|
+
*
|
|
134
|
+
* @returns The array to decode.
|
|
135
|
+
*/
|
|
136
|
+
const emptyArrayElement = (node: SchemaAST.Arrays): XmlValue => (arrayMemberIsStructural(node) ? [{}] : []);
|
|
137
|
+
|
|
138
|
+
/**
|
|
139
|
+
* @description Whether a parsed element carries nothing at all: no character data, no attributes and no children. The parser reduces such an element to an empty
|
|
140
|
+
* string, and the codec reduces one whose only attributes were namespace declarations to an empty record. An array field reads either as one empty
|
|
141
|
+
* object or as an empty array, depending on whether its member is structural.
|
|
142
|
+
*
|
|
143
|
+
* @param value - The parsed value.
|
|
144
|
+
*
|
|
145
|
+
* @returns Whether the element is empty.
|
|
146
|
+
*/
|
|
147
|
+
const isEmptyElement = (value: XmlValue): boolean => value === '' || (Predicate.isReadonlyObject(value) && Object.keys(value).length === 0);
|
|
148
|
+
|
|
149
|
+
/**
|
|
150
|
+
* @description The path of a field, for a failure message. The root has no name, so its fields are named on their own.
|
|
151
|
+
*
|
|
152
|
+
* @param parent - The parent's path.
|
|
153
|
+
* @param key - The field's key.
|
|
154
|
+
*
|
|
155
|
+
* @returns The field's path.
|
|
156
|
+
*/
|
|
157
|
+
const childPath = (parent: string, key: string): string => (parent === '' ? key : `${parent}.${key}`);
|
|
158
|
+
|
|
159
|
+
/**
|
|
160
|
+
* @description Reads the character data of a record that wants a plain value, discarding the attributes around it. A record that also carries a child element is
|
|
161
|
+
* refused, because a plain value has nowhere to put one and dropping it would lose a field the document carried. A record with no `#text` at all is
|
|
162
|
+
* left for the decoder to refuse, which reports the attributes it found instead of a value.
|
|
163
|
+
*
|
|
164
|
+
* @param record - The record to read.
|
|
165
|
+
* @param path - The path of the value, for the failure message.
|
|
166
|
+
*
|
|
167
|
+
* @returns The `#text` value, or the message describing the child element that makes a plain value impossible.
|
|
168
|
+
*/
|
|
169
|
+
const readCharacterData = (record: XmlRecord, path: string): Result.Result<XmlValue, string> => {
|
|
170
|
+
if (!(TEXT_KEY in record)) return Result.succeed(record);
|
|
171
|
+
|
|
172
|
+
const child = Object.keys(record).find(key => key !== TEXT_KEY && !isAttributeKey(key));
|
|
173
|
+
if (child !== undefined) {
|
|
174
|
+
return Result.fail(
|
|
175
|
+
`the field "${path === '' ? 'root' : path}" wants a plain value, but the element also carries the child element "${child}"; only attributes are discarded alongside ${TEXT_KEY}`
|
|
176
|
+
);
|
|
177
|
+
}
|
|
178
|
+
|
|
179
|
+
return Result.succeed(record[TEXT_KEY]);
|
|
180
|
+
};
|
|
181
|
+
|
|
182
|
+
/**
|
|
183
|
+
* @description Folds the fields of a record against the object node that describes them, so each field is normalized by its own AST.
|
|
184
|
+
*
|
|
185
|
+
* @param record - The record to fold.
|
|
186
|
+
* @param node - The object node.
|
|
187
|
+
* @param path - The path of the record, for a failure message.
|
|
188
|
+
*
|
|
189
|
+
* @returns The folded record, or the first field's failure.
|
|
190
|
+
*/
|
|
191
|
+
const normalizeFields = (record: XmlRecord, node: SchemaAST.Objects, path: string): Result.Result<XmlValue, string> => {
|
|
192
|
+
const out: Record<string, XmlValue> = {};
|
|
193
|
+
for (const [key, child] of Object.entries(record)) {
|
|
194
|
+
const field = fieldAst(node, key);
|
|
195
|
+
if (field === undefined || child === undefined) {
|
|
196
|
+
out[key] = child;
|
|
197
|
+
continue;
|
|
198
|
+
}
|
|
199
|
+
const normalized = normalizePlainValue(child, field, childPath(path, key));
|
|
200
|
+
if (Result.isFailure(normalized)) return normalized;
|
|
201
|
+
out[key] = normalized.success;
|
|
202
|
+
}
|
|
203
|
+
return Result.succeed(out);
|
|
204
|
+
};
|
|
205
|
+
|
|
206
|
+
/**
|
|
207
|
+
* @description Folds every member of an array against the array node that describes them. A member that is an empty element - the parser reduces it to an empty
|
|
208
|
+
* string, or to an empty record once the codec drops the namespace declarations it carried - cannot be a structural member, so it is kept as one
|
|
209
|
+
* empty object. That is how the renderer writes an empty array, and how a repeated empty tag reads back as one empty object per element.
|
|
210
|
+
*
|
|
211
|
+
* @param value - The array to fold.
|
|
212
|
+
* @param node - The array node.
|
|
213
|
+
* @param path - The path of the array, for a failure message.
|
|
214
|
+
*
|
|
215
|
+
* @returns The folded array, or the first member's failure.
|
|
216
|
+
*/
|
|
217
|
+
const normalizeMembers = (value: ReadonlyArray<XmlValue>, node: SchemaAST.Arrays, path: string): Result.Result<XmlValue, string> => {
|
|
218
|
+
const out: Array<XmlValue> = [];
|
|
219
|
+
for (let index = 0; index < value.length; index++) {
|
|
220
|
+
const element = memberAst(node, index);
|
|
221
|
+
if (element === undefined) {
|
|
222
|
+
out.push(value[index]);
|
|
223
|
+
continue;
|
|
224
|
+
}
|
|
225
|
+
if (isEmptyElement(value[index]) && isStructural(element)) {
|
|
226
|
+
out.push({});
|
|
227
|
+
continue;
|
|
228
|
+
}
|
|
229
|
+
const normalized = normalizePlainValue(value[index], element, `${path}[${index}]`);
|
|
230
|
+
if (Result.isFailure(normalized)) return normalized;
|
|
231
|
+
out.push(normalized.success);
|
|
232
|
+
}
|
|
233
|
+
return Result.succeed(out);
|
|
234
|
+
};
|
|
235
|
+
|
|
236
|
+
/**
|
|
237
|
+
* @description Folds a record against every object member of a union, so a plain value nested under any branch is still read from its character data. Which branch
|
|
238
|
+
* the value belongs to is the decoder's job, so the first branch that folds the record cleanly wins; a branch that refuses the record - because a
|
|
239
|
+
* field it wants as a plain value also carries a child element - is skipped. When the union has no object member, or none of them folds the record,
|
|
240
|
+
* the record is left for the decoder to read as a plain value.
|
|
241
|
+
*
|
|
242
|
+
* @param record - The record to fold.
|
|
243
|
+
* @param node - The union node.
|
|
244
|
+
* @param path - The path of the record, for a failure message.
|
|
245
|
+
*
|
|
246
|
+
* @returns The folded value, the first object member's failure, or the record unchanged.
|
|
247
|
+
*/
|
|
248
|
+
const normalizeUnion = (record: XmlRecord, node: SchemaAST.Union, path: string): Result.Result<XmlValue, string> => {
|
|
249
|
+
let failure: Result.Result<XmlValue, string> | undefined;
|
|
250
|
+
let sawObject = false;
|
|
251
|
+
for (const member of node.types) {
|
|
252
|
+
const resolved = resolveNode(member);
|
|
253
|
+
if (!SchemaAST.isObjects(resolved)) continue;
|
|
254
|
+
sawObject = true;
|
|
255
|
+
const normalized = normalizeFields(record, resolved, path);
|
|
256
|
+
if (Result.isSuccess(normalized)) return normalized;
|
|
257
|
+
failure ??= normalized;
|
|
258
|
+
}
|
|
259
|
+
if (failure !== undefined) return failure;
|
|
260
|
+
return sawObject ? Result.succeed(record) : readCharacterData(record, path);
|
|
261
|
+
};
|
|
262
|
+
|
|
263
|
+
/**
|
|
264
|
+
* @description Folds a record against the node that describes it. An object node is folded field by field; a union folds against its object members so a plain
|
|
265
|
+
* value nested under any branch is read; anything else wants a plain value and reads the character data.
|
|
266
|
+
*
|
|
267
|
+
* @param record - The record to fold.
|
|
268
|
+
* @param node - The resolved node.
|
|
269
|
+
* @param path - The path of the record, for a failure message.
|
|
270
|
+
*
|
|
271
|
+
* @returns The folded value, or the failure that makes it impossible.
|
|
272
|
+
*/
|
|
273
|
+
const normalizeRecord = (record: XmlRecord, node: SchemaAST.AST, path: string): Result.Result<XmlValue, string> => {
|
|
274
|
+
if (SchemaAST.isObjects(node)) return normalizeFields(record, node, path);
|
|
275
|
+
if (SchemaAST.isUnion(node)) return normalizeUnion(record, node, path);
|
|
276
|
+
if (isStructural(node)) return Result.succeed(record);
|
|
277
|
+
return readCharacterData(record, path);
|
|
278
|
+
};
|
|
279
|
+
|
|
280
|
+
/**
|
|
281
|
+
* @description Folds a parsed value against the derived StringTree AST, so a plain value can be read from an element that carries attributes. An empty element
|
|
282
|
+
* under an array field reads as one empty object when the member is structural, and as an empty array otherwise.
|
|
283
|
+
*
|
|
284
|
+
* @param value - The value the parser produced, after namespaces were resolved.
|
|
285
|
+
* @param ast - The derived StringTree AST the decoder will read the value with.
|
|
286
|
+
* @param path - The path of this value, for a failure message.
|
|
287
|
+
*
|
|
288
|
+
* @returns The value to decode, or the message describing why a plain value could not be read.
|
|
289
|
+
*/
|
|
290
|
+
export const normalizePlainValue = (value: XmlValue, ast: SchemaAST.AST, path: string): Result.Result<XmlValue, string> => {
|
|
291
|
+
if (Predicate.isUndefined(value)) return Result.succeed(value);
|
|
292
|
+
|
|
293
|
+
const node = resolveNode(ast);
|
|
294
|
+
const arrays = arrayNode(node);
|
|
295
|
+
|
|
296
|
+
if (Predicate.isString(value)) {
|
|
297
|
+
// An empty element under an array field is how the renderer writes an empty array; a structural member still reads as one empty object.
|
|
298
|
+
if (arrays !== undefined && value === '') return Result.succeed(emptyArrayElement(arrays));
|
|
299
|
+
return Result.succeed(value);
|
|
300
|
+
}
|
|
301
|
+
|
|
302
|
+
if (Array.isArray(value)) {
|
|
303
|
+
return arrays !== undefined ? normalizeMembers(value, arrays, path) : Result.succeed(value);
|
|
304
|
+
}
|
|
305
|
+
|
|
306
|
+
if (!Predicate.isReadonlyObject(value)) return Result.succeed(value);
|
|
307
|
+
|
|
308
|
+
// The same empty element, after the codec dropped the namespace declarations it carried, arrives as an empty record.
|
|
309
|
+
if (arrays !== undefined && isEmptyElement(value)) return Result.succeed(emptyArrayElement(arrays));
|
|
310
|
+
|
|
311
|
+
return normalizeRecord(value as XmlRecord, node, path);
|
|
312
|
+
};
|
package/src/render.ts
CHANGED
|
@@ -7,23 +7,23 @@
|
|
|
7
7
|
// string.
|
|
8
8
|
//
|
|
9
9
|
// The walk itself is plain synchronous functions rather than a chain of
|
|
10
|
-
// `yield*`es. Publicly `renderXml` is still an `Effect
|
|
10
|
+
// `yield*`es. Publicly `renderXml` is still an `Effect`: it suspends the walk so
|
|
11
11
|
// it runs lazily, and folds the one failure the walk can report into the typed
|
|
12
|
-
// error channel
|
|
13
|
-
// or per attribute. A 500-row report is thousands of elements, and
|
|
14
|
-
//
|
|
15
|
-
//
|
|
16
|
-
//
|
|
12
|
+
// error channel, but inside a document there is no effect boundary per element
|
|
13
|
+
// or per attribute. A 500-row report is thousands of elements, and one fiber
|
|
14
|
+
// step per element dominated the `render 500 rows` benchmark. The typed failure
|
|
15
|
+
// survives: the walk throws an {@link XmlRenderError} and `renderXml` catches
|
|
16
|
+
// it into `Effect.fail`.
|
|
17
17
|
//
|
|
18
18
|
// Escaping is the part that scales with the size of the document rather than
|
|
19
19
|
// with its structure, and it is written out here rather than delegated, for a
|
|
20
20
|
// measured reason. The entity encoder that used to live beside this package
|
|
21
21
|
// escaped by applying five sequential global replacements, one per character,
|
|
22
22
|
// so a document with a single `&` in twenty thousand characters was scanned
|
|
23
|
-
// five times over to change one byte
|
|
24
|
-
// `bench/codec.bench.ts` measure. The table below covers
|
|
25
|
-
// characters that encoder escaped, and the explicit expectations
|
|
26
|
-
// `test/render.spec.ts` pin the fast path.
|
|
23
|
+
// five times over to change one byte. The `render 20k` rows in
|
|
24
|
+
// `bench/codec.bench.ts` measure that five-pass cost. The table below covers
|
|
25
|
+
// the same five characters that encoder escaped, and the explicit expectations
|
|
26
|
+
// in `test/render.spec.ts` pin the fast path.
|
|
27
27
|
|
|
28
28
|
import { Effect, Predicate, Result } from 'effect';
|
|
29
29
|
|
|
@@ -65,11 +65,12 @@ const TEXT_TABLE = buildTable({});
|
|
|
65
65
|
const ATTRIBUTE_TABLE = buildTable(ATTRIBUTE_WHITESPACE);
|
|
66
66
|
|
|
67
67
|
/**
|
|
68
|
-
* @description The characters each table escapes, as a pattern rather than as a set of replacement passes.
|
|
69
|
-
* clean text cheap: V8 compiles a single character class into a scan that is several times faster than a JavaScript loop
|
|
70
|
-
* code unit at a time, and clean text is most text. `render 20k of clean text` in `bench/codec.bench.ts`
|
|
71
|
-
* loop over the same twenty thousand characters is roughly two and a half times slower. Neither pattern is
|
|
72
|
-
* always starts at the beginning. One module-level instance of each is therefore safe to reuse, and nothing
|
|
68
|
+
* @description The characters each table escapes, as a pattern rather than as a set of replacement passes. A single pattern finds the first character that needs
|
|
69
|
+
* replacing, which keeps clean text cheap: V8 compiles a single character class into a scan that is several times faster than a JavaScript loop
|
|
70
|
+
* reading the same string a code unit at a time, and clean text is most text. The `render 20k of clean text` benchmark in `bench/codec.bench.ts`
|
|
71
|
+
* measures that difference: a hand-written loop over the same twenty thousand characters is roughly two and a half times slower. Neither pattern is
|
|
72
|
+
* global, so `exec` ignores `lastIndex` and always starts at the beginning. One module-level instance of each is therefore safe to reuse, and nothing
|
|
73
|
+
* has to be reset between calls.
|
|
73
74
|
*/
|
|
74
75
|
const TEXT_UNSAFE = /[<>&"']/;
|
|
75
76
|
const ATTRIBUTE_UNSAFE = /[<>&"'\n\r\t]/;
|
|
@@ -171,9 +172,9 @@ interface ResolvedOptions {
|
|
|
171
172
|
/**
|
|
172
173
|
* @description Builds the name resolver for one render. Every element and every attribute name goes through here, and a document repeats names: a thousand
|
|
173
174
|
* `<item>` elements, or the same `id` on every row. A validator that runs a regex per occurrence pays that cost a thousand times for one answer, so
|
|
174
|
-
* the first result is remembered and the rest are lookups.
|
|
175
|
-
*
|
|
176
|
-
*
|
|
175
|
+
* the first result is remembered and the rest are lookups. Keeping the mode and version in one place also stops a caller from resolving a name with
|
|
176
|
+
* different settings than the render it is part of. A cache miss calls {@link resolveNameSync}, which throws an {@link XmlParseError} in `'error'`
|
|
177
|
+
* mode; {@link renderXml} catches it and reports it as an {@link XmlRenderError}.
|
|
177
178
|
*
|
|
178
179
|
* @param options - Resolved render options.
|
|
179
180
|
*
|
|
@@ -193,8 +194,7 @@ const makeNamer = (options: Omit<ResolvedOptions, 'namer' | 'lineAt'>): ((name:
|
|
|
193
194
|
/**
|
|
194
195
|
* @description A boolean option's value, with an absent one read as the default. The three boolean options are spelled through here rather than through a `??` of
|
|
195
196
|
* their own, so the table below reads as a list of what each option _is_ instead of a list of nine separate decisions about what an omitted option
|
|
196
|
-
* means
|
|
197
|
-
* read.
|
|
197
|
+
* means, and a reader looking for "which options are on by default" finds three words rather than three mixes of `?? true` and `?? false` to read.
|
|
198
198
|
*
|
|
199
199
|
* @param value - The option as the caller wrote it, or `undefined` when the caller left it out.
|
|
200
200
|
* @param fallback - The value to use when the caller left it out.
|
|
@@ -265,10 +265,10 @@ export const escapeAttribute = (value: string): string => escape(value, ATTRIBUT
|
|
|
265
265
|
|
|
266
266
|
/**
|
|
267
267
|
* @description Replaces every character the table has an entry for, in one pass over the string. The pattern finds the first character that needs replacing, and a
|
|
268
|
-
* string with none is handed straight back
|
|
269
|
-
*
|
|
270
|
-
*
|
|
271
|
-
*
|
|
268
|
+
* string with none is handed straight back. That is the common case, and the pattern exists to keep it fast. From there the rest of the string is
|
|
269
|
+
* copied in runs between the replacements rather than a character at a time, so the cost is one pattern scan, one copy, and one concatenation per
|
|
270
|
+
* replacement, rather than a whole pass per character class. Only ASCII is looked up. XML carries every other character natively, and a code unit
|
|
271
|
+
* above 127 has no entity an XML parser is required to know.
|
|
272
272
|
*
|
|
273
273
|
* @param value - The text to escape.
|
|
274
274
|
* @param pattern - Matches the first character that needs replacing.
|
|
@@ -302,7 +302,7 @@ const escape = (value: string, pattern: RegExp, table: ReadonlyArray<string | un
|
|
|
302
302
|
* A record becomes an element:
|
|
303
303
|
*
|
|
304
304
|
* - `@`-prefixed keys become attributes, the reserved `#text` key becomes character data, and every other key becomes a child element.
|
|
305
|
-
* - An array repeats its name
|
|
305
|
+
* - An array repeats its name. A document whose root value is an array wraps it in the root element and names each member `itemName`.
|
|
306
306
|
* - A string is character data. The walk is synchronous, and what can go wrong is reported by throwing an {@link XmlRenderError}; {@link renderXml}
|
|
307
307
|
* folds that into the effect's typed error channel. A caller not already in an `Effect` runs it with `Effect.runSync`, which throws the failure it
|
|
308
308
|
* produced.
|
|
@@ -379,9 +379,9 @@ const render = (value: XmlValue, options: XmlRenderOptions): string => {
|
|
|
379
379
|
};
|
|
380
380
|
|
|
381
381
|
/**
|
|
382
|
-
* @description Renders one named element and its subtree. The value an {@link XmlValue} holds decides which of the four shapes below it takes
|
|
383
|
-
* children, character data, an absent field, or a record
|
|
384
|
-
*
|
|
382
|
+
* @description Renders one named element and its subtree. The value an {@link XmlValue} holds decides which of the four shapes below it takes: a repeated run of
|
|
383
|
+
* children, character data, an absent field, or a record. Each of those is written by a function of its own, so this one is the dispatch rather than
|
|
384
|
+
* the document.
|
|
385
385
|
*
|
|
386
386
|
* @param out - The chunk buffer to append to.
|
|
387
387
|
* @param name - The element name, not yet resolved.
|
|
@@ -397,8 +397,8 @@ const renderElement = (out: Array<string>, name: string, value: XmlValue, depth:
|
|
|
397
397
|
return;
|
|
398
398
|
}
|
|
399
399
|
|
|
400
|
-
// An element opens its own line rather than having its caller do it,
|
|
401
|
-
//
|
|
400
|
+
// An element opens its own line rather than having its caller do it, so a
|
|
401
|
+
// repeated run of children stays on separate lines. The root is the
|
|
402
402
|
// one element that has nothing in front of it.
|
|
403
403
|
if (options.format && depth > 0) openLine(out, depth, options);
|
|
404
404
|
|
|
@@ -417,7 +417,7 @@ const renderElement = (out: Array<string>, name: string, value: XmlValue, depth:
|
|
|
417
417
|
|
|
418
418
|
/**
|
|
419
419
|
* @description Refuses to walk deeper than the render allows. A value can nest without end, and every one of those levels costs a stack frame here, so the cap is
|
|
420
|
-
* checked
|
|
420
|
+
* checked as the walk descends rather than trusted to the caller.
|
|
421
421
|
*
|
|
422
422
|
* @param depth - The depth about to be written.
|
|
423
423
|
* @param options - Resolved render options.
|
|
@@ -459,8 +459,8 @@ const renderRepeated = (out: Array<string>, name: string, members: ReadonlyArray
|
|
|
459
459
|
|
|
460
460
|
/**
|
|
461
461
|
* @description Renders an element whose value is character data, or nothing. An empty string is character data that happens to be empty, and an element holding
|
|
462
|
-
* none of it is the same element as one holding nothing at all
|
|
463
|
-
* that never went through the schema
|
|
462
|
+
* none of it is the same element as one holding nothing at all, as is an `undefined` element, which is an absent one. The renderer is handed values
|
|
463
|
+
* that never went through the schema (a caller building a document by hand), so the absent case is reachable, and an empty element is the correct
|
|
464
464
|
* rendering of both.
|
|
465
465
|
*
|
|
466
466
|
* @param out - The chunk buffer to append to.
|
|
@@ -538,9 +538,9 @@ interface Fields {
|
|
|
538
538
|
* @description One pass over a record's keys, collecting all three roles at once: the attributes are rendered as they are found, the child names are set aside for
|
|
539
539
|
* the pass that writes them, and the text key is left to {@link textOf}. A pass for the attributes, a pass for the children and an index for the text
|
|
540
540
|
* instead walks the keys three times and allocates the key array twice, which on a document of a few thousand elements is thousands of allocations
|
|
541
|
-
* for nothing. Sorting is off by default, and the default path is the one
|
|
541
|
+
* for nothing. Sorting is off by default, and the default path is the important one, so the attributes are built as they are found and there is
|
|
542
542
|
* nothing to sort. When it is on, the attribute keys are collected instead and rendered afterwards in sorted order, which costs an array per element
|
|
543
|
-
* and
|
|
543
|
+
* and gives output independent of the order the fields happened to be declared in.
|
|
544
544
|
*
|
|
545
545
|
* @param record - The element's value.
|
|
546
546
|
* @param options - Resolved render options.
|
|
@@ -557,10 +557,10 @@ const collectFields = (record: XmlRecord, options: ResolvedOptions): Fields => {
|
|
|
557
557
|
const key = keys[i] as string;
|
|
558
558
|
const child = record[key];
|
|
559
559
|
|
|
560
|
-
// An absent field is not written at all,
|
|
561
|
-
//
|
|
562
|
-
//
|
|
563
|
-
//
|
|
560
|
+
// An absent field is not written at all, so an unset optional attribute
|
|
561
|
+
// stays out of the document rather than appearing as `a=""`, and an absent
|
|
562
|
+
// child stays out rather than appearing as `<a/>`. The text key is read by
|
|
563
|
+
// `textOf` either way, so skipping it here costs nothing.
|
|
564
564
|
if (child === undefined) continue;
|
|
565
565
|
|
|
566
566
|
if (isAttributeKey(key)) {
|
|
@@ -647,9 +647,9 @@ const openLine = (out: Array<string>, depth: number, options: ResolvedOptions):
|
|
|
647
647
|
|
|
648
648
|
/**
|
|
649
649
|
* @description Writes an element with no content, in whichever of the two forms the options ask for. Every path that produces an element with nothing in it goes
|
|
650
|
-
* through here, so the self-closing decision is made in exactly one place. That
|
|
651
|
-
* empty string, an absent value, an empty array, and a record whose fields are all absent
|
|
652
|
-
*
|
|
650
|
+
* through here, so the self-closing decision is made in exactly one place. That is important because "nothing in it" arrives four different ways: an
|
|
651
|
+
* empty string, an absent value, an empty array, and a record whose fields are all absent. Four separate decisions are four chances for one of them
|
|
652
|
+
* to write the long form by accident.
|
|
653
653
|
*
|
|
654
654
|
* @param out - The chunk buffer to append to.
|
|
655
655
|
* @param tag - The element's name, already resolved.
|
|
@@ -690,8 +690,8 @@ const attributeText = (value: XmlValue): string => {
|
|
|
690
690
|
};
|
|
691
691
|
|
|
692
692
|
/**
|
|
693
|
-
* @description Renders a leaf that is not a string as the character data an XML document can hold. A schema-derived value never reaches here
|
|
694
|
-
* `Schema.toCodecStringTree` has already turned every scalar into a string
|
|
693
|
+
* @description Renders a leaf that is not a string as the character data an XML document can hold. A schema-derived value never reaches here, because
|
|
694
|
+
* `Schema.toCodecStringTree` has already turned every scalar into a string, so this is for values a caller built by hand. A value with no sensible
|
|
695
695
|
* text form is rendered as nothing rather than as `[object Object]`, which would silently write a document that parses back to something else.
|
|
696
696
|
*
|
|
697
697
|
* @param value - The leaf to render.
|
package/src/xml-error.ts
CHANGED
|
@@ -6,38 +6,38 @@
|
|
|
6
6
|
*
|
|
7
7
|
* ## Why one error with a `reason`, and not one error per cause
|
|
8
8
|
*
|
|
9
|
-
* The causes would mean a class each, and a caller handling "any of these" would need `catchTags` with all of them.
|
|
10
|
-
*
|
|
11
|
-
*
|
|
9
|
+
* The causes would mean a class each, and a caller handling "any of these" would need `catchTags` with all of them. A caller does not usually want to
|
|
10
|
+
* decide between the causes separately. They are all "this XML input was not acceptable", and the useful split is coarse: a bad _configuration_
|
|
11
|
+
* versus a bad _document_. So there is one error, and `reason` narrows to the specific cause. Recovery is a `catchReason` away:
|
|
12
12
|
*
|
|
13
13
|
* @example
|
|
14
14
|
* ```typescript
|
|
15
15
|
* import { Effect } from 'effect';
|
|
16
|
-
* import { EntityDecoder, XmlError } from '@endevops/effect-xml
|
|
16
|
+
* import { EntityDecoder, XmlError } from '@endevops/effect-codec-xml';
|
|
17
17
|
*
|
|
18
18
|
* const limited = new EntityDecoder({ limit: { maxTotalExpansions: 2 } }).decode('&&&').pipe(
|
|
19
19
|
* Effect.catchReason('XmlError', 'ExpansionLimitExceeded', reason => Effect.succeed(`gave up after ${reason.actual}`)),
|
|
20
20
|
* );
|
|
21
21
|
* ```;
|
|
22
22
|
*
|
|
23
|
-
* The message is kept alongside `reason` and is part of the schema, because the text
|
|
24
|
-
* `[EntityReplacer]` prefix in particular is documented as something callers match on, so
|
|
25
|
-
* rather than
|
|
23
|
+
* The message is kept alongside `reason` and is part of the schema, because callers depend on the text. The
|
|
24
|
+
* `[EntityReplacer]` prefix in particular is documented as something callers match on, so the codec reproduces it exactly
|
|
25
|
+
* rather than rewording it.
|
|
26
26
|
*/
|
|
27
27
|
|
|
28
28
|
import { Schema } from 'effect';
|
|
29
29
|
|
|
30
30
|
/**
|
|
31
31
|
* @description The specific cause of an {@link XmlError}, as a tagged union. The `_tag` on each member is the discriminant `Effect.catchReason` matches on, and
|
|
32
|
-
* the payload is what a handler needs in order to decide or to report. Every member is a case the decoder or the name validators actually raise
|
|
33
|
-
*
|
|
32
|
+
* the payload is what a handler needs in order to decide or to report. Every member is a case the decoder or the name validators actually raise.
|
|
33
|
+
* There is no catch-all member, so an exhaustive `match` stays exhaustive as causes are added.
|
|
34
34
|
*/
|
|
35
35
|
export const XmlErrorReason = Schema.TaggedUnion({
|
|
36
36
|
/**
|
|
37
37
|
* @description A required argument was `null` or another non-value where the package requires a real one. Raised by the factories that take caller-supplied
|
|
38
|
-
* input
|
|
39
|
-
*
|
|
40
|
-
*
|
|
38
|
+
* input, {@link EntityDecoder.make} among them, when they are handed `null` for an options object that has no meaningful default. It is a distinct
|
|
39
|
+
* case from the rest because the argument is not _wrong_, it is _absent_, and a caller who wrote `make(null)` meant something the type system does
|
|
40
|
+
* not allow. A decoder with every default is `make({})`, and saying so here is more useful than silently producing one.
|
|
41
41
|
*/
|
|
42
42
|
MissingOptions: {
|
|
43
43
|
/**
|
|
@@ -48,7 +48,7 @@ export const XmlErrorReason = Schema.TaggedUnion({
|
|
|
48
48
|
|
|
49
49
|
/**
|
|
50
50
|
* @description A name was checked against one of the five XML name productions and given a different one. Unreachable from TypeScript, where `Production` is a
|
|
51
|
-
* closed union
|
|
51
|
+
* closed union. It is the guard for untyped JavaScript callers and for values that crossed a boundary as `unknown`.
|
|
52
52
|
*/
|
|
53
53
|
InvalidProduction: {
|
|
54
54
|
production: Schema.String,
|
|
@@ -78,7 +78,7 @@ export const XmlErrorReason = Schema.TaggedUnion({
|
|
|
78
78
|
*/
|
|
79
79
|
EntityRejected: {
|
|
80
80
|
/**
|
|
81
|
-
* @description Which registration was in progress.
|
|
81
|
+
* @description Which registration was in progress. The runtime injects both, so they share a tier for limit accounting.
|
|
82
82
|
*/
|
|
83
83
|
context: Schema.Literals(['external', 'input']),
|
|
84
84
|
/**
|
|
@@ -93,7 +93,7 @@ export const XmlErrorReason = Schema.TaggedUnion({
|
|
|
93
93
|
*/
|
|
94
94
|
ExpansionLimitExceeded: {
|
|
95
95
|
/**
|
|
96
|
-
* @description The count that tripped the limit.
|
|
96
|
+
* @description The count that tripped the limit. The counter is not reset on failure, so this is the real over-limit total rather than the ceiling.
|
|
97
97
|
*/
|
|
98
98
|
actual: Schema.Number,
|
|
99
99
|
/**
|
|
@@ -139,14 +139,14 @@ export const XmlErrorReason = Schema.TaggedUnion({
|
|
|
139
139
|
export type XmlErrorReason = typeof XmlErrorReason.Type;
|
|
140
140
|
|
|
141
141
|
/**
|
|
142
|
-
* @description Every failure this package can report, in the `E` channel of the effects that can fail. Carries both a `reason
|
|
142
|
+
* @description Every failure this package can report, in the `E` channel of the effects that can fail. Carries both a `reason`, the typed and matchable cause, and
|
|
143
143
|
* a `message`, which is the human-readable form the package has always produced. Both are part of the schema, so an error survives a round-trip
|
|
144
144
|
* through a serialised boundary without losing either.
|
|
145
145
|
*
|
|
146
146
|
* @example
|
|
147
147
|
* ```typescript
|
|
148
148
|
* import { Effect } from 'effect';
|
|
149
|
-
* import { EntityDecoder } from '@endevops/effect-xml
|
|
149
|
+
* import { EntityDecoder } from '@endevops/effect-codec-xml';
|
|
150
150
|
*
|
|
151
151
|
* const program = Effect.gen(function*() {
|
|
152
152
|
* const decoder = yield* EntityDecoder.make({});
|
|
@@ -162,7 +162,7 @@ export class XmlError extends Schema.TaggedError<XmlError>()('XmlError', {
|
|
|
162
162
|
|
|
163
163
|
/**
|
|
164
164
|
* @description Human-readable description. The `[EntityReplacer]` and `[EntityDecoder]` prefixes are reproduced verbatim from the original throw sites, because
|
|
165
|
-
*
|
|
165
|
+
* anything matching on them depends on them.
|
|
166
166
|
*/
|
|
167
167
|
message: Schema.String,
|
|
168
168
|
}) {}
|