@mlightcad/data-model 1.14.11 → 1.14.13

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (70) hide show
  1. package/dist/data-model.cjs +17 -17
  2. package/lib/base/AcDbDxfPairReader.d.ts +26 -8
  3. package/lib/base/AcDbDxfPairReader.d.ts.map +1 -1
  4. package/lib/base/AcDbDxfPairReader.js +627 -128
  5. package/lib/base/AcDbDxfPairReader.js.map +1 -1
  6. package/lib/converter/worker/AcDbBaseWorker.d.ts +18 -1
  7. package/lib/converter/worker/AcDbBaseWorker.d.ts.map +1 -1
  8. package/lib/converter/worker/AcDbBaseWorker.js +30 -4
  9. package/lib/converter/worker/AcDbBaseWorker.js.map +1 -1
  10. package/lib/converter/worker/index.d.ts +1 -1
  11. package/lib/converter/worker/index.d.ts.map +1 -1
  12. package/lib/converter/worker/index.js +1 -1
  13. package/lib/converter/worker/index.js.map +1 -1
  14. package/lib/database/AcDbDatabase.d.ts +11 -0
  15. package/lib/database/AcDbDatabase.d.ts.map +1 -1
  16. package/lib/database/AcDbDatabase.js +63 -14
  17. package/lib/database/AcDbDatabase.js.map +1 -1
  18. package/lib/database/AcDbOpenDatabaseError.d.ts +3 -7
  19. package/lib/database/AcDbOpenDatabaseError.d.ts.map +1 -1
  20. package/lib/database/AcDbOpenDatabaseError.js +5 -16
  21. package/lib/database/AcDbOpenDatabaseError.js.map +1 -1
  22. package/lib/dxf/AcDbDxfEntityFactory.d.ts.map +1 -1
  23. package/lib/dxf/AcDbDxfEntityFactory.js +6 -14
  24. package/lib/dxf/AcDbDxfEntityFactory.js.map +1 -1
  25. package/lib/entity/AcDbArc.d.ts +9 -2
  26. package/lib/entity/AcDbArc.d.ts.map +1 -1
  27. package/lib/entity/AcDbArc.js +42 -42
  28. package/lib/entity/AcDbArc.js.map +1 -1
  29. package/lib/entity/AcDbCircle.d.ts +9 -2
  30. package/lib/entity/AcDbCircle.d.ts.map +1 -1
  31. package/lib/entity/AcDbCircle.js +37 -37
  32. package/lib/entity/AcDbCircle.js.map +1 -1
  33. package/lib/entity/AcDbEllipse.d.ts +9 -2
  34. package/lib/entity/AcDbEllipse.d.ts.map +1 -1
  35. package/lib/entity/AcDbEllipse.js +29 -41
  36. package/lib/entity/AcDbEllipse.js.map +1 -1
  37. package/lib/entity/AcDbEntity.d.ts +27 -2
  38. package/lib/entity/AcDbEntity.d.ts.map +1 -1
  39. package/lib/entity/AcDbEntity.js +46 -14
  40. package/lib/entity/AcDbEntity.js.map +1 -1
  41. package/lib/entity/AcDbLeader.d.ts.map +1 -1
  42. package/lib/entity/AcDbLeader.js +1 -0
  43. package/lib/entity/AcDbLeader.js.map +1 -1
  44. package/lib/entity/AcDbLine.d.ts +20 -3
  45. package/lib/entity/AcDbLine.d.ts.map +1 -1
  46. package/lib/entity/AcDbLine.js +77 -37
  47. package/lib/entity/AcDbLine.js.map +1 -1
  48. package/lib/entity/AcDbMLeader.d.ts.map +1 -1
  49. package/lib/entity/AcDbMLeader.js +2 -1
  50. package/lib/entity/AcDbMLeader.js.map +1 -1
  51. package/lib/entity/AcDbMText.d.ts.map +1 -1
  52. package/lib/entity/AcDbMText.js +3 -2
  53. package/lib/entity/AcDbMText.js.map +1 -1
  54. package/lib/entity/AcDbPolyline.d.ts +1 -1
  55. package/lib/entity/AcDbPolyline.d.ts.map +1 -1
  56. package/lib/entity/AcDbPolyline.js +54 -5
  57. package/lib/entity/AcDbPolyline.js.map +1 -1
  58. package/lib/entity/AcDbSpline.d.ts +9 -2
  59. package/lib/entity/AcDbSpline.d.ts.map +1 -1
  60. package/lib/entity/AcDbSpline.js +32 -0
  61. package/lib/entity/AcDbSpline.js.map +1 -1
  62. package/lib/entity/AcDbTextExtentsHelpers.d.ts +10 -3
  63. package/lib/entity/AcDbTextExtentsHelpers.d.ts.map +1 -1
  64. package/lib/entity/AcDbTextExtentsHelpers.js +15 -7
  65. package/lib/entity/AcDbTextExtentsHelpers.js.map +1 -1
  66. package/lib/misc/AcDbRenderingCache.d.ts +35 -0
  67. package/lib/misc/AcDbRenderingCache.d.ts.map +1 -1
  68. package/lib/misc/AcDbRenderingCache.js +116 -0
  69. package/lib/misc/AcDbRenderingCache.js.map +1 -1
  70. package/package.json +4 -4
@@ -11,6 +11,7 @@ var BINARY_DXF_MAGIC = (function () {
11
11
  bytes[21] = 0x00;
12
12
  return bytes;
13
13
  })();
14
+ var UTF8_DECODER = new TextDecoder('utf-8');
14
15
  var HEX_NIBBLE = (function () {
15
16
  var t = new Int8Array(128);
16
17
  for (var i = 0; i < 10; i++)
@@ -30,6 +31,520 @@ export function acdbIsBinaryDxf(data) {
30
31
  }
31
32
  return true;
32
33
  }
34
+ /** ASCII whitespace used around DXF numeric fields (String.trim's ASCII set). */
35
+ function acdbIsAsciiWhitespace(c) {
36
+ return c === 0x20 || (c >= 0x09 && c <= 0x0d);
37
+ }
38
+ /** True for every character `String.prototype.trim` strips. */
39
+ function acdbIsTrimWhitespace(c) {
40
+ return (acdbIsAsciiWhitespace(c) ||
41
+ c === 0xa0 ||
42
+ c === 0x1680 ||
43
+ c === 0x2028 ||
44
+ c === 0x2029 ||
45
+ c === 0x202f ||
46
+ c === 0x205f ||
47
+ c === 0x3000 ||
48
+ c === 0xfeff ||
49
+ (c >= 0x2000 && c <= 0x200a));
50
+ }
51
+ /**
52
+ * Decodes a pure-ASCII byte span without constructing a `TextDecoder` call or
53
+ * an intermediate `Uint8Array` copy. DXF keywords, handles and numeric text
54
+ * are overwhelmingly ASCII.
55
+ */
56
+ function acdbDecodeAsciiSpan(bytes, start, end) {
57
+ var length = end - start;
58
+ if (length <= 0)
59
+ return '';
60
+ if (length <= 16) {
61
+ var text_1 = '';
62
+ for (var i = start; i < end; i++)
63
+ text_1 += String.fromCharCode(bytes[i]);
64
+ return text_1;
65
+ }
66
+ var CHUNK = 4096;
67
+ var text = '';
68
+ for (var pos = start; pos < end; pos += CHUNK) {
69
+ var chunkEnd = Math.min(pos + CHUNK, end);
70
+ text += String.fromCharCode.apply(null, bytes.subarray(pos, chunkEnd));
71
+ }
72
+ return text;
73
+ }
74
+ /**
75
+ * Decodes a UTF-8 byte span, taking the ASCII fast path when possible.
76
+ *
77
+ * `nonAscii` must be true exactly when the span holds a byte `>= 0x80`. It is
78
+ * produced by the line scanner, so this function never re-scans the span.
79
+ */
80
+ function acdbDecodeUtf8Span(bytes, start, end, nonAscii) {
81
+ if (!nonAscii) {
82
+ return acdbDecodeAsciiSpan(bytes, start, end);
83
+ }
84
+ return UTF8_DECODER.decode(bytes.subarray(start, end));
85
+ }
86
+ /**
87
+ * Parses an integer group code straight from the bytes of a code line,
88
+ * without allocating the line string or a trimmed copy.
89
+ *
90
+ * Returns NaN for blank or malformed code lines. DXF group code lines are
91
+ * pure ASCII; the rare non-ASCII line falls back to `Number` + `isFinite`.
92
+ */
93
+ function acdbReadDxfCodeFromBytes(bytes, start, end, nonAscii) {
94
+ if (nonAscii) {
95
+ var n = Number(acdbDecodeUtf8Span(bytes, start, end, true).trim());
96
+ return Number.isFinite(n) ? n : NaN;
97
+ }
98
+ var i = start;
99
+ while (i < end && acdbIsAsciiWhitespace(bytes[i]))
100
+ i++;
101
+ if (i >= end)
102
+ return NaN;
103
+ var sign = 1;
104
+ var c0 = bytes[i];
105
+ if (c0 === 0x2d) {
106
+ sign = -1;
107
+ i++;
108
+ }
109
+ else if (c0 === 0x2b) {
110
+ i++;
111
+ }
112
+ var value = 0;
113
+ var digits = 0;
114
+ while (i < end) {
115
+ var c = bytes[i];
116
+ if (c >= 0x30 && c <= 0x39) {
117
+ value = value * 10 + (c - 0x30);
118
+ digits++;
119
+ i++;
120
+ }
121
+ else {
122
+ break;
123
+ }
124
+ }
125
+ if (digits === 0)
126
+ return NaN;
127
+ while (i < end) {
128
+ if (!acdbIsAsciiWhitespace(bytes[i]))
129
+ return NaN;
130
+ i++;
131
+ }
132
+ return sign * value;
133
+ }
134
+ /**
135
+ * Largest integer mantissa for which one more accumulation step is still exact
136
+ * as a double: below (2^53 - 9) / 10 we always stay representable.
137
+ */
138
+ var MAX_EXACT_DOUBLE_MANTISSA = (Number.MAX_SAFE_INTEGER - 9) / 10;
139
+ /**
140
+ * Fast path for double value lines: parses `[+-]?digits[.digits][eE[+-]digits]`
141
+ * straight from the byte span, without slicing the line or calling `Number()`.
142
+ * Returns undefined outside its exact domain (caller falls back).
143
+ */
144
+ function acdbParseDoubleSpan(bytes, start, end) {
145
+ var i = start;
146
+ while (i < end && acdbIsAsciiWhitespace(bytes[i]))
147
+ i++;
148
+ if (i >= end)
149
+ return 0;
150
+ var sign = 1;
151
+ var c0 = bytes[i];
152
+ if (c0 === 0x2d) {
153
+ sign = -1;
154
+ i++;
155
+ }
156
+ else if (c0 === 0x2b) {
157
+ i++;
158
+ }
159
+ var mantissa = 0;
160
+ var exp10 = 0;
161
+ var anyDigit = false;
162
+ var tooLong = false;
163
+ while (i < end) {
164
+ var c = bytes[i];
165
+ if (c < 0x30 || c > 0x39)
166
+ break;
167
+ anyDigit = true;
168
+ i++;
169
+ if (mantissa === 0 && c === 0x30)
170
+ continue;
171
+ if (mantissa > MAX_EXACT_DOUBLE_MANTISSA) {
172
+ tooLong = true;
173
+ continue;
174
+ }
175
+ mantissa = mantissa * 10 + (c - 0x30);
176
+ }
177
+ if (i < end && bytes[i] === 0x2e) {
178
+ i++;
179
+ while (i < end) {
180
+ var c = bytes[i];
181
+ if (c < 0x30 || c > 0x39)
182
+ break;
183
+ anyDigit = true;
184
+ i++;
185
+ if (mantissa === 0 && c === 0x30) {
186
+ exp10--;
187
+ continue;
188
+ }
189
+ if (mantissa > MAX_EXACT_DOUBLE_MANTISSA) {
190
+ tooLong = true;
191
+ continue;
192
+ }
193
+ mantissa = mantissa * 10 + (c - 0x30);
194
+ exp10--;
195
+ }
196
+ }
197
+ if (i < end && (bytes[i] === 0x65 || bytes[i] === 0x45)) {
198
+ i++;
199
+ var expSign = 1;
200
+ var sc = bytes[i];
201
+ if (sc === 0x2d) {
202
+ expSign = -1;
203
+ i++;
204
+ }
205
+ else if (sc === 0x2b) {
206
+ i++;
207
+ }
208
+ var expVal = 0;
209
+ var expDigits = 0;
210
+ while (i < end) {
211
+ var c = bytes[i];
212
+ if (c < 0x30 || c > 0x39)
213
+ break;
214
+ i++;
215
+ expDigits++;
216
+ if (expVal <= 10000)
217
+ expVal = expVal * 10 + (c - 0x30);
218
+ }
219
+ if (expDigits === 0)
220
+ return undefined;
221
+ exp10 += expSign * expVal;
222
+ }
223
+ while (i < end) {
224
+ if (!acdbIsAsciiWhitespace(bytes[i]))
225
+ return undefined;
226
+ i++;
227
+ }
228
+ if (!anyDigit)
229
+ return 0;
230
+ if (tooLong)
231
+ return undefined;
232
+ if (mantissa === 0)
233
+ return sign * mantissa;
234
+ if (exp10 > 22 || exp10 < -22)
235
+ return undefined;
236
+ var scale = 1;
237
+ var k = exp10 < 0 ? -exp10 : exp10;
238
+ for (var n = 0; n < k; n++)
239
+ scale *= 10;
240
+ return sign * (exp10 < 0 ? mantissa / scale : mantissa * scale);
241
+ }
242
+ /**
243
+ * Fast path for int value lines with parseInt semantics: skips leading
244
+ * whitespace, takes the longest digit run, ignores the rest. Returns
245
+ * undefined when non-ASCII whitespace may precede the digits or for digit
246
+ * runs longer than 15 (so the caller falls back).
247
+ */
248
+ function acdbParseIntSpan(bytes, start, end) {
249
+ var i = start;
250
+ for (;;) {
251
+ if (i >= end)
252
+ break;
253
+ var c = bytes[i];
254
+ if (c >= 0x80)
255
+ return undefined;
256
+ if (!acdbIsAsciiWhitespace(c))
257
+ break;
258
+ i++;
259
+ }
260
+ var sign = 1;
261
+ var c0 = bytes[i];
262
+ if (c0 === 0x2d) {
263
+ sign = -1;
264
+ i++;
265
+ }
266
+ else if (c0 === 0x2b) {
267
+ i++;
268
+ }
269
+ if (i < end && bytes[i] >= 0x80)
270
+ return undefined;
271
+ var value = 0;
272
+ var digits = 0;
273
+ var anyDigit = false;
274
+ while (i < end) {
275
+ var c = bytes[i];
276
+ if (c < 0x30 || c > 0x39)
277
+ break;
278
+ i++;
279
+ anyDigit = true;
280
+ if (value === 0 && c === 0x30)
281
+ continue;
282
+ if (digits >= 15)
283
+ return undefined;
284
+ value = value * 10 + (c - 0x30);
285
+ digits++;
286
+ }
287
+ return anyDigit ? sign * value : 0;
288
+ }
289
+ /**
290
+ * Fast path for long value lines with Number semantics (whole line must be
291
+ * numeric, unlike parseInt). Returns the integer when it has at most 15
292
+ * significant digits; otherwise undefined (caller falls back to Number/BigInt).
293
+ */
294
+ function acdbParseLongSpan(bytes, start, end) {
295
+ var i = start;
296
+ while (i < end && acdbIsAsciiWhitespace(bytes[i]))
297
+ i++;
298
+ var sign = 1;
299
+ var c0 = bytes[i];
300
+ if (c0 === 0x2d) {
301
+ sign = -1;
302
+ i++;
303
+ }
304
+ else if (c0 === 0x2b) {
305
+ i++;
306
+ }
307
+ var value = 0;
308
+ var digits = 0;
309
+ var anyDigit = false;
310
+ while (i < end) {
311
+ var c = bytes[i];
312
+ if (c < 0x30 || c > 0x39)
313
+ break;
314
+ i++;
315
+ anyDigit = true;
316
+ if (value === 0 && c === 0x30)
317
+ continue;
318
+ if (digits >= 15)
319
+ return undefined;
320
+ value = value * 10 + (c - 0x30);
321
+ digits++;
322
+ }
323
+ while (i < end) {
324
+ if (!acdbIsAsciiWhitespace(bytes[i]))
325
+ return undefined;
326
+ i++;
327
+ }
328
+ return anyDigit ? sign * value : 0;
329
+ }
330
+ /**
331
+ * Equivalent to `trimmed !== '' && trimmed !== '0'` without allocating.
332
+ * nonAscii comes from the line scanner; non-ASCII spans bail out so the caller
333
+ * can use the exact String.prototype.trim semantics.
334
+ */
335
+ function acdbDxfRawBoolIsTrue(bytes, start, end, nonAscii) {
336
+ if (nonAscii)
337
+ return undefined;
338
+ while (start < end && acdbIsAsciiWhitespace(bytes[start]))
339
+ start++;
340
+ while (end > start && acdbIsAsciiWhitespace(bytes[end - 1]))
341
+ end--;
342
+ if (start >= end)
343
+ return false;
344
+ return !(end - start === 1 && bytes[start] === 0x30);
345
+ }
346
+ /**
347
+ * Decodes a hex pair value straight from a character span. Used only as a rare
348
+ * fallback when a binary value line contains non-ASCII bytes.
349
+ */
350
+ function acdbDecodeHexBinaryText(text, start, end) {
351
+ while (start < end && acdbIsTrimWhitespace(text.charCodeAt(start)))
352
+ start++;
353
+ while (end > start && acdbIsTrimWhitespace(text.charCodeAt(end - 1)))
354
+ end--;
355
+ var byteLength = (end - start) >>> 1;
356
+ var bytes = new Uint8Array(byteLength);
357
+ for (var j = 0; j < byteLength; j++) {
358
+ var hi = HEX_NIBBLE[text.charCodeAt(start + j * 2) & 0x7f];
359
+ var lo = HEX_NIBBLE[text.charCodeAt(start + j * 2 + 1) & 0x7f];
360
+ bytes[j] = (hi << 4) | lo;
361
+ }
362
+ return bytes;
363
+ }
364
+ /**
365
+ * Decodes a code-310 hex line from bytes, trimming ASCII whitespace.
366
+ */
367
+ function acdbDecodeHexBinarySpan(bytes, start, end, nonAscii) {
368
+ while (start < end && acdbIsAsciiWhitespace(bytes[start]))
369
+ start++;
370
+ while (end > start && acdbIsAsciiWhitespace(bytes[end - 1]))
371
+ end--;
372
+ if (nonAscii) {
373
+ var text = acdbDecodeUtf8Span(bytes, start, end, true);
374
+ return acdbDecodeHexBinaryText(text, 0, text.length);
375
+ }
376
+ var byteLength = (end - start) >>> 1;
377
+ var out = new Uint8Array(byteLength);
378
+ for (var j = 0; j < byteLength; j++) {
379
+ var hi = HEX_NIBBLE[bytes[start + j * 2] & 0x7f];
380
+ var lo = HEX_NIBBLE[bytes[start + j * 2 + 1] & 0x7f];
381
+ out[j] = (hi << 4) | lo;
382
+ }
383
+ return out;
384
+ }
385
+ /**
386
+ * Parses one ASCII value line.
387
+ *
388
+ * nonAscii is the flag produced by the line scanner for this exact span; it
389
+ * is forwarded to every decode helper so no helper has to re-scan the bytes.
390
+ */
391
+ function parseAsciiValueSpan(code, bytes, start, end, nonAscii) {
392
+ var type = acdbDxfValueType(code);
393
+ if (type === 'comment')
394
+ return null;
395
+ switch (type) {
396
+ case 'string':
397
+ return {
398
+ code: code,
399
+ type: type,
400
+ value: acdbDecodeUtf8Span(bytes, start, end, nonAscii)
401
+ };
402
+ case 'int': {
403
+ var fast = acdbParseIntSpan(bytes, start, end);
404
+ var n = fast === undefined
405
+ ? parseInt(acdbDecodeUtf8Span(bytes, start, end, nonAscii), 10)
406
+ : fast;
407
+ return { code: code, type: type, value: Number.isFinite(n) ? n : 0 };
408
+ }
409
+ case 'long': {
410
+ var fast = acdbParseLongSpan(bytes, start, end);
411
+ if (fast !== undefined)
412
+ return { code: code, type: type, value: fast };
413
+ var raw = acdbDecodeUtf8Span(bytes, start, end, nonAscii);
414
+ var n = Number(raw);
415
+ if (Number.isSafeInteger(n))
416
+ return { code: code, type: type, value: n };
417
+ try {
418
+ return { code: code, type: type, value: BigInt(raw.trim()) };
419
+ }
420
+ catch (_a) {
421
+ return { code: code, type: type, value: 0 };
422
+ }
423
+ }
424
+ case 'double': {
425
+ var fast = acdbParseDoubleSpan(bytes, start, end);
426
+ var n = fast === undefined
427
+ ? Number(acdbDecodeUtf8Span(bytes, start, end, nonAscii))
428
+ : fast;
429
+ return { code: code, type: type, value: Number.isFinite(n) ? n : 0 };
430
+ }
431
+ case 'bool': {
432
+ var fast = acdbDxfRawBoolIsTrue(bytes, start, end, nonAscii);
433
+ if (fast !== undefined)
434
+ return { code: code, type: type, value: fast };
435
+ var trimmed = acdbDecodeUtf8Span(bytes, start, end, nonAscii).trim();
436
+ return { code: code, type: type, value: trimmed !== '' && trimmed !== '0' };
437
+ }
438
+ case 'handle': {
439
+ var value = acdbDecodeUtf8Span(bytes, start, end, nonAscii);
440
+ if (value.length === 0)
441
+ return { code: code, type: type, value: value };
442
+ var first = value.charCodeAt(0);
443
+ var last = value.charCodeAt(value.length - 1);
444
+ if (!acdbIsTrimWhitespace(first) && !acdbIsTrimWhitespace(last)) {
445
+ return { code: code, type: type, value: value };
446
+ }
447
+ return { code: code, type: type, value: value.trim() };
448
+ }
449
+ case 'binary':
450
+ return {
451
+ code: code,
452
+ type: type,
453
+ value: acdbDecodeHexBinarySpan(bytes, start, end, nonAscii)
454
+ };
455
+ default:
456
+ return null;
457
+ }
458
+ }
459
+ /**
460
+ * ASCII/UTF-8 pair reader over raw bytes.
461
+ *
462
+ * Line breaks are single bytes and UTF-8 continuation bytes never equal
463
+ * 0x0A/0x0D, so non-ASCII text values never straddle line boundaries.
464
+ */
465
+ function acdbMakeUtf8DxfPairReader(bytes) {
466
+ var pos = bytes.length >= 3 &&
467
+ bytes[0] === 0xef &&
468
+ bytes[1] === 0xbb &&
469
+ bytes[2] === 0xbf
470
+ ? 3
471
+ : 0;
472
+ var lineNumber = 1;
473
+ var lookahead;
474
+ var lookaheadValid = false;
475
+ /**
476
+ * Advances past one line and returns its content byte range.
477
+ *
478
+ * The scan also ORs every content byte, so nonAscii ("this line holds a
479
+ * byte >= 0x80") comes for free and the value/code parsers never have to walk
480
+ * the line a second time.
481
+ */
482
+ function readLineSpan() {
483
+ if (pos >= bytes.length)
484
+ return undefined;
485
+ var start = pos;
486
+ var contentEnd = pos;
487
+ var flags = 0;
488
+ while (contentEnd < bytes.length) {
489
+ var c = bytes[contentEnd];
490
+ if (c === 0x0a || c === 0x0d)
491
+ break;
492
+ flags |= c;
493
+ contentEnd++;
494
+ }
495
+ var end = contentEnd;
496
+ if (end < bytes.length && bytes[end] === 0x0d)
497
+ end++;
498
+ if (end < bytes.length && bytes[end] === 0x0a)
499
+ end++;
500
+ pos = end;
501
+ lineNumber++;
502
+ return { start: start, end: contentEnd, nonAscii: flags >= 0x80 };
503
+ }
504
+ function readRaw() {
505
+ for (;;) {
506
+ var codeSpan = readLineSpan();
507
+ if (codeSpan === undefined)
508
+ return undefined;
509
+ var code = acdbReadDxfCodeFromBytes(bytes, codeSpan.start, codeSpan.end, codeSpan.nonAscii);
510
+ if (Number.isNaN(code))
511
+ continue;
512
+ if (code === 999) {
513
+ if (readLineSpan() === undefined)
514
+ return undefined;
515
+ continue;
516
+ }
517
+ var valueSpan = readLineSpan();
518
+ if (valueSpan === undefined)
519
+ return undefined;
520
+ var pair = parseAsciiValueSpan(code, bytes, valueSpan.start, valueSpan.end, valueSpan.nonAscii);
521
+ if (pair)
522
+ return pair;
523
+ }
524
+ }
525
+ return {
526
+ kind: 'ascii',
527
+ next: function () {
528
+ if (lookaheadValid) {
529
+ var pair = lookahead;
530
+ lookahead = undefined;
531
+ lookaheadValid = false;
532
+ return pair;
533
+ }
534
+ return readRaw();
535
+ },
536
+ peek: function () {
537
+ if (!lookaheadValid) {
538
+ lookahead = readRaw();
539
+ lookaheadValid = true;
540
+ }
541
+ return lookahead;
542
+ },
543
+ position: function () {
544
+ return { line: lineNumber, byteOffset: pos };
545
+ }
546
+ };
547
+ }
33
548
  /**
34
549
  * Peek `$ACADVER` / `$DWGCODEPAGE` from the HEADER section without decoding
35
550
  * the whole file. Uses 64 KiB UTF-8 chunks (same strategy as AcDbDxfParser).
@@ -210,7 +725,6 @@ function isUtf8Encoding(encoding) {
210
725
  * decode calls rather than one per line, small enough that windows still holding
211
726
  * a retained value slice do not pin much memory.
212
727
  */
213
- var UTF8_DECODE_WINDOW_BYTES = 64 * 1024;
214
728
  /**
215
729
  * ASCII pair reader that decodes UTF-8 bytes one line-aligned window at a time,
216
730
  * instead of allocating a full-file decoded string (peak memory ≈ input bytes
@@ -220,112 +734,7 @@ var UTF8_DECODE_WINDOW_BYTES = 64 * 1024;
220
734
  * after a full `TextDecoder` pass.
221
735
  */
222
736
  export function acdbMakeUtf8AsciiDxfPairReader(bytes) {
223
- // Skip UTF-8 BOM when present.
224
- var start = bytes.length >= 3 &&
225
- bytes[0] === 0xef &&
226
- bytes[1] === 0xbb &&
227
- bytes[2] === 0xbf
228
- ? 3
229
- : 0;
230
- var lineNumber = 1;
231
- var lookahead;
232
- var lookaheadValid = false;
233
- var decoder = new TextDecoder('utf-8');
234
- // Decoded window covering bytes [windowStart, windowEnd), scanned by a
235
- // character cursor. Windows end just past a line break, and a break byte can
236
- // never appear inside a multi-byte UTF-8 sequence, so each window decodes
237
- // standalone and no line ever straddles two windows.
238
- var windowStart = start;
239
- var windowEnd = start;
240
- var text = '';
241
- var textPos = 0;
242
- /** Decodes the next window. Returns `false` once the input is exhausted. */
243
- function advanceWindow() {
244
- if (windowEnd >= bytes.length)
245
- return false;
246
- windowStart = windowEnd;
247
- var end = Math.min(windowStart + UTF8_DECODE_WINDOW_BYTES, bytes.length);
248
- while (end < bytes.length && bytes[end] !== 10 && bytes[end] !== 13)
249
- end++;
250
- if (end < bytes.length && bytes[end] === 13)
251
- end++;
252
- if (end < bytes.length && bytes[end] === 10)
253
- end++;
254
- windowEnd = end;
255
- text = decoder.decode(bytes.subarray(windowStart, windowEnd));
256
- textPos = 0;
257
- return true;
258
- }
259
- function readLine() {
260
- while (textPos >= text.length) {
261
- if (!advanceWindow())
262
- return undefined;
263
- }
264
- var end = textPos;
265
- while (end < text.length) {
266
- var c = text.charCodeAt(end);
267
- if (c === 10 || c === 13)
268
- break;
269
- end++;
270
- }
271
- var line = text.slice(textPos, end);
272
- if (end < text.length && text.charCodeAt(end) === 13)
273
- end++;
274
- if (end < text.length && text.charCodeAt(end) === 10)
275
- end++;
276
- textPos = end;
277
- lineNumber++;
278
- return line;
279
- }
280
- function readRaw() {
281
- for (;;) {
282
- var codeRaw = readLine();
283
- if (codeRaw === undefined)
284
- return undefined;
285
- var codeTrimmed = codeRaw.trim();
286
- if (codeTrimmed === '')
287
- continue;
288
- var valueRaw = readLine();
289
- if (valueRaw === undefined)
290
- return undefined;
291
- var code = Number(codeTrimmed);
292
- if (!Number.isFinite(code))
293
- continue;
294
- if (code === 999)
295
- continue;
296
- var pair = parseAsciiValue(code, valueRaw);
297
- if (pair)
298
- return pair;
299
- }
300
- }
301
- return {
302
- kind: 'ascii',
303
- next: function () {
304
- if (lookaheadValid) {
305
- var p = lookahead;
306
- lookahead = undefined;
307
- lookaheadValid = false;
308
- return p;
309
- }
310
- return readRaw();
311
- },
312
- peek: function () {
313
- if (!lookaheadValid) {
314
- lookahead = readRaw();
315
- lookaheadValid = true;
316
- }
317
- return lookahead;
318
- },
319
- position: function () {
320
- // Interpolated inside the current window: callers use this only to report
321
- // parse progress, so window-level precision is enough.
322
- var span = windowEnd - windowStart;
323
- var byteOffset = span > 0 && text.length > 0
324
- ? windowStart + Math.round((textPos / text.length) * span)
325
- : windowEnd;
326
- return { line: lineNumber, byteOffset: byteOffset };
327
- }
328
- };
737
+ return acdbMakeUtf8DxfPairReader(bytes);
329
738
  }
330
739
  function safeBigIntToNumber(v) {
331
740
  var max = BigInt(Number.MAX_SAFE_INTEGER);
@@ -520,12 +929,85 @@ export function acdbMakeBinaryDxfPairReader(data, options) {
520
929
  }
521
930
  };
522
931
  }
932
+ /**
933
+ * Strict UTF-8 validation (rejects stray continuation bytes, overlong forms,
934
+ * surrogate halves and anything beyond U+10FFFF). Used to tell genuine legacy
935
+ * ANSI bytes apart from UTF-8 content hiding behind a stale pre-2007
936
+ * `$DWGCODEPAGE`: real ANSI multi-byte text essentially never forms valid
937
+ * UTF-8 sequences, while UTF-8 content always validates.
938
+ */
939
+ function acdbIsValidUtf8(bytes) {
940
+ var n = bytes.length;
941
+ var i = 0;
942
+ while (i < n) {
943
+ var b0 = bytes[i];
944
+ if (b0 < 0x80) {
945
+ i++;
946
+ continue;
947
+ }
948
+ if (b0 < 0xc2)
949
+ return false; // stray continuation byte or overlong 2-byte lead
950
+ if (b0 < 0xe0) {
951
+ // 2-byte sequence
952
+ if (i + 1 >= n || (bytes[i + 1] & 0xc0) !== 0x80)
953
+ return false;
954
+ i += 2;
955
+ continue;
956
+ }
957
+ if (b0 < 0xf0) {
958
+ // 3-byte sequence
959
+ if (i + 2 >= n)
960
+ return false;
961
+ var b1 = bytes[i + 1];
962
+ if ((b1 & 0xc0) !== 0x80)
963
+ return false;
964
+ if (b0 === 0xe0 && b1 < 0xa0)
965
+ return false; // overlong encoding
966
+ if (b0 === 0xed && b1 >= 0xa0)
967
+ return false; // surrogate half
968
+ if ((bytes[i + 2] & 0xc0) !== 0x80)
969
+ return false;
970
+ i += 3;
971
+ continue;
972
+ }
973
+ if (b0 < 0xf5) {
974
+ // 4-byte sequence
975
+ if (i + 3 >= n)
976
+ return false;
977
+ var b1 = bytes[i + 1];
978
+ if ((b1 & 0xc0) !== 0x80)
979
+ return false;
980
+ if (b0 === 0xf0 && b1 < 0x90)
981
+ return false; // overlong encoding
982
+ if (b0 === 0xf4 && b1 >= 0x90)
983
+ return false; // beyond U+10FFFF
984
+ if ((bytes[i + 2] & 0xc0) !== 0x80)
985
+ return false;
986
+ if ((bytes[i + 3] & 0xc0) !== 0x80)
987
+ return false;
988
+ i += 4;
989
+ continue;
990
+ }
991
+ return false; // 0xf5..0xff are never valid lead bytes
992
+ }
993
+ return true;
994
+ }
523
995
  /**
524
996
  * Create a pair reader from DXF bytes (ASCII or binary).
525
997
  *
526
- * ASCII path: peek HEADER for version/codepage when needed. UTF-8 drawings
527
- * decode one line-aligned window at a time (no full-file string). Legacy code
528
- * pages still decode once via `TextDecoder`, then scan with a character cursor.
998
+ * ASCII path encoding strategy (hybrid of byte-trust and header sniffing):
999
+ *
1000
+ * 1. An explicit `options.encoding` always wins — the caller asserts the
1001
+ * encoding, no sniffing contradicts it.
1002
+ * 2. Otherwise the bytes are strictly validated as UTF-8. Valid UTF-8
1003
+ * (including pure ASCII) decodes correctly regardless of any stale
1004
+ * `$DWGCODEPAGE`, so the header is never consulted — modern files skip
1005
+ * the header pre-scan entirely and stream straight from bytes.
1006
+ * 3. Invalid UTF-8 means the file is not a spec-conformant modern DXF. The
1007
+ * HEADER is then peeked for `$ACADVER`/`$DWGCODEPAGE`: pre-2007 drawings
1008
+ * with a declared code page decode through it (matching AutoCAD), while
1009
+ * R2007+ or headerless files fall back to UTF-8 (replacement chars mark
1010
+ * the broken bytes).
529
1011
  */
530
1012
  export function acdbCreateDxfPairReader(data, options) {
531
1013
  if (options === void 0) { options = {}; }
@@ -534,10 +1016,10 @@ export function acdbCreateDxfPairReader(data, options) {
534
1016
  ? acdbNormalizeTextEncoding(options.encoding)
535
1017
  : undefined;
536
1018
  if (acdbIsBinaryDxf(bytes)) {
537
- var encoding_1 = overrideEncoding;
1019
+ var encoding = overrideEncoding;
538
1020
  var legacyR12 = options.legacyR12;
539
- if (encoding_1 == null || legacyR12 == null) {
540
- encoding_1 = encoding_1 !== null && encoding_1 !== void 0 ? encoding_1 : 'utf-8';
1021
+ if (encoding == null || legacyR12 == null) {
1022
+ encoding = encoding !== null && encoding !== void 0 ? encoding : 'utf-8';
541
1023
  if (legacyR12 == null) {
542
1024
  // After the 22-byte magic: R12 uses 1-byte codes (`0,'S'`), modern
543
1025
  // uses 2-byte LE codes (`0,0,'S'`) for the first SECTION marker.
@@ -556,28 +1038,45 @@ export function acdbCreateDxfPairReader(data, options) {
556
1038
  }
557
1039
  }
558
1040
  }
559
- return acdbMakeBinaryDxfPairReader(bytes, { encoding: encoding_1, legacyR12: legacyR12 });
1041
+ return acdbMakeBinaryDxfPairReader(bytes, { encoding: encoding, legacyR12: legacyR12 });
1042
+ }
1043
+ // 1. Explicit override wins: the caller asserts the encoding.
1044
+ if (overrideEncoding) {
1045
+ if (isUtf8Encoding(overrideEncoding)) {
1046
+ return acdbMakeUtf8DxfPairReader(bytes);
1047
+ }
1048
+ var text = new TextDecoder(overrideEncoding).decode(bytes);
1049
+ return acdbMakeAsciiDxfPairReader(text);
560
1050
  }
561
- var buffer = bytes.byteOffset === 0 && bytes.byteLength === bytes.buffer.byteLength
562
- ? bytes.buffer
563
- : bytes.buffer.slice(bytes.byteOffset, bytes.byteOffset + bytes.byteLength);
564
- var encoding = overrideEncoding;
565
- if (encoding == null) {
1051
+ // 2. Trust the bytes first: valid UTF-8 (including pure ASCII) is decoded
1052
+ // correctly no matter what a stale $DWGCODEPAGE claims, so modern files
1053
+ // never pay for a header pre-scan.
1054
+ if (acdbIsValidUtf8(bytes)) {
1055
+ return acdbMakeUtf8DxfPairReader(bytes);
1056
+ }
1057
+ // 3. Invalid UTF-8: fall back to the pre-2007 $DWGCODEPAGE when one is
1058
+ // declared (genuine legacy ANSI content). R2007+ or headerless files
1059
+ // stay on UTF-8, where replacement characters mark the broken bytes.
1060
+ var legacyEncoding = null;
1061
+ try {
1062
+ var buffer = bytes.byteOffset === 0 && bytes.byteLength === bytes.buffer.byteLength
1063
+ ? bytes.buffer
1064
+ : bytes.buffer.slice(bytes.byteOffset, bytes.byteOffset + bytes.byteLength);
566
1065
  var info = acdbPeekDxfHeaderInfo(buffer);
567
- // Pre-2007 drawings may declare a non-UTF-8 `$DWGCODEPAGE`.
568
1066
  if (info.version &&
569
1067
  !info.version.capabilities.supportsUtf8CodePage &&
570
1068
  info.encoding) {
571
- encoding = info.encoding;
572
- }
573
- else {
574
- encoding = 'utf-8';
1069
+ legacyEncoding = info.encoding;
575
1070
  }
576
1071
  }
577
- if (isUtf8Encoding(encoding)) {
578
- return acdbMakeUtf8AsciiDxfPairReader(bytes);
1072
+ catch (_a) {
1073
+ // Unrecognized $ACADVER spelling: treat as version-less and stay UTF-8
1074
+ // instead of failing the whole read.
1075
+ }
1076
+ if (legacyEncoding) {
1077
+ var text = new TextDecoder(legacyEncoding).decode(bytes);
1078
+ return acdbMakeAsciiDxfPairReader(text);
579
1079
  }
580
- var text = new TextDecoder(encoding).decode(bytes);
581
- return acdbMakeAsciiDxfPairReader(text);
1080
+ return acdbMakeUtf8DxfPairReader(bytes);
582
1081
  }
583
1082
  //# sourceMappingURL=AcDbDxfPairReader.js.map