@mlightcad/data-model 1.15.2 → 1.15.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (41) hide show
  1. package/dist/data-model.cjs +17 -17
  2. package/lib/base/AcDbDxfPairReader.d.ts +24 -24
  3. package/lib/base/AcDbDxfPairReader.d.ts.map +1 -1
  4. package/lib/base/AcDbDxfPairReader.js +181 -81
  5. package/lib/base/AcDbDxfPairReader.js.map +1 -1
  6. package/lib/database/AcDbBlockTableRecord.d.ts.map +1 -1
  7. package/lib/database/AcDbBlockTableRecord.js +3 -1
  8. package/lib/database/AcDbBlockTableRecord.js.map +1 -1
  9. package/lib/database/AcDbDatabase.d.ts.map +1 -1
  10. package/lib/database/AcDbDatabase.js +110 -70
  11. package/lib/database/AcDbDatabase.js.map +1 -1
  12. package/lib/database/AcDbSymbolTable.d.ts.map +1 -1
  13. package/lib/database/AcDbSymbolTable.js +4 -1
  14. package/lib/database/AcDbSymbolTable.js.map +1 -1
  15. package/lib/entity/AcDbAttribute.d.ts.map +1 -1
  16. package/lib/entity/AcDbAttribute.js +13 -0
  17. package/lib/entity/AcDbAttribute.js.map +1 -1
  18. package/lib/entity/AcDbAttributeDefinition.d.ts.map +1 -1
  19. package/lib/entity/AcDbAttributeDefinition.js +13 -0
  20. package/lib/entity/AcDbAttributeDefinition.js.map +1 -1
  21. package/lib/entity/AcDbBlockReference.js +1 -1
  22. package/lib/entity/AcDbBlockReference.js.map +1 -1
  23. package/lib/entity/AcDbEntity.d.ts.map +1 -1
  24. package/lib/entity/AcDbEntity.js +13 -2
  25. package/lib/entity/AcDbEntity.js.map +1 -1
  26. package/lib/entity/AcDbMLeader.d.ts.map +1 -1
  27. package/lib/entity/AcDbMLeader.js +18 -4
  28. package/lib/entity/AcDbMLeader.js.map +1 -1
  29. package/lib/entity/AcDbViewport.d.ts.map +1 -1
  30. package/lib/entity/AcDbViewport.js +33 -9
  31. package/lib/entity/AcDbViewport.js.map +1 -1
  32. package/lib/entity/dimension/AcDbAlignedDimension.d.ts.map +1 -1
  33. package/lib/entity/dimension/AcDbAlignedDimension.js +3 -2
  34. package/lib/entity/dimension/AcDbAlignedDimension.js.map +1 -1
  35. package/lib/entity/dimension/AcDbDimension.d.ts.map +1 -1
  36. package/lib/entity/dimension/AcDbDimension.js +27 -3
  37. package/lib/entity/dimension/AcDbDimension.js.map +1 -1
  38. package/lib/object/AcDbDictionary.d.ts.map +1 -1
  39. package/lib/object/AcDbDictionary.js +3 -1
  40. package/lib/object/AcDbDictionary.js.map +1 -1
  41. package/package.json +5 -5
@@ -32,19 +32,10 @@ export declare function acdbPeekDxfHeaderInfo(buffer: ArrayBuffer): AcDbDxfHeade
32
32
  */
33
33
  export declare function acdbMakeAsciiDxfPairReader(text: string): AcDbDxfPairReader;
34
34
  /**
35
- * Bytes decoded per `TextDecoder` call in {@link acdbMakeUtf8AsciiDxfPairReader}.
36
- *
37
- * Sized as a compromise: large enough that a multi-MB DXF costs hundreds of
38
- * decode calls rather than one per line, small enough that windows still holding
39
- * a retained value slice do not pin much memory.
40
- */
41
- /**
42
- * ASCII pair reader that decodes UTF-8 bytes one line-aligned window at a time,
43
- * instead of allocating a full-file decoded string (peak memory ≈ input bytes
44
- * plus one window).
45
- *
46
- * Non-UTF-8 code pages still go through {@link acdbMakeAsciiDxfPairReader}
47
- * after a full `TextDecoder` pass.
35
+ * ASCII pair reader that decodes UTF-8 bytes one line at a time, instead of
36
+ * allocating a full-file decoded string (peak memory ≈ input bytes plus one
37
+ * line). Legacy code pages use the same span path via
38
+ * {@link acdbCreateDxfPairReader}.
48
39
  */
49
40
  export declare function acdbMakeUtf8AsciiDxfPairReader(bytes: Uint8Array): AcDbDxfPairReader;
50
41
  /**
@@ -58,14 +49,16 @@ export declare function acdbMakeBinaryDxfPairReader(data: Uint8Array, options?:
58
49
  }): AcDbDxfPairReader;
59
50
  export interface AcDbCreateDxfPairReaderOptions {
60
51
  /**
61
- * Override text encoding for ASCII DXF.
52
+ * Override text encoding for ASCII and binary DXF.
62
53
  *
63
- * When omitted, the encoding is chosen automatically: bytes that validate
64
- * as UTF-8 decode as UTF-8; otherwise a declared pre-2007 `$DWGCODEPAGE`
65
- * is honored (see {@link acdbCreateDxfPairReader}). When provided, it wins
66
- * over detection — use it to force e.g. `'cp949'` for Korean drawings whose
67
- * header lies. Common non-WHATWG aliases such as `'cp949'` are normalized
68
- * to a supported `TextDecoder` label (`'euc-kr'`).
54
+ * When omitted, the encoding is chosen automatically: ASCII bytes that
55
+ * validate as UTF-8 decode as UTF-8; otherwise a declared pre-2007
56
+ * `$DWGCODEPAGE` is honored. Binary DXF peeks the same header fields and
57
+ * applies a pre-2007 code page to null-terminated strings (see
58
+ * {@link acdbCreateDxfPairReader}). When provided, it wins over detection
59
+ * — use it to force e.g. `'cp949'` for Korean drawings whose header lies.
60
+ * Common non-WHATWG aliases such as `'cp949'` are normalized to a
61
+ * supported `TextDecoder` label (`'euc-kr'`).
69
62
  */
70
63
  encoding?: string;
71
64
  /** Force R12 1-byte group codes for binary DXF. */
@@ -77,16 +70,23 @@ export interface AcDbCreateDxfPairReaderOptions {
77
70
  * ASCII path encoding strategy (hybrid of byte-trust and header sniffing):
78
71
  *
79
72
  * 1. An explicit `options.encoding` always wins — the caller asserts the
80
- * encoding, no sniffing contradicts it.
73
+ * encoding, no sniffing contradicts it. Non-UTF-8 overrides decode
74
+ * line-by-line (never a single full-file string).
81
75
  * 2. Otherwise the bytes are strictly validated as UTF-8. Valid UTF-8
82
76
  * (including pure ASCII) decodes correctly regardless of any stale
83
77
  * `$DWGCODEPAGE`, so the header is never consulted — modern files skip
84
78
  * the header pre-scan entirely and stream straight from bytes.
85
79
  * 3. Invalid UTF-8 means the file is not a spec-conformant modern DXF. The
86
80
  * HEADER is then peeked for `$ACADVER`/`$DWGCODEPAGE`: pre-2007 drawings
87
- * with a declared code page decode through it (matching AutoCAD), while
88
- * R2007+ or headerless files fall back to UTF-8 (replacement chars mark
89
- * the broken bytes).
81
+ * with a declared code page decode through it span-by-span (matching
82
+ * AutoCAD, without materializing one giant string), while R2007+ or
83
+ * headerless files fall back to UTF-8 (replacement chars mark the
84
+ * broken bytes).
85
+ *
86
+ * Binary path: when `options.encoding` is omitted, the HEADER is peeked for
87
+ * a pre-2007 `$DWGCODEPAGE` and that code page is used for null-terminated
88
+ * string values (same AutoCAD rule as ASCII). R2007+ / missing header keep
89
+ * UTF-8.
90
90
  */
91
91
  export declare function acdbCreateDxfPairReader(data: ArrayBuffer | Uint8Array, options?: AcDbCreateDxfPairReaderOptions): AcDbDxfPairReader;
92
92
  //# sourceMappingURL=AcDbDxfPairReader.d.ts.map
@@ -1 +1 @@
1
- {"version":3,"file":"AcDbDxfPairReader.d.ts","sourceRoot":"","sources":["../../src/base/AcDbDxfPairReader.ts"],"names":[],"mappings":"AAAA,OAAO,EAAE,cAAc,EAAE,MAAM,4BAA4B,CAAA;AAU3D,OAAO,KAAK,EAAE,WAAW,EAAE,MAAM,eAAe,CAAA;AAwBhD;;;;;GAKG;AACH,MAAM,WAAW,iBAAiB;IAChC,QAAQ,CAAC,IAAI,EAAE,OAAO,GAAG,QAAQ,CAAA;IACjC,IAAI,IAAI,WAAW,GAAG,SAAS,CAAA;IAC/B,IAAI,IAAI,WAAW,GAAG,SAAS,CAAA;IAC/B,QAAQ,IAAI;QAAE,IAAI,CAAC,EAAE,MAAM,CAAC;QAAC,UAAU,EAAE,MAAM,CAAA;KAAE,CAAA;CAClD;AAED,MAAM,WAAW,iBAAiB;IAChC,OAAO,EAAE,cAAc,GAAG,IAAI,CAAA;IAC9B,QAAQ,EAAE,MAAM,GAAG,IAAI,CAAA;CACxB;AAED,wBAAgB,eAAe,CAAC,IAAI,EAAE,UAAU,GAAG,OAAO,CAMzD;AAkjBD;;;GAGG;AACH,wBAAgB,qBAAqB,CAAC,MAAM,EAAE,WAAW,GAAG,iBAAiB,CA0C5E;AAoDD;;;;GAIG;AACH,wBAAgB,0BAA0B,CAAC,IAAI,EAAE,MAAM,GAAG,iBAAiB,CA+D1E;AAOD;;;;;;GAMG;AAEH;;;;;;;GAOG;AACH,wBAAgB,8BAA8B,CAC5C,KAAK,EAAE,UAAU,GAChB,iBAAiB,CAEnB;AASD;;;;GAIG;AACH,wBAAgB,2BAA2B,CACzC,IAAI,EAAE,UAAU,EAChB,OAAO,GAAE;IAAE,QAAQ,CAAC,EAAE,MAAM,CAAC;IAAC,SAAS,CAAC,EAAE,OAAO,CAAA;CAAO,GACvD,iBAAiB,CAoKnB;AAED,MAAM,WAAW,8BAA8B;IAC7C;;;;;;;;;OASG;IACH,QAAQ,CAAC,EAAE,MAAM,CAAA;IACjB,mDAAmD;IACnD,SAAS,CAAC,EAAE,OAAO,CAAA;CACpB;AAqDD;;;;;;;;;;;;;;;;GAgBG;AACH,wBAAgB,uBAAuB,CACrC,IAAI,EAAE,WAAW,GAAG,UAAU,EAC9B,OAAO,GAAE,8BAAmC,GAC3C,iBAAiB,CA4EnB"}
1
+ {"version":3,"file":"AcDbDxfPairReader.d.ts","sourceRoot":"","sources":["../../src/base/AcDbDxfPairReader.ts"],"names":[],"mappings":"AAAA,OAAO,EAAE,cAAc,EAAE,MAAM,4BAA4B,CAAA;AAU3D,OAAO,KAAK,EAAE,WAAW,EAAE,MAAM,eAAe,CAAA;AAwBhD;;;;;GAKG;AACH,MAAM,WAAW,iBAAiB;IAChC,QAAQ,CAAC,IAAI,EAAE,OAAO,GAAG,QAAQ,CAAA;IACjC,IAAI,IAAI,WAAW,GAAG,SAAS,CAAA;IAC/B,IAAI,IAAI,WAAW,GAAG,SAAS,CAAA;IAC/B,QAAQ,IAAI;QAAE,IAAI,CAAC,EAAE,MAAM,CAAC;QAAC,UAAU,EAAE,MAAM,CAAA;KAAE,CAAA;CAClD;AAED,MAAM,WAAW,iBAAiB;IAChC,OAAO,EAAE,cAAc,GAAG,IAAI,CAAA;IAC9B,QAAQ,EAAE,MAAM,GAAG,IAAI,CAAA;CACxB;AAED,wBAAgB,eAAe,CAAC,IAAI,EAAE,UAAU,GAAG,OAAO,CAMzD;AA+rBD;;;GAGG;AACH,wBAAgB,qBAAqB,CAAC,MAAM,EAAE,WAAW,GAAG,iBAAiB,CAuC5E;AAoDD;;;;GAIG;AACH,wBAAgB,0BAA0B,CAAC,IAAI,EAAE,MAAM,GAAG,iBAAiB,CA+D1E;AAED;;;;;GAKG;AACH,wBAAgB,8BAA8B,CAC5C,KAAK,EAAE,UAAU,GAChB,iBAAiB,CAEnB;AASD;;;;GAIG;AACH,wBAAgB,2BAA2B,CACzC,IAAI,EAAE,UAAU,EAChB,OAAO,GAAE;IAAE,QAAQ,CAAC,EAAE,MAAM,CAAC;IAAC,SAAS,CAAC,EAAE,OAAO,CAAA;CAAO,GACvD,iBAAiB,CAqKnB;AAED,MAAM,WAAW,8BAA8B;IAC7C;;;;;;;;;;;OAWG;IACH,QAAQ,CAAC,EAAE,MAAM,CAAA;IACjB,mDAAmD;IACnD,SAAS,CAAC,EAAE,OAAO,CAAA;CACpB;AAqDD;;;;;;;;;;;;;;;;;;;;;;;GAuBG;AACH,wBAAgB,uBAAuB,CACrC,IAAI,EAAE,WAAW,GAAG,UAAU,EAC9B,OAAO,GAAE,8BAAmC,GAC3C,iBAAiB,CA0EnB"}
@@ -72,16 +72,21 @@ function acdbDecodeAsciiSpan(bytes, start, end) {
72
72
  return text;
73
73
  }
74
74
  /**
75
- * Decodes a UTF-8 byte span, taking the ASCII fast path when possible.
75
+ * Decodes a text byte span, taking the ASCII fast path when possible.
76
76
  *
77
77
  * `nonAscii` must be true exactly when the span holds a byte `>= 0x80`. It is
78
78
  * produced by the line scanner, so this function never re-scans the span.
79
+ *
80
+ * For legacy DXF code pages (windows-1252, euc-kr, gbk, shift-jis, …) the
81
+ * trail-byte ranges never include `0x0A`/`0x0D`, so each line is a complete
82
+ * character sequence and may be decoded independently — same invariant the
83
+ * UTF-8 reader relies on.
79
84
  */
80
- function acdbDecodeUtf8Span(bytes, start, end, nonAscii) {
85
+ function acdbDecodeTextSpan(bytes, start, end, nonAscii, decoder) {
81
86
  if (!nonAscii) {
82
87
  return acdbDecodeAsciiSpan(bytes, start, end);
83
88
  }
84
- return UTF8_DECODER.decode(bytes.subarray(start, end));
89
+ return decoder.decode(bytes.subarray(start, end));
85
90
  }
86
91
  /**
87
92
  * Parses an integer group code straight from the bytes of a code line,
@@ -90,9 +95,9 @@ function acdbDecodeUtf8Span(bytes, start, end, nonAscii) {
90
95
  * Returns NaN for blank or malformed code lines. DXF group code lines are
91
96
  * pure ASCII; the rare non-ASCII line falls back to `Number` + `isFinite`.
92
97
  */
93
- function acdbReadDxfCodeFromBytes(bytes, start, end, nonAscii) {
98
+ function acdbReadDxfCodeFromBytes(bytes, start, end, nonAscii, decoder) {
94
99
  if (nonAscii) {
95
- var n = Number(acdbDecodeUtf8Span(bytes, start, end, true).trim());
100
+ var n = Number(acdbDecodeTextSpan(bytes, start, end, true, decoder).trim());
96
101
  return Number.isFinite(n) ? n : NaN;
97
102
  }
98
103
  var i = start;
@@ -364,13 +369,13 @@ function acdbDecodeHexBinaryText(text, start, end) {
364
369
  /**
365
370
  * Decodes a code-310 hex line from bytes, trimming ASCII whitespace.
366
371
  */
367
- function acdbDecodeHexBinarySpan(bytes, start, end, nonAscii) {
372
+ function acdbDecodeHexBinarySpan(bytes, start, end, nonAscii, decoder) {
368
373
  while (start < end && acdbIsAsciiWhitespace(bytes[start]))
369
374
  start++;
370
375
  while (end > start && acdbIsAsciiWhitespace(bytes[end - 1]))
371
376
  end--;
372
377
  if (nonAscii) {
373
- var text = acdbDecodeUtf8Span(bytes, start, end, true);
378
+ var text = acdbDecodeTextSpan(bytes, start, end, true, decoder);
374
379
  return acdbDecodeHexBinaryText(text, 0, text.length);
375
380
  }
376
381
  var byteLength = (end - start) >>> 1;
@@ -388,7 +393,7 @@ function acdbDecodeHexBinarySpan(bytes, start, end, nonAscii) {
388
393
  * nonAscii is the flag produced by the line scanner for this exact span; it
389
394
  * is forwarded to every decode helper so no helper has to re-scan the bytes.
390
395
  */
391
- function parseAsciiValueSpan(code, bytes, start, end, nonAscii) {
396
+ function parseAsciiValueSpan(code, bytes, start, end, nonAscii, decoder) {
392
397
  var type = acdbDxfValueType(code);
393
398
  if (type === 'comment')
394
399
  return null;
@@ -397,12 +402,12 @@ function parseAsciiValueSpan(code, bytes, start, end, nonAscii) {
397
402
  return {
398
403
  code: code,
399
404
  type: type,
400
- value: acdbDecodeUtf8Span(bytes, start, end, nonAscii)
405
+ value: acdbDecodeTextSpan(bytes, start, end, nonAscii, decoder)
401
406
  };
402
407
  case 'int': {
403
408
  var fast = acdbParseIntSpan(bytes, start, end);
404
409
  var n = fast === undefined
405
- ? parseInt(acdbDecodeUtf8Span(bytes, start, end, nonAscii), 10)
410
+ ? parseInt(acdbDecodeTextSpan(bytes, start, end, nonAscii, decoder), 10)
406
411
  : fast;
407
412
  return { code: code, type: type, value: Number.isFinite(n) ? n : 0 };
408
413
  }
@@ -410,7 +415,7 @@ function parseAsciiValueSpan(code, bytes, start, end, nonAscii) {
410
415
  var fast = acdbParseLongSpan(bytes, start, end);
411
416
  if (fast !== undefined)
412
417
  return { code: code, type: type, value: fast };
413
- var raw = acdbDecodeUtf8Span(bytes, start, end, nonAscii);
418
+ var raw = acdbDecodeTextSpan(bytes, start, end, nonAscii, decoder);
414
419
  var n = Number(raw);
415
420
  if (Number.isSafeInteger(n))
416
421
  return { code: code, type: type, value: n };
@@ -424,7 +429,7 @@ function parseAsciiValueSpan(code, bytes, start, end, nonAscii) {
424
429
  case 'double': {
425
430
  var fast = acdbParseDoubleSpan(bytes, start, end);
426
431
  var n = fast === undefined
427
- ? Number(acdbDecodeUtf8Span(bytes, start, end, nonAscii))
432
+ ? Number(acdbDecodeTextSpan(bytes, start, end, nonAscii, decoder))
428
433
  : fast;
429
434
  return { code: code, type: type, value: Number.isFinite(n) ? n : 0 };
430
435
  }
@@ -432,11 +437,11 @@ function parseAsciiValueSpan(code, bytes, start, end, nonAscii) {
432
437
  var fast = acdbDxfRawBoolIsTrue(bytes, start, end, nonAscii);
433
438
  if (fast !== undefined)
434
439
  return { code: code, type: type, value: fast };
435
- var trimmed = acdbDecodeUtf8Span(bytes, start, end, nonAscii).trim();
440
+ var trimmed = acdbDecodeTextSpan(bytes, start, end, nonAscii, decoder).trim();
436
441
  return { code: code, type: type, value: trimmed !== '' && trimmed !== '0' };
437
442
  }
438
443
  case 'handle': {
439
- var value = acdbDecodeUtf8Span(bytes, start, end, nonAscii);
444
+ var value = acdbDecodeTextSpan(bytes, start, end, nonAscii, decoder);
440
445
  if (value.length === 0)
441
446
  return { code: code, type: type, value: value };
442
447
  var first = value.charCodeAt(0);
@@ -450,19 +455,32 @@ function parseAsciiValueSpan(code, bytes, start, end, nonAscii) {
450
455
  return {
451
456
  code: code,
452
457
  type: type,
453
- value: acdbDecodeHexBinarySpan(bytes, start, end, nonAscii)
458
+ value: acdbDecodeHexBinarySpan(bytes, start, end, nonAscii, decoder)
454
459
  };
455
460
  default:
456
461
  return null;
457
462
  }
458
463
  }
464
+ function isUtf8Encoding(encoding) {
465
+ var e = encoding.toLowerCase().replace(/_/g, '-');
466
+ return e === 'utf-8' || e === 'utf8' || e === 'unicode-1-1-utf-8';
467
+ }
459
468
  /**
460
- * ASCII/UTF-8 pair reader over raw bytes.
469
+ * ASCII pair reader over raw bytes, decoding non-ASCII spans with `encoding`.
461
470
  *
462
- * Line breaks are single bytes and UTF-8 continuation bytes never equal
463
- * 0x0A/0x0D, so non-ASCII text values never straddle line boundaries.
471
+ * Line breaks are single bytes. For UTF-8, continuation bytes never equal
472
+ * `0x0A`/`0x0D`. The same holds for the legacy DXF code pages we support
473
+ * (windows-125x, euc-kr/cp949, gbk, shift-jis, big5, …): their trail-byte
474
+ * ranges exclude CR/LF, so non-ASCII text values never straddle line
475
+ * boundaries and each line may be decoded independently.
476
+ *
477
+ * @param encoding - WHATWG / `TextDecoder` label. Defaults to UTF-8.
464
478
  */
465
- function acdbMakeUtf8DxfPairReader(bytes) {
479
+ function acdbMakeByteAsciiDxfPairReader(bytes, encoding) {
480
+ if (encoding === void 0) { encoding = 'utf-8'; }
481
+ var decoder = isUtf8Encoding(encoding)
482
+ ? UTF8_DECODER
483
+ : new TextDecoder(encoding);
466
484
  var pos = bytes.length >= 3 &&
467
485
  bytes[0] === 0xef &&
468
486
  bytes[1] === 0xbb &&
@@ -506,7 +524,7 @@ function acdbMakeUtf8DxfPairReader(bytes) {
506
524
  var codeSpan = readLineSpan();
507
525
  if (codeSpan === undefined)
508
526
  return undefined;
509
- var code = acdbReadDxfCodeFromBytes(bytes, codeSpan.start, codeSpan.end, codeSpan.nonAscii);
527
+ var code = acdbReadDxfCodeFromBytes(bytes, codeSpan.start, codeSpan.end, codeSpan.nonAscii, decoder);
510
528
  if (Number.isNaN(code))
511
529
  continue;
512
530
  if (code === 999) {
@@ -517,7 +535,7 @@ function acdbMakeUtf8DxfPairReader(bytes) {
517
535
  var valueSpan = readLineSpan();
518
536
  if (valueSpan === undefined)
519
537
  return undefined;
520
- var pair = parseAsciiValueSpan(code, bytes, valueSpan.start, valueSpan.end, valueSpan.nonAscii);
538
+ var pair = parseAsciiValueSpan(code, bytes, valueSpan.start, valueSpan.end, valueSpan.nonAscii, decoder);
521
539
  if (pair)
522
540
  return pair;
523
541
  }
@@ -545,6 +563,95 @@ function acdbMakeUtf8DxfPairReader(bytes) {
545
563
  }
546
564
  };
547
565
  }
566
+ function acdbMakeUtf8DxfPairReader(bytes) {
567
+ return acdbMakeByteAsciiDxfPairReader(bytes, 'utf-8');
568
+ }
569
+ /**
570
+ * Resolve a `$DWGCODEPAGE` header value to a `TextDecoder` label.
571
+ * Unknown names yield `null` so callers keep their UTF-8 fallback.
572
+ */
573
+ function acdbResolveDwgCodePageEncoding(value) {
574
+ var _a;
575
+ if (!value)
576
+ return null;
577
+ var codePage = AcDbCodePage[value];
578
+ if (codePage === undefined)
579
+ return null;
580
+ return (_a = acdbDwgCodePageToEncoding(codePage)) !== null && _a !== void 0 ? _a : null;
581
+ }
582
+ /**
583
+ * Peek `$ACADVER` / `$DWGCODEPAGE` from a binary DXF HEADER section.
584
+ *
585
+ * Header keywords and code-page names are ASCII, so the temporary reader
586
+ * uses UTF-8 regardless of the drawing's text encoding.
587
+ */
588
+ function acdbPeekBinaryDxfHeaderInfo(data, legacyR12) {
589
+ var reader = acdbMakeBinaryDxfPairReader(data, {
590
+ encoding: 'utf-8',
591
+ legacyR12: legacyR12
592
+ });
593
+ var version = null;
594
+ var encoding = null;
595
+ var inHeader = false;
596
+ var pendingVar = null;
597
+ for (var pair = reader.next(); pair; pair = reader.next()) {
598
+ if (pair.code === 0 && pair.type === 'string') {
599
+ var name_1 = pair.value;
600
+ if (name_1 === 'SECTION') {
601
+ var section = reader.next();
602
+ if ((section === null || section === void 0 ? void 0 : section.code) === 2 &&
603
+ section.type === 'string' &&
604
+ section.value === 'HEADER') {
605
+ inHeader = true;
606
+ }
607
+ pendingVar = null;
608
+ continue;
609
+ }
610
+ if (name_1 === 'ENDSEC' && inHeader) {
611
+ return { version: version, encoding: encoding };
612
+ }
613
+ pendingVar = null;
614
+ continue;
615
+ }
616
+ if (!inHeader)
617
+ continue;
618
+ if (pair.code === 9 && pair.type === 'string') {
619
+ pendingVar = pair.value;
620
+ continue;
621
+ }
622
+ if (pendingVar === '$ACADVER' && pair.code === 1 && pair.type === 'string') {
623
+ try {
624
+ version = new AcDbDwgVersion(pair.value);
625
+ }
626
+ catch (_a) {
627
+ // Unrecognized spelling: leave version null.
628
+ }
629
+ }
630
+ else if (pendingVar === '$DWGCODEPAGE' &&
631
+ pair.code === 3 &&
632
+ pair.type === 'string') {
633
+ encoding = acdbResolveDwgCodePageEncoding(pair.value);
634
+ }
635
+ if (pendingVar !== null) {
636
+ pendingVar = null;
637
+ if (version && encoding)
638
+ return { version: version, encoding: encoding };
639
+ }
640
+ }
641
+ return { version: version, encoding: encoding };
642
+ }
643
+ /**
644
+ * Prefer a pre-2007 `$DWGCODEPAGE` when the header declares both a legacy
645
+ * version and a known code page; otherwise return null (caller keeps UTF-8).
646
+ */
647
+ function acdbLegacyEncodingFromHeader(info) {
648
+ if (info.version &&
649
+ !info.version.capabilities.supportsUtf8CodePage &&
650
+ info.encoding) {
651
+ return info.encoding;
652
+ }
653
+ return null;
654
+ }
548
655
  /**
549
656
  * Peek `$ACADVER` / `$DWGCODEPAGE` from the HEADER section without decoding
550
657
  * the whole file. Uses 64 KiB UTF-8 chunks (same strategy as AcDbDxfParser).
@@ -580,10 +687,8 @@ export function acdbPeekDxfHeaderInfo(buffer) {
580
687
  }
581
688
  else if (inHeader && line === '$DWGCODEPAGE') {
582
689
  var value = (_d = lines[i + 2]) === null || _d === void 0 ? void 0 : _d.trim();
583
- if (value) {
584
- var codePage = AcDbCodePage[value];
585
- encoding = acdbDwgCodePageToEncoding(codePage);
586
- }
690
+ if (value)
691
+ encoding = acdbResolveDwgCodePageEncoding(value);
587
692
  }
588
693
  if (version && encoding)
589
694
  return { version: version, encoding: encoding };
@@ -714,24 +819,11 @@ export function acdbMakeAsciiDxfPairReader(text) {
714
819
  }
715
820
  };
716
821
  }
717
- function isUtf8Encoding(encoding) {
718
- var e = encoding.toLowerCase().replace(/_/g, '-');
719
- return e === 'utf-8' || e === 'utf8' || e === 'unicode-1-1-utf-8';
720
- }
721
822
  /**
722
- * Bytes decoded per `TextDecoder` call in {@link acdbMakeUtf8AsciiDxfPairReader}.
723
- *
724
- * Sized as a compromise: large enough that a multi-MB DXF costs hundreds of
725
- * decode calls rather than one per line, small enough that windows still holding
726
- * a retained value slice do not pin much memory.
727
- */
728
- /**
729
- * ASCII pair reader that decodes UTF-8 bytes one line-aligned window at a time,
730
- * instead of allocating a full-file decoded string (peak memory ≈ input bytes
731
- * plus one window).
732
- *
733
- * Non-UTF-8 code pages still go through {@link acdbMakeAsciiDxfPairReader}
734
- * after a full `TextDecoder` pass.
823
+ * ASCII pair reader that decodes UTF-8 bytes one line at a time, instead of
824
+ * allocating a full-file decoded string (peak memory ≈ input bytes plus one
825
+ * line). Legacy code pages use the same span path via
826
+ * {@link acdbCreateDxfPairReader}.
735
827
  */
736
828
  export function acdbMakeUtf8AsciiDxfPairReader(bytes) {
737
829
  return acdbMakeUtf8DxfPairReader(bytes);
@@ -753,6 +845,7 @@ export function acdbMakeBinaryDxfPairReader(data, options) {
753
845
  if (options === void 0) { options = {}; }
754
846
  var encoding = (_a = options.encoding) !== null && _a !== void 0 ? _a : 'utf-8';
755
847
  var legacyR12 = (_b = options.legacyR12) !== null && _b !== void 0 ? _b : false;
848
+ var decoder = new TextDecoder(encoding);
756
849
  var PREFIX = 22;
757
850
  var view = new DataView(data.buffer, data.byteOffset, data.byteLength);
758
851
  var offset = data.length >= PREFIX ? PREFIX : data.length;
@@ -791,7 +884,7 @@ export function acdbMakeBinaryDxfPairReader(data, options) {
791
884
  return undefined;
792
885
  var bytes = data.subarray(start, offset);
793
886
  offset += 1;
794
- return new TextDecoder(encoding).decode(bytes);
887
+ return decoder.decode(bytes);
795
888
  }
796
889
  function readInt16() {
797
890
  if (offset + 2 > data.length)
@@ -998,18 +1091,26 @@ function acdbIsValidUtf8(bytes) {
998
1091
  * ASCII path encoding strategy (hybrid of byte-trust and header sniffing):
999
1092
  *
1000
1093
  * 1. An explicit `options.encoding` always wins — the caller asserts the
1001
- * encoding, no sniffing contradicts it.
1094
+ * encoding, no sniffing contradicts it. Non-UTF-8 overrides decode
1095
+ * line-by-line (never a single full-file string).
1002
1096
  * 2. Otherwise the bytes are strictly validated as UTF-8. Valid UTF-8
1003
1097
  * (including pure ASCII) decodes correctly regardless of any stale
1004
1098
  * `$DWGCODEPAGE`, so the header is never consulted — modern files skip
1005
1099
  * the header pre-scan entirely and stream straight from bytes.
1006
1100
  * 3. Invalid UTF-8 means the file is not a spec-conformant modern DXF. The
1007
1101
  * HEADER is then peeked for `$ACADVER`/`$DWGCODEPAGE`: pre-2007 drawings
1008
- * with a declared code page decode through it (matching AutoCAD), while
1009
- * R2007+ or headerless files fall back to UTF-8 (replacement chars mark
1010
- * the broken bytes).
1102
+ * with a declared code page decode through it span-by-span (matching
1103
+ * AutoCAD, without materializing one giant string), while R2007+ or
1104
+ * headerless files fall back to UTF-8 (replacement chars mark the
1105
+ * broken bytes).
1106
+ *
1107
+ * Binary path: when `options.encoding` is omitted, the HEADER is peeked for
1108
+ * a pre-2007 `$DWGCODEPAGE` and that code page is used for null-terminated
1109
+ * string values (same AutoCAD rule as ASCII). R2007+ / missing header keep
1110
+ * UTF-8.
1011
1111
  */
1012
1112
  export function acdbCreateDxfPairReader(data, options) {
1113
+ var _a;
1013
1114
  if (options === void 0) { options = {}; }
1014
1115
  var bytes = data instanceof Uint8Array ? data : new Uint8Array(data);
1015
1116
  var overrideEncoding = options.encoding
@@ -1018,35 +1119,38 @@ export function acdbCreateDxfPairReader(data, options) {
1018
1119
  if (acdbIsBinaryDxf(bytes)) {
1019
1120
  var encoding = overrideEncoding;
1020
1121
  var legacyR12 = options.legacyR12;
1021
- if (encoding == null || legacyR12 == null) {
1022
- encoding = encoding !== null && encoding !== void 0 ? encoding : 'utf-8';
1023
- if (legacyR12 == null) {
1024
- // After the 22-byte magic: R12 uses 1-byte codes (`0,'S'`), modern
1025
- // uses 2-byte LE codes (`0,0,'S'`) for the first SECTION marker.
1026
- var PREFIX = 22;
1027
- var b0 = bytes[PREFIX];
1028
- var b1 = bytes[PREFIX + 1];
1029
- var b2 = bytes[PREFIX + 2];
1030
- if (b0 === 0 && b1 === 0x53 /* 'S' */) {
1031
- legacyR12 = true;
1032
- }
1033
- else if (b0 === 0 && b1 === 0 && b2 === 0x53 /* 'S' */) {
1034
- legacyR12 = false;
1035
- }
1036
- else {
1037
- legacyR12 = false;
1038
- }
1122
+ if (legacyR12 == null) {
1123
+ // After the 22-byte magic: R12 uses 1-byte codes (`0,'S'`), modern
1124
+ // uses 2-byte LE codes (`0,0,'S'`) for the first SECTION marker.
1125
+ var PREFIX = 22;
1126
+ var b0 = bytes[PREFIX];
1127
+ var b1 = bytes[PREFIX + 1];
1128
+ var b2 = bytes[PREFIX + 2];
1129
+ if (b0 === 0 && b1 === 0x53 /* 'S' */) {
1130
+ legacyR12 = true;
1131
+ }
1132
+ else if (b0 === 0 && b1 === 0 && b2 === 0x53 /* 'S' */) {
1133
+ legacyR12 = false;
1039
1134
  }
1135
+ else {
1136
+ legacyR12 = false;
1137
+ }
1138
+ }
1139
+ if (encoding == null) {
1140
+ try {
1141
+ encoding =
1142
+ (_a = acdbLegacyEncodingFromHeader(acdbPeekBinaryDxfHeaderInfo(bytes, legacyR12))) !== null && _a !== void 0 ? _a : undefined;
1143
+ }
1144
+ catch (_b) {
1145
+ // Unrecognized $ACADVER: stay on UTF-8.
1146
+ }
1147
+ encoding = encoding !== null && encoding !== void 0 ? encoding : 'utf-8';
1040
1148
  }
1041
1149
  return acdbMakeBinaryDxfPairReader(bytes, { encoding: encoding, legacyR12: legacyR12 });
1042
1150
  }
1043
1151
  // 1. Explicit override wins: the caller asserts the encoding.
1044
1152
  if (overrideEncoding) {
1045
- if (isUtf8Encoding(overrideEncoding)) {
1046
- return acdbMakeUtf8DxfPairReader(bytes);
1047
- }
1048
- var text = new TextDecoder(overrideEncoding).decode(bytes);
1049
- return acdbMakeAsciiDxfPairReader(text);
1153
+ return acdbMakeByteAsciiDxfPairReader(bytes, overrideEncoding);
1050
1154
  }
1051
1155
  // 2. Trust the bytes first: valid UTF-8 (including pure ASCII) is decoded
1052
1156
  // correctly no matter what a stale $DWGCODEPAGE claims, so modern files
@@ -1055,27 +1159,23 @@ export function acdbCreateDxfPairReader(data, options) {
1055
1159
  return acdbMakeUtf8DxfPairReader(bytes);
1056
1160
  }
1057
1161
  // 3. Invalid UTF-8: fall back to the pre-2007 $DWGCODEPAGE when one is
1058
- // declared (genuine legacy ANSI content). R2007+ or headerless files
1059
- // stay on UTF-8, where replacement characters mark the broken bytes.
1162
+ // declared (genuine legacy ANSI content). Decode span-by-span so large
1163
+ // drawings never hit the V8 max-string-length ceiling. R2007+ or
1164
+ // headerless files stay on UTF-8, where replacement characters mark
1165
+ // the broken bytes.
1060
1166
  var legacyEncoding = null;
1061
1167
  try {
1062
1168
  var buffer = bytes.byteOffset === 0 && bytes.byteLength === bytes.buffer.byteLength
1063
1169
  ? bytes.buffer
1064
1170
  : bytes.buffer.slice(bytes.byteOffset, bytes.byteOffset + bytes.byteLength);
1065
- var info = acdbPeekDxfHeaderInfo(buffer);
1066
- if (info.version &&
1067
- !info.version.capabilities.supportsUtf8CodePage &&
1068
- info.encoding) {
1069
- legacyEncoding = info.encoding;
1070
- }
1171
+ legacyEncoding = acdbLegacyEncodingFromHeader(acdbPeekDxfHeaderInfo(buffer));
1071
1172
  }
1072
- catch (_a) {
1173
+ catch (_c) {
1073
1174
  // Unrecognized $ACADVER spelling: treat as version-less and stay UTF-8
1074
1175
  // instead of failing the whole read.
1075
1176
  }
1076
1177
  if (legacyEncoding) {
1077
- var text = new TextDecoder(legacyEncoding).decode(bytes);
1078
- return acdbMakeAsciiDxfPairReader(text);
1178
+ return acdbMakeByteAsciiDxfPairReader(bytes, legacyEncoding);
1079
1179
  }
1080
1180
  return acdbMakeUtf8DxfPairReader(bytes);
1081
1181
  }