@mlightcad/data-model 1.15.2 → 1.15.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/data-model.cjs +17 -17
- package/lib/base/AcDbDxfPairReader.d.ts +24 -24
- package/lib/base/AcDbDxfPairReader.d.ts.map +1 -1
- package/lib/base/AcDbDxfPairReader.js +181 -81
- package/lib/base/AcDbDxfPairReader.js.map +1 -1
- package/lib/database/AcDbBlockTableRecord.d.ts.map +1 -1
- package/lib/database/AcDbBlockTableRecord.js +3 -1
- package/lib/database/AcDbBlockTableRecord.js.map +1 -1
- package/lib/database/AcDbDatabase.d.ts.map +1 -1
- package/lib/database/AcDbDatabase.js +110 -70
- package/lib/database/AcDbDatabase.js.map +1 -1
- package/lib/database/AcDbSymbolTable.d.ts.map +1 -1
- package/lib/database/AcDbSymbolTable.js +4 -1
- package/lib/database/AcDbSymbolTable.js.map +1 -1
- package/lib/entity/AcDbAttribute.d.ts.map +1 -1
- package/lib/entity/AcDbAttribute.js +13 -0
- package/lib/entity/AcDbAttribute.js.map +1 -1
- package/lib/entity/AcDbAttributeDefinition.d.ts.map +1 -1
- package/lib/entity/AcDbAttributeDefinition.js +13 -0
- package/lib/entity/AcDbAttributeDefinition.js.map +1 -1
- package/lib/entity/AcDbBlockReference.js +1 -1
- package/lib/entity/AcDbBlockReference.js.map +1 -1
- package/lib/entity/AcDbEntity.d.ts.map +1 -1
- package/lib/entity/AcDbEntity.js +13 -2
- package/lib/entity/AcDbEntity.js.map +1 -1
- package/lib/entity/AcDbMLeader.d.ts.map +1 -1
- package/lib/entity/AcDbMLeader.js +18 -4
- package/lib/entity/AcDbMLeader.js.map +1 -1
- package/lib/entity/AcDbViewport.js +4 -4
- package/lib/entity/AcDbViewport.js.map +1 -1
- package/lib/entity/dimension/AcDbAlignedDimension.d.ts.map +1 -1
- package/lib/entity/dimension/AcDbAlignedDimension.js +3 -2
- package/lib/entity/dimension/AcDbAlignedDimension.js.map +1 -1
- package/lib/entity/dimension/AcDbDimension.d.ts.map +1 -1
- package/lib/entity/dimension/AcDbDimension.js +27 -3
- package/lib/entity/dimension/AcDbDimension.js.map +1 -1
- package/lib/object/AcDbDictionary.d.ts.map +1 -1
- package/lib/object/AcDbDictionary.js +3 -1
- package/lib/object/AcDbDictionary.js.map +1 -1
- package/package.json +5 -5
|
@@ -32,19 +32,10 @@ export declare function acdbPeekDxfHeaderInfo(buffer: ArrayBuffer): AcDbDxfHeade
|
|
|
32
32
|
*/
|
|
33
33
|
export declare function acdbMakeAsciiDxfPairReader(text: string): AcDbDxfPairReader;
|
|
34
34
|
/**
|
|
35
|
-
*
|
|
36
|
-
*
|
|
37
|
-
*
|
|
38
|
-
*
|
|
39
|
-
* a retained value slice do not pin much memory.
|
|
40
|
-
*/
|
|
41
|
-
/**
|
|
42
|
-
* ASCII pair reader that decodes UTF-8 bytes one line-aligned window at a time,
|
|
43
|
-
* instead of allocating a full-file decoded string (peak memory ≈ input bytes
|
|
44
|
-
* plus one window).
|
|
45
|
-
*
|
|
46
|
-
* Non-UTF-8 code pages still go through {@link acdbMakeAsciiDxfPairReader}
|
|
47
|
-
* after a full `TextDecoder` pass.
|
|
35
|
+
* ASCII pair reader that decodes UTF-8 bytes one line at a time, instead of
|
|
36
|
+
* allocating a full-file decoded string (peak memory ≈ input bytes plus one
|
|
37
|
+
* line). Legacy code pages use the same span path via
|
|
38
|
+
* {@link acdbCreateDxfPairReader}.
|
|
48
39
|
*/
|
|
49
40
|
export declare function acdbMakeUtf8AsciiDxfPairReader(bytes: Uint8Array): AcDbDxfPairReader;
|
|
50
41
|
/**
|
|
@@ -58,14 +49,16 @@ export declare function acdbMakeBinaryDxfPairReader(data: Uint8Array, options?:
|
|
|
58
49
|
}): AcDbDxfPairReader;
|
|
59
50
|
export interface AcDbCreateDxfPairReaderOptions {
|
|
60
51
|
/**
|
|
61
|
-
* Override text encoding for ASCII DXF.
|
|
52
|
+
* Override text encoding for ASCII and binary DXF.
|
|
62
53
|
*
|
|
63
|
-
* When omitted, the encoding is chosen automatically: bytes that
|
|
64
|
-
* as UTF-8 decode as UTF-8; otherwise a declared pre-2007
|
|
65
|
-
* is honored
|
|
66
|
-
*
|
|
67
|
-
*
|
|
68
|
-
*
|
|
54
|
+
* When omitted, the encoding is chosen automatically: ASCII bytes that
|
|
55
|
+
* validate as UTF-8 decode as UTF-8; otherwise a declared pre-2007
|
|
56
|
+
* `$DWGCODEPAGE` is honored. Binary DXF peeks the same header fields and
|
|
57
|
+
* applies a pre-2007 code page to null-terminated strings (see
|
|
58
|
+
* {@link acdbCreateDxfPairReader}). When provided, it wins over detection
|
|
59
|
+
* — use it to force e.g. `'cp949'` for Korean drawings whose header lies.
|
|
60
|
+
* Common non-WHATWG aliases such as `'cp949'` are normalized to a
|
|
61
|
+
* supported `TextDecoder` label (`'euc-kr'`).
|
|
69
62
|
*/
|
|
70
63
|
encoding?: string;
|
|
71
64
|
/** Force R12 1-byte group codes for binary DXF. */
|
|
@@ -77,16 +70,23 @@ export interface AcDbCreateDxfPairReaderOptions {
|
|
|
77
70
|
* ASCII path encoding strategy (hybrid of byte-trust and header sniffing):
|
|
78
71
|
*
|
|
79
72
|
* 1. An explicit `options.encoding` always wins — the caller asserts the
|
|
80
|
-
* encoding, no sniffing contradicts it.
|
|
73
|
+
* encoding, no sniffing contradicts it. Non-UTF-8 overrides decode
|
|
74
|
+
* line-by-line (never a single full-file string).
|
|
81
75
|
* 2. Otherwise the bytes are strictly validated as UTF-8. Valid UTF-8
|
|
82
76
|
* (including pure ASCII) decodes correctly regardless of any stale
|
|
83
77
|
* `$DWGCODEPAGE`, so the header is never consulted — modern files skip
|
|
84
78
|
* the header pre-scan entirely and stream straight from bytes.
|
|
85
79
|
* 3. Invalid UTF-8 means the file is not a spec-conformant modern DXF. The
|
|
86
80
|
* HEADER is then peeked for `$ACADVER`/`$DWGCODEPAGE`: pre-2007 drawings
|
|
87
|
-
* with a declared code page decode through it (matching
|
|
88
|
-
*
|
|
89
|
-
*
|
|
81
|
+
* with a declared code page decode through it span-by-span (matching
|
|
82
|
+
* AutoCAD, without materializing one giant string), while R2007+ or
|
|
83
|
+
* headerless files fall back to UTF-8 (replacement chars mark the
|
|
84
|
+
* broken bytes).
|
|
85
|
+
*
|
|
86
|
+
* Binary path: when `options.encoding` is omitted, the HEADER is peeked for
|
|
87
|
+
* a pre-2007 `$DWGCODEPAGE` and that code page is used for null-terminated
|
|
88
|
+
* string values (same AutoCAD rule as ASCII). R2007+ / missing header keep
|
|
89
|
+
* UTF-8.
|
|
90
90
|
*/
|
|
91
91
|
export declare function acdbCreateDxfPairReader(data: ArrayBuffer | Uint8Array, options?: AcDbCreateDxfPairReaderOptions): AcDbDxfPairReader;
|
|
92
92
|
//# sourceMappingURL=AcDbDxfPairReader.d.ts.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"AcDbDxfPairReader.d.ts","sourceRoot":"","sources":["../../src/base/AcDbDxfPairReader.ts"],"names":[],"mappings":"AAAA,OAAO,EAAE,cAAc,EAAE,MAAM,4BAA4B,CAAA;AAU3D,OAAO,KAAK,EAAE,WAAW,EAAE,MAAM,eAAe,CAAA;AAwBhD;;;;;GAKG;AACH,MAAM,WAAW,iBAAiB;IAChC,QAAQ,CAAC,IAAI,EAAE,OAAO,GAAG,QAAQ,CAAA;IACjC,IAAI,IAAI,WAAW,GAAG,SAAS,CAAA;IAC/B,IAAI,IAAI,WAAW,GAAG,SAAS,CAAA;IAC/B,QAAQ,IAAI;QAAE,IAAI,CAAC,EAAE,MAAM,CAAC;QAAC,UAAU,EAAE,MAAM,CAAA;KAAE,CAAA;CAClD;AAED,MAAM,WAAW,iBAAiB;IAChC,OAAO,EAAE,cAAc,GAAG,IAAI,CAAA;IAC9B,QAAQ,EAAE,MAAM,GAAG,IAAI,CAAA;CACxB;AAED,wBAAgB,eAAe,CAAC,IAAI,EAAE,UAAU,GAAG,OAAO,CAMzD;
|
|
1
|
+
{"version":3,"file":"AcDbDxfPairReader.d.ts","sourceRoot":"","sources":["../../src/base/AcDbDxfPairReader.ts"],"names":[],"mappings":"AAAA,OAAO,EAAE,cAAc,EAAE,MAAM,4BAA4B,CAAA;AAU3D,OAAO,KAAK,EAAE,WAAW,EAAE,MAAM,eAAe,CAAA;AAwBhD;;;;;GAKG;AACH,MAAM,WAAW,iBAAiB;IAChC,QAAQ,CAAC,IAAI,EAAE,OAAO,GAAG,QAAQ,CAAA;IACjC,IAAI,IAAI,WAAW,GAAG,SAAS,CAAA;IAC/B,IAAI,IAAI,WAAW,GAAG,SAAS,CAAA;IAC/B,QAAQ,IAAI;QAAE,IAAI,CAAC,EAAE,MAAM,CAAC;QAAC,UAAU,EAAE,MAAM,CAAA;KAAE,CAAA;CAClD;AAED,MAAM,WAAW,iBAAiB;IAChC,OAAO,EAAE,cAAc,GAAG,IAAI,CAAA;IAC9B,QAAQ,EAAE,MAAM,GAAG,IAAI,CAAA;CACxB;AAED,wBAAgB,eAAe,CAAC,IAAI,EAAE,UAAU,GAAG,OAAO,CAMzD;AA+rBD;;;GAGG;AACH,wBAAgB,qBAAqB,CAAC,MAAM,EAAE,WAAW,GAAG,iBAAiB,CAuC5E;AAoDD;;;;GAIG;AACH,wBAAgB,0BAA0B,CAAC,IAAI,EAAE,MAAM,GAAG,iBAAiB,CA+D1E;AAED;;;;;GAKG;AACH,wBAAgB,8BAA8B,CAC5C,KAAK,EAAE,UAAU,GAChB,iBAAiB,CAEnB;AASD;;;;GAIG;AACH,wBAAgB,2BAA2B,CACzC,IAAI,EAAE,UAAU,EAChB,OAAO,GAAE;IAAE,QAAQ,CAAC,EAAE,MAAM,CAAC;IAAC,SAAS,CAAC,EAAE,OAAO,CAAA;CAAO,GACvD,iBAAiB,CAqKnB;AAED,MAAM,WAAW,8BAA8B;IAC7C;;;;;;;;;;;OAWG;IACH,QAAQ,CAAC,EAAE,MAAM,CAAA;IACjB,mDAAmD;IACnD,SAAS,CAAC,EAAE,OAAO,CAAA;CACpB;AAqDD;;;;;;;;;;;;;;;;;;;;;;;GAuBG;AACH,wBAAgB,uBAAuB,CACrC,IAAI,EAAE,WAAW,GAAG,UAAU,EAC9B,OAAO,GAAE,8BAAmC,GAC3C,iBAAiB,CA0EnB"}
|
|
@@ -72,16 +72,21 @@ function acdbDecodeAsciiSpan(bytes, start, end) {
|
|
|
72
72
|
return text;
|
|
73
73
|
}
|
|
74
74
|
/**
|
|
75
|
-
* Decodes a
|
|
75
|
+
* Decodes a text byte span, taking the ASCII fast path when possible.
|
|
76
76
|
*
|
|
77
77
|
* `nonAscii` must be true exactly when the span holds a byte `>= 0x80`. It is
|
|
78
78
|
* produced by the line scanner, so this function never re-scans the span.
|
|
79
|
+
*
|
|
80
|
+
* For legacy DXF code pages (windows-1252, euc-kr, gbk, shift-jis, …) the
|
|
81
|
+
* trail-byte ranges never include `0x0A`/`0x0D`, so each line is a complete
|
|
82
|
+
* character sequence and may be decoded independently — same invariant the
|
|
83
|
+
* UTF-8 reader relies on.
|
|
79
84
|
*/
|
|
80
|
-
function
|
|
85
|
+
function acdbDecodeTextSpan(bytes, start, end, nonAscii, decoder) {
|
|
81
86
|
if (!nonAscii) {
|
|
82
87
|
return acdbDecodeAsciiSpan(bytes, start, end);
|
|
83
88
|
}
|
|
84
|
-
return
|
|
89
|
+
return decoder.decode(bytes.subarray(start, end));
|
|
85
90
|
}
|
|
86
91
|
/**
|
|
87
92
|
* Parses an integer group code straight from the bytes of a code line,
|
|
@@ -90,9 +95,9 @@ function acdbDecodeUtf8Span(bytes, start, end, nonAscii) {
|
|
|
90
95
|
* Returns NaN for blank or malformed code lines. DXF group code lines are
|
|
91
96
|
* pure ASCII; the rare non-ASCII line falls back to `Number` + `isFinite`.
|
|
92
97
|
*/
|
|
93
|
-
function acdbReadDxfCodeFromBytes(bytes, start, end, nonAscii) {
|
|
98
|
+
function acdbReadDxfCodeFromBytes(bytes, start, end, nonAscii, decoder) {
|
|
94
99
|
if (nonAscii) {
|
|
95
|
-
var n = Number(
|
|
100
|
+
var n = Number(acdbDecodeTextSpan(bytes, start, end, true, decoder).trim());
|
|
96
101
|
return Number.isFinite(n) ? n : NaN;
|
|
97
102
|
}
|
|
98
103
|
var i = start;
|
|
@@ -364,13 +369,13 @@ function acdbDecodeHexBinaryText(text, start, end) {
|
|
|
364
369
|
/**
|
|
365
370
|
* Decodes a code-310 hex line from bytes, trimming ASCII whitespace.
|
|
366
371
|
*/
|
|
367
|
-
function acdbDecodeHexBinarySpan(bytes, start, end, nonAscii) {
|
|
372
|
+
function acdbDecodeHexBinarySpan(bytes, start, end, nonAscii, decoder) {
|
|
368
373
|
while (start < end && acdbIsAsciiWhitespace(bytes[start]))
|
|
369
374
|
start++;
|
|
370
375
|
while (end > start && acdbIsAsciiWhitespace(bytes[end - 1]))
|
|
371
376
|
end--;
|
|
372
377
|
if (nonAscii) {
|
|
373
|
-
var text =
|
|
378
|
+
var text = acdbDecodeTextSpan(bytes, start, end, true, decoder);
|
|
374
379
|
return acdbDecodeHexBinaryText(text, 0, text.length);
|
|
375
380
|
}
|
|
376
381
|
var byteLength = (end - start) >>> 1;
|
|
@@ -388,7 +393,7 @@ function acdbDecodeHexBinarySpan(bytes, start, end, nonAscii) {
|
|
|
388
393
|
* nonAscii is the flag produced by the line scanner for this exact span; it
|
|
389
394
|
* is forwarded to every decode helper so no helper has to re-scan the bytes.
|
|
390
395
|
*/
|
|
391
|
-
function parseAsciiValueSpan(code, bytes, start, end, nonAscii) {
|
|
396
|
+
function parseAsciiValueSpan(code, bytes, start, end, nonAscii, decoder) {
|
|
392
397
|
var type = acdbDxfValueType(code);
|
|
393
398
|
if (type === 'comment')
|
|
394
399
|
return null;
|
|
@@ -397,12 +402,12 @@ function parseAsciiValueSpan(code, bytes, start, end, nonAscii) {
|
|
|
397
402
|
return {
|
|
398
403
|
code: code,
|
|
399
404
|
type: type,
|
|
400
|
-
value:
|
|
405
|
+
value: acdbDecodeTextSpan(bytes, start, end, nonAscii, decoder)
|
|
401
406
|
};
|
|
402
407
|
case 'int': {
|
|
403
408
|
var fast = acdbParseIntSpan(bytes, start, end);
|
|
404
409
|
var n = fast === undefined
|
|
405
|
-
? parseInt(
|
|
410
|
+
? parseInt(acdbDecodeTextSpan(bytes, start, end, nonAscii, decoder), 10)
|
|
406
411
|
: fast;
|
|
407
412
|
return { code: code, type: type, value: Number.isFinite(n) ? n : 0 };
|
|
408
413
|
}
|
|
@@ -410,7 +415,7 @@ function parseAsciiValueSpan(code, bytes, start, end, nonAscii) {
|
|
|
410
415
|
var fast = acdbParseLongSpan(bytes, start, end);
|
|
411
416
|
if (fast !== undefined)
|
|
412
417
|
return { code: code, type: type, value: fast };
|
|
413
|
-
var raw =
|
|
418
|
+
var raw = acdbDecodeTextSpan(bytes, start, end, nonAscii, decoder);
|
|
414
419
|
var n = Number(raw);
|
|
415
420
|
if (Number.isSafeInteger(n))
|
|
416
421
|
return { code: code, type: type, value: n };
|
|
@@ -424,7 +429,7 @@ function parseAsciiValueSpan(code, bytes, start, end, nonAscii) {
|
|
|
424
429
|
case 'double': {
|
|
425
430
|
var fast = acdbParseDoubleSpan(bytes, start, end);
|
|
426
431
|
var n = fast === undefined
|
|
427
|
-
? Number(
|
|
432
|
+
? Number(acdbDecodeTextSpan(bytes, start, end, nonAscii, decoder))
|
|
428
433
|
: fast;
|
|
429
434
|
return { code: code, type: type, value: Number.isFinite(n) ? n : 0 };
|
|
430
435
|
}
|
|
@@ -432,11 +437,11 @@ function parseAsciiValueSpan(code, bytes, start, end, nonAscii) {
|
|
|
432
437
|
var fast = acdbDxfRawBoolIsTrue(bytes, start, end, nonAscii);
|
|
433
438
|
if (fast !== undefined)
|
|
434
439
|
return { code: code, type: type, value: fast };
|
|
435
|
-
var trimmed =
|
|
440
|
+
var trimmed = acdbDecodeTextSpan(bytes, start, end, nonAscii, decoder).trim();
|
|
436
441
|
return { code: code, type: type, value: trimmed !== '' && trimmed !== '0' };
|
|
437
442
|
}
|
|
438
443
|
case 'handle': {
|
|
439
|
-
var value =
|
|
444
|
+
var value = acdbDecodeTextSpan(bytes, start, end, nonAscii, decoder);
|
|
440
445
|
if (value.length === 0)
|
|
441
446
|
return { code: code, type: type, value: value };
|
|
442
447
|
var first = value.charCodeAt(0);
|
|
@@ -450,19 +455,32 @@ function parseAsciiValueSpan(code, bytes, start, end, nonAscii) {
|
|
|
450
455
|
return {
|
|
451
456
|
code: code,
|
|
452
457
|
type: type,
|
|
453
|
-
value: acdbDecodeHexBinarySpan(bytes, start, end, nonAscii)
|
|
458
|
+
value: acdbDecodeHexBinarySpan(bytes, start, end, nonAscii, decoder)
|
|
454
459
|
};
|
|
455
460
|
default:
|
|
456
461
|
return null;
|
|
457
462
|
}
|
|
458
463
|
}
|
|
464
|
+
function isUtf8Encoding(encoding) {
|
|
465
|
+
var e = encoding.toLowerCase().replace(/_/g, '-');
|
|
466
|
+
return e === 'utf-8' || e === 'utf8' || e === 'unicode-1-1-utf-8';
|
|
467
|
+
}
|
|
459
468
|
/**
|
|
460
|
-
* ASCII
|
|
469
|
+
* ASCII pair reader over raw bytes, decoding non-ASCII spans with `encoding`.
|
|
461
470
|
*
|
|
462
|
-
* Line breaks are single bytes
|
|
463
|
-
* 0x0A
|
|
471
|
+
* Line breaks are single bytes. For UTF-8, continuation bytes never equal
|
|
472
|
+
* `0x0A`/`0x0D`. The same holds for the legacy DXF code pages we support
|
|
473
|
+
* (windows-125x, euc-kr/cp949, gbk, shift-jis, big5, …): their trail-byte
|
|
474
|
+
* ranges exclude CR/LF, so non-ASCII text values never straddle line
|
|
475
|
+
* boundaries and each line may be decoded independently.
|
|
476
|
+
*
|
|
477
|
+
* @param encoding - WHATWG / `TextDecoder` label. Defaults to UTF-8.
|
|
464
478
|
*/
|
|
465
|
-
function
|
|
479
|
+
function acdbMakeByteAsciiDxfPairReader(bytes, encoding) {
|
|
480
|
+
if (encoding === void 0) { encoding = 'utf-8'; }
|
|
481
|
+
var decoder = isUtf8Encoding(encoding)
|
|
482
|
+
? UTF8_DECODER
|
|
483
|
+
: new TextDecoder(encoding);
|
|
466
484
|
var pos = bytes.length >= 3 &&
|
|
467
485
|
bytes[0] === 0xef &&
|
|
468
486
|
bytes[1] === 0xbb &&
|
|
@@ -506,7 +524,7 @@ function acdbMakeUtf8DxfPairReader(bytes) {
|
|
|
506
524
|
var codeSpan = readLineSpan();
|
|
507
525
|
if (codeSpan === undefined)
|
|
508
526
|
return undefined;
|
|
509
|
-
var code = acdbReadDxfCodeFromBytes(bytes, codeSpan.start, codeSpan.end, codeSpan.nonAscii);
|
|
527
|
+
var code = acdbReadDxfCodeFromBytes(bytes, codeSpan.start, codeSpan.end, codeSpan.nonAscii, decoder);
|
|
510
528
|
if (Number.isNaN(code))
|
|
511
529
|
continue;
|
|
512
530
|
if (code === 999) {
|
|
@@ -517,7 +535,7 @@ function acdbMakeUtf8DxfPairReader(bytes) {
|
|
|
517
535
|
var valueSpan = readLineSpan();
|
|
518
536
|
if (valueSpan === undefined)
|
|
519
537
|
return undefined;
|
|
520
|
-
var pair = parseAsciiValueSpan(code, bytes, valueSpan.start, valueSpan.end, valueSpan.nonAscii);
|
|
538
|
+
var pair = parseAsciiValueSpan(code, bytes, valueSpan.start, valueSpan.end, valueSpan.nonAscii, decoder);
|
|
521
539
|
if (pair)
|
|
522
540
|
return pair;
|
|
523
541
|
}
|
|
@@ -545,6 +563,95 @@ function acdbMakeUtf8DxfPairReader(bytes) {
|
|
|
545
563
|
}
|
|
546
564
|
};
|
|
547
565
|
}
|
|
566
|
+
function acdbMakeUtf8DxfPairReader(bytes) {
|
|
567
|
+
return acdbMakeByteAsciiDxfPairReader(bytes, 'utf-8');
|
|
568
|
+
}
|
|
569
|
+
/**
|
|
570
|
+
* Resolve a `$DWGCODEPAGE` header value to a `TextDecoder` label.
|
|
571
|
+
* Unknown names yield `null` so callers keep their UTF-8 fallback.
|
|
572
|
+
*/
|
|
573
|
+
function acdbResolveDwgCodePageEncoding(value) {
|
|
574
|
+
var _a;
|
|
575
|
+
if (!value)
|
|
576
|
+
return null;
|
|
577
|
+
var codePage = AcDbCodePage[value];
|
|
578
|
+
if (codePage === undefined)
|
|
579
|
+
return null;
|
|
580
|
+
return (_a = acdbDwgCodePageToEncoding(codePage)) !== null && _a !== void 0 ? _a : null;
|
|
581
|
+
}
|
|
582
|
+
/**
|
|
583
|
+
* Peek `$ACADVER` / `$DWGCODEPAGE` from a binary DXF HEADER section.
|
|
584
|
+
*
|
|
585
|
+
* Header keywords and code-page names are ASCII, so the temporary reader
|
|
586
|
+
* uses UTF-8 regardless of the drawing's text encoding.
|
|
587
|
+
*/
|
|
588
|
+
function acdbPeekBinaryDxfHeaderInfo(data, legacyR12) {
|
|
589
|
+
var reader = acdbMakeBinaryDxfPairReader(data, {
|
|
590
|
+
encoding: 'utf-8',
|
|
591
|
+
legacyR12: legacyR12
|
|
592
|
+
});
|
|
593
|
+
var version = null;
|
|
594
|
+
var encoding = null;
|
|
595
|
+
var inHeader = false;
|
|
596
|
+
var pendingVar = null;
|
|
597
|
+
for (var pair = reader.next(); pair; pair = reader.next()) {
|
|
598
|
+
if (pair.code === 0 && pair.type === 'string') {
|
|
599
|
+
var name_1 = pair.value;
|
|
600
|
+
if (name_1 === 'SECTION') {
|
|
601
|
+
var section = reader.next();
|
|
602
|
+
if ((section === null || section === void 0 ? void 0 : section.code) === 2 &&
|
|
603
|
+
section.type === 'string' &&
|
|
604
|
+
section.value === 'HEADER') {
|
|
605
|
+
inHeader = true;
|
|
606
|
+
}
|
|
607
|
+
pendingVar = null;
|
|
608
|
+
continue;
|
|
609
|
+
}
|
|
610
|
+
if (name_1 === 'ENDSEC' && inHeader) {
|
|
611
|
+
return { version: version, encoding: encoding };
|
|
612
|
+
}
|
|
613
|
+
pendingVar = null;
|
|
614
|
+
continue;
|
|
615
|
+
}
|
|
616
|
+
if (!inHeader)
|
|
617
|
+
continue;
|
|
618
|
+
if (pair.code === 9 && pair.type === 'string') {
|
|
619
|
+
pendingVar = pair.value;
|
|
620
|
+
continue;
|
|
621
|
+
}
|
|
622
|
+
if (pendingVar === '$ACADVER' && pair.code === 1 && pair.type === 'string') {
|
|
623
|
+
try {
|
|
624
|
+
version = new AcDbDwgVersion(pair.value);
|
|
625
|
+
}
|
|
626
|
+
catch (_a) {
|
|
627
|
+
// Unrecognized spelling: leave version null.
|
|
628
|
+
}
|
|
629
|
+
}
|
|
630
|
+
else if (pendingVar === '$DWGCODEPAGE' &&
|
|
631
|
+
pair.code === 3 &&
|
|
632
|
+
pair.type === 'string') {
|
|
633
|
+
encoding = acdbResolveDwgCodePageEncoding(pair.value);
|
|
634
|
+
}
|
|
635
|
+
if (pendingVar !== null) {
|
|
636
|
+
pendingVar = null;
|
|
637
|
+
if (version && encoding)
|
|
638
|
+
return { version: version, encoding: encoding };
|
|
639
|
+
}
|
|
640
|
+
}
|
|
641
|
+
return { version: version, encoding: encoding };
|
|
642
|
+
}
|
|
643
|
+
/**
|
|
644
|
+
* Prefer a pre-2007 `$DWGCODEPAGE` when the header declares both a legacy
|
|
645
|
+
* version and a known code page; otherwise return null (caller keeps UTF-8).
|
|
646
|
+
*/
|
|
647
|
+
function acdbLegacyEncodingFromHeader(info) {
|
|
648
|
+
if (info.version &&
|
|
649
|
+
!info.version.capabilities.supportsUtf8CodePage &&
|
|
650
|
+
info.encoding) {
|
|
651
|
+
return info.encoding;
|
|
652
|
+
}
|
|
653
|
+
return null;
|
|
654
|
+
}
|
|
548
655
|
/**
|
|
549
656
|
* Peek `$ACADVER` / `$DWGCODEPAGE` from the HEADER section without decoding
|
|
550
657
|
* the whole file. Uses 64 KiB UTF-8 chunks (same strategy as AcDbDxfParser).
|
|
@@ -580,10 +687,8 @@ export function acdbPeekDxfHeaderInfo(buffer) {
|
|
|
580
687
|
}
|
|
581
688
|
else if (inHeader && line === '$DWGCODEPAGE') {
|
|
582
689
|
var value = (_d = lines[i + 2]) === null || _d === void 0 ? void 0 : _d.trim();
|
|
583
|
-
if (value)
|
|
584
|
-
|
|
585
|
-
encoding = acdbDwgCodePageToEncoding(codePage);
|
|
586
|
-
}
|
|
690
|
+
if (value)
|
|
691
|
+
encoding = acdbResolveDwgCodePageEncoding(value);
|
|
587
692
|
}
|
|
588
693
|
if (version && encoding)
|
|
589
694
|
return { version: version, encoding: encoding };
|
|
@@ -714,24 +819,11 @@ export function acdbMakeAsciiDxfPairReader(text) {
|
|
|
714
819
|
}
|
|
715
820
|
};
|
|
716
821
|
}
|
|
717
|
-
function isUtf8Encoding(encoding) {
|
|
718
|
-
var e = encoding.toLowerCase().replace(/_/g, '-');
|
|
719
|
-
return e === 'utf-8' || e === 'utf8' || e === 'unicode-1-1-utf-8';
|
|
720
|
-
}
|
|
721
822
|
/**
|
|
722
|
-
*
|
|
723
|
-
*
|
|
724
|
-
*
|
|
725
|
-
*
|
|
726
|
-
* a retained value slice do not pin much memory.
|
|
727
|
-
*/
|
|
728
|
-
/**
|
|
729
|
-
* ASCII pair reader that decodes UTF-8 bytes one line-aligned window at a time,
|
|
730
|
-
* instead of allocating a full-file decoded string (peak memory ≈ input bytes
|
|
731
|
-
* plus one window).
|
|
732
|
-
*
|
|
733
|
-
* Non-UTF-8 code pages still go through {@link acdbMakeAsciiDxfPairReader}
|
|
734
|
-
* after a full `TextDecoder` pass.
|
|
823
|
+
* ASCII pair reader that decodes UTF-8 bytes one line at a time, instead of
|
|
824
|
+
* allocating a full-file decoded string (peak memory ≈ input bytes plus one
|
|
825
|
+
* line). Legacy code pages use the same span path via
|
|
826
|
+
* {@link acdbCreateDxfPairReader}.
|
|
735
827
|
*/
|
|
736
828
|
export function acdbMakeUtf8AsciiDxfPairReader(bytes) {
|
|
737
829
|
return acdbMakeUtf8DxfPairReader(bytes);
|
|
@@ -753,6 +845,7 @@ export function acdbMakeBinaryDxfPairReader(data, options) {
|
|
|
753
845
|
if (options === void 0) { options = {}; }
|
|
754
846
|
var encoding = (_a = options.encoding) !== null && _a !== void 0 ? _a : 'utf-8';
|
|
755
847
|
var legacyR12 = (_b = options.legacyR12) !== null && _b !== void 0 ? _b : false;
|
|
848
|
+
var decoder = new TextDecoder(encoding);
|
|
756
849
|
var PREFIX = 22;
|
|
757
850
|
var view = new DataView(data.buffer, data.byteOffset, data.byteLength);
|
|
758
851
|
var offset = data.length >= PREFIX ? PREFIX : data.length;
|
|
@@ -791,7 +884,7 @@ export function acdbMakeBinaryDxfPairReader(data, options) {
|
|
|
791
884
|
return undefined;
|
|
792
885
|
var bytes = data.subarray(start, offset);
|
|
793
886
|
offset += 1;
|
|
794
|
-
return
|
|
887
|
+
return decoder.decode(bytes);
|
|
795
888
|
}
|
|
796
889
|
function readInt16() {
|
|
797
890
|
if (offset + 2 > data.length)
|
|
@@ -998,18 +1091,26 @@ function acdbIsValidUtf8(bytes) {
|
|
|
998
1091
|
* ASCII path encoding strategy (hybrid of byte-trust and header sniffing):
|
|
999
1092
|
*
|
|
1000
1093
|
* 1. An explicit `options.encoding` always wins — the caller asserts the
|
|
1001
|
-
* encoding, no sniffing contradicts it.
|
|
1094
|
+
* encoding, no sniffing contradicts it. Non-UTF-8 overrides decode
|
|
1095
|
+
* line-by-line (never a single full-file string).
|
|
1002
1096
|
* 2. Otherwise the bytes are strictly validated as UTF-8. Valid UTF-8
|
|
1003
1097
|
* (including pure ASCII) decodes correctly regardless of any stale
|
|
1004
1098
|
* `$DWGCODEPAGE`, so the header is never consulted — modern files skip
|
|
1005
1099
|
* the header pre-scan entirely and stream straight from bytes.
|
|
1006
1100
|
* 3. Invalid UTF-8 means the file is not a spec-conformant modern DXF. The
|
|
1007
1101
|
* HEADER is then peeked for `$ACADVER`/`$DWGCODEPAGE`: pre-2007 drawings
|
|
1008
|
-
* with a declared code page decode through it (matching
|
|
1009
|
-
*
|
|
1010
|
-
*
|
|
1102
|
+
* with a declared code page decode through it span-by-span (matching
|
|
1103
|
+
* AutoCAD, without materializing one giant string), while R2007+ or
|
|
1104
|
+
* headerless files fall back to UTF-8 (replacement chars mark the
|
|
1105
|
+
* broken bytes).
|
|
1106
|
+
*
|
|
1107
|
+
* Binary path: when `options.encoding` is omitted, the HEADER is peeked for
|
|
1108
|
+
* a pre-2007 `$DWGCODEPAGE` and that code page is used for null-terminated
|
|
1109
|
+
* string values (same AutoCAD rule as ASCII). R2007+ / missing header keep
|
|
1110
|
+
* UTF-8.
|
|
1011
1111
|
*/
|
|
1012
1112
|
export function acdbCreateDxfPairReader(data, options) {
|
|
1113
|
+
var _a;
|
|
1013
1114
|
if (options === void 0) { options = {}; }
|
|
1014
1115
|
var bytes = data instanceof Uint8Array ? data : new Uint8Array(data);
|
|
1015
1116
|
var overrideEncoding = options.encoding
|
|
@@ -1018,35 +1119,38 @@ export function acdbCreateDxfPairReader(data, options) {
|
|
|
1018
1119
|
if (acdbIsBinaryDxf(bytes)) {
|
|
1019
1120
|
var encoding = overrideEncoding;
|
|
1020
1121
|
var legacyR12 = options.legacyR12;
|
|
1021
|
-
if (
|
|
1022
|
-
|
|
1023
|
-
|
|
1024
|
-
|
|
1025
|
-
|
|
1026
|
-
|
|
1027
|
-
|
|
1028
|
-
|
|
1029
|
-
|
|
1030
|
-
|
|
1031
|
-
|
|
1032
|
-
|
|
1033
|
-
else if (b0 === 0 && b1 === 0 && b2 === 0x53 /* 'S' */) {
|
|
1034
|
-
legacyR12 = false;
|
|
1035
|
-
}
|
|
1036
|
-
else {
|
|
1037
|
-
legacyR12 = false;
|
|
1038
|
-
}
|
|
1122
|
+
if (legacyR12 == null) {
|
|
1123
|
+
// After the 22-byte magic: R12 uses 1-byte codes (`0,'S'`), modern
|
|
1124
|
+
// uses 2-byte LE codes (`0,0,'S'`) for the first SECTION marker.
|
|
1125
|
+
var PREFIX = 22;
|
|
1126
|
+
var b0 = bytes[PREFIX];
|
|
1127
|
+
var b1 = bytes[PREFIX + 1];
|
|
1128
|
+
var b2 = bytes[PREFIX + 2];
|
|
1129
|
+
if (b0 === 0 && b1 === 0x53 /* 'S' */) {
|
|
1130
|
+
legacyR12 = true;
|
|
1131
|
+
}
|
|
1132
|
+
else if (b0 === 0 && b1 === 0 && b2 === 0x53 /* 'S' */) {
|
|
1133
|
+
legacyR12 = false;
|
|
1039
1134
|
}
|
|
1135
|
+
else {
|
|
1136
|
+
legacyR12 = false;
|
|
1137
|
+
}
|
|
1138
|
+
}
|
|
1139
|
+
if (encoding == null) {
|
|
1140
|
+
try {
|
|
1141
|
+
encoding =
|
|
1142
|
+
(_a = acdbLegacyEncodingFromHeader(acdbPeekBinaryDxfHeaderInfo(bytes, legacyR12))) !== null && _a !== void 0 ? _a : undefined;
|
|
1143
|
+
}
|
|
1144
|
+
catch (_b) {
|
|
1145
|
+
// Unrecognized $ACADVER: stay on UTF-8.
|
|
1146
|
+
}
|
|
1147
|
+
encoding = encoding !== null && encoding !== void 0 ? encoding : 'utf-8';
|
|
1040
1148
|
}
|
|
1041
1149
|
return acdbMakeBinaryDxfPairReader(bytes, { encoding: encoding, legacyR12: legacyR12 });
|
|
1042
1150
|
}
|
|
1043
1151
|
// 1. Explicit override wins: the caller asserts the encoding.
|
|
1044
1152
|
if (overrideEncoding) {
|
|
1045
|
-
|
|
1046
|
-
return acdbMakeUtf8DxfPairReader(bytes);
|
|
1047
|
-
}
|
|
1048
|
-
var text = new TextDecoder(overrideEncoding).decode(bytes);
|
|
1049
|
-
return acdbMakeAsciiDxfPairReader(text);
|
|
1153
|
+
return acdbMakeByteAsciiDxfPairReader(bytes, overrideEncoding);
|
|
1050
1154
|
}
|
|
1051
1155
|
// 2. Trust the bytes first: valid UTF-8 (including pure ASCII) is decoded
|
|
1052
1156
|
// correctly no matter what a stale $DWGCODEPAGE claims, so modern files
|
|
@@ -1055,27 +1159,23 @@ export function acdbCreateDxfPairReader(data, options) {
|
|
|
1055
1159
|
return acdbMakeUtf8DxfPairReader(bytes);
|
|
1056
1160
|
}
|
|
1057
1161
|
// 3. Invalid UTF-8: fall back to the pre-2007 $DWGCODEPAGE when one is
|
|
1058
|
-
// declared (genuine legacy ANSI content).
|
|
1059
|
-
//
|
|
1162
|
+
// declared (genuine legacy ANSI content). Decode span-by-span so large
|
|
1163
|
+
// drawings never hit the V8 max-string-length ceiling. R2007+ or
|
|
1164
|
+
// headerless files stay on UTF-8, where replacement characters mark
|
|
1165
|
+
// the broken bytes.
|
|
1060
1166
|
var legacyEncoding = null;
|
|
1061
1167
|
try {
|
|
1062
1168
|
var buffer = bytes.byteOffset === 0 && bytes.byteLength === bytes.buffer.byteLength
|
|
1063
1169
|
? bytes.buffer
|
|
1064
1170
|
: bytes.buffer.slice(bytes.byteOffset, bytes.byteOffset + bytes.byteLength);
|
|
1065
|
-
|
|
1066
|
-
if (info.version &&
|
|
1067
|
-
!info.version.capabilities.supportsUtf8CodePage &&
|
|
1068
|
-
info.encoding) {
|
|
1069
|
-
legacyEncoding = info.encoding;
|
|
1070
|
-
}
|
|
1171
|
+
legacyEncoding = acdbLegacyEncodingFromHeader(acdbPeekDxfHeaderInfo(buffer));
|
|
1071
1172
|
}
|
|
1072
|
-
catch (
|
|
1173
|
+
catch (_c) {
|
|
1073
1174
|
// Unrecognized $ACADVER spelling: treat as version-less and stay UTF-8
|
|
1074
1175
|
// instead of failing the whole read.
|
|
1075
1176
|
}
|
|
1076
1177
|
if (legacyEncoding) {
|
|
1077
|
-
|
|
1078
|
-
return acdbMakeAsciiDxfPairReader(text);
|
|
1178
|
+
return acdbMakeByteAsciiDxfPairReader(bytes, legacyEncoding);
|
|
1079
1179
|
}
|
|
1080
1180
|
return acdbMakeUtf8DxfPairReader(bytes);
|
|
1081
1181
|
}
|