@xberg-io/xberg-wasm 1.2.8 → 1.2.9
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json
CHANGED
package/pkg/web/package.json
CHANGED
package/pkg/web/xberg_wasm.d.ts
CHANGED
|
@@ -2527,17 +2527,28 @@ export class WasmExtractionConfig {
|
|
|
2527
2527
|
* image I/O and processing when results won't be used.
|
|
2528
2528
|
*/
|
|
2529
2529
|
needsImageProcessing(): boolean;
|
|
2530
|
-
constructor(mimeDetectionPolicy?: WasmMimeDetectionPolicy | null, useCache?: boolean | null, enableQualityProcessing?: boolean | null, forceOcr?: boolean | null, ocrStrategy?: any | null, disableOcr?: boolean | null, resultFormat?: WasmResultFormat | null, outputFormat?: any | null, escapeMarkdown?: boolean | null, tableAnchors?: boolean | null, jupyterCellRendering?: WasmJupyterCellRendering | null, applyNotebookCellTags?: boolean | null, useLayoutForMarkdown?: boolean | null, includeDocumentStructure?: boolean | null, url?: WasmUrlExtractionConfig | null, maxArchiveDepth?: number | null, ocr?: WasmOcrConfig | null, forceOcrPages?: Uint32Array | null, chunking?: WasmChunkingConfig | null, contentFilter?: WasmContentFilterConfig | null, images?: WasmImageExtractionConfig | null, pdfOptions?: WasmPdfConfig | null, tokenReduction?: WasmTokenReductionOptions | null, languageDetection?: WasmLanguageDetectionConfig | null, pages?: WasmPageConfig | null, keywords?: WasmKeywordConfig | null, postprocessor?: WasmPostProcessorConfig | null, htmlOptions?: WasmConversionOptions | null, htmlOutput?: WasmHtmlOutputConfig | null, extractionTimeoutSecs?: bigint | null, maxConcurrentExtractions?: number | null, securityLimits?: WasmSecurityLimits | null, maxEmbeddedFileBytes?: bigint | null, layout?: WasmLayoutDetectionConfig | null, transcription?: WasmTranscriptionConfig | null, acceleration?: WasmAccelerationConfig | null, cacheNamespace?: string | null, cacheTtlSecs?: bigint | null, email?: WasmEmailConfig | null, csv?: WasmCsvConfig | null, geojson?: WasmGeoJsonExtractionConfig | null, concurrency?: WasmConcurrencyConfig | null, structuredExtraction?: WasmStructuredExtractionConfig | null, ner?: WasmNerConfig | null, redaction?: WasmRedactionConfig | null, summarization?: WasmSummarizationConfig | null, translation?: WasmTranslationConfig | null, pageClassification?: WasmPageClassificationConfig | null, chunkClassification?: WasmChunkClassificationConfig | null, captioning?: WasmCaptioningConfig | null, qrCodes?: boolean | null);
|
|
2530
|
+
constructor(mimeDetectionPolicy?: WasmMimeDetectionPolicy | null, useCache?: boolean | null, enableQualityProcessing?: boolean | null, forceOcr?: boolean | null, ocrStrategy?: any | null, disableOcr?: boolean | null, resultFormat?: WasmResultFormat | null, outputFormat?: any | null, escapeMarkdown?: boolean | null, tableAnchors?: boolean | null, jupyterCellRendering?: WasmJupyterCellRendering | null, applyNotebookCellTags?: boolean | null, useLayoutForMarkdown?: boolean | null, includeDocumentStructure?: boolean | null, url?: WasmUrlExtractionConfig | null, maxArchiveDepth?: number | null, ocr?: WasmOcrConfig | null, forceOcrPages?: Uint32Array | null, ocrNearEmptyFallback?: boolean | null, ocrScannedPageQualityGate?: boolean | null, ocrEmbeddedImages?: boolean | null, chunking?: WasmChunkingConfig | null, contentFilter?: WasmContentFilterConfig | null, images?: WasmImageExtractionConfig | null, pdfOptions?: WasmPdfConfig | null, tokenReduction?: WasmTokenReductionOptions | null, languageDetection?: WasmLanguageDetectionConfig | null, pages?: WasmPageConfig | null, keywords?: WasmKeywordConfig | null, postprocessor?: WasmPostProcessorConfig | null, htmlOptions?: WasmConversionOptions | null, htmlOutput?: WasmHtmlOutputConfig | null, extractionTimeoutSecs?: bigint | null, maxConcurrentExtractions?: number | null, securityLimits?: WasmSecurityLimits | null, maxEmbeddedFileBytes?: bigint | null, layout?: WasmLayoutDetectionConfig | null, transcription?: WasmTranscriptionConfig | null, acceleration?: WasmAccelerationConfig | null, cacheNamespace?: string | null, cacheTtlSecs?: bigint | null, email?: WasmEmailConfig | null, csv?: WasmCsvConfig | null, geojson?: WasmGeoJsonExtractionConfig | null, concurrency?: WasmConcurrencyConfig | null, structuredExtraction?: WasmStructuredExtractionConfig | null, ner?: WasmNerConfig | null, redaction?: WasmRedactionConfig | null, summarization?: WasmSummarizationConfig | null, translation?: WasmTranslationConfig | null, pageClassification?: WasmPageClassificationConfig | null, chunkClassification?: WasmChunkClassificationConfig | null, captioning?: WasmCaptioningConfig | null, qrCodes?: boolean | null);
|
|
2531
2531
|
/**
|
|
2532
2532
|
* Whether embedded images get OCR'd.
|
|
2533
2533
|
*
|
|
2534
|
-
* This
|
|
2535
|
-
* `image_ocr.process_images_with_ocr`
|
|
2536
|
-
*
|
|
2537
|
-
*
|
|
2538
|
-
*
|
|
2539
|
-
*
|
|
2540
|
-
*
|
|
2534
|
+
* This is THE condition -- `core/pipeline/mod.rs` and
|
|
2535
|
+
* `extraction.image_ocr.process_images_with_ocr` both call this method rather than
|
|
2536
|
+
* re-deriving it, because when two copies of it drifted apart a container extractor
|
|
2537
|
+
* asked `needs_image_data` and was told no, so it attached an image with an empty
|
|
2538
|
+
* buffer, and the OCR path then ran on those zero bytes and reported `Could not
|
|
2539
|
+
* determine image format` (GH#1662). A re-derived copy anywhere reopens that defect;
|
|
2540
|
+
* `embedded_image_ocr_gate_has_no_second_copy` in `core/pipeline/tests.rs` fails if
|
|
2541
|
+
* one appears.
|
|
2542
|
+
*
|
|
2543
|
+
* `Self.ocr_embedded_images` is the caller's explicit answer to the OCR half;
|
|
2544
|
+
* `None` derives it from whether an `ocr` block is present, which is what this
|
|
2545
|
+
* condition was before GH#1752 gave the behaviour a setting of its own.
|
|
2546
|
+
*
|
|
2547
|
+
* `disable_ocr` still wins regardless: it is documented as skipping OCR "for all
|
|
2548
|
+
* document types", and `ocr_embedded_images` (like the plain presence of an `ocr`
|
|
2549
|
+
* block before it) is an AUTOMATIC trigger, not an explicit request like `force_ocr` --
|
|
2550
|
+
* see `Self.effective_disable_ocr`'s callers elsewhere (`needs_image_processing`,
|
|
2551
|
+
* `extractors/image.rs`, `engine/extract_impl.rs`) for the same precedent.
|
|
2541
2552
|
*/
|
|
2542
2553
|
runsOcrOnEmbeddedImages(): boolean;
|
|
2543
2554
|
/**
|
|
@@ -2639,6 +2650,12 @@ export class WasmExtractionConfig {
|
|
|
2639
2650
|
set mimeDetectionPolicy(value: WasmMimeDetectionPolicy);
|
|
2640
2651
|
get ner(): WasmNerConfig | undefined;
|
|
2641
2652
|
set ner(value: WasmNerConfig | null | undefined);
|
|
2653
|
+
get ocrEmbeddedImages(): boolean | undefined;
|
|
2654
|
+
set ocrEmbeddedImages(value: boolean | null | undefined);
|
|
2655
|
+
get ocrNearEmptyFallback(): boolean | undefined;
|
|
2656
|
+
set ocrNearEmptyFallback(value: boolean | null | undefined);
|
|
2657
|
+
get ocrScannedPageQualityGate(): boolean | undefined;
|
|
2658
|
+
set ocrScannedPageQualityGate(value: boolean | null | undefined);
|
|
2642
2659
|
ocrStrategy: any;
|
|
2643
2660
|
get ocr(): WasmOcrConfig | undefined;
|
|
2644
2661
|
set ocr(value: WasmOcrConfig | null | undefined);
|
|
@@ -9018,8 +9035,11 @@ export interface InitOutput {
|
|
|
9018
9035
|
readonly wasmextractionconfig_needsImageData: (a: number) => number;
|
|
9019
9036
|
readonly wasmextractionconfig_needsImageProcessing: (a: number) => number;
|
|
9020
9037
|
readonly wasmextractionconfig_ner: (a: number) => number;
|
|
9021
|
-
readonly wasmextractionconfig_new: (a: number, b: number, c: number, d: number, e: number, f: number, g: number, h: number, i: number, j: number, k: number, l: number, m: number, n: number, o: number, p: number, q: number, r: number, s: number, t: number, u: number, v: number, w: number, x: number, y: number, z: number, a1: number, b1: number, c1: number, d1: number, e1: number, f1:
|
|
9038
|
+
readonly wasmextractionconfig_new: (a: number, b: number, c: number, d: number, e: number, f: number, g: number, h: number, i: number, j: number, k: number, l: number, m: number, n: number, o: number, p: number, q: number, r: number, s: number, t: number, u: number, v: number, w: number, x: number, y: number, z: number, a1: number, b1: number, c1: number, d1: number, e1: number, f1: number, g1: number, h1: number, i1: bigint, j1: number, k1: number, l1: number, m1: bigint, n1: number, o1: number, p1: number, q1: number, r1: number, s1: number, t1: bigint, u1: number, v1: number, w1: number, x1: number, y1: number, z1: number, a2: number, b2: number, c2: number, d2: number, e2: number, f2: number, g2: number) => number;
|
|
9022
9039
|
readonly wasmextractionconfig_ocr: (a: number) => number;
|
|
9040
|
+
readonly wasmextractionconfig_ocrEmbeddedImages: (a: number) => number;
|
|
9041
|
+
readonly wasmextractionconfig_ocrNearEmptyFallback: (a: number) => number;
|
|
9042
|
+
readonly wasmextractionconfig_ocrScannedPageQualityGate: (a: number) => number;
|
|
9023
9043
|
readonly wasmextractionconfig_ocrStrategy: (a: number) => any;
|
|
9024
9044
|
readonly wasmextractionconfig_outputFormat: (a: number) => any;
|
|
9025
9045
|
readonly wasmextractionconfig_pageClassification: (a: number) => number;
|
|
@@ -9063,6 +9083,9 @@ export interface InitOutput {
|
|
|
9063
9083
|
readonly wasmextractionconfig_set_mimeDetectionPolicy: (a: number, b: number) => void;
|
|
9064
9084
|
readonly wasmextractionconfig_set_ner: (a: number, b: number) => void;
|
|
9065
9085
|
readonly wasmextractionconfig_set_ocr: (a: number, b: number) => void;
|
|
9086
|
+
readonly wasmextractionconfig_set_ocrEmbeddedImages: (a: number, b: number) => void;
|
|
9087
|
+
readonly wasmextractionconfig_set_ocrNearEmptyFallback: (a: number, b: number) => void;
|
|
9088
|
+
readonly wasmextractionconfig_set_ocrScannedPageQualityGate: (a: number, b: number) => void;
|
|
9066
9089
|
readonly wasmextractionconfig_set_ocrStrategy: (a: number, b: any) => void;
|
|
9067
9090
|
readonly wasmextractionconfig_set_outputFormat: (a: number, b: any) => void;
|
|
9068
9091
|
readonly wasmextractionconfig_set_pageClassification: (a: number, b: number) => void;
|
package/pkg/web/xberg_wasm.js
CHANGED
|
@@ -13946,6 +13946,9 @@ export class WasmExtractionConfig {
|
|
|
13946
13946
|
* @param {number | null} [maxArchiveDepth]
|
|
13947
13947
|
* @param {WasmOcrConfig | null} [ocr]
|
|
13948
13948
|
* @param {Uint32Array | null} [forceOcrPages]
|
|
13949
|
+
* @param {boolean | null} [ocrNearEmptyFallback]
|
|
13950
|
+
* @param {boolean | null} [ocrScannedPageQualityGate]
|
|
13951
|
+
* @param {boolean | null} [ocrEmbeddedImages]
|
|
13949
13952
|
* @param {WasmChunkingConfig | null} [chunking]
|
|
13950
13953
|
* @param {WasmContentFilterConfig | null} [contentFilter]
|
|
13951
13954
|
* @param {WasmImageExtractionConfig | null} [images]
|
|
@@ -13980,7 +13983,7 @@ export class WasmExtractionConfig {
|
|
|
13980
13983
|
* @param {WasmCaptioningConfig | null} [captioning]
|
|
13981
13984
|
* @param {boolean | null} [qrCodes]
|
|
13982
13985
|
*/
|
|
13983
|
-
constructor(mimeDetectionPolicy, useCache, enableQualityProcessing, forceOcr, ocrStrategy, disableOcr, resultFormat, outputFormat, escapeMarkdown, tableAnchors, jupyterCellRendering, applyNotebookCellTags, useLayoutForMarkdown, includeDocumentStructure, url, maxArchiveDepth, ocr, forceOcrPages, chunking, contentFilter, images, pdfOptions, tokenReduction, languageDetection, pages, keywords, postprocessor, htmlOptions, htmlOutput, extractionTimeoutSecs, maxConcurrentExtractions, securityLimits, maxEmbeddedFileBytes, layout, transcription, acceleration, cacheNamespace, cacheTtlSecs, email, csv, geojson, concurrency, structuredExtraction, ner, redaction, summarization, translation, pageClassification, chunkClassification, captioning, qrCodes) {
|
|
13986
|
+
constructor(mimeDetectionPolicy, useCache, enableQualityProcessing, forceOcr, ocrStrategy, disableOcr, resultFormat, outputFormat, escapeMarkdown, tableAnchors, jupyterCellRendering, applyNotebookCellTags, useLayoutForMarkdown, includeDocumentStructure, url, maxArchiveDepth, ocr, forceOcrPages, ocrNearEmptyFallback, ocrScannedPageQualityGate, ocrEmbeddedImages, chunking, contentFilter, images, pdfOptions, tokenReduction, languageDetection, pages, keywords, postprocessor, htmlOptions, htmlOutput, extractionTimeoutSecs, maxConcurrentExtractions, securityLimits, maxEmbeddedFileBytes, layout, transcription, acceleration, cacheNamespace, cacheTtlSecs, email, csv, geojson, concurrency, structuredExtraction, ner, redaction, summarization, translation, pageClassification, chunkClassification, captioning, qrCodes) {
|
|
13984
13987
|
let ptr0 = 0;
|
|
13985
13988
|
if (!isLikeNone(url)) {
|
|
13986
13989
|
_assertClass(url, WasmUrlExtractionConfig);
|
|
@@ -14130,11 +14133,32 @@ export class WasmExtractionConfig {
|
|
|
14130
14133
|
_assertClass(captioning, WasmCaptioningConfig);
|
|
14131
14134
|
ptr30 = captioning.__destroy_into_raw();
|
|
14132
14135
|
}
|
|
14133
|
-
const ret = wasm.wasmextractionconfig_new(isLikeNone(mimeDetectionPolicy) ? 3 : mimeDetectionPolicy, isLikeNone(useCache) ? 0xFFFFFF : useCache ? 1 : 0, isLikeNone(enableQualityProcessing) ? 0xFFFFFF : enableQualityProcessing ? 1 : 0, isLikeNone(forceOcr) ? 0xFFFFFF : forceOcr ? 1 : 0, isLikeNone(ocrStrategy) ? 0 : addToExternrefTable0(ocrStrategy), isLikeNone(disableOcr) ? 0xFFFFFF : disableOcr ? 1 : 0, isLikeNone(resultFormat) ? 2 : resultFormat, isLikeNone(outputFormat) ? 0 : addToExternrefTable0(outputFormat), isLikeNone(escapeMarkdown) ? 0xFFFFFF : escapeMarkdown ? 1 : 0, isLikeNone(tableAnchors) ? 0xFFFFFF : tableAnchors ? 1 : 0, isLikeNone(jupyterCellRendering) ? 3 : jupyterCellRendering, isLikeNone(applyNotebookCellTags) ? 0xFFFFFF : applyNotebookCellTags ? 1 : 0, isLikeNone(useLayoutForMarkdown) ? 0xFFFFFF : useLayoutForMarkdown ? 1 : 0, isLikeNone(includeDocumentStructure) ? 0xFFFFFF : includeDocumentStructure ? 1 : 0, ptr0, isLikeNone(maxArchiveDepth) ? Number.MAX_SAFE_INTEGER : (maxArchiveDepth) >>> 0, ptr1, ptr2, len2, ptr3, ptr4, ptr5, ptr6, ptr7, ptr8, ptr9, ptr10, ptr11, ptr12, ptr13, !isLikeNone(extractionTimeoutSecs), isLikeNone(extractionTimeoutSecs) ? BigInt(0) : extractionTimeoutSecs, isLikeNone(maxConcurrentExtractions) ? Number.MAX_SAFE_INTEGER : (maxConcurrentExtractions) >>> 0, ptr14, !isLikeNone(maxEmbeddedFileBytes), isLikeNone(maxEmbeddedFileBytes) ? BigInt(0) : maxEmbeddedFileBytes, ptr15, ptr16, ptr17, ptr18, len18, !isLikeNone(cacheTtlSecs), isLikeNone(cacheTtlSecs) ? BigInt(0) : cacheTtlSecs, ptr19, ptr20, ptr21, ptr22, ptr23, ptr24, ptr25, ptr26, ptr27, ptr28, ptr29, ptr30, isLikeNone(qrCodes) ? 0xFFFFFF : qrCodes ? 1 : 0);
|
|
14136
|
+
const ret = wasm.wasmextractionconfig_new(isLikeNone(mimeDetectionPolicy) ? 3 : mimeDetectionPolicy, isLikeNone(useCache) ? 0xFFFFFF : useCache ? 1 : 0, isLikeNone(enableQualityProcessing) ? 0xFFFFFF : enableQualityProcessing ? 1 : 0, isLikeNone(forceOcr) ? 0xFFFFFF : forceOcr ? 1 : 0, isLikeNone(ocrStrategy) ? 0 : addToExternrefTable0(ocrStrategy), isLikeNone(disableOcr) ? 0xFFFFFF : disableOcr ? 1 : 0, isLikeNone(resultFormat) ? 2 : resultFormat, isLikeNone(outputFormat) ? 0 : addToExternrefTable0(outputFormat), isLikeNone(escapeMarkdown) ? 0xFFFFFF : escapeMarkdown ? 1 : 0, isLikeNone(tableAnchors) ? 0xFFFFFF : tableAnchors ? 1 : 0, isLikeNone(jupyterCellRendering) ? 3 : jupyterCellRendering, isLikeNone(applyNotebookCellTags) ? 0xFFFFFF : applyNotebookCellTags ? 1 : 0, isLikeNone(useLayoutForMarkdown) ? 0xFFFFFF : useLayoutForMarkdown ? 1 : 0, isLikeNone(includeDocumentStructure) ? 0xFFFFFF : includeDocumentStructure ? 1 : 0, ptr0, isLikeNone(maxArchiveDepth) ? Number.MAX_SAFE_INTEGER : (maxArchiveDepth) >>> 0, ptr1, ptr2, len2, isLikeNone(ocrNearEmptyFallback) ? 0xFFFFFF : ocrNearEmptyFallback ? 1 : 0, isLikeNone(ocrScannedPageQualityGate) ? 0xFFFFFF : ocrScannedPageQualityGate ? 1 : 0, isLikeNone(ocrEmbeddedImages) ? 0xFFFFFF : ocrEmbeddedImages ? 1 : 0, ptr3, ptr4, ptr5, ptr6, ptr7, ptr8, ptr9, ptr10, ptr11, ptr12, ptr13, !isLikeNone(extractionTimeoutSecs), isLikeNone(extractionTimeoutSecs) ? BigInt(0) : extractionTimeoutSecs, isLikeNone(maxConcurrentExtractions) ? Number.MAX_SAFE_INTEGER : (maxConcurrentExtractions) >>> 0, ptr14, !isLikeNone(maxEmbeddedFileBytes), isLikeNone(maxEmbeddedFileBytes) ? BigInt(0) : maxEmbeddedFileBytes, ptr15, ptr16, ptr17, ptr18, len18, !isLikeNone(cacheTtlSecs), isLikeNone(cacheTtlSecs) ? BigInt(0) : cacheTtlSecs, ptr19, ptr20, ptr21, ptr22, ptr23, ptr24, ptr25, ptr26, ptr27, ptr28, ptr29, ptr30, isLikeNone(qrCodes) ? 0xFFFFFF : qrCodes ? 1 : 0);
|
|
14134
14137
|
this.__wbg_ptr = ret;
|
|
14135
14138
|
WasmExtractionConfigFinalization.register(this, this.__wbg_ptr, this);
|
|
14136
14139
|
return this;
|
|
14137
14140
|
}
|
|
14141
|
+
/**
|
|
14142
|
+
* @returns {boolean | undefined}
|
|
14143
|
+
*/
|
|
14144
|
+
get ocrEmbeddedImages() {
|
|
14145
|
+
const ret = wasm.wasmextractionconfig_ocrEmbeddedImages(this.__wbg_ptr);
|
|
14146
|
+
return ret === 0xFFFFFF ? undefined : ret !== 0;
|
|
14147
|
+
}
|
|
14148
|
+
/**
|
|
14149
|
+
* @returns {boolean | undefined}
|
|
14150
|
+
*/
|
|
14151
|
+
get ocrNearEmptyFallback() {
|
|
14152
|
+
const ret = wasm.wasmextractionconfig_ocrNearEmptyFallback(this.__wbg_ptr);
|
|
14153
|
+
return ret === 0xFFFFFF ? undefined : ret !== 0;
|
|
14154
|
+
}
|
|
14155
|
+
/**
|
|
14156
|
+
* @returns {boolean | undefined}
|
|
14157
|
+
*/
|
|
14158
|
+
get ocrScannedPageQualityGate() {
|
|
14159
|
+
const ret = wasm.wasmextractionconfig_ocrScannedPageQualityGate(this.__wbg_ptr);
|
|
14160
|
+
return ret === 0xFFFFFF ? undefined : ret !== 0;
|
|
14161
|
+
}
|
|
14138
14162
|
/**
|
|
14139
14163
|
* @returns {any}
|
|
14140
14164
|
*/
|
|
@@ -14216,13 +14240,24 @@ export class WasmExtractionConfig {
|
|
|
14216
14240
|
/**
|
|
14217
14241
|
* Whether embedded images get OCR'd.
|
|
14218
14242
|
*
|
|
14219
|
-
* This
|
|
14220
|
-
* `image_ocr.process_images_with_ocr`
|
|
14221
|
-
*
|
|
14222
|
-
*
|
|
14223
|
-
*
|
|
14224
|
-
*
|
|
14225
|
-
*
|
|
14243
|
+
* This is THE condition -- `core/pipeline/mod.rs` and
|
|
14244
|
+
* `extraction.image_ocr.process_images_with_ocr` both call this method rather than
|
|
14245
|
+
* re-deriving it, because when two copies of it drifted apart a container extractor
|
|
14246
|
+
* asked `needs_image_data` and was told no, so it attached an image with an empty
|
|
14247
|
+
* buffer, and the OCR path then ran on those zero bytes and reported `Could not
|
|
14248
|
+
* determine image format` (GH#1662). A re-derived copy anywhere reopens that defect;
|
|
14249
|
+
* `embedded_image_ocr_gate_has_no_second_copy` in `core/pipeline/tests.rs` fails if
|
|
14250
|
+
* one appears.
|
|
14251
|
+
*
|
|
14252
|
+
* `Self.ocr_embedded_images` is the caller's explicit answer to the OCR half;
|
|
14253
|
+
* `None` derives it from whether an `ocr` block is present, which is what this
|
|
14254
|
+
* condition was before GH#1752 gave the behaviour a setting of its own.
|
|
14255
|
+
*
|
|
14256
|
+
* `disable_ocr` still wins regardless: it is documented as skipping OCR "for all
|
|
14257
|
+
* document types", and `ocr_embedded_images` (like the plain presence of an `ocr`
|
|
14258
|
+
* block before it) is an AUTOMATIC trigger, not an explicit request like `force_ocr` --
|
|
14259
|
+
* see `Self.effective_disable_ocr`'s callers elsewhere (`needs_image_processing`,
|
|
14260
|
+
* `extractors/image.rs`, `engine/extract_impl.rs`) for the same precedent.
|
|
14226
14261
|
* @returns {boolean}
|
|
14227
14262
|
*/
|
|
14228
14263
|
runsOcrOnEmbeddedImages() {
|
|
@@ -14506,6 +14541,24 @@ export class WasmExtractionConfig {
|
|
|
14506
14541
|
}
|
|
14507
14542
|
wasm.wasmextractionconfig_set_ner(this.__wbg_ptr, ptr0);
|
|
14508
14543
|
}
|
|
14544
|
+
/**
|
|
14545
|
+
* @param {boolean | null} [value]
|
|
14546
|
+
*/
|
|
14547
|
+
set ocrEmbeddedImages(value) {
|
|
14548
|
+
wasm.wasmextractionconfig_set_ocrEmbeddedImages(this.__wbg_ptr, isLikeNone(value) ? 0xFFFFFF : value ? 1 : 0);
|
|
14549
|
+
}
|
|
14550
|
+
/**
|
|
14551
|
+
* @param {boolean | null} [value]
|
|
14552
|
+
*/
|
|
14553
|
+
set ocrNearEmptyFallback(value) {
|
|
14554
|
+
wasm.wasmextractionconfig_set_ocrNearEmptyFallback(this.__wbg_ptr, isLikeNone(value) ? 0xFFFFFF : value ? 1 : 0);
|
|
14555
|
+
}
|
|
14556
|
+
/**
|
|
14557
|
+
* @param {boolean | null} [value]
|
|
14558
|
+
*/
|
|
14559
|
+
set ocrScannedPageQualityGate(value) {
|
|
14560
|
+
wasm.wasmextractionconfig_set_ocrScannedPageQualityGate(this.__wbg_ptr, isLikeNone(value) ? 0xFFFFFF : value ? 1 : 0);
|
|
14561
|
+
}
|
|
14509
14562
|
/**
|
|
14510
14563
|
* @param {any} value
|
|
14511
14564
|
*/
|
|
@@ -41590,7 +41643,7 @@ function __wbg_get_imports() {
|
|
|
41590
41643
|
const ret = arg0.value;
|
|
41591
41644
|
return ret;
|
|
41592
41645
|
},
|
|
41593
|
-
|
|
41646
|
+
__wbg_warn_ef083b0b8c8c0df4: function(arg0, arg1) {
|
|
41594
41647
|
console.warn(getStringFromWasm0(arg0, arg1));
|
|
41595
41648
|
},
|
|
41596
41649
|
__wbg_wasmarchiveentry_new: function(arg0) {
|
|
@@ -1,4 +1,4 @@
|
|
|
1
1
|
[diffend] Oversized file quarantined before diffing.
|
|
2
2
|
name: package/pkg/web/xberg_wasm_bg.wasm
|
|
3
|
-
size:
|
|
4
|
-
sha256:
|
|
3
|
+
size: 36061962 bytes
|
|
4
|
+
sha256: 5f03732f27f9ff4262949256035d8c4a6c60af5823e85a20bb7a33c763f72e3a
|
|
@@ -1524,8 +1524,11 @@ export const wasmextractionconfig_mimeDetectionPolicy: (a: number) => [number, n
|
|
|
1524
1524
|
export const wasmextractionconfig_needsImageData: (a: number) => number;
|
|
1525
1525
|
export const wasmextractionconfig_needsImageProcessing: (a: number) => number;
|
|
1526
1526
|
export const wasmextractionconfig_ner: (a: number) => number;
|
|
1527
|
-
export const wasmextractionconfig_new: (a: number, b: number, c: number, d: number, e: number, f: number, g: number, h: number, i: number, j: number, k: number, l: number, m: number, n: number, o: number, p: number, q: number, r: number, s: number, t: number, u: number, v: number, w: number, x: number, y: number, z: number, a1: number, b1: number, c1: number, d1: number, e1: number, f1:
|
|
1527
|
+
export const wasmextractionconfig_new: (a: number, b: number, c: number, d: number, e: number, f: number, g: number, h: number, i: number, j: number, k: number, l: number, m: number, n: number, o: number, p: number, q: number, r: number, s: number, t: number, u: number, v: number, w: number, x: number, y: number, z: number, a1: number, b1: number, c1: number, d1: number, e1: number, f1: number, g1: number, h1: number, i1: bigint, j1: number, k1: number, l1: number, m1: bigint, n1: number, o1: number, p1: number, q1: number, r1: number, s1: number, t1: bigint, u1: number, v1: number, w1: number, x1: number, y1: number, z1: number, a2: number, b2: number, c2: number, d2: number, e2: number, f2: number, g2: number) => number;
|
|
1528
1528
|
export const wasmextractionconfig_ocr: (a: number) => number;
|
|
1529
|
+
export const wasmextractionconfig_ocrEmbeddedImages: (a: number) => number;
|
|
1530
|
+
export const wasmextractionconfig_ocrNearEmptyFallback: (a: number) => number;
|
|
1531
|
+
export const wasmextractionconfig_ocrScannedPageQualityGate: (a: number) => number;
|
|
1529
1532
|
export const wasmextractionconfig_ocrStrategy: (a: number) => any;
|
|
1530
1533
|
export const wasmextractionconfig_outputFormat: (a: number) => any;
|
|
1531
1534
|
export const wasmextractionconfig_pageClassification: (a: number) => number;
|
|
@@ -1569,6 +1572,9 @@ export const wasmextractionconfig_set_maxEmbeddedFileBytes: (a: number, b: numbe
|
|
|
1569
1572
|
export const wasmextractionconfig_set_mimeDetectionPolicy: (a: number, b: number) => void;
|
|
1570
1573
|
export const wasmextractionconfig_set_ner: (a: number, b: number) => void;
|
|
1571
1574
|
export const wasmextractionconfig_set_ocr: (a: number, b: number) => void;
|
|
1575
|
+
export const wasmextractionconfig_set_ocrEmbeddedImages: (a: number, b: number) => void;
|
|
1576
|
+
export const wasmextractionconfig_set_ocrNearEmptyFallback: (a: number, b: number) => void;
|
|
1577
|
+
export const wasmextractionconfig_set_ocrScannedPageQualityGate: (a: number, b: number) => void;
|
|
1572
1578
|
export const wasmextractionconfig_set_ocrStrategy: (a: number, b: any) => void;
|
|
1573
1579
|
export const wasmextractionconfig_set_outputFormat: (a: number, b: any) => void;
|
|
1574
1580
|
export const wasmextractionconfig_set_pageClassification: (a: number, b: number) => void;
|