@xberg-io/langchain-xberg 0.0.1 → 1.0.0-rc.39
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +81 -2
- package/dist/index.cjs +284 -0
- package/dist/index.cjs.map +1 -0
- package/dist/index.d.cts +37 -0
- package/dist/index.d.ts +37 -0
- package/dist/index.js +247 -0
- package/dist/index.js.map +1 -0
- package/package.json +60 -4
package/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2025-2026 Kreuzberg, Inc.
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
package/README.md
CHANGED
|
@@ -1,4 +1,83 @@
|
|
|
1
|
+
<p align="center">
|
|
2
|
+
<picture>
|
|
3
|
+
<source media="(prefers-color-scheme: dark)" srcset="https://cdn.jsdelivr.net/gh/xberg-io/assets@v1/banner/readme-banner-dark.svg">
|
|
4
|
+
<img alt="Xberg" width="420" src="https://cdn.jsdelivr.net/gh/xberg-io/assets@v1/banner/readme-banner-light.svg">
|
|
5
|
+
</picture>
|
|
6
|
+
</p>
|
|
7
|
+
|
|
1
8
|
# @xberg-io/langchain-xberg
|
|
2
9
|
|
|
3
|
-
|
|
4
|
-
|
|
10
|
+
[](https://www.npmjs.com/package/@xberg-io/langchain-xberg)
|
|
11
|
+
|
|
12
|
+
A [LangChain.js](https://js.langchain.com) document loader for [Xberg](https://github.com/xberg-io/xberg).
|
|
13
|
+
Point it at a file, a directory, or raw bytes and it returns LangChain `Document`s with the extracted
|
|
14
|
+
text, tables, and rich metadata from 90+ document formats — with optional OCR for scans and images.
|
|
15
|
+
|
|
16
|
+
Extraction runs locally in-process through the `@xberg-io/xberg` native binding. No API key, no cloud
|
|
17
|
+
call, no data leaves your machine.
|
|
18
|
+
|
|
19
|
+
## Installation
|
|
20
|
+
|
|
21
|
+
`@langchain/core` is a peer dependency — install it alongside the loader:
|
|
22
|
+
|
|
23
|
+
```bash
|
|
24
|
+
npm install @xberg-io/langchain-xberg @langchain/core
|
|
25
|
+
```
|
|
26
|
+
|
|
27
|
+
Node.js 20.15+ is required on a platform for which `@xberg-io/xberg` ships a prebuilt binary (Linux
|
|
28
|
+
x64/arm64 glibc or musl, macOS arm64, Windows x64/arm64).
|
|
29
|
+
|
|
30
|
+
## Quick start
|
|
31
|
+
|
|
32
|
+
```ts
|
|
33
|
+
import { XbergLoader } from "@xberg-io/langchain-xberg";
|
|
34
|
+
|
|
35
|
+
// Single file — one Document
|
|
36
|
+
const loader = new XbergLoader({ filePath: "report.pdf" });
|
|
37
|
+
const docs = await loader.load();
|
|
38
|
+
console.log(docs[0].pageContent, docs[0].metadata);
|
|
39
|
+
|
|
40
|
+
// Multiple files or a directory — one batched extraction
|
|
41
|
+
const many = new XbergLoader({ filePath: "./docs", glob: "**/*.pdf" });
|
|
42
|
+
|
|
43
|
+
// Raw bytes — mimeType is required
|
|
44
|
+
const bytes = new XbergLoader({ data: fileBytes, mimeType: "application/pdf" });
|
|
45
|
+
```
|
|
46
|
+
|
|
47
|
+
### One Document per chunk (retrieval)
|
|
48
|
+
|
|
49
|
+
Enable `chunking` on the `ExtractionConfig` to emit one `Document` per chunk, each carrying heading
|
|
50
|
+
path, page span, and token-count metadata ready to embed:
|
|
51
|
+
|
|
52
|
+
```ts
|
|
53
|
+
const loader = new XbergLoader({
|
|
54
|
+
filePath: "report.pdf",
|
|
55
|
+
config: { chunking: { max_chars: 1000, max_overlap: 200 } },
|
|
56
|
+
});
|
|
57
|
+
const chunks = await loader.load();
|
|
58
|
+
```
|
|
59
|
+
|
|
60
|
+
### One Document per page
|
|
61
|
+
|
|
62
|
+
```ts
|
|
63
|
+
const loader = new XbergLoader({
|
|
64
|
+
filePath: "report.pdf",
|
|
65
|
+
config: { pages: { extractPages: true } },
|
|
66
|
+
});
|
|
67
|
+
const pages = await loader.load(); // metadata.page is 0-indexed
|
|
68
|
+
```
|
|
69
|
+
|
|
70
|
+
## Supported formats
|
|
71
|
+
|
|
72
|
+
Xberg extracts from 90+ formats including PDF, DOCX, PPTX, XLSX, HTML, EPUB, images, and more. See the
|
|
73
|
+
[Xberg documentation](https://docs.xberg.io) for the full list and the extraction configuration
|
|
74
|
+
reference.
|
|
75
|
+
|
|
76
|
+
## Part of Xberg.io
|
|
77
|
+
|
|
78
|
+
- [Xberg](https://github.com/xberg-io/xberg) — document intelligence: text, tables, metadata from 91+ formats with optional OCR.
|
|
79
|
+
- [Xberg Enterprise](https://github.com/xberg-io/xberg-enterprise) — managed extraction API with SDKs, dashboards, and observability.
|
|
80
|
+
|
|
81
|
+
## License
|
|
82
|
+
|
|
83
|
+
[MIT](./LICENSE)
|
package/dist/index.cjs
ADDED
|
@@ -0,0 +1,284 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
var __create = Object.create;
|
|
3
|
+
var __defProp = Object.defineProperty;
|
|
4
|
+
var __getOwnPropDesc = Object.getOwnPropertyDescriptor;
|
|
5
|
+
var __getOwnPropNames = Object.getOwnPropertyNames;
|
|
6
|
+
var __getProtoOf = Object.getPrototypeOf;
|
|
7
|
+
var __hasOwnProp = Object.prototype.hasOwnProperty;
|
|
8
|
+
var __export = (target, all) => {
|
|
9
|
+
for (var name in all)
|
|
10
|
+
__defProp(target, name, { get: all[name], enumerable: true });
|
|
11
|
+
};
|
|
12
|
+
var __copyProps = (to, from, except, desc) => {
|
|
13
|
+
if (from && typeof from === "object" || typeof from === "function") {
|
|
14
|
+
for (let key of __getOwnPropNames(from))
|
|
15
|
+
if (!__hasOwnProp.call(to, key) && key !== except)
|
|
16
|
+
__defProp(to, key, { get: () => from[key], enumerable: !(desc = __getOwnPropDesc(from, key)) || desc.enumerable });
|
|
17
|
+
}
|
|
18
|
+
return to;
|
|
19
|
+
};
|
|
20
|
+
var __toESM = (mod, isNodeMode, target) => (target = mod != null ? __create(__getProtoOf(mod)) : {}, __copyProps(
|
|
21
|
+
// If the importer is in node compatibility mode or this is not an ESM
|
|
22
|
+
// file that has been converted to a CommonJS file using a Babel-
|
|
23
|
+
// compatible transform (i.e. "__esModule" has not been set), then set
|
|
24
|
+
// "default" to the CommonJS "module.exports" for node compatibility.
|
|
25
|
+
isNodeMode || !mod || !mod.__esModule ? __defProp(target, "default", { value: mod, enumerable: true }) : target,
|
|
26
|
+
mod
|
|
27
|
+
));
|
|
28
|
+
var __toCommonJS = (mod) => __copyProps(__defProp({}, "__esModule", { value: true }), mod);
|
|
29
|
+
|
|
30
|
+
// src/index.ts
|
|
31
|
+
var index_exports = {};
|
|
32
|
+
__export(index_exports, {
|
|
33
|
+
XbergLoader: () => XbergLoader
|
|
34
|
+
});
|
|
35
|
+
module.exports = __toCommonJS(index_exports);
|
|
36
|
+
|
|
37
|
+
// src/loader.ts
|
|
38
|
+
var import_promises = require("fs/promises");
|
|
39
|
+
var import_base = require("@langchain/core/document_loaders/base");
|
|
40
|
+
var import_xberg = require("@xberg-io/xberg");
|
|
41
|
+
var import_fast_glob = __toESM(require("fast-glob"), 1);
|
|
42
|
+
|
|
43
|
+
// src/mapping.ts
|
|
44
|
+
var import_documents = require("@langchain/core/documents");
|
|
45
|
+
var METADATA_FIELDS = [
|
|
46
|
+
["title", "title"],
|
|
47
|
+
["subject", "subject"],
|
|
48
|
+
["authors", "authors"],
|
|
49
|
+
["keywords", "keywords"],
|
|
50
|
+
["language", "language"],
|
|
51
|
+
["created_at", "createdAt"],
|
|
52
|
+
["modified_at", "modifiedAt"],
|
|
53
|
+
["created_by", "createdBy"],
|
|
54
|
+
["modified_by", "modifiedBy"],
|
|
55
|
+
["category", "category"],
|
|
56
|
+
["tags", "tags"],
|
|
57
|
+
["document_version", "documentVersion"],
|
|
58
|
+
["abstract_text", "abstractText"],
|
|
59
|
+
["output_format", "outputFormat"],
|
|
60
|
+
["ocr_used", "ocrUsed"],
|
|
61
|
+
["extraction_duration_ms", "extractionDurationMs"]
|
|
62
|
+
];
|
|
63
|
+
var CONTENT_SEPARATOR = "\n\n";
|
|
64
|
+
function isChunkingEnabled(config) {
|
|
65
|
+
return config?.chunking != null;
|
|
66
|
+
}
|
|
67
|
+
function isPerPageEnabled(config) {
|
|
68
|
+
return Boolean(config?.pages?.extractPages);
|
|
69
|
+
}
|
|
70
|
+
function flattenMetadata(metadata) {
|
|
71
|
+
const flat = {};
|
|
72
|
+
if (!metadata) {
|
|
73
|
+
return flat;
|
|
74
|
+
}
|
|
75
|
+
for (const [snakeKey, camelKey] of METADATA_FIELDS) {
|
|
76
|
+
const value = metadata[camelKey];
|
|
77
|
+
if (value != null) {
|
|
78
|
+
flat[snakeKey] = value;
|
|
79
|
+
}
|
|
80
|
+
}
|
|
81
|
+
const additional = metadata.additional;
|
|
82
|
+
if (additional && Object.keys(additional).length > 0) {
|
|
83
|
+
flat.additional = { ...additional };
|
|
84
|
+
}
|
|
85
|
+
return flat;
|
|
86
|
+
}
|
|
87
|
+
function assembleContent(content, tables) {
|
|
88
|
+
const text = content ?? "";
|
|
89
|
+
if (!tables || tables.length === 0) {
|
|
90
|
+
return text;
|
|
91
|
+
}
|
|
92
|
+
const parts = tables.map((table) => table.markdown ?? "").filter((markdown) => markdown.length > 0);
|
|
93
|
+
if (parts.length === 0) {
|
|
94
|
+
return text;
|
|
95
|
+
}
|
|
96
|
+
return [text, ...parts].join(CONTENT_SEPARATOR);
|
|
97
|
+
}
|
|
98
|
+
function buildMetadata(document, source) {
|
|
99
|
+
const metadata = flattenMetadata(document.metadata);
|
|
100
|
+
metadata.mime_type = document.mimeType;
|
|
101
|
+
if (document.qualityScore != null) {
|
|
102
|
+
metadata.quality_score = document.qualityScore;
|
|
103
|
+
}
|
|
104
|
+
if (document.detectedLanguages && document.detectedLanguages.length > 0) {
|
|
105
|
+
metadata.detected_languages = document.detectedLanguages;
|
|
106
|
+
}
|
|
107
|
+
if (document.counts != null) {
|
|
108
|
+
metadata.page_count = document.counts.pages;
|
|
109
|
+
}
|
|
110
|
+
if (document.extractedKeywords && document.extractedKeywords.length > 0) {
|
|
111
|
+
metadata.extracted_keywords = document.extractedKeywords.map((keyword) => ({
|
|
112
|
+
text: keyword.text,
|
|
113
|
+
score: keyword.score,
|
|
114
|
+
algorithm: String(keyword.algorithm)
|
|
115
|
+
}));
|
|
116
|
+
}
|
|
117
|
+
const tables = document.tables ?? [];
|
|
118
|
+
metadata.table_count = tables.length;
|
|
119
|
+
if (tables.length > 0) {
|
|
120
|
+
metadata.tables = tables.map((table) => ({
|
|
121
|
+
cells: table.cells,
|
|
122
|
+
markdown: table.markdown,
|
|
123
|
+
page_number: table.pageNumber
|
|
124
|
+
}));
|
|
125
|
+
}
|
|
126
|
+
if (document.processingWarnings && document.processingWarnings.length > 0) {
|
|
127
|
+
metadata.processing_warnings = document.processingWarnings.map((warning) => ({
|
|
128
|
+
source: warning.source,
|
|
129
|
+
message: warning.message
|
|
130
|
+
}));
|
|
131
|
+
}
|
|
132
|
+
metadata.source = source;
|
|
133
|
+
return metadata;
|
|
134
|
+
}
|
|
135
|
+
function chunksToDocuments(document, source) {
|
|
136
|
+
const baseMetadata = buildMetadata(document, source);
|
|
137
|
+
const documents = [];
|
|
138
|
+
for (const chunk of document.chunks ?? []) {
|
|
139
|
+
const metadata = { ...baseMetadata };
|
|
140
|
+
const chunkMetadata = chunk.metadata;
|
|
141
|
+
metadata.chunk_index = chunkMetadata.chunkIndex;
|
|
142
|
+
metadata.total_chunks = chunkMetadata.totalChunks;
|
|
143
|
+
metadata.chunk_type = String(chunk.chunkType);
|
|
144
|
+
if (chunkMetadata.headingPath && chunkMetadata.headingPath.length > 0) {
|
|
145
|
+
metadata.heading_path = [...chunkMetadata.headingPath];
|
|
146
|
+
}
|
|
147
|
+
if (chunkMetadata.tokenCount != null) {
|
|
148
|
+
metadata.token_count = chunkMetadata.tokenCount;
|
|
149
|
+
}
|
|
150
|
+
if (chunkMetadata.firstPage != null) {
|
|
151
|
+
metadata.page = chunkMetadata.firstPage - 1;
|
|
152
|
+
metadata.first_page = chunkMetadata.firstPage;
|
|
153
|
+
}
|
|
154
|
+
if (chunkMetadata.lastPage != null) {
|
|
155
|
+
metadata.last_page = chunkMetadata.lastPage;
|
|
156
|
+
}
|
|
157
|
+
documents.push(new import_documents.Document({ pageContent: chunk.content, metadata }));
|
|
158
|
+
}
|
|
159
|
+
return documents;
|
|
160
|
+
}
|
|
161
|
+
function pagesToDocuments(document, source) {
|
|
162
|
+
const baseMetadata = buildMetadata(document, source);
|
|
163
|
+
const documents = [];
|
|
164
|
+
for (const page of document.pages ?? []) {
|
|
165
|
+
const metadata = { ...baseMetadata };
|
|
166
|
+
metadata.page = page.pageNumber - 1;
|
|
167
|
+
if (page.isBlank != null) {
|
|
168
|
+
metadata.is_blank = page.isBlank;
|
|
169
|
+
}
|
|
170
|
+
const pageContent = assembleContent(page.content, page.tables);
|
|
171
|
+
documents.push(new import_documents.Document({ pageContent, metadata }));
|
|
172
|
+
}
|
|
173
|
+
return documents;
|
|
174
|
+
}
|
|
175
|
+
function documentToDocuments(document, source, options) {
|
|
176
|
+
if (options.chunking && document.chunks && document.chunks.length > 0) {
|
|
177
|
+
return chunksToDocuments(document, source);
|
|
178
|
+
}
|
|
179
|
+
if (options.perPage && document.pages && document.pages.length > 0) {
|
|
180
|
+
return pagesToDocuments(document, source);
|
|
181
|
+
}
|
|
182
|
+
const metadata = buildMetadata(document, source);
|
|
183
|
+
const pageContent = assembleContent(document.content, document.tables);
|
|
184
|
+
return [new import_documents.Document({ pageContent, metadata })];
|
|
185
|
+
}
|
|
186
|
+
function resultToDocuments(result, sources, options) {
|
|
187
|
+
if (result.errors && result.errors.length > 0) {
|
|
188
|
+
const error = result.errors[0];
|
|
189
|
+
throw new Error(`Failed to extract '${error.source}': ${error.message}`);
|
|
190
|
+
}
|
|
191
|
+
const documents = [];
|
|
192
|
+
const results = result.results ?? [];
|
|
193
|
+
results.forEach((document, index) => {
|
|
194
|
+
const source = sources[index] ?? (sources.length > 0 ? sources[sources.length - 1] : "");
|
|
195
|
+
documents.push(...documentToDocuments(document, source, options));
|
|
196
|
+
});
|
|
197
|
+
return documents;
|
|
198
|
+
}
|
|
199
|
+
|
|
200
|
+
// src/loader.ts
|
|
201
|
+
var DEFAULT_GLOB = "**/*";
|
|
202
|
+
var XbergLoader = class extends import_base.BaseDocumentLoader {
|
|
203
|
+
filePath;
|
|
204
|
+
data;
|
|
205
|
+
mimeType;
|
|
206
|
+
glob;
|
|
207
|
+
config;
|
|
208
|
+
constructor(options) {
|
|
209
|
+
super();
|
|
210
|
+
const { filePath, data, mimeType, glob, config } = options;
|
|
211
|
+
if (filePath === void 0 && data === void 0) {
|
|
212
|
+
throw new Error("Either 'filePath' or 'data' must be provided.");
|
|
213
|
+
}
|
|
214
|
+
if (filePath !== void 0 && data !== void 0) {
|
|
215
|
+
throw new Error("Cannot specify both 'filePath' and 'data'. Use one or the other.");
|
|
216
|
+
}
|
|
217
|
+
if (data !== void 0 && mimeType === void 0) {
|
|
218
|
+
throw new Error("'mimeType' is required when using 'data'.");
|
|
219
|
+
}
|
|
220
|
+
this.filePath = filePath;
|
|
221
|
+
this.data = data;
|
|
222
|
+
this.mimeType = mimeType;
|
|
223
|
+
this.glob = glob;
|
|
224
|
+
this.config = config;
|
|
225
|
+
}
|
|
226
|
+
async load() {
|
|
227
|
+
const { inputs, sources, batch } = await this.buildInputs();
|
|
228
|
+
if (inputs.length === 0) {
|
|
229
|
+
return [];
|
|
230
|
+
}
|
|
231
|
+
let result;
|
|
232
|
+
try {
|
|
233
|
+
result = batch ? await (0, import_xberg.extractBatch)(inputs, this.config ?? null) : await (0, import_xberg.extract)(inputs[0], this.config ?? null);
|
|
234
|
+
} catch (error) {
|
|
235
|
+
const source = sources[0] ?? "input";
|
|
236
|
+
throw new Error(`Failed to extract '${source}': ${errorMessage(error)}`);
|
|
237
|
+
}
|
|
238
|
+
return resultToDocuments(result, sources, {
|
|
239
|
+
chunking: isChunkingEnabled(this.config),
|
|
240
|
+
perPage: isPerPageEnabled(this.config)
|
|
241
|
+
});
|
|
242
|
+
}
|
|
243
|
+
async buildInputs() {
|
|
244
|
+
if (this.data !== void 0) {
|
|
245
|
+
const source = `bytes://${this.mimeType}`;
|
|
246
|
+
const input = { kind: "bytes", bytes: this.data, mimeType: this.mimeType };
|
|
247
|
+
return { inputs: [input], sources: [source], batch: false };
|
|
248
|
+
}
|
|
249
|
+
const { paths, batch } = await this.resolvePaths();
|
|
250
|
+
const inputs = paths.map((path) => ({ kind: "uri", uri: path, mimeType: this.mimeType }));
|
|
251
|
+
return { inputs, sources: paths, batch };
|
|
252
|
+
}
|
|
253
|
+
async resolvePaths() {
|
|
254
|
+
const filePath = this.filePath;
|
|
255
|
+
if (Array.isArray(filePath)) {
|
|
256
|
+
return { paths: filePath, batch: true };
|
|
257
|
+
}
|
|
258
|
+
if (typeof filePath === "string") {
|
|
259
|
+
if (await isDirectory(filePath)) {
|
|
260
|
+
const pattern = this.glob ?? DEFAULT_GLOB;
|
|
261
|
+
const matches = await (0, import_fast_glob.default)(pattern, { cwd: filePath, onlyFiles: true, absolute: true });
|
|
262
|
+
return { paths: matches.sort(), batch: true };
|
|
263
|
+
}
|
|
264
|
+
return { paths: [filePath], batch: false };
|
|
265
|
+
}
|
|
266
|
+
return { paths: [], batch: false };
|
|
267
|
+
}
|
|
268
|
+
};
|
|
269
|
+
async function isDirectory(path) {
|
|
270
|
+
try {
|
|
271
|
+
const stats = await (0, import_promises.stat)(path);
|
|
272
|
+
return stats.isDirectory();
|
|
273
|
+
} catch {
|
|
274
|
+
return false;
|
|
275
|
+
}
|
|
276
|
+
}
|
|
277
|
+
function errorMessage(error) {
|
|
278
|
+
return error instanceof Error ? error.message : String(error);
|
|
279
|
+
}
|
|
280
|
+
// Annotate the CommonJS export names for ESM import in node:
|
|
281
|
+
0 && (module.exports = {
|
|
282
|
+
XbergLoader
|
|
283
|
+
});
|
|
284
|
+
//# sourceMappingURL=index.cjs.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"sources":["../src/index.ts","../src/loader.ts","../src/mapping.ts"],"sourcesContent":["export { XbergLoader, type XbergLoaderOptions } from \"./loader\";\n","import { stat } from \"node:fs/promises\";\nimport { BaseDocumentLoader } from \"@langchain/core/document_loaders/base\";\nimport type { Document } from \"@langchain/core/documents\";\nimport { extract, extractBatch } from \"@xberg-io/xberg\";\nimport type { ExtractInput, ExtractionConfig } from \"@xberg-io/xberg\";\nimport fastGlob from \"fast-glob\";\nimport { isChunkingEnabled, isPerPageEnabled, resultToDocuments, type ResultEnvelope } from \"./mapping\";\n\nconst DEFAULT_GLOB = \"**/*\";\n\nexport interface XbergLoaderOptions {\n /** File path, list of file paths, or directory path to load. */\n filePath?: string | string[];\n /** Raw bytes to extract from. Mutually exclusive with `filePath`. */\n data?: Uint8Array;\n /** MIME type hint. Required when using `data`, optional for `filePath`. */\n mimeType?: string;\n /** Glob pattern for directory mode. Defaults to matching all files. */\n glob?: string;\n /** Xberg extraction configuration controlling output format, OCR, pages, chunking, etc. */\n config?: ExtractionConfig;\n}\n\n/**\n * Load documents using Xberg, supporting 90+ file formats with true async extraction.\n *\n * By default each source becomes one Document. Enable `chunking` on the\n * `ExtractionConfig` to emit one Document per chunk, or `pages` for one Document\n * per page. Multiple paths (a list or a directory glob) are extracted with a single\n * `extractBatch` call so concurrency happens Rust-side.\n */\nexport class XbergLoader extends BaseDocumentLoader {\n private readonly filePath?: string | string[];\n private readonly data?: Uint8Array;\n private readonly mimeType?: string;\n private readonly glob?: string;\n private readonly config?: ExtractionConfig;\n\n constructor(options: XbergLoaderOptions) {\n super();\n const { filePath, data, mimeType, glob, config } = options;\n if (filePath === undefined && data === undefined) {\n throw new Error(\"Either 'filePath' or 'data' must be provided.\");\n }\n if (filePath !== undefined && data !== undefined) {\n throw new Error(\"Cannot specify both 'filePath' and 'data'. Use one or the other.\");\n }\n if (data !== undefined && mimeType === undefined) {\n throw new Error(\"'mimeType' is required when using 'data'.\");\n }\n\n this.filePath = filePath;\n this.data = data;\n this.mimeType = mimeType;\n this.glob = glob;\n this.config = config;\n }\n\n async load(): Promise<Document[]> {\n const { inputs, sources, batch } = await this.buildInputs();\n if (inputs.length === 0) {\n return [];\n }\n\n let result;\n try {\n result = batch ? await extractBatch(inputs, this.config ?? null) : await extract(inputs[0], this.config ?? null);\n } catch (error) {\n const source = sources[0] ?? \"input\";\n throw new Error(`Failed to extract '${source}': ${errorMessage(error)}`);\n }\n\n return resultToDocuments(result as unknown as ResultEnvelope, sources, {\n chunking: isChunkingEnabled(this.config),\n perPage: isPerPageEnabled(this.config),\n });\n }\n\n private async buildInputs(): Promise<{ inputs: ExtractInput[]; sources: string[]; batch: boolean }> {\n if (this.data !== undefined) {\n const source = `bytes://${this.mimeType}`;\n const input: ExtractInput = { kind: \"bytes\", bytes: this.data, mimeType: this.mimeType };\n return { inputs: [input], sources: [source], batch: false };\n }\n\n const { paths, batch } = await this.resolvePaths();\n const inputs: ExtractInput[] = paths.map((path) => ({ kind: \"uri\", uri: path, mimeType: this.mimeType }));\n return { inputs, sources: paths, batch };\n }\n\n private async resolvePaths(): Promise<{ paths: string[]; batch: boolean }> {\n const filePath = this.filePath;\n if (Array.isArray(filePath)) {\n return { paths: filePath, batch: true };\n }\n if (typeof filePath === \"string\") {\n if (await isDirectory(filePath)) {\n const pattern = this.glob ?? DEFAULT_GLOB;\n const matches = await fastGlob(pattern, { cwd: filePath, onlyFiles: true, absolute: true });\n return { paths: matches.sort(), batch: true };\n }\n return { paths: [filePath], batch: false };\n }\n return { paths: [], batch: false };\n }\n}\n\nasync function isDirectory(path: string): Promise<boolean> {\n try {\n const stats = await stat(path);\n return stats.isDirectory();\n } catch {\n return false;\n }\n}\n\nfunction errorMessage(error: unknown): string {\n return error instanceof Error ? error.message : String(error);\n}\n","import { Document } from \"@langchain/core/documents\";\nimport type { DocumentCounts, ExtractionConfig, Keyword, Metadata, ProcessingWarning, Table } from \"@xberg-io/xberg\";\n\n// Metadata fields (from Xberg Metadata) that carry JSON-friendly scalar/list values,\n// mapped from the binding's camelCase keys to the snake_case output keys used by the\n// Python adapter. Opaque nested fields (pages, format, imagePreprocessing, jsonSchema,\n// error) are skipped because they are native objects, not plain data. ~keep\nconst METADATA_FIELDS: ReadonlyArray<readonly [string, keyof Metadata]> = [\n [\"title\", \"title\"],\n [\"subject\", \"subject\"],\n [\"authors\", \"authors\"],\n [\"keywords\", \"keywords\"],\n [\"language\", \"language\"],\n [\"created_at\", \"createdAt\"],\n [\"modified_at\", \"modifiedAt\"],\n [\"created_by\", \"createdBy\"],\n [\"modified_by\", \"modifiedBy\"],\n [\"category\", \"category\"],\n [\"tags\", \"tags\"],\n [\"document_version\", \"documentVersion\"],\n [\"abstract_text\", \"abstractText\"],\n [\"output_format\", \"outputFormat\"],\n [\"ocr_used\", \"ocrUsed\"],\n [\"extraction_duration_ms\", \"extractionDurationMs\"],\n];\n\nconst CONTENT_SEPARATOR = \"\\n\\n\";\n\nexport interface ChunkMetadataShape {\n chunkIndex: number;\n totalChunks: number;\n headingPath?: string[];\n tokenCount?: number;\n firstPage?: number;\n lastPage?: number;\n}\n\nexport interface ChunkShape {\n content: string;\n chunkType: unknown;\n metadata: ChunkMetadataShape;\n}\n\nexport interface PageShape {\n pageNumber: number;\n content: string;\n tables?: Table[];\n isBlank?: boolean;\n}\n\n// Structural view of a binding ExtractedDocument. The binding types its own fields\n// through unexported `Js*` aliases; this interface uses the clean exported types plus\n// local chunk/page shapes so the mapper is fully typed and testable without the addon. ~keep\nexport interface ExtractedDoc {\n content?: string;\n mimeType?: string;\n metadata?: Metadata;\n tables?: Table[];\n counts?: DocumentCounts;\n detectedLanguages?: string[];\n chunks?: ChunkShape[];\n pages?: PageShape[];\n extractedKeywords?: Keyword[];\n qualityScore?: number;\n processingWarnings?: ProcessingWarning[];\n}\n\nexport interface ResultEnvelope {\n errors?: Array<{ source: string; message: string }>;\n results?: ExtractedDoc[];\n}\n\nexport interface SplitOptions {\n chunking: boolean;\n perPage: boolean;\n}\n\nexport function isChunkingEnabled(config?: ExtractionConfig | null): boolean {\n return config?.chunking != null;\n}\n\nexport function isPerPageEnabled(config?: ExtractionConfig | null): boolean {\n return Boolean(config?.pages?.extractPages);\n}\n\nexport function flattenMetadata(metadata?: Metadata): Record<string, unknown> {\n const flat: Record<string, unknown> = {};\n if (!metadata) {\n return flat;\n }\n\n for (const [snakeKey, camelKey] of METADATA_FIELDS) {\n const value = metadata[camelKey];\n if (value != null) {\n flat[snakeKey] = value;\n }\n }\n\n const additional = metadata.additional;\n if (additional && Object.keys(additional).length > 0) {\n flat.additional = { ...additional };\n }\n\n return flat;\n}\n\nexport function assembleContent(content: string | undefined, tables: Table[] | undefined): string {\n const text = content ?? \"\";\n if (!tables || tables.length === 0) {\n return text;\n }\n const parts = tables.map((table) => table.markdown ?? \"\").filter((markdown) => markdown.length > 0);\n if (parts.length === 0) {\n return text;\n }\n return [text, ...parts].join(CONTENT_SEPARATOR);\n}\n\nexport function buildMetadata(document: ExtractedDoc, source: string): Record<string, unknown> {\n const metadata = flattenMetadata(document.metadata);\n\n metadata.mime_type = document.mimeType;\n if (document.qualityScore != null) {\n metadata.quality_score = document.qualityScore;\n }\n if (document.detectedLanguages && document.detectedLanguages.length > 0) {\n metadata.detected_languages = document.detectedLanguages;\n }\n if (document.counts != null) {\n metadata.page_count = document.counts.pages;\n }\n\n if (document.extractedKeywords && document.extractedKeywords.length > 0) {\n metadata.extracted_keywords = document.extractedKeywords.map((keyword) => ({\n text: keyword.text,\n score: keyword.score,\n algorithm: String(keyword.algorithm),\n }));\n }\n\n const tables = document.tables ?? [];\n metadata.table_count = tables.length;\n if (tables.length > 0) {\n metadata.tables = tables.map((table) => ({\n cells: table.cells,\n markdown: table.markdown,\n page_number: table.pageNumber,\n }));\n }\n\n if (document.processingWarnings && document.processingWarnings.length > 0) {\n metadata.processing_warnings = document.processingWarnings.map((warning) => ({\n source: warning.source,\n message: warning.message,\n }));\n }\n\n metadata.source = source;\n return metadata;\n}\n\nfunction chunksToDocuments(document: ExtractedDoc, source: string): Document[] {\n const baseMetadata = buildMetadata(document, source);\n const documents: Document[] = [];\n\n for (const chunk of document.chunks ?? []) {\n const metadata: Record<string, unknown> = { ...baseMetadata };\n const chunkMetadata = chunk.metadata;\n metadata.chunk_index = chunkMetadata.chunkIndex;\n metadata.total_chunks = chunkMetadata.totalChunks;\n metadata.chunk_type = String(chunk.chunkType);\n if (chunkMetadata.headingPath && chunkMetadata.headingPath.length > 0) {\n metadata.heading_path = [...chunkMetadata.headingPath];\n }\n if (chunkMetadata.tokenCount != null) {\n metadata.token_count = chunkMetadata.tokenCount;\n }\n if (chunkMetadata.firstPage != null) {\n // Xberg uses 1-indexed pages; LangChain convention is 0-indexed. ~keep\n metadata.page = chunkMetadata.firstPage - 1;\n metadata.first_page = chunkMetadata.firstPage;\n }\n if (chunkMetadata.lastPage != null) {\n metadata.last_page = chunkMetadata.lastPage;\n }\n documents.push(new Document({ pageContent: chunk.content, metadata }));\n }\n\n return documents;\n}\n\nfunction pagesToDocuments(document: ExtractedDoc, source: string): Document[] {\n const baseMetadata = buildMetadata(document, source);\n const documents: Document[] = [];\n\n for (const page of document.pages ?? []) {\n const metadata: Record<string, unknown> = { ...baseMetadata };\n // Xberg uses 1-indexed pages; LangChain convention is 0-indexed. ~keep\n metadata.page = page.pageNumber - 1;\n if (page.isBlank != null) {\n metadata.is_blank = page.isBlank;\n }\n const pageContent = assembleContent(page.content, page.tables);\n documents.push(new Document({ pageContent, metadata }));\n }\n\n return documents;\n}\n\nexport function documentToDocuments(document: ExtractedDoc, source: string, options: SplitOptions): Document[] {\n if (options.chunking && document.chunks && document.chunks.length > 0) {\n return chunksToDocuments(document, source);\n }\n if (options.perPage && document.pages && document.pages.length > 0) {\n return pagesToDocuments(document, source);\n }\n const metadata = buildMetadata(document, source);\n const pageContent = assembleContent(document.content, document.tables);\n return [new Document({ pageContent, metadata })];\n}\n\nexport function resultToDocuments(result: ResultEnvelope, sources: string[], options: SplitOptions): Document[] {\n if (result.errors && result.errors.length > 0) {\n const error = result.errors[0];\n throw new Error(`Failed to extract '${error.source}': ${error.message}`);\n }\n\n const documents: Document[] = [];\n const results = result.results ?? [];\n results.forEach((document, index) => {\n const source = sources[index] ?? (sources.length > 0 ? sources[sources.length - 1] : \"\");\n documents.push(...documentToDocuments(document, source, options));\n });\n return documents;\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AAAA;AAAA;AAAA;AAAA;AAAA;;;ACAA,sBAAqB;AACrB,kBAAmC;AAEnC,mBAAsC;AAEtC,uBAAqB;;;ACLrB,uBAAyB;AAOzB,IAAM,kBAAoE;AAAA,EACxE,CAAC,SAAS,OAAO;AAAA,EACjB,CAAC,WAAW,SAAS;AAAA,EACrB,CAAC,WAAW,SAAS;AAAA,EACrB,CAAC,YAAY,UAAU;AAAA,EACvB,CAAC,YAAY,UAAU;AAAA,EACvB,CAAC,cAAc,WAAW;AAAA,EAC1B,CAAC,eAAe,YAAY;AAAA,EAC5B,CAAC,cAAc,WAAW;AAAA,EAC1B,CAAC,eAAe,YAAY;AAAA,EAC5B,CAAC,YAAY,UAAU;AAAA,EACvB,CAAC,QAAQ,MAAM;AAAA,EACf,CAAC,oBAAoB,iBAAiB;AAAA,EACtC,CAAC,iBAAiB,cAAc;AAAA,EAChC,CAAC,iBAAiB,cAAc;AAAA,EAChC,CAAC,YAAY,SAAS;AAAA,EACtB,CAAC,0BAA0B,sBAAsB;AACnD;AAEA,IAAM,oBAAoB;AAmDnB,SAAS,kBAAkB,QAA2C;AAC3E,SAAO,QAAQ,YAAY;AAC7B;AAEO,SAAS,iBAAiB,QAA2C;AAC1E,SAAO,QAAQ,QAAQ,OAAO,YAAY;AAC5C;AAEO,SAAS,gBAAgB,UAA8C;AAC5E,QAAM,OAAgC,CAAC;AACvC,MAAI,CAAC,UAAU;AACb,WAAO;AAAA,EACT;AAEA,aAAW,CAAC,UAAU,QAAQ,KAAK,iBAAiB;AAClD,UAAM,QAAQ,SAAS,QAAQ;AAC/B,QAAI,SAAS,MAAM;AACjB,WAAK,QAAQ,IAAI;AAAA,IACnB;AAAA,EACF;AAEA,QAAM,aAAa,SAAS;AAC5B,MAAI,cAAc,OAAO,KAAK,UAAU,EAAE,SAAS,GAAG;AACpD,SAAK,aAAa,EAAE,GAAG,WAAW;AAAA,EACpC;AAEA,SAAO;AACT;AAEO,SAAS,gBAAgB,SAA6B,QAAqC;AAChG,QAAM,OAAO,WAAW;AACxB,MAAI,CAAC,UAAU,OAAO,WAAW,GAAG;AAClC,WAAO;AAAA,EACT;AACA,QAAM,QAAQ,OAAO,IAAI,CAAC,UAAU,MAAM,YAAY,EAAE,EAAE,OAAO,CAAC,aAAa,SAAS,SAAS,CAAC;AAClG,MAAI,MAAM,WAAW,GAAG;AACtB,WAAO;AAAA,EACT;AACA,SAAO,CAAC,MAAM,GAAG,KAAK,EAAE,KAAK,iBAAiB;AAChD;AAEO,SAAS,cAAc,UAAwB,QAAyC;AAC7F,QAAM,WAAW,gBAAgB,SAAS,QAAQ;AAElD,WAAS,YAAY,SAAS;AAC9B,MAAI,SAAS,gBAAgB,MAAM;AACjC,aAAS,gBAAgB,SAAS;AAAA,EACpC;AACA,MAAI,SAAS,qBAAqB,SAAS,kBAAkB,SAAS,GAAG;AACvE,aAAS,qBAAqB,SAAS;AAAA,EACzC;AACA,MAAI,SAAS,UAAU,MAAM;AAC3B,aAAS,aAAa,SAAS,OAAO;AAAA,EACxC;AAEA,MAAI,SAAS,qBAAqB,SAAS,kBAAkB,SAAS,GAAG;AACvE,aAAS,qBAAqB,SAAS,kBAAkB,IAAI,CAAC,aAAa;AAAA,MACzE,MAAM,QAAQ;AAAA,MACd,OAAO,QAAQ;AAAA,MACf,WAAW,OAAO,QAAQ,SAAS;AAAA,IACrC,EAAE;AAAA,EACJ;AAEA,QAAM,SAAS,SAAS,UAAU,CAAC;AACnC,WAAS,cAAc,OAAO;AAC9B,MAAI,OAAO,SAAS,GAAG;AACrB,aAAS,SAAS,OAAO,IAAI,CAAC,WAAW;AAAA,MACvC,OAAO,MAAM;AAAA,MACb,UAAU,MAAM;AAAA,MAChB,aAAa,MAAM;AAAA,IACrB,EAAE;AAAA,EACJ;AAEA,MAAI,SAAS,sBAAsB,SAAS,mBAAmB,SAAS,GAAG;AACzE,aAAS,sBAAsB,SAAS,mBAAmB,IAAI,CAAC,aAAa;AAAA,MAC3E,QAAQ,QAAQ;AAAA,MAChB,SAAS,QAAQ;AAAA,IACnB,EAAE;AAAA,EACJ;AAEA,WAAS,SAAS;AAClB,SAAO;AACT;AAEA,SAAS,kBAAkB,UAAwB,QAA4B;AAC7E,QAAM,eAAe,cAAc,UAAU,MAAM;AACnD,QAAM,YAAwB,CAAC;AAE/B,aAAW,SAAS,SAAS,UAAU,CAAC,GAAG;AACzC,UAAM,WAAoC,EAAE,GAAG,aAAa;AAC5D,UAAM,gBAAgB,MAAM;AAC5B,aAAS,cAAc,cAAc;AACrC,aAAS,eAAe,cAAc;AACtC,aAAS,aAAa,OAAO,MAAM,SAAS;AAC5C,QAAI,cAAc,eAAe,cAAc,YAAY,SAAS,GAAG;AACrE,eAAS,eAAe,CAAC,GAAG,cAAc,WAAW;AAAA,IACvD;AACA,QAAI,cAAc,cAAc,MAAM;AACpC,eAAS,cAAc,cAAc;AAAA,IACvC;AACA,QAAI,cAAc,aAAa,MAAM;AAEnC,eAAS,OAAO,cAAc,YAAY;AAC1C,eAAS,aAAa,cAAc;AAAA,IACtC;AACA,QAAI,cAAc,YAAY,MAAM;AAClC,eAAS,YAAY,cAAc;AAAA,IACrC;AACA,cAAU,KAAK,IAAI,0BAAS,EAAE,aAAa,MAAM,SAAS,SAAS,CAAC,CAAC;AAAA,EACvE;AAEA,SAAO;AACT;AAEA,SAAS,iBAAiB,UAAwB,QAA4B;AAC5E,QAAM,eAAe,cAAc,UAAU,MAAM;AACnD,QAAM,YAAwB,CAAC;AAE/B,aAAW,QAAQ,SAAS,SAAS,CAAC,GAAG;AACvC,UAAM,WAAoC,EAAE,GAAG,aAAa;AAE5D,aAAS,OAAO,KAAK,aAAa;AAClC,QAAI,KAAK,WAAW,MAAM;AACxB,eAAS,WAAW,KAAK;AAAA,IAC3B;AACA,UAAM,cAAc,gBAAgB,KAAK,SAAS,KAAK,MAAM;AAC7D,cAAU,KAAK,IAAI,0BAAS,EAAE,aAAa,SAAS,CAAC,CAAC;AAAA,EACxD;AAEA,SAAO;AACT;AAEO,SAAS,oBAAoB,UAAwB,QAAgB,SAAmC;AAC7G,MAAI,QAAQ,YAAY,SAAS,UAAU,SAAS,OAAO,SAAS,GAAG;AACrE,WAAO,kBAAkB,UAAU,MAAM;AAAA,EAC3C;AACA,MAAI,QAAQ,WAAW,SAAS,SAAS,SAAS,MAAM,SAAS,GAAG;AAClE,WAAO,iBAAiB,UAAU,MAAM;AAAA,EAC1C;AACA,QAAM,WAAW,cAAc,UAAU,MAAM;AAC/C,QAAM,cAAc,gBAAgB,SAAS,SAAS,SAAS,MAAM;AACrE,SAAO,CAAC,IAAI,0BAAS,EAAE,aAAa,SAAS,CAAC,CAAC;AACjD;AAEO,SAAS,kBAAkB,QAAwB,SAAmB,SAAmC;AAC9G,MAAI,OAAO,UAAU,OAAO,OAAO,SAAS,GAAG;AAC7C,UAAM,QAAQ,OAAO,OAAO,CAAC;AAC7B,UAAM,IAAI,MAAM,sBAAsB,MAAM,MAAM,MAAM,MAAM,OAAO,EAAE;AAAA,EACzE;AAEA,QAAM,YAAwB,CAAC;AAC/B,QAAM,UAAU,OAAO,WAAW,CAAC;AACnC,UAAQ,QAAQ,CAAC,UAAU,UAAU;AACnC,UAAM,SAAS,QAAQ,KAAK,MAAM,QAAQ,SAAS,IAAI,QAAQ,QAAQ,SAAS,CAAC,IAAI;AACrF,cAAU,KAAK,GAAG,oBAAoB,UAAU,QAAQ,OAAO,CAAC;AAAA,EAClE,CAAC;AACD,SAAO;AACT;;;ADlOA,IAAM,eAAe;AAuBd,IAAM,cAAN,cAA0B,+BAAmB;AAAA,EACjC;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EAEjB,YAAY,SAA6B;AACvC,UAAM;AACN,UAAM,EAAE,UAAU,MAAM,UAAU,MAAM,OAAO,IAAI;AACnD,QAAI,aAAa,UAAa,SAAS,QAAW;AAChD,YAAM,IAAI,MAAM,+CAA+C;AAAA,IACjE;AACA,QAAI,aAAa,UAAa,SAAS,QAAW;AAChD,YAAM,IAAI,MAAM,kEAAkE;AAAA,IACpF;AACA,QAAI,SAAS,UAAa,aAAa,QAAW;AAChD,YAAM,IAAI,MAAM,2CAA2C;AAAA,IAC7D;AAEA,SAAK,WAAW;AAChB,SAAK,OAAO;AACZ,SAAK,WAAW;AAChB,SAAK,OAAO;AACZ,SAAK,SAAS;AAAA,EAChB;AAAA,EAEA,MAAM,OAA4B;AAChC,UAAM,EAAE,QAAQ,SAAS,MAAM,IAAI,MAAM,KAAK,YAAY;AAC1D,QAAI,OAAO,WAAW,GAAG;AACvB,aAAO,CAAC;AAAA,IACV;AAEA,QAAI;AACJ,QAAI;AACF,eAAS,QAAQ,UAAM,2BAAa,QAAQ,KAAK,UAAU,IAAI,IAAI,UAAM,sBAAQ,OAAO,CAAC,GAAG,KAAK,UAAU,IAAI;AAAA,IACjH,SAAS,OAAO;AACd,YAAM,SAAS,QAAQ,CAAC,KAAK;AAC7B,YAAM,IAAI,MAAM,sBAAsB,MAAM,MAAM,aAAa,KAAK,CAAC,EAAE;AAAA,IACzE;AAEA,WAAO,kBAAkB,QAAqC,SAAS;AAAA,MACrE,UAAU,kBAAkB,KAAK,MAAM;AAAA,MACvC,SAAS,iBAAiB,KAAK,MAAM;AAAA,IACvC,CAAC;AAAA,EACH;AAAA,EAEA,MAAc,cAAsF;AAClG,QAAI,KAAK,SAAS,QAAW;AAC3B,YAAM,SAAS,WAAW,KAAK,QAAQ;AACvC,YAAM,QAAsB,EAAE,MAAM,SAAS,OAAO,KAAK,MAAM,UAAU,KAAK,SAAS;AACvF,aAAO,EAAE,QAAQ,CAAC,KAAK,GAAG,SAAS,CAAC,MAAM,GAAG,OAAO,MAAM;AAAA,IAC5D;AAEA,UAAM,EAAE,OAAO,MAAM,IAAI,MAAM,KAAK,aAAa;AACjD,UAAM,SAAyB,MAAM,IAAI,CAAC,UAAU,EAAE,MAAM,OAAO,KAAK,MAAM,UAAU,KAAK,SAAS,EAAE;AACxG,WAAO,EAAE,QAAQ,SAAS,OAAO,MAAM;AAAA,EACzC;AAAA,EAEA,MAAc,eAA6D;AACzE,UAAM,WAAW,KAAK;AACtB,QAAI,MAAM,QAAQ,QAAQ,GAAG;AAC3B,aAAO,EAAE,OAAO,UAAU,OAAO,KAAK;AAAA,IACxC;AACA,QAAI,OAAO,aAAa,UAAU;AAChC,UAAI,MAAM,YAAY,QAAQ,GAAG;AAC/B,cAAM,UAAU,KAAK,QAAQ;AAC7B,cAAM,UAAU,UAAM,iBAAAA,SAAS,SAAS,EAAE,KAAK,UAAU,WAAW,MAAM,UAAU,KAAK,CAAC;AAC1F,eAAO,EAAE,OAAO,QAAQ,KAAK,GAAG,OAAO,KAAK;AAAA,MAC9C;AACA,aAAO,EAAE,OAAO,CAAC,QAAQ,GAAG,OAAO,MAAM;AAAA,IAC3C;AACA,WAAO,EAAE,OAAO,CAAC,GAAG,OAAO,MAAM;AAAA,EACnC;AACF;AAEA,eAAe,YAAY,MAAgC;AACzD,MAAI;AACF,UAAM,QAAQ,UAAM,sBAAK,IAAI;AAC7B,WAAO,MAAM,YAAY;AAAA,EAC3B,QAAQ;AACN,WAAO;AAAA,EACT;AACF;AAEA,SAAS,aAAa,OAAwB;AAC5C,SAAO,iBAAiB,QAAQ,MAAM,UAAU,OAAO,KAAK;AAC9D;","names":["fastGlob"]}
|
package/dist/index.d.cts
ADDED
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
import { BaseDocumentLoader } from '@langchain/core/document_loaders/base';
|
|
2
|
+
import { Document } from '@langchain/core/documents';
|
|
3
|
+
import { ExtractionConfig } from '@xberg-io/xberg';
|
|
4
|
+
|
|
5
|
+
interface XbergLoaderOptions {
|
|
6
|
+
/** File path, list of file paths, or directory path to load. */
|
|
7
|
+
filePath?: string | string[];
|
|
8
|
+
/** Raw bytes to extract from. Mutually exclusive with `filePath`. */
|
|
9
|
+
data?: Uint8Array;
|
|
10
|
+
/** MIME type hint. Required when using `data`, optional for `filePath`. */
|
|
11
|
+
mimeType?: string;
|
|
12
|
+
/** Glob pattern for directory mode. Defaults to matching all files. */
|
|
13
|
+
glob?: string;
|
|
14
|
+
/** Xberg extraction configuration controlling output format, OCR, pages, chunking, etc. */
|
|
15
|
+
config?: ExtractionConfig;
|
|
16
|
+
}
|
|
17
|
+
/**
|
|
18
|
+
* Load documents using Xberg, supporting 90+ file formats with true async extraction.
|
|
19
|
+
*
|
|
20
|
+
* By default each source becomes one Document. Enable `chunking` on the
|
|
21
|
+
* `ExtractionConfig` to emit one Document per chunk, or `pages` for one Document
|
|
22
|
+
* per page. Multiple paths (a list or a directory glob) are extracted with a single
|
|
23
|
+
* `extractBatch` call so concurrency happens Rust-side.
|
|
24
|
+
*/
|
|
25
|
+
declare class XbergLoader extends BaseDocumentLoader {
|
|
26
|
+
private readonly filePath?;
|
|
27
|
+
private readonly data?;
|
|
28
|
+
private readonly mimeType?;
|
|
29
|
+
private readonly glob?;
|
|
30
|
+
private readonly config?;
|
|
31
|
+
constructor(options: XbergLoaderOptions);
|
|
32
|
+
load(): Promise<Document[]>;
|
|
33
|
+
private buildInputs;
|
|
34
|
+
private resolvePaths;
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
export { XbergLoader, type XbergLoaderOptions };
|
package/dist/index.d.ts
ADDED
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
import { BaseDocumentLoader } from '@langchain/core/document_loaders/base';
|
|
2
|
+
import { Document } from '@langchain/core/documents';
|
|
3
|
+
import { ExtractionConfig } from '@xberg-io/xberg';
|
|
4
|
+
|
|
5
|
+
interface XbergLoaderOptions {
|
|
6
|
+
/** File path, list of file paths, or directory path to load. */
|
|
7
|
+
filePath?: string | string[];
|
|
8
|
+
/** Raw bytes to extract from. Mutually exclusive with `filePath`. */
|
|
9
|
+
data?: Uint8Array;
|
|
10
|
+
/** MIME type hint. Required when using `data`, optional for `filePath`. */
|
|
11
|
+
mimeType?: string;
|
|
12
|
+
/** Glob pattern for directory mode. Defaults to matching all files. */
|
|
13
|
+
glob?: string;
|
|
14
|
+
/** Xberg extraction configuration controlling output format, OCR, pages, chunking, etc. */
|
|
15
|
+
config?: ExtractionConfig;
|
|
16
|
+
}
|
|
17
|
+
/**
|
|
18
|
+
* Load documents using Xberg, supporting 90+ file formats with true async extraction.
|
|
19
|
+
*
|
|
20
|
+
* By default each source becomes one Document. Enable `chunking` on the
|
|
21
|
+
* `ExtractionConfig` to emit one Document per chunk, or `pages` for one Document
|
|
22
|
+
* per page. Multiple paths (a list or a directory glob) are extracted with a single
|
|
23
|
+
* `extractBatch` call so concurrency happens Rust-side.
|
|
24
|
+
*/
|
|
25
|
+
declare class XbergLoader extends BaseDocumentLoader {
|
|
26
|
+
private readonly filePath?;
|
|
27
|
+
private readonly data?;
|
|
28
|
+
private readonly mimeType?;
|
|
29
|
+
private readonly glob?;
|
|
30
|
+
private readonly config?;
|
|
31
|
+
constructor(options: XbergLoaderOptions);
|
|
32
|
+
load(): Promise<Document[]>;
|
|
33
|
+
private buildInputs;
|
|
34
|
+
private resolvePaths;
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
export { XbergLoader, type XbergLoaderOptions };
|
package/dist/index.js
ADDED
|
@@ -0,0 +1,247 @@
|
|
|
1
|
+
// src/loader.ts
|
|
2
|
+
import { stat } from "fs/promises";
|
|
3
|
+
import { BaseDocumentLoader } from "@langchain/core/document_loaders/base";
|
|
4
|
+
import { extract, extractBatch } from "@xberg-io/xberg";
|
|
5
|
+
import fastGlob from "fast-glob";
|
|
6
|
+
|
|
7
|
+
// src/mapping.ts
|
|
8
|
+
import { Document } from "@langchain/core/documents";
|
|
9
|
+
var METADATA_FIELDS = [
|
|
10
|
+
["title", "title"],
|
|
11
|
+
["subject", "subject"],
|
|
12
|
+
["authors", "authors"],
|
|
13
|
+
["keywords", "keywords"],
|
|
14
|
+
["language", "language"],
|
|
15
|
+
["created_at", "createdAt"],
|
|
16
|
+
["modified_at", "modifiedAt"],
|
|
17
|
+
["created_by", "createdBy"],
|
|
18
|
+
["modified_by", "modifiedBy"],
|
|
19
|
+
["category", "category"],
|
|
20
|
+
["tags", "tags"],
|
|
21
|
+
["document_version", "documentVersion"],
|
|
22
|
+
["abstract_text", "abstractText"],
|
|
23
|
+
["output_format", "outputFormat"],
|
|
24
|
+
["ocr_used", "ocrUsed"],
|
|
25
|
+
["extraction_duration_ms", "extractionDurationMs"]
|
|
26
|
+
];
|
|
27
|
+
var CONTENT_SEPARATOR = "\n\n";
|
|
28
|
+
function isChunkingEnabled(config) {
|
|
29
|
+
return config?.chunking != null;
|
|
30
|
+
}
|
|
31
|
+
function isPerPageEnabled(config) {
|
|
32
|
+
return Boolean(config?.pages?.extractPages);
|
|
33
|
+
}
|
|
34
|
+
function flattenMetadata(metadata) {
|
|
35
|
+
const flat = {};
|
|
36
|
+
if (!metadata) {
|
|
37
|
+
return flat;
|
|
38
|
+
}
|
|
39
|
+
for (const [snakeKey, camelKey] of METADATA_FIELDS) {
|
|
40
|
+
const value = metadata[camelKey];
|
|
41
|
+
if (value != null) {
|
|
42
|
+
flat[snakeKey] = value;
|
|
43
|
+
}
|
|
44
|
+
}
|
|
45
|
+
const additional = metadata.additional;
|
|
46
|
+
if (additional && Object.keys(additional).length > 0) {
|
|
47
|
+
flat.additional = { ...additional };
|
|
48
|
+
}
|
|
49
|
+
return flat;
|
|
50
|
+
}
|
|
51
|
+
function assembleContent(content, tables) {
|
|
52
|
+
const text = content ?? "";
|
|
53
|
+
if (!tables || tables.length === 0) {
|
|
54
|
+
return text;
|
|
55
|
+
}
|
|
56
|
+
const parts = tables.map((table) => table.markdown ?? "").filter((markdown) => markdown.length > 0);
|
|
57
|
+
if (parts.length === 0) {
|
|
58
|
+
return text;
|
|
59
|
+
}
|
|
60
|
+
return [text, ...parts].join(CONTENT_SEPARATOR);
|
|
61
|
+
}
|
|
62
|
+
function buildMetadata(document, source) {
|
|
63
|
+
const metadata = flattenMetadata(document.metadata);
|
|
64
|
+
metadata.mime_type = document.mimeType;
|
|
65
|
+
if (document.qualityScore != null) {
|
|
66
|
+
metadata.quality_score = document.qualityScore;
|
|
67
|
+
}
|
|
68
|
+
if (document.detectedLanguages && document.detectedLanguages.length > 0) {
|
|
69
|
+
metadata.detected_languages = document.detectedLanguages;
|
|
70
|
+
}
|
|
71
|
+
if (document.counts != null) {
|
|
72
|
+
metadata.page_count = document.counts.pages;
|
|
73
|
+
}
|
|
74
|
+
if (document.extractedKeywords && document.extractedKeywords.length > 0) {
|
|
75
|
+
metadata.extracted_keywords = document.extractedKeywords.map((keyword) => ({
|
|
76
|
+
text: keyword.text,
|
|
77
|
+
score: keyword.score,
|
|
78
|
+
algorithm: String(keyword.algorithm)
|
|
79
|
+
}));
|
|
80
|
+
}
|
|
81
|
+
const tables = document.tables ?? [];
|
|
82
|
+
metadata.table_count = tables.length;
|
|
83
|
+
if (tables.length > 0) {
|
|
84
|
+
metadata.tables = tables.map((table) => ({
|
|
85
|
+
cells: table.cells,
|
|
86
|
+
markdown: table.markdown,
|
|
87
|
+
page_number: table.pageNumber
|
|
88
|
+
}));
|
|
89
|
+
}
|
|
90
|
+
if (document.processingWarnings && document.processingWarnings.length > 0) {
|
|
91
|
+
metadata.processing_warnings = document.processingWarnings.map((warning) => ({
|
|
92
|
+
source: warning.source,
|
|
93
|
+
message: warning.message
|
|
94
|
+
}));
|
|
95
|
+
}
|
|
96
|
+
metadata.source = source;
|
|
97
|
+
return metadata;
|
|
98
|
+
}
|
|
99
|
+
function chunksToDocuments(document, source) {
|
|
100
|
+
const baseMetadata = buildMetadata(document, source);
|
|
101
|
+
const documents = [];
|
|
102
|
+
for (const chunk of document.chunks ?? []) {
|
|
103
|
+
const metadata = { ...baseMetadata };
|
|
104
|
+
const chunkMetadata = chunk.metadata;
|
|
105
|
+
metadata.chunk_index = chunkMetadata.chunkIndex;
|
|
106
|
+
metadata.total_chunks = chunkMetadata.totalChunks;
|
|
107
|
+
metadata.chunk_type = String(chunk.chunkType);
|
|
108
|
+
if (chunkMetadata.headingPath && chunkMetadata.headingPath.length > 0) {
|
|
109
|
+
metadata.heading_path = [...chunkMetadata.headingPath];
|
|
110
|
+
}
|
|
111
|
+
if (chunkMetadata.tokenCount != null) {
|
|
112
|
+
metadata.token_count = chunkMetadata.tokenCount;
|
|
113
|
+
}
|
|
114
|
+
if (chunkMetadata.firstPage != null) {
|
|
115
|
+
metadata.page = chunkMetadata.firstPage - 1;
|
|
116
|
+
metadata.first_page = chunkMetadata.firstPage;
|
|
117
|
+
}
|
|
118
|
+
if (chunkMetadata.lastPage != null) {
|
|
119
|
+
metadata.last_page = chunkMetadata.lastPage;
|
|
120
|
+
}
|
|
121
|
+
documents.push(new Document({ pageContent: chunk.content, metadata }));
|
|
122
|
+
}
|
|
123
|
+
return documents;
|
|
124
|
+
}
|
|
125
|
+
function pagesToDocuments(document, source) {
|
|
126
|
+
const baseMetadata = buildMetadata(document, source);
|
|
127
|
+
const documents = [];
|
|
128
|
+
for (const page of document.pages ?? []) {
|
|
129
|
+
const metadata = { ...baseMetadata };
|
|
130
|
+
metadata.page = page.pageNumber - 1;
|
|
131
|
+
if (page.isBlank != null) {
|
|
132
|
+
metadata.is_blank = page.isBlank;
|
|
133
|
+
}
|
|
134
|
+
const pageContent = assembleContent(page.content, page.tables);
|
|
135
|
+
documents.push(new Document({ pageContent, metadata }));
|
|
136
|
+
}
|
|
137
|
+
return documents;
|
|
138
|
+
}
|
|
139
|
+
function documentToDocuments(document, source, options) {
|
|
140
|
+
if (options.chunking && document.chunks && document.chunks.length > 0) {
|
|
141
|
+
return chunksToDocuments(document, source);
|
|
142
|
+
}
|
|
143
|
+
if (options.perPage && document.pages && document.pages.length > 0) {
|
|
144
|
+
return pagesToDocuments(document, source);
|
|
145
|
+
}
|
|
146
|
+
const metadata = buildMetadata(document, source);
|
|
147
|
+
const pageContent = assembleContent(document.content, document.tables);
|
|
148
|
+
return [new Document({ pageContent, metadata })];
|
|
149
|
+
}
|
|
150
|
+
function resultToDocuments(result, sources, options) {
|
|
151
|
+
if (result.errors && result.errors.length > 0) {
|
|
152
|
+
const error = result.errors[0];
|
|
153
|
+
throw new Error(`Failed to extract '${error.source}': ${error.message}`);
|
|
154
|
+
}
|
|
155
|
+
const documents = [];
|
|
156
|
+
const results = result.results ?? [];
|
|
157
|
+
results.forEach((document, index) => {
|
|
158
|
+
const source = sources[index] ?? (sources.length > 0 ? sources[sources.length - 1] : "");
|
|
159
|
+
documents.push(...documentToDocuments(document, source, options));
|
|
160
|
+
});
|
|
161
|
+
return documents;
|
|
162
|
+
}
|
|
163
|
+
|
|
164
|
+
// src/loader.ts
|
|
165
|
+
var DEFAULT_GLOB = "**/*";
|
|
166
|
+
var XbergLoader = class extends BaseDocumentLoader {
|
|
167
|
+
filePath;
|
|
168
|
+
data;
|
|
169
|
+
mimeType;
|
|
170
|
+
glob;
|
|
171
|
+
config;
|
|
172
|
+
constructor(options) {
|
|
173
|
+
super();
|
|
174
|
+
const { filePath, data, mimeType, glob, config } = options;
|
|
175
|
+
if (filePath === void 0 && data === void 0) {
|
|
176
|
+
throw new Error("Either 'filePath' or 'data' must be provided.");
|
|
177
|
+
}
|
|
178
|
+
if (filePath !== void 0 && data !== void 0) {
|
|
179
|
+
throw new Error("Cannot specify both 'filePath' and 'data'. Use one or the other.");
|
|
180
|
+
}
|
|
181
|
+
if (data !== void 0 && mimeType === void 0) {
|
|
182
|
+
throw new Error("'mimeType' is required when using 'data'.");
|
|
183
|
+
}
|
|
184
|
+
this.filePath = filePath;
|
|
185
|
+
this.data = data;
|
|
186
|
+
this.mimeType = mimeType;
|
|
187
|
+
this.glob = glob;
|
|
188
|
+
this.config = config;
|
|
189
|
+
}
|
|
190
|
+
async load() {
|
|
191
|
+
const { inputs, sources, batch } = await this.buildInputs();
|
|
192
|
+
if (inputs.length === 0) {
|
|
193
|
+
return [];
|
|
194
|
+
}
|
|
195
|
+
let result;
|
|
196
|
+
try {
|
|
197
|
+
result = batch ? await extractBatch(inputs, this.config ?? null) : await extract(inputs[0], this.config ?? null);
|
|
198
|
+
} catch (error) {
|
|
199
|
+
const source = sources[0] ?? "input";
|
|
200
|
+
throw new Error(`Failed to extract '${source}': ${errorMessage(error)}`);
|
|
201
|
+
}
|
|
202
|
+
return resultToDocuments(result, sources, {
|
|
203
|
+
chunking: isChunkingEnabled(this.config),
|
|
204
|
+
perPage: isPerPageEnabled(this.config)
|
|
205
|
+
});
|
|
206
|
+
}
|
|
207
|
+
async buildInputs() {
|
|
208
|
+
if (this.data !== void 0) {
|
|
209
|
+
const source = `bytes://${this.mimeType}`;
|
|
210
|
+
const input = { kind: "bytes", bytes: this.data, mimeType: this.mimeType };
|
|
211
|
+
return { inputs: [input], sources: [source], batch: false };
|
|
212
|
+
}
|
|
213
|
+
const { paths, batch } = await this.resolvePaths();
|
|
214
|
+
const inputs = paths.map((path) => ({ kind: "uri", uri: path, mimeType: this.mimeType }));
|
|
215
|
+
return { inputs, sources: paths, batch };
|
|
216
|
+
}
|
|
217
|
+
async resolvePaths() {
|
|
218
|
+
const filePath = this.filePath;
|
|
219
|
+
if (Array.isArray(filePath)) {
|
|
220
|
+
return { paths: filePath, batch: true };
|
|
221
|
+
}
|
|
222
|
+
if (typeof filePath === "string") {
|
|
223
|
+
if (await isDirectory(filePath)) {
|
|
224
|
+
const pattern = this.glob ?? DEFAULT_GLOB;
|
|
225
|
+
const matches = await fastGlob(pattern, { cwd: filePath, onlyFiles: true, absolute: true });
|
|
226
|
+
return { paths: matches.sort(), batch: true };
|
|
227
|
+
}
|
|
228
|
+
return { paths: [filePath], batch: false };
|
|
229
|
+
}
|
|
230
|
+
return { paths: [], batch: false };
|
|
231
|
+
}
|
|
232
|
+
};
|
|
233
|
+
async function isDirectory(path) {
|
|
234
|
+
try {
|
|
235
|
+
const stats = await stat(path);
|
|
236
|
+
return stats.isDirectory();
|
|
237
|
+
} catch {
|
|
238
|
+
return false;
|
|
239
|
+
}
|
|
240
|
+
}
|
|
241
|
+
function errorMessage(error) {
|
|
242
|
+
return error instanceof Error ? error.message : String(error);
|
|
243
|
+
}
|
|
244
|
+
export {
|
|
245
|
+
XbergLoader
|
|
246
|
+
};
|
|
247
|
+
//# sourceMappingURL=index.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"sources":["../src/loader.ts","../src/mapping.ts"],"sourcesContent":["import { stat } from \"node:fs/promises\";\nimport { BaseDocumentLoader } from \"@langchain/core/document_loaders/base\";\nimport type { Document } from \"@langchain/core/documents\";\nimport { extract, extractBatch } from \"@xberg-io/xberg\";\nimport type { ExtractInput, ExtractionConfig } from \"@xberg-io/xberg\";\nimport fastGlob from \"fast-glob\";\nimport { isChunkingEnabled, isPerPageEnabled, resultToDocuments, type ResultEnvelope } from \"./mapping\";\n\nconst DEFAULT_GLOB = \"**/*\";\n\nexport interface XbergLoaderOptions {\n /** File path, list of file paths, or directory path to load. */\n filePath?: string | string[];\n /** Raw bytes to extract from. Mutually exclusive with `filePath`. */\n data?: Uint8Array;\n /** MIME type hint. Required when using `data`, optional for `filePath`. */\n mimeType?: string;\n /** Glob pattern for directory mode. Defaults to matching all files. */\n glob?: string;\n /** Xberg extraction configuration controlling output format, OCR, pages, chunking, etc. */\n config?: ExtractionConfig;\n}\n\n/**\n * Load documents using Xberg, supporting 90+ file formats with true async extraction.\n *\n * By default each source becomes one Document. Enable `chunking` on the\n * `ExtractionConfig` to emit one Document per chunk, or `pages` for one Document\n * per page. Multiple paths (a list or a directory glob) are extracted with a single\n * `extractBatch` call so concurrency happens Rust-side.\n */\nexport class XbergLoader extends BaseDocumentLoader {\n private readonly filePath?: string | string[];\n private readonly data?: Uint8Array;\n private readonly mimeType?: string;\n private readonly glob?: string;\n private readonly config?: ExtractionConfig;\n\n constructor(options: XbergLoaderOptions) {\n super();\n const { filePath, data, mimeType, glob, config } = options;\n if (filePath === undefined && data === undefined) {\n throw new Error(\"Either 'filePath' or 'data' must be provided.\");\n }\n if (filePath !== undefined && data !== undefined) {\n throw new Error(\"Cannot specify both 'filePath' and 'data'. Use one or the other.\");\n }\n if (data !== undefined && mimeType === undefined) {\n throw new Error(\"'mimeType' is required when using 'data'.\");\n }\n\n this.filePath = filePath;\n this.data = data;\n this.mimeType = mimeType;\n this.glob = glob;\n this.config = config;\n }\n\n async load(): Promise<Document[]> {\n const { inputs, sources, batch } = await this.buildInputs();\n if (inputs.length === 0) {\n return [];\n }\n\n let result;\n try {\n result = batch ? await extractBatch(inputs, this.config ?? null) : await extract(inputs[0], this.config ?? null);\n } catch (error) {\n const source = sources[0] ?? \"input\";\n throw new Error(`Failed to extract '${source}': ${errorMessage(error)}`);\n }\n\n return resultToDocuments(result as unknown as ResultEnvelope, sources, {\n chunking: isChunkingEnabled(this.config),\n perPage: isPerPageEnabled(this.config),\n });\n }\n\n private async buildInputs(): Promise<{ inputs: ExtractInput[]; sources: string[]; batch: boolean }> {\n if (this.data !== undefined) {\n const source = `bytes://${this.mimeType}`;\n const input: ExtractInput = { kind: \"bytes\", bytes: this.data, mimeType: this.mimeType };\n return { inputs: [input], sources: [source], batch: false };\n }\n\n const { paths, batch } = await this.resolvePaths();\n const inputs: ExtractInput[] = paths.map((path) => ({ kind: \"uri\", uri: path, mimeType: this.mimeType }));\n return { inputs, sources: paths, batch };\n }\n\n private async resolvePaths(): Promise<{ paths: string[]; batch: boolean }> {\n const filePath = this.filePath;\n if (Array.isArray(filePath)) {\n return { paths: filePath, batch: true };\n }\n if (typeof filePath === \"string\") {\n if (await isDirectory(filePath)) {\n const pattern = this.glob ?? DEFAULT_GLOB;\n const matches = await fastGlob(pattern, { cwd: filePath, onlyFiles: true, absolute: true });\n return { paths: matches.sort(), batch: true };\n }\n return { paths: [filePath], batch: false };\n }\n return { paths: [], batch: false };\n }\n}\n\nasync function isDirectory(path: string): Promise<boolean> {\n try {\n const stats = await stat(path);\n return stats.isDirectory();\n } catch {\n return false;\n }\n}\n\nfunction errorMessage(error: unknown): string {\n return error instanceof Error ? error.message : String(error);\n}\n","import { Document } from \"@langchain/core/documents\";\nimport type { DocumentCounts, ExtractionConfig, Keyword, Metadata, ProcessingWarning, Table } from \"@xberg-io/xberg\";\n\n// Metadata fields (from Xberg Metadata) that carry JSON-friendly scalar/list values,\n// mapped from the binding's camelCase keys to the snake_case output keys used by the\n// Python adapter. Opaque nested fields (pages, format, imagePreprocessing, jsonSchema,\n// error) are skipped because they are native objects, not plain data. ~keep\nconst METADATA_FIELDS: ReadonlyArray<readonly [string, keyof Metadata]> = [\n [\"title\", \"title\"],\n [\"subject\", \"subject\"],\n [\"authors\", \"authors\"],\n [\"keywords\", \"keywords\"],\n [\"language\", \"language\"],\n [\"created_at\", \"createdAt\"],\n [\"modified_at\", \"modifiedAt\"],\n [\"created_by\", \"createdBy\"],\n [\"modified_by\", \"modifiedBy\"],\n [\"category\", \"category\"],\n [\"tags\", \"tags\"],\n [\"document_version\", \"documentVersion\"],\n [\"abstract_text\", \"abstractText\"],\n [\"output_format\", \"outputFormat\"],\n [\"ocr_used\", \"ocrUsed\"],\n [\"extraction_duration_ms\", \"extractionDurationMs\"],\n];\n\nconst CONTENT_SEPARATOR = \"\\n\\n\";\n\nexport interface ChunkMetadataShape {\n chunkIndex: number;\n totalChunks: number;\n headingPath?: string[];\n tokenCount?: number;\n firstPage?: number;\n lastPage?: number;\n}\n\nexport interface ChunkShape {\n content: string;\n chunkType: unknown;\n metadata: ChunkMetadataShape;\n}\n\nexport interface PageShape {\n pageNumber: number;\n content: string;\n tables?: Table[];\n isBlank?: boolean;\n}\n\n// Structural view of a binding ExtractedDocument. The binding types its own fields\n// through unexported `Js*` aliases; this interface uses the clean exported types plus\n// local chunk/page shapes so the mapper is fully typed and testable without the addon. ~keep\nexport interface ExtractedDoc {\n content?: string;\n mimeType?: string;\n metadata?: Metadata;\n tables?: Table[];\n counts?: DocumentCounts;\n detectedLanguages?: string[];\n chunks?: ChunkShape[];\n pages?: PageShape[];\n extractedKeywords?: Keyword[];\n qualityScore?: number;\n processingWarnings?: ProcessingWarning[];\n}\n\nexport interface ResultEnvelope {\n errors?: Array<{ source: string; message: string }>;\n results?: ExtractedDoc[];\n}\n\nexport interface SplitOptions {\n chunking: boolean;\n perPage: boolean;\n}\n\nexport function isChunkingEnabled(config?: ExtractionConfig | null): boolean {\n return config?.chunking != null;\n}\n\nexport function isPerPageEnabled(config?: ExtractionConfig | null): boolean {\n return Boolean(config?.pages?.extractPages);\n}\n\nexport function flattenMetadata(metadata?: Metadata): Record<string, unknown> {\n const flat: Record<string, unknown> = {};\n if (!metadata) {\n return flat;\n }\n\n for (const [snakeKey, camelKey] of METADATA_FIELDS) {\n const value = metadata[camelKey];\n if (value != null) {\n flat[snakeKey] = value;\n }\n }\n\n const additional = metadata.additional;\n if (additional && Object.keys(additional).length > 0) {\n flat.additional = { ...additional };\n }\n\n return flat;\n}\n\nexport function assembleContent(content: string | undefined, tables: Table[] | undefined): string {\n const text = content ?? \"\";\n if (!tables || tables.length === 0) {\n return text;\n }\n const parts = tables.map((table) => table.markdown ?? \"\").filter((markdown) => markdown.length > 0);\n if (parts.length === 0) {\n return text;\n }\n return [text, ...parts].join(CONTENT_SEPARATOR);\n}\n\nexport function buildMetadata(document: ExtractedDoc, source: string): Record<string, unknown> {\n const metadata = flattenMetadata(document.metadata);\n\n metadata.mime_type = document.mimeType;\n if (document.qualityScore != null) {\n metadata.quality_score = document.qualityScore;\n }\n if (document.detectedLanguages && document.detectedLanguages.length > 0) {\n metadata.detected_languages = document.detectedLanguages;\n }\n if (document.counts != null) {\n metadata.page_count = document.counts.pages;\n }\n\n if (document.extractedKeywords && document.extractedKeywords.length > 0) {\n metadata.extracted_keywords = document.extractedKeywords.map((keyword) => ({\n text: keyword.text,\n score: keyword.score,\n algorithm: String(keyword.algorithm),\n }));\n }\n\n const tables = document.tables ?? [];\n metadata.table_count = tables.length;\n if (tables.length > 0) {\n metadata.tables = tables.map((table) => ({\n cells: table.cells,\n markdown: table.markdown,\n page_number: table.pageNumber,\n }));\n }\n\n if (document.processingWarnings && document.processingWarnings.length > 0) {\n metadata.processing_warnings = document.processingWarnings.map((warning) => ({\n source: warning.source,\n message: warning.message,\n }));\n }\n\n metadata.source = source;\n return metadata;\n}\n\nfunction chunksToDocuments(document: ExtractedDoc, source: string): Document[] {\n const baseMetadata = buildMetadata(document, source);\n const documents: Document[] = [];\n\n for (const chunk of document.chunks ?? []) {\n const metadata: Record<string, unknown> = { ...baseMetadata };\n const chunkMetadata = chunk.metadata;\n metadata.chunk_index = chunkMetadata.chunkIndex;\n metadata.total_chunks = chunkMetadata.totalChunks;\n metadata.chunk_type = String(chunk.chunkType);\n if (chunkMetadata.headingPath && chunkMetadata.headingPath.length > 0) {\n metadata.heading_path = [...chunkMetadata.headingPath];\n }\n if (chunkMetadata.tokenCount != null) {\n metadata.token_count = chunkMetadata.tokenCount;\n }\n if (chunkMetadata.firstPage != null) {\n // Xberg uses 1-indexed pages; LangChain convention is 0-indexed. ~keep\n metadata.page = chunkMetadata.firstPage - 1;\n metadata.first_page = chunkMetadata.firstPage;\n }\n if (chunkMetadata.lastPage != null) {\n metadata.last_page = chunkMetadata.lastPage;\n }\n documents.push(new Document({ pageContent: chunk.content, metadata }));\n }\n\n return documents;\n}\n\nfunction pagesToDocuments(document: ExtractedDoc, source: string): Document[] {\n const baseMetadata = buildMetadata(document, source);\n const documents: Document[] = [];\n\n for (const page of document.pages ?? []) {\n const metadata: Record<string, unknown> = { ...baseMetadata };\n // Xberg uses 1-indexed pages; LangChain convention is 0-indexed. ~keep\n metadata.page = page.pageNumber - 1;\n if (page.isBlank != null) {\n metadata.is_blank = page.isBlank;\n }\n const pageContent = assembleContent(page.content, page.tables);\n documents.push(new Document({ pageContent, metadata }));\n }\n\n return documents;\n}\n\nexport function documentToDocuments(document: ExtractedDoc, source: string, options: SplitOptions): Document[] {\n if (options.chunking && document.chunks && document.chunks.length > 0) {\n return chunksToDocuments(document, source);\n }\n if (options.perPage && document.pages && document.pages.length > 0) {\n return pagesToDocuments(document, source);\n }\n const metadata = buildMetadata(document, source);\n const pageContent = assembleContent(document.content, document.tables);\n return [new Document({ pageContent, metadata })];\n}\n\nexport function resultToDocuments(result: ResultEnvelope, sources: string[], options: SplitOptions): Document[] {\n if (result.errors && result.errors.length > 0) {\n const error = result.errors[0];\n throw new Error(`Failed to extract '${error.source}': ${error.message}`);\n }\n\n const documents: Document[] = [];\n const results = result.results ?? [];\n results.forEach((document, index) => {\n const source = sources[index] ?? (sources.length > 0 ? sources[sources.length - 1] : \"\");\n documents.push(...documentToDocuments(document, source, options));\n });\n return documents;\n}\n"],"mappings":";AAAA,SAAS,YAAY;AACrB,SAAS,0BAA0B;AAEnC,SAAS,SAAS,oBAAoB;AAEtC,OAAO,cAAc;;;ACLrB,SAAS,gBAAgB;AAOzB,IAAM,kBAAoE;AAAA,EACxE,CAAC,SAAS,OAAO;AAAA,EACjB,CAAC,WAAW,SAAS;AAAA,EACrB,CAAC,WAAW,SAAS;AAAA,EACrB,CAAC,YAAY,UAAU;AAAA,EACvB,CAAC,YAAY,UAAU;AAAA,EACvB,CAAC,cAAc,WAAW;AAAA,EAC1B,CAAC,eAAe,YAAY;AAAA,EAC5B,CAAC,cAAc,WAAW;AAAA,EAC1B,CAAC,eAAe,YAAY;AAAA,EAC5B,CAAC,YAAY,UAAU;AAAA,EACvB,CAAC,QAAQ,MAAM;AAAA,EACf,CAAC,oBAAoB,iBAAiB;AAAA,EACtC,CAAC,iBAAiB,cAAc;AAAA,EAChC,CAAC,iBAAiB,cAAc;AAAA,EAChC,CAAC,YAAY,SAAS;AAAA,EACtB,CAAC,0BAA0B,sBAAsB;AACnD;AAEA,IAAM,oBAAoB;AAmDnB,SAAS,kBAAkB,QAA2C;AAC3E,SAAO,QAAQ,YAAY;AAC7B;AAEO,SAAS,iBAAiB,QAA2C;AAC1E,SAAO,QAAQ,QAAQ,OAAO,YAAY;AAC5C;AAEO,SAAS,gBAAgB,UAA8C;AAC5E,QAAM,OAAgC,CAAC;AACvC,MAAI,CAAC,UAAU;AACb,WAAO;AAAA,EACT;AAEA,aAAW,CAAC,UAAU,QAAQ,KAAK,iBAAiB;AAClD,UAAM,QAAQ,SAAS,QAAQ;AAC/B,QAAI,SAAS,MAAM;AACjB,WAAK,QAAQ,IAAI;AAAA,IACnB;AAAA,EACF;AAEA,QAAM,aAAa,SAAS;AAC5B,MAAI,cAAc,OAAO,KAAK,UAAU,EAAE,SAAS,GAAG;AACpD,SAAK,aAAa,EAAE,GAAG,WAAW;AAAA,EACpC;AAEA,SAAO;AACT;AAEO,SAAS,gBAAgB,SAA6B,QAAqC;AAChG,QAAM,OAAO,WAAW;AACxB,MAAI,CAAC,UAAU,OAAO,WAAW,GAAG;AAClC,WAAO;AAAA,EACT;AACA,QAAM,QAAQ,OAAO,IAAI,CAAC,UAAU,MAAM,YAAY,EAAE,EAAE,OAAO,CAAC,aAAa,SAAS,SAAS,CAAC;AAClG,MAAI,MAAM,WAAW,GAAG;AACtB,WAAO;AAAA,EACT;AACA,SAAO,CAAC,MAAM,GAAG,KAAK,EAAE,KAAK,iBAAiB;AAChD;AAEO,SAAS,cAAc,UAAwB,QAAyC;AAC7F,QAAM,WAAW,gBAAgB,SAAS,QAAQ;AAElD,WAAS,YAAY,SAAS;AAC9B,MAAI,SAAS,gBAAgB,MAAM;AACjC,aAAS,gBAAgB,SAAS;AAAA,EACpC;AACA,MAAI,SAAS,qBAAqB,SAAS,kBAAkB,SAAS,GAAG;AACvE,aAAS,qBAAqB,SAAS;AAAA,EACzC;AACA,MAAI,SAAS,UAAU,MAAM;AAC3B,aAAS,aAAa,SAAS,OAAO;AAAA,EACxC;AAEA,MAAI,SAAS,qBAAqB,SAAS,kBAAkB,SAAS,GAAG;AACvE,aAAS,qBAAqB,SAAS,kBAAkB,IAAI,CAAC,aAAa;AAAA,MACzE,MAAM,QAAQ;AAAA,MACd,OAAO,QAAQ;AAAA,MACf,WAAW,OAAO,QAAQ,SAAS;AAAA,IACrC,EAAE;AAAA,EACJ;AAEA,QAAM,SAAS,SAAS,UAAU,CAAC;AACnC,WAAS,cAAc,OAAO;AAC9B,MAAI,OAAO,SAAS,GAAG;AACrB,aAAS,SAAS,OAAO,IAAI,CAAC,WAAW;AAAA,MACvC,OAAO,MAAM;AAAA,MACb,UAAU,MAAM;AAAA,MAChB,aAAa,MAAM;AAAA,IACrB,EAAE;AAAA,EACJ;AAEA,MAAI,SAAS,sBAAsB,SAAS,mBAAmB,SAAS,GAAG;AACzE,aAAS,sBAAsB,SAAS,mBAAmB,IAAI,CAAC,aAAa;AAAA,MAC3E,QAAQ,QAAQ;AAAA,MAChB,SAAS,QAAQ;AAAA,IACnB,EAAE;AAAA,EACJ;AAEA,WAAS,SAAS;AAClB,SAAO;AACT;AAEA,SAAS,kBAAkB,UAAwB,QAA4B;AAC7E,QAAM,eAAe,cAAc,UAAU,MAAM;AACnD,QAAM,YAAwB,CAAC;AAE/B,aAAW,SAAS,SAAS,UAAU,CAAC,GAAG;AACzC,UAAM,WAAoC,EAAE,GAAG,aAAa;AAC5D,UAAM,gBAAgB,MAAM;AAC5B,aAAS,cAAc,cAAc;AACrC,aAAS,eAAe,cAAc;AACtC,aAAS,aAAa,OAAO,MAAM,SAAS;AAC5C,QAAI,cAAc,eAAe,cAAc,YAAY,SAAS,GAAG;AACrE,eAAS,eAAe,CAAC,GAAG,cAAc,WAAW;AAAA,IACvD;AACA,QAAI,cAAc,cAAc,MAAM;AACpC,eAAS,cAAc,cAAc;AAAA,IACvC;AACA,QAAI,cAAc,aAAa,MAAM;AAEnC,eAAS,OAAO,cAAc,YAAY;AAC1C,eAAS,aAAa,cAAc;AAAA,IACtC;AACA,QAAI,cAAc,YAAY,MAAM;AAClC,eAAS,YAAY,cAAc;AAAA,IACrC;AACA,cAAU,KAAK,IAAI,SAAS,EAAE,aAAa,MAAM,SAAS,SAAS,CAAC,CAAC;AAAA,EACvE;AAEA,SAAO;AACT;AAEA,SAAS,iBAAiB,UAAwB,QAA4B;AAC5E,QAAM,eAAe,cAAc,UAAU,MAAM;AACnD,QAAM,YAAwB,CAAC;AAE/B,aAAW,QAAQ,SAAS,SAAS,CAAC,GAAG;AACvC,UAAM,WAAoC,EAAE,GAAG,aAAa;AAE5D,aAAS,OAAO,KAAK,aAAa;AAClC,QAAI,KAAK,WAAW,MAAM;AACxB,eAAS,WAAW,KAAK;AAAA,IAC3B;AACA,UAAM,cAAc,gBAAgB,KAAK,SAAS,KAAK,MAAM;AAC7D,cAAU,KAAK,IAAI,SAAS,EAAE,aAAa,SAAS,CAAC,CAAC;AAAA,EACxD;AAEA,SAAO;AACT;AAEO,SAAS,oBAAoB,UAAwB,QAAgB,SAAmC;AAC7G,MAAI,QAAQ,YAAY,SAAS,UAAU,SAAS,OAAO,SAAS,GAAG;AACrE,WAAO,kBAAkB,UAAU,MAAM;AAAA,EAC3C;AACA,MAAI,QAAQ,WAAW,SAAS,SAAS,SAAS,MAAM,SAAS,GAAG;AAClE,WAAO,iBAAiB,UAAU,MAAM;AAAA,EAC1C;AACA,QAAM,WAAW,cAAc,UAAU,MAAM;AAC/C,QAAM,cAAc,gBAAgB,SAAS,SAAS,SAAS,MAAM;AACrE,SAAO,CAAC,IAAI,SAAS,EAAE,aAAa,SAAS,CAAC,CAAC;AACjD;AAEO,SAAS,kBAAkB,QAAwB,SAAmB,SAAmC;AAC9G,MAAI,OAAO,UAAU,OAAO,OAAO,SAAS,GAAG;AAC7C,UAAM,QAAQ,OAAO,OAAO,CAAC;AAC7B,UAAM,IAAI,MAAM,sBAAsB,MAAM,MAAM,MAAM,MAAM,OAAO,EAAE;AAAA,EACzE;AAEA,QAAM,YAAwB,CAAC;AAC/B,QAAM,UAAU,OAAO,WAAW,CAAC;AACnC,UAAQ,QAAQ,CAAC,UAAU,UAAU;AACnC,UAAM,SAAS,QAAQ,KAAK,MAAM,QAAQ,SAAS,IAAI,QAAQ,QAAQ,SAAS,CAAC,IAAI;AACrF,cAAU,KAAK,GAAG,oBAAoB,UAAU,QAAQ,OAAO,CAAC;AAAA,EAClE,CAAC;AACD,SAAO;AACT;;;ADlOA,IAAM,eAAe;AAuBd,IAAM,cAAN,cAA0B,mBAAmB;AAAA,EACjC;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EAEjB,YAAY,SAA6B;AACvC,UAAM;AACN,UAAM,EAAE,UAAU,MAAM,UAAU,MAAM,OAAO,IAAI;AACnD,QAAI,aAAa,UAAa,SAAS,QAAW;AAChD,YAAM,IAAI,MAAM,+CAA+C;AAAA,IACjE;AACA,QAAI,aAAa,UAAa,SAAS,QAAW;AAChD,YAAM,IAAI,MAAM,kEAAkE;AAAA,IACpF;AACA,QAAI,SAAS,UAAa,aAAa,QAAW;AAChD,YAAM,IAAI,MAAM,2CAA2C;AAAA,IAC7D;AAEA,SAAK,WAAW;AAChB,SAAK,OAAO;AACZ,SAAK,WAAW;AAChB,SAAK,OAAO;AACZ,SAAK,SAAS;AAAA,EAChB;AAAA,EAEA,MAAM,OAA4B;AAChC,UAAM,EAAE,QAAQ,SAAS,MAAM,IAAI,MAAM,KAAK,YAAY;AAC1D,QAAI,OAAO,WAAW,GAAG;AACvB,aAAO,CAAC;AAAA,IACV;AAEA,QAAI;AACJ,QAAI;AACF,eAAS,QAAQ,MAAM,aAAa,QAAQ,KAAK,UAAU,IAAI,IAAI,MAAM,QAAQ,OAAO,CAAC,GAAG,KAAK,UAAU,IAAI;AAAA,IACjH,SAAS,OAAO;AACd,YAAM,SAAS,QAAQ,CAAC,KAAK;AAC7B,YAAM,IAAI,MAAM,sBAAsB,MAAM,MAAM,aAAa,KAAK,CAAC,EAAE;AAAA,IACzE;AAEA,WAAO,kBAAkB,QAAqC,SAAS;AAAA,MACrE,UAAU,kBAAkB,KAAK,MAAM;AAAA,MACvC,SAAS,iBAAiB,KAAK,MAAM;AAAA,IACvC,CAAC;AAAA,EACH;AAAA,EAEA,MAAc,cAAsF;AAClG,QAAI,KAAK,SAAS,QAAW;AAC3B,YAAM,SAAS,WAAW,KAAK,QAAQ;AACvC,YAAM,QAAsB,EAAE,MAAM,SAAS,OAAO,KAAK,MAAM,UAAU,KAAK,SAAS;AACvF,aAAO,EAAE,QAAQ,CAAC,KAAK,GAAG,SAAS,CAAC,MAAM,GAAG,OAAO,MAAM;AAAA,IAC5D;AAEA,UAAM,EAAE,OAAO,MAAM,IAAI,MAAM,KAAK,aAAa;AACjD,UAAM,SAAyB,MAAM,IAAI,CAAC,UAAU,EAAE,MAAM,OAAO,KAAK,MAAM,UAAU,KAAK,SAAS,EAAE;AACxG,WAAO,EAAE,QAAQ,SAAS,OAAO,MAAM;AAAA,EACzC;AAAA,EAEA,MAAc,eAA6D;AACzE,UAAM,WAAW,KAAK;AACtB,QAAI,MAAM,QAAQ,QAAQ,GAAG;AAC3B,aAAO,EAAE,OAAO,UAAU,OAAO,KAAK;AAAA,IACxC;AACA,QAAI,OAAO,aAAa,UAAU;AAChC,UAAI,MAAM,YAAY,QAAQ,GAAG;AAC/B,cAAM,UAAU,KAAK,QAAQ;AAC7B,cAAM,UAAU,MAAM,SAAS,SAAS,EAAE,KAAK,UAAU,WAAW,MAAM,UAAU,KAAK,CAAC;AAC1F,eAAO,EAAE,OAAO,QAAQ,KAAK,GAAG,OAAO,KAAK;AAAA,MAC9C;AACA,aAAO,EAAE,OAAO,CAAC,QAAQ,GAAG,OAAO,MAAM;AAAA,IAC3C;AACA,WAAO,EAAE,OAAO,CAAC,GAAG,OAAO,MAAM;AAAA,EACnC;AACF;AAEA,eAAe,YAAY,MAAgC;AACzD,MAAI;AACF,UAAM,QAAQ,MAAM,KAAK,IAAI;AAC7B,WAAO,MAAM,YAAY;AAAA,EAC3B,QAAQ;AACN,WAAO;AAAA,EACT;AACF;AAEA,SAAS,aAAa,OAAwB;AAC5C,SAAO,iBAAiB,QAAQ,MAAM,UAAU,OAAO,KAAK;AAC9D;","names":[]}
|
package/package.json
CHANGED
|
@@ -1,8 +1,64 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@xberg-io/langchain-xberg",
|
|
3
|
-
"version": "0.0.
|
|
4
|
-
"description": "LangChain.js document loader for Xberg document
|
|
3
|
+
"version": "1.0.0-rc.39",
|
|
4
|
+
"description": "LangChain.js document loader for Xberg — extract text, tables, and metadata from 90+ document formats with optional OCR.",
|
|
5
|
+
"keywords": [
|
|
6
|
+
"langchain",
|
|
7
|
+
"langchainjs",
|
|
8
|
+
"document-loader",
|
|
9
|
+
"xberg",
|
|
10
|
+
"document-extraction",
|
|
11
|
+
"ocr",
|
|
12
|
+
"pdf",
|
|
13
|
+
"text-extraction",
|
|
14
|
+
"rag"
|
|
15
|
+
],
|
|
5
16
|
"license": "MIT",
|
|
6
|
-
"homepage": "https://github.com/xberg-io/xberg",
|
|
7
|
-
"
|
|
17
|
+
"homepage": "https://github.com/xberg-io/xberg/tree/main/integrations/node/langchain-xberg",
|
|
18
|
+
"author": {
|
|
19
|
+
"name": "Xberg.dev",
|
|
20
|
+
"email": "contact@xberg.io"
|
|
21
|
+
},
|
|
22
|
+
"repository": {
|
|
23
|
+
"type": "git",
|
|
24
|
+
"url": "git+https://github.com/xberg-io/xberg.git",
|
|
25
|
+
"directory": "integrations/node/langchain-xberg"
|
|
26
|
+
},
|
|
27
|
+
"engines": {
|
|
28
|
+
"node": ">=20.15"
|
|
29
|
+
},
|
|
30
|
+
"type": "module",
|
|
31
|
+
"main": "./dist/index.cjs",
|
|
32
|
+
"module": "./dist/index.js",
|
|
33
|
+
"types": "./dist/index.d.ts",
|
|
34
|
+
"exports": {
|
|
35
|
+
".": {
|
|
36
|
+
"types": "./dist/index.d.ts",
|
|
37
|
+
"import": "./dist/index.js",
|
|
38
|
+
"require": "./dist/index.cjs"
|
|
39
|
+
}
|
|
40
|
+
},
|
|
41
|
+
"files": ["dist"],
|
|
42
|
+
"scripts": {
|
|
43
|
+
"build": "tsup",
|
|
44
|
+
"typecheck": "tsc --noEmit",
|
|
45
|
+
"test": "vitest run"
|
|
46
|
+
},
|
|
47
|
+
"dependencies": {
|
|
48
|
+
"@xberg-io/xberg": "1.0.0-rc.39",
|
|
49
|
+
"fast-glob": "^3.3.2"
|
|
50
|
+
},
|
|
51
|
+
"peerDependencies": {
|
|
52
|
+
"@langchain/core": ">=0.3.0"
|
|
53
|
+
},
|
|
54
|
+
"devDependencies": {
|
|
55
|
+
"@langchain/core": "^0.3.0",
|
|
56
|
+
"@types/node": "^20.19.0",
|
|
57
|
+
"tsup": "^8.3.5",
|
|
58
|
+
"typescript": "^5.7.3",
|
|
59
|
+
"vitest": "^2.1.8"
|
|
60
|
+
},
|
|
61
|
+
"publishConfig": {
|
|
62
|
+
"access": "public"
|
|
63
|
+
}
|
|
8
64
|
}
|