@lumifai/harness-tool-pack-file-processing 0.1.0 → 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +8 -2
- package/dist/file-processing.js +113 -10
- package/dist/file-processing.js.map +1 -1
- package/dist/utils.d.ts +4 -1
- package/dist/utils.js +17 -25
- package/dist/utils.js.map +1 -1
- package/package.json +10 -5
- package/skills/file-processing/SKILL.md +10 -10
package/README.md
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
# @lumifai/harness-tool-pack-file-processing
|
|
2
2
|
|
|
3
|
-
Extract
|
|
3
|
+
Extract workspace PDF files into Markdown or text files, with optional document metadata.
|
|
4
4
|
|
|
5
5
|
```ts
|
|
6
6
|
import { createFileProcessingToolPack } from '@lumifai/harness-tool-pack-file-processing';
|
|
@@ -8,6 +8,12 @@ import { createFileProcessingToolPack } from '@lumifai/harness-tool-pack-file-pr
|
|
|
8
8
|
createFileProcessingToolPack();
|
|
9
9
|
```
|
|
10
10
|
|
|
11
|
-
`process_pdf` reads a workspace PDF, runs OCR-aware extraction via LiteParse, and returns an `output_path` reference instead of inline content. Bundled agent skills ship in this package and are merged into `workspace.skills` when the pack is enabled.
|
|
11
|
+
`process_pdf` reads a workspace PDF, runs OCR-aware extraction via LiteParse, and returns an `output_path` reference instead of inline content. Markdown is the default format and preserves more document structure, including tables and links. Bundled agent skills ship in this package and are merged into `workspace.skills` when the pack is enabled.
|
|
12
|
+
|
|
13
|
+
The tool applies consistent OCR-aware parsing and returns metadata, annotations, form fields, and page complexity by default. The small set of meaningful optional inputs is:
|
|
14
|
+
|
|
15
|
+
- `format`: `markdown` (default) or `text`.
|
|
16
|
+
- `include_images`: extracted image metadata and workspace paths to image files.
|
|
17
|
+
- `include_complexity`: per-page OCR and layout complexity signals (default `true`).
|
|
12
18
|
|
|
13
19
|
Tool pack id: `file-processing`. Disabled by default (`enabledByDefault: false`). Default policy: `process_pdf` → `ask`.
|
package/dist/file-processing.js
CHANGED
|
@@ -42,30 +42,133 @@ const utils_1 = require("./utils");
|
|
|
42
42
|
const workspaceFilePathSchema = zod_1.z
|
|
43
43
|
.string()
|
|
44
44
|
.min(1)
|
|
45
|
-
.describe('
|
|
45
|
+
.describe('Workspace-relative path to a file, e.g. documents/report.pdf');
|
|
46
|
+
const processPdfInputSchema = zod_1.z.object({
|
|
47
|
+
path: workspaceFilePathSchema,
|
|
48
|
+
format: zod_1.z
|
|
49
|
+
.enum(['markdown', 'text'])
|
|
50
|
+
.default('markdown')
|
|
51
|
+
.describe('Output format. Markdown preserves document structure and improves table extraction.'),
|
|
52
|
+
include_images: zod_1.z
|
|
53
|
+
.boolean()
|
|
54
|
+
.default(false)
|
|
55
|
+
.describe('Extract embedded images and write them beside the processed output.'),
|
|
56
|
+
include_complexity: zod_1.z
|
|
57
|
+
.boolean()
|
|
58
|
+
.default(true)
|
|
59
|
+
.describe('Include per-page layout and OCR complexity signals.'),
|
|
60
|
+
});
|
|
61
|
+
function safeImageFileName(name, id, index) {
|
|
62
|
+
const extension = name.split('.').pop() || 'bin';
|
|
63
|
+
const safeId = id.replace(/[^a-zA-Z0-9_-]/g, '_') || `image-${index + 1}`;
|
|
64
|
+
return `${safeId}.${extension.replace(/[^a-zA-Z0-9]/g, '') || 'bin'}`;
|
|
65
|
+
}
|
|
66
|
+
function serializeComplexity(pageNum, complexity) {
|
|
67
|
+
if (!complexity || typeof complexity !== 'object') {
|
|
68
|
+
return { page_number: pageNum };
|
|
69
|
+
}
|
|
70
|
+
const value = complexity;
|
|
71
|
+
return {
|
|
72
|
+
page_number: pageNum,
|
|
73
|
+
text_length: value.textLength,
|
|
74
|
+
text_coverage: value.textCoverage,
|
|
75
|
+
has_substantial_images: value.hasSubstantialImages,
|
|
76
|
+
image_block_count: value.imageBlockCount,
|
|
77
|
+
image_coverage: value.imageCoverage,
|
|
78
|
+
largest_image_coverage: value.largestImageCoverage,
|
|
79
|
+
full_page_image: value.fullPageImage,
|
|
80
|
+
uncovered_vector_area: value.uncoveredVectorArea,
|
|
81
|
+
is_garbled: value.isGarbled,
|
|
82
|
+
page_area: value.pageArea,
|
|
83
|
+
needs_ocr: value.needsOcr,
|
|
84
|
+
reasons: value.reasons,
|
|
85
|
+
layout: value.layout,
|
|
86
|
+
};
|
|
87
|
+
}
|
|
46
88
|
function createFileProcessingTools() {
|
|
47
89
|
const processPdf = (0, tools_1.createTool)({
|
|
48
90
|
id: 'process_pdf',
|
|
49
|
-
description: 'Extract
|
|
50
|
-
inputSchema:
|
|
51
|
-
path: workspaceFilePathSchema,
|
|
52
|
-
}),
|
|
91
|
+
description: 'Extract a PDF into a workspace Markdown or text file. Markdown preserves structure and tables. The result includes document metadata, annotations, form fields, and page complexity signals; image extraction is optional because it can be expensive.',
|
|
92
|
+
inputSchema: processPdfInputSchema,
|
|
53
93
|
execute: async (inputData, context) => {
|
|
54
94
|
const { LiteParse } = await Promise.resolve().then(() => __importStar(require('@llamaindex/liteparse')));
|
|
55
95
|
const { normalizedPath, buffer } = await (0, utils_1.readWorkspaceBinaryFile)(context, inputData.path);
|
|
56
|
-
const parser = new LiteParse({
|
|
96
|
+
const parser = new LiteParse({
|
|
97
|
+
ocrEnabled: true,
|
|
98
|
+
quiet: true,
|
|
99
|
+
outputFormat: inputData.format,
|
|
100
|
+
extractLinks: inputData.format === 'markdown',
|
|
101
|
+
imageMode: inputData.format === 'markdown' ? 'placeholder' : 'off',
|
|
102
|
+
extractImages: inputData.include_images,
|
|
103
|
+
extractAnnotations: true,
|
|
104
|
+
extractFormFields: true,
|
|
105
|
+
extractDocumentMetadata: true,
|
|
106
|
+
includeComplexity: inputData.include_complexity,
|
|
107
|
+
});
|
|
57
108
|
const result = await parser.parse(buffer);
|
|
58
|
-
const
|
|
59
|
-
|
|
109
|
+
const outputPath = (0, utils_1.deriveOutputPath)(normalizedPath, inputData.format === 'markdown' ? 'md' : 'txt');
|
|
110
|
+
const imageDirectory = `${outputPath}.images`;
|
|
111
|
+
const imageOutputs = [];
|
|
112
|
+
let outputContent = result.pages
|
|
113
|
+
.map((page) => {
|
|
114
|
+
const pageContent = inputData.format === 'markdown' ? page.markdown : page.text;
|
|
115
|
+
return `--- Page ${page.pageNum} ---\n${pageContent}`;
|
|
116
|
+
})
|
|
60
117
|
.join('\n\n');
|
|
61
|
-
|
|
118
|
+
if (inputData.include_images) {
|
|
119
|
+
const relativeImageDirectory = imageDirectory.substring(imageDirectory.lastIndexOf('/') + 1);
|
|
120
|
+
for (const [index, image] of result.images.entries()) {
|
|
121
|
+
const imageName = safeImageFileName(image.name, image.id, index);
|
|
122
|
+
const imagePath = `${imageDirectory}/${imageName}`;
|
|
123
|
+
await (0, utils_1.writeProcessedBinary)(context, imagePath, image.bytes);
|
|
124
|
+
imageOutputs.push({
|
|
125
|
+
id: image.id,
|
|
126
|
+
name: image.name,
|
|
127
|
+
output_path: imagePath,
|
|
128
|
+
page: image.page,
|
|
129
|
+
bbox: image.bbox,
|
|
130
|
+
width: image.width,
|
|
131
|
+
height: image.height,
|
|
132
|
+
rotation: image.rotation,
|
|
133
|
+
format: image.format,
|
|
134
|
+
duplicate_of: image.duplicateOf,
|
|
135
|
+
});
|
|
136
|
+
if (inputData.format === 'markdown') {
|
|
137
|
+
const relativeImagePath = `./${relativeImageDirectory}/${imageName}`;
|
|
138
|
+
if (image.name)
|
|
139
|
+
outputContent = outputContent.replaceAll(image.name, relativeImagePath);
|
|
140
|
+
if (image.id && image.id !== image.name)
|
|
141
|
+
outputContent = outputContent.replaceAll(image.id, relativeImagePath);
|
|
142
|
+
}
|
|
143
|
+
}
|
|
144
|
+
}
|
|
62
145
|
await (0, utils_1.writeProcessedText)(context, outputPath, outputContent);
|
|
63
|
-
|
|
146
|
+
const toolResult = {
|
|
64
147
|
source_path: normalizedPath,
|
|
65
148
|
output_path: outputPath,
|
|
66
149
|
page_count: result.pages.length,
|
|
67
150
|
char_count: outputContent.length,
|
|
68
151
|
};
|
|
152
|
+
toolResult.document_metadata = result.docMeta ?? null;
|
|
153
|
+
toolResult.creator = result.creator ?? null;
|
|
154
|
+
toolResult.producer = result.producer ?? null;
|
|
155
|
+
toolResult.annotations = result.pages.flatMap((page) => (page.annotations ?? []).map((annotation) => ({
|
|
156
|
+
page_number: page.pageNum,
|
|
157
|
+
...annotation,
|
|
158
|
+
})));
|
|
159
|
+
toolResult.form_fields = result.pages.flatMap((page) => (page.formFields ?? []).map((field) => ({
|
|
160
|
+
page_number: page.pageNum,
|
|
161
|
+
...field,
|
|
162
|
+
})));
|
|
163
|
+
toolResult.form_type = result.formType ?? null;
|
|
164
|
+
if (inputData.include_complexity) {
|
|
165
|
+
toolResult.complexity = result.pages.map((page) => serializeComplexity(page.pageNum, page.complexity));
|
|
166
|
+
}
|
|
167
|
+
if (inputData.include_images) {
|
|
168
|
+
toolResult.images = imageOutputs;
|
|
169
|
+
toolResult.image_error_count = result.imageErrorCount;
|
|
170
|
+
}
|
|
171
|
+
return toolResult;
|
|
69
172
|
},
|
|
70
173
|
});
|
|
71
174
|
return {
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"file-processing.js","sourceRoot":"","sources":["../src/file-processing.ts"],"names":[],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;
|
|
1
|
+
{"version":3,"file":"file-processing.js","sourceRoot":"","sources":["../src/file-processing.ts"],"names":[],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AAmEA,8DAiHC;AAED,oEAUC;AA9LD,8CAAgD;AAChD,6BAAwB;AAExB,oEAA6D;AAE7D,mCAKiB;AAEjB,MAAM,uBAAuB,GAAG,OAAC;KAC9B,MAAM,EAAE;KACR,GAAG,CAAC,CAAC,CAAC;KACN,QAAQ,CAAC,8DAA8D,CAAC,CAAC;AAE5E,MAAM,qBAAqB,GAAG,OAAC,CAAC,MAAM,CAAC;IACrC,IAAI,EAAE,uBAAuB;IAC7B,MAAM,EAAE,OAAC;SACN,IAAI,CAAC,CAAC,UAAU,EAAE,MAAM,CAAC,CAAC;SAC1B,OAAO,CAAC,UAAU,CAAC;SACnB,QAAQ,CACP,qFAAqF,CACtF;IACH,cAAc,EAAE,OAAC;SACd,OAAO,EAAE;SACT,OAAO,CAAC,KAAK,CAAC;SACd,QAAQ,CAAC,qEAAqE,CAAC;IAClF,kBAAkB,EAAE,OAAC;SAClB,OAAO,EAAE;SACT,OAAO,CAAC,IAAI,CAAC;SACb,QAAQ,CAAC,qDAAqD,CAAC;CACnE,CAAC,CAAC;AAEH,SAAS,iBAAiB,CAAC,IAAY,EAAE,EAAU,EAAE,KAAa;IAChE,MAAM,SAAS,GAAG,IAAI,CAAC,KAAK,CAAC,GAAG,CAAC,CAAC,GAAG,EAAE,IAAI,KAAK,CAAC;IACjD,MAAM,MAAM,GAAG,EAAE,CAAC,OAAO,CAAC,iBAAiB,EAAE,GAAG,CAAC,IAAI,SAAS,KAAK,GAAG,CAAC,EAAE,CAAC;IAC1E,OAAO,GAAG,MAAM,IAAI,SAAS,CAAC,OAAO,CAAC,eAAe,EAAE,EAAE,CAAC,IAAI,KAAK,EAAE,CAAC;AACxE,CAAC;AAED,SAAS,mBAAmB,CAAC,OAAe,EAAE,UAA2C;IACvF,IAAI,CAAC,UAAU,IAAI,OAAO,UAAU,KAAK,QAAQ,EAAE,CAAC;QAClD,OAAO,EAAE,WAAW,EAAE,OAAO,EAAE,CAAC;IAClC,CAAC;IAED,MAAM,KAAK,GAAG,UAAgD,CAAC;IAC/D,OAAO;QACL,WAAW,EAAE,OAAO;QACpB,WAAW,EAAE,KAAK,CAAC,UAAU;QAC7B,aAAa,EAAE,KAAK,CAAC,YAAY;QACjC,sBAAsB,EAAE,KAAK,CAAC,oBAAoB;QAClD,iBAAiB,EAAE,KAAK,CAAC,eAAe;QACxC,cAAc,EAAE,KAAK,CAAC,aAAa;QACnC,sBAAsB,EAAE,KAAK,CAAC,oBAAoB;QAClD,eAAe,EAAE,KAAK,CAAC,aAAa;QACpC,qBAAqB,EAAE,KAAK,CAAC,mBAAmB;QAChD,UAAU,EAAE,KAAK,CAAC,SAAS;QAC3B,SAAS,EAAE,KAAK,CAAC,QAAQ;QACzB,SAAS,EAAE,KAAK,CAAC,QAAQ;QACzB,OAAO,EAAE,KAAK,CAAC,OAAO;QACtB,MAAM,EAAE,KAAK,CAAC,MAAM;KACrB,CAAC;AACJ,CAAC;AAED,SAAgB,yBAAyB;IACvC,MAAM,UAAU,GAAG,IAAA,kBAAU,EAAC;QAC5B,EAAE,EAAE,aAAa;QACjB,WAAW,EACT,wPAAwP;QAC1P,WAAW,EAAE,qBAAqB;QAClC,OAAO,EAAE,KAAK,EAAE,SAAS,EAAE,OAAO,EAAE,EAAE;YACpC,MAAM,EAAE,SAAS,EAAE,GAAG,wDAAa,uBAAuB,GAAC,CAAC;YAC5D,MAAM,EAAE,cAAc,EAAE,MAAM,EAAE,GAAG,MAAM,IAAA,+BAAuB,EAAC,OAAO,EAAE,SAAS,CAAC,IAAI,CAAC,CAAC;YAE1F,MAAM,MAAM,GAAG,IAAI,SAAS,CAAC;gBAC3B,UAAU,EAAE,IAAI;gBAChB,KAAK,EAAE,IAAI;gBACX,YAAY,EAAE,SAAS,CAAC,MAAM;gBAC9B,YAAY,EAAE,SAAS,CAAC,MAAM,KAAK,UAAU;gBAC7C,SAAS,EAAE,SAAS,CAAC,MAAM,KAAK,UAAU,CAAC,CAAC,CAAC,aAAa,CAAC,CAAC,CAAC,KAAK;gBAClE,aAAa,EAAE,SAAS,CAAC,cAAc;gBACvC,kBAAkB,EAAE,IAAI;gBACxB,iBAAiB,EAAE,IAAI;gBACvB,uBAAuB,EAAE,IAAI;gBAC7B,iBAAiB,EAAE,SAAS,CAAC,kBAAkB;aAChD,CAAC,CAAC;YACH,MAAM,MAAM,GAAG,MAAM,MAAM,CAAC,KAAK,CAAC,MAAM,CAAC,CAAC;YAE1C,MAAM,UAAU,GAAG,IAAA,wBAAgB,EACjC,cAAc,EACd,SAAS,CAAC,MAAM,KAAK,UAAU,CAAC,CAAC,CAAC,IAAI,CAAC,CAAC,CAAC,KAAK,CAC/C,CAAC;YACF,MAAM,cAAc,GAAG,GAAG,UAAU,SAAS,CAAC;YAC9C,MAAM,YAAY,GAAG,EAAE,CAAC;YAExB,IAAI,aAAa,GAAG,MAAM,CAAC,KAAK;iBAC7B,GAAG,CAAC,CAAC,IAAI,EAAE,EAAE;gBACZ,MAAM,WAAW,GAAG,SAAS,CAAC,MAAM,KAAK,UAAU,CAAC,CAAC,CAAC,IAAI,CAAC,QAAQ,CAAC,CAAC,CAAC,IAAI,CAAC,IAAI,CAAC;gBAChF,OAAO,YAAY,IAAI,CAAC,OAAO,SAAS,WAAW,EAAE,CAAC;YACxD,CAAC,CAAC;iBACD,IAAI,CAAC,MAAM,CAAC,CAAC;YAEhB,IAAI,SAAS,CAAC,cAAc,EAAE,CAAC;gBAC7B,MAAM,sBAAsB,GAAG,cAAc,CAAC,SAAS,CACrD,cAAc,CAAC,WAAW,CAAC,GAAG,CAAC,GAAG,CAAC,CACpC,CAAC;gBAEF,KAAK,MAAM,CAAC,KAAK,EAAE,KAAK,CAAC,IAAI,MAAM,CAAC,MAAM,CAAC,OAAO,EAAE,EAAE,CAAC;oBACrD,MAAM,SAAS,GAAG,iBAAiB,CAAC,KAAK,CAAC,IAAI,EAAE,KAAK,CAAC,EAAE,EAAE,KAAK,CAAC,CAAC;oBACjE,MAAM,SAAS,GAAG,GAAG,cAAc,IAAI,SAAS,EAAE,CAAC;oBACnD,MAAM,IAAA,4BAAoB,EAAC,OAAO,EAAE,SAAS,EAAE,KAAK,CAAC,KAAK,CAAC,CAAC;oBAC5D,YAAY,CAAC,IAAI,CAAC;wBAChB,EAAE,EAAE,KAAK,CAAC,EAAE;wBACZ,IAAI,EAAE,KAAK,CAAC,IAAI;wBAChB,WAAW,EAAE,SAAS;wBACtB,IAAI,EAAE,KAAK,CAAC,IAAI;wBAChB,IAAI,EAAE,KAAK,CAAC,IAAI;wBAChB,KAAK,EAAE,KAAK,CAAC,KAAK;wBAClB,MAAM,EAAE,KAAK,CAAC,MAAM;wBACpB,QAAQ,EAAE,KAAK,CAAC,QAAQ;wBACxB,MAAM,EAAE,KAAK,CAAC,MAAM;wBACpB,YAAY,EAAE,KAAK,CAAC,WAAW;qBAChC,CAAC,CAAC;oBAEH,IAAI,SAAS,CAAC,MAAM,KAAK,UAAU,EAAE,CAAC;wBACpC,MAAM,iBAAiB,GAAG,KAAK,sBAAsB,IAAI,SAAS,EAAE,CAAC;wBACrE,IAAI,KAAK,CAAC,IAAI;4BAAE,aAAa,GAAG,aAAa,CAAC,UAAU,CAAC,KAAK,CAAC,IAAI,EAAE,iBAAiB,CAAC,CAAC;wBACxF,IAAI,KAAK,CAAC,EAAE,IAAI,KAAK,CAAC,EAAE,KAAK,KAAK,CAAC,IAAI;4BACrC,aAAa,GAAG,aAAa,CAAC,UAAU,CAAC,KAAK,CAAC,EAAE,EAAE,iBAAiB,CAAC,CAAC;oBAC1E,CAAC;gBACH,CAAC;YACH,CAAC;YAED,MAAM,IAAA,0BAAkB,EAAC,OAAO,EAAE,UAAU,EAAE,aAAa,CAAC,CAAC;YAE7D,MAAM,UAAU,GAA4B;gBAC1C,WAAW,EAAE,cAAc;gBAC3B,WAAW,EAAE,UAAU;gBACvB,UAAU,EAAE,MAAM,CAAC,KAAK,CAAC,MAAM;gBAC/B,UAAU,EAAE,aAAa,CAAC,MAAM;aACjC,CAAC;YAEF,UAAU,CAAC,iBAAiB,GAAG,MAAM,CAAC,OAAO,IAAI,IAAI,CAAC;YACtD,UAAU,CAAC,OAAO,GAAG,MAAM,CAAC,OAAO,IAAI,IAAI,CAAC;YAC5C,UAAU,CAAC,QAAQ,GAAG,MAAM,CAAC,QAAQ,IAAI,IAAI,CAAC;YAC9C,UAAU,CAAC,WAAW,GAAG,MAAM,CAAC,KAAK,CAAC,OAAO,CAAC,CAAC,IAAI,EAAE,EAAE,CACrD,CAAC,IAAI,CAAC,WAAW,IAAI,EAAE,CAAC,CAAC,GAAG,CAAC,CAAC,UAAU,EAAE,EAAE,CAAC,CAAC;gBAC5C,WAAW,EAAE,IAAI,CAAC,OAAO;gBACzB,GAAG,UAAU;aACd,CAAC,CAAC,CACJ,CAAC;YACF,UAAU,CAAC,WAAW,GAAG,MAAM,CAAC,KAAK,CAAC,OAAO,CAAC,CAAC,IAAI,EAAE,EAAE,CACrD,CAAC,IAAI,CAAC,UAAU,IAAI,EAAE,CAAC,CAAC,GAAG,CAAC,CAAC,KAAK,EAAE,EAAE,CAAC,CAAC;gBACtC,WAAW,EAAE,IAAI,CAAC,OAAO;gBACzB,GAAG,KAAK;aACT,CAAC,CAAC,CACJ,CAAC;YACF,UAAU,CAAC,SAAS,GAAG,MAAM,CAAC,QAAQ,IAAI,IAAI,CAAC;YAE/C,IAAI,SAAS,CAAC,kBAAkB,EAAE,CAAC;gBACjC,UAAU,CAAC,UAAU,GAAG,MAAM,CAAC,KAAK,CAAC,GAAG,CAAC,CAAC,IAAI,EAAE,EAAE,CAChD,mBAAmB,CAAC,IAAI,CAAC,OAAO,EAAE,IAAI,CAAC,UAAU,CAAC,CACnD,CAAC;YACJ,CAAC;YAED,IAAI,SAAS,CAAC,cAAc,EAAE,CAAC;gBAC7B,UAAU,CAAC,MAAM,GAAG,YAAY,CAAC;gBACjC,UAAU,CAAC,iBAAiB,GAAG,MAAM,CAAC,eAAe,CAAC;YACxD,CAAC;YAED,OAAO,UAAU,CAAC;QACpB,CAAC;KACF,CAAC,CAAC;IAEH,OAAO;QACL,UAAU;KACX,CAAC;AACJ,CAAC;AAED,SAAgB,4BAA4B;IAC1C,OAAO,IAAA,mCAAc,EAAC;QACpB,EAAE,EAAE,iBAAiB;QACrB,WAAW,EAAE,8BAA8B;QAC3C,gBAAgB,EAAE,KAAK;QACvB,KAAK,EAAE,yBAAyB,EAAE;QAClC,kBAAkB,EAAE;YAClB,WAAW,EAAE,KAAK;SACnB;KACF,CAAC,CAAC;AACL,CAAC"}
|
package/dist/utils.d.ts
CHANGED
|
@@ -1,8 +1,11 @@
|
|
|
1
1
|
import type { ToolExecutionContext } from '@mastra/core/tools';
|
|
2
|
+
/** Hard cap for files loaded into memory for PDF OCR / LiteParse. */
|
|
3
|
+
export declare const DEFAULT_MAX_FILE_BYTES: number;
|
|
2
4
|
export declare function normalizeWorkspaceFilePath(path: string): string;
|
|
3
|
-
export declare function readWorkspaceBinaryFile(context: ToolExecutionContext, path: string): Promise<{
|
|
5
|
+
export declare function readWorkspaceBinaryFile(context: ToolExecutionContext, path: string, maxBytes?: number): Promise<{
|
|
4
6
|
normalizedPath: string;
|
|
5
7
|
buffer: Buffer;
|
|
6
8
|
}>;
|
|
7
9
|
export declare function writeProcessedText(context: ToolExecutionContext, outputPath: string, content: string): Promise<void>;
|
|
10
|
+
export declare function writeProcessedBinary(context: ToolExecutionContext, outputPath: string, content: Uint8Array): Promise<void>;
|
|
8
11
|
export declare function deriveOutputPath(sourcePath: string, suffix: string): string;
|
package/dist/utils.js
CHANGED
|
@@ -1,40 +1,32 @@
|
|
|
1
1
|
"use strict";
|
|
2
2
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
3
|
+
exports.DEFAULT_MAX_FILE_BYTES = void 0;
|
|
3
4
|
exports.normalizeWorkspaceFilePath = normalizeWorkspaceFilePath;
|
|
4
5
|
exports.readWorkspaceBinaryFile = readWorkspaceBinaryFile;
|
|
5
6
|
exports.writeProcessedText = writeProcessedText;
|
|
7
|
+
exports.writeProcessedBinary = writeProcessedBinary;
|
|
6
8
|
exports.deriveOutputPath = deriveOutputPath;
|
|
9
|
+
const harness_tool_packs_1 = require("@lumifai/harness-tool-packs");
|
|
10
|
+
/** Hard cap for files loaded into memory for PDF OCR / LiteParse. */
|
|
11
|
+
exports.DEFAULT_MAX_FILE_BYTES = 64 * 1024 * 1024;
|
|
12
|
+
const workspaceFiles = (0, harness_tool_packs_1.createWorkspaceFiles)({ maxFileBytes: exports.DEFAULT_MAX_FILE_BYTES });
|
|
7
13
|
function normalizeWorkspaceFilePath(path) {
|
|
8
|
-
|
|
9
|
-
if (!trimmed.startsWith('/')) {
|
|
10
|
-
throw new Error(`Path must be absolute within the workspace: ${path}`);
|
|
11
|
-
}
|
|
12
|
-
return trimmed;
|
|
14
|
+
return (0, harness_tool_packs_1.normalizeWorkspacePath)(path);
|
|
13
15
|
}
|
|
14
|
-
async function readWorkspaceBinaryFile(context, path) {
|
|
15
|
-
const
|
|
16
|
-
|
|
17
|
-
if (!workspace?.filesystem?.readFile) {
|
|
18
|
-
throw new Error(`File processing tools require a workspace filesystem to read ${normalizedPath}`);
|
|
19
|
-
}
|
|
20
|
-
const fileContent = await workspace.filesystem.readFile(normalizedPath);
|
|
21
|
-
const buffer = Buffer.isBuffer(fileContent) ? fileContent : Buffer.from(fileContent);
|
|
22
|
-
return { normalizedPath, buffer };
|
|
16
|
+
async function readWorkspaceBinaryFile(context, path, maxBytes = exports.DEFAULT_MAX_FILE_BYTES) {
|
|
17
|
+
const file = await workspaceFiles.read(context, path, { maxBytes });
|
|
18
|
+
return { normalizedPath: file.path, buffer: file.bytes };
|
|
23
19
|
}
|
|
24
20
|
async function writeProcessedText(context, outputPath, content) {
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
}
|
|
29
|
-
const dir = outputPath.substring(0, outputPath.lastIndexOf('/'));
|
|
30
|
-
if (dir && dir !== '/') {
|
|
31
|
-
await workspace.filesystem.mkdir(dir, { recursive: true });
|
|
32
|
-
}
|
|
33
|
-
await workspace.filesystem.writeFile(outputPath, content);
|
|
21
|
+
await workspaceFiles.write(context, outputPath, content, { overwrite: true });
|
|
22
|
+
}
|
|
23
|
+
async function writeProcessedBinary(context, outputPath, content) {
|
|
24
|
+
await workspaceFiles.write(context, outputPath, Buffer.from(content), { overwrite: true });
|
|
34
25
|
}
|
|
35
26
|
function deriveOutputPath(sourcePath, suffix) {
|
|
36
|
-
const
|
|
37
|
-
const
|
|
27
|
+
const normalized = (0, harness_tool_packs_1.normalizeWorkspacePath)(sourcePath);
|
|
28
|
+
const parts = normalized.split('/');
|
|
29
|
+
const filename = parts[parts.length - 1] ?? normalized;
|
|
38
30
|
parts[parts.length - 1] = `${filename}.${suffix}`;
|
|
39
31
|
return parts.join('/');
|
|
40
32
|
}
|
package/dist/utils.js.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"utils.js","sourceRoot":"","sources":["../src/utils.ts"],"names":[],"mappings":"
|
|
1
|
+
{"version":3,"file":"utils.js","sourceRoot":"","sources":["../src/utils.ts"],"names":[],"mappings":";;;AAQA,gEAEC;AAED,0DAOC;AAED,gDAMC;AAED,oDAMC;AAED,4CAMC;AA1CD,oEAA2F;AAE3F,qEAAqE;AACxD,QAAA,sBAAsB,GAAG,EAAE,GAAG,IAAI,GAAG,IAAI,CAAC;AAEvD,MAAM,cAAc,GAAG,IAAA,yCAAoB,EAAC,EAAE,YAAY,EAAE,8BAAsB,EAAE,CAAC,CAAC;AAEtF,SAAgB,0BAA0B,CAAC,IAAY;IACrD,OAAO,IAAA,2CAAsB,EAAC,IAAI,CAAC,CAAC;AACtC,CAAC;AAEM,KAAK,UAAU,uBAAuB,CAC3C,OAA6B,EAC7B,IAAY,EACZ,WAAmB,8BAAsB;IAEzC,MAAM,IAAI,GAAG,MAAM,cAAc,CAAC,IAAI,CAAC,OAAO,EAAE,IAAI,EAAE,EAAE,QAAQ,EAAE,CAAC,CAAC;IACpE,OAAO,EAAE,cAAc,EAAE,IAAI,CAAC,IAAI,EAAE,MAAM,EAAE,IAAI,CAAC,KAAK,EAAE,CAAC;AAC3D,CAAC;AAEM,KAAK,UAAU,kBAAkB,CACtC,OAA6B,EAC7B,UAAkB,EAClB,OAAe;IAEf,MAAM,cAAc,CAAC,KAAK,CAAC,OAAO,EAAE,UAAU,EAAE,OAAO,EAAE,EAAE,SAAS,EAAE,IAAI,EAAE,CAAC,CAAC;AAChF,CAAC;AAEM,KAAK,UAAU,oBAAoB,CACxC,OAA6B,EAC7B,UAAkB,EAClB,OAAmB;IAEnB,MAAM,cAAc,CAAC,KAAK,CAAC,OAAO,EAAE,UAAU,EAAE,MAAM,CAAC,IAAI,CAAC,OAAO,CAAC,EAAE,EAAE,SAAS,EAAE,IAAI,EAAE,CAAC,CAAC;AAC7F,CAAC;AAED,SAAgB,gBAAgB,CAAC,UAAkB,EAAE,MAAc;IACjE,MAAM,UAAU,GAAG,IAAA,2CAAsB,EAAC,UAAU,CAAC,CAAC;IACtD,MAAM,KAAK,GAAG,UAAU,CAAC,KAAK,CAAC,GAAG,CAAC,CAAC;IACpC,MAAM,QAAQ,GAAG,KAAK,CAAC,KAAK,CAAC,MAAM,GAAG,CAAC,CAAC,IAAI,UAAU,CAAC;IACvD,KAAK,CAAC,KAAK,CAAC,MAAM,GAAG,CAAC,CAAC,GAAG,GAAG,QAAQ,IAAI,MAAM,EAAE,CAAC;IAClD,OAAO,KAAK,CAAC,IAAI,CAAC,GAAG,CAAC,CAAC;AACzB,CAAC"}
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@lumifai/harness-tool-pack-file-processing",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.3.0",
|
|
4
4
|
"license": "UNLICENSED",
|
|
5
5
|
"repository": {
|
|
6
6
|
"type": "git",
|
|
@@ -27,15 +27,20 @@
|
|
|
27
27
|
"skills"
|
|
28
28
|
],
|
|
29
29
|
"dependencies": {
|
|
30
|
-
"@llamaindex/liteparse": "^2.0
|
|
31
|
-
"@lumifai/harness-tool-packs": "0.
|
|
32
|
-
"
|
|
33
|
-
"zod": "^4.4.3"
|
|
30
|
+
"@llamaindex/liteparse": "^2.11.0",
|
|
31
|
+
"@lumifai/harness-tool-packs": "0.3.0",
|
|
32
|
+
"zod": "^4.6.5"
|
|
34
33
|
},
|
|
35
34
|
"publishConfig": {
|
|
36
35
|
"access": "public",
|
|
37
36
|
"registry": "https://registry.npmjs.org/"
|
|
38
37
|
},
|
|
38
|
+
"peerDependencies": {
|
|
39
|
+
"@mastra/core": "^1.73.0"
|
|
40
|
+
},
|
|
41
|
+
"devDependencies": {
|
|
42
|
+
"@mastra/core": "1.73.0"
|
|
43
|
+
},
|
|
39
44
|
"scripts": {
|
|
40
45
|
"build": "tsc -p tsconfig.json"
|
|
41
46
|
}
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
---
|
|
2
2
|
name: file-processing
|
|
3
|
-
description: Extract text from PDF files for LLM consumption
|
|
4
|
-
version: 1.
|
|
3
|
+
description: Extract structured Markdown or text from PDF files for LLM consumption
|
|
4
|
+
version: 1.2.0
|
|
5
5
|
tags:
|
|
6
6
|
- file-processing
|
|
7
7
|
- pdf
|
|
@@ -9,21 +9,21 @@ tags:
|
|
|
9
9
|
|
|
10
10
|
# File Processing Tools
|
|
11
11
|
|
|
12
|
-
Use the file-processing tool pack to extract
|
|
12
|
+
Use the file-processing tool pack to extract structured content from PDF documents. Extracted content is written to the workspace filesystem, keeping your context window clean.
|
|
13
13
|
|
|
14
14
|
## Available Tools
|
|
15
15
|
|
|
16
|
-
| Tool | Format | Output
|
|
17
|
-
| ------------- | ------ |
|
|
18
|
-
| `process_pdf` | PDF |
|
|
16
|
+
| Tool | Format | Output |
|
|
17
|
+
| ------------- | ------ | ---------------------------------- |
|
|
18
|
+
| `process_pdf` | PDF | Markdown or text with page markers |
|
|
19
19
|
|
|
20
20
|
## Workflow
|
|
21
21
|
|
|
22
|
-
1. Call `process_pdf` with the
|
|
23
|
-
2. The tool writes
|
|
22
|
+
1. Call `process_pdf` with the workspace-relative path to the file.
|
|
23
|
+
2. The tool writes Markdown to `<source>.md` by default. Pass `format: "text"` for the legacy plain-text format.
|
|
24
24
|
3. Use the returned `output_path` to reference the extracted content in subsequent operations.
|
|
25
25
|
|
|
26
26
|
## Tips
|
|
27
27
|
|
|
28
|
-
- All tools return metadata only
|
|
29
|
-
-
|
|
28
|
+
- All tools return metadata only — the extracted document body stays in the filesystem.
|
|
29
|
+
- Metadata, annotations, form fields, and complexity signals are included by default. Pass `include_images: true` only when embedded image files are needed; image extraction can be expensive.
|