@nebutra/document-pipeline 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,93 @@
1
+ import { mkdtemp, rm, writeFile } from "node:fs/promises";
2
+ import { tmpdir } from "node:os";
3
+ import { join } from "node:path";
4
+ import { ContentStore } from "@nebutra/content-store";
5
+ import { afterEach, describe, expect, it } from "vitest";
6
+ import { DocumentPipeline, readDocumentPipelineDebug } from "./index";
7
+
8
+ let root: string | undefined;
9
+ let store: ContentStore | undefined;
10
+
11
+ afterEach(async () => {
12
+ if (store) await store.close();
13
+ if (root) await rm(root, { recursive: true, force: true });
14
+ root = undefined;
15
+ store = undefined;
16
+ });
17
+
18
+ describe("DocumentPipeline", () => {
19
+ it("ingests frontmatter-aware markdown into content-store", async () => {
20
+ root = await mkdtemp(join(tmpdir(), "document-pipeline-"));
21
+ store = await ContentStore.open(join(root, "content"), { tenantId: "tenant_a" });
22
+ const pipeline = new DocumentPipeline({ tenantId: "tenant_a", contentStore: store });
23
+
24
+ const result = await pipeline.ingest({
25
+ tenantId: "tenant_a",
26
+ source: {
27
+ type: "inline",
28
+ path: "research/BRAND.md",
29
+ content: "---\nschema: note\n---\nretrieval note",
30
+ },
31
+ });
32
+
33
+ expect(result).toMatchObject({ path: "research/BRAND.md", chunkCount: 1 });
34
+ await expect(
35
+ store.search().query("retrieval").filter({ schema: "note" }).topK(5),
36
+ ).resolves.toHaveLength(1);
37
+ });
38
+
39
+ it("parses html through the lightweight deterministic parser", async () => {
40
+ root = await mkdtemp(join(tmpdir(), "document-pipeline-"));
41
+ store = await ContentStore.open(join(root, "content"), { tenantId: "tenant_a" });
42
+ const pipeline = new DocumentPipeline({ tenantId: "tenant_a", contentStore: store });
43
+
44
+ const parsed = await pipeline.parse({
45
+ tenantId: "tenant_a",
46
+ source: { type: "inline", path: "page.html", content: "<h1>Title</h1><p>Useful body</p>" },
47
+ });
48
+
49
+ expect(parsed.chunks[0]?.text).toContain("Title");
50
+ expect(parsed.chunks[0]?.metadata.mimeType).toBe("text/html");
51
+ });
52
+
53
+ it("routes complex binary formats to parser sidecars with a fix suggestion", async () => {
54
+ root = await mkdtemp(join(tmpdir(), "document-pipeline-"));
55
+ const pdfPath = join(root, "paper.pdf");
56
+ await writeFile(pdfPath, "%PDF-1.7\n", "utf8");
57
+ store = await ContentStore.open(join(root, "content"), { tenantId: "tenant_a" });
58
+ const pipeline = new DocumentPipeline({ tenantId: "tenant_a", contentStore: store });
59
+
60
+ await expect(
61
+ pipeline.parse({ tenantId: "tenant_a", source: { type: "file", path: pdfPath } }),
62
+ ).rejects.toMatchObject({
63
+ capability: "document-pipeline",
64
+ suggestion: expect.stringContaining("parser sidecar"),
65
+ });
66
+ });
67
+
68
+ it("requires tenant context for persistent ingestion", async () => {
69
+ const pipeline = new DocumentPipeline();
70
+ await expect(
71
+ pipeline.ingest({ source: { type: "inline", path: "note.md", content: "hello" } }),
72
+ ).rejects.toMatchObject({
73
+ capability: "document-pipeline",
74
+ suggestion: expect.stringContaining("tenantId"),
75
+ });
76
+ });
77
+
78
+ it("writes inspectable debug records", async () => {
79
+ root = await mkdtemp(join(tmpdir(), "document-pipeline-"));
80
+ store = await ContentStore.open(join(root, "content"), { tenantId: "tenant_a" });
81
+ const pipeline = new DocumentPipeline({
82
+ tenantId: "tenant_a",
83
+ contentStore: store,
84
+ debugRoot: root,
85
+ });
86
+ await pipeline.ingest({
87
+ tenantId: "tenant_a",
88
+ source: { type: "inline", path: "notes/a.md", content: "debuggable chunk" },
89
+ });
90
+
91
+ expect(await readDocumentPipelineDebug(root)).toEqual(expect.any(Array));
92
+ });
93
+ });
package/src/index.ts ADDED
@@ -0,0 +1,376 @@
1
+ import { mkdir, readFile, writeFile } from "node:fs/promises";
2
+ import { dirname, extname, join } from "node:path";
3
+ import { ContentStore } from "@nebutra/content-store";
4
+ import { CapabilityError } from "@nebutra/errors";
5
+
6
+ export type ParserChoice = "auto" | "markdown" | "html" | "text" | "sidecar";
7
+ export type ChunkerChoice = "paragraph" | "recursive" | "hierarchical";
8
+
9
+ export interface InlineSource {
10
+ readonly type: "inline";
11
+ readonly path: string;
12
+ readonly content: string;
13
+ readonly mimeType?: string;
14
+ }
15
+
16
+ export interface FileSource {
17
+ readonly type: "file";
18
+ readonly path: string;
19
+ readonly mimeType?: string;
20
+ }
21
+
22
+ export type DocumentSource = InlineSource | FileSource;
23
+
24
+ export interface IngestRequest {
25
+ readonly tenantId?: string;
26
+ readonly source: DocumentSource;
27
+ readonly parser?: ParserChoice;
28
+ readonly chunker?: ChunkerChoice;
29
+ readonly targetPath?: string;
30
+ readonly metadata?: Record<string, string>;
31
+ }
32
+
33
+ export interface ParseRequest {
34
+ readonly tenantId?: string;
35
+ readonly source: DocumentSource;
36
+ readonly parser?: ParserChoice;
37
+ readonly chunker?: ChunkerChoice;
38
+ readonly metadata?: Record<string, string>;
39
+ }
40
+
41
+ export interface ParsedChunk {
42
+ readonly id: string;
43
+ readonly text: string;
44
+ readonly metadata: {
45
+ readonly tenantId: string;
46
+ readonly sourcePath: string;
47
+ readonly chunkIndex: number;
48
+ readonly mimeType: string;
49
+ readonly parser: ParserChoice;
50
+ readonly heading?: string;
51
+ } & Record<string, string | number>;
52
+ }
53
+
54
+ export interface ParsedDocument {
55
+ readonly tenantId: string;
56
+ readonly path: string;
57
+ readonly mimeType: string;
58
+ readonly parser: ParserChoice;
59
+ readonly chunks: readonly ParsedChunk[];
60
+ readonly frontmatter: Record<string, string>;
61
+ }
62
+
63
+ export interface IngestResult {
64
+ readonly tenantId: string;
65
+ readonly path: string;
66
+ readonly parser: ParserChoice;
67
+ readonly chunkCount: number;
68
+ readonly contentIndexed: true;
69
+ }
70
+
71
+ export interface DocumentPipelineDoctorReport {
72
+ readonly capability: "document-pipeline";
73
+ readonly nativeParsers: readonly ParserChoice[];
74
+ readonly sidecarConfigured: boolean;
75
+ readonly contentStore: Awaited<ReturnType<ContentStore["doctor"]>>;
76
+ readonly suggestion?: string;
77
+ }
78
+
79
+ export interface DocumentParser {
80
+ parse(request: RequiredTenant<ParseRequest>): Promise<ParsedDocument>;
81
+ doctor(): Promise<{ readonly ok: boolean; readonly suggestion?: string }>;
82
+ }
83
+
84
+ export interface DocumentPipelineOptions {
85
+ readonly tenantId?: string;
86
+ readonly root?: string;
87
+ readonly debugRoot?: string;
88
+ readonly contentStore?: ContentStore;
89
+ readonly sidecarParser?: DocumentParser;
90
+ }
91
+
92
+ type RequiredTenant<T extends { readonly tenantId?: string }> = Omit<T, "tenantId"> & {
93
+ readonly tenantId: string;
94
+ };
95
+
96
+ function debugPath(root = process.cwd()): string {
97
+ return join(root, ".nebutra", "debug", "document-pipeline.jsonl");
98
+ }
99
+
100
+ async function appendDebug(root: string, entry: Record<string, unknown>): Promise<void> {
101
+ const path = debugPath(root);
102
+ await mkdir(dirname(path), { recursive: true });
103
+ await writeFile(path, `${JSON.stringify({ at: new Date().toISOString(), ...entry })}\n`, {
104
+ flag: "a",
105
+ });
106
+ }
107
+
108
+ export async function readDocumentPipelineDebug(
109
+ root = process.cwd(),
110
+ limit = 20,
111
+ ): Promise<unknown[]> {
112
+ try {
113
+ const raw = await readFile(debugPath(root), "utf8");
114
+ return raw
115
+ .trim()
116
+ .split("\n")
117
+ .filter(Boolean)
118
+ .slice(-limit)
119
+ .map((line) => JSON.parse(line) as unknown);
120
+ } catch {
121
+ return [];
122
+ }
123
+ }
124
+
125
+ function requireTenant(explicit: string | undefined, fallback: string | undefined): string {
126
+ const tenantId = explicit ?? fallback;
127
+ if (!tenantId) {
128
+ throw new CapabilityError("document-pipeline", "Document ingestion requires tenant context", {
129
+ suggestion:
130
+ "Pass tenantId on the request or construct DocumentPipeline with a tenantId default.",
131
+ statusCode: 400,
132
+ });
133
+ }
134
+ return tenantId;
135
+ }
136
+
137
+ export class DocumentPipeline {
138
+ readonly #tenantId: string | undefined;
139
+ readonly #root: string;
140
+ readonly #debugRoot: string;
141
+ readonly #contentStore: ContentStore | undefined;
142
+ readonly #sidecarParser: DocumentParser | undefined;
143
+
144
+ constructor(options: DocumentPipelineOptions = {}) {
145
+ this.#tenantId = options.tenantId;
146
+ this.#root = options.root ?? join(process.cwd(), ".nebutra", "document-pipeline");
147
+ this.#debugRoot = options.debugRoot ?? process.cwd();
148
+ this.#contentStore = options.contentStore;
149
+ this.#sidecarParser = options.sidecarParser;
150
+ }
151
+
152
+ static async open(
153
+ root = ".nebutra/document-pipeline",
154
+ options: Omit<DocumentPipelineOptions, "root" | "contentStore"> = {},
155
+ ): Promise<DocumentPipeline> {
156
+ const tenantId = options.tenantId ?? "local";
157
+ const contentStore = await ContentStore.open(join(root, "content"), { tenantId });
158
+ return new DocumentPipeline({ ...options, tenantId, root, contentStore });
159
+ }
160
+
161
+ async parse(request: ParseRequest): Promise<ParsedDocument> {
162
+ const tenantId = requireTenant(request.tenantId, this.#tenantId);
163
+ const required: RequiredTenant<ParseRequest> = { ...request, tenantId };
164
+ const parser = this.#chooseParser(required);
165
+ if (parser === "sidecar") {
166
+ if (!this.#sidecarParser) {
167
+ throw new CapabilityError("document-pipeline", "Parser sidecar is not configured", {
168
+ suggestion:
169
+ "Configure the parser sidecar for PDFs, office files, OCR, and layout-aware extraction.",
170
+ statusCode: 503,
171
+ metadata: { path: request.source.path },
172
+ });
173
+ }
174
+ const parsed = await this.#sidecarParser.parse(required);
175
+ await appendDebug(this.#debugRoot, {
176
+ type: "parse",
177
+ tenantId,
178
+ parser: "sidecar",
179
+ path: parsed.path,
180
+ });
181
+ return parsed;
182
+ }
183
+
184
+ const raw = await readSource(request.source);
185
+ const parsed = parseNativeDocument(required, raw, parser);
186
+ await appendDebug(this.#debugRoot, {
187
+ type: "parse",
188
+ tenantId,
189
+ parser,
190
+ path: parsed.path,
191
+ chunks: parsed.chunks.length,
192
+ });
193
+ return parsed;
194
+ }
195
+
196
+ async ingest(request: IngestRequest): Promise<IngestResult> {
197
+ const tenantId = requireTenant(request.tenantId, this.#tenantId);
198
+ const contentStore = await this.#resolveContentStore(tenantId);
199
+ const parsed = await this.parse({ ...request, tenantId });
200
+ const targetPath = request.targetPath ?? parsed.path;
201
+ const content = serializeForContentStore(parsed, request.metadata);
202
+ await contentStore.write(targetPath, content);
203
+ await appendDebug(this.#debugRoot, {
204
+ type: "ingest",
205
+ tenantId,
206
+ path: targetPath,
207
+ parser: parsed.parser,
208
+ chunks: parsed.chunks.length,
209
+ });
210
+ return {
211
+ tenantId,
212
+ path: targetPath,
213
+ parser: parsed.parser,
214
+ chunkCount: parsed.chunks.length,
215
+ contentIndexed: true,
216
+ };
217
+ }
218
+
219
+ async ingestFile(
220
+ path: string,
221
+ options: Omit<IngestRequest, "source"> = {},
222
+ ): Promise<IngestResult> {
223
+ return this.ingest({ ...options, source: { type: "file", path } });
224
+ }
225
+
226
+ async doctor(): Promise<DocumentPipelineDoctorReport> {
227
+ const contentStore = await this.#resolveContentStore(this.#tenantId ?? "local");
228
+ const sidecar = this.#sidecarParser ? await this.#sidecarParser.doctor() : { ok: false };
229
+ return {
230
+ capability: "document-pipeline",
231
+ nativeParsers: ["markdown", "html", "text"],
232
+ sidecarConfigured: sidecar.ok,
233
+ contentStore: await contentStore.doctor(),
234
+ ...(sidecar.ok
235
+ ? {}
236
+ : {
237
+ suggestion:
238
+ "Native parsers are active; configure the parser sidecar before ingesting PDFs, office files, OCR, or complex layout documents.",
239
+ }),
240
+ };
241
+ }
242
+
243
+ #chooseParser(request: RequiredTenant<ParseRequest>): ParserChoice {
244
+ if (request.parser && request.parser !== "auto") return request.parser;
245
+ const mime = request.source.mimeType ?? mimeFromPath(request.source.path);
246
+ if (mime === "text/markdown") return "markdown";
247
+ if (mime === "text/html") return "html";
248
+ if (mime.startsWith("text/")) return "text";
249
+ return "sidecar";
250
+ }
251
+
252
+ async #resolveContentStore(tenantId: string): Promise<ContentStore> {
253
+ if (this.#contentStore) return this.#contentStore;
254
+ return ContentStore.open(join(this.#root, "content"), { tenantId });
255
+ }
256
+ }
257
+
258
+ async function readSource(source: DocumentSource): Promise<string> {
259
+ return source.type === "inline" ? source.content : readFile(source.path, "utf8");
260
+ }
261
+
262
+ function mimeFromPath(path: string): string {
263
+ const extension = extname(path).toLowerCase();
264
+ if (extension === ".md" || extension === ".mdx") return "text/markdown";
265
+ if (extension === ".html" || extension === ".htm") return "text/html";
266
+ if (extension === ".txt" || extension === ".csv" || extension === ".json") return "text/plain";
267
+ if (extension === ".pdf") return "application/pdf";
268
+ if (extension === ".docx")
269
+ return "application/vnd.openxmlformats-officedocument.wordprocessingml.document";
270
+ if (extension === ".pptx")
271
+ return "application/vnd.openxmlformats-officedocument.presentationml.presentation";
272
+ if (extension === ".png" || extension === ".jpg" || extension === ".jpeg") return "image";
273
+ return "application/octet-stream";
274
+ }
275
+
276
+ function parseNativeDocument(
277
+ request: RequiredTenant<ParseRequest>,
278
+ raw: string,
279
+ parser: ParserChoice,
280
+ ): ParsedDocument {
281
+ const mimeType = request.source.mimeType ?? mimeFromPath(request.source.path);
282
+ const { frontmatter, body } =
283
+ parser === "markdown" ? parseFrontmatter(raw) : { frontmatter: {}, body: raw };
284
+ const text = parser === "html" ? htmlToText(body) : body;
285
+ const chunks = chunkParagraphs(text).map((chunk, index) => ({
286
+ id: `${request.tenantId}:${request.source.path}:${index}`,
287
+ text: chunk,
288
+ metadata: {
289
+ tenantId: request.tenantId,
290
+ sourcePath: request.source.path,
291
+ chunkIndex: index,
292
+ mimeType,
293
+ parser,
294
+ ...(request.metadata ?? {}),
295
+ },
296
+ }));
297
+ return {
298
+ tenantId: request.tenantId,
299
+ path: request.source.path,
300
+ mimeType,
301
+ parser,
302
+ chunks:
303
+ chunks.length > 0
304
+ ? chunks
305
+ : [
306
+ {
307
+ id: `${request.tenantId}:${request.source.path}:0`,
308
+ text,
309
+ metadata: {
310
+ tenantId: request.tenantId,
311
+ sourcePath: request.source.path,
312
+ chunkIndex: 0,
313
+ mimeType,
314
+ parser,
315
+ ...(request.metadata ?? {}),
316
+ },
317
+ },
318
+ ],
319
+ frontmatter,
320
+ };
321
+ }
322
+
323
+ function serializeForContentStore(
324
+ parsed: ParsedDocument,
325
+ metadata: Record<string, string> | undefined,
326
+ ): string {
327
+ const frontmatter = {
328
+ ...parsed.frontmatter,
329
+ ...(metadata ?? {}),
330
+ source_mime: parsed.mimeType,
331
+ parser: parsed.parser,
332
+ };
333
+ const frontmatterLines = Object.entries(frontmatter)
334
+ .sort(([left], [right]) => left.localeCompare(right))
335
+ .map(([key, value]) => `${key}: ${value}`);
336
+ const body = parsed.chunks.map((chunk) => chunk.text).join("\n\n");
337
+ return frontmatterLines.length > 0 ? `---\n${frontmatterLines.join("\n")}\n---\n${body}` : body;
338
+ }
339
+
340
+ function parseFrontmatter(content: string): {
341
+ readonly frontmatter: Record<string, string>;
342
+ readonly body: string;
343
+ } {
344
+ if (!content.startsWith("---\n")) return { frontmatter: {}, body: content };
345
+ const end = content.indexOf("\n---", 4);
346
+ if (end < 0) return { frontmatter: {}, body: content };
347
+ const raw = content.slice(4, end).trim();
348
+ const frontmatter: Record<string, string> = {};
349
+ for (const line of raw.split("\n")) {
350
+ const idx = line.indexOf(":");
351
+ if (idx > 0) frontmatter[line.slice(0, idx).trim()] = line.slice(idx + 1).trim();
352
+ }
353
+ return { frontmatter, body: content.slice(end + 4).trimStart() };
354
+ }
355
+
356
+ function chunkParagraphs(content: string): string[] {
357
+ return content
358
+ .split(/\n\s*\n/)
359
+ .map((part) => part.trim())
360
+ .filter(Boolean);
361
+ }
362
+
363
+ function htmlToText(html: string): string {
364
+ return html
365
+ .replace(/<script[\s\S]*?<\/script>/gi, " ")
366
+ .replace(/<style[\s\S]*?<\/style>/gi, " ")
367
+ .replace(/<\/(h[1-6]|p|li|tr|div)>/gi, "\n\n")
368
+ .replace(/<[^>]+>/g, " ")
369
+ .replace(/&nbsp;/g, " ")
370
+ .replace(/&amp;/g, "&")
371
+ .replace(/&lt;/g, "<")
372
+ .replace(/&gt;/g, ">")
373
+ .replace(/[ \t]+/g, " ")
374
+ .replace(/\n\s+/g, "\n")
375
+ .trim();
376
+ }
package/tsconfig.json ADDED
@@ -0,0 +1,12 @@
1
+ {
2
+ "extends": "../../../tsconfig.base.json",
3
+ "compilerOptions": {
4
+ "module": "ESNext",
5
+ "moduleResolution": "bundler",
6
+ "target": "esnext",
7
+ "types": ["node"],
8
+ "incremental": false
9
+ },
10
+ "include": ["src", "examples"],
11
+ "exclude": ["node_modules", "dist"]
12
+ }
package/tsup.config.ts ADDED
@@ -0,0 +1,11 @@
1
+ import { defineConfig } from "tsup";
2
+
3
+ export default defineConfig({
4
+ entry: ["src/index.ts", "src/cli.ts"],
5
+ format: ["esm"],
6
+ dts: true,
7
+ sourcemap: true,
8
+ clean: true,
9
+ target: "es2022",
10
+ external: ["@nebutra/content-store", "@nebutra/errors", "@nebutra/sandbox-runtime"],
11
+ });