@nebutra/document-pipeline 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md ADDED
@@ -0,0 +1,29 @@
1
+ # @nebutra/document-pipeline
2
+
3
+ Status: WIP — Not yet integrated into any production app.
4
+
5
+ `@nebutra/document-pipeline` owns document parser routing, metadata-preserving
6
+ chunks, content-store ingestion, parser health checks, doctor output, and debug
7
+ inspection. File truth remains in content-store; this package translates
8
+ documents into indexed content units.
9
+
10
+ It does not own Thread/Turn/Item state, prompt generation, model calls,
11
+ sub-agent scheduling, or approval lifecycle. Complex parsing and OCR are
12
+ sidecar-backed capability ports, not runtime logic.
13
+
14
+ ## Commands
15
+
16
+ ```bash
17
+ pnpm docs:doctor
18
+ pnpm docs:debug <job_id>
19
+ pnpm docs:ingest <path>
20
+ pnpm docs:inspect <chunk_id>
21
+ ```
22
+
23
+ ## Examples
24
+
25
+ Executable examples live under `examples/`:
26
+
27
+ - `ingest-markdown.ts`
28
+ - `parse-html.ts`
29
+ - `sidecar-gate.ts`
@@ -0,0 +1,227 @@
1
+ // src/index.ts
2
+ import { mkdir, readFile, writeFile } from "fs/promises";
3
+ import { dirname, extname, join } from "path";
4
+ import { ContentStore } from "@nebutra/content-store";
5
+ import { CapabilityError } from "@nebutra/errors";
6
+ function debugPath(root = process.cwd()) {
7
+ return join(root, ".nebutra", "debug", "document-pipeline.jsonl");
8
+ }
9
+ async function appendDebug(root, entry) {
10
+ const path = debugPath(root);
11
+ await mkdir(dirname(path), { recursive: true });
12
+ await writeFile(path, `${JSON.stringify({ at: (/* @__PURE__ */ new Date()).toISOString(), ...entry })}
13
+ `, {
14
+ flag: "a"
15
+ });
16
+ }
17
+ async function readDocumentPipelineDebug(root = process.cwd(), limit = 20) {
18
+ try {
19
+ const raw = await readFile(debugPath(root), "utf8");
20
+ return raw.trim().split("\n").filter(Boolean).slice(-limit).map((line) => JSON.parse(line));
21
+ } catch {
22
+ return [];
23
+ }
24
+ }
25
+ function requireTenant(explicit, fallback) {
26
+ const tenantId = explicit ?? fallback;
27
+ if (!tenantId) {
28
+ throw new CapabilityError("document-pipeline", "Document ingestion requires tenant context", {
29
+ suggestion: "Pass tenantId on the request or construct DocumentPipeline with a tenantId default.",
30
+ statusCode: 400
31
+ });
32
+ }
33
+ return tenantId;
34
+ }
35
+ var DocumentPipeline = class _DocumentPipeline {
36
+ #tenantId;
37
+ #root;
38
+ #debugRoot;
39
+ #contentStore;
40
+ #sidecarParser;
41
+ constructor(options = {}) {
42
+ this.#tenantId = options.tenantId;
43
+ this.#root = options.root ?? join(process.cwd(), ".nebutra", "document-pipeline");
44
+ this.#debugRoot = options.debugRoot ?? process.cwd();
45
+ this.#contentStore = options.contentStore;
46
+ this.#sidecarParser = options.sidecarParser;
47
+ }
48
+ static async open(root = ".nebutra/document-pipeline", options = {}) {
49
+ const tenantId = options.tenantId ?? "local";
50
+ const contentStore = await ContentStore.open(join(root, "content"), { tenantId });
51
+ return new _DocumentPipeline({ ...options, tenantId, root, contentStore });
52
+ }
53
+ async parse(request) {
54
+ const tenantId = requireTenant(request.tenantId, this.#tenantId);
55
+ const required = { ...request, tenantId };
56
+ const parser = this.#chooseParser(required);
57
+ if (parser === "sidecar") {
58
+ if (!this.#sidecarParser) {
59
+ throw new CapabilityError("document-pipeline", "Parser sidecar is not configured", {
60
+ suggestion: "Configure the parser sidecar for PDFs, office files, OCR, and layout-aware extraction.",
61
+ statusCode: 503,
62
+ metadata: { path: request.source.path }
63
+ });
64
+ }
65
+ const parsed2 = await this.#sidecarParser.parse(required);
66
+ await appendDebug(this.#debugRoot, {
67
+ type: "parse",
68
+ tenantId,
69
+ parser: "sidecar",
70
+ path: parsed2.path
71
+ });
72
+ return parsed2;
73
+ }
74
+ const raw = await readSource(request.source);
75
+ const parsed = parseNativeDocument(required, raw, parser);
76
+ await appendDebug(this.#debugRoot, {
77
+ type: "parse",
78
+ tenantId,
79
+ parser,
80
+ path: parsed.path,
81
+ chunks: parsed.chunks.length
82
+ });
83
+ return parsed;
84
+ }
85
+ async ingest(request) {
86
+ const tenantId = requireTenant(request.tenantId, this.#tenantId);
87
+ const contentStore = await this.#resolveContentStore(tenantId);
88
+ const parsed = await this.parse({ ...request, tenantId });
89
+ const targetPath = request.targetPath ?? parsed.path;
90
+ const content = serializeForContentStore(parsed, request.metadata);
91
+ await contentStore.write(targetPath, content);
92
+ await appendDebug(this.#debugRoot, {
93
+ type: "ingest",
94
+ tenantId,
95
+ path: targetPath,
96
+ parser: parsed.parser,
97
+ chunks: parsed.chunks.length
98
+ });
99
+ return {
100
+ tenantId,
101
+ path: targetPath,
102
+ parser: parsed.parser,
103
+ chunkCount: parsed.chunks.length,
104
+ contentIndexed: true
105
+ };
106
+ }
107
+ async ingestFile(path, options = {}) {
108
+ return this.ingest({ ...options, source: { type: "file", path } });
109
+ }
110
+ async doctor() {
111
+ const contentStore = await this.#resolveContentStore(this.#tenantId ?? "local");
112
+ const sidecar = this.#sidecarParser ? await this.#sidecarParser.doctor() : { ok: false };
113
+ return {
114
+ capability: "document-pipeline",
115
+ nativeParsers: ["markdown", "html", "text"],
116
+ sidecarConfigured: sidecar.ok,
117
+ contentStore: await contentStore.doctor(),
118
+ ...sidecar.ok ? {} : {
119
+ suggestion: "Native parsers are active; configure the parser sidecar before ingesting PDFs, office files, OCR, or complex layout documents."
120
+ }
121
+ };
122
+ }
123
+ #chooseParser(request) {
124
+ if (request.parser && request.parser !== "auto") return request.parser;
125
+ const mime = request.source.mimeType ?? mimeFromPath(request.source.path);
126
+ if (mime === "text/markdown") return "markdown";
127
+ if (mime === "text/html") return "html";
128
+ if (mime.startsWith("text/")) return "text";
129
+ return "sidecar";
130
+ }
131
+ async #resolveContentStore(tenantId) {
132
+ if (this.#contentStore) return this.#contentStore;
133
+ return ContentStore.open(join(this.#root, "content"), { tenantId });
134
+ }
135
+ };
136
+ async function readSource(source) {
137
+ return source.type === "inline" ? source.content : readFile(source.path, "utf8");
138
+ }
139
+ function mimeFromPath(path) {
140
+ const extension = extname(path).toLowerCase();
141
+ if (extension === ".md" || extension === ".mdx") return "text/markdown";
142
+ if (extension === ".html" || extension === ".htm") return "text/html";
143
+ if (extension === ".txt" || extension === ".csv" || extension === ".json") return "text/plain";
144
+ if (extension === ".pdf") return "application/pdf";
145
+ if (extension === ".docx")
146
+ return "application/vnd.openxmlformats-officedocument.wordprocessingml.document";
147
+ if (extension === ".pptx")
148
+ return "application/vnd.openxmlformats-officedocument.presentationml.presentation";
149
+ if (extension === ".png" || extension === ".jpg" || extension === ".jpeg") return "image";
150
+ return "application/octet-stream";
151
+ }
152
+ function parseNativeDocument(request, raw, parser) {
153
+ const mimeType = request.source.mimeType ?? mimeFromPath(request.source.path);
154
+ const { frontmatter, body } = parser === "markdown" ? parseFrontmatter(raw) : { frontmatter: {}, body: raw };
155
+ const text = parser === "html" ? htmlToText(body) : body;
156
+ const chunks = chunkParagraphs(text).map((chunk, index) => ({
157
+ id: `${request.tenantId}:${request.source.path}:${index}`,
158
+ text: chunk,
159
+ metadata: {
160
+ tenantId: request.tenantId,
161
+ sourcePath: request.source.path,
162
+ chunkIndex: index,
163
+ mimeType,
164
+ parser,
165
+ ...request.metadata ?? {}
166
+ }
167
+ }));
168
+ return {
169
+ tenantId: request.tenantId,
170
+ path: request.source.path,
171
+ mimeType,
172
+ parser,
173
+ chunks: chunks.length > 0 ? chunks : [
174
+ {
175
+ id: `${request.tenantId}:${request.source.path}:0`,
176
+ text,
177
+ metadata: {
178
+ tenantId: request.tenantId,
179
+ sourcePath: request.source.path,
180
+ chunkIndex: 0,
181
+ mimeType,
182
+ parser,
183
+ ...request.metadata ?? {}
184
+ }
185
+ }
186
+ ],
187
+ frontmatter
188
+ };
189
+ }
190
+ function serializeForContentStore(parsed, metadata) {
191
+ const frontmatter = {
192
+ ...parsed.frontmatter,
193
+ ...metadata ?? {},
194
+ source_mime: parsed.mimeType,
195
+ parser: parsed.parser
196
+ };
197
+ const frontmatterLines = Object.entries(frontmatter).sort(([left], [right]) => left.localeCompare(right)).map(([key, value]) => `${key}: ${value}`);
198
+ const body = parsed.chunks.map((chunk) => chunk.text).join("\n\n");
199
+ return frontmatterLines.length > 0 ? `---
200
+ ${frontmatterLines.join("\n")}
201
+ ---
202
+ ${body}` : body;
203
+ }
204
+ function parseFrontmatter(content) {
205
+ if (!content.startsWith("---\n")) return { frontmatter: {}, body: content };
206
+ const end = content.indexOf("\n---", 4);
207
+ if (end < 0) return { frontmatter: {}, body: content };
208
+ const raw = content.slice(4, end).trim();
209
+ const frontmatter = {};
210
+ for (const line of raw.split("\n")) {
211
+ const idx = line.indexOf(":");
212
+ if (idx > 0) frontmatter[line.slice(0, idx).trim()] = line.slice(idx + 1).trim();
213
+ }
214
+ return { frontmatter, body: content.slice(end + 4).trimStart() };
215
+ }
216
+ function chunkParagraphs(content) {
217
+ return content.split(/\n\s*\n/).map((part) => part.trim()).filter(Boolean);
218
+ }
219
+ function htmlToText(html) {
220
+ return html.replace(/<script[\s\S]*?<\/script>/gi, " ").replace(/<style[\s\S]*?<\/style>/gi, " ").replace(/<\/(h[1-6]|p|li|tr|div)>/gi, "\n\n").replace(/<[^>]+>/g, " ").replace(/&nbsp;/g, " ").replace(/&amp;/g, "&").replace(/&lt;/g, "<").replace(/&gt;/g, ">").replace(/[ \t]+/g, " ").replace(/\n\s+/g, "\n").trim();
221
+ }
222
+
223
+ export {
224
+ readDocumentPipelineDebug,
225
+ DocumentPipeline
226
+ };
227
+ //# sourceMappingURL=chunk-HZF62UJB.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"sources":["../src/index.ts"],"sourcesContent":["import { mkdir, readFile, writeFile } from \"node:fs/promises\";\nimport { dirname, extname, join } from \"node:path\";\nimport { ContentStore } from \"@nebutra/content-store\";\nimport { CapabilityError } from \"@nebutra/errors\";\n\nexport type ParserChoice = \"auto\" | \"markdown\" | \"html\" | \"text\" | \"sidecar\";\nexport type ChunkerChoice = \"paragraph\" | \"recursive\" | \"hierarchical\";\n\nexport interface InlineSource {\n readonly type: \"inline\";\n readonly path: string;\n readonly content: string;\n readonly mimeType?: string;\n}\n\nexport interface FileSource {\n readonly type: \"file\";\n readonly path: string;\n readonly mimeType?: string;\n}\n\nexport type DocumentSource = InlineSource | FileSource;\n\nexport interface IngestRequest {\n readonly tenantId?: string;\n readonly source: DocumentSource;\n readonly parser?: ParserChoice;\n readonly chunker?: ChunkerChoice;\n readonly targetPath?: string;\n readonly metadata?: Record<string, string>;\n}\n\nexport interface ParseRequest {\n readonly tenantId?: string;\n readonly source: DocumentSource;\n readonly parser?: ParserChoice;\n readonly chunker?: ChunkerChoice;\n readonly metadata?: Record<string, string>;\n}\n\nexport interface ParsedChunk {\n readonly id: string;\n readonly text: string;\n readonly metadata: {\n readonly tenantId: string;\n readonly sourcePath: string;\n readonly chunkIndex: number;\n readonly mimeType: string;\n readonly parser: ParserChoice;\n readonly heading?: string;\n } & Record<string, string | number>;\n}\n\nexport interface ParsedDocument {\n readonly tenantId: string;\n readonly path: string;\n readonly mimeType: string;\n readonly parser: ParserChoice;\n readonly chunks: readonly ParsedChunk[];\n readonly frontmatter: Record<string, string>;\n}\n\nexport interface IngestResult {\n readonly tenantId: string;\n readonly path: string;\n readonly parser: ParserChoice;\n readonly chunkCount: number;\n readonly contentIndexed: true;\n}\n\nexport interface DocumentPipelineDoctorReport {\n readonly capability: \"document-pipeline\";\n readonly nativeParsers: readonly ParserChoice[];\n readonly sidecarConfigured: boolean;\n readonly contentStore: Awaited<ReturnType<ContentStore[\"doctor\"]>>;\n readonly suggestion?: string;\n}\n\nexport interface DocumentParser {\n parse(request: RequiredTenant<ParseRequest>): Promise<ParsedDocument>;\n doctor(): Promise<{ readonly ok: boolean; readonly suggestion?: string }>;\n}\n\nexport interface DocumentPipelineOptions {\n readonly tenantId?: string;\n readonly root?: string;\n readonly debugRoot?: string;\n readonly contentStore?: ContentStore;\n readonly sidecarParser?: DocumentParser;\n}\n\ntype RequiredTenant<T extends { readonly tenantId?: string }> = Omit<T, \"tenantId\"> & {\n readonly tenantId: string;\n};\n\nfunction debugPath(root = process.cwd()): string {\n return join(root, \".nebutra\", \"debug\", \"document-pipeline.jsonl\");\n}\n\nasync function appendDebug(root: string, entry: Record<string, unknown>): Promise<void> {\n const path = debugPath(root);\n await mkdir(dirname(path), { recursive: true });\n await writeFile(path, `${JSON.stringify({ at: new Date().toISOString(), ...entry })}\\n`, {\n flag: \"a\",\n });\n}\n\nexport async function readDocumentPipelineDebug(\n root = process.cwd(),\n limit = 20,\n): Promise<unknown[]> {\n try {\n const raw = await readFile(debugPath(root), \"utf8\");\n return raw\n .trim()\n .split(\"\\n\")\n .filter(Boolean)\n .slice(-limit)\n .map((line) => JSON.parse(line) as unknown);\n } catch {\n return [];\n }\n}\n\nfunction requireTenant(explicit: string | undefined, fallback: string | undefined): string {\n const tenantId = explicit ?? fallback;\n if (!tenantId) {\n throw new CapabilityError(\"document-pipeline\", \"Document ingestion requires tenant context\", {\n suggestion:\n \"Pass tenantId on the request or construct DocumentPipeline with a tenantId default.\",\n statusCode: 400,\n });\n }\n return tenantId;\n}\n\nexport class DocumentPipeline {\n readonly #tenantId: string | undefined;\n readonly #root: string;\n readonly #debugRoot: string;\n readonly #contentStore: ContentStore | undefined;\n readonly #sidecarParser: DocumentParser | undefined;\n\n constructor(options: DocumentPipelineOptions = {}) {\n this.#tenantId = options.tenantId;\n this.#root = options.root ?? join(process.cwd(), \".nebutra\", \"document-pipeline\");\n this.#debugRoot = options.debugRoot ?? process.cwd();\n this.#contentStore = options.contentStore;\n this.#sidecarParser = options.sidecarParser;\n }\n\n static async open(\n root = \".nebutra/document-pipeline\",\n options: Omit<DocumentPipelineOptions, \"root\" | \"contentStore\"> = {},\n ): Promise<DocumentPipeline> {\n const tenantId = options.tenantId ?? \"local\";\n const contentStore = await ContentStore.open(join(root, \"content\"), { tenantId });\n return new DocumentPipeline({ ...options, tenantId, root, contentStore });\n }\n\n async parse(request: ParseRequest): Promise<ParsedDocument> {\n const tenantId = requireTenant(request.tenantId, this.#tenantId);\n const required: RequiredTenant<ParseRequest> = { ...request, tenantId };\n const parser = this.#chooseParser(required);\n if (parser === \"sidecar\") {\n if (!this.#sidecarParser) {\n throw new CapabilityError(\"document-pipeline\", \"Parser sidecar is not configured\", {\n suggestion:\n \"Configure the parser sidecar for PDFs, office files, OCR, and layout-aware extraction.\",\n statusCode: 503,\n metadata: { path: request.source.path },\n });\n }\n const parsed = await this.#sidecarParser.parse(required);\n await appendDebug(this.#debugRoot, {\n type: \"parse\",\n tenantId,\n parser: \"sidecar\",\n path: parsed.path,\n });\n return parsed;\n }\n\n const raw = await readSource(request.source);\n const parsed = parseNativeDocument(required, raw, parser);\n await appendDebug(this.#debugRoot, {\n type: \"parse\",\n tenantId,\n parser,\n path: parsed.path,\n chunks: parsed.chunks.length,\n });\n return parsed;\n }\n\n async ingest(request: IngestRequest): Promise<IngestResult> {\n const tenantId = requireTenant(request.tenantId, this.#tenantId);\n const contentStore = await this.#resolveContentStore(tenantId);\n const parsed = await this.parse({ ...request, tenantId });\n const targetPath = request.targetPath ?? parsed.path;\n const content = serializeForContentStore(parsed, request.metadata);\n await contentStore.write(targetPath, content);\n await appendDebug(this.#debugRoot, {\n type: \"ingest\",\n tenantId,\n path: targetPath,\n parser: parsed.parser,\n chunks: parsed.chunks.length,\n });\n return {\n tenantId,\n path: targetPath,\n parser: parsed.parser,\n chunkCount: parsed.chunks.length,\n contentIndexed: true,\n };\n }\n\n async ingestFile(\n path: string,\n options: Omit<IngestRequest, \"source\"> = {},\n ): Promise<IngestResult> {\n return this.ingest({ ...options, source: { type: \"file\", path } });\n }\n\n async doctor(): Promise<DocumentPipelineDoctorReport> {\n const contentStore = await this.#resolveContentStore(this.#tenantId ?? \"local\");\n const sidecar = this.#sidecarParser ? await this.#sidecarParser.doctor() : { ok: false };\n return {\n capability: \"document-pipeline\",\n nativeParsers: [\"markdown\", \"html\", \"text\"],\n sidecarConfigured: sidecar.ok,\n contentStore: await contentStore.doctor(),\n ...(sidecar.ok\n ? {}\n : {\n suggestion:\n \"Native parsers are active; configure the parser sidecar before ingesting PDFs, office files, OCR, or complex layout documents.\",\n }),\n };\n }\n\n #chooseParser(request: RequiredTenant<ParseRequest>): ParserChoice {\n if (request.parser && request.parser !== \"auto\") return request.parser;\n const mime = request.source.mimeType ?? mimeFromPath(request.source.path);\n if (mime === \"text/markdown\") return \"markdown\";\n if (mime === \"text/html\") return \"html\";\n if (mime.startsWith(\"text/\")) return \"text\";\n return \"sidecar\";\n }\n\n async #resolveContentStore(tenantId: string): Promise<ContentStore> {\n if (this.#contentStore) return this.#contentStore;\n return ContentStore.open(join(this.#root, \"content\"), { tenantId });\n }\n}\n\nasync function readSource(source: DocumentSource): Promise<string> {\n return source.type === \"inline\" ? source.content : readFile(source.path, \"utf8\");\n}\n\nfunction mimeFromPath(path: string): string {\n const extension = extname(path).toLowerCase();\n if (extension === \".md\" || extension === \".mdx\") return \"text/markdown\";\n if (extension === \".html\" || extension === \".htm\") return \"text/html\";\n if (extension === \".txt\" || extension === \".csv\" || extension === \".json\") return \"text/plain\";\n if (extension === \".pdf\") return \"application/pdf\";\n if (extension === \".docx\")\n return \"application/vnd.openxmlformats-officedocument.wordprocessingml.document\";\n if (extension === \".pptx\")\n return \"application/vnd.openxmlformats-officedocument.presentationml.presentation\";\n if (extension === \".png\" || extension === \".jpg\" || extension === \".jpeg\") return \"image\";\n return \"application/octet-stream\";\n}\n\nfunction parseNativeDocument(\n request: RequiredTenant<ParseRequest>,\n raw: string,\n parser: ParserChoice,\n): ParsedDocument {\n const mimeType = request.source.mimeType ?? mimeFromPath(request.source.path);\n const { frontmatter, body } =\n parser === \"markdown\" ? parseFrontmatter(raw) : { frontmatter: {}, body: raw };\n const text = parser === \"html\" ? htmlToText(body) : body;\n const chunks = chunkParagraphs(text).map((chunk, index) => ({\n id: `${request.tenantId}:${request.source.path}:${index}`,\n text: chunk,\n metadata: {\n tenantId: request.tenantId,\n sourcePath: request.source.path,\n chunkIndex: index,\n mimeType,\n parser,\n ...(request.metadata ?? {}),\n },\n }));\n return {\n tenantId: request.tenantId,\n path: request.source.path,\n mimeType,\n parser,\n chunks:\n chunks.length > 0\n ? chunks\n : [\n {\n id: `${request.tenantId}:${request.source.path}:0`,\n text,\n metadata: {\n tenantId: request.tenantId,\n sourcePath: request.source.path,\n chunkIndex: 0,\n mimeType,\n parser,\n ...(request.metadata ?? {}),\n },\n },\n ],\n frontmatter,\n };\n}\n\nfunction serializeForContentStore(\n parsed: ParsedDocument,\n metadata: Record<string, string> | undefined,\n): string {\n const frontmatter = {\n ...parsed.frontmatter,\n ...(metadata ?? {}),\n source_mime: parsed.mimeType,\n parser: parsed.parser,\n };\n const frontmatterLines = Object.entries(frontmatter)\n .sort(([left], [right]) => left.localeCompare(right))\n .map(([key, value]) => `${key}: ${value}`);\n const body = parsed.chunks.map((chunk) => chunk.text).join(\"\\n\\n\");\n return frontmatterLines.length > 0 ? `---\\n${frontmatterLines.join(\"\\n\")}\\n---\\n${body}` : body;\n}\n\nfunction parseFrontmatter(content: string): {\n readonly frontmatter: Record<string, string>;\n readonly body: string;\n} {\n if (!content.startsWith(\"---\\n\")) return { frontmatter: {}, body: content };\n const end = content.indexOf(\"\\n---\", 4);\n if (end < 0) return { frontmatter: {}, body: content };\n const raw = content.slice(4, end).trim();\n const frontmatter: Record<string, string> = {};\n for (const line of raw.split(\"\\n\")) {\n const idx = line.indexOf(\":\");\n if (idx > 0) frontmatter[line.slice(0, idx).trim()] = line.slice(idx + 1).trim();\n }\n return { frontmatter, body: content.slice(end + 4).trimStart() };\n}\n\nfunction chunkParagraphs(content: string): string[] {\n return content\n .split(/\\n\\s*\\n/)\n .map((part) => part.trim())\n .filter(Boolean);\n}\n\nfunction htmlToText(html: string): string {\n return html\n .replace(/<script[\\s\\S]*?<\\/script>/gi, \" \")\n .replace(/<style[\\s\\S]*?<\\/style>/gi, \" \")\n .replace(/<\\/(h[1-6]|p|li|tr|div)>/gi, \"\\n\\n\")\n .replace(/<[^>]+>/g, \" \")\n .replace(/&nbsp;/g, \" \")\n .replace(/&amp;/g, \"&\")\n .replace(/&lt;/g, \"<\")\n .replace(/&gt;/g, \">\")\n .replace(/[ \\t]+/g, \" \")\n .replace(/\\n\\s+/g, \"\\n\")\n .trim();\n}\n"],"mappings":";AAAA,SAAS,OAAO,UAAU,iBAAiB;AAC3C,SAAS,SAAS,SAAS,YAAY;AACvC,SAAS,oBAAoB;AAC7B,SAAS,uBAAuB;AA4FhC,SAAS,UAAU,OAAO,QAAQ,IAAI,GAAW;AAC/C,SAAO,KAAK,MAAM,YAAY,SAAS,yBAAyB;AAClE;AAEA,eAAe,YAAY,MAAc,OAA+C;AACtF,QAAM,OAAO,UAAU,IAAI;AAC3B,QAAM,MAAM,QAAQ,IAAI,GAAG,EAAE,WAAW,KAAK,CAAC;AAC9C,QAAM,UAAU,MAAM,GAAG,KAAK,UAAU,EAAE,KAAI,oBAAI,KAAK,GAAE,YAAY,GAAG,GAAG,MAAM,CAAC,CAAC;AAAA,GAAM;AAAA,IACvF,MAAM;AAAA,EACR,CAAC;AACH;AAEA,eAAsB,0BACpB,OAAO,QAAQ,IAAI,GACnB,QAAQ,IACY;AACpB,MAAI;AACF,UAAM,MAAM,MAAM,SAAS,UAAU,IAAI,GAAG,MAAM;AAClD,WAAO,IACJ,KAAK,EACL,MAAM,IAAI,EACV,OAAO,OAAO,EACd,MAAM,CAAC,KAAK,EACZ,IAAI,CAAC,SAAS,KAAK,MAAM,IAAI,CAAY;AAAA,EAC9C,QAAQ;AACN,WAAO,CAAC;AAAA,EACV;AACF;AAEA,SAAS,cAAc,UAA8B,UAAsC;AACzF,QAAM,WAAW,YAAY;AAC7B,MAAI,CAAC,UAAU;AACb,UAAM,IAAI,gBAAgB,qBAAqB,8CAA8C;AAAA,MAC3F,YACE;AAAA,MACF,YAAY;AAAA,IACd,CAAC;AAAA,EACH;AACA,SAAO;AACT;AAEO,IAAM,mBAAN,MAAM,kBAAiB;AAAA,EACnB;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EAET,YAAY,UAAmC,CAAC,GAAG;AACjD,SAAK,YAAY,QAAQ;AACzB,SAAK,QAAQ,QAAQ,QAAQ,KAAK,QAAQ,IAAI,GAAG,YAAY,mBAAmB;AAChF,SAAK,aAAa,QAAQ,aAAa,QAAQ,IAAI;AACnD,SAAK,gBAAgB,QAAQ;AAC7B,SAAK,iBAAiB,QAAQ;AAAA,EAChC;AAAA,EAEA,aAAa,KACX,OAAO,8BACP,UAAkE,CAAC,GACxC;AAC3B,UAAM,WAAW,QAAQ,YAAY;AACrC,UAAM,eAAe,MAAM,aAAa,KAAK,KAAK,MAAM,SAAS,GAAG,EAAE,SAAS,CAAC;AAChF,WAAO,IAAI,kBAAiB,EAAE,GAAG,SAAS,UAAU,MAAM,aAAa,CAAC;AAAA,EAC1E;AAAA,EAEA,MAAM,MAAM,SAAgD;AAC1D,UAAM,WAAW,cAAc,QAAQ,UAAU,KAAK,SAAS;AAC/D,UAAM,WAAyC,EAAE,GAAG,SAAS,SAAS;AACtE,UAAM,SAAS,KAAK,cAAc,QAAQ;AAC1C,QAAI,WAAW,WAAW;AACxB,UAAI,CAAC,KAAK,gBAAgB;AACxB,cAAM,IAAI,gBAAgB,qBAAqB,oCAAoC;AAAA,UACjF,YACE;AAAA,UACF,YAAY;AAAA,UACZ,UAAU,EAAE,MAAM,QAAQ,OAAO,KAAK;AAAA,QACxC,CAAC;AAAA,MACH;AACA,YAAMA,UAAS,MAAM,KAAK,eAAe,MAAM,QAAQ;AACvD,YAAM,YAAY,KAAK,YAAY;AAAA,QACjC,MAAM;AAAA,QACN;AAAA,QACA,QAAQ;AAAA,QACR,MAAMA,QAAO;AAAA,MACf,CAAC;AACD,aAAOA;AAAA,IACT;AAEA,UAAM,MAAM,MAAM,WAAW,QAAQ,MAAM;AAC3C,UAAM,SAAS,oBAAoB,UAAU,KAAK,MAAM;AACxD,UAAM,YAAY,KAAK,YAAY;AAAA,MACjC,MAAM;AAAA,MACN;AAAA,MACA;AAAA,MACA,MAAM,OAAO;AAAA,MACb,QAAQ,OAAO,OAAO;AAAA,IACxB,CAAC;AACD,WAAO;AAAA,EACT;AAAA,EAEA,MAAM,OAAO,SAA+C;AAC1D,UAAM,WAAW,cAAc,QAAQ,UAAU,KAAK,SAAS;AAC/D,UAAM,eAAe,MAAM,KAAK,qBAAqB,QAAQ;AAC7D,UAAM,SAAS,MAAM,KAAK,MAAM,EAAE,GAAG,SAAS,SAAS,CAAC;AACxD,UAAM,aAAa,QAAQ,cAAc,OAAO;AAChD,UAAM,UAAU,yBAAyB,QAAQ,QAAQ,QAAQ;AACjE,UAAM,aAAa,MAAM,YAAY,OAAO;AAC5C,UAAM,YAAY,KAAK,YAAY;AAAA,MACjC,MAAM;AAAA,MACN;AAAA,MACA,MAAM;AAAA,MACN,QAAQ,OAAO;AAAA,MACf,QAAQ,OAAO,OAAO;AAAA,IACxB,CAAC;AACD,WAAO;AAAA,MACL;AAAA,MACA,MAAM;AAAA,MACN,QAAQ,OAAO;AAAA,MACf,YAAY,OAAO,OAAO;AAAA,MAC1B,gBAAgB;AAAA,IAClB;AAAA,EACF;AAAA,EAEA,MAAM,WACJ,MACA,UAAyC,CAAC,GACnB;AACvB,WAAO,KAAK,OAAO,EAAE,GAAG,SAAS,QAAQ,EAAE,MAAM,QAAQ,KAAK,EAAE,CAAC;AAAA,EACnE;AAAA,EAEA,MAAM,SAAgD;AACpD,UAAM,eAAe,MAAM,KAAK,qBAAqB,KAAK,aAAa,OAAO;AAC9E,UAAM,UAAU,KAAK,iBAAiB,MAAM,KAAK,eAAe,OAAO,IAAI,EAAE,IAAI,MAAM;AACvF,WAAO;AAAA,MACL,YAAY;AAAA,MACZ,eAAe,CAAC,YAAY,QAAQ,MAAM;AAAA,MAC1C,mBAAmB,QAAQ;AAAA,MAC3B,cAAc,MAAM,aAAa,OAAO;AAAA,MACxC,GAAI,QAAQ,KACR,CAAC,IACD;AAAA,QACE,YACE;AAAA,MACJ;AAAA,IACN;AAAA,EACF;AAAA,EAEA,cAAc,SAAqD;AACjE,QAAI,QAAQ,UAAU,QAAQ,WAAW,OAAQ,QAAO,QAAQ;AAChE,UAAM,OAAO,QAAQ,OAAO,YAAY,aAAa,QAAQ,OAAO,IAAI;AACxE,QAAI,SAAS,gBAAiB,QAAO;AACrC,QAAI,SAAS,YAAa,QAAO;AACjC,QAAI,KAAK,WAAW,OAAO,EAAG,QAAO;AACrC,WAAO;AAAA,EACT;AAAA,EAEA,MAAM,qBAAqB,UAAyC;AAClE,QAAI,KAAK,cAAe,QAAO,KAAK;AACpC,WAAO,aAAa,KAAK,KAAK,KAAK,OAAO,SAAS,GAAG,EAAE,SAAS,CAAC;AAAA,EACpE;AACF;AAEA,eAAe,WAAW,QAAyC;AACjE,SAAO,OAAO,SAAS,WAAW,OAAO,UAAU,SAAS,OAAO,MAAM,MAAM;AACjF;AAEA,SAAS,aAAa,MAAsB;AAC1C,QAAM,YAAY,QAAQ,IAAI,EAAE,YAAY;AAC5C,MAAI,cAAc,SAAS,cAAc,OAAQ,QAAO;AACxD,MAAI,cAAc,WAAW,cAAc,OAAQ,QAAO;AAC1D,MAAI,cAAc,UAAU,cAAc,UAAU,cAAc,QAAS,QAAO;AAClF,MAAI,cAAc,OAAQ,QAAO;AACjC,MAAI,cAAc;AAChB,WAAO;AACT,MAAI,cAAc;AAChB,WAAO;AACT,MAAI,cAAc,UAAU,cAAc,UAAU,cAAc,QAAS,QAAO;AAClF,SAAO;AACT;AAEA,SAAS,oBACP,SACA,KACA,QACgB;AAChB,QAAM,WAAW,QAAQ,OAAO,YAAY,aAAa,QAAQ,OAAO,IAAI;AAC5E,QAAM,EAAE,aAAa,KAAK,IACxB,WAAW,aAAa,iBAAiB,GAAG,IAAI,EAAE,aAAa,CAAC,GAAG,MAAM,IAAI;AAC/E,QAAM,OAAO,WAAW,SAAS,WAAW,IAAI,IAAI;AACpD,QAAM,SAAS,gBAAgB,IAAI,EAAE,IAAI,CAAC,OAAO,WAAW;AAAA,IAC1D,IAAI,GAAG,QAAQ,QAAQ,IAAI,QAAQ,OAAO,IAAI,IAAI,KAAK;AAAA,IACvD,MAAM;AAAA,IACN,UAAU;AAAA,MACR,UAAU,QAAQ;AAAA,MAClB,YAAY,QAAQ,OAAO;AAAA,MAC3B,YAAY;AAAA,MACZ;AAAA,MACA;AAAA,MACA,GAAI,QAAQ,YAAY,CAAC;AAAA,IAC3B;AAAA,EACF,EAAE;AACF,SAAO;AAAA,IACL,UAAU,QAAQ;AAAA,IAClB,MAAM,QAAQ,OAAO;AAAA,IACrB;AAAA,IACA;AAAA,IACA,QACE,OAAO,SAAS,IACZ,SACA;AAAA,MACE;AAAA,QACE,IAAI,GAAG,QAAQ,QAAQ,IAAI,QAAQ,OAAO,IAAI;AAAA,QAC9C;AAAA,QACA,UAAU;AAAA,UACR,UAAU,QAAQ;AAAA,UAClB,YAAY,QAAQ,OAAO;AAAA,UAC3B,YAAY;AAAA,UACZ;AAAA,UACA;AAAA,UACA,GAAI,QAAQ,YAAY,CAAC;AAAA,QAC3B;AAAA,MACF;AAAA,IACF;AAAA,IACN;AAAA,EACF;AACF;AAEA,SAAS,yBACP,QACA,UACQ;AACR,QAAM,cAAc;AAAA,IAClB,GAAG,OAAO;AAAA,IACV,GAAI,YAAY,CAAC;AAAA,IACjB,aAAa,OAAO;AAAA,IACpB,QAAQ,OAAO;AAAA,EACjB;AACA,QAAM,mBAAmB,OAAO,QAAQ,WAAW,EAChD,KAAK,CAAC,CAAC,IAAI,GAAG,CAAC,KAAK,MAAM,KAAK,cAAc,KAAK,CAAC,EACnD,IAAI,CAAC,CAAC,KAAK,KAAK,MAAM,GAAG,GAAG,KAAK,KAAK,EAAE;AAC3C,QAAM,OAAO,OAAO,OAAO,IAAI,CAAC,UAAU,MAAM,IAAI,EAAE,KAAK,MAAM;AACjE,SAAO,iBAAiB,SAAS,IAAI;AAAA,EAAQ,iBAAiB,KAAK,IAAI,CAAC;AAAA;AAAA,EAAU,IAAI,KAAK;AAC7F;AAEA,SAAS,iBAAiB,SAGxB;AACA,MAAI,CAAC,QAAQ,WAAW,OAAO,EAAG,QAAO,EAAE,aAAa,CAAC,GAAG,MAAM,QAAQ;AAC1E,QAAM,MAAM,QAAQ,QAAQ,SAAS,CAAC;AACtC,MAAI,MAAM,EAAG,QAAO,EAAE,aAAa,CAAC,GAAG,MAAM,QAAQ;AACrD,QAAM,MAAM,QAAQ,MAAM,GAAG,GAAG,EAAE,KAAK;AACvC,QAAM,cAAsC,CAAC;AAC7C,aAAW,QAAQ,IAAI,MAAM,IAAI,GAAG;AAClC,UAAM,MAAM,KAAK,QAAQ,GAAG;AAC5B,QAAI,MAAM,EAAG,aAAY,KAAK,MAAM,GAAG,GAAG,EAAE,KAAK,CAAC,IAAI,KAAK,MAAM,MAAM,CAAC,EAAE,KAAK;AAAA,EACjF;AACA,SAAO,EAAE,aAAa,MAAM,QAAQ,MAAM,MAAM,CAAC,EAAE,UAAU,EAAE;AACjE;AAEA,SAAS,gBAAgB,SAA2B;AAClD,SAAO,QACJ,MAAM,SAAS,EACf,IAAI,CAAC,SAAS,KAAK,KAAK,CAAC,EACzB,OAAO,OAAO;AACnB;AAEA,SAAS,WAAW,MAAsB;AACxC,SAAO,KACJ,QAAQ,+BAA+B,GAAG,EAC1C,QAAQ,6BAA6B,GAAG,EACxC,QAAQ,8BAA8B,MAAM,EAC5C,QAAQ,YAAY,GAAG,EACvB,QAAQ,WAAW,GAAG,EACtB,QAAQ,UAAU,GAAG,EACrB,QAAQ,SAAS,GAAG,EACpB,QAAQ,SAAS,GAAG,EACpB,QAAQ,WAAW,GAAG,EACtB,QAAQ,UAAU,IAAI,EACtB,KAAK;AACV;","names":["parsed"]}
package/dist/cli.d.ts ADDED
@@ -0,0 +1,2 @@
1
+
2
+ export { }
package/dist/cli.js ADDED
@@ -0,0 +1,39 @@
1
+ import {
2
+ DocumentPipeline,
3
+ readDocumentPipelineDebug
4
+ } from "./chunk-HZF62UJB.js";
5
+
6
+ // src/cli.ts
7
+ var command = process.argv[2] ?? "doctor";
8
+ var pipeline = await DocumentPipeline.open(
9
+ process.env.DOCUMENT_PIPELINE_ROOT ?? ".nebutra/document-pipeline",
10
+ {
11
+ tenantId: process.env.NEBUTRA_TENANT_ID ?? "local"
12
+ }
13
+ );
14
+ if (command === "doctor") {
15
+ process.stdout.write(`${JSON.stringify(await pipeline.doctor(), null, 2)}
16
+ `);
17
+ } else if (command === "ingest") {
18
+ const path = process.argv[3];
19
+ if (!path) {
20
+ process.stderr.write("Usage: pnpm docs:ingest <path>\n");
21
+ process.exitCode = 1;
22
+ } else {
23
+ process.stdout.write(
24
+ `${JSON.stringify({ capability: "document-pipeline", result: await pipeline.ingestFile(path) }, null, 2)}
25
+ `
26
+ );
27
+ }
28
+ } else if (command === "inspect" || command === "debug") {
29
+ const entries = await readDocumentPipelineDebug();
30
+ process.stdout.write(
31
+ `${JSON.stringify({ capability: "document-pipeline", entries }, null, 2)}
32
+ `
33
+ );
34
+ } else {
35
+ process.stderr.write(`Unknown document-pipeline command: ${command}
36
+ `);
37
+ process.exitCode = 1;
38
+ }
39
+ //# sourceMappingURL=cli.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"sources":["../src/cli.ts"],"sourcesContent":["import { DocumentPipeline, readDocumentPipelineDebug } from \"./index\";\n\nconst command = process.argv[2] ?? \"doctor\";\nconst pipeline = await DocumentPipeline.open(\n process.env.DOCUMENT_PIPELINE_ROOT ?? \".nebutra/document-pipeline\",\n {\n tenantId: process.env.NEBUTRA_TENANT_ID ?? \"local\",\n },\n);\n\nif (command === \"doctor\") {\n process.stdout.write(`${JSON.stringify(await pipeline.doctor(), null, 2)}\\n`);\n} else if (command === \"ingest\") {\n const path = process.argv[3];\n if (!path) {\n process.stderr.write(\"Usage: pnpm docs:ingest <path>\\n\");\n process.exitCode = 1;\n } else {\n process.stdout.write(\n `${JSON.stringify({ capability: \"document-pipeline\", result: await pipeline.ingestFile(path) }, null, 2)}\\n`,\n );\n }\n} else if (command === \"inspect\" || command === \"debug\") {\n const entries = await readDocumentPipelineDebug();\n process.stdout.write(\n `${JSON.stringify({ capability: \"document-pipeline\", entries }, null, 2)}\\n`,\n );\n} else {\n process.stderr.write(`Unknown document-pipeline command: ${command}\\n`);\n process.exitCode = 1;\n}\n"],"mappings":";;;;;;AAEA,IAAM,UAAU,QAAQ,KAAK,CAAC,KAAK;AACnC,IAAM,WAAW,MAAM,iBAAiB;AAAA,EACtC,QAAQ,IAAI,0BAA0B;AAAA,EACtC;AAAA,IACE,UAAU,QAAQ,IAAI,qBAAqB;AAAA,EAC7C;AACF;AAEA,IAAI,YAAY,UAAU;AACxB,UAAQ,OAAO,MAAM,GAAG,KAAK,UAAU,MAAM,SAAS,OAAO,GAAG,MAAM,CAAC,CAAC;AAAA,CAAI;AAC9E,WAAW,YAAY,UAAU;AAC/B,QAAM,OAAO,QAAQ,KAAK,CAAC;AAC3B,MAAI,CAAC,MAAM;AACT,YAAQ,OAAO,MAAM,kCAAkC;AACvD,YAAQ,WAAW;AAAA,EACrB,OAAO;AACL,YAAQ,OAAO;AAAA,MACb,GAAG,KAAK,UAAU,EAAE,YAAY,qBAAqB,QAAQ,MAAM,SAAS,WAAW,IAAI,EAAE,GAAG,MAAM,CAAC,CAAC;AAAA;AAAA,IAC1G;AAAA,EACF;AACF,WAAW,YAAY,aAAa,YAAY,SAAS;AACvD,QAAM,UAAU,MAAM,0BAA0B;AAChD,UAAQ,OAAO;AAAA,IACb,GAAG,KAAK,UAAU,EAAE,YAAY,qBAAqB,QAAQ,GAAG,MAAM,CAAC,CAAC;AAAA;AAAA,EAC1E;AACF,OAAO;AACL,UAAQ,OAAO,MAAM,sCAAsC,OAAO;AAAA,CAAI;AACtE,UAAQ,WAAW;AACrB;","names":[]}
@@ -0,0 +1,96 @@
1
+ import { ContentStore } from '@nebutra/content-store';
2
+
3
+ type ParserChoice = "auto" | "markdown" | "html" | "text" | "sidecar";
4
+ type ChunkerChoice = "paragraph" | "recursive" | "hierarchical";
5
+ interface InlineSource {
6
+ readonly type: "inline";
7
+ readonly path: string;
8
+ readonly content: string;
9
+ readonly mimeType?: string;
10
+ }
11
+ interface FileSource {
12
+ readonly type: "file";
13
+ readonly path: string;
14
+ readonly mimeType?: string;
15
+ }
16
+ type DocumentSource = InlineSource | FileSource;
17
+ interface IngestRequest {
18
+ readonly tenantId?: string;
19
+ readonly source: DocumentSource;
20
+ readonly parser?: ParserChoice;
21
+ readonly chunker?: ChunkerChoice;
22
+ readonly targetPath?: string;
23
+ readonly metadata?: Record<string, string>;
24
+ }
25
+ interface ParseRequest {
26
+ readonly tenantId?: string;
27
+ readonly source: DocumentSource;
28
+ readonly parser?: ParserChoice;
29
+ readonly chunker?: ChunkerChoice;
30
+ readonly metadata?: Record<string, string>;
31
+ }
32
+ interface ParsedChunk {
33
+ readonly id: string;
34
+ readonly text: string;
35
+ readonly metadata: {
36
+ readonly tenantId: string;
37
+ readonly sourcePath: string;
38
+ readonly chunkIndex: number;
39
+ readonly mimeType: string;
40
+ readonly parser: ParserChoice;
41
+ readonly heading?: string;
42
+ } & Record<string, string | number>;
43
+ }
44
+ interface ParsedDocument {
45
+ readonly tenantId: string;
46
+ readonly path: string;
47
+ readonly mimeType: string;
48
+ readonly parser: ParserChoice;
49
+ readonly chunks: readonly ParsedChunk[];
50
+ readonly frontmatter: Record<string, string>;
51
+ }
52
+ interface IngestResult {
53
+ readonly tenantId: string;
54
+ readonly path: string;
55
+ readonly parser: ParserChoice;
56
+ readonly chunkCount: number;
57
+ readonly contentIndexed: true;
58
+ }
59
+ interface DocumentPipelineDoctorReport {
60
+ readonly capability: "document-pipeline";
61
+ readonly nativeParsers: readonly ParserChoice[];
62
+ readonly sidecarConfigured: boolean;
63
+ readonly contentStore: Awaited<ReturnType<ContentStore["doctor"]>>;
64
+ readonly suggestion?: string;
65
+ }
66
+ interface DocumentParser {
67
+ parse(request: RequiredTenant<ParseRequest>): Promise<ParsedDocument>;
68
+ doctor(): Promise<{
69
+ readonly ok: boolean;
70
+ readonly suggestion?: string;
71
+ }>;
72
+ }
73
+ interface DocumentPipelineOptions {
74
+ readonly tenantId?: string;
75
+ readonly root?: string;
76
+ readonly debugRoot?: string;
77
+ readonly contentStore?: ContentStore;
78
+ readonly sidecarParser?: DocumentParser;
79
+ }
80
+ type RequiredTenant<T extends {
81
+ readonly tenantId?: string;
82
+ }> = Omit<T, "tenantId"> & {
83
+ readonly tenantId: string;
84
+ };
85
+ declare function readDocumentPipelineDebug(root?: string, limit?: number): Promise<unknown[]>;
86
+ declare class DocumentPipeline {
87
+ #private;
88
+ constructor(options?: DocumentPipelineOptions);
89
+ static open(root?: string, options?: Omit<DocumentPipelineOptions, "root" | "contentStore">): Promise<DocumentPipeline>;
90
+ parse(request: ParseRequest): Promise<ParsedDocument>;
91
+ ingest(request: IngestRequest): Promise<IngestResult>;
92
+ ingestFile(path: string, options?: Omit<IngestRequest, "source">): Promise<IngestResult>;
93
+ doctor(): Promise<DocumentPipelineDoctorReport>;
94
+ }
95
+
96
+ export { type ChunkerChoice, type DocumentParser, DocumentPipeline, type DocumentPipelineDoctorReport, type DocumentPipelineOptions, type DocumentSource, type FileSource, type IngestRequest, type IngestResult, type InlineSource, type ParseRequest, type ParsedChunk, type ParsedDocument, type ParserChoice, readDocumentPipelineDebug };
package/dist/index.js ADDED
@@ -0,0 +1,9 @@
1
+ import {
2
+ DocumentPipeline,
3
+ readDocumentPipelineDebug
4
+ } from "./chunk-HZF62UJB.js";
5
+ export {
6
+ DocumentPipeline,
7
+ readDocumentPipelineDebug
8
+ };
9
+ //# sourceMappingURL=index.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"sources":[],"sourcesContent":[],"mappings":"","names":[]}
@@ -0,0 +1,15 @@
1
+ import { DocumentPipeline } from "@nebutra/document-pipeline";
2
+
3
+ const pipeline = await DocumentPipeline.open(".nebutra/document-pipeline-example", {
4
+ tenantId: "local",
5
+ });
6
+
7
+ const result = await pipeline.ingest({
8
+ tenantId: "local",
9
+ source: {
10
+ type: "inline",
11
+ path: "research/note.md",
12
+ content: "---\nschema: research_note\n---\nretrieval augmented generation notes",
13
+ },
14
+ });
15
+ process.stdout.write(`${JSON.stringify(result, null, 2)}\n`);
@@ -0,0 +1,15 @@
1
+ import { DocumentPipeline } from "@nebutra/document-pipeline";
2
+
3
+ const pipeline = await DocumentPipeline.open(".nebutra/document-pipeline-example", {
4
+ tenantId: "local",
5
+ });
6
+
7
+ const result = await pipeline.parse({
8
+ tenantId: "local",
9
+ source: {
10
+ type: "inline",
11
+ path: "landing.html",
12
+ content: "<h1>Launch</h1><p>Clear operator language</p>",
13
+ },
14
+ });
15
+ process.stdout.write(`${JSON.stringify(result, null, 2)}\n`);
@@ -0,0 +1,14 @@
1
+ import { DocumentPipeline } from "@nebutra/document-pipeline";
2
+
3
+ const pipeline = await DocumentPipeline.open(".nebutra/document-pipeline-example", {
4
+ tenantId: "local",
5
+ });
6
+
7
+ try {
8
+ await pipeline.parse({
9
+ tenantId: "local",
10
+ source: { type: "file", path: "paper.pdf" },
11
+ });
12
+ } catch (error) {
13
+ process.stdout.write(`${JSON.stringify(error, null, 2)}\n`);
14
+ }
package/package.json ADDED
@@ -0,0 +1,58 @@
1
+ {
2
+ "name": "@nebutra/document-pipeline",
3
+ "version": "0.1.0",
4
+ "description": "Document parse, chunk, embed-port, and content-store ingestion pipeline",
5
+ "private": false,
6
+ "license": "MIT",
7
+ "type": "module",
8
+ "nebutra": {
9
+ "status": "wip",
10
+ "productionReady": false,
11
+ "surface": "execution-capability",
12
+ "requires": [
13
+ "@nebutra/content-store for file-truth indexing",
14
+ "@nebutra/sandbox-runtime for parser sidecars"
15
+ ],
16
+ "gaps": [
17
+ "Complex PDF, office, and OCR parsing require a configured parser sidecar",
18
+ "Embedding is delegated to content-store/model ports rather than owned here",
19
+ "Async ingestion jobs are not yet backed by a durable queue"
20
+ ],
21
+ "featureId": "document-pipeline",
22
+ "category": "ai",
23
+ "summary": "Parser router, metadata-preserving chunks, content-store ingest, doctor, and debug tools"
24
+ },
25
+ "main": "./src/index.ts",
26
+ "types": "./src/index.ts",
27
+ "exports": {
28
+ ".": "./src/index.ts"
29
+ },
30
+ "dependencies": {
31
+ "@nebutra/content-store": "0.1.0",
32
+ "@nebutra/errors": "0.1.0",
33
+ "@nebutra/sandbox-runtime": "0.1.0"
34
+ },
35
+ "devDependencies": {
36
+ "@types/node": "^22.19.15",
37
+ "tsup": "^8.5.1",
38
+ "typescript": "^5.9.3",
39
+ "vitest": "^4.1.4"
40
+ },
41
+ "homepage": "https://github.com/Nebutra/Nebutra-Sailor/tree/main/packages/ai/document-pipeline#readme",
42
+ "repository": {
43
+ "type": "git",
44
+ "url": "git+https://github.com/Nebutra/Nebutra-Sailor.git",
45
+ "directory": "packages/ai/document-pipeline"
46
+ },
47
+ "bugs": {
48
+ "url": "https://github.com/Nebutra/Nebutra-Sailor/issues"
49
+ },
50
+ "publishConfig": {
51
+ "access": "public"
52
+ },
53
+ "scripts": {
54
+ "build": "tsup",
55
+ "test": "vitest run",
56
+ "typecheck": "tsc --noEmit"
57
+ }
58
+ }
package/src/cli.ts ADDED
@@ -0,0 +1,31 @@
1
+ import { DocumentPipeline, readDocumentPipelineDebug } from "./index";
2
+
3
+ const command = process.argv[2] ?? "doctor";
4
+ const pipeline = await DocumentPipeline.open(
5
+ process.env.DOCUMENT_PIPELINE_ROOT ?? ".nebutra/document-pipeline",
6
+ {
7
+ tenantId: process.env.NEBUTRA_TENANT_ID ?? "local",
8
+ },
9
+ );
10
+
11
+ if (command === "doctor") {
12
+ process.stdout.write(`${JSON.stringify(await pipeline.doctor(), null, 2)}\n`);
13
+ } else if (command === "ingest") {
14
+ const path = process.argv[3];
15
+ if (!path) {
16
+ process.stderr.write("Usage: pnpm docs:ingest <path>\n");
17
+ process.exitCode = 1;
18
+ } else {
19
+ process.stdout.write(
20
+ `${JSON.stringify({ capability: "document-pipeline", result: await pipeline.ingestFile(path) }, null, 2)}\n`,
21
+ );
22
+ }
23
+ } else if (command === "inspect" || command === "debug") {
24
+ const entries = await readDocumentPipelineDebug();
25
+ process.stdout.write(
26
+ `${JSON.stringify({ capability: "document-pipeline", entries }, null, 2)}\n`,
27
+ );
28
+ } else {
29
+ process.stderr.write(`Unknown document-pipeline command: ${command}\n`);
30
+ process.exitCode = 1;
31
+ }