dsh-convert-core 0.1.0-alpha.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +149 -0
- package/bin/media-to-md.mjs +168 -0
- package/bin/tita-convert.mjs +85 -0
- package/docs/media-to-md.md +143 -0
- package/lib/adapters/anydoc.d.ts +73 -0
- package/lib/adapters/anydoc.d.ts.map +1 -0
- package/lib/adapters/anydoc.js +295 -0
- package/lib/adapters/anydoc.js.map +1 -0
- package/lib/adapters/iddoc.d.ts +28 -0
- package/lib/adapters/iddoc.d.ts.map +1 -0
- package/lib/adapters/iddoc.js +153 -0
- package/lib/adapters/iddoc.js.map +1 -0
- package/lib/adapters/index.d.ts +41 -0
- package/lib/adapters/index.d.ts.map +1 -0
- package/lib/adapters/index.js +115 -0
- package/lib/adapters/index.js.map +1 -0
- package/lib/adapters/scanned.d.ts +28 -0
- package/lib/adapters/scanned.d.ts.map +1 -0
- package/lib/adapters/scanned.js +121 -0
- package/lib/adapters/scanned.js.map +1 -0
- package/lib/adapters/text.d.ts +29 -0
- package/lib/adapters/text.d.ts.map +1 -0
- package/lib/adapters/text.js +135 -0
- package/lib/adapters/text.js.map +1 -0
- package/lib/adapters/types.d.ts +65 -0
- package/lib/adapters/types.d.ts.map +1 -0
- package/lib/adapters/types.js +11 -0
- package/lib/adapters/types.js.map +1 -0
- package/lib/contract.d.ts +185 -0
- package/lib/contract.d.ts.map +1 -0
- package/lib/contract.js +59 -0
- package/lib/contract.js.map +1 -0
- package/lib/delivery.d.ts +36 -0
- package/lib/delivery.d.ts.map +1 -0
- package/lib/delivery.js +97 -0
- package/lib/delivery.js.map +1 -0
- package/lib/http.d.ts +34 -0
- package/lib/http.d.ts.map +1 -0
- package/lib/http.js +372 -0
- package/lib/http.js.map +1 -0
- package/lib/markdown.d.ts +32 -0
- package/lib/markdown.d.ts.map +1 -0
- package/lib/markdown.js +145 -0
- package/lib/markdown.js.map +1 -0
- package/lib/media/binaries.d.ts +26 -0
- package/lib/media/binaries.d.ts.map +1 -0
- package/lib/media/binaries.js +60 -0
- package/lib/media/binaries.js.map +1 -0
- package/lib/media/convert.d.ts +19 -0
- package/lib/media/convert.d.ts.map +1 -0
- package/lib/media/convert.js +210 -0
- package/lib/media/convert.js.map +1 -0
- package/lib/media/index.d.ts +27 -0
- package/lib/media/index.d.ts.map +1 -0
- package/lib/media/index.js +27 -0
- package/lib/media/index.js.map +1 -0
- package/lib/media/markdown.d.ts +40 -0
- package/lib/media/markdown.d.ts.map +1 -0
- package/lib/media/markdown.js +89 -0
- package/lib/media/markdown.js.map +1 -0
- package/lib/media/media.d.ts +45 -0
- package/lib/media/media.d.ts.map +1 -0
- package/lib/media/media.js +167 -0
- package/lib/media/media.js.map +1 -0
- package/lib/media/models.d.ts +31 -0
- package/lib/media/models.d.ts.map +1 -0
- package/lib/media/models.js +87 -0
- package/lib/media/models.js.map +1 -0
- package/lib/media/scanned.d.ts +18 -0
- package/lib/media/scanned.d.ts.map +1 -0
- package/lib/media/scanned.js +179 -0
- package/lib/media/scanned.js.map +1 -0
- package/lib/media/types.d.ts +154 -0
- package/lib/media/types.d.ts.map +1 -0
- package/lib/media/types.js +23 -0
- package/lib/media/types.js.map +1 -0
- package/lib/server.d.ts +35 -0
- package/lib/server.d.ts.map +1 -0
- package/lib/server.js +64 -0
- package/lib/server.js.map +1 -0
- package/lib/service.d.ts +170 -0
- package/lib/service.d.ts.map +1 -0
- package/lib/service.js +562 -0
- package/lib/service.js.map +1 -0
- package/package.json +59 -0
- package/scripts/asr.py +55 -0
- package/scripts/ocr.py +40 -0
- package/scripts/pdf_pages.py +49 -0
- package/scripts/vendor.mjs +76 -0
- package/src/adapters/anydoc.ts +356 -0
- package/src/adapters/iddoc.ts +171 -0
- package/src/adapters/index.ts +147 -0
- package/src/adapters/scanned.ts +139 -0
- package/src/adapters/text.ts +146 -0
- package/src/adapters/types.ts +76 -0
- package/src/contract.ts +230 -0
- package/src/delivery.ts +124 -0
- package/src/http.ts +394 -0
- package/src/markdown.ts +141 -0
- package/src/media/binaries.ts +67 -0
- package/src/media/convert.ts +232 -0
- package/src/media/index.ts +43 -0
- package/src/media/markdown.ts +105 -0
- package/src/media/media.ts +204 -0
- package/src/media/models.ts +124 -0
- package/src/media/scanned.ts +200 -0
- package/src/media/types.ts +171 -0
- package/src/server.ts +85 -0
- package/src/service.ts +639 -0
- package/web/app.js +382 -0
- package/web/index.html +217 -0
- package/web/styles.css +263 -0
- package/web/vendor/icons.js +25 -0
- package/web/vendor/vue.global.prod.js +14 -0
package/src/http.ts
ADDED
|
@@ -0,0 +1,394 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Transport for the kernel: a plain `node:http` handler.
|
|
3
|
+
*
|
|
4
|
+
* It is mounted three ways — standalone (`bin/tita-convert.mjs`), inside the
|
|
5
|
+
* DSH web server (`/convert/*`), and from tests — which is exactly why the
|
|
6
|
+
* kernel owns a handler instead of a server.
|
|
7
|
+
*/
|
|
8
|
+
|
|
9
|
+
import { readFile } from 'node:fs/promises'
|
|
10
|
+
import type { IncomingMessage, ServerResponse } from 'node:http'
|
|
11
|
+
import { extname, normalize, resolve, sep } from 'node:path'
|
|
12
|
+
import { fileURLToPath } from 'node:url'
|
|
13
|
+
import { AdapterError } from './adapters/types.js'
|
|
14
|
+
import type { ConvertRequest, ConvertTask } from './contract.js'
|
|
15
|
+
import { ConvertService, contentTypeOf } from './service.js'
|
|
16
|
+
|
|
17
|
+
export interface HandlerOptions {
|
|
18
|
+
service: ConvertService
|
|
19
|
+
/** Path prefix the handler answers on, e.g. `/convert`. Empty means `/`. */
|
|
20
|
+
basePath?: string
|
|
21
|
+
/** Directory holding the Vue console. Resolved relative to this module by default. */
|
|
22
|
+
webRoot?: string
|
|
23
|
+
/** Unauthenticated-local default; when set, every /api call must present it. */
|
|
24
|
+
apiKey?: string
|
|
25
|
+
}
|
|
26
|
+
|
|
27
|
+
const MAX_BODY_BYTES = 512 * 1024 * 1024
|
|
28
|
+
|
|
29
|
+
export function defaultWebRoot(): string {
|
|
30
|
+
return resolve(fileURLToPath(new URL('../web/', import.meta.url)))
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
function sendJson(res: ServerResponse, status: number, body: unknown): void {
|
|
34
|
+
const payload = JSON.stringify(body)
|
|
35
|
+
res.statusCode = status
|
|
36
|
+
res.setHeader('content-type', 'application/json; charset=utf-8')
|
|
37
|
+
res.setHeader('cache-control', 'no-store')
|
|
38
|
+
res.end(payload)
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
function sendError(res: ServerResponse, status: number, error: AdapterError | Error): void {
|
|
42
|
+
const code = error instanceof AdapterError ? error.code : 'INTERNAL_ERROR'
|
|
43
|
+
const suggestion = error instanceof AdapterError ? error.suggestion : undefined
|
|
44
|
+
sendJson(res, status, {
|
|
45
|
+
ok: false,
|
|
46
|
+
error: { code, message: error.message, ...(suggestion ? { suggestion } : {}) },
|
|
47
|
+
})
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
function ok<T>(data: T): { ok: true; data: T } {
|
|
51
|
+
return { ok: true, data }
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
async function readBody(req: IncomingMessage, limit = MAX_BODY_BYTES): Promise<Buffer> {
|
|
55
|
+
const chunks: Buffer[] = []
|
|
56
|
+
let bytes = 0
|
|
57
|
+
for await (const chunk of req) {
|
|
58
|
+
const buffer = Buffer.isBuffer(chunk) ? chunk : Buffer.from(String(chunk))
|
|
59
|
+
bytes += buffer.length
|
|
60
|
+
if (bytes > limit) throw new AdapterError('REQUEST_TOO_LARGE', `请求体超过上限 ${limit} 字节。`)
|
|
61
|
+
chunks.push(buffer)
|
|
62
|
+
}
|
|
63
|
+
return Buffer.concat(chunks)
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
interface MultipartPart {
|
|
67
|
+
name: string
|
|
68
|
+
filename?: string
|
|
69
|
+
contentType?: string
|
|
70
|
+
data: Buffer
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
/** Minimal multipart/form-data reader: no dependency, no streaming, no surprises. */
|
|
74
|
+
export function parseMultipart(body: Buffer, boundary: string): MultipartPart[] {
|
|
75
|
+
const delimiter = Buffer.from(`--${boundary}`)
|
|
76
|
+
const parts: MultipartPart[] = []
|
|
77
|
+
let index = body.indexOf(delimiter)
|
|
78
|
+
while (index >= 0) {
|
|
79
|
+
const start = index + delimiter.length
|
|
80
|
+
if (body.subarray(start, start + 2).toString() === '--') break
|
|
81
|
+
const headerEnd = body.indexOf('\r\n\r\n', start)
|
|
82
|
+
if (headerEnd < 0) break
|
|
83
|
+
const rawHeaders = body.subarray(start, headerEnd).toString('utf8')
|
|
84
|
+
const next = body.indexOf(delimiter, headerEnd)
|
|
85
|
+
if (next < 0) break
|
|
86
|
+
const content = body.subarray(headerEnd + 4, next - 2)
|
|
87
|
+
const name = /name="([^"]*)"/.exec(rawHeaders)?.[1] ?? ''
|
|
88
|
+
const filename = /filename="([^"]*)"/.exec(rawHeaders)?.[1]
|
|
89
|
+
const contentType = /content-type:\s*([^\r\n]+)/i.exec(rawHeaders)?.[1]?.trim()
|
|
90
|
+
parts.push({
|
|
91
|
+
name,
|
|
92
|
+
data: content,
|
|
93
|
+
...(filename ? { filename } : {}),
|
|
94
|
+
...(contentType ? { contentType } : {}),
|
|
95
|
+
})
|
|
96
|
+
index = next
|
|
97
|
+
}
|
|
98
|
+
return parts
|
|
99
|
+
}
|
|
100
|
+
|
|
101
|
+
function taskSummary(task: ConvertTask): Record<string, unknown> {
|
|
102
|
+
return {
|
|
103
|
+
id: task.id,
|
|
104
|
+
state: task.state,
|
|
105
|
+
inputKind: task.inputKind,
|
|
106
|
+
label: task.label,
|
|
107
|
+
sourceFormat: task.sourceFormat,
|
|
108
|
+
createdAt: task.createdAt,
|
|
109
|
+
updatedAt: task.updatedAt,
|
|
110
|
+
startedAt: task.startedAt,
|
|
111
|
+
finishedAt: task.finishedAt,
|
|
112
|
+
progress: task.progress,
|
|
113
|
+
stage: task.stage,
|
|
114
|
+
message: task.message,
|
|
115
|
+
error: task.error,
|
|
116
|
+
warnings: task.warnings,
|
|
117
|
+
attempts: task.attempts,
|
|
118
|
+
deliveryDir: task.deliveryDir,
|
|
119
|
+
assetCount: task.manifest?.assets.length ?? 0,
|
|
120
|
+
wordCount: task.manifest?.wordCount ?? 0,
|
|
121
|
+
parser: task.manifest?.parser,
|
|
122
|
+
contentHash: task.manifest?.contentHash,
|
|
123
|
+
externalId: task.manifest?.externalId,
|
|
124
|
+
title: task.manifest?.title ?? task.label,
|
|
125
|
+
}
|
|
126
|
+
}
|
|
127
|
+
|
|
128
|
+
/** Full task view, including the ingest payload the DSH layer forwards. */
|
|
129
|
+
function taskDetail(task: ConvertTask): Record<string, unknown> {
|
|
130
|
+
return {
|
|
131
|
+
...taskSummary(task),
|
|
132
|
+
manifest: task.manifest ?? null,
|
|
133
|
+
ingest: task.manifest?.ingest ?? null,
|
|
134
|
+
request: { ...task.request, text: task.request.text ? `${task.request.text.slice(0, 400)}…` : undefined },
|
|
135
|
+
}
|
|
136
|
+
}
|
|
137
|
+
|
|
138
|
+
function requestFromQuery(url: URL): ConvertRequest {
|
|
139
|
+
const kind = (url.searchParams.get('kind') ?? 'file') as ConvertRequest['kind']
|
|
140
|
+
const request: ConvertRequest = { kind }
|
|
141
|
+
const set = (key: keyof ConvertRequest, value: string | null): void => {
|
|
142
|
+
if (value === null || value === '') return
|
|
143
|
+
;(request as unknown as Record<string, unknown>)[key] = value
|
|
144
|
+
}
|
|
145
|
+
set('title', url.searchParams.get('title'))
|
|
146
|
+
set('sourceType', url.searchParams.get('source_type'))
|
|
147
|
+
set('externalId', url.searchParams.get('external_id'))
|
|
148
|
+
set('adapter', url.searchParams.get('adapter'))
|
|
149
|
+
return request
|
|
150
|
+
}
|
|
151
|
+
|
|
152
|
+
export function createRequestHandler(options: HandlerOptions): (req: IncomingMessage, res: ServerResponse) => Promise<void> {
|
|
153
|
+
const { service } = options
|
|
154
|
+
const basePath = (options.basePath ?? '').replace(/\/+$/, '')
|
|
155
|
+
const webRoot = options.webRoot ?? defaultWebRoot()
|
|
156
|
+
|
|
157
|
+
const serveStatic = async (res: ServerResponse, relative: string): Promise<boolean> => {
|
|
158
|
+
const target = resolve(webRoot, normalize(relative).replace(/^([/\\])+/, ''))
|
|
159
|
+
if (target !== webRoot && !target.startsWith(webRoot + sep)) return false
|
|
160
|
+
try {
|
|
161
|
+
let data = await readFile(target)
|
|
162
|
+
// The console is served under an arbitrary prefix (standalone at `/`,
|
|
163
|
+
// inside DSH at `/convert`), so asset URLs are resolved at serve time.
|
|
164
|
+
if (relative === 'index.html') {
|
|
165
|
+
data = Buffer.from(data.toString('utf8').replaceAll('__CONVERT_BASE__', basePath), 'utf8')
|
|
166
|
+
}
|
|
167
|
+
res.statusCode = 200
|
|
168
|
+
res.setHeader('content-type', contentTypeOf(target) === 'application/octet-stream' ? guessWebType(target) : contentTypeOf(target))
|
|
169
|
+
res.setHeader('cache-control', 'no-cache')
|
|
170
|
+
res.end(data)
|
|
171
|
+
return true
|
|
172
|
+
} catch {
|
|
173
|
+
return false
|
|
174
|
+
}
|
|
175
|
+
}
|
|
176
|
+
|
|
177
|
+
return async function handle(req: IncomingMessage, res: ServerResponse): Promise<void> {
|
|
178
|
+
const url = new URL(req.url ?? '/', 'http://127.0.0.1')
|
|
179
|
+
let pathname = url.pathname
|
|
180
|
+
if (basePath) {
|
|
181
|
+
if (pathname === basePath) pathname = '/'
|
|
182
|
+
else if (pathname.startsWith(`${basePath}/`)) pathname = pathname.slice(basePath.length)
|
|
183
|
+
else {
|
|
184
|
+
sendJson(res, 404, { ok: false, error: { code: 'NOT_FOUND', message: `未挂载在 ${basePath} 下` } })
|
|
185
|
+
return
|
|
186
|
+
}
|
|
187
|
+
}
|
|
188
|
+
|
|
189
|
+
try {
|
|
190
|
+
if (options.apiKey && pathname.startsWith('/api/') && req.headers['x-api-key'] !== options.apiKey) {
|
|
191
|
+
sendJson(res, 401, { ok: false, error: { code: 'UNAUTHORIZED', message: '缺少或错误的 X-Api-Key。' } })
|
|
192
|
+
return
|
|
193
|
+
}
|
|
194
|
+
|
|
195
|
+
// ---- console ---------------------------------------------------------
|
|
196
|
+
if (req.method === 'GET' && (pathname === '/' || pathname === '/index.html')) {
|
|
197
|
+
if (await serveStatic(res, 'index.html')) return
|
|
198
|
+
sendJson(res, 500, { ok: false, error: { code: 'CONSOLE_MISSING', message: `未找到内核控制台页面(${webRoot})。请先执行 pnpm build。` } })
|
|
199
|
+
return
|
|
200
|
+
}
|
|
201
|
+
if (req.method === 'GET' && /^\/(app\.js|styles\.css|vendor\/[\w.-]+)$/.test(pathname)) {
|
|
202
|
+
if (await serveStatic(res, pathname.slice(1))) return
|
|
203
|
+
sendJson(res, 404, { ok: false, error: { code: 'NOT_FOUND', message: pathname } })
|
|
204
|
+
return
|
|
205
|
+
}
|
|
206
|
+
|
|
207
|
+
// ---- health & capabilities -------------------------------------------
|
|
208
|
+
if (req.method === 'GET' && pathname === '/api/health') {
|
|
209
|
+
sendJson(res, 200, ok({
|
|
210
|
+
service: 'dsh-convert-core',
|
|
211
|
+
version: '0.1.0',
|
|
212
|
+
rootDir: service.rootDir,
|
|
213
|
+
outDir: service.outDir,
|
|
214
|
+
tasks: service.list().length,
|
|
215
|
+
}))
|
|
216
|
+
return
|
|
217
|
+
}
|
|
218
|
+
if (req.method === 'GET' && pathname === '/api/capabilities') {
|
|
219
|
+
sendJson(res, 200, ok(await service.capabilities()))
|
|
220
|
+
return
|
|
221
|
+
}
|
|
222
|
+
|
|
223
|
+
// ---- server-sent events ----------------------------------------------
|
|
224
|
+
if (req.method === 'GET' && pathname === '/api/events') {
|
|
225
|
+
res.statusCode = 200
|
|
226
|
+
res.setHeader('content-type', 'text/event-stream; charset=utf-8')
|
|
227
|
+
res.setHeader('cache-control', 'no-cache')
|
|
228
|
+
res.setHeader('connection', 'keep-alive')
|
|
229
|
+
res.write(`event: hello\ndata: ${JSON.stringify({ tasks: service.list().map(taskSummary) })}\n\n`)
|
|
230
|
+
const unsubscribe = service.subscribe(task => {
|
|
231
|
+
res.write(`event: task\ndata: ${JSON.stringify(taskSummary(task))}\n\n`)
|
|
232
|
+
})
|
|
233
|
+
const heartbeat = setInterval(() => { res.write(': ping\n\n') }, 20_000)
|
|
234
|
+
req.on('close', () => {
|
|
235
|
+
clearInterval(heartbeat)
|
|
236
|
+
unsubscribe()
|
|
237
|
+
})
|
|
238
|
+
return
|
|
239
|
+
}
|
|
240
|
+
|
|
241
|
+
// ---- tasks ------------------------------------------------------------
|
|
242
|
+
if (pathname === '/api/tasks' && req.method === 'GET') {
|
|
243
|
+
sendJson(res, 200, ok({ tasks: service.list().map(taskSummary) }))
|
|
244
|
+
return
|
|
245
|
+
}
|
|
246
|
+
|
|
247
|
+
if (pathname === '/api/tasks' && req.method === 'POST') {
|
|
248
|
+
const task = await createTask(service, req, url)
|
|
249
|
+
sendJson(res, 202, ok(taskDetail(task)))
|
|
250
|
+
return
|
|
251
|
+
}
|
|
252
|
+
|
|
253
|
+
const taskMatch = /^\/api\/tasks\/([^/]+)(\/.*)?$/.exec(pathname)
|
|
254
|
+
if (taskMatch) {
|
|
255
|
+
const id = decodeURIComponent(taskMatch[1]!)
|
|
256
|
+
const rest = taskMatch[2] ?? ''
|
|
257
|
+
if (rest === '' && req.method === 'GET') {
|
|
258
|
+
const task = service.get(id)
|
|
259
|
+
if (!task) throw new AdapterError('TASK_NOT_FOUND', `任务不存在:${id}`)
|
|
260
|
+
sendJson(res, 200, ok(taskDetail(task)))
|
|
261
|
+
return
|
|
262
|
+
}
|
|
263
|
+
if (rest === '/cancel' && req.method === 'POST') {
|
|
264
|
+
sendJson(res, 200, ok(taskSummary(service.cancel(id))))
|
|
265
|
+
return
|
|
266
|
+
}
|
|
267
|
+
if (rest === '/retry' && req.method === 'POST') {
|
|
268
|
+
sendJson(res, 200, ok(taskSummary(await service.retry(id))))
|
|
269
|
+
return
|
|
270
|
+
}
|
|
271
|
+
if (rest === '/ingest' && req.method === 'GET') {
|
|
272
|
+
sendJson(res, 200, ok(await service.ingestPlan(id)))
|
|
273
|
+
return
|
|
274
|
+
}
|
|
275
|
+
if (rest === '/markdown' && req.method === 'GET') {
|
|
276
|
+
const file = await service.readDeliveryFile(id, 'index.md')
|
|
277
|
+
if (!file) throw new AdapterError('ARTIFACT_NOT_FOUND', '任务没有 index.md(可能未成功或已被清理)。')
|
|
278
|
+
res.statusCode = 200
|
|
279
|
+
res.setHeader('content-type', 'text/markdown; charset=utf-8')
|
|
280
|
+
res.end(file.data)
|
|
281
|
+
return
|
|
282
|
+
}
|
|
283
|
+
if (rest.startsWith('/files/') && req.method === 'GET') {
|
|
284
|
+
const relative = decodeURIComponent(rest.slice('/files/'.length))
|
|
285
|
+
const file = await service.readDeliveryFile(id, relative)
|
|
286
|
+
if (!file) throw new AdapterError('ARTIFACT_NOT_FOUND', `交付目录里没有该文件:${relative}`)
|
|
287
|
+
res.statusCode = 200
|
|
288
|
+
res.setHeader('content-type', file.contentType)
|
|
289
|
+
res.setHeader('cache-control', 'no-store')
|
|
290
|
+
res.end(file.data)
|
|
291
|
+
return
|
|
292
|
+
}
|
|
293
|
+
sendJson(res, 405, { ok: false, error: { code: 'METHOD_NOT_ALLOWED', message: `${req.method} ${pathname}` } })
|
|
294
|
+
return
|
|
295
|
+
}
|
|
296
|
+
|
|
297
|
+
sendJson(res, 404, { ok: false, error: { code: 'NOT_FOUND', message: pathname } })
|
|
298
|
+
} catch (cause) {
|
|
299
|
+
const error = cause instanceof Error ? cause : new Error(String(cause))
|
|
300
|
+
const status = error instanceof AdapterError
|
|
301
|
+
? error.code === 'TASK_NOT_FOUND' || error.code === 'ARTIFACT_NOT_FOUND' ? 404
|
|
302
|
+
: error.code === 'REQUEST_TOO_LARGE' || error.code === 'INPUT_TOO_LARGE' ? 413
|
|
303
|
+
: error.code === 'REQUEST_INVALID' || error.code === 'CONTENT_REQUIRED' || error.code === 'URL_REQUIRED' ? 400
|
|
304
|
+
: 409
|
|
305
|
+
: 500
|
|
306
|
+
sendError(res, status, error)
|
|
307
|
+
}
|
|
308
|
+
}
|
|
309
|
+
}
|
|
310
|
+
|
|
311
|
+
async function createTask(service: ConvertService, req: IncomingMessage, url: URL): Promise<ConvertTask> {
|
|
312
|
+
// Keep the raw header: a multipart boundary is case-sensitive, so lowercasing
|
|
313
|
+
// the whole content type silently breaks every upload.
|
|
314
|
+
const rawContentType = req.headers['content-type'] ?? ''
|
|
315
|
+
const contentType = rawContentType.toLowerCase()
|
|
316
|
+
|
|
317
|
+
if (contentType.startsWith('application/json')) {
|
|
318
|
+
const body = JSON.parse((await readBody(req)).toString('utf8')) as Record<string, unknown>
|
|
319
|
+
const request: ConvertRequest = {
|
|
320
|
+
kind: (body.kind as ConvertRequest['kind']) ?? 'text',
|
|
321
|
+
...(typeof body.text === 'string' ? { text: body.text } : {}),
|
|
322
|
+
...(typeof body.title === 'string' ? { title: body.title } : {}),
|
|
323
|
+
...(typeof body.source_type === 'string' ? { sourceType: body.source_type } : {}),
|
|
324
|
+
...(typeof body.adapter === 'string' ? { adapter: body.adapter } : {}),
|
|
325
|
+
...(typeof body.local_path === 'string' && body.local_path.trim() ? { localPath: body.local_path.trim() } : {}),
|
|
326
|
+
...(typeof body.external_id === 'string' ? { externalId: body.external_id } : {}),
|
|
327
|
+
...(Array.isArray(body.source_type_candidates) ? { sourceTypeCandidates: body.source_type_candidates.map(String) } : {}),
|
|
328
|
+
...(body.lineage && typeof body.lineage === 'object' ? { lineage: body.lineage as ConvertRequest['lineage'] } : {}),
|
|
329
|
+
}
|
|
330
|
+
if (typeof body.local_path === 'string' && body.local_path.trim()) {
|
|
331
|
+
return await service.submit({ ...request, kind: 'file' }, {
|
|
332
|
+
...(typeof body.filename === 'string' ? { filename: body.filename } : {}),
|
|
333
|
+
})
|
|
334
|
+
}
|
|
335
|
+
if (typeof body.data === 'string' && body.data.length > 0) {
|
|
336
|
+
const data = Buffer.from(body.data, 'base64')
|
|
337
|
+
return await service.submit({ ...request, kind: request.kind ?? 'file' }, { data, ...(typeof body.filename === 'string' ? { filename: body.filename } : {}) })
|
|
338
|
+
}
|
|
339
|
+
return await service.submit(request)
|
|
340
|
+
}
|
|
341
|
+
|
|
342
|
+
const request = requestFromQuery(url)
|
|
343
|
+
|
|
344
|
+
if (contentType.startsWith('multipart/form-data')) {
|
|
345
|
+
const boundaryMatch = /boundary=(?:"([^"]+)"|([^;]+))/i.exec(rawContentType)
|
|
346
|
+
const boundary = (boundaryMatch?.[1] ?? boundaryMatch?.[2])?.trim()
|
|
347
|
+
if (!boundary) throw new AdapterError('REQUEST_INVALID', 'multipart 请求缺少 boundary。')
|
|
348
|
+
const parts = parseMultipart(await readBody(req), boundary)
|
|
349
|
+
const file = parts.find(part => part.filename)
|
|
350
|
+
for (const part of parts) {
|
|
351
|
+
if (part.filename) continue
|
|
352
|
+
const value = part.data.toString('utf8')
|
|
353
|
+
if (part.name === 'kind') request.kind = value as ConvertRequest['kind']
|
|
354
|
+
else if (part.name === 'title') request.title = value
|
|
355
|
+
else if (part.name === 'source_type') request.sourceType = value
|
|
356
|
+
else if (part.name === 'adapter') request.adapter = value
|
|
357
|
+
else if (part.name === 'text') request.text = value
|
|
358
|
+
else if (part.name === 'external_id') request.externalId = value
|
|
359
|
+
}
|
|
360
|
+
if (!file) {
|
|
361
|
+
if (request.kind === 'text' && request.text) return await service.submit(request)
|
|
362
|
+
throw new AdapterError('INPUT_REQUIRED', 'multipart 请求没有文件部件。')
|
|
363
|
+
}
|
|
364
|
+
return await service.submit(request, { data: file.data, filename: file.filename ?? 'upload.bin' })
|
|
365
|
+
}
|
|
366
|
+
|
|
367
|
+
// Raw body upload: the whole request body is the file.
|
|
368
|
+
const filename = url.searchParams.get('filename') ?? dispositionFilename(req.headers['content-disposition']) ?? 'upload.bin'
|
|
369
|
+
const data = await readBody(req)
|
|
370
|
+
if (data.length === 0) {
|
|
371
|
+
if (request.kind === 'text') return await service.submit(request)
|
|
372
|
+
throw new AdapterError('INPUT_REQUIRED', '请求体为空。')
|
|
373
|
+
}
|
|
374
|
+
return await service.submit({ ...request, kind: 'file' }, { data, filename })
|
|
375
|
+
}
|
|
376
|
+
|
|
377
|
+
function dispositionFilename(header: string | undefined): string | undefined {
|
|
378
|
+
if (!header) return undefined
|
|
379
|
+
const match = /filename\*?=(?:UTF-8'')?"?([^";]+)"?/i.exec(header)
|
|
380
|
+
return match ? decodeURIComponent(match[1]!) : undefined
|
|
381
|
+
}
|
|
382
|
+
|
|
383
|
+
function guessWebType(path: string): string {
|
|
384
|
+
switch (extname(path).toLowerCase()) {
|
|
385
|
+
case '.html': return 'text/html; charset=utf-8'
|
|
386
|
+
case '.js': return 'text/javascript; charset=utf-8'
|
|
387
|
+
case '.css': return 'text/css; charset=utf-8'
|
|
388
|
+
case '.svg': return 'image/svg+xml'
|
|
389
|
+
case '.json': return 'application/json; charset=utf-8'
|
|
390
|
+
default: return 'application/octet-stream'
|
|
391
|
+
}
|
|
392
|
+
}
|
|
393
|
+
|
|
394
|
+
export { taskSummary, taskDetail }
|
package/src/markdown.ts
ADDED
|
@@ -0,0 +1,141 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Markdown / front matter helpers.
|
|
3
|
+
*
|
|
4
|
+
* The serialization below is byte-compatible with the reader inside
|
|
5
|
+
* `plugins/dsh-data-domain-knowledge/src/runtime.ts` (`parseMarkdown`): one
|
|
6
|
+
* `key: value` line per field, strings raw, everything else JSON-encoded, and
|
|
7
|
+
* `yamlValue()` on the reading side un-quotes plain strings and JSON-parses
|
|
8
|
+
* `{...}` / `[...]` shapes.
|
|
9
|
+
*/
|
|
10
|
+
|
|
11
|
+
import { createHash } from 'node:crypto'
|
|
12
|
+
import { mkdir, rename, unlink, writeFile } from 'node:fs/promises'
|
|
13
|
+
import { dirname } from 'node:path'
|
|
14
|
+
|
|
15
|
+
/** Same shape the knowledge plugin accepts for identifiers. */
|
|
16
|
+
export const SAFE_ID = /^[a-zA-Z0-9._:-]{1,160}$/
|
|
17
|
+
|
|
18
|
+
export function sha256(value: string | Buffer): string {
|
|
19
|
+
return createHash('sha256').update(value).digest('hex')
|
|
20
|
+
}
|
|
21
|
+
|
|
22
|
+
export function sha1(value: string | Buffer): string {
|
|
23
|
+
return createHash('sha1').update(value).digest('hex')
|
|
24
|
+
}
|
|
25
|
+
|
|
26
|
+
export function shortHash(value: string | Buffer, length = 24): string {
|
|
27
|
+
return sha256(value).slice(0, length)
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
/** Mirrors the knowledge plugin slugger so paths look the same on both sides. */
|
|
31
|
+
export function slug(value: string, fallback = 'item'): string {
|
|
32
|
+
const normalized = value
|
|
33
|
+
.trim()
|
|
34
|
+
.replace(/[^\p{L}\p{N}._-]+/gu, '-')
|
|
35
|
+
.replace(/^-+|-+$/g, '')
|
|
36
|
+
return normalized.slice(0, 120) || `${fallback}-${Date.now().toString(36)}`
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
export function serializeFrontMatter(frontmatter: Record<string, unknown>): string {
|
|
40
|
+
const lines: string[] = []
|
|
41
|
+
for (const [key, value] of Object.entries(frontmatter)) {
|
|
42
|
+
if (value === undefined || value === null) continue
|
|
43
|
+
if (Array.isArray(value) && value.length === 0) continue
|
|
44
|
+
if (typeof value === 'object' && !Array.isArray(value) && Object.keys(value).length === 0) continue
|
|
45
|
+
// A raw newline would break the line-based reader on the knowledge side.
|
|
46
|
+
const encoded = typeof value === 'string' ? value.replace(/\r?\n/g, ' ') : JSON.stringify(value)
|
|
47
|
+
if (encoded.length === 0) continue
|
|
48
|
+
lines.push(`${key}: ${encoded}`)
|
|
49
|
+
}
|
|
50
|
+
return `---\n${lines.join('\n')}\n---`
|
|
51
|
+
}
|
|
52
|
+
|
|
53
|
+
export function toMarkdownDocument(frontmatter: Record<string, unknown>, content: string): string {
|
|
54
|
+
return `${serializeFrontMatter(frontmatter)}\n\n${content.trim()}\n`
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
export function parseMarkdownDocument(source: string): { frontmatter: Record<string, unknown>; content: string } {
|
|
58
|
+
if (!source.startsWith('---')) return { frontmatter: {}, content: source.trim() }
|
|
59
|
+
const end = source.indexOf('\n---', 3)
|
|
60
|
+
if (end < 0) return { frontmatter: {}, content: source.trim() }
|
|
61
|
+
const header = source.slice(3, end).replace(/^\n/, '')
|
|
62
|
+
const frontmatter: Record<string, unknown> = {}
|
|
63
|
+
for (const line of header.split(/\r?\n/)) {
|
|
64
|
+
const separator = line.indexOf(':')
|
|
65
|
+
if (separator <= 0) continue
|
|
66
|
+
const key = line.slice(0, separator).trim()
|
|
67
|
+
frontmatter[key] = yamlValue(line.slice(separator + 1))
|
|
68
|
+
}
|
|
69
|
+
return { frontmatter, content: source.slice(end + 4).trim() }
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
function yamlValue(value: string): unknown {
|
|
73
|
+
const text = value.trim()
|
|
74
|
+
if (text === 'true') return true
|
|
75
|
+
if (text === 'false') return false
|
|
76
|
+
if (text === 'null') return null
|
|
77
|
+
if (/^-?\d+(\.\d+)?$/.test(text)) return Number(text)
|
|
78
|
+
if ((text.startsWith('[') && text.endsWith(']')) || (text.startsWith('{') && text.endsWith('}'))) {
|
|
79
|
+
try {
|
|
80
|
+
return JSON.parse(text)
|
|
81
|
+
} catch {
|
|
82
|
+
/* keep the raw string */
|
|
83
|
+
}
|
|
84
|
+
}
|
|
85
|
+
return text.replace(/^['"]|['"]$/g, '')
|
|
86
|
+
}
|
|
87
|
+
|
|
88
|
+
/**
|
|
89
|
+
* Local image references in a Markdown body, resolved against `baseDir`.
|
|
90
|
+
* Same regex and same skip rules as the knowledge plugin, so the kernel can
|
|
91
|
+
* predict exactly what will reach the multimodal embedder.
|
|
92
|
+
*/
|
|
93
|
+
export function localImagePaths(content: string, baseDir: string): string[] {
|
|
94
|
+
const pattern = /!\[[^\]]*\]\(\s*(?:<([^>]+)>|([^\s)]+))/g
|
|
95
|
+
const found: string[] = []
|
|
96
|
+
for (const match of content.matchAll(pattern)) {
|
|
97
|
+
const raw = (match[1] ?? match[2] ?? '').trim()
|
|
98
|
+
if (!raw || raw.startsWith('data:') || /^https?:\/\//i.test(raw) || raw.startsWith('#')) continue
|
|
99
|
+
found.push(raw)
|
|
100
|
+
}
|
|
101
|
+
return [...new Set(found)]
|
|
102
|
+
}
|
|
103
|
+
|
|
104
|
+
export function wordCount(value: string): number {
|
|
105
|
+
const cjk = (value.match(/[\u3400-\u4dbf\u4e00-\u9fff\uf900-\ufaff]/g) ?? []).length
|
|
106
|
+
const latin = value
|
|
107
|
+
.replace(/[\u3400-\u4dbf\u4e00-\u9fff\uf900-\ufaff]/g, ' ')
|
|
108
|
+
.split(/\s+/)
|
|
109
|
+
.filter(Boolean).length
|
|
110
|
+
return cjk + latin
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
export async function writeTextAtomic(path: string, content: string): Promise<void> {
|
|
114
|
+
await mkdir(dirname(path), { recursive: true })
|
|
115
|
+
const temporary = `${path}.${process.pid}.${Date.now().toString(36)}.tmp`
|
|
116
|
+
try {
|
|
117
|
+
await writeFile(temporary, content, 'utf8')
|
|
118
|
+
await rename(temporary, path)
|
|
119
|
+
} finally {
|
|
120
|
+
try {
|
|
121
|
+
await unlink(temporary)
|
|
122
|
+
} catch {
|
|
123
|
+
/* rename consumed it */
|
|
124
|
+
}
|
|
125
|
+
}
|
|
126
|
+
}
|
|
127
|
+
|
|
128
|
+
export async function writeBufferAtomic(path: string, content: Buffer): Promise<void> {
|
|
129
|
+
await mkdir(dirname(path), { recursive: true })
|
|
130
|
+
const temporary = `${path}.${process.pid}.${Date.now().toString(36)}.tmp`
|
|
131
|
+
try {
|
|
132
|
+
await writeFile(temporary, content)
|
|
133
|
+
await rename(temporary, path)
|
|
134
|
+
} finally {
|
|
135
|
+
try {
|
|
136
|
+
await unlink(temporary)
|
|
137
|
+
} catch {
|
|
138
|
+
/* rename consumed it */
|
|
139
|
+
}
|
|
140
|
+
}
|
|
141
|
+
}
|
|
@@ -0,0 +1,67 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* 外部二进制探测。
|
|
3
|
+
*
|
|
4
|
+
* 这个包与文档类转换器的根本差别就在这里:它需要 `ffmpeg`/`ffprobe`,转写档还需要
|
|
5
|
+
* `uvx`(用来按需拉起 faster-whisper / RapidOCR)。探测结果既用于 `probe()`(宿主上报能力),
|
|
6
|
+
* 也用于错误信息里给出**可操作**的安装命令 —— 缺依赖要吵,但不能静默降级。
|
|
7
|
+
*/
|
|
8
|
+
|
|
9
|
+
import { execFile } from 'node:child_process'
|
|
10
|
+
import { promisify } from 'node:util'
|
|
11
|
+
|
|
12
|
+
const execFileAsync = promisify(execFile)
|
|
13
|
+
|
|
14
|
+
export interface ToolVersions {
|
|
15
|
+
ffmpeg?: string
|
|
16
|
+
ffprobe?: string
|
|
17
|
+
uvx?: string
|
|
18
|
+
}
|
|
19
|
+
|
|
20
|
+
export const FFMPEG_HINT = '装 ffmpeg(含 ffprobe):macOS `brew install ffmpeg`,Debian/Ubuntu `apt install ffmpeg`。'
|
|
21
|
+
export const UVX_HINT = '安装 uv(内含 uvx):`curl -LsSf https://astral.sh/uv/install.sh | sh`;或用 tools.uvxPath 指向已有的 uvx。'
|
|
22
|
+
|
|
23
|
+
export interface ProbeOptions {
|
|
24
|
+
uvxPath?: string
|
|
25
|
+
}
|
|
26
|
+
|
|
27
|
+
/** 每次都真跑一次 `-version`;调用方(宿主)自己缓存,别在这里藏状态。 */
|
|
28
|
+
export async function probeTools(options: ProbeOptions = {}): Promise<ToolVersions> {
|
|
29
|
+
const versions: ToolVersions = {}
|
|
30
|
+
try {
|
|
31
|
+
const { stdout } = await execFileAsync('ffprobe', ['-version'])
|
|
32
|
+
const found = /ffprobe version (\S+)/.exec(stdout)?.[1]
|
|
33
|
+
if (found) versions.ffprobe = found
|
|
34
|
+
} catch {
|
|
35
|
+
/* 缺失由调用方汇报 */
|
|
36
|
+
}
|
|
37
|
+
try {
|
|
38
|
+
const { stdout } = await execFileAsync('ffmpeg', ['-version'])
|
|
39
|
+
const found = /ffmpeg version (\S+)/.exec(stdout)?.[1]
|
|
40
|
+
if (found) versions.ffmpeg = found
|
|
41
|
+
} catch {
|
|
42
|
+
/* 缺失由调用方汇报 */
|
|
43
|
+
}
|
|
44
|
+
try {
|
|
45
|
+
const uvx = options.uvxPath?.trim() || process.env.MEDIA_TO_MD_UVX?.trim() || 'uvx'
|
|
46
|
+
const { stdout } = await execFileAsync(uvx, ['--version'])
|
|
47
|
+
// `uvx --version` 形如 `uvx 0.11.6 (65950801c 2026-04-09 aarch64-apple-darwin)`,只取版本号。
|
|
48
|
+
const found = /(\d+\.\d+\.\d+)/.exec(stdout)?.[1]
|
|
49
|
+
if (found) versions.uvx = found
|
|
50
|
+
} catch {
|
|
51
|
+
/* 可选依赖:只有转写/OCR 档才要求 */
|
|
52
|
+
}
|
|
53
|
+
return versions
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
export async function runFfmpeg(args: string[], options: { allowFailure?: boolean } = {}): Promise<{ stdout: string; stderr: string }> {
|
|
57
|
+
try {
|
|
58
|
+
const { stdout, stderr } = await execFileAsync('ffmpeg', args, { maxBuffer: 32 * 1024 * 1024 })
|
|
59
|
+
return { stdout, stderr }
|
|
60
|
+
} catch (cause) {
|
|
61
|
+
if (options.allowFailure) {
|
|
62
|
+
const error = cause as { stdout?: string; stderr?: string }
|
|
63
|
+
return { stdout: error.stdout ?? '', stderr: error.stderr ?? '' }
|
|
64
|
+
}
|
|
65
|
+
throw cause
|
|
66
|
+
}
|
|
67
|
+
}
|