dsh-skill-importer 0.2.0 → 0.2.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/server.ts ADDED
@@ -0,0 +1,603 @@
1
+ /**
2
+ * Host-side skill importer logic: direct filesystem writes, skill-root
3
+ * scanning, and URL fetching. Pure Node — no Cordis imports — so the route
4
+ * handlers stay unit-testable and the plugin body is only wiring.
5
+ *
6
+ * This is the "direct write" path: the host process owns the filesystem
7
+ * (no agent sandbox, no approval), so an import lands the file immediately
8
+ * and the skill-filesystem watcher discovers it in place.
9
+ */
10
+
11
+ import { cpSync, existsSync, lstatSync, mkdirSync, readFileSync, readdirSync, realpathSync, renameSync, rmSync, statSync, writeFileSync } from 'node:fs'
12
+ import { createHash, randomUUID } from 'node:crypto'
13
+ import { lookup } from 'node:dns/promises'
14
+ import { isIP } from 'node:net'
15
+ import { homedir } from 'node:os'
16
+ import { basename, dirname, extname, isAbsolute, join, relative } from 'node:path'
17
+ import type { IncomingMessage, ServerResponse } from 'node:http'
18
+ import { isValidSkillName, parseSkillFile } from './frontmatter.ts'
19
+ import type {
20
+ BatchCommitEntry, BatchScanEntry, BatchScanRequest,
21
+ ImportRequest, ImportTarget, ImportUrlRequest, SkillListEntry,
22
+ } from './types.ts'
23
+
24
+ /** Hard cap for one imported skill body (matches the client preview limit). */
25
+ export const MAX_CONTENT_BYTES = 256 * 1024
26
+
27
+ /** Cap for one request body read (JSON overhead above the content cap). */
28
+ export const MAX_BODY_BYTES = 1024 * 1024
29
+
30
+ /** Batch safety bounds: resources are copied, but one selection stays finite. */
31
+ export const MAX_BATCH_SKILLS = 200
32
+ export const MAX_BATCH_FILES_PER_SKILL = 2_000
33
+ export const MAX_BATCH_SKILL_BYTES = 10 * 1024 * 1024
34
+ export const BATCH_SCAN_TTL_MS = 10 * 60 * 1000
35
+
36
+ const KEBAB_CASE = /^[a-z0-9]+(?:-[a-z0-9]+)*$/
37
+ const MAX_URL_REDIRECTS = 5
38
+
39
+ /** The harness home (`$DSH_HOME`, defaulting to `~/.dsh`). */
40
+ export function dshHomeDir(): string {
41
+ return process.env.DSH_HOME ?? join(homedir(), '.dsh')
42
+ }
43
+
44
+ /** Absolute skill root for one target under one workspace. */
45
+ export function skillRoot(target: ImportTarget, workspacePath: string): string {
46
+ switch (target) {
47
+ case 'user':
48
+ return join(dshHomeDir(), 'skills')
49
+ case 'project-agents':
50
+ return join(workspacePath, '.agents', 'skills')
51
+ }
52
+ }
53
+
54
+ /** Discovery rank per root (lower wins), mirroring dsh-skill-filesystem. */
55
+ const ROOT_RANK: Record<ImportTarget, number> = { 'project-agents': 200, user: 400 }
56
+
57
+ /** Scan order: project roots first so project skills win duplicate names. */
58
+ const SCAN_ORDER: readonly ImportTarget[] = ['project-agents', 'user']
59
+
60
+ /**
61
+ * Quote one frontmatter string value when plain YAML would mis-parse it:
62
+ * a value containing `: ` (or any colon), leading/trailing whitespace, a
63
+ * leading YAML special character, or ` #` would fail or change meaning
64
+ * under the strict YAML parser `dsh-skill-filesystem` uses — such a file is
65
+ * silently skipped by discovery. `JSON.stringify` emits a YAML-compatible
66
+ * double-quoted scalar.
67
+ */
68
+ function yamlScalar(value: string): string {
69
+ if (value.length === 0
70
+ || /^[\s]/.test(value)
71
+ || /[\s]$/.test(value)
72
+ || /:/.test(value)
73
+ || /#/.test(value)
74
+ || /^[!&*{}\[\],|>'"%@`?]/.test(value)) {
75
+ return JSON.stringify(value)
76
+ }
77
+ return value
78
+ }
79
+
80
+ /**
81
+ * Rebuild a skill file's frontmatter in the canonical, strictly-YAML-valid
82
+ * form the harness's provider parses. Unknown keys are dropped (the harness
83
+ * only consumes name/description/whenToUse and the two invocation flags);
84
+ * the body is preserved verbatim (trimmed).
85
+ */
86
+ export function normalizeSkillText(text: string): string {
87
+ const { frontmatter, body } = parseSkillFile(text)
88
+ if (frontmatter.name === undefined || frontmatter.description === undefined) {
89
+ throw new Error('frontmatter 缺少 name 或 description 字段')
90
+ }
91
+ const lines = ['---']
92
+ lines.push(`name: ${frontmatter.name}`)
93
+ lines.push(`description: ${yamlScalar(frontmatter.description)}`)
94
+ if (frontmatter.whenToUse !== undefined) lines.push(`whenToUse: ${yamlScalar(frontmatter.whenToUse)}`)
95
+ if (frontmatter.disableModelInvocation !== undefined) lines.push(`disable-model-invocation: ${frontmatter.disableModelInvocation}`)
96
+ if (frontmatter.userInvocable !== undefined) lines.push(`user-invocable: ${frontmatter.userInvocable}`)
97
+ lines.push('---')
98
+ const normalizedBody = body.trim()
99
+ return lines.join('\n') + (normalizedBody.length > 0 ? `\n\n${normalizedBody}\n` : '\n')
100
+ }
101
+
102
+ /**
103
+ * Write one skill file atomically: `<root>/<name>/SKILL.md`, created via a
104
+ * same-directory temp file plus rename so a crash never leaves a torn file.
105
+ * The content's frontmatter is normalized first so the harness's strict YAML
106
+ * discovery always finds the skill.
107
+ * @param name - kebab-case skill name (validated).
108
+ * @param target - which skill root to write into.
109
+ * @param content - full Markdown text (frontmatter included).
110
+ * @param workspacePath - canonical workspace path; required for project targets.
111
+ * @returns the absolute path of the written file.
112
+ */
113
+ export function writeSkillFile(name: string, target: ImportTarget, content: string, workspacePath?: string): string {
114
+ if (!KEBAB_CASE.test(name)) throw new Error('技能名必须是 kebab-case(小写字母、数字、短横线)')
115
+ if (Buffer.byteLength(content, 'utf8') > MAX_CONTENT_BYTES) throw new Error(`内容超过 ${MAX_CONTENT_BYTES / 1024} KB`)
116
+ if (target !== 'user' && workspacePath === undefined) {
117
+ throw new Error('项目目标需要 workspacePath(当前工作区路径)')
118
+ }
119
+ const normalized = normalizeSkillText(content)
120
+ const directory = join(skillRoot(target, workspacePath ?? ''), name)
121
+ mkdirSync(directory, { recursive: true })
122
+ const file = join(directory, 'SKILL.md')
123
+ const temporary = `${file}.tmp`
124
+ writeFileSync(temporary, normalized, 'utf8')
125
+ renameSync(temporary, file)
126
+ return file
127
+ }
128
+
129
+ /** Scan one skill root and fold its skills into the name-keyed map (rank-aware). */
130
+ function scanRoot(root: string, source: ImportTarget, out: Map<string, SkillListEntry>): void {
131
+ if (!existsSync(root)) return
132
+ for (const entry of readdirSync(root, { withFileTypes: true })) {
133
+ const entryPath = join(root, entry.name)
134
+ let file: string | undefined
135
+ if (entry.isDirectory()) {
136
+ const candidate = join(entryPath, 'SKILL.md')
137
+ if (existsSync(candidate)) file = candidate
138
+ } else if (entry.isFile() && entry.name.endsWith('.md')) {
139
+ file = entryPath
140
+ }
141
+ if (file === undefined) continue
142
+ const { frontmatter } = parseSkillFile(readFileSync(file, 'utf8'))
143
+ if (frontmatter.name === undefined || frontmatter.description === undefined) continue
144
+ if (!isValidSkillName(frontmatter.name)) continue
145
+ const existing = out.get(frontmatter.name)
146
+ if (existing !== undefined && ROOT_RANK[existing.source] <= ROOT_RANK[source]) continue
147
+ out.set(frontmatter.name, {
148
+ name: frontmatter.name,
149
+ description: frontmatter.description,
150
+ ...(frontmatter.whenToUse !== undefined ? { whenToUse: frontmatter.whenToUse } : {}),
151
+ modelInvocable: frontmatter.disableModelInvocation !== true,
152
+ userInvocable: frontmatter.userInvocable !== false,
153
+ source,
154
+ })
155
+ }
156
+ }
157
+
158
+ /**
159
+ * List every installed skill across every registered workspace's project
160
+ * roots plus the user root. No rank deduplication: the management surface
161
+ * shows every location's copy (the framework's own catalog still applies
162
+ * rank at discovery time). Display order groups by source, then name.
163
+ * @param workspacePaths - canonical paths of the registered workspaces.
164
+ */
165
+ export function listSkills(workspacePaths: readonly string[]): SkillListEntry[] {
166
+ const rows: SkillListEntry[] = []
167
+ for (const target of SCAN_ORDER) {
168
+ const out = new Map<string, SkillListEntry>()
169
+ for (const path of target === 'user' ? [dshHomeDir()] : workspacePaths) {
170
+ scanRoot(skillRoot(target, path), target, out)
171
+ }
172
+ rows.push(...out.values())
173
+ }
174
+ return rows.sort((a, b) => a.name.localeCompare(b.name))
175
+ }
176
+
177
+ /**
178
+ * Delete one installed skill (its whole bundle directory `<root>/<name>/`,
179
+ * or the flat `<root>/<name>.md`), scoped to the skill roots only.
180
+ * @param name - kebab-case skill name (validated).
181
+ * @param source - which root the copy lives in.
182
+ * @param workspacePath - canonical workspace path; required for project sources.
183
+ * @returns true when something was removed, false when nothing matched.
184
+ */
185
+ export function deleteSkillFile(name: string, source: ImportTarget, workspacePath?: string): boolean {
186
+ if (!KEBAB_CASE.test(name)) throw new Error('技能名必须是 kebab-case(小写字母、数字、短横线)')
187
+ if (source !== 'user' && workspacePath === undefined) {
188
+ throw new Error('项目来源需要 workspacePath(当前工作区路径)')
189
+ }
190
+ const root = skillRoot(source, workspacePath ?? '')
191
+ const directory = join(root, name)
192
+ if (existsSync(directory)) {
193
+ rmSync(directory, { recursive: true, force: true })
194
+ return true
195
+ }
196
+ const flat = join(root, `${name}.md`)
197
+ if (existsSync(flat)) {
198
+ rmSync(flat)
199
+ return true
200
+ }
201
+ return false
202
+ }
203
+
204
+ /** Host-only source candidate retained between scan and the one-time commit. */
205
+ interface BatchCandidate {
206
+ readonly id: string
207
+ readonly name: string
208
+ readonly description: string
209
+ readonly sourcePath: string
210
+ readonly sourceKind: 'directory' | 'file'
211
+ readonly fingerprint: string
212
+ }
213
+
214
+ /** One short-lived preflight. The HTTP layer owns the map and single-use lifecycle. */
215
+ export interface BatchScanSession {
216
+ readonly scanId: string
217
+ readonly sourcePath: string
218
+ readonly target: ImportTarget
219
+ readonly workspacePath?: string
220
+ readonly expiresAt: number
221
+ readonly entries: readonly BatchScanEntry[]
222
+ readonly candidates: ReadonlyMap<string, BatchCandidate>
223
+ }
224
+
225
+ interface SourceStats {
226
+ readonly fingerprint: string
227
+ readonly fileCount: number
228
+ readonly bytes: number
229
+ }
230
+
231
+ /** Hash names and bytes while refusing symlinks and oversized skill bundles. */
232
+ function sourceStats(sourcePath: string, kind: 'directory' | 'file'): SourceStats {
233
+ const hash = createHash('sha256')
234
+ let fileCount = 0
235
+ let bytes = 0
236
+ const visit = (path: string, relativePath: string): void => {
237
+ const stat = lstatSync(path)
238
+ if (stat.isSymbolicLink()) throw new Error('技能目录不能包含符号链接')
239
+ if (stat.isDirectory()) {
240
+ for (const entry of readdirSync(path, { withFileTypes: true }).sort((a, b) => a.name.localeCompare(b.name))) {
241
+ visit(join(path, entry.name), relativePath.length === 0 ? entry.name : join(relativePath, entry.name))
242
+ }
243
+ return
244
+ }
245
+ if (!stat.isFile()) throw new Error('技能目录包含不支持的文件类型')
246
+ fileCount += 1
247
+ bytes += stat.size
248
+ if (fileCount > MAX_BATCH_FILES_PER_SKILL) throw new Error(`技能文件数量超过 ${MAX_BATCH_FILES_PER_SKILL}`)
249
+ if (bytes > MAX_BATCH_SKILL_BYTES) throw new Error(`技能目录超过 ${MAX_BATCH_SKILL_BYTES / 1024 / 1024} MB`)
250
+ hash.update(relativePath)
251
+ hash.update('\0')
252
+ hash.update(readFileSync(path))
253
+ hash.update('\0')
254
+ }
255
+ visit(sourcePath, kind === 'file' ? basename(sourcePath) : '')
256
+ return { fingerprint: hash.digest('hex'), fileCount, bytes }
257
+ }
258
+
259
+ function destinationExists(name: string, target: ImportTarget, workspacePath?: string): boolean {
260
+ const root = skillRoot(target, workspacePath ?? '')
261
+ return existsSync(join(root, name)) || existsSync(join(root, `${name}.md`))
262
+ }
263
+
264
+ /** Resolve immediate child skill directories and flat Markdown skills. */
265
+ function batchSources(root: string): Array<{ path: string; kind: 'directory' | 'file'; label: string }> {
266
+ const ownSkill = join(root, 'SKILL.md')
267
+ if (existsSync(ownSkill)) return [{ path: root, kind: 'directory', label: basename(root) }]
268
+ const sources: Array<{ path: string; kind: 'directory' | 'file'; label: string }> = []
269
+ for (const entry of readdirSync(root, { withFileTypes: true }).sort((a, b) => a.name.localeCompare(b.name))) {
270
+ const path = join(root, entry.name)
271
+ if (entry.isSymbolicLink()) continue
272
+ if (entry.isDirectory() && existsSync(join(path, 'SKILL.md'))) {
273
+ sources.push({ path, kind: 'directory', label: entry.name })
274
+ } else if (entry.isFile() && ['.md', '.markdown'].includes(extname(entry.name).toLowerCase())) {
275
+ sources.push({ path, kind: 'file', label: basename(entry.name, extname(entry.name)) })
276
+ }
277
+ }
278
+ return sources
279
+ }
280
+
281
+ /** Validate a selected skills root without writing anything. */
282
+ export function scanBatch(request: BatchScanRequest): BatchScanSession {
283
+ if (!isAbsolute(request.sourcePath)) throw new Error('批量导入目录必须是绝对路径')
284
+ if (!existsSync(request.sourcePath) || !statSync(request.sourcePath).isDirectory()) throw new Error('批量导入目录不存在或不可读')
285
+ const sourcePath = realpathSync(request.sourcePath)
286
+ const leaf = basename(sourcePath).toLowerCase()
287
+ const parent = basename(dirname(sourcePath)).toLowerCase()
288
+ const isSkillsRoot = leaf === 'skills'
289
+ const isSkillDirectory = parent === 'skills'
290
+ if (!isSkillsRoot && !isSkillDirectory) {
291
+ throw new Error('批量导入仅支持名为 skills 的目录或其中的单个技能目录')
292
+ }
293
+ if (request.target !== 'user' && request.workspacePath === undefined) throw new Error('项目目标需要 workspacePath(当前工作区路径)')
294
+ const sources = batchSources(sourcePath)
295
+ if (sources.length === 0) throw new Error('所选目录中没有找到可导入的技能')
296
+ if (sources.length > MAX_BATCH_SKILLS) throw new Error(`一次最多扫描 ${MAX_BATCH_SKILLS} 个技能`)
297
+ const entries: BatchScanEntry[] = []
298
+ const candidates = new Map<string, BatchCandidate>()
299
+ const nameRows = new Map<string, number[]>()
300
+
301
+ for (const [index, source] of sources.entries()) {
302
+ const id = String(index + 1)
303
+ const relativePath = relative(sourcePath, source.path) || '.'
304
+ try {
305
+ const skillFile = source.kind === 'directory' ? join(source.path, 'SKILL.md') : source.path
306
+ const text = readFileSync(skillFile, 'utf8')
307
+ if (Buffer.byteLength(text, 'utf8') > MAX_CONTENT_BYTES) throw new Error(`SKILL.md 超过 ${MAX_CONTENT_BYTES / 1024} KB`)
308
+ const { frontmatter } = parseSkillFile(text)
309
+ if (frontmatter.name === undefined) throw new Error('frontmatter 缺少 name 字段')
310
+ if (!isValidSkillName(frontmatter.name)) throw new Error('name 必须是 kebab-case(小写字母、数字、短横线)')
311
+ if (frontmatter.description === undefined || frontmatter.description.trim().length === 0) throw new Error('frontmatter 缺少 description 字段')
312
+ // Reuse the canonical normalizer so scan and single-file import accept the same format.
313
+ normalizeSkillText(text)
314
+ const stats = sourceStats(source.path, source.kind)
315
+ const warnings = source.label === frontmatter.name
316
+ ? undefined
317
+ : [`目录或文件名“${source.label}”与技能名“${frontmatter.name}”不一致`]
318
+ entries.push({
319
+ id,
320
+ name: frontmatter.name,
321
+ description: frontmatter.description,
322
+ relativePath,
323
+ status: 'ready',
324
+ conflict: destinationExists(frontmatter.name, request.target, request.workspacePath),
325
+ ...(warnings === undefined ? {} : { warnings }),
326
+ })
327
+ candidates.set(id, {
328
+ id,
329
+ name: frontmatter.name,
330
+ description: frontmatter.description,
331
+ sourcePath: source.path,
332
+ sourceKind: source.kind,
333
+ fingerprint: stats.fingerprint,
334
+ })
335
+ const at = entries.length - 1
336
+ nameRows.set(frontmatter.name, [...(nameRows.get(frontmatter.name) ?? []), at])
337
+ } catch (error) {
338
+ entries.push({
339
+ id,
340
+ name: source.label,
341
+ relativePath,
342
+ status: 'error',
343
+ error: error instanceof Error ? error.message : String(error),
344
+ conflict: false,
345
+ })
346
+ }
347
+ }
348
+
349
+ // A duplicate inside the selected batch is ambiguous: every copy is refused.
350
+ for (const [name, indexes] of nameRows) {
351
+ if (indexes.length < 2) continue
352
+ for (const index of indexes) {
353
+ const entry = entries[index]
354
+ if (entry === undefined) continue
355
+ entries[index] = { ...entry, status: 'error', conflict: false, error: `批次内存在重复技能名:${name}` }
356
+ candidates.delete(entry.id)
357
+ }
358
+ }
359
+
360
+ return {
361
+ scanId: randomUUID(),
362
+ sourcePath,
363
+ target: request.target,
364
+ ...(request.workspacePath === undefined ? {} : { workspacePath: request.workspacePath }),
365
+ expiresAt: Date.now() + BATCH_SCAN_TTL_MS,
366
+ entries,
367
+ candidates,
368
+ }
369
+ }
370
+
371
+ /** Copy one validated candidate into a same-root staging directory, then atomically swap it in. */
372
+ function installBatchCandidate(candidate: BatchCandidate, session: BatchScanSession, replace: boolean): 'imported' | 'replaced' {
373
+ const fresh = sourceStats(candidate.sourcePath, candidate.sourceKind)
374
+ if (fresh.fingerprint !== candidate.fingerprint) throw new Error('源技能在确认期间发生变化,请重新扫描')
375
+ const root = skillRoot(session.target, session.workspacePath ?? '')
376
+ mkdirSync(root, { recursive: true })
377
+ const finalDirectory = join(root, candidate.name)
378
+ const flatFile = join(root, `${candidate.name}.md`)
379
+ const conflicts = [finalDirectory, flatFile].filter(existsSync)
380
+ if (conflicts.length > 0 && !replace) throw new Error('目标中已存在同名技能')
381
+ const staging = join(root, `.${candidate.name}.import-${randomUUID()}`)
382
+ const backups: Array<{ original: string; backup: string }> = []
383
+ try {
384
+ if (candidate.sourceKind === 'directory') {
385
+ cpSync(candidate.sourcePath, staging, { recursive: true, errorOnExist: true, dereference: false })
386
+ const skillFile = join(staging, 'SKILL.md')
387
+ writeFileSync(skillFile, normalizeSkillText(readFileSync(skillFile, 'utf8')), 'utf8')
388
+ } else {
389
+ mkdirSync(staging, { recursive: false })
390
+ writeFileSync(join(staging, 'SKILL.md'), normalizeSkillText(readFileSync(candidate.sourcePath, 'utf8')), 'utf8')
391
+ }
392
+ for (const original of conflicts) {
393
+ const backup = join(root, `.${basename(original)}.backup-${randomUUID()}`)
394
+ renameSync(original, backup)
395
+ backups.push({ original, backup })
396
+ }
397
+ renameSync(staging, finalDirectory)
398
+ for (const { backup } of backups) rmSync(backup, { recursive: true, force: true })
399
+ return conflicts.length > 0 ? 'replaced' : 'imported'
400
+ } catch (error) {
401
+ rmSync(staging, { recursive: true, force: true })
402
+ if (existsSync(finalDirectory) && backups.length > 0) rmSync(finalDirectory, { recursive: true, force: true })
403
+ for (const { original, backup } of backups.reverse()) {
404
+ if (existsSync(backup)) renameSync(backup, original)
405
+ }
406
+ throw error
407
+ }
408
+ }
409
+
410
+ /** Commit a valid, unexpired scan once; invalid rows are returned as errors and never written. */
411
+ export function commitBatch(session: BatchScanSession, replaceNames: ReadonlySet<string>): BatchCommitEntry[] {
412
+ if (Date.now() > session.expiresAt) throw new Error('批量扫描结果已过期,请重新扫描')
413
+ const results: BatchCommitEntry[] = []
414
+ for (const entry of session.entries) {
415
+ if (entry.status === 'error') {
416
+ results.push({ name: entry.name, status: 'error', message: entry.error })
417
+ continue
418
+ }
419
+ const candidate = session.candidates.get(entry.id)
420
+ if (candidate === undefined) {
421
+ results.push({ name: entry.name, status: 'error', message: '扫描记录不完整,请重新扫描' })
422
+ continue
423
+ }
424
+ const conflict = destinationExists(entry.name, session.target, session.workspacePath)
425
+ if (conflict && !replaceNames.has(entry.name)) {
426
+ results.push({ name: entry.name, status: 'skipped', message: '目标中已存在同名技能,未选择替换' })
427
+ continue
428
+ }
429
+ try {
430
+ results.push({ name: entry.name, status: installBatchCandidate(candidate, session, conflict && replaceNames.has(entry.name)) })
431
+ } catch (error) {
432
+ results.push({ name: entry.name, status: 'error', message: error instanceof Error ? error.message : String(error) })
433
+ }
434
+ }
435
+ return results
436
+ }
437
+
438
+ /** Strip HTML down to a plain-Markdown-ish text body (best effort). */
439
+ function extractHtmlText(html: string): string {
440
+ const withoutBlocks = html
441
+ .replace(/<script[\s\S]*?<\/script>/gi, '')
442
+ .replace(/<style[\s\S]*?<\/style>/gi, '')
443
+ .replace(/<!--[\s\S]*?-->/g, '')
444
+ const title = withoutBlocks.match(/<title[^>]*>([\s\S]*?)<\/title>/i)?.[1]?.trim() ?? ''
445
+ const body = withoutBlocks
446
+ .replace(/<[^>]+>/g, ' ')
447
+ .replace(/&nbsp;/g, ' ')
448
+ .replace(/&amp;/g, '&')
449
+ .replace(/&lt;/g, '<')
450
+ .replace(/&gt;/g, '>')
451
+ .replace(/&quot;/g, '"')
452
+ .replace(/[ \t]{2,}/g, ' ')
453
+ .replace(/\n{3,}/g, '\n\n')
454
+ .trim()
455
+ return title.length > 0 ? `# ${title}\n\n${body}` : body
456
+ }
457
+
458
+ /**
459
+ * Fetch a URL's content for import. Markdown/plain responses pass through
460
+ * verbatim; HTML is roughly extracted to text (the import UI recommends
461
+ * `.md` sources).
462
+ * @param url - the source URL.
463
+ * @returns the text to write as the skill body.
464
+ */
465
+ function privateIpv4(address: string): boolean {
466
+ const octets = address.split('.').map(Number)
467
+ if (octets.length !== 4 || octets.some(value => !Number.isInteger(value) || value < 0 || value > 255)) return true
468
+ const [a = 0, b = 0, c = 0] = octets
469
+ return a === 0 || a === 10 || a === 127 || a >= 224
470
+ || (a === 100 && b >= 64 && b <= 127)
471
+ || (a === 169 && b === 254)
472
+ || (a === 172 && b >= 16 && b <= 31)
473
+ || (a === 192 && (b === 0 || b === 168 || (b === 0 && c === 2)))
474
+ || (a === 198 && (b === 18 || b === 19 || (b === 51 && c === 100)))
475
+ || (a === 203 && b === 0 && c === 113)
476
+ }
477
+
478
+ /** Whether an address is unsafe for a host-side URL import. */
479
+ export function isPrivateAddress(address: string): boolean {
480
+ const normalized = address.toLowerCase().split('%')[0] ?? ''
481
+ if (isIP(normalized) === 4) return privateIpv4(normalized)
482
+ if (isIP(normalized) !== 6) return true
483
+ if (normalized.startsWith('::ffff:')) return privateIpv4(normalized.slice(7))
484
+ return normalized === '::' || normalized === '::1'
485
+ || normalized.startsWith('fc') || normalized.startsWith('fd')
486
+ || /^fe[89ab]/.test(normalized)
487
+ || normalized.startsWith('ff')
488
+ || normalized.startsWith('2001:db8:')
489
+ }
490
+
491
+ type ResolveHost = (hostname: string) => Promise<readonly { readonly address: string }[]>
492
+
493
+ const resolveHost: ResolveHost = async hostname => lookup(hostname, { all: true, verbatim: true })
494
+
495
+ /** Parse and resolve one URL before the host is allowed to request it. */
496
+ export async function assertSafeImportUrl(input: string, resolver: ResolveHost = resolveHost): Promise<URL> {
497
+ let url: URL
498
+ try {
499
+ url = new URL(input)
500
+ } catch {
501
+ throw new Error('URL 格式无效')
502
+ }
503
+ if (url.protocol !== 'https:') throw new Error('URL 导入仅支持 HTTPS')
504
+ if (url.username.length > 0 || url.password.length > 0) throw new Error('URL 不能包含登录凭据')
505
+ const hostname = url.hostname.toLowerCase().replace(/^\[|\]$/g, '')
506
+ if (hostname === 'localhost' || hostname.endsWith('.localhost')) throw new Error('URL 不能指向本机或私有网络')
507
+ const addresses = isIP(hostname) === 0 ? await resolver(hostname) : [{ address: hostname }]
508
+ if (addresses.length === 0 || addresses.some(({ address }) => isPrivateAddress(address))) {
509
+ throw new Error('URL 不能指向本机、私有网络或保留地址')
510
+ }
511
+ return url
512
+ }
513
+
514
+ export async function fetchUrlContent(url: string): Promise<string> {
515
+ let current = await assertSafeImportUrl(url)
516
+ let response: Response | undefined
517
+ for (let redirects = 0; redirects <= MAX_URL_REDIRECTS; redirects += 1) {
518
+ response = await fetch(current, { redirect: 'manual', signal: AbortSignal.timeout(15_000) })
519
+ if (![301, 302, 303, 307, 308].includes(response.status)) break
520
+ if (redirects === MAX_URL_REDIRECTS) throw new Error(`重定向次数超过 ${MAX_URL_REDIRECTS}`)
521
+ const location = response.headers.get('location')
522
+ if (location === null) throw new Error('重定向响应缺少 Location')
523
+ current = await assertSafeImportUrl(new URL(location, current).href)
524
+ }
525
+ if (response === undefined) throw new Error('抓取失败')
526
+ if (!response.ok) throw new Error(`抓取失败:HTTP ${response.status}`)
527
+ const type = response.headers.get('content-type') ?? ''
528
+ const text = await response.text()
529
+ return type.includes('html') ? extractHtmlText(text) : text
530
+ }
531
+
532
+ /**
533
+ * Resolve one import request into a written file path. Shared by the file
534
+ * and URL routes; URL imports fetch first, then reuse the same write path.
535
+ * @param request - validated import body.
536
+ * @returns the absolute path of the written SKILL.md.
537
+ */
538
+ export async function resolveImport(request: ImportRequest | ImportUrlRequest): Promise<string> {
539
+ const content = 'content' in request
540
+ ? request.content
541
+ : await fetchUrlContent(request.url)
542
+ return writeSkillFile(request.name, request.target, content, request.workspacePath)
543
+ }
544
+
545
+ // ---- HTTP layer -----------------------------------------------------------
546
+
547
+ /** Read and parse a JSON request body within a byte cap. */
548
+ export function readJsonBody(req: IncomingMessage, limit: number = MAX_BODY_BYTES): Promise<unknown> {
549
+ return new Promise((resolve, reject) => {
550
+ const chunks: Buffer[] = []
551
+ let size = 0
552
+ req.on('data', (chunk: Buffer) => {
553
+ size += chunk.length
554
+ if (size > limit) {
555
+ reject(new Error('请求体过大'))
556
+ req.destroy()
557
+ return
558
+ }
559
+ chunks.push(chunk)
560
+ })
561
+ req.on('end', () => {
562
+ try {
563
+ resolve(JSON.parse(Buffer.concat(chunks).toString('utf8')))
564
+ } catch {
565
+ reject(new Error('无效的 JSON'))
566
+ }
567
+ })
568
+ req.on('error', reject)
569
+ })
570
+ }
571
+
572
+ /**
573
+ * Origin fence: the routes are served on the harness's loopback-only web
574
+ * server, so only the browser page itself (or a local curl) reaches them.
575
+ * A cross-origin page (any other website) is refused. Requests without an
576
+ * Origin header is rejected: all state-changing browser requests include it,
577
+ * while accepting an absent header would let non-browser clients bypass the fence.
578
+ */
579
+ export function originAllowed(req: IncomingMessage): boolean {
580
+ const origin = req.headers.origin
581
+ if (origin === undefined) return false
582
+ try {
583
+ const hostname = new URL(origin).hostname
584
+ return hostname === '127.0.0.1' || hostname === 'localhost'
585
+ } catch {
586
+ return false
587
+ }
588
+ }
589
+
590
+ /** Send one JSON response. */
591
+ export function sendJson(res: ServerResponse, status: number, body: unknown): void {
592
+ const payload = JSON.stringify(body)
593
+ res.writeHead(status, {
594
+ 'content-type': 'application/json; charset=utf-8',
595
+ 'content-length': Buffer.byteLength(payload),
596
+ })
597
+ res.end(payload)
598
+ }
599
+
600
+ /** Send a JSON error response with the given status. */
601
+ export function sendError(res: ServerResponse, status: number, error: string): void {
602
+ sendJson(res, status, { ok: false, error })
603
+ }