@h-ai/ai 0.1.0-alpha.34 → 0.1.0-alpha.35
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +125 -3
- package/dist/{ai-types-BZjo_rWW.d.ts → ai-audio-ws-protocol-C-V8N8YL.d.ts} +408 -2
- package/dist/{ai-reasoning-types-Cm3-HVVN.d.ts → ai-reasoning-types-BGf-x2Qj.d.ts} +391 -1
- package/dist/browser.d.ts +3 -3
- package/dist/browser.js +2 -2
- package/dist/{chunk-Y5BQR7QA.js → chunk-AXIZBZMV.js} +230 -4
- package/dist/chunk-AXIZBZMV.js.map +1 -0
- package/dist/{chunk-CXO3YSIG.js → chunk-URXCNMJW.js} +153 -4
- package/dist/chunk-URXCNMJW.js.map +1 -0
- package/dist/client/index.d.ts +56 -2
- package/dist/client/index.js +1 -1
- package/dist/index.d.ts +3 -3
- package/dist/index.js +1607 -318
- package/dist/index.js.map +1 -1
- package/package.json +7 -5
- package/dist/chunk-CXO3YSIG.js.map +0 -1
- package/dist/chunk-Y5BQR7QA.js.map +0 -1
|
@@ -91,6 +91,13 @@ interface StoreFilter<T> {
|
|
|
91
91
|
status?: string | string[];
|
|
92
92
|
/** 按 ref_id 索引列过滤(需要 AIRelStoreOptions.hasRefId 启用) */
|
|
93
93
|
refId?: string;
|
|
94
|
+
/**
|
|
95
|
+
* 按业务作用域过滤(`data.scope` JSON 字段的 key-value 包含匹配)。
|
|
96
|
+
*
|
|
97
|
+
* 仅在 PostgreSQL 上下推为 `data @> ?::jsonb`(配合 GIN 索引加速);
|
|
98
|
+
* SQLite / MySQL 上为 no-op(不影响结果,由调用方在内存中完成 scope 匹配)。
|
|
99
|
+
*/
|
|
100
|
+
scope?: Record<string, unknown>;
|
|
94
101
|
/** 排序 */
|
|
95
102
|
orderBy?: {
|
|
96
103
|
field: keyof T;
|
|
@@ -122,6 +129,14 @@ interface AIRelStoreOptions {
|
|
|
122
129
|
hasStatus?: boolean;
|
|
123
130
|
/** 是否创建 ref_id 索引列 */
|
|
124
131
|
hasRefId?: boolean;
|
|
132
|
+
/**
|
|
133
|
+
* 是否为业务作用域(记录内 `data.scope` JSON 字段)建立索引以加速 scope 过滤。
|
|
134
|
+
*
|
|
135
|
+
* 仅 PostgreSQL 生效:在 JSONB `data` 列上建 GIN 索引,配合 `StoreFilter.scope`
|
|
136
|
+
* 的 `data @> ?::jsonb` 包含查询下推,避免加载全部候选再在内存过滤。
|
|
137
|
+
* SQLite / MySQL 不建索引,`scope` 过滤退回调用方内存匹配(当前行为不变)。
|
|
138
|
+
*/
|
|
139
|
+
hasScopeIndex?: boolean;
|
|
125
140
|
}
|
|
126
141
|
/**
|
|
127
142
|
* AI 关系存储适配器
|
|
@@ -969,6 +984,32 @@ interface LLMProvider {
|
|
|
969
984
|
/** 获取可用模型列表 */
|
|
970
985
|
listModels: () => Promise<HaiResult<string[]>>;
|
|
971
986
|
}
|
|
987
|
+
/**
|
|
988
|
+
* 结构化输出请求
|
|
989
|
+
*
|
|
990
|
+
* 给定 Zod schema,让模型返回严格符合结构的 JSON(内部使用 `json_schema` response_format),
|
|
991
|
+
* 解析失败时自动带错误修复重试,避免调用方手写 `JSON.parse` 的不稳定性。
|
|
992
|
+
*/
|
|
993
|
+
interface GenerateObjectRequest<T> {
|
|
994
|
+
/** 输出结构的 Zod schema */
|
|
995
|
+
schema: ZodType<T>;
|
|
996
|
+
/** 对话消息 */
|
|
997
|
+
messages: ChatMessage[];
|
|
998
|
+
/** 模型名(可选,不传时使用默认模型) */
|
|
999
|
+
model?: string;
|
|
1000
|
+
/** 系统提示词(可选,追加为首条 system 消息) */
|
|
1001
|
+
systemPrompt?: string;
|
|
1002
|
+
/** 温度覆盖 */
|
|
1003
|
+
temperature?: number;
|
|
1004
|
+
/** 临时模型配置 */
|
|
1005
|
+
tempModel?: TempModelConfig;
|
|
1006
|
+
/** schema 名称(用于 json_schema response_format,默认 `result`) */
|
|
1007
|
+
schemaName?: string;
|
|
1008
|
+
/** 解析失败时的修复重试次数(默认 1) */
|
|
1009
|
+
maxRepairs?: number;
|
|
1010
|
+
/** 取消信号 */
|
|
1011
|
+
signal?: AbortSignal;
|
|
1012
|
+
}
|
|
972
1013
|
/**
|
|
973
1014
|
* LLM 操作接口(通过 `ai.llm` 访问)
|
|
974
1015
|
*
|
|
@@ -1003,6 +1044,15 @@ interface LLMOperations {
|
|
|
1003
1044
|
* @returns 文本片段的异步迭代器
|
|
1004
1045
|
*/
|
|
1005
1046
|
askStream: (question: string, options?: AskOptions) => AsyncIterable<string>;
|
|
1047
|
+
/**
|
|
1048
|
+
* 结构化输出:按 Zod schema 约束模型输出并解析为对象
|
|
1049
|
+
*
|
|
1050
|
+
* 内部使用 `json_schema` response_format 约束输出,解析/校验失败时自动带错误提示重试。
|
|
1051
|
+
*
|
|
1052
|
+
* @param request - 包含 schema、messages 及可选模型/修复次数等
|
|
1053
|
+
* @returns 符合 schema 的对象;多次重试仍无法解析时返回 `INVALID_REQUEST`
|
|
1054
|
+
*/
|
|
1055
|
+
generateObject: <T>(request: GenerateObjectRequest<T>) => Promise<HaiResult<T>>;
|
|
1006
1056
|
}
|
|
1007
1057
|
/** LLM 子功能工厂依赖(内部使用) */
|
|
1008
1058
|
interface AILLMFunctionsDeps {
|
|
@@ -1768,6 +1818,13 @@ interface MemoryExtractOptions {
|
|
|
1768
1818
|
interface MemoryRecallOptions {
|
|
1769
1819
|
/** 返回数量(默认使用配置的 defaultTopK) */
|
|
1770
1820
|
topK?: number;
|
|
1821
|
+
/**
|
|
1822
|
+
* 候选池倍数(覆盖配置的 candidateMultiplier)
|
|
1823
|
+
*
|
|
1824
|
+
* 检索时先取回 `topK × candidateMultiplier` 条候选再按 scope / 类型 / 重要性过滤,
|
|
1825
|
+
* 最后截取 topK。scope 过滤较严(如按 topic / persona 隔离)时应调大,避免漏召回。
|
|
1826
|
+
*/
|
|
1827
|
+
candidateMultiplier?: number;
|
|
1771
1828
|
/** 过滤类型 */
|
|
1772
1829
|
types?: MemoryType[];
|
|
1773
1830
|
/** 最低重要性 */
|
|
@@ -1787,6 +1844,12 @@ interface MemoryRecallOptions {
|
|
|
1787
1844
|
interface MemoryInjectionOptions {
|
|
1788
1845
|
/** 注入的记忆数量(默认 5) */
|
|
1789
1846
|
topK?: number;
|
|
1847
|
+
/**
|
|
1848
|
+
* 候选池倍数(覆盖配置的 candidateMultiplier)
|
|
1849
|
+
*
|
|
1850
|
+
* 透传给底层 `recall`,scope 过滤较严时应调大以避免漏召回。
|
|
1851
|
+
*/
|
|
1852
|
+
candidateMultiplier?: number;
|
|
1790
1853
|
/** 记忆占用的最大 token 预算(默认不限) */
|
|
1791
1854
|
maxTokens?: number;
|
|
1792
1855
|
/** 注入位置:system = 追加到 system 消息末尾,before-last = 插入在最后一条用户消息之前 */
|
|
@@ -2346,6 +2409,7 @@ declare const MemoryConfigSchema: z.ZodObject<{
|
|
|
2346
2409
|
recencyDecay: z.ZodDefault<z.ZodNumber>;
|
|
2347
2410
|
embeddingEnabled: z.ZodDefault<z.ZodBoolean>;
|
|
2348
2411
|
defaultTopK: z.ZodDefault<z.ZodNumber>;
|
|
2412
|
+
candidateMultiplier: z.ZodDefault<z.ZodNumber>;
|
|
2349
2413
|
writebackRelatedTopK: z.ZodDefault<z.ZodNumber>;
|
|
2350
2414
|
}, z.core.$strip>;
|
|
2351
2415
|
/** Memory 配置类型 */
|
|
@@ -2518,6 +2582,135 @@ declare const A2AConfigSchema: z.ZodObject<{
|
|
|
2518
2582
|
}, z.core.$strip>;
|
|
2519
2583
|
/** A2A 配置类型 */
|
|
2520
2584
|
type A2AConfig = z.infer<typeof A2AConfigSchema>;
|
|
2585
|
+
/**
|
|
2586
|
+
* 语音平台枚举
|
|
2587
|
+
*
|
|
2588
|
+
* 决定 `ai.audio` 底层调用哪个厂商(对使用方透明,公共请求/响应形状保持一致):
|
|
2589
|
+
*
|
|
2590
|
+
* - `openai` — OpenAI Audio API(transcriptions / speech)
|
|
2591
|
+
* - `mimo` — 小米 MiMo(Chat Completions 风格 ASR / TTS)
|
|
2592
|
+
* - `qwen` — 阿里云百炼 Qwen Realtime(DashScope WebSocket ASR / TTS)
|
|
2593
|
+
* - `doubao` — 火山引擎豆包语音(二进制 WebSocket ASR / TTS)
|
|
2594
|
+
*/
|
|
2595
|
+
declare const AudioProviderSchema: z.ZodEnum<{
|
|
2596
|
+
openai: "openai";
|
|
2597
|
+
mimo: "mimo";
|
|
2598
|
+
qwen: "qwen";
|
|
2599
|
+
doubao: "doubao";
|
|
2600
|
+
}>;
|
|
2601
|
+
/** 语音平台类型 */
|
|
2602
|
+
type AudioProviderName = z.infer<typeof AudioProviderSchema>;
|
|
2603
|
+
/**
|
|
2604
|
+
* 语音模型条目 Schema
|
|
2605
|
+
*
|
|
2606
|
+
* 定义单个语音模型:唯一 ID、所属平台、厂商模型名及凭据。凭据未提供时回退到对应平台的环境变量。
|
|
2607
|
+
*
|
|
2608
|
+
* @example
|
|
2609
|
+
* ```ts
|
|
2610
|
+
* const model = { id: 'asr', provider: 'qwen', model: 'qwen3-asr-flash-realtime' }
|
|
2611
|
+
* ```
|
|
2612
|
+
*/
|
|
2613
|
+
declare const AudioModelEntrySchema: z.ZodObject<{
|
|
2614
|
+
id: z.ZodString;
|
|
2615
|
+
provider: z.ZodEnum<{
|
|
2616
|
+
openai: "openai";
|
|
2617
|
+
mimo: "mimo";
|
|
2618
|
+
qwen: "qwen";
|
|
2619
|
+
doubao: "doubao";
|
|
2620
|
+
}>;
|
|
2621
|
+
model: z.ZodString;
|
|
2622
|
+
apiKey: z.ZodOptional<z.ZodString>;
|
|
2623
|
+
baseUrl: z.ZodOptional<z.ZodString>;
|
|
2624
|
+
appKey: z.ZodOptional<z.ZodString>;
|
|
2625
|
+
accessKey: z.ZodOptional<z.ZodString>;
|
|
2626
|
+
resourceId: z.ZodOptional<z.ZodString>;
|
|
2627
|
+
workspaceId: z.ZodOptional<z.ZodString>;
|
|
2628
|
+
timeout: z.ZodOptional<z.ZodNumber>;
|
|
2629
|
+
}, z.core.$strip>;
|
|
2630
|
+
/** 语音模型条目类型 */
|
|
2631
|
+
type AudioModelEntry = z.infer<typeof AudioModelEntrySchema>;
|
|
2632
|
+
/**
|
|
2633
|
+
* Audio 配置 Schema
|
|
2634
|
+
*
|
|
2635
|
+
* 注册语音模型并映射默认识别 / 合成模型。调用方通常无需指定模型,仅在临时切换时通过 `request.model` 覆盖。
|
|
2636
|
+
*
|
|
2637
|
+
* @example
|
|
2638
|
+
* ```ts
|
|
2639
|
+
* ai.init({
|
|
2640
|
+
* audio: {
|
|
2641
|
+
* models: [
|
|
2642
|
+
* { id: 'asr', provider: 'qwen', model: 'qwen3-asr-flash-realtime' },
|
|
2643
|
+
* { id: 'tts', provider: 'qwen', model: 'qwen3-tts-flash-realtime' },
|
|
2644
|
+
* ],
|
|
2645
|
+
* transcribeModel: 'asr',
|
|
2646
|
+
* synthesizeModel: 'tts',
|
|
2647
|
+
* },
|
|
2648
|
+
* })
|
|
2649
|
+
* ```
|
|
2650
|
+
*/
|
|
2651
|
+
declare const AudioConfigSchema: z.ZodObject<{
|
|
2652
|
+
models: z.ZodOptional<z.ZodArray<z.ZodObject<{
|
|
2653
|
+
id: z.ZodString;
|
|
2654
|
+
provider: z.ZodEnum<{
|
|
2655
|
+
openai: "openai";
|
|
2656
|
+
mimo: "mimo";
|
|
2657
|
+
qwen: "qwen";
|
|
2658
|
+
doubao: "doubao";
|
|
2659
|
+
}>;
|
|
2660
|
+
model: z.ZodString;
|
|
2661
|
+
apiKey: z.ZodOptional<z.ZodString>;
|
|
2662
|
+
baseUrl: z.ZodOptional<z.ZodString>;
|
|
2663
|
+
appKey: z.ZodOptional<z.ZodString>;
|
|
2664
|
+
accessKey: z.ZodOptional<z.ZodString>;
|
|
2665
|
+
resourceId: z.ZodOptional<z.ZodString>;
|
|
2666
|
+
workspaceId: z.ZodOptional<z.ZodString>;
|
|
2667
|
+
timeout: z.ZodOptional<z.ZodNumber>;
|
|
2668
|
+
}, z.core.$strip>>>;
|
|
2669
|
+
transcribeModel: z.ZodOptional<z.ZodString>;
|
|
2670
|
+
synthesizeModel: z.ZodOptional<z.ZodString>;
|
|
2671
|
+
maxAudioBytes: z.ZodDefault<z.ZodNumber>;
|
|
2672
|
+
maxStreamDurationMs: z.ZodDefault<z.ZodNumber>;
|
|
2673
|
+
}, z.core.$strip>;
|
|
2674
|
+
/** Audio 配置类型 */
|
|
2675
|
+
type AudioConfig = z.infer<typeof AudioConfigSchema>;
|
|
2676
|
+
/**
|
|
2677
|
+
* 已解析的语音模型配置
|
|
2678
|
+
*
|
|
2679
|
+
* 由 `resolveAudioModel()` 返回,凭据已合并环境变量、端点已应用平台默认值。
|
|
2680
|
+
*/
|
|
2681
|
+
interface ResolvedAudioModel {
|
|
2682
|
+
/** 模型条目 ID */
|
|
2683
|
+
id: string;
|
|
2684
|
+
/** 所属平台 */
|
|
2685
|
+
provider: AudioProviderName;
|
|
2686
|
+
/** 厂商模型名 */
|
|
2687
|
+
model: string;
|
|
2688
|
+
/** API Key(条目 > 平台环境变量) */
|
|
2689
|
+
apiKey: string | undefined;
|
|
2690
|
+
/** 端点(条目 > 平台默认) */
|
|
2691
|
+
baseUrl: string;
|
|
2692
|
+
/** 火山引擎 App Key */
|
|
2693
|
+
appKey: string | undefined;
|
|
2694
|
+
/** 火山引擎 Access Key */
|
|
2695
|
+
accessKey: string | undefined;
|
|
2696
|
+
/** 火山引擎资源 ID(条目 > 平台默认) */
|
|
2697
|
+
resourceId: string;
|
|
2698
|
+
/** 阿里云百炼业务空间 ID */
|
|
2699
|
+
workspaceId: string | undefined;
|
|
2700
|
+
/** 请求超时(毫秒) */
|
|
2701
|
+
timeout: number;
|
|
2702
|
+
}
|
|
2703
|
+
/**
|
|
2704
|
+
* 解析语音操作应使用的模型配置
|
|
2705
|
+
*
|
|
2706
|
+
* 解析优先级:请求显式 `model` > 场景默认(transcribe/synthesize);凭据回退到平台环境变量。
|
|
2707
|
+
*
|
|
2708
|
+
* @param audioConfig - Audio 配置
|
|
2709
|
+
* @param operation - 操作类型(识别 / 合成),决定使用哪个默认模型
|
|
2710
|
+
* @param explicit - 请求显式指定的模型 ID(最高优先级)
|
|
2711
|
+
* @returns 成功返回已解析模型;无匹配模型返回 `AUDIO_MODEL_NOT_FOUND`;缺少凭据返回 `CONFIGURATION_ERROR`
|
|
2712
|
+
*/
|
|
2713
|
+
declare function resolveAudioModel(audioConfig: AudioConfig, operation: 'transcribe' | 'synthesize', explicit?: string): HaiResult<ResolvedAudioModel>;
|
|
2521
2714
|
/**
|
|
2522
2715
|
* AI 配置 Schema
|
|
2523
2716
|
*
|
|
@@ -2664,6 +2857,7 @@ declare const AIConfigSchema: z.ZodObject<{
|
|
|
2664
2857
|
recencyDecay: z.ZodDefault<z.ZodNumber>;
|
|
2665
2858
|
embeddingEnabled: z.ZodDefault<z.ZodBoolean>;
|
|
2666
2859
|
defaultTopK: z.ZodDefault<z.ZodNumber>;
|
|
2860
|
+
candidateMultiplier: z.ZodDefault<z.ZodNumber>;
|
|
2667
2861
|
writebackRelatedTopK: z.ZodDefault<z.ZodNumber>;
|
|
2668
2862
|
}, z.core.$strip>>;
|
|
2669
2863
|
token: z.ZodOptional<z.ZodObject<{
|
|
@@ -2718,12 +2912,204 @@ declare const AIConfigSchema: z.ZodObject<{
|
|
|
2718
2912
|
}, z.core.$strip>>;
|
|
2719
2913
|
}, z.core.$strip>>;
|
|
2720
2914
|
}, z.core.$strip>>;
|
|
2915
|
+
audio: z.ZodOptional<z.ZodObject<{
|
|
2916
|
+
models: z.ZodOptional<z.ZodArray<z.ZodObject<{
|
|
2917
|
+
id: z.ZodString;
|
|
2918
|
+
provider: z.ZodEnum<{
|
|
2919
|
+
openai: "openai";
|
|
2920
|
+
mimo: "mimo";
|
|
2921
|
+
qwen: "qwen";
|
|
2922
|
+
doubao: "doubao";
|
|
2923
|
+
}>;
|
|
2924
|
+
model: z.ZodString;
|
|
2925
|
+
apiKey: z.ZodOptional<z.ZodString>;
|
|
2926
|
+
baseUrl: z.ZodOptional<z.ZodString>;
|
|
2927
|
+
appKey: z.ZodOptional<z.ZodString>;
|
|
2928
|
+
accessKey: z.ZodOptional<z.ZodString>;
|
|
2929
|
+
resourceId: z.ZodOptional<z.ZodString>;
|
|
2930
|
+
workspaceId: z.ZodOptional<z.ZodString>;
|
|
2931
|
+
timeout: z.ZodOptional<z.ZodNumber>;
|
|
2932
|
+
}, z.core.$strip>>>;
|
|
2933
|
+
transcribeModel: z.ZodOptional<z.ZodString>;
|
|
2934
|
+
synthesizeModel: z.ZodOptional<z.ZodString>;
|
|
2935
|
+
maxAudioBytes: z.ZodDefault<z.ZodNumber>;
|
|
2936
|
+
maxStreamDurationMs: z.ZodDefault<z.ZodNumber>;
|
|
2937
|
+
}, z.core.$strip>>;
|
|
2721
2938
|
}, z.core.$strip>;
|
|
2722
2939
|
/** AI 配置类型(校验后的完整类型) */
|
|
2723
2940
|
type AIConfig = z.infer<typeof AIConfigSchema>;
|
|
2724
2941
|
/** AI 配置输入类型(允许部分字段) */
|
|
2725
2942
|
type AIConfigInput = z.input<typeof AIConfigSchema>;
|
|
2726
2943
|
|
|
2944
|
+
/**
|
|
2945
|
+
* @h-ai/ai — Audio 子功能公共类型
|
|
2946
|
+
*
|
|
2947
|
+
* 仅定义语音识别(ASR)与语音合成(TTS)的对外业务类型。
|
|
2948
|
+
* 不暴露 WebSocket、SSE、厂商事件、Session/Commit、二进制帧等传输细节,
|
|
2949
|
+
* 也不暴露内部 Provider 接口与厂商报文结构。
|
|
2950
|
+
* @module audio/ai-audio-types
|
|
2951
|
+
*/
|
|
2952
|
+
|
|
2953
|
+
/**
|
|
2954
|
+
* 音频编码格式
|
|
2955
|
+
*
|
|
2956
|
+
* - `pcm16` — 16bit 小端裸 PCM(必须提供 `sampleRate`)
|
|
2957
|
+
* - `wav` — WAV 容器(自描述采样率,可不传 `sampleRate`)
|
|
2958
|
+
* - `mp3` — MP3 容器
|
|
2959
|
+
* - `opus` — Opus 编码
|
|
2960
|
+
*/
|
|
2961
|
+
type AudioFormat = 'pcm16' | 'wav' | 'mp3' | 'opus';
|
|
2962
|
+
/**
|
|
2963
|
+
* 完整音频内容(一次性输入/输出)
|
|
2964
|
+
*
|
|
2965
|
+
* `data` 统一使用二进制字节;Node.js 中的 `Buffer` 可直接作为 `Uint8Array` 传入。
|
|
2966
|
+
* Provider 内部会按厂商要求转换为 Base64 / Multipart / 二进制帧,调用方无需感知。
|
|
2967
|
+
*/
|
|
2968
|
+
interface AudioContent {
|
|
2969
|
+
/** 音频二进制数据 */
|
|
2970
|
+
data: Uint8Array;
|
|
2971
|
+
/** 音频编码格式 */
|
|
2972
|
+
format: AudioFormat;
|
|
2973
|
+
/** 采样率(Hz);`pcm16` 等裸音频必填,`wav` / `mp3` 等自描述格式可省略 */
|
|
2974
|
+
sampleRate?: number;
|
|
2975
|
+
/** 声道数(默认单声道) */
|
|
2976
|
+
channels?: 1 | 2;
|
|
2977
|
+
}
|
|
2978
|
+
/**
|
|
2979
|
+
* 实时音频输入流(持续到达的音频分片)
|
|
2980
|
+
*
|
|
2981
|
+
* 仅表达「音频分片持续到达」这一业务语义,不包含 WebSocket、Session、Commit 等传输概念。
|
|
2982
|
+
*/
|
|
2983
|
+
interface AudioInputStream {
|
|
2984
|
+
/** 持续到达的音频分片 */
|
|
2985
|
+
chunks: AsyncIterable<Uint8Array>;
|
|
2986
|
+
/** 音频编码格式 */
|
|
2987
|
+
format: AudioFormat;
|
|
2988
|
+
/** 采样率(Hz) */
|
|
2989
|
+
sampleRate: number;
|
|
2990
|
+
/** 声道数(默认单声道) */
|
|
2991
|
+
channels?: 1 | 2;
|
|
2992
|
+
}
|
|
2993
|
+
/** 完整语音识别请求 */
|
|
2994
|
+
interface TranscriptionRequest {
|
|
2995
|
+
/** 待识别的完整音频 */
|
|
2996
|
+
audio: AudioContent;
|
|
2997
|
+
/** 语言提示(如 `zh` / `en`;不传时由模型自动检测) */
|
|
2998
|
+
language?: string;
|
|
2999
|
+
/**
|
|
3000
|
+
* 领域提示词 / 热词(如角色名、专有名词、当前主题关键词)
|
|
3001
|
+
*
|
|
3002
|
+
* Provider 按能力映射为热词表 / phrase list / vocabulary / 提示词;不支持的平台会忽略。
|
|
3003
|
+
*/
|
|
3004
|
+
contextHints?: string[];
|
|
3005
|
+
/** 模型 ID(不传时使用配置中的默认识别模型) */
|
|
3006
|
+
model?: string;
|
|
3007
|
+
/** 取消信号 */
|
|
3008
|
+
signal?: AbortSignal;
|
|
3009
|
+
}
|
|
3010
|
+
/** 流式语音识别请求(支持完整音频或持续音频输入) */
|
|
3011
|
+
interface TranscriptionStreamRequest {
|
|
3012
|
+
/** 完整音频,或持续到达的音频输入流 */
|
|
3013
|
+
audio: AudioContent | AudioInputStream;
|
|
3014
|
+
/** 语言提示(如 `zh` / `en`;不传时由模型自动检测) */
|
|
3015
|
+
language?: string;
|
|
3016
|
+
/**
|
|
3017
|
+
* 领域提示词 / 热词(如角色名、专有名词、当前主题关键词)
|
|
3018
|
+
*
|
|
3019
|
+
* Provider 按能力映射为热词表 / phrase list / vocabulary / 提示词;不支持的平台会忽略。
|
|
3020
|
+
*/
|
|
3021
|
+
contextHints?: string[];
|
|
3022
|
+
/** 模型 ID(不传时使用配置中的默认识别模型) */
|
|
3023
|
+
model?: string;
|
|
3024
|
+
/** 取消信号 */
|
|
3025
|
+
signal?: AbortSignal;
|
|
3026
|
+
}
|
|
3027
|
+
/** 完整语音识别结果 */
|
|
3028
|
+
interface TranscriptionResult {
|
|
3029
|
+
/** 识别文本 */
|
|
3030
|
+
text: string;
|
|
3031
|
+
}
|
|
3032
|
+
/**
|
|
3033
|
+
* 流式语音识别领域事件
|
|
3034
|
+
*
|
|
3035
|
+
* 统一的语音领域事件(非厂商协议)。支持服务端 VAD 的平台(Qwen / 豆包)会在检测到语音
|
|
3036
|
+
* 起止时额外产出 `speech_started` / `speech_stopped`,使调用方可在「开始说话」的瞬间做出反应
|
|
3037
|
+
* (如取消当前上游生成),而无需自行运行 VAD;不支持 VAD 的平台仅产出 `transcript`。
|
|
3038
|
+
*
|
|
3039
|
+
* `transcript.text` 表示当前语句的完整识别文本(非字符增量),实时 ASR 会修订前一次临时结果,
|
|
3040
|
+
* 调用方可直接用 `text` 覆盖当前临时文本。
|
|
3041
|
+
*/
|
|
3042
|
+
type TranscriptionEvent = {
|
|
3043
|
+
type: 'speech_started';
|
|
3044
|
+
} | {
|
|
3045
|
+
type: 'transcript';
|
|
3046
|
+
text: string;
|
|
3047
|
+
final: boolean;
|
|
3048
|
+
} | {
|
|
3049
|
+
type: 'speech_stopped';
|
|
3050
|
+
};
|
|
3051
|
+
/** 完整语音合成请求 */
|
|
3052
|
+
interface SynthesisRequest {
|
|
3053
|
+
/** 待合成文本 */
|
|
3054
|
+
text: string;
|
|
3055
|
+
/** 音色(厂商音色名,不传时使用模型默认音色) */
|
|
3056
|
+
voice?: string;
|
|
3057
|
+
/**
|
|
3058
|
+
* 自然语言风格指令(如语速、情绪、角色语气)
|
|
3059
|
+
*
|
|
3060
|
+
* Provider 按能力映射(如 MiMo 放入 user 消息、Qwen instructions);不支持的平台会忽略。
|
|
3061
|
+
*/
|
|
3062
|
+
instruction?: string;
|
|
3063
|
+
/** 输出音频格式(不传时使用模型默认格式) */
|
|
3064
|
+
format?: AudioFormat;
|
|
3065
|
+
/** 输出采样率(Hz) */
|
|
3066
|
+
sampleRate?: number;
|
|
3067
|
+
/** 模型 ID(不传时使用配置中的默认合成模型) */
|
|
3068
|
+
model?: string;
|
|
3069
|
+
/** 取消信号 */
|
|
3070
|
+
signal?: AbortSignal;
|
|
3071
|
+
}
|
|
3072
|
+
/** 流式语音合成请求(支持完整文本或持续文本输入) */
|
|
3073
|
+
interface SynthesisStreamRequest {
|
|
3074
|
+
/** 完整文本,或持续到达的文本流(可直接连接 LLM 文本流实现边生成边合成) */
|
|
3075
|
+
text: string | AsyncIterable<string>;
|
|
3076
|
+
/** 音色(厂商音色名,不传时使用模型默认音色) */
|
|
3077
|
+
voice?: string;
|
|
3078
|
+
/**
|
|
3079
|
+
* 自然语言风格指令(如语速、情绪、角色语气)
|
|
3080
|
+
*
|
|
3081
|
+
* Provider 按能力映射(如 MiMo 放入 user 消息、Qwen instructions);不支持的平台会忽略。
|
|
3082
|
+
*/
|
|
3083
|
+
instruction?: string;
|
|
3084
|
+
/** 输出音频格式(不传时使用模型默认格式) */
|
|
3085
|
+
format?: AudioFormat;
|
|
3086
|
+
/** 输出采样率(Hz) */
|
|
3087
|
+
sampleRate?: number;
|
|
3088
|
+
/** 模型 ID(不传时使用配置中的默认合成模型) */
|
|
3089
|
+
model?: string;
|
|
3090
|
+
/** 取消信号 */
|
|
3091
|
+
signal?: AbortSignal;
|
|
3092
|
+
}
|
|
3093
|
+
/** 完整语音合成结果 */
|
|
3094
|
+
interface SynthesisResult extends AudioContent {
|
|
3095
|
+
}
|
|
3096
|
+
/**
|
|
3097
|
+
* Audio 操作接口(通过 `ai.audio` 访问)
|
|
3098
|
+
*
|
|
3099
|
+
* 提供语音识别与语音合成的完整与流式能力。普通方法返回 `HaiResult`;
|
|
3100
|
+
* 流式方法返回 `AsyncIterable`,迭代期间发生的连接/协议/上游错误会终止异步迭代(抛出异常)。
|
|
3101
|
+
*/
|
|
3102
|
+
interface AudioOperations {
|
|
3103
|
+
/** 将完整音频识别为完整文本 */
|
|
3104
|
+
transcribe: (request: TranscriptionRequest) => Promise<HaiResult<TranscriptionResult>>;
|
|
3105
|
+
/** 持续输入音频或增量返回识别文本(含语音起止领域事件) */
|
|
3106
|
+
transcribeStream: (request: TranscriptionStreamRequest) => AsyncIterable<TranscriptionEvent>;
|
|
3107
|
+
/** 将完整文本合成为完整音频 */
|
|
3108
|
+
synthesize: (request: SynthesisRequest) => Promise<HaiResult<SynthesisResult>>;
|
|
3109
|
+
/** 持续输入文本或增量输出音频 */
|
|
3110
|
+
synthesizeStream: (request: SynthesisStreamRequest) => AsyncIterable<Uint8Array>;
|
|
3111
|
+
}
|
|
3112
|
+
|
|
2727
3113
|
/**
|
|
2728
3114
|
* @h-ai/ai — RAG(Retrieval-Augmented Generation)子功能类型
|
|
2729
3115
|
*
|
|
@@ -2761,6 +3147,8 @@ interface RagOptions {
|
|
|
2761
3147
|
messages?: ChatMessage[];
|
|
2762
3148
|
/** 是否启用内部 LLM 调用的持久化(默认 true;Context 层调用时传 false 避免重复记录) */
|
|
2763
3149
|
enablePersist?: boolean;
|
|
3150
|
+
/** 请求取消信号(透传给内部 LLM 调用,支持中途取消生成) */
|
|
3151
|
+
signal?: AbortSignal;
|
|
2764
3152
|
}
|
|
2765
3153
|
/**
|
|
2766
3154
|
* RAG 上下文项
|
|
@@ -2913,6 +3301,8 @@ interface ReasoningOptions {
|
|
|
2913
3301
|
tools?: ToolRegistryOperations;
|
|
2914
3302
|
/** 温度覆盖 */
|
|
2915
3303
|
temperature?: number;
|
|
3304
|
+
/** 请求取消信号(透传给内部 LLM 调用,支持中途取消推理) */
|
|
3305
|
+
signal?: AbortSignal;
|
|
2916
3306
|
}
|
|
2917
3307
|
/**
|
|
2918
3308
|
* 推理步骤类型
|
|
@@ -3013,4 +3403,4 @@ interface ReasoningOperations {
|
|
|
3013
3403
|
runStream: (query: string, options?: ReasoningOptions) => AsyncIterable<ReasoningStreamEvent>;
|
|
3014
3404
|
}
|
|
3015
3405
|
|
|
3016
|
-
export { type
|
|
3406
|
+
export { type SynthesisRequest as $, type A2AConfig as A, MCPServerCapabilitiesSchema as B, type CompressConfig as C, type MCPServerConfig as D, type EmbeddingConfig as E, type FileConfig as F, MCPServerConfigSchema as G, type MemoryConfig as H, MemoryConfigSchema as I, type MemoryType as J, type KnowledgeConfig as K, type LLMConfig as L, type MCPConfig as M, MemoryTypeSchema as N, type ModelEntry as O, ModelEntrySchema as P, type ModelScenario as Q, ModelScenarioSchema as R, type ResolveRequiredModelEntryOptions as S, type ResolvedAudioModel as T, type ResolvedModelConfig as U, type RetrievalConfig as V, RetrievalConfigSchema as W, type RetrievalSourceConfig as X, RetrievalSourceSchema as Y, type SummaryConfig as Z, SummaryConfigSchema as _, A2AConfigSchema as a, type KnowledgeRetrieveOptions as a$, type SynthesisResult as a0, type SynthesisStreamRequest as a1, type TokenConfig as a2, TokenConfigSchema as a3, type TranscriptionEvent as a4, type TranscriptionRequest as a5, type TranscriptionResult as a6, type TranscriptionStreamRequest as a7, resolveAudioModel as a8, resolveModelApi as a9, type ChatCompletionResponse as aA, type ChatHistoryOptions as aB, type ChatMessage as aC, type ChatRecord as aD, type Citation as aE, type DefineToolOptions as aF, type DeveloperMessage as aG, type EntityDocumentRelation as aH, type EntityDocumentResult as aI, type EntityListOptions as aJ, type EntityQueryOptions as aK, type GenerateObjectRequest as aL, type ImageContent as aM, type InteractionScope as aN, type KnowledgeAskOptions as aO, type KnowledgeAskResult as aP, type KnowledgeDocumentInfo as aQ, type KnowledgeDocumentListOptions as aR, type KnowledgeDocumentRemoveOptions as aS, type KnowledgeEntity as aT, type KnowledgeIngestBatchProgress as aU, type KnowledgeIngestBatchResult as aV, type KnowledgeIngestFileInput as aW, type KnowledgeIngestInput as aX, type KnowledgeIngestResult as aY, type KnowledgeOperations as aZ, type KnowledgeRetrieveItem as a_, resolveModelEntry as aa, type A2AAgentCardConfig as ab, type A2AApiKeySecurity as ac, type A2AAuthenticator as ad, type A2ACallOptions as ae, type A2ACallResult as af, type A2ACallerIdentity as ag, type A2AClientCallRecord as ah, type A2AContextInfo as ai, type A2AHandleResult as aj, type A2AMessageRecord as ak, type A2AOperations as al, type A2ASecurityConfig as am, type A2ATaskFilter as an, type AILLMFunctionsDeps as ao, type AIRelStore as ap, type AIRelStoreOptions as aq, type AIStoreProvider as ar, type AIVectorBackend as as, type AIVectorStore as at, type AskOptions as au, type AssistantMessage as av, type ChatCompletionChoice as aw, type ChatCompletionChunk as ax, type ChatCompletionDelta as ay, type ChatCompletionRequest as az, A2ASkillConfigSchema as b, type KnowledgeRetrieveResult as b0, type KnowledgeSetupOptions as b1, type KnowledgeStore as b2, type LLMOperations as b3, type LLMProvider as b4, type MemoryClearOptions as b5, type MemoryEntry as b6, type MemoryEntryInput as b7, type MemoryExtractOptions as b8, type MemoryInjectionOptions as b9, type SSEEvent as bA, type SessionInfo as bB, type StoreFilter as bC, type StorePage as bD, type StoreScope as bE, type StreamOperations as bF, type StreamProcessor as bG, type StreamResult as bH, type SystemMessage as bI, type TempModelConfig as bJ, type TextContent as bK, type TokenUsage as bL, type Tool as bM, type ToolCall as bN, type ToolDefinition as bO, type ToolErrorType as bP, type ToolMessage as bQ, type ToolRegistryOperations as bR, type ToolsOperations as bS, type UserMessage as bT, type WhereClause as bU, type WhereOperator as bV, type WhereValue as bW, type MemoryListOptions as ba, type MemoryListPageOptions as bb, type MemoryOperations as bc, type MemoryRecallOptions as bd, type MemoryUpdateInput as be, type MessageContent as bf, type MessageRole as bg, type ObjectRef as bh, type RagContextItem as bi, type RagOperations as bj, type RagOptions as bk, type RagResult as bl, type RagStreamEvent as bm, type ReasoningOperations as bn, type ReasoningOptions as bo, type ReasoningResult as bp, type ReasoningStep as bq, type ReasoningStepType as br, type ReasoningStrategy as bs, type ReasoningStreamEvent as bt, type RetrievalOperations as bu, type RetrievalRequest as bv, type RetrievalResult as bw, type RetrievalResultItem as bx, type RetrievalSource as by, type SSEDecoder as bz, type AIConfig as c, type AIConfigInput as d, AIConfigSchema as e, type ApiType as f, ApiTypeSchema as g, type AudioConfig as h, AudioConfigSchema as i, type AudioContent as j, type AudioFormat as k, type AudioInputStream as l, type AudioModelEntry as m, AudioModelEntrySchema as n, type AudioOperations as o, type AudioProviderName as p, AudioProviderSchema as q, CompressConfigSchema as r, EmbeddingConfigSchema as s, type EntityType as t, EntityTypeSchema as u, FileConfigSchema as v, KnowledgeConfigSchema as w, LLMConfigSchema as x, MCPConfigSchema as y, type MCPServerCapabilities as z };
|
package/dist/browser.d.ts
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
|
-
export { A as A2AConfig, a as A2AConfigSchema, b as A2ASkillConfigSchema, c as AIConfig, d as AIConfigInput, e as AIConfigSchema, f as ApiType, g as ApiTypeSchema, C as CompressConfig,
|
|
2
|
-
export { A as AIFunctions, a as AIInitOptions, C as CompressionStrategy,
|
|
3
|
-
export { A2AClientOperations, AIApiAdapter, AIClient, AIClientConfig, StreamOptions, StreamProgress, collectStreamContent, createA2AClient, createAIClient, parseSSE } from './client/index.js';
|
|
1
|
+
export { A as A2AConfig, a as A2AConfigSchema, b as A2ASkillConfigSchema, c as AIConfig, d as AIConfigInput, e as AIConfigSchema, f as ApiType, g as ApiTypeSchema, h as AudioConfig, i as AudioConfigSchema, j as AudioContent, k as AudioFormat, l as AudioInputStream, m as AudioModelEntry, n as AudioModelEntrySchema, o as AudioOperations, p as AudioProviderName, q as AudioProviderSchema, C as CompressConfig, r as CompressConfigSchema, E as EmbeddingConfig, s as EmbeddingConfigSchema, t as EntityType, u as EntityTypeSchema, F as FileConfig, v as FileConfigSchema, K as KnowledgeConfig, w as KnowledgeConfigSchema, L as LLMConfig, x as LLMConfigSchema, M as MCPConfig, y as MCPConfigSchema, z as MCPServerCapabilities, B as MCPServerCapabilitiesSchema, D as MCPServerConfig, G as MCPServerConfigSchema, H as MemoryConfig, I as MemoryConfigSchema, J as MemoryType, N as MemoryTypeSchema, O as ModelEntry, P as ModelEntrySchema, Q as ModelScenario, R as ModelScenarioSchema, S as ResolveRequiredModelEntryOptions, T as ResolvedAudioModel, U as ResolvedModelConfig, V as RetrievalConfig, W as RetrievalConfigSchema, X as RetrievalSourceConfig, Y as RetrievalSourceSchema, Z as SummaryConfig, _ as SummaryConfigSchema, $ as SynthesisRequest, a0 as SynthesisResult, a1 as SynthesisStreamRequest, a2 as TokenConfig, a3 as TokenConfigSchema, a4 as TranscriptionEvent, a5 as TranscriptionRequest, a6 as TranscriptionResult, a7 as TranscriptionStreamRequest, a8 as resolveAudioModel, a9 as resolveModelApi, aa as resolveModelEntry } from './ai-reasoning-types-BGf-x2Qj.js';
|
|
2
|
+
export { A as AIFunctions, a as AIInitOptions, b as AUDIO_WS_PATH, c as AudioWsClientMessage, d as AudioWsDoneMessage, e as AudioWsEndMessage, f as AudioWsErrorMessage, g as AudioWsServerMessage, h as AudioWsSpeechMessage, i as AudioWsStartMessage, j as AudioWsTextMessage, k as AudioWsTranscriptMessage, C as CompressionStrategy, l as CompressionStrategySchema, H as HaiAIError } from './ai-audio-ws-protocol-C-V8N8YL.js';
|
|
3
|
+
export { A2AClientOperations, AIApiAdapter, AIClient, AIClientConfig, AudioClientConfig, AudioClientOperations, StreamOptions, StreamProgress, collectStreamContent, createA2AClient, createAIClient, createAudioClient, createUnconfiguredAudioClient, parseSSE } from './client/index.js';
|
|
4
4
|
import '@a2a-js/sdk/server';
|
|
5
5
|
import '@h-ai/core';
|
|
6
6
|
import '@h-ai/datapipe';
|
package/dist/browser.js
CHANGED
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
export { A2AConfigSchema, A2ASkillConfigSchema, AIConfigSchema, ApiTypeSchema, CompressConfigSchema, CompressionStrategySchema, EmbeddingConfigSchema, EntityTypeSchema, FileConfigSchema, HaiAIError, KnowledgeConfigSchema, LLMConfigSchema, MCPConfigSchema, MCPServerCapabilitiesSchema, MCPServerConfigSchema, MemoryConfigSchema, MemoryTypeSchema, ModelEntrySchema, ModelScenarioSchema, RetrievalConfigSchema, RetrievalSourceSchema, SummaryConfigSchema, TokenConfigSchema, resolveModelApi, resolveModelEntry } from './chunk-
|
|
2
|
-
export { collectStreamContent, createA2AClient, createAIClient, parseSSE } from './chunk-
|
|
1
|
+
export { A2AConfigSchema, A2ASkillConfigSchema, AIConfigSchema, AUDIO_WS_PATH, ApiTypeSchema, AudioConfigSchema, AudioModelEntrySchema, AudioProviderSchema, CompressConfigSchema, CompressionStrategySchema, EmbeddingConfigSchema, EntityTypeSchema, FileConfigSchema, HaiAIError, KnowledgeConfigSchema, LLMConfigSchema, MCPConfigSchema, MCPServerCapabilitiesSchema, MCPServerConfigSchema, MemoryConfigSchema, MemoryTypeSchema, ModelEntrySchema, ModelScenarioSchema, RetrievalConfigSchema, RetrievalSourceSchema, SummaryConfigSchema, TokenConfigSchema, resolveAudioModel, resolveModelApi, resolveModelEntry } from './chunk-URXCNMJW.js';
|
|
2
|
+
export { collectStreamContent, createA2AClient, createAIClient, createAudioClient, createUnconfiguredAudioClient, parseSSE } from './chunk-AXIZBZMV.js';
|
|
3
3
|
//# sourceMappingURL=browser.js.map
|
|
4
4
|
//# sourceMappingURL=browser.js.map
|