bilisum 1.13.2 → 1.13.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +4 -4
- package/bin/bilisum.js +157 -76
- package/package.json +7 -2
- package/runtime/VERSION +1 -0
- package/runtime/apps/service/pyproject.toml +29 -0
- package/runtime/apps/service/src/video_sum_service/__init__.py +1 -0
- package/runtime/apps/service/src/video_sum_service/__main__.py +5 -0
- package/runtime/apps/service/src/video_sum_service/app.py +480 -0
- package/runtime/apps/service/src/video_sum_service/context.py +24 -0
- package/runtime/apps/service/src/video_sum_service/integrations.py +264 -0
- package/runtime/apps/service/src/video_sum_service/knowledge/__init__.py +5 -0
- package/runtime/apps/service/src/video_sum_service/knowledge/index_service.py +408 -0
- package/runtime/apps/service/src/video_sum_service/knowledge/local_llm.py +319 -0
- package/runtime/apps/service/src/video_sum_service/knowledge/rag_service.py +492 -0
- package/runtime/apps/service/src/video_sum_service/knowledge/tag_service.py +242 -0
- package/runtime/apps/service/src/video_sum_service/main.py +31 -0
- package/runtime/apps/service/src/video_sum_service/repository.py +942 -0
- package/runtime/apps/service/src/video_sum_service/routers/__init__.py +1 -0
- package/runtime/apps/service/src/video_sum_service/routers/knowledge.py +272 -0
- package/runtime/apps/service/src/video_sum_service/routers/system.py +280 -0
- package/runtime/apps/service/src/video_sum_service/routers/tasks.py +287 -0
- package/runtime/apps/service/src/video_sum_service/routers/videos.py +766 -0
- package/runtime/apps/service/src/video_sum_service/runtime_support.py +1007 -0
- package/runtime/apps/service/src/video_sum_service/schemas.py +418 -0
- package/runtime/apps/service/src/video_sum_service/settings_manager.py +130 -0
- package/runtime/apps/service/src/video_sum_service/task_artifacts.py +102 -0
- package/runtime/apps/service/src/video_sum_service/task_exports.py +108 -0
- package/runtime/apps/service/src/video_sum_service/transcribe_worker.py +9 -0
- package/runtime/apps/service/src/video_sum_service/video_assets.py +623 -0
- package/runtime/apps/service/src/video_sum_service/worker.py +381 -0
- package/runtime/apps/web/static/apple-touch-icon.png +0 -0
- package/runtime/apps/web/static/assets/KaTeX_AMS-Regular-BQhdFMY1.woff2 +0 -0
- package/runtime/apps/web/static/assets/KaTeX_AMS-Regular-DMm9YOAa.woff +0 -0
- package/runtime/apps/web/static/assets/KaTeX_AMS-Regular-DRggAlZN.ttf +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Caligraphic-Bold-ATXxdsX0.ttf +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Caligraphic-Bold-BEiXGLvX.woff +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Caligraphic-Bold-Dq_IR9rO.woff2 +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Caligraphic-Regular-CTRA-rTL.woff +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Caligraphic-Regular-Di6jR-x-.woff2 +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Caligraphic-Regular-wX97UBjC.ttf +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Fraktur-Bold-BdnERNNW.ttf +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Fraktur-Bold-BsDP51OF.woff +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Fraktur-Bold-CL6g_b3V.woff2 +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Fraktur-Regular-CB_wures.ttf +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Fraktur-Regular-CTYiF6lA.woff2 +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Fraktur-Regular-Dxdc4cR9.woff +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Main-Bold-Cx986IdX.woff2 +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Main-Bold-Jm3AIy58.woff +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Main-Bold-waoOVXN0.ttf +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Main-BoldItalic-DxDJ3AOS.woff2 +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Main-BoldItalic-DzxPMmG6.ttf +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Main-BoldItalic-SpSLRI95.woff +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Main-Italic-3WenGoN9.ttf +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Main-Italic-BMLOBm91.woff +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Main-Italic-NWA7e6Wa.woff2 +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Main-Regular-B22Nviop.woff2 +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Main-Regular-Dr94JaBh.woff +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Main-Regular-ypZvNtVU.ttf +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Math-BoldItalic-B3XSjfu4.ttf +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Math-BoldItalic-CZnvNsCZ.woff2 +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Math-BoldItalic-iY-2wyZ7.woff +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Math-Italic-DA0__PXp.woff +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Math-Italic-flOr_0UB.ttf +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Math-Italic-t53AETM-.woff2 +0 -0
- package/runtime/apps/web/static/assets/KaTeX_SansSerif-Bold-CFMepnvq.ttf +0 -0
- package/runtime/apps/web/static/assets/KaTeX_SansSerif-Bold-D1sUS0GD.woff2 +0 -0
- package/runtime/apps/web/static/assets/KaTeX_SansSerif-Bold-DbIhKOiC.woff +0 -0
- package/runtime/apps/web/static/assets/KaTeX_SansSerif-Italic-C3H0VqGB.woff2 +0 -0
- package/runtime/apps/web/static/assets/KaTeX_SansSerif-Italic-DN2j7dab.woff +0 -0
- package/runtime/apps/web/static/assets/KaTeX_SansSerif-Italic-YYjJ1zSn.ttf +0 -0
- package/runtime/apps/web/static/assets/KaTeX_SansSerif-Regular-BNo7hRIc.ttf +0 -0
- package/runtime/apps/web/static/assets/KaTeX_SansSerif-Regular-CS6fqUqJ.woff +0 -0
- package/runtime/apps/web/static/assets/KaTeX_SansSerif-Regular-DDBCnlJ7.woff2 +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Script-Regular-C5JkGWo-.ttf +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Script-Regular-D3wIWfF6.woff2 +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Script-Regular-D5yQViql.woff +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Size1-Regular-C195tn64.woff +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Size1-Regular-Dbsnue_I.ttf +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Size1-Regular-mCD8mA8B.woff2 +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Size2-Regular-B7gKUWhC.ttf +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Size2-Regular-Dy4dx90m.woff2 +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Size2-Regular-oD1tc_U0.woff +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Size3-Regular-CTq5MqoE.woff +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Size3-Regular-DgpXs0kz.ttf +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Size4-Regular-BF-4gkZK.woff +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Size4-Regular-DWFBv043.ttf +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Size4-Regular-Dl5lxZxV.woff2 +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Typewriter-Regular-C0xS9mPB.woff +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Typewriter-Regular-CO6r4hn1.woff2 +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Typewriter-Regular-D3Ib7_Hf.ttf +0 -0
- package/runtime/apps/web/static/assets/icons/icon-180.png +0 -0
- package/runtime/apps/web/static/assets/icons/icon-512.png +0 -0
- package/runtime/apps/web/static/assets/icons/icon.svg +17 -0
- package/runtime/apps/web/static/assets/index-Cj5PVjQg.js +388 -0
- package/runtime/apps/web/static/assets/index-DU2t4_7s.css +1 -0
- package/runtime/apps/web/static/favicon-32x32.png +0 -0
- package/runtime/apps/web/static/favicon.ico +0 -0
- package/runtime/apps/web/static/favicon.svg +17 -0
- package/runtime/apps/web/static/index.html +19 -0
- package/runtime/apps/web/static/js/api.js +123 -0
- package/runtime/apps/web/static/js/main.js +932 -0
- package/runtime/apps/web/static/js/state.js +28 -0
- package/runtime/apps/web/static/js/utils.js +81 -0
- package/runtime/apps/web/static/js/views/home.js +604 -0
- package/runtime/apps/web/static/js/views/settings.js +477 -0
- package/runtime/apps/web/static/styles.css +2206 -0
- package/runtime/packages/core/pyproject.toml +18 -0
- package/runtime/packages/core/src/video_sum_core/__init__.py +1 -0
- package/runtime/packages/core/src/video_sum_core/errors.py +22 -0
- package/runtime/packages/core/src/video_sum_core/markdown_exports.py +154 -0
- package/runtime/packages/core/src/video_sum_core/models/__init__.py +1 -0
- package/runtime/packages/core/src/video_sum_core/models/tasks.py +70 -0
- package/runtime/packages/core/src/video_sum_core/pipeline/__init__.py +1 -0
- package/runtime/packages/core/src/video_sum_core/pipeline/base.py +46 -0
- package/runtime/packages/core/src/video_sum_core/pipeline/real.py +3467 -0
- package/runtime/packages/core/src/video_sum_core/transcribe_subprocess.py +173 -0
- package/runtime/packages/core/src/video_sum_core/utils.py +127 -0
- package/runtime/packages/infra/pyproject.toml +16 -0
- package/runtime/packages/infra/src/video_sum_infra/__init__.py +1 -0
- package/runtime/packages/infra/src/video_sum_infra/app.py +39 -0
- package/runtime/packages/infra/src/video_sum_infra/config.py +393 -0
- package/runtime/packages/infra/src/video_sum_infra/db.py +34 -0
- package/runtime/packages/infra/src/video_sum_infra/logging.py +54 -0
- package/runtime/packages/infra/src/video_sum_infra/paths.py +6 -0
- package/runtime/packages/infra/src/video_sum_infra/runtime.py +485 -0
|
@@ -0,0 +1,319 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import json
|
|
4
|
+
import logging
|
|
5
|
+
import time
|
|
6
|
+
from collections.abc import Callable, Iterator
|
|
7
|
+
from dataclasses import dataclass
|
|
8
|
+
|
|
9
|
+
import httpx
|
|
10
|
+
from fastapi import HTTPException
|
|
11
|
+
|
|
12
|
+
from video_sum_infra.config import ServiceSettings
|
|
13
|
+
from video_sum_service.integrations import extract_http_error_detail, extract_llm_message_content
|
|
14
|
+
|
|
15
|
+
logger = logging.getLogger(__name__)
|
|
16
|
+
|
|
17
|
+
KNOWLEDGE_LLM_TIMEOUT = httpx.Timeout(connect=15.0, read=45.0, write=30.0, pool=30.0)
|
|
18
|
+
KNOWLEDGE_LLM_STREAM_TIMEOUT = httpx.Timeout(connect=15.0, read=12.0, write=30.0, pool=30.0)
|
|
19
|
+
KNOWLEDGE_LLM_FIRST_CONTENT_TIMEOUT_SECONDS = 18.0
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
@dataclass(frozen=True)
|
|
23
|
+
class KnowledgeLlmStreamEvent:
|
|
24
|
+
kind: str
|
|
25
|
+
delta: str
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def resolve_knowledge_llm_settings(settings: ServiceSettings) -> tuple[bool, str, str, str]:
|
|
29
|
+
mode = str(getattr(settings, "knowledge_llm_mode", "same_as_main") or "same_as_main").strip().lower()
|
|
30
|
+
if mode == "custom":
|
|
31
|
+
enabled = bool(getattr(settings, "knowledge_llm_enabled", False))
|
|
32
|
+
base_url = str(getattr(settings, "knowledge_llm_base_url", "") or "").strip().rstrip("/")
|
|
33
|
+
model = str(getattr(settings, "knowledge_llm_model", "") or "").strip()
|
|
34
|
+
api_key = str(getattr(settings, "knowledge_llm_api_key", "") or "").strip()
|
|
35
|
+
return enabled, base_url, model, api_key
|
|
36
|
+
|
|
37
|
+
enabled = bool(getattr(settings, "llm_enabled", False))
|
|
38
|
+
base_url = str(getattr(settings, "llm_base_url", "") or "").strip().rstrip("/")
|
|
39
|
+
model = str(getattr(settings, "llm_model", "") or "").strip()
|
|
40
|
+
api_key = str(getattr(settings, "llm_api_key", "") or "").strip()
|
|
41
|
+
return enabled, base_url, model, api_key
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def knowledge_llm_available(settings: ServiceSettings) -> bool:
|
|
45
|
+
enabled, base_url, model, _api_key = resolve_knowledge_llm_settings(settings)
|
|
46
|
+
return bool(enabled and base_url and model)
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def ensure_knowledge_llm_settings(settings: ServiceSettings) -> tuple[str, str, str]:
|
|
50
|
+
enabled, base_url, model, api_key = resolve_knowledge_llm_settings(settings)
|
|
51
|
+
if not enabled:
|
|
52
|
+
raise HTTPException(status_code=400, detail="知识库问答和自动打标需要先启用知识库 LLM。")
|
|
53
|
+
if not base_url or not model:
|
|
54
|
+
raise HTTPException(status_code=400, detail="知识库问答和自动打标需要先填写知识库 LLM 的地址和模型名。")
|
|
55
|
+
return base_url, model, api_key
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def chat_knowledge_llm(
|
|
59
|
+
settings: ServiceSettings,
|
|
60
|
+
*,
|
|
61
|
+
system_prompt: str,
|
|
62
|
+
user_prompt: str,
|
|
63
|
+
max_tokens: int = 800,
|
|
64
|
+
temperature: float = 0.2,
|
|
65
|
+
require_json: bool = False,
|
|
66
|
+
) -> tuple[str, dict[str, object] | None]:
|
|
67
|
+
base_url, model, api_key = ensure_knowledge_llm_settings(settings)
|
|
68
|
+
headers = {"Content-Type": "application/json"}
|
|
69
|
+
if api_key:
|
|
70
|
+
headers["Authorization"] = f"Bearer {api_key}"
|
|
71
|
+
|
|
72
|
+
payload: dict[str, object] = {
|
|
73
|
+
"model": model,
|
|
74
|
+
"messages": [
|
|
75
|
+
{"role": "system", "content": system_prompt},
|
|
76
|
+
{"role": "user", "content": user_prompt},
|
|
77
|
+
],
|
|
78
|
+
"temperature": temperature,
|
|
79
|
+
"max_tokens": max_tokens,
|
|
80
|
+
}
|
|
81
|
+
if require_json:
|
|
82
|
+
payload["response_format"] = {"type": "json_object"}
|
|
83
|
+
payload["enable_thinking"] = False
|
|
84
|
+
payload["chat_template_kwargs"] = {"enable_thinking": False}
|
|
85
|
+
|
|
86
|
+
started_at = time.monotonic()
|
|
87
|
+
logger.info("knowledge llm request start mode=chat base_url=%s model=%s max_tokens=%s", base_url, model, max_tokens)
|
|
88
|
+
try:
|
|
89
|
+
with httpx.Client(timeout=KNOWLEDGE_LLM_TIMEOUT, follow_redirects=True) as client:
|
|
90
|
+
response = client.post(f"{base_url}/chat/completions", headers=headers, json=payload)
|
|
91
|
+
except httpx.ReadTimeout as exc:
|
|
92
|
+
elapsed_ms = int((time.monotonic() - started_at) * 1000)
|
|
93
|
+
logger.warning("knowledge llm request timeout mode=chat model=%s elapsed_ms=%s", model, elapsed_ms)
|
|
94
|
+
raise HTTPException(
|
|
95
|
+
status_code=504,
|
|
96
|
+
detail="知识库 LLM 响应超时:模型返回过慢,请稍后重试,或换更快的模型 / 减少上下文。",
|
|
97
|
+
) from exc
|
|
98
|
+
except httpx.HTTPError as exc:
|
|
99
|
+
elapsed_ms = int((time.monotonic() - started_at) * 1000)
|
|
100
|
+
logger.warning("knowledge llm request failed mode=chat model=%s elapsed_ms=%s error=%s", model, elapsed_ms, exc)
|
|
101
|
+
raise HTTPException(status_code=502, detail=f"知识库 LLM 连接失败:{exc}") from exc
|
|
102
|
+
|
|
103
|
+
elapsed_ms = int((time.monotonic() - started_at) * 1000)
|
|
104
|
+
logger.info(
|
|
105
|
+
"knowledge llm response received mode=chat model=%s status_code=%s elapsed_ms=%s",
|
|
106
|
+
model,
|
|
107
|
+
response.status_code,
|
|
108
|
+
elapsed_ms,
|
|
109
|
+
)
|
|
110
|
+
|
|
111
|
+
if response.status_code >= 400:
|
|
112
|
+
detail = extract_http_error_detail(response)
|
|
113
|
+
raise HTTPException(status_code=response.status_code, detail=f"知识库 LLM 调用失败:{detail}")
|
|
114
|
+
|
|
115
|
+
try:
|
|
116
|
+
body = response.json()
|
|
117
|
+
except ValueError:
|
|
118
|
+
body = None
|
|
119
|
+
|
|
120
|
+
content = extract_llm_message_content(body)
|
|
121
|
+
if not content:
|
|
122
|
+
raise HTTPException(status_code=502, detail="知识库 LLM 没有返回可读取内容。")
|
|
123
|
+
return content, body if isinstance(body, dict) else None
|
|
124
|
+
|
|
125
|
+
|
|
126
|
+
def _extract_stream_reasoning_delta(payload: dict[str, object]) -> str:
|
|
127
|
+
message = payload.get("message")
|
|
128
|
+
if isinstance(message, dict):
|
|
129
|
+
reasoning = message.get("reasoning_content")
|
|
130
|
+
if isinstance(reasoning, str):
|
|
131
|
+
return reasoning
|
|
132
|
+
|
|
133
|
+
choices = payload.get("choices")
|
|
134
|
+
if not isinstance(choices, list) or not choices:
|
|
135
|
+
return ""
|
|
136
|
+
first = choices[0]
|
|
137
|
+
if not isinstance(first, dict):
|
|
138
|
+
return ""
|
|
139
|
+
delta = first.get("delta")
|
|
140
|
+
if isinstance(delta, dict):
|
|
141
|
+
reasoning = delta.get("reasoning_content")
|
|
142
|
+
if isinstance(reasoning, str):
|
|
143
|
+
return reasoning
|
|
144
|
+
message = first.get("message")
|
|
145
|
+
if isinstance(message, dict):
|
|
146
|
+
reasoning = message.get("reasoning_content")
|
|
147
|
+
if isinstance(reasoning, str):
|
|
148
|
+
return reasoning
|
|
149
|
+
return ""
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
def _extract_stream_delta(payload: dict[str, object]) -> str:
|
|
153
|
+
message = payload.get("message")
|
|
154
|
+
if isinstance(message, dict):
|
|
155
|
+
content = message.get("content")
|
|
156
|
+
if isinstance(content, str):
|
|
157
|
+
return content
|
|
158
|
+
|
|
159
|
+
for key in ("response", "text", "content"):
|
|
160
|
+
value = payload.get(key)
|
|
161
|
+
if isinstance(value, str):
|
|
162
|
+
return value
|
|
163
|
+
|
|
164
|
+
choices = payload.get("choices")
|
|
165
|
+
if not isinstance(choices, list) or not choices:
|
|
166
|
+
return ""
|
|
167
|
+
first = choices[0]
|
|
168
|
+
if not isinstance(first, dict):
|
|
169
|
+
return ""
|
|
170
|
+
delta = first.get("delta")
|
|
171
|
+
if not isinstance(delta, dict):
|
|
172
|
+
return ""
|
|
173
|
+
content = delta.get("content")
|
|
174
|
+
if isinstance(content, str):
|
|
175
|
+
return content
|
|
176
|
+
if isinstance(content, list):
|
|
177
|
+
text_parts: list[str] = []
|
|
178
|
+
for item in content:
|
|
179
|
+
if not isinstance(item, dict):
|
|
180
|
+
continue
|
|
181
|
+
text = item.get("text")
|
|
182
|
+
if isinstance(text, str):
|
|
183
|
+
text_parts.append(text)
|
|
184
|
+
return "".join(text_parts)
|
|
185
|
+
return ""
|
|
186
|
+
|
|
187
|
+
|
|
188
|
+
def stream_knowledge_llm(
|
|
189
|
+
settings: ServiceSettings,
|
|
190
|
+
*,
|
|
191
|
+
system_prompt: str,
|
|
192
|
+
user_prompt: str,
|
|
193
|
+
max_tokens: int = 800,
|
|
194
|
+
temperature: float = 0.2,
|
|
195
|
+
should_cancel: Callable[[], bool] | None = None,
|
|
196
|
+
) -> Iterator[KnowledgeLlmStreamEvent]:
|
|
197
|
+
base_url, model, api_key = ensure_knowledge_llm_settings(settings)
|
|
198
|
+
headers = {"Content-Type": "application/json"}
|
|
199
|
+
if api_key:
|
|
200
|
+
headers["Authorization"] = f"Bearer {api_key}"
|
|
201
|
+
|
|
202
|
+
payload: dict[str, object] = {
|
|
203
|
+
"model": model,
|
|
204
|
+
"messages": [
|
|
205
|
+
{"role": "system", "content": system_prompt},
|
|
206
|
+
{"role": "user", "content": user_prompt},
|
|
207
|
+
],
|
|
208
|
+
"temperature": temperature,
|
|
209
|
+
"max_tokens": max_tokens,
|
|
210
|
+
"stream": True,
|
|
211
|
+
}
|
|
212
|
+
should_stop = should_cancel or (lambda: False)
|
|
213
|
+
|
|
214
|
+
started_at = time.monotonic()
|
|
215
|
+
opened_at: float | None = None
|
|
216
|
+
first_delta_logged = False
|
|
217
|
+
first_activity_logged = False
|
|
218
|
+
delta_count = 0
|
|
219
|
+
reasoning_delta_count = 0
|
|
220
|
+
logger.info("knowledge llm request start mode=stream base_url=%s model=%s max_tokens=%s", base_url, model, max_tokens)
|
|
221
|
+
try:
|
|
222
|
+
with httpx.Client(timeout=KNOWLEDGE_LLM_STREAM_TIMEOUT, follow_redirects=True) as client:
|
|
223
|
+
with client.stream("POST", f"{base_url}/chat/completions", headers=headers, json=payload) as response:
|
|
224
|
+
opened_at = time.monotonic()
|
|
225
|
+
elapsed_ms = int((time.monotonic() - started_at) * 1000)
|
|
226
|
+
logger.info(
|
|
227
|
+
"knowledge llm stream opened model=%s status_code=%s elapsed_ms=%s",
|
|
228
|
+
model,
|
|
229
|
+
response.status_code,
|
|
230
|
+
elapsed_ms,
|
|
231
|
+
)
|
|
232
|
+
if response.status_code >= 400:
|
|
233
|
+
detail = extract_http_error_detail(response)
|
|
234
|
+
raise HTTPException(status_code=response.status_code, detail=f"知识库 LLM 调用失败:{detail}")
|
|
235
|
+
|
|
236
|
+
for raw_line in response.iter_lines():
|
|
237
|
+
if should_stop():
|
|
238
|
+
return
|
|
239
|
+
if not first_activity_logged and opened_at is not None:
|
|
240
|
+
wait_seconds = time.monotonic() - opened_at
|
|
241
|
+
if wait_seconds > KNOWLEDGE_LLM_FIRST_CONTENT_TIMEOUT_SECONDS:
|
|
242
|
+
elapsed_ms = int((time.monotonic() - started_at) * 1000)
|
|
243
|
+
logger.warning(
|
|
244
|
+
"knowledge llm first activity timeout model=%s elapsed_ms=%s wait_seconds=%.1f",
|
|
245
|
+
model,
|
|
246
|
+
elapsed_ms,
|
|
247
|
+
wait_seconds,
|
|
248
|
+
)
|
|
249
|
+
raise HTTPException(
|
|
250
|
+
status_code=504,
|
|
251
|
+
detail="知识库 LLM 已建立连接但迟迟没有输出内容:当前模型首包过慢,请稍后重试或换更快的知识库模型。",
|
|
252
|
+
)
|
|
253
|
+
line = str(raw_line or "").strip()
|
|
254
|
+
if not line or line.startswith(":"):
|
|
255
|
+
continue
|
|
256
|
+
if not line.startswith("data:"):
|
|
257
|
+
continue
|
|
258
|
+
chunk = line[5:].strip()
|
|
259
|
+
if not chunk or chunk == "[DONE]":
|
|
260
|
+
continue
|
|
261
|
+
try:
|
|
262
|
+
body = json.loads(chunk)
|
|
263
|
+
except json.JSONDecodeError:
|
|
264
|
+
continue
|
|
265
|
+
if not isinstance(body, dict):
|
|
266
|
+
continue
|
|
267
|
+
reasoning_delta = _extract_stream_reasoning_delta(body)
|
|
268
|
+
if reasoning_delta:
|
|
269
|
+
reasoning_delta_count += 1
|
|
270
|
+
if not first_activity_logged:
|
|
271
|
+
first_activity_logged = True
|
|
272
|
+
elapsed_ms = int((time.monotonic() - started_at) * 1000)
|
|
273
|
+
logger.info("knowledge llm first reasoning delta model=%s elapsed_ms=%s", model, elapsed_ms)
|
|
274
|
+
yield KnowledgeLlmStreamEvent(kind="reasoning", delta=reasoning_delta)
|
|
275
|
+
delta = _extract_stream_delta(body)
|
|
276
|
+
if delta:
|
|
277
|
+
delta_count += 1
|
|
278
|
+
first_activity_logged = True
|
|
279
|
+
if not first_delta_logged:
|
|
280
|
+
first_delta_logged = True
|
|
281
|
+
elapsed_ms = int((time.monotonic() - started_at) * 1000)
|
|
282
|
+
logger.info("knowledge llm first stream delta model=%s elapsed_ms=%s", model, elapsed_ms)
|
|
283
|
+
yield KnowledgeLlmStreamEvent(kind="content", delta=delta)
|
|
284
|
+
except httpx.ReadTimeout as exc:
|
|
285
|
+
elapsed_ms = int((time.monotonic() - started_at) * 1000)
|
|
286
|
+
logger.warning(
|
|
287
|
+
"knowledge llm request timeout mode=stream model=%s elapsed_ms=%s delta_count=%s reasoning_delta_count=%s",
|
|
288
|
+
model,
|
|
289
|
+
elapsed_ms,
|
|
290
|
+
delta_count,
|
|
291
|
+
reasoning_delta_count,
|
|
292
|
+
)
|
|
293
|
+
raise HTTPException(
|
|
294
|
+
status_code=504,
|
|
295
|
+
detail="知识库 LLM 响应超时:模型返回过慢,请稍后重试,或换更快的模型 / 减少上下文。",
|
|
296
|
+
) from exc
|
|
297
|
+
except httpx.HTTPError as exc:
|
|
298
|
+
elapsed_ms = int((time.monotonic() - started_at) * 1000)
|
|
299
|
+
logger.warning("knowledge llm request failed mode=stream model=%s elapsed_ms=%s error=%s", model, elapsed_ms, exc)
|
|
300
|
+
raise HTTPException(status_code=502, detail=f"知识库 LLM 连接失败:{exc}") from exc
|
|
301
|
+
finally:
|
|
302
|
+
elapsed_ms = int((time.monotonic() - started_at) * 1000)
|
|
303
|
+
logger.info(
|
|
304
|
+
"knowledge llm stream finished model=%s elapsed_ms=%s delta_count=%s reasoning_delta_count=%s",
|
|
305
|
+
model,
|
|
306
|
+
elapsed_ms,
|
|
307
|
+
delta_count,
|
|
308
|
+
reasoning_delta_count,
|
|
309
|
+
)
|
|
310
|
+
|
|
311
|
+
|
|
312
|
+
def parse_json_payload(text: str) -> dict[str, object]:
|
|
313
|
+
try:
|
|
314
|
+
payload = json.loads(text)
|
|
315
|
+
except json.JSONDecodeError as exc:
|
|
316
|
+
raise HTTPException(status_code=502, detail=f"知识库 LLM 没有返回合法 JSON:{exc.msg}") from exc
|
|
317
|
+
if not isinstance(payload, dict):
|
|
318
|
+
raise HTTPException(status_code=502, detail="知识库 LLM 返回的 JSON 结构不符合预期。")
|
|
319
|
+
return payload
|