bilisum 1.13.2 → 1.13.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (125) hide show
  1. package/README.md +4 -4
  2. package/bin/bilisum.js +157 -76
  3. package/package.json +7 -2
  4. package/runtime/VERSION +1 -0
  5. package/runtime/apps/service/pyproject.toml +29 -0
  6. package/runtime/apps/service/src/video_sum_service/__init__.py +1 -0
  7. package/runtime/apps/service/src/video_sum_service/__main__.py +5 -0
  8. package/runtime/apps/service/src/video_sum_service/app.py +480 -0
  9. package/runtime/apps/service/src/video_sum_service/context.py +24 -0
  10. package/runtime/apps/service/src/video_sum_service/integrations.py +264 -0
  11. package/runtime/apps/service/src/video_sum_service/knowledge/__init__.py +5 -0
  12. package/runtime/apps/service/src/video_sum_service/knowledge/index_service.py +408 -0
  13. package/runtime/apps/service/src/video_sum_service/knowledge/local_llm.py +319 -0
  14. package/runtime/apps/service/src/video_sum_service/knowledge/rag_service.py +492 -0
  15. package/runtime/apps/service/src/video_sum_service/knowledge/tag_service.py +242 -0
  16. package/runtime/apps/service/src/video_sum_service/main.py +31 -0
  17. package/runtime/apps/service/src/video_sum_service/repository.py +942 -0
  18. package/runtime/apps/service/src/video_sum_service/routers/__init__.py +1 -0
  19. package/runtime/apps/service/src/video_sum_service/routers/knowledge.py +272 -0
  20. package/runtime/apps/service/src/video_sum_service/routers/system.py +280 -0
  21. package/runtime/apps/service/src/video_sum_service/routers/tasks.py +287 -0
  22. package/runtime/apps/service/src/video_sum_service/routers/videos.py +766 -0
  23. package/runtime/apps/service/src/video_sum_service/runtime_support.py +1007 -0
  24. package/runtime/apps/service/src/video_sum_service/schemas.py +418 -0
  25. package/runtime/apps/service/src/video_sum_service/settings_manager.py +130 -0
  26. package/runtime/apps/service/src/video_sum_service/task_artifacts.py +102 -0
  27. package/runtime/apps/service/src/video_sum_service/task_exports.py +108 -0
  28. package/runtime/apps/service/src/video_sum_service/transcribe_worker.py +9 -0
  29. package/runtime/apps/service/src/video_sum_service/video_assets.py +623 -0
  30. package/runtime/apps/service/src/video_sum_service/worker.py +381 -0
  31. package/runtime/apps/web/static/apple-touch-icon.png +0 -0
  32. package/runtime/apps/web/static/assets/KaTeX_AMS-Regular-BQhdFMY1.woff2 +0 -0
  33. package/runtime/apps/web/static/assets/KaTeX_AMS-Regular-DMm9YOAa.woff +0 -0
  34. package/runtime/apps/web/static/assets/KaTeX_AMS-Regular-DRggAlZN.ttf +0 -0
  35. package/runtime/apps/web/static/assets/KaTeX_Caligraphic-Bold-ATXxdsX0.ttf +0 -0
  36. package/runtime/apps/web/static/assets/KaTeX_Caligraphic-Bold-BEiXGLvX.woff +0 -0
  37. package/runtime/apps/web/static/assets/KaTeX_Caligraphic-Bold-Dq_IR9rO.woff2 +0 -0
  38. package/runtime/apps/web/static/assets/KaTeX_Caligraphic-Regular-CTRA-rTL.woff +0 -0
  39. package/runtime/apps/web/static/assets/KaTeX_Caligraphic-Regular-Di6jR-x-.woff2 +0 -0
  40. package/runtime/apps/web/static/assets/KaTeX_Caligraphic-Regular-wX97UBjC.ttf +0 -0
  41. package/runtime/apps/web/static/assets/KaTeX_Fraktur-Bold-BdnERNNW.ttf +0 -0
  42. package/runtime/apps/web/static/assets/KaTeX_Fraktur-Bold-BsDP51OF.woff +0 -0
  43. package/runtime/apps/web/static/assets/KaTeX_Fraktur-Bold-CL6g_b3V.woff2 +0 -0
  44. package/runtime/apps/web/static/assets/KaTeX_Fraktur-Regular-CB_wures.ttf +0 -0
  45. package/runtime/apps/web/static/assets/KaTeX_Fraktur-Regular-CTYiF6lA.woff2 +0 -0
  46. package/runtime/apps/web/static/assets/KaTeX_Fraktur-Regular-Dxdc4cR9.woff +0 -0
  47. package/runtime/apps/web/static/assets/KaTeX_Main-Bold-Cx986IdX.woff2 +0 -0
  48. package/runtime/apps/web/static/assets/KaTeX_Main-Bold-Jm3AIy58.woff +0 -0
  49. package/runtime/apps/web/static/assets/KaTeX_Main-Bold-waoOVXN0.ttf +0 -0
  50. package/runtime/apps/web/static/assets/KaTeX_Main-BoldItalic-DxDJ3AOS.woff2 +0 -0
  51. package/runtime/apps/web/static/assets/KaTeX_Main-BoldItalic-DzxPMmG6.ttf +0 -0
  52. package/runtime/apps/web/static/assets/KaTeX_Main-BoldItalic-SpSLRI95.woff +0 -0
  53. package/runtime/apps/web/static/assets/KaTeX_Main-Italic-3WenGoN9.ttf +0 -0
  54. package/runtime/apps/web/static/assets/KaTeX_Main-Italic-BMLOBm91.woff +0 -0
  55. package/runtime/apps/web/static/assets/KaTeX_Main-Italic-NWA7e6Wa.woff2 +0 -0
  56. package/runtime/apps/web/static/assets/KaTeX_Main-Regular-B22Nviop.woff2 +0 -0
  57. package/runtime/apps/web/static/assets/KaTeX_Main-Regular-Dr94JaBh.woff +0 -0
  58. package/runtime/apps/web/static/assets/KaTeX_Main-Regular-ypZvNtVU.ttf +0 -0
  59. package/runtime/apps/web/static/assets/KaTeX_Math-BoldItalic-B3XSjfu4.ttf +0 -0
  60. package/runtime/apps/web/static/assets/KaTeX_Math-BoldItalic-CZnvNsCZ.woff2 +0 -0
  61. package/runtime/apps/web/static/assets/KaTeX_Math-BoldItalic-iY-2wyZ7.woff +0 -0
  62. package/runtime/apps/web/static/assets/KaTeX_Math-Italic-DA0__PXp.woff +0 -0
  63. package/runtime/apps/web/static/assets/KaTeX_Math-Italic-flOr_0UB.ttf +0 -0
  64. package/runtime/apps/web/static/assets/KaTeX_Math-Italic-t53AETM-.woff2 +0 -0
  65. package/runtime/apps/web/static/assets/KaTeX_SansSerif-Bold-CFMepnvq.ttf +0 -0
  66. package/runtime/apps/web/static/assets/KaTeX_SansSerif-Bold-D1sUS0GD.woff2 +0 -0
  67. package/runtime/apps/web/static/assets/KaTeX_SansSerif-Bold-DbIhKOiC.woff +0 -0
  68. package/runtime/apps/web/static/assets/KaTeX_SansSerif-Italic-C3H0VqGB.woff2 +0 -0
  69. package/runtime/apps/web/static/assets/KaTeX_SansSerif-Italic-DN2j7dab.woff +0 -0
  70. package/runtime/apps/web/static/assets/KaTeX_SansSerif-Italic-YYjJ1zSn.ttf +0 -0
  71. package/runtime/apps/web/static/assets/KaTeX_SansSerif-Regular-BNo7hRIc.ttf +0 -0
  72. package/runtime/apps/web/static/assets/KaTeX_SansSerif-Regular-CS6fqUqJ.woff +0 -0
  73. package/runtime/apps/web/static/assets/KaTeX_SansSerif-Regular-DDBCnlJ7.woff2 +0 -0
  74. package/runtime/apps/web/static/assets/KaTeX_Script-Regular-C5JkGWo-.ttf +0 -0
  75. package/runtime/apps/web/static/assets/KaTeX_Script-Regular-D3wIWfF6.woff2 +0 -0
  76. package/runtime/apps/web/static/assets/KaTeX_Script-Regular-D5yQViql.woff +0 -0
  77. package/runtime/apps/web/static/assets/KaTeX_Size1-Regular-C195tn64.woff +0 -0
  78. package/runtime/apps/web/static/assets/KaTeX_Size1-Regular-Dbsnue_I.ttf +0 -0
  79. package/runtime/apps/web/static/assets/KaTeX_Size1-Regular-mCD8mA8B.woff2 +0 -0
  80. package/runtime/apps/web/static/assets/KaTeX_Size2-Regular-B7gKUWhC.ttf +0 -0
  81. package/runtime/apps/web/static/assets/KaTeX_Size2-Regular-Dy4dx90m.woff2 +0 -0
  82. package/runtime/apps/web/static/assets/KaTeX_Size2-Regular-oD1tc_U0.woff +0 -0
  83. package/runtime/apps/web/static/assets/KaTeX_Size3-Regular-CTq5MqoE.woff +0 -0
  84. package/runtime/apps/web/static/assets/KaTeX_Size3-Regular-DgpXs0kz.ttf +0 -0
  85. package/runtime/apps/web/static/assets/KaTeX_Size4-Regular-BF-4gkZK.woff +0 -0
  86. package/runtime/apps/web/static/assets/KaTeX_Size4-Regular-DWFBv043.ttf +0 -0
  87. package/runtime/apps/web/static/assets/KaTeX_Size4-Regular-Dl5lxZxV.woff2 +0 -0
  88. package/runtime/apps/web/static/assets/KaTeX_Typewriter-Regular-C0xS9mPB.woff +0 -0
  89. package/runtime/apps/web/static/assets/KaTeX_Typewriter-Regular-CO6r4hn1.woff2 +0 -0
  90. package/runtime/apps/web/static/assets/KaTeX_Typewriter-Regular-D3Ib7_Hf.ttf +0 -0
  91. package/runtime/apps/web/static/assets/icons/icon-180.png +0 -0
  92. package/runtime/apps/web/static/assets/icons/icon-512.png +0 -0
  93. package/runtime/apps/web/static/assets/icons/icon.svg +17 -0
  94. package/runtime/apps/web/static/assets/index-Cj5PVjQg.js +388 -0
  95. package/runtime/apps/web/static/assets/index-DU2t4_7s.css +1 -0
  96. package/runtime/apps/web/static/favicon-32x32.png +0 -0
  97. package/runtime/apps/web/static/favicon.ico +0 -0
  98. package/runtime/apps/web/static/favicon.svg +17 -0
  99. package/runtime/apps/web/static/index.html +19 -0
  100. package/runtime/apps/web/static/js/api.js +123 -0
  101. package/runtime/apps/web/static/js/main.js +932 -0
  102. package/runtime/apps/web/static/js/state.js +28 -0
  103. package/runtime/apps/web/static/js/utils.js +81 -0
  104. package/runtime/apps/web/static/js/views/home.js +604 -0
  105. package/runtime/apps/web/static/js/views/settings.js +477 -0
  106. package/runtime/apps/web/static/styles.css +2206 -0
  107. package/runtime/packages/core/pyproject.toml +18 -0
  108. package/runtime/packages/core/src/video_sum_core/__init__.py +1 -0
  109. package/runtime/packages/core/src/video_sum_core/errors.py +22 -0
  110. package/runtime/packages/core/src/video_sum_core/markdown_exports.py +154 -0
  111. package/runtime/packages/core/src/video_sum_core/models/__init__.py +1 -0
  112. package/runtime/packages/core/src/video_sum_core/models/tasks.py +70 -0
  113. package/runtime/packages/core/src/video_sum_core/pipeline/__init__.py +1 -0
  114. package/runtime/packages/core/src/video_sum_core/pipeline/base.py +46 -0
  115. package/runtime/packages/core/src/video_sum_core/pipeline/real.py +3467 -0
  116. package/runtime/packages/core/src/video_sum_core/transcribe_subprocess.py +173 -0
  117. package/runtime/packages/core/src/video_sum_core/utils.py +127 -0
  118. package/runtime/packages/infra/pyproject.toml +16 -0
  119. package/runtime/packages/infra/src/video_sum_infra/__init__.py +1 -0
  120. package/runtime/packages/infra/src/video_sum_infra/app.py +39 -0
  121. package/runtime/packages/infra/src/video_sum_infra/config.py +393 -0
  122. package/runtime/packages/infra/src/video_sum_infra/db.py +34 -0
  123. package/runtime/packages/infra/src/video_sum_infra/logging.py +54 -0
  124. package/runtime/packages/infra/src/video_sum_infra/paths.py +6 -0
  125. package/runtime/packages/infra/src/video_sum_infra/runtime.py +485 -0
@@ -0,0 +1,264 @@
1
+ import io
2
+ import json
3
+ import wave
4
+
5
+ import httpx
6
+ from fastapi import HTTPException
7
+
8
+ from video_sum_infra.config import ServiceSettings
9
+ from video_sum_infra.runtime import service_log_path
10
+
11
+ from video_sum_service.context import COVER_CACHE_DIR, settings_manager
12
+ from video_sum_service.settings_manager import SettingsUpdatePayload
13
+
14
+ MAX_LOG_CHARS = 20_000
15
+ MAX_LOG_LINE_CHARS = 1_000
16
+
17
+
18
+ def trim_log_text(content: str, *, max_chars: int = MAX_LOG_CHARS, max_line_chars: int = MAX_LOG_LINE_CHARS) -> str:
19
+ lines = content.splitlines()
20
+ trimmed_lines = [
21
+ f"{line[:max_line_chars]}... [line truncated]"
22
+ if len(line) > max_line_chars
23
+ else line
24
+ for line in lines
25
+ ]
26
+ trimmed = "\n".join(trimmed_lines)
27
+ if len(trimmed) <= max_chars:
28
+ return trimmed
29
+ return f"... [log truncated, showing last {max_chars} chars]\n{trimmed[-max_chars:]}"
30
+
31
+
32
+ def read_log_tail(max_lines: int = 200) -> str:
33
+ log_path = service_log_path()
34
+ if not log_path.exists():
35
+ return ""
36
+ try:
37
+ lines = log_path.read_text(encoding="utf-8", errors="replace").splitlines()
38
+ except OSError:
39
+ return ""
40
+ return trim_log_text("\n".join(lines[-max(1, max_lines) :]))
41
+
42
+
43
+ def cache_cover_image(source_url: str, canonical_id: str, referer_url: str | None = None) -> str:
44
+ if not source_url:
45
+ return ""
46
+ normalized_source = source_url.replace("http://", "https://")
47
+ COVER_CACHE_DIR.mkdir(parents=True, exist_ok=True)
48
+ target = COVER_CACHE_DIR / f"{canonical_id}.jpg"
49
+ if target.exists():
50
+ return f"/media/covers/{target.name}"
51
+ try:
52
+ headers = {
53
+ "User-Agent": (
54
+ "Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
55
+ "AppleWebKit/537.36 (KHTML, like Gecko) Chrome/123.0 Safari/537.36"
56
+ ),
57
+ "Referer": referer_url or "https://www.bilibili.com/",
58
+ }
59
+ with httpx.Client(timeout=30, follow_redirects=True, headers=headers) as client:
60
+ response = client.get(normalized_source)
61
+ response.raise_for_status()
62
+ target.write_bytes(response.content)
63
+ return f"/media/covers/{target.name}"
64
+ except Exception:
65
+ return normalized_source
66
+
67
+
68
+ def extract_http_error_detail(response: httpx.Response) -> str:
69
+ try:
70
+ payload = response.json()
71
+ except ValueError:
72
+ payload = None
73
+
74
+ if isinstance(payload, dict):
75
+ for key in ("detail", "message", "error", "msg"):
76
+ value = payload.get(key)
77
+ if isinstance(value, str) and value.strip():
78
+ return value.strip()
79
+ text = (response.text or "").strip()
80
+ return text or f"HTTP {response.status_code}"
81
+
82
+
83
+ def extract_llm_message_content(body: dict[str, object] | None) -> str:
84
+ if not isinstance(body, dict):
85
+ return ""
86
+ choices = body.get("choices")
87
+ if not isinstance(choices, list) or not choices:
88
+ return ""
89
+ first = choices[0]
90
+ if not isinstance(first, dict):
91
+ return ""
92
+ message = first.get("message")
93
+ if not isinstance(message, dict):
94
+ return ""
95
+ content = message.get("content")
96
+ if isinstance(content, str):
97
+ return content.strip()
98
+ if isinstance(content, list):
99
+ text_parts: list[str] = []
100
+ for item in content:
101
+ if isinstance(item, dict):
102
+ text = item.get("text")
103
+ if isinstance(text, str) and text.strip():
104
+ text_parts.append(text.strip())
105
+ return "\n".join(text_parts).strip()
106
+ return ""
107
+
108
+
109
+ def build_test_wav_bytes(duration_ms: int = 250, sample_rate: int = 16000) -> bytes:
110
+ frame_count = max(1, int(sample_rate * duration_ms / 1000))
111
+ buffer = io.BytesIO()
112
+ with wave.open(buffer, "wb") as wav_file:
113
+ wav_file.setnchannels(1)
114
+ wav_file.setsampwidth(2)
115
+ wav_file.setframerate(sample_rate)
116
+ wav_file.writeframes(b"\x00\x00" * frame_count)
117
+ return buffer.getvalue()
118
+
119
+
120
+ def probe_llm_connection(payload: SettingsUpdatePayload | None = None) -> dict[str, object]:
121
+ current_settings = settings_manager.current
122
+ updates = payload.model_dump(exclude_none=True) if payload is not None else {}
123
+ effective_settings = ServiceSettings.model_validate(
124
+ {**current_settings.model_dump(mode="json"), **updates}
125
+ )
126
+
127
+ base_url = str(effective_settings.llm_base_url or "").strip().rstrip("/")
128
+ api_key = str(effective_settings.llm_api_key or "").strip()
129
+ model = str(effective_settings.llm_model or "").strip()
130
+
131
+ if not base_url:
132
+ raise HTTPException(status_code=400, detail="请先填写 API Base URL。")
133
+ if not api_key:
134
+ raise HTTPException(status_code=400, detail="请先填写 API Key。")
135
+ if not model:
136
+ raise HTTPException(status_code=400, detail="请先填写模型名称。")
137
+
138
+ headers = {
139
+ "Authorization": f"Bearer {api_key}",
140
+ "Content-Type": "application/json",
141
+ }
142
+ request_payload = {
143
+ "model": model,
144
+ "messages": [
145
+ {
146
+ "role": "system",
147
+ "content": "你是一个 JSON 测试助手。你只能返回合法 JSON,不能输出任何额外文字。",
148
+ },
149
+ {
150
+ "role": "user",
151
+ "content": (
152
+ "请只返回一个合法 JSON 对象,不要带 markdown 代码块,不要带解释。"
153
+ '格式必须是:{"ok":true,"message":"test"}'
154
+ ),
155
+ },
156
+ ],
157
+ "temperature": 0,
158
+ "max_tokens": 64,
159
+ }
160
+
161
+ try:
162
+ with httpx.Client(timeout=20, follow_redirects=True) as client:
163
+ response = client.post(
164
+ f"{base_url}/chat/completions",
165
+ headers=headers,
166
+ json=request_payload,
167
+ )
168
+ except httpx.HTTPError as exc:
169
+ raise HTTPException(status_code=502, detail=f"LLM 连接失败:{exc}") from exc
170
+
171
+ if response.status_code >= 400:
172
+ detail = extract_http_error_detail(response)
173
+ raise HTTPException(
174
+ status_code=response.status_code,
175
+ detail=f"LLM 测试失败:{detail}",
176
+ )
177
+
178
+ try:
179
+ body = response.json()
180
+ except ValueError:
181
+ body = None
182
+
183
+ preview = extract_llm_message_content(body)
184
+ if not preview:
185
+ raise HTTPException(status_code=502, detail="LLM 测试失败:接口已连通,但没有返回可读取的消息内容。")
186
+
187
+ try:
188
+ parsed_json = json.loads(preview)
189
+ except json.JSONDecodeError as exc:
190
+ raise HTTPException(
191
+ status_code=502,
192
+ detail=f"LLM 测试失败:接口可访问,但当前模型未返回合法 JSON。{exc.msg}",
193
+ ) from exc
194
+
195
+ preview = preview[:200]
196
+ return {
197
+ "ok": True,
198
+ "message": f"LLM 连接与 JSON 输出测试成功:{model}",
199
+ "model": model,
200
+ "baseUrl": base_url,
201
+ "responsePreview": preview,
202
+ "jsonOutputAvailable": True,
203
+ "jsonPreview": json.dumps(parsed_json, ensure_ascii=False)[:200],
204
+ }
205
+
206
+
207
+ def probe_asr_connection(payload: SettingsUpdatePayload | None = None) -> dict[str, object]:
208
+ current_settings = settings_manager.current
209
+ updates = payload.model_dump(exclude_none=True) if payload is not None else {}
210
+ effective_settings = ServiceSettings.model_validate(
211
+ {**current_settings.model_dump(mode="json"), **updates}
212
+ )
213
+
214
+ base_url = str(effective_settings.siliconflow_asr_base_url or "").strip().rstrip("/")
215
+ api_key = str(effective_settings.siliconflow_asr_api_key or "").strip()
216
+ model = str(effective_settings.siliconflow_asr_model or "").strip()
217
+
218
+ if not base_url:
219
+ raise HTTPException(status_code=400, detail="请先填写 SiliconFlow Base URL。")
220
+ if not api_key:
221
+ raise HTTPException(status_code=400, detail="请先填写 SiliconFlow API Key。")
222
+ if not model:
223
+ raise HTTPException(status_code=400, detail="请先填写 ASR 模型名称。")
224
+
225
+ headers = {"Authorization": f"Bearer {api_key}"}
226
+ request_url = f"{base_url}/audio/transcriptions"
227
+ audio_bytes = build_test_wav_bytes()
228
+
229
+ try:
230
+ timeout = httpx.Timeout(connect=20.0, read=90.0, write=90.0, pool=20.0)
231
+ with httpx.Client(timeout=timeout) as client:
232
+ response = client.post(
233
+ request_url,
234
+ headers=headers,
235
+ data={"model": model},
236
+ files={"file": ("bilisum-asr-test.wav", audio_bytes, "audio/wav")},
237
+ )
238
+ except httpx.HTTPError as exc:
239
+ raise HTTPException(status_code=502, detail=f"ASR 连接失败:{exc}") from exc
240
+
241
+ if response.status_code in {401, 403}:
242
+ detail = extract_http_error_detail(response)
243
+ raise HTTPException(status_code=response.status_code, detail=f"ASR 测试失败:认证失败,{detail}")
244
+ if response.status_code >= 400:
245
+ detail = extract_http_error_detail(response)
246
+ raise HTTPException(status_code=response.status_code, detail=f"ASR 测试失败:{detail}")
247
+
248
+ try:
249
+ body = response.json()
250
+ except ValueError:
251
+ body = None
252
+
253
+ transcript = str(body.get("text") or "").strip() if isinstance(body, dict) else ""
254
+ return {
255
+ "ok": True,
256
+ "message": (
257
+ f"ASR 连接测试成功:{model}"
258
+ if transcript
259
+ else f"ASR 连接测试成功:{model}(接口已响应,但测试音频未返回文本)"
260
+ ),
261
+ "model": model,
262
+ "baseUrl": base_url,
263
+ "responsePreview": transcript[:120],
264
+ }
@@ -0,0 +1,5 @@
1
+ from video_sum_service.knowledge.index_service import KnowledgeIndexService
2
+ from video_sum_service.knowledge.rag_service import RagService
3
+ from video_sum_service.knowledge.tag_service import TagService
4
+
5
+ __all__ = ["KnowledgeIndexService", "RagService", "TagService"]
@@ -0,0 +1,408 @@
1
+ from __future__ import annotations
2
+
3
+ import importlib
4
+ import importlib.util
5
+ import json
6
+ import logging
7
+ import re
8
+ import sys
9
+ from datetime import datetime, timezone
10
+ from pathlib import Path
11
+ from typing import Any
12
+
13
+ from fastapi import HTTPException
14
+
15
+ from video_sum_infra.config import ServiceSettings
16
+ from video_sum_infra.runtime import activate_runtime_pythonpath, app_data_root
17
+ from video_sum_service.repository import SqliteTaskRepository
18
+ from video_sum_service.schemas import KnowledgeIndexChunkRecord, KnowledgeSearchResult
19
+
20
+ logger = logging.getLogger("video_sum_service.knowledge")
21
+
22
+
23
+ def format_anchor_seconds(seconds: float | None) -> str | None:
24
+ if seconds is None:
25
+ return None
26
+ total = max(0, int(seconds))
27
+ minutes, sec = divmod(total, 60)
28
+ hours, minutes = divmod(minutes, 60)
29
+ if hours:
30
+ return f"{hours:02d}:{minutes:02d}:{sec:02d}"
31
+ return f"{minutes:02d}:{sec:02d}"
32
+
33
+
34
+ class KnowledgeIndexService:
35
+ def __init__(
36
+ self,
37
+ repository: SqliteTaskRepository,
38
+ settings: ServiceSettings,
39
+ chroma_path: str | Path | None = None,
40
+ model_name: str = "BAAI/bge-small-zh-v1.5",
41
+ ) -> None:
42
+ self._repository = repository
43
+ self._settings = settings
44
+ self._model_name = model_name
45
+ self._chroma_path = Path(chroma_path or (app_data_root() / "knowledge_index"))
46
+ self._embedder = None
47
+ self._collection = None
48
+
49
+ def _ensure_runtime_import_path(self) -> None:
50
+ activate_runtime_pythonpath(self._settings.runtime_channel)
51
+
52
+ def _short_error(self, exc: BaseException, *, limit: int = 500) -> str:
53
+ message = str(exc).strip() or exc.__class__.__name__
54
+ return message if len(message) <= limit else f"{message[:limit]}..."
55
+
56
+ def _import_runtime_dependency(self, module_name: str, package_label: str, missing_detail: str):
57
+ self._ensure_runtime_import_path()
58
+ spec = importlib.util.find_spec(module_name)
59
+ if spec is None:
60
+ logger.warning(
61
+ "knowledge dependency missing package=%s module=%s runtime_channel=%s sys_path_tail=%s",
62
+ package_label,
63
+ module_name,
64
+ self._settings.runtime_channel,
65
+ sys.path[-5:],
66
+ )
67
+ raise HTTPException(status_code=500, detail=missing_detail)
68
+ try:
69
+ return importlib.import_module(module_name)
70
+ except Exception as exc:
71
+ logger.exception(
72
+ "knowledge dependency import failed package=%s module=%s runtime_channel=%s origin=%s sys_path_tail=%s",
73
+ package_label,
74
+ module_name,
75
+ self._settings.runtime_channel,
76
+ getattr(spec, "origin", ""),
77
+ sys.path[-5:],
78
+ )
79
+ raise HTTPException(
80
+ status_code=500,
81
+ detail=(
82
+ f"{package_label} 已安装但加载失败:{self._short_error(exc)}。"
83
+ "请在设置的知识库板块重新安装依赖,或先同步当前 runtime。"
84
+ ),
85
+ ) from exc
86
+
87
+ def _get_embedder(self):
88
+ if self._embedder is None:
89
+ sentence_transformers = self._import_runtime_dependency(
90
+ "sentence_transformers",
91
+ "sentence-transformers",
92
+ "缺少 sentence-transformers 依赖,无法构建知识库索引。",
93
+ )
94
+ try:
95
+ self._embedder = sentence_transformers.SentenceTransformer(self._model_name)
96
+ except Exception as exc:
97
+ logger.exception(
98
+ "knowledge embedding model load failed model=%s runtime_channel=%s",
99
+ self._model_name,
100
+ self._settings.runtime_channel,
101
+ )
102
+ raise HTTPException(
103
+ status_code=500,
104
+ detail=(
105
+ f"知识库向量模型加载失败:{self._short_error(exc)}。"
106
+ "请检查网络 / HuggingFace 缓存,或稍后重试。"
107
+ ),
108
+ ) from exc
109
+ return self._embedder
110
+
111
+ def _get_collection(self):
112
+ if self._collection is None:
113
+ chromadb = self._import_runtime_dependency(
114
+ "chromadb",
115
+ "chromadb",
116
+ "缺少 chromadb 依赖,无法构建知识库索引。",
117
+ )
118
+ try:
119
+ self._chroma_path.mkdir(parents=True, exist_ok=True)
120
+ client = chromadb.PersistentClient(path=str(self._chroma_path))
121
+ self._collection = client.get_or_create_collection(
122
+ name="bilisum_knowledge",
123
+ metadata={"hnsw:space": "cosine"},
124
+ )
125
+ except Exception as exc:
126
+ logger.exception(
127
+ "knowledge chromadb collection init failed path=%s runtime_channel=%s",
128
+ self._chroma_path,
129
+ self._settings.runtime_channel,
130
+ )
131
+ raise HTTPException(
132
+ status_code=500,
133
+ detail=f"知识库索引存储初始化失败:{self._short_error(exc)}。",
134
+ ) from exc
135
+ return self._collection
136
+
137
+ def _embed_texts(self, texts: list[str]) -> list[list[float]]:
138
+ if not texts:
139
+ return []
140
+ vectors = self._get_embedder().encode(texts, normalize_embeddings=True)
141
+ return [list(map(float, vector)) for vector in vectors]
142
+
143
+ def _split_markdown_sections(self, markdown: str) -> list[tuple[str, str]]:
144
+ content = str(markdown or "").strip()
145
+ if not content:
146
+ return []
147
+ sections: list[tuple[str, str]] = []
148
+ current_title = "知识笔记"
149
+ current_lines: list[str] = []
150
+ for line in content.splitlines():
151
+ if re.match(r"^\s{0,3}#{1,3}\s+", line):
152
+ if current_lines:
153
+ section_body = "\n".join(current_lines).strip()
154
+ if section_body:
155
+ sections.append((current_title, section_body))
156
+ current_title = re.sub(r"^\s{0,3}#{1,3}\s+", "", line).strip() or "知识笔记"
157
+ current_lines = []
158
+ continue
159
+ current_lines.append(line)
160
+ if current_lines:
161
+ section_body = "\n".join(current_lines).strip()
162
+ if section_body:
163
+ sections.append((current_title, section_body))
164
+ return sections
165
+
166
+ def _build_chunks_for_video(self, video_id: str) -> tuple[Any, list[KnowledgeIndexChunkRecord]]:
167
+ asset = self._repository.get_video_asset(video_id)
168
+ if asset is None or asset.latest_result is None:
169
+ return None, []
170
+ result = asset.latest_result
171
+ task = self._repository.get_task(asset.latest_task_id) if asset.latest_task_id else None
172
+ page_title = str(task.page_title or task.task_input.title or "").strip() if task is not None else ""
173
+ display_title = page_title or asset.title
174
+ now = datetime.now(timezone.utc)
175
+ chunk_specs: list[dict[str, object]] = []
176
+
177
+ overview_parts = [str(result.overview or "").strip(), *[str(item).strip() for item in result.key_points if str(item).strip()]]
178
+ overview_text = "\n".join([part for part in overview_parts if part]).strip()
179
+ if overview_text:
180
+ chunk_specs.append(
181
+ {
182
+ "index_type": "video_summary",
183
+ "segment_order": 0,
184
+ "anchor_label": None,
185
+ "anchor_seconds": None,
186
+ "content": f"{display_title}\n{overview_text}",
187
+ }
188
+ )
189
+
190
+ for index, chapter in enumerate(result.timeline):
191
+ if not isinstance(chapter, dict):
192
+ continue
193
+ title = str(chapter.get("title") or f"章节 {index + 1}").strip()
194
+ summary = str(chapter.get("summary") or "").strip()
195
+ start_raw = chapter.get("start")
196
+ start = float(start_raw) if isinstance(start_raw, (int, float)) else None
197
+ content = "\n".join([display_title, title, summary]).strip()
198
+ if summary:
199
+ chunk_specs.append(
200
+ {
201
+ "index_type": "chapter",
202
+ "segment_order": index + 1,
203
+ "anchor_label": title,
204
+ "anchor_seconds": start,
205
+ "content": content,
206
+ }
207
+ )
208
+
209
+ for index, (title, body) in enumerate(self._split_markdown_sections(result.knowledge_note_markdown)):
210
+ chunk_specs.append(
211
+ {
212
+ "index_type": "knowledge_note",
213
+ "segment_order": index + 1,
214
+ "anchor_label": title,
215
+ "anchor_seconds": None,
216
+ "content": f"{display_title}\n{title}\n{body}".strip(),
217
+ }
218
+ )
219
+
220
+ chunk_specs = [item for item in chunk_specs if str(item["content"]).strip()]
221
+ if not chunk_specs:
222
+ return asset, []
223
+
224
+ vectors = self._embed_texts([str(item["content"]) for item in chunk_specs])
225
+ chunks = [
226
+ KnowledgeIndexChunkRecord(
227
+ chunk_id=f"{video_id}:{item['index_type']}:{item['segment_order'] or 0}",
228
+ video_id=video_id,
229
+ embedding_json=json.dumps(vectors[index], ensure_ascii=False),
230
+ indexed_content=str(item["content"]),
231
+ index_type=str(item["index_type"]),
232
+ segment_order=int(item["segment_order"]) if item["segment_order"] is not None else None,
233
+ anchor_label=str(item["anchor_label"]) if item["anchor_label"] else None,
234
+ anchor_seconds=float(item["anchor_seconds"]) if item["anchor_seconds"] is not None else None,
235
+ created_at=now,
236
+ updated_at=now,
237
+ )
238
+ for index, item in enumerate(chunk_specs)
239
+ ]
240
+ return asset, chunks
241
+
242
+ def index_video(self, video_id: str, content: str | None = None) -> bool:
243
+ del content
244
+ asset, chunks = self._build_chunks_for_video(video_id)
245
+ if asset is None:
246
+ return False
247
+
248
+ collection = self._get_collection()
249
+ try:
250
+ collection.delete(where={"video_id": video_id})
251
+ except Exception:
252
+ pass
253
+
254
+ self._repository.replace_knowledge_chunks(video_id, chunks)
255
+ if not chunks:
256
+ return True
257
+
258
+ tags = [item.tag for item in self._repository.list_video_tags(video_id)]
259
+ task = self._repository.get_task(asset.latest_task_id) if asset.latest_task_id else None
260
+ page_number = int(task.page_number) if task is not None and task.page_number is not None else None
261
+ page_title = str(task.page_title or task.task_input.title or "").strip() if task is not None else ""
262
+ collection.add(
263
+ ids=[chunk.chunk_id for chunk in chunks],
264
+ documents=[chunk.indexed_content for chunk in chunks],
265
+ embeddings=[json.loads(chunk.embedding_json) for chunk in chunks],
266
+ metadatas=[
267
+ {
268
+ "video_id": chunk.video_id,
269
+ "index_type": chunk.index_type,
270
+ "anchor_label": chunk.anchor_label or "",
271
+ "anchor_seconds": float(chunk.anchor_seconds) if chunk.anchor_seconds is not None else -1.0,
272
+ "title": asset.title,
273
+ "page_number": page_number if page_number is not None else -1,
274
+ "page_title": page_title,
275
+ "display_title": page_title or asset.title,
276
+ "cover_url": asset.cover_url or "",
277
+ "tags_json": json.dumps(tags, ensure_ascii=False),
278
+ }
279
+ for chunk in chunks
280
+ ],
281
+ )
282
+ return True
283
+
284
+ def remove_video(self, video_id: str) -> bool:
285
+ self._repository.delete_knowledge_chunks(video_id)
286
+ try:
287
+ self._get_collection().delete(where={"video_id": video_id})
288
+ except HTTPException:
289
+ raise
290
+ except Exception:
291
+ pass
292
+ return True
293
+
294
+ def rebuild_index(self, force: bool = False) -> int:
295
+ del force
296
+ videos = [video for video in self._repository.list_video_assets() if video.latest_result is not None]
297
+ collection = self._get_collection()
298
+ try:
299
+ existing = collection.get()
300
+ ids = existing.get("ids") if isinstance(existing, dict) else None
301
+ if ids:
302
+ collection.delete(ids=ids)
303
+ except Exception:
304
+ pass
305
+
306
+ self._repository.clear_knowledge_chunks()
307
+ count = 0
308
+ for video in videos:
309
+ if self.index_video(video.video_id):
310
+ count += 1
311
+ return count
312
+
313
+ def _fetch_candidate_chunks(self, query: str, limit: int = 10, tag_filter: list[str] | None = None) -> list[dict[str, object]]:
314
+ cleaned_query = str(query or "").strip()
315
+ if not cleaned_query:
316
+ return []
317
+ if self._repository.get_knowledge_chunk_count() == 0:
318
+ self.rebuild_index()
319
+
320
+ effective_limit = max(limit * 4, 12)
321
+ query_embedding = self._embed_texts([cleaned_query])[0]
322
+ query_result = self._get_collection().query(
323
+ query_embeddings=[query_embedding],
324
+ n_results=effective_limit,
325
+ )
326
+ ids = query_result.get("ids", [[]])[0] if isinstance(query_result, dict) else []
327
+ documents = query_result.get("documents", [[]])[0] if isinstance(query_result, dict) else []
328
+ metadatas = query_result.get("metadatas", [[]])[0] if isinstance(query_result, dict) else []
329
+ distances = query_result.get("distances", [[]])[0] if isinstance(query_result, dict) else []
330
+
331
+ allowed_video_ids: set[str] | None = None
332
+ if tag_filter:
333
+ selected = {tag.strip() for tag in tag_filter if str(tag).strip()}
334
+ if selected:
335
+ allowed_video_ids = {
336
+ item.video_id
337
+ for item in self._repository.list_all_video_tags()
338
+ if item.tag in selected
339
+ }
340
+
341
+ candidates: list[dict[str, object]] = []
342
+ for index, chunk_id in enumerate(ids):
343
+ metadata = metadatas[index] if index < len(metadatas) and isinstance(metadatas[index], dict) else {}
344
+ video_id = str(metadata.get("video_id") or "")
345
+ if allowed_video_ids is not None and video_id not in allowed_video_ids:
346
+ continue
347
+ distance = float(distances[index]) if index < len(distances) else 1.0
348
+ relevance = max(0.0, 1.0 - distance / 2.0)
349
+ candidates.append(
350
+ {
351
+ "chunk_id": chunk_id,
352
+ "video_id": video_id,
353
+ "document": documents[index] if index < len(documents) else "",
354
+ "metadata": metadata,
355
+ "relevance_score": round(relevance, 4),
356
+ }
357
+ )
358
+ return candidates
359
+
360
+ def search(self, query: str, limit: int = 10, tag_filter: list[str] | None = None) -> list[KnowledgeSearchResult]:
361
+ candidates = self._fetch_candidate_chunks(query, limit=limit, tag_filter=tag_filter)
362
+ if not candidates:
363
+ return []
364
+
365
+ grouped: dict[str, dict[str, object]] = {}
366
+ for candidate in candidates:
367
+ video_id = str(candidate["video_id"])
368
+ current = grouped.get(video_id)
369
+ if current is None or float(candidate["relevance_score"]) > float(current["relevance_score"]):
370
+ grouped[video_id] = candidate
371
+
372
+ ordered = sorted(grouped.values(), key=lambda item: float(item["relevance_score"]), reverse=True)[: max(1, limit)]
373
+ results: list[KnowledgeSearchResult] = []
374
+ for candidate in ordered:
375
+ video_id = str(candidate["video_id"])
376
+ asset = self._repository.get_video_asset(video_id)
377
+ metadata = candidate["metadata"] if isinstance(candidate["metadata"], dict) else {}
378
+ video_title = asset.title if asset is not None else str(metadata.get("title") or "未知视频")
379
+ page_title = str(metadata.get("page_title") or metadata.get("display_title") or "").strip()
380
+ if page_title == video_title:
381
+ page_title = ""
382
+ page_number_raw = metadata.get("page_number")
383
+ page_number = int(page_number_raw) if isinstance(page_number_raw, (int, float)) and int(page_number_raw) > 0 else None
384
+ snippet = str(candidate["document"]).strip().replace("\n", " ")
385
+ snippet = snippet[:180].rstrip() + ("..." if len(snippet) > 180 else "")
386
+ tags = [item.tag for item in self._repository.list_video_tags(video_id)]
387
+ results.append(
388
+ KnowledgeSearchResult(
389
+ video_id=video_id,
390
+ title=page_title or video_title,
391
+ relevance_score=float(candidate["relevance_score"]),
392
+ snippet=snippet,
393
+ tags=tags,
394
+ cover_url=asset.cover_url if asset is not None else str(metadata.get("cover_url") or ""),
395
+ timestamp=format_anchor_seconds(
396
+ float(metadata["anchor_seconds"])
397
+ if metadata.get("anchor_seconds") not in {None, "", -1, -1.0}
398
+ else None
399
+ ),
400
+ video_title=video_title,
401
+ page_title=page_title or None,
402
+ page_number=page_number,
403
+ )
404
+ )
405
+ return results
406
+
407
+ def search_chunks(self, query: str, limit: int = 5, tag_filter: list[str] | None = None) -> list[dict[str, object]]:
408
+ return self._fetch_candidate_chunks(query, limit=limit, tag_filter=tag_filter)[: max(1, limit)]