bilisum 1.13.2 → 1.13.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +4 -4
- package/bin/bilisum.js +157 -76
- package/package.json +7 -2
- package/runtime/VERSION +1 -0
- package/runtime/apps/service/pyproject.toml +29 -0
- package/runtime/apps/service/src/video_sum_service/__init__.py +1 -0
- package/runtime/apps/service/src/video_sum_service/__main__.py +5 -0
- package/runtime/apps/service/src/video_sum_service/app.py +480 -0
- package/runtime/apps/service/src/video_sum_service/context.py +24 -0
- package/runtime/apps/service/src/video_sum_service/integrations.py +264 -0
- package/runtime/apps/service/src/video_sum_service/knowledge/__init__.py +5 -0
- package/runtime/apps/service/src/video_sum_service/knowledge/index_service.py +408 -0
- package/runtime/apps/service/src/video_sum_service/knowledge/local_llm.py +319 -0
- package/runtime/apps/service/src/video_sum_service/knowledge/rag_service.py +492 -0
- package/runtime/apps/service/src/video_sum_service/knowledge/tag_service.py +242 -0
- package/runtime/apps/service/src/video_sum_service/main.py +31 -0
- package/runtime/apps/service/src/video_sum_service/repository.py +942 -0
- package/runtime/apps/service/src/video_sum_service/routers/__init__.py +1 -0
- package/runtime/apps/service/src/video_sum_service/routers/knowledge.py +272 -0
- package/runtime/apps/service/src/video_sum_service/routers/system.py +280 -0
- package/runtime/apps/service/src/video_sum_service/routers/tasks.py +287 -0
- package/runtime/apps/service/src/video_sum_service/routers/videos.py +766 -0
- package/runtime/apps/service/src/video_sum_service/runtime_support.py +1007 -0
- package/runtime/apps/service/src/video_sum_service/schemas.py +418 -0
- package/runtime/apps/service/src/video_sum_service/settings_manager.py +130 -0
- package/runtime/apps/service/src/video_sum_service/task_artifacts.py +102 -0
- package/runtime/apps/service/src/video_sum_service/task_exports.py +108 -0
- package/runtime/apps/service/src/video_sum_service/transcribe_worker.py +9 -0
- package/runtime/apps/service/src/video_sum_service/video_assets.py +623 -0
- package/runtime/apps/service/src/video_sum_service/worker.py +381 -0
- package/runtime/apps/web/static/apple-touch-icon.png +0 -0
- package/runtime/apps/web/static/assets/KaTeX_AMS-Regular-BQhdFMY1.woff2 +0 -0
- package/runtime/apps/web/static/assets/KaTeX_AMS-Regular-DMm9YOAa.woff +0 -0
- package/runtime/apps/web/static/assets/KaTeX_AMS-Regular-DRggAlZN.ttf +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Caligraphic-Bold-ATXxdsX0.ttf +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Caligraphic-Bold-BEiXGLvX.woff +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Caligraphic-Bold-Dq_IR9rO.woff2 +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Caligraphic-Regular-CTRA-rTL.woff +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Caligraphic-Regular-Di6jR-x-.woff2 +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Caligraphic-Regular-wX97UBjC.ttf +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Fraktur-Bold-BdnERNNW.ttf +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Fraktur-Bold-BsDP51OF.woff +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Fraktur-Bold-CL6g_b3V.woff2 +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Fraktur-Regular-CB_wures.ttf +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Fraktur-Regular-CTYiF6lA.woff2 +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Fraktur-Regular-Dxdc4cR9.woff +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Main-Bold-Cx986IdX.woff2 +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Main-Bold-Jm3AIy58.woff +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Main-Bold-waoOVXN0.ttf +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Main-BoldItalic-DxDJ3AOS.woff2 +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Main-BoldItalic-DzxPMmG6.ttf +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Main-BoldItalic-SpSLRI95.woff +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Main-Italic-3WenGoN9.ttf +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Main-Italic-BMLOBm91.woff +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Main-Italic-NWA7e6Wa.woff2 +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Main-Regular-B22Nviop.woff2 +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Main-Regular-Dr94JaBh.woff +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Main-Regular-ypZvNtVU.ttf +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Math-BoldItalic-B3XSjfu4.ttf +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Math-BoldItalic-CZnvNsCZ.woff2 +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Math-BoldItalic-iY-2wyZ7.woff +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Math-Italic-DA0__PXp.woff +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Math-Italic-flOr_0UB.ttf +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Math-Italic-t53AETM-.woff2 +0 -0
- package/runtime/apps/web/static/assets/KaTeX_SansSerif-Bold-CFMepnvq.ttf +0 -0
- package/runtime/apps/web/static/assets/KaTeX_SansSerif-Bold-D1sUS0GD.woff2 +0 -0
- package/runtime/apps/web/static/assets/KaTeX_SansSerif-Bold-DbIhKOiC.woff +0 -0
- package/runtime/apps/web/static/assets/KaTeX_SansSerif-Italic-C3H0VqGB.woff2 +0 -0
- package/runtime/apps/web/static/assets/KaTeX_SansSerif-Italic-DN2j7dab.woff +0 -0
- package/runtime/apps/web/static/assets/KaTeX_SansSerif-Italic-YYjJ1zSn.ttf +0 -0
- package/runtime/apps/web/static/assets/KaTeX_SansSerif-Regular-BNo7hRIc.ttf +0 -0
- package/runtime/apps/web/static/assets/KaTeX_SansSerif-Regular-CS6fqUqJ.woff +0 -0
- package/runtime/apps/web/static/assets/KaTeX_SansSerif-Regular-DDBCnlJ7.woff2 +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Script-Regular-C5JkGWo-.ttf +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Script-Regular-D3wIWfF6.woff2 +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Script-Regular-D5yQViql.woff +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Size1-Regular-C195tn64.woff +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Size1-Regular-Dbsnue_I.ttf +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Size1-Regular-mCD8mA8B.woff2 +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Size2-Regular-B7gKUWhC.ttf +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Size2-Regular-Dy4dx90m.woff2 +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Size2-Regular-oD1tc_U0.woff +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Size3-Regular-CTq5MqoE.woff +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Size3-Regular-DgpXs0kz.ttf +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Size4-Regular-BF-4gkZK.woff +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Size4-Regular-DWFBv043.ttf +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Size4-Regular-Dl5lxZxV.woff2 +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Typewriter-Regular-C0xS9mPB.woff +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Typewriter-Regular-CO6r4hn1.woff2 +0 -0
- package/runtime/apps/web/static/assets/KaTeX_Typewriter-Regular-D3Ib7_Hf.ttf +0 -0
- package/runtime/apps/web/static/assets/icons/icon-180.png +0 -0
- package/runtime/apps/web/static/assets/icons/icon-512.png +0 -0
- package/runtime/apps/web/static/assets/icons/icon.svg +17 -0
- package/runtime/apps/web/static/assets/index-Cj5PVjQg.js +388 -0
- package/runtime/apps/web/static/assets/index-DU2t4_7s.css +1 -0
- package/runtime/apps/web/static/favicon-32x32.png +0 -0
- package/runtime/apps/web/static/favicon.ico +0 -0
- package/runtime/apps/web/static/favicon.svg +17 -0
- package/runtime/apps/web/static/index.html +19 -0
- package/runtime/apps/web/static/js/api.js +123 -0
- package/runtime/apps/web/static/js/main.js +932 -0
- package/runtime/apps/web/static/js/state.js +28 -0
- package/runtime/apps/web/static/js/utils.js +81 -0
- package/runtime/apps/web/static/js/views/home.js +604 -0
- package/runtime/apps/web/static/js/views/settings.js +477 -0
- package/runtime/apps/web/static/styles.css +2206 -0
- package/runtime/packages/core/pyproject.toml +18 -0
- package/runtime/packages/core/src/video_sum_core/__init__.py +1 -0
- package/runtime/packages/core/src/video_sum_core/errors.py +22 -0
- package/runtime/packages/core/src/video_sum_core/markdown_exports.py +154 -0
- package/runtime/packages/core/src/video_sum_core/models/__init__.py +1 -0
- package/runtime/packages/core/src/video_sum_core/models/tasks.py +70 -0
- package/runtime/packages/core/src/video_sum_core/pipeline/__init__.py +1 -0
- package/runtime/packages/core/src/video_sum_core/pipeline/base.py +46 -0
- package/runtime/packages/core/src/video_sum_core/pipeline/real.py +3467 -0
- package/runtime/packages/core/src/video_sum_core/transcribe_subprocess.py +173 -0
- package/runtime/packages/core/src/video_sum_core/utils.py +127 -0
- package/runtime/packages/infra/pyproject.toml +16 -0
- package/runtime/packages/infra/src/video_sum_infra/__init__.py +1 -0
- package/runtime/packages/infra/src/video_sum_infra/app.py +39 -0
- package/runtime/packages/infra/src/video_sum_infra/config.py +393 -0
- package/runtime/packages/infra/src/video_sum_infra/db.py +34 -0
- package/runtime/packages/infra/src/video_sum_infra/logging.py +54 -0
- package/runtime/packages/infra/src/video_sum_infra/paths.py +6 -0
- package/runtime/packages/infra/src/video_sum_infra/runtime.py +485 -0
|
@@ -0,0 +1,264 @@
|
|
|
1
|
+
import io
|
|
2
|
+
import json
|
|
3
|
+
import wave
|
|
4
|
+
|
|
5
|
+
import httpx
|
|
6
|
+
from fastapi import HTTPException
|
|
7
|
+
|
|
8
|
+
from video_sum_infra.config import ServiceSettings
|
|
9
|
+
from video_sum_infra.runtime import service_log_path
|
|
10
|
+
|
|
11
|
+
from video_sum_service.context import COVER_CACHE_DIR, settings_manager
|
|
12
|
+
from video_sum_service.settings_manager import SettingsUpdatePayload
|
|
13
|
+
|
|
14
|
+
MAX_LOG_CHARS = 20_000
|
|
15
|
+
MAX_LOG_LINE_CHARS = 1_000
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def trim_log_text(content: str, *, max_chars: int = MAX_LOG_CHARS, max_line_chars: int = MAX_LOG_LINE_CHARS) -> str:
|
|
19
|
+
lines = content.splitlines()
|
|
20
|
+
trimmed_lines = [
|
|
21
|
+
f"{line[:max_line_chars]}... [line truncated]"
|
|
22
|
+
if len(line) > max_line_chars
|
|
23
|
+
else line
|
|
24
|
+
for line in lines
|
|
25
|
+
]
|
|
26
|
+
trimmed = "\n".join(trimmed_lines)
|
|
27
|
+
if len(trimmed) <= max_chars:
|
|
28
|
+
return trimmed
|
|
29
|
+
return f"... [log truncated, showing last {max_chars} chars]\n{trimmed[-max_chars:]}"
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def read_log_tail(max_lines: int = 200) -> str:
|
|
33
|
+
log_path = service_log_path()
|
|
34
|
+
if not log_path.exists():
|
|
35
|
+
return ""
|
|
36
|
+
try:
|
|
37
|
+
lines = log_path.read_text(encoding="utf-8", errors="replace").splitlines()
|
|
38
|
+
except OSError:
|
|
39
|
+
return ""
|
|
40
|
+
return trim_log_text("\n".join(lines[-max(1, max_lines) :]))
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def cache_cover_image(source_url: str, canonical_id: str, referer_url: str | None = None) -> str:
|
|
44
|
+
if not source_url:
|
|
45
|
+
return ""
|
|
46
|
+
normalized_source = source_url.replace("http://", "https://")
|
|
47
|
+
COVER_CACHE_DIR.mkdir(parents=True, exist_ok=True)
|
|
48
|
+
target = COVER_CACHE_DIR / f"{canonical_id}.jpg"
|
|
49
|
+
if target.exists():
|
|
50
|
+
return f"/media/covers/{target.name}"
|
|
51
|
+
try:
|
|
52
|
+
headers = {
|
|
53
|
+
"User-Agent": (
|
|
54
|
+
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
|
|
55
|
+
"AppleWebKit/537.36 (KHTML, like Gecko) Chrome/123.0 Safari/537.36"
|
|
56
|
+
),
|
|
57
|
+
"Referer": referer_url or "https://www.bilibili.com/",
|
|
58
|
+
}
|
|
59
|
+
with httpx.Client(timeout=30, follow_redirects=True, headers=headers) as client:
|
|
60
|
+
response = client.get(normalized_source)
|
|
61
|
+
response.raise_for_status()
|
|
62
|
+
target.write_bytes(response.content)
|
|
63
|
+
return f"/media/covers/{target.name}"
|
|
64
|
+
except Exception:
|
|
65
|
+
return normalized_source
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def extract_http_error_detail(response: httpx.Response) -> str:
|
|
69
|
+
try:
|
|
70
|
+
payload = response.json()
|
|
71
|
+
except ValueError:
|
|
72
|
+
payload = None
|
|
73
|
+
|
|
74
|
+
if isinstance(payload, dict):
|
|
75
|
+
for key in ("detail", "message", "error", "msg"):
|
|
76
|
+
value = payload.get(key)
|
|
77
|
+
if isinstance(value, str) and value.strip():
|
|
78
|
+
return value.strip()
|
|
79
|
+
text = (response.text or "").strip()
|
|
80
|
+
return text or f"HTTP {response.status_code}"
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def extract_llm_message_content(body: dict[str, object] | None) -> str:
|
|
84
|
+
if not isinstance(body, dict):
|
|
85
|
+
return ""
|
|
86
|
+
choices = body.get("choices")
|
|
87
|
+
if not isinstance(choices, list) or not choices:
|
|
88
|
+
return ""
|
|
89
|
+
first = choices[0]
|
|
90
|
+
if not isinstance(first, dict):
|
|
91
|
+
return ""
|
|
92
|
+
message = first.get("message")
|
|
93
|
+
if not isinstance(message, dict):
|
|
94
|
+
return ""
|
|
95
|
+
content = message.get("content")
|
|
96
|
+
if isinstance(content, str):
|
|
97
|
+
return content.strip()
|
|
98
|
+
if isinstance(content, list):
|
|
99
|
+
text_parts: list[str] = []
|
|
100
|
+
for item in content:
|
|
101
|
+
if isinstance(item, dict):
|
|
102
|
+
text = item.get("text")
|
|
103
|
+
if isinstance(text, str) and text.strip():
|
|
104
|
+
text_parts.append(text.strip())
|
|
105
|
+
return "\n".join(text_parts).strip()
|
|
106
|
+
return ""
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
def build_test_wav_bytes(duration_ms: int = 250, sample_rate: int = 16000) -> bytes:
|
|
110
|
+
frame_count = max(1, int(sample_rate * duration_ms / 1000))
|
|
111
|
+
buffer = io.BytesIO()
|
|
112
|
+
with wave.open(buffer, "wb") as wav_file:
|
|
113
|
+
wav_file.setnchannels(1)
|
|
114
|
+
wav_file.setsampwidth(2)
|
|
115
|
+
wav_file.setframerate(sample_rate)
|
|
116
|
+
wav_file.writeframes(b"\x00\x00" * frame_count)
|
|
117
|
+
return buffer.getvalue()
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
def probe_llm_connection(payload: SettingsUpdatePayload | None = None) -> dict[str, object]:
|
|
121
|
+
current_settings = settings_manager.current
|
|
122
|
+
updates = payload.model_dump(exclude_none=True) if payload is not None else {}
|
|
123
|
+
effective_settings = ServiceSettings.model_validate(
|
|
124
|
+
{**current_settings.model_dump(mode="json"), **updates}
|
|
125
|
+
)
|
|
126
|
+
|
|
127
|
+
base_url = str(effective_settings.llm_base_url or "").strip().rstrip("/")
|
|
128
|
+
api_key = str(effective_settings.llm_api_key or "").strip()
|
|
129
|
+
model = str(effective_settings.llm_model or "").strip()
|
|
130
|
+
|
|
131
|
+
if not base_url:
|
|
132
|
+
raise HTTPException(status_code=400, detail="请先填写 API Base URL。")
|
|
133
|
+
if not api_key:
|
|
134
|
+
raise HTTPException(status_code=400, detail="请先填写 API Key。")
|
|
135
|
+
if not model:
|
|
136
|
+
raise HTTPException(status_code=400, detail="请先填写模型名称。")
|
|
137
|
+
|
|
138
|
+
headers = {
|
|
139
|
+
"Authorization": f"Bearer {api_key}",
|
|
140
|
+
"Content-Type": "application/json",
|
|
141
|
+
}
|
|
142
|
+
request_payload = {
|
|
143
|
+
"model": model,
|
|
144
|
+
"messages": [
|
|
145
|
+
{
|
|
146
|
+
"role": "system",
|
|
147
|
+
"content": "你是一个 JSON 测试助手。你只能返回合法 JSON,不能输出任何额外文字。",
|
|
148
|
+
},
|
|
149
|
+
{
|
|
150
|
+
"role": "user",
|
|
151
|
+
"content": (
|
|
152
|
+
"请只返回一个合法 JSON 对象,不要带 markdown 代码块,不要带解释。"
|
|
153
|
+
'格式必须是:{"ok":true,"message":"test"}'
|
|
154
|
+
),
|
|
155
|
+
},
|
|
156
|
+
],
|
|
157
|
+
"temperature": 0,
|
|
158
|
+
"max_tokens": 64,
|
|
159
|
+
}
|
|
160
|
+
|
|
161
|
+
try:
|
|
162
|
+
with httpx.Client(timeout=20, follow_redirects=True) as client:
|
|
163
|
+
response = client.post(
|
|
164
|
+
f"{base_url}/chat/completions",
|
|
165
|
+
headers=headers,
|
|
166
|
+
json=request_payload,
|
|
167
|
+
)
|
|
168
|
+
except httpx.HTTPError as exc:
|
|
169
|
+
raise HTTPException(status_code=502, detail=f"LLM 连接失败:{exc}") from exc
|
|
170
|
+
|
|
171
|
+
if response.status_code >= 400:
|
|
172
|
+
detail = extract_http_error_detail(response)
|
|
173
|
+
raise HTTPException(
|
|
174
|
+
status_code=response.status_code,
|
|
175
|
+
detail=f"LLM 测试失败:{detail}",
|
|
176
|
+
)
|
|
177
|
+
|
|
178
|
+
try:
|
|
179
|
+
body = response.json()
|
|
180
|
+
except ValueError:
|
|
181
|
+
body = None
|
|
182
|
+
|
|
183
|
+
preview = extract_llm_message_content(body)
|
|
184
|
+
if not preview:
|
|
185
|
+
raise HTTPException(status_code=502, detail="LLM 测试失败:接口已连通,但没有返回可读取的消息内容。")
|
|
186
|
+
|
|
187
|
+
try:
|
|
188
|
+
parsed_json = json.loads(preview)
|
|
189
|
+
except json.JSONDecodeError as exc:
|
|
190
|
+
raise HTTPException(
|
|
191
|
+
status_code=502,
|
|
192
|
+
detail=f"LLM 测试失败:接口可访问,但当前模型未返回合法 JSON。{exc.msg}",
|
|
193
|
+
) from exc
|
|
194
|
+
|
|
195
|
+
preview = preview[:200]
|
|
196
|
+
return {
|
|
197
|
+
"ok": True,
|
|
198
|
+
"message": f"LLM 连接与 JSON 输出测试成功:{model}",
|
|
199
|
+
"model": model,
|
|
200
|
+
"baseUrl": base_url,
|
|
201
|
+
"responsePreview": preview,
|
|
202
|
+
"jsonOutputAvailable": True,
|
|
203
|
+
"jsonPreview": json.dumps(parsed_json, ensure_ascii=False)[:200],
|
|
204
|
+
}
|
|
205
|
+
|
|
206
|
+
|
|
207
|
+
def probe_asr_connection(payload: SettingsUpdatePayload | None = None) -> dict[str, object]:
|
|
208
|
+
current_settings = settings_manager.current
|
|
209
|
+
updates = payload.model_dump(exclude_none=True) if payload is not None else {}
|
|
210
|
+
effective_settings = ServiceSettings.model_validate(
|
|
211
|
+
{**current_settings.model_dump(mode="json"), **updates}
|
|
212
|
+
)
|
|
213
|
+
|
|
214
|
+
base_url = str(effective_settings.siliconflow_asr_base_url or "").strip().rstrip("/")
|
|
215
|
+
api_key = str(effective_settings.siliconflow_asr_api_key or "").strip()
|
|
216
|
+
model = str(effective_settings.siliconflow_asr_model or "").strip()
|
|
217
|
+
|
|
218
|
+
if not base_url:
|
|
219
|
+
raise HTTPException(status_code=400, detail="请先填写 SiliconFlow Base URL。")
|
|
220
|
+
if not api_key:
|
|
221
|
+
raise HTTPException(status_code=400, detail="请先填写 SiliconFlow API Key。")
|
|
222
|
+
if not model:
|
|
223
|
+
raise HTTPException(status_code=400, detail="请先填写 ASR 模型名称。")
|
|
224
|
+
|
|
225
|
+
headers = {"Authorization": f"Bearer {api_key}"}
|
|
226
|
+
request_url = f"{base_url}/audio/transcriptions"
|
|
227
|
+
audio_bytes = build_test_wav_bytes()
|
|
228
|
+
|
|
229
|
+
try:
|
|
230
|
+
timeout = httpx.Timeout(connect=20.0, read=90.0, write=90.0, pool=20.0)
|
|
231
|
+
with httpx.Client(timeout=timeout) as client:
|
|
232
|
+
response = client.post(
|
|
233
|
+
request_url,
|
|
234
|
+
headers=headers,
|
|
235
|
+
data={"model": model},
|
|
236
|
+
files={"file": ("bilisum-asr-test.wav", audio_bytes, "audio/wav")},
|
|
237
|
+
)
|
|
238
|
+
except httpx.HTTPError as exc:
|
|
239
|
+
raise HTTPException(status_code=502, detail=f"ASR 连接失败:{exc}") from exc
|
|
240
|
+
|
|
241
|
+
if response.status_code in {401, 403}:
|
|
242
|
+
detail = extract_http_error_detail(response)
|
|
243
|
+
raise HTTPException(status_code=response.status_code, detail=f"ASR 测试失败:认证失败,{detail}")
|
|
244
|
+
if response.status_code >= 400:
|
|
245
|
+
detail = extract_http_error_detail(response)
|
|
246
|
+
raise HTTPException(status_code=response.status_code, detail=f"ASR 测试失败:{detail}")
|
|
247
|
+
|
|
248
|
+
try:
|
|
249
|
+
body = response.json()
|
|
250
|
+
except ValueError:
|
|
251
|
+
body = None
|
|
252
|
+
|
|
253
|
+
transcript = str(body.get("text") or "").strip() if isinstance(body, dict) else ""
|
|
254
|
+
return {
|
|
255
|
+
"ok": True,
|
|
256
|
+
"message": (
|
|
257
|
+
f"ASR 连接测试成功:{model}"
|
|
258
|
+
if transcript
|
|
259
|
+
else f"ASR 连接测试成功:{model}(接口已响应,但测试音频未返回文本)"
|
|
260
|
+
),
|
|
261
|
+
"model": model,
|
|
262
|
+
"baseUrl": base_url,
|
|
263
|
+
"responsePreview": transcript[:120],
|
|
264
|
+
}
|
|
@@ -0,0 +1,5 @@
|
|
|
1
|
+
from video_sum_service.knowledge.index_service import KnowledgeIndexService
|
|
2
|
+
from video_sum_service.knowledge.rag_service import RagService
|
|
3
|
+
from video_sum_service.knowledge.tag_service import TagService
|
|
4
|
+
|
|
5
|
+
__all__ = ["KnowledgeIndexService", "RagService", "TagService"]
|
|
@@ -0,0 +1,408 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import importlib
|
|
4
|
+
import importlib.util
|
|
5
|
+
import json
|
|
6
|
+
import logging
|
|
7
|
+
import re
|
|
8
|
+
import sys
|
|
9
|
+
from datetime import datetime, timezone
|
|
10
|
+
from pathlib import Path
|
|
11
|
+
from typing import Any
|
|
12
|
+
|
|
13
|
+
from fastapi import HTTPException
|
|
14
|
+
|
|
15
|
+
from video_sum_infra.config import ServiceSettings
|
|
16
|
+
from video_sum_infra.runtime import activate_runtime_pythonpath, app_data_root
|
|
17
|
+
from video_sum_service.repository import SqliteTaskRepository
|
|
18
|
+
from video_sum_service.schemas import KnowledgeIndexChunkRecord, KnowledgeSearchResult
|
|
19
|
+
|
|
20
|
+
logger = logging.getLogger("video_sum_service.knowledge")
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def format_anchor_seconds(seconds: float | None) -> str | None:
|
|
24
|
+
if seconds is None:
|
|
25
|
+
return None
|
|
26
|
+
total = max(0, int(seconds))
|
|
27
|
+
minutes, sec = divmod(total, 60)
|
|
28
|
+
hours, minutes = divmod(minutes, 60)
|
|
29
|
+
if hours:
|
|
30
|
+
return f"{hours:02d}:{minutes:02d}:{sec:02d}"
|
|
31
|
+
return f"{minutes:02d}:{sec:02d}"
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
class KnowledgeIndexService:
|
|
35
|
+
def __init__(
|
|
36
|
+
self,
|
|
37
|
+
repository: SqliteTaskRepository,
|
|
38
|
+
settings: ServiceSettings,
|
|
39
|
+
chroma_path: str | Path | None = None,
|
|
40
|
+
model_name: str = "BAAI/bge-small-zh-v1.5",
|
|
41
|
+
) -> None:
|
|
42
|
+
self._repository = repository
|
|
43
|
+
self._settings = settings
|
|
44
|
+
self._model_name = model_name
|
|
45
|
+
self._chroma_path = Path(chroma_path or (app_data_root() / "knowledge_index"))
|
|
46
|
+
self._embedder = None
|
|
47
|
+
self._collection = None
|
|
48
|
+
|
|
49
|
+
def _ensure_runtime_import_path(self) -> None:
|
|
50
|
+
activate_runtime_pythonpath(self._settings.runtime_channel)
|
|
51
|
+
|
|
52
|
+
def _short_error(self, exc: BaseException, *, limit: int = 500) -> str:
|
|
53
|
+
message = str(exc).strip() or exc.__class__.__name__
|
|
54
|
+
return message if len(message) <= limit else f"{message[:limit]}..."
|
|
55
|
+
|
|
56
|
+
def _import_runtime_dependency(self, module_name: str, package_label: str, missing_detail: str):
|
|
57
|
+
self._ensure_runtime_import_path()
|
|
58
|
+
spec = importlib.util.find_spec(module_name)
|
|
59
|
+
if spec is None:
|
|
60
|
+
logger.warning(
|
|
61
|
+
"knowledge dependency missing package=%s module=%s runtime_channel=%s sys_path_tail=%s",
|
|
62
|
+
package_label,
|
|
63
|
+
module_name,
|
|
64
|
+
self._settings.runtime_channel,
|
|
65
|
+
sys.path[-5:],
|
|
66
|
+
)
|
|
67
|
+
raise HTTPException(status_code=500, detail=missing_detail)
|
|
68
|
+
try:
|
|
69
|
+
return importlib.import_module(module_name)
|
|
70
|
+
except Exception as exc:
|
|
71
|
+
logger.exception(
|
|
72
|
+
"knowledge dependency import failed package=%s module=%s runtime_channel=%s origin=%s sys_path_tail=%s",
|
|
73
|
+
package_label,
|
|
74
|
+
module_name,
|
|
75
|
+
self._settings.runtime_channel,
|
|
76
|
+
getattr(spec, "origin", ""),
|
|
77
|
+
sys.path[-5:],
|
|
78
|
+
)
|
|
79
|
+
raise HTTPException(
|
|
80
|
+
status_code=500,
|
|
81
|
+
detail=(
|
|
82
|
+
f"{package_label} 已安装但加载失败:{self._short_error(exc)}。"
|
|
83
|
+
"请在设置的知识库板块重新安装依赖,或先同步当前 runtime。"
|
|
84
|
+
),
|
|
85
|
+
) from exc
|
|
86
|
+
|
|
87
|
+
def _get_embedder(self):
|
|
88
|
+
if self._embedder is None:
|
|
89
|
+
sentence_transformers = self._import_runtime_dependency(
|
|
90
|
+
"sentence_transformers",
|
|
91
|
+
"sentence-transformers",
|
|
92
|
+
"缺少 sentence-transformers 依赖,无法构建知识库索引。",
|
|
93
|
+
)
|
|
94
|
+
try:
|
|
95
|
+
self._embedder = sentence_transformers.SentenceTransformer(self._model_name)
|
|
96
|
+
except Exception as exc:
|
|
97
|
+
logger.exception(
|
|
98
|
+
"knowledge embedding model load failed model=%s runtime_channel=%s",
|
|
99
|
+
self._model_name,
|
|
100
|
+
self._settings.runtime_channel,
|
|
101
|
+
)
|
|
102
|
+
raise HTTPException(
|
|
103
|
+
status_code=500,
|
|
104
|
+
detail=(
|
|
105
|
+
f"知识库向量模型加载失败:{self._short_error(exc)}。"
|
|
106
|
+
"请检查网络 / HuggingFace 缓存,或稍后重试。"
|
|
107
|
+
),
|
|
108
|
+
) from exc
|
|
109
|
+
return self._embedder
|
|
110
|
+
|
|
111
|
+
def _get_collection(self):
|
|
112
|
+
if self._collection is None:
|
|
113
|
+
chromadb = self._import_runtime_dependency(
|
|
114
|
+
"chromadb",
|
|
115
|
+
"chromadb",
|
|
116
|
+
"缺少 chromadb 依赖,无法构建知识库索引。",
|
|
117
|
+
)
|
|
118
|
+
try:
|
|
119
|
+
self._chroma_path.mkdir(parents=True, exist_ok=True)
|
|
120
|
+
client = chromadb.PersistentClient(path=str(self._chroma_path))
|
|
121
|
+
self._collection = client.get_or_create_collection(
|
|
122
|
+
name="bilisum_knowledge",
|
|
123
|
+
metadata={"hnsw:space": "cosine"},
|
|
124
|
+
)
|
|
125
|
+
except Exception as exc:
|
|
126
|
+
logger.exception(
|
|
127
|
+
"knowledge chromadb collection init failed path=%s runtime_channel=%s",
|
|
128
|
+
self._chroma_path,
|
|
129
|
+
self._settings.runtime_channel,
|
|
130
|
+
)
|
|
131
|
+
raise HTTPException(
|
|
132
|
+
status_code=500,
|
|
133
|
+
detail=f"知识库索引存储初始化失败:{self._short_error(exc)}。",
|
|
134
|
+
) from exc
|
|
135
|
+
return self._collection
|
|
136
|
+
|
|
137
|
+
def _embed_texts(self, texts: list[str]) -> list[list[float]]:
|
|
138
|
+
if not texts:
|
|
139
|
+
return []
|
|
140
|
+
vectors = self._get_embedder().encode(texts, normalize_embeddings=True)
|
|
141
|
+
return [list(map(float, vector)) for vector in vectors]
|
|
142
|
+
|
|
143
|
+
def _split_markdown_sections(self, markdown: str) -> list[tuple[str, str]]:
|
|
144
|
+
content = str(markdown or "").strip()
|
|
145
|
+
if not content:
|
|
146
|
+
return []
|
|
147
|
+
sections: list[tuple[str, str]] = []
|
|
148
|
+
current_title = "知识笔记"
|
|
149
|
+
current_lines: list[str] = []
|
|
150
|
+
for line in content.splitlines():
|
|
151
|
+
if re.match(r"^\s{0,3}#{1,3}\s+", line):
|
|
152
|
+
if current_lines:
|
|
153
|
+
section_body = "\n".join(current_lines).strip()
|
|
154
|
+
if section_body:
|
|
155
|
+
sections.append((current_title, section_body))
|
|
156
|
+
current_title = re.sub(r"^\s{0,3}#{1,3}\s+", "", line).strip() or "知识笔记"
|
|
157
|
+
current_lines = []
|
|
158
|
+
continue
|
|
159
|
+
current_lines.append(line)
|
|
160
|
+
if current_lines:
|
|
161
|
+
section_body = "\n".join(current_lines).strip()
|
|
162
|
+
if section_body:
|
|
163
|
+
sections.append((current_title, section_body))
|
|
164
|
+
return sections
|
|
165
|
+
|
|
166
|
+
def _build_chunks_for_video(self, video_id: str) -> tuple[Any, list[KnowledgeIndexChunkRecord]]:
|
|
167
|
+
asset = self._repository.get_video_asset(video_id)
|
|
168
|
+
if asset is None or asset.latest_result is None:
|
|
169
|
+
return None, []
|
|
170
|
+
result = asset.latest_result
|
|
171
|
+
task = self._repository.get_task(asset.latest_task_id) if asset.latest_task_id else None
|
|
172
|
+
page_title = str(task.page_title or task.task_input.title or "").strip() if task is not None else ""
|
|
173
|
+
display_title = page_title or asset.title
|
|
174
|
+
now = datetime.now(timezone.utc)
|
|
175
|
+
chunk_specs: list[dict[str, object]] = []
|
|
176
|
+
|
|
177
|
+
overview_parts = [str(result.overview or "").strip(), *[str(item).strip() for item in result.key_points if str(item).strip()]]
|
|
178
|
+
overview_text = "\n".join([part for part in overview_parts if part]).strip()
|
|
179
|
+
if overview_text:
|
|
180
|
+
chunk_specs.append(
|
|
181
|
+
{
|
|
182
|
+
"index_type": "video_summary",
|
|
183
|
+
"segment_order": 0,
|
|
184
|
+
"anchor_label": None,
|
|
185
|
+
"anchor_seconds": None,
|
|
186
|
+
"content": f"{display_title}\n{overview_text}",
|
|
187
|
+
}
|
|
188
|
+
)
|
|
189
|
+
|
|
190
|
+
for index, chapter in enumerate(result.timeline):
|
|
191
|
+
if not isinstance(chapter, dict):
|
|
192
|
+
continue
|
|
193
|
+
title = str(chapter.get("title") or f"章节 {index + 1}").strip()
|
|
194
|
+
summary = str(chapter.get("summary") or "").strip()
|
|
195
|
+
start_raw = chapter.get("start")
|
|
196
|
+
start = float(start_raw) if isinstance(start_raw, (int, float)) else None
|
|
197
|
+
content = "\n".join([display_title, title, summary]).strip()
|
|
198
|
+
if summary:
|
|
199
|
+
chunk_specs.append(
|
|
200
|
+
{
|
|
201
|
+
"index_type": "chapter",
|
|
202
|
+
"segment_order": index + 1,
|
|
203
|
+
"anchor_label": title,
|
|
204
|
+
"anchor_seconds": start,
|
|
205
|
+
"content": content,
|
|
206
|
+
}
|
|
207
|
+
)
|
|
208
|
+
|
|
209
|
+
for index, (title, body) in enumerate(self._split_markdown_sections(result.knowledge_note_markdown)):
|
|
210
|
+
chunk_specs.append(
|
|
211
|
+
{
|
|
212
|
+
"index_type": "knowledge_note",
|
|
213
|
+
"segment_order": index + 1,
|
|
214
|
+
"anchor_label": title,
|
|
215
|
+
"anchor_seconds": None,
|
|
216
|
+
"content": f"{display_title}\n{title}\n{body}".strip(),
|
|
217
|
+
}
|
|
218
|
+
)
|
|
219
|
+
|
|
220
|
+
chunk_specs = [item for item in chunk_specs if str(item["content"]).strip()]
|
|
221
|
+
if not chunk_specs:
|
|
222
|
+
return asset, []
|
|
223
|
+
|
|
224
|
+
vectors = self._embed_texts([str(item["content"]) for item in chunk_specs])
|
|
225
|
+
chunks = [
|
|
226
|
+
KnowledgeIndexChunkRecord(
|
|
227
|
+
chunk_id=f"{video_id}:{item['index_type']}:{item['segment_order'] or 0}",
|
|
228
|
+
video_id=video_id,
|
|
229
|
+
embedding_json=json.dumps(vectors[index], ensure_ascii=False),
|
|
230
|
+
indexed_content=str(item["content"]),
|
|
231
|
+
index_type=str(item["index_type"]),
|
|
232
|
+
segment_order=int(item["segment_order"]) if item["segment_order"] is not None else None,
|
|
233
|
+
anchor_label=str(item["anchor_label"]) if item["anchor_label"] else None,
|
|
234
|
+
anchor_seconds=float(item["anchor_seconds"]) if item["anchor_seconds"] is not None else None,
|
|
235
|
+
created_at=now,
|
|
236
|
+
updated_at=now,
|
|
237
|
+
)
|
|
238
|
+
for index, item in enumerate(chunk_specs)
|
|
239
|
+
]
|
|
240
|
+
return asset, chunks
|
|
241
|
+
|
|
242
|
+
def index_video(self, video_id: str, content: str | None = None) -> bool:
|
|
243
|
+
del content
|
|
244
|
+
asset, chunks = self._build_chunks_for_video(video_id)
|
|
245
|
+
if asset is None:
|
|
246
|
+
return False
|
|
247
|
+
|
|
248
|
+
collection = self._get_collection()
|
|
249
|
+
try:
|
|
250
|
+
collection.delete(where={"video_id": video_id})
|
|
251
|
+
except Exception:
|
|
252
|
+
pass
|
|
253
|
+
|
|
254
|
+
self._repository.replace_knowledge_chunks(video_id, chunks)
|
|
255
|
+
if not chunks:
|
|
256
|
+
return True
|
|
257
|
+
|
|
258
|
+
tags = [item.tag for item in self._repository.list_video_tags(video_id)]
|
|
259
|
+
task = self._repository.get_task(asset.latest_task_id) if asset.latest_task_id else None
|
|
260
|
+
page_number = int(task.page_number) if task is not None and task.page_number is not None else None
|
|
261
|
+
page_title = str(task.page_title or task.task_input.title or "").strip() if task is not None else ""
|
|
262
|
+
collection.add(
|
|
263
|
+
ids=[chunk.chunk_id for chunk in chunks],
|
|
264
|
+
documents=[chunk.indexed_content for chunk in chunks],
|
|
265
|
+
embeddings=[json.loads(chunk.embedding_json) for chunk in chunks],
|
|
266
|
+
metadatas=[
|
|
267
|
+
{
|
|
268
|
+
"video_id": chunk.video_id,
|
|
269
|
+
"index_type": chunk.index_type,
|
|
270
|
+
"anchor_label": chunk.anchor_label or "",
|
|
271
|
+
"anchor_seconds": float(chunk.anchor_seconds) if chunk.anchor_seconds is not None else -1.0,
|
|
272
|
+
"title": asset.title,
|
|
273
|
+
"page_number": page_number if page_number is not None else -1,
|
|
274
|
+
"page_title": page_title,
|
|
275
|
+
"display_title": page_title or asset.title,
|
|
276
|
+
"cover_url": asset.cover_url or "",
|
|
277
|
+
"tags_json": json.dumps(tags, ensure_ascii=False),
|
|
278
|
+
}
|
|
279
|
+
for chunk in chunks
|
|
280
|
+
],
|
|
281
|
+
)
|
|
282
|
+
return True
|
|
283
|
+
|
|
284
|
+
def remove_video(self, video_id: str) -> bool:
|
|
285
|
+
self._repository.delete_knowledge_chunks(video_id)
|
|
286
|
+
try:
|
|
287
|
+
self._get_collection().delete(where={"video_id": video_id})
|
|
288
|
+
except HTTPException:
|
|
289
|
+
raise
|
|
290
|
+
except Exception:
|
|
291
|
+
pass
|
|
292
|
+
return True
|
|
293
|
+
|
|
294
|
+
def rebuild_index(self, force: bool = False) -> int:
|
|
295
|
+
del force
|
|
296
|
+
videos = [video for video in self._repository.list_video_assets() if video.latest_result is not None]
|
|
297
|
+
collection = self._get_collection()
|
|
298
|
+
try:
|
|
299
|
+
existing = collection.get()
|
|
300
|
+
ids = existing.get("ids") if isinstance(existing, dict) else None
|
|
301
|
+
if ids:
|
|
302
|
+
collection.delete(ids=ids)
|
|
303
|
+
except Exception:
|
|
304
|
+
pass
|
|
305
|
+
|
|
306
|
+
self._repository.clear_knowledge_chunks()
|
|
307
|
+
count = 0
|
|
308
|
+
for video in videos:
|
|
309
|
+
if self.index_video(video.video_id):
|
|
310
|
+
count += 1
|
|
311
|
+
return count
|
|
312
|
+
|
|
313
|
+
def _fetch_candidate_chunks(self, query: str, limit: int = 10, tag_filter: list[str] | None = None) -> list[dict[str, object]]:
|
|
314
|
+
cleaned_query = str(query or "").strip()
|
|
315
|
+
if not cleaned_query:
|
|
316
|
+
return []
|
|
317
|
+
if self._repository.get_knowledge_chunk_count() == 0:
|
|
318
|
+
self.rebuild_index()
|
|
319
|
+
|
|
320
|
+
effective_limit = max(limit * 4, 12)
|
|
321
|
+
query_embedding = self._embed_texts([cleaned_query])[0]
|
|
322
|
+
query_result = self._get_collection().query(
|
|
323
|
+
query_embeddings=[query_embedding],
|
|
324
|
+
n_results=effective_limit,
|
|
325
|
+
)
|
|
326
|
+
ids = query_result.get("ids", [[]])[0] if isinstance(query_result, dict) else []
|
|
327
|
+
documents = query_result.get("documents", [[]])[0] if isinstance(query_result, dict) else []
|
|
328
|
+
metadatas = query_result.get("metadatas", [[]])[0] if isinstance(query_result, dict) else []
|
|
329
|
+
distances = query_result.get("distances", [[]])[0] if isinstance(query_result, dict) else []
|
|
330
|
+
|
|
331
|
+
allowed_video_ids: set[str] | None = None
|
|
332
|
+
if tag_filter:
|
|
333
|
+
selected = {tag.strip() for tag in tag_filter if str(tag).strip()}
|
|
334
|
+
if selected:
|
|
335
|
+
allowed_video_ids = {
|
|
336
|
+
item.video_id
|
|
337
|
+
for item in self._repository.list_all_video_tags()
|
|
338
|
+
if item.tag in selected
|
|
339
|
+
}
|
|
340
|
+
|
|
341
|
+
candidates: list[dict[str, object]] = []
|
|
342
|
+
for index, chunk_id in enumerate(ids):
|
|
343
|
+
metadata = metadatas[index] if index < len(metadatas) and isinstance(metadatas[index], dict) else {}
|
|
344
|
+
video_id = str(metadata.get("video_id") or "")
|
|
345
|
+
if allowed_video_ids is not None and video_id not in allowed_video_ids:
|
|
346
|
+
continue
|
|
347
|
+
distance = float(distances[index]) if index < len(distances) else 1.0
|
|
348
|
+
relevance = max(0.0, 1.0 - distance / 2.0)
|
|
349
|
+
candidates.append(
|
|
350
|
+
{
|
|
351
|
+
"chunk_id": chunk_id,
|
|
352
|
+
"video_id": video_id,
|
|
353
|
+
"document": documents[index] if index < len(documents) else "",
|
|
354
|
+
"metadata": metadata,
|
|
355
|
+
"relevance_score": round(relevance, 4),
|
|
356
|
+
}
|
|
357
|
+
)
|
|
358
|
+
return candidates
|
|
359
|
+
|
|
360
|
+
def search(self, query: str, limit: int = 10, tag_filter: list[str] | None = None) -> list[KnowledgeSearchResult]:
|
|
361
|
+
candidates = self._fetch_candidate_chunks(query, limit=limit, tag_filter=tag_filter)
|
|
362
|
+
if not candidates:
|
|
363
|
+
return []
|
|
364
|
+
|
|
365
|
+
grouped: dict[str, dict[str, object]] = {}
|
|
366
|
+
for candidate in candidates:
|
|
367
|
+
video_id = str(candidate["video_id"])
|
|
368
|
+
current = grouped.get(video_id)
|
|
369
|
+
if current is None or float(candidate["relevance_score"]) > float(current["relevance_score"]):
|
|
370
|
+
grouped[video_id] = candidate
|
|
371
|
+
|
|
372
|
+
ordered = sorted(grouped.values(), key=lambda item: float(item["relevance_score"]), reverse=True)[: max(1, limit)]
|
|
373
|
+
results: list[KnowledgeSearchResult] = []
|
|
374
|
+
for candidate in ordered:
|
|
375
|
+
video_id = str(candidate["video_id"])
|
|
376
|
+
asset = self._repository.get_video_asset(video_id)
|
|
377
|
+
metadata = candidate["metadata"] if isinstance(candidate["metadata"], dict) else {}
|
|
378
|
+
video_title = asset.title if asset is not None else str(metadata.get("title") or "未知视频")
|
|
379
|
+
page_title = str(metadata.get("page_title") or metadata.get("display_title") or "").strip()
|
|
380
|
+
if page_title == video_title:
|
|
381
|
+
page_title = ""
|
|
382
|
+
page_number_raw = metadata.get("page_number")
|
|
383
|
+
page_number = int(page_number_raw) if isinstance(page_number_raw, (int, float)) and int(page_number_raw) > 0 else None
|
|
384
|
+
snippet = str(candidate["document"]).strip().replace("\n", " ")
|
|
385
|
+
snippet = snippet[:180].rstrip() + ("..." if len(snippet) > 180 else "")
|
|
386
|
+
tags = [item.tag for item in self._repository.list_video_tags(video_id)]
|
|
387
|
+
results.append(
|
|
388
|
+
KnowledgeSearchResult(
|
|
389
|
+
video_id=video_id,
|
|
390
|
+
title=page_title or video_title,
|
|
391
|
+
relevance_score=float(candidate["relevance_score"]),
|
|
392
|
+
snippet=snippet,
|
|
393
|
+
tags=tags,
|
|
394
|
+
cover_url=asset.cover_url if asset is not None else str(metadata.get("cover_url") or ""),
|
|
395
|
+
timestamp=format_anchor_seconds(
|
|
396
|
+
float(metadata["anchor_seconds"])
|
|
397
|
+
if metadata.get("anchor_seconds") not in {None, "", -1, -1.0}
|
|
398
|
+
else None
|
|
399
|
+
),
|
|
400
|
+
video_title=video_title,
|
|
401
|
+
page_title=page_title or None,
|
|
402
|
+
page_number=page_number,
|
|
403
|
+
)
|
|
404
|
+
)
|
|
405
|
+
return results
|
|
406
|
+
|
|
407
|
+
def search_chunks(self, query: str, limit: int = 5, tag_filter: list[str] | None = None) -> list[dict[str, object]]:
|
|
408
|
+
return self._fetch_candidate_chunks(query, limit=limit, tag_filter=tag_filter)[: max(1, limit)]
|