free-short-video 6.3.0 → 6.4.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.example +10 -0
- package/README.md +2 -25
- package/core/api/rate_limiter.py +12 -7
- package/core/audio/subtitle/generator.py +15 -0
- package/core/audio/tashkeel.py +117 -0
- package/core/audio/voices.py +59 -0
- package/core/compositor/concatenator/audio_overlay.py +10 -9
- package/core/compositor/concatenator/concat.py +10 -3
- package/core/compositor/ffmpeg_tool.py +102 -0
- package/core/compositor/processor.py +6 -4
- package/core/compositor/watermark.py +2 -1
- package/core/config.py +20 -2
- package/core/pipelines/creative/pipeline.py +23 -1
- package/core/pipelines/creative/steps_audio.py +26 -9
- package/core/pipelines/creative/steps_script.py +2 -2
- package/core/pipelines/manuscript_video.py +135 -64
- package/core/screenwriter/__init__.py +66 -0
- package/core/screenwriter/story.py +23 -7
- package/images/home.png +0 -0
- package/models/task.py +5 -1
- package/package.json +1 -1
- package/requirements.txt +7 -0
- package/server.py +47 -0
- package/static/assets/index-Cg4TCtV5.js +45 -0
- package/static/index.html +1 -1
- package/web/routes/config_routes.py +13 -5
- package/web/routes/image_routes.py +1 -1
- package/web/routes/preview_routes.py +279 -0
- package/web/routes/task_creation_routes.py +42 -0
- package/static/assets/index-SVQHMn30.js +0 -45
package/.env.example
CHANGED
|
@@ -77,3 +77,13 @@ PORT=8765
|
|
|
77
77
|
|
|
78
78
|
# 提示词语言(zh/en,影响 LLM meta-prompt 语言)
|
|
79
79
|
# PROMPT_LANGUAGE=zh
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
# ── 7. CORS 跨源白名单(可选,供独立本地伴侣工具调用本服务 API)──────────
|
|
83
|
+
# 场景:本地跑一个独立的简化前端(如 agnes-simple-ui,监听 :8787),
|
|
84
|
+
# 从浏览器直接调用本服务的 /api/* 接口。设置后服务仅允许列出的源跨源调用。
|
|
85
|
+
# 默认空 = 不启用 CORS(攻击面不变);同源页面不受影响。
|
|
86
|
+
# AGNES_CORS_ORIGINS=http://localhost:8787,http://127.0.0.1:3000
|
|
87
|
+
#
|
|
88
|
+
# 显式禁用(即使设置了 origins 也不启用中间件):
|
|
89
|
+
# AGNES_CORS_ENABLED=false
|
package/README.md
CHANGED
|
@@ -1,36 +1,13 @@
|
|
|
1
1
|
---
|
|
2
2
|
|
|
3
|
-
# What's New in v6.
|
|
3
|
+
# What's New in v6.4.1
|
|
4
4
|
|
|
5
5
|
## What's New
|
|
6
6
|
|
|
7
|
-
### Features & Improvements
|
|
8
|
-
|
|
9
|
-
- **Complete v6 optimization roadmap (29/29 items)** — every batch of the v6 roadmap is now shipped:
|
|
10
|
-
- **Performance (batch 2)**: the final compositing chain is now ffmpeg-based — identical-parameter scene concatenation uses `-c copy`, audio alignment/volume/silence-padding merge into a single filter pass, and subtitles render through the ASS path with per-entry styles (`AGNES_SUBTITLE_ASS`, with automatic fallback to the moviepy path). Poetry videos compose all scenes in one pass instead of re-encoding per scene. A dedicated encoding thread pool isolates heavy ffmpeg/moviepy work from API requests, and the token-bucket rate limiter gained a native async path so stopping a task during rate-limit waits is instant.
|
|
11
|
-
- **Reliability & engineering (batch 1)**: task state follows a single-writer principle with per-task locking, resume supports persisted word-level TTS cues (no re-synthesis on resume), video polling is adaptive and multi-scene waits run concurrently, task listing is indexed with `limit/offset/status` pagination, stale artifacts/error logs are governed, and the frontend stops polling in background tabs with exponential backoff and a connection-loss banner.
|
|
12
|
-
- **Frontend & i18n**: translations are split into per-language lazy-loaded chunks — the first-screen JS bundle drops from ~721 kB to ~305 kB (gzip 226 kB → 97 kB, **-58%**). Form submission/confirm/toast flows were unified into shared composables, mobile layout, focus-trap modals, `prefers-reduced-motion` and form drafts were added.
|
|
13
|
-
- **Observability & ops (batch 3)**: new `GET /api/health` and `GET /api/metrics` endpoints, optional rotating file logging (`AGNES_LOG_FILE`), and a Docker `HEALTHCHECK`. Runtime settings are now converged through typed `pydantic-settings` (with `.env` support) so concurrency limits scale dynamically with API-key count.
|
|
14
|
-
- **Immediate defect fixes (batch 0)**: stop now cancels instantly without retry backoff, event-loop blocking (watermark re-encode, sync downloads) is moved off the loop, multi-key delete works correctly, a frontend `v-html` XSS vector is closed, and image generation got a duplicate-submit guard.
|
|
15
|
-
- **Full 22-language support incl. Arabic** — the UI already had 22 languages; this release completes the voice catalog for all of them. Arabic UI is fully supported (PR #32), and 8 UI languages (Turkish, Vietnamese, Thai, Tagalog, Hindi, Persian, Bengali, Urdu) now have edge_tts voice groupings with native-voice name display, script-detection regexes (Thai/Devanagari/Bengali) and per-script subtitle font fallback (new bundled Noto fonts; Persian/Urdu reuse the Arabic reshape+bidi pipeline).
|
|
16
|
-
- **Transparent analytics disclosure & privacy controls** — the settings panel now shows a clear, collapsible privacy card listing exactly what usage statistics are reported (and what is never uploaded: prompts, manuscripts, poems, API keys and reference images are redacted before reporting). Analytics can be turned off entirely from the panel.
|
|
17
|
-
- **Complete error tracebacks in the feedback report** — pipeline failures now persist the full `traceback` into the task state; the diagnostics endpoint and the in-app feedback report include it, so you can paste complete error details (e.g. environment-level `[WinError 2]`) into GitHub issues without checking the server console.
|
|
18
|
-
|
|
19
|
-
### Refactoring & Optimizations
|
|
20
|
-
|
|
21
|
-
- **ffmpeg-first compositing chain** — the final assembly path for creative/manuscript/anchor/poetry videos was reworked from 3-4 full re-encodes into copy-concat + a single filter pass (with graceful fallback to the previous moviepy path). This is the largest performance win in the v6 line, cutting final-assembly time by roughly 3-10x on typical outputs.
|
|
22
|
-
- **Asynchronous rate limiting with dedicated encoding thread pool** — the token bucket now offers a native async acquire path (stop-aware), and heavy encoding runs on a dedicated executor so long encoding jobs no longer starve the request path.
|
|
23
|
-
|
|
24
7
|
### Bug Fixes
|
|
25
8
|
|
|
26
|
-
|
|
27
|
-
- **Fixed multi-Key configuration** — key IDs are now hashed from the actual key so deleting one Key from multiple configured Keys removes exactly that Key.
|
|
28
|
-
- **Fixed event-loop freezes** — watermark re-encoding and synchronous downloads no longer block the whole service; a semaphore release bug that could permanently break the concurrency cap under low-rate-limit configurations is fixed.
|
|
29
|
-
- **Fixed frontend issues** — a stored-XSS vector via unescaped `v-html` is closed, duplicate image-submit without guard is prevented, and fetch errors now surface readable backend messages instead of silent failures.
|
|
30
|
-
|
|
31
|
-
---
|
|
9
|
+
* **Fixed "ffmpeg not found" failure (`FileNotFoundError`) during video composition** — on Windows or systems where `ffmpeg` is not on the `PATH`, creative / manuscript / anchor / poetry composition (concatenation, audio overlay, watermark) could fail at the composition step with `[WinError 2]`. The server now automatically falls back to the static ffmpeg binary bundled with `imageio-ffmpeg`, so no manual ffmpeg installation is required. A non-blocking startup check logs which ffmpeg source is being used. (Issues #35, #36)
|
|
32
10
|
|
|
33
|
-
No configuration changes are required. Existing tasks remain resumable; task state files are unchanged in format.
|
|
34
11
|
|
|
35
12
|
---
|
|
36
13
|
|
package/core/api/rate_limiter.py
CHANGED
|
@@ -167,14 +167,19 @@ class AgnesRateLimiter:
|
|
|
167
167
|
def acquire(self) -> None:
|
|
168
168
|
"""阻塞式获取一个令牌(同步场景 / 脚本 / 测试用)。
|
|
169
169
|
|
|
170
|
-
如果桶中有令牌,立即消耗并返回;否则 ``time.sleep()``
|
|
170
|
+
如果桶中有令牌,立即消耗并返回;否则 ``time.sleep()`` 等待令牌可用。
|
|
171
|
+
|
|
172
|
+
⚠️ 修复(回归活锁):此处采用与 ``acquire_async`` 一致的**预支语义**——
|
|
173
|
+
``_try_acquire`` 已把 ``last_refill`` 预留到 ``now + wait_time`` 并清空令牌,
|
|
174
|
+
sleep 期间令牌尚未重新累积,再次循环调用 ``_try_acquire`` 会算出 ``elapsed≈0``
|
|
175
|
+
而**永远不足 1 个令牌**,导致令牌永不补充、等待者永久卡死(多并发时逐个推挤
|
|
176
|
+
``last_refill`` 还使 wait_time 越滚越大)。因此 sleep 完成后直接视为已获取。
|
|
171
177
|
"""
|
|
172
|
-
|
|
173
|
-
|
|
174
|
-
|
|
175
|
-
|
|
176
|
-
|
|
177
|
-
time.sleep(wait_time)
|
|
178
|
+
wait_time = self._try_acquire()
|
|
179
|
+
if wait_time is None:
|
|
180
|
+
return
|
|
181
|
+
self._record_wait(wait_time)
|
|
182
|
+
time.sleep(wait_time)
|
|
178
183
|
|
|
179
184
|
async def acquire_async(self, stop_event: asyncio.Event | None = None) -> None:
|
|
180
185
|
"""异步原生获取令牌(优化路线图 2.3)。
|
|
@@ -253,6 +253,14 @@ class SubtitleSrtMixin:
|
|
|
253
253
|
if text:
|
|
254
254
|
items.append((start_s, end_s, text))
|
|
255
255
|
|
|
256
|
+
# PRD 1.2a:TTS 若送入了加 tashkeel 的阿拉伯语文本,cues 文本会连带
|
|
257
|
+
# 变音符号——字幕显示前统一剥离(仅阿拉伯语变音符号,其他语言无副作用)。
|
|
258
|
+
from core.audio.tashkeel import strip_diacritics
|
|
259
|
+
|
|
260
|
+
items = [(s, e, strip_diacritics(t)) for s, e, t in items]
|
|
261
|
+
if not items:
|
|
262
|
+
return ""
|
|
263
|
+
|
|
256
264
|
if not items:
|
|
257
265
|
return ""
|
|
258
266
|
|
|
@@ -594,6 +602,13 @@ class SubtitleSrtMixin:
|
|
|
594
602
|
if not items:
|
|
595
603
|
return ""
|
|
596
604
|
|
|
605
|
+
# PRD 1.2a:TTS 若送入了加 tashkeel 的阿拉伯语文本,cues 文本会连带
|
|
606
|
+
# 变音符号——字幕显示前统一剥离(仅阿拉伯语变音符号,其他语言无副作用)。
|
|
607
|
+
# 剥离后再做归一化对齐,保证与 plain 的 segment_texts 策略 A 字符区间匹配。
|
|
608
|
+
from core.audio.tashkeel import strip_diacritics
|
|
609
|
+
|
|
610
|
+
items = [(s, e, strip_diacritics(t)) for s, e, t in items]
|
|
611
|
+
|
|
597
612
|
# 残余归一化:把最后一条 cue 的 end 钳到实际音频时长(避免尾差留白未覆盖)
|
|
598
613
|
if audio_duration and audio_duration > 0 and items[-1][1] > audio_duration:
|
|
599
614
|
items[-1] = (items[-1][0], audio_duration, items[-1][2])
|
|
@@ -0,0 +1,117 @@
|
|
|
1
|
+
"""阿拉伯语变音符号(tashkeel/harakat)自动标注,供 TTS 更准确地朗读。
|
|
2
|
+
|
|
3
|
+
使用 mishkal(基于规则的阿拉伯语形态分析库)而非 LLM 来生成变音符号:
|
|
4
|
+
LLM 方案实测不可靠——即使明确要求"只加符号、不改字母",模型仍会偶发替换
|
|
5
|
+
借词拼写(如 فيديو -> ويديو)或替换同义连词(如 إذا -> إن),这对旁白配音
|
|
6
|
+
是不可接受的(读出的内容会与原文不同)。mishkal 只做形态学标注,天然不会
|
|
7
|
+
引入新字母,因此改用"取 mishkal 的变音符号 + 保留原文逐字符结构"的合并算法,
|
|
8
|
+
以程序方式保证结果与原文在去除变音符号后完全一致。
|
|
9
|
+
|
|
10
|
+
依赖模式:**optional-dependency**(与 json-repair 相同的"缺失自动降级"模式)。
|
|
11
|
+
本模块在导入时不加载 mishkal(``_get_vocalizer`` 内延迟导入),因此即使
|
|
12
|
+
mishkal 未安装,``strip_diacritics`` 等公共函数仍可用;``add_tashkeel_safe``
|
|
13
|
+
在 mishkal 缺失时静默回退原文。``requirements.txt`` 默认安装 mishkal 保证
|
|
14
|
+
开箱即用。
|
|
15
|
+
|
|
16
|
+
Port: PR #33 by @Khaled97Sho(新增公共 ``strip_diacritics`` 供字幕隔离复用)。
|
|
17
|
+
"""
|
|
18
|
+
from __future__ import annotations
|
|
19
|
+
|
|
20
|
+
import logging
|
|
21
|
+
import re
|
|
22
|
+
|
|
23
|
+
logger = logging.getLogger(__name__)
|
|
24
|
+
|
|
25
|
+
# 阿拉伯字母(U+0621–U+064A)
|
|
26
|
+
_ARABIC_LETTER_RE = re.compile(r"[ء-ي]")
|
|
27
|
+
# 阿拉伯语变音符号(harakat/tashkeel,U+064B–U+0670):fatha/damma/kasra 等,
|
|
28
|
+
# 仅标注发音、不产生语音时长。字幕与旁白导出产物必须剥离,避免污染显示。
|
|
29
|
+
_DIACRITIC_RE = re.compile(r"[ً-ْٰ]")
|
|
30
|
+
|
|
31
|
+
_vocalizer = None
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def strip_diacritics(text: str) -> str:
|
|
35
|
+
"""去除文本中的阿拉伯语变音符号(tashkeel/harakat)。
|
|
36
|
+
|
|
37
|
+
用于字幕 / 旁白导出产物:加 tashkeel 的文本只应送入 TTS,字幕显示与
|
|
38
|
+
``narration.txt`` 必须使用剥离变音符号的干净版本(PRD 1.2a)。
|
|
39
|
+
对不含变音符号的文本(其他语言)为无副作用操作。
|
|
40
|
+
|
|
41
|
+
Args:
|
|
42
|
+
text: 原始文本(可为空)。
|
|
43
|
+
|
|
44
|
+
Returns:
|
|
45
|
+
剥离阿拉伯语变音符号后的文本。
|
|
46
|
+
"""
|
|
47
|
+
if not text:
|
|
48
|
+
return text
|
|
49
|
+
return _DIACRITIC_RE.sub("", text)
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def _get_vocalizer():
|
|
53
|
+
global _vocalizer
|
|
54
|
+
if _vocalizer is None:
|
|
55
|
+
from mishkal.tashkeel import TashkeelClass
|
|
56
|
+
|
|
57
|
+
_vocalizer = TashkeelClass()
|
|
58
|
+
return _vocalizer
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def _merge_tashkeel(original: str, diacritized: str) -> str:
|
|
62
|
+
"""按原文逐字符重建结果:阿拉伯字母取自 mishkal 输出(含其变音符号),
|
|
63
|
+
其余字符(标点、空格、数字等)严格保留原文,不采用 mishkal 对它们的改写。
|
|
64
|
+
"""
|
|
65
|
+
d_tokens: list[tuple[str, str]] = []
|
|
66
|
+
i = 0
|
|
67
|
+
while i < len(diacritized):
|
|
68
|
+
ch = diacritized[i]
|
|
69
|
+
if _ARABIC_LETTER_RE.match(ch):
|
|
70
|
+
j = i + 1
|
|
71
|
+
diac = ""
|
|
72
|
+
while j < len(diacritized) and _DIACRITIC_RE.match(diacritized[j]):
|
|
73
|
+
diac += diacritized[j]
|
|
74
|
+
j += 1
|
|
75
|
+
d_tokens.append((ch, diac))
|
|
76
|
+
i = j
|
|
77
|
+
else:
|
|
78
|
+
i += 1
|
|
79
|
+
|
|
80
|
+
out = []
|
|
81
|
+
ti = 0
|
|
82
|
+
for ch in original:
|
|
83
|
+
if _ARABIC_LETTER_RE.match(ch):
|
|
84
|
+
if ti < len(d_tokens):
|
|
85
|
+
letter, diac = d_tokens[ti]
|
|
86
|
+
out.append(letter + diac)
|
|
87
|
+
ti += 1
|
|
88
|
+
else:
|
|
89
|
+
out.append(ch)
|
|
90
|
+
else:
|
|
91
|
+
out.append(ch)
|
|
92
|
+
return "".join(out)
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def add_tashkeel_safe(text: str) -> str:
|
|
96
|
+
"""为阿拉伯文文本添加变音符号,失败或校验不通过时原样返回。
|
|
97
|
+
|
|
98
|
+
校验:合并结果去除变音符号后必须与输入(同样去除变音符号后)逐字符相等,
|
|
99
|
+
否则说明 mishkal 内部出现异常输出,这种情况下放弃标注、返回原文,
|
|
100
|
+
避免把错误内容送入 TTS。
|
|
101
|
+
"""
|
|
102
|
+
if not text or not _ARABIC_LETTER_RE.search(text):
|
|
103
|
+
return text
|
|
104
|
+
|
|
105
|
+
base_text = strip_diacritics(text)
|
|
106
|
+
try:
|
|
107
|
+
vocalizer = _get_vocalizer()
|
|
108
|
+
raw_result = vocalizer.tashkeel(base_text)
|
|
109
|
+
except Exception as e:
|
|
110
|
+
logger.warning(f"[Tashkeel] mishkal failed, returning original text: {e}")
|
|
111
|
+
return text
|
|
112
|
+
|
|
113
|
+
merged = _merge_tashkeel(base_text, raw_result)
|
|
114
|
+
if strip_diacritics(merged) != base_text:
|
|
115
|
+
logger.warning("[Tashkeel] validation failed (letters changed), returning original text")
|
|
116
|
+
return text
|
|
117
|
+
return merged
|
package/core/audio/voices.py
CHANGED
|
@@ -180,6 +180,65 @@ def detect_text_script(text: str) -> str:
|
|
|
180
180
|
return "unknown"
|
|
181
181
|
|
|
182
182
|
|
|
183
|
+
# ═══════════════════════════════════════════════════
|
|
184
|
+
# 语速估算(PR #33 吸收:跨脚本朗读速率差异 + 变音符号剥离)
|
|
185
|
+
# ═══════════════════════════════════════════════════
|
|
186
|
+
# 单一公共实现(PRD 1.3a):story.py / manuscript_video.py / preview_routes
|
|
187
|
+
# 均从本模块导入,消除此前两处 4.0/13.0 常量与脚本判断的重复副本。
|
|
188
|
+
|
|
189
|
+
# CJK 字符密度高(一字近一音节),约 4 字/秒(实测 zh-CN-Xiaoxiao ≈ 4.7,取保守值)
|
|
190
|
+
_CHARS_PER_SEC_CJK = 4.0
|
|
191
|
+
# 阿拉伯文(2026-08-31 实测校准:ar-SA-Hamed/Zariyah ≈ 10.4-10.6、ar-EG-Shakir ≈ 12.0
|
|
192
|
+
# 字符/秒,真实旁白含句间停顿更低,取 10.5;PR #33 原沿用的统一 13 偏快,会导致
|
|
193
|
+
# 旁白比画面长出约 40%,视频被迫定格等待旁白结束)
|
|
194
|
+
_CHARS_PER_SEC_ARABIC = 10.5
|
|
195
|
+
# 其余字母文字(拉丁/西里尔/泰文/天城文/孟加拉文等)统一 13 字符/秒
|
|
196
|
+
# (英文实测 15.7-16.8 偏快、法/西/俄未实测,13 为折中保守值,后续可分档校准)
|
|
197
|
+
_CHARS_PER_SEC_ALPHABETIC = 13.0
|
|
198
|
+
|
|
199
|
+
|
|
200
|
+
def estimate_chars_per_sec(text: str) -> float:
|
|
201
|
+
"""按文本主要文字体系估算朗读语速(字符/秒)。
|
|
202
|
+
|
|
203
|
+
中文/日文/韩文按 CJK 速率(4.0 字/秒);阿拉伯文按实测速率(10.5 字符/秒,
|
|
204
|
+
见 ``_CHARS_PER_SEC_ARABIC`` 校准记录);其余脚本(拉丁/西里尔/泰文/
|
|
205
|
+
天城文/孟加拉文等)统一按字母文字速率(13.0 字符/秒,泰文等未实测脚本
|
|
206
|
+
先用统一值,后续可校准)。
|
|
207
|
+
|
|
208
|
+
Args:
|
|
209
|
+
text: 待估算朗读时长的文本。
|
|
210
|
+
|
|
211
|
+
Returns:
|
|
212
|
+
语速(字符/秒)。
|
|
213
|
+
"""
|
|
214
|
+
script = detect_text_script(text)
|
|
215
|
+
if script in ("zh", "ja", "ko"):
|
|
216
|
+
return _CHARS_PER_SEC_CJK
|
|
217
|
+
if script == "arabic":
|
|
218
|
+
return _CHARS_PER_SEC_ARABIC
|
|
219
|
+
return _CHARS_PER_SEC_ALPHABETIC
|
|
220
|
+
|
|
221
|
+
|
|
222
|
+
def duration_len(text: str) -> int:
|
|
223
|
+
"""用于时长估算的字符数(剥离阿拉伯语变音符号后计数)。
|
|
224
|
+
|
|
225
|
+
阿拉伯语变音符号(harakat/tashkeel)只标注发音、不产生语音时长,
|
|
226
|
+
估算前必须剥离,否则加全变音符号的文本(codepoint 增加 40-60%)
|
|
227
|
+
会让时长估算严重偏长,导致生成视频尾部大片静音/定格。
|
|
228
|
+
|
|
229
|
+
Args:
|
|
230
|
+
text: 原始文本(可为空)。
|
|
231
|
+
|
|
232
|
+
Returns:
|
|
233
|
+
剥离变音符号后的字符数。
|
|
234
|
+
"""
|
|
235
|
+
if not text:
|
|
236
|
+
return 0
|
|
237
|
+
from core.audio.tashkeel import strip_diacritics
|
|
238
|
+
|
|
239
|
+
return len(strip_diacritics(text))
|
|
240
|
+
|
|
241
|
+
|
|
183
242
|
# edge-tts voice id 前缀 → 项目语言 code 映射。
|
|
184
243
|
# 例外:Tagalog 音色在 edge-tts 中用 ISO 639-3 代码 `fil`(如 fil-PH-AngeloNeural),
|
|
185
244
|
# 而项目/前端 UI 用 ISO 639-1 `tl`,故需显式归一。
|
|
@@ -13,6 +13,7 @@ from typing import List, Optional, Tuple
|
|
|
13
13
|
import srt as srt_lib
|
|
14
14
|
from moviepy import AudioFileClip, CompositeVideoClip, VideoFileClip
|
|
15
15
|
|
|
16
|
+
from core.compositor.ffmpeg_tool import resolve_binary
|
|
16
17
|
from models.task import SubtitleStyle
|
|
17
18
|
|
|
18
19
|
from .concat import _AUDIO_BITRATE, _AUDIO_CODEC, _AUDIO_FPS, _VIDEO_FPS
|
|
@@ -137,7 +138,7 @@ class AudioOverlayMixin:
|
|
|
137
138
|
tmp_files.append(extend_path)
|
|
138
139
|
pad_dur = final_dur - video_dur
|
|
139
140
|
VideoConcatenator._run_ffmpeg(
|
|
140
|
-
["ffmpeg", "-y",
|
|
141
|
+
[resolve_binary("ffmpeg"), "-y",
|
|
141
142
|
"-i", silent_path,
|
|
142
143
|
"-vf", f"tpad=stop_mode=clone:stop_duration={pad_dur:.2f}",
|
|
143
144
|
"-c:v", "libx264", "-pix_fmt", "yuv420p",
|
|
@@ -154,7 +155,7 @@ class AudioOverlayMixin:
|
|
|
154
155
|
tmp_files.append(apad_path)
|
|
155
156
|
pad_dur = final_dur - audio_dur
|
|
156
157
|
VideoConcatenator._run_ffmpeg(
|
|
157
|
-
["ffmpeg", "-y",
|
|
158
|
+
[resolve_binary("ffmpeg"), "-y",
|
|
158
159
|
"-i", audio_path,
|
|
159
160
|
"-af", f"apad=pad_dur={pad_dur:.2f},volume=1.5",
|
|
160
161
|
"-c:a", "libmp3lame", "-q:a", "2",
|
|
@@ -167,7 +168,7 @@ class AudioOverlayMixin:
|
|
|
167
168
|
vol_path = audio_path.replace(".mp3", "_vol.mp3")
|
|
168
169
|
tmp_files.append(vol_path)
|
|
169
170
|
VideoConcatenator._run_ffmpeg(
|
|
170
|
-
["ffmpeg", "-y",
|
|
171
|
+
[resolve_binary("ffmpeg"), "-y",
|
|
171
172
|
"-i", audio_path,
|
|
172
173
|
"-af", "volume=1.5",
|
|
173
174
|
"-c:a", "libmp3lame", "-q:a", "2",
|
|
@@ -290,7 +291,7 @@ class AudioOverlayMixin:
|
|
|
290
291
|
else:
|
|
291
292
|
v_chain = f"[0:v]tpad=stop_mode=clone:stop_duration={final_dur:.2f}[v]"
|
|
292
293
|
cmd = [
|
|
293
|
-
"ffmpeg", "-y",
|
|
294
|
+
resolve_binary("ffmpeg"), "-y",
|
|
294
295
|
"-i", silent_path,
|
|
295
296
|
"-i", audio_path,
|
|
296
297
|
"-filter_complex",
|
|
@@ -315,7 +316,7 @@ class AudioOverlayMixin:
|
|
|
315
316
|
"""ffprobe 获取视频宽高,失败回退 (768, 1152)。"""
|
|
316
317
|
try:
|
|
317
318
|
r = subprocess.run(
|
|
318
|
-
["ffprobe", "-v", "error", "-select_streams", "v:0",
|
|
319
|
+
[resolve_binary("ffprobe"), "-v", "error", "-select_streams", "v:0",
|
|
319
320
|
"-show_entries", "stream=width,height",
|
|
320
321
|
"-of", "csv=s=x:p=0", video_path],
|
|
321
322
|
stdin=subprocess.DEVNULL, capture_output=True, text=True, timeout=15,
|
|
@@ -706,7 +707,7 @@ class AudioOverlayMixin:
|
|
|
706
707
|
"""
|
|
707
708
|
try:
|
|
708
709
|
n = len(audio_paths)
|
|
709
|
-
cmd = ["ffmpeg", "-y"]
|
|
710
|
+
cmd = [resolve_binary("ffmpeg"), "-y"]
|
|
710
711
|
for ap in audio_paths:
|
|
711
712
|
cmd += ["-i", ap]
|
|
712
713
|
filters = []
|
|
@@ -811,7 +812,7 @@ class AudioOverlayMixin:
|
|
|
811
812
|
|
|
812
813
|
# Step 1: Get clip duration
|
|
813
814
|
probe = subprocess.run(
|
|
814
|
-
["ffprobe", "-v", "error", "-show_entries", "format=duration",
|
|
815
|
+
[resolve_binary("ffprobe"), "-v", "error", "-show_entries", "format=duration",
|
|
815
816
|
"-of", "csv=p=0", clip_path],
|
|
816
817
|
stdin=subprocess.DEVNULL,
|
|
817
818
|
capture_output=True, text=True, timeout=15,
|
|
@@ -836,7 +837,7 @@ class AudioOverlayMixin:
|
|
|
836
837
|
# Step 4: Concatenate with xfade cross-fade transitions
|
|
837
838
|
try:
|
|
838
839
|
subprocess.run(
|
|
839
|
-
["ffmpeg", "-y", "-f", "concat", "-safe", "0",
|
|
840
|
+
[resolve_binary("ffmpeg"), "-y", "-f", "concat", "-safe", "0",
|
|
840
841
|
"-i", concat_file,
|
|
841
842
|
"-c", "copy",
|
|
842
843
|
"-t", str(needed),
|
|
@@ -850,7 +851,7 @@ class AudioOverlayMixin:
|
|
|
850
851
|
# xfade filter 构建已由上方 trim 循环拼接替代(死代码,3.3 清理)
|
|
851
852
|
|
|
852
853
|
subprocess.run(
|
|
853
|
-
["ffmpeg", "-y",
|
|
854
|
+
[resolve_binary("ffmpeg"), "-y",
|
|
854
855
|
"-stream_loop", str(n - 1), "-i", clip_path,
|
|
855
856
|
"-filter_complex",
|
|
856
857
|
f"[0:v]trim=duration={needed}[v]",
|
|
@@ -11,6 +11,7 @@ from typing import List, Optional
|
|
|
11
11
|
import srt as srt_lib
|
|
12
12
|
from moviepy import VideoFileClip, concatenate_videoclips
|
|
13
13
|
|
|
14
|
+
from core.compositor.ffmpeg_tool import resolve_binary
|
|
14
15
|
from models.task import SubtitleStyle
|
|
15
16
|
|
|
16
17
|
logger = logging.getLogger(__name__)
|
|
@@ -114,7 +115,7 @@ class ConcatMixin:
|
|
|
114
115
|
sigs = set()
|
|
115
116
|
for p in video_paths:
|
|
116
117
|
r = subprocess.run(
|
|
117
|
-
["ffprobe", "-v", "error", "-select_streams", "v:0",
|
|
118
|
+
[resolve_binary("ffprobe"), "-v", "error", "-select_streams", "v:0",
|
|
118
119
|
"-show_entries", "stream=width,height,avg_frame_rate",
|
|
119
120
|
"-of", "csv=s=x:p=0", p],
|
|
120
121
|
stdin=subprocess.DEVNULL, capture_output=True, text=True, timeout=15,
|
|
@@ -137,7 +138,7 @@ class ConcatMixin:
|
|
|
137
138
|
f.write(f"file '{esc}'\n")
|
|
138
139
|
try:
|
|
139
140
|
r = subprocess.run(
|
|
140
|
-
["ffmpeg", "-y", "-f", "concat", "-safe", "0",
|
|
141
|
+
[resolve_binary("ffmpeg"), "-y", "-f", "concat", "-safe", "0",
|
|
141
142
|
"-i", concat_file, "-c", "copy", output_path],
|
|
142
143
|
stdin=subprocess.DEVNULL, capture_output=True, text=True, timeout=600,
|
|
143
144
|
)
|
|
@@ -383,7 +384,7 @@ class ConcatMixin:
|
|
|
383
384
|
"""用 ffprobe 获取媒体文件时长(秒)。"""
|
|
384
385
|
try:
|
|
385
386
|
r = subprocess.run(
|
|
386
|
-
["ffprobe", "-v", "error", "-show_entries", "format=duration",
|
|
387
|
+
[resolve_binary("ffprobe"), "-v", "error", "-show_entries", "format=duration",
|
|
387
388
|
"-of", "csv=p=0", path],
|
|
388
389
|
stdin=subprocess.DEVNULL,
|
|
389
390
|
capture_output=True, text=True, timeout=15,
|
|
@@ -396,6 +397,12 @@ class ConcatMixin:
|
|
|
396
397
|
def _run_ffmpeg(cmd: list, desc: str = "") -> None:
|
|
397
398
|
"""执行 ffmpeg 命令,失败时抛 RuntimeError。"""
|
|
398
399
|
logger.info(f"[Compositor] ffmpeg: {desc}")
|
|
400
|
+
if not cmd or cmd[0] is None:
|
|
401
|
+
raise RuntimeError(
|
|
402
|
+
"ffmpeg not available (no system PATH ffmpeg nor builtin "
|
|
403
|
+
"imageio-ffmpeg). Please install ffmpeg or verify imageio-ffmpeg "
|
|
404
|
+
"is installed."
|
|
405
|
+
)
|
|
399
406
|
try:
|
|
400
407
|
r = subprocess.run(
|
|
401
408
|
cmd, stdin=subprocess.DEVNULL,
|
|
@@ -0,0 +1,102 @@
|
|
|
1
|
+
"""core.compositor.ffmpeg_tool — ffmpeg / ffprobe 可执行文件统一解析。
|
|
2
|
+
|
|
3
|
+
背景:此前所有 ffmpeg/ffprobe 调用都用裸命令字符串经 ``subprocess.run`` 执行,
|
|
4
|
+
依赖系统 PATH 搜索可执行文件。Windows 未安装 ffmpeg 时 ``CreateProcess`` 抛
|
|
5
|
+
``[WinError 2] The system cannot find the file specified``,且发生在任务运行到
|
|
6
|
+
拼接步骤而非启动时(见 Issue #36 的完整 traceback)。
|
|
7
|
+
|
|
8
|
+
解析优先级(推荐,但不强制用户安装系统 ffmpeg):
|
|
9
|
+
1. 环境变量显式指定:``FFMPEG_BINARY`` / ``FFPROBE_BINARY`` —— 尊重用户意图
|
|
10
|
+
2. 系统 PATH(``shutil.which``)—— 已安装则用之,绝不覆盖
|
|
11
|
+
3. ``imageio-ffmpeg`` 内置静态二进制兜底 —— wheel 自带,无需用户操作
|
|
12
|
+
|
|
13
|
+
结果带进程级缓存;全部不可用时返回 None,由启动检测 / 调用方给出清晰指引
|
|
14
|
+
而非裸的 ``[WinError 2]``。
|
|
15
|
+
"""
|
|
16
|
+
import logging
|
|
17
|
+
import os
|
|
18
|
+
import shutil
|
|
19
|
+
|
|
20
|
+
logger = logging.getLogger(__name__)
|
|
21
|
+
|
|
22
|
+
# 二进制名 → 覆盖环境变量名
|
|
23
|
+
_EXE_OVERRIDE = {"ffmpeg": "FFMPEG_BINARY", "ffprobe": "FFPROBE_BINARY"}
|
|
24
|
+
_cache: "dict[str, str | None]" = {}
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def resolve_binary(name: str) -> "str | None":
|
|
28
|
+
"""按优先级解析 ffmpeg/ffprobe 可执行文件绝对路径,进程内缓存。
|
|
29
|
+
|
|
30
|
+
Args:
|
|
31
|
+
name: ``"ffmpeg"`` 或 ``"ffprobe"``。
|
|
32
|
+
|
|
33
|
+
Returns:
|
|
34
|
+
可用可执行文件绝对路径;全部不可用时返回 None。
|
|
35
|
+
"""
|
|
36
|
+
if name not in _EXE_OVERRIDE:
|
|
37
|
+
raise ValueError(f"unknown binary: {name}")
|
|
38
|
+
if name in _cache:
|
|
39
|
+
return _cache[name]
|
|
40
|
+
path = _resolve(name)
|
|
41
|
+
_cache[name] = path
|
|
42
|
+
return path
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def resolve_ffmpeg() -> "str | None":
|
|
46
|
+
"""便捷:解析 ffmpeg。"""
|
|
47
|
+
return resolve_binary("ffmpeg")
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def resolve_ffprobe() -> "str | None":
|
|
51
|
+
"""便捷:解析 ffprobe(可能为 None,仅时长/尺寸探测用,调用方自带兜底)。"""
|
|
52
|
+
return resolve_binary("ffprobe")
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def _resolve(name: str) -> "str | None":
|
|
56
|
+
# 1) 显式指定
|
|
57
|
+
override = os.environ.get(_EXE_OVERRIDE[name])
|
|
58
|
+
if override and os.path.exists(override):
|
|
59
|
+
logger.info(f"[Compositor] {name}: explicit {override}")
|
|
60
|
+
return override
|
|
61
|
+
|
|
62
|
+
# 2) 系统 PATH
|
|
63
|
+
found = shutil.which(name)
|
|
64
|
+
if found:
|
|
65
|
+
logger.info(f"[Compositor] {name}: system {found}")
|
|
66
|
+
return found
|
|
67
|
+
|
|
68
|
+
# 3) imageio-ffmpeg 内置(仅自带 ffmpeg;ffprobe 从同目录推导)
|
|
69
|
+
base = _cache.get("ffmpeg") or _resolve_builtin_ffmpeg()
|
|
70
|
+
if base:
|
|
71
|
+
if name == "ffmpeg":
|
|
72
|
+
return base
|
|
73
|
+
probe = _sibling(base, "ffprobe")
|
|
74
|
+
if probe:
|
|
75
|
+
logger.info(f"[Compositor] {name}: builtin {probe}")
|
|
76
|
+
return probe
|
|
77
|
+
|
|
78
|
+
logger.warning(f"[Compositor] {name}: not found (no system PATH nor builtin)")
|
|
79
|
+
return None
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def _resolve_builtin_ffmpeg() -> "str | None":
|
|
83
|
+
"""取 imageio-ffmpeg 内置静态二进制;加载/失败时记录并返回 None。"""
|
|
84
|
+
try:
|
|
85
|
+
import imageio_ffmpeg
|
|
86
|
+
|
|
87
|
+
exe = imageio_ffmpeg.get_ffmpeg_exe()
|
|
88
|
+
except Exception as e: # 加载失败 / 内置缺失 / 首次下载失败
|
|
89
|
+
logger.warning(f"[Compositor] builtin ffmpeg unavailable: {e}")
|
|
90
|
+
return None
|
|
91
|
+
if exe and os.path.exists(exe):
|
|
92
|
+
logger.info(f"[Compositor] ffmpeg: builtin {exe}")
|
|
93
|
+
return exe
|
|
94
|
+
return None
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
def _sibling(exe: str, stem: str) -> "str | None":
|
|
98
|
+
"""由可执行文件路径推导同目录兄弟程序(Windows 补 .exe)。"""
|
|
99
|
+
candidate = os.path.join(
|
|
100
|
+
os.path.dirname(exe), stem + (".exe" if os.name == "nt" else "")
|
|
101
|
+
)
|
|
102
|
+
return candidate if os.path.exists(candidate) else None
|
|
@@ -6,6 +6,8 @@
|
|
|
6
6
|
import logging
|
|
7
7
|
import os
|
|
8
8
|
|
|
9
|
+
from core.compositor.ffmpeg_tool import resolve_binary
|
|
10
|
+
|
|
9
11
|
logger = logging.getLogger(__name__)
|
|
10
12
|
|
|
11
13
|
|
|
@@ -20,7 +22,7 @@ class VideoProcessor:
|
|
|
20
22
|
|
|
21
23
|
import subprocess
|
|
22
24
|
subprocess.run([
|
|
23
|
-
"ffmpeg", "-y", "-i", input_path,
|
|
25
|
+
resolve_binary("ffmpeg"), "-y", "-i", input_path,
|
|
24
26
|
"-vf", f"scale={width}:{height}:force_original_aspect_ratio=decrease,"
|
|
25
27
|
f"pad={width}:{height}:(ow-iw)/2:(oh-ih)/2",
|
|
26
28
|
"-c:v", "libx264", "-preset", "fast",
|
|
@@ -37,7 +39,7 @@ class VideoProcessor:
|
|
|
37
39
|
|
|
38
40
|
import subprocess
|
|
39
41
|
subprocess.run([
|
|
40
|
-
"ffmpeg", "-y",
|
|
42
|
+
resolve_binary("ffmpeg"), "-y",
|
|
41
43
|
"-sseof", "-1",
|
|
42
44
|
"-i", video_path,
|
|
43
45
|
"-frames:v", "1",
|
|
@@ -55,7 +57,7 @@ class VideoProcessor:
|
|
|
55
57
|
|
|
56
58
|
import subprocess
|
|
57
59
|
subprocess.run([
|
|
58
|
-
"ffmpeg", "-y",
|
|
60
|
+
resolve_binary("ffmpeg"), "-y",
|
|
59
61
|
"-f", "lavfi",
|
|
60
62
|
"-i", "anullsrc=r=44100:cl=mono",
|
|
61
63
|
"-t", str(duration_sec),
|
|
@@ -87,7 +89,7 @@ class VideoProcessor:
|
|
|
87
89
|
# 注意:勿再用 -t {freeze_duration},那会把整个输出截断成只剩冻结段。
|
|
88
90
|
try:
|
|
89
91
|
subprocess.run([
|
|
90
|
-
"ffmpeg", "-y",
|
|
92
|
+
resolve_binary("ffmpeg"), "-y",
|
|
91
93
|
"-i", video_path,
|
|
92
94
|
"-vf", f"tpad=stop_mode=clone:stop_duration={freeze_duration}",
|
|
93
95
|
"-c:v", "libx264", "-preset", "fast",
|
|
@@ -11,6 +11,7 @@ import subprocess
|
|
|
11
11
|
import tempfile
|
|
12
12
|
from typing import NamedTuple, Optional
|
|
13
13
|
|
|
14
|
+
from core.compositor.ffmpeg_tool import resolve_binary
|
|
14
15
|
from core.config import resolve_font_path
|
|
15
16
|
|
|
16
17
|
logger = logging.getLogger(__name__)
|
|
@@ -194,7 +195,7 @@ def add_watermark(
|
|
|
194
195
|
|
|
195
196
|
# ffmpeg overlay: 将水印 PNG 叠加到视频上,无损流式拷贝
|
|
196
197
|
cmd = [
|
|
197
|
-
"ffmpeg", "-y",
|
|
198
|
+
resolve_binary("ffmpeg"), "-y",
|
|
198
199
|
"-i", input_path,
|
|
199
200
|
"-i", wm_png_path,
|
|
200
201
|
"-filter_complex", f"[0:v][1:v]overlay={layout.pos_x}:{layout.pos_y}[v]",
|
package/core/config.py
CHANGED
|
@@ -18,7 +18,7 @@ CONFIG_FILE = os.path.join(CONFIG_DIR, "config.json")
|
|
|
18
18
|
# ═══════════════════════════════════════════════════
|
|
19
19
|
# 应用版本号(v6.1 新增:发版时同步更新,见 docs/dev/release_process.md)
|
|
20
20
|
# ═══════════════════════════════════════════════════
|
|
21
|
-
APP_VERSION = "6.
|
|
21
|
+
APP_VERSION = "6.4.1"
|
|
22
22
|
|
|
23
23
|
# 未配置 API Key 时的统一报错文案(含免费获取与在线体验兜底,全站路由共用)
|
|
24
24
|
API_KEY_MISSING_MSG = (
|
|
@@ -202,6 +202,13 @@ try:
|
|
|
202
202
|
agnes_subtitle_ass: bool = True # 2.1c 字幕 ASS 单链灰度开关
|
|
203
203
|
agnes_video_poll_timeout: int = 1800 # 1.2 视频轮询总超时
|
|
204
204
|
|
|
205
|
+
# ── CORS(PR #33 吸收 Phase 2:可配置跨源白名单)──
|
|
206
|
+
# 供独立本地伴侣工具(如 agnes-simple-ui)从浏览器跨源调用本服务 API。
|
|
207
|
+
# agnes_cors_origins: 逗号分隔的允许源列表,空 = 不启用 CORS(默认,攻击面不变)。
|
|
208
|
+
# agnes_cors_enabled: None=auto(设置了 origins 才启用);显式 "false" 即使设置了 origins 也禁用。
|
|
209
|
+
agnes_cors_origins: str = ""
|
|
210
|
+
agnes_cors_enabled: bool | None = None
|
|
211
|
+
|
|
205
212
|
# ── 运维 ──
|
|
206
213
|
agnes_log_file: str = ""
|
|
207
214
|
agnes_sweep_age_days: int | None = None
|
|
@@ -235,6 +242,14 @@ except ImportError: # pragma: no cover - pydantic-settings 为必备依赖,
|
|
|
235
242
|
"0", "false", "off",
|
|
236
243
|
)
|
|
237
244
|
self.agnes_video_poll_timeout = int(os.environ.get("AGNES_VIDEO_POLL_TIMEOUT", "1800"))
|
|
245
|
+
self.agnes_cors_origins = os.environ.get("AGNES_CORS_ORIGINS", "")
|
|
246
|
+
_cors_enabled = os.environ.get("AGNES_CORS_ENABLED", "").strip().lower()
|
|
247
|
+
if _cors_enabled in ("0", "false", "off"):
|
|
248
|
+
self.agnes_cors_enabled = False
|
|
249
|
+
elif _cors_enabled in ("1", "true", "on"):
|
|
250
|
+
self.agnes_cors_enabled = True
|
|
251
|
+
else:
|
|
252
|
+
self.agnes_cors_enabled = None
|
|
238
253
|
self.agnes_log_file = os.environ.get("AGNES_LOG_FILE", "")
|
|
239
254
|
self.agnes_sweep_age_days = _env_int("AGNES_SWEEP_AGE_DAYS")
|
|
240
255
|
self.agnes_config_id_hmac_key = os.environ.get(
|
|
@@ -951,9 +966,12 @@ def get_video_model_capabilities() -> dict:
|
|
|
951
966
|
# ═══════════════════════════════════════════════════
|
|
952
967
|
|
|
953
968
|
# 可用域名映射
|
|
969
|
+
# 注:Agnes 官方域名划分(来源:AgnesAI-Labs skills 的 model_catalog 参考值):
|
|
970
|
+
# - com 国际站(主):apihub.agnes-ai.com
|
|
971
|
+
# - cn 国内站(中国站):api.agnes-ai.cn(apihub.agnes-ai.cn 仅作国际站备用,非国内站)
|
|
954
972
|
AGNES_DOMAIN_MAP = {
|
|
955
973
|
"com": "https://apihub.agnes-ai.com",
|
|
956
|
-
"cn": "https://
|
|
974
|
+
"cn": "https://api.agnes-ai.cn",
|
|
957
975
|
}
|
|
958
976
|
|
|
959
977
|
_DEFAULT_DOMAIN = "com"
|