yueying 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- yueying-0.1.0/LICENSE +21 -0
- yueying-0.1.0/PKG-INFO +142 -0
- yueying-0.1.0/README.md +113 -0
- yueying-0.1.0/pyproject.toml +45 -0
- yueying-0.1.0/setup.cfg +4 -0
- yueying-0.1.0/src/yueying/__init__.py +2 -0
- yueying-0.1.0/src/yueying/asr.py +105 -0
- yueying-0.1.0/src/yueying/cli.py +183 -0
- yueying-0.1.0/src/yueying/download.py +96 -0
- yueying-0.1.0/src/yueying/ffm.py +57 -0
- yueying-0.1.0/src/yueying/frames.py +126 -0
- yueying-0.1.0/src/yueying/report.py +118 -0
- yueying-0.1.0/src/yueying/skill/SKILL.md +34 -0
- yueying-0.1.0/src/yueying/subs.py +83 -0
- yueying-0.1.0/src/yueying.egg-info/PKG-INFO +142 -0
- yueying-0.1.0/src/yueying.egg-info/SOURCES.txt +18 -0
- yueying-0.1.0/src/yueying.egg-info/dependency_links.txt +1 -0
- yueying-0.1.0/src/yueying.egg-info/entry_points.txt +2 -0
- yueying-0.1.0/src/yueying.egg-info/requires.txt +8 -0
- yueying-0.1.0/src/yueying.egg-info/top_level.txt +1 -0
yueying-0.1.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 yueying contributors
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
yueying-0.1.0/PKG-INFO
ADDED
|
@@ -0,0 +1,142 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: yueying
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: 阅影:把视频变成 AI 能读的文字稿和关键帧,让 Claude Code / Cursor 等编程助手看懂视频
|
|
5
|
+
Author: vsh5dvsch7-png
|
|
6
|
+
License: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/vsh5dvsch7-png/yueying
|
|
8
|
+
Project-URL: Repository, https://github.com/vsh5dvsch7-png/yueying
|
|
9
|
+
Project-URL: Issues, https://github.com/vsh5dvsch7-png/yueying/issues
|
|
10
|
+
Keywords: video,whisper,claude-code,agent-skill,transcript,keyframes,bilibili,youtube
|
|
11
|
+
Classifier: Development Status :: 3 - Alpha
|
|
12
|
+
Classifier: Environment :: Console
|
|
13
|
+
Classifier: Intended Audience :: Developers
|
|
14
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
15
|
+
Classifier: Operating System :: OS Independent
|
|
16
|
+
Classifier: Programming Language :: Python :: 3
|
|
17
|
+
Classifier: Topic :: Multimedia :: Video
|
|
18
|
+
Requires-Python: >=3.9
|
|
19
|
+
Description-Content-Type: text/markdown
|
|
20
|
+
License-File: LICENSE
|
|
21
|
+
Requires-Dist: faster-whisper>=1.1
|
|
22
|
+
Requires-Dist: yt-dlp>=2025.1.1
|
|
23
|
+
Requires-Dist: imageio-ffmpeg>=0.5
|
|
24
|
+
Requires-Dist: pillow>=10.1
|
|
25
|
+
Provides-Extra: cuda
|
|
26
|
+
Requires-Dist: nvidia-cublas-cu12; extra == "cuda"
|
|
27
|
+
Requires-Dist: nvidia-cudnn-cu12; extra == "cuda"
|
|
28
|
+
Dynamic: license-file
|
|
29
|
+
|
|
30
|
+
# 阅影 yueying
|
|
31
|
+
|
|
32
|
+
**让 AI 编程助手看懂视频。** Claude Code、Cursor 这些工具读得了图片和文字,却读不了视频。阅影把一段视频(本地文件,或 B站 / YouTube / 抖音 / 小红书等链接)拆成 AI 能读的东西:
|
|
33
|
+
|
|
34
|
+
- **文字稿**:优先抓平台自带字幕;没有就用本地语音识别(faster-whisper),中、英、日、韩、法、德、俄等几十种语言自动检测,全程离线,不上传、不花钱
|
|
35
|
+
- **关键帧**:在画面明显变化的地方抽帧,每张左下角烧上编号和时间;再拼成九宫格总览图,AI 一眼看完全片走向
|
|
36
|
+
- **一份 report.md**:简介、章节、画面总览、关键帧清单、按段落带时间戳的文字稿,AI 读这一个文件就能开始干活
|
|
37
|
+
|
|
38
|
+
然后你就可以对 Claude 说:「帮我看看这个视频,把里面的步骤整理成笔记」「照着这个教程把代码写出来」「这个视频 10 分钟处讲了什么」。
|
|
39
|
+
|
|
40
|
+
## 安装
|
|
41
|
+
|
|
42
|
+
需要 Python 3.9 以上。ffmpeg 会自动带上,不用单独装。
|
|
43
|
+
|
|
44
|
+
**Windows**:下载本仓库,双击 `install.cmd`。它会创建独立的 Python 环境、装好依赖、把 Claude Code 技能放到 `~/.claude/skills/yueying`。
|
|
45
|
+
|
|
46
|
+
**手动安装(任何系统)**:
|
|
47
|
+
|
|
48
|
+
```bash
|
|
49
|
+
pip install yueying # 或 pipx install yueying
|
|
50
|
+
pip install "yueying[cuda]" # 有 NVIDIA 显卡再加这个,语音识别快很多
|
|
51
|
+
yueying --install-skill # 把 Claude Code 技能装到 ~/.claude/skills/yueying
|
|
52
|
+
```
|
|
53
|
+
|
|
54
|
+
想装 GitHub 上的最新版:`pip install git+https://github.com/vsh5dvsch7-png/yueying.git`。
|
|
55
|
+
|
|
56
|
+
装好技能后,Claude Code 会在需要看视频时自动调用它。Cursor 等支持 Agent Skills 标准的工具,把 `~/.claude/skills/yueying/SKILL.md` 复制到对应目录即可。
|
|
57
|
+
|
|
58
|
+
第一次做语音识别会下载模型(默认 large-v3-turbo,约 1.6 GB)。国内访问 HuggingFace 慢的话,程序会自动切换到 hf-mirror.com 镜像。
|
|
59
|
+
|
|
60
|
+
## 用法
|
|
61
|
+
|
|
62
|
+
```bash
|
|
63
|
+
yueying 视频.mp4
|
|
64
|
+
yueying "https://www.bilibili.com/video/BVxxxx"
|
|
65
|
+
yueying "https://www.youtube.com/watch?v=xxxx" --out ./notes/xxx
|
|
66
|
+
```
|
|
67
|
+
|
|
68
|
+
输出目录默认是 `./yueying_out/<视频名>/`:
|
|
69
|
+
|
|
70
|
+
```
|
|
71
|
+
report.md 给 AI 读的总索引
|
|
72
|
+
transcript.txt 按段落、带时间戳的文字稿
|
|
73
|
+
transcript.srt 字幕文件,可直接给播放器用
|
|
74
|
+
grid_01.jpg ... 九宫格总览图
|
|
75
|
+
frames/ 单帧大图,f003_00m15s.jpg 这样命名
|
|
76
|
+
manifest.json 给程序读的结构化结果
|
|
77
|
+
```
|
|
78
|
+
|
|
79
|
+
常用参数:
|
|
80
|
+
|
|
81
|
+
| 参数 | 作用 |
|
|
82
|
+
|---|---|
|
|
83
|
+
| `--lang zh` | 指定语言(默认自动检测) |
|
|
84
|
+
| `--model small` | 换小模型,没显卡的机器快很多;可选 tiny / base / small / medium / large-v3 / large-v3-turbo |
|
|
85
|
+
| `--device cpu` | 强制用 CPU |
|
|
86
|
+
| `--frames 30` | 最多抽多少帧(默认按时长自动,上限 60) |
|
|
87
|
+
| `--scene 0.2` | 场景切换灵敏度,越小越敏感 |
|
|
88
|
+
| `--no-asr` | 没字幕也不做语音识别(只要画面) |
|
|
89
|
+
| `--no-frames` | 不抽画面(只要文字) |
|
|
90
|
+
| `--force-asr` | 有字幕也重新识别 |
|
|
91
|
+
| `--cookies-from-browser chrome` | 用浏览器登录态下载(B站 高清 / 会员视频) |
|
|
92
|
+
| `--keep` | 保留下载的原视频 |
|
|
93
|
+
|
|
94
|
+
## 在 Claude Code 里用
|
|
95
|
+
|
|
96
|
+
装好技能后,直接说:
|
|
97
|
+
|
|
98
|
+
> 帮我看看 D:\课程\第3课.mp4,把知识点整理成笔记
|
|
99
|
+
|
|
100
|
+
> 这个视频 https://www.bilibili.com/video/BVxxxx 里的代码帮我抄下来
|
|
101
|
+
|
|
102
|
+
Claude 会自己跑 `yueying`,读 report.md 和关键帧,然后按你的要求整理。引用视频内容时它会带上时间点,方便你回看。
|
|
103
|
+
|
|
104
|
+
## 工作原理
|
|
105
|
+
|
|
106
|
+
```
|
|
107
|
+
视频/链接 ──► yt-dlp 下载(≤720p)+ 抓字幕
|
|
108
|
+
──► ffmpeg 抽 16k 音频 ──► faster-whisper 识别(有字幕则跳过)
|
|
109
|
+
──► ffmpeg 场景检测 ──► 抽帧 ──► Pillow 烧时间戳、拼九宫格
|
|
110
|
+
──► report.md / transcript / manifest.json
|
|
111
|
+
```
|
|
112
|
+
|
|
113
|
+
代码都在 `src/yueying/`,每个文件一件事:
|
|
114
|
+
|
|
115
|
+
| 文件 | 管什么 |
|
|
116
|
+
|---|---|
|
|
117
|
+
| `cli.py` | 命令行入口,串起下面几步 |
|
|
118
|
+
| `download.py` | yt-dlp 下载、挑字幕语言 |
|
|
119
|
+
| `ffm.py` | ffmpeg 封装:探测时长分辨率、抽音频、导内嵌字幕 |
|
|
120
|
+
| `subs.py` | 解析 srt / vtt / B站 json 字幕,去掉 YouTube 自动字幕的重复 |
|
|
121
|
+
| `asr.py` | faster-whisper 语音识别,自动选 GPU / CPU,失败回退 |
|
|
122
|
+
| `frames.py` | 场景检测、定抽帧时间点、抽帧、烧时间戳、拼九宫格 |
|
|
123
|
+
| `report.py` | 生成 report.md、transcript、manifest |
|
|
124
|
+
|
|
125
|
+
## 常见问题
|
|
126
|
+
|
|
127
|
+
**识别不准?** 换大模型 `--model large-v3`,或指定语言 `--lang ja`。专有名词和代码容易听错,以画面为准,AI 会对照关键帧。
|
|
128
|
+
|
|
129
|
+
**B站 下载失败?** 试试 `--cookies-from-browser chrome`(先在 Chrome 里登录 B站)。
|
|
130
|
+
|
|
131
|
+
**没有 NVIDIA 显卡很慢?** 用 `--model small`,一段 10 分钟的视频大约 1 到 2 分钟识别完。
|
|
132
|
+
|
|
133
|
+
**想要更多画面?** `--frames 60 --scene 0.2`。
|
|
134
|
+
|
|
135
|
+
## 致谢
|
|
136
|
+
|
|
137
|
+
- [faster-whisper](https://github.com/SYSTRAN/faster-whisper)、[yt-dlp](https://github.com/yt-dlp/yt-dlp)、[imageio-ffmpeg](https://github.com/imageio/imageio-ffmpeg)、[Pillow](https://python-pillow.org/)
|
|
138
|
+
- 九宫格烧时间戳的做法参考了 [video-vision-mcp](https://github.com/OAMaestro/video-vision-mcp),按时长分档抽帧参考了 [video-analyzer-skill](https://github.com/bsisduck/video-analyzer-skill)
|
|
139
|
+
|
|
140
|
+
## 许可
|
|
141
|
+
|
|
142
|
+
MIT
|
yueying-0.1.0/README.md
ADDED
|
@@ -0,0 +1,113 @@
|
|
|
1
|
+
# 阅影 yueying
|
|
2
|
+
|
|
3
|
+
**让 AI 编程助手看懂视频。** Claude Code、Cursor 这些工具读得了图片和文字,却读不了视频。阅影把一段视频(本地文件,或 B站 / YouTube / 抖音 / 小红书等链接)拆成 AI 能读的东西:
|
|
4
|
+
|
|
5
|
+
- **文字稿**:优先抓平台自带字幕;没有就用本地语音识别(faster-whisper),中、英、日、韩、法、德、俄等几十种语言自动检测,全程离线,不上传、不花钱
|
|
6
|
+
- **关键帧**:在画面明显变化的地方抽帧,每张左下角烧上编号和时间;再拼成九宫格总览图,AI 一眼看完全片走向
|
|
7
|
+
- **一份 report.md**:简介、章节、画面总览、关键帧清单、按段落带时间戳的文字稿,AI 读这一个文件就能开始干活
|
|
8
|
+
|
|
9
|
+
然后你就可以对 Claude 说:「帮我看看这个视频,把里面的步骤整理成笔记」「照着这个教程把代码写出来」「这个视频 10 分钟处讲了什么」。
|
|
10
|
+
|
|
11
|
+
## 安装
|
|
12
|
+
|
|
13
|
+
需要 Python 3.9 以上。ffmpeg 会自动带上,不用单独装。
|
|
14
|
+
|
|
15
|
+
**Windows**:下载本仓库,双击 `install.cmd`。它会创建独立的 Python 环境、装好依赖、把 Claude Code 技能放到 `~/.claude/skills/yueying`。
|
|
16
|
+
|
|
17
|
+
**手动安装(任何系统)**:
|
|
18
|
+
|
|
19
|
+
```bash
|
|
20
|
+
pip install yueying # 或 pipx install yueying
|
|
21
|
+
pip install "yueying[cuda]" # 有 NVIDIA 显卡再加这个,语音识别快很多
|
|
22
|
+
yueying --install-skill # 把 Claude Code 技能装到 ~/.claude/skills/yueying
|
|
23
|
+
```
|
|
24
|
+
|
|
25
|
+
想装 GitHub 上的最新版:`pip install git+https://github.com/vsh5dvsch7-png/yueying.git`。
|
|
26
|
+
|
|
27
|
+
装好技能后,Claude Code 会在需要看视频时自动调用它。Cursor 等支持 Agent Skills 标准的工具,把 `~/.claude/skills/yueying/SKILL.md` 复制到对应目录即可。
|
|
28
|
+
|
|
29
|
+
第一次做语音识别会下载模型(默认 large-v3-turbo,约 1.6 GB)。国内访问 HuggingFace 慢的话,程序会自动切换到 hf-mirror.com 镜像。
|
|
30
|
+
|
|
31
|
+
## 用法
|
|
32
|
+
|
|
33
|
+
```bash
|
|
34
|
+
yueying 视频.mp4
|
|
35
|
+
yueying "https://www.bilibili.com/video/BVxxxx"
|
|
36
|
+
yueying "https://www.youtube.com/watch?v=xxxx" --out ./notes/xxx
|
|
37
|
+
```
|
|
38
|
+
|
|
39
|
+
输出目录默认是 `./yueying_out/<视频名>/`:
|
|
40
|
+
|
|
41
|
+
```
|
|
42
|
+
report.md 给 AI 读的总索引
|
|
43
|
+
transcript.txt 按段落、带时间戳的文字稿
|
|
44
|
+
transcript.srt 字幕文件,可直接给播放器用
|
|
45
|
+
grid_01.jpg ... 九宫格总览图
|
|
46
|
+
frames/ 单帧大图,f003_00m15s.jpg 这样命名
|
|
47
|
+
manifest.json 给程序读的结构化结果
|
|
48
|
+
```
|
|
49
|
+
|
|
50
|
+
常用参数:
|
|
51
|
+
|
|
52
|
+
| 参数 | 作用 |
|
|
53
|
+
|---|---|
|
|
54
|
+
| `--lang zh` | 指定语言(默认自动检测) |
|
|
55
|
+
| `--model small` | 换小模型,没显卡的机器快很多;可选 tiny / base / small / medium / large-v3 / large-v3-turbo |
|
|
56
|
+
| `--device cpu` | 强制用 CPU |
|
|
57
|
+
| `--frames 30` | 最多抽多少帧(默认按时长自动,上限 60) |
|
|
58
|
+
| `--scene 0.2` | 场景切换灵敏度,越小越敏感 |
|
|
59
|
+
| `--no-asr` | 没字幕也不做语音识别(只要画面) |
|
|
60
|
+
| `--no-frames` | 不抽画面(只要文字) |
|
|
61
|
+
| `--force-asr` | 有字幕也重新识别 |
|
|
62
|
+
| `--cookies-from-browser chrome` | 用浏览器登录态下载(B站 高清 / 会员视频) |
|
|
63
|
+
| `--keep` | 保留下载的原视频 |
|
|
64
|
+
|
|
65
|
+
## 在 Claude Code 里用
|
|
66
|
+
|
|
67
|
+
装好技能后,直接说:
|
|
68
|
+
|
|
69
|
+
> 帮我看看 D:\课程\第3课.mp4,把知识点整理成笔记
|
|
70
|
+
|
|
71
|
+
> 这个视频 https://www.bilibili.com/video/BVxxxx 里的代码帮我抄下来
|
|
72
|
+
|
|
73
|
+
Claude 会自己跑 `yueying`,读 report.md 和关键帧,然后按你的要求整理。引用视频内容时它会带上时间点,方便你回看。
|
|
74
|
+
|
|
75
|
+
## 工作原理
|
|
76
|
+
|
|
77
|
+
```
|
|
78
|
+
视频/链接 ──► yt-dlp 下载(≤720p)+ 抓字幕
|
|
79
|
+
──► ffmpeg 抽 16k 音频 ──► faster-whisper 识别(有字幕则跳过)
|
|
80
|
+
──► ffmpeg 场景检测 ──► 抽帧 ──► Pillow 烧时间戳、拼九宫格
|
|
81
|
+
──► report.md / transcript / manifest.json
|
|
82
|
+
```
|
|
83
|
+
|
|
84
|
+
代码都在 `src/yueying/`,每个文件一件事:
|
|
85
|
+
|
|
86
|
+
| 文件 | 管什么 |
|
|
87
|
+
|---|---|
|
|
88
|
+
| `cli.py` | 命令行入口,串起下面几步 |
|
|
89
|
+
| `download.py` | yt-dlp 下载、挑字幕语言 |
|
|
90
|
+
| `ffm.py` | ffmpeg 封装:探测时长分辨率、抽音频、导内嵌字幕 |
|
|
91
|
+
| `subs.py` | 解析 srt / vtt / B站 json 字幕,去掉 YouTube 自动字幕的重复 |
|
|
92
|
+
| `asr.py` | faster-whisper 语音识别,自动选 GPU / CPU,失败回退 |
|
|
93
|
+
| `frames.py` | 场景检测、定抽帧时间点、抽帧、烧时间戳、拼九宫格 |
|
|
94
|
+
| `report.py` | 生成 report.md、transcript、manifest |
|
|
95
|
+
|
|
96
|
+
## 常见问题
|
|
97
|
+
|
|
98
|
+
**识别不准?** 换大模型 `--model large-v3`,或指定语言 `--lang ja`。专有名词和代码容易听错,以画面为准,AI 会对照关键帧。
|
|
99
|
+
|
|
100
|
+
**B站 下载失败?** 试试 `--cookies-from-browser chrome`(先在 Chrome 里登录 B站)。
|
|
101
|
+
|
|
102
|
+
**没有 NVIDIA 显卡很慢?** 用 `--model small`,一段 10 分钟的视频大约 1 到 2 分钟识别完。
|
|
103
|
+
|
|
104
|
+
**想要更多画面?** `--frames 60 --scene 0.2`。
|
|
105
|
+
|
|
106
|
+
## 致谢
|
|
107
|
+
|
|
108
|
+
- [faster-whisper](https://github.com/SYSTRAN/faster-whisper)、[yt-dlp](https://github.com/yt-dlp/yt-dlp)、[imageio-ffmpeg](https://github.com/imageio/imageio-ffmpeg)、[Pillow](https://python-pillow.org/)
|
|
109
|
+
- 九宫格烧时间戳的做法参考了 [video-vision-mcp](https://github.com/OAMaestro/video-vision-mcp),按时长分档抽帧参考了 [video-analyzer-skill](https://github.com/bsisduck/video-analyzer-skill)
|
|
110
|
+
|
|
111
|
+
## 许可
|
|
112
|
+
|
|
113
|
+
MIT
|
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=68"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "yueying"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "阅影:把视频变成 AI 能读的文字稿和关键帧,让 Claude Code / Cursor 等编程助手看懂视频"
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.9"
|
|
11
|
+
license = { text = "MIT" }
|
|
12
|
+
authors = [{ name = "vsh5dvsch7-png" }]
|
|
13
|
+
keywords = ["video", "whisper", "claude-code", "agent-skill", "transcript", "keyframes", "bilibili", "youtube"]
|
|
14
|
+
classifiers = [
|
|
15
|
+
"Development Status :: 3 - Alpha",
|
|
16
|
+
"Environment :: Console",
|
|
17
|
+
"Intended Audience :: Developers",
|
|
18
|
+
"License :: OSI Approved :: MIT License",
|
|
19
|
+
"Operating System :: OS Independent",
|
|
20
|
+
"Programming Language :: Python :: 3",
|
|
21
|
+
"Topic :: Multimedia :: Video",
|
|
22
|
+
]
|
|
23
|
+
dependencies = [
|
|
24
|
+
"faster-whisper>=1.1",
|
|
25
|
+
"yt-dlp>=2025.1.1",
|
|
26
|
+
"imageio-ffmpeg>=0.5",
|
|
27
|
+
"pillow>=10.1",
|
|
28
|
+
]
|
|
29
|
+
|
|
30
|
+
[project.optional-dependencies]
|
|
31
|
+
cuda = ["nvidia-cublas-cu12", "nvidia-cudnn-cu12"]
|
|
32
|
+
|
|
33
|
+
[project.urls]
|
|
34
|
+
Homepage = "https://github.com/vsh5dvsch7-png/yueying"
|
|
35
|
+
Repository = "https://github.com/vsh5dvsch7-png/yueying"
|
|
36
|
+
Issues = "https://github.com/vsh5dvsch7-png/yueying/issues"
|
|
37
|
+
|
|
38
|
+
[project.scripts]
|
|
39
|
+
yueying = "yueying.cli:main"
|
|
40
|
+
|
|
41
|
+
[tool.setuptools.packages.find]
|
|
42
|
+
where = ["src"]
|
|
43
|
+
|
|
44
|
+
[tool.setuptools.package-data]
|
|
45
|
+
yueying = ["skill/SKILL.md"]
|
yueying-0.1.0/setup.cfg
ADDED
|
@@ -0,0 +1,105 @@
|
|
|
1
|
+
"""语音识别:faster-whisper,本地离线;有 NVIDIA 显卡自动用 GPU,失败退回 CPU。"""
|
|
2
|
+
import glob
|
|
3
|
+
import os
|
|
4
|
+
import sys
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
def _add_cuda_dlls() -> None:
|
|
8
|
+
"""pip 装的 nvidia-cublas-cu12 / nvidia-cudnn-cu12 把 DLL 放在 site-packages/nvidia/*/bin,
|
|
9
|
+
Windows 下要手动加进搜索路径 ctranslate2 才找得到。"""
|
|
10
|
+
if not sys.platform.startswith("win"):
|
|
11
|
+
return
|
|
12
|
+
for sp in sys.path:
|
|
13
|
+
for d in glob.glob(os.path.join(sp, "nvidia", "*", "bin")):
|
|
14
|
+
try:
|
|
15
|
+
os.add_dll_directory(d)
|
|
16
|
+
except Exception:
|
|
17
|
+
pass
|
|
18
|
+
if d not in os.environ.get("PATH", ""):
|
|
19
|
+
os.environ["PATH"] = d + os.pathsep + os.environ.get("PATH", "")
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def pick_device(device: str):
|
|
23
|
+
"""返回 (device, compute_type)。"""
|
|
24
|
+
if device in ("cuda", "cpu"):
|
|
25
|
+
return device, ("float16" if device == "cuda" else "int8")
|
|
26
|
+
try:
|
|
27
|
+
import ctranslate2
|
|
28
|
+
if ctranslate2.get_cuda_device_count() > 0:
|
|
29
|
+
return "cuda", "float16"
|
|
30
|
+
except Exception:
|
|
31
|
+
pass
|
|
32
|
+
return "cpu", "int8"
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def _load(model_name: str, device: str, compute_type: str, log):
|
|
36
|
+
from faster_whisper import WhisperModel
|
|
37
|
+
try:
|
|
38
|
+
return WhisperModel(model_name, device=device, compute_type=compute_type)
|
|
39
|
+
except Exception as e:
|
|
40
|
+
msg = str(e).lower()
|
|
41
|
+
# 国内访问 HuggingFace 常失败,自动换镜像再试一次
|
|
42
|
+
if "HF_ENDPOINT" not in os.environ and ("huggingface" in msg or "connect" in msg or "timed out" in msg):
|
|
43
|
+
log(" 下载模型失败,改用镜像 hf-mirror.com 重试…")
|
|
44
|
+
os.environ["HF_ENDPOINT"] = "https://hf-mirror.com"
|
|
45
|
+
return WhisperModel(model_name, device=device, compute_type=compute_type)
|
|
46
|
+
raise
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
# whisper 输出中文时经常不带标点,用 hotwords 在每个窗口都塞一句带标点的提示,引导它加标点
|
|
50
|
+
STYLE_HINT = {
|
|
51
|
+
"zh": "以下是普通话的句子,带有标点符号。",
|
|
52
|
+
"yue": "以下係廣東話嘅句子,有標點符號。",
|
|
53
|
+
"ja": "以下は日本語の文章です。句読点があります。",
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def _run(model, wav, language):
|
|
58
|
+
from faster_whisper.audio import decode_audio
|
|
59
|
+
audio = decode_audio(wav, sampling_rate=16000)
|
|
60
|
+
if not language:
|
|
61
|
+
language, _, _ = model.detect_language(audio, vad_filter=True)
|
|
62
|
+
return model.transcribe(audio, language=language, beam_size=5, vad_filter=True,
|
|
63
|
+
condition_on_previous_text=False, hotwords=STYLE_HINT.get(language))
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def _chain(first, rest):
|
|
67
|
+
if first is not None:
|
|
68
|
+
yield first
|
|
69
|
+
yield from rest
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def transcribe(wav: str, model_name: str, device: str, language, duration: float, log=print) -> tuple:
|
|
73
|
+
"""返回 (segments, info)。segments = [{start, end, text}]。"""
|
|
74
|
+
_add_cuda_dlls()
|
|
75
|
+
dev, ct = pick_device(device)
|
|
76
|
+
log(f" 模型 {model_name},设备 {dev} ({ct});首次使用会下载模型,请耐心等待")
|
|
77
|
+
model = gen = info = None
|
|
78
|
+
if dev == "cuda":
|
|
79
|
+
try:
|
|
80
|
+
model = _load(model_name, dev, ct, log)
|
|
81
|
+
gen, info = _run(model, wav, language)
|
|
82
|
+
# 真正算出第一段才知道 GPU 能不能用
|
|
83
|
+
gen = iter(gen)
|
|
84
|
+
gen = _chain(next(gen, None), gen)
|
|
85
|
+
except Exception as e:
|
|
86
|
+
log(f" GPU 不可用({type(e).__name__}: {str(e)[:150]}),退回 CPU")
|
|
87
|
+
dev, ct = "cpu", "int8"
|
|
88
|
+
model = None
|
|
89
|
+
if model is None:
|
|
90
|
+
model = _load(model_name, dev, ct, log)
|
|
91
|
+
gen, info = _run(model, wav, language)
|
|
92
|
+
log(f" 检测语言 {info.language}(置信度 {info.language_probability:.0%})")
|
|
93
|
+
segs = []
|
|
94
|
+
last_pct = -1
|
|
95
|
+
for s in gen:
|
|
96
|
+
text = s.text.strip()
|
|
97
|
+
if text:
|
|
98
|
+
segs.append({"start": float(s.start), "end": float(s.end), "text": text})
|
|
99
|
+
if duration:
|
|
100
|
+
pct = int(min(s.end, duration) / duration * 100)
|
|
101
|
+
if pct // 10 != last_pct // 10:
|
|
102
|
+
last_pct = pct
|
|
103
|
+
log(f" 识别进度 {pct}%")
|
|
104
|
+
return segs, {"language": info.language, "language_probability": float(info.language_probability),
|
|
105
|
+
"model": model_name, "device": dev, "compute_type": ct}
|
|
@@ -0,0 +1,183 @@
|
|
|
1
|
+
"""命令行入口:yueying <视频文件或链接> [--out 目录] ..."""
|
|
2
|
+
import argparse
|
|
3
|
+
import glob
|
|
4
|
+
import json
|
|
5
|
+
import os
|
|
6
|
+
import re
|
|
7
|
+
import sys
|
|
8
|
+
import time
|
|
9
|
+
|
|
10
|
+
from . import __version__, ffm, subs, frames as fr, report
|
|
11
|
+
from .download import is_url, download
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def log(msg: str) -> None:
|
|
15
|
+
print(msg, flush=True)
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def slug(s: str) -> str:
|
|
19
|
+
s = re.sub(r"[\\/:*?\"<>|\s]+", "_", s).strip("._")
|
|
20
|
+
return s[:60] or "video"
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def find_sidecar_subs(video: str) -> list:
|
|
24
|
+
base = os.path.splitext(video)[0]
|
|
25
|
+
found = []
|
|
26
|
+
for ext in (".srt", ".vtt", ".ass", ".json"):
|
|
27
|
+
found += glob.glob(glob.escape(base) + "*" + ext)
|
|
28
|
+
return sorted(found)
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def find_exe() -> str:
|
|
32
|
+
"""找到 yueying 命令的绝对路径,写进技能文件,免得 Claude 的 shell 里 PATH 不一样找不到。"""
|
|
33
|
+
import shutil
|
|
34
|
+
exe = shutil.which("yueying")
|
|
35
|
+
if exe:
|
|
36
|
+
return os.path.abspath(exe)
|
|
37
|
+
cand = os.path.join(os.path.dirname(sys.executable), "yueying.exe" if os.name == "nt" else "yueying")
|
|
38
|
+
return cand if os.path.exists(cand) else "yueying"
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def install_skill() -> int:
|
|
42
|
+
"""把 SKILL.md 装到 ~/.claude/skills/yueying/,Claude Code 需要看视频时就会自动用。"""
|
|
43
|
+
from importlib import resources
|
|
44
|
+
text = resources.files("yueying").joinpath("skill/SKILL.md").read_text(encoding="utf-8")
|
|
45
|
+
exe = find_exe()
|
|
46
|
+
if exe != "yueying":
|
|
47
|
+
text = text.replace(' yueying "<', f' "{exe}" "<')
|
|
48
|
+
dst_dir = os.path.join(os.path.expanduser("~"), ".claude", "skills", "yueying")
|
|
49
|
+
os.makedirs(dst_dir, exist_ok=True)
|
|
50
|
+
dst = os.path.join(dst_dir, "SKILL.md")
|
|
51
|
+
with open(dst, "w", encoding="utf-8") as f:
|
|
52
|
+
f.write(text)
|
|
53
|
+
log(f"技能已安装:{dst}")
|
|
54
|
+
log(f"技能里使用的命令:{exe}")
|
|
55
|
+
log("现在可以在 Claude Code 里说:帮我看看这个视频 <路径或链接>")
|
|
56
|
+
return 0
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def main(argv=None) -> int:
|
|
60
|
+
for stream in (sys.stdout, sys.stderr):
|
|
61
|
+
try:
|
|
62
|
+
stream.reconfigure(encoding="utf-8", errors="replace")
|
|
63
|
+
except Exception:
|
|
64
|
+
pass
|
|
65
|
+
|
|
66
|
+
ap = argparse.ArgumentParser(prog="yueying", description="阅影:把视频变成 AI 能读的文字稿和关键帧")
|
|
67
|
+
ap.add_argument("input", nargs="?", help="本地视频文件,或 B站 / YouTube / 抖音等视频链接")
|
|
68
|
+
ap.add_argument("--install-skill", action="store_true", help="把 Claude Code 技能装到 ~/.claude/skills/yueying,装一次即可")
|
|
69
|
+
ap.add_argument("--out", help="输出目录(默认 ./yueying_out/<视频名>)")
|
|
70
|
+
ap.add_argument("--lang", help="语音语言代码,如 zh / en / ja;默认自动检测")
|
|
71
|
+
ap.add_argument("--model", default="large-v3-turbo", help="whisper 模型:tiny/base/small/medium/large-v3/large-v3-turbo(默认)")
|
|
72
|
+
ap.add_argument("--device", default="auto", choices=["auto", "cuda", "cpu"], help="识别设备,默认有显卡用显卡")
|
|
73
|
+
ap.add_argument("--frames", type=int, help="最多抽多少关键帧(默认按时长自动,上限 60)")
|
|
74
|
+
ap.add_argument("--scene", type=float, default=0.3, help="场景切换灵敏度 0~1,越小越敏感(默认 0.3)")
|
|
75
|
+
ap.add_argument("--no-frames", action="store_true", help="不抽画面")
|
|
76
|
+
ap.add_argument("--no-asr", action="store_true", help="没有字幕也不做语音识别")
|
|
77
|
+
ap.add_argument("--force-asr", action="store_true", help="即使有字幕也重新做语音识别")
|
|
78
|
+
ap.add_argument("--cookies-from-browser", metavar="BROWSER", help="从浏览器读登录 cookie(chrome/edge/firefox),B站高清或会员视频需要")
|
|
79
|
+
ap.add_argument("--keep", action="store_true", help="保留下载的原视频(默认处理完删掉)")
|
|
80
|
+
ap.add_argument("--json", action="store_true", help="最后额外打印一行 manifest JSON")
|
|
81
|
+
ap.add_argument("--version", action="version", version=f"yueying {__version__}")
|
|
82
|
+
a = ap.parse_args(argv)
|
|
83
|
+
if a.install_skill:
|
|
84
|
+
return install_skill()
|
|
85
|
+
if not a.input:
|
|
86
|
+
ap.error("请给出视频文件路径或链接,例如:yueying 视频.mp4")
|
|
87
|
+
|
|
88
|
+
t0 = time.time()
|
|
89
|
+
meta = {"source": a.input}
|
|
90
|
+
downloaded = None
|
|
91
|
+
if is_url(a.input):
|
|
92
|
+
out_dir = a.out or os.path.join("yueying_out", "pending")
|
|
93
|
+
log(f"[1/4] 下载 {a.input}")
|
|
94
|
+
tmp_dir = os.path.join(a.out or "yueying_out", "_download")
|
|
95
|
+
downloaded = download(a.input, tmp_dir, a.cookies_from_browser, log)
|
|
96
|
+
video = downloaded["video"]
|
|
97
|
+
meta.update({k: downloaded[k] for k in ("title", "uploader", "url", "description", "chapters", "language")})
|
|
98
|
+
out_dir = a.out or os.path.join("yueying_out", slug(meta["title"]))
|
|
99
|
+
sub_files = downloaded["subs"]
|
|
100
|
+
else:
|
|
101
|
+
video = os.path.abspath(a.input)
|
|
102
|
+
if not os.path.exists(video):
|
|
103
|
+
log(f"找不到文件:{video}")
|
|
104
|
+
return 2
|
|
105
|
+
log(f"[1/4] 本地文件 {video}")
|
|
106
|
+
meta["title"] = os.path.splitext(os.path.basename(video))[0]
|
|
107
|
+
out_dir = a.out or os.path.join("yueying_out", slug(meta["title"]))
|
|
108
|
+
sub_files = find_sidecar_subs(video)
|
|
109
|
+
out_dir = os.path.abspath(out_dir)
|
|
110
|
+
os.makedirs(out_dir, exist_ok=True)
|
|
111
|
+
|
|
112
|
+
info = ffm.probe(video)
|
|
113
|
+
meta.update(info)
|
|
114
|
+
if downloaded and downloaded.get("duration") and not info["duration"]:
|
|
115
|
+
meta["duration"] = downloaded["duration"]
|
|
116
|
+
log(f" 时长 {fr.fmt_time(meta['duration'])},{info['width']}x{info['height']},"
|
|
117
|
+
f"{'有' if info['has_audio'] else '无'}音轨,内嵌字幕 {info['subtitle_streams']} 条")
|
|
118
|
+
|
|
119
|
+
# ---- 文字 ----
|
|
120
|
+
segs, text_source = [], {"kind": "none", "desc": "无"}
|
|
121
|
+
log("[2/4] 文字")
|
|
122
|
+
if not a.force_asr:
|
|
123
|
+
if not sub_files and info["subtitle_streams"]:
|
|
124
|
+
emb = os.path.join(out_dir, "embedded.srt")
|
|
125
|
+
if ffm.extract_embedded_subtitle(video, emb):
|
|
126
|
+
sub_files = [emb]
|
|
127
|
+
for sf in sub_files:
|
|
128
|
+
try:
|
|
129
|
+
segs = subs.parse_file(sf)
|
|
130
|
+
except Exception as e:
|
|
131
|
+
log(f" 字幕 {os.path.basename(sf)} 解析失败:{e}")
|
|
132
|
+
continue
|
|
133
|
+
if segs:
|
|
134
|
+
text_source = {"kind": "subtitle", "file": sf, "desc": f"字幕文件 {os.path.basename(sf)}({len(segs)} 条)"}
|
|
135
|
+
log(f" 用字幕 {os.path.basename(sf)},{len(segs)} 条")
|
|
136
|
+
break
|
|
137
|
+
if not segs and not a.no_asr:
|
|
138
|
+
if not info["has_audio"]:
|
|
139
|
+
log(" 没有音轨,跳过语音识别")
|
|
140
|
+
else:
|
|
141
|
+
log(" 没有可用字幕,做本地语音识别")
|
|
142
|
+
from . import asr
|
|
143
|
+
wav = os.path.join(out_dir, "audio.wav")
|
|
144
|
+
ffm.extract_audio(video, wav)
|
|
145
|
+
segs, ainfo = asr.transcribe(wav, a.model, a.device, a.lang, meta["duration"], log)
|
|
146
|
+
try:
|
|
147
|
+
os.remove(wav)
|
|
148
|
+
except OSError:
|
|
149
|
+
pass
|
|
150
|
+
text_source = {"kind": "asr", **ainfo,
|
|
151
|
+
"desc": f"本地语音识别 faster-whisper {ainfo['model']}({ainfo['device']}),"
|
|
152
|
+
f"检测语言 {ainfo['language']}({ainfo['language_probability']:.0%}),{len(segs)} 段"}
|
|
153
|
+
elif not segs:
|
|
154
|
+
log(" 无字幕且已跳过语音识别")
|
|
155
|
+
|
|
156
|
+
# ---- 画面 ----
|
|
157
|
+
frame_list, grids = [], []
|
|
158
|
+
log("[3/4] 画面")
|
|
159
|
+
if a.no_frames or not info["width"]:
|
|
160
|
+
log(" 跳过")
|
|
161
|
+
else:
|
|
162
|
+
target = a.frames or fr.default_target(meta["duration"])
|
|
163
|
+
scenes = fr.scene_times(video, a.scene)
|
|
164
|
+
times = fr.plan_times(meta["duration"], scenes, target)
|
|
165
|
+
log(f" 场景切换 {len(scenes)} 处,抽 {len(times)} 帧")
|
|
166
|
+
frame_list = fr.extract(video, times, os.path.join(out_dir, "frames"), log=log)
|
|
167
|
+
grids = fr.make_grids(frame_list, out_dir)
|
|
168
|
+
|
|
169
|
+
# ---- 报告 ----
|
|
170
|
+
log("[4/4] 写报告")
|
|
171
|
+
manifest = report.write_all(out_dir, meta, segs, frame_list, grids, text_source)
|
|
172
|
+
if downloaded and not a.keep:
|
|
173
|
+
import shutil
|
|
174
|
+
shutil.rmtree(os.path.dirname(downloaded["video"]), ignore_errors=True)
|
|
175
|
+
log(f"完成,用时 {time.time() - t0:.0f} 秒")
|
|
176
|
+
log(f"报告:{manifest['report']}")
|
|
177
|
+
if a.json:
|
|
178
|
+
print(json.dumps({k: v for k, v in manifest.items() if k != "segments"}, ensure_ascii=False))
|
|
179
|
+
return 0
|
|
180
|
+
|
|
181
|
+
|
|
182
|
+
if __name__ == "__main__":
|
|
183
|
+
sys.exit(main())
|