vidwit 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- vidwit/__init__.py +1 -0
- vidwit/__main__.py +4 -0
- vidwit/assembler.py +43 -0
- vidwit/chunker.py +37 -0
- vidwit/cli.py +239 -0
- vidwit/config.py +139 -0
- vidwit/ffmpeg_io.py +102 -0
- vidwit/init_cmd.py +71 -0
- vidwit/llm.py +281 -0
- vidwit/pipeline.py +403 -0
- vidwit/scratch.py +73 -0
- vidwit/templates/__init__.py +0 -0
- vidwit/templates/vidwit.toml +67 -0
- vidwit/templates/vidwit_prompt.md +84 -0
- vidwit/transcribe.py +98 -0
- vidwit-1.0.0.dist-info/METADATA +463 -0
- vidwit-1.0.0.dist-info/RECORD +20 -0
- vidwit-1.0.0.dist-info/WHEEL +4 -0
- vidwit-1.0.0.dist-info/entry_points.txt +2 -0
- vidwit-1.0.0.dist-info/licenses/LICENSE +504 -0
vidwit/__init__.py
ADDED
|
@@ -0,0 +1 @@
|
|
|
1
|
+
__version__ = "1.0.0"
|
vidwit/__main__.py
ADDED
vidwit/assembler.py
ADDED
|
@@ -0,0 +1,43 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import re
|
|
4
|
+
from pathlib import Path
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
_WARNING_RE = re.compile(r"\[⚠[^\]]*\]")
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
def assemble(chunks_dir: Path, video_name: str) -> str:
|
|
11
|
+
"""Concatenate chunk_*.md into one witness document with TOC + warnings index."""
|
|
12
|
+
files = sorted(chunks_dir.glob("chunk_*.md"))
|
|
13
|
+
if not files:
|
|
14
|
+
raise FileNotFoundError(f"no chunks under {chunks_dir}")
|
|
15
|
+
|
|
16
|
+
bodies = [p.read_text(encoding="utf-8").rstrip() for p in files]
|
|
17
|
+
full_body = "\n\n".join(bodies)
|
|
18
|
+
|
|
19
|
+
warnings = sorted(set(_WARNING_RE.findall(full_body)))
|
|
20
|
+
toc = _build_toc(bodies)
|
|
21
|
+
|
|
22
|
+
header = [f"# vidwit — {video_name}", ""]
|
|
23
|
+
if warnings:
|
|
24
|
+
header += ["## Content warnings", ""]
|
|
25
|
+
header += [f"- {w}" for w in warnings]
|
|
26
|
+
header += [""]
|
|
27
|
+
if toc:
|
|
28
|
+
header += ["## Table of contents", ""]
|
|
29
|
+
header += toc
|
|
30
|
+
header += [""]
|
|
31
|
+
header += ["---", ""]
|
|
32
|
+
return "\n".join(header) + full_body + "\n"
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def _build_toc(bodies: list[str]) -> list[str]:
|
|
36
|
+
lines = []
|
|
37
|
+
for b in bodies:
|
|
38
|
+
for ln in b.splitlines():
|
|
39
|
+
m = re.match(r"^###\s+(.*)$", ln)
|
|
40
|
+
if m:
|
|
41
|
+
lines.append(f"- {m.group(1).strip()}")
|
|
42
|
+
break
|
|
43
|
+
return lines
|
vidwit/chunker.py
ADDED
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from dataclasses import dataclass
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
@dataclass(slots=True, frozen=True)
|
|
7
|
+
class Window:
|
|
8
|
+
index: int
|
|
9
|
+
start: float # seconds, inclusive
|
|
10
|
+
end: float # seconds, half-open
|
|
11
|
+
|
|
12
|
+
@property
|
|
13
|
+
def label(self) -> str:
|
|
14
|
+
return f"chunk_{self.index:04d}"
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def windows(duration_s: float, window_s: float, overlap_s: float) -> list[Window]:
|
|
18
|
+
"""Half-open windows [start, end) covering duration. Stride = window - overlap.
|
|
19
|
+
|
|
20
|
+
Adjacent windows overlap by overlap_s. Final window is clamped to duration.
|
|
21
|
+
"""
|
|
22
|
+
if window_s <= 0:
|
|
23
|
+
raise ValueError("window_s must be > 0")
|
|
24
|
+
if not 0 <= overlap_s < window_s:
|
|
25
|
+
raise ValueError("overlap_s must satisfy 0 <= overlap < window")
|
|
26
|
+
stride = window_s - overlap_s
|
|
27
|
+
out: list[Window] = []
|
|
28
|
+
i = 0
|
|
29
|
+
start = 0.0
|
|
30
|
+
while start < duration_s:
|
|
31
|
+
end = min(start + window_s, duration_s)
|
|
32
|
+
out.append(Window(index=i, start=start, end=end))
|
|
33
|
+
if end >= duration_s:
|
|
34
|
+
break
|
|
35
|
+
start += stride
|
|
36
|
+
i += 1
|
|
37
|
+
return out
|
vidwit/cli.py
ADDED
|
@@ -0,0 +1,239 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import argparse
|
|
4
|
+
import json
|
|
5
|
+
import logging
|
|
6
|
+
import sys
|
|
7
|
+
from dataclasses import replace
|
|
8
|
+
from pathlib import Path
|
|
9
|
+
|
|
10
|
+
from . import __version__, config as cfg_mod, init_cmd, pipeline
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def main(argv: list[str] | None = None) -> int:
|
|
14
|
+
raw = sys.argv[1:] if argv is None else argv
|
|
15
|
+
if raw and raw[0] == "init":
|
|
16
|
+
return init_cmd.run(raw[1:])
|
|
17
|
+
|
|
18
|
+
parser = _build_parser()
|
|
19
|
+
args = parser.parse_args(argv)
|
|
20
|
+
|
|
21
|
+
logging.basicConfig(
|
|
22
|
+
level=logging.DEBUG if args.verbose else logging.INFO,
|
|
23
|
+
format="%(asctime)s %(levelname)s %(name)s | %(message)s",
|
|
24
|
+
datefmt="%H:%M:%S",
|
|
25
|
+
)
|
|
26
|
+
|
|
27
|
+
cfg = _make_config(args)
|
|
28
|
+
|
|
29
|
+
inputs = _expand_inputs(args.inputs, cfg.video_exts)
|
|
30
|
+
if not inputs:
|
|
31
|
+
print("vidwit: no video inputs", file=sys.stderr)
|
|
32
|
+
return 2
|
|
33
|
+
if cfg.output_override is not None and len(inputs) > 1:
|
|
34
|
+
print(
|
|
35
|
+
f"vidwit: -o/--output works only with a single input video "
|
|
36
|
+
f"(got {len(inputs)}). Use --paths home:DIR to redirect a batch.",
|
|
37
|
+
file=sys.stderr,
|
|
38
|
+
)
|
|
39
|
+
return 2
|
|
40
|
+
|
|
41
|
+
rc = 0
|
|
42
|
+
for v in inputs:
|
|
43
|
+
try:
|
|
44
|
+
pipeline.run_one(v, cfg)
|
|
45
|
+
except FileExistsError as e:
|
|
46
|
+
print(f"vidwit: {e}", file=sys.stderr)
|
|
47
|
+
rc = 1
|
|
48
|
+
except Exception as e:
|
|
49
|
+
logging.exception("failed: %s", v)
|
|
50
|
+
print(f"vidwit: error on {v}: {e}", file=sys.stderr)
|
|
51
|
+
rc = 1
|
|
52
|
+
return rc
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def _build_parser() -> argparse.ArgumentParser:
|
|
56
|
+
p = argparse.ArgumentParser(
|
|
57
|
+
prog="vidwit",
|
|
58
|
+
description="Multimodal video witness — emit exhaustive markdown next to each video.",
|
|
59
|
+
epilog="Run `vidwit init` to drop a starter vidwit.toml into the current "
|
|
60
|
+
"directory (or `vidwit init --user` for ~/.config/vidwit/).",
|
|
61
|
+
)
|
|
62
|
+
p.add_argument("inputs", nargs="+", help="video file(s) or directory(ies)")
|
|
63
|
+
p.add_argument("--config", type=Path, default=None,
|
|
64
|
+
help="path to vidwit.toml (default: ./vidwit.toml or ~/.config/vidwit/vidwit.toml)")
|
|
65
|
+
p.add_argument("--fps", type=float, default=None, help="frame sampling rate (default 1.0)")
|
|
66
|
+
p.add_argument("--window", type=float, default=None, help="window length seconds (default 10.0)")
|
|
67
|
+
p.add_argument("--overlap", type=float, default=None, help="window overlap seconds (default 1.0)")
|
|
68
|
+
p.add_argument("--overwrite", action="store_true", help="replace existing .md")
|
|
69
|
+
p.add_argument(
|
|
70
|
+
"--no-resume", action="store_true",
|
|
71
|
+
help="force re-run; ignore cached transcript/frames/chunks in scratch dir",
|
|
72
|
+
)
|
|
73
|
+
p.add_argument("--keep-scratch", action="store_true", help="don't delete scratch dir on success")
|
|
74
|
+
p.add_argument("--jobs", type=int, default=None, help="threads for ffmpeg/whisper (default nproc)")
|
|
75
|
+
p.add_argument(
|
|
76
|
+
"--paths",
|
|
77
|
+
action="append",
|
|
78
|
+
default=[],
|
|
79
|
+
metavar="KEY:VAL",
|
|
80
|
+
help="yt-dlp style overrides: home:<dir> for final .md, temp:<dir> for scratch. Repeatable.",
|
|
81
|
+
)
|
|
82
|
+
p.add_argument("--ext", action="append", default=[], help="extra video extension (repeatable)")
|
|
83
|
+
p.add_argument("--default-speaker", default=None, help="label transcribed words with this name")
|
|
84
|
+
p.add_argument("--prompt", type=Path, default=None, help="path to system prompt markdown")
|
|
85
|
+
p.add_argument("--audio-language", default=None,
|
|
86
|
+
help="ISO language hint for whisper (e.g. 'de'); skips auto-detect")
|
|
87
|
+
p.add_argument("--notes", default=None,
|
|
88
|
+
help="free-text context forwarded to the LLM in every chunk")
|
|
89
|
+
p.add_argument("-o", "--output", type=Path, default=None,
|
|
90
|
+
help="explicit output path (single-input only); relative or absolute")
|
|
91
|
+
p.add_argument("--frame-width", type=int, default=None,
|
|
92
|
+
help="downscale frames to fit within this width (default 256)")
|
|
93
|
+
p.add_argument("--frame-height", type=int, default=None,
|
|
94
|
+
help="downscale frames to fit within this height (default 144)")
|
|
95
|
+
p.add_argument(
|
|
96
|
+
"--max-tokens", type=int, default=None,
|
|
97
|
+
help="cumulative LLM token cap per video; abort and assemble what is "
|
|
98
|
+
"done when exceeded. None = no cap.",
|
|
99
|
+
)
|
|
100
|
+
|
|
101
|
+
llm = p.add_argument_group("LLM")
|
|
102
|
+
llm.add_argument("--llm", dest="llm_provider", default=None,
|
|
103
|
+
choices=["dummy", "anthropic", "openai", "lmstudio"])
|
|
104
|
+
llm.add_argument("--model", dest="llm_model", default=None)
|
|
105
|
+
llm.add_argument("--base-url", dest="llm_base_url", default=None,
|
|
106
|
+
help="OpenAI-compatible endpoint (e.g. http://localhost:1234/v1)")
|
|
107
|
+
llm.add_argument("--timeout", dest="llm_timeout", type=float, default=None,
|
|
108
|
+
help="HTTP timeout for LLM requests in seconds (default 600)")
|
|
109
|
+
llm.add_argument(
|
|
110
|
+
"--extra-body", dest="llm_extra_body", action="append", default=[],
|
|
111
|
+
metavar="KEY=JSON",
|
|
112
|
+
help="extra field merged into the chat-completions request payload. "
|
|
113
|
+
"VALUE is parsed as JSON; falls back to string on parse failure. "
|
|
114
|
+
"Repeatable. Example: "
|
|
115
|
+
"--extra-body 'chat_template_kwargs={\"enable_thinking\":false}'",
|
|
116
|
+
)
|
|
117
|
+
|
|
118
|
+
p.add_argument("--whisper-model", default=None,
|
|
119
|
+
help="whisper model name (tiny, base, small, medium, large-v3)")
|
|
120
|
+
p.add_argument("--whisper-device", default=None, choices=["auto", "cpu", "cuda"])
|
|
121
|
+
|
|
122
|
+
p.add_argument("-v", "--verbose", action="store_true")
|
|
123
|
+
p.add_argument("--version", action="version", version=f"vidwit {__version__}")
|
|
124
|
+
return p
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
def _make_config(args: argparse.Namespace) -> cfg_mod.Config:
|
|
128
|
+
cfg = cfg_mod.Config()
|
|
129
|
+
config_path = args.config or cfg_mod.find_config_file()
|
|
130
|
+
if config_path is not None:
|
|
131
|
+
if args.config is not None and not config_path.exists():
|
|
132
|
+
raise SystemExit(f"vidwit: --config {config_path} not found")
|
|
133
|
+
cfg = cfg_mod.from_file(config_path, cfg)
|
|
134
|
+
logging.getLogger("vidwit").info("loaded config: %s", config_path)
|
|
135
|
+
cfg = cfg_mod.from_env(cfg)
|
|
136
|
+
|
|
137
|
+
if args.fps is not None: cfg = replace(cfg, fps=args.fps)
|
|
138
|
+
if args.window is not None: cfg = replace(cfg, window=args.window)
|
|
139
|
+
if args.overlap is not None: cfg = replace(cfg, overlap=args.overlap)
|
|
140
|
+
if args.overwrite: cfg = replace(cfg, overwrite=True)
|
|
141
|
+
if args.no_resume: cfg = replace(cfg, resume=False)
|
|
142
|
+
if args.keep_scratch: cfg = replace(cfg, keep_scratch=True)
|
|
143
|
+
if args.jobs is not None: cfg = replace(cfg, jobs=args.jobs)
|
|
144
|
+
if args.default_speaker: cfg = replace(cfg, default_speaker=args.default_speaker)
|
|
145
|
+
if args.prompt: cfg = replace(cfg, prompt_path=args.prompt)
|
|
146
|
+
if args.audio_language: cfg = replace(cfg, audio_language=args.audio_language)
|
|
147
|
+
if args.notes: cfg = replace(cfg, notes=args.notes)
|
|
148
|
+
if args.output: cfg = replace(cfg, output_override=args.output.expanduser())
|
|
149
|
+
if args.frame_width is not None: cfg = replace(cfg, frame_width=args.frame_width)
|
|
150
|
+
if args.frame_height is not None: cfg = replace(cfg, frame_height=args.frame_height)
|
|
151
|
+
if args.max_tokens is not None: cfg = replace(cfg, max_tokens=args.max_tokens)
|
|
152
|
+
if args.whisper_model: cfg = replace(cfg, whisper_model=args.whisper_model)
|
|
153
|
+
if args.whisper_device: cfg = replace(cfg, whisper_device=args.whisper_device)
|
|
154
|
+
|
|
155
|
+
if args.ext:
|
|
156
|
+
exts = tuple({*cfg.video_exts, *(_normalise_ext(e) for e in args.ext)})
|
|
157
|
+
cfg = replace(cfg, video_exts=exts)
|
|
158
|
+
|
|
159
|
+
home, temp = _parse_paths(args.paths)
|
|
160
|
+
if home: cfg = replace(cfg, paths_home=home)
|
|
161
|
+
if temp: cfg = replace(cfg, paths_temp=temp)
|
|
162
|
+
|
|
163
|
+
llm = cfg.llm
|
|
164
|
+
if args.llm_provider: llm = replace(llm, provider=args.llm_provider)
|
|
165
|
+
if args.llm_model: llm = replace(llm, model=args.llm_model)
|
|
166
|
+
if args.llm_base_url: llm = replace(llm, base_url=args.llm_base_url)
|
|
167
|
+
if args.llm_timeout is not None: llm = replace(llm, request_timeout=args.llm_timeout)
|
|
168
|
+
if args.llm_extra_body:
|
|
169
|
+
merged = {**llm.extra_body, **_parse_extra_body(args.llm_extra_body)}
|
|
170
|
+
llm = replace(llm, extra_body=merged)
|
|
171
|
+
cfg = replace(cfg, llm=llm)
|
|
172
|
+
|
|
173
|
+
return cfg
|
|
174
|
+
|
|
175
|
+
|
|
176
|
+
def _parse_paths(items: list[str]) -> tuple[Path | None, Path | None]:
|
|
177
|
+
home = temp = None
|
|
178
|
+
for raw in items:
|
|
179
|
+
if ":" not in raw:
|
|
180
|
+
raise SystemExit(f"--paths expects KEY:VAL, got {raw!r}")
|
|
181
|
+
key, val = raw.split(":", 1)
|
|
182
|
+
key = key.strip().lower()
|
|
183
|
+
path = Path(val).expanduser()
|
|
184
|
+
if key == "home":
|
|
185
|
+
home = path
|
|
186
|
+
elif key == "temp":
|
|
187
|
+
temp = path
|
|
188
|
+
else:
|
|
189
|
+
raise SystemExit(f"--paths unknown key {key!r} (want home or temp)")
|
|
190
|
+
return home, temp
|
|
191
|
+
|
|
192
|
+
|
|
193
|
+
def _parse_extra_body(items: list[str]) -> dict:
|
|
194
|
+
out: dict = {}
|
|
195
|
+
for raw in items:
|
|
196
|
+
if "=" not in raw:
|
|
197
|
+
raise SystemExit(f"--extra-body expects KEY=VALUE, got {raw!r}")
|
|
198
|
+
key, val = raw.split("=", 1)
|
|
199
|
+
key = key.strip()
|
|
200
|
+
try:
|
|
201
|
+
parsed = json.loads(val)
|
|
202
|
+
except json.JSONDecodeError:
|
|
203
|
+
parsed = val
|
|
204
|
+
out[key] = parsed
|
|
205
|
+
return out
|
|
206
|
+
|
|
207
|
+
|
|
208
|
+
def _normalise_ext(ext: str) -> str:
|
|
209
|
+
ext = ext.strip().lower()
|
|
210
|
+
return ext if ext.startswith(".") else f".{ext}"
|
|
211
|
+
|
|
212
|
+
|
|
213
|
+
def _expand_inputs(args: list[str], exts: tuple[str, ...]) -> list[Path]:
|
|
214
|
+
out: list[Path] = []
|
|
215
|
+
ext_set = {e.lower() for e in exts}
|
|
216
|
+
for raw in args:
|
|
217
|
+
p = Path(raw).expanduser()
|
|
218
|
+
if not p.exists():
|
|
219
|
+
print(f"vidwit: not found: {p}", file=sys.stderr)
|
|
220
|
+
continue
|
|
221
|
+
if p.is_dir():
|
|
222
|
+
for f in sorted(p.rglob("*")):
|
|
223
|
+
if f.is_file() and f.suffix.lower() in ext_set:
|
|
224
|
+
out.append(f)
|
|
225
|
+
elif p.is_file():
|
|
226
|
+
out.append(p)
|
|
227
|
+
# Dedupe, keep order.
|
|
228
|
+
seen: set[Path] = set()
|
|
229
|
+
unique: list[Path] = []
|
|
230
|
+
for f in out:
|
|
231
|
+
r = f.resolve()
|
|
232
|
+
if r not in seen:
|
|
233
|
+
seen.add(r)
|
|
234
|
+
unique.append(f)
|
|
235
|
+
return unique
|
|
236
|
+
|
|
237
|
+
|
|
238
|
+
if __name__ == "__main__":
|
|
239
|
+
raise SystemExit(main())
|
vidwit/config.py
ADDED
|
@@ -0,0 +1,139 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import os
|
|
4
|
+
import tomllib
|
|
5
|
+
from dataclasses import dataclass, field, replace
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
DEFAULT_VIDEO_EXTS = (".mp4", ".mkv", ".mov", ".webm", ".avi")
|
|
10
|
+
|
|
11
|
+
CONFIG_FILENAME = "vidwit.toml"
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
@dataclass(slots=True)
|
|
15
|
+
class LLMConfig:
|
|
16
|
+
provider: str = "dummy" # "anthropic" | "openai" | "lmstudio" | "dummy"
|
|
17
|
+
model: str = ""
|
|
18
|
+
base_url: str | None = None
|
|
19
|
+
api_key: str | None = None
|
|
20
|
+
max_output_tokens: int = 2048
|
|
21
|
+
request_timeout: float = 600.0 # seconds; bumped for slow local models
|
|
22
|
+
extra_body: dict = field(default_factory=dict) # merged into chat-completions payload
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
@dataclass(slots=True)
|
|
26
|
+
class Config:
|
|
27
|
+
fps: float = 1.0
|
|
28
|
+
window: float = 10.0
|
|
29
|
+
overlap: float = 1.0
|
|
30
|
+
overwrite: bool = False
|
|
31
|
+
resume: bool = True # reuse cached transcript/frames/chunks from scratch dir
|
|
32
|
+
keep_scratch: bool = False
|
|
33
|
+
jobs: int = field(default_factory=lambda: max(1, os.cpu_count() or 1))
|
|
34
|
+
paths_home: Path | None = None # final .md output dir override
|
|
35
|
+
paths_temp: Path | None = None # scratch dir override
|
|
36
|
+
video_exts: tuple[str, ...] = DEFAULT_VIDEO_EXTS
|
|
37
|
+
default_speaker: str | None = None
|
|
38
|
+
prompt_path: Path | None = None
|
|
39
|
+
whisper_model: str = "small"
|
|
40
|
+
whisper_device: str = "auto" # "auto" | "cpu" | "cuda"
|
|
41
|
+
audio_language: str | None = None # ISO code, e.g. "de"; forces whisper language
|
|
42
|
+
notes: str | None = None # free-text forwarded to LLM capture metadata
|
|
43
|
+
output_override: Path | None = None # explicit -o/--output path; single-input only
|
|
44
|
+
frame_width: int = 256 # downscale frames to fit within W x H (aspect preserved)
|
|
45
|
+
frame_height: int = 144
|
|
46
|
+
max_tokens: int | None = None # cumulative LLM token cap per video; abort + assemble when exceeded
|
|
47
|
+
llm: LLMConfig = field(default_factory=LLMConfig)
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def from_env(base: Config | None = None) -> Config:
|
|
51
|
+
cfg = base or Config()
|
|
52
|
+
env = os.environ
|
|
53
|
+
return replace(
|
|
54
|
+
cfg,
|
|
55
|
+
whisper_model=env.get("VIDWIT_WHISPER_MODEL", cfg.whisper_model),
|
|
56
|
+
whisper_device=env.get("VIDWIT_WHISPER_DEVICE", cfg.whisper_device),
|
|
57
|
+
llm=LLMConfig(
|
|
58
|
+
provider=env.get("VIDWIT_LLM_PROVIDER", cfg.llm.provider),
|
|
59
|
+
model=env.get("VIDWIT_LLM_MODEL", cfg.llm.model),
|
|
60
|
+
base_url=env.get("VIDWIT_LLM_BASE_URL", cfg.llm.base_url),
|
|
61
|
+
api_key=env.get(
|
|
62
|
+
"VIDWIT_LLM_API_KEY",
|
|
63
|
+
env.get("ANTHROPIC_API_KEY") or env.get("OPENAI_API_KEY") or cfg.llm.api_key,
|
|
64
|
+
),
|
|
65
|
+
max_output_tokens=int(
|
|
66
|
+
env.get("VIDWIT_LLM_MAX_OUTPUT_TOKENS", cfg.llm.max_output_tokens)
|
|
67
|
+
),
|
|
68
|
+
),
|
|
69
|
+
)
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def from_file(path: Path, base: Config | None = None) -> Config:
|
|
73
|
+
"""Load TOML config. Unknown keys ignored. Empty/missing file → base unchanged."""
|
|
74
|
+
cfg = base or Config()
|
|
75
|
+
if not path.exists():
|
|
76
|
+
return cfg
|
|
77
|
+
data = tomllib.loads(path.read_text(encoding="utf-8"))
|
|
78
|
+
|
|
79
|
+
defaults = data.get("defaults", {}) or {}
|
|
80
|
+
cfg = replace(
|
|
81
|
+
cfg,
|
|
82
|
+
fps=float(defaults.get("fps", cfg.fps)),
|
|
83
|
+
window=float(defaults.get("window", cfg.window)),
|
|
84
|
+
overlap=float(defaults.get("overlap", cfg.overlap)),
|
|
85
|
+
jobs=int(defaults.get("jobs", cfg.jobs)),
|
|
86
|
+
default_speaker=defaults.get("default_speaker", cfg.default_speaker),
|
|
87
|
+
prompt_path=Path(defaults["prompt"]).expanduser() if "prompt" in defaults else cfg.prompt_path,
|
|
88
|
+
frame_width=int(defaults.get("frame_width", cfg.frame_width)),
|
|
89
|
+
frame_height=int(defaults.get("frame_height", cfg.frame_height)),
|
|
90
|
+
max_tokens=defaults.get("max_tokens", cfg.max_tokens),
|
|
91
|
+
)
|
|
92
|
+
|
|
93
|
+
whisper = data.get("whisper", {}) or {}
|
|
94
|
+
cfg = replace(
|
|
95
|
+
cfg,
|
|
96
|
+
whisper_model=whisper.get("model", cfg.whisper_model),
|
|
97
|
+
whisper_device=whisper.get("device", cfg.whisper_device),
|
|
98
|
+
)
|
|
99
|
+
|
|
100
|
+
video = data.get("video", {}) or {}
|
|
101
|
+
cfg = replace(
|
|
102
|
+
cfg,
|
|
103
|
+
audio_language=video.get("audio_language", cfg.audio_language),
|
|
104
|
+
notes=video.get("notes", cfg.notes),
|
|
105
|
+
)
|
|
106
|
+
|
|
107
|
+
llm_data = data.get("llm", {}) or {}
|
|
108
|
+
eb = llm_data.get("extra_body") or {}
|
|
109
|
+
cfg = replace(
|
|
110
|
+
cfg,
|
|
111
|
+
llm=LLMConfig(
|
|
112
|
+
provider=llm_data.get("provider", cfg.llm.provider),
|
|
113
|
+
model=llm_data.get("model", cfg.llm.model),
|
|
114
|
+
base_url=llm_data.get("base_url", cfg.llm.base_url),
|
|
115
|
+
api_key=llm_data.get("api_key", cfg.llm.api_key),
|
|
116
|
+
max_output_tokens=int(llm_data.get("max_output_tokens", cfg.llm.max_output_tokens)),
|
|
117
|
+
request_timeout=float(llm_data.get("request_timeout", cfg.llm.request_timeout)),
|
|
118
|
+
extra_body=dict(eb) if isinstance(eb, dict) else {},
|
|
119
|
+
),
|
|
120
|
+
)
|
|
121
|
+
|
|
122
|
+
paths = data.get("paths", {}) or {}
|
|
123
|
+
cfg = replace(
|
|
124
|
+
cfg,
|
|
125
|
+
paths_home=Path(paths["home"]).expanduser() if "home" in paths else cfg.paths_home,
|
|
126
|
+
paths_temp=Path(paths["temp"]).expanduser() if "temp" in paths else cfg.paths_temp,
|
|
127
|
+
)
|
|
128
|
+
return cfg
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
def find_config_file() -> Path | None:
|
|
132
|
+
"""Search standard locations for a vidwit.toml. First hit wins."""
|
|
133
|
+
cwd = Path.cwd() / CONFIG_FILENAME
|
|
134
|
+
xdg = Path(os.environ.get("XDG_CONFIG_HOME") or (Path.home() / ".config"))
|
|
135
|
+
home = xdg / "vidwit" / CONFIG_FILENAME
|
|
136
|
+
for cand in (cwd, home):
|
|
137
|
+
if cand.is_file():
|
|
138
|
+
return cand
|
|
139
|
+
return None
|
vidwit/ffmpeg_io.py
ADDED
|
@@ -0,0 +1,102 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import json
|
|
4
|
+
import shutil
|
|
5
|
+
import subprocess
|
|
6
|
+
from dataclasses import dataclass
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
class FFmpegMissingError(RuntimeError):
|
|
11
|
+
pass
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def require_ffmpeg() -> None:
|
|
15
|
+
if shutil.which("ffmpeg") is None or shutil.which("ffprobe") is None:
|
|
16
|
+
raise FFmpegMissingError("ffmpeg + ffprobe required on PATH")
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
@dataclass(slots=True, frozen=True)
|
|
20
|
+
class VideoInfo:
|
|
21
|
+
path: Path
|
|
22
|
+
duration_s: float
|
|
23
|
+
width: int
|
|
24
|
+
height: int
|
|
25
|
+
has_audio: bool
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def probe(path: Path) -> VideoInfo:
|
|
29
|
+
require_ffmpeg()
|
|
30
|
+
out = subprocess.run(
|
|
31
|
+
[
|
|
32
|
+
"ffprobe", "-v", "error",
|
|
33
|
+
"-print_format", "json",
|
|
34
|
+
"-show_format",
|
|
35
|
+
"-show_streams",
|
|
36
|
+
str(path),
|
|
37
|
+
],
|
|
38
|
+
check=True, capture_output=True, text=True,
|
|
39
|
+
).stdout
|
|
40
|
+
data = json.loads(out)
|
|
41
|
+
duration = float(data.get("format", {}).get("duration", 0.0))
|
|
42
|
+
width = height = 0
|
|
43
|
+
has_audio = False
|
|
44
|
+
for s in data.get("streams", []):
|
|
45
|
+
if s.get("codec_type") == "video" and width == 0:
|
|
46
|
+
width = int(s.get("width", 0))
|
|
47
|
+
height = int(s.get("height", 0))
|
|
48
|
+
if s.get("codec_type") == "audio":
|
|
49
|
+
has_audio = True
|
|
50
|
+
return VideoInfo(path=path, duration_s=duration, width=width, height=height, has_audio=has_audio)
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def extract_audio(src: Path, dst: Path, sample_rate: int = 16000, threads: int = 0) -> Path:
|
|
54
|
+
"""Extract mono PCM WAV at sample_rate; whisper expects 16k mono."""
|
|
55
|
+
require_ffmpeg()
|
|
56
|
+
dst.parent.mkdir(parents=True, exist_ok=True)
|
|
57
|
+
subprocess.run(
|
|
58
|
+
[
|
|
59
|
+
"ffmpeg", "-v", "error", "-y",
|
|
60
|
+
"-i", str(src),
|
|
61
|
+
"-vn", "-ac", "1", "-ar", str(sample_rate),
|
|
62
|
+
"-c:a", "pcm_s16le",
|
|
63
|
+
"-threads", str(threads),
|
|
64
|
+
str(dst),
|
|
65
|
+
],
|
|
66
|
+
check=True,
|
|
67
|
+
)
|
|
68
|
+
return dst
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def extract_frames(
|
|
72
|
+
src: Path,
|
|
73
|
+
dst_dir: Path,
|
|
74
|
+
fps: float,
|
|
75
|
+
threads: int = 0,
|
|
76
|
+
quality: int = 4,
|
|
77
|
+
max_width: int | None = None,
|
|
78
|
+
max_height: int | None = None,
|
|
79
|
+
) -> list[Path]:
|
|
80
|
+
"""Extract frames at given fps as JPEG. Returns paths sorted by index.
|
|
81
|
+
|
|
82
|
+
If max_width and max_height are set, frames are downscaled to fit within
|
|
83
|
+
that box while preserving aspect ratio (no padding). 0 / None disables.
|
|
84
|
+
"""
|
|
85
|
+
require_ffmpeg()
|
|
86
|
+
dst_dir.mkdir(parents=True, exist_ok=True)
|
|
87
|
+
pattern = dst_dir / "f_%08d.jpg"
|
|
88
|
+
vf = f"fps={fps}"
|
|
89
|
+
if max_width and max_height:
|
|
90
|
+
vf += f",scale={max_width}:{max_height}:force_original_aspect_ratio=decrease"
|
|
91
|
+
subprocess.run(
|
|
92
|
+
[
|
|
93
|
+
"ffmpeg", "-v", "error", "-y",
|
|
94
|
+
"-i", str(src),
|
|
95
|
+
"-vf", vf,
|
|
96
|
+
"-q:v", str(quality),
|
|
97
|
+
"-threads", str(threads),
|
|
98
|
+
str(pattern),
|
|
99
|
+
],
|
|
100
|
+
check=True,
|
|
101
|
+
)
|
|
102
|
+
return sorted(dst_dir.glob("f_*.jpg"))
|
vidwit/init_cmd.py
ADDED
|
@@ -0,0 +1,71 @@
|
|
|
1
|
+
"""`vidwit init` — drop the bundled sample config (and optional prompt)
|
|
2
|
+
into the current directory or `~/.config/vidwit/` so a fresh
|
|
3
|
+
`pip install vidwit` user can start configuring without cloning the repo.
|
|
4
|
+
"""
|
|
5
|
+
from __future__ import annotations
|
|
6
|
+
|
|
7
|
+
import argparse
|
|
8
|
+
import os
|
|
9
|
+
import sys
|
|
10
|
+
from importlib.resources import files
|
|
11
|
+
from pathlib import Path
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def run(argv: list[str]) -> int:
|
|
15
|
+
p = argparse.ArgumentParser(
|
|
16
|
+
prog="vidwit init",
|
|
17
|
+
description="Write a starter vidwit.toml (and optionally a prompt template) "
|
|
18
|
+
"to the current directory or ~/.config/vidwit/.",
|
|
19
|
+
)
|
|
20
|
+
p.add_argument(
|
|
21
|
+
"--user", action="store_true",
|
|
22
|
+
help="write to ~/.config/vidwit/ instead of the current directory",
|
|
23
|
+
)
|
|
24
|
+
p.add_argument(
|
|
25
|
+
"--prompt", action="store_true",
|
|
26
|
+
help="also drop vidwit_prompt.md alongside vidwit.toml",
|
|
27
|
+
)
|
|
28
|
+
p.add_argument(
|
|
29
|
+
"--force", "-f", action="store_true",
|
|
30
|
+
help="overwrite existing files",
|
|
31
|
+
)
|
|
32
|
+
args = p.parse_args(argv)
|
|
33
|
+
|
|
34
|
+
if args.user:
|
|
35
|
+
xdg = Path(os.environ.get("XDG_CONFIG_HOME") or (Path.home() / ".config"))
|
|
36
|
+
target_dir = xdg / "vidwit"
|
|
37
|
+
else:
|
|
38
|
+
target_dir = Path.cwd()
|
|
39
|
+
target_dir.mkdir(parents=True, exist_ok=True)
|
|
40
|
+
|
|
41
|
+
written: list[Path] = []
|
|
42
|
+
skipped: list[Path] = []
|
|
43
|
+
|
|
44
|
+
items = [("vidwit.toml", "vidwit.toml")]
|
|
45
|
+
if args.prompt:
|
|
46
|
+
items.append(("vidwit_prompt.md", "vidwit_prompt.md"))
|
|
47
|
+
|
|
48
|
+
templates = files("vidwit.templates")
|
|
49
|
+
for src_name, dst_name in items:
|
|
50
|
+
dst = target_dir / dst_name
|
|
51
|
+
if dst.exists() and not args.force:
|
|
52
|
+
skipped.append(dst)
|
|
53
|
+
continue
|
|
54
|
+
dst.write_text(templates.joinpath(src_name).read_text(encoding="utf-8"),
|
|
55
|
+
encoding="utf-8")
|
|
56
|
+
written.append(dst)
|
|
57
|
+
|
|
58
|
+
for f in written:
|
|
59
|
+
print(f"wrote {f}")
|
|
60
|
+
for f in skipped:
|
|
61
|
+
print(f"exists, skipped {f} (use --force to overwrite)", file=sys.stderr)
|
|
62
|
+
|
|
63
|
+
if skipped and not written:
|
|
64
|
+
return 1
|
|
65
|
+
|
|
66
|
+
if written:
|
|
67
|
+
toml = next((f for f in written if f.name == "vidwit.toml"), None)
|
|
68
|
+
if toml is not None:
|
|
69
|
+
print()
|
|
70
|
+
print(f"Next: edit {toml} and set [llm] api_key.")
|
|
71
|
+
return 0
|