subxx 0.3.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- __main__.py +642 -0
- subxx-0.3.0.dist-info/METADATA +896 -0
- subxx-0.3.0.dist-info/RECORD +7 -0
- subxx-0.3.0.dist-info/WHEEL +4 -0
- subxx-0.3.0.dist-info/entry_points.txt +2 -0
- subxx-0.3.0.dist-info/licenses/LICENSE +190 -0
- subxx.py +1691 -0
__main__.py
ADDED
|
@@ -0,0 +1,642 @@
|
|
|
1
|
+
"""
|
|
2
|
+
__main__.py - CLI and HTTP API entry point for subxx
|
|
3
|
+
|
|
4
|
+
Provides typer-based CLI commands and optional FastAPI HTTP server.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
import sys
|
|
8
|
+
from pathlib import Path
|
|
9
|
+
from typing import Optional
|
|
10
|
+
import typer
|
|
11
|
+
|
|
12
|
+
# Fix Windows console encoding for emoji support
|
|
13
|
+
if sys.platform == "win32":
|
|
14
|
+
try:
|
|
15
|
+
|
|
16
|
+
sys.stdout.reconfigure(encoding="utf-8")
|
|
17
|
+
sys.stderr.reconfigure(encoding="utf-8")
|
|
18
|
+
except Exception:
|
|
19
|
+
pass # If reconfigure fails, continue with default encoding
|
|
20
|
+
from subxx import (
|
|
21
|
+
fetch_subs,
|
|
22
|
+
load_config,
|
|
23
|
+
get_default,
|
|
24
|
+
setup_logging,
|
|
25
|
+
generate_file_hash,
|
|
26
|
+
EXIT_SUCCESS,
|
|
27
|
+
EXIT_USER_CANCELLED,
|
|
28
|
+
EXIT_NO_SUBTITLES,
|
|
29
|
+
EXIT_NETWORK_ERROR,
|
|
30
|
+
EXIT_INVALID_URL,
|
|
31
|
+
EXIT_CONFIG_ERROR,
|
|
32
|
+
EXIT_FILE_ERROR,
|
|
33
|
+
)
|
|
34
|
+
|
|
35
|
+
app = typer.Typer(
|
|
36
|
+
name="subxx",
|
|
37
|
+
help="Subtitle fetching toolkit - Download subtitles from video URLs",
|
|
38
|
+
add_completion=True,
|
|
39
|
+
)
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
@app.command()
|
|
43
|
+
def version():
|
|
44
|
+
"""Show version information."""
|
|
45
|
+
try:
|
|
46
|
+
import importlib.metadata
|
|
47
|
+
|
|
48
|
+
ver = importlib.metadata.version("subxx")
|
|
49
|
+
except importlib.metadata.PackageNotFoundError:
|
|
50
|
+
ver = "0.1.0"
|
|
51
|
+
|
|
52
|
+
typer.echo(f"subxx {ver}")
|
|
53
|
+
typer.echo("Subtitle fetching toolkit")
|
|
54
|
+
typer.echo("https://gist.github.com/cprima/subxx")
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
@app.command()
|
|
58
|
+
def list(
|
|
59
|
+
url: str,
|
|
60
|
+
quiet: bool = typer.Option(False, "--quiet", "-q", help="Errors only"),
|
|
61
|
+
verbose: bool = typer.Option(False, "--verbose", "-v", help="Debug output"),
|
|
62
|
+
):
|
|
63
|
+
"""List available subtitles for a video without downloading.
|
|
64
|
+
|
|
65
|
+
Examples:
|
|
66
|
+
uv run subxx list https://youtu.be/VIDEO_ID
|
|
67
|
+
"""
|
|
68
|
+
import yt_dlp
|
|
69
|
+
|
|
70
|
+
# Load config for log file
|
|
71
|
+
config = load_config()
|
|
72
|
+
log_file = config.get("logging", {}).get("log_file")
|
|
73
|
+
|
|
74
|
+
# Setup logging
|
|
75
|
+
logger = setup_logging(verbose=verbose, quiet=quiet, log_file=log_file)
|
|
76
|
+
|
|
77
|
+
ydl_opts = {
|
|
78
|
+
"skip_download": True,
|
|
79
|
+
"quiet": True,
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
try:
|
|
83
|
+
with yt_dlp.YoutubeDL(ydl_opts) as ydl:
|
|
84
|
+
info = ydl.extract_info(url, download=False)
|
|
85
|
+
|
|
86
|
+
typer.echo(f"\nš¹ Video: {info.get('title', 'Unknown')}")
|
|
87
|
+
duration = info.get("duration", 0)
|
|
88
|
+
typer.echo(f"š Duration: {duration // 60}:{duration % 60:02d}\n")
|
|
89
|
+
|
|
90
|
+
# Manual subtitles
|
|
91
|
+
manual_subs = info.get("subtitles", {})
|
|
92
|
+
if manual_subs:
|
|
93
|
+
typer.echo("ā
Manual subtitles:")
|
|
94
|
+
for lang in sorted(manual_subs.keys()):
|
|
95
|
+
typer.echo(f" - {lang}")
|
|
96
|
+
|
|
97
|
+
# Auto-generated subtitles
|
|
98
|
+
auto_subs = info.get("automatic_captions", {})
|
|
99
|
+
if auto_subs:
|
|
100
|
+
if manual_subs:
|
|
101
|
+
typer.echo("")
|
|
102
|
+
typer.echo("š¤ Auto-generated subtitles:")
|
|
103
|
+
for lang in sorted(auto_subs.keys()):
|
|
104
|
+
typer.echo(f" - {lang}")
|
|
105
|
+
|
|
106
|
+
if not manual_subs and not auto_subs:
|
|
107
|
+
typer.echo("ā No subtitles available")
|
|
108
|
+
raise typer.Exit(code=EXIT_NO_SUBTITLES)
|
|
109
|
+
|
|
110
|
+
except typer.Exit:
|
|
111
|
+
raise # Re-raise typer.Exit without catching
|
|
112
|
+
except Exception as e:
|
|
113
|
+
logger.error(f"Failed to list subtitles: {e}")
|
|
114
|
+
raise typer.Exit(code=EXIT_NETWORK_ERROR)
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
@app.command()
|
|
118
|
+
def subs(
|
|
119
|
+
url: str,
|
|
120
|
+
langs: Optional[str] = typer.Option(
|
|
121
|
+
None, "--langs", "-l", help="Language codes (en,de,fr or all)"
|
|
122
|
+
),
|
|
123
|
+
# Format selection (one of these)
|
|
124
|
+
fmt: Optional[str] = typer.Option(
|
|
125
|
+
None, "--fmt", "-f", help="Output format (srt/vtt/txt/md/pdf)"
|
|
126
|
+
),
|
|
127
|
+
srt: bool = typer.Option(
|
|
128
|
+
False, "--srt", help="Download SRT subtitle file (default)"
|
|
129
|
+
),
|
|
130
|
+
vtt: bool = typer.Option(False, "--vtt", help="Download VTT subtitle file"),
|
|
131
|
+
txt: bool = typer.Option(False, "--txt", help="Extract to plain text"),
|
|
132
|
+
md: bool = typer.Option(False, "--md", help="Extract to Markdown"),
|
|
133
|
+
pdf: bool = typer.Option(False, "--pdf", help="Extract to PDF"),
|
|
134
|
+
# Other options
|
|
135
|
+
auto: Optional[bool] = typer.Option(
|
|
136
|
+
None, "--auto/--no-auto", help="Include auto-generated subs"
|
|
137
|
+
),
|
|
138
|
+
output_dir: Optional[str] = typer.Option(
|
|
139
|
+
None, "--output-dir", "-o", help="Output directory"
|
|
140
|
+
),
|
|
141
|
+
output_folder: Optional[str] = typer.Option(
|
|
142
|
+
None,
|
|
143
|
+
"--output-folder",
|
|
144
|
+
help="Subfolder name for organized output (creates folder with original.srt and transcript.md)",
|
|
145
|
+
),
|
|
146
|
+
sanitize: Optional[str] = typer.Option(
|
|
147
|
+
None, "--sanitize", help="Filename sanitization (safe/nospaces/slugify)"
|
|
148
|
+
),
|
|
149
|
+
timestamps: Optional[int] = typer.Option(
|
|
150
|
+
None,
|
|
151
|
+
"--timestamps",
|
|
152
|
+
"-t",
|
|
153
|
+
help="Add timestamp markers every N seconds (for txt/md/pdf)",
|
|
154
|
+
),
|
|
155
|
+
chapters: bool = typer.Option(
|
|
156
|
+
False,
|
|
157
|
+
"--chapters",
|
|
158
|
+
help="Use chapter markers from metadata (YouTube chapters, for txt/md/pdf)",
|
|
159
|
+
),
|
|
160
|
+
force: bool = typer.Option(False, "--force", help="Overwrite without prompting"),
|
|
161
|
+
skip_existing: bool = typer.Option(
|
|
162
|
+
False, "--skip-existing", help="Skip existing files"
|
|
163
|
+
),
|
|
164
|
+
dry_run: bool = typer.Option(
|
|
165
|
+
False, "--dry-run", help="Preview without downloading"
|
|
166
|
+
),
|
|
167
|
+
quiet: bool = typer.Option(False, "--quiet", "-q", help="Errors only"),
|
|
168
|
+
verbose: bool = typer.Option(False, "--verbose", "-v", help="Debug output"),
|
|
169
|
+
):
|
|
170
|
+
"""Fetch subtitles for a video URL.
|
|
171
|
+
|
|
172
|
+
Format determines behavior:
|
|
173
|
+
srt/vtt: Download subtitle file
|
|
174
|
+
txt/md/pdf: Download SRT ā Extract text ā Delete SRT
|
|
175
|
+
|
|
176
|
+
Examples:
|
|
177
|
+
Subtitle file: uv run subxx subs <url>
|
|
178
|
+
Plain text: uv run subxx subs <url> --txt
|
|
179
|
+
Markdown: uv run subxx subs <url> --md
|
|
180
|
+
With timestamps: uv run subxx subs <url> --md --timestamps 300
|
|
181
|
+
With chapters: uv run subxx subs <url> --md --chapters
|
|
182
|
+
PDF: uv run subxx subs <url> --pdf
|
|
183
|
+
"""
|
|
184
|
+
# Load config
|
|
185
|
+
config = load_config()
|
|
186
|
+
|
|
187
|
+
# Get log file from config
|
|
188
|
+
log_file = config.get("logging", {}).get("log_file")
|
|
189
|
+
|
|
190
|
+
# Determine verbosity level
|
|
191
|
+
if quiet:
|
|
192
|
+
verbosity = "quiet"
|
|
193
|
+
elif verbose:
|
|
194
|
+
verbosity = "verbose"
|
|
195
|
+
else:
|
|
196
|
+
verbosity = "normal"
|
|
197
|
+
|
|
198
|
+
# Setup logging with appropriate level
|
|
199
|
+
logger = setup_logging(verbose=verbose, quiet=quiet, log_file=log_file)
|
|
200
|
+
|
|
201
|
+
# Determine format from flags (priority: specific flag > --fmt > config > default)
|
|
202
|
+
if md:
|
|
203
|
+
fmt = "md"
|
|
204
|
+
elif txt:
|
|
205
|
+
fmt = "txt"
|
|
206
|
+
elif pdf:
|
|
207
|
+
fmt = "pdf"
|
|
208
|
+
elif vtt:
|
|
209
|
+
fmt = "vtt"
|
|
210
|
+
elif srt:
|
|
211
|
+
fmt = "srt"
|
|
212
|
+
elif fmt:
|
|
213
|
+
# Use --fmt value
|
|
214
|
+
pass
|
|
215
|
+
else:
|
|
216
|
+
# Use config or default
|
|
217
|
+
fmt = get_default(config, "fmt", "srt")
|
|
218
|
+
|
|
219
|
+
# Apply other config defaults
|
|
220
|
+
langs = langs or get_default(config, "langs", "en")
|
|
221
|
+
auto = auto if auto is not None else get_default(config, "auto", True)
|
|
222
|
+
output_dir = output_dir or get_default(config, "output_dir", ".")
|
|
223
|
+
sanitize = sanitize or get_default(config, "sanitize", "safe")
|
|
224
|
+
|
|
225
|
+
# Determine if we need extraction
|
|
226
|
+
text_formats = {"txt", "md", "pdf"}
|
|
227
|
+
need_extraction = fmt in text_formats
|
|
228
|
+
download_fmt = (
|
|
229
|
+
"srt" if need_extraction else fmt
|
|
230
|
+
) # Always download SRT for extraction
|
|
231
|
+
|
|
232
|
+
# Handle prompting logic
|
|
233
|
+
prompt_overwrite = not force and not skip_existing
|
|
234
|
+
|
|
235
|
+
# Call fetch_subs with verbosity parameter
|
|
236
|
+
exit_code = fetch_subs(
|
|
237
|
+
url=url,
|
|
238
|
+
langs=langs,
|
|
239
|
+
fmt=download_fmt,
|
|
240
|
+
auto=auto,
|
|
241
|
+
output_dir=output_dir,
|
|
242
|
+
prompt_overwrite=prompt_overwrite,
|
|
243
|
+
skip_existing=skip_existing,
|
|
244
|
+
dry_run=dry_run,
|
|
245
|
+
verbosity=verbosity,
|
|
246
|
+
sanitize=sanitize,
|
|
247
|
+
)
|
|
248
|
+
|
|
249
|
+
# If download successful and extraction needed, extract text
|
|
250
|
+
if exit_code == EXIT_SUCCESS and need_extraction and not dry_run:
|
|
251
|
+
try:
|
|
252
|
+
# Check if extract dependencies are available
|
|
253
|
+
try:
|
|
254
|
+
import srt # noqa: F401
|
|
255
|
+
from fpdf import FPDF # noqa: F401
|
|
256
|
+
except ImportError:
|
|
257
|
+
logger.error("ā Error: Missing dependencies for text extraction")
|
|
258
|
+
logger.info("š” Install with: uv sync --extra extract")
|
|
259
|
+
raise typer.Exit(code=EXIT_CONFIG_ERROR)
|
|
260
|
+
|
|
261
|
+
# Import extract function
|
|
262
|
+
from subxx import extract_text
|
|
263
|
+
import yt_dlp
|
|
264
|
+
|
|
265
|
+
# Extract video ID from URL to generate hash for filenames
|
|
266
|
+
video_id = None
|
|
267
|
+
file_hash = None
|
|
268
|
+
try:
|
|
269
|
+
with yt_dlp.YoutubeDL({"quiet": True}) as ydl:
|
|
270
|
+
info = ydl.extract_info(url, download=False)
|
|
271
|
+
video_id = info.get("id")
|
|
272
|
+
if video_id:
|
|
273
|
+
file_hash = generate_file_hash(video_id)
|
|
274
|
+
logger.debug(
|
|
275
|
+
f"Generated file hash: {file_hash} from video ID: {video_id}"
|
|
276
|
+
)
|
|
277
|
+
except Exception as e:
|
|
278
|
+
logger.warning(f"Could not extract video ID for hash generation: {e}")
|
|
279
|
+
# Continue without hash
|
|
280
|
+
|
|
281
|
+
# Find downloaded subtitle files for this video
|
|
282
|
+
output_path_obj = Path(output_dir).expanduser()
|
|
283
|
+
# Get all subtitle files (SRT, since we always download SRT for extraction)
|
|
284
|
+
# sorted by modification time (most recent first)
|
|
285
|
+
all_subtitle_files = sorted(
|
|
286
|
+
output_path_obj.glob(
|
|
287
|
+
f"*.{download_fmt}"
|
|
288
|
+
), # Use download_fmt (srt), not final fmt (md/txt/pdf)
|
|
289
|
+
key=lambda p: p.stat().st_mtime,
|
|
290
|
+
reverse=True,
|
|
291
|
+
)
|
|
292
|
+
# Take only the most recent file(s) - assume they're from this download
|
|
293
|
+
subtitle_files = all_subtitle_files[:1] if all_subtitle_files else []
|
|
294
|
+
logger.debug(f"Found {len(subtitle_files)} recent subtitle file(s)")
|
|
295
|
+
|
|
296
|
+
if not subtitle_files:
|
|
297
|
+
logger.error("No subtitle files found to extract")
|
|
298
|
+
raise typer.Exit(code=EXIT_FILE_ERROR)
|
|
299
|
+
|
|
300
|
+
# Extract text from each subtitle file
|
|
301
|
+
extract_failed = []
|
|
302
|
+
for subtitle_file in subtitle_files:
|
|
303
|
+
logger.info(f"š Extracting text from: {subtitle_file.name}")
|
|
304
|
+
|
|
305
|
+
# Handle output_folder if specified
|
|
306
|
+
if output_folder:
|
|
307
|
+
# Create organized folder structure
|
|
308
|
+
folder_path = output_path_obj / output_folder
|
|
309
|
+
folder_path.mkdir(parents=True, exist_ok=True)
|
|
310
|
+
|
|
311
|
+
# Generate filenames with hash suffix if available
|
|
312
|
+
if file_hash:
|
|
313
|
+
original_name = f"original-{file_hash}.{download_fmt}"
|
|
314
|
+
transcript_name = f"transcript-{file_hash}.{fmt}"
|
|
315
|
+
else:
|
|
316
|
+
original_name = f"original.{download_fmt}"
|
|
317
|
+
transcript_name = f"transcript.{fmt}"
|
|
318
|
+
|
|
319
|
+
# Move SRT to folder as original-{hash}.srt
|
|
320
|
+
original_srt = folder_path / original_name
|
|
321
|
+
import shutil
|
|
322
|
+
|
|
323
|
+
shutil.move(str(subtitle_file), str(original_srt))
|
|
324
|
+
logger.debug(f"Moved {subtitle_file.name} to {original_srt}")
|
|
325
|
+
|
|
326
|
+
# Set output file as transcript-{hash}.{fmt}
|
|
327
|
+
output_file = str(folder_path / transcript_name)
|
|
328
|
+
# Update subtitle_file to new location
|
|
329
|
+
subtitle_file = original_srt
|
|
330
|
+
else:
|
|
331
|
+
output_file = None # Auto-generate
|
|
332
|
+
|
|
333
|
+
extract_exit = extract_text(
|
|
334
|
+
subtitle_file=str(subtitle_file),
|
|
335
|
+
output_format=fmt, # Use the final format (txt/md/pdf)
|
|
336
|
+
timestamp_interval=timestamps,
|
|
337
|
+
output_file=output_file,
|
|
338
|
+
force=force,
|
|
339
|
+
use_chapters=chapters,
|
|
340
|
+
)
|
|
341
|
+
|
|
342
|
+
if extract_exit != EXIT_SUCCESS:
|
|
343
|
+
extract_failed.append(subtitle_file)
|
|
344
|
+
else:
|
|
345
|
+
# Only delete SRT file if NOT using output_folder (keep original for re-processing)
|
|
346
|
+
if not output_folder:
|
|
347
|
+
subtitle_file.unlink()
|
|
348
|
+
logger.debug(
|
|
349
|
+
f"Deleted intermediate SRT file: {subtitle_file.name}"
|
|
350
|
+
)
|
|
351
|
+
|
|
352
|
+
if extract_failed:
|
|
353
|
+
logger.error(f"Failed to extract {len(extract_failed)} file(s)")
|
|
354
|
+
raise typer.Exit(code=EXIT_FILE_ERROR)
|
|
355
|
+
except typer.Exit:
|
|
356
|
+
raise # Re-raise typer.Exit
|
|
357
|
+
except Exception as e:
|
|
358
|
+
import traceback
|
|
359
|
+
|
|
360
|
+
logger.error(f"ā Extraction failed: {e}")
|
|
361
|
+
logger.debug(f"Traceback: {traceback.format_exc()}")
|
|
362
|
+
raise typer.Exit(code=EXIT_FILE_ERROR)
|
|
363
|
+
|
|
364
|
+
raise typer.Exit(code=exit_code)
|
|
365
|
+
|
|
366
|
+
|
|
367
|
+
@app.command()
|
|
368
|
+
def batch(
|
|
369
|
+
urls_file: str,
|
|
370
|
+
langs: str = typer.Option("en", "--langs", "-l", help="Language codes"),
|
|
371
|
+
fmt: str = typer.Option("srt", "--fmt", "-f", help="Output format"),
|
|
372
|
+
auto: bool = typer.Option(True, "--auto/--no-auto", help="Include auto-generated"),
|
|
373
|
+
output_dir: str = typer.Option(".", "--output-dir", "-o", help="Output directory"),
|
|
374
|
+
sanitize: str = typer.Option(
|
|
375
|
+
"safe", "--sanitize", help="Filename sanitization (safe/nospaces/slugify)"
|
|
376
|
+
),
|
|
377
|
+
quiet: bool = typer.Option(False, "--quiet", "-q", help="Errors only"),
|
|
378
|
+
verbose: bool = typer.Option(False, "--verbose", "-v", help="Debug output"),
|
|
379
|
+
):
|
|
380
|
+
"""Download subtitles for multiple URLs from a file.
|
|
381
|
+
|
|
382
|
+
URL file format (yt-dlp standard):
|
|
383
|
+
One URL per line
|
|
384
|
+
Lines starting with # are comments
|
|
385
|
+
Empty lines are ignored
|
|
386
|
+
|
|
387
|
+
Example:
|
|
388
|
+
uv run subxx batch urls.txt --langs en,de
|
|
389
|
+
"""
|
|
390
|
+
# Load config
|
|
391
|
+
config = load_config()
|
|
392
|
+
log_file = config.get("logging", {}).get("log_file")
|
|
393
|
+
|
|
394
|
+
# Determine verbosity level
|
|
395
|
+
if quiet:
|
|
396
|
+
verbosity = "quiet"
|
|
397
|
+
elif verbose:
|
|
398
|
+
verbosity = "verbose"
|
|
399
|
+
else:
|
|
400
|
+
verbosity = "normal"
|
|
401
|
+
|
|
402
|
+
# Setup logging
|
|
403
|
+
logger = setup_logging(verbose=verbose, quiet=quiet, log_file=log_file)
|
|
404
|
+
|
|
405
|
+
# Read URLs from file (yt-dlp format)
|
|
406
|
+
urls_path = Path(urls_file).expanduser()
|
|
407
|
+
if not urls_path.exists():
|
|
408
|
+
logger.error(f"File not found: {urls_file}")
|
|
409
|
+
raise typer.Exit(code=EXIT_FILE_ERROR)
|
|
410
|
+
|
|
411
|
+
urls = []
|
|
412
|
+
with open(urls_path) as f:
|
|
413
|
+
for line in f:
|
|
414
|
+
line = line.strip()
|
|
415
|
+
# Skip comments and empty lines (yt-dlp standard)
|
|
416
|
+
if line and not line.startswith("#"):
|
|
417
|
+
urls.append(line)
|
|
418
|
+
|
|
419
|
+
if not urls:
|
|
420
|
+
logger.error("No URLs found in file")
|
|
421
|
+
raise typer.Exit(code=EXIT_INVALID_URL)
|
|
422
|
+
|
|
423
|
+
logger.info(f"Processing {len(urls)} URLs from {urls_file}")
|
|
424
|
+
|
|
425
|
+
# Process each URL
|
|
426
|
+
failed = []
|
|
427
|
+
for i, url in enumerate(urls, 1):
|
|
428
|
+
logger.info(f"[{i}/{len(urls)}] {url}")
|
|
429
|
+
|
|
430
|
+
exit_code = fetch_subs(
|
|
431
|
+
url=url,
|
|
432
|
+
langs=langs,
|
|
433
|
+
fmt=fmt,
|
|
434
|
+
auto=auto,
|
|
435
|
+
output_dir=output_dir,
|
|
436
|
+
prompt_overwrite=False, # No prompts in batch mode
|
|
437
|
+
skip_existing=True, # Skip existing by default
|
|
438
|
+
verbosity=verbosity,
|
|
439
|
+
sanitize=sanitize,
|
|
440
|
+
)
|
|
441
|
+
|
|
442
|
+
if exit_code != 0:
|
|
443
|
+
failed.append(url)
|
|
444
|
+
|
|
445
|
+
# Summary
|
|
446
|
+
if failed:
|
|
447
|
+
logger.error(f"ā Failed to download {len(failed)}/{len(urls)} URLs")
|
|
448
|
+
for url in failed:
|
|
449
|
+
logger.error(f" - {url}")
|
|
450
|
+
raise typer.Exit(code=EXIT_NETWORK_ERROR)
|
|
451
|
+
else:
|
|
452
|
+
logger.info(f"ā
Successfully downloaded {len(urls)} subtitle sets")
|
|
453
|
+
raise typer.Exit(code=EXIT_SUCCESS)
|
|
454
|
+
|
|
455
|
+
|
|
456
|
+
@app.command()
|
|
457
|
+
def extract(
|
|
458
|
+
subtitle_file: str = typer.Argument(..., help="Subtitle file (.srt or .vtt)"),
|
|
459
|
+
output_format: str = typer.Option(
|
|
460
|
+
"txt", "--format", "-f", help="Output format (txt, md, pdf)"
|
|
461
|
+
),
|
|
462
|
+
timestamp_interval: Optional[int] = typer.Option(
|
|
463
|
+
None,
|
|
464
|
+
"--timestamps",
|
|
465
|
+
"-t",
|
|
466
|
+
help="Add timestamp every N seconds (e.g., 300 for 5min)",
|
|
467
|
+
),
|
|
468
|
+
chapters: bool = typer.Option(
|
|
469
|
+
False, "--chapters", help="Use chapter markers from metadata (YouTube chapters)"
|
|
470
|
+
),
|
|
471
|
+
auto_structure: bool = typer.Option(
|
|
472
|
+
False,
|
|
473
|
+
"--auto-structure",
|
|
474
|
+
help="Auto-detect best structure (YouTube chapters ā virtual chapters ā plain)",
|
|
475
|
+
),
|
|
476
|
+
fallback_timestamps: Optional[int] = typer.Option(
|
|
477
|
+
None,
|
|
478
|
+
"--fallback-timestamps",
|
|
479
|
+
help="Fallback to virtual chapters with this interval if YouTube chapters unavailable/insufficient",
|
|
480
|
+
),
|
|
481
|
+
min_chapters: Optional[int] = typer.Option(
|
|
482
|
+
None,
|
|
483
|
+
"--min-chapters",
|
|
484
|
+
help="Minimum required YouTube chapters (triggers fallback if fewer)",
|
|
485
|
+
),
|
|
486
|
+
output_file: Optional[str] = typer.Option(
|
|
487
|
+
None, "--output", "-o", help="Output file path"
|
|
488
|
+
),
|
|
489
|
+
force: bool = typer.Option(False, "--force", help="Overwrite existing output"),
|
|
490
|
+
quiet: bool = typer.Option(False, "--quiet", "-q", help="Suppress output"),
|
|
491
|
+
verbose: bool = typer.Option(False, "--verbose", "-v", help="Verbose output"),
|
|
492
|
+
):
|
|
493
|
+
"""Extract text from subtitle files.
|
|
494
|
+
|
|
495
|
+
Removes timestamps and formatting to create readable text documents.
|
|
496
|
+
|
|
497
|
+
Examples:
|
|
498
|
+
Basic: uv run subxx extract video.srt
|
|
499
|
+
Markdown: uv run subxx extract video.srt -f md
|
|
500
|
+
Auto-structure: uv run subxx extract video.srt --auto-structure -f md
|
|
501
|
+
With timestamps: uv run subxx extract video.srt -t 300
|
|
502
|
+
With chapters: uv run subxx extract video.srt --chapters -f md
|
|
503
|
+
With fallback: uv run subxx extract video.srt --chapters --fallback-timestamps 300 -f md
|
|
504
|
+
Minimum chapters: uv run subxx extract video.srt --chapters --min-chapters 5 --fallback-timestamps 300 -f md
|
|
505
|
+
PDF output: uv run subxx extract video.srt -f pdf
|
|
506
|
+
"""
|
|
507
|
+
|
|
508
|
+
# Check if extract dependencies are installed
|
|
509
|
+
try:
|
|
510
|
+
import srt # noqa: F401
|
|
511
|
+
from fpdf import FPDF # noqa: F401
|
|
512
|
+
except ImportError:
|
|
513
|
+
typer.echo("ā Error: Missing dependencies for text extraction")
|
|
514
|
+
typer.echo("Install with: uv sync --extra extract")
|
|
515
|
+
typer.echo("Or run with: uv run --extra extract subxx extract <file>")
|
|
516
|
+
raise typer.Exit(code=EXIT_CONFIG_ERROR)
|
|
517
|
+
|
|
518
|
+
# Load config
|
|
519
|
+
config = load_config()
|
|
520
|
+
log_file = config.get("logging", {}).get("log_file")
|
|
521
|
+
|
|
522
|
+
# Setup logging
|
|
523
|
+
logger = setup_logging(verbose=verbose, quiet=quiet, log_file=log_file)
|
|
524
|
+
|
|
525
|
+
# Apply auto-structure defaults
|
|
526
|
+
if auto_structure:
|
|
527
|
+
chapters = True
|
|
528
|
+
if fallback_timestamps is None:
|
|
529
|
+
fallback_timestamps = 300 # Default: 5-minute virtual chapters
|
|
530
|
+
logger.info(
|
|
531
|
+
"Auto-structure mode: will try YouTube chapters ā virtual chapters (5min) ā plain"
|
|
532
|
+
)
|
|
533
|
+
|
|
534
|
+
# Import and call extract function
|
|
535
|
+
from subxx import extract_text
|
|
536
|
+
|
|
537
|
+
exit_code = extract_text(
|
|
538
|
+
subtitle_file=subtitle_file,
|
|
539
|
+
output_format=output_format,
|
|
540
|
+
timestamp_interval=timestamp_interval,
|
|
541
|
+
output_file=output_file,
|
|
542
|
+
force=force,
|
|
543
|
+
use_chapters=chapters,
|
|
544
|
+
fallback_timestamps=fallback_timestamps,
|
|
545
|
+
min_chapters=min_chapters,
|
|
546
|
+
)
|
|
547
|
+
|
|
548
|
+
raise typer.Exit(code=exit_code)
|
|
549
|
+
|
|
550
|
+
|
|
551
|
+
@app.command()
|
|
552
|
+
def serve(
|
|
553
|
+
host: str = typer.Option(
|
|
554
|
+
"127.0.0.1", "--host", help="Bind address (ALWAYS use 127.0.0.1)"
|
|
555
|
+
),
|
|
556
|
+
port: int = typer.Option(8000, "--port", help="Port to listen on"),
|
|
557
|
+
):
|
|
558
|
+
"""Start HTTP API server (requires fastapi, uvicorn).
|
|
559
|
+
|
|
560
|
+
ā ļø WARNING: API has NO authentication. ONLY run on localhost!
|
|
561
|
+
|
|
562
|
+
Example:
|
|
563
|
+
uv run --extra api subxx serve --host 127.0.0.1 --port 8000
|
|
564
|
+
"""
|
|
565
|
+
try:
|
|
566
|
+
import uvicorn
|
|
567
|
+
from fastapi import FastAPI, HTTPException
|
|
568
|
+
from fastapi.responses import PlainTextResponse
|
|
569
|
+
from pydantic import BaseModel
|
|
570
|
+
import anyio
|
|
571
|
+
import tempfile
|
|
572
|
+
from pathlib import Path
|
|
573
|
+
except ImportError:
|
|
574
|
+
typer.echo("ā Error: API dependencies not installed")
|
|
575
|
+
typer.echo("Install with: uv sync --extra api")
|
|
576
|
+
typer.echo("Or run with: uv run --extra api subxx serve")
|
|
577
|
+
raise typer.Exit(code=EXIT_CONFIG_ERROR)
|
|
578
|
+
|
|
579
|
+
# Security check
|
|
580
|
+
if host != "127.0.0.1" and host != "localhost":
|
|
581
|
+
typer.echo(f"ā ļø WARNING: Binding to {host} exposes API to network!")
|
|
582
|
+
typer.echo("The API has NO authentication and should ONLY run on localhost.")
|
|
583
|
+
if not typer.confirm("Continue anyway?", default=False):
|
|
584
|
+
raise typer.Exit(code=EXIT_USER_CANCELLED)
|
|
585
|
+
|
|
586
|
+
class SubsRequest(BaseModel):
|
|
587
|
+
url: str
|
|
588
|
+
langs: str = "en"
|
|
589
|
+
fmt: str = "srt"
|
|
590
|
+
auto: bool = True
|
|
591
|
+
sanitize: str = "safe"
|
|
592
|
+
|
|
593
|
+
api = FastAPI(
|
|
594
|
+
title="subxx API", description="Subtitle fetching HTTP API", version="0.1.0"
|
|
595
|
+
)
|
|
596
|
+
|
|
597
|
+
@api.post("/subs", response_class=PlainTextResponse)
|
|
598
|
+
async def fetch_subs_endpoint(req: SubsRequest):
|
|
599
|
+
"""Fetch subtitles and return content directly."""
|
|
600
|
+
with tempfile.TemporaryDirectory() as tmpdir:
|
|
601
|
+
# Download to temp directory
|
|
602
|
+
exit_code = await anyio.to_thread.run_sync(
|
|
603
|
+
fetch_subs,
|
|
604
|
+
url=req.url,
|
|
605
|
+
langs=req.langs,
|
|
606
|
+
fmt=req.fmt,
|
|
607
|
+
auto=req.auto,
|
|
608
|
+
output_dir=tmpdir,
|
|
609
|
+
out_template="%(title)s.%(id)s.%(lang)s.%(ext)s",
|
|
610
|
+
prompt_overwrite=False,
|
|
611
|
+
skip_existing=False,
|
|
612
|
+
dry_run=False,
|
|
613
|
+
verbosity="quiet",
|
|
614
|
+
sanitize=req.sanitize,
|
|
615
|
+
)
|
|
616
|
+
|
|
617
|
+
if exit_code != 0:
|
|
618
|
+
raise HTTPException(status_code=500, detail="Subtitle fetch failed")
|
|
619
|
+
|
|
620
|
+
# Find downloaded file(s) and return content
|
|
621
|
+
subtitle_files = list(Path(tmpdir).glob(f"*.{req.fmt}"))
|
|
622
|
+
if not subtitle_files:
|
|
623
|
+
raise HTTPException(status_code=404, detail="No subtitles found")
|
|
624
|
+
|
|
625
|
+
# Return first subtitle
|
|
626
|
+
return subtitle_files[0].read_text(encoding="utf-8")
|
|
627
|
+
|
|
628
|
+
@api.get("/health")
|
|
629
|
+
async def health():
|
|
630
|
+
"""Health check endpoint."""
|
|
631
|
+
return {"status": "ok", "service": "subxx"}
|
|
632
|
+
|
|
633
|
+
typer.echo(f"š Starting subxx API server on http://{host}:{port}")
|
|
634
|
+
typer.echo(f"š API docs: http://{host}:{port}/docs")
|
|
635
|
+
typer.echo("ā ļø Security: NO authentication - localhost only!")
|
|
636
|
+
typer.echo("")
|
|
637
|
+
|
|
638
|
+
uvicorn.run(api, host=host, port=port)
|
|
639
|
+
|
|
640
|
+
|
|
641
|
+
if __name__ == "__main__":
|
|
642
|
+
app()
|