watch-skill 1.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- watch_skill/__init__.py +19 -0
- watch_skill/acquire/__init__.py +12 -0
- watch_skill/acquire/cache.py +150 -0
- watch_skill/acquire/cobalt.py +96 -0
- watch_skill/acquire/direct.py +50 -0
- watch_skill/acquire/resolver.py +176 -0
- watch_skill/acquire/sources.py +54 -0
- watch_skill/acquire/types.py +26 -0
- watch_skill/acquire/ytdlp.py +208 -0
- watch_skill/answer/__init__.py +11 -0
- watch_skill/answer/cache.py +172 -0
- watch_skill/answer/confidence.py +122 -0
- watch_skill/answer/crops.py +75 -0
- watch_skill/answer/engine.py +397 -0
- watch_skill/answer/ladder.py +182 -0
- watch_skill/answer/localize.py +218 -0
- watch_skill/answer/types.py +62 -0
- watch_skill/batch.py +158 -0
- watch_skill/bench/__init__.py +4 -0
- watch_skill/bench/perception.py +175 -0
- watch_skill/config.py +295 -0
- watch_skill/errors.py +98 -0
- watch_skill/extract/__init__.py +11 -0
- watch_skill/extract/bug_report.py +139 -0
- watch_skill/extract/chapters.py +123 -0
- watch_skill/extract/hook.py +152 -0
- watch_skill/health/__init__.py +6 -0
- watch_skill/health/agents_setup.py +151 -0
- watch_skill/health/binaries.py +277 -0
- watch_skill/health/clean.py +114 -0
- watch_skill/health/doctor.py +633 -0
- watch_skill/health/log.py +46 -0
- watch_skill/health/vision_setup.py +297 -0
- watch_skill/index/__init__.py +29 -0
- watch_skill/index/db.py +299 -0
- watch_skill/index/embeddings.py +74 -0
- watch_skill/index/retrieval.py +266 -0
- watch_skill/index/store.py +340 -0
- watch_skill/index/textnorm.py +208 -0
- watch_skill/integrations/__init__.py +13 -0
- watch_skill/integrations/_core.py +88 -0
- watch_skill/integrations/autogen.py +26 -0
- watch_skill/integrations/crewai.py +40 -0
- watch_skill/integrations/langchain.py +30 -0
- watch_skill/integrations/llamaindex.py +27 -0
- watch_skill/integrations/openai_agents.py +29 -0
- watch_skill/jobs.py +115 -0
- watch_skill/lessons/__init__.py +23 -0
- watch_skill/lessons/classify.py +105 -0
- watch_skill/lessons/evals.py +192 -0
- watch_skill/lessons/inject.py +81 -0
- watch_skill/lessons/profiles.py +94 -0
- watch_skill/lessons/report.py +96 -0
- watch_skill/lessons/store.py +184 -0
- watch_skill/library/__init__.py +23 -0
- watch_skill/library/notes.py +204 -0
- watch_skill/library/synthesize.py +349 -0
- watch_skill/loop/__init__.py +61 -0
- watch_skill/loop/artifact.py +82 -0
- watch_skill/loop/capture.py +226 -0
- watch_skill/loop/critic.py +331 -0
- watch_skill/loop/diff.py +125 -0
- watch_skill/loop/framework.py +215 -0
- watch_skill/loop/monitor.py +198 -0
- watch_skill/loop/reportfmt.py +42 -0
- watch_skill/loop/runner.py +293 -0
- watch_skill/loop/webhook.py +95 -0
- watch_skill/perceive/__init__.py +23 -0
- watch_skill/perceive/budget.py +97 -0
- watch_skill/perceive/engine.py +213 -0
- watch_skill/perceive/media.py +91 -0
- watch_skill/perceive/ocr.py +151 -0
- watch_skill/perceive/ocr_backends.py +315 -0
- watch_skill/perceive/scenes.py +87 -0
- watch_skill/perceive/types.py +88 -0
- watch_skill/report.py +108 -0
- watch_skill/surfaces/__init__.py +5 -0
- watch_skill/surfaces/api/__init__.py +9 -0
- watch_skill/surfaces/api/app.py +337 -0
- watch_skill/surfaces/cli/__init__.py +1 -0
- watch_skill/surfaces/cli/main.py +1035 -0
- watch_skill/surfaces/mcp/__init__.py +1 -0
- watch_skill/surfaces/mcp/server.py +619 -0
- watch_skill/transcribe/__init__.py +12 -0
- watch_skill/transcribe/audio.py +131 -0
- watch_skill/transcribe/cloud.py +190 -0
- watch_skill/transcribe/diarize.py +116 -0
- watch_skill/transcribe/ladder.py +71 -0
- watch_skill/transcribe/local.py +113 -0
- watch_skill/transcribe/types.py +66 -0
- watch_skill/transcribe/vtt.py +64 -0
- watch_skill/viewer.py +213 -0
- watch_skill/vision/__init__.py +22 -0
- watch_skill/vision/client.py +304 -0
- watch_skill/vision/cost.py +52 -0
- watch_skill/vision/local_health.py +119 -0
- watch_skill/vision/model.py +149 -0
- watch_skill/vision/prices.json +19 -0
- watch_skill/vision/registry.py +144 -0
- watch_skill/watch.py +177 -0
- watch_skill-1.2.0.dist-info/METADATA +410 -0
- watch_skill-1.2.0.dist-info/RECORD +105 -0
- watch_skill-1.2.0.dist-info/WHEEL +4 -0
- watch_skill-1.2.0.dist-info/entry_points.txt +2 -0
- watch_skill-1.2.0.dist-info/licenses/LICENSE +21 -0
watch_skill/__init__.py
ADDED
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
"""Watch Skill — give any agent a video input.
|
|
2
|
+
|
|
3
|
+
Core engine: acquisition, perception, transcription, indexing, vision, and
|
|
4
|
+
the autonomous watch-critique-iterate loop. Surfaces (MCP, CLI, REST) are
|
|
5
|
+
thin wrappers around this package and live outside it.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from importlib.metadata import PackageNotFoundError
|
|
9
|
+
from importlib.metadata import version as _pkg_version
|
|
10
|
+
|
|
11
|
+
# Read the version from installed metadata instead of restating it here.
|
|
12
|
+
# v1.0.0 shipped reporting "0.6.0" because the release bumped every manifest
|
|
13
|
+
# and missed this line; deriving it removes the chance of a second drift.
|
|
14
|
+
try:
|
|
15
|
+
__version__ = _pkg_version("watch-skill")
|
|
16
|
+
except PackageNotFoundError: # running from a source tree with no install
|
|
17
|
+
__version__ = "0.0.0+unknown"
|
|
18
|
+
|
|
19
|
+
__all__ = ["__version__"]
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
"""Source acquisition: resolve ANY input to a local video file.
|
|
2
|
+
|
|
3
|
+
Self-healing chain: yt-dlp -> auto-update + retry -> self-hosted cobalt
|
|
4
|
+
(opt-in) -> direct ffmpeg pull. Content-addressed cache with LRU eviction.
|
|
5
|
+
Privacy invariants: no cookies/logins, video files never leave the machine.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from watch_skill.acquire.resolver import acquire, fetch_captions_only
|
|
9
|
+
from watch_skill.acquire.sources import SourceKind, classify_source
|
|
10
|
+
from watch_skill.acquire.types import AcquireResult
|
|
11
|
+
|
|
12
|
+
__all__ = ["AcquireResult", "SourceKind", "acquire", "classify_source", "fetch_captions_only"]
|
|
@@ -0,0 +1,150 @@
|
|
|
1
|
+
"""Content-addressed download cache: never re-download an unchanged video.
|
|
2
|
+
|
|
3
|
+
Layout: ``<data_dir>/cache/<key>/`` where ``key`` is a hash of the normalized
|
|
4
|
+
source URL. Each entry holds the media file(s), optional subtitles, and an
|
|
5
|
+
``entry.json`` with source metadata. Access time is tracked by touching
|
|
6
|
+
``entry.json`` so LRU eviction can order entries without OS atime support.
|
|
7
|
+
"""
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import hashlib
|
|
11
|
+
import json
|
|
12
|
+
import os
|
|
13
|
+
import shutil
|
|
14
|
+
import time
|
|
15
|
+
from dataclasses import dataclass
|
|
16
|
+
from pathlib import Path
|
|
17
|
+
from typing import Any
|
|
18
|
+
|
|
19
|
+
from watch_skill.config import get_settings
|
|
20
|
+
|
|
21
|
+
ENTRY_FILE = "entry.json"
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
@dataclass
|
|
25
|
+
class CacheEntry:
|
|
26
|
+
"""One cached download."""
|
|
27
|
+
|
|
28
|
+
key: str
|
|
29
|
+
dir: Path
|
|
30
|
+
video_path: Path | None
|
|
31
|
+
subtitle_path: Path | None
|
|
32
|
+
info: dict[str, Any]
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def cache_key(source: str) -> str:
|
|
36
|
+
"""Stable key for a source URL (case-preserved, whitespace-trimmed)."""
|
|
37
|
+
return hashlib.sha256(source.strip().encode("utf-8")).hexdigest()[:24]
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def entry_dir(source: str, create: bool = False) -> Path:
|
|
41
|
+
"""Directory where this source's download lives (created on demand)."""
|
|
42
|
+
directory = get_settings().cache_dir / cache_key(source)
|
|
43
|
+
if create:
|
|
44
|
+
directory.mkdir(parents=True, exist_ok=True)
|
|
45
|
+
return directory
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def _load_entry(directory: Path) -> CacheEntry | None:
|
|
49
|
+
meta_path = directory / ENTRY_FILE
|
|
50
|
+
if not meta_path.is_file():
|
|
51
|
+
return None
|
|
52
|
+
try:
|
|
53
|
+
meta = json.loads(meta_path.read_text(encoding="utf-8"))
|
|
54
|
+
except (OSError, json.JSONDecodeError):
|
|
55
|
+
return None
|
|
56
|
+
video = directory / meta["video"] if meta.get("video") else None
|
|
57
|
+
subs = directory / meta["subtitle"] if meta.get("subtitle") else None
|
|
58
|
+
if video is not None and not video.is_file():
|
|
59
|
+
return None
|
|
60
|
+
return CacheEntry(
|
|
61
|
+
key=directory.name,
|
|
62
|
+
dir=directory,
|
|
63
|
+
video_path=video,
|
|
64
|
+
subtitle_path=subs if subs and subs.is_file() else None,
|
|
65
|
+
info=meta.get("info", {}),
|
|
66
|
+
)
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def lookup(source: str) -> CacheEntry | None:
|
|
70
|
+
"""Return the cached entry for ``source`` and mark it recently used."""
|
|
71
|
+
directory = entry_dir(source)
|
|
72
|
+
entry = _load_entry(directory)
|
|
73
|
+
if entry is not None:
|
|
74
|
+
try:
|
|
75
|
+
os.utime(directory / ENTRY_FILE)
|
|
76
|
+
except OSError:
|
|
77
|
+
pass
|
|
78
|
+
return entry
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def commit(
|
|
82
|
+
source: str,
|
|
83
|
+
video_path: Path | None,
|
|
84
|
+
subtitle_path: Path | None = None,
|
|
85
|
+
info: dict[str, Any] | None = None,
|
|
86
|
+
) -> CacheEntry:
|
|
87
|
+
"""Record a completed download that already lives inside the entry dir.
|
|
88
|
+
|
|
89
|
+
Acquirers download *into* :func:`entry_dir` so committing is just writing
|
|
90
|
+
the manifest — no copy of a multi-GB file.
|
|
91
|
+
"""
|
|
92
|
+
directory = entry_dir(source, create=True)
|
|
93
|
+
meta = {
|
|
94
|
+
"source": source,
|
|
95
|
+
"video": video_path.name if video_path else None,
|
|
96
|
+
"subtitle": subtitle_path.name if subtitle_path else None,
|
|
97
|
+
"info": info or {},
|
|
98
|
+
"committed_at": time.time(),
|
|
99
|
+
}
|
|
100
|
+
(directory / ENTRY_FILE).write_text(
|
|
101
|
+
json.dumps(meta, ensure_ascii=False, indent=2), encoding="utf-8"
|
|
102
|
+
)
|
|
103
|
+
evict_lru()
|
|
104
|
+
return CacheEntry(
|
|
105
|
+
key=directory.name,
|
|
106
|
+
dir=directory,
|
|
107
|
+
video_path=video_path,
|
|
108
|
+
subtitle_path=subtitle_path,
|
|
109
|
+
info=info or {},
|
|
110
|
+
)
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
def _dir_size(directory: Path) -> int:
|
|
114
|
+
return sum(p.stat().st_size for p in directory.rglob("*") if p.is_file())
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
def evict_lru(max_bytes: int | None = None) -> int:
|
|
118
|
+
"""Delete least-recently-used entries until the cache fits the cap.
|
|
119
|
+
|
|
120
|
+
Returns the number of evicted entries. Recency = mtime of ``entry.json``
|
|
121
|
+
(touched on every :func:`lookup` hit).
|
|
122
|
+
"""
|
|
123
|
+
cap = max_bytes if max_bytes is not None else get_settings().cache_max_bytes
|
|
124
|
+
cache_dir = get_settings().cache_dir
|
|
125
|
+
if not cache_dir.is_dir():
|
|
126
|
+
return 0
|
|
127
|
+
|
|
128
|
+
entries: list[tuple[float, Path, int]] = []
|
|
129
|
+
total = 0
|
|
130
|
+
for directory in cache_dir.iterdir():
|
|
131
|
+
meta = directory / ENTRY_FILE
|
|
132
|
+
if not directory.is_dir() or not meta.is_file():
|
|
133
|
+
continue
|
|
134
|
+
size = _dir_size(directory)
|
|
135
|
+
total += size
|
|
136
|
+
entries.append((meta.stat().st_mtime, directory, size))
|
|
137
|
+
|
|
138
|
+
evicted = 0
|
|
139
|
+
entries.sort() # oldest first
|
|
140
|
+
while total > cap and entries:
|
|
141
|
+
_, directory, size = entries.pop(0)
|
|
142
|
+
shutil.rmtree(directory, ignore_errors=True)
|
|
143
|
+
total -= size
|
|
144
|
+
evicted += 1
|
|
145
|
+
return evicted
|
|
146
|
+
|
|
147
|
+
|
|
148
|
+
def clear() -> None:
|
|
149
|
+
"""Remove the entire cache (used by `watch-skill cache clear`)."""
|
|
150
|
+
shutil.rmtree(get_settings().cache_dir, ignore_errors=True)
|
|
@@ -0,0 +1,96 @@
|
|
|
1
|
+
"""cobalt fallback acquirer (self-hosted instances only).
|
|
2
|
+
|
|
3
|
+
Used only after yt-dlp (and its self-update retry) failed, and only when the
|
|
4
|
+
user configured ``WATCHSKILL_COBALT_API_URL``. The public api.cobalt.tools
|
|
5
|
+
now requires JWT auth (verified 2026-07-05: anonymous POST returns
|
|
6
|
+
``error.api.auth.jwt.missing``), so without a configured self-hosted
|
|
7
|
+
instance the chain skips straight to the direct-URL fallback instead of
|
|
8
|
+
burning a doomed network round-trip.
|
|
9
|
+
"""
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import os
|
|
13
|
+
from pathlib import Path
|
|
14
|
+
from typing import Any
|
|
15
|
+
|
|
16
|
+
import httpx
|
|
17
|
+
|
|
18
|
+
from watch_skill.errors import AcquisitionError
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def _api_url() -> str | None:
|
|
22
|
+
return os.environ.get("WATCHSKILL_COBALT_API_URL") or None
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def is_configured() -> bool:
|
|
26
|
+
"""cobalt participates in the fallback chain only when an instance is set."""
|
|
27
|
+
return _api_url() is not None
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def _request_media_url(url: str, timeout: float = 60.0) -> str:
|
|
31
|
+
"""Ask the cobalt instance to resolve ``url`` to a downloadable media URL."""
|
|
32
|
+
api_url = _api_url()
|
|
33
|
+
if api_url is None:
|
|
34
|
+
raise AcquisitionError(
|
|
35
|
+
"no cobalt instance configured",
|
|
36
|
+
code="acquire.cobalt_not_configured",
|
|
37
|
+
fix="set WATCHSKILL_COBALT_API_URL to a self-hosted cobalt instance "
|
|
38
|
+
"(the public API requires auth); the chain continues without it",
|
|
39
|
+
details={"url": url},
|
|
40
|
+
)
|
|
41
|
+
try:
|
|
42
|
+
response = httpx.post(
|
|
43
|
+
api_url,
|
|
44
|
+
json={"url": url, "videoQuality": "720", "filenameStyle": "basic"},
|
|
45
|
+
headers={"Accept": "application/json", "Content-Type": "application/json"},
|
|
46
|
+
timeout=timeout,
|
|
47
|
+
follow_redirects=True,
|
|
48
|
+
)
|
|
49
|
+
payload: dict[str, Any] = response.json()
|
|
50
|
+
except (httpx.HTTPError, ValueError) as exc:
|
|
51
|
+
raise AcquisitionError(
|
|
52
|
+
f"cobalt API unreachable or returned non-JSON: {exc}",
|
|
53
|
+
code="acquire.cobalt_unreachable",
|
|
54
|
+
fix="set WATCHSKILL_COBALT_API_URL to a working cobalt instance, or skip",
|
|
55
|
+
details={"url": url, "api": _api_url()},
|
|
56
|
+
) from exc
|
|
57
|
+
|
|
58
|
+
status = payload.get("status")
|
|
59
|
+
if status in ("tunnel", "redirect", "stream") and payload.get("url"):
|
|
60
|
+
return str(payload["url"])
|
|
61
|
+
raise AcquisitionError(
|
|
62
|
+
f"cobalt could not resolve this URL (status={status!r})",
|
|
63
|
+
code="acquire.cobalt_failed",
|
|
64
|
+
fix="the resolver will continue down the fallback chain",
|
|
65
|
+
details={"url": url, "response": {k: payload.get(k) for k in ("status", "error")}},
|
|
66
|
+
)
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def download(url: str, dest: Path, timeout: float = 1800.0) -> Path:
|
|
70
|
+
"""Resolve via cobalt and stream the media to ``dest``."""
|
|
71
|
+
media_url = _request_media_url(url)
|
|
72
|
+
dest.parent.mkdir(parents=True, exist_ok=True)
|
|
73
|
+
tmp = dest.with_suffix(dest.suffix + ".part")
|
|
74
|
+
try:
|
|
75
|
+
with httpx.stream("GET", media_url, follow_redirects=True, timeout=timeout) as resp:
|
|
76
|
+
resp.raise_for_status()
|
|
77
|
+
with tmp.open("wb") as fh:
|
|
78
|
+
for chunk in resp.iter_bytes(1024 * 256):
|
|
79
|
+
fh.write(chunk)
|
|
80
|
+
tmp.replace(dest)
|
|
81
|
+
except httpx.HTTPError as exc:
|
|
82
|
+
tmp.unlink(missing_ok=True)
|
|
83
|
+
raise AcquisitionError(
|
|
84
|
+
f"cobalt media download failed: {exc}",
|
|
85
|
+
code="acquire.cobalt_download_failed",
|
|
86
|
+
fix="the resolver will continue down the fallback chain",
|
|
87
|
+
details={"url": url},
|
|
88
|
+
) from exc
|
|
89
|
+
if not dest.is_file() or dest.stat().st_size == 0:
|
|
90
|
+
raise AcquisitionError(
|
|
91
|
+
"cobalt returned an empty file",
|
|
92
|
+
code="acquire.cobalt_empty",
|
|
93
|
+
fix="the resolver continues down the fallback chain on its own",
|
|
94
|
+
details={"url": url},
|
|
95
|
+
)
|
|
96
|
+
return dest
|
|
@@ -0,0 +1,50 @@
|
|
|
1
|
+
"""Direct ffmpeg pull: media URLs and HLS/DASH manifests, incl. bounded live capture."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
import subprocess
|
|
5
|
+
import sys
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
|
|
8
|
+
from watch_skill.errors import AcquisitionError
|
|
9
|
+
from watch_skill.health.binaries import require_binary
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def ffmpeg_pull(
|
|
13
|
+
url: str,
|
|
14
|
+
dest: Path,
|
|
15
|
+
duration_seconds: float | None = None,
|
|
16
|
+
timeout: float = 3600.0,
|
|
17
|
+
) -> Path:
|
|
18
|
+
"""Fetch a direct media URL or HLS/DASH manifest into a local mp4.
|
|
19
|
+
|
|
20
|
+
``duration_seconds`` bounds the capture — required for live streams,
|
|
21
|
+
which would otherwise download forever. Stream-copies when possible and
|
|
22
|
+
falls back to transcoding only if the container rejects the codecs.
|
|
23
|
+
"""
|
|
24
|
+
ffmpeg = require_binary("ffmpeg")
|
|
25
|
+
dest.parent.mkdir(parents=True, exist_ok=True)
|
|
26
|
+
|
|
27
|
+
base: list[str] = [str(ffmpeg), "-hide_banner", "-loglevel", "error", "-y", "-i", url]
|
|
28
|
+
if duration_seconds is not None:
|
|
29
|
+
base += ["-t", f"{duration_seconds:.3f}"]
|
|
30
|
+
|
|
31
|
+
copy_cmd = base + ["-c", "copy", "-movflags", "+faststart", str(dest)]
|
|
32
|
+
result = subprocess.run(
|
|
33
|
+
copy_cmd, capture_output=True, text=True, timeout=timeout,
|
|
34
|
+
encoding="utf-8", errors="replace",
|
|
35
|
+
)
|
|
36
|
+
if result.returncode != 0 or not dest.is_file() or dest.stat().st_size == 0:
|
|
37
|
+
print("[watch-skill] stream copy failed — transcoding…", file=sys.stderr)
|
|
38
|
+
transcode_cmd = base + ["-c:v", "libx264", "-preset", "veryfast", "-c:a", "aac", str(dest)]
|
|
39
|
+
result = subprocess.run(
|
|
40
|
+
transcode_cmd, capture_output=True, text=True, timeout=timeout,
|
|
41
|
+
encoding="utf-8", errors="replace",
|
|
42
|
+
)
|
|
43
|
+
if result.returncode != 0 or not dest.is_file() or dest.stat().st_size == 0:
|
|
44
|
+
raise AcquisitionError(
|
|
45
|
+
"ffmpeg could not fetch the media URL",
|
|
46
|
+
code="acquire.ffmpeg_pull_failed",
|
|
47
|
+
fix="check the URL is reachable and actually points at media/a manifest",
|
|
48
|
+
details={"url": url, "stderr_tail": result.stderr[-2000:]},
|
|
49
|
+
)
|
|
50
|
+
return dest
|
|
@@ -0,0 +1,176 @@
|
|
|
1
|
+
"""Resolve ANY input to a local video file, with cache and fallback chain.
|
|
2
|
+
|
|
3
|
+
Chain for network sources (each hop logs why the previous one failed):
|
|
4
|
+
yt-dlp -> (auto-update + retry, inside ytdlp.download) -> self-hosted cobalt
|
|
5
|
+
(only when WATCHSKILL_COBALT_API_URL is set — the public instance requires
|
|
6
|
+
auth) -> direct ffmpeg pull. Direct/manifest URLs try yt-dlp first too (it handles
|
|
7
|
+
both), with ffmpeg as the reliable second step. Screen/window capture is
|
|
8
|
+
provided by ``watch_skill.loop.capture`` (Milestone 3) — the resolver returns
|
|
9
|
+
a structured error pointing there until then.
|
|
10
|
+
"""
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
import sys
|
|
14
|
+
from pathlib import Path
|
|
15
|
+
|
|
16
|
+
from watch_skill.acquire import cache, cobalt, direct, ytdlp
|
|
17
|
+
from watch_skill.acquire.sources import SourceKind, classify_source, is_url_kind
|
|
18
|
+
from watch_skill.acquire.types import AcquireResult
|
|
19
|
+
from watch_skill.errors import AcquisitionError
|
|
20
|
+
from watch_skill.health.log import record_incident
|
|
21
|
+
|
|
22
|
+
VIDEO_EXTS = ytdlp.VIDEO_EXTS
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def _resolve_local(source: str) -> AcquireResult:
|
|
26
|
+
path = Path(source).expanduser().resolve()
|
|
27
|
+
if path.is_dir():
|
|
28
|
+
# "file not found" is wrong and unhelpful here: the path exists, and
|
|
29
|
+
# a folder of recordings is what `batch` is for.
|
|
30
|
+
raise AcquisitionError(
|
|
31
|
+
f"{path} is a directory, not a video file",
|
|
32
|
+
code="acquire.is_a_directory",
|
|
33
|
+
fix=f'watch every video in it with: watch-skill batch "{path}"',
|
|
34
|
+
details={"source": source},
|
|
35
|
+
)
|
|
36
|
+
if not path.is_file():
|
|
37
|
+
raise AcquisitionError(
|
|
38
|
+
f"file not found: {path}",
|
|
39
|
+
code="acquire.file_not_found",
|
|
40
|
+
fix="check the path; remember to quote paths containing spaces",
|
|
41
|
+
details={"source": source},
|
|
42
|
+
)
|
|
43
|
+
if path.suffix.lower() not in VIDEO_EXTS:
|
|
44
|
+
print(
|
|
45
|
+
f"[watch-skill] warning: {path.suffix} is not a known video extension — proceeding",
|
|
46
|
+
file=sys.stderr,
|
|
47
|
+
)
|
|
48
|
+
return AcquireResult(
|
|
49
|
+
source=source,
|
|
50
|
+
kind=SourceKind.LOCAL_FILE,
|
|
51
|
+
video_path=path,
|
|
52
|
+
info={"title": path.name, "url": str(path)},
|
|
53
|
+
acquirer="local",
|
|
54
|
+
)
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def _from_cache(source: str, kind: SourceKind) -> AcquireResult | None:
|
|
58
|
+
entry = cache.lookup(source)
|
|
59
|
+
if entry is None or entry.video_path is None:
|
|
60
|
+
return None
|
|
61
|
+
print(f"[watch-skill] cache hit: {entry.dir}", file=sys.stderr)
|
|
62
|
+
return AcquireResult(
|
|
63
|
+
source=source,
|
|
64
|
+
kind=kind,
|
|
65
|
+
video_path=entry.video_path,
|
|
66
|
+
subtitle_path=entry.subtitle_path,
|
|
67
|
+
info=entry.info,
|
|
68
|
+
from_cache=True,
|
|
69
|
+
acquirer="cache",
|
|
70
|
+
)
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def _try_chain(
|
|
74
|
+
source: str, kind: SourceKind, out_dir: Path,
|
|
75
|
+
audio_only: bool, duration_cap: float | None,
|
|
76
|
+
) -> AcquireResult:
|
|
77
|
+
"""Walk the fallback chain, recording each hop's failure."""
|
|
78
|
+
failures: list[str] = []
|
|
79
|
+
|
|
80
|
+
try:
|
|
81
|
+
dl = ytdlp.download(source, out_dir, audio_only=audio_only)
|
|
82
|
+
return AcquireResult(
|
|
83
|
+
source=source, kind=kind, video_path=Path(dl["video_path"]),
|
|
84
|
+
subtitle_path=dl["subtitle_path"], info=dl["info"], acquirer="yt-dlp",
|
|
85
|
+
)
|
|
86
|
+
except AcquisitionError as exc:
|
|
87
|
+
failures.append(f"yt-dlp: {exc.message}")
|
|
88
|
+
record_incident("acquire_fallback", "yt-dlp failed, trying fallbacks", url=source)
|
|
89
|
+
print(f"[watch-skill] yt-dlp failed ({exc.code}) — trying fallbacks…", file=sys.stderr)
|
|
90
|
+
|
|
91
|
+
if kind == SourceKind.PAGE_URL and cobalt.is_configured():
|
|
92
|
+
try:
|
|
93
|
+
video = cobalt.download(source, out_dir / "media.mp4")
|
|
94
|
+
return AcquireResult(
|
|
95
|
+
source=source, kind=kind, video_path=video, acquirer="cobalt",
|
|
96
|
+
info={"url": source},
|
|
97
|
+
)
|
|
98
|
+
except AcquisitionError as exc:
|
|
99
|
+
failures.append(f"cobalt: {exc.message}")
|
|
100
|
+
record_incident("acquire_fallback", "cobalt failed, trying ffmpeg", url=source)
|
|
101
|
+
print(f"[watch-skill] cobalt failed ({exc.code}) — trying direct ffmpeg…", file=sys.stderr)
|
|
102
|
+
|
|
103
|
+
try:
|
|
104
|
+
video = direct.ffmpeg_pull(source, out_dir / "media.mp4", duration_seconds=duration_cap)
|
|
105
|
+
return AcquireResult(
|
|
106
|
+
source=source, kind=kind, video_path=video, acquirer="ffmpeg",
|
|
107
|
+
info={"url": source},
|
|
108
|
+
)
|
|
109
|
+
except AcquisitionError as exc:
|
|
110
|
+
failures.append(f"ffmpeg: {exc.message}")
|
|
111
|
+
|
|
112
|
+
raise AcquisitionError(
|
|
113
|
+
"every acquirer in the fallback chain failed",
|
|
114
|
+
code="acquire.chain_exhausted",
|
|
115
|
+
fix="check the URL; if it needs login or is region-locked, Watch Skill "
|
|
116
|
+
"will not bypass that (privacy invariant: no cookies, no logins)",
|
|
117
|
+
details={"source": source, "failures": failures},
|
|
118
|
+
)
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
def acquire(
|
|
122
|
+
source: str,
|
|
123
|
+
audio_only: bool = False,
|
|
124
|
+
duration_cap: float | None = None,
|
|
125
|
+
use_cache: bool = True,
|
|
126
|
+
) -> AcquireResult:
|
|
127
|
+
"""Resolve ``source`` (URL / manifest / local path / capture spec) to local files.
|
|
128
|
+
|
|
129
|
+
Network downloads land in the content-addressed cache and are reused on
|
|
130
|
+
the next call. ``duration_cap`` bounds live-stream capture.
|
|
131
|
+
"""
|
|
132
|
+
kind = classify_source(source)
|
|
133
|
+
|
|
134
|
+
if kind in (SourceKind.SCREEN, SourceKind.WINDOW):
|
|
135
|
+
raise AcquisitionError(
|
|
136
|
+
"screen/window capture is provided by the loop module",
|
|
137
|
+
code="acquire.capture_required",
|
|
138
|
+
fix="use `watch-skill capture` / the capture() API (Milestone 3)",
|
|
139
|
+
details={"source": source, "kind": kind.value},
|
|
140
|
+
)
|
|
141
|
+
|
|
142
|
+
if kind == SourceKind.LOCAL_FILE:
|
|
143
|
+
return _resolve_local(source)
|
|
144
|
+
|
|
145
|
+
assert is_url_kind(kind)
|
|
146
|
+
if use_cache:
|
|
147
|
+
cached = _from_cache(source, kind)
|
|
148
|
+
if cached is not None:
|
|
149
|
+
return cached
|
|
150
|
+
|
|
151
|
+
out_dir = cache.entry_dir(source, create=True)
|
|
152
|
+
result = _try_chain(source, kind, out_dir, audio_only, duration_cap)
|
|
153
|
+
# Audio-only fetches are not committed: a later full fetch must not be
|
|
154
|
+
# shadowed by a cache entry that has no video frames in it.
|
|
155
|
+
if use_cache and not audio_only:
|
|
156
|
+
cache.commit(source, result.video_path, result.subtitle_path, result.info)
|
|
157
|
+
return result
|
|
158
|
+
|
|
159
|
+
|
|
160
|
+
def fetch_captions_only(source: str) -> AcquireResult:
|
|
161
|
+
"""Probe captions + metadata for a URL without downloading media."""
|
|
162
|
+
kind = classify_source(source)
|
|
163
|
+
if not is_url_kind(kind):
|
|
164
|
+
raise AcquisitionError(
|
|
165
|
+
"captions probe only applies to URLs",
|
|
166
|
+
code="acquire.not_a_url",
|
|
167
|
+
fix="local files have no platform captions; transcription falls "
|
|
168
|
+
"back to local whisper automatically",
|
|
169
|
+
details={"source": source},
|
|
170
|
+
)
|
|
171
|
+
out_dir = cache.entry_dir(source, create=True)
|
|
172
|
+
probe = ytdlp.fetch_captions(source, out_dir)
|
|
173
|
+
return AcquireResult(
|
|
174
|
+
source=source, kind=kind, video_path=None,
|
|
175
|
+
subtitle_path=probe["subtitle_path"], info=probe["info"], acquirer="yt-dlp",
|
|
176
|
+
)
|
|
@@ -0,0 +1,54 @@
|
|
|
1
|
+
"""Classify any input string into a source kind the resolver can act on."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
from enum import Enum
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
from urllib.parse import urlparse
|
|
7
|
+
|
|
8
|
+
MEDIA_EXTS = {
|
|
9
|
+
".mp4", ".mkv", ".webm", ".mov", ".m4v", ".avi", ".flv", ".wmv",
|
|
10
|
+
".ts", ".m4a", ".mp3", ".opus", ".ogg", ".wav",
|
|
11
|
+
}
|
|
12
|
+
MANIFEST_EXTS = {".m3u8", ".mpd"}
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
class SourceKind(str, Enum): # noqa: UP042 — StrEnum needs 3.11+, str+Enum reads the same
|
|
16
|
+
"""What kind of input the user handed us."""
|
|
17
|
+
|
|
18
|
+
PAGE_URL = "page_url" # any website; yt-dlp's 1800+ extractors
|
|
19
|
+
DIRECT_URL = "direct_url" # URL pointing straight at a media file
|
|
20
|
+
MANIFEST_URL = "manifest_url" # HLS (.m3u8) / DASH (.mpd) manifest
|
|
21
|
+
LOCAL_FILE = "local_file"
|
|
22
|
+
SCREEN = "screen" # `screen:` — capture the screen (Milestone 3)
|
|
23
|
+
WINDOW = "window" # `window:<title>` — capture one window (Milestone 3)
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def classify_source(source: str) -> SourceKind:
|
|
27
|
+
"""Map a raw source string to a :class:`SourceKind`.
|
|
28
|
+
|
|
29
|
+
Rules: ``screen:`` / ``window:`` prefixes are capture requests; http(s)
|
|
30
|
+
URLs are split by path suffix into manifest / direct-media / page URLs;
|
|
31
|
+
everything else is treated as a local file path.
|
|
32
|
+
"""
|
|
33
|
+
stripped = source.strip()
|
|
34
|
+
lowered = stripped.lower()
|
|
35
|
+
if lowered == "screen:" or lowered.startswith("screen:"):
|
|
36
|
+
return SourceKind.SCREEN
|
|
37
|
+
if lowered.startswith("window:"):
|
|
38
|
+
return SourceKind.WINDOW
|
|
39
|
+
|
|
40
|
+
parsed = urlparse(stripped)
|
|
41
|
+
if parsed.scheme in ("http", "https") and parsed.netloc:
|
|
42
|
+
suffix = Path(parsed.path).suffix.lower()
|
|
43
|
+
if suffix in MANIFEST_EXTS:
|
|
44
|
+
return SourceKind.MANIFEST_URL
|
|
45
|
+
if suffix in MEDIA_EXTS:
|
|
46
|
+
return SourceKind.DIRECT_URL
|
|
47
|
+
return SourceKind.PAGE_URL
|
|
48
|
+
|
|
49
|
+
return SourceKind.LOCAL_FILE
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def is_url_kind(kind: SourceKind) -> bool:
|
|
53
|
+
"""True for any source that must be fetched over the network."""
|
|
54
|
+
return kind in (SourceKind.PAGE_URL, SourceKind.DIRECT_URL, SourceKind.MANIFEST_URL)
|
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
"""Shared acquisition data types."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
from dataclasses import dataclass, field
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
from typing import Any
|
|
7
|
+
|
|
8
|
+
from watch_skill.acquire.sources import SourceKind
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
@dataclass
|
|
12
|
+
class AcquireResult:
|
|
13
|
+
"""A source resolved to local files.
|
|
14
|
+
|
|
15
|
+
``video_path`` is ``None`` only for captions-only probes (no media was
|
|
16
|
+
needed). ``info`` carries whatever metadata the acquirer learned (title,
|
|
17
|
+
uploader, duration, webpage URL).
|
|
18
|
+
"""
|
|
19
|
+
|
|
20
|
+
source: str
|
|
21
|
+
kind: SourceKind
|
|
22
|
+
video_path: Path | None
|
|
23
|
+
subtitle_path: Path | None = None
|
|
24
|
+
info: dict[str, Any] = field(default_factory=dict)
|
|
25
|
+
from_cache: bool = False
|
|
26
|
+
acquirer: str = "unknown" # which chain step produced this: yt-dlp | cobalt | ffmpeg | local
|