watch-skill 1.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (105) hide show
  1. watch_skill/__init__.py +19 -0
  2. watch_skill/acquire/__init__.py +12 -0
  3. watch_skill/acquire/cache.py +150 -0
  4. watch_skill/acquire/cobalt.py +96 -0
  5. watch_skill/acquire/direct.py +50 -0
  6. watch_skill/acquire/resolver.py +176 -0
  7. watch_skill/acquire/sources.py +54 -0
  8. watch_skill/acquire/types.py +26 -0
  9. watch_skill/acquire/ytdlp.py +208 -0
  10. watch_skill/answer/__init__.py +11 -0
  11. watch_skill/answer/cache.py +172 -0
  12. watch_skill/answer/confidence.py +122 -0
  13. watch_skill/answer/crops.py +75 -0
  14. watch_skill/answer/engine.py +397 -0
  15. watch_skill/answer/ladder.py +182 -0
  16. watch_skill/answer/localize.py +218 -0
  17. watch_skill/answer/types.py +62 -0
  18. watch_skill/batch.py +158 -0
  19. watch_skill/bench/__init__.py +4 -0
  20. watch_skill/bench/perception.py +175 -0
  21. watch_skill/config.py +295 -0
  22. watch_skill/errors.py +98 -0
  23. watch_skill/extract/__init__.py +11 -0
  24. watch_skill/extract/bug_report.py +139 -0
  25. watch_skill/extract/chapters.py +123 -0
  26. watch_skill/extract/hook.py +152 -0
  27. watch_skill/health/__init__.py +6 -0
  28. watch_skill/health/agents_setup.py +151 -0
  29. watch_skill/health/binaries.py +277 -0
  30. watch_skill/health/clean.py +114 -0
  31. watch_skill/health/doctor.py +633 -0
  32. watch_skill/health/log.py +46 -0
  33. watch_skill/health/vision_setup.py +297 -0
  34. watch_skill/index/__init__.py +29 -0
  35. watch_skill/index/db.py +299 -0
  36. watch_skill/index/embeddings.py +74 -0
  37. watch_skill/index/retrieval.py +266 -0
  38. watch_skill/index/store.py +340 -0
  39. watch_skill/index/textnorm.py +208 -0
  40. watch_skill/integrations/__init__.py +13 -0
  41. watch_skill/integrations/_core.py +88 -0
  42. watch_skill/integrations/autogen.py +26 -0
  43. watch_skill/integrations/crewai.py +40 -0
  44. watch_skill/integrations/langchain.py +30 -0
  45. watch_skill/integrations/llamaindex.py +27 -0
  46. watch_skill/integrations/openai_agents.py +29 -0
  47. watch_skill/jobs.py +115 -0
  48. watch_skill/lessons/__init__.py +23 -0
  49. watch_skill/lessons/classify.py +105 -0
  50. watch_skill/lessons/evals.py +192 -0
  51. watch_skill/lessons/inject.py +81 -0
  52. watch_skill/lessons/profiles.py +94 -0
  53. watch_skill/lessons/report.py +96 -0
  54. watch_skill/lessons/store.py +184 -0
  55. watch_skill/library/__init__.py +23 -0
  56. watch_skill/library/notes.py +204 -0
  57. watch_skill/library/synthesize.py +349 -0
  58. watch_skill/loop/__init__.py +61 -0
  59. watch_skill/loop/artifact.py +82 -0
  60. watch_skill/loop/capture.py +226 -0
  61. watch_skill/loop/critic.py +331 -0
  62. watch_skill/loop/diff.py +125 -0
  63. watch_skill/loop/framework.py +215 -0
  64. watch_skill/loop/monitor.py +198 -0
  65. watch_skill/loop/reportfmt.py +42 -0
  66. watch_skill/loop/runner.py +293 -0
  67. watch_skill/loop/webhook.py +95 -0
  68. watch_skill/perceive/__init__.py +23 -0
  69. watch_skill/perceive/budget.py +97 -0
  70. watch_skill/perceive/engine.py +213 -0
  71. watch_skill/perceive/media.py +91 -0
  72. watch_skill/perceive/ocr.py +151 -0
  73. watch_skill/perceive/ocr_backends.py +315 -0
  74. watch_skill/perceive/scenes.py +87 -0
  75. watch_skill/perceive/types.py +88 -0
  76. watch_skill/report.py +108 -0
  77. watch_skill/surfaces/__init__.py +5 -0
  78. watch_skill/surfaces/api/__init__.py +9 -0
  79. watch_skill/surfaces/api/app.py +337 -0
  80. watch_skill/surfaces/cli/__init__.py +1 -0
  81. watch_skill/surfaces/cli/main.py +1035 -0
  82. watch_skill/surfaces/mcp/__init__.py +1 -0
  83. watch_skill/surfaces/mcp/server.py +619 -0
  84. watch_skill/transcribe/__init__.py +12 -0
  85. watch_skill/transcribe/audio.py +131 -0
  86. watch_skill/transcribe/cloud.py +190 -0
  87. watch_skill/transcribe/diarize.py +116 -0
  88. watch_skill/transcribe/ladder.py +71 -0
  89. watch_skill/transcribe/local.py +113 -0
  90. watch_skill/transcribe/types.py +66 -0
  91. watch_skill/transcribe/vtt.py +64 -0
  92. watch_skill/viewer.py +213 -0
  93. watch_skill/vision/__init__.py +22 -0
  94. watch_skill/vision/client.py +304 -0
  95. watch_skill/vision/cost.py +52 -0
  96. watch_skill/vision/local_health.py +119 -0
  97. watch_skill/vision/model.py +149 -0
  98. watch_skill/vision/prices.json +19 -0
  99. watch_skill/vision/registry.py +144 -0
  100. watch_skill/watch.py +177 -0
  101. watch_skill-1.2.0.dist-info/METADATA +410 -0
  102. watch_skill-1.2.0.dist-info/RECORD +105 -0
  103. watch_skill-1.2.0.dist-info/WHEEL +4 -0
  104. watch_skill-1.2.0.dist-info/entry_points.txt +2 -0
  105. watch_skill-1.2.0.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,19 @@
1
+ """Watch Skill — give any agent a video input.
2
+
3
+ Core engine: acquisition, perception, transcription, indexing, vision, and
4
+ the autonomous watch-critique-iterate loop. Surfaces (MCP, CLI, REST) are
5
+ thin wrappers around this package and live outside it.
6
+ """
7
+
8
+ from importlib.metadata import PackageNotFoundError
9
+ from importlib.metadata import version as _pkg_version
10
+
11
+ # Read the version from installed metadata instead of restating it here.
12
+ # v1.0.0 shipped reporting "0.6.0" because the release bumped every manifest
13
+ # and missed this line; deriving it removes the chance of a second drift.
14
+ try:
15
+ __version__ = _pkg_version("watch-skill")
16
+ except PackageNotFoundError: # running from a source tree with no install
17
+ __version__ = "0.0.0+unknown"
18
+
19
+ __all__ = ["__version__"]
@@ -0,0 +1,12 @@
1
+ """Source acquisition: resolve ANY input to a local video file.
2
+
3
+ Self-healing chain: yt-dlp -> auto-update + retry -> self-hosted cobalt
4
+ (opt-in) -> direct ffmpeg pull. Content-addressed cache with LRU eviction.
5
+ Privacy invariants: no cookies/logins, video files never leave the machine.
6
+ """
7
+
8
+ from watch_skill.acquire.resolver import acquire, fetch_captions_only
9
+ from watch_skill.acquire.sources import SourceKind, classify_source
10
+ from watch_skill.acquire.types import AcquireResult
11
+
12
+ __all__ = ["AcquireResult", "SourceKind", "acquire", "classify_source", "fetch_captions_only"]
@@ -0,0 +1,150 @@
1
+ """Content-addressed download cache: never re-download an unchanged video.
2
+
3
+ Layout: ``<data_dir>/cache/<key>/`` where ``key`` is a hash of the normalized
4
+ source URL. Each entry holds the media file(s), optional subtitles, and an
5
+ ``entry.json`` with source metadata. Access time is tracked by touching
6
+ ``entry.json`` so LRU eviction can order entries without OS atime support.
7
+ """
8
+ from __future__ import annotations
9
+
10
+ import hashlib
11
+ import json
12
+ import os
13
+ import shutil
14
+ import time
15
+ from dataclasses import dataclass
16
+ from pathlib import Path
17
+ from typing import Any
18
+
19
+ from watch_skill.config import get_settings
20
+
21
+ ENTRY_FILE = "entry.json"
22
+
23
+
24
+ @dataclass
25
+ class CacheEntry:
26
+ """One cached download."""
27
+
28
+ key: str
29
+ dir: Path
30
+ video_path: Path | None
31
+ subtitle_path: Path | None
32
+ info: dict[str, Any]
33
+
34
+
35
+ def cache_key(source: str) -> str:
36
+ """Stable key for a source URL (case-preserved, whitespace-trimmed)."""
37
+ return hashlib.sha256(source.strip().encode("utf-8")).hexdigest()[:24]
38
+
39
+
40
+ def entry_dir(source: str, create: bool = False) -> Path:
41
+ """Directory where this source's download lives (created on demand)."""
42
+ directory = get_settings().cache_dir / cache_key(source)
43
+ if create:
44
+ directory.mkdir(parents=True, exist_ok=True)
45
+ return directory
46
+
47
+
48
+ def _load_entry(directory: Path) -> CacheEntry | None:
49
+ meta_path = directory / ENTRY_FILE
50
+ if not meta_path.is_file():
51
+ return None
52
+ try:
53
+ meta = json.loads(meta_path.read_text(encoding="utf-8"))
54
+ except (OSError, json.JSONDecodeError):
55
+ return None
56
+ video = directory / meta["video"] if meta.get("video") else None
57
+ subs = directory / meta["subtitle"] if meta.get("subtitle") else None
58
+ if video is not None and not video.is_file():
59
+ return None
60
+ return CacheEntry(
61
+ key=directory.name,
62
+ dir=directory,
63
+ video_path=video,
64
+ subtitle_path=subs if subs and subs.is_file() else None,
65
+ info=meta.get("info", {}),
66
+ )
67
+
68
+
69
+ def lookup(source: str) -> CacheEntry | None:
70
+ """Return the cached entry for ``source`` and mark it recently used."""
71
+ directory = entry_dir(source)
72
+ entry = _load_entry(directory)
73
+ if entry is not None:
74
+ try:
75
+ os.utime(directory / ENTRY_FILE)
76
+ except OSError:
77
+ pass
78
+ return entry
79
+
80
+
81
+ def commit(
82
+ source: str,
83
+ video_path: Path | None,
84
+ subtitle_path: Path | None = None,
85
+ info: dict[str, Any] | None = None,
86
+ ) -> CacheEntry:
87
+ """Record a completed download that already lives inside the entry dir.
88
+
89
+ Acquirers download *into* :func:`entry_dir` so committing is just writing
90
+ the manifest — no copy of a multi-GB file.
91
+ """
92
+ directory = entry_dir(source, create=True)
93
+ meta = {
94
+ "source": source,
95
+ "video": video_path.name if video_path else None,
96
+ "subtitle": subtitle_path.name if subtitle_path else None,
97
+ "info": info or {},
98
+ "committed_at": time.time(),
99
+ }
100
+ (directory / ENTRY_FILE).write_text(
101
+ json.dumps(meta, ensure_ascii=False, indent=2), encoding="utf-8"
102
+ )
103
+ evict_lru()
104
+ return CacheEntry(
105
+ key=directory.name,
106
+ dir=directory,
107
+ video_path=video_path,
108
+ subtitle_path=subtitle_path,
109
+ info=info or {},
110
+ )
111
+
112
+
113
+ def _dir_size(directory: Path) -> int:
114
+ return sum(p.stat().st_size for p in directory.rglob("*") if p.is_file())
115
+
116
+
117
+ def evict_lru(max_bytes: int | None = None) -> int:
118
+ """Delete least-recently-used entries until the cache fits the cap.
119
+
120
+ Returns the number of evicted entries. Recency = mtime of ``entry.json``
121
+ (touched on every :func:`lookup` hit).
122
+ """
123
+ cap = max_bytes if max_bytes is not None else get_settings().cache_max_bytes
124
+ cache_dir = get_settings().cache_dir
125
+ if not cache_dir.is_dir():
126
+ return 0
127
+
128
+ entries: list[tuple[float, Path, int]] = []
129
+ total = 0
130
+ for directory in cache_dir.iterdir():
131
+ meta = directory / ENTRY_FILE
132
+ if not directory.is_dir() or not meta.is_file():
133
+ continue
134
+ size = _dir_size(directory)
135
+ total += size
136
+ entries.append((meta.stat().st_mtime, directory, size))
137
+
138
+ evicted = 0
139
+ entries.sort() # oldest first
140
+ while total > cap and entries:
141
+ _, directory, size = entries.pop(0)
142
+ shutil.rmtree(directory, ignore_errors=True)
143
+ total -= size
144
+ evicted += 1
145
+ return evicted
146
+
147
+
148
+ def clear() -> None:
149
+ """Remove the entire cache (used by `watch-skill cache clear`)."""
150
+ shutil.rmtree(get_settings().cache_dir, ignore_errors=True)
@@ -0,0 +1,96 @@
1
+ """cobalt fallback acquirer (self-hosted instances only).
2
+
3
+ Used only after yt-dlp (and its self-update retry) failed, and only when the
4
+ user configured ``WATCHSKILL_COBALT_API_URL``. The public api.cobalt.tools
5
+ now requires JWT auth (verified 2026-07-05: anonymous POST returns
6
+ ``error.api.auth.jwt.missing``), so without a configured self-hosted
7
+ instance the chain skips straight to the direct-URL fallback instead of
8
+ burning a doomed network round-trip.
9
+ """
10
+ from __future__ import annotations
11
+
12
+ import os
13
+ from pathlib import Path
14
+ from typing import Any
15
+
16
+ import httpx
17
+
18
+ from watch_skill.errors import AcquisitionError
19
+
20
+
21
+ def _api_url() -> str | None:
22
+ return os.environ.get("WATCHSKILL_COBALT_API_URL") or None
23
+
24
+
25
+ def is_configured() -> bool:
26
+ """cobalt participates in the fallback chain only when an instance is set."""
27
+ return _api_url() is not None
28
+
29
+
30
+ def _request_media_url(url: str, timeout: float = 60.0) -> str:
31
+ """Ask the cobalt instance to resolve ``url`` to a downloadable media URL."""
32
+ api_url = _api_url()
33
+ if api_url is None:
34
+ raise AcquisitionError(
35
+ "no cobalt instance configured",
36
+ code="acquire.cobalt_not_configured",
37
+ fix="set WATCHSKILL_COBALT_API_URL to a self-hosted cobalt instance "
38
+ "(the public API requires auth); the chain continues without it",
39
+ details={"url": url},
40
+ )
41
+ try:
42
+ response = httpx.post(
43
+ api_url,
44
+ json={"url": url, "videoQuality": "720", "filenameStyle": "basic"},
45
+ headers={"Accept": "application/json", "Content-Type": "application/json"},
46
+ timeout=timeout,
47
+ follow_redirects=True,
48
+ )
49
+ payload: dict[str, Any] = response.json()
50
+ except (httpx.HTTPError, ValueError) as exc:
51
+ raise AcquisitionError(
52
+ f"cobalt API unreachable or returned non-JSON: {exc}",
53
+ code="acquire.cobalt_unreachable",
54
+ fix="set WATCHSKILL_COBALT_API_URL to a working cobalt instance, or skip",
55
+ details={"url": url, "api": _api_url()},
56
+ ) from exc
57
+
58
+ status = payload.get("status")
59
+ if status in ("tunnel", "redirect", "stream") and payload.get("url"):
60
+ return str(payload["url"])
61
+ raise AcquisitionError(
62
+ f"cobalt could not resolve this URL (status={status!r})",
63
+ code="acquire.cobalt_failed",
64
+ fix="the resolver will continue down the fallback chain",
65
+ details={"url": url, "response": {k: payload.get(k) for k in ("status", "error")}},
66
+ )
67
+
68
+
69
+ def download(url: str, dest: Path, timeout: float = 1800.0) -> Path:
70
+ """Resolve via cobalt and stream the media to ``dest``."""
71
+ media_url = _request_media_url(url)
72
+ dest.parent.mkdir(parents=True, exist_ok=True)
73
+ tmp = dest.with_suffix(dest.suffix + ".part")
74
+ try:
75
+ with httpx.stream("GET", media_url, follow_redirects=True, timeout=timeout) as resp:
76
+ resp.raise_for_status()
77
+ with tmp.open("wb") as fh:
78
+ for chunk in resp.iter_bytes(1024 * 256):
79
+ fh.write(chunk)
80
+ tmp.replace(dest)
81
+ except httpx.HTTPError as exc:
82
+ tmp.unlink(missing_ok=True)
83
+ raise AcquisitionError(
84
+ f"cobalt media download failed: {exc}",
85
+ code="acquire.cobalt_download_failed",
86
+ fix="the resolver will continue down the fallback chain",
87
+ details={"url": url},
88
+ ) from exc
89
+ if not dest.is_file() or dest.stat().st_size == 0:
90
+ raise AcquisitionError(
91
+ "cobalt returned an empty file",
92
+ code="acquire.cobalt_empty",
93
+ fix="the resolver continues down the fallback chain on its own",
94
+ details={"url": url},
95
+ )
96
+ return dest
@@ -0,0 +1,50 @@
1
+ """Direct ffmpeg pull: media URLs and HLS/DASH manifests, incl. bounded live capture."""
2
+ from __future__ import annotations
3
+
4
+ import subprocess
5
+ import sys
6
+ from pathlib import Path
7
+
8
+ from watch_skill.errors import AcquisitionError
9
+ from watch_skill.health.binaries import require_binary
10
+
11
+
12
+ def ffmpeg_pull(
13
+ url: str,
14
+ dest: Path,
15
+ duration_seconds: float | None = None,
16
+ timeout: float = 3600.0,
17
+ ) -> Path:
18
+ """Fetch a direct media URL or HLS/DASH manifest into a local mp4.
19
+
20
+ ``duration_seconds`` bounds the capture — required for live streams,
21
+ which would otherwise download forever. Stream-copies when possible and
22
+ falls back to transcoding only if the container rejects the codecs.
23
+ """
24
+ ffmpeg = require_binary("ffmpeg")
25
+ dest.parent.mkdir(parents=True, exist_ok=True)
26
+
27
+ base: list[str] = [str(ffmpeg), "-hide_banner", "-loglevel", "error", "-y", "-i", url]
28
+ if duration_seconds is not None:
29
+ base += ["-t", f"{duration_seconds:.3f}"]
30
+
31
+ copy_cmd = base + ["-c", "copy", "-movflags", "+faststart", str(dest)]
32
+ result = subprocess.run(
33
+ copy_cmd, capture_output=True, text=True, timeout=timeout,
34
+ encoding="utf-8", errors="replace",
35
+ )
36
+ if result.returncode != 0 or not dest.is_file() or dest.stat().st_size == 0:
37
+ print("[watch-skill] stream copy failed — transcoding…", file=sys.stderr)
38
+ transcode_cmd = base + ["-c:v", "libx264", "-preset", "veryfast", "-c:a", "aac", str(dest)]
39
+ result = subprocess.run(
40
+ transcode_cmd, capture_output=True, text=True, timeout=timeout,
41
+ encoding="utf-8", errors="replace",
42
+ )
43
+ if result.returncode != 0 or not dest.is_file() or dest.stat().st_size == 0:
44
+ raise AcquisitionError(
45
+ "ffmpeg could not fetch the media URL",
46
+ code="acquire.ffmpeg_pull_failed",
47
+ fix="check the URL is reachable and actually points at media/a manifest",
48
+ details={"url": url, "stderr_tail": result.stderr[-2000:]},
49
+ )
50
+ return dest
@@ -0,0 +1,176 @@
1
+ """Resolve ANY input to a local video file, with cache and fallback chain.
2
+
3
+ Chain for network sources (each hop logs why the previous one failed):
4
+ yt-dlp -> (auto-update + retry, inside ytdlp.download) -> self-hosted cobalt
5
+ (only when WATCHSKILL_COBALT_API_URL is set — the public instance requires
6
+ auth) -> direct ffmpeg pull. Direct/manifest URLs try yt-dlp first too (it handles
7
+ both), with ffmpeg as the reliable second step. Screen/window capture is
8
+ provided by ``watch_skill.loop.capture`` (Milestone 3) — the resolver returns
9
+ a structured error pointing there until then.
10
+ """
11
+ from __future__ import annotations
12
+
13
+ import sys
14
+ from pathlib import Path
15
+
16
+ from watch_skill.acquire import cache, cobalt, direct, ytdlp
17
+ from watch_skill.acquire.sources import SourceKind, classify_source, is_url_kind
18
+ from watch_skill.acquire.types import AcquireResult
19
+ from watch_skill.errors import AcquisitionError
20
+ from watch_skill.health.log import record_incident
21
+
22
+ VIDEO_EXTS = ytdlp.VIDEO_EXTS
23
+
24
+
25
+ def _resolve_local(source: str) -> AcquireResult:
26
+ path = Path(source).expanduser().resolve()
27
+ if path.is_dir():
28
+ # "file not found" is wrong and unhelpful here: the path exists, and
29
+ # a folder of recordings is what `batch` is for.
30
+ raise AcquisitionError(
31
+ f"{path} is a directory, not a video file",
32
+ code="acquire.is_a_directory",
33
+ fix=f'watch every video in it with: watch-skill batch "{path}"',
34
+ details={"source": source},
35
+ )
36
+ if not path.is_file():
37
+ raise AcquisitionError(
38
+ f"file not found: {path}",
39
+ code="acquire.file_not_found",
40
+ fix="check the path; remember to quote paths containing spaces",
41
+ details={"source": source},
42
+ )
43
+ if path.suffix.lower() not in VIDEO_EXTS:
44
+ print(
45
+ f"[watch-skill] warning: {path.suffix} is not a known video extension — proceeding",
46
+ file=sys.stderr,
47
+ )
48
+ return AcquireResult(
49
+ source=source,
50
+ kind=SourceKind.LOCAL_FILE,
51
+ video_path=path,
52
+ info={"title": path.name, "url": str(path)},
53
+ acquirer="local",
54
+ )
55
+
56
+
57
+ def _from_cache(source: str, kind: SourceKind) -> AcquireResult | None:
58
+ entry = cache.lookup(source)
59
+ if entry is None or entry.video_path is None:
60
+ return None
61
+ print(f"[watch-skill] cache hit: {entry.dir}", file=sys.stderr)
62
+ return AcquireResult(
63
+ source=source,
64
+ kind=kind,
65
+ video_path=entry.video_path,
66
+ subtitle_path=entry.subtitle_path,
67
+ info=entry.info,
68
+ from_cache=True,
69
+ acquirer="cache",
70
+ )
71
+
72
+
73
+ def _try_chain(
74
+ source: str, kind: SourceKind, out_dir: Path,
75
+ audio_only: bool, duration_cap: float | None,
76
+ ) -> AcquireResult:
77
+ """Walk the fallback chain, recording each hop's failure."""
78
+ failures: list[str] = []
79
+
80
+ try:
81
+ dl = ytdlp.download(source, out_dir, audio_only=audio_only)
82
+ return AcquireResult(
83
+ source=source, kind=kind, video_path=Path(dl["video_path"]),
84
+ subtitle_path=dl["subtitle_path"], info=dl["info"], acquirer="yt-dlp",
85
+ )
86
+ except AcquisitionError as exc:
87
+ failures.append(f"yt-dlp: {exc.message}")
88
+ record_incident("acquire_fallback", "yt-dlp failed, trying fallbacks", url=source)
89
+ print(f"[watch-skill] yt-dlp failed ({exc.code}) — trying fallbacks…", file=sys.stderr)
90
+
91
+ if kind == SourceKind.PAGE_URL and cobalt.is_configured():
92
+ try:
93
+ video = cobalt.download(source, out_dir / "media.mp4")
94
+ return AcquireResult(
95
+ source=source, kind=kind, video_path=video, acquirer="cobalt",
96
+ info={"url": source},
97
+ )
98
+ except AcquisitionError as exc:
99
+ failures.append(f"cobalt: {exc.message}")
100
+ record_incident("acquire_fallback", "cobalt failed, trying ffmpeg", url=source)
101
+ print(f"[watch-skill] cobalt failed ({exc.code}) — trying direct ffmpeg…", file=sys.stderr)
102
+
103
+ try:
104
+ video = direct.ffmpeg_pull(source, out_dir / "media.mp4", duration_seconds=duration_cap)
105
+ return AcquireResult(
106
+ source=source, kind=kind, video_path=video, acquirer="ffmpeg",
107
+ info={"url": source},
108
+ )
109
+ except AcquisitionError as exc:
110
+ failures.append(f"ffmpeg: {exc.message}")
111
+
112
+ raise AcquisitionError(
113
+ "every acquirer in the fallback chain failed",
114
+ code="acquire.chain_exhausted",
115
+ fix="check the URL; if it needs login or is region-locked, Watch Skill "
116
+ "will not bypass that (privacy invariant: no cookies, no logins)",
117
+ details={"source": source, "failures": failures},
118
+ )
119
+
120
+
121
+ def acquire(
122
+ source: str,
123
+ audio_only: bool = False,
124
+ duration_cap: float | None = None,
125
+ use_cache: bool = True,
126
+ ) -> AcquireResult:
127
+ """Resolve ``source`` (URL / manifest / local path / capture spec) to local files.
128
+
129
+ Network downloads land in the content-addressed cache and are reused on
130
+ the next call. ``duration_cap`` bounds live-stream capture.
131
+ """
132
+ kind = classify_source(source)
133
+
134
+ if kind in (SourceKind.SCREEN, SourceKind.WINDOW):
135
+ raise AcquisitionError(
136
+ "screen/window capture is provided by the loop module",
137
+ code="acquire.capture_required",
138
+ fix="use `watch-skill capture` / the capture() API (Milestone 3)",
139
+ details={"source": source, "kind": kind.value},
140
+ )
141
+
142
+ if kind == SourceKind.LOCAL_FILE:
143
+ return _resolve_local(source)
144
+
145
+ assert is_url_kind(kind)
146
+ if use_cache:
147
+ cached = _from_cache(source, kind)
148
+ if cached is not None:
149
+ return cached
150
+
151
+ out_dir = cache.entry_dir(source, create=True)
152
+ result = _try_chain(source, kind, out_dir, audio_only, duration_cap)
153
+ # Audio-only fetches are not committed: a later full fetch must not be
154
+ # shadowed by a cache entry that has no video frames in it.
155
+ if use_cache and not audio_only:
156
+ cache.commit(source, result.video_path, result.subtitle_path, result.info)
157
+ return result
158
+
159
+
160
+ def fetch_captions_only(source: str) -> AcquireResult:
161
+ """Probe captions + metadata for a URL without downloading media."""
162
+ kind = classify_source(source)
163
+ if not is_url_kind(kind):
164
+ raise AcquisitionError(
165
+ "captions probe only applies to URLs",
166
+ code="acquire.not_a_url",
167
+ fix="local files have no platform captions; transcription falls "
168
+ "back to local whisper automatically",
169
+ details={"source": source},
170
+ )
171
+ out_dir = cache.entry_dir(source, create=True)
172
+ probe = ytdlp.fetch_captions(source, out_dir)
173
+ return AcquireResult(
174
+ source=source, kind=kind, video_path=None,
175
+ subtitle_path=probe["subtitle_path"], info=probe["info"], acquirer="yt-dlp",
176
+ )
@@ -0,0 +1,54 @@
1
+ """Classify any input string into a source kind the resolver can act on."""
2
+ from __future__ import annotations
3
+
4
+ from enum import Enum
5
+ from pathlib import Path
6
+ from urllib.parse import urlparse
7
+
8
+ MEDIA_EXTS = {
9
+ ".mp4", ".mkv", ".webm", ".mov", ".m4v", ".avi", ".flv", ".wmv",
10
+ ".ts", ".m4a", ".mp3", ".opus", ".ogg", ".wav",
11
+ }
12
+ MANIFEST_EXTS = {".m3u8", ".mpd"}
13
+
14
+
15
+ class SourceKind(str, Enum): # noqa: UP042 — StrEnum needs 3.11+, str+Enum reads the same
16
+ """What kind of input the user handed us."""
17
+
18
+ PAGE_URL = "page_url" # any website; yt-dlp's 1800+ extractors
19
+ DIRECT_URL = "direct_url" # URL pointing straight at a media file
20
+ MANIFEST_URL = "manifest_url" # HLS (.m3u8) / DASH (.mpd) manifest
21
+ LOCAL_FILE = "local_file"
22
+ SCREEN = "screen" # `screen:` — capture the screen (Milestone 3)
23
+ WINDOW = "window" # `window:<title>` — capture one window (Milestone 3)
24
+
25
+
26
+ def classify_source(source: str) -> SourceKind:
27
+ """Map a raw source string to a :class:`SourceKind`.
28
+
29
+ Rules: ``screen:`` / ``window:`` prefixes are capture requests; http(s)
30
+ URLs are split by path suffix into manifest / direct-media / page URLs;
31
+ everything else is treated as a local file path.
32
+ """
33
+ stripped = source.strip()
34
+ lowered = stripped.lower()
35
+ if lowered == "screen:" or lowered.startswith("screen:"):
36
+ return SourceKind.SCREEN
37
+ if lowered.startswith("window:"):
38
+ return SourceKind.WINDOW
39
+
40
+ parsed = urlparse(stripped)
41
+ if parsed.scheme in ("http", "https") and parsed.netloc:
42
+ suffix = Path(parsed.path).suffix.lower()
43
+ if suffix in MANIFEST_EXTS:
44
+ return SourceKind.MANIFEST_URL
45
+ if suffix in MEDIA_EXTS:
46
+ return SourceKind.DIRECT_URL
47
+ return SourceKind.PAGE_URL
48
+
49
+ return SourceKind.LOCAL_FILE
50
+
51
+
52
+ def is_url_kind(kind: SourceKind) -> bool:
53
+ """True for any source that must be fetched over the network."""
54
+ return kind in (SourceKind.PAGE_URL, SourceKind.DIRECT_URL, SourceKind.MANIFEST_URL)
@@ -0,0 +1,26 @@
1
+ """Shared acquisition data types."""
2
+ from __future__ import annotations
3
+
4
+ from dataclasses import dataclass, field
5
+ from pathlib import Path
6
+ from typing import Any
7
+
8
+ from watch_skill.acquire.sources import SourceKind
9
+
10
+
11
+ @dataclass
12
+ class AcquireResult:
13
+ """A source resolved to local files.
14
+
15
+ ``video_path`` is ``None`` only for captions-only probes (no media was
16
+ needed). ``info`` carries whatever metadata the acquirer learned (title,
17
+ uploader, duration, webpage URL).
18
+ """
19
+
20
+ source: str
21
+ kind: SourceKind
22
+ video_path: Path | None
23
+ subtitle_path: Path | None = None
24
+ info: dict[str, Any] = field(default_factory=dict)
25
+ from_cache: bool = False
26
+ acquirer: str = "unknown" # which chain step produced this: yt-dlp | cobalt | ffmpeg | local