scripthaul 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
scripthaul/__init__.py ADDED
@@ -0,0 +1,616 @@
1
+ """Dependency-free ScriptHaul API client."""
2
+
3
+ import json
4
+ import os
5
+ import re
6
+ import time
7
+ import uuid
8
+ from datetime import datetime, timezone
9
+ from email.utils import parsedate_to_datetime
10
+ from importlib.metadata import PackageNotFoundError
11
+ from importlib.metadata import version as _installed_version
12
+ from urllib.error import HTTPError, URLError
13
+ from urllib.parse import quote, urlencode, urljoin, urlparse
14
+ from urllib.request import HTTPRedirectHandler, Request, build_opener
15
+
16
+ DEFAULT_BASE_URL = "https://api.scripthaul.com"
17
+ RETRY_STATUSES = {429, 502, 503}
18
+ TERMINAL_JOB_STATUSES = {"completed", "failed", "cancelled"}
19
+ # Polls are bounded separately from retries: a 202 Retry-After is the server's
20
+ # pacing hint, never a reason to give up.
21
+ MAX_POLL_DELAY = 30.0
22
+ # Fallback for a source checkout that was never installed; the test suite pins
23
+ # it to pyproject.toml so the two cannot drift.
24
+ _FALLBACK_VERSION = "0.2.0"
25
+
26
+
27
+ def _detect_version():
28
+ try:
29
+ return _installed_version("scripthaul")
30
+ except PackageNotFoundError:
31
+ return _FALLBACK_VERSION
32
+
33
+
34
+ __version__ = _detect_version()
35
+ SDK_HEADER = "python/%s" % __version__
36
+
37
+ _UNSET = object()
38
+
39
+
40
+ class ScriptHaulError(Exception):
41
+ """Base API exception; request_id is always present, or None client-side."""
42
+
43
+ error_class = "UpstreamError"
44
+
45
+ def __init__(self, message, *, error_class=_UNSET, code="unexpected_error",
46
+ status=None, retryable=False, retry_after=None, request_id=None,
47
+ docs_url=None, details=None):
48
+ super().__init__(message)
49
+ # Server classes pin their error_class on the class. Client-only
50
+ # classes pin None, so a local socket failure or an undersized timeout
51
+ # can never be mistaken for a relay verdict.
52
+ self.error_class = (self.__class__.error_class if error_class is _UNSET
53
+ else error_class)
54
+ self.errorClass = self.error_class
55
+ self.code = code
56
+ self.status = status
57
+ self.retryable = bool(retryable)
58
+ self.retry_after = retry_after
59
+ self.request_id = request_id
60
+ self.docs_url = docs_url
61
+ self.details = details or {}
62
+
63
+
64
+ class IpBlocked(ScriptHaulError): error_class = "IpBlocked"
65
+ class PoTokenRequired(ScriptHaulError): error_class = "PoTokenRequired"
66
+ class RateLimited(ScriptHaulError): error_class = "RateLimited"
67
+ class Unavailable(ScriptHaulError): error_class = "Unavailable"
68
+ class NoCaptions(ScriptHaulError): error_class = "NoCaptions"
69
+ class UpstreamError(ScriptHaulError): error_class = "UpstreamError"
70
+ class RelayUnavailable(ScriptHaulError): error_class = "RelayUnavailable"
71
+ class BudgetExhausted(ScriptHaulError): error_class = "BudgetExhausted"
72
+ class InvalidInput(ScriptHaulError): error_class = "InvalidInput"
73
+ class NotFound(ScriptHaulError): error_class = "NotFound"
74
+ class QuotaExceeded(ScriptHaulError): error_class = "QuotaExceeded"
75
+ class TurnstileFailed(ScriptHaulError): error_class = "TurnstileFailed"
76
+ class Forbidden(ScriptHaulError): error_class = "Forbidden"
77
+ class InsufficientCredits(ScriptHaulError): error_class = "InsufficientCredits"
78
+ class AccountFrozen(ScriptHaulError): error_class = "AccountFrozen"
79
+ class ConfigurationError(ScriptHaulError): error_class = "ConfigurationError"
80
+
81
+
82
+ # Client-only failures: no server verdict exists because the request did not
83
+ # complete, or the caller's own budget ran out first.
84
+ class ClientNetworkError(ScriptHaulError): error_class = None
85
+ class ClientTimeout(ScriptHaulError): error_class = None
86
+ class JobPaused(ScriptHaulError): error_class = None
87
+
88
+
89
+ ERROR_CLASSES = {
90
+ cls.error_class: cls for cls in (
91
+ IpBlocked, PoTokenRequired, RateLimited, Unavailable, NoCaptions,
92
+ UpstreamError, RelayUnavailable, BudgetExhausted, InvalidInput, NotFound,
93
+ QuotaExceeded, TurnstileFailed, Forbidden, InsufficientCredits,
94
+ AccountFrozen, ConfigurationError,
95
+ )
96
+ }
97
+
98
+
99
+ def _valid_idempotency_key(value):
100
+ return (isinstance(value, str) and bool(value.strip()) and len(value) <= 128
101
+ and re.search(r"[\x00-\x08\x0a-\x1f\x7f\u0100-\U0010ffff]", value) is None)
102
+
103
+
104
+ def _monitor_idempotency_key(value):
105
+ if value is None:
106
+ return str(uuid.uuid4())
107
+ if not _valid_idempotency_key(value):
108
+ raise InvalidInput("Idempotency-Key must contain 1 to 128 header-safe characters.",
109
+ code="invalid_idempotency_key")
110
+ return value
111
+
112
+
113
+ def _has_idempotency_key(headers):
114
+ values = [value for name, value in (headers or {}).items()
115
+ if str(name).lower() == "idempotency-key"]
116
+ # An ambiguous pair of differently-cased headers is not a replay identity.
117
+ return len(values) == 1 and _valid_idempotency_key(values[0])
118
+
119
+
120
+ class ResponseMeta:
121
+ """The header contract of one response: cache verdict, credits, limits, id."""
122
+
123
+ __slots__ = ("status", "cache", "credits_charged", "credits_balance",
124
+ "rate_limit", "request_id", "library_coverage")
125
+
126
+ def __init__(self, status, headers):
127
+ self.status = status
128
+ self.cache = headers.get("x-cache") or None
129
+ self.credits_charged = _header_number(headers, "x-credits-charged")
130
+ self.credits_balance = _header_number(headers, "x-credits-balance")
131
+ self.rate_limit = _header_number(headers, "x-ratelimit-limit")
132
+ self.request_id = headers.get("x-request-id") or None
133
+ self.library_coverage = headers.get("x-library-coverage") or None
134
+
135
+ def __repr__(self):
136
+ return ("ResponseMeta(status=%r, cache=%r, credits_charged=%r, "
137
+ "credits_balance=%r, rate_limit=%r, request_id=%r)" % (
138
+ self.status, self.cache, self.credits_charged,
139
+ self.credits_balance, self.rate_limit, self.request_id))
140
+
141
+
142
+ class ApiResponse(dict):
143
+ """A JSON body that also carries the response headers as ``meta``."""
144
+
145
+ __slots__ = ("meta",)
146
+
147
+ def __init__(self, payload, meta=None):
148
+ super().__init__(payload)
149
+ self.meta = meta
150
+
151
+
152
+ class _Secret:
153
+ """Holds the key so ``vars(client)``, ``repr`` and pretty-printers show a prefix only."""
154
+
155
+ __slots__ = ("_value",)
156
+
157
+ def __init__(self, value):
158
+ self._value = value
159
+
160
+ def get(self):
161
+ return self._value
162
+
163
+ def __repr__(self):
164
+ return "<secret %s...>" % self._value[:12]
165
+
166
+ __str__ = __repr__
167
+
168
+
169
+ def _header_number(headers, name):
170
+ value = headers.get(name)
171
+ if value in (None, ""):
172
+ return None
173
+ try:
174
+ number = float(value)
175
+ except (TypeError, ValueError):
176
+ return None
177
+ return int(number) if number.is_integer() else number
178
+
179
+
180
+ class _NoRedirect(HTTPRedirectHandler):
181
+ def redirect_request(self, *_args, **_kwargs):
182
+ return None
183
+
184
+
185
+ _NO_REDIRECT_OPENER = build_opener(_NoRedirect)
186
+
187
+
188
+ def _default_transport(method, url, headers, body, timeout):
189
+ request = Request(url, data=body, headers=headers, method=method)
190
+ try:
191
+ # Treat every redirect as an API error. urllib otherwise reuses the
192
+ # Request headers and can disclose the Bearer key to another origin.
193
+ with _NO_REDIRECT_OPENER.open(request, timeout=timeout) as response:
194
+ return response.status, dict(response.headers.items()), response.read()
195
+ except HTTPError as error:
196
+ return error.code, dict(error.headers.items()), error.read()
197
+
198
+
199
+ def _retry_after_seconds(headers, payload):
200
+ """Retry-After from the header (seconds or HTTP-date), else the envelope; None without a hint."""
201
+ value = headers.get("retry-after")
202
+ if value:
203
+ try:
204
+ return max(0.0, float(value))
205
+ except ValueError:
206
+ try:
207
+ parsed = parsedate_to_datetime(value)
208
+ if parsed.tzinfo is None:
209
+ parsed = parsed.replace(tzinfo=timezone.utc)
210
+ return max(0.0, (parsed - datetime.now(timezone.utc)).total_seconds())
211
+ except (TypeError, ValueError, OverflowError):
212
+ pass
213
+ if isinstance(payload, dict):
214
+ try:
215
+ return max(0.0, float(payload["retry_after"]))
216
+ except (KeyError, TypeError, ValueError):
217
+ return None
218
+ return None
219
+
220
+
221
+ def _poll_delay(headers, fallback):
222
+ hinted = _retry_after_seconds(headers, None)
223
+ return min(MAX_POLL_DELAY, fallback if hinted is None else hinted)
224
+
225
+
226
+ def _query(values):
227
+ clean = {}
228
+ for key, value in values.items():
229
+ if value is None:
230
+ continue
231
+ clean[key] = str(value).lower() if isinstance(value, bool) else value
232
+ return urlencode(clean)
233
+
234
+
235
+ def _paused_error(job):
236
+ reason = job.data.get("pause_reason")
237
+ details = {"pause_reason": reason, "credits": job.data.get("credits"), "job": job.data}
238
+ request_id = job.meta.request_id if job.meta else None
239
+ if reason == "insufficient_credits":
240
+ error = InsufficientCredits(
241
+ "Job %s is paused until the account buys credits "
242
+ "(pause_reason insufficient_credits)." % job.id,
243
+ code="job_paused_insufficient_credits", retryable=False,
244
+ details=details, request_id=request_id,
245
+ )
246
+ else:
247
+ error = JobPaused(
248
+ "Job %s is paused (%s); pass wait_through_pauses=True to keep polling."
249
+ % (job.id, reason or "reason not given"),
250
+ code="job_paused", retryable=True, details=details, request_id=request_id,
251
+ )
252
+ error.job = job
253
+ return error
254
+
255
+
256
+ class Job:
257
+ """Mutable view of one bulk job."""
258
+
259
+ def __init__(self, client, data):
260
+ if not isinstance(data, dict) or not data.get("id"):
261
+ raise InvalidInput("The API returned a job without an id.",
262
+ code="invalid_job_response")
263
+ self._client = client
264
+ self.data = data
265
+
266
+ @property
267
+ def id(self):
268
+ return self.data["id"]
269
+
270
+ @property
271
+ def status(self):
272
+ return self.data.get("status")
273
+
274
+ @property
275
+ def meta(self):
276
+ return getattr(self.data, "meta", None)
277
+
278
+ def refresh(self):
279
+ payload = self._client._request("GET", "/v1/jobs/%s" % self.id)
280
+ self.data = _unwrap_job(payload)
281
+ return self
282
+
283
+ def cancel(self):
284
+ """Cancel the job, releasing its unsettled reservation; delivered rows keep their charge."""
285
+ payload = self._client._request("POST", "/v1/jobs/%s/cancel" % self.id)
286
+ job = payload.get("job") if isinstance(payload, dict) else None
287
+ if isinstance(job, dict) and job.get("id"):
288
+ self.data = _unwrap_job(payload)
289
+ return self
290
+ return self.refresh()
291
+
292
+ def wait(self, *, poll_interval=None, timeout=None, wait_through_pauses=None):
293
+ interval = (self._client.job_poll_interval if poll_interval is None
294
+ else poll_interval)
295
+ limit = self._client.job_timeout if timeout is None else timeout
296
+ through_pauses = (self._client.wait_through_pauses if wait_through_pauses is None
297
+ else wait_through_pauses)
298
+ deadline = self._client._monotonic() + limit
299
+ while True:
300
+ if self.status in TERMINAL_JOB_STATUSES:
301
+ return self
302
+ # A paused job waits for a purchase or the cron, not for this loop:
303
+ # polling it every two seconds for ten minutes is ~300 metered
304
+ # reads that cannot change the outcome.
305
+ if self.status == "paused" and not through_pauses:
306
+ raise _paused_error(self)
307
+ if self._client._monotonic() >= deadline:
308
+ raise ClientTimeout(
309
+ "Timed out waiting for job %s." % self.id,
310
+ code="job_wait_timeout", retryable=True,
311
+ request_id=self.meta.request_id if self.meta else None,
312
+ )
313
+ if interval > 0:
314
+ self._client._sleep(interval)
315
+ self.refresh()
316
+
317
+ def __repr__(self):
318
+ return "Job(id=%r, status=%r)" % (self.id, self.status)
319
+
320
+
321
+ def _unwrap_job(payload):
322
+ job = payload.get("job", payload)
323
+ if isinstance(job, dict) and not isinstance(job, ApiResponse):
324
+ job = ApiResponse(job, getattr(payload, "meta", None))
325
+ return job
326
+
327
+
328
+ class Jobs:
329
+ def __init__(self, client):
330
+ self._client = client
331
+
332
+ def create(self, input=None, *, translate=None, idempotency_key=None, **options):
333
+ if not input:
334
+ raise InvalidInput("jobs.create() requires an input.",
335
+ code="missing_job_input")
336
+ if isinstance(input, dict):
337
+ payload = {**input, **options}
338
+ else:
339
+ payload = {"input": input, **options}
340
+ if translate is not None:
341
+ payload["translate"] = translate
342
+ headers = {"Idempotency-Key": idempotency_key or str(uuid.uuid4())}
343
+ result = self._client._request("POST", "/v1/jobs", body=payload,
344
+ headers=headers)
345
+ return Job(self._client, _unwrap_job(result))
346
+
347
+
348
+ class Channels:
349
+ def __init__(self, client):
350
+ self._client = client
351
+
352
+ def resolve(self, input):
353
+ if not input:
354
+ raise InvalidInput("channels.resolve() requires a channel input.",
355
+ code="missing_channel_input")
356
+ return self._client._request("GET", "/v1/channels/resolve?" + _query({"input": input}))
357
+
358
+ def latest(self, *, channel=None, playlist=None):
359
+ if bool(channel) == bool(playlist):
360
+ raise InvalidInput("channels.latest() requires exactly one channel or playlist.",
361
+ code="invalid_latest_scope")
362
+ return self._client._request("GET", "/v1/channels/latest?" + _query({
363
+ "channel": channel, "playlist": playlist,
364
+ }))
365
+
366
+
367
+ class Library:
368
+ def __init__(self, client):
369
+ self._client = client
370
+
371
+ def coverage(self, *, channel=None, playlist=None):
372
+ if bool(channel) == bool(playlist):
373
+ raise InvalidInput("library.coverage() requires exactly one channel or playlist.",
374
+ code="invalid_library_scope")
375
+ return self._client._request("GET", "/v1/library?" + _query({
376
+ "channel": channel, "playlist": playlist,
377
+ }))
378
+
379
+ def search(self, query, *, channel=None, video_id=None, limit=None, offset=None):
380
+ if not query or bool(channel) == bool(video_id):
381
+ raise InvalidInput("library.search() requires a query and exactly one channel or video_id.",
382
+ code="invalid_library_search")
383
+ return self._client._request("GET", "/v1/library/search?" + _query({
384
+ "q": query, "channel": channel, "video_id": video_id,
385
+ "limit": limit, "offset": offset,
386
+ }))
387
+
388
+
389
+ class Monitors:
390
+ def __init__(self, client):
391
+ self._client = client
392
+
393
+ def create(self, *, channel=None, playlist=None, auto_fetch=False,
394
+ language=None, fallback=None, webhook_url=None, idempotency_key=None):
395
+ if bool(channel) == bool(playlist):
396
+ raise InvalidInput("monitors.create() requires exactly one channel or playlist.",
397
+ code="invalid_monitor_scope")
398
+ body = {key: value for key, value in {
399
+ "channel": channel, "playlist": playlist, "auto_fetch": auto_fetch,
400
+ "language": language, "fallback": fallback, "webhook_url": webhook_url,
401
+ }.items() if value is not None}
402
+ headers = {"Idempotency-Key": _monitor_idempotency_key(idempotency_key)}
403
+ return self._client._request("POST", "/v1/monitors", body=body, headers=headers)
404
+
405
+ def list(self):
406
+ return self._client._request("GET", "/v1/monitors")
407
+
408
+ def get(self, id):
409
+ if not id:
410
+ raise InvalidInput("monitors.get() requires a monitor id.", code="missing_monitor_id")
411
+ return self._client._request("GET", "/v1/monitors/" + quote(str(id), safe=""))
412
+
413
+ def cancel(self, id):
414
+ if not id:
415
+ raise InvalidInput("monitors.cancel() requires a monitor id.", code="missing_monitor_id")
416
+ return self._client._request("DELETE", "/v1/monitors/" + quote(str(id), safe=""))
417
+
418
+
419
+ class ScriptHaul:
420
+ """Synchronous client for discovery, transcripts, listings, and bulk jobs."""
421
+
422
+ def __init__(self, api_key=None, *, base_url=None, timeout=30,
423
+ max_retries=3, retry_delay=0.5, max_retry_after=30.0,
424
+ max_polls=120, poll_interval=1.0, job_poll_interval=2.0,
425
+ job_timeout=600, wait_through_pauses=False,
426
+ transport=None, sleep=time.sleep, monotonic=time.monotonic):
427
+ key = api_key or os.environ.get("SCRIPTHAUL_API_KEY")
428
+ if not key:
429
+ raise InvalidInput("Set api_key or SCRIPTHAUL_API_KEY.",
430
+ code="missing_api_key")
431
+ # Wrapped so nothing that walks the instance (vars, pprint, a
432
+ # structured logger) can print the key. Keys are never logged.
433
+ self._api_key = _Secret(str(key))
434
+ self.key_prefix = str(key)[:12]
435
+ selected_url = base_url or os.environ.get("SCRIPTHAUL_BASE_URL")
436
+ self.base_url = (selected_url or DEFAULT_BASE_URL).rstrip("/")
437
+ parsed = urlparse(self.base_url)
438
+ if parsed.scheme not in {"http", "https"} or not parsed.netloc:
439
+ raise InvalidInput("base_url must be an absolute HTTP URL.",
440
+ code="invalid_base_url")
441
+ self.timeout = timeout
442
+ self.max_retries = max_retries
443
+ self.retry_delay = retry_delay
444
+ # Longest Retry-After the client sleeps through. Daily caps advertise
445
+ # a day; nobody wants a call that blocks until UTC midnight.
446
+ self.max_retry_after = max_retry_after
447
+ self.max_polls = max_polls
448
+ self.poll_interval = poll_interval
449
+ self.job_poll_interval = job_poll_interval
450
+ self.job_timeout = job_timeout
451
+ self.wait_through_pauses = bool(wait_through_pauses)
452
+ self._transport = transport or _default_transport
453
+ self._sleep = sleep
454
+ self._monotonic = monotonic
455
+ self.jobs = Jobs(self)
456
+ self.channels = Channels(self)
457
+ self.library = Library(self)
458
+ self.monitors = Monitors(self)
459
+
460
+ @property
461
+ def api_key(self):
462
+ return self._api_key.get()
463
+
464
+ def __repr__(self):
465
+ return "ScriptHaul(base_url=%r, api_key='%s...')" % (self.base_url, self.key_prefix)
466
+
467
+ __str__ = __repr__
468
+
469
+ def _url(self, path):
470
+ url = urljoin(self.base_url + "/", path)
471
+ if urlparse(url).netloc != urlparse(self.base_url).netloc or \
472
+ urlparse(url).scheme != urlparse(self.base_url).scheme:
473
+ raise InvalidInput("The API returned a cross-origin polling URL.",
474
+ code="unsafe_poll_url")
475
+ return url
476
+
477
+ @staticmethod
478
+ def _payload(raw):
479
+ if not raw:
480
+ return None
481
+ text = raw.decode("utf-8")
482
+ try:
483
+ return json.loads(text)
484
+ except json.JSONDecodeError:
485
+ return text
486
+
487
+ @staticmethod
488
+ def _error(status, headers, payload):
489
+ body = payload if isinstance(payload, dict) else {}
490
+ error_class = body.get("errorClass", "UpstreamError")
491
+ error_type = ERROR_CLASSES.get(error_class, ScriptHaulError)
492
+ hinted = _retry_after_seconds(headers, body)
493
+ return error_type(
494
+ body.get("message", "ScriptHaul returned HTTP %s." % status),
495
+ error_class=error_class, code=body.get("code", "unexpected_error"),
496
+ status=status, retryable=body.get("retryable", False),
497
+ retry_after=None if hinted is None else int(-(-hinted // 1)),
498
+ request_id=body.get("request_id") or headers.get("x-request-id"),
499
+ docs_url=body.get("docs_url"), details=body,
500
+ )
501
+
502
+ def _request(self, method, path, *, body=None, headers=None, poll_202=False):
503
+ url = self._url(path)
504
+ request_method = str(method).upper()
505
+ request_body = body
506
+ polls = 0
507
+ while True:
508
+ # Without a replay key, an interrupted POST may already have
509
+ # committed. A later 202-to-GET poll keeps the ordinary retry cap.
510
+ # Discovery search failures still spend proxy bytes, so their
511
+ # automatic retry is capped at one even with a higher global cap.
512
+ retry_limit = (0 if request_method == "POST" and not _has_idempotency_key(headers)
513
+ else min(self.max_retries, 1) if request_method == "GET"
514
+ and urlparse(url).path.endswith("/v1/search")
515
+ else self.max_retries)
516
+ response = None
517
+ payload = None
518
+ payload_read = False
519
+ for attempt in range(retry_limit + 1):
520
+ payload = None
521
+ payload_read = False
522
+ request_headers = {
523
+ "Authorization": "Bearer %s" % self._api_key.get(),
524
+ "Accept": "application/json",
525
+ "X-ScriptHaul-SDK": SDK_HEADER,
526
+ **(headers or {}),
527
+ }
528
+ encoded = None
529
+ if request_body is not None:
530
+ request_headers["Content-Type"] = "application/json"
531
+ encoded = json.dumps(request_body, separators=(",", ":")).encode()
532
+ try:
533
+ status, response_headers, raw = self._transport(
534
+ request_method, url, request_headers, encoded, self.timeout,
535
+ )
536
+ normalized = {str(key).lower(): value
537
+ for key, value in response_headers.items()}
538
+ response = (status, normalized, raw)
539
+ except (OSError, URLError) as cause:
540
+ if attempt < retry_limit:
541
+ self._sleep(self.retry_delay * (2 ** attempt))
542
+ continue
543
+ raise ClientNetworkError(
544
+ "The ScriptHaul API could not be reached.",
545
+ code="network_error", retryable=True,
546
+ details={"cause": str(cause)},
547
+ ) from cause
548
+ if status not in RETRY_STATUSES or attempt >= retry_limit:
549
+ break
550
+ payload = self._payload(raw)
551
+ payload_read = True
552
+ # The envelope's verdict wins: a daily cap says retryable false
553
+ # with a day-long Retry-After, and three more requests cannot
554
+ # change that.
555
+ if isinstance(payload, dict) and payload.get("retryable") is False:
556
+ break
557
+ delay = _retry_after_seconds(normalized, payload)
558
+ if delay is not None and delay > self.max_retry_after:
559
+ break
560
+ self._sleep(self.retry_delay * (2 ** attempt) if delay is None else delay)
561
+
562
+ status, response_headers, raw = response
563
+ if not payload_read:
564
+ payload = self._payload(raw)
565
+ if status == 202 and poll_202:
566
+ polls += 1
567
+ if polls > self.max_polls:
568
+ request_id = payload.get("request_id") if isinstance(payload, dict) else None
569
+ raise ClientTimeout(
570
+ "Timed out waiting for the transcript.", code="poll_timeout",
571
+ retryable=True, request_id=request_id or response_headers.get("x-request-id"),
572
+ )
573
+ if not isinstance(payload, dict) or not payload.get("poll_url"):
574
+ raise self._error(status, response_headers, payload)
575
+ self._sleep(_poll_delay(response_headers, self.poll_interval))
576
+ url = self._url(payload["poll_url"])
577
+ request_method, request_body = "GET", None
578
+ continue
579
+ if not 200 <= status < 300:
580
+ raise self._error(status, response_headers, payload)
581
+ if isinstance(payload, dict):
582
+ return ApiResponse(payload, ResponseMeta(status, response_headers))
583
+ return payload
584
+
585
+ def transcript(self, video_id, **options):
586
+ if not video_id:
587
+ raise InvalidInput("transcript() requires a video id.",
588
+ code="missing_video_id")
589
+ allowed = {key: options[key] for key in
590
+ ("format", "language", "fallback", "raw", "fresh", "translate") if key in options}
591
+ query = _query({"video_id": video_id, **allowed})
592
+ return self._request("GET", "/v1/transcript?" + query, poll_202=True)
593
+
594
+ def videos(self, input, **options):
595
+ if not input:
596
+ raise InvalidInput(
597
+ "videos() requires a channel, playlist, or list input.",
598
+ code="missing_video_input",
599
+ )
600
+ allowed = {key: options[key] for key in ("contents", "offset") if key in options}
601
+ query = _query({"url": input, **allowed})
602
+ return self._request("GET", "/v1/videos?" + query)
603
+
604
+ def search(self, query, *, type=None, channel=None, continuation=None):
605
+ if not query:
606
+ raise InvalidInput("search() requires a query.", code="missing_search_query")
607
+ return self._request("GET", "/v1/search?" + _query({
608
+ "q": query, "type": type, "channel": channel, "continuation": continuation,
609
+ }))
610
+
611
+
612
+ __all__ = [
613
+ "ScriptHaul", "Job", "ApiResponse", "ResponseMeta", "ScriptHaulError",
614
+ "ClientNetworkError", "ClientTimeout", "JobPaused", "SDK_HEADER",
615
+ "__version__", *ERROR_CLASSES.keys(),
616
+ ]
@@ -0,0 +1,106 @@
1
+ Metadata-Version: 2.4
2
+ Name: scripthaul
3
+ Version: 0.2.0
4
+ Summary: Thin Python client for the ScriptHaul transcript API
5
+ Author-email: ScriptHaul <support@scripthaul.com>
6
+ License-Expression: MIT
7
+ Project-URL: Homepage, https://api.scripthaul.com/docs
8
+ Project-URL: Documentation, https://api.scripthaul.com/docs
9
+ Project-URL: Support, https://api.scripthaul.com/docs
10
+ Project-URL: Repository, https://github.com/scripthaul/scripthaul-python
11
+ Project-URL: Issues, https://github.com/scripthaul/scripthaul-python/issues
12
+ Keywords: youtube,transcript,captions,api
13
+ Classifier: Programming Language :: Python :: 3
14
+ Requires-Python: >=3.9
15
+ Description-Content-Type: text/markdown
16
+ License-File: LICENSE
17
+ Dynamic: license-file
18
+
19
+ # ScriptHaul for Python
20
+
21
+ Dependency-free, synchronous ScriptHaul API client for Python 3.9+. MIT licensed.
22
+
23
+ ```sh
24
+ pip install scripthaul
25
+ ```
26
+
27
+ Until the first PyPI release, install straight from GitHub: `pip install git+https://github.com/scripthaul/scripthaul-python`. Issues and pull requests are welcome in this repository; the API itself is documented at [api.scripthaul.com/docs](https://api.scripthaul.com/docs).
28
+
29
+ ```python
30
+ from scripthaul import ScriptHaul
31
+
32
+ client = ScriptHaul(api_key="sh_live_...")
33
+ transcript = client.transcript("dQw4w9WgXcQ")
34
+ print(transcript.meta.cache, transcript.meta.credits_charged) # HIT 0
35
+ videos = client.videos("https://www.youtube.com/@example")
36
+ job = client.jobs.create(input="https://www.youtube.com/@example").wait()
37
+ ```
38
+
39
+ ## Transcripts
40
+
41
+ `client.transcript(video_id, **options)` accepts `translate` (default `False`). Set `language="es", translate=True` to receive the requested language when YouTube declines to translate the captions, if it can be delivered. A transcript delivered this way costs 300 credits per started 10,000 characters of transcript (at most 1,500), on top of the delivery credit; nothing extra is charged when it cannot be delivered. Without `translate`, nothing changes.
42
+
43
+ ## Jobs
44
+
45
+ `client.jobs.create()` accepts `translate` (default `False`) alongside `input`, `language`, `fallback`, and `webhook_url`. Set `translate=True` to ask for the job's language on videos whose captions YouTube declines to translate. A transcript delivered this way costs 300 credits per started 10,000 characters of transcript (at most 1,500), on top of the delivery credit; nothing extra is charged when it cannot be delivered. These translations are charged per video when delivered, never reserved. Opted-in jobs carry `translate: true` and a `translation_credits` total; their video rows carry `translation_credits`. Without `translate`, nothing changes.
46
+
47
+ ## Discovery
48
+
49
+ Search YouTube or one channel for one credit per delivered page. Identical requests within 60 seconds are free retries. Channel resolution and the latest 15 channel or playlist entries cost zero credits; latest entries come from a feed cached for ten minutes.
50
+
51
+ ```python
52
+ page = client.search("climate science", type="video", channel="@TED")
53
+ print(page["results"], page.meta.credits_charged) # video results include cached: True/False
54
+ if page["has_more"]:
55
+ next_page = client.search("climate science", type="video", channel="@TED",
56
+ continuation=page["continuation"])
57
+ channel = client.channels.resolve("@TED")
58
+ latest = client.channels.latest(channel=channel["channel_id"])
59
+ playlist_latest = client.channels.latest(playlist=channel["uploads_playlist_id"])
60
+ ```
61
+
62
+ `type` accepts `video` (default), `channel`, or `playlist`. Keep continuation strings unchanged and repeat the same query, type, and channel. Resolve a handle first and pass its `channel_id` to latest uploads so that reading the feed needs no Data API call. `examples/discovery.py` requests two search pages, one credit per delivered page.
63
+
64
+ ## Library and monitors
65
+
66
+ Library coverage and scoped passage search cost zero credits. Fetch any missing transcripts through a normal job first; each successful cold delivery costs one credit and joins the shared library. Search supports quoted phrases and prefix terms and requires one channel or video scope.
67
+
68
+ Coverage reports `counts.uncaptioned` for known `no_captions` or `unavailable` videos without a searchable index. Those videos are excluded from `missing` and `counts.cold`; an existing index takes precedence. `complete: true` means the source snapshot is complete and nothing indexable is missing, so it can coexist with a positive uncaptioned count. Report that count without promising captions for every video.
69
+
70
+ ```python
71
+ coverage = client.library.coverage(channel="@TED")
72
+ passages = client.library.search('"solar energy"', channel="@TED", limit=10, offset=0)
73
+ print(passages.meta.library_coverage) # indexed/known, for example "37/412"
74
+ watched = client.monitors.create(channel="@TED", auto_fetch=False)
75
+ monitors = client.monitors.list()
76
+ monitor = client.monitors.get(watched["monitor"]["id"])
77
+ client.monitors.cancel(watched["monitor"]["id"])
78
+ ```
79
+
80
+ Coverage also accepts `playlist` instead of `channel`; passage search accepts `video_id` instead of `channel`. Monitors accept either `channel` or `playlist`, plus `language`, `fallback`, and `webhook_url`. Set `auto_fetch=True` to authorize ordinary billable transcript jobs for new uploads; monitoring itself costs zero credits. The executable `examples/library.py` uses `auto_fetch=False` and cancels its monitor before exiting.
81
+
82
+ Monitor creation resolves handles on the server through the channel-resolution allowance; direct UC IDs need no resolution call. Each `monitors.create()` generates one idempotency key and reuses it for retries. To recover a creation across separate calls, pass `client.monitors.create(channel="@TED", idempotency_key="my-watch-creation")`. An explicit key must be a nonblank header-safe string of at most 128 characters; invalid keys fail locally. Keys are sent only in the `Idempotency-Key` header. A new watch returns status 201 and `idempotent: false`; replaying a key or creating an already active/paused account source returns 200 and `idempotent: true`, with the existing monitor unchanged. Replaying a key still returns that original monitor after cancellation; use a fresh key for a replacement.
83
+
84
+ ## Response metadata
85
+
86
+ Every successful JSON body is returned as an `ApiResponse` (a `dict`) whose `meta` attribute carries the header contract: `cache` (`HIT`/`MISS`), `credits_charged`, `credits_balance`, `rate_limit`, `request_id`, `library_coverage`, and `status`. It is an attribute, not a key, so it never appears in `json.dumps` or in dict iteration. Jobs expose the same object as `job.meta`, refreshed on every read.
87
+
88
+ ## Errors
89
+
90
+ API failures are subclasses of `ScriptHaulError` named exactly after the server's `errorClass`. Every exception has `error_class`, `code`, `status`, `retryable`, `retry_after` (seconds, from `Retry-After` or the envelope), `request_id`, `docs_url`, and `details`.
91
+
92
+ `NotFound` with status 404 and code `feed_not_found` means YouTube returned 404 for a well-formed channel or playlist RSS feed. Latest uploads and monitor creation raise it without automatic retries; check the source ID before trying again.
93
+
94
+ Failures the server never saw use client-only classes whose `error_class` is `None`, so a local problem is never mistaken for a relay verdict: `ClientNetworkError` (`code="network_error"`, the socket failed), `ClientTimeout` (`poll_timeout` or `job_wait_timeout`, the client's own budget ran out), and `JobPaused` (`job_paused`). All three are `retryable`.
95
+
96
+ ## Retries and polling
97
+
98
+ POST requests are retried only when they carry a valid `Idempotency-Key`; an unkeyed POST gets one transport attempt even if the connection fails or the response is retryable. Both `jobs.create()` and `monitors.create()` supply stable keys. `job.cancel()` has no key and is not automatically retried.
99
+
100
+ Retryable requests use `max_retries` (default three) for network failures, 502, 503, and retryable 429 responses. YouTube and channel search (`GET /v1/search`) share a stricter total limit of **at most one retry**, even when `max_retries` is higher, because each failed attempt can spend proxy bytes. `max_retries=0` disables that retry too. The query and continuation stay unchanged. Other GETs, including library search, retain the configured limit. A response whose envelope says `retryable: false` (every daily cap, and the per-key rpm guard) is raised immediately with `retry_after` preserved; so is one whose `Retry-After` exceeds `max_retry_after` (default 30 s). Transcript 202 responses are polled along the API's `poll_url` (same origin only) up to `max_polls` times.
101
+
102
+ `job.wait()` polls until the job is `completed`, `failed`, or `cancelled`. A `paused` job stops the wait: `InsufficientCredits` (`code="job_paused_insufficient_credits"`, with `pause_reason`, `credits`, and `job` in `details`) when the account must buy credits, otherwise `JobPaused`. Pass `wait_through_pauses=True` (per call or on the client) to keep polling instead. `job.cancel()` releases the unsettled reservation.
103
+
104
+ ## Key handling
105
+
106
+ `repr(client)`, `str(client)`, and `vars(client)` show only the 12-character display prefix (`client.key_prefix`); the full key is reachable as `client.api_key` and is sent solely as `Authorization: Bearer` to `base_url`. The client refuses cross-origin poll URLs and never follows redirects.
@@ -0,0 +1,6 @@
1
+ scripthaul/__init__.py,sha256=JkQu69UO-nekxQZb44D2-EbVcKgtBj1HgCXyYlyLqFk,26168
2
+ scripthaul-0.2.0.dist-info/licenses/LICENSE,sha256=k47RN9xmUUjuptrfDg7VWo3xOMsUWYSAnvbaVgCZgKk,1067
3
+ scripthaul-0.2.0.dist-info/METADATA,sha256=BE-ohSX-I4mv8ynJhj9SaIWtGhhmuec6Zgpg4biQcdA,9401
4
+ scripthaul-0.2.0.dist-info/WHEEL,sha256=YVMoNqKzERt-wjUZwJ33xBGAwnFl-4cqbYkTtWa4itE,91
5
+ scripthaul-0.2.0.dist-info/top_level.txt,sha256=Gb7XdheFHbLiUChDpeQj4lx-DAZaCE3Q-24bMMuDSFY,11
6
+ scripthaul-0.2.0.dist-info/RECORD,,
@@ -0,0 +1,5 @@
1
+ Wheel-Version: 1.0
2
+ Generator: setuptools (84.0.0)
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any
5
+
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 ScriptHaul
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1 @@
1
+ scripthaul