scripthaul 0.2.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 ScriptHaul
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,106 @@
1
+ Metadata-Version: 2.4
2
+ Name: scripthaul
3
+ Version: 0.2.0
4
+ Summary: Thin Python client for the ScriptHaul transcript API
5
+ Author-email: ScriptHaul <support@scripthaul.com>
6
+ License-Expression: MIT
7
+ Project-URL: Homepage, https://api.scripthaul.com/docs
8
+ Project-URL: Documentation, https://api.scripthaul.com/docs
9
+ Project-URL: Support, https://api.scripthaul.com/docs
10
+ Project-URL: Repository, https://github.com/scripthaul/scripthaul-python
11
+ Project-URL: Issues, https://github.com/scripthaul/scripthaul-python/issues
12
+ Keywords: youtube,transcript,captions,api
13
+ Classifier: Programming Language :: Python :: 3
14
+ Requires-Python: >=3.9
15
+ Description-Content-Type: text/markdown
16
+ License-File: LICENSE
17
+ Dynamic: license-file
18
+
19
+ # ScriptHaul for Python
20
+
21
+ Dependency-free, synchronous ScriptHaul API client for Python 3.9+. MIT licensed.
22
+
23
+ ```sh
24
+ pip install scripthaul
25
+ ```
26
+
27
+ Until the first PyPI release, install straight from GitHub: `pip install git+https://github.com/scripthaul/scripthaul-python`. Issues and pull requests are welcome in this repository; the API itself is documented at [api.scripthaul.com/docs](https://api.scripthaul.com/docs).
28
+
29
+ ```python
30
+ from scripthaul import ScriptHaul
31
+
32
+ client = ScriptHaul(api_key="sh_live_...")
33
+ transcript = client.transcript("dQw4w9WgXcQ")
34
+ print(transcript.meta.cache, transcript.meta.credits_charged) # HIT 0
35
+ videos = client.videos("https://www.youtube.com/@example")
36
+ job = client.jobs.create(input="https://www.youtube.com/@example").wait()
37
+ ```
38
+
39
+ ## Transcripts
40
+
41
+ `client.transcript(video_id, **options)` accepts `translate` (default `False`). Set `language="es", translate=True` to receive the requested language when YouTube declines to translate the captions, if it can be delivered. A transcript delivered this way costs 300 credits per started 10,000 characters of transcript (at most 1,500), on top of the delivery credit; nothing extra is charged when it cannot be delivered. Without `translate`, nothing changes.
42
+
43
+ ## Jobs
44
+
45
+ `client.jobs.create()` accepts `translate` (default `False`) alongside `input`, `language`, `fallback`, and `webhook_url`. Set `translate=True` to ask for the job's language on videos whose captions YouTube declines to translate. A transcript delivered this way costs 300 credits per started 10,000 characters of transcript (at most 1,500), on top of the delivery credit; nothing extra is charged when it cannot be delivered. These translations are charged per video when delivered, never reserved. Opted-in jobs carry `translate: true` and a `translation_credits` total; their video rows carry `translation_credits`. Without `translate`, nothing changes.
46
+
47
+ ## Discovery
48
+
49
+ Search YouTube or one channel for one credit per delivered page. Identical requests within 60 seconds are free retries. Channel resolution and the latest 15 channel or playlist entries cost zero credits; latest entries come from a feed cached for ten minutes.
50
+
51
+ ```python
52
+ page = client.search("climate science", type="video", channel="@TED")
53
+ print(page["results"], page.meta.credits_charged) # video results include cached: True/False
54
+ if page["has_more"]:
55
+ next_page = client.search("climate science", type="video", channel="@TED",
56
+ continuation=page["continuation"])
57
+ channel = client.channels.resolve("@TED")
58
+ latest = client.channels.latest(channel=channel["channel_id"])
59
+ playlist_latest = client.channels.latest(playlist=channel["uploads_playlist_id"])
60
+ ```
61
+
62
+ `type` accepts `video` (default), `channel`, or `playlist`. Keep continuation strings unchanged and repeat the same query, type, and channel. Resolve a handle first and pass its `channel_id` to latest uploads so that reading the feed needs no Data API call. `examples/discovery.py` requests two search pages, one credit per delivered page.
63
+
64
+ ## Library and monitors
65
+
66
+ Library coverage and scoped passage search cost zero credits. Fetch any missing transcripts through a normal job first; each successful cold delivery costs one credit and joins the shared library. Search supports quoted phrases and prefix terms and requires one channel or video scope.
67
+
68
+ Coverage reports `counts.uncaptioned` for known `no_captions` or `unavailable` videos without a searchable index. Those videos are excluded from `missing` and `counts.cold`; an existing index takes precedence. `complete: true` means the source snapshot is complete and nothing indexable is missing, so it can coexist with a positive uncaptioned count. Report that count without promising captions for every video.
69
+
70
+ ```python
71
+ coverage = client.library.coverage(channel="@TED")
72
+ passages = client.library.search('"solar energy"', channel="@TED", limit=10, offset=0)
73
+ print(passages.meta.library_coverage) # indexed/known, for example "37/412"
74
+ watched = client.monitors.create(channel="@TED", auto_fetch=False)
75
+ monitors = client.monitors.list()
76
+ monitor = client.monitors.get(watched["monitor"]["id"])
77
+ client.monitors.cancel(watched["monitor"]["id"])
78
+ ```
79
+
80
+ Coverage also accepts `playlist` instead of `channel`; passage search accepts `video_id` instead of `channel`. Monitors accept either `channel` or `playlist`, plus `language`, `fallback`, and `webhook_url`. Set `auto_fetch=True` to authorize ordinary billable transcript jobs for new uploads; monitoring itself costs zero credits. The executable `examples/library.py` uses `auto_fetch=False` and cancels its monitor before exiting.
81
+
82
+ Monitor creation resolves handles on the server through the channel-resolution allowance; direct UC IDs need no resolution call. Each `monitors.create()` generates one idempotency key and reuses it for retries. To recover a creation across separate calls, pass `client.monitors.create(channel="@TED", idempotency_key="my-watch-creation")`. An explicit key must be a nonblank header-safe string of at most 128 characters; invalid keys fail locally. Keys are sent only in the `Idempotency-Key` header. A new watch returns status 201 and `idempotent: false`; replaying a key or creating an already active/paused account source returns 200 and `idempotent: true`, with the existing monitor unchanged. Replaying a key still returns that original monitor after cancellation; use a fresh key for a replacement.
83
+
84
+ ## Response metadata
85
+
86
+ Every successful JSON body is returned as an `ApiResponse` (a `dict`) whose `meta` attribute carries the header contract: `cache` (`HIT`/`MISS`), `credits_charged`, `credits_balance`, `rate_limit`, `request_id`, `library_coverage`, and `status`. It is an attribute, not a key, so it never appears in `json.dumps` or in dict iteration. Jobs expose the same object as `job.meta`, refreshed on every read.
87
+
88
+ ## Errors
89
+
90
+ API failures are subclasses of `ScriptHaulError` named exactly after the server's `errorClass`. Every exception has `error_class`, `code`, `status`, `retryable`, `retry_after` (seconds, from `Retry-After` or the envelope), `request_id`, `docs_url`, and `details`.
91
+
92
+ `NotFound` with status 404 and code `feed_not_found` means YouTube returned 404 for a well-formed channel or playlist RSS feed. Latest uploads and monitor creation raise it without automatic retries; check the source ID before trying again.
93
+
94
+ Failures the server never saw use client-only classes whose `error_class` is `None`, so a local problem is never mistaken for a relay verdict: `ClientNetworkError` (`code="network_error"`, the socket failed), `ClientTimeout` (`poll_timeout` or `job_wait_timeout`, the client's own budget ran out), and `JobPaused` (`job_paused`). All three are `retryable`.
95
+
96
+ ## Retries and polling
97
+
98
+ POST requests are retried only when they carry a valid `Idempotency-Key`; an unkeyed POST gets one transport attempt even if the connection fails or the response is retryable. Both `jobs.create()` and `monitors.create()` supply stable keys. `job.cancel()` has no key and is not automatically retried.
99
+
100
+ Retryable requests use `max_retries` (default three) for network failures, 502, 503, and retryable 429 responses. YouTube and channel search (`GET /v1/search`) share a stricter total limit of **at most one retry**, even when `max_retries` is higher, because each failed attempt can spend proxy bytes. `max_retries=0` disables that retry too. The query and continuation stay unchanged. Other GETs, including library search, retain the configured limit. A response whose envelope says `retryable: false` (every daily cap, and the per-key rpm guard) is raised immediately with `retry_after` preserved; so is one whose `Retry-After` exceeds `max_retry_after` (default 30 s). Transcript 202 responses are polled along the API's `poll_url` (same origin only) up to `max_polls` times.
101
+
102
+ `job.wait()` polls until the job is `completed`, `failed`, or `cancelled`. A `paused` job stops the wait: `InsufficientCredits` (`code="job_paused_insufficient_credits"`, with `pause_reason`, `credits`, and `job` in `details`) when the account must buy credits, otherwise `JobPaused`. Pass `wait_through_pauses=True` (per call or on the client) to keep polling instead. `job.cancel()` releases the unsettled reservation.
103
+
104
+ ## Key handling
105
+
106
+ `repr(client)`, `str(client)`, and `vars(client)` show only the 12-character display prefix (`client.key_prefix`); the full key is reachable as `client.api_key` and is sent solely as `Authorization: Bearer` to `base_url`. The client refuses cross-origin poll URLs and never follows redirects.
@@ -0,0 +1,88 @@
1
+ # ScriptHaul for Python
2
+
3
+ Dependency-free, synchronous ScriptHaul API client for Python 3.9+. MIT licensed.
4
+
5
+ ```sh
6
+ pip install scripthaul
7
+ ```
8
+
9
+ Until the first PyPI release, install straight from GitHub: `pip install git+https://github.com/scripthaul/scripthaul-python`. Issues and pull requests are welcome in this repository; the API itself is documented at [api.scripthaul.com/docs](https://api.scripthaul.com/docs).
10
+
11
+ ```python
12
+ from scripthaul import ScriptHaul
13
+
14
+ client = ScriptHaul(api_key="sh_live_...")
15
+ transcript = client.transcript("dQw4w9WgXcQ")
16
+ print(transcript.meta.cache, transcript.meta.credits_charged) # HIT 0
17
+ videos = client.videos("https://www.youtube.com/@example")
18
+ job = client.jobs.create(input="https://www.youtube.com/@example").wait()
19
+ ```
20
+
21
+ ## Transcripts
22
+
23
+ `client.transcript(video_id, **options)` accepts `translate` (default `False`). Set `language="es", translate=True` to receive the requested language when YouTube declines to translate the captions, if it can be delivered. A transcript delivered this way costs 300 credits per started 10,000 characters of transcript (at most 1,500), on top of the delivery credit; nothing extra is charged when it cannot be delivered. Without `translate`, nothing changes.
24
+
25
+ ## Jobs
26
+
27
+ `client.jobs.create()` accepts `translate` (default `False`) alongside `input`, `language`, `fallback`, and `webhook_url`. Set `translate=True` to ask for the job's language on videos whose captions YouTube declines to translate. A transcript delivered this way costs 300 credits per started 10,000 characters of transcript (at most 1,500), on top of the delivery credit; nothing extra is charged when it cannot be delivered. These translations are charged per video when delivered, never reserved. Opted-in jobs carry `translate: true` and a `translation_credits` total; their video rows carry `translation_credits`. Without `translate`, nothing changes.
28
+
29
+ ## Discovery
30
+
31
+ Search YouTube or one channel for one credit per delivered page. Identical requests within 60 seconds are free retries. Channel resolution and the latest 15 channel or playlist entries cost zero credits; latest entries come from a feed cached for ten minutes.
32
+
33
+ ```python
34
+ page = client.search("climate science", type="video", channel="@TED")
35
+ print(page["results"], page.meta.credits_charged) # video results include cached: True/False
36
+ if page["has_more"]:
37
+ next_page = client.search("climate science", type="video", channel="@TED",
38
+ continuation=page["continuation"])
39
+ channel = client.channels.resolve("@TED")
40
+ latest = client.channels.latest(channel=channel["channel_id"])
41
+ playlist_latest = client.channels.latest(playlist=channel["uploads_playlist_id"])
42
+ ```
43
+
44
+ `type` accepts `video` (default), `channel`, or `playlist`. Keep continuation strings unchanged and repeat the same query, type, and channel. Resolve a handle first and pass its `channel_id` to latest uploads so that reading the feed needs no Data API call. `examples/discovery.py` requests two search pages, one credit per delivered page.
45
+
46
+ ## Library and monitors
47
+
48
+ Library coverage and scoped passage search cost zero credits. Fetch any missing transcripts through a normal job first; each successful cold delivery costs one credit and joins the shared library. Search supports quoted phrases and prefix terms and requires one channel or video scope.
49
+
50
+ Coverage reports `counts.uncaptioned` for known `no_captions` or `unavailable` videos without a searchable index. Those videos are excluded from `missing` and `counts.cold`; an existing index takes precedence. `complete: true` means the source snapshot is complete and nothing indexable is missing, so it can coexist with a positive uncaptioned count. Report that count without promising captions for every video.
51
+
52
+ ```python
53
+ coverage = client.library.coverage(channel="@TED")
54
+ passages = client.library.search('"solar energy"', channel="@TED", limit=10, offset=0)
55
+ print(passages.meta.library_coverage) # indexed/known, for example "37/412"
56
+ watched = client.monitors.create(channel="@TED", auto_fetch=False)
57
+ monitors = client.monitors.list()
58
+ monitor = client.monitors.get(watched["monitor"]["id"])
59
+ client.monitors.cancel(watched["monitor"]["id"])
60
+ ```
61
+
62
+ Coverage also accepts `playlist` instead of `channel`; passage search accepts `video_id` instead of `channel`. Monitors accept either `channel` or `playlist`, plus `language`, `fallback`, and `webhook_url`. Set `auto_fetch=True` to authorize ordinary billable transcript jobs for new uploads; monitoring itself costs zero credits. The executable `examples/library.py` uses `auto_fetch=False` and cancels its monitor before exiting.
63
+
64
+ Monitor creation resolves handles on the server through the channel-resolution allowance; direct UC IDs need no resolution call. Each `monitors.create()` generates one idempotency key and reuses it for retries. To recover a creation across separate calls, pass `client.monitors.create(channel="@TED", idempotency_key="my-watch-creation")`. An explicit key must be a nonblank header-safe string of at most 128 characters; invalid keys fail locally. Keys are sent only in the `Idempotency-Key` header. A new watch returns status 201 and `idempotent: false`; replaying a key or creating an already active/paused account source returns 200 and `idempotent: true`, with the existing monitor unchanged. Replaying a key still returns that original monitor after cancellation; use a fresh key for a replacement.
65
+
66
+ ## Response metadata
67
+
68
+ Every successful JSON body is returned as an `ApiResponse` (a `dict`) whose `meta` attribute carries the header contract: `cache` (`HIT`/`MISS`), `credits_charged`, `credits_balance`, `rate_limit`, `request_id`, `library_coverage`, and `status`. It is an attribute, not a key, so it never appears in `json.dumps` or in dict iteration. Jobs expose the same object as `job.meta`, refreshed on every read.
69
+
70
+ ## Errors
71
+
72
+ API failures are subclasses of `ScriptHaulError` named exactly after the server's `errorClass`. Every exception has `error_class`, `code`, `status`, `retryable`, `retry_after` (seconds, from `Retry-After` or the envelope), `request_id`, `docs_url`, and `details`.
73
+
74
+ `NotFound` with status 404 and code `feed_not_found` means YouTube returned 404 for a well-formed channel or playlist RSS feed. Latest uploads and monitor creation raise it without automatic retries; check the source ID before trying again.
75
+
76
+ Failures the server never saw use client-only classes whose `error_class` is `None`, so a local problem is never mistaken for a relay verdict: `ClientNetworkError` (`code="network_error"`, the socket failed), `ClientTimeout` (`poll_timeout` or `job_wait_timeout`, the client's own budget ran out), and `JobPaused` (`job_paused`). All three are `retryable`.
77
+
78
+ ## Retries and polling
79
+
80
+ POST requests are retried only when they carry a valid `Idempotency-Key`; an unkeyed POST gets one transport attempt even if the connection fails or the response is retryable. Both `jobs.create()` and `monitors.create()` supply stable keys. `job.cancel()` has no key and is not automatically retried.
81
+
82
+ Retryable requests use `max_retries` (default three) for network failures, 502, 503, and retryable 429 responses. YouTube and channel search (`GET /v1/search`) share a stricter total limit of **at most one retry**, even when `max_retries` is higher, because each failed attempt can spend proxy bytes. `max_retries=0` disables that retry too. The query and continuation stay unchanged. Other GETs, including library search, retain the configured limit. A response whose envelope says `retryable: false` (every daily cap, and the per-key rpm guard) is raised immediately with `retry_after` preserved; so is one whose `Retry-After` exceeds `max_retry_after` (default 30 s). Transcript 202 responses are polled along the API's `poll_url` (same origin only) up to `max_polls` times.
83
+
84
+ `job.wait()` polls until the job is `completed`, `failed`, or `cancelled`. A `paused` job stops the wait: `InsufficientCredits` (`code="job_paused_insufficient_credits"`, with `pause_reason`, `credits`, and `job` in `details`) when the account must buy credits, otherwise `JobPaused`. Pass `wait_through_pauses=True` (per call or on the client) to keep polling instead. `job.cancel()` releases the unsettled reservation.
85
+
86
+ ## Key handling
87
+
88
+ `repr(client)`, `str(client)`, and `vars(client)` show only the 12-character display prefix (`client.key_prefix`); the full key is reachable as `client.api_key` and is sent solely as `Authorization: Bearer` to `base_url`. The client refuses cross-origin poll URLs and never follows redirects.
@@ -0,0 +1,30 @@
1
+ [build-system]
2
+ # setuptools 77 is the first release that understands the PEP 639 SPDX
3
+ # `license` string and `license-files` below.
4
+ requires = ["setuptools>=77"]
5
+ build-backend = "setuptools.build_meta"
6
+
7
+ [project]
8
+ name = "scripthaul"
9
+ version = "0.2.0"
10
+ description = "Thin Python client for the ScriptHaul transcript API"
11
+ readme = "README.md"
12
+ requires-python = ">=3.9"
13
+ license = "MIT"
14
+ license-files = ["LICENSE"]
15
+ authors = [{ name = "ScriptHaul", email = "support@scripthaul.com" }]
16
+ keywords = ["youtube", "transcript", "captions", "api"]
17
+ classifiers = [
18
+ "Programming Language :: Python :: 3",
19
+ ]
20
+ dependencies = []
21
+
22
+ [project.urls]
23
+ Homepage = "https://api.scripthaul.com/docs"
24
+ Documentation = "https://api.scripthaul.com/docs"
25
+ Support = "https://api.scripthaul.com/docs"
26
+ Repository = "https://github.com/scripthaul/scripthaul-python"
27
+ Issues = "https://github.com/scripthaul/scripthaul-python/issues"
28
+
29
+ [tool.setuptools.packages.find]
30
+ where = ["src"]
@@ -0,0 +1,4 @@
1
+ [egg_info]
2
+ tag_build =
3
+ tag_date = 0
4
+