yt-briefing 0.10.0 → 0.11.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude/skills/yt-search/SKILL.md +3 -3
- package/dist/yt-search.js +30 -12
- package/dist/yt-transcript.js +21 -1
- package/package.json +1 -1
|
@@ -1,12 +1,12 @@
|
|
|
1
1
|
---
|
|
2
2
|
name: yt-search
|
|
3
3
|
description: Search WITHIN one YouTube channel by intent — name a channel and what you're after; the engine lists that channel's uploads, ranks them against your intent (metadata only, no transcript yet), then lazily yields ONE matching video at a time with a rich summary. You keep or skip each; at the end it synthesizes a comparison from everything you kept. Channel-scoped, not whole-YouTube. Same transcript engine + proxy as /yt; lazy on purpose (no transcript bursts → no IP block). Summaries and prompts use the language chosen at onboarding.
|
|
4
|
-
argument-hint: A channel (@handle or URL) and a descriptive intent, e.g. "@t3dotgg which terminal for AI coding". Optional --top N (default 10)
|
|
4
|
+
argument-hint: A channel (@handle or URL) and a descriptive intent, e.g. "@t3dotgg which terminal for AI coding". Searches the channel's whole upload history. Optional --top N (default 10).
|
|
5
5
|
---
|
|
6
6
|
|
|
7
7
|
## How it works
|
|
8
8
|
|
|
9
|
-
`src/yt-search.ts` is the whole engine: list
|
|
9
|
+
`src/yt-search.ts` is the whole engine: list the channel's **full upload history** (cheap — `playlistItems`, ~1 quota unit/page; NOT `search.list`) → re-rank it against your intent on metadata only (title/description, no transcript; ranked in chunks and merged so a 1000+ video channel never overflows the LLM) → **lazy** one-candidate-at-a-time yield with a rich summary → record keep/skip → on demand synthesize a comparison from everything kept. **Channel-scoped on purpose** — you choose where to look; it does NOT search all of YouTube. Matching is descriptive: the LLM filters the channel's videos by intent. This skill is a thin loop — paste the summary, collect keep/skip, show the final comparison.
|
|
10
10
|
|
|
11
11
|
**Lazy on purpose:** one transcript per step, never a burst — a burst looks like scraping and gets the IP blocked (same reason `/yt` is lazy). Run the engine bare — stdout is a single JSON line, stderr empty; never redirect.
|
|
12
12
|
|
|
@@ -53,5 +53,5 @@ while true:
|
|
|
53
53
|
|
|
54
54
|
- **Verbatim:** paste `summary` and `comparison` exactly as returned; never paste a raw transcript.
|
|
55
55
|
- **Language:** question text + option descriptions follow `output_lang`; button labels stay `Keep` / `Skip`.
|
|
56
|
-
- **Scope:** one channel per search
|
|
56
|
+
- **Scope:** one channel per search, ranked across its **whole upload history**. Listing is cheap; the cost is the lazy transcript fetches, so let the user keep/skip rather than pulling everything. A bare resume (no `--reset`) continues the same ranked queue. `--top N` (default 10) caps how many of the top re-ranked matches get triaged — it is the only flag.
|
|
57
57
|
- **Stateless triage:** independent of `/yt` (no channel profiles, no ratings written). For the recurring multi-channel briefing use `/yt`; for one known video use `/yt-transcribe`.
|
package/dist/yt-search.js
CHANGED
|
@@ -5,8 +5,9 @@
|
|
|
5
5
|
*
|
|
6
6
|
* You point it at a channel and describe what you're after ("which terminal does he recommend
|
|
7
7
|
* for AI coding"); the engine:
|
|
8
|
-
* 1. lists that channel's
|
|
9
|
-
* 2. re-ranks
|
|
8
|
+
* 1. lists that channel's FULL upload history (cheap — playlistItems, 1 quota unit/page; NOT search.list),
|
|
9
|
+
* 2. re-ranks it against your intent on metadata only — title/description, NO transcript yet
|
|
10
|
+
* (the history is ranked in chunks and merged, so a 1000+ video channel never overflows the LLM),
|
|
10
11
|
* 3. yields ONE matching video at a time with a rich summary, lazily — never a burst of
|
|
11
12
|
* transcript fetches (a burst looks like scraping and gets the IP blocked),
|
|
12
13
|
* 4. records your keep/skip decision; kept summaries accumulate in a cache,
|
|
@@ -16,7 +17,7 @@
|
|
|
16
17
|
* the channel's videos by intent (no exact-keyword needed).
|
|
17
18
|
*
|
|
18
19
|
* Usage (the skill / CLI drives these; one JSON line per call):
|
|
19
|
-
* yt-search "<intent>" --channel <@handle|url> [--reset] [--top N] [--
|
|
20
|
+
* yt-search "<intent>" --channel <@handle|url> [--reset] [--top N] [--lang auto]
|
|
20
21
|
* yt-search --keep record the pending candidate, advance, yield next
|
|
21
22
|
* yt-search --skip drop the pending candidate, advance, yield next
|
|
22
23
|
* yt-search --compare synthesize a comparison from everything kept
|
|
@@ -45,7 +46,7 @@ mkdirSync(CACHE_DIR, { recursive: true });
|
|
|
45
46
|
const RUNTIME = process.execPath;
|
|
46
47
|
const LANG = outputLang();
|
|
47
48
|
const argv = process.argv.slice(2);
|
|
48
|
-
const VALUE_FLAGS = new Set(['--channel', '--top', '--
|
|
49
|
+
const VALUE_FLAGS = new Set(['--channel', '--top', '--lang']);
|
|
49
50
|
const has = (f) => argv.includes(f);
|
|
50
51
|
const flagVal = (f) => {
|
|
51
52
|
const i = argv.indexOf(f);
|
|
@@ -73,8 +74,6 @@ const SKIP = has('--skip');
|
|
|
73
74
|
const COMPARE = has('--compare');
|
|
74
75
|
const CHANNEL = flagVal('--channel');
|
|
75
76
|
const TOP = Math.max(1, parseInt(flagVal('--top') || '10', 10));
|
|
76
|
-
const SCAN = Math.max(1, parseInt(flagVal('--scan') || '50', 10)); // recent uploads to consider when no --since
|
|
77
|
-
const SINCE = flagVal('--since');
|
|
78
77
|
const LANGTRACK = flagVal('--lang') || 'auto';
|
|
79
78
|
function emit(obj) {
|
|
80
79
|
process.stdout.write(JSON.stringify(obj));
|
|
@@ -116,8 +115,12 @@ function parseJsonArray(out) {
|
|
|
116
115
|
return null;
|
|
117
116
|
}
|
|
118
117
|
}
|
|
119
|
-
|
|
120
|
-
|
|
118
|
+
// The whole channel history is ranked, so the pool can be large (a years-old channel is 1000+
|
|
119
|
+
// uploads). One LLM prompt would overflow context, so the pool is ranked in chunks and merged.
|
|
120
|
+
const RERANK_CHUNK = 200;
|
|
121
|
+
/** Re-rank ONE chunk against the intent (metadata only). Returns [] on a parse/LLM failure so a
|
|
122
|
+
* bad chunk is dropped rather than flooding the queue with unranked videos. */
|
|
123
|
+
async function rerankBatch(intent, items) {
|
|
121
124
|
if (items.length === 0)
|
|
122
125
|
return [];
|
|
123
126
|
const compact = items.map(h => ({ id: h.videoId, title: h.title, published: h.publishedAt, desc: (h.description || '').slice(0, 280) }));
|
|
@@ -135,7 +138,7 @@ Set keep=false for anything not relevant to the intent.`;
|
|
|
135
138
|
const out = await chat(prompt, { system: 'You output ONLY a raw JSON array as instructed.', temperature: 0 });
|
|
136
139
|
const arr = parseJsonArray(out);
|
|
137
140
|
if (!arr)
|
|
138
|
-
return
|
|
141
|
+
return [];
|
|
139
142
|
const byId = new Map(items.map(h => [h.videoId, h]));
|
|
140
143
|
const ranked = [];
|
|
141
144
|
for (const r of arr) {
|
|
@@ -145,11 +148,26 @@ Set keep=false for anything not relevant to the intent.`;
|
|
|
145
148
|
if (h)
|
|
146
149
|
ranked.push({ ...h, score: typeof r.score === 'number' ? r.score : undefined, reason: r.reason });
|
|
147
150
|
}
|
|
148
|
-
return ranked
|
|
151
|
+
return ranked;
|
|
149
152
|
}
|
|
150
153
|
catch {
|
|
151
|
-
return
|
|
154
|
+
return [];
|
|
155
|
+
}
|
|
156
|
+
}
|
|
157
|
+
/** Re-rank the channel's full history against the intent: chunk the pool, rank each chunk, merge
|
|
158
|
+
* the keepers globally by score (highest first), dedup by id. */
|
|
159
|
+
async function rerank(intent, items) {
|
|
160
|
+
if (items.length === 0)
|
|
161
|
+
return [];
|
|
162
|
+
const merged = [];
|
|
163
|
+
for (let i = 0; i < items.length; i += RERANK_CHUNK) {
|
|
164
|
+
merged.push(...await rerankBatch(intent, items.slice(i, i + RERANK_CHUNK)));
|
|
152
165
|
}
|
|
166
|
+
const byId = new Map();
|
|
167
|
+
for (const c of merged)
|
|
168
|
+
if (!byId.has(c.videoId))
|
|
169
|
+
byId.set(c.videoId, c);
|
|
170
|
+
return [...byId.values()].sort((a, b) => (b.score ?? -1) - (a.score ?? -1));
|
|
153
171
|
}
|
|
154
172
|
/** Rich, standalone summary for one candidate — the triage artifact + compare input. */
|
|
155
173
|
async function megaSummary(c, transcript, intent) {
|
|
@@ -278,7 +296,7 @@ async function main() {
|
|
|
278
296
|
emit({ status: 'error', error: `Could not read a channel handle from "${CHANNEL}" — use @name or the channel URL.` });
|
|
279
297
|
let videos;
|
|
280
298
|
try {
|
|
281
|
-
videos = await fetchChannelVideos(handle, {
|
|
299
|
+
videos = await fetchChannelVideos(handle, { limit: null, enrich: false }); // full channel history
|
|
282
300
|
}
|
|
283
301
|
catch (e) {
|
|
284
302
|
emit({ status: 'error', error: e.message });
|
package/dist/yt-transcript.js
CHANGED
|
@@ -91,6 +91,7 @@ const ytdlpArgs = [
|
|
|
91
91
|
'--sub-langs', subLangs,
|
|
92
92
|
'--sub-format', 'vtt',
|
|
93
93
|
'--ignore-errors', // continue if one language track fails (e.g. 429 on pl for en-only video)
|
|
94
|
+
'--extractor-retries', '3', // yt-dlp's own retry on the extractor-level bot-check / 429
|
|
94
95
|
'--no-warnings',
|
|
95
96
|
'--quiet',
|
|
96
97
|
'-o', outTemplate,
|
|
@@ -98,7 +99,26 @@ const ytdlpArgs = [
|
|
|
98
99
|
if (proxyUrl)
|
|
99
100
|
ytdlpArgs.push('--proxy', proxyUrl);
|
|
100
101
|
ytdlpArgs.push(`https://www.youtube.com/watch?v=${videoId}`);
|
|
101
|
-
|
|
102
|
+
// The first request through a cold WARP/shared egress IP intermittently trips YouTube's
|
|
103
|
+
// "Sign in to confirm you're not a bot" check; the very next request seconds later succeeds
|
|
104
|
+
// (the extractor session warms up). Without a retry that transient flake exits 2, which the
|
|
105
|
+
// sweep treats as fatal (emit → exit) — so a cold `/yt` needed two calls to get going. Retry
|
|
106
|
+
// ONLY on rate-limit-with-no-subtitles: a genuine IP block still exhausts the attempts and
|
|
107
|
+
// exits 2, and real tooling errors / captionless videos fall straight through untouched.
|
|
108
|
+
const sleepSync = (ms) => Atomics.wait(new Int32Array(new SharedArrayBuffer(4)), 0, 0, ms);
|
|
109
|
+
let result;
|
|
110
|
+
for (let attempt = 1; attempt <= 3; attempt++) {
|
|
111
|
+
result = spawnSync(YT_DLP, ytdlpArgs, { encoding: 'utf8', timeout: 60_000 });
|
|
112
|
+
if (result.error)
|
|
113
|
+
break; // spawn / ENOENT — retrying won't help
|
|
114
|
+
const stderr = (result.stderr ?? '').toLowerCase();
|
|
115
|
+
const limited = stderr.includes('sign in') || stderr.includes('429') || stderr.includes('too many') || stderr.includes('captcha');
|
|
116
|
+
const gotSubs = existsSync(tmpDir) && readdirSync(tmpDir).some(f => f.endsWith('.vtt'));
|
|
117
|
+
if (gotSubs || !limited)
|
|
118
|
+
break; // success, or a non-rate-limit outcome → stop
|
|
119
|
+
if (attempt < 3)
|
|
120
|
+
sleepSync(attempt * 2000); // 2s, then 4s backoff between cold retries
|
|
121
|
+
}
|
|
102
122
|
const cleanup = () => { try {
|
|
103
123
|
rmSync(tmpDir, { recursive: true });
|
|
104
124
|
}
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "yt-briefing",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.11.1",
|
|
4
4
|
"description": "A self-learning YouTube briefing engine: it sweeps the channels you follow, filters noise in two stages (title, then transcript), summarizes the rest in your language, and adapts to your ratings — one video at a time.",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"bin": {
|