yt-briefing 0.12.3 → 0.13.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude/skills/yt/SKILL.md +3 -1
- package/README.md +1 -1
- package/dist/yt-channel-pending.js +13 -3
- package/dist/yt-sweep.js +37 -15
- package/package.json +1 -1
|
@@ -22,10 +22,12 @@ Read `data/config.json` → `output_lang` once at the start of the loop. Phrase
|
|
|
22
22
|
```
|
|
23
23
|
out = JSON.parse(`bun run src/yt-sweep.ts --reset`) // Bash — first call: --reset rebuilds the queue fresh
|
|
24
24
|
while true:
|
|
25
|
+
if out.skipped: FIRST, in output_lang, tell the user one line — "Skipped {out.skipped}:" + a `; `-joined list of `{channel} «{title}» — {reason}` from out.skips (videos the title/content filter dropped on the way here, including your channel directives). Then handle out.status below.
|
|
25
26
|
out.status:
|
|
26
27
|
"done" → sweep finished — tell the user, stop
|
|
27
28
|
"error" → setup/config problem (e.g. missing API key) — show `out.error` to the user verbatim, stop
|
|
28
|
-
"rate_limited" →
|
|
29
|
+
"rate_limited" → YouTube is blocking the egress IP (429 / captcha) — tell the user, stop; recovery in README.md → Proxy
|
|
30
|
+
"tooling_error" → transcript toolchain failed (proxy down / yt-dlp / network) — tell the user, stop; check proxy health, then README.md → Proxy
|
|
29
31
|
"rating_needed" → steps A–E
|
|
30
32
|
```
|
|
31
33
|
|
package/README.md
CHANGED
|
@@ -150,7 +150,7 @@ YT_BRIEFING_LLM_MODEL=gemini-2.5-flash
|
|
|
150
150
|
|
|
151
151
|
Want something else? Change those three lines for OpenRouter (`https://openrouter.ai/api/v1`),
|
|
152
152
|
OpenAI (`https://api.openai.com/v1`), Anthropic (`https://api.anthropic.com/v1/`,
|
|
153
|
-
e.g. `claude-sonnet-
|
|
153
|
+
e.g. `claude-sonnet-5`), or a local Ollama (`http://localhost:11434/v1`). Set
|
|
154
154
|
`YT_BRIEFING_LLM_BASE_URL`, `_API_KEY`, and `_MODEL` in your root `.env` (see [Setup](#setup)).
|
|
155
155
|
|
|
156
156
|
## Why an API, not the agent's native model
|
|
@@ -2,8 +2,11 @@
|
|
|
2
2
|
/**
|
|
3
3
|
* Usage: bun src/yt-channel-pending.ts @HANDLE
|
|
4
4
|
*
|
|
5
|
-
* For one channel: reads state.md pointers
|
|
6
|
-
*
|
|
5
|
+
* For one channel: reads state.md pointers, fetches the channel's videos in-process via
|
|
6
|
+
* lib/yt-api.ts (--since a fixed lookback, NOT state.md's `updated` — that column is a
|
|
7
|
+
* write-timestamp, not the pointer video's publish date; using it as the fetch cutoff
|
|
8
|
+
* silently orphaned backlog videos on channels swept more than once per day, see
|
|
9
|
+
* CHANGELOG),
|
|
7
10
|
* filters to videos NEWER than each type's pointer (or baseline if pointer null),
|
|
8
11
|
* sorts ASC by publishedAt (process oldest first → state pointer advances monotonically),
|
|
9
12
|
* outputs JSON array: [{videoId, title, publishedAt, type, is_baseline}].
|
|
@@ -28,7 +31,14 @@ if (!row) {
|
|
|
28
31
|
console.error(`Channel ${handle} not found in state.md`);
|
|
29
32
|
process.exit(1);
|
|
30
33
|
}
|
|
31
|
-
|
|
34
|
+
// `row.updated` is when state.md was last WRITTEN (stamped to today on every bump),
|
|
35
|
+
// not when the pointer video was PUBLISHED — reusing it as the fetch cutoff shrinks the
|
|
36
|
+
// window to "today" after the first bump of the day, so a second same-day sweep can
|
|
37
|
+
// silently drop any not-yet-processed backlog between the pointer and today. A fixed
|
|
38
|
+
// lookback avoids that: the pointer-cutoff logic below still does the real filtering,
|
|
39
|
+
// this just has to be wide enough to always re-include the pointer video itself.
|
|
40
|
+
const LOOKBACK_DAYS = 45;
|
|
41
|
+
const since = new Date(Date.now() - LOOKBACK_DAYS * 24 * 60 * 60 * 1000).toISOString().slice(0, 10);
|
|
32
42
|
let videos;
|
|
33
43
|
try {
|
|
34
44
|
videos = await fetchChannelVideos(handle, { since });
|
package/dist/yt-sweep.js
CHANGED
|
@@ -77,6 +77,15 @@ const DEBUG = !!process.env.YT_BRIEFING_DEBUG;
|
|
|
77
77
|
const T0 = Date.now();
|
|
78
78
|
const log = (msg) => { if (DEBUG)
|
|
79
79
|
appendFileSync(LOG_FILE, `⏱ ${msg} (+${Date.now() - T0}ms)\n`); };
|
|
80
|
+
const skipLedger = [];
|
|
81
|
+
const recordedSkips = new Set();
|
|
82
|
+
/** Record a skip once per videoId — the fg/bg expansion race can apply a title-skip twice. */
|
|
83
|
+
function recordSkip(log) {
|
|
84
|
+
if (recordedSkips.has(log.videoId))
|
|
85
|
+
return;
|
|
86
|
+
recordedSkips.add(log.videoId);
|
|
87
|
+
skipLedger.push(log);
|
|
88
|
+
}
|
|
80
89
|
// ---------- subprocess helper ----------
|
|
81
90
|
function run(cmd) {
|
|
82
91
|
return new Promise((resolve, reject) => {
|
|
@@ -114,8 +123,10 @@ function persistSkip(item, status) {
|
|
|
114
123
|
* Idempotent — re-applying the same skip is a no-op. ONLY the foreground calls this.
|
|
115
124
|
*/
|
|
116
125
|
function applyTitleSkips(skips) {
|
|
117
|
-
for (const s of skips)
|
|
126
|
+
for (const s of skips) {
|
|
118
127
|
bumpState(s.channel, s.type, s.videoId);
|
|
128
|
+
recordSkip({ stage: 'title', channel: s.channel, videoId: s.videoId, title: s.title, reason: s.reason || 'title filter' });
|
|
129
|
+
}
|
|
119
130
|
}
|
|
120
131
|
/** Append items, skipping any videoId already queued OR already resolved (seen). */
|
|
121
132
|
function enqueue(queue, add) {
|
|
@@ -207,17 +218,18 @@ function spawnBackgroundFill() {
|
|
|
207
218
|
}
|
|
208
219
|
function emit(obj) {
|
|
209
220
|
log(`EXIT status=${obj.status}`);
|
|
210
|
-
|
|
221
|
+
const ledger = skipLedger.length ? { skipped: skipLedger.length, skips: skipLedger } : {};
|
|
222
|
+
process.stdout.write(JSON.stringify({ ...obj, ...ledger }));
|
|
211
223
|
process.exit(0);
|
|
212
224
|
}
|
|
213
225
|
// ---------- LLM gates ----------
|
|
214
226
|
/** Title filter: batch-classify a channel's non-baseline titles. Falls back to keep-all on any error. */
|
|
215
227
|
async function runTitleFilter(profilePathAbs, videos) {
|
|
216
|
-
const skip = new
|
|
228
|
+
const skip = new Map();
|
|
217
229
|
if (!existsSync(profilePathAbs))
|
|
218
230
|
return skip;
|
|
219
231
|
const profile = readFileSync(profilePathAbs, 'utf8');
|
|
220
|
-
if (!/##\s*Skip titles/.test(profile))
|
|
232
|
+
if (!/##\s*(Skip titles|Notes|Channel policy)/.test(profile))
|
|
221
233
|
return skip;
|
|
222
234
|
const toClassify = videos.filter(v => !v.is_baseline);
|
|
223
235
|
if (toClassify.length === 0)
|
|
@@ -227,8 +239,8 @@ async function runTitleFilter(profilePathAbs, videos) {
|
|
|
227
239
|
Channel profile:
|
|
228
240
|
${profile}
|
|
229
241
|
|
|
230
|
-
|
|
231
|
-
Keep by default
|
|
242
|
+
Skip a title when it clearly matches a skip rule from the profile — a '## Skip titles' example, or a skip/reject directive in '## Notes' or '## Channel policy' that is judgeable from the title alone (defer transcript-dependent rejections to the content stage).
|
|
243
|
+
Keep by default: skip only on a clear match, and when unsure, keep.
|
|
232
244
|
|
|
233
245
|
Videos:
|
|
234
246
|
${JSON.stringify(toClassify.map(v => ({ id: v.videoId, title: v.title, type: v.type })))}
|
|
@@ -253,7 +265,7 @@ Output ONLY a raw JSON array (no markdown fences, no explanation):
|
|
|
253
265
|
const parsed = JSON.parse(out.slice(start, end + 1));
|
|
254
266
|
for (const r of parsed)
|
|
255
267
|
if (r.result === 'skip')
|
|
256
|
-
skip.
|
|
268
|
+
skip.set(r.id, (r.reason ?? '').trim());
|
|
257
269
|
}
|
|
258
270
|
catch { /* keep-all */ }
|
|
259
271
|
return skip;
|
|
@@ -263,7 +275,7 @@ async function runContentFilter(item, transcript) {
|
|
|
263
275
|
const profile = existsSync(item.profile_path) ? readFileSync(item.profile_path, 'utf8') : '';
|
|
264
276
|
const baselineNote = item.is_baseline ? ' · baseline' : '';
|
|
265
277
|
const profileSection = profile
|
|
266
|
-
? `\nChannel
|
|
278
|
+
? `\nChannel directives — standing instructions the user set for THIS channel (sections: Channel policy, Summary format, Cut sections, Episode types, Notes). Carry them out as written: a directive may reject the video outright (see the skip check in step 1), or — for a video that passes — add a section, shift emphasis or tone, or scrutinize/flag specific things within the briefing format (never changing its skeleton). If a directive asks you to verify or scrutinize claims, assess them against your own background knowledge: state what you can corroborate, what looks overstated, cherry-picked or misattributed, and what you cannot place. Always mark your confidence, and say plainly this is a from-memory assessment, not live verification — for anything recent or outside your knowledge, flag that instead of guessing. Keep "what the video claims" separate from your own assessment.\n${profile}\n`
|
|
267
279
|
: '';
|
|
268
280
|
const prompt = `Write a summary of this YouTube video in ${LANG}, OR return 'OFFTOPIC: <reason>' if the transcript clearly does not match what the title/channel promises.
|
|
269
281
|
|
|
@@ -279,17 +291,18 @@ Transcript:
|
|
|
279
291
|
${transcript}
|
|
280
292
|
${profileSection}
|
|
281
293
|
Steps:
|
|
282
|
-
1. Substance check:
|
|
294
|
+
1. Substance / skip check — output ONLY 'OFFTOPIC: <short reason>' and stop if EITHER (a) the transcript clearly does not deliver what the title promises, OR (b) a channel directive clearly instructs rejecting this kind of content and the transcript clearly meets that rejection criterion. When unsure, keep. Otherwise continue.
|
|
283
295
|
2. Otherwise write the summary:
|
|
284
296
|
- Header: ### ${item.channel} — "${item.title}"
|
|
285
297
|
- Subtitle: _${item.publishedAt} · ${item.type} · https://youtube.com/watch?v=${item.videoId}${baselineNote}_
|
|
286
298
|
- 2-5 numbered thematic sections × 2-5 sentences each
|
|
287
299
|
- At most 5-8 short quotes from the transcript
|
|
288
300
|
- No timestamps
|
|
289
|
-
3.
|
|
290
|
-
4.
|
|
301
|
+
3. Apply the channel directives above, if any — they shape the briefing's content, emphasis, and tone, but not the header/subtitle/section skeleton. Put any directive-driven assessment in its own short, clearly-labeled section.
|
|
302
|
+
4. Language: natural ${LANG}. Avoid calques/anglicisms; use foreign words only for proper nouns or established technical terms. Section headers should be verb phrases, not noun stacks.
|
|
303
|
+
5. Output: ONLY the briefing itself OR 'OFFTOPIC: ...'. No preamble, and no meta-commentary about these instructions.`;
|
|
291
304
|
return await chat(prompt, {
|
|
292
|
-
system: `You
|
|
305
|
+
system: `You write channel briefings in ${LANG}, following the task instructions and the channel's standing directives exactly. Output only the briefing or 'OFFTOPIC: <reason>' — no preamble, no meta-commentary about the instructions.`,
|
|
293
306
|
model: getModel(),
|
|
294
307
|
});
|
|
295
308
|
}
|
|
@@ -321,7 +334,7 @@ async function expandChannel(ref) {
|
|
|
321
334
|
const skips = [];
|
|
322
335
|
for (const it of candidates) {
|
|
323
336
|
if (titleSkip.has(it.videoId))
|
|
324
|
-
skips.push({ channel: it.channel, type: it.type, videoId: it.videoId });
|
|
337
|
+
skips.push({ channel: it.channel, type: it.type, videoId: it.videoId, title: it.title, reason: titleSkip.get(it.videoId) || 'title filter' });
|
|
325
338
|
else
|
|
326
339
|
items.push(it);
|
|
327
340
|
}
|
|
@@ -418,15 +431,19 @@ async function processItem(item) {
|
|
|
418
431
|
log(`transcript ${item.videoId} ${Date.now() - tT}ms (exit ${t.code})`);
|
|
419
432
|
if (t.code === 2)
|
|
420
433
|
return { kind: 'rate_limited' };
|
|
434
|
+
// exit 3 = tooling/proxy failure (503, tunnel down, missing yt-dlp) — NOT a missing-captions
|
|
435
|
+
// case. Surface it distinctly so the operator looks at the toolchain, not at YouTube/IP.
|
|
436
|
+
if (t.code === 3)
|
|
437
|
+
return { kind: 'tooling_error' };
|
|
421
438
|
if (t.code !== 0)
|
|
422
439
|
return { kind: 'skip', status: 'no_transcript' };
|
|
423
440
|
if (!t.stdout.trim())
|
|
424
|
-
return { kind: 'skip', status: 'content_skip' };
|
|
441
|
+
return { kind: 'skip', status: 'content_skip', reason: 'empty transcript' };
|
|
425
442
|
const tC = Date.now();
|
|
426
443
|
const summary = await runContentFilter(item, t.stdout);
|
|
427
444
|
log(`content ${item.videoId} ${Date.now() - tC}ms`);
|
|
428
445
|
if (summary.startsWith('OFFTOPIC:'))
|
|
429
|
-
return { kind: 'skip', status: 'content_skip' };
|
|
446
|
+
return { kind: 'skip', status: 'content_skip', reason: summary.slice(9).trim() };
|
|
430
447
|
return { kind: 'ratable', summary };
|
|
431
448
|
}
|
|
432
449
|
async function advance(queue) {
|
|
@@ -454,8 +471,13 @@ async function advance(queue) {
|
|
|
454
471
|
writeFileSync(QUEUE_FILE, JSON.stringify(queue));
|
|
455
472
|
emit({ status: 'rate_limited' });
|
|
456
473
|
}
|
|
474
|
+
if (result.kind === 'tooling_error') {
|
|
475
|
+
writeFileSync(QUEUE_FILE, JSON.stringify(queue));
|
|
476
|
+
emit({ status: 'tooling_error' });
|
|
477
|
+
}
|
|
457
478
|
if (result.kind === 'skip') {
|
|
458
479
|
persistSkip(item, result.status);
|
|
480
|
+
recordSkip({ stage: 'content', channel: item.channel, videoId: item.videoId, title: item.title, reason: result.reason ?? (result.status === 'no_transcript' ? 'no transcript' : 'off-topic') });
|
|
459
481
|
dropHead(queue);
|
|
460
482
|
continue;
|
|
461
483
|
}
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "yt-briefing",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.13.1",
|
|
4
4
|
"description": "A self-learning YouTube briefing engine: it sweeps the channels you follow, filters noise in two stages (title, then transcript), summarizes the rest in your language, and adapts to your ratings — one video at a time.",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"bin": {
|