yt-briefing 0.12.2 → 0.13.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude/skills/yt/SKILL.md +3 -1
- package/dist/lib/yt-lib.js +10 -0
- package/dist/yt-rating.js +15 -1
- package/dist/yt-sweep.js +42 -19
- package/package.json +1 -1
|
@@ -22,10 +22,12 @@ Read `data/config.json` → `output_lang` once at the start of the loop. Phrase
|
|
|
22
22
|
```
|
|
23
23
|
out = JSON.parse(`bun run src/yt-sweep.ts --reset`) // Bash — first call: --reset rebuilds the queue fresh
|
|
24
24
|
while true:
|
|
25
|
+
if out.skipped: FIRST, in output_lang, tell the user one line — "Skipped {out.skipped}:" + a `; `-joined list of `{channel} «{title}» — {reason}` from out.skips (videos the title/content filter dropped on the way here, including your channel directives). Then handle out.status below.
|
|
25
26
|
out.status:
|
|
26
27
|
"done" → sweep finished — tell the user, stop
|
|
27
28
|
"error" → setup/config problem (e.g. missing API key) — show `out.error` to the user verbatim, stop
|
|
28
|
-
"rate_limited" →
|
|
29
|
+
"rate_limited" → YouTube is blocking the egress IP (429 / captcha) — tell the user, stop; recovery in README.md → Proxy
|
|
30
|
+
"tooling_error" → transcript toolchain failed (proxy down / yt-dlp / network) — tell the user, stop; check proxy health, then README.md → Proxy
|
|
29
31
|
"rating_needed" → steps A–E
|
|
30
32
|
```
|
|
31
33
|
|
package/dist/lib/yt-lib.js
CHANGED
|
@@ -87,6 +87,16 @@ export function bumpStatePointer(content, handle, type, videoId, date) {
|
|
|
87
87
|
throw new Error(`bumpStatePointer: row for ${handle} not found`);
|
|
88
88
|
return bumpStateFrontmatterDate(lines.join('\n'), date);
|
|
89
89
|
}
|
|
90
|
+
/**
|
|
91
|
+
* Is this video already resolved for the current run? True when the per-(channel,type)
|
|
92
|
+
* state pointer has landed on it, OR it is in the run's `seen` set. The `seen` arm is the
|
|
93
|
+
* race-proof guard: `bumpStatePointer` is not monotonic, so a detached `--fill` re-bumping
|
|
94
|
+
* a stale pointer can transiently regress it and make a pointer-only check re-emit an
|
|
95
|
+
* already-rated video. `seen` is authoritative and pointer-independent.
|
|
96
|
+
*/
|
|
97
|
+
export function isResolved(pointer, videoId, seen) {
|
|
98
|
+
return pointer === videoId || seen.includes(videoId);
|
|
99
|
+
}
|
|
90
100
|
// ---------- channel profile: durable signal writes ----------
|
|
91
101
|
/**
|
|
92
102
|
* Append `line` under `## <section>`, creating the section if missing (before `## Notes`
|
package/dist/yt-rating.js
CHANGED
|
@@ -19,7 +19,7 @@
|
|
|
19
19
|
import { readFileSync, writeFileSync, existsSync } from 'fs';
|
|
20
20
|
import { loadEnv } from "./lib/env.js";
|
|
21
21
|
import { parseChannels, appendSkipTitle, appendNote, bumpStatePointer } from "./lib/yt-lib.js";
|
|
22
|
-
import { CHANNELS_MD, STATE_MD, PENDING_FILE, profilePath } from "./lib/paths.js";
|
|
22
|
+
import { CHANNELS_MD, STATE_MD, PENDING_FILE, QUEUE_FILE, profilePath } from "./lib/paths.js";
|
|
23
23
|
loadEnv();
|
|
24
24
|
function getArg(args, name) {
|
|
25
25
|
const idx = args.indexOf(name);
|
|
@@ -103,6 +103,20 @@ if (!args.noState) {
|
|
|
103
103
|
stateBumped = true;
|
|
104
104
|
}
|
|
105
105
|
}
|
|
106
|
+
// 3. Mark the rated video resolved in the run queue (`seen`) — independent of the state
|
|
107
|
+
// pointer, so a cursor-regression race in yt-sweep (a detached `--fill` re-bumping a
|
|
108
|
+
// stale pointer) can't re-emit an already-rated video. Best-effort: the queue is a
|
|
109
|
+
// same-day throwaway and no sweep process writes it while the user rates.
|
|
110
|
+
if (!args.noState && existsSync(QUEUE_FILE)) {
|
|
111
|
+
try {
|
|
112
|
+
const q = JSON.parse(readFileSync(QUEUE_FILE, 'utf8'));
|
|
113
|
+
if (q?.built_at === date && Array.isArray(q.seen) && !q.seen.includes(args.id)) {
|
|
114
|
+
q.seen.push(args.id);
|
|
115
|
+
writeFileSync(QUEUE_FILE, JSON.stringify(q), 'utf8');
|
|
116
|
+
}
|
|
117
|
+
}
|
|
118
|
+
catch { /* corrupt / foreign queue → ignore; the next sweep rebuilds it */ }
|
|
119
|
+
}
|
|
106
120
|
console.log(JSON.stringify({
|
|
107
121
|
ok: true,
|
|
108
122
|
profile: `channels/${ch.slug}.md`,
|
package/dist/yt-sweep.js
CHANGED
|
@@ -50,7 +50,7 @@
|
|
|
50
50
|
import { readFileSync, writeFileSync, existsSync, rmSync, mkdirSync, renameSync, appendFileSync } from 'node:fs';
|
|
51
51
|
import { spawn } from 'node:child_process';
|
|
52
52
|
import { loadEnv, missingEnv, missingEnvMessage, REQUIRED_LLM, REQUIRED_YOUTUBE } from "./lib/env.js";
|
|
53
|
-
import { parseChannels, parseState, bumpStatePointer } from "./lib/yt-lib.js";
|
|
53
|
+
import { parseChannels, parseState, bumpStatePointer, isResolved } from "./lib/yt-lib.js";
|
|
54
54
|
import { chat, getModel } from "./lib/llm.js";
|
|
55
55
|
import { outputLang } from "./lib/config.js";
|
|
56
56
|
import { PKG_ROOT, CHANNELS_MD, STATE_MD, CACHE_DIR, QUEUE_FILE, REST_FILE, PENDING_FILE, PREFETCH_FILE, LOG_FILE, profilePath, script, } from "./lib/paths.js";
|
|
@@ -77,6 +77,15 @@ const DEBUG = !!process.env.YT_BRIEFING_DEBUG;
|
|
|
77
77
|
const T0 = Date.now();
|
|
78
78
|
const log = (msg) => { if (DEBUG)
|
|
79
79
|
appendFileSync(LOG_FILE, `⏱ ${msg} (+${Date.now() - T0}ms)\n`); };
|
|
80
|
+
const skipLedger = [];
|
|
81
|
+
const recordedSkips = new Set();
|
|
82
|
+
/** Record a skip once per videoId — the fg/bg expansion race can apply a title-skip twice. */
|
|
83
|
+
function recordSkip(log) {
|
|
84
|
+
if (recordedSkips.has(log.videoId))
|
|
85
|
+
return;
|
|
86
|
+
recordedSkips.add(log.videoId);
|
|
87
|
+
skipLedger.push(log);
|
|
88
|
+
}
|
|
80
89
|
// ---------- subprocess helper ----------
|
|
81
90
|
function run(cmd) {
|
|
82
91
|
return new Promise((resolve, reject) => {
|
|
@@ -114,8 +123,10 @@ function persistSkip(item, status) {
|
|
|
114
123
|
* Idempotent — re-applying the same skip is a no-op. ONLY the foreground calls this.
|
|
115
124
|
*/
|
|
116
125
|
function applyTitleSkips(skips) {
|
|
117
|
-
for (const s of skips)
|
|
126
|
+
for (const s of skips) {
|
|
118
127
|
bumpState(s.channel, s.type, s.videoId);
|
|
128
|
+
recordSkip({ stage: 'title', channel: s.channel, videoId: s.videoId, title: s.title, reason: s.reason || 'title filter' });
|
|
129
|
+
}
|
|
119
130
|
}
|
|
120
131
|
/** Append items, skipping any videoId already queued OR already resolved (seen). */
|
|
121
132
|
function enqueue(queue, add) {
|
|
@@ -207,17 +218,18 @@ function spawnBackgroundFill() {
|
|
|
207
218
|
}
|
|
208
219
|
function emit(obj) {
|
|
209
220
|
log(`EXIT status=${obj.status}`);
|
|
210
|
-
|
|
221
|
+
const ledger = skipLedger.length ? { skipped: skipLedger.length, skips: skipLedger } : {};
|
|
222
|
+
process.stdout.write(JSON.stringify({ ...obj, ...ledger }));
|
|
211
223
|
process.exit(0);
|
|
212
224
|
}
|
|
213
225
|
// ---------- LLM gates ----------
|
|
214
226
|
/** Title filter: batch-classify a channel's non-baseline titles. Falls back to keep-all on any error. */
|
|
215
227
|
async function runTitleFilter(profilePathAbs, videos) {
|
|
216
|
-
const skip = new
|
|
228
|
+
const skip = new Map();
|
|
217
229
|
if (!existsSync(profilePathAbs))
|
|
218
230
|
return skip;
|
|
219
231
|
const profile = readFileSync(profilePathAbs, 'utf8');
|
|
220
|
-
if (!/##\s*Skip titles/.test(profile))
|
|
232
|
+
if (!/##\s*(Skip titles|Notes|Channel policy)/.test(profile))
|
|
221
233
|
return skip;
|
|
222
234
|
const toClassify = videos.filter(v => !v.is_baseline);
|
|
223
235
|
if (toClassify.length === 0)
|
|
@@ -227,8 +239,8 @@ async function runTitleFilter(profilePathAbs, videos) {
|
|
|
227
239
|
Channel profile:
|
|
228
240
|
${profile}
|
|
229
241
|
|
|
230
|
-
|
|
231
|
-
Keep by default
|
|
242
|
+
Skip a title when it clearly matches a skip rule from the profile — a '## Skip titles' example, or a skip/reject directive in '## Notes' or '## Channel policy' that is judgeable from the title alone (defer transcript-dependent rejections to the content stage).
|
|
243
|
+
Keep by default: skip only on a clear match, and when unsure, keep.
|
|
232
244
|
|
|
233
245
|
Videos:
|
|
234
246
|
${JSON.stringify(toClassify.map(v => ({ id: v.videoId, title: v.title, type: v.type })))}
|
|
@@ -253,7 +265,7 @@ Output ONLY a raw JSON array (no markdown fences, no explanation):
|
|
|
253
265
|
const parsed = JSON.parse(out.slice(start, end + 1));
|
|
254
266
|
for (const r of parsed)
|
|
255
267
|
if (r.result === 'skip')
|
|
256
|
-
skip.
|
|
268
|
+
skip.set(r.id, (r.reason ?? '').trim());
|
|
257
269
|
}
|
|
258
270
|
catch { /* keep-all */ }
|
|
259
271
|
return skip;
|
|
@@ -263,7 +275,7 @@ async function runContentFilter(item, transcript) {
|
|
|
263
275
|
const profile = existsSync(item.profile_path) ? readFileSync(item.profile_path, 'utf8') : '';
|
|
264
276
|
const baselineNote = item.is_baseline ? ' · baseline' : '';
|
|
265
277
|
const profileSection = profile
|
|
266
|
-
? `\nChannel
|
|
278
|
+
? `\nChannel directives — standing instructions the user set for THIS channel (sections: Channel policy, Summary format, Cut sections, Episode types, Notes). Carry them out as written: a directive may reject the video outright (see the skip check in step 1), or — for a video that passes — add a section, shift emphasis or tone, or scrutinize/flag specific things within the briefing format (never changing its skeleton). If a directive asks you to verify or scrutinize claims, assess them against your own background knowledge: state what you can corroborate, what looks overstated, cherry-picked or misattributed, and what you cannot place. Always mark your confidence, and say plainly this is a from-memory assessment, not live verification — for anything recent or outside your knowledge, flag that instead of guessing. Keep "what the video claims" separate from your own assessment.\n${profile}\n`
|
|
267
279
|
: '';
|
|
268
280
|
const prompt = `Write a summary of this YouTube video in ${LANG}, OR return 'OFFTOPIC: <reason>' if the transcript clearly does not match what the title/channel promises.
|
|
269
281
|
|
|
@@ -279,17 +291,18 @@ Transcript:
|
|
|
279
291
|
${transcript}
|
|
280
292
|
${profileSection}
|
|
281
293
|
Steps:
|
|
282
|
-
1. Substance check:
|
|
294
|
+
1. Substance / skip check — output ONLY 'OFFTOPIC: <short reason>' and stop if EITHER (a) the transcript clearly does not deliver what the title promises, OR (b) a channel directive clearly instructs rejecting this kind of content and the transcript clearly meets that rejection criterion. When unsure, keep. Otherwise continue.
|
|
283
295
|
2. Otherwise write the summary:
|
|
284
296
|
- Header: ### ${item.channel} — "${item.title}"
|
|
285
297
|
- Subtitle: _${item.publishedAt} · ${item.type} · https://youtube.com/watch?v=${item.videoId}${baselineNote}_
|
|
286
298
|
- 2-5 numbered thematic sections × 2-5 sentences each
|
|
287
299
|
- At most 5-8 short quotes from the transcript
|
|
288
300
|
- No timestamps
|
|
289
|
-
3.
|
|
290
|
-
4.
|
|
301
|
+
3. Apply the channel directives above, if any — they shape the briefing's content, emphasis, and tone, but not the header/subtitle/section skeleton. Put any directive-driven assessment in its own short, clearly-labeled section.
|
|
302
|
+
4. Language: natural ${LANG}. Avoid calques/anglicisms; use foreign words only for proper nouns or established technical terms. Section headers should be verb phrases, not noun stacks.
|
|
303
|
+
5. Output: ONLY the briefing itself OR 'OFFTOPIC: ...'. No preamble, and no meta-commentary about these instructions.`;
|
|
291
304
|
return await chat(prompt, {
|
|
292
|
-
system: `You
|
|
305
|
+
system: `You write channel briefings in ${LANG}, following the task instructions and the channel's standing directives exactly. Output only the briefing or 'OFFTOPIC: <reason>' — no preamble, no meta-commentary about the instructions.`,
|
|
293
306
|
model: getModel(),
|
|
294
307
|
});
|
|
295
308
|
}
|
|
@@ -321,7 +334,7 @@ async function expandChannel(ref) {
|
|
|
321
334
|
const skips = [];
|
|
322
335
|
for (const it of candidates) {
|
|
323
336
|
if (titleSkip.has(it.videoId))
|
|
324
|
-
skips.push({ channel: it.channel, type: it.type, videoId: it.videoId });
|
|
337
|
+
skips.push({ channel: it.channel, type: it.type, videoId: it.videoId, title: it.title, reason: titleSkip.get(it.videoId) || 'title filter' });
|
|
325
338
|
else
|
|
326
339
|
items.push(it);
|
|
327
340
|
}
|
|
@@ -418,15 +431,19 @@ async function processItem(item) {
|
|
|
418
431
|
log(`transcript ${item.videoId} ${Date.now() - tT}ms (exit ${t.code})`);
|
|
419
432
|
if (t.code === 2)
|
|
420
433
|
return { kind: 'rate_limited' };
|
|
434
|
+
// exit 3 = tooling/proxy failure (503, tunnel down, missing yt-dlp) — NOT a missing-captions
|
|
435
|
+
// case. Surface it distinctly so the operator looks at the toolchain, not at YouTube/IP.
|
|
436
|
+
if (t.code === 3)
|
|
437
|
+
return { kind: 'tooling_error' };
|
|
421
438
|
if (t.code !== 0)
|
|
422
439
|
return { kind: 'skip', status: 'no_transcript' };
|
|
423
440
|
if (!t.stdout.trim())
|
|
424
|
-
return { kind: 'skip', status: 'content_skip' };
|
|
441
|
+
return { kind: 'skip', status: 'content_skip', reason: 'empty transcript' };
|
|
425
442
|
const tC = Date.now();
|
|
426
443
|
const summary = await runContentFilter(item, t.stdout);
|
|
427
444
|
log(`content ${item.videoId} ${Date.now() - tC}ms`);
|
|
428
445
|
if (summary.startsWith('OFFTOPIC:'))
|
|
429
|
-
return { kind: 'skip', status: 'content_skip' };
|
|
446
|
+
return { kind: 'skip', status: 'content_skip', reason: summary.slice(9).trim() };
|
|
430
447
|
return { kind: 'ratable', summary };
|
|
431
448
|
}
|
|
432
449
|
async function advance(queue) {
|
|
@@ -437,8 +454,9 @@ async function advance(queue) {
|
|
|
437
454
|
if (queue.items.length === 0)
|
|
438
455
|
break;
|
|
439
456
|
const item = queue.items[0];
|
|
440
|
-
// Head already resolved last round (rated/skipped → pointer landed on it
|
|
441
|
-
|
|
457
|
+
// Head already resolved last round (rated/skipped → pointer landed on it, or it's in
|
|
458
|
+
// `seen`) → drop. The `seen` arm survives a --fill pointer-regression race (isResolved).
|
|
459
|
+
if (isResolved(statePointer(item), item.videoId, queue.seen)) {
|
|
442
460
|
dropHead(queue);
|
|
443
461
|
continue;
|
|
444
462
|
}
|
|
@@ -453,8 +471,13 @@ async function advance(queue) {
|
|
|
453
471
|
writeFileSync(QUEUE_FILE, JSON.stringify(queue));
|
|
454
472
|
emit({ status: 'rate_limited' });
|
|
455
473
|
}
|
|
474
|
+
if (result.kind === 'tooling_error') {
|
|
475
|
+
writeFileSync(QUEUE_FILE, JSON.stringify(queue));
|
|
476
|
+
emit({ status: 'tooling_error' });
|
|
477
|
+
}
|
|
456
478
|
if (result.kind === 'skip') {
|
|
457
479
|
persistSkip(item, result.status);
|
|
480
|
+
recordSkip({ stage: 'content', channel: item.channel, videoId: item.videoId, title: item.title, reason: result.reason ?? (result.status === 'no_transcript' ? 'no transcript' : 'off-topic') });
|
|
458
481
|
dropHead(queue);
|
|
459
482
|
continue;
|
|
460
483
|
}
|
|
@@ -489,7 +512,7 @@ async function runPrefetch(videoId) {
|
|
|
489
512
|
const item = queue.items.find(i => i.videoId === videoId);
|
|
490
513
|
if (!item)
|
|
491
514
|
process.exit(0);
|
|
492
|
-
if (statePointer(item)
|
|
515
|
+
if (isResolved(statePointer(item), item.videoId, queue.seen))
|
|
493
516
|
process.exit(0); // already resolved
|
|
494
517
|
if (loadPrefetch(item.videoId))
|
|
495
518
|
process.exit(0); // already warm
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "yt-briefing",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.13.0",
|
|
4
4
|
"description": "A self-learning YouTube briefing engine: it sweeps the channels you follow, filters noise in two stages (title, then transcript), summarizes the rest in your language, and adapts to your ratings — one video at a time.",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"bin": {
|