yt-briefing 0.12.3 → 0.13.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -22,10 +22,12 @@ Read `data/config.json` → `output_lang` once at the start of the loop. Phrase
22
22
  ```
23
23
  out = JSON.parse(`bun run src/yt-sweep.ts --reset`) // Bash — first call: --reset rebuilds the queue fresh
24
24
  while true:
25
+ if out.skipped: FIRST, in output_lang, tell the user one line — "Skipped {out.skipped}:" + a `; `-joined list of `{channel} «{title}» — {reason}` from out.skips (videos the title/content filter dropped on the way here, including your channel directives). Then handle out.status below.
25
26
  out.status:
26
27
  "done" → sweep finished — tell the user, stop
27
28
  "error" → setup/config problem (e.g. missing API key) — show `out.error` to the user verbatim, stop
28
- "rate_limited" → transcript fetch blocked (usually a blocked egress IP) — tell the user, stop; recovery in README.md → Proxy
29
+ "rate_limited" → YouTube is blocking the egress IP (429 / captcha) — tell the user, stop; recovery in README.md → Proxy
30
+ "tooling_error" → transcript toolchain failed (proxy down / yt-dlp / network) — tell the user, stop; check proxy health, then README.md → Proxy
29
31
  "rating_needed" → steps A–E
30
32
  ```
31
33
 
package/dist/yt-sweep.js CHANGED
@@ -77,6 +77,15 @@ const DEBUG = !!process.env.YT_BRIEFING_DEBUG;
77
77
  const T0 = Date.now();
78
78
  const log = (msg) => { if (DEBUG)
79
79
  appendFileSync(LOG_FILE, `⏱ ${msg} (+${Date.now() - T0}ms)\n`); };
80
+ const skipLedger = [];
81
+ const recordedSkips = new Set();
82
+ /** Record a skip once per videoId — the fg/bg expansion race can apply a title-skip twice. */
83
+ function recordSkip(log) {
84
+ if (recordedSkips.has(log.videoId))
85
+ return;
86
+ recordedSkips.add(log.videoId);
87
+ skipLedger.push(log);
88
+ }
80
89
  // ---------- subprocess helper ----------
81
90
  function run(cmd) {
82
91
  return new Promise((resolve, reject) => {
@@ -114,8 +123,10 @@ function persistSkip(item, status) {
114
123
  * Idempotent — re-applying the same skip is a no-op. ONLY the foreground calls this.
115
124
  */
116
125
  function applyTitleSkips(skips) {
117
- for (const s of skips)
126
+ for (const s of skips) {
118
127
  bumpState(s.channel, s.type, s.videoId);
128
+ recordSkip({ stage: 'title', channel: s.channel, videoId: s.videoId, title: s.title, reason: s.reason || 'title filter' });
129
+ }
119
130
  }
120
131
  /** Append items, skipping any videoId already queued OR already resolved (seen). */
121
132
  function enqueue(queue, add) {
@@ -207,17 +218,18 @@ function spawnBackgroundFill() {
207
218
  }
208
219
  function emit(obj) {
209
220
  log(`EXIT status=${obj.status}`);
210
- process.stdout.write(JSON.stringify(obj));
221
+ const ledger = skipLedger.length ? { skipped: skipLedger.length, skips: skipLedger } : {};
222
+ process.stdout.write(JSON.stringify({ ...obj, ...ledger }));
211
223
  process.exit(0);
212
224
  }
213
225
  // ---------- LLM gates ----------
214
226
  /** Title filter: batch-classify a channel's non-baseline titles. Falls back to keep-all on any error. */
215
227
  async function runTitleFilter(profilePathAbs, videos) {
216
- const skip = new Set();
228
+ const skip = new Map();
217
229
  if (!existsSync(profilePathAbs))
218
230
  return skip;
219
231
  const profile = readFileSync(profilePathAbs, 'utf8');
220
- if (!/##\s*Skip titles/.test(profile))
232
+ if (!/##\s*(Skip titles|Notes|Channel policy)/.test(profile))
221
233
  return skip;
222
234
  const toClassify = videos.filter(v => !v.is_baseline);
223
235
  if (toClassify.length === 0)
@@ -227,8 +239,8 @@ async function runTitleFilter(profilePathAbs, videos) {
227
239
  Channel profile:
228
240
  ${profile}
229
241
 
230
- Focus on the '## Skip titles' section (titles to skip) and any '## Notes' rules.
231
- Keep by default — only skip a title that clearly matches the worthless pattern. If that section is missing or empty: classify all as keep.
242
+ Skip a title when it clearly matches a skip rule from the profile — a '## Skip titles' example, or a skip/reject directive in '## Notes' or '## Channel policy' that is judgeable from the title alone (defer transcript-dependent rejections to the content stage).
243
+ Keep by default: skip only on a clear match, and when unsure, keep.
232
244
 
233
245
  Videos:
234
246
  ${JSON.stringify(toClassify.map(v => ({ id: v.videoId, title: v.title, type: v.type })))}
@@ -253,7 +265,7 @@ Output ONLY a raw JSON array (no markdown fences, no explanation):
253
265
  const parsed = JSON.parse(out.slice(start, end + 1));
254
266
  for (const r of parsed)
255
267
  if (r.result === 'skip')
256
- skip.add(r.id);
268
+ skip.set(r.id, (r.reason ?? '').trim());
257
269
  }
258
270
  catch { /* keep-all */ }
259
271
  return skip;
@@ -263,7 +275,7 @@ async function runContentFilter(item, transcript) {
263
275
  const profile = existsSync(item.profile_path) ? readFileSync(item.profile_path, 'utf8') : '';
264
276
  const baselineNote = item.is_baseline ? ' · baseline' : '';
265
277
  const profileSection = profile
266
- ? `\nChannel profile (sections to honor: Channel policy, Summary format, Cut sections, Episode types, Notes):\n${profile}\n`
278
+ ? `\nChannel directives — standing instructions the user set for THIS channel (sections: Channel policy, Summary format, Cut sections, Episode types, Notes). Carry them out as written: a directive may reject the video outright (see the skip check in step 1), or — for a video that passes — add a section, shift emphasis or tone, or scrutinize/flag specific things within the briefing format (never changing its skeleton). If a directive asks you to verify or scrutinize claims, assess them against your own background knowledge: state what you can corroborate, what looks overstated, cherry-picked or misattributed, and what you cannot place. Always mark your confidence, and say plainly this is a from-memory assessment, not live verification — for anything recent or outside your knowledge, flag that instead of guessing. Keep "what the video claims" separate from your own assessment.\n${profile}\n`
267
279
  : '';
268
280
  const prompt = `Write a summary of this YouTube video in ${LANG}, OR return 'OFFTOPIC: <reason>' if the transcript clearly does not match what the title/channel promises.
269
281
 
@@ -279,17 +291,18 @@ Transcript:
279
291
  ${transcript}
280
292
  ${profileSection}
281
293
  Steps:
282
- 1. Substance check: does the transcript actually deliver what the title promises? If clearly not → output ONLY 'OFFTOPIC: <short reason>' and stop.
294
+ 1. Substance / skip check — output ONLY 'OFFTOPIC: <short reason>' and stop if EITHER (a) the transcript clearly does not deliver what the title promises, OR (b) a channel directive clearly instructs rejecting this kind of content and the transcript clearly meets that rejection criterion. When unsure, keep. Otherwise continue.
283
295
  2. Otherwise write the summary:
284
296
  - Header: ### ${item.channel} — "${item.title}"
285
297
  - Subtitle: _${item.publishedAt} · ${item.type} · https://youtube.com/watch?v=${item.videoId}${baselineNote}_
286
298
  - 2-5 numbered thematic sections × 2-5 sentences each
287
299
  - At most 5-8 short quotes from the transcript
288
300
  - No timestamps
289
- 3. Language: natural ${LANG}. Avoid calques/anglicisms; use foreign words only for proper nouns or established technical terms. Section headers should be verb phrases, not noun stacks.
290
- 4. Output: ONLY the summary OR 'OFFTOPIC: ...'. No preamble, no trailing commentary.`;
301
+ 3. Apply the channel directives above, if any — they shape the briefing's content, emphasis, and tone, but not the header/subtitle/section skeleton. Put any directive-driven assessment in its own short, clearly-labeled section.
302
+ 4. Language: natural ${LANG}. Avoid calques/anglicisms; use foreign words only for proper nouns or established technical terms. Section headers should be verb phrases, not noun stacks.
303
+ 5. Output: ONLY the briefing itself OR 'OFFTOPIC: ...'. No preamble, and no meta-commentary about these instructions.`;
291
304
  return await chat(prompt, {
292
- system: `You are a video summarizer writing in ${LANG}. Follow the task instructions exactly. Output only the summary or 'OFFTOPIC: <reason>'. No preamble, no commentary.`,
305
+ system: `You write channel briefings in ${LANG}, following the task instructions and the channel's standing directives exactly. Output only the briefing or 'OFFTOPIC: <reason>' — no preamble, no meta-commentary about the instructions.`,
293
306
  model: getModel(),
294
307
  });
295
308
  }
@@ -321,7 +334,7 @@ async function expandChannel(ref) {
321
334
  const skips = [];
322
335
  for (const it of candidates) {
323
336
  if (titleSkip.has(it.videoId))
324
- skips.push({ channel: it.channel, type: it.type, videoId: it.videoId });
337
+ skips.push({ channel: it.channel, type: it.type, videoId: it.videoId, title: it.title, reason: titleSkip.get(it.videoId) || 'title filter' });
325
338
  else
326
339
  items.push(it);
327
340
  }
@@ -418,15 +431,19 @@ async function processItem(item) {
418
431
  log(`transcript ${item.videoId} ${Date.now() - tT}ms (exit ${t.code})`);
419
432
  if (t.code === 2)
420
433
  return { kind: 'rate_limited' };
434
+ // exit 3 = tooling/proxy failure (503, tunnel down, missing yt-dlp) — NOT a missing-captions
435
+ // case. Surface it distinctly so the operator looks at the toolchain, not at YouTube/IP.
436
+ if (t.code === 3)
437
+ return { kind: 'tooling_error' };
421
438
  if (t.code !== 0)
422
439
  return { kind: 'skip', status: 'no_transcript' };
423
440
  if (!t.stdout.trim())
424
- return { kind: 'skip', status: 'content_skip' };
441
+ return { kind: 'skip', status: 'content_skip', reason: 'empty transcript' };
425
442
  const tC = Date.now();
426
443
  const summary = await runContentFilter(item, t.stdout);
427
444
  log(`content ${item.videoId} ${Date.now() - tC}ms`);
428
445
  if (summary.startsWith('OFFTOPIC:'))
429
- return { kind: 'skip', status: 'content_skip' };
446
+ return { kind: 'skip', status: 'content_skip', reason: summary.slice(9).trim() };
430
447
  return { kind: 'ratable', summary };
431
448
  }
432
449
  async function advance(queue) {
@@ -454,8 +471,13 @@ async function advance(queue) {
454
471
  writeFileSync(QUEUE_FILE, JSON.stringify(queue));
455
472
  emit({ status: 'rate_limited' });
456
473
  }
474
+ if (result.kind === 'tooling_error') {
475
+ writeFileSync(QUEUE_FILE, JSON.stringify(queue));
476
+ emit({ status: 'tooling_error' });
477
+ }
457
478
  if (result.kind === 'skip') {
458
479
  persistSkip(item, result.status);
480
+ recordSkip({ stage: 'content', channel: item.channel, videoId: item.videoId, title: item.title, reason: result.reason ?? (result.status === 'no_transcript' ? 'no transcript' : 'off-topic') });
459
481
  dropHead(queue);
460
482
  continue;
461
483
  }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "yt-briefing",
3
- "version": "0.12.3",
3
+ "version": "0.13.0",
4
4
  "description": "A self-learning YouTube briefing engine: it sweeps the channels you follow, filters noise in two stages (title, then transcript), summarizes the rest in your language, and adapts to your ratings — one video at a time.",
5
5
  "type": "module",
6
6
  "bin": {