@graphlearning/shell 0.7.0 → 0.10.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,193 @@
1
+ #!/usr/bin/env node
2
+ // record-all.mjs — batch 4K capture across courses, pulling narration from GitHub as it lands.
3
+ //
4
+ // npm run record:all # every course, in syllabus order
5
+ // npm run record:all -- --courses internals,performance
6
+ // npm run record:all -- --wait 0 # never wait; skip any course missing audio
7
+ // npm run record:all -- --no-pull # use the working tree as-is
8
+ //
9
+ // WHY THIS EXISTS: the narration wavs are produced by a Colab + Chatterbox pass that commits them
10
+ // to the repo's branch one section at a time, so the audio for a course arrives WHILE this is
11
+ // running. A plain `for c in …; do npm run record -- $c; done` would reach `internals` before its
12
+ // nine wavs existed and silently record six sections of 3s silence (record-course.mjs's documented
13
+ // fallback — it keeps the pipeline from hanging, which is right for one missing clip and wrong for a
14
+ // whole chapter). So before each course this does: git pull → count that course's wavs against
15
+ // scripts/audio-manifest.json → if any are missing, keep pulling every --poll until they land or
16
+ // --wait expires. A course is only recorded when it is COMPLETE; an incomplete one is skipped and
17
+ // named in the summary, never half-recorded.
18
+ //
19
+ // Everything else is record-course.mjs's job, unchanged: the 2.4s pulse-period loop window, the
20
+ // bell lead-in, the per-segment fingerprint cache (so a re-run after a skip costs nothing for the
21
+ // courses already done) and the final concat into scripts/out/<course>.mp4.
22
+ //
23
+ // SLEEP: the repo's `record:all` script wraps this in `caffeinate -ims` so a multi-hour batch
24
+ // survives the machine going idle — `"record:all": "caffeinate -ims graphl-record-all"`. -i no idle
25
+ // sleep, -m no disk sleep, -s no system sleep (AC power only). -d is deliberately NOT set: capture
26
+ // is headless, so forcing the display on all night buys nothing. The wrapper stays in the repo
27
+ // rather than inside this script because `caffeinate` is macOS-only and re-exec'ing ourselves under
28
+ // it would hide a process layer from anyone reading the batch's output.
29
+
30
+ import { execFileSync, spawn } from 'node:child_process'
31
+ import { existsSync, readFileSync } from 'node:fs'
32
+ import { join } from 'node:path'
33
+
34
+ import { repoDir, dataDir, pkgDir } from './_paths.mjs'
35
+
36
+ const manifestPath = join(dataDir, 'audio-manifest.json')
37
+ if (!existsSync(manifestPath)) {
38
+ console.error(`No scripts/audio-manifest.json in ${repoDir} — generate it first: npm run gen:audio`)
39
+ process.exit(1)
40
+ }
41
+ const entries = JSON.parse(readFileSync(manifestPath, 'utf8')).entries ?? []
42
+
43
+ // The recorder is PACKAGE-owned, so resolving it from this file is correct (see _paths.mjs: the rule
44
+ // is never to resolve repo data that way). Spawned directly rather than through `npm run record` —
45
+ // one less process between the batch and ffmpeg/puppeteer for signals to cross, and it does not
46
+ // require the consuming repo to have declared a `record` script.
47
+ const RECORDER = join(pkgDir, 'record-course.mjs')
48
+
49
+ // ---- CLI ---------------------------------------------------------------------------------
50
+ function parse(argv) {
51
+ const o = { courses: [], waitMin: 240, pollS: 120, pull: true, force: false }
52
+ for (let i = 0; i < argv.length; i++) {
53
+ const a = argv[i]
54
+ if (a === '--courses') o.courses.push(...(argv[++i] ?? '').split(',').filter(Boolean))
55
+ else if (a.startsWith('--courses=')) o.courses.push(...a.slice(10).split(',').filter(Boolean))
56
+ else if (a === '--wait') o.waitMin = +(argv[++i] ?? 0)
57
+ else if (a.startsWith('--wait=')) o.waitMin = +a.slice(7)
58
+ else if (a === '--poll') o.pollS = +(argv[++i] ?? 120)
59
+ else if (a.startsWith('--poll=')) o.pollS = +a.slice(7)
60
+ else if (a === '--no-pull') o.pull = false
61
+ else if (a === '--force') o.force = true
62
+ else { console.error(`unknown argument: ${a}`); process.exit(2) }
63
+ }
64
+ return o
65
+ }
66
+ const opt = parse(process.argv.slice(2))
67
+
68
+ // Syllabus order = the manifest's own order (it is generated from COURSES), not alphabetical.
69
+ const allCourses = [...new Set(entries.map((e) => e.course))]
70
+ const courses = opt.courses.length ? opt.courses : allCourses
71
+ const unknown = courses.filter((c) => !allCourses.includes(c))
72
+ if (unknown.length) {
73
+ console.error(`unknown course(s): ${unknown.join(', ')}\nknown: ${allCourses.join(', ')}`)
74
+ process.exit(2)
75
+ }
76
+
77
+ const sleep = (ms) => new Promise((r) => setTimeout(r, ms))
78
+ const hhmm = () => new Date().toTimeString().slice(0, 8)
79
+
80
+ // ---- narration state ---------------------------------------------------------------------
81
+ // A course is recordable when every section the manifest lists for it has a wav on disk.
82
+ function audioState(course) {
83
+ const want = entries.filter((e) => e.course === course)
84
+ const missing = want.filter((e) => !existsSync(join(repoDir, 'public', 'audio', e.file)))
85
+ return { total: want.length, have: want.length - missing.length, missing: missing.map((e) => e.section) }
86
+ }
87
+
88
+ // ---- git ----------------------------------------------------------------------------------
89
+ // The branch is read rather than assumed: most repos record from `main`, but not all of them sit
90
+ // there (aws-lab is on `rebuild/atom-library`), and pulling a branch the repo is not on would
91
+ // either fail or quietly fetch the wrong wavs. A detached HEAD has nothing to track, so pulling is
92
+ // skipped instead of guessed at.
93
+ function currentBranch() {
94
+ try {
95
+ // stderr ignored: outside a git repo this prints git's own "fatal: not a git repository" ahead
96
+ // of the note below, which reads like the batch broke when it is a supported case.
97
+ const b = execFileSync('git', ['rev-parse', '--abbrev-ref', 'HEAD'],
98
+ { cwd: repoDir, encoding: 'utf8', stdio: ['ignore', 'pipe', 'ignore'] }).trim()
99
+ return b && b !== 'HEAD' ? b : null
100
+ } catch {
101
+ return null
102
+ }
103
+ }
104
+ const branch = opt.pull ? currentBranch() : null
105
+
106
+ // Fast-forward only: this repo's working tree is never the place Colab's commits get merged, so a
107
+ // non-ff situation means something unexpected and the batch should say so rather than resolve it.
108
+ function pull() {
109
+ if (!opt.pull) return { ok: true, note: 'skipped (--no-pull)' }
110
+ if (!branch) return { ok: true, note: 'skipped (detached HEAD or not a git repo)' }
111
+ try {
112
+ const before = execFileSync('git', ['rev-parse', 'HEAD'], { cwd: repoDir, encoding: 'utf8' }).trim()
113
+ execFileSync('git', ['pull', '--ff-only', 'origin', branch], { cwd: repoDir, stdio: 'pipe' })
114
+ const after = execFileSync('git', ['rev-parse', 'HEAD'], { cwd: repoDir, encoding: 'utf8' }).trim()
115
+ return { ok: true, note: before === after ? 'already current' : `${before.slice(0, 7)} → ${after.slice(0, 7)}` }
116
+ } catch (e) {
117
+ // Network blips and transient GitHub errors are expected over a multi-hour run — report and
118
+ // carry on with whatever the working tree already has rather than aborting the batch.
119
+ return { ok: false, note: (e.stderr?.toString() || e.message).trim().split('\n').pop() }
120
+ }
121
+ }
122
+
123
+ // ---- record ---------------------------------------------------------------------------------
124
+ function record(course) {
125
+ return new Promise((done) => {
126
+ const args = [RECORDER, course, ...(opt.force ? ['--force'] : [])]
127
+ const child = spawn(process.execPath, args, { cwd: repoDir, stdio: 'inherit', env: process.env })
128
+ current = child
129
+ child.on('exit', (code) => { current = null; done(code ?? 1) })
130
+ child.on('error', () => { current = null; done(1) })
131
+ })
132
+ }
133
+
134
+ // Forward Ctrl-C to the running recorder so ffmpeg/puppeteer/vite do not outlive the batch.
135
+ let current = null
136
+ for (const sig of ['SIGINT', 'SIGTERM']) {
137
+ process.on(sig, () => { current?.kill(sig); console.log(`\n${sig} — stopping batch.`); process.exit(130) })
138
+ }
139
+
140
+ // ---- the batch --------------------------------------------------------------------------------
141
+ const results = []
142
+ console.log(`Batch 4K capture — ${courses.length} course(s): ${courses.join(', ')}`)
143
+ console.log(` pull=${opt.pull}${branch ? ` (origin/${branch})` : ''} wait=${opt.waitMin}min poll=${opt.pollS}s force=${opt.force}\n`)
144
+
145
+ for (const course of courses) {
146
+ console.log(`${'─'.repeat(72)}\n${course} [${hhmm()}]`)
147
+
148
+ const p = pull()
149
+ console.log(` git pull: ${p.ok ? p.note : `FAILED — ${p.note} (continuing with local tree)`}`)
150
+
151
+ let st = audioState(course)
152
+ // Colab commits one section at a time, so an incomplete course is usually just EARLY. Keep
153
+ // pulling until it completes or the budget runs out; --wait 0 turns this into a plain skip.
154
+ const deadline = Date.now() + opt.waitMin * 60_000
155
+ while (st.missing.length && Date.now() < deadline) {
156
+ console.log(` audio ${st.have}/${st.total} — waiting ${opt.pollS}s for: ${st.missing.join(', ')}`)
157
+ await sleep(opt.pollS * 1000)
158
+ const q = pull()
159
+ if (!q.ok) console.log(` git pull: FAILED — ${q.note}`)
160
+ st = audioState(course)
161
+ }
162
+
163
+ if (st.missing.length) {
164
+ console.log(` ⊘ SKIP — audio ${st.have}/${st.total}, still missing: ${st.missing.join(', ')}`)
165
+ results.push({ course, status: 'skipped', detail: `${st.have}/${st.total} wavs` })
166
+ continue
167
+ }
168
+
169
+ console.log(` audio ${st.have}/${st.total} ✓ — recording\n`)
170
+ const t0 = Date.now()
171
+ const code = await record(course)
172
+ const mins = ((Date.now() - t0) / 60_000).toFixed(1)
173
+ if (code === 0) {
174
+ console.log(`\n ✅ ${course} — scripts/out/${course}.mp4 (${mins} min)`)
175
+ results.push({ course, status: 'ok', detail: `${mins} min` })
176
+ } else {
177
+ console.log(`\n ✗ ${course} — recorder exited ${code} (${mins} min)`)
178
+ results.push({ course, status: 'failed', detail: `exit ${code}` })
179
+ }
180
+ }
181
+
182
+ // ---- summary ------------------------------------------------------------------------------------
183
+ console.log(`\n${'═'.repeat(72)}\nSummary [${hhmm()}]`)
184
+ for (const r of results) {
185
+ const mark = r.status === 'ok' ? '✅' : r.status === 'skipped' ? '⊘ ' : '✗ '
186
+ console.log(` ${mark} ${r.course.padEnd(16)} ${r.status.padEnd(8)} ${r.detail}`)
187
+ }
188
+ const skipped = results.filter((r) => r.status === 'skipped').map((r) => r.course)
189
+ if (skipped.length) {
190
+ console.log(`\nRe-run once Colab finishes — completed courses are reused from their segment cache:`)
191
+ console.log(` npm run record:all -- --courses ${skipped.join(',')}`)
192
+ }
193
+ process.exit(results.some((r) => r.status === 'failed') ? 1 : 0)