@hraness/dawg 0.0.0-stage → 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (113) hide show
  1. package/CHANGELOG.md +126 -0
  2. package/DAWG.md +327 -0
  3. package/LICENSE +21 -0
  4. package/README.md +213 -2
  5. package/core/diff.ts +249 -0
  6. package/core/drums.ts +102 -0
  7. package/core/key.ts +43 -0
  8. package/core/loop.ts +78 -0
  9. package/core/pitch.ts +60 -0
  10. package/core/score.ts +1388 -0
  11. package/core/sdk/eval-child.ts +113 -0
  12. package/core/sdk/eval.ts +257 -0
  13. package/core/sdk/print.ts +393 -0
  14. package/core/sdk/v1.ts +954 -0
  15. package/core/slug.ts +19 -0
  16. package/package.json +45 -4
  17. package/src/agent/agent.ts +853 -0
  18. package/src/agent/brief.ts +160 -0
  19. package/src/agent/gateway.ts +441 -0
  20. package/src/agent/models.ts +633 -0
  21. package/src/agent/ops.ts +157 -0
  22. package/src/agent/planner.ts +259 -0
  23. package/src/agent/provider.ts +454 -0
  24. package/src/agent/sse.ts +114 -0
  25. package/src/agent/tools.ts +1373 -0
  26. package/src/agent/usage.ts +296 -0
  27. package/src/agent/workspace.ts +683 -0
  28. package/src/agent/xcb-agent.ts +262 -0
  29. package/src/agent/xcb.ts +579 -0
  30. package/src/audio/click.ts +125 -0
  31. package/src/audio/clock.ts +68 -0
  32. package/src/audio/engine.ts +841 -0
  33. package/src/audio/live.ts +152 -0
  34. package/src/audio/lock.ts +57 -0
  35. package/src/audio/player.ts +134 -0
  36. package/src/audio/render-worker.ts +68 -0
  37. package/src/audio/renderer.ts +174 -0
  38. package/src/audio/sampler.ts +292 -0
  39. package/src/audio/samples.ts +683 -0
  40. package/src/audio/wav.ts +861 -0
  41. package/src/auth/cli.ts +231 -0
  42. package/src/auth/credentials.ts +411 -0
  43. package/src/auth/discover.ts +481 -0
  44. package/src/auth/login.ts +1191 -0
  45. package/src/auth/openrouter.ts +206 -0
  46. package/src/auth/picker.ts +282 -0
  47. package/src/auth/runner.ts +207 -0
  48. package/src/auth/tui.ts +107 -0
  49. package/src/commands/edit.ts +170 -0
  50. package/src/commands/help.ts +247 -0
  51. package/src/commands/history.ts +69 -0
  52. package/src/commands/music.ts +461 -0
  53. package/src/commands/sample.ts +302 -0
  54. package/src/daemon.ts +31 -0
  55. package/src/main.ts +2209 -0
  56. package/src/media/analyze.ts +364 -0
  57. package/src/media/backend.ts +253 -0
  58. package/src/media/cli.ts +173 -0
  59. package/src/media/download.ts +281 -0
  60. package/src/media/dsp.ts +281 -0
  61. package/src/media/import.ts +130 -0
  62. package/src/media/lyrics.ts +201 -0
  63. package/src/media/notes.ts +363 -0
  64. package/src/media/paths.ts +168 -0
  65. package/src/media/process.ts +226 -0
  66. package/src/media/registry.ts +9 -0
  67. package/src/media/sidecar.ts +72 -0
  68. package/src/media/stemdeck.ts +254 -0
  69. package/src/media/stems.ts +173 -0
  70. package/src/media/tools.ts +292 -0
  71. package/src/media/types.ts +92 -0
  72. package/src/media/vendor/basic-pitch.ts +261 -0
  73. package/src/media/vendor/drums.ts +817 -0
  74. package/src/media/vendor/grid.ts +203 -0
  75. package/src/media/vendor/util.ts +139 -0
  76. package/src/media/vendor/wav.ts +233 -0
  77. package/src/project/check.ts +80 -0
  78. package/src/project/init.ts +253 -0
  79. package/src/project/sync.ts +432 -0
  80. package/src/project/typecheck.ts +149 -0
  81. package/src/render.ts +121 -0
  82. package/src/session/attach.ts +181 -0
  83. package/src/session/client.ts +498 -0
  84. package/src/session/daemon.ts +740 -0
  85. package/src/session/delta.ts +249 -0
  86. package/src/session/list.ts +180 -0
  87. package/src/session/lock.ts +92 -0
  88. package/src/session/meta.ts +253 -0
  89. package/src/session/naming.ts +430 -0
  90. package/src/session/port.ts +481 -0
  91. package/src/session/presence.ts +159 -0
  92. package/src/session/protocol.ts +618 -0
  93. package/src/session/rebase.ts +168 -0
  94. package/src/session/store.ts +581 -0
  95. package/src/tui/menu.ts +1083 -0
  96. package/src/tui/play-mode.ts +442 -0
  97. package/src/tui/play-session.ts +636 -0
  98. package/src/web/fetch.ts +340 -0
  99. package/src/web/http.ts +137 -0
  100. package/src/web/search.ts +681 -0
  101. package/tui/activity.ts +364 -0
  102. package/tui/app.ts +1372 -0
  103. package/tui/drums.ts +65 -0
  104. package/tui/highway.ts +921 -0
  105. package/tui/input.ts +63 -0
  106. package/tui/keys.ts +102 -0
  107. package/tui/layers.ts +80 -0
  108. package/tui/play-strip.ts +143 -0
  109. package/tui/prompt.ts +609 -0
  110. package/tui/render.ts +124 -0
  111. package/tui/screen.ts +247 -0
  112. package/tui/text.ts +72 -0
  113. package/tui/theme.ts +451 -0
@@ -0,0 +1,281 @@
1
+ /**
2
+ * Pure-TypeScript analysis used when no StemDeck beat grid exists: an onset
3
+ * envelope → autocorrelation tempo estimate with beat phase, a pitch-class
4
+ * histogram for key estimation, and a 240-bucket RMS waveform. Everything
5
+ * works on a mono float signal and is O(n) in samples plus a bounded search.
6
+ */
7
+ import { estimateKey } from "../../core/key.ts";
8
+ import { MEDIA_LIMITS } from "./types.ts";
9
+ import type { ParsedWav } from "./vendor/wav.ts";
10
+
11
+ export const MIN_BPM = 60;
12
+ export const MAX_BPM = 200;
13
+ const HOP = 512;
14
+ /** Analyse at most this much audio for tempo and key (the first minutes decide). */
15
+ const MAX_ANALYSIS_SECONDS = 240;
16
+
17
+ /** Downmixed mono samples, bounded to MAX_ANALYSIS_SECONDS. */
18
+ export function monoSignal(wav: ParsedWav, maxSeconds = MAX_ANALYSIS_SECONDS) {
19
+ const count = Math.min(
20
+ wav.sampleCount,
21
+ Math.floor(maxSeconds * wav.sampleRate),
22
+ );
23
+ const mono = new Float32Array(count);
24
+ for (let frame = 0; frame < count; frame += 1) {
25
+ let sum = 0;
26
+ for (let channel = 0; channel < wav.channels; channel += 1)
27
+ sum += wav.sample(frame, channel);
28
+ mono[frame] = sum / wav.channels;
29
+ }
30
+ return mono;
31
+ }
32
+
33
+ /** Half-wave-rectified log-energy difference per hop: a cheap onset envelope. */
34
+ export function onsetEnvelope(mono: Float32Array, sampleRate: number) {
35
+ const frames = Math.floor(mono.length / HOP);
36
+ const envelope = new Float32Array(frames);
37
+ let previous = 0;
38
+ for (let frame = 0; frame < frames; frame += 1) {
39
+ let energy = 0;
40
+ const start = frame * HOP;
41
+ for (let index = 0; index < HOP; index += 1) {
42
+ const value = mono[start + index]!;
43
+ energy += value * value;
44
+ }
45
+ const level = Math.log1p(energy * 1e3);
46
+ envelope[frame] = Math.max(0, level - previous);
47
+ previous = level;
48
+ }
49
+ // Remove the local mean so sustained loudness does not look like onsets.
50
+ const window = Math.max(1, Math.round((0.25 * sampleRate) / HOP));
51
+ const smoothed = new Float32Array(frames);
52
+ let running = 0;
53
+ for (let frame = 0; frame < frames; frame += 1) {
54
+ running += envelope[frame]!;
55
+ if (frame >= window) running -= envelope[frame - window]!;
56
+ const mean = running / Math.min(window, frame + 1);
57
+ smoothed[frame] = Math.max(0, envelope[frame]! - mean);
58
+ }
59
+ return { envelope: smoothed, hopSeconds: HOP / sampleRate };
60
+ }
61
+
62
+ export type TempoEstimate = Readonly<{
63
+ bpm: number;
64
+ /** 0..1, how far the winning lag stands above the autocorrelation floor. */
65
+ confidence: number;
66
+ /** Seconds of the first beat, from the phase that best matches onsets. */
67
+ offsetSeconds: number;
68
+ }>;
69
+
70
+ /**
71
+ * Autocorrelation over the 60–200 BPM lag range with a mild preference for
72
+ * ~120 BPM, so half/double-time ambiguities resolve toward the usual octave.
73
+ */
74
+ export function estimateTempo(
75
+ envelope: Float32Array,
76
+ hopSeconds: number,
77
+ ): TempoEstimate | undefined {
78
+ const frames = envelope.length;
79
+ // Spread each onset over ±2 frames so a beat period that is not a whole
80
+ // number of hops still correlates with itself (else 2× the period wins).
81
+ envelope = smoothEnvelope(envelope);
82
+ const minLag = Math.max(1, Math.floor(60 / MAX_BPM / hopSeconds));
83
+ const maxLag = Math.ceil(60 / MIN_BPM / hopSeconds);
84
+ if (frames < maxLag * 4) return undefined;
85
+ let total = 0;
86
+ for (const value of envelope) total += value;
87
+ if (!(total > 0)) return undefined;
88
+ const scores = new Float64Array(maxLag + 1);
89
+ let best = -1;
90
+ let bestScore = 0;
91
+ let floor = 0;
92
+ for (let lag = minLag; lag <= maxLag; lag += 1) {
93
+ let sum = 0;
94
+ for (let frame = lag; frame < frames; frame += 1)
95
+ sum += envelope[frame]! * envelope[frame - lag]!;
96
+ const bpm = 60 / (lag * hopSeconds);
97
+ // Gaussian weight centred on 120 BPM in log tempo (Ellis-style prior).
98
+ const weight = Math.exp(-0.5 * (Math.log2(bpm / 120) / 1.0) ** 2);
99
+ const score = (sum / (frames - lag)) * weight;
100
+ scores[lag] = score;
101
+ floor += score;
102
+ if (score > bestScore) {
103
+ bestScore = score;
104
+ best = lag;
105
+ }
106
+ }
107
+ if (best < 0 || bestScore <= 0) return undefined;
108
+ floor /= maxLag - minLag + 1;
109
+ // Refine with a parabolic fit between neighbouring lags.
110
+ let lag = best;
111
+ if (best > minLag && best < maxLag) {
112
+ const left = scores[best - 1]!;
113
+ const right = scores[best + 1]!;
114
+ const denominator = left - 2 * bestScore + right;
115
+ if (denominator < 0) lag = best + (0.5 * (left - right)) / denominator;
116
+ }
117
+ const bpm = Math.round((60 / (lag * hopSeconds)) * 10) / 10;
118
+ const confidence = Math.max(0, Math.min(1, 1 - floor / bestScore));
119
+ // Phase: the offset whose comb of beats collects the most onset energy.
120
+ const period = lag;
121
+ const steps = Math.max(1, Math.round(period));
122
+ let bestOffset = 0;
123
+ let bestComb = -1;
124
+ for (let step = 0; step < steps; step += 1) {
125
+ let comb = 0;
126
+ for (let position = step; position < frames; position += period)
127
+ comb += envelope[Math.round(position)] ?? 0;
128
+ if (comb > bestComb) {
129
+ bestComb = comb;
130
+ bestOffset = step;
131
+ }
132
+ }
133
+ return { bpm, confidence, offsetSeconds: bestOffset * hopSeconds };
134
+ }
135
+
136
+ const SMOOTH_KERNEL = [1, 2, 3, 2, 1].map((weight) => weight / 9);
137
+
138
+ function smoothEnvelope(envelope: Float32Array): Float32Array {
139
+ const out = new Float32Array(envelope.length);
140
+ const half = (SMOOTH_KERNEL.length - 1) / 2;
141
+ for (let frame = 0; frame < envelope.length; frame += 1) {
142
+ let sum = 0;
143
+ for (let offset = -half; offset <= half; offset += 1) {
144
+ const value = envelope[frame + offset];
145
+ if (value !== undefined) sum += value * SMOOTH_KERNEL[offset + half]!;
146
+ }
147
+ out[frame] = sum;
148
+ }
149
+ return out;
150
+ }
151
+
152
+ type Fft = {
153
+ size: number;
154
+ reverse: Uint32Array;
155
+ cos: Float64Array;
156
+ sin: Float64Array;
157
+ };
158
+
159
+ function createFft(size: number): Fft {
160
+ const reverse = new Uint32Array(size);
161
+ const bits = Math.log2(size);
162
+ for (let index = 0; index < size; index += 1) {
163
+ let value = 0;
164
+ for (let bit = 0; bit < bits; bit += 1)
165
+ value |= ((index >> bit) & 1) << (bits - 1 - bit);
166
+ reverse[index] = value;
167
+ }
168
+ const cos = new Float64Array(size / 2);
169
+ const sin = new Float64Array(size / 2);
170
+ for (let index = 0; index < size / 2; index += 1) {
171
+ cos[index] = Math.cos((2 * Math.PI * index) / size);
172
+ sin[index] = -Math.sin((2 * Math.PI * index) / size);
173
+ }
174
+ return { size, reverse, cos, sin };
175
+ }
176
+
177
+ function fftInPlace(fft: Fft, re: Float64Array, im: Float64Array): void {
178
+ const size = fft.size;
179
+ for (let index = 0; index < size; index += 1) {
180
+ const target = fft.reverse[index]!;
181
+ if (target > index) {
182
+ [re[index], re[target]] = [re[target]!, re[index]!];
183
+ [im[index], im[target]] = [im[target]!, im[index]!];
184
+ }
185
+ }
186
+ for (let width = 2; width <= size; width *= 2) {
187
+ const half = width / 2;
188
+ const step = size / width;
189
+ for (let start = 0; start < size; start += width) {
190
+ for (let offset = 0; offset < half; offset += 1) {
191
+ const twiddle = offset * step;
192
+ const cos = fft.cos[twiddle]!;
193
+ const sin = fft.sin[twiddle]!;
194
+ const even = start + offset;
195
+ const odd = even + half;
196
+ const oddRe = re[odd]! * cos - im[odd]! * sin;
197
+ const oddIm = re[odd]! * sin + im[odd]! * cos;
198
+ re[odd] = re[even]! - oddRe;
199
+ im[odd] = im[even]! - oddIm;
200
+ re[even] = re[even]! + oddRe;
201
+ im[even] = im[even]! + oddIm;
202
+ }
203
+ }
204
+ }
205
+ }
206
+
207
+ /**
208
+ * 12-bin pitch-class histogram from sparse 4096-point frames (one every
209
+ * half second) over 55–1760 Hz, weighted by log magnitude.
210
+ */
211
+ export function pitchClassHistogram(mono: Float32Array, sampleRate: number) {
212
+ const size = 4096;
213
+ const histogram = new Array<number>(12).fill(0);
214
+ if (mono.length < size) return histogram;
215
+ const fft = createFft(size);
216
+ const window = new Float64Array(size);
217
+ for (let index = 0; index < size; index += 1)
218
+ window[index] = 0.5 - 0.5 * Math.cos((2 * Math.PI * index) / (size - 1));
219
+ const re = new Float64Array(size);
220
+ const im = new Float64Array(size);
221
+ const hop = Math.max(size, Math.round(sampleRate / 2));
222
+ const binHz = sampleRate / size;
223
+ const lowBin = Math.max(1, Math.floor(55 / binHz));
224
+ const highBin = Math.min(size / 2 - 1, Math.ceil(1760 / binHz));
225
+ for (let start = 0; start + size <= mono.length; start += hop) {
226
+ for (let index = 0; index < size; index += 1) {
227
+ re[index] = mono[start + index]! * window[index]!;
228
+ im[index] = 0;
229
+ }
230
+ fftInPlace(fft, re, im);
231
+ for (let bin = lowBin; bin <= highBin; bin += 1) {
232
+ const magnitude = Math.hypot(re[bin]!, im[bin]!);
233
+ if (magnitude < 1e-4) continue;
234
+ // Only spectral peaks vote, so harmonics spread less onto neighbours.
235
+ const left = Math.hypot(re[bin - 1]!, im[bin - 1]!);
236
+ const right = Math.hypot(re[bin + 1]!, im[bin + 1]!);
237
+ if (magnitude < left || magnitude < right) continue;
238
+ const midi = 69 + 12 * Math.log2((bin * binHz) / 440);
239
+ const pitchClass = ((Math.round(midi) % 12) + 12) % 12;
240
+ histogram[pitchClass] =
241
+ histogram[pitchClass]! + Math.log1p(magnitude * 100);
242
+ }
243
+ }
244
+ return histogram;
245
+ }
246
+
247
+ export function keyFromHistogram(histogram: readonly number[]): string | null {
248
+ return estimateKey(histogram);
249
+ }
250
+
251
+ /** 240 normalised RMS buckets over the whole file, rounded to 3 decimals. */
252
+ export function waveformPeaks(
253
+ wav: ParsedWav,
254
+ buckets = MEDIA_LIMITS.waveformBuckets,
255
+ ): readonly number[] {
256
+ const count = Math.max(1, Math.min(buckets, wav.sampleCount));
257
+ const peaks = new Array<number>(count).fill(0);
258
+ const perBucket = wav.sampleCount / count;
259
+ let maximum = 0;
260
+ for (let bucket = 0; bucket < count; bucket += 1) {
261
+ const start = Math.floor(bucket * perBucket);
262
+ const end = Math.max(start + 1, Math.floor((bucket + 1) * perBucket));
263
+ // Stride long buckets so a 10-minute file costs the same as a short one.
264
+ const stride = Math.max(1, Math.floor((end - start) / 2_048));
265
+ let energy = 0;
266
+ let samples = 0;
267
+ for (let frame = start; frame < end; frame += stride) {
268
+ for (let channel = 0; channel < wav.channels; channel += 1) {
269
+ const value = wav.sample(frame, channel);
270
+ energy += value * value;
271
+ samples += 1;
272
+ }
273
+ }
274
+ const rms = samples > 0 ? Math.sqrt(energy / samples) : 0;
275
+ peaks[bucket] = rms;
276
+ if (rms > maximum) maximum = rms;
277
+ }
278
+ return peaks.map((value) =>
279
+ maximum > 0 ? Math.round((value / maximum) * 1000) / 1000 : 0,
280
+ );
281
+ }
@@ -0,0 +1,130 @@
1
+ /**
2
+ * `import_sample(file, name, {begin?, end?, root?})`: ffmpeg converts any
3
+ * audio file to a 48 kHz PCM16 wav at `tracks/<slug>/samples/<name>.wav`
4
+ * and the result carries the `sampler({...})` snippet for `track.ts`. The
5
+ * sampler schema itself belongs to the project-format lane; this tool only
6
+ * copies the file and returns the snippet.
7
+ */
8
+ import { SCORE_LIMITS } from "../../core/score.ts";
9
+ import { join } from "node:path";
10
+ import { missingTool } from "./backend.ts";
11
+ import { MediaToolError, runHelper } from "./process.ts";
12
+ import {
13
+ ensureDir,
14
+ formatBytes,
15
+ projectPath,
16
+ resolveInput,
17
+ samplesDir,
18
+ sha256File,
19
+ } from "./paths.ts";
20
+ import {
21
+ MEDIA_LIMITS,
22
+ type MediaResult,
23
+ type MediaRunContext,
24
+ } from "./types.ts";
25
+ import { parseWav } from "./vendor/wav.ts";
26
+
27
+ export const SAMPLE_LIMITS = Object.freeze({
28
+ maxBytes: SCORE_LIMITS.maxSampleFileBytes,
29
+ maxSeconds: SCORE_LIMITS.maxSampleSeconds,
30
+ sampleRate: 48_000,
31
+ });
32
+
33
+ export type ImportSampleArgs = Readonly<{
34
+ file: string;
35
+ name: string;
36
+ /** Loop/playback window as fractions 0..1 of the file. */
37
+ begin?: number;
38
+ end?: number;
39
+ /** Root note name or MIDI number the sample is pitched at (default C4). */
40
+ root?: string | number;
41
+ }>;
42
+
43
+ export async function importSample(
44
+ args: ImportSampleArgs,
45
+ context: MediaRunContext,
46
+ ): Promise<MediaResult> {
47
+ if (!context.runner.which("ffmpeg"))
48
+ throw new MediaToolError(missingTool("ffmpeg"));
49
+ const input = await resolveInput(context, args.file);
50
+ const dir = await ensureDir(samplesDir(context));
51
+ const target = join(dir, `${args.name}.wav`);
52
+ context.progress(`ffmpeg → samples/${args.name}.wav`);
53
+ const temp = `${target}.part.wav`;
54
+ await runHelper(
55
+ context,
56
+ [
57
+ "ffmpeg",
58
+ "-hide_banner",
59
+ "-nostdin",
60
+ "-loglevel",
61
+ "error",
62
+ "-y",
63
+ "-i",
64
+ input.absolute,
65
+ "-vn",
66
+ "-t",
67
+ String(SAMPLE_LIMITS.maxSeconds),
68
+ "-ar",
69
+ String(SAMPLE_LIMITS.sampleRate),
70
+ "-ac",
71
+ "2",
72
+ "-c:a",
73
+ "pcm_s16le",
74
+ temp,
75
+ ],
76
+ { timeoutMs: MEDIA_LIMITS.sampleTimeoutMs },
77
+ );
78
+ const file = Bun.file(temp);
79
+ if (file.size > SAMPLE_LIMITS.maxBytes) {
80
+ await file.delete().catch(() => undefined);
81
+ throw new MediaToolError(
82
+ `${args.name}.wav would be ${formatBytes(file.size)}, over the ${formatBytes(SAMPLE_LIMITS.maxBytes)} sample cap; trim with begin/end on a shorter source`,
83
+ );
84
+ }
85
+ const wav = parseWav(new Uint8Array(await file.arrayBuffer()), {
86
+ maximumBytes: SAMPLE_LIMITS.maxBytes,
87
+ maximumDurationSeconds: SAMPLE_LIMITS.maxSeconds + 1,
88
+ maximumChannels: 2,
89
+ });
90
+ const { rename } = await import("node:fs/promises");
91
+ await rename(temp, target);
92
+ const sha256 = await sha256File(target);
93
+ const durationSeconds =
94
+ Math.round((wav.sampleCount / wav.sampleRate) * 1000) / 1000;
95
+ const src = `samples/${args.name}.wav`;
96
+ const ref: Record<string, unknown> = { src };
97
+ if (args.root !== undefined) ref.root = args.root;
98
+ if (args.begin !== undefined) ref.begin = args.begin;
99
+ if (args.end !== undefined) ref.end = args.end;
100
+ const inner =
101
+ Object.keys(ref).length === 1
102
+ ? JSON.stringify(src)
103
+ : `{ ${Object.entries(ref)
104
+ .map(([key, value]) => `${key}: ${JSON.stringify(value)}`)
105
+ .join(", ")} }`;
106
+ // A root means the sample is pitched: keyed mode resamples it across notes.
107
+ const snippet = `instrument: sampler({ ${args.name}: ${inner} }${args.root !== undefined ? ', { mode: "keyed" }' : ""})`;
108
+ const relativePath = projectPath(context.projectRoot, target);
109
+ return {
110
+ summary: `imported ${src} (${durationSeconds}s, ${formatBytes(file.size)})`,
111
+ outputs: [relativePath],
112
+ content: {
113
+ sample: {
114
+ name: args.name,
115
+ src,
116
+ path: relativePath,
117
+ sha256,
118
+ durationSeconds,
119
+ sampleRate: wav.sampleRate,
120
+ channels: wav.channels,
121
+ bytes: file.size,
122
+ ...(args.root !== undefined ? { root: args.root } : {}),
123
+ ...(args.begin !== undefined ? { begin: args.begin } : {}),
124
+ ...(args.end !== undefined ? { end: args.end } : {}),
125
+ },
126
+ snippet,
127
+ note: "Paste the snippet into the track's instrument in track.ts; the sampler schema (voices, root, begin, end, gain, loop, choke) is defined in DAWG.md.",
128
+ },
129
+ };
130
+ }
@@ -0,0 +1,201 @@
1
+ /**
2
+ * `transcribe_lyrics(file, {lang?})`: whisper.cpp (`whisper-cli`) on a 16 kHz
3
+ * mono copy of the file. The ggml model lives in `~/.cache/dawg/whisper/`;
4
+ * when it is missing a progress card names the size and destination before
5
+ * dawg fetches it from Hugging Face. Output: `<name>.lyrics.json` (segments
6
+ * with seconds) and `<name>.lyrics.txt`.
7
+ */
8
+ import { homedir } from "node:os";
9
+ import { join } from "node:path";
10
+ import { missingTool } from "./backend.ts";
11
+ import {
12
+ ensureDir,
13
+ exists,
14
+ formatBytes,
15
+ projectPath,
16
+ resolveInput,
17
+ writeJsonAtomic,
18
+ } from "./paths.ts";
19
+ import {
20
+ MediaToolError,
21
+ downloadToFile,
22
+ runHelper,
23
+ throwIfAborted,
24
+ withTempDir,
25
+ } from "./process.ts";
26
+ import {
27
+ MEDIA_LIMITS,
28
+ type MediaResult,
29
+ type MediaRunContext,
30
+ } from "./types.ts";
31
+ import { finiteNumber, isRecord, optionalString } from "./vendor/util.ts";
32
+
33
+ export const WHISPER_MODEL_BASE_URL =
34
+ "https://huggingface.co/ggerganov/whisper.cpp/resolve/main/";
35
+ const MODEL_MAX_BYTES = 200 * 1024 * 1024;
36
+ const MAX_SEGMENTS = 4_000;
37
+ const MAX_TEXT_CHARS = 4_000;
38
+ const LANGUAGE_PATTERN = /^[a-z]{2,3}$|^auto$/;
39
+
40
+ export type LyricsArgs = Readonly<{ file: string; lang?: string }>;
41
+
42
+ export function whisperModelName(lang: string | undefined): string {
43
+ return lang === undefined || lang === "en"
44
+ ? "ggml-base.en.bin"
45
+ : "ggml-base.bin";
46
+ }
47
+
48
+ export function whisperModelDir(homeDir: string | undefined): string {
49
+ return join(homeDir ?? homedir(), ".cache", "dawg", "whisper");
50
+ }
51
+
52
+ export type LyricSegment = Readonly<{
53
+ start: number;
54
+ end: number;
55
+ text: string;
56
+ }>;
57
+
58
+ /** Parse whisper-cli's `-oj` JSON (`transcription[].offsets` are milliseconds). */
59
+ export function parseWhisperJson(value: unknown): readonly LyricSegment[] {
60
+ if (!isRecord(value) || !Array.isArray(value.transcription)) return [];
61
+ const segments: LyricSegment[] = [];
62
+ for (const entry of value.transcription.slice(0, MAX_SEGMENTS)) {
63
+ if (!isRecord(entry)) continue;
64
+ const offsets = isRecord(entry.offsets) ? entry.offsets : {};
65
+ const from = finiteNumber(offsets.from);
66
+ const to = finiteNumber(offsets.to);
67
+ const text = optionalString(entry.text, 2_000);
68
+ if (from === undefined || to === undefined || !text) continue;
69
+ segments.push({ start: from / 1000, end: to / 1000, text });
70
+ }
71
+ return segments;
72
+ }
73
+
74
+ export async function transcribeLyrics(
75
+ args: LyricsArgs,
76
+ context: MediaRunContext,
77
+ ): Promise<MediaResult> {
78
+ if (args.lang !== undefined && !LANGUAGE_PATTERN.test(args.lang))
79
+ throw new MediaToolError(
80
+ "lang must be a two- or three-letter code or auto",
81
+ );
82
+ if (!context.runner.which("whisper-cli"))
83
+ throw new MediaToolError(missingTool("whisper-cli"));
84
+ if (!context.runner.which("ffmpeg"))
85
+ throw new MediaToolError(missingTool("ffmpeg"));
86
+ const input = await resolveInput(context, args.file);
87
+ const modelName = whisperModelName(args.lang);
88
+ const modelDir = whisperModelDir(context.homeDir ?? context.env?.HOME);
89
+ const modelPath = join(modelDir, modelName);
90
+ if (!(await exists(modelPath))) {
91
+ await ensureDir(modelDir);
92
+ context.progress(
93
+ `downloading ${modelName} (~${formatBytes(MEDIA_LIMITS.whisperModelBytes)}) to ${modelDir} from huggingface.co/ggerganov/whisper.cpp`,
94
+ );
95
+ await downloadToFile(
96
+ context.fetch ?? fetch,
97
+ `${WHISPER_MODEL_BASE_URL}${modelName}`,
98
+ modelPath,
99
+ {
100
+ maxBytes: MODEL_MAX_BYTES,
101
+ signal: context.signal,
102
+ progress: (received, total) =>
103
+ context.progress(
104
+ total
105
+ ? `model ${Math.round((received / total) * 100)}%`
106
+ : `model ${formatBytes(received)}`,
107
+ ),
108
+ },
109
+ );
110
+ }
111
+ throwIfAborted(context.signal);
112
+ return withTempDir("lyrics", async (temp) => {
113
+ const pcm = join(temp, "audio16k.wav");
114
+ context.progress("ffmpeg → 16 kHz mono");
115
+ await runHelper(
116
+ context,
117
+ [
118
+ "ffmpeg",
119
+ "-hide_banner",
120
+ "-nostdin",
121
+ "-loglevel",
122
+ "error",
123
+ "-y",
124
+ "-i",
125
+ input.absolute,
126
+ "-vn",
127
+ "-t",
128
+ "3600",
129
+ "-ar",
130
+ "16000",
131
+ "-ac",
132
+ "1",
133
+ "-c:a",
134
+ "pcm_s16le",
135
+ pcm,
136
+ ],
137
+ { timeoutMs: MEDIA_LIMITS.sampleTimeoutMs },
138
+ );
139
+ const outBase = join(temp, "transcript");
140
+ context.progress("whisper transcribing");
141
+ await runHelper(
142
+ context,
143
+ [
144
+ "whisper-cli",
145
+ "-m",
146
+ modelPath,
147
+ "-f",
148
+ pcm,
149
+ "-l",
150
+ args.lang ?? "en",
151
+ "-np",
152
+ "-oj",
153
+ "-of",
154
+ outBase,
155
+ ],
156
+ {
157
+ timeoutMs: MEDIA_LIMITS.lyricsTimeoutMs,
158
+ progress: (chunk) => {
159
+ const match = chunk.match(/progress\s*=\s*(\d{1,3})%/);
160
+ return match ? `whisper ${match[1]}%` : undefined;
161
+ },
162
+ },
163
+ );
164
+ const jsonFile = Bun.file(`${outBase}.json`);
165
+ if (jsonFile.size > 32 * 1024 * 1024)
166
+ throw new MediaToolError("whisper output too large");
167
+ let parsed: unknown;
168
+ try {
169
+ parsed = JSON.parse(await jsonFile.text()) as unknown;
170
+ } catch {
171
+ throw new MediaToolError("whisper-cli wrote no JSON transcript");
172
+ }
173
+ const segments = parseWhisperJson(parsed);
174
+ const base = input.absolute.replace(/\.[A-Za-z0-9]{1,5}$/, "");
175
+ const jsonOut = `${base}.lyrics.json`;
176
+ const textOut = `${base}.lyrics.txt`;
177
+ const text = segments.map((segment) => segment.text.trim()).join("\n");
178
+ await writeJsonAtomic(jsonOut, {
179
+ file: input.relative,
180
+ model: modelName,
181
+ language: args.lang ?? "en",
182
+ segments,
183
+ });
184
+ await Bun.write(textOut, `${text}\n`);
185
+ const outputs = [jsonOut, textOut].map((path) =>
186
+ projectPath(context.projectRoot, path),
187
+ );
188
+ return {
189
+ summary: `transcribed ${segments.length} segments from ${input.relative}`,
190
+ outputs,
191
+ content: {
192
+ file: input.relative,
193
+ lyrics: outputs[0],
194
+ text: outputs[1],
195
+ segments: segments.length,
196
+ preview: text.slice(0, MAX_TEXT_CHARS),
197
+ ...(text.length > MAX_TEXT_CHARS ? { truncated: true } : {}),
198
+ },
199
+ };
200
+ });
201
+ }