reelkit-cli 0.3.1 → 0.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/skill/SKILL.md +3 -0
- package/skill/reference/clips.md +1 -0
- package/skill/reference/references.md +32 -0
- package/src/cli.ts +12 -0
- package/src/commands/assets.ts +13 -6
- package/src/commands/auth.ts +1 -1
- package/src/commands/plan.ts +22 -2
- package/src/commands/ref.ts +307 -0
- package/src/contract/index.ts +25 -1
- package/src/pipeline/schema.ts +6 -0
- package/src/project/beats.ts +130 -0
- package/src/project/chromakey.ts +14 -4
- package/src/project/refmeasure.ts +148 -0
- package/src/testing/conformance.ts +107 -1
- package/src/testing/fake-api.ts +44 -6
package/package.json
CHANGED
package/skill/SKILL.md
CHANGED
|
@@ -25,6 +25,8 @@ For each file the user gives, look at it, then register it with a description of
|
|
|
25
25
|
`reelkit assets upload ./logo.png --describe "Acme logo, white wordmark on blue"`
|
|
26
26
|
Add `--footage` for a video the motion design should be laid over. User files stay on this machine. A video of someone or something on a plain background can be made transparent: `--green` keys a green background locally for free, and `--cutout` removes any background on the server (the video is sent to Reelkit, so ask first; see `reference/clips.md`).
|
|
27
27
|
|
|
28
|
+
If the user points at an existing video ("make one like this", a link or a file), it is a reference: you learn how it is built and make something new in that spirit. Read `reference/references.md`, then `reelkit ref download <url-or-file>` and `reelkit ref analyze <id>`, and look at the frames it saves. Take its structure, pace and motion; never its footage, music or words. A link is fetched on the user's machine and they are responsible for the right to download it; ask before running `reelkit ref analyze` without `--no-transcript`, because the audio (never the video) is sent to Reelkit to be transcribed.
|
|
29
|
+
|
|
28
30
|
### 2. Look
|
|
29
31
|
Read `reference/styles.md` and agree the video's look with the user: one of its looks, or their own. If they already described what they want, match it and confirm in one sentence. Ask anything still open in one message, each question with a default. If any on-screen text will be Hebrew, read `reference/hebrew-rtl.md` now: it changes how words may enter and how lines are written.
|
|
30
32
|
|
|
@@ -63,6 +65,7 @@ Write `plan.json`:
|
|
|
63
65
|
- `clipPrompt` is set only for `clip` scenes (a generated or reused video clip is the scene's picture; the rest of the scene uses `imageTags` and `shareable` as an illustration does). Clips are scarce: most videos have none or one or two.
|
|
64
66
|
- `userAssetIds` lists the ids of the user's files shown in that scene.
|
|
65
67
|
- `pace` is `slow`, `normal` or `fast`.
|
|
68
|
+
- When the video follows a reference, add `"reference": { "id": "<id>", "take": ["fast cuts every ~1.2s"] }` (1 to 6 notes on what you took).
|
|
66
69
|
|
|
67
70
|
Run `reelkit plan check`; it prints the estimated length to tell the user. Fix everything under "Fix these". Act on "Worth improving" unless you have a good reason not to.
|
|
68
71
|
|
package/skill/reference/clips.md
CHANGED
|
@@ -35,6 +35,7 @@ A clip is a few seconds of generated video. It is the most expensive thing Reelk
|
|
|
35
35
|
A green-screen clip is a subject filmed on a flat pure green background. It is for a person or an object that you place over something of your own: a presenter pointing at the user's screenshot, a character beside your title, a product over a gradient.
|
|
36
36
|
- Generate it with `reelkit assets gen clip --scene <sceneId> --green`. The original is saved as `assets/clip-<sceneId>.mp4` and the green is keyed out locally into `assets/clip-<sceneId>.webm`, a video with a transparent background. The manifest has both: `s.clipKey` and `s.clipKeyedKey`. Use the `.webm`.
|
|
37
37
|
- Prompt it as a subject, not a scene: "A woman in a blue jacket, waist up, waves and then points to her left. Flat, evenly lit, pure green background (#00FF00)." Keep the whole subject inside the frame with room around it; no green anywhere on the subject (clothes, props, eyes); no shadows, floor or reflections on the background; no camera move, because keying a moving frame edge flickers.
|
|
38
|
+
- A video model rarely produces a true chroma green; it usually gives a soft sage. When the green is too dull to key cleanly, the command does not key it: it cuts the subject out on the server instead (see "No green screen" below), says so, and uses cutout seconds from the quota. The result is used the same way.
|
|
38
39
|
- If the keyed edges look green or ragged in the preview, the prompt was at fault: regenerate once with a flatter green and an evenly lit subject, using `--redo`. A clip left unkeyed (the command says so) is keyed again for free by running the same command again.
|
|
39
40
|
- Layer it above the scene's background and below the captions:
|
|
40
41
|
|
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: references
|
|
3
|
+
description: Use when the user says "make one like this" or gives a link or file of an existing video to match - how to bring it in, measure it, look at it, and turn how it is built into a plan, without copying anything from it.
|
|
4
|
+
---
|
|
5
|
+
|
|
6
|
+
# References: "make one like this"
|
|
7
|
+
|
|
8
|
+
A reference is an existing video the user wants the new one to feel like. You learn how it is built (its structure, pace and motion) and make something new in that spirit. Nothing of it is ever reused.
|
|
9
|
+
|
|
10
|
+
## Bring it in
|
|
11
|
+
1. Ask for the video as a file if the user has it. A file on this machine is the clean path: `reelkit ref download ./example.mp4` copies it. A link works too, but it is fetched on the user's machine with yt-dlp (which they must have installed; the command says how if not), and the user is responsible for having the right to download that video. Say so in one sentence when you use a link. A reference is at most 10 minutes.
|
|
12
|
+
2. `reelkit ref analyze <id>` measures it on this machine: scenes, cut times, pace, colours, loudness and tempo. It writes `refs/<id>/breakdown.json` and two sample frames per scene in `refs/<id>/frames/`. `reelkit ref list` shows what is in the project.
|
|
13
|
+
3. Ask before it sends anything. The transcript is made on the Reelkit server, so the reference's audio (never the video) leaves the user's machine and is deleted there when the job ends. Ask "I can also transcribe its audio on the Reelkit server to understand the structure: OK?" and run `reelkit ref analyze <id> --no-transcript` if they say no. `reelkit whoami` shows the monthly transcription seconds left. If the transcript is refused (quota, not logged in), the breakdown is still saved; `reelkit ref analyze <id> --redo` adds the transcript later.
|
|
14
|
+
|
|
15
|
+
## Look, then read
|
|
16
|
+
Open the frames in `refs/<id>/frames/` and look at them one scene after another: how big the type is, where it sits, how much of the screen is picture and how much is colour, how busy each scene is. Then read `breakdown.json`: `pacing.averageShotSec`, `pacing.firstCutSec`, `pacing.cutsPerSecond`, the `palette`, and the `transcript` if there is one.
|
|
17
|
+
|
|
18
|
+
## What to take, and what never to
|
|
19
|
+
Take:
|
|
20
|
+
- the structure: how it opens, how many beats it has, how it ends
|
|
21
|
+
- the pace and the cut rhythm
|
|
22
|
+
- how type and layout are used (big words on colour, small captions over footage, one idea per screen)
|
|
23
|
+
- the kind of motion (snappy, floaty, hard cuts, slow pushes)
|
|
24
|
+
- the mood of its music, as a description to search the library with
|
|
25
|
+
|
|
26
|
+
Never take its footage, its music, its exact words, its logos or its faces. The transcript is for understanding the structure (where the hook ends, where the point lands), never for copying the script: write your own words about the user's subject.
|
|
27
|
+
|
|
28
|
+
## Turn the pace into the plan
|
|
29
|
+
- Scene count and length come from `pacing.averageShotSec`: a video of 30 seconds with shots of about 1.5 seconds is many quick scenes; since a plan has 3 to 8 scenes, group several reference shots into one scene that changes its picture inside (stacked text, quick swaps) rather than stretching it.
|
|
30
|
+
- The hook's length comes from `pacing.firstCutSec`: if the reference holds its first shot for 1.2 seconds, your first scene is one short line.
|
|
31
|
+
- Write the plan's `reference`: `{ "id": "<id>", "take": ["fast cuts every ~1.2s", "big type on colour fields"] }`, one to six short notes of what you took. `reelkit plan check` needs the analysis to exist and says when your scenes are much slower or faster than the reference.
|
|
32
|
+
- When the reference cuts on the beat (`pacing.cutsOnBeat` high, say 0.7 or more, with `audio.tempoBpm`), plan the new video's scene changes on a regular grid at a similar tempo, and search the library for music of a similar tempo and mood. The reference's own music is never used.
|
package/src/cli.ts
CHANGED
|
@@ -7,6 +7,7 @@ import { init } from "./commands/init";
|
|
|
7
7
|
import { install } from "./commands/install";
|
|
8
8
|
import { openInBrowser } from "./open";
|
|
9
9
|
import { planCheck } from "./commands/plan";
|
|
10
|
+
import { refAnalyze, refAudio, refDownload, refList } from "./commands/ref";
|
|
10
11
|
import type { Ctx, Result } from "./context";
|
|
11
12
|
import { readFileSync } from "node:fs";
|
|
12
13
|
|
|
@@ -90,6 +91,17 @@ gen.command("clip [prompt]").description("Generate a scene's video clip (takes m
|
|
|
90
91
|
.option("--share", "also add it to the shared library (only a generic clip)").option("--redo", "generate a new clip even if the scene has one").option("--resume <id>", "keep waiting for a clip already started, without paying again")
|
|
91
92
|
.action(run((ctx, prompt, opts) => assetsGenClip(ctx, prompt, opts)));
|
|
92
93
|
|
|
94
|
+
const ref = program.command("ref").description("Use an existing video as a reference for how a new one should feel (it is measured here; nothing of it goes into the output)");
|
|
95
|
+
ref.command("download <url-or-file>").description("Bring a video into the project as a reference: a link is fetched on this machine with yt-dlp (you need the right to download it), a file is copied")
|
|
96
|
+
.option("--redo", "download a link again even if it is already in the project")
|
|
97
|
+
.action(run((ctx, input: string, opts: { redo?: boolean }) => refDownload(ctx, input, opts)));
|
|
98
|
+
ref.command("audio <id>").description("Save the reference's audio as refs/<id>/audio.mp3 (on this machine)").action(run((ctx, id: string) => refAudio(ctx, id)));
|
|
99
|
+
ref.command("analyze <id>").description("Measure how the reference is built: scenes, cuts, pacing, colours, sound and tempo, with sample frames. Its audio (never the video) is sent to the Reelkit server to be transcribed unless --no-transcript")
|
|
100
|
+
.option("--no-transcript", "do not send the audio to the Reelkit server; everything else is measured here")
|
|
101
|
+
.option("--redo", "measure again even if a breakdown exists")
|
|
102
|
+
.action(run((ctx, id: string, opts: { transcript?: boolean; redo?: boolean }) => refAnalyze(ctx, id, opts)));
|
|
103
|
+
ref.command("list").description("List the references in this project").action(run((ctx) => refList(ctx)));
|
|
104
|
+
|
|
93
105
|
program.command("plan").description("Work with plan.json").command("check").description("Validate plan.json and list what to fix or improve").action(run(planCheck));
|
|
94
106
|
|
|
95
107
|
program.command("check").description("Check the composition in src/ without rendering").action(run(check));
|
package/src/commands/assets.ts
CHANGED
|
@@ -5,7 +5,7 @@ import { ApiFailure, download, uploadTo } from "../api/client";
|
|
|
5
5
|
import { client, type Ctx, type Result } from "../context";
|
|
6
6
|
import { LibraryKindSchema, MAX_CUTOUT_SECONDS, MAX_UPLOAD_BYTES, UploadKindSchema, type CutoutType } from "../contract";
|
|
7
7
|
import { PACE_SPEED, type AssetManifest, type AssetRecord, type ScenePlan } from "../pipeline/schema";
|
|
8
|
-
import { keyGreen } from "../project/chromakey";
|
|
8
|
+
import { GreenTooDullError, keyGreen } from "../project/chromakey";
|
|
9
9
|
import { buildManifest, voiceoverStale, type ClipRecord, type Voiceover } from "../project/manifest";
|
|
10
10
|
import { loadPlan } from "./plan";
|
|
11
11
|
import { FILE_NAME } from "../render/validate";
|
|
@@ -388,6 +388,10 @@ export async function assetsGenClip(
|
|
|
388
388
|
if (opts.green && have.greenScreen && !have.keyedKey) {
|
|
389
389
|
// The clip was paid for but not keyed (ffmpeg failed or --green came later): keying is free, so do it now.
|
|
390
390
|
const keyed = await keyScene(project, scene.id, have, key);
|
|
391
|
+
if (!keyed.ok && (keyed.data as { dull?: boolean } | undefined)?.dull) {
|
|
392
|
+
const cut = await cutoutScene(ctx, project, scene.id, have, now);
|
|
393
|
+
return { ...cut, data: { path: have.key, ...(cut.ok ? { keyedPath: keyedKey } : {}) }, summary: cut.ok ? `The green in scene ${scene.id}'s clip was too dull to key on this machine, so the subject was cut out on the server instead. ${cut.summary}` : cut.summary };
|
|
394
|
+
}
|
|
391
395
|
if (!keyed.ok) return keyed;
|
|
392
396
|
return { ok: true, data: { path: have.key, keyedPath: keyedKey }, summary: `Keyed the green out of scene ${scene.id}'s clip into ${keyedKey}.` };
|
|
393
397
|
}
|
|
@@ -421,20 +425,23 @@ export async function assetsGenClip(
|
|
|
421
425
|
const record: ClipRecord = { key: path, greenScreen: green, durationSec: st.durationSec ?? seconds };
|
|
422
426
|
setSceneClip(project, scene.id, record);
|
|
423
427
|
let keyedPath: string | undefined;
|
|
428
|
+
let dullGreen = false;
|
|
424
429
|
if (green || opts.green) {
|
|
425
430
|
// The paid clip is already saved above; if keying fails, running the command again keys it without a new charge.
|
|
426
431
|
const keyed = await keyScene(project, scene.id, { ...record, greenScreen: true }, key);
|
|
427
|
-
|
|
428
|
-
|
|
432
|
+
// A model's green is often too dull to key. The subject is then cut out on the server, which needs no green at all.
|
|
433
|
+
if (!keyed.ok && (keyed.data as { dull?: boolean } | undefined)?.dull) dullGreen = true;
|
|
434
|
+
else if (!keyed.ok) return keyed;
|
|
435
|
+
else keyedPath = keyedKey;
|
|
429
436
|
}
|
|
430
437
|
let cutoutNote: string | undefined;
|
|
431
|
-
if (opts.cutout) {
|
|
438
|
+
if (opts.cutout || dullGreen) {
|
|
432
439
|
// The paid clip is already saved above; if the cutout fails, running the command again does only the cutout.
|
|
433
440
|
const cut = await cutoutScene(ctx, project, scene.id, record, now);
|
|
434
441
|
const shared = st.libraryId ? `, and shared it to the library as ${st.libraryId} (awaiting review)` : "";
|
|
435
442
|
if (!cut.ok) return { ...cut, data: { id, path, durationSec: record.durationSec, libraryId: st.libraryId ?? null }, summary: `Generated the clip for scene ${scene.id}${shared}. ${cut.summary}` };
|
|
436
443
|
keyedPath = keyedKey;
|
|
437
|
-
cutoutNote = cut.summary;
|
|
444
|
+
cutoutNote = dullGreen ? `The generated green was too dull to key on this machine, so the subject was cut out on the server instead. ${cut.summary}` : cut.summary;
|
|
438
445
|
}
|
|
439
446
|
const { note } = tryManifest(project, plan);
|
|
440
447
|
return {
|
|
@@ -508,7 +515,7 @@ async function keyScene(project: Project, sceneId: string, record: ClipRecord, k
|
|
|
508
515
|
try {
|
|
509
516
|
await key(project.path(record.key), project.path(keyedKey));
|
|
510
517
|
} catch (e) {
|
|
511
|
-
return { ok: false, data: { path: record.key }, summary: `The clip is saved at ${record.key}, but the green could not be keyed out: ${e instanceof Error ? e.message : String(e)}` };
|
|
518
|
+
return { ok: false, data: { path: record.key, ...(e instanceof GreenTooDullError ? { dull: true } : {}) }, summary: `The clip is saved at ${record.key}, but the green could not be keyed out: ${e instanceof Error ? e.message : String(e)}` };
|
|
512
519
|
}
|
|
513
520
|
setSceneClip(project, sceneId, { ...record, greenScreen: true, keyedKey });
|
|
514
521
|
return { ok: true, summary: "" };
|
package/src/commands/auth.ts
CHANGED
|
@@ -113,6 +113,6 @@ export async function whoami(ctx: Ctx): Promise<Result> {
|
|
|
113
113
|
const q = me.quota;
|
|
114
114
|
return {
|
|
115
115
|
ok: true, data: me,
|
|
116
|
-
summary: `${me.handle}\nVoiceover: ${q.voiceoverChars.used}/${q.voiceoverChars.limit} characters\nImages: ${q.images.used}/${q.images.limit}\nClips: ${q.clips.used}/${q.clips.limit}\nCutouts: ${q.cutoutSeconds.used}/${q.cutoutSeconds.limit} seconds\nResets: ${q.resetsAt.slice(0, 10)}\nContributions: ${me.contributions}`,
|
|
116
|
+
summary: `${me.handle}\nVoiceover: ${q.voiceoverChars.used}/${q.voiceoverChars.limit} characters\nImages: ${q.images.used}/${q.images.limit}\nClips: ${q.clips.used}/${q.clips.limit}\nCutouts: ${q.cutoutSeconds.used}/${q.cutoutSeconds.limit} seconds\nTranscription: ${q.transcribeSeconds.used}/${q.transcribeSeconds.limit} seconds\nResets: ${q.resetsAt.slice(0, 10)}\nContributions: ${me.contributions}`,
|
|
117
117
|
};
|
|
118
118
|
}
|
package/src/commands/plan.ts
CHANGED
|
@@ -15,6 +15,25 @@ export function loadPlan(project: Project): ScenePlan {
|
|
|
15
15
|
return parsed.data;
|
|
16
16
|
}
|
|
17
17
|
|
|
18
|
+
// What a plan that follows a reference must satisfy: the reference has to be measured, and the pace should be of the same order.
|
|
19
|
+
function referenceChecks(project: Project, plan: ScenePlan): { mustFix: string[]; improve: string[] } {
|
|
20
|
+
if (!plan.reference) return { mustFix: [], improve: [] };
|
|
21
|
+
const id = plan.reference.id;
|
|
22
|
+
if (!/^r-[0-9a-f]{8}$/.test(id) || !project.exists(`refs/${id}/ref.json`))
|
|
23
|
+
return { mustFix: [`reference ${id} is not in this project. Run \`reelkit ref download <url-or-file>\` and use the id it prints, or remove "reference" from plan.json.`], improve: [] };
|
|
24
|
+
const file = `refs/${id}/breakdown.json`;
|
|
25
|
+
if (!project.exists(file)) return { mustFix: [`reference ${id} has not been analysed yet. Run \`reelkit ref analyze ${id}\`.`], improve: [] };
|
|
26
|
+
let average: unknown;
|
|
27
|
+
try { average = project.readJson<{ pacing?: { averageShotSec?: unknown } }>(file).pacing?.averageShotSec; } catch { average = undefined; }
|
|
28
|
+
if (typeof average !== "number" || !(average > 0)) return { mustFix: [`${file} has no average shot length. Run \`reelkit ref analyze ${id} --redo\`.`], improve: [] };
|
|
29
|
+
const ours = estimateLength(plan).seconds / plan.scenes.length;
|
|
30
|
+
const ratio = ours / average;
|
|
31
|
+
const improve = ratio > 2 || ratio < 0.5
|
|
32
|
+
? [`The plan's scenes average ${ours.toFixed(1)}s but the reference's shots average ${average.toFixed(1)}s: ${ratio > 2 ? "the new video will feel much slower" : "the new video will feel much faster"}. Change the number of scenes or the narration length, or say in reference.take why the pace differs.`]
|
|
33
|
+
: [];
|
|
34
|
+
return { mustFix: [], improve };
|
|
35
|
+
}
|
|
36
|
+
|
|
18
37
|
// Hard rules (mustFix) stop the video; soft notes (shouldImprove) are advice for a better script.
|
|
19
38
|
export async function planCheck(ctx: Ctx): Promise<Result> {
|
|
20
39
|
const project = openProject(ctx.cwd);
|
|
@@ -34,8 +53,9 @@ export async function planCheck(ctx: Ctx): Promise<Result> {
|
|
|
34
53
|
const footage = project.footage();
|
|
35
54
|
const assets = project.assets().filter((a) => a.id !== footage?.id);
|
|
36
55
|
const voiceIds = loadCredentials(ctx.env) ? (await client(ctx)("voices", {})).voices.map((v) => v.id) : undefined;
|
|
37
|
-
const
|
|
38
|
-
const
|
|
56
|
+
const ref = referenceChecks(project, parsed.data);
|
|
57
|
+
const mustFix = [...validatePlan(parsed.data, { aspect: project.config().aspect, footage, assets, voiceIds }), ...ref.mustFix];
|
|
58
|
+
const shouldImprove = mustFix.length ? [] : [...reviewPlan(parsed.data, { footage }), ...ref.improve];
|
|
39
59
|
const length = estimateLength(parsed.data);
|
|
40
60
|
const estimatedSeconds = Math.round(length.seconds);
|
|
41
61
|
const lines = [
|
|
@@ -0,0 +1,307 @@
|
|
|
1
|
+
import { execFile } from "node:child_process";
|
|
2
|
+
import { createHash, randomUUID } from "node:crypto";
|
|
3
|
+
import { copyFileSync, existsSync, mkdirSync, readdirSync, readFileSync, rmSync, statSync } from "node:fs";
|
|
4
|
+
import { basename, extname, resolve } from "node:path";
|
|
5
|
+
import { z } from "zod";
|
|
6
|
+
import { ApiFailure, uploadTo } from "../api/client";
|
|
7
|
+
import { client, type Ctx, type Result } from "../context";
|
|
8
|
+
import { MAX_TRANSCRIBE_SECONDS } from "../contract";
|
|
9
|
+
import { loadCredentials } from "../credentials";
|
|
10
|
+
import { analyzeBeats, cutsOnBeat } from "../project/beats";
|
|
11
|
+
import { openProject, type Project } from "../project/project";
|
|
12
|
+
import {
|
|
13
|
+
audioDuration, decodeMono, detectCuts, extractFrame, extractMp3, loudnessLufs, nearestAspect, paletteOfFrames, paletteOfVideo, probeVideo, REF_ASPECTS, toMp4,
|
|
14
|
+
} from "../project/refmeasure";
|
|
15
|
+
|
|
16
|
+
// A reference is for learning how a video is built, so ten minutes is more than it needs and keeps every step quick.
|
|
17
|
+
export const MAX_REF_SECONDS = 600;
|
|
18
|
+
const MAX_SCENES = 40;
|
|
19
|
+
const REF_ID = /^r-[0-9a-f]{8}$/;
|
|
20
|
+
|
|
21
|
+
export const RefRecordSchema = z.object({
|
|
22
|
+
id: z.string(), source: z.string(), urlHash: z.string().optional(),
|
|
23
|
+
durationSec: z.number(), width: z.number(), height: z.number(), aspect: z.enum(REF_ASPECTS), fps: z.number(), hasAudio: z.boolean(), createdAt: z.string(),
|
|
24
|
+
});
|
|
25
|
+
export type RefRecord = z.infer<typeof RefRecordSchema>;
|
|
26
|
+
|
|
27
|
+
export type Breakdown = {
|
|
28
|
+
source: string; durationSec: number; width: number; height: number; aspect: RefRecord["aspect"]; fps: number;
|
|
29
|
+
scenes: { index: number; startSec: number; endSec: number; durationSec: number; frames: string[]; palette: string[] }[];
|
|
30
|
+
// True when the video has more cuts than the 40 scenes listed: the last scene then holds everything after the 39th cut.
|
|
31
|
+
scenesCapped?: boolean;
|
|
32
|
+
pacing: { cuts: number; averageShotSec: number; shortestShotSec: number; longestShotSec: number; cutsPerSecond: number; firstCutSec?: number; cutsOnBeat?: number };
|
|
33
|
+
palette: string[];
|
|
34
|
+
audio: { hasAudio: boolean; loudnessLufs?: number; tempoBpm?: number; beatConfidence: number };
|
|
35
|
+
beats?: number[];
|
|
36
|
+
transcript?: { language?: string; text: string; segments: { text: string; startSec: number; endSec: number }[] };
|
|
37
|
+
};
|
|
38
|
+
|
|
39
|
+
export const REF_NOTICE = "You are responsible for having the right to download this video; it is used only as a reference: none of its footage or music goes into the output.";
|
|
40
|
+
|
|
41
|
+
// The external program that fetches a link. Tests give their own so that nothing touches the network.
|
|
42
|
+
export type YtdlpRunner = { available(): Promise<boolean>; run(args: string[]): Promise<void> };
|
|
43
|
+
export type RefDeps = { ytdlp?: YtdlpRunner };
|
|
44
|
+
|
|
45
|
+
const realYtdlp: YtdlpRunner = {
|
|
46
|
+
available: () => new Promise((done) => { execFile("yt-dlp", ["--version"], (err) => done(!err)); }),
|
|
47
|
+
run: (args) => new Promise((done, fail) => {
|
|
48
|
+
// No shell: the link is one element of the argument list and can never become a command.
|
|
49
|
+
execFile("yt-dlp", args, { maxBuffer: 64 * 1024 * 1024 }, (err, _out, stderr) => {
|
|
50
|
+
if (!err) return done();
|
|
51
|
+
const last = String(stderr).trim().split("\n").filter(Boolean).at(-1) ?? err.message;
|
|
52
|
+
fail(new Error(last.replace(/^ERROR:\s*/, "")));
|
|
53
|
+
});
|
|
54
|
+
}),
|
|
55
|
+
};
|
|
56
|
+
|
|
57
|
+
const YTDLP_MISSING = "yt-dlp is not installed, and links need it. Install it (macOS: `brew install yt-dlp`, or `pipx install yt-dlp`) and run the command again; a video file on this machine works without it.";
|
|
58
|
+
|
|
59
|
+
// One video, at most 1080p, as mp4, and nothing longer than the limit: the flags that keep a link from fetching a playlist, a channel or a huge file.
|
|
60
|
+
export function ytdlpArgs(url: string, dest: string): string[] {
|
|
61
|
+
return [
|
|
62
|
+
"--ignore-config", "--no-playlist", "--max-downloads", "1", "--no-progress", "--no-warnings",
|
|
63
|
+
"--match-filter", `duration<=${MAX_REF_SECONDS}`,
|
|
64
|
+
"-f", "bv*[height<=1080][ext=mp4]+ba[ext=m4a]/b[height<=1080][ext=mp4]/bv*[height<=1080]+ba/b[height<=1080]",
|
|
65
|
+
"--merge-output-format", "mp4", "-o", dest, "--", url,
|
|
66
|
+
];
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
const dirOf = (id: string) => `refs/${id}`;
|
|
70
|
+
const fileOf = (id: string, name: string) => `refs/${id}/${name}`;
|
|
71
|
+
|
|
72
|
+
function loadRef(project: Project, id: string): RefRecord {
|
|
73
|
+
if (!REF_ID.test(id)) throw new Error(`"${id}" is not a reference id. Run \`reelkit ref list\` to see them.`);
|
|
74
|
+
if (!project.exists(fileOf(id, "ref.json"))) throw new Error(`There is no reference ${id} in this project. Run \`reelkit ref list\`, or \`reelkit ref download <url-or-file>\`.`);
|
|
75
|
+
const parsed = RefRecordSchema.safeParse(project.readJson(fileOf(id, "ref.json")));
|
|
76
|
+
if (!parsed.success) throw new Error(`refs/${id}/ref.json is not valid. Download the reference again with \`reelkit ref download\`.`);
|
|
77
|
+
return parsed.data;
|
|
78
|
+
}
|
|
79
|
+
|
|
80
|
+
function allRefs(project: Project): RefRecord[] {
|
|
81
|
+
if (!project.exists("refs")) return [];
|
|
82
|
+
const out: RefRecord[] = [];
|
|
83
|
+
for (const name of readdirSync(project.path("refs")).sort()) {
|
|
84
|
+
if (!REF_ID.test(name) || !project.exists(fileOf(name, "ref.json"))) continue;
|
|
85
|
+
const parsed = RefRecordSchema.safeParse(project.readJsonOr(fileOf(name, "ref.json"), null));
|
|
86
|
+
if (parsed.success) out.push(parsed.data);
|
|
87
|
+
}
|
|
88
|
+
return out;
|
|
89
|
+
}
|
|
90
|
+
|
|
91
|
+
const fmt = (n: number) => String(Math.round(n * 10) / 10);
|
|
92
|
+
const minutes = (sec: number) => (sec >= 90 ? `${Math.round(sec / 60)} minutes` : `${Math.round(sec)} seconds`);
|
|
93
|
+
const tooLong = (what: string, sec: number) => `${what} is ${minutes(sec)} long and a reference is limited to ${MAX_REF_SECONDS / 60} minutes. Pick a shorter video, or trim it and give the file.`;
|
|
94
|
+
|
|
95
|
+
// A link has a scheme of two or more letters; "C:\\clip.mp4" is a path.
|
|
96
|
+
const looksLikeLink = (s: string) => /^[a-z][a-z0-9+.-]+:/i.test(s);
|
|
97
|
+
|
|
98
|
+
export async function refDownload(ctx: Ctx, input: string, opts: { redo?: boolean } = {}, deps: RefDeps = {}): Promise<Result> {
|
|
99
|
+
const project = openProject(ctx.cwd);
|
|
100
|
+
input = input.trim();
|
|
101
|
+
const isLink = looksLikeLink(input);
|
|
102
|
+
let url: URL | undefined;
|
|
103
|
+
if (isLink) {
|
|
104
|
+
try { url = new URL(input); } catch { return { ok: false, summary: `"${input}" is not a link that can be downloaded. Give an http or https link, or a video file.` }; }
|
|
105
|
+
if (url.protocol !== "http:" && url.protocol !== "https:") return { ok: false, summary: `Only http and https links can be downloaded, not ${url.protocol}. Give an http or https link, or a video file.` };
|
|
106
|
+
}
|
|
107
|
+
|
|
108
|
+
const urlHash = url ? createHash("sha256").update(url.href).digest("hex") : undefined;
|
|
109
|
+
let id = `r-${randomUUID().replace(/-/g, "").slice(0, 8)}`;
|
|
110
|
+
if (urlHash) {
|
|
111
|
+
const have = allRefs(project).find((r) => r.urlHash === urlHash && project.exists(fileOf(r.id, "video.mp4")));
|
|
112
|
+
if (have && !opts.redo) return { ok: true, data: have, summary: `This link is already downloaded as ${have.id} (refs/${have.id}/video.mp4). Use --redo to download it again. Next: \`reelkit ref analyze ${have.id}\`.\n${REF_NOTICE}` };
|
|
113
|
+
if (have) { id = have.id; rmSync(project.path(dirOf(id)), { recursive: true, force: true }); }
|
|
114
|
+
}
|
|
115
|
+
|
|
116
|
+
const video = project.path(fileOf(id, "video.mp4"));
|
|
117
|
+
const cleanup = () => rmSync(project.path(dirOf(id)), { recursive: true, force: true });
|
|
118
|
+
let source: string;
|
|
119
|
+
try {
|
|
120
|
+
if (url) {
|
|
121
|
+
const runner = deps.ytdlp ?? realYtdlp;
|
|
122
|
+
if (!(await runner.available())) return { ok: false, summary: YTDLP_MISSING };
|
|
123
|
+
source = url.href;
|
|
124
|
+
mkdirSync(project.path(dirOf(id)), { recursive: true });
|
|
125
|
+
ctx.log(`Downloading ${url.href} with yt-dlp.`);
|
|
126
|
+
try { await runner.run(ytdlpArgs(url.href, video)); }
|
|
127
|
+
catch (e) { cleanup(); return { ok: false, summary: `yt-dlp could not download this link: ${e instanceof Error ? e.message : String(e)} Check the link, or download the video yourself and give its file.` }; }
|
|
128
|
+
if (!existsSync(video)) { cleanup(); return { ok: false, summary: "yt-dlp finished but saved no video. The link may be a page without a video, or longer than 10 minutes. Check the link, or give a video file." }; }
|
|
129
|
+
} else {
|
|
130
|
+
const from = resolve(ctx.cwd, input);
|
|
131
|
+
if (!existsSync(from) || !statSync(from).isFile()) return { ok: false, summary: `${basename(from)} does not exist. Check the path.` };
|
|
132
|
+
source = basename(from);
|
|
133
|
+
const pre = await probeVideo(from, basename(from));
|
|
134
|
+
if (pre.durationSec > MAX_REF_SECONDS) return { ok: false, summary: tooLong(basename(from), pre.durationSec) };
|
|
135
|
+
mkdirSync(project.path(dirOf(id)), { recursive: true });
|
|
136
|
+
// mp4 is copied as it is; any other container is converted, so that refs/<id>/video.mp4 is an mp4 whatever was given.
|
|
137
|
+
if ([".mp4", ".m4v"].includes(extname(from).toLowerCase())) copyFileSync(from, video);
|
|
138
|
+
else await toMp4(from, video);
|
|
139
|
+
}
|
|
140
|
+
const info = await probeVideo(video, source);
|
|
141
|
+
if (info.durationSec > MAX_REF_SECONDS) { cleanup(); return { ok: false, summary: tooLong("This video", info.durationSec) }; }
|
|
142
|
+
const record: RefRecord = {
|
|
143
|
+
id, source, ...(urlHash ? { urlHash } : {}), ...info, aspect: nearestAspect(info.width, info.height), createdAt: new Date().toISOString(),
|
|
144
|
+
};
|
|
145
|
+
project.writeJson(fileOf(id, "ref.json"), record);
|
|
146
|
+
return {
|
|
147
|
+
ok: true, data: record,
|
|
148
|
+
summary: `Saved reference ${id}: ${fmt(info.durationSec)}s, ${info.width}x${info.height} (${record.aspect}), ${fmt(info.fps)} fps, ${info.hasAudio ? "with audio" : "no audio"}, at refs/${id}/video.mp4. Next: \`reelkit ref analyze ${id}\`.\n${REF_NOTICE}`,
|
|
149
|
+
};
|
|
150
|
+
} catch (e) {
|
|
151
|
+
cleanup();
|
|
152
|
+
// Every failure here is already a plain line written for the person who gave the video.
|
|
153
|
+
return { ok: false, summary: e instanceof Error ? e.message : String(e) };
|
|
154
|
+
}
|
|
155
|
+
}
|
|
156
|
+
|
|
157
|
+
// The audio of a reference as a small mp3, made once.
|
|
158
|
+
async function ensureAudio(project: Project, ref: RefRecord): Promise<string> {
|
|
159
|
+
const rel = fileOf(ref.id, "audio.mp3");
|
|
160
|
+
if (!project.exists(rel)) await extractMp3(project.path(fileOf(ref.id, "video.mp4")), project.path(rel));
|
|
161
|
+
return rel;
|
|
162
|
+
}
|
|
163
|
+
|
|
164
|
+
export async function refAudio(ctx: Ctx, id: string): Promise<Result> {
|
|
165
|
+
const project = openProject(ctx.cwd);
|
|
166
|
+
const ref = loadRef(project, id);
|
|
167
|
+
if (!project.exists(fileOf(id, "video.mp4"))) return { ok: false, summary: `refs/${id}/video.mp4 is missing. Download the reference again with \`reelkit ref download --redo\`.` };
|
|
168
|
+
if (!ref.hasAudio) return { ok: false, summary: `Reference ${id} has no audio track, so there is no audio to extract.` };
|
|
169
|
+
const had = project.exists(fileOf(id, "audio.mp3"));
|
|
170
|
+
const rel = await ensureAudio(project, ref);
|
|
171
|
+
const bytes = statSync(project.path(rel)).size;
|
|
172
|
+
return { ok: true, data: { path: rel, bytes, skipped: had }, summary: had ? `The audio of ${id} is already at ${rel}.` : `Saved the audio of ${id} at ${rel} (mono, 16 kHz, ${Math.round(bytes / 1024)} KB).` };
|
|
173
|
+
}
|
|
174
|
+
|
|
175
|
+
export async function refList(ctx: Ctx): Promise<Result> {
|
|
176
|
+
const project = openProject(ctx.cwd);
|
|
177
|
+
const refs = allRefs(project).map((r) => ({ id: r.id, source: r.source, durationSec: r.durationSec, analyzed: project.exists(fileOf(r.id, "breakdown.json")) }));
|
|
178
|
+
return {
|
|
179
|
+
ok: true, data: { references: refs },
|
|
180
|
+
summary: refs.length
|
|
181
|
+
? refs.map((r) => `${r.id} ${fmt(r.durationSec)}s ${r.analyzed ? "analysed" : "not analysed"} ${r.source}`).join("\n")
|
|
182
|
+
: "No references in this project yet. Run `reelkit ref download <url-or-file>`.",
|
|
183
|
+
};
|
|
184
|
+
}
|
|
185
|
+
|
|
186
|
+
type TranscriptOutcome = { transcript?: NonNullable<Breakdown["transcript"]>; note: string };
|
|
187
|
+
|
|
188
|
+
// The audio (never the video) goes to the server and comes back as text. Whatever goes wrong here costs the transcript only: the measurements
|
|
189
|
+
// of the video do not depend on it.
|
|
190
|
+
async function transcribe(ctx: Ctx, project: Project, ref: RefRecord): Promise<TranscriptOutcome> {
|
|
191
|
+
const retry = `\`reelkit ref analyze ${ref.id} --redo\``;
|
|
192
|
+
if (!loadCredentials(ctx.env)) return { note: `No transcript: you are not logged in. Run \`reelkit auth login\`, then ${retry}.` };
|
|
193
|
+
try {
|
|
194
|
+
const rel = await ensureAudio(project, ref);
|
|
195
|
+
const path = project.path(rel);
|
|
196
|
+
const bytes = readFileSync(path);
|
|
197
|
+
const api = client(ctx);
|
|
198
|
+
ctx.log("Sending the audio (not the video) to the Reelkit server to be transcribed.");
|
|
199
|
+
const durationSec = Math.min(MAX_TRANSCRIBE_SECONDS, Math.max(0.1, await audioDuration(path)));
|
|
200
|
+
const up = await api("transcribeStart", { filename: "audio.mp3", contentType: "audio/mpeg", bytes: bytes.length, durationSec });
|
|
201
|
+
await uploadTo(up.uploadUrl, new Uint8Array(bytes), "audio/mpeg");
|
|
202
|
+
const t = await api("transcribeRun", { id: up.id });
|
|
203
|
+
return {
|
|
204
|
+
transcript: { ...(t.language ? { language: t.language } : {}), text: t.text, segments: t.segments },
|
|
205
|
+
note: `The audio (not the video) was sent to the Reelkit server to be transcribed and was deleted there. It used ${Math.ceil(t.durationSec)} seconds of your monthly transcription quota (\`reelkit whoami\`).`,
|
|
206
|
+
};
|
|
207
|
+
} catch (e) {
|
|
208
|
+
const why = e instanceof ApiFailure ? e.message : `the Reelkit server could not be reached (${e instanceof Error ? e.message : String(e)}).`;
|
|
209
|
+
return { note: `No transcript: ${why.endsWith(".") ? why : `${why}.`} Retry with ${retry}.` };
|
|
210
|
+
}
|
|
211
|
+
}
|
|
212
|
+
|
|
213
|
+
// Runs `fn` over the items, `size` at a time, and keeps the order of the results.
|
|
214
|
+
async function inBatches<T, R>(items: T[], size: number, fn: (item: T, index: number) => Promise<R>): Promise<R[]> {
|
|
215
|
+
const out: R[] = [];
|
|
216
|
+
for (let i = 0; i < items.length; i += size) out.push(...(await Promise.all(items.slice(i, i + size).map((x, j) => fn(x, i + j)))));
|
|
217
|
+
return out;
|
|
218
|
+
}
|
|
219
|
+
|
|
220
|
+
const round2 = (n: number) => Math.round(n * 100) / 100;
|
|
221
|
+
|
|
222
|
+
async function measure(ctx: Ctx, project: Project, ref: RefRecord): Promise<Breakdown> {
|
|
223
|
+
const video = project.path(fileOf(ref.id, "video.mp4"));
|
|
224
|
+
ctx.log("Finding the cuts.");
|
|
225
|
+
const cuts = await detectCuts(video, ref.durationSec);
|
|
226
|
+
// The scenes listed are capped; the pacing below is measured on every cut.
|
|
227
|
+
const sceneCuts = cuts.slice(0, MAX_SCENES - 1);
|
|
228
|
+
const bounds = [0, ...sceneCuts, ref.durationSec];
|
|
229
|
+
rmSync(project.path(fileOf(ref.id, "frames")), { recursive: true, force: true });
|
|
230
|
+
ctx.log(`Saving sample frames of ${bounds.length - 1} scene(s).`);
|
|
231
|
+
// Each scene is a few short ffmpeg runs; running several scenes at once keeps a video with many cuts quick.
|
|
232
|
+
const scenes = await inBatches(bounds.slice(0, -1), 6, async (start, i) => {
|
|
233
|
+
const end = bounds[i + 1]!, len = end - start;
|
|
234
|
+
const n = String(i + 1).padStart(2, "0");
|
|
235
|
+
const frames = [`${dirOf(ref.id)}/frames/scene-${n}-1.jpg`, `${dirOf(ref.id)}/frames/scene-${n}-2.jpg`];
|
|
236
|
+
// A little inside the end, so that the last frame of the video can be read.
|
|
237
|
+
await extractFrame(video, start + len * 0.25, project.path(frames[0]!));
|
|
238
|
+
await extractFrame(video, Math.min(start + len * 0.75, ref.durationSec - 0.05), project.path(frames[1]!));
|
|
239
|
+
return { index: i + 1, startSec: round2(start), endSec: round2(end), durationSec: round2(len), frames, palette: await paletteOfFrames(frames.map((f) => project.path(f))) };
|
|
240
|
+
});
|
|
241
|
+
ctx.log("Measuring colours and sound.");
|
|
242
|
+
const palette = await paletteOfVideo(video, ref.durationSec);
|
|
243
|
+
|
|
244
|
+
const shots = [0, ...cuts, ref.durationSec].map((t, i, all) => (i ? t - all[i - 1]! : 0)).slice(1);
|
|
245
|
+
const pacing: Breakdown["pacing"] = {
|
|
246
|
+
cuts: cuts.length, averageShotSec: round2(ref.durationSec / shots.length), shortestShotSec: round2(Math.min(...shots)), longestShotSec: round2(Math.max(...shots)),
|
|
247
|
+
cutsPerSecond: round2(cuts.length / ref.durationSec), ...(cuts.length ? { firstCutSec: round2(cuts[0]!) } : {}),
|
|
248
|
+
};
|
|
249
|
+
|
|
250
|
+
const audio: Breakdown["audio"] = { hasAudio: ref.hasAudio, beatConfidence: 0 };
|
|
251
|
+
let beats: number[] | undefined;
|
|
252
|
+
if (ref.hasAudio) {
|
|
253
|
+
// A track that cannot be measured is a reference without a tempo, not a failed analysis.
|
|
254
|
+
const loud = await loudnessLufs(video).catch(() => undefined);
|
|
255
|
+
if (loud !== undefined) audio.loudnessLufs = loud;
|
|
256
|
+
try {
|
|
257
|
+
const rate = 11025;
|
|
258
|
+
const r = analyzeBeats(await decodeMono(video, rate), rate);
|
|
259
|
+
audio.beatConfidence = r.confidence;
|
|
260
|
+
if (r.tempoBpm !== undefined && r.beats) {
|
|
261
|
+
audio.tempoBpm = r.tempoBpm;
|
|
262
|
+
beats = r.beats.filter((t) => t <= ref.durationSec);
|
|
263
|
+
if (cuts.length) pacing.cutsOnBeat = round2(cutsOnBeat(cuts, beats));
|
|
264
|
+
}
|
|
265
|
+
} catch { /* no tempo */ }
|
|
266
|
+
}
|
|
267
|
+
return { source: ref.source, durationSec: ref.durationSec, width: ref.width, height: ref.height, aspect: ref.aspect, fps: ref.fps, scenes, ...(cuts.length > MAX_SCENES - 1 ? { scenesCapped: true } : {}), pacing, palette, audio, ...(beats ? { beats } : {}) };
|
|
268
|
+
}
|
|
269
|
+
|
|
270
|
+
// A short text a person or an agent can read at once.
|
|
271
|
+
function digest(id: string, b: Breakdown, notes: string[]): string {
|
|
272
|
+
const lines = [
|
|
273
|
+
`Reference ${id}: ${fmt(b.durationSec)}s, ${b.width}x${b.height} (${b.aspect}), ${fmt(b.fps)} fps.`,
|
|
274
|
+
b.pacing.cuts
|
|
275
|
+
? `Shape: ${b.scenes.length}${b.scenesCapped ? "+" : ""} scenes, ${b.pacing.cuts} cuts; average shot ${fmt(b.pacing.averageShotSec)}s (shortest ${fmt(b.pacing.shortestShotSec)}s, longest ${fmt(b.pacing.longestShotSec)}s); first cut at ${fmt(b.pacing.firstCutSec!)}s.`
|
|
276
|
+
: "Shape: one continuous shot, no cuts found.",
|
|
277
|
+
`Palette: ${b.palette.join(", ")}.`,
|
|
278
|
+
!b.audio.hasAudio ? "Audio: none."
|
|
279
|
+
: `Audio:${b.audio.loudnessLufs !== undefined ? ` ${b.audio.loudnessLufs} LUFS;` : ""} ${b.audio.tempoBpm !== undefined ? `tempo about ${Math.round(b.audio.tempoBpm)} BPM (confidence ${b.audio.beatConfidence})` : "no clear beat"}${b.pacing.cutsOnBeat !== undefined ? `; ${Math.round(b.pacing.cutsOnBeat * b.pacing.cuts)} of ${b.pacing.cuts} cuts land on a beat` : ""}.`,
|
|
280
|
+
];
|
|
281
|
+
if (b.transcript) lines.push(`Transcript${b.transcript.language ? ` (${b.transcript.language})` : ""}: "${b.transcript.text.slice(0, 200)}${b.transcript.text.length > 200 ? "..." : ""}"`);
|
|
282
|
+
lines.push(...notes);
|
|
283
|
+
lines.push(`Look at the frames now: refs/${id}/frames/ has two per scene (scene-01-1.jpg, scene-01-2.jpg, ...). The full numbers are in refs/${id}/breakdown.json.`);
|
|
284
|
+
return lines.join("\n");
|
|
285
|
+
}
|
|
286
|
+
|
|
287
|
+
export async function refAnalyze(ctx: Ctx, id: string, opts: { transcript?: boolean; redo?: boolean } = {}): Promise<Result> {
|
|
288
|
+
const project = openProject(ctx.cwd);
|
|
289
|
+
const ref = loadRef(project, id);
|
|
290
|
+
if (!project.exists(fileOf(id, "video.mp4"))) return { ok: false, summary: `refs/${id}/video.mp4 is missing. Download the reference again with \`reelkit ref download --redo\`.` };
|
|
291
|
+
const file = fileOf(id, "breakdown.json");
|
|
292
|
+
if (project.exists(file) && !opts.redo) {
|
|
293
|
+
const have = project.readJson<Breakdown>(file);
|
|
294
|
+
return { ok: true, data: have, summary: digest(id, have, [`Already analysed: this is the saved breakdown. Use --redo to measure again.`]) };
|
|
295
|
+
}
|
|
296
|
+
const breakdown = await measure(ctx, project, ref);
|
|
297
|
+
const notes: string[] = [];
|
|
298
|
+
if (opts.transcript === false) notes.push(`No transcript (--no-transcript). Nothing was sent anywhere. To add it later: \`reelkit ref analyze ${id} --redo\`.`);
|
|
299
|
+
else if (!ref.hasAudio) notes.push("No transcript: the reference has no audio. Nothing was sent anywhere.");
|
|
300
|
+
else {
|
|
301
|
+
const t = await transcribe(ctx, project, ref);
|
|
302
|
+
if (t.transcript) breakdown.transcript = t.transcript;
|
|
303
|
+
notes.push(t.note);
|
|
304
|
+
}
|
|
305
|
+
project.writeJson(file, breakdown);
|
|
306
|
+
return { ok: true, data: breakdown, summary: digest(id, breakdown, notes) };
|
|
307
|
+
}
|
package/src/contract/index.ts
CHANGED
|
@@ -67,6 +67,15 @@
|
|
|
67
67
|
// 20 seconds is refused by the request schema.
|
|
68
68
|
// - When the cutout quota is used up, `cutoutRun` is 429 `quota_exceeded` with the reset date. When the server has no GPU service configured,
|
|
69
69
|
// `cutoutStart` and `cutoutRun` are 400 `invalid_request` with the message "Background removal is not available on this server yet."
|
|
70
|
+
// - A transcript of a reference video's audio is a job with two calls. `transcribeStart` checks the request and answers with an id and an
|
|
71
|
+
// `uploadUrl` (charging nothing; the client sends the audio there exactly as for a library upload); `transcribeRun` checks that the audio
|
|
72
|
+
// arrived (else 400 `invalid_request`), reserves `ceil(durationSec)` seconds from the caller's monthly transcription quota, transcribes,
|
|
73
|
+
// deletes the audio and answers with the transcript. Only the audio is sent, never the video. The audio is deleted from the server when the
|
|
74
|
+
// call ends, whatever the result. The transcript is kept with the job, so calling `transcribeRun` again for the same id returns the same
|
|
75
|
+
// transcript and does not charge again. A failed run is not charged: the seconds are given back. It is charged in whole seconds of audio
|
|
76
|
+
// (`ceil(durationSec)`), and audio longer than 600 seconds is refused by the request schema. Segment times are in seconds from the start
|
|
77
|
+
// of the audio. A transcribe id belongs to its caller: another user's id, or an unknown one, is `not_found`.
|
|
78
|
+
// - When the transcription quota is used up, `transcribeRun` is 429 `quota_exceeded` with the reset date.
|
|
70
79
|
// - `durationSec` is the decoded audio length; `words` are in seconds from the start of that audio.
|
|
71
80
|
// - Search returns published items only, and leaves out items that are not matches at all; the default `limit` is 8.
|
|
72
81
|
import { z } from "zod";
|
|
@@ -101,6 +110,12 @@ export const CutoutTypeSchema = z.enum(["video/mp4", "video/quicktime", "video/w
|
|
|
101
110
|
export type CutoutType = z.infer<typeof CutoutTypeSchema>;
|
|
102
111
|
// The longest video a cutout takes, in seconds.
|
|
103
112
|
export const MAX_CUTOUT_SECONDS = 20;
|
|
113
|
+
// The audio types a transcription takes, the most bytes it takes (25 MB) and the longest audio in seconds.
|
|
114
|
+
export const TranscribeTypeSchema = z.enum(["audio/mpeg", "audio/mp4", "audio/wav"]);
|
|
115
|
+
export type TranscribeType = z.infer<typeof TranscribeTypeSchema>;
|
|
116
|
+
export const MAX_TRANSCRIBE_BYTES = 26_214_400;
|
|
117
|
+
export const MAX_TRANSCRIBE_SECONDS = 600;
|
|
118
|
+
export const TranscriptSegmentSchema = z.object({ text: z.string(), startSec: z.number(), endSec: z.number() });
|
|
104
119
|
|
|
105
120
|
export const ErrorCodeSchema = z.enum(["unauthenticated", "quota_exceeded", "not_found", "invalid_request", "server_error"]);
|
|
106
121
|
export type ErrorCode = z.infer<typeof ErrorCodeSchema>;
|
|
@@ -136,7 +151,7 @@ export type Voice = z.infer<typeof VoiceSchema>;
|
|
|
136
151
|
const Meter = z.object({ used: z.number(), limit: z.number() });
|
|
137
152
|
export const MeSchema = z.object({
|
|
138
153
|
userId: z.string(), handle: z.string(),
|
|
139
|
-
quota: z.object({ voiceoverChars: Meter, images: Meter, clips: Meter, cutoutSeconds: Meter, resetsAt: z.string() }),
|
|
154
|
+
quota: z.object({ voiceoverChars: Meter, images: Meter, clips: Meter, cutoutSeconds: Meter, transcribeSeconds: Meter, resetsAt: z.string() }),
|
|
140
155
|
// How many of the user's own items are published in the shared library. An item still in review, or sent back, is not counted.
|
|
141
156
|
contributions: z.number(),
|
|
142
157
|
});
|
|
@@ -207,6 +222,15 @@ export const routes = {
|
|
|
207
222
|
url: z.string().optional(), ext: z.enum(["webm"]).optional(), contentType: z.literal("video/webm").optional(),
|
|
208
223
|
}).refine((r) => (r.status === "done") === (r.url !== undefined && r.ext !== undefined && r.contentType !== undefined) && (r.status === "failed") === (r.message !== undefined),
|
|
209
224
|
"url, ext and contentType belong to done and message to failed")),
|
|
225
|
+
transcribeStart: route("POST", "/transcripts", true,
|
|
226
|
+
z.object({
|
|
227
|
+
filename: text(200), contentType: TranscribeTypeSchema,
|
|
228
|
+
bytes: z.number().int().min(1).max(MAX_TRANSCRIBE_BYTES), durationSec: z.number().positive().max(MAX_TRANSCRIBE_SECONDS),
|
|
229
|
+
}),
|
|
230
|
+
z.object({ id: z.string(), uploadUrl: z.string() })),
|
|
231
|
+
transcribeRun: route("POST", "/transcripts/run", true,
|
|
232
|
+
z.object({ id: ItemId }),
|
|
233
|
+
z.object({ id: z.string(), text: z.string(), language: z.string().optional(), segments: z.array(TranscriptSegmentSchema), durationSec: z.number().positive() })),
|
|
210
234
|
publicLibrary: route("GET", "/public/library", false,
|
|
211
235
|
z.object({ q: text(200).optional(), kind: LibraryKindSchema.optional(), page: z.coerce.number().int().min(1).max(10000).optional() }),
|
|
212
236
|
// With `q` the items are the best matches by meaning, best first, each with `match` (0..1), in one page (`hasMore` false). Under heavy
|
package/src/pipeline/schema.ts
CHANGED
|
@@ -28,6 +28,12 @@ export const ScenePlanSchema = z.object({
|
|
|
28
28
|
// Optional only so plans stored before voices existed still load; a new plan must choose one.
|
|
29
29
|
voiceId: z.string().optional().describe("id of the narration voice, one of the ids from `reelkit assets voices`, chosen to suit the idea, audience and language"),
|
|
30
30
|
pace: z.enum(["slow", "normal", "fast"]).optional().describe("speaking pace: slow for calm or emotional, normal by default, fast for high-energy"),
|
|
31
|
+
// A video the new one is made "like": what is taken from it is how it feels (structure, pacing, motion), never its footage, music or words.
|
|
32
|
+
// Optional, so plans stored before references existed still load.
|
|
33
|
+
reference: z.object({
|
|
34
|
+
id: z.string().describe("the id of a reference in this project, from `reelkit ref list`"),
|
|
35
|
+
take: z.array(z.string()).min(1).max(6).describe("short notes on what is taken from it, e.g. \"fast cuts every ~1.2s\", \"big type on colour fields\""),
|
|
36
|
+
}).optional(),
|
|
31
37
|
scenes: z.array(SceneSchema).min(3).max(8),
|
|
32
38
|
});
|
|
33
39
|
|
|
@@ -0,0 +1,130 @@
|
|
|
1
|
+
// Finds the tempo and the beat times of a piece of audio, on this machine, with no libraries. The audio is given as mono samples.
|
|
2
|
+
//
|
|
3
|
+
// The idea: loud moments (a drum hit, a click, a word) make the energy of the sound jump. The list of those jumps over time is the onset curve.
|
|
4
|
+
// If the music has a pulse, the curve repeats itself after one beat, so the tempo is the delay at which the curve best matches a copy of itself.
|
|
5
|
+
|
|
6
|
+
export type BeatAnalysis = {
|
|
7
|
+
// How sure the tempo is, from 0 (nothing repeats) to 1 (a perfect pulse). Real music is usually 0.3 to 0.7.
|
|
8
|
+
confidence: number;
|
|
9
|
+
// Present only when the pulse is clear enough to trust; a made-up tempo is worse than none.
|
|
10
|
+
tempoBpm?: number;
|
|
11
|
+
// Beat times in seconds from the start of the samples, on the grid of `tempoBpm`. Present with `tempoBpm`.
|
|
12
|
+
beats?: number[];
|
|
13
|
+
};
|
|
14
|
+
|
|
15
|
+
// Below this the autocorrelation peak is what noise produces by chance, so no tempo is reported.
|
|
16
|
+
export const MIN_BEAT_CONFIDENCE = 0.2;
|
|
17
|
+
|
|
18
|
+
const WINDOW = 512;
|
|
19
|
+
const HOP = 256;
|
|
20
|
+
const MIN_BPM = 60;
|
|
21
|
+
const MAX_BPM = 200;
|
|
22
|
+
// The range people actually move to; when a tempo outside it scores nearly as well, it is the half or double of one inside it.
|
|
23
|
+
const PREFERRED_LOW = 80;
|
|
24
|
+
const PREFERRED_HIGH = 160;
|
|
25
|
+
const BPM_STEP = 0.5;
|
|
26
|
+
// How many multiples of the beat length are compared: a real pulse matches at two, three and four beats too, noise does not.
|
|
27
|
+
const HARMONICS = 4;
|
|
28
|
+
|
|
29
|
+
// Energy per window, as a curve that jumps up where the sound gets louder. Quiet onsets count as much as loud ones (log scale), and the
|
|
30
|
+
// curve is smoothed a little so a hit that falls between two windows still shows up in both.
|
|
31
|
+
function onsetCurve(samples: Float32Array): Float64Array {
|
|
32
|
+
const frames = Math.max(0, Math.floor((samples.length - WINDOW) / HOP) + 1);
|
|
33
|
+
const energy = new Float64Array(frames);
|
|
34
|
+
for (let f = 0; f < frames; f++) {
|
|
35
|
+
let sum = 0;
|
|
36
|
+
const start = f * HOP;
|
|
37
|
+
for (let i = 0; i < WINDOW; i++) { const s = samples[start + i]!; sum += s * s; }
|
|
38
|
+
energy[f] = Math.log1p(100 * Math.sqrt(sum / WINDOW));
|
|
39
|
+
}
|
|
40
|
+
const rise = new Float64Array(frames);
|
|
41
|
+
for (let f = 1; f < frames; f++) rise[f] = Math.max(0, energy[f]! - energy[f - 1]!);
|
|
42
|
+
const smooth = new Float64Array(frames);
|
|
43
|
+
for (let f = 0; f < frames; f++) smooth[f] = 0.25 * (rise[f - 1] ?? 0) + 0.5 * rise[f]! + 0.25 * (rise[f + 1] ?? 0);
|
|
44
|
+
return smooth;
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
// The value of the curve at a position between two windows.
|
|
48
|
+
function at(curve: Float64Array, pos: number): number {
|
|
49
|
+
if (pos < 0 || pos > curve.length - 1) return 0;
|
|
50
|
+
const i = Math.floor(pos), frac = pos - i;
|
|
51
|
+
return curve[i]! * (1 - frac) + (curve[i + 1] ?? curve[i]!) * frac;
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
// How much the curve looks like itself `lag` windows later, from 0 to about 1.
|
|
55
|
+
function selfSimilarity(curve: Float64Array, lag: number): number {
|
|
56
|
+
const n = Math.floor(curve.length - lag);
|
|
57
|
+
if (n < 8) return 0;
|
|
58
|
+
let cross = 0, own = 0, shifted = 0;
|
|
59
|
+
for (let i = 0; i < n; i++) {
|
|
60
|
+
const a = curve[i]!, b = at(curve, i + lag);
|
|
61
|
+
cross += a * b; own += a * a; shifted += b * b;
|
|
62
|
+
}
|
|
63
|
+
return own > 0 && shifted > 0 ? cross / Math.sqrt(own * shifted) : 0;
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
export function analyzeBeats(samples: Float32Array, sampleRate: number): BeatAnalysis {
|
|
67
|
+
const curve = onsetCurve(samples);
|
|
68
|
+
if (curve.length < 16) return { confidence: 0 };
|
|
69
|
+
const mean = curve.reduce((s, v) => s + v, 0) / curve.length;
|
|
70
|
+
// The mean is taken out so that a steady loudness (which has no beat) does not look like a match.
|
|
71
|
+
for (let i = 0; i < curve.length; i++) curve[i] = curve[i]! - mean;
|
|
72
|
+
let energy = 0;
|
|
73
|
+
for (const v of curve) energy += v * v;
|
|
74
|
+
if (energy < 1e-9) return { confidence: 0 };
|
|
75
|
+
|
|
76
|
+
const framesPerSec = sampleRate / HOP;
|
|
77
|
+
const candidates: { bpm: number; score: number }[] = [];
|
|
78
|
+
for (let bpm = MIN_BPM; bpm <= MAX_BPM + 1e-9; bpm += BPM_STEP) {
|
|
79
|
+
const period = (60 / bpm) * framesPerSec;
|
|
80
|
+
let total = 0, used = 0;
|
|
81
|
+
for (let m = 1; m <= HARMONICS; m++) {
|
|
82
|
+
if (m * period > curve.length / 2) break;
|
|
83
|
+
total += selfSimilarity(curve, m * period);
|
|
84
|
+
used++;
|
|
85
|
+
}
|
|
86
|
+
candidates.push({ bpm, score: used ? total / used : 0 });
|
|
87
|
+
}
|
|
88
|
+
const isPeak = (i: number) => candidates[i]!.score >= (candidates[i - 1]?.score ?? -Infinity) && candidates[i]!.score >= (candidates[i + 1]?.score ?? -Infinity);
|
|
89
|
+
let best = 0;
|
|
90
|
+
for (let i = 1; i < candidates.length; i++) if (candidates[i]!.score > candidates[best]!.score) best = i;
|
|
91
|
+
const inPreferred = (i: number) => candidates[i]!.bpm >= PREFERRED_LOW && candidates[i]!.bpm <= PREFERRED_HIGH;
|
|
92
|
+
if (!inPreferred(best)) {
|
|
93
|
+
let alt = -1;
|
|
94
|
+
for (let i = 0; i < candidates.length; i++) if (inPreferred(i) && isPeak(i) && (alt < 0 || candidates[i]!.score > candidates[alt]!.score)) alt = i;
|
|
95
|
+
if (alt >= 0 && candidates[alt]!.score >= 0.8 * candidates[best]!.score) best = alt;
|
|
96
|
+
}
|
|
97
|
+
|
|
98
|
+
const confidence = Math.max(0, Math.min(1, candidates[best]!.score));
|
|
99
|
+
if (confidence < MIN_BEAT_CONFIDENCE) return { confidence: round(confidence, 2) };
|
|
100
|
+
|
|
101
|
+
// The peak lies between two of the tested tempos: a parabola through the three scores finds where it really is.
|
|
102
|
+
let bpm = candidates[best]!.bpm;
|
|
103
|
+
const l = candidates[best - 1]?.score, c = candidates[best]!.score, r = candidates[best + 1]?.score;
|
|
104
|
+
if (l !== undefined && r !== undefined && l - 2 * c + r < 0) bpm += (BPM_STEP * 0.5 * (l - r)) / (l - 2 * c + r);
|
|
105
|
+
|
|
106
|
+
// Where the beats fall: the shift of the grid that lands on the most onset.
|
|
107
|
+
const periodSec = 60 / bpm;
|
|
108
|
+
// Frame f describes the window that starts at f * HOP; a rise there means the sound arrived in the later part of that window.
|
|
109
|
+
const posOf = (t: number) => (t * sampleRate - 0.75 * WINDOW) / HOP;
|
|
110
|
+
const duration = samples.length / sampleRate;
|
|
111
|
+
let bestPhase = 0, bestHit = -Infinity;
|
|
112
|
+
for (let phase = 0; phase < periodSec; phase += HOP / 4 / sampleRate) {
|
|
113
|
+
let hit = 0;
|
|
114
|
+
for (let t = phase; t < duration; t += periodSec) hit += at(curve, posOf(t));
|
|
115
|
+
if (hit > bestHit) { bestHit = hit; bestPhase = phase; }
|
|
116
|
+
}
|
|
117
|
+
const beats: number[] = [];
|
|
118
|
+
// A beat that lands just before the first sample (a track that starts on a hit) is the first beat, at 0.
|
|
119
|
+
for (let t = bestPhase - periodSec; t < duration; t += periodSec) if (t >= -0.03) beats.push(round(Math.max(0, t), 3));
|
|
120
|
+
return { confidence: round(confidence, 2), tempoBpm: round(bpm, 1), beats };
|
|
121
|
+
}
|
|
122
|
+
|
|
123
|
+
// The share of cuts (0..1) that fall within `toleranceSec` of a beat.
|
|
124
|
+
export function cutsOnBeat(cuts: number[], beats: number[], toleranceSec = 0.08): number {
|
|
125
|
+
if (!cuts.length || !beats.length) return 0;
|
|
126
|
+
const near = cuts.filter((c) => beats.some((b) => Math.abs(b - c) <= toleranceSec)).length;
|
|
127
|
+
return near / cuts.length;
|
|
128
|
+
}
|
|
129
|
+
|
|
130
|
+
const round = (n: number, digits: number) => Math.round(n * 10 ** digits) / 10 ** digits;
|
package/src/project/chromakey.ts
CHANGED
|
@@ -12,12 +12,19 @@ export type KeyOptions = {
|
|
|
12
12
|
requireGreen?: boolean;
|
|
13
13
|
};
|
|
14
14
|
|
|
15
|
+
// Thrown when the background is green but not a saturated chroma green, or (for a clip that was asked to be on green) not green at
|
|
16
|
+
// all. Video models rarely produce a true chroma green: they give a soft sage or olive that sits too close to skin and grey to be
|
|
17
|
+
// keyed without taking the subject with it. The caller should cut the subject out on the server instead.
|
|
18
|
+
export class GreenTooDullError extends Error {
|
|
19
|
+
constructor() { super("The green background is too dull to key cleanly: it is not a saturated chroma green. Use --cutout instead, which removes any background."); this.name = "GreenTooDullError"; }
|
|
20
|
+
}
|
|
21
|
+
|
|
15
22
|
const NOT_GREEN = "This video does not have a green background in its corners, so there is nothing to key out. Film the subject in front of an evenly lit green background that fills the frame.";
|
|
16
23
|
|
|
17
24
|
// The colour to key on, read from the first frame's four corners. A real green screen, filmed or generated, is never exactly
|
|
18
25
|
// #00FF00: it is whatever green the light and the encoder made of it, and keying on the wrong green leaves a haze. Undefined when
|
|
19
26
|
// the corners are not green.
|
|
20
|
-
async function cornerGreen(input: string): Promise<string | undefined> {
|
|
27
|
+
async function cornerGreen(input: string): Promise<{ colour: string; saturated: boolean } | undefined> {
|
|
21
28
|
const corner = (x: string, y: string) => `crop=24:24:${x}:${y},scale=1:1:flags=area`;
|
|
22
29
|
const picks = ["0:0", "iw-24:0", "0:ih-24", "iw-24:ih-24"].map((c) => { const [x, y] = c.split(":") as [string, string]; return corner(x, y); });
|
|
23
30
|
const colours: [number, number, number][] = [];
|
|
@@ -29,7 +36,9 @@ async function cornerGreen(input: string): Promise<string | undefined> {
|
|
|
29
36
|
// Most corners must agree: a subject may reach one corner, but not three.
|
|
30
37
|
if (green.length < 3) return undefined;
|
|
31
38
|
const avg = (i: 0 | 1 | 2) => Math.round(green.reduce((sum, c) => sum + c[i], 0) / green.length);
|
|
32
|
-
|
|
39
|
+
const [r, g, b] = [avg(0), avg(1), avg(2)];
|
|
40
|
+
// How far the green stands above the other two channels. Studio and pure greens are far above 60; a sage green is around 30.
|
|
41
|
+
return { colour: `0x${[r, g, b].map((n) => n.toString(16).padStart(2, "0")).join("")}`, saturated: g - Math.max(r, b) >= 60 };
|
|
33
42
|
}
|
|
34
43
|
|
|
35
44
|
// Turns the green of a green-screen clip into transparency: VP9 with an alpha channel, which Remotion plays with `transparent`.
|
|
@@ -39,12 +48,13 @@ export async function keyGreen(input: string, output: string, opts: KeyOptions =
|
|
|
39
48
|
try {
|
|
40
49
|
const found = await cornerGreen(input);
|
|
41
50
|
if (!found && opts.requireGreen) throw new Error(NOT_GREEN);
|
|
42
|
-
|
|
51
|
+
if (!found || !found.saturated) throw new GreenTooDullError();
|
|
52
|
+
const filter = `chromakey=color=${found.colour ?? PURE_GREEN}:similarity=0.16:blend=0.08,despill=type=green:mix=0.5:expand=0,format=yuva420p`;
|
|
43
53
|
const audio = opts.keepAudio ? ["-c:a", "libopus", "-b:a", "96k"] : ["-an"];
|
|
44
54
|
await run("ffmpeg", ["-v", "error", "-y", "-i", input, "-vf", filter, "-c:v", "libvpx-vp9", "-pix_fmt", "yuva420p", "-b:v", "0", "-crf", "30", ...audio, output]);
|
|
45
55
|
} catch (e) {
|
|
46
56
|
rmSync(output, { force: true });
|
|
47
|
-
if (e instanceof Error && e.message === NOT_GREEN) throw e;
|
|
57
|
+
if (e instanceof GreenTooDullError || (e instanceof Error && e.message === NOT_GREEN)) throw e;
|
|
48
58
|
if ((e as NodeJS.ErrnoException)?.code === "ENOENT") throw new Error("ffmpeg was not found. Install ffmpeg (macOS: `brew install ffmpeg`) and run the command again.");
|
|
49
59
|
throw new Error("ffmpeg could not key the green out of the clip. The original clip is saved; run the command again to retry.");
|
|
50
60
|
}
|
|
@@ -0,0 +1,148 @@
|
|
|
1
|
+
// Measuring a reference video with ffmpeg and ffprobe, all on this machine. Every failure is a one-line Error for the person who gave the video.
|
|
2
|
+
import { execFile } from "node:child_process";
|
|
3
|
+
import { mkdirSync } from "node:fs";
|
|
4
|
+
import { dirname } from "node:path";
|
|
5
|
+
import { promisify } from "node:util";
|
|
6
|
+
|
|
7
|
+
const exec = promisify(execFile);
|
|
8
|
+
|
|
9
|
+
export const REF_ASPECTS = ["9:16", "16:9", "1:1", "4:5"] as const;
|
|
10
|
+
export type RefAspect = (typeof REF_ASPECTS)[number];
|
|
11
|
+
|
|
12
|
+
async function tool(cmd: "ffmpeg" | "ffprobe", args: string[]): Promise<{ stdout: Buffer; stderr: string }> {
|
|
13
|
+
try {
|
|
14
|
+
const r = await exec(cmd, args, { maxBuffer: 256 * 1024 * 1024, encoding: "buffer" });
|
|
15
|
+
return { stdout: r.stdout, stderr: r.stderr.toString("utf8") };
|
|
16
|
+
} catch (e) {
|
|
17
|
+
if ((e as NodeJS.ErrnoException)?.code === "ENOENT") throw new Error(`${cmd} was not found. Install ffmpeg (macOS: \`brew install ffmpeg\`) and try again.`);
|
|
18
|
+
throw e;
|
|
19
|
+
}
|
|
20
|
+
}
|
|
21
|
+
|
|
22
|
+
export type VideoInfo = { durationSec: number; width: number; height: number; fps: number; hasAudio: boolean };
|
|
23
|
+
|
|
24
|
+
const rate = (r: string | undefined): number => {
|
|
25
|
+
const [a, b] = (r ?? "").split("/").map(Number);
|
|
26
|
+
return a && b ? a / b : a || 0;
|
|
27
|
+
};
|
|
28
|
+
|
|
29
|
+
export async function probeVideo(path: string, name: string): Promise<VideoInfo> {
|
|
30
|
+
const unreadable = new Error(`${name} could not be read as a video. It may be damaged or not a video: try another file or link.`);
|
|
31
|
+
let info: { streams?: { codec_type?: string; width?: number; height?: number; avg_frame_rate?: string; r_frame_rate?: string; duration?: string }[]; format?: { duration?: string } };
|
|
32
|
+
try {
|
|
33
|
+
info = JSON.parse((await tool("ffprobe", ["-v", "error", "-print_format", "json", "-show_format", "-show_streams", path])).stdout.toString("utf8"));
|
|
34
|
+
} catch (e) {
|
|
35
|
+
if (e instanceof Error && e.message.startsWith("ffprobe was not found")) throw e;
|
|
36
|
+
throw unreadable;
|
|
37
|
+
}
|
|
38
|
+
const v = info.streams?.find((s) => s.codec_type === "video");
|
|
39
|
+
const duration = Number(info.format?.duration ?? v?.duration);
|
|
40
|
+
if (!v || !(v.width! > 0) || !(v.height! > 0) || !(duration > 0)) throw unreadable;
|
|
41
|
+
const fps = rate(v.avg_frame_rate) || rate(v.r_frame_rate);
|
|
42
|
+
return { durationSec: Math.round(duration * 100) / 100, width: v.width!, height: v.height!, fps: Math.round(fps * 100) / 100, hasAudio: Boolean(info.streams?.some((s) => s.codec_type === "audio")) };
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
// The nearest of the four shapes, by how far the ratios are apart on a log scale (so 2:1 is as far from 1:1 as 1:2).
|
|
46
|
+
export function nearestAspect(width: number, height: number): RefAspect {
|
|
47
|
+
const ratio = width / height;
|
|
48
|
+
const value: Record<RefAspect, number> = { "9:16": 9 / 16, "16:9": 16 / 9, "1:1": 1, "4:5": 4 / 5 };
|
|
49
|
+
return REF_ASPECTS.reduce((best, a) => (Math.abs(Math.log(ratio / value[a])) < Math.abs(Math.log(ratio / value[best])) ? a : best));
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
const MIN_SHOT_SEC = 0.25;
|
|
53
|
+
|
|
54
|
+
// The times where the picture changes completely. Cuts closer than a quarter of a second to each other, to the start or to the end are one
|
|
55
|
+
// flash or a fade, not a shot of their own.
|
|
56
|
+
export async function detectCuts(video: string, durationSec: number, threshold = 0.3): Promise<number[]> {
|
|
57
|
+
const { stderr } = await tool("ffmpeg", ["-hide_banner", "-nostats", "-i", video, "-an", "-vf", `select='gt(scene,${threshold})',showinfo`, "-f", "null", "-"]);
|
|
58
|
+
const times = [...stderr.matchAll(/pts_time:\s*([0-9.]+)/g)].map((m) => Number(m[1])).sort((a, b) => a - b);
|
|
59
|
+
const cuts: number[] = [];
|
|
60
|
+
for (const t of times) {
|
|
61
|
+
if (t < MIN_SHOT_SEC || durationSec - t < MIN_SHOT_SEC) continue;
|
|
62
|
+
if (cuts.length && t - cuts[cuts.length - 1]! < MIN_SHOT_SEC) continue;
|
|
63
|
+
cuts.push(Math.round(t * 1000) / 1000);
|
|
64
|
+
}
|
|
65
|
+
return cuts;
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
// One JPEG of the video at `atSec`, at most 540 pixels on the long side.
|
|
69
|
+
export async function extractFrame(video: string, atSec: number, dest: string): Promise<void> {
|
|
70
|
+
mkdirSync(dirname(dest), { recursive: true });
|
|
71
|
+
await tool("ffmpeg", ["-v", "error", "-y", "-ss", atSec.toFixed(3), "-i", video, "-frames:v", "1",
|
|
72
|
+
"-vf", "scale=w='if(gt(iw,ih),min(540,iw),-2)':h='if(gt(iw,ih),-2,min(540,ih))'", "-q:v", "4", dest]);
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
type Pixel = [number, number, number];
|
|
76
|
+
|
|
77
|
+
// Averages of the main colours: each pixel goes in one of 64 buckets (four levels per channel), and the fullest buckets, skipping one whose
|
|
78
|
+
// colour is close to a colour already taken, are named by the average of their pixels.
|
|
79
|
+
export function dominantColors(pixels: Pixel[], max: number): string[] {
|
|
80
|
+
const buckets = new Map<number, { n: number; sum: Pixel }>();
|
|
81
|
+
for (const [r, g, b] of pixels) {
|
|
82
|
+
const key = (r >> 6) * 16 + (g >> 6) * 4 + (b >> 6);
|
|
83
|
+
const e = buckets.get(key) ?? { n: 0, sum: [0, 0, 0] as Pixel };
|
|
84
|
+
e.n++; e.sum[0] += r; e.sum[1] += g; e.sum[2] += b;
|
|
85
|
+
buckets.set(key, e);
|
|
86
|
+
}
|
|
87
|
+
const chosen: Pixel[] = [];
|
|
88
|
+
for (const e of [...buckets.values()].sort((a, b) => b.n - a.n)) {
|
|
89
|
+
const c = e.sum.map((v) => Math.round(v / e.n)) as Pixel;
|
|
90
|
+
if (chosen.some((o) => Math.hypot(o[0] - c[0], o[1] - c[1], o[2] - c[2]) < 48)) continue;
|
|
91
|
+
chosen.push(c);
|
|
92
|
+
if (chosen.length === max) break;
|
|
93
|
+
}
|
|
94
|
+
return chosen.map(([r, g, b]) => `#${[r, g, b].map((v) => v.toString(16).padStart(2, "0")).join("")}`);
|
|
95
|
+
}
|
|
96
|
+
|
|
97
|
+
// Every picture is shrunk to 4 by 4 pixels, which averages away detail and leaves the colour of each part of the frame.
|
|
98
|
+
async function tinyPixels(inputArgs: string[], filter: string): Promise<Pixel[]> {
|
|
99
|
+
const { stdout } = await tool("ffmpeg", ["-v", "error", ...inputArgs, "-an", "-vf", `${filter}scale=4:4:flags=area`, "-f", "rawvideo", "-pix_fmt", "rgb24", "-"]);
|
|
100
|
+
const out: Pixel[] = [];
|
|
101
|
+
for (let i = 0; i + 2 < stdout.length; i += 3) out.push([stdout[i]!, stdout[i + 1]!, stdout[i + 2]!]);
|
|
102
|
+
return out;
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
export async function paletteOfVideo(video: string, durationSec: number, max = 5): Promise<string[]> {
|
|
106
|
+
// At most about 120 pictures however long the video is.
|
|
107
|
+
const fps = Math.max(0.15, Math.min(2, 120 / durationSec));
|
|
108
|
+
return dominantColors(await tinyPixels(["-i", video], `fps=${fps.toFixed(3)},`), max);
|
|
109
|
+
}
|
|
110
|
+
|
|
111
|
+
export async function paletteOfFrames(frames: string[], max = 3): Promise<string[]> {
|
|
112
|
+
const pixels: Pixel[] = [];
|
|
113
|
+
for (const f of frames) pixels.push(...(await tinyPixels(["-i", f], "")));
|
|
114
|
+
return dominantColors(pixels, max);
|
|
115
|
+
}
|
|
116
|
+
|
|
117
|
+
// The integrated loudness in LUFS, or undefined when there is none (silence is minus infinity).
|
|
118
|
+
export async function loudnessLufs(video: string): Promise<number | undefined> {
|
|
119
|
+
const { stderr } = await tool("ffmpeg", ["-hide_banner", "-nostats", "-i", video, "-vn", "-af", "ebur128=framelog=quiet", "-f", "null", "-"]);
|
|
120
|
+
const all = [...stderr.matchAll(/\bI:\s+(-?[0-9.]+)\s+LUFS/g)];
|
|
121
|
+
const value = Number(all.at(-1)?.[1]);
|
|
122
|
+
return Number.isFinite(value) ? Math.round(value * 10) / 10 : undefined;
|
|
123
|
+
}
|
|
124
|
+
|
|
125
|
+
// The audio as mono samples between -1 and 1, for the beat analysis.
|
|
126
|
+
export async function decodeMono(video: string, sampleRate: number): Promise<Float32Array> {
|
|
127
|
+
const { stdout } = await tool("ffmpeg", ["-v", "error", "-i", video, "-vn", "-ac", "1", "-ar", String(sampleRate), "-f", "s16le", "-"]);
|
|
128
|
+
const n = Math.floor(stdout.length / 2);
|
|
129
|
+
const out = new Float32Array(n);
|
|
130
|
+
for (let i = 0; i < n; i++) out[i] = stdout.readInt16LE(i * 2) / 32768;
|
|
131
|
+
return out;
|
|
132
|
+
}
|
|
133
|
+
|
|
134
|
+
// Speech needs little: mono, 16 kHz and 64 kbit/s keep a ten-minute reference under 5 MB.
|
|
135
|
+
export async function extractMp3(video: string, dest: string): Promise<void> {
|
|
136
|
+
mkdirSync(dirname(dest), { recursive: true });
|
|
137
|
+
await tool("ffmpeg", ["-v", "error", "-y", "-i", video, "-vn", "-ac", "1", "-ar", "16000", "-b:a", "64k", "-c:a", "libmp3lame", dest]);
|
|
138
|
+
}
|
|
139
|
+
|
|
140
|
+
export async function toMp4(source: string, dest: string): Promise<void> {
|
|
141
|
+
mkdirSync(dirname(dest), { recursive: true });
|
|
142
|
+
await tool("ffmpeg", ["-v", "error", "-y", "-i", source, "-c:v", "libx264", "-pix_fmt", "yuv420p", "-c:a", "aac", "-movflags", "+faststart", dest]);
|
|
143
|
+
}
|
|
144
|
+
|
|
145
|
+
export async function audioDuration(path: string): Promise<number> {
|
|
146
|
+
const { stdout } = await tool("ffprobe", ["-v", "error", "-show_entries", "format=duration", "-of", "csv=p=0", path]);
|
|
147
|
+
return Number(stdout.toString("utf8").trim());
|
|
148
|
+
}
|
|
@@ -23,8 +23,11 @@ export type Harness = {
|
|
|
23
23
|
// Optional, for the cutout tests. A file name that makes a cutout job fail (the fake fails any name containing "cutout-fail"); without
|
|
24
24
|
// it the failing-job case is skipped. The cutout tests wait between status asks for `clipPollMs` too.
|
|
25
25
|
cutoutFailFilename?: string;
|
|
26
|
+
// Optional, for the transcript tests. A file name that makes a transcription fail (the fake fails any name containing "transcribe-fail");
|
|
27
|
+
// without it the failing-job case is skipped.
|
|
28
|
+
transcribeFailFilename?: string;
|
|
26
29
|
};
|
|
27
|
-
export type StartOptions = { voiceoverChars?: number; images?: number; clips?: number; cutoutSeconds?: number };
|
|
30
|
+
export type StartOptions = { voiceoverChars?: number; images?: number; clips?: number; cutoutSeconds?: number; transcribeSeconds?: number };
|
|
28
31
|
|
|
29
32
|
const PNG = new Uint8Array(Buffer.from("iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mNkYPhfDwAChwGA60e6kgAAAABJRU5ErkJggg==", "base64"));
|
|
30
33
|
const DATE = /\d{4}-\d{2}-\d{2}/;
|
|
@@ -87,6 +90,8 @@ export function runConformance(name: string, start: (opts: StartOptions) => Prom
|
|
|
87
90
|
expect(me.quota.clips.limit).toBeGreaterThan(0);
|
|
88
91
|
expect(me.quota.cutoutSeconds.used).toBe(0);
|
|
89
92
|
expect(me.quota.cutoutSeconds.limit).toBeGreaterThan(0);
|
|
93
|
+
expect(me.quota.transcribeSeconds.used).toBe(0);
|
|
94
|
+
expect(me.quota.transcribeSeconds.limit).toBeGreaterThan(0);
|
|
90
95
|
expect(me.quota.voiceoverChars.limit).toBeGreaterThan(0);
|
|
91
96
|
expect(me.contributions).toBe(0);
|
|
92
97
|
const resets = new Date(me.quota.resetsAt);
|
|
@@ -522,6 +527,107 @@ export function runConformance(name: string, start: (opts: StartOptions) => Prom
|
|
|
522
527
|
});
|
|
523
528
|
});
|
|
524
529
|
|
|
530
|
+
describe("transcripts", () => {
|
|
531
|
+
const AUDIO = new Uint8Array([73, 68, 51, 4, ...new Array(2000).fill(9)]);
|
|
532
|
+
const meta = (over: Record<string, unknown> = {}) => ({ filename: "talk.mp3", contentType: "audio/mpeg" as const, bytes: AUDIO.length, durationSec: 12.4, ...over });
|
|
533
|
+
const sent = async (api: Api, over: Record<string, unknown> = {}) => {
|
|
534
|
+
const started = await api("transcribeStart", meta(over) as never);
|
|
535
|
+
await put(started.uploadUrl, AUDIO, (over.contentType as string | undefined) ?? "audio/mpeg");
|
|
536
|
+
return started;
|
|
537
|
+
};
|
|
538
|
+
|
|
539
|
+
it("start, upload, run: text and segments with increasing times inside the audio; charged in whole seconds to this user only", async () => {
|
|
540
|
+
const api = await as("user-trn0000001"), other = await as("user-trn0000002");
|
|
541
|
+
const started = await api("transcribeStart", meta());
|
|
542
|
+
expect(started.id).toBeTruthy();
|
|
543
|
+
expect(started.uploadUrl).toBeTruthy();
|
|
544
|
+
expect((await api("me", {})).quota.transcribeSeconds.used).toBe(0);
|
|
545
|
+
expect((await put(started.uploadUrl, AUDIO, "audio/mpeg")).ok).toBe(true);
|
|
546
|
+
const t = await api("transcribeRun", { id: started.id });
|
|
547
|
+
expect(t.id).toBe(started.id);
|
|
548
|
+
expect(t.text.length).toBeGreaterThan(0);
|
|
549
|
+
expect(t.durationSec).toBeGreaterThan(0);
|
|
550
|
+
expect(t.segments.length).toBeGreaterThan(0);
|
|
551
|
+
let last = 0;
|
|
552
|
+
for (const s of t.segments) {
|
|
553
|
+
expect(s.text.length).toBeGreaterThan(0);
|
|
554
|
+
expect(s.startSec).toBeGreaterThanOrEqual(last);
|
|
555
|
+
expect(s.endSec).toBeGreaterThanOrEqual(s.startSec);
|
|
556
|
+
expect(s.endSec).toBeLessThanOrEqual(t.durationSec + 0.5);
|
|
557
|
+
last = s.endSec;
|
|
558
|
+
}
|
|
559
|
+
expect((await api("me", {})).quota.transcribeSeconds.used).toBe(13);
|
|
560
|
+
expect((await other("me", {})).quota.transcribeSeconds.used).toBe(0);
|
|
561
|
+
});
|
|
562
|
+
|
|
563
|
+
it("a run before the audio arrived is invalid_request and charges nothing", async () => {
|
|
564
|
+
const api = await as("user-trn0000003");
|
|
565
|
+
const started = await api("transcribeStart", meta());
|
|
566
|
+
expect((await failure(api("transcribeRun", { id: started.id }))).code).toBe("invalid_request");
|
|
567
|
+
expect((await api("me", {})).quota.transcribeSeconds.used).toBe(0);
|
|
568
|
+
await put(started.uploadUrl, AUDIO, "audio/mpeg");
|
|
569
|
+
expect((await api("transcribeRun", { id: started.id })).text.length).toBeGreaterThan(0);
|
|
570
|
+
});
|
|
571
|
+
|
|
572
|
+
it("running twice returns the same transcript and charges once", async () => {
|
|
573
|
+
const api = await as("user-trn0000004");
|
|
574
|
+
const started = await sent(api);
|
|
575
|
+
const first = await api("transcribeRun", { id: started.id });
|
|
576
|
+
const second = await api("transcribeRun", { id: started.id });
|
|
577
|
+
expect(second).toEqual(first);
|
|
578
|
+
expect((await api("me", {})).quota.transcribeSeconds.used).toBe(13);
|
|
579
|
+
});
|
|
580
|
+
|
|
581
|
+
it("a transcript id belongs to its caller: another user, or an unknown id, is not_found", async () => {
|
|
582
|
+
const api = await as("user-trn0000005"), other = await as("user-trn0000006");
|
|
583
|
+
const started = await sent(api);
|
|
584
|
+
await api("transcribeRun", { id: started.id });
|
|
585
|
+
expect((await failure(other("transcribeRun", { id: started.id }))).code).toBe("not_found");
|
|
586
|
+
expect((await failure(api("transcribeRun", { id: "tr-nosuchjob" }))).code).toBe("not_found");
|
|
587
|
+
expect((await other("me", {})).quota.transcribeSeconds.used).toBe(0);
|
|
588
|
+
});
|
|
589
|
+
|
|
590
|
+
it("a failing job is a server error and is not charged", async () => {
|
|
591
|
+
if (!h.transcribeFailFilename) return;
|
|
592
|
+
const api = await as("user-trn0000007");
|
|
593
|
+
const started = await sent(api, { filename: h.transcribeFailFilename });
|
|
594
|
+
expect((await failure(api("transcribeRun", { id: started.id }))).code).toBe("server_error");
|
|
595
|
+
expect((await api("me", {})).quota.transcribeSeconds.used).toBe(0);
|
|
596
|
+
});
|
|
597
|
+
|
|
598
|
+
it("at the limit a run is quota_exceeded with the reset date, charged nothing; other users are unaffected", async () => {
|
|
599
|
+
const t = await start({ transcribeSeconds: 5 });
|
|
600
|
+
try {
|
|
601
|
+
const api = createClient({ baseUrl: t.baseUrl, token: await t.login("user-trnq000001") });
|
|
602
|
+
const other = createClient({ baseUrl: t.baseUrl, token: await t.login("user-trnq000002") });
|
|
603
|
+
const first = await sent(api, { durationSec: 3 });
|
|
604
|
+
await api("transcribeRun", { id: first.id });
|
|
605
|
+
const second = await sent(api, { durationSec: 3 });
|
|
606
|
+
const e = await failure(api("transcribeRun", { id: second.id }));
|
|
607
|
+
expect(e.code).toBe("quota_exceeded");
|
|
608
|
+
expect(e.message).toMatch(DATE);
|
|
609
|
+
const me = await api("me", {});
|
|
610
|
+
expect(me.quota.transcribeSeconds).toEqual({ used: 3, limit: 5 });
|
|
611
|
+
expect(e.message).toContain(me.quota.resetsAt.slice(0, 10));
|
|
612
|
+
const third = await sent(other, { durationSec: 3 });
|
|
613
|
+
expect((await other("transcribeRun", { id: third.id })).text.length).toBeGreaterThan(0);
|
|
614
|
+
} finally { await t.close(); }
|
|
615
|
+
});
|
|
616
|
+
|
|
617
|
+
it("refuses audio over 600 seconds, a type that is not audio, a size out of range and a missing login", async () => {
|
|
618
|
+
const api = await as("user-trn0000008");
|
|
619
|
+
for (const bad of [{ durationSec: 600.5 }, { durationSec: 0 }, { durationSec: -1 }, { contentType: "video/mp4" }, { bytes: 0 }, { bytes: 26_214_401 }, { filename: "a\u0000b.mp3" }]) {
|
|
620
|
+
expect((await failure(api("transcribeStart", meta(bad) as never))).code, JSON.stringify(bad)).toBe("invalid_request");
|
|
621
|
+
}
|
|
622
|
+
for (const ok of [{ durationSec: 600 }, { contentType: "audio/mp4" as const }, { contentType: "audio/wav" as const }, { bytes: 26_214_400 }]) {
|
|
623
|
+
expect((await api("transcribeStart", meta(ok))).id).toBeTruthy();
|
|
624
|
+
}
|
|
625
|
+
expect((await failure(api("transcribeRun", { id: "../x" }))).code).toBe("invalid_request");
|
|
626
|
+
expect((await failure(anon("transcribeStart", meta()))).code).toBe("unauthenticated");
|
|
627
|
+
expect((await failure(anon("transcribeRun", { id: "tr-x" }))).code).toBe("unauthenticated");
|
|
628
|
+
});
|
|
629
|
+
});
|
|
630
|
+
|
|
525
631
|
describe("quota", () => {
|
|
526
632
|
it("passes exactly at the limit; one over is quota_exceeded with the reset date; other users are unaffected", async () => {
|
|
527
633
|
const t = await start({ voiceoverChars: 10, images: 1 });
|
package/src/testing/fake-api.ts
CHANGED
|
@@ -81,10 +81,14 @@ function makeCutout(): Blob {
|
|
|
81
81
|
} finally { rmSync(dir, { recursive: true, force: true }); }
|
|
82
82
|
}
|
|
83
83
|
|
|
84
|
+
// A transcript job: the audio arrives at the upload URL; `run` reserves the seconds, deletes the audio and keeps the answer, so a second run is free.
|
|
85
|
+
type TranscriptJob = { owner: string; filename: string; durationSec: number; file?: Uint8Array; result?: { id: string; text: string; language: string; segments: { text: string; startSec: number; endSec: number }[]; durationSec: number } };
|
|
86
|
+
const TRANSCRIBE_FAIL = "transcribe-fail";
|
|
87
|
+
|
|
84
88
|
export type FakeApi = Awaited<ReturnType<typeof startFakeApi>>;
|
|
85
89
|
|
|
86
90
|
// An in-memory stand-in for the Reelkit API. It implements every route in the contract.
|
|
87
|
-
export async function startFakeApi(opts: { voiceoverCharLimit?: number; imageLimit?: number; clipLimit?: number; noClipProvider?: boolean; cutoutSecondsLimit?: number; noCutoutService?: boolean; autoApprove?: boolean } = {}) {
|
|
91
|
+
export async function startFakeApi(opts: { voiceoverCharLimit?: number; imageLimit?: number; clipLimit?: number; noClipProvider?: boolean; cutoutSecondsLimit?: number; noCutoutService?: boolean; transcribeSecondsLimit?: number; autoApprove?: boolean } = {}) {
|
|
88
92
|
// token -> the user it belongs to
|
|
89
93
|
const tokens = new Map<string, string>();
|
|
90
94
|
const devices = new Map<string, { userCode: string; approved: boolean; userId: string }>();
|
|
@@ -96,15 +100,16 @@ export async function startFakeApi(opts: { voiceoverCharLimit?: number; imageLim
|
|
|
96
100
|
// `sink` is where the accepted bytes go: a library item, or a cutout job.
|
|
97
101
|
const uploads = new Map<string, { contentType: string; bytes: number; used: boolean; sink: { has(): boolean; put(bytes: Uint8Array): void } | undefined }>();
|
|
98
102
|
// Totals over every user (for tests of the CLI), and the same counts per user (what a quota is measured against).
|
|
99
|
-
const usage = { chars: 0, images: 0, clips: 0, cutoutSeconds: 0, pulls: 0, uploads: 0 };
|
|
100
|
-
const meters = new Map<string, { chars: number; images: number; clips: number; cutoutSeconds: number; uploads: number }>();
|
|
103
|
+
const usage = { chars: 0, images: 0, clips: 0, cutoutSeconds: 0, transcribeSeconds: 0, pulls: 0, uploads: 0 };
|
|
104
|
+
const meters = new Map<string, { chars: number; images: number; clips: number; cutoutSeconds: number; transcribeSeconds: number; uploads: number }>();
|
|
101
105
|
const meter = (userId: string | undefined) => {
|
|
102
106
|
const id = userId ?? "u-test";
|
|
103
|
-
if (!meters.has(id)) meters.set(id, { chars: 0, images: 0, clips: 0, cutoutSeconds: 0, uploads: 0 });
|
|
107
|
+
if (!meters.has(id)) meters.set(id, { chars: 0, images: 0, clips: 0, cutoutSeconds: 0, transcribeSeconds: 0, uploads: 0 });
|
|
104
108
|
return meters.get(id)!;
|
|
105
109
|
};
|
|
106
110
|
const resetDate = () => nextMonthStart().toISOString();
|
|
107
|
-
const limits = { chars: opts.voiceoverCharLimit ?? 10_000, images: opts.imageLimit ?? 30, clips: opts.clipLimit ?? 5, cutoutSeconds: opts.cutoutSecondsLimit ?? 120 };
|
|
111
|
+
const limits = { chars: opts.voiceoverCharLimit ?? 10_000, images: opts.imageLimit ?? 30, clips: opts.clipLimit ?? 5, cutoutSeconds: opts.cutoutSecondsLimit ?? 120, transcribeSeconds: opts.transcribeSecondsLimit ?? 1800 };
|
|
112
|
+
const transcripts = new Map<string, TranscriptJob>();
|
|
108
113
|
const jobs = new Map<string, ClipJob>();
|
|
109
114
|
const cutouts = new Map<string, CutoutJob>();
|
|
110
115
|
let cutoutFile: Blob | undefined;
|
|
@@ -141,7 +146,7 @@ export async function startFakeApi(opts: { voiceoverCharLimit?: number; imageLim
|
|
|
141
146
|
const m = meter(userId);
|
|
142
147
|
return {
|
|
143
148
|
userId: userId ?? "u-test", handle: handleOf(userId ?? "u-test"),
|
|
144
|
-
quota: { voiceoverChars: { used: m.chars, limit: limits.chars }, images: { used: m.images, limit: limits.images }, clips: { used: m.clips, limit: limits.clips }, cutoutSeconds: { used: m.cutoutSeconds, limit: limits.cutoutSeconds }, resetsAt: resetDate() },
|
|
149
|
+
quota: { voiceoverChars: { used: m.chars, limit: limits.chars }, images: { used: m.images, limit: limits.images }, clips: { used: m.clips, limit: limits.clips }, cutoutSeconds: { used: m.cutoutSeconds, limit: limits.cutoutSeconds }, transcribeSeconds: { used: m.transcribeSeconds, limit: limits.transcribeSeconds }, resetsAt: resetDate() },
|
|
145
150
|
// What the user gave the shared library that was accepted: their own items that are published. One waiting for review does not count.
|
|
146
151
|
contributions: [...store.values()].filter((x) => x.owner === userId && x.committed && x.item.visibility === "published").length,
|
|
147
152
|
};
|
|
@@ -281,6 +286,37 @@ export async function startFakeApi(opts: { voiceoverCharLimit?: number; imageLim
|
|
|
281
286
|
}
|
|
282
287
|
return { id, status: "done", url: fileUrl((cutoutFile ??= makeCutout())), ext: "webm", contentType: "video/webm" };
|
|
283
288
|
},
|
|
289
|
+
transcribeStart: (input, { userId }) => {
|
|
290
|
+
const id = nextId("tr");
|
|
291
|
+
const job: TranscriptJob = { owner: userId ?? "u-test", filename: input.filename, durationSec: input.durationSec };
|
|
292
|
+
transcripts.set(id, job);
|
|
293
|
+
const token = secret();
|
|
294
|
+
uploads.set(token, { contentType: input.contentType, bytes: input.bytes, used: false, sink: { has: () => Boolean(job.file), put: (bytes) => { job.file = bytes; } } });
|
|
295
|
+
return { id, uploadUrl: `${origin}/upload/${token}` };
|
|
296
|
+
},
|
|
297
|
+
transcribeRun: ({ id }, { userId }) => {
|
|
298
|
+
const job = transcripts.get(id);
|
|
299
|
+
if (!job || job.owner !== (userId ?? "u-test")) throw new Fail(404, "not_found", `No transcript with id ${id}.`);
|
|
300
|
+
// Asked again, it answers the same transcript without a second charge.
|
|
301
|
+
if (job.result) return job.result;
|
|
302
|
+
if (!job.file) throw new Fail(400, "invalid_request", `Nothing was uploaded for ${id}. Send the audio to the upload URL first.`);
|
|
303
|
+
const seconds = Math.ceil(job.durationSec);
|
|
304
|
+
const m = meter(userId);
|
|
305
|
+
if (m.transcribeSeconds + seconds > limits.transcribeSeconds) throw new Fail(429, "quota_exceeded", `Transcription quota used up. It resets on ${resetDate().slice(0, 10)}.`);
|
|
306
|
+
// The audio is deleted when the call ends, whatever the result; a failure never keeps the seconds.
|
|
307
|
+
job.file = undefined;
|
|
308
|
+
if (job.filename.includes(TRANSCRIBE_FAIL)) throw new Fail(500, "server_error", "The audio could not be transcribed.");
|
|
309
|
+
m.transcribeSeconds += seconds;
|
|
310
|
+
usage.transcribeSeconds += seconds;
|
|
311
|
+
const base = job.filename.replace(/\.[^.]*$/, "").replace(/[^A-Za-z0-9]+/g, " ").trim() || "audio";
|
|
312
|
+
const half = Math.round((job.durationSec / 2) * 100) / 100;
|
|
313
|
+
job.result = {
|
|
314
|
+
id, language: "en", durationSec: job.durationSec,
|
|
315
|
+
text: `This is ${base}. It ends here.`,
|
|
316
|
+
segments: [{ text: `This is ${base}.`, startSec: 0, endSec: half }, { text: "It ends here.", startSec: half, endSec: job.durationSec }],
|
|
317
|
+
};
|
|
318
|
+
return job.result;
|
|
319
|
+
},
|
|
284
320
|
publicLibrary: ({ q, kind, page }) => {
|
|
285
321
|
// Published items only. Newest first; with a query, best match first and equal matches newest first.
|
|
286
322
|
const want = q ? words(q) : undefined;
|
|
@@ -348,6 +384,8 @@ export async function startFakeApi(opts: { voiceoverCharLimit?: number; imageLim
|
|
|
348
384
|
setClipLimit(n: number) { limits.clips = n; },
|
|
349
385
|
// The same for the seconds of video a user may have cut out.
|
|
350
386
|
setCutoutLimit(n: number) { limits.cutoutSeconds = n; },
|
|
387
|
+
// The same for the seconds of audio a user may have transcribed.
|
|
388
|
+
setTranscribeLimit(n: number) { limits.transcribeSeconds = n; },
|
|
351
389
|
// Stops cutout jobs from finishing (true) or lets them finish again (false).
|
|
352
390
|
holdCutouts(hold: boolean) { cutoutsHeld = hold; },
|
|
353
391
|
approve(userCode: string, userId = "u-test") { for (const d of devices.values()) if (d.userCode === userCode) { d.approved = true; d.userId = userId; } },
|