nixamp 0.17.0 → 0.18.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/channels.d.ts +54 -1
- package/dist/channels.js +160 -23
- package/dist/compression/benchmark.d.ts +97 -0
- package/dist/compression/benchmark.js +241 -0
- package/dist/compression/cli.js +59 -1
- package/dist/compression/receiver.d.ts +25 -5
- package/dist/compression/receiver.js +54 -5
- package/dist/compression/routes.js +7 -2
- package/dist/compression/service.d.ts +18 -6
- package/dist/compression/service.js +110 -19
- package/dist/compression/source.d.ts +17 -0
- package/dist/compression/source.js +131 -0
- package/dist/links.js +14 -3
- package/dist/live-api.js +1 -0
- package/dist/live-events.js +11 -3
- package/dist/owner.js +10 -0
- package/dist/server.js +118 -10
- package/dist/share.js +6 -0
- package/package.json +1 -1
- package/src/channels.ts +180 -22
- package/src/compression/benchmark.ts +297 -0
- package/src/compression/cli.ts +56 -1
- package/src/compression/receiver.ts +68 -7
- package/src/compression/routes.ts +7 -2
- package/src/compression/service.ts +121 -24
- package/src/compression/source.ts +130 -0
- package/src/links.ts +12 -3
- package/src/live-api.ts +1 -0
- package/src/live-events.ts +14 -3
- package/src/owner.ts +8 -0
- package/src/server.ts +113 -11
- package/src/share.ts +5 -0
- package/web/dist/assets/{hls-3VKVEQE3-eV54kXE3.js → hls-3VKVEQE3-B2kl0-z1.js} +1 -1
- package/web/dist/assets/{index-CurZFzlH.css → index-D1QxpRmE.css} +1 -1
- package/web/dist/assets/index-DF6O2leR.js +1 -0
- package/web/dist/assets/{mpegts-Buc3Odv6.js → mpegts-4pBNK_Yr.js} +1 -1
- package/web/dist/assets/{mpegts-LO6RVLD6-DcDKPB4P.js → mpegts-LO6RVLD6-B4RLM-_e.js} +1 -1
- package/web/dist/index.html +15 -10
- package/web/dist/sw.js +6 -6
- package/web/dist/assets/index-DDzutJ75.js +0 -1
package/package.json
CHANGED
package/src/channels.ts
CHANGED
|
@@ -80,6 +80,13 @@ export interface ChannelInfo {
|
|
|
80
80
|
* a member may have on at once.
|
|
81
81
|
*/
|
|
82
82
|
startedBy?: string;
|
|
83
|
+
/**
|
|
84
|
+
* Set when nixamp reads the source itself and hands ffmpeg the bytes down
|
|
85
|
+
* a pipe, so the original bytes pass through this process and can be
|
|
86
|
+
* relayed exactly as they came. Only a transport stream from a file or a
|
|
87
|
+
* plain URL is read this way, and only when a policy asks for it.
|
|
88
|
+
*/
|
|
89
|
+
teed?: boolean;
|
|
83
90
|
}
|
|
84
91
|
|
|
85
92
|
/** Where a pulled source is picked up from, and whether it can be at all. */
|
|
@@ -90,6 +97,24 @@ export interface PullResume {
|
|
|
90
97
|
position: number;
|
|
91
98
|
}
|
|
92
99
|
|
|
100
|
+
/**
|
|
101
|
+
* A source read by us rather than by ffmpeg: what to tell ffmpeg it is,
|
|
102
|
+
* and how to open it. Opened once per dial; the signal is pulled when that
|
|
103
|
+
* dial is over.
|
|
104
|
+
*/
|
|
105
|
+
export interface PullThrough {
|
|
106
|
+
/** ffmpeg's name for the container, e.g. `mpegts`. */
|
|
107
|
+
format: string;
|
|
108
|
+
open(signal: AbortSignal): Promise<AsyncIterable<Uint8Array>>;
|
|
109
|
+
}
|
|
110
|
+
|
|
111
|
+
/**
|
|
112
|
+
* Asked at every dial whether this source should be read through us. Null
|
|
113
|
+
* means ffmpeg dials it as it always has. `from` is where a film is being
|
|
114
|
+
* picked up from; a source read through us cannot be joined mid-way.
|
|
115
|
+
*/
|
|
116
|
+
export type ThroughProvider = (info: ChannelInfo, from: number, input: string[], audio: string) => PullThrough | null;
|
|
117
|
+
|
|
93
118
|
/**
|
|
94
119
|
* How far back of the saved place a film is picked up from, in seconds. The
|
|
95
120
|
* place is written down every so often and a restart lands between two
|
|
@@ -203,6 +228,8 @@ export interface ChannelOptions {
|
|
|
203
228
|
maxListenerQueueBytes?: number;
|
|
204
229
|
/** How long a rate is measured over before the backlog is sized off it. Tests shorten it. */
|
|
205
230
|
rateWindowMs?: number;
|
|
231
|
+
/** Whether a pulled source is read here and piped to ffmpeg. See `Channels.setThrough`. */
|
|
232
|
+
through?: ThroughProvider;
|
|
206
233
|
}
|
|
207
234
|
|
|
208
235
|
/**
|
|
@@ -249,6 +276,10 @@ export class Channel {
|
|
|
249
276
|
*/
|
|
250
277
|
ephemeral = false;
|
|
251
278
|
private idle: ReturnType<typeof setTimeout> | null = null;
|
|
279
|
+
/** Whoever wants the source's own bytes, when the source is read through us. */
|
|
280
|
+
private readonly sourceTaps = new Set<Listener>();
|
|
281
|
+
/** Pulls the plug on the current read-through, when there is one. */
|
|
282
|
+
private throughAbort: AbortController | null = null;
|
|
252
283
|
|
|
253
284
|
constructor(
|
|
254
285
|
readonly info: ChannelInfo,
|
|
@@ -337,6 +368,15 @@ export class Channel {
|
|
|
337
368
|
// and a film that has barely started is started.
|
|
338
369
|
const from = resume.live ? 0 : Math.max(0, Math.floor((this.info.position ?? 0) - REWIND));
|
|
339
370
|
const seek = from > 0 ? ["-ss", String(from)] : [];
|
|
371
|
+
// Read the source here rather than in ffmpeg, when a policy wants the
|
|
372
|
+
// original bytes and the source is the kind that can be. ffmpeg then
|
|
373
|
+
// reads a pipe, and every byte that goes down it is also handed to
|
|
374
|
+
// whoever has tapped the source. Asked again at every dial, so a
|
|
375
|
+
// policy set after the channel started applies at its next restart.
|
|
376
|
+
this.throughAbort?.abort();
|
|
377
|
+
this.throughAbort = null;
|
|
378
|
+
const through = this.options.through?.(this.info, from, input, audio) ?? null;
|
|
379
|
+
this.info.teed = through !== null;
|
|
340
380
|
const child = spawn(
|
|
341
381
|
command,
|
|
342
382
|
[
|
|
@@ -348,36 +388,54 @@ export class Channel {
|
|
|
348
388
|
// to. Not stderr, which is for what went wrong.
|
|
349
389
|
"-progress", "pipe:3",
|
|
350
390
|
"-stats_period", "1",
|
|
351
|
-
|
|
352
|
-
|
|
353
|
-
|
|
354
|
-
|
|
355
|
-
|
|
356
|
-
|
|
357
|
-
|
|
358
|
-
|
|
359
|
-
|
|
360
|
-
|
|
361
|
-
|
|
362
|
-
|
|
363
|
-
|
|
364
|
-
|
|
365
|
-
|
|
366
|
-
|
|
367
|
-
|
|
368
|
-
|
|
369
|
-
|
|
370
|
-
|
|
371
|
-
|
|
391
|
+
...(through
|
|
392
|
+
? [
|
|
393
|
+
// Stated, as for a publisher: ffmpeg mis-probes a pipe. Paced
|
|
394
|
+
// the same way, since a pipe is read as fast as it is written.
|
|
395
|
+
// The source's own input tuning still applies -- a transport
|
|
396
|
+
// stream read from a pipe needs the same probe depth and
|
|
397
|
+
// generated timestamps it would off a socket -- but request
|
|
398
|
+
// headers, which only a source read through us leaves out,
|
|
399
|
+
// are no use to a pipe and are not here (a source that needs
|
|
400
|
+
// them is not read through us in the first place).
|
|
401
|
+
...input,
|
|
402
|
+
"-f", through.format,
|
|
403
|
+
...(paced ? ["-re"] : []),
|
|
404
|
+
"-i", "pipe:0",
|
|
405
|
+
]
|
|
406
|
+
: [
|
|
407
|
+
// A dropped source is normal over hours, and a channel that dies
|
|
408
|
+
// the first time a CDN hiccups is not a channel anybody can rely
|
|
409
|
+
// on. ffmpeg redials on its own before we have to.
|
|
410
|
+
// A connection that stops answering is an error after this long,
|
|
411
|
+
// and an error is a thing the reconnect knows what to do with.
|
|
412
|
+
// Without it a silent socket is waited on for ever. In
|
|
413
|
+
// microseconds, as ffmpeg wants it.
|
|
414
|
+
...remoteArgs,
|
|
415
|
+
// Real time, always. A file read as fast as the disk allows is an
|
|
416
|
+
// hour of film in ninety seconds and a room that cannot be in it
|
|
417
|
+
// together; a live source is already paced and loses nothing.
|
|
418
|
+
...(paced ? ["-re"] : []),
|
|
419
|
+
// What the source's site expects on the request: a user agent, a
|
|
420
|
+
// referer, a cookie. A link resolved by yt-dlp comes with these,
|
|
421
|
+
// and a CDN that got them from yt-dlp and not from us answers 403.
|
|
422
|
+
...input,
|
|
423
|
+
...seek,
|
|
424
|
+
"-i", source,
|
|
425
|
+
// The sound, when the site keeps it apart from the picture: a
|
|
426
|
+
// second input, dialled the same way, that the encode maps in.
|
|
427
|
+
...(audio ? [...remoteArgs, ...(paced ? ["-re"] : []), ...input, ...seek, "-i", audio] : []),
|
|
428
|
+
]),
|
|
372
429
|
...encode,
|
|
373
430
|
"pipe:1",
|
|
374
431
|
],
|
|
375
|
-
{ stdio: ["ignore", "pipe", "pipe", "pipe"] },
|
|
432
|
+
{ stdio: [through ? "pipe" : "ignore", "pipe", "pipe", "pipe"] },
|
|
376
433
|
);
|
|
377
434
|
|
|
378
435
|
let sent = false;
|
|
379
436
|
this.child = child;
|
|
380
437
|
this.rearm(child);
|
|
438
|
+
if (through) void this.feedThrough(child, through);
|
|
381
439
|
// ffmpeg's progress: key=value lines, out_time_us being how much it
|
|
382
440
|
// has written, from where it was told to start. Read whole lines,
|
|
383
441
|
// since a chunk can end mid-number. Drained whatever it says, for
|
|
@@ -463,6 +521,22 @@ export class Channel {
|
|
|
463
521
|
this.rateBytes = 0;
|
|
464
522
|
this.rate = 0;
|
|
465
523
|
this.hangUp();
|
|
524
|
+
// The source's own bytes start over too: a new dial is a new stream
|
|
525
|
+
// from its beginning, and whoever was tapping it must not be handed the
|
|
526
|
+
// new beginning after the old middle.
|
|
527
|
+
this.endTaps();
|
|
528
|
+
}
|
|
529
|
+
|
|
530
|
+
/** Everybody tapping the source is told it ended. */
|
|
531
|
+
private endTaps(): void {
|
|
532
|
+
for (const tap of this.sourceTaps) {
|
|
533
|
+
try {
|
|
534
|
+
tap.end();
|
|
535
|
+
} catch {
|
|
536
|
+
// Gone already.
|
|
537
|
+
}
|
|
538
|
+
}
|
|
539
|
+
this.sourceTaps.clear();
|
|
466
540
|
}
|
|
467
541
|
|
|
468
542
|
/** Expect output within STALL, or treat the source as gone and dial again. */
|
|
@@ -671,6 +745,69 @@ export class Channel {
|
|
|
671
745
|
this.startOver();
|
|
672
746
|
}
|
|
673
747
|
|
|
748
|
+
/**
|
|
749
|
+
* Hear the source's own bytes, as they go down the pipe to ffmpeg. Only
|
|
750
|
+
* a channel read through us has any; for the rest this attaches nothing
|
|
751
|
+
* and returns null. Ended, like a listener, when the source starts over.
|
|
752
|
+
*/
|
|
753
|
+
tapSource(tap: Listener): (() => void) | null {
|
|
754
|
+
if (!this.info.teed || this.closing) return null;
|
|
755
|
+
this.sourceTaps.add(tap);
|
|
756
|
+
return () => {
|
|
757
|
+
this.sourceTaps.delete(tap);
|
|
758
|
+
};
|
|
759
|
+
}
|
|
760
|
+
|
|
761
|
+
private tap(chunk: Buffer): void {
|
|
762
|
+
for (const tap of this.sourceTaps) {
|
|
763
|
+
try {
|
|
764
|
+
tap.write(chunk);
|
|
765
|
+
} catch {
|
|
766
|
+
this.sourceTaps.delete(tap);
|
|
767
|
+
}
|
|
768
|
+
}
|
|
769
|
+
}
|
|
770
|
+
|
|
771
|
+
/**
|
|
772
|
+
* Read the source and write it to this ffmpeg's stdin, at the rate ffmpeg
|
|
773
|
+
* takes it. Every chunk is handed to the taps first, so a tap sees exactly
|
|
774
|
+
* the bytes ffmpeg does. When the source ends, stdin is ended, ffmpeg
|
|
775
|
+
* finishes, and its close handler dials again -- the same path a source
|
|
776
|
+
* that ffmpeg read itself takes when it drops.
|
|
777
|
+
*/
|
|
778
|
+
private async feedThrough(child: ChildProcess, through: PullThrough): Promise<void> {
|
|
779
|
+
const controller = new AbortController();
|
|
780
|
+
this.throughAbort = controller;
|
|
781
|
+
const stdin = child.stdin;
|
|
782
|
+
if (!stdin) return;
|
|
783
|
+
stdin.on("error", () => undefined);
|
|
784
|
+
try {
|
|
785
|
+
const body = await through.open(controller.signal);
|
|
786
|
+
for await (const raw of body) {
|
|
787
|
+
if (this.child !== child || this.closing || controller.signal.aborted) break;
|
|
788
|
+
const chunk = Buffer.isBuffer(raw) ? raw : Buffer.from(raw);
|
|
789
|
+
this.tap(chunk);
|
|
790
|
+
if (!stdin.write(chunk)) {
|
|
791
|
+
// Wait for ffmpeg to take it, or for the pipe to go: a pipe that
|
|
792
|
+
// closed never drains, and waiting on it would hold the read open.
|
|
793
|
+
await new Promise<void>((done) => {
|
|
794
|
+
stdin.once("drain", done);
|
|
795
|
+
stdin.once("close", done);
|
|
796
|
+
});
|
|
797
|
+
}
|
|
798
|
+
}
|
|
799
|
+
} catch (error) {
|
|
800
|
+
if (this.child === child && !controller.signal.aborted) this.info.error = (error as Error).message;
|
|
801
|
+
} finally {
|
|
802
|
+
try {
|
|
803
|
+
stdin.end();
|
|
804
|
+
} catch {
|
|
805
|
+
// Already gone.
|
|
806
|
+
}
|
|
807
|
+
if (this.throughAbort === controller) this.throughAbort = null;
|
|
808
|
+
}
|
|
809
|
+
}
|
|
810
|
+
|
|
674
811
|
listen(listener: Listener): () => void {
|
|
675
812
|
// What the stream is, before any of what it is currently saying. Without
|
|
676
813
|
// this a listener who arrives after the first second gets fragments that
|
|
@@ -735,6 +872,9 @@ export class Channel {
|
|
|
735
872
|
if (said && !this.info.error) this.info.error = said;
|
|
736
873
|
const child = this.child;
|
|
737
874
|
this.child = null;
|
|
875
|
+
this.throughAbort?.abort();
|
|
876
|
+
this.throughAbort = null;
|
|
877
|
+
this.endTaps();
|
|
738
878
|
try {
|
|
739
879
|
child?.stdin?.end();
|
|
740
880
|
} catch {
|
|
@@ -838,6 +978,7 @@ export class Channels {
|
|
|
838
978
|
input: string[] = [],
|
|
839
979
|
audio = "",
|
|
840
980
|
resume: PullResume = { live: true, position: 0 },
|
|
981
|
+
codecs?: ChannelInfo["codecs"],
|
|
841
982
|
): Channel | null {
|
|
842
983
|
if (this.open.has(id)) return null;
|
|
843
984
|
const channel = new Channel(
|
|
@@ -851,6 +992,9 @@ export class Channels {
|
|
|
851
992
|
listeners: 0,
|
|
852
993
|
kind,
|
|
853
994
|
source,
|
|
995
|
+
// Known before the first dial, so whether to read the source here
|
|
996
|
+
// can be decided from what it holds.
|
|
997
|
+
...(codecs ? { codecs } : {}),
|
|
854
998
|
},
|
|
855
999
|
this.options,
|
|
856
1000
|
(gone) => this.open.delete(gone),
|
|
@@ -965,6 +1109,20 @@ export class Channels {
|
|
|
965
1109
|
return this.open.get(id)?.opening() ?? [];
|
|
966
1110
|
}
|
|
967
1111
|
|
|
1112
|
+
/** Hear a channel's source bytes, when it is read through us. Null otherwise. */
|
|
1113
|
+
tapSource(id: string, tap: Listener): (() => void) | null {
|
|
1114
|
+
return this.open.get(id)?.tapSource(tap) ?? null;
|
|
1115
|
+
}
|
|
1116
|
+
|
|
1117
|
+
/**
|
|
1118
|
+
* Who decides whether a pulled source is read here and piped to ffmpeg.
|
|
1119
|
+
* Set once by whoever owns the policies; asked at every dial.
|
|
1120
|
+
*/
|
|
1121
|
+
setThrough(provider: ThroughProvider | null): void {
|
|
1122
|
+
if (provider) this.options.through = provider;
|
|
1123
|
+
else delete this.options.through;
|
|
1124
|
+
}
|
|
1125
|
+
|
|
968
1126
|
/** The kind of a channel, for a relay to say what it is carrying. */
|
|
969
1127
|
kindOf(id: string): "audio" | "video" | undefined {
|
|
970
1128
|
return this.open.get(id)?.info.kind;
|
|
@@ -0,0 +1,297 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The OpenStream benchmark: what a codec makes of a defined corpus, on this
|
|
3
|
+
* machine, with every number reproducible and nothing claimed that was not
|
|
4
|
+
* measured.
|
|
5
|
+
*
|
|
6
|
+
* It proves the honest things: that decompression restores every byte, that
|
|
7
|
+
* an incompressible sample costs only the envelope overhead and never more,
|
|
8
|
+
* that a compressible one saves what it says against the complete wire size,
|
|
9
|
+
* and how long each takes. It does not prove a production saving -- a
|
|
10
|
+
* synthetic padded stream compresses to almost nothing, which says more about
|
|
11
|
+
* the padding than the codec, and the report labels it so. Point it at real
|
|
12
|
+
* authorized samples with `--corpus` for numbers that mean something.
|
|
13
|
+
*
|
|
14
|
+
* The corpus is built from bytes alone by default, so anyone can run it with
|
|
15
|
+
* no ffmpeg and no media: random data, zeros, repetitive text, tiny and empty
|
|
16
|
+
* inputs, and hand-built transport-stream packets with a known share of null
|
|
17
|
+
* padding. ffmpeg, when present, adds real encoded media; a directory of your
|
|
18
|
+
* own files replaces the lot.
|
|
19
|
+
*/
|
|
20
|
+
import { createHash } from "node:crypto";
|
|
21
|
+
import { readdirSync, readFileSync, statSync } from "node:fs";
|
|
22
|
+
import { arch, cpus, platform, release, totalmem } from "node:os";
|
|
23
|
+
import { join } from "node:path";
|
|
24
|
+
import { type BenchRow, benchmark } from "./analyze.ts";
|
|
25
|
+
import { toolVersions } from "./codec.ts";
|
|
26
|
+
import { ENVELOPE_VERSION, FRAME_HEADER_BYTES, MAGIC, STREAM_HEADER_BYTES } from "./envelope.ts";
|
|
27
|
+
import { DEFAULT_LOSSLESS } from "./policy.ts";
|
|
28
|
+
import { SYNC, TS_PACKET } from "./ts-transform.ts";
|
|
29
|
+
|
|
30
|
+
/** The report schema version, bumped when the shape below changes. */
|
|
31
|
+
export const REPORT_SCHEMA = 1;
|
|
32
|
+
|
|
33
|
+
export interface Sample {
|
|
34
|
+
name: string;
|
|
35
|
+
/** synthetic bytes we generated, or a real file the runner supplied. */
|
|
36
|
+
kind: "synthetic" | "real";
|
|
37
|
+
bytes: Buffer;
|
|
38
|
+
/** A word on what it is and why it is in the corpus. */
|
|
39
|
+
notes: string;
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
export interface SampleResult {
|
|
43
|
+
sample: string;
|
|
44
|
+
kind: Sample["kind"];
|
|
45
|
+
inputBytes: number;
|
|
46
|
+
sha256: string;
|
|
47
|
+
container: string;
|
|
48
|
+
rows: BenchRow[];
|
|
49
|
+
/** The mode the policy would pick for this sample, and why. */
|
|
50
|
+
recommendation: { mode: string; level: number; reason: string };
|
|
51
|
+
}
|
|
52
|
+
|
|
53
|
+
export interface BenchmarkReport {
|
|
54
|
+
schema: number;
|
|
55
|
+
spec: "openstream";
|
|
56
|
+
specVersion: string;
|
|
57
|
+
generatedAt: string;
|
|
58
|
+
implementation: { name: string; version: string };
|
|
59
|
+
environment: {
|
|
60
|
+
runtime: string;
|
|
61
|
+
zstd: string;
|
|
62
|
+
zlib: string;
|
|
63
|
+
os: string;
|
|
64
|
+
arch: string;
|
|
65
|
+
cpu: string;
|
|
66
|
+
cores: number;
|
|
67
|
+
memoryGiB: number;
|
|
68
|
+
};
|
|
69
|
+
envelope: { magic: string; streamHeaderBytes: number; frameHeaderBytes: number };
|
|
70
|
+
policy: { minSavingsPercent: number; minSavingsBytes: number; maxBlockBytes: number; zstdLevels: number[] };
|
|
71
|
+
samples: SampleResult[];
|
|
72
|
+
/** Per mode, summed across the corpus: the honest aggregate. */
|
|
73
|
+
summary: {
|
|
74
|
+
corpusBytes: number;
|
|
75
|
+
byMode: { mode: string; level: number; wireBytes: number; savingsPercent: number; roundTrip: boolean; encodeMs: number; decodeMs: number }[];
|
|
76
|
+
};
|
|
77
|
+
caveats: string[];
|
|
78
|
+
}
|
|
79
|
+
|
|
80
|
+
/**
|
|
81
|
+
* A deterministic, genuinely incompressible fill: a SHA-256 keystream from a
|
|
82
|
+
* fixed seed. Reproducible byte-for-byte across runs (so two reports compare)
|
|
83
|
+
* and uncompressible (so the "floor" sample is really the floor, unlike a
|
|
84
|
+
* linear-congruential stream, whose periodicity a compressor crushes).
|
|
85
|
+
*/
|
|
86
|
+
function keystream(size: number, seed: string): Buffer {
|
|
87
|
+
const out = Buffer.alloc(size);
|
|
88
|
+
let block = createHash("sha256").update(seed).digest();
|
|
89
|
+
let at = 0;
|
|
90
|
+
while (at < size) {
|
|
91
|
+
const take = Math.min(block.length, size - at);
|
|
92
|
+
block.copy(out, at, 0, take);
|
|
93
|
+
at += take;
|
|
94
|
+
block = createHash("sha256").update(block).digest();
|
|
95
|
+
}
|
|
96
|
+
return out;
|
|
97
|
+
}
|
|
98
|
+
|
|
99
|
+
/** Transport-stream packets, a given share of them null padding, payloads incompressible. */
|
|
100
|
+
function fakeTs(packets: number, nullEvery: number, seed = "ts"): Buffer {
|
|
101
|
+
const out = Buffer.alloc(packets * TS_PACKET);
|
|
102
|
+
const fill = keystream(packets * TS_PACKET, seed);
|
|
103
|
+
for (let i = 0; i < packets; i += 1) {
|
|
104
|
+
const at = i * TS_PACKET;
|
|
105
|
+
const isNull = nullEvery > 0 && i % nullEvery === 0;
|
|
106
|
+
const pid = isNull ? 0x1fff : [0x100, 0x101, 0x102][i % 3] as number;
|
|
107
|
+
out[at] = SYNC;
|
|
108
|
+
out[at + 1] = (pid >> 8) & 0x1f;
|
|
109
|
+
out[at + 2] = pid & 0xff;
|
|
110
|
+
out[at + 3] = 0x10 | (i & 0xf);
|
|
111
|
+
for (let j = at + 4; j < at + TS_PACKET; j += 1) out[j] = isNull ? 0xff : (fill[j] as number);
|
|
112
|
+
}
|
|
113
|
+
return out;
|
|
114
|
+
}
|
|
115
|
+
|
|
116
|
+
/** The default corpus: bytes only, no ffmpeg, deterministic. */
|
|
117
|
+
export function syntheticCorpus(): Sample[] {
|
|
118
|
+
return [
|
|
119
|
+
{ name: "random-1mib", kind: "synthetic", bytes: keystream(1024 * 1024, "random"), notes: "incompressible: the floor, where a codec must not grow the data beyond envelope overhead" },
|
|
120
|
+
{ name: "zeros-1mib", kind: "synthetic", bytes: Buffer.alloc(1024 * 1024, 0), notes: "maximally compressible: the ceiling" },
|
|
121
|
+
{ name: "text-repeat-1mib", kind: "synthetic", bytes: Buffer.from("the quick brown fox jumps over the lazy dog\n".repeat(24000)).subarray(0, 1024 * 1024), notes: "repetitive text: ordinary redundancy" },
|
|
122
|
+
{ name: "ts-padded-50pct", kind: "synthetic", bytes: fakeTs(4000, 2), notes: "transport stream, half null packets: padding a copy must keep and a codec removes; the synthetic case that flatters a codec" },
|
|
123
|
+
{ name: "ts-unpadded", kind: "synthetic", bytes: fakeTs(4000, 0), notes: "transport stream, no padding: closer to an efficient real feed" },
|
|
124
|
+
{ name: "tiny-3b", kind: "synthetic", bytes: Buffer.from("abc"), notes: "smaller than a frame header: proves overhead is reported honestly" },
|
|
125
|
+
{ name: "empty", kind: "synthetic", bytes: Buffer.alloc(0), notes: "the empty stream" },
|
|
126
|
+
];
|
|
127
|
+
}
|
|
128
|
+
|
|
129
|
+
/** Real files a runner points us at, each read whole (capped) as a sample. */
|
|
130
|
+
export function fileCorpus(dir: string, capBytes = 64 * 1024 * 1024): Sample[] {
|
|
131
|
+
const out: Sample[] = [];
|
|
132
|
+
for (const name of readdirSync(dir).sort()) {
|
|
133
|
+
const path = join(dir, name);
|
|
134
|
+
try {
|
|
135
|
+
if (!statSync(path).isFile()) continue;
|
|
136
|
+
const whole = readFileSync(path);
|
|
137
|
+
out.push({ name, kind: "real", bytes: whole.subarray(0, capBytes), notes: whole.length > capBytes ? `real file, first ${capBytes} bytes of ${whole.length}` : "real file" });
|
|
138
|
+
} catch {
|
|
139
|
+
// Unreadable entries are skipped, not fatal.
|
|
140
|
+
}
|
|
141
|
+
}
|
|
142
|
+
return out;
|
|
143
|
+
}
|
|
144
|
+
|
|
145
|
+
export interface RunOptions {
|
|
146
|
+
implementationVersion: string;
|
|
147
|
+
zstdLevels?: number[];
|
|
148
|
+
blockBytes?: number;
|
|
149
|
+
signal?: AbortSignal;
|
|
150
|
+
}
|
|
151
|
+
|
|
152
|
+
/** Run the corpus and build the report. */
|
|
153
|
+
export async function runBenchmark(corpus: Sample[], options: RunOptions): Promise<BenchmarkReport> {
|
|
154
|
+
const zstdLevels = options.zstdLevels ?? [1, 3, 9];
|
|
155
|
+
const blockBytes = options.blockBytes ?? DEFAULT_LOSSLESS.maxBlockBytes;
|
|
156
|
+
const policy = { minSavingsPercent: DEFAULT_LOSSLESS.minSavingsPercent, minSavingsBytes: DEFAULT_LOSSLESS.minSavingsBytes };
|
|
157
|
+
const samples: SampleResult[] = [];
|
|
158
|
+
for (const sample of corpus) {
|
|
159
|
+
if (options.signal?.aborted) break;
|
|
160
|
+
const rows = await benchmark(sample.bytes, { blockBytes, zstdLevels, tsAware: true, policy, ...(options.signal ? { signal: options.signal } : {}) });
|
|
161
|
+
const stored = rows.find((r) => r.mode === "stored");
|
|
162
|
+
const best = rows.filter((r) => r.roundTrip && r.mode !== "stored").sort((a, b) => a.wireBytes - b.wireBytes)[0];
|
|
163
|
+
const beats = stored && best && stored.wireBytes - best.wireBytes >= policy.minSavingsBytes && ((stored.wireBytes - best.wireBytes) * 100) / stored.wireBytes >= policy.minSavingsPercent;
|
|
164
|
+
samples.push({
|
|
165
|
+
sample: sample.name,
|
|
166
|
+
kind: sample.kind,
|
|
167
|
+
inputBytes: sample.bytes.length,
|
|
168
|
+
sha256: createHash("sha256").update(sample.bytes).digest("hex"),
|
|
169
|
+
container: rows.length ? sniff(sample.bytes) : "empty",
|
|
170
|
+
rows,
|
|
171
|
+
recommendation: beats && best
|
|
172
|
+
? { mode: best.mode, level: best.level, reason: `saves ${best.savingsPercent}% of the complete wire size` }
|
|
173
|
+
: { mode: "stored", level: 0, reason: "no codec beat stored by the configured margin" },
|
|
174
|
+
});
|
|
175
|
+
}
|
|
176
|
+
const byMode = aggregate(samples);
|
|
177
|
+
const cpu = cpus()[0]?.model ?? "unknown";
|
|
178
|
+
return {
|
|
179
|
+
schema: REPORT_SCHEMA,
|
|
180
|
+
spec: "openstream",
|
|
181
|
+
specVersion: MAGIC,
|
|
182
|
+
generatedAt: new Date().toISOString(),
|
|
183
|
+
implementation: { name: "nixamp", version: options.implementationVersion },
|
|
184
|
+
environment: {
|
|
185
|
+
...toolVersions(),
|
|
186
|
+
os: `${platform()} ${release()}`,
|
|
187
|
+
arch: arch(),
|
|
188
|
+
cpu,
|
|
189
|
+
cores: cpus().length,
|
|
190
|
+
memoryGiB: Math.round((totalmem() / 1024 ** 3) * 10) / 10,
|
|
191
|
+
},
|
|
192
|
+
envelope: { magic: MAGIC, streamHeaderBytes: STREAM_HEADER_BYTES, frameHeaderBytes: FRAME_HEADER_BYTES },
|
|
193
|
+
policy: { ...policy, maxBlockBytes: blockBytes, zstdLevels },
|
|
194
|
+
samples,
|
|
195
|
+
summary: { corpusBytes: corpus.reduce((n, s) => n + s.bytes.length, 0), byMode },
|
|
196
|
+
caveats: [
|
|
197
|
+
`Envelope v${ENVELOPE_VERSION}: a ${STREAM_HEADER_BYTES}-byte stream header, a ${FRAME_HEADER_BYTES}-byte header per frame, plus one end frame.`,
|
|
198
|
+
"OpenStream is a framing envelope over Zstandard and gzip, not a new compression algorithm; these numbers are those codecs at the block boundary, honestly framed.",
|
|
199
|
+
"Synthetic samples do not predict production savings. A padded transport stream flatters a codec by its padding; an efficient real feed saves far less. Use --corpus with authorized real samples for numbers that mean something.",
|
|
200
|
+
"Timings are wall-clock on the machine and runtime named in `environment` and do not transfer to other hardware.",
|
|
201
|
+
"roundTrip:false in any row is a failure of exactness and must block a release.",
|
|
202
|
+
],
|
|
203
|
+
};
|
|
204
|
+
}
|
|
205
|
+
|
|
206
|
+
function sniff(bytes: Buffer): string {
|
|
207
|
+
// A light container guess for the report; the full analyzer is elsewhere.
|
|
208
|
+
if (bytes.length >= 8 && bytes.toString("latin1", 4, 8) === "ftyp") return "mp4";
|
|
209
|
+
if (bytes.length >= TS_PACKET && bytes[0] === SYNC && bytes[TS_PACKET] === SYNC) return "mpegts";
|
|
210
|
+
return "bytes";
|
|
211
|
+
}
|
|
212
|
+
|
|
213
|
+
function aggregate(samples: SampleResult[]): BenchmarkReport["summary"]["byMode"] {
|
|
214
|
+
const modes = new Map<string, { mode: string; level: number; wireBytes: number; input: number; roundTrip: boolean; encodeMs: number; decodeMs: number }>();
|
|
215
|
+
for (const sample of samples) {
|
|
216
|
+
for (const row of sample.rows) {
|
|
217
|
+
// A row with a note is a mode that did not apply to this sample (ts-zstd
|
|
218
|
+
// on non-transport bytes, say). It is not a saving and not a failure, so
|
|
219
|
+
// it is left out of the aggregate rather than dragging a mode down.
|
|
220
|
+
if (row.note) continue;
|
|
221
|
+
const key = `${row.mode}:${row.level}`;
|
|
222
|
+
const acc = modes.get(key) ?? { mode: row.mode, level: row.level, wireBytes: 0, input: 0, roundTrip: true, encodeMs: 0, decodeMs: 0 };
|
|
223
|
+
acc.wireBytes += row.wireBytes;
|
|
224
|
+
acc.input += row.inputBytes;
|
|
225
|
+
acc.roundTrip = acc.roundTrip && row.roundTrip;
|
|
226
|
+
acc.encodeMs += row.encodeMs;
|
|
227
|
+
acc.decodeMs += row.decodeMs;
|
|
228
|
+
modes.set(key, acc);
|
|
229
|
+
}
|
|
230
|
+
}
|
|
231
|
+
return [...modes.values()].map((m) => ({
|
|
232
|
+
mode: m.mode,
|
|
233
|
+
level: m.level,
|
|
234
|
+
wireBytes: m.wireBytes,
|
|
235
|
+
savingsPercent: m.input === 0 ? 0 : Math.round(((m.input - m.wireBytes) * 100) / m.input * 100) / 100,
|
|
236
|
+
roundTrip: m.roundTrip,
|
|
237
|
+
encodeMs: Math.round(m.encodeMs * 100) / 100,
|
|
238
|
+
decodeMs: Math.round(m.decodeMs * 100) / 100,
|
|
239
|
+
}));
|
|
240
|
+
}
|
|
241
|
+
|
|
242
|
+
/** The report as Markdown, for a human and for the reports page. */
|
|
243
|
+
export function reportMarkdown(report: BenchmarkReport): string {
|
|
244
|
+
const e = report.environment;
|
|
245
|
+
const lines: string[] = [];
|
|
246
|
+
lines.push(`# OpenStream benchmark — ${report.implementation.name} ${report.implementation.version}`);
|
|
247
|
+
lines.push("");
|
|
248
|
+
lines.push(`Generated ${report.generatedAt} · envelope ${report.specVersion} · schema ${report.schema}`);
|
|
249
|
+
lines.push("");
|
|
250
|
+
lines.push(`**Environment.** ${e.runtime}, zstd ${e.zstd}, zlib ${e.zlib}, on ${e.os} ${e.arch}, ${e.cores}× ${e.cpu}, ${e.memoryGiB} GiB.`);
|
|
251
|
+
lines.push("");
|
|
252
|
+
lines.push(`**Policy.** block ${report.policy.maxBlockBytes} B; eligible at ${report.policy.minSavingsPercent}% and ${report.policy.minSavingsBytes} B; zstd levels ${report.policy.zstdLevels.join(", ")}.`);
|
|
253
|
+
lines.push("");
|
|
254
|
+
lines.push("## Corpus");
|
|
255
|
+
lines.push("");
|
|
256
|
+
lines.push("| sample | kind | bytes | container | best mode | saves |");
|
|
257
|
+
lines.push("| --- | --- | ---: | --- | --- | ---: |");
|
|
258
|
+
for (const s of report.samples) {
|
|
259
|
+
const best = s.rows.filter((r) => r.roundTrip && r.mode !== "stored").sort((a, b) => a.wireBytes - b.wireBytes)[0];
|
|
260
|
+
const saves = s.recommendation.mode === "stored" ? "stored" : `${best?.savingsPercent ?? 0}%`;
|
|
261
|
+
lines.push(`| ${s.sample} | ${s.kind} | ${s.inputBytes} | ${s.container} | ${s.recommendation.mode}${s.recommendation.level ? ` L${s.recommendation.level}` : ""} | ${saves} |`);
|
|
262
|
+
}
|
|
263
|
+
lines.push("");
|
|
264
|
+
lines.push("## Aggregate, per mode across the corpus");
|
|
265
|
+
lines.push("");
|
|
266
|
+
lines.push("| mode | wire bytes | saving | enc ms | dec ms | round trip |");
|
|
267
|
+
lines.push("| --- | ---: | ---: | ---: | ---: | --- |");
|
|
268
|
+
for (const m of report.summary.byMode) {
|
|
269
|
+
lines.push(`| ${m.mode}${m.level ? ` L${m.level}` : ""} | ${m.wireBytes} | ${m.savingsPercent >= 0 ? "+" : ""}${m.savingsPercent}% | ${m.encodeMs} | ${m.decodeMs} | ${m.roundTrip ? "ok" : "FAILED"} |`);
|
|
270
|
+
}
|
|
271
|
+
lines.push("");
|
|
272
|
+
lines.push("## Caveats");
|
|
273
|
+
lines.push("");
|
|
274
|
+
for (const c of report.caveats) lines.push(`- ${c}`);
|
|
275
|
+
lines.push("");
|
|
276
|
+
return lines.join("\n");
|
|
277
|
+
}
|
|
278
|
+
|
|
279
|
+
/**
|
|
280
|
+
* Whether every applicable codec restored exactly: the gate a release must
|
|
281
|
+
* pass. A row with a note is a mode that did not apply to that sample, not a
|
|
282
|
+
* corrupted round trip, so it does not fail the gate.
|
|
283
|
+
*/
|
|
284
|
+
export function reportPasses(report: BenchmarkReport): boolean {
|
|
285
|
+
return report.samples.every((s) => s.rows.every((r) => r.roundTrip || Boolean(r.note)));
|
|
286
|
+
}
|
|
287
|
+
|
|
288
|
+
/** The rows that are a real exactness failure: ran, and did not restore. */
|
|
289
|
+
export function exactnessFailures(report: BenchmarkReport): { sample: string; mode: string; level: number }[] {
|
|
290
|
+
const out: { sample: string; mode: string; level: number }[] = [];
|
|
291
|
+
for (const s of report.samples) {
|
|
292
|
+
for (const r of s.rows) {
|
|
293
|
+
if (!r.roundTrip && !r.note) out.push({ sample: s.sample, mode: r.mode, level: r.level });
|
|
294
|
+
}
|
|
295
|
+
}
|
|
296
|
+
return out;
|
|
297
|
+
}
|