Compare commits

...
7 Commits
36 changed files with 945 additions and 214 deletions
+14 -8
View File
@@ -3,10 +3,12 @@
<script src="../dist/bundles/mediabunny.cjs"></script>
<script src="../packages/mp3-encoder/dist/bundles/mediabunny-mp3-encoder.js"></script>
<script src="../packages/ac3/dist/bundles/mediabunny-ac3.js"></script>
<script src="../packages/flac-encoder/dist/bundles/mediabunny-flac-encoder.js"></script>
<script type="module">
//MediabunnyMp3Encoder.registerMp3Encoder();
MediabunnyAc3.registerAc3Decoder();
MediabunnyFlacEncoder.registerFlacEncoder();
const fileInput = document.createElement('input');
fileInput.type = 'file';
@@ -23,7 +25,7 @@
chunked: true,
chunkSize: 2**20
});
const outputFormat = new Mediabunny.WavOutputFormat();
const outputFormat = new Mediabunny.FlacOutputFormat();
const p = document.createElement('p');
p.textContent = 'Capturing...';
@@ -57,7 +59,7 @@
const tracks = [];
let start = 0;
if (true) {
if (false) {
input = new Mediabunny.Input({
source: new Mediabunny.UrlSource('http://localhost:8000/index.m3u8'),
formats: Mediabunny.ALL_FORMATS,
@@ -85,15 +87,18 @@
});
}
const primaryTrack = await input.getPrimaryAudioTrack();
const startTime = await primaryTrack.getFirstTimestamp();
console.log(startTime)
//const primaryTrack = await input.getPrimaryAudioTrack();
//const startTime = await primaryTrack.getFirstTimestamp();
//console.log(startTime)
let ctx = null;
let conversion = await Mediabunny.Conversion.init({
input,
output,
audio: (track) => ({ discard: track.number !== primaryTrack.number }),
audio: {
//forceTranscode: true,
//sampleFormat: 's16',
},
/*
video: {
discard: true,
@@ -139,11 +144,12 @@
}
},
trim: {
start: startTime,
end: startTime + 2,
//start: startTime,
//end: startTime + 2,
},
});
//console.log(conversion);
console.log(conversion.discardedTracks);
let progress = 0;
conversion.onProgress = newProgress => progress = newProgress;
+34 -6
View File
@@ -11,7 +11,6 @@
document.body.append(fileInput);
fileInput.addEventListener('change', async () => {
/*
const file = fileInput.files[0];
const input = new Mediabunny.Input({
formats: Mediabunny.ALL_FORMATS,
@@ -19,15 +18,43 @@
});
const track = await input.getPrimaryAudioTrack();
const sink3 = new Mediabunny.EncodedPacketSink(track);
console.log((await sink3.getFirstPacket()).data.join(', '));
return;
console.log(await track.getDurationFromMetadata(), await track.computeDuration());
const sink = new Mediabunny.EncodedPacketSink(track);
for await (const packet of sink.packets()) {
console.log(packet);
}
const output = new Mediabunny.Output({
format: new Mediabunny.Mp4OutputFormat(),
target: new Mediabunny.BufferTarget(),
});
const conversion = await Mediabunny.Conversion.init({ input, output });
await conversion.execute();
//console.log(await input.getDurationFromMetadata(), await input.computeDuration());
return;
// Download it now
const blob = new Blob([output.target.buffer]);
const url = URL.createObjectURL(blob);
const a = document.createElement('a');
a.href = url;
a.download = file.name.replace(/\.\w+$/, '.mp4');
a.click();
URL.revokeObjectURL(url);
/*
const track = await input.getPrimaryAudioTrack();
const sink = new Mediabunny.EncodedPacketSink(track);
for await (const packet of sink.packets()) {
console.log(packet);
}
console.log("Done")
*/
/*
const input = new Mediabunny.Input({
source: new Mediabunny.UrlSource('https://storage.googleapis.com/shaka-demo-assets/angel-one-widevine-hls/hls.m3u8'),
formats: Mediabunny.ALL_FORMATS,
@@ -36,6 +63,7 @@
const videoTrack = await input.getPrimaryVideoTrack();
const sink = new Mediabunny.EncodedPacketSink(videoTrack);
console.log(await sink.getFirstPacket());
*/
/*
return
+4
View File
@@ -196,10 +196,14 @@ export default withMermaid({
if (title !== 'Mediabunny') {
title += ' | Mediabunny';
}
const canonicalUrl = `https://mediabunny.dev/${pageData.relativePath}`
.replace(/index\.md$/, '')
.replace(/\.md$/, '');
((pageData.frontmatter['head'] ??= []) as HeadConfig[]).push(
['meta', { property: 'og:title', content: title }],
['meta', { property: 'twitter:title', content: title }],
['link', { rel: 'canonical', href: canonicalUrl }],
);
},
});
+3 -3
View File
@@ -77,7 +77,7 @@ The API surface added by the HLS update is vast and I obviously can't cover it i
By using the Conversion API, you can just do this:
<div class="text-xs">
<div class="text-[13.7142857143px]">
```ts
import { ... } from 'mediabunny';
@@ -107,7 +107,7 @@ That's it. This will stream-download the entire HLS playlist, transcode it if ne
This is basically the inverse of the previous example. Just like we're able to read HLS and turn it into an MP4, we're able to read any input file and turn it into a full HLS playlist including master playlist, media playlists and segments:
<div class="text-xs">
<div class="text-[13.7142857143px]">
```ts
import { ... } from 'mediabunny';
@@ -163,7 +163,7 @@ No transcode server is needed here, it's all handled by the client, and the serv
You could build an OBS-like broadcasting system where a user records their screen, facecam or microphone, encodes multiple variants locally, and then broadcasts finished HLS segments directly to the server, meaning no transcoding is needed.
<div class="text-xs overflow-auto">
<div class="text-[13.7142857143px] overflow-auto">
```ts
// Get the screen and mic
+1
View File
@@ -242,6 +242,7 @@ type ConversionAudioOptions = {
bitrate?: number | Quality;
numberOfChannels?: number;
sampleRate?: number;
sampleFormat?: 'u8' | 's16' | 's32' | 'f32';
forceTranscode?: boolean;
process?: (sample: AudioSample) => MaybePromise<
AudioSample | AudioSample[] | null
+1
View File
@@ -32,6 +32,7 @@ export default tseslint.config(
'@typescript-eslint/no-unsafe-enum-comparison': 'off',
'@typescript-eslint/no-unsafe-unary-minus': 'off',
'@typescript-eslint/no-deprecated': 'error',
'@typescript-eslint/consistent-type-exports': 'error',
},
},
{
+1
View File
@@ -9,6 +9,7 @@
<script type="module" src="./file-compression.ts"></script>
<link rel="stylesheet" href="../base.css">
<link rel="icon" href="../../docs/public/mediabunny-logo.svg">
<link rel="canonical" href="https://mediabunny.dev/examples/file-compression/">
</head>
<body class="flex flex-col items-center py-10 bg-zinc-50 text-zinc-800 dark:bg-zinc-900 dark:text-zinc-200 px-2">
+1
View File
@@ -9,6 +9,7 @@
<script type="module" src="./hls-transcoding.ts"></script>
<link rel="stylesheet" href="../base.css">
<link rel="icon" href="../../docs/public/mediabunny-logo.svg">
<link rel="canonical" href="https://mediabunny.dev/examples/hls-transcoding/">
</head>
<body class="flex flex-col items-center bg-zinc-50 px-2 py-10 text-zinc-800 dark:bg-zinc-900 dark:text-zinc-200">
+1
View File
@@ -9,6 +9,7 @@
<script type="module" src="./live-recording.ts"></script>
<link rel="stylesheet" href="../base.css">
<link rel="icon" href="../../docs/public/mediabunny-logo.svg">
<link rel="canonical" href="https://mediabunny.dev/examples/live-recording/">
</head>
<body class="flex flex-col items-center py-10 bg-zinc-50 text-zinc-800 dark:bg-zinc-900 dark:text-zinc-200 px-2">
+1
View File
@@ -9,6 +9,7 @@
<script type="module" src="./media-player.ts"></script>
<link rel="stylesheet" href="../base.css">
<link rel="icon" href="../../docs/public/mediabunny-logo.svg">
<link rel="canonical" href="https://mediabunny.dev/examples/media-player/">
</head>
<body class="flex flex-col items-center py-10 bg-zinc-50 text-zinc-800 dark:bg-zinc-900 dark:text-zinc-200 px-2 h-svh">
+1
View File
@@ -9,6 +9,7 @@
<script type="module" src="./metadata-extraction.ts"></script>
<link rel="stylesheet" href="../base.css">
<link rel="icon" href="../../docs/public/mediabunny-logo.svg">
<link rel="canonical" href="https://mediabunny.dev/examples/metadata-extraction/">
</head>
<body class="flex flex-col items-center py-10 bg-zinc-50 text-zinc-800 dark:bg-zinc-900 dark:text-zinc-200 px-2">
@@ -9,6 +9,7 @@
<script type="module" src="./procedural-generation.ts"></script>
<link rel="stylesheet" href="../base.css">
<link rel="icon" href="../../docs/public/mediabunny-logo.svg">
<link rel="canonical" href="https://mediabunny.dev/examples/procedural-generation/">
</head>
<body class="flex flex-col items-center py-10 bg-zinc-50 text-zinc-800 dark:bg-zinc-900 dark:text-zinc-200 px-2">
+1
View File
@@ -9,6 +9,7 @@
<script type="module" src="./thumbnail-generation.ts"></script>
<link rel="stylesheet" href="../base.css">
<link rel="icon" href="../../docs/public/mediabunny-logo.svg">
<link rel="canonical" href="https://mediabunny.dev/examples/thumbnail-generation/">
</head>
<body class="flex flex-col items-center py-10 bg-gray-50 text-gray-800 dark:bg-zinc-900 dark:text-zinc-200 px-2">
+9 -9
View File
@@ -1,12 +1,12 @@
{
"name": "mediabunny",
"version": "1.42.0",
"version": "1.43.0",
"lockfileVersion": 3,
"requires": true,
"packages": {
"": {
"name": "mediabunny",
"version": "1.42.0",
"version": "1.43.0",
"license": "MPL-2.0",
"workspaces": [
"packages/*"
@@ -7751,9 +7751,9 @@
}
},
"node_modules/mediabunny": {
"version": "1.41.0",
"resolved": "https://registry.npmjs.org/mediabunny/-/mediabunny-1.41.0.tgz",
"integrity": "sha512-cJWBHvAyRNgTbsx8Z2oHptX/PdiywwsS+GIvv2k9eioXSOzzAemBsHwf7jM6fvW3b65azvZGP9RLuO1DI8JmdA==",
"version": "1.42.0",
"resolved": "https://registry.npmjs.org/mediabunny/-/mediabunny-1.42.0.tgz",
"integrity": "sha512-s9ypTqLi6kbh95gC+YaJlG0PkLvMxu37Q/wO/pFZx0fUCA5Ym5mp+2dWoa83mKQ3Uo18aNlgev5iJ5ESZqWwgQ==",
"license": "MPL-2.0",
"peer": true,
"workspaces": [
@@ -12077,7 +12077,7 @@
},
"packages/aac-encoder": {
"name": "@mediabunny/aac-encoder",
"version": "1.42.0",
"version": "1.43.0",
"license": "MPL-2.0",
"devDependencies": {
"@types/emscripten": "^1.40.1"
@@ -12092,7 +12092,7 @@
},
"packages/ac3": {
"name": "@mediabunny/ac3",
"version": "1.42.0",
"version": "1.43.0",
"license": "MPL-2.0",
"devDependencies": {
"@types/emscripten": "^1.40.1"
@@ -12107,7 +12107,7 @@
},
"packages/flac-encoder": {
"name": "@mediabunny/flac-encoder",
"version": "1.42.0",
"version": "1.43.0",
"license": "MPL-2.0",
"devDependencies": {
"@types/emscripten": "^1.40.1"
@@ -12122,7 +12122,7 @@
},
"packages/mp3-encoder": {
"name": "@mediabunny/mp3-encoder",
"version": "1.42.0",
"version": "1.43.0",
"license": "MPL-2.0",
"devDependencies": {
"@types/emscripten": "^1.40.1"
+1 -1
View File
@@ -1,7 +1,7 @@
{
"name": "mediabunny",
"author": "Vanilagy",
"version": "1.42.0",
"version": "1.43.0",
"description": "Pure TypeScript media toolkit for reading, writing, and converting media files, directly in the browser.",
"type": "module",
"workspaces": [
+1 -1
View File
@@ -1,7 +1,7 @@
{
"name": "@mediabunny/aac-encoder",
"author": "Vanilagy",
"version": "1.42.0",
"version": "1.43.0",
"description": "AAC encoder extension for Mediabunny, based on FFmpeg.",
"main": "./dist/bundles/mediabunny-aac-encoder.mjs",
"module": "./dist/bundles/mediabunny-aac-encoder.mjs",
+1 -1
View File
@@ -1,7 +1,7 @@
{
"name": "@mediabunny/ac3",
"author": "Vanilagy",
"version": "1.42.0",
"version": "1.43.0",
"description": "AC-3 and E-AC-3 (Dolby Digital) decoder and encoder extension for Mediabunny, based on FFmpeg.",
"main": "./dist/bundles/mediabunny-ac3.mjs",
"module": "./dist/bundles/mediabunny-ac3.mjs",
Binary file not shown.
+1 -1
View File
@@ -1,7 +1,7 @@
{
"name": "@mediabunny/flac-encoder",
"author": "Vanilagy",
"version": "1.42.0",
"version": "1.43.0",
"description": "FLAC encoder extension for Mediabunny, based on libFLAC.",
"main": "./dist/bundles/mediabunny-flac-encoder.mjs",
"module": "./dist/bundles/mediabunny-flac-encoder.mjs",
+9 -16
View File
@@ -12,7 +12,6 @@
#include <stdlib.h>
#include <string.h>
#define BITS_PER_SAMPLE 16
#define COMPRESSION_LEVEL 5
typedef struct {
@@ -23,14 +22,10 @@ typedef struct {
typedef struct {
FLAC__StreamEncoder *encoder;
// Input buffer for interleaved int16 samples from JS
int16_t *input_buffer;
// Input buffer for interleaved int32 samples from JS
FLAC__int32 *input_buffer;
int input_buffer_size;
// Widened to int32 for libFLAC
FLAC__int32 *int32_buffer;
int int32_buffer_size;
// Contiguous output buffer for encoded frame data
uint8_t *output_buffer;
int output_size;
@@ -48,6 +43,7 @@ typedef struct {
bool header_done;
int channels;
int bits_per_sample;
} EncoderContext;
static void ensure_output_capacity(EncoderContext *ctx, int needed) {
@@ -120,13 +116,14 @@ static void reset_output(EncoderContext *ctx) {
}
EMSCRIPTEN_KEEPALIVE
int init_encoder(int channels, int sample_rate) {
int init_encoder(int channels, int sample_rate, int bits_per_sample) {
EncoderContext *ctx = calloc(1, sizeof(EncoderContext));
if (!ctx) {
return 0;
}
ctx->channels = channels;
ctx->bits_per_sample = bits_per_sample;
ctx->encoder = FLAC__stream_encoder_new();
if (!ctx->encoder) {
@@ -136,7 +133,7 @@ int init_encoder(int channels, int sample_rate) {
FLAC__stream_encoder_set_channels(ctx->encoder, channels);
FLAC__stream_encoder_set_sample_rate(ctx->encoder, sample_rate);
FLAC__stream_encoder_set_bits_per_sample(ctx->encoder, BITS_PER_SAMPLE);
FLAC__stream_encoder_set_bits_per_sample(ctx->encoder, bits_per_sample);
FLAC__stream_encoder_set_compression_level(ctx->encoder, COMPRESSION_LEVEL);
FLAC__stream_encoder_set_verify(ctx->encoder, false);
@@ -174,19 +171,15 @@ EMSCRIPTEN_KEEPALIVE
int send_samples(int ctx_ptr, int num_samples) {
EncoderContext *ctx = (EncoderContext *)ctx_ptr;
// Widen int16 to int32 for libFLAC
int total = num_samples * ctx->channels;
if (total > ctx->int32_buffer_size) {
ctx->int32_buffer = realloc(ctx->int32_buffer, total * sizeof(FLAC__int32));
ctx->int32_buffer_size = total;
}
int shift = 32 - ctx->bits_per_sample;
for (int i = 0; i < total; i++) {
ctx->int32_buffer[i] = ctx->input_buffer[i];
ctx->input_buffer[i] >>= shift;
}
reset_output(ctx);
FLAC__bool ok = FLAC__stream_encoder_process_interleaved(ctx->encoder, ctx->int32_buffer, num_samples);
FLAC__bool ok = FLAC__stream_encoder_process_interleaved(ctx->encoder, ctx->input_buffer, num_samples);
return ok ? 0 : -1;
}
+5 -4
View File
@@ -16,7 +16,7 @@ type ExtendedEmscriptenModule = EmscriptenModule & {
let module: ExtendedEmscriptenModule;
let modulePromise: Promise<ExtendedEmscriptenModule> | null = null;
let initEncoderFn: (channels: number, sampleRate: number) => number;
let initEncoderFn: (channels: number, sampleRate: number, bitsPerSample: number) => number;
let getEncodeInputPtr: (ctx: number, size: number) => number;
let sendSamplesFn: (ctx: number, numSamples: number) => number;
let getOutputData: (ctx: number) => number;
@@ -37,7 +37,7 @@ const ensureModule = async () => {
module = await modulePromise;
modulePromise = null;
initEncoderFn = module.cwrap('init_encoder', 'number', ['number', 'number']);
initEncoderFn = module.cwrap('init_encoder', 'number', ['number', 'number', 'number']);
getEncodeInputPtr = module.cwrap('get_encode_input_ptr', 'number', ['number', 'number']);
sendSamplesFn = module.cwrap('send_samples', 'number', ['number', 'number']);
getOutputData = module.cwrap('get_output_data', 'number', ['number']);
@@ -50,10 +50,10 @@ const ensureModule = async () => {
}
};
const initEncoder = async (numberOfChannels: number, sampleRate: number) => {
const initEncoder = async (numberOfChannels: number, sampleRate: number, bitsPerSample: 16 | 24) => {
await ensureModule();
const ctx = initEncoderFn(numberOfChannels, sampleRate);
const ctx = initEncoderFn(numberOfChannels, sampleRate, bitsPerSample);
if (ctx === 0) {
throw new Error('Failed to initialize FLAC encoder.');
}
@@ -121,6 +121,7 @@ const onMessage = (data: { id: number; command: WorkerCommand }) => {
const { ctx, header } = await initEncoder(
command.data.numberOfChannels,
command.data.sampleRate,
command.data.bitsPerSample,
);
result = { type: command.type, ctx, header };
transferables.push(header);
+51 -19
View File
@@ -29,7 +29,7 @@ class FlacEncoder extends CustomAudioEncoder {
reject: (reason?: unknown) => void;
}>();
private ctx = 0;
private ctx: number | null = null;
private chunkMetadata: EncodedAudioChunkMetadata = {};
private description: Uint8Array | null = null;
private nextTimestampInSamples: number | null = null;
@@ -65,19 +65,6 @@ class FlacEncoder extends CustomAudioEncoder {
};
nodeWorker.on('message', onMessage);
}
const result = await this.sendCommand({
type: 'init',
data: {
numberOfChannels: this.config.numberOfChannels,
sampleRate: this.config.sampleRate,
},
});
this.ctx = result.ctx;
this.description = new Uint8Array(result.header);
this.resetInternalState();
}
private resetInternalState() {
@@ -94,15 +81,50 @@ class FlacEncoder extends CustomAudioEncoder {
}
async encode(audioSample: AudioSample) {
if (this.ctx === null) {
// This is the first sample, let's do some init
let bitsPerSample: 16 | 24;
switch (audioSample.format) {
case 'u8':
case 'u8-planar':
case 's16':
case 's16-planar':
bitsPerSample = 16;
break;
case 's32':
case 's32-planar':
case 'f32':
case 'f32-planar':
bitsPerSample = 24;
break;
default:
assertNever(audioSample.format);
assert(false);
}
const result = await this.sendCommand({
type: 'init',
data: {
numberOfChannels: this.config.numberOfChannels,
sampleRate: this.config.sampleRate,
bitsPerSample,
},
});
this.ctx = result.ctx;
this.description = new Uint8Array(result.header);
this.resetInternalState();
}
if (this.nextTimestampInSamples === null) {
this.nextTimestampInSamples = Math.round(audioSample.timestamp * this.config.sampleRate);
}
const totalBytes = audioSample.allocationSize({ format: 's16', planeIndex: 0 });
const audioBytes = new Uint8Array(totalBytes);
audioSample.copyTo(audioBytes, { format: 's16', planeIndex: 0 });
const totalBytes = audioSample.allocationSize({ format: 's32', planeIndex: 0 });
const audioData = new ArrayBuffer(totalBytes);
audioSample.copyTo(audioData, { format: 's32', planeIndex: 0 });
const audioData = audioBytes.buffer;
const result = await this.sendCommand({
type: 'encode',
data: {
@@ -116,6 +138,10 @@ class FlacEncoder extends CustomAudioEncoder {
}
async flush() {
if (this.ctx === null) {
return;
}
const result = await this.sendCommand({ type: 'flush', data: { ctx: this.ctx } });
this.emitPackets(result.packets);
@@ -173,7 +199,8 @@ class FlacEncoder extends CustomAudioEncoder {
/**
* Registers the FLAC encoder, which Mediabunny will then use automatically when applicable. Make sure to call this
* function before starting any encoding task.
* function before starting any encoding task. The FLAC encoder will automatically determine the output bit depth
* (16 or 24) based on the sample format of incoming `AudioSample` instances.
*
* Preferably, wrap the call in a condition to avoid overriding any native FLAC encoder:
*
@@ -198,3 +225,8 @@ function assert(x: unknown): asserts x {
throw new Error('Assertion failed.');
}
}
export const assertNever = (x: never) => {
// eslint-disable-next-line @typescript-eslint/restrict-template-expressions
throw new Error(`Unexpected value: ${x}`);
};
+1
View File
@@ -16,6 +16,7 @@ export type WorkerCommand = {
data: {
numberOfChannels: number;
sampleRate: number;
bitsPerSample: 16 | 24;
};
} | {
type: 'encode';
+1 -1
View File
@@ -1,7 +1,7 @@
{
"name": "@mediabunny/mp3-encoder",
"author": "Vanilagy",
"version": "1.42.0",
"version": "1.43.0",
"description": "MP3 encoder extension for Mediabunny, based on LAME.",
"main": "./dist/bundles/mediabunny-mp3-encoder.mjs",
"module": "./dist/bundles/mediabunny-mp3-encoder.mjs",
+112
View File
@@ -866,6 +866,21 @@ export type HevcSpsInfo = {
minSpatialSegmentationIdc: number;
};
export const concatHevcNalUnits = (nalUnits: Uint8Array[], decoderConfig: VideoDecoderConfig) => {
if (decoderConfig.description) {
// Stream is length-prefixed. Let's extract the size of the length prefix from the decoder config
const bytes = toUint8Array(decoderConfig.description);
const lengthSizeMinusOne = bytes[21]! & 0b11;
const lengthSize = (lengthSizeMinusOne + 1) as 1 | 2 | 3 | 4;
return concatNalUnitsInLengthPrefixed(nalUnits, lengthSize);
} else {
// Stream is in Annex B format
return concatNalUnitsInAnnexB(nalUnits);
}
};
export const iterateHevcNalUnits = (packetData: Uint8Array, decoderConfig: VideoDecoderConfig) => {
if (decoderConfig.description) {
const bytes = toUint8Array(decoderConfig.description);
@@ -1602,6 +1617,103 @@ export const deserializeHevcDecoderConfigurationRecord = (data: Uint8Array): Hev
}
};
enum HevcNaluOrderState {
audAllowed,
beforeFirstVcl,
afterFirstVcl,
eoBitstreamAllowed,
noMoreDataAllowed,
}
// This function sanitzes the contents of an HEVC packet such that
// https://source.chromium.org/chromium/chromium/src/+/main:media/formats/mp4/hevc.cc's validation logic does not trip
// up on its contents. The validation is often too strict and rejects packets that Chromium could decode just fine.
// Chromium code retrieved on 2026-04-29.
// See https://issues.chromium.org/issues/507611247.
export const sanitizeHevcPacketForChromium = (
packetData: Uint8Array,
decoderConfig: VideoDecoderConfig,
): Uint8Array | null => {
const removedNalUnits = new Set<number>();
let orderState: HevcNaluOrderState = HevcNaluOrderState.audAllowed;
for (const loc of iterateHevcNalUnits(packetData, decoderConfig)) {
if (orderState === HevcNaluOrderState.noMoreDataAllowed) {
removedNalUnits.add(loc.offset);
continue;
}
const type = extractNalUnitTypeForHevc(packetData[loc.offset]!);
if (orderState === HevcNaluOrderState.eoBitstreamAllowed && type !== 37 /* EOB_NUT */) {
removedNalUnits.add(loc.offset);
continue;
}
let remove = false;
if (type === 35) { // AUD_NUT
if (orderState > HevcNaluOrderState.audAllowed) {
remove = true;
} else {
orderState = HevcNaluOrderState.beforeFirstVcl;
}
} else if (type <= 31) { // VCL (0-31)
if (orderState > HevcNaluOrderState.afterFirstVcl) {
remove = true;
} else {
orderState = HevcNaluOrderState.afterFirstVcl;
}
} else if (type === 36) { // EOS_NUT
if (orderState !== HevcNaluOrderState.afterFirstVcl) {
remove = true;
} else {
orderState = HevcNaluOrderState.eoBitstreamAllowed;
}
} else if (type === 37) { // EOB_NUT
if (orderState < HevcNaluOrderState.afterFirstVcl) {
remove = true;
} else {
orderState = HevcNaluOrderState.noMoreDataAllowed;
}
} else if (
type === 32 || type === 33 || type === 34 || type === 39
|| (type >= 41 && type <= 44) || (type >= 48 && type <= 55)
) { // VPS, SPS, PPS, PREFIX_SEI, RSV_NVCL41..44, UNSPEC48..55
if (orderState > HevcNaluOrderState.beforeFirstVcl) {
remove = true;
} else {
orderState = HevcNaluOrderState.beforeFirstVcl;
}
} else if (
type === 38 || type === 40
|| (type >= 45 && type <= 47) || (type >= 56 && type <= 63)
) { // FD, SUFFIX_SEI, RSV_NVCL45..47, UNSPEC56..63
if (orderState < HevcNaluOrderState.afterFirstVcl) {
remove = true;
}
}
if (remove) {
removedNalUnits.add(loc.offset);
}
}
// If nothing violated the rules, return null to signal that
if (removedNalUnits.size === 0) {
return null;
}
const filteredNalUnits: Uint8Array[] = [];
for (const loc of iterateHevcNalUnits(packetData, decoderConfig)) {
if (!removedNalUnits.has(loc.offset)) {
filteredNalUnits.push(packetData.subarray(loc.offset, loc.offset + loc.length));
}
}
return concatHevcNalUnits(filteredNalUnits, decoderConfig);
};
export type Vp9CodecInfo = {
profile: number;
level: number;
+45 -11
View File
@@ -50,7 +50,15 @@ import {
} from './misc';
import { Output, OutputTrackGroup, TrackType } from './output';
import { Mp4OutputFormat } from './output-format';
import { AudioSample, clampCropRectangle, CropRectangle, validateCropRectangle, VideoSample } from './sample';
import {
AudioSample,
audioSampleToInterleavedFormat,
clampCropRectangle,
CropRectangle,
toInterleavedAudioFormat,
validateCropRectangle,
VideoSample,
} from './sample';
import { MetadataTags, validateMetadataTags } from './metadata';
import { NullTarget } from './target';
import { AudioResampler } from './resample';
@@ -261,6 +269,13 @@ export type ConversionAudioOptions = {
numberOfChannels?: number;
/** The desired sample rate of the output audio, in hertz. */
sampleRate?: number;
/**
* The desired sample format (and therefore bit depth) of the audio samples before they are passed to the encoder.
* Can be used to control bit depth with certain output codecs such as FLAC.
*
* Setting this field forces audio transcoding.
*/
sampleFormat?: 'u8' | 's16' | 's32' | 'f32';
/** The desired output audio codec. */
codec?: AudioCodec;
/** The desired bitrate of the output audio. */
@@ -442,6 +457,12 @@ const validateAudioOptions = (audioOptions: ConversionAudioOptions) => {
) {
throw new TypeError('options.audio.sampleRate, when provided, must be a positive integer.');
}
if (
audioOptions?.sampleFormat !== undefined
&& !['u8', 's16', 's32', 'f32'].includes(audioOptions.sampleFormat)
) {
throw new TypeError('options.audio.sampleFormat, when provided, must be one of: u8, s16, s32, f32.');
}
if (audioOptions?.process !== undefined && typeof audioOptions.process !== 'function') {
throw new TypeError('options.audio.process, when provided, must be a function.');
}
@@ -1347,7 +1368,7 @@ export class Conversion {
timestamp: lastCanvasTimestamp! + i / frameRate,
duration: 1 / frameRate,
});
await this._registerVideoSample(track, trackOptions, outputTrackId, source, sample);
await this._registerVideoSample(trackOptions, outputTrackId, source, sample);
sample.close();
}
};
@@ -1384,7 +1405,7 @@ export class Conversion {
timestamp: adjustedSampleTimestamp,
duration: frameRate !== undefined ? 1 / frameRate : duration,
});
await this._registerVideoSample(track, trackOptions, outputTrackId, source, sample);
await this._registerVideoSample(trackOptions, outputTrackId, source, sample);
sample.close();
if (frameRate !== undefined) {
@@ -1425,7 +1446,7 @@ export class Conversion {
for (let i = 1; i < frameDifference; i++) {
lastSample.setTimestamp(lastSampleTimestamp! + i / frameRate);
lastSample.setDuration(1 / frameRate);
await this._registerVideoSample(track, trackOptions, outputTrackId, source, lastSample);
await this._registerVideoSample(trackOptions, outputTrackId, source, lastSample);
}
lastSample.close();
@@ -1464,7 +1485,7 @@ export class Conversion {
}
sample.setTimestamp(adjustedSampleTimestamp);
await this._registerVideoSample(track, trackOptions, outputTrackId, source, sample);
await this._registerVideoSample(trackOptions, outputTrackId, source, sample);
if (frameRate !== undefined) {
lastSample = sample;
@@ -1513,7 +1534,6 @@ export class Conversion {
/** @internal */
async _registerVideoSample(
track: InputVideoTrack,
trackOptions: ConversionVideoOptions,
outputTrackId: number,
source: VideoSampleSource,
@@ -1609,6 +1629,7 @@ export class Conversion {
&& audioCodecs.includes(sourceCodec)
&& (!trackOptions.codec || trackOptions.codec === sourceCodec)
&& !trackOptions.process
&& trackOptions.sampleFormat === undefined
) {
// Fast path, we can simply copy over the encoded packets
@@ -1745,7 +1766,7 @@ export class Conversion {
// Offset the timestamp as needed
sample.setTimestamp(sample.timestamp - this._startTimestamp);
await this._registerAudioSample(track, trackOptions, outputTrackId, source, sample);
await this._registerAudioSample(trackOptions, outputTrackId, source, sample);
sample.close();
}
@@ -1778,16 +1799,25 @@ export class Conversion {
/** @internal */
async _registerAudioSample(
track: InputAudioTrack,
trackOptions: ConversionAudioOptions,
outputTrackId: number,
source: AudioSampleSource,
sample: AudioSample,
inputSample: AudioSample,
) {
if (this._canceled) {
return;
}
let sample = inputSample;
if (
trackOptions.sampleFormat !== undefined
&& toInterleavedAudioFormat(sample.format) !== trackOptions.sampleFormat
) {
// Do a sample format conversion
sample = audioSampleToInterleavedFormat(sample, trackOptions.sampleFormat);
}
this._reportProgress(outputTrackId, sample.timestamp + sample.duration);
let finalSamples: AudioSample[];
@@ -1823,8 +1853,12 @@ export class Conversion {
}
}
} finally {
if (sample !== inputSample) {
sample.close();
}
for (const finalSample of finalSamples) {
if (finalSample !== sample) {
if (finalSample !== inputSample) {
finalSample.close();
}
}
@@ -1857,7 +1891,7 @@ export class Conversion {
onSample: async (sample) => {
sample.setTimestamp(sample.timestamp - this._startTimestamp);
await this._registerAudioSample(track, trackOptions, outputTrackId, source, sample);
await this._registerAudioSample(trackOptions, outputTrackId, source, sample);
sample.close();
},
});
+12 -1
View File
@@ -386,6 +386,11 @@ export type AudioTransformOptions = {
numberOfChannels?: number;
/** The desired output sample rate in hertz to resample to. */
sampleRate?: number;
/**
* The desired sample format (and therefore bit depth) of the audio samples before they are passed to the encoder.
* Can be used to control bit depth with certain output codecs such as FLAC.
*/
sampleFormat?: 'u8' | 's16' | 's32' | 'f32';
/**
* Allows for custom user-defined processing of audio samples, e.g. for applying audio effects or timestamp
* modifications. Called for each audio sample after resampling and remixing.
@@ -406,7 +411,7 @@ export const validateAudioEncodingConfig = (config: AudioEncodingConfig) => {
}
if (
config.bitrate === undefined
&& (!(PCM_AUDIO_CODECS as readonly string[]).includes(config.codec) || config.codec === 'flac')
&& !((PCM_AUDIO_CODECS as readonly string[]).includes(config.codec) || config.codec === 'flac')
) {
throw new TypeError('config.bitrate must be provided for compressed audio codecs.');
}
@@ -433,6 +438,12 @@ export const validateAudioEncodingConfig = (config: AudioEncodingConfig) => {
) {
throw new TypeError('config.transform.sampleRate, when provided, must be a positive integer.');
}
if (
config.transform.sampleFormat !== undefined
&& !['u8', 's16', 's32', 'f32'].includes(config.transform.sampleFormat)
) {
throw new TypeError('config.transform.sampleFormat, when provided, must be one of: u8, s16, s32, f32.');
}
if (config.transform.process !== undefined && typeof config.transform.process !== 'function') {
throw new TypeError('config.transform.process, when provided, must be a function.');
}
+82 -82
View File
@@ -22,48 +22,48 @@ if ((globalThis as Record<symbol, unknown>)[MEDIABUNNY_LOADED_SYMBOL]) {
export {
Output,
OutputOptions,
type OutputOptions,
OutputTrack,
OutputVideoTrack,
OutputAudioTrack,
OutputSubtitleTrack,
OutputTrackGroup,
BaseTrackMetadata,
VideoTrackMetadata,
AudioTrackMetadata,
SubtitleTrackMetadata,
OutputEvents,
type BaseTrackMetadata,
type VideoTrackMetadata,
type AudioTrackMetadata,
type SubtitleTrackMetadata,
type OutputEvents,
} from './output';
export {
OutputFormat,
AdtsOutputFormat,
AdtsOutputFormatOptions,
type AdtsOutputFormatOptions,
CmafOutputFormat,
CmafOutputFormatOptions,
type CmafOutputFormatOptions,
FlacOutputFormat,
FlacOutputFormatOptions,
type FlacOutputFormatOptions,
HlsOutputFormat,
HlsOutputFormatOptions,
HlsOutputPlaylistInfo,
HlsOutputSegmentInfo,
type HlsOutputFormatOptions,
type HlsOutputPlaylistInfo,
type HlsOutputSegmentInfo,
IsobmffOutputFormat,
IsobmffOutputFormatOptions,
type IsobmffOutputFormatOptions,
MkvOutputFormat,
MkvOutputFormatOptions,
type MkvOutputFormatOptions,
MovOutputFormat,
Mp3OutputFormat,
Mp3OutputFormatOptions,
type Mp3OutputFormatOptions,
Mp4OutputFormat,
MpegTsOutputFormat,
MpegTsOutputFormatOptions,
type MpegTsOutputFormatOptions,
OggOutputFormat,
OggOutputFormatOptions,
type OggOutputFormatOptions,
WavOutputFormat,
WavOutputFormatOptions,
type WavOutputFormatOptions,
WebMOutputFormat,
WebMOutputFormatOptions,
InclusiveIntegerRange,
TrackCountLimits,
type WebMOutputFormatOptions,
type InclusiveIntegerRange,
type TrackCountLimits,
} from './output-format';
export {
MediaSource,
@@ -76,17 +76,17 @@ export {
EncodedAudioPacketSource,
EncodedVideoPacketSource,
MediaStreamAudioTrackSource,
MediaStreamAudioTrackSourceOptions,
type MediaStreamAudioTrackSourceOptions,
MediaStreamVideoTrackSource,
MediaStreamVideoTrackSourceOptions,
type MediaStreamVideoTrackSourceOptions,
TextSubtitleSource,
VideoSampleSource,
} from './media-source';
export {
MediaCodec,
VideoCodec,
AudioCodec,
SubtitleCodec,
type MediaCodec,
type VideoCodec,
type AudioCodec,
type SubtitleCodec,
VIDEO_CODECS,
AUDIO_CODECS,
PCM_AUDIO_CODECS,
@@ -102,12 +102,12 @@ export {
getDecodableAudioCodecs,
} from './decode';
export {
VideoEncodingConfig,
VideoEncodingAdditionalOptions,
VideoTransformOptions,
AudioEncodingConfig,
AudioEncodingAdditionalOptions,
AudioTransformOptions,
type VideoEncodingConfig,
type VideoEncodingAdditionalOptions,
type VideoTransformOptions,
type AudioEncodingConfig,
type AudioEncodingAdditionalOptions,
type AudioTransformOptions,
canEncode,
canEncodeVideo,
canEncodeAudio,
@@ -128,69 +128,69 @@ export {
} from './encode';
export {
Target,
TargetEvents,
TargetRequest,
type TargetEvents,
type TargetRequest,
AppendOnlyStreamTarget,
BufferTarget,
BufferTargetOptions,
type BufferTargetOptions,
FilePathTarget,
FilePathTargetOptions,
type FilePathTargetOptions,
NullTarget,
PathedTarget,
RangedTarget,
StreamTarget,
StreamTargetOptions,
StreamTargetChunk,
type StreamTargetOptions,
type StreamTargetChunk,
} from './target';
export {
AnyIterable,
type AnyIterable,
ConcurrentRunner,
EventEmitter,
EventListenerOptions,
FilePath,
MaybePromise,
type EventListenerOptions,
type FilePath,
type MaybePromise,
} from './misc';
export {
PsshBox,
type PsshBox,
} from './isobmff/isobmff-misc';
export {
Rational,
Rectangle,
Rotation,
SetOptional,
SetRequired,
type Rational,
type Rectangle,
type Rotation,
type SetOptional,
type SetRequired,
} from './misc';
export {
TrackType,
type TrackType,
ALL_TRACK_TYPES,
} from './output';
export {
Source,
SourceEvents,
type SourceEvents,
SourceRef,
SourceRequest,
type SourceRequest,
BlobSource,
BlobSourceOptions,
type BlobSourceOptions,
BufferSource,
CustomPathedSource,
FilePathSource,
FilePathSourceOptions,
type FilePathSourceOptions,
PathedSource,
StreamSource,
StreamSourceOptions,
type StreamSourceOptions,
RangedSource,
ReadableStreamSource,
ReadableStreamSourceOptions,
type ReadableStreamSourceOptions,
UrlSource,
UrlSourceOptions,
type UrlSourceOptions,
} from './source';
export {
InputFormat,
InputFormatOptions,
type InputFormatOptions,
AdtsInputFormat,
FlacInputFormat,
IsobmffInputFormat,
IsobmffInputFormatOptions,
type IsobmffInputFormatOptions,
HlsInputFormat,
MatroskaInputFormat,
Mp3InputFormat,
@@ -216,38 +216,38 @@ export {
} from './input-format';
export {
Input,
InputOptions,
InputEvents,
type InputOptions,
type InputEvents,
InputDisposedError,
UnsupportedInputFormatError,
} from './input';
export {
DurationMetadataRequestOptions,
type DurationMetadataRequestOptions,
} from './demuxer';
export {
InputTrack,
InputVideoTrack,
InputAudioTrack,
InputTrackQuery,
PacketStats,
type InputTrackQuery,
type PacketStats,
asc,
desc,
prefer,
} from './input-track';
export {
EncodedPacket,
EncodedPacketSideData,
PacketType,
type EncodedPacketSideData,
type PacketType,
} from './packet';
export {
AudioSample,
AudioSampleInit,
AudioSampleCopyToOptions,
type AudioSampleInit,
type AudioSampleCopyToOptions,
VideoSample,
VideoSampleInit,
VideoSamplePixelFormat,
type VideoSampleInit,
type VideoSamplePixelFormat,
VideoSampleColorSpace,
CropRectangle,
type CropRectangle,
VIDEO_SAMPLE_PIXEL_FORMATS,
} from './sample';
export {
@@ -255,20 +255,20 @@ export {
AudioSampleSink,
BaseMediaSampleSink,
CanvasSink,
CanvasSinkOptions,
type CanvasSinkOptions,
EncodedPacketSink,
PacketRetrievalOptions,
type PacketRetrievalOptions,
VideoSampleSink,
WrappedAudioBuffer,
WrappedCanvas,
type WrappedAudioBuffer,
type WrappedCanvas,
} from './media-sink';
export {
Conversion,
ConversionOptions,
ConversionVideoOptions,
ConversionAudioOptions,
type ConversionOptions,
type ConversionVideoOptions,
type ConversionAudioOptions,
ConversionCanceledError,
DiscardedTrack,
type DiscardedTrack,
} from './conversion';
export {
CustomVideoDecoder,
@@ -279,11 +279,11 @@ export {
registerEncoder,
} from './custom-coder';
export {
MetadataTags,
AttachedImage,
type MetadataTags,
type AttachedImage,
RichImageData,
AttachedFile,
TrackDisposition,
type TrackDisposition,
} from './metadata';
// 🐡🦔
+46 -8
View File
@@ -174,6 +174,12 @@ const u64 = (value: number) => {
return [bytes[0], bytes[1], bytes[2], bytes[3], bytes[4], bytes[5], bytes[6], bytes[7]] as number[];
};
const i64 = (value: number) => {
view.setInt32(0, Math.floor(value / 2 ** 32), false);
view.setUint32(4, value, false);
return [bytes[0], bytes[1], bytes[2], bytes[3], bytes[4], bytes[5], bytes[6], bytes[7]] as number[];
};
const fixed_8_8 = (value: number) => {
view.setInt16(0, 2 ** 8 * value, false);
return [bytes[0], bytes[1]] as number[];
@@ -384,11 +390,14 @@ export const mvhd = (
creationTime: number,
trackDatas: IsobmffTrackData[],
) => {
const duration = intoTimescale(Math.max(
const duration = Math.max(
0,
...trackDatas
.map(x => presentationSpan(x)),
), GLOBAL_TIMESCALE);
.map(trackData => (
intoTimescale(presentationSpan(trackData), GLOBAL_TIMESCALE)
+ intoTimescale(trackData.startTimestampOffset ?? 0, GLOBAL_TIMESCALE)
)),
);
const nextTrackId = Math.max(0, ...trackDatas.map(x => x.track.id)) + 1;
// Conditionally use u64 if u32 isn't enough
@@ -417,7 +426,9 @@ const presentationSpan = (trackData: IsobmffTrackData) => {
let minTimestamp = Infinity;
let maxEndTimestamp = -Infinity;
for (const sample of trackData.samples) {
for (let i = 0; i < trackData.samples.length; i++) {
const sample = trackData.samples[i]!;
if (sample.timestamp < minTimestamp) {
minTimestamp = sample.timestamp;
}
@@ -440,9 +451,11 @@ const presentationSpan = (trackData: IsobmffTrackData) => {
*/
export const trak = (trackData: IsobmffTrackData, creationTime: number) => {
const trackMetadata = getTrackMetadata(trackData);
const needsEditList = trackData.startTimestampOffset !== null && trackData.startTimestampOffset > 0;
return box('trak', undefined, [
tkhd(trackData, creationTime),
needsEditList ? edts(trackData, trackData.startTimestampOffset!) : null,
mdia(trackData, creationTime),
trackMetadata.name !== undefined
? box('udta', undefined, [
@@ -459,10 +472,8 @@ export const tkhd = (
trackData: IsobmffTrackData,
creationTime: number,
) => {
const durationInGlobalTimescale = intoTimescale(
presentationSpan(trackData),
GLOBAL_TIMESCALE,
);
const durationInGlobalTimescale = intoTimescale(presentationSpan(trackData), GLOBAL_TIMESCALE)
+ intoTimescale(trackData.startTimestampOffset ?? 0, GLOBAL_TIMESCALE);
const needsU64 = !isU32(creationTime) || !isU32(durationInGlobalTimescale);
const u32OrU64 = needsU64 ? u64 : u32;
@@ -497,6 +508,32 @@ export const tkhd = (
]);
};
/** Edit Box: Specifies edits to the track's media. */
export const edts = (trackData: IsobmffTrackData, offset: number) => {
const startOffset = intoTimescale(offset, GLOBAL_TIMESCALE);
const mediaDuration = intoTimescale(presentationSpan(trackData), GLOBAL_TIMESCALE);
const needs64Bits = !isU32(startOffset) || !isU32(mediaDuration);
const u32OrU64 = needs64Bits ? u64 : u32;
const i32OrI64 = needs64Bits ? i64 : i32;
return box('edts', undefined, [
fullBox('elst', needs64Bits ? 1 : 0, 0, [
u32(2), // Entry count
// #1
u32OrU64(startOffset), // Segment duration
i32OrI64(-1), // Media time
fixed_16_16(1), // Media rate
// #2
u32OrU64(mediaDuration), // Segment duration
i32OrI64(0), // Media time
fixed_16_16(1), // Media rate
]),
]);
};
/** Media Box: Describes and define a track's media type and sample data. */
export const mdia = (trackData: IsobmffTrackData, creationTime: number) => box('mdia', undefined, [
mdhd(trackData, creationTime),
@@ -509,6 +546,7 @@ export const mdhd = (
trackData: IsobmffTrackData,
creationTime: number,
) => {
// Since the duration represents the raw media duration, edit list offsets are not taken into account here
const localDuration = intoTimescale(
presentationSpan(trackData),
trackData.timescale,
+71 -28
View File
@@ -52,7 +52,7 @@ import {
import { buildIsobmffMimeType } from './isobmff-misc';
import { MAX_BOX_HEADER_SIZE, MIN_BOX_HEADER_SIZE } from './isobmff-reader';
export const GLOBAL_TIMESCALE = 1000;
export const GLOBAL_TIMESCALE = 57600; // LCM of a bunch of common frame rates (24, 25, 30, 60, 144, ...)
const TIMESTAMP_OFFSET = 2_082_844_800; // Seconds between Jan 1 1904 and Jan 1 1970
export type Sample = {
@@ -85,6 +85,7 @@ export type IsobmffTrackData = {
compositionTimeOffsetTable: { sampleCount: number; sampleCompositionTimeOffset: number }[];
lastTimescaleUnits: number | null;
lastSample: Sample | null;
startTimestampOffset: number | null;
finalizedChunks: Chunk[];
currentChunk: Chunk | null;
@@ -120,6 +121,7 @@ export type IsobmffTrackData = {
* Some players expect this for PCM audio.
*/
requiresPcmTransformation: boolean;
expectedNextPcmPacketTimestamp: number | null;
/**
* The "ADTS stripping" involves removing the ADTS header from each AAC packet. SOBMFF stores raw AAC data, not
* ADTS-wrapped data.
@@ -395,7 +397,10 @@ export class IsobmffMuxer extends Muxer {
// The frame rate set by the user may not be an integer. Since timescale is an integer, we'll approximate the
// frame time (inverse of frame rate) with a rational number, then use that approximation's denominator
// as the timescale.
const timescale = computeRationalApproximation(1 / (track.metadata.frameRate ?? 57600), 1e6).denominator;
const timescale = computeRationalApproximation(
1 / (track.metadata.frameRate ?? GLOBAL_TIMESCALE),
1e6,
).denominator;
const displayAspectWidth = decoderConfig.displayAspectWidth;
const displayAspectHeight = decoderConfig.displayAspectHeight;
@@ -425,6 +430,7 @@ export class IsobmffMuxer extends Muxer {
compositionTimeOffsetTable: [],
lastTimescaleUnits: null,
lastSample: null,
startTimestampOffset: null,
finalizedChunks: [],
currentChunk: null,
compactlyCodedChunkTable: [],
@@ -494,6 +500,7 @@ export class IsobmffMuxer extends Muxer {
requiresPcmTransformation:
!this.isFragmented
&& (PCM_AUDIO_CODECS as readonly string[]).includes(track.source._codec),
expectedNextPcmPacketTimestamp: null,
requiresAdtsStripping,
firstPacket: packet,
},
@@ -505,6 +512,7 @@ export class IsobmffMuxer extends Muxer {
compositionTimeOffsetTable: [],
lastTimescaleUnits: null,
lastSample: null,
startTimestampOffset: null,
finalizedChunks: [],
currentChunk: null,
compactlyCodedChunkTable: [],
@@ -547,6 +555,7 @@ export class IsobmffMuxer extends Muxer {
compositionTimeOffsetTable: [],
lastTimescaleUnits: null,
lastSample: null,
startTimestampOffset: null,
finalizedChunks: [],
currentChunk: null,
compactlyCodedChunkTable: [],
@@ -629,41 +638,59 @@ export class IsobmffMuxer extends Muxer {
packetData = packetData.subarray(headerLength);
}
const timestamp = this.validateAndNormalizeTimestamp(
let timestamp = this.validateAndNormalizeTimestamp(
trackData.track,
packet.timestamp,
packet.type === 'key',
);
let duration = packet.duration;
if (trackData.info.requiresPcmTransformation) {
// Packets may have only approximate timestamp/duration information, but for our PCM logic, we need it
// to be precise. So here, we refine the values.
const pcmInfo = parsePcmCodec(
trackData.info.decoderConfig.codec as PcmAudioCodec,
);
const frameSize = pcmInfo.sampleSize * trackData.info.numberOfChannels;
// Compute the precise duration
duration = packetData.byteLength / frameSize / trackData.info.sampleRate;
if (trackData.info.expectedNextPcmPacketTimestamp !== null) {
const diff = timestamp - trackData.info.expectedNextPcmPacketTimestamp;
if (diff < 0.01) {
timestamp = trackData.info.expectedNextPcmPacketTimestamp;
} else {
const paddedDuration = await this.padWithSilence(
trackData,
trackData.info.expectedNextPcmPacketTimestamp,
diff,
);
timestamp = trackData.info.expectedNextPcmPacketTimestamp + paddedDuration;
}
}
trackData.info.expectedNextPcmPacketTimestamp = timestamp + duration;
}
const internalSample = this.createSampleForTrack(
trackData,
packetData,
timestamp,
packet.duration,
duration,
packet.type,
);
if (trackData.info.requiresPcmTransformation) {
await this.maybePadWithSilence(trackData, timestamp);
}
await this.registerSample(trackData, internalSample);
} finally {
release();
}
}
private async maybePadWithSilence(trackData: IsobmffAudioTrackData, untilTimestamp: number) {
// The PCM transformation assumes that all samples are contiguous. This is not something that is enforced, so
// we need to pad the "holes" in between samples (and before the first sample) with additional
// "silence samples".
const lastSample = last(trackData.samples);
const lastEndTimestamp = lastSample
? lastSample.timestamp + lastSample.duration
: 0;
const delta = untilTimestamp - lastEndTimestamp;
const deltaInTimescale = intoTimescale(delta, trackData.timescale);
private async padWithSilence(trackData: IsobmffAudioTrackData, timestamp: number, duration: number) {
const deltaInTimescale = intoTimescale(duration, trackData.timescale);
duration = deltaInTimescale / trackData.timescale;
if (deltaInTimescale > 0) {
const { sampleSize, silentValue } = parsePcmCodec(
@@ -675,12 +702,14 @@ export class IsobmffMuxer extends Muxer {
const paddingSample = this.createSampleForTrack(
trackData,
new Uint8Array(data.buffer),
lastEndTimestamp,
delta,
timestamp,
duration,
'key',
);
await this.registerSample(trackData, paddingSample);
}
return duration;
}
async addSubtitleCue(track: OutputSubtitleTrack, cue: SubtitleCue, meta?: SubtitleMetadata) {
@@ -821,6 +850,11 @@ export class IsobmffMuxer extends Muxer {
}
if (trackData.type === 'audio' && trackData.info.requiresPcmTransformation) {
if (!this.isFragmented) {
// The first timestamp is the lowest
trackData.startTimestampOffset ??= trackData.timestampProcessingQueue[0]!.timestamp;
}
let totalDuration = 0;
// Compute the total duration in the track timescale (which is equal to the amount of PCM audio samples)
@@ -848,6 +882,10 @@ export class IsobmffMuxer extends Muxer {
const sortedTimestamps = trackData.timestampProcessingQueue.map(x => x.timestamp).sort((a, b) => a - b);
if (!this.isFragmented) {
trackData.startTimestampOffset ??= sortedTimestamps[0]!;
}
for (let i = 0; i < trackData.timestampProcessingQueue.length; i++) {
const sample = trackData.timestampProcessingQueue[i]!;
@@ -857,12 +895,6 @@ export class IsobmffMuxer extends Muxer {
// model it.
sample.decodeTimestamp = sortedTimestamps[i]!;
if (!this.isFragmented && trackData.lastTimescaleUnits === null) {
// In non-fragmented files, the first decode timestamp is always zero. If the first presentation
// timestamp isn't zero, we'll simply use the composition time offset to achieve it.
sample.decodeTimestamp = 0;
}
const sampleCompositionTimeOffset
= intoTimescale(sample.timestamp - sample.decodeTimestamp, trackData.timescale);
const durationInTimescale = intoTimescale(sample.duration, trackData.timescale);
@@ -1377,6 +1409,17 @@ export class IsobmffMuxer extends Muxer {
} else {
for (const trackData of this.trackDatas) {
await this.finalizeCurrentChunk(trackData);
// Must hold because we will have processed at least one sample
assert(trackData.startTimestampOffset !== null);
// Shift all of the samples by the start offset. We'll then write out an edit list that will shift them
// back to their proper spot in the composition.
for (let i = 0; i < trackData.samples.length; i++) {
const sample = trackData.samples[i]!;
sample.timestamp -= trackData.startTimestampOffset;
sample.decodeTimestamp -= trackData.startTimestampOffset;
}
}
}
+20 -11
View File
@@ -17,6 +17,7 @@ import {
iterateAvcNalUnits,
iterateHevcNalUnits,
parseAvcSps,
sanitizeHevcPacketForChromium,
} from './codec-data';
import { CustomVideoDecoder, customVideoDecoders, CustomAudioDecoder, customAudioDecoders } from './custom-coder';
import { InputDisposedError } from './input';
@@ -980,20 +981,28 @@ class VideoDecoderWrapper extends DecoderWrapper<VideoSample> {
insertSorted(this.inputTimestamps, packet.timestamp, x => x);
}
// Workaround for https://issues.chromium.org/issues/470109459
if (isChromium() && this.currentPacketIndex === 0 && this.codec === 'avc') {
const filteredNalUnits: Uint8Array[] = [];
if (isChromium() && this.currentPacketIndex === 0) {
if (this.codec === 'avc') {
// Workaround for https://issues.chromium.org/issues/470109459
const filteredNalUnits: Uint8Array[] = [];
for (const loc of iterateAvcNalUnits(packet.data, this.decoderConfig)) {
const type = extractNalUnitTypeForAvc(packet.data[loc.offset]!);
// These trip up Chromium's key frame detection, so let's strip them
if (!(type >= 20 && type <= 31)) {
filteredNalUnits.push(packet.data.subarray(loc.offset, loc.offset + loc.length));
for (const loc of iterateAvcNalUnits(packet.data, this.decoderConfig)) {
const type = extractNalUnitTypeForAvc(packet.data[loc.offset]!);
// These trip up Chromium's key frame detection, so let's strip them
if (!(type >= 20 && type <= 31)) {
filteredNalUnits.push(packet.data.subarray(loc.offset, loc.offset + loc.length));
}
}
const newData = concatAvcNalUnits(filteredNalUnits, this.decoderConfig);
packet = new EncodedPacket(newData, packet.type, packet.timestamp, packet.duration);
} else if (this.codec === 'hevc') {
// Workaround for https://issues.chromium.org/issues/507611247
const sanitizedData = sanitizeHevcPacketForChromium(packet.data, this.decoderConfig);
if (sanitizedData) {
packet = new EncodedPacket(sanitizedData, packet.type, packet.timestamp, packet.duration);
}
}
const newData = concatAvcNalUnits(filteredNalUnits, this.decoderConfig);
packet = new EncodedPacket(newData, packet.type, packet.timestamp, packet.duration);
}
this.decoder.decode(packet.toEncodedVideoChunk());
+26 -1
View File
@@ -50,7 +50,13 @@ import {
customAudioEncoders,
} from './custom-coder';
import { EncodedPacket, EncodedPacketSideData } from './packet';
import { AudioSample, clampCropRectangle, VideoSample } from './sample';
import {
AudioSample,
audioSampleToInterleavedFormat,
clampCropRectangle,
toInterleavedAudioFormat,
VideoSample,
} from './sample';
import {
AudioEncodingConfig,
buildAudioEncoderConfig,
@@ -1888,6 +1894,21 @@ class AudioEncoderWrapper {
private async processAndEncode(audioSample: AudioSample, shouldClose: boolean) {
const config = this.encodingConfig;
if (
config.transform?.sampleFormat !== undefined
&& toInterleavedAudioFormat(audioSample.format) !== config.transform.sampleFormat
) {
// Do a sample format conversion
const newSample = audioSampleToInterleavedFormat(audioSample, config.transform.sampleFormat);
if (shouldClose) {
audioSample.close();
}
audioSample = newSample;
shouldClose = true;
}
if (config.transform?.process) {
let processed = config.transform.process(audioSample);
if (processed instanceof Promise) {
@@ -1910,6 +1931,10 @@ class AudioEncoderWrapper {
}
await this.encodeSample(sample, true);
}
if (shouldClose) {
audioSample.close();
}
} else {
await this.encodeSample(audioSample, shouldClose);
}
+30
View File
@@ -1953,6 +1953,21 @@ const isAudioData = (x: unknown): x is AudioData => {
return typeof AudioData !== 'undefined' && x instanceof AudioData;
};
export const toInterleavedAudioFormat = (format: AudioSampleFormat): 'u8' | 's16' | 's32' | 'f32' => {
switch (format) {
case 'u8-planar':
return 'u8';
case 's16-planar':
return 's16';
case 's32-planar':
return 's32';
case 'f32-planar':
return 'f32';
default:
return format;
}
};
/**
* WebKit has a bug where calling AudioData.copyTo with a format different from the source format
* crashes the tab when there are more than 2 channels. This function works around that by always
@@ -2061,3 +2076,18 @@ const doAudioDataCopyToWebKitWorkaround = (
}
}
};
export const audioSampleToInterleavedFormat = (sample: AudioSample, format: 'u8' | 's16' | 's32' | 'f32') => {
const size = sample.allocationSize({ format, planeIndex: 0 });
const buffer = new ArrayBuffer(size);
sample.copyTo(buffer, { format, planeIndex: 0 });
return new AudioSample({
data: buffer,
format,
numberOfChannels: sample.numberOfChannels,
sampleRate: sample.sampleRate,
timestamp: sample.timestamp,
duration: sample.duration,
});
};
+107
View File
@@ -0,0 +1,107 @@
import { expect, test } from 'vitest';
import { Input } from '../../src/input.js';
import { ALL_FORMATS } from '../../src/input-format.js';
import { AudioSampleSource } from '../../src/media-source.js';
import { AudioSampleSink } from '../../src/media-sink.js';
import { assert } from '../../src/misc.js';
import { Output } from '../../src/output.js';
import { FlacOutputFormat } from '../../src/output-format.js';
import { AudioSample } from '../../src/sample.js';
import { BufferSource } from '../../src/source.js';
import { BufferTarget } from '../../src/target.js';
import { registerFlacEncoder } from '@mediabunny/flac-encoder';
test('FLAC encoder, 24-bit', async () => {
registerFlacEncoder();
const sampleRate = 48000;
const channels = 2;
const durationSeconds = 2;
const data = createF32SineWave(sampleRate, channels, durationSeconds);
using sample = await encodeAndDecodeFirstSample(new AudioSample({
data,
format: 'f32',
numberOfChannels: channels,
sampleRate,
timestamp: 0,
}));
expect(sample.format).toBe('s32');
});
test('FLAC encoder, 16-bit', async () => {
registerFlacEncoder();
const sampleRate = 48000;
const channels = 2;
const durationSeconds = 2;
const data = createS16SineWave(sampleRate, channels, durationSeconds);
using sample = await encodeAndDecodeFirstSample(new AudioSample({
data,
format: 's16',
numberOfChannels: channels,
sampleRate,
timestamp: 0,
}));
expect(sample.format).toBe('s16');
});
const createF32SineWave = (sampleRate: number, channels: number, durationSeconds: number) => {
const totalFrames = sampleRate * durationSeconds;
const data = new Float32Array(totalFrames * channels);
for (let i = 0; i < totalFrames; i++) {
const value = Math.sin(2 * Math.PI * 440 * i / sampleRate);
for (let ch = 0; ch < channels; ch++) {
data[i * channels + ch] = value;
}
}
return data;
};
const createS16SineWave = (sampleRate: number, channels: number, durationSeconds: number) => {
const totalFrames = sampleRate * durationSeconds;
const data = new Int16Array(totalFrames * channels);
for (let i = 0; i < totalFrames; i++) {
const value = Math.round(Math.sin(2 * Math.PI * 440 * i / sampleRate) * 32767);
for (let ch = 0; ch < channels; ch++) {
data[i * channels + ch] = value;
}
}
return data;
};
const encodeAndDecodeFirstSample = async (audioSample: AudioSample) => {
const output = new Output({
format: new FlacOutputFormat(),
target: new BufferTarget(),
});
const audioSource = new AudioSampleSource({ codec: 'flac' });
output.addAudioTrack(audioSource);
await output.start();
await audioSource.add(audioSample);
audioSource.close();
await output.finalize();
using input = new Input({
source: new BufferSource(output.target.buffer!),
formats: ALL_FORMATS,
});
const track = await input.getPrimaryAudioTrack();
assert(track);
const sink = new AudioSampleSink(track);
const sample = await sink.getSample(0);
assert(sample);
return sample;
};
+2 -2
View File
@@ -47,7 +47,7 @@ test('FLAC encoding', async () => {
target: new BufferTarget(),
});
const audioSource = new AudioSampleSource({ codec: 'flac', bitrate: 1 });
const audioSource = new AudioSampleSource({ codec: 'flac' });
output.addAudioTrack(audioSource);
await output.start();
@@ -97,7 +97,7 @@ test('FLAC with huge timestamps', async () => {
target: new BufferTarget(),
});
const audioSource = new AudioSampleSource({ codec: 'flac', bitrate: 1 });
const audioSource = new AudioSampleSource({ codec: 'flac' });
output.addAudioTrack(audioSource);
await output.start();
+248
View File
@@ -9,6 +9,8 @@ import { BufferTarget } from '../../src/target.js';
import { Mp4OutputFormat } from '../../src/output-format.js';
import { Conversion } from '../../src/conversion.js';
import { assert } from '../../src/misc.js';
import { EncodedAudioPacketSource, EncodedVideoPacketSource } from '../../src/media-source.js';
import { EncodedPacket } from '../../src/packet.js';
const __dirname = new URL('.', import.meta.url).pathname;
@@ -104,3 +106,249 @@ test('Fragmented fMP4 with video+audio preserves B-frame CTS', async () => {
expect(timestamps).toEqual(originalTimestamps);
});
test('Zero start timestamp, regular MP4', async () => {
const output = new Output({
format: new Mp4OutputFormat(),
target: new BufferTarget(),
});
const source = new EncodedVideoPacketSource('vp8');
output.addVideoTrack(source);
await output.start();
const meta = { decoderConfig: { codec: 'vp8', codedWidth: 1280, codedHeight: 720 } };
await source.add(new EncodedPacket(new Uint8Array(1024), 'key', 0, 0.1), meta);
await source.add(new EncodedPacket(new Uint8Array(1024), 'delta', 0.1, 0.1));
await source.add(new EncodedPacket(new Uint8Array(1024), 'delta', 0.2, 0.1));
await source.add(new EncodedPacket(new Uint8Array(1024), 'delta', 0.3, 0.1));
await output.finalize();
// Hacky but works
const str = String.fromCharCode(...new Uint8Array(output.target.buffer!));
expect(str.includes('edts') || str.includes('elst')).toBe(false);
});
test('Non-zero start timestamp, regular MP4', async () => {
const output = new Output({
format: new Mp4OutputFormat(),
target: new BufferTarget(),
});
const source = new EncodedVideoPacketSource('vp8');
output.addVideoTrack(source);
await output.start();
const meta = { decoderConfig: { codec: 'vp8', codedWidth: 1280, codedHeight: 720 } };
await source.add(new EncodedPacket(new Uint8Array(1024), 'key', 1, 0.1), meta);
await source.add(new EncodedPacket(new Uint8Array(1024), 'delta', 1.1, 0.1));
await source.add(new EncodedPacket(new Uint8Array(1024), 'delta', 1.2, 0.1));
await source.add(new EncodedPacket(new Uint8Array(1024), 'delta', 1.3, 0.1));
await output.finalize();
// Hacky but works
const str = String.fromCharCode(...new Uint8Array(output.target.buffer!));
expect(str.includes('edts') && str.includes('elst')).toBe(true);
using input = new Input({
source: new BufferSource(output.target.buffer!),
formats: ALL_FORMATS,
});
const track = await input.getPrimaryVideoTrack();
assert(track);
const sink = new EncodedPacketSink(track);
const timestamps: number[] = [];
const durations: number[] = [];
for await (const packet of sink.packets()) {
timestamps.push(packet.timestamp);
durations.push(packet.duration);
}
expect(timestamps).toEqual([1, 1.1, 1.2, 1.3]);
expect(durations).toEqual([0.1, 0.1, 0.1, 0.1]);
});
test('Non-zero start timestamp, fragmented MP4', async () => {
const output = new Output({
format: new Mp4OutputFormat({ fastStart: 'fragmented' }),
target: new BufferTarget(),
});
const source = new EncodedVideoPacketSource('vp8');
output.addVideoTrack(source);
await output.start();
const meta = { decoderConfig: { codec: 'vp8', codedWidth: 1280, codedHeight: 720 } };
await source.add(new EncodedPacket(new Uint8Array(1024), 'key', 1, 0.1), meta);
await source.add(new EncodedPacket(new Uint8Array(1024), 'delta', 1.1, 0.1));
await source.add(new EncodedPacket(new Uint8Array(1024), 'delta', 1.2, 0.1));
await source.add(new EncodedPacket(new Uint8Array(1024), 'delta', 1.3, 0.1));
await output.finalize();
// Hacky but works
const str = String.fromCharCode(...new Uint8Array(output.target.buffer!));
expect(str.includes('edts') || str.includes('elst')).toBe(false);
using input = new Input({
source: new BufferSource(output.target.buffer!),
formats: ALL_FORMATS,
});
const track = await input.getPrimaryVideoTrack();
assert(track);
const sink = new EncodedPacketSink(track);
const timestamps: number[] = [];
const durations: number[] = [];
for await (const packet of sink.packets()) {
timestamps.push(packet.timestamp);
durations.push(packet.duration);
}
expect(timestamps).toEqual([1, 1.1, 1.2, 1.3]);
expect(durations).toEqual([0.1, 0.1, 0.1, 0.1]);
});
test('PCM audio', async () => {
const output = new Output({
format: new Mp4OutputFormat(),
target: new BufferTarget(),
});
const source = new EncodedAudioPacketSource('pcm-s16');
output.addAudioTrack(source);
await output.start();
const meta = { decoderConfig: { codec: 'pcm-s16', numberOfChannels: 2, sampleRate: 48000 } };
await source.add(new EncodedPacket(new Uint8Array(1024), 'key', 0, 1024 / 2 / 2 / 48000), meta);
await output.finalize();
const input = new Input({
source: new BufferSource(output.target.buffer!),
formats: ALL_FORMATS,
});
const audioTrack = await input.getPrimaryAudioTrack();
assert(audioTrack);
expect(await audioTrack.getFirstTimestamp()).toBe(0);
});
test('PCM audio with non-zero timestamp', async () => {
const output = new Output({
format: new Mp4OutputFormat(),
target: new BufferTarget(),
});
const source = new EncodedAudioPacketSource('pcm-s16');
output.addAudioTrack(source);
await output.start();
const meta = { decoderConfig: { codec: 'pcm-s16', numberOfChannels: 2, sampleRate: 48000 } };
await source.add(new EncodedPacket(new Uint8Array(1024), 'key', 1, 1024 / 2 / 2 / 48000), meta);
await output.finalize();
const input = new Input({
source: new BufferSource(output.target.buffer!),
formats: ALL_FORMATS,
});
const audioTrack = await input.getPrimaryAudioTrack();
assert(audioTrack);
expect(await audioTrack.getFirstTimestamp()).toBe(1);
});
test('PCM audio, silence padding', async () => {
const output = new Output({
format: new Mp4OutputFormat(),
target: new BufferTarget(),
});
const source = new EncodedAudioPacketSource('pcm-s16');
output.addAudioTrack(source);
await output.start();
const meta = { decoderConfig: { codec: 'pcm-s16', numberOfChannels: 2, sampleRate: 48000 } };
await source.add(new EncodedPacket(new Uint8Array(1024), 'key', 0, 1024 / 2 / 2 / 48000), meta);
await source.add(new EncodedPacket(new Uint8Array(1024), 'key', 1, 1024 / 2 / 2 / 48000), meta);
await output.finalize();
const input = new Input({
source: new BufferSource(output.target.buffer!),
formats: ALL_FORMATS,
});
const audioTrack = await input.getPrimaryAudioTrack();
assert(audioTrack);
expect(await audioTrack.getCodec()).toBe('pcm-s16');
const numChannels = await audioTrack.getNumberOfChannels();
const expectedFrameCount = 48000 + 256;
const sink = new EncodedPacketSink(audioTrack);
let frameCount = 0;
for await (const packet of sink.packets()) {
frameCount += packet.byteLength / 2 / numChannels;
}
expect(frameCount).toBe(expectedFrameCount);
});
test('PCM audio, no silence padding with approximate timestamps', async () => {
const output = new Output({
format: new Mp4OutputFormat(),
target: new BufferTarget(),
});
const source = new EncodedAudioPacketSource('pcm-s16');
output.addAudioTrack(source);
await output.start();
const meta = { decoderConfig: { codec: 'pcm-s16', numberOfChannels: 2, sampleRate: 48000 } };
await source.add(new EncodedPacket(new Uint8Array(1024), 'key', 0, 1024 / 2 / 2 / 48000), meta);
// 0.006 is 256/48000 "rounded up", but it's close enough for silence padding not to kick in
await source.add(new EncodedPacket(new Uint8Array(1024), 'key', 0.006, 1024 / 2 / 2 / 48000), meta);
await output.finalize();
const input = new Input({
source: new BufferSource(output.target.buffer!),
formats: ALL_FORMATS,
});
const audioTrack = await input.getPrimaryAudioTrack();
assert(audioTrack);
expect(await audioTrack.getCodec()).toBe('pcm-s16');
const numChannels = await audioTrack.getNumberOfChannels();
const expectedFrameCount = 256 + 256;
const sink = new EncodedPacketSink(audioTrack);
let frameCount = 0;
for await (const packet of sink.packets()) {
frameCount += packet.byteLength / 2 / numChannels;
}
expect(frameCount).toBe(expectedFrameCount);
});