mirror of
https://github.com/arcodange-org/mediabunny.git
synced 2026-10-06 07:13:47 +02:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
62dfc5dd1c | ||
|
|
782d3e134b | ||
|
|
9224fb886c | ||
|
|
a61631a299 | ||
|
|
f3dec587fd | ||
|
|
2d49122277 | ||
|
|
06a89ed085 |
+14
-8
@@ -3,10 +3,12 @@
|
||||
<script src="../dist/bundles/mediabunny.cjs"></script>
|
||||
<script src="../packages/mp3-encoder/dist/bundles/mediabunny-mp3-encoder.js"></script>
|
||||
<script src="../packages/ac3/dist/bundles/mediabunny-ac3.js"></script>
|
||||
<script src="../packages/flac-encoder/dist/bundles/mediabunny-flac-encoder.js"></script>
|
||||
|
||||
<script type="module">
|
||||
//MediabunnyMp3Encoder.registerMp3Encoder();
|
||||
MediabunnyAc3.registerAc3Decoder();
|
||||
MediabunnyFlacEncoder.registerFlacEncoder();
|
||||
|
||||
const fileInput = document.createElement('input');
|
||||
fileInput.type = 'file';
|
||||
@@ -23,7 +25,7 @@
|
||||
chunked: true,
|
||||
chunkSize: 2**20
|
||||
});
|
||||
const outputFormat = new Mediabunny.WavOutputFormat();
|
||||
const outputFormat = new Mediabunny.FlacOutputFormat();
|
||||
|
||||
const p = document.createElement('p');
|
||||
p.textContent = 'Capturing...';
|
||||
@@ -57,7 +59,7 @@
|
||||
const tracks = [];
|
||||
let start = 0;
|
||||
|
||||
if (true) {
|
||||
if (false) {
|
||||
input = new Mediabunny.Input({
|
||||
source: new Mediabunny.UrlSource('http://localhost:8000/index.m3u8'),
|
||||
formats: Mediabunny.ALL_FORMATS,
|
||||
@@ -85,15 +87,18 @@
|
||||
});
|
||||
}
|
||||
|
||||
const primaryTrack = await input.getPrimaryAudioTrack();
|
||||
const startTime = await primaryTrack.getFirstTimestamp();
|
||||
console.log(startTime)
|
||||
//const primaryTrack = await input.getPrimaryAudioTrack();
|
||||
//const startTime = await primaryTrack.getFirstTimestamp();
|
||||
//console.log(startTime)
|
||||
|
||||
let ctx = null;
|
||||
let conversion = await Mediabunny.Conversion.init({
|
||||
input,
|
||||
output,
|
||||
audio: (track) => ({ discard: track.number !== primaryTrack.number }),
|
||||
audio: {
|
||||
//forceTranscode: true,
|
||||
//sampleFormat: 's16',
|
||||
},
|
||||
/*
|
||||
video: {
|
||||
discard: true,
|
||||
@@ -139,11 +144,12 @@
|
||||
}
|
||||
},
|
||||
trim: {
|
||||
start: startTime,
|
||||
end: startTime + 2,
|
||||
//start: startTime,
|
||||
//end: startTime + 2,
|
||||
},
|
||||
});
|
||||
//console.log(conversion);
|
||||
console.log(conversion.discardedTracks);
|
||||
|
||||
let progress = 0;
|
||||
conversion.onProgress = newProgress => progress = newProgress;
|
||||
|
||||
+34
-6
@@ -11,7 +11,6 @@
|
||||
document.body.append(fileInput);
|
||||
|
||||
fileInput.addEventListener('change', async () => {
|
||||
/*
|
||||
const file = fileInput.files[0];
|
||||
const input = new Mediabunny.Input({
|
||||
formats: Mediabunny.ALL_FORMATS,
|
||||
@@ -19,15 +18,43 @@
|
||||
});
|
||||
|
||||
const track = await input.getPrimaryAudioTrack();
|
||||
const sink3 = new Mediabunny.EncodedPacketSink(track);
|
||||
console.log((await sink3.getFirstPacket()).data.join(', '));
|
||||
return;
|
||||
console.log(await track.getDurationFromMetadata(), await track.computeDuration());
|
||||
const sink = new Mediabunny.EncodedPacketSink(track);
|
||||
|
||||
for await (const packet of sink.packets()) {
|
||||
console.log(packet);
|
||||
}
|
||||
|
||||
const output = new Mediabunny.Output({
|
||||
format: new Mediabunny.Mp4OutputFormat(),
|
||||
target: new Mediabunny.BufferTarget(),
|
||||
});
|
||||
|
||||
const conversion = await Mediabunny.Conversion.init({ input, output });
|
||||
await conversion.execute();
|
||||
|
||||
//console.log(await input.getDurationFromMetadata(), await input.computeDuration());
|
||||
return;
|
||||
|
||||
// Download it now
|
||||
const blob = new Blob([output.target.buffer]);
|
||||
const url = URL.createObjectURL(blob);
|
||||
const a = document.createElement('a');
|
||||
a.href = url;
|
||||
a.download = file.name.replace(/\.\w+$/, '.mp4');
|
||||
a.click();
|
||||
URL.revokeObjectURL(url);
|
||||
|
||||
/*
|
||||
const track = await input.getPrimaryAudioTrack();
|
||||
const sink = new Mediabunny.EncodedPacketSink(track);
|
||||
|
||||
for await (const packet of sink.packets()) {
|
||||
console.log(packet);
|
||||
}
|
||||
|
||||
console.log("Done")
|
||||
*/
|
||||
|
||||
/*
|
||||
const input = new Mediabunny.Input({
|
||||
source: new Mediabunny.UrlSource('https://storage.googleapis.com/shaka-demo-assets/angel-one-widevine-hls/hls.m3u8'),
|
||||
formats: Mediabunny.ALL_FORMATS,
|
||||
@@ -36,6 +63,7 @@
|
||||
const videoTrack = await input.getPrimaryVideoTrack();
|
||||
const sink = new Mediabunny.EncodedPacketSink(videoTrack);
|
||||
console.log(await sink.getFirstPacket());
|
||||
*/
|
||||
|
||||
/*
|
||||
return
|
||||
|
||||
@@ -196,10 +196,14 @@ export default withMermaid({
|
||||
if (title !== 'Mediabunny') {
|
||||
title += ' | Mediabunny';
|
||||
}
|
||||
const canonicalUrl = `https://mediabunny.dev/${pageData.relativePath}`
|
||||
.replace(/index\.md$/, '')
|
||||
.replace(/\.md$/, '');
|
||||
|
||||
((pageData.frontmatter['head'] ??= []) as HeadConfig[]).push(
|
||||
['meta', { property: 'og:title', content: title }],
|
||||
['meta', { property: 'twitter:title', content: title }],
|
||||
['link', { rel: 'canonical', href: canonicalUrl }],
|
||||
);
|
||||
},
|
||||
});
|
||||
|
||||
@@ -77,7 +77,7 @@ The API surface added by the HLS update is vast and I obviously can't cover it i
|
||||
|
||||
By using the Conversion API, you can just do this:
|
||||
|
||||
<div class="text-xs">
|
||||
<div class="text-[13.7142857143px]">
|
||||
|
||||
```ts
|
||||
import { ... } from 'mediabunny';
|
||||
@@ -107,7 +107,7 @@ That's it. This will stream-download the entire HLS playlist, transcode it if ne
|
||||
|
||||
This is basically the inverse of the previous example. Just like we're able to read HLS and turn it into an MP4, we're able to read any input file and turn it into a full HLS playlist including master playlist, media playlists and segments:
|
||||
|
||||
<div class="text-xs">
|
||||
<div class="text-[13.7142857143px]">
|
||||
|
||||
```ts
|
||||
import { ... } from 'mediabunny';
|
||||
@@ -163,7 +163,7 @@ No transcode server is needed here, it's all handled by the client, and the serv
|
||||
|
||||
You could build an OBS-like broadcasting system where a user records their screen, facecam or microphone, encodes multiple variants locally, and then broadcasts finished HLS segments directly to the server, meaning no transcoding is needed.
|
||||
|
||||
<div class="text-xs overflow-auto">
|
||||
<div class="text-[13.7142857143px] overflow-auto">
|
||||
|
||||
```ts
|
||||
// Get the screen and mic
|
||||
|
||||
@@ -242,6 +242,7 @@ type ConversionAudioOptions = {
|
||||
bitrate?: number | Quality;
|
||||
numberOfChannels?: number;
|
||||
sampleRate?: number;
|
||||
sampleFormat?: 'u8' | 's16' | 's32' | 'f32';
|
||||
forceTranscode?: boolean;
|
||||
process?: (sample: AudioSample) => MaybePromise<
|
||||
AudioSample | AudioSample[] | null
|
||||
|
||||
@@ -32,6 +32,7 @@ export default tseslint.config(
|
||||
'@typescript-eslint/no-unsafe-enum-comparison': 'off',
|
||||
'@typescript-eslint/no-unsafe-unary-minus': 'off',
|
||||
'@typescript-eslint/no-deprecated': 'error',
|
||||
'@typescript-eslint/consistent-type-exports': 'error',
|
||||
},
|
||||
},
|
||||
{
|
||||
|
||||
@@ -9,6 +9,7 @@
|
||||
<script type="module" src="./file-compression.ts"></script>
|
||||
<link rel="stylesheet" href="../base.css">
|
||||
<link rel="icon" href="../../docs/public/mediabunny-logo.svg">
|
||||
<link rel="canonical" href="https://mediabunny.dev/examples/file-compression/">
|
||||
</head>
|
||||
|
||||
<body class="flex flex-col items-center py-10 bg-zinc-50 text-zinc-800 dark:bg-zinc-900 dark:text-zinc-200 px-2">
|
||||
|
||||
@@ -9,6 +9,7 @@
|
||||
<script type="module" src="./hls-transcoding.ts"></script>
|
||||
<link rel="stylesheet" href="../base.css">
|
||||
<link rel="icon" href="../../docs/public/mediabunny-logo.svg">
|
||||
<link rel="canonical" href="https://mediabunny.dev/examples/hls-transcoding/">
|
||||
</head>
|
||||
|
||||
<body class="flex flex-col items-center bg-zinc-50 px-2 py-10 text-zinc-800 dark:bg-zinc-900 dark:text-zinc-200">
|
||||
|
||||
@@ -9,6 +9,7 @@
|
||||
<script type="module" src="./live-recording.ts"></script>
|
||||
<link rel="stylesheet" href="../base.css">
|
||||
<link rel="icon" href="../../docs/public/mediabunny-logo.svg">
|
||||
<link rel="canonical" href="https://mediabunny.dev/examples/live-recording/">
|
||||
</head>
|
||||
|
||||
<body class="flex flex-col items-center py-10 bg-zinc-50 text-zinc-800 dark:bg-zinc-900 dark:text-zinc-200 px-2">
|
||||
|
||||
@@ -9,6 +9,7 @@
|
||||
<script type="module" src="./media-player.ts"></script>
|
||||
<link rel="stylesheet" href="../base.css">
|
||||
<link rel="icon" href="../../docs/public/mediabunny-logo.svg">
|
||||
<link rel="canonical" href="https://mediabunny.dev/examples/media-player/">
|
||||
</head>
|
||||
|
||||
<body class="flex flex-col items-center py-10 bg-zinc-50 text-zinc-800 dark:bg-zinc-900 dark:text-zinc-200 px-2 h-svh">
|
||||
|
||||
@@ -9,6 +9,7 @@
|
||||
<script type="module" src="./metadata-extraction.ts"></script>
|
||||
<link rel="stylesheet" href="../base.css">
|
||||
<link rel="icon" href="../../docs/public/mediabunny-logo.svg">
|
||||
<link rel="canonical" href="https://mediabunny.dev/examples/metadata-extraction/">
|
||||
</head>
|
||||
|
||||
<body class="flex flex-col items-center py-10 bg-zinc-50 text-zinc-800 dark:bg-zinc-900 dark:text-zinc-200 px-2">
|
||||
|
||||
@@ -9,6 +9,7 @@
|
||||
<script type="module" src="./procedural-generation.ts"></script>
|
||||
<link rel="stylesheet" href="../base.css">
|
||||
<link rel="icon" href="../../docs/public/mediabunny-logo.svg">
|
||||
<link rel="canonical" href="https://mediabunny.dev/examples/procedural-generation/">
|
||||
</head>
|
||||
|
||||
<body class="flex flex-col items-center py-10 bg-zinc-50 text-zinc-800 dark:bg-zinc-900 dark:text-zinc-200 px-2">
|
||||
|
||||
@@ -9,6 +9,7 @@
|
||||
<script type="module" src="./thumbnail-generation.ts"></script>
|
||||
<link rel="stylesheet" href="../base.css">
|
||||
<link rel="icon" href="../../docs/public/mediabunny-logo.svg">
|
||||
<link rel="canonical" href="https://mediabunny.dev/examples/thumbnail-generation/">
|
||||
</head>
|
||||
|
||||
<body class="flex flex-col items-center py-10 bg-gray-50 text-gray-800 dark:bg-zinc-900 dark:text-zinc-200 px-2">
|
||||
|
||||
Generated
+9
-9
@@ -1,12 +1,12 @@
|
||||
{
|
||||
"name": "mediabunny",
|
||||
"version": "1.42.0",
|
||||
"version": "1.43.0",
|
||||
"lockfileVersion": 3,
|
||||
"requires": true,
|
||||
"packages": {
|
||||
"": {
|
||||
"name": "mediabunny",
|
||||
"version": "1.42.0",
|
||||
"version": "1.43.0",
|
||||
"license": "MPL-2.0",
|
||||
"workspaces": [
|
||||
"packages/*"
|
||||
@@ -7751,9 +7751,9 @@
|
||||
}
|
||||
},
|
||||
"node_modules/mediabunny": {
|
||||
"version": "1.41.0",
|
||||
"resolved": "https://registry.npmjs.org/mediabunny/-/mediabunny-1.41.0.tgz",
|
||||
"integrity": "sha512-cJWBHvAyRNgTbsx8Z2oHptX/PdiywwsS+GIvv2k9eioXSOzzAemBsHwf7jM6fvW3b65azvZGP9RLuO1DI8JmdA==",
|
||||
"version": "1.42.0",
|
||||
"resolved": "https://registry.npmjs.org/mediabunny/-/mediabunny-1.42.0.tgz",
|
||||
"integrity": "sha512-s9ypTqLi6kbh95gC+YaJlG0PkLvMxu37Q/wO/pFZx0fUCA5Ym5mp+2dWoa83mKQ3Uo18aNlgev5iJ5ESZqWwgQ==",
|
||||
"license": "MPL-2.0",
|
||||
"peer": true,
|
||||
"workspaces": [
|
||||
@@ -12077,7 +12077,7 @@
|
||||
},
|
||||
"packages/aac-encoder": {
|
||||
"name": "@mediabunny/aac-encoder",
|
||||
"version": "1.42.0",
|
||||
"version": "1.43.0",
|
||||
"license": "MPL-2.0",
|
||||
"devDependencies": {
|
||||
"@types/emscripten": "^1.40.1"
|
||||
@@ -12092,7 +12092,7 @@
|
||||
},
|
||||
"packages/ac3": {
|
||||
"name": "@mediabunny/ac3",
|
||||
"version": "1.42.0",
|
||||
"version": "1.43.0",
|
||||
"license": "MPL-2.0",
|
||||
"devDependencies": {
|
||||
"@types/emscripten": "^1.40.1"
|
||||
@@ -12107,7 +12107,7 @@
|
||||
},
|
||||
"packages/flac-encoder": {
|
||||
"name": "@mediabunny/flac-encoder",
|
||||
"version": "1.42.0",
|
||||
"version": "1.43.0",
|
||||
"license": "MPL-2.0",
|
||||
"devDependencies": {
|
||||
"@types/emscripten": "^1.40.1"
|
||||
@@ -12122,7 +12122,7 @@
|
||||
},
|
||||
"packages/mp3-encoder": {
|
||||
"name": "@mediabunny/mp3-encoder",
|
||||
"version": "1.42.0",
|
||||
"version": "1.43.0",
|
||||
"license": "MPL-2.0",
|
||||
"devDependencies": {
|
||||
"@types/emscripten": "^1.40.1"
|
||||
|
||||
+1
-1
@@ -1,7 +1,7 @@
|
||||
{
|
||||
"name": "mediabunny",
|
||||
"author": "Vanilagy",
|
||||
"version": "1.42.0",
|
||||
"version": "1.43.0",
|
||||
"description": "Pure TypeScript media toolkit for reading, writing, and converting media files, directly in the browser.",
|
||||
"type": "module",
|
||||
"workspaces": [
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
{
|
||||
"name": "@mediabunny/aac-encoder",
|
||||
"author": "Vanilagy",
|
||||
"version": "1.42.0",
|
||||
"version": "1.43.0",
|
||||
"description": "AAC encoder extension for Mediabunny, based on FFmpeg.",
|
||||
"main": "./dist/bundles/mediabunny-aac-encoder.mjs",
|
||||
"module": "./dist/bundles/mediabunny-aac-encoder.mjs",
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
{
|
||||
"name": "@mediabunny/ac3",
|
||||
"author": "Vanilagy",
|
||||
"version": "1.42.0",
|
||||
"version": "1.43.0",
|
||||
"description": "AC-3 and E-AC-3 (Dolby Digital) decoder and encoder extension for Mediabunny, based on FFmpeg.",
|
||||
"main": "./dist/bundles/mediabunny-ac3.mjs",
|
||||
"module": "./dist/bundles/mediabunny-ac3.mjs",
|
||||
|
||||
Generated
BIN
Binary file not shown.
@@ -1,7 +1,7 @@
|
||||
{
|
||||
"name": "@mediabunny/flac-encoder",
|
||||
"author": "Vanilagy",
|
||||
"version": "1.42.0",
|
||||
"version": "1.43.0",
|
||||
"description": "FLAC encoder extension for Mediabunny, based on libFLAC.",
|
||||
"main": "./dist/bundles/mediabunny-flac-encoder.mjs",
|
||||
"module": "./dist/bundles/mediabunny-flac-encoder.mjs",
|
||||
|
||||
@@ -12,7 +12,6 @@
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
|
||||
#define BITS_PER_SAMPLE 16
|
||||
#define COMPRESSION_LEVEL 5
|
||||
|
||||
typedef struct {
|
||||
@@ -23,14 +22,10 @@ typedef struct {
|
||||
typedef struct {
|
||||
FLAC__StreamEncoder *encoder;
|
||||
|
||||
// Input buffer for interleaved int16 samples from JS
|
||||
int16_t *input_buffer;
|
||||
// Input buffer for interleaved int32 samples from JS
|
||||
FLAC__int32 *input_buffer;
|
||||
int input_buffer_size;
|
||||
|
||||
// Widened to int32 for libFLAC
|
||||
FLAC__int32 *int32_buffer;
|
||||
int int32_buffer_size;
|
||||
|
||||
// Contiguous output buffer for encoded frame data
|
||||
uint8_t *output_buffer;
|
||||
int output_size;
|
||||
@@ -48,6 +43,7 @@ typedef struct {
|
||||
bool header_done;
|
||||
|
||||
int channels;
|
||||
int bits_per_sample;
|
||||
} EncoderContext;
|
||||
|
||||
static void ensure_output_capacity(EncoderContext *ctx, int needed) {
|
||||
@@ -120,13 +116,14 @@ static void reset_output(EncoderContext *ctx) {
|
||||
}
|
||||
|
||||
EMSCRIPTEN_KEEPALIVE
|
||||
int init_encoder(int channels, int sample_rate) {
|
||||
int init_encoder(int channels, int sample_rate, int bits_per_sample) {
|
||||
EncoderContext *ctx = calloc(1, sizeof(EncoderContext));
|
||||
if (!ctx) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
ctx->channels = channels;
|
||||
ctx->bits_per_sample = bits_per_sample;
|
||||
|
||||
ctx->encoder = FLAC__stream_encoder_new();
|
||||
if (!ctx->encoder) {
|
||||
@@ -136,7 +133,7 @@ int init_encoder(int channels, int sample_rate) {
|
||||
|
||||
FLAC__stream_encoder_set_channels(ctx->encoder, channels);
|
||||
FLAC__stream_encoder_set_sample_rate(ctx->encoder, sample_rate);
|
||||
FLAC__stream_encoder_set_bits_per_sample(ctx->encoder, BITS_PER_SAMPLE);
|
||||
FLAC__stream_encoder_set_bits_per_sample(ctx->encoder, bits_per_sample);
|
||||
FLAC__stream_encoder_set_compression_level(ctx->encoder, COMPRESSION_LEVEL);
|
||||
FLAC__stream_encoder_set_verify(ctx->encoder, false);
|
||||
|
||||
@@ -174,19 +171,15 @@ EMSCRIPTEN_KEEPALIVE
|
||||
int send_samples(int ctx_ptr, int num_samples) {
|
||||
EncoderContext *ctx = (EncoderContext *)ctx_ptr;
|
||||
|
||||
// Widen int16 to int32 for libFLAC
|
||||
int total = num_samples * ctx->channels;
|
||||
if (total > ctx->int32_buffer_size) {
|
||||
ctx->int32_buffer = realloc(ctx->int32_buffer, total * sizeof(FLAC__int32));
|
||||
ctx->int32_buffer_size = total;
|
||||
}
|
||||
int shift = 32 - ctx->bits_per_sample;
|
||||
for (int i = 0; i < total; i++) {
|
||||
ctx->int32_buffer[i] = ctx->input_buffer[i];
|
||||
ctx->input_buffer[i] >>= shift;
|
||||
}
|
||||
|
||||
reset_output(ctx);
|
||||
|
||||
FLAC__bool ok = FLAC__stream_encoder_process_interleaved(ctx->encoder, ctx->int32_buffer, num_samples);
|
||||
FLAC__bool ok = FLAC__stream_encoder_process_interleaved(ctx->encoder, ctx->input_buffer, num_samples);
|
||||
return ok ? 0 : -1;
|
||||
}
|
||||
|
||||
|
||||
@@ -16,7 +16,7 @@ type ExtendedEmscriptenModule = EmscriptenModule & {
|
||||
let module: ExtendedEmscriptenModule;
|
||||
let modulePromise: Promise<ExtendedEmscriptenModule> | null = null;
|
||||
|
||||
let initEncoderFn: (channels: number, sampleRate: number) => number;
|
||||
let initEncoderFn: (channels: number, sampleRate: number, bitsPerSample: number) => number;
|
||||
let getEncodeInputPtr: (ctx: number, size: number) => number;
|
||||
let sendSamplesFn: (ctx: number, numSamples: number) => number;
|
||||
let getOutputData: (ctx: number) => number;
|
||||
@@ -37,7 +37,7 @@ const ensureModule = async () => {
|
||||
module = await modulePromise;
|
||||
modulePromise = null;
|
||||
|
||||
initEncoderFn = module.cwrap('init_encoder', 'number', ['number', 'number']);
|
||||
initEncoderFn = module.cwrap('init_encoder', 'number', ['number', 'number', 'number']);
|
||||
getEncodeInputPtr = module.cwrap('get_encode_input_ptr', 'number', ['number', 'number']);
|
||||
sendSamplesFn = module.cwrap('send_samples', 'number', ['number', 'number']);
|
||||
getOutputData = module.cwrap('get_output_data', 'number', ['number']);
|
||||
@@ -50,10 +50,10 @@ const ensureModule = async () => {
|
||||
}
|
||||
};
|
||||
|
||||
const initEncoder = async (numberOfChannels: number, sampleRate: number) => {
|
||||
const initEncoder = async (numberOfChannels: number, sampleRate: number, bitsPerSample: 16 | 24) => {
|
||||
await ensureModule();
|
||||
|
||||
const ctx = initEncoderFn(numberOfChannels, sampleRate);
|
||||
const ctx = initEncoderFn(numberOfChannels, sampleRate, bitsPerSample);
|
||||
if (ctx === 0) {
|
||||
throw new Error('Failed to initialize FLAC encoder.');
|
||||
}
|
||||
@@ -121,6 +121,7 @@ const onMessage = (data: { id: number; command: WorkerCommand }) => {
|
||||
const { ctx, header } = await initEncoder(
|
||||
command.data.numberOfChannels,
|
||||
command.data.sampleRate,
|
||||
command.data.bitsPerSample,
|
||||
);
|
||||
result = { type: command.type, ctx, header };
|
||||
transferables.push(header);
|
||||
|
||||
@@ -29,7 +29,7 @@ class FlacEncoder extends CustomAudioEncoder {
|
||||
reject: (reason?: unknown) => void;
|
||||
}>();
|
||||
|
||||
private ctx = 0;
|
||||
private ctx: number | null = null;
|
||||
private chunkMetadata: EncodedAudioChunkMetadata = {};
|
||||
private description: Uint8Array | null = null;
|
||||
private nextTimestampInSamples: number | null = null;
|
||||
@@ -65,19 +65,6 @@ class FlacEncoder extends CustomAudioEncoder {
|
||||
};
|
||||
nodeWorker.on('message', onMessage);
|
||||
}
|
||||
|
||||
const result = await this.sendCommand({
|
||||
type: 'init',
|
||||
data: {
|
||||
numberOfChannels: this.config.numberOfChannels,
|
||||
sampleRate: this.config.sampleRate,
|
||||
},
|
||||
});
|
||||
|
||||
this.ctx = result.ctx;
|
||||
|
||||
this.description = new Uint8Array(result.header);
|
||||
this.resetInternalState();
|
||||
}
|
||||
|
||||
private resetInternalState() {
|
||||
@@ -94,15 +81,50 @@ class FlacEncoder extends CustomAudioEncoder {
|
||||
}
|
||||
|
||||
async encode(audioSample: AudioSample) {
|
||||
if (this.ctx === null) {
|
||||
// This is the first sample, let's do some init
|
||||
|
||||
let bitsPerSample: 16 | 24;
|
||||
switch (audioSample.format) {
|
||||
case 'u8':
|
||||
case 'u8-planar':
|
||||
case 's16':
|
||||
case 's16-planar':
|
||||
bitsPerSample = 16;
|
||||
break;
|
||||
case 's32':
|
||||
case 's32-planar':
|
||||
case 'f32':
|
||||
case 'f32-planar':
|
||||
bitsPerSample = 24;
|
||||
break;
|
||||
default:
|
||||
assertNever(audioSample.format);
|
||||
assert(false);
|
||||
}
|
||||
|
||||
const result = await this.sendCommand({
|
||||
type: 'init',
|
||||
data: {
|
||||
numberOfChannels: this.config.numberOfChannels,
|
||||
sampleRate: this.config.sampleRate,
|
||||
bitsPerSample,
|
||||
},
|
||||
});
|
||||
|
||||
this.ctx = result.ctx;
|
||||
this.description = new Uint8Array(result.header);
|
||||
this.resetInternalState();
|
||||
}
|
||||
|
||||
if (this.nextTimestampInSamples === null) {
|
||||
this.nextTimestampInSamples = Math.round(audioSample.timestamp * this.config.sampleRate);
|
||||
}
|
||||
|
||||
const totalBytes = audioSample.allocationSize({ format: 's16', planeIndex: 0 });
|
||||
const audioBytes = new Uint8Array(totalBytes);
|
||||
audioSample.copyTo(audioBytes, { format: 's16', planeIndex: 0 });
|
||||
const totalBytes = audioSample.allocationSize({ format: 's32', planeIndex: 0 });
|
||||
const audioData = new ArrayBuffer(totalBytes);
|
||||
audioSample.copyTo(audioData, { format: 's32', planeIndex: 0 });
|
||||
|
||||
const audioData = audioBytes.buffer;
|
||||
const result = await this.sendCommand({
|
||||
type: 'encode',
|
||||
data: {
|
||||
@@ -116,6 +138,10 @@ class FlacEncoder extends CustomAudioEncoder {
|
||||
}
|
||||
|
||||
async flush() {
|
||||
if (this.ctx === null) {
|
||||
return;
|
||||
}
|
||||
|
||||
const result = await this.sendCommand({ type: 'flush', data: { ctx: this.ctx } });
|
||||
this.emitPackets(result.packets);
|
||||
|
||||
@@ -173,7 +199,8 @@ class FlacEncoder extends CustomAudioEncoder {
|
||||
|
||||
/**
|
||||
* Registers the FLAC encoder, which Mediabunny will then use automatically when applicable. Make sure to call this
|
||||
* function before starting any encoding task.
|
||||
* function before starting any encoding task. The FLAC encoder will automatically determine the output bit depth
|
||||
* (16 or 24) based on the sample format of incoming `AudioSample` instances.
|
||||
*
|
||||
* Preferably, wrap the call in a condition to avoid overriding any native FLAC encoder:
|
||||
*
|
||||
@@ -198,3 +225,8 @@ function assert(x: unknown): asserts x {
|
||||
throw new Error('Assertion failed.');
|
||||
}
|
||||
}
|
||||
|
||||
export const assertNever = (x: never) => {
|
||||
// eslint-disable-next-line @typescript-eslint/restrict-template-expressions
|
||||
throw new Error(`Unexpected value: ${x}`);
|
||||
};
|
||||
|
||||
@@ -16,6 +16,7 @@ export type WorkerCommand = {
|
||||
data: {
|
||||
numberOfChannels: number;
|
||||
sampleRate: number;
|
||||
bitsPerSample: 16 | 24;
|
||||
};
|
||||
} | {
|
||||
type: 'encode';
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
{
|
||||
"name": "@mediabunny/mp3-encoder",
|
||||
"author": "Vanilagy",
|
||||
"version": "1.42.0",
|
||||
"version": "1.43.0",
|
||||
"description": "MP3 encoder extension for Mediabunny, based on LAME.",
|
||||
"main": "./dist/bundles/mediabunny-mp3-encoder.mjs",
|
||||
"module": "./dist/bundles/mediabunny-mp3-encoder.mjs",
|
||||
|
||||
@@ -866,6 +866,21 @@ export type HevcSpsInfo = {
|
||||
minSpatialSegmentationIdc: number;
|
||||
};
|
||||
|
||||
export const concatHevcNalUnits = (nalUnits: Uint8Array[], decoderConfig: VideoDecoderConfig) => {
|
||||
if (decoderConfig.description) {
|
||||
// Stream is length-prefixed. Let's extract the size of the length prefix from the decoder config
|
||||
|
||||
const bytes = toUint8Array(decoderConfig.description);
|
||||
const lengthSizeMinusOne = bytes[21]! & 0b11;
|
||||
const lengthSize = (lengthSizeMinusOne + 1) as 1 | 2 | 3 | 4;
|
||||
|
||||
return concatNalUnitsInLengthPrefixed(nalUnits, lengthSize);
|
||||
} else {
|
||||
// Stream is in Annex B format
|
||||
return concatNalUnitsInAnnexB(nalUnits);
|
||||
}
|
||||
};
|
||||
|
||||
export const iterateHevcNalUnits = (packetData: Uint8Array, decoderConfig: VideoDecoderConfig) => {
|
||||
if (decoderConfig.description) {
|
||||
const bytes = toUint8Array(decoderConfig.description);
|
||||
@@ -1602,6 +1617,103 @@ export const deserializeHevcDecoderConfigurationRecord = (data: Uint8Array): Hev
|
||||
}
|
||||
};
|
||||
|
||||
enum HevcNaluOrderState {
|
||||
audAllowed,
|
||||
beforeFirstVcl,
|
||||
afterFirstVcl,
|
||||
eoBitstreamAllowed,
|
||||
noMoreDataAllowed,
|
||||
}
|
||||
|
||||
// This function sanitzes the contents of an HEVC packet such that
|
||||
// https://source.chromium.org/chromium/chromium/src/+/main:media/formats/mp4/hevc.cc's validation logic does not trip
|
||||
// up on its contents. The validation is often too strict and rejects packets that Chromium could decode just fine.
|
||||
// Chromium code retrieved on 2026-04-29.
|
||||
// See https://issues.chromium.org/issues/507611247.
|
||||
export const sanitizeHevcPacketForChromium = (
|
||||
packetData: Uint8Array,
|
||||
decoderConfig: VideoDecoderConfig,
|
||||
): Uint8Array | null => {
|
||||
const removedNalUnits = new Set<number>();
|
||||
let orderState: HevcNaluOrderState = HevcNaluOrderState.audAllowed;
|
||||
|
||||
for (const loc of iterateHevcNalUnits(packetData, decoderConfig)) {
|
||||
if (orderState === HevcNaluOrderState.noMoreDataAllowed) {
|
||||
removedNalUnits.add(loc.offset);
|
||||
continue;
|
||||
}
|
||||
|
||||
const type = extractNalUnitTypeForHevc(packetData[loc.offset]!);
|
||||
|
||||
if (orderState === HevcNaluOrderState.eoBitstreamAllowed && type !== 37 /* EOB_NUT */) {
|
||||
removedNalUnits.add(loc.offset);
|
||||
continue;
|
||||
}
|
||||
|
||||
let remove = false;
|
||||
|
||||
if (type === 35) { // AUD_NUT
|
||||
if (orderState > HevcNaluOrderState.audAllowed) {
|
||||
remove = true;
|
||||
} else {
|
||||
orderState = HevcNaluOrderState.beforeFirstVcl;
|
||||
}
|
||||
} else if (type <= 31) { // VCL (0-31)
|
||||
if (orderState > HevcNaluOrderState.afterFirstVcl) {
|
||||
remove = true;
|
||||
} else {
|
||||
orderState = HevcNaluOrderState.afterFirstVcl;
|
||||
}
|
||||
} else if (type === 36) { // EOS_NUT
|
||||
if (orderState !== HevcNaluOrderState.afterFirstVcl) {
|
||||
remove = true;
|
||||
} else {
|
||||
orderState = HevcNaluOrderState.eoBitstreamAllowed;
|
||||
}
|
||||
} else if (type === 37) { // EOB_NUT
|
||||
if (orderState < HevcNaluOrderState.afterFirstVcl) {
|
||||
remove = true;
|
||||
} else {
|
||||
orderState = HevcNaluOrderState.noMoreDataAllowed;
|
||||
}
|
||||
} else if (
|
||||
type === 32 || type === 33 || type === 34 || type === 39
|
||||
|| (type >= 41 && type <= 44) || (type >= 48 && type <= 55)
|
||||
) { // VPS, SPS, PPS, PREFIX_SEI, RSV_NVCL41..44, UNSPEC48..55
|
||||
if (orderState > HevcNaluOrderState.beforeFirstVcl) {
|
||||
remove = true;
|
||||
} else {
|
||||
orderState = HevcNaluOrderState.beforeFirstVcl;
|
||||
}
|
||||
} else if (
|
||||
type === 38 || type === 40
|
||||
|| (type >= 45 && type <= 47) || (type >= 56 && type <= 63)
|
||||
) { // FD, SUFFIX_SEI, RSV_NVCL45..47, UNSPEC56..63
|
||||
if (orderState < HevcNaluOrderState.afterFirstVcl) {
|
||||
remove = true;
|
||||
}
|
||||
}
|
||||
|
||||
if (remove) {
|
||||
removedNalUnits.add(loc.offset);
|
||||
}
|
||||
}
|
||||
|
||||
// If nothing violated the rules, return null to signal that
|
||||
if (removedNalUnits.size === 0) {
|
||||
return null;
|
||||
}
|
||||
|
||||
const filteredNalUnits: Uint8Array[] = [];
|
||||
for (const loc of iterateHevcNalUnits(packetData, decoderConfig)) {
|
||||
if (!removedNalUnits.has(loc.offset)) {
|
||||
filteredNalUnits.push(packetData.subarray(loc.offset, loc.offset + loc.length));
|
||||
}
|
||||
}
|
||||
|
||||
return concatHevcNalUnits(filteredNalUnits, decoderConfig);
|
||||
};
|
||||
|
||||
export type Vp9CodecInfo = {
|
||||
profile: number;
|
||||
level: number;
|
||||
|
||||
+45
-11
@@ -50,7 +50,15 @@ import {
|
||||
} from './misc';
|
||||
import { Output, OutputTrackGroup, TrackType } from './output';
|
||||
import { Mp4OutputFormat } from './output-format';
|
||||
import { AudioSample, clampCropRectangle, CropRectangle, validateCropRectangle, VideoSample } from './sample';
|
||||
import {
|
||||
AudioSample,
|
||||
audioSampleToInterleavedFormat,
|
||||
clampCropRectangle,
|
||||
CropRectangle,
|
||||
toInterleavedAudioFormat,
|
||||
validateCropRectangle,
|
||||
VideoSample,
|
||||
} from './sample';
|
||||
import { MetadataTags, validateMetadataTags } from './metadata';
|
||||
import { NullTarget } from './target';
|
||||
import { AudioResampler } from './resample';
|
||||
@@ -261,6 +269,13 @@ export type ConversionAudioOptions = {
|
||||
numberOfChannels?: number;
|
||||
/** The desired sample rate of the output audio, in hertz. */
|
||||
sampleRate?: number;
|
||||
/**
|
||||
* The desired sample format (and therefore bit depth) of the audio samples before they are passed to the encoder.
|
||||
* Can be used to control bit depth with certain output codecs such as FLAC.
|
||||
*
|
||||
* Setting this field forces audio transcoding.
|
||||
*/
|
||||
sampleFormat?: 'u8' | 's16' | 's32' | 'f32';
|
||||
/** The desired output audio codec. */
|
||||
codec?: AudioCodec;
|
||||
/** The desired bitrate of the output audio. */
|
||||
@@ -442,6 +457,12 @@ const validateAudioOptions = (audioOptions: ConversionAudioOptions) => {
|
||||
) {
|
||||
throw new TypeError('options.audio.sampleRate, when provided, must be a positive integer.');
|
||||
}
|
||||
if (
|
||||
audioOptions?.sampleFormat !== undefined
|
||||
&& !['u8', 's16', 's32', 'f32'].includes(audioOptions.sampleFormat)
|
||||
) {
|
||||
throw new TypeError('options.audio.sampleFormat, when provided, must be one of: u8, s16, s32, f32.');
|
||||
}
|
||||
if (audioOptions?.process !== undefined && typeof audioOptions.process !== 'function') {
|
||||
throw new TypeError('options.audio.process, when provided, must be a function.');
|
||||
}
|
||||
@@ -1347,7 +1368,7 @@ export class Conversion {
|
||||
timestamp: lastCanvasTimestamp! + i / frameRate,
|
||||
duration: 1 / frameRate,
|
||||
});
|
||||
await this._registerVideoSample(track, trackOptions, outputTrackId, source, sample);
|
||||
await this._registerVideoSample(trackOptions, outputTrackId, source, sample);
|
||||
sample.close();
|
||||
}
|
||||
};
|
||||
@@ -1384,7 +1405,7 @@ export class Conversion {
|
||||
timestamp: adjustedSampleTimestamp,
|
||||
duration: frameRate !== undefined ? 1 / frameRate : duration,
|
||||
});
|
||||
await this._registerVideoSample(track, trackOptions, outputTrackId, source, sample);
|
||||
await this._registerVideoSample(trackOptions, outputTrackId, source, sample);
|
||||
sample.close();
|
||||
|
||||
if (frameRate !== undefined) {
|
||||
@@ -1425,7 +1446,7 @@ export class Conversion {
|
||||
for (let i = 1; i < frameDifference; i++) {
|
||||
lastSample.setTimestamp(lastSampleTimestamp! + i / frameRate);
|
||||
lastSample.setDuration(1 / frameRate);
|
||||
await this._registerVideoSample(track, trackOptions, outputTrackId, source, lastSample);
|
||||
await this._registerVideoSample(trackOptions, outputTrackId, source, lastSample);
|
||||
}
|
||||
|
||||
lastSample.close();
|
||||
@@ -1464,7 +1485,7 @@ export class Conversion {
|
||||
}
|
||||
|
||||
sample.setTimestamp(adjustedSampleTimestamp);
|
||||
await this._registerVideoSample(track, trackOptions, outputTrackId, source, sample);
|
||||
await this._registerVideoSample(trackOptions, outputTrackId, source, sample);
|
||||
|
||||
if (frameRate !== undefined) {
|
||||
lastSample = sample;
|
||||
@@ -1513,7 +1534,6 @@ export class Conversion {
|
||||
|
||||
/** @internal */
|
||||
async _registerVideoSample(
|
||||
track: InputVideoTrack,
|
||||
trackOptions: ConversionVideoOptions,
|
||||
outputTrackId: number,
|
||||
source: VideoSampleSource,
|
||||
@@ -1609,6 +1629,7 @@ export class Conversion {
|
||||
&& audioCodecs.includes(sourceCodec)
|
||||
&& (!trackOptions.codec || trackOptions.codec === sourceCodec)
|
||||
&& !trackOptions.process
|
||||
&& trackOptions.sampleFormat === undefined
|
||||
) {
|
||||
// Fast path, we can simply copy over the encoded packets
|
||||
|
||||
@@ -1745,7 +1766,7 @@ export class Conversion {
|
||||
// Offset the timestamp as needed
|
||||
sample.setTimestamp(sample.timestamp - this._startTimestamp);
|
||||
|
||||
await this._registerAudioSample(track, trackOptions, outputTrackId, source, sample);
|
||||
await this._registerAudioSample(trackOptions, outputTrackId, source, sample);
|
||||
sample.close();
|
||||
}
|
||||
|
||||
@@ -1778,16 +1799,25 @@ export class Conversion {
|
||||
|
||||
/** @internal */
|
||||
async _registerAudioSample(
|
||||
track: InputAudioTrack,
|
||||
trackOptions: ConversionAudioOptions,
|
||||
outputTrackId: number,
|
||||
source: AudioSampleSource,
|
||||
sample: AudioSample,
|
||||
inputSample: AudioSample,
|
||||
) {
|
||||
if (this._canceled) {
|
||||
return;
|
||||
}
|
||||
|
||||
let sample = inputSample;
|
||||
|
||||
if (
|
||||
trackOptions.sampleFormat !== undefined
|
||||
&& toInterleavedAudioFormat(sample.format) !== trackOptions.sampleFormat
|
||||
) {
|
||||
// Do a sample format conversion
|
||||
sample = audioSampleToInterleavedFormat(sample, trackOptions.sampleFormat);
|
||||
}
|
||||
|
||||
this._reportProgress(outputTrackId, sample.timestamp + sample.duration);
|
||||
|
||||
let finalSamples: AudioSample[];
|
||||
@@ -1823,8 +1853,12 @@ export class Conversion {
|
||||
}
|
||||
}
|
||||
} finally {
|
||||
if (sample !== inputSample) {
|
||||
sample.close();
|
||||
}
|
||||
|
||||
for (const finalSample of finalSamples) {
|
||||
if (finalSample !== sample) {
|
||||
if (finalSample !== inputSample) {
|
||||
finalSample.close();
|
||||
}
|
||||
}
|
||||
@@ -1857,7 +1891,7 @@ export class Conversion {
|
||||
onSample: async (sample) => {
|
||||
sample.setTimestamp(sample.timestamp - this._startTimestamp);
|
||||
|
||||
await this._registerAudioSample(track, trackOptions, outputTrackId, source, sample);
|
||||
await this._registerAudioSample(trackOptions, outputTrackId, source, sample);
|
||||
sample.close();
|
||||
},
|
||||
});
|
||||
|
||||
+12
-1
@@ -386,6 +386,11 @@ export type AudioTransformOptions = {
|
||||
numberOfChannels?: number;
|
||||
/** The desired output sample rate in hertz to resample to. */
|
||||
sampleRate?: number;
|
||||
/**
|
||||
* The desired sample format (and therefore bit depth) of the audio samples before they are passed to the encoder.
|
||||
* Can be used to control bit depth with certain output codecs such as FLAC.
|
||||
*/
|
||||
sampleFormat?: 'u8' | 's16' | 's32' | 'f32';
|
||||
/**
|
||||
* Allows for custom user-defined processing of audio samples, e.g. for applying audio effects or timestamp
|
||||
* modifications. Called for each audio sample after resampling and remixing.
|
||||
@@ -406,7 +411,7 @@ export const validateAudioEncodingConfig = (config: AudioEncodingConfig) => {
|
||||
}
|
||||
if (
|
||||
config.bitrate === undefined
|
||||
&& (!(PCM_AUDIO_CODECS as readonly string[]).includes(config.codec) || config.codec === 'flac')
|
||||
&& !((PCM_AUDIO_CODECS as readonly string[]).includes(config.codec) || config.codec === 'flac')
|
||||
) {
|
||||
throw new TypeError('config.bitrate must be provided for compressed audio codecs.');
|
||||
}
|
||||
@@ -433,6 +438,12 @@ export const validateAudioEncodingConfig = (config: AudioEncodingConfig) => {
|
||||
) {
|
||||
throw new TypeError('config.transform.sampleRate, when provided, must be a positive integer.');
|
||||
}
|
||||
if (
|
||||
config.transform.sampleFormat !== undefined
|
||||
&& !['u8', 's16', 's32', 'f32'].includes(config.transform.sampleFormat)
|
||||
) {
|
||||
throw new TypeError('config.transform.sampleFormat, when provided, must be one of: u8, s16, s32, f32.');
|
||||
}
|
||||
if (config.transform.process !== undefined && typeof config.transform.process !== 'function') {
|
||||
throw new TypeError('config.transform.process, when provided, must be a function.');
|
||||
}
|
||||
|
||||
+82
-82
@@ -22,48 +22,48 @@ if ((globalThis as Record<symbol, unknown>)[MEDIABUNNY_LOADED_SYMBOL]) {
|
||||
|
||||
export {
|
||||
Output,
|
||||
OutputOptions,
|
||||
type OutputOptions,
|
||||
OutputTrack,
|
||||
OutputVideoTrack,
|
||||
OutputAudioTrack,
|
||||
OutputSubtitleTrack,
|
||||
OutputTrackGroup,
|
||||
BaseTrackMetadata,
|
||||
VideoTrackMetadata,
|
||||
AudioTrackMetadata,
|
||||
SubtitleTrackMetadata,
|
||||
OutputEvents,
|
||||
type BaseTrackMetadata,
|
||||
type VideoTrackMetadata,
|
||||
type AudioTrackMetadata,
|
||||
type SubtitleTrackMetadata,
|
||||
type OutputEvents,
|
||||
} from './output';
|
||||
export {
|
||||
OutputFormat,
|
||||
AdtsOutputFormat,
|
||||
AdtsOutputFormatOptions,
|
||||
type AdtsOutputFormatOptions,
|
||||
CmafOutputFormat,
|
||||
CmafOutputFormatOptions,
|
||||
type CmafOutputFormatOptions,
|
||||
FlacOutputFormat,
|
||||
FlacOutputFormatOptions,
|
||||
type FlacOutputFormatOptions,
|
||||
HlsOutputFormat,
|
||||
HlsOutputFormatOptions,
|
||||
HlsOutputPlaylistInfo,
|
||||
HlsOutputSegmentInfo,
|
||||
type HlsOutputFormatOptions,
|
||||
type HlsOutputPlaylistInfo,
|
||||
type HlsOutputSegmentInfo,
|
||||
IsobmffOutputFormat,
|
||||
IsobmffOutputFormatOptions,
|
||||
type IsobmffOutputFormatOptions,
|
||||
MkvOutputFormat,
|
||||
MkvOutputFormatOptions,
|
||||
type MkvOutputFormatOptions,
|
||||
MovOutputFormat,
|
||||
Mp3OutputFormat,
|
||||
Mp3OutputFormatOptions,
|
||||
type Mp3OutputFormatOptions,
|
||||
Mp4OutputFormat,
|
||||
MpegTsOutputFormat,
|
||||
MpegTsOutputFormatOptions,
|
||||
type MpegTsOutputFormatOptions,
|
||||
OggOutputFormat,
|
||||
OggOutputFormatOptions,
|
||||
type OggOutputFormatOptions,
|
||||
WavOutputFormat,
|
||||
WavOutputFormatOptions,
|
||||
type WavOutputFormatOptions,
|
||||
WebMOutputFormat,
|
||||
WebMOutputFormatOptions,
|
||||
InclusiveIntegerRange,
|
||||
TrackCountLimits,
|
||||
type WebMOutputFormatOptions,
|
||||
type InclusiveIntegerRange,
|
||||
type TrackCountLimits,
|
||||
} from './output-format';
|
||||
export {
|
||||
MediaSource,
|
||||
@@ -76,17 +76,17 @@ export {
|
||||
EncodedAudioPacketSource,
|
||||
EncodedVideoPacketSource,
|
||||
MediaStreamAudioTrackSource,
|
||||
MediaStreamAudioTrackSourceOptions,
|
||||
type MediaStreamAudioTrackSourceOptions,
|
||||
MediaStreamVideoTrackSource,
|
||||
MediaStreamVideoTrackSourceOptions,
|
||||
type MediaStreamVideoTrackSourceOptions,
|
||||
TextSubtitleSource,
|
||||
VideoSampleSource,
|
||||
} from './media-source';
|
||||
export {
|
||||
MediaCodec,
|
||||
VideoCodec,
|
||||
AudioCodec,
|
||||
SubtitleCodec,
|
||||
type MediaCodec,
|
||||
type VideoCodec,
|
||||
type AudioCodec,
|
||||
type SubtitleCodec,
|
||||
VIDEO_CODECS,
|
||||
AUDIO_CODECS,
|
||||
PCM_AUDIO_CODECS,
|
||||
@@ -102,12 +102,12 @@ export {
|
||||
getDecodableAudioCodecs,
|
||||
} from './decode';
|
||||
export {
|
||||
VideoEncodingConfig,
|
||||
VideoEncodingAdditionalOptions,
|
||||
VideoTransformOptions,
|
||||
AudioEncodingConfig,
|
||||
AudioEncodingAdditionalOptions,
|
||||
AudioTransformOptions,
|
||||
type VideoEncodingConfig,
|
||||
type VideoEncodingAdditionalOptions,
|
||||
type VideoTransformOptions,
|
||||
type AudioEncodingConfig,
|
||||
type AudioEncodingAdditionalOptions,
|
||||
type AudioTransformOptions,
|
||||
canEncode,
|
||||
canEncodeVideo,
|
||||
canEncodeAudio,
|
||||
@@ -128,69 +128,69 @@ export {
|
||||
} from './encode';
|
||||
export {
|
||||
Target,
|
||||
TargetEvents,
|
||||
TargetRequest,
|
||||
type TargetEvents,
|
||||
type TargetRequest,
|
||||
AppendOnlyStreamTarget,
|
||||
BufferTarget,
|
||||
BufferTargetOptions,
|
||||
type BufferTargetOptions,
|
||||
FilePathTarget,
|
||||
FilePathTargetOptions,
|
||||
type FilePathTargetOptions,
|
||||
NullTarget,
|
||||
PathedTarget,
|
||||
RangedTarget,
|
||||
StreamTarget,
|
||||
StreamTargetOptions,
|
||||
StreamTargetChunk,
|
||||
type StreamTargetOptions,
|
||||
type StreamTargetChunk,
|
||||
} from './target';
|
||||
export {
|
||||
AnyIterable,
|
||||
type AnyIterable,
|
||||
ConcurrentRunner,
|
||||
EventEmitter,
|
||||
EventListenerOptions,
|
||||
FilePath,
|
||||
MaybePromise,
|
||||
type EventListenerOptions,
|
||||
type FilePath,
|
||||
type MaybePromise,
|
||||
} from './misc';
|
||||
export {
|
||||
PsshBox,
|
||||
type PsshBox,
|
||||
} from './isobmff/isobmff-misc';
|
||||
export {
|
||||
Rational,
|
||||
Rectangle,
|
||||
Rotation,
|
||||
SetOptional,
|
||||
SetRequired,
|
||||
type Rational,
|
||||
type Rectangle,
|
||||
type Rotation,
|
||||
type SetOptional,
|
||||
type SetRequired,
|
||||
} from './misc';
|
||||
export {
|
||||
TrackType,
|
||||
type TrackType,
|
||||
ALL_TRACK_TYPES,
|
||||
} from './output';
|
||||
export {
|
||||
Source,
|
||||
SourceEvents,
|
||||
type SourceEvents,
|
||||
SourceRef,
|
||||
SourceRequest,
|
||||
type SourceRequest,
|
||||
BlobSource,
|
||||
BlobSourceOptions,
|
||||
type BlobSourceOptions,
|
||||
BufferSource,
|
||||
CustomPathedSource,
|
||||
FilePathSource,
|
||||
FilePathSourceOptions,
|
||||
type FilePathSourceOptions,
|
||||
PathedSource,
|
||||
StreamSource,
|
||||
StreamSourceOptions,
|
||||
type StreamSourceOptions,
|
||||
RangedSource,
|
||||
ReadableStreamSource,
|
||||
ReadableStreamSourceOptions,
|
||||
type ReadableStreamSourceOptions,
|
||||
UrlSource,
|
||||
UrlSourceOptions,
|
||||
type UrlSourceOptions,
|
||||
} from './source';
|
||||
export {
|
||||
InputFormat,
|
||||
InputFormatOptions,
|
||||
type InputFormatOptions,
|
||||
AdtsInputFormat,
|
||||
FlacInputFormat,
|
||||
IsobmffInputFormat,
|
||||
IsobmffInputFormatOptions,
|
||||
type IsobmffInputFormatOptions,
|
||||
HlsInputFormat,
|
||||
MatroskaInputFormat,
|
||||
Mp3InputFormat,
|
||||
@@ -216,38 +216,38 @@ export {
|
||||
} from './input-format';
|
||||
export {
|
||||
Input,
|
||||
InputOptions,
|
||||
InputEvents,
|
||||
type InputOptions,
|
||||
type InputEvents,
|
||||
InputDisposedError,
|
||||
UnsupportedInputFormatError,
|
||||
} from './input';
|
||||
export {
|
||||
DurationMetadataRequestOptions,
|
||||
type DurationMetadataRequestOptions,
|
||||
} from './demuxer';
|
||||
export {
|
||||
InputTrack,
|
||||
InputVideoTrack,
|
||||
InputAudioTrack,
|
||||
InputTrackQuery,
|
||||
PacketStats,
|
||||
type InputTrackQuery,
|
||||
type PacketStats,
|
||||
asc,
|
||||
desc,
|
||||
prefer,
|
||||
} from './input-track';
|
||||
export {
|
||||
EncodedPacket,
|
||||
EncodedPacketSideData,
|
||||
PacketType,
|
||||
type EncodedPacketSideData,
|
||||
type PacketType,
|
||||
} from './packet';
|
||||
export {
|
||||
AudioSample,
|
||||
AudioSampleInit,
|
||||
AudioSampleCopyToOptions,
|
||||
type AudioSampleInit,
|
||||
type AudioSampleCopyToOptions,
|
||||
VideoSample,
|
||||
VideoSampleInit,
|
||||
VideoSamplePixelFormat,
|
||||
type VideoSampleInit,
|
||||
type VideoSamplePixelFormat,
|
||||
VideoSampleColorSpace,
|
||||
CropRectangle,
|
||||
type CropRectangle,
|
||||
VIDEO_SAMPLE_PIXEL_FORMATS,
|
||||
} from './sample';
|
||||
export {
|
||||
@@ -255,20 +255,20 @@ export {
|
||||
AudioSampleSink,
|
||||
BaseMediaSampleSink,
|
||||
CanvasSink,
|
||||
CanvasSinkOptions,
|
||||
type CanvasSinkOptions,
|
||||
EncodedPacketSink,
|
||||
PacketRetrievalOptions,
|
||||
type PacketRetrievalOptions,
|
||||
VideoSampleSink,
|
||||
WrappedAudioBuffer,
|
||||
WrappedCanvas,
|
||||
type WrappedAudioBuffer,
|
||||
type WrappedCanvas,
|
||||
} from './media-sink';
|
||||
export {
|
||||
Conversion,
|
||||
ConversionOptions,
|
||||
ConversionVideoOptions,
|
||||
ConversionAudioOptions,
|
||||
type ConversionOptions,
|
||||
type ConversionVideoOptions,
|
||||
type ConversionAudioOptions,
|
||||
ConversionCanceledError,
|
||||
DiscardedTrack,
|
||||
type DiscardedTrack,
|
||||
} from './conversion';
|
||||
export {
|
||||
CustomVideoDecoder,
|
||||
@@ -279,11 +279,11 @@ export {
|
||||
registerEncoder,
|
||||
} from './custom-coder';
|
||||
export {
|
||||
MetadataTags,
|
||||
AttachedImage,
|
||||
type MetadataTags,
|
||||
type AttachedImage,
|
||||
RichImageData,
|
||||
AttachedFile,
|
||||
TrackDisposition,
|
||||
type TrackDisposition,
|
||||
} from './metadata';
|
||||
|
||||
// 🐡🦔
|
||||
|
||||
@@ -174,6 +174,12 @@ const u64 = (value: number) => {
|
||||
return [bytes[0], bytes[1], bytes[2], bytes[3], bytes[4], bytes[5], bytes[6], bytes[7]] as number[];
|
||||
};
|
||||
|
||||
const i64 = (value: number) => {
|
||||
view.setInt32(0, Math.floor(value / 2 ** 32), false);
|
||||
view.setUint32(4, value, false);
|
||||
return [bytes[0], bytes[1], bytes[2], bytes[3], bytes[4], bytes[5], bytes[6], bytes[7]] as number[];
|
||||
};
|
||||
|
||||
const fixed_8_8 = (value: number) => {
|
||||
view.setInt16(0, 2 ** 8 * value, false);
|
||||
return [bytes[0], bytes[1]] as number[];
|
||||
@@ -384,11 +390,14 @@ export const mvhd = (
|
||||
creationTime: number,
|
||||
trackDatas: IsobmffTrackData[],
|
||||
) => {
|
||||
const duration = intoTimescale(Math.max(
|
||||
const duration = Math.max(
|
||||
0,
|
||||
...trackDatas
|
||||
.map(x => presentationSpan(x)),
|
||||
), GLOBAL_TIMESCALE);
|
||||
.map(trackData => (
|
||||
intoTimescale(presentationSpan(trackData), GLOBAL_TIMESCALE)
|
||||
+ intoTimescale(trackData.startTimestampOffset ?? 0, GLOBAL_TIMESCALE)
|
||||
)),
|
||||
);
|
||||
const nextTrackId = Math.max(0, ...trackDatas.map(x => x.track.id)) + 1;
|
||||
|
||||
// Conditionally use u64 if u32 isn't enough
|
||||
@@ -417,7 +426,9 @@ const presentationSpan = (trackData: IsobmffTrackData) => {
|
||||
let minTimestamp = Infinity;
|
||||
let maxEndTimestamp = -Infinity;
|
||||
|
||||
for (const sample of trackData.samples) {
|
||||
for (let i = 0; i < trackData.samples.length; i++) {
|
||||
const sample = trackData.samples[i]!;
|
||||
|
||||
if (sample.timestamp < minTimestamp) {
|
||||
minTimestamp = sample.timestamp;
|
||||
}
|
||||
@@ -440,9 +451,11 @@ const presentationSpan = (trackData: IsobmffTrackData) => {
|
||||
*/
|
||||
export const trak = (trackData: IsobmffTrackData, creationTime: number) => {
|
||||
const trackMetadata = getTrackMetadata(trackData);
|
||||
const needsEditList = trackData.startTimestampOffset !== null && trackData.startTimestampOffset > 0;
|
||||
|
||||
return box('trak', undefined, [
|
||||
tkhd(trackData, creationTime),
|
||||
needsEditList ? edts(trackData, trackData.startTimestampOffset!) : null,
|
||||
mdia(trackData, creationTime),
|
||||
trackMetadata.name !== undefined
|
||||
? box('udta', undefined, [
|
||||
@@ -459,10 +472,8 @@ export const tkhd = (
|
||||
trackData: IsobmffTrackData,
|
||||
creationTime: number,
|
||||
) => {
|
||||
const durationInGlobalTimescale = intoTimescale(
|
||||
presentationSpan(trackData),
|
||||
GLOBAL_TIMESCALE,
|
||||
);
|
||||
const durationInGlobalTimescale = intoTimescale(presentationSpan(trackData), GLOBAL_TIMESCALE)
|
||||
+ intoTimescale(trackData.startTimestampOffset ?? 0, GLOBAL_TIMESCALE);
|
||||
|
||||
const needsU64 = !isU32(creationTime) || !isU32(durationInGlobalTimescale);
|
||||
const u32OrU64 = needsU64 ? u64 : u32;
|
||||
@@ -497,6 +508,32 @@ export const tkhd = (
|
||||
]);
|
||||
};
|
||||
|
||||
/** Edit Box: Specifies edits to the track's media. */
|
||||
export const edts = (trackData: IsobmffTrackData, offset: number) => {
|
||||
const startOffset = intoTimescale(offset, GLOBAL_TIMESCALE);
|
||||
const mediaDuration = intoTimescale(presentationSpan(trackData), GLOBAL_TIMESCALE);
|
||||
|
||||
const needs64Bits = !isU32(startOffset) || !isU32(mediaDuration);
|
||||
const u32OrU64 = needs64Bits ? u64 : u32;
|
||||
const i32OrI64 = needs64Bits ? i64 : i32;
|
||||
|
||||
return box('edts', undefined, [
|
||||
fullBox('elst', needs64Bits ? 1 : 0, 0, [
|
||||
u32(2), // Entry count
|
||||
|
||||
// #1
|
||||
u32OrU64(startOffset), // Segment duration
|
||||
i32OrI64(-1), // Media time
|
||||
fixed_16_16(1), // Media rate
|
||||
|
||||
// #2
|
||||
u32OrU64(mediaDuration), // Segment duration
|
||||
i32OrI64(0), // Media time
|
||||
fixed_16_16(1), // Media rate
|
||||
]),
|
||||
]);
|
||||
};
|
||||
|
||||
/** Media Box: Describes and define a track's media type and sample data. */
|
||||
export const mdia = (trackData: IsobmffTrackData, creationTime: number) => box('mdia', undefined, [
|
||||
mdhd(trackData, creationTime),
|
||||
@@ -509,6 +546,7 @@ export const mdhd = (
|
||||
trackData: IsobmffTrackData,
|
||||
creationTime: number,
|
||||
) => {
|
||||
// Since the duration represents the raw media duration, edit list offsets are not taken into account here
|
||||
const localDuration = intoTimescale(
|
||||
presentationSpan(trackData),
|
||||
trackData.timescale,
|
||||
|
||||
@@ -52,7 +52,7 @@ import {
|
||||
import { buildIsobmffMimeType } from './isobmff-misc';
|
||||
import { MAX_BOX_HEADER_SIZE, MIN_BOX_HEADER_SIZE } from './isobmff-reader';
|
||||
|
||||
export const GLOBAL_TIMESCALE = 1000;
|
||||
export const GLOBAL_TIMESCALE = 57600; // LCM of a bunch of common frame rates (24, 25, 30, 60, 144, ...)
|
||||
const TIMESTAMP_OFFSET = 2_082_844_800; // Seconds between Jan 1 1904 and Jan 1 1970
|
||||
|
||||
export type Sample = {
|
||||
@@ -85,6 +85,7 @@ export type IsobmffTrackData = {
|
||||
compositionTimeOffsetTable: { sampleCount: number; sampleCompositionTimeOffset: number }[];
|
||||
lastTimescaleUnits: number | null;
|
||||
lastSample: Sample | null;
|
||||
startTimestampOffset: number | null;
|
||||
|
||||
finalizedChunks: Chunk[];
|
||||
currentChunk: Chunk | null;
|
||||
@@ -120,6 +121,7 @@ export type IsobmffTrackData = {
|
||||
* Some players expect this for PCM audio.
|
||||
*/
|
||||
requiresPcmTransformation: boolean;
|
||||
expectedNextPcmPacketTimestamp: number | null;
|
||||
/**
|
||||
* The "ADTS stripping" involves removing the ADTS header from each AAC packet. SOBMFF stores raw AAC data, not
|
||||
* ADTS-wrapped data.
|
||||
@@ -395,7 +397,10 @@ export class IsobmffMuxer extends Muxer {
|
||||
// The frame rate set by the user may not be an integer. Since timescale is an integer, we'll approximate the
|
||||
// frame time (inverse of frame rate) with a rational number, then use that approximation's denominator
|
||||
// as the timescale.
|
||||
const timescale = computeRationalApproximation(1 / (track.metadata.frameRate ?? 57600), 1e6).denominator;
|
||||
const timescale = computeRationalApproximation(
|
||||
1 / (track.metadata.frameRate ?? GLOBAL_TIMESCALE),
|
||||
1e6,
|
||||
).denominator;
|
||||
|
||||
const displayAspectWidth = decoderConfig.displayAspectWidth;
|
||||
const displayAspectHeight = decoderConfig.displayAspectHeight;
|
||||
@@ -425,6 +430,7 @@ export class IsobmffMuxer extends Muxer {
|
||||
compositionTimeOffsetTable: [],
|
||||
lastTimescaleUnits: null,
|
||||
lastSample: null,
|
||||
startTimestampOffset: null,
|
||||
finalizedChunks: [],
|
||||
currentChunk: null,
|
||||
compactlyCodedChunkTable: [],
|
||||
@@ -494,6 +500,7 @@ export class IsobmffMuxer extends Muxer {
|
||||
requiresPcmTransformation:
|
||||
!this.isFragmented
|
||||
&& (PCM_AUDIO_CODECS as readonly string[]).includes(track.source._codec),
|
||||
expectedNextPcmPacketTimestamp: null,
|
||||
requiresAdtsStripping,
|
||||
firstPacket: packet,
|
||||
},
|
||||
@@ -505,6 +512,7 @@ export class IsobmffMuxer extends Muxer {
|
||||
compositionTimeOffsetTable: [],
|
||||
lastTimescaleUnits: null,
|
||||
lastSample: null,
|
||||
startTimestampOffset: null,
|
||||
finalizedChunks: [],
|
||||
currentChunk: null,
|
||||
compactlyCodedChunkTable: [],
|
||||
@@ -547,6 +555,7 @@ export class IsobmffMuxer extends Muxer {
|
||||
compositionTimeOffsetTable: [],
|
||||
lastTimescaleUnits: null,
|
||||
lastSample: null,
|
||||
startTimestampOffset: null,
|
||||
finalizedChunks: [],
|
||||
currentChunk: null,
|
||||
compactlyCodedChunkTable: [],
|
||||
@@ -629,41 +638,59 @@ export class IsobmffMuxer extends Muxer {
|
||||
packetData = packetData.subarray(headerLength);
|
||||
}
|
||||
|
||||
const timestamp = this.validateAndNormalizeTimestamp(
|
||||
let timestamp = this.validateAndNormalizeTimestamp(
|
||||
trackData.track,
|
||||
packet.timestamp,
|
||||
packet.type === 'key',
|
||||
);
|
||||
let duration = packet.duration;
|
||||
|
||||
if (trackData.info.requiresPcmTransformation) {
|
||||
// Packets may have only approximate timestamp/duration information, but for our PCM logic, we need it
|
||||
// to be precise. So here, we refine the values.
|
||||
|
||||
const pcmInfo = parsePcmCodec(
|
||||
trackData.info.decoderConfig.codec as PcmAudioCodec,
|
||||
);
|
||||
const frameSize = pcmInfo.sampleSize * trackData.info.numberOfChannels;
|
||||
|
||||
// Compute the precise duration
|
||||
duration = packetData.byteLength / frameSize / trackData.info.sampleRate;
|
||||
|
||||
if (trackData.info.expectedNextPcmPacketTimestamp !== null) {
|
||||
const diff = timestamp - trackData.info.expectedNextPcmPacketTimestamp;
|
||||
if (diff < 0.01) {
|
||||
timestamp = trackData.info.expectedNextPcmPacketTimestamp;
|
||||
} else {
|
||||
const paddedDuration = await this.padWithSilence(
|
||||
trackData,
|
||||
trackData.info.expectedNextPcmPacketTimestamp,
|
||||
diff,
|
||||
);
|
||||
timestamp = trackData.info.expectedNextPcmPacketTimestamp + paddedDuration;
|
||||
}
|
||||
}
|
||||
|
||||
trackData.info.expectedNextPcmPacketTimestamp = timestamp + duration;
|
||||
}
|
||||
|
||||
const internalSample = this.createSampleForTrack(
|
||||
trackData,
|
||||
packetData,
|
||||
timestamp,
|
||||
packet.duration,
|
||||
duration,
|
||||
packet.type,
|
||||
);
|
||||
|
||||
if (trackData.info.requiresPcmTransformation) {
|
||||
await this.maybePadWithSilence(trackData, timestamp);
|
||||
}
|
||||
|
||||
await this.registerSample(trackData, internalSample);
|
||||
} finally {
|
||||
release();
|
||||
}
|
||||
}
|
||||
|
||||
private async maybePadWithSilence(trackData: IsobmffAudioTrackData, untilTimestamp: number) {
|
||||
// The PCM transformation assumes that all samples are contiguous. This is not something that is enforced, so
|
||||
// we need to pad the "holes" in between samples (and before the first sample) with additional
|
||||
// "silence samples".
|
||||
|
||||
const lastSample = last(trackData.samples);
|
||||
const lastEndTimestamp = lastSample
|
||||
? lastSample.timestamp + lastSample.duration
|
||||
: 0;
|
||||
|
||||
const delta = untilTimestamp - lastEndTimestamp;
|
||||
const deltaInTimescale = intoTimescale(delta, trackData.timescale);
|
||||
private async padWithSilence(trackData: IsobmffAudioTrackData, timestamp: number, duration: number) {
|
||||
const deltaInTimescale = intoTimescale(duration, trackData.timescale);
|
||||
duration = deltaInTimescale / trackData.timescale;
|
||||
|
||||
if (deltaInTimescale > 0) {
|
||||
const { sampleSize, silentValue } = parsePcmCodec(
|
||||
@@ -675,12 +702,14 @@ export class IsobmffMuxer extends Muxer {
|
||||
const paddingSample = this.createSampleForTrack(
|
||||
trackData,
|
||||
new Uint8Array(data.buffer),
|
||||
lastEndTimestamp,
|
||||
delta,
|
||||
timestamp,
|
||||
duration,
|
||||
'key',
|
||||
);
|
||||
await this.registerSample(trackData, paddingSample);
|
||||
}
|
||||
|
||||
return duration;
|
||||
}
|
||||
|
||||
async addSubtitleCue(track: OutputSubtitleTrack, cue: SubtitleCue, meta?: SubtitleMetadata) {
|
||||
@@ -821,6 +850,11 @@ export class IsobmffMuxer extends Muxer {
|
||||
}
|
||||
|
||||
if (trackData.type === 'audio' && trackData.info.requiresPcmTransformation) {
|
||||
if (!this.isFragmented) {
|
||||
// The first timestamp is the lowest
|
||||
trackData.startTimestampOffset ??= trackData.timestampProcessingQueue[0]!.timestamp;
|
||||
}
|
||||
|
||||
let totalDuration = 0;
|
||||
|
||||
// Compute the total duration in the track timescale (which is equal to the amount of PCM audio samples)
|
||||
@@ -848,6 +882,10 @@ export class IsobmffMuxer extends Muxer {
|
||||
|
||||
const sortedTimestamps = trackData.timestampProcessingQueue.map(x => x.timestamp).sort((a, b) => a - b);
|
||||
|
||||
if (!this.isFragmented) {
|
||||
trackData.startTimestampOffset ??= sortedTimestamps[0]!;
|
||||
}
|
||||
|
||||
for (let i = 0; i < trackData.timestampProcessingQueue.length; i++) {
|
||||
const sample = trackData.timestampProcessingQueue[i]!;
|
||||
|
||||
@@ -857,12 +895,6 @@ export class IsobmffMuxer extends Muxer {
|
||||
// model it.
|
||||
sample.decodeTimestamp = sortedTimestamps[i]!;
|
||||
|
||||
if (!this.isFragmented && trackData.lastTimescaleUnits === null) {
|
||||
// In non-fragmented files, the first decode timestamp is always zero. If the first presentation
|
||||
// timestamp isn't zero, we'll simply use the composition time offset to achieve it.
|
||||
sample.decodeTimestamp = 0;
|
||||
}
|
||||
|
||||
const sampleCompositionTimeOffset
|
||||
= intoTimescale(sample.timestamp - sample.decodeTimestamp, trackData.timescale);
|
||||
const durationInTimescale = intoTimescale(sample.duration, trackData.timescale);
|
||||
@@ -1377,6 +1409,17 @@ export class IsobmffMuxer extends Muxer {
|
||||
} else {
|
||||
for (const trackData of this.trackDatas) {
|
||||
await this.finalizeCurrentChunk(trackData);
|
||||
|
||||
// Must hold because we will have processed at least one sample
|
||||
assert(trackData.startTimestampOffset !== null);
|
||||
|
||||
// Shift all of the samples by the start offset. We'll then write out an edit list that will shift them
|
||||
// back to their proper spot in the composition.
|
||||
for (let i = 0; i < trackData.samples.length; i++) {
|
||||
const sample = trackData.samples[i]!;
|
||||
sample.timestamp -= trackData.startTimestampOffset;
|
||||
sample.decodeTimestamp -= trackData.startTimestampOffset;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
+20
-11
@@ -17,6 +17,7 @@ import {
|
||||
iterateAvcNalUnits,
|
||||
iterateHevcNalUnits,
|
||||
parseAvcSps,
|
||||
sanitizeHevcPacketForChromium,
|
||||
} from './codec-data';
|
||||
import { CustomVideoDecoder, customVideoDecoders, CustomAudioDecoder, customAudioDecoders } from './custom-coder';
|
||||
import { InputDisposedError } from './input';
|
||||
@@ -980,20 +981,28 @@ class VideoDecoderWrapper extends DecoderWrapper<VideoSample> {
|
||||
insertSorted(this.inputTimestamps, packet.timestamp, x => x);
|
||||
}
|
||||
|
||||
// Workaround for https://issues.chromium.org/issues/470109459
|
||||
if (isChromium() && this.currentPacketIndex === 0 && this.codec === 'avc') {
|
||||
const filteredNalUnits: Uint8Array[] = [];
|
||||
if (isChromium() && this.currentPacketIndex === 0) {
|
||||
if (this.codec === 'avc') {
|
||||
// Workaround for https://issues.chromium.org/issues/470109459
|
||||
const filteredNalUnits: Uint8Array[] = [];
|
||||
|
||||
for (const loc of iterateAvcNalUnits(packet.data, this.decoderConfig)) {
|
||||
const type = extractNalUnitTypeForAvc(packet.data[loc.offset]!);
|
||||
// These trip up Chromium's key frame detection, so let's strip them
|
||||
if (!(type >= 20 && type <= 31)) {
|
||||
filteredNalUnits.push(packet.data.subarray(loc.offset, loc.offset + loc.length));
|
||||
for (const loc of iterateAvcNalUnits(packet.data, this.decoderConfig)) {
|
||||
const type = extractNalUnitTypeForAvc(packet.data[loc.offset]!);
|
||||
// These trip up Chromium's key frame detection, so let's strip them
|
||||
if (!(type >= 20 && type <= 31)) {
|
||||
filteredNalUnits.push(packet.data.subarray(loc.offset, loc.offset + loc.length));
|
||||
}
|
||||
}
|
||||
|
||||
const newData = concatAvcNalUnits(filteredNalUnits, this.decoderConfig);
|
||||
packet = new EncodedPacket(newData, packet.type, packet.timestamp, packet.duration);
|
||||
} else if (this.codec === 'hevc') {
|
||||
// Workaround for https://issues.chromium.org/issues/507611247
|
||||
const sanitizedData = sanitizeHevcPacketForChromium(packet.data, this.decoderConfig);
|
||||
if (sanitizedData) {
|
||||
packet = new EncodedPacket(sanitizedData, packet.type, packet.timestamp, packet.duration);
|
||||
}
|
||||
}
|
||||
|
||||
const newData = concatAvcNalUnits(filteredNalUnits, this.decoderConfig);
|
||||
packet = new EncodedPacket(newData, packet.type, packet.timestamp, packet.duration);
|
||||
}
|
||||
|
||||
this.decoder.decode(packet.toEncodedVideoChunk());
|
||||
|
||||
+26
-1
@@ -50,7 +50,13 @@ import {
|
||||
customAudioEncoders,
|
||||
} from './custom-coder';
|
||||
import { EncodedPacket, EncodedPacketSideData } from './packet';
|
||||
import { AudioSample, clampCropRectangle, VideoSample } from './sample';
|
||||
import {
|
||||
AudioSample,
|
||||
audioSampleToInterleavedFormat,
|
||||
clampCropRectangle,
|
||||
toInterleavedAudioFormat,
|
||||
VideoSample,
|
||||
} from './sample';
|
||||
import {
|
||||
AudioEncodingConfig,
|
||||
buildAudioEncoderConfig,
|
||||
@@ -1888,6 +1894,21 @@ class AudioEncoderWrapper {
|
||||
private async processAndEncode(audioSample: AudioSample, shouldClose: boolean) {
|
||||
const config = this.encodingConfig;
|
||||
|
||||
if (
|
||||
config.transform?.sampleFormat !== undefined
|
||||
&& toInterleavedAudioFormat(audioSample.format) !== config.transform.sampleFormat
|
||||
) {
|
||||
// Do a sample format conversion
|
||||
const newSample = audioSampleToInterleavedFormat(audioSample, config.transform.sampleFormat);
|
||||
|
||||
if (shouldClose) {
|
||||
audioSample.close();
|
||||
}
|
||||
|
||||
audioSample = newSample;
|
||||
shouldClose = true;
|
||||
}
|
||||
|
||||
if (config.transform?.process) {
|
||||
let processed = config.transform.process(audioSample);
|
||||
if (processed instanceof Promise) {
|
||||
@@ -1910,6 +1931,10 @@ class AudioEncoderWrapper {
|
||||
}
|
||||
await this.encodeSample(sample, true);
|
||||
}
|
||||
|
||||
if (shouldClose) {
|
||||
audioSample.close();
|
||||
}
|
||||
} else {
|
||||
await this.encodeSample(audioSample, shouldClose);
|
||||
}
|
||||
|
||||
@@ -1953,6 +1953,21 @@ const isAudioData = (x: unknown): x is AudioData => {
|
||||
return typeof AudioData !== 'undefined' && x instanceof AudioData;
|
||||
};
|
||||
|
||||
export const toInterleavedAudioFormat = (format: AudioSampleFormat): 'u8' | 's16' | 's32' | 'f32' => {
|
||||
switch (format) {
|
||||
case 'u8-planar':
|
||||
return 'u8';
|
||||
case 's16-planar':
|
||||
return 's16';
|
||||
case 's32-planar':
|
||||
return 's32';
|
||||
case 'f32-planar':
|
||||
return 'f32';
|
||||
default:
|
||||
return format;
|
||||
}
|
||||
};
|
||||
|
||||
/**
|
||||
* WebKit has a bug where calling AudioData.copyTo with a format different from the source format
|
||||
* crashes the tab when there are more than 2 channels. This function works around that by always
|
||||
@@ -2061,3 +2076,18 @@ const doAudioDataCopyToWebKitWorkaround = (
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
export const audioSampleToInterleavedFormat = (sample: AudioSample, format: 'u8' | 's16' | 's32' | 'f32') => {
|
||||
const size = sample.allocationSize({ format, planeIndex: 0 });
|
||||
const buffer = new ArrayBuffer(size);
|
||||
sample.copyTo(buffer, { format, planeIndex: 0 });
|
||||
|
||||
return new AudioSample({
|
||||
data: buffer,
|
||||
format,
|
||||
numberOfChannels: sample.numberOfChannels,
|
||||
sampleRate: sample.sampleRate,
|
||||
timestamp: sample.timestamp,
|
||||
duration: sample.duration,
|
||||
});
|
||||
};
|
||||
|
||||
@@ -0,0 +1,107 @@
|
||||
import { expect, test } from 'vitest';
|
||||
import { Input } from '../../src/input.js';
|
||||
import { ALL_FORMATS } from '../../src/input-format.js';
|
||||
import { AudioSampleSource } from '../../src/media-source.js';
|
||||
import { AudioSampleSink } from '../../src/media-sink.js';
|
||||
import { assert } from '../../src/misc.js';
|
||||
import { Output } from '../../src/output.js';
|
||||
import { FlacOutputFormat } from '../../src/output-format.js';
|
||||
import { AudioSample } from '../../src/sample.js';
|
||||
import { BufferSource } from '../../src/source.js';
|
||||
import { BufferTarget } from '../../src/target.js';
|
||||
import { registerFlacEncoder } from '@mediabunny/flac-encoder';
|
||||
|
||||
test('FLAC encoder, 24-bit', async () => {
|
||||
registerFlacEncoder();
|
||||
|
||||
const sampleRate = 48000;
|
||||
const channels = 2;
|
||||
const durationSeconds = 2;
|
||||
const data = createF32SineWave(sampleRate, channels, durationSeconds);
|
||||
|
||||
using sample = await encodeAndDecodeFirstSample(new AudioSample({
|
||||
data,
|
||||
format: 'f32',
|
||||
numberOfChannels: channels,
|
||||
sampleRate,
|
||||
timestamp: 0,
|
||||
}));
|
||||
|
||||
expect(sample.format).toBe('s32');
|
||||
});
|
||||
|
||||
test('FLAC encoder, 16-bit', async () => {
|
||||
registerFlacEncoder();
|
||||
|
||||
const sampleRate = 48000;
|
||||
const channels = 2;
|
||||
const durationSeconds = 2;
|
||||
const data = createS16SineWave(sampleRate, channels, durationSeconds);
|
||||
|
||||
using sample = await encodeAndDecodeFirstSample(new AudioSample({
|
||||
data,
|
||||
format: 's16',
|
||||
numberOfChannels: channels,
|
||||
sampleRate,
|
||||
timestamp: 0,
|
||||
}));
|
||||
|
||||
expect(sample.format).toBe('s16');
|
||||
});
|
||||
|
||||
const createF32SineWave = (sampleRate: number, channels: number, durationSeconds: number) => {
|
||||
const totalFrames = sampleRate * durationSeconds;
|
||||
const data = new Float32Array(totalFrames * channels);
|
||||
|
||||
for (let i = 0; i < totalFrames; i++) {
|
||||
const value = Math.sin(2 * Math.PI * 440 * i / sampleRate);
|
||||
for (let ch = 0; ch < channels; ch++) {
|
||||
data[i * channels + ch] = value;
|
||||
}
|
||||
}
|
||||
|
||||
return data;
|
||||
};
|
||||
|
||||
const createS16SineWave = (sampleRate: number, channels: number, durationSeconds: number) => {
|
||||
const totalFrames = sampleRate * durationSeconds;
|
||||
const data = new Int16Array(totalFrames * channels);
|
||||
|
||||
for (let i = 0; i < totalFrames; i++) {
|
||||
const value = Math.round(Math.sin(2 * Math.PI * 440 * i / sampleRate) * 32767);
|
||||
for (let ch = 0; ch < channels; ch++) {
|
||||
data[i * channels + ch] = value;
|
||||
}
|
||||
}
|
||||
|
||||
return data;
|
||||
};
|
||||
|
||||
const encodeAndDecodeFirstSample = async (audioSample: AudioSample) => {
|
||||
const output = new Output({
|
||||
format: new FlacOutputFormat(),
|
||||
target: new BufferTarget(),
|
||||
});
|
||||
|
||||
const audioSource = new AudioSampleSource({ codec: 'flac' });
|
||||
output.addAudioTrack(audioSource);
|
||||
|
||||
await output.start();
|
||||
await audioSource.add(audioSample);
|
||||
audioSource.close();
|
||||
await output.finalize();
|
||||
|
||||
using input = new Input({
|
||||
source: new BufferSource(output.target.buffer!),
|
||||
formats: ALL_FORMATS,
|
||||
});
|
||||
|
||||
const track = await input.getPrimaryAudioTrack();
|
||||
assert(track);
|
||||
|
||||
const sink = new AudioSampleSink(track);
|
||||
const sample = await sink.getSample(0);
|
||||
assert(sample);
|
||||
|
||||
return sample;
|
||||
};
|
||||
@@ -47,7 +47,7 @@ test('FLAC encoding', async () => {
|
||||
target: new BufferTarget(),
|
||||
});
|
||||
|
||||
const audioSource = new AudioSampleSource({ codec: 'flac', bitrate: 1 });
|
||||
const audioSource = new AudioSampleSource({ codec: 'flac' });
|
||||
output.addAudioTrack(audioSource);
|
||||
|
||||
await output.start();
|
||||
@@ -97,7 +97,7 @@ test('FLAC with huge timestamps', async () => {
|
||||
target: new BufferTarget(),
|
||||
});
|
||||
|
||||
const audioSource = new AudioSampleSource({ codec: 'flac', bitrate: 1 });
|
||||
const audioSource = new AudioSampleSource({ codec: 'flac' });
|
||||
output.addAudioTrack(audioSource);
|
||||
|
||||
await output.start();
|
||||
|
||||
@@ -9,6 +9,8 @@ import { BufferTarget } from '../../src/target.js';
|
||||
import { Mp4OutputFormat } from '../../src/output-format.js';
|
||||
import { Conversion } from '../../src/conversion.js';
|
||||
import { assert } from '../../src/misc.js';
|
||||
import { EncodedAudioPacketSource, EncodedVideoPacketSource } from '../../src/media-source.js';
|
||||
import { EncodedPacket } from '../../src/packet.js';
|
||||
|
||||
const __dirname = new URL('.', import.meta.url).pathname;
|
||||
|
||||
@@ -104,3 +106,249 @@ test('Fragmented fMP4 with video+audio preserves B-frame CTS', async () => {
|
||||
|
||||
expect(timestamps).toEqual(originalTimestamps);
|
||||
});
|
||||
|
||||
test('Zero start timestamp, regular MP4', async () => {
|
||||
const output = new Output({
|
||||
format: new Mp4OutputFormat(),
|
||||
target: new BufferTarget(),
|
||||
});
|
||||
|
||||
const source = new EncodedVideoPacketSource('vp8');
|
||||
output.addVideoTrack(source);
|
||||
|
||||
await output.start();
|
||||
|
||||
const meta = { decoderConfig: { codec: 'vp8', codedWidth: 1280, codedHeight: 720 } };
|
||||
|
||||
await source.add(new EncodedPacket(new Uint8Array(1024), 'key', 0, 0.1), meta);
|
||||
await source.add(new EncodedPacket(new Uint8Array(1024), 'delta', 0.1, 0.1));
|
||||
await source.add(new EncodedPacket(new Uint8Array(1024), 'delta', 0.2, 0.1));
|
||||
await source.add(new EncodedPacket(new Uint8Array(1024), 'delta', 0.3, 0.1));
|
||||
|
||||
await output.finalize();
|
||||
|
||||
// Hacky but works
|
||||
const str = String.fromCharCode(...new Uint8Array(output.target.buffer!));
|
||||
expect(str.includes('edts') || str.includes('elst')).toBe(false);
|
||||
});
|
||||
|
||||
test('Non-zero start timestamp, regular MP4', async () => {
|
||||
const output = new Output({
|
||||
format: new Mp4OutputFormat(),
|
||||
target: new BufferTarget(),
|
||||
});
|
||||
|
||||
const source = new EncodedVideoPacketSource('vp8');
|
||||
output.addVideoTrack(source);
|
||||
|
||||
await output.start();
|
||||
|
||||
const meta = { decoderConfig: { codec: 'vp8', codedWidth: 1280, codedHeight: 720 } };
|
||||
|
||||
await source.add(new EncodedPacket(new Uint8Array(1024), 'key', 1, 0.1), meta);
|
||||
await source.add(new EncodedPacket(new Uint8Array(1024), 'delta', 1.1, 0.1));
|
||||
await source.add(new EncodedPacket(new Uint8Array(1024), 'delta', 1.2, 0.1));
|
||||
await source.add(new EncodedPacket(new Uint8Array(1024), 'delta', 1.3, 0.1));
|
||||
|
||||
await output.finalize();
|
||||
|
||||
// Hacky but works
|
||||
const str = String.fromCharCode(...new Uint8Array(output.target.buffer!));
|
||||
expect(str.includes('edts') && str.includes('elst')).toBe(true);
|
||||
|
||||
using input = new Input({
|
||||
source: new BufferSource(output.target.buffer!),
|
||||
formats: ALL_FORMATS,
|
||||
});
|
||||
|
||||
const track = await input.getPrimaryVideoTrack();
|
||||
assert(track);
|
||||
const sink = new EncodedPacketSink(track);
|
||||
|
||||
const timestamps: number[] = [];
|
||||
const durations: number[] = [];
|
||||
for await (const packet of sink.packets()) {
|
||||
timestamps.push(packet.timestamp);
|
||||
durations.push(packet.duration);
|
||||
}
|
||||
|
||||
expect(timestamps).toEqual([1, 1.1, 1.2, 1.3]);
|
||||
expect(durations).toEqual([0.1, 0.1, 0.1, 0.1]);
|
||||
});
|
||||
|
||||
test('Non-zero start timestamp, fragmented MP4', async () => {
|
||||
const output = new Output({
|
||||
format: new Mp4OutputFormat({ fastStart: 'fragmented' }),
|
||||
target: new BufferTarget(),
|
||||
});
|
||||
|
||||
const source = new EncodedVideoPacketSource('vp8');
|
||||
output.addVideoTrack(source);
|
||||
|
||||
await output.start();
|
||||
|
||||
const meta = { decoderConfig: { codec: 'vp8', codedWidth: 1280, codedHeight: 720 } };
|
||||
|
||||
await source.add(new EncodedPacket(new Uint8Array(1024), 'key', 1, 0.1), meta);
|
||||
await source.add(new EncodedPacket(new Uint8Array(1024), 'delta', 1.1, 0.1));
|
||||
await source.add(new EncodedPacket(new Uint8Array(1024), 'delta', 1.2, 0.1));
|
||||
await source.add(new EncodedPacket(new Uint8Array(1024), 'delta', 1.3, 0.1));
|
||||
|
||||
await output.finalize();
|
||||
|
||||
// Hacky but works
|
||||
const str = String.fromCharCode(...new Uint8Array(output.target.buffer!));
|
||||
expect(str.includes('edts') || str.includes('elst')).toBe(false);
|
||||
|
||||
using input = new Input({
|
||||
source: new BufferSource(output.target.buffer!),
|
||||
formats: ALL_FORMATS,
|
||||
});
|
||||
|
||||
const track = await input.getPrimaryVideoTrack();
|
||||
assert(track);
|
||||
const sink = new EncodedPacketSink(track);
|
||||
|
||||
const timestamps: number[] = [];
|
||||
const durations: number[] = [];
|
||||
for await (const packet of sink.packets()) {
|
||||
timestamps.push(packet.timestamp);
|
||||
durations.push(packet.duration);
|
||||
}
|
||||
|
||||
expect(timestamps).toEqual([1, 1.1, 1.2, 1.3]);
|
||||
expect(durations).toEqual([0.1, 0.1, 0.1, 0.1]);
|
||||
});
|
||||
|
||||
test('PCM audio', async () => {
|
||||
const output = new Output({
|
||||
format: new Mp4OutputFormat(),
|
||||
target: new BufferTarget(),
|
||||
});
|
||||
|
||||
const source = new EncodedAudioPacketSource('pcm-s16');
|
||||
output.addAudioTrack(source);
|
||||
|
||||
await output.start();
|
||||
|
||||
const meta = { decoderConfig: { codec: 'pcm-s16', numberOfChannels: 2, sampleRate: 48000 } };
|
||||
|
||||
await source.add(new EncodedPacket(new Uint8Array(1024), 'key', 0, 1024 / 2 / 2 / 48000), meta);
|
||||
|
||||
await output.finalize();
|
||||
|
||||
const input = new Input({
|
||||
source: new BufferSource(output.target.buffer!),
|
||||
formats: ALL_FORMATS,
|
||||
});
|
||||
const audioTrack = await input.getPrimaryAudioTrack();
|
||||
assert(audioTrack);
|
||||
|
||||
expect(await audioTrack.getFirstTimestamp()).toBe(0);
|
||||
});
|
||||
|
||||
test('PCM audio with non-zero timestamp', async () => {
|
||||
const output = new Output({
|
||||
format: new Mp4OutputFormat(),
|
||||
target: new BufferTarget(),
|
||||
});
|
||||
|
||||
const source = new EncodedAudioPacketSource('pcm-s16');
|
||||
output.addAudioTrack(source);
|
||||
|
||||
await output.start();
|
||||
|
||||
const meta = { decoderConfig: { codec: 'pcm-s16', numberOfChannels: 2, sampleRate: 48000 } };
|
||||
|
||||
await source.add(new EncodedPacket(new Uint8Array(1024), 'key', 1, 1024 / 2 / 2 / 48000), meta);
|
||||
|
||||
await output.finalize();
|
||||
|
||||
const input = new Input({
|
||||
source: new BufferSource(output.target.buffer!),
|
||||
formats: ALL_FORMATS,
|
||||
});
|
||||
const audioTrack = await input.getPrimaryAudioTrack();
|
||||
assert(audioTrack);
|
||||
|
||||
expect(await audioTrack.getFirstTimestamp()).toBe(1);
|
||||
});
|
||||
|
||||
test('PCM audio, silence padding', async () => {
|
||||
const output = new Output({
|
||||
format: new Mp4OutputFormat(),
|
||||
target: new BufferTarget(),
|
||||
});
|
||||
|
||||
const source = new EncodedAudioPacketSource('pcm-s16');
|
||||
output.addAudioTrack(source);
|
||||
|
||||
await output.start();
|
||||
|
||||
const meta = { decoderConfig: { codec: 'pcm-s16', numberOfChannels: 2, sampleRate: 48000 } };
|
||||
|
||||
await source.add(new EncodedPacket(new Uint8Array(1024), 'key', 0, 1024 / 2 / 2 / 48000), meta);
|
||||
await source.add(new EncodedPacket(new Uint8Array(1024), 'key', 1, 1024 / 2 / 2 / 48000), meta);
|
||||
|
||||
await output.finalize();
|
||||
|
||||
const input = new Input({
|
||||
source: new BufferSource(output.target.buffer!),
|
||||
formats: ALL_FORMATS,
|
||||
});
|
||||
const audioTrack = await input.getPrimaryAudioTrack();
|
||||
assert(audioTrack);
|
||||
|
||||
expect(await audioTrack.getCodec()).toBe('pcm-s16');
|
||||
const numChannels = await audioTrack.getNumberOfChannels();
|
||||
|
||||
const expectedFrameCount = 48000 + 256;
|
||||
const sink = new EncodedPacketSink(audioTrack);
|
||||
let frameCount = 0;
|
||||
|
||||
for await (const packet of sink.packets()) {
|
||||
frameCount += packet.byteLength / 2 / numChannels;
|
||||
}
|
||||
|
||||
expect(frameCount).toBe(expectedFrameCount);
|
||||
});
|
||||
|
||||
test('PCM audio, no silence padding with approximate timestamps', async () => {
|
||||
const output = new Output({
|
||||
format: new Mp4OutputFormat(),
|
||||
target: new BufferTarget(),
|
||||
});
|
||||
|
||||
const source = new EncodedAudioPacketSource('pcm-s16');
|
||||
output.addAudioTrack(source);
|
||||
|
||||
await output.start();
|
||||
|
||||
const meta = { decoderConfig: { codec: 'pcm-s16', numberOfChannels: 2, sampleRate: 48000 } };
|
||||
|
||||
await source.add(new EncodedPacket(new Uint8Array(1024), 'key', 0, 1024 / 2 / 2 / 48000), meta);
|
||||
// 0.006 is 256/48000 "rounded up", but it's close enough for silence padding not to kick in
|
||||
await source.add(new EncodedPacket(new Uint8Array(1024), 'key', 0.006, 1024 / 2 / 2 / 48000), meta);
|
||||
|
||||
await output.finalize();
|
||||
|
||||
const input = new Input({
|
||||
source: new BufferSource(output.target.buffer!),
|
||||
formats: ALL_FORMATS,
|
||||
});
|
||||
const audioTrack = await input.getPrimaryAudioTrack();
|
||||
assert(audioTrack);
|
||||
|
||||
expect(await audioTrack.getCodec()).toBe('pcm-s16');
|
||||
const numChannels = await audioTrack.getNumberOfChannels();
|
||||
|
||||
const expectedFrameCount = 256 + 256;
|
||||
const sink = new EncodedPacketSink(audioTrack);
|
||||
let frameCount = 0;
|
||||
|
||||
for await (const packet of sink.packets()) {
|
||||
frameCount += packet.byteLength / 2 / numChannels;
|
||||
}
|
||||
|
||||
expect(frameCount).toBe(expectedFrameCount);
|
||||
});
|
||||
|
||||
Reference in New Issue
Block a user