mirror of
https://github.com/arcodange-org/mediabunny.git
synced 2026-09-28 11:23:45 +02:00
Add support for remaining audio codecs to Matroska muxer
This commit is contained in:
+13
-7
@@ -46,10 +46,12 @@
|
||||
|
||||
const target = new Metamuxer.BufferTarget();
|
||||
const output = new Metamuxer.Output({
|
||||
format: new Metamuxer.MovOutputFormat({ fastStart: undefined }),
|
||||
format: new Metamuxer.MkvOutputFormat(),
|
||||
target
|
||||
});
|
||||
|
||||
console.log(await input.getFormat());
|
||||
|
||||
const audioTrack = await input.getPrimaryAudioTrack();
|
||||
const videoTrack = await input.getPrimaryVideoTrack();
|
||||
|
||||
@@ -57,9 +59,9 @@
|
||||
|
||||
const decoderConfig = await audioTrack.getDecoderConfig();
|
||||
const sampleSource = new Metamuxer.EncodedAudioSampleSource(await audioTrack.getCodec());
|
||||
const audioDataSource = new Metamuxer.AudioDataSource({ codec: 'alaw', bitrate: 128e3 });
|
||||
const audioDataSource = new Metamuxer.AudioDataSource({ codec: 'pcm-s16', bitrate: 128e3 });
|
||||
const videoSampleSource = new Metamuxer.EncodedVideoSampleSource(await videoTrack.getCodec());
|
||||
output.addAudioTrack(audioDataSource ?? sampleSource);
|
||||
output.addAudioTrack(sampleSource);
|
||||
output.addVideoTrack(videoSampleSource);
|
||||
|
||||
output.start();
|
||||
@@ -67,18 +69,22 @@
|
||||
const videoDecoderConfig = await videoTrack.getDecoderConfig();
|
||||
|
||||
|
||||
/*
|
||||
|
||||
|
||||
for await (const sample of drain.samples()) {
|
||||
//console.log(sample.timestamp);
|
||||
//sample.timestamp *= 2;
|
||||
//console.log(sample);
|
||||
//console.log(sample)
|
||||
await sampleSource.digest(sample, { decoderConfig });
|
||||
}
|
||||
*/
|
||||
for await (const { data } of new Metamuxer.AudioDataDrain(audioTrack).data()) {
|
||||
/*
|
||||
for await (const { data, timestamp } of new Metamuxer.AudioDataDrain(audioTrack).data()) {
|
||||
//console.log("wow", data.timestamp);
|
||||
//console.log(sample)
|
||||
await audioDataSource.digest(data);
|
||||
}
|
||||
*/
|
||||
for await (const sample of new Metamuxer.EncodedVideoSampleDrain(videoTrack).samples()) {
|
||||
//console.log(sample)
|
||||
await videoSampleSource.digest(sample, { decoderConfig: videoDecoderConfig });
|
||||
@@ -96,7 +102,7 @@
|
||||
a.click();
|
||||
URL.revokeObjectURL(url);
|
||||
}
|
||||
download(new Blob([target.buffer]), 'converted.mov');
|
||||
download(new Blob([target.buffer]), 'converted.mkv');
|
||||
|
||||
document.body.textContent = performance.now() - start;
|
||||
|
||||
|
||||
+1
-1
@@ -45,7 +45,7 @@ export class Mp4InputFormat extends IsobmffInputFormat {
|
||||
/** @internal */
|
||||
override async _canReadInput(input: Input) {
|
||||
const majorBrand = await this._getMajorBrand(input);
|
||||
return majorBrand !== 'qt ';
|
||||
return !!majorBrand && majorBrand !== 'qt ';
|
||||
}
|
||||
|
||||
getName() {
|
||||
|
||||
@@ -32,8 +32,11 @@ import {
|
||||
} from '../subtitles';
|
||||
import {
|
||||
AudioCodec,
|
||||
PCM_CODECS,
|
||||
PcmAudioCodec,
|
||||
SubtitleCodec,
|
||||
VideoCodec,
|
||||
parsePcmCodec,
|
||||
validateAudioChunkMetadata,
|
||||
validateSubtitleMetadata,
|
||||
validateVideoChunkMetadata,
|
||||
@@ -103,15 +106,27 @@ type MatroskaAudioTrackData = MatroskaTrackData & { type: 'audio' };
|
||||
type MatroskaSubtitleTrackData = MatroskaTrackData & { type: 'subtitle' };
|
||||
|
||||
const CODEC_STRING_MAP: Partial<Record<VideoCodec | AudioCodec | SubtitleCodec, string>> = {
|
||||
avc: 'V_MPEG4/ISO/AVC',
|
||||
hevc: 'V_MPEGH/ISO/HEVC',
|
||||
vp8: 'V_VP8',
|
||||
vp9: 'V_VP9',
|
||||
av1: 'V_AV1',
|
||||
aac: 'A_AAC', // TODO is this even correct
|
||||
opus: 'A_OPUS',
|
||||
vorbis: 'A_VORBIS',
|
||||
webvtt: 'S_TEXT/WEBVTT',
|
||||
'avc': 'V_MPEG4/ISO/AVC',
|
||||
'hevc': 'V_MPEGH/ISO/HEVC',
|
||||
'vp8': 'V_VP8',
|
||||
'vp9': 'V_VP9',
|
||||
'av1': 'V_AV1',
|
||||
|
||||
'aac': 'A_AAC',
|
||||
'mp3': 'A_MPEG/L3',
|
||||
'opus': 'A_OPUS',
|
||||
'vorbis': 'A_VORBIS',
|
||||
'flac': 'A_FLAC',
|
||||
'pcm-u8': 'A_PCM/INT/LIT',
|
||||
'pcm-s16': 'A_PCM/INT/LIT',
|
||||
'pcm-s16be': 'A_PCM/INT/BIG',
|
||||
'pcm-s24': 'A_PCM/INT/LIT',
|
||||
'pcm-s24be': 'A_PCM/INT/BIG',
|
||||
'pcm-s32': 'A_PCM/INT/LIT',
|
||||
'pcm-s32be': 'A_PCM/INT/BIG',
|
||||
'pcm-f32': 'A_PCM/FLOAT/IEEE',
|
||||
|
||||
'webvtt': 'S_TEXT/WEBVTT',
|
||||
};
|
||||
|
||||
const TRACK_TYPE_MAP: Record<OutputTrack['type'], number> = {
|
||||
@@ -470,6 +485,10 @@ export class MatroskaMuxer extends Muxer {
|
||||
}
|
||||
|
||||
private audioSpecificTrackInfo(trackData: MatroskaAudioTrackData) {
|
||||
const pcmInfo = (PCM_CODECS as readonly string[]).includes(trackData.track.source._codec)
|
||||
? parsePcmCodec(trackData.track.source._codec as PcmAudioCodec)
|
||||
: null;
|
||||
|
||||
return [
|
||||
(trackData.info.decoderConfig.description
|
||||
? {
|
||||
@@ -480,7 +499,7 @@ export class MatroskaMuxer extends Muxer {
|
||||
{ id: EBMLId.Audio, data: [
|
||||
{ id: EBMLId.SamplingFrequency, data: new EBMLFloat32(trackData.info.sampleRate) },
|
||||
{ id: EBMLId.Channels, data: trackData.info.numberOfChannels },
|
||||
// TODO Bit depth for when PCM is a thing
|
||||
pcmInfo ? { id: EBMLId.BitDepth, data: 8 * pcmInfo.sampleSize } : null,
|
||||
] },
|
||||
];
|
||||
}
|
||||
|
||||
+1
-1
@@ -995,7 +995,7 @@ class PcmAudioDecoderWrapper extends DecoderWrapper<EncodedAudioSample, AudioDat
|
||||
numberOfChannels: this.decoderConfig.numberOfChannels,
|
||||
sampleRate: this.decoderConfig.sampleRate,
|
||||
numberOfFrames,
|
||||
timestamp: sample.timestamp,
|
||||
timestamp: sample.microsecondTimestamp,
|
||||
});
|
||||
|
||||
// Since all other decoders are async, we'll make this one behave async as well
|
||||
|
||||
+48
-19
@@ -499,39 +499,68 @@ class AudioEncoderWrapper {
|
||||
assert(this.outputSampleSize);
|
||||
assert(this.writeOutputValue);
|
||||
|
||||
// Need to extract data from the audio data before it's closed
|
||||
const { numberOfChannels, numberOfFrames, sampleRate, timestamp } = audioData;
|
||||
|
||||
const CHUNK_SIZE = 2048;
|
||||
const outputs: {
|
||||
frameCount: number;
|
||||
view: DataView;
|
||||
}[] = [];
|
||||
|
||||
// Prepare all of the output buffers, each being bounded by CHUNK_SIZE so we don't generate huge samples
|
||||
for (let frame = 0; frame < numberOfFrames; frame += CHUNK_SIZE) {
|
||||
const frameCount = Math.min(CHUNK_SIZE, audioData.numberOfFrames - frame);
|
||||
const outputSize = frameCount * numberOfChannels * this.outputSampleSize;
|
||||
const outputBuffer = new ArrayBuffer(outputSize);
|
||||
const outputView = new DataView(outputBuffer);
|
||||
|
||||
outputs.push({ frameCount, view: outputView });
|
||||
}
|
||||
|
||||
// All user agents are required to support conversion to f32-planar
|
||||
const allocationSize = audioData.allocationSize(({ planeIndex: 0, format: 'f32-planar' }));
|
||||
const floats = new Float32Array(allocationSize / Float32Array.BYTES_PER_ELEMENT);
|
||||
|
||||
const channelCount = audioData.numberOfChannels;
|
||||
const outputSize = audioData.numberOfFrames * channelCount * this.outputSampleSize;
|
||||
const outputBuffer = new ArrayBuffer(outputSize);
|
||||
const outputView = new DataView(outputBuffer);
|
||||
|
||||
for (let i = 0; i < channelCount; i++) {
|
||||
for (let i = 0; i < numberOfChannels; i++) {
|
||||
audioData.copyTo(floats, { planeIndex: i, format: 'f32-planar' });
|
||||
for (let j = 0; j < floats.length; j++) {
|
||||
// Write it interleaved... interleavedly?
|
||||
this.writeOutputValue(outputView, (j * channelCount + i) * this.outputSampleSize, floats[j]!);
|
||||
|
||||
for (let j = 0; j < outputs.length; j++) {
|
||||
const { frameCount, view } = outputs[j]!;
|
||||
|
||||
for (let k = 0; k < frameCount; k++) {
|
||||
this.writeOutputValue(
|
||||
view,
|
||||
(k * numberOfChannels + i) * this.outputSampleSize,
|
||||
floats[j * CHUNK_SIZE + k]!,
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
const sample = new EncodedAudioSample(
|
||||
new Uint8Array(outputBuffer),
|
||||
'key',
|
||||
audioData.timestamp / 1e6,
|
||||
audioData.duration / 1e6,
|
||||
);
|
||||
const meta: EncodedAudioChunkMetadata = {
|
||||
decoderConfig: {
|
||||
codec: this.encodingConfig.codec,
|
||||
numberOfChannels: audioData.numberOfChannels,
|
||||
sampleRate: audioData.sampleRate,
|
||||
numberOfChannels,
|
||||
sampleRate,
|
||||
},
|
||||
};
|
||||
|
||||
this.encodingConfig.onEncodedSample?.(sample, meta);
|
||||
await this.muxer!.addEncodedAudioSample(this.source._connectedTrack!, sample, meta); // With backpressure
|
||||
for (let i = 0; i < outputs.length; i++) {
|
||||
const { frameCount, view } = outputs[i]!;
|
||||
const outputBuffer = view.buffer;
|
||||
const startFrame = i * CHUNK_SIZE;
|
||||
|
||||
const sample = new EncodedAudioSample(
|
||||
new Uint8Array(outputBuffer),
|
||||
'key',
|
||||
timestamp / 1e6 + startFrame / sampleRate,
|
||||
frameCount / sampleRate,
|
||||
);
|
||||
|
||||
this.encodingConfig.onEncodedSample?.(sample, meta);
|
||||
await this.muxer!.addEncodedAudioSample(this.source._connectedTrack!, sample, meta); // With backpressure
|
||||
}
|
||||
}
|
||||
|
||||
private ensureEncoder(audioData: AudioData) {
|
||||
|
||||
@@ -158,7 +158,9 @@ export class MkvOutputFormat extends OutputFormat {
|
||||
static getSupportedCodecs(): MediaCodec[] {
|
||||
return [
|
||||
'avc', 'hevc', 'vp8', 'vp9', 'av1',
|
||||
'aac', 'opus', 'vorbis',
|
||||
'aac', 'mp3', 'opus', 'vorbis', 'flac',
|
||||
// pcm-s8, pcm-f32be, ulaw and alaw are not supported
|
||||
'pcm-u8', 'pcm-s16', 'pcm-s16be', 'pcm-s24', 'pcm-s24be', 'pcm-s32', 'pcm-s32be', 'pcm-f32',
|
||||
'webvtt',
|
||||
];
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user