diff --git a/dev/demux.html b/dev/demux.html index 5354400..6d93d39 100644 --- a/dev/demux.html +++ b/dev/demux.html @@ -17,8 +17,13 @@ source: new Mediabunny.BlobSource(file), }); - const track = await input.getPrimaryVideoTrack(); - console.log(await track.getDecoderConfig()); + const track = await input.getPrimaryAudioTrack(); + const sink = new Mediabunny.AudioSampleSink(track); + + for await (const sample of sink.samples()) { + console.log(sample); + sample.close() + } /* diff --git a/shared/aac-misc.ts b/shared/aac-misc.ts index 5c0fc71..e1ecaf7 100644 --- a/shared/aac-misc.ts +++ b/shared/aac-misc.ts @@ -10,10 +10,11 @@ import { Bitstream } from './bitstream'; export type AacAudioSpecificConfig = { objectType: number; + coreObjectType: number; frequencyIndex: number; - sampleRate: number | null; channelConfiguration: number; - numberOfChannels: number | null; + outputSampleRate: number | null; + outputNumberOfChannels: number | null; }; export const aacFrequencyTable = [ @@ -30,20 +31,8 @@ export const parseAacAudioSpecificConfig = (bytes: Uint8Array | null): AacAudioS const bitstream = new Bitstream(bytes); - let objectType = bitstream.readBits(5); - if (objectType === 31) { - objectType = 32 + bitstream.readBits(6); - } - - const frequencyIndex = bitstream.readBits(4); - let sampleRate: number | null = null; - if (frequencyIndex === 15) { - sampleRate = bitstream.readBits(24); - } else { - if (frequencyIndex < aacFrequencyTable.length) { - sampleRate = aacFrequencyTable[frequencyIndex]!; - } - } + const objectType = readAacObjectType(bitstream); + const { frequencyIndex, sampleRate } = readAacSamplingFrequency(bitstream); const channelConfiguration = bitstream.readBits(4); let numberOfChannels: number | null = null; @@ -51,61 +40,155 @@ export const parseAacAudioSpecificConfig = (bytes: Uint8Array | null): AacAudioS numberOfChannels = aacChannelMap[channelConfiguration]!; } + let coreObjectType = objectType; + let psPresent = false; + let outputSampleRate = sampleRate; + + if (objectType === 5 || objectType === 29) { + // Explicit hierarchical signaling: everything read so far describes the core coder, and the rate the + // decoder actually outputs follows right here + psPresent = objectType === 29; + + outputSampleRate = readAacSamplingFrequency(bitstream).sampleRate; + coreObjectType = readAacObjectType(bitstream); + + if (coreObjectType === 22) { + bitstream.skipBits(4); // extensionChannelConfiguration + } + } else { + // There may be SBR/PS flags sitting behind a sync word after the config of the core coder. We find them by + // scanning for the sync word, just like FFmpeg does. + while (bitstream.getBitsLeft() > 15) { + const searchStart = bitstream.pos; + + if (bitstream.readBits(11) !== 0x2b7) { + bitstream.pos = searchStart + 1; + continue; + } + + if (readAacObjectType(bitstream) === 5 && bitstream.readBits(1)) { + outputSampleRate = readAacSamplingFrequency(bitstream).sampleRate; + + if (bitstream.getBitsLeft() > 11 && bitstream.readBits(11) === 0x548) { + psPresent = !!bitstream.readBits(1); + } + } + + break; + } + } + + if (numberOfChannels !== null && numberOfChannels > 1) { + // PS only ever upmixes a mono core. A stereo signal means whatever we read was bogus + psPresent = false; + } + return { objectType, + coreObjectType, frequencyIndex, - sampleRate, channelConfiguration, - numberOfChannels, + outputSampleRate, + outputNumberOfChannels: psPresent && numberOfChannels === 1 + ? 2 + : numberOfChannels, + }; +}; + +const readAacObjectType = (bitstream: Bitstream) => { + const objectType = bitstream.readBits(5); + return objectType === 31 ? 32 + bitstream.readBits(6) : objectType; +}; + +const readAacSamplingFrequency = (bitstream: Bitstream) => { + const frequencyIndex = bitstream.readBits(4); + + if (frequencyIndex === 15) { + return { + frequencyIndex, + sampleRate: bitstream.readBits(24), + }; + } + + return { + frequencyIndex, + sampleRate: frequencyIndex < aacFrequencyTable.length + ? aacFrequencyTable[frequencyIndex]! + : null, }; }; export const buildAacAudioSpecificConfig = (config: { objectType: number; - sampleRate: number; - numberOfChannels: number; + outputSampleRate: number; + outputNumberOfChannels: number; }) => { - let frequencyIndex = aacFrequencyTable.indexOf(config.sampleRate); - let customSampleRate: number | null = null; + const usesSbr = config.objectType === 5 || config.objectType === 29; + const usesPs = config.objectType === 29; - if (frequencyIndex === -1) { - frequencyIndex = 15; - customSampleRate = config.sampleRate; - } + // SBR runs the core coder at half the output rate, and PS upmixes a mono core to stereo + const coreSampleRate = usesSbr ? config.outputSampleRate / 2 : config.outputSampleRate; + const coreNumberOfChannels = usesPs ? 1 : config.outputNumberOfChannels; - const channelConfiguration = aacChannelMap.indexOf(config.numberOfChannels); + const channelConfiguration = aacChannelMap.indexOf(coreNumberOfChannels); if (channelConfiguration === -1) { - throw new TypeError(`Unsupported number of channels: ${config.numberOfChannels}`); + throw new TypeError(`Unsupported number of channels: ${config.outputNumberOfChannels}`); } - let bitCount = 5 + 4 + 4; + // Object type, sampling frequency, channel configuration, then a bare GASpecificConfig + let bitCount = 5 + 4 + 4 + 3; if (config.objectType >= 32) { bitCount += 6; } - if (frequencyIndex === 15) { + if (findAacFrequencyIndex(coreSampleRate) === 15) { bitCount += 24; } + if (usesSbr) { + bitCount += 4 + 5; // Extension sampling frequency and the object type of the core coder + if (findAacFrequencyIndex(config.outputSampleRate) === 15) { + bitCount += 24; + } + } const byteCount = Math.ceil(bitCount / 8); const bytes = new Uint8Array(byteCount); const bitstream = new Bitstream(bytes); - if (config.objectType < 32) { - bitstream.writeBits(5, config.objectType); - } else { - bitstream.writeBits(5, 31); - bitstream.writeBits(6, config.objectType - 32); + writeAacObjectType(bitstream, config.objectType); + writeAacSamplingFrequency(bitstream, coreSampleRate); + bitstream.writeBits(4, channelConfiguration); + + if (usesSbr) { + writeAacSamplingFrequency(bitstream, config.outputSampleRate); + writeAacObjectType(bitstream, 2); // AAC-LC underneath } + bitstream.writeBits(3, 0); // frameLengthFlag, dependsOnCoreCoder, extensionFlag + + return bytes; +}; + +const writeAacObjectType = (bitstream: Bitstream, objectType: number) => { + if (objectType < 32) { + bitstream.writeBits(5, objectType); + } else { + bitstream.writeBits(5, 31); + bitstream.writeBits(6, objectType - 32); + } +}; + +const writeAacSamplingFrequency = (bitstream: Bitstream, sampleRate: number) => { + const frequencyIndex = findAacFrequencyIndex(sampleRate); bitstream.writeBits(4, frequencyIndex); if (frequencyIndex === 15) { - bitstream.writeBits(24, customSampleRate!); + bitstream.writeBits(24, sampleRate); } +}; - bitstream.writeBits(4, channelConfiguration); - - return bytes; +const findAacFrequencyIndex = (sampleRate: number) => { + const index = aacFrequencyTable.indexOf(sampleRate); + return index === -1 ? 15 : index; }; export type AdtsHeaderTemplate = { @@ -117,8 +200,8 @@ export const buildAdtsHeaderTemplate = (config: AacAudioSpecificConfig): AdtsHea const header = new Uint8Array(7); const bitstream = new Bitstream(header); - const { objectType, frequencyIndex, channelConfiguration } = config; - const profile = objectType - 1; + const { coreObjectType, frequencyIndex, channelConfiguration } = config; + const profile = coreObjectType - 1; bitstream.writeBits(12, 0b1111_11111111); // Syncword bitstream.writeBits(1, 0); // MPEG Version diff --git a/src/isobmff/isobmff-demuxer.ts b/src/isobmff/isobmff-demuxer.ts index 2f9ba57..0e0fe60 100644 --- a/src/isobmff/isobmff-demuxer.ts +++ b/src/isobmff/isobmff-demuxer.ts @@ -1609,11 +1609,11 @@ export class IsobmffDemuxer extends Demuxer { if (track.info.codec === 'aac') { // Let's try to deduce more accurate values directly from the AudioSpecificConfig: const audioSpecificConfig = parseAacAudioSpecificConfig(track.info.codecDescription); - if (audioSpecificConfig.numberOfChannels !== null) { - track.info.numberOfChannels = audioSpecificConfig.numberOfChannels; + if (audioSpecificConfig.outputNumberOfChannels !== null) { + track.info.numberOfChannels = audioSpecificConfig.outputNumberOfChannels; } - if (audioSpecificConfig.sampleRate !== null) { - track.info.sampleRate = audioSpecificConfig.sampleRate; + if (audioSpecificConfig.outputSampleRate !== null) { + track.info.sampleRate = audioSpecificConfig.outputSampleRate; } } } diff --git a/src/isobmff/isobmff-muxer.ts b/src/isobmff/isobmff-muxer.ts index 4d6c8c5..1e3006c 100644 --- a/src/isobmff/isobmff-muxer.ts +++ b/src/isobmff/isobmff-muxer.ts @@ -525,8 +525,8 @@ export class IsobmffMuxer extends Muxer { decoderConfig.description = buildAacAudioSpecificConfig({ objectType: adtsFrame.objectType, - sampleRate, - numberOfChannels, + outputSampleRate: sampleRate, + outputNumberOfChannels: numberOfChannels, }); requiresAdtsStripping = true; } diff --git a/src/matroska/matroska-muxer.ts b/src/matroska/matroska-muxer.ts index 7763df3..dc213a8 100644 --- a/src/matroska/matroska-muxer.ts +++ b/src/matroska/matroska-muxer.ts @@ -858,8 +858,8 @@ export class MatroskaMuxer extends Muxer { decoderConfig.description = buildAacAudioSpecificConfig({ objectType: adtsFrame.objectType, - sampleRate, - numberOfChannels, + outputSampleRate: sampleRate, + outputNumberOfChannels: numberOfChannels, }); requiresAdtsStripping = true; } diff --git a/src/media-source.ts b/src/media-source.ts index 8d82533..5260e12 100644 --- a/src/media-source.ts +++ b/src/media-source.ts @@ -2257,8 +2257,8 @@ class AudioEncoderWrapper { meta.decoderConfig.description = buildAacAudioSpecificConfig({ objectType, - numberOfChannels: meta.decoderConfig.numberOfChannels, - sampleRate: meta.decoderConfig.sampleRate, + outputNumberOfChannels: meta.decoderConfig.numberOfChannels, + outputSampleRate: meta.decoderConfig.sampleRate, }); } } diff --git a/test/node/read-mp4.test.ts b/test/node/read-mp4.test.ts index 6b09ed5..12a2677 100644 --- a/test/node/read-mp4.test.ts +++ b/test/node/read-mp4.test.ts @@ -14,6 +14,7 @@ import { Mp4OutputFormat, Output, } from '../../src/index.js'; +import { assert, toUint8Array } from '../../src/misc.js'; const __dirname = new URL('.', import.meta.url).pathname; @@ -160,3 +161,23 @@ test('Annex B', async () => { const firstPacket = (await sink.getFirstPacket())!; expect([...firstPacket.data.slice(0, 4)]).toEqual([0, 0, 0, 1]); }); + +test('HE-AAC v2 audio config is parsed correctly', async () => { + const filePath = path.join(__dirname, '..', 'public/he-aac-v2.mp4'); + using input = new Input({ + source: new FilePathSource(filePath), + formats: ALL_FORMATS, + }); + + const track = await input.getPrimaryAudioTrack(); + assert(track); + + expect(await track.getSampleRate()).toBe(44100); + expect(await track.getNumberOfChannels()).toBe(2); + + const decoderConfig = (await track.getDecoderConfig())!; + expect(decoderConfig.codec).toBe('mp4a.40.29'); + expect(decoderConfig.sampleRate).toBe(44100); + expect(decoderConfig.numberOfChannels).toBe(2); + expect([...toUint8Array(decoderConfig.description!)]).toEqual([0xeb, 0x8a, 0x08, 0x00]); +}); diff --git a/test/public/he-aac-v2.mp4 b/test/public/he-aac-v2.mp4 new file mode 100644 index 0000000..5811026 Binary files /dev/null and b/test/public/he-aac-v2.mp4 differ