diff --git a/docs/.vitepress/config.mts b/docs/.vitepress/config.mts index 9f6a58e..9720265 100644 --- a/docs/.vitepress/config.mts +++ b/docs/.vitepress/config.mts @@ -162,6 +162,7 @@ export default withMermaid({ { text: 'FLAC', link: '/codec-registry/flac' }, { text: 'AC-3', link: '/codec-registry/ac3' }, { text: 'E-AC-3', link: '/codec-registry/eac3' }, + { text: 'DTS', link: '/codec-registry/dts' }, { text: 'Linear PCM', link: '/codec-registry/pcm' }, { text: 'μ-law PCM', link: '/codec-registry/ulaw' }, { text: 'A-law PCM', link: '/codec-registry/alaw' }, diff --git a/docs/codec-registry/dts.md b/docs/codec-registry/dts.md new file mode 100644 index 0000000..3f0d5b6 --- /dev/null +++ b/docs/codec-registry/dts.md @@ -0,0 +1,44 @@ +--- +description: DTS (DTS Coherent Acoustics) audio codec definition, defining legal codec strings, decoder configs, and packet data formats. +--- + + + + + +# DTS codec registration + +## Description + +The DTS audio codec (DTS Coherent Acoustics), specified in [ETSI TS 102 114](https://www.etsi.org/deliver/etsi_ts/102100_102199/102114/01.06.01_60/ts_102114v010601p.pdf). + +## Codec ID + +```ts +'dts' +``` + +## `EncodedPacket` data + +The packet's data must be a core substream frame as defined in Section 5 of [ETSI TS 102 114](https://www.etsi.org/deliver/etsi_ts/102100_102199/102114/01.06.01_60/ts_102114v010601p.pdf), beginning with the sync word `0x7FFE8001`, followed by any number of extension substreams as defined in Section 7.4.1 of [ETSI TS 102 114](https://www.etsi.org/deliver/etsi_ts/102100_102199/102114/01.06.01_60/ts_102114v010601p.pdf), each beginning with the sync word `0x64582025`. The core substream frame may be omitted. + +The bitstream must use 16-bit big-endian packing. + +## `EncodedPacket` type + +The packet's type is always `'key'`. + +## `AudioDecoderConfig` codec string + +The codec string must be one of the four four-character codes identifying which substreams the bitstream is built from: + +- `'dtsc'` - a core substream only +- `'dtsh'` - a core substream with extension substreams +- `'dtsl'` - extension substreams carrying lossless audio, with no core substream +- `'dtse'` - DTS Express + +## `AudioDecoderConfig` description + +`description` is not used for this codec. diff --git a/docs/codec-registry/overview.md b/docs/codec-registry/overview.md index bae0fb4..85e5b27 100644 --- a/docs/codec-registry/overview.md +++ b/docs/codec-registry/overview.md @@ -26,6 +26,7 @@ The registry is an extension of the [WebCodecs Codec Registry](https://www.w3.or - [FLAC](./flac) - [AC-3](./ac3) - [E-AC-3](./eac3) +- [DTS](./dts) - [Linear PCM](./pcm) - [μ-law PCM](./ulaw) - [A-law PCM](./alaw) diff --git a/docs/guide/extensions/server.md b/docs/guide/extensions/server.md index 9d3754a..aec0681 100644 --- a/docs/guide/extensions/server.md +++ b/docs/guide/extensions/server.md @@ -8,7 +8,7 @@ By default, Mediabunny requires a browser environment for full access to decoder Features added by this package include: - Video decoders and encoders for AVC (H.264), HEVC (H.265), VP8, VP9, AV1, and ProRes. Supports both length-prefixed and Annex B AVC/HEVC as well as transparent video via VP9 and ProRes. -- Audio decoders and encoders for AAC, MP3, Vorbis, Opus, FLAC, AC-3 and E-AC-3. Supports AAC in both AAC and ADTS formats. +- Audio decoders and encoders for AAC, MP3, Vorbis, Opus, FLAC, AC-3, E-AC-3 and DTS. Supports AAC in both AAC and ADTS formats. - Video frame transformation support (resize, rotate, crop) - Automatic hardware acceleration on all platforms (macOS, Linux, Windows) - Built-in multithreading diff --git a/docs/guide/introduction.md b/docs/guide/introduction.md index 0e66c84..d2ec32a 100644 --- a/docs/guide/introduction.md +++ b/docs/guide/introduction.md @@ -64,7 +64,7 @@ Mediabunny's simple yet flexible API provides a modern alternative to traditiona The extension enables: - Video decoders and encoders for AVC (H.264), HEVC (H.265), VP8, VP9, and AV1. Supports both length-prefixed and Annex B AVC/HEVC as well as transparent video via VP9. -- Audio decoders and encoders for AAC, MP3, Vorbis, Opus, FLAC, AC-3 and E-AC-3. Supports AAC in both AAC and ADTS formats. +- Audio decoders and encoders for AAC, MP3, Vorbis, Opus, FLAC, AC-3, E-AC-3 and DTS. Supports AAC in both AAC and ADTS formats. - Video frame transformation support (resize, rotate, crop) - Automatic hardware acceleration on all platforms (macOS, Linux, Windows) - Built-in multithreading diff --git a/docs/guide/supported-formats-and-codecs.md b/docs/guide/supported-formats-and-codecs.md index d420bf0..6158de2 100644 --- a/docs/guide/supported-formats-and-codecs.md +++ b/docs/guide/supported-formats-and-codecs.md @@ -51,6 +51,7 @@ Mediabunny ships with built-in decoders and encoders for all audio PCM codecs, m - `'flac'` - Free Lossless Audio Codec (FLAC) [^flac] - `'ac3'` - Dolby Digital (AC-3) [^ac3] - `'eac3'` - Dolby Digital Plus (E-AC-3) [^ac3] +- `'dts'` - DTS Coherent Acoustics [^dts] - `'pcm-u8'` - 8-bit unsigned PCM - `'pcm-s8'` - 8-bit signed PCM - `'pcm-s16'` - 16-bit little-endian signed PCM @@ -89,6 +90,7 @@ Not all codecs can be used with all containers. The following table specifies th | `'flac'` | ✓ | ✓ | ✓ | | | | | | ✓ | | | `'ac3'` | ✓ | ✓ | ✓ | | | | | | | ✓ | | `'eac3'` | ✓ | ✓ | ✓ | | | | | | | ✓ | +| `'dts'` | ✓ | ✓ | ✓ | | | | | | | ✓ | | `'pcm-u8'` | | ✓ | ✓ | | | | ✓ | | | | | `'pcm-s8'` | | ✓ | | | | | | | | | | `'pcm-s16'` | ✓ | ✓ | ✓ | | | | ✓ | | | | @@ -112,6 +114,7 @@ For HLS, the supported codecs depend on the segment format chosen. [^mp3]: MP3 encoding is not supported by WebCodecs. You can polyfill it with the [`@mediabunny/mp3-encoder`](./extensions/mp3-encoder) extension package. [^flac]: FLAC encoding is not supported by WebCodecs. You can polyfill it with the [`@mediabunny/flac-encoder`](./extensions/flac-encoder) extension package. [^ac3]: AC-3 and E-AC-3 are not natively supported by WebCodecs. To encode or decode these codecs, you can use the [`@mediabunny/ac3`](./extensions/ac3) extension package. +[^dts]: DTS is not supported by WebCodecs. The [`@mediabunny/server`](./extensions/server) extension package provides both decoding and encoding support for server-side environments. [^webm]: WebM only supports a small subset of the codecs supported by Matroska. However, this library can technically read all codecs from a WebM that are supported by Matroska. [^webvtt]: WebVTT can only be written, not read. @@ -127,7 +130,7 @@ import { canEncode } from 'mediabunny'; canEncode('avc'); // => Promise canEncode('opus'); // => Promise ``` -Video codecs are checked using 1280x720 @1Mbps, while audio codecs are checked using 2 channels, 48 kHz @128kbps. +Video codecs are checked using 1280x720, while audio codecs are checked using 2 channels at 48 kHz. You can also check encodability using specific configurations: ```ts diff --git a/packages/server/src/audio-decoder.ts b/packages/server/src/audio-decoder.ts index 22bb420..247d16b 100644 --- a/packages/server/src/audio-decoder.ts +++ b/packages/server/src/audio-decoder.ts @@ -25,7 +25,8 @@ export class NodeAvAudioDecoder extends CustomAudioDecoder { || codec === 'vorbis' || codec === 'flac' || codec === 'ac3' - || codec === 'eac3'; + || codec === 'eac3' + || codec === 'dts'; } async init(): Promise { diff --git a/packages/server/src/audio-encoder.ts b/packages/server/src/audio-encoder.ts index 9b94106..172a0f5 100644 --- a/packages/server/src/audio-encoder.ts +++ b/packages/server/src/audio-encoder.ts @@ -15,7 +15,7 @@ import { EncodedPacket, } from 'mediabunny'; import * as NodeAv from 'node-av'; -import { CODEC_TO_CODEC_ID, getChannelLayout } from './misc'; +import { CODEC_TO_CODEC_ID, getChannelLayout, getDtsChannelLayout } from './misc'; import { assert, toUint8Array } from '../../../src/misc'; import { copyAudioSampleToAvFrame, AvFrameAudioSampleResource } from './audio-sample'; import { @@ -29,6 +29,11 @@ const AAC_SAMPLE_RATES const OPUS_SAMPLE_RATES = [8000, 12000, 16000, 24000, 48000]; const MP3_SAMPLE_RATES = [8000, 11025, 12000, 16000, 22050, 24000, 32000, 44100, 48000]; const AC3_SAMPLE_RATES = [32000, 44100, 48000]; +const DTS_SAMPLE_RATES = [8000, 11025, 12000, 16000, 22050, 24000, 32000, 44100, 48000]; + +/** DTS only has layouts for these channel counts; 3 and 7 have no mapping. */ +const DTS_CHANNEL_COUNTS = [1, 2, 4, 5, 6]; +const DTS_MAX_FRAME_SIZE = 16384; const FRAME_SIZE_FALLBACK = 1024; // Just 'cause @@ -72,6 +77,10 @@ export class NodeAvAudioEncoder extends CustomAudioEncoder { ) || ( codec === 'eac3' && numberOfChannels >= 1 && numberOfChannels <= 16 && AC3_SAMPLE_RATES.includes(sampleRate) + ) || ( + codec === 'dts' && DTS_CHANNEL_COUNTS.includes(numberOfChannels) + && DTS_SAMPLE_RATES.includes(sampleRate) + && dtsBitrateFits(resolveBitrate(config, codec), sampleRate, numberOfChannels) ); } @@ -107,21 +116,26 @@ export class NodeAvAudioEncoder extends CustomAudioEncoder { } codecContext.sampleRate = this.config.sampleRate; - codecContext.channelLayout = getChannelLayout(this.config.numberOfChannels); + codecContext.channelLayout = this.codec === 'dts' + ? getDtsChannelLayout(this.config.numberOfChannels) + : getChannelLayout(this.config.numberOfChannels); codecContext.codecType = NodeAv.AVMEDIA_TYPE_AUDIO; codecContext.codecId = CODEC_TO_CODEC_ID[this.codec]!; codecContext.sampleFormat = sampleFormat; codecContext.timeBase = new NodeAv.Rational(1, this.config.sampleRate); - codecContext.bitRate = BigInt( - this.config.bitrate ?? new Quality('medium')._toAudioBitrate(this.codec) ?? 0, - ); + codecContext.bitRate = BigInt(resolveBitrate(this.config, this.codec)); if (this.config.bitrateMode === 'constant') { codecContext.rcMinRate = codecContext.bitRate; codecContext.rcMaxRate = codecContext.bitRate; } - const ret = await codecContext.open2(); + // libav's DTS encoder is marked experimental, so it refuses to open at the default compliance level + const options = this.codec === 'dts' + ? NodeAv.Dictionary.fromObject({ strict: -2 }) + : null; + + const ret = await codecContext.open2(this.avCodec, options); NodeAv.FFmpegError.throwIfError(ret, 'Open codec context'); this.codecContext = codecContext; @@ -172,7 +186,9 @@ export class NodeAvAudioEncoder extends CustomAudioEncoder { this.resampler = new NodeAv.SoftwareResampleContext(); this.resamplerInputSampleRate = this.frame.sampleRate; - const outLayout = getChannelLayout(this.codecContext.channels); + // Resample straight to whatever layout the encoder was opened with, since not every codec accepts + // the layout that getChannelLayout hands out for a given channel count + const outLayout = this.codecContext.channelLayout; const inLayout = getChannelLayout(this.frame.channels); const ret = this.resampler.allocSetOpts2( @@ -400,3 +416,25 @@ export class NodeAvAudioEncoder extends CustomAudioEncoder { this.resampler?.free(); } } + +const resolveBitrate = (config: AudioEncoderConfig, codec: AudioCodec) => { + return config.bitrate ?? new Quality('medium')._toAudioBitrate(codec) ?? 0; +}; + +/** + * The DTS encoder needs each frame to be big enough to hold the per-channel side info and rejects the whole + * configuration when it isn't, so this mirrors the check from FFmpeg's dcaenc.c. + */ +const dtsBitrateFits = (bitrate: number, sampleRate: number, numberOfChannels: number) => { + if (bitrate < 32000 || bitrate > 3840000) { + return false; + } + + const hasLfe = numberOfChannels === 6; + const fullbandChannels = numberOfChannels - (hasLfe ? 1 : 0); + + const frameBits = 32 * Math.ceil(Math.ceil(bitrate * 512 / sampleRate) / 32); + const minFrameBits = 132 + (493 + 28 * 32) * fullbandChannels + (hasLfe ? 72 : 0); + + return frameBits >= minFrameBits && frameBits <= 8 * DTS_MAX_FRAME_SIZE; +}; diff --git a/packages/server/src/misc.ts b/packages/server/src/misc.ts index 3c152ae..ab4998a 100644 --- a/packages/server/src/misc.ts +++ b/packages/server/src/misc.ts @@ -25,6 +25,7 @@ export const CODEC_TO_CODEC_ID: Partial> = flac: NodeAv.AV_CODEC_ID_FLAC, ac3: NodeAv.AV_CODEC_ID_AC3, eac3: NodeAv.AV_CODEC_ID_EAC3, + dts: NodeAv.AV_CODEC_ID_DTS, }; let cachedHardwareContext: NodeAv.HardwareContext | null | undefined = undefined; @@ -267,3 +268,21 @@ export const getChannelLayout = (numChannels: number): NodeAv.ChannelLayout => { default: return { nbChannels: numChannels, order: NodeAv.AV_CHANNEL_ORDER_UNSPEC, mask: 0n }; } }; + +/** FL + FR + SL + SR, which libav has no constant for. */ +const QUAD_SIDE_MASK = 0x603n; + +/** + * DTS insists on the side-based surround layouts and rejects the back-based ones that {@link getChannelLayout} + * hands out, so it gets its own mapping. + */ +export const getDtsChannelLayout = (numChannels: number): NodeAv.ChannelLayout => { + switch (numChannels) { + case 1: return NodeAv.AV_CHANNEL_LAYOUT_MONO; + case 2: return NodeAv.AV_CHANNEL_LAYOUT_STEREO; + case 4: return { nbChannels: 4, order: NodeAv.AV_CHANNEL_ORDER_NATIVE, mask: QUAD_SIDE_MASK }; + case 5: return NodeAv.AV_CHANNEL_LAYOUT_5POINT0; + case 6: return NodeAv.AV_CHANNEL_LAYOUT_5POINT1; + default: return { nbChannels: numChannels, order: NodeAv.AV_CHANNEL_ORDER_UNSPEC, mask: 0n }; + } +}; diff --git a/src/codec-data.ts b/src/codec-data.ts index 8319af3..071720d 100644 --- a/src/codec-data.ts +++ b/src/codec-data.ts @@ -6,7 +6,7 @@ * file, You can obtain one at https://mozilla.org/MPL/2.0/. */ -import { AVC_LEVEL_TABLE, VideoCodec, VP9_LEVEL_TABLE } from './codec'; +import { AVC_LEVEL_TABLE, DtsFourCc, VideoCodec, VP9_LEVEL_TABLE } from './codec'; import { assert, assertNever, @@ -24,6 +24,7 @@ import { toUint8Array, getChromiumVersion, isChromium, + popcount, setUint24, } from './misc'; import { Logging } from './logging'; @@ -3176,3 +3177,495 @@ export const getEac3ChannelCount = (config: Eac3FrameInfo): number => { return channels; }; + +// ============================================================================ +// DTS Parsing +// Reference: ETSI TS 102 114 V1.6.1 +// ============================================================================ + +/** Core substream sync word, in the 16-bit big-endian packing that the registry mandates. */ +export const DTS_CORE_SYNC_WORD = 0x7ffe8001; + +/** Extension substream sync word. Section 7.4.1 */ +export const DTS_EXSS_SYNC_WORD = 0x64582025; + +/** The core frame header never reaches beyond this many bytes. */ +export const DTS_CORE_FRAME_HEADER_SIZE = 18; + +/** An extension substream always declares its own size within this many bytes. */ +export const DTS_EXSS_HEADER_PREFIX_SIZE = 10; + +/** The largest nuExtSSHeaderSize can get, and therefore how far the asset descriptors can reach. */ +export const DTS_EXSS_MAX_HEADER_SIZE = 4096; + +/** Number of PCM samples in one core PCM block; the core codes its length as a count of these. */ +export const DTS_PCM_BLOCK_SAMPLES = 32; + +/** Size of the DTSSpecificBox (ddts) payload in bytes. */ +export const DTS_SPECIFIC_BOX_SIZE = 20; + +/** Number of PCM blocks that a core frame's length must be a multiple of. */ +const DTS_SUBBAND_SAMPLES = 8; + +/** Core sample rates indexed by SFREQ. Zeroes mark invalid codes. Table 5-4 */ +const DTS_CORE_SAMPLE_RATES = [ + 0, 8000, 16000, 32000, 0, 0, 11025, 22050, + 44100, 0, 0, 12000, 24000, 48000, 96000, 192000, +]; + +/** + * Core bit rates in bps indexed by RATE, where a zero means the code isn't a constant rate. Table 5-7 + * + * Note that FFmpeg's ff_dca_bit_rates has 896000 where the spec has 960, and defines rates for codes 25 to 28 + * which this revision of the spec calls invalid. We keep the latter, since they cost nothing and some content + * predating the spec revision uses them. + */ +const DTS_CORE_BIT_RATES = [ + 32000, 56000, 64000, 96000, 112000, 128000, 192000, 224000, + 256000, 320000, 384000, 448000, 512000, 576000, 640000, 768000, + 960000, 1024000, 1152000, 1280000, 1344000, 1408000, 1411200, 1472000, + 1536000, 1920000, 2048000, 3072000, 3840000, 0, 0, 0, +]; + +/** Source PCM resolutions in bits indexed by PCMR. Zeroes mark invalid codes. */ +const DTS_PCM_RESOLUTIONS = [16, 16, 20, 20, 0, 24, 24, 0]; + +/** Channel counts indexed by AMODE, not counting LFE. */ +const DTS_AMODE_CHANNEL_COUNTS = [1, 2, 2, 2, 2, 3, 3, 4, 4, 5, 6, 6, 6, 7, 8, 8]; + +/** + * Speaker layout masks indexed by AMODE, expressed with the same bits as `ChannelLayout` in the DTSSpecificBox + * and `nuSpkrActivityMask` in an extension substream asset descriptor. + */ +const DTS_AMODE_CHANNEL_LAYOUTS = [ + 0x0001, 0x0002, 0x0002, 0x0002, 0x0002, 0x0003, 0x0012, 0x0013, + 0x0006, 0x0007, 0x0206, 0x0143, 0x0053, 0x0207, 0x0246, 0x0217, +]; + +/** The LFE1 speaker bit in a channel layout mask. */ +const DTS_CHANNEL_LAYOUT_LFE1 = 0x0008; + +/** The channel layout bits that stand for a pair of speakers rather than a single one. */ +const DTS_CHANNEL_LAYOUT_PAIR_MASK = 0xae66; + +/** Reference clock rates indexed by nuRefClockCode. The last code is unused. Table 7-3 */ +const DTS_EXSS_REF_CLOCKS = [32000, 44100, 48000, 0]; + +/** Sample rates used by extension substream assets, indexed by nuMaxSampleRate. */ +const DTS_EXSS_SAMPLE_RATES = [ + 8000, 16000, 32000, 64000, 128000, 22050, 44100, 88200, + 176400, 352800, 12000, 24000, 48000, 96000, 192000, 384000, +]; + +/** Frame durations that the DTSSpecificBox can express, indexed by FrameDuration. */ +const DTS_SPECIFIC_BOX_FRAME_DURATIONS = [512, 1024, 2048, 4096]; + +export type DtsCoreFrameInfo = { + /** Size of the core substream frame in bytes */ + frameSize: number; + sampleRate: number; + numberOfChannels: number; + /** Number of PCM samples the frame decodes to */ + sampleCount: number; + /** Speaker layout mask */ + channelLayout: number; + /** Audio channel arrangement (AMODE) */ + amode: number; + /** Whether an LFE channel is present */ + lfePresent: boolean; + /** Constant bit rate in bps, or 0 when the stream doesn't run at one */ + bitRate: number; + /** Source PCM resolution in bits */ + pcmResolution: number; +}; + +export type DtsExssAssetInfo = { + sampleRate: number; + numberOfChannels: number; + /** Number of PCM samples the frame decodes to */ + sampleCount: number; + /** Speaker layout mask, or 0 when the asset doesn't declare one */ + channelLayout: number; + /** Source PCM resolution in bits */ + pcmResolution: number; +}; + +export type DtsExssInfo = { + /** Size of the extension substream in bytes */ + frameSize: number; + /** Null when this substream omits the static fields that describe the stream */ + asset: DtsExssAssetInfo | null; +}; + +export type DtsFrameInfo = { + /** Size of the entire frame in bytes, core substream plus any extension substreams */ + frameSize: number; + sampleRate: number; + numberOfChannels: number; + /** Number of PCM samples the frame decodes to */ + sampleCount: number; + /** Speaker layout mask */ + channelLayout: number; + /** Source PCM resolution in bits */ + pcmResolution: number; + /** Constant bit rate in bps, or 0 when the stream doesn't run at one */ + bitRate: number; + /** The leading core substream, or null for extension-only streams such as DTS Express */ + core: DtsCoreFrameInfo | null; + /** Whether the frame carries extension substreams on top of the core */ + hasExtensions: boolean; +}; + +/** + * Parse one complete DTS frame, being a core substream frame followed by any number of extension substreams, + * or an extension substream on its own. Section 5 and Section 7.4.1 + */ +export const parseDtsFrame = (data: Uint8Array): DtsFrameInfo | null => { + const core = parseDtsCoreFrameHeader(data); + const view = toDataView(data); + + // The core substream is padded out to a 4-byte boundary before the first extension substream starts + let offset = core ? Math.ceil(core.frameSize / 4) * 4 : 0; + let firstExss: DtsExssInfo | null = null; + + while (offset + 4 <= data.length && view.getUint32(offset) === DTS_EXSS_SYNC_WORD) { + const exss = parseDtsExssHeader(data.subarray(offset)); + if (!exss) { + break; + } + + firstExss ??= exss; + offset += exss.frameSize; + } + + if (core) { + // The core describes what every DTS decoder can play back; the extension substreams only build on top of + // it, so the core's parameters are the ones we report. + return { + frameSize: firstExss ? offset : core.frameSize, + sampleRate: core.sampleRate, + numberOfChannels: core.numberOfChannels, + sampleCount: core.sampleCount, + channelLayout: core.channelLayout, + pcmResolution: core.pcmResolution, + bitRate: core.bitRate, + core, + hasExtensions: firstExss !== null, + }; + } + + if (!firstExss?.asset) { + return null; + } + + const { asset } = firstExss; + + return { + frameSize: offset, + sampleRate: asset.sampleRate, + numberOfChannels: asset.numberOfChannels, + sampleCount: asset.sampleCount, + channelLayout: asset.channelLayout, + pcmResolution: asset.pcmResolution, + bitRate: 0, + core: null, + hasExtensions: true, + }; +}; + +/** + * Works out which four-character code describes a packet, or null when the packet doesn't say. Telling 'dtsl' + * from 'dtse' would mean working out whether the asset holds XLL or LBR data, which sits behind the speaker + * remapping and mixing metadata deep in the asset descriptor, so we don't do it. + */ +export const extractDtsFourCcFromPacket = (data: Uint8Array): DtsFourCc | null => { + const frameInfo = parseDtsFrame(data); + if (!frameInfo?.core) { + return null; + } + + return frameInfo.hasExtensions ? 'dtsh' : 'dtsc'; +}; + +/** Parse the header of a core substream frame. Section 5.3 */ +export const parseDtsCoreFrameHeader = (data: Uint8Array): DtsCoreFrameInfo | null => { + if (data.length < DTS_CORE_FRAME_HEADER_SIZE) { + return null; + } + if (data[0] !== 0x7f || data[1] !== 0xfe || data[2] !== 0x80 || data[3] !== 0x01) { + return null; + } + + const bitstream = new Bitstream(data); + bitstream.skipBits(32); // SYNC + + bitstream.skipBits(1); // FTYPE + + // Terminating frames carry fewer samples than a full PCM block; we don't handle those + if (bitstream.readBits(5) !== DTS_PCM_BLOCK_SAMPLES - 1) { + return null; + } + + const cpf = bitstream.readBits(1); + const npcmblocks = bitstream.readBits(7) + 1; + if (npcmblocks % DTS_SUBBAND_SAMPLES !== 0) { + return null; + } + + const frameSize = bitstream.readBits(14) + 1; + if (frameSize < 96) { + return null; + } + + const amode = bitstream.readBits(6); + if (amode >= DTS_AMODE_CHANNEL_COUNTS.length) { + return null; + } + + const sampleRate = DTS_CORE_SAMPLE_RATES[bitstream.readBits(4)]!; + if (sampleRate === 0) { + return null; + } + + const bitRate = DTS_CORE_BIT_RATES[bitstream.readBits(5)]!; + + if (bitstream.readBits(1) !== 0) { + return null; // A reserved bit that must be zero, so a cheap way to reject false sync words + } + + bitstream.skipBits(1 + 1 + 1 + 1); // DYNF, TIMEF, AUXF, HDCD + bitstream.skipBits(3 + 1 + 1); // EXT_AUDIO_ID, EXT_AUDIO, ASPF + + const lff = bitstream.readBits(2); + if (lff === 3) { + return null; + } + + bitstream.skipBits(1); // HFLAG + if (cpf) { + bitstream.skipBits(16); // HCRC + } + + bitstream.skipBits(1 + 4 + 2); // FILTS, VERNUM, CHIST + + const pcmResolution = DTS_PCM_RESOLUTIONS[bitstream.readBits(3)]!; + if (pcmResolution === 0) { + return null; + } + + const lfePresent = lff !== 0; + + return { + frameSize, + sampleRate, + numberOfChannels: DTS_AMODE_CHANNEL_COUNTS[amode]! + (lfePresent ? 1 : 0), + sampleCount: npcmblocks * DTS_PCM_BLOCK_SAMPLES, + channelLayout: DTS_AMODE_CHANNEL_LAYOUTS[amode]! | (lfePresent ? DTS_CHANNEL_LAYOUT_LFE1 : 0), + amode, + lfePresent, + bitRate, + pcmResolution, + }; +}; + +/** Parse the header of an extension substream, along with its first audio asset descriptor. Section 7.4.1 */ +export const parseDtsExssHeader = (data: Uint8Array): DtsExssInfo | null => { + if (data.length < DTS_EXSS_HEADER_PREFIX_SIZE) { + return null; + } + if (data[0] !== 0x64 || data[1] !== 0x58 || data[2] !== 0x20 || data[3] !== 0x25) { + return null; + } + + const bitstream = new Bitstream(data); + bitstream.skipBits(32); // SYNC + bitstream.skipBits(8); // nuUserDefinedBits + + const extSsIndex = bitstream.readBits(2); + const wideHeader = bitstream.readBits(1); + const headerSizeBits = 8 + 4 * wideHeader; + const frameSizeBits = 16 + 4 * wideHeader; + + bitstream.skipBits(headerSizeBits); // nuExtSSHeaderSize + const frameSize = bitstream.readBits(frameSizeBits) + 1; + + // Everything past this point can run off the end of what we were given, in which case the Bitstream keeps + // handing out zeroes; the bounds check further down catches that + const incomplete: DtsExssInfo = { frameSize, asset: null }; + + if (!bitstream.readBits(1)) { // bStaticFieldsPresent + return incomplete; + } + + const refClock = DTS_EXSS_REF_CLOCKS[bitstream.readBits(2)]!; // nuRefClockCode + // The frame duration is a count of reference clock cycles, not of samples + const frameDurationCycles = 512 * (bitstream.readBits(3) + 1); // nuExSSFrameDurationCode + + if (bitstream.readBits(1)) { // bTimeStampFlag + bitstream.skipBits(32 + 4); // nuTimeStamp, nLSB + } + + const numAudioPresentations = bitstream.readBits(3) + 1; + const numAssets = bitstream.readBits(3) + 1; + + const activeExssMasks: number[] = []; + for (let i = 0; i < numAudioPresentations; i++) { + activeExssMasks.push(bitstream.readBits(extSsIndex + 1)); + } + for (const mask of activeExssMasks) { + bitstream.skipBits(8 * popcount(mask)); // nuActiveAssetMask + } + + if (bitstream.readBits(1)) { // bMixMetadataEnbl + bitstream.skipBits(2); // nuMixMetadataAdjLevel + const spkrMaskBits = (bitstream.readBits(2) + 1) << 2; + const numMixOutConfigs = bitstream.readBits(2) + 1; + bitstream.skipBits(numMixOutConfigs * spkrMaskBits); // nuMixOutChMask + } + + for (let i = 0; i < numAssets; i++) { + bitstream.skipBits(frameSizeBits); // nuAssetFsize + } + + // From here on we're inside the first audio asset descriptor + bitstream.skipBits(9); // nuAssetDescriptFsize + bitstream.skipBits(3); // nuAssetIndex + + if (bitstream.readBits(1)) { // bAssetTypeDescrPresent + bitstream.skipBits(4); // nuAssetTypeDescriptor + } + if (bitstream.readBits(1)) { // bLanguageDescrPresent + bitstream.skipBits(24); // LanguageDescriptor + } + if (bitstream.readBits(1)) { // bInfoTextPresent + bitstream.skipBits(8 * (bitstream.readBits(10) + 1)); // nuInfoTextByteSize, InfoTextString + } + + const pcmResolution = bitstream.readBits(5) + 1; + const sampleRate = DTS_EXSS_SAMPLE_RATES[bitstream.readBits(4)]!; + const numberOfChannels = bitstream.readBits(8) + 1; + + let channelLayout = 0; + + if (bitstream.readBits(1)) { // bOne2OneMapChannels2Speakers + if (numberOfChannels > 2) { + bitstream.skipBits(1); // bEmbeddedStereoFlag + } + if (numberOfChannels > 6) { + bitstream.skipBits(1); // bEmbeddedSixChFlag + } + + if (bitstream.readBits(1)) { // bSpkrMaskEnabled + const spkrMaskBits = (bitstream.readBits(2) + 1) << 2; + channelLayout = bitstream.readBits(spkrMaskBits); // nuSpkrActivityMask + } + } + + if (refClock === 0 || bitstream.getBitsLeft() < 0) { + return incomplete; + } + + return { + frameSize, + asset: { + sampleRate, + numberOfChannels, + sampleCount: Math.round(frameDurationCycles * sampleRate / refClock), + channelLayout, + pcmResolution, + }, + }; +}; + +export type DtsSpecificBoxInfo = { + sampleRate: number; + maxBitrate: number; + avgBitrate: number; + pcmSampleDepth: number; + /** Number of PCM samples one frame decodes to */ + sampleCount: number; + /** Speaker layout mask, or 0 when the box doesn't declare one */ + channelLayout: number; + /** Null when the box carries nothing we can derive a channel count from */ + numberOfChannels: number | null; +}; + +/** Parse a DTSSpecificBox (ddts). */ +export const parseDtsSpecificBox = (data: Uint8Array): DtsSpecificBoxInfo | null => { + if (data.length < DTS_SPECIFIC_BOX_SIZE) { + return null; + } + + const view = toDataView(data); + const sampleRate = view.getUint32(0); + if (sampleRate === 0) { + return null; + } + + const bitstream = new Bitstream(data); + bitstream.seekToByte(13); + + const frameDuration = bitstream.readBits(2); + bitstream.skipBits(5); // StreamConstruction + const coreLfePresent = bitstream.readBits(1); + const coreLayout = bitstream.readBits(6); + bitstream.skipBits(14); // CoreSize + bitstream.skipBits(1); // StereoDownmix + bitstream.skipBits(3); // RepresentationType + const channelLayout = bitstream.readBits(16); + + let numberOfChannels: number | null = null; + if (channelLayout !== 0) { + numberOfChannels = getDtsChannelCount(channelLayout); + } else if (coreLayout < DTS_AMODE_CHANNEL_COUNTS.length) { + numberOfChannels = DTS_AMODE_CHANNEL_COUNTS[coreLayout]! + coreLfePresent; + } + + return { + sampleRate, + maxBitrate: view.getUint32(4), + avgBitrate: view.getUint32(8), + pcmSampleDepth: data[12]!, + sampleCount: DTS_SPECIFIC_BOX_FRAME_DURATIONS[frameDuration]!, + channelLayout, + numberOfChannels, + }; +}; + +/** Build the payload of a DTSSpecificBox (ddts) from a frame of the stream it describes. */ +export const buildDtsSpecificBox = (frameInfo: DtsFrameInfo) => { + const bytes = new Uint8Array(DTS_SPECIFIC_BOX_SIZE); + const view = toDataView(bytes); + + view.setUint32(0, frameInfo.sampleRate); + view.setUint32(4, frameInfo.bitRate); + view.setUint32(8, frameInfo.bitRate); + bytes[12] = frameInfo.pcmResolution; + + // The spec only defines codes for streams built out of a known set of substreams. Anything else is required + // to be signaled as 0, meaning the construction is left unspecified. + const streamConstruction = frameInfo.core && !frameInfo.hasExtensions ? 1 : 0; + + const bitstream = new Bitstream(bytes); + bitstream.seekToByte(13); + + bitstream.writeBits(2, Math.max(DTS_SPECIFIC_BOX_FRAME_DURATIONS.indexOf(frameInfo.sampleCount), 0)); + bitstream.writeBits(5, streamConstruction); + bitstream.writeBits(1, frameInfo.core?.lfePresent ? 1 : 0); + bitstream.writeBits(6, frameInfo.core?.amode ?? 0); + bitstream.writeBits(14, frameInfo.core ? frameInfo.core.frameSize - 1 : 0); + bitstream.writeBits(1, 0); // StereoDownmix + bitstream.writeBits(3, 0); // RepresentationType + bitstream.writeBits(16, frameInfo.channelLayout); + bitstream.writeBits(1, 0); // MultiAssetFlag + bitstream.writeBits(1, 0); // LBRDurationMod + bitstream.writeBits(1, 0); // ReservedBoxPresent + bitstream.writeBits(5, 0); // Reserved + + return bytes; +}; + +/** Count the channels in a DTS speaker layout mask, where some bits stand for a pair of speakers. */ +const getDtsChannelCount = (channelLayout: number) => { + return popcount(channelLayout) + popcount(channelLayout & DTS_CHANNEL_LAYOUT_PAIR_MASK); +}; diff --git a/src/codec.ts b/src/codec.ts index 08f1b52..63c41d1 100644 --- a/src/codec.ts +++ b/src/codec.ts @@ -75,6 +75,7 @@ export const NON_PCM_AUDIO_CODECS = [ 'flac', 'ac3', 'eac3', + 'dts', ] as const; /** * List of known audio codecs, ordered by encoding preference. @@ -227,6 +228,14 @@ export const PRORES_FOURCCS = [ ] as const; export type ProresFourCc = typeof PRORES_FOURCCS[number]; +export const DTS_FOURCCS = [ + 'dtsc', // DTS core + 'dtsh', // DTS-HD, core plus extension substreams + 'dtsl', // DTS-HD Lossless, no core + 'dtse', // DTS Express +] as const; +export type DtsFourCc = typeof DTS_FOURCCS[number]; + // Target data rates of the ProRes profiles at 1920x1080 ~30fps, as published by Apple const PRORES_PROFILE_TARGET_BITRATES: { fourCc: ProresFourCc; bitrate: number; alpha: boolean }[] = [ { fourCc: 'apco', bitrate: 45_000_000, alpha: false }, // 422 Proxy @@ -595,6 +604,8 @@ export const buildAudioCodecString = (codec: AudioCodec, numberOfChannels: numbe return 'ac-3'; } else if (codec === 'eac3') { return 'ec-3'; + } else if (codec === 'dts') { + return 'dtsc'; } else if ((PCM_AUDIO_CODECS as readonly string[]).includes(codec)) { return codec; } @@ -611,8 +622,9 @@ export const extractAudioCodecString = (trackInfo: { codec: AudioCodec | null; codecDescription: Uint8Array | null; aacCodecInfo: AacCodecInfo | null; + dtsFormat: DtsFourCc | null; }) => { - const { codec, codecDescription, aacCodecInfo } = trackInfo; + const { codec, codecDescription, aacCodecInfo, dtsFormat } = trackInfo; if (codec === 'aac') { if (!aacCodecInfo) { @@ -644,6 +656,8 @@ export const extractAudioCodecString = (trackInfo: { return 'ac-3'; } else if (codec === 'eac3') { return 'ec-3'; + } else if (codec === 'dts') { + return dtsFormat ?? 'dtsc'; } else if (codec && (PCM_AUDIO_CODECS as readonly string[]).includes(codec)) { return codec; } @@ -756,6 +770,8 @@ export const inferCodecFromCodecString = (codecString: string): MediaCodec | nul return 'ac3'; } else if (codecString === 'ec-3' || codecString === 'eac3') { return 'eac3'; + } else if ((DTS_FOURCCS as readonly string[]).includes(codecString)) { + return 'dts'; } else if (codecString === 'ulaw') { return 'ulaw'; } else if (codecString === 'alaw') { @@ -998,7 +1014,7 @@ export const validateVideoChunkMetadata = ( }; const VALID_AUDIO_CODEC_STRING_PREFIXES = [ - 'mp4a', 'mp3', 'opus', 'vorbis', 'flac', 'ulaw', 'alaw', 'pcm', 'ac-3', 'ec-3', + 'mp4a', 'mp3', 'opus', 'vorbis', 'flac', 'ulaw', 'alaw', 'pcm', 'ac-3', 'ec-3', 'dts', ]; export const validateAudioChunkMetadata = ( @@ -1132,6 +1148,15 @@ export const validateAudioChunkMetadata = ( if (metadata.decoderConfig.codec !== 'ec-3') { throw new TypeError('Audio chunk metadata decoder configuration codec string for EC-3 must be "ec-3".'); } + } else if (metadata.decoderConfig.codec.startsWith('dts')) { + // DTS-specific validation + + if (!(DTS_FOURCCS as readonly string[]).includes(metadata.decoderConfig.codec)) { + throw new TypeError( + 'Audio chunk metadata decoder configuration codec string for DTS must be one of the following' + + ` four-character codes: ${DTS_FOURCCS.join(', ')}.`, + ); + } } else if ( metadata.decoderConfig.codec.startsWith('pcm') || metadata.decoderConfig.codec.startsWith('ulaw') diff --git a/src/encode.ts b/src/encode.ts index 656122d..a30fd9f 100644 --- a/src/encode.ts +++ b/src/encode.ts @@ -858,6 +858,7 @@ export class Quality { vorbis: 64000, // 64kbps base for Vorbis ac3: 384000, // 384kbps base for AC-3 eac3: 192000, // 192kbps base for E-AC-3 + dts: 768000, // 768kbps base for DTS }; const baseBitrate = baseRates[codec as keyof typeof baseRates]; @@ -1071,7 +1072,7 @@ export const canEncodeVideo = async ( } validateVideoEncodingAdditionalOptions(codec, restOptions); - const resolvedQuality = resolveQuality(quality, bitrate) ?? new Quality({ bitrate: 1e6 }); + const resolvedQuality = resolveQuality(quality, bitrate) ?? new Quality('medium'); let candidates: VideoEncoderConfigCandidate[]; try { @@ -1222,7 +1223,7 @@ export const canEncodeAudio = async ( } validateAudioEncodingAdditionalOptions(codec, restOptions); - const resolvedQuality = resolveQuality(quality, bitrate) ?? new Quality({ bitrate: 128e3 }); + const resolvedQuality = resolveQuality(quality, bitrate) ?? new Quality('medium'); const encoderConfig = buildAudioEncoderConfig({ codec, diff --git a/src/isobmff/isobmff-boxes.ts b/src/isobmff/isobmff-boxes.ts index 3867fc4..ca3ba8b 100644 --- a/src/isobmff/isobmff-boxes.ts +++ b/src/isobmff/isobmff-boxes.ts @@ -43,7 +43,13 @@ import { IsobmffVideoTrackData, Sample, } from './isobmff-muxer'; -import { parseAc3SyncFrame, parseEac3SyncFrame, parseOpusIdentificationHeader } from '../codec-data'; +import { + buildDtsSpecificBox, + parseAc3SyncFrame, + parseDtsFrame, + parseEac3SyncFrame, + parseOpusIdentificationHeader, +} from '../codec-data'; import { MetadataTags, RichImageData } from '../metadata'; import { Bitstream } from '../../shared/bitstream'; @@ -690,7 +696,11 @@ export const stsd = (trackData: IsobmffTrackData) => { trackData, ); } else if (trackData.type === 'audio') { - const boxName = audioCodecToBoxName(trackData.track.source._codec, trackData.muxer.isQuickTime); + const boxName = audioCodecToBoxName( + trackData.track.source._codec, + trackData.info.decoderConfig.codec, + trackData.muxer.isQuickTime, + ); assert(boxName); sampleDescription = soundSampleDescription( @@ -968,7 +978,11 @@ export const wave = (trackData: IsobmffAudioTrackData) => { export const frma = (trackData: IsobmffAudioTrackData) => { return box('frma', [ - ascii(audioCodecToBoxName(trackData.track.source._codec, trackData.muxer.isQuickTime)), + ascii(audioCodecToBoxName( + trackData.track.source._codec, + trackData.info.decoderConfig.codec, + trackData.muxer.isQuickTime, + )), ]); }; @@ -1123,6 +1137,21 @@ const dec3 = (trackData: IsobmffAudioTrackData) => { return box('dec3', [...bytes]); }; +/** DTSSpecificBox */ +const ddts = (trackData: IsobmffAudioTrackData) => { + assert(trackData.info.primingPacket); + + const frameInfo = parseDtsFrame(trackData.info.primingPacket.data); + if (!frameInfo) { + throw new Error( + 'Couldn\'t extract DTS frame info from the audio packet. ' + + 'Ensure the packets contain valid DTS frames as specified in ETSI TS 102 114.', + ); + } + + return box('ddts', [...buildDtsSpecificBox(frameInfo)]); +}; + export const subtitleSampleDescription = ( compressionType: string, trackData: IsobmffSubtitleTrackData, @@ -1816,7 +1845,7 @@ const VIDEO_CODEC_TO_CONFIGURATION_BOX: Record< prores: null, }; -const audioCodecToBoxName = (codec: AudioCodec, isQuickTime: boolean): string => { +const audioCodecToBoxName = (codec: AudioCodec, fullCodecString: string, isQuickTime: boolean): string => { switch (codec) { case 'aac': return 'mp4a'; case 'mp3': return 'mp4a'; @@ -1829,6 +1858,7 @@ const audioCodecToBoxName = (codec: AudioCodec, isQuickTime: boolean): string => case 'pcm-s8': return 'sowt'; case 'ac3': return 'ac-3'; case 'eac3': return 'ec-3'; + case 'dts': return fullCodecString; } // Logic diverges here @@ -1870,6 +1900,7 @@ const audioCodecToConfigurationBox = (codec: AudioCodec, isQuickTime: boolean) = case 'flac': return dfLa; case 'ac3': return dac3; case 'eac3': return dec3; + case 'dts': return ddts; } // Logic diverges here diff --git a/src/isobmff/isobmff-demuxer.ts b/src/isobmff/isobmff-demuxer.ts index 98ab435..8419780 100644 --- a/src/isobmff/isobmff-demuxer.ts +++ b/src/isobmff/isobmff-demuxer.ts @@ -11,6 +11,8 @@ import { parseAacAudioSpecificConfig } from '../../shared/aac-misc'; import { AacCodecInfo, AudioCodec, + DTS_FOURCCS, + DtsFourCc, extractAudioCodecString, extractVideoCodecString, MediaCodec, @@ -33,6 +35,9 @@ import { parseEac3Config, getEac3SampleRate, getEac3ChannelCount, + extractDtsFourCcFromPacket, + parseDtsSpecificBox, + DTS_SPECIFIC_BOX_SIZE, AC3_ACMOD_CHANNEL_COUNTS, } from '../codec-data'; import { Demuxer } from '../demuxer'; @@ -159,6 +164,7 @@ type InternalTrack = { codec: AudioCodec | null; codecDescription: Uint8Array | null; aacCodecInfo: AacCodecInfo | null; + dtsFormat: DtsFourCc | null; pcmLittleEndian: boolean; pcmSampleSize: number | null; }; @@ -1034,6 +1040,7 @@ export class IsobmffDemuxer extends Demuxer { codec: null, codecDescription: null, aacCodecInfo: null, + dtsFormat: null, pcmLittleEndian: false, pcmSampleSize: null, }; @@ -1185,6 +1192,9 @@ export class IsobmffDemuxer extends Demuxer { track.info.codec = 'ac3'; } else if (codecName === 'ec-3') { track.info.codec = 'eac3'; + } else if ((DTS_FOURCCS as readonly string[]).includes(codecName!)) { + track.info.codec = 'dts'; + track.info.dtsFormat = codecName as DtsFourCc; } else if (codecName === 'twos') { if (sampleSize === 8) { track.info.codec = 'pcm-s8'; @@ -1564,6 +1574,8 @@ export class IsobmffDemuxer extends Demuxer { track.info.codec = 'mp3'; } else if (objectTypeIndication === 0xdd) { track.info.codec = 'vorbis'; // "nonstandard, gpac uses it" - FFmpeg + } else if (objectTypeIndication === 0xa9) { + track.info.codec = 'dts'; } else { Logging._warn( `Unsupported audio codec (objectTypeIndication ${objectTypeIndication}) - discarding track.`, @@ -1763,6 +1775,28 @@ export class IsobmffDemuxer extends Demuxer { track.info.numberOfChannels = getEac3ChannelCount(config); }; break; + case 'ddts': { // DTSSpecificBox + const track = this.currentTrack; + if (!track) { + break; + } + assert(track.info?.type === 'audio'); + + const bytes = readBytes(slice, Math.min(boxInfo.contentSize, DTS_SPECIFIC_BOX_SIZE)); + const config = parseDtsSpecificBox(bytes); + + if (!config) { + Logging._warn('Invalid ddts box contents, ignoring.'); + break; + } + + track.info.sampleRate = config.sampleRate; + + if (config.numberOfChannels !== null) { + track.info.numberOfChannels = config.numberOfChannels; + } + }; break; + case 'stts': { const track = this.currentTrack; if (!track) { @@ -3404,7 +3438,7 @@ class IsobmffVideoTrackBacking extends IsobmffTrackBacking implements InputVideo class IsobmffAudioTrackBacking extends IsobmffTrackBacking implements InputAudioTrackBacking { override internalTrack: InternalAudioTrack; - decoderConfig: AudioDecoderConfig | null = null; + decoderConfigPromise: Promise | null = null; constructor(internalTrack: InternalAudioTrack) { super(internalTrack); @@ -3432,12 +3466,20 @@ class IsobmffAudioTrackBacking extends IsobmffTrackBacking implements InputAudio return null; } - return this.decoderConfig ??= { - codec: extractAudioCodecString(this.internalTrack.info), - numberOfChannels: this.internalTrack.info.numberOfChannels, - sampleRate: this.internalTrack.info.sampleRate, - description: this.internalTrack.info.codecDescription ?? undefined, - }; + return this.decoderConfigPromise ??= (async (): Promise => { + if (this.internalTrack.info.codec === 'dts' && !this.internalTrack.info.dtsFormat) { + // Gotta check the packet to determine the DTS variant + const firstPacket = await this.getFirstPacket({}); + this.internalTrack.info.dtsFormat = firstPacket && extractDtsFourCcFromPacket(firstPacket.data); + } + + return { + codec: extractAudioCodecString(this.internalTrack.info), + numberOfChannels: this.internalTrack.info.numberOfChannels, + sampleRate: this.internalTrack.info.sampleRate, + description: this.internalTrack.info.codecDescription ?? undefined, + }; + })(); } } diff --git a/src/isobmff/isobmff-muxer.ts b/src/isobmff/isobmff-muxer.ts index 06b5f15..4d6c8c5 100644 --- a/src/isobmff/isobmff-muxer.ts +++ b/src/isobmff/isobmff-muxer.ts @@ -531,10 +531,14 @@ export class IsobmffMuxer extends Muxer { requiresAdtsStripping = true; } - if (track.source._codec === 'ac3' || track.source._codec === 'eac3') { - if (!packet) { + if (!packet) { + if (track.source._codec === 'ac3' || track.source._codec === 'eac3') { throw new Error('AC-3/E-AC-3 require a priming packet.'); } + + if (track.source._codec === 'dts') { + throw new Error('DTS requires a priming packet.'); + } } const newTrackData: IsobmffAudioTrackData = { diff --git a/src/matroska/ebml.ts b/src/matroska/ebml.ts index 08a07c7..6cc5a88 100644 --- a/src/matroska/ebml.ts +++ b/src/matroska/ebml.ts @@ -750,6 +750,7 @@ export const CODEC_STRING_MAP: Partial> = { 'flac': 'A_FLAC', 'ac3': 'A_AC3', 'eac3': 'A_EAC3', + 'dts': 'A_DTS', 'pcm-u8': 'A_PCM/INT/LIT', 'pcm-s16': 'A_PCM/INT/LIT', 'pcm-s16be': 'A_PCM/INT/BIG', diff --git a/src/matroska/matroska-demuxer.ts b/src/matroska/matroska-demuxer.ts index 26d538b..9fbbc3f 100644 --- a/src/matroska/matroska-demuxer.ts +++ b/src/matroska/matroska-demuxer.ts @@ -9,6 +9,7 @@ import { TrackType } from '../output'; import { extractAv1CodecInfoFromPacket, + extractDtsFourCcFromPacket, extractAvcDecoderConfigurationRecord, extractHevcDecoderConfigurationRecord, extractVp9CodecInfoFromPacket, @@ -16,6 +17,7 @@ import { import { AacCodecInfo, AudioCodec, + DtsFourCc, extractAudioCodecString, extractVideoCodecString, MediaCodec, @@ -229,6 +231,7 @@ type InternalTrack = { codec: AudioCodec | null; codecDescription: Uint8Array | null; aacCodecInfo: AacCodecInfo | null; + dtsFormat: DtsFourCc | null; }; }; type InternalVideoTrack = InternalTrack & { info: { type: 'video' } }; @@ -1127,6 +1130,14 @@ export class MatroskaDemuxer extends Demuxer { } else if (codecIdWithoutSuffix === CODEC_STRING_MAP.eac3) { this.currentTrack.info.codec = 'eac3'; this.currentTrack.info.codecDescription = this.currentTrack.codecPrivate; + } else if (codecIdWithoutSuffix === CODEC_STRING_MAP.dts) { + this.currentTrack.info.codec = 'dts'; + + if (this.currentTrack.codecId === 'A_DTS/EXPRESS') { + this.currentTrack.info.dtsFormat = 'dtse'; + } else if (this.currentTrack.codecId === 'A_DTS/LOSSLESS') { + this.currentTrack.info.dtsFormat = 'dtsl'; + } } else if (this.currentTrack.codecId === 'A_PCM/INT/LIT') { if (this.currentTrack.info.bitDepth === 8) { this.currentTrack.info.codec = 'pcm-u8'; @@ -1200,6 +1211,7 @@ export class MatroskaDemuxer extends Demuxer { codec: null, codecDescription: null, aacCodecInfo: null, + dtsFormat: null, }; } }; break; @@ -2556,7 +2568,7 @@ class MatroskaVideoTrackBacking extends MatroskaTrackBacking implements InputVid class MatroskaAudioTrackBacking extends MatroskaTrackBacking implements InputAudioTrackBacking { override internalTrack: InternalAudioTrack; - decoderConfig: AudioDecoderConfig | null = null; + decoderConfigPromise: Promise | null = null; constructor(internalTrack: InternalAudioTrack) { super(internalTrack); @@ -2584,15 +2596,24 @@ class MatroskaAudioTrackBacking extends MatroskaTrackBacking implements InputAud return null; } - return this.decoderConfig ??= { - codec: extractAudioCodecString({ - codec: this.internalTrack.info.codec, - codecDescription: this.internalTrack.info.codecDescription, - aacCodecInfo: this.internalTrack.info.aacCodecInfo, - }), - numberOfChannels: this.internalTrack.info.numberOfChannels, - sampleRate: this.internalTrack.info.sampleRate, - description: this.internalTrack.info.codecDescription ?? undefined, - }; + return this.decoderConfigPromise ??= (async (): Promise => { + if (this.internalTrack.info.codec === 'dts' && !this.internalTrack.info.dtsFormat) { + // Gotta check the packet to determine the DTS variant + const firstPacket = await this.getFirstPacket({}); + this.internalTrack.info.dtsFormat = firstPacket && extractDtsFourCcFromPacket(firstPacket.data); + } + + return { + codec: extractAudioCodecString({ + codec: this.internalTrack.info.codec, + codecDescription: this.internalTrack.info.codecDescription, + aacCodecInfo: this.internalTrack.info.aacCodecInfo, + dtsFormat: this.internalTrack.info.dtsFormat, + }), + numberOfChannels: this.internalTrack.info.numberOfChannels, + sampleRate: this.internalTrack.info.sampleRate, + description: this.internalTrack.info.codecDescription ?? undefined, + }; + })(); } } diff --git a/src/matroska/matroska-muxer.ts b/src/matroska/matroska-muxer.ts index 41b0525..7763df3 100644 --- a/src/matroska/matroska-muxer.ts +++ b/src/matroska/matroska-muxer.ts @@ -310,9 +310,18 @@ export class MatroskaMuxer extends Muxer { this.tracksElement = tracksElement; for (const trackData of this.trackDatas) { - const codecId = CODEC_STRING_MAP[trackData.track.source._codec]; + let codecId = CODEC_STRING_MAP[trackData.track.source._codec]; assert(codecId); + if (trackData.type === 'audio' && trackData.track.source._codec === 'dts') { + // We can further refine the Codec ID + if (trackData.info.decoderConfig.codec === 'dtse') { + codecId = 'A_DTS/EXPRESS'; + } else if (trackData.info.decoderConfig.codec === 'dtsl') { + codecId = 'A_DTS/LOSSLESS'; + } + } + let seekPreRollNs = 0; if (trackData.type === 'audio' && trackData.track.source._codec === 'opus') { seekPreRollNs = 1e6 * 80; // In "Matroska ticks" (nanoseconds) diff --git a/src/misc.ts b/src/misc.ts index bd6979d..60fea06 100644 --- a/src/misc.ts +++ b/src/misc.ts @@ -472,6 +472,17 @@ export const ilog = (x: number) => { return ret; }; +export const popcount = (value: number) => { + let count = 0; + + while (value !== 0) { + value &= value - 1; + count++; + } + + return count; +}; + const ISO_639_2_REGEX = /^[a-z]{3}$/; export const isIso639Dash2LanguageCode = (x: string) => { return ISO_639_2_REGEX.test(x); diff --git a/src/mpeg-ts/mpeg-ts-demuxer.ts b/src/mpeg-ts/mpeg-ts-demuxer.ts index f82a09c..54635cd 100644 --- a/src/mpeg-ts/mpeg-ts-demuxer.ts +++ b/src/mpeg-ts/mpeg-ts-demuxer.ts @@ -13,6 +13,7 @@ import { aacChannelMap, aacFrequencyTable } from '../../shared/aac-misc'; import { AacCodecInfo, AudioCodec, + DtsFourCc, extractAudioCodecString, extractVideoCodecString, MediaCodec, @@ -38,6 +39,12 @@ import { AC3_FRAME_SIZES, extractNalUnitTypeForAvc, extractNalUnitTypeForHevc, + parseDtsFrame, + parseDtsCoreFrameHeader, + parseDtsExssHeader, + DTS_CORE_FRAME_HEADER_SIZE, + DTS_EXSS_HEADER_PREFIX_SIZE, + DTS_EXSS_MAX_HEADER_SIZE, } from '../codec-data'; import { Demuxer } from '../demuxer'; import { Input } from '../input'; @@ -82,6 +89,22 @@ import { Bitstream } from '../../shared/bitstream'; const MISSING_PTS_ERROR_MESSAGE = 'PES packet is missing PTS where it was expected. PES packets without PTS are not' + ' currently supported. If you think this file should be supported, please report it.'; +const REGISTRATION_DESCRIPTOR_TAG = 0x05; + +// The 'HDMV' and 'HDPR' format identifiers, which mark a program as Blu-ray-derived +const HDMV_FORMAT_IDENTIFIER = 0x48444d56; +const HDPR_FORMAT_IDENTIFIER = 0x48445052; + +// 'DTS1', 'DTS2' and 'DTS3' all share this prefix +const DTS_FORMAT_IDENTIFIER_PREFIX = 0x44545300; + +// Stream types that only mean DTS within a Blu-ray-derived program +const BLU_RAY_DTS_STREAM_TYPES = new Set([ + MpegTsStreamType.BLU_RAY_DTS_HD, + MpegTsStreamType.BLU_RAY_DTS_HD_MASTER, + MpegTsStreamType.BLU_RAY_DTS_EXPRESS_SECONDARY, +]); + type ElementaryStream = { demuxer: MpegTsDemuxer; pid: number; @@ -110,6 +133,7 @@ type ElementaryStream = { codec: AudioCodec; decoderConfig: AudioDecoderConfig | null; aacCodecInfo: AacCodecInfo | null; + dtsFormat: DtsFourCc | null; numberOfChannels: number; sampleRate: number; }; @@ -294,7 +318,27 @@ export class MpegTsDemuxer extends Demuxer { // "The remaining 10 bits specify the number of bytes of the descriptors immediately following the // program_info_length field" const programInfoLength = bitstream.readBits(10); - bitstream.skipBits(8 * programInfoLength); + + // A few stream types only mean what Blu-ray says they mean, and the program's registration + // descriptor is what tells us we're looking at a Blu-ray-derived stream. + const programInfoEndPos = bitstream.pos + 8 * programInfoLength; + let isBluRayProgram = false; + + while (bitstream.pos < programInfoEndPos) { + const descriptorTag = bitstream.readBits(8); + const descriptorLength = bitstream.readBits(8); + const descriptorEndPos = bitstream.pos + 8 * descriptorLength; + + if (descriptorTag === REGISTRATION_DESCRIPTOR_TAG && descriptorLength >= 4) { + const formatIdentifier = bitstream.readBits(32); + isBluRayProgram ||= formatIdentifier === HDMV_FORMAT_IDENTIFIER + || formatIdentifier === HDPR_FORMAT_IDENTIFIER; + } + + bitstream.pos = descriptorEndPos; + } + + bitstream.pos = programInfoEndPos; while (8 * (sectionLength + BYTES_BEFORE_SECTION_LENGTH) - bitstream.pos > BITS_IN_CRC_32) { const streamType = bitstream.readBits(8); @@ -304,24 +348,40 @@ export class MpegTsDemuxer extends Demuxer { bitstream.skipBits(6); const esInfoLength = bitstream.readBits(10); - // Check ES descriptors to detect AC-3/E-AC-3 in System B + // Check ES descriptors to detect AC-3/E-AC-3/DTS in System B const esInfoEndPos = bitstream.pos + 8 * esInfoLength; let hasAc3Descriptor = false; let hasEac3Descriptor = false; + let hasDtsDescriptor = false; while (bitstream.pos < esInfoEndPos) { const descriptorTag = bitstream.readBits(8); const descriptorLength = bitstream.readBits(8); + const descriptorEndPos = bitstream.pos + 8 * descriptorLength; + if (descriptorTag === 0x6a) { hasAc3Descriptor = true; } else if (descriptorTag === 0x7a || descriptorTag === 0xcc) { hasEac3Descriptor = true; + } else if (descriptorTag === 0x7b) { + hasDtsDescriptor = true; + } else if (descriptorTag === REGISTRATION_DESCRIPTOR_TAG && descriptorLength >= 4) { + // DTS uses 'DTS1', 'DTS2' and 'DTS3' to also signal the stream's frame size, + // which we work out from the bitstream anyway + const formatIdentifier = bitstream.readBits(32); + hasDtsDescriptor ||= (formatIdentifier & 0xffffff00) === DTS_FORMAT_IDENTIFIER_PREFIX; } - bitstream.skipBits(8 * descriptorLength); + + bitstream.pos = descriptorEndPos; } let info: ElementaryStream['info'] | null = null; - switch (streamType) { + // Blu-ray has its own stream types for the DTS extensions + const effectiveStreamType = isBluRayProgram && BLU_RAY_DTS_STREAM_TYPES.has(streamType) + ? MpegTsStreamType.DTS + : streamType; + + switch (effectiveStreamType) { case MpegTsStreamType.AVC: case MpegTsStreamType.HEVC: { const codec = streamType === MpegTsStreamType.AVC ? 'avc' : 'hevc'; @@ -350,21 +410,23 @@ export class MpegTsDemuxer extends Demuxer { case MpegTsStreamType.MP3_MPEG2: case MpegTsStreamType.AAC: case MpegTsStreamType.AC3_SYSTEM_A: - case MpegTsStreamType.EAC3_SYSTEM_A: { + case MpegTsStreamType.EAC3_SYSTEM_A: + case MpegTsStreamType.DTS: + case MpegTsStreamType.DTS_ATSC: { let codec: AudioCodec; if ( - streamType === MpegTsStreamType.MP3_MPEG1 - || streamType === MpegTsStreamType.MP3_MPEG2 + effectiveStreamType === MpegTsStreamType.MP3_MPEG1 + || effectiveStreamType === MpegTsStreamType.MP3_MPEG2 ) { codec = 'mp3'; - } else if (streamType === MpegTsStreamType.AAC) { + } else if (effectiveStreamType === MpegTsStreamType.AAC) { codec = 'aac'; - } else if (streamType === MpegTsStreamType.AC3_SYSTEM_A) { + } else if (effectiveStreamType === MpegTsStreamType.AC3_SYSTEM_A) { codec = 'ac3'; - } else if (streamType === MpegTsStreamType.EAC3_SYSTEM_A) { + } else if (effectiveStreamType === MpegTsStreamType.EAC3_SYSTEM_A) { codec = 'eac3'; } else { - throw new Error('Unreachable.'); + codec = 'dts'; } info = { @@ -372,6 +434,7 @@ export class MpegTsDemuxer extends Demuxer { codec, decoderConfig: null, aacCodecInfo: null, + dtsFormat: null, numberOfChannels: -1, sampleRate: -1, }; @@ -384,6 +447,7 @@ export class MpegTsDemuxer extends Demuxer { codec: 'eac3', decoderConfig: null, aacCodecInfo: null, + dtsFormat: null, numberOfChannels: -1, sampleRate: -1, }; @@ -393,6 +457,17 @@ export class MpegTsDemuxer extends Demuxer { codec: 'ac3', decoderConfig: null, aacCodecInfo: null, + dtsFormat: null, + numberOfChannels: -1, + sampleRate: -1, + }; + } else if (hasDtsDescriptor) { + info = { + type: 'audio', + codec: 'dts', + decoderConfig: null, + aacCodecInfo: null, + dtsFormat: null, numberOfChannels: -1, sampleRate: -1, }; @@ -678,6 +753,22 @@ export class MpegTsDemuxer extends Demuxer { elementaryStream.info.numberOfChannels = getEac3ChannelCount(frameInfo); elementaryStream.info.sampleRate = sampleRate; + } else if (elementaryStream.info.codec === 'dts') { + const frameInfo = parseDtsFrame(context.suppliedPacket.data); + if (!frameInfo) { + throw new Error( + 'Invalid DTS audio stream; could not read frame header from first packet.', + ); + } + + elementaryStream.info.numberOfChannels = frameInfo.numberOfChannels; + elementaryStream.info.sampleRate = frameInfo.sampleRate; + + if (frameInfo.core) { + // Telling the two extension-only variants apart would need us to work out + // whether the asset holds XLL or LBR data, so we leave those unlabeled + elementaryStream.info.dtsFormat = frameInfo.hasExtensions ? 'dtsh' : 'dtsc'; + } } else { throw new Error('Unhandled.'); } @@ -687,6 +778,7 @@ export class MpegTsDemuxer extends Demuxer { codec: elementaryStream.info.codec, codecDescription: null, aacCodecInfo: elementaryStream.info.aacCodecInfo, + dtsFormat: elementaryStream.info.dtsFormat, }), numberOfChannels: elementaryStream.info.numberOfChannels, sampleRate: elementaryStream.info.sampleRate, @@ -2315,6 +2407,89 @@ class PacketReadingContext { samplesPerFrame * TIMESCALE / elementaryStream.info.sampleRate, ); return this.supplyPacket(remaining, duration); + } else if (codec === 'dts') { + if (byte !== 0x7f && byte !== 0x64) { + continue; + } + + this.skip(-1); + const possibleSyncPos = this.currentPos; + + let remaining = this.ensureBuffered(DTS_CORE_FRAME_HEADER_SIZE); + if (remaining instanceof Promise) remaining = await remaining; + + if (remaining < DTS_CORE_FRAME_HEADER_SIZE) { + return; + } + + const headerBytes = this.readBytes(DTS_CORE_FRAME_HEADER_SIZE); + const core = parseDtsCoreFrameHeader(headerBytes); + let leadingExss = core ? null : parseDtsExssHeader(headerBytes); + + if (!core && !leadingExss) { + this.seekTo(possibleSyncPos + 1); + continue; + } + + if (leadingExss && !leadingExss.asset) { + // The static fields and the asset descriptor reach past the bytes we read for the + // core header, so take another look with the substream's whole header available + this.seekTo(possibleSyncPos); + + const headerBound = Math.min(leadingExss.frameSize, DTS_EXSS_MAX_HEADER_SIZE); + let remaining = this.ensureBuffered(headerBound); + if (remaining instanceof Promise) remaining = await remaining; + + leadingExss = parseDtsExssHeader(this.readBytes(remaining)) ?? leadingExss; + } + + let frameSize = core ? core.frameSize : leadingExss!.frameSize; + + // Extension-only streams put exactly one substream in each frame, so it's only a core + // frame that needs us to go looking for the extensions riding along with it. The core is + // padded out to a 4-byte boundary before the first of them starts. + if (core) { + let nextSubstreamPos = Math.ceil(core.frameSize / 4) * 4; + + while (true) { + this.seekTo(possibleSyncPos); + + const neededBytes = nextSubstreamPos + DTS_EXSS_HEADER_PREFIX_SIZE; + let remaining = this.ensureBuffered(neededBytes); + if (remaining instanceof Promise) remaining = await remaining; + + if (remaining < neededBytes) { + break; + } + + this.seekTo(possibleSyncPos + nextSubstreamPos); + + const exss = parseDtsExssHeader(this.readBytes(DTS_EXSS_HEADER_PREFIX_SIZE)); + if (!exss) { + break; + } + + nextSubstreamPos += exss.frameSize; + frameSize = nextSubstreamPos; + } + } + + // Only the substream leading the frame declares how many samples the frame holds + const sampleCount = core?.sampleCount ?? leadingExss!.asset?.sampleCount; + if (sampleCount === undefined) { + this.seekTo(possibleSyncPos + 1); + continue; + } + + this.seekTo(possibleSyncPos); + + remaining = this.ensureBuffered(frameSize); + if (remaining instanceof Promise) remaining = await remaining; + + const duration = Math.round( + sampleCount * TIMESCALE / elementaryStream.info.sampleRate, + ); + return this.supplyPacket(remaining, duration); } else { throw new Error('Unhandled.'); } diff --git a/src/mpeg-ts/mpeg-ts-misc.ts b/src/mpeg-ts/mpeg-ts-misc.ts index d19d062..23b2d4d 100644 --- a/src/mpeg-ts/mpeg-ts-misc.ts +++ b/src/mpeg-ts/mpeg-ts-misc.ts @@ -15,6 +15,11 @@ export const enum MpegTsStreamType { AAC = 0x0f, AC3_SYSTEM_A = 0x81, EAC3_SYSTEM_A = 0x87, + DTS = 0x82, + DTS_ATSC = 0x8a, + BLU_RAY_DTS_HD = 0x85, + BLU_RAY_DTS_HD_MASTER = 0x86, + BLU_RAY_DTS_EXPRESS_SECONDARY = 0xa2, PRIVATE_DATA = 0x06, AVC = 0x1b, HEVC = 0x24, diff --git a/src/mpeg-ts/mpeg-ts-muxer.ts b/src/mpeg-ts/mpeg-ts-muxer.ts index d48f99e..8320a7f 100644 --- a/src/mpeg-ts/mpeg-ts-muxer.ts +++ b/src/mpeg-ts/mpeg-ts-muxer.ts @@ -161,7 +161,7 @@ export class MpegTsMuxer extends Muxer { assert(meta?.decoderConfig); const codec = track.source._codec; - assert(codec === 'aac' || codec === 'mp3' || codec === 'ac3' || codec === 'eac3'); + assert(codec === 'aac' || codec === 'mp3' || codec === 'ac3' || codec === 'eac3' || codec === 'dts'); let streamType: MpegTsStreamType; let streamId: number; @@ -186,6 +186,11 @@ export class MpegTsMuxer extends Muxer { streamType = MpegTsStreamType.EAC3_SYSTEM_A; streamId = 0xbd; }; break; + + case 'dts': { + streamType = MpegTsStreamType.DTS; + streamId = 0xbd; + }; break; } const pid = FIRST_TRACK_PID + this.trackDatas.length; @@ -414,7 +419,7 @@ export class MpegTsMuxer extends Muxer { ): Uint8Array { const codec = (trackData.track as OutputAudioTrack).source._codec; - if (codec === 'mp3' || codec === 'ac3' || codec === 'eac3') { + if (codec === 'mp3' || codec === 'ac3' || codec === 'eac3' || codec === 'dts') { // We're good return packet.data; } diff --git a/src/output-format.ts b/src/output-format.ts index 9098282..96186a9 100644 --- a/src/output-format.ts +++ b/src/output-format.ts @@ -1153,7 +1153,7 @@ export class MpegTsOutputFormat extends OutputFormat { getSupportedCodecs(): MediaCodec[] { return [ ...VIDEO_CODECS.filter(codec => ['avc', 'hevc'].includes(codec)), - ...AUDIO_CODECS.filter(codec => ['aac', 'mp3', 'ac3', 'eac3'].includes(codec)), + ...AUDIO_CODECS.filter(codec => ['aac', 'mp3', 'ac3', 'eac3', 'dts'].includes(codec)), ]; } diff --git a/test/node/dts.test.ts b/test/node/dts.test.ts new file mode 100644 index 0000000..6ddfa76 --- /dev/null +++ b/test/node/dts.test.ts @@ -0,0 +1,171 @@ +import { expect, test } from 'vitest'; +import path from 'node:path'; +import { Input } from '../../src/input.js'; +import { BufferSource, FilePathSource } from '../../src/source.js'; +import { ALL_FORMATS } from '../../src/input-format.js'; +import { Output } from '../../src/output.js'; +import { MkvOutputFormat, Mp4OutputFormat, MpegTsOutputFormat, OutputFormat } from '../../src/output-format.js'; +import { BufferTarget } from '../../src/target.js'; +import { Conversion } from '../../src/conversion.js'; +import { EncodedPacketSink } from '../../src/media-sink.js'; +import { EncodedPacket } from '../../src/packet.js'; +import { assert, uint8ArraysAreEqual } from '../../src/misc.js'; + +const __dirname = new URL('.', import.meta.url).pathname; + +const DTSC_FILE = 'toothsome-dts.mp4'; +const ESDS_FILE = 'toothsome-dts-esds.mp4'; + +const MP4_TIMESTAMP_TOLERANCE = 0; +const MATROSKA_TIMESTAMP_TOLERANCE = 1 / 1000; +const MPEG_TS_TIMESTAMP_TOLERANCE = 1 / 90_000; + +test('Read from MP4 with a dtsc sample entry', async () => { + using input = new Input({ + source: new FilePathSource(path.join(__dirname, '..', 'public', DTSC_FILE)), + formats: ALL_FORMATS, + }); + + const track = await input.getPrimaryAudioTrack(); + assert(track); + + const decoderConfig = await track.getDecoderConfig(); + assert(decoderConfig); + + expect(await track.getCodec()).toBe('dts'); + expect(decoderConfig.codec).toBe('dtsc'); + expect(decoderConfig.description).toBeUndefined(); + + // This file declares 48000 in its sample entry, which the ddts box then corrects to the real rate + expect(await track.getSampleRate()).toBe(24000); + expect(await track.getNumberOfChannels()).toBe(2); + + await expectStreamShape(input); +}); + +test('Read from MP4 with an esds sample entry', async () => { + using input = new Input({ + source: new FilePathSource(path.join(__dirname, '..', 'public', ESDS_FILE)), + formats: ALL_FORMATS, + }); + + const track = await input.getPrimaryAudioTrack(); + assert(track); + + const decoderConfig = await track.getDecoderConfig(); + assert(decoderConfig); + + expect(await track.getCodec()).toBe('dts'); + expect(decoderConfig.codec).toBe('dtsc'); + expect(decoderConfig.description).toBeUndefined(); + + expect(await track.getSampleRate()).toBe(24000); + expect(await track.getNumberOfChannels()).toBe(2); + + await expectStreamShape(input); +}); + +test('Transmux dtsc MP4 into MP4', async () => { + await expectTransmuxToPreservePackets(DTSC_FILE, new Mp4OutputFormat(), MP4_TIMESTAMP_TOLERANCE); +}); + +test('Transmux dtsc MP4 into Matroska', async () => { + await expectTransmuxToPreservePackets(DTSC_FILE, new MkvOutputFormat(), MATROSKA_TIMESTAMP_TOLERANCE); +}); + +test('Transmux dtsc MP4 into MPEG-TS', async () => { + await expectTransmuxToPreservePackets(DTSC_FILE, new MpegTsOutputFormat(), MPEG_TS_TIMESTAMP_TOLERANCE); +}); + +test('Transmux esds MP4 into MP4', async () => { + await expectTransmuxToPreservePackets(ESDS_FILE, new Mp4OutputFormat(), MP4_TIMESTAMP_TOLERANCE); +}); + +test('Transmux esds MP4 into Matroska', async () => { + await expectTransmuxToPreservePackets(ESDS_FILE, new MkvOutputFormat(), MATROSKA_TIMESTAMP_TOLERANCE); +}); + +test('Transmux esds MP4 into MPEG-TS', async () => { + await expectTransmuxToPreservePackets(ESDS_FILE, new MpegTsOutputFormat(), MPEG_TS_TIMESTAMP_TOLERANCE); +}); + +const expectStreamShape = async (input: Input) => { + const packets = await readAllPackets(input); + + expect(packets.length).toBe(1805); + expect(packets.every(packet => packet.type === 'key')).toBe(true); + expect(packets.every(packet => packet.data.byteLength === 512)).toBe(true); +}; + +/** Converts the file without re-encoding and checks that every packet comes back out unchanged. */ +const expectTransmuxToPreservePackets = async ( + fileName: string, + outputFormat: OutputFormat, + timestampTolerance: number, +) => { + using originalInput = new Input({ + source: new FilePathSource(path.join(__dirname, '..', 'public', fileName)), + formats: ALL_FORMATS, + }); + + const originalPackets = await readAllPackets(originalInput); + + const output = new Output({ + format: outputFormat, + target: new BufferTarget(), + }); + + const conversion = await Conversion.init({ input: originalInput, output }); + await conversion.execute(); + + const { buffer } = output.target; + assert(buffer); + + using newInput = new Input({ + source: new BufferSource(buffer), + formats: ALL_FORMATS, + }); + + const newTrack = await newInput.getPrimaryAudioTrack(); + assert(newTrack); + + const newDecoderConfig = await newTrack.getDecoderConfig(); + assert(newDecoderConfig); + + expect(await newTrack.getCodec()).toBe('dts'); + expect(await newTrack.getSampleRate()).toBe(24000); + expect(await newTrack.getNumberOfChannels()).toBe(2); + + // Only the MP4 sample entry can carry this, so for the other formats it comes back out of the bitstream + expect(newDecoderConfig.codec).toBe('dtsc'); + + const newPackets = await readAllPackets(newInput); + expectPacketsToMatch(newPackets, originalPackets, timestampTolerance); +}; + +const expectPacketsToMatch = (actual: EncodedPacket[], expected: EncodedPacket[], timestampTolerance: number) => { + expect(actual.length).toBe(expected.length); + + for (let i = 0; i < expected.length; i++) { + const actualPacket = actual[i]!; + const expectedPacket = expected[i]!; + + expect(actualPacket.type).toBe(expectedPacket.type); + expect(uint8ArraysAreEqual(actualPacket.data, expectedPacket.data)).toBe(true); + expect(Math.abs(actualPacket.timestamp - expectedPacket.timestamp)).toBeLessThanOrEqual(timestampTolerance); + } +}; + +const readAllPackets = async (input: Input) => { + const track = await input.getPrimaryAudioTrack(); + assert(track); + + const sink = new EncodedPacketSink(track); + const packets: EncodedPacket[] = []; + + for await (const packet of sink.packets()) { + packets.push(packet); + } + + return packets; +}; diff --git a/test/node/server-extension.test.ts b/test/node/server-extension.test.ts index 88033d2..bc28490 100644 --- a/test/node/server-extension.test.ts +++ b/test/node/server-extension.test.ts @@ -1400,6 +1400,9 @@ describe('Video', async () => { }); describe('Audio', async () => { + /** A DTS bitrate that clears the encoder's per-frame minimum at the 48 kHz stereo the tests here use. */ + const DTS_BITRATE = 768000; + test('Decoder lifecycle', async () => { using input = new Input({ source: new FilePathSource('./test/public/trim-buck-bunny-ffmpeg.ts'), @@ -1726,6 +1729,25 @@ describe('Audio', async () => { }); }); + test('DTS encode & decode', async () => { + await encodeDecodeTest('dts', { bitrate: DTS_BITRATE }, async (packet, meta, i) => { + expect(packet.type).toBe('key'); + expect(packet.duration).toBeCloseTo(512 / 48000); + + if (i === 0) { + expect(meta.decoderConfig).toBeDefined(); + expect(meta.decoderConfig!.codec).toBe('dtsc'); + expect(meta.decoderConfig!.numberOfChannels).toBe(2); + expect(meta.decoderConfig!.sampleRate).toBe(48000); + expect(meta.decoderConfig!.description).toBeUndefined(); + } + }, async (sample) => { + expect(sample.numberOfChannels).toBe(2); + expect(sample.sampleRate).toBe(48000); + expect(sample.duration).toBeCloseTo(512 / 48000); + }); + }); + for (const codec of NON_PCM_AUDIO_CODECS) { test(`${codec} encode & decode, negative timestamps`, async () => { await timestampTest(codec, -1); @@ -1747,7 +1769,10 @@ describe('Audio', async () => { const testDuration = codec !== 'opus'; const testSampleStart = codec !== 'vorbis'; - await encodeDecodeTest(codec, {}, async (packet, _meta, i) => { + // The harness' default bitrate is below what DTS needs to fit its per-channel side info into a frame + const extraConfig = codec === 'dts' ? { bitrate: DTS_BITRATE } : {}; + + await encodeDecodeTest(codec, extraConfig, async (packet, _meta, i) => { if (i === 0) { expect(packet.timestamp).toBe(startTimestamp); } else if (testDuration) { diff --git a/test/public/toothsome-dts-esds.mp4 b/test/public/toothsome-dts-esds.mp4 new file mode 100644 index 0000000..ab704ed Binary files /dev/null and b/test/public/toothsome-dts-esds.mp4 differ diff --git a/test/public/toothsome-dts.mp4 b/test/public/toothsome-dts.mp4 new file mode 100644 index 0000000..f7e1ade Binary files /dev/null and b/test/public/toothsome-dts.mp4 differ