mirror of
https://github.com/arcodange-org/mediabunny.git
synced 2026-09-27 10:53:50 +02:00
Define DTS codec, add muxing and demuxing support across ISOBMFF, Matroska, and MPEG-TS, add DTS encode/decode support to @mediabunny/server
This commit is contained in:
@@ -162,6 +162,7 @@ export default withMermaid({
|
||||
{ text: 'FLAC', link: '/codec-registry/flac' },
|
||||
{ text: 'AC-3', link: '/codec-registry/ac3' },
|
||||
{ text: 'E-AC-3', link: '/codec-registry/eac3' },
|
||||
{ text: 'DTS', link: '/codec-registry/dts' },
|
||||
{ text: 'Linear PCM', link: '/codec-registry/pcm' },
|
||||
{ text: 'μ-law PCM', link: '/codec-registry/ulaw' },
|
||||
{ text: 'A-law PCM', link: '/codec-registry/alaw' },
|
||||
|
||||
@@ -0,0 +1,44 @@
|
||||
---
|
||||
description: DTS (DTS Coherent Acoustics) audio codec definition, defining legal codec strings, decoder configs, and packet data formats.
|
||||
---
|
||||
|
||||
<script setup>
|
||||
import { VPBadge } from 'vitepress/theme'
|
||||
</script>
|
||||
|
||||
<VPBadge type="info" text="Audio codec" />
|
||||
|
||||
# DTS codec registration
|
||||
|
||||
## Description
|
||||
|
||||
The DTS audio codec (DTS Coherent Acoustics), specified in [ETSI TS 102 114](https://www.etsi.org/deliver/etsi_ts/102100_102199/102114/01.06.01_60/ts_102114v010601p.pdf).
|
||||
|
||||
## Codec ID
|
||||
|
||||
```ts
|
||||
'dts'
|
||||
```
|
||||
|
||||
## `EncodedPacket` data
|
||||
|
||||
The packet's data must be a core substream frame as defined in Section 5 of [ETSI TS 102 114](https://www.etsi.org/deliver/etsi_ts/102100_102199/102114/01.06.01_60/ts_102114v010601p.pdf), beginning with the sync word `0x7FFE8001`, followed by any number of extension substreams as defined in Section 7.4.1 of [ETSI TS 102 114](https://www.etsi.org/deliver/etsi_ts/102100_102199/102114/01.06.01_60/ts_102114v010601p.pdf), each beginning with the sync word `0x64582025`. The core substream frame may be omitted.
|
||||
|
||||
The bitstream must use 16-bit big-endian packing.
|
||||
|
||||
## `EncodedPacket` type
|
||||
|
||||
The packet's type is always `'key'`.
|
||||
|
||||
## `AudioDecoderConfig` codec string
|
||||
|
||||
The codec string must be one of the four four-character codes identifying which substreams the bitstream is built from:
|
||||
|
||||
- `'dtsc'` - a core substream only
|
||||
- `'dtsh'` - a core substream with extension substreams
|
||||
- `'dtsl'` - extension substreams carrying lossless audio, with no core substream
|
||||
- `'dtse'` - DTS Express
|
||||
|
||||
## `AudioDecoderConfig` description
|
||||
|
||||
`description` is not used for this codec.
|
||||
@@ -26,6 +26,7 @@ The registry is an extension of the [WebCodecs Codec Registry](https://www.w3.or
|
||||
- [FLAC](./flac)
|
||||
- [AC-3](./ac3)
|
||||
- [E-AC-3](./eac3)
|
||||
- [DTS](./dts)
|
||||
- [Linear PCM](./pcm)
|
||||
- [μ-law PCM](./ulaw)
|
||||
- [A-law PCM](./alaw)
|
||||
|
||||
@@ -8,7 +8,7 @@ By default, Mediabunny requires a browser environment for full access to decoder
|
||||
|
||||
Features added by this package include:
|
||||
- Video decoders and encoders for AVC (H.264), HEVC (H.265), VP8, VP9, AV1, and ProRes. Supports both length-prefixed and Annex B AVC/HEVC as well as transparent video via VP9 and ProRes.
|
||||
- Audio decoders and encoders for AAC, MP3, Vorbis, Opus, FLAC, AC-3 and E-AC-3. Supports AAC in both AAC and ADTS formats.
|
||||
- Audio decoders and encoders for AAC, MP3, Vorbis, Opus, FLAC, AC-3, E-AC-3 and DTS. Supports AAC in both AAC and ADTS formats.
|
||||
- Video frame transformation support (resize, rotate, crop)
|
||||
- Automatic hardware acceleration on all platforms (macOS, Linux, Windows)
|
||||
- Built-in multithreading
|
||||
|
||||
@@ -64,7 +64,7 @@ Mediabunny's simple yet flexible API provides a modern alternative to traditiona
|
||||
|
||||
The extension enables:
|
||||
- Video decoders and encoders for AVC (H.264), HEVC (H.265), VP8, VP9, and AV1. Supports both length-prefixed and Annex B AVC/HEVC as well as transparent video via VP9.
|
||||
- Audio decoders and encoders for AAC, MP3, Vorbis, Opus, FLAC, AC-3 and E-AC-3. Supports AAC in both AAC and ADTS formats.
|
||||
- Audio decoders and encoders for AAC, MP3, Vorbis, Opus, FLAC, AC-3, E-AC-3 and DTS. Supports AAC in both AAC and ADTS formats.
|
||||
- Video frame transformation support (resize, rotate, crop)
|
||||
- Automatic hardware acceleration on all platforms (macOS, Linux, Windows)
|
||||
- Built-in multithreading
|
||||
|
||||
@@ -51,6 +51,7 @@ Mediabunny ships with built-in decoders and encoders for all audio PCM codecs, m
|
||||
- `'flac'` - Free Lossless Audio Codec (FLAC) [^flac]
|
||||
- `'ac3'` - Dolby Digital (AC-3) [^ac3]
|
||||
- `'eac3'` - Dolby Digital Plus (E-AC-3) [^ac3]
|
||||
- `'dts'` - DTS Coherent Acoustics [^dts]
|
||||
- `'pcm-u8'` - 8-bit unsigned PCM
|
||||
- `'pcm-s8'` - 8-bit signed PCM
|
||||
- `'pcm-s16'` - 16-bit little-endian signed PCM
|
||||
@@ -89,6 +90,7 @@ Not all codecs can be used with all containers. The following table specifies th
|
||||
| `'flac'` | ✓ | ✓ | ✓ | | | | | | ✓ | |
|
||||
| `'ac3'` | ✓ | ✓ | ✓ | | | | | | | ✓ |
|
||||
| `'eac3'` | ✓ | ✓ | ✓ | | | | | | | ✓ |
|
||||
| `'dts'` | ✓ | ✓ | ✓ | | | | | | | ✓ |
|
||||
| `'pcm-u8'` | | ✓ | ✓ | | | | ✓ | | | |
|
||||
| `'pcm-s8'` | | ✓ | | | | | | | | |
|
||||
| `'pcm-s16'` | ✓ | ✓ | ✓ | | | | ✓ | | | |
|
||||
@@ -112,6 +114,7 @@ For HLS, the supported codecs depend on the segment format chosen.
|
||||
[^mp3]: MP3 encoding is not supported by WebCodecs. You can polyfill it with the [`@mediabunny/mp3-encoder`](./extensions/mp3-encoder) extension package.
|
||||
[^flac]: FLAC encoding is not supported by WebCodecs. You can polyfill it with the [`@mediabunny/flac-encoder`](./extensions/flac-encoder) extension package.
|
||||
[^ac3]: AC-3 and E-AC-3 are not natively supported by WebCodecs. To encode or decode these codecs, you can use the [`@mediabunny/ac3`](./extensions/ac3) extension package.
|
||||
[^dts]: DTS is not supported by WebCodecs. The [`@mediabunny/server`](./extensions/server) extension package provides both decoding and encoding support for server-side environments.
|
||||
[^webm]: WebM only supports a small subset of the codecs supported by Matroska. However, this library can technically read all codecs from a WebM that are supported by Matroska.
|
||||
[^webvtt]: WebVTT can only be written, not read.
|
||||
|
||||
@@ -127,7 +130,7 @@ import { canEncode } from 'mediabunny';
|
||||
canEncode('avc'); // => Promise<boolean>
|
||||
canEncode('opus'); // => Promise<boolean>
|
||||
```
|
||||
Video codecs are checked using 1280x720 @1Mbps, while audio codecs are checked using 2 channels, 48 kHz @128kbps.
|
||||
Video codecs are checked using 1280x720, while audio codecs are checked using 2 channels at 48 kHz.
|
||||
|
||||
You can also check encodability using specific configurations:
|
||||
```ts
|
||||
|
||||
@@ -25,7 +25,8 @@ export class NodeAvAudioDecoder extends CustomAudioDecoder {
|
||||
|| codec === 'vorbis'
|
||||
|| codec === 'flac'
|
||||
|| codec === 'ac3'
|
||||
|| codec === 'eac3';
|
||||
|| codec === 'eac3'
|
||||
|| codec === 'dts';
|
||||
}
|
||||
|
||||
async init(): Promise<void> {
|
||||
|
||||
@@ -15,7 +15,7 @@ import {
|
||||
EncodedPacket,
|
||||
} from 'mediabunny';
|
||||
import * as NodeAv from 'node-av';
|
||||
import { CODEC_TO_CODEC_ID, getChannelLayout } from './misc';
|
||||
import { CODEC_TO_CODEC_ID, getChannelLayout, getDtsChannelLayout } from './misc';
|
||||
import { assert, toUint8Array } from '../../../src/misc';
|
||||
import { copyAudioSampleToAvFrame, AvFrameAudioSampleResource } from './audio-sample';
|
||||
import {
|
||||
@@ -29,6 +29,11 @@ const AAC_SAMPLE_RATES
|
||||
const OPUS_SAMPLE_RATES = [8000, 12000, 16000, 24000, 48000];
|
||||
const MP3_SAMPLE_RATES = [8000, 11025, 12000, 16000, 22050, 24000, 32000, 44100, 48000];
|
||||
const AC3_SAMPLE_RATES = [32000, 44100, 48000];
|
||||
const DTS_SAMPLE_RATES = [8000, 11025, 12000, 16000, 22050, 24000, 32000, 44100, 48000];
|
||||
|
||||
/** DTS only has layouts for these channel counts; 3 and 7 have no mapping. */
|
||||
const DTS_CHANNEL_COUNTS = [1, 2, 4, 5, 6];
|
||||
const DTS_MAX_FRAME_SIZE = 16384;
|
||||
|
||||
const FRAME_SIZE_FALLBACK = 1024; // Just 'cause
|
||||
|
||||
@@ -72,6 +77,10 @@ export class NodeAvAudioEncoder extends CustomAudioEncoder {
|
||||
) || (
|
||||
codec === 'eac3' && numberOfChannels >= 1 && numberOfChannels <= 16
|
||||
&& AC3_SAMPLE_RATES.includes(sampleRate)
|
||||
) || (
|
||||
codec === 'dts' && DTS_CHANNEL_COUNTS.includes(numberOfChannels)
|
||||
&& DTS_SAMPLE_RATES.includes(sampleRate)
|
||||
&& dtsBitrateFits(resolveBitrate(config, codec), sampleRate, numberOfChannels)
|
||||
);
|
||||
}
|
||||
|
||||
@@ -107,21 +116,26 @@ export class NodeAvAudioEncoder extends CustomAudioEncoder {
|
||||
}
|
||||
|
||||
codecContext.sampleRate = this.config.sampleRate;
|
||||
codecContext.channelLayout = getChannelLayout(this.config.numberOfChannels);
|
||||
codecContext.channelLayout = this.codec === 'dts'
|
||||
? getDtsChannelLayout(this.config.numberOfChannels)
|
||||
: getChannelLayout(this.config.numberOfChannels);
|
||||
codecContext.codecType = NodeAv.AVMEDIA_TYPE_AUDIO;
|
||||
codecContext.codecId = CODEC_TO_CODEC_ID[this.codec]!;
|
||||
codecContext.sampleFormat = sampleFormat;
|
||||
codecContext.timeBase = new NodeAv.Rational(1, this.config.sampleRate);
|
||||
codecContext.bitRate = BigInt(
|
||||
this.config.bitrate ?? new Quality('medium')._toAudioBitrate(this.codec) ?? 0,
|
||||
);
|
||||
codecContext.bitRate = BigInt(resolveBitrate(this.config, this.codec));
|
||||
|
||||
if (this.config.bitrateMode === 'constant') {
|
||||
codecContext.rcMinRate = codecContext.bitRate;
|
||||
codecContext.rcMaxRate = codecContext.bitRate;
|
||||
}
|
||||
|
||||
const ret = await codecContext.open2();
|
||||
// libav's DTS encoder is marked experimental, so it refuses to open at the default compliance level
|
||||
const options = this.codec === 'dts'
|
||||
? NodeAv.Dictionary.fromObject({ strict: -2 })
|
||||
: null;
|
||||
|
||||
const ret = await codecContext.open2(this.avCodec, options);
|
||||
NodeAv.FFmpegError.throwIfError(ret, 'Open codec context');
|
||||
|
||||
this.codecContext = codecContext;
|
||||
@@ -172,7 +186,9 @@ export class NodeAvAudioEncoder extends CustomAudioEncoder {
|
||||
this.resampler = new NodeAv.SoftwareResampleContext();
|
||||
this.resamplerInputSampleRate = this.frame.sampleRate;
|
||||
|
||||
const outLayout = getChannelLayout(this.codecContext.channels);
|
||||
// Resample straight to whatever layout the encoder was opened with, since not every codec accepts
|
||||
// the layout that getChannelLayout hands out for a given channel count
|
||||
const outLayout = this.codecContext.channelLayout;
|
||||
const inLayout = getChannelLayout(this.frame.channels);
|
||||
|
||||
const ret = this.resampler.allocSetOpts2(
|
||||
@@ -400,3 +416,25 @@ export class NodeAvAudioEncoder extends CustomAudioEncoder {
|
||||
this.resampler?.free();
|
||||
}
|
||||
}
|
||||
|
||||
const resolveBitrate = (config: AudioEncoderConfig, codec: AudioCodec) => {
|
||||
return config.bitrate ?? new Quality('medium')._toAudioBitrate(codec) ?? 0;
|
||||
};
|
||||
|
||||
/**
|
||||
* The DTS encoder needs each frame to be big enough to hold the per-channel side info and rejects the whole
|
||||
* configuration when it isn't, so this mirrors the check from FFmpeg's dcaenc.c.
|
||||
*/
|
||||
const dtsBitrateFits = (bitrate: number, sampleRate: number, numberOfChannels: number) => {
|
||||
if (bitrate < 32000 || bitrate > 3840000) {
|
||||
return false;
|
||||
}
|
||||
|
||||
const hasLfe = numberOfChannels === 6;
|
||||
const fullbandChannels = numberOfChannels - (hasLfe ? 1 : 0);
|
||||
|
||||
const frameBits = 32 * Math.ceil(Math.ceil(bitrate * 512 / sampleRate) / 32);
|
||||
const minFrameBits = 132 + (493 + 28 * 32) * fullbandChannels + (hasLfe ? 72 : 0);
|
||||
|
||||
return frameBits >= minFrameBits && frameBits <= 8 * DTS_MAX_FRAME_SIZE;
|
||||
};
|
||||
|
||||
@@ -25,6 +25,7 @@ export const CODEC_TO_CODEC_ID: Partial<Record<MediaCodec, NodeAv.AVCodecID>> =
|
||||
flac: NodeAv.AV_CODEC_ID_FLAC,
|
||||
ac3: NodeAv.AV_CODEC_ID_AC3,
|
||||
eac3: NodeAv.AV_CODEC_ID_EAC3,
|
||||
dts: NodeAv.AV_CODEC_ID_DTS,
|
||||
};
|
||||
|
||||
let cachedHardwareContext: NodeAv.HardwareContext | null | undefined = undefined;
|
||||
@@ -267,3 +268,21 @@ export const getChannelLayout = (numChannels: number): NodeAv.ChannelLayout => {
|
||||
default: return { nbChannels: numChannels, order: NodeAv.AV_CHANNEL_ORDER_UNSPEC, mask: 0n };
|
||||
}
|
||||
};
|
||||
|
||||
/** FL + FR + SL + SR, which libav has no constant for. */
|
||||
const QUAD_SIDE_MASK = 0x603n;
|
||||
|
||||
/**
|
||||
* DTS insists on the side-based surround layouts and rejects the back-based ones that {@link getChannelLayout}
|
||||
* hands out, so it gets its own mapping.
|
||||
*/
|
||||
export const getDtsChannelLayout = (numChannels: number): NodeAv.ChannelLayout => {
|
||||
switch (numChannels) {
|
||||
case 1: return NodeAv.AV_CHANNEL_LAYOUT_MONO;
|
||||
case 2: return NodeAv.AV_CHANNEL_LAYOUT_STEREO;
|
||||
case 4: return { nbChannels: 4, order: NodeAv.AV_CHANNEL_ORDER_NATIVE, mask: QUAD_SIDE_MASK };
|
||||
case 5: return NodeAv.AV_CHANNEL_LAYOUT_5POINT0;
|
||||
case 6: return NodeAv.AV_CHANNEL_LAYOUT_5POINT1;
|
||||
default: return { nbChannels: numChannels, order: NodeAv.AV_CHANNEL_ORDER_UNSPEC, mask: 0n };
|
||||
}
|
||||
};
|
||||
|
||||
+494
-1
@@ -6,7 +6,7 @@
|
||||
* file, You can obtain one at https://mozilla.org/MPL/2.0/.
|
||||
*/
|
||||
|
||||
import { AVC_LEVEL_TABLE, VideoCodec, VP9_LEVEL_TABLE } from './codec';
|
||||
import { AVC_LEVEL_TABLE, DtsFourCc, VideoCodec, VP9_LEVEL_TABLE } from './codec';
|
||||
import {
|
||||
assert,
|
||||
assertNever,
|
||||
@@ -24,6 +24,7 @@ import {
|
||||
toUint8Array,
|
||||
getChromiumVersion,
|
||||
isChromium,
|
||||
popcount,
|
||||
setUint24,
|
||||
} from './misc';
|
||||
import { Logging } from './logging';
|
||||
@@ -3176,3 +3177,495 @@ export const getEac3ChannelCount = (config: Eac3FrameInfo): number => {
|
||||
|
||||
return channels;
|
||||
};
|
||||
|
||||
// ============================================================================
|
||||
// DTS Parsing
|
||||
// Reference: ETSI TS 102 114 V1.6.1
|
||||
// ============================================================================
|
||||
|
||||
/** Core substream sync word, in the 16-bit big-endian packing that the registry mandates. */
|
||||
export const DTS_CORE_SYNC_WORD = 0x7ffe8001;
|
||||
|
||||
/** Extension substream sync word. Section 7.4.1 */
|
||||
export const DTS_EXSS_SYNC_WORD = 0x64582025;
|
||||
|
||||
/** The core frame header never reaches beyond this many bytes. */
|
||||
export const DTS_CORE_FRAME_HEADER_SIZE = 18;
|
||||
|
||||
/** An extension substream always declares its own size within this many bytes. */
|
||||
export const DTS_EXSS_HEADER_PREFIX_SIZE = 10;
|
||||
|
||||
/** The largest nuExtSSHeaderSize can get, and therefore how far the asset descriptors can reach. */
|
||||
export const DTS_EXSS_MAX_HEADER_SIZE = 4096;
|
||||
|
||||
/** Number of PCM samples in one core PCM block; the core codes its length as a count of these. */
|
||||
export const DTS_PCM_BLOCK_SAMPLES = 32;
|
||||
|
||||
/** Size of the DTSSpecificBox (ddts) payload in bytes. */
|
||||
export const DTS_SPECIFIC_BOX_SIZE = 20;
|
||||
|
||||
/** Number of PCM blocks that a core frame's length must be a multiple of. */
|
||||
const DTS_SUBBAND_SAMPLES = 8;
|
||||
|
||||
/** Core sample rates indexed by SFREQ. Zeroes mark invalid codes. Table 5-4 */
|
||||
const DTS_CORE_SAMPLE_RATES = [
|
||||
0, 8000, 16000, 32000, 0, 0, 11025, 22050,
|
||||
44100, 0, 0, 12000, 24000, 48000, 96000, 192000,
|
||||
];
|
||||
|
||||
/**
|
||||
* Core bit rates in bps indexed by RATE, where a zero means the code isn't a constant rate. Table 5-7
|
||||
*
|
||||
* Note that FFmpeg's ff_dca_bit_rates has 896000 where the spec has 960, and defines rates for codes 25 to 28
|
||||
* which this revision of the spec calls invalid. We keep the latter, since they cost nothing and some content
|
||||
* predating the spec revision uses them.
|
||||
*/
|
||||
const DTS_CORE_BIT_RATES = [
|
||||
32000, 56000, 64000, 96000, 112000, 128000, 192000, 224000,
|
||||
256000, 320000, 384000, 448000, 512000, 576000, 640000, 768000,
|
||||
960000, 1024000, 1152000, 1280000, 1344000, 1408000, 1411200, 1472000,
|
||||
1536000, 1920000, 2048000, 3072000, 3840000, 0, 0, 0,
|
||||
];
|
||||
|
||||
/** Source PCM resolutions in bits indexed by PCMR. Zeroes mark invalid codes. */
|
||||
const DTS_PCM_RESOLUTIONS = [16, 16, 20, 20, 0, 24, 24, 0];
|
||||
|
||||
/** Channel counts indexed by AMODE, not counting LFE. */
|
||||
const DTS_AMODE_CHANNEL_COUNTS = [1, 2, 2, 2, 2, 3, 3, 4, 4, 5, 6, 6, 6, 7, 8, 8];
|
||||
|
||||
/**
|
||||
* Speaker layout masks indexed by AMODE, expressed with the same bits as `ChannelLayout` in the DTSSpecificBox
|
||||
* and `nuSpkrActivityMask` in an extension substream asset descriptor.
|
||||
*/
|
||||
const DTS_AMODE_CHANNEL_LAYOUTS = [
|
||||
0x0001, 0x0002, 0x0002, 0x0002, 0x0002, 0x0003, 0x0012, 0x0013,
|
||||
0x0006, 0x0007, 0x0206, 0x0143, 0x0053, 0x0207, 0x0246, 0x0217,
|
||||
];
|
||||
|
||||
/** The LFE1 speaker bit in a channel layout mask. */
|
||||
const DTS_CHANNEL_LAYOUT_LFE1 = 0x0008;
|
||||
|
||||
/** The channel layout bits that stand for a pair of speakers rather than a single one. */
|
||||
const DTS_CHANNEL_LAYOUT_PAIR_MASK = 0xae66;
|
||||
|
||||
/** Reference clock rates indexed by nuRefClockCode. The last code is unused. Table 7-3 */
|
||||
const DTS_EXSS_REF_CLOCKS = [32000, 44100, 48000, 0];
|
||||
|
||||
/** Sample rates used by extension substream assets, indexed by nuMaxSampleRate. */
|
||||
const DTS_EXSS_SAMPLE_RATES = [
|
||||
8000, 16000, 32000, 64000, 128000, 22050, 44100, 88200,
|
||||
176400, 352800, 12000, 24000, 48000, 96000, 192000, 384000,
|
||||
];
|
||||
|
||||
/** Frame durations that the DTSSpecificBox can express, indexed by FrameDuration. */
|
||||
const DTS_SPECIFIC_BOX_FRAME_DURATIONS = [512, 1024, 2048, 4096];
|
||||
|
||||
export type DtsCoreFrameInfo = {
|
||||
/** Size of the core substream frame in bytes */
|
||||
frameSize: number;
|
||||
sampleRate: number;
|
||||
numberOfChannels: number;
|
||||
/** Number of PCM samples the frame decodes to */
|
||||
sampleCount: number;
|
||||
/** Speaker layout mask */
|
||||
channelLayout: number;
|
||||
/** Audio channel arrangement (AMODE) */
|
||||
amode: number;
|
||||
/** Whether an LFE channel is present */
|
||||
lfePresent: boolean;
|
||||
/** Constant bit rate in bps, or 0 when the stream doesn't run at one */
|
||||
bitRate: number;
|
||||
/** Source PCM resolution in bits */
|
||||
pcmResolution: number;
|
||||
};
|
||||
|
||||
export type DtsExssAssetInfo = {
|
||||
sampleRate: number;
|
||||
numberOfChannels: number;
|
||||
/** Number of PCM samples the frame decodes to */
|
||||
sampleCount: number;
|
||||
/** Speaker layout mask, or 0 when the asset doesn't declare one */
|
||||
channelLayout: number;
|
||||
/** Source PCM resolution in bits */
|
||||
pcmResolution: number;
|
||||
};
|
||||
|
||||
export type DtsExssInfo = {
|
||||
/** Size of the extension substream in bytes */
|
||||
frameSize: number;
|
||||
/** Null when this substream omits the static fields that describe the stream */
|
||||
asset: DtsExssAssetInfo | null;
|
||||
};
|
||||
|
||||
export type DtsFrameInfo = {
|
||||
/** Size of the entire frame in bytes, core substream plus any extension substreams */
|
||||
frameSize: number;
|
||||
sampleRate: number;
|
||||
numberOfChannels: number;
|
||||
/** Number of PCM samples the frame decodes to */
|
||||
sampleCount: number;
|
||||
/** Speaker layout mask */
|
||||
channelLayout: number;
|
||||
/** Source PCM resolution in bits */
|
||||
pcmResolution: number;
|
||||
/** Constant bit rate in bps, or 0 when the stream doesn't run at one */
|
||||
bitRate: number;
|
||||
/** The leading core substream, or null for extension-only streams such as DTS Express */
|
||||
core: DtsCoreFrameInfo | null;
|
||||
/** Whether the frame carries extension substreams on top of the core */
|
||||
hasExtensions: boolean;
|
||||
};
|
||||
|
||||
/**
|
||||
* Parse one complete DTS frame, being a core substream frame followed by any number of extension substreams,
|
||||
* or an extension substream on its own. Section 5 and Section 7.4.1
|
||||
*/
|
||||
export const parseDtsFrame = (data: Uint8Array): DtsFrameInfo | null => {
|
||||
const core = parseDtsCoreFrameHeader(data);
|
||||
const view = toDataView(data);
|
||||
|
||||
// The core substream is padded out to a 4-byte boundary before the first extension substream starts
|
||||
let offset = core ? Math.ceil(core.frameSize / 4) * 4 : 0;
|
||||
let firstExss: DtsExssInfo | null = null;
|
||||
|
||||
while (offset + 4 <= data.length && view.getUint32(offset) === DTS_EXSS_SYNC_WORD) {
|
||||
const exss = parseDtsExssHeader(data.subarray(offset));
|
||||
if (!exss) {
|
||||
break;
|
||||
}
|
||||
|
||||
firstExss ??= exss;
|
||||
offset += exss.frameSize;
|
||||
}
|
||||
|
||||
if (core) {
|
||||
// The core describes what every DTS decoder can play back; the extension substreams only build on top of
|
||||
// it, so the core's parameters are the ones we report.
|
||||
return {
|
||||
frameSize: firstExss ? offset : core.frameSize,
|
||||
sampleRate: core.sampleRate,
|
||||
numberOfChannels: core.numberOfChannels,
|
||||
sampleCount: core.sampleCount,
|
||||
channelLayout: core.channelLayout,
|
||||
pcmResolution: core.pcmResolution,
|
||||
bitRate: core.bitRate,
|
||||
core,
|
||||
hasExtensions: firstExss !== null,
|
||||
};
|
||||
}
|
||||
|
||||
if (!firstExss?.asset) {
|
||||
return null;
|
||||
}
|
||||
|
||||
const { asset } = firstExss;
|
||||
|
||||
return {
|
||||
frameSize: offset,
|
||||
sampleRate: asset.sampleRate,
|
||||
numberOfChannels: asset.numberOfChannels,
|
||||
sampleCount: asset.sampleCount,
|
||||
channelLayout: asset.channelLayout,
|
||||
pcmResolution: asset.pcmResolution,
|
||||
bitRate: 0,
|
||||
core: null,
|
||||
hasExtensions: true,
|
||||
};
|
||||
};
|
||||
|
||||
/**
|
||||
* Works out which four-character code describes a packet, or null when the packet doesn't say. Telling 'dtsl'
|
||||
* from 'dtse' would mean working out whether the asset holds XLL or LBR data, which sits behind the speaker
|
||||
* remapping and mixing metadata deep in the asset descriptor, so we don't do it.
|
||||
*/
|
||||
export const extractDtsFourCcFromPacket = (data: Uint8Array): DtsFourCc | null => {
|
||||
const frameInfo = parseDtsFrame(data);
|
||||
if (!frameInfo?.core) {
|
||||
return null;
|
||||
}
|
||||
|
||||
return frameInfo.hasExtensions ? 'dtsh' : 'dtsc';
|
||||
};
|
||||
|
||||
/** Parse the header of a core substream frame. Section 5.3 */
|
||||
export const parseDtsCoreFrameHeader = (data: Uint8Array): DtsCoreFrameInfo | null => {
|
||||
if (data.length < DTS_CORE_FRAME_HEADER_SIZE) {
|
||||
return null;
|
||||
}
|
||||
if (data[0] !== 0x7f || data[1] !== 0xfe || data[2] !== 0x80 || data[3] !== 0x01) {
|
||||
return null;
|
||||
}
|
||||
|
||||
const bitstream = new Bitstream(data);
|
||||
bitstream.skipBits(32); // SYNC
|
||||
|
||||
bitstream.skipBits(1); // FTYPE
|
||||
|
||||
// Terminating frames carry fewer samples than a full PCM block; we don't handle those
|
||||
if (bitstream.readBits(5) !== DTS_PCM_BLOCK_SAMPLES - 1) {
|
||||
return null;
|
||||
}
|
||||
|
||||
const cpf = bitstream.readBits(1);
|
||||
const npcmblocks = bitstream.readBits(7) + 1;
|
||||
if (npcmblocks % DTS_SUBBAND_SAMPLES !== 0) {
|
||||
return null;
|
||||
}
|
||||
|
||||
const frameSize = bitstream.readBits(14) + 1;
|
||||
if (frameSize < 96) {
|
||||
return null;
|
||||
}
|
||||
|
||||
const amode = bitstream.readBits(6);
|
||||
if (amode >= DTS_AMODE_CHANNEL_COUNTS.length) {
|
||||
return null;
|
||||
}
|
||||
|
||||
const sampleRate = DTS_CORE_SAMPLE_RATES[bitstream.readBits(4)]!;
|
||||
if (sampleRate === 0) {
|
||||
return null;
|
||||
}
|
||||
|
||||
const bitRate = DTS_CORE_BIT_RATES[bitstream.readBits(5)]!;
|
||||
|
||||
if (bitstream.readBits(1) !== 0) {
|
||||
return null; // A reserved bit that must be zero, so a cheap way to reject false sync words
|
||||
}
|
||||
|
||||
bitstream.skipBits(1 + 1 + 1 + 1); // DYNF, TIMEF, AUXF, HDCD
|
||||
bitstream.skipBits(3 + 1 + 1); // EXT_AUDIO_ID, EXT_AUDIO, ASPF
|
||||
|
||||
const lff = bitstream.readBits(2);
|
||||
if (lff === 3) {
|
||||
return null;
|
||||
}
|
||||
|
||||
bitstream.skipBits(1); // HFLAG
|
||||
if (cpf) {
|
||||
bitstream.skipBits(16); // HCRC
|
||||
}
|
||||
|
||||
bitstream.skipBits(1 + 4 + 2); // FILTS, VERNUM, CHIST
|
||||
|
||||
const pcmResolution = DTS_PCM_RESOLUTIONS[bitstream.readBits(3)]!;
|
||||
if (pcmResolution === 0) {
|
||||
return null;
|
||||
}
|
||||
|
||||
const lfePresent = lff !== 0;
|
||||
|
||||
return {
|
||||
frameSize,
|
||||
sampleRate,
|
||||
numberOfChannels: DTS_AMODE_CHANNEL_COUNTS[amode]! + (lfePresent ? 1 : 0),
|
||||
sampleCount: npcmblocks * DTS_PCM_BLOCK_SAMPLES,
|
||||
channelLayout: DTS_AMODE_CHANNEL_LAYOUTS[amode]! | (lfePresent ? DTS_CHANNEL_LAYOUT_LFE1 : 0),
|
||||
amode,
|
||||
lfePresent,
|
||||
bitRate,
|
||||
pcmResolution,
|
||||
};
|
||||
};
|
||||
|
||||
/** Parse the header of an extension substream, along with its first audio asset descriptor. Section 7.4.1 */
|
||||
export const parseDtsExssHeader = (data: Uint8Array): DtsExssInfo | null => {
|
||||
if (data.length < DTS_EXSS_HEADER_PREFIX_SIZE) {
|
||||
return null;
|
||||
}
|
||||
if (data[0] !== 0x64 || data[1] !== 0x58 || data[2] !== 0x20 || data[3] !== 0x25) {
|
||||
return null;
|
||||
}
|
||||
|
||||
const bitstream = new Bitstream(data);
|
||||
bitstream.skipBits(32); // SYNC
|
||||
bitstream.skipBits(8); // nuUserDefinedBits
|
||||
|
||||
const extSsIndex = bitstream.readBits(2);
|
||||
const wideHeader = bitstream.readBits(1);
|
||||
const headerSizeBits = 8 + 4 * wideHeader;
|
||||
const frameSizeBits = 16 + 4 * wideHeader;
|
||||
|
||||
bitstream.skipBits(headerSizeBits); // nuExtSSHeaderSize
|
||||
const frameSize = bitstream.readBits(frameSizeBits) + 1;
|
||||
|
||||
// Everything past this point can run off the end of what we were given, in which case the Bitstream keeps
|
||||
// handing out zeroes; the bounds check further down catches that
|
||||
const incomplete: DtsExssInfo = { frameSize, asset: null };
|
||||
|
||||
if (!bitstream.readBits(1)) { // bStaticFieldsPresent
|
||||
return incomplete;
|
||||
}
|
||||
|
||||
const refClock = DTS_EXSS_REF_CLOCKS[bitstream.readBits(2)]!; // nuRefClockCode
|
||||
// The frame duration is a count of reference clock cycles, not of samples
|
||||
const frameDurationCycles = 512 * (bitstream.readBits(3) + 1); // nuExSSFrameDurationCode
|
||||
|
||||
if (bitstream.readBits(1)) { // bTimeStampFlag
|
||||
bitstream.skipBits(32 + 4); // nuTimeStamp, nLSB
|
||||
}
|
||||
|
||||
const numAudioPresentations = bitstream.readBits(3) + 1;
|
||||
const numAssets = bitstream.readBits(3) + 1;
|
||||
|
||||
const activeExssMasks: number[] = [];
|
||||
for (let i = 0; i < numAudioPresentations; i++) {
|
||||
activeExssMasks.push(bitstream.readBits(extSsIndex + 1));
|
||||
}
|
||||
for (const mask of activeExssMasks) {
|
||||
bitstream.skipBits(8 * popcount(mask)); // nuActiveAssetMask
|
||||
}
|
||||
|
||||
if (bitstream.readBits(1)) { // bMixMetadataEnbl
|
||||
bitstream.skipBits(2); // nuMixMetadataAdjLevel
|
||||
const spkrMaskBits = (bitstream.readBits(2) + 1) << 2;
|
||||
const numMixOutConfigs = bitstream.readBits(2) + 1;
|
||||
bitstream.skipBits(numMixOutConfigs * spkrMaskBits); // nuMixOutChMask
|
||||
}
|
||||
|
||||
for (let i = 0; i < numAssets; i++) {
|
||||
bitstream.skipBits(frameSizeBits); // nuAssetFsize
|
||||
}
|
||||
|
||||
// From here on we're inside the first audio asset descriptor
|
||||
bitstream.skipBits(9); // nuAssetDescriptFsize
|
||||
bitstream.skipBits(3); // nuAssetIndex
|
||||
|
||||
if (bitstream.readBits(1)) { // bAssetTypeDescrPresent
|
||||
bitstream.skipBits(4); // nuAssetTypeDescriptor
|
||||
}
|
||||
if (bitstream.readBits(1)) { // bLanguageDescrPresent
|
||||
bitstream.skipBits(24); // LanguageDescriptor
|
||||
}
|
||||
if (bitstream.readBits(1)) { // bInfoTextPresent
|
||||
bitstream.skipBits(8 * (bitstream.readBits(10) + 1)); // nuInfoTextByteSize, InfoTextString
|
||||
}
|
||||
|
||||
const pcmResolution = bitstream.readBits(5) + 1;
|
||||
const sampleRate = DTS_EXSS_SAMPLE_RATES[bitstream.readBits(4)]!;
|
||||
const numberOfChannels = bitstream.readBits(8) + 1;
|
||||
|
||||
let channelLayout = 0;
|
||||
|
||||
if (bitstream.readBits(1)) { // bOne2OneMapChannels2Speakers
|
||||
if (numberOfChannels > 2) {
|
||||
bitstream.skipBits(1); // bEmbeddedStereoFlag
|
||||
}
|
||||
if (numberOfChannels > 6) {
|
||||
bitstream.skipBits(1); // bEmbeddedSixChFlag
|
||||
}
|
||||
|
||||
if (bitstream.readBits(1)) { // bSpkrMaskEnabled
|
||||
const spkrMaskBits = (bitstream.readBits(2) + 1) << 2;
|
||||
channelLayout = bitstream.readBits(spkrMaskBits); // nuSpkrActivityMask
|
||||
}
|
||||
}
|
||||
|
||||
if (refClock === 0 || bitstream.getBitsLeft() < 0) {
|
||||
return incomplete;
|
||||
}
|
||||
|
||||
return {
|
||||
frameSize,
|
||||
asset: {
|
||||
sampleRate,
|
||||
numberOfChannels,
|
||||
sampleCount: Math.round(frameDurationCycles * sampleRate / refClock),
|
||||
channelLayout,
|
||||
pcmResolution,
|
||||
},
|
||||
};
|
||||
};
|
||||
|
||||
export type DtsSpecificBoxInfo = {
|
||||
sampleRate: number;
|
||||
maxBitrate: number;
|
||||
avgBitrate: number;
|
||||
pcmSampleDepth: number;
|
||||
/** Number of PCM samples one frame decodes to */
|
||||
sampleCount: number;
|
||||
/** Speaker layout mask, or 0 when the box doesn't declare one */
|
||||
channelLayout: number;
|
||||
/** Null when the box carries nothing we can derive a channel count from */
|
||||
numberOfChannels: number | null;
|
||||
};
|
||||
|
||||
/** Parse a DTSSpecificBox (ddts). */
|
||||
export const parseDtsSpecificBox = (data: Uint8Array): DtsSpecificBoxInfo | null => {
|
||||
if (data.length < DTS_SPECIFIC_BOX_SIZE) {
|
||||
return null;
|
||||
}
|
||||
|
||||
const view = toDataView(data);
|
||||
const sampleRate = view.getUint32(0);
|
||||
if (sampleRate === 0) {
|
||||
return null;
|
||||
}
|
||||
|
||||
const bitstream = new Bitstream(data);
|
||||
bitstream.seekToByte(13);
|
||||
|
||||
const frameDuration = bitstream.readBits(2);
|
||||
bitstream.skipBits(5); // StreamConstruction
|
||||
const coreLfePresent = bitstream.readBits(1);
|
||||
const coreLayout = bitstream.readBits(6);
|
||||
bitstream.skipBits(14); // CoreSize
|
||||
bitstream.skipBits(1); // StereoDownmix
|
||||
bitstream.skipBits(3); // RepresentationType
|
||||
const channelLayout = bitstream.readBits(16);
|
||||
|
||||
let numberOfChannels: number | null = null;
|
||||
if (channelLayout !== 0) {
|
||||
numberOfChannels = getDtsChannelCount(channelLayout);
|
||||
} else if (coreLayout < DTS_AMODE_CHANNEL_COUNTS.length) {
|
||||
numberOfChannels = DTS_AMODE_CHANNEL_COUNTS[coreLayout]! + coreLfePresent;
|
||||
}
|
||||
|
||||
return {
|
||||
sampleRate,
|
||||
maxBitrate: view.getUint32(4),
|
||||
avgBitrate: view.getUint32(8),
|
||||
pcmSampleDepth: data[12]!,
|
||||
sampleCount: DTS_SPECIFIC_BOX_FRAME_DURATIONS[frameDuration]!,
|
||||
channelLayout,
|
||||
numberOfChannels,
|
||||
};
|
||||
};
|
||||
|
||||
/** Build the payload of a DTSSpecificBox (ddts) from a frame of the stream it describes. */
|
||||
export const buildDtsSpecificBox = (frameInfo: DtsFrameInfo) => {
|
||||
const bytes = new Uint8Array(DTS_SPECIFIC_BOX_SIZE);
|
||||
const view = toDataView(bytes);
|
||||
|
||||
view.setUint32(0, frameInfo.sampleRate);
|
||||
view.setUint32(4, frameInfo.bitRate);
|
||||
view.setUint32(8, frameInfo.bitRate);
|
||||
bytes[12] = frameInfo.pcmResolution;
|
||||
|
||||
// The spec only defines codes for streams built out of a known set of substreams. Anything else is required
|
||||
// to be signaled as 0, meaning the construction is left unspecified.
|
||||
const streamConstruction = frameInfo.core && !frameInfo.hasExtensions ? 1 : 0;
|
||||
|
||||
const bitstream = new Bitstream(bytes);
|
||||
bitstream.seekToByte(13);
|
||||
|
||||
bitstream.writeBits(2, Math.max(DTS_SPECIFIC_BOX_FRAME_DURATIONS.indexOf(frameInfo.sampleCount), 0));
|
||||
bitstream.writeBits(5, streamConstruction);
|
||||
bitstream.writeBits(1, frameInfo.core?.lfePresent ? 1 : 0);
|
||||
bitstream.writeBits(6, frameInfo.core?.amode ?? 0);
|
||||
bitstream.writeBits(14, frameInfo.core ? frameInfo.core.frameSize - 1 : 0);
|
||||
bitstream.writeBits(1, 0); // StereoDownmix
|
||||
bitstream.writeBits(3, 0); // RepresentationType
|
||||
bitstream.writeBits(16, frameInfo.channelLayout);
|
||||
bitstream.writeBits(1, 0); // MultiAssetFlag
|
||||
bitstream.writeBits(1, 0); // LBRDurationMod
|
||||
bitstream.writeBits(1, 0); // ReservedBoxPresent
|
||||
bitstream.writeBits(5, 0); // Reserved
|
||||
|
||||
return bytes;
|
||||
};
|
||||
|
||||
/** Count the channels in a DTS speaker layout mask, where some bits stand for a pair of speakers. */
|
||||
const getDtsChannelCount = (channelLayout: number) => {
|
||||
return popcount(channelLayout) + popcount(channelLayout & DTS_CHANNEL_LAYOUT_PAIR_MASK);
|
||||
};
|
||||
|
||||
+27
-2
@@ -75,6 +75,7 @@ export const NON_PCM_AUDIO_CODECS = [
|
||||
'flac',
|
||||
'ac3',
|
||||
'eac3',
|
||||
'dts',
|
||||
] as const;
|
||||
/**
|
||||
* List of known audio codecs, ordered by encoding preference.
|
||||
@@ -227,6 +228,14 @@ export const PRORES_FOURCCS = [
|
||||
] as const;
|
||||
export type ProresFourCc = typeof PRORES_FOURCCS[number];
|
||||
|
||||
export const DTS_FOURCCS = [
|
||||
'dtsc', // DTS core
|
||||
'dtsh', // DTS-HD, core plus extension substreams
|
||||
'dtsl', // DTS-HD Lossless, no core
|
||||
'dtse', // DTS Express
|
||||
] as const;
|
||||
export type DtsFourCc = typeof DTS_FOURCCS[number];
|
||||
|
||||
// Target data rates of the ProRes profiles at 1920x1080 ~30fps, as published by Apple
|
||||
const PRORES_PROFILE_TARGET_BITRATES: { fourCc: ProresFourCc; bitrate: number; alpha: boolean }[] = [
|
||||
{ fourCc: 'apco', bitrate: 45_000_000, alpha: false }, // 422 Proxy
|
||||
@@ -595,6 +604,8 @@ export const buildAudioCodecString = (codec: AudioCodec, numberOfChannels: numbe
|
||||
return 'ac-3';
|
||||
} else if (codec === 'eac3') {
|
||||
return 'ec-3';
|
||||
} else if (codec === 'dts') {
|
||||
return 'dtsc';
|
||||
} else if ((PCM_AUDIO_CODECS as readonly string[]).includes(codec)) {
|
||||
return codec;
|
||||
}
|
||||
@@ -611,8 +622,9 @@ export const extractAudioCodecString = (trackInfo: {
|
||||
codec: AudioCodec | null;
|
||||
codecDescription: Uint8Array | null;
|
||||
aacCodecInfo: AacCodecInfo | null;
|
||||
dtsFormat: DtsFourCc | null;
|
||||
}) => {
|
||||
const { codec, codecDescription, aacCodecInfo } = trackInfo;
|
||||
const { codec, codecDescription, aacCodecInfo, dtsFormat } = trackInfo;
|
||||
|
||||
if (codec === 'aac') {
|
||||
if (!aacCodecInfo) {
|
||||
@@ -644,6 +656,8 @@ export const extractAudioCodecString = (trackInfo: {
|
||||
return 'ac-3';
|
||||
} else if (codec === 'eac3') {
|
||||
return 'ec-3';
|
||||
} else if (codec === 'dts') {
|
||||
return dtsFormat ?? 'dtsc';
|
||||
} else if (codec && (PCM_AUDIO_CODECS as readonly string[]).includes(codec)) {
|
||||
return codec;
|
||||
}
|
||||
@@ -756,6 +770,8 @@ export const inferCodecFromCodecString = (codecString: string): MediaCodec | nul
|
||||
return 'ac3';
|
||||
} else if (codecString === 'ec-3' || codecString === 'eac3') {
|
||||
return 'eac3';
|
||||
} else if ((DTS_FOURCCS as readonly string[]).includes(codecString)) {
|
||||
return 'dts';
|
||||
} else if (codecString === 'ulaw') {
|
||||
return 'ulaw';
|
||||
} else if (codecString === 'alaw') {
|
||||
@@ -998,7 +1014,7 @@ export const validateVideoChunkMetadata = (
|
||||
};
|
||||
|
||||
const VALID_AUDIO_CODEC_STRING_PREFIXES = [
|
||||
'mp4a', 'mp3', 'opus', 'vorbis', 'flac', 'ulaw', 'alaw', 'pcm', 'ac-3', 'ec-3',
|
||||
'mp4a', 'mp3', 'opus', 'vorbis', 'flac', 'ulaw', 'alaw', 'pcm', 'ac-3', 'ec-3', 'dts',
|
||||
];
|
||||
|
||||
export const validateAudioChunkMetadata = (
|
||||
@@ -1132,6 +1148,15 @@ export const validateAudioChunkMetadata = (
|
||||
if (metadata.decoderConfig.codec !== 'ec-3') {
|
||||
throw new TypeError('Audio chunk metadata decoder configuration codec string for EC-3 must be "ec-3".');
|
||||
}
|
||||
} else if (metadata.decoderConfig.codec.startsWith('dts')) {
|
||||
// DTS-specific validation
|
||||
|
||||
if (!(DTS_FOURCCS as readonly string[]).includes(metadata.decoderConfig.codec)) {
|
||||
throw new TypeError(
|
||||
'Audio chunk metadata decoder configuration codec string for DTS must be one of the following'
|
||||
+ ` four-character codes: ${DTS_FOURCCS.join(', ')}.`,
|
||||
);
|
||||
}
|
||||
} else if (
|
||||
metadata.decoderConfig.codec.startsWith('pcm')
|
||||
|| metadata.decoderConfig.codec.startsWith('ulaw')
|
||||
|
||||
+3
-2
@@ -858,6 +858,7 @@ export class Quality {
|
||||
vorbis: 64000, // 64kbps base for Vorbis
|
||||
ac3: 384000, // 384kbps base for AC-3
|
||||
eac3: 192000, // 192kbps base for E-AC-3
|
||||
dts: 768000, // 768kbps base for DTS
|
||||
};
|
||||
|
||||
const baseBitrate = baseRates[codec as keyof typeof baseRates];
|
||||
@@ -1071,7 +1072,7 @@ export const canEncodeVideo = async (
|
||||
}
|
||||
validateVideoEncodingAdditionalOptions(codec, restOptions);
|
||||
|
||||
const resolvedQuality = resolveQuality(quality, bitrate) ?? new Quality({ bitrate: 1e6 });
|
||||
const resolvedQuality = resolveQuality(quality, bitrate) ?? new Quality('medium');
|
||||
|
||||
let candidates: VideoEncoderConfigCandidate[];
|
||||
try {
|
||||
@@ -1222,7 +1223,7 @@ export const canEncodeAudio = async (
|
||||
}
|
||||
validateAudioEncodingAdditionalOptions(codec, restOptions);
|
||||
|
||||
const resolvedQuality = resolveQuality(quality, bitrate) ?? new Quality({ bitrate: 128e3 });
|
||||
const resolvedQuality = resolveQuality(quality, bitrate) ?? new Quality('medium');
|
||||
|
||||
const encoderConfig = buildAudioEncoderConfig({
|
||||
codec,
|
||||
|
||||
@@ -43,7 +43,13 @@ import {
|
||||
IsobmffVideoTrackData,
|
||||
Sample,
|
||||
} from './isobmff-muxer';
|
||||
import { parseAc3SyncFrame, parseEac3SyncFrame, parseOpusIdentificationHeader } from '../codec-data';
|
||||
import {
|
||||
buildDtsSpecificBox,
|
||||
parseAc3SyncFrame,
|
||||
parseDtsFrame,
|
||||
parseEac3SyncFrame,
|
||||
parseOpusIdentificationHeader,
|
||||
} from '../codec-data';
|
||||
import { MetadataTags, RichImageData } from '../metadata';
|
||||
import { Bitstream } from '../../shared/bitstream';
|
||||
|
||||
@@ -690,7 +696,11 @@ export const stsd = (trackData: IsobmffTrackData) => {
|
||||
trackData,
|
||||
);
|
||||
} else if (trackData.type === 'audio') {
|
||||
const boxName = audioCodecToBoxName(trackData.track.source._codec, trackData.muxer.isQuickTime);
|
||||
const boxName = audioCodecToBoxName(
|
||||
trackData.track.source._codec,
|
||||
trackData.info.decoderConfig.codec,
|
||||
trackData.muxer.isQuickTime,
|
||||
);
|
||||
assert(boxName);
|
||||
|
||||
sampleDescription = soundSampleDescription(
|
||||
@@ -968,7 +978,11 @@ export const wave = (trackData: IsobmffAudioTrackData) => {
|
||||
|
||||
export const frma = (trackData: IsobmffAudioTrackData) => {
|
||||
return box('frma', [
|
||||
ascii(audioCodecToBoxName(trackData.track.source._codec, trackData.muxer.isQuickTime)),
|
||||
ascii(audioCodecToBoxName(
|
||||
trackData.track.source._codec,
|
||||
trackData.info.decoderConfig.codec,
|
||||
trackData.muxer.isQuickTime,
|
||||
)),
|
||||
]);
|
||||
};
|
||||
|
||||
@@ -1123,6 +1137,21 @@ const dec3 = (trackData: IsobmffAudioTrackData) => {
|
||||
return box('dec3', [...bytes]);
|
||||
};
|
||||
|
||||
/** DTSSpecificBox */
|
||||
const ddts = (trackData: IsobmffAudioTrackData) => {
|
||||
assert(trackData.info.primingPacket);
|
||||
|
||||
const frameInfo = parseDtsFrame(trackData.info.primingPacket.data);
|
||||
if (!frameInfo) {
|
||||
throw new Error(
|
||||
'Couldn\'t extract DTS frame info from the audio packet. '
|
||||
+ 'Ensure the packets contain valid DTS frames as specified in ETSI TS 102 114.',
|
||||
);
|
||||
}
|
||||
|
||||
return box('ddts', [...buildDtsSpecificBox(frameInfo)]);
|
||||
};
|
||||
|
||||
export const subtitleSampleDescription = (
|
||||
compressionType: string,
|
||||
trackData: IsobmffSubtitleTrackData,
|
||||
@@ -1816,7 +1845,7 @@ const VIDEO_CODEC_TO_CONFIGURATION_BOX: Record<
|
||||
prores: null,
|
||||
};
|
||||
|
||||
const audioCodecToBoxName = (codec: AudioCodec, isQuickTime: boolean): string => {
|
||||
const audioCodecToBoxName = (codec: AudioCodec, fullCodecString: string, isQuickTime: boolean): string => {
|
||||
switch (codec) {
|
||||
case 'aac': return 'mp4a';
|
||||
case 'mp3': return 'mp4a';
|
||||
@@ -1829,6 +1858,7 @@ const audioCodecToBoxName = (codec: AudioCodec, isQuickTime: boolean): string =>
|
||||
case 'pcm-s8': return 'sowt';
|
||||
case 'ac3': return 'ac-3';
|
||||
case 'eac3': return 'ec-3';
|
||||
case 'dts': return fullCodecString;
|
||||
}
|
||||
|
||||
// Logic diverges here
|
||||
@@ -1870,6 +1900,7 @@ const audioCodecToConfigurationBox = (codec: AudioCodec, isQuickTime: boolean) =
|
||||
case 'flac': return dfLa;
|
||||
case 'ac3': return dac3;
|
||||
case 'eac3': return dec3;
|
||||
case 'dts': return ddts;
|
||||
}
|
||||
|
||||
// Logic diverges here
|
||||
|
||||
@@ -11,6 +11,8 @@ import { parseAacAudioSpecificConfig } from '../../shared/aac-misc';
|
||||
import {
|
||||
AacCodecInfo,
|
||||
AudioCodec,
|
||||
DTS_FOURCCS,
|
||||
DtsFourCc,
|
||||
extractAudioCodecString,
|
||||
extractVideoCodecString,
|
||||
MediaCodec,
|
||||
@@ -33,6 +35,9 @@ import {
|
||||
parseEac3Config,
|
||||
getEac3SampleRate,
|
||||
getEac3ChannelCount,
|
||||
extractDtsFourCcFromPacket,
|
||||
parseDtsSpecificBox,
|
||||
DTS_SPECIFIC_BOX_SIZE,
|
||||
AC3_ACMOD_CHANNEL_COUNTS,
|
||||
} from '../codec-data';
|
||||
import { Demuxer } from '../demuxer';
|
||||
@@ -159,6 +164,7 @@ type InternalTrack = {
|
||||
codec: AudioCodec | null;
|
||||
codecDescription: Uint8Array | null;
|
||||
aacCodecInfo: AacCodecInfo | null;
|
||||
dtsFormat: DtsFourCc | null;
|
||||
pcmLittleEndian: boolean;
|
||||
pcmSampleSize: number | null;
|
||||
};
|
||||
@@ -1034,6 +1040,7 @@ export class IsobmffDemuxer extends Demuxer {
|
||||
codec: null,
|
||||
codecDescription: null,
|
||||
aacCodecInfo: null,
|
||||
dtsFormat: null,
|
||||
pcmLittleEndian: false,
|
||||
pcmSampleSize: null,
|
||||
};
|
||||
@@ -1185,6 +1192,9 @@ export class IsobmffDemuxer extends Demuxer {
|
||||
track.info.codec = 'ac3';
|
||||
} else if (codecName === 'ec-3') {
|
||||
track.info.codec = 'eac3';
|
||||
} else if ((DTS_FOURCCS as readonly string[]).includes(codecName!)) {
|
||||
track.info.codec = 'dts';
|
||||
track.info.dtsFormat = codecName as DtsFourCc;
|
||||
} else if (codecName === 'twos') {
|
||||
if (sampleSize === 8) {
|
||||
track.info.codec = 'pcm-s8';
|
||||
@@ -1564,6 +1574,8 @@ export class IsobmffDemuxer extends Demuxer {
|
||||
track.info.codec = 'mp3';
|
||||
} else if (objectTypeIndication === 0xdd) {
|
||||
track.info.codec = 'vorbis'; // "nonstandard, gpac uses it" - FFmpeg
|
||||
} else if (objectTypeIndication === 0xa9) {
|
||||
track.info.codec = 'dts';
|
||||
} else {
|
||||
Logging._warn(
|
||||
`Unsupported audio codec (objectTypeIndication ${objectTypeIndication}) - discarding track.`,
|
||||
@@ -1763,6 +1775,28 @@ export class IsobmffDemuxer extends Demuxer {
|
||||
track.info.numberOfChannels = getEac3ChannelCount(config);
|
||||
}; break;
|
||||
|
||||
case 'ddts': { // DTSSpecificBox
|
||||
const track = this.currentTrack;
|
||||
if (!track) {
|
||||
break;
|
||||
}
|
||||
assert(track.info?.type === 'audio');
|
||||
|
||||
const bytes = readBytes(slice, Math.min(boxInfo.contentSize, DTS_SPECIFIC_BOX_SIZE));
|
||||
const config = parseDtsSpecificBox(bytes);
|
||||
|
||||
if (!config) {
|
||||
Logging._warn('Invalid ddts box contents, ignoring.');
|
||||
break;
|
||||
}
|
||||
|
||||
track.info.sampleRate = config.sampleRate;
|
||||
|
||||
if (config.numberOfChannels !== null) {
|
||||
track.info.numberOfChannels = config.numberOfChannels;
|
||||
}
|
||||
}; break;
|
||||
|
||||
case 'stts': {
|
||||
const track = this.currentTrack;
|
||||
if (!track) {
|
||||
@@ -3404,7 +3438,7 @@ class IsobmffVideoTrackBacking extends IsobmffTrackBacking implements InputVideo
|
||||
|
||||
class IsobmffAudioTrackBacking extends IsobmffTrackBacking implements InputAudioTrackBacking {
|
||||
override internalTrack: InternalAudioTrack;
|
||||
decoderConfig: AudioDecoderConfig | null = null;
|
||||
decoderConfigPromise: Promise<AudioDecoderConfig> | null = null;
|
||||
|
||||
constructor(internalTrack: InternalAudioTrack) {
|
||||
super(internalTrack);
|
||||
@@ -3432,12 +3466,20 @@ class IsobmffAudioTrackBacking extends IsobmffTrackBacking implements InputAudio
|
||||
return null;
|
||||
}
|
||||
|
||||
return this.decoderConfig ??= {
|
||||
codec: extractAudioCodecString(this.internalTrack.info),
|
||||
numberOfChannels: this.internalTrack.info.numberOfChannels,
|
||||
sampleRate: this.internalTrack.info.sampleRate,
|
||||
description: this.internalTrack.info.codecDescription ?? undefined,
|
||||
};
|
||||
return this.decoderConfigPromise ??= (async (): Promise<AudioDecoderConfig> => {
|
||||
if (this.internalTrack.info.codec === 'dts' && !this.internalTrack.info.dtsFormat) {
|
||||
// Gotta check the packet to determine the DTS variant
|
||||
const firstPacket = await this.getFirstPacket({});
|
||||
this.internalTrack.info.dtsFormat = firstPacket && extractDtsFourCcFromPacket(firstPacket.data);
|
||||
}
|
||||
|
||||
return {
|
||||
codec: extractAudioCodecString(this.internalTrack.info),
|
||||
numberOfChannels: this.internalTrack.info.numberOfChannels,
|
||||
sampleRate: this.internalTrack.info.sampleRate,
|
||||
description: this.internalTrack.info.codecDescription ?? undefined,
|
||||
};
|
||||
})();
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -531,10 +531,14 @@ export class IsobmffMuxer extends Muxer {
|
||||
requiresAdtsStripping = true;
|
||||
}
|
||||
|
||||
if (track.source._codec === 'ac3' || track.source._codec === 'eac3') {
|
||||
if (!packet) {
|
||||
if (!packet) {
|
||||
if (track.source._codec === 'ac3' || track.source._codec === 'eac3') {
|
||||
throw new Error('AC-3/E-AC-3 require a priming packet.');
|
||||
}
|
||||
|
||||
if (track.source._codec === 'dts') {
|
||||
throw new Error('DTS requires a priming packet.');
|
||||
}
|
||||
}
|
||||
|
||||
const newTrackData: IsobmffAudioTrackData = {
|
||||
|
||||
@@ -750,6 +750,7 @@ export const CODEC_STRING_MAP: Partial<Record<MediaCodec, string>> = {
|
||||
'flac': 'A_FLAC',
|
||||
'ac3': 'A_AC3',
|
||||
'eac3': 'A_EAC3',
|
||||
'dts': 'A_DTS',
|
||||
'pcm-u8': 'A_PCM/INT/LIT',
|
||||
'pcm-s16': 'A_PCM/INT/LIT',
|
||||
'pcm-s16be': 'A_PCM/INT/BIG',
|
||||
|
||||
@@ -9,6 +9,7 @@
|
||||
import { TrackType } from '../output';
|
||||
import {
|
||||
extractAv1CodecInfoFromPacket,
|
||||
extractDtsFourCcFromPacket,
|
||||
extractAvcDecoderConfigurationRecord,
|
||||
extractHevcDecoderConfigurationRecord,
|
||||
extractVp9CodecInfoFromPacket,
|
||||
@@ -16,6 +17,7 @@ import {
|
||||
import {
|
||||
AacCodecInfo,
|
||||
AudioCodec,
|
||||
DtsFourCc,
|
||||
extractAudioCodecString,
|
||||
extractVideoCodecString,
|
||||
MediaCodec,
|
||||
@@ -229,6 +231,7 @@ type InternalTrack = {
|
||||
codec: AudioCodec | null;
|
||||
codecDescription: Uint8Array | null;
|
||||
aacCodecInfo: AacCodecInfo | null;
|
||||
dtsFormat: DtsFourCc | null;
|
||||
};
|
||||
};
|
||||
type InternalVideoTrack = InternalTrack & { info: { type: 'video' } };
|
||||
@@ -1127,6 +1130,14 @@ export class MatroskaDemuxer extends Demuxer {
|
||||
} else if (codecIdWithoutSuffix === CODEC_STRING_MAP.eac3) {
|
||||
this.currentTrack.info.codec = 'eac3';
|
||||
this.currentTrack.info.codecDescription = this.currentTrack.codecPrivate;
|
||||
} else if (codecIdWithoutSuffix === CODEC_STRING_MAP.dts) {
|
||||
this.currentTrack.info.codec = 'dts';
|
||||
|
||||
if (this.currentTrack.codecId === 'A_DTS/EXPRESS') {
|
||||
this.currentTrack.info.dtsFormat = 'dtse';
|
||||
} else if (this.currentTrack.codecId === 'A_DTS/LOSSLESS') {
|
||||
this.currentTrack.info.dtsFormat = 'dtsl';
|
||||
}
|
||||
} else if (this.currentTrack.codecId === 'A_PCM/INT/LIT') {
|
||||
if (this.currentTrack.info.bitDepth === 8) {
|
||||
this.currentTrack.info.codec = 'pcm-u8';
|
||||
@@ -1200,6 +1211,7 @@ export class MatroskaDemuxer extends Demuxer {
|
||||
codec: null,
|
||||
codecDescription: null,
|
||||
aacCodecInfo: null,
|
||||
dtsFormat: null,
|
||||
};
|
||||
}
|
||||
}; break;
|
||||
@@ -2556,7 +2568,7 @@ class MatroskaVideoTrackBacking extends MatroskaTrackBacking implements InputVid
|
||||
|
||||
class MatroskaAudioTrackBacking extends MatroskaTrackBacking implements InputAudioTrackBacking {
|
||||
override internalTrack: InternalAudioTrack;
|
||||
decoderConfig: AudioDecoderConfig | null = null;
|
||||
decoderConfigPromise: Promise<AudioDecoderConfig> | null = null;
|
||||
|
||||
constructor(internalTrack: InternalAudioTrack) {
|
||||
super(internalTrack);
|
||||
@@ -2584,15 +2596,24 @@ class MatroskaAudioTrackBacking extends MatroskaTrackBacking implements InputAud
|
||||
return null;
|
||||
}
|
||||
|
||||
return this.decoderConfig ??= {
|
||||
codec: extractAudioCodecString({
|
||||
codec: this.internalTrack.info.codec,
|
||||
codecDescription: this.internalTrack.info.codecDescription,
|
||||
aacCodecInfo: this.internalTrack.info.aacCodecInfo,
|
||||
}),
|
||||
numberOfChannels: this.internalTrack.info.numberOfChannels,
|
||||
sampleRate: this.internalTrack.info.sampleRate,
|
||||
description: this.internalTrack.info.codecDescription ?? undefined,
|
||||
};
|
||||
return this.decoderConfigPromise ??= (async (): Promise<AudioDecoderConfig> => {
|
||||
if (this.internalTrack.info.codec === 'dts' && !this.internalTrack.info.dtsFormat) {
|
||||
// Gotta check the packet to determine the DTS variant
|
||||
const firstPacket = await this.getFirstPacket({});
|
||||
this.internalTrack.info.dtsFormat = firstPacket && extractDtsFourCcFromPacket(firstPacket.data);
|
||||
}
|
||||
|
||||
return {
|
||||
codec: extractAudioCodecString({
|
||||
codec: this.internalTrack.info.codec,
|
||||
codecDescription: this.internalTrack.info.codecDescription,
|
||||
aacCodecInfo: this.internalTrack.info.aacCodecInfo,
|
||||
dtsFormat: this.internalTrack.info.dtsFormat,
|
||||
}),
|
||||
numberOfChannels: this.internalTrack.info.numberOfChannels,
|
||||
sampleRate: this.internalTrack.info.sampleRate,
|
||||
description: this.internalTrack.info.codecDescription ?? undefined,
|
||||
};
|
||||
})();
|
||||
}
|
||||
}
|
||||
|
||||
@@ -310,9 +310,18 @@ export class MatroskaMuxer extends Muxer {
|
||||
this.tracksElement = tracksElement;
|
||||
|
||||
for (const trackData of this.trackDatas) {
|
||||
const codecId = CODEC_STRING_MAP[trackData.track.source._codec];
|
||||
let codecId = CODEC_STRING_MAP[trackData.track.source._codec];
|
||||
assert(codecId);
|
||||
|
||||
if (trackData.type === 'audio' && trackData.track.source._codec === 'dts') {
|
||||
// We can further refine the Codec ID
|
||||
if (trackData.info.decoderConfig.codec === 'dtse') {
|
||||
codecId = 'A_DTS/EXPRESS';
|
||||
} else if (trackData.info.decoderConfig.codec === 'dtsl') {
|
||||
codecId = 'A_DTS/LOSSLESS';
|
||||
}
|
||||
}
|
||||
|
||||
let seekPreRollNs = 0;
|
||||
if (trackData.type === 'audio' && trackData.track.source._codec === 'opus') {
|
||||
seekPreRollNs = 1e6 * 80; // In "Matroska ticks" (nanoseconds)
|
||||
|
||||
+11
@@ -472,6 +472,17 @@ export const ilog = (x: number) => {
|
||||
return ret;
|
||||
};
|
||||
|
||||
export const popcount = (value: number) => {
|
||||
let count = 0;
|
||||
|
||||
while (value !== 0) {
|
||||
value &= value - 1;
|
||||
count++;
|
||||
}
|
||||
|
||||
return count;
|
||||
};
|
||||
|
||||
const ISO_639_2_REGEX = /^[a-z]{3}$/;
|
||||
export const isIso639Dash2LanguageCode = (x: string) => {
|
||||
return ISO_639_2_REGEX.test(x);
|
||||
|
||||
+186
-11
@@ -13,6 +13,7 @@ import { aacChannelMap, aacFrequencyTable } from '../../shared/aac-misc';
|
||||
import {
|
||||
AacCodecInfo,
|
||||
AudioCodec,
|
||||
DtsFourCc,
|
||||
extractAudioCodecString,
|
||||
extractVideoCodecString,
|
||||
MediaCodec,
|
||||
@@ -38,6 +39,12 @@ import {
|
||||
AC3_FRAME_SIZES,
|
||||
extractNalUnitTypeForAvc,
|
||||
extractNalUnitTypeForHevc,
|
||||
parseDtsFrame,
|
||||
parseDtsCoreFrameHeader,
|
||||
parseDtsExssHeader,
|
||||
DTS_CORE_FRAME_HEADER_SIZE,
|
||||
DTS_EXSS_HEADER_PREFIX_SIZE,
|
||||
DTS_EXSS_MAX_HEADER_SIZE,
|
||||
} from '../codec-data';
|
||||
import { Demuxer } from '../demuxer';
|
||||
import { Input } from '../input';
|
||||
@@ -82,6 +89,22 @@ import { Bitstream } from '../../shared/bitstream';
|
||||
const MISSING_PTS_ERROR_MESSAGE = 'PES packet is missing PTS where it was expected. PES packets without PTS are not'
|
||||
+ ' currently supported. If you think this file should be supported, please report it.';
|
||||
|
||||
const REGISTRATION_DESCRIPTOR_TAG = 0x05;
|
||||
|
||||
// The 'HDMV' and 'HDPR' format identifiers, which mark a program as Blu-ray-derived
|
||||
const HDMV_FORMAT_IDENTIFIER = 0x48444d56;
|
||||
const HDPR_FORMAT_IDENTIFIER = 0x48445052;
|
||||
|
||||
// 'DTS1', 'DTS2' and 'DTS3' all share this prefix
|
||||
const DTS_FORMAT_IDENTIFIER_PREFIX = 0x44545300;
|
||||
|
||||
// Stream types that only mean DTS within a Blu-ray-derived program
|
||||
const BLU_RAY_DTS_STREAM_TYPES = new Set<number>([
|
||||
MpegTsStreamType.BLU_RAY_DTS_HD,
|
||||
MpegTsStreamType.BLU_RAY_DTS_HD_MASTER,
|
||||
MpegTsStreamType.BLU_RAY_DTS_EXPRESS_SECONDARY,
|
||||
]);
|
||||
|
||||
type ElementaryStream = {
|
||||
demuxer: MpegTsDemuxer;
|
||||
pid: number;
|
||||
@@ -110,6 +133,7 @@ type ElementaryStream = {
|
||||
codec: AudioCodec;
|
||||
decoderConfig: AudioDecoderConfig | null;
|
||||
aacCodecInfo: AacCodecInfo | null;
|
||||
dtsFormat: DtsFourCc | null;
|
||||
numberOfChannels: number;
|
||||
sampleRate: number;
|
||||
};
|
||||
@@ -294,7 +318,27 @@ export class MpegTsDemuxer extends Demuxer {
|
||||
// "The remaining 10 bits specify the number of bytes of the descriptors immediately following the
|
||||
// program_info_length field"
|
||||
const programInfoLength = bitstream.readBits(10);
|
||||
bitstream.skipBits(8 * programInfoLength);
|
||||
|
||||
// A few stream types only mean what Blu-ray says they mean, and the program's registration
|
||||
// descriptor is what tells us we're looking at a Blu-ray-derived stream.
|
||||
const programInfoEndPos = bitstream.pos + 8 * programInfoLength;
|
||||
let isBluRayProgram = false;
|
||||
|
||||
while (bitstream.pos < programInfoEndPos) {
|
||||
const descriptorTag = bitstream.readBits(8);
|
||||
const descriptorLength = bitstream.readBits(8);
|
||||
const descriptorEndPos = bitstream.pos + 8 * descriptorLength;
|
||||
|
||||
if (descriptorTag === REGISTRATION_DESCRIPTOR_TAG && descriptorLength >= 4) {
|
||||
const formatIdentifier = bitstream.readBits(32);
|
||||
isBluRayProgram ||= formatIdentifier === HDMV_FORMAT_IDENTIFIER
|
||||
|| formatIdentifier === HDPR_FORMAT_IDENTIFIER;
|
||||
}
|
||||
|
||||
bitstream.pos = descriptorEndPos;
|
||||
}
|
||||
|
||||
bitstream.pos = programInfoEndPos;
|
||||
|
||||
while (8 * (sectionLength + BYTES_BEFORE_SECTION_LENGTH) - bitstream.pos > BITS_IN_CRC_32) {
|
||||
const streamType = bitstream.readBits(8);
|
||||
@@ -304,24 +348,40 @@ export class MpegTsDemuxer extends Demuxer {
|
||||
bitstream.skipBits(6);
|
||||
const esInfoLength = bitstream.readBits(10);
|
||||
|
||||
// Check ES descriptors to detect AC-3/E-AC-3 in System B
|
||||
// Check ES descriptors to detect AC-3/E-AC-3/DTS in System B
|
||||
const esInfoEndPos = bitstream.pos + 8 * esInfoLength;
|
||||
let hasAc3Descriptor = false;
|
||||
let hasEac3Descriptor = false;
|
||||
let hasDtsDescriptor = false;
|
||||
while (bitstream.pos < esInfoEndPos) {
|
||||
const descriptorTag = bitstream.readBits(8);
|
||||
const descriptorLength = bitstream.readBits(8);
|
||||
const descriptorEndPos = bitstream.pos + 8 * descriptorLength;
|
||||
|
||||
if (descriptorTag === 0x6a) {
|
||||
hasAc3Descriptor = true;
|
||||
} else if (descriptorTag === 0x7a || descriptorTag === 0xcc) {
|
||||
hasEac3Descriptor = true;
|
||||
} else if (descriptorTag === 0x7b) {
|
||||
hasDtsDescriptor = true;
|
||||
} else if (descriptorTag === REGISTRATION_DESCRIPTOR_TAG && descriptorLength >= 4) {
|
||||
// DTS uses 'DTS1', 'DTS2' and 'DTS3' to also signal the stream's frame size,
|
||||
// which we work out from the bitstream anyway
|
||||
const formatIdentifier = bitstream.readBits(32);
|
||||
hasDtsDescriptor ||= (formatIdentifier & 0xffffff00) === DTS_FORMAT_IDENTIFIER_PREFIX;
|
||||
}
|
||||
bitstream.skipBits(8 * descriptorLength);
|
||||
|
||||
bitstream.pos = descriptorEndPos;
|
||||
}
|
||||
|
||||
let info: ElementaryStream['info'] | null = null;
|
||||
|
||||
switch (streamType) {
|
||||
// Blu-ray has its own stream types for the DTS extensions
|
||||
const effectiveStreamType = isBluRayProgram && BLU_RAY_DTS_STREAM_TYPES.has(streamType)
|
||||
? MpegTsStreamType.DTS
|
||||
: streamType;
|
||||
|
||||
switch (effectiveStreamType) {
|
||||
case MpegTsStreamType.AVC:
|
||||
case MpegTsStreamType.HEVC: {
|
||||
const codec = streamType === MpegTsStreamType.AVC ? 'avc' : 'hevc';
|
||||
@@ -350,21 +410,23 @@ export class MpegTsDemuxer extends Demuxer {
|
||||
case MpegTsStreamType.MP3_MPEG2:
|
||||
case MpegTsStreamType.AAC:
|
||||
case MpegTsStreamType.AC3_SYSTEM_A:
|
||||
case MpegTsStreamType.EAC3_SYSTEM_A: {
|
||||
case MpegTsStreamType.EAC3_SYSTEM_A:
|
||||
case MpegTsStreamType.DTS:
|
||||
case MpegTsStreamType.DTS_ATSC: {
|
||||
let codec: AudioCodec;
|
||||
if (
|
||||
streamType === MpegTsStreamType.MP3_MPEG1
|
||||
|| streamType === MpegTsStreamType.MP3_MPEG2
|
||||
effectiveStreamType === MpegTsStreamType.MP3_MPEG1
|
||||
|| effectiveStreamType === MpegTsStreamType.MP3_MPEG2
|
||||
) {
|
||||
codec = 'mp3';
|
||||
} else if (streamType === MpegTsStreamType.AAC) {
|
||||
} else if (effectiveStreamType === MpegTsStreamType.AAC) {
|
||||
codec = 'aac';
|
||||
} else if (streamType === MpegTsStreamType.AC3_SYSTEM_A) {
|
||||
} else if (effectiveStreamType === MpegTsStreamType.AC3_SYSTEM_A) {
|
||||
codec = 'ac3';
|
||||
} else if (streamType === MpegTsStreamType.EAC3_SYSTEM_A) {
|
||||
} else if (effectiveStreamType === MpegTsStreamType.EAC3_SYSTEM_A) {
|
||||
codec = 'eac3';
|
||||
} else {
|
||||
throw new Error('Unreachable.');
|
||||
codec = 'dts';
|
||||
}
|
||||
|
||||
info = {
|
||||
@@ -372,6 +434,7 @@ export class MpegTsDemuxer extends Demuxer {
|
||||
codec,
|
||||
decoderConfig: null,
|
||||
aacCodecInfo: null,
|
||||
dtsFormat: null,
|
||||
numberOfChannels: -1,
|
||||
sampleRate: -1,
|
||||
};
|
||||
@@ -384,6 +447,7 @@ export class MpegTsDemuxer extends Demuxer {
|
||||
codec: 'eac3',
|
||||
decoderConfig: null,
|
||||
aacCodecInfo: null,
|
||||
dtsFormat: null,
|
||||
numberOfChannels: -1,
|
||||
sampleRate: -1,
|
||||
};
|
||||
@@ -393,6 +457,17 @@ export class MpegTsDemuxer extends Demuxer {
|
||||
codec: 'ac3',
|
||||
decoderConfig: null,
|
||||
aacCodecInfo: null,
|
||||
dtsFormat: null,
|
||||
numberOfChannels: -1,
|
||||
sampleRate: -1,
|
||||
};
|
||||
} else if (hasDtsDescriptor) {
|
||||
info = {
|
||||
type: 'audio',
|
||||
codec: 'dts',
|
||||
decoderConfig: null,
|
||||
aacCodecInfo: null,
|
||||
dtsFormat: null,
|
||||
numberOfChannels: -1,
|
||||
sampleRate: -1,
|
||||
};
|
||||
@@ -678,6 +753,22 @@ export class MpegTsDemuxer extends Demuxer {
|
||||
|
||||
elementaryStream.info.numberOfChannels = getEac3ChannelCount(frameInfo);
|
||||
elementaryStream.info.sampleRate = sampleRate;
|
||||
} else if (elementaryStream.info.codec === 'dts') {
|
||||
const frameInfo = parseDtsFrame(context.suppliedPacket.data);
|
||||
if (!frameInfo) {
|
||||
throw new Error(
|
||||
'Invalid DTS audio stream; could not read frame header from first packet.',
|
||||
);
|
||||
}
|
||||
|
||||
elementaryStream.info.numberOfChannels = frameInfo.numberOfChannels;
|
||||
elementaryStream.info.sampleRate = frameInfo.sampleRate;
|
||||
|
||||
if (frameInfo.core) {
|
||||
// Telling the two extension-only variants apart would need us to work out
|
||||
// whether the asset holds XLL or LBR data, so we leave those unlabeled
|
||||
elementaryStream.info.dtsFormat = frameInfo.hasExtensions ? 'dtsh' : 'dtsc';
|
||||
}
|
||||
} else {
|
||||
throw new Error('Unhandled.');
|
||||
}
|
||||
@@ -687,6 +778,7 @@ export class MpegTsDemuxer extends Demuxer {
|
||||
codec: elementaryStream.info.codec,
|
||||
codecDescription: null,
|
||||
aacCodecInfo: elementaryStream.info.aacCodecInfo,
|
||||
dtsFormat: elementaryStream.info.dtsFormat,
|
||||
}),
|
||||
numberOfChannels: elementaryStream.info.numberOfChannels,
|
||||
sampleRate: elementaryStream.info.sampleRate,
|
||||
@@ -2315,6 +2407,89 @@ class PacketReadingContext {
|
||||
samplesPerFrame * TIMESCALE / elementaryStream.info.sampleRate,
|
||||
);
|
||||
return this.supplyPacket(remaining, duration);
|
||||
} else if (codec === 'dts') {
|
||||
if (byte !== 0x7f && byte !== 0x64) {
|
||||
continue;
|
||||
}
|
||||
|
||||
this.skip(-1);
|
||||
const possibleSyncPos = this.currentPos;
|
||||
|
||||
let remaining = this.ensureBuffered(DTS_CORE_FRAME_HEADER_SIZE);
|
||||
if (remaining instanceof Promise) remaining = await remaining;
|
||||
|
||||
if (remaining < DTS_CORE_FRAME_HEADER_SIZE) {
|
||||
return;
|
||||
}
|
||||
|
||||
const headerBytes = this.readBytes(DTS_CORE_FRAME_HEADER_SIZE);
|
||||
const core = parseDtsCoreFrameHeader(headerBytes);
|
||||
let leadingExss = core ? null : parseDtsExssHeader(headerBytes);
|
||||
|
||||
if (!core && !leadingExss) {
|
||||
this.seekTo(possibleSyncPos + 1);
|
||||
continue;
|
||||
}
|
||||
|
||||
if (leadingExss && !leadingExss.asset) {
|
||||
// The static fields and the asset descriptor reach past the bytes we read for the
|
||||
// core header, so take another look with the substream's whole header available
|
||||
this.seekTo(possibleSyncPos);
|
||||
|
||||
const headerBound = Math.min(leadingExss.frameSize, DTS_EXSS_MAX_HEADER_SIZE);
|
||||
let remaining = this.ensureBuffered(headerBound);
|
||||
if (remaining instanceof Promise) remaining = await remaining;
|
||||
|
||||
leadingExss = parseDtsExssHeader(this.readBytes(remaining)) ?? leadingExss;
|
||||
}
|
||||
|
||||
let frameSize = core ? core.frameSize : leadingExss!.frameSize;
|
||||
|
||||
// Extension-only streams put exactly one substream in each frame, so it's only a core
|
||||
// frame that needs us to go looking for the extensions riding along with it. The core is
|
||||
// padded out to a 4-byte boundary before the first of them starts.
|
||||
if (core) {
|
||||
let nextSubstreamPos = Math.ceil(core.frameSize / 4) * 4;
|
||||
|
||||
while (true) {
|
||||
this.seekTo(possibleSyncPos);
|
||||
|
||||
const neededBytes = nextSubstreamPos + DTS_EXSS_HEADER_PREFIX_SIZE;
|
||||
let remaining = this.ensureBuffered(neededBytes);
|
||||
if (remaining instanceof Promise) remaining = await remaining;
|
||||
|
||||
if (remaining < neededBytes) {
|
||||
break;
|
||||
}
|
||||
|
||||
this.seekTo(possibleSyncPos + nextSubstreamPos);
|
||||
|
||||
const exss = parseDtsExssHeader(this.readBytes(DTS_EXSS_HEADER_PREFIX_SIZE));
|
||||
if (!exss) {
|
||||
break;
|
||||
}
|
||||
|
||||
nextSubstreamPos += exss.frameSize;
|
||||
frameSize = nextSubstreamPos;
|
||||
}
|
||||
}
|
||||
|
||||
// Only the substream leading the frame declares how many samples the frame holds
|
||||
const sampleCount = core?.sampleCount ?? leadingExss!.asset?.sampleCount;
|
||||
if (sampleCount === undefined) {
|
||||
this.seekTo(possibleSyncPos + 1);
|
||||
continue;
|
||||
}
|
||||
|
||||
this.seekTo(possibleSyncPos);
|
||||
|
||||
remaining = this.ensureBuffered(frameSize);
|
||||
if (remaining instanceof Promise) remaining = await remaining;
|
||||
|
||||
const duration = Math.round(
|
||||
sampleCount * TIMESCALE / elementaryStream.info.sampleRate,
|
||||
);
|
||||
return this.supplyPacket(remaining, duration);
|
||||
} else {
|
||||
throw new Error('Unhandled.');
|
||||
}
|
||||
|
||||
@@ -15,6 +15,11 @@ export const enum MpegTsStreamType {
|
||||
AAC = 0x0f,
|
||||
AC3_SYSTEM_A = 0x81,
|
||||
EAC3_SYSTEM_A = 0x87,
|
||||
DTS = 0x82,
|
||||
DTS_ATSC = 0x8a,
|
||||
BLU_RAY_DTS_HD = 0x85,
|
||||
BLU_RAY_DTS_HD_MASTER = 0x86,
|
||||
BLU_RAY_DTS_EXPRESS_SECONDARY = 0xa2,
|
||||
PRIVATE_DATA = 0x06,
|
||||
AVC = 0x1b,
|
||||
HEVC = 0x24,
|
||||
|
||||
@@ -161,7 +161,7 @@ export class MpegTsMuxer extends Muxer {
|
||||
assert(meta?.decoderConfig);
|
||||
|
||||
const codec = track.source._codec;
|
||||
assert(codec === 'aac' || codec === 'mp3' || codec === 'ac3' || codec === 'eac3');
|
||||
assert(codec === 'aac' || codec === 'mp3' || codec === 'ac3' || codec === 'eac3' || codec === 'dts');
|
||||
|
||||
let streamType: MpegTsStreamType;
|
||||
let streamId: number;
|
||||
@@ -186,6 +186,11 @@ export class MpegTsMuxer extends Muxer {
|
||||
streamType = MpegTsStreamType.EAC3_SYSTEM_A;
|
||||
streamId = 0xbd;
|
||||
}; break;
|
||||
|
||||
case 'dts': {
|
||||
streamType = MpegTsStreamType.DTS;
|
||||
streamId = 0xbd;
|
||||
}; break;
|
||||
}
|
||||
|
||||
const pid = FIRST_TRACK_PID + this.trackDatas.length;
|
||||
@@ -414,7 +419,7 @@ export class MpegTsMuxer extends Muxer {
|
||||
): Uint8Array {
|
||||
const codec = (trackData.track as OutputAudioTrack).source._codec;
|
||||
|
||||
if (codec === 'mp3' || codec === 'ac3' || codec === 'eac3') {
|
||||
if (codec === 'mp3' || codec === 'ac3' || codec === 'eac3' || codec === 'dts') {
|
||||
// We're good
|
||||
return packet.data;
|
||||
}
|
||||
|
||||
@@ -1153,7 +1153,7 @@ export class MpegTsOutputFormat extends OutputFormat {
|
||||
getSupportedCodecs(): MediaCodec[] {
|
||||
return [
|
||||
...VIDEO_CODECS.filter(codec => ['avc', 'hevc'].includes(codec)),
|
||||
...AUDIO_CODECS.filter(codec => ['aac', 'mp3', 'ac3', 'eac3'].includes(codec)),
|
||||
...AUDIO_CODECS.filter(codec => ['aac', 'mp3', 'ac3', 'eac3', 'dts'].includes(codec)),
|
||||
];
|
||||
}
|
||||
|
||||
|
||||
@@ -0,0 +1,171 @@
|
||||
import { expect, test } from 'vitest';
|
||||
import path from 'node:path';
|
||||
import { Input } from '../../src/input.js';
|
||||
import { BufferSource, FilePathSource } from '../../src/source.js';
|
||||
import { ALL_FORMATS } from '../../src/input-format.js';
|
||||
import { Output } from '../../src/output.js';
|
||||
import { MkvOutputFormat, Mp4OutputFormat, MpegTsOutputFormat, OutputFormat } from '../../src/output-format.js';
|
||||
import { BufferTarget } from '../../src/target.js';
|
||||
import { Conversion } from '../../src/conversion.js';
|
||||
import { EncodedPacketSink } from '../../src/media-sink.js';
|
||||
import { EncodedPacket } from '../../src/packet.js';
|
||||
import { assert, uint8ArraysAreEqual } from '../../src/misc.js';
|
||||
|
||||
const __dirname = new URL('.', import.meta.url).pathname;
|
||||
|
||||
const DTSC_FILE = 'toothsome-dts.mp4';
|
||||
const ESDS_FILE = 'toothsome-dts-esds.mp4';
|
||||
|
||||
const MP4_TIMESTAMP_TOLERANCE = 0;
|
||||
const MATROSKA_TIMESTAMP_TOLERANCE = 1 / 1000;
|
||||
const MPEG_TS_TIMESTAMP_TOLERANCE = 1 / 90_000;
|
||||
|
||||
test('Read from MP4 with a dtsc sample entry', async () => {
|
||||
using input = new Input({
|
||||
source: new FilePathSource(path.join(__dirname, '..', 'public', DTSC_FILE)),
|
||||
formats: ALL_FORMATS,
|
||||
});
|
||||
|
||||
const track = await input.getPrimaryAudioTrack();
|
||||
assert(track);
|
||||
|
||||
const decoderConfig = await track.getDecoderConfig();
|
||||
assert(decoderConfig);
|
||||
|
||||
expect(await track.getCodec()).toBe('dts');
|
||||
expect(decoderConfig.codec).toBe('dtsc');
|
||||
expect(decoderConfig.description).toBeUndefined();
|
||||
|
||||
// This file declares 48000 in its sample entry, which the ddts box then corrects to the real rate
|
||||
expect(await track.getSampleRate()).toBe(24000);
|
||||
expect(await track.getNumberOfChannels()).toBe(2);
|
||||
|
||||
await expectStreamShape(input);
|
||||
});
|
||||
|
||||
test('Read from MP4 with an esds sample entry', async () => {
|
||||
using input = new Input({
|
||||
source: new FilePathSource(path.join(__dirname, '..', 'public', ESDS_FILE)),
|
||||
formats: ALL_FORMATS,
|
||||
});
|
||||
|
||||
const track = await input.getPrimaryAudioTrack();
|
||||
assert(track);
|
||||
|
||||
const decoderConfig = await track.getDecoderConfig();
|
||||
assert(decoderConfig);
|
||||
|
||||
expect(await track.getCodec()).toBe('dts');
|
||||
expect(decoderConfig.codec).toBe('dtsc');
|
||||
expect(decoderConfig.description).toBeUndefined();
|
||||
|
||||
expect(await track.getSampleRate()).toBe(24000);
|
||||
expect(await track.getNumberOfChannels()).toBe(2);
|
||||
|
||||
await expectStreamShape(input);
|
||||
});
|
||||
|
||||
test('Transmux dtsc MP4 into MP4', async () => {
|
||||
await expectTransmuxToPreservePackets(DTSC_FILE, new Mp4OutputFormat(), MP4_TIMESTAMP_TOLERANCE);
|
||||
});
|
||||
|
||||
test('Transmux dtsc MP4 into Matroska', async () => {
|
||||
await expectTransmuxToPreservePackets(DTSC_FILE, new MkvOutputFormat(), MATROSKA_TIMESTAMP_TOLERANCE);
|
||||
});
|
||||
|
||||
test('Transmux dtsc MP4 into MPEG-TS', async () => {
|
||||
await expectTransmuxToPreservePackets(DTSC_FILE, new MpegTsOutputFormat(), MPEG_TS_TIMESTAMP_TOLERANCE);
|
||||
});
|
||||
|
||||
test('Transmux esds MP4 into MP4', async () => {
|
||||
await expectTransmuxToPreservePackets(ESDS_FILE, new Mp4OutputFormat(), MP4_TIMESTAMP_TOLERANCE);
|
||||
});
|
||||
|
||||
test('Transmux esds MP4 into Matroska', async () => {
|
||||
await expectTransmuxToPreservePackets(ESDS_FILE, new MkvOutputFormat(), MATROSKA_TIMESTAMP_TOLERANCE);
|
||||
});
|
||||
|
||||
test('Transmux esds MP4 into MPEG-TS', async () => {
|
||||
await expectTransmuxToPreservePackets(ESDS_FILE, new MpegTsOutputFormat(), MPEG_TS_TIMESTAMP_TOLERANCE);
|
||||
});
|
||||
|
||||
const expectStreamShape = async (input: Input) => {
|
||||
const packets = await readAllPackets(input);
|
||||
|
||||
expect(packets.length).toBe(1805);
|
||||
expect(packets.every(packet => packet.type === 'key')).toBe(true);
|
||||
expect(packets.every(packet => packet.data.byteLength === 512)).toBe(true);
|
||||
};
|
||||
|
||||
/** Converts the file without re-encoding and checks that every packet comes back out unchanged. */
|
||||
const expectTransmuxToPreservePackets = async (
|
||||
fileName: string,
|
||||
outputFormat: OutputFormat,
|
||||
timestampTolerance: number,
|
||||
) => {
|
||||
using originalInput = new Input({
|
||||
source: new FilePathSource(path.join(__dirname, '..', 'public', fileName)),
|
||||
formats: ALL_FORMATS,
|
||||
});
|
||||
|
||||
const originalPackets = await readAllPackets(originalInput);
|
||||
|
||||
const output = new Output({
|
||||
format: outputFormat,
|
||||
target: new BufferTarget(),
|
||||
});
|
||||
|
||||
const conversion = await Conversion.init({ input: originalInput, output });
|
||||
await conversion.execute();
|
||||
|
||||
const { buffer } = output.target;
|
||||
assert(buffer);
|
||||
|
||||
using newInput = new Input({
|
||||
source: new BufferSource(buffer),
|
||||
formats: ALL_FORMATS,
|
||||
});
|
||||
|
||||
const newTrack = await newInput.getPrimaryAudioTrack();
|
||||
assert(newTrack);
|
||||
|
||||
const newDecoderConfig = await newTrack.getDecoderConfig();
|
||||
assert(newDecoderConfig);
|
||||
|
||||
expect(await newTrack.getCodec()).toBe('dts');
|
||||
expect(await newTrack.getSampleRate()).toBe(24000);
|
||||
expect(await newTrack.getNumberOfChannels()).toBe(2);
|
||||
|
||||
// Only the MP4 sample entry can carry this, so for the other formats it comes back out of the bitstream
|
||||
expect(newDecoderConfig.codec).toBe('dtsc');
|
||||
|
||||
const newPackets = await readAllPackets(newInput);
|
||||
expectPacketsToMatch(newPackets, originalPackets, timestampTolerance);
|
||||
};
|
||||
|
||||
const expectPacketsToMatch = (actual: EncodedPacket[], expected: EncodedPacket[], timestampTolerance: number) => {
|
||||
expect(actual.length).toBe(expected.length);
|
||||
|
||||
for (let i = 0; i < expected.length; i++) {
|
||||
const actualPacket = actual[i]!;
|
||||
const expectedPacket = expected[i]!;
|
||||
|
||||
expect(actualPacket.type).toBe(expectedPacket.type);
|
||||
expect(uint8ArraysAreEqual(actualPacket.data, expectedPacket.data)).toBe(true);
|
||||
expect(Math.abs(actualPacket.timestamp - expectedPacket.timestamp)).toBeLessThanOrEqual(timestampTolerance);
|
||||
}
|
||||
};
|
||||
|
||||
const readAllPackets = async (input: Input) => {
|
||||
const track = await input.getPrimaryAudioTrack();
|
||||
assert(track);
|
||||
|
||||
const sink = new EncodedPacketSink(track);
|
||||
const packets: EncodedPacket[] = [];
|
||||
|
||||
for await (const packet of sink.packets()) {
|
||||
packets.push(packet);
|
||||
}
|
||||
|
||||
return packets;
|
||||
};
|
||||
@@ -1400,6 +1400,9 @@ describe('Video', async () => {
|
||||
});
|
||||
|
||||
describe('Audio', async () => {
|
||||
/** A DTS bitrate that clears the encoder's per-frame minimum at the 48 kHz stereo the tests here use. */
|
||||
const DTS_BITRATE = 768000;
|
||||
|
||||
test('Decoder lifecycle', async () => {
|
||||
using input = new Input({
|
||||
source: new FilePathSource('./test/public/trim-buck-bunny-ffmpeg.ts'),
|
||||
@@ -1726,6 +1729,25 @@ describe('Audio', async () => {
|
||||
});
|
||||
});
|
||||
|
||||
test('DTS encode & decode', async () => {
|
||||
await encodeDecodeTest('dts', { bitrate: DTS_BITRATE }, async (packet, meta, i) => {
|
||||
expect(packet.type).toBe('key');
|
||||
expect(packet.duration).toBeCloseTo(512 / 48000);
|
||||
|
||||
if (i === 0) {
|
||||
expect(meta.decoderConfig).toBeDefined();
|
||||
expect(meta.decoderConfig!.codec).toBe('dtsc');
|
||||
expect(meta.decoderConfig!.numberOfChannels).toBe(2);
|
||||
expect(meta.decoderConfig!.sampleRate).toBe(48000);
|
||||
expect(meta.decoderConfig!.description).toBeUndefined();
|
||||
}
|
||||
}, async (sample) => {
|
||||
expect(sample.numberOfChannels).toBe(2);
|
||||
expect(sample.sampleRate).toBe(48000);
|
||||
expect(sample.duration).toBeCloseTo(512 / 48000);
|
||||
});
|
||||
});
|
||||
|
||||
for (const codec of NON_PCM_AUDIO_CODECS) {
|
||||
test(`${codec} encode & decode, negative timestamps`, async () => {
|
||||
await timestampTest(codec, -1);
|
||||
@@ -1747,7 +1769,10 @@ describe('Audio', async () => {
|
||||
const testDuration = codec !== 'opus';
|
||||
const testSampleStart = codec !== 'vorbis';
|
||||
|
||||
await encodeDecodeTest(codec, {}, async (packet, _meta, i) => {
|
||||
// The harness' default bitrate is below what DTS needs to fit its per-channel side info into a frame
|
||||
const extraConfig = codec === 'dts' ? { bitrate: DTS_BITRATE } : {};
|
||||
|
||||
await encodeDecodeTest(codec, extraConfig, async (packet, _meta, i) => {
|
||||
if (i === 0) {
|
||||
expect(packet.timestamp).toBe(startTimestamp);
|
||||
} else if (testDuration) {
|
||||
|
||||
Binary file not shown.
Binary file not shown.
Reference in New Issue
Block a user