Address a bunch of TODOs and add parameter validation everywhere

This commit is contained in:
Vanilagy
2024-11-24 21:17:00 +01:00
parent 29749451bb
commit b51b2b1b45
14 changed files with 1846 additions and 1059 deletions
+61 -32
View File
@@ -146,6 +146,31 @@ const fixed_2_30 = (value: number) => {
return [bytes[0], bytes[1], bytes[2], bytes[3]] as number[];
};
const variableUnsignedInt = (value: number, byteLength?: number) => {
let bytes: number[] = [];
let remaining = value;
do {
let byte = remaining & 0x7f;
remaining >>= 7;
// If this isn't the first byte we're adding (meaning there will be more bytes after it
// when we reverse the array), set the continuation bit
if (bytes.length > 0) {
byte |= 0x80;
}
bytes.push(byte);
if (byteLength !== undefined) {
byteLength--;
}
} while (remaining > 0 || byteLength);
// Reverse the array since we built it backwards
return bytes.reverse();
};
const ascii = (text: string, nullTerminated = false) => {
let bytes = Array(text.length).fill(null).map((_, i) => text.charCodeAt(i));
if (nullTerminated) bytes.push(0x00);
@@ -279,7 +304,7 @@ export const mvhd = (
return lastSample.timestamp + lastSample.duration;
})
), GLOBAL_TIMESCALE);
let nextTrackId = Math.max(...trackDatas.map(x => x.track.id)) + 1;
let nextTrackId = Math.max(0, ...trackDatas.map(x => x.track.id)) + 1;
// Conditionally use u64 if u32 isn't enough
let needsU64 = !isU32(creationTime) || !isU32(duration);
@@ -538,9 +563,6 @@ export const colr = (trackData: IsobmffVideoTrackData) => box('colr', [
u8((trackData.info.decoderConfig.colorSpace!.fullRange ? 1 : 0) << 7) // Full range flag
]);
// TODO: All muxers should ensure that the decoder config description is provided for the codecs that require it. This
// is relevant when the user skips WebCodecs and uses their own encoder.
/** AVC Configuration Box: Provides additional information to the decoder. */
export const avcC = (trackData: IsobmffVideoTrackData) => trackData.info.decoderConfig && box('avcC', [
// For AVC, description is an AVCDecoderConfigurationRecord, so nothing else to do here
@@ -549,7 +571,7 @@ export const avcC = (trackData: IsobmffVideoTrackData) => trackData.info.decoder
/** HEVC Configuration Box: Provides additional information to the decoder. */
export const hvcC = (trackData: IsobmffVideoTrackData) => trackData.info.decoderConfig && box('hvcC', [
// For HEVC, description is a HEVCDecoderConfigurationRecord, so nothing else to do here
// For HEVC, description is an HEVCDecoderConfigurationRecord, so nothing else to do here
...toUint8Array(trackData.info.decoderConfig.description!)
]);
@@ -562,9 +584,7 @@ export const vpcC = (trackData: IsobmffVideoTrackData) => {
}
let decoderConfig = trackData.info.decoderConfig;
if (!decoderConfig.colorSpace) {
throw new Error(`'colorSpace' is required in the decoder config for VP8/VP9.`);
}
assert(decoderConfig.colorSpace); // This is guaranteed by an earlier validation step
let parts = decoderConfig.codec.split('.');
let profile = Number(parts[1]);
@@ -631,28 +651,39 @@ export const soundSampleDescription = (
export const esds = (trackData: IsobmffAudioTrackData) => {
let description = toUint8Array(trackData.info.decoderConfig.description ?? new ArrayBuffer(0));
// TODO Compact the 808080 stuff, it's superfluous
// Adapted from https://stackoverflow.com/a/54803118
// We build up the bytes in a layered way which reflects the nested structure
return fullBox('esds', 0, 0, [
// https://stackoverflow.com/a/54803118
u32(0x03808080), // TAG(3) = Object Descriptor ([2])
u8(0x20 + description.byteLength), // length of this OD (which includes the next 2 tags)
u16(1), // ES_ID = 1
u8(0x00), // flags etc = 0
u32(0x04808080), // TAG(4) = ES Descriptor ([2]) embedded in above OD
u8(0x12 + description.byteLength), // length of this ESD
u8(0x40), // MPEG-4 Audio
u8(0x15), // stream type(6bits)=5 audio, flags(2bits)=1
u24(0), // 24bit buffer size
u32(0x0001FC17), // max bitrate
u32(0x0001FC17), // avg bitrate
u32(0x05808080), // TAG(5) = ASC ([2],[3]) embedded in above OD
u8(description.byteLength), // length
...description,
u32(0x06808080), // TAG(6)
u8(0x01), // length
u8(0x02) // data
]);
let bytes = [
...description
];
bytes = [
...u8(0x40), // MPEG-4 Audio
...u8(0x15), // stream type(6bits)=5 audio, flags(2bits)=1
...u24(0), // 24bit buffer size
...u32(0), // max bitrate
...u32(0), // avg bitrate
...u8(0x05), // TAG(5) = ASC ([2],[3]) embedded in above OD
...variableUnsignedInt(bytes.length),
...bytes
];
bytes = [
...u16(1), // ES_ID = 1
...u8(0x00), // flags etc = 0
...u8(0x04), // TAG(4) = ES Descriptor ([2]) embedded in above OD
...variableUnsignedInt(bytes.length),
...bytes,
...u8(0x06), // TAG(6)
...u8(0x01), // length
...u8(0x02) // data
];
bytes = [
...u8(0x03), // TAG(3) = Object Descriptor ([2])
...variableUnsignedInt(bytes.length),
...bytes
];
return fullBox('esds', 0, 0, bytes);
};
/** Opus Specific Box. */
@@ -665,9 +696,7 @@ export const dOps = (trackData: IsobmffAudioTrackData) => {
// https://www.rfc-editor.org/rfc/rfc7845#section-5
const description = trackData.info.decoderConfig?.description;
if (description) {
if (description.byteLength < 18) {
throw new TypeError('Invalid decoder description provided for Opus; must be at least 18 bytes long.');
}
assert(description.byteLength < 18); // Is validated in an earlier step
const view = ArrayBuffer.isView(description)
? new DataView(description.buffer, description.byteOffset, description.byteLength)
+48 -112
View File
@@ -1,11 +1,12 @@
import { Box, free, ftyp, IsobmffBoxWriter, mdat, mfra, moof, moov, vtta, vttc, vtte } from './isobmff_boxes';
import { Muxer } from '../muxer';
import { Output, OutputAudioTrack, OutputSubtitleTrack, OutputTrack, OutputVideoTrack } from '../output';
import { Writer } from '../writer';
import { ArrayBufferTargetWriter, Writer } from '../writer';
import { assert, last, TransformationMatrix } from '../misc';
import { Mp4OutputFormat } from '../output_format';
import { inlineTimestampRegex, SubtitleConfig, SubtitleCue, SubtitleMetadata } from '../subtitles';
import { ArrayBufferTarget } from '../target';
import { validateAudioChunkMetadata, validateSubtitleMetadata, validateVideoChunkMetadata } from '../codec';
export const GLOBAL_TIMESCALE = 1000;
const TIMESTAMP_OFFSET = 2_082_844_800; // Seconds between Jan 1 1904 and Jan 1 1970
@@ -88,6 +89,7 @@ export class IsobmffMuxer extends Muxer {
#writer: Writer;
#boxWriter: IsobmffBoxWriter;
#format: Mp4OutputFormat;
#fastStart: NonNullable<Mp4OutputFormat['options']['fastStart']>;
#auxTarget = new ArrayBufferTarget();
#auxWriter = this.#auxTarget.createWriter();
@@ -109,6 +111,14 @@ export class IsobmffMuxer extends Muxer {
this.#writer = output.writer;
this.#boxWriter = new IsobmffBoxWriter(this.#writer);
this.#format = format;
// If the fastStart option isn't defined, enable in-memory fast start if the target is an ArrayBuffer, as the
// memory usage remains identical
this.#fastStart = format.options.fastStart ?? (this.#writer instanceof ArrayBufferTargetWriter ? 'in-memory' : false);
if (this.#fastStart === 'in-memory' || this.#fastStart === 'fragmented') {
this.#writer.ensureMonotonicity = true;
}
}
start() {
@@ -117,21 +127,16 @@ export class IsobmffMuxer extends Muxer {
// Write the header
this.#boxWriter.writeBox(ftyp({
holdsAvc: holdsAvc,
fragmented: this.#format.options.fastStart === 'fragmented'
fragmented: this.#fastStart === 'fragmented'
}));
this.#ftypSize = this.#writer.getPos();
if (this.#format.options.fastStart === 'in-memory') {
if (this.#fastStart === 'in-memory') {
this.#mdat = mdat(false);
} else if (this.#format.options.fastStart === 'fragmented') {
} else if (this.#fastStart === 'fragmented') {
// We write the moov box once we write out the first fragment to make sure we get the decoder configs
} else {
if (typeof this.#format.options.fastStart === 'object') {
let moovSizeUpperBound = this.#computeMoovSizeUpperBound();
this.#writer.seek(this.#writer.getPos() + moovSizeUpperBound);
}
this.#mdat = mdat(true); // Reserve large size by default, can refine this when finalizing.
this.#boxWriter.writeBox(this.#mdat);
}
@@ -139,45 +144,14 @@ export class IsobmffMuxer extends Muxer {
this.#writer.flush();
}
#computeMoovSizeUpperBound() {
assert(typeof this.#format.options.fastStart === 'object');
let upperBound = 0;
let sampleCounts = [
this.#format.options.fastStart.expectedVideoChunks,
this.#format.options.fastStart.expectedAudioChunks
];
for (let n of sampleCounts) {
if (!n) continue;
// Given the max allowed sample count, compute the space they'll take up in the Sample Table Box, assuming
// the worst case for each individual box:
// stts box - since it is compactly coded, the maximum length of this table will be 2/3n
upperBound += (4 + 4) * Math.ceil(2/3 * n);
// stss box - 1 entry per sample
upperBound += 4 * n;
// stsc box - since it is compactly coded, the maximum length of this table will be 2/3n
upperBound += (4 + 4 + 4) * Math.ceil(2/3 * n);
// stsz box - 1 entry per sample
upperBound += 4 * n;
// co64 box - we assume 1 sample per chunk and 64-bit chunk offsets
upperBound += 8 * n;
}
upperBound += 4096; // Assume a generous 4 kB for everything else: Track metadata, codec descriptors, etc.
return upperBound;
}
#getVideoTrackData(track: OutputVideoTrack, meta?: EncodedVideoChunkMetadata) {
const existingTrackData = this.#trackDatas.find(x => x.track === track);
if (existingTrackData) {
return existingTrackData as IsobmffVideoTrackData;
}
// TODO Make proper errors for these
validateVideoChunkMetadata(meta);
assert(meta);
assert(meta.decoderConfig);
assert(meta.decoderConfig.codedWidth !== undefined);
@@ -216,7 +190,8 @@ export class IsobmffMuxer extends Muxer {
return existingTrackData as IsobmffAudioTrackData;
}
// TODO Make proper errors for these
validateAudioChunkMetadata(meta);
assert(meta);
assert(meta.decoderConfig);
@@ -253,7 +228,8 @@ export class IsobmffMuxer extends Muxer {
return existingTrackData as IsobmffSubtitleTrackData;
}
// TODO Make proper errors for these
validateSubtitleMetadata(meta);
assert(meta);
assert(meta.config);
@@ -293,62 +269,30 @@ export class IsobmffMuxer extends Muxer {
addEncodedVideoChunk(track: OutputVideoTrack, chunk: EncodedVideoChunk, meta?: EncodedVideoChunkMetadata) {
const trackData = this.#getVideoTrackData(track, meta);
if (
typeof this.#format.options.fastStart === 'object' &&
trackData.samples.length === this.#format.options.fastStart.expectedVideoChunks
) {
// TODO reference track id
throw new Error(`Cannot add more video chunks than specified in 'fastStart' (${
this.#format.options.fastStart.expectedVideoChunks
}).`);
}
let data = new Uint8Array(chunk.byteLength);
chunk.copyTo(data);
let timestamp = this.validateAndNormalizeTimestamp(trackData.track, chunk.timestamp, chunk.type === 'key');
let sample = this.#createSampleForTrack(trackData, data, timestamp, (chunk.duration ?? 0) / 1e6, chunk.type);
if (this.#format.options.fastStart === 'fragmented') {
trackData.sampleQueue.push(sample);
this.#interleaveSamples();
} else {
this.#addSampleToTrack(trackData, sample);
}
this.#registerSample(trackData, sample);
}
addEncodedAudioChunk(track: OutputAudioTrack, chunk: EncodedAudioChunk, meta?: EncodedAudioChunkMetadata) {
const trackData = this.#getAudioTrackData(track, meta);
if (
typeof this.#format.options.fastStart === 'object' &&
trackData.samples.length === this.#format.options.fastStart.expectedAudioChunks
) {
// TODO reference track id
throw new Error(`Cannot add more audio chunks than specified in 'fastStart' (${
this.#format.options.fastStart.expectedAudioChunks
}).`);
}
let data = new Uint8Array(chunk.byteLength);
chunk.copyTo(data);
let timestamp = this.validateAndNormalizeTimestamp(trackData.track, chunk.timestamp, chunk.type === 'key');
let sample = this.#createSampleForTrack(trackData, data, timestamp, (chunk.duration ?? 0) / 1e6, chunk.type);
if (this.#format.options.fastStart === 'fragmented') {
trackData.sampleQueue.push(sample);
this.#interleaveSamples();
} else {
this.#addSampleToTrack(trackData, sample);
}
this.#registerSample(trackData, sample);
}
addSubtitleCue(track: OutputSubtitleTrack, cue: SubtitleCue, meta?: SubtitleMetadata) {
const trackData = this.#getSubtitleTrackData(track, meta);
// TODO the expectedSubtitleChunks thing
this.validateAndNormalizeTimestamp(trackData.track, 1e6 * cue.timestamp, true);
if (track.source.codec === 'webvtt') {
@@ -392,14 +336,7 @@ export class IsobmffMuxer extends Muxer {
let body = this.#auxWriter.getSlice(0, this.#auxWriter.getPos());
let sample = this.#createSampleForTrack(trackData, body, trackData.lastCueEndTimestamp, sampleStart - trackData.lastCueEndTimestamp, 'key');
// todo extract this into reusable thing
if (this.#format.options.fastStart === 'fragmented') {
trackData.sampleQueue.push(sample);
this.#interleaveSamples();
} else {
this.#addSampleToTrack(trackData, sample);
}
this.#registerSample(trackData, sample);
trackData.lastCueEndTimestamp = sampleStart;
}
@@ -442,14 +379,7 @@ export class IsobmffMuxer extends Muxer {
let body = this.#auxWriter.getSlice(0, this.#auxWriter.getPos());
let sample = this.#createSampleForTrack(trackData, body, sampleStart, sampleEnd - sampleStart, 'key');
// todo extract this into reusable thing
if (this.#format.options.fastStart === 'fragmented') {
trackData.sampleQueue.push(sample);
this.#interleaveSamples();
} else {
this.#addSampleToTrack(trackData, sample);
}
this.#registerSample(trackData, sample);
trackData.lastCueEndTimestamp = sampleEnd;
}
}
@@ -503,7 +433,7 @@ export class IsobmffMuxer extends Muxer {
trackData.lastTimescaleUnits += delta;
trackData.lastSample.timescaleUnitsToNextSample = delta;
if (this.#format.options.fastStart !== 'fragmented') {
if (this.#fastStart !== 'fragmented') {
let lastTableEntry = last(trackData.timeToSampleTable);
assert(lastTableEntry);
@@ -555,7 +485,7 @@ export class IsobmffMuxer extends Muxer {
} else {
trackData.lastTimescaleUnits = 0;
if (this.#format.options.fastStart !== 'fragmented') {
if (this.#fastStart !== 'fragmented') {
trackData.timeToSampleTable.push({
sampleCount: 1,
sampleDelta: durationInTimescale
@@ -573,15 +503,21 @@ export class IsobmffMuxer extends Muxer {
trackData.timestampProcessingQueue.length = 0;
}
#addSampleToTrack(
trackData: IsobmffTrackData,
sample: Sample
) {
#registerSample(trackData: IsobmffTrackData, sample: Sample) {
if (this.#fastStart === 'fragmented') {
trackData.sampleQueue.push(sample);
this.#interleaveSamples();
} else {
this.#addSampleToTrack(trackData, sample);
}
}
#addSampleToTrack(trackData: IsobmffTrackData, sample: Sample) {
if (sample.type === 'key') {
this.#processTimestamps(trackData);
}
if (this.#format.options.fastStart !== 'fragmented') {
if (this.#fastStart !== 'fragmented') {
trackData.samples.push(sample);
}
@@ -591,7 +527,7 @@ export class IsobmffMuxer extends Muxer {
} else {
let currentChunkDuration = sample.timestamp - trackData.currentChunk.startTimestamp;
if (this.#format.options.fastStart === 'fragmented') {
if (this.#fastStart === 'fragmented') {
// We can only finalize this fragment (and begin a new one) if we know that each track will be able to
// start the new one with a key frame.
const keyFrameQueuedEverywhere = this.#trackDatas.every(otherTrackData => {
@@ -631,7 +567,7 @@ export class IsobmffMuxer extends Muxer {
}
#finalizeCurrentChunk(trackData: IsobmffTrackData) {
assert(this.#format.options.fastStart !== 'fragmented');
assert(this.#fastStart !== 'fragmented');
if (!trackData.currentChunk) return;
@@ -648,7 +584,7 @@ export class IsobmffMuxer extends Muxer {
});
}
if (this.#format.options.fastStart === 'in-memory') {
if (this.#fastStart === 'in-memory') {
trackData.currentChunk.offset = 0; // We'll compute the proper offset when finalizing
return;
}
@@ -665,7 +601,7 @@ export class IsobmffMuxer extends Muxer {
}
#interleaveSamples() {
assert(this.#format.options.fastStart === 'fragmented');
assert(this.#fastStart === 'fragmented');
for (const track of this.output.tracks) {
if (!track.source.closed && !this.#trackDatas.some(x => x.track === track)) {
@@ -699,7 +635,7 @@ export class IsobmffMuxer extends Muxer {
}
#finalizeFragment(flushWriter = true) {
assert(this.#format.options.fastStart === 'fragmented');
assert(this.#fastStart === 'fragmented');
let fragmentNumber = this.#nextFragmentNumber++;
@@ -775,7 +711,7 @@ export class IsobmffMuxer extends Muxer {
}
}
if (this.#format.options.fastStart === 'fragmented') {
if (this.#fastStart === 'fragmented') {
// Since a track is now closed, we may be able to write out chunks that were previously waiting
this.#interleaveSamples();
}
@@ -789,7 +725,7 @@ export class IsobmffMuxer extends Muxer {
}
}
if (this.#format.options.fastStart === 'fragmented') {
if (this.#fastStart === 'fragmented') {
for (let trackData of this.#trackDatas) {
for (let sample of trackData.sampleQueue) {
this.#addSampleToTrack(trackData, sample);
@@ -800,13 +736,13 @@ export class IsobmffMuxer extends Muxer {
this.#finalizeFragment(false); // Don't flush the last fragment as we will flush it with the mfra box soon
} else {
for (let trackData of this.#trackDatas) {
for (let trackData of this.#trackDatas) {
this.#processTimestamps(trackData);
this.#finalizeCurrentChunk(trackData);
}
}
if (this.#format.options.fastStart === 'in-memory') {
if (this.#fastStart === 'in-memory') {
assert(this.#mdat);
let mdatSize: number;
@@ -850,7 +786,7 @@ export class IsobmffMuxer extends Muxer {
sample.data = null;
}
}
} else if (this.#format.options.fastStart === 'fragmented') {
} else if (this.#fastStart === 'fragmented') {
// Append the mfra box to the end of the file for better random access
let startPos = this.#writer.getPos();
let mfraBox = mfra(this.#trackDatas);
@@ -873,7 +809,7 @@ export class IsobmffMuxer extends Muxer {
let movieBox = moov(this.#trackDatas, this.#creationTime);
if (typeof this.#format.options.fastStart === 'object') {
if (typeof this.#fastStart === 'object') {
this.#writer.seek(this.#ftypSize);
this.#boxWriter.writeBox(movieBox);