diff --git a/dev/convert.html b/dev/convert.html index 9f1a6dc..459e554 100644 --- a/dev/convert.html +++ b/dev/convert.html @@ -21,7 +21,7 @@ chunked: true, chunkSize: 2**20 }); - const outputFormat = new Metamuxer.WavOutputFormat(); + const outputFormat = new Metamuxer.Mp4OutputFormat(); const button = document.createElement('button'); button.textContent = 'Cancel'; @@ -103,12 +103,12 @@ console.log("Done", target.buffer); const video = document.createElement('video'); - video.src = URL.createObjectURL(new Blob([target.buffer])); + video.src = URL.createObjectURL(new Blob([target.buffer], { type: outputFormat.mimeType })); video.controls = true; document.body.append(video); video.play(); - download(new Blob([target.buffer]), 'converted' + outputFormat.fileExtension); + //download(new Blob([target.buffer]), 'converted' + outputFormat.fileExtension); function download(blob, filename) { const url = URL.createObjectURL(blob); diff --git a/src/avc.ts b/src/avc.ts new file mode 100644 index 0000000..28a9d1d --- /dev/null +++ b/src/avc.ts @@ -0,0 +1,281 @@ +import { assert, Bitstream, readExpGolomb } from './misc'; + +// References: +// ISO 14496-15 +// ITU-T-REC-H.264 +// https://stackoverflow.com/questions/24884827 + +/** Finds all NAL units in an AVC packet in Annex B format. */ +const findNalUnitsInAnnexB = (packetData: Uint8Array) => { + const nalUnits: { type: number; data: Uint8Array }[] = []; + let i = 0; + + while (i < packetData.length) { + let startCodePos = -1; + let startCodeLength = 0; + + for (let j = i; j < packetData.length - 3; j++) { + // Check for 3-byte start code (0x000001) + if (packetData[j] === 0 && packetData[j + 1] === 0 && packetData[j + 2] === 1) { + startCodePos = j; + startCodeLength = 3; + break; + } + + // Check for 4-byte start code (0x00000001) + if ( + j < packetData.length - 4 + && packetData[j] === 0 + && packetData[j + 1] === 0 + && packetData[j + 2] === 0 + && packetData[j + 3] === 1 + ) { + startCodePos = j; + startCodeLength = 4; + break; + } + } + + if (startCodePos === -1) { + break; // No more start codes found + } + + // If this isn't the first start code, extract the previous NAL unit + if (i > 0 && startCodePos > i) { + const nalData = packetData.subarray(i, startCodePos); + if (nalData.length > 0) { + const nalType = nalData[0]! & 0x1f; + nalUnits.push({ type: nalType, data: nalData }); + } + } + + i = startCodePos + startCodeLength; + } + + // Extract the last NAL unit if there is one + if (i < packetData.length) { + const nalData = packetData.subarray(i); + if (nalData.length > 0) { + const nalType = nalData[0]! & 0x1f; + nalUnits.push({ type: nalType, data: nalData }); + } + } + + return nalUnits; +}; + +// Data specified in ISO 14496-15 +export type AvcDecoderConfigurationRecord = { + configurationVersion: number; + avcProfileIndication: number; + profileCompatibility: number; + avcLevelIndication: number; + lengthSizeMinusOne: number; + sequenceParameterSets: { + sequenceParameterSetLength: number; + sequenceParameterSetNalUnit: Uint8Array; + }[]; + pictureParameterSets: { + pictureParameterSetLength: number; + pictureParameterSetNalUnit: Uint8Array; + }[]; + + // Fields only for specific profiles: + chromaFormat: number | null; + bitDepthLumaMinus8: number | null; + bitDepthChromaMinus8: number | null; + sequenceParameterSetExt: { + sequenceParameterSetExtLength: number; + sequenceParameterSetExtNalUnit: Uint8Array; + }[] | null; +}; + +/** Builds an AvcDecoderConfigurationRecord from an AVC packet in Annex B format. */ +export const extractAvcDecoderConfigurationRecord = (packetData: Uint8Array) => { + try { + const nalUnits = findNalUnitsInAnnexB(packetData); + + const spsUnits = nalUnits.filter(unit => unit.type === 7); + const ppsUnits = nalUnits.filter(unit => unit.type === 8); + const spsExtUnits = nalUnits.filter(unit => unit.type === 13); + + if (spsUnits.length === 0) { + return null; + } + + if (ppsUnits.length === 0) { + return null; + } + + // Let's get the first SPS for profile and level information + const spsData = spsUnits[0]!.data; + const bitstream = new Bitstream(spsData); + + bitstream.skipBits(1); // forbidden_zero_bit + bitstream.skipBits(2); // nal_ref_idc + const nal_unit_type = bitstream.readBits(5); + + if (nal_unit_type !== 7) { // SPS NAL unit type is 7 + console.error('Invalid SPS NAL unit type'); + return null; + } + + const profile_idc = bitstream.readAlignedByte(); + const constraint_flags = bitstream.readAlignedByte(); + const level_idc = bitstream.readAlignedByte(); + + const record: AvcDecoderConfigurationRecord = { + configurationVersion: 1, + avcProfileIndication: profile_idc, + profileCompatibility: constraint_flags, + avcLevelIndication: level_idc, + lengthSizeMinusOne: 3, // Typically 4 bytes for length field + sequenceParameterSets: spsUnits.map(unit => ({ + sequenceParameterSetLength: unit.data.length, + sequenceParameterSetNalUnit: unit.data, + })), + pictureParameterSets: ppsUnits.map(unit => ({ + pictureParameterSetLength: unit.data.length, + pictureParameterSetNalUnit: unit.data, + })), + chromaFormat: null, + bitDepthLumaMinus8: null, + bitDepthChromaMinus8: null, + sequenceParameterSetExt: null, + }; + + if ( + profile_idc === 100 + || profile_idc === 110 + || profile_idc === 122 + || profile_idc === 144 + ) { + readExpGolomb(bitstream); // seq_parameter_set_id + + const chroma_format_idc = readExpGolomb(bitstream); + + if (chroma_format_idc === 3) { + bitstream.skipBits(1); // separate_colour_plane_flag + } + + const bit_depth_luma_minus8 = readExpGolomb(bitstream); + + const bit_depth_chroma_minus8 = readExpGolomb(bitstream); + + record.chromaFormat = chroma_format_idc; + record.bitDepthLumaMinus8 = bit_depth_luma_minus8; + record.bitDepthChromaMinus8 = bit_depth_chroma_minus8; + + record.sequenceParameterSetExt = spsExtUnits.map(unit => ({ + sequenceParameterSetExtLength: unit.data.length, + sequenceParameterSetExtNalUnit: unit.data, + })); + } + + return record; + } catch (error) { + console.error('Error building AVC Decoder Configuration Record:', error); + return null; + } +}; + +/** Serializes an AvcDecoderConfigurationRecord into the format specified in Section 5.3.3.1 of ISO 14496-15. */ +export const serializeAvcDecoderConfigurationRecord = (record: AvcDecoderConfigurationRecord) => { + const bytes: number[] = []; + + // Write header + bytes.push(record.configurationVersion); + bytes.push(record.avcProfileIndication); + bytes.push(record.profileCompatibility); + bytes.push(record.avcLevelIndication); + bytes.push(0xFC | (record.lengthSizeMinusOne & 0x03)); // Reserved bits (6) + lengthSizeMinusOne (2) + + // Reserved bits (3) + numOfSequenceParameterSets (5) + bytes.push(0xE0 | (record.sequenceParameterSets.length & 0x1F)); + + // Write SPS + for (const sps of record.sequenceParameterSets) { + bytes.push(sps.sequenceParameterSetLength >> 8); // High byte + bytes.push(sps.sequenceParameterSetLength & 0xFF); // Low byte + + for (let i = 0; i < sps.sequenceParameterSetLength; i++) { + bytes.push(sps.sequenceParameterSetNalUnit[i]!); + } + } + + bytes.push(record.pictureParameterSets.length); + + // Write PPS + for (const pps of record.pictureParameterSets) { + bytes.push(pps.pictureParameterSetLength >> 8); // High byte + bytes.push(pps.pictureParameterSetLength & 0xFF); // Low byte + + for (let i = 0; i < pps.pictureParameterSetLength; i++) { + bytes.push(pps.pictureParameterSetNalUnit[i]!); + } + } + + if ( + record.avcProfileIndication === 100 + || record.avcProfileIndication === 110 + || record.avcProfileIndication === 122 + || record.avcProfileIndication === 144 + ) { + assert(record.chromaFormat !== null); + assert(record.bitDepthLumaMinus8 !== null); + assert(record.bitDepthChromaMinus8 !== null); + assert(record.sequenceParameterSetExt !== null); + + bytes.push(0xFC | (record.chromaFormat & 0x03)); // Reserved bits + chroma_format + bytes.push(0xF8 | (record.bitDepthLumaMinus8 & 0x07)); // Reserved bits + bit_depth_luma_minus8 + bytes.push(0xF8 | (record.bitDepthChromaMinus8 & 0x07)); // Reserved bits + bit_depth_chroma_minus8 + + bytes.push(record.sequenceParameterSetExt.length); + + // Write SPS Ext + for (const spsExt of record.sequenceParameterSetExt) { + bytes.push(spsExt.sequenceParameterSetExtLength >> 8); // High byte + bytes.push(spsExt.sequenceParameterSetExtLength & 0xFF); // Low byte + + for (let i = 0; i < spsExt.sequenceParameterSetExtLength; i++) { + bytes.push(spsExt.sequenceParameterSetExtNalUnit[i]!); + } + } + } + + return new Uint8Array(bytes); +}; + +/** Converts an AVC packet in Annex B format to AVCC format. */ +export const transformAnnexBToAvcc = (packetData: Uint8Array) => { + const NAL_UNIT_LENGTH_SIZE = 4; + + const nalUnits = findNalUnitsInAnnexB(packetData); + + if (nalUnits.length === 0) { + // If no NAL units were found, it's not valid Annex B data + return null; + } + + let totalSize = 0; + for (const nalUnit of nalUnits) { + totalSize += NAL_UNIT_LENGTH_SIZE + nalUnit.data.length; + } + + const avccData = new Uint8Array(totalSize); + const dataView = new DataView(avccData.buffer); + let offset = 0; + + // Write each NAL unit with its length prefix + for (const nalUnit of nalUnits) { + const length = nalUnit.data.length; + + dataView.setUint32(offset, length, false); + offset += 4; + + avccData.set(nalUnit.data, offset); + offset += nalUnit.data.length; + } + + return avccData; +}; diff --git a/src/codec.ts b/src/codec.ts index 5cb92db..8414b69 100644 --- a/src/codec.ts +++ b/src/codec.ts @@ -1,5 +1,7 @@ +import { AvcDecoderConfigurationRecord } from './avc'; import { customAudioEncoders, customVideoEncoders } from './custom-coder'; import { + Bitstream, COLOR_PRIMARIES_MAP, MATRIX_COEFFICIENTS_MAP, TRANSFER_CHARACTERISTICS_MAP, @@ -7,7 +9,6 @@ import { bytesToHexString, isAllowSharedBufferSource, last, - readBits, reverseBitsU32, toDataView, } from './misc'; @@ -335,20 +336,31 @@ export const extractVideoCodecString = (trackInfo: { codec: VideoCodec | null; codecDescription: Uint8Array | null; colorSpace: VideoColorSpaceInit | null; + avcCodecInfo: AvcDecoderConfigurationRecord | null; vp9CodecInfo: Vp9CodecInfo | null; av1CodecInfo: Av1CodecInfo | null; }) => { - const { codec, codecDescription: description, colorSpace, vp9CodecInfo, av1CodecInfo } = trackInfo; + const { codec, codecDescription: description, colorSpace, avcCodecInfo, vp9CodecInfo, av1CodecInfo } = trackInfo; if (codec === 'avc') { + if (avcCodecInfo) { + const bytes = new Uint8Array([ + avcCodecInfo.avcProfileIndication, + avcCodecInfo.profileCompatibility, + avcCodecInfo.avcLevelIndication, + ]); + + return `avc1.${bytesToHexString(bytes)}`; + } + if (!description || description.byteLength < 4) { - throw new TypeError('AVC description must be at least 4 bytes long.'); + throw new TypeError('AVC decoder description is not provided or is not at least 4 bytes long.'); } return `avc1.${bytesToHexString(description.subarray(1, 4))}`; } else if (codec === 'hevc') { if (!description) { - throw new TypeError('HEVC description must be provided.'); + throw new TypeError('HEVC decoder description is not provided.'); } const view = toDataView(description); @@ -480,11 +492,9 @@ export const extractVideoCodecString = (trackInfo: { throw new TypeError(`Unhandled codec '${codec}'.`); }; -/** - * When VP9 codec information is not provided by the container, we can still try to extract the information by digging - * into the VP9 bitstream. - */ -export const extractVp9CodecInfoFromFrame = (frame: Uint8Array): Vp9CodecInfo | null => { +export const extractVp9CodecInfoFromFrame = ( + frame: Uint8Array, +): Vp9CodecInfo | null => { // eslint-disable-next-line @stylistic/max-len // https://storage.googleapis.com/downloads.webmproject.org/docs/vp9/vp9-bitstream-specification-v0.7-20170222-draft.pdf // http://downloads.webmproject.org/docs/vp9/vp9-bitstream_superframe-and-uncompressed-header_v1.0.pdf @@ -513,91 +523,80 @@ export const extractVp9CodecInfoFromFrame = (frame: Uint8Array): Vp9CodecInfo | frame = frame.subarray(0, frameSize); } - let bitPos = 0; + const bitstream = new Bitstream(frame); // Frame marker (0b10) - const frameMarker = readBits(frame, bitPos, bitPos + 2); + const frameMarker = bitstream.readBits(2); if (frameMarker !== 2) { return null; } - bitPos += 2; // Profile - const profileLowBit = readBits(frame, bitPos, bitPos + 1); - bitPos += 1; - const profileHighBit = readBits(frame, bitPos, bitPos + 1); - bitPos += 1; + const profileLowBit = bitstream.readBits(1); + const profileHighBit = bitstream.readBits(1); const profile = (profileHighBit << 1) + profileLowBit; // Skip reserved bit for profile 3 if (profile === 3) { - bitPos += 1; + bitstream.skipBits(1); } // show_existing_frame - const showExistingFrame = readBits(frame, bitPos, bitPos + 1); - bitPos += 1; + const showExistingFrame = bitstream.readBits(1); if (showExistingFrame === 1) { return null; } // frame_type (0 = key frame) - const frameType = readBits(frame, bitPos, bitPos + 1); - bitPos += 1; + const frameType = bitstream.readBits(1); if (frameType !== 0) { return null; } // Skip show_frame and error_resilient_mode - bitPos += 2; + bitstream.skipBits(2); // Sync code (0x498342) - const syncCode = readBits(frame, bitPos, bitPos + 24); + const syncCode = bitstream.readBits(24); if (syncCode !== 0x498342) { return null; } - bitPos += 24; // Color config let bitDepth = 8; if (profile >= 2) { - const tenOrTwelveBit = readBits(frame, bitPos, bitPos + 1); - bitPos += 1; + const tenOrTwelveBit = bitstream.readBits(1); bitDepth = tenOrTwelveBit ? 12 : 10; } // Color space - const colorSpace = readBits(frame, bitPos, bitPos + 3); - bitPos += 3; + const colorSpace = bitstream.readBits(3); let chromaSubsampling = 0; let videoFullRangeFlag = 0; if (colorSpace !== 7) { // 7 is CS_RGB - const colorRange = readBits(frame, bitPos, bitPos + 1); - bitPos += 1; + const colorRange = bitstream.readBits(1); videoFullRangeFlag = colorRange; if (profile === 1 || profile === 3) { - const subsamplingX = readBits(frame, bitPos, bitPos + 1); - bitPos += 1; - const subsamplingY = readBits(frame, bitPos, bitPos + 1); - bitPos += 1; + const subsamplingX = bitstream.readBits(1); + const subsamplingY = bitstream.readBits(1); // 0 = 4:2:0 vertical // 1 = 4:2:0 colocated // 2 = 4:2:2 // 3 = 4:4:4 - chromaSubsampling = (!subsamplingX && !subsamplingY) + chromaSubsampling = !subsamplingX && !subsamplingY ? 3 // 0,0 = 4:4:4 - : (subsamplingX && !subsamplingY) - ? 2 // 1,0 = 4:2:2 - : 1; // 1,1 = 4:2:0 colocated (default) + : subsamplingX && !subsamplingY + ? 2 // 1,0 = 4:2:2 + : 1; // 1,1 = 4:2:0 colocated (default) // Skip reserved bit - bitPos += 1; + bitstream.skipBits(1); } else { // For profile 0 and 2, always 4:2:0 chromaSubsampling = 1; // Using colocated as default @@ -609,10 +608,8 @@ export const extractVp9CodecInfoFromFrame = (frame: Uint8Array): Vp9CodecInfo | } // Parse frame size - const widthMinusOne = readBits(frame, bitPos, bitPos + 16); - bitPos += 16; - const heightMinusOne = readBits(frame, bitPos, bitPos + 16); - bitPos += 16; + const widthMinusOne = bitstream.readBits(16); + const heightMinusOne = bitstream.readBits(16); const width = widthMinusOne + 1; const height = heightMinusOne + 1; @@ -632,15 +629,21 @@ export const extractVp9CodecInfoFromFrame = (frame: Uint8Array): Vp9CodecInfo | ? 0 : colorSpace === 2 ? 1 - : colorSpace === 1 ? 6 : 2; + : colorSpace === 1 + ? 6 + : 2; const colourPrimaries = colorSpace === 2 ? 1 - : colorSpace === 1 ? 6 : 2; + : colorSpace === 1 + ? 6 + : 2; const transferCharacteristics = colorSpace === 2 ? 1 - : colorSpace === 1 ? 6 : 2; + : colorSpace === 1 + ? 6 + : 2; return { profile, @@ -658,16 +661,18 @@ export const extractVp9CodecInfoFromFrame = (frame: Uint8Array): Vp9CodecInfo | * When AV1 codec information is not provided by the container, we can still try to extract the information by digging * into the AV1 bitstream. */ -export const extractAv1CodecInfoFromFrame = (data: Uint8Array): Av1CodecInfo | null => { +export const extractAv1CodecInfoFromFrame = ( + data: Uint8Array, +): Av1CodecInfo | null => { // https://aomediacodec.github.io/av1-spec/av1-spec.pdf - let offset = 0; + const bitstream = new Bitstream(data); - const readLeb128 = () => { + const readLeb128 = (): number | null => { let value = 0; for (let i = 0; i < 8; i++) { - const byte = data[offset++]; + const byte = bitstream.readAlignedByte(); if (byte === undefined) return 0; value |= ((byte & 0x7f) << (i * 7)); @@ -690,10 +695,9 @@ export const extractAv1CodecInfoFromFrame = (data: Uint8Array): Av1CodecInfo | n return value; }; - while (offset < data.length) { + while (bitstream.getBitsLeft() >= 8) { // Parse OBU header - const obuHeader = readBits(data, offset * 8, offset * 8 + 8); - offset += 1; + const obuHeader = bitstream.readBits(8); const obuType = (obuHeader >> 3) & 0xf; const obuExtension = (obuHeader >> 2) & 0x1; @@ -701,7 +705,7 @@ export const extractAv1CodecInfoFromFrame = (data: Uint8Array): Av1CodecInfo | n // Skip extension header if present if (obuExtension) { - offset += 1; + bitstream.skipBits(8); } // Read OBU size if present @@ -711,43 +715,35 @@ export const extractAv1CodecInfoFromFrame = (data: Uint8Array): Av1CodecInfo | n if (obuSizeValue === null) return null; // It was invalid obuSize = obuSizeValue; } else { - obuSize = data.length - offset; + // Calculate remaining bits and convert to bytes, rounding down + obuSize = Math.floor(bitstream.getBitsLeft() / 8); } // We're only interested in Sequence Header OBU (type 1) if (obuType === 1) { - const startBit = offset * 8; - let bitOffset = 0; - // Read sequence header fields - const seqProfile = readBits(data, startBit + bitOffset, startBit + bitOffset + 3); - bitOffset += 3; + const seqProfile = bitstream.readBits(3); // eslint-disable-next-line @typescript-eslint/no-unused-vars - const stillPicture = readBits(data, startBit + bitOffset, startBit + bitOffset + 1); - bitOffset += 1; + const stillPicture = bitstream.readBits(1); - const reducedStillPictureHeader = readBits(data, startBit + bitOffset, startBit + bitOffset + 1); - bitOffset += 1; + const reducedStillPictureHeader = bitstream.readBits(1); let seqLevel = 0; let seqTier = 0; let bufferDelayLengthMinus1 = 0; if (reducedStillPictureHeader) { - seqLevel = readBits(data, startBit + bitOffset, startBit + bitOffset + 5); - bitOffset += 5; + seqLevel = bitstream.readBits(5); } else { // Parse timing_info_present_flag - const timingInfoPresentFlag = readBits(data, startBit + bitOffset, startBit + bitOffset + 1); - bitOffset += 1; + const timingInfoPresentFlag = bitstream.readBits(1); if (timingInfoPresentFlag) { // Skip timing info (num_units_in_display_tick, time_scale, equal_picture_interval) - bitOffset += 32; // num_units_in_display_tick - bitOffset += 32; // time_scale - const equalPictureInterval = readBits(data, startBit + bitOffset, startBit + bitOffset + 1); - bitOffset += 1; + bitstream.skipBits(32); // num_units_in_display_tick + bitstream.skipBits(32); // time_scale + const equalPictureInterval = bitstream.readBits(1); if (equalPictureInterval) { // Skip num_ticks_per_picture_minus_1 (uvlc) @@ -758,30 +754,26 @@ export const extractAv1CodecInfoFromFrame = (data: Uint8Array): Av1CodecInfo | n } // Parse decoder_model_info_present_flag - const decoderModelInfoPresentFlag = readBits(data, startBit + bitOffset, startBit + bitOffset + 1); - bitOffset += 1; + const decoderModelInfoPresentFlag = bitstream.readBits(1); if (decoderModelInfoPresentFlag) { // Store buffer_delay_length_minus_1 instead of just skipping - bufferDelayLengthMinus1 = readBits(data, startBit + bitOffset, startBit + bitOffset + 5); - bitOffset += 5; - bitOffset += 32; // num_units_in_decoding_tick - bitOffset += 5; // buffer_removal_time_length_minus_1 - bitOffset += 5; // frame_presentation_time_length_minus_1 + bufferDelayLengthMinus1 = bitstream.readBits(5); + bitstream.skipBits(32); // num_units_in_decoding_tick + bitstream.skipBits(5); // buffer_removal_time_length_minus_1 + bitstream.skipBits(5); // frame_presentation_time_length_minus_1 } // Parse operating_points_cnt_minus_1 - const operatingPointsCntMinus1 = readBits(data, startBit + bitOffset, startBit + bitOffset + 5); - bitOffset += 5; + const operatingPointsCntMinus1 = bitstream.readBits(5); // For each operating point for (let i = 0; i <= operatingPointsCntMinus1; i++) { // operating_point_idc[i] - bitOffset += 12; + bitstream.skipBits(12); // seq_level_idx[i] - const seqLevelIdx = readBits(data, startBit + bitOffset, startBit + bitOffset + 5); - bitOffset += 5; + const seqLevelIdx = bitstream.readBits(5); if (i === 0) { seqLevel = seqLevelIdx; @@ -789,8 +781,7 @@ export const extractAv1CodecInfoFromFrame = (data: Uint8Array): Av1CodecInfo | n if (seqLevelIdx > 7) { // seq_tier[i] - const seqTierTemp = readBits(data, startBit + bitOffset, startBit + bitOffset + 1); - bitOffset += 1; + const seqTierTemp = bitstream.readBits(1); if (i === 0) { seqTier = seqTierTemp; } @@ -798,37 +789,31 @@ export const extractAv1CodecInfoFromFrame = (data: Uint8Array): Av1CodecInfo | n if (decoderModelInfoPresentFlag) { // decoder_model_present_for_this_op[i] - const decoderModelPresentForThisOp - = readBits(data, startBit + bitOffset, startBit + bitOffset + 1); - bitOffset += 1; + const decoderModelPresentForThisOp = bitstream.readBits(1); if (decoderModelPresentForThisOp) { const n = bufferDelayLengthMinus1 + 1; - bitOffset += n; // decoder_buffer_delay[op] - bitOffset += n; // encoder_buffer_delay[op] - bitOffset += 1; // low_delay_mode_flag[op] + bitstream.skipBits(n); // decoder_buffer_delay[op] + bitstream.skipBits(n); // encoder_buffer_delay[op] + bitstream.skipBits(1); // low_delay_mode_flag[op] } } // initial_display_delay_present_flag - const initialDisplayDelayPresentFlag - = readBits(data, startBit + bitOffset, startBit + bitOffset + 1); - bitOffset += 1; + const initialDisplayDelayPresentFlag = bitstream.readBits(1); if (initialDisplayDelayPresentFlag) { // initial_display_delay_minus_1[i] - bitOffset += 4; + bitstream.skipBits(4); } } } - const highBitdepth = readBits(data, startBit + bitOffset, startBit + bitOffset + 1); - bitOffset += 1; + const highBitdepth = bitstream.readBits(1); let bitDepth = 8; if (seqProfile === 2 && highBitdepth) { - const twelveBit = readBits(data, startBit + bitOffset, startBit + bitOffset + 1); - bitOffset += 1; + const twelveBit = bitstream.readBits(1); bitDepth = twelveBit ? 12 : 10; } else if (seqProfile <= 2) { bitDepth = highBitdepth ? 10 : 8; @@ -836,8 +821,7 @@ export const extractAv1CodecInfoFromFrame = (data: Uint8Array): Av1CodecInfo | n let monochrome = 0; if (seqProfile !== 1) { - monochrome = readBits(data, startBit + bitOffset, startBit + bitOffset + 1); - bitOffset += 1; + monochrome = bitstream.readBits(1); } let chromaSubsamplingX = 1; @@ -853,18 +837,15 @@ export const extractAv1CodecInfoFromFrame = (data: Uint8Array): Av1CodecInfo | n chromaSubsamplingY = 0; } else { if (bitDepth === 12) { - chromaSubsamplingX = readBits(data, startBit + bitOffset, startBit + bitOffset + 1); - bitOffset += 1; + chromaSubsamplingX = bitstream.readBits(1); if (chromaSubsamplingX) { - chromaSubsamplingY = readBits(data, startBit + bitOffset, startBit + bitOffset + 1); - bitOffset += 1; + chromaSubsamplingY = bitstream.readBits(1); } } } if (chromaSubsamplingX && chromaSubsamplingY) { - chromaSamplePosition = readBits(data, startBit + bitOffset, startBit + bitOffset + 2); - bitOffset += 2; + chromaSamplePosition = bitstream.readBits(2); } } @@ -881,7 +862,8 @@ export const extractAv1CodecInfoFromFrame = (data: Uint8Array): Av1CodecInfo | n } // Move to next OBU - offset += obuSize; + // The OBU size is in bytes, so skip that many bytes. + bitstream.skipBits(obuSize * 8); } return null; @@ -958,30 +940,25 @@ export const parseAacAudioSpecificConfig = (bytes: Uint8Array | null) => { throw new TypeError('AAC description must be at least 2 bytes long.'); } - let bitOffset = 0; + const bitstream = new Bitstream(bytes); // 1) Audio Object Type (5 bits) - let objectType = readBits(bytes, bitOffset, bitOffset + 5); - bitOffset += 5; + let objectType = bitstream.readBits(5); if (objectType === 31) { // (Escape value) -> 6 bits + 32 - objectType = 32 + readBits(bytes, bitOffset, bitOffset + 6); - bitOffset += 6; + objectType = 32 + bitstream.readBits(6); } // 2) Sampling Frequency Index (4 bits) - const frequencyIndex = readBits(bytes, bitOffset, bitOffset + 4); - bitOffset += 4; + const frequencyIndex = bitstream.readBits(4); let sampleRate: number | null = null; if (frequencyIndex === 15) { // Explicit sample rate (24 bits) - sampleRate = readBits(bytes, bitOffset, bitOffset + 24); - bitOffset += 24; + sampleRate = bitstream.readBits(24); } else { const freqTable = [ - 96000, 88200, 64000, 48000, 44100, - 32000, 24000, 22050, 16000, 12000, - 11025, 8000, 7350, + 96000, 88200, 64000, 48000, 44100, 32000, 24000, 22050, + 16000, 12000, 11025, 8000, 7350, ]; if (frequencyIndex < freqTable.length) { sampleRate = freqTable[frequencyIndex]!; @@ -989,13 +966,16 @@ export const parseAacAudioSpecificConfig = (bytes: Uint8Array | null) => { } // 3) Channel Configuration (4 bits) - const channelConfiguration = readBits(bytes, bitOffset, bitOffset + 4); - bitOffset += 4; + const channelConfiguration = bitstream.readBits(4); let numberOfChannels: number | null = null; if (channelConfiguration >= 1 && channelConfiguration <= 7) { const channelMap = { - 1: 1, 2: 2, 3: 3, - 4: 4, 5: 5, 6: 6, + 1: 1, + 2: 2, + 3: 3, + 4: 4, + 5: 5, + 6: 6, 7: 8, }; numberOfChannels = channelMap[channelConfiguration as keyof typeof channelMap]; @@ -1374,12 +1354,9 @@ export const validateVideoChunkMetadata = (metadata: EncodedVideoChunkMetadata | ); } - if (!metadata.decoderConfig.description) { - throw new TypeError( - 'Video chunk metadata decoder configuration for AVC must include a description, which is expected to be' - + ' an AVCDecoderConfigurationRecord as specified in ISO 14496-15.', - ); - } + // `description` may or may not be set, depending on if the format is AVCC or Annex B, so don't perform any + // validation for it. + // https://www.w3.org/TR/webcodecs-avc-codec-registration } else if (metadata.decoderConfig.codec.startsWith('hev1') || metadata.decoderConfig.codec.startsWith('hvc1')) { // HEVC-specific validation diff --git a/src/input-format.ts b/src/input-format.ts index cc7e8aa..391b9d0 100644 --- a/src/input-format.ts +++ b/src/input-format.ts @@ -121,7 +121,7 @@ export class MatroskaInputFormat extends InputFormat { } const dataSize = ebmlReader.readElementSize(); - if (dataSize === -1) { + if (dataSize === null) { return false; // Miss me with that shit } @@ -129,7 +129,7 @@ export class MatroskaInputFormat extends InputFormat { while (ebmlReader.pos < startPos + dataSize) { const { id, size } = ebmlReader.readElementHeader(); const dataStartPos = ebmlReader.pos; - if (size === -1) return false; + if (size === null) return false; switch (id) { case EBMLId.EBMLVersion: { diff --git a/src/isobmff/isobmff-boxes.ts b/src/isobmff/isobmff-boxes.ts index c18a210..4331d20 100644 --- a/src/isobmff/isobmff-boxes.ts +++ b/src/isobmff/isobmff-boxes.ts @@ -931,8 +931,7 @@ export const stsc = (trackData: IsobmffTrackData) => { /** Sample Size Box: Specifies the byte size of each sample in the media. */ export const stsz = (trackData: IsobmffTrackData) => { - if (trackData.requiresPcmTransformation) { - assert(trackData.type === 'audio'); + if (trackData.type === 'audio' && trackData.info.requiresPcmTransformation) { const { sampleSize } = parsePcmCodec(trackData.track.source._codec as PcmAudioCodec); // With PCM, every sample has the same size diff --git a/src/isobmff/isobmff-demuxer.ts b/src/isobmff/isobmff-demuxer.ts index 87e2573..b961eed 100644 --- a/src/isobmff/isobmff-demuxer.ts +++ b/src/isobmff/isobmff-demuxer.ts @@ -1,3 +1,4 @@ +import { AvcDecoderConfigurationRecord } from '../avc'; import { AacCodecInfo, AudioCodec, @@ -73,6 +74,7 @@ type InternalTrack = { codec: VideoCodec | null; codecDescription: Uint8Array | null; colorSpace: VideoColorSpaceInit | null; + avcCodecInfo: AvcDecoderConfigurationRecord | null; vp9CodecInfo: Vp9CodecInfo | null; av1CodecInfo: Av1CodecInfo | null; }; @@ -445,7 +447,8 @@ export class IsobmffDemuxer extends Demuxer { const moofBoxInfo = this.metadataReader.readBoxHeader(); assert(moofBoxInfo.name === 'moof'); - await this.metadataReader.reader.loadRange(startPos, startPos + moofBoxInfo.totalSize); + const contentStart = this.metadataReader.pos; + await this.metadataReader.reader.loadRange(contentStart, contentStart + moofBoxInfo.contentSize); this.metadataReader.pos = startPos; this.traverseBox(); @@ -456,7 +459,9 @@ export class IsobmffDemuxer extends Demuxer { const fragment = this.fragments[index]!; assert(fragment.moofOffset === startPos); - this.metadataReader.reader.forgetRange(startPos, startPos + moofBoxInfo.totalSize); + // We have read everything in the moof box, there's no need to keep the data around anymore + // (keep the header tho) + this.metadataReader.reader.forgetRange(contentStart, contentStart + moofBoxInfo.contentSize); // It may be that some tracks don't define the base decode time, i.e. when the fragment begins. This means the // only other option is to sum up the duration of all previous fragments. @@ -756,6 +761,7 @@ export class IsobmffDemuxer extends Demuxer { codec: null, codecDescription: null, colorSpace: null, + avcCodecInfo: null, vp9CodecInfo: null, av1CodecInfo: null, }; diff --git a/src/isobmff/isobmff-muxer.ts b/src/isobmff/isobmff-muxer.ts index b5c76fc..5a3c9a7 100644 --- a/src/isobmff/isobmff-muxer.ts +++ b/src/isobmff/isobmff-muxer.ts @@ -15,6 +15,11 @@ import { } from '../codec'; import { BufferTarget } from '../target'; import { EncodedPacket, PacketType } from '../packet'; +import { + extractAvcDecoderConfigurationRecord, + serializeAvcDecoderConfigurationRecord, + transformAnnexBToAvcc, +} from '../avc'; export const GLOBAL_TIMESCALE = 1000; const TIMESTAMP_OFFSET = 2_082_844_800; // Seconds between Jan 1 1904 and Jan 1 1970 @@ -47,11 +52,6 @@ export type IsobmffTrackData = { compositionTimeOffsetTable: { sampleCount: number; sampleCompositionTimeOffset: number }[]; lastTimescaleUnits: number | null; lastSample: Sample | null; - /** - * The "PCM transformation" is making every sample in the sample table be exactly one PCM audio sample long. - * Some players expect this for PCM audio. - */ - requiresPcmTransformation: boolean; finalizedChunks: Chunk[]; currentChunk: Chunk | null; @@ -66,6 +66,11 @@ export type IsobmffTrackData = { width: number; height: number; decoderConfig: VideoDecoderConfig; + /** + * The "AVC transformation" involves converting the raw AVC packet data from Annex B to AVCC format. + * https://stackoverflow.com/questions/24884827 + */ + requiresAvcTransformation: boolean; }; } | { track: OutputAudioTrack; @@ -74,6 +79,11 @@ export type IsobmffTrackData = { numberOfChannels: number; sampleRate: number; decoderConfig: AudioDecoderConfig; + /** + * The "PCM transformation" is making every sample in the sample table be exactly one PCM audio sample long. + * Some players expect this for PCM audio. + */ + requiresPcmTransformation: boolean; }; } | { track: OutputSubtitleTrack; @@ -183,7 +193,7 @@ export class IsobmffMuxer extends Muxer { release(); } - private getVideoTrackData(track: OutputVideoTrack, meta?: EncodedVideoChunkMetadata) { + private getVideoTrackData(track: OutputVideoTrack, packet: EncodedPacket, meta?: EncodedVideoChunkMetadata) { const existingTrackData = this.trackDatas.find(x => x.track === track); if (existingTrackData) { return existingTrackData as IsobmffVideoTrackData; @@ -193,8 +203,30 @@ export class IsobmffMuxer extends Muxer { assert(meta); assert(meta.decoderConfig); - assert(meta.decoderConfig.codedWidth !== undefined); - assert(meta.decoderConfig.codedHeight !== undefined); + + const decoderConfig = { ...meta.decoderConfig }; + assert(decoderConfig.codedWidth !== undefined); + assert(decoderConfig.codedHeight !== undefined); + + let requiresAvcTransformation = false; + + if (track.source._codec === 'avc' && !decoderConfig.description) { + // ISOBMFF can only hold AVC in the AVCC format, not in Annex B, but the missing description indicates + // Annex B. This means we'll need to do some converterino. + + const decoderConfigurationRecord = extractAvcDecoderConfigurationRecord(packet.data); + if (!decoderConfigurationRecord) { + throw new Error( + 'Couldn\'t extract an AVCDecoderConfigurationRecord from the AVC packet. Make sure the packets are' + + ' in Annex B format (as specified in ITU-T-REC-H.264) when not providing a description, or' + + ' provide a description (must be an AVCDecoderConfigurationRecord as specified in ISO 14496-15)' + + ' and ensure the packets are in AVCC format.', + ); + } + + decoderConfig.description = serializeAvcDecoderConfigurationRecord(decoderConfigurationRecord); + requiresAvcTransformation = true; + } // The frame rate set by the user may not be an integer. Since timescale is an integer, we'll approximate the // frame time (inverse of frame rate) with a rational number, then use that approximation's denominator @@ -205,9 +237,10 @@ export class IsobmffMuxer extends Muxer { track, type: 'video', info: { - width: meta.decoderConfig.codedWidth, - height: meta.decoderConfig.codedHeight, - decoderConfig: meta.decoderConfig, + width: decoderConfig.codedWidth, + height: decoderConfig.codedHeight, + decoderConfig: decoderConfig, + requiresAvcTransformation, }, timescale, samples: [], @@ -220,7 +253,6 @@ export class IsobmffMuxer extends Muxer { finalizedChunks: [], currentChunk: null, compactlyCodedChunkTable: [], - requiresPcmTransformation: false, }; this.trackDatas.push(newTrackData); @@ -247,6 +279,9 @@ export class IsobmffMuxer extends Muxer { numberOfChannels: meta.decoderConfig.numberOfChannels, sampleRate: meta.decoderConfig.sampleRate, decoderConfig: meta.decoderConfig, + requiresPcmTransformation: + !this.isFragmented + && (PCM_AUDIO_CODECS as readonly string[]).includes(track.source._codec), }, timescale: meta.decoderConfig.sampleRate, samples: [], @@ -259,9 +294,6 @@ export class IsobmffMuxer extends Muxer { finalizedChunks: [], currentChunk: null, compactlyCodedChunkTable: [], - requiresPcmTransformation: - !this.isFragmented - && (PCM_AUDIO_CODECS as readonly string[]).includes(track.source._codec), }; this.trackDatas.push(newTrackData); @@ -298,7 +330,6 @@ export class IsobmffMuxer extends Muxer { finalizedChunks: [], currentChunk: null, compactlyCodedChunkTable: [], - requiresPcmTransformation: false, lastCueEndTimestamp: 0, cueQueue: [], @@ -316,7 +347,20 @@ export class IsobmffMuxer extends Muxer { const release = await this.mutex.acquire(); try { - const trackData = this.getVideoTrackData(track, meta); + const trackData = this.getVideoTrackData(track, packet, meta); + + let packetData = packet.data; + if (trackData.info.requiresAvcTransformation) { + const transformedData = transformAnnexBToAvcc(packetData); + if (!transformedData) { + throw new Error( + 'Failed to transform packet data. Make sure all packets are provided in Annex B format, as' + + ' specified in ITU-T-REC-H.264.', + ); + } + + packetData = transformedData; + } const timestamp = this.validateAndNormalizeTimestamp( trackData.track, @@ -325,7 +369,7 @@ export class IsobmffMuxer extends Muxer { ); const internalSample = this.createSampleForTrack( trackData, - packet.data, + packetData, timestamp, packet.duration, packet.type, @@ -356,7 +400,7 @@ export class IsobmffMuxer extends Muxer { packet.type, ); - if (trackData.requiresPcmTransformation) { + if (trackData.info.requiresPcmTransformation) { await this.maybePadWithSilence(trackData, timestamp); } @@ -534,7 +578,7 @@ export class IsobmffMuxer extends Muxer { return; } - if (trackData.requiresPcmTransformation) { + if (trackData.type === 'audio' && trackData.info.requiresPcmTransformation) { let totalDuration = 0; // Compute the total duration in the track timescale (which is equal to the amount of PCM audio samples) @@ -748,7 +792,7 @@ export class IsobmffMuxer extends Muxer { this.finalizedChunks.push(trackData.currentChunk); let sampleCount = trackData.currentChunk.samples.length; - if (trackData.requiresPcmTransformation) { + if (trackData.type === 'audio' && trackData.info.requiresPcmTransformation) { sampleCount = trackData.currentChunk.samples .reduce((acc, sample) => acc + intoTimescale(sample.duration, trackData.timescale), 0); } diff --git a/src/matroska/ebml.ts b/src/matroska/ebml.ts index 6b95f56..290e1ab 100644 --- a/src/matroska/ebml.ts +++ b/src/matroska/ebml.ts @@ -104,8 +104,34 @@ export enum EBMLId { Projection = 0x7670, ProjectionType = 0x7671, ProjectionPoseRoll = 0x7675, + Attachments = 0x1941a469, + Chapters = 0x1043a770, + Tags = 0x1254c367, } +export const LEVEL_0_EBML_IDS: EBMLId[] = [ + EBMLId.EBML, + EBMLId.Segment, +]; + +export const LEVEL_1_EBML_IDS: EBMLId[] = [ + EBMLId.EBMLMaxIDLength, + EBMLId.EBMLMaxSizeLength, + EBMLId.SeekHead, + EBMLId.Info, + EBMLId.Cluster, + EBMLId.Tracks, + EBMLId.Cues, + EBMLId.Attachments, + EBMLId.Chapters, + EBMLId.Tags, +]; + +export const LEVEL_0_AND_1_EBML_IDS = [ + ...LEVEL_0_EBML_IDS, + ...LEVEL_1_EBML_IDS, +]; + export const measureUnsignedInt = (value: number) => { if (value < (1 << 8)) { return 1; @@ -476,13 +502,22 @@ export class EBMLReader { } readElementSize() { - let size = this.readU8(); + let size: number | null = this.readU8(); if (size === 0xff) { - size = -1; + size = null; } else { this.pos--; size = this.readVarInt(); + + // In some (livestreamed) files, this is the value of the size field. While this technically is just a very + // large number, it is intended to behave like the reserved size 0xFF, meaning the size is undefined. We + // catch the number here. Note that it cannot be perfectly represented as a double, but the comparison works + // nonetheless. + // eslint-disable-next-line no-loss-of-precision + if (size === 0x00ffffffffffffff) { + size = null; + } } return size; @@ -494,6 +529,30 @@ export class EBMLReader { return { id, size }; } + + /** Returns the byte offset in the file of the next element with a matching ID. */ + async searchForNextElementId(ids: EBMLId[], until: number) { + const loadChunkSize = 2 ** 20; // 1 MiB + const idsSet = new Set(ids); + + while (this.pos < until - MAX_HEADER_SIZE) { + if (!this.reader.rangeIsLoaded(this.pos, this.pos + MAX_HEADER_SIZE)) { + await this.reader.loadRange(this.pos, Math.min(this.pos + loadChunkSize, until)); + } + + const elementStartPos = this.pos; + const elementHeader = this.readElementHeader(); + if (idsSet.has(elementHeader.id)) { + return elementStartPos; + } + + assertDefinedSize(elementHeader.size); + + this.pos += elementHeader.size; + } + + return null; + } } export const CODEC_STRING_MAP: Partial> = { @@ -519,3 +578,9 @@ export const CODEC_STRING_MAP: Partial> = { 'webvtt': 'S_TEXT/WEBVTT', }; + +export function assertDefinedSize(size: number | null): asserts size is number { + if (size === null) { + throw new Error('Undefined element size is used in a place where it is not supported.'); + } +}; diff --git a/src/matroska/matroska-demuxer.ts b/src/matroska/matroska-demuxer.ts index d7be74f..095b351 100644 --- a/src/matroska/matroska-demuxer.ts +++ b/src/matroska/matroska-demuxer.ts @@ -1,3 +1,4 @@ +import { extractAvcDecoderConfigurationRecord } from '../avc'; import { AacCodecInfo, AudioCodec, @@ -37,7 +38,15 @@ import { } from '../misc'; import { EncodedPacket, PLACEHOLDER_DATA } from '../packet'; import { Reader } from '../reader'; -import { CODEC_STRING_MAP, EBMLId, EBMLReader, MAX_HEADER_SIZE, MIN_HEADER_SIZE } from './ebml'; +import { + assertDefinedSize, + CODEC_STRING_MAP, + EBMLId, + EBMLReader, + LEVEL_0_AND_1_EBML_IDS, + MAX_HEADER_SIZE, + MIN_HEADER_SIZE, +} from './ebml'; type Segment = { seekHeadSeen: boolean; @@ -201,12 +210,6 @@ export class MatroskaDemuxer extends Demuxer { return string; } - assertDefinedSize(size: number) { - if (size === -1) { - throw new Error('Undefined element size is used in a place where it is not supported.'); - } - } - readMetadata() { return this.readMetadataPromise ??= (async () => { this.metadataReader.pos = 0; @@ -223,24 +226,27 @@ export class MatroskaDemuxer extends Demuxer { const startPos = this.metadataReader.pos; if (id === EBMLId.EBML) { + assertDefinedSize(size); + await this.metadataReader.reader.loadRange(this.metadataReader.pos, this.metadataReader.pos + size); this.readContiguousElements(this.metadataReader, size); } else if (id === EBMLId.Segment) { // Segment found! await this.readSegment(size); - if (size === -1) { - // Stop searching for other segments if this is the case + if (size === null) { + // Segment sizes can be undefined (common in livestreamed files), so assume this is the last + // and only segment break; } } - this.assertDefinedSize(size); + assertDefinedSize(size); this.metadataReader.pos = startPos + size; } })(); } - async readSegment(dataSize: number) { + async readSegment(dataSize: number | null) { const segmentDataStart = this.metadataReader.pos; this.currentSegment = { @@ -257,8 +263,8 @@ export class MatroskaDemuxer extends Demuxer { cuePoints: [], dataStartPos: segmentDataStart, - elementEndPos: dataSize === -1 - ? (await this.input.source.getSize() - MIN_HEADER_SIZE) + elementEndPos: dataSize === null + ? await this.input.source.getSize() // Assume it goes until the end of the file : segmentDataStart + dataSize, clusterSeekStartPos: segmentDataStart, @@ -289,6 +295,7 @@ export class MatroskaDemuxer extends Demuxer { const field = METADATA_ELEMENTS[metadataElementIndex]!.flag; this.currentSegment[field] = true; + assertDefinedSize(size); await this.metadataReader.reader.loadRange(this.metadataReader.pos, this.metadataReader.pos + size); this.readContiguousElements(this.metadataReader, size); } else if (id === EBMLId.Cluster) { @@ -324,7 +331,10 @@ export class MatroskaDemuxer extends Demuxer { } } - this.assertDefinedSize(size); + if (size === null) { + break; + } + this.metadataReader.pos = dataStartPos + size; if (!clusterEncountered) { @@ -347,6 +357,8 @@ export class MatroskaDemuxer extends Demuxer { const { id, size } = this.metadataReader.readElementHeader(); if (id !== target.id) continue; + assertDefinedSize(size); + this.currentSegment[target.flag] = true; await this.metadataReader.reader.loadRange(this.metadataReader.pos, this.metadataReader.pos + size); this.readContiguousElements(this.metadataReader, size); @@ -411,9 +423,24 @@ export class MatroskaDemuxer extends Demuxer { await this.metadataReader.reader.loadRange(this.metadataReader.pos, this.metadataReader.pos + MAX_HEADER_SIZE); const elementStartPos = this.metadataReader.pos; - const { id, size } = this.metadataReader.readElementHeader(); + const elementHeader = this.metadataReader.readElementHeader(); + const id = elementHeader.id; + let size = elementHeader.size; const dataStartPos = this.metadataReader.pos; + if (size === null) { + // The cluster's size is undefined (can happen in livestreamed files). We'd still like to know the size of + // it, so we have no other choice but to iterate over the EBML structure until we find an element at level + // 0 or 1, indicating the end of the cluster (all elements inside the cluster are at level 2). + this.clusterReader.pos = dataStartPos; + const nextElementPos = await this.clusterReader.searchForNextElementId( + LEVEL_0_AND_1_EBML_IDS, + segment.elementEndPos, + ); + + size = (nextElementPos ?? segment.elementEndPos) - dataStartPos; + } + assert(id === EBMLId.Cluster); // Load the entire cluster @@ -539,6 +566,7 @@ export class MatroskaDemuxer extends Demuxer { traverseElement(reader: EBMLReader) { const { id, size } = reader.readElementHeader(); const dataStartPos = reader.pos; + assertDefinedSize(size); switch (id) { case EBMLId.DocType: { @@ -981,7 +1009,6 @@ export class MatroskaDemuxer extends Demuxer { }; break; } - this.assertDefinedSize(size); reader.pos = dataStartPos + size; } } @@ -1329,6 +1356,7 @@ abstract class MatroskaTrackBacking implements InputTrackBacking { // We use the metadata reader to find the cluster, but the cluster reader to load the cluster const metadataReader = demuxer.metadataReader; + const clusterReader = demuxer.clusterReader; let prevCluster: Cluster | null = null; let bestClusterIndex = clusterIndex; @@ -1379,7 +1407,9 @@ abstract class MatroskaTrackBacking implements InputTrackBacking { // Load the header await metadataReader.reader.loadRange(metadataReader.pos, metadataReader.pos + MAX_HEADER_SIZE); const elementStartPos = metadataReader.pos; - const { id, size } = metadataReader.readElementHeader(); + const elementHeader = metadataReader.readElementHeader(); + const id = elementHeader.id; + let size = elementHeader.size; const dataStartPos = metadataReader.pos; if (id === EBMLId.Cluster) { @@ -1415,6 +1445,42 @@ abstract class MatroskaTrackBacking implements InputTrackBacking { } } + if (size === null) { + // Undefined element size (can happen in livestreamed files). In this case, we need to do some + // searching to determine the actual size of the element. + + if (id === EBMLId.Cluster) { + // The cluster should have already computed its length, we can just copy that result + assert(prevCluster); + size = prevCluster.elementEndPos - dataStartPos; + } else { + // Search for the next element at level 0 or 1 + clusterReader.pos = dataStartPos; + const nextElementPos = await clusterReader.searchForNextElementId( + LEVEL_0_AND_1_EBML_IDS, + segment.elementEndPos, + ); + + size = (nextElementPos ?? segment.elementEndPos) - dataStartPos; + } + + const endPos = dataStartPos + size; + if (endPos >= segment.elementEndPos - MIN_HEADER_SIZE) { + // No more elements fit in this segment + break; + } else { + // Check the next element. If it's a new segment, we know this segment ends here. The new + // segment is just ignored, since we're likely in a livestreamed file and thus only care about + // the first segment. + clusterReader.pos = endPos; + const elementId = clusterReader.readElementId(); + if (elementId === EBMLId.Segment) { + segment.elementEndPos = endPos; + break; + } + } + } + metadataReader.pos = dataStartPos + size; } @@ -1483,7 +1549,10 @@ class MatroskaVideoTrackBacking extends MatroskaTrackBacking implements InputVid return this.decoderConfigPromise ??= (async (): Promise => { let firstPacket: EncodedPacket | null = null; const needsPacketForAdditionalInfo - = this.internalTrack.info.codec === 'vp9' || this.internalTrack.info.codec === 'av1'; + = this.internalTrack.info.codec === 'vp9' + || this.internalTrack.info.codec === 'av1' + // Packets are in Annex B format: + || (this.internalTrack.info.codec === 'avc' && !this.internalTrack.info.codecDescription); if (needsPacketForAdditionalInfo) { firstPacket = await this.getFirstPacket({}); @@ -1496,6 +1565,9 @@ class MatroskaVideoTrackBacking extends MatroskaTrackBacking implements InputVid codec: this.internalTrack.info.codec, codecDescription: this.internalTrack.info.codecDescription, colorSpace: this.internalTrack.info.colorSpace, + avcCodecInfo: this.internalTrack.info.codec === 'avc' && firstPacket + ? extractAvcDecoderConfigurationRecord(firstPacket.data) + : null, vp9CodecInfo: this.internalTrack.info.codec === 'vp9' && firstPacket ? extractVp9CodecInfoFromFrame(firstPacket.data) : null, diff --git a/src/matroska/matroska-muxer.ts b/src/matroska/matroska-muxer.ts index 60d231e..fbe0e17 100644 --- a/src/matroska/matroska-muxer.ts +++ b/src/matroska/matroska-muxer.ts @@ -1,4 +1,5 @@ import { + Bitstream, COLOR_PRIMARIES_MAP, MATRIX_COEFFICIENTS_MAP, TRANSFER_CHARACTERISTICS_MAP, @@ -6,7 +7,6 @@ import { assert, colorSpaceIsComplete, normalizeRotation, - readBits, roundToMultiple, textEncoder, toUint8Array, @@ -623,37 +623,42 @@ export class MatroskaMuxer extends Muxer { } } - /** Due to [a bug in Chromium](https://bugs.chromium.org/p/chromium/issues/detail?id=1377842), VP9 streams often - * lack color space information. This method patches in that information. */ - // http://downloads.webmproject.org/docs/vp9/vp9-bitstream_superframe-and-uncompressed-header_v1.0.pdf - private fixVP9ColorSpace(trackData: MatroskaVideoTrackData, chunk: InternalMediaChunk) { + /** + * Due to [a bug in Chromium](https://bugs.chromium.org/p/chromium/issues/detail?id=1377842), VP9 streams often + * lack color space information. This method patches in that information. + */ + private fixVP9ColorSpace( + trackData: MatroskaVideoTrackData, + chunk: InternalMediaChunk, + ) { + // http://downloads.webmproject.org/docs/vp9/vp9-bitstream_superframe-and-uncompressed-header_v1.0.pdf + if (chunk.type !== 'key') return; if (!trackData.info.decoderConfig.colorSpace || !trackData.info.decoderConfig.colorSpace.matrix) return; - let i = 0; + const bitstream = new Bitstream(chunk.data); + // Check if it's a "superframe" - if (readBits(chunk.data, 0, 2) !== 0b10) return; - i += 2; + if (bitstream.readBits(2) !== 0b10) return; - const profile = (readBits(chunk.data, i + 1, i + 2) << 1) + readBits(chunk.data, i + 0, i + 1); - i += 2; - if (profile === 3) i++; + const profileLowBit = bitstream.readBits(1); + const profileHighBit = bitstream.readBits(1); + const profile = (profileHighBit << 1) + profileLowBit; - const showExistingFrame = readBits(chunk.data, i + 0, i + 1); - i++; + if (profile === 3) bitstream.skipBits(1); + + const showExistingFrame = bitstream.readBits(1); if (showExistingFrame) return; - const frameType = readBits(chunk.data, i + 0, i + 1); - i++; + const frameType = bitstream.readBits(1); if (frameType !== 0) return; // Just to be sure - i += 2; + bitstream.skipBits(2); - const syncCode = readBits(chunk.data, i + 0, i + 24); - i += 24; + const syncCode = bitstream.readBits(24); if (syncCode !== 0x498342) return; - if (profile >= 2) i++; + if (profile >= 2) bitstream.skipBits(1); const colorSpaceID = { rgb: 7, @@ -661,7 +666,10 @@ export class MatroskaMuxer extends Muxer { bt470bg: 1, smpte170m: 3, }[trackData.info.decoderConfig.colorSpace.matrix]; - writeBits(chunk.data, i + 0, i + 3, colorSpaceID); + + // The bitstream position is now at the start of the color space bits. + // We can use the global writeBits function here as requested. + writeBits(chunk.data, bitstream.pos, bitstream.pos + 3, colorSpaceID); } /** Converts a read-only external chunk into an internal one for easier use. */ diff --git a/src/misc.ts b/src/misc.ts index 44d9814..6701e31 100644 --- a/src/misc.ts +++ b/src/misc.ts @@ -30,19 +30,77 @@ export const isU32 = (value: number) => { return value >= 0 && value < 2 ** 32; }; -export const readBits = (bytes: Uint8Array, start: number, end: number) => { - let result = 0; +export class Bitstream { + /** Current offset in bits. */ + pos = 0; - for (let i = start; i < end; i++) { - const byteIndex = Math.floor(i / 8); - const byte = bytes[byteIndex]!; - const bitIndex = 0b111 - (i & 0b111); - const bit = (byte & (1 << bitIndex)) >> bitIndex; + constructor(public bytes: Uint8Array) {} - result <<= 1; - result |= bit; + seekToByte(byteOffset: number) { + this.pos = 8 * byteOffset; } + readBit() { + const byteIndex = Math.floor(this.pos / 8); + const byte = this.bytes[byteIndex] ?? 0; + const bitIndex = 0b111 - (this.pos & 0b111); + const bit = (byte & (1 << bitIndex)) >> bitIndex; + + this.pos++; + return bit; + } + + readBits(n: number) { + let result = 0; + + for (let i = 0; i < n; i++) { + result <<= 1; + result |= this.readBit(); + } + + return result; + } + + readAlignedByte() { + // Ensure we're byte-aligned + if (this.pos % 8 !== 0) { + throw new Error('Bitstream is not byte-aligned.'); + } + + const byteIndex = this.pos / 8; + const byte = this.bytes[byteIndex] ?? 0; + + this.pos += 8; + return byte; + } + + skipBits(n: number) { + this.pos += n; + } + + getBitsLeft() { + return this.bytes.length * 8 - this.pos; + } + + clone() { + const clone = new Bitstream(this.bytes); + clone.pos = this.pos; + return clone; + } +} + +/** Reads an exponential-Golomb universal code from a Bitstream. */ +export const readExpGolomb = (bitstream: Bitstream) => { + let leadingZeroBits = 0; + while (bitstream.readBit() === 0 && leadingZeroBits < 32) { + leadingZeroBits++; + } + + if (leadingZeroBits >= 32) { + throw new Error('Invalid exponential-Golomb code.'); + } + + const result = (1 << leadingZeroBits) - 1 + bitstream.readBits(leadingZeroBits); return result; }; diff --git a/src/ogg/ogg-misc.ts b/src/ogg/ogg-misc.ts index d8c18ff..fe3839c 100644 --- a/src/ogg/ogg-misc.ts +++ b/src/ogg/ogg-misc.ts @@ -1,5 +1,5 @@ import { parseOpusTocByte } from '../codec'; -import { assert, ilog, readBits, toDataView } from '../misc'; +import { assert, Bitstream, ilog, toDataView } from '../misc'; export const OGGS = 0x5367674f; // 'OggS' @@ -96,40 +96,6 @@ export const extractSampleMetadata = ( // Based on vorbis_parser.c from FFmpeg. export const parseModesFromVorbisSetupPacket = (setupHeader: Uint8Array) => { - class Bitstream { - bytes: Uint8Array; - pos: number; - - constructor(bytes: Uint8Array) { - this.bytes = bytes; - this.pos = 0; - } - - read(n: number): number { - const result = readBits(this.bytes, this.pos, this.pos + n); - this.pos += n; - return result; - } - - skip(n: number): void { - this.pos += n; - } - - getBitsLeft(): number { - return this.bytes.length * 8 - this.pos; - } - - getBitCount(): number { - return this.pos; - } - - clone(): Bitstream { - const clone = new Bitstream(this.bytes); - clone.pos = this.pos; - return clone; - } - } - // Verify that this is a Setup header. if (setupHeader.length < 7) { throw new Error('Setup header is too short.'); @@ -150,14 +116,14 @@ export const parseModesFromVorbisSetupPacket = (setupHeader: Uint8Array) => { } // Initialize a Bitstream on the reversed buffer. - const bs = new Bitstream(revBuffer); + const bitstream = new Bitstream(revBuffer); // --- Find the framing bit. // In FFmpeg code, we scan until get_bits1() returns 1. let gotFramingBit = 0; - while (bs.getBitsLeft() > 97) { - if (bs.read(1) === 1) { - gotFramingBit = bs.getBitCount(); + while (bitstream.getBitsLeft() > 97) { + if (bitstream.readBits(1) === 1) { + gotFramingBit = bitstream.pos; break; } } @@ -170,23 +136,23 @@ export const parseModesFromVorbisSetupPacket = (setupHeader: Uint8Array) => { let modeCount = 0; let gotModeHeader = false; let lastModeCount = 0; - while (bs.getBitsLeft() >= 97) { - const tempPos = bs.pos; - const a = bs.read(8); - const b = bs.read(16); - const c = bs.read(16); + while (bitstream.getBitsLeft() >= 97) { + const tempPos = bitstream.pos; + const a = bitstream.readBits(8); + const b = bitstream.readBits(16); + const c = bitstream.readBits(16); // If a > 63 or b or c nonzero, assume we’ve gone too far. if (a > 63 || b !== 0 || c !== 0) { - bs.pos = tempPos; + bitstream.pos = tempPos; break; } - bs.skip(1); + bitstream.skipBits(1); modeCount++; if (modeCount > 64) { break; } - const bsClone = bs.clone(); - const candidate = bsClone.read(6) + 1; + const bsClone = bitstream.clone(); + const candidate = bsClone.readBits(6) + 1; if (candidate === modeCount) { gotModeHeader = true; lastModeCount = modeCount; @@ -201,16 +167,16 @@ export const parseModesFromVorbisSetupPacket = (setupHeader: Uint8Array) => { const finalModeCount = lastModeCount; // --- Reinitialize the bitstream. - bs.pos = 0; + bitstream.pos = 0; // Skip the bits up to the found framing bit. - bs.skip(gotFramingBit); + bitstream.skipBits(gotFramingBit); // --- Now read, for each mode (in reverse order), 40 bits then one bit. // That one bit is the mode blockflag. const modeBlockflags = Array(finalModeCount).fill(0) as number[]; for (let i = finalModeCount - 1; i >= 0; i--) { - bs.skip(40); - modeBlockflags[i] = bs.read(1); + bitstream.skipBits(40); + modeBlockflags[i] = bitstream.readBits(1); } return { modeBlockflags }; diff --git a/src/output-format.ts b/src/output-format.ts index f4676d8..ec202d8 100644 --- a/src/output-format.ts +++ b/src/output-format.ts @@ -52,6 +52,8 @@ export abstract class OutputFormat { /** The file extension used by this output format, beginning with a dot. */ abstract get fileExtension(): string; + /** The base MIME type of the output format. */ + abstract get mimeType(): string; /** Returns a list of media codecs that this output format can contain. */ abstract getSupportedCodecs(): MediaCodec[]; /** Returns the number of tracks that this output format supports. */ @@ -227,6 +229,10 @@ export class Mp4OutputFormat extends IsobmffOutputFormat { return '.mp4'; } + get mimeType() { + return 'video/mp4'; + } + getSupportedCodecs(): MediaCodec[] { return [ ...VIDEO_CODECS, @@ -259,6 +265,10 @@ export class MovOutputFormat extends IsobmffOutputFormat { return '.mov'; } + get mimeType() { + return 'video/quicktime'; + } + getSupportedCodecs(): MediaCodec[] { return [ ...VIDEO_CODECS, @@ -380,6 +390,10 @@ export class MkvOutputFormat extends OutputFormat { return '.mkv'; } + get mimeType() { + return 'video/x-matroska'; + } + getSupportedCodecs(): MediaCodec[] { return [ ...VIDEO_CODECS, @@ -423,6 +437,10 @@ export class WebMOutputFormat extends MkvOutputFormat { return '.webm'; } + override get mimeType() { + return 'video/webm'; + } + /** @internal */ override _codecUnsupportedHint(codec: MediaCodec) { if (new MkvOutputFormat().getSupportedCodecs().includes(codec)) { @@ -491,6 +509,10 @@ export class Mp3OutputFormat extends OutputFormat { return '.mp3'; } + get mimeType() { + return 'audio/mpeg'; + } + getSupportedCodecs(): MediaCodec[] { return ['mp3']; } @@ -556,6 +578,10 @@ export class WavOutputFormat extends OutputFormat { return '.wav'; } + get mimeType() { + return 'audio/wav'; + } + getSupportedCodecs(): MediaCodec[] { return [ ...PCM_AUDIO_CODECS.filter(codec => @@ -628,6 +654,10 @@ export class OggOutputFormat extends OutputFormat { return '.ogg'; } + get mimeType() { + return 'application/ogg'; + } + getSupportedCodecs(): MediaCodec[] { return [ ...AUDIO_CODECS.filter(codec => ['vorbis', 'opus'].includes(codec)), diff --git a/src/reader.ts b/src/reader.ts index 4f720a6..809a21a 100644 --- a/src/reader.ts +++ b/src/reader.ts @@ -69,6 +69,31 @@ export class Reader { this.insertIntoLoadedSegments(start, bytes); } + rangeIsLoaded(start: number, end: number) { + if (end <= start) { + return true; + } + + const index = binarySearchLessOrEqual(this.loadedSegments, start, x => x.start); + if (index === -1) { + return false; + } + + for (let i = index; i < this.loadedSegments.length; i++) { + const segment = this.loadedSegments[i]!; + if (segment.start > start) { + break; + } + + const segmentEncasesRequestedRange = segment.end >= end; + if (segmentEncasesRequestedRange) { + return true; + } + } + + return false; + } + private insertIntoLoadedSegments(start: number, bytes: Uint8Array) { const segment: ReadSegment = { start, diff --git a/todo.txt b/todo.txt index 0bb2534..41415e1 100644 --- a/todo.txt +++ b/todo.txt @@ -1,3 +1,4 @@ - https://github.com/Vanilagy/mp4-muxer/issues/83 tell him it's possible now - textsubtitlesource, chunked piping -- More efficient MP3 loading when reading sequentially \ No newline at end of file +- More efficient MP3 loading when reading sequentially +- Redo readers: I think better to have a single reader with ephemeral/non-ephemeral loading methods than to have multiple readers. This way, they can still share data. \ No newline at end of file