diff --git a/src/input-format.ts b/src/input-format.ts index 1748f27..db1e5b7 100644 --- a/src/input-format.ts +++ b/src/input-format.ts @@ -155,7 +155,7 @@ export class MatroskaInputFormat extends InputFormat { } const dataSize = readElementSize(headerSlice); - if (dataSize === null) { + if (typeof dataSize !== 'number') { return false; // Miss me with that shit } @@ -171,7 +171,7 @@ export class MatroskaInputFormat extends InputFormat { const { id, size } = header; const dataStartPos = dataSlice.filePos; - if (size === null) return false; + if (size === undefined) return false; switch (id) { case EBMLId.EBMLVersion: { diff --git a/src/matroska/ebml.ts b/src/matroska/ebml.ts index eb27c40..891adab 100644 --- a/src/matroska/ebml.ts +++ b/src/matroska/ebml.ts @@ -7,7 +7,7 @@ */ import { MediaCodec } from '../codec'; -import { assertNever, textDecoder, textEncoder } from '../misc'; +import { assert, assertNever, textDecoder, textEncoder } from '../misc'; import { FileSlice, readBytes, Reader, readF32Be, readF64Be, readU8 } from '../reader'; import { Writer } from '../writer'; @@ -470,6 +470,10 @@ export const MIN_HEADER_SIZE = 2; // 1-byte ID and 1-byte size export const MAX_HEADER_SIZE = 2 * MAX_VAR_INT_SIZE; // 8-byte ID and 8-byte size export const readVarIntSize = (slice: FileSlice) => { + if (slice.remainingLength < 1) { + return null; + } + const firstByte = readU8(slice); slice.skip(-1); @@ -484,10 +488,19 @@ export const readVarIntSize = (slice: FileSlice) => { mask >>= 1; } + // Check if we have enough bytes to read the full varint + if (slice.remainingLength < width) { + return null; + } + return width; }; export const readVarInt = (slice: FileSlice) => { + if (slice.remainingLength < 1) { + return null; + } + // Read the first byte to determine the width of the variable-length integer const firstByte = readU8(slice); @@ -503,6 +516,11 @@ export const readVarInt = (slice: FileSlice) => { mask >>= 1; } + if (slice.remainingLength < width - 1) { + // Not enough bytes + return null; + } + // First byte's value needs the marker bit cleared let value = firstByte & (mask - 1); @@ -563,39 +581,58 @@ export const readElementId = (slice: FileSlice) => { return null; } + if (slice.remainingLength < size) { + return null; // It don't fit + } + const id = readUnsignedInt(slice, size); return id; }; -export const readElementSize = (slice: FileSlice) => { - let size: number | null = readU8(slice); +/** Returns `undefined` to indicate the EBML undefined size. Returns `null` if the size couldn't be read. */ +export const readElementSize = (slice: FileSlice): number | undefined | null => { + // Need at least 1 byte to read the size + if (slice.remainingLength < 1) { + return null; + } - if (size === 0xff) { - size = null; - } else { - slice.skip(-1); - size = readVarInt(slice); + const firstByte = readU8(slice); - // In some (livestreamed) files, this is the value of the size field. While this technically is just a very - // large number, it is intended to behave like the reserved size 0xFF, meaning the size is undefined. We - // catch the number here. Note that it cannot be perfectly represented as a double, but the comparison works - // nonetheless. - // eslint-disable-next-line no-loss-of-precision - if (size === 0x00ffffffffffffff) { - size = null; - } + if (firstByte === 0xff) { + return undefined; + } + + slice.skip(-1); + const size = readVarInt(slice); + + if (size === null) { + return null; + } + + // In some (livestreamed) files, this is the value of the size field. While this technically is just a very + // large number, it is intended to behave like the reserved size 0xFF, meaning the size is undefined. We + // catch the number here. Note that it cannot be perfectly represented as a double, but the comparison works + // nonetheless. + // eslint-disable-next-line no-loss-of-precision + if (size === 0x00ffffffffffffff) { + return undefined; } return size; }; export const readElementHeader = (slice: FileSlice) => { + assert(slice.remainingLength >= MIN_HEADER_SIZE); + const id = readElementId(slice); if (id === null) { return null; } const size = readElementSize(slice); + if (size === null) { + return null; + } return { id, size }; }; @@ -720,8 +757,8 @@ export const CODEC_STRING_MAP: Partial> = { 'webvtt': 'S_TEXT/WEBVTT', }; -export function assertDefinedSize(size: number | null): asserts size is number { - if (size === null) { +export function assertDefinedSize(size: number | undefined): asserts size is number { + if (size === undefined) { throw new Error('Undefined element size is used in a place where it is not supported.'); } }; diff --git a/src/matroska/matroska-demuxer.ts b/src/matroska/matroska-demuxer.ts index 965ccf8..1c47d09 100644 --- a/src/matroska/matroska-demuxer.ts +++ b/src/matroska/matroska-demuxer.ts @@ -347,7 +347,7 @@ export class MatroskaDemuxer extends Demuxer { } else if (id === EBMLId.Segment) { // Segment found! await this.readSegment(dataStartPos, size); - if (size === null) { + if (size === undefined) { // Segment sizes can be undefined (common in livestreamed files), so assume this is the last // and only segment break; @@ -365,7 +365,7 @@ export class MatroskaDemuxer extends Demuxer { // doesn't contain any of the clusters that follow it. In the case, we apply the following logic: if // we find a top-level cluster, attribute it to the previous segment. - if (size === null) { + if (size === undefined) { // Just in case this is one of those weird sizeless clusters, let's do our best and still try to // determine its size. const nextElementPos = await searchForNextElementId( @@ -390,7 +390,7 @@ export class MatroskaDemuxer extends Demuxer { })(); } - async readSegment(segmentDataStart: number, dataSize: number | null) { + async readSegment(segmentDataStart: number, dataSize: number | undefined) { this.currentSegment = { seekHeadSeen: false, infoSeen: false, @@ -407,7 +407,7 @@ export class MatroskaDemuxer extends Demuxer { cuePoints: [], dataStartPos: segmentDataStart, - elementEndPos: dataSize === null + elementEndPos: dataSize === undefined ? null // Assume it goes until the end of the file : segmentDataStart + dataSize, clusterSeekStartPos: segmentDataStart, @@ -484,7 +484,7 @@ export class MatroskaDemuxer extends Demuxer { break; // Stop at the first cluster } - if (size === null) { + if (size === undefined) { break; } else { currentPos = dataStartPos + size; @@ -614,7 +614,7 @@ export class MatroskaDemuxer extends Demuxer { let size = elementHeader.size; const dataStartPos = headerSlice.filePos; - if (size === null) { + if (size === undefined) { // The cluster's size is undefined (can happen in livestreamed files). We'd still like to know the size of // it, so we have no other choice but to iterate over the EBML structure until we find an element at level // 0 or 1, indicating the end of the cluster (all elements inside the cluster are at level 2). @@ -916,9 +916,7 @@ export class MatroskaDemuxer extends Demuxer { } readContiguousElements(slice: FileSlice, stopIds?: number[]) { - const startIndex = slice.filePos; - - while (slice.filePos - startIndex <= slice.length - MIN_HEADER_SIZE) { + while (slice.remainingLength >= MIN_HEADER_SIZE) { const startPos = slice.filePos; const foundElement = this.traverseElement(slice, stopIds); @@ -2229,7 +2227,7 @@ abstract class MatroskaTrackBacking implements InputTrackBacking { } } - if (size === null) { + if (size === undefined) { // Undefined element size (can happen in livestreamed files). In this case, we need to do some // searching to determine the actual size of the element.