From 5bb9f8609cd078a721a9e00a5e4eb2854c4f6020 Mon Sep 17 00:00:00 2001 From: Vanilagy <1696106+Vanilagy@users.noreply.github.com> Date: Sat, 4 Jan 2025 14:02:45 +0100 Subject: [PATCH] Allow non-zero initial timestamps for ISOBMFF muxer --- dev/demux.html | 56 +++++++++++++++++++++++++++++++--- src/isobmff/isobmff-demuxer.ts | 36 +++++++++++++++++++++- src/isobmff/isobmff-muxer.ts | 22 +++++++------ src/matroska/matroska-muxer.ts | 2 -- src/media-source.ts | 6 ++++ src/muxer.ts | 5 --- todo.txt | 4 ++- 7 files changed, 109 insertions(+), 22 deletions(-) diff --git a/dev/demux.html b/dev/demux.html index 3924f4d..1ee7517 100644 --- a/dev/demux.html +++ b/dev/demux.html @@ -32,7 +32,7 @@ fileInput.addEventListener('change', async () => { const file = fileInput.files[0]; - const source = new Metamuxer.BufferSource(await file.arrayBuffer()) ?? new Metamuxer.BlobSource(file); + const source = new Metamuxer.BlobSource(file); const start = performance.now(); const input = new Metamuxer.Input({ @@ -40,14 +40,62 @@ source }); + const target = new Metamuxer.ArrayBufferTarget(); + const output = new Metamuxer.Output({ + format: new Metamuxer.Mp4OutputFormat({ fastStart: undefined }), + target + }); + const videoTrack = await input.getPrimaryVideoTrack(); const drain = new Metamuxer.EncodedVideoSampleDrain(videoTrack); - for await (const sample of drain.samples()) { - //console.log(chunk); + const decoderConfig = await videoTrack.getDecoderConfig(); + const sampleSource = new Metamuxer.EncodedVideoSampleSource(await videoTrack.getCodec()); + output.addVideoTrack(sampleSource); + + output.start(); + + let j = 0; + for await (const sample of drain.samples(undefined, undefined)) { + await sampleSource.digest(sample, { decoderConfig }); + + if (j++ < 12) { + console.log(sample); + } + } + + await output.finalize(); + document.body.textContent = performance.now() - start; + + console.log(target) + + function download(blob, filename) { + const url = URL.createObjectURL(blob); + const a = document.createElement('a'); + a.href = url; + a.download = filename; + a.click(); + URL.revokeObjectURL(url); + } + //download(new Blob([target.buffer]), 'converted.mp4'); + + const input2 = new Metamuxer.Input({ + formats: Metamuxer.ALL_FORMATS, + source: new Metamuxer.BufferSource(target.buffer) + }); + + const videoTrack2 = await input2.getPrimaryVideoTrack(); + const drain2 = new Metamuxer.EncodedVideoSampleDrain(videoTrack2); + + let i = 0; + for await (const sample of drain2.samples(undefined, undefined)) { + console.log(sample); + if (i++ > 10) { + break; + } } - document.body.textContent = performance.now() - start; + /* const videoTrack = await input.getPrimaryVideoTrack(); diff --git a/src/isobmff/isobmff-demuxer.ts b/src/isobmff/isobmff-demuxer.ts index ecc41dc..d5a61d3 100644 --- a/src/isobmff/isobmff-demuxer.ts +++ b/src/isobmff/isobmff-demuxer.ts @@ -88,6 +88,11 @@ type SampleTable = { presentationTimestamp: number; sampleIndex: number; }[] | null; + /** + * Provides a fast map from sample index to index in the sorted presentation timestamps array - so, a fast map from + * decode order to presentation order. + */ + presentationTimestampIndexMap: number[] | null; }; type SampleTimingEntry = { startIndex: number; @@ -272,6 +277,7 @@ export class IsobmffDemuxer extends Demuxer { chunkOffsets: [], sampleToChunk: [], presentationTimestamps: null, + presentationTimestampIndexMap: null, }; internalTrack.sampleTable = sampleTable; @@ -389,6 +395,11 @@ export class IsobmffDemuxer extends Demuxer { } sampleTable.presentationTimestamps.sort((a, b) => a.presentationTimestamp - b.presentationTimestamp); + + sampleTable.presentationTimestampIndexMap = Array(sampleTable.presentationTimestamps.length).fill(-1); + for (let i = 0; i < sampleTable.presentationTimestamps.length; i++) { + sampleTable.presentationTimestampIndexMap[sampleTable.presentationTimestamps[i]!.sampleIndex] = i; + } } else { // If they're not defined, we can simply use the decode timestamps as presentation timestamps } @@ -1624,6 +1635,15 @@ export class IsobmffDemuxer extends Demuxer { .map((x, i) => ({ presentationTimestamp: x.presentationTimestamp, sampleIndex: i })) .sort((a, b) => a.presentationTimestamp - b.presentationTimestamp); + // Update sample durations based on presentation order + for (let i = 0; i < trackData.presentationTimestamps.length - 1; i++) { + const current = trackData.presentationTimestamps[i]!; + const next = trackData.presentationTimestamps[i + 1]!; + + const duration = next.presentationTimestamp - current.presentationTimestamp; + trackData.samples[current.sampleIndex]!.duration = duration; + } + const firstSample = trackData.samples[trackData.presentationTimestamps[0]!.sampleIndex]!; const lastSample = trackData.samples[last(trackData.presentationTimestamps)!.sampleIndex]!; @@ -2319,9 +2339,23 @@ const getSampleInfo = (sampleTable: SampleTable, sampleIndex: number): SampleInf } } + let duration = timingEntry.delta; + if (sampleTable.presentationTimestamps) { + // In order to accurately compute the duration, we need to take the duration to the next sample in presentation + // order, not in decode order + const presentationIndex = sampleTable.presentationTimestampIndexMap![sampleIndex]; + assert(presentationIndex !== undefined); + + if (presentationIndex < sampleTable.presentationTimestamps.length - 1) { + const nextEntry = sampleTable.presentationTimestamps[presentationIndex + 1]!; + const nextPresentationTimestamp = nextEntry.presentationTimestamp; + duration = nextPresentationTimestamp - presentationTimestamp; + } + } + return { presentationTimestamp, - duration: timingEntry.delta, + duration, sampleOffset, sampleSize, chunkOffset, diff --git a/src/isobmff/isobmff-muxer.ts b/src/isobmff/isobmff-muxer.ts index e280ec7..dd96480 100644 --- a/src/isobmff/isobmff-muxer.ts +++ b/src/isobmff/isobmff-muxer.ts @@ -85,8 +85,6 @@ export const intoTimescale = (timeInSeconds: number, timescale: number, round = }; export class IsobmffMuxer extends Muxer { - override timestampsMustStartAtZero = true; - private writer: Writer; private boxWriter: IsobmffBoxWriter; private fastStart: NonNullable; @@ -263,10 +261,6 @@ export class IsobmffMuxer extends Muxer { this.trackDatas.push(newTrackData); this.trackDatas.sort((a, b) => a.track.id - b.track.id); - // Subtitle cues don't need to start at 0, so let's register a timestamp at 0 to satisfy the - // "timestamps must start a zero" constraint - this.validateAndNormalizeTimestamp(track, 0, true); - return newTrackData; } @@ -446,8 +440,7 @@ export class IsobmffMuxer extends Muxer { data, size: data.byteLength, type, - // Will be refined once the next sample comes in - timescaleUnitsToNextSample: intoTimescale(duration, trackData.timescale), + timescaleUnitsToNextSample: intoTimescale(duration, trackData.timescale), // Will be refined }; return sample; @@ -469,6 +462,12 @@ export class IsobmffMuxer extends Muxer { // model it. sample.decodeTimestamp = sortedTimestamps[i]!; + if (this.fastStart !== 'fragmented' && trackData.lastTimescaleUnits === null) { + // In non-fragmented files, the first decode timestamp is always zero. If the first presentation + // timestamp isn't zero, we'll simply use the composition time offset to achieve it. + sample.decodeTimestamp = 0; + } + const sampleCompositionTimeOffset = intoTimescale(sample.timestamp - sample.decodeTimestamp, trackData.timescale); const durationInTimescale = intoTimescale(sample.duration, trackData.timescale); @@ -533,7 +532,8 @@ export class IsobmffMuxer extends Muxer { } } } else { - trackData.lastTimescaleUnits = 0; + // Decode timestamp of the first sample + trackData.lastTimescaleUnits = intoTimescale(sample.decodeTimestamp, trackData.timescale, false); if (this.fastStart !== 'fragmented') { trackData.timeToSampleTable.push({ @@ -581,6 +581,10 @@ export class IsobmffMuxer extends Muxer { // We can only finalize this fragment (and begin a new one) if we know that each track will be able to // start the new one with a key frame. const keyFrameQueuedEverywhere = this.trackDatas.every((otherTrackData) => { + if (otherTrackData.track.source._closed) { + return true; + } + if (trackData === otherTrackData) { return sample.type === 'key'; } diff --git a/src/matroska/matroska-muxer.ts b/src/matroska/matroska-muxer.ts index f14bb50..549ac6b 100644 --- a/src/matroska/matroska-muxer.ts +++ b/src/matroska/matroska-muxer.ts @@ -121,8 +121,6 @@ const TRACK_TYPE_MAP: Record = { }; export class MatroskaMuxer extends Muxer { - override timestampsMustStartAtZero = false; - private writer: Writer; private format: WebMOutputFormat | MkvOutputFormat; diff --git a/src/media-source.ts b/src/media-source.ts index 33bc9f4..7d18cbf 100644 --- a/src/media-source.ts +++ b/src/media-source.ts @@ -110,6 +110,9 @@ export class EncodedVideoSampleSource extends VideoSource { if (!(sample instanceof EncodedVideoSample)) { throw new TypeError('sample must be an EncodedVideoSample.'); } + if (sample.isMetadataOnly) { + throw new TypeError('Metadata-only samples cannot be digested.'); + } this._ensureValidDigest(); return this._connectedTrack!.output._muxer.addEncodedVideoSample(this._connectedTrack!, sample, meta); @@ -404,6 +407,9 @@ export class EncodedAudioSampleSource extends AudioSource { if (!(sample instanceof EncodedAudioSample)) { throw new TypeError('chunk must be an EncodedAudioSample.'); } + if (sample.isMetadataOnly) { + throw new TypeError('Metadata-only samples cannot be digested.'); + } this._ensureValidDigest(); return this._connectedTrack!.output._muxer.addEncodedAudioSample(this._connectedTrack!, sample, meta); diff --git a/src/muxer.ts b/src/muxer.ts index f7c1802..8afc366 100644 --- a/src/muxer.ts +++ b/src/muxer.ts @@ -34,7 +34,6 @@ export abstract class Muxer { lastKeyFrameTimestamp: number; }>(); - abstract timestampsMustStartAtZero: boolean; protected validateAndNormalizeTimestamp(track: OutputTrack, timestampInSeconds: number, isKeyFrame: boolean) { let timestampInfo = this.trackTimestampInfo.get(track); if (!timestampInfo) { @@ -42,10 +41,6 @@ export abstract class Muxer { throw new Error('First frame must be a key frame.'); } - if (this.timestampsMustStartAtZero && timestampInSeconds > 0) { - throw new Error(`Timestamps must start at zero (got ${timestampInSeconds}s).`); - } - timestampInfo = { timestampOffset: timestampInSeconds, maxTimestamp: track.source._offsetTimestamps ? 0 : timestampInSeconds, diff --git a/todo.txt b/todo.txt index 84501e1..30470b0 100644 --- a/todo.txt +++ b/todo.txt @@ -4,4 +4,6 @@ - Mov muxer!!! Only mov can hold PCM audio, MP4 cannot - Metadata methods for computing average fps and bitrate - A stream source?? Or like a callback-driven source -- onRead callback for Source \ No newline at end of file +- onRead callback for Source +- onHeader, etc callbacks for Matroska +- Add cslg box \ No newline at end of file