diff --git a/dev/convert.html b/dev/convert.html index bc1f9d1..bb9b7a4 100644 --- a/dev/convert.html +++ b/dev/convert.html @@ -48,6 +48,7 @@ target }); + let ctx = null; const conversion = await Mediabunny.Conversion.init({ input: new Mediabunny.Input({ formats: Mediabunny.ALL_FORMATS, @@ -56,6 +57,8 @@ output, audio: (_, n) => ({ discard: n > 1, + codec: 'aac', + //codec: 'opus', /* process: (sample) => { return sample; @@ -70,7 +73,7 @@ //numberOfChannels: 1, //sampleRate: 4000 //discard: true - //forceTranscode: true, + forceTranscode: true, }), /* video: { @@ -97,6 +100,7 @@ }, */ video: () => ({ + //discard: true, /* process: (sample) => { if (!ctx) { @@ -108,9 +112,11 @@ ctx = canvas.getContext('2d'); } + console.log(ctx.canvas.width, ctx.canvas.height); + ctx.clearRect(0, 0, ctx.canvas.width, ctx.canvas.height); - sample.draw(ctx, 0, 0); - ctx.drawImage(watermark, 32, 32); + sample.drawWithFit(ctx, { fit: 'fill' }); + //ctx.drawImage(watermark, 32, 32); return ctx.canvas; }, @@ -170,8 +176,8 @@ } }, trim: { - start: 0, - end: 10 + //start: 0, + //end: 10 }, }); console.log(conversion); diff --git a/dev/demux.html b/dev/demux.html index 1a864a3..c423bfe 100644 --- a/dev/demux.html +++ b/dev/demux.html @@ -14,7 +14,21 @@ source: new Mediabunny.BlobSource(file), }); - const videoTrack = await input.getPrimaryAudioTrack(); + const audioTrack = await input.getPrimaryAudioTrack(); + const sink = new Mediabunny.AudioSampleSink(audioTrack); + + let lastEnd = 0; + for await (const sample of sink.samples()) { + if (sample.timestamp - lastEnd > 0) { + console.warn(sample.timestamp - lastEnd) + console.log(sample.timestamp, sample.duration, "diffie", sample.timestamp - lastEnd); + } + + lastEnd = sample.timestamp + sample.duration; + sample.close(); + } + + /* const sink = new Mediabunny.EncodedPacketSink(videoTrack); for await (const packet of sink.packets()) { @@ -22,6 +36,7 @@ if (packet.timestamp > 10) break; } + */ /* const sink = new Mediabunny.VideoSampleSink(videoTrack); diff --git a/package-lock.json b/package-lock.json index 271da7e..e991436 100644 --- a/package-lock.json +++ b/package-lock.json @@ -1,12 +1,12 @@ { "name": "mediabunny", - "version": "1.24.5", + "version": "1.24.6", "lockfileVersion": 3, "requires": true, "packages": { "": { "name": "mediabunny", - "version": "1.24.5", + "version": "1.24.6", "license": "MPL-2.0", "workspaces": [ "packages/*" @@ -7749,9 +7749,9 @@ } }, "node_modules/mediabunny": { - "version": "1.24.4", - "resolved": "https://registry.npmjs.org/mediabunny/-/mediabunny-1.24.4.tgz", - "integrity": "sha512-dpWYBPTtMg152yNLXZQ7xb6hsXdYbKp9EuK8qq4npS+SZ08FVc1XHlXYhrOm31T+tUVJKgm95Yaqy69wTpZP9Q==", + "version": "1.24.5", + "resolved": "https://registry.npmjs.org/mediabunny/-/mediabunny-1.24.5.tgz", + "integrity": "sha512-4Sd5x8WXQiuDKZcJDAiQjyD0Lg/cg9p+dBShCM02F1pMvKdB9R6+WHZ9NFCFSqViVPX2fYSu95ZIpBUg9+TwTQ==", "license": "MPL-2.0", "peer": true, "workspaces": [ @@ -12242,7 +12242,7 @@ }, "packages/mp3-encoder": { "name": "@mediabunny/mp3-encoder", - "version": "1.24.5", + "version": "1.24.6", "license": "MPL-2.0", "devDependencies": { "@types/emscripten": "^1.40.1" diff --git a/package.json b/package.json index 3b11949..acc0a8c 100644 --- a/package.json +++ b/package.json @@ -1,7 +1,7 @@ { "name": "mediabunny", "author": "Vanilagy", - "version": "1.24.5", + "version": "1.24.6", "description": "Pure TypeScript media toolkit for reading, writing, and converting media files, directly in the browser.", "type": "module", "workspaces": [ diff --git a/packages/mp3-encoder/package.json b/packages/mp3-encoder/package.json index b7d133b..64bc7ab 100644 --- a/packages/mp3-encoder/package.json +++ b/packages/mp3-encoder/package.json @@ -1,7 +1,7 @@ { "name": "@mediabunny/mp3-encoder", "author": "Vanilagy", - "version": "1.24.5", + "version": "1.24.6", "description": "MP3 encoder extension for Mediabunny, based on LAME.", "main": "./dist/bundles/mediabunny-mp3-encoder.mjs", "module": "./dist/bundles/mediabunny-mp3-encoder.mjs", diff --git a/src/codec.ts b/src/codec.ts index a396e70..b9414fd 100644 --- a/src/codec.ts +++ b/src/codec.ts @@ -337,6 +337,7 @@ export const extractVideoCodecString = (trackInfo: { codec: VideoCodec | null; codecDescription: Uint8Array | null; colorSpace: VideoColorSpaceInit | null; + avcType: 1 | 3 | null; avcCodecInfo: AvcDecoderConfigurationRecord | null; hevcCodecInfo: HevcDecoderConfigurationRecord | null; vp9CodecInfo: Vp9CodecInfo | null; @@ -345,6 +346,8 @@ export const extractVideoCodecString = (trackInfo: { const { codec, codecDescription, colorSpace, avcCodecInfo, hevcCodecInfo, vp9CodecInfo, av1CodecInfo } = trackInfo; if (codec === 'avc') { + assert(trackInfo.avcType !== null); + if (avcCodecInfo) { const bytes = new Uint8Array([ avcCodecInfo.avcProfileIndication, @@ -352,14 +355,14 @@ export const extractVideoCodecString = (trackInfo: { avcCodecInfo.avcLevelIndication, ]); - return `avc1.${bytesToHexString(bytes)}`; + return `avc${trackInfo.avcType}.${bytesToHexString(bytes)}`; } if (!codecDescription || codecDescription.byteLength < 4) { throw new TypeError('AVC decoder description is not provided or is not at least 4 bytes long.'); } - return `avc1.${bytesToHexString(codecDescription.subarray(1, 4))}`; + return `avc${trackInfo.avcType}.${bytesToHexString(codecDescription.subarray(1, 4))}`; } else if (codec === 'hevc') { let generalProfileSpace: number; let generalProfileIdc: number; diff --git a/src/conversion.ts b/src/conversion.ts index abf820a..0cabe1a 100644 --- a/src/conversion.ts +++ b/src/conversion.ts @@ -182,7 +182,8 @@ export type ConversionVideoOptions = { * corrections. * * Must return a {@link VideoSample} or a `CanvasImageSource`, an array of them, or `null` for dropping the frame. - * When non-timestamped data is returned, the timestamp and duration from the source sample will be used. + * When non-timestamped data is returned, the timestamp and duration from the source sample will be used. Rotation + * metadata of the returned sample will be ignored. * * This function can also be used to manually resize frames. When doing so, you should signal the post-process * dimensions using the `processedWidth` and `processedHeight` fields, which enables the encoder to better know what @@ -873,7 +874,10 @@ export class Conversion { || trackOptions.process !== undefined; let needsRerender = width !== originalWidth || height !== originalHeight - || (totalRotation !== 0 && !outputSupportsRotation) + // TODO This is suboptimal: Forcing a rerender when both rotation and process are set is not + // performance-optimal, but right now there's no other way because we can't change the track rotation + // metadata after the output has already started. Should be possible with API changes in v2, though! + || (totalRotation !== 0 && (!outputSupportsRotation || trackOptions.process !== undefined)) || !!crop; const alpha = trackOptions.alpha ?? 'discard'; diff --git a/src/flac/flac-demuxer.ts b/src/flac/flac-demuxer.ts index 5248394..4303fd4 100644 --- a/src/flac/flac-demuxer.ts +++ b/src/flac/flac-demuxer.ts @@ -328,11 +328,13 @@ export class FlacDemuxer extends Demuxer { const nextByte = readU8(slice); if (nextByte === 0xff) { + const positionBeforeReading = slice.filePos; + const byteAfterNextByte = readU8(slice); const expected = this.blockingBit === 1 ? 0b1111_1001 : 0b1111_1000; if (byteAfterNextByte !== expected) { - slice.skip(-1); + slice.filePos = positionBeforeReading; continue; } @@ -345,7 +347,7 @@ export class FlacDemuxer extends Demuxer { }); if (!nextFrameHeader) { - slice.skip(-1); + slice.filePos = positionBeforeReading; continue; } @@ -355,14 +357,14 @@ export class FlacDemuxer extends Demuxer { if (this.blockingBit === 0) { // Case A: If the stream is fixed block size, this is the frame number, which increments by 1 if (nextFrameHeader.num - frameHeader.num !== 1) { - slice.skip(-1); + slice.filePos = positionBeforeReading; continue; } } else { // Case B: If the stream is variable block size, this is the sample number, which increments by // amount of samples in a frame. if (nextFrameHeader.num - frameHeader.num !== frameHeader.blockSize) { - slice.skip(-1); + slice.filePos = positionBeforeReading; continue; } } diff --git a/src/isobmff/isobmff-boxes.ts b/src/isobmff/isobmff-boxes.ts index 96900d0..d7a5a38 100644 --- a/src/isobmff/isobmff-boxes.ts +++ b/src/isobmff/isobmff-boxes.ts @@ -577,7 +577,7 @@ export const stsd = (trackData: IsobmffTrackData) => { if (trackData.type === 'video') { sampleDescription = videoSampleDescription( - VIDEO_CODEC_TO_BOX_NAME[trackData.track.source._codec], + videoCodecToBoxName(trackData.track.source._codec, trackData.info.decoderConfig.codec), trackData, ); } else if (trackData.type === 'audio') { @@ -1572,12 +1572,14 @@ const dataStringBoxLong = (value: string) => { ]); }; -const VIDEO_CODEC_TO_BOX_NAME: Record = { - avc: 'avc1', - hevc: 'hvc1', - vp8: 'vp08', - vp9: 'vp09', - av1: 'av01', +const videoCodecToBoxName = (codec: VideoCodec, fullCodecString: string) => { + switch (codec) { + case 'avc': return fullCodecString.startsWith('avc3') ? 'avc3' : 'avc1'; + case 'hevc': return 'hvc1'; + case 'vp8': return 'vp08'; + case 'vp9': return 'vp09'; + case 'av1': return 'av01'; + } }; const VIDEO_CODEC_TO_CONFIGURATION_BOX: Record Box | null> = { diff --git a/src/isobmff/isobmff-demuxer.ts b/src/isobmff/isobmff-demuxer.ts index e1d047c..124c7b1 100644 --- a/src/isobmff/isobmff-demuxer.ts +++ b/src/isobmff/isobmff-demuxer.ts @@ -126,6 +126,7 @@ type InternalTrack = { codec: VideoCodec | null; codecDescription: Uint8Array | null; colorSpace: VideoColorSpaceInit | null; + avcType: 1 | 3 | null; avcCodecInfo: AvcDecoderConfigurationRecord | null; hevcCodecInfo: HevcDecoderConfigurationRecord | null; vp9CodecInfo: Vp9CodecInfo | null; @@ -836,6 +837,7 @@ export class IsobmffDemuxer extends Demuxer { codec: null, codecDescription: null, colorSpace: null, + avcType: null, avcCodecInfo: null, hevcCodecInfo: null, vp9CodecInfo: null, @@ -890,8 +892,9 @@ export class IsobmffDemuxer extends Demuxer { const lowercaseBoxName = sampleBoxInfo.name.toLowerCase(); if (track.info.type === 'video') { - if (lowercaseBoxName === 'avc1') { + if (lowercaseBoxName === 'avc1' || lowercaseBoxName === 'avc3') { track.info.codec = 'avc'; + track.info.avcType = lowercaseBoxName === 'avc1' ? 1 : 3; } else if (lowercaseBoxName === 'hvc1' || lowercaseBoxName === 'hev1') { track.info.codec = 'hevc'; } else if (lowercaseBoxName === 'vp08') { diff --git a/src/matroska/matroska-demuxer.ts b/src/matroska/matroska-demuxer.ts index e964bef..1992205 100644 --- a/src/matroska/matroska-demuxer.ts +++ b/src/matroska/matroska-demuxer.ts @@ -2319,6 +2319,7 @@ class MatroskaVideoTrackBacking extends MatroskaTrackBacking implements InputVid codec: this.internalTrack.info.codec, codecDescription: this.internalTrack.info.codecDescription, colorSpace: this.internalTrack.info.colorSpace, + avcType: 1, // We don't know better (or do we?) so just assume 'avc1' avcCodecInfo: this.internalTrack.info.codec === 'avc' && firstPacket ? extractAvcDecoderConfigurationRecord(firstPacket.data) : null, diff --git a/src/media-source.ts b/src/media-source.ts index 86989a2..c2b4a6b 100644 --- a/src/media-source.ts +++ b/src/media-source.ts @@ -1294,6 +1294,8 @@ class AudioEncoderWrapper { private customEncoderCallSerializer = new CallSerializer(); private customEncoderQueueSize = 0; + private lastEndSampleIndex: number | null = null; + /** * Encoders typically throw their errors "out of band", meaning asynchronously in some other execution context. * However, we want to surface these errors to the user within the normal control flow, so they don't go uncaught. @@ -1342,6 +1344,35 @@ class AudioEncoderWrapper { } assert(this.encoderInitialized); + // Handle padding of gaps with silence to avoid audio drift over time, like in + // https://github.com/Vanilagy/mediabunny/issues/176 + // TODO An open question is how encoders deal with the first AudioData having a non-zero timestamp, and with + // AudioDatas that have an overlapping timestamp range. + { + const startSampleIndex = Math.round( + audioSample.timestamp * audioSample.sampleRate, + ); + const endSampleIndex = Math.round( + (audioSample.timestamp + audioSample.duration) * audioSample.sampleRate, + ); + + if (this.lastEndSampleIndex !== null && startSampleIndex > this.lastEndSampleIndex) { + const sampleCount = startSampleIndex - this.lastEndSampleIndex; + const fillSample = new AudioSample({ + data: new Float32Array(sampleCount * audioSample.numberOfChannels), + format: 'f32-planar', + sampleRate: audioSample.sampleRate, + numberOfChannels: audioSample.numberOfChannels, + numberOfFrames: sampleCount, + timestamp: this.lastEndSampleIndex / audioSample.sampleRate, + }); + + await this.add(fillSample, true); // Recursive call + } + + this.lastEndSampleIndex = endSampleIndex; + } + if (this.customEncoder) { this.customEncoderQueueSize++; diff --git a/src/sample.ts b/src/sample.ts index 805aebe..35d7800 100644 --- a/src/sample.ts +++ b/src/sample.ts @@ -552,7 +552,6 @@ export class VideoSample implements Disposable { dHeight, ); - // Restore the previous transformation state context.restore(); } @@ -644,6 +643,8 @@ export class VideoSample implements Disposable { dy = (canvasHeight - newHeight) / 2; } + context.save(); + const aspectRatioChange = rotation % 180 === 0 ? 1 : newWidth / newHeight; context.translate(canvasWidth / 2, canvasHeight / 2); context.rotate(rotation * Math.PI / 180); @@ -655,6 +656,8 @@ export class VideoSample implements Disposable { // Important that we don't use .draw() here since that would take rotation into account, but we wanna handle it // ourselves here context.drawImage(this.toCanvasImageSource(), sx, sy, sWidth, sHeight, dx, dy, newWidth, newHeight); + + context.restore(); } /** @internal */