diff --git a/docs/guide/output-formats.md b/docs/guide/output-formats.md index 98e61d3..6e125cb 100644 --- a/docs/guide/output-formats.md +++ b/docs/guide/output-formats.md @@ -272,6 +272,10 @@ const output = new Output({ }); ``` +::: info +This format ensures [append-only writing](#append-only-writing). +::: + The following options are available: ```ts type AdtsOutputFormatOptions = { diff --git a/docs/index.md b/docs/index.md index 2addbdb..6bb3731 100644 --- a/docs/index.md +++ b/docs/index.md @@ -106,6 +106,8 @@ const sponsors = { { image: '/sponsors/jellypod.png', name: 'Jellypod', url: 'https://jellypod.ai/' }, ], individual: [ + { image: 'https://avatars.githubusercontent.com/u/82552321', name: 'Polotno', url: 'https://github.com/polotno-project' }, + { image: 'https://avatars.githubusercontent.com/u/489051', name: 'Roman Rädle', url: 'https://github.com/raedle' }, { image: 'https://avatars.githubusercontent.com/u/197597', name: 'Christopher Chedeau', url: 'https://github.com/vjeux' }, { image: 'https://avatars.githubusercontent.com/u/84167135', name: 'Memenome', url: 'https://github.com/memenome' }, { image: 'https://avatars.githubusercontent.com/u/5913254', name: 'Brandon McConnell', url: 'https://github.com/brandonmcconnell' }, diff --git a/package-lock.json b/package-lock.json index 796e73f..62971f4 100644 --- a/package-lock.json +++ b/package-lock.json @@ -1,12 +1,12 @@ { "name": "mediabunny", - "version": "1.27.0", + "version": "1.27.3", "lockfileVersion": 3, "requires": true, "packages": { "": { "name": "mediabunny", - "version": "1.27.0", + "version": "1.27.3", "license": "MPL-2.0", "workspaces": [ "packages/*" @@ -7739,9 +7739,9 @@ } }, "node_modules/mediabunny": { - "version": "1.26.0", - "resolved": "https://registry.npmjs.org/mediabunny/-/mediabunny-1.26.0.tgz", - "integrity": "sha512-0duZPn/vVpx+mKjOn3PNb/tBrtKUoq4n5+uO1JwyQm3cdaKYy7tVTEvTpXJxbpQ/CdnHn5YMVRg8mwi7fIgSYA==", + "version": "1.27.2", + "resolved": "https://registry.npmjs.org/mediabunny/-/mediabunny-1.27.2.tgz", + "integrity": "sha512-0g/vmb6X0xmnzqW0U44weF9CmzcUFFQRHrjzoVpSL2KXuhWIWcW3u9wIcohHiTY78jfr0SauYKLxjOQReaMxXQ==", "license": "MPL-2.0", "peer": true, "workspaces": [ @@ -12065,7 +12065,7 @@ }, "packages/mp3-encoder": { "name": "@mediabunny/mp3-encoder", - "version": "1.27.0", + "version": "1.27.3", "license": "MPL-2.0", "devDependencies": { "@types/emscripten": "^1.40.1" diff --git a/package.json b/package.json index 02f0c56..c03709d 100644 --- a/package.json +++ b/package.json @@ -1,7 +1,7 @@ { "name": "mediabunny", "author": "Vanilagy", - "version": "1.27.0", + "version": "1.27.3", "description": "Pure TypeScript media toolkit for reading, writing, and converting media files, directly in the browser.", "type": "module", "workspaces": [ diff --git a/packages/mp3-encoder/package.json b/packages/mp3-encoder/package.json index ae9eaeb..b416c9b 100644 --- a/packages/mp3-encoder/package.json +++ b/packages/mp3-encoder/package.json @@ -1,7 +1,7 @@ { "name": "@mediabunny/mp3-encoder", "author": "Vanilagy", - "version": "1.27.0", + "version": "1.27.3", "description": "MP3 encoder extension for Mediabunny, based on LAME.", "main": "./dist/bundles/mediabunny-mp3-encoder.mjs", "module": "./dist/bundles/mediabunny-mp3-encoder.mjs", diff --git a/src/codec-data.ts b/src/codec-data.ts index 2fd8ca7..4882752 100644 --- a/src/codec-data.ts +++ b/src/codec-data.ts @@ -24,6 +24,7 @@ import { toUint8Array, getChromiumVersion, isChromium, + setUint24, } from './misc'; import { PacketType } from './packet'; import { MetadataTags } from './metadata'; @@ -161,38 +162,54 @@ const removeEmulationPreventionBytes = (data: Uint8Array) => { return new Uint8Array(result); }; -/** Converts an AVC packet in Annex B format to length-prefixed format. */ -export const transformAnnexBToLengthPrefixed = (packetData: Uint8Array) => { - const NAL_UNIT_LENGTH_SIZE = 4; +const ANNEX_B_START_CODE = new Uint8Array([0, 0, 0, 1]); - const nalUnits = findNalUnitsInAnnexB(packetData); - - if (nalUnits.length === 0) { - // If no NAL units were found, it's not valid Annex B data - return null; - } - - let totalSize = 0; - for (const nalUnit of nalUnits) { - totalSize += NAL_UNIT_LENGTH_SIZE + nalUnit.byteLength; - } - - const avccData = new Uint8Array(totalSize); - const dataView = new DataView(avccData.buffer); +export const concatNalUnitsInAnnexB = (nalUnits: Uint8Array[]) => { + const totalLength = nalUnits.reduce((a, b) => a + ANNEX_B_START_CODE.byteLength + b.byteLength, 0); + const result = new Uint8Array(totalLength); let offset = 0; - // Write each NAL unit with its length prefix for (const nalUnit of nalUnits) { - const length = nalUnit.byteLength; + result.set(ANNEX_B_START_CODE, offset); + offset += ANNEX_B_START_CODE.byteLength; - dataView.setUint32(offset, length, false); - offset += 4; - - avccData.set(nalUnit, offset); + result.set(nalUnit, offset); offset += nalUnit.byteLength; } - return avccData; + return result; +}; + +export const concatNalUnitsInLengthPrefixed = (nalUnits: Uint8Array[], lengthSize: 1 | 2 | 3 | 4) => { + const totalLength = nalUnits.reduce((a, b) => a + lengthSize + b.byteLength, 0); + const result = new Uint8Array(totalLength); + let offset = 0; + + for (const nalUnit of nalUnits) { + const dataView = new DataView(result.buffer, result.byteOffset, result.byteLength); + + switch (lengthSize) { + case 1: + dataView.setUint8(offset, nalUnit.byteLength); + break; + case 2: + dataView.setUint16(offset, nalUnit.byteLength, false); + break; + case 3: + setUint24(dataView, offset, nalUnit.byteLength, false); + break; + case 4: + dataView.setUint32(offset, nalUnit.byteLength, false); + break; + } + + offset += lengthSize; + + result.set(nalUnit, offset); + offset += nalUnit.byteLength; + } + + return result; }; // Data specified in ISO 14496-15 @@ -227,7 +244,22 @@ export const extractAvcNalUnits = (packetData: Uint8Array, decoderConfig: VideoD } }; -const extractNalUnitTypeForAvc = (data: Uint8Array) => { +export const concatAvcNalUnits = (nalUnits: Uint8Array[], decoderConfig: VideoDecoderConfig) => { + if (decoderConfig.description) { + // Stream is length-prefixed. Let's extract the size of the length prefix from the decoder config + + const bytes = toUint8Array(decoderConfig.description); + const lengthSizeMinusOne = bytes[4]! & 0b11; + const lengthSize = (lengthSizeMinusOne + 1) as 1 | 2 | 3 | 4; + + return concatNalUnitsInLengthPrefixed(nalUnits, lengthSize); + } else { + // Stream is in Annex B format + return concatNalUnitsInAnnexB(nalUnits); + } +}; + +export const extractNalUnitTypeForAvc = (data: Uint8Array) => { return data[0]! & 0x1F; }; diff --git a/src/isobmff/isobmff-muxer.ts b/src/isobmff/isobmff-muxer.ts index e866f9d..fd73b50 100644 --- a/src/isobmff/isobmff-muxer.ts +++ b/src/isobmff/isobmff-muxer.ts @@ -25,11 +25,12 @@ import { import { BufferTarget } from '../target'; import { EncodedPacket, PacketType } from '../packet'; import { + concatNalUnitsInLengthPrefixed, extractAvcDecoderConfigurationRecord, extractHevcDecoderConfigurationRecord, + findNalUnitsInAnnexB, serializeAvcDecoderConfigurationRecord, serializeHevcDecoderConfigurationRecord, - transformAnnexBToLengthPrefixed, } from '../codec-data'; import { buildIsobmffMimeType } from './isobmff-misc'; import { MAX_BOX_HEADER_SIZE, MIN_BOX_HEADER_SIZE } from './isobmff-reader'; @@ -464,15 +465,18 @@ export class IsobmffMuxer extends Muxer { let packetData = packet.data; if (trackData.info.requiresAnnexBTransformation) { - const transformedData = transformAnnexBToLengthPrefixed(packetData); - if (!transformedData) { + const nalUnits = findNalUnitsInAnnexB(packetData); + if (nalUnits.length === 0) { + // It's not valid Annex B data throw new Error( 'Failed to transform packet data. Make sure all packets are provided in Annex B format, as' + ' specified in ITU-T-REC-H.264 and ITU-T-REC-H.265.', ); } - packetData = transformedData; + // We don't strip things like SPS or PPS NALUs here, mainly because they can also appear in the middle + // of a stream and potentially modify the parameters of it. So, let's just leave them in to be sure. + packetData = concatNalUnitsInLengthPrefixed(nalUnits, 4); } const timestamp = this.validateAndNormalizeTimestamp( diff --git a/src/media-sink.ts b/src/media-sink.ts index 711d0c4..539606e 100644 --- a/src/media-sink.ts +++ b/src/media-sink.ts @@ -8,9 +8,12 @@ import { parsePcmCodec, PCM_AUDIO_CODECS, PcmAudioCodec, VideoCodec, AudioCodec } from './codec'; import { + concatAvcNalUnits, deserializeAvcDecoderConfigurationRecord, determineVideoPacketType, + extractAvcNalUnits, extractHevcNalUnits, + extractNalUnitTypeForAvc, extractNalUnitTypeForHevc, HevcNalUnitType, parseAvcSps, @@ -928,8 +931,6 @@ class VideoDecoderWrapper extends DecoderWrapper { this.raslSkipped = true; } - this.currentPacketIndex++; - if (this.customDecoder) { this.customDecoderQueueSize++; void this.customDecoderCallSerializer @@ -942,9 +943,24 @@ class VideoDecoderWrapper extends DecoderWrapper { insertSorted(this.inputTimestamps, packet.timestamp, x => x); } + // Workaround for https://issues.chromium.org/issues/470109459 + if (isChromium() && this.currentPacketIndex === 0 && this.codec === 'avc') { + const nalUnits = extractAvcNalUnits(packet.data, this.decoderConfig); + const filteredNalUnits = nalUnits.filter((x) => { + const type = extractNalUnitTypeForAvc(x); + // These trip up Chromium's key frame detection, so let's strip them + return !(type >= 20 && type <= 31); + }); + + const newData = concatAvcNalUnits(filteredNalUnits, this.decoderConfig); + packet = new EncodedPacket(newData, packet.type, packet.timestamp, packet.duration); + } + this.decoder.decode(packet.toEncodedVideoChunk()); this.decodeAlphaData(packet); } + + this.currentPacketIndex++; } decodeAlphaData(packet: EncodedPacket) { diff --git a/src/sample.ts b/src/sample.ts index 94e47ef..42dc131 100644 --- a/src/sample.ts +++ b/src/sample.ts @@ -18,6 +18,7 @@ import { isFirefox, polyfillSymbolDispose, assertNever, + isWebKit, } from './misc'; polyfillSymbolDispose(); @@ -1590,6 +1591,7 @@ export class AudioSample implements Disposable { const { planeIndex, format, frameCount: optFrameCount, frameOffset: optFrameOffset } = options; + const srcFormat = this.format; const destFormat = format ?? this.format; if (!destFormat) throw new Error('Destination format not determined'); @@ -1624,58 +1626,31 @@ export class AudioSample implements Disposable { const writeFn = getWriteFunction(destFormat); if (isAudioData(this._data)) { - if (destIsPlanar) { - if (destFormat === 'f32-planar') { - // Simple, since the browser must support f32-planar, we can just delegate here - this._data.copyTo(destination, { - planeIndex, - frameOffset, - frameCount: copyFrameCount, - format: 'f32-planar', - }); - } else { - // Allocate temporary buffer for f32-planar data - const tempBuffer = new ArrayBuffer(copyFrameCount * 4); - const tempArray = new Float32Array(tempBuffer); - this._data.copyTo(tempArray, { - planeIndex, - frameOffset, - frameCount: copyFrameCount, - format: 'f32-planar', - }); - - // Convert each f32 sample to destination format - const tempView = new DataView(tempBuffer); - for (let i = 0; i < copyFrameCount; i++) { - const destOffset = i * destBytesPerSample; - const sample = tempView.getFloat32(i * 4, true); - writeFn(destView, destOffset, sample); - } - } + if (isWebKit() && numChannels > 2 && destFormat !== srcFormat) { + // WebKit bug workaround + doAudioDataCopyToWebKitWorkaround( + this._data, + destView, + srcFormat, + destFormat, + numChannels, + planeIndex, + frameOffset, + copyFrameCount, + ); } else { - // Destination is interleaved. - // Allocate a temporary Float32Array to hold one channel's worth of data. - const numCh = numChannels; - const temp = new Float32Array(copyFrameCount); - for (let ch = 0; ch < numCh; ch++) { - this._data.copyTo(temp, { - planeIndex: ch, - frameOffset, - frameCount: copyFrameCount, - format: 'f32-planar', - }); - for (let i = 0; i < copyFrameCount; i++) { - const destIndex = i * numCh + ch; - const destOffset = destIndex * destBytesPerSample; - writeFn(destView, destOffset, temp[i]!); - } - } + // Per spec, only f32-planar conversion must be supported, but in practice, all browsers support all + // destination formats, so let's just delegate here: + this._data.copyTo(destination, { + planeIndex, + frameOffset, + frameCount: copyFrameCount, + format: destFormat, + }); } } else { const uint8Data = this._data; const srcView = toDataView(uint8Data); - - const srcFormat = this.format; const readFn = getReadFunction(srcFormat); const srcBytesPerSample = getBytesPerSample(srcFormat); const srcIsPlanar = formatIsPlanar(srcFormat); @@ -2035,3 +2010,112 @@ const getWriteFunction = (format: AudioSampleFormat): (view: DataView, offset: n const isAudioData = (x: unknown): x is AudioData => { return typeof AudioData !== 'undefined' && x instanceof AudioData; }; + +/** + * WebKit has a bug where calling AudioData.copyTo with a format different from the source format + * crashes the tab when there are more than 2 channels. This function works around that by always + * copying with the source format and then manually converting to the destination format. + * + * See https://bugs.webkit.org/show_bug.cgi?id=302521. + */ +const doAudioDataCopyToWebKitWorkaround = ( + audioData: AudioData, + destView: DataView, + srcFormat: AudioSampleFormat, + destFormat: AudioSampleFormat, + numChannels: number, + planeIndex: number, + frameOffset: number, + copyFrameCount: number, +) => { + const readFn = getReadFunction(srcFormat); + const writeFn = getWriteFunction(destFormat); + const srcBytesPerSample = getBytesPerSample(srcFormat); + const destBytesPerSample = getBytesPerSample(destFormat); + const srcIsPlanar = formatIsPlanar(srcFormat); + const destIsPlanar = formatIsPlanar(destFormat); + + if (destIsPlanar) { + if (srcIsPlanar) { + // src planar -> dest planar: copy single plane and convert + const data = new ArrayBuffer(copyFrameCount * srcBytesPerSample); + const dataView = toDataView(data); + + audioData.copyTo(data, { + planeIndex, + frameOffset, + frameCount: copyFrameCount, + format: srcFormat, + }); + + for (let i = 0; i < copyFrameCount; i++) { + const srcOffset = i * srcBytesPerSample; + const destOffset = i * destBytesPerSample; + const sample = readFn(dataView, srcOffset); + writeFn(destView, destOffset, sample); + } + } else { + // src interleaved -> dest planar: copy all interleaved data, extract one channel + const data = new ArrayBuffer(copyFrameCount * numChannels * srcBytesPerSample); + const dataView = toDataView(data); + + audioData.copyTo(data, { + planeIndex: 0, + frameOffset, + frameCount: copyFrameCount, + format: srcFormat, + }); + + for (let i = 0; i < copyFrameCount; i++) { + const srcOffset = (i * numChannels + planeIndex) * srcBytesPerSample; + const destOffset = i * destBytesPerSample; + const sample = readFn(dataView, srcOffset); + writeFn(destView, destOffset, sample); + } + } + } else { + if (srcIsPlanar) { + // src planar -> dest interleaved: copy each plane and interleave + const planeSize = copyFrameCount * srcBytesPerSample; + const data = new ArrayBuffer(planeSize); + const dataView = toDataView(data); + + for (let ch = 0; ch < numChannels; ch++) { + audioData.copyTo(data, { + planeIndex: ch, + frameOffset, + frameCount: copyFrameCount, + format: srcFormat, + }); + + for (let i = 0; i < copyFrameCount; i++) { + const srcOffset = i * srcBytesPerSample; + const destOffset = (i * numChannels + ch) * destBytesPerSample; + const sample = readFn(dataView, srcOffset); + writeFn(destView, destOffset, sample); + } + } + } else { + // src interleaved -> dest interleaved: copy all and convert + const data = new ArrayBuffer(copyFrameCount * numChannels * srcBytesPerSample); + const dataView = toDataView(data); + + audioData.copyTo(data, { + planeIndex: 0, + frameOffset, + frameCount: copyFrameCount, + format: srcFormat, + }); + + for (let i = 0; i < copyFrameCount; i++) { + for (let ch = 0; ch < numChannels; ch++) { + const idx = i * numChannels + ch; + const srcOffset = idx * srcBytesPerSample; + const destOffset = idx * destBytesPerSample; + const sample = readFn(dataView, srcOffset); + writeFn(destView, destOffset, sample); + } + } + } + } +}; diff --git a/src/source.ts b/src/source.ts index efd4879..0059189 100644 --- a/src/source.ts +++ b/src/source.ts @@ -263,6 +263,11 @@ export class BlobSource extends Source { } worker.running = false; + + if (worker.aborted) { + // MDN: "Calling this method signals a loss of interest in the stream by a consumer." + await reader?.cancel(); + } } /** @internal */ @@ -556,7 +561,7 @@ export class UrlSource extends Source { } if (worker.aborted) { - break; + continue; // Cleanup happens in next iteration } const { done, value } = readResult; @@ -578,14 +583,8 @@ export class UrlSource extends Source { this.onread?.(worker.currentPos, worker.currentPos + value.length); this._orchestrator.supplyWorkerData(worker, value); } - - if (worker.aborted) { - break; - } } - worker.running = false; - // The previous UrlSource had logic for circumventing https://issues.chromium.org/issues/436025873; I haven't // been able to observe this bug with the new UrlSource (maybe because we're using response streaming), so the // logic for that has vanished for now. Leaving a comment here if this becomes relevant again. diff --git a/test/node/annex-b-conversion.test.ts b/test/node/annex-b-conversion.test.ts new file mode 100644 index 0000000..3e86555 --- /dev/null +++ b/test/node/annex-b-conversion.test.ts @@ -0,0 +1,54 @@ +import { expect, test } from 'vitest'; +import { Input } from '../../src/input.js'; +import { BufferSource, FilePathSource } from '../../src/source.js'; +import path from 'node:path'; +import { ALL_FORMATS } from '../../src/input-format.js'; +import { Output } from '../../src/output.js'; +import { Mp4OutputFormat } from '../../src/output-format.js'; +import { BufferTarget } from '../../src/target.js'; +import { Conversion } from '../../src/conversion.js'; +import { EncodedPacketSink } from '../../src/media-sink.js'; +import { extractAvcNalUnits } from '../../src/codec-data.js'; + +const __dirname = new URL('.', import.meta.url).pathname; + +test('Annex B to length-prefixed conversion, MP4', async () => { + using originalInput = new Input({ + source: new FilePathSource(path.join(__dirname, '..', 'public/annex-b-avc.mkv')), + formats: ALL_FORMATS, + }); + const originalVideoTrack = (await originalInput.getPrimaryVideoTrack())!; + const originalDecoderConfig = (await originalVideoTrack.getDecoderConfig())!; + expect(originalDecoderConfig.description).toBeUndefined(); + expect(originalVideoTrack.codec).toBe('avc'); + + const originalSink = new EncodedPacketSink(originalVideoTrack); + const originalFirstPacket = await originalSink.getFirstPacket(); + expect([...originalFirstPacket!.data.slice(0, 4)]).toEqual([0, 0, 0, 1]); + + const originalNalUnits = extractAvcNalUnits(originalFirstPacket!.data, originalDecoderConfig); + + const output = new Output({ + format: new Mp4OutputFormat(), + target: new BufferTarget(), + }); + + const conversion = await Conversion.init({ input: originalInput, output }); + await conversion.execute(); + + using newInput = new Input({ + source: new BufferSource(output.target.buffer!), + formats: ALL_FORMATS, + }); + const newVideoTrack = (await newInput.getPrimaryVideoTrack())!; + const newDecoderConfig = (await newVideoTrack.getDecoderConfig())!; + expect(newDecoderConfig.description).toBeDefined(); + expect(newVideoTrack.codec).toBe('avc'); + + const newSink = new EncodedPacketSink(newVideoTrack); + const newFirstPacket = await newSink.getFirstPacket(); + expect([...newFirstPacket!.data.slice(0, 4)]).not.toEqual([0, 0, 0, 1]); // Successfully converted + + const newNalUnits = extractAvcNalUnits(newFirstPacket!.data, newDecoderConfig); + expect(newNalUnits).toEqual(originalNalUnits); // Content is the same though +}); diff --git a/test/public/annex-b-avc.mkv b/test/public/annex-b-avc.mkv new file mode 100644 index 0000000..699dfd2 Binary files /dev/null and b/test/public/annex-b-avc.mkv differ