Transparent video read/write support (#145)

* Add support for reading transparent Matroska and implement alpha side data & decode

* Few fixes

* Implement alpha encoding

* Gracefully handle inability to acquire WebGL context, properly clean up WebGL contexts

* Improve touch device detection

* Test test

* Test test #2

* Test test 3

* Test test 4

* Test test 5

* Test test 6

* Test test 7

* Test test 8

* Test test 9

* Test test 10

* Test test 11

* Test test 12

* Test test 13

* Test test 14

* Test test 15

* Test test 16

* Test test 17

* Test test 18

* Test test 19

* Test test 20

* Fix type errors, improve MetadataTags docs

* Do a bunch of docs work

* Test test?

* "Unexpected only modifier 🤓"

* Add InputVideoTrack.canBeTransparent()

* Some clean-up

* Adjust CI to be less spammy in PRs

* Fix broken license headers
This commit is contained in:
David P.
2025-09-24 21:20:14 +02:00
committed by GitHub
parent 0878dd2ca9
commit fa4e064345
34 changed files with 1537 additions and 181 deletions
+423 -63
View File
@@ -7,7 +7,12 @@
*/
import { parsePcmCodec, PCM_AUDIO_CODECS, PcmAudioCodec, VideoCodec, AudioCodec } from './codec';
import { extractHevcNalUnits, extractNalUnitTypeForHevc, HevcNalUnitType } from './codec-data';
import {
determineVideoPacketType,
extractHevcNalUnits,
extractNalUnitTypeForHevc,
HevcNalUnitType,
} from './codec-data';
import { CustomVideoDecoder, customVideoDecoders, CustomAudioDecoder, customAudioDecoders } from './custom-coder';
import { InputDisposedError } from './input';
import { InputAudioTrack, InputTrack, InputVideoTrack } from './input-track';
@@ -365,7 +370,7 @@ abstract class DecoderWrapper<
> {
constructor(
public onSample: (sample: MediaSample) => unknown,
public onError: (error: DOMException) => unknown,
public onError: (error: Error) => unknown,
) {}
abstract getDecodeQueueSize(): number;
@@ -388,7 +393,7 @@ export abstract class BaseMediaSampleSink<
/** @internal */
abstract _createDecoder(
onSample: (sample: MediaSample) => unknown,
onError: (error: DOMException) => unknown
onError: (error: Error) => unknown
): Promise<DecoderWrapper<MediaSample>>;
/** @internal */
abstract _createPacketSink(): EncodedPacketSink;
@@ -811,9 +816,23 @@ class VideoDecoderWrapper extends DecoderWrapper<VideoSample> {
currentPacketIndex = 0;
raslSkipped = false; // For HEVC stuff
// Alpha stuff
alphaDecoder: VideoDecoder | null = null;
alphaHadKeyframe = false;
colorQueue: VideoFrame[] = [];
alphaQueue: (VideoFrame | null)[] = [];
merger: ColorAlphaMerger | null = null;
mergerCreationFailed = false;
decodedAlphaChunkCount = 0;
alphaDecoderQueueSize = 0;
/** Each value is the number of decoded alpha chunks at which a null alpha frame should be added. */
nullAlphaFrameQueue: number[] = [];
currentAlphaPacketIndex = 0;
alphaRaslSkipped = false; // For HEVC stuff
constructor(
onSample: (sample: VideoSample) => unknown,
onError: (error: DOMException) => unknown,
onError: (error: Error) => unknown,
public codec: VideoCodec,
public decoderConfig: VideoDecoderConfig,
public rotation: Rotation,
@@ -840,79 +859,48 @@ class VideoDecoderWrapper extends DecoderWrapper<VideoSample> {
void this.customDecoderCallSerializer.call(() => this.customDecoder!.init());
} else {
// Specific handler for the WebCodecs VideoDecoder to iron out browser differences
const sampleHandler = (sample: VideoSample) => {
if (isSafari()) {
// For correct B-frame handling, we don't just hand over the frames directly but instead add them to
// a queue, because we want to ensure frames are emitted in presentation order. We flush the queue
// each time we receive a frame with a timestamp larger than the highest we've seen so far, as we
// can sure that is not a B-frame. Typically, WebCodecs automatically guarantees that frames are
// emitted in presentation order, but Safari doesn't always follow this rule.
if (this.sampleQueue.length > 0 && (sample.timestamp >= last(this.sampleQueue)!.timestamp)) {
for (const sample of this.sampleQueue) {
this.finalizeAndEmitSample(sample);
}
const colorHandler = (frame: VideoFrame) => {
if (this.alphaQueue.length > 0) {
// Even when no alpha data is present (most of the time), there will be nulls in this queue
const alphaFrame = this.alphaQueue.shift();
assert(alphaFrame !== undefined);
this.sampleQueue.length = 0;
}
insertSorted(this.sampleQueue, sample, x => x.timestamp);
this.mergeAlpha(frame, alphaFrame);
} else {
// Assign it the next earliest timestamp from the input. We do this because browsers, by spec, are
// required to emit decoded frames in presentation order *while* retaining the timestamp of their
// originating EncodedVideoChunk. For files with B-frames but no out-of-order timestamps (like a
// missing ctts box, for example), this causes a mismatch. We therefore fix the timestamps and
// ensure they are sorted by doing this.
const timestamp = this.inputTimestamps.shift();
// There's no way we'd have more decoded frames than encoded packets we passed in. Actually, the
// correspondence should be 1:1.
assert(timestamp !== undefined);
sample.setTimestamp(timestamp);
this.finalizeAndEmitSample(sample);
this.colorQueue.push(frame);
}
};
this.decoder = new VideoDecoder({
output: frame => sampleHandler(new VideoSample(frame)),
output: (frame) => {
try {
colorHandler(frame);
} catch (error) {
this.onError(error as Error);
}
},
error: onError,
});
this.decoder.configure(decoderConfig);
}
}
finalizeAndEmitSample(sample: VideoSample) {
// Round the timestamps to the time resolution
sample.setTimestamp(Math.round(sample.timestamp * this.timeResolution) / this.timeResolution);
sample.setDuration(Math.round(sample.duration * this.timeResolution) / this.timeResolution);
sample.setRotation(this.rotation);
this.onSample(sample);
}
getDecodeQueueSize() {
if (this.customDecoder) {
return this.customDecoderQueueSize;
} else {
assert(this.decoder);
return this.decoder.decodeQueueSize;
return Math.max(
this.decoder.decodeQueueSize,
this.alphaDecoder?.decodeQueueSize ?? 0,
);
}
}
decode(packet: EncodedPacket) {
if (this.codec === 'hevc' && this.currentPacketIndex > 0 && !this.raslSkipped) {
// If we're using HEVC, we need to make sure to skip any RASL slices that follow a non-IDR key frame such as
// CRA_NUT. This is because RASL slices cannot be decoded without data before the CRA_NUT. Browsers behave
// differently here: Chromium drops the packets, Safari throws a decoder error. Either way, it's not good
// and causes bugs upstream. So, let's take the dropping into our own hands.
const nalUnits = extractHevcNalUnits(packet.data, this.decoderConfig);
const hasRaslPicture = nalUnits.some((x) => {
const type = extractNalUnitTypeForHevc(x);
return type === HevcNalUnitType.RASL_N || type === HevcNalUnitType.RASL_R;
});
if (hasRaslPicture) {
if (this.hasHevcRaslPicture(packet.data)) {
return; // Drop
}
@@ -934,15 +922,218 @@ class VideoDecoderWrapper extends DecoderWrapper<VideoSample> {
}
this.decoder.decode(packet.toEncodedVideoChunk());
this.decodeAlphaData(packet);
}
}
decodeAlphaData(packet: EncodedPacket) {
if (!packet.sideData.alpha || this.mergerCreationFailed) {
// No alpha side data in the packet, most common case
this.pushNullAlphaFrame();
return;
}
if (!this.merger) {
try {
this.merger = new ColorAlphaMerger();
} catch (error) {
console.error('Due to an error, only color data will be decoded.', error);
this.mergerCreationFailed = true;
this.decodeAlphaData(packet); // Go again
return;
}
}
// Check if we need to set up the alpha decoder
if (!this.alphaDecoder) {
const alphaHandler = (frame: VideoFrame) => {
this.alphaDecoderQueueSize--;
if (this.colorQueue.length > 0) {
const colorFrame = this.colorQueue.shift();
assert(colorFrame !== undefined);
this.mergeAlpha(colorFrame, frame);
} else {
this.alphaQueue.push(frame);
}
// Check if any null frames have been queued for this point
this.decodedAlphaChunkCount++;
while (
this.nullAlphaFrameQueue.length > 0
&& this.nullAlphaFrameQueue[0] === this.decodedAlphaChunkCount
) {
this.nullAlphaFrameQueue.shift();
if (this.colorQueue.length > 0) {
const colorFrame = this.colorQueue.shift();
assert(colorFrame !== undefined);
this.mergeAlpha(colorFrame, null);
} else {
this.alphaQueue.push(null);
}
}
};
this.alphaDecoder = new VideoDecoder({
output: (frame) => {
try {
alphaHandler(frame);
} catch (error) {
this.onError(error as Error);
}
},
error: this.onError,
});
this.alphaDecoder.configure(this.decoderConfig);
}
const type = determineVideoPacketType(this.codec, this.decoderConfig, packet.sideData.alpha);
// Alpha packets might follow a different key frame rhythm than the main packets. Therefore, before we start
// decoding, we must first find a packet that's actually a key frame. Until then, we treat the image as opaque.
if (!this.alphaHadKeyframe) {
this.alphaHadKeyframe = type === 'key';
}
if (this.alphaHadKeyframe) {
// Same RASL skipping logic as for color, unlikely to be hit (since who uses HEVC with separate alpha??) but
// here for symmetry.
if (this.codec === 'hevc' && this.currentAlphaPacketIndex > 0 && !this.alphaRaslSkipped) {
if (this.hasHevcRaslPicture(packet.sideData.alpha)) {
this.pushNullAlphaFrame();
return;
}
this.alphaRaslSkipped = true;
}
this.currentAlphaPacketIndex++;
this.alphaDecoder.decode(packet.alphaToEncodedVideoChunk(type ?? packet.type));
this.alphaDecoderQueueSize++;
} else {
this.pushNullAlphaFrame();
}
}
pushNullAlphaFrame() {
if (this.alphaDecoderQueueSize === 0) {
// Easy
this.alphaQueue.push(null);
} else {
// There are still alpha chunks being decoded, so pushing `null` immediately would result in out-of-order
// data and be incorrect. Instead, we need to enqueue a "null frame" for when the current decoder workload
// has finished.
this.nullAlphaFrameQueue.push(this.decodedAlphaChunkCount + this.alphaDecoderQueueSize);
}
}
/**
* If we're using HEVC, we need to make sure to skip any RASL slices that follow a non-IDR key frame such as
* CRA_NUT. This is because RASL slices cannot be decoded without data before the CRA_NUT. Browsers behave
* differently here: Chromium drops the packets, Safari throws a decoder error. Either way, it's not good
* and causes bugs upstream. So, let's take the dropping into our own hands.
*/
hasHevcRaslPicture(packetData: Uint8Array) {
const nalUnits = extractHevcNalUnits(packetData, this.decoderConfig);
return nalUnits.some((x) => {
const type = extractNalUnitTypeForHevc(x);
return type === HevcNalUnitType.RASL_N || type === HevcNalUnitType.RASL_R;
});
}
/** Handler for the WebCodecs VideoDecoder for ironing out browser differences. */
sampleHandler(sample: VideoSample) {
if (isSafari()) {
// For correct B-frame handling, we don't just hand over the frames directly but instead add them to
// a queue, because we want to ensure frames are emitted in presentation order. We flush the queue
// each time we receive a frame with a timestamp larger than the highest we've seen so far, as we
// can sure that is not a B-frame. Typically, WebCodecs automatically guarantees that frames are
// emitted in presentation order, but Safari doesn't always follow this rule.
if (this.sampleQueue.length > 0 && (sample.timestamp >= last(this.sampleQueue)!.timestamp)) {
for (const sample of this.sampleQueue) {
this.finalizeAndEmitSample(sample);
}
this.sampleQueue.length = 0;
}
insertSorted(this.sampleQueue, sample, x => x.timestamp);
} else {
// Assign it the next earliest timestamp from the input. We do this because browsers, by spec, are
// required to emit decoded frames in presentation order *while* retaining the timestamp of their
// originating EncodedVideoChunk. For files with B-frames but no out-of-order timestamps (like a
// missing ctts box, for example), this causes a mismatch. We therefore fix the timestamps and
// ensure they are sorted by doing this.
const timestamp = this.inputTimestamps.shift();
// There's no way we'd have more decoded frames than encoded packets we passed in. Actually, the
// correspondence should be 1:1.
assert(timestamp !== undefined);
sample.setTimestamp(timestamp);
this.finalizeAndEmitSample(sample);
}
}
finalizeAndEmitSample(sample: VideoSample) {
// Round the timestamps to the time resolution
sample.setTimestamp(Math.round(sample.timestamp * this.timeResolution) / this.timeResolution);
sample.setDuration(Math.round(sample.duration * this.timeResolution) / this.timeResolution);
sample.setRotation(this.rotation);
this.onSample(sample);
}
mergeAlpha(color: VideoFrame, alpha: VideoFrame | null) {
if (!alpha) {
// Nothing needs to be merged
const finalSample = new VideoSample(color);
this.sampleHandler(finalSample);
return;
}
assert(this.merger);
this.merger.update(color, alpha);
color.close();
alpha.close();
const finalFrame = new VideoFrame(this.merger.canvas, {
timestamp: color.timestamp,
duration: color.duration ?? undefined,
});
const finalSample = new VideoSample(finalFrame);
this.sampleHandler(finalSample);
}
async flush() {
if (this.customDecoder) {
await this.customDecoderCallSerializer.call(() => this.customDecoder!.flush());
} else {
assert(this.decoder);
await this.decoder.flush();
await Promise.all([
this.decoder.flush(),
this.alphaDecoder?.flush(),
]);
this.colorQueue.forEach(x => x.close());
this.colorQueue.length = 0;
this.alphaQueue.forEach(x => x?.close());
this.alphaQueue.length = 0;
this.alphaHadKeyframe = false;
this.decodedAlphaChunkCount = 0;
this.alphaDecoderQueueSize = 0;
this.nullAlphaFrameQueue.length = 0;
this.currentAlphaPacketIndex = 0;
this.alphaRaslSkipped = false;
}
if (isSafari()) {
@@ -963,6 +1154,14 @@ class VideoDecoderWrapper extends DecoderWrapper<VideoSample> {
} else {
assert(this.decoder);
this.decoder.close();
this.alphaDecoder?.close();
this.colorQueue.forEach(x => x.close());
this.colorQueue.length = 0;
this.alphaQueue.forEach(x => x?.close());
this.alphaQueue.length = 0;
this.merger?.close();
}
for (const sample of this.sampleQueue) {
@@ -972,6 +1171,150 @@ class VideoDecoderWrapper extends DecoderWrapper<VideoSample> {
}
}
/** Utility class that merges together color and alpha information using simple WebGL 2 shaders. */
class ColorAlphaMerger {
canvas: OffscreenCanvas | HTMLCanvasElement;
private gl: WebGL2RenderingContext;
private program: WebGLProgram;
private vao: WebGLVertexArrayObject;
private colorTexture: WebGLTexture;
private alphaTexture: WebGLTexture;
constructor() {
// Canvas will be resized later
if (typeof OffscreenCanvas !== 'undefined') {
// Prefer OffscreenCanvas for Worker environments
this.canvas = new OffscreenCanvas(300, 150);
} else {
this.canvas = document.createElement('canvas');
}
const gl = this.canvas.getContext('webgl2', {
premultipliedAlpha: false,
}) as unknown as WebGL2RenderingContext | null; // Casting because of some TypeScript weirdness
if (!gl) {
throw new Error('Couldn\'t acquire WebGL 2 context.');
}
this.gl = gl;
this.program = this.createProgram();
this.vao = this.createVAO();
this.colorTexture = this.createTexture();
this.alphaTexture = this.createTexture();
this.gl.useProgram(this.program);
this.gl.uniform1i(this.gl.getUniformLocation(this.program, 'u_colorTexture'), 0);
this.gl.uniform1i(this.gl.getUniformLocation(this.program, 'u_alphaTexture'), 1);
}
private createProgram(): WebGLProgram {
const vertexShader = this.createShader(this.gl.VERTEX_SHADER, `#version 300 es
in vec2 a_position;
in vec2 a_texCoord;
out vec2 v_texCoord;
void main() {
gl_Position = vec4(a_position, 0.0, 1.0);
v_texCoord = a_texCoord;
}
`);
const fragmentShader = this.createShader(this.gl.FRAGMENT_SHADER, `#version 300 es
precision highp float;
uniform sampler2D u_colorTexture;
uniform sampler2D u_alphaTexture;
in vec2 v_texCoord;
out vec4 fragColor;
void main() {
vec3 color = texture(u_colorTexture, v_texCoord).rgb;
float alpha = texture(u_alphaTexture, v_texCoord).r;
fragColor = vec4(color, alpha);
}
`);
const program = this.gl.createProgram();
this.gl.attachShader(program, vertexShader);
this.gl.attachShader(program, fragmentShader);
this.gl.linkProgram(program);
return program;
}
private createShader(type: number, source: string): WebGLShader {
const shader = this.gl.createShader(type)!;
this.gl.shaderSource(shader, source);
this.gl.compileShader(shader);
return shader;
}
private createVAO(): WebGLVertexArrayObject {
const vao = this.gl.createVertexArray();
this.gl.bindVertexArray(vao);
const vertices = new Float32Array([
-1, -1, 0, 1,
1, -1, 1, 1,
-1, 1, 0, 0,
1, 1, 1, 0,
]);
const buffer = this.gl.createBuffer();
this.gl.bindBuffer(this.gl.ARRAY_BUFFER, buffer);
this.gl.bufferData(this.gl.ARRAY_BUFFER, vertices, this.gl.STATIC_DRAW);
const positionLocation = this.gl.getAttribLocation(this.program, 'a_position');
const texCoordLocation = this.gl.getAttribLocation(this.program, 'a_texCoord');
this.gl.enableVertexAttribArray(positionLocation);
this.gl.vertexAttribPointer(positionLocation, 2, this.gl.FLOAT, false, 16, 0);
this.gl.enableVertexAttribArray(texCoordLocation);
this.gl.vertexAttribPointer(texCoordLocation, 2, this.gl.FLOAT, false, 16, 8);
return vao;
}
private createTexture(): WebGLTexture {
const texture = this.gl.createTexture();
this.gl.bindTexture(this.gl.TEXTURE_2D, texture);
this.gl.texParameteri(this.gl.TEXTURE_2D, this.gl.TEXTURE_WRAP_S, this.gl.CLAMP_TO_EDGE);
this.gl.texParameteri(this.gl.TEXTURE_2D, this.gl.TEXTURE_WRAP_T, this.gl.CLAMP_TO_EDGE);
this.gl.texParameteri(this.gl.TEXTURE_2D, this.gl.TEXTURE_MIN_FILTER, this.gl.LINEAR);
this.gl.texParameteri(this.gl.TEXTURE_2D, this.gl.TEXTURE_MAG_FILTER, this.gl.LINEAR);
return texture;
}
update(color: VideoFrame, alpha: VideoFrame): void {
if (color.displayWidth !== this.canvas.width || color.displayHeight !== this.canvas.height) {
this.canvas.width = color.displayWidth;
this.canvas.height = color.displayHeight;
}
this.gl.activeTexture(this.gl.TEXTURE0);
this.gl.bindTexture(this.gl.TEXTURE_2D, this.colorTexture);
this.gl.texImage2D(this.gl.TEXTURE_2D, 0, this.gl.RGBA, this.gl.RGBA, this.gl.UNSIGNED_BYTE, color);
this.gl.activeTexture(this.gl.TEXTURE1);
this.gl.bindTexture(this.gl.TEXTURE_2D, this.alphaTexture);
this.gl.texImage2D(this.gl.TEXTURE_2D, 0, this.gl.RGBA, this.gl.RGBA, this.gl.UNSIGNED_BYTE, alpha);
this.gl.viewport(0, 0, this.canvas.width, this.canvas.height);
this.gl.clear(this.gl.COLOR_BUFFER_BIT);
this.gl.bindVertexArray(this.vao);
this.gl.drawArrays(this.gl.TRIANGLE_STRIP, 0, 4);
}
close() {
this.gl.getExtension('WEBGL_lose_context')?.loseContext();
this.gl = null as unknown as WebGL2RenderingContext;
}
}
/**
* A sink that retrieves decoded video samples (video frames) from a video track.
* @group Media sinks
@@ -995,7 +1338,7 @@ export class VideoSampleSink extends BaseMediaSampleSink<VideoSample> {
/** @internal */
async _createDecoder(
onSample: (sample: VideoSample) => unknown,
onError: (error: DOMException) => unknown,
onError: (error: Error) => unknown,
) {
if (!(await this._track.canDecode())) {
throw new Error(
@@ -1078,6 +1421,11 @@ export type WrappedCanvas = {
* @public
*/
export type CanvasSinkOptions = {
/**
* Whether the output canvases should have transparency instead of a black background. Defaults to `false`. Set
* this to `true` when using this sink to read transparent videos.
*/
alpha?: boolean;
/**
* The width of the output canvas in pixels, defaulting to the display width of the video track. If height is not
* set, it will be deduced automatically based on aspect ratio.
@@ -1130,6 +1478,8 @@ export class CanvasSink {
/** @internal */
_videoTrack: InputVideoTrack;
/** @internal */
_alpha: boolean;
/** @internal */
_width: number;
/** @internal */
_height: number;
@@ -1154,6 +1504,9 @@ export class CanvasSink {
if (options && typeof options !== 'object') {
throw new TypeError('options must be an object.');
}
if (options.alpha !== undefined && typeof options.alpha !== 'boolean') {
throw new TypeError('options.alpha, when provided, must be a boolean.');
}
if (options.width !== undefined && (!Number.isInteger(options.width) || options.width <= 0)) {
throw new TypeError('options.width, when defined, must be a positive integer.');
}
@@ -1214,6 +1567,7 @@ export class CanvasSink {
}
this._videoTrack = videoTrack;
this._alpha = options.alpha ?? false;
this._width = width;
this._height = height;
this._rotation = rotation;
@@ -1250,14 +1604,14 @@ export class CanvasSink {
}
const context = canvas.getContext('2d', {
alpha: isFirefox(), // Firefox has VideoFrame glitches with opaque canvases
alpha: this._alpha || isFirefox(), // Firefox has VideoFrame glitches with opaque canvases
}) as CanvasRenderingContext2D | OffscreenCanvasRenderingContext2D;
assert(context);
context.resetTransform();
if (!canvasIsNew) {
if (isFirefox()) {
if (!this._alpha && isFirefox()) {
context.fillStyle = 'black';
context.fillRect(0, 0, this._width, this._height);
} else {
@@ -1338,7 +1692,7 @@ class AudioDecoderWrapper extends DecoderWrapper<AudioSample> {
constructor(
onSample: (sample: AudioSample) => unknown,
onError: (error: DOMException) => unknown,
onError: (error: Error) => unknown,
codec: AudioCodec,
decoderConfig: AudioDecoderConfig,
) {
@@ -1390,7 +1744,13 @@ class AudioDecoderWrapper extends DecoderWrapper<AudioSample> {
void this.customDecoderCallSerializer.call(() => this.customDecoder!.init());
} else {
this.decoder = new AudioDecoder({
output: data => sampleHandler(new AudioSample(data)),
output: (data) => {
try {
sampleHandler(new AudioSample(data));
} catch (error) {
this.onError(error as Error);
}
},
error: onError,
});
this.decoder.configure(decoderConfig);
@@ -1455,7 +1815,7 @@ class PcmAudioDecoderWrapper extends DecoderWrapper<AudioSample> {
constructor(
onSample: (sample: AudioSample) => unknown,
onError: (error: DOMException) => unknown,
onError: (error: Error) => unknown,
public decoderConfig: AudioDecoderConfig,
) {
super(onSample, onError);
@@ -1645,7 +2005,7 @@ export class AudioSampleSink extends BaseMediaSampleSink<AudioSample> {
/** @internal */
async _createDecoder(
onSample: (sample: AudioSample) => unknown,
onError: (error: DOMException) => unknown,
onError: (error: Error) => unknown,
) {
if (!(await this._track.canDecode())) {
throw new Error(