Add WebVTT subtitle support to MP4 & a bunch of misc changes

This commit is contained in:
Vanilagy
2024-11-24 14:57:31 +01:00
parent 39aea03565
commit 29ecccb26a
15 changed files with 2355 additions and 1557 deletions
+55 -5
View File
@@ -36,7 +36,8 @@
canvas.height = 720;
const context = canvas.getContext('2d');
let format = new Metamuxer.WebMOutputFormat();// new Metamuxer.MkvOutputFormat();// new Metamuxer.Mp4OutputFormat({ fastStart: false });
let format = new Metamuxer.WebMOutputFormat();
format = new Metamuxer.Mp4OutputFormat({ fastStart: false }); // new Metamuxer.MkvOutputFormat();// new Metamuxer.Mp4OutputFormat({ fastStart: false });
let target = new Metamuxer.ArrayBufferTarget();
let output = new Metamuxer.Output({ format, target });
@@ -71,11 +72,11 @@
*/
let videoSource = new Metamuxer.CanvasSource(canvas, {
codec: 'vp9',
codec: 'avc',
bitrate: 1e6
});
let audioSource = new Metamuxer.AudioBufferSource({
codec: 'opus',
codec: 'aac',
bitrate: 128e3,
});
let subtitleSource = new Metamuxer.TextSubtitleSource('webvtt');
@@ -89,9 +90,58 @@
let simpleWebvttFile =
`WEBVTT
00:00:00.000 --> 00:00:10.000
1
00:00:02.000 --> 00:00:10.000
Example entry 1: Hello <b>world</b>.
`;
if (false) simpleWebvttFile =
`WEBVTT
1
00:05.000 --> 00:12.500 align:start line:10
<v Roger Bingham>We are in New York City.
We are looking straight down 5th Avenue.
00:13.000 --> 00:18.000
<v Neil DeGrass Tyson>Didn't you already say that?
2
00:17.000 --> 00:20.000
Testing... <00:17.350>One... <00:18.125>Two...
`
simpleWebvttFile =
`WEBVTT
00:00:01.000 --> 00:00:01.500 align:start line:1
1. <b>justify (top, left)</b>,
00:00:01.500 --> 00:00:02.000 align:center line:1.5
2. <b>justify (top, center)</b>,
00:00:02.000 --> 00:00:02.500 align:end line:2
3. <b>justify (top, right)</b>,
00:00:02.500 --> 00:00:03.000 align:start line:48%
4. <b>justify (center, left)</b>,
00:00:03.000 --> 00:00:03.500 align:center line:50%
5. <b>justify (center, center)</b>,
00:00:03.500 --> 00:00:04.000 align:end line:52%
6. <b>justify (center, right)</b>,
00:00:04.000 --> 00:00:04.500 align:start line:-1
7. <b>justify (bottom, left)</b>,
00:00:04.500 --> 00:00:05.000 align:center line:-1.5
8. <b>justify (bottom, center)</b>,
00:00:05.000 --> 00:00:05.500 align:end line:-2
9. <b>justify (bottom, right)</b>.
`;
subtitleSource.digest(simpleWebvttFile);
subtitleSource.close();
@@ -112,5 +162,5 @@ Example entry 1: Hello <b>world</b>.
await output.finalize();
console.log(target);
download(new Blob([target.buffer]), 'test.webm');
download(new Blob([target.buffer]), 'test.mp4');
</script>
+810 -592
View File
File diff suppressed because it is too large Load Diff
+7 -7
View File
File diff suppressed because one or more lines are too long
+7 -7
View File
File diff suppressed because one or more lines are too long
+810 -592
View File
File diff suppressed because it is too large Load Diff
+188 -20
View File
@@ -1,6 +1,97 @@
import { toUint8Array, assert, isU32, last, TransformationMatrix } from '../misc';
import { AudioCodec, AudioSource, VideoCodec, VideoSource } from '../source';
import { GLOBAL_TIMESCALE, intoTimescale, IsobmffAudioTrackData, IsobmffTrackData, IsobmffVideoTrackData, Sample } from './isobmff_muxer';
import { toUint8Array, assert, isU32, last, TransformationMatrix, textEncoder } from '../misc';
import { AudioCodec, AudioSource, SubtitleCodec, VideoCodec, VideoSource } from '../source';
import { formatSubtitleTimestamp } from '../subtitles';
import { Writer } from '../writer';
import { GLOBAL_TIMESCALE, intoTimescale, IsobmffAudioTrackData, IsobmffSubtitleTrackData, IsobmffTrackData, IsobmffVideoTrackData, Sample } from './isobmff_muxer';
export class IsobmffBoxWriter {
private helper = new Uint8Array(8);
private helperView = new DataView(this.helper.buffer);
/**
* Stores the position from the start of the file to where boxes elements have been written. This is used to
* rewrite/edit elements that were already added before, and to measure sizes of things.
*/
offsets = new WeakMap<Box, number>();
constructor(private writer: Writer) {}
writeU32(value: number) {
this.helperView.setUint32(0, value, false);
this.writer.write(this.helper.subarray(0, 4));
}
writeU64(value: number) {
this.helperView.setUint32(0, Math.floor(value / 2**32), false);
this.helperView.setUint32(4, value, false);
this.writer.write(this.helper.subarray(0, 8));
}
writeAscii(text: string) {
for (let i = 0; i < text.length; i++) {
this.helperView.setUint8(i % 8, text.charCodeAt(i));
if (i % 8 === 7) this.writer.write(this.helper);
}
if (text.length % 8 !== 0) {
this.writer.write(this.helper.subarray(0, text.length % 8));
}
}
writeBox(box: Box) {
this.offsets.set(box, this.writer.getPos());
if (box.contents && !box.children) {
this.writeBoxHeader(box, box.size ?? box.contents.byteLength + 8);
this.writer.write(box.contents);
} else {
let startPos = this.writer.getPos();
this.writeBoxHeader(box, 0);
if (box.contents) this.writer.write(box.contents);
if (box.children) for (let child of box.children) if (child) this.writeBox(child);
let endPos = this.writer.getPos();
let size = box.size ?? endPos - startPos;
this.writer.seek(startPos);
this.writeBoxHeader(box, size);
this.writer.seek(endPos);
}
}
writeBoxHeader(box: Box, size: number) {
this.writeU32(box.largeSize ? 1 : size);
this.writeAscii(box.type);
if (box.largeSize) this.writeU64(size);
}
measureBoxHeader(box: Box) {
return 8 + (box.largeSize ? 8 : 0);
}
patchBox(box: Box) {
const boxOffset = this.offsets.get(box);
assert(boxOffset !== undefined);
let endPos = this.writer.getPos();
this.writer.seek(boxOffset);
this.writeBox(box);
this.writer.seek(endPos);
}
measureBox(box: Box) {
if (box.contents && !box.children) {
let headerSize = this.measureBoxHeader(box);
return headerSize + box.contents.byteLength;
} else {
let result = this.measureBoxHeader(box);
if (box.contents) result += box.contents.byteLength;
if (box.children) for (let child of box.children) if (child) result += this.measureBox(child);
return result;
}
}
}
let bytes = new Uint8Array(8);
let view = new DataView(bytes.buffer);
@@ -248,7 +339,7 @@ export const tkhd = (
u32OrU64(durationInGlobalTimescale), // Duration
Array(8).fill(0), // Reserved
u16(0), // Layer
u16(0), // Alternate group
u16(trackData.track.id), // Alternate group
fixed_8_8(trackData.type === 'audio' ? 1 : 0), // Volume
u16(0), // Reserved
matrixToBytes(matrix), // Matrix
@@ -260,7 +351,7 @@ export const tkhd = (
/** Media Box: Describes and define a track's media type and sample data. */
export const mdia = (trackData: IsobmffTrackData, creationTime: number) => box('mdia', undefined, [
mdhd(trackData, creationTime),
hdlr(trackData.type === 'video' ? 'vide' : 'soun'),
hdlr(trackData),
minf(trackData)
]);
@@ -288,15 +379,26 @@ export const mdhd = (
]);
};
const TRACK_TYPE_TO_COMPONENT_SUBTYPE: Record<IsobmffTrackData['type'], string> = {
video: 'vide',
audio: 'soun',
subtitle: 'text'
};
const TRACK_TYPE_TO_HANDLER_NAME: Record<IsobmffTrackData['type'], string> = {
video: 'VideoHandler',
audio: 'SoundHandler',
subtitle: 'TextHandler'
};
/** Handler Reference Box: Specifies the media handler component that is to be used to interpret the media's data. */
export const hdlr = (componentSubtype: string) => fullBox('hdlr', 0, 0, [
export const hdlr = (trackData: IsobmffTrackData) => fullBox('hdlr', 0, 0, [
ascii('mhlr'), // Component type
ascii(componentSubtype), // Component subtype
ascii(TRACK_TYPE_TO_COMPONENT_SUBTYPE[trackData.type]), // Component subtype
u32(0), // Component manufacturer
u32(0), // Component flags
u32(0), // Component flags mask
// TODO:
ascii('mp4-muxer-hdlr', true) // Component name
ascii(TRACK_TYPE_TO_HANDLER_NAME[trackData.type], true) // Component name
]);
/**
@@ -304,7 +406,7 @@ export const hdlr = (componentSubtype: string) => fullBox('hdlr', 0, 0, [
* information to map from media time to media data and to process the media data.
*/
export const minf = (trackData: IsobmffTrackData) => box('minf', undefined, [
trackData.type === 'video' ? vmhd() : smhd(),
TRACK_TYPE_TO_HEADER_BOX[trackData.type](),
dinf(),
stbl(trackData)
]);
@@ -323,6 +425,15 @@ export const smhd = () => fullBox('smhd', 0, 0, [
u16(0) // Reserved
]);
/** Null Media Header Box. */
export const nmhd = () => fullBox('nmhd', 0, 0);
const TRACK_TYPE_TO_HEADER_BOX: Record<IsobmffTrackData['type'], () => Box> = {
video: vmhd,
audio: smhd,
subtitle: nmhd
};
/**
* Data Information Box: Contains information specifying the data handler component that provides access to the
* media data. The data handler component uses the Data Information Box to interpret the media's data.
@@ -365,19 +476,34 @@ export const stbl = (trackData: IsobmffTrackData) => {
* Sample Description Box: Stores information that allows you to decode samples in the media. The data stored in the
* sample description varies, depending on the media type.
*/
export const stsd = (trackData: IsobmffTrackData) => fullBox('stsd', 0, 0, [
u32(1) // Entry count
], [
trackData.type === 'video'
? videoSampleDescription(
export const stsd = (trackData: IsobmffTrackData) => {
let sampleDescription: Box;
if (trackData.type === 'video') {
sampleDescription = videoSampleDescription(
VIDEO_CODEC_TO_BOX_NAME[trackData.track.source.codec],
trackData as IsobmffVideoTrackData
trackData
)
: soundSampleDescription(
} else if (trackData.type === 'audio') {
sampleDescription = soundSampleDescription(
AUDIO_CODEC_TO_BOX_NAME[trackData.track.source.codec],
trackData as IsobmffAudioTrackData
)
]);
trackData
);
} else if (trackData.type === 'subtitle') {
sampleDescription = subtitleSampleDescription(
SUBTITLE_CODEC_TO_BOX_NAME[trackData.track.source.codec],
trackData
);
}
assert(sampleDescription!);
return fullBox('stsd', 0, 0, [
u32(1) // Entry count
], [
sampleDescription
]);
};
/** Video Sample Description Box: Contains information that defines how to interpret video media data. */
export const videoSampleDescription = (
@@ -400,6 +526,7 @@ export const videoSampleDescription = (
i16(0xffff) // Pre-defined
], [
VIDEO_CODEC_TO_CONFIGURATION_BOX[trackData.track.source.codec](trackData)
// TODO colr
]);
// TODO: All muxers should ensure that the decoder config description is provided for the codecs that require it. This
@@ -550,6 +677,24 @@ export const dOps = (trackData: IsobmffAudioTrackData) => {
]);
};
export const subtitleSampleDescription = (
compressionType: string,
trackData: IsobmffSubtitleTrackData
) => box(compressionType, [
Array(6).fill(0), // Reserved
u16(1), // Data reference index
], [
SUBTITLE_CODEC_TO_CONFIGURATION_BOX[trackData.track.source.codec](trackData)
])
export const vttC = (trackData: IsobmffSubtitleTrackData) => box('vttC', [
...textEncoder.encode(trackData.info.config.description)
]);
export const txtC = (textConfig: Uint8Array) => fullBox('txtC', 0, 0, [
...textConfig, 0 // Text config (null-terminated)
]);
/**
* Time-To-Sample Box: Stores duration information for a media's samples, providing a mapping from a time in a media
* to the corresponding data sample. The table is compact, meaning that consecutive samples with the same time delta
@@ -817,6 +962,21 @@ export const mfro = () => {
]);
};
/** VTT Empty Cue Box */
export const vtte = () => box('vtte');
/** VTT Cue Box */
export const vttc = (payload: string, timestamp: number | null, identifier: string | null, settings: string | null, sourceId: number | null) => box('vttc', undefined, [
sourceId !== null ? box('vsid', [i32(sourceId)]) : null,
identifier !== null ? box('iden', [...textEncoder.encode(identifier)]) : null,
timestamp !== null ? box('ctim', [...textEncoder.encode(formatSubtitleTimestamp(timestamp))]) : null,
settings !== null ? box('sttg', [...textEncoder.encode(settings)]) : null,
box('payl', [...textEncoder.encode(payload)])
]);
/** VTT Additional Text Box */
export const vtta = (notes: string) => box('vtta', [...textEncoder.encode(notes)]);
const VIDEO_CODEC_TO_BOX_NAME: Record<VideoCodec, string> = {
'avc': 'avc1',
'hevc': 'hvc1',
@@ -842,3 +1002,11 @@ const AUDIO_CODEC_TO_CONFIGURATION_BOX: Record<AudioCodec, (trackData: IsobmffAu
'aac': esds,
'opus': dOps
};
const SUBTITLE_CODEC_TO_BOX_NAME: Record<SubtitleCodec, string> = {
'webvtt': 'wvtt'
};
const SUBTITLE_CODEC_TO_CONFIGURATION_BOX: Record<SubtitleCodec, (trackData: IsobmffSubtitleTrackData) => Box | null> = {
'webvtt': vttC
};
+276 -175
View File
@@ -1,9 +1,11 @@
import { Box, free, ftyp, mdat, mfra, moof, moov } from './isobmff_boxes';
import { Box, free, ftyp, IsobmffBoxWriter, mdat, mfra, moof, moov, vtta, vttc, vtte } from './isobmff_boxes';
import { Muxer } from '../muxer';
import { Output, OutputAudioTrack, OutputTrack, OutputVideoTrack } from '../output';
import { Output, OutputAudioTrack, OutputSubtitleTrack, OutputTrack, OutputVideoTrack } from '../output';
import { Writer } from '../writer';
import { assert, last, TransformationMatrix } from '../misc';
import { Mp4OutputFormat } from '../output_format';
import { inlineTimestampRegex, SubtitleConfig, SubtitleCue, SubtitleMetadata } from '../subtitles';
import { ArrayBufferTarget } from '../target';
export const GLOBAL_TIMESCALE = 1000;
const TIMESTAMP_OFFSET = 2_082_844_800; // Seconds between Jan 1 1904 and Jan 1 1970
@@ -32,9 +34,6 @@ export type IsobmffTrackData = {
sampleQueue: Sample[], // For fragmented files
timestampProcessingQueue: Sample[],
firstTimestamp: number | null,
lastKeyFrameTimestamp: number | null,
timeToSampleTable: { sampleCount: number, sampleDelta: number }[];
compositionTimeOffsetTable: { sampleCount: number, sampleCompositionTimeOffset: number }[];
lastTimescaleUnits: number | null,
@@ -62,10 +61,21 @@ export type IsobmffTrackData = {
sampleRate: number,
decoderConfig: AudioDecoderConfig
}
} | {
track: OutputSubtitleTrack,
type: 'subtitle',
info: {
config: SubtitleConfig
},
lastCueEndTimestamp: number,
cueQueue: SubtitleCue[],
nextSourceId: number,
cueToSourceId: WeakMap<SubtitleCue, number>
});
export type IsobmffVideoTrackData = IsobmffTrackData & { type: 'video' };
export type IsobmffAudioTrackData = IsobmffTrackData & { type: 'audio' };
export type IsobmffSubtitleTrackData = IsobmffTrackData & { type: 'subtitle' };
export const intoTimescale = (timeInSeconds: number, timescale: number, round = true) => {
let value = timeInSeconds * timescale;
@@ -73,16 +83,15 @@ export const intoTimescale = (timeInSeconds: number, timescale: number, round =
};
export class IsobmffMuxer extends Muxer {
#writer: Writer;
#format: Mp4OutputFormat;
#helper = new Uint8Array(8);
#helperView = new DataView(this.#helper.buffer);
override timestampsMustStartAtZero = true;
/**
* Stores the position from the start of the file to where boxes elements have been written. This is used to
* rewrite/edit elements that were already added before, and to measure sizes of things.
*/
offsets = new WeakMap<Box, number>();
#writer: Writer;
#boxWriter: IsobmffBoxWriter;
#format: Mp4OutputFormat;
#auxTarget = new ArrayBufferTarget();
#auxWriter = this.#auxTarget.createWriter();
#auxBoxWriter = new IsobmffBoxWriter(this.#auxWriter);
#ftypSize: number | null = null;
#mdat: Box | null = null;
@@ -98,90 +107,15 @@ export class IsobmffMuxer extends Muxer {
super(output);
this.#writer = output.writer;
this.#boxWriter = new IsobmffBoxWriter(this.#writer);
this.#format = format;
}
writeU32(value: number) {
this.#helperView.setUint32(0, value, false);
this.#writer.write(this.#helper.subarray(0, 4));
}
writeU64(value: number) {
this.#helperView.setUint32(0, Math.floor(value / 2**32), false);
this.#helperView.setUint32(4, value, false);
this.#writer.write(this.#helper.subarray(0, 8));
}
writeAscii(text: string) {
for (let i = 0; i < text.length; i++) {
this.#helperView.setUint8(i % 8, text.charCodeAt(i));
if (i % 8 === 7) this.#writer.write(this.#helper);
}
if (text.length % 8 !== 0) {
this.#writer.write(this.#helper.subarray(0, text.length % 8));
}
}
writeBox(box: Box) {
this.offsets.set(box, this.#writer.getPos());
if (box.contents && !box.children) {
this.writeBoxHeader(box, box.size ?? box.contents.byteLength + 8);
this.#writer.write(box.contents);
} else {
let startPos = this.#writer.getPos();
this.writeBoxHeader(box, 0);
if (box.contents) this.#writer.write(box.contents);
if (box.children) for (let child of box.children) if (child) this.writeBox(child);
let endPos = this.#writer.getPos();
let size = box.size ?? endPos - startPos;
this.#writer.seek(startPos);
this.writeBoxHeader(box, size);
this.#writer.seek(endPos);
}
}
writeBoxHeader(box: Box, size: number) {
this.writeU32(box.largeSize ? 1 : size);
this.writeAscii(box.type);
if (box.largeSize) this.writeU64(size);
}
measureBoxHeader(box: Box) {
return 8 + (box.largeSize ? 8 : 0);
}
patchBox(box: Box) {
const boxOffset = this.offsets.get(box);
assert(boxOffset !== undefined);
let endPos = this.#writer.getPos();
this.#writer.seek(boxOffset);
this.writeBox(box);
this.#writer.seek(endPos);
}
measureBox(box: Box) {
if (box.contents && !box.children) {
let headerSize = this.measureBoxHeader(box);
return headerSize + box.contents.byteLength;
} else {
let result = this.measureBoxHeader(box);
if (box.contents) result += box.contents.byteLength;
if (box.children) for (let child of box.children) if (child) result += this.measureBox(child);
return result;
}
}
start() {
const holdsAvc = this.output.tracks.some(x => x.type === 'video' && x.source.codec === 'avc');
// Write the header
this.writeBox(ftyp({
this.#boxWriter.writeBox(ftyp({
holdsAvc: holdsAvc,
fragmented: this.#format.options.fastStart === 'fragmented'
}));
@@ -199,7 +133,7 @@ export class IsobmffMuxer extends Muxer {
}
this.#mdat = mdat(true); // Reserve large size by default, can refine this when finalizing.
this.writeBox(this.#mdat);
this.#boxWriter.writeBox(this.#mdat);
}
this.#writer.flush();
@@ -240,7 +174,7 @@ export class IsobmffMuxer extends Muxer {
#getVideoTrackData(track: OutputVideoTrack, meta?: EncodedVideoChunkMetadata) {
const existingTrackData = this.#trackDatas.find(x => x.track === track);
if (existingTrackData) {
return existingTrackData;
return existingTrackData as IsobmffVideoTrackData;
}
// TODO Make proper errors for these
@@ -249,7 +183,7 @@ export class IsobmffMuxer extends Muxer {
assert(meta.decoderConfig.codedWidth !== undefined);
assert(meta.decoderConfig.codedHeight !== undefined);
const newTrackData: IsobmffTrackData = {
const newTrackData: IsobmffVideoTrackData = {
track,
type: 'video',
info: {
@@ -261,8 +195,6 @@ export class IsobmffMuxer extends Muxer {
samples: [],
sampleQueue: [],
timestampProcessingQueue: [],
firstTimestamp: null,
lastKeyFrameTimestamp: null,
timeToSampleTable: [],
compositionTimeOffsetTable: [],
lastTimescaleUnits: null,
@@ -281,14 +213,14 @@ export class IsobmffMuxer extends Muxer {
#getAudioTrackData(track: OutputAudioTrack, meta?: EncodedAudioChunkMetadata) {
const existingTrackData = this.#trackDatas.find(x => x.track === track);
if (existingTrackData) {
return existingTrackData;
return existingTrackData as IsobmffAudioTrackData;
}
// TODO Make proper errors for these
assert(meta);
assert(meta.decoderConfig);
const newTrackData: IsobmffTrackData = {
const newTrackData: IsobmffAudioTrackData = {
track,
type: 'audio',
info: {
@@ -300,8 +232,6 @@ export class IsobmffMuxer extends Muxer {
samples: [],
sampleQueue: [],
timestampProcessingQueue: [],
firstTimestamp: null,
lastKeyFrameTimestamp: null,
timeToSampleTable: [],
compositionTimeOffsetTable: [],
lastTimescaleUnits: null,
@@ -317,6 +247,49 @@ export class IsobmffMuxer extends Muxer {
return newTrackData;
}
#getSubtitleTrackData(track: OutputSubtitleTrack, meta?: SubtitleMetadata) {
const existingTrackData = this.#trackDatas.find(x => x.track === track);
if (existingTrackData) {
return existingTrackData as IsobmffSubtitleTrackData;
}
// TODO Make proper errors for these
assert(meta);
assert(meta.config);
const newTrackData: IsobmffSubtitleTrackData = {
track,
type: 'subtitle',
info: {
config: meta.config
},
timescale: 1000, // Reasonable
samples: [],
sampleQueue: [],
timestampProcessingQueue: [],
timeToSampleTable: [],
compositionTimeOffsetTable: [],
lastTimescaleUnits: null,
lastSample: null,
finalizedChunks: [],
currentChunk: null,
compactlyCodedChunkTable: [],
lastCueEndTimestamp: 0,
cueQueue: [],
nextSourceId: 0,
cueToSourceId: new WeakMap()
};
this.#trackDatas.push(newTrackData);
this.#trackDatas.sort((a, b) => a.track.id - b.track.id);
// Subtitle cues don't need to start at 0, so let's register a timestamp at 0 to satisfy the
// "timestamps must start a zero" constraint
this.validateAndNormalizeTimestamp(track, 0, true);
return newTrackData;
}
addEncodedVideoChunk(track: OutputVideoTrack, chunk: EncodedVideoChunk, meta?: EncodedVideoChunkMetadata) {
const trackData = this.#getVideoTrackData(track, meta);
@@ -330,13 +303,17 @@ export class IsobmffMuxer extends Muxer {
}).`);
}
let videoSample = this.#createSampleForTrack(trackData, chunk);
let data = new Uint8Array(chunk.byteLength);
chunk.copyTo(data);
let timestamp = this.validateAndNormalizeTimestamp(trackData.track, chunk.timestamp, chunk.type === 'key');
let sample = this.#createSampleForTrack(trackData, data, timestamp, (chunk.duration ?? 0) / 1e6, chunk.type);
if (this.#format.options.fastStart === 'fragmented') {
trackData.sampleQueue.push(videoSample);
trackData.sampleQueue.push(sample);
this.#interleaveSamples();
} else {
this.#addSampleToTrack(trackData, videoSample);
this.#addSampleToTrack(trackData, sample);
}
}
@@ -353,35 +330,146 @@ export class IsobmffMuxer extends Muxer {
}).`);
}
let audioSample = this.#createSampleForTrack(trackData, chunk);
let data = new Uint8Array(chunk.byteLength);
chunk.copyTo(data);
let timestamp = this.validateAndNormalizeTimestamp(trackData.track, chunk.timestamp, chunk.type === 'key');
let sample = this.#createSampleForTrack(trackData, data, timestamp, (chunk.duration ?? 0) / 1e6, chunk.type);
if (this.#format.options.fastStart === 'fragmented') {
trackData.sampleQueue.push(audioSample);
trackData.sampleQueue.push(sample);
this.#interleaveSamples();
} else {
this.#addSampleToTrack(trackData, audioSample);
this.#addSampleToTrack(trackData, sample);
}
}
addSubtitleCue(track: OutputSubtitleTrack, cue: SubtitleCue, meta?: SubtitleMetadata) {
const trackData = this.#getSubtitleTrackData(track, meta);
// TODO the expectedSubtitleChunks thing
this.validateAndNormalizeTimestamp(trackData.track, 1e6 * cue.timestamp, true);
if (track.source.codec === 'webvtt') {
trackData.cueQueue.push(cue);
this.#processWebVTTCues(trackData, cue.timestamp);
} else {
// TODO
}
}
#processWebVTTCues(trackData: IsobmffSubtitleTrackData, until: number) {
// WebVTT cues need to undergo special processing as empty sections need to be padded out with samples, and
// overlapping samples require special logic. The algorithm produces the format specified in ISO 14496-30.
while (trackData.cueQueue.length > 0) {
let timestamps = new Set<number>([]);
for (let cue of trackData.cueQueue) {
assert(cue.timestamp <= until);
assert(trackData.lastCueEndTimestamp <= cue.timestamp + cue.duration);
timestamps.add(Math.max(cue.timestamp, trackData.lastCueEndTimestamp)); // Start timestamp
timestamps.add(cue.timestamp + cue.duration); // End timestamp
}
let sortedTimestamps = [...timestamps].sort((a, b) => a - b);
// These are the timestamps of the next sample we'll create:
let sampleStart = sortedTimestamps[0]!;
let sampleEnd = sortedTimestamps[1] ?? sampleStart;
if (until < sampleEnd) {
break;
}
// We may need to pad out empty space with an vtte box
if (trackData.lastCueEndTimestamp < sampleStart) {
this.#auxWriter.seek(0);
let box = vtte();
this.#auxBoxWriter.writeBox(box);
let body = this.#auxWriter.getSlice(0, this.#auxWriter.getPos());
let sample = this.#createSampleForTrack(trackData, body, trackData.lastCueEndTimestamp, sampleStart - trackData.lastCueEndTimestamp, 'key');
// todo extract this into reusable thing
if (this.#format.options.fastStart === 'fragmented') {
trackData.sampleQueue.push(sample);
this.#interleaveSamples();
} else {
this.#addSampleToTrack(trackData, sample);
}
trackData.lastCueEndTimestamp = sampleStart;
}
this.#auxWriter.seek(0);
for (let i = 0; i < trackData.cueQueue.length; i++) {
let cue = trackData.cueQueue[i]!
if (cue.timestamp >= sampleEnd) {
break;
}
inlineTimestampRegex.lastIndex = 0;
let containsTimestamp = inlineTimestampRegex.test(cue.text);
let endTimestamp = cue.timestamp + cue.duration;
let sourceId = trackData.cueToSourceId.get(cue);
if (sourceId === undefined && sampleEnd < endTimestamp) {
// We know this cue will appear in more than one sample, therefore we need to mark it with a
// unique ID
sourceId = trackData.nextSourceId++;
trackData.cueToSourceId.set(cue, sourceId);
}
if (cue.notes) {
// Any notes/comments are included in a special vtta box
let box = vtta(cue.notes);
this.#auxBoxWriter.writeBox(box);
}
let box = vttc(cue.text, containsTimestamp ? sampleStart : null, cue.identifier ?? null, cue.settings ?? null, sourceId ?? null);
this.#auxBoxWriter.writeBox(box);
if (endTimestamp === sampleEnd) {
// The cue won't appear in any future sample, so we're done with it
trackData.cueQueue.splice(i--, 1);
}
}
let body = this.#auxWriter.getSlice(0, this.#auxWriter.getPos());
let sample = this.#createSampleForTrack(trackData, body, sampleStart, sampleEnd - sampleStart, 'key');
// todo extract this into reusable thing
if (this.#format.options.fastStart === 'fragmented') {
trackData.sampleQueue.push(sample);
this.#interleaveSamples();
} else {
this.#addSampleToTrack(trackData, sample);
}
trackData.lastCueEndTimestamp = sampleEnd;
}
}
#createSampleForTrack(
trackData: IsobmffTrackData,
chunk: EncodedVideoChunk | EncodedAudioChunk
data: Uint8Array,
timestamp: number,
duration: number,
type: 'key' | 'delta'
) {
let timestampInSeconds = this.#validateTimestamp(trackData, chunk);
let durationInSeconds = (chunk.duration ?? 0) / 1e6;
let data = new Uint8Array(chunk.byteLength);
chunk.copyTo(data);
let sample: Sample = {
timestamp: timestampInSeconds,
decodeTimestamp: timestampInSeconds, // We may refine this later
duration: durationInSeconds,
data: data,
timestamp,
decodeTimestamp: timestamp, // This may be refined later
duration,
data,
size: data.byteLength,
type: chunk.type,
type,
// Will be refined once the next sample comes in
timescaleUnitsToNextSample: intoTimescale(durationInSeconds, trackData.timescale)
timescaleUnitsToNextSample: intoTimescale(duration, trackData.timescale)
};
return sample;
@@ -405,6 +493,7 @@ export class IsobmffMuxer extends Muxer {
const sampleCompositionTimeOffset =
intoTimescale(sample.timestamp - sample.decodeTimestamp, trackData.timescale);
const durationInTimescale = intoTimescale(sample.duration, trackData.timescale);
if (trackData.lastTimescaleUnits !== null) {
assert(trackData.lastSample);
@@ -417,20 +506,34 @@ export class IsobmffMuxer extends Muxer {
if (this.#format.options.fastStart !== 'fragmented') {
let lastTableEntry = last(trackData.timeToSampleTable);
assert(lastTableEntry);
if (lastTableEntry.sampleCount === 1) {
// If we hit this case, we're the second sample
lastTableEntry.sampleDelta = delta;
lastTableEntry.sampleCount++;
} else if (lastTableEntry.sampleDelta === delta) {
// Simply increment the count
let entryBefore = trackData.timeToSampleTable[trackData.timeToSampleTable.length - 2];
if (entryBefore && entryBefore.sampleDelta === delta) {
// If the delta is the same as the previous one, merge the two entries
entryBefore.sampleCount++;
trackData.timeToSampleTable.pop();
lastTableEntry = entryBefore;
}
} else if (lastTableEntry.sampleDelta !== delta) {
// The delta has changed, so we need a new entry to reach the current sample
lastTableEntry.sampleCount--;
trackData.timeToSampleTable.push(lastTableEntry = {
sampleCount: 1,
sampleDelta: delta
});
}
if (lastTableEntry.sampleDelta === durationInTimescale) {
// The sample's duration matches the delta, so we can increment the count
lastTableEntry.sampleCount++;
} else {
// The delta has changed, subtract one from the previous run and create a new run with the new delta
lastTableEntry.sampleCount--;
// Add a new entry in order to maintain the last sample's true duration
trackData.timeToSampleTable.push({
sampleCount: 2,
sampleDelta: delta
sampleCount: 1,
sampleDelta: durationInTimescale
});
}
@@ -455,7 +558,7 @@ export class IsobmffMuxer extends Muxer {
if (this.#format.options.fastStart !== 'fragmented') {
trackData.timeToSampleTable.push({
sampleCount: 1,
sampleDelta: intoTimescale(sample.duration, trackData.timescale)
sampleDelta: durationInTimescale
});
trackData.compositionTimeOffsetTable.push({
sampleCount: 1,
@@ -527,30 +630,6 @@ export class IsobmffMuxer extends Muxer {
trackData.timestampProcessingQueue.push(sample);
}
#validateTimestamp(trackData: IsobmffTrackData, chunk: EncodedVideoChunk | EncodedAudioChunk) {
let timestampInSeconds = chunk.timestamp / 1e6;
if (timestampInSeconds < 0) {
throw new Error(`Timestamps must be non-negative (got ${timestampInSeconds}s).`);
}
if (trackData.firstTimestamp === null) {
trackData.firstTimestamp = timestampInSeconds;
}
timestampInSeconds -= trackData.firstTimestamp;
if (trackData.lastKeyFrameTimestamp !== null && timestampInSeconds < trackData.lastKeyFrameTimestamp) {
throw new Error(`Timestamp cannot be before last key frame's timestamp (got ${timestampInSeconds}s, last key frame at ${trackData.lastKeyFrameTimestamp}s).`);
}
if (chunk.type === 'key') {
trackData.lastKeyFrameTimestamp = timestampInSeconds;
}
return timestampInSeconds;
}
#finalizeCurrentChunk(trackData: IsobmffTrackData) {
assert(this.#format.options.fastStart !== 'fragmented');
@@ -588,8 +667,10 @@ export class IsobmffMuxer extends Muxer {
#interleaveSamples() {
assert(this.#format.options.fastStart === 'fragmented');
if (this.#trackDatas.length < this.output.tracks.length) {
return; // We haven't seen a sample from each track yet
for (const track of this.output.tracks) {
if (!track.source.closed && !this.#trackDatas.some(x => x.track === track)) {
return; // We haven't seen a sample from this open track yet
}
}
outer:
@@ -598,11 +679,11 @@ export class IsobmffMuxer extends Muxer {
let minTimestamp = Infinity;
for (let trackData of this.#trackDatas) {
if (trackData.sampleQueue.length === 0) {
if (trackData.sampleQueue.length === 0 && !trackData.track.source.closed) {
break outer;
}
if (trackData.sampleQueue[0]!.timestamp < minTimestamp) {
if (trackData.sampleQueue.length > 0 && trackData.sampleQueue[0]!.timestamp < minTimestamp) {
trackWithMinTimestamp = trackData;
minTimestamp = trackData.sampleQueue[0]!.timestamp;
}
@@ -625,13 +706,13 @@ export class IsobmffMuxer extends Muxer {
if (fragmentNumber === 1) {
// Write the moov box now that we have all decoder configs
let movieBox = moov(this.#trackDatas, this.#creationTime, true);
this.writeBox(movieBox);
this.#boxWriter.writeBox(movieBox);
}
// Write out an initial moof box; will be overwritten later once actual chunk offsets are known
let moofOffset = this.#writer.getPos();
let moofBox = moof(fragmentNumber, this.#trackDatas);
this.writeBox(moofBox);
this.#boxWriter.writeBox(moofBox);
// Create the mdat box
{
@@ -646,15 +727,15 @@ export class IsobmffMuxer extends Muxer {
}
}
let mdatSize = this.measureBox(mdatBox) + totalTrackSampleSize;
let mdatSize = this.#boxWriter.measureBox(mdatBox) + totalTrackSampleSize;
if (mdatSize >= 2**32) {
// Fragment is larger than 4 GiB, we need to use the large size
mdatBox.largeSize = true;
mdatSize = this.measureBox(mdatBox) + totalTrackSampleSize;
mdatSize = this.#boxWriter.measureBox(mdatBox) + totalTrackSampleSize;
}
mdatBox.size = mdatSize;
this.writeBox(mdatBox);
this.#boxWriter.writeBox(mdatBox);
}
// Write sample data
@@ -670,9 +751,9 @@ export class IsobmffMuxer extends Muxer {
// Now that we set the actual chunk offsets, fix the moof box
let endPos = this.#writer.getPos();
this.#writer.seek(this.offsets.get(moofBox)!);
this.#writer.seek(this.#boxWriter.offsets.get(moofBox)!);
let newMoofBox = moof(fragmentNumber, this.#trackDatas);
this.writeBox(newMoofBox);
this.#boxWriter.writeBox(newMoofBox);
this.#writer.seek(endPos);
for (let trackData of this.#trackDatas) {
@@ -686,8 +767,28 @@ export class IsobmffMuxer extends Muxer {
}
}
override onTrackClose(track: OutputTrack) {
if (track.type === 'subtitle' && track.source.codec === 'webvtt') {
let trackData = this.#trackDatas.find(x => x.track === track) as IsobmffSubtitleTrackData;
if (trackData) {
this.#processWebVTTCues(trackData, Infinity);
}
}
if (this.#format.options.fastStart === 'fragmented') {
// Since a track is now closed, we may be able to write out chunks that were previously waiting
this.#interleaveSamples();
}
}
/** Finalizes the file, making it ready for use. Must be called after all video and audio chunks have been added. */
finalize() {
for (let trackData of this.#trackDatas) {
if (trackData.type === 'subtitle' && trackData.track.source.codec === 'webvtt') {
this.#processWebVTTCues(trackData, Infinity);
}
}
if (this.#format.options.fastStart === 'fragmented') {
for (let trackData of this.#trackDatas) {
for (let sample of trackData.sampleQueue) {
@@ -719,8 +820,8 @@ export class IsobmffMuxer extends Muxer {
for (let i = 0; i < 2; i++) {
let movieBox = moov(this.#trackDatas, this.#creationTime);
let movieBoxSize = this.measureBox(movieBox);
mdatSize = this.measureBox(this.#mdat);
let movieBoxSize = this.#boxWriter.measureBox(movieBox);
mdatSize = this.#boxWriter.measureBox(this.#mdat);
let currentChunkPos = this.#writer.getPos() + movieBoxSize + mdatSize;
for (let chunk of this.#finalizedChunks) {
@@ -737,10 +838,10 @@ export class IsobmffMuxer extends Muxer {
}
let movieBox = moov(this.#trackDatas, this.#creationTime);
this.writeBox(movieBox);
this.#boxWriter.writeBox(movieBox);
this.#mdat.size = mdatSize!;
this.writeBox(this.#mdat);
this.#boxWriter.writeBox(this.#mdat);
for (let chunk of this.#finalizedChunks) {
for (let sample of chunk.samples) {
@@ -753,33 +854,33 @@ export class IsobmffMuxer extends Muxer {
// Append the mfra box to the end of the file for better random access
let startPos = this.#writer.getPos();
let mfraBox = mfra(this.#trackDatas);
this.writeBox(mfraBox);
this.#boxWriter.writeBox(mfraBox);
// Patch the 'size' field of the mfro box at the end of the mfra box now that we know its actual size
let mfraBoxSize = this.#writer.getPos() - startPos;
this.#writer.seek(this.#writer.getPos() - 4);
this.writeU32(mfraBoxSize);
this.#boxWriter.writeU32(mfraBoxSize);
} else {
assert(this.#mdat);
assert(this.#ftypSize !== null);
let mdatPos = this.offsets.get(this.#mdat);
let mdatPos = this.#boxWriter.offsets.get(this.#mdat);
assert(mdatPos !== undefined);
let mdatSize = this.#writer.getPos() - mdatPos;
this.#mdat.size = mdatSize;
this.#mdat.largeSize = mdatSize >= 2**32; // Only use the large size if we need it
this.patchBox(this.#mdat);
this.#boxWriter.patchBox(this.#mdat);
let movieBox = moov(this.#trackDatas, this.#creationTime);
if (typeof this.#format.options.fastStart === 'object') {
this.#writer.seek(this.#ftypSize);
this.writeBox(movieBox);
this.#boxWriter.writeBox(movieBox);
let remainingBytes = mdatPos - this.#writer.getPos();
this.writeBox(free(remainingBytes));
this.#boxWriter.writeBox(free(remainingBytes));
} else {
this.writeBox(movieBox);
this.#boxWriter.writeBox(movieBox);
}
}
}
+4 -3
View File
@@ -63,7 +63,6 @@ export enum EBMLId {
Video = 0xe0,
PixelWidth = 0xb0,
PixelHeight = 0xba,
Void = 0xec,
Audio = 0xe1,
SamplingFrequency = 0xb5,
Channels = 0x9f,
@@ -73,6 +72,9 @@ export enum EBMLId {
BlockGroup = 0xa0,
Block = 0xa1,
BlockAdditions = 0x75a1,
BlockMore = 0xa6,
BlockAdditional = 0xa5,
BlockAddID = 0xee,
BlockDuration = 0x9b,
ReferenceBlock = 0xfb,
Cluster = 0x1f43b675,
@@ -87,8 +89,7 @@ export enum EBMLId {
MatrixCoefficients = 0x55b1,
TransferCharacteristics = 0x55ba,
Primaries = 0x55bb,
Range = 0x55b9,
AlphaMode = 0x53c0
Range = 0x55b9
}
export const measureUnsignedInt = (value: number) => {
+54 -58
View File
@@ -1,9 +1,9 @@
import { assert, readBits, toUint8Array, writeBits } from '../misc';
import { assert, readBits, textEncoder, toUint8Array, writeBits } from '../misc';
import { Muxer } from '../muxer';
import { Output, OutputAudioTrack, OutputSubtitleTrack, OutputTrack, OutputVideoTrack } from '../output';
import { MkvOutputFormat, WebMOutputFormat } from '../output_format';
import { AudioCodec, SubtitleCodec, VideoCodec } from '../source';
import { EncodedSubtitleChunk, EncodedSubtitleChunkMetadata, SubtitleDecoderConfig } from '../subtitles';
import { formatSubtitleTimestamp, inlineTimestampRegex, parseSubtitleTimestamp, SubtitleConfig, SubtitleCue, SubtitleMetadata } from '../subtitles';
import { Writer } from '../writer';
import { EBML, EBMLElement, EBMLFloat32, EBMLFloat64, EBMLId, EBMLSignedInt, measureEBMLVarInt, measureSignedInt, measureUnsignedInt } from './ebml';
@@ -38,9 +38,6 @@ type SeekHead = {
type MatroskaTrackData = {
chunkQueue: InternalMediaChunk[],
firstTimestamp: number | null,
lastKeyFrameTimestamp: number | null,
lastWrittenMsTimestamp: number | null
} & ({
track: OutputVideoTrack,
@@ -62,7 +59,7 @@ type MatroskaTrackData = {
track: OutputSubtitleTrack,
type: 'subtitle',
info: {
decoderConfig: SubtitleDecoderConfig
config: SubtitleConfig
}
});
@@ -78,7 +75,6 @@ const CODEC_STRING_MAP: Record<VideoCodec | AudioCodec | SubtitleCodec, string>
av1: 'V_AV1',
aac: 'A_AAC',
opus: 'A_OPUS',
vorbis: 'A_VORBIS',
webvtt: 'S_TEXT/WEBVTT'
};
@@ -94,6 +90,8 @@ const TRACK_TYPE_MAP: Record<OutputTrack['type'], number> = {
// Update: Not really. There are duration fields and seek fields that are just uneditable if streaming is required.
export class MatroskaMuxer extends Muxer {
override timestampsMustStartAtZero = true;
#writer: Writer;
#format: WebMOutputFormat | MkvOutputFormat;
@@ -391,8 +389,8 @@ export class MatroskaMuxer extends Muxer {
{ id: EBMLId.TrackUID, data: trackData.track.id },
{ id: EBMLId.TrackType, data: TRACK_TYPE_MAP[trackData.type] }, // TODO Subtitle case
{ id: EBMLId.CodecID, data: CODEC_STRING_MAP[trackData.track.source.codec] },
(trackData.info.decoderConfig.description ? { id: EBMLId.CodecPrivate, data: toUint8Array(trackData.info.decoderConfig.description) } : null),
...(trackData.type === 'video' ? [
(trackData.info.decoderConfig.description ? { id: EBMLId.CodecPrivate, data: toUint8Array(trackData.info.decoderConfig.description) } : null),
(trackData.track.metadata.frameRate ? { id: EBMLId.DefaultDuration, data: 1e9 / trackData.track.metadata.frameRate } : null),
{ id: EBMLId.Video, data: [
{ id: EBMLId.PixelWidth, data: trackData.info.width },
@@ -430,12 +428,16 @@ export class MatroskaMuxer extends Muxer {
] }
] : []),
...(trackData.type === 'audio' ? [
(trackData.info.decoderConfig.description ? { id: EBMLId.CodecPrivate, data: toUint8Array(trackData.info.decoderConfig.description) } : null),
{ id: EBMLId.Audio, data: [
{ id: EBMLId.SamplingFrequency, data: new EBMLFloat32(trackData.info.sampleRate) },
{ id: EBMLId.Channels, data: trackData.info.numberOfChannels },
// Bit depth for when PCM is a thing
] }
] : [])
] : []),
...(trackData.type === 'subtitle' ? [
{ id: EBMLId.CodecPrivate, data: textEncoder.encode(trackData.info.config.description) }
] : []),
] })
}
}
@@ -492,8 +494,6 @@ export class MatroskaMuxer extends Muxer {
decoderConfig: meta.decoderConfig
},
chunkQueue: [],
firstTimestamp: null,
lastKeyFrameTimestamp: null,
lastWrittenMsTimestamp: null
};
@@ -522,8 +522,6 @@ export class MatroskaMuxer extends Muxer {
decoderConfig: meta.decoderConfig
},
chunkQueue: [],
firstTimestamp: null,
lastKeyFrameTimestamp: null,
lastWrittenMsTimestamp: null
};
@@ -533,7 +531,7 @@ export class MatroskaMuxer extends Muxer {
return newTrackData;
}
#getSubtitleTrackData(track: OutputSubtitleTrack, meta?: EncodedSubtitleChunkMetadata) {
#getSubtitleTrackData(track: OutputSubtitleTrack, meta?: SubtitleMetadata) {
const existingTrackData = this.#trackDatas.find(x => x.track === track);
if (existingTrackData) {
return existingTrackData as MatroskaAudioTrackData;
@@ -541,17 +539,15 @@ export class MatroskaMuxer extends Muxer {
// TODO Make proper errors for these
assert(meta);
assert(meta.decoderConfig);
assert(meta.config);
const newTrackData: MatroskaSubtitleTrackData = {
track,
type: 'subtitle',
info: {
decoderConfig: meta.decoderConfig
config: meta.config
},
chunkQueue: [],
firstTimestamp: null,
lastKeyFrameTimestamp: null,
lastWrittenMsTimestamp: null
};
@@ -567,7 +563,8 @@ export class MatroskaMuxer extends Muxer {
let data = new Uint8Array(chunk.byteLength);
chunk.copyTo(data);
let videoChunk = this.#createInternalChunk(trackData, data, chunk.timestamp, chunk.duration ?? 0, chunk.type);
let timestamp = this.validateAndNormalizeTimestamp(trackData.track, chunk.timestamp, chunk.type === 'key');
let videoChunk = this.#createInternalChunk(data, timestamp, (chunk.duration ?? 0) / 1e6, chunk.type);
if (track.source.codec === 'vp9') this.#fixVP9ColorSpace(trackData, videoChunk);
trackData.chunkQueue.push(videoChunk);
@@ -580,27 +577,44 @@ export class MatroskaMuxer extends Muxer {
let data = new Uint8Array(chunk.byteLength);
chunk.copyTo(data);
let audioChunk = this.#createInternalChunk(trackData, data, chunk.timestamp, chunk.duration ?? 0, chunk.type);
let timestamp = this.validateAndNormalizeTimestamp(trackData.track, chunk.timestamp, chunk.type === 'key');
let audioChunk = this.#createInternalChunk(data, timestamp, (chunk.duration ?? 0) / 1e6, chunk.type);
trackData.chunkQueue.push(audioChunk);
this.#interleaveChunks();
}
addEncodedSubtitleChunk(track: OutputSubtitleTrack, chunk: EncodedSubtitleChunk, meta?: EncodedSubtitleChunkMetadata) {
addSubtitleCue(track: OutputSubtitleTrack, cue: SubtitleCue, meta?: SubtitleMetadata) {
const trackData = this.#getSubtitleTrackData(track, meta);
let subtitleChunk = this.#createInternalChunk(trackData, chunk.body, chunk.timestamp, chunk.duration, 'key', chunk.additions);
const timestamp = this.validateAndNormalizeTimestamp(trackData.track, 1e6 * cue.timestamp, true);
let bodyText = cue.text;
const timestampMs = Math.floor(timestamp * 1000);
// Replace in-body timestamps so that they're relative to the cue start time
inlineTimestampRegex.lastIndex = 0;
bodyText = bodyText.replace(inlineTimestampRegex, (match) => {
let time = parseSubtitleTimestamp(match.slice(1, -1));
let offsetTime = time - timestampMs;
return `<${formatSubtitleTimestamp(offsetTime)}>`;
});
const body = textEncoder.encode(bodyText);
const additions = `${cue.settings ?? ''}\n${cue.identifier ?? ''}\n${cue.notes ?? ''}`;
let subtitleChunk = this.#createInternalChunk(body, timestamp, cue.duration, 'key', additions.trim() ? textEncoder.encode(additions) : null);
trackData.chunkQueue.push(subtitleChunk);
this.#interleaveChunks();
}
#interleaveChunks() {
let openTrackCount = 0;
for (const trackData of this.#trackDatas) if (!trackData.track.source.closed) openTrackCount++;
if (this.#trackDatas.length < openTrackCount) {
return; // We haven't seen a sample from every open track yet
for (const track of this.output.tracks) {
if (!track.source.closed && !this.#trackDatas.some(x => x.track === track)) {
return; // We haven't seen a sample from this open track yet
}
}
outer:
@@ -668,50 +682,23 @@ export class MatroskaMuxer extends Muxer {
/** Converts a read-only external chunk into an internal one for easier use. */
#createInternalChunk(
trackData: MatroskaTrackData,
data: Uint8Array,
timestamp: number,
duration: number,
type: 'key' | 'delta',
additions: Uint8Array | null = null
) {
let adjustedTimestamp = this.#validateTimestamp(trackData, timestamp, type === 'key');
let internalChunk: InternalMediaChunk = {
data,
type,
timestamp: adjustedTimestamp,
duration: duration / 1e6,
timestamp,
duration,
additions
};
return internalChunk;
}
#validateTimestamp(trackData: MatroskaTrackData, timestamp: number, isKeyFrame: boolean) {
let timestampInSeconds = timestamp / 1e6;
if (timestampInSeconds < 0) {
throw new Error(`Timestamps must be non-negative (got ${timestampInSeconds}s).`);
}
if (trackData.firstTimestamp === null) {
trackData.firstTimestamp = timestampInSeconds;
}
timestampInSeconds -= trackData.firstTimestamp;
if (trackData.lastKeyFrameTimestamp !== null && timestampInSeconds < trackData.lastKeyFrameTimestamp) {
throw new Error(`Timestamp cannot be before last key frame's timestamp (got ${timestampInSeconds}s, last key frame at ${trackData.lastKeyFrameTimestamp}s).`);
}
if (isKeyFrame) {
trackData.lastKeyFrameTimestamp = timestampInSeconds;
}
return timestampInSeconds;
}
/** Writes a block containing media data to the file. */
#writeBlock(trackData: MatroskaTrackData, chunk: InternalMediaChunk) {
// TODO Update this comment. This code always runs now
@@ -726,6 +713,10 @@ export class MatroskaMuxer extends Muxer {
// We can only finalize this fragment (and begin a new one) if we know that each track will be able to
// start the new one with a key frame.
const keyFrameQueuedEverywhere = this.#trackDatas.every(otherTrackData => {
if (otherTrackData.track.source.closed) {
return true;
}
if (trackData === otherTrackData) {
return chunk.type === 'key';
}
@@ -780,8 +771,13 @@ export class MatroskaMuxer extends Muxer {
chunk.data
] },
chunk.type === 'delta' ? { id: EBMLId.ReferenceBlock, data: new EBMLSignedInt(trackData.lastWrittenMsTimestamp! - msTimestamp) } : null,
msDuration > 0 ? { id: EBMLId.BlockDuration, data: msDuration } : null,
chunk.additions ? { id: EBMLId.BlockAdditions, data: chunk.additions } : null
chunk.additions ? { id: EBMLId.BlockAdditions, data: [
{ id: EBMLId.BlockMore, data: [
{ id: EBMLId.BlockAdditional, data: chunk.additions },
{ id: EBMLId.BlockAddID, data: 1 }
] }
] } : null,
msDuration > 0 ? { id: EBMLId.BlockDuration, data: msDuration } : null
] };
this.writeEBML(blockGroup);
}
+3 -1
View File
@@ -48,4 +48,6 @@ export const toUint8Array = (source: AllowSharedBufferSource): Uint8Array => {
} else {
return new Uint8Array(source.buffer, source.byteOffset, source.byteLength);
}
};
};
export const textEncoder = new TextEncoder();
+55 -2
View File
@@ -1,5 +1,5 @@
import { Output, OutputAudioTrack, OutputSubtitleTrack, OutputTrack, OutputVideoTrack } from "./output";
import { EncodedSubtitleChunk, EncodedSubtitleChunkMetadata } from "./subtitles";
import { SubtitleCue, SubtitleMetadata } from "./subtitles";
export abstract class Muxer {
output: Output;
@@ -11,9 +11,62 @@ export abstract class Muxer {
abstract start(): void;
abstract addEncodedVideoChunk(track: OutputVideoTrack, chunk: EncodedVideoChunk, meta?: EncodedVideoChunkMetadata): void;
abstract addEncodedAudioChunk(track: OutputAudioTrack, chunk: EncodedAudioChunk, meta?: EncodedAudioChunkMetadata): void;
abstract addEncodedSubtitleChunk(track: OutputSubtitleTrack, chunk: EncodedSubtitleChunk, meta?: EncodedSubtitleChunkMetadata): void;
abstract addSubtitleCue(track: OutputSubtitleTrack, cue: SubtitleCue, meta?: SubtitleMetadata): void;
abstract finalize(): void;
beforeTrackAdd(track: OutputTrack) {}
onTrackClose(track: OutputTrack) {}
private trackTimestampInfo = new WeakMap<OutputTrack, {
timestampOffset: number,
maxTimestamp: number,
lastKeyFrameTimestamp: number
}>();
abstract timestampsMustStartAtZero: boolean;
protected validateAndNormalizeTimestamp(track: OutputTrack, rawTimestampInUs: number, isKeyFrame: boolean) {
let timestampInSeconds = rawTimestampInUs / 1e6;
let timestampInfo = this.trackTimestampInfo.get(track);
if (!timestampInfo) {
if (!isKeyFrame) {
throw new Error('First frame must be a key frame.');
}
if (this.timestampsMustStartAtZero && timestampInSeconds > 0) {
throw new Error(`Timestamps must start at zero (got ${timestampInSeconds}s).`);
}
timestampInfo = {
timestampOffset: timestampInSeconds,
maxTimestamp: track.source.offsetTimestamps ? 0 : timestampInSeconds,
lastKeyFrameTimestamp: track.source.offsetTimestamps ? 0 : timestampInSeconds
};
this.trackTimestampInfo.set(track, timestampInfo);
}
if (track.source.offsetTimestamps) {
timestampInSeconds -= timestampInfo.timestampOffset;
}
if (timestampInSeconds < 0) {
throw new Error(`Timestamps must be non-negative (got ${timestampInSeconds}s).`);
}
if (timestampInSeconds < timestampInfo.lastKeyFrameTimestamp) {
throw new Error(`Timestamp cannot be smaller than last key frame's timestamp (got ${timestampInSeconds}s, last key frame at ${timestampInfo.lastKeyFrameTimestamp}s).`);
}
if (isKeyFrame) {
if (timestampInSeconds < timestampInfo.maxTimestamp) {
throw new Error(`Key frame timestamps cannot be smaller than any timestamp that came before (got ${timestampInSeconds}s, max timestamp was ${timestampInfo.maxTimestamp}s).`);
}
timestampInfo.lastKeyFrameTimestamp = timestampInSeconds;
}
timestampInfo.maxTimestamp = Math.max(timestampInfo.maxTimestamp, timestampInSeconds);
return timestampInSeconds;
}
}
+2
View File
@@ -9,7 +9,9 @@ export abstract class OutputFormat {
export class Mp4OutputFormat extends OutputFormat {
constructor(public options: {
// TODO: Make this optional with a smart default
fastStart: false | 'in-memory' | 'fragmented' | {
// TODO: This is too simple now that multiple tracks can exist.
expectedVideoChunks?: number,
expectedAudioChunks?: number
},
+20 -12
View File
@@ -1,15 +1,16 @@
import { buildAudioCodecString, buildVideoCodecString } from "./codec";
import { assert, TransformationMatrix } from "./misc";
import { assert } from "./misc";
import { OutputAudioTrack, OutputSubtitleTrack, OutputTrack, OutputVideoTrack } from "./output";
import { SubtitleEncoder } from "./subtitles";
import { SubtitleParser } from "./subtitles";
export type VideoCodec = 'avc' | 'hevc' | 'vp8' | 'vp9' | 'av1';
export type AudioCodec = 'aac' | 'opus' | 'vorbis'; // TODO add the rest
export type AudioCodec = 'aac' | 'opus'; // TODO add the rest
export type SubtitleCodec = 'webvtt';
export abstract class MediaSource {
connectedTrack: OutputTrack | null = null;
closed = false;
offsetTimestamps = false;
ensureValidDigest() {
if (!this.connectedTrack) {
@@ -83,8 +84,9 @@ export class EncodedVideoChunkSource extends VideoSource {
const KEY_FRAME_INTERVAL = 5;
type VideoCodecConfig = {
codec: 'avc' | 'hevc' | 'vp8' | 'vp9' | 'av1',
bitrate: number
codec: VideoCodec,
bitrate: number,
latencyMode?: VideoEncoderConfig['latencyMode']
};
class VideoEncoderWrapper {
@@ -125,6 +127,8 @@ class VideoEncoderWrapper {
width: videoFrame.codedWidth,
height: videoFrame.codedHeight,
bitrate: this.codecConfig.bitrate,
framerate: this.source.connectedTrack?.metadata.frameRate,
latencyMode: this.codecConfig.latencyMode,
});
}
@@ -162,6 +166,7 @@ export class CanvasSource extends VideoSource {
const frame = new VideoFrame(this.canvas, {
timestamp: Math.round(1e6 * timestamp),
duration: Math.round(1e6 * duration),
alpha: 'discard',
});
this.encoder.digest(frame);
@@ -177,6 +182,8 @@ export class MediaStreamVideoTrackSource extends VideoSource {
private encoder: VideoEncoderWrapper;
private abortController: AbortController | null = null;
override offsetTimestamps = true;
constructor(private track: MediaStreamVideoTrack, codecConfig: VideoCodecConfig) {
super(codecConfig.codec);
this.encoder = new VideoEncoderWrapper(this, codecConfig);
@@ -236,7 +243,7 @@ export class EncodedAudioChunkSource extends AudioSource {
}
type AudioCodecConfig = {
codec: 'aac' | 'opus' | 'vorbis',
codec: AudioCodec,
bitrate: number
};
@@ -340,6 +347,8 @@ export class MediaStreamAudioTrackSource extends AudioSource {
private encoder: AudioEncoderWrapper;
private abortController: AbortController | null = null;
override offsetTimestamps = true;
constructor(private track: MediaStreamAudioTrack, codecConfig: AudioCodecConfig) {
super(codecConfig.codec);
this.encoder = new AudioEncoderWrapper(this, codecConfig);
@@ -388,21 +397,20 @@ export abstract class SubtitleSource extends MediaSource {
}
export class TextSubtitleSource extends SubtitleSource {
private encoder: SubtitleEncoder;
private parser: SubtitleParser;
constructor(codec: SubtitleCodec) {
super(codec);
this.encoder = new SubtitleEncoder({
output: (chunk, metadata) => this.connectedTrack?.output.muxer.addEncodedSubtitleChunk(this.connectedTrack, chunk, metadata),
this.parser = new SubtitleParser({
codec,
output: (cue, metadata) => this.connectedTrack?.output.muxer.addSubtitleCue(this.connectedTrack, cue, metadata),
error: (error) => console.error(error) // TODO
});
this.encoder.configure({ codec });
}
digest(text: string) {
this.ensureValidDigest();
this.encoder.encode(text);
this.parser.parse(text);
}
}
+60 -83
View File
@@ -1,62 +1,46 @@
export type EncodedSubtitleChunk = {
body: Uint8Array,
additions: Uint8Array | null,
timestamp: number,
duration: number
export type SubtitleCue = {
timestamp: number, // in seconds
duration: number, // in seconds
text: string,
identifier?: string,
settings?: string,
notes?: string
};
export type SubtitleDecoderConfig = {
description: Uint8Array
export type SubtitleConfig = {
description: string
};
export type EncodedSubtitleChunkMetadata = {
decoderConfig?: SubtitleDecoderConfig
export type SubtitleMetadata = {
config?: SubtitleConfig
};
interface SubtitleEncoderOptions {
output: (chunk: EncodedSubtitleChunk, metadata: EncodedSubtitleChunkMetadata) => unknown,
type SubtitleParserOptions = {
codec: 'webvtt',
output: (cue: SubtitleCue, metadata: SubtitleMetadata) => unknown,
error: (error: Error) => unknown
}
interface SubtitleEncoderConfig {
codec: 'webvtt'
}
};
const cueBlockHeaderRegex = /(?:(.+?)\n)?((?:\d{2}:)?\d{2}:\d{2}.\d{3})\s+-->\s+((?:\d{2}:)?\d{2}:\d{2}.\d{3})/g;
const preambleStartRegex = /^WEBVTT.*?\n{2}/;
const timestampRegex = /(?:(\d{2}):)?(\d{2}):(\d{2}).(\d{3})/;
const inlineTimestampRegex = /<(?:(\d{2}):)?(\d{2}):(\d{2}).(\d{3})>/g;
const textEncoder = new TextEncoder();
const preambleStartRegex = /^WEBVTT(.|\n)*?\n{2}/;
export const inlineTimestampRegex = /<(?:(\d{2}):)?(\d{2}):(\d{2}).(\d{3})>/g;
export class SubtitleEncoder {
#options: SubtitleEncoderOptions;
#config: SubtitleEncoderConfig | null = null;
#preambleBytes: Uint8Array | null = null;
export class SubtitleParser {
#options: SubtitleParserOptions;
#preambleText: string | null = null;
#preambleEmitted = false;
constructor(options: SubtitleEncoderOptions) {
constructor(options: SubtitleParserOptions) {
this.#options = options;
}
configure(config: SubtitleEncoderConfig) {
if (config.codec !== 'webvtt') {
throw new Error("Codec must be 'webvtt'.");
}
this.#config = config;
}
encode(text: string) {
if (!this.#config) {
throw new Error('Encoder not configured.');
}
text = text.replace('\r\n', '\n').replace('\r', '\n');
parse(text: string) {
text = text.replaceAll('\r\n', '\n').replaceAll('\r', '\n');
cueBlockHeaderRegex.lastIndex = 0;
let match: RegExpMatchArray | null;
if (!this.#preambleBytes) {
if (!this.#preambleText) {
if (!preambleStartRegex.test(text)) {
let error = new Error('WebVTT preamble incorrect.');
this.#options.error(error);
@@ -72,7 +56,7 @@ export class SubtitleEncoder {
throw error;
}
this.#preambleBytes = textEncoder.encode(preamble);
this.#preambleText = preamble;
if (match) {
text = text.slice(match.index);
@@ -82,70 +66,63 @@ export class SubtitleEncoder {
while (match = cueBlockHeaderRegex.exec(text)) {
let notes = text.slice(0, match.index);
let cueIdentifier = match[1] || '';
let cueIdentifier = match[1];
let matchEnd = match.index! + match[0].length;
let bodyStart = text.indexOf('\n', matchEnd) + 1;
let cueSettings = text.slice(matchEnd, bodyStart).trim();
let bodyEnd = text.indexOf('\n\n', matchEnd);
if (bodyEnd === -1) bodyEnd = text.length;
let startTime = this.#parseTimestamp(match[2]!);
let endTime = this.#parseTimestamp(match[3]!);
let startTime = parseSubtitleTimestamp(match[2]!);
let endTime = parseSubtitleTimestamp(match[3]!);
let duration = endTime - startTime;
let body = text.slice(bodyStart, bodyEnd);
let additions = `${cueSettings}\n${cueIdentifier}\n${notes}`;
// Replace in-body timestamps so that they're relative to the cue start time
inlineTimestampRegex.lastIndex = 0;
body = body.replace(inlineTimestampRegex, (match) => {
let time = this.#parseTimestamp(match.slice(1, -1));
let offsetTime = time - startTime;
return `<${this.#formatTimestamp(offsetTime)}>`;
});
let body = text.slice(bodyStart, bodyEnd).trim();
text = text.slice(bodyEnd).trimStart();
cueBlockHeaderRegex.lastIndex = 0;
let chunk: EncodedSubtitleChunk = {
body: textEncoder.encode(body),
additions: additions.trim() === '' ? null : textEncoder.encode(additions),
timestamp: startTime * 1000,
duration: duration * 1000
let cue: SubtitleCue = {
timestamp: startTime / 1000,
duration: duration / 1000,
text: body,
identifier: cueIdentifier,
settings: cueSettings,
notes
};
let meta: EncodedSubtitleChunkMetadata = {};
let meta: SubtitleMetadata = {};
if (!this.#preambleEmitted) {
meta.decoderConfig = {
description: this.#preambleBytes
meta.config = {
description: this.#preambleText
};
this.#preambleEmitted = true;
}
this.#options.output(chunk, meta);
this.#options.output(cue, meta);
}
}
}
#parseTimestamp(string: string) {
let match = timestampRegex.exec(string);
if (!match) throw new Error('Expected match.');
const timestampRegex = /(?:(\d{2}):)?(\d{2}):(\d{2}).(\d{3})/;
export const parseSubtitleTimestamp = (string: string) => {
let match = timestampRegex.exec(string);
if (!match) throw new Error('Expected match.');
return 60 * 60 * 1000 * Number(match[1] || '0') +
60 * 1000 * Number(match[2]) +
1000 * Number(match[3]) +
Number(match[4]);
}
return 60 * 60 * 1000 * Number(match[1] || '0') +
60 * 1000 * Number(match[2]) +
1000 * Number(match[3]) +
Number(match[4]);
};
#formatTimestamp(timestamp: number) {
let hours = Math.floor(timestamp / (60 * 60 * 1000));
let minutes = Math.floor((timestamp % (60 * 60 * 1000)) / (60 * 1000));
let seconds = Math.floor((timestamp % (60 * 1000)) / 1000);
let milliseconds = timestamp % 1000;
export const formatSubtitleTimestamp = (timestamp: number) => {
let hours = Math.floor(timestamp / (60 * 60 * 1000));
let minutes = Math.floor((timestamp % (60 * 60 * 1000)) / (60 * 1000));
let seconds = Math.floor((timestamp % (60 * 1000)) / 1000);
let milliseconds = timestamp % 1000;
return hours.toString().padStart(2, '0') + ':' +
minutes.toString().padStart(2, '0') + ':' +
seconds.toString().padStart(2, '0') + '.' +
milliseconds.toString().padStart(3, '0');
}
}
return hours.toString().padStart(2, '0') + ':' +
minutes.toString().padStart(2, '0') + ':' +
seconds.toString().padStart(2, '0') + '.' +
milliseconds.toString().padStart(3, '0');
};
+4
View File
@@ -67,6 +67,10 @@ export class ArrayBufferTargetWriter extends Writer {
this.#ensureSize(this.#pos);
this.#target.buffer = this.#buffer.slice(0, Math.max(this.#maxPos, this.#pos));
}
getSlice(start: number, end: number) {
return this.#bytes.slice(start, end);
}
}
/**