mirror of
https://github.com/arcodange-org/mediabunny.git
synced 2026-10-02 05:13:50 +02:00
Add WebVTT subtitle support to MP4 & a bunch of misc changes
This commit is contained in:
+55
-5
@@ -36,7 +36,8 @@
|
||||
canvas.height = 720;
|
||||
const context = canvas.getContext('2d');
|
||||
|
||||
let format = new Metamuxer.WebMOutputFormat();// new Metamuxer.MkvOutputFormat();// new Metamuxer.Mp4OutputFormat({ fastStart: false });
|
||||
let format = new Metamuxer.WebMOutputFormat();
|
||||
format = new Metamuxer.Mp4OutputFormat({ fastStart: false }); // new Metamuxer.MkvOutputFormat();// new Metamuxer.Mp4OutputFormat({ fastStart: false });
|
||||
let target = new Metamuxer.ArrayBufferTarget();
|
||||
|
||||
let output = new Metamuxer.Output({ format, target });
|
||||
@@ -71,11 +72,11 @@
|
||||
*/
|
||||
|
||||
let videoSource = new Metamuxer.CanvasSource(canvas, {
|
||||
codec: 'vp9',
|
||||
codec: 'avc',
|
||||
bitrate: 1e6
|
||||
});
|
||||
let audioSource = new Metamuxer.AudioBufferSource({
|
||||
codec: 'opus',
|
||||
codec: 'aac',
|
||||
bitrate: 128e3,
|
||||
});
|
||||
let subtitleSource = new Metamuxer.TextSubtitleSource('webvtt');
|
||||
@@ -89,9 +90,58 @@
|
||||
let simpleWebvttFile =
|
||||
`WEBVTT
|
||||
|
||||
00:00:00.000 --> 00:00:10.000
|
||||
1
|
||||
00:00:02.000 --> 00:00:10.000
|
||||
Example entry 1: Hello <b>world</b>.
|
||||
`;
|
||||
|
||||
if (false) simpleWebvttFile =
|
||||
`WEBVTT
|
||||
|
||||
1
|
||||
00:05.000 --> 00:12.500 align:start line:10
|
||||
<v Roger Bingham>We are in New York City.
|
||||
We are looking straight down 5th Avenue.
|
||||
|
||||
00:13.000 --> 00:18.000
|
||||
<v Neil DeGrass Tyson>Didn't you already say that?
|
||||
|
||||
2
|
||||
00:17.000 --> 00:20.000
|
||||
Testing... <00:17.350>One... <00:18.125>Two...
|
||||
`
|
||||
|
||||
simpleWebvttFile =
|
||||
`WEBVTT
|
||||
|
||||
00:00:01.000 --> 00:00:01.500 align:start line:1
|
||||
1. <b>justify (top, left)</b>,
|
||||
|
||||
00:00:01.500 --> 00:00:02.000 align:center line:1.5
|
||||
2. <b>justify (top, center)</b>,
|
||||
|
||||
00:00:02.000 --> 00:00:02.500 align:end line:2
|
||||
3. <b>justify (top, right)</b>,
|
||||
|
||||
00:00:02.500 --> 00:00:03.000 align:start line:48%
|
||||
4. <b>justify (center, left)</b>,
|
||||
|
||||
00:00:03.000 --> 00:00:03.500 align:center line:50%
|
||||
5. <b>justify (center, center)</b>,
|
||||
|
||||
00:00:03.500 --> 00:00:04.000 align:end line:52%
|
||||
6. <b>justify (center, right)</b>,
|
||||
|
||||
00:00:04.000 --> 00:00:04.500 align:start line:-1
|
||||
7. <b>justify (bottom, left)</b>,
|
||||
|
||||
00:00:04.500 --> 00:00:05.000 align:center line:-1.5
|
||||
8. <b>justify (bottom, center)</b>,
|
||||
|
||||
00:00:05.000 --> 00:00:05.500 align:end line:-2
|
||||
9. <b>justify (bottom, right)</b>.
|
||||
`;
|
||||
|
||||
subtitleSource.digest(simpleWebvttFile);
|
||||
subtitleSource.close();
|
||||
|
||||
@@ -112,5 +162,5 @@ Example entry 1: Hello <b>world</b>.
|
||||
await output.finalize();
|
||||
|
||||
console.log(target);
|
||||
download(new Blob([target.buffer]), 'test.webm');
|
||||
download(new Blob([target.buffer]), 'test.mp4');
|
||||
</script>
|
||||
Vendored
+810
-592
File diff suppressed because it is too large
Load Diff
Vendored
+7
-7
File diff suppressed because one or more lines are too long
Vendored
+7
-7
File diff suppressed because one or more lines are too long
Vendored
+810
-592
File diff suppressed because it is too large
Load Diff
+188
-20
@@ -1,6 +1,97 @@
|
||||
import { toUint8Array, assert, isU32, last, TransformationMatrix } from '../misc';
|
||||
import { AudioCodec, AudioSource, VideoCodec, VideoSource } from '../source';
|
||||
import { GLOBAL_TIMESCALE, intoTimescale, IsobmffAudioTrackData, IsobmffTrackData, IsobmffVideoTrackData, Sample } from './isobmff_muxer';
|
||||
import { toUint8Array, assert, isU32, last, TransformationMatrix, textEncoder } from '../misc';
|
||||
import { AudioCodec, AudioSource, SubtitleCodec, VideoCodec, VideoSource } from '../source';
|
||||
import { formatSubtitleTimestamp } from '../subtitles';
|
||||
import { Writer } from '../writer';
|
||||
import { GLOBAL_TIMESCALE, intoTimescale, IsobmffAudioTrackData, IsobmffSubtitleTrackData, IsobmffTrackData, IsobmffVideoTrackData, Sample } from './isobmff_muxer';
|
||||
|
||||
export class IsobmffBoxWriter {
|
||||
private helper = new Uint8Array(8);
|
||||
private helperView = new DataView(this.helper.buffer);
|
||||
|
||||
/**
|
||||
* Stores the position from the start of the file to where boxes elements have been written. This is used to
|
||||
* rewrite/edit elements that were already added before, and to measure sizes of things.
|
||||
*/
|
||||
offsets = new WeakMap<Box, number>();
|
||||
|
||||
constructor(private writer: Writer) {}
|
||||
|
||||
writeU32(value: number) {
|
||||
this.helperView.setUint32(0, value, false);
|
||||
this.writer.write(this.helper.subarray(0, 4));
|
||||
}
|
||||
|
||||
writeU64(value: number) {
|
||||
this.helperView.setUint32(0, Math.floor(value / 2**32), false);
|
||||
this.helperView.setUint32(4, value, false);
|
||||
this.writer.write(this.helper.subarray(0, 8));
|
||||
}
|
||||
|
||||
writeAscii(text: string) {
|
||||
for (let i = 0; i < text.length; i++) {
|
||||
this.helperView.setUint8(i % 8, text.charCodeAt(i));
|
||||
if (i % 8 === 7) this.writer.write(this.helper);
|
||||
}
|
||||
|
||||
if (text.length % 8 !== 0) {
|
||||
this.writer.write(this.helper.subarray(0, text.length % 8));
|
||||
}
|
||||
}
|
||||
|
||||
writeBox(box: Box) {
|
||||
this.offsets.set(box, this.writer.getPos());
|
||||
|
||||
if (box.contents && !box.children) {
|
||||
this.writeBoxHeader(box, box.size ?? box.contents.byteLength + 8);
|
||||
this.writer.write(box.contents);
|
||||
} else {
|
||||
let startPos = this.writer.getPos();
|
||||
this.writeBoxHeader(box, 0);
|
||||
|
||||
if (box.contents) this.writer.write(box.contents);
|
||||
if (box.children) for (let child of box.children) if (child) this.writeBox(child);
|
||||
|
||||
let endPos = this.writer.getPos();
|
||||
let size = box.size ?? endPos - startPos;
|
||||
this.writer.seek(startPos);
|
||||
this.writeBoxHeader(box, size);
|
||||
this.writer.seek(endPos);
|
||||
}
|
||||
}
|
||||
|
||||
writeBoxHeader(box: Box, size: number) {
|
||||
this.writeU32(box.largeSize ? 1 : size);
|
||||
this.writeAscii(box.type);
|
||||
if (box.largeSize) this.writeU64(size);
|
||||
}
|
||||
|
||||
measureBoxHeader(box: Box) {
|
||||
return 8 + (box.largeSize ? 8 : 0);
|
||||
}
|
||||
|
||||
patchBox(box: Box) {
|
||||
const boxOffset = this.offsets.get(box);
|
||||
assert(boxOffset !== undefined);
|
||||
|
||||
let endPos = this.writer.getPos();
|
||||
this.writer.seek(boxOffset);
|
||||
this.writeBox(box);
|
||||
this.writer.seek(endPos);
|
||||
}
|
||||
|
||||
measureBox(box: Box) {
|
||||
if (box.contents && !box.children) {
|
||||
let headerSize = this.measureBoxHeader(box);
|
||||
return headerSize + box.contents.byteLength;
|
||||
} else {
|
||||
let result = this.measureBoxHeader(box);
|
||||
if (box.contents) result += box.contents.byteLength;
|
||||
if (box.children) for (let child of box.children) if (child) result += this.measureBox(child);
|
||||
|
||||
return result;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
let bytes = new Uint8Array(8);
|
||||
let view = new DataView(bytes.buffer);
|
||||
@@ -248,7 +339,7 @@ export const tkhd = (
|
||||
u32OrU64(durationInGlobalTimescale), // Duration
|
||||
Array(8).fill(0), // Reserved
|
||||
u16(0), // Layer
|
||||
u16(0), // Alternate group
|
||||
u16(trackData.track.id), // Alternate group
|
||||
fixed_8_8(trackData.type === 'audio' ? 1 : 0), // Volume
|
||||
u16(0), // Reserved
|
||||
matrixToBytes(matrix), // Matrix
|
||||
@@ -260,7 +351,7 @@ export const tkhd = (
|
||||
/** Media Box: Describes and define a track's media type and sample data. */
|
||||
export const mdia = (trackData: IsobmffTrackData, creationTime: number) => box('mdia', undefined, [
|
||||
mdhd(trackData, creationTime),
|
||||
hdlr(trackData.type === 'video' ? 'vide' : 'soun'),
|
||||
hdlr(trackData),
|
||||
minf(trackData)
|
||||
]);
|
||||
|
||||
@@ -288,15 +379,26 @@ export const mdhd = (
|
||||
]);
|
||||
};
|
||||
|
||||
const TRACK_TYPE_TO_COMPONENT_SUBTYPE: Record<IsobmffTrackData['type'], string> = {
|
||||
video: 'vide',
|
||||
audio: 'soun',
|
||||
subtitle: 'text'
|
||||
};
|
||||
|
||||
const TRACK_TYPE_TO_HANDLER_NAME: Record<IsobmffTrackData['type'], string> = {
|
||||
video: 'VideoHandler',
|
||||
audio: 'SoundHandler',
|
||||
subtitle: 'TextHandler'
|
||||
};
|
||||
|
||||
/** Handler Reference Box: Specifies the media handler component that is to be used to interpret the media's data. */
|
||||
export const hdlr = (componentSubtype: string) => fullBox('hdlr', 0, 0, [
|
||||
export const hdlr = (trackData: IsobmffTrackData) => fullBox('hdlr', 0, 0, [
|
||||
ascii('mhlr'), // Component type
|
||||
ascii(componentSubtype), // Component subtype
|
||||
ascii(TRACK_TYPE_TO_COMPONENT_SUBTYPE[trackData.type]), // Component subtype
|
||||
u32(0), // Component manufacturer
|
||||
u32(0), // Component flags
|
||||
u32(0), // Component flags mask
|
||||
// TODO:
|
||||
ascii('mp4-muxer-hdlr', true) // Component name
|
||||
ascii(TRACK_TYPE_TO_HANDLER_NAME[trackData.type], true) // Component name
|
||||
]);
|
||||
|
||||
/**
|
||||
@@ -304,7 +406,7 @@ export const hdlr = (componentSubtype: string) => fullBox('hdlr', 0, 0, [
|
||||
* information to map from media time to media data and to process the media data.
|
||||
*/
|
||||
export const minf = (trackData: IsobmffTrackData) => box('minf', undefined, [
|
||||
trackData.type === 'video' ? vmhd() : smhd(),
|
||||
TRACK_TYPE_TO_HEADER_BOX[trackData.type](),
|
||||
dinf(),
|
||||
stbl(trackData)
|
||||
]);
|
||||
@@ -323,6 +425,15 @@ export const smhd = () => fullBox('smhd', 0, 0, [
|
||||
u16(0) // Reserved
|
||||
]);
|
||||
|
||||
/** Null Media Header Box. */
|
||||
export const nmhd = () => fullBox('nmhd', 0, 0);
|
||||
|
||||
const TRACK_TYPE_TO_HEADER_BOX: Record<IsobmffTrackData['type'], () => Box> = {
|
||||
video: vmhd,
|
||||
audio: smhd,
|
||||
subtitle: nmhd
|
||||
};
|
||||
|
||||
/**
|
||||
* Data Information Box: Contains information specifying the data handler component that provides access to the
|
||||
* media data. The data handler component uses the Data Information Box to interpret the media's data.
|
||||
@@ -365,19 +476,34 @@ export const stbl = (trackData: IsobmffTrackData) => {
|
||||
* Sample Description Box: Stores information that allows you to decode samples in the media. The data stored in the
|
||||
* sample description varies, depending on the media type.
|
||||
*/
|
||||
export const stsd = (trackData: IsobmffTrackData) => fullBox('stsd', 0, 0, [
|
||||
u32(1) // Entry count
|
||||
], [
|
||||
trackData.type === 'video'
|
||||
? videoSampleDescription(
|
||||
export const stsd = (trackData: IsobmffTrackData) => {
|
||||
let sampleDescription: Box;
|
||||
|
||||
if (trackData.type === 'video') {
|
||||
sampleDescription = videoSampleDescription(
|
||||
VIDEO_CODEC_TO_BOX_NAME[trackData.track.source.codec],
|
||||
trackData as IsobmffVideoTrackData
|
||||
trackData
|
||||
)
|
||||
: soundSampleDescription(
|
||||
} else if (trackData.type === 'audio') {
|
||||
sampleDescription = soundSampleDescription(
|
||||
AUDIO_CODEC_TO_BOX_NAME[trackData.track.source.codec],
|
||||
trackData as IsobmffAudioTrackData
|
||||
)
|
||||
]);
|
||||
trackData
|
||||
);
|
||||
} else if (trackData.type === 'subtitle') {
|
||||
sampleDescription = subtitleSampleDescription(
|
||||
SUBTITLE_CODEC_TO_BOX_NAME[trackData.track.source.codec],
|
||||
trackData
|
||||
);
|
||||
}
|
||||
|
||||
assert(sampleDescription!);
|
||||
|
||||
return fullBox('stsd', 0, 0, [
|
||||
u32(1) // Entry count
|
||||
], [
|
||||
sampleDescription
|
||||
]);
|
||||
};
|
||||
|
||||
/** Video Sample Description Box: Contains information that defines how to interpret video media data. */
|
||||
export const videoSampleDescription = (
|
||||
@@ -400,6 +526,7 @@ export const videoSampleDescription = (
|
||||
i16(0xffff) // Pre-defined
|
||||
], [
|
||||
VIDEO_CODEC_TO_CONFIGURATION_BOX[trackData.track.source.codec](trackData)
|
||||
// TODO colr
|
||||
]);
|
||||
|
||||
// TODO: All muxers should ensure that the decoder config description is provided for the codecs that require it. This
|
||||
@@ -550,6 +677,24 @@ export const dOps = (trackData: IsobmffAudioTrackData) => {
|
||||
]);
|
||||
};
|
||||
|
||||
export const subtitleSampleDescription = (
|
||||
compressionType: string,
|
||||
trackData: IsobmffSubtitleTrackData
|
||||
) => box(compressionType, [
|
||||
Array(6).fill(0), // Reserved
|
||||
u16(1), // Data reference index
|
||||
], [
|
||||
SUBTITLE_CODEC_TO_CONFIGURATION_BOX[trackData.track.source.codec](trackData)
|
||||
])
|
||||
|
||||
export const vttC = (trackData: IsobmffSubtitleTrackData) => box('vttC', [
|
||||
...textEncoder.encode(trackData.info.config.description)
|
||||
]);
|
||||
|
||||
export const txtC = (textConfig: Uint8Array) => fullBox('txtC', 0, 0, [
|
||||
...textConfig, 0 // Text config (null-terminated)
|
||||
]);
|
||||
|
||||
/**
|
||||
* Time-To-Sample Box: Stores duration information for a media's samples, providing a mapping from a time in a media
|
||||
* to the corresponding data sample. The table is compact, meaning that consecutive samples with the same time delta
|
||||
@@ -817,6 +962,21 @@ export const mfro = () => {
|
||||
]);
|
||||
};
|
||||
|
||||
/** VTT Empty Cue Box */
|
||||
export const vtte = () => box('vtte');
|
||||
|
||||
/** VTT Cue Box */
|
||||
export const vttc = (payload: string, timestamp: number | null, identifier: string | null, settings: string | null, sourceId: number | null) => box('vttc', undefined, [
|
||||
sourceId !== null ? box('vsid', [i32(sourceId)]) : null,
|
||||
identifier !== null ? box('iden', [...textEncoder.encode(identifier)]) : null,
|
||||
timestamp !== null ? box('ctim', [...textEncoder.encode(formatSubtitleTimestamp(timestamp))]) : null,
|
||||
settings !== null ? box('sttg', [...textEncoder.encode(settings)]) : null,
|
||||
box('payl', [...textEncoder.encode(payload)])
|
||||
]);
|
||||
|
||||
/** VTT Additional Text Box */
|
||||
export const vtta = (notes: string) => box('vtta', [...textEncoder.encode(notes)]);
|
||||
|
||||
const VIDEO_CODEC_TO_BOX_NAME: Record<VideoCodec, string> = {
|
||||
'avc': 'avc1',
|
||||
'hevc': 'hvc1',
|
||||
@@ -842,3 +1002,11 @@ const AUDIO_CODEC_TO_CONFIGURATION_BOX: Record<AudioCodec, (trackData: IsobmffAu
|
||||
'aac': esds,
|
||||
'opus': dOps
|
||||
};
|
||||
|
||||
const SUBTITLE_CODEC_TO_BOX_NAME: Record<SubtitleCodec, string> = {
|
||||
'webvtt': 'wvtt'
|
||||
};
|
||||
|
||||
const SUBTITLE_CODEC_TO_CONFIGURATION_BOX: Record<SubtitleCodec, (trackData: IsobmffSubtitleTrackData) => Box | null> = {
|
||||
'webvtt': vttC
|
||||
};
|
||||
+276
-175
@@ -1,9 +1,11 @@
|
||||
import { Box, free, ftyp, mdat, mfra, moof, moov } from './isobmff_boxes';
|
||||
import { Box, free, ftyp, IsobmffBoxWriter, mdat, mfra, moof, moov, vtta, vttc, vtte } from './isobmff_boxes';
|
||||
import { Muxer } from '../muxer';
|
||||
import { Output, OutputAudioTrack, OutputTrack, OutputVideoTrack } from '../output';
|
||||
import { Output, OutputAudioTrack, OutputSubtitleTrack, OutputTrack, OutputVideoTrack } from '../output';
|
||||
import { Writer } from '../writer';
|
||||
import { assert, last, TransformationMatrix } from '../misc';
|
||||
import { Mp4OutputFormat } from '../output_format';
|
||||
import { inlineTimestampRegex, SubtitleConfig, SubtitleCue, SubtitleMetadata } from '../subtitles';
|
||||
import { ArrayBufferTarget } from '../target';
|
||||
|
||||
export const GLOBAL_TIMESCALE = 1000;
|
||||
const TIMESTAMP_OFFSET = 2_082_844_800; // Seconds between Jan 1 1904 and Jan 1 1970
|
||||
@@ -32,9 +34,6 @@ export type IsobmffTrackData = {
|
||||
sampleQueue: Sample[], // For fragmented files
|
||||
timestampProcessingQueue: Sample[],
|
||||
|
||||
firstTimestamp: number | null,
|
||||
lastKeyFrameTimestamp: number | null,
|
||||
|
||||
timeToSampleTable: { sampleCount: number, sampleDelta: number }[];
|
||||
compositionTimeOffsetTable: { sampleCount: number, sampleCompositionTimeOffset: number }[];
|
||||
lastTimescaleUnits: number | null,
|
||||
@@ -62,10 +61,21 @@ export type IsobmffTrackData = {
|
||||
sampleRate: number,
|
||||
decoderConfig: AudioDecoderConfig
|
||||
}
|
||||
} | {
|
||||
track: OutputSubtitleTrack,
|
||||
type: 'subtitle',
|
||||
info: {
|
||||
config: SubtitleConfig
|
||||
},
|
||||
lastCueEndTimestamp: number,
|
||||
cueQueue: SubtitleCue[],
|
||||
nextSourceId: number,
|
||||
cueToSourceId: WeakMap<SubtitleCue, number>
|
||||
});
|
||||
|
||||
export type IsobmffVideoTrackData = IsobmffTrackData & { type: 'video' };
|
||||
export type IsobmffAudioTrackData = IsobmffTrackData & { type: 'audio' };
|
||||
export type IsobmffSubtitleTrackData = IsobmffTrackData & { type: 'subtitle' };
|
||||
|
||||
export const intoTimescale = (timeInSeconds: number, timescale: number, round = true) => {
|
||||
let value = timeInSeconds * timescale;
|
||||
@@ -73,16 +83,15 @@ export const intoTimescale = (timeInSeconds: number, timescale: number, round =
|
||||
};
|
||||
|
||||
export class IsobmffMuxer extends Muxer {
|
||||
#writer: Writer;
|
||||
#format: Mp4OutputFormat;
|
||||
#helper = new Uint8Array(8);
|
||||
#helperView = new DataView(this.#helper.buffer);
|
||||
override timestampsMustStartAtZero = true;
|
||||
|
||||
/**
|
||||
* Stores the position from the start of the file to where boxes elements have been written. This is used to
|
||||
* rewrite/edit elements that were already added before, and to measure sizes of things.
|
||||
*/
|
||||
offsets = new WeakMap<Box, number>();
|
||||
#writer: Writer;
|
||||
#boxWriter: IsobmffBoxWriter;
|
||||
#format: Mp4OutputFormat;
|
||||
|
||||
#auxTarget = new ArrayBufferTarget();
|
||||
#auxWriter = this.#auxTarget.createWriter();
|
||||
#auxBoxWriter = new IsobmffBoxWriter(this.#auxWriter);
|
||||
|
||||
#ftypSize: number | null = null;
|
||||
#mdat: Box | null = null;
|
||||
@@ -98,90 +107,15 @@ export class IsobmffMuxer extends Muxer {
|
||||
super(output);
|
||||
|
||||
this.#writer = output.writer;
|
||||
this.#boxWriter = new IsobmffBoxWriter(this.#writer);
|
||||
this.#format = format;
|
||||
}
|
||||
|
||||
writeU32(value: number) {
|
||||
this.#helperView.setUint32(0, value, false);
|
||||
this.#writer.write(this.#helper.subarray(0, 4));
|
||||
}
|
||||
|
||||
writeU64(value: number) {
|
||||
this.#helperView.setUint32(0, Math.floor(value / 2**32), false);
|
||||
this.#helperView.setUint32(4, value, false);
|
||||
this.#writer.write(this.#helper.subarray(0, 8));
|
||||
}
|
||||
|
||||
writeAscii(text: string) {
|
||||
for (let i = 0; i < text.length; i++) {
|
||||
this.#helperView.setUint8(i % 8, text.charCodeAt(i));
|
||||
if (i % 8 === 7) this.#writer.write(this.#helper);
|
||||
}
|
||||
|
||||
if (text.length % 8 !== 0) {
|
||||
this.#writer.write(this.#helper.subarray(0, text.length % 8));
|
||||
}
|
||||
}
|
||||
|
||||
writeBox(box: Box) {
|
||||
this.offsets.set(box, this.#writer.getPos());
|
||||
|
||||
if (box.contents && !box.children) {
|
||||
this.writeBoxHeader(box, box.size ?? box.contents.byteLength + 8);
|
||||
this.#writer.write(box.contents);
|
||||
} else {
|
||||
let startPos = this.#writer.getPos();
|
||||
this.writeBoxHeader(box, 0);
|
||||
|
||||
if (box.contents) this.#writer.write(box.contents);
|
||||
if (box.children) for (let child of box.children) if (child) this.writeBox(child);
|
||||
|
||||
let endPos = this.#writer.getPos();
|
||||
let size = box.size ?? endPos - startPos;
|
||||
this.#writer.seek(startPos);
|
||||
this.writeBoxHeader(box, size);
|
||||
this.#writer.seek(endPos);
|
||||
}
|
||||
}
|
||||
|
||||
writeBoxHeader(box: Box, size: number) {
|
||||
this.writeU32(box.largeSize ? 1 : size);
|
||||
this.writeAscii(box.type);
|
||||
if (box.largeSize) this.writeU64(size);
|
||||
}
|
||||
|
||||
measureBoxHeader(box: Box) {
|
||||
return 8 + (box.largeSize ? 8 : 0);
|
||||
}
|
||||
|
||||
patchBox(box: Box) {
|
||||
const boxOffset = this.offsets.get(box);
|
||||
assert(boxOffset !== undefined);
|
||||
|
||||
let endPos = this.#writer.getPos();
|
||||
this.#writer.seek(boxOffset);
|
||||
this.writeBox(box);
|
||||
this.#writer.seek(endPos);
|
||||
}
|
||||
|
||||
measureBox(box: Box) {
|
||||
if (box.contents && !box.children) {
|
||||
let headerSize = this.measureBoxHeader(box);
|
||||
return headerSize + box.contents.byteLength;
|
||||
} else {
|
||||
let result = this.measureBoxHeader(box);
|
||||
if (box.contents) result += box.contents.byteLength;
|
||||
if (box.children) for (let child of box.children) if (child) result += this.measureBox(child);
|
||||
|
||||
return result;
|
||||
}
|
||||
}
|
||||
|
||||
start() {
|
||||
const holdsAvc = this.output.tracks.some(x => x.type === 'video' && x.source.codec === 'avc');
|
||||
|
||||
// Write the header
|
||||
this.writeBox(ftyp({
|
||||
this.#boxWriter.writeBox(ftyp({
|
||||
holdsAvc: holdsAvc,
|
||||
fragmented: this.#format.options.fastStart === 'fragmented'
|
||||
}));
|
||||
@@ -199,7 +133,7 @@ export class IsobmffMuxer extends Muxer {
|
||||
}
|
||||
|
||||
this.#mdat = mdat(true); // Reserve large size by default, can refine this when finalizing.
|
||||
this.writeBox(this.#mdat);
|
||||
this.#boxWriter.writeBox(this.#mdat);
|
||||
}
|
||||
|
||||
this.#writer.flush();
|
||||
@@ -240,7 +174,7 @@ export class IsobmffMuxer extends Muxer {
|
||||
#getVideoTrackData(track: OutputVideoTrack, meta?: EncodedVideoChunkMetadata) {
|
||||
const existingTrackData = this.#trackDatas.find(x => x.track === track);
|
||||
if (existingTrackData) {
|
||||
return existingTrackData;
|
||||
return existingTrackData as IsobmffVideoTrackData;
|
||||
}
|
||||
|
||||
// TODO Make proper errors for these
|
||||
@@ -249,7 +183,7 @@ export class IsobmffMuxer extends Muxer {
|
||||
assert(meta.decoderConfig.codedWidth !== undefined);
|
||||
assert(meta.decoderConfig.codedHeight !== undefined);
|
||||
|
||||
const newTrackData: IsobmffTrackData = {
|
||||
const newTrackData: IsobmffVideoTrackData = {
|
||||
track,
|
||||
type: 'video',
|
||||
info: {
|
||||
@@ -261,8 +195,6 @@ export class IsobmffMuxer extends Muxer {
|
||||
samples: [],
|
||||
sampleQueue: [],
|
||||
timestampProcessingQueue: [],
|
||||
firstTimestamp: null,
|
||||
lastKeyFrameTimestamp: null,
|
||||
timeToSampleTable: [],
|
||||
compositionTimeOffsetTable: [],
|
||||
lastTimescaleUnits: null,
|
||||
@@ -281,14 +213,14 @@ export class IsobmffMuxer extends Muxer {
|
||||
#getAudioTrackData(track: OutputAudioTrack, meta?: EncodedAudioChunkMetadata) {
|
||||
const existingTrackData = this.#trackDatas.find(x => x.track === track);
|
||||
if (existingTrackData) {
|
||||
return existingTrackData;
|
||||
return existingTrackData as IsobmffAudioTrackData;
|
||||
}
|
||||
|
||||
// TODO Make proper errors for these
|
||||
assert(meta);
|
||||
assert(meta.decoderConfig);
|
||||
|
||||
const newTrackData: IsobmffTrackData = {
|
||||
const newTrackData: IsobmffAudioTrackData = {
|
||||
track,
|
||||
type: 'audio',
|
||||
info: {
|
||||
@@ -300,8 +232,6 @@ export class IsobmffMuxer extends Muxer {
|
||||
samples: [],
|
||||
sampleQueue: [],
|
||||
timestampProcessingQueue: [],
|
||||
firstTimestamp: null,
|
||||
lastKeyFrameTimestamp: null,
|
||||
timeToSampleTable: [],
|
||||
compositionTimeOffsetTable: [],
|
||||
lastTimescaleUnits: null,
|
||||
@@ -317,6 +247,49 @@ export class IsobmffMuxer extends Muxer {
|
||||
return newTrackData;
|
||||
}
|
||||
|
||||
#getSubtitleTrackData(track: OutputSubtitleTrack, meta?: SubtitleMetadata) {
|
||||
const existingTrackData = this.#trackDatas.find(x => x.track === track);
|
||||
if (existingTrackData) {
|
||||
return existingTrackData as IsobmffSubtitleTrackData;
|
||||
}
|
||||
|
||||
// TODO Make proper errors for these
|
||||
assert(meta);
|
||||
assert(meta.config);
|
||||
|
||||
const newTrackData: IsobmffSubtitleTrackData = {
|
||||
track,
|
||||
type: 'subtitle',
|
||||
info: {
|
||||
config: meta.config
|
||||
},
|
||||
timescale: 1000, // Reasonable
|
||||
samples: [],
|
||||
sampleQueue: [],
|
||||
timestampProcessingQueue: [],
|
||||
timeToSampleTable: [],
|
||||
compositionTimeOffsetTable: [],
|
||||
lastTimescaleUnits: null,
|
||||
lastSample: null,
|
||||
finalizedChunks: [],
|
||||
currentChunk: null,
|
||||
compactlyCodedChunkTable: [],
|
||||
lastCueEndTimestamp: 0,
|
||||
cueQueue: [],
|
||||
nextSourceId: 0,
|
||||
cueToSourceId: new WeakMap()
|
||||
};
|
||||
|
||||
this.#trackDatas.push(newTrackData);
|
||||
this.#trackDatas.sort((a, b) => a.track.id - b.track.id);
|
||||
|
||||
// Subtitle cues don't need to start at 0, so let's register a timestamp at 0 to satisfy the
|
||||
// "timestamps must start a zero" constraint
|
||||
this.validateAndNormalizeTimestamp(track, 0, true);
|
||||
|
||||
return newTrackData;
|
||||
}
|
||||
|
||||
addEncodedVideoChunk(track: OutputVideoTrack, chunk: EncodedVideoChunk, meta?: EncodedVideoChunkMetadata) {
|
||||
const trackData = this.#getVideoTrackData(track, meta);
|
||||
|
||||
@@ -330,13 +303,17 @@ export class IsobmffMuxer extends Muxer {
|
||||
}).`);
|
||||
}
|
||||
|
||||
let videoSample = this.#createSampleForTrack(trackData, chunk);
|
||||
let data = new Uint8Array(chunk.byteLength);
|
||||
chunk.copyTo(data);
|
||||
|
||||
let timestamp = this.validateAndNormalizeTimestamp(trackData.track, chunk.timestamp, chunk.type === 'key');
|
||||
let sample = this.#createSampleForTrack(trackData, data, timestamp, (chunk.duration ?? 0) / 1e6, chunk.type);
|
||||
|
||||
if (this.#format.options.fastStart === 'fragmented') {
|
||||
trackData.sampleQueue.push(videoSample);
|
||||
trackData.sampleQueue.push(sample);
|
||||
this.#interleaveSamples();
|
||||
} else {
|
||||
this.#addSampleToTrack(trackData, videoSample);
|
||||
this.#addSampleToTrack(trackData, sample);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -353,35 +330,146 @@ export class IsobmffMuxer extends Muxer {
|
||||
}).`);
|
||||
}
|
||||
|
||||
let audioSample = this.#createSampleForTrack(trackData, chunk);
|
||||
let data = new Uint8Array(chunk.byteLength);
|
||||
chunk.copyTo(data);
|
||||
|
||||
let timestamp = this.validateAndNormalizeTimestamp(trackData.track, chunk.timestamp, chunk.type === 'key');
|
||||
let sample = this.#createSampleForTrack(trackData, data, timestamp, (chunk.duration ?? 0) / 1e6, chunk.type);
|
||||
|
||||
if (this.#format.options.fastStart === 'fragmented') {
|
||||
trackData.sampleQueue.push(audioSample);
|
||||
trackData.sampleQueue.push(sample);
|
||||
this.#interleaveSamples();
|
||||
} else {
|
||||
this.#addSampleToTrack(trackData, audioSample);
|
||||
this.#addSampleToTrack(trackData, sample);
|
||||
}
|
||||
}
|
||||
|
||||
addSubtitleCue(track: OutputSubtitleTrack, cue: SubtitleCue, meta?: SubtitleMetadata) {
|
||||
const trackData = this.#getSubtitleTrackData(track, meta);
|
||||
|
||||
// TODO the expectedSubtitleChunks thing
|
||||
|
||||
this.validateAndNormalizeTimestamp(trackData.track, 1e6 * cue.timestamp, true);
|
||||
|
||||
if (track.source.codec === 'webvtt') {
|
||||
trackData.cueQueue.push(cue);
|
||||
this.#processWebVTTCues(trackData, cue.timestamp);
|
||||
} else {
|
||||
// TODO
|
||||
}
|
||||
}
|
||||
|
||||
#processWebVTTCues(trackData: IsobmffSubtitleTrackData, until: number) {
|
||||
// WebVTT cues need to undergo special processing as empty sections need to be padded out with samples, and
|
||||
// overlapping samples require special logic. The algorithm produces the format specified in ISO 14496-30.
|
||||
|
||||
while (trackData.cueQueue.length > 0) {
|
||||
let timestamps = new Set<number>([]);
|
||||
for (let cue of trackData.cueQueue) {
|
||||
assert(cue.timestamp <= until);
|
||||
assert(trackData.lastCueEndTimestamp <= cue.timestamp + cue.duration);
|
||||
|
||||
timestamps.add(Math.max(cue.timestamp, trackData.lastCueEndTimestamp)); // Start timestamp
|
||||
timestamps.add(cue.timestamp + cue.duration); // End timestamp
|
||||
}
|
||||
|
||||
let sortedTimestamps = [...timestamps].sort((a, b) => a - b);
|
||||
|
||||
// These are the timestamps of the next sample we'll create:
|
||||
let sampleStart = sortedTimestamps[0]!;
|
||||
let sampleEnd = sortedTimestamps[1] ?? sampleStart;
|
||||
|
||||
if (until < sampleEnd) {
|
||||
break;
|
||||
}
|
||||
|
||||
// We may need to pad out empty space with an vtte box
|
||||
if (trackData.lastCueEndTimestamp < sampleStart) {
|
||||
this.#auxWriter.seek(0);
|
||||
let box = vtte();
|
||||
this.#auxBoxWriter.writeBox(box);
|
||||
|
||||
let body = this.#auxWriter.getSlice(0, this.#auxWriter.getPos());
|
||||
let sample = this.#createSampleForTrack(trackData, body, trackData.lastCueEndTimestamp, sampleStart - trackData.lastCueEndTimestamp, 'key');
|
||||
|
||||
// todo extract this into reusable thing
|
||||
if (this.#format.options.fastStart === 'fragmented') {
|
||||
trackData.sampleQueue.push(sample);
|
||||
this.#interleaveSamples();
|
||||
} else {
|
||||
this.#addSampleToTrack(trackData, sample);
|
||||
}
|
||||
|
||||
trackData.lastCueEndTimestamp = sampleStart;
|
||||
}
|
||||
|
||||
this.#auxWriter.seek(0);
|
||||
|
||||
for (let i = 0; i < trackData.cueQueue.length; i++) {
|
||||
let cue = trackData.cueQueue[i]!
|
||||
|
||||
if (cue.timestamp >= sampleEnd) {
|
||||
break;
|
||||
}
|
||||
|
||||
inlineTimestampRegex.lastIndex = 0;
|
||||
let containsTimestamp = inlineTimestampRegex.test(cue.text);
|
||||
|
||||
let endTimestamp = cue.timestamp + cue.duration;
|
||||
let sourceId = trackData.cueToSourceId.get(cue);
|
||||
if (sourceId === undefined && sampleEnd < endTimestamp) {
|
||||
// We know this cue will appear in more than one sample, therefore we need to mark it with a
|
||||
// unique ID
|
||||
sourceId = trackData.nextSourceId++;
|
||||
trackData.cueToSourceId.set(cue, sourceId);
|
||||
}
|
||||
|
||||
if (cue.notes) {
|
||||
// Any notes/comments are included in a special vtta box
|
||||
let box = vtta(cue.notes);
|
||||
this.#auxBoxWriter.writeBox(box);
|
||||
}
|
||||
|
||||
let box = vttc(cue.text, containsTimestamp ? sampleStart : null, cue.identifier ?? null, cue.settings ?? null, sourceId ?? null);
|
||||
this.#auxBoxWriter.writeBox(box);
|
||||
|
||||
if (endTimestamp === sampleEnd) {
|
||||
// The cue won't appear in any future sample, so we're done with it
|
||||
trackData.cueQueue.splice(i--, 1);
|
||||
}
|
||||
}
|
||||
|
||||
let body = this.#auxWriter.getSlice(0, this.#auxWriter.getPos());
|
||||
let sample = this.#createSampleForTrack(trackData, body, sampleStart, sampleEnd - sampleStart, 'key');
|
||||
|
||||
// todo extract this into reusable thing
|
||||
if (this.#format.options.fastStart === 'fragmented') {
|
||||
trackData.sampleQueue.push(sample);
|
||||
this.#interleaveSamples();
|
||||
} else {
|
||||
this.#addSampleToTrack(trackData, sample);
|
||||
}
|
||||
|
||||
trackData.lastCueEndTimestamp = sampleEnd;
|
||||
}
|
||||
}
|
||||
|
||||
#createSampleForTrack(
|
||||
trackData: IsobmffTrackData,
|
||||
chunk: EncodedVideoChunk | EncodedAudioChunk
|
||||
data: Uint8Array,
|
||||
timestamp: number,
|
||||
duration: number,
|
||||
type: 'key' | 'delta'
|
||||
) {
|
||||
let timestampInSeconds = this.#validateTimestamp(trackData, chunk);
|
||||
let durationInSeconds = (chunk.duration ?? 0) / 1e6;
|
||||
|
||||
let data = new Uint8Array(chunk.byteLength);
|
||||
chunk.copyTo(data);
|
||||
|
||||
let sample: Sample = {
|
||||
timestamp: timestampInSeconds,
|
||||
decodeTimestamp: timestampInSeconds, // We may refine this later
|
||||
duration: durationInSeconds,
|
||||
data: data,
|
||||
timestamp,
|
||||
decodeTimestamp: timestamp, // This may be refined later
|
||||
duration,
|
||||
data,
|
||||
size: data.byteLength,
|
||||
type: chunk.type,
|
||||
type,
|
||||
// Will be refined once the next sample comes in
|
||||
timescaleUnitsToNextSample: intoTimescale(durationInSeconds, trackData.timescale)
|
||||
timescaleUnitsToNextSample: intoTimescale(duration, trackData.timescale)
|
||||
};
|
||||
|
||||
return sample;
|
||||
@@ -405,6 +493,7 @@ export class IsobmffMuxer extends Muxer {
|
||||
|
||||
const sampleCompositionTimeOffset =
|
||||
intoTimescale(sample.timestamp - sample.decodeTimestamp, trackData.timescale);
|
||||
const durationInTimescale = intoTimescale(sample.duration, trackData.timescale);
|
||||
|
||||
if (trackData.lastTimescaleUnits !== null) {
|
||||
assert(trackData.lastSample);
|
||||
@@ -417,20 +506,34 @@ export class IsobmffMuxer extends Muxer {
|
||||
if (this.#format.options.fastStart !== 'fragmented') {
|
||||
let lastTableEntry = last(trackData.timeToSampleTable);
|
||||
assert(lastTableEntry);
|
||||
|
||||
|
||||
if (lastTableEntry.sampleCount === 1) {
|
||||
// If we hit this case, we're the second sample
|
||||
lastTableEntry.sampleDelta = delta;
|
||||
lastTableEntry.sampleCount++;
|
||||
} else if (lastTableEntry.sampleDelta === delta) {
|
||||
// Simply increment the count
|
||||
|
||||
let entryBefore = trackData.timeToSampleTable[trackData.timeToSampleTable.length - 2];
|
||||
if (entryBefore && entryBefore.sampleDelta === delta) {
|
||||
// If the delta is the same as the previous one, merge the two entries
|
||||
entryBefore.sampleCount++;
|
||||
trackData.timeToSampleTable.pop();
|
||||
lastTableEntry = entryBefore;
|
||||
}
|
||||
} else if (lastTableEntry.sampleDelta !== delta) {
|
||||
// The delta has changed, so we need a new entry to reach the current sample
|
||||
lastTableEntry.sampleCount--;
|
||||
trackData.timeToSampleTable.push(lastTableEntry = {
|
||||
sampleCount: 1,
|
||||
sampleDelta: delta
|
||||
});
|
||||
}
|
||||
|
||||
if (lastTableEntry.sampleDelta === durationInTimescale) {
|
||||
// The sample's duration matches the delta, so we can increment the count
|
||||
lastTableEntry.sampleCount++;
|
||||
} else {
|
||||
// The delta has changed, subtract one from the previous run and create a new run with the new delta
|
||||
lastTableEntry.sampleCount--;
|
||||
// Add a new entry in order to maintain the last sample's true duration
|
||||
trackData.timeToSampleTable.push({
|
||||
sampleCount: 2,
|
||||
sampleDelta: delta
|
||||
sampleCount: 1,
|
||||
sampleDelta: durationInTimescale
|
||||
});
|
||||
}
|
||||
|
||||
@@ -455,7 +558,7 @@ export class IsobmffMuxer extends Muxer {
|
||||
if (this.#format.options.fastStart !== 'fragmented') {
|
||||
trackData.timeToSampleTable.push({
|
||||
sampleCount: 1,
|
||||
sampleDelta: intoTimescale(sample.duration, trackData.timescale)
|
||||
sampleDelta: durationInTimescale
|
||||
});
|
||||
trackData.compositionTimeOffsetTable.push({
|
||||
sampleCount: 1,
|
||||
@@ -527,30 +630,6 @@ export class IsobmffMuxer extends Muxer {
|
||||
trackData.timestampProcessingQueue.push(sample);
|
||||
}
|
||||
|
||||
#validateTimestamp(trackData: IsobmffTrackData, chunk: EncodedVideoChunk | EncodedAudioChunk) {
|
||||
let timestampInSeconds = chunk.timestamp / 1e6;
|
||||
|
||||
if (timestampInSeconds < 0) {
|
||||
throw new Error(`Timestamps must be non-negative (got ${timestampInSeconds}s).`);
|
||||
}
|
||||
|
||||
if (trackData.firstTimestamp === null) {
|
||||
trackData.firstTimestamp = timestampInSeconds;
|
||||
}
|
||||
|
||||
timestampInSeconds -= trackData.firstTimestamp;
|
||||
|
||||
if (trackData.lastKeyFrameTimestamp !== null && timestampInSeconds < trackData.lastKeyFrameTimestamp) {
|
||||
throw new Error(`Timestamp cannot be before last key frame's timestamp (got ${timestampInSeconds}s, last key frame at ${trackData.lastKeyFrameTimestamp}s).`);
|
||||
}
|
||||
|
||||
if (chunk.type === 'key') {
|
||||
trackData.lastKeyFrameTimestamp = timestampInSeconds;
|
||||
}
|
||||
|
||||
return timestampInSeconds;
|
||||
}
|
||||
|
||||
#finalizeCurrentChunk(trackData: IsobmffTrackData) {
|
||||
assert(this.#format.options.fastStart !== 'fragmented');
|
||||
|
||||
@@ -588,8 +667,10 @@ export class IsobmffMuxer extends Muxer {
|
||||
#interleaveSamples() {
|
||||
assert(this.#format.options.fastStart === 'fragmented');
|
||||
|
||||
if (this.#trackDatas.length < this.output.tracks.length) {
|
||||
return; // We haven't seen a sample from each track yet
|
||||
for (const track of this.output.tracks) {
|
||||
if (!track.source.closed && !this.#trackDatas.some(x => x.track === track)) {
|
||||
return; // We haven't seen a sample from this open track yet
|
||||
}
|
||||
}
|
||||
|
||||
outer:
|
||||
@@ -598,11 +679,11 @@ export class IsobmffMuxer extends Muxer {
|
||||
let minTimestamp = Infinity;
|
||||
|
||||
for (let trackData of this.#trackDatas) {
|
||||
if (trackData.sampleQueue.length === 0) {
|
||||
if (trackData.sampleQueue.length === 0 && !trackData.track.source.closed) {
|
||||
break outer;
|
||||
}
|
||||
|
||||
if (trackData.sampleQueue[0]!.timestamp < minTimestamp) {
|
||||
if (trackData.sampleQueue.length > 0 && trackData.sampleQueue[0]!.timestamp < minTimestamp) {
|
||||
trackWithMinTimestamp = trackData;
|
||||
minTimestamp = trackData.sampleQueue[0]!.timestamp;
|
||||
}
|
||||
@@ -625,13 +706,13 @@ export class IsobmffMuxer extends Muxer {
|
||||
if (fragmentNumber === 1) {
|
||||
// Write the moov box now that we have all decoder configs
|
||||
let movieBox = moov(this.#trackDatas, this.#creationTime, true);
|
||||
this.writeBox(movieBox);
|
||||
this.#boxWriter.writeBox(movieBox);
|
||||
}
|
||||
|
||||
// Write out an initial moof box; will be overwritten later once actual chunk offsets are known
|
||||
let moofOffset = this.#writer.getPos();
|
||||
let moofBox = moof(fragmentNumber, this.#trackDatas);
|
||||
this.writeBox(moofBox);
|
||||
this.#boxWriter.writeBox(moofBox);
|
||||
|
||||
// Create the mdat box
|
||||
{
|
||||
@@ -646,15 +727,15 @@ export class IsobmffMuxer extends Muxer {
|
||||
}
|
||||
}
|
||||
|
||||
let mdatSize = this.measureBox(mdatBox) + totalTrackSampleSize;
|
||||
let mdatSize = this.#boxWriter.measureBox(mdatBox) + totalTrackSampleSize;
|
||||
if (mdatSize >= 2**32) {
|
||||
// Fragment is larger than 4 GiB, we need to use the large size
|
||||
mdatBox.largeSize = true;
|
||||
mdatSize = this.measureBox(mdatBox) + totalTrackSampleSize;
|
||||
mdatSize = this.#boxWriter.measureBox(mdatBox) + totalTrackSampleSize;
|
||||
}
|
||||
|
||||
mdatBox.size = mdatSize;
|
||||
this.writeBox(mdatBox);
|
||||
this.#boxWriter.writeBox(mdatBox);
|
||||
}
|
||||
|
||||
// Write sample data
|
||||
@@ -670,9 +751,9 @@ export class IsobmffMuxer extends Muxer {
|
||||
|
||||
// Now that we set the actual chunk offsets, fix the moof box
|
||||
let endPos = this.#writer.getPos();
|
||||
this.#writer.seek(this.offsets.get(moofBox)!);
|
||||
this.#writer.seek(this.#boxWriter.offsets.get(moofBox)!);
|
||||
let newMoofBox = moof(fragmentNumber, this.#trackDatas);
|
||||
this.writeBox(newMoofBox);
|
||||
this.#boxWriter.writeBox(newMoofBox);
|
||||
this.#writer.seek(endPos);
|
||||
|
||||
for (let trackData of this.#trackDatas) {
|
||||
@@ -686,8 +767,28 @@ export class IsobmffMuxer extends Muxer {
|
||||
}
|
||||
}
|
||||
|
||||
override onTrackClose(track: OutputTrack) {
|
||||
if (track.type === 'subtitle' && track.source.codec === 'webvtt') {
|
||||
let trackData = this.#trackDatas.find(x => x.track === track) as IsobmffSubtitleTrackData;
|
||||
if (trackData) {
|
||||
this.#processWebVTTCues(trackData, Infinity);
|
||||
}
|
||||
}
|
||||
|
||||
if (this.#format.options.fastStart === 'fragmented') {
|
||||
// Since a track is now closed, we may be able to write out chunks that were previously waiting
|
||||
this.#interleaveSamples();
|
||||
}
|
||||
}
|
||||
|
||||
/** Finalizes the file, making it ready for use. Must be called after all video and audio chunks have been added. */
|
||||
finalize() {
|
||||
for (let trackData of this.#trackDatas) {
|
||||
if (trackData.type === 'subtitle' && trackData.track.source.codec === 'webvtt') {
|
||||
this.#processWebVTTCues(trackData, Infinity);
|
||||
}
|
||||
}
|
||||
|
||||
if (this.#format.options.fastStart === 'fragmented') {
|
||||
for (let trackData of this.#trackDatas) {
|
||||
for (let sample of trackData.sampleQueue) {
|
||||
@@ -719,8 +820,8 @@ export class IsobmffMuxer extends Muxer {
|
||||
|
||||
for (let i = 0; i < 2; i++) {
|
||||
let movieBox = moov(this.#trackDatas, this.#creationTime);
|
||||
let movieBoxSize = this.measureBox(movieBox);
|
||||
mdatSize = this.measureBox(this.#mdat);
|
||||
let movieBoxSize = this.#boxWriter.measureBox(movieBox);
|
||||
mdatSize = this.#boxWriter.measureBox(this.#mdat);
|
||||
let currentChunkPos = this.#writer.getPos() + movieBoxSize + mdatSize;
|
||||
|
||||
for (let chunk of this.#finalizedChunks) {
|
||||
@@ -737,10 +838,10 @@ export class IsobmffMuxer extends Muxer {
|
||||
}
|
||||
|
||||
let movieBox = moov(this.#trackDatas, this.#creationTime);
|
||||
this.writeBox(movieBox);
|
||||
this.#boxWriter.writeBox(movieBox);
|
||||
|
||||
this.#mdat.size = mdatSize!;
|
||||
this.writeBox(this.#mdat);
|
||||
this.#boxWriter.writeBox(this.#mdat);
|
||||
|
||||
for (let chunk of this.#finalizedChunks) {
|
||||
for (let sample of chunk.samples) {
|
||||
@@ -753,33 +854,33 @@ export class IsobmffMuxer extends Muxer {
|
||||
// Append the mfra box to the end of the file for better random access
|
||||
let startPos = this.#writer.getPos();
|
||||
let mfraBox = mfra(this.#trackDatas);
|
||||
this.writeBox(mfraBox);
|
||||
this.#boxWriter.writeBox(mfraBox);
|
||||
|
||||
// Patch the 'size' field of the mfro box at the end of the mfra box now that we know its actual size
|
||||
let mfraBoxSize = this.#writer.getPos() - startPos;
|
||||
this.#writer.seek(this.#writer.getPos() - 4);
|
||||
this.writeU32(mfraBoxSize);
|
||||
this.#boxWriter.writeU32(mfraBoxSize);
|
||||
} else {
|
||||
assert(this.#mdat);
|
||||
assert(this.#ftypSize !== null);
|
||||
|
||||
let mdatPos = this.offsets.get(this.#mdat);
|
||||
let mdatPos = this.#boxWriter.offsets.get(this.#mdat);
|
||||
assert(mdatPos !== undefined);
|
||||
let mdatSize = this.#writer.getPos() - mdatPos;
|
||||
this.#mdat.size = mdatSize;
|
||||
this.#mdat.largeSize = mdatSize >= 2**32; // Only use the large size if we need it
|
||||
this.patchBox(this.#mdat);
|
||||
this.#boxWriter.patchBox(this.#mdat);
|
||||
|
||||
let movieBox = moov(this.#trackDatas, this.#creationTime);
|
||||
|
||||
if (typeof this.#format.options.fastStart === 'object') {
|
||||
this.#writer.seek(this.#ftypSize);
|
||||
this.writeBox(movieBox);
|
||||
this.#boxWriter.writeBox(movieBox);
|
||||
|
||||
let remainingBytes = mdatPos - this.#writer.getPos();
|
||||
this.writeBox(free(remainingBytes));
|
||||
this.#boxWriter.writeBox(free(remainingBytes));
|
||||
} else {
|
||||
this.writeBox(movieBox);
|
||||
this.#boxWriter.writeBox(movieBox);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -63,7 +63,6 @@ export enum EBMLId {
|
||||
Video = 0xe0,
|
||||
PixelWidth = 0xb0,
|
||||
PixelHeight = 0xba,
|
||||
Void = 0xec,
|
||||
Audio = 0xe1,
|
||||
SamplingFrequency = 0xb5,
|
||||
Channels = 0x9f,
|
||||
@@ -73,6 +72,9 @@ export enum EBMLId {
|
||||
BlockGroup = 0xa0,
|
||||
Block = 0xa1,
|
||||
BlockAdditions = 0x75a1,
|
||||
BlockMore = 0xa6,
|
||||
BlockAdditional = 0xa5,
|
||||
BlockAddID = 0xee,
|
||||
BlockDuration = 0x9b,
|
||||
ReferenceBlock = 0xfb,
|
||||
Cluster = 0x1f43b675,
|
||||
@@ -87,8 +89,7 @@ export enum EBMLId {
|
||||
MatrixCoefficients = 0x55b1,
|
||||
TransferCharacteristics = 0x55ba,
|
||||
Primaries = 0x55bb,
|
||||
Range = 0x55b9,
|
||||
AlphaMode = 0x53c0
|
||||
Range = 0x55b9
|
||||
}
|
||||
|
||||
export const measureUnsignedInt = (value: number) => {
|
||||
|
||||
@@ -1,9 +1,9 @@
|
||||
import { assert, readBits, toUint8Array, writeBits } from '../misc';
|
||||
import { assert, readBits, textEncoder, toUint8Array, writeBits } from '../misc';
|
||||
import { Muxer } from '../muxer';
|
||||
import { Output, OutputAudioTrack, OutputSubtitleTrack, OutputTrack, OutputVideoTrack } from '../output';
|
||||
import { MkvOutputFormat, WebMOutputFormat } from '../output_format';
|
||||
import { AudioCodec, SubtitleCodec, VideoCodec } from '../source';
|
||||
import { EncodedSubtitleChunk, EncodedSubtitleChunkMetadata, SubtitleDecoderConfig } from '../subtitles';
|
||||
import { formatSubtitleTimestamp, inlineTimestampRegex, parseSubtitleTimestamp, SubtitleConfig, SubtitleCue, SubtitleMetadata } from '../subtitles';
|
||||
import { Writer } from '../writer';
|
||||
import { EBML, EBMLElement, EBMLFloat32, EBMLFloat64, EBMLId, EBMLSignedInt, measureEBMLVarInt, measureSignedInt, measureUnsignedInt } from './ebml';
|
||||
|
||||
@@ -38,9 +38,6 @@ type SeekHead = {
|
||||
|
||||
type MatroskaTrackData = {
|
||||
chunkQueue: InternalMediaChunk[],
|
||||
|
||||
firstTimestamp: number | null,
|
||||
lastKeyFrameTimestamp: number | null,
|
||||
lastWrittenMsTimestamp: number | null
|
||||
} & ({
|
||||
track: OutputVideoTrack,
|
||||
@@ -62,7 +59,7 @@ type MatroskaTrackData = {
|
||||
track: OutputSubtitleTrack,
|
||||
type: 'subtitle',
|
||||
info: {
|
||||
decoderConfig: SubtitleDecoderConfig
|
||||
config: SubtitleConfig
|
||||
}
|
||||
});
|
||||
|
||||
@@ -78,7 +75,6 @@ const CODEC_STRING_MAP: Record<VideoCodec | AudioCodec | SubtitleCodec, string>
|
||||
av1: 'V_AV1',
|
||||
aac: 'A_AAC',
|
||||
opus: 'A_OPUS',
|
||||
vorbis: 'A_VORBIS',
|
||||
webvtt: 'S_TEXT/WEBVTT'
|
||||
};
|
||||
|
||||
@@ -94,6 +90,8 @@ const TRACK_TYPE_MAP: Record<OutputTrack['type'], number> = {
|
||||
// Update: Not really. There are duration fields and seek fields that are just uneditable if streaming is required.
|
||||
|
||||
export class MatroskaMuxer extends Muxer {
|
||||
override timestampsMustStartAtZero = true;
|
||||
|
||||
#writer: Writer;
|
||||
#format: WebMOutputFormat | MkvOutputFormat;
|
||||
|
||||
@@ -391,8 +389,8 @@ export class MatroskaMuxer extends Muxer {
|
||||
{ id: EBMLId.TrackUID, data: trackData.track.id },
|
||||
{ id: EBMLId.TrackType, data: TRACK_TYPE_MAP[trackData.type] }, // TODO Subtitle case
|
||||
{ id: EBMLId.CodecID, data: CODEC_STRING_MAP[trackData.track.source.codec] },
|
||||
(trackData.info.decoderConfig.description ? { id: EBMLId.CodecPrivate, data: toUint8Array(trackData.info.decoderConfig.description) } : null),
|
||||
...(trackData.type === 'video' ? [
|
||||
(trackData.info.decoderConfig.description ? { id: EBMLId.CodecPrivate, data: toUint8Array(trackData.info.decoderConfig.description) } : null),
|
||||
(trackData.track.metadata.frameRate ? { id: EBMLId.DefaultDuration, data: 1e9 / trackData.track.metadata.frameRate } : null),
|
||||
{ id: EBMLId.Video, data: [
|
||||
{ id: EBMLId.PixelWidth, data: trackData.info.width },
|
||||
@@ -430,12 +428,16 @@ export class MatroskaMuxer extends Muxer {
|
||||
] }
|
||||
] : []),
|
||||
...(trackData.type === 'audio' ? [
|
||||
(trackData.info.decoderConfig.description ? { id: EBMLId.CodecPrivate, data: toUint8Array(trackData.info.decoderConfig.description) } : null),
|
||||
{ id: EBMLId.Audio, data: [
|
||||
{ id: EBMLId.SamplingFrequency, data: new EBMLFloat32(trackData.info.sampleRate) },
|
||||
{ id: EBMLId.Channels, data: trackData.info.numberOfChannels },
|
||||
// Bit depth for when PCM is a thing
|
||||
] }
|
||||
] : [])
|
||||
] : []),
|
||||
...(trackData.type === 'subtitle' ? [
|
||||
{ id: EBMLId.CodecPrivate, data: textEncoder.encode(trackData.info.config.description) }
|
||||
] : []),
|
||||
] })
|
||||
}
|
||||
}
|
||||
@@ -492,8 +494,6 @@ export class MatroskaMuxer extends Muxer {
|
||||
decoderConfig: meta.decoderConfig
|
||||
},
|
||||
chunkQueue: [],
|
||||
firstTimestamp: null,
|
||||
lastKeyFrameTimestamp: null,
|
||||
lastWrittenMsTimestamp: null
|
||||
};
|
||||
|
||||
@@ -522,8 +522,6 @@ export class MatroskaMuxer extends Muxer {
|
||||
decoderConfig: meta.decoderConfig
|
||||
},
|
||||
chunkQueue: [],
|
||||
firstTimestamp: null,
|
||||
lastKeyFrameTimestamp: null,
|
||||
lastWrittenMsTimestamp: null
|
||||
};
|
||||
|
||||
@@ -533,7 +531,7 @@ export class MatroskaMuxer extends Muxer {
|
||||
return newTrackData;
|
||||
}
|
||||
|
||||
#getSubtitleTrackData(track: OutputSubtitleTrack, meta?: EncodedSubtitleChunkMetadata) {
|
||||
#getSubtitleTrackData(track: OutputSubtitleTrack, meta?: SubtitleMetadata) {
|
||||
const existingTrackData = this.#trackDatas.find(x => x.track === track);
|
||||
if (existingTrackData) {
|
||||
return existingTrackData as MatroskaAudioTrackData;
|
||||
@@ -541,17 +539,15 @@ export class MatroskaMuxer extends Muxer {
|
||||
|
||||
// TODO Make proper errors for these
|
||||
assert(meta);
|
||||
assert(meta.decoderConfig);
|
||||
assert(meta.config);
|
||||
|
||||
const newTrackData: MatroskaSubtitleTrackData = {
|
||||
track,
|
||||
type: 'subtitle',
|
||||
info: {
|
||||
decoderConfig: meta.decoderConfig
|
||||
config: meta.config
|
||||
},
|
||||
chunkQueue: [],
|
||||
firstTimestamp: null,
|
||||
lastKeyFrameTimestamp: null,
|
||||
lastWrittenMsTimestamp: null
|
||||
};
|
||||
|
||||
@@ -567,7 +563,8 @@ export class MatroskaMuxer extends Muxer {
|
||||
let data = new Uint8Array(chunk.byteLength);
|
||||
chunk.copyTo(data);
|
||||
|
||||
let videoChunk = this.#createInternalChunk(trackData, data, chunk.timestamp, chunk.duration ?? 0, chunk.type);
|
||||
let timestamp = this.validateAndNormalizeTimestamp(trackData.track, chunk.timestamp, chunk.type === 'key');
|
||||
let videoChunk = this.#createInternalChunk(data, timestamp, (chunk.duration ?? 0) / 1e6, chunk.type);
|
||||
if (track.source.codec === 'vp9') this.#fixVP9ColorSpace(trackData, videoChunk);
|
||||
|
||||
trackData.chunkQueue.push(videoChunk);
|
||||
@@ -580,27 +577,44 @@ export class MatroskaMuxer extends Muxer {
|
||||
let data = new Uint8Array(chunk.byteLength);
|
||||
chunk.copyTo(data);
|
||||
|
||||
let audioChunk = this.#createInternalChunk(trackData, data, chunk.timestamp, chunk.duration ?? 0, chunk.type);
|
||||
let timestamp = this.validateAndNormalizeTimestamp(trackData.track, chunk.timestamp, chunk.type === 'key');
|
||||
let audioChunk = this.#createInternalChunk(data, timestamp, (chunk.duration ?? 0) / 1e6, chunk.type);
|
||||
|
||||
trackData.chunkQueue.push(audioChunk);
|
||||
this.#interleaveChunks();
|
||||
}
|
||||
|
||||
addEncodedSubtitleChunk(track: OutputSubtitleTrack, chunk: EncodedSubtitleChunk, meta?: EncodedSubtitleChunkMetadata) {
|
||||
addSubtitleCue(track: OutputSubtitleTrack, cue: SubtitleCue, meta?: SubtitleMetadata) {
|
||||
const trackData = this.#getSubtitleTrackData(track, meta);
|
||||
|
||||
let subtitleChunk = this.#createInternalChunk(trackData, chunk.body, chunk.timestamp, chunk.duration, 'key', chunk.additions);
|
||||
const timestamp = this.validateAndNormalizeTimestamp(trackData.track, 1e6 * cue.timestamp, true);
|
||||
|
||||
let bodyText = cue.text;
|
||||
const timestampMs = Math.floor(timestamp * 1000);
|
||||
|
||||
// Replace in-body timestamps so that they're relative to the cue start time
|
||||
inlineTimestampRegex.lastIndex = 0;
|
||||
bodyText = bodyText.replace(inlineTimestampRegex, (match) => {
|
||||
let time = parseSubtitleTimestamp(match.slice(1, -1));
|
||||
let offsetTime = time - timestampMs;
|
||||
|
||||
return `<${formatSubtitleTimestamp(offsetTime)}>`;
|
||||
});
|
||||
|
||||
const body = textEncoder.encode(bodyText);
|
||||
const additions = `${cue.settings ?? ''}\n${cue.identifier ?? ''}\n${cue.notes ?? ''}`;
|
||||
|
||||
let subtitleChunk = this.#createInternalChunk(body, timestamp, cue.duration, 'key', additions.trim() ? textEncoder.encode(additions) : null);
|
||||
|
||||
trackData.chunkQueue.push(subtitleChunk);
|
||||
this.#interleaveChunks();
|
||||
}
|
||||
|
||||
#interleaveChunks() {
|
||||
let openTrackCount = 0;
|
||||
for (const trackData of this.#trackDatas) if (!trackData.track.source.closed) openTrackCount++;
|
||||
|
||||
if (this.#trackDatas.length < openTrackCount) {
|
||||
return; // We haven't seen a sample from every open track yet
|
||||
for (const track of this.output.tracks) {
|
||||
if (!track.source.closed && !this.#trackDatas.some(x => x.track === track)) {
|
||||
return; // We haven't seen a sample from this open track yet
|
||||
}
|
||||
}
|
||||
|
||||
outer:
|
||||
@@ -668,50 +682,23 @@ export class MatroskaMuxer extends Muxer {
|
||||
|
||||
/** Converts a read-only external chunk into an internal one for easier use. */
|
||||
#createInternalChunk(
|
||||
trackData: MatroskaTrackData,
|
||||
data: Uint8Array,
|
||||
timestamp: number,
|
||||
duration: number,
|
||||
type: 'key' | 'delta',
|
||||
additions: Uint8Array | null = null
|
||||
) {
|
||||
let adjustedTimestamp = this.#validateTimestamp(trackData, timestamp, type === 'key');
|
||||
|
||||
let internalChunk: InternalMediaChunk = {
|
||||
data,
|
||||
type,
|
||||
timestamp: adjustedTimestamp,
|
||||
duration: duration / 1e6,
|
||||
timestamp,
|
||||
duration,
|
||||
additions
|
||||
};
|
||||
|
||||
return internalChunk;
|
||||
}
|
||||
|
||||
#validateTimestamp(trackData: MatroskaTrackData, timestamp: number, isKeyFrame: boolean) {
|
||||
let timestampInSeconds = timestamp / 1e6;
|
||||
|
||||
if (timestampInSeconds < 0) {
|
||||
throw new Error(`Timestamps must be non-negative (got ${timestampInSeconds}s).`);
|
||||
}
|
||||
|
||||
if (trackData.firstTimestamp === null) {
|
||||
trackData.firstTimestamp = timestampInSeconds;
|
||||
}
|
||||
|
||||
timestampInSeconds -= trackData.firstTimestamp;
|
||||
|
||||
if (trackData.lastKeyFrameTimestamp !== null && timestampInSeconds < trackData.lastKeyFrameTimestamp) {
|
||||
throw new Error(`Timestamp cannot be before last key frame's timestamp (got ${timestampInSeconds}s, last key frame at ${trackData.lastKeyFrameTimestamp}s).`);
|
||||
}
|
||||
|
||||
if (isKeyFrame) {
|
||||
trackData.lastKeyFrameTimestamp = timestampInSeconds;
|
||||
}
|
||||
|
||||
return timestampInSeconds;
|
||||
}
|
||||
|
||||
/** Writes a block containing media data to the file. */
|
||||
#writeBlock(trackData: MatroskaTrackData, chunk: InternalMediaChunk) {
|
||||
// TODO Update this comment. This code always runs now
|
||||
@@ -726,6 +713,10 @@ export class MatroskaMuxer extends Muxer {
|
||||
// We can only finalize this fragment (and begin a new one) if we know that each track will be able to
|
||||
// start the new one with a key frame.
|
||||
const keyFrameQueuedEverywhere = this.#trackDatas.every(otherTrackData => {
|
||||
if (otherTrackData.track.source.closed) {
|
||||
return true;
|
||||
}
|
||||
|
||||
if (trackData === otherTrackData) {
|
||||
return chunk.type === 'key';
|
||||
}
|
||||
@@ -780,8 +771,13 @@ export class MatroskaMuxer extends Muxer {
|
||||
chunk.data
|
||||
] },
|
||||
chunk.type === 'delta' ? { id: EBMLId.ReferenceBlock, data: new EBMLSignedInt(trackData.lastWrittenMsTimestamp! - msTimestamp) } : null,
|
||||
msDuration > 0 ? { id: EBMLId.BlockDuration, data: msDuration } : null,
|
||||
chunk.additions ? { id: EBMLId.BlockAdditions, data: chunk.additions } : null
|
||||
chunk.additions ? { id: EBMLId.BlockAdditions, data: [
|
||||
{ id: EBMLId.BlockMore, data: [
|
||||
{ id: EBMLId.BlockAdditional, data: chunk.additions },
|
||||
{ id: EBMLId.BlockAddID, data: 1 }
|
||||
] }
|
||||
] } : null,
|
||||
msDuration > 0 ? { id: EBMLId.BlockDuration, data: msDuration } : null
|
||||
] };
|
||||
this.writeEBML(blockGroup);
|
||||
}
|
||||
|
||||
+3
-1
@@ -48,4 +48,6 @@ export const toUint8Array = (source: AllowSharedBufferSource): Uint8Array => {
|
||||
} else {
|
||||
return new Uint8Array(source.buffer, source.byteOffset, source.byteLength);
|
||||
}
|
||||
};
|
||||
};
|
||||
|
||||
export const textEncoder = new TextEncoder();
|
||||
+55
-2
@@ -1,5 +1,5 @@
|
||||
import { Output, OutputAudioTrack, OutputSubtitleTrack, OutputTrack, OutputVideoTrack } from "./output";
|
||||
import { EncodedSubtitleChunk, EncodedSubtitleChunkMetadata } from "./subtitles";
|
||||
import { SubtitleCue, SubtitleMetadata } from "./subtitles";
|
||||
|
||||
export abstract class Muxer {
|
||||
output: Output;
|
||||
@@ -11,9 +11,62 @@ export abstract class Muxer {
|
||||
abstract start(): void;
|
||||
abstract addEncodedVideoChunk(track: OutputVideoTrack, chunk: EncodedVideoChunk, meta?: EncodedVideoChunkMetadata): void;
|
||||
abstract addEncodedAudioChunk(track: OutputAudioTrack, chunk: EncodedAudioChunk, meta?: EncodedAudioChunkMetadata): void;
|
||||
abstract addEncodedSubtitleChunk(track: OutputSubtitleTrack, chunk: EncodedSubtitleChunk, meta?: EncodedSubtitleChunkMetadata): void;
|
||||
abstract addSubtitleCue(track: OutputSubtitleTrack, cue: SubtitleCue, meta?: SubtitleMetadata): void;
|
||||
abstract finalize(): void;
|
||||
|
||||
beforeTrackAdd(track: OutputTrack) {}
|
||||
onTrackClose(track: OutputTrack) {}
|
||||
|
||||
private trackTimestampInfo = new WeakMap<OutputTrack, {
|
||||
timestampOffset: number,
|
||||
maxTimestamp: number,
|
||||
lastKeyFrameTimestamp: number
|
||||
}>();
|
||||
|
||||
abstract timestampsMustStartAtZero: boolean;
|
||||
protected validateAndNormalizeTimestamp(track: OutputTrack, rawTimestampInUs: number, isKeyFrame: boolean) {
|
||||
let timestampInSeconds = rawTimestampInUs / 1e6;
|
||||
|
||||
let timestampInfo = this.trackTimestampInfo.get(track);
|
||||
if (!timestampInfo) {
|
||||
if (!isKeyFrame) {
|
||||
throw new Error('First frame must be a key frame.');
|
||||
}
|
||||
|
||||
if (this.timestampsMustStartAtZero && timestampInSeconds > 0) {
|
||||
throw new Error(`Timestamps must start at zero (got ${timestampInSeconds}s).`);
|
||||
}
|
||||
|
||||
timestampInfo = {
|
||||
timestampOffset: timestampInSeconds,
|
||||
maxTimestamp: track.source.offsetTimestamps ? 0 : timestampInSeconds,
|
||||
lastKeyFrameTimestamp: track.source.offsetTimestamps ? 0 : timestampInSeconds
|
||||
};
|
||||
this.trackTimestampInfo.set(track, timestampInfo);
|
||||
}
|
||||
|
||||
if (track.source.offsetTimestamps) {
|
||||
timestampInSeconds -= timestampInfo.timestampOffset;
|
||||
}
|
||||
|
||||
if (timestampInSeconds < 0) {
|
||||
throw new Error(`Timestamps must be non-negative (got ${timestampInSeconds}s).`);
|
||||
}
|
||||
|
||||
if (timestampInSeconds < timestampInfo.lastKeyFrameTimestamp) {
|
||||
throw new Error(`Timestamp cannot be smaller than last key frame's timestamp (got ${timestampInSeconds}s, last key frame at ${timestampInfo.lastKeyFrameTimestamp}s).`);
|
||||
}
|
||||
|
||||
if (isKeyFrame) {
|
||||
if (timestampInSeconds < timestampInfo.maxTimestamp) {
|
||||
throw new Error(`Key frame timestamps cannot be smaller than any timestamp that came before (got ${timestampInSeconds}s, max timestamp was ${timestampInfo.maxTimestamp}s).`);
|
||||
}
|
||||
|
||||
timestampInfo.lastKeyFrameTimestamp = timestampInSeconds;
|
||||
}
|
||||
|
||||
timestampInfo.maxTimestamp = Math.max(timestampInfo.maxTimestamp, timestampInSeconds);
|
||||
|
||||
return timestampInSeconds;
|
||||
}
|
||||
}
|
||||
@@ -9,7 +9,9 @@ export abstract class OutputFormat {
|
||||
|
||||
export class Mp4OutputFormat extends OutputFormat {
|
||||
constructor(public options: {
|
||||
// TODO: Make this optional with a smart default
|
||||
fastStart: false | 'in-memory' | 'fragmented' | {
|
||||
// TODO: This is too simple now that multiple tracks can exist.
|
||||
expectedVideoChunks?: number,
|
||||
expectedAudioChunks?: number
|
||||
},
|
||||
|
||||
+20
-12
@@ -1,15 +1,16 @@
|
||||
import { buildAudioCodecString, buildVideoCodecString } from "./codec";
|
||||
import { assert, TransformationMatrix } from "./misc";
|
||||
import { assert } from "./misc";
|
||||
import { OutputAudioTrack, OutputSubtitleTrack, OutputTrack, OutputVideoTrack } from "./output";
|
||||
import { SubtitleEncoder } from "./subtitles";
|
||||
import { SubtitleParser } from "./subtitles";
|
||||
|
||||
export type VideoCodec = 'avc' | 'hevc' | 'vp8' | 'vp9' | 'av1';
|
||||
export type AudioCodec = 'aac' | 'opus' | 'vorbis'; // TODO add the rest
|
||||
export type AudioCodec = 'aac' | 'opus'; // TODO add the rest
|
||||
export type SubtitleCodec = 'webvtt';
|
||||
|
||||
export abstract class MediaSource {
|
||||
connectedTrack: OutputTrack | null = null;
|
||||
closed = false;
|
||||
offsetTimestamps = false;
|
||||
|
||||
ensureValidDigest() {
|
||||
if (!this.connectedTrack) {
|
||||
@@ -83,8 +84,9 @@ export class EncodedVideoChunkSource extends VideoSource {
|
||||
const KEY_FRAME_INTERVAL = 5;
|
||||
|
||||
type VideoCodecConfig = {
|
||||
codec: 'avc' | 'hevc' | 'vp8' | 'vp9' | 'av1',
|
||||
bitrate: number
|
||||
codec: VideoCodec,
|
||||
bitrate: number,
|
||||
latencyMode?: VideoEncoderConfig['latencyMode']
|
||||
};
|
||||
|
||||
class VideoEncoderWrapper {
|
||||
@@ -125,6 +127,8 @@ class VideoEncoderWrapper {
|
||||
width: videoFrame.codedWidth,
|
||||
height: videoFrame.codedHeight,
|
||||
bitrate: this.codecConfig.bitrate,
|
||||
framerate: this.source.connectedTrack?.metadata.frameRate,
|
||||
latencyMode: this.codecConfig.latencyMode,
|
||||
});
|
||||
}
|
||||
|
||||
@@ -162,6 +166,7 @@ export class CanvasSource extends VideoSource {
|
||||
const frame = new VideoFrame(this.canvas, {
|
||||
timestamp: Math.round(1e6 * timestamp),
|
||||
duration: Math.round(1e6 * duration),
|
||||
alpha: 'discard',
|
||||
});
|
||||
|
||||
this.encoder.digest(frame);
|
||||
@@ -177,6 +182,8 @@ export class MediaStreamVideoTrackSource extends VideoSource {
|
||||
private encoder: VideoEncoderWrapper;
|
||||
private abortController: AbortController | null = null;
|
||||
|
||||
override offsetTimestamps = true;
|
||||
|
||||
constructor(private track: MediaStreamVideoTrack, codecConfig: VideoCodecConfig) {
|
||||
super(codecConfig.codec);
|
||||
this.encoder = new VideoEncoderWrapper(this, codecConfig);
|
||||
@@ -236,7 +243,7 @@ export class EncodedAudioChunkSource extends AudioSource {
|
||||
}
|
||||
|
||||
type AudioCodecConfig = {
|
||||
codec: 'aac' | 'opus' | 'vorbis',
|
||||
codec: AudioCodec,
|
||||
bitrate: number
|
||||
};
|
||||
|
||||
@@ -340,6 +347,8 @@ export class MediaStreamAudioTrackSource extends AudioSource {
|
||||
private encoder: AudioEncoderWrapper;
|
||||
private abortController: AbortController | null = null;
|
||||
|
||||
override offsetTimestamps = true;
|
||||
|
||||
constructor(private track: MediaStreamAudioTrack, codecConfig: AudioCodecConfig) {
|
||||
super(codecConfig.codec);
|
||||
this.encoder = new AudioEncoderWrapper(this, codecConfig);
|
||||
@@ -388,21 +397,20 @@ export abstract class SubtitleSource extends MediaSource {
|
||||
}
|
||||
|
||||
export class TextSubtitleSource extends SubtitleSource {
|
||||
private encoder: SubtitleEncoder;
|
||||
private parser: SubtitleParser;
|
||||
|
||||
constructor(codec: SubtitleCodec) {
|
||||
super(codec);
|
||||
|
||||
this.encoder = new SubtitleEncoder({
|
||||
output: (chunk, metadata) => this.connectedTrack?.output.muxer.addEncodedSubtitleChunk(this.connectedTrack, chunk, metadata),
|
||||
this.parser = new SubtitleParser({
|
||||
codec,
|
||||
output: (cue, metadata) => this.connectedTrack?.output.muxer.addSubtitleCue(this.connectedTrack, cue, metadata),
|
||||
error: (error) => console.error(error) // TODO
|
||||
});
|
||||
|
||||
this.encoder.configure({ codec });
|
||||
}
|
||||
|
||||
digest(text: string) {
|
||||
this.ensureValidDigest();
|
||||
this.encoder.encode(text);
|
||||
this.parser.parse(text);
|
||||
}
|
||||
}
|
||||
+60
-83
@@ -1,62 +1,46 @@
|
||||
export type EncodedSubtitleChunk = {
|
||||
body: Uint8Array,
|
||||
additions: Uint8Array | null,
|
||||
timestamp: number,
|
||||
duration: number
|
||||
export type SubtitleCue = {
|
||||
timestamp: number, // in seconds
|
||||
duration: number, // in seconds
|
||||
text: string,
|
||||
identifier?: string,
|
||||
settings?: string,
|
||||
notes?: string
|
||||
};
|
||||
|
||||
export type SubtitleDecoderConfig = {
|
||||
description: Uint8Array
|
||||
export type SubtitleConfig = {
|
||||
description: string
|
||||
};
|
||||
|
||||
export type EncodedSubtitleChunkMetadata = {
|
||||
decoderConfig?: SubtitleDecoderConfig
|
||||
export type SubtitleMetadata = {
|
||||
config?: SubtitleConfig
|
||||
};
|
||||
|
||||
interface SubtitleEncoderOptions {
|
||||
output: (chunk: EncodedSubtitleChunk, metadata: EncodedSubtitleChunkMetadata) => unknown,
|
||||
type SubtitleParserOptions = {
|
||||
codec: 'webvtt',
|
||||
output: (cue: SubtitleCue, metadata: SubtitleMetadata) => unknown,
|
||||
error: (error: Error) => unknown
|
||||
}
|
||||
|
||||
interface SubtitleEncoderConfig {
|
||||
codec: 'webvtt'
|
||||
}
|
||||
};
|
||||
|
||||
const cueBlockHeaderRegex = /(?:(.+?)\n)?((?:\d{2}:)?\d{2}:\d{2}.\d{3})\s+-->\s+((?:\d{2}:)?\d{2}:\d{2}.\d{3})/g;
|
||||
const preambleStartRegex = /^WEBVTT.*?\n{2}/;
|
||||
const timestampRegex = /(?:(\d{2}):)?(\d{2}):(\d{2}).(\d{3})/;
|
||||
const inlineTimestampRegex = /<(?:(\d{2}):)?(\d{2}):(\d{2}).(\d{3})>/g;
|
||||
const textEncoder = new TextEncoder();
|
||||
const preambleStartRegex = /^WEBVTT(.|\n)*?\n{2}/;
|
||||
export const inlineTimestampRegex = /<(?:(\d{2}):)?(\d{2}):(\d{2}).(\d{3})>/g;
|
||||
|
||||
export class SubtitleEncoder {
|
||||
#options: SubtitleEncoderOptions;
|
||||
#config: SubtitleEncoderConfig | null = null;
|
||||
#preambleBytes: Uint8Array | null = null;
|
||||
export class SubtitleParser {
|
||||
#options: SubtitleParserOptions;
|
||||
#preambleText: string | null = null;
|
||||
#preambleEmitted = false;
|
||||
|
||||
constructor(options: SubtitleEncoderOptions) {
|
||||
constructor(options: SubtitleParserOptions) {
|
||||
this.#options = options;
|
||||
}
|
||||
|
||||
configure(config: SubtitleEncoderConfig) {
|
||||
if (config.codec !== 'webvtt') {
|
||||
throw new Error("Codec must be 'webvtt'.");
|
||||
}
|
||||
|
||||
this.#config = config;
|
||||
}
|
||||
|
||||
encode(text: string) {
|
||||
if (!this.#config) {
|
||||
throw new Error('Encoder not configured.');
|
||||
}
|
||||
|
||||
text = text.replace('\r\n', '\n').replace('\r', '\n');
|
||||
parse(text: string) {
|
||||
text = text.replaceAll('\r\n', '\n').replaceAll('\r', '\n');
|
||||
|
||||
cueBlockHeaderRegex.lastIndex = 0;
|
||||
let match: RegExpMatchArray | null;
|
||||
|
||||
if (!this.#preambleBytes) {
|
||||
if (!this.#preambleText) {
|
||||
if (!preambleStartRegex.test(text)) {
|
||||
let error = new Error('WebVTT preamble incorrect.');
|
||||
this.#options.error(error);
|
||||
@@ -72,7 +56,7 @@ export class SubtitleEncoder {
|
||||
throw error;
|
||||
}
|
||||
|
||||
this.#preambleBytes = textEncoder.encode(preamble);
|
||||
this.#preambleText = preamble;
|
||||
|
||||
if (match) {
|
||||
text = text.slice(match.index);
|
||||
@@ -82,70 +66,63 @@ export class SubtitleEncoder {
|
||||
|
||||
while (match = cueBlockHeaderRegex.exec(text)) {
|
||||
let notes = text.slice(0, match.index);
|
||||
let cueIdentifier = match[1] || '';
|
||||
let cueIdentifier = match[1];
|
||||
let matchEnd = match.index! + match[0].length;
|
||||
let bodyStart = text.indexOf('\n', matchEnd) + 1;
|
||||
let cueSettings = text.slice(matchEnd, bodyStart).trim();
|
||||
let bodyEnd = text.indexOf('\n\n', matchEnd);
|
||||
if (bodyEnd === -1) bodyEnd = text.length;
|
||||
|
||||
let startTime = this.#parseTimestamp(match[2]!);
|
||||
let endTime = this.#parseTimestamp(match[3]!);
|
||||
let startTime = parseSubtitleTimestamp(match[2]!);
|
||||
let endTime = parseSubtitleTimestamp(match[3]!);
|
||||
let duration = endTime - startTime;
|
||||
|
||||
let body = text.slice(bodyStart, bodyEnd);
|
||||
let additions = `${cueSettings}\n${cueIdentifier}\n${notes}`;
|
||||
|
||||
// Replace in-body timestamps so that they're relative to the cue start time
|
||||
inlineTimestampRegex.lastIndex = 0;
|
||||
body = body.replace(inlineTimestampRegex, (match) => {
|
||||
let time = this.#parseTimestamp(match.slice(1, -1));
|
||||
let offsetTime = time - startTime;
|
||||
|
||||
return `<${this.#formatTimestamp(offsetTime)}>`;
|
||||
});
|
||||
let body = text.slice(bodyStart, bodyEnd).trim();
|
||||
|
||||
text = text.slice(bodyEnd).trimStart();
|
||||
cueBlockHeaderRegex.lastIndex = 0;
|
||||
|
||||
let chunk: EncodedSubtitleChunk = {
|
||||
body: textEncoder.encode(body),
|
||||
additions: additions.trim() === '' ? null : textEncoder.encode(additions),
|
||||
timestamp: startTime * 1000,
|
||||
duration: duration * 1000
|
||||
let cue: SubtitleCue = {
|
||||
timestamp: startTime / 1000,
|
||||
duration: duration / 1000,
|
||||
text: body,
|
||||
identifier: cueIdentifier,
|
||||
settings: cueSettings,
|
||||
notes
|
||||
};
|
||||
|
||||
let meta: EncodedSubtitleChunkMetadata = {};
|
||||
let meta: SubtitleMetadata = {};
|
||||
if (!this.#preambleEmitted) {
|
||||
meta.decoderConfig = {
|
||||
description: this.#preambleBytes
|
||||
meta.config = {
|
||||
description: this.#preambleText
|
||||
};
|
||||
this.#preambleEmitted = true;
|
||||
}
|
||||
|
||||
this.#options.output(chunk, meta);
|
||||
this.#options.output(cue, meta);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#parseTimestamp(string: string) {
|
||||
let match = timestampRegex.exec(string);
|
||||
if (!match) throw new Error('Expected match.');
|
||||
const timestampRegex = /(?:(\d{2}):)?(\d{2}):(\d{2}).(\d{3})/;
|
||||
export const parseSubtitleTimestamp = (string: string) => {
|
||||
let match = timestampRegex.exec(string);
|
||||
if (!match) throw new Error('Expected match.');
|
||||
|
||||
return 60 * 60 * 1000 * Number(match[1] || '0') +
|
||||
60 * 1000 * Number(match[2]) +
|
||||
1000 * Number(match[3]) +
|
||||
Number(match[4]);
|
||||
}
|
||||
return 60 * 60 * 1000 * Number(match[1] || '0') +
|
||||
60 * 1000 * Number(match[2]) +
|
||||
1000 * Number(match[3]) +
|
||||
Number(match[4]);
|
||||
};
|
||||
|
||||
#formatTimestamp(timestamp: number) {
|
||||
let hours = Math.floor(timestamp / (60 * 60 * 1000));
|
||||
let minutes = Math.floor((timestamp % (60 * 60 * 1000)) / (60 * 1000));
|
||||
let seconds = Math.floor((timestamp % (60 * 1000)) / 1000);
|
||||
let milliseconds = timestamp % 1000;
|
||||
export const formatSubtitleTimestamp = (timestamp: number) => {
|
||||
let hours = Math.floor(timestamp / (60 * 60 * 1000));
|
||||
let minutes = Math.floor((timestamp % (60 * 60 * 1000)) / (60 * 1000));
|
||||
let seconds = Math.floor((timestamp % (60 * 1000)) / 1000);
|
||||
let milliseconds = timestamp % 1000;
|
||||
|
||||
return hours.toString().padStart(2, '0') + ':' +
|
||||
minutes.toString().padStart(2, '0') + ':' +
|
||||
seconds.toString().padStart(2, '0') + '.' +
|
||||
milliseconds.toString().padStart(3, '0');
|
||||
}
|
||||
}
|
||||
return hours.toString().padStart(2, '0') + ':' +
|
||||
minutes.toString().padStart(2, '0') + ':' +
|
||||
seconds.toString().padStart(2, '0') + '.' +
|
||||
milliseconds.toString().padStart(3, '0');
|
||||
};
|
||||
@@ -67,6 +67,10 @@ export class ArrayBufferTargetWriter extends Writer {
|
||||
this.#ensureSize(this.#pos);
|
||||
this.#target.buffer = this.#buffer.slice(0, Math.max(this.#maxPos, this.#pos));
|
||||
}
|
||||
|
||||
getSlice(start: number, end: number) {
|
||||
return this.#bytes.slice(start, end);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
|
||||
Reference in New Issue
Block a user