Compare commits

...
64 Commits
Author SHA1 Message Date
Vanilagy 7a2a23bcb0 And another one 2025-08-08 18:57:10 +02:00
Vanilagy b63d86226f Fix release workflow? 2025-08-08 18:53:06 +02:00
Vanilagy 3371813c33 Bump version to 1.5.0, update release workflow 2025-08-08 18:46:12 +02:00
Vanilagy 1fe115ac83 Implement frame rate adjustment logic for Conversion API, fix source not being closed 2025-08-08 18:39:08 +02:00
David P.andGitHub 7e42e298c5 Merge pull request #45 from yonathan06/main
Custom fps convertion WIP
2025-08-08 17:20:04 +02:00
Vanilagy 597b68b299 Add support for Matroska lacing, fix packet lookup logic 2025-08-08 17:10:03 +02:00
David P.andGitHub f5803598e8 Merge pull request #44 from frw/patch-1
fix: check string for null terminators
2025-08-08 12:16:12 +02:00
Yonatan Bendahan b98d8e58ca remove console.log 2025-08-07 12:00:29 +03:00
Yonatan Bendahan dd60ca71b3 fix lint error 2025-08-07 11:56:44 +03:00
Yonatan Bendahan 4b99fd08ca with custom fps convertion 2025-08-07 11:47:50 +03:00
Frederick Widjaja 5b238c6d0f fix: check string for null terminators 2025-08-07 11:09:44 +07:00
David P.andGitHub 2f83e07a55 Merge pull request #42 from yonathan06/main
Add "duplex: 'half'" to fetch stream example
2025-08-05 14:45:31 +02:00
Yonatan Bendahan 1aaded9241 add "duplex: 'half'" to fetch stream example 2025-08-05 15:20:12 +03:00
Vanilagy 5a6b849ff8 Fix fragmented MP4 files with a non-empty sample table 2025-08-03 22:34:38 +02:00
Vanilagy b60d4cafed Fix example blob MIME types 2025-08-01 11:11:37 +02:00
Vanilagy 95091cce9c Add special top-level Cluster logic for Matroska demuxer 2025-07-28 12:03:19 +02:00
Vanilagy 72ed933bf4 Fix discarded rotation when doing VideoSample.clone() 2025-07-28 01:50:39 +02:00
Vanilagy ef3b53ff8a Add Geef <3 2025-07-27 21:27:16 +02:00
Vanilagy ac8baa4873 Fall back to default timescale when not provided by the Matroska file 2025-07-27 13:03:43 +02:00
Vanilagy 232d1a6cd7 Bump version to 1.4.0 2025-07-27 12:20:11 +02:00
Vanilagy 50fe065852 Document API changes 2025-07-27 12:19:44 +02:00
Vanilagy d38ad22559 Big realtime playback refactor
- Added Worker- or AudioContext-based fallbacks in case MediaStreamTrackProcessor is not available
- Fixed fMP4-streamed files not playing in Safari
- Added better error handling for MediaStream sources
- Improved the live recording example to be more robust
2025-07-27 12:13:03 +02:00
Vanilagy 9cc38329f2 Fix link 2025-07-27 11:46:36 +02:00
Vanilagy 09ed583b78 Smol change 2025-07-27 00:57:57 +02:00
Vanilagy 9de5b24ec5 Add new sponsors 2025-07-27 00:56:22 +02:00
Vanilagy 03b4843c1d Fix incorrectly fixed .js imports 2025-07-26 00:50:47 +02:00
Vanilagy 9bfa855f25 Downgrade @types/web-codecs for better type support, specify requirements in README and docs 2025-07-26 00:34:24 +02:00
Vanilagy d768f3c63b Add section in docs for Jonny 2025-07-25 13:22:18 +02:00
Vanilagy c4d13ed698 Improve UrlSource fetching logic 2025-07-25 12:07:54 +02:00
Vanilagy 5e933d9362 Add packet type derivation logic, new verifyKeyPackets options, new determinePacketType method 2025-07-24 16:09:01 +02:00
Vanilagy 904ac2e366 Improve error messages in examples 2025-07-19 17:28:39 +02:00
Vanilagy 4b12468008 Add support for reading and writing RF64 WAVE files 2025-07-18 16:48:23 +02:00
Vanilagy 2277810853 Add sponsor 2025-07-18 14:24:21 +02:00
Vanilagy 19b1c015b6 Add full URLs for OpenGraph image, add more graph tags 2025-07-18 09:25:24 +02:00
Vanilagy b1b493ab10 Fix out of bounds reads for Matroska demuxer 2025-07-15 16:03:00 +02:00
Vanilagy cf0bbe896a Add longer variant to VideoSample.draw 2025-07-15 11:49:33 +02:00
Vanilagy c5d299c4ec QuickTime 😩 2025-07-14 09:53:26 +02:00
Vanilagy b06f510382 Bump patch version 2025-07-11 10:58:58 +02:00
Vanilagy 14ad80df5d Improve ISOBMFF demuxer to support fragments without a key frame 2025-07-11 10:57:24 +02:00
Vanilagy 514a533445 Reduce MP3 detection false positives 2025-07-11 10:07:45 +02:00
Vanilagy 9072a7a9f7 Be even more graceful with invalid ISOBMFF files 2025-07-10 17:47:56 +02:00
Vanilagy fba463f46b Be more graceful with unsupported edit lists 2025-07-10 17:33:10 +02:00
Vanilagy 3621c59034 Center bunny, add bisky 2025-07-10 11:39:02 +02:00
Vanilagy a16911874c Clarify built distribution files in docs 2025-07-09 12:14:29 +02:00
Vanilagy f59a79dcf7 Bump permissions for Release workflow 2025-07-09 12:05:33 +02:00
Vanilagy 19b7036d82 Bump version to 1.0.4 2025-07-09 11:59:32 +02:00
Vanilagy 47bae059db Add build artifact upload step to Release workflow 2025-07-09 11:58:15 +02:00
Vanilagy 84c9d37cb7 Revise build step and dist structure for better Webpack, CJS and pnpm support 2025-07-09 11:53:19 +02:00
David P.andGitHub 3e96184a6f Merge pull request #11 from studnitz/patch-1
Fix typo
2025-07-09 10:29:23 +02:00
Alexander von StudnitzandGitHub ee59837857 Fix typo 2025-07-09 09:47:26 +02:00
Vanilagy 08cf34ee0d Adjust README image sizes 2025-07-08 15:42:32 +02:00
Vanilagy 1b382d3e57 Improve unsupported encoder configuration error messages 2025-07-07 11:21:42 +02:00
Vanilagy f78fc697cd Allow DataView as AllowSharedBufferSource 2025-07-07 10:40:28 +02:00
Vanilagy 39df195ada Adjust some meta tags 2025-07-03 20:54:42 +02:00
Vanilagy 326e03cfc9 Link migration guides in docs 2025-07-03 12:42:27 +02:00
Vanilagy 23aef27154 Oops 2025-07-03 12:22:27 +02:00
Vanilagy 9daab77bb5 Clarify live streaming 2025-07-03 12:19:08 +02:00
Vanilagy 5a6e3eebda Adjust README 2025-07-03 12:16:42 +02:00
Vanilagy c4fb325329 Clarify "media" = video/audio 2025-07-03 11:39:46 +02:00
Vanilagy 1071bf9d49 Merge branch 'main' of https://github.com/Vanilagy/metamuxer 2025-07-03 11:36:07 +02:00
Vanilagy a8fa2e4cec Modify examples to only accept video and audio files 2025-07-03 11:36:04 +02:00
Vanilagy 12f14b1f66 Remove .gitattributes 2025-07-03 00:35:24 +02:00
Vanilagy d4d93a7007 PINK PROGRESS BAR 2025-07-02 23:07:47 +02:00
Vanilagy f44cb26d76 Fix styling, add new sponsor 2025-07-02 23:01:21 +02:00
65 changed files with 2543 additions and 884 deletions
+1 -1
View File
@@ -10,6 +10,6 @@ indent_size = 2
indent_style = space
indent_size = 4
[*.yaml]
[*.yml]
indent_style = space
indent_size = 2
-1
View File
@@ -1 +0,0 @@
build/* linguist-generated
@@ -2,6 +2,7 @@ name: Lint
on:
push:
pull_request:
jobs:
lint:
+16 -3
View File
@@ -11,12 +11,20 @@ jobs:
release:
name: Release
runs-on: ubuntu-latest
permissions:
contents: read
id-token: write
permissions: write-all
steps:
- name: Checkout repository
uses: actions/checkout@v4
with:
fetch-depth: 0
- name: Merge main into release branch
run: |
git config user.name "github-actions[bot]"
git config user.email "github-actions[bot]@users.noreply.github.com"
git checkout release
git merge origin/main --no-ff -m "Merge main into release for tag ${{ github.event.release.tag_name }}"
git push origin release
- name: Set up Node.js
uses: actions/setup-node@v4
@@ -41,6 +49,11 @@ jobs:
- name: Run build
run: npm run build
- name: Upload build artifacts
env:
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
run: gh release upload ${{ github.event.release.tag_name }} dist/bundles/mediabunny.cjs dist/bundles/mediabunny.min.cjs dist/bundles/mediabunny.mjs dist/bundles/mediabunny.min.mjs dist/mediabunny.d.ts
- name: Create Publish to npm
run: npm publish --provenance
env:
-1
View File
@@ -1,5 +1,4 @@
/node_modules
/build
/dist
/dist-docs
.DS_Store
+12 -6
View File
@@ -5,10 +5,10 @@
[![](https://img.shields.io/npm/dm/mediabunny)](https://www.npmjs.com/package/mediabunny)
<div align="center">
<img src="./docs/public/mediabunny-logo.svg" height="180">
<img src="./docs/public/mediabunny-logo.svg" width="180" height="180">
</div>
Mediabunny is a JavaScript library for reading, writing, and converting media files (like MP4 or WebM), directly in the browser. It aims to be a complete toolkit for high-performance media operations on the web. It's written from scratch in pure TypeScript, has zero dependencies, is very performant, and is extremely tree-shakable, meaning you only include what you use. You can think of it a bit like [FFmpeg](https://ffmpeg.org/), but built from the ground up for the web.
Mediabunny is a JavaScript library for reading, writing, and converting media files (like MP4, WebM, MP3), directly in the browser. It aims to be a complete toolkit for high-performance media operations on the web. It's written from scratch in pure TypeScript, has zero dependencies, is very performant, and is extremely tree-shakable, meaning you only include what you use. You can think of it a bit like [FFmpeg](https://ffmpeg.org/), but built from the ground up for the web.
[Documentation](https://mediabunny.dev) | [Examples](https://mediabunny.dev/examples) | [Sponsoring](#sponsoring) | [License](#license) | [Discord](https://discord.gg/hmpkyYuS4U)
@@ -16,15 +16,19 @@ Mediabunny is a JavaScript library for reading, writing, and converting media fi
<div align="center">
<a href="https://www.gling.ai/" target="_blank">
<img src="./docs/public/sponsors/gling.svg" height="60" alt="Gling AI">
<img src="./docs/public/sponsors/gling.svg" width="60" height="60" alt="Gling AI">
</a>
&nbsp;&nbsp;&nbsp;&nbsp;
<a href="https://diffusion.studio/" target="_blank">
<img src="./docs/public/sponsors/diffusionstudio.png" height="60" alt="Diffusion Studio">
<img src="./docs/public/sponsors/diffusionstudio.png" width="60" height="60" alt="Diffusion Studio">
</a>
&nbsp;&nbsp;&nbsp;&nbsp;
<a href="https://kino.ai/" target="_blank">
<img src="./docs/public/sponsors/kino.jpg" width="60" height="60" alt="Kino">
</a>
</div>
[Get featured](https://github.com/sponsors/Vanilagy)
[Sponsor Mediabunny's development](https://github.com/sponsors/Vanilagy)
## Features
@@ -49,6 +53,8 @@ Core features include:
npm install mediabunny
```
Requires any JavaScript environment that can run ECMAScript 2021 or later. Mediabunny is expected to be run in modern browsers. For types, TypeScript 5.7 or later is required.
### Read file metadata
```js
@@ -152,7 +158,7 @@ For development, clone this repository and install it using a modern version of
```bash
npm install # Install dependencies
npm run watch # Development build with watch mode
npm run watch # Build bundles on watch mode
npm run build # Production build with type definitions
+1 -1
View File
@@ -1,6 +1,6 @@
{
"$schema": "https://developer.microsoft.com/json-schemas/api-extractor/v7/api-extractor.schema.json",
"mainEntryPointFilePath": "build/index.d.ts",
"mainEntryPointFilePath": "dist/modules/index.d.ts",
"bundledPackages": [],
"compiler": {},
"apiReport": {
+4 -4
View File
@@ -37,20 +37,20 @@ const esmConfig = {
const ctxUmd = await esbuild.context({
...umdConfig,
outfile: 'dist/mediabunny.js',
outfile: 'dist/bundles/mediabunny.cjs',
});
const ctxEsm = await esbuild.context({
...esmConfig,
outfile: 'dist/mediabunny.mjs',
outfile: 'dist/bundles/mediabunny.mjs',
});
const ctxUmdMinified = await esbuild.context({
...umdConfig,
outfile: 'dist/mediabunny.min.js',
outfile: 'dist/bundles/mediabunny.min.cjs',
minify: true,
});
const ctxEsmMinified = await esbuild.context({
...esmConfig,
outfile: 'dist/mediabunny.min.mjs',
outfile: 'dist/bundles/mediabunny.min.mjs',
minify: true,
});
+9 -4
View File
@@ -1,6 +1,6 @@
<!DOCTYPE html>
<script src="../dist/mediabunny.js"></script>
<script src="../dist/bundles/mediabunny.cjs"></script>
<script type="module">
const fileInput = document.createElement('input');
@@ -21,7 +21,7 @@
chunked: true,
chunkSize: 2**20
});
const outputFormat = new Mediabunny.MovOutputFormat();
const outputFormat = new Mediabunny.Mp4OutputFormat({});
const button = document.createElement('button');
button.textContent = 'Cancel';
@@ -38,7 +38,8 @@
target
}),
audio: {
codec: 'pcm-f64'
//codec: 'opus',
//bitrate: 128000,
//numberOfChannels: 1,
//sampleRate: 4000
//discard: true
@@ -69,6 +70,10 @@
},
*/
video: {
frameRate: 27.123,
//width: 320,
//forceTranscode: true,
//codec: 'av1',
//discard: true,
//width: 1280,
//discard: true,
@@ -86,7 +91,7 @@
},
trim: {
start: 0,
end: 10
end: 20
},
});
console.log(conversion);
+93 -6
View File
@@ -1,6 +1,6 @@
<!DOCTYPE html>
<script src="../dist/mediabunny.js"></script>
<script src="../dist/bundles/mediabunny.cjs"></script>
<script type="module">
const fileInput = document.createElement('input');
@@ -15,14 +15,101 @@
formats: Mediabunny.ALL_FORMATS,
source
});
const audioTrack = await input.getPrimaryAudioTrack();
const sink = new Mediabunny.EncodedPacketSink(audioTrack);
let thing = await sink.getFirstPacket();
while (thing) {
console.log(thing);
if (thing.timestamp >= 2.4) break;
thing = await sink.getNextPacket(thing);
}
console.log("done")
/*
for await (const packet of sink.packets()) {
console.log(packet);
if (packet.timestamp >= 2.4) break;
}
*/
/*
const videoTrack = await input.getPrimaryVideoTrack();
const packetSink = new Mediabunny.EncodedPacketSink(videoTrack);
const sampleSink = new Mediabunny.VideoSampleSink(videoTrack);
for await (const packet of packetSink.packets(undefined, undefined, {verifyType: true})) {
const guess = packet.type;
const real = await videoTrack.determinePacketType(packet);
if (guess !== real) {
console.log(guess, real, packet);
}
}
console.log("don")
*/
/*
console.time()
for await (const packet of packetSink.packets(undefined, undefined, { verifyType: true })) {
//console.log(packet)
}
console.timeEnd()
*/
//console.log(await packetSink.getPacket(6.666666666666667, { verifyType: true }))
/*
for await (const packet of packetSink.packets()) {
const guess = packet.type;
const real = await videoTrack.determinePacketType(packet);
if (guess !== real) {
console.log(guess, real, packet);
}
}
console.log("done")
*/
/*
const timestamp = 6.666666666666667;
const thePacket = await packetSink.getPacket(timestamp);
console.log(videoTrack.codec, thePacket, await videoTrack.determinePacketType(thePacket));
sampleSink.getSample(thePacket.timestamp);
*/
/*
let packet = await packetSink.getFirstPacket();
while (packet) {
console.log(packet)
const sample = await sampleSink.getSample(packet.timestamp);
packet = await packetSink.getNextKeyPacket(packet);
}
*/
/*
const canvas = document.createElement('canvas');
canvas.width = 1920;
canvas.height = 1080;
canvas.style.background = 'ghostwhite';
const ctx = canvas.getContext('2d');
document.body.append(canvas);
const videoTrack = await input.getPrimaryVideoTrack();
const sink = new Mediabunny.VideoSampleSink(videoTrack);
const sink = new Mediabunny.EncodedPacketSink(videoTrack);
for await (const packet of sink.packets()) {
if (packet.timestamp >= 2) break;
console.log(packet);
}
const sample = await sink.getSample(3);
console.log(sample);
sample.draw(ctx, 1500, 500, 50, 50, 0, 0);
*/
/*
let timestamps = [];
+14 -6
View File
@@ -1,6 +1,6 @@
<button>Go</button>
<script src="../dist/mediabunny.js"></script>
<script src="../dist/bundles/mediabunny.cjs"></script>
<script type="module">
function download(blob, filename) {
@@ -23,21 +23,29 @@
format: new Mediabunny.Mp4OutputFormat(),
});
if (videoTrack) {
output.addVideoTrack(new Mediabunny.MediaStreamVideoTrackSource(videoTrack, {
const source = new Mediabunny.MediaStreamVideoTrackSource(videoTrack, {
codec: 'avc',
bitrate: Mediabunny.QUALITY_MEDIUM
}));
});
source.errorPromise.catch((d) => console.log("Hello?????", d));
output.addVideoTrack(source);
}
if (audioTrack) {
output.addAudioTrack(new Mediabunny.MediaStreamAudioTrackSource(audioTrack, {
const source = new Mediabunny.MediaStreamAudioTrackSource(audioTrack, {
codec: 'aac',
bitrate: Mediabunny.QUALITY_MEDIUM
}));
});
source.errorPromise.catch((d) => console.log("Hello!!???", d));
output.addAudioTrack(source);
}
await output.start();
await new Promise(resolve => setTimeout(resolve, 3000));
await new Promise(resolve => setTimeout(resolve, 5000));
await output.finalize();
+1 -1
View File
@@ -1,6 +1,6 @@
<!DOCTYPE html>
<script src="../dist/mediabunny.js"></script>
<script src="../dist/bundles/mediabunny.cjs"></script>
<script type="module">
function download(blob, filename) {
+1 -1
View File
@@ -15,7 +15,7 @@
</div>
</div>
<script src="../dist/mediabunny.js"></script>
<script src="../dist/bundles/mediabunny.cjs"></script>
<script type="module">
const fileInput = document.querySelector('input[type="file"]');
+15 -4
View File
@@ -4,18 +4,27 @@ import tailwindcss from '@tailwindcss/vite';
import llmstxt from 'vitepress-plugin-llms';
import { HeadConfig } from 'vitepress';
const DESCRIPTION = 'A JavaScript library for reading, writing, and converting media files. Directly in the browser,'
+ ' and faster than anybunny else.';
// https://vitepress.dev/reference/site-config
export default withMermaid({
title: 'Mediabunny',
description: 'A JavaScript library for reading, writing, and converting media files. Directly in the browser, and'
+ ' faster than anybunny else.',
description: DESCRIPTION,
cleanUrls: true,
head: [
['link', { rel: 'icon', type: 'image/svg+xml', href: '/mediabunny-logo.svg' }],
['link', { rel: 'icon', type: 'image/png', href: '/mediabunny-logo.png' }],
['link', { rel: 'icon', type: 'image/svg+xml', href: '/mediabunny-logo.svg' }],
['meta', { property: 'og:type', content: 'website' }],
['meta', { property: 'og:site_name', content: 'Mediabunny' }],
['meta', { property: 'og:image', content: '/mediabunny-og-image.png' }],
['meta', { property: 'og:url', content: 'https://mediabunny.dev/' }],
['meta', { property: 'og:image', content: 'https://mediabunny.dev/mediabunny-og-image.png' }],
['meta', { property: 'og:locale', content: 'en-US' }],
['meta', { property: 'og:description', content: DESCRIPTION }],
['meta', { name: 'twitter:image', content: 'https://mediabunny.dev/mediabunny-og-image.png' }],
['meta', { name: 'twitter:card', content: 'summary_large_image' }],
['meta', { name: 'twitter:site', content: '@vanilagy' }],
['meta', { name: 'twitter:description', content: DESCRIPTION }],
],
themeConfig: {
logo: '/mediabunny-logo.svg',
@@ -72,6 +81,7 @@ export default withMermaid({
{ icon: 'github', link: 'https://github.com/Vanilagy/mediabunny' },
{ icon: 'discord', link: 'https://discord.gg/hmpkyYuS4U' },
{ icon: 'x', link: 'https://x.com/vanilagy' },
{ icon: 'bluesky', link: 'https://bsky.app/profile/vanilagy.bsky.social' },
],
search: {
@@ -110,6 +120,7 @@ export default withMermaid({
((pageData.frontmatter['head'] ??= []) as HeadConfig[]).push(
['meta', { property: 'og:title', content: title }],
['meta', { property: 'twitter:title', content: title }],
);
},
});
+1 -1
View File
@@ -37,7 +37,7 @@ features:
target: _self
icon:
src: /mingcute--magic-3-line.svg
- title: Live recording
- title: Live recording & streaming
details: Record a video from live sources and stream it to a video element.
link: /examples/live-recording
target: _self
+6
View File
@@ -11,6 +11,7 @@ It has the following features:
- Trimming
- Video resizing & fitting
- Video rotation
- Video frame rate adjustment
- Audio resampling
- Audio up/downmixing
@@ -101,6 +102,7 @@ type ConversionOptions = {
height?: number;
fit?: 'fill' | 'contain' | 'cover';
rotate?: 0 | 90 | 180 | 270;
frameRate?: number;
codec?: VideoCodec;
bitrate?: number | Quality;
forceTranscode?: boolean;
@@ -141,6 +143,10 @@ The `width`, `height` and `fit` properties control how the video is resized. If
If `width` or `height` is used in conjunction with `rotation`, they control the post-rotation dimensions.
### Adjusting frame rate
The `frameRate` property can be used to set the frame rate of the output video in Hz. If not specified, the original input frame rate will be used (which may be variable).
### Transcoding video
Use the `codec` property to control the codec of the output track. This should be set to a [codec](./supported-formats-and-codecs#video-codecs) supported by the output file, or else the track will be [discarded](#discarded-tracks).
+8 -2
View File
@@ -17,6 +17,10 @@ bun add mediabunny
```
:::
::: info
Requires any JavaScript environment that can run ECMAScript 2021 or later. Mediabunny is expected to be run in modern browsers. For types, TypeScript 5.7 or later is required.
:::
Then, simply import it like this:
```ts
import { ... } from 'mediabunny'; // ESM
@@ -27,7 +31,9 @@ ESM is preferred because it gives you tree shaking.
You can also just include the library using a script tag in your HTML:
```html
<script src="path/to/mediabunny.js"></script>
<script src="mediabunny.cjs"></script>
```
You can download the built distribution file from the [releases page](https://github.com/Vanilagy/mediabunny/releases).
This will add a `Mediabunny` object to the global scope.
You can download a built distribution file from the [releases page](https://github.com/Vanilagy/mediabunny/releases). Use the `*.cjs` builds for normal script tag inclusion, or the `*.mjs` builds for script tags with `type="module"` or direct imports via ESM. Including the `mediabunny.d.ts` declaration file in your TypeScript project will declare a global `Mediabunny` namespace.
+7
View File
@@ -66,6 +66,13 @@ This library is the result of unifying these libraries into one, solving all the
Due to tree shaking, if you only need an MP4 or WebM muxer, this library's bundle size will still be very small.
### Migration
If you're coming from mp4-muxer or webm-muxer, you should migrate to Mediabunny. For that, refer to these guides:
- [Guide: Migrating from mp4-muxer to Mediabunny](https://github.com/Vanilagy/mp4-muxer/blob/main/MIGRATION-GUIDE.md)
- [Guide: Migrating from webm-muxer to Mediabunny](https://github.com/Vanilagy/webm-muxer/blob/main/MIGRATION-GUIDE.md)
## Technical overview
At its core, Mediabunny is a collection of multiplexers and demultiplexers, one of each for every container format. Demultiplexers stream data from *sources*, while multiplexers stream data to *targets*. Every demultiplexer is capable of extracting file metadata as well as compressed media data, while multiplexers write metadata and encoded media data into a new file.
+26
View File
@@ -144,6 +144,32 @@ for await (const packet of sink.packets(start, end)) {
The `packets` method is more performant than manual iteration as it will intelligently preload future packets before they are needed.
#### Verifying key packets
By default, packet types are determined using the metadata provided by the containing file. Some files can erroneously label some delta packets as key packets, leading to potential decoder errors. To be guaranteed that a key packet is actually a key packet, you can enable the `verifyKeyPackets` option:
```ts
// If the packet returned by this method has type: 'key', it's guaranteed
// to be a key packet.
await sink.getPacket(5, { verifyKeyPackets: true });
// Returned packets are guaranteed to be key packets
await sink.getKeyPacket(10, { verifyKeyPackets: true });
await sink.getNextKeyPacket(packet, { verifyKeyPackets: true });
// Also works for the iterator:
for await (const packet of sink.packets(
undefined,
undefined,
{ verifyKeyPackets: true },
)) {
// ...
}
```
::: info
`verifyKeyPackets` only works when `metadataOnly` is not also enabled.
:::
#### Metadata-only packet retrieval
Sometimes, you're only interested in a packet's metadata (timestamp, duration, type, ...) and not in its encoded media data. All methods on `EncodedPacketSink` accept a final `options` parameter which you can use to retrieve [metadata-only packets](./packets-and-samples#metadata-only-packets):
+14
View File
@@ -163,6 +163,9 @@ const videoTrackSource = new MediaStreamVideoTrackSource(videoTrack, {
codec: 'vp9',
bitrate: 1e7,
});
// Make sure to allow any internal errors to properly bubble up
videoTrackSource.errorPromise.catch((error) => ...);
```
This source requires no additional method calls; data will automatically be captured and piped to the output file as soon as `start()` is called on the `Output`. Make sure to `stop()` on `videoTrack` after finalizing the `Output` if you don't need the user's media anymore.
@@ -171,6 +174,10 @@ This source requires no additional method calls; data will automatically be capt
If this source is the only MediaStreamTrack source in the `Output`, then the first video sample added by it starts at timestamp 0. If there are multiple, then the earliest media sample across all tracks starts at timestamp 0, and all tracks will be perfectly synchronized with each other.
:::
::: warning
`MediaStreamVideoTrackSource`'s internals are detached from the typical code flow but can still throw, so make sure to utilize `errorPromise` to deal with any errors and to stop the `Output`.
:::
### `EncodedVideoPacketSource`
The most barebones of all video sources, this source can be used to directly pipe [encoded packets](./packets-and-samples#encodedpacket) of video data to the output. This source requires that you take care of the encoding process yourself, which enables you to use the WebCodecs API manually or to plug in your own encoding stack. Alternatively, you may retrieve the encoded packets directly by reading them from another media file, allowing you to skip decoding and reencoding video data.
@@ -312,6 +319,9 @@ const audioTrackSource = new MediaStreamAudioTrackSource(audioTrack, {
codec: 'opus',
bitrate: 128e3,
});
// Make sure to allow any internal errors to properly bubble up
audioTrackSource.errorPromise.catch((error) => ...);
```
This source requires no additional method calls; data will automatically be captured and piped to the output file as soon as `start()` is called on the `Output`. Make sure to `stop()` on `audioTrack` after finalizing the `Output` if you don't need the user's media anymore.
@@ -320,6 +330,10 @@ This source requires no additional method calls; data will automatically be capt
If this source is the only MediaStreamTrack source in the `Output`, then the first audio sample added by it starts at timestamp 0. If there are multiple, then the earliest media sample across all tracks starts at timestamp 0, and all tracks will be perfectly synchronized with each other.
:::
::: warning
`MediaStreamAudioTrackSource`'s internals are detached from the typical code flow but can still throw, so make sure to utilize `errorPromise` to deal with any errors and to stop the `Output`.
:::
### `EncodedAudioPacketSource`
The most barebones of all audio sources, this source can be used to directly pipe [encoded packets](./packets-and-samples#encodedpacket) of audio data to the output. This source requires that you take care of the encoding process yourself, which enables you to use the WebCodecs API manually or to plug in your own encoding stack. Alternatively, you may retrieve the encoded packets directly by reading them from another media file, allowing you to skip decoding and reencoding audio data.
+3
View File
@@ -227,8 +227,11 @@ const output = new Output({
The following options are available:
```ts
type WavOutputFormatOptions = {
large?: boolean;
onHeader?: (data: Uint8Array, position: number) => unknown;
};
```
- `large`\
When enabled, an RF64 file be written, allowing for file sizes to exceed 4 GiB, which is otherwise not possible for regular WAVE files.
- `onHeader`\
Will be called once the file header is written. The header consists of the RIFF header, the format chunk, and the start of the data chunk (with a placeholder size of 0).
+32 -5
View File
@@ -133,6 +133,16 @@ encodedPacket.type; // => PacketType ('key' | 'delta')
For example, in a video track, it is common to have a key frame about every few seconds. When seeking, if the user seeks to a position shortly after a key frame, the decoded data can be shown quickly; if they seek far away from a key frame, the decoder must first crunch through many delta frames before it can show anything.
#### Determining a packet's actual type
The `type` field is derived from metadata in the containing file, which can sometimes (in rare cases) be incorrect. To determine a packet's actual type with certainty, you can do this:
```ts
// `packet` must come from the InputTrack `track`
const type = await track.determinePacketType(packet); // => PacketType | null
```
This determines the packet's type by looking into its bitstream. `null` is returned when the type couldn't be determined.
---
You can query the packet's timing information:
@@ -309,17 +319,29 @@ The `VideoFrame` returned by this method **must** be closed separately from the
---
It's also common to draw video samples to a `<canvas>` element or an `OffscreenCanvas`. For this, you can use the following method:
It's also common to draw video samples to a `<canvas>` element or an `OffscreenCanvas`. For this, you can use the following methods:
```ts
draw(
context: CanvasRenderingContext2D | OffscreenCanvasRenderingContext2D,
dx: number,
dy: number,
dWidth?: number,
dHeight?: number,
dWidth?: number, // defaults to displayWidth
dHeight?: number, // defaults to displayHeight
): void;
draw(
context: CanvasRenderingContext2D | OffscreenCanvasRenderingContext2D,
sx: number,
sy: number,
sWidth: number,
sHeight: number,
dx: number,
dy: number,
dWidth?: number, // defaults to sWidth
dHeight?: number, // defaults to sHeight
): void;
```
This method is similar to [drawImage](https://developer.mozilla.org/en-US/docs/Web/API/CanvasRenderingContext2D/drawImage) and paints the video frame at the given position with the given dimensions. If the dimensions aren't specified, they will default to the display dimensions. This method will automatically draw the frame with the correct rotation based on its `rotation` property.
These methods behave like [drawImage](https://developer.mozilla.org/en-US/docs/Web/API/CanvasRenderingContext2D/drawImage) and paint the video frame at the given position with the given dimensions. This method will automatically draw the frame with the correct rotation based on its `rotation` property.
If you want to draw the raw underlying image to a canvas directly (without respecting the rotation metadata), then you can use the following method:
```ts
@@ -379,7 +401,7 @@ An audio sample represents a section of audio data. It can be created directly f
### Creating audio samples
Audio samples can be constructed either from an `AudioData` instance or an initialization object:
Audio samples can be constructed either from an `AudioData` instance, an initialization object, or an `AudioBuffer`:
```ts
import { AudioSample } from 'mediabunny';
@@ -395,6 +417,11 @@ const sample = new AudioSample({
sampleRate: 44100, // in Hz
timestamp: 0, // in seconds
});
// From AudioBuffer:
const timestamp = 0; // in seconds
const samples = AudioSample.fromAudioBuffer(audioBuffer, timestamp);
// => Returns multiple AudioSamples if the AudioBuffer is very long
```
The following audio sample formats are supported:
+1
View File
@@ -312,6 +312,7 @@ const output = new Output({
const uploadComplete = fetch('https://example.com/upload', {
method: 'POST',
body: readable,
duplex: 'half',
headers: {
'Content-Type': output.format.mimeType,
},
+12
View File
@@ -286,6 +286,18 @@ See [Media sinks](./media-sinks) for a full list of sinks.
### Examples
Loop over all raw encoded packets of a track:
```ts
import { EncodedPacketSink } from 'mediabunny';
const videoTrack = await input.getPrimaryVideoTrack();
const sink = new EncodedPacketSink(videoTrack);
for await (const packet of sink.packets()) {
console.log(packet.timestamp);
}
```
Here we iterate over all samples (frames) of a video track:
```ts
import { VideoSampleSink } from 'mediabunny';
+10 -5
View File
@@ -6,7 +6,7 @@ title: Mediabunny
hero:
name: Mediabunny
text: Complete media toolkit
tagline: A JavaScript library for reading, writing, and converting media files. Directly in the browser, and faster than anybunny else.
tagline: A JavaScript library for reading, writing, and converting video and audio files. Directly in the browser, and faster than anybunny else.
image:
src: /mediabunny-logo.svg
alt: Mediabunny logo
@@ -90,10 +90,15 @@ const sponsors = {
gold: [
{ image: '/sponsors/gling.svg', name: 'Gling AI', url: 'https://www.gling.ai/' },
{ image: '/sponsors/diffusionstudio.png', name: 'Diffusion Studio', url: 'https://diffusion.studio/' },
{ image: '/sponsors/kino.jpg', name: 'Kino', url: 'https://kino.ai/' },
],
individual: [
{ image: 'https://avatars.githubusercontent.com/u/84167135', name: 'Memenome', url: 'https://github.com/memenome' },
{ image: 'https://avatars.githubusercontent.com/u/9549394', name: 'studnitz', url: 'https://github.com/studnitz' },
{ image: 'https://avatars.githubusercontent.com/u/30229596', name: 'Pablo Bonilla', url: 'https://github.com/devPablo' },
{ image: 'https://avatars.githubusercontent.com/u/58149663', name: 'H7GhosT', url: 'https://github.com/H7GhosT' },
{ image: 'https://avatars.githubusercontent.com/u/91711202', name: 'ihasq', url: 'https://github.com/ihasq' },
{ image: 'https://avatars.githubusercontent.com/u/61233224', name: 'Allwhy', url: 'https://github.com/Allwhy' },
],
};
</script>
@@ -123,7 +128,7 @@ npm install mediabunny
<div class="flex flex-col lg:flex-row lg:gap-20 lg:items-center">
<div class="flex-1 min-w-0">
<h1 class="inline-block" style="background: -webkit-linear-gradient(-30deg, #ff45ac, #ff78c2); -webkit-background-clip: text; color: transparent;">Read any media file, efficiently</h1>
<p class="text-lg">Mediabunny allows you efficiently read data from any media file, no matter the size: duration, resolution, rotation, tracks, codecs and other metadata, as well as raw or decoded media data from anywhere in the file. Load only what you need.</p>
<p class="text-lg">Mediabunny allows you efficiently read data from any video or audio file, no matter the size: duration, resolution, rotation, tracks, codecs and other metadata, as well as raw or decoded media data from anywhere in the file. Load only what you need.</p>
<a class="!no-underline inline-flex items-center gap-1.5" :no-icon="true" href="/guide/reading-media-files">
Docs
<span class="vpi-arrow-right" />
@@ -236,7 +241,7 @@ await conversion.execute();
</div>
</div>
<div class="flex flex-col-reverse lg:flex-row gap-4 lg:gap-20 lg:items-center">
<div class="flex flex-col-reverse lg:flex-row gap-4 lg:gap-20 items-center">
<div class="relative flex-1 min-w-0">
<div class="absolute size-70 rounded-full bg-[#ff45ac]/0 top-1/2 left-1/2 -translate-x-1/2 -translate-y-1/2 blur-[200px]" />
<img class="relative" src="./assets/inspiring-io.svg">
@@ -323,7 +328,7 @@ await conversion.execute();
<template v-if="sponsors.gold.length > 0">
<h3 class="!text-2xl">Gold sponsors</h3>
<div class="flex flex-wrap mt-1 justify-center gap-1">
<a v-for="sponsor in sponsors.gold" :href="sponsor.url" target="_blank" class="flex items-center p-2 rounded-full hover:bg-(--vp-c-gray-3) !text-[initial] !no-underline">
<a v-for="sponsor in sponsors.gold" :href="sponsor.url" target="_blank" class="flex items-center p-2 rounded-full hover:bg-(--vp-c-gray-3) !text-(--vp-c-text-1) !no-underline">
<img :src="sponsor.image" class="size-16 rounded-full">
<p class="!my-0 !font-medium px-3">{{ sponsor.name }}</p>
</a>
@@ -332,7 +337,7 @@ await conversion.execute();
<template v-if="sponsors.individual.length > 0">
<h4>Individual sponsors</h4>
<div class="flex flex-wrap mt-1 justify-center">
<a v-for="sponsor in sponsors.individual" :href="sponsor.url" target="_blank" class="flex gap-1 w-24 flex-col items-center p-2 rounded-xl hover:bg-(--vp-c-gray-3) !text-[initial] !no-underline">
<a v-for="sponsor in sponsors.individual" :href="sponsor.url" target="_blank" class="flex gap-1 w-24 flex-col items-center p-2 rounded-xl hover:bg-(--vp-c-gray-3) !text-(--vp-c-text-1) !no-underline">
<img :src="sponsor.image" class="size-8 rounded-full">
<p class="!my-0 !font-medium text-xs !leading-4">{{ sponsor.name }}</p>
</a>
Binary file not shown.

After

Width:  |  Height:  |  Size: 6.0 KiB

@@ -91,13 +91,17 @@ const compressFile = async (file: File) => {
// Display the final media file
videoElement.style.display = '';
videoElement.src = URL.createObjectURL(new Blob([output.target.buffer!]));
videoElement.src = URL.createObjectURL(new Blob([output.target.buffer!], { type: output.format.mimeType }));
void videoElement.play();
compressionFacts.style.display = '';
compressionFacts.textContent
= `${(output.target.buffer!.byteLength / file.size * 100).toPrecision(3)}% of original size`;
} catch (error) {
console.error(error);
await currentConversion?.cancel();
errorElement.textContent = String(error);
clearInterval(currentIntervalId);
@@ -113,6 +117,7 @@ const compressFile = async (file: File) => {
selectMediaButton.addEventListener('click', () => {
const fileInput = document.createElement('input');
fileInput.type = 'file';
fileInput.accept = 'video/*,video/x-matroska,audio/*';
fileInput.addEventListener('change', () => {
const file = fileInput.files?.[0];
if (!file) {
+3 -2
View File
@@ -4,7 +4,7 @@
<meta charset="UTF-8">
<meta http-equiv="X-UA-Compatible" content="IE=edge">
<meta name="viewport" content="width=device-width, initial-scale=1.0">
<title>Live recording example | Mediabunny</title>
<title>Live recording & streaming example | Mediabunny</title>
<script type="module" src="../base.ts"></script>
<script type="module" src="./live-recording.ts"></script>
<link rel="stylesheet" href="../base.css">
@@ -12,7 +12,7 @@
</head>
<body class="flex flex-col items-center py-10 bg-zinc-50 text-zinc-800 dark:bg-zinc-900 dark:text-zinc-200 px-2">
<h1 class="text-3xl font-bold text-orange-500 text-center">Live recording example</h1>
<h1 class="text-3xl font-bold text-orange-500 text-center">Live recording & streaming example</h1>
<p class="max-w-lg text-center">The live canvas state and your microphone input will be written into a fragmented MP4 file and live-streamed to a <code>&lt;video&gt;</code> element.</p>
<button id="toggle-button" class="rounded-lg bg-zinc-200 dark:bg-zinc-750 hover:bg-zinc-300 dark:hover:bg-zinc-700 px-5 py-2 mt-4">
@@ -22,6 +22,7 @@
<hr class="w-full max-w-96 my-4 border-zinc-300 dark:border-zinc-700" style="display: none;">
<p id="error-element" class="text-red-500"></p>
<p id="warning-element" class="text-amber-500"></p>
<div class="flex gap-4" id="main-container" style="display: none;">
<div class="flex flex-col items-center">
+44 -15
View File
@@ -1,4 +1,5 @@
import {
canEncodeAudio,
CanvasSource,
MediaStreamAudioTrackSource,
Mp4OutputFormat,
@@ -13,6 +14,7 @@ const mainContainer = document.querySelector('#main-container') as HTMLDivElemen
const videoElement = document.querySelector('video') as HTMLVideoElement;
const downloadButton = document.querySelector('#download-button') as HTMLAnchorElement;
const errorElement = document.querySelector('#error-element') as HTMLParagraphElement;
const warningElement = document.querySelector('#warning-element') as HTMLParagraphElement;
const canvas = document.querySelector('canvas') as HTMLCanvasElement;
const context = canvas.getContext('2d', { alpha: false, desynchronized: true })!;
@@ -38,19 +40,30 @@ const startRecording = async () => {
mainContainer.style.display = 'none';
videoElement.src = '';
downloadButton.style.display = 'none';
errorElement.textContent = '';
warningElement.textContent = '';
// Paint a white background to the canvas
context.fillStyle = 'white';
context.fillRect(0, 0, canvas.width, canvas.height);
// Get user microphone
mediaStream = await navigator.mediaDevices.getUserMedia({ audio: true });
const audioIsEncodable = await canEncodeAudio('opus', {
bitrate: QUALITY_MEDIUM,
});
let audioTrack: MediaStreamAudioTrack | null = null;
if (audioIsEncodable) {
// Get user microphone
mediaStream = await navigator.mediaDevices.getUserMedia({ audio: true });
audioTrack = mediaStream.getAudioTracks()[0] ?? null;
} else {
warningElement.textContent
= 'Audio is not yet encodable by your browser, so the audio track has been omitted.';
}
horizontalRule.style.display = '';
mainContainer.style.display = '';
const audioTrack = mediaStream.getAudioTracks()[0];
// Create a new output file
output = new Output({
// We're using fragmented MP4 here; streamable WebM would also work
@@ -85,7 +98,7 @@ const startRecording = async () => {
// Add the video track, with the canvas as the source
videoSource = new CanvasSource(canvas, {
codec: 'vp9',
codec: 'avc',
bitrate: QUALITY_MEDIUM,
keyFrameInterval: 0.5,
latencyMode: 'realtime', // Allow the encoder to skip frames to keep up with real-time constraints
@@ -98,6 +111,8 @@ const startRecording = async () => {
codec: 'opus',
bitrate: QUALITY_MEDIUM,
});
audioSource.errorPromise.catch(cancelRecording); // Make sure errors are bubbled up
output.addAudioTrack(audioSource);
}
@@ -107,9 +122,9 @@ const startRecording = async () => {
readyForMoreFrames = true;
lastFrameNumber = -1;
// Start the video frame capture loop
void addVideoFrame();
videoCaptureInterval = window.setInterval(() => void addVideoFrame(), 1000 / frameRate);
// Start the video frame capture loop, making sure errors are caught
void addVideoFrame().catch(cancelRecording);
videoCaptureInterval = window.setInterval(() => void addVideoFrame().catch(cancelRecording), 1000 / frameRate);
const mimeType = await output.getMimeType();
sourceBuffer = mediaSource.addSourceBuffer(mimeType);
@@ -120,21 +135,35 @@ const startRecording = async () => {
toggleRecordingButton.textContent = 'Stop recording';
toggleRecordingButton.disabled = false;
} catch (error) {
errorElement.textContent = String(error);
mainContainer.style.display = 'none';
toggleRecordingButton.textContent = 'Start recording';
toggleRecordingButton.disabled = false;
recording = false;
await cancelRecording(error);
}
};
const cancelRecording = async (error: unknown) => {
if (!recording) {
return; // Already canceled
}
console.error(error);
errorElement.textContent = String(error);
clearInterval(videoCaptureInterval);
mainContainer.style.display = 'none';
toggleRecordingButton.textContent = 'Start recording';
toggleRecordingButton.disabled = false;
recording = false;
await output?.cancel();
mediaStream?.getTracks().forEach(track => track.stop());
};
const stopRecording = async () => {
toggleRecordingButton.textContent = 'Stopping...';
toggleRecordingButton.disabled = true;
clearInterval(videoCaptureInterval);
mediaStream.getTracks().forEach(track => track.stop());
mediaStream?.getTracks().forEach(track => track.stop());
await output.finalize();
+2
View File
@@ -29,6 +29,8 @@
<hr class="w-full max-w-96 my-4 border-zinc-300 dark:border-zinc-700" style="display: none;">
<p id="error-element" class="text-red-500"></p>
<p id="warning-element" class="text-amber-500 mb-1"></p>
<div id="player" class="relative bg-black rounded-xl shrink min-h-14 min-w-0 w-full max-w-5xl overflow-hidden select-none" style="display: none;">
<canvas class="size-full object-contain" width="1280" height="720"></canvas>
+37 -10
View File
@@ -30,6 +30,7 @@ const volumeIconWrapper = document.querySelector('#volume-icon-wrapper') as HTML
const volumeButton = document.querySelector('#volume-button') as HTMLButtonElement;
const fullscreenButton = document.querySelector('#fullscreen-button') as HTMLButtonElement;
const errorElement = document.querySelector('#error-element') as HTMLDivElement;
const warningElement = document.querySelector('#warning-element') as HTMLDivElement;
const context = canvas.getContext('2d', { alpha: false, desynchronized: true })!;
@@ -81,6 +82,8 @@ const initMediaPlayer = async (file: File) => {
fileNameElement.textContent = file.name;
horizontalRule.style.display = '';
playerContainer.style.display = 'none';
errorElement.textContent = '';
warningElement.textContent = '';
// Create an Input from the file
const input = new Input({
@@ -95,17 +98,38 @@ const initMediaPlayer = async (file: File) => {
let videoTrack = await input.getPrimaryVideoTrack();
let audioTrack = await input.getPrimaryAudioTrack();
if (!(await videoTrack?.canDecode())) {
// We can't decode the video track, so treat it like there is no video track
videoTrack = null;
let problemMessage = '';
if (videoTrack) {
if (videoTrack.codec === null) {
problemMessage += 'Unsupported video codec. ';
videoTrack = null;
} else if (!(await videoTrack.canDecode())) {
problemMessage += 'Unable to decode the video track. ';
videoTrack = null;
}
}
if (!(await audioTrack?.canDecode())) {
// We can't decode the audio track, so treat it like there is no audio track
audioTrack = null;
if (audioTrack) {
if (audioTrack.codec === null) {
problemMessage += 'Unsupported audio codec. ';
audioTrack = null;
} else if (!(await audioTrack.canDecode())) {
problemMessage += 'Unable to decode the audio track. ';
audioTrack = null;
}
}
if (!videoTrack && !audioTrack) {
throw new Error('Media file has no playable video or audio track.');
if (!problemMessage) {
problemMessage = 'No audio or video track found.';
}
throw new Error(problemMessage);
}
if (problemMessage) {
warningElement.textContent = problemMessage;
}
// We must create the audio context with the matching sample rate for correct acoustic results
@@ -151,8 +175,10 @@ const initMediaPlayer = async (file: File) => {
controlsElement.style.opacity = '1';
playerContainer.style.cursor = '';
}
} catch (e) {
errorElement.textContent = String(e);
} catch (error) {
console.error(error);
errorElement.textContent = String(error);
playerContainer.style.display = 'none';
}
};
@@ -253,7 +279,7 @@ const runAudioIterator = async () => {
}
// To play back audio, we loop over all audio chunks (typically very short) of the file and play them at the correct
// timestamp. The result is a continuous, uninteruppted audio signal.
// timestamp. The result is a continuous, uninterrupted audio signal.
for await (const { buffer, timestamp } of audioBufferIterator!) {
const node = audioContext!.createBufferSource();
node.buffer = buffer;
@@ -544,6 +570,7 @@ const formatSeconds = (seconds: number) => {
selectMediaButton.addEventListener('click', () => {
const fileInput = document.createElement('input');
fileInput.type = 'file';
fileInput.accept = 'video/*,video/x-matroska,audio/*';
fileInput.addEventListener('change', () => {
const file = fileInput.files?.[0];
if (!file) {
@@ -129,6 +129,8 @@ const renderObject = (object: Record<string, unknown>) => {
listItem.removeChild(loadingSpan);
listItem.appendChild(renderValue(resolvedValue));
}).catch((error) => {
console.error(error);
// Show the promise error
listItem.removeChild(loadingSpan);
const errorSpan = document.createElement('span');
@@ -155,6 +157,7 @@ const shortDelay = () => {
selectMediaButton.addEventListener('click', () => {
const fileInput = document.createElement('input');
fileInput.type = 'file';
fileInput.accept = 'video/*,video/x-matroska,audio/*';
fileInput.addEventListener('change', () => {
const file = fileInput.files?.[0];
if (!file) {
+1 -1
View File
@@ -38,7 +38,7 @@
<p id="error-element" class="text-red-500"></p>
<div class="w-full max-w-80 h-2 rounded-full bg-zinc-200 dark:bg-zinc-750 overflow-hidden" id="progress-bar-container" style="display: none;">
<div class="h-full bg-teal-500 w-0" id="progress-bar"></div>
<div class="h-full bg-pink-500 w-0" id="progress-bar"></div>
</div>
<p class="text-xs font-medium mt-1.5 tabular-nums" id="progress-text" style="display: none;"></p>
@@ -7,6 +7,7 @@ import {
QUALITY_HIGH,
getFirstEncodableAudioCodec,
getFirstEncodableVideoCodec,
OutputFormat,
} from 'mediabunny';
const durationSlider = document.querySelector('#duration-slider') as HTMLInputElement;
@@ -57,6 +58,8 @@ let currentScaleIndex = 0;
let collisionCount = 0;
let collisionsPerScale = 0;
let output: Output<OutputFormat, BufferTarget>;
/** === MAIN VIDEO FILE GENERATION LOGIC === */
const generateVideo = async () => {
@@ -82,7 +85,7 @@ const generateVideo = async () => {
initScene(duration);
// Create a new output file
const output = new Output({
output = new Output({
target: new BufferTarget(), // Stored in memory
format: new Mp4OutputFormat(),
});
@@ -177,13 +180,17 @@ const generateVideo = async () => {
videoInfo.style.display = '';
// Display and play the resulting media file
const videoBlob = new Blob([output.target.buffer!], { type: 'video/mp4' });
const videoBlob = new Blob([output.target.buffer!], { type: output.format.mimeType });
resultVideo.src = URL.createObjectURL(videoBlob);
void resultVideo.play();
const fileSizeMiB = (videoBlob.size / (1024 * 1024)).toPrecision(3);
videoInfo.textContent = `File size: ${fileSizeMiB} MiB`;
} catch (error) {
console.error(error);
await output?.cancel();
clearInterval(progressInterval);
errorElement.textContent = String(error);
progressBarContainer.style.display = 'none';
@@ -15,7 +15,7 @@ const THUMBNAIL_SIZE = 200;
const generateThumbnails = async (file: File) => {
fileNameElement.textContent = file.name;
horizontalRule.style.display = '';
errorElement.innerHTML = '';
errorElement.textContent = '';
thumbnailContainer.innerHTML = '';
try {
@@ -30,6 +30,14 @@ const generateThumbnails = async (file: File) => {
throw new Error('File has no video track.');
}
if (videoTrack.codec === null) {
throw new Error('Unsupported video codec.');
}
if (!(await videoTrack.canDecode())) {
throw new Error('Unable to decode the video track.');
}
// Compute width and height of the thumbnails such that the larger dimension is equal to THUMBNAIL_SIZE
const width = videoTrack.displayWidth > videoTrack.displayHeight
? THUMBNAIL_SIZE
@@ -88,8 +96,10 @@ const generateThumbnails = async (file: File) => {
i++;
}
} catch (e) {
errorElement.textContent = String(e);
} catch (error) {
console.error(error);
errorElement.textContent = String(error);
thumbnailContainer.innerHTML = '';
}
};
@@ -99,6 +109,7 @@ const generateThumbnails = async (file: File) => {
selectMediaButton.addEventListener('click', () => {
const fileInput = document.createElement('input');
fileInput.type = 'file';
fileInput.accept = 'video/*,video/x-matroska,audio/*';
fileInput.addEventListener('change', () => {
const file = fileInput.files?.[0];
if (!file) {
+28 -42
View File
@@ -1,16 +1,16 @@
{
"name": "mediabunny",
"version": "0.1.0",
"version": "1.4.4",
"lockfileVersion": 3,
"requires": true,
"packages": {
"": {
"name": "mediabunny",
"version": "0.1.0",
"version": "1.4.4",
"license": "MPL-2.0",
"dependencies": {
"@types/dom-mediacapture-transform": "^0.1.11",
"@types/dom-webcodecs": "^0.1.15"
"@types/dom-webcodecs": "0.1.13"
},
"devDependencies": {
"@eslint/js": "^9.22.0",
@@ -319,9 +319,9 @@
}
},
"node_modules/@babel/helper-string-parser": {
"version": "7.25.9",
"resolved": "https://registry.npmjs.org/@babel/helper-string-parser/-/helper-string-parser-7.25.9.tgz",
"integrity": "sha512-4A/SCr/2KLd5jrtOMFzaKjVtAei3+2r/NChoBNoZ3EyP/+GlhoaEGoWOZUmFmoITP7zOJyHIMm+DYRd8o3PvHA==",
"version": "7.27.1",
"resolved": "https://registry.npmjs.org/@babel/helper-string-parser/-/helper-string-parser-7.27.1.tgz",
"integrity": "sha512-qMlSxKbpRlAridDExk92nSobyDdpPijUq2DW6oDnUqd0iOGxmQjyqhMIihI9+zv4LPyZdRje2cavWPbCbWm3eA==",
"dev": true,
"license": "MIT",
"engines": {
@@ -329,9 +329,9 @@
}
},
"node_modules/@babel/helper-validator-identifier": {
"version": "7.25.9",
"resolved": "https://registry.npmjs.org/@babel/helper-validator-identifier/-/helper-validator-identifier-7.25.9.tgz",
"integrity": "sha512-Ed61U6XJc3CVRfkERJWDz4dJwKe7iLmmJsbOGu9wSloNSFttHV0I8g6UAgb7qnK5ly5bGLPd4oXZlxCdANBOWQ==",
"version": "7.27.1",
"resolved": "https://registry.npmjs.org/@babel/helper-validator-identifier/-/helper-validator-identifier-7.27.1.tgz",
"integrity": "sha512-D2hP9eA+Sqx1kBZgzxZh0y1trbuU+JoDkiEwqhQ36nodYqJwyEIhPSdMNd7lOm/4io72luTPWH20Yda0xOuUow==",
"dev": true,
"license": "MIT",
"engines": {
@@ -339,13 +339,13 @@
}
},
"node_modules/@babel/parser": {
"version": "7.27.0",
"resolved": "https://registry.npmjs.org/@babel/parser/-/parser-7.27.0.tgz",
"integrity": "sha512-iaepho73/2Pz7w2eMS0Q5f83+0RKI7i4xmiYeBmDzfRVbQtTOG7Ts0S4HzJVsTMGI9keU8rNfuZr8DKfSt7Yyg==",
"version": "7.28.0",
"resolved": "https://registry.npmjs.org/@babel/parser/-/parser-7.28.0.tgz",
"integrity": "sha512-jVZGvOxOuNSsuQuLRTh13nU0AogFlw32w/MT+LV6D3sP5WdbW61E77RnkbaO2dUvmPAYrBDJXGn5gGS6tH4j8g==",
"dev": true,
"license": "MIT",
"dependencies": {
"@babel/types": "^7.27.0"
"@babel/types": "^7.28.0"
},
"bin": {
"parser": "bin/babel-parser.js"
@@ -355,14 +355,14 @@
}
},
"node_modules/@babel/types": {
"version": "7.27.0",
"resolved": "https://registry.npmjs.org/@babel/types/-/types-7.27.0.tgz",
"integrity": "sha512-H45s8fVLYjbhFH62dIJ3WtmJ6RSPt/3DRO0ZcT2SUiYiQyz3BLVb9ADEnLl91m74aQPS3AzzeajZHYOalWe3bg==",
"version": "7.28.0",
"resolved": "https://registry.npmjs.org/@babel/types/-/types-7.28.0.tgz",
"integrity": "sha512-jYnje+JyZG5YThjHiF28oT4SIZLnYOcSBb6+SDaFIyzDVSkXQmQQYclJ2R+YxcdmK0AX6x1E5OQNtuh3jHDrUg==",
"dev": true,
"license": "MIT",
"dependencies": {
"@babel/helper-string-parser": "^7.25.9",
"@babel/helper-validator-identifier": "^7.25.9"
"@babel/helper-string-parser": "^7.27.1",
"@babel/helper-validator-identifier": "^7.27.1"
},
"engines": {
"node": ">=6.9.0"
@@ -1238,18 +1238,14 @@
}
},
"node_modules/@jridgewell/gen-mapping": {
"version": "0.3.8",
"resolved": "https://registry.npmjs.org/@jridgewell/gen-mapping/-/gen-mapping-0.3.8.tgz",
"integrity": "sha512-imAbBGkb+ebQyxKgzv5Hu2nmROxoDOXHh80evxdoXNOrvAnVx7zimzc1Oo5h9RlfV4vPXaE2iM5pOFbvOCClWA==",
"version": "0.3.12",
"resolved": "https://registry.npmjs.org/@jridgewell/gen-mapping/-/gen-mapping-0.3.12.tgz",
"integrity": "sha512-OuLGC46TjB5BbN1dH8JULVVZY4WTdkF7tV9Ys6wLL1rubZnCMstOhNHueU5bLCrnRuDhKPDM4g6sw4Bel5Gzqg==",
"dev": true,
"license": "MIT",
"dependencies": {
"@jridgewell/set-array": "^1.2.1",
"@jridgewell/sourcemap-codec": "^1.4.10",
"@jridgewell/sourcemap-codec": "^1.5.0",
"@jridgewell/trace-mapping": "^0.3.24"
},
"engines": {
"node": ">=6.0.0"
}
},
"node_modules/@jridgewell/resolve-uri": {
@@ -1262,16 +1258,6 @@
"node": ">=6.0.0"
}
},
"node_modules/@jridgewell/set-array": {
"version": "1.2.1",
"resolved": "https://registry.npmjs.org/@jridgewell/set-array/-/set-array-1.2.1.tgz",
"integrity": "sha512-R8gLRTZeyp03ymzP/6Lil/28tGeGEzhx1q2k703KGWRAI1VdvPIXdG70VJc2pAMw3NA6JKL5hhFu1sJX0Mnn/A==",
"dev": true,
"license": "MIT",
"engines": {
"node": ">=6.0.0"
}
},
"node_modules/@jridgewell/sourcemap-codec": {
"version": "1.5.0",
"resolved": "https://registry.npmjs.org/@jridgewell/sourcemap-codec/-/sourcemap-codec-1.5.0.tgz",
@@ -1280,9 +1266,9 @@
"license": "MIT"
},
"node_modules/@jridgewell/trace-mapping": {
"version": "0.3.25",
"resolved": "https://registry.npmjs.org/@jridgewell/trace-mapping/-/trace-mapping-0.3.25.tgz",
"integrity": "sha512-vNk6aEwybGtawWmy/PzwnGDOjCkLWSD2wqvjGGAgOAwCGWySYXfYoxt00IJkTF+8Lb57DwOb3Aa0o9CApepiYQ==",
"version": "0.3.29",
"resolved": "https://registry.npmjs.org/@jridgewell/trace-mapping/-/trace-mapping-0.3.29.tgz",
"integrity": "sha512-uw6guiW/gcAGPDhLmd77/6lW8QLeiV5RUTsAX46Db6oLhGaVj4lhnPwb184s1bkc8kdVg/+h988dro8GRDpmYQ==",
"dev": true,
"license": "MIT",
"dependencies": {
@@ -2457,9 +2443,9 @@
}
},
"node_modules/@types/dom-webcodecs": {
"version": "0.1.15",
"resolved": "https://registry.npmjs.org/@types/dom-webcodecs/-/dom-webcodecs-0.1.15.tgz",
"integrity": "sha512-omOlCPvTWyPm4ZE5bZUhlSvnHM2ZWM2U+1cPiYFL/e8aV5O9MouELp+L4dMKNTON0nTeHqEg+KWDfFQMY5Wkaw==",
"version": "0.1.13",
"resolved": "https://registry.npmjs.org/@types/dom-webcodecs/-/dom-webcodecs-0.1.13.tgz",
"integrity": "sha512-O5hkiFIcjjszPIYyUSyvScyvrBoV3NOEEZx/pMlsu44TKzWNkLVBBxnxJz42in5n3QIolYOcBYFCPZZ0h8SkwQ==",
"license": "MIT"
},
"node_modules/@types/estree": {
+15 -11
View File
@@ -1,25 +1,27 @@
{
"name": "mediabunny",
"author": "Vanilagy",
"version": "1.0.0",
"version": "1.5.0",
"description": "Pure TypeScript media toolkit for reading, writing, and converting media files, directly in the browser.",
"type": "module",
"main": "./dist/mediabunny.js",
"module": "./dist/mediabunny.mjs",
"types": "./dist/mediabunny.d.ts",
"main": "./dist/bundles/mediabunny.cjs",
"module": "./dist/modules/index.js",
"types": "./dist/modules/index.d.ts",
"exports": {
"types": "./dist/mediabunny.d.ts",
"import": "./dist/mediabunny.mjs",
"require": "./dist/mediabunny.js"
"types": "./dist/modules/index.d.ts",
"import": "./dist/modules/index.js",
"require": "./dist/bundles/mediabunny.cjs"
},
"files": [
"README.md",
"package.json",
"LICENSE",
"dist"
"dist",
"src"
],
"sideEffects": false,
"scripts": {
"build": "tsx scripts/ensure-license-headers.ts && node build.mjs && tsc -p src && api-extractor run && npm run check-docblocks && tsx scripts/append-namespace.ts",
"build": "rm -rf dist && tsx scripts/ensure-license-headers.ts && tsc -p src && npm run fix-build-import-paths && node build.mjs && api-extractor run && npm run check-docblocks && npm run append-namespace",
"watch": "node build.mjs --watch",
"lint": "eslint .",
"check": "tsc -p src --noEmit && tsc -p scripts --noEmit && tsc -p tsconfig.vite.json --noEmit && rm tsconfig.vite.tsbuildinfo",
@@ -28,7 +30,9 @@
"docs:build": "vitepress build docs && npm run examples:build",
"docs:preview": "vitepress preview docs",
"dev": "vite",
"examples:build": "vite build"
"examples:build": "vite build",
"fix-build-import-paths": "tsx scripts/add-import-extensions.ts",
"append-namespace": "echo 'export as namespace Mediabunny;' >> dist/mediabunny.d.ts"
},
"license": "MPL-2.0",
"repository": {
@@ -45,7 +49,7 @@
},
"dependencies": {
"@types/dom-mediacapture-transform": "^0.1.11",
"@types/dom-webcodecs": "^0.1.15"
"@types/dom-webcodecs": "0.1.13"
},
"devDependencies": {
"@eslint/js": "^9.22.0",
+37
View File
@@ -0,0 +1,37 @@
import * as fs from 'fs';
import * as path from 'path';
// .js extensions are technically required in compliant ECMAScript, and Webpack needs them, so we add them here.
const walkDir = (dir: string) => {
const files: string[] = [];
const items = fs.readdirSync(dir);
for (const item of items) {
const fullPath = path.join(dir, item);
const stat = fs.statSync(fullPath);
if (stat.isDirectory()) {
files.push(...walkDir(fullPath));
} else if (item.endsWith('.js')) {
files.push(fullPath);
}
}
return files;
};
const fixFile = (filePath: string) => {
const content = fs.readFileSync(filePath, 'utf8');
const fixed = content.replace(
/(\s+from\s+['"])([^'"]*)(['"])/g,
'$1$2.js$3',
);
if (content !== fixed) {
fs.writeFileSync(filePath, fixed);
}
};
const jsFiles = walkDir('dist');
jsFiles.forEach(fixFile);
-3
View File
@@ -1,3 +0,0 @@
import { appendFileSync } from 'fs';
appendFileSync('dist/mediabunny.d.ts', '\nexport as namespace Mediabunny;');
+3 -1
View File
@@ -1,5 +1,5 @@
import ts from 'typescript';
import * as fs from 'node:fs';
import * as fs from 'fs';
const checkDocblocks = (filePath: string) => {
const program = ts.createProgram([filePath], {});
@@ -17,6 +17,8 @@ const checkDocblocks = (filePath: string) => {
ts.isInterfaceDeclaration(node)
|| ts.isClassDeclaration(node)
|| ts.isMethodDeclaration(node)
|| ts.isGetAccessorDeclaration(node)
|| ts.isSetAccessorDeclaration(node)
|| ts.isPropertyDeclaration(node)
|| ts.isFunctionDeclaration(node)
|| ts.isTypeAliasDeclaration(node)
+342 -180
View File
@@ -7,7 +7,18 @@
*/
import { VP9_LEVEL_TABLE } from './codec';
import { assert, Bitstream, last, readExpGolomb, readSignedExpGolomb, toDataView } from './misc';
import { InputVideoTrack } from './input-track';
import {
assert,
assertNever,
Bitstream,
last,
readExpGolomb,
readSignedExpGolomb,
toDataView,
toUint8Array,
} from './misc';
import { EncodedPacket, PacketType } from './packet';
// References for AVC/HEVC code:
// ISO 14496-15
@@ -72,6 +83,39 @@ const findNalUnitsInAnnexB = (packetData: Uint8Array) => {
return nalUnits;
};
/** Finds all NAL units in an AVC packet in length-prefixed format. */
const findNalUnitsInLengthPrefixed = (packetData: Uint8Array, lengthSize: 1 | 2 | 3 | 4) => {
const nalUnits: Uint8Array[] = [];
let offset = 0;
const dataView = new DataView(packetData.buffer, packetData.byteOffset, packetData.byteLength);
while (offset + lengthSize <= packetData.length) {
let nalUnitLength: number;
if (lengthSize === 1) {
nalUnitLength = dataView.getUint8(offset);
} else if (lengthSize === 2) {
nalUnitLength = dataView.getUint16(offset, false);
} else if (lengthSize === 3) {
nalUnitLength = (dataView.getUint16(offset, false) << 8) + dataView.getUint8(offset + 2);
} else if (lengthSize === 4) {
nalUnitLength = dataView.getUint32(offset, false);
} else {
assertNever(lengthSize);
assert(false);
}
offset += lengthSize;
const nalUnit = packetData.subarray(offset, offset + nalUnitLength);
nalUnits.push(nalUnit);
offset += nalUnitLength;
}
return nalUnits;
};
const removeEmulationPreventionBytes = (data: Uint8Array) => {
const result: number[] = [];
const len = data.length;
@@ -879,30 +923,6 @@ export const extractVp9CodecInfoFromPacket = (
// https://storage.googleapis.com/downloads.webmproject.org/docs/vp9/vp9-bitstream-specification-v0.7-20170222-draft.pdf
// http://downloads.webmproject.org/docs/vp9/vp9-bitstream_superframe-and-uncompressed-header_v1.0.pdf
// Handle superframe
const lastByte = packet[packet.length - 1];
if (lastByte && (lastByte & 0xe0) === 0xc0) { // Is superframe
const bytesPerFrameSize = ((lastByte & 0x18) >> 3) + 1;
const numFrames = (lastByte & 0x07) + 1;
const indexSize = 2 + numFrames * bytesPerFrameSize;
// Verify matching marker bytes
if (packet[packet.length - indexSize] !== lastByte) {
return null;
}
// Get first frame size
let frameSize = 0;
const offset = packet.length - indexSize + 1;
for (let i = 0; i < bytesPerFrameSize; i++) {
if (!packet[offset + i]) return null;
frameSize |= packet[offset + i]! << (8 * i);
}
packet = packet.subarray(0, frameSize);
}
const bitstream = new Bitstream(packet);
// Frame marker (0b10)
@@ -1048,13 +1068,8 @@ export type Av1CodecInfo = {
chromaSamplePosition: number;
};
/**
* When AV1 codec information is not provided by the container, we can still try to extract the information by digging
* into the AV1 bitstream.
*/
export const extractAv1CodecInfoFromPacket = (
packet: Uint8Array,
): Av1CodecInfo | null => {
/** Iterates over all OBUs in an AV1 packet bistream. */
export function* iterateAv1PacketObus(packet: Uint8Array) {
// https://aomediacodec.github.io/av1-spec/av1-spec.pdf
const bitstream = new Bitstream(packet);
@@ -1064,7 +1079,6 @@ export const extractAv1CodecInfoFromPacket = (
for (let i = 0; i < 8; i++) {
const byte = bitstream.readAlignedByte();
if (byte === undefined) return 0;
value |= ((byte & 0x7f) << (i * 7));
@@ -1088,11 +1102,11 @@ export const extractAv1CodecInfoFromPacket = (
while (bitstream.getBitsLeft() >= 8) {
// Parse OBU header
const obuHeader = bitstream.readBits(8);
const obuType = (obuHeader >> 3) & 0xf;
const obuExtension = (obuHeader >> 2) & 0x1;
const obuHasSizeField = (obuHeader >> 1) & 0x1;
bitstream.skipBits(1);
const obuType = bitstream.readBits(4);
const obuExtension = bitstream.readBits(1);
const obuHasSizeField = bitstream.readBits(1);
bitstream.skipBits(1);
// Skip extension header if present
if (obuExtension) {
@@ -1103,159 +1117,180 @@ export const extractAv1CodecInfoFromPacket = (
let obuSize: number;
if (obuHasSizeField) {
const obuSizeValue = readLeb128();
if (obuSizeValue === null) return null; // It was invalid
if (obuSizeValue === null) return; // It was invalid
obuSize = obuSizeValue;
} else {
// Calculate remaining bits and convert to bytes, rounding down
obuSize = Math.floor(bitstream.getBitsLeft() / 8);
}
// We're only interested in Sequence Header OBU (type 1)
if (obuType === 1) {
// Read sequence header fields
const seqProfile = bitstream.readBits(3);
assert(bitstream.pos % 8 === 0);
// eslint-disable-next-line @typescript-eslint/no-unused-vars
const stillPicture = bitstream.readBits(1);
const reducedStillPictureHeader = bitstream.readBits(1);
let seqLevel = 0;
let seqTier = 0;
let bufferDelayLengthMinus1 = 0;
if (reducedStillPictureHeader) {
seqLevel = bitstream.readBits(5);
} else {
// Parse timing_info_present_flag
const timingInfoPresentFlag = bitstream.readBits(1);
if (timingInfoPresentFlag) {
// Skip timing info (num_units_in_display_tick, time_scale, equal_picture_interval)
bitstream.skipBits(32); // num_units_in_display_tick
bitstream.skipBits(32); // time_scale
const equalPictureInterval = bitstream.readBits(1);
if (equalPictureInterval) {
// Skip num_ticks_per_picture_minus_1 (uvlc)
// Since this is variable length, we'd need to implement uvlc reading
// For now, we'll return null as this is rare
return null;
}
}
// Parse decoder_model_info_present_flag
const decoderModelInfoPresentFlag = bitstream.readBits(1);
if (decoderModelInfoPresentFlag) {
// Store buffer_delay_length_minus_1 instead of just skipping
bufferDelayLengthMinus1 = bitstream.readBits(5);
bitstream.skipBits(32); // num_units_in_decoding_tick
bitstream.skipBits(5); // buffer_removal_time_length_minus_1
bitstream.skipBits(5); // frame_presentation_time_length_minus_1
}
// Parse operating_points_cnt_minus_1
const operatingPointsCntMinus1 = bitstream.readBits(5);
// For each operating point
for (let i = 0; i <= operatingPointsCntMinus1; i++) {
// operating_point_idc[i]
bitstream.skipBits(12);
// seq_level_idx[i]
const seqLevelIdx = bitstream.readBits(5);
if (i === 0) {
seqLevel = seqLevelIdx;
}
if (seqLevelIdx > 7) {
// seq_tier[i]
const seqTierTemp = bitstream.readBits(1);
if (i === 0) {
seqTier = seqTierTemp;
}
}
if (decoderModelInfoPresentFlag) {
// decoder_model_present_for_this_op[i]
const decoderModelPresentForThisOp = bitstream.readBits(1);
if (decoderModelPresentForThisOp) {
const n = bufferDelayLengthMinus1 + 1;
bitstream.skipBits(n); // decoder_buffer_delay[op]
bitstream.skipBits(n); // encoder_buffer_delay[op]
bitstream.skipBits(1); // low_delay_mode_flag[op]
}
}
// initial_display_delay_present_flag
const initialDisplayDelayPresentFlag = bitstream.readBits(1);
if (initialDisplayDelayPresentFlag) {
// initial_display_delay_minus_1[i]
bitstream.skipBits(4);
}
}
}
const highBitdepth = bitstream.readBits(1);
let bitDepth = 8;
if (seqProfile === 2 && highBitdepth) {
const twelveBit = bitstream.readBits(1);
bitDepth = twelveBit ? 12 : 10;
} else if (seqProfile <= 2) {
bitDepth = highBitdepth ? 10 : 8;
}
let monochrome = 0;
if (seqProfile !== 1) {
monochrome = bitstream.readBits(1);
}
let chromaSubsamplingX = 1;
let chromaSubsamplingY = 1;
let chromaSamplePosition = 0;
if (!monochrome) {
if (seqProfile === 0) {
chromaSubsamplingX = 1;
chromaSubsamplingY = 1;
} else if (seqProfile === 1) {
chromaSubsamplingX = 0;
chromaSubsamplingY = 0;
} else {
if (bitDepth === 12) {
chromaSubsamplingX = bitstream.readBits(1);
if (chromaSubsamplingX) {
chromaSubsamplingY = bitstream.readBits(1);
}
}
}
if (chromaSubsamplingX && chromaSubsamplingY) {
chromaSamplePosition = bitstream.readBits(2);
}
}
return {
profile: seqProfile,
level: seqLevel,
tier: seqTier,
bitDepth,
monochrome,
chromaSubsamplingX,
chromaSubsamplingY,
chromaSamplePosition,
};
}
yield {
type: obuType,
data: packet.subarray(bitstream.pos / 8, bitstream.pos / 8 + obuSize),
};
// Move to next OBU
// The OBU size is in bytes, so skip that many bytes.
bitstream.skipBits(obuSize * 8);
}
};
/**
* When AV1 codec information is not provided by the container, we can still try to extract the information by digging
* into the AV1 bitstream.
*/
export const extractAv1CodecInfoFromPacket = (
packet: Uint8Array,
): Av1CodecInfo | null => {
// https://aomediacodec.github.io/av1-spec/av1-spec.pdf
for (const { type, data } of iterateAv1PacketObus(packet)) {
if (type !== 1) {
continue; // 1 == OBU_SEQUENCE_HEADER
}
const bitstream = new Bitstream(data);
// Read sequence header fields
const seqProfile = bitstream.readBits(3);
// eslint-disable-next-line @typescript-eslint/no-unused-vars
const stillPicture = bitstream.readBits(1);
const reducedStillPictureHeader = bitstream.readBits(1);
let seqLevel = 0;
let seqTier = 0;
let bufferDelayLengthMinus1 = 0;
if (reducedStillPictureHeader) {
seqLevel = bitstream.readBits(5);
} else {
// Parse timing_info_present_flag
const timingInfoPresentFlag = bitstream.readBits(1);
if (timingInfoPresentFlag) {
// Skip timing info (num_units_in_display_tick, time_scale, equal_picture_interval)
bitstream.skipBits(32); // num_units_in_display_tick
bitstream.skipBits(32); // time_scale
const equalPictureInterval = bitstream.readBits(1);
if (equalPictureInterval) {
// Skip num_ticks_per_picture_minus_1 (uvlc)
// Since this is variable length, we'd need to implement uvlc reading
// For now, we'll return null as this is rare
return null;
}
}
// Parse decoder_model_info_present_flag
const decoderModelInfoPresentFlag = bitstream.readBits(1);
if (decoderModelInfoPresentFlag) {
// Store buffer_delay_length_minus_1 instead of just skipping
bufferDelayLengthMinus1 = bitstream.readBits(5);
bitstream.skipBits(32); // num_units_in_decoding_tick
bitstream.skipBits(5); // buffer_removal_time_length_minus_1
bitstream.skipBits(5); // frame_presentation_time_length_minus_1
}
// Parse operating_points_cnt_minus_1
const operatingPointsCntMinus1 = bitstream.readBits(5);
// For each operating point
for (let i = 0; i <= operatingPointsCntMinus1; i++) {
// operating_point_idc[i]
bitstream.skipBits(12);
// seq_level_idx[i]
const seqLevelIdx = bitstream.readBits(5);
if (i === 0) {
seqLevel = seqLevelIdx;
}
if (seqLevelIdx > 7) {
// seq_tier[i]
const seqTierTemp = bitstream.readBits(1);
if (i === 0) {
seqTier = seqTierTemp;
}
}
if (decoderModelInfoPresentFlag) {
// decoder_model_present_for_this_op[i]
const decoderModelPresentForThisOp = bitstream.readBits(1);
if (decoderModelPresentForThisOp) {
const n = bufferDelayLengthMinus1 + 1;
bitstream.skipBits(n); // decoder_buffer_delay[op]
bitstream.skipBits(n); // encoder_buffer_delay[op]
bitstream.skipBits(1); // low_delay_mode_flag[op]
}
}
// initial_display_delay_present_flag
const initialDisplayDelayPresentFlag = bitstream.readBits(1);
if (initialDisplayDelayPresentFlag) {
// initial_display_delay_minus_1[i]
bitstream.skipBits(4);
}
}
}
const highBitdepth = bitstream.readBits(1);
let bitDepth = 8;
if (seqProfile === 2 && highBitdepth) {
const twelveBit = bitstream.readBits(1);
bitDepth = twelveBit ? 12 : 10;
} else if (seqProfile <= 2) {
bitDepth = highBitdepth ? 10 : 8;
}
let monochrome = 0;
if (seqProfile !== 1) {
monochrome = bitstream.readBits(1);
}
let chromaSubsamplingX = 1;
let chromaSubsamplingY = 1;
let chromaSamplePosition = 0;
if (!monochrome) {
if (seqProfile === 0) {
chromaSubsamplingX = 1;
chromaSubsamplingY = 1;
} else if (seqProfile === 1) {
chromaSubsamplingX = 0;
chromaSubsamplingY = 0;
} else {
if (bitDepth === 12) {
chromaSubsamplingX = bitstream.readBits(1);
if (chromaSubsamplingX) {
chromaSubsamplingY = bitstream.readBits(1);
}
}
}
if (chromaSubsamplingX && chromaSubsamplingY) {
chromaSamplePosition = bitstream.readBits(2);
}
}
return {
profile: seqProfile,
level: seqLevel,
tier: seqTier,
bitDepth,
monochrome,
chromaSubsamplingX,
chromaSubsamplingY,
chromaSamplePosition,
};
}
return null;
};
@@ -1392,3 +1427,130 @@ export const parseModesFromVorbisSetupPacket = (setupHeader: Uint8Array) => {
return { modeBlockflags };
};
/** Determines a packet's type (key or delta) by digging into the packet bitstream. */
export const determineVideoPacketType = async (
videoTrack: InputVideoTrack,
packet: EncodedPacket,
): Promise<PacketType | null> => {
assert(videoTrack.codec);
switch (videoTrack.codec) {
case 'avc': {
const decoderConfig = await videoTrack.getDecoderConfig();
assert(decoderConfig);
let nalUnits: Uint8Array[];
if (decoderConfig.description) {
// Stream is length-prefixed. Let's extract the size of the length prefix from the decoder config
const bytes = toUint8Array(decoderConfig.description);
const lengthSizeMinusOne = bytes[4]! & 0b11;
const lengthSize = (lengthSizeMinusOne + 1) as 1 | 2 | 3 | 4;
nalUnits = findNalUnitsInLengthPrefixed(packet.data, lengthSize);
} else {
// Stream is in Annex B format
nalUnits = findNalUnitsInAnnexB(packet.data);
}
const isKeyframe = nalUnits.some(x => extractNalUnitTypeForAvc(x) === 5);
return isKeyframe ? 'key' : 'delta';
};
case 'hevc': {
const decoderConfig = await videoTrack.getDecoderConfig();
assert(decoderConfig);
let nalUnits: Uint8Array[];
if (decoderConfig.description) {
// Stream is length-prefixed. Let's extract the size of the length prefix from the decoder config
const bytes = toUint8Array(decoderConfig.description);
const lengthSizeMinusOne = bytes[21]! & 0b11;
const lengthSize = (lengthSizeMinusOne + 1) as 1 | 2 | 3 | 4;
nalUnits = findNalUnitsInLengthPrefixed(packet.data, lengthSize);
} else {
// Stream is in Annex B format
nalUnits = findNalUnitsInAnnexB(packet.data);
}
const isKeyframe = nalUnits.some((x) => {
const type = extractNalUnitTypeForHevc(x);
return 16 <= type && type <= 23;
});
return isKeyframe ? 'key' : 'delta';
};
case 'vp8': {
// VP8, once again, by far the easiest to deal with.
const frameType = packet.data[0]! & 0b1;
return frameType === 0 ? 'key' : 'delta';
};
case 'vp9': {
const bitstream = new Bitstream(packet.data);
if (bitstream.readBits(2) !== 2) {
return null;
};
const profileLowBit = bitstream.readBits(1);
const profileHighBit = bitstream.readBits(1);
const profile = (profileHighBit << 1) + profileLowBit;
// Skip reserved bit for profile 3
if (profile === 3) {
bitstream.skipBits(1);
}
const showExistingFrame = bitstream.readBits(1);
if (showExistingFrame) {
return null;
}
const frameType = bitstream.readBits(1);
return frameType === 0 ? 'key' : 'delta';
};
case 'av1': {
let reducedStillPictureHeader = false;
for (const { type, data } of iterateAv1PacketObus(packet.data)) {
if (type === 1) { // OBU_SEQUENCE_HEADER
const bitstream = new Bitstream(data);
bitstream.skipBits(4);
reducedStillPictureHeader = !!bitstream.readBits(1);
} else if (
type === 3 // OBU_FRAME_HEADER
|| type === 6 // OBU_FRAME
|| type === 7 // OBU_REDUNDANT_FRAME_HEADER
) {
if (reducedStillPictureHeader) {
return 'key';
}
const bitstream = new Bitstream(data);
const showExistingFrame = bitstream.readBits(1);
if (showExistingFrame) {
return null;
}
const frameType = bitstream.readBits(2);
return frameType === 0 ? 'key' : 'delta';
}
}
return null;
};
default: {
assertNever(videoTrack.codec);
assert(false);
};
}
};
+148 -15
View File
@@ -77,6 +77,11 @@ export type ConversionOptions = {
* rotation is _in addition to_ the natural rotation of the input video as specified in input file's metadata.
*/
rotate?: Rotation;
/**
* The desired frame rate of the output video, in hertz. If not specified, the original input frame rate will
* be used (which may be variable).
*/
frameRate?: number;
/** The desired output video codec. */
codec?: VideoCodec;
/** The desired bitrate of the output video. */
@@ -261,8 +266,14 @@ export class Conversion {
if (options.video?.rotate !== undefined && ![0, 90, 180, 270].includes(options.video.rotate)) {
throw new TypeError('options.video.rotate, when provided, must be 0, 90, 180 or 270.');
}
if (
options.video?.frameRate !== undefined
&& (!Number.isFinite(options.video.frameRate) || options.video.frameRate <= 0)
) {
throw new TypeError('options.video.frameRate, when provided, must be a finite positive number.');
}
if (options.audio !== undefined && (!options.audio || typeof options.audio !== 'object')) {
throw new TypeError('options.video, when provided, must be an object.');
throw new TypeError('options.audio, when provided, must be an object.');
}
if (options.audio?.discard !== undefined && typeof options.audio.discard !== 'boolean') {
throw new TypeError('options.audio.discard, when provided, must be a boolean.');
@@ -470,7 +481,10 @@ export class Conversion {
}
const firstTimestamp = await track.getFirstTimestamp();
const needsTranscode = !!this._options.video?.forceTranscode || this._startTimestamp > 0 || firstTimestamp < 0;
const needsTranscode = !!this._options.video?.forceTranscode
|| this._startTimestamp > 0
|| firstTimestamp < 0
|| !!this._options.video?.frameRate;
const needsRerender = width !== originalWidth
|| height !== originalHeight
|| (totalRotation !== 0 && !outputSupportsRotation);
@@ -498,7 +512,7 @@ export class Conversion {
? await sink.getPacket(this._endTimestamp, { metadataOnly: true }) ?? undefined
: undefined;
for await (const packet of sink.packets(undefined, endPacket)) {
for await (const packet of sink.packets(undefined, endPacket, { verifyKeyPackets: true })) {
if (this._synchronizer.shouldWait(track.id, packet.timestamp)) {
await this._synchronizer.wait(packet.timestamp);
}
@@ -547,10 +561,10 @@ export class Conversion {
onEncodedPacket: sample => this._reportProgress(track.id, sample.timestamp + sample.duration),
};
if (needsRerender) {
const source = new VideoSampleSource(encodingConfig);
videoSource = source;
const source = new VideoSampleSource(encodingConfig);
videoSource = source;
if (needsRerender) {
this._trackPromises.push((async () => {
await this._started;
@@ -562,6 +576,27 @@ export class Conversion {
poolSize: 1,
});
const iterator = sink.canvases(this._startTimestamp, this._endTimestamp);
const frameRate = this._options.video?.frameRate;
let lastCanvas: HTMLCanvasElement | OffscreenCanvas | null = null;
let lastCanvasTimestamp: number | null = null;
let lastCanvasEndTimestamp: number | null = null;
/** Repeats the last sample to pad out the time until the specified timestamp. */
const padFrames = async (until: number) => {
assert(lastCanvas);
assert(frameRate !== undefined);
const frameDifference = Math.round((until - lastCanvasTimestamp!) * frameRate);
for (let i = 1; i < frameDifference; i++) {
const sample = new VideoSample(lastCanvas, {
timestamp: lastCanvasTimestamp! + i / frameRate,
duration: 1 / frameRate,
});
await source.add(sample);
}
};
for await (const { canvas, timestamp, duration } of iterator) {
if (this._synchronizer.shouldWait(track.id, timestamp)) {
@@ -572,37 +607,134 @@ export class Conversion {
return;
}
let adjustedSampleTimestamp = Math.max(timestamp - this._startTimestamp, 0);
lastCanvasEndTimestamp = timestamp + duration;
if (frameRate !== undefined) {
// Logic for skipping/repeating frames when a frame rate is set
const alignedTimestamp = Math.floor(adjustedSampleTimestamp * frameRate) / frameRate;
if (lastCanvas !== null) {
if (alignedTimestamp <= lastCanvasTimestamp!) {
lastCanvas = canvas;
lastCanvasTimestamp = alignedTimestamp;
// Skip this sample, since we already added one for this frame
continue;
} else {
// Check if we may need to repeat the previous frame
await padFrames(alignedTimestamp);
}
}
adjustedSampleTimestamp = alignedTimestamp;
}
const sample = new VideoSample(canvas, {
timestamp: Math.max(timestamp - this._startTimestamp, 0),
duration,
timestamp: adjustedSampleTimestamp,
duration: frameRate !== undefined ? 1 / frameRate : duration,
});
await source.add(sample);
sample.close();
if (frameRate !== undefined) {
lastCanvas = canvas;
lastCanvasTimestamp = adjustedSampleTimestamp;
} else {
sample.close();
}
}
if (lastCanvas) {
assert(lastCanvasEndTimestamp !== null);
assert(frameRate !== undefined);
// If necessary, pad until the end timestamp of the last sample
await padFrames(Math.floor(lastCanvasEndTimestamp * frameRate) / frameRate);
}
source.close();
this._synchronizer.closeTrack(track.id);
})());
} else {
const source = new VideoSampleSource(encodingConfig);
videoSource = source;
this._trackPromises.push((async () => {
await this._started;
const sink = new VideoSampleSink(track);
const frameRate = this._options.video?.frameRate;
let lastSample: VideoSample | null = null;
let lastSampleTimestamp: number | null = null;
let lastSampleEndTimestamp: number | null = null;
/** Repeats the last sample to pad out the time until the specified timestamp. */
const padFrames = async (until: number) => {
assert(lastSample);
assert(frameRate !== undefined);
const frameDifference = Math.round((until - lastSampleTimestamp!) * frameRate);
for (let i = 1; i < frameDifference; i++) {
lastSample.setTimestamp(lastSampleTimestamp! + i / frameRate);
lastSample.setDuration(1 / frameRate);
await source.add(lastSample);
}
lastSample.close();
};
for await (const sample of sink.samples(this._startTimestamp, this._endTimestamp)) {
if (this._synchronizer.shouldWait(track.id, sample.timestamp)) {
await this._synchronizer.wait(sample.timestamp);
}
sample.setTimestamp(Math.max(sample.timestamp - this._startTimestamp, 0));
if (this._canceled) {
lastSample?.close();
return;
}
let adjustedSampleTimestamp = Math.max(sample.timestamp - this._startTimestamp, 0);
lastSampleEndTimestamp = sample.timestamp + sample.duration;
if (frameRate !== undefined) {
// Logic for skipping/repeating frames when a frame rate is set
const alignedTimestamp = Math.floor(adjustedSampleTimestamp * frameRate) / frameRate;
if (lastSample !== null) {
if (alignedTimestamp <= lastSampleTimestamp!) {
lastSample.close();
lastSample = sample;
lastSampleTimestamp = alignedTimestamp;
// Skip this sample, since we already added one for this frame
continue;
} else {
// Check if we may need to repeat the previous frame
await padFrames(alignedTimestamp);
}
}
adjustedSampleTimestamp = alignedTimestamp;
sample.setDuration(1 / frameRate);
}
sample.setTimestamp(adjustedSampleTimestamp);
await source.add(sample);
sample.close();
if (frameRate !== undefined) {
lastSample = sample;
lastSampleTimestamp = adjustedSampleTimestamp;
} else {
sample.close();
}
}
if (lastSample) {
assert(lastSampleEndTimestamp !== null);
assert(frameRate !== undefined);
// If necessary, pad until the end timestamp of the last sample
await padFrames(Math.floor(lastSampleEndTimestamp * frameRate) / frameRate);
}
source.close();
@@ -612,6 +744,7 @@ export class Conversion {
}
this.output.addVideoTrack(videoSource, {
frameRate: this._options.video?.frameRate,
languageCode: track.languageCode,
rotation: needsRerender ? 0 : totalRotation, // Rerendering will bake the rotation into the output
});
+3
View File
@@ -6,6 +6,9 @@
* file, You can obtain one at https://mozilla.org/MPL/2.0/.
*/
/// <reference types="dom-mediacapture-transform" preserve="true" />
/// <reference types="dom-webcodecs" preserve="true" />
export {
Output,
OutputOptions,
+6 -4
View File
@@ -13,6 +13,7 @@ import { IsobmffReader } from './isobmff/isobmff-reader';
import { EBMLId, EBMLReader } from './matroska/ebml';
import { MatroskaDemuxer } from './matroska/matroska-demuxer';
import { Mp3Demuxer } from './mp3/mp3-demuxer';
import { FRAME_HEADER_SIZE } from './mp3/mp3-misc';
import { Mp3Reader } from './mp3/mp3-reader';
import { OggDemuxer } from './ogg/ogg-demuxer';
import { OggReader } from './ogg/ogg-reader';
@@ -245,9 +246,10 @@ export class Mp3InputFormat extends InputFormat {
}
// Fine, we found one frame header, but we're still not entirely sure this is MP3. Let's check if we can find
// another header nearby:
// another header right after it:
mp3Reader.pos = firstHeader.startPos + firstHeader.totalSize;
const secondHeader = mp3Reader.readNextFrameHeader(Math.min(framesStartPos + 4096, sourceSize));
await mp3Reader.reader.loadRange(mp3Reader.pos, mp3Reader.pos + FRAME_HEADER_SIZE);
const secondHeader = mp3Reader.readNextFrameHeader(mp3Reader.pos + FRAME_HEADER_SIZE);
if (!secondHeader) {
return false;
}
@@ -257,7 +259,7 @@ export class Mp3InputFormat extends InputFormat {
return false;
}
// We have found two matching MP3 frames, a strong indicator that this is an MP3 file
// We have found two matching consecutive MP3 frames, a strong indicator that this is an MP3 file
return true;
}
@@ -289,7 +291,7 @@ export class WaveInputFormat extends InputFormat {
const riffReader = new RiffReader(input._mainReader);
const riffType = riffReader.readAscii(4);
if (riffType !== 'RIFF' && riffType !== 'RIFX') {
if (riffType !== 'RIFF' && riffType !== 'RIFX' && riffType !== 'RF64') {
return false;
}
+42 -3
View File
@@ -7,11 +7,12 @@
*/
import { AudioCodec, MediaCodec, VideoCodec } from './codec';
import { determineVideoPacketType } from './codec-data';
import { customAudioDecoders, customVideoDecoders } from './custom-coder';
import { EncodedPacketSink, PacketRetrievalOptions } from './media-sink';
import { assert, Rotation } from './misc';
import { TrackType } from './output';
import { EncodedPacket } from './packet';
import { EncodedPacket, PacketType } from './packet';
/**
* Contains aggregate statistics about the encoded packets of a track.
@@ -62,6 +63,11 @@ export abstract class InputTrack {
abstract getCodecParameterString(): Promise<string | null>;
/** Checks if this track's packets can be decoded by the browser. */
abstract canDecode(): Promise<boolean>;
/**
* For a given packet of this track, this method determines the actual type of this packet (key/delta) by looking
* into its bitstream. Returns null if the type couldn't be determined.
*/
abstract determinePacketType(packet: EncodedPacket): Promise<PacketType | null>;
/** Returns true iff this track is a video track. */
isVideoTrack(): this is InputVideoTrack {
@@ -222,7 +228,10 @@ export class InputVideoTrack extends InputTrack {
|| (colorSpace.matrix as string) === 'bt2020-ncl';
}
/** Returns the decoder configuration for decoding the track's packets using a VideoDecoder. */
/**
* Returns the decoder configuration for decoding the track's packets using a VideoDecoder. Returns null if the
* track's codec is unknown.
*/
getDecoderConfig() {
return this._backing.getDecoderConfig();
}
@@ -257,6 +266,21 @@ export class InputVideoTrack extends InputTrack {
return false;
}
}
async determinePacketType(packet: EncodedPacket): Promise<PacketType | null> {
if (!(packet instanceof EncodedPacket)) {
throw new TypeError('packet must be an EncodedPacket.');
}
if (packet.isMetadataOnly) {
throw new TypeError('packet must not be metadata-only to determine its type.');
}
if (this.codec === null) {
return null;
}
return determineVideoPacketType(this, packet);
}
}
export interface InputAudioTrackBacking extends InputTrackBacking {
@@ -299,7 +323,10 @@ export class InputAudioTrack extends InputTrack {
return this._backing.getSampleRate();
}
/** Returns the decoder configuration for decoding the track's packets using an AudioDecoder. */
/**
* Returns the decoder configuration for decoding the track's packets using an AudioDecoder. Returns null if the
* track's codec is unknown.
*/
getDecoderConfig() {
return this._backing.getDecoderConfig();
}
@@ -338,4 +365,16 @@ export class InputAudioTrack extends InputTrack {
return false;
}
}
async determinePacketType(packet: EncodedPacket): Promise<PacketType | null> {
if (!(packet instanceof EncodedPacket)) {
throw new TypeError('packet must be an EncodedPacket.');
}
if (this.codec === null) {
return null;
}
return 'key'; // No audio codec with delta packets
}
}
+280 -216
View File
@@ -54,6 +54,7 @@ import {
isIso639Dash2LanguageCode,
roundToMultiple,
normalizeRotation,
Bitstream,
} from '../misc';
import { EncodedPacket, PLACEHOLDER_DATA } from '../packet';
import { Reader } from '../reader';
@@ -74,6 +75,7 @@ type InternalTrack = {
fragmentLookupTable: FragmentLookupTableEntry[] | null;
currentFragmentState: FragmentTrackState | null;
fragments: Fragment[];
fragmentsWithKeyFrame: Fragment[];
/** The segment durations of all edit list entries leading up to the main one (from which the offset is taken.) */
editListPreviousSegmentDurations: number;
/** The media time offset of the main edit list entry (with media time !== -1) */
@@ -167,6 +169,7 @@ type FragmentTrackState = {
type FragmentTrackData = {
startTimestamp: number;
endTimestamp: number;
firstKeyFrameTimestamp: number | null;
samples: FragmentTrackSample[];
presentationTimestamps: {
presentationTimestamp: number;
@@ -607,6 +610,7 @@ export class IsobmffDemuxer extends Demuxer {
fragmentLookupTable: null,
currentFragmentState: null,
fragments: [],
fragmentsWithKeyFrame: [],
editListPreviousSegmentDurations: 0,
editListOffset: 0,
} satisfies InternalTrack as InternalTrack;
@@ -701,7 +705,10 @@ export class IsobmffDemuxer extends Demuxer {
}
if (relevantEntryFound) {
throw new Error('Unsupported edit list: multiple edits are not supported.');
console.warn(
'Unsupported edit list: multiple edits are not currently supported. Only using first edit.',
);
break;
}
if (mediaTime === -1) {
@@ -710,7 +717,8 @@ export class IsobmffDemuxer extends Demuxer {
}
if (mediaRate !== 1) {
throw new Error('Unsupported edit list: media rate must be 1.');
console.warn('Unsupported edit list entry: media rate must be 1.');
break;
}
track.editListPreviousSegmentDurations = previousSegmentDurations;
@@ -942,7 +950,8 @@ export class IsobmffDemuxer extends Demuxer {
} else if (sampleSize === 16) {
track.info.codec = 'pcm-s16be';
} else {
throw new Error(`Unsupported sample size ${sampleSize} for codec 'twos'.`);
console.warn(`Unsupported sample size ${sampleSize} for codec 'twos'.`);
track.info.codec = null;
}
} else if (lowercaseBoxName === 'sowt') {
if (sampleSize === 8) {
@@ -950,7 +959,8 @@ export class IsobmffDemuxer extends Demuxer {
} else if (sampleSize === 16) {
track.info.codec = 'pcm-s16';
} else {
throw new Error(`Unsupported sample size ${sampleSize} for codec 'sowt'.`);
console.warn(`Unsupported sample size ${sampleSize} for codec 'sowt'.`);
track.info.codec = null;
}
} else if (lowercaseBoxName === 'raw ') {
track.info.codec = 'pcm-u8';
@@ -1193,7 +1203,8 @@ export class IsobmffDemuxer extends Demuxer {
} else if (pcmSampleSize === 32) {
track.info.codec = 'pcm-s32';
} else {
throw new Error(`Invalid ipcm sample size ${pcmSampleSize}.`);
console.warn(`Invalid ipcm sample size ${pcmSampleSize}.`);
track.info.codec = null;
}
} else {
if (pcmSampleSize === 16) {
@@ -1203,7 +1214,8 @@ export class IsobmffDemuxer extends Demuxer {
} else if (pcmSampleSize === 32) {
track.info.codec = 'pcm-s32be';
} else {
throw new Error(`Invalid ipcm sample size ${pcmSampleSize}.`);
console.warn(`Invalid ipcm sample size ${pcmSampleSize}.`);
track.info.codec = null;
}
}
} else if (track.info.codec === 'pcm-f32be') {
@@ -1215,7 +1227,8 @@ export class IsobmffDemuxer extends Demuxer {
} else if (pcmSampleSize === 64) {
track.info.codec = 'pcm-f64';
} else {
throw new Error(`Invalid fpcm sample size ${pcmSampleSize}.`);
console.warn(`Invalid fpcm sample size ${pcmSampleSize}.`);
track.info.codec = null;
}
} else {
if (pcmSampleSize === 32) {
@@ -1223,7 +1236,8 @@ export class IsobmffDemuxer extends Demuxer {
} else if (pcmSampleSize === 64) {
track.info.codec = 'pcm-f64be';
} else {
throw new Error(`Invalid fpcm sample size ${pcmSampleSize}.`);
console.warn(`Invalid fpcm sample size ${pcmSampleSize}.`);
track.info.codec = null;
}
}
}
@@ -1405,8 +1419,27 @@ export class IsobmffDemuxer extends Demuxer {
}; break;
case 'stz2': {
throw new Error('Unsupported.');
};
const track = this.currentTrack;
assert(track);
if (!track.sampleTable) {
break;
}
this.metadataReader.pos += 4; // Version + flags
this.metadataReader.pos += 3; // Reserved
const fieldSize = this.metadataReader.readU8(); // in bits
const sampleCount = this.metadataReader.readU32();
const bytes = this.metadataReader.readBytes(Math.ceil(sampleCount * fieldSize / 8));
const bitstream = new Bitstream(bytes);
for (let i = 0; i < sampleCount; i++) {
const sampleSize = bitstream.readBits(fieldSize);
track.sampleTable.sampleSizes.push(sampleSize);
}
}; break;
case 'stss': {
const track = this.currentTrack;
@@ -1635,6 +1668,16 @@ export class IsobmffDemuxer extends Demuxer {
);
this.currentTrack.fragments.splice(insertionIndex + 1, 0, this.currentFragment);
const hasKeyFrame = trackData.firstKeyFrameTimestamp !== null;
if (hasKeyFrame) {
const insertionIndex = binarySearchLessOrEqual(
this.currentTrack.fragmentsWithKeyFrame,
this.currentFragment.moofOffset,
x => x.moofOffset,
);
this.currentTrack.fragmentsWithKeyFrame.splice(insertionIndex + 1, 0, this.currentFragment);
}
const { currentFragmentState } = this.currentTrack;
assert(currentFragmentState);
@@ -1733,7 +1776,8 @@ export class IsobmffDemuxer extends Demuxer {
assert(track.currentFragmentState);
if (this.currentFragment.trackData.has(track.id)) {
throw new Error('Can\'t have two trun boxes for the same track in one fragment.');
console.warn('Can\'t have two trun boxes for the same track in one fragment. Ignoring...');
break;
}
const version = this.metadataReader.readU8();
@@ -1770,6 +1814,7 @@ export class IsobmffDemuxer extends Demuxer {
const trackData: FragmentTrackData = {
startTimestamp: 0,
endTimestamp: 0,
firstKeyFrameTimestamp: null,
samples: [],
presentationTimestamps: [],
startTimestampIsFinal: false,
@@ -1831,13 +1876,19 @@ export class IsobmffDemuxer extends Demuxer {
.map((x, i) => ({ presentationTimestamp: x.presentationTimestamp, sampleIndex: i }))
.sort((a, b) => a.presentationTimestamp - b.presentationTimestamp);
// Update sample durations based on presentation order
for (let i = 0; i < trackData.presentationTimestamps.length - 1; i++) {
const current = trackData.presentationTimestamps[i]!;
const next = trackData.presentationTimestamps[i + 1]!;
for (let i = 0; i < trackData.presentationTimestamps.length; i++) {
const currentEntry = trackData.presentationTimestamps[i]!;
const currentSample = trackData.samples[currentEntry.sampleIndex]!;
const duration = next.presentationTimestamp - current.presentationTimestamp;
trackData.samples[current.sampleIndex]!.duration = duration;
if (trackData.firstKeyFrameTimestamp === null && currentSample.isKeyFrame) {
trackData.firstKeyFrameTimestamp = currentSample.presentationTimestamp;
}
if (i < trackData.presentationTimestamps.length - 1) {
// Update sample durations based on presentation order
const nextEntry = trackData.presentationTimestamps[i + 1]!;
currentSample.duration = nextEntry.presentationTimestamp - currentEntry.presentationTimestamp;
}
}
const firstSample = trackData.samples[trackData.presentationTimestamps[0]!.sampleIndex]!;
@@ -1890,44 +1941,46 @@ abstract class IsobmffTrackBacking implements InputTrackBacking {
}
async getFirstPacket(options: PacketRetrievalOptions) {
if (this.internalTrack.demuxer.isFragmented) {
return this.performFragmentedLookup(
() => {
const startFragment = this.internalTrack.demuxer.fragments[0] ?? null;
if (startFragment?.isKnownToBeFirstFragment) {
// Walk from the very first fragment in the file until we find one with our track in it
let currentFragment: Fragment | null = startFragment;
while (currentFragment) {
const trackData = currentFragment.trackData.get(this.internalTrack.id);
if (trackData) {
return {
fragmentIndex: binarySearchExact(
this.internalTrack.fragments,
currentFragment.moofOffset,
x => x.moofOffset,
),
sampleIndex: 0,
correctSampleFound: true,
};
}
currentFragment = currentFragment.nextFragment;
}
}
return {
fragmentIndex: -1,
sampleIndex: -1,
correctSampleFound: false,
};
},
-Infinity, // Use -Infinity as a search timestamp to avoid using the lookup entries
Infinity,
options,
);
const regularPacket = await this.fetchPacketForSampleIndex(0, options);
if (regularPacket || !this.internalTrack.demuxer.isFragmented) {
// If there's a non-fragmented packet, always prefer that
return regularPacket;
}
return this.fetchPacketForSampleIndex(0, options);
return this.performFragmentedLookup(
() => {
const startFragment = this.internalTrack.demuxer.fragments[0] ?? null;
if (startFragment?.isKnownToBeFirstFragment) {
// Walk from the very first fragment in the file until we find one with our track in it
let currentFragment: Fragment | null = startFragment;
while (currentFragment) {
const trackData = currentFragment.trackData.get(this.internalTrack.id);
if (trackData) {
return {
fragmentIndex: binarySearchExact(
this.internalTrack.fragments,
currentFragment.moofOffset,
x => x.moofOffset,
),
sampleIndex: 0,
correctSampleFound: true,
};
}
currentFragment = currentFragment.nextFragment;
}
}
return {
fragmentIndex: -1,
sampleIndex: -1,
correctSampleFound: false,
};
},
-Infinity, // Use -Infinity as a search timestamp to avoid using the lookup entries
Infinity,
options,
);
}
private mapTimestampIntoTimescale(timestamp: number) {
@@ -1940,187 +1993,186 @@ abstract class IsobmffTrackBacking implements InputTrackBacking {
async getPacket(timestamp: number, options: PacketRetrievalOptions) {
const timestampInTimescale = this.mapTimestampIntoTimescale(timestamp);
if (this.internalTrack.demuxer.isFragmented) {
return this.performFragmentedLookup(
() => this.findSampleInFragmentsForTimestamp(timestampInTimescale),
timestampInTimescale,
timestampInTimescale,
options,
);
} else {
const sampleTable = this.internalTrack.demuxer.getSampleTableForTrack(this.internalTrack);
const sampleIndex = getSampleIndexForTimestamp(sampleTable, timestampInTimescale);
return this.fetchPacketForSampleIndex(sampleIndex, options);
const sampleTable = this.internalTrack.demuxer.getSampleTableForTrack(this.internalTrack);
const sampleIndex = getSampleIndexForTimestamp(sampleTable, timestampInTimescale);
const regularPacket = await this.fetchPacketForSampleIndex(sampleIndex, options);
if (!sampleTableIsEmpty(sampleTable) || !this.internalTrack.demuxer.isFragmented) {
// Prefer the non-fragmented packet
return regularPacket;
}
return this.performFragmentedLookup(
() => this.findSampleInFragmentsForTimestamp(timestampInTimescale),
timestampInTimescale,
timestampInTimescale,
options,
);
}
async getNextPacket(packet: EncodedPacket, options: PacketRetrievalOptions) {
if (this.internalTrack.demuxer.isFragmented) {
const locationInFragment = this.packetToFragmentLocation.get(packet);
if (locationInFragment === undefined) {
throw new Error('Packet was not created from this track.');
}
const regularSampleIndex = this.packetToSampleIndex.get(packet);
const trackData = locationInFragment.fragment.trackData.get(this.internalTrack.id)!;
const fragmentSample = trackData.samples[locationInFragment.sampleIndex]!;
const fragmentIndex = binarySearchExact(
this.internalTrack.fragments,
locationInFragment.fragment.moofOffset,
x => x.moofOffset,
);
assert(fragmentIndex !== -1);
return this.performFragmentedLookup(
() => {
if (locationInFragment.sampleIndex + 1 < trackData.samples.length) {
// We can simply take the next sample in the fragment
return {
fragmentIndex,
sampleIndex: locationInFragment.sampleIndex + 1,
correctSampleFound: true,
};
} else {
// Walk the list of fragments until we find the next fragment for this track
let currentFragment = locationInFragment.fragment;
while (currentFragment.nextFragment) {
currentFragment = currentFragment.nextFragment;
const trackData = currentFragment.trackData.get(this.internalTrack.id);
if (trackData) {
const fragmentIndex = binarySearchExact(
this.internalTrack.fragments,
currentFragment.moofOffset,
x => x.moofOffset,
);
assert(fragmentIndex !== -1);
return {
fragmentIndex,
sampleIndex: 0,
correctSampleFound: true,
};
}
}
return {
fragmentIndex,
sampleIndex: -1,
correctSampleFound: false,
};
}
},
fragmentSample.presentationTimestamp,
Infinity,
options,
);
if (regularSampleIndex !== undefined) {
// Prefer the non-fragmented packet
return this.fetchPacketForSampleIndex(regularSampleIndex + 1, options);
}
const sampleIndex = this.packetToSampleIndex.get(packet);
if (sampleIndex === undefined) {
const locationInFragment = this.packetToFragmentLocation.get(packet);
if (locationInFragment === undefined) {
throw new Error('Packet was not created from this track.');
}
return this.fetchPacketForSampleIndex(sampleIndex + 1, options);
const trackData = locationInFragment.fragment.trackData.get(this.internalTrack.id)!;
const fragmentIndex = binarySearchExact(
this.internalTrack.fragments,
locationInFragment.fragment.moofOffset,
x => x.moofOffset,
);
assert(fragmentIndex !== -1);
return this.performFragmentedLookup(
() => {
if (locationInFragment.sampleIndex + 1 < trackData.samples.length) {
// We can simply take the next sample in the fragment
return {
fragmentIndex,
sampleIndex: locationInFragment.sampleIndex + 1,
correctSampleFound: true,
};
} else {
// Walk the list of fragments until we find the next fragment for this track
let currentFragment = locationInFragment.fragment;
while (currentFragment.nextFragment) {
currentFragment = currentFragment.nextFragment;
const trackData = currentFragment.trackData.get(this.internalTrack.id);
if (trackData) {
const fragmentIndex = binarySearchExact(
this.internalTrack.fragments,
currentFragment.moofOffset,
x => x.moofOffset,
);
assert(fragmentIndex !== -1);
return {
fragmentIndex,
sampleIndex: 0,
correctSampleFound: true,
};
}
}
return {
fragmentIndex,
sampleIndex: -1,
correctSampleFound: false,
};
}
},
-Infinity, // Use -Infinity as a search timestamp to avoid using the lookup entries
Infinity,
options,
);
}
async getKeyPacket(timestamp: number, options: PacketRetrievalOptions) {
const timestampInTimescale = this.mapTimestampIntoTimescale(timestamp);
if (this.internalTrack.demuxer.isFragmented) {
return this.performFragmentedLookup(
() => this.findKeySampleInFragmentsForTimestamp(timestampInTimescale),
timestampInTimescale,
timestampInTimescale,
options,
);
}
const sampleTable = this.internalTrack.demuxer.getSampleTableForTrack(this.internalTrack);
const sampleIndex = getSampleIndexForTimestamp(sampleTable, timestampInTimescale);
const keyFrameSampleIndex = sampleIndex === -1
? -1
: getRelevantKeyframeIndexForSample(sampleTable, sampleIndex);
return this.fetchPacketForSampleIndex(keyFrameSampleIndex, options);
const regularPacket = await this.fetchPacketForSampleIndex(keyFrameSampleIndex, options);
if (!sampleTableIsEmpty(sampleTable) || !this.internalTrack.demuxer.isFragmented) {
// Prefer the non-fragmented packet
return regularPacket;
}
return this.performFragmentedLookup(
() => this.findKeySampleInFragmentsForTimestamp(timestampInTimescale),
timestampInTimescale,
timestampInTimescale,
options,
);
}
async getNextKeyPacket(packet: EncodedPacket, options: PacketRetrievalOptions) {
if (this.internalTrack.demuxer.isFragmented) {
const locationInFragment = this.packetToFragmentLocation.get(packet);
if (locationInFragment === undefined) {
throw new Error('Packet was not created from this track.');
}
const trackData = locationInFragment.fragment.trackData.get(this.internalTrack.id)!;
const fragmentSample = trackData.samples[locationInFragment.sampleIndex]!;
const fragmentIndex = binarySearchExact(
this.internalTrack.fragments,
locationInFragment.fragment.moofOffset,
x => x.moofOffset,
);
assert(fragmentIndex !== -1);
return this.performFragmentedLookup(
() => {
const nextKeyFrameIndex = trackData.samples.findIndex(
(x, i) => x.isKeyFrame && i > locationInFragment.sampleIndex,
);
if (nextKeyFrameIndex !== -1) {
// We can simply take the next key frame in the fragment
return {
fragmentIndex,
sampleIndex: nextKeyFrameIndex,
correctSampleFound: true,
};
} else {
// Walk the list of fragments until we find the next fragment for this track
let currentFragment = locationInFragment.fragment;
while (currentFragment.nextFragment) {
currentFragment = currentFragment.nextFragment;
const trackData = currentFragment.trackData.get(this.internalTrack.id);
if (trackData) {
const fragmentIndex = binarySearchExact(
this.internalTrack.fragments,
currentFragment.moofOffset,
x => x.moofOffset,
);
assert(fragmentIndex !== -1);
const keyFrameIndex = trackData.samples.findIndex(x => x.isKeyFrame);
if (keyFrameIndex === -1) {
throw new Error('Not supported: Fragment does not contain key sample.');
}
return {
fragmentIndex,
sampleIndex: keyFrameIndex,
correctSampleFound: true,
};
}
}
return {
fragmentIndex,
sampleIndex: -1,
correctSampleFound: false,
};
}
},
fragmentSample.presentationTimestamp,
Infinity,
options,
);
const regularSampleIndex = this.packetToSampleIndex.get(packet);
if (regularSampleIndex !== undefined) {
// Prefer the non-fragmented packet
const sampleTable = this.internalTrack.demuxer.getSampleTableForTrack(this.internalTrack);
const nextKeyFrameSampleIndex = getNextKeyframeIndexForSample(sampleTable, regularSampleIndex);
return this.fetchPacketForSampleIndex(nextKeyFrameSampleIndex, options);
}
const sampleIndex = this.packetToSampleIndex.get(packet);
if (sampleIndex === undefined) {
const locationInFragment = this.packetToFragmentLocation.get(packet);
if (locationInFragment === undefined) {
throw new Error('Packet was not created from this track.');
}
const sampleTable = this.internalTrack.demuxer.getSampleTableForTrack(this.internalTrack);
const nextKeyFrameSampleIndex = getNextKeyframeIndexForSample(sampleTable, sampleIndex);
return this.fetchPacketForSampleIndex(nextKeyFrameSampleIndex, options);
const trackData = locationInFragment.fragment.trackData.get(this.internalTrack.id)!;
const fragmentIndex = binarySearchExact(
this.internalTrack.fragments,
locationInFragment.fragment.moofOffset,
x => x.moofOffset,
);
assert(fragmentIndex !== -1);
return this.performFragmentedLookup(
() => {
const nextKeyFrameIndex = trackData.samples.findIndex(
(x, i) => x.isKeyFrame && i > locationInFragment.sampleIndex,
);
if (nextKeyFrameIndex !== -1) {
// We can simply take the next key frame in the fragment
return {
fragmentIndex,
sampleIndex: nextKeyFrameIndex,
correctSampleFound: true,
};
} else {
// Walk the list of fragments until we find the next fragment for this track with a key frame
let currentFragment = locationInFragment.fragment;
while (currentFragment.nextFragment) {
currentFragment = currentFragment.nextFragment;
const trackData = currentFragment.trackData.get(this.internalTrack.id);
if (trackData && trackData.firstKeyFrameTimestamp !== null) {
const fragmentIndex = binarySearchExact(
this.internalTrack.fragments,
currentFragment.moofOffset,
x => x.moofOffset,
);
assert(fragmentIndex !== -1);
const keyFrameIndex = trackData.samples.findIndex(x => x.isKeyFrame);
assert(keyFrameIndex !== -1); // There must be one
return {
fragmentIndex,
sampleIndex: keyFrameIndex,
correctSampleFound: true,
};
}
}
return {
fragmentIndex,
sampleIndex: -1,
correctSampleFound: false,
};
}
},
-Infinity, // Use -Infinity as a search timestamp to avoid using the lookup entries
Infinity,
options,
);
}
private async fetchPacketForSampleIndex(sampleIndex: number, options: PacketRetrievalOptions) {
@@ -2231,26 +2283,34 @@ abstract class IsobmffTrackBacking implements InputTrackBacking {
}
private findKeySampleInFragmentsForTimestamp(timestampInTimescale: number) {
const fragmentIndex = binarySearchLessOrEqual(
const indexInKeyFrameFragments = binarySearchLessOrEqual(
// This array is technically not sorted by start timestamp, but for any reasonable file, it basically is.
this.internalTrack.fragments,
this.internalTrack.fragmentsWithKeyFrame,
timestampInTimescale,
x => x.trackData.get(this.internalTrack.id)!.startTimestamp,
);
let fragmentIndex = -1;
let sampleIndex = -1;
let correctSampleFound = false;
if (fragmentIndex !== -1) {
const fragment = this.internalTrack.fragments[fragmentIndex]!;
if (indexInKeyFrameFragments !== -1) {
const fragment = this.internalTrack.fragmentsWithKeyFrame[indexInKeyFrameFragments]!;
// Now, let's find the actual index of the fragment in the list of ALL fragments, not just key frame ones
fragmentIndex = binarySearchExact(
this.internalTrack.fragments,
fragment.moofOffset,
x => x.moofOffset,
);
assert(fragmentIndex !== -1);
const trackData = fragment.trackData.get(this.internalTrack.id)!;
const index = findLastIndex(trackData.presentationTimestamps, (x) => {
const sample = trackData.samples[x.sampleIndex]!;
return sample.isKeyFrame && x.presentationTimestamp <= timestampInTimescale;
});
if (index === -1) {
throw new Error('Not supported: Fragment does not begin with a key sample.');
}
assert(index !== -1); // It's a key frame fragment, so there must be a key frame
const entry = trackData.presentationTimestamps[index]!;
sampleIndex = entry.sampleIndex;
@@ -2644,3 +2704,7 @@ const extractRotationFromMatrix = (matrix: TransformationMatrix) => {
// Invert the rotation because matrices are post-multiplied in ISOBMFF
return -Math.atan2(sinTheta, cosTheta) * (180 / Math.PI);
};
const sampleTableIsEmpty = (sampleTable: SampleTable) => {
return sampleTable.sampleSizes.length === 0;
};
+13 -6
View File
@@ -32,6 +32,7 @@ import {
transformAnnexBToLengthPrefixed,
} from '../codec-data';
import { buildIsobmffMimeType } from './isobmff-misc';
import { MAX_BOX_HEADER_SIZE, MIN_BOX_HEADER_SIZE } from './isobmff-reader';
export const GLOBAL_TIMESCALE = 1000;
const TIMESTAMP_OFFSET = 2_082_844_800; // Seconds between Jan 1 1904 and Jan 1 1970
@@ -989,10 +990,7 @@ export class IsobmffMuxer extends Muxer {
const moofOffset = this.writer.getPos();
const mdatStartPos = moofOffset + this.boxWriter.measureBox(moofBox);
// Header with large size. We always reserve 16 bytes for it even if we don't end up using the large size.
const mdatHeaderSize = 16;
let currentPos = mdatStartPos + mdatHeaderSize;
let currentPos = mdatStartPos + MIN_BOX_HEADER_SIZE;
let fragmentStartTimestamp = Infinity;
for (const trackData of tracksInFragment) {
trackData.currentChunk!.offset = currentPos;
@@ -1006,6 +1004,15 @@ export class IsobmffMuxer extends Muxer {
}
const mdatSize = currentPos - mdatStartPos;
const needsLargeMdatSize = mdatSize >= 2 ** 32;
if (needsLargeMdatSize) {
// Shift all offsets by 8. Previously, all chunks were shifted assuming the large box size, but due to what
// I suspect is a bug in WebKit, it failed in Safari (when livestreaming with MSE, not for static playback).
for (const trackData of tracksInFragment) {
trackData.currentChunk!.offset! += MAX_BOX_HEADER_SIZE - MIN_BOX_HEADER_SIZE;
}
}
if (this.format._options.onMoof) {
this.writer.startTrackingWrites();
@@ -1025,11 +1032,11 @@ export class IsobmffMuxer extends Muxer {
this.writer.startTrackingWrites();
}
const mdatBox = mdat(mdatSize >= 2 ** 32);
const mdatBox = mdat(needsLargeMdatSize);
mdatBox.size = mdatSize;
this.boxWriter.writeBox(mdatBox);
this.writer.seek(mdatStartPos + mdatHeaderSize);
this.writer.seek(mdatStartPos + (needsLargeMdatSize ? MAX_BOX_HEADER_SIZE : MIN_BOX_HEADER_SIZE));
// Write sample data
for (const trackData of tracksInFragment) {
+39 -1
View File
@@ -499,7 +499,13 @@ export class EBMLReader {
const { view, offset } = this.reader.getViewAndOffset(this.pos, this.pos + length);
this.pos += length;
return String.fromCharCode(...new Uint8Array(view.buffer, offset, length));
// Actual string length might be shorter due to null terminators
let strLength = 0;
while (strLength < length && view.getUint8(offset + strLength) !== 0) {
strLength += 1;
}
return String.fromCharCode(...new Uint8Array(view.buffer, offset, strLength));
}
readElementId() {
@@ -588,6 +594,38 @@ export const CODEC_STRING_MAP: Partial<Record<MediaCodec, string>> = {
'webvtt': 'S_TEXT/WEBVTT',
};
export const readVarInt = (data: Uint8Array, offset: number) => {
if (offset >= data.length) {
throw new Error('Offset out of bounds.');
}
// Read the first byte to determine the width of the variable-length integer
const firstByte = data[offset]!;
// Find the position of VINT_MARKER, which determines the width
let width = 1;
let mask = 1 << 7;
while ((firstByte & mask) === 0 && width < 8) {
width++;
mask >>= 1;
}
if (offset + width > data.length) {
throw new Error('VarInt extends beyond data bounds.');
}
// First byte's value needs the marker bit cleared
let value = firstByte & (mask - 1);
// Read remaining bytes
for (let i = 1; i < width; i++) {
value *= 1 << 8;
value += data[offset + i]!;
}
return { value, width };
};
export function assertDefinedSize(size: number | null): asserts size is number {
if (size === null) {
throw new Error('Undefined element size is used in a place where it is not supported.');
+206 -25
View File
@@ -57,6 +57,7 @@ import {
LEVEL_0_AND_1_EBML_IDS,
MAX_HEADER_SIZE,
MIN_HEADER_SIZE,
readVarInt,
} from './ebml';
import { buildMatroskaMimeType } from './matroska-misc';
@@ -107,12 +108,20 @@ type ClusterTrackData = {
}[];
};
enum BlockLacing {
None,
Xiph,
FixedSize,
Ebml,
}
type ClusterBlock = {
timestamp: number;
duration: number;
isKeyFrame: boolean;
referencedTimestamps: number[];
data: Uint8Array;
lacing: BlockLacing;
};
type CuePoint = {
@@ -133,6 +142,7 @@ type InternalTrack = {
inputTrack: InputTrack | null;
codecId: string | null;
codecPrivate: Uint8Array | null;
defaultDuration: number | null;
languageCode: string;
info:
| null
@@ -220,13 +230,16 @@ export class MatroskaDemuxer extends Demuxer {
const fileSize = await this.input.source.getSize();
while (this.metadataReader.pos < fileSize - MIN_HEADER_SIZE) {
// Loop over all top-level elements in the file
while (this.metadataReader.pos <= fileSize - MIN_HEADER_SIZE) {
await this.metadataReader.reader.loadRange(
this.metadataReader.pos,
this.metadataReader.pos + MAX_HEADER_SIZE,
);
const { id, size } = this.metadataReader.readElementHeader();
const header = this.metadataReader.readElementHeader();
const id = header.id;
let size = header.size;
const startPos = this.metadataReader.pos;
if (id === EBMLId.EBML) {
@@ -242,6 +255,26 @@ export class MatroskaDemuxer extends Demuxer {
// and only segment
break;
}
} else if (id === EBMLId.Cluster) {
// Clusters are not a top-level element in Matroska, but some files contain a Segment whose size
// doesn't contain any of the clusters that follow it. In the case, we apply the following logic: if
// we find a top-level cluster, attribute it to the previous segment.
if (size === null) {
// Just in case this is one of those weird sizeless clusters, let's do our best and still try to
// determine its size.
const nextElementPos = await this.clusterReader.searchForNextElementId(
LEVEL_0_AND_1_EBML_IDS,
fileSize,
);
size = (nextElementPos ?? fileSize) - startPos;
}
const lastSegment = last(this.segments);
if (lastSegment) {
// Extend the previous segment's size
lastSegment.elementEndPos = startPos + size;
}
}
assertDefinedSize(size);
@@ -368,6 +401,13 @@ export class MatroskaDemuxer extends Demuxer {
this.readContiguousElements(this.metadataReader, size);
}
if (this.currentSegment.timestampScale === -1) {
// TimestampScale element is missing. Technically an invalid file, but let's default to the typical value,
// which is 1e6.
this.currentSegment.timestampScale = 1e6;
this.currentSegment.timestampFactor = 1e9 / 1e6;
}
// Put default tracks first
this.currentSegment.tracks.sort((a, b) => Number(b.isDefault) - Number(a.isDefault));
@@ -464,16 +504,20 @@ export class MatroskaDemuxer extends Demuxer {
this.readContiguousElements(this.clusterReader, size);
for (const [trackId, trackData] of cluster.trackData) {
let blockReferencesExist = false;
const track = segment.tracks.find(x => x.id === trackId) ?? null;
// This must hold, as track datas only get created if a block for that track is encountered
assert(trackData.blocks.length > 0);
let blockReferencesExist = false;
let hasLacedBlocks = false;
for (let i = 0; i < trackData.blocks.length; i++) {
const block = trackData.blocks[i]!;
block.timestamp += cluster.timestamp;
blockReferencesExist ||= block.referencedTimestamps.length > 0;
hasLacedBlocks ||= block.lacing !== BlockLacing.None;
}
if (blockReferencesExist) {
@@ -484,34 +528,47 @@ export class MatroskaDemuxer extends Demuxer {
.map((block, i) => ({ timestamp: block.timestamp, blockIndex: i }))
.sort((a, b) => a.timestamp - b.timestamp);
let hasKeyFrame = false;
for (let i = 0; i < trackData.presentationTimestamps.length; i++) {
const entry = trackData.presentationTimestamps[i]!;
const block = trackData.blocks[entry.blockIndex]!;
const currentEntry = trackData.presentationTimestamps[i]!;
const currentBlock = trackData.blocks[currentEntry.blockIndex]!;
if (block.isKeyFrame) {
hasKeyFrame = true;
if (trackData.firstKeyFrameTimestamp === null && block.isKeyFrame) {
trackData.firstKeyFrameTimestamp = block.timestamp;
}
if (trackData.firstKeyFrameTimestamp === null && currentBlock.isKeyFrame) {
trackData.firstKeyFrameTimestamp = currentBlock.timestamp;
}
if (i < trackData.presentationTimestamps.length - 1) {
// Update block durations based on presentation order
const nextEntry = trackData.presentationTimestamps[i + 1]!;
const nextBlock = trackData.blocks[nextEntry.blockIndex]!;
block.duration = nextBlock.timestamp - block.timestamp;
currentBlock.duration = nextEntry.timestamp - currentBlock.timestamp;
} else if (currentBlock.duration === 0) {
if (track?.defaultDuration != null) {
if (currentBlock.lacing === BlockLacing.None) {
currentBlock.duration = track.defaultDuration;
} else {
// Handled by the lace resolution code
}
}
}
}
if (hasLacedBlocks) {
// Perform lace resolution. Here, we expand each laced block into multiple blocks where each contains
// one frame of the lace. We do this after determining block timestamps so we can properly distribute
// the block's duration across the laced frames.
this.expandLacedBlocks(trackData.blocks, track);
// Recompute since blocks have changed
trackData.presentationTimestamps = trackData.blocks
.map((block, i) => ({ timestamp: block.timestamp, blockIndex: i }))
.sort((a, b) => a.timestamp - b.timestamp);
}
const firstBlock = trackData.blocks[trackData.presentationTimestamps[0]!.blockIndex]!;
const lastBlock = trackData.blocks[last(trackData.presentationTimestamps)!.blockIndex]!;
trackData.startTimestamp = firstBlock.timestamp;
trackData.endTimestamp = lastBlock.timestamp + lastBlock.duration;
const track = segment.tracks.find(x => x.id === trackId);
if (track) {
const insertionIndex = binarySearchLessOrEqual(
track.clusters,
@@ -520,6 +577,7 @@ export class MatroskaDemuxer extends Demuxer {
);
track.clusters.splice(insertionIndex + 1, 0, cluster);
const hasKeyFrame = trackData.firstKeyFrameTimestamp !== null;
if (hasKeyFrame) {
const insertionIndex = binarySearchLessOrEqual(
track.clustersWithKeyFrame,
@@ -559,10 +617,124 @@ export class MatroskaDemuxer extends Demuxer {
return trackData;
}
expandLacedBlocks(blocks: ClusterBlock[], track: InternalTrack | null) {
// https://www.matroska.org/technical/notes.html#block-lacing
for (let blockIndex = 0; blockIndex < blocks.length; blockIndex++) {
const originalBlock = blocks[blockIndex]!;
if (originalBlock.lacing === BlockLacing.None) {
continue;
}
const data = originalBlock.data;
let pos = 0;
const frameSizes: number[] = [];
const frameCount = data[pos]! + 1;
pos++;
switch (originalBlock.lacing) {
case BlockLacing.Xiph: {
let totalUsedSize = 0;
// Xiph lacing, just like in Ogg
for (let i = 0; i < frameCount - 1; i++) {
let frameSize = 0;
while (pos < data.length) {
const value = data[pos]!;
frameSize += value;
pos++;
if (value < 255) {
frameSizes.push(frameSize);
totalUsedSize += frameSize;
break;
}
}
}
// Compute the last frame's size from whatever's left
frameSizes.push(data.length - (pos + totalUsedSize));
}; break;
case BlockLacing.FixedSize: {
// Fixed size lacing: all frames have same size
const totalDataSize = data.length - 1; // Minus the frame count byte
const frameSize = Math.floor(totalDataSize / frameCount);
for (let i = 0; i < frameCount; i++) {
frameSizes.push(frameSize);
}
}; break;
case BlockLacing.Ebml: {
// EBML lacing: first size absolute, subsequent ones are coded as signed differences from the last
const firstResult = readVarInt(data, pos);
let currentSize = firstResult.value;
frameSizes.push(currentSize);
pos += firstResult.width;
let totalUsedSize = currentSize;
for (let i = 1; i < frameCount - 1; i++) {
const diffResult = readVarInt(data, pos);
const unsignedDiff = diffResult.value;
const bias = (1 << (diffResult.width * 7 - 1)) - 1; // Typo-corrected version of 2^((7*n)-1)^-1
const diff = unsignedDiff - bias;
currentSize += diff;
frameSizes.push(currentSize);
pos += diffResult.width;
totalUsedSize += currentSize;
}
// Compute the last frame's size from whatever's left
frameSizes.push(data.length - (pos + totalUsedSize));
}; break;
default: assert(false);
}
assert(frameSizes.length === frameCount);
blocks.splice(blockIndex, 1); // Remove the original block
let dataOffset = pos;
// Now, let's insert each frame as its own block
for (let i = 0; i < frameCount; i++) {
const frameSize = frameSizes[i]!;
const frameData = data.subarray(dataOffset, dataOffset + frameSize);
const blockDuration = originalBlock.duration || (frameCount * (track?.defaultDuration ?? 0));
// Distribute timestamps evenly across the block duration
const frameTimestamp = originalBlock.timestamp + (blockDuration * i / frameCount);
const frameDuration = blockDuration / frameCount;
blocks.splice(blockIndex + i, 0, {
timestamp: frameTimestamp,
duration: frameDuration,
isKeyFrame: originalBlock.isKeyFrame,
referencedTimestamps: originalBlock.referencedTimestamps,
data: frameData,
lacing: BlockLacing.None,
});
dataOffset += frameSize;
}
blockIndex += frameCount; // Skip the blocks we just added
blockIndex--;
}
}
readContiguousElements(reader: EBMLReader, totalSize: number) {
const startIndex = reader.pos;
while (reader.pos - startIndex < totalSize) {
while (reader.pos - startIndex <= totalSize - MIN_HEADER_SIZE) {
this.traverseElement(reader);
}
}
@@ -630,6 +802,7 @@ export class MatroskaDemuxer extends Demuxer {
inputTrack: null,
codecId: null,
codecPrivate: null,
defaultDuration: null,
languageCode: UNDETERMINED_LANGUAGE,
info: null,
};
@@ -791,6 +964,13 @@ export class MatroskaDemuxer extends Demuxer {
this.currentTrack.codecPrivate = reader.readBytes(size);
}; break;
case EBMLId.DefaultDuration: {
if (!this.currentTrack) break;
this.currentTrack.defaultDuration
= this.currentTrack.segment.timestampFactor * reader.readUnsignedInt(size) / 1e9;
}; break;
case EBMLId.Language: {
if (!this.currentTrack) break;
@@ -952,14 +1132,16 @@ export class MatroskaDemuxer extends Demuxer {
const flags = reader.readU8();
const isKeyFrame = !!(flags & 0x80);
const lacing = (flags >> 1) & 0x3 as BlockLacing; // If the block is laced, we'll expand it later
const trackData = this.getTrackDataInCluster(this.currentCluster, trackNumber);
trackData.blocks.push({
timestamp: relativeTimestamp, // We'll add the cluster's timestamp to this later
duration: 0,
duration: 0, // Will set later
isKeyFrame,
referencedTimestamps: [],
data: reader.readBytes(size - (reader.pos - dataStartPos)),
lacing,
});
}; break;
@@ -983,16 +1165,17 @@ export class MatroskaDemuxer extends Demuxer {
const trackNumber = reader.readVarInt();
const relativeTimestamp = reader.readS16();
// eslint-disable-next-line @typescript-eslint/no-unused-vars
const flags = reader.readU8();
const lacing = (flags >> 1) & 0x3 as BlockLacing; // If the block is laced, we'll expand it later
const trackData = this.getTrackDataInCluster(this.currentCluster, trackNumber);
this.currentBlock = {
timestamp: relativeTimestamp, // We'll add the cluster's timestamp to this later
duration: 0,
duration: 0, // Will set later
isKeyFrame: true,
referencedTimestamps: [],
data: reader.readBytes(size - (reader.pos - dataStartPos)),
lacing,
};
trackData.blocks.push(this.currentBlock);
}; break;
@@ -1115,7 +1298,6 @@ abstract class MatroskaTrackBacking implements InputTrackBacking {
}
const trackData = locationInCluster.cluster.trackData.get(this.internalTrack.id)!;
const block = trackData.blocks[locationInCluster.blockIndex]!;
const clusterIndex = binarySearchExact(
this.internalTrack.clusters,
@@ -1163,7 +1345,7 @@ abstract class MatroskaTrackBacking implements InputTrackBacking {
};
}
},
block.timestamp,
-Infinity, // Use -Infinity as a search timestamp to avoid using the cues
Infinity,
options,
);
@@ -1187,7 +1369,6 @@ abstract class MatroskaTrackBacking implements InputTrackBacking {
}
const trackData = locationInCluster.cluster.trackData.get(this.internalTrack.id)!;
const block = trackData.blocks[locationInCluster.blockIndex]!;
const clusterIndex = binarySearchExact(
this.internalTrack.clusters,
@@ -1210,7 +1391,7 @@ abstract class MatroskaTrackBacking implements InputTrackBacking {
correctBlockFound: true,
};
} else {
// Walk the list of clusters until we find the next cluster for this track
// Walk the list of clusters until we find the next cluster for this track with a key frame
let currentCluster = locationInCluster.cluster;
while (currentCluster.nextCluster) {
currentCluster = currentCluster.nextCluster;
@@ -1242,7 +1423,7 @@ abstract class MatroskaTrackBacking implements InputTrackBacking {
};
}
},
block.timestamp,
-Infinity, // Use -Infinity as a search timestamp to avoid using the cues
Infinity,
options,
);
@@ -1473,7 +1654,7 @@ abstract class MatroskaTrackBacking implements InputTrackBacking {
}
const endPos = dataStartPos + size;
if (endPos >= segment.elementEndPos - MIN_HEADER_SIZE) {
if (endPos > segment.elementEndPos - MIN_HEADER_SIZE) {
// No more elements fit in this segment
break;
} else {
+1 -2
View File
@@ -694,8 +694,7 @@ export class MatroskaMuxer extends Muxer {
const bitstream = new Bitstream(chunk.data);
// Check if it's a "superframe"
if (bitstream.readBits(2) !== 0b10) return;
bitstream.skipBits(2);
const profileLowBit = bitstream.readBits(1);
const profileHighBit = bitstream.readBits(1);
+89 -10
View File
@@ -39,6 +39,14 @@ export type PacketRetrievalOptions = {
* be loaded.
*/
metadataOnly?: boolean;
/**
* When set to true, key packets will be verified upon retrieval by looking into the packet's bitstream.
* If not enabled, the packet types will be determined solely by what's stored in the containing file and may be
* incorrect, potentially leading to decoder errors. Since determining a packet's actual type requires looking into
* its data, this option cannot be enabled together with `metadataOnly`.
*/
verifyKeyPackets?: boolean;
};
const validatePacketRetrievalOptions = (options: PacketRetrievalOptions) => {
@@ -48,6 +56,12 @@ const validatePacketRetrievalOptions = (options: PacketRetrievalOptions) => {
if (options.metadataOnly !== undefined && typeof options.metadataOnly !== 'boolean') {
throw new TypeError('options.metadataOnly, when defined, must be a boolean.');
}
if (options.verifyKeyPackets !== undefined && typeof options.verifyKeyPackets !== 'boolean') {
throw new TypeError('options.verifyKeyPackets, when defined, must be a boolean.');
}
if (options.verifyKeyPackets && options.metadataOnly) {
throw new TypeError('options.verifyKeyPackets and options.metadataOnly cannot be enabled together.');
}
};
const validateTimestamp = (timestamp: number) => {
@@ -56,6 +70,30 @@ const validateTimestamp = (timestamp: number) => {
}
};
const maybeFixPacketType = (
track: InputTrack,
promise: Promise<EncodedPacket | null>,
options: PacketRetrievalOptions,
) => {
if (options.verifyKeyPackets) {
return promise.then(async (packet) => {
if (!packet || packet.type === 'delta') {
return packet;
}
const determinedType = await track.determinePacketType(packet);
if (determinedType) {
// @ts-expect-error Technically readonly
packet.type = determinedType;
}
return packet;
});
} else {
return promise;
}
};
/**
* Sink for retrieving encoded packets from an input track.
* @public
@@ -78,7 +116,8 @@ export class EncodedPacketSink {
*/
getFirstPacket(options: PacketRetrievalOptions = {}) {
validatePacketRetrievalOptions(options);
return this._track._backing.getFirstPacket(options);
return maybeFixPacketType(this._track, this._track._backing.getFirstPacket(options), options);
}
/**
@@ -92,7 +131,8 @@ export class EncodedPacketSink {
getPacket(timestamp: number, options: PacketRetrievalOptions = {}) {
validateTimestamp(timestamp);
validatePacketRetrievalOptions(options);
return this._track._backing.getPacket(timestamp, options);
return maybeFixPacketType(this._track, this._track._backing.getPacket(timestamp, options), options);
}
/**
@@ -104,7 +144,8 @@ export class EncodedPacketSink {
throw new TypeError('packet must be an EncodedPacket.');
}
validatePacketRetrievalOptions(options);
return this._track._backing.getNextPacket(packet, options);
return maybeFixPacketType(this._track, this._track._backing.getNextPacket(packet, options), options);
}
/**
@@ -114,24 +155,60 @@ export class EncodedPacketSink {
* last key packet using `getKeyPacket(Infinity)`. The method returns null if the timestamp is before the first
* key packet in the track.
*
* To ensure that the returned packet is guaranteed to be a real key frame, enable `options.verifyKeyPackets`.
*
* @param timestamp - The timestamp used for retrieval, in seconds.
*/
getKeyPacket(timestamp: number, options: PacketRetrievalOptions = {}) {
async getKeyPacket(timestamp: number, options: PacketRetrievalOptions = {}): Promise<EncodedPacket | null> {
validateTimestamp(timestamp);
validatePacketRetrievalOptions(options);
return this._track._backing.getKeyPacket(timestamp, options);
if (!options.verifyKeyPackets) {
return this._track._backing.getKeyPacket(timestamp, options);
}
const packet = await this._track._backing.getKeyPacket(timestamp, options);
if (!packet || packet.type === 'delta') {
return packet;
}
const determinedType = await this._track.determinePacketType(packet);
if (determinedType === 'delta') {
// Try returning the previous key packet (in hopes that it's actually a key packet)
return this.getKeyPacket(packet.timestamp - 1 / this._track.timeResolution, options);
}
return packet;
}
/**
* Retrieves the key packet following the given packet (in decode order), or null if the given packet is the last
* key packet.
*
* To ensure that the returned packet is guaranteed to be a real key frame, enable `options.verifyKeyPackets`.
*/
getNextKeyPacket(packet: EncodedPacket, options: PacketRetrievalOptions = {}) {
async getNextKeyPacket(packet: EncodedPacket, options: PacketRetrievalOptions = {}): Promise<EncodedPacket | null> {
if (!(packet instanceof EncodedPacket)) {
throw new TypeError('packet must be an EncodedPacket.');
}
validatePacketRetrievalOptions(options);
return this._track._backing.getNextKeyPacket(packet, options);
if (!options.verifyKeyPackets) {
return this._track._backing.getNextKeyPacket(packet, options);
}
const nextPacket = await this._track._backing.getNextKeyPacket(packet, options);
if (!nextPacket || nextPacket.type === 'delta') {
return nextPacket;
}
const determinedType = await this._track.determinePacketType(nextPacket);
if (determinedType === 'delta') {
// Try returning the next key packet (in hopes that it's actually a key packet)
return this.getNextKeyPacket(nextPacket, options);
}
return nextPacket;
}
/**
@@ -346,7 +423,8 @@ export abstract class BaseMediaSampleSink<
});
const packetSink = this._createPacketSink();
const keyPacket = await packetSink.getKeyPacket(startTimestamp) ?? await packetSink.getFirstPacket();
const keyPacket = await packetSink.getKeyPacket(startTimestamp, { verifyKeyPackets: true })
?? await packetSink.getFirstPacket();
if (!keyPacket) {
return;
}
@@ -364,7 +442,7 @@ export abstract class BaseMediaSampleSink<
? null
: packet.type === 'key' && packet.timestamp === endTimestamp
? packet
: await packetSink.getNextKeyPacket(packet);
: await packetSink.getNextKeyPacket(packet, { verifyKeyPackets: true });
if (keyPacket) {
endPacket = keyPacket;
@@ -567,7 +645,7 @@ export abstract class BaseMediaSampleSink<
}
const targetPacket = await packetSink.getPacket(timestamp);
const keyPacket = targetPacket && await packetSink.getKeyPacket(timestamp);
const keyPacket = targetPacket && await packetSink.getKeyPacket(timestamp, { verifyKeyPackets: true });
if (!keyPacket) {
if (maxSequenceNumber !== -1) {
@@ -737,6 +815,7 @@ class VideoDecoderWrapper extends DecoderWrapper<VideoSample> {
// Round the timestamps to the time resolution
sample.setTimestamp(Math.round(sample.timestamp * this.timeResolution) / this.timeResolution);
sample.setDuration(Math.round(sample.duration * this.timeResolution) / this.timeResolution);
sample.setRotation(this.rotation);
this.onSample(sample);
}
+551 -239
View File
@@ -24,7 +24,7 @@ import {
VideoCodec,
} from './codec';
import { OutputAudioTrack, OutputSubtitleTrack, OutputTrack, OutputVideoTrack } from './output';
import { assert, assertNever, CallSerializer, clamp, setInt24, setUint24 } from './misc';
import { assert, assertNever, CallSerializer, clamp, promiseWithResolvers, setInt24, setUint24 } from './misc';
import { Muxer } from './muxer';
import { SubtitleParser } from './subtitles';
import { toAlaw, toUlaw } from './pcm';
@@ -78,7 +78,7 @@ export abstract class MediaSource {
}
/** @internal */
_start() {}
async _start() {}
/** @internal */
async _flushAndClose() {}
@@ -272,86 +272,94 @@ class VideoEncoderWrapper {
constructor(private source: VideoSource, private encodingConfig: VideoEncodingConfig) {}
async add(videoSample: VideoSample, shouldClose: boolean, encodeOptions?: VideoEncoderEncodeOptions) {
this.checkForEncoderError();
this.source._ensureValidAdd();
try {
this.checkForEncoderError();
this.source._ensureValidAdd();
// Ensure video sample size remains constant
if (this.lastWidth !== null && this.lastHeight !== null) {
if (videoSample.codedWidth !== this.lastWidth || videoSample.codedHeight !== this.lastHeight) {
throw new Error(
`Video sample size must remain constant. Expected ${this.lastWidth}x${this.lastHeight},`
+ ` got ${videoSample.codedWidth}x${videoSample.codedHeight}.`,
);
}
} else {
this.lastWidth = videoSample.codedWidth;
this.lastHeight = videoSample.codedHeight;
}
if (!this.encoderInitialized) {
if (!this.ensureEncoderPromise) {
void this.ensureEncoder(videoSample);
// Ensure video sample size remains constant
if (this.lastWidth !== null && this.lastHeight !== null) {
if (videoSample.codedWidth !== this.lastWidth || videoSample.codedHeight !== this.lastHeight) {
throw new Error(
`Video sample size must remain constant. Expected ${this.lastWidth}x${this.lastHeight},`
+ ` got ${videoSample.codedWidth}x${videoSample.codedHeight}.`,
);
}
} else {
this.lastWidth = videoSample.codedWidth;
this.lastHeight = videoSample.codedHeight;
}
// No, this "if" statement is not useless. Sometimes, the above call to `ensureEncoder` might have
// synchronously completed and the encoder is already initialized. In this case, we don't need to await the
// promise anymore. This also fixes nasty async race condition bugs when multiple code paths are calling
// this method: It's important that the call that initialized the encoder go through this code first.
if (!this.encoderInitialized) {
await this.ensureEncoderPromise;
if (!this.ensureEncoderPromise) {
void this.ensureEncoder(videoSample);
}
// No, this "if" statement is not useless. Sometimes, the above call to `ensureEncoder` might have
// synchronously completed and the encoder is already initialized. In this case, we don't need to await
// the promise anymore. This also fixes nasty async race condition bugs when multiple code paths are
// calling this method: It's important that the call that initialized the encoder go through this
// code first.
if (!this.encoderInitialized) {
await this.ensureEncoderPromise;
}
}
}
assert(this.encoderInitialized);
assert(this.encoderInitialized);
const keyFrameInterval = this.encodingConfig.keyFrameInterval ?? 5;
const multipleOfKeyFrameInterval = Math.floor(videoSample.timestamp / keyFrameInterval);
const keyFrameInterval = this.encodingConfig.keyFrameInterval ?? 5;
const multipleOfKeyFrameInterval = Math.floor(videoSample.timestamp / keyFrameInterval);
// Ensure a key frame every keyFrameInterval seconds. It is important that all video tracks follow the same
// "key frame" rhythm, because aligned key frames are required to start new fragments in ISOBMFF or clusters
// in Matroska (or at least desirable).
const finalEncodeOptions = {
...encodeOptions,
keyFrame: encodeOptions?.keyFrame
|| keyFrameInterval === 0
|| multipleOfKeyFrameInterval !== this.lastMultipleOfKeyFrameInterval,
};
this.lastMultipleOfKeyFrameInterval = multipleOfKeyFrameInterval;
// Ensure a key frame every keyFrameInterval seconds. It is important that all video tracks follow the same
// "key frame" rhythm, because aligned key frames are required to start new fragments in ISOBMFF or clusters
// in Matroska (or at least desirable).
const finalEncodeOptions = {
...encodeOptions,
keyFrame: encodeOptions?.keyFrame
|| keyFrameInterval === 0
|| multipleOfKeyFrameInterval !== this.lastMultipleOfKeyFrameInterval,
};
this.lastMultipleOfKeyFrameInterval = multipleOfKeyFrameInterval;
if (this.customEncoder) {
this.customEncoderQueueSize++;
const promise = this.customEncoderCallSerializer
.call(() => this.customEncoder!.encode(videoSample, finalEncodeOptions))
.then(() => {
this.customEncoderQueueSize--;
if (this.customEncoder) {
this.customEncoderQueueSize++;
const promise = this.customEncoderCallSerializer
.call(() => this.customEncoder!.encode(videoSample, finalEncodeOptions))
.then(() => {
this.customEncoderQueueSize--;
if (shouldClose) {
videoSample.close();
}
})
.catch((error: Error) => {
this.encoderError ??= error;
});
if (shouldClose) {
videoSample.close();
}
})
.catch((error: Error) => {
this.encoderError ??= error;
});
if (this.customEncoderQueueSize >= 4) {
await promise;
if (this.customEncoderQueueSize >= 4) {
await promise;
}
} else {
assert(this.encoder);
const videoFrame = videoSample.toVideoFrame();
this.encoder.encode(videoFrame, finalEncodeOptions);
videoFrame.close();
if (shouldClose) {
videoSample.close();
}
// We need to do this after sending the frame to the encoder as the frame otherwise might be closed
if (this.encoder.encodeQueueSize >= 4) {
await new Promise(resolve => this.encoder!.addEventListener('dequeue', resolve, { once: true }));
}
}
} else {
assert(this.encoder);
const videoFrame = videoSample.toVideoFrame();
this.encoder.encode(videoFrame, finalEncodeOptions);
videoFrame.close();
await this.muxer!.mutex.currentPromise; // Allow the writer to apply backpressure
} finally {
if (shouldClose) {
// Make sure it's always closed, even if there was an error
videoSample.close();
}
// We need to do this after sending the frame to the encoder as the frame otherwise might be closed
if (this.encoder.encodeQueueSize >= 4) {
await new Promise(resolve => this.encoder!.addEventListener('dequeue', resolve, { once: true }));
}
}
await this.muxer!.mutex.currentPromise; // Allow the writer to apply backpressure
}
private async ensureEncoder(videoSample: VideoSample) {
@@ -416,8 +424,9 @@ class VideoEncoderWrapper {
const support = await VideoEncoder.isConfigSupported(encoderConfig);
if (!support.supported) {
throw new Error(
'This specific encoder configuration is not supported by this browser. Consider using another'
+ ' codec or changing your video parameters.',
`This specific encoder configuration (${encoderConfig.codec}, ${encoderConfig.bitrate} bps,`
+ ` ${encoderConfig.width}x${encoderConfig.height}) is not supported by this browser. Consider`
+ ` using another codec or changing your video parameters.`,
);
}
@@ -574,6 +583,20 @@ export class MediaStreamVideoTrackSource extends VideoSource {
private _abortController: AbortController | null = null;
/** @internal */
private _track: MediaStreamVideoTrack;
/** @internal */
private _workerTrackId: number | null = null;
/** @internal */
private _workerListener: ((event: MessageEvent) => void) | null = null;
/** @internal */
private _promiseWithResolvers = promiseWithResolvers();
/** @internal */
private _errorPromiseAccessed = false;
/** A promise that rejects upon any error within this source. This promise never resolves. */
get errorPromise() {
this._errorPromiseAccessed = true;
return this._promiseWithResolvers.promise;
}
constructor(track: MediaStreamVideoTrack, encodingConfig: VideoEncodingConfig) {
if (!(track instanceof MediaStreamTrack) || track.kind !== 'video') {
@@ -592,41 +615,102 @@ export class MediaStreamVideoTrackSource extends VideoSource {
}
/** @internal */
override _start() {
override async _start() {
if (!this._errorPromiseAccessed) {
console.warn(
'Make sure not to ignore the `errorPromise` field on MediaStreamVideoTrackSource, so that any internal'
+ ' errors get bubbled up properly.',
);
}
this._abortController = new AbortController();
let frameReceived = false;
let firstVideoFrameTimestamp: number | null = null;
let errored = false;
const processor = new MediaStreamTrackProcessor({ track: this._track });
const consumer = new WritableStream<VideoFrame>({
write: (videoFrame) => {
if (!frameReceived) {
setMediaStreamTimestampOffset(this, videoFrame);
frameReceived = true;
const onVideoFrame = (videoFrame: VideoFrame) => {
if (errored) {
videoFrame.close();
return;
}
if (firstVideoFrameTimestamp === null) {
firstVideoFrameTimestamp = videoFrame.timestamp / 1e6;
const muxer = this._connectedTrack!.output._muxer;
if (muxer.firstMediaStreamTimestamp === null) {
muxer.firstMediaStreamTimestamp = performance.now() / 1000;
this._timestampOffset = -firstVideoFrameTimestamp;
} else {
this._timestampOffset = (performance.now() / 1000 - muxer.firstMediaStreamTimestamp)
- firstVideoFrameTimestamp;
}
}
if (this._encoder.getQueueSize() >= 4) {
// Drop frames if the encoder is overloaded
videoFrame.close();
return;
}
if (this._encoder.getQueueSize() >= 4) {
// Drop frames if the encoder is overloaded
videoFrame.close();
return;
}
void this._encoder.add(new VideoSample(videoFrame), true)
.catch((error) => {
this._abortController?.abort();
throw error;
});
},
});
void this._encoder.add(new VideoSample(videoFrame), true)
.catch((error) => {
errored = true;
processor.readable.pipeTo(consumer, {
signal: this._abortController.signal,
}).catch((err) => {
// Handle abort error silently
if (err instanceof DOMException && err.name === 'AbortError') return;
// Handle other errors
console.error('Pipe error:', err);
});
this._abortController?.abort();
this._promiseWithResolvers.reject(error);
if (this._workerTrackId !== null) {
// Tell the worker to stop the track
sendMessageToMediaStreamTrackProcessorWorker({
type: 'stopTrack',
trackId: this._workerTrackId,
});
}
});
};
if (typeof MediaStreamTrackProcessor !== 'undefined') {
// We can do it here directly, perfect
const processor = new MediaStreamTrackProcessor({ track: this._track });
const consumer = new WritableStream<VideoFrame>({ write: onVideoFrame });
processor.readable.pipeTo(consumer, {
signal: this._abortController.signal,
}).catch((error) => {
// Handle AbortError silently
if (error instanceof DOMException && error.name === 'AbortError') return;
this._promiseWithResolvers.reject(error);
});
} else {
// It might still be supported in a worker, so let's check that
const supportedInWorker = await mediaStreamTrackProcessorIsSupportedInWorker();
if (supportedInWorker) {
this._workerTrackId = nextMediaStreamTrackProcessorWorkerId++;
sendMessageToMediaStreamTrackProcessorWorker({
type: 'videoTrack',
trackId: this._workerTrackId,
track: this._track,
}, [this._track]);
this._workerListener = (event: MessageEvent) => {
const message = event.data as MediaStreamTrackProcessorWorkerMessage;
if (message.type === 'videoFrame' && message.trackId === this._workerTrackId) {
onVideoFrame(message.videoFrame);
} else if (message.type === 'error' && message.trackId === this._workerTrackId) {
this._promiseWithResolvers.reject(message.error);
}
};
mediaStreamTrackProcessorWorker!.addEventListener('message', this._workerListener);
} else {
throw new Error('MediaStreamTrackProcessor is required but not supported by this browser.');
}
}
}
/** @internal */
@@ -636,6 +720,32 @@ export class MediaStreamVideoTrackSource extends VideoSource {
this._abortController = null;
}
if (this._workerTrackId !== null) {
assert(this._workerListener);
sendMessageToMediaStreamTrackProcessorWorker({
type: 'stopTrack',
trackId: this._workerTrackId,
});
// Wait for the worker to stop the track
await new Promise<void>((resolve) => {
const listener = (event: MessageEvent) => {
const message = event.data as MediaStreamTrackProcessorWorkerMessage;
if (message.type === 'trackStopped' && message.trackId === this._workerTrackId) {
assert(this._workerListener);
mediaStreamTrackProcessorWorker!.removeEventListener('message', this._workerListener);
mediaStreamTrackProcessorWorker!.removeEventListener('message', listener);
resolve();
}
};
mediaStreamTrackProcessorWorker!.addEventListener('message', listener);
});
}
await this._encoder.flushAndClose();
}
}
@@ -782,78 +892,86 @@ class AudioEncoderWrapper {
constructor(private source: AudioSource, private encodingConfig: AudioEncodingConfig) {}
async add(audioSample: AudioSample, shouldClose: boolean) {
this.checkForEncoderError();
this.source._ensureValidAdd();
try {
this.checkForEncoderError();
this.source._ensureValidAdd();
// Ensure audio parameters remain constant
if (this.lastNumberOfChannels !== null && this.lastSampleRate !== null) {
if (
audioSample.numberOfChannels !== this.lastNumberOfChannels
|| audioSample.sampleRate !== this.lastSampleRate
) {
throw new Error(
`Audio parameters must remain constant. Expected ${this.lastNumberOfChannels} channels at`
+ ` ${this.lastSampleRate} Hz, got ${audioSample.numberOfChannels} channels at`
+ ` ${audioSample.sampleRate} Hz.`,
);
}
} else {
this.lastNumberOfChannels = audioSample.numberOfChannels;
this.lastSampleRate = audioSample.sampleRate;
}
if (!this.encoderInitialized) {
if (!this.ensureEncoderPromise) {
void this.ensureEncoder(audioSample);
// Ensure audio parameters remain constant
if (this.lastNumberOfChannels !== null && this.lastSampleRate !== null) {
if (
audioSample.numberOfChannels !== this.lastNumberOfChannels
|| audioSample.sampleRate !== this.lastSampleRate
) {
throw new Error(
`Audio parameters must remain constant. Expected ${this.lastNumberOfChannels} channels at`
+ ` ${this.lastSampleRate} Hz, got ${audioSample.numberOfChannels} channels at`
+ ` ${audioSample.sampleRate} Hz.`,
);
}
} else {
this.lastNumberOfChannels = audioSample.numberOfChannels;
this.lastSampleRate = audioSample.sampleRate;
}
// No, this "if" statement is not useless. Sometimes, the above call to `ensureEncoder` might have
// synchronously completed and the encoder is already initialized. In this case, we don't need to await the
// promise anymore. This also fixes nasty async race condition bugs when multiple code paths are calling
// this method: It's important that the call that initialized the encoder go through this code first.
if (!this.encoderInitialized) {
await this.ensureEncoderPromise;
if (!this.ensureEncoderPromise) {
void this.ensureEncoder(audioSample);
}
// No, this "if" statement is not useless. Sometimes, the above call to `ensureEncoder` might have
// synchronously completed and the encoder is already initialized. In this case, we don't need to await
// the promise anymore. This also fixes nasty async race condition bugs when multiple code paths are
// calling this method: It's important that the call that initialized the encoder go through this
// code first.
if (!this.encoderInitialized) {
await this.ensureEncoderPromise;
}
}
}
assert(this.encoderInitialized);
assert(this.encoderInitialized);
if (this.customEncoder) {
this.customEncoderQueueSize++;
const promise = this.customEncoderCallSerializer
.call(() => this.customEncoder!.encode(audioSample))
.then(() => {
this.customEncoderQueueSize--;
if (this.customEncoder) {
this.customEncoderQueueSize++;
const promise = this.customEncoderCallSerializer
.call(() => this.customEncoder!.encode(audioSample))
.then(() => {
this.customEncoderQueueSize--;
if (shouldClose) {
audioSample.close();
}
})
.catch((error: Error) => {
this.encoderError ??= error;
});
if (shouldClose) {
audioSample.close();
}
})
.catch((error: Error) => {
this.encoderError ??= error;
});
if (this.customEncoderQueueSize >= 4) {
await promise;
if (this.customEncoderQueueSize >= 4) {
await promise;
}
await this.muxer!.mutex.currentPromise; // Allow the writer to apply backpressure
} else if (this.isPcmEncoder) {
await this.doPcmEncoding(audioSample, shouldClose);
} else {
assert(this.encoder);
const audioData = audioSample.toAudioData();
this.encoder.encode(audioData);
audioData.close();
if (shouldClose) {
audioSample.close();
}
if (this.encoder.encodeQueueSize >= 4) {
await new Promise(resolve => this.encoder!.addEventListener('dequeue', resolve, { once: true }));
}
await this.muxer!.mutex.currentPromise; // Allow the writer to apply backpressure
}
await this.muxer!.mutex.currentPromise; // Allow the writer to apply backpressure
} else if (this.isPcmEncoder) {
await this.doPcmEncoding(audioSample, shouldClose);
} else {
assert(this.encoder);
const audioData = audioSample.toAudioData();
this.encoder.encode(audioData);
audioData.close();
} finally {
if (shouldClose) {
// Make sure it's always closed, even if there was an error
audioSample.close();
}
if (this.encoder.encodeQueueSize >= 4) {
await new Promise(resolve => this.encoder!.addEventListener('dequeue', resolve, { once: true }));
}
await this.muxer!.mutex.currentPromise; // Allow the writer to apply backpressure
}
}
@@ -988,8 +1106,9 @@ class AudioEncoderWrapper {
const support = await AudioEncoder.isConfigSupported(encoderConfig);
if (!support.supported) {
throw new Error(
'This specific encoder configuration not supported by this browser. Consider using another'
+ ' codec or changing your audio parameters.',
`This specific encoder configuration (${encoderConfig.codec}, ${encoderConfig.bitrate} bps,`
+ ` ${encoderConfig.numberOfChannels} channels, ${encoderConfig.sampleRate} Hz) is not`
+ ` supported by this browser. Consider using another codec or changing your audio parameters.`,
);
}
@@ -1184,7 +1303,7 @@ export class AudioBufferSource extends AudioSource {
/** @internal */
private _encoder: AudioEncoderWrapper;
/** @internal */
private _accumulatedFrameCount = 0;
private _accumulatedTime = 0;
constructor(encodingConfig: AudioEncodingConfig) {
validateAudioEncodingConfig(encodingConfig);
@@ -1206,47 +1325,10 @@ export class AudioBufferSource extends AudioSource {
throw new TypeError('audioBuffer must be an AudioBuffer.');
}
const MAX_FLOAT_COUNT = 64 * 1024 * 1024;
const audioSamples = AudioSample.fromAudioBuffer(audioBuffer, this._accumulatedTime);
const promises = audioSamples.map(sample => this._encoder.add(sample, true));
const numberOfChannels = audioBuffer.numberOfChannels;
const sampleRate = audioBuffer.sampleRate;
const totalFrames = audioBuffer.length;
const maxFramesPerChunk = Math.floor(MAX_FLOAT_COUNT / numberOfChannels);
let currentRelativeFrame = 0;
let remainingFrames = totalFrames;
const promises: Promise<void>[] = [];
// Create AudioData in a chunked fashion so we don't create huge Float32Arrays
while (remainingFrames > 0) {
const framesToCopy = Math.min(maxFramesPerChunk, remainingFrames);
const chunkData = new Float32Array(numberOfChannels * framesToCopy);
for (let channel = 0; channel < numberOfChannels; channel++) {
audioBuffer.copyFromChannel(
chunkData.subarray(channel * framesToCopy, channel * framesToCopy + framesToCopy),
channel,
currentRelativeFrame,
);
}
const audioSample = new AudioSample({
format: 'f32-planar',
sampleRate,
numberOfFrames: framesToCopy,
numberOfChannels,
timestamp: (this._accumulatedFrameCount + currentRelativeFrame) / sampleRate,
data: chunkData,
});
promises.push(this._encoder.add(audioSample, true));
currentRelativeFrame += framesToCopy;
remainingFrames -= framesToCopy;
}
this._accumulatedFrameCount += totalFrames;
this._accumulatedTime += audioBuffer.duration;
return Promise.all(promises);
}
@@ -1270,6 +1352,20 @@ export class MediaStreamAudioTrackSource extends AudioSource {
private _abortController: AbortController | null = null;
/** @internal */
private _track: MediaStreamAudioTrack;
/** @internal */
private _audioContext: AudioContext | null = null;
/** @internal */
private _scriptProcessorNode: ScriptProcessorNode | null = null; // Deprecated but goated
/** @internal */
private _promiseWithResolvers = promiseWithResolvers();
/** @internal */
private _errorPromiseAccessed = false;
/** A promise that rejects upon any error within this source. This promise never resolves. */
get errorPromise() {
this._errorPromiseAccessed = true;
return this._promiseWithResolvers.promise;
}
constructor(track: MediaStreamAudioTrack, encodingConfig: AudioEncodingConfig) {
if (!(track instanceof MediaStreamTrack) || track.kind !== 'audio') {
@@ -1283,41 +1379,104 @@ export class MediaStreamAudioTrackSource extends AudioSource {
}
/** @internal */
override _start() {
override async _start() {
if (!this._errorPromiseAccessed) {
console.warn(
'Make sure not to ignore the `errorPromise` field on MediaStreamVideoTrackSource, so that any internal'
+ ' errors get bubbled up properly.',
);
}
this._abortController = new AbortController();
let dataReceived = false;
if (typeof MediaStreamTrackProcessor !== 'undefined') {
// Great, MediaStreamTrackProcessor is supported, this is the preferred way of doing things
let firstAudioDataTimestamp: number | null = null;
const processor = new MediaStreamTrackProcessor({ track: this._track });
const consumer = new WritableStream<AudioData>({
write: (audioData) => {
if (!dataReceived) {
setMediaStreamTimestampOffset(this, audioData);
dataReceived = true;
const processor = new MediaStreamTrackProcessor({ track: this._track });
const consumer = new WritableStream<AudioData>({
write: (audioData) => {
if (firstAudioDataTimestamp === null) {
firstAudioDataTimestamp = audioData.timestamp / 1e6;
const muxer = this._connectedTrack!.output._muxer;
if (muxer.firstMediaStreamTimestamp === null) {
muxer.firstMediaStreamTimestamp = performance.now() / 1000;
this._timestampOffset = -firstAudioDataTimestamp;
} else {
this._timestampOffset = (performance.now() / 1000 - muxer.firstMediaStreamTimestamp)
- firstAudioDataTimestamp;
}
}
if (this._encoder.getQueueSize() >= 4) {
// Drop data if the encoder is overloaded
audioData.close();
return;
}
void this._encoder.add(new AudioSample(audioData), true)
.catch((error) => {
this._abortController?.abort();
this._promiseWithResolvers.reject(error);
});
},
});
processor.readable.pipeTo(consumer, {
signal: this._abortController.signal,
}).catch((error) => {
// Handle AbortError silently
if (error instanceof DOMException && error.name === 'AbortError') return;
this._promiseWithResolvers.reject(error);
});
} else {
// Let's fall back to an AudioContext approach
this._audioContext = new AudioContext({ sampleRate: this._track.getSettings().sampleRate });
const sourceNode = this._audioContext.createMediaStreamSource(new MediaStream([this._track]));
this._scriptProcessorNode = this._audioContext.createScriptProcessor(4096);
if (this._audioContext.state === 'suspended') {
await this._audioContext.resume();
}
sourceNode.connect(this._scriptProcessorNode);
this._scriptProcessorNode.connect(this._audioContext.destination);
let audioReceived = false;
let totalDuration = 0;
this._scriptProcessorNode.onaudioprocess = (event) => {
const audioSamples = AudioSample.fromAudioBuffer(event.inputBuffer, totalDuration);
totalDuration += event.inputBuffer.duration;
for (const audioSample of audioSamples) {
if (!audioReceived) {
audioReceived = true;
const muxer = this._connectedTrack!.output._muxer;
if (muxer.firstMediaStreamTimestamp === null) {
muxer.firstMediaStreamTimestamp = performance.now() / 1000;
} else {
this._timestampOffset = performance.now() / 1000 - muxer.firstMediaStreamTimestamp;
}
}
if (this._encoder.getQueueSize() >= 4) {
// Drop data if the encoder is overloaded
audioSample.close();
continue;
}
void this._encoder.add(audioSample, true)
.catch((error) => {
void this._audioContext!.suspend();
this._promiseWithResolvers.reject(error);
});
}
if (this._encoder.getQueueSize() >= 4) {
// Drop data if the encoder is overloaded
audioData.close();
return;
}
void this._encoder.add(new AudioSample(audioData), true)
.catch((error) => {
this._abortController?.abort();
throw error;
});
},
});
processor.readable.pipeTo(consumer, {
signal: this._abortController.signal,
}).catch((err) => {
// Handle abort error silently
if (err instanceof DOMException && err.name === 'AbortError') return;
// Handle other errors
console.error('Pipe error:', err);
});
};
}
}
/** @internal */
@@ -1327,22 +1486,175 @@ export class MediaStreamAudioTrackSource extends AudioSource {
this._abortController = null;
}
if (this._audioContext) {
assert(this._scriptProcessorNode);
this._scriptProcessorNode.disconnect();
await this._audioContext.suspend();
}
await this._encoder.flushAndClose();
}
}
const setMediaStreamTimestampOffset = (source: MediaSource, sample: VideoFrame | AudioData) => {
const timestampInSeconds = sample.timestamp / 1e6;
// === MEDIA STREAM TRACK PROCESSOR WORKER ===
assert(source._connectedTrack);
const muxer = source._connectedTrack.output._muxer;
if (muxer.firstMediaStreamTimestamp === null) {
// We're the first MediaStreamTrack of this output to receive data
muxer.firstMediaStreamTimestamp = timestampInSeconds;
type MediaStreamTrackProcessorWorkerMessage = {
type: 'support';
supported: boolean;
} | {
type: 'videoFrame';
trackId: number;
videoFrame: VideoFrame;
} | {
type: 'trackStopped';
trackId: number;
} | {
type: 'error';
trackId: number;
error: Error;
};
type MediaStreamTrackProcessorControllerMessage = {
type: 'videoTrack';
trackId: number;
track: MediaStreamVideoTrack;
} | {
type: 'stopTrack';
trackId: number;
};
const mediaStreamTrackProcessorWorkerCode = () => {
const sendMessage = (message: MediaStreamTrackProcessorWorkerMessage, transfer?: Transferable[]) => {
if (transfer) {
// The error is bullshit, it's using the wrong postMessage
// eslint-disable-next-line @typescript-eslint/no-explicit-any, @typescript-eslint/no-unsafe-argument
self.postMessage(message, transfer as any);
} else {
self.postMessage(message);
}
};
// Immediately send a message to the main thread, letting them know of the support
sendMessage({
type: 'support',
supported: typeof MediaStreamTrackProcessor !== 'undefined',
});
const abortControllers = new Map<number, AbortController>();
const stoppedTracks = new Set<number>();
self.addEventListener('message', (event) => {
const message = event.data as MediaStreamTrackProcessorControllerMessage;
switch (message.type) {
case 'videoTrack': {
const processor = new MediaStreamTrackProcessor({ track: message.track });
const consumer = new WritableStream<VideoFrame>({
write: (videoFrame) => {
if (stoppedTracks.has(message.trackId)) {
videoFrame.close();
return;
}
// Send it to the main thread
sendMessage({
type: 'videoFrame',
trackId: message.trackId,
videoFrame,
}, [videoFrame]);
},
});
const abortController = new AbortController();
abortControllers.set(message.trackId, abortController);
processor.readable.pipeTo(consumer, {
signal: abortController.signal,
}).catch((error: Error) => {
// Handle AbortError silently
if (error instanceof DOMException && error.name === 'AbortError') return;
sendMessage({
type: 'error',
trackId: message.trackId,
error,
});
});
}; break;
case 'stopTrack': {
const abortController = abortControllers.get(message.trackId);
if (abortController) {
abortController.abort();
abortControllers.delete(message.trackId);
}
stoppedTracks.add(message.trackId);
sendMessage({
type: 'trackStopped',
trackId: message.trackId,
});
}; break;
default: assertNever(message);
}
});
};
let nextMediaStreamTrackProcessorWorkerId = 0;
let mediaStreamTrackProcessorWorker: Worker | null = null;
const initMediaStreamTrackProcessorWorker = () => {
const blob = new Blob(
[`(${mediaStreamTrackProcessorWorkerCode.toString()})()`],
{ type: 'application/javascript' },
);
const url = URL.createObjectURL(blob);
mediaStreamTrackProcessorWorker = new Worker(url);
};
let mediaStreamTrackProcessorIsSupportedInWorkerCache: boolean | null = null;
const mediaStreamTrackProcessorIsSupportedInWorker = async () => {
if (mediaStreamTrackProcessorIsSupportedInWorkerCache !== null) {
return mediaStreamTrackProcessorIsSupportedInWorkerCache;
}
// Math.min to ensure the timestamps can't get negative
source._timestampOffset = -Math.min(muxer.firstMediaStreamTimestamp, timestampInSeconds);
if (!mediaStreamTrackProcessorWorker) {
initMediaStreamTrackProcessorWorker();
}
return new Promise<boolean>((resolve) => {
assert(mediaStreamTrackProcessorWorker);
const listener = (event: MessageEvent) => {
const message = event.data as MediaStreamTrackProcessorWorkerMessage;
if (message.type === 'support') {
mediaStreamTrackProcessorIsSupportedInWorkerCache = message.supported;
mediaStreamTrackProcessorWorker!.removeEventListener('message', listener);
resolve(message.supported);
}
};
mediaStreamTrackProcessorWorker.addEventListener('message', listener);
});
};
const sendMessageToMediaStreamTrackProcessorWorker = (
message: MediaStreamTrackProcessorControllerMessage,
transfer?: Transferable[],
) => {
assert(mediaStreamTrackProcessorWorker);
if (transfer) {
mediaStreamTrackProcessorWorker.postMessage(message, transfer);
} else {
mediaStreamTrackProcessorWorker.postMessage(message);
}
};
/**
+16 -9
View File
@@ -48,7 +48,7 @@ export class Bitstream {
this.pos = 8 * byteOffset;
}
readBit() {
private readBit() {
const byteIndex = Math.floor(this.pos / 8);
const byte = this.bytes[byteIndex] ?? 0;
const bitIndex = 0b111 - (this.pos & 0b111);
@@ -59,6 +59,10 @@ export class Bitstream {
}
readBits(n: number) {
if (n === 1) {
return this.readBit();
}
let result = 0;
for (let i = 0; i < n; i++) {
@@ -100,7 +104,7 @@ export class Bitstream {
/** Reads an exponential-Golomb universal code from a Bitstream. */
export const readExpGolomb = (bitstream: Bitstream) => {
let leadingZeroBits = 0;
while (bitstream.readBit() === 0 && leadingZeroBits < 32) {
while (bitstream.readBits(1) === 0 && leadingZeroBits < 32) {
leadingZeroBits++;
}
@@ -134,7 +138,9 @@ export const writeBits = (bytes: Uint8Array, start: number, end: number, value:
};
export const toUint8Array = (source: AllowSharedBufferSource): Uint8Array => {
if (source instanceof ArrayBuffer) {
if (source instanceof Uint8Array) {
return source;
} else if (source instanceof ArrayBuffer) {
return new Uint8Array(source);
} else {
return new Uint8Array(source.buffer, source.byteOffset, source.byteLength);
@@ -142,7 +148,9 @@ export const toUint8Array = (source: AllowSharedBufferSource): Uint8Array => {
};
export const toDataView = (source: AllowSharedBufferSource) => {
if (source instanceof ArrayBuffer) {
if (source instanceof DataView) {
return source;
} else if (source instanceof ArrayBuffer) {
return new DataView(source);
} else {
return new DataView(source.buffer, source.byteOffset, source.byteLength);
@@ -202,11 +210,10 @@ export const colorSpaceIsComplete = (
};
export const isAllowSharedBufferSource = (x: unknown) => {
// Quite a mouthful:
return (
x instanceof ArrayBuffer
|| (typeof SharedArrayBuffer !== 'undefined' && x instanceof SharedArrayBuffer)
|| (ArrayBuffer.isView(x) && !(x instanceof DataView))
|| ArrayBuffer.isView(x)
);
};
@@ -508,15 +515,15 @@ export const retriedFetch = async (
try {
return await fetch(url, requestInit);
} catch (error) {
console.error('Retrying failed fetch. Error:', error);
attempts++;
const retryDelayInSeconds = getRetryDelay(attempts);
if (retryDelayInSeconds === null) {
throw error;
}
console.error('Retrying failed fetch. Error:', error);
if (!Number.isFinite(retryDelayInSeconds) || retryDelayInSeconds < 0) {
throw new TypeError('Retry delay must be a non-negative finite number.');
}
+4
View File
@@ -85,6 +85,10 @@ export const readFrameHeader = (word: number, reader: { pos: number; fileSize: n
reader.pos += 1;
if (firstByte !== 0xff) {
return null;
}
if ((secondByte & 0xe0) !== 0xe0) {
return null;
}
+1 -1
View File
@@ -65,7 +65,7 @@ export class Mp3Reader {
assert(this.fileSize);
until ??= this.fileSize;
while (this.pos < until - FRAME_HEADER_SIZE) {
while (this.pos <= until - FRAME_HEADER_SIZE) {
const word = this.readU32();
this.pos -= 4;
+9
View File
@@ -546,6 +546,12 @@ export class Mp3OutputFormat extends OutputFormat {
* @public
*/
export type WavOutputFormatOptions = {
/**
* When enabled, an RF64 file be written, allowing for file sizes to exceed 4 GiB, which is otherwise not possible
* for regular WAVE files.
*/
large?: boolean;
/**
* Will be called once the file header is written. The header consists of the RIFF header, the format chunk, and the
* start of the data chunk (with a placeholder size of 0).
@@ -565,6 +571,9 @@ export class WavOutputFormat extends OutputFormat {
if (!options || typeof options !== 'object') {
throw new TypeError('options must be an object.');
}
if (options.large !== undefined && typeof options.large !== 'boolean') {
throw new TypeError('options.large, when provided, must be a boolean.');
}
if (options.onHeader !== undefined && typeof options.onHeader !== 'function') {
throw new TypeError('options.onHeader, when provided, must be a function.');
}
+2 -3
View File
@@ -344,9 +344,8 @@ export class Output<
await this._muxer.start();
for (const track of this._tracks) {
track.source._start();
}
const promises = this._tracks.map(track => track.source._start());
await Promise.all(promises);
release();
})();
+167 -7
View File
@@ -236,6 +236,7 @@ export class VideoSample {
return new VideoSample(this._data.clone(), {
timestamp: this.timestamp,
duration: this.duration,
rotation: this.rotation,
});
} else if (this._data instanceof Uint8Array) {
return new VideoSample(this._data.slice(), {
@@ -245,6 +246,7 @@ export class VideoSample {
timestamp: this.timestamp,
duration: this.duration,
colorSpace: this.colorSpace,
rotation: this.rotation,
});
} else {
return new VideoSample(this._data, {
@@ -254,6 +256,7 @@ export class VideoSample {
timestamp: this.timestamp,
duration: this.duration,
colorSpace: this.colorSpace,
rotation: this.rotation,
});
}
}
@@ -366,9 +369,79 @@ export class VideoSample {
context: CanvasRenderingContext2D | OffscreenCanvasRenderingContext2D,
dx: number,
dy: number,
dWidth: number = this.displayWidth,
dHeight: number = this.displayHeight,
dWidth?: number,
dHeight?: number,
): void;
/**
* Draws the video sample to a 2D canvas context. Rotation metadata will be taken into account.
*
* @param sx - The x-coordinate of the top left corner of the sub-rectangle of the source image to draw into the
* destination context.
* @param sy - The y-coordinate of the top left corner of the sub-rectangle of the source image to draw into the
* destination context.
* @param sWidth - The width of the sub-rectangle of the source image to draw into the destination context.
* @param sHeight - The height of the sub-rectangle of the source image to draw into the destination context.
* @param dx - The x-coordinate in the destination canvas at which to place the top-left corner of the source image.
* @param dy - The y-coordinate in the destination canvas at which to place the top-left corner of the source image.
* @param dWidth - The width in pixels with which to draw the image in the destination canvas.
* @param dHeight - The height in pixels with which to draw the image in the destination canvas.
*/
draw(
context: CanvasRenderingContext2D | OffscreenCanvasRenderingContext2D,
sx: number,
sy: number,
sWidth: number,
sHeight: number,
dx: number,
dy: number,
dWidth?: number,
dHeight?: number,
): void;
draw(
context: CanvasRenderingContext2D | OffscreenCanvasRenderingContext2D,
arg1: number,
arg2: number,
arg3?: number,
arg4?: number,
arg5?: number,
arg6?: number,
arg7?: number,
arg8?: number,
) {
let sx = 0;
let sy = 0;
let sWidth = this.displayWidth;
let sHeight = this.displayHeight;
let dx = 0;
let dy = 0;
let dWidth = this.displayWidth;
let dHeight = this.displayHeight;
if (arg5 !== undefined) {
sx = arg1!;
sy = arg2!;
sWidth = arg3!;
sHeight = arg4!;
dx = arg5;
dy = arg6!;
if (arg7 !== undefined) {
dWidth = arg7;
dHeight = arg8!;
} else {
dWidth = sWidth;
dHeight = sHeight;
}
} else {
dx = arg1;
dy = arg2;
if (arg3 !== undefined) {
dWidth = arg3;
dHeight = arg4!;
}
}
if (!(
(typeof CanvasRenderingContext2D !== 'undefined' && context instanceof CanvasRenderingContext2D)
|| (
@@ -378,6 +451,18 @@ export class VideoSample {
)) {
throw new TypeError('context must be a CanvasRenderingContext2D or OffscreenCanvasRenderingContext2D.');
}
if (!Number.isFinite(sx)) {
throw new TypeError('sx must be a number.');
}
if (!Number.isFinite(sy)) {
throw new TypeError('sy must be a number.');
}
if (!Number.isFinite(sWidth) || sWidth < 0) {
throw new TypeError('sWidth must be a non-negative number.');
}
if (!Number.isFinite(sHeight) || sHeight < 0) {
throw new TypeError('sHeight must be a non-negative number.');
}
if (!Number.isFinite(dx)) {
throw new TypeError('dx must be a number.');
}
@@ -395,6 +480,29 @@ export class VideoSample {
throw new Error('VideoSample is closed.');
}
// The provided sx,sy,sWidth,sHeight refer to the final rotated image, but that's not actually how the image is
// stored. Therefore, we must map these back onto the original, pre-rotation image.
if (this.rotation === 90) {
[sx, sy, sWidth, sHeight] = [
sy,
this.codedHeight - sx - sWidth,
sHeight,
sWidth,
];
} else if (this.rotation === 180) {
[sx, sy] = [
this.codedWidth - sx - sWidth,
this.codedHeight - sy - sHeight,
];
} else if (this.rotation === 270) {
[sx, sy, sWidth, sHeight] = [
this.codedWidth - sy - sHeight,
sx,
sHeight,
sWidth,
];
}
const source = this.toCanvasImageSource();
context.save();
@@ -412,6 +520,10 @@ export class VideoSample {
context.drawImage(
source,
sx,
sy,
sWidth,
sHeight,
-dWidth / 2,
-dHeight / 2,
dWidth,
@@ -890,7 +1002,6 @@ export class AudioSample {
}
return new AudioData({
format: this.format,
sampleRate: this.sampleRate,
numberOfFrames: this.numberOfFrames,
@@ -900,11 +1011,9 @@ export class AudioSample {
});
} else {
const data = new ArrayBuffer(this.allocationSize({ planeIndex: 0, format: this.format }));
this.copyTo(new DataView(data), { planeIndex: 0, format: this.format });
this.copyTo(data, { planeIndex: 0, format: this.format });
return new AudioData({
format: this.format,
sampleRate: this.sampleRate,
numberOfFrames: this.numberOfFrames,
@@ -916,7 +1025,6 @@ export class AudioSample {
}
} else {
return new AudioData({
format: this.format,
sampleRate: this.sampleRate,
numberOfFrames: this.numberOfFrames,
@@ -958,6 +1066,58 @@ export class AudioSample {
// eslint-disable-next-line @typescript-eslint/no-unnecessary-type-assertion
(this.timestamp as number) = newTimestamp;
}
/**
* Creates AudioSamples from an AudioBuffer, starting at the given timestamp in seconds. Typically creates exactly
* one sample, but may create multiple if the AudioBuffer is exceedingly large.
*/
static fromAudioBuffer(audioBuffer: AudioBuffer, timestamp: number) {
if (!(audioBuffer instanceof AudioBuffer)) {
throw new TypeError('audioBuffer must be an AudioBuffer.');
}
const MAX_FLOAT_COUNT = 64 * 1024 * 1024;
const numberOfChannels = audioBuffer.numberOfChannels;
const sampleRate = audioBuffer.sampleRate;
const totalFrames = audioBuffer.length;
const maxFramesPerChunk = Math.floor(MAX_FLOAT_COUNT / numberOfChannels);
let currentRelativeFrame = 0;
let remainingFrames = totalFrames;
const result: AudioSample[] = [];
// Create AudioData in a chunked fashion so we don't create huge Float32Arrays
while (remainingFrames > 0) {
const framesToCopy = Math.min(maxFramesPerChunk, remainingFrames);
const chunkData = new Float32Array(numberOfChannels * framesToCopy);
for (let channel = 0; channel < numberOfChannels; channel++) {
audioBuffer.copyFromChannel(
chunkData.subarray(channel * framesToCopy, channel * framesToCopy + framesToCopy),
channel,
currentRelativeFrame,
);
}
const audioSample = new AudioSample({
format: 'f32-planar',
sampleRate,
numberOfFrames: framesToCopy,
numberOfChannels,
timestamp: timestamp + currentRelativeFrame / sampleRate,
data: chunkData,
});
result.push(audioSample);
currentRelativeFrame += framesToCopy;
remainingFrames -= framesToCopy;
}
return result;
}
}
const getBytesPerSample = (format: AudioSampleFormat): number => {
+27 -2
View File
@@ -218,7 +218,8 @@ export class UrlSource extends Source {
const buffer = await response.arrayBuffer();
if (!range) {
if (response.status === 200) {
// The server didn't return 206 Partial Content, so it's not a range response
this._fullData = buffer;
}
@@ -252,6 +253,26 @@ export class UrlSource extends Source {
return this._fullData.byteLength;
}
// First, try a HEAD request to get the size
try {
const headResponse = await retriedFetch(
this._url,
mergeObjectsDeeply(this._options.requestInit ?? {}, {
method: 'HEAD',
}),
this._options.getRetryDelay ?? (() => null),
);
if (headResponse.ok) {
const contentLength = headResponse.headers.get('Content-Length');
if (contentLength) {
return parseInt(contentLength);
}
}
} catch {
// We tried
}
// Try a range request to get the Content-Range header
const rangeResponse = await retriedFetch(
this._url,
@@ -267,9 +288,13 @@ export class UrlSource extends Source {
if (contentRange) {
const match = contentRange.match(/bytes \d+-\d+\/(\d+)/);
if (match && match[1]) {
return parseInt(match[1], 10);
return parseInt(match[1]);
}
}
} else if (rangeResponse.status === 200) {
// The server just returned the whole thing
this._fullData = await rangeResponse.arrayBuffer();
return this._fullData.byteLength;
}
// If the range request didn't provide the size, make a full GET request
+1 -2
View File
@@ -1,9 +1,8 @@
{
"extends": "../tsconfig.json",
"compilerOptions": {
"outDir": "../build",
"outDir": "../dist/modules",
"declaration": true,
"sourceMap": true,
"declarationMap": true,
"stripInternal": true,
"noEmit": false
+15
View File
@@ -35,6 +35,21 @@ export class RiffReader {
return view.getUint32(offset, this.littleEndian);
}
readU64() {
let low: number;
let high: number;
if (this.littleEndian) {
low = this.readU32();
high = this.readU32();
} else {
high = this.readU32();
low = this.readU32();
}
return high * 0x100000000 + low;
}
readAscii(length: number) {
const { view, offset } = this.reader.getViewAndOffset(this.pos, this.pos + length);
this.pos += length;
+9 -3
View File
@@ -14,14 +14,20 @@ export class RiffWriter {
constructor(private writer: Writer) {}
writeU16(value: number) {
this.helperView.setUint16(0, value, true);
this.writer.write(this.helper.subarray(0, 2));
}
writeU32(value: number) {
this.helperView.setUint32(0, value, true);
this.writer.write(this.helper.subarray(0, 4));
}
writeU16(value: number) {
this.helperView.setUint16(0, value, true);
this.writer.write(this.helper.subarray(0, 2));
writeU64(value: number) {
this.helperView.setUint32(0, value, true);
this.helperView.setUint32(4, Math.floor(value / 2 ** 32), true);
this.writer.write(this.helper);
}
writeAscii(text: string) {
+24 -3
View File
@@ -53,9 +53,13 @@ export class WaveDemuxer extends Demuxer {
const actualFileSize = await this.metadataReader.reader.source.getSize();
const riffType = this.metadataReader.readAscii(4);
this.metadataReader.littleEndian = riffType === 'RIFF';
this.metadataReader.littleEndian = riffType !== 'RIFX';
const totalFileSize = Math.min(this.metadataReader.readU32() + 8, actualFileSize);
const isRf64 = riffType === 'RF64';
const outerChunkSize = this.metadataReader.readU32();
let totalFileSize = isRf64 ? actualFileSize : Math.min(outerChunkSize + 8, actualFileSize);
const format = this.metadataReader.readAscii(4);
if (format !== 'WAVE') {
@@ -63,6 +67,9 @@ export class WaveDemuxer extends Demuxer {
}
this.metadataReader.pos = 12;
let chunksRead = 0;
let dataChunkSize: number | null = null;
while (this.metadataReader.pos < totalFileSize) {
await this.metadataReader.reader.loadRange(this.metadataReader.pos, this.metadataReader.pos + 8);
@@ -70,14 +77,28 @@ export class WaveDemuxer extends Demuxer {
const chunkSize = this.metadataReader.readU32();
const startPos = this.metadataReader.pos;
if (isRf64 && chunksRead === 0 && chunkId !== 'ds64') {
throw new Error('Invalid RF64 file: First chunk must be "ds64".');
}
if (chunkId === 'fmt ') {
await this.parseFmtChunk(chunkSize);
} else if (chunkId === 'data') {
dataChunkSize ??= chunkSize;
this.dataStart = this.metadataReader.pos;
this.dataSize = Math.min(chunkSize, totalFileSize - this.dataStart);
this.dataSize = Math.min(dataChunkSize, totalFileSize - this.dataStart);
} else if (chunkId === 'ds64') {
// File and data chunk sizes are defined in here instead
const riffChunkSize = this.metadataReader.readU64();
dataChunkSize = this.metadataReader.readU64();
totalFileSize = Math.min(riffChunkSize + 8, actualFileSize);
}
this.metadataReader.pos = startPos + chunkSize + (chunkSize & 1); // Handle padding
chunksRead++;
}
if (!this.audioInfo) {
+57 -9
View File
@@ -18,10 +18,13 @@ import { assert } from '../misc';
export class WaveMuxer extends Muxer {
private format: WavOutputFormat;
private isRf64: boolean;
private writer: Writer;
private riffWriter: RiffWriter;
private headerWritten = false;
private dataSize = 0;
private sampleRate: number | null = null;
private sampleCount = 0;
constructor(output: Output, format: WavOutputFormat) {
super(output);
@@ -29,6 +32,7 @@ export class WaveMuxer extends Muxer {
this.format = format;
this.writer = output._writer;
this.riffWriter = new RiffWriter(output._writer);
this.isRf64 = !!format._options.large;
}
async start() {
@@ -58,13 +62,22 @@ export class WaveMuxer extends Muxer {
assert(meta.decoderConfig);
this.writeHeader(track, meta.decoderConfig);
this.sampleRate = meta.decoderConfig.sampleRate;
this.headerWritten = true;
}
this.validateAndNormalizeTimestamp(track, packet.timestamp, packet.type === 'key');
if (!this.isRf64 && this.writer.getPos() + packet.data.byteLength >= 2 ** 32) {
throw new Error(
'Adding more audio data would exceed the maximum RIFF size of 4 GiB. To write larger files, use'
+ ' RF64 by setting `large: true` in the WavOutputFormatOptions.',
);
}
this.writer.write(packet.data);
this.dataSize += packet.data.byteLength;
this.sampleCount += Math.round(packet.duration * this.sampleRate!);
await this.writer.flush();
} finally {
@@ -101,10 +114,26 @@ export class WaveMuxer extends Muxer {
const blockSize = pcmInfo.sampleSize * channels;
// RIFF header
this.riffWriter.writeAscii('RIFF');
this.riffWriter.writeU32(0); // File size placeholder
this.riffWriter.writeAscii(this.isRf64 ? 'RF64' : 'RIFF');
if (this.isRf64) {
this.riffWriter.writeU32(0xffffffff); // Not used in RF64
} else {
this.riffWriter.writeU32(0); // File size placeholder
}
this.riffWriter.writeAscii('WAVE');
if (this.isRf64) {
this.riffWriter.writeAscii('ds64');
this.riffWriter.writeU32(28); // Chunk size
this.riffWriter.writeU64(0); // RIFF size placeholder
this.riffWriter.writeU64(0); // Data size placeholder
this.riffWriter.writeU64(0); // Sample count placeholder
this.riffWriter.writeU32(0); // Table length
// Empty table
}
// fmt chunk
this.riffWriter.writeAscii('fmt ');
this.riffWriter.writeU32(16); // Chunk size
@@ -117,7 +146,12 @@ export class WaveMuxer extends Muxer {
// data chunk
this.riffWriter.writeAscii('data');
this.riffWriter.writeU32(0); // Data size placeholder
if (this.isRf64) {
this.riffWriter.writeU32(0xffffffff); // Not used in RF64
} else {
this.riffWriter.writeU32(0); // Data size placeholder
}
if (this.format._options.onHeader) {
const { data, start } = this.writer.stopTrackingWrites();
@@ -130,13 +164,27 @@ export class WaveMuxer extends Muxer {
const endPos = this.writer.getPos();
// Write file size
this.writer.seek(4);
this.riffWriter.writeU32(this.dataSize + 36); // File size - 8
if (this.isRf64) {
// Write riff size
this.writer.seek(20);
this.riffWriter.writeU64(endPos - 8);
// Write data chunk size
this.writer.seek(40);
this.riffWriter.writeU32(this.dataSize);
// Write data size
this.writer.seek(28);
this.riffWriter.writeU64(this.dataSize);
// Write sample count
this.writer.seek(36);
this.riffWriter.writeU64(this.sampleCount);
} else {
// Write file size
this.writer.seek(4);
this.riffWriter.writeU32(endPos - 8);
// Write data chunk size
this.writer.seek(40);
this.riffWriter.writeU32(this.dataSize);
}
this.writer.seek(endPos);
+1 -1
View File
@@ -20,7 +20,7 @@ const rollupInput = Object.fromEntries(
export default defineConfig({
resolve: {
alias: {
mediabunny: path.resolve(__dirname, './dist/mediabunny.mjs'),
mediabunny: path.resolve(__dirname, './dist/bundles/mediabunny.mjs'),
},
},
plugins: [