Compare commits

...
25 Commits
Author SHA1 Message Date
Vanilagy c9633d26d7 Bump minor version 2025-08-18 22:57:17 +02:00
David P.andGitHub 6e802bdf0b Merge pull request #65 from Allwhy/feature/aac-support
Implement ADTS :3
2025-08-18 22:56:00 +02:00
Ally 9054b8e381 Rewrite ADTS muxer to use Bitstream 2025-08-18 22:47:43 +02:00
Vanilagy 8b9f22ba46 eqeqeq 2025-08-18 17:17:18 +02:00
Ally 8da619a624 Implement ADTS demuxer, Fix MP3 demuxer race conditions 2025-08-18 01:29:52 +02:00
Ally 1ce5e108a7 Implement ADTS muxer :3 2025-08-17 16:14:15 +02:00
Vanilagy 0cb466ebb1 Add taf2000 2025-08-15 21:17:21 +02:00
Vanilagy 3be94cc788 Decrease AudioBuffer chunking size, support more MP3 sample rates, lift MP3 encoder memory limit 2025-08-15 14:19:06 +02:00
Vanilagy 89df8fdae2 text-center 2025-08-15 12:00:17 +02:00
Vanilagy 3acb9e8f99 Add sponsor 2025-08-15 11:36:58 +02:00
Vanilagy 716b864fbf Bump patch 2025-08-14 10:56:23 +02:00
David P.andGitHub bea53d9a2b Merge pull request #60 from devPablo/main
Fix UrlSource single-byte buffer on Range Request HTTP 200
2025-08-14 03:12:21 +02:00
Pablo Bonilla 4521a65587 Add descriptive comment on why range request response must be skipped 2025-08-13 19:05:03 -06:00
Pablo Bonilla 5c598e35c2 Skip single-byte buffer when reading through UrlSource for range request 2025-08-13 18:22:59 -06:00
Vanilagy c2292c8304 Bump patch 2025-08-13 17:14:19 +02:00
David P.andGitHub 84219b3ed4 Merge pull request #59 from Yukiniro/e_q6qy_i
fix(conversion): fixed using wrong timestamp
2025-08-13 17:13:38 +02:00
Yukiniro 88b0b5f46f fix(conversion): fix sample end timestamp calculation error 2025-08-13 23:06:11 +08:00
Yukiniro 0348eda8a9 fix(conversion): fixed using wrong timestamp when calculating lastCanvasEndTimestamp 2025-08-13 22:48:08 +08:00
Vanilagy 93cbe69abc Fix remaining type errors, bump version 2025-08-13 09:58:52 +02:00
Vanilagy 16c8a6c255 Gracefully handle invalid EBML headers 2025-08-13 09:46:27 +02:00
Vanilagy 7e546c2cc7 Fix sample closing before it reaches the custom encoder 2025-08-12 15:36:17 +02:00
Vanilagy 7e89511ca6 Bump patch 2025-08-12 10:09:34 +02:00
Vanilagy e0a4169bf8 Handle missing starting key sample in ISOBMFF 2025-08-12 09:29:57 +02:00
Vanilagy c9dcebed6f Skip zero-length AudioSamples 2025-08-12 08:53:25 +02:00
Vanilagy ae0266df52 Clarify decode order thing for encoded packet sources 2025-08-11 18:41:15 +02:00
35 changed files with 1017 additions and 197 deletions
+6 -4
View File
@@ -24,7 +24,9 @@
chunked: true,
chunkSize: 2**20
});
const outputFormat = new Mediabunny.Mp4OutputFormat({});
const outputFormat = new Mediabunny.AdtsOutputFormat({
onFrame: console.log
});
const button = document.createElement('button');
button.textContent = 'Cancel';
@@ -41,7 +43,7 @@
target
}),
audio: {
discard: true,
//discard: true,
//codec: 'opus',
//bitrate: 128000,
//numberOfChannels: 1,
@@ -94,8 +96,8 @@
//height: 100,
}),
trim: {
start: 0,
end: 20
//start: 0,
//end: 20
},
});
console.log(conversion);
+9 -9
View File
@@ -8,15 +8,6 @@
document.body.append(fileInput);
fileInput.addEventListener('change', async () => {
const videoUrl = "https://upload.wikimedia.org/wikipedia/commons/5/53/1941._%D0%9A%D0%BE%D0%BD%D1%91%D0%BA-%D0%B3%D0%BE%D1%80%D0%B1%D1%83%D0%BD%D0%BE%D0%BA.webm"
const source = new Mediabunny.UrlSource(videoUrl)
const input = new Mediabunny.Input({ formats: Mediabunny.ALL_FORMATS, source });
const videoTrack = await input.getPrimaryVideoTrack();
console.log(videoTrack);
/*
const file = fileInput.files[0];
const source = new Mediabunny.BlobSource(file);
@@ -25,6 +16,15 @@
source
});
const videoTrack = await input.getPrimaryVideoTrack();
const sink = new Mediabunny.EncodedPacketSink(videoTrack);
for await (const packet of sink.packets(undefined, undefined, { verifyKeyPackets: false })) {
console.log(packet);
if (packet.timestamp >= 2.4) break;
}
/*
const audioTrack = await input.getPrimaryAudioTrack();
const sink = new Mediabunny.EncodedPacketSink(audioTrack);
+1 -1
View File
@@ -12,7 +12,7 @@ Here's a long list of stuff this library does:
- Converting media files
- Hardware-accelerated decoding & encoding (via the WebCodecs API)
- Support for multiple video, audio and subtitle tracks
- Read & write support for many container formats (.mp4, .mov, .webm, .mkv, .mp3, .wav, .ogg), including variations such as MP4 with Fast Start, fragmented MP4, or streamable Matroska
- Read & write support for many container formats (.mp4, .mov, .webm, .mkv, .mp3, .wav, .ogg, .aac), including variations such as MP4 with Fast Start, fragmented MP4, or streamable Matroska
- Support for 25 different codecs
- Lazy, optimized, on-demand file reading
- Input and output streaming, arbitrary file size support
+2
View File
@@ -123,6 +123,7 @@ const sampleSource = new VideoSampleSource({
});
await sampleSource.add(videoSample);
videoSample.close(); // If it's not needed anymore
// You may optionally force samples to be encoded as key frames:
await sampleSource.add(videoSample, { keyFrame: true });
@@ -285,6 +286,7 @@ const sampleSource = new AudioSampleSource({
});
await sampleSource.add(audioSample);
audioSample.close(); // If it's not needed anymore
```
### `AudioBufferSource`
+22 -1
View File
@@ -244,4 +244,25 @@ type WavOutputFormatOptions = {
- `large`\
When enabled, an RF64 file be written, allowing for file sizes to exceed 4 GiB, which is otherwise not possible for regular WAVE files.
- `onHeader`\
Will be called once the file header is written. The header consists of the RIFF header, the format chunk, and the start of the data chunk (with a placeholder size of 0).
Will be called once the file header is written. The header consists of the RIFF header, the format chunk, and the start of the data chunk (with a placeholder size of 0).
## ADTS
This output format creates ADTS (.aac) files.
```ts
import { Output, AdtsOutputFormat } from 'mediabunny';
const output = new Output({
format: new AdtsOutputFormat(options),
// ...
});
```
The following options are available:
```ts
type AdtsOutputFormatOptions = {
onFrame?: (data: Uint8Array, position: number) => unknown;
};
```
- `onFrame`\
Will be called for each ADTS frame that is written.
+28 -27
View File
@@ -11,6 +11,7 @@ Mediabunny supports many commonly used media container formats, all of which are
- Ogg (.ogg)
- MP3 (.mp3)
- WAVE (.wav)
- ADTS (.aac)
## Codecs
@@ -60,33 +61,33 @@ Mediabunny ships with built-in decoders and encoders for all audio PCM codecs, m
Not all codecs can be used with all containers. The following table specifies the supported codec-container combinations:
| | .mp4 | .mov | .mkv | .webm[^1] | .ogg | .mp3 | .wav |
|:--------------:|:--------:|:-----:|:-----:|:---------:|:-----:|:-----:|:-----:|
| `'avc'` | ✓ | ✓ | ✓ | | | | |
| `'hevc'` | ✓ | ✓ | ✓ | | | | |
| `'vp8'` | ✓ | ✓ | ✓ | ✓ | | | |
| `'vp9'` | ✓ | ✓ | ✓ | ✓ | | | |
| `'av1'` | ✓ | ✓ | ✓ | ✓ | | | |
| `'aac'` | ✓ | ✓ | ✓ | | | | |
| `'opus'` | ✓ | ✓ | ✓ | ✓ | ✓ | | |
| `'mp3'` | ✓ | ✓ | ✓ | | | ✓ | |
| `'vorbis'` | ✓ | ✓ | ✓ | ✓ | ✓ | | |
| `'flac'` | ✓ | ✓ | ✓ | | | | |
| `'pcm-u8'` | | ✓ | ✓ | | | | ✓ |
| `'pcm-s8'` | | ✓ | | | | | |
| `'pcm-s16'` | ✓ | ✓ | ✓ | | | | ✓ |
| `'pcm-s16be'` | ✓ | ✓ | ✓ | | | | |
| `'pcm-s24'` | ✓ | ✓ | ✓ | | | | ✓ |
| `'pcm-s24be'` | ✓ | ✓ | ✓ | | | | |
| `'pcm-s32'` | ✓ | ✓ | ✓ | | | | ✓ |
| `'pcm-s32be'` | ✓ | ✓ | ✓ | | | | |
| `'pcm-f32'` | ✓ | ✓ | ✓ | | | | ✓ |
| `'pcm-f32be'` | ✓ | ✓ | | | | | |
| `'pcm-f64'` | ✓ | ✓ | ✓ | | | | |
| `'pcm-f64be'` | ✓ | ✓ | | | | | |
| `'ulaw'` | | ✓ | | | | | ✓ |
| `'alaw'` | | ✓ | | | | | ✓ |
| `'webvtt'`[^2] | (✓) | | (✓) | (✓) | | | |
| | .mp4 | .mov | .mkv | .webm[^1] | .ogg | .mp3 | .wav | .aac |
|:--------------:|:--------:|:-----:|:-----:|:---------:|:-----:|:-----:|:-----:|:-----:|
| `'avc'` | ✓ | ✓ | ✓ | | | | | |
| `'hevc'` | ✓ | ✓ | ✓ | | | | | |
| `'vp8'` | ✓ | ✓ | ✓ | ✓ | | | | |
| `'vp9'` | ✓ | ✓ | ✓ | ✓ | | | | |
| `'av1'` | ✓ | ✓ | ✓ | ✓ | | | | |
| `'aac'` | ✓ | ✓ | ✓ | | | | | ✓ |
| `'opus'` | ✓ | ✓ | ✓ | ✓ | ✓ | | | |
| `'mp3'` | ✓ | ✓ | ✓ | | | ✓ | | |
| `'vorbis'` | ✓ | ✓ | ✓ | ✓ | ✓ | | | |
| `'flac'` | ✓ | ✓ | ✓ | | | | | |
| `'pcm-u8'` | | ✓ | ✓ | | | | ✓ | |
| `'pcm-s8'` | | ✓ | | | | | | |
| `'pcm-s16'` | ✓ | ✓ | ✓ | | | | ✓ | |
| `'pcm-s16be'` | ✓ | ✓ | ✓ | | | | | |
| `'pcm-s24'` | ✓ | ✓ | ✓ | | | | ✓ | |
| `'pcm-s24be'` | ✓ | ✓ | ✓ | | | | | |
| `'pcm-s32'` | ✓ | ✓ | ✓ | | | | ✓ | |
| `'pcm-s32be'` | ✓ | ✓ | ✓ | | | | | |
| `'pcm-f32'` | ✓ | ✓ | ✓ | | | | ✓ | |
| `'pcm-f32be'` | ✓ | ✓ | | | | | | |
| `'pcm-f64'` | ✓ | ✓ | ✓ | | | | | |
| `'pcm-f64be'` | ✓ | ✓ | | | | | | |
| `'ulaw'` | | ✓ | | | | | ✓ | |
| `'alaw'` | | ✓ | | | | | ✓ | |
| `'webvtt'`[^2] | (✓) | | (✓) | (✓) | | | | |
[^1]: WebM only supports a small subset of the codecs supported by Matroska. However, this library can technically read all codecs from a WebM that are supported by Matroska.
+3 -1
View File
@@ -94,8 +94,10 @@ const sponsors = {
],
individual: [
{ image: 'https://avatars.githubusercontent.com/u/84167135', name: 'Memenome', url: 'https://github.com/memenome' },
{ image: 'https://avatars.githubusercontent.com/u/5913254', name: 'Brandon McConnell', url: 'https://github.com/brandonmcconnell' },
{ image: 'https://avatars.githubusercontent.com/u/9549394', name: 'studnitz', url: 'https://github.com/studnitz' },
{ image: 'https://avatars.githubusercontent.com/u/30229596', name: 'Pablo Bonilla', url: 'https://github.com/devPablo' },
{ image: 'https://avatars.githubusercontent.com/u/63088713', name: 'taf2000', url: 'https://github.com/taf2000' },
{ image: 'https://avatars.githubusercontent.com/u/58149663', name: 'H7GhosT', url: 'https://github.com/H7GhosT' },
{ image: 'https://avatars.githubusercontent.com/u/91711202', name: 'ihasq', url: 'https://github.com/ihasq' },
{ image: 'https://avatars.githubusercontent.com/u/61233224', name: 'Allwhy', url: 'https://github.com/Allwhy' },
@@ -339,7 +341,7 @@ await conversion.execute();
<div class="flex flex-wrap mt-1 justify-center">
<a v-for="sponsor in sponsors.individual" :href="sponsor.url" target="_blank" class="flex gap-1 w-24 flex-col items-center p-2 rounded-xl hover:bg-(--vp-c-gray-3) !text-(--vp-c-text-1) !no-underline">
<img :src="sponsor.image" class="size-8 rounded-full">
<p class="!my-0 !font-medium text-xs !leading-4">{{ sponsor.name }}</p>
<p class="!my-0 !font-medium text-xs !leading-4 text-center">{{ sponsor.name }}</p>
</a>
</div>
</template>
+2 -1
View File
@@ -25,10 +25,11 @@ export default tseslint.config(
code: 120,
}],
'curly': ['error', 'multi-line'],
'eqeqeq': ['error', 'always', { null: 'ignore' }],
'@typescript-eslint/no-empty-object-type': 'off',
'@typescript-eslint/require-await': 'off',
'@stylistic/yield-star-spacing': ['error', { before: false, after: true }],
'@typescript-eslint/no-unsafe-enum-comparison': 'off',
'@typescript-eslint/no-unsafe-enum-comparison': 'off',
},
},
{
@@ -117,7 +117,7 @@ const compressFile = async (file: File) => {
selectMediaButton.addEventListener('click', () => {
const fileInput = document.createElement('input');
fileInput.type = 'file';
fileInput.accept = 'video/*,video/x-matroska,audio/*';
fileInput.accept = 'video/*,video/x-matroska,audio/*,audio/aac';
fileInput.addEventListener('change', () => {
const file = fileInput.files?.[0];
if (!file) {
+1 -1
View File
@@ -581,7 +581,7 @@ const formatSeconds = (seconds: number) => {
selectMediaButton.addEventListener('click', () => {
const fileInput = document.createElement('input');
fileInput.type = 'file';
fileInput.accept = 'video/*,video/x-matroska,audio/*';
fileInput.accept = 'video/*,video/x-matroska,audio/*,audio/aac';
fileInput.addEventListener('change', () => {
const file = fileInput.files?.[0];
if (!file) {
@@ -157,7 +157,7 @@ const shortDelay = () => {
selectMediaButton.addEventListener('click', () => {
const fileInput = document.createElement('input');
fileInput.type = 'file';
fileInput.accept = 'video/*,video/x-matroska,audio/*';
fileInput.accept = 'video/*,video/x-matroska,audio/*,audio/aac';
fileInput.addEventListener('change', () => {
const file = fileInput.files?.[0];
if (!file) {
@@ -109,7 +109,7 @@ const generateThumbnails = async (file: File) => {
selectMediaButton.addEventListener('click', () => {
const fileInput = document.createElement('input');
fileInput.type = 'file';
fileInput.accept = 'video/*,video/x-matroska,audio/*';
fileInput.accept = 'video/*,video/x-matroska,audio/*,audio/aac';
fileInput.addEventListener('change', () => {
const file = fileInput.files?.[0];
if (!file) {
+6 -6
View File
@@ -1,12 +1,12 @@
{
"name": "mediabunny",
"version": "1.7.1",
"version": "1.9.0",
"lockfileVersion": 3,
"requires": true,
"packages": {
"": {
"name": "mediabunny",
"version": "1.7.1",
"version": "1.9.0",
"license": "MPL-2.0",
"workspaces": [
"packages/*"
@@ -5900,9 +5900,9 @@
}
},
"node_modules/mediabunny": {
"version": "1.7.0",
"resolved": "https://registry.npmjs.org/mediabunny/-/mediabunny-1.7.0.tgz",
"integrity": "sha512-QcTdptOtvjAHb4KQpgWWWHRS0+DgGwTXelTwPWaZwQu85iYy11bzsLS4e29rfH4nLL/eq8K/dBsjj1LuApc2Qw==",
"version": "1.8.0",
"resolved": "https://registry.npmjs.org/mediabunny/-/mediabunny-1.8.0.tgz",
"integrity": "sha512-+JjE+A2NxtjxnBsEy4yCw9CYx8yDyJzivAx28xU5vWqEy4Xu9gE8qiIklKWfLEAoiyudJ+wn0gn3rl1HtK8vcw==",
"license": "MPL-2.0",
"peer": true,
"workspaces": [
@@ -9017,7 +9017,7 @@
},
"packages/mp3-encoder": {
"name": "@mediabunny/mp3-encoder",
"version": "1.7.1",
"version": "1.9.0",
"license": "MPL-2.0",
"devDependencies": {
"@types/emscripten": "^1.40.1"
+1 -1
View File
@@ -1,7 +1,7 @@
{
"name": "mediabunny",
"author": "Vanilagy",
"version": "1.7.1",
"version": "1.9.0",
"description": "Pure TypeScript media toolkit for reading, writing, and converting media files, directly in the browser.",
"type": "module",
"workspaces": [
+2 -1
View File
@@ -108,7 +108,7 @@ For simplicity, all built WASM artifacts are included in the repo, since these r
### Prerequisites
[Install Emscripten](https://emscripten.org/docs/getting_started/downloads.html). The recommended way is using the emsdk, which involves cloning a repo and running a few commands.
[Install Emscripten](https://emscripten.org/docs/getting_started/downloads.html). The recommended way is using the emsdk, which involves cloning a repo and running a few commands. The following commands assume Emscripten is sourced in.
### Compiling LAME:
@@ -139,6 +139,7 @@ emcc src/lame-bridge.c build/libmp3lame.a \
-s MODULARIZE=1 \
-s EXPORT_ES6=1 \
-s SINGLE_FILE=1 \
-s ALLOW_MEMORY_GROWTH=1 \
-s ENVIRONMENT=web,worker \
-s EXPORTED_RUNTIME_METHODS=cwrap,HEAPU8 \
-s EXPORTED_FUNCTIONS=_malloc,_free \
+1 -1
View File
File diff suppressed because one or more lines are too long
+1 -1
View File
@@ -1,7 +1,7 @@
{
"name": "@mediabunny/mp3-encoder",
"author": "Vanilagy",
"version": "1.7.1",
"version": "1.9.0",
"description": "MP3 encoder extension for Mediabunny, based on LAME.",
"main": "./dist/bundles/mediabunny-mp3-encoder.mjs",
"module": "./dist/bundles/mediabunny-mp3-encoder.mjs",
+2 -2
View File
@@ -7,7 +7,7 @@
*/
import { CustomAudioEncoder, AudioCodec, AudioSample, EncodedPacket, registerEncoder } from 'mediabunny';
import { FRAME_HEADER_SIZE, readFrameHeader } from '../../../shared/mp3-misc';
import { FRAME_HEADER_SIZE, readFrameHeader, SAMPLING_RATES } from '../../../shared/mp3-misc';
import type { WorkerCommand, WorkerResponse, WorkerResponseData } from './shared';
// @ts-expect-error An esbuild plugin handles this, TypeScript doesn't need to understand
import createWorker from './encode.worker';
@@ -28,7 +28,7 @@ class Mp3Encoder extends CustomAudioEncoder {
static override supports(codec: AudioCodec, config: AudioDecoderConfig): boolean {
return codec === 'mp3'
&& (config.numberOfChannels === 1 || config.numberOfChannels === 2)
&& (config.sampleRate === 32000 || config.sampleRate === 44100 || config.sampleRate === 48000);
&& Object.values(SAMPLING_RATES).some(x => x.includes(config.sampleRate));
}
async init() {
+312
View File
@@ -0,0 +1,312 @@
/*!
* Copyright (c) 2025-present, Vanilagy and contributors
*
* This Source Code Form is subject to the terms of the Mozilla Public
* License, v. 2.0. If a copy of the MPL was not distributed with this
* file, You can obtain one at https://mozilla.org/MPL/2.0/.
*/
import { aacChannelMap, aacFrequencyTable, AudioCodec } from '../codec';
import { Demuxer } from '../demuxer';
import { Input } from '../input';
import { InputAudioTrack, InputAudioTrackBacking } from '../input-track';
import { PacketRetrievalOptions } from '../media-sink';
import {
assert,
AsyncMutex,
binarySearchExact,
binarySearchLessOrEqual,
Bitstream,
UNDETERMINED_LANGUAGE,
} from '../misc';
import { EncodedPacket, PLACEHOLDER_DATA } from '../packet';
import { AdtsReader, FrameHeader, MAX_FRAME_HEADER_SIZE } from './adts-reader';
const SAMPLES_PER_AAC_FRAME = 1024;
type Sample = {
timestamp: number;
duration: number;
dataStart: number;
dataSize: number;
};
export class AdtsDemuxer extends Demuxer {
reader: AdtsReader;
metadataPromise: Promise<void> | null = null;
firstFrameHeader: FrameHeader | null = null;
loadedSamples: Sample[] = []; // All samples from the start of the file to lastLoadedPos
tracks: InputAudioTrack[] = [];
readingMutex = new AsyncMutex();
lastLoadedPos = 0;
fileSize = 0;
nextTimestampInSamples = 0;
constructor(input: Input) {
super(input);
this.reader = new AdtsReader(input._mainReader);
}
async readMetadata() {
return this.metadataPromise ??= (async () => {
this.fileSize = await this.input.source.getSize();
await this.loadNextChunk();
// There has to be a frame if this demuxer got selected
assert(this.firstFrameHeader);
// Create the single audio track
this.tracks = [new InputAudioTrack(new AdtsAudioTrackBacking(this))];
})();
}
async loadNextChunk() {
assert(this.lastLoadedPos < this.fileSize);
const chunkSize = 0.5 * 1024 * 1024; // 0.5 MiB
const endPos = Math.min(this.lastLoadedPos + chunkSize, this.fileSize);
await this.reader.reader.loadRange(this.lastLoadedPos, endPos);
this.lastLoadedPos = endPos;
assert(this.lastLoadedPos <= this.fileSize);
this.parseFramesFromLoadedData();
}
private parseFramesFromLoadedData() {
while (this.reader.pos <= this.fileSize - MAX_FRAME_HEADER_SIZE) {
const startPos = this.reader.pos;
const header = this.reader.readFrameHeader();
if (!header) {
break;
}
// Check if the entire frame fits in the loaded data
if (startPos + header.frameLength > this.lastLoadedPos) {
// Frame doesn't fit, reset positions and stop
this.reader.pos = startPos;
this.lastLoadedPos = startPos;
break;
}
if (!this.firstFrameHeader) {
this.firstFrameHeader = header;
}
const sampleRate = aacFrequencyTable[header.samplingFrequencyIndex];
assert(sampleRate !== undefined);
const sampleDuration = SAMPLES_PER_AAC_FRAME / sampleRate;
const headerSize = header.crcCheck ? MAX_FRAME_HEADER_SIZE : MAX_FRAME_HEADER_SIZE - 2;
const sample: Sample = {
timestamp: this.nextTimestampInSamples / sampleRate,
duration: sampleDuration,
dataStart: startPos + headerSize,
dataSize: header.frameLength - headerSize,
};
this.loadedSamples.push(sample);
this.nextTimestampInSamples += SAMPLES_PER_AAC_FRAME;
this.reader.pos = startPos + header.frameLength;
}
}
async getMimeType() {
return 'audio/aac';
}
async getTracks() {
await this.readMetadata();
return this.tracks;
}
async computeDuration() {
await this.readMetadata();
const track = this.tracks[0];
assert(track);
return track.computeDuration();
}
}
class AdtsAudioTrackBacking implements InputAudioTrackBacking {
constructor(public demuxer: AdtsDemuxer) {}
getId() {
return 1;
}
async getFirstTimestamp() {
return 0;
}
getTimeResolution() {
const sampleRate = this.getSampleRate();
return sampleRate / SAMPLES_PER_AAC_FRAME;
}
async computeDuration() {
const lastPacket = await this.getPacket(Infinity, { metadataOnly: true });
return (lastPacket?.timestamp ?? 0) + (lastPacket?.duration ?? 0);
}
getLanguageCode() {
return UNDETERMINED_LANGUAGE;
}
getCodec(): AudioCodec {
return 'aac';
}
getNumberOfChannels() {
assert(this.demuxer.firstFrameHeader);
const numberOfChannels = aacChannelMap[this.demuxer.firstFrameHeader.channelConfiguration];
assert(numberOfChannels !== undefined);
return numberOfChannels;
}
getSampleRate() {
assert(this.demuxer.firstFrameHeader);
const sampleRate = aacFrequencyTable[this.demuxer.firstFrameHeader.samplingFrequencyIndex];
assert(sampleRate !== undefined);
return sampleRate;
}
async getDecoderConfig(): Promise<AudioDecoderConfig> {
assert(this.demuxer.firstFrameHeader);
const bytes = new Uint8Array(3); // 19 bits max
const bitstream = new Bitstream(bytes);
const { objectType, samplingFrequencyIndex, channelConfiguration } = this.demuxer.firstFrameHeader;
if (objectType > 31) {
bitstream.writeBits(5, 31);
bitstream.writeBits(6, objectType - 32);
} else {
bitstream.writeBits(5, objectType);
}
bitstream.writeBits(4, samplingFrequencyIndex); // samplingFrequencyIndex === 15 is forbidden
bitstream.writeBits(4, channelConfiguration);
return {
codec: `mp4a.40.${this.demuxer.firstFrameHeader.objectType}`,
numberOfChannels: this.getNumberOfChannels(),
sampleRate: this.getSampleRate(),
description: bytes.subarray(0, Math.ceil((bitstream.pos - 1) / 8)),
};
}
getPacketAtIndex(sampleIndex: number, options: PacketRetrievalOptions) {
if (sampleIndex === -1) {
return null;
}
const rawSample = this.demuxer.loadedSamples[sampleIndex];
if (!rawSample) {
return null;
}
let data: Uint8Array;
if (options.metadataOnly) {
data = PLACEHOLDER_DATA;
} else {
this.demuxer.reader.pos = rawSample.dataStart;
data = this.demuxer.reader.readBytes(rawSample.dataSize);
}
return new EncodedPacket(
data,
'key',
rawSample.timestamp,
rawSample.duration,
sampleIndex,
rawSample.dataSize,
);
}
async getFirstPacket(options: PacketRetrievalOptions) {
return this.getPacketAtIndex(0, options);
}
async getNextPacket(packet: EncodedPacket, options: PacketRetrievalOptions) {
const release = await this.demuxer.readingMutex.acquire();
try {
const sampleIndex = binarySearchExact(
this.demuxer.loadedSamples,
packet.timestamp,
x => x.timestamp,
);
if (sampleIndex === -1) {
throw new Error('Packet was not created from this track.');
}
const nextIndex = sampleIndex + 1;
// Ensure the next sample exists
while (
nextIndex >= this.demuxer.loadedSamples.length
&& this.demuxer.lastLoadedPos < this.demuxer.fileSize
) {
await this.demuxer.loadNextChunk();
}
return this.getPacketAtIndex(nextIndex, options);
} finally {
release();
}
}
async getPacket(timestamp: number, options: PacketRetrievalOptions) {
const release = await this.demuxer.readingMutex.acquire();
try {
while (true) {
const index = binarySearchLessOrEqual(
this.demuxer.loadedSamples,
timestamp,
x => x.timestamp,
);
if (index === -1 && this.demuxer.loadedSamples.length > 0) {
// We're before the first sample
return null;
}
if (this.demuxer.lastLoadedPos === this.demuxer.fileSize) {
// All data is loaded, return what we found
return this.getPacketAtIndex(index, options);
}
if (index >= 0 && index + 1 < this.demuxer.loadedSamples.length) {
// The next packet also exists, we're done
return this.getPacketAtIndex(index, options);
}
// Otherwise, keep loading data
await this.demuxer.loadNextChunk();
}
} finally {
release();
}
}
getKeyPacket(timestamp: number, options: PacketRetrievalOptions) {
return this.getPacket(timestamp, options);
}
getNextKeyPacket(packet: EncodedPacket, options: PacketRetrievalOptions) {
return this.getNextPacket(packet, options);
}
}
+109
View File
@@ -0,0 +1,109 @@
/*!
* Copyright (c) 2025-present, Vanilagy and contributors
*
* This Source Code Form is subject to the terms of the Mozilla Public
* License, v. 2.0. If a copy of the MPL was not distributed with this
* file, You can obtain one at https://mozilla.org/MPL/2.0/.
*/
import { AacAudioSpecificConfig, parseAacAudioSpecificConfig, validateAudioChunkMetadata } from '../codec';
import { assert, Bitstream, toUint8Array } from '../misc';
import { Muxer } from '../muxer';
import { Output, OutputAudioTrack } from '../output';
import { AdtsOutputFormat } from '../output-format';
import { EncodedPacket } from '../packet';
import { Writer } from '../writer';
export class AdtsMuxer extends Muxer {
private format: AdtsOutputFormat;
private writer: Writer;
private header = new Uint8Array(7);
private headerBitstream = new Bitstream(this.header);
private audioSpecificConfig: AacAudioSpecificConfig | null = null;
constructor(output: Output, format: AdtsOutputFormat) {
super(output);
this.format = format;
this.writer = output._writer;
}
async start() {
// Nothing needed here
}
async getMimeType() {
return 'audio/aac';
}
async addEncodedVideoPacket() {
throw new Error('ADTS does not support video.');
}
async addEncodedAudioPacket(
track: OutputAudioTrack,
packet: EncodedPacket,
meta?: EncodedAudioChunkMetadata,
) {
// https://wiki.multimedia.cx/index.php/ADTS (last visited: 2025/08/17)
const release = await this.mutex.acquire();
try {
if (!this.audioSpecificConfig) {
validateAudioChunkMetadata(meta);
const description = meta?.decoderConfig?.description;
assert(description);
this.audioSpecificConfig = parseAacAudioSpecificConfig(toUint8Array(description));
const { objectType, frequencyIndex, channelConfiguration } = this.audioSpecificConfig;
const profile = objectType - 1;
this.headerBitstream.writeBits(12, 0b1111_11111111); // Syncword
this.headerBitstream.writeBits(1, 0); // MPEG Version
this.headerBitstream.writeBits(2, 0); // Layer
this.headerBitstream.writeBits(1, 1); // Protection absence
this.headerBitstream.writeBits(2, profile); // Profile
this.headerBitstream.writeBits(4, frequencyIndex); // MPEG-4 Sampling Frequency Index
this.headerBitstream.writeBits(1, 0); // Private bit
this.headerBitstream.writeBits(3, channelConfiguration); // MPEG-4 Channel Configuration
this.headerBitstream.writeBits(1, 0); // Originality
this.headerBitstream.writeBits(1, 0); // Home
this.headerBitstream.writeBits(1, 0); // Copyright ID bit
this.headerBitstream.writeBits(1, 0); // Copyright ID start
this.headerBitstream.skipBits(13); // Frame length
this.headerBitstream.writeBits(11, 0x7ff); // Buffer fullness
this.headerBitstream.writeBits(2, 0); // Number of AAC frames minus 1
// Omit CRC check
}
const frameLength = packet.data.byteLength + this.header.byteLength;
this.headerBitstream.pos = 30;
this.headerBitstream.writeBits(13, frameLength);
const startPos = this.writer.getPos();
this.writer.write(this.header);
this.writer.write(packet.data);
if (this.format._options.onFrame) {
const frameBytes = new Uint8Array(frameLength);
frameBytes.set(this.header, 0);
frameBytes.set(packet.data, this.header.byteLength);
this.format._options.onFrame(frameBytes, startPos);
}
await this.writer.flush();
} finally {
release();
}
}
async addSubtitleCue() {
throw new Error('ADTS does not support subtitles.');
}
async finalize() {}
}
+96
View File
@@ -0,0 +1,96 @@
/*!
* Copyright (c) 2025-present, Vanilagy and contributors
*
* This Source Code Form is subject to the terms of the Mozilla Public
* License, v. 2.0. If a copy of the MPL was not distributed with this
* file, You can obtain one at https://mozilla.org/MPL/2.0/.
*/
import { Bitstream } from '../misc';
import { Reader } from '../reader';
export const MAX_FRAME_HEADER_SIZE = 9;
export type FrameHeader = {
objectType: number;
samplingFrequencyIndex: number;
channelConfiguration: number;
frameLength: number;
numberOfAacFrames: number;
crcCheck: number | null;
startPos: number;
};
export class AdtsReader {
pos = 0;
constructor(public reader: Reader) {}
readBytes(length: number) {
const { view, offset } = this.reader.getViewAndOffset(this.pos, this.pos + length);
this.pos += length;
return new Uint8Array(view.buffer, offset, length);
}
readFrameHeader(): FrameHeader | null {
// https://wiki.multimedia.cx/index.php/ADTS (last visited: 2025/08/17)
const startPos = this.pos;
const bytes = this.readBytes(9); // 9 with CRC, 7 without CRC
const bitstream = new Bitstream(bytes);
const syncword = bitstream.readBits(12);
if (syncword !== 0b1111_11111111) {
return null;
}
bitstream.skipBits(1); // MPEG version
const layer = bitstream.readBits(2);
if (layer !== 0) {
return null;
}
const protectionAbsence = bitstream.readBits(1);
const objectType = bitstream.readBits(2) + 1;
const samplingFrequencyIndex = bitstream.readBits(4);
if (samplingFrequencyIndex === 15) {
return null;
}
bitstream.skipBits(1); // Private bit
const channelConfiguration = bitstream.readBits(3);
if (channelConfiguration === 0) {
throw new Error('ADTS frames with channel configuration 0 are not supported.');
}
bitstream.skipBits(1); // Originality
bitstream.skipBits(1); // Home
bitstream.skipBits(1); // Copyright ID bit
bitstream.skipBits(1); // Copyright ID start
const frameLength = bitstream.readBits(13);
bitstream.skipBits(11); // Buffer fullness
const numberOfAacFrames = bitstream.readBits(2) + 1;
if (numberOfAacFrames !== 1) {
throw new Error('ADTS frames with more than one AAC frame are not supported.');
}
let crcCheck: number | null = null;
if (protectionAbsence === 1) { // No CRC
this.pos -= 2;
} else { // CRC
crcCheck = bitstream.readBits(16);
}
return {
objectType,
samplingFrequencyIndex,
channelConfiguration,
frameLength,
numberOfAacFrames,
crcCheck,
startPos,
};
}
}
+19 -17
View File
@@ -561,7 +561,22 @@ export const extractAudioCodecString = (trackInfo: {
throw new TypeError(`Unhandled codec '${codec}'.`);
};
export const parseAacAudioSpecificConfig = (bytes: Uint8Array | null) => {
export type AacAudioSpecificConfig = {
objectType: number;
frequencyIndex: number;
sampleRate: number | null;
channelConfiguration: number;
numberOfChannels: number | null;
};
export const aacFrequencyTable = [
96000, 88200, 64000, 48000, 44100, 32000,
24000, 22050, 16000, 12000, 11025, 8000, 7350,
];
export const aacChannelMap = [-1, 1, 2, 3, 4, 5, 6, 8];
export const parseAacAudioSpecificConfig = (bytes: Uint8Array | null): AacAudioSpecificConfig => {
if (!bytes || bytes.byteLength < 2) {
throw new TypeError('AAC description must be at least 2 bytes long.');
}
@@ -578,28 +593,15 @@ export const parseAacAudioSpecificConfig = (bytes: Uint8Array | null) => {
if (frequencyIndex === 15) {
sampleRate = bitstream.readBits(24);
} else {
const freqTable = [
96000, 88200, 64000, 48000, 44100, 32000, 24000, 22050,
16000, 12000, 11025, 8000, 7350,
];
if (frequencyIndex < freqTable.length) {
sampleRate = freqTable[frequencyIndex]!;
if (frequencyIndex < aacFrequencyTable.length) {
sampleRate = aacFrequencyTable[frequencyIndex]!;
}
}
const channelConfiguration = bitstream.readBits(4);
let numberOfChannels: number | null = null;
if (channelConfiguration >= 1 && channelConfiguration <= 7) {
const channelMap = {
1: 1,
2: 2,
3: 3,
4: 4,
5: 5,
6: 6,
7: 8,
};
numberOfChannels = channelMap[channelConfiguration as keyof typeof channelMap];
numberOfChannels = aacChannelMap[channelConfiguration]!;
}
return {
+2 -2
View File
@@ -671,7 +671,7 @@ export class Conversion {
}
let adjustedSampleTimestamp = Math.max(timestamp - this._startTimestamp, 0);
lastCanvasEndTimestamp = timestamp + duration;
lastCanvasEndTimestamp = adjustedSampleTimestamp + duration;
if (frameRate !== undefined) {
// Logic for skipping/repeating frames when a frame rate is set
@@ -757,7 +757,7 @@ export class Conversion {
}
let adjustedSampleTimestamp = Math.max(sample.timestamp - this._startTimestamp, 0);
lastSampleEndTimestamp = sample.timestamp + sample.duration;
lastSampleEndTimestamp = adjustedSampleTimestamp + sample.duration;
if (frameRate !== undefined) {
// Logic for skipping/repeating frames when a frame rate is set
+2
View File
@@ -35,6 +35,8 @@ export {
WavOutputFormatOptions,
OggOutputFormat,
OggOutputFormatOptions,
AdtsOutputFormat,
AdtsOutputFormatOptions,
TrackCountLimits,
InclusiveIntegerRange,
} from './output-format';
+72 -4
View File
@@ -10,7 +10,7 @@ import { Demuxer } from './demuxer';
import { Input } from './input';
import { IsobmffDemuxer } from './isobmff/isobmff-demuxer';
import { IsobmffReader } from './isobmff/isobmff-reader';
import { EBMLId, EBMLReader } from './matroska/ebml';
import { EBMLId, EBMLReader, MIN_HEADER_SIZE } from './matroska/ebml';
import { MatroskaDemuxer } from './matroska/matroska-demuxer';
import { Mp3Demuxer } from './mp3/mp3-demuxer';
import { FRAME_HEADER_SIZE } from '../shared/mp3-misc';
@@ -19,6 +19,8 @@ import { OggDemuxer } from './ogg/ogg-demuxer';
import { OggReader } from './ogg/ogg-reader';
import { RiffReader } from './wave/riff-reader';
import { WaveDemuxer } from './wave/wave-demuxer';
import { AdtsReader, MAX_FRAME_HEADER_SIZE } from './adts/adts-reader';
import { AdtsDemuxer } from './adts/adts-demuxer';
/**
* Base class representing an input media file format.
@@ -106,6 +108,10 @@ export class QuickTimeInputFormat extends IsobmffInputFormat {
}
}
function foo() {
return 5;
}
/**
* Matroska file format.
* @public
@@ -120,6 +126,12 @@ export class MatroskaInputFormat extends InputFormat {
const ebmlReader = new EBMLReader(input._mainReader);
const varIntSize = ebmlReader.readVarIntSize();
if (varIntSize === null) {
return false;
}
foo();
if (varIntSize < 1 || varIntSize > 8) {
return false;
}
@@ -135,8 +147,11 @@ export class MatroskaInputFormat extends InputFormat {
}
const startPos = ebmlReader.pos;
while (ebmlReader.pos < startPos + dataSize) {
const { id, size } = ebmlReader.readElementHeader();
while (ebmlReader.pos <= startPos + dataSize - MIN_HEADER_SIZE) {
const header = ebmlReader.readElementHeader();
if (!header) break;
const { id, size } = header;
const dataStartPos = ebmlReader.pos;
if (size === null) return false;
@@ -344,6 +359,54 @@ export class OggInputFormat extends InputFormat {
}
}
/**
* ADTS file format.
* @public
*/
export class AdtsInputFormat extends InputFormat {
/** @internal */
async _canReadInput(input: Input) {
const sourceSize = await input._mainReader.source.getSize();
if (sourceSize < MAX_FRAME_HEADER_SIZE) {
return false;
}
const adtsReader = new AdtsReader(input._mainReader);
const firstHeader = adtsReader.readFrameHeader();
if (!firstHeader) {
return false;
}
if (sourceSize < firstHeader.frameLength + MAX_FRAME_HEADER_SIZE) {
return false;
}
adtsReader.pos = firstHeader.frameLength;
await adtsReader.reader.loadRange(adtsReader.pos, adtsReader.pos + MAX_FRAME_HEADER_SIZE);
const secondHeader = adtsReader.readFrameHeader();
if (!secondHeader) {
return false;
}
return firstHeader.objectType === secondHeader.objectType
&& firstHeader.samplingFrequencyIndex === secondHeader.samplingFrequencyIndex
&& firstHeader.channelConfiguration === secondHeader.channelConfiguration;
}
/** @internal */
_createDemuxer(input: Input) {
return new AdtsDemuxer(input);
}
get name() {
return 'ADTS';
}
get mimeType() {
return 'audio/aac';
}
}
/**
* MP4 input format singleton.
* @public
@@ -379,10 +442,15 @@ export const WAVE = new WaveInputFormat();
* @public
*/
export const OGG = new OggInputFormat();
/**
* ADTS input format singleton.
* @public
*/
export const ADTS = new AdtsInputFormat();
/**
* List of all input format singletons. If you don't need to support all input formats, you should specify the
* formats individually for better tree shaking.
* @public
*/
export const ALL_FORMATS: InputFormat[] = [MP4, QTFF, MATROSKA, WEBM, WAVE, OGG, MP3];
export const ALL_FORMATS: InputFormat[] = [MP4, QTFF, MATROSKA, WEBM, WAVE, OGG, MP3, ADTS];
+7 -1
View File
@@ -1046,7 +1046,7 @@ export class IsobmffDemuxer extends Demuxer {
const chromaSamplePosition = thirdByte & 0b11;
// Logic from https://aomediacodec.github.io/av1-spec/av1-spec.pdf
const bitDepth = profile == 2 && highBitDepth ? (twelveBit ? 12 : 10) : (highBitDepth ? 10 : 8);
const bitDepth = profile === 2 && highBitDepth ? (twelveBit ? 12 : 10) : (highBitDepth ? 10 : 8);
track.info.av1CodecInfo = {
profile,
@@ -1459,6 +1459,12 @@ export class IsobmffDemuxer extends Demuxer {
const sampleIndex = this.metadataReader.readU32() - 1; // Convert to 0-indexed
track.sampleTable.keySampleIndices.push(sampleIndex);
}
if (track.sampleTable.keySampleIndices[0] !== 0) {
// Some files don't mark the first sample a key sample, which is basically almost always incorrect.
// Here, we correct for that mistake:
track.sampleTable.keySampleIndices.unshift(0);
}
}; break;
case 'stsc': {
+25 -6
View File
@@ -379,7 +379,7 @@ export class EBMLWriter {
const MAX_VAR_INT_SIZE = 8;
export const MIN_HEADER_SIZE = 2; // 1-byte ID and 1-byte size
export const MAX_HEADER_SIZE = 4 + MAX_VAR_INT_SIZE; // 4-byte ID and 8-byte size
export const MAX_HEADER_SIZE = 2 * MAX_VAR_INT_SIZE; // 8-byte ID and 8-byte size
export class EBMLReader {
pos = 0;
@@ -411,9 +411,13 @@ export class EBMLReader {
const { view, offset } = this.reader.getViewAndOffset(this.pos, this.pos + 1);
const firstByte = view.getUint8(offset);
if (firstByte === 0) {
return null; // Invalid VINT
}
let width = 1;
let mask = 0x80;
while ((firstByte & mask) === 0 && width < 8) {
while ((firstByte & mask) === 0) {
width++;
mask >>= 1;
}
@@ -426,10 +430,14 @@ export class EBMLReader {
const { view, offset } = this.reader.getViewAndOffset(this.pos, this.pos + 1);
const firstByte = view.getUint8(offset);
if (firstByte === 0) {
return null; // Invalid VINT
}
// Find the position of VINT_MARKER, which determines the width
let width = 1;
let mask = 1 << 7;
while ((firstByte & mask) === 0 && width < MAX_VAR_INT_SIZE) {
while ((firstByte & mask) === 0) {
width++;
mask >>= 1;
}
@@ -509,8 +517,11 @@ export class EBMLReader {
readElementId() {
const size = this.readVarIntSize();
const id = this.readUnsignedInt(size);
if (size === null) {
return null;
}
const id = this.readUnsignedInt(size);
return id;
}
@@ -538,6 +549,10 @@ export class EBMLReader {
readElementHeader() {
const id = this.readElementId();
if (id === null) {
return null;
}
const size = this.readElementSize();
return { id, size };
@@ -548,13 +563,17 @@ export class EBMLReader {
const loadChunkSize = 2 ** 20; // 1 MiB
const idsSet = new Set(ids);
while (this.pos < until - MAX_HEADER_SIZE) {
if (!this.reader.rangeIsLoaded(this.pos, this.pos + MAX_HEADER_SIZE)) {
while (this.pos <= until - MIN_HEADER_SIZE) {
if (!this.reader.rangeIsLoaded(this.pos, Math.min(this.pos + MAX_HEADER_SIZE, until))) {
await this.reader.loadRange(this.pos, Math.min(this.pos + loadChunkSize, until));
}
const elementStartPos = this.pos;
const elementHeader = this.readElementHeader();
if (!elementHeader) {
break;
}
if (idsSet.has(elementHeader.id)) {
return elementStartPos;
}
+38 -7
View File
@@ -239,6 +239,10 @@ export class MatroskaDemuxer extends Demuxer {
);
const header = this.metadataReader.readElementHeader();
if (!header) {
break; // Zero padding at the end of the file triggers this, for example
}
const id = header.id;
let size = header.size;
const startPos = this.metadataReader.pos;
@@ -318,14 +322,19 @@ export class MatroskaDemuxer extends Demuxer {
);
let clusterEncountered = false;
while (this.metadataReader.pos < this.currentSegment.elementEndPos) {
while (this.metadataReader.pos <= this.currentSegment.elementEndPos - MIN_HEADER_SIZE) {
await this.metadataReader.reader.loadRange(
this.metadataReader.pos,
this.metadataReader.pos + MAX_HEADER_SIZE,
);
const elementStartPos = this.metadataReader.pos;
const { id, size } = this.metadataReader.readElementHeader();
const header = this.metadataReader.readElementHeader();
if (!header) {
break;
}
const { id, size } = header;
const dataStartPos = this.metadataReader.pos;
const metadataElementIndex = METADATA_ELEMENTS.findIndex(x => x.id === id);
@@ -392,7 +401,10 @@ export class MatroskaDemuxer extends Demuxer {
this.metadataReader.pos,
this.metadataReader.pos + 2 ** 12, // Load a larger range, assuming the correct element will be there
);
const { id, size } = this.metadataReader.readElementHeader();
const header = this.metadataReader.readElementHeader();
if (!header) continue;
const { id, size } = header;
if (id !== target.id) continue;
assertDefinedSize(size);
@@ -469,6 +481,8 @@ export class MatroskaDemuxer extends Demuxer {
const elementStartPos = this.metadataReader.pos;
const elementHeader = this.metadataReader.readElementHeader();
assert(elementHeader);
const id = elementHeader.id;
let size = elementHeader.size;
const dataStartPos = this.metadataReader.pos;
@@ -720,12 +734,20 @@ export class MatroskaDemuxer extends Demuxer {
const startIndex = reader.pos;
while (reader.pos - startIndex <= totalSize - MIN_HEADER_SIZE) {
this.traverseElement(reader);
const foundElement = this.traverseElement(reader);
if (!foundElement) {
break;
}
}
}
traverseElement(reader: EBMLReader) {
const { id, size } = reader.readElementHeader();
traverseElement(reader: EBMLReader): boolean {
const header = reader.readElementHeader();
if (!header) {
return false;
}
const { id, size } = header;
const dataStartPos = reader.pos;
assertDefinedSize(size);
@@ -1113,6 +1135,8 @@ export class MatroskaDemuxer extends Demuxer {
if (!this.currentCluster) break;
const trackNumber = reader.readVarInt();
if (trackNumber === null) break;
const relativeTimestamp = reader.readS16();
const flags = reader.readU8();
@@ -1148,6 +1172,8 @@ export class MatroskaDemuxer extends Demuxer {
if (!this.currentCluster) break;
const trackNumber = reader.readVarInt();
if (trackNumber === null) break;
const relativeTimestamp = reader.readS16();
const flags = reader.readU8();
@@ -1184,6 +1210,7 @@ export class MatroskaDemuxer extends Demuxer {
}
reader.pos = dataStartPos + size;
return true;
}
}
@@ -1562,7 +1589,7 @@ abstract class MatroskaTrackBacking implements InputTrackBacking {
}
}
while (metadataReader.pos < segment.elementEndPos) {
while (metadataReader.pos <= segment.elementEndPos - MIN_HEADER_SIZE) {
if (prevCluster) {
const trackData = prevCluster.trackData.get(this.internalTrack.id);
if (trackData && trackData.startTimestamp > latestTimestamp) {
@@ -1582,6 +1609,10 @@ abstract class MatroskaTrackBacking implements InputTrackBacking {
await metadataReader.reader.loadRange(metadataReader.pos, metadataReader.pos + MAX_HEADER_SIZE);
const elementStartPos = metadataReader.pos;
const elementHeader = metadataReader.readElementHeader();
if (!elementHeader) {
break;
}
const id = elementHeader.id;
let size = elementHeader.size;
const dataStartPos = metadataReader.pos;
+7 -1
View File
@@ -1248,9 +1248,15 @@ class AudioDecoderWrapper extends DecoderWrapper<AudioSample> {
super(onSample, onError);
const sampleHandler = (sample: AudioSample) => {
const sampleRate = decoderConfig.sampleRate;
if (sample.numberOfFrames === 0) {
// We skip zero-data (empty) AudioSamples. These are sometimes emitted, for example, by Firefox when it
// decodes Vorbis (at the start).
sample.close();
return;
}
// Round the timestamp to the sample rate
const sampleRate = decoderConfig.sampleRate;
sample.setTimestamp(Math.round(sample.timestamp * sampleRate) / sampleRate);
onSample(sample);
+33 -31
View File
@@ -158,7 +158,8 @@ export class EncodedVideoPacketSource extends VideoSource {
}
/**
* Adds an encoded packet to the output video track.
* Adds an encoded packet to the output video track. Packets must be added in *decode order*, while a packet's
* timestamp must be its *presentation timestamp*. B-frames are handled automatically.
*
* @param meta - Additional metadata from the encoder. You should pass this for the first call, including a valid
* decoder config.
@@ -322,17 +323,17 @@ class VideoEncoderWrapper {
if (this.customEncoder) {
this.customEncoderQueueSize++;
const promise = this.customEncoderCallSerializer
.call(() => this.customEncoder!.encode(videoSample, finalEncodeOptions))
.then(() => {
this.customEncoderQueueSize--;
if (shouldClose) {
videoSample.close();
}
})
.catch((error: Error) => {
this.encoderError ??= error;
// We clone the sample so it cannot be closed on us from the outside before it reaches the encoder
const clonedSample = videoSample.clone();
const promise = this.customEncoderCallSerializer
.call(() => this.customEncoder!.encode(clonedSample, finalEncodeOptions))
.then(() => this.customEncoderQueueSize--)
.catch((error: Error) => this.encoderError ??= error)
.finally(() => {
clonedSample.close();
// `videoSample` gets closed in the finally block at the end of the method
});
if (this.customEncoderQueueSize >= 4) {
@@ -439,6 +440,7 @@ class VideoEncoderWrapper {
void this.muxer!.addEncodedVideoPacket(this.source._connectedTrack!, packet, meta);
},
error: (error) => {
error.stack = new Error().stack; // Provide a more useful stack trace
this.encoderError ??= error;
},
});
@@ -482,7 +484,6 @@ class VideoEncoderWrapper {
checkForEncoderError() {
if (this.encoderError) {
this.encoderError.stack = new Error().stack; // Provide a more useful stack trace
throw this.encoderError;
}
}
@@ -788,7 +789,7 @@ export class EncodedAudioPacketSource extends AudioSource {
}
/**
* Adds an encoded packet to the output audio track.
* Adds an encoded packet to the output audio track. Packets must be added in *decode order*.
*
* @param meta - Additional metadata from the encoder. You should pass this for the first call, including a valid
* decoder config.
@@ -938,17 +939,17 @@ class AudioEncoderWrapper {
if (this.customEncoder) {
this.customEncoderQueueSize++;
const promise = this.customEncoderCallSerializer
.call(() => this.customEncoder!.encode(audioSample))
.then(() => {
this.customEncoderQueueSize--;
if (shouldClose) {
audioSample.close();
}
})
.catch((error: Error) => {
this.encoderError ??= error;
// We clone the sample so it cannot be closed on us from the outside before it reaches the encoder
const clonedSample = audioSample.clone();
const promise = this.customEncoderCallSerializer
.call(() => this.customEncoder!.encode(clonedSample))
.then(() => this.customEncoderQueueSize--)
.catch((error: Error) => this.encoderError ??= error)
.finally(() => {
clonedSample.close();
// `audioSample` gets closed in the finally block at the end of the method
});
if (this.customEncoderQueueSize >= 4) {
@@ -1127,6 +1128,7 @@ class AudioEncoderWrapper {
void this.muxer!.addEncodedAudioPacket(this.source._connectedTrack!, packet, meta);
},
error: (error) => {
error.stack = new Error().stack; // Provide a more useful stack trace
this.encoderError ??= error;
},
});
@@ -1265,7 +1267,6 @@ class AudioEncoderWrapper {
checkForEncoderError() {
if (this.encoderError) {
this.encoderError.stack = new Error().stack; // Provide a more useful stack trace
throw this.encoderError;
}
}
@@ -1333,16 +1334,17 @@ export class AudioBufferSource extends AudioSource {
* @returns A Promise that resolves once the output is ready to receive more samples. You should await this Promise
* to respect writer and encoder backpressure.
*/
add(audioBuffer: AudioBuffer) {
async add(audioBuffer: AudioBuffer) {
if (!(audioBuffer instanceof AudioBuffer)) {
throw new TypeError('audioBuffer must be an AudioBuffer.');
}
const audioSamples = AudioSample.fromAudioBuffer(audioBuffer, this._accumulatedTime);
const promises = audioSamples.map(sample => this._encoder.add(sample, true));
const iterator = AudioSample._fromAudioBuffer(audioBuffer, this._accumulatedTime);
this._accumulatedTime += audioBuffer.duration;
return Promise.all(promises);
for (const audioSample of iterator) {
await this._encoder.add(audioSample, true);
}
}
/** @internal */
@@ -1465,10 +1467,10 @@ export class MediaStreamAudioTrackSource extends AudioSource {
let totalDuration = 0;
this._scriptProcessorNode.onaudioprocess = (event) => {
const audioSamples = AudioSample.fromAudioBuffer(event.inputBuffer, totalDuration);
const iterator = AudioSample._fromAudioBuffer(event.inputBuffer, totalDuration);
totalDuration += event.inputBuffer.duration;
for (const audioSample of audioSamples) {
for (const audioSample of iterator) {
if (!audioReceived) {
audioReceived = true;
+16
View File
@@ -73,6 +73,22 @@ export class Bitstream {
return result;
}
writeBits(n: number, value: number) {
const end = this.pos + n;
for (let i = this.pos; i < end; i++) {
const byteIndex = Math.floor(i / 8);
let byte = this.bytes[byteIndex]!;
const bitIndex = 0b111 - (i & 0b111);
byte &= ~(1 << bitIndex);
byte |= ((value & (1 << (end - i - 1))) >> (end - i - 1)) << bitIndex;
this.bytes[byteIndex] = byte;
}
this.pos = end;
};
readAlignedByte() {
// Ensure we're byte-aligned
if (this.pos % 8 !== 0) {
+66 -64
View File
@@ -32,7 +32,7 @@ export class Mp3Demuxer extends Demuxer {
tracks: InputAudioTrack[] = [];
loadingMutex = new AsyncMutex();
readingMutex = new AsyncMutex();
lastLoadedPos = 0;
fileSize = 0;
nextTimestampInSamples = 0;
@@ -53,9 +53,8 @@ export class Mp3Demuxer extends Demuxer {
await this.loadNextChunk();
}
if (!this.firstFrameHeader) {
throw new Error('No MP3 frames found.');
}
// There has to be a frame if this demuxer got selected
assert(this.firstFrameHeader);
this.tracks = [new InputAudioTrack(new Mp3AudioTrackBacking(this))];
})();
@@ -63,30 +62,24 @@ export class Mp3Demuxer extends Demuxer {
/** Loads the next 0.5 MiB of frames. */
async loadNextChunk() {
const release = await this.loadingMutex.acquire();
assert(this.lastLoadedPos < this.fileSize);
try {
assert(this.lastLoadedPos < this.fileSize);
const chunkSize = 0.5 * 1024 * 1024; // 0.5 MiB
const endPos = Math.min(this.lastLoadedPos + chunkSize, this.fileSize);
await this.reader.reader.loadRange(this.lastLoadedPos, endPos);
const chunkSize = 0.5 * 1024 * 1024; // 0.5 MiB
const endPos = Math.min(this.lastLoadedPos + chunkSize, this.fileSize);
await this.reader.reader.loadRange(this.lastLoadedPos, endPos);
this.lastLoadedPos = endPos;
assert(this.lastLoadedPos <= this.fileSize);
this.lastLoadedPos = endPos;
assert(this.lastLoadedPos <= this.fileSize);
if (this.reader.pos === 0) {
// First time, let's see if there's an ID3 tag
const id3Tag = this.reader.readId3();
if (id3Tag) {
this.reader.pos += id3Tag.size;
}
if (this.reader.pos === 0) {
// First time, let's see if there's an ID3 tag
const id3Tag = this.reader.readId3();
if (id3Tag) {
this.reader.pos += id3Tag.size;
}
this.parseFramesFromLoadedData();
} finally {
release();
}
this.parseFramesFromLoadedData();
}
private parseFramesFromLoadedData() {
@@ -232,58 +225,67 @@ class Mp3AudioTrackBacking implements InputAudioTrackBacking {
}
async getFirstPacket(options: PacketRetrievalOptions) {
// Ensure we have at least one frame loaded
while (this.demuxer.loadedSamples.length === 0 && this.demuxer.lastLoadedPos < this.demuxer.fileSize) {
await this.demuxer.loadNextChunk();
}
return this.getPacketAtIndex(0, options);
}
async getNextPacket(packet: EncodedPacket, options: PacketRetrievalOptions) {
const sampleIndex = binarySearchExact(
this.demuxer.loadedSamples,
packet.timestamp,
x => x.timestamp,
);
if (sampleIndex === -1) {
throw new Error('Packet was not created from this track.');
}
const release = await this.demuxer.readingMutex.acquire();
const nextIndex = sampleIndex + 1;
// Ensure the next sample exists
while (nextIndex >= this.demuxer.loadedSamples.length && this.demuxer.lastLoadedPos < this.demuxer.fileSize) {
await this.demuxer.loadNextChunk();
}
try {
const sampleIndex = binarySearchExact(
this.demuxer.loadedSamples,
packet.timestamp,
x => x.timestamp,
);
if (sampleIndex === -1) {
throw new Error('Packet was not created from this track.');
}
return this.getPacketAtIndex(nextIndex, options);
const nextIndex = sampleIndex + 1;
// Ensure the next sample exists
while (
nextIndex >= this.demuxer.loadedSamples.length
&& this.demuxer.lastLoadedPos < this.demuxer.fileSize
) {
await this.demuxer.loadNextChunk();
}
return this.getPacketAtIndex(nextIndex, options);
} finally {
release();
}
}
async getPacket(timestamp: number, options: PacketRetrievalOptions) {
while (true) {
const index = binarySearchLessOrEqual(
this.demuxer.loadedSamples,
timestamp,
x => x.timestamp,
);
const release = await this.demuxer.readingMutex.acquire();
try {
while (true) {
const index = binarySearchLessOrEqual(
this.demuxer.loadedSamples,
timestamp,
x => x.timestamp,
);
if (index === -1 && this.demuxer.loadedSamples.length > 0) {
// We're before the first sample
return null;
if (index === -1 && this.demuxer.loadedSamples.length > 0) {
// We're before the first sample
return null;
}
if (this.demuxer.lastLoadedPos === this.demuxer.fileSize) {
// All data is loaded, return what we found
return this.getPacketAtIndex(index, options);
}
if (index >= 0 && index + 1 < this.demuxer.loadedSamples.length) {
// The next packet also exists, we're done
return this.getPacketAtIndex(index, options);
}
// Otherwise, keep loading data
await this.demuxer.loadNextChunk();
}
if (this.demuxer.lastLoadedPos === this.demuxer.fileSize) {
// All data is loaded, return what we found
return this.getPacketAtIndex(index, options);
}
if (index >= 0 && index + 1 < this.demuxer.loadedSamples.length) {
// The next packet also exists, we're done
return this.getPacketAtIndex(index, options);
}
// Otherwise, keep loading data
await this.demuxer.loadNextChunk();
} finally {
release();
}
}
+72
View File
@@ -6,6 +6,7 @@
* file, You can obtain one at https://mozilla.org/MPL/2.0/.
*/
import { AdtsMuxer } from './adts/adts-muxer';
import {
AUDIO_CODECS,
AudioCodec,
@@ -705,3 +706,74 @@ export class OggOutputFormat extends OutputFormat {
return false;
}
}
/**
* ADTS-specific output options.
* @public
*/
export type AdtsOutputFormatOptions = {
/**
* Will be called for each ADTS frame that is written.
*
* @param data - The raw bytes.
* @param position - The byte offset of the data in the file.
*/
onFrame?: (data: Uint8Array, position: number) => unknown;
};
/**
* ADTS file format.
* @public
*/
export class AdtsOutputFormat extends OutputFormat {
/** @internal */
_options: AdtsOutputFormatOptions;
constructor(options: AdtsOutputFormatOptions = {}) {
if (!options || typeof options !== 'object') {
throw new TypeError('options must be an object.');
}
if (options.onFrame !== undefined && typeof options.onFrame !== 'function') {
throw new TypeError('options.onFrame, when provided, must be a function.');
}
super();
this._options = options;
}
/** @internal */
_createMuxer(output: Output) {
return new AdtsMuxer(output, this);
}
/** @internal */
get _name() {
return 'ADTS';
}
getSupportedTrackCounts(): TrackCountLimits {
return {
video: { min: 0, max: 0 },
audio: { min: 1, max: 1 },
subtitle: { min: 0, max: 0 },
total: { min: 1, max: 1 },
};
}
get fileExtension() {
return '.aac';
}
get mimeType() {
return 'audio/aac';
}
getSupportedCodecs(): MediaCodec[] {
return ['aac'];
}
get supportsVideoRotationMetadata() {
return false;
}
}
+46 -3
View File
@@ -1067,6 +1067,49 @@ export class AudioSample {
(this.timestamp as number) = newTimestamp;
}
/** @internal */
static* _fromAudioBuffer(audioBuffer: AudioBuffer, timestamp: number) {
if (!(audioBuffer instanceof AudioBuffer)) {
throw new TypeError('audioBuffer must be an AudioBuffer.');
}
const MAX_FLOAT_COUNT = 48000 * 5; // 5 seconds of mono 48 kHz audio per sample
const numberOfChannels = audioBuffer.numberOfChannels;
const sampleRate = audioBuffer.sampleRate;
const totalFrames = audioBuffer.length;
const maxFramesPerChunk = Math.floor(MAX_FLOAT_COUNT / numberOfChannels);
let currentRelativeFrame = 0;
let remainingFrames = totalFrames;
// Create AudioSamples in a chunked fashion so we don't create huge Float32Arrays
while (remainingFrames > 0) {
const framesToCopy = Math.min(maxFramesPerChunk, remainingFrames);
const chunkData = new Float32Array(numberOfChannels * framesToCopy);
for (let channel = 0; channel < numberOfChannels; channel++) {
audioBuffer.copyFromChannel(
chunkData.subarray(channel * framesToCopy, (channel + 1) * framesToCopy),
channel,
currentRelativeFrame,
);
}
yield new AudioSample({
format: 'f32-planar',
sampleRate,
numberOfFrames: framesToCopy,
numberOfChannels,
timestamp: timestamp + currentRelativeFrame / sampleRate,
data: chunkData,
});
currentRelativeFrame += framesToCopy;
remainingFrames -= framesToCopy;
}
}
/**
* Creates AudioSamples from an AudioBuffer, starting at the given timestamp in seconds. Typically creates exactly
* one sample, but may create multiple if the AudioBuffer is exceedingly large.
@@ -1076,7 +1119,7 @@ export class AudioSample {
throw new TypeError('audioBuffer must be an AudioBuffer.');
}
const MAX_FLOAT_COUNT = 64 * 1024 * 1024;
const MAX_FLOAT_COUNT = 48000 * 5; // 5 seconds of mono 48 kHz audio per sample
const numberOfChannels = audioBuffer.numberOfChannels;
const sampleRate = audioBuffer.sampleRate;
@@ -1088,14 +1131,14 @@ export class AudioSample {
const result: AudioSample[] = [];
// Create AudioData in a chunked fashion so we don't create huge Float32Arrays
// Create AudioSamples in a chunked fashion so we don't create huge Float32Arrays
while (remainingFrames > 0) {
const framesToCopy = Math.min(maxFramesPerChunk, remainingFrames);
const chunkData = new Float32Array(numberOfChannels * framesToCopy);
for (let channel = 0; channel < numberOfChannels; channel++) {
audioBuffer.copyFromChannel(
chunkData.subarray(channel * framesToCopy, channel * framesToCopy + framesToCopy),
chunkData.subarray(channel * framesToCopy, (channel + 1) * framesToCopy),
channel,
currentRelativeFrame,
);
+5 -1
View File
@@ -314,7 +314,11 @@ export class UrlSource extends Source {
} else if (rangeResponse.status === 200) {
// The server just returned the whole thing
this._fullData = await rangeResponse.arrayBuffer();
return this._fullData.byteLength;
if (this._fullData.byteLength !== 1) {
return this._fullData.byteLength;
} else {
// The server responded with 200, but returned only the requested range, so skip the response
}
}
// If the range request didn't provide the size, make a full GET request