diff --git a/dev/convert.html b/dev/convert.html
index 7f9fde7..e58be51 100644
--- a/dev/convert.html
+++ b/dev/convert.html
@@ -11,9 +11,275 @@
progress.max = 1;
document.body.append(progress);
+
+
+
+
+ /**
+ * Naive audio resampler using linear interpolation between samples
+ * Much faster but lower quality than windowed sinc - causes aliasing and imaging artifacts
+ * @param {AudioBuffer} inputBuffer - The input audio buffer
+ * @param {number} targetSampleRate - The desired output sample rate
+ * @returns {AudioBuffer} - New resampled audio buffer
+ */
+function naiveResample(inputBuffer, targetSampleRate) {
+ const inputSampleRate = inputBuffer.sampleRate;
+ const ratio = targetSampleRate / inputSampleRate;
+ const inputLength = inputBuffer.length;
+ const outputLength = Math.floor(inputLength * ratio);
+
+ // Create output buffer
+ const audioContext = new (window.AudioContext || window.webkitAudioContext)();
+ const outputBuffer = audioContext.createBuffer(
+ inputBuffer.numberOfChannels,
+ outputLength,
+ targetSampleRate
+ );
+
+ // Process each channel independently
+ for (let channel = 0; channel < inputBuffer.numberOfChannels; channel++) {
+ const inputData = inputBuffer.getChannelData(channel);
+ const outputData = outputBuffer.getChannelData(channel);
+
+ naiveResampleChannel(inputData, outputData, inputSampleRate, targetSampleRate);
+ }
+
+ return outputBuffer;
+}
+
+/**
+ * Naive resample a single channel using linear interpolation
+ */
+function naiveResampleChannel(inputData, outputData, inputSampleRate, targetSampleRate) {
+ const inputLength = inputData.length;
+ const outputLength = outputData.length;
+
+ for (let n = 0; n < outputLength; n++) {
+ // Calculate the corresponding position in the input signal
+ const inputPosition = n * inputSampleRate / targetSampleRate;
+
+ // Get the floor and ceiling indices
+ const lowerIndex = Math.floor(inputPosition);
+ const upperIndex = Math.ceil(inputPosition);
+
+ // Handle edge cases
+ if (lowerIndex >= inputLength - 1) {
+ // At or past the end - just use the last sample
+ outputData[n] = inputData[inputLength - 1];
+ } else if (lowerIndex < 0) {
+ // Before the start - use first sample (shouldn't happen with our calculation)
+ outputData[n] = inputData[0];
+ } else if (lowerIndex === upperIndex) {
+ // Exact sample alignment - no interpolation needed
+ outputData[n] = inputData[lowerIndex];
+ } else {
+ // Linear interpolation between floor and ceil samples
+ const fraction = inputPosition - lowerIndex;
+ const lowerSample = inputData[lowerIndex];
+ const upperSample = inputData[upperIndex];
+
+ // Linear interpolation: lerp(a, b, t) = a + t * (b - a)
+ outputData[n] = lowerSample + fraction * (upperSample - lowerSample);
+ }
+ }
+}
+
+
+
+
+
+
+
+/**
+ * Resample an AudioBuffer to a new sample rate using windowed sinc interpolation
+ * @param {AudioBuffer} inputBuffer - The input audio buffer
+ * @param {number} targetSampleRate - The desired output sample rate
+ * @param {number} windowSize - Half-width of the sinc window (default: 6)
+ * @returns {AudioBuffer} - New resampled audio buffer
+ */
+function resampleAudioBuffer(inputBuffer, targetSampleRate, windowSize = 1) {
+ const inputSampleRate = inputBuffer.sampleRate;
+ const ratio = targetSampleRate / inputSampleRate;
+ const inputLength = inputBuffer.length;
+ const outputLength = Math.floor(inputLength * ratio);
+
+ // Scale window size for anti-aliasing when downsampling
+ const effectiveWindowSize = windowSize * Math.max(1, inputSampleRate / targetSampleRate);
+
+ // Create output buffer
+ const audioContext = new (window.AudioContext || window.webkitAudioContext)();
+ const outputBuffer = audioContext.createBuffer(
+ inputBuffer.numberOfChannels,
+ outputLength,
+ targetSampleRate
+ );
+
+ // Process each channel independently
+ for (let channel = 0; channel < inputBuffer.numberOfChannels; channel++) {
+ const inputData = inputBuffer.getChannelData(channel);
+ const outputData = outputBuffer.getChannelData(channel);
+
+ resampleChannel(inputData, outputData, inputSampleRate, targetSampleRate, effectiveWindowSize);
+ }
+
+ return outputBuffer;
+}
+
+/**
+ * Resample a single channel of audio data
+ */
+function resampleChannel(inputData, outputData, inputSampleRate, targetSampleRate, windowSize) {
+ const inputLength = inputData.length;
+ const outputLength = outputData.length;
+
+ for (let n = 0; n < outputLength; n++) {
+ // Current output time in input sample units
+ const inputTime = n * inputSampleRate / targetSampleRate;
+
+ let sum = 0;
+ const windowRadius = Math.ceil(windowSize);
+
+ // Convolve with windowed sinc kernel
+ for (let k = -windowRadius; k <= windowRadius; k++) {
+ const inputIndex = Math.floor(inputTime) + k;
+
+ // Handle edges with zero padding
+ if (inputIndex < 0 || inputIndex >= inputLength) {
+ continue;
+ }
+
+ // Time difference for sinc calculation
+ const timeDiff = inputTime - inputIndex;
+
+ // Calculate windowed sinc weight
+ const weight = windowedSinc(timeDiff, windowSize);
+
+ sum += inputData[inputIndex] * weight;
+ }
+
+ outputData[n] = sum;
+ }
+}
+
+/**
+ * Windowed sinc function using Kaiser window
+ * @param {number} x - Input value
+ * @param {number} windowSize - Window size parameter
+ * @returns {number} - Windowed sinc value
+ */
+function windowedSinc(x, windowSize) {
+ if (Math.abs(x) > windowSize) {
+ return 0;
+ }
+
+ // Sinc function
+ let sincValue;
+ if (Math.abs(x) < 1e-10) {
+ sincValue = 1; // lim(x->0) sinc(x) = 1
+ } else {
+ const piX = Math.PI * x;
+ sincValue = Math.sin(piX) / piX;
+ }
+
+ // Kaiser window (beta = 8 for good balance of main lobe width vs side lobe suppression)
+ const beta = 8;
+ const windowValue = kaiserWindow(x / windowSize, beta);
+
+ return sincValue * windowValue;
+}
+
+/**
+ * Kaiser window function
+ * @param {number} n - Normalized position (-1 to 1)
+ * @param {number} beta - Kaiser beta parameter
+ * @returns {number} - Window value
+ */
+function kaiserWindow(n, beta) {
+ if (Math.abs(n) > 1) {
+ return 0;
+ }
+
+ const arg = beta * Math.sqrt(1 - n * n);
+ return modifiedBesselI0(arg) / modifiedBesselI0(beta);
+}
+
+/**
+ * Modified Bessel function of the first kind, order 0
+ * Using series approximation
+ */
+function modifiedBesselI0(x) {
+ let sum = 1;
+ let term = 1;
+ const xSquaredOver4 = (x * x) / 4;
+
+ for (let k = 1; k < 50; k++) {
+ term *= xSquaredOver4 / (k * k);
+ sum += term;
+
+ if (term < 1e-12) break; // Convergence check
+ }
+
+ return sum;
+}
+
+// Example usage:
+// const resampledBuffer = resampleAudioBuffer(originalBuffer, 44100);
+
+// For testing - create a simple test signal
+function createTestBuffer(sampleRate = 48000, duration = 1, frequency = 440) {
+ const audioContext = new (window.AudioContext || window.webkitAudioContext)();
+ const length = Math.floor(sampleRate * duration);
+ const buffer = audioContext.createBuffer(1, length, sampleRate);
+ const data = buffer.getChannelData(0);
+
+ for (let i = 0; i < length; i++) {
+ data[i] = Math.sin(2 * Math.PI * frequency * i / sampleRate) * 0.5;
+ }
+
+ return buffer;
+}
+
+// Test example:
+// const testBuffer = createTestBuffer(48000, 1, 440);
+// const resampled = resampleAudioBuffer(testBuffer, 44100);
+// console.log(`Original: ${testBuffer.sampleRate}Hz, ${testBuffer.length} samples`);
+// console.log(`Resampled: ${resampled.sampleRate}Hz, ${resampled.length} samples`);
+
fileInput.addEventListener('change', async () => {
const file = fileInput.files[0];
+ /*
+ const context = new AudioContext();
+ const buffer = await context.decodeAudioData(await file.arrayBuffer());
+
+ const resampled = naiveResample(buffer, 16000);
+ console.log(resampled)
+
+ const node = context.createBufferSource();
+ node.buffer = resampled;
+ node.connect(context.destination);
+ node.start();
+ */
+
+ /*
+ const cursedOutput = new Metamuxer.Output({
+ format: new Metamuxer.WavOutputFormat(),
+ target: new Metamuxer.BufferTarget()
+ });
+ const cursedSource = new Metamuxer.AudioBufferSource({codec: 'pcm-s16'});
+ cursedOutput.addAudioTrack(cursedSource);
+
+ await cursedOutput.start();
+
+ await cursedSource.add(resampled);
+ await cursedOutput.finalize();
+
+ console.log(cursedOutput.target.buffer);
+ download(new Blob([cursedOutput.target.buffer]), 'cursed.wav')
+ */
+
+ //return;
+
const source = new Metamuxer.BlobSource(file);
const target = new Metamuxer.BufferTarget() ?? new Metamuxer.StreamTarget(new WritableStream({
write: console.log
@@ -21,7 +287,7 @@
chunked: true,
chunkSize: 2**20
});
- const outputFormat = new Metamuxer.Mp4OutputFormat();
+ const outputFormat = new Metamuxer.WavOutputFormat();
const button = document.createElement('button');
button.textContent = 'Cancel';
@@ -38,6 +304,8 @@
target
}),
audio: {
+ numberOfChannels: 1,
+ sampleRate: 16000
//discard: true
//forceReencode: true,
},
@@ -66,7 +334,8 @@
},
*/
video: {
- width: 640
+ discard: true,
+ //width: 640
//forceReencode: true,
//rotate: 90
//width: 720 ?? 2160,
diff --git a/src/conversion.ts b/src/conversion.ts
index 1bf4052..940f8f7 100644
--- a/src/conversion.ts
+++ b/src/conversion.ts
@@ -12,14 +12,12 @@ import {
import { Input } from './input';
import { InputAudioTrack, InputTrack, InputVideoTrack } from './input-track';
import {
- AudioBufferSink,
AudioSampleSink,
CanvasSink,
EncodedPacketSink,
VideoSampleSink,
} from './media-sink';
import {
- AudioBufferSource,
AudioEncodingConfig,
AudioSource,
EncodedVideoPacketSource,
@@ -31,7 +29,7 @@ import {
} from './media-source';
import { assert, clamp, normalizeRotation, promiseWithResolvers, Rotation } from './misc';
import { Output, TrackType } from './output';
-import { VideoSample } from './sample';
+import { AudioSample, VideoSample } from './sample';
/**
* The options for media file conversion.
@@ -782,18 +780,14 @@ export class Conversion {
this.utilizedTracks.push(track);
}
- /**
- * Resamples the audio by decoding it, playing it onto an OfflineAudioContext and encoding the
- * resulting AudioBuffer.
- * @internal
- */
+ /** @internal */
_resampleAudio(
track: InputAudioTrack,
codec: AudioCodec,
targetNumberOfChannels: number,
targetSampleRate: number,
) {
- const source = new AudioBufferSource({
+ const source = new AudioSampleSource({
codec,
bitrate: this._options.audio?.bitrate ?? QUALITY_HIGH,
onEncodedPacket: packet => this._reportProgress(track.id, packet.timestamp + packet.duration),
@@ -802,89 +796,33 @@ export class Conversion {
this._trackPromises.push((async () => {
await this._started;
- const trackDuration = Math.min(
- await track.computeDuration() - this._startTimestamp,
- this._endTimestamp - this._startTimestamp,
- );
- const totalFrameCount = Math.round(trackDuration * targetSampleRate);
- const maxChunkLength = 5 * targetSampleRate;
-
- let currentContextStartFrame = 0;
- let currentContext: OfflineAudioContext | null = new OfflineAudioContext({
- length: Math.min(totalFrameCount - currentContextStartFrame, maxChunkLength),
- numberOfChannels: targetNumberOfChannels,
- sampleRate: targetSampleRate,
+ const resampler = new AudioResampler({
+ sourceNumberOfChannels: track.numberOfChannels,
+ sourceSampleRate: track.sampleRate,
+ targetNumberOfChannels,
+ targetSampleRate,
+ startTime: this._startTimestamp,
+ endTime: this._endTimestamp,
+ onSample: sample => source.add(sample),
});
- const sink = new AudioBufferSink(track);
- const iterator = sink.buffers(this._startTimestamp, this._endTimestamp);
+ const sink = new AudioSampleSink(track);
+ const iterator = sink.samples(this._startTimestamp, this._endTimestamp); // Todo make sure timestamps work
- for await (const { buffer, timestamp, duration } of iterator) {
- if (this._synchronizer.shouldWait(track.id, timestamp)) {
- await this._synchronizer.wait(timestamp);
+ for await (const sample of iterator) {
+ if (this._synchronizer.shouldWait(track.id, sample.timestamp)) {
+ await this._synchronizer.wait(sample.timestamp);
}
- const offsetTimestamp = timestamp - this._startTimestamp;
- const endTimestamp = offsetTimestamp + duration;
-
- // while loop, as a single source buffer may span multiple audio contexts
- while (currentContext) {
- const currentContextStartTime = currentContextStartFrame / targetSampleRate;
- const currentContextEndTime
- = (currentContextStartFrame + currentContext.length) / targetSampleRate;
-
- if (offsetTimestamp < currentContextEndTime) {
- // The buffer lies within the context, let's play it
- const node = currentContext.createBufferSource();
- node.buffer = buffer;
- node.connect(currentContext.destination);
-
- if (offsetTimestamp < currentContextStartTime) {
- node.start(0, currentContextStartTime - offsetTimestamp);
- } else {
- node.start(offsetTimestamp - currentContextStartTime);
- }
- }
-
- if (endTimestamp >= currentContextEndTime) {
- // Render the audio
- const renderedBuffer = await currentContext.startRendering();
-
- if (this._canceled) {
- return;
- }
-
- await source.add(renderedBuffer);
-
- currentContextStartFrame += currentContext.length;
-
- const newLength = Math.min(
- totalFrameCount - currentContextStartFrame,
- maxChunkLength,
- );
- currentContext = newLength > 0
- ? new OfflineAudioContext({
- length: newLength,
- numberOfChannels: targetNumberOfChannels,
- sampleRate: targetSampleRate,
- })
- : null;
- } else {
- break;
- }
- }
- }
-
- if (currentContext) {
- const renderedBuffer = await currentContext.startRendering();
-
if (this._canceled) {
return;
}
- await source.add(renderedBuffer);
+ await resampler.add(sample);
}
+ await resampler.finalize();
+
await source.close();
this._synchronizer.closeTrack(track.id);
})());
@@ -971,3 +909,278 @@ class TrackSynchronizer {
this.computeMinAndMaybeResolve();
}
}
+
+/**
+ * Utility class to handle audio resampling, handling both sample rate resampling as well as channel up/downmixing.
+ * The advantage over doing this manually rather than using OfflineAudioContext to do it for us is the artifact-free
+ * handling of putting multiple resampled audio samples back to back, which produces flaky results using
+ * OfflineAudioContext.
+ */
+export class AudioResampler {
+ sourceSampleRate: number;
+ targetSampleRate: number;
+ sourceNumberOfChannels: number;
+ targetNumberOfChannels: number;
+ startTime: number;
+ endTime: number;
+ onSample: (sample: AudioSample) => Promise;
+
+ bufferSizeInFrames: number;
+ bufferSizeInSamples: number;
+ outputBuffer: Float32Array;
+ /** Start frame of current buffer */
+ bufferStartFrame: number;
+ /** The highest index written to in the current buffer */
+ maxWrittenFrame: number;
+ channelMixer!: (sourceData: Float32Array, sourceFrameIndex: number, targetChannelIndex: number) => number;
+ tempSourceBuffer: Float32Array;
+
+ constructor(options: {
+ sourceSampleRate: number;
+ targetSampleRate: number;
+ sourceNumberOfChannels: number;
+ targetNumberOfChannels: number;
+ startTime: number;
+ endTime: number;
+ onSample: (sample: AudioSample) => Promise;
+ }) {
+ this.sourceSampleRate = options.sourceSampleRate;
+ this.targetSampleRate = options.targetSampleRate;
+ this.sourceNumberOfChannels = options.sourceNumberOfChannels;
+ this.targetNumberOfChannels = options.targetNumberOfChannels;
+ this.startTime = options.startTime;
+ this.endTime = options.endTime;
+ this.onSample = options.onSample;
+
+ this.bufferSizeInFrames = Math.floor(this.targetSampleRate * 5.0); // 5 seconds
+ this.bufferSizeInSamples = this.bufferSizeInFrames * this.targetNumberOfChannels;
+
+ this.outputBuffer = new Float32Array(this.bufferSizeInSamples);
+ this.bufferStartFrame = 0;
+ this.maxWrittenFrame = -1;
+
+ this.setupChannelMixer();
+
+ // Pre-allocate temporary buffer for source data
+ this.tempSourceBuffer = new Float32Array(this.sourceSampleRate * this.sourceNumberOfChannels);
+ }
+
+ /**
+ * Sets up the channel mixer to handle up/downmixing in the case where input and output channel counts don't match.
+ */
+ setupChannelMixer(): void {
+ const sourceNum = this.sourceNumberOfChannels;
+ const targetNum = this.targetNumberOfChannels;
+
+ // Logic taken from
+ // https://developer.mozilla.org/en-US/docs/Web/API/Web_Audio_API/Basic_concepts_behind_Web_Audio_API
+ // Most of the mapping functions are branchless.
+
+ if (sourceNum === 1 && targetNum === 2) {
+ // Mono to Stereo: M -> L, M -> R
+ this.channelMixer = (sourceData: Float32Array, sourceFrameIndex: number) => {
+ return sourceData[sourceFrameIndex * sourceNum]!;
+ };
+ } else if (sourceNum === 1 && targetNum === 4) {
+ // Mono to Quad: M -> L, M -> R, 0 -> SL, 0 -> SR
+ this.channelMixer = (sourceData: Float32Array, sourceFrameIndex: number, targetChannelIndex: number) => {
+ return sourceData[sourceFrameIndex * sourceNum]! * +(targetChannelIndex < 2);
+ };
+ } else if (sourceNum === 1 && targetNum === 6) {
+ // Mono to 5.1: 0 -> L, 0 -> R, M -> C, 0 -> LFE, 0 -> SL, 0 -> SR
+ this.channelMixer = (sourceData: Float32Array, sourceFrameIndex: number, targetChannelIndex: number) => {
+ return sourceData[sourceFrameIndex * sourceNum]! * +(targetChannelIndex === 2);
+ };
+ } else if (sourceNum === 2 && targetNum === 1) {
+ // Stereo to Mono: 0.5 * (L + R)
+ this.channelMixer = (sourceData: Float32Array, sourceFrameIndex: number) => {
+ const baseIdx = sourceFrameIndex * sourceNum;
+ return 0.5 * (sourceData[baseIdx]! + sourceData[baseIdx + 1]!);
+ };
+ } else if (sourceNum === 2 && targetNum === 4) {
+ // Stereo to Quad: L -> L, R -> R, 0 -> SL, 0 -> SR
+ this.channelMixer = (sourceData: Float32Array, sourceFrameIndex: number, targetChannelIndex: number) => {
+ return sourceData[sourceFrameIndex * sourceNum + targetChannelIndex]! * +(targetChannelIndex < 2);
+ };
+ } else if (sourceNum === 2 && targetNum === 6) {
+ // Stereo to 5.1: L -> L, R -> R, 0 -> C, 0 -> LFE, 0 -> SL, 0 -> SR
+ this.channelMixer = (sourceData: Float32Array, sourceFrameIndex: number, targetChannelIndex: number) => {
+ return sourceData[sourceFrameIndex * sourceNum + targetChannelIndex]! * +(targetChannelIndex < 2);
+ };
+ } else if (sourceNum === 4 && targetNum === 1) {
+ // Quad to Mono: 0.25 * (L + R + SL + SR)
+ this.channelMixer = (sourceData: Float32Array, sourceFrameIndex: number) => {
+ const baseIdx = sourceFrameIndex * sourceNum;
+ return 0.25 * (
+ sourceData[baseIdx]! + sourceData[baseIdx + 1]!
+ + sourceData[baseIdx + 2]! + sourceData[baseIdx + 3]!
+ );
+ };
+ } else if (sourceNum === 4 && targetNum === 2) {
+ // Quad to Stereo: 0.5 * (L + SL), 0.5 * (R + SR)
+ this.channelMixer = (sourceData: Float32Array, sourceFrameIndex: number, targetChannelIndex: number) => {
+ const baseIdx = sourceFrameIndex * sourceNum;
+ return 0.5 * (
+ sourceData[baseIdx + targetChannelIndex]!
+ + sourceData[baseIdx + targetChannelIndex + 2]!
+ );
+ };
+ } else if (sourceNum === 4 && targetNum === 6) {
+ // Quad to 5.1: L -> L, R -> R, 0 -> C, 0 -> LFE, SL -> SL, SR -> SR
+ this.channelMixer = (sourceData: Float32Array, sourceFrameIndex: number, targetChannelIndex: number) => {
+ const baseIdx = sourceFrameIndex * sourceNum;
+
+ // It's a bit harder to do this one branchlessly
+ if (targetChannelIndex < 2) return sourceData[baseIdx + targetChannelIndex]!; // L, R
+ if (targetChannelIndex === 2 || targetChannelIndex === 3) return 0; // C, LFE
+ return sourceData[baseIdx + targetChannelIndex - 2]!; // SL, SR
+ };
+ } else if (sourceNum === 6 && targetNum === 1) {
+ // 5.1 to Mono: sqrt(1/2) * (L + R) + C + 0.5 * (SL + SR)
+ this.channelMixer = (sourceData: Float32Array, sourceFrameIndex: number) => {
+ const baseIdx = sourceFrameIndex * sourceNum;
+ return Math.SQRT1_2 * (sourceData[baseIdx]! + sourceData[baseIdx + 1]!)
+ + sourceData[baseIdx + 2]!
+ + 0.5 * (sourceData[baseIdx + 4]! + sourceData[baseIdx + 5]!);
+ };
+ } else if (sourceNum === 6 && targetNum === 2) {
+ // 5.1 to Stereo: L + sqrt(1/2) * (C + SL), R + sqrt(1/2) * (C + SR)
+ this.channelMixer = (sourceData: Float32Array, sourceFrameIndex: number, targetChannelIndex: number) => {
+ const baseIdx = sourceFrameIndex * sourceNum;
+ return sourceData[baseIdx + targetChannelIndex]!
+ + Math.SQRT1_2 * (sourceData[baseIdx + 2]! + sourceData[baseIdx + targetChannelIndex + 4]!);
+ };
+ } else if (sourceNum === 6 && targetNum === 4) {
+ // 5.1 to Quad: L + sqrt(1/2) * C, R + sqrt(1/2) * C, SL, SR
+ this.channelMixer = (sourceData: Float32Array, sourceFrameIndex: number, targetChannelIndex: number) => {
+ const baseIdx = sourceFrameIndex * sourceNum;
+
+ // It's a bit harder to do this one branchlessly
+ if (targetChannelIndex < 2) {
+ return sourceData[baseIdx + targetChannelIndex]! + Math.SQRT1_2 * sourceData[baseIdx + 2]!;
+ }
+ return sourceData[baseIdx + targetChannelIndex + 2]!; // SL, SR
+ };
+ } else {
+ // Discrete fallback: direct mapping with zero-fill or drop
+ this.channelMixer = (sourceData: Float32Array, sourceFrameIndex: number, targetChannelIndex: number) => {
+ return targetChannelIndex < sourceNum
+ ? sourceData[sourceFrameIndex * sourceNum + targetChannelIndex]!
+ : 0;
+ };
+ }
+ }
+
+ ensureTempBufferSize(requiredSamples: number): void {
+ let length = this.tempSourceBuffer.length;
+
+ while (length < requiredSamples) {
+ length *= 2;
+ }
+
+ if (length !== this.tempSourceBuffer.length) {
+ const newBuffer = new Float32Array(length);
+ newBuffer.set(this.tempSourceBuffer);
+ this.tempSourceBuffer = newBuffer;
+ }
+ }
+
+ async add(audioSample: AudioSample) {
+ if (!audioSample || audioSample._closed) {
+ return;
+ }
+
+ const requiredSamples = audioSample.numberOfFrames * audioSample.numberOfChannels;
+ this.ensureTempBufferSize(requiredSamples);
+
+ // Copy the audio data to the temp buffer
+ const sourceDataSize = audioSample.allocationSize({ planeIndex: 0, format: 'f32' });
+ const sourceView = new Float32Array(this.tempSourceBuffer.buffer, 0, sourceDataSize / 4);
+ audioSample.copyTo(sourceView, { planeIndex: 0, format: 'f32' });
+
+ const inputStartTime = audioSample.timestamp - this.startTime;
+ const inputDuration = audioSample.numberOfFrames / this.sourceSampleRate;
+ const inputEndTime = Math.min(inputStartTime + inputDuration, this.endTime - this.startTime);
+
+ // Compute which output frames are affected by this sample
+ const outputStartFrame = Math.floor(inputStartTime * this.targetSampleRate);
+ const outputEndFrame = Math.ceil(inputEndTime * this.targetSampleRate);
+
+ for (let outputFrame = outputStartFrame; outputFrame < outputEndFrame; outputFrame++) {
+ if (outputFrame < this.bufferStartFrame) {
+ continue; // Skip writes to the past
+ }
+
+ while (outputFrame >= this.bufferStartFrame + this.bufferSizeInFrames) {
+ // The write is after the current buffer, so finalize it
+ await this.finalizeCurrentBuffer();
+ this.bufferStartFrame += this.bufferSizeInFrames;
+ }
+
+ const bufferFrameIndex = outputFrame - this.bufferStartFrame;
+ assert(bufferFrameIndex < this.bufferSizeInFrames);
+
+ const outputTime = outputFrame / this.targetSampleRate;
+ const inputTime = outputTime - inputStartTime;
+ const sourcePosition = inputTime * this.sourceSampleRate;
+
+ const sourceLowerFrame = Math.floor(sourcePosition);
+ const sourceUpperFrame = Math.ceil(sourcePosition);
+ const fraction = sourcePosition - sourceLowerFrame;
+
+ // Process each output channel
+ for (let targetChannel = 0; targetChannel < this.targetNumberOfChannels; targetChannel++) {
+ let lowerSample = 0;
+ let upperSample = 0;
+
+ if (sourceLowerFrame >= 0 && sourceLowerFrame < audioSample.numberOfFrames) {
+ lowerSample = this.channelMixer(sourceView, sourceLowerFrame, targetChannel);
+ }
+
+ if (sourceUpperFrame >= 0 && sourceUpperFrame < audioSample.numberOfFrames) {
+ upperSample = this.channelMixer(sourceView, sourceUpperFrame, targetChannel);
+ }
+
+ // For resampling, we do naive linear interpolation to find the in-between sample. This produces
+ // suboptimal results especially for downsampling (for which a low-pass filter would first need to be
+ // applied), but AudioContext doesn't do this either so, whatever.
+ const outputSample = lowerSample + fraction * (upperSample - lowerSample);
+
+ // Write to output buffer (interleaved)
+ const outputIndex = bufferFrameIndex * this.targetNumberOfChannels + targetChannel;
+ this.outputBuffer[outputIndex]! += outputSample; // Add in case of overlapping samples
+ }
+
+ this.maxWrittenFrame = Math.max(this.maxWrittenFrame, bufferFrameIndex);
+ }
+ }
+
+ async finalizeCurrentBuffer() {
+ if (this.maxWrittenFrame < 0) {
+ return; // Nothing to finalize
+ }
+
+ const samplesWritten = (this.maxWrittenFrame + 1) * this.targetNumberOfChannels;
+
+ const outputData = new Float32Array(samplesWritten);
+ outputData.set(this.outputBuffer.subarray(0, samplesWritten));
+
+ const timestampSeconds = this.bufferStartFrame / this.targetSampleRate;
+ const audioSample = new AudioSample({
+ format: 'f32',
+ sampleRate: this.targetSampleRate,
+ numberOfChannels: this.targetNumberOfChannels,
+ timestamp: timestampSeconds,
+ data: outputData,
+ });
+
+ await this.onSample(audioSample);
+
+ this.outputBuffer.fill(0);
+ this.maxWrittenFrame = -1;
+ }
+
+ finalize() {
+ return this.finalizeCurrentBuffer();
+ }
+}